Account for the real Adv.SIMD cost of SAD and ABD reductions.

For SAD_EXPR, Adv.SIMD has to build a 128-bit result from two 64-bit
halves.  Account for that extra work when costing 128-bit Adv.SIMD
vectors.

For SAD-style ABD reductions, SVE can keep the byte absolute difference as
uabd and then use udot for the widening sum.  Without dot-product support,
the Adv.SIMD fallback is a longer dependent sequence, using uabdl/uabal
followed by the widening reduction.  Charge that extra Adv.SIMD work in
both the statement cost and the issue model, so scalar profitability and
vector-loop comparison stay in sync.

This recovers costing fallouts from patch 1 in the series where the costs
makes us pick Adv. SIMD in cases where SVE would be much faster.

Bootstrapped Regtested on aarch64-none-linux-gnu and no issues.

OK for master?

Thanks,
Tamar

gcc/ChangeLog:

        * config/aarch64/aarch64.cc (aarch64_ifn_vect_stmt_p): New function.
        (aarch64_adjust_stmt_cost): Account for extra Adv.SIMD SAD and ABD
        reduction costs.
        (aarch64_vector_costs::count_ops): Likewise.

---
diff --git a/gcc/config/aarch64/aarch64.cc b/gcc/config/aarch64/aarch64.cc
index 
c168941be4e173d29bb864d937b4f80b02931f62..3e1fc9bac5fe6f24773854db2b310421a3864bb5
 100644
--- a/gcc/config/aarch64/aarch64.cc
+++ b/gcc/config/aarch64/aarch64.cc
@@ -18447,6 +18447,74 @@ aarch64_sve_adjust_stmt_cost (class vec_info *vinfo, 
vect_cost_for_stmt kind,
   return stmt_cost;
 }
 
+/* Return > 0 if STMT_INFO is the instruction denoted by CODE and return the
+   number of instructions the given CODE is implemented using.  */
+static unsigned int
+aarch64_ifn_vect_stmt_p (stmt_vec_info stmt_info, tree vectype,
+                        unsigned int vec_flags)
+{
+  stmt_info = vect_stmt_to_vectorize (stmt_info);
+  gcall *call = dyn_cast<gcall *> (STMT_VINFO_STMT (stmt_info));
+  if (!vectype)
+    return 0;
+
+  gassign *assign = dyn_cast <gassign *> (STMT_VINFO_STMT (stmt_info));
+  if (!call && !assign)
+    return 0;
+
+  bool advsimd_p = vec_flags & VEC_ADVSIMD;
+  bool is_64bit_p = known_gt (GET_MODE_BITSIZE (TYPE_MODE (vectype)), 64);
+  if (assign)
+    {
+      switch (gimple_assign_rhs_code (assign))
+      {
+       /* SAD for Adv. SIMD is emulated using two instuctions per 64-bit
+          quantities.  So 128-bit ADB requires 4 INSN.  Account for that.  */
+       case SAD_EXPR:
+         return advsimd_p && is_64bit_p ? 2 : 0;
+       case WIDEN_SUM_EXPR:
+         {
+           tree rhs = gimple_assign_rhs1 (assign);
+           if (!advsimd_p
+               || TARGET_DOTPROD
+               || !vect_is_reduction (stmt_info)
+               || TREE_CODE (rhs) != SSA_NAME)
+             return 0;
+
+           gimple *def_stmt = SSA_NAME_DEF_STMT (rhs);
+           gassign *def_assign = dyn_cast<gassign *> (def_stmt);
+           if (def_assign
+               && CONVERT_EXPR_CODE_P (gimple_assign_rhs_code (def_assign))
+               && TREE_CODE (gimple_assign_rhs1 (def_assign)) == SSA_NAME)
+             def_stmt = SSA_NAME_DEF_STMT (gimple_assign_rhs1 (def_assign));
+
+           gcall *def = dyn_cast<gcall *> (def_stmt);
+           if (!def || gimple_call_combined_fn (def) != CFN_ABD)
+             return 0;
+
+           for (unsigned int i = 0; i < 2; ++i)
+             {
+               tree arg = gimple_call_arg (def, i);
+               if (TREE_CODE (arg) != SSA_NAME
+                   || !gimple_assign_load_p (SSA_NAME_DEF_STMT (arg)))
+                 return 0;
+             }
+           return 2;
+         }
+       default:
+         break;
+      }
+      return 0;
+    }
+
+  switch (gimple_call_combined_fn (call))
+  {
+    default:
+      break;
+    }
+  return 0;
+}
+
 /* STMT_COST is the cost calculated for STMT_INFO, which has cost kind KIND
    and which when vectorized would operate on vector type VECTYPE.  Add the
    cost of any embedded operations.  */
@@ -18572,6 +18640,14 @@ aarch64_vector_costs::count_ops (unsigned int count, 
vect_cost_for_stmt kind,
        return;
     }
 
+  unsigned int n_insn = 0;
+  if (stmt_info
+      && kind == vector_stmt
+      && (n_insn = aarch64_ifn_vect_stmt_p (stmt_info,
+                                           STMT_VINFO_VECTYPE (stmt_info),
+                                           m_vec_flags)))
+    ops->general_ops += n_insn * count;
+
   /* Detect the case where we are using an emulated gather/scatter.  When a
      target does not support gathers and scatters directly the vectorizer
      emulates these by constructing an index vector and then issuing an
@@ -19151,6 +19227,15 @@ aarch64_vector_costs::add_stmt_cost (int count, 
vect_cost_for_stmt kind,
         to the base cost calculated above.  */
       stmt_cost = aarch64_adjust_stmt_cost (m_vinfo, kind, stmt_info, node,
                                            vectype, m_vec_flags, stmt_cost);
+      unsigned int n_insn = 0;
+      if (vectype
+         && kind == vector_stmt
+         && (n_insn = aarch64_ifn_vect_stmt_p (stmt_info, vectype,
+                                               m_vec_flags)))
+       {
+         const simd_vec_cost *simd_costs = aarch64_simd_vec_costs (vectype);
+         stmt_cost += count * n_insn * simd_costs->int_stmt_cost;
+       }
 
       /* If we're applying the SVE vs. Advanced SIMD unrolling heuristic,
         estimate the number of statements in the unrolled Advanced SIMD


-- 
diff --git a/gcc/config/aarch64/aarch64.cc b/gcc/config/aarch64/aarch64.cc
index c168941be4e173d29bb864d937b4f80b02931f62..3e1fc9bac5fe6f24773854db2b310421a3864bb5 100644
--- a/gcc/config/aarch64/aarch64.cc
+++ b/gcc/config/aarch64/aarch64.cc
@@ -18447,6 +18447,74 @@ aarch64_sve_adjust_stmt_cost (class vec_info *vinfo, vect_cost_for_stmt kind,
   return stmt_cost;
 }
 
+/* Return > 0 if STMT_INFO is the instruction denoted by CODE and return the
+   number of instructions the given CODE is implemented using.  */
+static unsigned int
+aarch64_ifn_vect_stmt_p (stmt_vec_info stmt_info, tree vectype,
+			 unsigned int vec_flags)
+{
+  stmt_info = vect_stmt_to_vectorize (stmt_info);
+  gcall *call = dyn_cast<gcall *> (STMT_VINFO_STMT (stmt_info));
+  if (!vectype)
+    return 0;
+
+  gassign *assign = dyn_cast <gassign *> (STMT_VINFO_STMT (stmt_info));
+  if (!call && !assign)
+    return 0;
+
+  bool advsimd_p = vec_flags & VEC_ADVSIMD;
+  bool is_64bit_p = known_gt (GET_MODE_BITSIZE (TYPE_MODE (vectype)), 64);
+  if (assign)
+    {
+      switch (gimple_assign_rhs_code (assign))
+      {
+	/* SAD for Adv. SIMD is emulated using two instuctions per 64-bit
+	   quantities.  So 128-bit ADB requires 4 INSN.  Account for that.  */
+	case SAD_EXPR:
+	  return advsimd_p && is_64bit_p ? 2 : 0;
+	case WIDEN_SUM_EXPR:
+	  {
+	    tree rhs = gimple_assign_rhs1 (assign);
+	    if (!advsimd_p
+		|| TARGET_DOTPROD
+		|| !vect_is_reduction (stmt_info)
+		|| TREE_CODE (rhs) != SSA_NAME)
+	      return 0;
+
+	    gimple *def_stmt = SSA_NAME_DEF_STMT (rhs);
+	    gassign *def_assign = dyn_cast<gassign *> (def_stmt);
+	    if (def_assign
+		&& CONVERT_EXPR_CODE_P (gimple_assign_rhs_code (def_assign))
+		&& TREE_CODE (gimple_assign_rhs1 (def_assign)) == SSA_NAME)
+	      def_stmt = SSA_NAME_DEF_STMT (gimple_assign_rhs1 (def_assign));
+
+	    gcall *def = dyn_cast<gcall *> (def_stmt);
+	    if (!def || gimple_call_combined_fn (def) != CFN_ABD)
+	      return 0;
+
+	    for (unsigned int i = 0; i < 2; ++i)
+	      {
+		tree arg = gimple_call_arg (def, i);
+		if (TREE_CODE (arg) != SSA_NAME
+		    || !gimple_assign_load_p (SSA_NAME_DEF_STMT (arg)))
+		  return 0;
+	      }
+	    return 2;
+	  }
+	default:
+	  break;
+      }
+      return 0;
+    }
+
+  switch (gimple_call_combined_fn (call))
+  {
+    default:
+      break;
+    }
+  return 0;
+}
+
 /* STMT_COST is the cost calculated for STMT_INFO, which has cost kind KIND
    and which when vectorized would operate on vector type VECTYPE.  Add the
    cost of any embedded operations.  */
@@ -18572,6 +18640,14 @@ aarch64_vector_costs::count_ops (unsigned int count, vect_cost_for_stmt kind,
 	return;
     }
 
+  unsigned int n_insn = 0;
+  if (stmt_info
+      && kind == vector_stmt
+      && (n_insn = aarch64_ifn_vect_stmt_p (stmt_info,
+					    STMT_VINFO_VECTYPE (stmt_info),
+					    m_vec_flags)))
+    ops->general_ops += n_insn * count;
+
   /* Detect the case where we are using an emulated gather/scatter.  When a
      target does not support gathers and scatters directly the vectorizer
      emulates these by constructing an index vector and then issuing an
@@ -19151,6 +19227,15 @@ aarch64_vector_costs::add_stmt_cost (int count, vect_cost_for_stmt kind,
 	 to the base cost calculated above.  */
       stmt_cost = aarch64_adjust_stmt_cost (m_vinfo, kind, stmt_info, node,
 					    vectype, m_vec_flags, stmt_cost);
+      unsigned int n_insn = 0;
+      if (vectype
+	  && kind == vector_stmt
+	  && (n_insn = aarch64_ifn_vect_stmt_p (stmt_info, vectype,
+						m_vec_flags)))
+	{
+	  const simd_vec_cost *simd_costs = aarch64_simd_vec_costs (vectype);
+	  stmt_cost += count * n_insn * simd_costs->int_stmt_cost;
+	}
 
       /* If we're applying the SVE vs. Advanced SIMD unrolling heuristic,
 	 estimate the number of statements in the unrolled Advanced SIMD

Reply via email to