> On 18 Sep 2026, at 11:04, Tamar Christina <[email protected]> wrote:
> 
> Use the same VF cap in the AArch64 body-cost adjustment and in the
> main-loop comparison.
> 
> Known fixed-trip loops can execute fewer lanes than the architectural VF.
> The body-cost path already wanted to account for that, but the main-loop
> comparison still used the uncapped VF.  This made the two parts of the
> cost model rank the same loop differently.
> 
> This factors the cap into a small helper and uses it in both places.
> 
> Bootstrapped Regtested on aarch64-none-linux-gnu and no issues.
> 
> OK for master?
> 
> Thanks,
> Tamar
> 
> gcc/ChangeLog:
> 
> * config/aarch64/aarch64.cc (aarch64_vect_vf_for_cost): New
> function.
> (aarch64_vector_costs::adjust_body_cost): Use it.
> (aarch64_vector_costs::better_main_loop_than_p): Likewise.
> 
> ---
> diff --git a/gcc/config/aarch64/aarch64.cc b/gcc/config/aarch64/aarch64.cc
> index 
> f9d906f449f27906f8cf0b4b4edee898cb795e15..c168941be4e173d29bb864d937b4f80b02931f62
>  100644
> --- a/gcc/config/aarch64/aarch64.cc
> +++ b/gcc/config/aarch64/aarch64.cc
> @@ -18986,6 +18986,27 @@ aarch64_external_adjust_stmt_cost 
> (vect_cost_for_stmt kind, slp_tree node,
>   return stmt_cost;
> }
> 
> +static inline unsigned int
> +aarch64_vect_vf_for_cost (loop_vec_info loop_vinfo)
> +{

New function should have a function comment, especially as we have quite a lot 
of vect cost-related functions in the backend and it’s not trivial to keep 
track of them in one’s head.
Ok with that change.
Thanks,
Kyrill

> +  unsigned int estimated_vf = vect_vf_for_cost (loop_vinfo);
> +
> +  /* If we know we have a single partial vector iteration, cap the VF
> +     to the number of scalar iterations for costing purposes.  */
> +  if (LOOP_VINFO_NITERS_KNOWN_P (loop_vinfo))
> +    {
> +      auto niters = LOOP_VINFO_INT_NITERS (loop_vinfo);
> +      if (niters < estimated_vf && dump_enabled_p ())
> + dump_printf_loc (MSG_NOTE, vect_location,
> + "Scalar loop iterates at most %wd times.  Capping VF "
> + " from %d to %wd\n", niters, estimated_vf, niters);
> +
> +      estimated_vf = MIN (estimated_vf, niters);
> +    }
> +
> +  return estimated_vf;
> +}
> +
> unsigned
> aarch64_vector_costs::add_stmt_cost (int count, vect_cost_for_stmt kind,
>     stmt_vec_info stmt_info, slp_tree node,
> @@ -19349,7 +19370,7 @@ adjust_body_cost (loop_vec_info loop_vinfo,
> 
>   const auto &scalar_ops = scalar_costs->m_ops[0];
>   const auto &vector_ops = m_ops[0];
> -  unsigned int estimated_vf = vect_vf_for_cost (loop_vinfo);
> +  unsigned int estimated_vf = aarch64_vect_vf_for_cost (loop_vinfo);
>   unsigned int orig_body_cost = body_cost;
>   bool should_disparage = false;
> 
> @@ -19357,19 +19378,6 @@ adjust_body_cost (loop_vec_info loop_vinfo,
>     dump_printf_loc (MSG_NOTE, vect_location,
>     "Original vector body cost = %d\n", body_cost);
> 
> -  /* If we know we have a single partial vector iteration, cap the VF
> -     to the number of scalar iterations for costing purposes.  */
> -  if (LOOP_VINFO_NITERS_KNOWN_P (loop_vinfo))
> -    {
> -      auto niters = LOOP_VINFO_INT_NITERS (loop_vinfo);
> -      if (niters < estimated_vf && dump_enabled_p ())
> - dump_printf_loc (MSG_NOTE, vect_location,
> - "Scalar loop iterates at most %wd times.  Capping VF "
> - " from %d to %wd\n", niters, estimated_vf, niters);
> -
> -      estimated_vf = MIN (estimated_vf, niters);
> -    }
> -
>   fractional_cost scalar_cycles_per_iter
>     = scalar_ops.min_cycles_per_iter () * estimated_vf;
> 
> @@ -19561,14 +19569,16 @@ better_main_loop_than_p (const vector_costs 
> *uncast_other) const
> 
>   auto this_loop_vinfo = as_a<loop_vec_info> (this->m_vinfo);
>   auto other_loop_vinfo = as_a<loop_vec_info> (other->m_vinfo);
> +  auto this_loop_vf = aarch64_vect_vf_for_cost (this_loop_vinfo);
> +  auto other_loop_vf = aarch64_vect_vf_for_cost (other_loop_vinfo);
> 
>   if (dump_enabled_p ())
>     dump_printf_loc (MSG_NOTE, vect_location,
>     "Comparing two main loops (%s at VF %d vs %s at VF %d)\n",
>     GET_MODE_NAME (this_loop_vinfo->vector_mode),
> -     vect_vf_for_cost (this_loop_vinfo),
> +     this_loop_vf,
>     GET_MODE_NAME (other_loop_vinfo->vector_mode),
> -     vect_vf_for_cost (other_loop_vinfo));
> +     other_loop_vf);
> 
>   /* Apply the unrolling heuristic described above
>      m_unrolled_advsimd_niters.  */
> @@ -19604,10 +19614,8 @@ better_main_loop_than_p (const vector_costs 
> *uncast_other) const
>  other->m_ops[i].dump ();
> }
> 
> -      auto this_estimated_vf = (vect_vf_for_cost (this_loop_vinfo)
> - * this->m_ops[i].vf_factor ());
> -      auto other_estimated_vf = (vect_vf_for_cost (other_loop_vinfo)
> - * other->m_ops[i].vf_factor ());
> +      auto this_estimated_vf = (this_loop_vf * this->m_ops[i].vf_factor ());
> +      auto other_estimated_vf = (other_loop_vf * other->m_ops[i].vf_factor 
> ());
> 
>       /* If it appears that one loop could process the same amount of data
> in fewer cycles, prefer that loop over the other one.  */
> 
> 
> -- 
> <rb20905.patch>

Reply via email to