> On 18 Sep 2026, at 11:04, Tamar Christina <[email protected]> wrote:
>
> Use the same VF cap in the AArch64 body-cost adjustment and in the
> main-loop comparison.
>
> Known fixed-trip loops can execute fewer lanes than the architectural VF.
> The body-cost path already wanted to account for that, but the main-loop
> comparison still used the uncapped VF. This made the two parts of the
> cost model rank the same loop differently.
>
> This factors the cap into a small helper and uses it in both places.
>
> Bootstrapped Regtested on aarch64-none-linux-gnu and no issues.
>
> OK for master?
>
> Thanks,
> Tamar
>
> gcc/ChangeLog:
>
> * config/aarch64/aarch64.cc (aarch64_vect_vf_for_cost): New
> function.
> (aarch64_vector_costs::adjust_body_cost): Use it.
> (aarch64_vector_costs::better_main_loop_than_p): Likewise.
>
> ---
> diff --git a/gcc/config/aarch64/aarch64.cc b/gcc/config/aarch64/aarch64.cc
> index
> f9d906f449f27906f8cf0b4b4edee898cb795e15..c168941be4e173d29bb864d937b4f80b02931f62
> 100644
> --- a/gcc/config/aarch64/aarch64.cc
> +++ b/gcc/config/aarch64/aarch64.cc
> @@ -18986,6 +18986,27 @@ aarch64_external_adjust_stmt_cost
> (vect_cost_for_stmt kind, slp_tree node,
> return stmt_cost;
> }
>
> +static inline unsigned int
> +aarch64_vect_vf_for_cost (loop_vec_info loop_vinfo)
> +{
New function should have a function comment, especially as we have quite a lot
of vect cost-related functions in the backend and it’s not trivial to keep
track of them in one’s head.
Ok with that change.
Thanks,
Kyrill
> + unsigned int estimated_vf = vect_vf_for_cost (loop_vinfo);
> +
> + /* If we know we have a single partial vector iteration, cap the VF
> + to the number of scalar iterations for costing purposes. */
> + if (LOOP_VINFO_NITERS_KNOWN_P (loop_vinfo))
> + {
> + auto niters = LOOP_VINFO_INT_NITERS (loop_vinfo);
> + if (niters < estimated_vf && dump_enabled_p ())
> + dump_printf_loc (MSG_NOTE, vect_location,
> + "Scalar loop iterates at most %wd times. Capping VF "
> + " from %d to %wd\n", niters, estimated_vf, niters);
> +
> + estimated_vf = MIN (estimated_vf, niters);
> + }
> +
> + return estimated_vf;
> +}
> +
> unsigned
> aarch64_vector_costs::add_stmt_cost (int count, vect_cost_for_stmt kind,
> stmt_vec_info stmt_info, slp_tree node,
> @@ -19349,7 +19370,7 @@ adjust_body_cost (loop_vec_info loop_vinfo,
>
> const auto &scalar_ops = scalar_costs->m_ops[0];
> const auto &vector_ops = m_ops[0];
> - unsigned int estimated_vf = vect_vf_for_cost (loop_vinfo);
> + unsigned int estimated_vf = aarch64_vect_vf_for_cost (loop_vinfo);
> unsigned int orig_body_cost = body_cost;
> bool should_disparage = false;
>
> @@ -19357,19 +19378,6 @@ adjust_body_cost (loop_vec_info loop_vinfo,
> dump_printf_loc (MSG_NOTE, vect_location,
> "Original vector body cost = %d\n", body_cost);
>
> - /* If we know we have a single partial vector iteration, cap the VF
> - to the number of scalar iterations for costing purposes. */
> - if (LOOP_VINFO_NITERS_KNOWN_P (loop_vinfo))
> - {
> - auto niters = LOOP_VINFO_INT_NITERS (loop_vinfo);
> - if (niters < estimated_vf && dump_enabled_p ())
> - dump_printf_loc (MSG_NOTE, vect_location,
> - "Scalar loop iterates at most %wd times. Capping VF "
> - " from %d to %wd\n", niters, estimated_vf, niters);
> -
> - estimated_vf = MIN (estimated_vf, niters);
> - }
> -
> fractional_cost scalar_cycles_per_iter
> = scalar_ops.min_cycles_per_iter () * estimated_vf;
>
> @@ -19561,14 +19569,16 @@ better_main_loop_than_p (const vector_costs
> *uncast_other) const
>
> auto this_loop_vinfo = as_a<loop_vec_info> (this->m_vinfo);
> auto other_loop_vinfo = as_a<loop_vec_info> (other->m_vinfo);
> + auto this_loop_vf = aarch64_vect_vf_for_cost (this_loop_vinfo);
> + auto other_loop_vf = aarch64_vect_vf_for_cost (other_loop_vinfo);
>
> if (dump_enabled_p ())
> dump_printf_loc (MSG_NOTE, vect_location,
> "Comparing two main loops (%s at VF %d vs %s at VF %d)\n",
> GET_MODE_NAME (this_loop_vinfo->vector_mode),
> - vect_vf_for_cost (this_loop_vinfo),
> + this_loop_vf,
> GET_MODE_NAME (other_loop_vinfo->vector_mode),
> - vect_vf_for_cost (other_loop_vinfo));
> + other_loop_vf);
>
> /* Apply the unrolling heuristic described above
> m_unrolled_advsimd_niters. */
> @@ -19604,10 +19614,8 @@ better_main_loop_than_p (const vector_costs
> *uncast_other) const
> other->m_ops[i].dump ();
> }
>
> - auto this_estimated_vf = (vect_vf_for_cost (this_loop_vinfo)
> - * this->m_ops[i].vf_factor ());
> - auto other_estimated_vf = (vect_vf_for_cost (other_loop_vinfo)
> - * other->m_ops[i].vf_factor ());
> + auto this_estimated_vf = (this_loop_vf * this->m_ops[i].vf_factor ());
> + auto other_estimated_vf = (other_loop_vf * other->m_ops[i].vf_factor
> ());
>
> /* If it appears that one loop could process the same amount of data
> in fewer cycles, prefer that loop over the other one. */
>
>
> --
> <rb20905.patch>