Richard Biener <[email protected]> writes:
> SLP permute nodes (can) have no SLP_TREE_REPRESENTATIVE, avoid
> leaving vertex weight (later used for costing) at zero but
> assign weight based on the region entry block.
That's better than 0 :) but is there a plan to make the choice of
block more accurate in future? Using the entry block for something
that actually ends up in a loop would defeat the speed-based costing.
Richard
>
> This avoids regressing bb-slp-layout-18.c and bb-slp-pr54400.c
> with 2/2 which drops SLP_TREE_REPRESENTATIVE from all permute
> nodes.
>
> * tree-vect-slp.cc (vect_slp_node_weight): Get vinfo as
> context. For nodes without representative use the
> region entry block.
> (vect_optimize_slp_pass::start_choosing_layouts): Always
> assign vertex weight.
> ---
> gcc/tree-vect-slp.cc | 14 +++++++++-----
> 1 file changed, 9 insertions(+), 5 deletions(-)
>
> diff --git a/gcc/tree-vect-slp.cc b/gcc/tree-vect-slp.cc
> index 5e729e478ce..b2f14e6528a 100644
> --- a/gcc/tree-vect-slp.cc
> +++ b/gcc/tree-vect-slp.cc
> @@ -346,10 +346,14 @@ vect_free_oprnd_info (vec<slp_oprnd_info> &oprnds_info)
> a "more important" node when optimizing for speed). */
>
> static sreal
> -vect_slp_node_weight (slp_tree node)
> +vect_slp_node_weight (vec_info *vinfo, slp_tree node)
> {
> - stmt_vec_info stmt_info = vect_orig_stmt (SLP_TREE_REPRESENTATIVE (node));
> - basic_block bb = gimple_bb (stmt_info->stmt);
> + stmt_vec_info stmt_info = SLP_TREE_REPRESENTATIVE (node);
> + basic_block bb;
> + if (!stmt_info)
> + bb = vinfo->bbs[0];
> + else
> + bb = gimple_bb (vect_orig_stmt (stmt_info)->stmt);
> return bb->count.to_sreal_scale (ENTRY_BLOCK_PTR_FOR_FN (cfun)->count);
> }
>
> @@ -7520,10 +7524,10 @@ vect_optimize_slp_pass::start_choosing_layouts ()
> auto &partition = m_partitions[vertex.partition];
> slp_tree node = vertex.node;
>
> + vertex.weight = vect_slp_node_weight (m_vinfo, node);
> +
> if (stmt_vec_info rep = SLP_TREE_REPRESENTATIVE (node))
> {
> - vertex.weight = vect_slp_node_weight (node);
> -
> /* We do not handle stores with a permutation, so all
> incoming permutations must have been materialized.