On 20/07/2026 16:06, Christopher Bazley wrote:
Hi again,

On 19/07/2026 19:56, Christopher Bazley wrote:
On 24/06/2026 11:42, Richard Biener wrote:
On Wed, Jun 3, 2026 at 5:21 PM Christopher Bazley <[email protected]> wrote:
@@ -14106,7 +14159,36 @@ vect_maybe_update_slp_op_vectype (vec_info *vinfo, slp_tree op, tree vectype)
        && SLP_TREE_DEF_TYPE (op) == vect_external_def
        && SLP_TREE_LANES (op) > 1)
      return false;
-  (void) vinfo; /* FORNOW */
+
+  /* When the vectorizer falls back to building vector operands from scalars, +     it can create SLP trees with external defs that have a number of lanes not +     divisible by the number of subparts in a vector type naively inferred from +     the scalar type.  Reject such types to avoid ICE when later computing the
+     prologue cost for invariant operands.  */
+  if (SLP_TREE_DEF_TYPE (op) == vect_external_def)
+    {
+      poly_uint64 vf = 1;
+
+      if (loop_vec_info loop_vinfo = dyn_cast<loop_vec_info> (vinfo))
+       vf = LOOP_VINFO_VECT_FACTOR (loop_vinfo);
+
+      vf *= SLP_TREE_LANES (op);
+
+      if (maybe_lt (TYPE_VECTOR_SUBPARTS (vectype), vf)
+         && !multiple_p (vf, TYPE_VECTOR_SUBPARTS (vectype)))

So this seems to be a guard for vect_get_num_copies?  It seems to me
that if this is only a problem to costing we should fix up there, not here.

After digging out my notes from October last year, I can confirm that I added that guard to prevent vect_get_num_copies -> vect_get_num_vectors from crashing when called from vect_prologue_cost_for_slp:

Final SLP tree for instance 0x4cee590:
node 0x4d6a5d0 (nunits.min=4, nunits.max=4, refcnt=2) vector(4) unsigned int
op template: MEM[(struct inflate_state *)_1].ncode = _45;
stmt 0 MEM[(struct inflate_state *)_1].ncode = _45;
stmt 1 MEM[(struct inflate_state *)_1].nlen = _39;
stmt 2 MEM[(struct inflate_state *)_1].ndist = _42;
children 0x4d6a690

node 0x4d6a690 (nunits.min=4, nunits.max=4, refcnt=2) vector(4) unsigned int
op template: _45 = _44 + 4;
stmt 0 _45 = _44 + 4;
stmt 1 _39 = _38 + 257;
stmt 2 _42 = _41 + 1;
children 0x4d6a750 0x4d6ab10

node 0x4d6a750 (nunits.min=4, nunits.max=4, refcnt=2) vector(4) unsigned int
op template: _44 = _43 & 15;
stmt 0 _44 = _43 & 15;
stmt 1 _38 = _37 & 31;
stmt 2 _41 = _40 & 31;
children 0x4d6a810 0x4d6aa50

node 0x4d6a810 (nunits.min=4, nunits.max=4, refcnt=2) vector(4) unsigned int
op template: _43 = (unsigned int) hold_463;
stmt 0 _43 = (unsigned int) hold_463;
stmt 1 _37 = (unsigned int) hold_1020;
stmt 2 _40 = (unsigned int) hold_460;
children 0x4d6a990

node (external) 0x4d6a990 (nunits.min=18446744073709551615, nunits.max=1, refcnt=1)
{ hold_463, hold_1020, hold_460 }

node (constant) 0x4d6aa50 (nunits.min=18446744073709551615, nunits.max=1, refcnt=1)
{ 15, 31, 31 }

node (constant) 0x4d6ab10 (nunits.min=18446744073709551615, nunits.max=1, refcnt=1)
{ 4, 257, 1 }

(gdb)

#5  0x0000000001f9c498 in vect_prologue_cost_for_slp (
     vinfo=0x4e4e2e0, node=0x4d6a990, cost_vec=0xffffffffdca0)
     at ../../gcc/tree-vect-slp.cc:8269
8269      unsigned nvectors = vect_get_num_copies (vinfo, node);

(gdb) print *node
$42 = {
   children = {m_vec = 0x0},
   stmts = {m_vec = 0x0},
   ops = {m_vec = 0x4cbf1b0},
   representative = 0x0,
   cycle_info = {id = -1, reduc_idx = -1},
   load_permutation = {m_vec = 0x0},
   lane_permutation = {m_vec = 0x0},
   vectype = 0xfffff7436738,
   vec_defs = {m_vec = 0x0},
   refcnt = 1,
   nunits = {
     min = {coeffs = {18446744073709551615, 0}},
     max = {coeffs = {1, 0}}
   },
   def_type = vect_external_def,
   lanes = 3,
   code = ERROR_MARK,
   gs_scale = 0,
   gs_base = 0x0,
   ldst_lanes = false,
   avoid_stlf_fail = false,
   can_use_mask = false,
   can_use_len = false,
   can_use_partial_vectors = true,
   vertex = 4,
   type = undef_vec_info_type,
   data = 0x0,
   failed = 0x0,
   next_node = 0x4d6a810,
   prev_node = 0x4d6aa50
}

during GIMPLE pass: slp
dump file: libz_a-infback.c.192t.slp1

../../zlib/infback.c: In function ‘inflateBack’:
../../zlib/infback.c:242:13: internal compiler error: in exact_div, at poly-int.h:2179    242 | int ZEXPORT inflateBack(z_streamp strm, in_func in, void FAR *in_desc,
       |             ^~~~~~~~~~~

0x3a162bb internal_error(char const*, ...)
     ../../gcc/diagnostic-global-context.cc:786
0x3a25c2f fancy_abort(char const*, int, char const*)
     ../../gcc/diagnostics/context.cc:1803
0x151916f poly_int<2u, poly_result<unsigned long, if_nonpoly<unsigned long, unsigned long, poly_int_traits<unsigned long>::is_poly>::type, poly_coeff_pair_traits<unsigned long, if_nonpoly<unsigned long, unsigned long, poly_int_traits<unsigned long>::is_poly>::type>::result_kind>::type> exact_div<2u, unsigned long, unsigned long>(poly_int<2u, unsigned long> const&, unsigned long)
     ../../gcc/poly-int.h:2179
0x151494f poly_int<2u, poly_result<unsigned long, unsigned long, poly_coeff_pair_traits<unsigned long, unsigned long>::result_kind>::type> exact_div<2u, unsigned long, unsigned long>(poly_int<2u, unsigned long> const&, poly_int<2u, unsigned long> const&)
     ../../gcc/poly-int.h:2192
0x1f36dcb vect_get_num_vectors(poly_int<2u, unsigned long>, tree_node*)
     ../../gcc/tree-vectorizer.h:2315
0x1f36ed3 vect_get_num_copies(vec_info*, _slp_tree*)
     ../../gcc/tree-vectorizer.h:2336
0x1f9c497 vect_prologue_cost_for_slp
     ../../gcc/tree-vect-slp.cc:8269
0x1f9cccf vect_slp_analyze_node_operations
     ../../gcc/tree-vect-slp.cc:8432
0x1f9c8c7 vect_slp_analyze_node_operations
     ../../gcc/tree-vect-slp.cc:8355
0x1f9c8c7 vect_slp_analyze_node_operations
     ../../gcc/tree-vect-slp.cc:8355
0x1f9c8c7 vect_slp_analyze_node_operations
     ../../gcc/tree-vect-slp.cc:8355
0x1f9de5b vect_slp_analyze_operations(vec_info*)
     ../../gcc/tree-vect-slp.cc:8782
0x1fa1e13 vect_slp_analyze_bb_1
     ../../gcc/tree-vect-slp.cc:9829
0x1fa1feb vect_slp_region
     ../../gcc/tree-vect-slp.cc:9876
0x1fa3127 vect_slp_bbs
     ../../gcc/tree-vect-slp.cc:10099
0x1fa38c3 vect_slp_function(function*)
     ../../gcc/tree-vect-slp.cc:10221
0x1fd18bb execute
     ../../gcc/tree-vectorizer.cc:1568

As I understand for SVE you're always having a single vector given
you want to use AdvSIMD for full vector parts, like with 7 int lanes
you force split during SLP discovery to get 4 lanes AdvSIMD and
3 lanes with predicated tail and SVE?  On x86 we can share the

At the assembly language level, that is right:

     adrp    x1, .LANCHOR0
     add    x2, x1, :lo12:.LANCHOR0
     ldr    q29, [x0]              ; ASIMD load of 4 ints
     ptrue    p7.s, vl3
     ldr    q30, [x2, 16]
     ldr    q28, [x1, #:lo12:.LANCHOR0]
     add    z28.s, z28.s, z29.s
     str    q28, [x0], 16           ; ASIMD store of 4 ints
     ld1w    z31.s, p7/z, [x0]       ; SVE load of 7 ints
     add    z30.s, z31.s, z30.s
     st1w    z30.s, p7, [x0]         ; SVE store of 7 ints

But in GIMPLE, the same (VLA) vector type is used for both groups after the split:

note:   starting SLP discovery for node 0x5390530
note:   get vectype for scalar type (group size 7): uint32_t
note:   vectype: vector([4,4]) unsigned int
note:   nunits = [4,4]
missed:   Build SLP failed: unrolling required in basic block SLP
note:   Build SLP for *x_16(D) = _2;
note:   Build SLP for MEM[(uint32_t *)x_16(D) + 4B] = _4;
note:   Build SLP for MEM[(uint32_t *)x_16(D) + 8B] = _6;
note:   Build SLP for MEM[(uint32_t *)x_16(D) + 12B] = _8;
note:   Build SLP for MEM[(uint32_t *)x_16(D) + 16B] = _10;
note:   Build SLP for MEM[(uint32_t *)x_16(D) + 20B] = _12;
note:   Build SLP for MEM[(uint32_t *)x_16(D) + 24B] = _14;
note:   SLP discovery for node 0x5390530 failed
note:   Splitting SLP group at stmt 4
note:   Split group into 4 and 3

...

note:   Final SLP tree for instance 0x52ceac0:
note:   node 0x53905e8 (max_nunits=4, refcnt=2) vector([4,4]) unsigned int
note:   op template: *x_16(D) = _2;
note:       stmt 0 *x_16(D) = _2;
note:       stmt 1 MEM[(uint32_t *)x_16(D) + 4B] = _4;
note:       stmt 2 MEM[(uint32_t *)x_16(D) + 8B] = _6;
note:       stmt 3 MEM[(uint32_t *)x_16(D) + 12B] = _8;
note:       children 0x53906a0
note:   node 0x53906a0 (max_nunits=4, refcnt=2) vector([4,4]) unsigned int
note:   op template: _2 = _1 + 1;
note:       stmt 0 _2 = _1 + 1;
note:       stmt 1 _4 = _3 + 2;
note:       stmt 2 _6 = _5 + 1;
note:       stmt 3 _8 = _7 + 2;
note:       children 0x5390758 0x5390810
note:   node 0x5390758 (max_nunits=4, refcnt=2) vector([4,4]) unsigned int
note:   op template: _1 = *x_16(D);
note:       stmt 0 _1 = *x_16(D);
note:       stmt 1 _3 = MEM[(uint32_t *)x_16(D) + 4B];
note:       stmt 2 _5 = MEM[(uint32_t *)x_16(D) + 8B];
note:       stmt 3 _7 = MEM[(uint32_t *)x_16(D) + 12B];
note:       load permutation { 0 1 2 3 }
note:   node (constant) 0x5390810 (max_nunits=1, refcnt=1)
note:       { 1, 2, 1, 2 }

...

note:   Final SLP tree for instance 0x52e95d0:
note:   node 0x53908c8 (max_nunits=4, refcnt=2) vector([4,4]) unsigned int
note:   op template: MEM[(uint32_t *)x_16(D) + 16B] = _10;
note:       stmt 0 MEM[(uint32_t *)x_16(D) + 16B] = _10;
note:       stmt 1 MEM[(uint32_t *)x_16(D) + 20B] = _12;
note:       stmt 2 MEM[(uint32_t *)x_16(D) + 24B] = _14;
note:       children 0x5390980
note:   node 0x5390980 (max_nunits=4, refcnt=2) vector([4,4]) unsigned int
note:   op template: _10 = _9 + 1;
note:       stmt 0 _10 = _9 + 1;
note:       stmt 1 _12 = _11 + 2;
note:       stmt 2 _14 = _13 + 1;
note:       children 0x5390a38 0x5390af0
note:   node 0x5390a38 (max_nunits=4, refcnt=2) vector([4,4]) unsigned int
note:   op template: _9 = MEM[(uint32_t *)x_16(D) + 16B];
note:       stmt 0 _9 = MEM[(uint32_t *)x_16(D) + 16B];
note:       stmt 1 _11 = MEM[(uint32_t *)x_16(D) + 20B];
note:       stmt 2 _13 = MEM[(uint32_t *)x_16(D) + 24B];
note:       load permutation { 4 5 6 }
note:   node (constant) 0x5390af0 (max_nunits=1, refcnt=1)
note:       { 1, 2, 1 }

vector type so technically we do not need to force split there
(but I think there's no harm done if doing so).  So -- why does

   if (known_ge (TYPE_VECTOR_SUBPARTS (vectype), vf))
     return 1;

in vect_get_num_vectors not trigger and solve the issue?

Before my change in vect_maybe_update_slp_op_vectype, I think the call that provoked the ICE must have been vect_get_num_vectors(3, V4SI), which failed because 3 % 4 != 0.  I agree that known_ge

This was a false hypothesis.  I can find no evidence for it.

(TYPE_VECTOR_SUBPARTS (V4SI), 3) should have been true in that circumstance.  That suggests that I observed the ICE in exact_div (and then modified vect_maybe_update_slp_op_vectype to fix it) before I made an unrelated change to make vect_get_num_copies more robust (probably for other reasons).  I will try reverting my fix in vect_maybe_update_slp_op_vectype to confirm that it is now redundant. If it is redundant then I will remove it; otherwise, I will look into modifying vect_prologue_cost_for_slp to avoid the ICE in a different way.

Having experimentally reverted my change to vect_maybe_update_slp_op_vectype, I discovered that the ICE in vect_get_num_copies reoccurred with vect_get_num_vectors (3, V2DI) on x64 when called from the same place: vect_prologue_cost_for_slp.  Note that the group size is greater than the number of subparts in the vector type.  That means known_ge (TYPE_VECTOR_SUBPARTS (vectype), vf) is false in vect_get_num_copies, therefore that 'if' statement that you asked about cannot solve this issue.

This is the problematic part of the source code (in zlib/inflate.c):

#define NEEDBITS(n) \
     do { \
         while (bits < (unsigned)(n)) \
             PULLBYTE(); \
     } while (0)

/* Return the low n bits of the bit accumulator (n < 16) */
#define BITS(n) \
     ((unsigned)hold & ((1U << (n)) - 1))

/* Remove n bits from the bit accumulator */
#define DROPBITS(n) \
     do { \
         hold >>= (n); \
         bits -= (unsigned)(n); \
     } while (0)
...
unsigned long hold;         /* bit buffer */
...
         case TABLE:
             NEEDBITS(14);
             state->nlen = BITS(5) + 257;
             DROPBITS(5);
             state->ndist = BITS(5) + 1;
             DROPBITS(5);
             state->ncode = BITS(4) + 4;
             DROPBITS(4);

which operates on these struct members:

     unsigned ncode;             /* number of code length code lengths */
     unsigned nlen;              /* number of length code lengths */
     unsigned ndist;             /* number of distance code lengths */

The relevant part of the slp1 dump is:

note:   starting SLP discovery for node 0x62444e0
note:   get vectype for scalar type (group size 3): unsigned int
note:   vectype: vector(4) unsigned int
note:   nunits = 4
note:   Build SLP for state_848->ncode = _220;
note:   Build SLP for state_848->nlen = _214;
note:   Build SLP for state_848->ndist = _217;
note:   vect_is_simple_use: operand _219 + 4, type of def: internal
note:   vect_is_simple_use: operand _213 + 257, type of def: internal
note:   vect_is_simple_use: operand _216 + 1, type of def: internal

note:   starting SLP discovery for node 0x6244590
note:   get vectype for scalar type (group size 3): unsigned int
note:   vectype: vector(4) unsigned int
note:   nunits = 4
note:   Build SLP for _220 = _219 + 4;
note:   Build SLP for _214 = _213 + 257;
note:   Build SLP for _217 = _216 + 1;
note:   vect_is_simple_use: operand _218 & 15, type of def: internal
note:   vect_is_simple_use: operand 4, type of def: constant
note:   vect_is_simple_use: operand _212 & 31, type of def: internal
note:   vect_is_simple_use: operand 257, type of def: constant
note:   vect_is_simple_use: operand _215 & 31, type of def: internal
note:   vect_is_simple_use: operand 1, type of def: constant

note:   starting SLP discovery for node 0x6244640
note:   get vectype for scalar type (group size 3): unsigned int
note:   vectype: vector(4) unsigned int
note:   nunits = 4
note:   Build SLP for _219 = _218 & 15;
note:   Build SLP for _213 = _212 & 31;
note:   Build SLP for _216 = _215 & 31;
note:   vect_is_simple_use: operand (unsigned int) hold_892, type of def: internal
note:   vect_is_simple_use: operand 15, type of def: constant
note:   vect_is_simple_use: operand (unsigned int) hold_1726, type of def: internal
note:   vect_is_simple_use: operand 31, type of def: constant
note:   vect_is_simple_use: operand (unsigned int) hold_889, type of def: internal
note:   vect_is_simple_use: operand 31, type of def: constant

note:   starting SLP discovery for node 0x62446f0
note:   get vectype for scalar type (group size 3): unsigned int
note:   vectype: vector(4) unsigned int
note:   nunits = 4
note:   Build SLP for _218 = (unsigned int) hold_892;
note:   Build SLP for _212 = (unsigned int) hold_1726;
note:   Build SLP for _215 = (unsigned int) hold_889;
note:   vect_is_simple_use: operand hold_1726 >> 10, type of def: internal
note:   vect_is_simple_use: operand hold_1726 = PHI <hold_903(178), hold_641(19)>, type of def: internal
note:   vect_is_simple_use: operand hold_1726 >> 5, type of def: internal

note:   starting SLP discovery for node 0x62447a0
note:   get vectype for scalar type (group size 3): long unsigned int
note:   vectype: vector(2) long unsigned int
note:   nunits = 2
missed:   Build SLP failed: unrolling required in basic block SLP
note:   Build SLP for hold_892 = hold_1726 >> 10;
note:   Build SLP for hold_1726 = PHI <hold_903(178), hold_641(19)>
missed:   Build SLP failed: different operation in stmt hold_1726 = PHI <hold_903(178), hold_641(19)>
missed:   original stmt hold_892 = hold_1726 >> 10;
note:   Build SLP for hold_889 = hold_1726 >> 5;
missed:   Build SLP failed: different shift arguments in hold_889 = hold_1726 >> 5;
note:   SLP discovery for node 0x62447a0 failed
note:   Building vector operands from scalars

note:   SLP discovery for node 0x62446f0 succeeded
note:   SLP discovery for node 0x6244640 succeeded
note:   SLP discovery for node 0x6244590 succeeded
note:   SLP discovery for node 0x62444e0 succeeded

note:   SLP size 5 vs. limit 1240.
note:   Final SLP tree for instance 0x62808b0:
note:   node 0x62444e0 (max_nunits=4, refcnt=2) vector(4) unsigned int
note:   op template: state_848->ncode = _220;
note:       stmt 0 state_848->ncode = _220;
note:       stmt 1 state_848->nlen = _214;
note:       stmt 2 state_848->ndist = _217;
note:       children 0x6244590
note:   node 0x6244590 (max_nunits=4, refcnt=2) vector(4) unsigned int
note:   op template: _220 = _219 + 4;
note:       stmt 0 _220 = _219 + 4;
note:       stmt 1 _214 = _213 + 257;
note:       stmt 2 _217 = _216 + 1;
note:       children 0x6244640 0x62449b0
note:   node 0x6244640 (max_nunits=4, refcnt=2) vector(4) unsigned int
note:   op template: _219 = _218 & 15;
note:       stmt 0 _219 = _218 & 15;
note:       stmt 1 _213 = _212 & 31;
note:       stmt 2 _216 = _215 & 31;
note:       children 0x62446f0 0x6244900
note:   node 0x62446f0 (max_nunits=4, refcnt=2) vector(4) unsigned int
note:   op template: _218 = (unsigned int) hold_892;
note:       stmt 0 _218 = (unsigned int) hold_892;
note:       stmt 1 _212 = (unsigned int) hold_1726;
note:       stmt 2 _215 = (unsigned int) hold_889;
note:       children 0x6244850
note:   node (external) 0x6244850 (max_nunits=1, refcnt=1)
note:       { hold_892, hold_1726, hold_889 }
note:   node (constant) 0x6244900 (max_nunits=1, refcnt=1)
note:       { 15, 31, 31 }
note:   node (constant) 0x62449b0 (max_nunits=1, refcnt=1)
note:       { 4, 257, 1 }

The node that fails in vect_prologue_cost_for_slp has:

def_type = vect_external_def
lanes    = 3
vectype  = V2DI

It is a three-element external definition pack:

{ hold_892, hold_1726, hold_889 }

Where each element is defined as:

hold_892 = hold_1726 >> 10;
hold_1726 = PHI <hold_903(178), hold_641(19)>
hold_889 = hold_1726 >> 5

The problematic vectype of V2DI is assigned to the external definition node by vect_maybe_update_slp_op_vectype (vinfo, slp_op0, vectype_in). At this point, the parent SLP node (passed to vectorizable_conversion) has:

   lanes = 3
   vectype = V4SI
   def_type = vect_internal_def

And, before its vectype is updated, the operand SLP node has:
   lanes = 3
   vectype = NULL initially
   def_type = vect_external_def

vectorizable_conversion derives:

lhs_type (from TREE_TYPE (scalar_dest)) = unsigned int
vectype_out (from SLP_TREE_VECTYPE (slp_node)) = unsigned V4SI
vectype_in (from vect_is_simple_use) = NULL_TREE
rhs_type (from TREE_TYPE (op0)) = unsigned DI
vectype_in (from get_vectype_for_scalar_type (.., rhs_type, slp_node)) = unsigned V2DI

It then classifies the operation as:

nunits_out (from TYPE_VECTOR_SUBPARTS (vectype_out)) = 4
nunits_in (from TYPE_VECTOR_SUBPARTS (vectype_in))  = 2

multiple_p (4, 2) == true
modifier = NARROW_DST

Given that get_vectype_for_scalar_type is supposed to select a vector type compatible with slp_node, and given that the general rule for my patch series is that SLP nodes must either occupy a whole number of vectors or one partial vector, I am inclined to think that get_vectype_for_scalar_type is at fault for choosing unsigned V2DI.  I think it should have chosen NULL_TREE instead.

Does that sound reasonable?

After further experimentation, I have changed my mind. It is not the role of get_vectype_for_scalar_type to enforce the invariant that SLP nodes must either occupy a whole number of vectors or one partial vector. Attempting to enforce that rule in get_vectype_for_scalar_type leads to absurd consequences such as trying to vectorise a group of 17 using type vector(1) unsigned char (which leads to "Build SLP failed: not using single lane vector type vector(1) unsigned char"). Even in non-prime cases such as a group of 6, it is not clear to me that V2QI + V2QI + V2QI would be preferable to V4QI + V2QI.

So, given that V2DI is a legitimate vector type for the DI scalar type, the next question is whether a three-lane external-def node can validly use that type.

My understanding is that choice of vectype for most SLP nodes (including types that could not have as many subparts as the group size) happens in

  vect_slp_region ->
    vect_slp_analyze_bb_1 ->
      vect_analyze_slp ->
        vect_analyze_slp_instance (owns matches[]) ->
          vect_build_slp_tree ->
            vect_build_slp_tree_2 ->
              vect_build_slp_tree_1 ->
                vect_get_vector_types_for_stmt

Splitting of SLP nodes because they have inappropriate vectypes happens during "check the SLP opportunities in the basic block, analyze and build SLP trees", using the matches[] array in

  vect_slp_region ->
    vect_slp_analyze_bb_1 ->
      vect_analyze_slp ->
        vect_analyze_slp_instance

In contrast, the questionable vectype of V2DI is assigned after control has returned from vect_analyze_slp_instance during "analyze statements in SLP instances of VINFO" in

  vect_slp_region ->
    vect_slp_analyze_bb_1 ->
      vect_slp_analyze_operations ->
        vect_slp_analyze_node_operations ->
          vect_slp_analyze_node_operations_1 ->
            vect_analyze_stmt ->
              vectorizable_conversion ->
                vect_maybe_update_slp_op_vectype

The SLP tree has already been built by this time, and there is no longer any matches[] array, so it does not seem right to think in terms of splitting the external node.

The vect_maybe_update_slp_op_vectype function already has existing code to identify conflicting vector types when SLP_TREE_DEF_TYPE (op) == vect_external_def, based on SLP_TREE_LANES. That's why I believed this was the right place enforce the invariant.

If vect_maybe_update_slp_op_vectype does not enforce the invariant during the costing/analysis phase then vect_prologue_cost_for_slp crashes:

  vect_slp_region ->
    vect_slp_analyze_bb_1 ->
      vect_slp_analyze_operations ->
        vect_slp_analyze_node_operations ->
          vect_slp_analyze_node_operations_1 ->
             ... vect_maybe_update_slp_op_vectype
          vect_prologue_cost_for_slp ->
            vect_get_num_copies

If I modify vect_prologue_cost_for_slp to work around the fact that the expected invariant does not hold then vect_analyze_stmt fails for unrelated reasons:

note:   ==> examining statement: state_848->ncode = _220;
note:   vect_is_simple_use: operand _219 + 4, type of def: internal
note: the target doesn't have the appropriate partial vectorization store for vector data type vector(4) unsigned int. missed: can't operate on partial vectors because the target doesn't have the appropriate partial vectorization load or store.
missed:   SLP store needs but cannot use partial vectors.
missed: not vectorized: relevant stmt not supported: state_848->ncode = _220; note: removing SLP instance operations starting from: state_848->ncode = _220;

However, supposing that vect_analyze_stmt had not failed for unrelated reasons, I believe that vect_create_constant_vectors would have crashed later for the same reason as vect_prologue_cost_for_slp:

  vect_slp_region ->
    vect_slp_analyze_bb_1 ->
      vect_slp_analyze_operations ->
        ... vect_maybe_update_slp_op_vectype
    vect_schedule_slp ->
      vect_schedule_scc ->
        vect_schedule_slp_node ->
          vect_create_constant_vectors ->
            vect_get_num_copies

Would you still prefer this invalid lanes/vectype combination to be handled separately in both costing and constant vector creation, instead of rejecting the problematic vectype assignment in vect_maybe_update_slp_op_vectype? My concern is that doing so might not be sufficient, because vect_get_num_copies has many other callers.

I am doubtful about making vect_get_num_copies robust for this input either, because in this instance the failed exact division appears to expose an invalid SLP-node lanes/vectype combination rather than merely a costing limitation.

The thread we're pulling at seems related to whether SLP nodes such as "one and a half" vectors should be allowed. I wonder whether that is something that could be deferred to a future patch series.

Please let me know.

Thanks,
--
Christopher Bazley
Staff Software Engineer, GNU Tools Team.
Arm Ltd, 110 Fulbourn Road, Cambridge, CB1 9NJ, UK.
http://www.arm.com/

Reply via email to