The 08/20/2026 15:26, Kyrylo Tkachov wrote: > > > On 20 Aug 2026, at 11:04, Tamar Christina <[email protected]> wrote: > > Hi Kyrill, > > -----Original Message----- > From: [email protected]<mailto:[email protected]> > <[email protected]<mailto:[email protected]>> > Sent: 17 August 2026 15:53 > To: [email protected]<mailto:[email protected]> > Cc: Tamar Christina > <[email protected]<mailto:[email protected]>>; > [email protected]<mailto:[email protected]>; Kyrylo > Tkachov <[email protected]<mailto:[email protected]>> > Subject: [PATCH] middle-end: Try intermediate types for widening sums > [PR122069] > > From: Kyrylo Tkachov <[email protected]> > > vect_recog_widen_sum_pattern currently queries the widening-sum optab > only > for the scalar input type in the source. This misses target patterns that > start at an intermediate precision. For example, AArch64 SVE has a > VNx2DI <- VNx8HI widening sum but no VNx2DI <- VNx8QI widening sum. > > Keep the exact input query and operand first. If the query fails, try > successively wider full-element integer types up to half the accumulator > width. Preserve the input signedness and use vect_convert_input to make the > intermediate conversion. The recognized pattern can then use: > > patt_1 = (short int) byte; > patt_2 = patt_1 w+ sum_0; > > For an unsigned byte-to-64-bit reduction on AArch64 SVE, the old main loop > processes VL / 64 input bytes per iteration: > > cntd x3 > whilelo p7.d, xzr, x1 > movi d30, #0 > ptrue p6.b, all > .L3: > ld1b z29.d, p7/z, [x0, x2] > add x2, x2, x3 > add z30.d, p7/m, z30.d, z29.d > whilelo p7.d, x2, x1 > b.any .L3 > uaddv d31, p6, z30.d > > The new main loop processes VL / 16 input bytes per iteration: > > cnth x0 > movi d31, #0 > ptrue p7.b, all > mov z29.h, #1 > .L4: > ld1b z30.h, p7/z, [x3, x2] > add x2, x2, x0 > udot z31.d, z30.h, z29.h > cmp x4, x2 > bcs .L4 > uaddv d31, p7, z31.d > > For the same number of input bytes, the old main loop executes four times > and > issues four widened loads and four vector adds. The new loop executes once > and issues one widened load and one dot product. > The signed form similarly uses LD1SB and SDOT. > Direct byte-to-32-bit dot products do not change. > > Bootstrapped and tested on aarch64-none-linux-gnu. > Ok for trunk? > Thanks, > Kyrill > > gcc/ChangeLog: > > PR middle-end/122069 > * tree-vect-patterns.cc (vect_recog_widen_sum_pattern): Try wider > intermediate input types. > > gcc/testsuite/ChangeLog: > > PR middle-end/122069 > * gcc.target/aarch64/sve/reduc_3_costly.c: Update the expected > number > of horizontal reductions. > * gcc.target/aarch64/sve/widen_sum_1.c: New test. > > Signed-off-by: Kyrylo Tkachov <[email protected]> > --- > .../gcc.target/aarch64/sve/reduc_3_costly.c | 2 +- > .../gcc.target/aarch64/sve/widen_sum_1.c | 84 +++++++++++++++++++ > gcc/tree-vect-patterns.cc | 28 ++++++- > 3 files changed, 111 insertions(+), 3 deletions(-) > create mode 100644 gcc/testsuite/gcc.target/aarch64/sve/widen_sum_1.c > > diff --git a/gcc/testsuite/gcc.target/aarch64/sve/reduc_3_costly.c > b/gcc/testsuite/gcc.target/aarch64/sve/reduc_3_costly.c > index 988459df8af..7cef352a9b7 100644 > --- a/gcc/testsuite/gcc.target/aarch64/sve/reduc_3_costly.c > +++ b/gcc/testsuite/gcc.target/aarch64/sve/reduc_3_costly.c > @@ -27,6 +27,6 @@ REDUC_PTR (float, _Float16) > REDUC_PTR (double, float) > > /* { dg-final { scan-assembler-times {\tuaddv\td[0-9]+, p[0-7], z[0-9]+\.s\n} > 2 } } */ > -/* { dg-final { scan-assembler-times {\tuaddv\td[0-9]+, p[0-7], z[0-9]+\.d\n} > 3 } } */ > +/* { dg-final { scan-assembler-times {\tuaddv\td[0-9]+, p[0-7], z[0-9]+\.d\n} > 4 } } */ > /* { dg-final { scan-assembler-times {\tfaddv\ts[0-9]+, p[0-7], z[0-9]+\.s\n} > 1 > } } */ > /* { dg-final { scan-assembler-times {\tfaddv\td[0-9]+, p[0-7], z[0-9]+\.d\n} > 1 > } } */ > diff --git a/gcc/testsuite/gcc.target/aarch64/sve/widen_sum_1.c > b/gcc/testsuite/gcc.target/aarch64/sve/widen_sum_1.c > new file mode 100644 > index 00000000000..3ae9d14dafd > --- /dev/null > +++ b/gcc/testsuite/gcc.target/aarch64/sve/widen_sum_1.c > @@ -0,0 +1,84 @@ > +/* { dg-do compile } */ > +/* { dg-require-effective-target lp64 } */ > +/* { dg-options "-O3 -march=armv8-a+sve -mautovec-preference=sve-only -- > param vect-epilogues-nomask=0 -fdump-tree-vect-details" } */ > +/* { dg-final { check-function-bodies "**" "" } } */ > + > +#define DEF(NAME, INPUT, OUTPUT) \ > + OUTPUT \ > + NAME (const INPUT *a, long n) \ > + { \ > + OUTPUT sum = 0; \ > + for (long i = 0; i < n; ++i) \ > + sum += a[i]; \ > + return sum; \ > + } > + > +/* > +** sum_u8_long: > +** ... > +** cnth x[0-9]+ > +** ... > +** ptrue (p[0-7])\.b, all > +** mov (z[0-9]+\.h), #1 > +** ... > +** ld1b (z[0-9]+\.h), \1/z, \[[^]]*\] > +** ... > +** udot (z[0-9]+)\.d, \3, \2 > +** ... > +** uaddv d[0-9]+, \1, \4\.d > +** ... > +*/ > +DEF (sum_u8_long, unsigned char, long) > + > +/* > +** sum_s8_long: > +** ... > +** cnth x[0-9]+ > +** ... > +** ptrue (p[0-7])\.b, all > +** mov (z[0-9]+\.h), #1 > +** ... > +** ld1sb (z[0-9]+\.h), \1/z, \[[^]]*\] > +** ... > +** sdot (z[0-9]+)\.d, \3, \2 > +** ... > +** uaddv d[0-9]+, \1, \4\.d > +** ... > +*/ > +DEF (sum_s8_long, signed char, long) > + > +/* > +** sum_u8_int: > +** ... > +** cntb x[0-9]+ > +** ... > +** ptrue (p[0-7])\.b, all > +** mov (z[0-9]+\.b), #1 > +** ... > +** ld1b (z[0-9]+\.b), \1/z, \[[^]]*\] > +** ... > +** udot (z[0-9]+)\.s, \3, \2 > +** ... > +** uaddv d[0-9]+, \1, \4\.s > +** ... > +*/ > +DEF (sum_u8_int, unsigned char, int) > + > +/* > +** sum_s8_int: > +** ... > +** cntb x[0-9]+ > +** ... > +** ptrue (p[0-7])\.b, all > +** mov (z[0-9]+\.b), #1 > +** ... > +** ld1b (z[0-9]+\.b), \1/z, \[[^]]*\] > +** ... > +** sdot (z[0-9]+)\.s, \3, \2 > +** ... > +** uaddv d[0-9]+, \1, \4\.s > +** ... > +*/ > +DEF (sum_s8_int, signed char, int) > + > +/* { dg-final { scan-tree-dump-times "LOOP VECTORIZED" 4 "vect" } } */ > diff --git a/gcc/tree-vect-patterns.cc b/gcc/tree-vect-patterns.cc > index fd1232cc755..1ee96b9f225 100644 > --- a/gcc/tree-vect-patterns.cc > +++ b/gcc/tree-vect-patterns.cc > @@ -2570,12 +2570,36 @@ vect_recog_widen_sum_pattern (vec_info > *vinfo, > > vect_pattern_detected ("vect_recog_widen_sum_pattern", last_stmt); > > + tree input = unprom0.op; > if (!vect_supportable_conv_optab_p (vinfo, type, WIDEN_SUM_EXPR, > unprom0.type, type_out)) > - return NULL; > + { > + /* Try widening the input to an intermediate type before adding it to > + the accumulator. Start with the narrowest type in order to retain > + the largest vectorization factor. */ > + tree input_type = unprom0.type; > + tree input_vectype = NULL_TREE; > + for (unsigned int precision > + = vect_element_precision (TYPE_PRECISION (input_type) + 1); > + precision <= TYPE_PRECISION (type) / 2; > + precision *= 2) > + { > + input_type = build_nonstandard_integer_type > + (precision, TYPE_UNSIGNED (unprom0.type)); > + if (vect_supportable_conv_optab_p (vinfo, type, WIDEN_SUM_EXPR, > + input_type, type_out, > &input_vectype)) > + break; > + } > + > + if (!input_vectype) > + return NULL; > + > + input = vect_convert_input (vinfo, stmt_vinfo, input_type, &unprom0, > + input_vectype); > + } > > Patch itself is OK but I think we can simplify this code a bit, what about: > > tree input_type = unprom0.type; > tree input_vectype = NULL_TREE; > > while (!vect_supportable_conv_optab_p (vinfo, type, WIDEN_SUM_EXPR, > input_type, type_out, &input_vectype)) > { > unsigned int precision > = vect_element_precision (TYPE_PRECISION (input_type) + 1); > > if (precision > TYPE_PRECISION (type) / 2) > return NULL; > > input_type = build_nonstandard_integer_type > (precision, TYPE_UNSIGNED (unprom0.type)); > } > > tree input = vect_convert_input (vinfo, stmt_vinfo, input_type, &unprom0, > input_vectype); > > instead? > > OK with that change. > > Thanks, here’s what I’m pushing. > Kyrill
Is that not just the original patch without the changes above? Thanks, Tamar > > > Thanks, > Tamar > > > var = vect_recog_temp_ssa_var (type, NULL); > - pattern_stmt = gimple_build_assign (var, WIDEN_SUM_EXPR, unprom0.op, > oprnd1); > + pattern_stmt = gimple_build_assign (var, WIDEN_SUM_EXPR, input, > oprnd1); > > return pattern_stmt; > } > -- > 2.50.1 (Apple Git-155) > --
