> On 20 Aug 2026, at 16:32, Tamar Christina <[email protected]> wrote:
> 
> The 08/20/2026 15:26, Kyrylo Tkachov wrote:
>> 
>> 
>> On 20 Aug 2026, at 11:04, Tamar Christina <[email protected]> wrote:
>> 
>> Hi Kyrill,
>> 
>> -----Original Message-----
>> From: [email protected]<mailto:[email protected]> 
>> <[email protected]<mailto:[email protected]>>
>> Sent: 17 August 2026 15:53
>> To: [email protected]<mailto:[email protected]>
>> Cc: Tamar Christina 
>> <[email protected]<mailto:[email protected]>>; 
>> [email protected]<mailto:[email protected]>; Kyrylo
>> Tkachov <[email protected]<mailto:[email protected]>>
>> Subject: [PATCH] middle-end: Try intermediate types for widening sums
>> [PR122069]
>> 
>> From: Kyrylo Tkachov <[email protected]>
>> 
>> vect_recog_widen_sum_pattern currently queries the widening-sum optab
>> only
>> for the scalar input type in the source.  This misses target patterns that
>> start at an intermediate precision.  For example, AArch64 SVE has a
>> VNx2DI <- VNx8HI widening sum but no VNx2DI <- VNx8QI widening sum.
>> 
>> Keep the exact input query and operand first.  If the query fails, try
>> successively wider full-element integer types up to half the accumulator
>> width.  Preserve the input signedness and use vect_convert_input to make the
>> intermediate conversion.  The recognized pattern can then use:
>> 
>> patt_1 = (short int) byte;
>> patt_2 = patt_1 w+ sum_0;
>> 
>> For an unsigned byte-to-64-bit reduction on AArch64 SVE, the old main loop
>> processes VL / 64 input bytes per iteration:
>> 
>>       cntd    x3
>>       whilelo p7.d, xzr, x1
>>       movi    d30, #0
>>       ptrue   p6.b, all
>> .L3:
>>       ld1b    z29.d, p7/z, [x0, x2]
>>       add     x2, x2, x3
>>       add     z30.d, p7/m, z30.d, z29.d
>>       whilelo p7.d, x2, x1
>>       b.any   .L3
>>       uaddv   d31, p6, z30.d
>> 
>> The new main loop processes VL / 16 input bytes per iteration:
>> 
>>       cnth    x0
>>       movi    d31, #0
>>       ptrue   p7.b, all
>>       mov     z29.h, #1
>> .L4:
>>       ld1b    z30.h, p7/z, [x3, x2]
>>       add     x2, x2, x0
>>       udot    z31.d, z30.h, z29.h
>>       cmp     x4, x2
>>       bcs     .L4
>>       uaddv   d31, p7, z31.d
>> 
>> For the same number of input bytes, the old main loop executes four times
>> and
>> issues four widened loads and four vector adds.  The new loop executes once
>> and issues one widened load and one dot product.
>> The signed form similarly uses LD1SB and SDOT.
>> Direct byte-to-32-bit dot products do not change.
>> 
>> Bootstrapped and tested on aarch64-none-linux-gnu.
>> Ok for trunk?
>> Thanks,
>> Kyrill
>> 
>> gcc/ChangeLog:
>> 
>> PR middle-end/122069
>> * tree-vect-patterns.cc (vect_recog_widen_sum_pattern): Try wider
>> intermediate input types.
>> 
>> gcc/testsuite/ChangeLog:
>> 
>> PR middle-end/122069
>> * gcc.target/aarch64/sve/reduc_3_costly.c: Update the expected
>> number
>> of horizontal reductions.
>> * gcc.target/aarch64/sve/widen_sum_1.c: New test.
>> 
>> Signed-off-by: Kyrylo Tkachov <[email protected]>
>> ---
>> .../gcc.target/aarch64/sve/reduc_3_costly.c   |  2 +-
>> .../gcc.target/aarch64/sve/widen_sum_1.c      | 84 +++++++++++++++++++
>> gcc/tree-vect-patterns.cc                     | 28 ++++++-
>> 3 files changed, 111 insertions(+), 3 deletions(-)
>> create mode 100644 gcc/testsuite/gcc.target/aarch64/sve/widen_sum_1.c
>> 
>> diff --git a/gcc/testsuite/gcc.target/aarch64/sve/reduc_3_costly.c
>> b/gcc/testsuite/gcc.target/aarch64/sve/reduc_3_costly.c
>> index 988459df8af..7cef352a9b7 100644
>> --- a/gcc/testsuite/gcc.target/aarch64/sve/reduc_3_costly.c
>> +++ b/gcc/testsuite/gcc.target/aarch64/sve/reduc_3_costly.c
>> @@ -27,6 +27,6 @@ REDUC_PTR (float, _Float16)
>> REDUC_PTR (double, float)
>> 
>> /* { dg-final { scan-assembler-times {\tuaddv\td[0-9]+, p[0-7], z[0-9]+\.s\n}
>> 2 } } */
>> -/* { dg-final { scan-assembler-times {\tuaddv\td[0-9]+, p[0-7], 
>> z[0-9]+\.d\n}
>> 3 } } */
>> +/* { dg-final { scan-assembler-times {\tuaddv\td[0-9]+, p[0-7], 
>> z[0-9]+\.d\n}
>> 4 } } */
>> /* { dg-final { scan-assembler-times {\tfaddv\ts[0-9]+, p[0-7], 
>> z[0-9]+\.s\n} 1
>> } } */
>> /* { dg-final { scan-assembler-times {\tfaddv\td[0-9]+, p[0-7], 
>> z[0-9]+\.d\n} 1
>> } } */
>> diff --git a/gcc/testsuite/gcc.target/aarch64/sve/widen_sum_1.c
>> b/gcc/testsuite/gcc.target/aarch64/sve/widen_sum_1.c
>> new file mode 100644
>> index 00000000000..3ae9d14dafd
>> --- /dev/null
>> +++ b/gcc/testsuite/gcc.target/aarch64/sve/widen_sum_1.c
>> @@ -0,0 +1,84 @@
>> +/* { dg-do compile } */
>> +/* { dg-require-effective-target lp64 } */
>> +/* { dg-options "-O3 -march=armv8-a+sve -mautovec-preference=sve-only --
>> param vect-epilogues-nomask=0 -fdump-tree-vect-details" } */
>> +/* { dg-final { check-function-bodies "**" "" } } */
>> +
>> +#define DEF(NAME, INPUT, OUTPUT) \
>> +  OUTPUT                        \
>> +  NAME (const INPUT *a, long n) \
>> +  {                             \
>> +    OUTPUT sum = 0;             \
>> +    for (long i = 0; i < n; ++i) \
>> +      sum += a[i];              \
>> +    return sum;                 \
>> +  }
>> +
>> +/*
>> +** sum_u8_long:
>> +** ...
>> +** cnth x[0-9]+
>> +** ...
>> +** ptrue (p[0-7])\.b, all
>> +** mov (z[0-9]+\.h), #1
>> +** ...
>> +** ld1b (z[0-9]+\.h), \1/z, \[[^]]*\]
>> +** ...
>> +** udot (z[0-9]+)\.d, \3, \2
>> +** ...
>> +** uaddv d[0-9]+, \1, \4\.d
>> +** ...
>> +*/
>> +DEF (sum_u8_long, unsigned char, long)
>> +
>> +/*
>> +** sum_s8_long:
>> +** ...
>> +** cnth x[0-9]+
>> +** ...
>> +** ptrue (p[0-7])\.b, all
>> +** mov (z[0-9]+\.h), #1
>> +** ...
>> +** ld1sb (z[0-9]+\.h), \1/z, \[[^]]*\]
>> +** ...
>> +** sdot (z[0-9]+)\.d, \3, \2
>> +** ...
>> +** uaddv d[0-9]+, \1, \4\.d
>> +** ...
>> +*/
>> +DEF (sum_s8_long, signed char, long)
>> +
>> +/*
>> +** sum_u8_int:
>> +** ...
>> +** cntb x[0-9]+
>> +** ...
>> +** ptrue (p[0-7])\.b, all
>> +** mov (z[0-9]+\.b), #1
>> +** ...
>> +** ld1b (z[0-9]+\.b), \1/z, \[[^]]*\]
>> +** ...
>> +** udot (z[0-9]+)\.s, \3, \2
>> +** ...
>> +** uaddv d[0-9]+, \1, \4\.s
>> +** ...
>> +*/
>> +DEF (sum_u8_int, unsigned char, int)
>> +
>> +/*
>> +** sum_s8_int:
>> +** ...
>> +** cntb x[0-9]+
>> +** ...
>> +** ptrue (p[0-7])\.b, all
>> +** mov (z[0-9]+\.b), #1
>> +** ...
>> +** ld1b (z[0-9]+\.b), \1/z, \[[^]]*\]
>> +** ...
>> +** sdot (z[0-9]+)\.s, \3, \2
>> +** ...
>> +** uaddv d[0-9]+, \1, \4\.s
>> +** ...
>> +*/
>> +DEF (sum_s8_int, signed char, int)
>> +
>> +/* { dg-final { scan-tree-dump-times "LOOP VECTORIZED" 4 "vect" } } */
>> diff --git a/gcc/tree-vect-patterns.cc b/gcc/tree-vect-patterns.cc
>> index fd1232cc755..1ee96b9f225 100644
>> --- a/gcc/tree-vect-patterns.cc
>> +++ b/gcc/tree-vect-patterns.cc
>> @@ -2570,12 +2570,36 @@ vect_recog_widen_sum_pattern (vec_info
>> *vinfo,
>> 
>>  vect_pattern_detected ("vect_recog_widen_sum_pattern", last_stmt);
>> 
>> +  tree input = unprom0.op;
>>  if (!vect_supportable_conv_optab_p (vinfo, type, WIDEN_SUM_EXPR,
>>      unprom0.type, type_out))
>> -    return NULL;
>> +    {
>> +      /* Try widening the input to an intermediate type before adding it to
>> +  the accumulator.  Start with the narrowest type in order to retain
>> +  the largest vectorization factor.  */
>> +      tree input_type = unprom0.type;
>> +      tree input_vectype = NULL_TREE;
>> +      for (unsigned int precision
>> +      = vect_element_precision (TYPE_PRECISION (input_type) + 1);
>> +    precision <= TYPE_PRECISION (type) / 2;
>> +    precision *= 2)
>> + {
>> +   input_type = build_nonstandard_integer_type
>> +     (precision, TYPE_UNSIGNED (unprom0.type));
>> +   if (vect_supportable_conv_optab_p (vinfo, type, WIDEN_SUM_EXPR,
>> + input_type, type_out,
>> &input_vectype))
>> +     break;
>> + }
>> +
>> +      if (!input_vectype)
>> + return NULL;
>> +
>> +      input = vect_convert_input (vinfo, stmt_vinfo, input_type, &unprom0,
>> +   input_vectype);
>> +    }
>> 
>> Patch itself is OK but I think we can simplify this code a bit, what about:
>> 
>> tree input_type = unprom0.type;
>> tree input_vectype = NULL_TREE;
>> 
>> while (!vect_supportable_conv_optab_p (vinfo, type, WIDEN_SUM_EXPR,
>>                                      input_type, type_out, &input_vectype))
>>   {
>>     unsigned int precision
>>       = vect_element_precision (TYPE_PRECISION (input_type) + 1);
>> 
>>     if (precision > TYPE_PRECISION (type) / 2)
>>       return NULL;
>> 
>>     input_type = build_nonstandard_integer_type
>>       (precision, TYPE_UNSIGNED (unprom0.type));
>>   }
>> 
>> tree input = vect_convert_input (vinfo, stmt_vinfo, input_type, &unprom0,
>>                                input_vectype);
>> 
>> instead?
>> 
>> OK with that change.
>> 
>> Thanks, here’s what I’m pushing.
>> Kyrill
> 
> Is that not just the original patch without the changes above?
> 

Yikes, I had messed up a git pull on one of my internal mirrors.
I’ve reverted the original commit and pushed this one instead.
Thanks,
Kyrill

> Thanks,
> Tamar
> 
>> 
>> 
>> Thanks,
>> Tamar
>> 
>> 
>>  var = vect_recog_temp_ssa_var (type, NULL);
>> -  pattern_stmt = gimple_build_assign (var, WIDEN_SUM_EXPR, unprom0.op,
>> oprnd1);
>> +  pattern_stmt = gimple_build_assign (var, WIDEN_SUM_EXPR, input,
>> oprnd1);
>> 
>>  return pattern_stmt;
>> }
>> --
>> 2.50.1 (Apple Git-155)
>> 
> 
> 
> 
> -- 

Attachment: 225e8b0a8577f99b865ea3e57d8835092759c375.patch
Description: 225e8b0a8577f99b865ea3e57d8835092759c375.patch

Reply via email to