On Mon, Aug 10, 2026 at 6:07 PM H.J. Lu <[email protected]> wrote:
>
> On Mon, Aug 10, 2026 at 5:47 PM H.J. Lu <[email protected]> wrote:
> >
> > On Sun, Aug 9, 2026 at 6:19 AM Andrew Pinski <[email protected]> wrote:
> > >
> > > On Tue, Aug 4, 2026 at 3:00 AM <[email protected]> wrote:
> > > >
> > > > From: Kyrylo Tkachov <[email protected]>
> > > >
> > > > The rule folding (type) minmax ((wide_type) a, (wide_type) b) to
> > > > minmax (a, b) is restricted to scalars. Extension is monotone, so it
> > > > commutes with the comparison and the outer truncation is exact, and the
> > > > argument is lanewise, so it holds for vectors unchanged.
> > > >
> > > > The vector case additionally needs the narrow operation to be available,
> > > > otherwise vector lowering would scalarise what used to be a single wide
> > > > instruction.
> > > >
> > > > typedef int v2si __attribute__((vector_size (8)));
> > > > typedef long v2di __attribute__((vector_size (16)));
> > > > v2si h (v2si a, v2si b)
> > > > {
> > > > v2di x = __builtin_convertvector (a, v2di);
> > > > v2di y = __builtin_convertvector (b, v2di);
> > > > return __builtin_convertvector (x < y ? x : y, v2si);
> > > > }
> > > >
> > > > aarch64 -O3 before:
> > > >
> > > > sshll v0.2d, v0.2s, 0
> > > > sshll v1.2d, v1.2s, 0
> > > > cmgt v2.2d, v1.2d, v0.2d
> > > > bsl v2.16b, v0.16b, v1.16b
> > > > xtn v0.2s, v2.2d
> > > >
> > > > after:
> > > >
> > > > smin v0.2s, v0.2s, v1.2s
> > > >
> > > > Require the wide vector MIN or MAX to have one use. A shared wide
> > > > result would otherwise gain a second narrow operation.
> > > >
> > > > Bootstrapped and tested on aarch64-none-linux-gnu.
> > > > Ok for trunk?
> > > > Thanks,
> > > > Kyrill
> > > >
> > > > gcc/ChangeLog:
> > > >
> > > > * match.pd ((type) minmax ((wide_type) a, (wide_type) b)):
> > > > Accept
> > > > vector types, using element_precision and requiring the narrow
> > > > operation to be supported.
> > > >
> > > > gcc/testsuite/ChangeLog:
> > > >
> > > > * g++.dg/tree-ssa/vec-narrow-1.C: New test.
> > > > * g++.dg/tree-ssa/vec-narrow-minmax-2.C: New test.
> > > >
> > > > Signed-off-by: Kyrylo Tkachov <[email protected]>
> > > > ---
> > > > gcc/match.pd | 20 +++++++++++++------
> > > > gcc/testsuite/g++.dg/tree-ssa/vec-narrow-1.C | 11 ++++++++++
> > > > .../g++.dg/tree-ssa/vec-narrow-minmax-2.C | 18 +++++++++++++++++
> > > > 3 files changed, 43 insertions(+), 6 deletions(-)
> > > > create mode 100644 gcc/testsuite/g++.dg/tree-ssa/vec-narrow-1.C
> > > > create mode 100644 gcc/testsuite/g++.dg/tree-ssa/vec-narrow-minmax-2.C
> > > >
> > > > diff --git a/gcc/match.pd b/gcc/match.pd
> > > > index e0ec2626842..adbdda91b51 100644
> > > > --- a/gcc/match.pd
> > > > +++ b/gcc/match.pd
> > > > @@ -4649,17 +4649,25 @@ DEFINE_INT_AND_FLOAT_ROUND_FN (RINT)
> > > >
> > > > /* (type) minmax ((wide_type) a, (wide_type) b) -> minmax (a, b)
> > > > when type matches the type of a and b, and wide_type is a wider
> > > > - type with the same signedness as type. */
> > > > + type with the same signedness as type. Extension is monotone, so it
> > > > + commutes with the comparison, and the truncation is then exact. The
> > > > + same holds lanewise for a vector, where the narrow operation has to
> > > > be
> > > > + available so that lowering does not scalarise it. The wide
> > > > operation
> > > > + must become dead before a narrow vector operation is introduced. */
> > > > (for minmax (min max)
> > > > + MINMAX (MIN_EXPR MAX_EXPR)
> > > > (simplify
> > > > - (convert (minmax:c (convert@2 @0) (convert@3 @1)))
> > > > - (if (INTEGRAL_TYPE_P (type)
> > > > - && INTEGRAL_TYPE_P (TREE_TYPE (@2))
> > > > + (convert (minmax:c@4 (convert@2 @0) (convert@3 @1)))
> > > > + (if (ANY_INTEGRAL_TYPE_P (type)
> > > > + && ANY_INTEGRAL_TYPE_P (TREE_TYPE (@2))
> > > > && types_match (type, TREE_TYPE (@0))
> > > > && types_match (type, TREE_TYPE (@1))
> > > > && types_match (TREE_TYPE (@2), TREE_TYPE (@3))
> > > > - && TYPE_PRECISION (TREE_TYPE (@2)) > TYPE_PRECISION (type)
> > > > - && TYPE_UNSIGNED (TREE_TYPE (@2)) == TYPE_UNSIGNED (type))
> > > > + && element_precision (TREE_TYPE (@2)) > element_precision (type)
> > > > + && TYPE_UNSIGNED (TREE_TYPE (@2)) == TYPE_UNSIGNED (type)
> > > > + && (!VECTOR_TYPE_P (type)
> > > > + || (single_use (@4)
> > > > + && target_supports_op_p (type, MINMAX, optab_vector))))
> > > > (minmax @0 @1))))
> > > >
> > > > /* max (a, a + CST) -> a + CST where CST is positive. */
> > > > diff --git a/gcc/testsuite/g++.dg/tree-ssa/vec-narrow-1.C
> > > > b/gcc/testsuite/g++.dg/tree-ssa/vec-narrow-1.C
> > > > new file mode 100644
> > > > index 00000000000..de738e33c30
> > > > --- /dev/null
> > > > +++ b/gcc/testsuite/g++.dg/tree-ssa/vec-narrow-1.C
> > > > @@ -0,0 +1,11 @@
> > > > +// { dg-do compile }
> > > > +// { dg-options "-O2 -fdump-tree-optimized" }
> > > > +// Extension is monotone, so it commutes with the comparison and the
> > > > outer
> > > > +// truncation is exact. The argument is lanewise, so a widened vector
> > > > +// MIN/MAX feeding a truncating conversion narrows.
> > > > +typedef int v2si __attribute__((vector_size (8)));
> > > > +typedef long long v2di __attribute__((vector_size (16)));
> > > > +v2si f (v2si a, v2si b)
> > > > +{ v2di x = __builtin_convertvector (a, v2di), y =
> > > > __builtin_convertvector (b, v2di);
> > > > + return __builtin_convertvector (x < y ? x : y, v2si); }
> > > > +// { dg-final { scan-tree-dump-not "vector\\(2\\) long" "optimized" } }
> > >
> > >
> > > This testcase fails on x86_64.
> >
> > This is needed:
> >
> > diff --git a/gcc/testsuite/g++.dg/tree-ssa/vec-narrow-minmax-2.C
> > b/gcc/testsuite/g++.dg/tree-ssa/vec-narrow-minmax-2.C
> > index d7fb7f06f16..f51730f08a7 100644
> > --- a/gcc/testsuite/g++.dg/tree-ssa/vec-narrow-minmax-2.C
> > +++ b/gcc/testsuite/g++.dg/tree-ssa/vec-narrow-minmax-2.C
> > @@ -1,5 +1,6 @@
> > // { dg-do compile }
> > // { dg-options "-O2 -fdump-tree-optimized" }
> > +// { dg-additional-options "-mavx512vl -mavx512dq" { target {
> > i?86-*-* x86_64-*-* } } }
> >
> > typedef int v2si __attribute__((vector_size (8)));
> > typedef long long v2di __attribute__((vector_size (16)));
> >
> > I am checking it in.
>
> These instructions are in SSE4.1. But due to
>
> https://gcc.gnu.org/bugzilla/show_bug.cgi?id=126767
>
> -msse4 is needed.
-msse4 is correct. But -msse4 generates:
pmovsxdq %xmm0, %xmm2
pmovsxdq %xmm1, %xmm1
movdqa %xmm2, %xmm0
movdqa %xmm2, %xmm3
pcmpgtq %xmm1, %xmm0
pblendvb %xmm0, %xmm1, %xmm3
movdqa %xmm3, %xmm0
movaps %xmm3, (%rdi)
shufps $232, %xmm3, %xmm0
while clang generates
pminsd %xmm1, %xmm0
pmovsxdq %xmm0, %xmm1
movdqa %xmm1, (%rdi)
> > > > diff --git a/gcc/testsuite/g++.dg/tree-ssa/vec-narrow-minmax-2.C
> > > > b/gcc/testsuite/g++.dg/tree-ssa/vec-narrow-minmax-2.C
> > > > new file mode 100644
> > > > index 00000000000..d7fb7f06f16
> > > > --- /dev/null
> > > > +++ b/gcc/testsuite/g++.dg/tree-ssa/vec-narrow-minmax-2.C
> > > > @@ -0,0 +1,18 @@
> > > > +// { dg-do compile }
> > > > +// { dg-options "-O2 -fdump-tree-optimized" }
> > > > +
> > > > +typedef int v2si __attribute__((vector_size (8)));
> > > > +typedef long long v2di __attribute__((vector_size (16)));
> > > > +
> > > > +v2si
> > > > +f (v2si a, v2si b, v2di *p)
> > > > +{
> > > > + v2di x = __builtin_convertvector (a, v2di);
> > > > + v2di y = __builtin_convertvector (b, v2di);
> > > > + v2di z = x < y ? x : y;
> > > > + *p = z;
> > > > + return __builtin_convertvector (z, v2si);
> > > > +}
> > > > +
> > > > +// A shared wide MIN must not gain a second narrow MIN.
> > > > +// { dg-final { scan-tree-dump-times "MIN_EXPR" 1 "optimized" } }
> > > > --
> > > > 2.50.1 (Apple Git-155)
> > > >
> >
> >
> >
> > --
> > H.J.
>
>
>
> --
> H.J.
--
H.J.