Double unaligned load and store cost if the vector mode size is less
than 8 bytes and the number of vector element is less than 4 when SSE
is enabled to avoid using vector instructions on unaligned short data
with 2 vector elements so that for

extern char *var1;
extern int var2;

void
func (void)
{
  var2 = var1[1] + var1[0];
}

we generate

movq var1(%rip), %rdx
movsbl 1(%rdx), %eax
movsbl (%rdx), %edx
addl %edx, %eax
movl %eax, var2(%rip)

instead of

movq var1(%rip), %rax
pxor %xmm1, %xmm1
pinsrw $0, (%rax), %xmm0
pcmpgtb %xmm0, %xmm1
punpcklbw %xmm1, %xmm0
movdqa %xmm0, %xmm1
psraw $15, %xmm1
punpcklwd %xmm1, %xmm0
movd %xmm0, %edx
pshufd $0xe5, %xmm0, %xmm2
movd %xmm2, %eax
addl %edx, %eax
movl %eax, var2(%rip)

with -O2 -march=x86-64.

Compile PR 125100 tests with -mno-sse since unaligned V2QImode load is no
longer generated when SSE is enabled.

gcc/

PR target/126802
* config/i386/i386.cc (sse_adjust_unaligned_cost): New.
(ix86_default_vector_cost): Call sse_adjust_unaligned_cost for
unaligned load and store to adjust unaligned move cost.

testsuite/

PR target/126802
* gcc.target/i386/pr125100-1.c: Add -mno-sse.
* gcc.target/i386/pr125100-2.c: Likewise.
* gcc.target/i386/pr125100-3.c: Likewise.
* gcc.target/i386/pr126802-1a.c: New test.
* gcc.target/i386/pr126802-1b.c: Likewise.
* gcc.target/i386/pr126802-2a.c: Likewise.
* gcc.target/i386/pr126802-2b.c: Likewise.
* gcc.target/i386/pr126802-3a.c: Likewise.
* gcc.target/i386/pr126802-3b.c: Likewise.
* gcc.target/i386/pr126802-4a.c: Likewise.
* gcc.target/i386/pr126802-4b.c: Likewise.

Tested on Linux/x86-64 without any regressions.

--
H.J.
From 2f941db2bb2a27e223cd956b3ae4779eb291b820 Mon Sep 17 00:00:00 2001
From: "H.J. Lu" <[email protected]>
Date: Wed, 12 Aug 2026 06:57:32 +0800
Subject: [PATCH] x86: Double unaligned move cost for short data

Double unaligned load and store cost if the vector mode size is less
than 8 bytes and the number of vector element is less than 4 when SSE
is enabled to avoid using vector instructions on unaligned short data
with 2 vector elements so that for

extern char *var1;
extern int var2;

void
func (void)
{
  var2 = var1[1] + var1[0];
}

we generate

	movq	var1(%rip), %rdx
	movsbl	1(%rdx), %eax
	movsbl	(%rdx), %edx
	addl	%edx, %eax
	movl	%eax, var2(%rip)

instead of

	movq	var1(%rip), %rax
	pxor	%xmm1, %xmm1
	pinsrw	$0, (%rax), %xmm0
	pcmpgtb	%xmm0, %xmm1
	punpcklbw	%xmm1, %xmm0
	movdqa	%xmm0, %xmm1
	psraw	$15, %xmm1
	punpcklwd	%xmm1, %xmm0
	movd	%xmm0, %edx
	pshufd	$0xe5, %xmm0, %xmm2
	movd	%xmm2, %eax
	addl	%edx, %eax
	movl	%eax, var2(%rip)

with -O2 -march=x86-64.

Compile PR 125100 tests with -mno-sse since unaligned V2QImode load is no
longer generated when SSE is enabled.

gcc/

	PR target/126802
	* config/i386/i386.cc (sse_adjust_unaligned_cost): New.
	(ix86_default_vector_cost): Call sse_adjust_unaligned_cost for
	unaligned load and store to adjust unaligned move cost.

testsuite/

	PR target/126802
	* gcc.target/i386/pr125100-1.c: Add -mno-sse.
	* gcc.target/i386/pr125100-2.c: Likewise.
	* gcc.target/i386/pr125100-3.c: Likewise.
	* gcc.target/i386/pr126802-1a.c: New test.
	* gcc.target/i386/pr126802-1b.c: Likewise.
	* gcc.target/i386/pr126802-2a.c: Likewise.
	* gcc.target/i386/pr126802-2b.c: Likewise.
	* gcc.target/i386/pr126802-3a.c: Likewise.
	* gcc.target/i386/pr126802-3b.c: Likewise.
	* gcc.target/i386/pr126802-4a.c: Likewise.
	* gcc.target/i386/pr126802-4b.c: Likewise.

Signed-off-by: H.J. Lu <[email protected]>
---
 gcc/config/i386/i386.cc                     | 24 ++++++++++++--
 gcc/testsuite/gcc.target/i386/pr125100-1.c  |  2 +-
 gcc/testsuite/gcc.target/i386/pr125100-2.c  |  2 +-
 gcc/testsuite/gcc.target/i386/pr125100-3.c  |  2 +-
 gcc/testsuite/gcc.target/i386/pr126802-1a.c | 26 +++++++++++++++
 gcc/testsuite/gcc.target/i386/pr126802-1b.c | 19 +++++++++++
 gcc/testsuite/gcc.target/i386/pr126802-2a.c | 26 +++++++++++++++
 gcc/testsuite/gcc.target/i386/pr126802-2b.c | 19 +++++++++++
 gcc/testsuite/gcc.target/i386/pr126802-3a.c | 25 ++++++++++++++
 gcc/testsuite/gcc.target/i386/pr126802-3b.c | 18 +++++++++++
 gcc/testsuite/gcc.target/i386/pr126802-4a.c | 36 +++++++++++++++++++++
 gcc/testsuite/gcc.target/i386/pr126802-4b.c | 23 +++++++++++++
 12 files changed, 217 insertions(+), 5 deletions(-)
 create mode 100644 gcc/testsuite/gcc.target/i386/pr126802-1a.c
 create mode 100644 gcc/testsuite/gcc.target/i386/pr126802-1b.c
 create mode 100644 gcc/testsuite/gcc.target/i386/pr126802-2a.c
 create mode 100644 gcc/testsuite/gcc.target/i386/pr126802-2b.c
 create mode 100644 gcc/testsuite/gcc.target/i386/pr126802-3a.c
 create mode 100644 gcc/testsuite/gcc.target/i386/pr126802-3b.c
 create mode 100644 gcc/testsuite/gcc.target/i386/pr126802-4a.c
 create mode 100644 gcc/testsuite/gcc.target/i386/pr126802-4b.c

diff --git a/gcc/config/i386/i386.cc b/gcc/config/i386/i386.cc
index 780651ebf8c..21a09ab88d8 100644
--- a/gcc/config/i386/i386.cc
+++ b/gcc/config/i386/i386.cc
@@ -21517,6 +21517,22 @@ sse_store_index (machine_mode mode)
     }
 }
 
+/* Adjust unaligned load and store cost.  */
+
+static inline int
+sse_adjust_unaligned_cost (int cost, machine_mode mode)
+{
+  /* PR 126802 - Double unaligned load and store cost if the vector mode
+     size is less than 8 bytes and the number of vector element is less
+     than 4 when SSE is enabled.  */
+  if (TARGET_SSE
+      && VECTOR_MODE_P (mode)
+      && GET_MODE_SIZE (mode) < 8
+      && GET_MODE_NUNITS (mode) < 4)
+    cost += cost;
+  return cost;
+}
+
 /* Return the cost of moving data of mode M between a
    register and memory.  A value of 2 is the default; this cost is
    relative to those in `REGISTER_MOVE_COST'.
@@ -25989,14 +26005,18 @@ ix86_default_vector_cost (enum vect_cost_for_stmt type_of_cost,
 	/* See PR82713 - we may end up being called on non-vector type.  */
 	if (index < 0)
 	  index = 2;
-        return COSTS_N_INSNS (ix86_cost->sse_unaligned_load[index]) / 2;
+	return sse_adjust_unaligned_cost
+	  (COSTS_N_INSNS (ix86_cost->sse_unaligned_load[index]) / 2,
+	   mode);
 
       case unaligned_store:
 	index = sse_store_index (mode);
 	/* See PR82713 - we may end up being called on non-vector type.  */
 	if (index < 0)
 	  index = 2;
-        return COSTS_N_INSNS (ix86_cost->sse_unaligned_store[index]) / 2;
+	return sse_adjust_unaligned_cost
+	  (COSTS_N_INSNS (ix86_cost->sse_unaligned_store[index]) / 2,
+	   mode);
 
       case vector_gather_load:
         return ix86_vec_cost (mode,
diff --git a/gcc/testsuite/gcc.target/i386/pr125100-1.c b/gcc/testsuite/gcc.target/i386/pr125100-1.c
index 21765a5843e..d1f537e981b 100644
--- a/gcc/testsuite/gcc.target/i386/pr125100-1.c
+++ b/gcc/testsuite/gcc.target/i386/pr125100-1.c
@@ -1,5 +1,5 @@
 /* { dg-do compile { target fpic } } */
-/* { dg-options "-mtune=generic -O2 -fPIC" } */
+/* { dg-options "-mtune=generic -mno-sse -O2 -fPIC" } */
 /* { dg-additional-options "-march=pentiumpro" { target ia32 } } */
 
 struct desc {
diff --git a/gcc/testsuite/gcc.target/i386/pr125100-2.c b/gcc/testsuite/gcc.target/i386/pr125100-2.c
index d179c3f4050..23ca7d9a7a6 100644
--- a/gcc/testsuite/gcc.target/i386/pr125100-2.c
+++ b/gcc/testsuite/gcc.target/i386/pr125100-2.c
@@ -1,5 +1,5 @@
 /* { dg-do compile { target fpic } } */
-/* { dg-options "-mtune=generic -O2 -fPIC" } */
+/* { dg-options "-mtune=generic -mno-sse -O2 -fPIC" } */
 /* { dg-additional-options "-march=pentiumpro" { target ia32 } } */
 
 struct desc {
diff --git a/gcc/testsuite/gcc.target/i386/pr125100-3.c b/gcc/testsuite/gcc.target/i386/pr125100-3.c
index 2015ee819b3..0727086a578 100644
--- a/gcc/testsuite/gcc.target/i386/pr125100-3.c
+++ b/gcc/testsuite/gcc.target/i386/pr125100-3.c
@@ -1,5 +1,5 @@
 /* { dg-do compile { target fpic } } */
-/* { dg-options "-mtune=generic -O2 -fPIC" } */
+/* { dg-options "-mtune=generic -mno-sse -O2 -fPIC" } */
 /* { dg-additional-options "-march=pentiumpro" { target ia32 } } */
 
 struct desc {
diff --git a/gcc/testsuite/gcc.target/i386/pr126802-1a.c b/gcc/testsuite/gcc.target/i386/pr126802-1a.c
new file mode 100644
index 00000000000..37b9918f48f
--- /dev/null
+++ b/gcc/testsuite/gcc.target/i386/pr126802-1a.c
@@ -0,0 +1,26 @@
+/* { dg-do compile } */
+/* { dg-options "-O2 -march=x86-64" } */
+/* Keep labels and directives ('.cfi_startproc', '.cfi_endproc').  */
+/* { dg-final { check-function-bodies "**" "" "" { target { *-*-linux* && lp64 } } {^\t?\.} } } */
+
+/*
+**func:
+**.LFB0:
+**	.cfi_startproc
+**	movq	var1\(%rip\), %rdx
+**	movsbl	1\(%rdx\), %eax
+**	movsbl	\(%rdx\), %edx
+**	addl	%edx, %eax
+**	movl	%eax, var2\(%rip\)
+**	ret
+**...
+*/
+
+extern char *var1;
+extern int var2;
+
+void
+func (void)
+{
+  var2 = var1[1] + var1[0];
+}
diff --git a/gcc/testsuite/gcc.target/i386/pr126802-1b.c b/gcc/testsuite/gcc.target/i386/pr126802-1b.c
new file mode 100644
index 00000000000..f447c874902
--- /dev/null
+++ b/gcc/testsuite/gcc.target/i386/pr126802-1b.c
@@ -0,0 +1,19 @@
+/* { dg-do compile } */
+/* { dg-options "-O2 -march=x86-64-v4" } */
+/* Keep labels and directives ('.cfi_startproc', '.cfi_endproc').  */
+/* { dg-final { check-function-bodies "**" "" "" { target { *-*-linux* && lp64 } } {^\t?\.} } } */
+
+/*
+**func:
+**.LFB0:
+**	.cfi_startproc
+**	movq	var1\(%rip\), %rdx
+**	movsbl	1\(%rdx\), %eax
+**	movsbl	\(%rdx\), %edx
+**	addl	%edx, %eax
+**	movl	%eax, var2\(%rip\)
+**	ret
+**...
+*/
+
+#include "pr126802-1a.c"
diff --git a/gcc/testsuite/gcc.target/i386/pr126802-2a.c b/gcc/testsuite/gcc.target/i386/pr126802-2a.c
new file mode 100644
index 00000000000..cd833388530
--- /dev/null
+++ b/gcc/testsuite/gcc.target/i386/pr126802-2a.c
@@ -0,0 +1,26 @@
+/* { dg-do compile } */
+/* { dg-options "-O2 -march=x86-64" } */
+/* Keep labels and directives ('.cfi_startproc', '.cfi_endproc').  */
+/* { dg-final { check-function-bodies "**" "" "" { target { *-*-linux* && lp64 } } {^\t?\.} } } */
+
+/*
+**func:
+**.LFB0:
+**	.cfi_startproc
+**	movq	var1\(%rip\), %rdx
+**	movswl	2\(%rdx\), %eax
+**	movswl	\(%rdx\), %edx
+**	addl	%edx, %eax
+**	movl	%eax, var2\(%rip\)
+**	ret
+**...
+*/
+
+extern short *var1;
+extern int var2;
+
+void
+func (void)
+{
+  var2 = var1[1] + var1[0];
+}
diff --git a/gcc/testsuite/gcc.target/i386/pr126802-2b.c b/gcc/testsuite/gcc.target/i386/pr126802-2b.c
new file mode 100644
index 00000000000..b56b3daf33b
--- /dev/null
+++ b/gcc/testsuite/gcc.target/i386/pr126802-2b.c
@@ -0,0 +1,19 @@
+/* { dg-do compile } */
+/* { dg-options "-O2 -march=x86-64-v4" } */
+/* Keep labels and directives ('.cfi_startproc', '.cfi_endproc').  */
+/* { dg-final { check-function-bodies "**" "" "" { target { *-*-linux* && lp64 } } {^\t?\.} } } */
+
+/*
+**func:
+**.LFB0:
+**	.cfi_startproc
+**	movq	var1\(%rip\), %rdx
+**	movswl	2\(%rdx\), %eax
+**	movswl	\(%rdx\), %edx
+**	addl	%edx, %eax
+**	movl	%eax, var2\(%rip\)
+**	ret
+**...
+*/
+
+#include "pr126802-2a.c"
diff --git a/gcc/testsuite/gcc.target/i386/pr126802-3a.c b/gcc/testsuite/gcc.target/i386/pr126802-3a.c
new file mode 100644
index 00000000000..b2fd9b5e03e
--- /dev/null
+++ b/gcc/testsuite/gcc.target/i386/pr126802-3a.c
@@ -0,0 +1,25 @@
+/* { dg-do compile } */
+/* { dg-options "-O2 -march=x86-64" } */
+/* Keep labels and directives ('.cfi_startproc', '.cfi_endproc').  */
+/* { dg-final { check-function-bodies "**" "" "" { target { *-*-linux* && lp64 } } {^\t?\.} } } */
+
+/*
+**func:
+**.LFB0:
+**	.cfi_startproc
+**	movq	var1\(%rip\), %rdx
+**	movl	4\(%rdx\), %eax
+**	addl	\(%rdx\), %eax
+**	movl	%eax, var2\(%rip\)
+**	ret
+**...
+*/
+
+extern int *var1;
+extern int var2;
+
+void
+func (void)
+{
+  var2 = var1[1] + var1[0];
+}
diff --git a/gcc/testsuite/gcc.target/i386/pr126802-3b.c b/gcc/testsuite/gcc.target/i386/pr126802-3b.c
new file mode 100644
index 00000000000..4c9c9d6f137
--- /dev/null
+++ b/gcc/testsuite/gcc.target/i386/pr126802-3b.c
@@ -0,0 +1,18 @@
+/* { dg-do compile } */
+/* { dg-options "-O2 -march=x86-64-v4" } */
+/* Keep labels and directives ('.cfi_startproc', '.cfi_endproc').  */
+/* { dg-final { check-function-bodies "**" "" "" { target { *-*-linux* && lp64 } } {^\t?\.} } } */
+
+/*
+**func:
+**.LFB0:
+**	.cfi_startproc
+**	movq	var1\(%rip\), %rdx
+**	movl	4\(%rdx\), %eax
+**	addl	\(%rdx\), %eax
+**	movl	%eax, var2\(%rip\)
+**	ret
+**...
+*/
+
+#include "pr126802-3a.c"
diff --git a/gcc/testsuite/gcc.target/i386/pr126802-4a.c b/gcc/testsuite/gcc.target/i386/pr126802-4a.c
new file mode 100644
index 00000000000..6461fdfbc0a
--- /dev/null
+++ b/gcc/testsuite/gcc.target/i386/pr126802-4a.c
@@ -0,0 +1,36 @@
+/* { dg-do compile } */
+/* { dg-options "-O2 -march=x86-64" } */
+/* Keep labels and directives ('.cfi_startproc', '.cfi_endproc').  */
+/* { dg-final { check-function-bodies "**" "" "" { target { *-*-linux* && lp64 } } {^\t?\.} } } */
+
+/*
+**func:
+**.LFB0:
+**	.cfi_startproc
+**	movq	var1\(%rip\), %rax
+**	pxor	%xmm1, %xmm1
+**	movd	\(%rax\), %xmm0
+**	pcmpgtb	%xmm0, %xmm1
+**	punpcklbw	%xmm1, %xmm0
+**	movdqa	%xmm0, %xmm1
+**	psraw	\$15, %xmm1
+**	punpcklwd	%xmm1, %xmm0
+**	movdqa	%xmm0, %xmm1
+**	psrldq	\$8, %xmm1
+**	paddd	%xmm1, %xmm0
+**	movdqa	%xmm0, %xmm1
+**	psrldq	\$4, %xmm1
+**	paddd	%xmm1, %xmm0
+**	movd	%xmm0, var2\(%rip\)
+**	ret
+**...
+*/
+
+extern char *var1;
+extern int var2;
+
+void
+func (void)
+{
+  var2 = var1[1] + var1[0] + var1[2] + var1[3];
+}
diff --git a/gcc/testsuite/gcc.target/i386/pr126802-4b.c b/gcc/testsuite/gcc.target/i386/pr126802-4b.c
new file mode 100644
index 00000000000..04710ec9c6d
--- /dev/null
+++ b/gcc/testsuite/gcc.target/i386/pr126802-4b.c
@@ -0,0 +1,23 @@
+/* { dg-do compile } */
+/* { dg-options "-O2 -march=x86-64-v2" } */
+/* Keep labels and directives ('.cfi_startproc', '.cfi_endproc').  */
+/* { dg-final { check-function-bodies "**" "" "" { target { *-*-linux* && lp64 } } {^\t?\.} } } */
+
+/*
+**func:
+**.LFB0:
+**	.cfi_startproc
+**	movq	var1\(%rip\), %rax
+**	pmovsxbd	\(%rax\), %xmm0
+**	movdqa	%xmm0, %xmm1
+**	psrldq	\$8, %xmm1
+**	paddd	%xmm1, %xmm0
+**	movdqa	%xmm0, %xmm1
+**	psrldq	\$4, %xmm1
+**	paddd	%xmm1, %xmm0
+**	movd	%xmm0, var2\(%rip\)
+**	ret
+**...
+*/
+
+#include "pr126802-4a.c"
-- 
2.55.0

Reply via email to