The default cost of MULT/DIV/MOD in rtx_cost () is high: 5,7 insns
respectively. This causes even a trivial mpy by 7 to be synthesized.
Given these have direct equivalents in BPF ISA, fix the cost to generate
native BPF insns.
Note the existing divmod-licall-2.c test was a bit fragile as it forced
cast signed an actual unsigned int which is provably non-negative.
In the new cost model compiler would generate a native unsigned divide
even though it not available for -mcpu=v3, tripping up the test.
Fix by ensuring the arg is actually signed.
gcc/ChangeLog:
* config/bpf/bpf.cc (bpf_rtx_costs): Assign MPY/DIV/MOD cost 1.
gcc/testsuite/ChangeLog:
* gcc.target/bpf/divmod-libcall-2.c: Change args/ret to signed.
* gcc.target/bpf/mult-large.c: New test.
* gcc.target/bpf/mult-small.c: New test.
Signed-off-by: Vineet Gupta <[email protected]>
(cherry picked from commit c743ce6b75543fadea31ab8d518bba363bf5e44e)
---
gcc/config/bpf/bpf.cc | 24 ++++++++++++++++---
.../gcc.target/bpf/divmod-libcall-2.c | 15 +++++++-----
gcc/testsuite/gcc.target/bpf/mult-large.c | 17 +++++++++++++
gcc/testsuite/gcc.target/bpf/mult-small.c | 14 +++++++++++
4 files changed, 61 insertions(+), 9 deletions(-)
create mode 100644 gcc/testsuite/gcc.target/bpf/mult-large.c
create mode 100644 gcc/testsuite/gcc.target/bpf/mult-small.c
diff --git a/gcc/config/bpf/bpf.cc b/gcc/config/bpf/bpf.cc
index 89a4917cd0a..58d2318f972 100644
--- a/gcc/config/bpf/bpf.cc
+++ b/gcc/config/bpf/bpf.cc
@@ -602,14 +602,32 @@ bpf_legitimate_address_p (machine_mode mode,
`rtx_cost' should recurse. */
static bool
-bpf_rtx_costs (rtx x ATTRIBUTE_UNUSED,
+bpf_rtx_costs (rtx x,
enum machine_mode mode ATTRIBUTE_UNUSED,
int outer_code ATTRIBUTE_UNUSED,
int opno ATTRIBUTE_UNUSED,
- int *total ATTRIBUTE_UNUSED,
+ int *total,
bool speed ATTRIBUTE_UNUSED)
{
- /* To be written. */
+ switch (GET_CODE (x))
+ {
+ case MULT:
+ case DIV:
+ case UDIV:
+ case MOD:
+ case UMOD:
+ /* BPF implements these as a single instruction, so keep the native
+ operation cheaper than synthesized sequence.
+ Only influences choice between actually available alternatives;
+ if the operation has no insn (e.g. a 64-bit signed divide before
+ -mcpu=v4) expand_divmod () still falls back to a libcall.
+ Return false so caller rtx_cost keeps recursing for operands. */
+ *total = COSTS_N_INSNS (1);
+ return false;
+
+ default:
+ return false;
+ }
return false;
}
diff --git a/gcc/testsuite/gcc.target/bpf/divmod-libcall-2.c
b/gcc/testsuite/gcc.target/bpf/divmod-libcall-2.c
index 792d689395a..7296579171d 100644
--- a/gcc/testsuite/gcc.target/bpf/divmod-libcall-2.c
+++ b/gcc/testsuite/gcc.target/bpf/divmod-libcall-2.c
@@ -1,16 +1,19 @@
+/* Inverse of divmod-libcall-1.c. Ensure libcalls are generated for
+ -mcpu=v3 due to lack of signed div/mod. */
+
/* { dg-do compile } */
/* { dg-options "-O2 -mcpu=v3" } */
/* { dg-final { scan-assembler "global\t__divdi3" } } */
/* { dg-final { scan-assembler "global\t__moddi3" } } */
-int
-foo (unsigned int len)
+long
+foo (long len)
{
- return ((long)len) * 234 / 5;
+ return len * 234 / 5;
}
-int
-bar (unsigned int len)
+long
+bar (long len)
{
- return ((long)len) * 234 % 5;
+ return len * 234 % 5;
}
diff --git a/gcc/testsuite/gcc.target/bpf/mult-large.c
b/gcc/testsuite/gcc.target/bpf/mult-large.c
new file mode 100644
index 00000000000..cc0efb8bb52
--- /dev/null
+++ b/gcc/testsuite/gcc.target/bpf/mult-large.c
@@ -0,0 +1,17 @@
+/* Verify that a multiply by a complex constant uses BPF's native single
+ instruction mul rather than a synthesized shift/add(-sub) sequence.
+ This is extracted from memset-4.c but with 32-bit multiplicand vs. byte
+ as the constant multiplier could genuinely trigger a mpy synthesis as
+ 'x | x<<8 | x<<16 | x<<24'. */
+
+/* { dg-do compile } */
+/* { dg-options "-O2" } */
+
+unsigned
+bcast (unsigned x)
+{
+ return x * 0x01010101u;
+}
+
+/* { dg-final { scan-assembler {\*= 16843009} } } */
+/* { dg-final { scan-assembler-not {<<=} } } */
diff --git a/gcc/testsuite/gcc.target/bpf/mult-small.c
b/gcc/testsuite/gcc.target/bpf/mult-small.c
new file mode 100644
index 00000000000..3cebb1fdc1f
--- /dev/null
+++ b/gcc/testsuite/gcc.target/bpf/mult-small.c
@@ -0,0 +1,14 @@
+/* Verify that a multiply by a small constant uses BPF's native single
+ instruction mul rather than a synthesized shift/add(-sub) sequence. */
+
+/* { dg-do compile } */
+/* { dg-options "-O2 -mcpu=v4" } */
+
+unsigned
+mul7 (unsigned x)
+{
+ return x * 7;
+}
+
+/* { dg-final { scan-assembler {\*= 7} } } */
+/* { dg-final { scan-assembler-not {<<=} } } */
--
2.39.5