From: Kyrylo Tkachov <[email protected]>
vect_recog_sat_trunc_pattern only forms .SAT_TRUNC when the target has the
sstrunc or ustrunc optab for the mode pair, and the backend had neither, even
though aarch64_<su>qmovn<mode> already matches exactly the RTL those optabs
describe. A saturating narrowing loop was therefore vectorised as a pair of
clamps and a permute:
static inline unsigned char clip (unsigned short x)
{ return x & ~255 ? 255 : x; }
for (i) r[i] = clip (x[i]);
before after
movi v29.8h, 0xff ldp q30, q31, [x1], 32
ldp q31, q30, [x1], 32 uqxtn v30.8b, v30.8h
umin v31.8h, v31.8h, v29.8h uqxtn v31.8b, v31.8h
umin v30.8h, v30.8h, v29.8h stp d30, d31, [x0], 16
uzp1 v31.16b, v31.16b, v30.16b
str q31, [x0], 16
and the signed form was worse still, needing two constants and a pair of
compares per half. Only the standard names were missing, so this adds the
expander and leaves the existing insn to match it.
Bootstrapped and tested on aarch64-none-linux-gnu.
Ok for trunk?
Thanks,
Kyrill
gcc/ChangeLog:
* config/aarch64/iterators.md (vnarrowq): New mode attribute.
(sat_trunc_op): New code attribute.
* config/aarch64/aarch64-simd.md (<sat_trunc_op>trunc<mode><vnarrowq>2):
New expander.
gcc/testsuite/ChangeLog:
* gcc.target/aarch64/vect-sat-trunc-1.c: New test.
* gcc.target/aarch64/vect-sat-trunc-2.c: New test.
Signed-off-by: Kyrylo Tkachov <[email protected]>
---
gcc/config/aarch64/aarch64-simd.md | 8 ++
gcc/config/aarch64/iterators.md | 6 ++
.../gcc.target/aarch64/vect-sat-trunc-1.c | 73 +++++++++++++++++++
.../gcc.target/aarch64/vect-sat-trunc-2.c | 57 +++++++++++++++
4 files changed, 144 insertions(+)
create mode 100644 gcc/testsuite/gcc.target/aarch64/vect-sat-trunc-1.c
create mode 100644 gcc/testsuite/gcc.target/aarch64/vect-sat-trunc-2.c
diff --git a/gcc/config/aarch64/aarch64-simd.md
b/gcc/config/aarch64/aarch64-simd.md
index 12861333b22..527efe94084 100644
--- a/gcc/config/aarch64/aarch64-simd.md
+++ b/gcc/config/aarch64/aarch64-simd.md
@@ -6122,6 +6122,14 @@
[(set_attr "type" "neon_qadd<q>")]
)
+;; The saturating narrowing conversion that the vectoriser asks for through
+;; IFN_SAT_TRUNC. The RTL is what aarch64_<su>qmovn<mode> below matches.
+(define_expand "<sat_trunc_op>trunc<mode><vnarrowq>2"
+ [(set (match_operand:<VNARROWQ> 0 "register_operand")
+ (SAT_TRUNC:<VNARROWQ> (match_operand:VQN 1 "register_operand")))]
+ "TARGET_SIMD"
+)
+
;; sqmovn and uqmovn
(define_insn "aarch64_<su>qmovn<mode><vczle><vczbe>"
diff --git a/gcc/config/aarch64/iterators.md b/gcc/config/aarch64/iterators.md
index 12e65c0f230..7dbc19b7e05 100644
--- a/gcc/config/aarch64/iterators.md
+++ b/gcc/config/aarch64/iterators.md
@@ -2134,6 +2134,9 @@
(DI "v2si")])
;; Narrowed double-modes for VQN (Used for XTN).
+;; Lower case VNARROWQ, for the sstrunc and ustrunc optab names.
+(define_mode_attr vnarrowq [(V8HI "v8qi") (V4SI "v4hi") (V2DI "v2si")])
+
(define_mode_attr VNARROWQ [(V8HI "V8QI") (V4SI "V4HI")
(V2DI "V2SI")
(DI "SI") (SI "HI")
@@ -3407,6 +3410,9 @@
(define_code_attr TRUNC_SHIFT [(ss_truncate "ashiftrt")
(us_truncate "lshiftrt") (truncate "lshiftrt")])
+;; The optab prefix of a saturating truncation.
+(define_code_attr sat_trunc_op [(ss_truncate "ss") (us_truncate "us")])
+
(define_code_attr shrn_op [(ss_truncate "sq")
(us_truncate "uq") (truncate "")])
diff --git a/gcc/testsuite/gcc.target/aarch64/vect-sat-trunc-1.c
b/gcc/testsuite/gcc.target/aarch64/vect-sat-trunc-1.c
new file mode 100644
index 00000000000..b3128971356
--- /dev/null
+++ b/gcc/testsuite/gcc.target/aarch64/vect-sat-trunc-1.c
@@ -0,0 +1,73 @@
+/* { dg-do compile } */
+/* { dg-options "-O3 -march=armv8-a" } */
+/* { dg-additional-options "-fno-schedule-insns -fno-schedule-insns2" } */
+/* { dg-final { check-function-bodies "**" "" } } */
+
+typedef __UINT16_TYPE__ u16;
+typedef __UINT8_TYPE__ u8;
+typedef __INT16_TYPE__ i16;
+typedef __INT8_TYPE__ i8;
+
+static inline u8
+clip_u8 (u16 x)
+{
+ return x & (u16) ~(u16) 255 ? (u8) 255 : (u8) x;
+}
+
+/*
+** clu:
+** ...
+** ldp q[0-9]+, q[0-9]+, \[x[0-9]+\]
+** uqxtn v[0-9]+\.8b, v[0-9]+\.8h
+** uqxtn v[0-9]+\.8b, v[0-9]+\.8h
+** stp d[0-9]+, d[0-9]+, \[x[0-9]+\]
+** ret
+*/
+void
+clu (u8 *__restrict r, u16 *__restrict x)
+{
+ for (int i = 0; i < 16; i++)
+ r[i] = clip_u8 (x[i]);
+}
+
+static inline i8
+clip_i8 (i16 x)
+{
+ i8 t = (i8) x;
+ return (i16) -128 <= x && x <= (i16) 127 ? t : x < 0 ? -128 : 127;
+}
+
+/*
+** cls:
+** ...
+** ldp q[0-9]+, q[0-9]+, \[x[0-9]+\]
+** sqxtn v[0-9]+\.8b, v[0-9]+\.8h
+** sqxtn v[0-9]+\.8b, v[0-9]+\.8h
+** stp d[0-9]+, d[0-9]+, \[x[0-9]+\]
+** ret
+*/
+void
+cls (i8 *__restrict r, i16 *__restrict x)
+{
+ for (int i = 0; i < 16; i++)
+ r[i] = clip_i8 (x[i]);
+}
+
+/* The same in variable-length loops, and at the other two element widths. */
+void
+clu_n (u8 *__restrict r, u16 *__restrict x, int n)
+{
+ for (int i = 0; i < n; i++)
+ r[i] = clip_u8 (x[i]);
+}
+
+void
+cls_n (i8 *__restrict r, i16 *__restrict x, int n)
+{
+ for (int i = 0; i < n; i++)
+ r[i] = clip_i8 (x[i]);
+}
+
+/* { dg-final { scan-assembler-times {\tuqxtn\tv[0-9]+\.8b, v[0-9]+\.8h} 3 } }
*/
+/* { dg-final { scan-assembler-times {\tsqxtn\tv[0-9]+\.8b, v[0-9]+\.8h} 3 } }
*/
+/* { dg-final { scan-assembler-not {\tuzp1\t} } } */
diff --git a/gcc/testsuite/gcc.target/aarch64/vect-sat-trunc-2.c
b/gcc/testsuite/gcc.target/aarch64/vect-sat-trunc-2.c
new file mode 100644
index 00000000000..76d5d6e632f
--- /dev/null
+++ b/gcc/testsuite/gcc.target/aarch64/vect-sat-trunc-2.c
@@ -0,0 +1,57 @@
+/* { dg-do run } */
+/* { dg-options "-O3" } */
+
+typedef __UINT16_TYPE__ u16; typedef __UINT8_TYPE__ u8;
+typedef __UINT32_TYPE__ u32; typedef __UINT64_TYPE__ u64;
+typedef __INT16_TYPE__ i16; typedef __INT8_TYPE__ i8;
+typedef __INT32_TYPE__ i32; typedef __INT64_TYPE__ i64;
+#define N 137
+
+#define DEFU(name, WT, NT, NMAX) \
+ static inline NT clip_##name (WT x) \
+ { return x & (WT) ~(WT) NMAX ? (NT) NMAX : (NT) x; } \
+ __attribute__((noipa)) void name (NT *__restrict r, WT *__restrict x, int n)
\
+ { for (int i = 0; i < n; i++) r[i] = clip_##name (x[i]); } \
+ __attribute__((noipa, optimize ("O0"))) \
+ void name##_ref (NT *__restrict r, WT *__restrict x, int n) \
+ { for (int i = 0; i < n; i++) r[i] = clip_##name (x[i]); }
+
+#define DEFS(name, WT, NT, NMIN, NMAX) \
+ static inline NT clip_##name (WT x) \
+ { NT t = (NT) x; \
+ return (WT) NMIN <= x && x <= (WT) NMAX ? t : x < 0 ? NMIN : NMAX; } \
+ __attribute__((noipa)) void name (NT *__restrict r, WT *__restrict x, int n)
\
+ { for (int i = 0; i < n; i++) r[i] = clip_##name (x[i]); } \
+ __attribute__((noipa, optimize ("O0"))) \
+ void name##_ref (NT *__restrict r, WT *__restrict x, int n) \
+ { for (int i = 0; i < n; i++) r[i] = clip_##name (x[i]); }
+
+DEFU (u16to8, u16, u8, 255)
+DEFU (u32to16, u32, u16, 65535)
+DEFU (u64to32, u64, u32, 0xffffffffu)
+DEFS (i16to8, i16, i8, -128, 127)
+DEFS (i32to16, i32, i16, -32768, 32767)
+DEFS (i64to32, i64, i32, (i32) 0x80000000, 0x7fffffff)
+
+static u16 a16[N]; static u32 a32[N]; static u64 a64[N];
+static u8 d8[N], e8[N]; static u16 d16[N], e16[N]; static u32 d32[N], e32[N];
+static unsigned long seed = 7;
+static unsigned rnd (void) { seed = seed * 6364136223846793005UL + 1; return
(unsigned)(seed >> 33); }
+#define CHK(d, e, n) for (int i = 0; i < n; i++) if (d[i] != e[i])
__builtin_abort ();
+int main (void)
+{
+ for (int r = 0; r < 200; r++)
+ {
+ for (int i = 0; i < N; i++)
+ { unsigned v = rnd ();
+ a16[i] = (u16) v; a32[i] = v; a64[i] = ((u64) v << 32) | rnd ();
+ if ((i & 7) == 0) { a16[i] = 0xffff; a32[i] = 0xffffffffu; a64[i] =
~0UL; } }
+ u16to8 (d8, a16, N); u16to8_ref (e8, a16, N); CHK (d8, e8, N)
+ u32to16 (d16, a32, N); u32to16_ref (e16, a32, N); CHK (d16, e16, N)
+ u64to32 (d32, a64, N); u64to32_ref (e32, a64, N); CHK (d32, e32, N)
+ i16to8 ((i8 *) d8, (i16 *) a16, N); i16to8_ref ((i8 *) e8, (i16 *) a16,
N); CHK (d8, e8, N)
+ i32to16 ((i16 *) d16, (i32 *) a32, N); i32to16_ref ((i16 *) e16, (i32 *)
a32, N); CHK (d16, e16, N)
+ i64to32 ((i32 *) d32, (i64 *) a64, N); i64to32_ref ((i32 *) e32, (i64 *)
a64, N); CHK (d32, e32, N)
+ }
+ return 0;
+}
--
2.50.1 (Apple Git-155)