https://github.com/XChy created https://github.com/llvm/llvm-project/pull/228049
This patch supports packed slide 1 up/down intrinsics and codegen. I also noticed that the v2i32 packed pair intrinsics are missing while working on this patch. I will work on a follow-up patch to add them. >From bd025d8f80b15f50eeda1994f18125b333095407 Mon Sep 17 00:00:00 2001 From: XChy <[email protected]> Date: Mon, 28 Sep 2026 21:30:37 +0800 Subject: [PATCH] [RISCV][P-ext] Support Packed Slide 1 --- clang/lib/Headers/riscv_packed_simd.h | 31 ++ clang/test/CodeGen/RISCV/rvp-intrinsics.c | 444 ++++++++++++++++++ .../riscv_packed_simd.c | 134 ++++++ llvm/lib/Target/RISCV/RISCVISelLowering.cpp | 96 ++++ llvm/lib/Target/RISCV/RISCVInstrInfoP.td | 37 ++ llvm/test/CodeGen/RISCV/rvp-simd-32.ll | 68 +++ llvm/test/CodeGen/RISCV/rvp-simd-64.ll | 108 +++++ 7 files changed, 918 insertions(+) diff --git a/clang/lib/Headers/riscv_packed_simd.h b/clang/lib/Headers/riscv_packed_simd.h index 6bc610d74246b..58a7fda4e1919 100644 --- a/clang/lib/Headers/riscv_packed_simd.h +++ b/clang/lib/Headers/riscv_packed_simd.h @@ -274,6 +274,12 @@ typedef uint32_t uint32x2_t __attribute__((__vector_size__(8))); return __builtin_shufflevector(__lo, __hi, 0, 1, 2, 3, 4, 5, 6, 7); \ } +#define __packed_slide1(name, ty, elt_ty, ...) \ + static __inline__ ty __DEFAULT_FN_ATTRS __riscv_##name(ty __rd, \ + elt_ty __rs1) { \ + return __builtin_shufflevector(__rd, (ty){__rs1}, __VA_ARGS__); \ + } + #define __packed_pair_ee2(name, ty) \ static __inline__ ty __DEFAULT_FN_ATTRS __riscv_##name(ty __rs1, ty __rs2) { \ return __builtin_shufflevector(__rs1, __rs2, 0, 2); \ @@ -1304,6 +1310,30 @@ __packed_concat4(pjoin2_u8x8, uint8x8_t, uint8x4_t) __packed_concat2(pjoin2_i16x4, int16x4_t, int16x2_t) __packed_concat2(pjoin2_u16x4, uint16x4_t, uint16x2_t) +/* Packed Slide 1 up/down (32-bit) */ +__packed_slide1(pslide1up_i8x4, int8x4_t, int8_t, 4, 0, 1, 2) +__packed_slide1(pslide1up_u8x4, uint8x4_t, uint8_t, 4, 0, 1, 2) +__packed_slide1(pslide1up_i16x2, int16x2_t, int16_t, 2, 0) +__packed_slide1(pslide1up_u16x2, uint16x2_t, uint16_t, 2, 0) +__packed_slide1(pslide1down_i8x4, int8x4_t, int8_t, 1, 2, 3, 4) +__packed_slide1(pslide1down_u8x4, uint8x4_t, uint8_t, 1, 2, 3, 4) +__packed_slide1(pslide1down_i16x2, int16x2_t, int16_t, 1, 2) +__packed_slide1(pslide1down_u16x2, uint16x2_t, uint16_t, 1, 2) + +/* Packed Slide 1 up/down (64-bit) */ +__packed_slide1(pslide1up_i8x8, int8x8_t, int8_t, 8, 0, 1, 2, 3, 4, 5, 6) +__packed_slide1(pslide1up_u8x8, uint8x8_t, uint8_t, 8, 0, 1, 2, 3, 4, 5, 6) +__packed_slide1(pslide1up_i16x4, int16x4_t, int16_t, 4, 0, 1, 2) +__packed_slide1(pslide1up_u16x4, uint16x4_t, uint16_t, 4, 0, 1, 2) +__packed_slide1(pslide1up_i32x2, int32x2_t, int32_t, 2, 0) +__packed_slide1(pslide1up_u32x2, uint32x2_t, uint32_t, 2, 0) +__packed_slide1(pslide1down_i8x8, int8x8_t, int8_t, 1, 2, 3, 4, 5, 6, 7, 8) +__packed_slide1(pslide1down_u8x8, uint8x8_t, uint8_t, 1, 2, 3, 4, 5, 6, 7, 8) +__packed_slide1(pslide1down_i16x4, int16x4_t, int16_t, 1, 2, 3, 4) +__packed_slide1(pslide1down_u16x4, uint16x4_t, uint16_t, 1, 2, 3, 4) +__packed_slide1(pslide1down_i32x2, int32x2_t, int32_t, 1, 2) +__packed_slide1(pslide1down_u32x2, uint32x2_t, uint32_t, 1, 2) + /* Packed Subvector Extract */ __packed_subvector_extract8(pget_i8x8_i8x4, int8x4_t, int8x8_t) __packed_subvector_extract8(pget_u8x8_u8x4, uint8x4_t, uint8x8_t) @@ -1525,6 +1555,7 @@ __packed_reinterpret(u32x2_i32x2, int32x2_t, uint32x2_t) #undef __packed_unzipo4 #undef __packed_concat2 #undef __packed_concat4 +#undef __packed_slide1 #undef __packed_pair_ee2 #undef __packed_pair_eo2 #undef __packed_pair_oe2 diff --git a/clang/test/CodeGen/RISCV/rvp-intrinsics.c b/clang/test/CodeGen/RISCV/rvp-intrinsics.c index b26b75b3e1431..03e54f4e30839 100644 --- a/clang/test/CodeGen/RISCV/rvp-intrinsics.c +++ b/clang/test/CodeGen/RISCV/rvp-intrinsics.c @@ -9068,3 +9068,447 @@ int32x2_t test_pwunzipho_i32x2(int16x4_t a) { return __riscv_pwunzipho_i32x2(a); uint32x2_t test_pwunzipho_u32x2(uint16x4_t a) { return __riscv_pwunzipho_u32x2(a); } + +/* Packed Slide 1 up/down (32-bit) */ + +// RV32-LABEL: define dso_local i32 @test_pslide1up_i8x4( +// RV32-SAME: i32 noundef [[RD_COERCE:%.*]], i8 noundef signext [[RS1:%.*]]) #[[ATTR0]] { +// RV32-NEXT: [[ENTRY:.*:]] +// RV32-NEXT: [[TMP0:%.*]] = bitcast i32 [[RD_COERCE]] to <4 x i8> +// RV32-NEXT: [[VECINIT4_I:%.*]] = insertelement <4 x i8> poison, i8 [[RS1]], i64 0 +// RV32-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <4 x i8> [[TMP0]], <4 x i8> [[VECINIT4_I]], <4 x i32> <i32 4, i32 0, i32 1, i32 2> +// RV32-NEXT: [[TMP1:%.*]] = bitcast <4 x i8> [[SHUFFLE_I]] to i32 +// RV32-NEXT: ret i32 [[TMP1]] +// +// RV64-LABEL: define dso_local i32 @test_pslide1up_i8x4( +// RV64-SAME: i32 noundef [[RD_COERCE:%.*]], i8 noundef signext [[RS1:%.*]]) #[[ATTR0]] { +// RV64-NEXT: [[ENTRY:.*:]] +// RV64-NEXT: [[TMP0:%.*]] = bitcast i32 [[RD_COERCE]] to <4 x i8> +// RV64-NEXT: [[VECINIT4_I:%.*]] = insertelement <4 x i8> poison, i8 [[RS1]], i64 0 +// RV64-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <4 x i8> [[TMP0]], <4 x i8> [[VECINIT4_I]], <4 x i32> <i32 4, i32 0, i32 1, i32 2> +// RV64-NEXT: [[TMP1:%.*]] = bitcast <4 x i8> [[SHUFFLE_I]] to i32 +// RV64-NEXT: ret i32 [[TMP1]] +// +int8x4_t test_pslide1up_i8x4(int8x4_t rd, int8_t rs1) { + return __riscv_pslide1up_i8x4(rd, rs1); +} + +// RV32-LABEL: define dso_local i32 @test_pslide1up_u8x4( +// RV32-SAME: i32 noundef [[RD_COERCE:%.*]], i8 noundef zeroext [[RS1:%.*]]) #[[ATTR0]] { +// RV32-NEXT: [[ENTRY:.*:]] +// RV32-NEXT: [[TMP0:%.*]] = bitcast i32 [[RD_COERCE]] to <4 x i8> +// RV32-NEXT: [[VECINIT4_I:%.*]] = insertelement <4 x i8> poison, i8 [[RS1]], i64 0 +// RV32-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <4 x i8> [[TMP0]], <4 x i8> [[VECINIT4_I]], <4 x i32> <i32 4, i32 0, i32 1, i32 2> +// RV32-NEXT: [[TMP1:%.*]] = bitcast <4 x i8> [[SHUFFLE_I]] to i32 +// RV32-NEXT: ret i32 [[TMP1]] +// +// RV64-LABEL: define dso_local i32 @test_pslide1up_u8x4( +// RV64-SAME: i32 noundef [[RD_COERCE:%.*]], i8 noundef zeroext [[RS1:%.*]]) #[[ATTR0]] { +// RV64-NEXT: [[ENTRY:.*:]] +// RV64-NEXT: [[TMP0:%.*]] = bitcast i32 [[RD_COERCE]] to <4 x i8> +// RV64-NEXT: [[VECINIT4_I:%.*]] = insertelement <4 x i8> poison, i8 [[RS1]], i64 0 +// RV64-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <4 x i8> [[TMP0]], <4 x i8> [[VECINIT4_I]], <4 x i32> <i32 4, i32 0, i32 1, i32 2> +// RV64-NEXT: [[TMP1:%.*]] = bitcast <4 x i8> [[SHUFFLE_I]] to i32 +// RV64-NEXT: ret i32 [[TMP1]] +// +uint8x4_t test_pslide1up_u8x4(uint8x4_t rd, uint8_t rs1) { + return __riscv_pslide1up_u8x4(rd, rs1); +} + +// RV32-LABEL: define dso_local i32 @test_pslide1up_i16x2( +// RV32-SAME: i32 noundef [[RD_COERCE:%.*]], i16 noundef signext [[RS1:%.*]]) #[[ATTR0]] { +// RV32-NEXT: [[ENTRY:.*:]] +// RV32-NEXT: [[TMP0:%.*]] = bitcast i32 [[RD_COERCE]] to <2 x i16> +// RV32-NEXT: [[VECINIT2_I:%.*]] = insertelement <2 x i16> poison, i16 [[RS1]], i64 0 +// RV32-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <2 x i16> [[TMP0]], <2 x i16> [[VECINIT2_I]], <2 x i32> <i32 2, i32 0> +// RV32-NEXT: [[TMP1:%.*]] = bitcast <2 x i16> [[SHUFFLE_I]] to i32 +// RV32-NEXT: ret i32 [[TMP1]] +// +// RV64-LABEL: define dso_local i32 @test_pslide1up_i16x2( +// RV64-SAME: i32 noundef [[RD_COERCE:%.*]], i16 noundef signext [[RS1:%.*]]) #[[ATTR0]] { +// RV64-NEXT: [[ENTRY:.*:]] +// RV64-NEXT: [[TMP0:%.*]] = bitcast i32 [[RD_COERCE]] to <2 x i16> +// RV64-NEXT: [[VECINIT2_I:%.*]] = insertelement <2 x i16> poison, i16 [[RS1]], i64 0 +// RV64-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <2 x i16> [[TMP0]], <2 x i16> [[VECINIT2_I]], <2 x i32> <i32 2, i32 0> +// RV64-NEXT: [[TMP1:%.*]] = bitcast <2 x i16> [[SHUFFLE_I]] to i32 +// RV64-NEXT: ret i32 [[TMP1]] +// +int16x2_t test_pslide1up_i16x2(int16x2_t rd, int16_t rs1) { + return __riscv_pslide1up_i16x2(rd, rs1); +} + +// RV32-LABEL: define dso_local i32 @test_pslide1up_u16x2( +// RV32-SAME: i32 noundef [[RD_COERCE:%.*]], i16 noundef zeroext [[RS1:%.*]]) #[[ATTR0]] { +// RV32-NEXT: [[ENTRY:.*:]] +// RV32-NEXT: [[TMP0:%.*]] = bitcast i32 [[RD_COERCE]] to <2 x i16> +// RV32-NEXT: [[VECINIT2_I:%.*]] = insertelement <2 x i16> poison, i16 [[RS1]], i64 0 +// RV32-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <2 x i16> [[TMP0]], <2 x i16> [[VECINIT2_I]], <2 x i32> <i32 2, i32 0> +// RV32-NEXT: [[TMP1:%.*]] = bitcast <2 x i16> [[SHUFFLE_I]] to i32 +// RV32-NEXT: ret i32 [[TMP1]] +// +// RV64-LABEL: define dso_local i32 @test_pslide1up_u16x2( +// RV64-SAME: i32 noundef [[RD_COERCE:%.*]], i16 noundef zeroext [[RS1:%.*]]) #[[ATTR0]] { +// RV64-NEXT: [[ENTRY:.*:]] +// RV64-NEXT: [[TMP0:%.*]] = bitcast i32 [[RD_COERCE]] to <2 x i16> +// RV64-NEXT: [[VECINIT2_I:%.*]] = insertelement <2 x i16> poison, i16 [[RS1]], i64 0 +// RV64-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <2 x i16> [[TMP0]], <2 x i16> [[VECINIT2_I]], <2 x i32> <i32 2, i32 0> +// RV64-NEXT: [[TMP1:%.*]] = bitcast <2 x i16> [[SHUFFLE_I]] to i32 +// RV64-NEXT: ret i32 [[TMP1]] +// +uint16x2_t test_pslide1up_u16x2(uint16x2_t rd, uint16_t rs1) { + return __riscv_pslide1up_u16x2(rd, rs1); +} + +// RV32-LABEL: define dso_local i32 @test_pslide1down_i8x4( +// RV32-SAME: i32 noundef [[RD_COERCE:%.*]], i8 noundef signext [[RS1:%.*]]) #[[ATTR0]] { +// RV32-NEXT: [[ENTRY:.*:]] +// RV32-NEXT: [[TMP0:%.*]] = bitcast i32 [[RD_COERCE]] to <4 x i8> +// RV32-NEXT: [[VECINIT4_I:%.*]] = insertelement <4 x i8> poison, i8 [[RS1]], i64 0 +// RV32-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <4 x i8> [[TMP0]], <4 x i8> [[VECINIT4_I]], <4 x i32> <i32 1, i32 2, i32 3, i32 4> +// RV32-NEXT: [[TMP1:%.*]] = bitcast <4 x i8> [[SHUFFLE_I]] to i32 +// RV32-NEXT: ret i32 [[TMP1]] +// +// RV64-LABEL: define dso_local i32 @test_pslide1down_i8x4( +// RV64-SAME: i32 noundef [[RD_COERCE:%.*]], i8 noundef signext [[RS1:%.*]]) #[[ATTR0]] { +// RV64-NEXT: [[ENTRY:.*:]] +// RV64-NEXT: [[TMP0:%.*]] = bitcast i32 [[RD_COERCE]] to <4 x i8> +// RV64-NEXT: [[VECINIT4_I:%.*]] = insertelement <4 x i8> poison, i8 [[RS1]], i64 0 +// RV64-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <4 x i8> [[TMP0]], <4 x i8> [[VECINIT4_I]], <4 x i32> <i32 1, i32 2, i32 3, i32 4> +// RV64-NEXT: [[TMP1:%.*]] = bitcast <4 x i8> [[SHUFFLE_I]] to i32 +// RV64-NEXT: ret i32 [[TMP1]] +// +int8x4_t test_pslide1down_i8x4(int8x4_t rd, int8_t rs1) { + return __riscv_pslide1down_i8x4(rd, rs1); +} + +// RV32-LABEL: define dso_local i32 @test_pslide1down_u8x4( +// RV32-SAME: i32 noundef [[RD_COERCE:%.*]], i8 noundef zeroext [[RS1:%.*]]) #[[ATTR0]] { +// RV32-NEXT: [[ENTRY:.*:]] +// RV32-NEXT: [[TMP0:%.*]] = bitcast i32 [[RD_COERCE]] to <4 x i8> +// RV32-NEXT: [[VECINIT4_I:%.*]] = insertelement <4 x i8> poison, i8 [[RS1]], i64 0 +// RV32-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <4 x i8> [[TMP0]], <4 x i8> [[VECINIT4_I]], <4 x i32> <i32 1, i32 2, i32 3, i32 4> +// RV32-NEXT: [[TMP1:%.*]] = bitcast <4 x i8> [[SHUFFLE_I]] to i32 +// RV32-NEXT: ret i32 [[TMP1]] +// +// RV64-LABEL: define dso_local i32 @test_pslide1down_u8x4( +// RV64-SAME: i32 noundef [[RD_COERCE:%.*]], i8 noundef zeroext [[RS1:%.*]]) #[[ATTR0]] { +// RV64-NEXT: [[ENTRY:.*:]] +// RV64-NEXT: [[TMP0:%.*]] = bitcast i32 [[RD_COERCE]] to <4 x i8> +// RV64-NEXT: [[VECINIT4_I:%.*]] = insertelement <4 x i8> poison, i8 [[RS1]], i64 0 +// RV64-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <4 x i8> [[TMP0]], <4 x i8> [[VECINIT4_I]], <4 x i32> <i32 1, i32 2, i32 3, i32 4> +// RV64-NEXT: [[TMP1:%.*]] = bitcast <4 x i8> [[SHUFFLE_I]] to i32 +// RV64-NEXT: ret i32 [[TMP1]] +// +uint8x4_t test_pslide1down_u8x4(uint8x4_t rd, uint8_t rs1) { + return __riscv_pslide1down_u8x4(rd, rs1); +} + +// RV32-LABEL: define dso_local i32 @test_pslide1down_i16x2( +// RV32-SAME: i32 noundef [[RD_COERCE:%.*]], i16 noundef signext [[RS1:%.*]]) #[[ATTR0]] { +// RV32-NEXT: [[ENTRY:.*:]] +// RV32-NEXT: [[TMP0:%.*]] = bitcast i32 [[RD_COERCE]] to <2 x i16> +// RV32-NEXT: [[VECINIT2_I:%.*]] = insertelement <2 x i16> poison, i16 [[RS1]], i64 0 +// RV32-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <2 x i16> [[TMP0]], <2 x i16> [[VECINIT2_I]], <2 x i32> <i32 1, i32 2> +// RV32-NEXT: [[TMP1:%.*]] = bitcast <2 x i16> [[SHUFFLE_I]] to i32 +// RV32-NEXT: ret i32 [[TMP1]] +// +// RV64-LABEL: define dso_local i32 @test_pslide1down_i16x2( +// RV64-SAME: i32 noundef [[RD_COERCE:%.*]], i16 noundef signext [[RS1:%.*]]) #[[ATTR0]] { +// RV64-NEXT: [[ENTRY:.*:]] +// RV64-NEXT: [[TMP0:%.*]] = bitcast i32 [[RD_COERCE]] to <2 x i16> +// RV64-NEXT: [[VECINIT2_I:%.*]] = insertelement <2 x i16> poison, i16 [[RS1]], i64 0 +// RV64-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <2 x i16> [[TMP0]], <2 x i16> [[VECINIT2_I]], <2 x i32> <i32 1, i32 2> +// RV64-NEXT: [[TMP1:%.*]] = bitcast <2 x i16> [[SHUFFLE_I]] to i32 +// RV64-NEXT: ret i32 [[TMP1]] +// +int16x2_t test_pslide1down_i16x2(int16x2_t rd, int16_t rs1) { + return __riscv_pslide1down_i16x2(rd, rs1); +} + +// RV32-LABEL: define dso_local i32 @test_pslide1down_u16x2( +// RV32-SAME: i32 noundef [[RD_COERCE:%.*]], i16 noundef zeroext [[RS1:%.*]]) #[[ATTR0]] { +// RV32-NEXT: [[ENTRY:.*:]] +// RV32-NEXT: [[TMP0:%.*]] = bitcast i32 [[RD_COERCE]] to <2 x i16> +// RV32-NEXT: [[VECINIT2_I:%.*]] = insertelement <2 x i16> poison, i16 [[RS1]], i64 0 +// RV32-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <2 x i16> [[TMP0]], <2 x i16> [[VECINIT2_I]], <2 x i32> <i32 1, i32 2> +// RV32-NEXT: [[TMP1:%.*]] = bitcast <2 x i16> [[SHUFFLE_I]] to i32 +// RV32-NEXT: ret i32 [[TMP1]] +// +// RV64-LABEL: define dso_local i32 @test_pslide1down_u16x2( +// RV64-SAME: i32 noundef [[RD_COERCE:%.*]], i16 noundef zeroext [[RS1:%.*]]) #[[ATTR0]] { +// RV64-NEXT: [[ENTRY:.*:]] +// RV64-NEXT: [[TMP0:%.*]] = bitcast i32 [[RD_COERCE]] to <2 x i16> +// RV64-NEXT: [[VECINIT2_I:%.*]] = insertelement <2 x i16> poison, i16 [[RS1]], i64 0 +// RV64-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <2 x i16> [[TMP0]], <2 x i16> [[VECINIT2_I]], <2 x i32> <i32 1, i32 2> +// RV64-NEXT: [[TMP1:%.*]] = bitcast <2 x i16> [[SHUFFLE_I]] to i32 +// RV64-NEXT: ret i32 [[TMP1]] +// +uint16x2_t test_pslide1down_u16x2(uint16x2_t rd, uint16_t rs1) { + return __riscv_pslide1down_u16x2(rd, rs1); +} + +/* Packed Slide 1 up/down (64-bit) */ + +// RV32-LABEL: define dso_local i64 @test_pslide1up_i8x8( +// RV32-SAME: i64 noundef [[RD_COERCE:%.*]], i8 noundef signext [[RS1:%.*]]) #[[ATTR0]] { +// RV32-NEXT: [[ENTRY:.*:]] +// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <8 x i8> +// RV32-NEXT: [[VECINIT8_I:%.*]] = insertelement <8 x i8> poison, i8 [[RS1]], i64 0 +// RV32-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <8 x i8> [[TMP0]], <8 x i8> [[VECINIT8_I]], <8 x i32> <i32 8, i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6> +// RV32-NEXT: [[TMP1:%.*]] = bitcast <8 x i8> [[SHUFFLE_I]] to i64 +// RV32-NEXT: ret i64 [[TMP1]] +// +// RV64-LABEL: define dso_local i64 @test_pslide1up_i8x8( +// RV64-SAME: i64 noundef [[RD_COERCE:%.*]], i8 noundef signext [[RS1:%.*]]) #[[ATTR0]] { +// RV64-NEXT: [[ENTRY:.*:]] +// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <8 x i8> +// RV64-NEXT: [[VECINIT8_I:%.*]] = insertelement <8 x i8> poison, i8 [[RS1]], i64 0 +// RV64-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <8 x i8> [[TMP0]], <8 x i8> [[VECINIT8_I]], <8 x i32> <i32 8, i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6> +// RV64-NEXT: [[TMP1:%.*]] = bitcast <8 x i8> [[SHUFFLE_I]] to i64 +// RV64-NEXT: ret i64 [[TMP1]] +// +int8x8_t test_pslide1up_i8x8(int8x8_t rd, int8_t rs1) { + return __riscv_pslide1up_i8x8(rd, rs1); +} + +// RV32-LABEL: define dso_local i64 @test_pslide1up_u8x8( +// RV32-SAME: i64 noundef [[RD_COERCE:%.*]], i8 noundef zeroext [[RS1:%.*]]) #[[ATTR0]] { +// RV32-NEXT: [[ENTRY:.*:]] +// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <8 x i8> +// RV32-NEXT: [[VECINIT8_I:%.*]] = insertelement <8 x i8> poison, i8 [[RS1]], i64 0 +// RV32-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <8 x i8> [[TMP0]], <8 x i8> [[VECINIT8_I]], <8 x i32> <i32 8, i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6> +// RV32-NEXT: [[TMP1:%.*]] = bitcast <8 x i8> [[SHUFFLE_I]] to i64 +// RV32-NEXT: ret i64 [[TMP1]] +// +// RV64-LABEL: define dso_local i64 @test_pslide1up_u8x8( +// RV64-SAME: i64 noundef [[RD_COERCE:%.*]], i8 noundef zeroext [[RS1:%.*]]) #[[ATTR0]] { +// RV64-NEXT: [[ENTRY:.*:]] +// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <8 x i8> +// RV64-NEXT: [[VECINIT8_I:%.*]] = insertelement <8 x i8> poison, i8 [[RS1]], i64 0 +// RV64-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <8 x i8> [[TMP0]], <8 x i8> [[VECINIT8_I]], <8 x i32> <i32 8, i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6> +// RV64-NEXT: [[TMP1:%.*]] = bitcast <8 x i8> [[SHUFFLE_I]] to i64 +// RV64-NEXT: ret i64 [[TMP1]] +// +uint8x8_t test_pslide1up_u8x8(uint8x8_t rd, uint8_t rs1) { + return __riscv_pslide1up_u8x8(rd, rs1); +} + +// RV32-LABEL: define dso_local i64 @test_pslide1up_i16x4( +// RV32-SAME: i64 noundef [[RD_COERCE:%.*]], i16 noundef signext [[RS1:%.*]]) #[[ATTR0]] { +// RV32-NEXT: [[ENTRY:.*:]] +// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <4 x i16> +// RV32-NEXT: [[VECINIT4_I:%.*]] = insertelement <4 x i16> poison, i16 [[RS1]], i64 0 +// RV32-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <4 x i16> [[TMP0]], <4 x i16> [[VECINIT4_I]], <4 x i32> <i32 4, i32 0, i32 1, i32 2> +// RV32-NEXT: [[TMP1:%.*]] = bitcast <4 x i16> [[SHUFFLE_I]] to i64 +// RV32-NEXT: ret i64 [[TMP1]] +// +// RV64-LABEL: define dso_local i64 @test_pslide1up_i16x4( +// RV64-SAME: i64 noundef [[RD_COERCE:%.*]], i16 noundef signext [[RS1:%.*]]) #[[ATTR0]] { +// RV64-NEXT: [[ENTRY:.*:]] +// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <4 x i16> +// RV64-NEXT: [[VECINIT4_I:%.*]] = insertelement <4 x i16> poison, i16 [[RS1]], i64 0 +// RV64-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <4 x i16> [[TMP0]], <4 x i16> [[VECINIT4_I]], <4 x i32> <i32 4, i32 0, i32 1, i32 2> +// RV64-NEXT: [[TMP1:%.*]] = bitcast <4 x i16> [[SHUFFLE_I]] to i64 +// RV64-NEXT: ret i64 [[TMP1]] +// +int16x4_t test_pslide1up_i16x4(int16x4_t rd, int16_t rs1) { + return __riscv_pslide1up_i16x4(rd, rs1); +} + +// RV32-LABEL: define dso_local i64 @test_pslide1up_u16x4( +// RV32-SAME: i64 noundef [[RD_COERCE:%.*]], i16 noundef zeroext [[RS1:%.*]]) #[[ATTR0]] { +// RV32-NEXT: [[ENTRY:.*:]] +// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <4 x i16> +// RV32-NEXT: [[VECINIT4_I:%.*]] = insertelement <4 x i16> poison, i16 [[RS1]], i64 0 +// RV32-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <4 x i16> [[TMP0]], <4 x i16> [[VECINIT4_I]], <4 x i32> <i32 4, i32 0, i32 1, i32 2> +// RV32-NEXT: [[TMP1:%.*]] = bitcast <4 x i16> [[SHUFFLE_I]] to i64 +// RV32-NEXT: ret i64 [[TMP1]] +// +// RV64-LABEL: define dso_local i64 @test_pslide1up_u16x4( +// RV64-SAME: i64 noundef [[RD_COERCE:%.*]], i16 noundef zeroext [[RS1:%.*]]) #[[ATTR0]] { +// RV64-NEXT: [[ENTRY:.*:]] +// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <4 x i16> +// RV64-NEXT: [[VECINIT4_I:%.*]] = insertelement <4 x i16> poison, i16 [[RS1]], i64 0 +// RV64-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <4 x i16> [[TMP0]], <4 x i16> [[VECINIT4_I]], <4 x i32> <i32 4, i32 0, i32 1, i32 2> +// RV64-NEXT: [[TMP1:%.*]] = bitcast <4 x i16> [[SHUFFLE_I]] to i64 +// RV64-NEXT: ret i64 [[TMP1]] +// +uint16x4_t test_pslide1up_u16x4(uint16x4_t rd, uint16_t rs1) { + return __riscv_pslide1up_u16x4(rd, rs1); +} + +// RV32-LABEL: define dso_local i64 @test_pslide1up_i32x2( +// RV32-SAME: i64 noundef [[RD_COERCE:%.*]], i32 noundef [[RS1:%.*]]) #[[ATTR0]] { +// RV32-NEXT: [[ENTRY:.*:]] +// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <2 x i32> +// RV32-NEXT: [[VECINIT2_I:%.*]] = insertelement <2 x i32> poison, i32 [[RS1]], i64 0 +// RV32-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <2 x i32> [[TMP0]], <2 x i32> [[VECINIT2_I]], <2 x i32> <i32 2, i32 0> +// RV32-NEXT: [[TMP1:%.*]] = bitcast <2 x i32> [[SHUFFLE_I]] to i64 +// RV32-NEXT: ret i64 [[TMP1]] +// +// RV64-LABEL: define dso_local i64 @test_pslide1up_i32x2( +// RV64-SAME: i64 noundef [[RD_COERCE:%.*]], i32 noundef signext [[RS1:%.*]]) #[[ATTR0]] { +// RV64-NEXT: [[ENTRY:.*:]] +// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <2 x i32> +// RV64-NEXT: [[VECINIT2_I:%.*]] = insertelement <2 x i32> poison, i32 [[RS1]], i64 0 +// RV64-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <2 x i32> [[TMP0]], <2 x i32> [[VECINIT2_I]], <2 x i32> <i32 2, i32 0> +// RV64-NEXT: [[TMP1:%.*]] = bitcast <2 x i32> [[SHUFFLE_I]] to i64 +// RV64-NEXT: ret i64 [[TMP1]] +// +int32x2_t test_pslide1up_i32x2(int32x2_t rd, int32_t rs1) { + return __riscv_pslide1up_i32x2(rd, rs1); +} + +// RV32-LABEL: define dso_local i64 @test_pslide1up_u32x2( +// RV32-SAME: i64 noundef [[RD_COERCE:%.*]], i32 noundef [[RS1:%.*]]) #[[ATTR0]] { +// RV32-NEXT: [[ENTRY:.*:]] +// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <2 x i32> +// RV32-NEXT: [[VECINIT2_I:%.*]] = insertelement <2 x i32> poison, i32 [[RS1]], i64 0 +// RV32-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <2 x i32> [[TMP0]], <2 x i32> [[VECINIT2_I]], <2 x i32> <i32 2, i32 0> +// RV32-NEXT: [[TMP1:%.*]] = bitcast <2 x i32> [[SHUFFLE_I]] to i64 +// RV32-NEXT: ret i64 [[TMP1]] +// +// RV64-LABEL: define dso_local i64 @test_pslide1up_u32x2( +// RV64-SAME: i64 noundef [[RD_COERCE:%.*]], i32 noundef signext [[RS1:%.*]]) #[[ATTR0]] { +// RV64-NEXT: [[ENTRY:.*:]] +// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <2 x i32> +// RV64-NEXT: [[VECINIT2_I:%.*]] = insertelement <2 x i32> poison, i32 [[RS1]], i64 0 +// RV64-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <2 x i32> [[TMP0]], <2 x i32> [[VECINIT2_I]], <2 x i32> <i32 2, i32 0> +// RV64-NEXT: [[TMP1:%.*]] = bitcast <2 x i32> [[SHUFFLE_I]] to i64 +// RV64-NEXT: ret i64 [[TMP1]] +// +uint32x2_t test_pslide1up_u32x2(uint32x2_t rd, uint32_t rs1) { + return __riscv_pslide1up_u32x2(rd, rs1); +} + +// RV32-LABEL: define dso_local i64 @test_pslide1down_i8x8( +// RV32-SAME: i64 noundef [[RD_COERCE:%.*]], i8 noundef signext [[RS1:%.*]]) #[[ATTR0]] { +// RV32-NEXT: [[ENTRY:.*:]] +// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <8 x i8> +// RV32-NEXT: [[VECINIT8_I:%.*]] = insertelement <8 x i8> poison, i8 [[RS1]], i64 0 +// RV32-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <8 x i8> [[TMP0]], <8 x i8> [[VECINIT8_I]], <8 x i32> <i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8> +// RV32-NEXT: [[TMP1:%.*]] = bitcast <8 x i8> [[SHUFFLE_I]] to i64 +// RV32-NEXT: ret i64 [[TMP1]] +// +// RV64-LABEL: define dso_local i64 @test_pslide1down_i8x8( +// RV64-SAME: i64 noundef [[RD_COERCE:%.*]], i8 noundef signext [[RS1:%.*]]) #[[ATTR0]] { +// RV64-NEXT: [[ENTRY:.*:]] +// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <8 x i8> +// RV64-NEXT: [[VECINIT8_I:%.*]] = insertelement <8 x i8> poison, i8 [[RS1]], i64 0 +// RV64-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <8 x i8> [[TMP0]], <8 x i8> [[VECINIT8_I]], <8 x i32> <i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8> +// RV64-NEXT: [[TMP1:%.*]] = bitcast <8 x i8> [[SHUFFLE_I]] to i64 +// RV64-NEXT: ret i64 [[TMP1]] +// +int8x8_t test_pslide1down_i8x8(int8x8_t rd, int8_t rs1) { + return __riscv_pslide1down_i8x8(rd, rs1); +} + +// RV32-LABEL: define dso_local i64 @test_pslide1down_u8x8( +// RV32-SAME: i64 noundef [[RD_COERCE:%.*]], i8 noundef zeroext [[RS1:%.*]]) #[[ATTR0]] { +// RV32-NEXT: [[ENTRY:.*:]] +// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <8 x i8> +// RV32-NEXT: [[VECINIT8_I:%.*]] = insertelement <8 x i8> poison, i8 [[RS1]], i64 0 +// RV32-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <8 x i8> [[TMP0]], <8 x i8> [[VECINIT8_I]], <8 x i32> <i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8> +// RV32-NEXT: [[TMP1:%.*]] = bitcast <8 x i8> [[SHUFFLE_I]] to i64 +// RV32-NEXT: ret i64 [[TMP1]] +// +// RV64-LABEL: define dso_local i64 @test_pslide1down_u8x8( +// RV64-SAME: i64 noundef [[RD_COERCE:%.*]], i8 noundef zeroext [[RS1:%.*]]) #[[ATTR0]] { +// RV64-NEXT: [[ENTRY:.*:]] +// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <8 x i8> +// RV64-NEXT: [[VECINIT8_I:%.*]] = insertelement <8 x i8> poison, i8 [[RS1]], i64 0 +// RV64-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <8 x i8> [[TMP0]], <8 x i8> [[VECINIT8_I]], <8 x i32> <i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8> +// RV64-NEXT: [[TMP1:%.*]] = bitcast <8 x i8> [[SHUFFLE_I]] to i64 +// RV64-NEXT: ret i64 [[TMP1]] +// +uint8x8_t test_pslide1down_u8x8(uint8x8_t rd, uint8_t rs1) { + return __riscv_pslide1down_u8x8(rd, rs1); +} + +// RV32-LABEL: define dso_local i64 @test_pslide1down_i16x4( +// RV32-SAME: i64 noundef [[RD_COERCE:%.*]], i16 noundef signext [[RS1:%.*]]) #[[ATTR0]] { +// RV32-NEXT: [[ENTRY:.*:]] +// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <4 x i16> +// RV32-NEXT: [[VECINIT4_I:%.*]] = insertelement <4 x i16> poison, i16 [[RS1]], i64 0 +// RV32-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <4 x i16> [[TMP0]], <4 x i16> [[VECINIT4_I]], <4 x i32> <i32 1, i32 2, i32 3, i32 4> +// RV32-NEXT: [[TMP1:%.*]] = bitcast <4 x i16> [[SHUFFLE_I]] to i64 +// RV32-NEXT: ret i64 [[TMP1]] +// +// RV64-LABEL: define dso_local i64 @test_pslide1down_i16x4( +// RV64-SAME: i64 noundef [[RD_COERCE:%.*]], i16 noundef signext [[RS1:%.*]]) #[[ATTR0]] { +// RV64-NEXT: [[ENTRY:.*:]] +// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <4 x i16> +// RV64-NEXT: [[VECINIT4_I:%.*]] = insertelement <4 x i16> poison, i16 [[RS1]], i64 0 +// RV64-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <4 x i16> [[TMP0]], <4 x i16> [[VECINIT4_I]], <4 x i32> <i32 1, i32 2, i32 3, i32 4> +// RV64-NEXT: [[TMP1:%.*]] = bitcast <4 x i16> [[SHUFFLE_I]] to i64 +// RV64-NEXT: ret i64 [[TMP1]] +// +int16x4_t test_pslide1down_i16x4(int16x4_t rd, int16_t rs1) { + return __riscv_pslide1down_i16x4(rd, rs1); +} + +// RV32-LABEL: define dso_local i64 @test_pslide1down_u16x4( +// RV32-SAME: i64 noundef [[RD_COERCE:%.*]], i16 noundef zeroext [[RS1:%.*]]) #[[ATTR0]] { +// RV32-NEXT: [[ENTRY:.*:]] +// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <4 x i16> +// RV32-NEXT: [[VECINIT4_I:%.*]] = insertelement <4 x i16> poison, i16 [[RS1]], i64 0 +// RV32-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <4 x i16> [[TMP0]], <4 x i16> [[VECINIT4_I]], <4 x i32> <i32 1, i32 2, i32 3, i32 4> +// RV32-NEXT: [[TMP1:%.*]] = bitcast <4 x i16> [[SHUFFLE_I]] to i64 +// RV32-NEXT: ret i64 [[TMP1]] +// +// RV64-LABEL: define dso_local i64 @test_pslide1down_u16x4( +// RV64-SAME: i64 noundef [[RD_COERCE:%.*]], i16 noundef zeroext [[RS1:%.*]]) #[[ATTR0]] { +// RV64-NEXT: [[ENTRY:.*:]] +// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <4 x i16> +// RV64-NEXT: [[VECINIT4_I:%.*]] = insertelement <4 x i16> poison, i16 [[RS1]], i64 0 +// RV64-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <4 x i16> [[TMP0]], <4 x i16> [[VECINIT4_I]], <4 x i32> <i32 1, i32 2, i32 3, i32 4> +// RV64-NEXT: [[TMP1:%.*]] = bitcast <4 x i16> [[SHUFFLE_I]] to i64 +// RV64-NEXT: ret i64 [[TMP1]] +// +uint16x4_t test_pslide1down_u16x4(uint16x4_t rd, uint16_t rs1) { + return __riscv_pslide1down_u16x4(rd, rs1); +} + +// RV32-LABEL: define dso_local i64 @test_pslide1down_i32x2( +// RV32-SAME: i64 noundef [[RD_COERCE:%.*]], i32 noundef [[RS1:%.*]]) #[[ATTR0]] { +// RV32-NEXT: [[ENTRY:.*:]] +// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <2 x i32> +// RV32-NEXT: [[VECINIT2_I:%.*]] = insertelement <2 x i32> poison, i32 [[RS1]], i64 0 +// RV32-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <2 x i32> [[TMP0]], <2 x i32> [[VECINIT2_I]], <2 x i32> <i32 1, i32 2> +// RV32-NEXT: [[TMP1:%.*]] = bitcast <2 x i32> [[SHUFFLE_I]] to i64 +// RV32-NEXT: ret i64 [[TMP1]] +// +// RV64-LABEL: define dso_local i64 @test_pslide1down_i32x2( +// RV64-SAME: i64 noundef [[RD_COERCE:%.*]], i32 noundef signext [[RS1:%.*]]) #[[ATTR0]] { +// RV64-NEXT: [[ENTRY:.*:]] +// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <2 x i32> +// RV64-NEXT: [[VECINIT2_I:%.*]] = insertelement <2 x i32> poison, i32 [[RS1]], i64 0 +// RV64-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <2 x i32> [[TMP0]], <2 x i32> [[VECINIT2_I]], <2 x i32> <i32 1, i32 2> +// RV64-NEXT: [[TMP1:%.*]] = bitcast <2 x i32> [[SHUFFLE_I]] to i64 +// RV64-NEXT: ret i64 [[TMP1]] +// +int32x2_t test_pslide1down_i32x2(int32x2_t rd, int32_t rs1) { + return __riscv_pslide1down_i32x2(rd, rs1); +} + +// RV32-LABEL: define dso_local i64 @test_pslide1down_u32x2( +// RV32-SAME: i64 noundef [[RD_COERCE:%.*]], i32 noundef [[RS1:%.*]]) #[[ATTR0]] { +// RV32-NEXT: [[ENTRY:.*:]] +// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <2 x i32> +// RV32-NEXT: [[VECINIT2_I:%.*]] = insertelement <2 x i32> poison, i32 [[RS1]], i64 0 +// RV32-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <2 x i32> [[TMP0]], <2 x i32> [[VECINIT2_I]], <2 x i32> <i32 1, i32 2> +// RV32-NEXT: [[TMP1:%.*]] = bitcast <2 x i32> [[SHUFFLE_I]] to i64 +// RV32-NEXT: ret i64 [[TMP1]] +// +// RV64-LABEL: define dso_local i64 @test_pslide1down_u32x2( +// RV64-SAME: i64 noundef [[RD_COERCE:%.*]], i32 noundef signext [[RS1:%.*]]) #[[ATTR0]] { +// RV64-NEXT: [[ENTRY:.*:]] +// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <2 x i32> +// RV64-NEXT: [[VECINIT2_I:%.*]] = insertelement <2 x i32> poison, i32 [[RS1]], i64 0 +// RV64-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <2 x i32> [[TMP0]], <2 x i32> [[VECINIT2_I]], <2 x i32> <i32 1, i32 2> +// RV64-NEXT: [[TMP1:%.*]] = bitcast <2 x i32> [[SHUFFLE_I]] to i64 +// RV64-NEXT: ret i64 [[TMP1]] +// +uint32x2_t test_pslide1down_u32x2(uint32x2_t rd, uint32_t rs1) { + return __riscv_pslide1down_u32x2(rd, rs1); +} diff --git a/cross-project-tests/intrinsic-header-tests/riscv_packed_simd.c b/cross-project-tests/intrinsic-header-tests/riscv_packed_simd.c index 469c3d6846ed8..fb3f6f19af55a 100644 --- a/cross-project-tests/intrinsic-header-tests/riscv_packed_simd.c +++ b/cross-project-tests/intrinsic-header-tests/riscv_packed_simd.c @@ -5531,3 +5531,137 @@ int32x2_t test_pwunzipho_i32x2(int16x4_t a) { uint32x2_t test_pwunzipho_u32x2(uint16x4_t a) { return __riscv_pwunzipho_u32x2(a); } + +/* Packed Slide 1 up/down (32-bit) */ + +// CHECK-LABEL: test_pslide1up_i8x4: +// CHECK: slx +int8x4_t test_pslide1up_i8x4(int8x4_t rd, int8_t rs1) { + return __riscv_pslide1up_i8x4(rd, rs1); +} + +// CHECK-LABEL: test_pslide1up_u8x4: +// CHECK: slx +uint8x4_t test_pslide1up_u8x4(uint8x4_t rd, uint8_t rs1) { + return __riscv_pslide1up_u8x4(rd, rs1); +} + +// CHECK-LABEL: test_pslide1up_i16x2: +// RV32: pack +// RV64: ppaire.h +int16x2_t test_pslide1up_i16x2(int16x2_t rd, int16_t rs1) { + return __riscv_pslide1up_i16x2(rd, rs1); +} + +// CHECK-LABEL: test_pslide1up_u16x2: +// RV32: pack +// RV64: ppaire.h +uint16x2_t test_pslide1up_u16x2(uint16x2_t rd, uint16_t rs1) { + return __riscv_pslide1up_u16x2(rd, rs1); +} + +// CHECK-LABEL: test_pslide1down_i8x4: +// RV32: srx +// RV64: pack +// RV64: srli +int8x4_t test_pslide1down_i8x4(int8x4_t rd, int8_t rs1) { + return __riscv_pslide1down_i8x4(rd, rs1); +} + +// CHECK-LABEL: test_pslide1down_u8x4: +// RV32: srx +// RV64: pack +// RV64: srli +uint8x4_t test_pslide1down_u8x4(uint8x4_t rd, uint8_t rs1) { + return __riscv_pslide1down_u8x4(rd, rs1); +} + +// CHECK-LABEL: test_pslide1down_i16x2: +// CHECK: ppairoe.h +int16x2_t test_pslide1down_i16x2(int16x2_t rd, int16_t rs1) { + return __riscv_pslide1down_i16x2(rd, rs1); +} + +// CHECK-LABEL: test_pslide1down_u16x2: +// CHECK: ppairoe.h +uint16x2_t test_pslide1down_u16x2(uint16x2_t rd, uint16_t rs1) { + return __riscv_pslide1down_u16x2(rd, rs1); +} + +/* Packed Slide 1 up/down (64-bit) */ + +// CHECK-LABEL: test_pslide1up_i8x8: +// CHECK: slx +int8x8_t test_pslide1up_i8x8(int8x8_t rd, int8_t rs1) { + return __riscv_pslide1up_i8x8(rd, rs1); +} + +// CHECK-LABEL: test_pslide1up_u8x8: +// CHECK: slx +uint8x8_t test_pslide1up_u8x8(uint8x8_t rd, uint8_t rs1) { + return __riscv_pslide1up_u8x8(rd, rs1); +} + +// CHECK-LABEL: test_pslide1up_i16x4: +// CHECK: slx +int16x4_t test_pslide1up_i16x4(int16x4_t rd, int16_t rs1) { + return __riscv_pslide1up_i16x4(rd, rs1); +} + +// CHECK-LABEL: test_pslide1up_u16x4: +// CHECK: slx +uint16x4_t test_pslide1up_u16x4(uint16x4_t rd, uint16_t rs1) { + return __riscv_pslide1up_u16x4(rd, rs1); +} + +// CHECK-LABEL: test_pslide1up_i32x2: +// RV32: mv +// RV64: pack +int32x2_t test_pslide1up_i32x2(int32x2_t rd, int32_t rs1) { + return __riscv_pslide1up_i32x2(rd, rs1); +} + +// CHECK-LABEL: test_pslide1up_u32x2: +// RV32: mv +// RV64: pack +uint32x2_t test_pslide1up_u32x2(uint32x2_t rd, uint32_t rs1) { + return __riscv_pslide1up_u32x2(rd, rs1); +} + +// CHECK-LABEL: test_pslide1down_i8x8: +// CHECK: srx +int8x8_t test_pslide1down_i8x8(int8x8_t rd, int8_t rs1) { + return __riscv_pslide1down_i8x8(rd, rs1); +} + +// CHECK-LABEL: test_pslide1down_u8x8: +// CHECK: srx +uint8x8_t test_pslide1down_u8x8(uint8x8_t rd, uint8_t rs1) { + return __riscv_pslide1down_u8x8(rd, rs1); +} + +// CHECK-LABEL: test_pslide1down_i16x4: +// CHECK: srx +int16x4_t test_pslide1down_i16x4(int16x4_t rd, int16_t rs1) { + return __riscv_pslide1down_i16x4(rd, rs1); +} + +// CHECK-LABEL: test_pslide1down_u16x4: +// CHECK: srx +uint16x4_t test_pslide1down_u16x4(uint16x4_t rd, uint16_t rs1) { + return __riscv_pslide1down_u16x4(rd, rs1); +} + +// CHECK-LABEL: test_pslide1down_i32x2: +// RV32: mv +// RV64: ppairoe.w +int32x2_t test_pslide1down_i32x2(int32x2_t rd, int32_t rs1) { + return __riscv_pslide1down_i32x2(rd, rs1); +} + +// CHECK-LABEL: test_pslide1down_u32x2: +// RV32: mv +// RV64: ppairoe.w +uint32x2_t test_pslide1down_u32x2(uint32x2_t rd, uint32_t rs1) { + return __riscv_pslide1down_u32x2(rd, rs1); +} diff --git a/llvm/lib/Target/RISCV/RISCVISelLowering.cpp b/llvm/lib/Target/RISCV/RISCVISelLowering.cpp index 7730a52a2e26d..acf4ebf5508f6 100644 --- a/llvm/lib/Target/RISCV/RISCVISelLowering.cpp +++ b/llvm/lib/Target/RISCV/RISCVISelLowering.cpp @@ -6497,6 +6497,100 @@ static SDValue lowerVECTOR_SHUFFLEAsPUnzip(ShuffleVectorSDNode *SVN, return DAG.getNode(Opc, DL, VT, V1, V2); } +// Match a slide by one element with a scalar inserted at either end: +// <b0, a0, ..., aN-2> -> slide1up +// <a1, ..., aN-1, b0> -> slide1down +static SDValue lowerVECTOR_SHUFFLEAsPSlide1(ShuffleVectorSDNode *SVN, + const RISCVSubtarget &Subtarget, + SelectionDAG &DAG) { + MVT VT = SVN->getSimpleValueType(0); + if (VT != MVT::v4i8 && VT != MVT::v8i8 && VT != MVT::v2i16 && + VT != MVT::v4i16 && VT != MVT::v2i32) + return SDValue(); + if (!Subtarget.is64Bit() && VT == MVT::v2i32) + return SDValue(); + + unsigned NumElts = VT.getVectorNumElements(); + ArrayRef<int> Mask = SVN->getMask(); + unsigned ActiveElts = NumElts; + if (Subtarget.is64Bit() && + all_of(Mask.drop_front(NumElts / 2), [](int M) { return M < 0; })) + ActiveElts /= 2; + + ArrayRef<int> ActiveMask = Mask.take_front(ActiveElts); + bool SlideUp = ActiveMask[0] == (int)NumElts && + all_of(enumerate(ActiveMask.drop_front()), [](const auto &M) { + return M.value() == (int)M.index(); + }); + bool SlideDown = ActiveMask.back() == (int)NumElts && + all_of(enumerate(ActiveMask.drop_back()), [](const auto &M) { + return M.value() == (int)M.index() + 1; + }); + if (!SlideUp && !SlideDown) + return SDValue(); + + SDValue V1 = SVN->getOperand(0); + SDValue V2 = SVN->getOperand(1); + SDLoc DL(SVN); + unsigned EltBits = VT.getScalarSizeInBits(); + unsigned SlideBits = ActiveElts * EltBits; + MVT XLenVT = Subtarget.getXLenVT(); + SDValue Scalar = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, XLenVT, V2, + DAG.getVectorIdxConstant(0, DL)); + + if (SlideBits == Subtarget.getXLen()) + return DAG.getNode(SlideUp ? RISCVISD::PSLIDE1UP + : RISCVISD::PSLIDE1DOWN, + DL, VT, V1, Scalar); + + // A 32-bit packed slide on RV64 is carried in the low word of a GPR. Expand + // it using XLEN operations or packed pair instructions. + if (Subtarget.is64Bit()) { + assert(SlideBits == 32 && "Unexpected RV64 packed slide width"); + if (ActiveElts == 2) + return DAG.getNode(SlideUp ? RISCVISD::PPAIRE : RISCVISD::PPAIROE, DL, + VT, SlideUp ? V2 : V1, SlideUp ? V1 : V2); + + SDValue Shamt = DAG.getConstant(EltBits, DL, MVT::i64); + SDValue Bits = DAG.getBitcast(MVT::i64, V1); + if (SlideUp) { + Scalar = DAG.getNode(ISD::SHL, DL, MVT::i64, Scalar, + DAG.getConstant(64 - EltBits, DL, MVT::i64)); + Bits = DAG.getNode(ISD::FSHL, DL, MVT::i64, Bits, Scalar, Shamt); + } else { + SDValue Pair = DAG.getNode(RISCVISD::PPAIRE, DL, MVT::v2i32, + DAG.getBitcast(MVT::v2i32, V1), + DAG.getBitcast(MVT::v2i32, Scalar)); + Bits = DAG.getNode(ISD::SRL, DL, MVT::i64, + DAG.getBitcast(MVT::i64, Pair), Shamt); + } + return DAG.getBitcast(VT, Bits); + } + + // RV32 represents 64-bit packed values as two GPRs. Form each shifted half + // directly so the slide uses two XLEN funnel shifts. + assert(SlideBits == 64 && "Unexpected RV32 packed slide width"); + auto [LoVec, HiVec] = DAG.SplitVector(V1, DL); + MVT HalfVT = LoVec.getSimpleValueType(); + SDValue Lo = DAG.getBitcast(MVT::i32, LoVec); + SDValue Hi = DAG.getBitcast(MVT::i32, HiVec); + SDValue Shamt = DAG.getConstant(EltBits, DL, MVT::i32); + SDValue NewLo; + SDValue NewHi; + if (SlideUp) { + SDValue Insert = DAG.getNode(ISD::SHL, DL, MVT::i32, Scalar, + DAG.getConstant(32 - EltBits, DL, MVT::i32)); + NewLo = DAG.getNode(ISD::FSHL, DL, MVT::i32, Lo, Insert, Shamt); + NewHi = DAG.getNode(ISD::FSHL, DL, MVT::i32, Hi, Lo, Shamt); + } else { + NewLo = DAG.getNode(ISD::FSHR, DL, MVT::i32, Hi, Lo, Shamt); + NewHi = DAG.getNode(ISD::FSHR, DL, MVT::i32, Scalar, Hi, Shamt); + } + return DAG.getNode(ISD::CONCAT_VECTORS, DL, VT, + DAG.getBitcast(HalfVT, NewLo), + DAG.getBitcast(HalfVT, NewHi)); +} + // Match the packed zero-extend shuffle mask <0, N, 2, N+2, ...>: even result // lanes keep operand 0's even lanes and odd result lanes come from operand 1. // The odd lanes may select any element of operand 1, which is looser than a @@ -6670,6 +6764,8 @@ SDValue RISCVTargetLowering::lowerVECTOR_SHUFFLE(SDValue Op, return DAG.getBitcast(VT, Srl); } + if (SDValue V = lowerVECTOR_SHUFFLEAsPSlide1(SVN, Subtarget, DAG)) + return V; if (SDValue V = lowerVECTOR_SHUFFLEAsPUnzip(SVN, DAG, Subtarget.is64Bit())) return V; if (SDValue V = lowerVECTOR_SHUFFLEAsPZip(SVN, Subtarget, DAG)) diff --git a/llvm/lib/Target/RISCV/RISCVInstrInfoP.td b/llvm/lib/Target/RISCV/RISCVInstrInfoP.td index 8a6754b900028..d831cd67f13db 100644 --- a/llvm/lib/Target/RISCV/RISCVInstrInfoP.td +++ b/llvm/lib/Target/RISCV/RISCVInstrInfoP.td @@ -2007,6 +2007,14 @@ def riscv_ppaire : RVSDNode<"PPAIRE", SDT_RISCVPPair>; def riscv_ppaireo : RVSDNode<"PPAIREO", SDT_RISCVPPair>; def riscv_ppairoe : RVSDNode<"PPAIROE", SDT_RISCVPPair>; def riscv_ppairo : RVSDNode<"PPAIRO", SDT_RISCVPPair>; + +// Slide packed elements by one and insert an XLEN scalar at either end. +def SDT_RISCVPackedSlide1 : SDTypeProfile<1, 2, [SDTCisVec<0>, + SDTCisSameAs<0, 1>, + SDTCisVT<2, XLenVT>]>; +def riscv_pslide1up : RVSDNode<"PSLIDE1UP", SDT_RISCVPackedSlide1>; +def riscv_pslide1down : RVSDNode<"PSLIDE1DOWN", SDT_RISCVPackedSlide1>; + def SDT_RISCVPackedBinary : SDTypeProfile<1, 2, [SDTCisVec<0>, SDTCisSameAs<0, 1>, SDTCisSameAs<0, 2>]>; @@ -2769,6 +2777,17 @@ let append Predicates = [IsRV32] in { def : Pat<(v4i16 (riscv_pwzip (v2i16 GPR:$rs1), (v2i16 GPR:$rs2))), (WZIP16P GPR:$rs1, GPR:$rs2)>; + // Packed slide by one element. + def : Pat<(v4i8 (riscv_pslide1up (v4i8 GPR:$rd), GPR:$rs1)), + (SLX GPR:$rd, (SLLI GPR:$rs1, (XLenVT 24)), + (XLenVT (ADDI (XLenVT X0), 8)))>; + def : Pat<(v4i8 (riscv_pslide1down (v4i8 GPR:$rd), GPR:$rs1)), + (SRX GPR:$rd, GPR:$rs1, (XLenVT (ADDI (XLenVT X0), 8)))>; + def : Pat<(v2i16 (riscv_pslide1up (v2i16 GPR:$rd), GPR:$rs1)), + (PACK GPR:$rs1, GPR:$rd)>; + def : Pat<(v2i16 (riscv_pslide1down (v2i16 GPR:$rd), GPR:$rs1)), + (PPAIROE_H GPR:$rd, GPR:$rs1)>; + // Packed pair: pair the even/odd-position elements of rs1 and rs2. def : Pat<(v4i8 (riscv_ppaire (v4i8 GPR:$rs1), (v4i8 GPR:$rs2))), (PPAIRE_B GPR:$rs1, GPR:$rs2)>; @@ -3628,7 +3647,25 @@ let append Predicates = [IsRV64] in { def : Pat<(v4i16 (riscv_pzip (v4i16 GPR:$rs1), (v4i16 GPR:$rs2))), (ZIP16P GPR:$rs1, GPR:$rs2)>; + // Packed slide by one element. + def : Pat<(v8i8 (riscv_pslide1up (v8i8 GPR:$rd), GPR:$rs1)), + (SLX GPR:$rd, (SLLI GPR:$rs1, (XLenVT 56)), + (XLenVT (ADDI (XLenVT X0), 8)))>; + def : Pat<(v8i8 (riscv_pslide1down (v8i8 GPR:$rd), GPR:$rs1)), + (SRX GPR:$rd, GPR:$rs1, (XLenVT (ADDI (XLenVT X0), 8)))>; + def : Pat<(v4i16 (riscv_pslide1up (v4i16 GPR:$rd), GPR:$rs1)), + (SLX GPR:$rd, (SLLI GPR:$rs1, (XLenVT 48)), + (XLenVT (ADDI (XLenVT X0), 16)))>; + def : Pat<(v4i16 (riscv_pslide1down (v4i16 GPR:$rd), GPR:$rs1)), + (SRX GPR:$rd, GPR:$rs1, (XLenVT (ADDI (XLenVT X0), 16)))>; + def : Pat<(v2i32 (riscv_pslide1up (v2i32 GPR:$rd), GPR:$rs1)), + (PACK GPR:$rs1, GPR:$rd)>; + def : Pat<(v2i32 (riscv_pslide1down (v2i32 GPR:$rd), GPR:$rs1)), + (PPAIROE_W GPR:$rd, GPR:$rs1)>; + // Packed pair: pair the even/odd-position elements of rs1 and rs2. + def : Pat<(v2i32 (riscv_ppaire (v2i32 GPR:$rs1), (v2i32 GPR:$rs2))), + (PACK GPR:$rs1, GPR:$rs2)>; def : Pat<(v8i8 (riscv_ppaire (v8i8 GPR:$rs1), (v8i8 GPR:$rs2))), (PPAIRE_B GPR:$rs1, GPR:$rs2)>; def : Pat<(v8i8 (riscv_ppaireo (v8i8 GPR:$rs1), (v8i8 GPR:$rs2))), diff --git a/llvm/test/CodeGen/RISCV/rvp-simd-32.ll b/llvm/test/CodeGen/RISCV/rvp-simd-32.ll index e947cebb90a91..a862f31f9f9d4 100644 --- a/llvm/test/CodeGen/RISCV/rvp-simd-32.ll +++ b/llvm/test/CodeGen/RISCV/rvp-simd-32.ll @@ -4184,3 +4184,71 @@ define <2 x i16> @test_pusati_u16x2_max_width(<2 x i16> %a) { %res = call <2 x i16> @llvm.riscv.pusati.v2i16.i32(<2 x i16> %a, i32 15) ret <2 x i16> %res } + +define <4 x i8> @test_pslide1up_v4i8(<4 x i8> %rd, i8 %rs1) { +; RV32-LABEL: test_pslide1up_v4i8: +; RV32: # %bb.0: +; RV32-NEXT: slli a1, a1, 24 +; RV32-NEXT: li a2, 8 +; RV32-NEXT: slx a0, a1, a2 +; RV32-NEXT: ret +; +; RV64-LABEL: test_pslide1up_v4i8: +; RV64: # %bb.0: +; RV64-NEXT: slli a1, a1, 56 +; RV64-NEXT: li a2, 8 +; RV64-NEXT: slx a0, a1, a2 +; RV64-NEXT: ret + %scalar = insertelement <4 x i8> poison, i8 %rs1, i64 0 + %res = shufflevector <4 x i8> %rd, <4 x i8> %scalar, <4 x i32> <i32 4, i32 0, i32 1, i32 2> + ret <4 x i8> %res +} + +define <4 x i8> @test_pslide1down_v4i8(<4 x i8> %rd, i8 %rs1) { +; RV32-LABEL: test_pslide1down_v4i8: +; RV32: # %bb.0: +; RV32-NEXT: li a2, 8 +; RV32-NEXT: srx a0, a1, a2 +; RV32-NEXT: ret +; +; RV64-LABEL: test_pslide1down_v4i8: +; RV64: # %bb.0: +; RV64-NEXT: pack a0, a0, a1 +; RV64-NEXT: srli a0, a0, 8 +; RV64-NEXT: ret + %scalar = insertelement <4 x i8> poison, i8 %rs1, i64 0 + %res = shufflevector <4 x i8> %rd, <4 x i8> %scalar, <4 x i32> <i32 1, i32 2, i32 3, i32 4> + ret <4 x i8> %res +} + +define <2 x i16> @test_pslide1up_v2i16(<2 x i16> %rd, i16 %rs1) { +; RV32-LABEL: test_pslide1up_v2i16: +; RV32: # %bb.0: +; RV32-NEXT: pack a0, a1, a0 +; RV32-NEXT: ret +; +; RV64-LABEL: test_pslide1up_v2i16: +; RV64: # %bb.0: +; RV64-NEXT: pmv.hs a1, a1 +; RV64-NEXT: ppaire.h a0, a1, a0 +; RV64-NEXT: ret + %scalar = insertelement <2 x i16> poison, i16 %rs1, i64 0 + %res = shufflevector <2 x i16> %rd, <2 x i16> %scalar, <2 x i32> <i32 2, i32 0> + ret <2 x i16> %res +} + +define <2 x i16> @test_pslide1down_v2i16(<2 x i16> %rd, i16 %rs1) { +; RV32-LABEL: test_pslide1down_v2i16: +; RV32: # %bb.0: +; RV32-NEXT: ppairoe.h a0, a0, a1 +; RV32-NEXT: ret +; +; RV64-LABEL: test_pslide1down_v2i16: +; RV64: # %bb.0: +; RV64-NEXT: pmv.hs a1, a1 +; RV64-NEXT: ppairoe.h a0, a0, a1 +; RV64-NEXT: ret + %scalar = insertelement <2 x i16> poison, i16 %rs1, i64 0 + %res = shufflevector <2 x i16> %rd, <2 x i16> %scalar, <2 x i32> <i32 1, i32 2> + ret <2 x i16> %res +} diff --git a/llvm/test/CodeGen/RISCV/rvp-simd-64.ll b/llvm/test/CodeGen/RISCV/rvp-simd-64.ll index 77376296bfbdf..99ef53d3df7e6 100644 --- a/llvm/test/CodeGen/RISCV/rvp-simd-64.ll +++ b/llvm/test/CodeGen/RISCV/rvp-simd-64.ll @@ -8618,3 +8618,111 @@ define <2 x i32> @test_pusati_u32x2_max_width(<2 x i32> %a) { %res = call <2 x i32> @llvm.riscv.pusati.v2i32.i32(<2 x i32> %a, i32 31) ret <2 x i32> %res } + +define <8 x i8> @test_pslide1up_v8i8(<8 x i8> %rd, i8 %rs1) { +; RV32-LABEL: test_pslide1up_v8i8: +; RV32: # %bb.0: +; RV32-NEXT: li a3, 8 +; RV32-NEXT: slli a2, a2, 24 +; RV32-NEXT: slx a1, a0, a3 +; RV32-NEXT: slx a0, a2, a3 +; RV32-NEXT: ret +; +; RV64-LABEL: test_pslide1up_v8i8: +; RV64: # %bb.0: +; RV64-NEXT: slli a1, a1, 56 +; RV64-NEXT: li a2, 8 +; RV64-NEXT: slx a0, a1, a2 +; RV64-NEXT: ret + %scalar = insertelement <8 x i8> poison, i8 %rs1, i64 0 + %res = shufflevector <8 x i8> %rd, <8 x i8> %scalar, <8 x i32> <i32 8, i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6> + ret <8 x i8> %res +} + +define <8 x i8> @test_pslide1down_v8i8(<8 x i8> %rd, i8 %rs1) { +; RV32-LABEL: test_pslide1down_v8i8: +; RV32: # %bb.0: +; RV32-NEXT: li a3, 8 +; RV32-NEXT: srx a0, a1, a3 +; RV32-NEXT: srx a1, a2, a3 +; RV32-NEXT: ret +; +; RV64-LABEL: test_pslide1down_v8i8: +; RV64: # %bb.0: +; RV64-NEXT: li a2, 8 +; RV64-NEXT: srx a0, a1, a2 +; RV64-NEXT: ret + %scalar = insertelement <8 x i8> poison, i8 %rs1, i64 0 + %res = shufflevector <8 x i8> %rd, <8 x i8> %scalar, <8 x i32> <i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8> + ret <8 x i8> %res +} + +define <4 x i16> @test_pslide1up_v4i16(<4 x i16> %rd, i16 %rs1) { +; RV32-LABEL: test_pslide1up_v4i16: +; RV32: # %bb.0: +; RV32-NEXT: li a3, 16 +; RV32-NEXT: slli a2, a2, 16 +; RV32-NEXT: slx a1, a0, a3 +; RV32-NEXT: slx a0, a2, a3 +; RV32-NEXT: ret +; +; RV64-LABEL: test_pslide1up_v4i16: +; RV64: # %bb.0: +; RV64-NEXT: slli a1, a1, 48 +; RV64-NEXT: li a2, 16 +; RV64-NEXT: slx a0, a1, a2 +; RV64-NEXT: ret + %scalar = insertelement <4 x i16> poison, i16 %rs1, i64 0 + %res = shufflevector <4 x i16> %rd, <4 x i16> %scalar, <4 x i32> <i32 4, i32 0, i32 1, i32 2> + ret <4 x i16> %res +} + +define <4 x i16> @test_pslide1down_v4i16(<4 x i16> %rd, i16 %rs1) { +; RV32-LABEL: test_pslide1down_v4i16: +; RV32: # %bb.0: +; RV32-NEXT: li a3, 16 +; RV32-NEXT: srx a0, a1, a3 +; RV32-NEXT: srx a1, a2, a3 +; RV32-NEXT: ret +; +; RV64-LABEL: test_pslide1down_v4i16: +; RV64: # %bb.0: +; RV64-NEXT: li a2, 16 +; RV64-NEXT: srx a0, a1, a2 +; RV64-NEXT: ret + %scalar = insertelement <4 x i16> poison, i16 %rs1, i64 0 + %res = shufflevector <4 x i16> %rd, <4 x i16> %scalar, <4 x i32> <i32 1, i32 2, i32 3, i32 4> + ret <4 x i16> %res +} + +define <2 x i32> @test_pslide1up_v2i32(<2 x i32> %rd, i32 %rs1) { +; RV32-LABEL: test_pslide1up_v2i32: +; RV32: # %bb.0: +; RV32-NEXT: mv a1, a0 +; RV32-NEXT: mv a0, a2 +; RV32-NEXT: ret +; +; RV64-LABEL: test_pslide1up_v2i32: +; RV64: # %bb.0: +; RV64-NEXT: pack a0, a1, a0 +; RV64-NEXT: ret + %scalar = insertelement <2 x i32> poison, i32 %rs1, i64 0 + %res = shufflevector <2 x i32> %rd, <2 x i32> %scalar, <2 x i32> <i32 2, i32 0> + ret <2 x i32> %res +} + +define <2 x i32> @test_pslide1down_v2i32(<2 x i32> %rd, i32 %rs1) { +; RV32-LABEL: test_pslide1down_v2i32: +; RV32: # %bb.0: +; RV32-NEXT: mv a0, a1 +; RV32-NEXT: mv a1, a2 +; RV32-NEXT: ret +; +; RV64-LABEL: test_pslide1down_v2i32: +; RV64: # %bb.0: +; RV64-NEXT: ppairoe.w a0, a0, a1 +; RV64-NEXT: ret + %scalar = insertelement <2 x i32> poison, i32 %rs1, i64 0 + %res = shufflevector <2 x i32> %rd, <2 x i32> %scalar, <2 x i32> <i32 1, i32 2> + ret <2 x i32> %res +} _______________________________________________ cfe-commits mailing list [email protected] https://lists.llvm.org/cgi-bin/mailman/listinfo/cfe-commits
