https://github.com/StarryCSF created https://github.com/llvm/llvm-project/pull/226505
Add intrinsics, Clang builtins and SelectionDAG support for the packed widening multiply-accumulate operations. See https://github.com/riscv/riscv-p-spec/blob/master/P-ext-intrinsics.adoc#packed-widening-multiply-accumulate. On RV32, the pwmacc.h family accumulates into the register pair holding rd. On RV64, the operations are lowered according to the spec decomposition table, using zip16p and the pmacc.w.h01/pmaccu.w.h01 forms, or pwcvtu.wh followed by pmaccsu.w.h00 for the signed-unsigned form. >From bdcb7e54ff5362b6ddb5dd024cd6fd740f22bddc Mon Sep 17 00:00:00 2001 From: "ZhiQiang.Fan" <[email protected]> Date: Fri, 25 Sep 2026 22:05:06 +0800 Subject: [PATCH] [RISCV][P-ext] Add Packed Widening Multiply Accumulate intrinsics RV32 selects the pwmacc.h family, which accumulates into a register pair; RV64 follows the spec's RV64 decomposition sequences. --- clang/include/clang/Basic/BuiltinsRISCV.td | 5 ++ clang/lib/CodeGen/TargetBuiltins/RISCV.cpp | 21 ++++++ clang/lib/Headers/riscv_packed_simd.h | 5 ++ clang/test/CodeGen/RISCV/rvp-intrinsics.c | 73 +++++++++++++++++++ .../riscv_packed_simd.c | 25 +++++++ llvm/include/llvm/IR/IntrinsicsRISCV.td | 13 ++++ llvm/lib/Target/RISCV/RISCVISelDAGToDAG.cpp | 12 +++ llvm/lib/Target/RISCV/RISCVISelLowering.cpp | 58 +++++++++++++++ llvm/lib/Target/RISCV/RISCVInstrInfoP.td | 18 +++++ llvm/test/CodeGen/RISCV/rvp-simd-64.ll | 49 +++++++++++++ 10 files changed, 279 insertions(+) diff --git a/clang/include/clang/Basic/BuiltinsRISCV.td b/clang/include/clang/Basic/BuiltinsRISCV.td index 4cc4cc2169f01..b5275015fad8f 100644 --- a/clang/include/clang/Basic/BuiltinsRISCV.td +++ b/clang/include/clang/Basic/BuiltinsRISCV.td @@ -327,6 +327,11 @@ def pm4add_i16x4 : RISCVBuiltin<"int64_t(_Vector<4, short>, _Vector<4, short>)"> def pm4addu_u16x4 : RISCVBuiltin<"uint64_t(_Vector<4, unsigned short>, _Vector<4, unsigned short>)">; def pm4addsu_i16x4 : RISCVBuiltin<"int64_t(_Vector<4, short>, _Vector<4, unsigned short>)">; +// Packed Widening Multiply Accumulate +def pwmacc_i32x2 : RISCVBuiltin<"_Vector<2, int>(_Vector<2, int>, _Vector<2, short>, _Vector<2, short>)">; +def pwmaccu_u32x2 : RISCVBuiltin<"_Vector<2, unsigned int>(_Vector<2, unsigned int>, _Vector<2, unsigned short>, _Vector<2, unsigned short>)">; +def pwmaccsu_i32x2 : RISCVBuiltin<"_Vector<2, int>(_Vector<2, int>, _Vector<2, short>, _Vector<2, unsigned short>)">; + // Packed Absolute Difference Sum (32-bit) def pabdsumu_u8x4_u32 : RISCVBuiltin<"unsigned int(_Vector<4, unsigned char>, _Vector<4, unsigned char>)">; def pabdsumau_u8x4_u32 : RISCVBuiltin<"unsigned int(unsigned int, _Vector<4, unsigned char>, _Vector<4, unsigned char>)">; diff --git a/clang/lib/CodeGen/TargetBuiltins/RISCV.cpp b/clang/lib/CodeGen/TargetBuiltins/RISCV.cpp index dc44fd06e035a..3b804c8f2b571 100644 --- a/clang/lib/CodeGen/TargetBuiltins/RISCV.cpp +++ b/clang/lib/CodeGen/TargetBuiltins/RISCV.cpp @@ -1648,6 +1648,27 @@ Value *CodeGenFunction::EmitRISCVBuiltinExpr(unsigned BuiltinID, break; } + // Packed Widening Multiply Accumulate + case RISCV::BI__builtin_riscv_pwmacc_i32x2: + case RISCV::BI__builtin_riscv_pwmaccu_u32x2: + case RISCV::BI__builtin_riscv_pwmaccsu_i32x2: { + switch (BuiltinID) { + default: + llvm_unreachable("unexpected builtin ID"); + case RISCV::BI__builtin_riscv_pwmacc_i32x2: + ID = Intrinsic::riscv_pwmacc_i32x2; + break; + case RISCV::BI__builtin_riscv_pwmaccu_u32x2: + ID = Intrinsic::riscv_pwmaccu_u32x2; + break; + case RISCV::BI__builtin_riscv_pwmaccsu_i32x2: + ID = Intrinsic::riscv_pwmaccsu_i32x2; + break; + } + IntrinsicTypes = {ResultType, Ops[1]->getType()}; + break; + } + // Packed Reduction Sum case RISCV::BI__builtin_riscv_predsum_i8x4_i32: case RISCV::BI__builtin_riscv_predsum_i16x2_i32: diff --git a/clang/lib/Headers/riscv_packed_simd.h b/clang/lib/Headers/riscv_packed_simd.h index 99a6f7b1d6e9f..28c661612d89b 100644 --- a/clang/lib/Headers/riscv_packed_simd.h +++ b/clang/lib/Headers/riscv_packed_simd.h @@ -1021,6 +1021,11 @@ __packed_binary_builtin_mixed(pm4add_i16x4, int64_t, int16x4_t, int16x4_t, __bui __packed_binary_builtin_mixed(pm4addu_u16x4, uint64_t, uint16x4_t, uint16x4_t, __builtin_riscv_pm4addu_u16x4) __packed_binary_builtin_mixed(pm4addsu_i16x4, int64_t, int16x4_t, uint16x4_t, __builtin_riscv_pm4addsu_i16x4) +/* Packed Widening Multiply Accumulate */ +__packed_ternary_builtin_mixed(pwmacc_i32x2, int32x2_t, int16x2_t, int16x2_t, __builtin_riscv_pwmacc_i32x2) +__packed_ternary_builtin_mixed(pwmaccu_u32x2, uint32x2_t, uint16x2_t, uint16x2_t, __builtin_riscv_pwmaccu_u32x2) +__packed_ternary_builtin_mixed(pwmaccsu_i32x2, int32x2_t, int16x2_t, uint16x2_t, __builtin_riscv_pwmaccsu_i32x2) + /* Packed Absolute Difference Sum (32-bit) */ __packed_abdsum(pabdsumu_u8x4_u32, uint32_t, uint8x4_t, __builtin_riscv_pabdsumu_u8x4_u32) __packed_ternary_builtin_cast(pabdsumau_u8x4_u32, uint32_t, uint8x4_t, __builtin_riscv_pabdsumau_u8x4_u32) diff --git a/clang/test/CodeGen/RISCV/rvp-intrinsics.c b/clang/test/CodeGen/RISCV/rvp-intrinsics.c index 1184683f1e333..bdcf9ae0a4360 100644 --- a/clang/test/CodeGen/RISCV/rvp-intrinsics.c +++ b/clang/test/CodeGen/RISCV/rvp-intrinsics.c @@ -13116,3 +13116,76 @@ int16x4_t test_psati_i16x4(int16x4_t a) { return __riscv_psati_i16x4(a, 8); } // RV64-NEXT: ret i64 [[TMP2]] // int32x2_t test_psati_i32x2(int32x2_t a) { return __riscv_psati_i32x2(a, 16); } + +/* Packed Widening Multiply Accumulate */ +// RV32-LABEL: define dso_local i64 @test_pwmacc_i32x2( +// RV32-SAME: i64 noundef [[RD_COERCE:%.*]], i32 noundef [[RS1_COERCE:%.*]], i32 noundef [[RS2_COERCE:%.*]]) #[[ATTR0]] { +// RV32-NEXT: [[ENTRY:.*:]] +// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <2 x i32> +// RV32-NEXT: [[TMP1:%.*]] = bitcast i32 [[RS1_COERCE]] to <2 x i16> +// RV32-NEXT: [[TMP2:%.*]] = bitcast i32 [[RS2_COERCE]] to <2 x i16> +// RV32-NEXT: [[TMP3:%.*]] = call <2 x i32> @llvm.riscv.pwmacc.i32x2.v2i32.v2i16(<2 x i32> [[TMP0]], <2 x i16> [[TMP1]], <2 x i16> [[TMP2]]) +// RV32-NEXT: [[TMP4:%.*]] = bitcast <2 x i32> [[TMP3]] to i64 +// RV32-NEXT: ret i64 [[TMP4]] +// +// RV64-LABEL: define dso_local i64 @test_pwmacc_i32x2( +// RV64-SAME: i64 noundef [[RD_COERCE:%.*]], i32 noundef [[RS1_COERCE:%.*]], i32 noundef [[RS2_COERCE:%.*]]) #[[ATTR0]] { +// RV64-NEXT: [[ENTRY:.*:]] +// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <2 x i32> +// RV64-NEXT: [[TMP1:%.*]] = bitcast i32 [[RS1_COERCE]] to <2 x i16> +// RV64-NEXT: [[TMP2:%.*]] = bitcast i32 [[RS2_COERCE]] to <2 x i16> +// RV64-NEXT: [[TMP3:%.*]] = call <2 x i32> @llvm.riscv.pwmacc.i32x2.v2i32.v2i16(<2 x i32> [[TMP0]], <2 x i16> [[TMP1]], <2 x i16> [[TMP2]]) +// RV64-NEXT: [[TMP4:%.*]] = bitcast <2 x i32> [[TMP3]] to i64 +// RV64-NEXT: ret i64 [[TMP4]] +// +int32x2_t test_pwmacc_i32x2(int32x2_t rd, int16x2_t rs1, int16x2_t rs2) { + return __riscv_pwmacc_i32x2(rd, rs1, rs2); +} + +// RV32-LABEL: define dso_local i64 @test_pwmaccu_u32x2( +// RV32-SAME: i64 noundef [[RD_COERCE:%.*]], i32 noundef [[RS1_COERCE:%.*]], i32 noundef [[RS2_COERCE:%.*]]) #[[ATTR0]] { +// RV32-NEXT: [[ENTRY:.*:]] +// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <2 x i32> +// RV32-NEXT: [[TMP1:%.*]] = bitcast i32 [[RS1_COERCE]] to <2 x i16> +// RV32-NEXT: [[TMP2:%.*]] = bitcast i32 [[RS2_COERCE]] to <2 x i16> +// RV32-NEXT: [[TMP3:%.*]] = call <2 x i32> @llvm.riscv.pwmaccu.u32x2.v2i32.v2i16(<2 x i32> [[TMP0]], <2 x i16> [[TMP1]], <2 x i16> [[TMP2]]) +// RV32-NEXT: [[TMP4:%.*]] = bitcast <2 x i32> [[TMP3]] to i64 +// RV32-NEXT: ret i64 [[TMP4]] +// +// RV64-LABEL: define dso_local i64 @test_pwmaccu_u32x2( +// RV64-SAME: i64 noundef [[RD_COERCE:%.*]], i32 noundef [[RS1_COERCE:%.*]], i32 noundef [[RS2_COERCE:%.*]]) #[[ATTR0]] { +// RV64-NEXT: [[ENTRY:.*:]] +// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <2 x i32> +// RV64-NEXT: [[TMP1:%.*]] = bitcast i32 [[RS1_COERCE]] to <2 x i16> +// RV64-NEXT: [[TMP2:%.*]] = bitcast i32 [[RS2_COERCE]] to <2 x i16> +// RV64-NEXT: [[TMP3:%.*]] = call <2 x i32> @llvm.riscv.pwmaccu.u32x2.v2i32.v2i16(<2 x i32> [[TMP0]], <2 x i16> [[TMP1]], <2 x i16> [[TMP2]]) +// RV64-NEXT: [[TMP4:%.*]] = bitcast <2 x i32> [[TMP3]] to i64 +// RV64-NEXT: ret i64 [[TMP4]] +// +uint32x2_t test_pwmaccu_u32x2(uint32x2_t rd, uint16x2_t rs1, uint16x2_t rs2) { + return __riscv_pwmaccu_u32x2(rd, rs1, rs2); +} + +// RV32-LABEL: define dso_local i64 @test_pwmaccsu_i32x2( +// RV32-SAME: i64 noundef [[RD_COERCE:%.*]], i32 noundef [[RS1_COERCE:%.*]], i32 noundef [[RS2_COERCE:%.*]]) #[[ATTR0]] { +// RV32-NEXT: [[ENTRY:.*:]] +// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <2 x i32> +// RV32-NEXT: [[TMP1:%.*]] = bitcast i32 [[RS1_COERCE]] to <2 x i16> +// RV32-NEXT: [[TMP2:%.*]] = bitcast i32 [[RS2_COERCE]] to <2 x i16> +// RV32-NEXT: [[TMP3:%.*]] = call <2 x i32> @llvm.riscv.pwmaccsu.i32x2.v2i32.v2i16(<2 x i32> [[TMP0]], <2 x i16> [[TMP1]], <2 x i16> [[TMP2]]) +// RV32-NEXT: [[TMP4:%.*]] = bitcast <2 x i32> [[TMP3]] to i64 +// RV32-NEXT: ret i64 [[TMP4]] +// +// RV64-LABEL: define dso_local i64 @test_pwmaccsu_i32x2( +// RV64-SAME: i64 noundef [[RD_COERCE:%.*]], i32 noundef [[RS1_COERCE:%.*]], i32 noundef [[RS2_COERCE:%.*]]) #[[ATTR0]] { +// RV64-NEXT: [[ENTRY:.*:]] +// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <2 x i32> +// RV64-NEXT: [[TMP1:%.*]] = bitcast i32 [[RS1_COERCE]] to <2 x i16> +// RV64-NEXT: [[TMP2:%.*]] = bitcast i32 [[RS2_COERCE]] to <2 x i16> +// RV64-NEXT: [[TMP3:%.*]] = call <2 x i32> @llvm.riscv.pwmaccsu.i32x2.v2i32.v2i16(<2 x i32> [[TMP0]], <2 x i16> [[TMP1]], <2 x i16> [[TMP2]]) +// RV64-NEXT: [[TMP4:%.*]] = bitcast <2 x i32> [[TMP3]] to i64 +// RV64-NEXT: ret i64 [[TMP4]] +// +int32x2_t test_pwmaccsu_i32x2(int32x2_t rd, int16x2_t rs1, uint16x2_t rs2) { + return __riscv_pwmaccsu_i32x2(rd, rs1, rs2); +} diff --git a/cross-project-tests/intrinsic-header-tests/riscv_packed_simd.c b/cross-project-tests/intrinsic-header-tests/riscv_packed_simd.c index 1e8d1c02fa66c..c33d03324ac6f 100644 --- a/cross-project-tests/intrinsic-header-tests/riscv_packed_simd.c +++ b/cross-project-tests/intrinsic-header-tests/riscv_packed_simd.c @@ -4051,6 +4051,31 @@ int64_t test_pm4addsu_i16x4(int16x4_t rs1, uint16x4_t rs2) { return __riscv_pm4addsu_i16x4(rs1, rs2); } +// Packed Widening Multiply Accumulate. +// CHECK-LABEL: test_pwmacc_i32x2: +// RV32: pwmacc.h +// RV64: zip16p +// RV64: pmacc.w.h01 +int32x2_t test_pwmacc_i32x2(int32x2_t rd, int16x2_t rs1, int16x2_t rs2) { + return __riscv_pwmacc_i32x2(rd, rs1, rs2); +} + +// CHECK-LABEL: test_pwmaccu_u32x2: +// RV32: pwmaccu.h +// RV64: zip16p +// RV64: pmaccu.w.h01 +uint32x2_t test_pwmaccu_u32x2(uint32x2_t rd, uint16x2_t rs1, uint16x2_t rs2) { + return __riscv_pwmaccu_u32x2(rd, rs1, rs2); +} + +// CHECK-LABEL: test_pwmaccsu_i32x2: +// RV32: pwmaccsu.h +// RV64: pwcvtu.wh +// RV64: pmaccsu.w.h00 +int32x2_t test_pwmaccsu_i32x2(int32x2_t rd, int16x2_t rs1, uint16x2_t rs2) { + return __riscv_pwmaccsu_i32x2(rd, rs1, rs2); +} + // Packed Multiply Parts. // CHECK-LABEL: test_pmul_b00_i16x2: // RV32: pmul.h.b00 diff --git a/llvm/include/llvm/IR/IntrinsicsRISCV.td b/llvm/include/llvm/IR/IntrinsicsRISCV.td index 0ed8a3fce15c0..b589ea12375df 100644 --- a/llvm/include/llvm/IR/IntrinsicsRISCV.td +++ b/llvm/include/llvm/IR/IntrinsicsRISCV.td @@ -2228,6 +2228,19 @@ class RVPBinaryIntrinsic def int_riscv_pmaccsu_00 : RVPPackedMulPartsAccIntrinsic; def int_riscv_pmaccsu_11 : RVPPackedMulPartsAccIntrinsic; + // Packed Widening Multiply Accumulate. + // The pwmacc*/pwmaccsu* instructions are available on RV32. + // On RV64, these operations are synthesized using the instruction + // sequences specified by the P extension. + class RVPWideningMulAccIntrinsic + : DefaultAttrsIntrinsic<[llvm_anyvector_ty], + [LLVMMatchType<0>, llvm_anyvector_ty, + LLVMMatchType<1>], + [IntrNoMem, IntrSpeculatable]>; + def int_riscv_pwmacc_i32x2 : RVPWideningMulAccIntrinsic; + def int_riscv_pwmaccu_u32x2 : RVPWideningMulAccIntrinsic; + def int_riscv_pwmaccsu_i32x2 : RVPWideningMulAccIntrinsic; + class RVPScalarMulPartsAccIntrinsic : DefaultAttrsIntrinsic<[llvm_anyint_ty], [LLVMMatchType<0>, llvm_anyvector_ty, diff --git a/llvm/lib/Target/RISCV/RISCVISelDAGToDAG.cpp b/llvm/lib/Target/RISCV/RISCVISelDAGToDAG.cpp index 146aef612f387..3c2b8ee071c95 100644 --- a/llvm/lib/Target/RISCV/RISCVISelDAGToDAG.cpp +++ b/llvm/lib/Target/RISCV/RISCVISelDAGToDAG.cpp @@ -2149,6 +2149,9 @@ void RISCVDAGToDAGISel::Select(SDNode *Node) { } case RISCVISD::MQWACC: case RISCVISD::MQRWACC: + case RISCVISD::PWMACC_H: + case RISCVISD::PWMACCU_H: + case RISCVISD::PWMACCSU_H: case RISCVISD::WMACC: case RISCVISD::WMACCU: case RISCVISD::WMACCSU: { @@ -2167,6 +2170,15 @@ void RISCVDAGToDAGISel::Select(SDNode *Node) { case RISCVISD::MQRWACC: Opc = RISCV::MQRWACC; break; + case RISCVISD::PWMACC_H: + Opc = RISCV::PWMACC_H; + break; + case RISCVISD::PWMACCU_H: + Opc = RISCV::PWMACCU_H; + break; + case RISCVISD::PWMACCSU_H: + Opc = RISCV::PWMACCSU_H; + break; case RISCVISD::WMACC: Opc = RISCV::WMACC; break; diff --git a/llvm/lib/Target/RISCV/RISCVISelLowering.cpp b/llvm/lib/Target/RISCV/RISCVISelLowering.cpp index 7a5905fc4b3ea..0115e122703a2 100644 --- a/llvm/lib/Target/RISCV/RISCVISelLowering.cpp +++ b/llvm/lib/Target/RISCV/RISCVISelLowering.cpp @@ -12524,6 +12524,19 @@ static unsigned getRVPHorizontalMulOpcode(unsigned IntNo) { } } +static unsigned getRVPWideningMulAccPairOpcode(unsigned IntNo) { + switch (IntNo) { + default: + llvm_unreachable("Unexpected RISC-V packed pwmacc intrinsic"); + case Intrinsic::riscv_pwmacc_i32x2: + return RISCVISD::PWMACC_H; + case Intrinsic::riscv_pwmaccu_u32x2: + return RISCVISD::PWMACCU_H; + case Intrinsic::riscv_pwmaccsu_i32x2: + return RISCVISD::PWMACCSU_H; + } +} + static SDValue lowerRV32HorizontalMul64(unsigned IntNo, SDValue Rs1, SDValue Rs2, const SDLoc &DL, SelectionDAG &DAG) { @@ -13386,6 +13399,51 @@ SDValue RISCVTargetLowering::LowerINTRINSIC_WO_CHAIN(SDValue Op, return DAG.getNode(Opc, DL, VT, Rs1, Rs2); } + case Intrinsic::riscv_pwmacc_i32x2: + case Intrinsic::riscv_pwmaccu_u32x2: + case Intrinsic::riscv_pwmaccsu_i32x2: { + EVT VT = Op.getValueType(); + SDValue Acc = Op.getOperand(1); + SDValue Rs1 = Op.getOperand(2); + SDValue Rs2 = Op.getOperand(3); + MVT XLenVT = Subtarget.getXLenVT(); + + if (Subtarget.is64Bit()) { + // Per the spec decomposition table: zip16p duplicates each halfword + // into both positions of its 32-bit lane (the su form uses the + // pwcvtu.wh spelling, zip16p with x0), then pmacc.w.h01 or + // pmaccsu.w.h00 multiplies the lane-local halfwords into rd. The + // upper halfwords of the operands are not read. RV32 selects the + // pwmacc.h family on the register pair instead. + Rs1 = DAG.getNode(ISD::CONCAT_VECTORS, DL, MVT::v4i16, Rs1, + DAG.getUNDEF(MVT::v2i16)); + Rs2 = DAG.getNode(ISD::CONCAT_VECTORS, DL, MVT::v4i16, Rs2, + DAG.getUNDEF(MVT::v2i16)); + if (IntNo == Intrinsic::riscv_pwmaccsu_i32x2) { + SDValue Zero = DAG.getNode(ISD::SPLAT_VECTOR, DL, MVT::v4i16, + DAG.getConstant(0, DL, XLenVT)); + Rs1 = DAG.getNode(RISCVISD::ZIP16P, DL, MVT::v4i16, Rs1, Zero); + Rs2 = DAG.getNode(RISCVISD::ZIP16P, DL, MVT::v4i16, Rs2, Zero); + return DAG.getNode(RISCVISD::PMACCSU_HALVES_00, DL, VT, Acc, Rs1, Rs2); + } + Rs1 = DAG.getNode(RISCVISD::ZIP16P, DL, MVT::v4i16, Rs1, Rs1); + Rs2 = DAG.getNode(RISCVISD::ZIP16P, DL, MVT::v4i16, Rs2, Rs2); + unsigned Opc = IntNo == Intrinsic::riscv_pwmaccu_u32x2 + ? RISCVISD::PMACCU_HALVES_01 + : RISCVISD::PMACC_HALVES_01; + return DAG.getNode(Opc, DL, VT, Acc, Rs1, Rs2); + } + + // RV32 uses pwmacc.h/pwmaccsu.h on the 64-bit accumulator register + // pair, exposed as a two-result node over the accumulator words. + SDValue RdLo = DAG.getExtractVectorElt(DL, XLenVT, Acc, 0); + SDValue RdHi = DAG.getExtractVectorElt(DL, XLenVT, Acc, 1); + SDVTList VTs = DAG.getVTList(XLenVT, XLenVT); + SDValue Res = DAG.getNode(getRVPWideningMulAccPairOpcode(IntNo), DL, VTs, + {RdLo, RdHi, Rs1, Rs2}); + return DAG.getNode(ISD::BUILD_VECTOR, DL, VT, Res.getValue(0), + Res.getValue(1)); + } case Intrinsic::riscv_pmerge: { EVT VT = Op.getValueType(); auto buildMerge = [&](SDValue Rs1, SDValue Rs2, SDValue Mask, diff --git a/llvm/lib/Target/RISCV/RISCVInstrInfoP.td b/llvm/lib/Target/RISCV/RISCVInstrInfoP.td index 4d2c8d9be1073..e0f021fd652f2 100644 --- a/llvm/lib/Target/RISCV/RISCVInstrInfoP.td +++ b/llvm/lib/Target/RISCV/RISCVInstrInfoP.td @@ -1977,6 +1977,20 @@ def riscv_pmaccsu_halves_00 def riscv_pmaccsu_halves_11 : RVSDNode<"PMACCSU_HALVES_11", SDT_RISCVWideningMulAccByHalves>; +// The pwmacc*/pwmaccsu* instructions use a 64-bit accumulator on RV32. +// The accumulator and result are represented as two i32 values in the +// SelectionDAG; custom ISel combines and splits the register pair. +def SDT_RISCVPackedWideningMulAccPair + : SDTypeProfile<2, 4, [SDTCisVT<0, i32>, SDTCisVT<1, i32>, + SDTCisSameAs<0, 2>, SDTCisSameAs<1, 3>, + SDTCisVec<4>, SDTCisSameAs<4, 5>]>; +def riscv_pwmacc_h : RVSDNode<"PWMACC_H", + SDT_RISCVPackedWideningMulAccPair>; +def riscv_pwmaccu_h : RVSDNode<"PWMACCU_H", + SDT_RISCVPackedWideningMulAccPair>; +def riscv_pwmaccsu_h : RVSDNode<"PWMACCSU_H", + SDT_RISCVPackedWideningMulAccPair>; + def SDT_RISCVWideningShiftLeft : SDTypeProfile<2, 2, [SDTCisVT<0, i32>, SDTCisSameAs<0, 1>, SDTCisSameAs<0, 2>, @@ -2011,6 +2025,7 @@ def SDT_RISCVPackedBinary : SDTypeProfile<1, 2, [SDTCisVec<0>, SDTCisSameAs<0, 1>, SDTCisSameAs<0, 2>]>; def riscv_ppairoe_h : RVSDNode<"PPAIROE_H", SDT_RISCVPackedBinary>; +def riscv_zip16p : RVSDNode<"ZIP16P", SDT_RISCVPackedBinary>; // Averaging subtraction, (a - b) >> 2 def riscv_asub : RVSDNode<"ASUB", SDTIntBinOp>; @@ -3228,6 +3243,9 @@ let append Predicates = [IsRV64] in { def : PatMulPartsAcc<riscv_pmaccsu_halves_00, PMACCSU_W_H00, v2i32, v4i16>; def : PatMulPartsAcc<riscv_pmaccsu_halves_11, PMACCSU_W_H11, v2i32, v4i16>; + def : Pat<(v4i16 (riscv_zip16p (v4i16 GPR:$rs1), (v4i16 GPR:$rs2))), + (ZIP16P GPR:$rs1, GPR:$rs2)>; + // Scalar word multiply-parts accumulate patterns. def : PatMulPartsAcc<riscv_pmacc_halves_00, MACC_W00, i64, v2i32>; def : PatMulPartsAcc<riscv_pmacc_halves_01, MACC_W01, i64, v2i32>; diff --git a/llvm/test/CodeGen/RISCV/rvp-simd-64.ll b/llvm/test/CodeGen/RISCV/rvp-simd-64.ll index 34a00069a5597..fdfc97042ddc4 100644 --- a/llvm/test/CodeGen/RISCV/rvp-simd-64.ll +++ b/llvm/test/CodeGen/RISCV/rvp-simd-64.ll @@ -8004,6 +8004,55 @@ define <2 x i32> @test_pmaccsu_h11_v2i32(<2 x i32> %rd, <4 x i16> %a, <4 x i16> ret <2 x i32> %r } +; Packed Widening Multiply Accumulate +define <2 x i32> @test_pwmacc_i32x2(<2 x i32> %rd, <2 x i16> %rs1, <2 x i16> %rs2) { +; RV32-LABEL: test_pwmacc_i32x2: +; RV32: # %bb.0: +; RV32-NEXT: pwmacc.h a0, a2, a3 +; RV32-NEXT: ret +; +; RV64-LABEL: test_pwmacc_i32x2: +; RV64: # %bb.0: +; RV64-NEXT: zip16p a2, a2, a2 +; RV64-NEXT: zip16p a1, a1, a1 +; RV64-NEXT: pmacc.w.h01 a0, a1, a2 +; RV64-NEXT: ret + %res = call <2 x i32> @llvm.riscv.pwmacc.i32x2.v2i32.v2i16(<2 x i32> %rd, <2 x i16> %rs1, <2 x i16> %rs2) + ret <2 x i32> %res +} + +define <2 x i32> @test_pwmaccu_u32x2(<2 x i32> %rd, <2 x i16> %rs1, <2 x i16> %rs2) { +; RV32-LABEL: test_pwmaccu_u32x2: +; RV32: # %bb.0: +; RV32-NEXT: pwmaccu.h a0, a2, a3 +; RV32-NEXT: ret +; +; RV64-LABEL: test_pwmaccu_u32x2: +; RV64: # %bb.0: +; RV64-NEXT: zip16p a2, a2, a2 +; RV64-NEXT: zip16p a1, a1, a1 +; RV64-NEXT: pmaccu.w.h01 a0, a1, a2 +; RV64-NEXT: ret + %res = call <2 x i32> @llvm.riscv.pwmaccu.u32x2.v2i32.v2i16(<2 x i32> %rd, <2 x i16> %rs1, <2 x i16> %rs2) + ret <2 x i32> %res +} + +define <2 x i32> @test_pwmaccsu_i32x2(<2 x i32> %rd, <2 x i16> %rs1, <2 x i16> %rs2) { +; RV32-LABEL: test_pwmaccsu_i32x2: +; RV32: # %bb.0: +; RV32-NEXT: pwmaccsu.h a0, a2, a3 +; RV32-NEXT: ret +; +; RV64-LABEL: test_pwmaccsu_i32x2: +; RV64: # %bb.0: +; RV64-NEXT: pwcvtu.wh a2, a2 +; RV64-NEXT: pwcvtu.wh a1, a1 +; RV64-NEXT: pmaccsu.w.h00 a0, a1, a2 +; RV64-NEXT: ret + %res = call <2 x i32> @llvm.riscv.pwmaccsu.i32x2.v2i32.v2i16(<2 x i32> %rd, <2 x i16> %rs1, <2 x i16> %rs2) + ret <2 x i32> %res +} + define i64 @test_macc_w00_i64(i64 %rd, <2 x i32> %a, <2 x i32> %b) { ; RV32-LABEL: test_macc_w00_i64: ; RV32: # %bb.0: _______________________________________________ cfe-commits mailing list [email protected] https://lists.llvm.org/cgi-bin/mailman/listinfo/cfe-commits
