[clang] [llvm] [Clang][RISCV][P-ext] Add Packed Widening Multiply Accumulate intrinsics (PR #226505)
via llvm-commits
llvm-commits at lists.llvm.org
Mon Sep 28 04:00:45 PDT 2026
https://github.com/StarryCSF updated https://github.com/llvm/llvm-project/pull/226505
>From bdcb7e54ff5362b6ddb5dd024cd6fd740f22bddc Mon Sep 17 00:00:00 2001
From: "ZhiQiang.Fan" <571253675 at QQ.com>
Date: Fri, 25 Sep 2026 22:05:06 +0800
Subject: [PATCH 1/2] [RISCV][P-ext] Add Packed Widening Multiply Accumulate
intrinsics
RV32 selects the pwmacc.h family, which accumulates into a register
pair; RV64 follows the spec's RV64 decomposition sequences.
---
clang/include/clang/Basic/BuiltinsRISCV.td | 5 ++
clang/lib/CodeGen/TargetBuiltins/RISCV.cpp | 21 ++++++
clang/lib/Headers/riscv_packed_simd.h | 5 ++
clang/test/CodeGen/RISCV/rvp-intrinsics.c | 73 +++++++++++++++++++
.../riscv_packed_simd.c | 25 +++++++
llvm/include/llvm/IR/IntrinsicsRISCV.td | 13 ++++
llvm/lib/Target/RISCV/RISCVISelDAGToDAG.cpp | 12 +++
llvm/lib/Target/RISCV/RISCVISelLowering.cpp | 58 +++++++++++++++
llvm/lib/Target/RISCV/RISCVInstrInfoP.td | 18 +++++
llvm/test/CodeGen/RISCV/rvp-simd-64.ll | 49 +++++++++++++
10 files changed, 279 insertions(+)
diff --git a/clang/include/clang/Basic/BuiltinsRISCV.td b/clang/include/clang/Basic/BuiltinsRISCV.td
index 4cc4cc2169f01..b5275015fad8f 100644
--- a/clang/include/clang/Basic/BuiltinsRISCV.td
+++ b/clang/include/clang/Basic/BuiltinsRISCV.td
@@ -327,6 +327,11 @@ def pm4add_i16x4 : RISCVBuiltin<"int64_t(_Vector<4, short>, _Vector<4, short>)">
def pm4addu_u16x4 : RISCVBuiltin<"uint64_t(_Vector<4, unsigned short>, _Vector<4, unsigned short>)">;
def pm4addsu_i16x4 : RISCVBuiltin<"int64_t(_Vector<4, short>, _Vector<4, unsigned short>)">;
+// Packed Widening Multiply Accumulate
+def pwmacc_i32x2 : RISCVBuiltin<"_Vector<2, int>(_Vector<2, int>, _Vector<2, short>, _Vector<2, short>)">;
+def pwmaccu_u32x2 : RISCVBuiltin<"_Vector<2, unsigned int>(_Vector<2, unsigned int>, _Vector<2, unsigned short>, _Vector<2, unsigned short>)">;
+def pwmaccsu_i32x2 : RISCVBuiltin<"_Vector<2, int>(_Vector<2, int>, _Vector<2, short>, _Vector<2, unsigned short>)">;
+
// Packed Absolute Difference Sum (32-bit)
def pabdsumu_u8x4_u32 : RISCVBuiltin<"unsigned int(_Vector<4, unsigned char>, _Vector<4, unsigned char>)">;
def pabdsumau_u8x4_u32 : RISCVBuiltin<"unsigned int(unsigned int, _Vector<4, unsigned char>, _Vector<4, unsigned char>)">;
diff --git a/clang/lib/CodeGen/TargetBuiltins/RISCV.cpp b/clang/lib/CodeGen/TargetBuiltins/RISCV.cpp
index dc44fd06e035a..3b804c8f2b571 100644
--- a/clang/lib/CodeGen/TargetBuiltins/RISCV.cpp
+++ b/clang/lib/CodeGen/TargetBuiltins/RISCV.cpp
@@ -1648,6 +1648,27 @@ Value *CodeGenFunction::EmitRISCVBuiltinExpr(unsigned BuiltinID,
break;
}
+ // Packed Widening Multiply Accumulate
+ case RISCV::BI__builtin_riscv_pwmacc_i32x2:
+ case RISCV::BI__builtin_riscv_pwmaccu_u32x2:
+ case RISCV::BI__builtin_riscv_pwmaccsu_i32x2: {
+ switch (BuiltinID) {
+ default:
+ llvm_unreachable("unexpected builtin ID");
+ case RISCV::BI__builtin_riscv_pwmacc_i32x2:
+ ID = Intrinsic::riscv_pwmacc_i32x2;
+ break;
+ case RISCV::BI__builtin_riscv_pwmaccu_u32x2:
+ ID = Intrinsic::riscv_pwmaccu_u32x2;
+ break;
+ case RISCV::BI__builtin_riscv_pwmaccsu_i32x2:
+ ID = Intrinsic::riscv_pwmaccsu_i32x2;
+ break;
+ }
+ IntrinsicTypes = {ResultType, Ops[1]->getType()};
+ break;
+ }
+
// Packed Reduction Sum
case RISCV::BI__builtin_riscv_predsum_i8x4_i32:
case RISCV::BI__builtin_riscv_predsum_i16x2_i32:
diff --git a/clang/lib/Headers/riscv_packed_simd.h b/clang/lib/Headers/riscv_packed_simd.h
index 99a6f7b1d6e9f..28c661612d89b 100644
--- a/clang/lib/Headers/riscv_packed_simd.h
+++ b/clang/lib/Headers/riscv_packed_simd.h
@@ -1021,6 +1021,11 @@ __packed_binary_builtin_mixed(pm4add_i16x4, int64_t, int16x4_t, int16x4_t, __bui
__packed_binary_builtin_mixed(pm4addu_u16x4, uint64_t, uint16x4_t, uint16x4_t, __builtin_riscv_pm4addu_u16x4)
__packed_binary_builtin_mixed(pm4addsu_i16x4, int64_t, int16x4_t, uint16x4_t, __builtin_riscv_pm4addsu_i16x4)
+/* Packed Widening Multiply Accumulate */
+__packed_ternary_builtin_mixed(pwmacc_i32x2, int32x2_t, int16x2_t, int16x2_t, __builtin_riscv_pwmacc_i32x2)
+__packed_ternary_builtin_mixed(pwmaccu_u32x2, uint32x2_t, uint16x2_t, uint16x2_t, __builtin_riscv_pwmaccu_u32x2)
+__packed_ternary_builtin_mixed(pwmaccsu_i32x2, int32x2_t, int16x2_t, uint16x2_t, __builtin_riscv_pwmaccsu_i32x2)
+
/* Packed Absolute Difference Sum (32-bit) */
__packed_abdsum(pabdsumu_u8x4_u32, uint32_t, uint8x4_t, __builtin_riscv_pabdsumu_u8x4_u32)
__packed_ternary_builtin_cast(pabdsumau_u8x4_u32, uint32_t, uint8x4_t, __builtin_riscv_pabdsumau_u8x4_u32)
diff --git a/clang/test/CodeGen/RISCV/rvp-intrinsics.c b/clang/test/CodeGen/RISCV/rvp-intrinsics.c
index 1184683f1e333..bdcf9ae0a4360 100644
--- a/clang/test/CodeGen/RISCV/rvp-intrinsics.c
+++ b/clang/test/CodeGen/RISCV/rvp-intrinsics.c
@@ -13116,3 +13116,76 @@ int16x4_t test_psati_i16x4(int16x4_t a) { return __riscv_psati_i16x4(a, 8); }
// RV64-NEXT: ret i64 [[TMP2]]
//
int32x2_t test_psati_i32x2(int32x2_t a) { return __riscv_psati_i32x2(a, 16); }
+
+/* Packed Widening Multiply Accumulate */
+// RV32-LABEL: define dso_local i64 @test_pwmacc_i32x2(
+// RV32-SAME: i64 noundef [[RD_COERCE:%.*]], i32 noundef [[RS1_COERCE:%.*]], i32 noundef [[RS2_COERCE:%.*]]) #[[ATTR0]] {
+// RV32-NEXT: [[ENTRY:.*:]]
+// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <2 x i32>
+// RV32-NEXT: [[TMP1:%.*]] = bitcast i32 [[RS1_COERCE]] to <2 x i16>
+// RV32-NEXT: [[TMP2:%.*]] = bitcast i32 [[RS2_COERCE]] to <2 x i16>
+// RV32-NEXT: [[TMP3:%.*]] = call <2 x i32> @llvm.riscv.pwmacc.i32x2.v2i32.v2i16(<2 x i32> [[TMP0]], <2 x i16> [[TMP1]], <2 x i16> [[TMP2]])
+// RV32-NEXT: [[TMP4:%.*]] = bitcast <2 x i32> [[TMP3]] to i64
+// RV32-NEXT: ret i64 [[TMP4]]
+//
+// RV64-LABEL: define dso_local i64 @test_pwmacc_i32x2(
+// RV64-SAME: i64 noundef [[RD_COERCE:%.*]], i32 noundef [[RS1_COERCE:%.*]], i32 noundef [[RS2_COERCE:%.*]]) #[[ATTR0]] {
+// RV64-NEXT: [[ENTRY:.*:]]
+// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <2 x i32>
+// RV64-NEXT: [[TMP1:%.*]] = bitcast i32 [[RS1_COERCE]] to <2 x i16>
+// RV64-NEXT: [[TMP2:%.*]] = bitcast i32 [[RS2_COERCE]] to <2 x i16>
+// RV64-NEXT: [[TMP3:%.*]] = call <2 x i32> @llvm.riscv.pwmacc.i32x2.v2i32.v2i16(<2 x i32> [[TMP0]], <2 x i16> [[TMP1]], <2 x i16> [[TMP2]])
+// RV64-NEXT: [[TMP4:%.*]] = bitcast <2 x i32> [[TMP3]] to i64
+// RV64-NEXT: ret i64 [[TMP4]]
+//
+int32x2_t test_pwmacc_i32x2(int32x2_t rd, int16x2_t rs1, int16x2_t rs2) {
+ return __riscv_pwmacc_i32x2(rd, rs1, rs2);
+}
+
+// RV32-LABEL: define dso_local i64 @test_pwmaccu_u32x2(
+// RV32-SAME: i64 noundef [[RD_COERCE:%.*]], i32 noundef [[RS1_COERCE:%.*]], i32 noundef [[RS2_COERCE:%.*]]) #[[ATTR0]] {
+// RV32-NEXT: [[ENTRY:.*:]]
+// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <2 x i32>
+// RV32-NEXT: [[TMP1:%.*]] = bitcast i32 [[RS1_COERCE]] to <2 x i16>
+// RV32-NEXT: [[TMP2:%.*]] = bitcast i32 [[RS2_COERCE]] to <2 x i16>
+// RV32-NEXT: [[TMP3:%.*]] = call <2 x i32> @llvm.riscv.pwmaccu.u32x2.v2i32.v2i16(<2 x i32> [[TMP0]], <2 x i16> [[TMP1]], <2 x i16> [[TMP2]])
+// RV32-NEXT: [[TMP4:%.*]] = bitcast <2 x i32> [[TMP3]] to i64
+// RV32-NEXT: ret i64 [[TMP4]]
+//
+// RV64-LABEL: define dso_local i64 @test_pwmaccu_u32x2(
+// RV64-SAME: i64 noundef [[RD_COERCE:%.*]], i32 noundef [[RS1_COERCE:%.*]], i32 noundef [[RS2_COERCE:%.*]]) #[[ATTR0]] {
+// RV64-NEXT: [[ENTRY:.*:]]
+// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <2 x i32>
+// RV64-NEXT: [[TMP1:%.*]] = bitcast i32 [[RS1_COERCE]] to <2 x i16>
+// RV64-NEXT: [[TMP2:%.*]] = bitcast i32 [[RS2_COERCE]] to <2 x i16>
+// RV64-NEXT: [[TMP3:%.*]] = call <2 x i32> @llvm.riscv.pwmaccu.u32x2.v2i32.v2i16(<2 x i32> [[TMP0]], <2 x i16> [[TMP1]], <2 x i16> [[TMP2]])
+// RV64-NEXT: [[TMP4:%.*]] = bitcast <2 x i32> [[TMP3]] to i64
+// RV64-NEXT: ret i64 [[TMP4]]
+//
+uint32x2_t test_pwmaccu_u32x2(uint32x2_t rd, uint16x2_t rs1, uint16x2_t rs2) {
+ return __riscv_pwmaccu_u32x2(rd, rs1, rs2);
+}
+
+// RV32-LABEL: define dso_local i64 @test_pwmaccsu_i32x2(
+// RV32-SAME: i64 noundef [[RD_COERCE:%.*]], i32 noundef [[RS1_COERCE:%.*]], i32 noundef [[RS2_COERCE:%.*]]) #[[ATTR0]] {
+// RV32-NEXT: [[ENTRY:.*:]]
+// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <2 x i32>
+// RV32-NEXT: [[TMP1:%.*]] = bitcast i32 [[RS1_COERCE]] to <2 x i16>
+// RV32-NEXT: [[TMP2:%.*]] = bitcast i32 [[RS2_COERCE]] to <2 x i16>
+// RV32-NEXT: [[TMP3:%.*]] = call <2 x i32> @llvm.riscv.pwmaccsu.i32x2.v2i32.v2i16(<2 x i32> [[TMP0]], <2 x i16> [[TMP1]], <2 x i16> [[TMP2]])
+// RV32-NEXT: [[TMP4:%.*]] = bitcast <2 x i32> [[TMP3]] to i64
+// RV32-NEXT: ret i64 [[TMP4]]
+//
+// RV64-LABEL: define dso_local i64 @test_pwmaccsu_i32x2(
+// RV64-SAME: i64 noundef [[RD_COERCE:%.*]], i32 noundef [[RS1_COERCE:%.*]], i32 noundef [[RS2_COERCE:%.*]]) #[[ATTR0]] {
+// RV64-NEXT: [[ENTRY:.*:]]
+// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <2 x i32>
+// RV64-NEXT: [[TMP1:%.*]] = bitcast i32 [[RS1_COERCE]] to <2 x i16>
+// RV64-NEXT: [[TMP2:%.*]] = bitcast i32 [[RS2_COERCE]] to <2 x i16>
+// RV64-NEXT: [[TMP3:%.*]] = call <2 x i32> @llvm.riscv.pwmaccsu.i32x2.v2i32.v2i16(<2 x i32> [[TMP0]], <2 x i16> [[TMP1]], <2 x i16> [[TMP2]])
+// RV64-NEXT: [[TMP4:%.*]] = bitcast <2 x i32> [[TMP3]] to i64
+// RV64-NEXT: ret i64 [[TMP4]]
+//
+int32x2_t test_pwmaccsu_i32x2(int32x2_t rd, int16x2_t rs1, uint16x2_t rs2) {
+ return __riscv_pwmaccsu_i32x2(rd, rs1, rs2);
+}
diff --git a/cross-project-tests/intrinsic-header-tests/riscv_packed_simd.c b/cross-project-tests/intrinsic-header-tests/riscv_packed_simd.c
index 1e8d1c02fa66c..c33d03324ac6f 100644
--- a/cross-project-tests/intrinsic-header-tests/riscv_packed_simd.c
+++ b/cross-project-tests/intrinsic-header-tests/riscv_packed_simd.c
@@ -4051,6 +4051,31 @@ int64_t test_pm4addsu_i16x4(int16x4_t rs1, uint16x4_t rs2) {
return __riscv_pm4addsu_i16x4(rs1, rs2);
}
+// Packed Widening Multiply Accumulate.
+// CHECK-LABEL: test_pwmacc_i32x2:
+// RV32: pwmacc.h
+// RV64: zip16p
+// RV64: pmacc.w.h01
+int32x2_t test_pwmacc_i32x2(int32x2_t rd, int16x2_t rs1, int16x2_t rs2) {
+ return __riscv_pwmacc_i32x2(rd, rs1, rs2);
+}
+
+// CHECK-LABEL: test_pwmaccu_u32x2:
+// RV32: pwmaccu.h
+// RV64: zip16p
+// RV64: pmaccu.w.h01
+uint32x2_t test_pwmaccu_u32x2(uint32x2_t rd, uint16x2_t rs1, uint16x2_t rs2) {
+ return __riscv_pwmaccu_u32x2(rd, rs1, rs2);
+}
+
+// CHECK-LABEL: test_pwmaccsu_i32x2:
+// RV32: pwmaccsu.h
+// RV64: pwcvtu.wh
+// RV64: pmaccsu.w.h00
+int32x2_t test_pwmaccsu_i32x2(int32x2_t rd, int16x2_t rs1, uint16x2_t rs2) {
+ return __riscv_pwmaccsu_i32x2(rd, rs1, rs2);
+}
+
// Packed Multiply Parts.
// CHECK-LABEL: test_pmul_b00_i16x2:
// RV32: pmul.h.b00
diff --git a/llvm/include/llvm/IR/IntrinsicsRISCV.td b/llvm/include/llvm/IR/IntrinsicsRISCV.td
index 0ed8a3fce15c0..b589ea12375df 100644
--- a/llvm/include/llvm/IR/IntrinsicsRISCV.td
+++ b/llvm/include/llvm/IR/IntrinsicsRISCV.td
@@ -2228,6 +2228,19 @@ class RVPBinaryIntrinsic
def int_riscv_pmaccsu_00 : RVPPackedMulPartsAccIntrinsic;
def int_riscv_pmaccsu_11 : RVPPackedMulPartsAccIntrinsic;
+ // Packed Widening Multiply Accumulate.
+ // The pwmacc*/pwmaccsu* instructions are available on RV32.
+ // On RV64, these operations are synthesized using the instruction
+ // sequences specified by the P extension.
+ class RVPWideningMulAccIntrinsic
+ : DefaultAttrsIntrinsic<[llvm_anyvector_ty],
+ [LLVMMatchType<0>, llvm_anyvector_ty,
+ LLVMMatchType<1>],
+ [IntrNoMem, IntrSpeculatable]>;
+ def int_riscv_pwmacc_i32x2 : RVPWideningMulAccIntrinsic;
+ def int_riscv_pwmaccu_u32x2 : RVPWideningMulAccIntrinsic;
+ def int_riscv_pwmaccsu_i32x2 : RVPWideningMulAccIntrinsic;
+
class RVPScalarMulPartsAccIntrinsic
: DefaultAttrsIntrinsic<[llvm_anyint_ty],
[LLVMMatchType<0>, llvm_anyvector_ty,
diff --git a/llvm/lib/Target/RISCV/RISCVISelDAGToDAG.cpp b/llvm/lib/Target/RISCV/RISCVISelDAGToDAG.cpp
index 146aef612f387..3c2b8ee071c95 100644
--- a/llvm/lib/Target/RISCV/RISCVISelDAGToDAG.cpp
+++ b/llvm/lib/Target/RISCV/RISCVISelDAGToDAG.cpp
@@ -2149,6 +2149,9 @@ void RISCVDAGToDAGISel::Select(SDNode *Node) {
}
case RISCVISD::MQWACC:
case RISCVISD::MQRWACC:
+ case RISCVISD::PWMACC_H:
+ case RISCVISD::PWMACCU_H:
+ case RISCVISD::PWMACCSU_H:
case RISCVISD::WMACC:
case RISCVISD::WMACCU:
case RISCVISD::WMACCSU: {
@@ -2167,6 +2170,15 @@ void RISCVDAGToDAGISel::Select(SDNode *Node) {
case RISCVISD::MQRWACC:
Opc = RISCV::MQRWACC;
break;
+ case RISCVISD::PWMACC_H:
+ Opc = RISCV::PWMACC_H;
+ break;
+ case RISCVISD::PWMACCU_H:
+ Opc = RISCV::PWMACCU_H;
+ break;
+ case RISCVISD::PWMACCSU_H:
+ Opc = RISCV::PWMACCSU_H;
+ break;
case RISCVISD::WMACC:
Opc = RISCV::WMACC;
break;
diff --git a/llvm/lib/Target/RISCV/RISCVISelLowering.cpp b/llvm/lib/Target/RISCV/RISCVISelLowering.cpp
index 7a5905fc4b3ea..0115e122703a2 100644
--- a/llvm/lib/Target/RISCV/RISCVISelLowering.cpp
+++ b/llvm/lib/Target/RISCV/RISCVISelLowering.cpp
@@ -12524,6 +12524,19 @@ static unsigned getRVPHorizontalMulOpcode(unsigned IntNo) {
}
}
+static unsigned getRVPWideningMulAccPairOpcode(unsigned IntNo) {
+ switch (IntNo) {
+ default:
+ llvm_unreachable("Unexpected RISC-V packed pwmacc intrinsic");
+ case Intrinsic::riscv_pwmacc_i32x2:
+ return RISCVISD::PWMACC_H;
+ case Intrinsic::riscv_pwmaccu_u32x2:
+ return RISCVISD::PWMACCU_H;
+ case Intrinsic::riscv_pwmaccsu_i32x2:
+ return RISCVISD::PWMACCSU_H;
+ }
+}
+
static SDValue lowerRV32HorizontalMul64(unsigned IntNo, SDValue Rs1,
SDValue Rs2, const SDLoc &DL,
SelectionDAG &DAG) {
@@ -13386,6 +13399,51 @@ SDValue RISCVTargetLowering::LowerINTRINSIC_WO_CHAIN(SDValue Op,
return DAG.getNode(Opc, DL, VT, Rs1, Rs2);
}
+ case Intrinsic::riscv_pwmacc_i32x2:
+ case Intrinsic::riscv_pwmaccu_u32x2:
+ case Intrinsic::riscv_pwmaccsu_i32x2: {
+ EVT VT = Op.getValueType();
+ SDValue Acc = Op.getOperand(1);
+ SDValue Rs1 = Op.getOperand(2);
+ SDValue Rs2 = Op.getOperand(3);
+ MVT XLenVT = Subtarget.getXLenVT();
+
+ if (Subtarget.is64Bit()) {
+ // Per the spec decomposition table: zip16p duplicates each halfword
+ // into both positions of its 32-bit lane (the su form uses the
+ // pwcvtu.wh spelling, zip16p with x0), then pmacc.w.h01 or
+ // pmaccsu.w.h00 multiplies the lane-local halfwords into rd. The
+ // upper halfwords of the operands are not read. RV32 selects the
+ // pwmacc.h family on the register pair instead.
+ Rs1 = DAG.getNode(ISD::CONCAT_VECTORS, DL, MVT::v4i16, Rs1,
+ DAG.getUNDEF(MVT::v2i16));
+ Rs2 = DAG.getNode(ISD::CONCAT_VECTORS, DL, MVT::v4i16, Rs2,
+ DAG.getUNDEF(MVT::v2i16));
+ if (IntNo == Intrinsic::riscv_pwmaccsu_i32x2) {
+ SDValue Zero = DAG.getNode(ISD::SPLAT_VECTOR, DL, MVT::v4i16,
+ DAG.getConstant(0, DL, XLenVT));
+ Rs1 = DAG.getNode(RISCVISD::ZIP16P, DL, MVT::v4i16, Rs1, Zero);
+ Rs2 = DAG.getNode(RISCVISD::ZIP16P, DL, MVT::v4i16, Rs2, Zero);
+ return DAG.getNode(RISCVISD::PMACCSU_HALVES_00, DL, VT, Acc, Rs1, Rs2);
+ }
+ Rs1 = DAG.getNode(RISCVISD::ZIP16P, DL, MVT::v4i16, Rs1, Rs1);
+ Rs2 = DAG.getNode(RISCVISD::ZIP16P, DL, MVT::v4i16, Rs2, Rs2);
+ unsigned Opc = IntNo == Intrinsic::riscv_pwmaccu_u32x2
+ ? RISCVISD::PMACCU_HALVES_01
+ : RISCVISD::PMACC_HALVES_01;
+ return DAG.getNode(Opc, DL, VT, Acc, Rs1, Rs2);
+ }
+
+ // RV32 uses pwmacc.h/pwmaccsu.h on the 64-bit accumulator register
+ // pair, exposed as a two-result node over the accumulator words.
+ SDValue RdLo = DAG.getExtractVectorElt(DL, XLenVT, Acc, 0);
+ SDValue RdHi = DAG.getExtractVectorElt(DL, XLenVT, Acc, 1);
+ SDVTList VTs = DAG.getVTList(XLenVT, XLenVT);
+ SDValue Res = DAG.getNode(getRVPWideningMulAccPairOpcode(IntNo), DL, VTs,
+ {RdLo, RdHi, Rs1, Rs2});
+ return DAG.getNode(ISD::BUILD_VECTOR, DL, VT, Res.getValue(0),
+ Res.getValue(1));
+ }
case Intrinsic::riscv_pmerge: {
EVT VT = Op.getValueType();
auto buildMerge = [&](SDValue Rs1, SDValue Rs2, SDValue Mask,
diff --git a/llvm/lib/Target/RISCV/RISCVInstrInfoP.td b/llvm/lib/Target/RISCV/RISCVInstrInfoP.td
index 4d2c8d9be1073..e0f021fd652f2 100644
--- a/llvm/lib/Target/RISCV/RISCVInstrInfoP.td
+++ b/llvm/lib/Target/RISCV/RISCVInstrInfoP.td
@@ -1977,6 +1977,20 @@ def riscv_pmaccsu_halves_00
def riscv_pmaccsu_halves_11
: RVSDNode<"PMACCSU_HALVES_11", SDT_RISCVWideningMulAccByHalves>;
+// The pwmacc*/pwmaccsu* instructions use a 64-bit accumulator on RV32.
+// The accumulator and result are represented as two i32 values in the
+// SelectionDAG; custom ISel combines and splits the register pair.
+def SDT_RISCVPackedWideningMulAccPair
+ : SDTypeProfile<2, 4, [SDTCisVT<0, i32>, SDTCisVT<1, i32>,
+ SDTCisSameAs<0, 2>, SDTCisSameAs<1, 3>,
+ SDTCisVec<4>, SDTCisSameAs<4, 5>]>;
+def riscv_pwmacc_h : RVSDNode<"PWMACC_H",
+ SDT_RISCVPackedWideningMulAccPair>;
+def riscv_pwmaccu_h : RVSDNode<"PWMACCU_H",
+ SDT_RISCVPackedWideningMulAccPair>;
+def riscv_pwmaccsu_h : RVSDNode<"PWMACCSU_H",
+ SDT_RISCVPackedWideningMulAccPair>;
+
def SDT_RISCVWideningShiftLeft : SDTypeProfile<2, 2, [SDTCisVT<0, i32>,
SDTCisSameAs<0, 1>,
SDTCisSameAs<0, 2>,
@@ -2011,6 +2025,7 @@ def SDT_RISCVPackedBinary : SDTypeProfile<1, 2, [SDTCisVec<0>,
SDTCisSameAs<0, 1>,
SDTCisSameAs<0, 2>]>;
def riscv_ppairoe_h : RVSDNode<"PPAIROE_H", SDT_RISCVPackedBinary>;
+def riscv_zip16p : RVSDNode<"ZIP16P", SDT_RISCVPackedBinary>;
// Averaging subtraction, (a - b) >> 2
def riscv_asub : RVSDNode<"ASUB", SDTIntBinOp>;
@@ -3228,6 +3243,9 @@ let append Predicates = [IsRV64] in {
def : PatMulPartsAcc<riscv_pmaccsu_halves_00, PMACCSU_W_H00, v2i32, v4i16>;
def : PatMulPartsAcc<riscv_pmaccsu_halves_11, PMACCSU_W_H11, v2i32, v4i16>;
+ def : Pat<(v4i16 (riscv_zip16p (v4i16 GPR:$rs1), (v4i16 GPR:$rs2))),
+ (ZIP16P GPR:$rs1, GPR:$rs2)>;
+
// Scalar word multiply-parts accumulate patterns.
def : PatMulPartsAcc<riscv_pmacc_halves_00, MACC_W00, i64, v2i32>;
def : PatMulPartsAcc<riscv_pmacc_halves_01, MACC_W01, i64, v2i32>;
diff --git a/llvm/test/CodeGen/RISCV/rvp-simd-64.ll b/llvm/test/CodeGen/RISCV/rvp-simd-64.ll
index 34a00069a5597..fdfc97042ddc4 100644
--- a/llvm/test/CodeGen/RISCV/rvp-simd-64.ll
+++ b/llvm/test/CodeGen/RISCV/rvp-simd-64.ll
@@ -8004,6 +8004,55 @@ define <2 x i32> @test_pmaccsu_h11_v2i32(<2 x i32> %rd, <4 x i16> %a, <4 x i16>
ret <2 x i32> %r
}
+; Packed Widening Multiply Accumulate
+define <2 x i32> @test_pwmacc_i32x2(<2 x i32> %rd, <2 x i16> %rs1, <2 x i16> %rs2) {
+; RV32-LABEL: test_pwmacc_i32x2:
+; RV32: # %bb.0:
+; RV32-NEXT: pwmacc.h a0, a2, a3
+; RV32-NEXT: ret
+;
+; RV64-LABEL: test_pwmacc_i32x2:
+; RV64: # %bb.0:
+; RV64-NEXT: zip16p a2, a2, a2
+; RV64-NEXT: zip16p a1, a1, a1
+; RV64-NEXT: pmacc.w.h01 a0, a1, a2
+; RV64-NEXT: ret
+ %res = call <2 x i32> @llvm.riscv.pwmacc.i32x2.v2i32.v2i16(<2 x i32> %rd, <2 x i16> %rs1, <2 x i16> %rs2)
+ ret <2 x i32> %res
+}
+
+define <2 x i32> @test_pwmaccu_u32x2(<2 x i32> %rd, <2 x i16> %rs1, <2 x i16> %rs2) {
+; RV32-LABEL: test_pwmaccu_u32x2:
+; RV32: # %bb.0:
+; RV32-NEXT: pwmaccu.h a0, a2, a3
+; RV32-NEXT: ret
+;
+; RV64-LABEL: test_pwmaccu_u32x2:
+; RV64: # %bb.0:
+; RV64-NEXT: zip16p a2, a2, a2
+; RV64-NEXT: zip16p a1, a1, a1
+; RV64-NEXT: pmaccu.w.h01 a0, a1, a2
+; RV64-NEXT: ret
+ %res = call <2 x i32> @llvm.riscv.pwmaccu.u32x2.v2i32.v2i16(<2 x i32> %rd, <2 x i16> %rs1, <2 x i16> %rs2)
+ ret <2 x i32> %res
+}
+
+define <2 x i32> @test_pwmaccsu_i32x2(<2 x i32> %rd, <2 x i16> %rs1, <2 x i16> %rs2) {
+; RV32-LABEL: test_pwmaccsu_i32x2:
+; RV32: # %bb.0:
+; RV32-NEXT: pwmaccsu.h a0, a2, a3
+; RV32-NEXT: ret
+;
+; RV64-LABEL: test_pwmaccsu_i32x2:
+; RV64: # %bb.0:
+; RV64-NEXT: pwcvtu.wh a2, a2
+; RV64-NEXT: pwcvtu.wh a1, a1
+; RV64-NEXT: pmaccsu.w.h00 a0, a1, a2
+; RV64-NEXT: ret
+ %res = call <2 x i32> @llvm.riscv.pwmaccsu.i32x2.v2i32.v2i16(<2 x i32> %rd, <2 x i16> %rs1, <2 x i16> %rs2)
+ ret <2 x i32> %res
+}
+
define i64 @test_macc_w00_i64(i64 %rd, <2 x i32> %a, <2 x i32> %b) {
; RV32-LABEL: test_macc_w00_i64:
; RV32: # %bb.0:
>From d50f66d9c9fa250854f5bd5cf6dfa52fe46a072d Mon Sep 17 00:00:00 2001
From: "ZhiQiang.Fan" <571253675 at QQ.com>
Date: Mon, 28 Sep 2026 18:56:56 +0800
Subject: [PATCH 2/2] Resolve review: rv64 pwmacc
---
llvm/lib/Target/RISCV/RISCVISelDAGToDAG.cpp | 12 --------
llvm/lib/Target/RISCV/RISCVISelLowering.cpp | 32 +++++++++------------
llvm/lib/Target/RISCV/RISCVInstrInfoP.td | 25 +++++++++-------
llvm/test/CodeGen/RISCV/rvp-simd-64.ll | 10 +++----
4 files changed, 32 insertions(+), 47 deletions(-)
diff --git a/llvm/lib/Target/RISCV/RISCVISelDAGToDAG.cpp b/llvm/lib/Target/RISCV/RISCVISelDAGToDAG.cpp
index 3c2b8ee071c95..146aef612f387 100644
--- a/llvm/lib/Target/RISCV/RISCVISelDAGToDAG.cpp
+++ b/llvm/lib/Target/RISCV/RISCVISelDAGToDAG.cpp
@@ -2149,9 +2149,6 @@ void RISCVDAGToDAGISel::Select(SDNode *Node) {
}
case RISCVISD::MQWACC:
case RISCVISD::MQRWACC:
- case RISCVISD::PWMACC_H:
- case RISCVISD::PWMACCU_H:
- case RISCVISD::PWMACCSU_H:
case RISCVISD::WMACC:
case RISCVISD::WMACCU:
case RISCVISD::WMACCSU: {
@@ -2170,15 +2167,6 @@ void RISCVDAGToDAGISel::Select(SDNode *Node) {
case RISCVISD::MQRWACC:
Opc = RISCV::MQRWACC;
break;
- case RISCVISD::PWMACC_H:
- Opc = RISCV::PWMACC_H;
- break;
- case RISCVISD::PWMACCU_H:
- Opc = RISCV::PWMACCU_H;
- break;
- case RISCVISD::PWMACCSU_H:
- Opc = RISCV::PWMACCSU_H;
- break;
case RISCVISD::WMACC:
Opc = RISCV::WMACC;
break;
diff --git a/llvm/lib/Target/RISCV/RISCVISelLowering.cpp b/llvm/lib/Target/RISCV/RISCVISelLowering.cpp
index 0115e122703a2..f75186053c415 100644
--- a/llvm/lib/Target/RISCV/RISCVISelLowering.cpp
+++ b/llvm/lib/Target/RISCV/RISCVISelLowering.cpp
@@ -12524,7 +12524,7 @@ static unsigned getRVPHorizontalMulOpcode(unsigned IntNo) {
}
}
-static unsigned getRVPWideningMulAccPairOpcode(unsigned IntNo) {
+static unsigned getRVPWideningMulAccOpcode(unsigned IntNo) {
switch (IntNo) {
default:
llvm_unreachable("Unexpected RISC-V packed pwmacc intrinsic");
@@ -13409,11 +13409,11 @@ SDValue RISCVTargetLowering::LowerINTRINSIC_WO_CHAIN(SDValue Op,
MVT XLenVT = Subtarget.getXLenVT();
if (Subtarget.is64Bit()) {
- // Per the spec decomposition table: zip16p duplicates each halfword
- // into both positions of its 32-bit lane (the su form uses the
- // pwcvtu.wh spelling, zip16p with x0), then pmacc.w.h01 or
- // pmaccsu.w.h00 multiplies the lane-local halfwords into rd. The
- // upper halfwords of the operands are not read. RV32 selects the
+ // Per the spec decomposition table: a single zip16p puts the
+ // halfwords of Rs1 in the even elements and those of Rs2 in the odd
+ // elements (the su form uses the pwcvtu.wh spelling, zip16p with x0).
+ // pmacc.w.h01 then multiplies the even elements of the first source
+ // by the odd elements of the second source into rd. RV32 selects the
// pwmacc.h family on the register pair instead.
Rs1 = DAG.getNode(ISD::CONCAT_VECTORS, DL, MVT::v4i16, Rs1,
DAG.getUNDEF(MVT::v2i16));
@@ -13422,27 +13422,21 @@ SDValue RISCVTargetLowering::LowerINTRINSIC_WO_CHAIN(SDValue Op,
if (IntNo == Intrinsic::riscv_pwmaccsu_i32x2) {
SDValue Zero = DAG.getNode(ISD::SPLAT_VECTOR, DL, MVT::v4i16,
DAG.getConstant(0, DL, XLenVT));
- Rs1 = DAG.getNode(RISCVISD::ZIP16P, DL, MVT::v4i16, Rs1, Zero);
- Rs2 = DAG.getNode(RISCVISD::ZIP16P, DL, MVT::v4i16, Rs2, Zero);
+ Rs1 = DAG.getNode(RISCVISD::PZIP, DL, MVT::v4i16, Rs1, Zero);
+ Rs2 = DAG.getNode(RISCVISD::PZIP, DL, MVT::v4i16, Rs2, Zero);
return DAG.getNode(RISCVISD::PMACCSU_HALVES_00, DL, VT, Acc, Rs1, Rs2);
}
- Rs1 = DAG.getNode(RISCVISD::ZIP16P, DL, MVT::v4i16, Rs1, Rs1);
- Rs2 = DAG.getNode(RISCVISD::ZIP16P, DL, MVT::v4i16, Rs2, Rs2);
+ SDValue Zip = DAG.getNode(RISCVISD::PZIP, DL, MVT::v4i16, Rs1, Rs2);
unsigned Opc = IntNo == Intrinsic::riscv_pwmaccu_u32x2
? RISCVISD::PMACCU_HALVES_01
: RISCVISD::PMACC_HALVES_01;
- return DAG.getNode(Opc, DL, VT, Acc, Rs1, Rs2);
+ return DAG.getNode(Opc, DL, VT, Acc, Zip, Zip);
}
// RV32 uses pwmacc.h/pwmaccsu.h on the 64-bit accumulator register
- // pair, exposed as a two-result node over the accumulator words.
- SDValue RdLo = DAG.getExtractVectorElt(DL, XLenVT, Acc, 0);
- SDValue RdHi = DAG.getExtractVectorElt(DL, XLenVT, Acc, 1);
- SDVTList VTs = DAG.getVTList(XLenVT, XLenVT);
- SDValue Res = DAG.getNode(getRVPWideningMulAccPairOpcode(IntNo), DL, VTs,
- {RdLo, RdHi, Rs1, Rs2});
- return DAG.getNode(ISD::BUILD_VECTOR, DL, VT, Res.getValue(0),
- Res.getValue(1));
+ // pair, represented as a single v2i32 value.
+ return DAG.getNode(getRVPWideningMulAccOpcode(IntNo), DL, VT, Acc, Rs1,
+ Rs2);
}
case Intrinsic::riscv_pmerge: {
EVT VT = Op.getValueType();
diff --git a/llvm/lib/Target/RISCV/RISCVInstrInfoP.td b/llvm/lib/Target/RISCV/RISCVInstrInfoP.td
index e0f021fd652f2..cffd39fea701f 100644
--- a/llvm/lib/Target/RISCV/RISCVInstrInfoP.td
+++ b/llvm/lib/Target/RISCV/RISCVInstrInfoP.td
@@ -1977,13 +1977,11 @@ def riscv_pmaccsu_halves_00
def riscv_pmaccsu_halves_11
: RVSDNode<"PMACCSU_HALVES_11", SDT_RISCVWideningMulAccByHalves>;
-// The pwmacc*/pwmaccsu* instructions use a 64-bit accumulator on RV32.
-// The accumulator and result are represented as two i32 values in the
-// SelectionDAG; custom ISel combines and splits the register pair.
+// The pwmacc*/pwmaccsu* instructions accumulate into the 64-bit register
+// pair holding rd, represented as a v2i32 on RV32.
def SDT_RISCVPackedWideningMulAccPair
- : SDTypeProfile<2, 4, [SDTCisVT<0, i32>, SDTCisVT<1, i32>,
- SDTCisSameAs<0, 2>, SDTCisSameAs<1, 3>,
- SDTCisVec<4>, SDTCisSameAs<4, 5>]>;
+ : SDTypeProfile<1, 3, [SDTCisVT<0, v2i32>, SDTCisSameAs<0, 1>,
+ SDTCisVT<2, v2i16>, SDTCisSameAs<2, 3>]>;
def riscv_pwmacc_h : RVSDNode<"PWMACC_H",
SDT_RISCVPackedWideningMulAccPair>;
def riscv_pwmaccu_h : RVSDNode<"PWMACCU_H",
@@ -2025,7 +2023,6 @@ def SDT_RISCVPackedBinary : SDTypeProfile<1, 2, [SDTCisVec<0>,
SDTCisSameAs<0, 1>,
SDTCisSameAs<0, 2>]>;
def riscv_ppairoe_h : RVSDNode<"PPAIROE_H", SDT_RISCVPackedBinary>;
-def riscv_zip16p : RVSDNode<"ZIP16P", SDT_RISCVPackedBinary>;
// Averaging subtraction, (a - b) >> 2
def riscv_asub : RVSDNode<"ASUB", SDTIntBinOp>;
@@ -2704,6 +2701,17 @@ let append Predicates = [IsRV32] in {
(v2i16 GPR:$rs1), (v2i16 GPR:$rs2))),
(PM2WADDASU_H GPRPair:$rd, GPR:$rs1, GPR:$rs2)>;
+ // Packed widening multiply accumulate patterns.
+ def : Pat<(v2i32 (riscv_pwmacc_h (v2i32 GPRPair:$rd),
+ (v2i16 GPR:$rs1), (v2i16 GPR:$rs2))),
+ (PWMACC_H GPRPair:$rd, GPR:$rs1, GPR:$rs2)>;
+ def : Pat<(v2i32 (riscv_pwmaccu_h (v2i32 GPRPair:$rd),
+ (v2i16 GPR:$rs1), (v2i16 GPR:$rs2))),
+ (PWMACCU_H GPRPair:$rd, GPR:$rs1, GPR:$rs2)>;
+ def : Pat<(v2i32 (riscv_pwmaccsu_h (v2i32 GPRPair:$rd),
+ (v2i16 GPR:$rs1), (v2i16 GPR:$rs2))),
+ (PWMACCSU_H GPRPair:$rd, GPR:$rs1, GPR:$rs2)>;
+
// 8/16-bit bitreverse patterns
// With Zbkb, brev8 reverses the bits within each byte directly; otherwise
// reverse all bits then swap the bytes back.
@@ -3243,9 +3251,6 @@ let append Predicates = [IsRV64] in {
def : PatMulPartsAcc<riscv_pmaccsu_halves_00, PMACCSU_W_H00, v2i32, v4i16>;
def : PatMulPartsAcc<riscv_pmaccsu_halves_11, PMACCSU_W_H11, v2i32, v4i16>;
- def : Pat<(v4i16 (riscv_zip16p (v4i16 GPR:$rs1), (v4i16 GPR:$rs2))),
- (ZIP16P GPR:$rs1, GPR:$rs2)>;
-
// Scalar word multiply-parts accumulate patterns.
def : PatMulPartsAcc<riscv_pmacc_halves_00, MACC_W00, i64, v2i32>;
def : PatMulPartsAcc<riscv_pmacc_halves_01, MACC_W01, i64, v2i32>;
diff --git a/llvm/test/CodeGen/RISCV/rvp-simd-64.ll b/llvm/test/CodeGen/RISCV/rvp-simd-64.ll
index fdfc97042ddc4..455b745f92bcc 100644
--- a/llvm/test/CodeGen/RISCV/rvp-simd-64.ll
+++ b/llvm/test/CodeGen/RISCV/rvp-simd-64.ll
@@ -8013,9 +8013,8 @@ define <2 x i32> @test_pwmacc_i32x2(<2 x i32> %rd, <2 x i16> %rs1, <2 x i16> %rs
;
; RV64-LABEL: test_pwmacc_i32x2:
; RV64: # %bb.0:
-; RV64-NEXT: zip16p a2, a2, a2
-; RV64-NEXT: zip16p a1, a1, a1
-; RV64-NEXT: pmacc.w.h01 a0, a1, a2
+; RV64-NEXT: zip16p a1, a1, a2
+; RV64-NEXT: pmacc.w.h01 a0, a1, a1
; RV64-NEXT: ret
%res = call <2 x i32> @llvm.riscv.pwmacc.i32x2.v2i32.v2i16(<2 x i32> %rd, <2 x i16> %rs1, <2 x i16> %rs2)
ret <2 x i32> %res
@@ -8029,9 +8028,8 @@ define <2 x i32> @test_pwmaccu_u32x2(<2 x i32> %rd, <2 x i16> %rs1, <2 x i16> %r
;
; RV64-LABEL: test_pwmaccu_u32x2:
; RV64: # %bb.0:
-; RV64-NEXT: zip16p a2, a2, a2
-; RV64-NEXT: zip16p a1, a1, a1
-; RV64-NEXT: pmaccu.w.h01 a0, a1, a2
+; RV64-NEXT: zip16p a1, a1, a2
+; RV64-NEXT: pmaccu.w.h01 a0, a1, a1
; RV64-NEXT: ret
%res = call <2 x i32> @llvm.riscv.pwmaccu.u32x2.v2i32.v2i16(<2 x i32> %rd, <2 x i16> %rs1, <2 x i16> %rs2)
ret <2 x i32> %res
More information about the llvm-commits
mailing list