[clang] [llvm] [Clang][RISCV][P-ext] Add Packed Widening Multiply Accumulate intrinsics (PR #226505)

via llvm-commits llvm-commits at lists.llvm.org
Mon Sep 28 04:00:45 PDT 2026


https://github.com/StarryCSF updated https://github.com/llvm/llvm-project/pull/226505

>From bdcb7e54ff5362b6ddb5dd024cd6fd740f22bddc Mon Sep 17 00:00:00 2001
From: "ZhiQiang.Fan" <571253675 at QQ.com>
Date: Fri, 25 Sep 2026 22:05:06 +0800
Subject: [PATCH 1/2] [RISCV][P-ext] Add Packed Widening Multiply Accumulate
 intrinsics

RV32 selects the pwmacc.h family, which accumulates into a register
pair; RV64 follows the spec's RV64 decomposition sequences.
---
 clang/include/clang/Basic/BuiltinsRISCV.td    |  5 ++
 clang/lib/CodeGen/TargetBuiltins/RISCV.cpp    | 21 ++++++
 clang/lib/Headers/riscv_packed_simd.h         |  5 ++
 clang/test/CodeGen/RISCV/rvp-intrinsics.c     | 73 +++++++++++++++++++
 .../riscv_packed_simd.c                       | 25 +++++++
 llvm/include/llvm/IR/IntrinsicsRISCV.td       | 13 ++++
 llvm/lib/Target/RISCV/RISCVISelDAGToDAG.cpp   | 12 +++
 llvm/lib/Target/RISCV/RISCVISelLowering.cpp   | 58 +++++++++++++++
 llvm/lib/Target/RISCV/RISCVInstrInfoP.td      | 18 +++++
 llvm/test/CodeGen/RISCV/rvp-simd-64.ll        | 49 +++++++++++++
 10 files changed, 279 insertions(+)

diff --git a/clang/include/clang/Basic/BuiltinsRISCV.td b/clang/include/clang/Basic/BuiltinsRISCV.td
index 4cc4cc2169f01..b5275015fad8f 100644
--- a/clang/include/clang/Basic/BuiltinsRISCV.td
+++ b/clang/include/clang/Basic/BuiltinsRISCV.td
@@ -327,6 +327,11 @@ def pm4add_i16x4 : RISCVBuiltin<"int64_t(_Vector<4, short>, _Vector<4, short>)">
 def pm4addu_u16x4 : RISCVBuiltin<"uint64_t(_Vector<4, unsigned short>, _Vector<4, unsigned short>)">;
 def pm4addsu_i16x4 : RISCVBuiltin<"int64_t(_Vector<4, short>, _Vector<4, unsigned short>)">;
 
+// Packed Widening Multiply Accumulate
+def pwmacc_i32x2 : RISCVBuiltin<"_Vector<2, int>(_Vector<2, int>, _Vector<2, short>, _Vector<2, short>)">;
+def pwmaccu_u32x2 : RISCVBuiltin<"_Vector<2, unsigned int>(_Vector<2, unsigned int>, _Vector<2, unsigned short>, _Vector<2, unsigned short>)">;
+def pwmaccsu_i32x2 : RISCVBuiltin<"_Vector<2, int>(_Vector<2, int>, _Vector<2, short>, _Vector<2, unsigned short>)">;
+
 // Packed Absolute Difference Sum (32-bit)
 def pabdsumu_u8x4_u32 : RISCVBuiltin<"unsigned int(_Vector<4, unsigned char>, _Vector<4, unsigned char>)">;
 def pabdsumau_u8x4_u32 : RISCVBuiltin<"unsigned int(unsigned int, _Vector<4, unsigned char>, _Vector<4, unsigned char>)">;
diff --git a/clang/lib/CodeGen/TargetBuiltins/RISCV.cpp b/clang/lib/CodeGen/TargetBuiltins/RISCV.cpp
index dc44fd06e035a..3b804c8f2b571 100644
--- a/clang/lib/CodeGen/TargetBuiltins/RISCV.cpp
+++ b/clang/lib/CodeGen/TargetBuiltins/RISCV.cpp
@@ -1648,6 +1648,27 @@ Value *CodeGenFunction::EmitRISCVBuiltinExpr(unsigned BuiltinID,
     break;
   }
 
+  // Packed Widening Multiply Accumulate
+  case RISCV::BI__builtin_riscv_pwmacc_i32x2:
+  case RISCV::BI__builtin_riscv_pwmaccu_u32x2:
+  case RISCV::BI__builtin_riscv_pwmaccsu_i32x2: {
+    switch (BuiltinID) {
+    default:
+      llvm_unreachable("unexpected builtin ID");
+    case RISCV::BI__builtin_riscv_pwmacc_i32x2:
+      ID = Intrinsic::riscv_pwmacc_i32x2;
+      break;
+    case RISCV::BI__builtin_riscv_pwmaccu_u32x2:
+      ID = Intrinsic::riscv_pwmaccu_u32x2;
+      break;
+    case RISCV::BI__builtin_riscv_pwmaccsu_i32x2:
+      ID = Intrinsic::riscv_pwmaccsu_i32x2;
+      break;
+    }
+    IntrinsicTypes = {ResultType, Ops[1]->getType()};
+    break;
+  }
+
   // Packed Reduction Sum
   case RISCV::BI__builtin_riscv_predsum_i8x4_i32:
   case RISCV::BI__builtin_riscv_predsum_i16x2_i32:
diff --git a/clang/lib/Headers/riscv_packed_simd.h b/clang/lib/Headers/riscv_packed_simd.h
index 99a6f7b1d6e9f..28c661612d89b 100644
--- a/clang/lib/Headers/riscv_packed_simd.h
+++ b/clang/lib/Headers/riscv_packed_simd.h
@@ -1021,6 +1021,11 @@ __packed_binary_builtin_mixed(pm4add_i16x4, int64_t, int16x4_t, int16x4_t, __bui
 __packed_binary_builtin_mixed(pm4addu_u16x4, uint64_t, uint16x4_t, uint16x4_t, __builtin_riscv_pm4addu_u16x4)
 __packed_binary_builtin_mixed(pm4addsu_i16x4, int64_t, int16x4_t, uint16x4_t, __builtin_riscv_pm4addsu_i16x4)
 
+/* Packed Widening Multiply Accumulate */
+__packed_ternary_builtin_mixed(pwmacc_i32x2, int32x2_t, int16x2_t, int16x2_t, __builtin_riscv_pwmacc_i32x2)
+__packed_ternary_builtin_mixed(pwmaccu_u32x2, uint32x2_t, uint16x2_t, uint16x2_t, __builtin_riscv_pwmaccu_u32x2)
+__packed_ternary_builtin_mixed(pwmaccsu_i32x2, int32x2_t, int16x2_t, uint16x2_t, __builtin_riscv_pwmaccsu_i32x2)
+
 /* Packed Absolute Difference Sum (32-bit) */
 __packed_abdsum(pabdsumu_u8x4_u32, uint32_t, uint8x4_t, __builtin_riscv_pabdsumu_u8x4_u32)
 __packed_ternary_builtin_cast(pabdsumau_u8x4_u32, uint32_t, uint8x4_t, __builtin_riscv_pabdsumau_u8x4_u32)
diff --git a/clang/test/CodeGen/RISCV/rvp-intrinsics.c b/clang/test/CodeGen/RISCV/rvp-intrinsics.c
index 1184683f1e333..bdcf9ae0a4360 100644
--- a/clang/test/CodeGen/RISCV/rvp-intrinsics.c
+++ b/clang/test/CodeGen/RISCV/rvp-intrinsics.c
@@ -13116,3 +13116,76 @@ int16x4_t test_psati_i16x4(int16x4_t a) { return __riscv_psati_i16x4(a, 8); }
 // RV64-NEXT:    ret i64 [[TMP2]]
 //
 int32x2_t test_psati_i32x2(int32x2_t a) { return __riscv_psati_i32x2(a, 16); }
+
+/* Packed Widening Multiply Accumulate */
+// RV32-LABEL: define dso_local i64 @test_pwmacc_i32x2(
+// RV32-SAME: i64 noundef [[RD_COERCE:%.*]], i32 noundef [[RS1_COERCE:%.*]], i32 noundef [[RS2_COERCE:%.*]]) #[[ATTR0]] {
+// RV32-NEXT:  [[ENTRY:.*:]]
+// RV32-NEXT:    [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <2 x i32>
+// RV32-NEXT:    [[TMP1:%.*]] = bitcast i32 [[RS1_COERCE]] to <2 x i16>
+// RV32-NEXT:    [[TMP2:%.*]] = bitcast i32 [[RS2_COERCE]] to <2 x i16>
+// RV32-NEXT:    [[TMP3:%.*]] = call <2 x i32> @llvm.riscv.pwmacc.i32x2.v2i32.v2i16(<2 x i32> [[TMP0]], <2 x i16> [[TMP1]], <2 x i16> [[TMP2]])
+// RV32-NEXT:    [[TMP4:%.*]] = bitcast <2 x i32> [[TMP3]] to i64
+// RV32-NEXT:    ret i64 [[TMP4]]
+//
+// RV64-LABEL: define dso_local i64 @test_pwmacc_i32x2(
+// RV64-SAME: i64 noundef [[RD_COERCE:%.*]], i32 noundef [[RS1_COERCE:%.*]], i32 noundef [[RS2_COERCE:%.*]]) #[[ATTR0]] {
+// RV64-NEXT:  [[ENTRY:.*:]]
+// RV64-NEXT:    [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <2 x i32>
+// RV64-NEXT:    [[TMP1:%.*]] = bitcast i32 [[RS1_COERCE]] to <2 x i16>
+// RV64-NEXT:    [[TMP2:%.*]] = bitcast i32 [[RS2_COERCE]] to <2 x i16>
+// RV64-NEXT:    [[TMP3:%.*]] = call <2 x i32> @llvm.riscv.pwmacc.i32x2.v2i32.v2i16(<2 x i32> [[TMP0]], <2 x i16> [[TMP1]], <2 x i16> [[TMP2]])
+// RV64-NEXT:    [[TMP4:%.*]] = bitcast <2 x i32> [[TMP3]] to i64
+// RV64-NEXT:    ret i64 [[TMP4]]
+//
+int32x2_t test_pwmacc_i32x2(int32x2_t rd, int16x2_t rs1, int16x2_t rs2) {
+  return __riscv_pwmacc_i32x2(rd, rs1, rs2);
+}
+
+// RV32-LABEL: define dso_local i64 @test_pwmaccu_u32x2(
+// RV32-SAME: i64 noundef [[RD_COERCE:%.*]], i32 noundef [[RS1_COERCE:%.*]], i32 noundef [[RS2_COERCE:%.*]]) #[[ATTR0]] {
+// RV32-NEXT:  [[ENTRY:.*:]]
+// RV32-NEXT:    [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <2 x i32>
+// RV32-NEXT:    [[TMP1:%.*]] = bitcast i32 [[RS1_COERCE]] to <2 x i16>
+// RV32-NEXT:    [[TMP2:%.*]] = bitcast i32 [[RS2_COERCE]] to <2 x i16>
+// RV32-NEXT:    [[TMP3:%.*]] = call <2 x i32> @llvm.riscv.pwmaccu.u32x2.v2i32.v2i16(<2 x i32> [[TMP0]], <2 x i16> [[TMP1]], <2 x i16> [[TMP2]])
+// RV32-NEXT:    [[TMP4:%.*]] = bitcast <2 x i32> [[TMP3]] to i64
+// RV32-NEXT:    ret i64 [[TMP4]]
+//
+// RV64-LABEL: define dso_local i64 @test_pwmaccu_u32x2(
+// RV64-SAME: i64 noundef [[RD_COERCE:%.*]], i32 noundef [[RS1_COERCE:%.*]], i32 noundef [[RS2_COERCE:%.*]]) #[[ATTR0]] {
+// RV64-NEXT:  [[ENTRY:.*:]]
+// RV64-NEXT:    [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <2 x i32>
+// RV64-NEXT:    [[TMP1:%.*]] = bitcast i32 [[RS1_COERCE]] to <2 x i16>
+// RV64-NEXT:    [[TMP2:%.*]] = bitcast i32 [[RS2_COERCE]] to <2 x i16>
+// RV64-NEXT:    [[TMP3:%.*]] = call <2 x i32> @llvm.riscv.pwmaccu.u32x2.v2i32.v2i16(<2 x i32> [[TMP0]], <2 x i16> [[TMP1]], <2 x i16> [[TMP2]])
+// RV64-NEXT:    [[TMP4:%.*]] = bitcast <2 x i32> [[TMP3]] to i64
+// RV64-NEXT:    ret i64 [[TMP4]]
+//
+uint32x2_t test_pwmaccu_u32x2(uint32x2_t rd, uint16x2_t rs1, uint16x2_t rs2) {
+  return __riscv_pwmaccu_u32x2(rd, rs1, rs2);
+}
+
+// RV32-LABEL: define dso_local i64 @test_pwmaccsu_i32x2(
+// RV32-SAME: i64 noundef [[RD_COERCE:%.*]], i32 noundef [[RS1_COERCE:%.*]], i32 noundef [[RS2_COERCE:%.*]]) #[[ATTR0]] {
+// RV32-NEXT:  [[ENTRY:.*:]]
+// RV32-NEXT:    [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <2 x i32>
+// RV32-NEXT:    [[TMP1:%.*]] = bitcast i32 [[RS1_COERCE]] to <2 x i16>
+// RV32-NEXT:    [[TMP2:%.*]] = bitcast i32 [[RS2_COERCE]] to <2 x i16>
+// RV32-NEXT:    [[TMP3:%.*]] = call <2 x i32> @llvm.riscv.pwmaccsu.i32x2.v2i32.v2i16(<2 x i32> [[TMP0]], <2 x i16> [[TMP1]], <2 x i16> [[TMP2]])
+// RV32-NEXT:    [[TMP4:%.*]] = bitcast <2 x i32> [[TMP3]] to i64
+// RV32-NEXT:    ret i64 [[TMP4]]
+//
+// RV64-LABEL: define dso_local i64 @test_pwmaccsu_i32x2(
+// RV64-SAME: i64 noundef [[RD_COERCE:%.*]], i32 noundef [[RS1_COERCE:%.*]], i32 noundef [[RS2_COERCE:%.*]]) #[[ATTR0]] {
+// RV64-NEXT:  [[ENTRY:.*:]]
+// RV64-NEXT:    [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <2 x i32>
+// RV64-NEXT:    [[TMP1:%.*]] = bitcast i32 [[RS1_COERCE]] to <2 x i16>
+// RV64-NEXT:    [[TMP2:%.*]] = bitcast i32 [[RS2_COERCE]] to <2 x i16>
+// RV64-NEXT:    [[TMP3:%.*]] = call <2 x i32> @llvm.riscv.pwmaccsu.i32x2.v2i32.v2i16(<2 x i32> [[TMP0]], <2 x i16> [[TMP1]], <2 x i16> [[TMP2]])
+// RV64-NEXT:    [[TMP4:%.*]] = bitcast <2 x i32> [[TMP3]] to i64
+// RV64-NEXT:    ret i64 [[TMP4]]
+//
+int32x2_t test_pwmaccsu_i32x2(int32x2_t rd, int16x2_t rs1, uint16x2_t rs2) {
+  return __riscv_pwmaccsu_i32x2(rd, rs1, rs2);
+}
diff --git a/cross-project-tests/intrinsic-header-tests/riscv_packed_simd.c b/cross-project-tests/intrinsic-header-tests/riscv_packed_simd.c
index 1e8d1c02fa66c..c33d03324ac6f 100644
--- a/cross-project-tests/intrinsic-header-tests/riscv_packed_simd.c
+++ b/cross-project-tests/intrinsic-header-tests/riscv_packed_simd.c
@@ -4051,6 +4051,31 @@ int64_t test_pm4addsu_i16x4(int16x4_t rs1, uint16x4_t rs2) {
   return __riscv_pm4addsu_i16x4(rs1, rs2);
 }
 
+// Packed Widening Multiply Accumulate.
+// CHECK-LABEL: test_pwmacc_i32x2:
+// RV32:        pwmacc.h
+// RV64:        zip16p
+// RV64:        pmacc.w.h01
+int32x2_t test_pwmacc_i32x2(int32x2_t rd, int16x2_t rs1, int16x2_t rs2) {
+  return __riscv_pwmacc_i32x2(rd, rs1, rs2);
+}
+
+// CHECK-LABEL: test_pwmaccu_u32x2:
+// RV32:        pwmaccu.h
+// RV64:        zip16p
+// RV64:        pmaccu.w.h01
+uint32x2_t test_pwmaccu_u32x2(uint32x2_t rd, uint16x2_t rs1, uint16x2_t rs2) {
+  return __riscv_pwmaccu_u32x2(rd, rs1, rs2);
+}
+
+// CHECK-LABEL: test_pwmaccsu_i32x2:
+// RV32:        pwmaccsu.h
+// RV64:        pwcvtu.wh
+// RV64:        pmaccsu.w.h00
+int32x2_t test_pwmaccsu_i32x2(int32x2_t rd, int16x2_t rs1, uint16x2_t rs2) {
+  return __riscv_pwmaccsu_i32x2(rd, rs1, rs2);
+}
+
 // Packed Multiply Parts.
 // CHECK-LABEL: test_pmul_b00_i16x2:
 // RV32:        pmul.h.b00
diff --git a/llvm/include/llvm/IR/IntrinsicsRISCV.td b/llvm/include/llvm/IR/IntrinsicsRISCV.td
index 0ed8a3fce15c0..b589ea12375df 100644
--- a/llvm/include/llvm/IR/IntrinsicsRISCV.td
+++ b/llvm/include/llvm/IR/IntrinsicsRISCV.td
@@ -2228,6 +2228,19 @@ class RVPBinaryIntrinsic
   def int_riscv_pmaccsu_00 : RVPPackedMulPartsAccIntrinsic;
   def int_riscv_pmaccsu_11 : RVPPackedMulPartsAccIntrinsic;
 
+  // Packed Widening Multiply Accumulate.
+  // The pwmacc*/pwmaccsu* instructions are available on RV32.
+  // On RV64, these operations are synthesized using the instruction
+  // sequences specified by the P extension.
+  class RVPWideningMulAccIntrinsic
+      : DefaultAttrsIntrinsic<[llvm_anyvector_ty],
+                              [LLVMMatchType<0>, llvm_anyvector_ty,
+                               LLVMMatchType<1>],
+                              [IntrNoMem, IntrSpeculatable]>;
+  def int_riscv_pwmacc_i32x2   : RVPWideningMulAccIntrinsic;
+  def int_riscv_pwmaccu_u32x2  : RVPWideningMulAccIntrinsic;
+  def int_riscv_pwmaccsu_i32x2 : RVPWideningMulAccIntrinsic;
+
   class RVPScalarMulPartsAccIntrinsic
       : DefaultAttrsIntrinsic<[llvm_anyint_ty],
                               [LLVMMatchType<0>, llvm_anyvector_ty,
diff --git a/llvm/lib/Target/RISCV/RISCVISelDAGToDAG.cpp b/llvm/lib/Target/RISCV/RISCVISelDAGToDAG.cpp
index 146aef612f387..3c2b8ee071c95 100644
--- a/llvm/lib/Target/RISCV/RISCVISelDAGToDAG.cpp
+++ b/llvm/lib/Target/RISCV/RISCVISelDAGToDAG.cpp
@@ -2149,6 +2149,9 @@ void RISCVDAGToDAGISel::Select(SDNode *Node) {
   }
   case RISCVISD::MQWACC:
   case RISCVISD::MQRWACC:
+  case RISCVISD::PWMACC_H:
+  case RISCVISD::PWMACCU_H:
+  case RISCVISD::PWMACCSU_H:
   case RISCVISD::WMACC:
   case RISCVISD::WMACCU:
   case RISCVISD::WMACCSU: {
@@ -2167,6 +2170,15 @@ void RISCVDAGToDAGISel::Select(SDNode *Node) {
     case RISCVISD::MQRWACC:
       Opc = RISCV::MQRWACC;
       break;
+    case RISCVISD::PWMACC_H:
+      Opc = RISCV::PWMACC_H;
+      break;
+    case RISCVISD::PWMACCU_H:
+      Opc = RISCV::PWMACCU_H;
+      break;
+    case RISCVISD::PWMACCSU_H:
+      Opc = RISCV::PWMACCSU_H;
+      break;
     case RISCVISD::WMACC:
       Opc = RISCV::WMACC;
       break;
diff --git a/llvm/lib/Target/RISCV/RISCVISelLowering.cpp b/llvm/lib/Target/RISCV/RISCVISelLowering.cpp
index 7a5905fc4b3ea..0115e122703a2 100644
--- a/llvm/lib/Target/RISCV/RISCVISelLowering.cpp
+++ b/llvm/lib/Target/RISCV/RISCVISelLowering.cpp
@@ -12524,6 +12524,19 @@ static unsigned getRVPHorizontalMulOpcode(unsigned IntNo) {
   }
 }
 
+static unsigned getRVPWideningMulAccPairOpcode(unsigned IntNo) {
+  switch (IntNo) {
+  default:
+    llvm_unreachable("Unexpected RISC-V packed pwmacc intrinsic");
+  case Intrinsic::riscv_pwmacc_i32x2:
+    return RISCVISD::PWMACC_H;
+  case Intrinsic::riscv_pwmaccu_u32x2:
+    return RISCVISD::PWMACCU_H;
+  case Intrinsic::riscv_pwmaccsu_i32x2:
+    return RISCVISD::PWMACCSU_H;
+  }
+}
+
 static SDValue lowerRV32HorizontalMul64(unsigned IntNo, SDValue Rs1,
                                         SDValue Rs2, const SDLoc &DL,
                                         SelectionDAG &DAG) {
@@ -13386,6 +13399,51 @@ SDValue RISCVTargetLowering::LowerINTRINSIC_WO_CHAIN(SDValue Op,
 
     return DAG.getNode(Opc, DL, VT, Rs1, Rs2);
   }
+  case Intrinsic::riscv_pwmacc_i32x2:
+  case Intrinsic::riscv_pwmaccu_u32x2:
+  case Intrinsic::riscv_pwmaccsu_i32x2: {
+    EVT VT = Op.getValueType();
+    SDValue Acc = Op.getOperand(1);
+    SDValue Rs1 = Op.getOperand(2);
+    SDValue Rs2 = Op.getOperand(3);
+    MVT XLenVT = Subtarget.getXLenVT();
+
+    if (Subtarget.is64Bit()) {
+      // Per the spec decomposition table: zip16p duplicates each halfword
+      // into both positions of its 32-bit lane (the su form uses the
+      // pwcvtu.wh spelling, zip16p with x0), then pmacc.w.h01 or
+      // pmaccsu.w.h00 multiplies the lane-local halfwords into rd. The
+      // upper halfwords of the operands are not read. RV32 selects the
+      // pwmacc.h family on the register pair instead.
+      Rs1 = DAG.getNode(ISD::CONCAT_VECTORS, DL, MVT::v4i16, Rs1,
+                        DAG.getUNDEF(MVT::v2i16));
+      Rs2 = DAG.getNode(ISD::CONCAT_VECTORS, DL, MVT::v4i16, Rs2,
+                        DAG.getUNDEF(MVT::v2i16));
+      if (IntNo == Intrinsic::riscv_pwmaccsu_i32x2) {
+        SDValue Zero = DAG.getNode(ISD::SPLAT_VECTOR, DL, MVT::v4i16,
+                                   DAG.getConstant(0, DL, XLenVT));
+        Rs1 = DAG.getNode(RISCVISD::ZIP16P, DL, MVT::v4i16, Rs1, Zero);
+        Rs2 = DAG.getNode(RISCVISD::ZIP16P, DL, MVT::v4i16, Rs2, Zero);
+        return DAG.getNode(RISCVISD::PMACCSU_HALVES_00, DL, VT, Acc, Rs1, Rs2);
+      }
+      Rs1 = DAG.getNode(RISCVISD::ZIP16P, DL, MVT::v4i16, Rs1, Rs1);
+      Rs2 = DAG.getNode(RISCVISD::ZIP16P, DL, MVT::v4i16, Rs2, Rs2);
+      unsigned Opc = IntNo == Intrinsic::riscv_pwmaccu_u32x2
+                         ? RISCVISD::PMACCU_HALVES_01
+                         : RISCVISD::PMACC_HALVES_01;
+      return DAG.getNode(Opc, DL, VT, Acc, Rs1, Rs2);
+    }
+
+    // RV32 uses pwmacc.h/pwmaccsu.h on the 64-bit accumulator register
+    // pair, exposed as a two-result node over the accumulator words.
+    SDValue RdLo = DAG.getExtractVectorElt(DL, XLenVT, Acc, 0);
+    SDValue RdHi = DAG.getExtractVectorElt(DL, XLenVT, Acc, 1);
+    SDVTList VTs = DAG.getVTList(XLenVT, XLenVT);
+    SDValue Res = DAG.getNode(getRVPWideningMulAccPairOpcode(IntNo), DL, VTs,
+                              {RdLo, RdHi, Rs1, Rs2});
+    return DAG.getNode(ISD::BUILD_VECTOR, DL, VT, Res.getValue(0),
+                       Res.getValue(1));
+  }
   case Intrinsic::riscv_pmerge: {
     EVT VT = Op.getValueType();
     auto buildMerge = [&](SDValue Rs1, SDValue Rs2, SDValue Mask,
diff --git a/llvm/lib/Target/RISCV/RISCVInstrInfoP.td b/llvm/lib/Target/RISCV/RISCVInstrInfoP.td
index 4d2c8d9be1073..e0f021fd652f2 100644
--- a/llvm/lib/Target/RISCV/RISCVInstrInfoP.td
+++ b/llvm/lib/Target/RISCV/RISCVInstrInfoP.td
@@ -1977,6 +1977,20 @@ def riscv_pmaccsu_halves_00
 def riscv_pmaccsu_halves_11
     : RVSDNode<"PMACCSU_HALVES_11", SDT_RISCVWideningMulAccByHalves>;
 
+// The pwmacc*/pwmaccsu* instructions use a 64-bit accumulator on RV32.
+// The accumulator and result are represented as two i32 values in the
+// SelectionDAG; custom ISel combines and splits the register pair.
+def SDT_RISCVPackedWideningMulAccPair
+    : SDTypeProfile<2, 4, [SDTCisVT<0, i32>, SDTCisVT<1, i32>,
+                           SDTCisSameAs<0, 2>, SDTCisSameAs<1, 3>,
+                           SDTCisVec<4>, SDTCisSameAs<4, 5>]>;
+def riscv_pwmacc_h   : RVSDNode<"PWMACC_H",
+                                SDT_RISCVPackedWideningMulAccPair>;
+def riscv_pwmaccu_h  : RVSDNode<"PWMACCU_H",
+                                SDT_RISCVPackedWideningMulAccPair>;
+def riscv_pwmaccsu_h : RVSDNode<"PWMACCSU_H",
+                                SDT_RISCVPackedWideningMulAccPair>;
+
 def SDT_RISCVWideningShiftLeft : SDTypeProfile<2, 2, [SDTCisVT<0, i32>,
                                                       SDTCisSameAs<0, 1>,
                                                       SDTCisSameAs<0, 2>,
@@ -2011,6 +2025,7 @@ def SDT_RISCVPackedBinary : SDTypeProfile<1, 2, [SDTCisVec<0>,
                                                  SDTCisSameAs<0, 1>,
                                                  SDTCisSameAs<0, 2>]>;
 def riscv_ppairoe_h : RVSDNode<"PPAIROE_H", SDT_RISCVPackedBinary>;
+def riscv_zip16p : RVSDNode<"ZIP16P", SDT_RISCVPackedBinary>;
 
 // Averaging subtraction, (a - b) >> 2
 def riscv_asub : RVSDNode<"ASUB", SDTIntBinOp>;
@@ -3228,6 +3243,9 @@ let append Predicates = [IsRV64] in {
   def : PatMulPartsAcc<riscv_pmaccsu_halves_00, PMACCSU_W_H00, v2i32, v4i16>;
   def : PatMulPartsAcc<riscv_pmaccsu_halves_11, PMACCSU_W_H11, v2i32, v4i16>;
 
+  def : Pat<(v4i16 (riscv_zip16p (v4i16 GPR:$rs1), (v4i16 GPR:$rs2))),
+            (ZIP16P GPR:$rs1, GPR:$rs2)>;
+
   // Scalar word multiply-parts accumulate patterns.
   def : PatMulPartsAcc<riscv_pmacc_halves_00, MACC_W00, i64, v2i32>;
   def : PatMulPartsAcc<riscv_pmacc_halves_01, MACC_W01, i64, v2i32>;
diff --git a/llvm/test/CodeGen/RISCV/rvp-simd-64.ll b/llvm/test/CodeGen/RISCV/rvp-simd-64.ll
index 34a00069a5597..fdfc97042ddc4 100644
--- a/llvm/test/CodeGen/RISCV/rvp-simd-64.ll
+++ b/llvm/test/CodeGen/RISCV/rvp-simd-64.ll
@@ -8004,6 +8004,55 @@ define <2 x i32> @test_pmaccsu_h11_v2i32(<2 x i32> %rd, <4 x i16> %a, <4 x i16>
   ret <2 x i32> %r
 }
 
+; Packed Widening Multiply Accumulate
+define <2 x i32> @test_pwmacc_i32x2(<2 x i32> %rd, <2 x i16> %rs1, <2 x i16> %rs2) {
+; RV32-LABEL: test_pwmacc_i32x2:
+; RV32:       # %bb.0:
+; RV32-NEXT:    pwmacc.h a0, a2, a3
+; RV32-NEXT:    ret
+;
+; RV64-LABEL: test_pwmacc_i32x2:
+; RV64:       # %bb.0:
+; RV64-NEXT:    zip16p a2, a2, a2
+; RV64-NEXT:    zip16p a1, a1, a1
+; RV64-NEXT:    pmacc.w.h01 a0, a1, a2
+; RV64-NEXT:    ret
+  %res = call <2 x i32> @llvm.riscv.pwmacc.i32x2.v2i32.v2i16(<2 x i32> %rd, <2 x i16> %rs1, <2 x i16> %rs2)
+  ret <2 x i32> %res
+}
+
+define <2 x i32> @test_pwmaccu_u32x2(<2 x i32> %rd, <2 x i16> %rs1, <2 x i16> %rs2) {
+; RV32-LABEL: test_pwmaccu_u32x2:
+; RV32:       # %bb.0:
+; RV32-NEXT:    pwmaccu.h a0, a2, a3
+; RV32-NEXT:    ret
+;
+; RV64-LABEL: test_pwmaccu_u32x2:
+; RV64:       # %bb.0:
+; RV64-NEXT:    zip16p a2, a2, a2
+; RV64-NEXT:    zip16p a1, a1, a1
+; RV64-NEXT:    pmaccu.w.h01 a0, a1, a2
+; RV64-NEXT:    ret
+  %res = call <2 x i32> @llvm.riscv.pwmaccu.u32x2.v2i32.v2i16(<2 x i32> %rd, <2 x i16> %rs1, <2 x i16> %rs2)
+  ret <2 x i32> %res
+}
+
+define <2 x i32> @test_pwmaccsu_i32x2(<2 x i32> %rd, <2 x i16> %rs1, <2 x i16> %rs2) {
+; RV32-LABEL: test_pwmaccsu_i32x2:
+; RV32:       # %bb.0:
+; RV32-NEXT:    pwmaccsu.h a0, a2, a3
+; RV32-NEXT:    ret
+;
+; RV64-LABEL: test_pwmaccsu_i32x2:
+; RV64:       # %bb.0:
+; RV64-NEXT:    pwcvtu.wh a2, a2
+; RV64-NEXT:    pwcvtu.wh a1, a1
+; RV64-NEXT:    pmaccsu.w.h00 a0, a1, a2
+; RV64-NEXT:    ret
+  %res = call <2 x i32> @llvm.riscv.pwmaccsu.i32x2.v2i32.v2i16(<2 x i32> %rd, <2 x i16> %rs1, <2 x i16> %rs2)
+  ret <2 x i32> %res
+}
+
 define i64 @test_macc_w00_i64(i64 %rd, <2 x i32> %a, <2 x i32> %b) {
 ; RV32-LABEL: test_macc_w00_i64:
 ; RV32:       # %bb.0:

>From d50f66d9c9fa250854f5bd5cf6dfa52fe46a072d Mon Sep 17 00:00:00 2001
From: "ZhiQiang.Fan" <571253675 at QQ.com>
Date: Mon, 28 Sep 2026 18:56:56 +0800
Subject: [PATCH 2/2] Resolve review: rv64 pwmacc

---
 llvm/lib/Target/RISCV/RISCVISelDAGToDAG.cpp | 12 --------
 llvm/lib/Target/RISCV/RISCVISelLowering.cpp | 32 +++++++++------------
 llvm/lib/Target/RISCV/RISCVInstrInfoP.td    | 25 +++++++++-------
 llvm/test/CodeGen/RISCV/rvp-simd-64.ll      | 10 +++----
 4 files changed, 32 insertions(+), 47 deletions(-)

diff --git a/llvm/lib/Target/RISCV/RISCVISelDAGToDAG.cpp b/llvm/lib/Target/RISCV/RISCVISelDAGToDAG.cpp
index 3c2b8ee071c95..146aef612f387 100644
--- a/llvm/lib/Target/RISCV/RISCVISelDAGToDAG.cpp
+++ b/llvm/lib/Target/RISCV/RISCVISelDAGToDAG.cpp
@@ -2149,9 +2149,6 @@ void RISCVDAGToDAGISel::Select(SDNode *Node) {
   }
   case RISCVISD::MQWACC:
   case RISCVISD::MQRWACC:
-  case RISCVISD::PWMACC_H:
-  case RISCVISD::PWMACCU_H:
-  case RISCVISD::PWMACCSU_H:
   case RISCVISD::WMACC:
   case RISCVISD::WMACCU:
   case RISCVISD::WMACCSU: {
@@ -2170,15 +2167,6 @@ void RISCVDAGToDAGISel::Select(SDNode *Node) {
     case RISCVISD::MQRWACC:
       Opc = RISCV::MQRWACC;
       break;
-    case RISCVISD::PWMACC_H:
-      Opc = RISCV::PWMACC_H;
-      break;
-    case RISCVISD::PWMACCU_H:
-      Opc = RISCV::PWMACCU_H;
-      break;
-    case RISCVISD::PWMACCSU_H:
-      Opc = RISCV::PWMACCSU_H;
-      break;
     case RISCVISD::WMACC:
       Opc = RISCV::WMACC;
       break;
diff --git a/llvm/lib/Target/RISCV/RISCVISelLowering.cpp b/llvm/lib/Target/RISCV/RISCVISelLowering.cpp
index 0115e122703a2..f75186053c415 100644
--- a/llvm/lib/Target/RISCV/RISCVISelLowering.cpp
+++ b/llvm/lib/Target/RISCV/RISCVISelLowering.cpp
@@ -12524,7 +12524,7 @@ static unsigned getRVPHorizontalMulOpcode(unsigned IntNo) {
   }
 }
 
-static unsigned getRVPWideningMulAccPairOpcode(unsigned IntNo) {
+static unsigned getRVPWideningMulAccOpcode(unsigned IntNo) {
   switch (IntNo) {
   default:
     llvm_unreachable("Unexpected RISC-V packed pwmacc intrinsic");
@@ -13409,11 +13409,11 @@ SDValue RISCVTargetLowering::LowerINTRINSIC_WO_CHAIN(SDValue Op,
     MVT XLenVT = Subtarget.getXLenVT();
 
     if (Subtarget.is64Bit()) {
-      // Per the spec decomposition table: zip16p duplicates each halfword
-      // into both positions of its 32-bit lane (the su form uses the
-      // pwcvtu.wh spelling, zip16p with x0), then pmacc.w.h01 or
-      // pmaccsu.w.h00 multiplies the lane-local halfwords into rd. The
-      // upper halfwords of the operands are not read. RV32 selects the
+      // Per the spec decomposition table: a single zip16p puts the
+      // halfwords of Rs1 in the even elements and those of Rs2 in the odd
+      // elements (the su form uses the pwcvtu.wh spelling, zip16p with x0).
+      // pmacc.w.h01 then multiplies the even elements of the first source
+      // by the odd elements of the second source into rd. RV32 selects the
       // pwmacc.h family on the register pair instead.
       Rs1 = DAG.getNode(ISD::CONCAT_VECTORS, DL, MVT::v4i16, Rs1,
                         DAG.getUNDEF(MVT::v2i16));
@@ -13422,27 +13422,21 @@ SDValue RISCVTargetLowering::LowerINTRINSIC_WO_CHAIN(SDValue Op,
       if (IntNo == Intrinsic::riscv_pwmaccsu_i32x2) {
         SDValue Zero = DAG.getNode(ISD::SPLAT_VECTOR, DL, MVT::v4i16,
                                    DAG.getConstant(0, DL, XLenVT));
-        Rs1 = DAG.getNode(RISCVISD::ZIP16P, DL, MVT::v4i16, Rs1, Zero);
-        Rs2 = DAG.getNode(RISCVISD::ZIP16P, DL, MVT::v4i16, Rs2, Zero);
+        Rs1 = DAG.getNode(RISCVISD::PZIP, DL, MVT::v4i16, Rs1, Zero);
+        Rs2 = DAG.getNode(RISCVISD::PZIP, DL, MVT::v4i16, Rs2, Zero);
         return DAG.getNode(RISCVISD::PMACCSU_HALVES_00, DL, VT, Acc, Rs1, Rs2);
       }
-      Rs1 = DAG.getNode(RISCVISD::ZIP16P, DL, MVT::v4i16, Rs1, Rs1);
-      Rs2 = DAG.getNode(RISCVISD::ZIP16P, DL, MVT::v4i16, Rs2, Rs2);
+      SDValue Zip = DAG.getNode(RISCVISD::PZIP, DL, MVT::v4i16, Rs1, Rs2);
       unsigned Opc = IntNo == Intrinsic::riscv_pwmaccu_u32x2
                          ? RISCVISD::PMACCU_HALVES_01
                          : RISCVISD::PMACC_HALVES_01;
-      return DAG.getNode(Opc, DL, VT, Acc, Rs1, Rs2);
+      return DAG.getNode(Opc, DL, VT, Acc, Zip, Zip);
     }
 
     // RV32 uses pwmacc.h/pwmaccsu.h on the 64-bit accumulator register
-    // pair, exposed as a two-result node over the accumulator words.
-    SDValue RdLo = DAG.getExtractVectorElt(DL, XLenVT, Acc, 0);
-    SDValue RdHi = DAG.getExtractVectorElt(DL, XLenVT, Acc, 1);
-    SDVTList VTs = DAG.getVTList(XLenVT, XLenVT);
-    SDValue Res = DAG.getNode(getRVPWideningMulAccPairOpcode(IntNo), DL, VTs,
-                              {RdLo, RdHi, Rs1, Rs2});
-    return DAG.getNode(ISD::BUILD_VECTOR, DL, VT, Res.getValue(0),
-                       Res.getValue(1));
+    // pair, represented as a single v2i32 value.
+    return DAG.getNode(getRVPWideningMulAccOpcode(IntNo), DL, VT, Acc, Rs1,
+                       Rs2);
   }
   case Intrinsic::riscv_pmerge: {
     EVT VT = Op.getValueType();
diff --git a/llvm/lib/Target/RISCV/RISCVInstrInfoP.td b/llvm/lib/Target/RISCV/RISCVInstrInfoP.td
index e0f021fd652f2..cffd39fea701f 100644
--- a/llvm/lib/Target/RISCV/RISCVInstrInfoP.td
+++ b/llvm/lib/Target/RISCV/RISCVInstrInfoP.td
@@ -1977,13 +1977,11 @@ def riscv_pmaccsu_halves_00
 def riscv_pmaccsu_halves_11
     : RVSDNode<"PMACCSU_HALVES_11", SDT_RISCVWideningMulAccByHalves>;
 
-// The pwmacc*/pwmaccsu* instructions use a 64-bit accumulator on RV32.
-// The accumulator and result are represented as two i32 values in the
-// SelectionDAG; custom ISel combines and splits the register pair.
+// The pwmacc*/pwmaccsu* instructions accumulate into the 64-bit register
+// pair holding rd, represented as a v2i32 on RV32.
 def SDT_RISCVPackedWideningMulAccPair
-    : SDTypeProfile<2, 4, [SDTCisVT<0, i32>, SDTCisVT<1, i32>,
-                           SDTCisSameAs<0, 2>, SDTCisSameAs<1, 3>,
-                           SDTCisVec<4>, SDTCisSameAs<4, 5>]>;
+    : SDTypeProfile<1, 3, [SDTCisVT<0, v2i32>, SDTCisSameAs<0, 1>,
+                           SDTCisVT<2, v2i16>, SDTCisSameAs<2, 3>]>;
 def riscv_pwmacc_h   : RVSDNode<"PWMACC_H",
                                 SDT_RISCVPackedWideningMulAccPair>;
 def riscv_pwmaccu_h  : RVSDNode<"PWMACCU_H",
@@ -2025,7 +2023,6 @@ def SDT_RISCVPackedBinary : SDTypeProfile<1, 2, [SDTCisVec<0>,
                                                  SDTCisSameAs<0, 1>,
                                                  SDTCisSameAs<0, 2>]>;
 def riscv_ppairoe_h : RVSDNode<"PPAIROE_H", SDT_RISCVPackedBinary>;
-def riscv_zip16p : RVSDNode<"ZIP16P", SDT_RISCVPackedBinary>;
 
 // Averaging subtraction, (a - b) >> 2
 def riscv_asub : RVSDNode<"ASUB", SDTIntBinOp>;
@@ -2704,6 +2701,17 @@ let append Predicates = [IsRV32] in {
                                      (v2i16 GPR:$rs1), (v2i16 GPR:$rs2))),
             (PM2WADDASU_H GPRPair:$rd, GPR:$rs1, GPR:$rs2)>;
 
+  // Packed widening multiply accumulate patterns.
+  def : Pat<(v2i32 (riscv_pwmacc_h (v2i32 GPRPair:$rd),
+                                   (v2i16 GPR:$rs1), (v2i16 GPR:$rs2))),
+            (PWMACC_H GPRPair:$rd, GPR:$rs1, GPR:$rs2)>;
+  def : Pat<(v2i32 (riscv_pwmaccu_h (v2i32 GPRPair:$rd),
+                                    (v2i16 GPR:$rs1), (v2i16 GPR:$rs2))),
+            (PWMACCU_H GPRPair:$rd, GPR:$rs1, GPR:$rs2)>;
+  def : Pat<(v2i32 (riscv_pwmaccsu_h (v2i32 GPRPair:$rd),
+                                     (v2i16 GPR:$rs1), (v2i16 GPR:$rs2))),
+            (PWMACCSU_H GPRPair:$rd, GPR:$rs1, GPR:$rs2)>;
+
   // 8/16-bit bitreverse patterns
   // With Zbkb, brev8 reverses the bits within each byte directly; otherwise
   // reverse all bits then swap the bytes back.
@@ -3243,9 +3251,6 @@ let append Predicates = [IsRV64] in {
   def : PatMulPartsAcc<riscv_pmaccsu_halves_00, PMACCSU_W_H00, v2i32, v4i16>;
   def : PatMulPartsAcc<riscv_pmaccsu_halves_11, PMACCSU_W_H11, v2i32, v4i16>;
 
-  def : Pat<(v4i16 (riscv_zip16p (v4i16 GPR:$rs1), (v4i16 GPR:$rs2))),
-            (ZIP16P GPR:$rs1, GPR:$rs2)>;
-
   // Scalar word multiply-parts accumulate patterns.
   def : PatMulPartsAcc<riscv_pmacc_halves_00, MACC_W00, i64, v2i32>;
   def : PatMulPartsAcc<riscv_pmacc_halves_01, MACC_W01, i64, v2i32>;
diff --git a/llvm/test/CodeGen/RISCV/rvp-simd-64.ll b/llvm/test/CodeGen/RISCV/rvp-simd-64.ll
index fdfc97042ddc4..455b745f92bcc 100644
--- a/llvm/test/CodeGen/RISCV/rvp-simd-64.ll
+++ b/llvm/test/CodeGen/RISCV/rvp-simd-64.ll
@@ -8013,9 +8013,8 @@ define <2 x i32> @test_pwmacc_i32x2(<2 x i32> %rd, <2 x i16> %rs1, <2 x i16> %rs
 ;
 ; RV64-LABEL: test_pwmacc_i32x2:
 ; RV64:       # %bb.0:
-; RV64-NEXT:    zip16p a2, a2, a2
-; RV64-NEXT:    zip16p a1, a1, a1
-; RV64-NEXT:    pmacc.w.h01 a0, a1, a2
+; RV64-NEXT:    zip16p a1, a1, a2
+; RV64-NEXT:    pmacc.w.h01 a0, a1, a1
 ; RV64-NEXT:    ret
   %res = call <2 x i32> @llvm.riscv.pwmacc.i32x2.v2i32.v2i16(<2 x i32> %rd, <2 x i16> %rs1, <2 x i16> %rs2)
   ret <2 x i32> %res
@@ -8029,9 +8028,8 @@ define <2 x i32> @test_pwmaccu_u32x2(<2 x i32> %rd, <2 x i16> %rs1, <2 x i16> %r
 ;
 ; RV64-LABEL: test_pwmaccu_u32x2:
 ; RV64:       # %bb.0:
-; RV64-NEXT:    zip16p a2, a2, a2
-; RV64-NEXT:    zip16p a1, a1, a1
-; RV64-NEXT:    pmaccu.w.h01 a0, a1, a2
+; RV64-NEXT:    zip16p a1, a1, a2
+; RV64-NEXT:    pmaccu.w.h01 a0, a1, a1
 ; RV64-NEXT:    ret
   %res = call <2 x i32> @llvm.riscv.pwmaccu.u32x2.v2i32.v2i16(<2 x i32> %rd, <2 x i16> %rs1, <2 x i16> %rs2)
   ret <2 x i32> %res



More information about the llvm-commits mailing list