[clang] [llvm] [RISCV][P-ext] Add packed multiply-parts accumulate intrinsics (PR #222571)

via cfe-commits cfe-commits at lists.llvm.org
Thu Sep 10 02:49:09 PDT 2026


https://github.com/sihuan created https://github.com/llvm/llvm-project/pull/222571

Add SelectionDAG and intrinsic support for the RISC-V P multiply-parts
accumulate operations, which add the selected products into rd. See also
https://github.com/riscv/riscv-p-spec/blob/master/P-ext-intrinsics.adoc#packed-multiply-parts-accumulate

The intrinsics lower to `PMACC_HALVES_*` nodes rather than to `add` of the
plain products, so the selection does not depend on a fold. On RV32 the word
forms accumulate into a GPR pair with new `WMACC`/`WMACCU`/`WMACCSU` nodes.

Also adds the Clang builtins and the `riscv_packed_simd.h` wrappers.


>From 4003577d908022bcd1d614f5855fbe7cdd30822e Mon Sep 17 00:00:00 2001
From: SiHuaN <liyongtai at iscas.ac.cn>
Date: Fri, 28 Aug 2026 08:15:57 +0000
Subject: [PATCH] [RISCV][P-ext] Add packed multiply-parts accumulate
 intrinsics

Add SelectionDAG and intrinsic support for the RISC-V P multiply-parts
accumulate operations, which add the selected products into rd. See also
https://github.com/riscv/riscv-p-spec/blob/master/P-ext-intrinsics.adoc#packed-multiply-parts-accumulate

Also adds the Clang builtins and the `riscv_packed_simd.h` wrappers.
---
 clang/include/clang/Basic/BuiltinsRISCV.td    |  28 ++
 clang/lib/CodeGen/TargetBuiltins/RISCV.cpp    |  74 ++++++
 clang/lib/Headers/riscv_packed_simd.h         |  28 ++
 .../riscv_packed_simd.c                       | 172 ++++++++++++
 llvm/include/llvm/IR/IntrinsicsRISCV.td       |  30 +++
 llvm/lib/Target/RISCV/RISCVISelDAGToDAG.cpp   |  26 +-
 llvm/lib/Target/RISCV/RISCVISelLowering.cpp   | 139 ++++++++++
 llvm/lib/Target/RISCV/RISCVInstrInfoP.td      |  63 +++++
 llvm/test/CodeGen/RISCV/rvp-simd-32.ll        | 122 +++++++++
 llvm/test/CodeGen/RISCV/rvp-simd-64.ll        | 250 ++++++++++++++++++
 10 files changed, 930 insertions(+), 2 deletions(-)

diff --git a/clang/include/clang/Basic/BuiltinsRISCV.td b/clang/include/clang/Basic/BuiltinsRISCV.td
index 58184479d3c27..ee840e45a65ba 100644
--- a/clang/include/clang/Basic/BuiltinsRISCV.td
+++ b/clang/include/clang/Basic/BuiltinsRISCV.td
@@ -426,6 +426,34 @@ def mqracc_w00_i64 : RISCVBuiltin<"int64_t(int64_t, _Vector<2, int>, _Vector<2,
 def mqracc_w01_i64 : RISCVBuiltin<"int64_t(int64_t, _Vector<2, int>, _Vector<2, int>)">;
 def mqracc_w11_i64 : RISCVBuiltin<"int64_t(int64_t, _Vector<2, int>, _Vector<2, int>)">;
 
+// Packed Multiply Parts Accumulate (32-bit)
+def macc_h00_i32 : RISCVBuiltin<"int(int, _Vector<2, short>, _Vector<2, short>)">;
+def macc_h01_i32 : RISCVBuiltin<"int(int, _Vector<2, short>, _Vector<2, short>)">;
+def macc_h11_i32 : RISCVBuiltin<"int(int, _Vector<2, short>, _Vector<2, short>)">;
+def maccu_h00_u32 : RISCVBuiltin<"unsigned int(unsigned int, _Vector<2, unsigned short>, _Vector<2, unsigned short>)">;
+def maccu_h01_u32 : RISCVBuiltin<"unsigned int(unsigned int, _Vector<2, unsigned short>, _Vector<2, unsigned short>)">;
+def maccu_h11_u32 : RISCVBuiltin<"unsigned int(unsigned int, _Vector<2, unsigned short>, _Vector<2, unsigned short>)">;
+def maccsu_h00_i32 : RISCVBuiltin<"int(int, _Vector<2, short>, _Vector<2, unsigned short>)">;
+def maccsu_h11_i32 : RISCVBuiltin<"int(int, _Vector<2, short>, _Vector<2, unsigned short>)">;
+
+// Packed Multiply Parts Accumulate (64-bit)
+def pmacc_h00_i32x2 : RISCVBuiltin<"_Vector<2, int>(_Vector<2, int>, _Vector<4, short>, _Vector<4, short>)">;
+def pmacc_h01_i32x2 : RISCVBuiltin<"_Vector<2, int>(_Vector<2, int>, _Vector<4, short>, _Vector<4, short>)">;
+def pmacc_h11_i32x2 : RISCVBuiltin<"_Vector<2, int>(_Vector<2, int>, _Vector<4, short>, _Vector<4, short>)">;
+def pmaccu_h00_u32x2 : RISCVBuiltin<"_Vector<2, unsigned int>(_Vector<2, unsigned int>, _Vector<4, unsigned short>, _Vector<4, unsigned short>)">;
+def pmaccu_h01_u32x2 : RISCVBuiltin<"_Vector<2, unsigned int>(_Vector<2, unsigned int>, _Vector<4, unsigned short>, _Vector<4, unsigned short>)">;
+def pmaccu_h11_u32x2 : RISCVBuiltin<"_Vector<2, unsigned int>(_Vector<2, unsigned int>, _Vector<4, unsigned short>, _Vector<4, unsigned short>)">;
+def pmaccsu_h00_i32x2 : RISCVBuiltin<"_Vector<2, int>(_Vector<2, int>, _Vector<4, short>, _Vector<4, unsigned short>)">;
+def pmaccsu_h11_i32x2 : RISCVBuiltin<"_Vector<2, int>(_Vector<2, int>, _Vector<4, short>, _Vector<4, unsigned short>)">;
+def macc_w00_i64 : RISCVBuiltin<"int64_t(int64_t, _Vector<2, int>, _Vector<2, int>)">;
+def macc_w01_i64 : RISCVBuiltin<"int64_t(int64_t, _Vector<2, int>, _Vector<2, int>)">;
+def macc_w11_i64 : RISCVBuiltin<"int64_t(int64_t, _Vector<2, int>, _Vector<2, int>)">;
+def maccu_w00_u64 : RISCVBuiltin<"uint64_t(uint64_t, _Vector<2, unsigned int>, _Vector<2, unsigned int>)">;
+def maccu_w01_u64 : RISCVBuiltin<"uint64_t(uint64_t, _Vector<2, unsigned int>, _Vector<2, unsigned int>)">;
+def maccu_w11_u64 : RISCVBuiltin<"uint64_t(uint64_t, _Vector<2, unsigned int>, _Vector<2, unsigned int>)">;
+def maccsu_w00_i64 : RISCVBuiltin<"int64_t(int64_t, _Vector<2, int>, _Vector<2, unsigned int>)">;
+def maccsu_w11_i64 : RISCVBuiltin<"int64_t(int64_t, _Vector<2, int>, _Vector<2, unsigned int>)">;
+
 // Packed Sign and Zero Extend (32-bit)
 def psext_b_i16x2 : RISCVBuiltin<"_Vector<2, short>(_Vector<2, short>)">;
 def pzext_b_u16x2 : RISCVBuiltin<"_Vector<2, unsigned short>(_Vector<2, unsigned short>)">;
diff --git a/clang/lib/CodeGen/TargetBuiltins/RISCV.cpp b/clang/lib/CodeGen/TargetBuiltins/RISCV.cpp
index 2e9fbf771bed1..f99a05ce673aa 100644
--- a/clang/lib/CodeGen/TargetBuiltins/RISCV.cpp
+++ b/clang/lib/CodeGen/TargetBuiltins/RISCV.cpp
@@ -1938,6 +1938,80 @@ Value *CodeGenFunction::EmitRISCVBuiltinExpr(unsigned BuiltinID,
     break;
   }
 
+  // Packed Multiply Parts Accumulate.
+  case RISCV::BI__builtin_riscv_macc_h00_i32:
+  case RISCV::BI__builtin_riscv_pmacc_h00_i32x2:
+  case RISCV::BI__builtin_riscv_macc_w00_i64:
+  case RISCV::BI__builtin_riscv_macc_h01_i32:
+  case RISCV::BI__builtin_riscv_pmacc_h01_i32x2:
+  case RISCV::BI__builtin_riscv_macc_w01_i64:
+  case RISCV::BI__builtin_riscv_macc_h11_i32:
+  case RISCV::BI__builtin_riscv_pmacc_h11_i32x2:
+  case RISCV::BI__builtin_riscv_macc_w11_i64:
+  case RISCV::BI__builtin_riscv_maccu_h00_u32:
+  case RISCV::BI__builtin_riscv_pmaccu_h00_u32x2:
+  case RISCV::BI__builtin_riscv_maccu_w00_u64:
+  case RISCV::BI__builtin_riscv_maccu_h01_u32:
+  case RISCV::BI__builtin_riscv_pmaccu_h01_u32x2:
+  case RISCV::BI__builtin_riscv_maccu_w01_u64:
+  case RISCV::BI__builtin_riscv_maccu_h11_u32:
+  case RISCV::BI__builtin_riscv_pmaccu_h11_u32x2:
+  case RISCV::BI__builtin_riscv_maccu_w11_u64:
+  case RISCV::BI__builtin_riscv_maccsu_h00_i32:
+  case RISCV::BI__builtin_riscv_pmaccsu_h00_i32x2:
+  case RISCV::BI__builtin_riscv_maccsu_w00_i64:
+  case RISCV::BI__builtin_riscv_maccsu_h11_i32:
+  case RISCV::BI__builtin_riscv_pmaccsu_h11_i32x2:
+  case RISCV::BI__builtin_riscv_maccsu_w11_i64: {
+    switch (BuiltinID) {
+    default:
+      llvm_unreachable("unexpected builtin ID");
+    case RISCV::BI__builtin_riscv_macc_h00_i32:
+    case RISCV::BI__builtin_riscv_pmacc_h00_i32x2:
+    case RISCV::BI__builtin_riscv_macc_w00_i64:
+      ID = Intrinsic::riscv_macc_00;
+      break;
+    case RISCV::BI__builtin_riscv_macc_h01_i32:
+    case RISCV::BI__builtin_riscv_pmacc_h01_i32x2:
+    case RISCV::BI__builtin_riscv_macc_w01_i64:
+      ID = Intrinsic::riscv_macc_01;
+      break;
+    case RISCV::BI__builtin_riscv_macc_h11_i32:
+    case RISCV::BI__builtin_riscv_pmacc_h11_i32x2:
+    case RISCV::BI__builtin_riscv_macc_w11_i64:
+      ID = Intrinsic::riscv_macc_11;
+      break;
+    case RISCV::BI__builtin_riscv_maccu_h00_u32:
+    case RISCV::BI__builtin_riscv_pmaccu_h00_u32x2:
+    case RISCV::BI__builtin_riscv_maccu_w00_u64:
+      ID = Intrinsic::riscv_maccu_00;
+      break;
+    case RISCV::BI__builtin_riscv_maccu_h01_u32:
+    case RISCV::BI__builtin_riscv_pmaccu_h01_u32x2:
+    case RISCV::BI__builtin_riscv_maccu_w01_u64:
+      ID = Intrinsic::riscv_maccu_01;
+      break;
+    case RISCV::BI__builtin_riscv_maccu_h11_u32:
+    case RISCV::BI__builtin_riscv_pmaccu_h11_u32x2:
+    case RISCV::BI__builtin_riscv_maccu_w11_u64:
+      ID = Intrinsic::riscv_maccu_11;
+      break;
+    case RISCV::BI__builtin_riscv_maccsu_h00_i32:
+    case RISCV::BI__builtin_riscv_pmaccsu_h00_i32x2:
+    case RISCV::BI__builtin_riscv_maccsu_w00_i64:
+      ID = Intrinsic::riscv_maccsu_00;
+      break;
+    case RISCV::BI__builtin_riscv_maccsu_h11_i32:
+    case RISCV::BI__builtin_riscv_pmaccsu_h11_i32x2:
+    case RISCV::BI__builtin_riscv_maccsu_w11_i64:
+      ID = Intrinsic::riscv_maccsu_11;
+      break;
+    }
+
+    IntrinsicTypes = {ResultType, Ops[1]->getType()};
+    break;
+  }
+
   // Zk builtins
 
   // Zknh
diff --git a/clang/lib/Headers/riscv_packed_simd.h b/clang/lib/Headers/riscv_packed_simd.h
index 37ac140780d02..5c359337d8b7f 100644
--- a/clang/lib/Headers/riscv_packed_simd.h
+++ b/clang/lib/Headers/riscv_packed_simd.h
@@ -984,6 +984,34 @@ __packed_ternary_builtin_cast(mqracc_w00_i64, int64_t, int32x2_t, __builtin_risc
 __packed_ternary_builtin_cast(mqracc_w01_i64, int64_t, int32x2_t, __builtin_riscv_mqracc_w01_i64)
 __packed_ternary_builtin_cast(mqracc_w11_i64, int64_t, int32x2_t, __builtin_riscv_mqracc_w11_i64)
 
+/* Packed Multiply Parts Accumulate (32-bit) */
+__packed_ternary_builtin_mixed(macc_h00_i32, int32_t, int16x2_t, int16x2_t, __builtin_riscv_macc_h00_i32)
+__packed_ternary_builtin_mixed(macc_h01_i32, int32_t, int16x2_t, int16x2_t, __builtin_riscv_macc_h01_i32)
+__packed_ternary_builtin_mixed(macc_h11_i32, int32_t, int16x2_t, int16x2_t, __builtin_riscv_macc_h11_i32)
+__packed_ternary_builtin_mixed(maccu_h00_u32, uint32_t, uint16x2_t, uint16x2_t, __builtin_riscv_maccu_h00_u32)
+__packed_ternary_builtin_mixed(maccu_h01_u32, uint32_t, uint16x2_t, uint16x2_t, __builtin_riscv_maccu_h01_u32)
+__packed_ternary_builtin_mixed(maccu_h11_u32, uint32_t, uint16x2_t, uint16x2_t, __builtin_riscv_maccu_h11_u32)
+__packed_ternary_builtin_mixed(maccsu_h00_i32, int32_t, int16x2_t, uint16x2_t, __builtin_riscv_maccsu_h00_i32)
+__packed_ternary_builtin_mixed(maccsu_h11_i32, int32_t, int16x2_t, uint16x2_t, __builtin_riscv_maccsu_h11_i32)
+
+/* Packed Multiply Parts Accumulate (64-bit) */
+__packed_ternary_builtin_mixed(pmacc_h00_i32x2, int32x2_t, int16x4_t, int16x4_t, __builtin_riscv_pmacc_h00_i32x2)
+__packed_ternary_builtin_mixed(pmacc_h01_i32x2, int32x2_t, int16x4_t, int16x4_t, __builtin_riscv_pmacc_h01_i32x2)
+__packed_ternary_builtin_mixed(pmacc_h11_i32x2, int32x2_t, int16x4_t, int16x4_t, __builtin_riscv_pmacc_h11_i32x2)
+__packed_ternary_builtin_mixed(pmaccu_h00_u32x2, uint32x2_t, uint16x4_t, uint16x4_t, __builtin_riscv_pmaccu_h00_u32x2)
+__packed_ternary_builtin_mixed(pmaccu_h01_u32x2, uint32x2_t, uint16x4_t, uint16x4_t, __builtin_riscv_pmaccu_h01_u32x2)
+__packed_ternary_builtin_mixed(pmaccu_h11_u32x2, uint32x2_t, uint16x4_t, uint16x4_t, __builtin_riscv_pmaccu_h11_u32x2)
+__packed_ternary_builtin_mixed(pmaccsu_h00_i32x2, int32x2_t, int16x4_t, uint16x4_t, __builtin_riscv_pmaccsu_h00_i32x2)
+__packed_ternary_builtin_mixed(pmaccsu_h11_i32x2, int32x2_t, int16x4_t, uint16x4_t, __builtin_riscv_pmaccsu_h11_i32x2)
+__packed_ternary_builtin_mixed(macc_w00_i64, int64_t, int32x2_t, int32x2_t, __builtin_riscv_macc_w00_i64)
+__packed_ternary_builtin_mixed(macc_w01_i64, int64_t, int32x2_t, int32x2_t, __builtin_riscv_macc_w01_i64)
+__packed_ternary_builtin_mixed(macc_w11_i64, int64_t, int32x2_t, int32x2_t, __builtin_riscv_macc_w11_i64)
+__packed_ternary_builtin_mixed(maccu_w00_u64, uint64_t, uint32x2_t, uint32x2_t, __builtin_riscv_maccu_w00_u64)
+__packed_ternary_builtin_mixed(maccu_w01_u64, uint64_t, uint32x2_t, uint32x2_t, __builtin_riscv_maccu_w01_u64)
+__packed_ternary_builtin_mixed(maccu_w11_u64, uint64_t, uint32x2_t, uint32x2_t, __builtin_riscv_maccu_w11_u64)
+__packed_ternary_builtin_mixed(maccsu_w00_i64, int64_t, int32x2_t, uint32x2_t, __builtin_riscv_maccsu_w00_i64)
+__packed_ternary_builtin_mixed(maccsu_w11_i64, int64_t, int32x2_t, uint32x2_t, __builtin_riscv_maccsu_w11_i64)
+
 /* Packed Narrowing Clip Pair (32-bit) */
 __packed_binary_builtin_cast(pnclipp_i8x4, int16x2_t, int8x4_t, __builtin_riscv_pnclipp_i8x4)
 __packed_binary_builtin_cast(pnclipup_u8x4, uint16x2_t, uint8x4_t, __builtin_riscv_pnclipup_u8x4)
diff --git a/cross-project-tests/intrinsic-header-tests/riscv_packed_simd.c b/cross-project-tests/intrinsic-header-tests/riscv_packed_simd.c
index 48fcf036794ff..df9623edfa71a 100644
--- a/cross-project-tests/intrinsic-header-tests/riscv_packed_simd.c
+++ b/cross-project-tests/intrinsic-header-tests/riscv_packed_simd.c
@@ -4248,3 +4248,175 @@ int32_t test_pget_i32x2_i32(int32x2_t v) {
 uint32_t test_pget_u32x2_u32(uint32x2_t v) {
   return __riscv_pget_u32x2_u32(v, 1);
 }
+
+/* Packed Multiply Parts Accumulate (32-bit) */
+
+// CHECK-LABEL: test_macc_h00_i32:
+// RV32:        macc.h00
+// RV64:        pmacc.w.h00
+int32_t test_macc_h00_i32(int32_t rd, int16x2_t a, int16x2_t b) {
+  return __riscv_macc_h00_i32(rd, a, b);
+}
+
+// CHECK-LABEL: test_macc_h01_i32:
+// RV32:        macc.h01
+// RV64:        pmacc.w.h01
+int32_t test_macc_h01_i32(int32_t rd, int16x2_t a, int16x2_t b) {
+  return __riscv_macc_h01_i32(rd, a, b);
+}
+
+// CHECK-LABEL: test_macc_h11_i32:
+// RV32:        macc.h11
+// RV64:        pmacc.w.h11
+int32_t test_macc_h11_i32(int32_t rd, int16x2_t a, int16x2_t b) {
+  return __riscv_macc_h11_i32(rd, a, b);
+}
+
+// CHECK-LABEL: test_maccu_h00_u32:
+// RV32:        maccu.h00
+// RV64:        pmaccu.w.h00
+uint32_t test_maccu_h00_u32(uint32_t rd, uint16x2_t a, uint16x2_t b) {
+  return __riscv_maccu_h00_u32(rd, a, b);
+}
+
+// CHECK-LABEL: test_maccu_h01_u32:
+// RV32:        maccu.h01
+// RV64:        pmaccu.w.h01
+uint32_t test_maccu_h01_u32(uint32_t rd, uint16x2_t a, uint16x2_t b) {
+  return __riscv_maccu_h01_u32(rd, a, b);
+}
+
+// CHECK-LABEL: test_maccu_h11_u32:
+// RV32:        maccu.h11
+// RV64:        pmaccu.w.h11
+uint32_t test_maccu_h11_u32(uint32_t rd, uint16x2_t a, uint16x2_t b) {
+  return __riscv_maccu_h11_u32(rd, a, b);
+}
+
+// CHECK-LABEL: test_maccsu_h00_i32:
+// RV32:        maccsu.h00
+// RV64:        pmaccsu.w.h00
+int32_t test_maccsu_h00_i32(int32_t rd, int16x2_t a, uint16x2_t b) {
+  return __riscv_maccsu_h00_i32(rd, a, b);
+}
+
+// CHECK-LABEL: test_maccsu_h11_i32:
+// RV32:        maccsu.h11
+// RV64:        pmaccsu.w.h11
+int32_t test_maccsu_h11_i32(int32_t rd, int16x2_t a, uint16x2_t b) {
+  return __riscv_maccsu_h11_i32(rd, a, b);
+}
+
+/* Packed Multiply Parts Accumulate (64-bit) */
+
+// CHECK-LABEL: test_pmacc_h00_i32x2:
+// RV32-COUNT-2: macc.h00
+// RV64:         pmacc.w.h00
+int32x2_t test_pmacc_h00_i32x2(int32x2_t rd, int16x4_t a, int16x4_t b) {
+  return __riscv_pmacc_h00_i32x2(rd, a, b);
+}
+
+// CHECK-LABEL: test_pmacc_h01_i32x2:
+// RV32-COUNT-2: macc.h01
+// RV64:         pmacc.w.h01
+int32x2_t test_pmacc_h01_i32x2(int32x2_t rd, int16x4_t a, int16x4_t b) {
+  return __riscv_pmacc_h01_i32x2(rd, a, b);
+}
+
+// CHECK-LABEL: test_pmacc_h11_i32x2:
+// RV32-COUNT-2: macc.h11
+// RV64:         pmacc.w.h11
+int32x2_t test_pmacc_h11_i32x2(int32x2_t rd, int16x4_t a, int16x4_t b) {
+  return __riscv_pmacc_h11_i32x2(rd, a, b);
+}
+
+// CHECK-LABEL: test_pmaccu_h00_u32x2:
+// RV32-COUNT-2: maccu.h00
+// RV64:         pmaccu.w.h00
+uint32x2_t test_pmaccu_h00_u32x2(uint32x2_t rd, uint16x4_t a, uint16x4_t b) {
+  return __riscv_pmaccu_h00_u32x2(rd, a, b);
+}
+
+// CHECK-LABEL: test_pmaccu_h01_u32x2:
+// RV32-COUNT-2: maccu.h01
+// RV64:         pmaccu.w.h01
+uint32x2_t test_pmaccu_h01_u32x2(uint32x2_t rd, uint16x4_t a, uint16x4_t b) {
+  return __riscv_pmaccu_h01_u32x2(rd, a, b);
+}
+
+// CHECK-LABEL: test_pmaccu_h11_u32x2:
+// RV32-COUNT-2: maccu.h11
+// RV64:         pmaccu.w.h11
+uint32x2_t test_pmaccu_h11_u32x2(uint32x2_t rd, uint16x4_t a, uint16x4_t b) {
+  return __riscv_pmaccu_h11_u32x2(rd, a, b);
+}
+
+// CHECK-LABEL: test_pmaccsu_h00_i32x2:
+// RV32-COUNT-2: maccsu.h00
+// RV64:         pmaccsu.w.h00
+int32x2_t test_pmaccsu_h00_i32x2(int32x2_t rd, int16x4_t a, uint16x4_t b) {
+  return __riscv_pmaccsu_h00_i32x2(rd, a, b);
+}
+
+// CHECK-LABEL: test_pmaccsu_h11_i32x2:
+// RV32-COUNT-2: maccsu.h11
+// RV64:         pmaccsu.w.h11
+int32x2_t test_pmaccsu_h11_i32x2(int32x2_t rd, int16x4_t a, uint16x4_t b) {
+  return __riscv_pmaccsu_h11_i32x2(rd, a, b);
+}
+
+// CHECK-LABEL: test_macc_w00_i64:
+// RV32:        wmacc
+// RV64:        macc.w00
+int64_t test_macc_w00_i64(int64_t rd, int32x2_t a, int32x2_t b) {
+  return __riscv_macc_w00_i64(rd, a, b);
+}
+
+// CHECK-LABEL: test_macc_w01_i64:
+// RV32:        wmacc
+// RV64:        macc.w01
+int64_t test_macc_w01_i64(int64_t rd, int32x2_t a, int32x2_t b) {
+  return __riscv_macc_w01_i64(rd, a, b);
+}
+
+// CHECK-LABEL: test_macc_w11_i64:
+// RV32:        wmacc
+// RV64:        macc.w11
+int64_t test_macc_w11_i64(int64_t rd, int32x2_t a, int32x2_t b) {
+  return __riscv_macc_w11_i64(rd, a, b);
+}
+
+// CHECK-LABEL: test_maccu_w00_u64:
+// RV32:        wmaccu
+// RV64:        maccu.w00
+uint64_t test_maccu_w00_u64(uint64_t rd, uint32x2_t a, uint32x2_t b) {
+  return __riscv_maccu_w00_u64(rd, a, b);
+}
+
+// CHECK-LABEL: test_maccu_w01_u64:
+// RV32:        wmaccu
+// RV64:        maccu.w01
+uint64_t test_maccu_w01_u64(uint64_t rd, uint32x2_t a, uint32x2_t b) {
+  return __riscv_maccu_w01_u64(rd, a, b);
+}
+
+// CHECK-LABEL: test_maccu_w11_u64:
+// RV32:        wmaccu
+// RV64:        maccu.w11
+uint64_t test_maccu_w11_u64(uint64_t rd, uint32x2_t a, uint32x2_t b) {
+  return __riscv_maccu_w11_u64(rd, a, b);
+}
+
+// CHECK-LABEL: test_maccsu_w00_i64:
+// RV32:        wmaccsu
+// RV64:        maccsu.w00
+int64_t test_maccsu_w00_i64(int64_t rd, int32x2_t a, uint32x2_t b) {
+  return __riscv_maccsu_w00_i64(rd, a, b);
+}
+
+// CHECK-LABEL: test_maccsu_w11_i64:
+// RV32:        wmaccsu
+// RV64:        maccsu.w11
+int64_t test_maccsu_w11_i64(int64_t rd, int32x2_t a, uint32x2_t b) {
+  return __riscv_maccsu_w11_i64(rd, a, b);
+}
diff --git a/llvm/include/llvm/IR/IntrinsicsRISCV.td b/llvm/include/llvm/IR/IntrinsicsRISCV.td
index e4ec9e9beb5ee..e70caa8b09700 100644
--- a/llvm/include/llvm/IR/IntrinsicsRISCV.td
+++ b/llvm/include/llvm/IR/IntrinsicsRISCV.td
@@ -2186,6 +2186,36 @@ class RVPBinaryIntrinsic
   def int_riscv_mulsu_00 : RVPScalarMulPartsIntrinsic;
   def int_riscv_mulsu_11 : RVPScalarMulPartsIntrinsic;
 
+  // Packed Multiply Parts Accumulate.
+  class RVPPackedMulPartsAccIntrinsic
+      : DefaultAttrsIntrinsic<[llvm_anyvector_ty],
+                              [LLVMMatchType<0>,
+                               LLVMSubdivide2VectorType<0>,
+                               LLVMSubdivide2VectorType<0>],
+                              [IntrNoMem, IntrSpeculatable]>;
+  def int_riscv_pmacc_00   : RVPPackedMulPartsAccIntrinsic;
+  def int_riscv_pmacc_01   : RVPPackedMulPartsAccIntrinsic;
+  def int_riscv_pmacc_11   : RVPPackedMulPartsAccIntrinsic;
+  def int_riscv_pmaccu_00  : RVPPackedMulPartsAccIntrinsic;
+  def int_riscv_pmaccu_01  : RVPPackedMulPartsAccIntrinsic;
+  def int_riscv_pmaccu_11  : RVPPackedMulPartsAccIntrinsic;
+  def int_riscv_pmaccsu_00 : RVPPackedMulPartsAccIntrinsic;
+  def int_riscv_pmaccsu_11 : RVPPackedMulPartsAccIntrinsic;
+
+  class RVPScalarMulPartsAccIntrinsic
+      : DefaultAttrsIntrinsic<[llvm_anyint_ty],
+                              [LLVMMatchType<0>, llvm_anyvector_ty,
+                               LLVMMatchType<1>],
+                              [IntrNoMem, IntrSpeculatable]>;
+  def int_riscv_macc_00   : RVPScalarMulPartsAccIntrinsic;
+  def int_riscv_macc_01   : RVPScalarMulPartsAccIntrinsic;
+  def int_riscv_macc_11   : RVPScalarMulPartsAccIntrinsic;
+  def int_riscv_maccu_00  : RVPScalarMulPartsAccIntrinsic;
+  def int_riscv_maccu_01  : RVPScalarMulPartsAccIntrinsic;
+  def int_riscv_maccu_11  : RVPScalarMulPartsAccIntrinsic;
+  def int_riscv_maccsu_00 : RVPScalarMulPartsAccIntrinsic;
+  def int_riscv_maccsu_11 : RVPScalarMulPartsAccIntrinsic;
+
   // Packed Absolute Difference Sum.
   def int_riscv_pabdsumu
       : DefaultAttrsIntrinsic<[llvm_anyint_ty],
diff --git a/llvm/lib/Target/RISCV/RISCVISelDAGToDAG.cpp b/llvm/lib/Target/RISCV/RISCVISelDAGToDAG.cpp
index d263d0320839b..c1c7031a17e71 100644
--- a/llvm/lib/Target/RISCV/RISCVISelDAGToDAG.cpp
+++ b/llvm/lib/Target/RISCV/RISCVISelDAGToDAG.cpp
@@ -2073,13 +2073,35 @@ void RISCVDAGToDAGISel::Select(SDNode *Node) {
     return;
   }
   case RISCVISD::MQWACC:
-  case RISCVISD::MQRWACC: {
+  case RISCVISD::MQRWACC:
+  case RISCVISD::WMACC:
+  case RISCVISD::WMACCU:
+  case RISCVISD::WMACCSU: {
     assert(!Subtarget->is64Bit() && Subtarget->hasStdExtP() &&
            "Unexpected opcode");
 
     SDValue Op0 = buildGPRPair(CurDAG, DL, MVT::Untyped, Node->getOperand(0),
                                Node->getOperand(1));
-    unsigned Opc = Opcode == RISCVISD::MQRWACC ? RISCV::MQRWACC : RISCV::MQWACC;
+    unsigned Opc;
+    switch (Opcode) {
+    default:
+      llvm_unreachable("Unexpected opcode");
+    case RISCVISD::MQWACC:
+      Opc = RISCV::MQWACC;
+      break;
+    case RISCVISD::MQRWACC:
+      Opc = RISCV::MQRWACC;
+      break;
+    case RISCVISD::WMACC:
+      Opc = RISCV::WMACC;
+      break;
+    case RISCVISD::WMACCU:
+      Opc = RISCV::WMACCU;
+      break;
+    case RISCVISD::WMACCSU:
+      Opc = RISCV::WMACCSU;
+      break;
+    }
     MachineSDNode *New = CurDAG->getMachineNode(
         Opc, DL, MVT::Untyped, Op0, Node->getOperand(2), Node->getOperand(3));
     auto [Lo, Hi] = extractGPRPair(CurDAG, DL, SDValue(New, 0));
diff --git a/llvm/lib/Target/RISCV/RISCVISelLowering.cpp b/llvm/lib/Target/RISCV/RISCVISelLowering.cpp
index 5f8ad5da42da1..e2d91fc4c4cfa 100644
--- a/llvm/lib/Target/RISCV/RISCVISelLowering.cpp
+++ b/llvm/lib/Target/RISCV/RISCVISelLowering.cpp
@@ -12548,6 +12548,38 @@ static Intrinsic::ID getRVPScalarMulPartsIntrinsic(unsigned IntNo) {
   }
 }
 
+/// Return the multiply-parts accumulate node for \p IntNo.
+static unsigned getRVPMulAccHalvesOpcode(unsigned IntNo) {
+  switch (IntNo) {
+  default:
+    llvm_unreachable("Unexpected RISC-V multiply-parts accumulate intrinsic");
+  case Intrinsic::riscv_pmacc_00:
+  case Intrinsic::riscv_macc_00:
+    return RISCVISD::PMACC_HALVES_00;
+  case Intrinsic::riscv_pmacc_01:
+  case Intrinsic::riscv_macc_01:
+    return RISCVISD::PMACC_HALVES_01;
+  case Intrinsic::riscv_pmacc_11:
+  case Intrinsic::riscv_macc_11:
+    return RISCVISD::PMACC_HALVES_11;
+  case Intrinsic::riscv_pmaccu_00:
+  case Intrinsic::riscv_maccu_00:
+    return RISCVISD::PMACCU_HALVES_00;
+  case Intrinsic::riscv_pmaccu_01:
+  case Intrinsic::riscv_maccu_01:
+    return RISCVISD::PMACCU_HALVES_01;
+  case Intrinsic::riscv_pmaccu_11:
+  case Intrinsic::riscv_maccu_11:
+    return RISCVISD::PMACCU_HALVES_11;
+  case Intrinsic::riscv_pmaccsu_00:
+  case Intrinsic::riscv_maccsu_00:
+    return RISCVISD::PMACCSU_HALVES_00;
+  case Intrinsic::riscv_pmaccsu_11:
+  case Intrinsic::riscv_maccsu_11:
+    return RISCVISD::PMACCSU_HALVES_11;
+  }
+}
+
 /// Return {opcode, rs1 lane, rs2 lane} for the word form of \p IntNo.
 static std::tuple<unsigned, unsigned, unsigned>
 getRVPWordMulPartsOpcodeAndLanes(unsigned IntNo) {
@@ -12573,6 +12605,32 @@ getRVPWordMulPartsOpcodeAndLanes(unsigned IntNo) {
   }
 }
 
+/// Return {opcode, rs1 lane, rs2 lane} for the word form of accumulate
+/// intrinsic \p IntNo.
+static std::tuple<unsigned, unsigned, unsigned>
+getRVPWordMulPartsAccOpcodeAndLanes(unsigned IntNo) {
+  switch (IntNo) {
+  default:
+    llvm_unreachable("Unexpected RISC-V multiply-parts accumulate intrinsic");
+  case Intrinsic::riscv_macc_00:
+    return {RISCVISD::WMACC, 0, 0};
+  case Intrinsic::riscv_macc_01:
+    return {RISCVISD::WMACC, 0, 1};
+  case Intrinsic::riscv_macc_11:
+    return {RISCVISD::WMACC, 1, 1};
+  case Intrinsic::riscv_maccu_00:
+    return {RISCVISD::WMACCU, 0, 0};
+  case Intrinsic::riscv_maccu_01:
+    return {RISCVISD::WMACCU, 0, 1};
+  case Intrinsic::riscv_maccu_11:
+    return {RISCVISD::WMACCU, 1, 1};
+  case Intrinsic::riscv_maccsu_00:
+    return {RISCVISD::WMACCSU, 0, 0};
+  case Intrinsic::riscv_maccsu_11:
+    return {RISCVISD::WMACCSU, 1, 1};
+  }
+}
+
 SDValue RISCVTargetLowering::LowerINTRINSIC_WO_CHAIN(SDValue Op,
                                                      SelectionDAG &DAG) const {
   unsigned IntNo = Op.getConstantOperandVal(0);
@@ -12635,6 +12693,42 @@ SDValue RISCVTargetLowering::LowerINTRINSIC_WO_CHAIN(SDValue Op,
     SDValue Hi = DAG.getNode(Opc, DL, HalfVT, Rs1Hi, Rs2Hi);
     return DAG.getNode(ISD::CONCAT_VECTORS, DL, VT, Lo, Hi);
   }
+  case Intrinsic::riscv_pmacc_00:
+  case Intrinsic::riscv_pmacc_01:
+  case Intrinsic::riscv_pmacc_11:
+  case Intrinsic::riscv_pmaccu_00:
+  case Intrinsic::riscv_pmaccu_01:
+  case Intrinsic::riscv_pmaccu_11:
+  case Intrinsic::riscv_pmaccsu_00:
+  case Intrinsic::riscv_pmaccsu_11:
+  case Intrinsic::riscv_macc_00:
+  case Intrinsic::riscv_macc_01:
+  case Intrinsic::riscv_macc_11:
+  case Intrinsic::riscv_maccu_00:
+  case Intrinsic::riscv_maccu_01:
+  case Intrinsic::riscv_maccu_11:
+  case Intrinsic::riscv_maccsu_00:
+  case Intrinsic::riscv_maccsu_11: {
+    MVT VT = Op.getSimpleValueType();
+    SDValue Rd = Op.getOperand(1);
+    SDValue Rs1 = Op.getOperand(2);
+    SDValue Rs2 = Op.getOperand(3);
+    unsigned Opc = getRVPMulAccHalvesOpcode(IntNo);
+    if (VT != MVT::v2i32 || !Subtarget.isPExtPackedDoubleType(VT))
+      return DAG.getNode(Opc, DL, VT, Rd, Rs1, Rs2);
+
+    // On RV32 a 64-bit result lives in a GPR pair; accumulate each half with
+    // the 32-bit form of the same product.
+    auto [Rs1Lo, Rs1Hi] = DAG.SplitVector(Rs1, DL);
+    auto [Rs2Lo, Rs2Hi] = DAG.SplitVector(Rs2, DL);
+    SDValue Lo =
+        DAG.getNode(Opc, DL, MVT::i32,
+                    DAG.getExtractVectorElt(DL, MVT::i32, Rd, 0), Rs1Lo, Rs2Lo);
+    SDValue Hi =
+        DAG.getNode(Opc, DL, MVT::i32,
+                    DAG.getExtractVectorElt(DL, MVT::i32, Rd, 1), Rs1Hi, Rs2Hi);
+    return DAG.getNode(ISD::BUILD_VECTOR, DL, VT, Lo, Hi);
+  }
   case Intrinsic::riscv_pas:
   case Intrinsic::riscv_psa:
   case Intrinsic::riscv_psas:
@@ -17263,6 +17357,51 @@ void RISCVTargetLowering::ReplaceNodeResults(SDNode *N,
       }
       reportFatalUsageError("unsupported llvm.riscv multiply-parts intrinsic");
     }
+    case Intrinsic::riscv_macc_00:
+    case Intrinsic::riscv_macc_01:
+    case Intrinsic::riscv_macc_11:
+    case Intrinsic::riscv_maccu_00:
+    case Intrinsic::riscv_maccu_01:
+    case Intrinsic::riscv_maccu_11:
+    case Intrinsic::riscv_maccsu_00:
+    case Intrinsic::riscv_maccsu_11: {
+      // macc.hXX exists only on RV32 and macc.wXX only on RV64; the other XLEN
+      // has to build the product here.
+      MVT VT = N->getSimpleValueType(0);
+      MVT SrcVT = N->getOperand(2).getSimpleValueType();
+      if (Subtarget.hasStdExtP() && Subtarget.is64Bit() && VT == MVT::i32 &&
+          SrcVT == MVT::v2i16) {
+        // Accumulate into the first element of the packed product.
+        SDValue Undef = DAG.getUNDEF(SrcVT);
+        SDValue Rd = DAG.getNode(ISD::SCALAR_TO_VECTOR, DL, MVT::v2i32,
+                                 N->getOperand(1));
+        SDValue Rs1 = DAG.getNode(ISD::CONCAT_VECTORS, DL, MVT::v4i16,
+                                  N->getOperand(2), Undef);
+        SDValue Rs2 = DAG.getNode(ISD::CONCAT_VECTORS, DL, MVT::v4i16,
+                                  N->getOperand(3), Undef);
+        SDValue Res = DAG.getNode(getRVPMulAccHalvesOpcode(IntNo), DL,
+                                  MVT::v2i32, Rd, Rs1, Rs2);
+        Results.push_back(DAG.getExtractVectorElt(DL, MVT::i32, Res, 0));
+        return;
+      }
+      if (Subtarget.hasStdExtP() && !Subtarget.is64Bit() && VT == MVT::i64 &&
+          SrcVT == MVT::v2i32) {
+        auto [Opc, Rs1Lane, Rs2Lane] =
+            getRVPWordMulPartsAccOpcodeAndLanes(IntNo);
+        auto [RdLo, RdHi] =
+            DAG.SplitScalar(N->getOperand(1), DL, MVT::i32, MVT::i32);
+        SDValue Rs1 =
+            DAG.getExtractVectorElt(DL, MVT::i32, N->getOperand(2), Rs1Lane);
+        SDValue Rs2 =
+            DAG.getExtractVectorElt(DL, MVT::i32, N->getOperand(3), Rs2Lane);
+        SDValue Res = DAG.getNode(Opc, DL, DAG.getVTList(MVT::i32, MVT::i32),
+                                  RdLo, RdHi, Rs1, Rs2);
+        Results.push_back(
+            DAG.getNode(ISD::BUILD_PAIR, DL, MVT::i64, Res, Res.getValue(1)));
+        return;
+      }
+      reportFatalUsageError("unsupported llvm.riscv multiply-parts intrinsic");
+    }
     case Intrinsic::riscv_paadd:
     case Intrinsic::riscv_paaddu:
     case Intrinsic::riscv_pasub:
diff --git a/llvm/lib/Target/RISCV/RISCVInstrInfoP.td b/llvm/lib/Target/RISCV/RISCVInstrInfoP.td
index 3637ee98df990..dfdfb5435c3c3 100644
--- a/llvm/lib/Target/RISCV/RISCVInstrInfoP.td
+++ b/llvm/lib/Target/RISCV/RISCVInstrInfoP.td
@@ -1791,6 +1791,11 @@ class PatMulParts<SDPatternOperator OpNode, RVInst Inst, ValueType ResultVT,
                   ValueType SourceVT>
     : Pat<(ResultVT (OpNode (SourceVT GPR:$rs1), (SourceVT GPR:$rs2))),
           (Inst GPR:$rs1, GPR:$rs2)>;
+class PatMulPartsAcc<SDPatternOperator OpNode, RVInst Inst, ValueType ResultVT,
+                     ValueType SourceVT>
+    : Pat<(ResultVT (OpNode (ResultVT GPR:$rd), (SourceVT GPR:$rs1),
+                            (SourceVT GPR:$rs2))),
+          (Inst GPR:$rd, GPR:$rs1, GPR:$rs2)>;
 
 class PatGprGprGpr<SDPatternOperator OpNode, RVInst Inst, ValueType VT>
     : Pat<(VT (OpNode (VT GPR:$rd), (VT GPR:$rs1), (VT GPR:$rs2))),
@@ -1853,6 +1858,11 @@ def riscv_wsub  : RVSDNode<"WSUB",  SDTIntBinHiLoOp>;
 
 def riscv_wmulsu : RVSDNode<"WMULSU", SDTIntBinHiLoOp>;
 
+// Widening multiply-accumulate into a GPR pair: rd_p = rd_p + rs1 * rs2.
+def riscv_wmacc   : RVSDNode<"WMACC",   SDT_RISCVWideningAddSubAccumulate>;
+def riscv_wmaccu  : RVSDNode<"WMACCU",  SDT_RISCVWideningAddSubAccumulate>;
+def riscv_wmaccsu : RVSDNode<"WMACCSU", SDT_RISCVWideningAddSubAccumulate>;
+
 def SDT_RISCVPackedWideningMul : SDTypeProfile<1, 2, [SDTCisVec<0>,
                                                       SDTCisSameAs<1, 2>,
                                                       SDTCisOpSmallerThanOp<1, 0>,
@@ -1917,6 +1927,29 @@ def riscv_pm2addu_h
     : RVSDNode<"PM2ADDU_H", SDT_RISCVPM2Halfword, [SDNPCommutative]>;
 def riscv_pm2sub_h : RVSDNode<"PM2SUB_H", SDT_RISCVPM2Halfword>;
 
+def SDT_RISCVWideningMulAccByHalves
+    : SDTypeProfile<1, 3, [SDTCisSameAs<0, 1>,
+                           SDTCisSameAs<2, 3>,
+                           SDTCisSameSizeAs<0, 2>]>;
+def riscv_pmacc_halves_00
+    : RVSDNode<"PMACC_HALVES_00", SDT_RISCVWideningMulAccByHalves>;
+def riscv_pmacc_halves_01
+    : RVSDNode<"PMACC_HALVES_01", SDT_RISCVWideningMulAccByHalves>;
+def riscv_pmacc_halves_11
+    : RVSDNode<"PMACC_HALVES_11", SDT_RISCVWideningMulAccByHalves>;
+
+def riscv_pmaccu_halves_00
+    : RVSDNode<"PMACCU_HALVES_00", SDT_RISCVWideningMulAccByHalves>;
+def riscv_pmaccu_halves_01
+    : RVSDNode<"PMACCU_HALVES_01", SDT_RISCVWideningMulAccByHalves>;
+def riscv_pmaccu_halves_11
+    : RVSDNode<"PMACCU_HALVES_11", SDT_RISCVWideningMulAccByHalves>;
+
+def riscv_pmaccsu_halves_00
+    : RVSDNode<"PMACCSU_HALVES_00", SDT_RISCVWideningMulAccByHalves>;
+def riscv_pmaccsu_halves_11
+    : RVSDNode<"PMACCSU_HALVES_11", SDT_RISCVWideningMulAccByHalves>;
+
 def SDT_RISCVWideningShiftLeft : SDTypeProfile<2, 2, [SDTCisVT<0, i32>,
                                                       SDTCisSameAs<0, 1>,
                                                       SDTCisSameAs<0, 2>,
@@ -2382,6 +2415,16 @@ let Predicates = [HasStdExtP] in {
             (PMULSU_H_B11 GPR:$rs1, GPR:$rs2)>;
 
 let append Predicates = [IsRV32] in {
+  // Scalar halfword multiply-parts accumulate patterns.
+  def : PatMulPartsAcc<riscv_pmacc_halves_00, MACC_H00, i32, v2i16>;
+  def : PatMulPartsAcc<riscv_pmacc_halves_01, MACC_H01, i32, v2i16>;
+  def : PatMulPartsAcc<riscv_pmacc_halves_11, MACC_H11, i32, v2i16>;
+  def : PatMulPartsAcc<riscv_pmaccu_halves_00, MACCU_H00, i32, v2i16>;
+  def : PatMulPartsAcc<riscv_pmaccu_halves_01, MACCU_H01, i32, v2i16>;
+  def : PatMulPartsAcc<riscv_pmaccu_halves_11, MACCU_H11, i32, v2i16>;
+  def : PatMulPartsAcc<riscv_pmaccsu_halves_00, MACCSU_H00, i32, v2i16>;
+  def : PatMulPartsAcc<riscv_pmaccsu_halves_11, MACCSU_H11, i32, v2i16>;
+
   // Scalar halfword multiply-parts patterns.
   def : PatMulParts<int_riscv_mul_00, MUL_H00, i32, v2i16>;
   def : PatMulParts<int_riscv_mul_01, MUL_H01, i32, v2i16>;
@@ -3075,6 +3118,26 @@ let append Predicates = [IsRV64] in {
   def : PatGprGpr<riscv_asub, PASUB_W, v2i32>;
   def : PatGprGpr<riscv_asubu, PASUBU_W, v2i32>;
 
+  // Packed halfword multiply-parts accumulate patterns.
+  def : PatMulPartsAcc<riscv_pmacc_halves_00, PMACC_W_H00, v2i32, v4i16>;
+  def : PatMulPartsAcc<riscv_pmacc_halves_01, PMACC_W_H01, v2i32, v4i16>;
+  def : PatMulPartsAcc<riscv_pmacc_halves_11, PMACC_W_H11, v2i32, v4i16>;
+  def : PatMulPartsAcc<riscv_pmaccu_halves_00, PMACCU_W_H00, v2i32, v4i16>;
+  def : PatMulPartsAcc<riscv_pmaccu_halves_01, PMACCU_W_H01, v2i32, v4i16>;
+  def : PatMulPartsAcc<riscv_pmaccu_halves_11, PMACCU_W_H11, v2i32, v4i16>;
+  def : PatMulPartsAcc<riscv_pmaccsu_halves_00, PMACCSU_W_H00, v2i32, v4i16>;
+  def : PatMulPartsAcc<riscv_pmaccsu_halves_11, PMACCSU_W_H11, v2i32, v4i16>;
+
+  // Scalar word multiply-parts accumulate patterns.
+  def : PatMulPartsAcc<riscv_pmacc_halves_00, MACC_W00, i64, v2i32>;
+  def : PatMulPartsAcc<riscv_pmacc_halves_01, MACC_W01, i64, v2i32>;
+  def : PatMulPartsAcc<riscv_pmacc_halves_11, MACC_W11, i64, v2i32>;
+  def : PatMulPartsAcc<riscv_pmaccu_halves_00, MACCU_W00, i64, v2i32>;
+  def : PatMulPartsAcc<riscv_pmaccu_halves_01, MACCU_W01, i64, v2i32>;
+  def : PatMulPartsAcc<riscv_pmaccu_halves_11, MACCU_W11, i64, v2i32>;
+  def : PatMulPartsAcc<riscv_pmaccsu_halves_00, MACCSU_W00, i64, v2i32>;
+  def : PatMulPartsAcc<riscv_pmaccsu_halves_11, MACCSU_W11, i64, v2i32>;
+
   // Scalar word multiply-parts patterns.
   def : PatMulParts<int_riscv_mul_00, MUL_W00, i64, v2i32>;
   def : PatMulParts<int_riscv_mul_01, MUL_W01, i64, v2i32>;
diff --git a/llvm/test/CodeGen/RISCV/rvp-simd-32.ll b/llvm/test/CodeGen/RISCV/rvp-simd-32.ll
index c11d9716ec053..e9e04eb6b5ebe 100644
--- a/llvm/test/CodeGen/RISCV/rvp-simd-32.ll
+++ b/llvm/test/CodeGen/RISCV/rvp-simd-32.ll
@@ -3705,3 +3705,125 @@ define i32 @test_pm2addsu_v2i16_i32(<2 x i16> %a, <2 x i16> %b) {
   %r = call i32 @llvm.riscv.pm2addsu.i32.v2i16(<2 x i16> %a, <2 x i16> %b)
   ret i32 %r
 }
+
+; Packed Multiply Parts Accumulate.
+declare i32 @llvm.riscv.macc.00.i32.v2i16(i32, <2 x i16>, <2 x i16>)
+declare i32 @llvm.riscv.macc.01.i32.v2i16(i32, <2 x i16>, <2 x i16>)
+declare i32 @llvm.riscv.macc.11.i32.v2i16(i32, <2 x i16>, <2 x i16>)
+declare i32 @llvm.riscv.maccu.00.i32.v2i16(i32, <2 x i16>, <2 x i16>)
+declare i32 @llvm.riscv.maccu.01.i32.v2i16(i32, <2 x i16>, <2 x i16>)
+declare i32 @llvm.riscv.maccu.11.i32.v2i16(i32, <2 x i16>, <2 x i16>)
+declare i32 @llvm.riscv.maccsu.00.i32.v2i16(i32, <2 x i16>, <2 x i16>)
+declare i32 @llvm.riscv.maccsu.11.i32.v2i16(i32, <2 x i16>, <2 x i16>)
+
+define i32 @test_macc_h00_i32(i32 %rd, <2 x i16> %a, <2 x i16> %b) {
+; RV32-LABEL: test_macc_h00_i32:
+; RV32:       # %bb.0:
+; RV32-NEXT:    macc.h00 a0, a1, a2
+; RV32-NEXT:    ret
+;
+; RV64-LABEL: test_macc_h00_i32:
+; RV64:       # %bb.0:
+; RV64-NEXT:    pmacc.w.h00 a0, a1, a2
+; RV64-NEXT:    ret
+  %r = call i32 @llvm.riscv.macc.00.i32.v2i16(i32 %rd, <2 x i16> %a, <2 x i16> %b)
+  ret i32 %r
+}
+
+define i32 @test_macc_h01_i32(i32 %rd, <2 x i16> %a, <2 x i16> %b) {
+; RV32-LABEL: test_macc_h01_i32:
+; RV32:       # %bb.0:
+; RV32-NEXT:    macc.h01 a0, a1, a2
+; RV32-NEXT:    ret
+;
+; RV64-LABEL: test_macc_h01_i32:
+; RV64:       # %bb.0:
+; RV64-NEXT:    pmacc.w.h01 a0, a1, a2
+; RV64-NEXT:    ret
+  %r = call i32 @llvm.riscv.macc.01.i32.v2i16(i32 %rd, <2 x i16> %a, <2 x i16> %b)
+  ret i32 %r
+}
+
+define i32 @test_macc_h11_i32(i32 %rd, <2 x i16> %a, <2 x i16> %b) {
+; RV32-LABEL: test_macc_h11_i32:
+; RV32:       # %bb.0:
+; RV32-NEXT:    macc.h11 a0, a1, a2
+; RV32-NEXT:    ret
+;
+; RV64-LABEL: test_macc_h11_i32:
+; RV64:       # %bb.0:
+; RV64-NEXT:    pmacc.w.h11 a0, a1, a2
+; RV64-NEXT:    ret
+  %r = call i32 @llvm.riscv.macc.11.i32.v2i16(i32 %rd, <2 x i16> %a, <2 x i16> %b)
+  ret i32 %r
+}
+
+define i32 @test_maccu_h00_i32(i32 %rd, <2 x i16> %a, <2 x i16> %b) {
+; RV32-LABEL: test_maccu_h00_i32:
+; RV32:       # %bb.0:
+; RV32-NEXT:    maccu.h00 a0, a1, a2
+; RV32-NEXT:    ret
+;
+; RV64-LABEL: test_maccu_h00_i32:
+; RV64:       # %bb.0:
+; RV64-NEXT:    pmaccu.w.h00 a0, a1, a2
+; RV64-NEXT:    ret
+  %r = call i32 @llvm.riscv.maccu.00.i32.v2i16(i32 %rd, <2 x i16> %a, <2 x i16> %b)
+  ret i32 %r
+}
+
+define i32 @test_maccu_h01_i32(i32 %rd, <2 x i16> %a, <2 x i16> %b) {
+; RV32-LABEL: test_maccu_h01_i32:
+; RV32:       # %bb.0:
+; RV32-NEXT:    maccu.h01 a0, a1, a2
+; RV32-NEXT:    ret
+;
+; RV64-LABEL: test_maccu_h01_i32:
+; RV64:       # %bb.0:
+; RV64-NEXT:    pmaccu.w.h01 a0, a1, a2
+; RV64-NEXT:    ret
+  %r = call i32 @llvm.riscv.maccu.01.i32.v2i16(i32 %rd, <2 x i16> %a, <2 x i16> %b)
+  ret i32 %r
+}
+
+define i32 @test_maccu_h11_i32(i32 %rd, <2 x i16> %a, <2 x i16> %b) {
+; RV32-LABEL: test_maccu_h11_i32:
+; RV32:       # %bb.0:
+; RV32-NEXT:    maccu.h11 a0, a1, a2
+; RV32-NEXT:    ret
+;
+; RV64-LABEL: test_maccu_h11_i32:
+; RV64:       # %bb.0:
+; RV64-NEXT:    pmaccu.w.h11 a0, a1, a2
+; RV64-NEXT:    ret
+  %r = call i32 @llvm.riscv.maccu.11.i32.v2i16(i32 %rd, <2 x i16> %a, <2 x i16> %b)
+  ret i32 %r
+}
+
+define i32 @test_maccsu_h00_i32(i32 %rd, <2 x i16> %a, <2 x i16> %b) {
+; RV32-LABEL: test_maccsu_h00_i32:
+; RV32:       # %bb.0:
+; RV32-NEXT:    maccsu.h00 a0, a1, a2
+; RV32-NEXT:    ret
+;
+; RV64-LABEL: test_maccsu_h00_i32:
+; RV64:       # %bb.0:
+; RV64-NEXT:    pmaccsu.w.h00 a0, a1, a2
+; RV64-NEXT:    ret
+  %r = call i32 @llvm.riscv.maccsu.00.i32.v2i16(i32 %rd, <2 x i16> %a, <2 x i16> %b)
+  ret i32 %r
+}
+
+define i32 @test_maccsu_h11_i32(i32 %rd, <2 x i16> %a, <2 x i16> %b) {
+; RV32-LABEL: test_maccsu_h11_i32:
+; RV32:       # %bb.0:
+; RV32-NEXT:    maccsu.h11 a0, a1, a2
+; RV32-NEXT:    ret
+;
+; RV64-LABEL: test_maccsu_h11_i32:
+; RV64:       # %bb.0:
+; RV64-NEXT:    pmaccsu.w.h11 a0, a1, a2
+; RV64-NEXT:    ret
+  %r = call i32 @llvm.riscv.maccsu.11.i32.v2i16(i32 %rd, <2 x i16> %a, <2 x i16> %b)
+  ret i32 %r
+}
diff --git a/llvm/test/CodeGen/RISCV/rvp-simd-64.ll b/llvm/test/CodeGen/RISCV/rvp-simd-64.ll
index fd9cf65c2ef53..43107137fe1fa 100644
--- a/llvm/test/CodeGen/RISCV/rvp-simd-64.ll
+++ b/llvm/test/CodeGen/RISCV/rvp-simd-64.ll
@@ -7866,3 +7866,253 @@ define i64 @test_pm4addsu_v4i16_i64(<4 x i16> %a, <4 x i16> %b) {
   %r = call i64 @llvm.riscv.pm4addsu.i64.v4i16(<4 x i16> %a, <4 x i16> %b)
   ret i64 %r
 }
+
+; Packed Multiply Parts Accumulate.
+declare <2 x i32> @llvm.riscv.pmacc.00.v2i32(<2 x i32>, <4 x i16>, <4 x i16>)
+declare <2 x i32> @llvm.riscv.pmacc.01.v2i32(<2 x i32>, <4 x i16>, <4 x i16>)
+declare <2 x i32> @llvm.riscv.pmacc.11.v2i32(<2 x i32>, <4 x i16>, <4 x i16>)
+declare <2 x i32> @llvm.riscv.pmaccu.00.v2i32(<2 x i32>, <4 x i16>, <4 x i16>)
+declare <2 x i32> @llvm.riscv.pmaccu.01.v2i32(<2 x i32>, <4 x i16>, <4 x i16>)
+declare <2 x i32> @llvm.riscv.pmaccu.11.v2i32(<2 x i32>, <4 x i16>, <4 x i16>)
+declare <2 x i32> @llvm.riscv.pmaccsu.00.v2i32(<2 x i32>, <4 x i16>, <4 x i16>)
+declare <2 x i32> @llvm.riscv.pmaccsu.11.v2i32(<2 x i32>, <4 x i16>, <4 x i16>)
+declare i64 @llvm.riscv.macc.00.i64.v2i32(i64, <2 x i32>, <2 x i32>)
+declare i64 @llvm.riscv.macc.01.i64.v2i32(i64, <2 x i32>, <2 x i32>)
+declare i64 @llvm.riscv.macc.11.i64.v2i32(i64, <2 x i32>, <2 x i32>)
+declare i64 @llvm.riscv.maccu.00.i64.v2i32(i64, <2 x i32>, <2 x i32>)
+declare i64 @llvm.riscv.maccu.01.i64.v2i32(i64, <2 x i32>, <2 x i32>)
+declare i64 @llvm.riscv.maccu.11.i64.v2i32(i64, <2 x i32>, <2 x i32>)
+declare i64 @llvm.riscv.maccsu.00.i64.v2i32(i64, <2 x i32>, <2 x i32>)
+declare i64 @llvm.riscv.maccsu.11.i64.v2i32(i64, <2 x i32>, <2 x i32>)
+
+define <2 x i32> @test_pmacc_h00_v2i32(<2 x i32> %rd, <4 x i16> %a, <4 x i16> %b) {
+; RV32-LABEL: test_pmacc_h00_v2i32:
+; RV32:       # %bb.0:
+; RV32-NEXT:    macc.h00 a1, a3, a5
+; RV32-NEXT:    macc.h00 a0, a2, a4
+; RV32-NEXT:    ret
+;
+; RV64-LABEL: test_pmacc_h00_v2i32:
+; RV64:       # %bb.0:
+; RV64-NEXT:    pmacc.w.h00 a0, a1, a2
+; RV64-NEXT:    ret
+  %r = call <2 x i32> @llvm.riscv.pmacc.00.v2i32(<2 x i32> %rd, <4 x i16> %a, <4 x i16> %b)
+  ret <2 x i32> %r
+}
+
+define <2 x i32> @test_pmacc_h01_v2i32(<2 x i32> %rd, <4 x i16> %a, <4 x i16> %b) {
+; RV32-LABEL: test_pmacc_h01_v2i32:
+; RV32:       # %bb.0:
+; RV32-NEXT:    macc.h01 a1, a3, a5
+; RV32-NEXT:    macc.h01 a0, a2, a4
+; RV32-NEXT:    ret
+;
+; RV64-LABEL: test_pmacc_h01_v2i32:
+; RV64:       # %bb.0:
+; RV64-NEXT:    pmacc.w.h01 a0, a1, a2
+; RV64-NEXT:    ret
+  %r = call <2 x i32> @llvm.riscv.pmacc.01.v2i32(<2 x i32> %rd, <4 x i16> %a, <4 x i16> %b)
+  ret <2 x i32> %r
+}
+
+define <2 x i32> @test_pmacc_h11_v2i32(<2 x i32> %rd, <4 x i16> %a, <4 x i16> %b) {
+; RV32-LABEL: test_pmacc_h11_v2i32:
+; RV32:       # %bb.0:
+; RV32-NEXT:    macc.h11 a1, a3, a5
+; RV32-NEXT:    macc.h11 a0, a2, a4
+; RV32-NEXT:    ret
+;
+; RV64-LABEL: test_pmacc_h11_v2i32:
+; RV64:       # %bb.0:
+; RV64-NEXT:    pmacc.w.h11 a0, a1, a2
+; RV64-NEXT:    ret
+  %r = call <2 x i32> @llvm.riscv.pmacc.11.v2i32(<2 x i32> %rd, <4 x i16> %a, <4 x i16> %b)
+  ret <2 x i32> %r
+}
+
+define <2 x i32> @test_pmaccu_h00_v2i32(<2 x i32> %rd, <4 x i16> %a, <4 x i16> %b) {
+; RV32-LABEL: test_pmaccu_h00_v2i32:
+; RV32:       # %bb.0:
+; RV32-NEXT:    maccu.h00 a1, a3, a5
+; RV32-NEXT:    maccu.h00 a0, a2, a4
+; RV32-NEXT:    ret
+;
+; RV64-LABEL: test_pmaccu_h00_v2i32:
+; RV64:       # %bb.0:
+; RV64-NEXT:    pmaccu.w.h00 a0, a1, a2
+; RV64-NEXT:    ret
+  %r = call <2 x i32> @llvm.riscv.pmaccu.00.v2i32(<2 x i32> %rd, <4 x i16> %a, <4 x i16> %b)
+  ret <2 x i32> %r
+}
+
+define <2 x i32> @test_pmaccu_h01_v2i32(<2 x i32> %rd, <4 x i16> %a, <4 x i16> %b) {
+; RV32-LABEL: test_pmaccu_h01_v2i32:
+; RV32:       # %bb.0:
+; RV32-NEXT:    maccu.h01 a1, a3, a5
+; RV32-NEXT:    maccu.h01 a0, a2, a4
+; RV32-NEXT:    ret
+;
+; RV64-LABEL: test_pmaccu_h01_v2i32:
+; RV64:       # %bb.0:
+; RV64-NEXT:    pmaccu.w.h01 a0, a1, a2
+; RV64-NEXT:    ret
+  %r = call <2 x i32> @llvm.riscv.pmaccu.01.v2i32(<2 x i32> %rd, <4 x i16> %a, <4 x i16> %b)
+  ret <2 x i32> %r
+}
+
+define <2 x i32> @test_pmaccu_h11_v2i32(<2 x i32> %rd, <4 x i16> %a, <4 x i16> %b) {
+; RV32-LABEL: test_pmaccu_h11_v2i32:
+; RV32:       # %bb.0:
+; RV32-NEXT:    maccu.h11 a1, a3, a5
+; RV32-NEXT:    maccu.h11 a0, a2, a4
+; RV32-NEXT:    ret
+;
+; RV64-LABEL: test_pmaccu_h11_v2i32:
+; RV64:       # %bb.0:
+; RV64-NEXT:    pmaccu.w.h11 a0, a1, a2
+; RV64-NEXT:    ret
+  %r = call <2 x i32> @llvm.riscv.pmaccu.11.v2i32(<2 x i32> %rd, <4 x i16> %a, <4 x i16> %b)
+  ret <2 x i32> %r
+}
+
+define <2 x i32> @test_pmaccsu_h00_v2i32(<2 x i32> %rd, <4 x i16> %a, <4 x i16> %b) {
+; RV32-LABEL: test_pmaccsu_h00_v2i32:
+; RV32:       # %bb.0:
+; RV32-NEXT:    maccsu.h00 a1, a3, a5
+; RV32-NEXT:    maccsu.h00 a0, a2, a4
+; RV32-NEXT:    ret
+;
+; RV64-LABEL: test_pmaccsu_h00_v2i32:
+; RV64:       # %bb.0:
+; RV64-NEXT:    pmaccsu.w.h00 a0, a1, a2
+; RV64-NEXT:    ret
+  %r = call <2 x i32> @llvm.riscv.pmaccsu.00.v2i32(<2 x i32> %rd, <4 x i16> %a, <4 x i16> %b)
+  ret <2 x i32> %r
+}
+
+define <2 x i32> @test_pmaccsu_h11_v2i32(<2 x i32> %rd, <4 x i16> %a, <4 x i16> %b) {
+; RV32-LABEL: test_pmaccsu_h11_v2i32:
+; RV32:       # %bb.0:
+; RV32-NEXT:    maccsu.h11 a1, a3, a5
+; RV32-NEXT:    maccsu.h11 a0, a2, a4
+; RV32-NEXT:    ret
+;
+; RV64-LABEL: test_pmaccsu_h11_v2i32:
+; RV64:       # %bb.0:
+; RV64-NEXT:    pmaccsu.w.h11 a0, a1, a2
+; RV64-NEXT:    ret
+  %r = call <2 x i32> @llvm.riscv.pmaccsu.11.v2i32(<2 x i32> %rd, <4 x i16> %a, <4 x i16> %b)
+  ret <2 x i32> %r
+}
+
+define i64 @test_macc_w00_i64(i64 %rd, <2 x i32> %a, <2 x i32> %b) {
+; RV32-LABEL: test_macc_w00_i64:
+; RV32:       # %bb.0:
+; RV32-NEXT:    wmacc a0, a2, a4
+; RV32-NEXT:    ret
+;
+; RV64-LABEL: test_macc_w00_i64:
+; RV64:       # %bb.0:
+; RV64-NEXT:    macc.w00 a0, a1, a2
+; RV64-NEXT:    ret
+  %r = call i64 @llvm.riscv.macc.00.i64.v2i32(i64 %rd, <2 x i32> %a, <2 x i32> %b)
+  ret i64 %r
+}
+
+define i64 @test_macc_w01_i64(i64 %rd, <2 x i32> %a, <2 x i32> %b) {
+; RV32-LABEL: test_macc_w01_i64:
+; RV32:       # %bb.0:
+; RV32-NEXT:    wmacc a0, a2, a5
+; RV32-NEXT:    ret
+;
+; RV64-LABEL: test_macc_w01_i64:
+; RV64:       # %bb.0:
+; RV64-NEXT:    macc.w01 a0, a1, a2
+; RV64-NEXT:    ret
+  %r = call i64 @llvm.riscv.macc.01.i64.v2i32(i64 %rd, <2 x i32> %a, <2 x i32> %b)
+  ret i64 %r
+}
+
+define i64 @test_macc_w11_i64(i64 %rd, <2 x i32> %a, <2 x i32> %b) {
+; RV32-LABEL: test_macc_w11_i64:
+; RV32:       # %bb.0:
+; RV32-NEXT:    wmacc a0, a3, a5
+; RV32-NEXT:    ret
+;
+; RV64-LABEL: test_macc_w11_i64:
+; RV64:       # %bb.0:
+; RV64-NEXT:    macc.w11 a0, a1, a2
+; RV64-NEXT:    ret
+  %r = call i64 @llvm.riscv.macc.11.i64.v2i32(i64 %rd, <2 x i32> %a, <2 x i32> %b)
+  ret i64 %r
+}
+
+define i64 @test_maccu_w00_i64(i64 %rd, <2 x i32> %a, <2 x i32> %b) {
+; RV32-LABEL: test_maccu_w00_i64:
+; RV32:       # %bb.0:
+; RV32-NEXT:    wmaccu a0, a2, a4
+; RV32-NEXT:    ret
+;
+; RV64-LABEL: test_maccu_w00_i64:
+; RV64:       # %bb.0:
+; RV64-NEXT:    maccu.w00 a0, a1, a2
+; RV64-NEXT:    ret
+  %r = call i64 @llvm.riscv.maccu.00.i64.v2i32(i64 %rd, <2 x i32> %a, <2 x i32> %b)
+  ret i64 %r
+}
+
+define i64 @test_maccu_w01_i64(i64 %rd, <2 x i32> %a, <2 x i32> %b) {
+; RV32-LABEL: test_maccu_w01_i64:
+; RV32:       # %bb.0:
+; RV32-NEXT:    wmaccu a0, a2, a5
+; RV32-NEXT:    ret
+;
+; RV64-LABEL: test_maccu_w01_i64:
+; RV64:       # %bb.0:
+; RV64-NEXT:    maccu.w01 a0, a1, a2
+; RV64-NEXT:    ret
+  %r = call i64 @llvm.riscv.maccu.01.i64.v2i32(i64 %rd, <2 x i32> %a, <2 x i32> %b)
+  ret i64 %r
+}
+
+define i64 @test_maccu_w11_i64(i64 %rd, <2 x i32> %a, <2 x i32> %b) {
+; RV32-LABEL: test_maccu_w11_i64:
+; RV32:       # %bb.0:
+; RV32-NEXT:    wmaccu a0, a3, a5
+; RV32-NEXT:    ret
+;
+; RV64-LABEL: test_maccu_w11_i64:
+; RV64:       # %bb.0:
+; RV64-NEXT:    maccu.w11 a0, a1, a2
+; RV64-NEXT:    ret
+  %r = call i64 @llvm.riscv.maccu.11.i64.v2i32(i64 %rd, <2 x i32> %a, <2 x i32> %b)
+  ret i64 %r
+}
+
+define i64 @test_maccsu_w00_i64(i64 %rd, <2 x i32> %a, <2 x i32> %b) {
+; RV32-LABEL: test_maccsu_w00_i64:
+; RV32:       # %bb.0:
+; RV32-NEXT:    wmaccsu a0, a2, a4
+; RV32-NEXT:    ret
+;
+; RV64-LABEL: test_maccsu_w00_i64:
+; RV64:       # %bb.0:
+; RV64-NEXT:    maccsu.w00 a0, a1, a2
+; RV64-NEXT:    ret
+  %r = call i64 @llvm.riscv.maccsu.00.i64.v2i32(i64 %rd, <2 x i32> %a, <2 x i32> %b)
+  ret i64 %r
+}
+
+define i64 @test_maccsu_w11_i64(i64 %rd, <2 x i32> %a, <2 x i32> %b) {
+; RV32-LABEL: test_maccsu_w11_i64:
+; RV32:       # %bb.0:
+; RV32-NEXT:    wmaccsu a0, a3, a5
+; RV32-NEXT:    ret
+;
+; RV64-LABEL: test_maccsu_w11_i64:
+; RV64:       # %bb.0:
+; RV64-NEXT:    maccsu.w11 a0, a1, a2
+; RV64-NEXT:    ret
+  %r = call i64 @llvm.riscv.maccsu.11.i64.v2i32(i64 %rd, <2 x i32> %a, <2 x i32> %b)
+  ret i64 %r
+}



More information about the cfe-commits mailing list