[clang] [llvm] [RISCV][P-ext] Add packed multiply-parts accumulate intrinsics (PR #222571)
via cfe-commits
cfe-commits at lists.llvm.org
Thu Sep 10 02:49:09 PDT 2026
https://github.com/sihuan created https://github.com/llvm/llvm-project/pull/222571
Add SelectionDAG and intrinsic support for the RISC-V P multiply-parts
accumulate operations, which add the selected products into rd. See also
https://github.com/riscv/riscv-p-spec/blob/master/P-ext-intrinsics.adoc#packed-multiply-parts-accumulate
The intrinsics lower to `PMACC_HALVES_*` nodes rather than to `add` of the
plain products, so the selection does not depend on a fold. On RV32 the word
forms accumulate into a GPR pair with new `WMACC`/`WMACCU`/`WMACCSU` nodes.
Also adds the Clang builtins and the `riscv_packed_simd.h` wrappers.
>From 4003577d908022bcd1d614f5855fbe7cdd30822e Mon Sep 17 00:00:00 2001
From: SiHuaN <liyongtai at iscas.ac.cn>
Date: Fri, 28 Aug 2026 08:15:57 +0000
Subject: [PATCH] [RISCV][P-ext] Add packed multiply-parts accumulate
intrinsics
Add SelectionDAG and intrinsic support for the RISC-V P multiply-parts
accumulate operations, which add the selected products into rd. See also
https://github.com/riscv/riscv-p-spec/blob/master/P-ext-intrinsics.adoc#packed-multiply-parts-accumulate
Also adds the Clang builtins and the `riscv_packed_simd.h` wrappers.
---
clang/include/clang/Basic/BuiltinsRISCV.td | 28 ++
clang/lib/CodeGen/TargetBuiltins/RISCV.cpp | 74 ++++++
clang/lib/Headers/riscv_packed_simd.h | 28 ++
.../riscv_packed_simd.c | 172 ++++++++++++
llvm/include/llvm/IR/IntrinsicsRISCV.td | 30 +++
llvm/lib/Target/RISCV/RISCVISelDAGToDAG.cpp | 26 +-
llvm/lib/Target/RISCV/RISCVISelLowering.cpp | 139 ++++++++++
llvm/lib/Target/RISCV/RISCVInstrInfoP.td | 63 +++++
llvm/test/CodeGen/RISCV/rvp-simd-32.ll | 122 +++++++++
llvm/test/CodeGen/RISCV/rvp-simd-64.ll | 250 ++++++++++++++++++
10 files changed, 930 insertions(+), 2 deletions(-)
diff --git a/clang/include/clang/Basic/BuiltinsRISCV.td b/clang/include/clang/Basic/BuiltinsRISCV.td
index 58184479d3c27..ee840e45a65ba 100644
--- a/clang/include/clang/Basic/BuiltinsRISCV.td
+++ b/clang/include/clang/Basic/BuiltinsRISCV.td
@@ -426,6 +426,34 @@ def mqracc_w00_i64 : RISCVBuiltin<"int64_t(int64_t, _Vector<2, int>, _Vector<2,
def mqracc_w01_i64 : RISCVBuiltin<"int64_t(int64_t, _Vector<2, int>, _Vector<2, int>)">;
def mqracc_w11_i64 : RISCVBuiltin<"int64_t(int64_t, _Vector<2, int>, _Vector<2, int>)">;
+// Packed Multiply Parts Accumulate (32-bit)
+def macc_h00_i32 : RISCVBuiltin<"int(int, _Vector<2, short>, _Vector<2, short>)">;
+def macc_h01_i32 : RISCVBuiltin<"int(int, _Vector<2, short>, _Vector<2, short>)">;
+def macc_h11_i32 : RISCVBuiltin<"int(int, _Vector<2, short>, _Vector<2, short>)">;
+def maccu_h00_u32 : RISCVBuiltin<"unsigned int(unsigned int, _Vector<2, unsigned short>, _Vector<2, unsigned short>)">;
+def maccu_h01_u32 : RISCVBuiltin<"unsigned int(unsigned int, _Vector<2, unsigned short>, _Vector<2, unsigned short>)">;
+def maccu_h11_u32 : RISCVBuiltin<"unsigned int(unsigned int, _Vector<2, unsigned short>, _Vector<2, unsigned short>)">;
+def maccsu_h00_i32 : RISCVBuiltin<"int(int, _Vector<2, short>, _Vector<2, unsigned short>)">;
+def maccsu_h11_i32 : RISCVBuiltin<"int(int, _Vector<2, short>, _Vector<2, unsigned short>)">;
+
+// Packed Multiply Parts Accumulate (64-bit)
+def pmacc_h00_i32x2 : RISCVBuiltin<"_Vector<2, int>(_Vector<2, int>, _Vector<4, short>, _Vector<4, short>)">;
+def pmacc_h01_i32x2 : RISCVBuiltin<"_Vector<2, int>(_Vector<2, int>, _Vector<4, short>, _Vector<4, short>)">;
+def pmacc_h11_i32x2 : RISCVBuiltin<"_Vector<2, int>(_Vector<2, int>, _Vector<4, short>, _Vector<4, short>)">;
+def pmaccu_h00_u32x2 : RISCVBuiltin<"_Vector<2, unsigned int>(_Vector<2, unsigned int>, _Vector<4, unsigned short>, _Vector<4, unsigned short>)">;
+def pmaccu_h01_u32x2 : RISCVBuiltin<"_Vector<2, unsigned int>(_Vector<2, unsigned int>, _Vector<4, unsigned short>, _Vector<4, unsigned short>)">;
+def pmaccu_h11_u32x2 : RISCVBuiltin<"_Vector<2, unsigned int>(_Vector<2, unsigned int>, _Vector<4, unsigned short>, _Vector<4, unsigned short>)">;
+def pmaccsu_h00_i32x2 : RISCVBuiltin<"_Vector<2, int>(_Vector<2, int>, _Vector<4, short>, _Vector<4, unsigned short>)">;
+def pmaccsu_h11_i32x2 : RISCVBuiltin<"_Vector<2, int>(_Vector<2, int>, _Vector<4, short>, _Vector<4, unsigned short>)">;
+def macc_w00_i64 : RISCVBuiltin<"int64_t(int64_t, _Vector<2, int>, _Vector<2, int>)">;
+def macc_w01_i64 : RISCVBuiltin<"int64_t(int64_t, _Vector<2, int>, _Vector<2, int>)">;
+def macc_w11_i64 : RISCVBuiltin<"int64_t(int64_t, _Vector<2, int>, _Vector<2, int>)">;
+def maccu_w00_u64 : RISCVBuiltin<"uint64_t(uint64_t, _Vector<2, unsigned int>, _Vector<2, unsigned int>)">;
+def maccu_w01_u64 : RISCVBuiltin<"uint64_t(uint64_t, _Vector<2, unsigned int>, _Vector<2, unsigned int>)">;
+def maccu_w11_u64 : RISCVBuiltin<"uint64_t(uint64_t, _Vector<2, unsigned int>, _Vector<2, unsigned int>)">;
+def maccsu_w00_i64 : RISCVBuiltin<"int64_t(int64_t, _Vector<2, int>, _Vector<2, unsigned int>)">;
+def maccsu_w11_i64 : RISCVBuiltin<"int64_t(int64_t, _Vector<2, int>, _Vector<2, unsigned int>)">;
+
// Packed Sign and Zero Extend (32-bit)
def psext_b_i16x2 : RISCVBuiltin<"_Vector<2, short>(_Vector<2, short>)">;
def pzext_b_u16x2 : RISCVBuiltin<"_Vector<2, unsigned short>(_Vector<2, unsigned short>)">;
diff --git a/clang/lib/CodeGen/TargetBuiltins/RISCV.cpp b/clang/lib/CodeGen/TargetBuiltins/RISCV.cpp
index 2e9fbf771bed1..f99a05ce673aa 100644
--- a/clang/lib/CodeGen/TargetBuiltins/RISCV.cpp
+++ b/clang/lib/CodeGen/TargetBuiltins/RISCV.cpp
@@ -1938,6 +1938,80 @@ Value *CodeGenFunction::EmitRISCVBuiltinExpr(unsigned BuiltinID,
break;
}
+ // Packed Multiply Parts Accumulate.
+ case RISCV::BI__builtin_riscv_macc_h00_i32:
+ case RISCV::BI__builtin_riscv_pmacc_h00_i32x2:
+ case RISCV::BI__builtin_riscv_macc_w00_i64:
+ case RISCV::BI__builtin_riscv_macc_h01_i32:
+ case RISCV::BI__builtin_riscv_pmacc_h01_i32x2:
+ case RISCV::BI__builtin_riscv_macc_w01_i64:
+ case RISCV::BI__builtin_riscv_macc_h11_i32:
+ case RISCV::BI__builtin_riscv_pmacc_h11_i32x2:
+ case RISCV::BI__builtin_riscv_macc_w11_i64:
+ case RISCV::BI__builtin_riscv_maccu_h00_u32:
+ case RISCV::BI__builtin_riscv_pmaccu_h00_u32x2:
+ case RISCV::BI__builtin_riscv_maccu_w00_u64:
+ case RISCV::BI__builtin_riscv_maccu_h01_u32:
+ case RISCV::BI__builtin_riscv_pmaccu_h01_u32x2:
+ case RISCV::BI__builtin_riscv_maccu_w01_u64:
+ case RISCV::BI__builtin_riscv_maccu_h11_u32:
+ case RISCV::BI__builtin_riscv_pmaccu_h11_u32x2:
+ case RISCV::BI__builtin_riscv_maccu_w11_u64:
+ case RISCV::BI__builtin_riscv_maccsu_h00_i32:
+ case RISCV::BI__builtin_riscv_pmaccsu_h00_i32x2:
+ case RISCV::BI__builtin_riscv_maccsu_w00_i64:
+ case RISCV::BI__builtin_riscv_maccsu_h11_i32:
+ case RISCV::BI__builtin_riscv_pmaccsu_h11_i32x2:
+ case RISCV::BI__builtin_riscv_maccsu_w11_i64: {
+ switch (BuiltinID) {
+ default:
+ llvm_unreachable("unexpected builtin ID");
+ case RISCV::BI__builtin_riscv_macc_h00_i32:
+ case RISCV::BI__builtin_riscv_pmacc_h00_i32x2:
+ case RISCV::BI__builtin_riscv_macc_w00_i64:
+ ID = Intrinsic::riscv_macc_00;
+ break;
+ case RISCV::BI__builtin_riscv_macc_h01_i32:
+ case RISCV::BI__builtin_riscv_pmacc_h01_i32x2:
+ case RISCV::BI__builtin_riscv_macc_w01_i64:
+ ID = Intrinsic::riscv_macc_01;
+ break;
+ case RISCV::BI__builtin_riscv_macc_h11_i32:
+ case RISCV::BI__builtin_riscv_pmacc_h11_i32x2:
+ case RISCV::BI__builtin_riscv_macc_w11_i64:
+ ID = Intrinsic::riscv_macc_11;
+ break;
+ case RISCV::BI__builtin_riscv_maccu_h00_u32:
+ case RISCV::BI__builtin_riscv_pmaccu_h00_u32x2:
+ case RISCV::BI__builtin_riscv_maccu_w00_u64:
+ ID = Intrinsic::riscv_maccu_00;
+ break;
+ case RISCV::BI__builtin_riscv_maccu_h01_u32:
+ case RISCV::BI__builtin_riscv_pmaccu_h01_u32x2:
+ case RISCV::BI__builtin_riscv_maccu_w01_u64:
+ ID = Intrinsic::riscv_maccu_01;
+ break;
+ case RISCV::BI__builtin_riscv_maccu_h11_u32:
+ case RISCV::BI__builtin_riscv_pmaccu_h11_u32x2:
+ case RISCV::BI__builtin_riscv_maccu_w11_u64:
+ ID = Intrinsic::riscv_maccu_11;
+ break;
+ case RISCV::BI__builtin_riscv_maccsu_h00_i32:
+ case RISCV::BI__builtin_riscv_pmaccsu_h00_i32x2:
+ case RISCV::BI__builtin_riscv_maccsu_w00_i64:
+ ID = Intrinsic::riscv_maccsu_00;
+ break;
+ case RISCV::BI__builtin_riscv_maccsu_h11_i32:
+ case RISCV::BI__builtin_riscv_pmaccsu_h11_i32x2:
+ case RISCV::BI__builtin_riscv_maccsu_w11_i64:
+ ID = Intrinsic::riscv_maccsu_11;
+ break;
+ }
+
+ IntrinsicTypes = {ResultType, Ops[1]->getType()};
+ break;
+ }
+
// Zk builtins
// Zknh
diff --git a/clang/lib/Headers/riscv_packed_simd.h b/clang/lib/Headers/riscv_packed_simd.h
index 37ac140780d02..5c359337d8b7f 100644
--- a/clang/lib/Headers/riscv_packed_simd.h
+++ b/clang/lib/Headers/riscv_packed_simd.h
@@ -984,6 +984,34 @@ __packed_ternary_builtin_cast(mqracc_w00_i64, int64_t, int32x2_t, __builtin_risc
__packed_ternary_builtin_cast(mqracc_w01_i64, int64_t, int32x2_t, __builtin_riscv_mqracc_w01_i64)
__packed_ternary_builtin_cast(mqracc_w11_i64, int64_t, int32x2_t, __builtin_riscv_mqracc_w11_i64)
+/* Packed Multiply Parts Accumulate (32-bit) */
+__packed_ternary_builtin_mixed(macc_h00_i32, int32_t, int16x2_t, int16x2_t, __builtin_riscv_macc_h00_i32)
+__packed_ternary_builtin_mixed(macc_h01_i32, int32_t, int16x2_t, int16x2_t, __builtin_riscv_macc_h01_i32)
+__packed_ternary_builtin_mixed(macc_h11_i32, int32_t, int16x2_t, int16x2_t, __builtin_riscv_macc_h11_i32)
+__packed_ternary_builtin_mixed(maccu_h00_u32, uint32_t, uint16x2_t, uint16x2_t, __builtin_riscv_maccu_h00_u32)
+__packed_ternary_builtin_mixed(maccu_h01_u32, uint32_t, uint16x2_t, uint16x2_t, __builtin_riscv_maccu_h01_u32)
+__packed_ternary_builtin_mixed(maccu_h11_u32, uint32_t, uint16x2_t, uint16x2_t, __builtin_riscv_maccu_h11_u32)
+__packed_ternary_builtin_mixed(maccsu_h00_i32, int32_t, int16x2_t, uint16x2_t, __builtin_riscv_maccsu_h00_i32)
+__packed_ternary_builtin_mixed(maccsu_h11_i32, int32_t, int16x2_t, uint16x2_t, __builtin_riscv_maccsu_h11_i32)
+
+/* Packed Multiply Parts Accumulate (64-bit) */
+__packed_ternary_builtin_mixed(pmacc_h00_i32x2, int32x2_t, int16x4_t, int16x4_t, __builtin_riscv_pmacc_h00_i32x2)
+__packed_ternary_builtin_mixed(pmacc_h01_i32x2, int32x2_t, int16x4_t, int16x4_t, __builtin_riscv_pmacc_h01_i32x2)
+__packed_ternary_builtin_mixed(pmacc_h11_i32x2, int32x2_t, int16x4_t, int16x4_t, __builtin_riscv_pmacc_h11_i32x2)
+__packed_ternary_builtin_mixed(pmaccu_h00_u32x2, uint32x2_t, uint16x4_t, uint16x4_t, __builtin_riscv_pmaccu_h00_u32x2)
+__packed_ternary_builtin_mixed(pmaccu_h01_u32x2, uint32x2_t, uint16x4_t, uint16x4_t, __builtin_riscv_pmaccu_h01_u32x2)
+__packed_ternary_builtin_mixed(pmaccu_h11_u32x2, uint32x2_t, uint16x4_t, uint16x4_t, __builtin_riscv_pmaccu_h11_u32x2)
+__packed_ternary_builtin_mixed(pmaccsu_h00_i32x2, int32x2_t, int16x4_t, uint16x4_t, __builtin_riscv_pmaccsu_h00_i32x2)
+__packed_ternary_builtin_mixed(pmaccsu_h11_i32x2, int32x2_t, int16x4_t, uint16x4_t, __builtin_riscv_pmaccsu_h11_i32x2)
+__packed_ternary_builtin_mixed(macc_w00_i64, int64_t, int32x2_t, int32x2_t, __builtin_riscv_macc_w00_i64)
+__packed_ternary_builtin_mixed(macc_w01_i64, int64_t, int32x2_t, int32x2_t, __builtin_riscv_macc_w01_i64)
+__packed_ternary_builtin_mixed(macc_w11_i64, int64_t, int32x2_t, int32x2_t, __builtin_riscv_macc_w11_i64)
+__packed_ternary_builtin_mixed(maccu_w00_u64, uint64_t, uint32x2_t, uint32x2_t, __builtin_riscv_maccu_w00_u64)
+__packed_ternary_builtin_mixed(maccu_w01_u64, uint64_t, uint32x2_t, uint32x2_t, __builtin_riscv_maccu_w01_u64)
+__packed_ternary_builtin_mixed(maccu_w11_u64, uint64_t, uint32x2_t, uint32x2_t, __builtin_riscv_maccu_w11_u64)
+__packed_ternary_builtin_mixed(maccsu_w00_i64, int64_t, int32x2_t, uint32x2_t, __builtin_riscv_maccsu_w00_i64)
+__packed_ternary_builtin_mixed(maccsu_w11_i64, int64_t, int32x2_t, uint32x2_t, __builtin_riscv_maccsu_w11_i64)
+
/* Packed Narrowing Clip Pair (32-bit) */
__packed_binary_builtin_cast(pnclipp_i8x4, int16x2_t, int8x4_t, __builtin_riscv_pnclipp_i8x4)
__packed_binary_builtin_cast(pnclipup_u8x4, uint16x2_t, uint8x4_t, __builtin_riscv_pnclipup_u8x4)
diff --git a/cross-project-tests/intrinsic-header-tests/riscv_packed_simd.c b/cross-project-tests/intrinsic-header-tests/riscv_packed_simd.c
index 48fcf036794ff..df9623edfa71a 100644
--- a/cross-project-tests/intrinsic-header-tests/riscv_packed_simd.c
+++ b/cross-project-tests/intrinsic-header-tests/riscv_packed_simd.c
@@ -4248,3 +4248,175 @@ int32_t test_pget_i32x2_i32(int32x2_t v) {
uint32_t test_pget_u32x2_u32(uint32x2_t v) {
return __riscv_pget_u32x2_u32(v, 1);
}
+
+/* Packed Multiply Parts Accumulate (32-bit) */
+
+// CHECK-LABEL: test_macc_h00_i32:
+// RV32: macc.h00
+// RV64: pmacc.w.h00
+int32_t test_macc_h00_i32(int32_t rd, int16x2_t a, int16x2_t b) {
+ return __riscv_macc_h00_i32(rd, a, b);
+}
+
+// CHECK-LABEL: test_macc_h01_i32:
+// RV32: macc.h01
+// RV64: pmacc.w.h01
+int32_t test_macc_h01_i32(int32_t rd, int16x2_t a, int16x2_t b) {
+ return __riscv_macc_h01_i32(rd, a, b);
+}
+
+// CHECK-LABEL: test_macc_h11_i32:
+// RV32: macc.h11
+// RV64: pmacc.w.h11
+int32_t test_macc_h11_i32(int32_t rd, int16x2_t a, int16x2_t b) {
+ return __riscv_macc_h11_i32(rd, a, b);
+}
+
+// CHECK-LABEL: test_maccu_h00_u32:
+// RV32: maccu.h00
+// RV64: pmaccu.w.h00
+uint32_t test_maccu_h00_u32(uint32_t rd, uint16x2_t a, uint16x2_t b) {
+ return __riscv_maccu_h00_u32(rd, a, b);
+}
+
+// CHECK-LABEL: test_maccu_h01_u32:
+// RV32: maccu.h01
+// RV64: pmaccu.w.h01
+uint32_t test_maccu_h01_u32(uint32_t rd, uint16x2_t a, uint16x2_t b) {
+ return __riscv_maccu_h01_u32(rd, a, b);
+}
+
+// CHECK-LABEL: test_maccu_h11_u32:
+// RV32: maccu.h11
+// RV64: pmaccu.w.h11
+uint32_t test_maccu_h11_u32(uint32_t rd, uint16x2_t a, uint16x2_t b) {
+ return __riscv_maccu_h11_u32(rd, a, b);
+}
+
+// CHECK-LABEL: test_maccsu_h00_i32:
+// RV32: maccsu.h00
+// RV64: pmaccsu.w.h00
+int32_t test_maccsu_h00_i32(int32_t rd, int16x2_t a, uint16x2_t b) {
+ return __riscv_maccsu_h00_i32(rd, a, b);
+}
+
+// CHECK-LABEL: test_maccsu_h11_i32:
+// RV32: maccsu.h11
+// RV64: pmaccsu.w.h11
+int32_t test_maccsu_h11_i32(int32_t rd, int16x2_t a, uint16x2_t b) {
+ return __riscv_maccsu_h11_i32(rd, a, b);
+}
+
+/* Packed Multiply Parts Accumulate (64-bit) */
+
+// CHECK-LABEL: test_pmacc_h00_i32x2:
+// RV32-COUNT-2: macc.h00
+// RV64: pmacc.w.h00
+int32x2_t test_pmacc_h00_i32x2(int32x2_t rd, int16x4_t a, int16x4_t b) {
+ return __riscv_pmacc_h00_i32x2(rd, a, b);
+}
+
+// CHECK-LABEL: test_pmacc_h01_i32x2:
+// RV32-COUNT-2: macc.h01
+// RV64: pmacc.w.h01
+int32x2_t test_pmacc_h01_i32x2(int32x2_t rd, int16x4_t a, int16x4_t b) {
+ return __riscv_pmacc_h01_i32x2(rd, a, b);
+}
+
+// CHECK-LABEL: test_pmacc_h11_i32x2:
+// RV32-COUNT-2: macc.h11
+// RV64: pmacc.w.h11
+int32x2_t test_pmacc_h11_i32x2(int32x2_t rd, int16x4_t a, int16x4_t b) {
+ return __riscv_pmacc_h11_i32x2(rd, a, b);
+}
+
+// CHECK-LABEL: test_pmaccu_h00_u32x2:
+// RV32-COUNT-2: maccu.h00
+// RV64: pmaccu.w.h00
+uint32x2_t test_pmaccu_h00_u32x2(uint32x2_t rd, uint16x4_t a, uint16x4_t b) {
+ return __riscv_pmaccu_h00_u32x2(rd, a, b);
+}
+
+// CHECK-LABEL: test_pmaccu_h01_u32x2:
+// RV32-COUNT-2: maccu.h01
+// RV64: pmaccu.w.h01
+uint32x2_t test_pmaccu_h01_u32x2(uint32x2_t rd, uint16x4_t a, uint16x4_t b) {
+ return __riscv_pmaccu_h01_u32x2(rd, a, b);
+}
+
+// CHECK-LABEL: test_pmaccu_h11_u32x2:
+// RV32-COUNT-2: maccu.h11
+// RV64: pmaccu.w.h11
+uint32x2_t test_pmaccu_h11_u32x2(uint32x2_t rd, uint16x4_t a, uint16x4_t b) {
+ return __riscv_pmaccu_h11_u32x2(rd, a, b);
+}
+
+// CHECK-LABEL: test_pmaccsu_h00_i32x2:
+// RV32-COUNT-2: maccsu.h00
+// RV64: pmaccsu.w.h00
+int32x2_t test_pmaccsu_h00_i32x2(int32x2_t rd, int16x4_t a, uint16x4_t b) {
+ return __riscv_pmaccsu_h00_i32x2(rd, a, b);
+}
+
+// CHECK-LABEL: test_pmaccsu_h11_i32x2:
+// RV32-COUNT-2: maccsu.h11
+// RV64: pmaccsu.w.h11
+int32x2_t test_pmaccsu_h11_i32x2(int32x2_t rd, int16x4_t a, uint16x4_t b) {
+ return __riscv_pmaccsu_h11_i32x2(rd, a, b);
+}
+
+// CHECK-LABEL: test_macc_w00_i64:
+// RV32: wmacc
+// RV64: macc.w00
+int64_t test_macc_w00_i64(int64_t rd, int32x2_t a, int32x2_t b) {
+ return __riscv_macc_w00_i64(rd, a, b);
+}
+
+// CHECK-LABEL: test_macc_w01_i64:
+// RV32: wmacc
+// RV64: macc.w01
+int64_t test_macc_w01_i64(int64_t rd, int32x2_t a, int32x2_t b) {
+ return __riscv_macc_w01_i64(rd, a, b);
+}
+
+// CHECK-LABEL: test_macc_w11_i64:
+// RV32: wmacc
+// RV64: macc.w11
+int64_t test_macc_w11_i64(int64_t rd, int32x2_t a, int32x2_t b) {
+ return __riscv_macc_w11_i64(rd, a, b);
+}
+
+// CHECK-LABEL: test_maccu_w00_u64:
+// RV32: wmaccu
+// RV64: maccu.w00
+uint64_t test_maccu_w00_u64(uint64_t rd, uint32x2_t a, uint32x2_t b) {
+ return __riscv_maccu_w00_u64(rd, a, b);
+}
+
+// CHECK-LABEL: test_maccu_w01_u64:
+// RV32: wmaccu
+// RV64: maccu.w01
+uint64_t test_maccu_w01_u64(uint64_t rd, uint32x2_t a, uint32x2_t b) {
+ return __riscv_maccu_w01_u64(rd, a, b);
+}
+
+// CHECK-LABEL: test_maccu_w11_u64:
+// RV32: wmaccu
+// RV64: maccu.w11
+uint64_t test_maccu_w11_u64(uint64_t rd, uint32x2_t a, uint32x2_t b) {
+ return __riscv_maccu_w11_u64(rd, a, b);
+}
+
+// CHECK-LABEL: test_maccsu_w00_i64:
+// RV32: wmaccsu
+// RV64: maccsu.w00
+int64_t test_maccsu_w00_i64(int64_t rd, int32x2_t a, uint32x2_t b) {
+ return __riscv_maccsu_w00_i64(rd, a, b);
+}
+
+// CHECK-LABEL: test_maccsu_w11_i64:
+// RV32: wmaccsu
+// RV64: maccsu.w11
+int64_t test_maccsu_w11_i64(int64_t rd, int32x2_t a, uint32x2_t b) {
+ return __riscv_maccsu_w11_i64(rd, a, b);
+}
diff --git a/llvm/include/llvm/IR/IntrinsicsRISCV.td b/llvm/include/llvm/IR/IntrinsicsRISCV.td
index e4ec9e9beb5ee..e70caa8b09700 100644
--- a/llvm/include/llvm/IR/IntrinsicsRISCV.td
+++ b/llvm/include/llvm/IR/IntrinsicsRISCV.td
@@ -2186,6 +2186,36 @@ class RVPBinaryIntrinsic
def int_riscv_mulsu_00 : RVPScalarMulPartsIntrinsic;
def int_riscv_mulsu_11 : RVPScalarMulPartsIntrinsic;
+ // Packed Multiply Parts Accumulate.
+ class RVPPackedMulPartsAccIntrinsic
+ : DefaultAttrsIntrinsic<[llvm_anyvector_ty],
+ [LLVMMatchType<0>,
+ LLVMSubdivide2VectorType<0>,
+ LLVMSubdivide2VectorType<0>],
+ [IntrNoMem, IntrSpeculatable]>;
+ def int_riscv_pmacc_00 : RVPPackedMulPartsAccIntrinsic;
+ def int_riscv_pmacc_01 : RVPPackedMulPartsAccIntrinsic;
+ def int_riscv_pmacc_11 : RVPPackedMulPartsAccIntrinsic;
+ def int_riscv_pmaccu_00 : RVPPackedMulPartsAccIntrinsic;
+ def int_riscv_pmaccu_01 : RVPPackedMulPartsAccIntrinsic;
+ def int_riscv_pmaccu_11 : RVPPackedMulPartsAccIntrinsic;
+ def int_riscv_pmaccsu_00 : RVPPackedMulPartsAccIntrinsic;
+ def int_riscv_pmaccsu_11 : RVPPackedMulPartsAccIntrinsic;
+
+ class RVPScalarMulPartsAccIntrinsic
+ : DefaultAttrsIntrinsic<[llvm_anyint_ty],
+ [LLVMMatchType<0>, llvm_anyvector_ty,
+ LLVMMatchType<1>],
+ [IntrNoMem, IntrSpeculatable]>;
+ def int_riscv_macc_00 : RVPScalarMulPartsAccIntrinsic;
+ def int_riscv_macc_01 : RVPScalarMulPartsAccIntrinsic;
+ def int_riscv_macc_11 : RVPScalarMulPartsAccIntrinsic;
+ def int_riscv_maccu_00 : RVPScalarMulPartsAccIntrinsic;
+ def int_riscv_maccu_01 : RVPScalarMulPartsAccIntrinsic;
+ def int_riscv_maccu_11 : RVPScalarMulPartsAccIntrinsic;
+ def int_riscv_maccsu_00 : RVPScalarMulPartsAccIntrinsic;
+ def int_riscv_maccsu_11 : RVPScalarMulPartsAccIntrinsic;
+
// Packed Absolute Difference Sum.
def int_riscv_pabdsumu
: DefaultAttrsIntrinsic<[llvm_anyint_ty],
diff --git a/llvm/lib/Target/RISCV/RISCVISelDAGToDAG.cpp b/llvm/lib/Target/RISCV/RISCVISelDAGToDAG.cpp
index d263d0320839b..c1c7031a17e71 100644
--- a/llvm/lib/Target/RISCV/RISCVISelDAGToDAG.cpp
+++ b/llvm/lib/Target/RISCV/RISCVISelDAGToDAG.cpp
@@ -2073,13 +2073,35 @@ void RISCVDAGToDAGISel::Select(SDNode *Node) {
return;
}
case RISCVISD::MQWACC:
- case RISCVISD::MQRWACC: {
+ case RISCVISD::MQRWACC:
+ case RISCVISD::WMACC:
+ case RISCVISD::WMACCU:
+ case RISCVISD::WMACCSU: {
assert(!Subtarget->is64Bit() && Subtarget->hasStdExtP() &&
"Unexpected opcode");
SDValue Op0 = buildGPRPair(CurDAG, DL, MVT::Untyped, Node->getOperand(0),
Node->getOperand(1));
- unsigned Opc = Opcode == RISCVISD::MQRWACC ? RISCV::MQRWACC : RISCV::MQWACC;
+ unsigned Opc;
+ switch (Opcode) {
+ default:
+ llvm_unreachable("Unexpected opcode");
+ case RISCVISD::MQWACC:
+ Opc = RISCV::MQWACC;
+ break;
+ case RISCVISD::MQRWACC:
+ Opc = RISCV::MQRWACC;
+ break;
+ case RISCVISD::WMACC:
+ Opc = RISCV::WMACC;
+ break;
+ case RISCVISD::WMACCU:
+ Opc = RISCV::WMACCU;
+ break;
+ case RISCVISD::WMACCSU:
+ Opc = RISCV::WMACCSU;
+ break;
+ }
MachineSDNode *New = CurDAG->getMachineNode(
Opc, DL, MVT::Untyped, Op0, Node->getOperand(2), Node->getOperand(3));
auto [Lo, Hi] = extractGPRPair(CurDAG, DL, SDValue(New, 0));
diff --git a/llvm/lib/Target/RISCV/RISCVISelLowering.cpp b/llvm/lib/Target/RISCV/RISCVISelLowering.cpp
index 5f8ad5da42da1..e2d91fc4c4cfa 100644
--- a/llvm/lib/Target/RISCV/RISCVISelLowering.cpp
+++ b/llvm/lib/Target/RISCV/RISCVISelLowering.cpp
@@ -12548,6 +12548,38 @@ static Intrinsic::ID getRVPScalarMulPartsIntrinsic(unsigned IntNo) {
}
}
+/// Return the multiply-parts accumulate node for \p IntNo.
+static unsigned getRVPMulAccHalvesOpcode(unsigned IntNo) {
+ switch (IntNo) {
+ default:
+ llvm_unreachable("Unexpected RISC-V multiply-parts accumulate intrinsic");
+ case Intrinsic::riscv_pmacc_00:
+ case Intrinsic::riscv_macc_00:
+ return RISCVISD::PMACC_HALVES_00;
+ case Intrinsic::riscv_pmacc_01:
+ case Intrinsic::riscv_macc_01:
+ return RISCVISD::PMACC_HALVES_01;
+ case Intrinsic::riscv_pmacc_11:
+ case Intrinsic::riscv_macc_11:
+ return RISCVISD::PMACC_HALVES_11;
+ case Intrinsic::riscv_pmaccu_00:
+ case Intrinsic::riscv_maccu_00:
+ return RISCVISD::PMACCU_HALVES_00;
+ case Intrinsic::riscv_pmaccu_01:
+ case Intrinsic::riscv_maccu_01:
+ return RISCVISD::PMACCU_HALVES_01;
+ case Intrinsic::riscv_pmaccu_11:
+ case Intrinsic::riscv_maccu_11:
+ return RISCVISD::PMACCU_HALVES_11;
+ case Intrinsic::riscv_pmaccsu_00:
+ case Intrinsic::riscv_maccsu_00:
+ return RISCVISD::PMACCSU_HALVES_00;
+ case Intrinsic::riscv_pmaccsu_11:
+ case Intrinsic::riscv_maccsu_11:
+ return RISCVISD::PMACCSU_HALVES_11;
+ }
+}
+
/// Return {opcode, rs1 lane, rs2 lane} for the word form of \p IntNo.
static std::tuple<unsigned, unsigned, unsigned>
getRVPWordMulPartsOpcodeAndLanes(unsigned IntNo) {
@@ -12573,6 +12605,32 @@ getRVPWordMulPartsOpcodeAndLanes(unsigned IntNo) {
}
}
+/// Return {opcode, rs1 lane, rs2 lane} for the word form of accumulate
+/// intrinsic \p IntNo.
+static std::tuple<unsigned, unsigned, unsigned>
+getRVPWordMulPartsAccOpcodeAndLanes(unsigned IntNo) {
+ switch (IntNo) {
+ default:
+ llvm_unreachable("Unexpected RISC-V multiply-parts accumulate intrinsic");
+ case Intrinsic::riscv_macc_00:
+ return {RISCVISD::WMACC, 0, 0};
+ case Intrinsic::riscv_macc_01:
+ return {RISCVISD::WMACC, 0, 1};
+ case Intrinsic::riscv_macc_11:
+ return {RISCVISD::WMACC, 1, 1};
+ case Intrinsic::riscv_maccu_00:
+ return {RISCVISD::WMACCU, 0, 0};
+ case Intrinsic::riscv_maccu_01:
+ return {RISCVISD::WMACCU, 0, 1};
+ case Intrinsic::riscv_maccu_11:
+ return {RISCVISD::WMACCU, 1, 1};
+ case Intrinsic::riscv_maccsu_00:
+ return {RISCVISD::WMACCSU, 0, 0};
+ case Intrinsic::riscv_maccsu_11:
+ return {RISCVISD::WMACCSU, 1, 1};
+ }
+}
+
SDValue RISCVTargetLowering::LowerINTRINSIC_WO_CHAIN(SDValue Op,
SelectionDAG &DAG) const {
unsigned IntNo = Op.getConstantOperandVal(0);
@@ -12635,6 +12693,42 @@ SDValue RISCVTargetLowering::LowerINTRINSIC_WO_CHAIN(SDValue Op,
SDValue Hi = DAG.getNode(Opc, DL, HalfVT, Rs1Hi, Rs2Hi);
return DAG.getNode(ISD::CONCAT_VECTORS, DL, VT, Lo, Hi);
}
+ case Intrinsic::riscv_pmacc_00:
+ case Intrinsic::riscv_pmacc_01:
+ case Intrinsic::riscv_pmacc_11:
+ case Intrinsic::riscv_pmaccu_00:
+ case Intrinsic::riscv_pmaccu_01:
+ case Intrinsic::riscv_pmaccu_11:
+ case Intrinsic::riscv_pmaccsu_00:
+ case Intrinsic::riscv_pmaccsu_11:
+ case Intrinsic::riscv_macc_00:
+ case Intrinsic::riscv_macc_01:
+ case Intrinsic::riscv_macc_11:
+ case Intrinsic::riscv_maccu_00:
+ case Intrinsic::riscv_maccu_01:
+ case Intrinsic::riscv_maccu_11:
+ case Intrinsic::riscv_maccsu_00:
+ case Intrinsic::riscv_maccsu_11: {
+ MVT VT = Op.getSimpleValueType();
+ SDValue Rd = Op.getOperand(1);
+ SDValue Rs1 = Op.getOperand(2);
+ SDValue Rs2 = Op.getOperand(3);
+ unsigned Opc = getRVPMulAccHalvesOpcode(IntNo);
+ if (VT != MVT::v2i32 || !Subtarget.isPExtPackedDoubleType(VT))
+ return DAG.getNode(Opc, DL, VT, Rd, Rs1, Rs2);
+
+ // On RV32 a 64-bit result lives in a GPR pair; accumulate each half with
+ // the 32-bit form of the same product.
+ auto [Rs1Lo, Rs1Hi] = DAG.SplitVector(Rs1, DL);
+ auto [Rs2Lo, Rs2Hi] = DAG.SplitVector(Rs2, DL);
+ SDValue Lo =
+ DAG.getNode(Opc, DL, MVT::i32,
+ DAG.getExtractVectorElt(DL, MVT::i32, Rd, 0), Rs1Lo, Rs2Lo);
+ SDValue Hi =
+ DAG.getNode(Opc, DL, MVT::i32,
+ DAG.getExtractVectorElt(DL, MVT::i32, Rd, 1), Rs1Hi, Rs2Hi);
+ return DAG.getNode(ISD::BUILD_VECTOR, DL, VT, Lo, Hi);
+ }
case Intrinsic::riscv_pas:
case Intrinsic::riscv_psa:
case Intrinsic::riscv_psas:
@@ -17263,6 +17357,51 @@ void RISCVTargetLowering::ReplaceNodeResults(SDNode *N,
}
reportFatalUsageError("unsupported llvm.riscv multiply-parts intrinsic");
}
+ case Intrinsic::riscv_macc_00:
+ case Intrinsic::riscv_macc_01:
+ case Intrinsic::riscv_macc_11:
+ case Intrinsic::riscv_maccu_00:
+ case Intrinsic::riscv_maccu_01:
+ case Intrinsic::riscv_maccu_11:
+ case Intrinsic::riscv_maccsu_00:
+ case Intrinsic::riscv_maccsu_11: {
+ // macc.hXX exists only on RV32 and macc.wXX only on RV64; the other XLEN
+ // has to build the product here.
+ MVT VT = N->getSimpleValueType(0);
+ MVT SrcVT = N->getOperand(2).getSimpleValueType();
+ if (Subtarget.hasStdExtP() && Subtarget.is64Bit() && VT == MVT::i32 &&
+ SrcVT == MVT::v2i16) {
+ // Accumulate into the first element of the packed product.
+ SDValue Undef = DAG.getUNDEF(SrcVT);
+ SDValue Rd = DAG.getNode(ISD::SCALAR_TO_VECTOR, DL, MVT::v2i32,
+ N->getOperand(1));
+ SDValue Rs1 = DAG.getNode(ISD::CONCAT_VECTORS, DL, MVT::v4i16,
+ N->getOperand(2), Undef);
+ SDValue Rs2 = DAG.getNode(ISD::CONCAT_VECTORS, DL, MVT::v4i16,
+ N->getOperand(3), Undef);
+ SDValue Res = DAG.getNode(getRVPMulAccHalvesOpcode(IntNo), DL,
+ MVT::v2i32, Rd, Rs1, Rs2);
+ Results.push_back(DAG.getExtractVectorElt(DL, MVT::i32, Res, 0));
+ return;
+ }
+ if (Subtarget.hasStdExtP() && !Subtarget.is64Bit() && VT == MVT::i64 &&
+ SrcVT == MVT::v2i32) {
+ auto [Opc, Rs1Lane, Rs2Lane] =
+ getRVPWordMulPartsAccOpcodeAndLanes(IntNo);
+ auto [RdLo, RdHi] =
+ DAG.SplitScalar(N->getOperand(1), DL, MVT::i32, MVT::i32);
+ SDValue Rs1 =
+ DAG.getExtractVectorElt(DL, MVT::i32, N->getOperand(2), Rs1Lane);
+ SDValue Rs2 =
+ DAG.getExtractVectorElt(DL, MVT::i32, N->getOperand(3), Rs2Lane);
+ SDValue Res = DAG.getNode(Opc, DL, DAG.getVTList(MVT::i32, MVT::i32),
+ RdLo, RdHi, Rs1, Rs2);
+ Results.push_back(
+ DAG.getNode(ISD::BUILD_PAIR, DL, MVT::i64, Res, Res.getValue(1)));
+ return;
+ }
+ reportFatalUsageError("unsupported llvm.riscv multiply-parts intrinsic");
+ }
case Intrinsic::riscv_paadd:
case Intrinsic::riscv_paaddu:
case Intrinsic::riscv_pasub:
diff --git a/llvm/lib/Target/RISCV/RISCVInstrInfoP.td b/llvm/lib/Target/RISCV/RISCVInstrInfoP.td
index 3637ee98df990..dfdfb5435c3c3 100644
--- a/llvm/lib/Target/RISCV/RISCVInstrInfoP.td
+++ b/llvm/lib/Target/RISCV/RISCVInstrInfoP.td
@@ -1791,6 +1791,11 @@ class PatMulParts<SDPatternOperator OpNode, RVInst Inst, ValueType ResultVT,
ValueType SourceVT>
: Pat<(ResultVT (OpNode (SourceVT GPR:$rs1), (SourceVT GPR:$rs2))),
(Inst GPR:$rs1, GPR:$rs2)>;
+class PatMulPartsAcc<SDPatternOperator OpNode, RVInst Inst, ValueType ResultVT,
+ ValueType SourceVT>
+ : Pat<(ResultVT (OpNode (ResultVT GPR:$rd), (SourceVT GPR:$rs1),
+ (SourceVT GPR:$rs2))),
+ (Inst GPR:$rd, GPR:$rs1, GPR:$rs2)>;
class PatGprGprGpr<SDPatternOperator OpNode, RVInst Inst, ValueType VT>
: Pat<(VT (OpNode (VT GPR:$rd), (VT GPR:$rs1), (VT GPR:$rs2))),
@@ -1853,6 +1858,11 @@ def riscv_wsub : RVSDNode<"WSUB", SDTIntBinHiLoOp>;
def riscv_wmulsu : RVSDNode<"WMULSU", SDTIntBinHiLoOp>;
+// Widening multiply-accumulate into a GPR pair: rd_p = rd_p + rs1 * rs2.
+def riscv_wmacc : RVSDNode<"WMACC", SDT_RISCVWideningAddSubAccumulate>;
+def riscv_wmaccu : RVSDNode<"WMACCU", SDT_RISCVWideningAddSubAccumulate>;
+def riscv_wmaccsu : RVSDNode<"WMACCSU", SDT_RISCVWideningAddSubAccumulate>;
+
def SDT_RISCVPackedWideningMul : SDTypeProfile<1, 2, [SDTCisVec<0>,
SDTCisSameAs<1, 2>,
SDTCisOpSmallerThanOp<1, 0>,
@@ -1917,6 +1927,29 @@ def riscv_pm2addu_h
: RVSDNode<"PM2ADDU_H", SDT_RISCVPM2Halfword, [SDNPCommutative]>;
def riscv_pm2sub_h : RVSDNode<"PM2SUB_H", SDT_RISCVPM2Halfword>;
+def SDT_RISCVWideningMulAccByHalves
+ : SDTypeProfile<1, 3, [SDTCisSameAs<0, 1>,
+ SDTCisSameAs<2, 3>,
+ SDTCisSameSizeAs<0, 2>]>;
+def riscv_pmacc_halves_00
+ : RVSDNode<"PMACC_HALVES_00", SDT_RISCVWideningMulAccByHalves>;
+def riscv_pmacc_halves_01
+ : RVSDNode<"PMACC_HALVES_01", SDT_RISCVWideningMulAccByHalves>;
+def riscv_pmacc_halves_11
+ : RVSDNode<"PMACC_HALVES_11", SDT_RISCVWideningMulAccByHalves>;
+
+def riscv_pmaccu_halves_00
+ : RVSDNode<"PMACCU_HALVES_00", SDT_RISCVWideningMulAccByHalves>;
+def riscv_pmaccu_halves_01
+ : RVSDNode<"PMACCU_HALVES_01", SDT_RISCVWideningMulAccByHalves>;
+def riscv_pmaccu_halves_11
+ : RVSDNode<"PMACCU_HALVES_11", SDT_RISCVWideningMulAccByHalves>;
+
+def riscv_pmaccsu_halves_00
+ : RVSDNode<"PMACCSU_HALVES_00", SDT_RISCVWideningMulAccByHalves>;
+def riscv_pmaccsu_halves_11
+ : RVSDNode<"PMACCSU_HALVES_11", SDT_RISCVWideningMulAccByHalves>;
+
def SDT_RISCVWideningShiftLeft : SDTypeProfile<2, 2, [SDTCisVT<0, i32>,
SDTCisSameAs<0, 1>,
SDTCisSameAs<0, 2>,
@@ -2382,6 +2415,16 @@ let Predicates = [HasStdExtP] in {
(PMULSU_H_B11 GPR:$rs1, GPR:$rs2)>;
let append Predicates = [IsRV32] in {
+ // Scalar halfword multiply-parts accumulate patterns.
+ def : PatMulPartsAcc<riscv_pmacc_halves_00, MACC_H00, i32, v2i16>;
+ def : PatMulPartsAcc<riscv_pmacc_halves_01, MACC_H01, i32, v2i16>;
+ def : PatMulPartsAcc<riscv_pmacc_halves_11, MACC_H11, i32, v2i16>;
+ def : PatMulPartsAcc<riscv_pmaccu_halves_00, MACCU_H00, i32, v2i16>;
+ def : PatMulPartsAcc<riscv_pmaccu_halves_01, MACCU_H01, i32, v2i16>;
+ def : PatMulPartsAcc<riscv_pmaccu_halves_11, MACCU_H11, i32, v2i16>;
+ def : PatMulPartsAcc<riscv_pmaccsu_halves_00, MACCSU_H00, i32, v2i16>;
+ def : PatMulPartsAcc<riscv_pmaccsu_halves_11, MACCSU_H11, i32, v2i16>;
+
// Scalar halfword multiply-parts patterns.
def : PatMulParts<int_riscv_mul_00, MUL_H00, i32, v2i16>;
def : PatMulParts<int_riscv_mul_01, MUL_H01, i32, v2i16>;
@@ -3075,6 +3118,26 @@ let append Predicates = [IsRV64] in {
def : PatGprGpr<riscv_asub, PASUB_W, v2i32>;
def : PatGprGpr<riscv_asubu, PASUBU_W, v2i32>;
+ // Packed halfword multiply-parts accumulate patterns.
+ def : PatMulPartsAcc<riscv_pmacc_halves_00, PMACC_W_H00, v2i32, v4i16>;
+ def : PatMulPartsAcc<riscv_pmacc_halves_01, PMACC_W_H01, v2i32, v4i16>;
+ def : PatMulPartsAcc<riscv_pmacc_halves_11, PMACC_W_H11, v2i32, v4i16>;
+ def : PatMulPartsAcc<riscv_pmaccu_halves_00, PMACCU_W_H00, v2i32, v4i16>;
+ def : PatMulPartsAcc<riscv_pmaccu_halves_01, PMACCU_W_H01, v2i32, v4i16>;
+ def : PatMulPartsAcc<riscv_pmaccu_halves_11, PMACCU_W_H11, v2i32, v4i16>;
+ def : PatMulPartsAcc<riscv_pmaccsu_halves_00, PMACCSU_W_H00, v2i32, v4i16>;
+ def : PatMulPartsAcc<riscv_pmaccsu_halves_11, PMACCSU_W_H11, v2i32, v4i16>;
+
+ // Scalar word multiply-parts accumulate patterns.
+ def : PatMulPartsAcc<riscv_pmacc_halves_00, MACC_W00, i64, v2i32>;
+ def : PatMulPartsAcc<riscv_pmacc_halves_01, MACC_W01, i64, v2i32>;
+ def : PatMulPartsAcc<riscv_pmacc_halves_11, MACC_W11, i64, v2i32>;
+ def : PatMulPartsAcc<riscv_pmaccu_halves_00, MACCU_W00, i64, v2i32>;
+ def : PatMulPartsAcc<riscv_pmaccu_halves_01, MACCU_W01, i64, v2i32>;
+ def : PatMulPartsAcc<riscv_pmaccu_halves_11, MACCU_W11, i64, v2i32>;
+ def : PatMulPartsAcc<riscv_pmaccsu_halves_00, MACCSU_W00, i64, v2i32>;
+ def : PatMulPartsAcc<riscv_pmaccsu_halves_11, MACCSU_W11, i64, v2i32>;
+
// Scalar word multiply-parts patterns.
def : PatMulParts<int_riscv_mul_00, MUL_W00, i64, v2i32>;
def : PatMulParts<int_riscv_mul_01, MUL_W01, i64, v2i32>;
diff --git a/llvm/test/CodeGen/RISCV/rvp-simd-32.ll b/llvm/test/CodeGen/RISCV/rvp-simd-32.ll
index c11d9716ec053..e9e04eb6b5ebe 100644
--- a/llvm/test/CodeGen/RISCV/rvp-simd-32.ll
+++ b/llvm/test/CodeGen/RISCV/rvp-simd-32.ll
@@ -3705,3 +3705,125 @@ define i32 @test_pm2addsu_v2i16_i32(<2 x i16> %a, <2 x i16> %b) {
%r = call i32 @llvm.riscv.pm2addsu.i32.v2i16(<2 x i16> %a, <2 x i16> %b)
ret i32 %r
}
+
+; Packed Multiply Parts Accumulate.
+declare i32 @llvm.riscv.macc.00.i32.v2i16(i32, <2 x i16>, <2 x i16>)
+declare i32 @llvm.riscv.macc.01.i32.v2i16(i32, <2 x i16>, <2 x i16>)
+declare i32 @llvm.riscv.macc.11.i32.v2i16(i32, <2 x i16>, <2 x i16>)
+declare i32 @llvm.riscv.maccu.00.i32.v2i16(i32, <2 x i16>, <2 x i16>)
+declare i32 @llvm.riscv.maccu.01.i32.v2i16(i32, <2 x i16>, <2 x i16>)
+declare i32 @llvm.riscv.maccu.11.i32.v2i16(i32, <2 x i16>, <2 x i16>)
+declare i32 @llvm.riscv.maccsu.00.i32.v2i16(i32, <2 x i16>, <2 x i16>)
+declare i32 @llvm.riscv.maccsu.11.i32.v2i16(i32, <2 x i16>, <2 x i16>)
+
+define i32 @test_macc_h00_i32(i32 %rd, <2 x i16> %a, <2 x i16> %b) {
+; RV32-LABEL: test_macc_h00_i32:
+; RV32: # %bb.0:
+; RV32-NEXT: macc.h00 a0, a1, a2
+; RV32-NEXT: ret
+;
+; RV64-LABEL: test_macc_h00_i32:
+; RV64: # %bb.0:
+; RV64-NEXT: pmacc.w.h00 a0, a1, a2
+; RV64-NEXT: ret
+ %r = call i32 @llvm.riscv.macc.00.i32.v2i16(i32 %rd, <2 x i16> %a, <2 x i16> %b)
+ ret i32 %r
+}
+
+define i32 @test_macc_h01_i32(i32 %rd, <2 x i16> %a, <2 x i16> %b) {
+; RV32-LABEL: test_macc_h01_i32:
+; RV32: # %bb.0:
+; RV32-NEXT: macc.h01 a0, a1, a2
+; RV32-NEXT: ret
+;
+; RV64-LABEL: test_macc_h01_i32:
+; RV64: # %bb.0:
+; RV64-NEXT: pmacc.w.h01 a0, a1, a2
+; RV64-NEXT: ret
+ %r = call i32 @llvm.riscv.macc.01.i32.v2i16(i32 %rd, <2 x i16> %a, <2 x i16> %b)
+ ret i32 %r
+}
+
+define i32 @test_macc_h11_i32(i32 %rd, <2 x i16> %a, <2 x i16> %b) {
+; RV32-LABEL: test_macc_h11_i32:
+; RV32: # %bb.0:
+; RV32-NEXT: macc.h11 a0, a1, a2
+; RV32-NEXT: ret
+;
+; RV64-LABEL: test_macc_h11_i32:
+; RV64: # %bb.0:
+; RV64-NEXT: pmacc.w.h11 a0, a1, a2
+; RV64-NEXT: ret
+ %r = call i32 @llvm.riscv.macc.11.i32.v2i16(i32 %rd, <2 x i16> %a, <2 x i16> %b)
+ ret i32 %r
+}
+
+define i32 @test_maccu_h00_i32(i32 %rd, <2 x i16> %a, <2 x i16> %b) {
+; RV32-LABEL: test_maccu_h00_i32:
+; RV32: # %bb.0:
+; RV32-NEXT: maccu.h00 a0, a1, a2
+; RV32-NEXT: ret
+;
+; RV64-LABEL: test_maccu_h00_i32:
+; RV64: # %bb.0:
+; RV64-NEXT: pmaccu.w.h00 a0, a1, a2
+; RV64-NEXT: ret
+ %r = call i32 @llvm.riscv.maccu.00.i32.v2i16(i32 %rd, <2 x i16> %a, <2 x i16> %b)
+ ret i32 %r
+}
+
+define i32 @test_maccu_h01_i32(i32 %rd, <2 x i16> %a, <2 x i16> %b) {
+; RV32-LABEL: test_maccu_h01_i32:
+; RV32: # %bb.0:
+; RV32-NEXT: maccu.h01 a0, a1, a2
+; RV32-NEXT: ret
+;
+; RV64-LABEL: test_maccu_h01_i32:
+; RV64: # %bb.0:
+; RV64-NEXT: pmaccu.w.h01 a0, a1, a2
+; RV64-NEXT: ret
+ %r = call i32 @llvm.riscv.maccu.01.i32.v2i16(i32 %rd, <2 x i16> %a, <2 x i16> %b)
+ ret i32 %r
+}
+
+define i32 @test_maccu_h11_i32(i32 %rd, <2 x i16> %a, <2 x i16> %b) {
+; RV32-LABEL: test_maccu_h11_i32:
+; RV32: # %bb.0:
+; RV32-NEXT: maccu.h11 a0, a1, a2
+; RV32-NEXT: ret
+;
+; RV64-LABEL: test_maccu_h11_i32:
+; RV64: # %bb.0:
+; RV64-NEXT: pmaccu.w.h11 a0, a1, a2
+; RV64-NEXT: ret
+ %r = call i32 @llvm.riscv.maccu.11.i32.v2i16(i32 %rd, <2 x i16> %a, <2 x i16> %b)
+ ret i32 %r
+}
+
+define i32 @test_maccsu_h00_i32(i32 %rd, <2 x i16> %a, <2 x i16> %b) {
+; RV32-LABEL: test_maccsu_h00_i32:
+; RV32: # %bb.0:
+; RV32-NEXT: maccsu.h00 a0, a1, a2
+; RV32-NEXT: ret
+;
+; RV64-LABEL: test_maccsu_h00_i32:
+; RV64: # %bb.0:
+; RV64-NEXT: pmaccsu.w.h00 a0, a1, a2
+; RV64-NEXT: ret
+ %r = call i32 @llvm.riscv.maccsu.00.i32.v2i16(i32 %rd, <2 x i16> %a, <2 x i16> %b)
+ ret i32 %r
+}
+
+define i32 @test_maccsu_h11_i32(i32 %rd, <2 x i16> %a, <2 x i16> %b) {
+; RV32-LABEL: test_maccsu_h11_i32:
+; RV32: # %bb.0:
+; RV32-NEXT: maccsu.h11 a0, a1, a2
+; RV32-NEXT: ret
+;
+; RV64-LABEL: test_maccsu_h11_i32:
+; RV64: # %bb.0:
+; RV64-NEXT: pmaccsu.w.h11 a0, a1, a2
+; RV64-NEXT: ret
+ %r = call i32 @llvm.riscv.maccsu.11.i32.v2i16(i32 %rd, <2 x i16> %a, <2 x i16> %b)
+ ret i32 %r
+}
diff --git a/llvm/test/CodeGen/RISCV/rvp-simd-64.ll b/llvm/test/CodeGen/RISCV/rvp-simd-64.ll
index fd9cf65c2ef53..43107137fe1fa 100644
--- a/llvm/test/CodeGen/RISCV/rvp-simd-64.ll
+++ b/llvm/test/CodeGen/RISCV/rvp-simd-64.ll
@@ -7866,3 +7866,253 @@ define i64 @test_pm4addsu_v4i16_i64(<4 x i16> %a, <4 x i16> %b) {
%r = call i64 @llvm.riscv.pm4addsu.i64.v4i16(<4 x i16> %a, <4 x i16> %b)
ret i64 %r
}
+
+; Packed Multiply Parts Accumulate.
+declare <2 x i32> @llvm.riscv.pmacc.00.v2i32(<2 x i32>, <4 x i16>, <4 x i16>)
+declare <2 x i32> @llvm.riscv.pmacc.01.v2i32(<2 x i32>, <4 x i16>, <4 x i16>)
+declare <2 x i32> @llvm.riscv.pmacc.11.v2i32(<2 x i32>, <4 x i16>, <4 x i16>)
+declare <2 x i32> @llvm.riscv.pmaccu.00.v2i32(<2 x i32>, <4 x i16>, <4 x i16>)
+declare <2 x i32> @llvm.riscv.pmaccu.01.v2i32(<2 x i32>, <4 x i16>, <4 x i16>)
+declare <2 x i32> @llvm.riscv.pmaccu.11.v2i32(<2 x i32>, <4 x i16>, <4 x i16>)
+declare <2 x i32> @llvm.riscv.pmaccsu.00.v2i32(<2 x i32>, <4 x i16>, <4 x i16>)
+declare <2 x i32> @llvm.riscv.pmaccsu.11.v2i32(<2 x i32>, <4 x i16>, <4 x i16>)
+declare i64 @llvm.riscv.macc.00.i64.v2i32(i64, <2 x i32>, <2 x i32>)
+declare i64 @llvm.riscv.macc.01.i64.v2i32(i64, <2 x i32>, <2 x i32>)
+declare i64 @llvm.riscv.macc.11.i64.v2i32(i64, <2 x i32>, <2 x i32>)
+declare i64 @llvm.riscv.maccu.00.i64.v2i32(i64, <2 x i32>, <2 x i32>)
+declare i64 @llvm.riscv.maccu.01.i64.v2i32(i64, <2 x i32>, <2 x i32>)
+declare i64 @llvm.riscv.maccu.11.i64.v2i32(i64, <2 x i32>, <2 x i32>)
+declare i64 @llvm.riscv.maccsu.00.i64.v2i32(i64, <2 x i32>, <2 x i32>)
+declare i64 @llvm.riscv.maccsu.11.i64.v2i32(i64, <2 x i32>, <2 x i32>)
+
+define <2 x i32> @test_pmacc_h00_v2i32(<2 x i32> %rd, <4 x i16> %a, <4 x i16> %b) {
+; RV32-LABEL: test_pmacc_h00_v2i32:
+; RV32: # %bb.0:
+; RV32-NEXT: macc.h00 a1, a3, a5
+; RV32-NEXT: macc.h00 a0, a2, a4
+; RV32-NEXT: ret
+;
+; RV64-LABEL: test_pmacc_h00_v2i32:
+; RV64: # %bb.0:
+; RV64-NEXT: pmacc.w.h00 a0, a1, a2
+; RV64-NEXT: ret
+ %r = call <2 x i32> @llvm.riscv.pmacc.00.v2i32(<2 x i32> %rd, <4 x i16> %a, <4 x i16> %b)
+ ret <2 x i32> %r
+}
+
+define <2 x i32> @test_pmacc_h01_v2i32(<2 x i32> %rd, <4 x i16> %a, <4 x i16> %b) {
+; RV32-LABEL: test_pmacc_h01_v2i32:
+; RV32: # %bb.0:
+; RV32-NEXT: macc.h01 a1, a3, a5
+; RV32-NEXT: macc.h01 a0, a2, a4
+; RV32-NEXT: ret
+;
+; RV64-LABEL: test_pmacc_h01_v2i32:
+; RV64: # %bb.0:
+; RV64-NEXT: pmacc.w.h01 a0, a1, a2
+; RV64-NEXT: ret
+ %r = call <2 x i32> @llvm.riscv.pmacc.01.v2i32(<2 x i32> %rd, <4 x i16> %a, <4 x i16> %b)
+ ret <2 x i32> %r
+}
+
+define <2 x i32> @test_pmacc_h11_v2i32(<2 x i32> %rd, <4 x i16> %a, <4 x i16> %b) {
+; RV32-LABEL: test_pmacc_h11_v2i32:
+; RV32: # %bb.0:
+; RV32-NEXT: macc.h11 a1, a3, a5
+; RV32-NEXT: macc.h11 a0, a2, a4
+; RV32-NEXT: ret
+;
+; RV64-LABEL: test_pmacc_h11_v2i32:
+; RV64: # %bb.0:
+; RV64-NEXT: pmacc.w.h11 a0, a1, a2
+; RV64-NEXT: ret
+ %r = call <2 x i32> @llvm.riscv.pmacc.11.v2i32(<2 x i32> %rd, <4 x i16> %a, <4 x i16> %b)
+ ret <2 x i32> %r
+}
+
+define <2 x i32> @test_pmaccu_h00_v2i32(<2 x i32> %rd, <4 x i16> %a, <4 x i16> %b) {
+; RV32-LABEL: test_pmaccu_h00_v2i32:
+; RV32: # %bb.0:
+; RV32-NEXT: maccu.h00 a1, a3, a5
+; RV32-NEXT: maccu.h00 a0, a2, a4
+; RV32-NEXT: ret
+;
+; RV64-LABEL: test_pmaccu_h00_v2i32:
+; RV64: # %bb.0:
+; RV64-NEXT: pmaccu.w.h00 a0, a1, a2
+; RV64-NEXT: ret
+ %r = call <2 x i32> @llvm.riscv.pmaccu.00.v2i32(<2 x i32> %rd, <4 x i16> %a, <4 x i16> %b)
+ ret <2 x i32> %r
+}
+
+define <2 x i32> @test_pmaccu_h01_v2i32(<2 x i32> %rd, <4 x i16> %a, <4 x i16> %b) {
+; RV32-LABEL: test_pmaccu_h01_v2i32:
+; RV32: # %bb.0:
+; RV32-NEXT: maccu.h01 a1, a3, a5
+; RV32-NEXT: maccu.h01 a0, a2, a4
+; RV32-NEXT: ret
+;
+; RV64-LABEL: test_pmaccu_h01_v2i32:
+; RV64: # %bb.0:
+; RV64-NEXT: pmaccu.w.h01 a0, a1, a2
+; RV64-NEXT: ret
+ %r = call <2 x i32> @llvm.riscv.pmaccu.01.v2i32(<2 x i32> %rd, <4 x i16> %a, <4 x i16> %b)
+ ret <2 x i32> %r
+}
+
+define <2 x i32> @test_pmaccu_h11_v2i32(<2 x i32> %rd, <4 x i16> %a, <4 x i16> %b) {
+; RV32-LABEL: test_pmaccu_h11_v2i32:
+; RV32: # %bb.0:
+; RV32-NEXT: maccu.h11 a1, a3, a5
+; RV32-NEXT: maccu.h11 a0, a2, a4
+; RV32-NEXT: ret
+;
+; RV64-LABEL: test_pmaccu_h11_v2i32:
+; RV64: # %bb.0:
+; RV64-NEXT: pmaccu.w.h11 a0, a1, a2
+; RV64-NEXT: ret
+ %r = call <2 x i32> @llvm.riscv.pmaccu.11.v2i32(<2 x i32> %rd, <4 x i16> %a, <4 x i16> %b)
+ ret <2 x i32> %r
+}
+
+define <2 x i32> @test_pmaccsu_h00_v2i32(<2 x i32> %rd, <4 x i16> %a, <4 x i16> %b) {
+; RV32-LABEL: test_pmaccsu_h00_v2i32:
+; RV32: # %bb.0:
+; RV32-NEXT: maccsu.h00 a1, a3, a5
+; RV32-NEXT: maccsu.h00 a0, a2, a4
+; RV32-NEXT: ret
+;
+; RV64-LABEL: test_pmaccsu_h00_v2i32:
+; RV64: # %bb.0:
+; RV64-NEXT: pmaccsu.w.h00 a0, a1, a2
+; RV64-NEXT: ret
+ %r = call <2 x i32> @llvm.riscv.pmaccsu.00.v2i32(<2 x i32> %rd, <4 x i16> %a, <4 x i16> %b)
+ ret <2 x i32> %r
+}
+
+define <2 x i32> @test_pmaccsu_h11_v2i32(<2 x i32> %rd, <4 x i16> %a, <4 x i16> %b) {
+; RV32-LABEL: test_pmaccsu_h11_v2i32:
+; RV32: # %bb.0:
+; RV32-NEXT: maccsu.h11 a1, a3, a5
+; RV32-NEXT: maccsu.h11 a0, a2, a4
+; RV32-NEXT: ret
+;
+; RV64-LABEL: test_pmaccsu_h11_v2i32:
+; RV64: # %bb.0:
+; RV64-NEXT: pmaccsu.w.h11 a0, a1, a2
+; RV64-NEXT: ret
+ %r = call <2 x i32> @llvm.riscv.pmaccsu.11.v2i32(<2 x i32> %rd, <4 x i16> %a, <4 x i16> %b)
+ ret <2 x i32> %r
+}
+
+define i64 @test_macc_w00_i64(i64 %rd, <2 x i32> %a, <2 x i32> %b) {
+; RV32-LABEL: test_macc_w00_i64:
+; RV32: # %bb.0:
+; RV32-NEXT: wmacc a0, a2, a4
+; RV32-NEXT: ret
+;
+; RV64-LABEL: test_macc_w00_i64:
+; RV64: # %bb.0:
+; RV64-NEXT: macc.w00 a0, a1, a2
+; RV64-NEXT: ret
+ %r = call i64 @llvm.riscv.macc.00.i64.v2i32(i64 %rd, <2 x i32> %a, <2 x i32> %b)
+ ret i64 %r
+}
+
+define i64 @test_macc_w01_i64(i64 %rd, <2 x i32> %a, <2 x i32> %b) {
+; RV32-LABEL: test_macc_w01_i64:
+; RV32: # %bb.0:
+; RV32-NEXT: wmacc a0, a2, a5
+; RV32-NEXT: ret
+;
+; RV64-LABEL: test_macc_w01_i64:
+; RV64: # %bb.0:
+; RV64-NEXT: macc.w01 a0, a1, a2
+; RV64-NEXT: ret
+ %r = call i64 @llvm.riscv.macc.01.i64.v2i32(i64 %rd, <2 x i32> %a, <2 x i32> %b)
+ ret i64 %r
+}
+
+define i64 @test_macc_w11_i64(i64 %rd, <2 x i32> %a, <2 x i32> %b) {
+; RV32-LABEL: test_macc_w11_i64:
+; RV32: # %bb.0:
+; RV32-NEXT: wmacc a0, a3, a5
+; RV32-NEXT: ret
+;
+; RV64-LABEL: test_macc_w11_i64:
+; RV64: # %bb.0:
+; RV64-NEXT: macc.w11 a0, a1, a2
+; RV64-NEXT: ret
+ %r = call i64 @llvm.riscv.macc.11.i64.v2i32(i64 %rd, <2 x i32> %a, <2 x i32> %b)
+ ret i64 %r
+}
+
+define i64 @test_maccu_w00_i64(i64 %rd, <2 x i32> %a, <2 x i32> %b) {
+; RV32-LABEL: test_maccu_w00_i64:
+; RV32: # %bb.0:
+; RV32-NEXT: wmaccu a0, a2, a4
+; RV32-NEXT: ret
+;
+; RV64-LABEL: test_maccu_w00_i64:
+; RV64: # %bb.0:
+; RV64-NEXT: maccu.w00 a0, a1, a2
+; RV64-NEXT: ret
+ %r = call i64 @llvm.riscv.maccu.00.i64.v2i32(i64 %rd, <2 x i32> %a, <2 x i32> %b)
+ ret i64 %r
+}
+
+define i64 @test_maccu_w01_i64(i64 %rd, <2 x i32> %a, <2 x i32> %b) {
+; RV32-LABEL: test_maccu_w01_i64:
+; RV32: # %bb.0:
+; RV32-NEXT: wmaccu a0, a2, a5
+; RV32-NEXT: ret
+;
+; RV64-LABEL: test_maccu_w01_i64:
+; RV64: # %bb.0:
+; RV64-NEXT: maccu.w01 a0, a1, a2
+; RV64-NEXT: ret
+ %r = call i64 @llvm.riscv.maccu.01.i64.v2i32(i64 %rd, <2 x i32> %a, <2 x i32> %b)
+ ret i64 %r
+}
+
+define i64 @test_maccu_w11_i64(i64 %rd, <2 x i32> %a, <2 x i32> %b) {
+; RV32-LABEL: test_maccu_w11_i64:
+; RV32: # %bb.0:
+; RV32-NEXT: wmaccu a0, a3, a5
+; RV32-NEXT: ret
+;
+; RV64-LABEL: test_maccu_w11_i64:
+; RV64: # %bb.0:
+; RV64-NEXT: maccu.w11 a0, a1, a2
+; RV64-NEXT: ret
+ %r = call i64 @llvm.riscv.maccu.11.i64.v2i32(i64 %rd, <2 x i32> %a, <2 x i32> %b)
+ ret i64 %r
+}
+
+define i64 @test_maccsu_w00_i64(i64 %rd, <2 x i32> %a, <2 x i32> %b) {
+; RV32-LABEL: test_maccsu_w00_i64:
+; RV32: # %bb.0:
+; RV32-NEXT: wmaccsu a0, a2, a4
+; RV32-NEXT: ret
+;
+; RV64-LABEL: test_maccsu_w00_i64:
+; RV64: # %bb.0:
+; RV64-NEXT: maccsu.w00 a0, a1, a2
+; RV64-NEXT: ret
+ %r = call i64 @llvm.riscv.maccsu.00.i64.v2i32(i64 %rd, <2 x i32> %a, <2 x i32> %b)
+ ret i64 %r
+}
+
+define i64 @test_maccsu_w11_i64(i64 %rd, <2 x i32> %a, <2 x i32> %b) {
+; RV32-LABEL: test_maccsu_w11_i64:
+; RV32: # %bb.0:
+; RV32-NEXT: wmaccsu a0, a3, a5
+; RV32-NEXT: ret
+;
+; RV64-LABEL: test_maccsu_w11_i64:
+; RV64: # %bb.0:
+; RV64-NEXT: maccsu.w11 a0, a1, a2
+; RV64-NEXT: ret
+ %r = call i64 @llvm.riscv.maccsu.11.i64.v2i32(i64 %rd, <2 x i32> %a, <2 x i32> %b)
+ ret i64 %r
+}
More information about the cfe-commits
mailing list