[llvm] [RISCV][P-ext] packed shift-add codegen (PR #201294)
via llvm-commits
llvm-commits at lists.llvm.org
Wed Jun 3 23:55:54 PDT 2026
https://github.com/sihuan updated https://github.com/llvm/llvm-project/pull/201294
>From f11df98e4a263777b70c9caf54242163dd0a9229 Mon Sep 17 00:00:00 2001
From: SiHuaN <liyongtai at iscas.ac.cn>
Date: Wed, 3 Jun 2026 15:45:51 +0800
Subject: [PATCH 1/2] [RISCV][P-ext] packed shift-add codegen
Add isel patterns for psh1add, pssh1sadd and the scalar ssh1sadd, matching
(a << 1) + b and sadd.sat(sadd.sat(a, a), b).
---
llvm/lib/Target/RISCV/RISCVInstrInfoP.td | 38 +++++++++
llvm/test/CodeGen/RISCV/rv32p.ll | 10 +++
llvm/test/CodeGen/RISCV/rvp-simd-32.ll | 29 ++++++-
llvm/test/CodeGen/RISCV/rvp-simd-64.ll | 98 +++++++++++++++++++-----
4 files changed, 153 insertions(+), 22 deletions(-)
diff --git a/llvm/lib/Target/RISCV/RISCVInstrInfoP.td b/llvm/lib/Target/RISCV/RISCVInstrInfoP.td
index 4c07d054c2983..11f78a9da256f 100644
--- a/llvm/lib/Target/RISCV/RISCVInstrInfoP.td
+++ b/llvm/lib/Target/RISCV/RISCVInstrInfoP.td
@@ -1889,6 +1889,26 @@ def RISCVBuildPairGPRVec : RVSDNode<"BuildPairGPRVec",
SDT_RISCVBuildPairGPRVec>;
def RISCVSplitGPRVec : RVSDNode<"SplitGPRVec", SDT_RISCVSplitGPRVec>;
+// (rs1 << 1) + rs2
+class PatPSh1Add<RVInst Inst, ValueType vt>
+ : Pat<(vt (add (riscv_pshl (vt GPR:$rs1), (XLenVT 1)), (vt GPR:$rs2))),
+ (Inst GPR:$rs1, GPR:$rs2)>;
+class PatPSh1AddPair<RVInst Inst, ValueType vt>
+ : Pat<(vt (add (riscv_pshl (vt GPRPair:$rs1), (XLenVT 1)),
+ (vt GPRPair:$rs2))),
+ (Inst GPRPair:$rs1, GPRPair:$rs2)>;
+
+// Signed-saturating (rs1 << 1) then signed-saturating add rs2; a saturating
+// left-shift-by-1 is a saturating add of the operand with itself.
+class PatPSSh1SAdd<RVInst Inst, ValueType vt>
+ : Pat<(vt (saddsat (saddsat (vt GPR:$rs1), (vt GPR:$rs1)),
+ (vt GPR:$rs2))),
+ (Inst GPR:$rs1, GPR:$rs2)>;
+class PatPSSh1SAddPair<RVInst Inst, ValueType vt>
+ : Pat<(vt (saddsat (saddsat (vt GPRPair:$rs1), (vt GPRPair:$rs1)),
+ (vt GPRPair:$rs2))),
+ (Inst GPRPair:$rs1, GPRPair:$rs2)>;
+
let Predicates = [HasStdExtP] in {
def : PatGpr<abs, ABS>;
@@ -1959,6 +1979,10 @@ let Predicates = [HasStdExtP] in {
def : PatGprGpr<ssubsat, PSSUB_H, XLenVecI16VT>;
def : PatGprGpr<usubsat, PSSUBU_H, XLenVecI16VT>;
+ // 16-bit shift-add patterns
+ def : PatPSh1Add<PSH1ADD_H, XLenVecI16VT>;
+ def : PatPSSh1SAdd<PSSH1SADD_H, XLenVecI16VT>;
+
// 8-bit averaging patterns
def : PatGprGpr<avgfloors, PAADD_B, XLenVecI8VT>;
def : PatGprGpr<avgflooru, PAADDU_B, XLenVecI8VT>;
@@ -2072,6 +2096,8 @@ let append Predicates = [IsRV32] in {
def : PatGprUimmLog2XLen<sshlsat, SSLAI>;
// No SSLLI
+ def : PatPSSh1SAdd<SSH1SADD, XLenVT>;
+
// 32-bit averaging patterns
def : PatGprGpr<avgfloors, AADD, i32>;
def : PatGprGpr<avgflooru, AADDU, i32>;
@@ -2173,6 +2199,14 @@ let append Predicates = [IsRV32] in {
def : PatGprPairGprPair<ssubsat, PSSUB_DW, v2i32>;
def : PatGprPairGprPair<usubsat, PSSUBU_DW, v2i32>;
+ // 16-bit shift-add patterns
+ def : PatPSh1AddPair<PSH1ADD_DH, v4i16>;
+ def : PatPSSh1SAddPair<PSSH1SADD_DH, v4i16>;
+
+ // 32-bit shift-add patterns
+ def : PatPSh1AddPair<PSH1ADD_DW, v2i32>;
+ def : PatPSSh1SAddPair<PSSH1SADD_DW, v2i32>;
+
// 8-bit averaging patterns
def : PatGprPairGprPair<avgfloors, PAADD_DB, v8i8>;
def : PatGprPairGprPair<avgflooru, PAADDU_DB, v8i8>;
@@ -2377,6 +2411,10 @@ let append Predicates = [IsRV64] in {
def : PatGprGpr<ssubsat, PSSUB_W, v2i32>;
def : PatGprGpr<usubsat, PSSUBU_W, v2i32>;
+ // 32-bit shift-add patterns
+ def : PatPSh1Add<PSH1ADD_W, v2i32>;
+ def : PatPSSh1SAdd<PSSH1SADD_W, v2i32>;
+
// 32-bit averaging patterns
def : PatGprGpr<avgfloors, PAADD_W, v2i32>;
def : PatGprGpr<avgflooru, PAADDU_W, v2i32>;
diff --git a/llvm/test/CodeGen/RISCV/rv32p.ll b/llvm/test/CodeGen/RISCV/rv32p.ll
index e2f6fd7e0192d..e627be54d3a08 100644
--- a/llvm/test/CodeGen/RISCV/rv32p.ll
+++ b/llvm/test/CodeGen/RISCV/rv32p.ll
@@ -732,6 +732,16 @@ define i32 @sadd_i32(i32 %x, i32 %y) {
ret i32 %a
}
+define i32 @ssh1sadd_i32(i32 %x, i32 %y) {
+; CHECK-LABEL: ssh1sadd_i32:
+; CHECK: # %bb.0:
+; CHECK-NEXT: ssh1sadd a0, a0, a1
+; CHECK-NEXT: ret
+ %shl = call i32 @llvm.sadd.sat.i32(i32 %x, i32 %x)
+ %a = call i32 @llvm.sadd.sat.i32(i32 %shl, i32 %y)
+ ret i32 %a
+}
+
define i8 @ssub_i8(i8 %x, i8 %y) {
; CHECK-LABEL: ssub_i8:
; CHECK: # %bb.0:
diff --git a/llvm/test/CodeGen/RISCV/rvp-simd-32.ll b/llvm/test/CodeGen/RISCV/rvp-simd-32.ll
index f7078dad860f5..a1a386625528c 100644
--- a/llvm/test/CodeGen/RISCV/rvp-simd-32.ll
+++ b/llvm/test/CodeGen/RISCV/rvp-simd-32.ll
@@ -218,6 +218,27 @@ define <2 x i16> @test_pssubu_h(<2 x i16> %a, <2 x i16> %b) {
ret <2 x i16> %res
}
+; Test shift-add operations for v2i16
+define <2 x i16> @test_psh1add_h(<2 x i16> %a, <2 x i16> %b) {
+; CHECK-LABEL: test_psh1add_h:
+; CHECK: # %bb.0:
+; CHECK-NEXT: psh1add.h a0, a0, a1
+; CHECK-NEXT: ret
+ %shl = shl <2 x i16> %a, splat (i16 1)
+ %res = add <2 x i16> %shl, %b
+ ret <2 x i16> %res
+}
+
+define <2 x i16> @test_pssh1sadd_h(<2 x i16> %a, <2 x i16> %b) {
+; CHECK-LABEL: test_pssh1sadd_h:
+; CHECK: # %bb.0:
+; CHECK-NEXT: pssh1sadd.h a0, a0, a1
+; CHECK-NEXT: ret
+ %shl = call <2 x i16> @llvm.sadd.sat.v2i16(<2 x i16> %a, <2 x i16> %a)
+ %res = call <2 x i16> @llvm.sadd.sat.v2i16(<2 x i16> %shl, <2 x i16> %b)
+ ret <2 x i16> %res
+}
+
; Test saturating add operations for v4i8
define <4 x i8> @test_psadd_b(<4 x i8> %a, <4 x i8> %b) {
; CHECK-LABEL: test_psadd_b:
@@ -2012,10 +2033,10 @@ define <2 x i16> @test_select_v2i16(i1 %cond, <2 x i16> %a, <2 x i16> %b) {
; CHECK: # %bb.0:
; CHECK-NEXT: andi a3, a0, 1
; CHECK-NEXT: mv a0, a1
-; CHECK-NEXT: bnez a3, .LBB144_2
+; CHECK-NEXT: bnez a3, .LBB146_2
; CHECK-NEXT: # %bb.1:
; CHECK-NEXT: mv a0, a2
-; CHECK-NEXT: .LBB144_2:
+; CHECK-NEXT: .LBB146_2:
; CHECK-NEXT: ret
%res = select i1 %cond, <2 x i16> %a, <2 x i16> %b
ret <2 x i16> %res
@@ -2026,10 +2047,10 @@ define <4 x i8> @test_select_v4i8(i1 %cond, <4 x i8> %a, <4 x i8> %b) {
; CHECK: # %bb.0:
; CHECK-NEXT: andi a3, a0, 1
; CHECK-NEXT: mv a0, a1
-; CHECK-NEXT: bnez a3, .LBB145_2
+; CHECK-NEXT: bnez a3, .LBB147_2
; CHECK-NEXT: # %bb.1:
; CHECK-NEXT: mv a0, a2
-; CHECK-NEXT: .LBB145_2:
+; CHECK-NEXT: .LBB147_2:
; CHECK-NEXT: ret
%res = select i1 %cond, <4 x i8> %a, <4 x i8> %b
ret <4 x i8> %res
diff --git a/llvm/test/CodeGen/RISCV/rvp-simd-64.ll b/llvm/test/CodeGen/RISCV/rvp-simd-64.ll
index 8d2d0f278e8c9..0df9a7c0b6fa1 100644
--- a/llvm/test/CodeGen/RISCV/rvp-simd-64.ll
+++ b/llvm/test/CodeGen/RISCV/rvp-simd-64.ll
@@ -449,6 +449,37 @@ define <4 x i16> @test_pssubu_h(<4 x i16> %a, <4 x i16> %b) {
ret <4 x i16> %res
}
+; Test shift-add operations for v4i16
+define <4 x i16> @test_psh1add_h(<4 x i16> %a, <4 x i16> %b) {
+; RV32-LABEL: test_psh1add_h:
+; RV32: # %bb.0:
+; RV32-NEXT: psh1add.dh a0, a0, a2
+; RV32-NEXT: ret
+;
+; RV64-LABEL: test_psh1add_h:
+; RV64: # %bb.0:
+; RV64-NEXT: psh1add.h a0, a0, a1
+; RV64-NEXT: ret
+ %shl = shl <4 x i16> %a, splat (i16 1)
+ %res = add <4 x i16> %shl, %b
+ ret <4 x i16> %res
+}
+
+define <4 x i16> @test_pssh1sadd_h(<4 x i16> %a, <4 x i16> %b) {
+; RV32-LABEL: test_pssh1sadd_h:
+; RV32: # %bb.0:
+; RV32-NEXT: pssh1sadd.dh a0, a0, a2
+; RV32-NEXT: ret
+;
+; RV64-LABEL: test_pssh1sadd_h:
+; RV64: # %bb.0:
+; RV64-NEXT: pssh1sadd.h a0, a0, a1
+; RV64-NEXT: ret
+ %shl = call <4 x i16> @llvm.sadd.sat.v4i16(<4 x i16> %a, <4 x i16> %a)
+ %res = call <4 x i16> @llvm.sadd.sat.v4i16(<4 x i16> %shl, <4 x i16> %b)
+ ret <4 x i16> %res
+}
+
; Test saturating add operations for v8i8
define <8 x i8> @test_psadd_b(<8 x i8> %a, <8 x i8> %b) {
; RV32-LABEL: test_psadd_b:
@@ -1252,6 +1283,37 @@ define <2 x i32> @test_pssubu_w(<2 x i32> %a, <2 x i32> %b) {
ret <2 x i32> %res
}
+; Test shift-add operations for v2i32
+define <2 x i32> @test_psh1add_w(<2 x i32> %a, <2 x i32> %b) {
+; RV32-LABEL: test_psh1add_w:
+; RV32: # %bb.0:
+; RV32-NEXT: psh1add.dw a0, a0, a2
+; RV32-NEXT: ret
+;
+; RV64-LABEL: test_psh1add_w:
+; RV64: # %bb.0:
+; RV64-NEXT: psh1add.w a0, a0, a1
+; RV64-NEXT: ret
+ %shl = shl <2 x i32> %a, splat (i32 1)
+ %res = add <2 x i32> %shl, %b
+ ret <2 x i32> %res
+}
+
+define <2 x i32> @test_pssh1sadd_w(<2 x i32> %a, <2 x i32> %b) {
+; RV32-LABEL: test_pssh1sadd_w:
+; RV32: # %bb.0:
+; RV32-NEXT: pssh1sadd.dw a0, a0, a2
+; RV32-NEXT: ret
+;
+; RV64-LABEL: test_pssh1sadd_w:
+; RV64: # %bb.0:
+; RV64-NEXT: pssh1sadd.w a0, a0, a1
+; RV64-NEXT: ret
+ %shl = call <2 x i32> @llvm.sadd.sat.v2i32(<2 x i32> %a, <2 x i32> %a)
+ %res = call <2 x i32> @llvm.sadd.sat.v2i32(<2 x i32> %shl, <2 x i32> %b)
+ ret <2 x i32> %res
+}
+
; Test averaging floor signed operations for v2i32 (RV64 only)
; avgfloors pattern: (a + b) arithmetic shift right 1
define <2 x i32> @test_paadd_w(<2 x i32> %a, <2 x i32> %b) {
@@ -3988,12 +4050,12 @@ define <4 x i16> @test_select_v4i16(i1 %cond, <4 x i16> %a, <4 x i16> %b) {
; RV32-LABEL: test_select_v4i16:
; RV32: # %bb.0:
; RV32-NEXT: andi a5, a0, 1
-; RV32-NEXT: bnez a5, .LBB211_2
+; RV32-NEXT: bnez a5, .LBB215_2
; RV32-NEXT: # %bb.1:
; RV32-NEXT: mv a0, a3
; RV32-NEXT: mv a1, a4
; RV32-NEXT: ret
-; RV32-NEXT: .LBB211_2:
+; RV32-NEXT: .LBB215_2:
; RV32-NEXT: mv a0, a1
; RV32-NEXT: mv a1, a2
; RV32-NEXT: ret
@@ -4002,10 +4064,10 @@ define <4 x i16> @test_select_v4i16(i1 %cond, <4 x i16> %a, <4 x i16> %b) {
; RV64: # %bb.0:
; RV64-NEXT: andi a3, a0, 1
; RV64-NEXT: mv a0, a1
-; RV64-NEXT: bnez a3, .LBB211_2
+; RV64-NEXT: bnez a3, .LBB215_2
; RV64-NEXT: # %bb.1:
; RV64-NEXT: mv a0, a2
-; RV64-NEXT: .LBB211_2:
+; RV64-NEXT: .LBB215_2:
; RV64-NEXT: ret
%res = select i1 %cond, <4 x i16> %a, <4 x i16> %b
ret <4 x i16> %res
@@ -4015,12 +4077,12 @@ define <8 x i8> @test_select_v8i8(i1 %cond, <8 x i8> %a, <8 x i8> %b) {
; RV32-LABEL: test_select_v8i8:
; RV32: # %bb.0:
; RV32-NEXT: andi a5, a0, 1
-; RV32-NEXT: bnez a5, .LBB212_2
+; RV32-NEXT: bnez a5, .LBB216_2
; RV32-NEXT: # %bb.1:
; RV32-NEXT: mv a0, a3
; RV32-NEXT: mv a1, a4
; RV32-NEXT: ret
-; RV32-NEXT: .LBB212_2:
+; RV32-NEXT: .LBB216_2:
; RV32-NEXT: mv a0, a1
; RV32-NEXT: mv a1, a2
; RV32-NEXT: ret
@@ -4029,10 +4091,10 @@ define <8 x i8> @test_select_v8i8(i1 %cond, <8 x i8> %a, <8 x i8> %b) {
; RV64: # %bb.0:
; RV64-NEXT: andi a3, a0, 1
; RV64-NEXT: mv a0, a1
-; RV64-NEXT: bnez a3, .LBB212_2
+; RV64-NEXT: bnez a3, .LBB216_2
; RV64-NEXT: # %bb.1:
; RV64-NEXT: mv a0, a2
-; RV64-NEXT: .LBB212_2:
+; RV64-NEXT: .LBB216_2:
; RV64-NEXT: ret
%res = select i1 %cond, <8 x i8> %a, <8 x i8> %b
ret <8 x i8> %res
@@ -4042,12 +4104,12 @@ define <2 x i32> @test_select_v2i32(i1 %cond, <2 x i32> %a, <2 x i32> %b) {
; RV32-LABEL: test_select_v2i32:
; RV32: # %bb.0:
; RV32-NEXT: andi a5, a0, 1
-; RV32-NEXT: bnez a5, .LBB213_2
+; RV32-NEXT: bnez a5, .LBB217_2
; RV32-NEXT: # %bb.1:
; RV32-NEXT: mv a0, a3
; RV32-NEXT: mv a1, a4
; RV32-NEXT: ret
-; RV32-NEXT: .LBB213_2:
+; RV32-NEXT: .LBB217_2:
; RV32-NEXT: mv a0, a1
; RV32-NEXT: mv a1, a2
; RV32-NEXT: ret
@@ -4056,10 +4118,10 @@ define <2 x i32> @test_select_v2i32(i1 %cond, <2 x i32> %a, <2 x i32> %b) {
; RV64: # %bb.0:
; RV64-NEXT: andi a3, a0, 1
; RV64-NEXT: mv a0, a1
-; RV64-NEXT: bnez a3, .LBB213_2
+; RV64-NEXT: bnez a3, .LBB217_2
; RV64-NEXT: # %bb.1:
; RV64-NEXT: mv a0, a2
-; RV64-NEXT: .LBB213_2:
+; RV64-NEXT: .LBB217_2:
; RV64-NEXT: ret
%res = select i1 %cond, <2 x i32> %a, <2 x i32> %b
ret <2 x i32> %res
@@ -4107,16 +4169,16 @@ define <2 x i32> @test_vselect_v2i32(<2 x i32> %a, <2 x i32> %b, <2 x i32> %c) {
; RV32: # %bb.0:
; RV32-NEXT: pmslt.dw a6, a2, a0
; RV32-NEXT: mv a0, a4
-; RV32-NEXT: beqz a7, .LBB216_3
+; RV32-NEXT: beqz a7, .LBB220_3
; RV32-NEXT: # %bb.1:
-; RV32-NEXT: beqz a6, .LBB216_4
-; RV32-NEXT: .LBB216_2:
+; RV32-NEXT: beqz a6, .LBB220_4
+; RV32-NEXT: .LBB220_2:
; RV32-NEXT: mv a1, a5
; RV32-NEXT: ret
-; RV32-NEXT: .LBB216_3:
+; RV32-NEXT: .LBB220_3:
; RV32-NEXT: mv a5, a3
-; RV32-NEXT: bnez a6, .LBB216_2
-; RV32-NEXT: .LBB216_4:
+; RV32-NEXT: bnez a6, .LBB220_2
+; RV32-NEXT: .LBB220_4:
; RV32-NEXT: mv a0, a2
; RV32-NEXT: mv a1, a5
; RV32-NEXT: ret
>From 2adb366df63d64c5d5f25fc4d6255b31f8fe05ae Mon Sep 17 00:00:00 2001
From: SiHuaN <liyongtai at iscas.ac.cn>
Date: Thu, 4 Jun 2026 14:27:43 +0800
Subject: [PATCH 2/2] also match sshl.sat + sadd.sat
---
llvm/lib/Target/RISCV/RISCVInstrInfoP.td | 22 ++++----
llvm/test/CodeGen/RISCV/rv32p.ll | 10 ++++
llvm/test/CodeGen/RISCV/rvp-simd-32.ll | 18 +++++--
llvm/test/CodeGen/RISCV/rvp-simd-64.ll | 66 +++++++++++++++++-------
4 files changed, 85 insertions(+), 31 deletions(-)
diff --git a/llvm/lib/Target/RISCV/RISCVInstrInfoP.td b/llvm/lib/Target/RISCV/RISCVInstrInfoP.td
index 11f78a9da256f..ae8c21ecdbc51 100644
--- a/llvm/lib/Target/RISCV/RISCVInstrInfoP.td
+++ b/llvm/lib/Target/RISCV/RISCVInstrInfoP.td
@@ -1894,19 +1894,20 @@ class PatPSh1Add<RVInst Inst, ValueType vt>
: Pat<(vt (add (riscv_pshl (vt GPR:$rs1), (XLenVT 1)), (vt GPR:$rs2))),
(Inst GPR:$rs1, GPR:$rs2)>;
class PatPSh1AddPair<RVInst Inst, ValueType vt>
- : Pat<(vt (add (riscv_pshl (vt GPRPair:$rs1), (XLenVT 1)),
- (vt GPRPair:$rs2))),
+ : Pat<(vt (add (riscv_pshl (vt GPRPair:$rs1), (XLenVT 1)), (vt GPRPair:$rs2))),
(Inst GPRPair:$rs1, GPRPair:$rs2)>;
-// Signed-saturating (rs1 << 1) then signed-saturating add rs2; a saturating
-// left-shift-by-1 is a saturating add of the operand with itself.
+// Vector sshlsat is custom-lowered to riscv_pssha before isel; saddsat(a, a)
+// is an equivalent shape via the signed sshlsat(a, 1) identity.
+def saturating_shl1 : PatFrags<(ops node:$a),
+ [(riscv_pssha node:$a, (XLenVT 1)),
+ (saddsat node:$a, node:$a)]>;
+
class PatPSSh1SAdd<RVInst Inst, ValueType vt>
- : Pat<(vt (saddsat (saddsat (vt GPR:$rs1), (vt GPR:$rs1)),
- (vt GPR:$rs2))),
+ : Pat<(vt (saddsat (saturating_shl1 (vt GPR:$rs1)), (vt GPR:$rs2))),
(Inst GPR:$rs1, GPR:$rs2)>;
class PatPSSh1SAddPair<RVInst Inst, ValueType vt>
- : Pat<(vt (saddsat (saddsat (vt GPRPair:$rs1), (vt GPRPair:$rs1)),
- (vt GPRPair:$rs2))),
+ : Pat<(vt (saddsat (saturating_shl1 (vt GPRPair:$rs1)), (vt GPRPair:$rs2))),
(Inst GPRPair:$rs1, GPRPair:$rs2)>;
let Predicates = [HasStdExtP] in {
@@ -2096,7 +2097,10 @@ let append Predicates = [IsRV32] in {
def : PatGprUimmLog2XLen<sshlsat, SSLAI>;
// No SSLLI
- def : PatPSSh1SAdd<SSH1SADD, XLenVT>;
+ def : Pat<(XLenVT (saddsat (sshlsat GPR:$rs1, (XLenVT 1)), GPR:$rs2)),
+ (SSH1SADD GPR:$rs1, GPR:$rs2)>;
+ def : Pat<(XLenVT (saddsat (saddsat GPR:$rs1, GPR:$rs1), GPR:$rs2)),
+ (SSH1SADD GPR:$rs1, GPR:$rs2)>;
// 32-bit averaging patterns
def : PatGprGpr<avgfloors, AADD, i32>;
diff --git a/llvm/test/CodeGen/RISCV/rv32p.ll b/llvm/test/CodeGen/RISCV/rv32p.ll
index e627be54d3a08..a64db5e1efbbc 100644
--- a/llvm/test/CodeGen/RISCV/rv32p.ll
+++ b/llvm/test/CodeGen/RISCV/rv32p.ll
@@ -736,6 +736,16 @@ define i32 @ssh1sadd_i32(i32 %x, i32 %y) {
; CHECK-LABEL: ssh1sadd_i32:
; CHECK: # %bb.0:
; CHECK-NEXT: ssh1sadd a0, a0, a1
+; CHECK-NEXT: ret
+ %shl = call i32 @llvm.sshl.sat.i32(i32 %x, i32 1)
+ %a = call i32 @llvm.sadd.sat.i32(i32 %shl, i32 %y)
+ ret i32 %a
+}
+
+define i32 @ssh1sadd_i32_addself(i32 %x, i32 %y) {
+; CHECK-LABEL: ssh1sadd_i32_addself:
+; CHECK: # %bb.0:
+; CHECK-NEXT: ssh1sadd a0, a0, a1
; CHECK-NEXT: ret
%shl = call i32 @llvm.sadd.sat.i32(i32 %x, i32 %x)
%a = call i32 @llvm.sadd.sat.i32(i32 %shl, i32 %y)
diff --git a/llvm/test/CodeGen/RISCV/rvp-simd-32.ll b/llvm/test/CodeGen/RISCV/rvp-simd-32.ll
index a1a386625528c..88e4d0c0f6297 100644
--- a/llvm/test/CodeGen/RISCV/rvp-simd-32.ll
+++ b/llvm/test/CodeGen/RISCV/rvp-simd-32.ll
@@ -233,6 +233,16 @@ define <2 x i16> @test_pssh1sadd_h(<2 x i16> %a, <2 x i16> %b) {
; CHECK-LABEL: test_pssh1sadd_h:
; CHECK: # %bb.0:
; CHECK-NEXT: pssh1sadd.h a0, a0, a1
+; CHECK-NEXT: ret
+ %shl = call <2 x i16> @llvm.sshl.sat.v2i16(<2 x i16> %a, <2 x i16> splat (i16 1))
+ %res = call <2 x i16> @llvm.sadd.sat.v2i16(<2 x i16> %shl, <2 x i16> %b)
+ ret <2 x i16> %res
+}
+
+define <2 x i16> @test_pssh1sadd_h_addself(<2 x i16> %a, <2 x i16> %b) {
+; CHECK-LABEL: test_pssh1sadd_h_addself:
+; CHECK: # %bb.0:
+; CHECK-NEXT: pssh1sadd.h a0, a0, a1
; CHECK-NEXT: ret
%shl = call <2 x i16> @llvm.sadd.sat.v2i16(<2 x i16> %a, <2 x i16> %a)
%res = call <2 x i16> @llvm.sadd.sat.v2i16(<2 x i16> %shl, <2 x i16> %b)
@@ -2033,10 +2043,10 @@ define <2 x i16> @test_select_v2i16(i1 %cond, <2 x i16> %a, <2 x i16> %b) {
; CHECK: # %bb.0:
; CHECK-NEXT: andi a3, a0, 1
; CHECK-NEXT: mv a0, a1
-; CHECK-NEXT: bnez a3, .LBB146_2
+; CHECK-NEXT: bnez a3, .LBB147_2
; CHECK-NEXT: # %bb.1:
; CHECK-NEXT: mv a0, a2
-; CHECK-NEXT: .LBB146_2:
+; CHECK-NEXT: .LBB147_2:
; CHECK-NEXT: ret
%res = select i1 %cond, <2 x i16> %a, <2 x i16> %b
ret <2 x i16> %res
@@ -2047,10 +2057,10 @@ define <4 x i8> @test_select_v4i8(i1 %cond, <4 x i8> %a, <4 x i8> %b) {
; CHECK: # %bb.0:
; CHECK-NEXT: andi a3, a0, 1
; CHECK-NEXT: mv a0, a1
-; CHECK-NEXT: bnez a3, .LBB147_2
+; CHECK-NEXT: bnez a3, .LBB148_2
; CHECK-NEXT: # %bb.1:
; CHECK-NEXT: mv a0, a2
-; CHECK-NEXT: .LBB147_2:
+; CHECK-NEXT: .LBB148_2:
; CHECK-NEXT: ret
%res = select i1 %cond, <4 x i8> %a, <4 x i8> %b
ret <4 x i8> %res
diff --git a/llvm/test/CodeGen/RISCV/rvp-simd-64.ll b/llvm/test/CodeGen/RISCV/rvp-simd-64.ll
index 0df9a7c0b6fa1..db7db027f47c1 100644
--- a/llvm/test/CodeGen/RISCV/rvp-simd-64.ll
+++ b/llvm/test/CodeGen/RISCV/rvp-simd-64.ll
@@ -474,6 +474,21 @@ define <4 x i16> @test_pssh1sadd_h(<4 x i16> %a, <4 x i16> %b) {
; RV64-LABEL: test_pssh1sadd_h:
; RV64: # %bb.0:
; RV64-NEXT: pssh1sadd.h a0, a0, a1
+; RV64-NEXT: ret
+ %shl = call <4 x i16> @llvm.sshl.sat.v4i16(<4 x i16> %a, <4 x i16> splat (i16 1))
+ %res = call <4 x i16> @llvm.sadd.sat.v4i16(<4 x i16> %shl, <4 x i16> %b)
+ ret <4 x i16> %res
+}
+
+define <4 x i16> @test_pssh1sadd_h_addself(<4 x i16> %a, <4 x i16> %b) {
+; RV32-LABEL: test_pssh1sadd_h_addself:
+; RV32: # %bb.0:
+; RV32-NEXT: pssh1sadd.dh a0, a0, a2
+; RV32-NEXT: ret
+;
+; RV64-LABEL: test_pssh1sadd_h_addself:
+; RV64: # %bb.0:
+; RV64-NEXT: pssh1sadd.h a0, a0, a1
; RV64-NEXT: ret
%shl = call <4 x i16> @llvm.sadd.sat.v4i16(<4 x i16> %a, <4 x i16> %a)
%res = call <4 x i16> @llvm.sadd.sat.v4i16(<4 x i16> %shl, <4 x i16> %b)
@@ -1308,6 +1323,21 @@ define <2 x i32> @test_pssh1sadd_w(<2 x i32> %a, <2 x i32> %b) {
; RV64-LABEL: test_pssh1sadd_w:
; RV64: # %bb.0:
; RV64-NEXT: pssh1sadd.w a0, a0, a1
+; RV64-NEXT: ret
+ %shl = call <2 x i32> @llvm.sshl.sat.v2i32(<2 x i32> %a, <2 x i32> splat (i32 1))
+ %res = call <2 x i32> @llvm.sadd.sat.v2i32(<2 x i32> %shl, <2 x i32> %b)
+ ret <2 x i32> %res
+}
+
+define <2 x i32> @test_pssh1sadd_w_addself(<2 x i32> %a, <2 x i32> %b) {
+; RV32-LABEL: test_pssh1sadd_w_addself:
+; RV32: # %bb.0:
+; RV32-NEXT: pssh1sadd.dw a0, a0, a2
+; RV32-NEXT: ret
+;
+; RV64-LABEL: test_pssh1sadd_w_addself:
+; RV64: # %bb.0:
+; RV64-NEXT: pssh1sadd.w a0, a0, a1
; RV64-NEXT: ret
%shl = call <2 x i32> @llvm.sadd.sat.v2i32(<2 x i32> %a, <2 x i32> %a)
%res = call <2 x i32> @llvm.sadd.sat.v2i32(<2 x i32> %shl, <2 x i32> %b)
@@ -4050,12 +4080,12 @@ define <4 x i16> @test_select_v4i16(i1 %cond, <4 x i16> %a, <4 x i16> %b) {
; RV32-LABEL: test_select_v4i16:
; RV32: # %bb.0:
; RV32-NEXT: andi a5, a0, 1
-; RV32-NEXT: bnez a5, .LBB215_2
+; RV32-NEXT: bnez a5, .LBB217_2
; RV32-NEXT: # %bb.1:
; RV32-NEXT: mv a0, a3
; RV32-NEXT: mv a1, a4
; RV32-NEXT: ret
-; RV32-NEXT: .LBB215_2:
+; RV32-NEXT: .LBB217_2:
; RV32-NEXT: mv a0, a1
; RV32-NEXT: mv a1, a2
; RV32-NEXT: ret
@@ -4064,10 +4094,10 @@ define <4 x i16> @test_select_v4i16(i1 %cond, <4 x i16> %a, <4 x i16> %b) {
; RV64: # %bb.0:
; RV64-NEXT: andi a3, a0, 1
; RV64-NEXT: mv a0, a1
-; RV64-NEXT: bnez a3, .LBB215_2
+; RV64-NEXT: bnez a3, .LBB217_2
; RV64-NEXT: # %bb.1:
; RV64-NEXT: mv a0, a2
-; RV64-NEXT: .LBB215_2:
+; RV64-NEXT: .LBB217_2:
; RV64-NEXT: ret
%res = select i1 %cond, <4 x i16> %a, <4 x i16> %b
ret <4 x i16> %res
@@ -4077,12 +4107,12 @@ define <8 x i8> @test_select_v8i8(i1 %cond, <8 x i8> %a, <8 x i8> %b) {
; RV32-LABEL: test_select_v8i8:
; RV32: # %bb.0:
; RV32-NEXT: andi a5, a0, 1
-; RV32-NEXT: bnez a5, .LBB216_2
+; RV32-NEXT: bnez a5, .LBB218_2
; RV32-NEXT: # %bb.1:
; RV32-NEXT: mv a0, a3
; RV32-NEXT: mv a1, a4
; RV32-NEXT: ret
-; RV32-NEXT: .LBB216_2:
+; RV32-NEXT: .LBB218_2:
; RV32-NEXT: mv a0, a1
; RV32-NEXT: mv a1, a2
; RV32-NEXT: ret
@@ -4091,10 +4121,10 @@ define <8 x i8> @test_select_v8i8(i1 %cond, <8 x i8> %a, <8 x i8> %b) {
; RV64: # %bb.0:
; RV64-NEXT: andi a3, a0, 1
; RV64-NEXT: mv a0, a1
-; RV64-NEXT: bnez a3, .LBB216_2
+; RV64-NEXT: bnez a3, .LBB218_2
; RV64-NEXT: # %bb.1:
; RV64-NEXT: mv a0, a2
-; RV64-NEXT: .LBB216_2:
+; RV64-NEXT: .LBB218_2:
; RV64-NEXT: ret
%res = select i1 %cond, <8 x i8> %a, <8 x i8> %b
ret <8 x i8> %res
@@ -4104,12 +4134,12 @@ define <2 x i32> @test_select_v2i32(i1 %cond, <2 x i32> %a, <2 x i32> %b) {
; RV32-LABEL: test_select_v2i32:
; RV32: # %bb.0:
; RV32-NEXT: andi a5, a0, 1
-; RV32-NEXT: bnez a5, .LBB217_2
+; RV32-NEXT: bnez a5, .LBB219_2
; RV32-NEXT: # %bb.1:
; RV32-NEXT: mv a0, a3
; RV32-NEXT: mv a1, a4
; RV32-NEXT: ret
-; RV32-NEXT: .LBB217_2:
+; RV32-NEXT: .LBB219_2:
; RV32-NEXT: mv a0, a1
; RV32-NEXT: mv a1, a2
; RV32-NEXT: ret
@@ -4118,10 +4148,10 @@ define <2 x i32> @test_select_v2i32(i1 %cond, <2 x i32> %a, <2 x i32> %b) {
; RV64: # %bb.0:
; RV64-NEXT: andi a3, a0, 1
; RV64-NEXT: mv a0, a1
-; RV64-NEXT: bnez a3, .LBB217_2
+; RV64-NEXT: bnez a3, .LBB219_2
; RV64-NEXT: # %bb.1:
; RV64-NEXT: mv a0, a2
-; RV64-NEXT: .LBB217_2:
+; RV64-NEXT: .LBB219_2:
; RV64-NEXT: ret
%res = select i1 %cond, <2 x i32> %a, <2 x i32> %b
ret <2 x i32> %res
@@ -4169,16 +4199,16 @@ define <2 x i32> @test_vselect_v2i32(<2 x i32> %a, <2 x i32> %b, <2 x i32> %c) {
; RV32: # %bb.0:
; RV32-NEXT: pmslt.dw a6, a2, a0
; RV32-NEXT: mv a0, a4
-; RV32-NEXT: beqz a7, .LBB220_3
+; RV32-NEXT: beqz a7, .LBB222_3
; RV32-NEXT: # %bb.1:
-; RV32-NEXT: beqz a6, .LBB220_4
-; RV32-NEXT: .LBB220_2:
+; RV32-NEXT: beqz a6, .LBB222_4
+; RV32-NEXT: .LBB222_2:
; RV32-NEXT: mv a1, a5
; RV32-NEXT: ret
-; RV32-NEXT: .LBB220_3:
+; RV32-NEXT: .LBB222_3:
; RV32-NEXT: mv a5, a3
-; RV32-NEXT: bnez a6, .LBB220_2
-; RV32-NEXT: .LBB220_4:
+; RV32-NEXT: bnez a6, .LBB222_2
+; RV32-NEXT: .LBB222_4:
; RV32-NEXT: mv a0, a2
; RV32-NEXT: mv a1, a5
; RV32-NEXT: ret
More information about the llvm-commits
mailing list