[clang] [llvm] [Clang][RISCV] Add packed saturating and rounding shift intrinsics (PR #217692)
via llvm-commits
llvm-commits at lists.llvm.org
Thu Aug 20 09:52:21 PDT 2026
https://github.com/sihuan created https://github.com/llvm/llvm-project/pull/217692
Add builtins and header wrappers for `__riscv_pssha`, `__riscv_psshar`,
`__riscv_psshl` and `__riscv_psshlr`.
Depends on #217688, which the immediate-form tests need; only the last
commit is new.
>From 3c7e7ecc1fb90039b3d42d1f834c93f58dd66700 Mon Sep 17 00:00:00 2001
From: SiHuaN <liyongtai at iscas.ac.cn>
Date: Wed, 19 Aug 2026 04:56:38 +0000
Subject: [PATCH 1/2] [RISCV][P-ext] Select immediate forms for packed
saturating shifts
pssha/psshar take a signed shift amount, so a constant selects an immediate
form: psslai when non-negative, psrai/psrari when negative. Only pssha with
a non-negative constant was handled. Magnitudes that do not fit the field
keep the register form, where the hardware clamps them to a full shift that
psrari does not reproduce.
---
llvm/lib/Target/RISCV/RISCVInstrInfoP.td | 28 +++-
llvm/test/CodeGen/RISCV/rvp-simd-32.ll | 64 +++++++-
llvm/test/CodeGen/RISCV/rvp-simd-64.ll | 179 ++++++++++++++++++++---
3 files changed, 245 insertions(+), 26 deletions(-)
diff --git a/llvm/lib/Target/RISCV/RISCVInstrInfoP.td b/llvm/lib/Target/RISCV/RISCVInstrInfoP.td
index f09df843e0c40..e156287fc7706 100644
--- a/llvm/lib/Target/RISCV/RISCVInstrInfoP.td
+++ b/llvm/lib/Target/RISCV/RISCVInstrInfoP.td
@@ -118,6 +118,14 @@ def uimm4_plus1 : RISCVUImmPlus1Op<4>;
def uimm5_plus1 : RISCVUImmPlus1Op<5>;
def uimm6_plus1 : RISCVUImmPlus1Op<6>;
+// Matches the negation of an N-bit unsigned immediate, rendering the magnitude.
+class RISCVUImmNegLeaf<int bitsNum>
+ : ImmLeaf<XLenVT, "return Imm < 0 && isUInt<" # bitsNum # ">(-Imm);",
+ NegImm>;
+
+def uimm4_neg : RISCVUImmNegLeaf<4>;
+def uimm5_neg : RISCVUImmNegLeaf<5>;
+
//===----------------------------------------------------------------------===//
// Instruction class templates
//===----------------------------------------------------------------------===//
@@ -2144,8 +2152,11 @@ let Predicates = [HasStdExtP] in {
// 16-bit arithmetic shift right patterns
def : PatGprImm<riscv_psra, PSRAI_H, uimm4, XLenVecI16VT>;
- // 16-bit signed saturation shift left patterns
+ // 16-bit saturating shift patterns
def : PatGprImm<riscv_pssha, PSSLAI_H, uimm4, XLenVecI16VT>;
+ def : PatGprImm<riscv_psshar, PSSLAI_H, uimm4, XLenVecI16VT>;
+ def : PatGprImm<riscv_pssha, PSRAI_H, uimm4_neg, XLenVecI16VT>;
+ def : PatGprImm<riscv_psshar, PSRARI_H, uimm4_neg, XLenVecI16VT>;
def : PatGprShift<riscv_pssha, PSSHA_HS, XLenVecI16VT>;
def : PatGprShift<riscv_psshar, PSSHAR_HS, XLenVecI16VT>;
def : PatGprShift<riscv_psshl, PSSHL_HS, XLenVecI16VT>;
@@ -2500,15 +2511,21 @@ let append Predicates = [IsRV32] in {
// 32-bit arithmetic shift left/right patterns
def : PatGprPairImm<riscv_psra, PSRAI_DW, uimm5, v2i32>;
- // 16-bit signed saturation shift left patterns
+ // 16-bit saturating shift patterns
def : PatGprPairImm<riscv_pssha, PSSLAI_DH, uimm4, v4i16>;
+ def : PatGprPairImm<riscv_psshar, PSSLAI_DH, uimm4, v4i16>;
+ def : PatGprPairImm<riscv_pssha, PSRAI_DH, uimm4_neg, v4i16>;
+ def : PatGprPairImm<riscv_psshar, PSRARI_DH, uimm4_neg, v4i16>;
def : PatGprPairShift<riscv_pssha, PSSHA_DHS, v4i16>;
def : PatGprPairShift<riscv_psshar, PSSHAR_DHS, v4i16>;
def : PatGprPairShift<riscv_psshl, PSSHL_DHS, v4i16>;
def : PatGprPairShift<riscv_psshlr, PSSHLR_DHS, v4i16>;
- // 32-bit signed saturation shift left patterns
+ // 32-bit saturating shift patterns
def : PatGprPairImm<riscv_pssha, PSSLAI_DW, uimm5, v2i32>;
+ def : PatGprPairImm<riscv_psshar, PSSLAI_DW, uimm5, v2i32>;
+ def : PatGprPairImm<riscv_pssha, PSRAI_DW, uimm5_neg, v2i32>;
+ def : PatGprPairImm<riscv_psshar, PSRARI_DW, uimm5_neg, v2i32>;
def : PatGprPairShift<riscv_pssha, PSSHA_DWS, v2i32>;
def : PatGprPairShift<riscv_psshar, PSSHAR_DWS, v2i32>;
def : PatGprPairShift<riscv_psshl, PSSHL_DWS, v2i32>;
@@ -2810,8 +2827,11 @@ let append Predicates = [IsRV64] in {
// 32-bit arithmetic shift left/right patterns
def : PatGprImm<riscv_psra, PSRAI_W, uimm5, v2i32>;
- // 32-bit signed saturation shift left patterns
+ // 32-bit saturating shift patterns
def : PatGprImm<riscv_pssha, PSSLAI_W, uimm5, v2i32>;
+ def : PatGprImm<riscv_psshar, PSSLAI_W, uimm5, v2i32>;
+ def : PatGprImm<riscv_pssha, PSRAI_W, uimm5_neg, v2i32>;
+ def : PatGprImm<riscv_psshar, PSRARI_W, uimm5_neg, v2i32>;
def : PatGprShift<riscv_pssha, PSSHA_WS, v2i32>;
def : PatGprShift<riscv_psshar, PSSHAR_WS, v2i32>;
def : PatGprShift<riscv_psshl, PSSHL_WS, v2i32>;
diff --git a/llvm/test/CodeGen/RISCV/rvp-simd-32.ll b/llvm/test/CodeGen/RISCV/rvp-simd-32.ll
index 8f96ec6e8a811..5961c4f4ce940 100644
--- a/llvm/test/CodeGen/RISCV/rvp-simd-32.ll
+++ b/llvm/test/CodeGen/RISCV/rvp-simd-32.ll
@@ -1300,6 +1300,62 @@ define <2 x i16> @test_psshlr_s_u16x2(<2 x i16> %a, i32 %shamt) {
ret <2 x i16> %res
}
+; Constant shift amounts select the immediate forms.
+define <2 x i16> @test_pssha_s_i16x2_imm(<2 x i16> %a) {
+; CHECK-LABEL: test_pssha_s_i16x2_imm:
+; CHECK: # %bb.0:
+; CHECK-NEXT: psslai.h a0, a0, 3
+; CHECK-NEXT: ret
+ %res = call <2 x i16> @llvm.riscv.pssha.v2i16.i32(<2 x i16> %a, i32 3)
+ ret <2 x i16> %res
+}
+
+define <2 x i16> @test_psshar_s_i16x2_imm(<2 x i16> %a) {
+; CHECK-LABEL: test_psshar_s_i16x2_imm:
+; CHECK: # %bb.0:
+; CHECK-NEXT: psslai.h a0, a0, 3
+; CHECK-NEXT: ret
+ %res = call <2 x i16> @llvm.riscv.psshar.v2i16.i32(<2 x i16> %a, i32 3)
+ ret <2 x i16> %res
+}
+
+define <2 x i16> @test_pssha_s_i16x2_neg_imm(<2 x i16> %a) {
+; CHECK-LABEL: test_pssha_s_i16x2_neg_imm:
+; CHECK: # %bb.0:
+; CHECK-NEXT: psrai.h a0, a0, 3
+; CHECK-NEXT: ret
+ %res = call <2 x i16> @llvm.riscv.pssha.v2i16.i32(<2 x i16> %a, i32 -3)
+ ret <2 x i16> %res
+}
+
+define <2 x i16> @test_psshar_s_i16x2_neg_imm(<2 x i16> %a) {
+; CHECK-LABEL: test_psshar_s_i16x2_neg_imm:
+; CHECK: # %bb.0:
+; CHECK-NEXT: psrari.h a0, a0, 3
+; CHECK-NEXT: ret
+ %res = call <2 x i16> @llvm.riscv.psshar.v2i16.i32(<2 x i16> %a, i32 -3)
+ ret <2 x i16> %res
+}
+
+define <2 x i16> @test_pssha_s_i16x2_neg_imm_max(<2 x i16> %a) {
+; CHECK-LABEL: test_pssha_s_i16x2_neg_imm_max:
+; CHECK: # %bb.0:
+; CHECK-NEXT: psrai.h a0, a0, 15
+; CHECK-NEXT: ret
+ %res = call <2 x i16> @llvm.riscv.pssha.v2i16.i32(<2 x i16> %a, i32 -15)
+ ret <2 x i16> %res
+}
+
+define <2 x i16> @test_pssha_s_i16x2_neg_imm_too_large(<2 x i16> %a) {
+; CHECK-LABEL: test_pssha_s_i16x2_neg_imm_too_large:
+; CHECK: # %bb.0:
+; CHECK-NEXT: li a1, -16
+; CHECK-NEXT: pssha.hs a0, a0, a1
+; CHECK-NEXT: ret
+ %res = call <2 x i16> @llvm.riscv.pssha.v2i16.i32(<2 x i16> %a, i32 -16)
+ ret <2 x i16> %res
+}
+
; Test packed multiply high signed for v4i8
define <4 x i8> @test_pmulh_b(<4 x i8> %a, <4 x i8> %b) {
; RV32-LABEL: test_pmulh_b:
@@ -2258,10 +2314,10 @@ define <2 x i16> @test_select_v2i16(i1 %cond, <2 x i16> %a, <2 x i16> %b) {
; CHECK: # %bb.0:
; CHECK-NEXT: andi a3, a0, 1
; CHECK-NEXT: mv a0, a1
-; CHECK-NEXT: bnez a3, .LBB158_2
+; CHECK-NEXT: bnez a3, .LBB164_2
; CHECK-NEXT: # %bb.1:
; CHECK-NEXT: mv a0, a2
-; CHECK-NEXT: .LBB158_2:
+; CHECK-NEXT: .LBB164_2:
; CHECK-NEXT: ret
%res = select i1 %cond, <2 x i16> %a, <2 x i16> %b
ret <2 x i16> %res
@@ -2272,10 +2328,10 @@ define <4 x i8> @test_select_v4i8(i1 %cond, <4 x i8> %a, <4 x i8> %b) {
; CHECK: # %bb.0:
; CHECK-NEXT: andi a3, a0, 1
; CHECK-NEXT: mv a0, a1
-; CHECK-NEXT: bnez a3, .LBB159_2
+; CHECK-NEXT: bnez a3, .LBB165_2
; CHECK-NEXT: # %bb.1:
; CHECK-NEXT: mv a0, a2
-; CHECK-NEXT: .LBB159_2:
+; CHECK-NEXT: .LBB165_2:
; CHECK-NEXT: ret
%res = select i1 %cond, <4 x i8> %a, <4 x i8> %b
ret <4 x i8> %res
diff --git a/llvm/test/CodeGen/RISCV/rvp-simd-64.ll b/llvm/test/CodeGen/RISCV/rvp-simd-64.ll
index fe067d8225fc3..02ea76c17866e 100644
--- a/llvm/test/CodeGen/RISCV/rvp-simd-64.ll
+++ b/llvm/test/CodeGen/RISCV/rvp-simd-64.ll
@@ -2427,6 +2427,149 @@ define <2 x i32> @test_psshlr_s_u32x2(<2 x i32> %a, i32 %shamt) {
ret <2 x i32> %res
}
+; Constant shift amounts select the immediate forms.
+define <4 x i16> @test_pssha_s_i16x4_imm(<4 x i16> %a) {
+; RV32-LABEL: test_pssha_s_i16x4_imm:
+; RV32: # %bb.0:
+; RV32-NEXT: psslai.dh a0, a0, 3
+; RV32-NEXT: ret
+;
+; RV64-LABEL: test_pssha_s_i16x4_imm:
+; RV64: # %bb.0:
+; RV64-NEXT: psslai.h a0, a0, 3
+; RV64-NEXT: ret
+ %res = call <4 x i16> @llvm.riscv.pssha.v4i16.i32(<4 x i16> %a, i32 3)
+ ret <4 x i16> %res
+}
+
+define <4 x i16> @test_psshar_s_i16x4_imm(<4 x i16> %a) {
+; RV32-LABEL: test_psshar_s_i16x4_imm:
+; RV32: # %bb.0:
+; RV32-NEXT: psslai.dh a0, a0, 3
+; RV32-NEXT: ret
+;
+; RV64-LABEL: test_psshar_s_i16x4_imm:
+; RV64: # %bb.0:
+; RV64-NEXT: psslai.h a0, a0, 3
+; RV64-NEXT: ret
+ %res = call <4 x i16> @llvm.riscv.psshar.v4i16.i32(<4 x i16> %a, i32 3)
+ ret <4 x i16> %res
+}
+
+define <4 x i16> @test_pssha_s_i16x4_neg_imm(<4 x i16> %a) {
+; RV32-LABEL: test_pssha_s_i16x4_neg_imm:
+; RV32: # %bb.0:
+; RV32-NEXT: psrai.dh a0, a0, 3
+; RV32-NEXT: ret
+;
+; RV64-LABEL: test_pssha_s_i16x4_neg_imm:
+; RV64: # %bb.0:
+; RV64-NEXT: psrai.h a0, a0, 3
+; RV64-NEXT: ret
+ %res = call <4 x i16> @llvm.riscv.pssha.v4i16.i32(<4 x i16> %a, i32 -3)
+ ret <4 x i16> %res
+}
+
+define <4 x i16> @test_psshar_s_i16x4_neg_imm(<4 x i16> %a) {
+; RV32-LABEL: test_psshar_s_i16x4_neg_imm:
+; RV32: # %bb.0:
+; RV32-NEXT: psrari.dh a0, a0, 3
+; RV32-NEXT: ret
+;
+; RV64-LABEL: test_psshar_s_i16x4_neg_imm:
+; RV64: # %bb.0:
+; RV64-NEXT: psrari.h a0, a0, 3
+; RV64-NEXT: ret
+ %res = call <4 x i16> @llvm.riscv.psshar.v4i16.i32(<4 x i16> %a, i32 -3)
+ ret <4 x i16> %res
+}
+
+define <2 x i32> @test_pssha_s_i32x2_imm(<2 x i32> %a) {
+; RV32-LABEL: test_pssha_s_i32x2_imm:
+; RV32: # %bb.0:
+; RV32-NEXT: psslai.dw a0, a0, 3
+; RV32-NEXT: ret
+;
+; RV64-LABEL: test_pssha_s_i32x2_imm:
+; RV64: # %bb.0:
+; RV64-NEXT: psslai.w a0, a0, 3
+; RV64-NEXT: ret
+ %res = call <2 x i32> @llvm.riscv.pssha.v2i32.i32(<2 x i32> %a, i32 3)
+ ret <2 x i32> %res
+}
+
+define <2 x i32> @test_psshar_s_i32x2_imm(<2 x i32> %a) {
+; RV32-LABEL: test_psshar_s_i32x2_imm:
+; RV32: # %bb.0:
+; RV32-NEXT: psslai.dw a0, a0, 3
+; RV32-NEXT: ret
+;
+; RV64-LABEL: test_psshar_s_i32x2_imm:
+; RV64: # %bb.0:
+; RV64-NEXT: psslai.w a0, a0, 3
+; RV64-NEXT: ret
+ %res = call <2 x i32> @llvm.riscv.psshar.v2i32.i32(<2 x i32> %a, i32 3)
+ ret <2 x i32> %res
+}
+
+define <2 x i32> @test_pssha_s_i32x2_neg_imm(<2 x i32> %a) {
+; RV32-LABEL: test_pssha_s_i32x2_neg_imm:
+; RV32: # %bb.0:
+; RV32-NEXT: psrai.dw a0, a0, 3
+; RV32-NEXT: ret
+;
+; RV64-LABEL: test_pssha_s_i32x2_neg_imm:
+; RV64: # %bb.0:
+; RV64-NEXT: psrai.w a0, a0, 3
+; RV64-NEXT: ret
+ %res = call <2 x i32> @llvm.riscv.pssha.v2i32.i32(<2 x i32> %a, i32 -3)
+ ret <2 x i32> %res
+}
+
+define <2 x i32> @test_psshar_s_i32x2_neg_imm(<2 x i32> %a) {
+; RV32-LABEL: test_psshar_s_i32x2_neg_imm:
+; RV32: # %bb.0:
+; RV32-NEXT: psrari.dw a0, a0, 3
+; RV32-NEXT: ret
+;
+; RV64-LABEL: test_psshar_s_i32x2_neg_imm:
+; RV64: # %bb.0:
+; RV64-NEXT: psrari.w a0, a0, 3
+; RV64-NEXT: ret
+ %res = call <2 x i32> @llvm.riscv.psshar.v2i32.i32(<2 x i32> %a, i32 -3)
+ ret <2 x i32> %res
+}
+
+define <2 x i32> @test_pssha_s_i32x2_neg_imm_max(<2 x i32> %a) {
+; RV32-LABEL: test_pssha_s_i32x2_neg_imm_max:
+; RV32: # %bb.0:
+; RV32-NEXT: psrai.dw a0, a0, 31
+; RV32-NEXT: ret
+;
+; RV64-LABEL: test_pssha_s_i32x2_neg_imm_max:
+; RV64: # %bb.0:
+; RV64-NEXT: psrai.w a0, a0, 31
+; RV64-NEXT: ret
+ %res = call <2 x i32> @llvm.riscv.pssha.v2i32.i32(<2 x i32> %a, i32 -31)
+ ret <2 x i32> %res
+}
+
+define <2 x i32> @test_pssha_s_i32x2_neg_imm_too_large(<2 x i32> %a) {
+; RV32-LABEL: test_pssha_s_i32x2_neg_imm_too_large:
+; RV32: # %bb.0:
+; RV32-NEXT: li a2, -32
+; RV32-NEXT: pssha.dws a0, a0, a2
+; RV32-NEXT: ret
+;
+; RV64-LABEL: test_pssha_s_i32x2_neg_imm_too_large:
+; RV64: # %bb.0:
+; RV64-NEXT: li a1, -32
+; RV64-NEXT: pssha.ws a0, a0, a1
+; RV64-NEXT: ret
+ %res = call <2 x i32> @llvm.riscv.pssha.v2i32.i32(<2 x i32> %a, i32 -32)
+ ret <2 x i32> %res
+}
+
; Test packed multiply high signed
define <8 x i8> @test_pmulh_b(<8 x i8> %a, <8 x i8> %b) {
; RV32-LABEL: test_pmulh_b:
@@ -4534,12 +4677,12 @@ define <4 x i16> @test_select_v4i16(i1 %cond, <4 x i16> %a, <4 x i16> %b) {
; RV32-LABEL: test_select_v4i16:
; RV32: # %bb.0:
; RV32-NEXT: andi a5, a0, 1
-; RV32-NEXT: bnez a5, .LBB236_2
+; RV32-NEXT: bnez a5, .LBB246_2
; RV32-NEXT: # %bb.1:
; RV32-NEXT: mv a0, a3
; RV32-NEXT: mv a1, a4
; RV32-NEXT: ret
-; RV32-NEXT: .LBB236_2:
+; RV32-NEXT: .LBB246_2:
; RV32-NEXT: mv a0, a1
; RV32-NEXT: mv a1, a2
; RV32-NEXT: ret
@@ -4548,10 +4691,10 @@ define <4 x i16> @test_select_v4i16(i1 %cond, <4 x i16> %a, <4 x i16> %b) {
; RV64: # %bb.0:
; RV64-NEXT: andi a3, a0, 1
; RV64-NEXT: mv a0, a1
-; RV64-NEXT: bnez a3, .LBB236_2
+; RV64-NEXT: bnez a3, .LBB246_2
; RV64-NEXT: # %bb.1:
; RV64-NEXT: mv a0, a2
-; RV64-NEXT: .LBB236_2:
+; RV64-NEXT: .LBB246_2:
; RV64-NEXT: ret
%res = select i1 %cond, <4 x i16> %a, <4 x i16> %b
ret <4 x i16> %res
@@ -4561,12 +4704,12 @@ define <8 x i8> @test_select_v8i8(i1 %cond, <8 x i8> %a, <8 x i8> %b) {
; RV32-LABEL: test_select_v8i8:
; RV32: # %bb.0:
; RV32-NEXT: andi a5, a0, 1
-; RV32-NEXT: bnez a5, .LBB237_2
+; RV32-NEXT: bnez a5, .LBB247_2
; RV32-NEXT: # %bb.1:
; RV32-NEXT: mv a0, a3
; RV32-NEXT: mv a1, a4
; RV32-NEXT: ret
-; RV32-NEXT: .LBB237_2:
+; RV32-NEXT: .LBB247_2:
; RV32-NEXT: mv a0, a1
; RV32-NEXT: mv a1, a2
; RV32-NEXT: ret
@@ -4575,10 +4718,10 @@ define <8 x i8> @test_select_v8i8(i1 %cond, <8 x i8> %a, <8 x i8> %b) {
; RV64: # %bb.0:
; RV64-NEXT: andi a3, a0, 1
; RV64-NEXT: mv a0, a1
-; RV64-NEXT: bnez a3, .LBB237_2
+; RV64-NEXT: bnez a3, .LBB247_2
; RV64-NEXT: # %bb.1:
; RV64-NEXT: mv a0, a2
-; RV64-NEXT: .LBB237_2:
+; RV64-NEXT: .LBB247_2:
; RV64-NEXT: ret
%res = select i1 %cond, <8 x i8> %a, <8 x i8> %b
ret <8 x i8> %res
@@ -4588,12 +4731,12 @@ define <2 x i32> @test_select_v2i32(i1 %cond, <2 x i32> %a, <2 x i32> %b) {
; RV32-LABEL: test_select_v2i32:
; RV32: # %bb.0:
; RV32-NEXT: andi a5, a0, 1
-; RV32-NEXT: bnez a5, .LBB238_2
+; RV32-NEXT: bnez a5, .LBB248_2
; RV32-NEXT: # %bb.1:
; RV32-NEXT: mv a0, a3
; RV32-NEXT: mv a1, a4
; RV32-NEXT: ret
-; RV32-NEXT: .LBB238_2:
+; RV32-NEXT: .LBB248_2:
; RV32-NEXT: mv a0, a1
; RV32-NEXT: mv a1, a2
; RV32-NEXT: ret
@@ -4602,10 +4745,10 @@ define <2 x i32> @test_select_v2i32(i1 %cond, <2 x i32> %a, <2 x i32> %b) {
; RV64: # %bb.0:
; RV64-NEXT: andi a3, a0, 1
; RV64-NEXT: mv a0, a1
-; RV64-NEXT: bnez a3, .LBB238_2
+; RV64-NEXT: bnez a3, .LBB248_2
; RV64-NEXT: # %bb.1:
; RV64-NEXT: mv a0, a2
-; RV64-NEXT: .LBB238_2:
+; RV64-NEXT: .LBB248_2:
; RV64-NEXT: ret
%res = select i1 %cond, <2 x i32> %a, <2 x i32> %b
ret <2 x i32> %res
@@ -4653,16 +4796,16 @@ define <2 x i32> @test_vselect_v2i32(<2 x i32> %a, <2 x i32> %b, <2 x i32> %c) {
; RV32: # %bb.0:
; RV32-NEXT: pmslt.dw a6, a2, a0
; RV32-NEXT: mv a0, a4
-; RV32-NEXT: beqz a7, .LBB241_3
+; RV32-NEXT: beqz a7, .LBB251_3
; RV32-NEXT: # %bb.1:
-; RV32-NEXT: beqz a6, .LBB241_4
-; RV32-NEXT: .LBB241_2:
+; RV32-NEXT: beqz a6, .LBB251_4
+; RV32-NEXT: .LBB251_2:
; RV32-NEXT: mv a1, a5
; RV32-NEXT: ret
-; RV32-NEXT: .LBB241_3:
+; RV32-NEXT: .LBB251_3:
; RV32-NEXT: mv a5, a3
-; RV32-NEXT: bnez a6, .LBB241_2
-; RV32-NEXT: .LBB241_4:
+; RV32-NEXT: bnez a6, .LBB251_2
+; RV32-NEXT: .LBB251_4:
; RV32-NEXT: mv a0, a2
; RV32-NEXT: mv a1, a5
; RV32-NEXT: ret
>From 2cf3abd5834ba1e4fa64056c8065a21e08da6d43 Mon Sep 17 00:00:00 2001
From: SiHuaN <liyongtai at iscas.ac.cn>
Date: Mon, 17 Aug 2026 15:27:52 +0000
Subject: [PATCH 2/2] [Clang][RISCV] Add packed saturating and rounding shift
intrinsics
Add builtins and header wrappers for `__riscv_pssha`, `__riscv_psshar`,
`__riscv_psshl` and `__riscv_psshlr`, lowering to the `llvm.riscv.*`
intrinsics from #208630.
---
clang/include/clang/Basic/BuiltinsRISCV.td | 16 ++
clang/lib/CodeGen/TargetBuiltins/RISCV.cpp | 35 ++-
clang/lib/Headers/riscv_packed_simd.h | 16 ++
clang/test/CodeGen/RISCV/rvp-intrinsics.c | 244 ++++++++++++++++++
.../riscv_packed_simd.c | 160 ++++++++++++
5 files changed, 470 insertions(+), 1 deletion(-)
diff --git a/clang/include/clang/Basic/BuiltinsRISCV.td b/clang/include/clang/Basic/BuiltinsRISCV.td
index b91b356d7a25d..74d7520821b6e 100644
--- a/clang/include/clang/Basic/BuiltinsRISCV.td
+++ b/clang/include/clang/Basic/BuiltinsRISCV.td
@@ -323,6 +323,22 @@ def pnclipup_u16x4 : RISCVBuiltin<"_Vector<4, unsigned short>(_Vector<2, unsigne
def pnclipp_i32x2 : RISCVBuiltin<"_Vector<2, int>(int64_t, int64_t)">;
def pnclipup_u32x2 : RISCVBuiltin<"_Vector<2, unsigned int>(uint64_t, uint64_t)">;
+// Packed Saturating and Rounding Shifts (32-bit)
+def pssha_s_i16x2 : RISCVBuiltin<"_Vector<2, short>(_Vector<2, short>, int)">;
+def psshar_s_i16x2 : RISCVBuiltin<"_Vector<2, short>(_Vector<2, short>, int)">;
+def psshl_s_u16x2 : RISCVBuiltin<"_Vector<2, unsigned short>(_Vector<2, unsigned short>, int)">;
+def psshlr_s_u16x2 : RISCVBuiltin<"_Vector<2, unsigned short>(_Vector<2, unsigned short>, int)">;
+
+// Packed Saturating and Rounding Shifts (64-bit)
+def pssha_s_i16x4 : RISCVBuiltin<"_Vector<4, short>(_Vector<4, short>, int)">;
+def pssha_s_i32x2 : RISCVBuiltin<"_Vector<2, int>(_Vector<2, int>, int)">;
+def psshar_s_i16x4 : RISCVBuiltin<"_Vector<4, short>(_Vector<4, short>, int)">;
+def psshar_s_i32x2 : RISCVBuiltin<"_Vector<2, int>(_Vector<2, int>, int)">;
+def psshl_s_u16x4 : RISCVBuiltin<"_Vector<4, unsigned short>(_Vector<4, unsigned short>, int)">;
+def psshl_s_u32x2 : RISCVBuiltin<"_Vector<2, unsigned int>(_Vector<2, unsigned int>, int)">;
+def psshlr_s_u16x4 : RISCVBuiltin<"_Vector<4, unsigned short>(_Vector<4, unsigned short>, int)">;
+def psshlr_s_u32x2 : RISCVBuiltin<"_Vector<2, unsigned int>(_Vector<2, unsigned int>, int)">;
+
} // Features = "experimental-p"
//===----------------------------------------------------------------------===//
diff --git a/clang/lib/CodeGen/TargetBuiltins/RISCV.cpp b/clang/lib/CodeGen/TargetBuiltins/RISCV.cpp
index eb98b38974b21..eefe86e1f8cde 100644
--- a/clang/lib/CodeGen/TargetBuiltins/RISCV.cpp
+++ b/clang/lib/CodeGen/TargetBuiltins/RISCV.cpp
@@ -1296,7 +1296,20 @@ Value *CodeGenFunction::EmitRISCVBuiltinExpr(unsigned BuiltinID,
case RISCV::BI__builtin_riscv_psext_b_i32x2:
case RISCV::BI__builtin_riscv_psext_h_i32x2:
case RISCV::BI__builtin_riscv_pzext_b_u16x4:
- case RISCV::BI__builtin_riscv_pzext_h_u32x2: {
+ case RISCV::BI__builtin_riscv_pzext_h_u32x2:
+ // Packed Saturating and Rounding Shifts
+ case RISCV::BI__builtin_riscv_pssha_s_i16x2:
+ case RISCV::BI__builtin_riscv_psshar_s_i16x2:
+ case RISCV::BI__builtin_riscv_psshl_s_u16x2:
+ case RISCV::BI__builtin_riscv_psshlr_s_u16x2:
+ case RISCV::BI__builtin_riscv_pssha_s_i16x4:
+ case RISCV::BI__builtin_riscv_pssha_s_i32x2:
+ case RISCV::BI__builtin_riscv_psshar_s_i16x4:
+ case RISCV::BI__builtin_riscv_psshar_s_i32x2:
+ case RISCV::BI__builtin_riscv_psshl_s_u16x4:
+ case RISCV::BI__builtin_riscv_psshl_s_u32x2:
+ case RISCV::BI__builtin_riscv_psshlr_s_u16x4:
+ case RISCV::BI__builtin_riscv_psshlr_s_u32x2: {
switch (BuiltinID) {
default:
llvm_unreachable("unexpected builtin ID");
@@ -1443,6 +1456,26 @@ Value *CodeGenFunction::EmitRISCVBuiltinExpr(unsigned BuiltinID,
case RISCV::BI__builtin_riscv_pzext_h_u32x2:
ID = Intrinsic::riscv_pzext_h;
break;
+ case RISCV::BI__builtin_riscv_pssha_s_i16x2:
+ case RISCV::BI__builtin_riscv_pssha_s_i16x4:
+ case RISCV::BI__builtin_riscv_pssha_s_i32x2:
+ ID = Intrinsic::riscv_pssha;
+ break;
+ case RISCV::BI__builtin_riscv_psshar_s_i16x2:
+ case RISCV::BI__builtin_riscv_psshar_s_i16x4:
+ case RISCV::BI__builtin_riscv_psshar_s_i32x2:
+ ID = Intrinsic::riscv_psshar;
+ break;
+ case RISCV::BI__builtin_riscv_psshl_s_u16x2:
+ case RISCV::BI__builtin_riscv_psshl_s_u16x4:
+ case RISCV::BI__builtin_riscv_psshl_s_u32x2:
+ ID = Intrinsic::riscv_psshl;
+ break;
+ case RISCV::BI__builtin_riscv_psshlr_s_u16x2:
+ case RISCV::BI__builtin_riscv_psshlr_s_u16x4:
+ case RISCV::BI__builtin_riscv_psshlr_s_u32x2:
+ ID = Intrinsic::riscv_psshlr;
+ break;
}
IntrinsicTypes = {ResultType};
diff --git a/clang/lib/Headers/riscv_packed_simd.h b/clang/lib/Headers/riscv_packed_simd.h
index 491e219bec297..2a9a8f88a0af4 100644
--- a/clang/lib/Headers/riscv_packed_simd.h
+++ b/clang/lib/Headers/riscv_packed_simd.h
@@ -510,6 +510,22 @@ __packed_shift8(psra_s_i8x8, int8x8_t, >>)
__packed_shift16(psra_s_i16x4, int16x4_t, >>)
__packed_shift32(psra_s_i32x2, int32x2_t, >>)
+/* Packed Saturating and Rounding Shifts (32-bit) */
+__packed_binary_builtin_mixed(pssha_s_i16x2, int16x2_t, int16x2_t, int, __builtin_riscv_pssha_s_i16x2)
+__packed_binary_builtin_mixed(psshar_s_i16x2, int16x2_t, int16x2_t, int, __builtin_riscv_psshar_s_i16x2)
+__packed_binary_builtin_mixed(psshl_s_u16x2, uint16x2_t, uint16x2_t, int, __builtin_riscv_psshl_s_u16x2)
+__packed_binary_builtin_mixed(psshlr_s_u16x2, uint16x2_t, uint16x2_t, int, __builtin_riscv_psshlr_s_u16x2)
+
+/* Packed Saturating and Rounding Shifts (64-bit) */
+__packed_binary_builtin_mixed(pssha_s_i16x4, int16x4_t, int16x4_t, int, __builtin_riscv_pssha_s_i16x4)
+__packed_binary_builtin_mixed(pssha_s_i32x2, int32x2_t, int32x2_t, int, __builtin_riscv_pssha_s_i32x2)
+__packed_binary_builtin_mixed(psshar_s_i16x4, int16x4_t, int16x4_t, int, __builtin_riscv_psshar_s_i16x4)
+__packed_binary_builtin_mixed(psshar_s_i32x2, int32x2_t, int32x2_t, int, __builtin_riscv_psshar_s_i32x2)
+__packed_binary_builtin_mixed(psshl_s_u16x4, uint16x4_t, uint16x4_t, int, __builtin_riscv_psshl_s_u16x4)
+__packed_binary_builtin_mixed(psshl_s_u32x2, uint32x2_t, uint32x2_t, int, __builtin_riscv_psshl_s_u32x2)
+__packed_binary_builtin_mixed(psshlr_s_u16x4, uint16x4_t, uint16x4_t, int, __builtin_riscv_psshlr_s_u16x4)
+__packed_binary_builtin_mixed(psshlr_s_u32x2, uint32x2_t, uint32x2_t, int, __builtin_riscv_psshlr_s_u32x2)
+
/* Packed Logical Operations (32-bit) */
__packed_binary_op(pand_i8x4, int8x4_t, &)
__packed_binary_op(pand_u8x4, uint8x4_t, &)
diff --git a/clang/test/CodeGen/RISCV/rvp-intrinsics.c b/clang/test/CodeGen/RISCV/rvp-intrinsics.c
index d6afc4d18cc6a..3a558d9750db2 100644
--- a/clang/test/CodeGen/RISCV/rvp-intrinsics.c
+++ b/clang/test/CodeGen/RISCV/rvp-intrinsics.c
@@ -4506,6 +4506,250 @@ uint32x2_t test_psrl_s_u32x2(uint32x2_t a, unsigned shamt) {
return __riscv_psrl_s_u32x2(a, shamt);
}
+/* Packed Saturating and Rounding Shifts (32-bit) */
+
+// RV32-LABEL: define dso_local i32 @test_pssha_s_i16x2(
+// RV32-SAME: i32 noundef [[A_COERCE:%.*]], i32 noundef [[SHAMT:%.*]]) #[[ATTR0]] {
+// RV32-NEXT: [[ENTRY:.*:]]
+// RV32-NEXT: [[TMP0:%.*]] = bitcast i32 [[A_COERCE]] to <2 x i16>
+// RV32-NEXT: [[TMP1:%.*]] = call <2 x i16> @llvm.riscv.pssha.v2i16(<2 x i16> [[TMP0]], i32 [[SHAMT]])
+// RV32-NEXT: [[TMP2:%.*]] = bitcast <2 x i16> [[TMP1]] to i32
+// RV32-NEXT: ret i32 [[TMP2]]
+//
+// RV64-LABEL: define dso_local i32 @test_pssha_s_i16x2(
+// RV64-SAME: i32 noundef [[A_COERCE:%.*]], i32 noundef signext [[SHAMT:%.*]]) #[[ATTR0]] {
+// RV64-NEXT: [[ENTRY:.*:]]
+// RV64-NEXT: [[TMP0:%.*]] = bitcast i32 [[A_COERCE]] to <2 x i16>
+// RV64-NEXT: [[TMP1:%.*]] = call <2 x i16> @llvm.riscv.pssha.v2i16(<2 x i16> [[TMP0]], i32 [[SHAMT]])
+// RV64-NEXT: [[TMP2:%.*]] = bitcast <2 x i16> [[TMP1]] to i32
+// RV64-NEXT: ret i32 [[TMP2]]
+//
+int16x2_t test_pssha_s_i16x2(int16x2_t a, int shamt) {
+ return __riscv_pssha_s_i16x2(a, shamt);
+}
+
+// RV32-LABEL: define dso_local i32 @test_psshar_s_i16x2(
+// RV32-SAME: i32 noundef [[A_COERCE:%.*]], i32 noundef [[SHAMT:%.*]]) #[[ATTR0]] {
+// RV32-NEXT: [[ENTRY:.*:]]
+// RV32-NEXT: [[TMP0:%.*]] = bitcast i32 [[A_COERCE]] to <2 x i16>
+// RV32-NEXT: [[TMP1:%.*]] = call <2 x i16> @llvm.riscv.psshar.v2i16(<2 x i16> [[TMP0]], i32 [[SHAMT]])
+// RV32-NEXT: [[TMP2:%.*]] = bitcast <2 x i16> [[TMP1]] to i32
+// RV32-NEXT: ret i32 [[TMP2]]
+//
+// RV64-LABEL: define dso_local i32 @test_psshar_s_i16x2(
+// RV64-SAME: i32 noundef [[A_COERCE:%.*]], i32 noundef signext [[SHAMT:%.*]]) #[[ATTR0]] {
+// RV64-NEXT: [[ENTRY:.*:]]
+// RV64-NEXT: [[TMP0:%.*]] = bitcast i32 [[A_COERCE]] to <2 x i16>
+// RV64-NEXT: [[TMP1:%.*]] = call <2 x i16> @llvm.riscv.psshar.v2i16(<2 x i16> [[TMP0]], i32 [[SHAMT]])
+// RV64-NEXT: [[TMP2:%.*]] = bitcast <2 x i16> [[TMP1]] to i32
+// RV64-NEXT: ret i32 [[TMP2]]
+//
+int16x2_t test_psshar_s_i16x2(int16x2_t a, int shamt) {
+ return __riscv_psshar_s_i16x2(a, shamt);
+}
+
+// RV32-LABEL: define dso_local i32 @test_psshl_s_u16x2(
+// RV32-SAME: i32 noundef [[A_COERCE:%.*]], i32 noundef [[SHAMT:%.*]]) #[[ATTR0]] {
+// RV32-NEXT: [[ENTRY:.*:]]
+// RV32-NEXT: [[TMP0:%.*]] = bitcast i32 [[A_COERCE]] to <2 x i16>
+// RV32-NEXT: [[TMP1:%.*]] = call <2 x i16> @llvm.riscv.psshl.v2i16(<2 x i16> [[TMP0]], i32 [[SHAMT]])
+// RV32-NEXT: [[TMP2:%.*]] = bitcast <2 x i16> [[TMP1]] to i32
+// RV32-NEXT: ret i32 [[TMP2]]
+//
+// RV64-LABEL: define dso_local i32 @test_psshl_s_u16x2(
+// RV64-SAME: i32 noundef [[A_COERCE:%.*]], i32 noundef signext [[SHAMT:%.*]]) #[[ATTR0]] {
+// RV64-NEXT: [[ENTRY:.*:]]
+// RV64-NEXT: [[TMP0:%.*]] = bitcast i32 [[A_COERCE]] to <2 x i16>
+// RV64-NEXT: [[TMP1:%.*]] = call <2 x i16> @llvm.riscv.psshl.v2i16(<2 x i16> [[TMP0]], i32 [[SHAMT]])
+// RV64-NEXT: [[TMP2:%.*]] = bitcast <2 x i16> [[TMP1]] to i32
+// RV64-NEXT: ret i32 [[TMP2]]
+//
+uint16x2_t test_psshl_s_u16x2(uint16x2_t a, int shamt) {
+ return __riscv_psshl_s_u16x2(a, shamt);
+}
+
+// RV32-LABEL: define dso_local i32 @test_psshlr_s_u16x2(
+// RV32-SAME: i32 noundef [[A_COERCE:%.*]], i32 noundef [[SHAMT:%.*]]) #[[ATTR0]] {
+// RV32-NEXT: [[ENTRY:.*:]]
+// RV32-NEXT: [[TMP0:%.*]] = bitcast i32 [[A_COERCE]] to <2 x i16>
+// RV32-NEXT: [[TMP1:%.*]] = call <2 x i16> @llvm.riscv.psshlr.v2i16(<2 x i16> [[TMP0]], i32 [[SHAMT]])
+// RV32-NEXT: [[TMP2:%.*]] = bitcast <2 x i16> [[TMP1]] to i32
+// RV32-NEXT: ret i32 [[TMP2]]
+//
+// RV64-LABEL: define dso_local i32 @test_psshlr_s_u16x2(
+// RV64-SAME: i32 noundef [[A_COERCE:%.*]], i32 noundef signext [[SHAMT:%.*]]) #[[ATTR0]] {
+// RV64-NEXT: [[ENTRY:.*:]]
+// RV64-NEXT: [[TMP0:%.*]] = bitcast i32 [[A_COERCE]] to <2 x i16>
+// RV64-NEXT: [[TMP1:%.*]] = call <2 x i16> @llvm.riscv.psshlr.v2i16(<2 x i16> [[TMP0]], i32 [[SHAMT]])
+// RV64-NEXT: [[TMP2:%.*]] = bitcast <2 x i16> [[TMP1]] to i32
+// RV64-NEXT: ret i32 [[TMP2]]
+//
+uint16x2_t test_psshlr_s_u16x2(uint16x2_t a, int shamt) {
+ return __riscv_psshlr_s_u16x2(a, shamt);
+}
+
+/* Packed Saturating and Rounding Shifts (64-bit) */
+
+// RV32-LABEL: define dso_local i64 @test_pssha_s_i16x4(
+// RV32-SAME: i64 noundef [[A_COERCE:%.*]], i32 noundef [[SHAMT:%.*]]) #[[ATTR0]] {
+// RV32-NEXT: [[ENTRY:.*:]]
+// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[A_COERCE]] to <4 x i16>
+// RV32-NEXT: [[TMP1:%.*]] = call <4 x i16> @llvm.riscv.pssha.v4i16(<4 x i16> [[TMP0]], i32 [[SHAMT]])
+// RV32-NEXT: [[TMP2:%.*]] = bitcast <4 x i16> [[TMP1]] to i64
+// RV32-NEXT: ret i64 [[TMP2]]
+//
+// RV64-LABEL: define dso_local i64 @test_pssha_s_i16x4(
+// RV64-SAME: i64 noundef [[A_COERCE:%.*]], i32 noundef signext [[SHAMT:%.*]]) #[[ATTR0]] {
+// RV64-NEXT: [[ENTRY:.*:]]
+// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[A_COERCE]] to <4 x i16>
+// RV64-NEXT: [[TMP1:%.*]] = call <4 x i16> @llvm.riscv.pssha.v4i16(<4 x i16> [[TMP0]], i32 [[SHAMT]])
+// RV64-NEXT: [[TMP2:%.*]] = bitcast <4 x i16> [[TMP1]] to i64
+// RV64-NEXT: ret i64 [[TMP2]]
+//
+int16x4_t test_pssha_s_i16x4(int16x4_t a, int shamt) {
+ return __riscv_pssha_s_i16x4(a, shamt);
+}
+
+// RV32-LABEL: define dso_local i64 @test_pssha_s_i32x2(
+// RV32-SAME: i64 noundef [[A_COERCE:%.*]], i32 noundef [[SHAMT:%.*]]) #[[ATTR0]] {
+// RV32-NEXT: [[ENTRY:.*:]]
+// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[A_COERCE]] to <2 x i32>
+// RV32-NEXT: [[TMP1:%.*]] = call <2 x i32> @llvm.riscv.pssha.v2i32(<2 x i32> [[TMP0]], i32 [[SHAMT]])
+// RV32-NEXT: [[TMP2:%.*]] = bitcast <2 x i32> [[TMP1]] to i64
+// RV32-NEXT: ret i64 [[TMP2]]
+//
+// RV64-LABEL: define dso_local i64 @test_pssha_s_i32x2(
+// RV64-SAME: i64 noundef [[A_COERCE:%.*]], i32 noundef signext [[SHAMT:%.*]]) #[[ATTR0]] {
+// RV64-NEXT: [[ENTRY:.*:]]
+// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[A_COERCE]] to <2 x i32>
+// RV64-NEXT: [[TMP1:%.*]] = call <2 x i32> @llvm.riscv.pssha.v2i32(<2 x i32> [[TMP0]], i32 [[SHAMT]])
+// RV64-NEXT: [[TMP2:%.*]] = bitcast <2 x i32> [[TMP1]] to i64
+// RV64-NEXT: ret i64 [[TMP2]]
+//
+int32x2_t test_pssha_s_i32x2(int32x2_t a, int shamt) {
+ return __riscv_pssha_s_i32x2(a, shamt);
+}
+
+// RV32-LABEL: define dso_local i64 @test_psshar_s_i16x4(
+// RV32-SAME: i64 noundef [[A_COERCE:%.*]], i32 noundef [[SHAMT:%.*]]) #[[ATTR0]] {
+// RV32-NEXT: [[ENTRY:.*:]]
+// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[A_COERCE]] to <4 x i16>
+// RV32-NEXT: [[TMP1:%.*]] = call <4 x i16> @llvm.riscv.psshar.v4i16(<4 x i16> [[TMP0]], i32 [[SHAMT]])
+// RV32-NEXT: [[TMP2:%.*]] = bitcast <4 x i16> [[TMP1]] to i64
+// RV32-NEXT: ret i64 [[TMP2]]
+//
+// RV64-LABEL: define dso_local i64 @test_psshar_s_i16x4(
+// RV64-SAME: i64 noundef [[A_COERCE:%.*]], i32 noundef signext [[SHAMT:%.*]]) #[[ATTR0]] {
+// RV64-NEXT: [[ENTRY:.*:]]
+// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[A_COERCE]] to <4 x i16>
+// RV64-NEXT: [[TMP1:%.*]] = call <4 x i16> @llvm.riscv.psshar.v4i16(<4 x i16> [[TMP0]], i32 [[SHAMT]])
+// RV64-NEXT: [[TMP2:%.*]] = bitcast <4 x i16> [[TMP1]] to i64
+// RV64-NEXT: ret i64 [[TMP2]]
+//
+int16x4_t test_psshar_s_i16x4(int16x4_t a, int shamt) {
+ return __riscv_psshar_s_i16x4(a, shamt);
+}
+
+// RV32-LABEL: define dso_local i64 @test_psshar_s_i32x2(
+// RV32-SAME: i64 noundef [[A_COERCE:%.*]], i32 noundef [[SHAMT:%.*]]) #[[ATTR0]] {
+// RV32-NEXT: [[ENTRY:.*:]]
+// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[A_COERCE]] to <2 x i32>
+// RV32-NEXT: [[TMP1:%.*]] = call <2 x i32> @llvm.riscv.psshar.v2i32(<2 x i32> [[TMP0]], i32 [[SHAMT]])
+// RV32-NEXT: [[TMP2:%.*]] = bitcast <2 x i32> [[TMP1]] to i64
+// RV32-NEXT: ret i64 [[TMP2]]
+//
+// RV64-LABEL: define dso_local i64 @test_psshar_s_i32x2(
+// RV64-SAME: i64 noundef [[A_COERCE:%.*]], i32 noundef signext [[SHAMT:%.*]]) #[[ATTR0]] {
+// RV64-NEXT: [[ENTRY:.*:]]
+// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[A_COERCE]] to <2 x i32>
+// RV64-NEXT: [[TMP1:%.*]] = call <2 x i32> @llvm.riscv.psshar.v2i32(<2 x i32> [[TMP0]], i32 [[SHAMT]])
+// RV64-NEXT: [[TMP2:%.*]] = bitcast <2 x i32> [[TMP1]] to i64
+// RV64-NEXT: ret i64 [[TMP2]]
+//
+int32x2_t test_psshar_s_i32x2(int32x2_t a, int shamt) {
+ return __riscv_psshar_s_i32x2(a, shamt);
+}
+
+// RV32-LABEL: define dso_local i64 @test_psshl_s_u16x4(
+// RV32-SAME: i64 noundef [[A_COERCE:%.*]], i32 noundef [[SHAMT:%.*]]) #[[ATTR0]] {
+// RV32-NEXT: [[ENTRY:.*:]]
+// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[A_COERCE]] to <4 x i16>
+// RV32-NEXT: [[TMP1:%.*]] = call <4 x i16> @llvm.riscv.psshl.v4i16(<4 x i16> [[TMP0]], i32 [[SHAMT]])
+// RV32-NEXT: [[TMP2:%.*]] = bitcast <4 x i16> [[TMP1]] to i64
+// RV32-NEXT: ret i64 [[TMP2]]
+//
+// RV64-LABEL: define dso_local i64 @test_psshl_s_u16x4(
+// RV64-SAME: i64 noundef [[A_COERCE:%.*]], i32 noundef signext [[SHAMT:%.*]]) #[[ATTR0]] {
+// RV64-NEXT: [[ENTRY:.*:]]
+// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[A_COERCE]] to <4 x i16>
+// RV64-NEXT: [[TMP1:%.*]] = call <4 x i16> @llvm.riscv.psshl.v4i16(<4 x i16> [[TMP0]], i32 [[SHAMT]])
+// RV64-NEXT: [[TMP2:%.*]] = bitcast <4 x i16> [[TMP1]] to i64
+// RV64-NEXT: ret i64 [[TMP2]]
+//
+uint16x4_t test_psshl_s_u16x4(uint16x4_t a, int shamt) {
+ return __riscv_psshl_s_u16x4(a, shamt);
+}
+
+// RV32-LABEL: define dso_local i64 @test_psshl_s_u32x2(
+// RV32-SAME: i64 noundef [[A_COERCE:%.*]], i32 noundef [[SHAMT:%.*]]) #[[ATTR0]] {
+// RV32-NEXT: [[ENTRY:.*:]]
+// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[A_COERCE]] to <2 x i32>
+// RV32-NEXT: [[TMP1:%.*]] = call <2 x i32> @llvm.riscv.psshl.v2i32(<2 x i32> [[TMP0]], i32 [[SHAMT]])
+// RV32-NEXT: [[TMP2:%.*]] = bitcast <2 x i32> [[TMP1]] to i64
+// RV32-NEXT: ret i64 [[TMP2]]
+//
+// RV64-LABEL: define dso_local i64 @test_psshl_s_u32x2(
+// RV64-SAME: i64 noundef [[A_COERCE:%.*]], i32 noundef signext [[SHAMT:%.*]]) #[[ATTR0]] {
+// RV64-NEXT: [[ENTRY:.*:]]
+// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[A_COERCE]] to <2 x i32>
+// RV64-NEXT: [[TMP1:%.*]] = call <2 x i32> @llvm.riscv.psshl.v2i32(<2 x i32> [[TMP0]], i32 [[SHAMT]])
+// RV64-NEXT: [[TMP2:%.*]] = bitcast <2 x i32> [[TMP1]] to i64
+// RV64-NEXT: ret i64 [[TMP2]]
+//
+uint32x2_t test_psshl_s_u32x2(uint32x2_t a, int shamt) {
+ return __riscv_psshl_s_u32x2(a, shamt);
+}
+
+// RV32-LABEL: define dso_local i64 @test_psshlr_s_u16x4(
+// RV32-SAME: i64 noundef [[A_COERCE:%.*]], i32 noundef [[SHAMT:%.*]]) #[[ATTR0]] {
+// RV32-NEXT: [[ENTRY:.*:]]
+// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[A_COERCE]] to <4 x i16>
+// RV32-NEXT: [[TMP1:%.*]] = call <4 x i16> @llvm.riscv.psshlr.v4i16(<4 x i16> [[TMP0]], i32 [[SHAMT]])
+// RV32-NEXT: [[TMP2:%.*]] = bitcast <4 x i16> [[TMP1]] to i64
+// RV32-NEXT: ret i64 [[TMP2]]
+//
+// RV64-LABEL: define dso_local i64 @test_psshlr_s_u16x4(
+// RV64-SAME: i64 noundef [[A_COERCE:%.*]], i32 noundef signext [[SHAMT:%.*]]) #[[ATTR0]] {
+// RV64-NEXT: [[ENTRY:.*:]]
+// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[A_COERCE]] to <4 x i16>
+// RV64-NEXT: [[TMP1:%.*]] = call <4 x i16> @llvm.riscv.psshlr.v4i16(<4 x i16> [[TMP0]], i32 [[SHAMT]])
+// RV64-NEXT: [[TMP2:%.*]] = bitcast <4 x i16> [[TMP1]] to i64
+// RV64-NEXT: ret i64 [[TMP2]]
+//
+uint16x4_t test_psshlr_s_u16x4(uint16x4_t a, int shamt) {
+ return __riscv_psshlr_s_u16x4(a, shamt);
+}
+
+// RV32-LABEL: define dso_local i64 @test_psshlr_s_u32x2(
+// RV32-SAME: i64 noundef [[A_COERCE:%.*]], i32 noundef [[SHAMT:%.*]]) #[[ATTR0]] {
+// RV32-NEXT: [[ENTRY:.*:]]
+// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[A_COERCE]] to <2 x i32>
+// RV32-NEXT: [[TMP1:%.*]] = call <2 x i32> @llvm.riscv.psshlr.v2i32(<2 x i32> [[TMP0]], i32 [[SHAMT]])
+// RV32-NEXT: [[TMP2:%.*]] = bitcast <2 x i32> [[TMP1]] to i64
+// RV32-NEXT: ret i64 [[TMP2]]
+//
+// RV64-LABEL: define dso_local i64 @test_psshlr_s_u32x2(
+// RV64-SAME: i64 noundef [[A_COERCE:%.*]], i32 noundef signext [[SHAMT:%.*]]) #[[ATTR0]] {
+// RV64-NEXT: [[ENTRY:.*:]]
+// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[A_COERCE]] to <2 x i32>
+// RV64-NEXT: [[TMP1:%.*]] = call <2 x i32> @llvm.riscv.psshlr.v2i32(<2 x i32> [[TMP0]], i32 [[SHAMT]])
+// RV64-NEXT: [[TMP2:%.*]] = bitcast <2 x i32> [[TMP1]] to i64
+// RV64-NEXT: ret i64 [[TMP2]]
+//
+uint32x2_t test_psshlr_s_u32x2(uint32x2_t a, int shamt) {
+ return __riscv_psshlr_s_u32x2(a, shamt);
+}
+
/* Packed Logical Operations (32-bit) */
// RV32-LABEL: define dso_local i32 @test_pand_i8x4(
diff --git a/cross-project-tests/intrinsic-header-tests/riscv_packed_simd.c b/cross-project-tests/intrinsic-header-tests/riscv_packed_simd.c
index eac091c66bba0..10a2e508fa9ce 100644
--- a/cross-project-tests/intrinsic-header-tests/riscv_packed_simd.c
+++ b/cross-project-tests/intrinsic-header-tests/riscv_packed_simd.c
@@ -1079,6 +1079,166 @@ int32x2_t test_psra_s_i32x2_imm(int32x2_t a) {
return __riscv_psra_s_i32x2(a, 11);
}
+// CHECK-LABEL: test_pssha_s_i16x2:
+// CHECK: pssha.hs
+int16x2_t test_pssha_s_i16x2(int16x2_t a, int n) {
+ return __riscv_pssha_s_i16x2(a, n);
+}
+
+// CHECK-LABEL: test_psshar_s_i16x2:
+// CHECK: psshar.hs
+int16x2_t test_psshar_s_i16x2(int16x2_t a, int n) {
+ return __riscv_psshar_s_i16x2(a, n);
+}
+
+// CHECK-LABEL: test_psshl_s_u16x2:
+// CHECK: psshl.hs
+uint16x2_t test_psshl_s_u16x2(uint16x2_t a, int n) {
+ return __riscv_psshl_s_u16x2(a, n);
+}
+
+// CHECK-LABEL: test_psshlr_s_u16x2:
+// CHECK: psshlr.hs
+uint16x2_t test_psshlr_s_u16x2(uint16x2_t a, int n) {
+ return __riscv_psshlr_s_u16x2(a, n);
+}
+
+// CHECK-LABEL: test_pssha_s_i16x4:
+// RV32: pssha.dhs
+// RV64: pssha.hs
+int16x4_t test_pssha_s_i16x4(int16x4_t a, int n) {
+ return __riscv_pssha_s_i16x4(a, n);
+}
+
+// CHECK-LABEL: test_pssha_s_i32x2:
+// RV32: pssha.dws
+// RV64: pssha.ws
+int32x2_t test_pssha_s_i32x2(int32x2_t a, int n) {
+ return __riscv_pssha_s_i32x2(a, n);
+}
+
+// CHECK-LABEL: test_psshar_s_i16x4:
+// RV32: psshar.dhs
+// RV64: psshar.hs
+int16x4_t test_psshar_s_i16x4(int16x4_t a, int n) {
+ return __riscv_psshar_s_i16x4(a, n);
+}
+
+// CHECK-LABEL: test_psshar_s_i32x2:
+// RV32: psshar.dws
+// RV64: psshar.ws
+int32x2_t test_psshar_s_i32x2(int32x2_t a, int n) {
+ return __riscv_psshar_s_i32x2(a, n);
+}
+
+// CHECK-LABEL: test_psshl_s_u16x4:
+// RV32: psshl.dhs
+// RV64: psshl.hs
+uint16x4_t test_psshl_s_u16x4(uint16x4_t a, int n) {
+ return __riscv_psshl_s_u16x4(a, n);
+}
+
+// CHECK-LABEL: test_psshl_s_u32x2:
+// RV32: psshl.dws
+// RV64: psshl.ws
+uint32x2_t test_psshl_s_u32x2(uint32x2_t a, int n) {
+ return __riscv_psshl_s_u32x2(a, n);
+}
+
+// CHECK-LABEL: test_psshlr_s_u16x4:
+// RV32: psshlr.dhs
+// RV64: psshlr.hs
+uint16x4_t test_psshlr_s_u16x4(uint16x4_t a, int n) {
+ return __riscv_psshlr_s_u16x4(a, n);
+}
+
+// CHECK-LABEL: test_psshlr_s_u32x2:
+// RV32: psshlr.dws
+// RV64: psshlr.ws
+uint32x2_t test_psshlr_s_u32x2(uint32x2_t a, int n) {
+ return __riscv_psshlr_s_u32x2(a, n);
+}
+
+// CHECK-LABEL: test_pssha_s_i16x2_imm:
+// CHECK: psslai.h{{[[:space:]]+}}{{.*}}, 3
+int16x2_t test_pssha_s_i16x2_imm(int16x2_t a) {
+ return __riscv_pssha_s_i16x2(a, 3);
+}
+
+// CHECK-LABEL: test_pssha_s_i16x2_neg_imm:
+// CHECK: psrai.h{{[[:space:]]+}}{{.*}}, 5
+int16x2_t test_pssha_s_i16x2_neg_imm(int16x2_t a) {
+ return __riscv_pssha_s_i16x2(a, -5);
+}
+
+// CHECK-LABEL: test_psshar_s_i16x2_imm:
+// CHECK: psslai.h{{[[:space:]]+}}{{.*}}, 3
+int16x2_t test_psshar_s_i16x2_imm(int16x2_t a) {
+ return __riscv_psshar_s_i16x2(a, 3);
+}
+
+// CHECK-LABEL: test_psshar_s_i16x2_neg_imm:
+// CHECK: psrari.h{{[[:space:]]+}}{{.*}}, 5
+int16x2_t test_psshar_s_i16x2_neg_imm(int16x2_t a) {
+ return __riscv_psshar_s_i16x2(a, -5);
+}
+
+// CHECK-LABEL: test_pssha_s_i16x4_imm:
+// RV32: psslai.dh{{[[:space:]]+}}{{.*}}, 3
+// RV64: psslai.h{{[[:space:]]+}}{{.*}}, 3
+int16x4_t test_pssha_s_i16x4_imm(int16x4_t a) {
+ return __riscv_pssha_s_i16x4(a, 3);
+}
+
+// CHECK-LABEL: test_pssha_s_i16x4_neg_imm:
+// RV32: psrai.dh{{[[:space:]]+}}{{.*}}, 5
+// RV64: psrai.h{{[[:space:]]+}}{{.*}}, 5
+int16x4_t test_pssha_s_i16x4_neg_imm(int16x4_t a) {
+ return __riscv_pssha_s_i16x4(a, -5);
+}
+
+// CHECK-LABEL: test_pssha_s_i32x2_imm:
+// RV32: psslai.dw{{[[:space:]]+}}{{.*}}, 3
+// RV64: psslai.w{{[[:space:]]+}}{{.*}}, 3
+int32x2_t test_pssha_s_i32x2_imm(int32x2_t a) {
+ return __riscv_pssha_s_i32x2(a, 3);
+}
+
+// CHECK-LABEL: test_pssha_s_i32x2_neg_imm:
+// RV32: psrai.dw{{[[:space:]]+}}{{.*}}, 5
+// RV64: psrai.w{{[[:space:]]+}}{{.*}}, 5
+int32x2_t test_pssha_s_i32x2_neg_imm(int32x2_t a) {
+ return __riscv_pssha_s_i32x2(a, -5);
+}
+
+// CHECK-LABEL: test_psshar_s_i16x4_imm:
+// RV32: psslai.dh{{[[:space:]]+}}{{.*}}, 3
+// RV64: psslai.h{{[[:space:]]+}}{{.*}}, 3
+int16x4_t test_psshar_s_i16x4_imm(int16x4_t a) {
+ return __riscv_psshar_s_i16x4(a, 3);
+}
+
+// CHECK-LABEL: test_psshar_s_i16x4_neg_imm:
+// RV32: psrari.dh{{[[:space:]]+}}{{.*}}, 5
+// RV64: psrari.h{{[[:space:]]+}}{{.*}}, 5
+int16x4_t test_psshar_s_i16x4_neg_imm(int16x4_t a) {
+ return __riscv_psshar_s_i16x4(a, -5);
+}
+
+// CHECK-LABEL: test_psshar_s_i32x2_imm:
+// RV32: psslai.dw{{[[:space:]]+}}{{.*}}, 3
+// RV64: psslai.w{{[[:space:]]+}}{{.*}}, 3
+int32x2_t test_psshar_s_i32x2_imm(int32x2_t a) {
+ return __riscv_psshar_s_i32x2(a, 3);
+}
+
+// CHECK-LABEL: test_psshar_s_i32x2_neg_imm:
+// RV32: psrari.dw{{[[:space:]]+}}{{.*}}, 5
+// RV64: psrari.w{{[[:space:]]+}}{{.*}}, 5
+int32x2_t test_psshar_s_i32x2_neg_imm(int32x2_t a) {
+ return __riscv_psshar_s_i32x2(a, -5);
+}
+
// CHECK-LABEL: test_pand_i8x4:
// CHECK: and{{[[:space:]]}}
int8x4_t test_pand_i8x4(int8x4_t a, int8x4_t b) {
More information about the llvm-commits
mailing list