[clang] [llvm] [Clang][RISCV] Add packed narrowing shift intrinsics (PR #228726)
via cfe-commits
cfe-commits at lists.llvm.org
Sat Oct 3 10:33:42 PDT 2026
https://github.com/Michael-Chen-NJU updated https://github.com/llvm/llvm-project/pull/228726
>From 4e595b2557f54544f43b26d616217639826dde72 Mon Sep 17 00:00:00 2001
From: Michael-Chen-NJU <2802328816 at qq.com>
Date: Sat, 3 Oct 2026 23:50:06 +0800
Subject: [PATCH 1/2] [Clang][RISCV] Add packed narrowing shift intrinsics
---
clang/include/clang/Basic/BuiltinsRISCV.td | 8 +
clang/lib/CodeGen/TargetBuiltins/RISCV.cpp | 17 ++
clang/lib/Headers/riscv_packed_simd.h | 14 ++
clang/test/CodeGen/RISCV/rvp-intrinsics.c | 120 ++++++++++
.../riscv_packed_simd.c | 74 ++++++
llvm/include/llvm/IR/IntrinsicsRISCV.td | 9 +
llvm/lib/Target/RISCV/RISCVISelLowering.cpp | 69 ++++++
llvm/lib/Target/RISCV/RISCVInstrInfoP.td | 29 ++-
.../test/CodeGen/RISCV/rvp-narrowing-shift.ll | 219 ++++++++++++++++++
9 files changed, 558 insertions(+), 1 deletion(-)
create mode 100644 llvm/test/CodeGen/RISCV/rvp-narrowing-shift.ll
diff --git a/clang/include/clang/Basic/BuiltinsRISCV.td b/clang/include/clang/Basic/BuiltinsRISCV.td
index 3a3e85c25b664..c647c25c6ee4e 100644
--- a/clang/include/clang/Basic/BuiltinsRISCV.td
+++ b/clang/include/clang/Basic/BuiltinsRISCV.td
@@ -533,6 +533,14 @@ def pwsll_s_u32x2 : RISCVBuiltin<"_Vector<2, unsigned int>(_Vector<2, unsigned s
def pwsla_s_i16x4 : RISCVBuiltin<"_Vector<4, short>(_Vector<4, signed char>, unsigned int)">;
def pwsla_s_i32x2 : RISCVBuiltin<"_Vector<2, int>(_Vector<2, short>, unsigned int)">;
+// Packed Narrowing Shifts
+def pnsrl_s_u8x4 : RISCVBuiltin<"_Vector<4, unsigned char>(_Vector<4, unsigned short>, unsigned int)">;
+def pnsrl_s_u16x2 : RISCVBuiltin<"_Vector<2, unsigned short>(_Vector<2, unsigned int>, unsigned int)">;
+def pnsra_s_i8x4 : RISCVBuiltin<"_Vector<4, signed char>(_Vector<4, short>, unsigned int)">;
+def pnsra_s_i16x2 : RISCVBuiltin<"_Vector<2, short>(_Vector<2, int>, unsigned int)">;
+def pnsrar_s_i8x4 : RISCVBuiltin<"_Vector<4, signed char>(_Vector<4, short>, unsigned int)">;
+def pnsrar_s_i16x2 : RISCVBuiltin<"_Vector<2, short>(_Vector<2, int>, unsigned int)">;
+
// Packed Narrowing Clip Pair (32-bit)
def pnclipp_i8x4 : RISCVBuiltin<"_Vector<4, signed char>(_Vector<2, short>, _Vector<2, short>)">;
diff --git a/clang/lib/CodeGen/TargetBuiltins/RISCV.cpp b/clang/lib/CodeGen/TargetBuiltins/RISCV.cpp
index 74f85712e4828..8158543e232c3 100644
--- a/clang/lib/CodeGen/TargetBuiltins/RISCV.cpp
+++ b/clang/lib/CodeGen/TargetBuiltins/RISCV.cpp
@@ -1211,6 +1211,23 @@ Value *CodeGenFunction::EmitRISCVBuiltinExpr(unsigned BuiltinID,
IntrinsicTypes = {ResultType, Ops[0]->getType()};
break;
+ // Packed Narrowing Shifts
+ case RISCV::BI__builtin_riscv_pnsrl_s_u8x4:
+ case RISCV::BI__builtin_riscv_pnsrl_s_u16x2:
+ ID = Intrinsic::riscv_pnsrl;
+ IntrinsicTypes = {ResultType, Ops[0]->getType()};
+ break;
+ case RISCV::BI__builtin_riscv_pnsra_s_i8x4:
+ case RISCV::BI__builtin_riscv_pnsra_s_i16x2:
+ ID = Intrinsic::riscv_pnsra;
+ IntrinsicTypes = {ResultType, Ops[0]->getType()};
+ break;
+ case RISCV::BI__builtin_riscv_pnsrar_s_i8x4:
+ case RISCV::BI__builtin_riscv_pnsrar_s_i16x2:
+ ID = Intrinsic::riscv_pnsrar;
+ IntrinsicTypes = {ResultType, Ops[0]->getType()};
+ break;
+
// Packed Averaging Addition and Subtraction
case RISCV::BI__builtin_riscv_paadd_i8x4:
case RISCV::BI__builtin_riscv_paadd_i16x2:
diff --git a/clang/lib/Headers/riscv_packed_simd.h b/clang/lib/Headers/riscv_packed_simd.h
index 9ae0c45d60674..a155278342d23 100644
--- a/clang/lib/Headers/riscv_packed_simd.h
+++ b/clang/lib/Headers/riscv_packed_simd.h
@@ -801,6 +801,20 @@ __packed_binary_builtin_mixed(pwsla_s_i16x4, int16x4_t, int8x4_t, unsigned,
__packed_binary_builtin_mixed(pwsla_s_i32x2, int32x2_t, int16x2_t, unsigned,
__builtin_riscv_pwsla_s_i32x2)
+/* Packed Narrowing Shift */
+__packed_binary_builtin_mixed(pnsrl_s_u8x4, uint8x4_t, uint16x4_t, unsigned,
+ __builtin_riscv_pnsrl_s_u8x4)
+__packed_binary_builtin_mixed(pnsrl_s_u16x2, uint16x2_t, uint32x2_t, unsigned,
+ __builtin_riscv_pnsrl_s_u16x2)
+__packed_binary_builtin_mixed(pnsra_s_i8x4, int8x4_t, int16x4_t, unsigned,
+ __builtin_riscv_pnsra_s_i8x4)
+__packed_binary_builtin_mixed(pnsra_s_i16x2, int16x2_t, int32x2_t, unsigned,
+ __builtin_riscv_pnsra_s_i16x2)
+__packed_binary_builtin_mixed(pnsrar_s_i8x4, int8x4_t, int16x4_t, unsigned,
+ __builtin_riscv_pnsrar_s_i8x4)
+__packed_binary_builtin_mixed(pnsrar_s_i16x2, int16x2_t, int32x2_t, unsigned,
+ __builtin_riscv_pnsrar_s_i16x2)
+
/* Packed Widening Addition and Subtraction */
__packed_widen_binary_op(pwadd_i16x4, int16x4_t, int8x4_t, +)
__packed_widen_binary_op(pwadd_i32x2, int32x2_t, int16x2_t, +)
diff --git a/clang/test/CodeGen/RISCV/rvp-intrinsics.c b/clang/test/CodeGen/RISCV/rvp-intrinsics.c
index d9ad53c373da2..8b0868312cdd1 100644
--- a/clang/test/CodeGen/RISCV/rvp-intrinsics.c
+++ b/clang/test/CodeGen/RISCV/rvp-intrinsics.c
@@ -5512,6 +5512,126 @@ int32x2_t test_pwsla_s_i32x2(int16x2_t rs1, unsigned shamt) {
return __riscv_pwsla_s_i32x2(rs1, shamt);
}
+// RV32-LABEL: define dso_local i32 @test_pnsrl_s_u8x4(
+// RV32-SAME: i64 noundef [[RS1_COERCE:%.*]], i32 noundef [[SHAMT:%.*]]) #[[ATTR0]] {
+// RV32-NEXT: [[ENTRY:.*:]]
+// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[RS1_COERCE]] to <4 x i16>
+// RV32-NEXT: [[TMP1:%.*]] = call <4 x i8> @llvm.riscv.pnsrl.v4i8.v4i16(<4 x i16> [[TMP0]], i32 [[SHAMT]])
+// RV32-NEXT: [[TMP2:%.*]] = bitcast <4 x i8> [[TMP1]] to i32
+// RV32-NEXT: ret i32 [[TMP2]]
+//
+// RV64-LABEL: define dso_local i32 @test_pnsrl_s_u8x4(
+// RV64-SAME: i64 noundef [[RS1_COERCE:%.*]], i32 noundef signext [[SHAMT:%.*]]) #[[ATTR0]] {
+// RV64-NEXT: [[ENTRY:.*:]]
+// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[RS1_COERCE]] to <4 x i16>
+// RV64-NEXT: [[TMP1:%.*]] = call <4 x i8> @llvm.riscv.pnsrl.v4i8.v4i16(<4 x i16> [[TMP0]], i32 [[SHAMT]])
+// RV64-NEXT: [[TMP2:%.*]] = bitcast <4 x i8> [[TMP1]] to i32
+// RV64-NEXT: ret i32 [[TMP2]]
+//
+uint8x4_t test_pnsrl_s_u8x4(uint16x4_t rs1, unsigned shamt) {
+ return __riscv_pnsrl_s_u8x4(rs1, shamt);
+}
+
+// RV32-LABEL: define dso_local i32 @test_pnsrl_s_u16x2(
+// RV32-SAME: i64 noundef [[RS1_COERCE:%.*]], i32 noundef [[SHAMT:%.*]]) #[[ATTR0]] {
+// RV32-NEXT: [[ENTRY:.*:]]
+// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[RS1_COERCE]] to <2 x i32>
+// RV32-NEXT: [[TMP1:%.*]] = call <2 x i16> @llvm.riscv.pnsrl.v2i16.v2i32(<2 x i32> [[TMP0]], i32 [[SHAMT]])
+// RV32-NEXT: [[TMP2:%.*]] = bitcast <2 x i16> [[TMP1]] to i32
+// RV32-NEXT: ret i32 [[TMP2]]
+//
+// RV64-LABEL: define dso_local i32 @test_pnsrl_s_u16x2(
+// RV64-SAME: i64 noundef [[RS1_COERCE:%.*]], i32 noundef signext [[SHAMT:%.*]]) #[[ATTR0]] {
+// RV64-NEXT: [[ENTRY:.*:]]
+// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[RS1_COERCE]] to <2 x i32>
+// RV64-NEXT: [[TMP1:%.*]] = call <2 x i16> @llvm.riscv.pnsrl.v2i16.v2i32(<2 x i32> [[TMP0]], i32 [[SHAMT]])
+// RV64-NEXT: [[TMP2:%.*]] = bitcast <2 x i16> [[TMP1]] to i32
+// RV64-NEXT: ret i32 [[TMP2]]
+//
+uint16x2_t test_pnsrl_s_u16x2(uint32x2_t rs1, unsigned shamt) {
+ return __riscv_pnsrl_s_u16x2(rs1, shamt);
+}
+
+// RV32-LABEL: define dso_local i32 @test_pnsra_s_i8x4(
+// RV32-SAME: i64 noundef [[RS1_COERCE:%.*]], i32 noundef [[SHAMT:%.*]]) #[[ATTR0]] {
+// RV32-NEXT: [[ENTRY:.*:]]
+// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[RS1_COERCE]] to <4 x i16>
+// RV32-NEXT: [[TMP1:%.*]] = call <4 x i8> @llvm.riscv.pnsra.v4i8.v4i16(<4 x i16> [[TMP0]], i32 [[SHAMT]])
+// RV32-NEXT: [[TMP2:%.*]] = bitcast <4 x i8> [[TMP1]] to i32
+// RV32-NEXT: ret i32 [[TMP2]]
+//
+// RV64-LABEL: define dso_local i32 @test_pnsra_s_i8x4(
+// RV64-SAME: i64 noundef [[RS1_COERCE:%.*]], i32 noundef signext [[SHAMT:%.*]]) #[[ATTR0]] {
+// RV64-NEXT: [[ENTRY:.*:]]
+// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[RS1_COERCE]] to <4 x i16>
+// RV64-NEXT: [[TMP1:%.*]] = call <4 x i8> @llvm.riscv.pnsra.v4i8.v4i16(<4 x i16> [[TMP0]], i32 [[SHAMT]])
+// RV64-NEXT: [[TMP2:%.*]] = bitcast <4 x i8> [[TMP1]] to i32
+// RV64-NEXT: ret i32 [[TMP2]]
+//
+int8x4_t test_pnsra_s_i8x4(int16x4_t rs1, unsigned shamt) {
+ return __riscv_pnsra_s_i8x4(rs1, shamt);
+}
+
+// RV32-LABEL: define dso_local i32 @test_pnsra_s_i16x2(
+// RV32-SAME: i64 noundef [[RS1_COERCE:%.*]], i32 noundef [[SHAMT:%.*]]) #[[ATTR0]] {
+// RV32-NEXT: [[ENTRY:.*:]]
+// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[RS1_COERCE]] to <2 x i32>
+// RV32-NEXT: [[TMP1:%.*]] = call <2 x i16> @llvm.riscv.pnsra.v2i16.v2i32(<2 x i32> [[TMP0]], i32 [[SHAMT]])
+// RV32-NEXT: [[TMP2:%.*]] = bitcast <2 x i16> [[TMP1]] to i32
+// RV32-NEXT: ret i32 [[TMP2]]
+//
+// RV64-LABEL: define dso_local i32 @test_pnsra_s_i16x2(
+// RV64-SAME: i64 noundef [[RS1_COERCE:%.*]], i32 noundef signext [[SHAMT:%.*]]) #[[ATTR0]] {
+// RV64-NEXT: [[ENTRY:.*:]]
+// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[RS1_COERCE]] to <2 x i32>
+// RV64-NEXT: [[TMP1:%.*]] = call <2 x i16> @llvm.riscv.pnsra.v2i16.v2i32(<2 x i32> [[TMP0]], i32 [[SHAMT]])
+// RV64-NEXT: [[TMP2:%.*]] = bitcast <2 x i16> [[TMP1]] to i32
+// RV64-NEXT: ret i32 [[TMP2]]
+//
+int16x2_t test_pnsra_s_i16x2(int32x2_t rs1, unsigned shamt) {
+ return __riscv_pnsra_s_i16x2(rs1, shamt);
+}
+
+// RV32-LABEL: define dso_local i32 @test_pnsrar_s_i8x4(
+// RV32-SAME: i64 noundef [[RS1_COERCE:%.*]], i32 noundef [[SHAMT:%.*]]) #[[ATTR0]] {
+// RV32-NEXT: [[ENTRY:.*:]]
+// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[RS1_COERCE]] to <4 x i16>
+// RV32-NEXT: [[TMP1:%.*]] = call <4 x i8> @llvm.riscv.pnsrar.v4i8.v4i16(<4 x i16> [[TMP0]], i32 [[SHAMT]])
+// RV32-NEXT: [[TMP2:%.*]] = bitcast <4 x i8> [[TMP1]] to i32
+// RV32-NEXT: ret i32 [[TMP2]]
+//
+// RV64-LABEL: define dso_local i32 @test_pnsrar_s_i8x4(
+// RV64-SAME: i64 noundef [[RS1_COERCE:%.*]], i32 noundef signext [[SHAMT:%.*]]) #[[ATTR0]] {
+// RV64-NEXT: [[ENTRY:.*:]]
+// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[RS1_COERCE]] to <4 x i16>
+// RV64-NEXT: [[TMP1:%.*]] = call <4 x i8> @llvm.riscv.pnsrar.v4i8.v4i16(<4 x i16> [[TMP0]], i32 [[SHAMT]])
+// RV64-NEXT: [[TMP2:%.*]] = bitcast <4 x i8> [[TMP1]] to i32
+// RV64-NEXT: ret i32 [[TMP2]]
+//
+int8x4_t test_pnsrar_s_i8x4(int16x4_t rs1, unsigned shamt) {
+ return __riscv_pnsrar_s_i8x4(rs1, shamt);
+}
+
+// RV32-LABEL: define dso_local i32 @test_pnsrar_s_i16x2(
+// RV32-SAME: i64 noundef [[RS1_COERCE:%.*]], i32 noundef [[SHAMT:%.*]]) #[[ATTR0]] {
+// RV32-NEXT: [[ENTRY:.*:]]
+// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[RS1_COERCE]] to <2 x i32>
+// RV32-NEXT: [[TMP1:%.*]] = call <2 x i16> @llvm.riscv.pnsrar.v2i16.v2i32(<2 x i32> [[TMP0]], i32 [[SHAMT]])
+// RV32-NEXT: [[TMP2:%.*]] = bitcast <2 x i16> [[TMP1]] to i32
+// RV32-NEXT: ret i32 [[TMP2]]
+//
+// RV64-LABEL: define dso_local i32 @test_pnsrar_s_i16x2(
+// RV64-SAME: i64 noundef [[RS1_COERCE:%.*]], i32 noundef signext [[SHAMT:%.*]]) #[[ATTR0]] {
+// RV64-NEXT: [[ENTRY:.*:]]
+// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[RS1_COERCE]] to <2 x i32>
+// RV64-NEXT: [[TMP1:%.*]] = call <2 x i16> @llvm.riscv.pnsrar.v2i16.v2i32(<2 x i32> [[TMP0]], i32 [[SHAMT]])
+// RV64-NEXT: [[TMP2:%.*]] = bitcast <2 x i16> [[TMP1]] to i32
+// RV64-NEXT: ret i32 [[TMP2]]
+//
+int16x2_t test_pnsrar_s_i16x2(int32x2_t rs1, unsigned shamt) {
+ return __riscv_pnsrar_s_i16x2(rs1, shamt);
+}
+
// CHECK-LABEL: define dso_local i64 @test_pwadd_i16x4(
// CHECK-SAME: i32 noundef [[RS1_COERCE:%.*]], i32 noundef [[RS2_COERCE:%.*]]) #[[ATTR0]] {
// CHECK-NEXT: [[ENTRY:.*:]]
diff --git a/cross-project-tests/intrinsic-header-tests/riscv_packed_simd.c b/cross-project-tests/intrinsic-header-tests/riscv_packed_simd.c
index 6d92dd5990261..9b462245dc786 100644
--- a/cross-project-tests/intrinsic-header-tests/riscv_packed_simd.c
+++ b/cross-project-tests/intrinsic-header-tests/riscv_packed_simd.c
@@ -2439,6 +2439,80 @@ uint32x2_t test_pwsll_s_u32x2_low5(uint16x2_t rs1) {
return __riscv_pwsll_s_u32x2(rs1, 63);
}
+// CHECK-LABEL: test_pnsrl_s_u8x4:
+// RV32: pnsrl.bs
+// RV64: psrl.hs
+// RV64: pncvt.wb
+uint8x4_t test_pnsrl_s_u8x4(uint16x4_t rs1, unsigned shamt) {
+ return __riscv_pnsrl_s_u8x4(rs1, shamt);
+}
+
+// CHECK-LABEL: test_pnsrl_s_u16x2:
+// RV32: pnsrl.hs
+// RV64: psrl.ws
+// RV64: pncvt.wh
+uint16x2_t test_pnsrl_s_u16x2(uint32x2_t rs1, unsigned shamt) {
+ return __riscv_pnsrl_s_u16x2(rs1, shamt);
+}
+
+// CHECK-LABEL: test_pnsra_s_i8x4:
+// RV32: pnsra.bs
+// RV64: psra.hs
+// RV64: pncvt.wb
+int8x4_t test_pnsra_s_i8x4(int16x4_t rs1, unsigned shamt) {
+ return __riscv_pnsra_s_i8x4(rs1, shamt);
+}
+
+// CHECK-LABEL: test_pnsra_s_i16x2:
+// RV32: pnsra.hs
+// RV64: psra.ws
+// RV64: pncvt.wh
+int16x2_t test_pnsra_s_i16x2(int32x2_t rs1, unsigned shamt) {
+ return __riscv_pnsra_s_i16x2(rs1, shamt);
+}
+
+// CHECK-LABEL: test_pnsrar_s_i8x4:
+// RV32: pnsrar.bs
+// RV64: psshar.hs
+// RV64: pncvt.wb
+int8x4_t test_pnsrar_s_i8x4(int16x4_t rs1, unsigned shamt) {
+ return __riscv_pnsrar_s_i8x4(rs1, shamt);
+}
+
+// CHECK-LABEL: test_pnsrar_s_i16x2:
+// RV32: pnsrar.hs
+// RV64: psshar.ws
+// RV64: pncvt.wh
+int16x2_t test_pnsrar_s_i16x2(int32x2_t rs1, unsigned shamt) {
+ return __riscv_pnsrar_s_i16x2(rs1, shamt);
+}
+
+// Counts that exceed the result width still use the low five bits of shamt.
+// CHECK-LABEL: test_pnsrl_s_u8x4_16:
+// RV32: li{{[[:space:]]}}a2, 16
+// RV32-NEXT: pnsrl.bs{{[[:space:]]}}a0, a0, a2
+// RV64: li{{[[:space:]]}}a1, 16
+// RV64: psrl.hs{{[[:space:]]}}a0, a0, a1
+uint8x4_t test_pnsrl_s_u8x4_16(uint16x4_t rs1) {
+ return __riscv_pnsrl_s_u8x4(rs1, 16);
+}
+
+// CHECK-LABEL: test_pnsrl_s_u16x2_31:
+// RV32: pnsrli.h{{[[:space:]]}}a0, a0, 31
+// RV64: psrli.w{{[[:space:]]}}a0, a0, 31
+uint16x2_t test_pnsrl_s_u16x2_31(uint32x2_t rs1) {
+ return __riscv_pnsrl_s_u16x2(rs1, 31);
+}
+
+// CHECK-LABEL: test_pnsrar_s_i8x4_16:
+// RV32: li{{[[:space:]]}}a2, 16
+// RV32-NEXT: pnsrar.bs{{[[:space:]]}}a0, a0, a2
+// RV64: li{{[[:space:]]}}a1, -16
+// RV64-NEXT: psshar.hs{{[[:space:]]}}a0, a0, a1
+int8x4_t test_pnsrar_s_i8x4_16(int16x4_t rs1) {
+ return __riscv_pnsrar_s_i8x4(rs1, 16);
+}
+
// CHECK-LABEL: test_pwadd_i16x4:
// RV32: pwadd.b
// RV64: zip8p
diff --git a/llvm/include/llvm/IR/IntrinsicsRISCV.td b/llvm/include/llvm/IR/IntrinsicsRISCV.td
index 3ff0b10bb6f53..390e35b8e5d8c 100644
--- a/llvm/include/llvm/IR/IntrinsicsRISCV.td
+++ b/llvm/include/llvm/IR/IntrinsicsRISCV.td
@@ -2089,6 +2089,15 @@ class RVPBinaryIntrinsic
def int_riscv_pwsll : RVPWideningShiftIntrinsic;
def int_riscv_pwsla : RVPWideningShiftIntrinsic;
+ // Packed Narrowing Shifts.
+ class RVPNarrowingShiftIntrinsic
+ : DefaultAttrsIntrinsic<[llvm_anyvector_ty],
+ [llvm_anyvector_ty, llvm_i32_ty],
+ [IntrNoMem, IntrSpeculatable]>;
+ def int_riscv_pnsrl : RVPNarrowingShiftIntrinsic;
+ def int_riscv_pnsra : RVPNarrowingShiftIntrinsic;
+ def int_riscv_pnsrar : RVPNarrowingShiftIntrinsic;
+
// Packed Saturation.
class RVPSaturationIntrinsic
: DefaultAttrsIntrinsic<[llvm_anyvector_ty],
diff --git a/llvm/lib/Target/RISCV/RISCVISelLowering.cpp b/llvm/lib/Target/RISCV/RISCVISelLowering.cpp
index 3892e62f715cf..f416bd5b4f90b 100644
--- a/llvm/lib/Target/RISCV/RISCVISelLowering.cpp
+++ b/llvm/lib/Target/RISCV/RISCVISelLowering.cpp
@@ -12394,6 +12394,19 @@ static unsigned getRVPShiftOpcode(Intrinsic::ID IntNo) {
}
}
+static unsigned getRVPNarrowingShiftOpcode(Intrinsic::ID IntNo) {
+ switch (IntNo) {
+ default:
+ llvm_unreachable("Unexpected RISC-V packed narrowing shift intrinsic");
+ case Intrinsic::riscv_pnsrl:
+ return RISCVISD::PNSRL;
+ case Intrinsic::riscv_pnsra:
+ return RISCVISD::PNSRA;
+ case Intrinsic::riscv_pnsrar:
+ return RISCVISD::PNSRAR;
+ }
+}
+
static SDValue lowerPZExt(SDValue Src, const SDLoc &DL, SelectionDAG &DAG,
const RISCVSubtarget &Subtarget) {
MVT VT = Src.getSimpleValueType();
@@ -13275,6 +13288,19 @@ SDValue RISCVTargetLowering::LowerINTRINSIC_WO_CHAIN(SDValue Op,
SDValue Wide = DAG.getNode(ExtOpc, DL, VT, Src);
return DAG.getNode(RISCVISD::PSLL, DL, VT, Wide, ShAmt);
}
+ case Intrinsic::riscv_pnsrl:
+ case Intrinsic::riscv_pnsra:
+ case Intrinsic::riscv_pnsrar: {
+ MVT VT = Op.getSimpleValueType();
+ MVT SrcVT = Op.getOperand(1).getSimpleValueType();
+ if (!((VT == MVT::v4i8 && SrcVT == MVT::v4i16) ||
+ (VT == MVT::v2i16 && SrcVT == MVT::v2i32)))
+ reportFatalUsageError("unsupported packed narrowing shift intrinsic");
+
+ SDValue ShAmt = DAG.getNode(ISD::ANY_EXTEND, DL, XLenVT, Op.getOperand(2));
+ return DAG.getNode(getRVPNarrowingShiftOpcode(IntNo), DL, VT,
+ Op.getOperand(1), ShAmt);
+ }
case Intrinsic::riscv_psati:
case Intrinsic::riscv_pusati: {
bool IsSigned = IntNo == Intrinsic::riscv_psati;
@@ -18062,6 +18088,49 @@ void RISCVTargetLowering::ReplaceNodeResults(SDNode *N,
Results.push_back(DAG.getExtractSubvector(DL, VT, Res, 0));
return;
}
+ case Intrinsic::riscv_pnsrl:
+ case Intrinsic::riscv_pnsra:
+ case Intrinsic::riscv_pnsrar: {
+ MVT VT = N->getSimpleValueType(0);
+ if (!Subtarget.is64Bit() || (VT != MVT::v4i8 && VT != MVT::v2i16))
+ return;
+
+ SDValue Src = N->getOperand(1);
+ MVT SrcVT = Src.getSimpleValueType();
+ if (!((VT == MVT::v4i8 && SrcVT == MVT::v4i16) ||
+ (VT == MVT::v2i16 && SrcVT == MVT::v2i32)))
+ reportFatalUsageError("unsupported packed narrowing shift intrinsic");
+
+ MVT XLenVT = Subtarget.getXLenVT();
+ SDValue ShAmt =
+ DAG.getNode(ISD::ANY_EXTEND, DL, XLenVT, N->getOperand(2));
+ unsigned Opc;
+ switch (IntNo) {
+ default:
+ llvm_unreachable("Unexpected packed narrowing shift intrinsic");
+ case Intrinsic::riscv_pnsrl:
+ Opc = RISCVISD::PSRL;
+ break;
+ case Intrinsic::riscv_pnsra:
+ Opc = RISCVISD::PSRA;
+ break;
+ case Intrinsic::riscv_pnsrar:
+ Opc = RISCVISD::PSSHAR;
+ ShAmt = DAG.getNode(ISD::AND, DL, XLenVT, ShAmt,
+ DAG.getConstant(31, DL, XLenVT));
+ ShAmt = DAG.getNode(ISD::SUB, DL, XLenVT,
+ DAG.getConstant(0, DL, XLenVT), ShAmt);
+ break;
+ }
+
+ SDValue Shifted = DAG.getNode(Opc, DL, SrcVT, Src, ShAmt);
+ MVT UnzipVT = VT == MVT::v4i8 ? MVT::v8i8 : MVT::v4i16;
+ Shifted = DAG.getBitcast(UnzipVT, Shifted);
+ SDValue Unzip = DAG.getNode(RISCVISD::PUNZIPE, DL, UnzipVT, Shifted,
+ DAG.getUNDEF(UnzipVT));
+ Results.push_back(DAG.getExtractSubvector(DL, VT, Unzip, 0));
+ return;
+ }
case Intrinsic::riscv_predsum:
case Intrinsic::riscv_predsumu: {
bool IsSigned = IntNo == Intrinsic::riscv_predsum;
diff --git a/llvm/lib/Target/RISCV/RISCVInstrInfoP.td b/llvm/lib/Target/RISCV/RISCVInstrInfoP.td
index ee5b17749cb73..4a3b0be8949f6 100644
--- a/llvm/lib/Target/RISCV/RISCVInstrInfoP.td
+++ b/llvm/lib/Target/RISCV/RISCVInstrInfoP.td
@@ -2089,7 +2089,9 @@ def riscv_psshlr : RVSDNode<"PSSHLR", SDT_RISCVPackedShift>;
def SDT_RISCVPackedNarrowingShift
: SDTypeProfile<1, 2, [SDTCisVec<0>, SDTCisVec<1>,
SDTCisVT<2, XLenVT>]>;
-def riscv_pnsrl : RVSDNode<"PNSRL", SDT_RISCVPackedNarrowingShift>;
+def riscv_pnsrl : RVSDNode<"PNSRL", SDT_RISCVPackedNarrowingShift>;
+def riscv_pnsra : RVSDNode<"PNSRA", SDT_RISCVPackedNarrowingShift>;
+def riscv_pnsrar : RVSDNode<"PNSRAR", SDT_RISCVPackedNarrowingShift>;
def riscv_pnclip : RVSDNode<"PNCLIP", SDT_RISCVPackedNarrowingShift>;
def riscv_pnclipu : RVSDNode<"PNCLIPU", SDT_RISCVPackedNarrowingShift>;
@@ -2821,6 +2823,31 @@ let append Predicates = [IsRV32] in {
(PNSRLI_B GPRPair:$rs1, uimm4:$imm)>;
def : Pat<(v2i16 (riscv_pnsrl (v4i16 GPRPair:$rs1), uimm5:$imm)),
(PNSRLI_H GPRPair:$rs1, uimm5:$imm)>;
+ def : Pat<(v4i8 (riscv_pnsrl (v4i16 GPRPair:$rs1), uimm4:$imm)),
+ (PNSRLI_B GPRPair:$rs1, uimm4:$imm)>;
+ def : Pat<(v2i16 (riscv_pnsrl (v2i32 GPRPair:$rs1), uimm5:$imm)),
+ (PNSRLI_H GPRPair:$rs1, uimm5:$imm)>;
+ def : Pat<(v4i8 (riscv_pnsra (v4i16 GPRPair:$rs1), uimm4:$imm)),
+ (PNSRAI_B GPRPair:$rs1, uimm4:$imm)>;
+ def : Pat<(v2i16 (riscv_pnsra (v2i32 GPRPair:$rs1), uimm5:$imm)),
+ (PNSRAI_H GPRPair:$rs1, uimm5:$imm)>;
+ def : Pat<(v4i8 (riscv_pnsrar (v4i16 GPRPair:$rs1), uimm4:$imm)),
+ (PNSRARI_B GPRPair:$rs1, uimm4:$imm)>;
+ def : Pat<(v2i16 (riscv_pnsrar (v2i32 GPRPair:$rs1), uimm5:$imm)),
+ (PNSRARI_H GPRPair:$rs1, uimm5:$imm)>;
+
+ def : Pat<(v4i8 (riscv_pnsrl (v4i16 GPRPair:$rs1), shiftMask32:$rs2)),
+ (PNSRL_BS GPRPair:$rs1, shiftMask32:$rs2)>;
+ def : Pat<(v2i16 (riscv_pnsrl (v2i32 GPRPair:$rs1), shiftMask32:$rs2)),
+ (PNSRL_HS GPRPair:$rs1, shiftMask32:$rs2)>;
+ def : Pat<(v4i8 (riscv_pnsra (v4i16 GPRPair:$rs1), shiftMask32:$rs2)),
+ (PNSRA_BS GPRPair:$rs1, shiftMask32:$rs2)>;
+ def : Pat<(v2i16 (riscv_pnsra (v2i32 GPRPair:$rs1), shiftMask32:$rs2)),
+ (PNSRA_HS GPRPair:$rs1, shiftMask32:$rs2)>;
+ def : Pat<(v4i8 (riscv_pnsrar (v4i16 GPRPair:$rs1), shiftMask32:$rs2)),
+ (PNSRAR_BS GPRPair:$rs1, shiftMask32:$rs2)>;
+ def : Pat<(v2i16 (riscv_pnsrar (v2i32 GPRPair:$rs1), shiftMask32:$rs2)),
+ (PNSRAR_HS GPRPair:$rs1, shiftMask32:$rs2)>;
def : Pat<(v4i8 (trunc (v4i16 (riscv_psrl GPRPair:$rs1, uimm4:$imm)))),
(PNSRLI_B GPRPair:$rs1, uimm4:$imm)>;
diff --git a/llvm/test/CodeGen/RISCV/rvp-narrowing-shift.ll b/llvm/test/CodeGen/RISCV/rvp-narrowing-shift.ll
new file mode 100644
index 0000000000000..d5a7295fe12c8
--- /dev/null
+++ b/llvm/test/CodeGen/RISCV/rvp-narrowing-shift.ll
@@ -0,0 +1,219 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc -mtriple=riscv32 -mattr=+experimental-p -verify-machineinstrs < %s | \
+; RUN: FileCheck %s --check-prefix=RV32
+; RUN: llc -mtriple=riscv64 -mattr=+experimental-p -verify-machineinstrs < %s | \
+; RUN: FileCheck %s --check-prefix=RV64
+
+declare <4 x i8> @llvm.riscv.pnsrl.v4i8.v4i16(<4 x i16>, i32)
+declare <2 x i16> @llvm.riscv.pnsrl.v2i16.v2i32(<2 x i32>, i32)
+declare <4 x i8> @llvm.riscv.pnsra.v4i8.v4i16(<4 x i16>, i32)
+declare <2 x i16> @llvm.riscv.pnsra.v2i16.v2i32(<2 x i32>, i32)
+declare <4 x i8> @llvm.riscv.pnsrar.v4i8.v4i16(<4 x i16>, i32)
+declare <2 x i16> @llvm.riscv.pnsrar.v2i16.v2i32(<2 x i32>, i32)
+
+define <4 x i8> @pnsrl_v4i8(<4 x i16> %x, i32 %shamt) {
+; RV32-LABEL: pnsrl_v4i8:
+; RV32: # %bb.0:
+; RV32-NEXT: pnsrl.bs a0, a0, a2
+; RV32-NEXT: ret
+;
+; RV64-LABEL: pnsrl_v4i8:
+; RV64: # %bb.0:
+; RV64-NEXT: psrl.hs a0, a0, a1
+; RV64-NEXT: pncvt.wb a0, a0
+; RV64-NEXT: ret
+ %res = call <4 x i8> @llvm.riscv.pnsrl.v4i8.v4i16(<4 x i16> %x, i32 %shamt)
+ ret <4 x i8> %res
+}
+
+define <2 x i16> @pnsrl_v2i16(<2 x i32> %x, i32 %shamt) {
+; RV32-LABEL: pnsrl_v2i16:
+; RV32: # %bb.0:
+; RV32-NEXT: pnsrl.hs a0, a0, a2
+; RV32-NEXT: ret
+;
+; RV64-LABEL: pnsrl_v2i16:
+; RV64: # %bb.0:
+; RV64-NEXT: psrl.ws a0, a0, a1
+; RV64-NEXT: pncvt.wh a0, a0
+; RV64-NEXT: ret
+ %res = call <2 x i16> @llvm.riscv.pnsrl.v2i16.v2i32(<2 x i32> %x, i32 %shamt)
+ ret <2 x i16> %res
+}
+
+define <4 x i8> @pnsra_v4i8(<4 x i16> %x, i32 %shamt) {
+; RV32-LABEL: pnsra_v4i8:
+; RV32: # %bb.0:
+; RV32-NEXT: pnsra.bs a0, a0, a2
+; RV32-NEXT: ret
+;
+; RV64-LABEL: pnsra_v4i8:
+; RV64: # %bb.0:
+; RV64-NEXT: psra.hs a0, a0, a1
+; RV64-NEXT: pncvt.wb a0, a0
+; RV64-NEXT: ret
+ %res = call <4 x i8> @llvm.riscv.pnsra.v4i8.v4i16(<4 x i16> %x, i32 %shamt)
+ ret <4 x i8> %res
+}
+
+define <2 x i16> @pnsra_v2i16(<2 x i32> %x, i32 %shamt) {
+; RV32-LABEL: pnsra_v2i16:
+; RV32: # %bb.0:
+; RV32-NEXT: pnsra.hs a0, a0, a2
+; RV32-NEXT: ret
+;
+; RV64-LABEL: pnsra_v2i16:
+; RV64: # %bb.0:
+; RV64-NEXT: psra.ws a0, a0, a1
+; RV64-NEXT: pncvt.wh a0, a0
+; RV64-NEXT: ret
+ %res = call <2 x i16> @llvm.riscv.pnsra.v2i16.v2i32(<2 x i32> %x, i32 %shamt)
+ ret <2 x i16> %res
+}
+
+define <4 x i8> @pnsrar_v4i8(<4 x i16> %x, i32 %shamt) {
+; RV32-LABEL: pnsrar_v4i8:
+; RV32: # %bb.0:
+; RV32-NEXT: pnsrar.bs a0, a0, a2
+; RV32-NEXT: ret
+;
+; RV64-LABEL: pnsrar_v4i8:
+; RV64: # %bb.0:
+; RV64-NEXT: andi a1, a1, 31
+; RV64-NEXT: neg a1, a1
+; RV64-NEXT: psshar.hs a0, a0, a1
+; RV64-NEXT: pncvt.wb a0, a0
+; RV64-NEXT: ret
+ %res = call <4 x i8> @llvm.riscv.pnsrar.v4i8.v4i16(<4 x i16> %x, i32 %shamt)
+ ret <4 x i8> %res
+}
+
+define <2 x i16> @pnsrar_v2i16(<2 x i32> %x, i32 %shamt) {
+; RV32-LABEL: pnsrar_v2i16:
+; RV32: # %bb.0:
+; RV32-NEXT: pnsrar.hs a0, a0, a2
+; RV32-NEXT: ret
+;
+; RV64-LABEL: pnsrar_v2i16:
+; RV64: # %bb.0:
+; RV64-NEXT: andi a1, a1, 31
+; RV64-NEXT: neg a1, a1
+; RV64-NEXT: psshar.ws a0, a0, a1
+; RV64-NEXT: pncvt.wh a0, a0
+; RV64-NEXT: ret
+ %res = call <2 x i16> @llvm.riscv.pnsrar.v2i16.v2i32(<2 x i32> %x, i32 %shamt)
+ ret <2 x i16> %res
+}
+
+define <4 x i8> @pnsrl_v4i8_15(<4 x i16> %x) {
+; RV32-LABEL: pnsrl_v4i8_15:
+; RV32: # %bb.0:
+; RV32-NEXT: pnsrli.b a0, a0, 15
+; RV32-NEXT: ret
+;
+; RV64-LABEL: pnsrl_v4i8_15:
+; RV64: # %bb.0:
+; RV64-NEXT: psrli.h a0, a0, 15
+; RV64-NEXT: pncvt.wb a0, a0
+; RV64-NEXT: ret
+ %res = call <4 x i8> @llvm.riscv.pnsrl.v4i8.v4i16(<4 x i16> %x, i32 15)
+ ret <4 x i8> %res
+}
+
+define <4 x i8> @pnsrl_v4i8_16(<4 x i16> %x) {
+; RV32-LABEL: pnsrl_v4i8_16:
+; RV32: # %bb.0:
+; RV32-NEXT: li a2, 16
+; RV32-NEXT: pnsrl.bs a0, a0, a2
+; RV32-NEXT: ret
+;
+; RV64-LABEL: pnsrl_v4i8_16:
+; RV64: # %bb.0:
+; RV64-NEXT: li a1, 16
+; RV64-NEXT: psrl.hs a0, a0, a1
+; RV64-NEXT: pncvt.wb a0, a0
+; RV64-NEXT: ret
+ %res = call <4 x i8> @llvm.riscv.pnsrl.v4i8.v4i16(<4 x i16> %x, i32 16)
+ ret <4 x i8> %res
+}
+
+define <4 x i8> @pnsrl_v4i8_32(<4 x i16> %x) {
+; RV32-LABEL: pnsrl_v4i8_32:
+; RV32: # %bb.0:
+; RV32-NEXT: li a2, 32
+; RV32-NEXT: pnsrl.bs a0, a0, a2
+; RV32-NEXT: ret
+;
+; RV64-LABEL: pnsrl_v4i8_32:
+; RV64: # %bb.0:
+; RV64-NEXT: li a1, 32
+; RV64-NEXT: psrl.hs a0, a0, a1
+; RV64-NEXT: pncvt.wb a0, a0
+; RV64-NEXT: ret
+ %res = call <4 x i8> @llvm.riscv.pnsrl.v4i8.v4i16(<4 x i16> %x, i32 32)
+ ret <4 x i8> %res
+}
+
+define <2 x i16> @pnsra_v2i16_31(<2 x i32> %x) {
+; RV32-LABEL: pnsra_v2i16_31:
+; RV32: # %bb.0:
+; RV32-NEXT: pnsrai.h a0, a0, 31
+; RV32-NEXT: ret
+;
+; RV64-LABEL: pnsra_v2i16_31:
+; RV64: # %bb.0:
+; RV64-NEXT: psrai.w a0, a0, 31
+; RV64-NEXT: pncvt.wh a0, a0
+; RV64-NEXT: ret
+ %res = call <2 x i16> @llvm.riscv.pnsra.v2i16.v2i32(<2 x i32> %x, i32 31)
+ ret <2 x i16> %res
+}
+
+define <2 x i16> @pnsra_v2i16_32(<2 x i32> %x) {
+; RV32-LABEL: pnsra_v2i16_32:
+; RV32: # %bb.0:
+; RV32-NEXT: li a2, 32
+; RV32-NEXT: pnsra.hs a0, a0, a2
+; RV32-NEXT: ret
+;
+; RV64-LABEL: pnsra_v2i16_32:
+; RV64: # %bb.0:
+; RV64-NEXT: li a1, 32
+; RV64-NEXT: psra.ws a0, a0, a1
+; RV64-NEXT: pncvt.wh a0, a0
+; RV64-NEXT: ret
+ %res = call <2 x i16> @llvm.riscv.pnsra.v2i16.v2i32(<2 x i32> %x, i32 32)
+ ret <2 x i16> %res
+}
+
+define <4 x i8> @pnsrar_v4i8_15(<4 x i16> %x) {
+; RV32-LABEL: pnsrar_v4i8_15:
+; RV32: # %bb.0:
+; RV32-NEXT: pnsrari.b a0, a0, 15
+; RV32-NEXT: ret
+;
+; RV64-LABEL: pnsrar_v4i8_15:
+; RV64: # %bb.0:
+; RV64-NEXT: psrari.h a0, a0, 15
+; RV64-NEXT: pncvt.wb a0, a0
+; RV64-NEXT: ret
+ %res = call <4 x i8> @llvm.riscv.pnsrar.v4i8.v4i16(<4 x i16> %x, i32 15)
+ ret <4 x i8> %res
+}
+
+define <4 x i8> @pnsrar_v4i8_16(<4 x i16> %x) {
+; RV32-LABEL: pnsrar_v4i8_16:
+; RV32: # %bb.0:
+; RV32-NEXT: li a2, 16
+; RV32-NEXT: pnsrar.bs a0, a0, a2
+; RV32-NEXT: ret
+;
+; RV64-LABEL: pnsrar_v4i8_16:
+; RV64: # %bb.0:
+; RV64-NEXT: li a1, -16
+; RV64-NEXT: psshar.hs a0, a0, a1
+; RV64-NEXT: pncvt.wb a0, a0
+; RV64-NEXT: ret
+ %res = call <4 x i8> @llvm.riscv.pnsrar.v4i8.v4i16(<4 x i16> %x, i32 16)
+ ret <4 x i8> %res
+}
>From 300bfae49a7d954cc645fe3a9ee707518a799fe5 Mon Sep 17 00:00:00 2001
From: Michael-Chen-NJU <2802328816 at qq.com>
Date: Sun, 4 Oct 2026 01:33:18 +0800
Subject: [PATCH 2/2] [RISCV] Address packed narrowing shift review feedback
---
clang/test/CodeGen/RISCV/rvp-intrinsics.c | 114 +++++++-------------
llvm/lib/Target/RISCV/RISCVISelLowering.cpp | 3 +-
2 files changed, 37 insertions(+), 80 deletions(-)
diff --git a/clang/test/CodeGen/RISCV/rvp-intrinsics.c b/clang/test/CodeGen/RISCV/rvp-intrinsics.c
index 8b0868312cdd1..41f94d48d8de2 100644
--- a/clang/test/CodeGen/RISCV/rvp-intrinsics.c
+++ b/clang/test/CodeGen/RISCV/rvp-intrinsics.c
@@ -5512,121 +5512,79 @@ int32x2_t test_pwsla_s_i32x2(int16x2_t rs1, unsigned shamt) {
return __riscv_pwsla_s_i32x2(rs1, shamt);
}
-// RV32-LABEL: define dso_local i32 @test_pnsrl_s_u8x4(
+// CHECK-LABEL: define dso_local i32 @test_pnsrl_s_u8x4(
// RV32-SAME: i64 noundef [[RS1_COERCE:%.*]], i32 noundef [[SHAMT:%.*]]) #[[ATTR0]] {
-// RV32-NEXT: [[ENTRY:.*:]]
-// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[RS1_COERCE]] to <4 x i16>
-// RV32-NEXT: [[TMP1:%.*]] = call <4 x i8> @llvm.riscv.pnsrl.v4i8.v4i16(<4 x i16> [[TMP0]], i32 [[SHAMT]])
-// RV32-NEXT: [[TMP2:%.*]] = bitcast <4 x i8> [[TMP1]] to i32
-// RV32-NEXT: ret i32 [[TMP2]]
-//
-// RV64-LABEL: define dso_local i32 @test_pnsrl_s_u8x4(
// RV64-SAME: i64 noundef [[RS1_COERCE:%.*]], i32 noundef signext [[SHAMT:%.*]]) #[[ATTR0]] {
-// RV64-NEXT: [[ENTRY:.*:]]
-// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[RS1_COERCE]] to <4 x i16>
-// RV64-NEXT: [[TMP1:%.*]] = call <4 x i8> @llvm.riscv.pnsrl.v4i8.v4i16(<4 x i16> [[TMP0]], i32 [[SHAMT]])
-// RV64-NEXT: [[TMP2:%.*]] = bitcast <4 x i8> [[TMP1]] to i32
-// RV64-NEXT: ret i32 [[TMP2]]
+// CHECK-NEXT: [[ENTRY:.*:]]
+// CHECK-NEXT: [[TMP0:%.*]] = bitcast i64 [[RS1_COERCE]] to <4 x i16>
+// CHECK-NEXT: [[TMP1:%.*]] = call <4 x i8> @llvm.riscv.pnsrl.v4i8.v4i16(<4 x i16> [[TMP0]], i32 [[SHAMT]])
+// CHECK-NEXT: [[TMP2:%.*]] = bitcast <4 x i8> [[TMP1]] to i32
+// CHECK-NEXT: ret i32 [[TMP2]]
//
uint8x4_t test_pnsrl_s_u8x4(uint16x4_t rs1, unsigned shamt) {
return __riscv_pnsrl_s_u8x4(rs1, shamt);
}
-// RV32-LABEL: define dso_local i32 @test_pnsrl_s_u16x2(
+// CHECK-LABEL: define dso_local i32 @test_pnsrl_s_u16x2(
// RV32-SAME: i64 noundef [[RS1_COERCE:%.*]], i32 noundef [[SHAMT:%.*]]) #[[ATTR0]] {
-// RV32-NEXT: [[ENTRY:.*:]]
-// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[RS1_COERCE]] to <2 x i32>
-// RV32-NEXT: [[TMP1:%.*]] = call <2 x i16> @llvm.riscv.pnsrl.v2i16.v2i32(<2 x i32> [[TMP0]], i32 [[SHAMT]])
-// RV32-NEXT: [[TMP2:%.*]] = bitcast <2 x i16> [[TMP1]] to i32
-// RV32-NEXT: ret i32 [[TMP2]]
-//
-// RV64-LABEL: define dso_local i32 @test_pnsrl_s_u16x2(
// RV64-SAME: i64 noundef [[RS1_COERCE:%.*]], i32 noundef signext [[SHAMT:%.*]]) #[[ATTR0]] {
-// RV64-NEXT: [[ENTRY:.*:]]
-// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[RS1_COERCE]] to <2 x i32>
-// RV64-NEXT: [[TMP1:%.*]] = call <2 x i16> @llvm.riscv.pnsrl.v2i16.v2i32(<2 x i32> [[TMP0]], i32 [[SHAMT]])
-// RV64-NEXT: [[TMP2:%.*]] = bitcast <2 x i16> [[TMP1]] to i32
-// RV64-NEXT: ret i32 [[TMP2]]
+// CHECK-NEXT: [[ENTRY:.*:]]
+// CHECK-NEXT: [[TMP0:%.*]] = bitcast i64 [[RS1_COERCE]] to <2 x i32>
+// CHECK-NEXT: [[TMP1:%.*]] = call <2 x i16> @llvm.riscv.pnsrl.v2i16.v2i32(<2 x i32> [[TMP0]], i32 [[SHAMT]])
+// CHECK-NEXT: [[TMP2:%.*]] = bitcast <2 x i16> [[TMP1]] to i32
+// CHECK-NEXT: ret i32 [[TMP2]]
//
uint16x2_t test_pnsrl_s_u16x2(uint32x2_t rs1, unsigned shamt) {
return __riscv_pnsrl_s_u16x2(rs1, shamt);
}
-// RV32-LABEL: define dso_local i32 @test_pnsra_s_i8x4(
+// CHECK-LABEL: define dso_local i32 @test_pnsra_s_i8x4(
// RV32-SAME: i64 noundef [[RS1_COERCE:%.*]], i32 noundef [[SHAMT:%.*]]) #[[ATTR0]] {
-// RV32-NEXT: [[ENTRY:.*:]]
-// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[RS1_COERCE]] to <4 x i16>
-// RV32-NEXT: [[TMP1:%.*]] = call <4 x i8> @llvm.riscv.pnsra.v4i8.v4i16(<4 x i16> [[TMP0]], i32 [[SHAMT]])
-// RV32-NEXT: [[TMP2:%.*]] = bitcast <4 x i8> [[TMP1]] to i32
-// RV32-NEXT: ret i32 [[TMP2]]
-//
-// RV64-LABEL: define dso_local i32 @test_pnsra_s_i8x4(
// RV64-SAME: i64 noundef [[RS1_COERCE:%.*]], i32 noundef signext [[SHAMT:%.*]]) #[[ATTR0]] {
-// RV64-NEXT: [[ENTRY:.*:]]
-// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[RS1_COERCE]] to <4 x i16>
-// RV64-NEXT: [[TMP1:%.*]] = call <4 x i8> @llvm.riscv.pnsra.v4i8.v4i16(<4 x i16> [[TMP0]], i32 [[SHAMT]])
-// RV64-NEXT: [[TMP2:%.*]] = bitcast <4 x i8> [[TMP1]] to i32
-// RV64-NEXT: ret i32 [[TMP2]]
+// CHECK-NEXT: [[ENTRY:.*:]]
+// CHECK-NEXT: [[TMP0:%.*]] = bitcast i64 [[RS1_COERCE]] to <4 x i16>
+// CHECK-NEXT: [[TMP1:%.*]] = call <4 x i8> @llvm.riscv.pnsra.v4i8.v4i16(<4 x i16> [[TMP0]], i32 [[SHAMT]])
+// CHECK-NEXT: [[TMP2:%.*]] = bitcast <4 x i8> [[TMP1]] to i32
+// CHECK-NEXT: ret i32 [[TMP2]]
//
int8x4_t test_pnsra_s_i8x4(int16x4_t rs1, unsigned shamt) {
return __riscv_pnsra_s_i8x4(rs1, shamt);
}
-// RV32-LABEL: define dso_local i32 @test_pnsra_s_i16x2(
+// CHECK-LABEL: define dso_local i32 @test_pnsra_s_i16x2(
// RV32-SAME: i64 noundef [[RS1_COERCE:%.*]], i32 noundef [[SHAMT:%.*]]) #[[ATTR0]] {
-// RV32-NEXT: [[ENTRY:.*:]]
-// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[RS1_COERCE]] to <2 x i32>
-// RV32-NEXT: [[TMP1:%.*]] = call <2 x i16> @llvm.riscv.pnsra.v2i16.v2i32(<2 x i32> [[TMP0]], i32 [[SHAMT]])
-// RV32-NEXT: [[TMP2:%.*]] = bitcast <2 x i16> [[TMP1]] to i32
-// RV32-NEXT: ret i32 [[TMP2]]
-//
-// RV64-LABEL: define dso_local i32 @test_pnsra_s_i16x2(
// RV64-SAME: i64 noundef [[RS1_COERCE:%.*]], i32 noundef signext [[SHAMT:%.*]]) #[[ATTR0]] {
-// RV64-NEXT: [[ENTRY:.*:]]
-// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[RS1_COERCE]] to <2 x i32>
-// RV64-NEXT: [[TMP1:%.*]] = call <2 x i16> @llvm.riscv.pnsra.v2i16.v2i32(<2 x i32> [[TMP0]], i32 [[SHAMT]])
-// RV64-NEXT: [[TMP2:%.*]] = bitcast <2 x i16> [[TMP1]] to i32
-// RV64-NEXT: ret i32 [[TMP2]]
+// CHECK-NEXT: [[ENTRY:.*:]]
+// CHECK-NEXT: [[TMP0:%.*]] = bitcast i64 [[RS1_COERCE]] to <2 x i32>
+// CHECK-NEXT: [[TMP1:%.*]] = call <2 x i16> @llvm.riscv.pnsra.v2i16.v2i32(<2 x i32> [[TMP0]], i32 [[SHAMT]])
+// CHECK-NEXT: [[TMP2:%.*]] = bitcast <2 x i16> [[TMP1]] to i32
+// CHECK-NEXT: ret i32 [[TMP2]]
//
int16x2_t test_pnsra_s_i16x2(int32x2_t rs1, unsigned shamt) {
return __riscv_pnsra_s_i16x2(rs1, shamt);
}
-// RV32-LABEL: define dso_local i32 @test_pnsrar_s_i8x4(
+// CHECK-LABEL: define dso_local i32 @test_pnsrar_s_i8x4(
// RV32-SAME: i64 noundef [[RS1_COERCE:%.*]], i32 noundef [[SHAMT:%.*]]) #[[ATTR0]] {
-// RV32-NEXT: [[ENTRY:.*:]]
-// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[RS1_COERCE]] to <4 x i16>
-// RV32-NEXT: [[TMP1:%.*]] = call <4 x i8> @llvm.riscv.pnsrar.v4i8.v4i16(<4 x i16> [[TMP0]], i32 [[SHAMT]])
-// RV32-NEXT: [[TMP2:%.*]] = bitcast <4 x i8> [[TMP1]] to i32
-// RV32-NEXT: ret i32 [[TMP2]]
-//
-// RV64-LABEL: define dso_local i32 @test_pnsrar_s_i8x4(
// RV64-SAME: i64 noundef [[RS1_COERCE:%.*]], i32 noundef signext [[SHAMT:%.*]]) #[[ATTR0]] {
-// RV64-NEXT: [[ENTRY:.*:]]
-// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[RS1_COERCE]] to <4 x i16>
-// RV64-NEXT: [[TMP1:%.*]] = call <4 x i8> @llvm.riscv.pnsrar.v4i8.v4i16(<4 x i16> [[TMP0]], i32 [[SHAMT]])
-// RV64-NEXT: [[TMP2:%.*]] = bitcast <4 x i8> [[TMP1]] to i32
-// RV64-NEXT: ret i32 [[TMP2]]
+// CHECK-NEXT: [[ENTRY:.*:]]
+// CHECK-NEXT: [[TMP0:%.*]] = bitcast i64 [[RS1_COERCE]] to <4 x i16>
+// CHECK-NEXT: [[TMP1:%.*]] = call <4 x i8> @llvm.riscv.pnsrar.v4i8.v4i16(<4 x i16> [[TMP0]], i32 [[SHAMT]])
+// CHECK-NEXT: [[TMP2:%.*]] = bitcast <4 x i8> [[TMP1]] to i32
+// CHECK-NEXT: ret i32 [[TMP2]]
//
int8x4_t test_pnsrar_s_i8x4(int16x4_t rs1, unsigned shamt) {
return __riscv_pnsrar_s_i8x4(rs1, shamt);
}
-// RV32-LABEL: define dso_local i32 @test_pnsrar_s_i16x2(
+// CHECK-LABEL: define dso_local i32 @test_pnsrar_s_i16x2(
// RV32-SAME: i64 noundef [[RS1_COERCE:%.*]], i32 noundef [[SHAMT:%.*]]) #[[ATTR0]] {
-// RV32-NEXT: [[ENTRY:.*:]]
-// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[RS1_COERCE]] to <2 x i32>
-// RV32-NEXT: [[TMP1:%.*]] = call <2 x i16> @llvm.riscv.pnsrar.v2i16.v2i32(<2 x i32> [[TMP0]], i32 [[SHAMT]])
-// RV32-NEXT: [[TMP2:%.*]] = bitcast <2 x i16> [[TMP1]] to i32
-// RV32-NEXT: ret i32 [[TMP2]]
-//
-// RV64-LABEL: define dso_local i32 @test_pnsrar_s_i16x2(
// RV64-SAME: i64 noundef [[RS1_COERCE:%.*]], i32 noundef signext [[SHAMT:%.*]]) #[[ATTR0]] {
-// RV64-NEXT: [[ENTRY:.*:]]
-// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[RS1_COERCE]] to <2 x i32>
-// RV64-NEXT: [[TMP1:%.*]] = call <2 x i16> @llvm.riscv.pnsrar.v2i16.v2i32(<2 x i32> [[TMP0]], i32 [[SHAMT]])
-// RV64-NEXT: [[TMP2:%.*]] = bitcast <2 x i16> [[TMP1]] to i32
-// RV64-NEXT: ret i32 [[TMP2]]
+// CHECK-NEXT: [[ENTRY:.*:]]
+// CHECK-NEXT: [[TMP0:%.*]] = bitcast i64 [[RS1_COERCE]] to <2 x i32>
+// CHECK-NEXT: [[TMP1:%.*]] = call <2 x i16> @llvm.riscv.pnsrar.v2i16.v2i32(<2 x i32> [[TMP0]], i32 [[SHAMT]])
+// CHECK-NEXT: [[TMP2:%.*]] = bitcast <2 x i16> [[TMP1]] to i32
+// CHECK-NEXT: ret i32 [[TMP2]]
//
int16x2_t test_pnsrar_s_i16x2(int32x2_t rs1, unsigned shamt) {
return __riscv_pnsrar_s_i16x2(rs1, shamt);
diff --git a/llvm/lib/Target/RISCV/RISCVISelLowering.cpp b/llvm/lib/Target/RISCV/RISCVISelLowering.cpp
index f416bd5b4f90b..56ef620e9b1d0 100644
--- a/llvm/lib/Target/RISCV/RISCVISelLowering.cpp
+++ b/llvm/lib/Target/RISCV/RISCVISelLowering.cpp
@@ -18118,8 +18118,7 @@ void RISCVTargetLowering::ReplaceNodeResults(SDNode *N,
Opc = RISCVISD::PSSHAR;
ShAmt = DAG.getNode(ISD::AND, DL, XLenVT, ShAmt,
DAG.getConstant(31, DL, XLenVT));
- ShAmt = DAG.getNode(ISD::SUB, DL, XLenVT,
- DAG.getConstant(0, DL, XLenVT), ShAmt);
+ ShAmt = DAG.getNegative(ShAmt, DL, XLenVT);
break;
}
More information about the cfe-commits
mailing list