[clang] [llvm] [Clang][RISCV] Add packed narrowing shift intrinsics (PR #228726)
via cfe-commits
cfe-commits at lists.llvm.org
Sat Oct 3 08:51:34 PDT 2026
llvmorg-github-actions[bot] wrote:
<!--LLVM PR SUMMARY COMMENT-->
@llvm/pr-subscribers-backend-risc-v
Author: 陈子昂 (Michael-Chen-NJU)
<details>
<summary>Changes</summary>
This patch adds the Packed Narrowing Shift intrinsics:
- `__riscv_pnsrl_s_u8x4`
- `__riscv_pnsrl_s_u16x2`
- `__riscv_pnsra_s_i8x4`
- `__riscv_pnsra_s_i16x2`
- `__riscv_pnsrar_s_i8x4`
- `__riscv_pnsrar_s_i16x2`
---
Patch is 27.22 KiB, truncated to 20.00 KiB below, full version: https://github.com/llvm/llvm-project/pull/228726.diff
9 Files Affected:
- (modified) clang/include/clang/Basic/BuiltinsRISCV.td (+8)
- (modified) clang/lib/CodeGen/TargetBuiltins/RISCV.cpp (+17)
- (modified) clang/lib/Headers/riscv_packed_simd.h (+14)
- (modified) clang/test/CodeGen/RISCV/rvp-intrinsics.c (+120)
- (modified) cross-project-tests/intrinsic-header-tests/riscv_packed_simd.c (+74)
- (modified) llvm/include/llvm/IR/IntrinsicsRISCV.td (+9)
- (modified) llvm/lib/Target/RISCV/RISCVISelLowering.cpp (+69)
- (modified) llvm/lib/Target/RISCV/RISCVInstrInfoP.td (+28-1)
- (added) llvm/test/CodeGen/RISCV/rvp-narrowing-shift.ll (+219)
``````````diff
diff --git a/clang/include/clang/Basic/BuiltinsRISCV.td b/clang/include/clang/Basic/BuiltinsRISCV.td
index 3a3e85c25b664..c647c25c6ee4e 100644
--- a/clang/include/clang/Basic/BuiltinsRISCV.td
+++ b/clang/include/clang/Basic/BuiltinsRISCV.td
@@ -533,6 +533,14 @@ def pwsll_s_u32x2 : RISCVBuiltin<"_Vector<2, unsigned int>(_Vector<2, unsigned s
def pwsla_s_i16x4 : RISCVBuiltin<"_Vector<4, short>(_Vector<4, signed char>, unsigned int)">;
def pwsla_s_i32x2 : RISCVBuiltin<"_Vector<2, int>(_Vector<2, short>, unsigned int)">;
+// Packed Narrowing Shifts
+def pnsrl_s_u8x4 : RISCVBuiltin<"_Vector<4, unsigned char>(_Vector<4, unsigned short>, unsigned int)">;
+def pnsrl_s_u16x2 : RISCVBuiltin<"_Vector<2, unsigned short>(_Vector<2, unsigned int>, unsigned int)">;
+def pnsra_s_i8x4 : RISCVBuiltin<"_Vector<4, signed char>(_Vector<4, short>, unsigned int)">;
+def pnsra_s_i16x2 : RISCVBuiltin<"_Vector<2, short>(_Vector<2, int>, unsigned int)">;
+def pnsrar_s_i8x4 : RISCVBuiltin<"_Vector<4, signed char>(_Vector<4, short>, unsigned int)">;
+def pnsrar_s_i16x2 : RISCVBuiltin<"_Vector<2, short>(_Vector<2, int>, unsigned int)">;
+
// Packed Narrowing Clip Pair (32-bit)
def pnclipp_i8x4 : RISCVBuiltin<"_Vector<4, signed char>(_Vector<2, short>, _Vector<2, short>)">;
diff --git a/clang/lib/CodeGen/TargetBuiltins/RISCV.cpp b/clang/lib/CodeGen/TargetBuiltins/RISCV.cpp
index 74f85712e4828..8158543e232c3 100644
--- a/clang/lib/CodeGen/TargetBuiltins/RISCV.cpp
+++ b/clang/lib/CodeGen/TargetBuiltins/RISCV.cpp
@@ -1211,6 +1211,23 @@ Value *CodeGenFunction::EmitRISCVBuiltinExpr(unsigned BuiltinID,
IntrinsicTypes = {ResultType, Ops[0]->getType()};
break;
+ // Packed Narrowing Shifts
+ case RISCV::BI__builtin_riscv_pnsrl_s_u8x4:
+ case RISCV::BI__builtin_riscv_pnsrl_s_u16x2:
+ ID = Intrinsic::riscv_pnsrl;
+ IntrinsicTypes = {ResultType, Ops[0]->getType()};
+ break;
+ case RISCV::BI__builtin_riscv_pnsra_s_i8x4:
+ case RISCV::BI__builtin_riscv_pnsra_s_i16x2:
+ ID = Intrinsic::riscv_pnsra;
+ IntrinsicTypes = {ResultType, Ops[0]->getType()};
+ break;
+ case RISCV::BI__builtin_riscv_pnsrar_s_i8x4:
+ case RISCV::BI__builtin_riscv_pnsrar_s_i16x2:
+ ID = Intrinsic::riscv_pnsrar;
+ IntrinsicTypes = {ResultType, Ops[0]->getType()};
+ break;
+
// Packed Averaging Addition and Subtraction
case RISCV::BI__builtin_riscv_paadd_i8x4:
case RISCV::BI__builtin_riscv_paadd_i16x2:
diff --git a/clang/lib/Headers/riscv_packed_simd.h b/clang/lib/Headers/riscv_packed_simd.h
index 9ae0c45d60674..a155278342d23 100644
--- a/clang/lib/Headers/riscv_packed_simd.h
+++ b/clang/lib/Headers/riscv_packed_simd.h
@@ -801,6 +801,20 @@ __packed_binary_builtin_mixed(pwsla_s_i16x4, int16x4_t, int8x4_t, unsigned,
__packed_binary_builtin_mixed(pwsla_s_i32x2, int32x2_t, int16x2_t, unsigned,
__builtin_riscv_pwsla_s_i32x2)
+/* Packed Narrowing Shift */
+__packed_binary_builtin_mixed(pnsrl_s_u8x4, uint8x4_t, uint16x4_t, unsigned,
+ __builtin_riscv_pnsrl_s_u8x4)
+__packed_binary_builtin_mixed(pnsrl_s_u16x2, uint16x2_t, uint32x2_t, unsigned,
+ __builtin_riscv_pnsrl_s_u16x2)
+__packed_binary_builtin_mixed(pnsra_s_i8x4, int8x4_t, int16x4_t, unsigned,
+ __builtin_riscv_pnsra_s_i8x4)
+__packed_binary_builtin_mixed(pnsra_s_i16x2, int16x2_t, int32x2_t, unsigned,
+ __builtin_riscv_pnsra_s_i16x2)
+__packed_binary_builtin_mixed(pnsrar_s_i8x4, int8x4_t, int16x4_t, unsigned,
+ __builtin_riscv_pnsrar_s_i8x4)
+__packed_binary_builtin_mixed(pnsrar_s_i16x2, int16x2_t, int32x2_t, unsigned,
+ __builtin_riscv_pnsrar_s_i16x2)
+
/* Packed Widening Addition and Subtraction */
__packed_widen_binary_op(pwadd_i16x4, int16x4_t, int8x4_t, +)
__packed_widen_binary_op(pwadd_i32x2, int32x2_t, int16x2_t, +)
diff --git a/clang/test/CodeGen/RISCV/rvp-intrinsics.c b/clang/test/CodeGen/RISCV/rvp-intrinsics.c
index d9ad53c373da2..8b0868312cdd1 100644
--- a/clang/test/CodeGen/RISCV/rvp-intrinsics.c
+++ b/clang/test/CodeGen/RISCV/rvp-intrinsics.c
@@ -5512,6 +5512,126 @@ int32x2_t test_pwsla_s_i32x2(int16x2_t rs1, unsigned shamt) {
return __riscv_pwsla_s_i32x2(rs1, shamt);
}
+// RV32-LABEL: define dso_local i32 @test_pnsrl_s_u8x4(
+// RV32-SAME: i64 noundef [[RS1_COERCE:%.*]], i32 noundef [[SHAMT:%.*]]) #[[ATTR0]] {
+// RV32-NEXT: [[ENTRY:.*:]]
+// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[RS1_COERCE]] to <4 x i16>
+// RV32-NEXT: [[TMP1:%.*]] = call <4 x i8> @llvm.riscv.pnsrl.v4i8.v4i16(<4 x i16> [[TMP0]], i32 [[SHAMT]])
+// RV32-NEXT: [[TMP2:%.*]] = bitcast <4 x i8> [[TMP1]] to i32
+// RV32-NEXT: ret i32 [[TMP2]]
+//
+// RV64-LABEL: define dso_local i32 @test_pnsrl_s_u8x4(
+// RV64-SAME: i64 noundef [[RS1_COERCE:%.*]], i32 noundef signext [[SHAMT:%.*]]) #[[ATTR0]] {
+// RV64-NEXT: [[ENTRY:.*:]]
+// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[RS1_COERCE]] to <4 x i16>
+// RV64-NEXT: [[TMP1:%.*]] = call <4 x i8> @llvm.riscv.pnsrl.v4i8.v4i16(<4 x i16> [[TMP0]], i32 [[SHAMT]])
+// RV64-NEXT: [[TMP2:%.*]] = bitcast <4 x i8> [[TMP1]] to i32
+// RV64-NEXT: ret i32 [[TMP2]]
+//
+uint8x4_t test_pnsrl_s_u8x4(uint16x4_t rs1, unsigned shamt) {
+ return __riscv_pnsrl_s_u8x4(rs1, shamt);
+}
+
+// RV32-LABEL: define dso_local i32 @test_pnsrl_s_u16x2(
+// RV32-SAME: i64 noundef [[RS1_COERCE:%.*]], i32 noundef [[SHAMT:%.*]]) #[[ATTR0]] {
+// RV32-NEXT: [[ENTRY:.*:]]
+// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[RS1_COERCE]] to <2 x i32>
+// RV32-NEXT: [[TMP1:%.*]] = call <2 x i16> @llvm.riscv.pnsrl.v2i16.v2i32(<2 x i32> [[TMP0]], i32 [[SHAMT]])
+// RV32-NEXT: [[TMP2:%.*]] = bitcast <2 x i16> [[TMP1]] to i32
+// RV32-NEXT: ret i32 [[TMP2]]
+//
+// RV64-LABEL: define dso_local i32 @test_pnsrl_s_u16x2(
+// RV64-SAME: i64 noundef [[RS1_COERCE:%.*]], i32 noundef signext [[SHAMT:%.*]]) #[[ATTR0]] {
+// RV64-NEXT: [[ENTRY:.*:]]
+// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[RS1_COERCE]] to <2 x i32>
+// RV64-NEXT: [[TMP1:%.*]] = call <2 x i16> @llvm.riscv.pnsrl.v2i16.v2i32(<2 x i32> [[TMP0]], i32 [[SHAMT]])
+// RV64-NEXT: [[TMP2:%.*]] = bitcast <2 x i16> [[TMP1]] to i32
+// RV64-NEXT: ret i32 [[TMP2]]
+//
+uint16x2_t test_pnsrl_s_u16x2(uint32x2_t rs1, unsigned shamt) {
+ return __riscv_pnsrl_s_u16x2(rs1, shamt);
+}
+
+// RV32-LABEL: define dso_local i32 @test_pnsra_s_i8x4(
+// RV32-SAME: i64 noundef [[RS1_COERCE:%.*]], i32 noundef [[SHAMT:%.*]]) #[[ATTR0]] {
+// RV32-NEXT: [[ENTRY:.*:]]
+// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[RS1_COERCE]] to <4 x i16>
+// RV32-NEXT: [[TMP1:%.*]] = call <4 x i8> @llvm.riscv.pnsra.v4i8.v4i16(<4 x i16> [[TMP0]], i32 [[SHAMT]])
+// RV32-NEXT: [[TMP2:%.*]] = bitcast <4 x i8> [[TMP1]] to i32
+// RV32-NEXT: ret i32 [[TMP2]]
+//
+// RV64-LABEL: define dso_local i32 @test_pnsra_s_i8x4(
+// RV64-SAME: i64 noundef [[RS1_COERCE:%.*]], i32 noundef signext [[SHAMT:%.*]]) #[[ATTR0]] {
+// RV64-NEXT: [[ENTRY:.*:]]
+// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[RS1_COERCE]] to <4 x i16>
+// RV64-NEXT: [[TMP1:%.*]] = call <4 x i8> @llvm.riscv.pnsra.v4i8.v4i16(<4 x i16> [[TMP0]], i32 [[SHAMT]])
+// RV64-NEXT: [[TMP2:%.*]] = bitcast <4 x i8> [[TMP1]] to i32
+// RV64-NEXT: ret i32 [[TMP2]]
+//
+int8x4_t test_pnsra_s_i8x4(int16x4_t rs1, unsigned shamt) {
+ return __riscv_pnsra_s_i8x4(rs1, shamt);
+}
+
+// RV32-LABEL: define dso_local i32 @test_pnsra_s_i16x2(
+// RV32-SAME: i64 noundef [[RS1_COERCE:%.*]], i32 noundef [[SHAMT:%.*]]) #[[ATTR0]] {
+// RV32-NEXT: [[ENTRY:.*:]]
+// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[RS1_COERCE]] to <2 x i32>
+// RV32-NEXT: [[TMP1:%.*]] = call <2 x i16> @llvm.riscv.pnsra.v2i16.v2i32(<2 x i32> [[TMP0]], i32 [[SHAMT]])
+// RV32-NEXT: [[TMP2:%.*]] = bitcast <2 x i16> [[TMP1]] to i32
+// RV32-NEXT: ret i32 [[TMP2]]
+//
+// RV64-LABEL: define dso_local i32 @test_pnsra_s_i16x2(
+// RV64-SAME: i64 noundef [[RS1_COERCE:%.*]], i32 noundef signext [[SHAMT:%.*]]) #[[ATTR0]] {
+// RV64-NEXT: [[ENTRY:.*:]]
+// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[RS1_COERCE]] to <2 x i32>
+// RV64-NEXT: [[TMP1:%.*]] = call <2 x i16> @llvm.riscv.pnsra.v2i16.v2i32(<2 x i32> [[TMP0]], i32 [[SHAMT]])
+// RV64-NEXT: [[TMP2:%.*]] = bitcast <2 x i16> [[TMP1]] to i32
+// RV64-NEXT: ret i32 [[TMP2]]
+//
+int16x2_t test_pnsra_s_i16x2(int32x2_t rs1, unsigned shamt) {
+ return __riscv_pnsra_s_i16x2(rs1, shamt);
+}
+
+// RV32-LABEL: define dso_local i32 @test_pnsrar_s_i8x4(
+// RV32-SAME: i64 noundef [[RS1_COERCE:%.*]], i32 noundef [[SHAMT:%.*]]) #[[ATTR0]] {
+// RV32-NEXT: [[ENTRY:.*:]]
+// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[RS1_COERCE]] to <4 x i16>
+// RV32-NEXT: [[TMP1:%.*]] = call <4 x i8> @llvm.riscv.pnsrar.v4i8.v4i16(<4 x i16> [[TMP0]], i32 [[SHAMT]])
+// RV32-NEXT: [[TMP2:%.*]] = bitcast <4 x i8> [[TMP1]] to i32
+// RV32-NEXT: ret i32 [[TMP2]]
+//
+// RV64-LABEL: define dso_local i32 @test_pnsrar_s_i8x4(
+// RV64-SAME: i64 noundef [[RS1_COERCE:%.*]], i32 noundef signext [[SHAMT:%.*]]) #[[ATTR0]] {
+// RV64-NEXT: [[ENTRY:.*:]]
+// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[RS1_COERCE]] to <4 x i16>
+// RV64-NEXT: [[TMP1:%.*]] = call <4 x i8> @llvm.riscv.pnsrar.v4i8.v4i16(<4 x i16> [[TMP0]], i32 [[SHAMT]])
+// RV64-NEXT: [[TMP2:%.*]] = bitcast <4 x i8> [[TMP1]] to i32
+// RV64-NEXT: ret i32 [[TMP2]]
+//
+int8x4_t test_pnsrar_s_i8x4(int16x4_t rs1, unsigned shamt) {
+ return __riscv_pnsrar_s_i8x4(rs1, shamt);
+}
+
+// RV32-LABEL: define dso_local i32 @test_pnsrar_s_i16x2(
+// RV32-SAME: i64 noundef [[RS1_COERCE:%.*]], i32 noundef [[SHAMT:%.*]]) #[[ATTR0]] {
+// RV32-NEXT: [[ENTRY:.*:]]
+// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[RS1_COERCE]] to <2 x i32>
+// RV32-NEXT: [[TMP1:%.*]] = call <2 x i16> @llvm.riscv.pnsrar.v2i16.v2i32(<2 x i32> [[TMP0]], i32 [[SHAMT]])
+// RV32-NEXT: [[TMP2:%.*]] = bitcast <2 x i16> [[TMP1]] to i32
+// RV32-NEXT: ret i32 [[TMP2]]
+//
+// RV64-LABEL: define dso_local i32 @test_pnsrar_s_i16x2(
+// RV64-SAME: i64 noundef [[RS1_COERCE:%.*]], i32 noundef signext [[SHAMT:%.*]]) #[[ATTR0]] {
+// RV64-NEXT: [[ENTRY:.*:]]
+// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[RS1_COERCE]] to <2 x i32>
+// RV64-NEXT: [[TMP1:%.*]] = call <2 x i16> @llvm.riscv.pnsrar.v2i16.v2i32(<2 x i32> [[TMP0]], i32 [[SHAMT]])
+// RV64-NEXT: [[TMP2:%.*]] = bitcast <2 x i16> [[TMP1]] to i32
+// RV64-NEXT: ret i32 [[TMP2]]
+//
+int16x2_t test_pnsrar_s_i16x2(int32x2_t rs1, unsigned shamt) {
+ return __riscv_pnsrar_s_i16x2(rs1, shamt);
+}
+
// CHECK-LABEL: define dso_local i64 @test_pwadd_i16x4(
// CHECK-SAME: i32 noundef [[RS1_COERCE:%.*]], i32 noundef [[RS2_COERCE:%.*]]) #[[ATTR0]] {
// CHECK-NEXT: [[ENTRY:.*:]]
diff --git a/cross-project-tests/intrinsic-header-tests/riscv_packed_simd.c b/cross-project-tests/intrinsic-header-tests/riscv_packed_simd.c
index 6d92dd5990261..9b462245dc786 100644
--- a/cross-project-tests/intrinsic-header-tests/riscv_packed_simd.c
+++ b/cross-project-tests/intrinsic-header-tests/riscv_packed_simd.c
@@ -2439,6 +2439,80 @@ uint32x2_t test_pwsll_s_u32x2_low5(uint16x2_t rs1) {
return __riscv_pwsll_s_u32x2(rs1, 63);
}
+// CHECK-LABEL: test_pnsrl_s_u8x4:
+// RV32: pnsrl.bs
+// RV64: psrl.hs
+// RV64: pncvt.wb
+uint8x4_t test_pnsrl_s_u8x4(uint16x4_t rs1, unsigned shamt) {
+ return __riscv_pnsrl_s_u8x4(rs1, shamt);
+}
+
+// CHECK-LABEL: test_pnsrl_s_u16x2:
+// RV32: pnsrl.hs
+// RV64: psrl.ws
+// RV64: pncvt.wh
+uint16x2_t test_pnsrl_s_u16x2(uint32x2_t rs1, unsigned shamt) {
+ return __riscv_pnsrl_s_u16x2(rs1, shamt);
+}
+
+// CHECK-LABEL: test_pnsra_s_i8x4:
+// RV32: pnsra.bs
+// RV64: psra.hs
+// RV64: pncvt.wb
+int8x4_t test_pnsra_s_i8x4(int16x4_t rs1, unsigned shamt) {
+ return __riscv_pnsra_s_i8x4(rs1, shamt);
+}
+
+// CHECK-LABEL: test_pnsra_s_i16x2:
+// RV32: pnsra.hs
+// RV64: psra.ws
+// RV64: pncvt.wh
+int16x2_t test_pnsra_s_i16x2(int32x2_t rs1, unsigned shamt) {
+ return __riscv_pnsra_s_i16x2(rs1, shamt);
+}
+
+// CHECK-LABEL: test_pnsrar_s_i8x4:
+// RV32: pnsrar.bs
+// RV64: psshar.hs
+// RV64: pncvt.wb
+int8x4_t test_pnsrar_s_i8x4(int16x4_t rs1, unsigned shamt) {
+ return __riscv_pnsrar_s_i8x4(rs1, shamt);
+}
+
+// CHECK-LABEL: test_pnsrar_s_i16x2:
+// RV32: pnsrar.hs
+// RV64: psshar.ws
+// RV64: pncvt.wh
+int16x2_t test_pnsrar_s_i16x2(int32x2_t rs1, unsigned shamt) {
+ return __riscv_pnsrar_s_i16x2(rs1, shamt);
+}
+
+// Counts that exceed the result width still use the low five bits of shamt.
+// CHECK-LABEL: test_pnsrl_s_u8x4_16:
+// RV32: li{{[[:space:]]}}a2, 16
+// RV32-NEXT: pnsrl.bs{{[[:space:]]}}a0, a0, a2
+// RV64: li{{[[:space:]]}}a1, 16
+// RV64: psrl.hs{{[[:space:]]}}a0, a0, a1
+uint8x4_t test_pnsrl_s_u8x4_16(uint16x4_t rs1) {
+ return __riscv_pnsrl_s_u8x4(rs1, 16);
+}
+
+// CHECK-LABEL: test_pnsrl_s_u16x2_31:
+// RV32: pnsrli.h{{[[:space:]]}}a0, a0, 31
+// RV64: psrli.w{{[[:space:]]}}a0, a0, 31
+uint16x2_t test_pnsrl_s_u16x2_31(uint32x2_t rs1) {
+ return __riscv_pnsrl_s_u16x2(rs1, 31);
+}
+
+// CHECK-LABEL: test_pnsrar_s_i8x4_16:
+// RV32: li{{[[:space:]]}}a2, 16
+// RV32-NEXT: pnsrar.bs{{[[:space:]]}}a0, a0, a2
+// RV64: li{{[[:space:]]}}a1, -16
+// RV64-NEXT: psshar.hs{{[[:space:]]}}a0, a0, a1
+int8x4_t test_pnsrar_s_i8x4_16(int16x4_t rs1) {
+ return __riscv_pnsrar_s_i8x4(rs1, 16);
+}
+
// CHECK-LABEL: test_pwadd_i16x4:
// RV32: pwadd.b
// RV64: zip8p
diff --git a/llvm/include/llvm/IR/IntrinsicsRISCV.td b/llvm/include/llvm/IR/IntrinsicsRISCV.td
index 3ff0b10bb6f53..390e35b8e5d8c 100644
--- a/llvm/include/llvm/IR/IntrinsicsRISCV.td
+++ b/llvm/include/llvm/IR/IntrinsicsRISCV.td
@@ -2089,6 +2089,15 @@ class RVPBinaryIntrinsic
def int_riscv_pwsll : RVPWideningShiftIntrinsic;
def int_riscv_pwsla : RVPWideningShiftIntrinsic;
+ // Packed Narrowing Shifts.
+ class RVPNarrowingShiftIntrinsic
+ : DefaultAttrsIntrinsic<[llvm_anyvector_ty],
+ [llvm_anyvector_ty, llvm_i32_ty],
+ [IntrNoMem, IntrSpeculatable]>;
+ def int_riscv_pnsrl : RVPNarrowingShiftIntrinsic;
+ def int_riscv_pnsra : RVPNarrowingShiftIntrinsic;
+ def int_riscv_pnsrar : RVPNarrowingShiftIntrinsic;
+
// Packed Saturation.
class RVPSaturationIntrinsic
: DefaultAttrsIntrinsic<[llvm_anyvector_ty],
diff --git a/llvm/lib/Target/RISCV/RISCVISelLowering.cpp b/llvm/lib/Target/RISCV/RISCVISelLowering.cpp
index 3892e62f715cf..f416bd5b4f90b 100644
--- a/llvm/lib/Target/RISCV/RISCVISelLowering.cpp
+++ b/llvm/lib/Target/RISCV/RISCVISelLowering.cpp
@@ -12394,6 +12394,19 @@ static unsigned getRVPShiftOpcode(Intrinsic::ID IntNo) {
}
}
+static unsigned getRVPNarrowingShiftOpcode(Intrinsic::ID IntNo) {
+ switch (IntNo) {
+ default:
+ llvm_unreachable("Unexpected RISC-V packed narrowing shift intrinsic");
+ case Intrinsic::riscv_pnsrl:
+ return RISCVISD::PNSRL;
+ case Intrinsic::riscv_pnsra:
+ return RISCVISD::PNSRA;
+ case Intrinsic::riscv_pnsrar:
+ return RISCVISD::PNSRAR;
+ }
+}
+
static SDValue lowerPZExt(SDValue Src, const SDLoc &DL, SelectionDAG &DAG,
const RISCVSubtarget &Subtarget) {
MVT VT = Src.getSimpleValueType();
@@ -13275,6 +13288,19 @@ SDValue RISCVTargetLowering::LowerINTRINSIC_WO_CHAIN(SDValue Op,
SDValue Wide = DAG.getNode(ExtOpc, DL, VT, Src);
return DAG.getNode(RISCVISD::PSLL, DL, VT, Wide, ShAmt);
}
+ case Intrinsic::riscv_pnsrl:
+ case Intrinsic::riscv_pnsra:
+ case Intrinsic::riscv_pnsrar: {
+ MVT VT = Op.getSimpleValueType();
+ MVT SrcVT = Op.getOperand(1).getSimpleValueType();
+ if (!((VT == MVT::v4i8 && SrcVT == MVT::v4i16) ||
+ (VT == MVT::v2i16 && SrcVT == MVT::v2i32)))
+ reportFatalUsageError("unsupported packed narrowing shift intrinsic");
+
+ SDValue ShAmt = DAG.getNode(ISD::ANY_EXTEND, DL, XLenVT, Op.getOperand(2));
+ return DAG.getNode(getRVPNarrowingShiftOpcode(IntNo), DL, VT,
+ Op.getOperand(1), ShAmt);
+ }
case Intrinsic::riscv_psati:
case Intrinsic::riscv_pusati: {
bool IsSigned = IntNo == Intrinsic::riscv_psati;
@@ -18062,6 +18088,49 @@ void RISCVTargetLowering::ReplaceNodeResults(SDNode *N,
Results.push_back(DAG.getExtractSubvector(DL, VT, Res, 0));
return;
}
+ case Intrinsic::riscv_pnsrl:
+ case Intrinsic::riscv_pnsra:
+ case Intrinsic::riscv_pnsrar: {
+ MVT VT = N->getSimpleValueType(0);
+ if (!Subtarget.is64Bit() || (VT != MVT::v4i8 && VT != MVT::v2i16))
+ return;
+
+ SDValue Src = N->getOperand(1);
+ MVT SrcVT = Src.getSimpleValueType();
+ if (!((VT == MVT::v4i8 && SrcVT == MVT::v4i16) ||
+ (VT == MVT::v2i16 && SrcVT == MVT::v2i32)))
+ reportFatalUsageError("unsupported packed narrowing shift intrinsic");
+
+ MVT XLenVT = Subtarget.getXLenVT();
+ SDValue ShAmt =
+ DAG.getNode(ISD::ANY_EXTEND, DL, XLenVT, N->getOperand(2));
+ unsigned Opc;
+ switch (IntNo) {
+ default:
+ llvm_unreachable("Unexpected packed narrowing shift intrinsic");
+ case Intrinsic::riscv_pnsrl:
+ Opc = RISCVISD::PSRL;
+ break;
+ case Intrinsic::riscv_pnsra:
+ Opc = RISCVISD::PSRA;
+ break;
+ case Intrinsic::riscv_pnsrar:
+ Opc = RISCVISD::PSSHAR;
+ ShAmt = DAG.getNode(ISD::AND, DL, XLenVT, ShAmt,
+ DAG.getConstant(31, DL, XLenVT));
+ ShAmt = DAG.getNode(ISD::SUB, DL, XLenVT,
+ DAG.getConstant(0, DL, XLenVT), ShAmt);
+ break;
+ }
+
+ SDValue Shifted = DAG.getNode(Opc, DL, SrcVT, Src, ShAmt);
+ MVT UnzipVT = VT == MVT::v4i8 ? MVT::v8i8 : MVT::v4i16;
+ Shifted = DAG.getBitcast(UnzipVT, Shifted);
+ SDValue Unzip = DAG.getNode(RISCVISD::PUNZIPE, DL, UnzipVT, Shifted,
+ DAG.getUNDEF(UnzipVT));
+ Results.push_back(DAG.getExtractSubvector(DL, VT, Unzip, 0));
+ return;
+ }
case Intrinsic::riscv_predsum:
case Intrinsic::riscv_predsumu: {
bool IsSigned = IntNo == Intrinsic::riscv_predsum;
diff --git a/llvm/lib/Target/RISCV/RISCVInstrInfoP.td b/llvm/lib/Target/RISCV/RISCVInstrInfoP.td
index ee5b17749cb73..4a3b0be8949f6 100644
--- a/llvm/lib/Target/RISCV/RISCVInstrInfoP.td
+++ b/llvm/lib/Target/RISCV/RISCVInstrInfoP.td
@@ -2089,7 +2089,9 @@ def riscv_psshlr : RVSDNode<"PSSHLR", SDT_RISCVPackedShift>;
def SDT_RISCVPackedNarrowingShift
: SDTypeProfile<1, 2, [SDTCisVec<0>, SDTCisVec<1>,
SDTCisVT<2, XLenVT>]>;
-def riscv_pnsrl : RVSDNode<"PNSRL", SDT_RISCVPackedNarrowingShift>;
+def riscv_pnsrl : RVSDNode<"PNSRL", SDT_RISCVPackedNarrowingShift>;
+def riscv_pnsra : RVSDNode<"PNSRA", SDT_RISCVPackedNarrowingShift>;
+def riscv_pnsrar : RVSDNode<"PNSRAR", SDT_RISCVPackedNarrowingShift>;
def riscv_pnclip : RVSDNode<"PNCLIP", SDT_RISCVPackedNarrowingShift>;
def riscv_pnclipu : RVSDNode<"PNCLIPU", SDT_RISCVPackedNarrowingShift>;
@@ -2821,6 +2823,31 @@ let append Predicates = [IsRV32] in {
(PNSRLI_B GPRPair:$rs1, uimm4:$imm)>;
def : Pat<(v2i16 (riscv_pnsrl (v4i16 GPRPair:$rs1), uimm5:$imm)),
(PNSRLI_H GPRPair:$rs1, uimm5:$imm)>;
+ def : Pat<(v4i8 (riscv_pnsrl (v4i16 GPRPair:$rs1), uimm4:$imm)),
+ (PNSRLI_B GPRPair:$rs1, uimm4:$imm)>;
+ def : Pat<(v2i16 (riscv_pnsrl (v2i32 GPRPair:$rs1), uimm5:$imm)),
+ (PNSRLI_H GPRPair:$rs1, uimm5:$imm)>;
+ def : Pat<(v4i8 (riscv_pnsra (v4i16 GPRPair:$rs1), uimm4:$imm)),
+ (PNSRAI_B GPRPair:$rs1, uimm4:$imm)>;
+ def : Pat<(v2i16 (riscv_pnsra (v2i32 GPRPair:$rs1), uimm5:$imm)),
+ (PNSRAI_H GPRPair:$rs1, uimm5:$imm)>;
+ def : Pat<(v4i8 (riscv_pnsrar (v4i16 GPRPair:$rs1), uimm4:$imm)),
+ (PNSRARI_B GPRPair:$rs1, uimm4:$imm)>;
+ def : Pat<(v2i16 (riscv_pnsrar (v2i32 GPRPair:$rs1), uimm5:$imm)),
+ (PNSRARI_H GPRPair:$rs1, uimm5:$imm)>;
+
+ def : Pat<(v4i8 (riscv_pnsrl (v4i16 GPRPair:$rs1), shiftMask32:$rs2)),
+ (PNSRL_BS GPRPair:$rs1, shiftMask32:$rs2)>;
+ def : Pat<(v2i16 (riscv_pnsrl (v2i32 GPRPair:$rs1), shiftMask32:$rs2)),
+ (PNSRL_HS GPRPair:$rs1, shiftMask32:$rs2)>;
+ def ...
[truncated]
``````````
</details>
https://github.com/llvm/llvm-project/pull/228726
More information about the cfe-commits
mailing list