[clang] 08e2450 - [RISCV][P-ext] Support Packed Slide 1 up/down (#228049)
via cfe-commits
cfe-commits at lists.llvm.org
Fri Oct 2 10:02:10 PDT 2026
Author: Hongyu Chen
Date: 2026-10-02T17:01:56Z
New Revision: 08e245062b1028d09aadf8344d5d7e9177ce19d6
URL: https://github.com/llvm/llvm-project/commit/08e245062b1028d09aadf8344d5d7e9177ce19d6
DIFF: https://github.com/llvm/llvm-project/commit/08e245062b1028d09aadf8344d5d7e9177ce19d6.diff
LOG: [RISCV][P-ext] Support Packed Slide 1 up/down (#228049)
This patch supports packed slide 1 up/down intrinsics and codegen.
Added:
Modified:
clang/lib/Headers/riscv_packed_simd.h
clang/test/CodeGen/RISCV/rvp-intrinsics.c
cross-project-tests/intrinsic-header-tests/riscv_packed_simd.c
llvm/lib/Target/RISCV/RISCVISelLowering.cpp
llvm/lib/Target/RISCV/RISCVInstrInfoP.td
llvm/test/CodeGen/RISCV/rvp-simd-32.ll
llvm/test/CodeGen/RISCV/rvp-simd-64.ll
Removed:
################################################################################
diff --git a/clang/lib/Headers/riscv_packed_simd.h b/clang/lib/Headers/riscv_packed_simd.h
index 4148b165e28f3..9ae0c45d60674 100644
--- a/clang/lib/Headers/riscv_packed_simd.h
+++ b/clang/lib/Headers/riscv_packed_simd.h
@@ -274,6 +274,12 @@ typedef uint32_t uint32x2_t __attribute__((__vector_size__(8)));
return __builtin_shufflevector(__lo, __hi, 0, 1, 2, 3, 4, 5, 6, 7); \
}
+#define __packed_slide1(name, ty, elt_ty, ...) \
+ static __inline__ ty __DEFAULT_FN_ATTRS __riscv_##name(ty __rd, \
+ elt_ty __rs1) { \
+ return __builtin_shufflevector(__rd, (ty){__rs1}, __VA_ARGS__); \
+ }
+
#define __packed_pair_ee2(name, ty) \
static __inline__ ty __DEFAULT_FN_ATTRS __riscv_##name(ty __rs1, ty __rs2) { \
return __builtin_shufflevector(__rs1, __rs2, 0, 2); \
@@ -1310,6 +1316,30 @@ __packed_concat4(pjoin2_u8x8, uint8x8_t, uint8x4_t)
__packed_concat2(pjoin2_i16x4, int16x4_t, int16x2_t)
__packed_concat2(pjoin2_u16x4, uint16x4_t, uint16x2_t)
+/* Packed Slide 1 up/down (32-bit) */
+__packed_slide1(pslide1up_i8x4, int8x4_t, int8_t, 4, 0, 1, 2)
+__packed_slide1(pslide1up_u8x4, uint8x4_t, uint8_t, 4, 0, 1, 2)
+__packed_slide1(pslide1up_i16x2, int16x2_t, int16_t, 2, 0)
+__packed_slide1(pslide1up_u16x2, uint16x2_t, uint16_t, 2, 0)
+__packed_slide1(pslide1down_i8x4, int8x4_t, int8_t, 1, 2, 3, 4)
+__packed_slide1(pslide1down_u8x4, uint8x4_t, uint8_t, 1, 2, 3, 4)
+__packed_slide1(pslide1down_i16x2, int16x2_t, int16_t, 1, 2)
+__packed_slide1(pslide1down_u16x2, uint16x2_t, uint16_t, 1, 2)
+
+/* Packed Slide 1 up/down (64-bit) */
+__packed_slide1(pslide1up_i8x8, int8x8_t, int8_t, 8, 0, 1, 2, 3, 4, 5, 6)
+__packed_slide1(pslide1up_u8x8, uint8x8_t, uint8_t, 8, 0, 1, 2, 3, 4, 5, 6)
+__packed_slide1(pslide1up_i16x4, int16x4_t, int16_t, 4, 0, 1, 2)
+__packed_slide1(pslide1up_u16x4, uint16x4_t, uint16_t, 4, 0, 1, 2)
+__packed_slide1(pslide1up_i32x2, int32x2_t, int32_t, 2, 0)
+__packed_slide1(pslide1up_u32x2, uint32x2_t, uint32_t, 2, 0)
+__packed_slide1(pslide1down_i8x8, int8x8_t, int8_t, 1, 2, 3, 4, 5, 6, 7, 8)
+__packed_slide1(pslide1down_u8x8, uint8x8_t, uint8_t, 1, 2, 3, 4, 5, 6, 7, 8)
+__packed_slide1(pslide1down_i16x4, int16x4_t, int16_t, 1, 2, 3, 4)
+__packed_slide1(pslide1down_u16x4, uint16x4_t, uint16_t, 1, 2, 3, 4)
+__packed_slide1(pslide1down_i32x2, int32x2_t, int32_t, 1, 2)
+__packed_slide1(pslide1down_u32x2, uint32x2_t, uint32_t, 1, 2)
+
/* Packed Subvector Extract */
__packed_subvector_extract8(pget_i8x8_i8x4, int8x4_t, int8x8_t)
__packed_subvector_extract8(pget_u8x8_u8x4, uint8x4_t, uint8x8_t)
@@ -1531,6 +1561,7 @@ __packed_reinterpret(u32x2_i32x2, int32x2_t, uint32x2_t)
#undef __packed_unzipo4
#undef __packed_concat2
#undef __packed_concat4
+#undef __packed_slide1
#undef __packed_pair_ee2
#undef __packed_pair_eo2
#undef __packed_pair_oe2
diff --git a/clang/test/CodeGen/RISCV/rvp-intrinsics.c b/clang/test/CodeGen/RISCV/rvp-intrinsics.c
index 5f57545cf6c9f..d9ad53c373da2 100644
--- a/clang/test/CodeGen/RISCV/rvp-intrinsics.c
+++ b/clang/test/CodeGen/RISCV/rvp-intrinsics.c
@@ -9096,3 +9096,447 @@ int32x2_t test_pwunzipho_i32x2(int16x4_t a) { return __riscv_pwunzipho_i32x2(a);
uint32x2_t test_pwunzipho_u32x2(uint16x4_t a) {
return __riscv_pwunzipho_u32x2(a);
}
+
+/* Packed Slide 1 up/down (32-bit) */
+
+// RV32-LABEL: define dso_local i32 @test_pslide1up_i8x4(
+// RV32-SAME: i32 noundef [[RD_COERCE:%.*]], i8 noundef signext [[RS1:%.*]]) #[[ATTR0]] {
+// RV32-NEXT: [[ENTRY:.*:]]
+// RV32-NEXT: [[TMP0:%.*]] = bitcast i32 [[RD_COERCE]] to <4 x i8>
+// RV32-NEXT: [[VECINIT4_I:%.*]] = insertelement <4 x i8> poison, i8 [[RS1]], i64 0
+// RV32-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <4 x i8> [[TMP0]], <4 x i8> [[VECINIT4_I]], <4 x i32> <i32 4, i32 0, i32 1, i32 2>
+// RV32-NEXT: [[TMP1:%.*]] = bitcast <4 x i8> [[SHUFFLE_I]] to i32
+// RV32-NEXT: ret i32 [[TMP1]]
+//
+// RV64-LABEL: define dso_local i32 @test_pslide1up_i8x4(
+// RV64-SAME: i32 noundef [[RD_COERCE:%.*]], i8 noundef signext [[RS1:%.*]]) #[[ATTR0]] {
+// RV64-NEXT: [[ENTRY:.*:]]
+// RV64-NEXT: [[TMP0:%.*]] = bitcast i32 [[RD_COERCE]] to <4 x i8>
+// RV64-NEXT: [[VECINIT4_I:%.*]] = insertelement <4 x i8> poison, i8 [[RS1]], i64 0
+// RV64-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <4 x i8> [[TMP0]], <4 x i8> [[VECINIT4_I]], <4 x i32> <i32 4, i32 0, i32 1, i32 2>
+// RV64-NEXT: [[TMP1:%.*]] = bitcast <4 x i8> [[SHUFFLE_I]] to i32
+// RV64-NEXT: ret i32 [[TMP1]]
+//
+int8x4_t test_pslide1up_i8x4(int8x4_t rd, int8_t rs1) {
+ return __riscv_pslide1up_i8x4(rd, rs1);
+}
+
+// RV32-LABEL: define dso_local i32 @test_pslide1up_u8x4(
+// RV32-SAME: i32 noundef [[RD_COERCE:%.*]], i8 noundef zeroext [[RS1:%.*]]) #[[ATTR0]] {
+// RV32-NEXT: [[ENTRY:.*:]]
+// RV32-NEXT: [[TMP0:%.*]] = bitcast i32 [[RD_COERCE]] to <4 x i8>
+// RV32-NEXT: [[VECINIT4_I:%.*]] = insertelement <4 x i8> poison, i8 [[RS1]], i64 0
+// RV32-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <4 x i8> [[TMP0]], <4 x i8> [[VECINIT4_I]], <4 x i32> <i32 4, i32 0, i32 1, i32 2>
+// RV32-NEXT: [[TMP1:%.*]] = bitcast <4 x i8> [[SHUFFLE_I]] to i32
+// RV32-NEXT: ret i32 [[TMP1]]
+//
+// RV64-LABEL: define dso_local i32 @test_pslide1up_u8x4(
+// RV64-SAME: i32 noundef [[RD_COERCE:%.*]], i8 noundef zeroext [[RS1:%.*]]) #[[ATTR0]] {
+// RV64-NEXT: [[ENTRY:.*:]]
+// RV64-NEXT: [[TMP0:%.*]] = bitcast i32 [[RD_COERCE]] to <4 x i8>
+// RV64-NEXT: [[VECINIT4_I:%.*]] = insertelement <4 x i8> poison, i8 [[RS1]], i64 0
+// RV64-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <4 x i8> [[TMP0]], <4 x i8> [[VECINIT4_I]], <4 x i32> <i32 4, i32 0, i32 1, i32 2>
+// RV64-NEXT: [[TMP1:%.*]] = bitcast <4 x i8> [[SHUFFLE_I]] to i32
+// RV64-NEXT: ret i32 [[TMP1]]
+//
+uint8x4_t test_pslide1up_u8x4(uint8x4_t rd, uint8_t rs1) {
+ return __riscv_pslide1up_u8x4(rd, rs1);
+}
+
+// RV32-LABEL: define dso_local i32 @test_pslide1up_i16x2(
+// RV32-SAME: i32 noundef [[RD_COERCE:%.*]], i16 noundef signext [[RS1:%.*]]) #[[ATTR0]] {
+// RV32-NEXT: [[ENTRY:.*:]]
+// RV32-NEXT: [[TMP0:%.*]] = bitcast i32 [[RD_COERCE]] to <2 x i16>
+// RV32-NEXT: [[VECINIT2_I:%.*]] = insertelement <2 x i16> poison, i16 [[RS1]], i64 0
+// RV32-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <2 x i16> [[TMP0]], <2 x i16> [[VECINIT2_I]], <2 x i32> <i32 2, i32 0>
+// RV32-NEXT: [[TMP1:%.*]] = bitcast <2 x i16> [[SHUFFLE_I]] to i32
+// RV32-NEXT: ret i32 [[TMP1]]
+//
+// RV64-LABEL: define dso_local i32 @test_pslide1up_i16x2(
+// RV64-SAME: i32 noundef [[RD_COERCE:%.*]], i16 noundef signext [[RS1:%.*]]) #[[ATTR0]] {
+// RV64-NEXT: [[ENTRY:.*:]]
+// RV64-NEXT: [[TMP0:%.*]] = bitcast i32 [[RD_COERCE]] to <2 x i16>
+// RV64-NEXT: [[VECINIT2_I:%.*]] = insertelement <2 x i16> poison, i16 [[RS1]], i64 0
+// RV64-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <2 x i16> [[TMP0]], <2 x i16> [[VECINIT2_I]], <2 x i32> <i32 2, i32 0>
+// RV64-NEXT: [[TMP1:%.*]] = bitcast <2 x i16> [[SHUFFLE_I]] to i32
+// RV64-NEXT: ret i32 [[TMP1]]
+//
+int16x2_t test_pslide1up_i16x2(int16x2_t rd, int16_t rs1) {
+ return __riscv_pslide1up_i16x2(rd, rs1);
+}
+
+// RV32-LABEL: define dso_local i32 @test_pslide1up_u16x2(
+// RV32-SAME: i32 noundef [[RD_COERCE:%.*]], i16 noundef zeroext [[RS1:%.*]]) #[[ATTR0]] {
+// RV32-NEXT: [[ENTRY:.*:]]
+// RV32-NEXT: [[TMP0:%.*]] = bitcast i32 [[RD_COERCE]] to <2 x i16>
+// RV32-NEXT: [[VECINIT2_I:%.*]] = insertelement <2 x i16> poison, i16 [[RS1]], i64 0
+// RV32-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <2 x i16> [[TMP0]], <2 x i16> [[VECINIT2_I]], <2 x i32> <i32 2, i32 0>
+// RV32-NEXT: [[TMP1:%.*]] = bitcast <2 x i16> [[SHUFFLE_I]] to i32
+// RV32-NEXT: ret i32 [[TMP1]]
+//
+// RV64-LABEL: define dso_local i32 @test_pslide1up_u16x2(
+// RV64-SAME: i32 noundef [[RD_COERCE:%.*]], i16 noundef zeroext [[RS1:%.*]]) #[[ATTR0]] {
+// RV64-NEXT: [[ENTRY:.*:]]
+// RV64-NEXT: [[TMP0:%.*]] = bitcast i32 [[RD_COERCE]] to <2 x i16>
+// RV64-NEXT: [[VECINIT2_I:%.*]] = insertelement <2 x i16> poison, i16 [[RS1]], i64 0
+// RV64-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <2 x i16> [[TMP0]], <2 x i16> [[VECINIT2_I]], <2 x i32> <i32 2, i32 0>
+// RV64-NEXT: [[TMP1:%.*]] = bitcast <2 x i16> [[SHUFFLE_I]] to i32
+// RV64-NEXT: ret i32 [[TMP1]]
+//
+uint16x2_t test_pslide1up_u16x2(uint16x2_t rd, uint16_t rs1) {
+ return __riscv_pslide1up_u16x2(rd, rs1);
+}
+
+// RV32-LABEL: define dso_local i32 @test_pslide1down_i8x4(
+// RV32-SAME: i32 noundef [[RD_COERCE:%.*]], i8 noundef signext [[RS1:%.*]]) #[[ATTR0]] {
+// RV32-NEXT: [[ENTRY:.*:]]
+// RV32-NEXT: [[TMP0:%.*]] = bitcast i32 [[RD_COERCE]] to <4 x i8>
+// RV32-NEXT: [[VECINIT4_I:%.*]] = insertelement <4 x i8> poison, i8 [[RS1]], i64 0
+// RV32-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <4 x i8> [[TMP0]], <4 x i8> [[VECINIT4_I]], <4 x i32> <i32 1, i32 2, i32 3, i32 4>
+// RV32-NEXT: [[TMP1:%.*]] = bitcast <4 x i8> [[SHUFFLE_I]] to i32
+// RV32-NEXT: ret i32 [[TMP1]]
+//
+// RV64-LABEL: define dso_local i32 @test_pslide1down_i8x4(
+// RV64-SAME: i32 noundef [[RD_COERCE:%.*]], i8 noundef signext [[RS1:%.*]]) #[[ATTR0]] {
+// RV64-NEXT: [[ENTRY:.*:]]
+// RV64-NEXT: [[TMP0:%.*]] = bitcast i32 [[RD_COERCE]] to <4 x i8>
+// RV64-NEXT: [[VECINIT4_I:%.*]] = insertelement <4 x i8> poison, i8 [[RS1]], i64 0
+// RV64-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <4 x i8> [[TMP0]], <4 x i8> [[VECINIT4_I]], <4 x i32> <i32 1, i32 2, i32 3, i32 4>
+// RV64-NEXT: [[TMP1:%.*]] = bitcast <4 x i8> [[SHUFFLE_I]] to i32
+// RV64-NEXT: ret i32 [[TMP1]]
+//
+int8x4_t test_pslide1down_i8x4(int8x4_t rd, int8_t rs1) {
+ return __riscv_pslide1down_i8x4(rd, rs1);
+}
+
+// RV32-LABEL: define dso_local i32 @test_pslide1down_u8x4(
+// RV32-SAME: i32 noundef [[RD_COERCE:%.*]], i8 noundef zeroext [[RS1:%.*]]) #[[ATTR0]] {
+// RV32-NEXT: [[ENTRY:.*:]]
+// RV32-NEXT: [[TMP0:%.*]] = bitcast i32 [[RD_COERCE]] to <4 x i8>
+// RV32-NEXT: [[VECINIT4_I:%.*]] = insertelement <4 x i8> poison, i8 [[RS1]], i64 0
+// RV32-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <4 x i8> [[TMP0]], <4 x i8> [[VECINIT4_I]], <4 x i32> <i32 1, i32 2, i32 3, i32 4>
+// RV32-NEXT: [[TMP1:%.*]] = bitcast <4 x i8> [[SHUFFLE_I]] to i32
+// RV32-NEXT: ret i32 [[TMP1]]
+//
+// RV64-LABEL: define dso_local i32 @test_pslide1down_u8x4(
+// RV64-SAME: i32 noundef [[RD_COERCE:%.*]], i8 noundef zeroext [[RS1:%.*]]) #[[ATTR0]] {
+// RV64-NEXT: [[ENTRY:.*:]]
+// RV64-NEXT: [[TMP0:%.*]] = bitcast i32 [[RD_COERCE]] to <4 x i8>
+// RV64-NEXT: [[VECINIT4_I:%.*]] = insertelement <4 x i8> poison, i8 [[RS1]], i64 0
+// RV64-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <4 x i8> [[TMP0]], <4 x i8> [[VECINIT4_I]], <4 x i32> <i32 1, i32 2, i32 3, i32 4>
+// RV64-NEXT: [[TMP1:%.*]] = bitcast <4 x i8> [[SHUFFLE_I]] to i32
+// RV64-NEXT: ret i32 [[TMP1]]
+//
+uint8x4_t test_pslide1down_u8x4(uint8x4_t rd, uint8_t rs1) {
+ return __riscv_pslide1down_u8x4(rd, rs1);
+}
+
+// RV32-LABEL: define dso_local i32 @test_pslide1down_i16x2(
+// RV32-SAME: i32 noundef [[RD_COERCE:%.*]], i16 noundef signext [[RS1:%.*]]) #[[ATTR0]] {
+// RV32-NEXT: [[ENTRY:.*:]]
+// RV32-NEXT: [[TMP0:%.*]] = bitcast i32 [[RD_COERCE]] to <2 x i16>
+// RV32-NEXT: [[VECINIT2_I:%.*]] = insertelement <2 x i16> poison, i16 [[RS1]], i64 0
+// RV32-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <2 x i16> [[TMP0]], <2 x i16> [[VECINIT2_I]], <2 x i32> <i32 1, i32 2>
+// RV32-NEXT: [[TMP1:%.*]] = bitcast <2 x i16> [[SHUFFLE_I]] to i32
+// RV32-NEXT: ret i32 [[TMP1]]
+//
+// RV64-LABEL: define dso_local i32 @test_pslide1down_i16x2(
+// RV64-SAME: i32 noundef [[RD_COERCE:%.*]], i16 noundef signext [[RS1:%.*]]) #[[ATTR0]] {
+// RV64-NEXT: [[ENTRY:.*:]]
+// RV64-NEXT: [[TMP0:%.*]] = bitcast i32 [[RD_COERCE]] to <2 x i16>
+// RV64-NEXT: [[VECINIT2_I:%.*]] = insertelement <2 x i16> poison, i16 [[RS1]], i64 0
+// RV64-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <2 x i16> [[TMP0]], <2 x i16> [[VECINIT2_I]], <2 x i32> <i32 1, i32 2>
+// RV64-NEXT: [[TMP1:%.*]] = bitcast <2 x i16> [[SHUFFLE_I]] to i32
+// RV64-NEXT: ret i32 [[TMP1]]
+//
+int16x2_t test_pslide1down_i16x2(int16x2_t rd, int16_t rs1) {
+ return __riscv_pslide1down_i16x2(rd, rs1);
+}
+
+// RV32-LABEL: define dso_local i32 @test_pslide1down_u16x2(
+// RV32-SAME: i32 noundef [[RD_COERCE:%.*]], i16 noundef zeroext [[RS1:%.*]]) #[[ATTR0]] {
+// RV32-NEXT: [[ENTRY:.*:]]
+// RV32-NEXT: [[TMP0:%.*]] = bitcast i32 [[RD_COERCE]] to <2 x i16>
+// RV32-NEXT: [[VECINIT2_I:%.*]] = insertelement <2 x i16> poison, i16 [[RS1]], i64 0
+// RV32-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <2 x i16> [[TMP0]], <2 x i16> [[VECINIT2_I]], <2 x i32> <i32 1, i32 2>
+// RV32-NEXT: [[TMP1:%.*]] = bitcast <2 x i16> [[SHUFFLE_I]] to i32
+// RV32-NEXT: ret i32 [[TMP1]]
+//
+// RV64-LABEL: define dso_local i32 @test_pslide1down_u16x2(
+// RV64-SAME: i32 noundef [[RD_COERCE:%.*]], i16 noundef zeroext [[RS1:%.*]]) #[[ATTR0]] {
+// RV64-NEXT: [[ENTRY:.*:]]
+// RV64-NEXT: [[TMP0:%.*]] = bitcast i32 [[RD_COERCE]] to <2 x i16>
+// RV64-NEXT: [[VECINIT2_I:%.*]] = insertelement <2 x i16> poison, i16 [[RS1]], i64 0
+// RV64-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <2 x i16> [[TMP0]], <2 x i16> [[VECINIT2_I]], <2 x i32> <i32 1, i32 2>
+// RV64-NEXT: [[TMP1:%.*]] = bitcast <2 x i16> [[SHUFFLE_I]] to i32
+// RV64-NEXT: ret i32 [[TMP1]]
+//
+uint16x2_t test_pslide1down_u16x2(uint16x2_t rd, uint16_t rs1) {
+ return __riscv_pslide1down_u16x2(rd, rs1);
+}
+
+/* Packed Slide 1 up/down (64-bit) */
+
+// RV32-LABEL: define dso_local i64 @test_pslide1up_i8x8(
+// RV32-SAME: i64 noundef [[RD_COERCE:%.*]], i8 noundef signext [[RS1:%.*]]) #[[ATTR0]] {
+// RV32-NEXT: [[ENTRY:.*:]]
+// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <8 x i8>
+// RV32-NEXT: [[VECINIT8_I:%.*]] = insertelement <8 x i8> poison, i8 [[RS1]], i64 0
+// RV32-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <8 x i8> [[TMP0]], <8 x i8> [[VECINIT8_I]], <8 x i32> <i32 8, i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6>
+// RV32-NEXT: [[TMP1:%.*]] = bitcast <8 x i8> [[SHUFFLE_I]] to i64
+// RV32-NEXT: ret i64 [[TMP1]]
+//
+// RV64-LABEL: define dso_local i64 @test_pslide1up_i8x8(
+// RV64-SAME: i64 noundef [[RD_COERCE:%.*]], i8 noundef signext [[RS1:%.*]]) #[[ATTR0]] {
+// RV64-NEXT: [[ENTRY:.*:]]
+// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <8 x i8>
+// RV64-NEXT: [[VECINIT8_I:%.*]] = insertelement <8 x i8> poison, i8 [[RS1]], i64 0
+// RV64-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <8 x i8> [[TMP0]], <8 x i8> [[VECINIT8_I]], <8 x i32> <i32 8, i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6>
+// RV64-NEXT: [[TMP1:%.*]] = bitcast <8 x i8> [[SHUFFLE_I]] to i64
+// RV64-NEXT: ret i64 [[TMP1]]
+//
+int8x8_t test_pslide1up_i8x8(int8x8_t rd, int8_t rs1) {
+ return __riscv_pslide1up_i8x8(rd, rs1);
+}
+
+// RV32-LABEL: define dso_local i64 @test_pslide1up_u8x8(
+// RV32-SAME: i64 noundef [[RD_COERCE:%.*]], i8 noundef zeroext [[RS1:%.*]]) #[[ATTR0]] {
+// RV32-NEXT: [[ENTRY:.*:]]
+// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <8 x i8>
+// RV32-NEXT: [[VECINIT8_I:%.*]] = insertelement <8 x i8> poison, i8 [[RS1]], i64 0
+// RV32-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <8 x i8> [[TMP0]], <8 x i8> [[VECINIT8_I]], <8 x i32> <i32 8, i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6>
+// RV32-NEXT: [[TMP1:%.*]] = bitcast <8 x i8> [[SHUFFLE_I]] to i64
+// RV32-NEXT: ret i64 [[TMP1]]
+//
+// RV64-LABEL: define dso_local i64 @test_pslide1up_u8x8(
+// RV64-SAME: i64 noundef [[RD_COERCE:%.*]], i8 noundef zeroext [[RS1:%.*]]) #[[ATTR0]] {
+// RV64-NEXT: [[ENTRY:.*:]]
+// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <8 x i8>
+// RV64-NEXT: [[VECINIT8_I:%.*]] = insertelement <8 x i8> poison, i8 [[RS1]], i64 0
+// RV64-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <8 x i8> [[TMP0]], <8 x i8> [[VECINIT8_I]], <8 x i32> <i32 8, i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6>
+// RV64-NEXT: [[TMP1:%.*]] = bitcast <8 x i8> [[SHUFFLE_I]] to i64
+// RV64-NEXT: ret i64 [[TMP1]]
+//
+uint8x8_t test_pslide1up_u8x8(uint8x8_t rd, uint8_t rs1) {
+ return __riscv_pslide1up_u8x8(rd, rs1);
+}
+
+// RV32-LABEL: define dso_local i64 @test_pslide1up_i16x4(
+// RV32-SAME: i64 noundef [[RD_COERCE:%.*]], i16 noundef signext [[RS1:%.*]]) #[[ATTR0]] {
+// RV32-NEXT: [[ENTRY:.*:]]
+// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <4 x i16>
+// RV32-NEXT: [[VECINIT4_I:%.*]] = insertelement <4 x i16> poison, i16 [[RS1]], i64 0
+// RV32-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <4 x i16> [[TMP0]], <4 x i16> [[VECINIT4_I]], <4 x i32> <i32 4, i32 0, i32 1, i32 2>
+// RV32-NEXT: [[TMP1:%.*]] = bitcast <4 x i16> [[SHUFFLE_I]] to i64
+// RV32-NEXT: ret i64 [[TMP1]]
+//
+// RV64-LABEL: define dso_local i64 @test_pslide1up_i16x4(
+// RV64-SAME: i64 noundef [[RD_COERCE:%.*]], i16 noundef signext [[RS1:%.*]]) #[[ATTR0]] {
+// RV64-NEXT: [[ENTRY:.*:]]
+// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <4 x i16>
+// RV64-NEXT: [[VECINIT4_I:%.*]] = insertelement <4 x i16> poison, i16 [[RS1]], i64 0
+// RV64-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <4 x i16> [[TMP0]], <4 x i16> [[VECINIT4_I]], <4 x i32> <i32 4, i32 0, i32 1, i32 2>
+// RV64-NEXT: [[TMP1:%.*]] = bitcast <4 x i16> [[SHUFFLE_I]] to i64
+// RV64-NEXT: ret i64 [[TMP1]]
+//
+int16x4_t test_pslide1up_i16x4(int16x4_t rd, int16_t rs1) {
+ return __riscv_pslide1up_i16x4(rd, rs1);
+}
+
+// RV32-LABEL: define dso_local i64 @test_pslide1up_u16x4(
+// RV32-SAME: i64 noundef [[RD_COERCE:%.*]], i16 noundef zeroext [[RS1:%.*]]) #[[ATTR0]] {
+// RV32-NEXT: [[ENTRY:.*:]]
+// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <4 x i16>
+// RV32-NEXT: [[VECINIT4_I:%.*]] = insertelement <4 x i16> poison, i16 [[RS1]], i64 0
+// RV32-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <4 x i16> [[TMP0]], <4 x i16> [[VECINIT4_I]], <4 x i32> <i32 4, i32 0, i32 1, i32 2>
+// RV32-NEXT: [[TMP1:%.*]] = bitcast <4 x i16> [[SHUFFLE_I]] to i64
+// RV32-NEXT: ret i64 [[TMP1]]
+//
+// RV64-LABEL: define dso_local i64 @test_pslide1up_u16x4(
+// RV64-SAME: i64 noundef [[RD_COERCE:%.*]], i16 noundef zeroext [[RS1:%.*]]) #[[ATTR0]] {
+// RV64-NEXT: [[ENTRY:.*:]]
+// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <4 x i16>
+// RV64-NEXT: [[VECINIT4_I:%.*]] = insertelement <4 x i16> poison, i16 [[RS1]], i64 0
+// RV64-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <4 x i16> [[TMP0]], <4 x i16> [[VECINIT4_I]], <4 x i32> <i32 4, i32 0, i32 1, i32 2>
+// RV64-NEXT: [[TMP1:%.*]] = bitcast <4 x i16> [[SHUFFLE_I]] to i64
+// RV64-NEXT: ret i64 [[TMP1]]
+//
+uint16x4_t test_pslide1up_u16x4(uint16x4_t rd, uint16_t rs1) {
+ return __riscv_pslide1up_u16x4(rd, rs1);
+}
+
+// RV32-LABEL: define dso_local i64 @test_pslide1up_i32x2(
+// RV32-SAME: i64 noundef [[RD_COERCE:%.*]], i32 noundef [[RS1:%.*]]) #[[ATTR0]] {
+// RV32-NEXT: [[ENTRY:.*:]]
+// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <2 x i32>
+// RV32-NEXT: [[VECINIT2_I:%.*]] = insertelement <2 x i32> poison, i32 [[RS1]], i64 0
+// RV32-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <2 x i32> [[TMP0]], <2 x i32> [[VECINIT2_I]], <2 x i32> <i32 2, i32 0>
+// RV32-NEXT: [[TMP1:%.*]] = bitcast <2 x i32> [[SHUFFLE_I]] to i64
+// RV32-NEXT: ret i64 [[TMP1]]
+//
+// RV64-LABEL: define dso_local i64 @test_pslide1up_i32x2(
+// RV64-SAME: i64 noundef [[RD_COERCE:%.*]], i32 noundef signext [[RS1:%.*]]) #[[ATTR0]] {
+// RV64-NEXT: [[ENTRY:.*:]]
+// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <2 x i32>
+// RV64-NEXT: [[VECINIT2_I:%.*]] = insertelement <2 x i32> poison, i32 [[RS1]], i64 0
+// RV64-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <2 x i32> [[TMP0]], <2 x i32> [[VECINIT2_I]], <2 x i32> <i32 2, i32 0>
+// RV64-NEXT: [[TMP1:%.*]] = bitcast <2 x i32> [[SHUFFLE_I]] to i64
+// RV64-NEXT: ret i64 [[TMP1]]
+//
+int32x2_t test_pslide1up_i32x2(int32x2_t rd, int32_t rs1) {
+ return __riscv_pslide1up_i32x2(rd, rs1);
+}
+
+// RV32-LABEL: define dso_local i64 @test_pslide1up_u32x2(
+// RV32-SAME: i64 noundef [[RD_COERCE:%.*]], i32 noundef [[RS1:%.*]]) #[[ATTR0]] {
+// RV32-NEXT: [[ENTRY:.*:]]
+// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <2 x i32>
+// RV32-NEXT: [[VECINIT2_I:%.*]] = insertelement <2 x i32> poison, i32 [[RS1]], i64 0
+// RV32-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <2 x i32> [[TMP0]], <2 x i32> [[VECINIT2_I]], <2 x i32> <i32 2, i32 0>
+// RV32-NEXT: [[TMP1:%.*]] = bitcast <2 x i32> [[SHUFFLE_I]] to i64
+// RV32-NEXT: ret i64 [[TMP1]]
+//
+// RV64-LABEL: define dso_local i64 @test_pslide1up_u32x2(
+// RV64-SAME: i64 noundef [[RD_COERCE:%.*]], i32 noundef signext [[RS1:%.*]]) #[[ATTR0]] {
+// RV64-NEXT: [[ENTRY:.*:]]
+// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <2 x i32>
+// RV64-NEXT: [[VECINIT2_I:%.*]] = insertelement <2 x i32> poison, i32 [[RS1]], i64 0
+// RV64-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <2 x i32> [[TMP0]], <2 x i32> [[VECINIT2_I]], <2 x i32> <i32 2, i32 0>
+// RV64-NEXT: [[TMP1:%.*]] = bitcast <2 x i32> [[SHUFFLE_I]] to i64
+// RV64-NEXT: ret i64 [[TMP1]]
+//
+uint32x2_t test_pslide1up_u32x2(uint32x2_t rd, uint32_t rs1) {
+ return __riscv_pslide1up_u32x2(rd, rs1);
+}
+
+// RV32-LABEL: define dso_local i64 @test_pslide1down_i8x8(
+// RV32-SAME: i64 noundef [[RD_COERCE:%.*]], i8 noundef signext [[RS1:%.*]]) #[[ATTR0]] {
+// RV32-NEXT: [[ENTRY:.*:]]
+// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <8 x i8>
+// RV32-NEXT: [[VECINIT8_I:%.*]] = insertelement <8 x i8> poison, i8 [[RS1]], i64 0
+// RV32-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <8 x i8> [[TMP0]], <8 x i8> [[VECINIT8_I]], <8 x i32> <i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8>
+// RV32-NEXT: [[TMP1:%.*]] = bitcast <8 x i8> [[SHUFFLE_I]] to i64
+// RV32-NEXT: ret i64 [[TMP1]]
+//
+// RV64-LABEL: define dso_local i64 @test_pslide1down_i8x8(
+// RV64-SAME: i64 noundef [[RD_COERCE:%.*]], i8 noundef signext [[RS1:%.*]]) #[[ATTR0]] {
+// RV64-NEXT: [[ENTRY:.*:]]
+// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <8 x i8>
+// RV64-NEXT: [[VECINIT8_I:%.*]] = insertelement <8 x i8> poison, i8 [[RS1]], i64 0
+// RV64-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <8 x i8> [[TMP0]], <8 x i8> [[VECINIT8_I]], <8 x i32> <i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8>
+// RV64-NEXT: [[TMP1:%.*]] = bitcast <8 x i8> [[SHUFFLE_I]] to i64
+// RV64-NEXT: ret i64 [[TMP1]]
+//
+int8x8_t test_pslide1down_i8x8(int8x8_t rd, int8_t rs1) {
+ return __riscv_pslide1down_i8x8(rd, rs1);
+}
+
+// RV32-LABEL: define dso_local i64 @test_pslide1down_u8x8(
+// RV32-SAME: i64 noundef [[RD_COERCE:%.*]], i8 noundef zeroext [[RS1:%.*]]) #[[ATTR0]] {
+// RV32-NEXT: [[ENTRY:.*:]]
+// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <8 x i8>
+// RV32-NEXT: [[VECINIT8_I:%.*]] = insertelement <8 x i8> poison, i8 [[RS1]], i64 0
+// RV32-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <8 x i8> [[TMP0]], <8 x i8> [[VECINIT8_I]], <8 x i32> <i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8>
+// RV32-NEXT: [[TMP1:%.*]] = bitcast <8 x i8> [[SHUFFLE_I]] to i64
+// RV32-NEXT: ret i64 [[TMP1]]
+//
+// RV64-LABEL: define dso_local i64 @test_pslide1down_u8x8(
+// RV64-SAME: i64 noundef [[RD_COERCE:%.*]], i8 noundef zeroext [[RS1:%.*]]) #[[ATTR0]] {
+// RV64-NEXT: [[ENTRY:.*:]]
+// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <8 x i8>
+// RV64-NEXT: [[VECINIT8_I:%.*]] = insertelement <8 x i8> poison, i8 [[RS1]], i64 0
+// RV64-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <8 x i8> [[TMP0]], <8 x i8> [[VECINIT8_I]], <8 x i32> <i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8>
+// RV64-NEXT: [[TMP1:%.*]] = bitcast <8 x i8> [[SHUFFLE_I]] to i64
+// RV64-NEXT: ret i64 [[TMP1]]
+//
+uint8x8_t test_pslide1down_u8x8(uint8x8_t rd, uint8_t rs1) {
+ return __riscv_pslide1down_u8x8(rd, rs1);
+}
+
+// RV32-LABEL: define dso_local i64 @test_pslide1down_i16x4(
+// RV32-SAME: i64 noundef [[RD_COERCE:%.*]], i16 noundef signext [[RS1:%.*]]) #[[ATTR0]] {
+// RV32-NEXT: [[ENTRY:.*:]]
+// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <4 x i16>
+// RV32-NEXT: [[VECINIT4_I:%.*]] = insertelement <4 x i16> poison, i16 [[RS1]], i64 0
+// RV32-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <4 x i16> [[TMP0]], <4 x i16> [[VECINIT4_I]], <4 x i32> <i32 1, i32 2, i32 3, i32 4>
+// RV32-NEXT: [[TMP1:%.*]] = bitcast <4 x i16> [[SHUFFLE_I]] to i64
+// RV32-NEXT: ret i64 [[TMP1]]
+//
+// RV64-LABEL: define dso_local i64 @test_pslide1down_i16x4(
+// RV64-SAME: i64 noundef [[RD_COERCE:%.*]], i16 noundef signext [[RS1:%.*]]) #[[ATTR0]] {
+// RV64-NEXT: [[ENTRY:.*:]]
+// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <4 x i16>
+// RV64-NEXT: [[VECINIT4_I:%.*]] = insertelement <4 x i16> poison, i16 [[RS1]], i64 0
+// RV64-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <4 x i16> [[TMP0]], <4 x i16> [[VECINIT4_I]], <4 x i32> <i32 1, i32 2, i32 3, i32 4>
+// RV64-NEXT: [[TMP1:%.*]] = bitcast <4 x i16> [[SHUFFLE_I]] to i64
+// RV64-NEXT: ret i64 [[TMP1]]
+//
+int16x4_t test_pslide1down_i16x4(int16x4_t rd, int16_t rs1) {
+ return __riscv_pslide1down_i16x4(rd, rs1);
+}
+
+// RV32-LABEL: define dso_local i64 @test_pslide1down_u16x4(
+// RV32-SAME: i64 noundef [[RD_COERCE:%.*]], i16 noundef zeroext [[RS1:%.*]]) #[[ATTR0]] {
+// RV32-NEXT: [[ENTRY:.*:]]
+// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <4 x i16>
+// RV32-NEXT: [[VECINIT4_I:%.*]] = insertelement <4 x i16> poison, i16 [[RS1]], i64 0
+// RV32-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <4 x i16> [[TMP0]], <4 x i16> [[VECINIT4_I]], <4 x i32> <i32 1, i32 2, i32 3, i32 4>
+// RV32-NEXT: [[TMP1:%.*]] = bitcast <4 x i16> [[SHUFFLE_I]] to i64
+// RV32-NEXT: ret i64 [[TMP1]]
+//
+// RV64-LABEL: define dso_local i64 @test_pslide1down_u16x4(
+// RV64-SAME: i64 noundef [[RD_COERCE:%.*]], i16 noundef zeroext [[RS1:%.*]]) #[[ATTR0]] {
+// RV64-NEXT: [[ENTRY:.*:]]
+// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <4 x i16>
+// RV64-NEXT: [[VECINIT4_I:%.*]] = insertelement <4 x i16> poison, i16 [[RS1]], i64 0
+// RV64-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <4 x i16> [[TMP0]], <4 x i16> [[VECINIT4_I]], <4 x i32> <i32 1, i32 2, i32 3, i32 4>
+// RV64-NEXT: [[TMP1:%.*]] = bitcast <4 x i16> [[SHUFFLE_I]] to i64
+// RV64-NEXT: ret i64 [[TMP1]]
+//
+uint16x4_t test_pslide1down_u16x4(uint16x4_t rd, uint16_t rs1) {
+ return __riscv_pslide1down_u16x4(rd, rs1);
+}
+
+// RV32-LABEL: define dso_local i64 @test_pslide1down_i32x2(
+// RV32-SAME: i64 noundef [[RD_COERCE:%.*]], i32 noundef [[RS1:%.*]]) #[[ATTR0]] {
+// RV32-NEXT: [[ENTRY:.*:]]
+// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <2 x i32>
+// RV32-NEXT: [[VECINIT2_I:%.*]] = insertelement <2 x i32> poison, i32 [[RS1]], i64 0
+// RV32-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <2 x i32> [[TMP0]], <2 x i32> [[VECINIT2_I]], <2 x i32> <i32 1, i32 2>
+// RV32-NEXT: [[TMP1:%.*]] = bitcast <2 x i32> [[SHUFFLE_I]] to i64
+// RV32-NEXT: ret i64 [[TMP1]]
+//
+// RV64-LABEL: define dso_local i64 @test_pslide1down_i32x2(
+// RV64-SAME: i64 noundef [[RD_COERCE:%.*]], i32 noundef signext [[RS1:%.*]]) #[[ATTR0]] {
+// RV64-NEXT: [[ENTRY:.*:]]
+// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <2 x i32>
+// RV64-NEXT: [[VECINIT2_I:%.*]] = insertelement <2 x i32> poison, i32 [[RS1]], i64 0
+// RV64-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <2 x i32> [[TMP0]], <2 x i32> [[VECINIT2_I]], <2 x i32> <i32 1, i32 2>
+// RV64-NEXT: [[TMP1:%.*]] = bitcast <2 x i32> [[SHUFFLE_I]] to i64
+// RV64-NEXT: ret i64 [[TMP1]]
+//
+int32x2_t test_pslide1down_i32x2(int32x2_t rd, int32_t rs1) {
+ return __riscv_pslide1down_i32x2(rd, rs1);
+}
+
+// RV32-LABEL: define dso_local i64 @test_pslide1down_u32x2(
+// RV32-SAME: i64 noundef [[RD_COERCE:%.*]], i32 noundef [[RS1:%.*]]) #[[ATTR0]] {
+// RV32-NEXT: [[ENTRY:.*:]]
+// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <2 x i32>
+// RV32-NEXT: [[VECINIT2_I:%.*]] = insertelement <2 x i32> poison, i32 [[RS1]], i64 0
+// RV32-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <2 x i32> [[TMP0]], <2 x i32> [[VECINIT2_I]], <2 x i32> <i32 1, i32 2>
+// RV32-NEXT: [[TMP1:%.*]] = bitcast <2 x i32> [[SHUFFLE_I]] to i64
+// RV32-NEXT: ret i64 [[TMP1]]
+//
+// RV64-LABEL: define dso_local i64 @test_pslide1down_u32x2(
+// RV64-SAME: i64 noundef [[RD_COERCE:%.*]], i32 noundef signext [[RS1:%.*]]) #[[ATTR0]] {
+// RV64-NEXT: [[ENTRY:.*:]]
+// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <2 x i32>
+// RV64-NEXT: [[VECINIT2_I:%.*]] = insertelement <2 x i32> poison, i32 [[RS1]], i64 0
+// RV64-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <2 x i32> [[TMP0]], <2 x i32> [[VECINIT2_I]], <2 x i32> <i32 1, i32 2>
+// RV64-NEXT: [[TMP1:%.*]] = bitcast <2 x i32> [[SHUFFLE_I]] to i64
+// RV64-NEXT: ret i64 [[TMP1]]
+//
+uint32x2_t test_pslide1down_u32x2(uint32x2_t rd, uint32_t rs1) {
+ return __riscv_pslide1down_u32x2(rd, rs1);
+}
diff --git a/cross-project-tests/intrinsic-header-tests/riscv_packed_simd.c b/cross-project-tests/intrinsic-header-tests/riscv_packed_simd.c
index e7ad4af120270..6d92dd5990261 100644
--- a/cross-project-tests/intrinsic-header-tests/riscv_packed_simd.c
+++ b/cross-project-tests/intrinsic-header-tests/riscv_packed_simd.c
@@ -5547,3 +5547,137 @@ int32x2_t test_pwunzipho_i32x2(int16x4_t a) {
uint32x2_t test_pwunzipho_u32x2(uint16x4_t a) {
return __riscv_pwunzipho_u32x2(a);
}
+
+/* Packed Slide 1 up/down (32-bit) */
+
+// CHECK-LABEL: test_pslide1up_i8x4:
+// CHECK: slx
+int8x4_t test_pslide1up_i8x4(int8x4_t rd, int8_t rs1) {
+ return __riscv_pslide1up_i8x4(rd, rs1);
+}
+
+// CHECK-LABEL: test_pslide1up_u8x4:
+// CHECK: slx
+uint8x4_t test_pslide1up_u8x4(uint8x4_t rd, uint8_t rs1) {
+ return __riscv_pslide1up_u8x4(rd, rs1);
+}
+
+// CHECK-LABEL: test_pslide1up_i16x2:
+// RV32: pack
+// RV64: ppaire.h
+int16x2_t test_pslide1up_i16x2(int16x2_t rd, int16_t rs1) {
+ return __riscv_pslide1up_i16x2(rd, rs1);
+}
+
+// CHECK-LABEL: test_pslide1up_u16x2:
+// RV32: pack
+// RV64: ppaire.h
+uint16x2_t test_pslide1up_u16x2(uint16x2_t rd, uint16_t rs1) {
+ return __riscv_pslide1up_u16x2(rd, rs1);
+}
+
+// CHECK-LABEL: test_pslide1down_i8x4:
+// RV32: srx
+// RV64: pack
+// RV64: srli
+int8x4_t test_pslide1down_i8x4(int8x4_t rd, int8_t rs1) {
+ return __riscv_pslide1down_i8x4(rd, rs1);
+}
+
+// CHECK-LABEL: test_pslide1down_u8x4:
+// RV32: srx
+// RV64: pack
+// RV64: srli
+uint8x4_t test_pslide1down_u8x4(uint8x4_t rd, uint8_t rs1) {
+ return __riscv_pslide1down_u8x4(rd, rs1);
+}
+
+// CHECK-LABEL: test_pslide1down_i16x2:
+// CHECK: ppairoe.h
+int16x2_t test_pslide1down_i16x2(int16x2_t rd, int16_t rs1) {
+ return __riscv_pslide1down_i16x2(rd, rs1);
+}
+
+// CHECK-LABEL: test_pslide1down_u16x2:
+// CHECK: ppairoe.h
+uint16x2_t test_pslide1down_u16x2(uint16x2_t rd, uint16_t rs1) {
+ return __riscv_pslide1down_u16x2(rd, rs1);
+}
+
+/* Packed Slide 1 up/down (64-bit) */
+
+// CHECK-LABEL: test_pslide1up_i8x8:
+// CHECK: slx
+int8x8_t test_pslide1up_i8x8(int8x8_t rd, int8_t rs1) {
+ return __riscv_pslide1up_i8x8(rd, rs1);
+}
+
+// CHECK-LABEL: test_pslide1up_u8x8:
+// CHECK: slx
+uint8x8_t test_pslide1up_u8x8(uint8x8_t rd, uint8_t rs1) {
+ return __riscv_pslide1up_u8x8(rd, rs1);
+}
+
+// CHECK-LABEL: test_pslide1up_i16x4:
+// CHECK: slx
+int16x4_t test_pslide1up_i16x4(int16x4_t rd, int16_t rs1) {
+ return __riscv_pslide1up_i16x4(rd, rs1);
+}
+
+// CHECK-LABEL: test_pslide1up_u16x4:
+// CHECK: slx
+uint16x4_t test_pslide1up_u16x4(uint16x4_t rd, uint16_t rs1) {
+ return __riscv_pslide1up_u16x4(rd, rs1);
+}
+
+// CHECK-LABEL: test_pslide1up_i32x2:
+// RV32: mv
+// RV64: pack
+int32x2_t test_pslide1up_i32x2(int32x2_t rd, int32_t rs1) {
+ return __riscv_pslide1up_i32x2(rd, rs1);
+}
+
+// CHECK-LABEL: test_pslide1up_u32x2:
+// RV32: mv
+// RV64: pack
+uint32x2_t test_pslide1up_u32x2(uint32x2_t rd, uint32_t rs1) {
+ return __riscv_pslide1up_u32x2(rd, rs1);
+}
+
+// CHECK-LABEL: test_pslide1down_i8x8:
+// CHECK: srx
+int8x8_t test_pslide1down_i8x8(int8x8_t rd, int8_t rs1) {
+ return __riscv_pslide1down_i8x8(rd, rs1);
+}
+
+// CHECK-LABEL: test_pslide1down_u8x8:
+// CHECK: srx
+uint8x8_t test_pslide1down_u8x8(uint8x8_t rd, uint8_t rs1) {
+ return __riscv_pslide1down_u8x8(rd, rs1);
+}
+
+// CHECK-LABEL: test_pslide1down_i16x4:
+// CHECK: srx
+int16x4_t test_pslide1down_i16x4(int16x4_t rd, int16_t rs1) {
+ return __riscv_pslide1down_i16x4(rd, rs1);
+}
+
+// CHECK-LABEL: test_pslide1down_u16x4:
+// CHECK: srx
+uint16x4_t test_pslide1down_u16x4(uint16x4_t rd, uint16_t rs1) {
+ return __riscv_pslide1down_u16x4(rd, rs1);
+}
+
+// CHECK-LABEL: test_pslide1down_i32x2:
+// RV32: mv
+// RV64: ppairoe.w
+int32x2_t test_pslide1down_i32x2(int32x2_t rd, int32_t rs1) {
+ return __riscv_pslide1down_i32x2(rd, rs1);
+}
+
+// CHECK-LABEL: test_pslide1down_u32x2:
+// RV32: mv
+// RV64: ppairoe.w
+uint32x2_t test_pslide1down_u32x2(uint32x2_t rd, uint32_t rs1) {
+ return __riscv_pslide1down_u32x2(rd, rs1);
+}
diff --git a/llvm/lib/Target/RISCV/RISCVISelLowering.cpp b/llvm/lib/Target/RISCV/RISCVISelLowering.cpp
index fbdd4e48c9f63..c96a481705cec 100644
--- a/llvm/lib/Target/RISCV/RISCVISelLowering.cpp
+++ b/llvm/lib/Target/RISCV/RISCVISelLowering.cpp
@@ -6497,6 +6497,111 @@ static SDValue lowerVECTOR_SHUFFLEAsPUnzip(ShuffleVectorSDNode *SVN,
return DAG.getNode(Opc, DL, VT, V1, V2);
}
+// Match a slide by one element with a scalar inserted at either end:
+// <b0, a0, ..., aN-2> -> slide1up
+// <a1, ..., aN-1, b0> -> slide1down
+static SDValue lowerVECTOR_SHUFFLEAsPSlide1(ShuffleVectorSDNode *SVN,
+ const RISCVSubtarget &Subtarget,
+ SelectionDAG &DAG) {
+ MVT VT = SVN->getSimpleValueType(0);
+ if (VT != MVT::v4i8 && VT != MVT::v8i8 && VT != MVT::v2i16 &&
+ VT != MVT::v4i16 && VT != MVT::v2i32)
+ return SDValue();
+ if (!Subtarget.is64Bit() && VT == MVT::v2i32)
+ return SDValue();
+
+ unsigned NumElts = VT.getVectorNumElements();
+ ArrayRef<int> Mask = SVN->getMask();
+ unsigned ActiveElts = NumElts;
+ if (Subtarget.is64Bit() &&
+ all_of(Mask.drop_front(NumElts / 2), [](int M) { return M < 0; }))
+ ActiveElts /= 2;
+
+ ArrayRef<int> ActiveMask = Mask.take_front(ActiveElts);
+ bool SlideUp = ActiveMask[0] == (int)NumElts &&
+ all_of(enumerate(ActiveMask.drop_front()), [](const auto &M) {
+ return M.value() == (int)M.index();
+ });
+ bool SlideDown = ActiveMask.back() == (int)NumElts &&
+ all_of(enumerate(ActiveMask.drop_back()), [](const auto &M) {
+ return M.value() == (int)M.index() + 1;
+ });
+ if (!SlideUp && !SlideDown)
+ return SDValue();
+
+ SDValue V1 = SVN->getOperand(0);
+ SDValue V2 = SVN->getOperand(1);
+ SDLoc DL(SVN);
+ unsigned EltBits = VT.getScalarSizeInBits();
+ unsigned SlideBits = ActiveElts * EltBits;
+ MVT XLenVT = Subtarget.getXLenVT();
+ SDValue Scalar = DAG.getExtractVectorElt(DL, XLenVT, V2, 0);
+
+ // A two-element slide is exactly a packed pair operation.
+ if (ActiveElts == 2) {
+ SDValue ScalarVec = DAG.getBitcast(VT, Scalar);
+ return DAG.getNode(SlideUp ? RISCVISD::PPAIRE : RISCVISD::PPAIROE, DL, VT,
+ SlideUp ? ScalarVec : V1, SlideUp ? V1 : ScalarVec);
+ }
+
+ // Lower a full-register slide to the funnel shift instruction that
+ // implements it.
+ if (SlideBits == Subtarget.getXLen()) {
+ SDValue Bits = DAG.getBitcast(XLenVT, V1);
+ SDValue Shamt = DAG.getConstant(EltBits, DL, XLenVT);
+ if (SlideUp) {
+ Scalar = DAG.getNode(ISD::SHL, DL, XLenVT, Scalar,
+ DAG.getConstant(SlideBits - EltBits, DL, XLenVT));
+ Bits = DAG.getNode(ISD::FSHL, DL, XLenVT, Bits, Scalar, Shamt);
+ } else {
+ Bits = DAG.getNode(ISD::FSHR, DL, XLenVT, Scalar, Bits, Shamt);
+ }
+ return DAG.getBitcast(VT, Bits);
+ }
+
+ // A 32-bit packed slide on RV64 is carried in the low word of a GPR. Expand
+ // it using XLEN operations or packed pair instructions.
+ if (Subtarget.is64Bit()) {
+ assert(SlideBits == 32 && "Unexpected RV64 packed slide width");
+ SDValue Shamt = DAG.getConstant(EltBits, DL, MVT::i64);
+ SDValue Bits = DAG.getBitcast(MVT::i64, V1);
+ if (SlideUp) {
+ Scalar = DAG.getNode(ISD::SHL, DL, MVT::i64, Scalar,
+ DAG.getConstant(64 - EltBits, DL, MVT::i64));
+ Bits = DAG.getNode(ISD::FSHL, DL, MVT::i64, Bits, Scalar, Shamt);
+ } else {
+ SDValue Pair = DAG.getNode(RISCVISD::PPAIRE, DL, MVT::v2i32,
+ DAG.getBitcast(MVT::v2i32, V1),
+ DAG.getBitcast(MVT::v2i32, Scalar));
+ Bits = DAG.getNode(ISD::SRL, DL, MVT::i64, DAG.getBitcast(MVT::i64, Pair),
+ Shamt);
+ }
+ return DAG.getBitcast(VT, Bits);
+ }
+
+ // RV32 represents 64-bit packed values as two GPRs. Form each shifted half
+ // directly so the slide uses two XLEN funnel shifts.
+ assert(SlideBits == 64 && "Unexpected RV32 packed slide width");
+ auto [LoVec, HiVec] = DAG.SplitVector(V1, DL);
+ MVT HalfVT = LoVec.getSimpleValueType();
+ SDValue Lo = DAG.getBitcast(MVT::i32, LoVec);
+ SDValue Hi = DAG.getBitcast(MVT::i32, HiVec);
+ SDValue Shamt = DAG.getConstant(EltBits, DL, MVT::i32);
+ SDValue NewLo;
+ SDValue NewHi;
+ if (SlideUp) {
+ SDValue Insert = DAG.getNode(ISD::SHL, DL, MVT::i32, Scalar,
+ DAG.getConstant(32 - EltBits, DL, MVT::i32));
+ NewLo = DAG.getNode(ISD::FSHL, DL, MVT::i32, Lo, Insert, Shamt);
+ NewHi = DAG.getNode(ISD::FSHL, DL, MVT::i32, Hi, Lo, Shamt);
+ } else {
+ NewLo = DAG.getNode(ISD::FSHR, DL, MVT::i32, Hi, Lo, Shamt);
+ NewHi = DAG.getNode(ISD::FSHR, DL, MVT::i32, Scalar, Hi, Shamt);
+ }
+ return DAG.getNode(ISD::CONCAT_VECTORS, DL, VT, DAG.getBitcast(HalfVT, NewLo),
+ DAG.getBitcast(HalfVT, NewHi));
+}
+
// Match the packed zero-extend shuffle mask <0, N, 2, N+2, ...>: even result
// lanes keep operand 0's even lanes and odd result lanes come from operand 1.
// The odd lanes may select any element of operand 1, which is looser than a
@@ -6670,6 +6775,8 @@ SDValue RISCVTargetLowering::lowerVECTOR_SHUFFLE(SDValue Op,
return DAG.getBitcast(VT, Srl);
}
+ if (SDValue V = lowerVECTOR_SHUFFLEAsPSlide1(SVN, Subtarget, DAG))
+ return V;
if (SDValue V = lowerVECTOR_SHUFFLEAsPUnzip(SVN, DAG, Subtarget.is64Bit()))
return V;
if (SDValue V = lowerVECTOR_SHUFFLEAsPZip(SVN, Subtarget, DAG))
diff --git a/llvm/lib/Target/RISCV/RISCVInstrInfoP.td b/llvm/lib/Target/RISCV/RISCVInstrInfoP.td
index 67fee79db0582..ee5b17749cb73 100644
--- a/llvm/lib/Target/RISCV/RISCVInstrInfoP.td
+++ b/llvm/lib/Target/RISCV/RISCVInstrInfoP.td
@@ -2011,6 +2011,7 @@ def riscv_ppaire : RVSDNode<"PPAIRE", SDT_RISCVPPair>;
def riscv_ppaireo : RVSDNode<"PPAIREO", SDT_RISCVPPair>;
def riscv_ppairoe : RVSDNode<"PPAIROE", SDT_RISCVPPair>;
def riscv_ppairo : RVSDNode<"PPAIRO", SDT_RISCVPPair>;
+
def SDT_RISCVPackedBinary : SDTypeProfile<1, 2, [SDTCisVec<0>,
SDTCisSameAs<0, 1>,
SDTCisSameAs<0, 2>]>;
@@ -3641,6 +3642,10 @@ let append Predicates = [IsRV64] in {
(ZIP16P GPR:$rs1, GPR:$rs2)>;
// Packed pair: pair the even/odd-position elements of rs1 and rs2.
+ def : Pat<(v2i32 (riscv_ppaire (v2i32 GPR:$rs1), (v2i32 GPR:$rs2))),
+ (PACK GPR:$rs1, GPR:$rs2)>;
+ def : Pat<(v2i32 (riscv_ppairoe (v2i32 GPR:$rs1), (v2i32 GPR:$rs2))),
+ (PPAIROE_W GPR:$rs1, GPR:$rs2)>;
def : Pat<(v8i8 (riscv_ppaire (v8i8 GPR:$rs1), (v8i8 GPR:$rs2))),
(PPAIRE_B GPR:$rs1, GPR:$rs2)>;
def : Pat<(v8i8 (riscv_ppaireo (v8i8 GPR:$rs1), (v8i8 GPR:$rs2))),
diff --git a/llvm/test/CodeGen/RISCV/rvp-simd-32.ll b/llvm/test/CodeGen/RISCV/rvp-simd-32.ll
index e947cebb90a91..bdeef32a84cb9 100644
--- a/llvm/test/CodeGen/RISCV/rvp-simd-32.ll
+++ b/llvm/test/CodeGen/RISCV/rvp-simd-32.ll
@@ -4184,3 +4184,69 @@ define <2 x i16> @test_pusati_u16x2_max_width(<2 x i16> %a) {
%res = call <2 x i16> @llvm.riscv.pusati.v2i16.i32(<2 x i16> %a, i32 15)
ret <2 x i16> %res
}
+
+define <4 x i8> @test_pslide1up_v4i8(<4 x i8> %rd, i8 %rs1) {
+; RV32-LABEL: test_pslide1up_v4i8:
+; RV32: # %bb.0:
+; RV32-NEXT: slli a1, a1, 24
+; RV32-NEXT: li a2, 8
+; RV32-NEXT: slx a0, a1, a2
+; RV32-NEXT: ret
+;
+; RV64-LABEL: test_pslide1up_v4i8:
+; RV64: # %bb.0:
+; RV64-NEXT: slli a1, a1, 56
+; RV64-NEXT: li a2, 8
+; RV64-NEXT: slx a0, a1, a2
+; RV64-NEXT: ret
+ %scalar = insertelement <4 x i8> poison, i8 %rs1, i64 0
+ %res = shufflevector <4 x i8> %rd, <4 x i8> %scalar, <4 x i32> <i32 4, i32 0, i32 1, i32 2>
+ ret <4 x i8> %res
+}
+
+define <4 x i8> @test_pslide1down_v4i8(<4 x i8> %rd, i8 %rs1) {
+; RV32-LABEL: test_pslide1down_v4i8:
+; RV32: # %bb.0:
+; RV32-NEXT: li a2, 8
+; RV32-NEXT: srx a0, a1, a2
+; RV32-NEXT: ret
+;
+; RV64-LABEL: test_pslide1down_v4i8:
+; RV64: # %bb.0:
+; RV64-NEXT: pack a0, a0, a1
+; RV64-NEXT: srli a0, a0, 8
+; RV64-NEXT: ret
+ %scalar = insertelement <4 x i8> poison, i8 %rs1, i64 0
+ %res = shufflevector <4 x i8> %rd, <4 x i8> %scalar, <4 x i32> <i32 1, i32 2, i32 3, i32 4>
+ ret <4 x i8> %res
+}
+
+define <2 x i16> @test_pslide1up_v2i16(<2 x i16> %rd, i16 %rs1) {
+; RV32-LABEL: test_pslide1up_v2i16:
+; RV32: # %bb.0:
+; RV32-NEXT: pack a0, a1, a0
+; RV32-NEXT: ret
+;
+; RV64-LABEL: test_pslide1up_v2i16:
+; RV64: # %bb.0:
+; RV64-NEXT: ppaire.h a0, a1, a0
+; RV64-NEXT: ret
+ %scalar = insertelement <2 x i16> poison, i16 %rs1, i64 0
+ %res = shufflevector <2 x i16> %rd, <2 x i16> %scalar, <2 x i32> <i32 2, i32 0>
+ ret <2 x i16> %res
+}
+
+define <2 x i16> @test_pslide1down_v2i16(<2 x i16> %rd, i16 %rs1) {
+; RV32-LABEL: test_pslide1down_v2i16:
+; RV32: # %bb.0:
+; RV32-NEXT: ppairoe.h a0, a0, a1
+; RV32-NEXT: ret
+;
+; RV64-LABEL: test_pslide1down_v2i16:
+; RV64: # %bb.0:
+; RV64-NEXT: ppairoe.h a0, a0, a1
+; RV64-NEXT: ret
+ %scalar = insertelement <2 x i16> poison, i16 %rs1, i64 0
+ %res = shufflevector <2 x i16> %rd, <2 x i16> %scalar, <2 x i32> <i32 1, i32 2>
+ ret <2 x i16> %res
+}
diff --git a/llvm/test/CodeGen/RISCV/rvp-simd-64.ll b/llvm/test/CodeGen/RISCV/rvp-simd-64.ll
index 3ba711d2082e0..b930302b8bfe2 100644
--- a/llvm/test/CodeGen/RISCV/rvp-simd-64.ll
+++ b/llvm/test/CodeGen/RISCV/rvp-simd-64.ll
@@ -8650,3 +8650,111 @@ define <2 x i32> @test_pusati_u32x2_max_width(<2 x i32> %a) {
%res = call <2 x i32> @llvm.riscv.pusati.v2i32.i32(<2 x i32> %a, i32 31)
ret <2 x i32> %res
}
+
+define <8 x i8> @test_pslide1up_v8i8(<8 x i8> %rd, i8 %rs1) {
+; RV32-LABEL: test_pslide1up_v8i8:
+; RV32: # %bb.0:
+; RV32-NEXT: li a3, 8
+; RV32-NEXT: slli a2, a2, 24
+; RV32-NEXT: slx a1, a0, a3
+; RV32-NEXT: slx a0, a2, a3
+; RV32-NEXT: ret
+;
+; RV64-LABEL: test_pslide1up_v8i8:
+; RV64: # %bb.0:
+; RV64-NEXT: slli a1, a1, 56
+; RV64-NEXT: li a2, 8
+; RV64-NEXT: slx a0, a1, a2
+; RV64-NEXT: ret
+ %scalar = insertelement <8 x i8> poison, i8 %rs1, i64 0
+ %res = shufflevector <8 x i8> %rd, <8 x i8> %scalar, <8 x i32> <i32 8, i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6>
+ ret <8 x i8> %res
+}
+
+define <8 x i8> @test_pslide1down_v8i8(<8 x i8> %rd, i8 %rs1) {
+; RV32-LABEL: test_pslide1down_v8i8:
+; RV32: # %bb.0:
+; RV32-NEXT: li a3, 8
+; RV32-NEXT: srx a0, a1, a3
+; RV32-NEXT: srx a1, a2, a3
+; RV32-NEXT: ret
+;
+; RV64-LABEL: test_pslide1down_v8i8:
+; RV64: # %bb.0:
+; RV64-NEXT: li a2, 8
+; RV64-NEXT: srx a0, a1, a2
+; RV64-NEXT: ret
+ %scalar = insertelement <8 x i8> poison, i8 %rs1, i64 0
+ %res = shufflevector <8 x i8> %rd, <8 x i8> %scalar, <8 x i32> <i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8>
+ ret <8 x i8> %res
+}
+
+define <4 x i16> @test_pslide1up_v4i16(<4 x i16> %rd, i16 %rs1) {
+; RV32-LABEL: test_pslide1up_v4i16:
+; RV32: # %bb.0:
+; RV32-NEXT: li a3, 16
+; RV32-NEXT: slli a2, a2, 16
+; RV32-NEXT: slx a1, a0, a3
+; RV32-NEXT: slx a0, a2, a3
+; RV32-NEXT: ret
+;
+; RV64-LABEL: test_pslide1up_v4i16:
+; RV64: # %bb.0:
+; RV64-NEXT: slli a1, a1, 48
+; RV64-NEXT: li a2, 16
+; RV64-NEXT: slx a0, a1, a2
+; RV64-NEXT: ret
+ %scalar = insertelement <4 x i16> poison, i16 %rs1, i64 0
+ %res = shufflevector <4 x i16> %rd, <4 x i16> %scalar, <4 x i32> <i32 4, i32 0, i32 1, i32 2>
+ ret <4 x i16> %res
+}
+
+define <4 x i16> @test_pslide1down_v4i16(<4 x i16> %rd, i16 %rs1) {
+; RV32-LABEL: test_pslide1down_v4i16:
+; RV32: # %bb.0:
+; RV32-NEXT: li a3, 16
+; RV32-NEXT: srx a0, a1, a3
+; RV32-NEXT: srx a1, a2, a3
+; RV32-NEXT: ret
+;
+; RV64-LABEL: test_pslide1down_v4i16:
+; RV64: # %bb.0:
+; RV64-NEXT: li a2, 16
+; RV64-NEXT: srx a0, a1, a2
+; RV64-NEXT: ret
+ %scalar = insertelement <4 x i16> poison, i16 %rs1, i64 0
+ %res = shufflevector <4 x i16> %rd, <4 x i16> %scalar, <4 x i32> <i32 1, i32 2, i32 3, i32 4>
+ ret <4 x i16> %res
+}
+
+define <2 x i32> @test_pslide1up_v2i32(<2 x i32> %rd, i32 %rs1) {
+; RV32-LABEL: test_pslide1up_v2i32:
+; RV32: # %bb.0:
+; RV32-NEXT: mv a1, a0
+; RV32-NEXT: mv a0, a2
+; RV32-NEXT: ret
+;
+; RV64-LABEL: test_pslide1up_v2i32:
+; RV64: # %bb.0:
+; RV64-NEXT: pack a0, a1, a0
+; RV64-NEXT: ret
+ %scalar = insertelement <2 x i32> poison, i32 %rs1, i64 0
+ %res = shufflevector <2 x i32> %rd, <2 x i32> %scalar, <2 x i32> <i32 2, i32 0>
+ ret <2 x i32> %res
+}
+
+define <2 x i32> @test_pslide1down_v2i32(<2 x i32> %rd, i32 %rs1) {
+; RV32-LABEL: test_pslide1down_v2i32:
+; RV32: # %bb.0:
+; RV32-NEXT: mv a0, a1
+; RV32-NEXT: mv a1, a2
+; RV32-NEXT: ret
+;
+; RV64-LABEL: test_pslide1down_v2i32:
+; RV64: # %bb.0:
+; RV64-NEXT: ppairoe.w a0, a0, a1
+; RV64-NEXT: ret
+ %scalar = insertelement <2 x i32> poison, i32 %rs1, i64 0
+ %res = shufflevector <2 x i32> %rd, <2 x i32> %scalar, <2 x i32> <i32 1, i32 2>
+ ret <2 x i32> %res
+}
More information about the cfe-commits
mailing list