[llvm] bc9e7ae - [SelectionDAG] fold interleave of contiguous splats (#224318)
via llvm-commits
llvm-commits at lists.llvm.org
Mon Sep 21 06:46:04 PDT 2026
Author: Kamlesh Kumar
Date: 2026-09-21T14:45:55+01:00
New Revision: bc9e7ae8e828ffa5f59f700a8a0d5fb54e0f6ba9
URL: https://github.com/llvm/llvm-project/commit/bc9e7ae8e828ffa5f59f700a8a0d5fb54e0f6ba9
DIFF: https://github.com/llvm/llvm-project/commit/bc9e7ae8e828ffa5f59f700a8a0d5fb54e0f6ba9.diff
LOG: [SelectionDAG] fold interleave of contiguous splats (#224318)
When operands of vector_interleave is splats and each splat's is from a
contiguous source , interleave operations can be folded to shuffles of
source of splat.
i.e.
interleave(splat(S[J]), splat(S[J+1]), splat(S[J+2]), ..., splat(S[J +
Factor - 1]))
can be folded to
shuffles (S[J], S[J+1], S[J+Factor -1], S[J]....)
Added:
llvm/test/CodeGen/AArch64/vector-interleave-consecutive-splats.ll
Modified:
llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp
Removed:
################################################################################
diff --git a/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp b/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp
index 771e4a99111c8..17e2ab01bc11f 100644
--- a/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp
@@ -27895,6 +27895,7 @@ SDValue DAGCombiner::visitCONCAT_VECTORS(SDNode *N) {
SDValue DAGCombiner::visitVECTOR_INTERLEAVE(SDNode *N) {
EVT VT = N->getValueType(0);
SDValue Op0 = N->getOperand(0);
+ unsigned Factor = N->getNumOperands();
// Fold an interleave of fixed-length BUILD_VECTORs by rearranging their
// scalar operands directly.
@@ -27904,7 +27905,6 @@ SDValue DAGCombiner::visitVECTOR_INTERLEAVE(SDNode *N) {
return Op.getOpcode() == ISD::BUILD_VECTOR &&
Op.getOperand(0).getValueType() == EltVT;
})) {
- unsigned Factor = N->getNumOperands();
unsigned NumElts = VT.getVectorNumElements();
SDLoc DL(N);
SmallVector<SDValue, 4> Results;
@@ -27920,6 +27920,28 @@ SDValue DAGCombiner::visitVECTOR_INTERLEAVE(SDNode *N) {
}
}
+ // Fold interleave(splat(S[J]), ..., splat(S[J + Factor - 1])) to shuffles
+ // of S.
+ if (Op0.getOpcode() == ISD::VECTOR_SHUFFLE &&
+ VT.getVectorElementCount().isKnownMultipleOf(Factor)) {
+ int FirstIndex;
+ unsigned NumElts = VT.getVectorNumElements();
+ SDValue Source = DAG.getSplatSourceVector(Op0, FirstIndex);
+ if (Source && llvm::all_of(llvm::enumerate(N->op_values()), [&](auto Item) {
+ int SplatIndex;
+ return DAG.getSplatSourceVector(Item.value(), SplatIndex) == Source &&
+ SplatIndex == FirstIndex + static_cast<int>(Item.index());
+ })) {
+ SmallVector<int, 16> Mask;
+ for (unsigned I = 0; I != NumElts; ++I)
+ Mask.push_back(FirstIndex + I % Factor);
+ SDValue Shuffle =
+ DAG.getVectorShuffle(VT, SDLoc(N), Source, DAG.getPOISON(VT), Mask);
+ SmallVector<SDValue, 4> Results(Factor, Shuffle);
+ return CombineTo(N, &Results);
+ }
+ }
+
// Check to see if all operands are identical.
if (!llvm::all_equal(N->op_values()))
return SDValue();
diff --git a/llvm/test/CodeGen/AArch64/vector-interleave-consecutive-splats.ll b/llvm/test/CodeGen/AArch64/vector-interleave-consecutive-splats.ll
new file mode 100644
index 0000000000000..c768baf838ff5
--- /dev/null
+++ b/llvm/test/CodeGen/AArch64/vector-interleave-consecutive-splats.ll
@@ -0,0 +1,102 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 5
+; RUN: llc -mtriple=aarch64-none-linux-gnu < %s | FileCheck %s
+
+define void @interleave4_consecutive_splats(<4 x i16> %src, ptr %dst0, ptr %dst1) {
+; CHECK-LABEL: interleave4_consecutive_splats:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $d0 killed $d0 def $q0
+; CHECK-NEXT: mov v0.d[1], v0.d[0]
+; CHECK-NEXT: stp q0, q0, [x0]
+; CHECK-NEXT: stp q0, q0, [x0, #32]
+; CHECK-NEXT: stp q0, q0, [x1]
+; CHECK-NEXT: stp q0, q0, [x1, #32]
+; CHECK-NEXT: ret
+ %splat2 = shufflevector <4 x i16> %src, <4 x i16> poison, <8 x i32> splat (i32 2)
+ %splat1 = shufflevector <4 x i16> %src, <4 x i16> poison, <8 x i32> splat (i32 1)
+ %splat0 = shufflevector <4 x i16> %src, <4 x i16> poison, <8 x i32> zeroinitializer
+ %splat3 = shufflevector <4 x i16> %src, <4 x i16> poison, <8 x i32> splat (i32 3)
+ %interleave = tail call <32 x i16> @llvm.vector.interleave4.v32i16(
+ <8 x i16> %splat0, <8 x i16> %splat1, <8 x i16> %splat2,
+ <8 x i16> %splat3)
+ store <32 x i16> %interleave, ptr %dst0
+ store <32 x i16> %interleave, ptr %dst1
+ ret void
+}
+
+define void @interleave3_consecutive_splats_nonzero(<8 x i16> %src, ptr %dst0, ptr %dst1) {
+; CHECK-LABEL: interleave3_consecutive_splats_nonzero:
+; CHECK: // %bb.0:
+; CHECK-NEXT: sub sp, sp, #48
+; CHECK-NEXT: .cfi_def_cfa_offset 48
+; CHECK-NEXT: dup v1.8h, v0.h[2]
+; CHECK-NEXT: mov x8, sp
+; CHECK-NEXT: dup v2.8h, v0.h[3]
+; CHECK-NEXT: dup v3.8h, v0.h[4]
+; CHECK-NEXT: st3 { v1.8h, v2.8h, v3.8h }, [x8]
+; CHECK-NEXT: ldp q0, q1, [sp]
+; CHECK-NEXT: ldr q2, [sp, #32]
+; CHECK-NEXT: stp q0, q1, [x0]
+; CHECK-NEXT: str q2, [x0, #32]
+; CHECK-NEXT: stp q0, q1, [x1]
+; CHECK-NEXT: str q2, [x1, #32]
+; CHECK-NEXT: add sp, sp, #48
+; CHECK-NEXT: ret
+ %splat2 = shufflevector <8 x i16> %src, <8 x i16> poison, <8 x i32> splat (i32 2)
+ %splat3 = shufflevector <8 x i16> %src, <8 x i16> poison, <8 x i32> splat (i32 3)
+ %splat4 = shufflevector <8 x i16> %src, <8 x i16> poison, <8 x i32> splat (i32 4)
+
+ %interleave = tail call <24 x i16> @llvm.vector.interleave3.v24i16(
+ <8 x i16> %splat2, <8 x i16> %splat3,<8 x i16> %splat4)
+ store <24 x i16> %interleave, ptr %dst0
+ store <24 x i16> %interleave, ptr %dst1
+ ret void
+}
+
+define void @interleave2_consecutive_splats_nonzero(<8 x i16> %src, ptr %dst0, ptr %dst1) {
+; CHECK-LABEL: interleave2_consecutive_splats_nonzero:
+; CHECK: // %bb.0:
+; CHECK-NEXT: dup v0.4s, v0.s[1]
+; CHECK-NEXT: stp q0, q0, [x0]
+; CHECK-NEXT: stp q0, q0, [x1]
+; CHECK-NEXT: ret
+ %splat2 = shufflevector <8 x i16> %src, <8 x i16> poison, <8 x i32> splat (i32 2)
+ %splat3 = shufflevector <8 x i16> %src, <8 x i16> poison, <8 x i32> splat (i32 3)
+ %interleave = tail call <16 x i16> @llvm.vector.interleave2.v16i16(
+ <8 x i16> %splat2, <8 x i16> %splat3)
+ store <16 x i16> %interleave, ptr %dst0
+ store <16 x i16> %interleave, ptr %dst1
+ ret void
+}
+
+define void @interleave4_nonshuffle_operand(<8 x i16> %src, ptr %dst0, ptr %dst1) {
+; CHECK-LABEL: interleave4_nonshuffle_operand:
+; CHECK: // %bb.0:
+; CHECK-NEXT: movi v1.8h, #1
+; CHECK-NEXT: dup v2.8h, v0.h[0]
+; CHECK-NEXT: dup v3.8h, v0.h[2]
+; CHECK-NEXT: dup v4.8h, v0.h[3]
+; CHECK-NEXT: add v0.8h, v0.8h, v1.8h
+; CHECK-NEXT: zip2 v1.8h, v2.8h, v3.8h
+; CHECK-NEXT: zip1 v2.8h, v2.8h, v3.8h
+; CHECK-NEXT: zip2 v5.8h, v0.8h, v4.8h
+; CHECK-NEXT: zip1 v0.8h, v0.8h, v4.8h
+; CHECK-NEXT: zip1 v3.8h, v1.8h, v5.8h
+; CHECK-NEXT: zip2 v1.8h, v1.8h, v5.8h
+; CHECK-NEXT: zip1 v4.8h, v2.8h, v0.8h
+; CHECK-NEXT: zip2 v0.8h, v2.8h, v0.8h
+; CHECK-NEXT: stp q3, q1, [x0, #32]
+; CHECK-NEXT: stp q4, q0, [x0]
+; CHECK-NEXT: stp q4, q0, [x1]
+; CHECK-NEXT: stp q3, q1, [x1, #32]
+; CHECK-NEXT: ret
+ %splat0 = shufflevector <8 x i16> %src, <8 x i16> poison, <8 x i32> zeroinitializer
+ %added = add <8 x i16> %src, splat (i16 1)
+ %splat2 = shufflevector <8 x i16> %src, <8 x i16> poison, <8 x i32> splat (i32 2)
+ %splat3 = shufflevector <8 x i16> %src, <8 x i16> poison, <8 x i32> splat (i32 3)
+ %interleave = call <32 x i16> @llvm.vector.interleave4.v32i16(
+ <8 x i16> %splat0, <8 x i16> %added, <8 x i16> %splat2,
+ <8 x i16> %splat3)
+ store <32 x i16> %interleave, ptr %dst0
+ store <32 x i16> %interleave, ptr %dst1
+ ret void
+}
More information about the llvm-commits
mailing list