[llvm] bc9e7ae - [SelectionDAG] fold interleave of contiguous splats (#224318)

via llvm-commits llvm-commits at lists.llvm.org
Mon Sep 21 06:46:04 PDT 2026


Author: Kamlesh Kumar
Date: 2026-09-21T14:45:55+01:00
New Revision: bc9e7ae8e828ffa5f59f700a8a0d5fb54e0f6ba9

URL: https://github.com/llvm/llvm-project/commit/bc9e7ae8e828ffa5f59f700a8a0d5fb54e0f6ba9
DIFF: https://github.com/llvm/llvm-project/commit/bc9e7ae8e828ffa5f59f700a8a0d5fb54e0f6ba9.diff

LOG: [SelectionDAG] fold interleave of contiguous splats (#224318)

When operands of vector_interleave is splats and each splat's is from a
contiguous source , interleave operations can be folded to shuffles of
source of splat.
i.e.
interleave(splat(S[J]), splat(S[J+1]), splat(S[J+2]), ..., splat(S[J +
Factor - 1]))
can be folded to
shuffles (S[J], S[J+1], S[J+Factor -1], S[J]....)

Added: 
    llvm/test/CodeGen/AArch64/vector-interleave-consecutive-splats.ll

Modified: 
    llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp

Removed: 
    


################################################################################
diff  --git a/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp b/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp
index 771e4a99111c8..17e2ab01bc11f 100644
--- a/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp
@@ -27895,6 +27895,7 @@ SDValue DAGCombiner::visitCONCAT_VECTORS(SDNode *N) {
 SDValue DAGCombiner::visitVECTOR_INTERLEAVE(SDNode *N) {
   EVT VT = N->getValueType(0);
   SDValue Op0 = N->getOperand(0);
+  unsigned Factor = N->getNumOperands();
 
   // Fold an interleave of fixed-length BUILD_VECTORs by rearranging their
   // scalar operands directly.
@@ -27904,7 +27905,6 @@ SDValue DAGCombiner::visitVECTOR_INTERLEAVE(SDNode *N) {
           return Op.getOpcode() == ISD::BUILD_VECTOR &&
                  Op.getOperand(0).getValueType() == EltVT;
         })) {
-      unsigned Factor = N->getNumOperands();
       unsigned NumElts = VT.getVectorNumElements();
       SDLoc DL(N);
       SmallVector<SDValue, 4> Results;
@@ -27920,6 +27920,28 @@ SDValue DAGCombiner::visitVECTOR_INTERLEAVE(SDNode *N) {
     }
   }
 
+  // Fold interleave(splat(S[J]), ..., splat(S[J + Factor - 1])) to shuffles
+  // of S.
+  if (Op0.getOpcode() == ISD::VECTOR_SHUFFLE &&
+      VT.getVectorElementCount().isKnownMultipleOf(Factor)) {
+    int FirstIndex;
+    unsigned NumElts = VT.getVectorNumElements();
+    SDValue Source = DAG.getSplatSourceVector(Op0, FirstIndex);
+    if (Source && llvm::all_of(llvm::enumerate(N->op_values()), [&](auto Item) {
+          int SplatIndex;
+          return DAG.getSplatSourceVector(Item.value(), SplatIndex) == Source &&
+                 SplatIndex == FirstIndex + static_cast<int>(Item.index());
+        })) {
+      SmallVector<int, 16> Mask;
+      for (unsigned I = 0; I != NumElts; ++I)
+        Mask.push_back(FirstIndex + I % Factor);
+      SDValue Shuffle =
+          DAG.getVectorShuffle(VT, SDLoc(N), Source, DAG.getPOISON(VT), Mask);
+      SmallVector<SDValue, 4> Results(Factor, Shuffle);
+      return CombineTo(N, &Results);
+    }
+  }
+
   // Check to see if all operands are identical.
   if (!llvm::all_equal(N->op_values()))
     return SDValue();

diff  --git a/llvm/test/CodeGen/AArch64/vector-interleave-consecutive-splats.ll b/llvm/test/CodeGen/AArch64/vector-interleave-consecutive-splats.ll
new file mode 100644
index 0000000000000..c768baf838ff5
--- /dev/null
+++ b/llvm/test/CodeGen/AArch64/vector-interleave-consecutive-splats.ll
@@ -0,0 +1,102 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 5
+; RUN: llc -mtriple=aarch64-none-linux-gnu < %s | FileCheck %s
+
+define void @interleave4_consecutive_splats(<4 x i16> %src, ptr %dst0, ptr %dst1) {
+; CHECK-LABEL: interleave4_consecutive_splats:
+; CHECK:       // %bb.0:
+; CHECK-NEXT:    // kill: def $d0 killed $d0 def $q0
+; CHECK-NEXT:    mov v0.d[1], v0.d[0]
+; CHECK-NEXT:    stp q0, q0, [x0]
+; CHECK-NEXT:    stp q0, q0, [x0, #32]
+; CHECK-NEXT:    stp q0, q0, [x1]
+; CHECK-NEXT:    stp q0, q0, [x1, #32]
+; CHECK-NEXT:    ret
+  %splat2 = shufflevector <4 x i16> %src, <4 x i16> poison, <8 x i32> splat (i32 2)
+  %splat1 = shufflevector <4 x i16> %src, <4 x i16> poison, <8 x i32> splat (i32 1)
+  %splat0 = shufflevector <4 x i16> %src, <4 x i16> poison, <8 x i32> zeroinitializer
+  %splat3 = shufflevector <4 x i16> %src, <4 x i16> poison, <8 x i32> splat (i32 3)
+  %interleave = tail call <32 x i16> @llvm.vector.interleave4.v32i16(
+      <8 x i16> %splat0, <8 x i16> %splat1, <8 x i16> %splat2,
+      <8 x i16> %splat3)
+  store <32 x i16> %interleave, ptr %dst0
+  store <32 x i16> %interleave, ptr %dst1
+  ret void
+}
+
+define void @interleave3_consecutive_splats_nonzero(<8 x i16> %src, ptr %dst0, ptr %dst1) {
+; CHECK-LABEL: interleave3_consecutive_splats_nonzero:
+; CHECK:       // %bb.0:
+; CHECK-NEXT:    sub sp, sp, #48
+; CHECK-NEXT:    .cfi_def_cfa_offset 48
+; CHECK-NEXT:    dup v1.8h, v0.h[2]
+; CHECK-NEXT:    mov x8, sp
+; CHECK-NEXT:    dup v2.8h, v0.h[3]
+; CHECK-NEXT:    dup v3.8h, v0.h[4]
+; CHECK-NEXT:    st3 { v1.8h, v2.8h, v3.8h }, [x8]
+; CHECK-NEXT:    ldp q0, q1, [sp]
+; CHECK-NEXT:    ldr q2, [sp, #32]
+; CHECK-NEXT:    stp q0, q1, [x0]
+; CHECK-NEXT:    str q2, [x0, #32]
+; CHECK-NEXT:    stp q0, q1, [x1]
+; CHECK-NEXT:    str q2, [x1, #32]
+; CHECK-NEXT:    add sp, sp, #48
+; CHECK-NEXT:    ret
+  %splat2 = shufflevector <8 x i16> %src, <8 x i16> poison, <8 x i32> splat (i32 2)
+  %splat3 = shufflevector <8 x i16> %src, <8 x i16> poison, <8 x i32> splat (i32 3)
+  %splat4 = shufflevector <8 x i16> %src, <8 x i16> poison, <8 x i32> splat (i32 4)
+
+  %interleave = tail call <24 x i16> @llvm.vector.interleave3.v24i16(
+      <8 x i16> %splat2, <8 x i16> %splat3,<8 x i16> %splat4)
+  store <24 x i16> %interleave, ptr %dst0
+  store <24 x i16> %interleave, ptr %dst1
+  ret void
+}
+
+define void @interleave2_consecutive_splats_nonzero(<8 x i16> %src, ptr %dst0, ptr %dst1) {
+; CHECK-LABEL: interleave2_consecutive_splats_nonzero:
+; CHECK:       // %bb.0:
+; CHECK-NEXT:    dup v0.4s, v0.s[1]
+; CHECK-NEXT:    stp q0, q0, [x0]
+; CHECK-NEXT:    stp q0, q0, [x1]
+; CHECK-NEXT:    ret
+  %splat2 = shufflevector <8 x i16> %src, <8 x i16> poison, <8 x i32> splat (i32 2)
+  %splat3 = shufflevector <8 x i16> %src, <8 x i16> poison, <8 x i32> splat (i32 3)
+  %interleave = tail call <16 x i16> @llvm.vector.interleave2.v16i16(
+      <8 x i16> %splat2, <8 x i16> %splat3)
+  store <16 x i16> %interleave, ptr %dst0
+  store <16 x i16> %interleave, ptr %dst1
+  ret void
+}
+
+define void @interleave4_nonshuffle_operand(<8 x i16> %src, ptr %dst0, ptr %dst1) {
+; CHECK-LABEL: interleave4_nonshuffle_operand:
+; CHECK:       // %bb.0:
+; CHECK-NEXT:    movi v1.8h, #1
+; CHECK-NEXT:    dup v2.8h, v0.h[0]
+; CHECK-NEXT:    dup v3.8h, v0.h[2]
+; CHECK-NEXT:    dup v4.8h, v0.h[3]
+; CHECK-NEXT:    add v0.8h, v0.8h, v1.8h
+; CHECK-NEXT:    zip2 v1.8h, v2.8h, v3.8h
+; CHECK-NEXT:    zip1 v2.8h, v2.8h, v3.8h
+; CHECK-NEXT:    zip2 v5.8h, v0.8h, v4.8h
+; CHECK-NEXT:    zip1 v0.8h, v0.8h, v4.8h
+; CHECK-NEXT:    zip1 v3.8h, v1.8h, v5.8h
+; CHECK-NEXT:    zip2 v1.8h, v1.8h, v5.8h
+; CHECK-NEXT:    zip1 v4.8h, v2.8h, v0.8h
+; CHECK-NEXT:    zip2 v0.8h, v2.8h, v0.8h
+; CHECK-NEXT:    stp q3, q1, [x0, #32]
+; CHECK-NEXT:    stp q4, q0, [x0]
+; CHECK-NEXT:    stp q4, q0, [x1]
+; CHECK-NEXT:    stp q3, q1, [x1, #32]
+; CHECK-NEXT:    ret
+  %splat0 = shufflevector <8 x i16> %src, <8 x i16> poison, <8 x i32> zeroinitializer
+  %added = add <8 x i16> %src, splat (i16 1)
+  %splat2 = shufflevector <8 x i16> %src, <8 x i16> poison, <8 x i32> splat (i32 2)
+  %splat3 = shufflevector <8 x i16> %src, <8 x i16> poison, <8 x i32> splat (i32 3)
+  %interleave = call <32 x i16> @llvm.vector.interleave4.v32i16(
+      <8 x i16> %splat0, <8 x i16> %added, <8 x i16> %splat2,
+      <8 x i16> %splat3)
+  store <32 x i16> %interleave, ptr %dst0
+  store <32 x i16> %interleave, ptr %dst1
+  ret void
+}


        


More information about the llvm-commits mailing list