[llvm-branch-commits] [llvm] [CGP] Extend vector lsr to support interleaved/unrolled loops (PR #226201)

Graham Hunter via llvm-branch-commits llvm-branch-commits at lists.llvm.org
Fri Sep 25 07:06:39 PDT 2026


https://github.com/huntergr-arm updated https://github.com/llvm/llvm-project/pull/226201

>From ea9401e8f27bdeef2f762712a9571e9be3e3c669 Mon Sep 17 00:00:00 2001
From: Graham Hunter <graham.hunter at arm.com>
Date: Thu, 24 Sep 2026 14:48:50 +0000
Subject: [PATCH 1/2] Implement support for unrolled vector case

---
 llvm/lib/CodeGen/CodeGenPrepare.cpp           | 30 ++++++++++-
 .../AArch64/strength-reduce-vector-as-data.ll | 53 +++++++++----------
 2 files changed, 53 insertions(+), 30 deletions(-)

diff --git a/llvm/lib/CodeGen/CodeGenPrepare.cpp b/llvm/lib/CodeGen/CodeGenPrepare.cpp
index 29983cfbef024..885497b75fc47 100644
--- a/llvm/lib/CodeGen/CodeGenPrepare.cpp
+++ b/llvm/lib/CodeGen/CodeGenPrepare.cpp
@@ -8988,8 +8988,30 @@ static bool strengthReduceVectorPhiUsers(PHINode *Phi, LoopInfo *LI) {
   if (!match(Step, m_c_Add(m_Specific(Phi), m_Value(LoopStride))))
     return false;
 
+  unsigned AddCount = 0;
+  Value *ShiftedVScale = LoopStride;
+  const int DepthLimit = 4;
+  // We're looking for updates by a multiple of the number of elements in
+  // the vector.
+  // TODO: Support cases where the total stride is created directly by a
+  //       shifted vscale where we have interleaving.
+  if (match(LoopStride, m_c_Add(m_Value(LoopStride),
+                                m_Value(ShiftedVScale, m_Splat(m_Value())))))
+    for (int I = 0; I < DepthLimit; ++I) {
+      AddCount++;
+      if (!match(LoopStride,
+                 m_c_Add(m_Value(LoopStride), m_Specific(ShiftedVScale))))
+        break;
+    }
+
   const APInt *ShiftAmt = nullptr;
-  if (!match(LoopStride, m_Splat(m_Shl(m_VScale(), m_APInt(ShiftAmt)))))
+  if (!match(LoopStride, m_Splat(m_Shl(m_VScale(), m_APInt(ShiftAmt)))) ||
+      LoopStride != ShiftedVScale)
+    return false;
+
+  // Make sure the shift amount matches the minimum element count.
+  auto EltCnt = cast<VectorType>(Phi->getType())->getElementCount();
+  if (1 << ShiftAmt->getZExtValue() != EltCnt.getKnownMinValue())
     return false;
 
   // Record users of interest.
@@ -9057,11 +9079,15 @@ static bool strengthReduceVectorPhiUsers(PHINode *Phi, LoopInfo *LI) {
   // iteration.
   Value *StructStride = ConstantInt::get(ITy, ShiftAmt->getZExtValue());
   StructStride = PHBuilder.CreateMul(StructStride, PHBuilder.CreateVScale(ITy));
+  Value *TotalStride =
+      PHBuilder.CreateMul(StructStride, ConstantInt::get(ITy, AddCount + 1));
   StructStride =
       PHBuilder.CreateVectorSplat(VTy->getElementCount(), StructStride);
+  TotalStride =
+      PHBuilder.CreateVectorSplat(VTy->getElementCount(), TotalStride);
 
   IRBuilder<> LBuilder(cast<Instruction>(Step));
-  Value *NewStep = LBuilder.CreateAdd(Phi, StructStride);
+  Value *NewStep = LBuilder.CreateAdd(Phi, TotalStride);
 
   // Update the phi to the new start and step.
   Phi->setIncomingValueForBlock(PreHeader, NewStart);
diff --git a/llvm/test/CodeGen/AArch64/strength-reduce-vector-as-data.ll b/llvm/test/CodeGen/AArch64/strength-reduce-vector-as-data.ll
index e49c276ba38dd..dcad17b887347 100644
--- a/llvm/test/CodeGen/AArch64/strength-reduce-vector-as-data.ll
+++ b/llvm/test/CodeGen/AArch64/strength-reduce-vector-as-data.ll
@@ -51,38 +51,35 @@ middle.block:
 define void @init_array_of_ptrs_to_structs_interleave4(ptr noalias %arc_ptrs, ptr %arc_new, i64 %num_arcs) #0 {
 ; CHECK-LABEL: init_array_of_ptrs_to_structs_interleave4:
 ; CHECK:       // %bb.0: // %entry
-; CHECK-NEXT:    cntd x8
-; CHECK-NEXT:    mov w9, #2147483647 // =0x7fffffff
-; CHECK-NEXT:    cnth x10
-; CHECK-NEXT:    mov z0.d, x8
-; CHECK-NEXT:    cntw x8
-; CHECK-NEXT:    inch x9
-; CHECK-NEXT:    mov z1.d, x8
-; CHECK-NEXT:    cntd x8, all, mul #3
-; CHECK-NEXT:    index z2.d, #0, #1
-; CHECK-NEXT:    mov z3.d, x8
-; CHECK-NEXT:    mov z4.d, x10
-; CHECK-NEXT:    mov z5.d, x1
-; CHECK-NEXT:    mov z6.d, #72 // =0x48
-; CHECK-NEXT:    and x8, x9, x2
-; CHECK-NEXT:    ptrue p0.d
+; CHECK-NEXT:    rdvl x8, #1
+; CHECK-NEXT:    cntd x9, all, mul #3
+; CHECK-NEXT:    cntw x10
+; CHECK-NEXT:    lsr x8, x8, #4
+; CHECK-NEXT:    cntd x11
+; CHECK-NEXT:    mov w13, #2147483647 // =0x7fffffff
+; CHECK-NEXT:    inch x13
+; CHECK-NEXT:    mov z0.d, x10
+; CHECK-NEXT:    umull x9, w9, w8
+; CHECK-NEXT:    umull x12, w10, w8
+; CHECK-NEXT:    umull x8, w11, w8
+; CHECK-NEXT:    mov w11, #72 // =0x48
+; CHECK-NEXT:    index z1.d, x1, x11
+; CHECK-NEXT:    mov z2.d, x9
+; CHECK-NEXT:    mov z3.d, x12
+; CHECK-NEXT:    mov z4.d, x8
+; CHECK-NEXT:    and x8, x13, x2
 ; CHECK-NEXT:    sub x8, x8, x2
 ; CHECK-NEXT:  .LBB1_1: // %vector.body
 ; CHECK-NEXT:    // =>This Inner Loop Header: Depth=1
-; CHECK-NEXT:    add z7.d, z2.d, z0.d
-; CHECK-NEXT:    add z16.d, z2.d, z1.d
-; CHECK-NEXT:    movprfx z18, z5
-; CHECK-NEXT:    mla z18.d, p0/m, z2.d, z6.d
-; CHECK-NEXT:    add z17.d, z2.d, z3.d
+; CHECK-NEXT:    add z5.d, z1.d, z4.d
+; CHECK-NEXT:    add z6.d, z1.d, z3.d
+; CHECK-NEXT:    str z1, [x0]
+; CHECK-NEXT:    add z7.d, z1.d, z2.d
 ; CHECK-NEXT:    inch x8
-; CHECK-NEXT:    add z2.d, z2.d, z4.d
-; CHECK-NEXT:    mad z7.d, p0/m, z6.d, z5.d
-; CHECK-NEXT:    mad z16.d, p0/m, z6.d, z5.d
-; CHECK-NEXT:    mad z17.d, p0/m, z6.d, z5.d
-; CHECK-NEXT:    str z18, [x0]
-; CHECK-NEXT:    str z7, [x0, #1, mul vl]
-; CHECK-NEXT:    str z16, [x0, #2, mul vl]
-; CHECK-NEXT:    str z17, [x0, #3, mul vl]
+; CHECK-NEXT:    add z1.d, z1.d, z0.d
+; CHECK-NEXT:    str z5, [x0, #1, mul vl]
+; CHECK-NEXT:    str z6, [x0, #2, mul vl]
+; CHECK-NEXT:    str z7, [x0, #3, mul vl]
 ; CHECK-NEXT:    incb x0, all, mul #4
 ; CHECK-NEXT:    cbnz x8, .LBB1_1
 ; CHECK-NEXT:  // %bb.2: // %exit

>From ca5624d2188e126e70a11567f1ec35756dbd0511 Mon Sep 17 00:00:00 2001
From: Graham Hunter <graham.hunter at arm.com>
Date: Fri, 25 Sep 2026 13:54:40 +0000
Subject: [PATCH 2/2] update new IR test

---
 .../AArch64/strength-reduce-vector-as-data.ll | 42 +++++++++++++++----
 1 file changed, 34 insertions(+), 8 deletions(-)

diff --git a/llvm/test/Transforms/CodeGenPrepare/AArch64/strength-reduce-vector-as-data.ll b/llvm/test/Transforms/CodeGenPrepare/AArch64/strength-reduce-vector-as-data.ll
index 1bf5a3d2d8477..dcfe56a243057 100644
--- a/llvm/test/Transforms/CodeGenPrepare/AArch64/strength-reduce-vector-as-data.ll
+++ b/llvm/test/Transforms/CodeGenPrepare/AArch64/strength-reduce-vector-as-data.ll
@@ -20,12 +20,15 @@ define void @init_array_of_ptrs_to_structs(ptr noalias %arc_ptrs, ptr %arc_new,
 ; CHECK-NEXT:    [[TMP9:%.*]] = add <vscale x 2 x i64> [[TMP7]], [[DOTSPLAT]]
 ; CHECK-NEXT:    [[TMP11:%.*]] = call i64 @llvm.vscale.i64()
 ; CHECK-NEXT:    [[TMP5:%.*]] = mul i64 1, [[TMP11]]
+; CHECK-NEXT:    [[TMP10:%.*]] = mul i64 [[TMP5]], 1
 ; CHECK-NEXT:    [[DOTSPLATINSERT1:%.*]] = insertelement <vscale x 2 x i64> poison, i64 [[TMP5]], i64 0
 ; CHECK-NEXT:    [[DOTSPLAT2:%.*]] = shufflevector <vscale x 2 x i64> [[DOTSPLATINSERT1]], <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+; CHECK-NEXT:    [[DOTSPLATINSERT3:%.*]] = insertelement <vscale x 2 x i64> poison, i64 [[TMP10]], i64 0
+; CHECK-NEXT:    [[DOTSPLAT4:%.*]] = shufflevector <vscale x 2 x i64> [[DOTSPLATINSERT3]], <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
 ; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
 ; CHECK:       [[VECTOR_BODY]]:
 ; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT:    [[VEC_IND:%.*]] = phi <vscale x 2 x i64> [ [[TMP9]], %[[ENTRY]] ], [ [[TMP10:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[VEC_IND:%.*]] = phi <vscale x 2 x i64> [ [[TMP9]], %[[ENTRY]] ], [ [[TMP12:%.*]], %[[VECTOR_BODY]] ]
 ; CHECK-NEXT:    [[TMP6:%.*]] = inttoptr <vscale x 2 x i64> [[VEC_IND]] to <vscale x 2 x ptr>
 ; CHECK-NEXT:    [[WIDE_GEP:%.*]] = getelementptr inbounds nuw [72 x i8], ptr [[ARC_NEW]], <vscale x 2 x i64> [[VEC_IND]]
 ; CHECK-NEXT:    [[TMP1:%.*]] = getelementptr inbounds nuw [8 x i8], ptr [[ARC_PTRS]], i64 [[INDEX]]
@@ -33,7 +36,7 @@ define void @init_array_of_ptrs_to_structs(ptr noalias %arc_ptrs, ptr %arc_new,
 ; CHECK-NEXT:    [[TMP2:%.*]] = tail call i64 @llvm.vscale.i64()
 ; CHECK-NEXT:    [[TMP3:%.*]] = shl nuw nsw i64 [[TMP2]], 1
 ; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP3]]
-; CHECK-NEXT:    [[TMP10]] = add <vscale x 2 x i64> [[VEC_IND]], [[DOTSPLAT2]]
+; CHECK-NEXT:    [[TMP12]] = add <vscale x 2 x i64> [[VEC_IND]], [[DOTSPLAT4]]
 ; CHECK-NEXT:    [[VEC_IND_NEXT:%.*]] = add nuw nsw <vscale x 2 x i64> [[VEC_IND]], [[BROADCAST_SPLAT]]
 ; CHECK-NEXT:    [[TMP4:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
 ; CHECK-NEXT:    br i1 [[TMP4]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
@@ -83,17 +86,39 @@ define void @init_array_of_ptrs_to_structs_interleave4(ptr noalias %arc_ptrs, pt
 ; CHECK-NEXT:    [[INVARIANT_OP:%.*]] = add nuw <vscale x 2 x i64> [[BROADCAST_SPLAT]], [[BROADCAST_SPLAT]]
 ; CHECK-NEXT:    [[INVARIANT_OP26:%.*]] = add nuw <vscale x 2 x i64> [[INVARIANT_OP]], [[BROADCAST_SPLAT]]
 ; CHECK-NEXT:    [[INVARIANT_OP27:%.*]] = add nuw <vscale x 2 x i64> [[INVARIANT_OP26]], [[BROADCAST_SPLAT]]
+; CHECK-NEXT:    [[TMP22:%.*]] = mul <vscale x 2 x i64> [[TMP5]], splat (i64 72)
+; CHECK-NEXT:    [[TMP23:%.*]] = ptrtoint ptr [[ARC_NEW]] to i64
+; CHECK-NEXT:    [[DOTSPLATINSERT:%.*]] = insertelement <vscale x 2 x i64> poison, i64 [[TMP23]], i64 0
+; CHECK-NEXT:    [[DOTSPLAT:%.*]] = shufflevector <vscale x 2 x i64> [[DOTSPLATINSERT]], <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+; CHECK-NEXT:    [[TMP24:%.*]] = add <vscale x 2 x i64> [[TMP22]], [[DOTSPLAT]]
+; CHECK-NEXT:    [[TMP25:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-NEXT:    [[TMP26:%.*]] = mul i64 1, [[TMP25]]
+; CHECK-NEXT:    [[TMP27:%.*]] = mul i64 [[TMP26]], 4
+; CHECK-NEXT:    [[DOTSPLATINSERT1:%.*]] = insertelement <vscale x 2 x i64> poison, i64 [[TMP26]], i64 0
+; CHECK-NEXT:    [[DOTSPLAT2:%.*]] = shufflevector <vscale x 2 x i64> [[DOTSPLATINSERT1]], <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+; CHECK-NEXT:    [[DOTSPLATINSERT3:%.*]] = insertelement <vscale x 2 x i64> poison, i64 [[TMP27]], i64 0
+; CHECK-NEXT:    [[DOTSPLAT4:%.*]] = shufflevector <vscale x 2 x i64> [[DOTSPLATINSERT3]], <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+; CHECK-NEXT:    [[TMP29:%.*]] = mul <vscale x 2 x i64> [[INVARIANT_OP26]], [[DOTSPLAT2]]
+; CHECK-NEXT:    [[TMP30:%.*]] = mul <vscale x 2 x i64> [[INVARIANT_OP]], [[DOTSPLAT2]]
+; CHECK-NEXT:    [[TMP31:%.*]] = mul <vscale x 2 x i64> [[BROADCAST_SPLAT]], [[DOTSPLAT2]]
 ; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
 ; CHECK:       [[VECTOR_BODY]]:
 ; CHECK-NEXT:    [[LSR_IV36:%.*]] = phi ptr [ [[SCEVGEP37:%.*]], %[[VECTOR_BODY]] ], [ [[ARC_PTRS]], %[[ENTRY]] ]
 ; CHECK-NEXT:    [[LSR_IV34:%.*]] = phi i64 [ [[LSR_IV_NEXT35:%.*]], %[[VECTOR_BODY]] ], [ [[N_VEC]], %[[ENTRY]] ]
-; CHECK-NEXT:    [[VEC_IND:%.*]] = phi <vscale x 2 x i64> [ [[TMP5]], %[[ENTRY]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[VEC_IND:%.*]] = phi <vscale x 2 x i64> [ [[TMP24]], %[[ENTRY]] ], [ [[TMP28:%.*]], %[[VECTOR_BODY]] ]
 ; CHECK-NEXT:    [[STEP_ADD:%.*]] = add nuw <vscale x 2 x i64> [[VEC_IND]], [[BROADCAST_SPLAT]]
 ; CHECK-NEXT:    [[STEP_ADD_2_REASS:%.*]] = add nuw <vscale x 2 x i64> [[VEC_IND]], [[INVARIANT_OP]]
 ; CHECK-NEXT:    [[STEP_ADD_3_REASS:%.*]] = add nuw <vscale x 2 x i64> [[VEC_IND]], [[INVARIANT_OP26]]
+; CHECK-NEXT:    [[TMP32:%.*]] = inttoptr <vscale x 2 x i64> [[VEC_IND]] to <vscale x 2 x ptr>
 ; CHECK-NEXT:    [[WIDE_GEP:%.*]] = getelementptr inbounds nuw [72 x i8], ptr [[ARC_NEW]], <vscale x 2 x i64> [[VEC_IND]]
+; CHECK-NEXT:    [[TMP33:%.*]] = add <vscale x 2 x i64> [[VEC_IND]], [[TMP31]]
+; CHECK-NEXT:    [[TMP17:%.*]] = inttoptr <vscale x 2 x i64> [[TMP33]] to <vscale x 2 x ptr>
 ; CHECK-NEXT:    [[WIDE_GEP10:%.*]] = getelementptr inbounds nuw [72 x i8], ptr [[ARC_NEW]], <vscale x 2 x i64> [[STEP_ADD]]
+; CHECK-NEXT:    [[TMP18:%.*]] = add <vscale x 2 x i64> [[VEC_IND]], [[TMP30]]
+; CHECK-NEXT:    [[TMP19:%.*]] = inttoptr <vscale x 2 x i64> [[TMP18]] to <vscale x 2 x ptr>
 ; CHECK-NEXT:    [[WIDE_GEP11:%.*]] = getelementptr inbounds nuw [72 x i8], ptr [[ARC_NEW]], <vscale x 2 x i64> [[STEP_ADD_2_REASS]]
+; CHECK-NEXT:    [[TMP20:%.*]] = add <vscale x 2 x i64> [[VEC_IND]], [[TMP29]]
+; CHECK-NEXT:    [[TMP21:%.*]] = inttoptr <vscale x 2 x i64> [[TMP20]] to <vscale x 2 x ptr>
 ; CHECK-NEXT:    [[WIDE_GEP12:%.*]] = getelementptr inbounds nuw [72 x i8], ptr [[ARC_NEW]], <vscale x 2 x i64> [[STEP_ADD_3_REASS]]
 ; CHECK-NEXT:    [[TMP6:%.*]] = tail call i64 @llvm.vscale.i64()
 ; CHECK-NEXT:    [[TMP7:%.*]] = shl i64 [[TMP6]], 4
@@ -104,11 +129,12 @@ define void @init_array_of_ptrs_to_structs_interleave4(ptr noalias %arc_ptrs, pt
 ; CHECK-NEXT:    [[TMP10:%.*]] = tail call i64 @llvm.vscale.i64()
 ; CHECK-NEXT:    [[TMP11:%.*]] = mul i64 [[TMP10]], 48
 ; CHECK-NEXT:    [[SCEVGEP38:%.*]] = getelementptr i8, ptr [[LSR_IV36]], i64 [[TMP11]]
-; CHECK-NEXT:    store <vscale x 2 x ptr> [[WIDE_GEP]], ptr [[LSR_IV36]], align 8
-; CHECK-NEXT:    store <vscale x 2 x ptr> [[WIDE_GEP10]], ptr [[SCEVGEP40]], align 8
-; CHECK-NEXT:    store <vscale x 2 x ptr> [[WIDE_GEP11]], ptr [[SCEVGEP39]], align 8
-; CHECK-NEXT:    store <vscale x 2 x ptr> [[WIDE_GEP12]], ptr [[SCEVGEP38]], align 8
-; CHECK-NEXT:    [[VEC_IND_NEXT]] = add nuw <vscale x 2 x i64> [[VEC_IND]], [[INVARIANT_OP27]]
+; CHECK-NEXT:    store <vscale x 2 x ptr> [[TMP32]], ptr [[LSR_IV36]], align 8
+; CHECK-NEXT:    store <vscale x 2 x ptr> [[TMP17]], ptr [[SCEVGEP40]], align 8
+; CHECK-NEXT:    store <vscale x 2 x ptr> [[TMP19]], ptr [[SCEVGEP39]], align 8
+; CHECK-NEXT:    store <vscale x 2 x ptr> [[TMP21]], ptr [[SCEVGEP38]], align 8
+; CHECK-NEXT:    [[TMP28]] = add <vscale x 2 x i64> [[VEC_IND]], [[DOTSPLAT4]]
+; CHECK-NEXT:    [[VEC_IND_NEXT:%.*]] = add nuw <vscale x 2 x i64> [[VEC_IND]], [[INVARIANT_OP27]]
 ; CHECK-NEXT:    [[TMP12:%.*]] = tail call i64 @llvm.vscale.i64()
 ; CHECK-NEXT:    [[TMP13:%.*]] = shl nuw nsw i64 [[TMP12]], 3
 ; CHECK-NEXT:    [[LSR_IV_NEXT35]] = sub i64 [[LSR_IV34]], [[TMP13]]



More information about the llvm-branch-commits mailing list