[llvm-branch-commits] [llvm] [CGP] Simple loop strength reduction for vector values (PR #226197)

Graham Hunter via llvm-branch-commits llvm-branch-commits at lists.llvm.org
Wed Sep 30 05:59:16 PDT 2026


https://github.com/huntergr-arm updated https://github.com/llvm/llvm-project/pull/226197

>From 5142e67b0c4113a5ee5323abcc8f999e7d963490 Mon Sep 17 00:00:00 2001
From: Graham Hunter <graham.hunter at arm.com>
Date: Thu, 24 Sep 2026 12:40:37 +0000
Subject: [PATCH 1/6] Implement single gep case

---
 llvm/lib/CodeGen/CodeGenPrepare.cpp           | 145 ++++++++++++++++++
 .../AArch64/strength-reduce-vector-as-data.ll |  16 +-
 2 files changed, 152 insertions(+), 9 deletions(-)

diff --git a/llvm/lib/CodeGen/CodeGenPrepare.cpp b/llvm/lib/CodeGen/CodeGenPrepare.cpp
index ff40f210c5790..29983cfbef024 100644
--- a/llvm/lib/CodeGen/CodeGenPrepare.cpp
+++ b/llvm/lib/CodeGen/CodeGenPrepare.cpp
@@ -8944,6 +8944,146 @@ static bool optimizeBranch(CondBrInst *Branch, const TargetLowering &TLI,
   return false;
 }
 
+// Performs very basic loop strength reduction to vector values with a known
+// evolution from one iteration to the next that are used as the value operand
+// for a store.
+// A simple example to illustrate:
+//
+//  %vec.ind = phi <vscale x 2 x i64> [ %start, %entry ],
+//                                    [ %vec.ind.next, %vector.body ]
+//  %data.gep = getelementptr inbounds nuw [72 x i8], ptr %datap,
+//                            <vscale x 2 x i64> %vec.ind
+//  %addr.gep = getelementptr inbounds nuw [8 x i8], ptr %addrp, i64 %index
+//  store <vscale x 2 x ptr> %data.gep, ptr %addr.gep, align 8
+//  %vec.ind.next = add nuw nsw <vscale x 2 x i64> %vec.ind, %elt.cnt.splat
+//
+// This will end up with a multiply by a vscale-scaled term in the loop, as
+// well as adding to the base. Changing it to add a splat based on vscale *
+// min.elt.cnt * sizeof(ptrdiff) and remove the gep removes the multiply.
+//
+// TODO: Support more cases, such as the address instead of value operand for
+//       strided memory operations.
+static bool strengthReduceVectorPhiUsers(PHINode *Phi, LoopInfo *LI) {
+  // We're only interested in header phis in innermost loops.
+  // We want a loop with an identifiable preheader and single latch.
+  Loop *L = LI->getLoopFor(Phi->getParent());
+  if (!L || !L->isInnermost())
+    return false;
+
+  BasicBlock *Header = L->getHeader();
+  BasicBlock *PreHeader = L->getLoopPreheader();
+  BasicBlock *Latch = L->getLoopLatch();
+  if (!PreHeader || !Latch || Phi->getParent() != Header)
+    return false;
+
+  // Check the progression of the phi.
+  Value *Start = Phi->getIncomingValueForBlock(PreHeader);
+  Value *Step = Phi->getIncomingValueForBlock(Latch);
+
+  // TODO: Handle offsets from stepvector.
+  if (!match(Start, m_Intrinsic<Intrinsic::stepvector>()))
+    return false;
+
+  Value *LoopStride = nullptr;
+  if (!match(Step, m_c_Add(m_Specific(Phi), m_Value(LoopStride))))
+    return false;
+
+  const APInt *ShiftAmt = nullptr;
+  if (!match(LoopStride, m_Splat(m_Shl(m_VScale(), m_APInt(ShiftAmt)))))
+    return false;
+
+  // Record users of interest.
+  struct OffsetGEP {
+    GetElementPtrInst *GEP;
+    Value *Offset;
+  };
+  SmallVector<OffsetGEP, 4> Candidates;
+  Value *CommonBase = nullptr;
+  unsigned CommonSize = 0;
+  DataLayout DL = Phi->getFunction()->getDataLayout();
+  for (User *U : Phi->users()) {
+    Instruction *I = cast<Instruction>(U);
+    // Skip over the step, handled above.
+    if (Step == I)
+      continue;
+
+    // If we have an offset from the Phi values, record that and then look
+    // for a GEP.
+    Value *Offset = nullptr;
+    if (match(I, m_OneUse(m_c_Add(m_Specific(Phi), m_Value(Offset)))))
+      I = cast<Instruction>(I->getSingleUndroppableUse()->getUser());
+
+    // We're only interested in single index GEPs used only as the value
+    // operand in a store for now.
+    auto *GEP = dyn_cast<GetElementPtrInst>(I);
+    if (!GEP || GEP->getNumOperands() != 2)
+      return false;
+
+    Use *GEPUse = GEP->getSingleUndroppableUse();
+    if (!GEPUse || !isa<StoreInst>(GEPUse->getUser()) ||
+        GEPUse->getOperandNo() != 0)
+      return false;
+
+    // Reject anything with a loop-varying base or if we have different bases
+    // for different GEPs.
+    // TODO: Support multiple bases.
+    Value *Base = I->getOperand(0);
+    if (!L->isLoopInvariant(Base) || (CommonBase && CommonBase != Base))
+      return false;
+
+    CommonBase = Base;
+
+    // Check that the indexed size is also the same.
+    unsigned Size = DL.getTypeStoreSize(GEP->getResultElementType());
+    if (!Size || (CommonSize && CommonSize != Size))
+      return false;
+
+    CommonSize = Size;
+    Candidates.push_back({GEP, Offset});
+  }
+
+  // Multiply the start by the size of the struct, and add the base pointer.
+  IRBuilder<> PHBuilder(PreHeader->getTerminator());
+  VectorType *VTy = cast<VectorType>(Start->getType());
+  Type *ITy = VTy->getElementType();
+  Value *StructSize = ConstantInt::get(ITy, APInt(64, CommonSize));
+  StructSize = PHBuilder.CreateVectorSplat(VTy->getElementCount(), StructSize);
+  Value *NewStart = PHBuilder.CreateMul(Start, StructSize);
+  CommonBase = PHBuilder.CreatePtrToInt(CommonBase, ITy);
+  CommonBase = PHBuilder.CreateVectorSplat(VTy->getElementCount(), CommonBase);
+  NewStart = PHBuilder.CreateAdd(NewStart, CommonBase);
+
+  // Create a new step based on the total size of all struct addresses per
+  // iteration.
+  Value *StructStride = ConstantInt::get(ITy, ShiftAmt->getZExtValue());
+  StructStride = PHBuilder.CreateMul(StructStride, PHBuilder.CreateVScale(ITy));
+  StructStride =
+      PHBuilder.CreateVectorSplat(VTy->getElementCount(), StructStride);
+
+  IRBuilder<> LBuilder(cast<Instruction>(Step));
+  Value *NewStep = LBuilder.CreateAdd(Phi, StructStride);
+
+  // Update the phi to the new start and step.
+  Phi->setIncomingValueForBlock(PreHeader, NewStart);
+  Phi->setIncomingValueForBlock(Latch, NewStep);
+
+  // Replace the GEPs with casted adds to remove the multiplies from the loop.
+  // TODO: An alternative would be to order by offset, and just add from the
+  // previous term in the loop. Requires fewer registers, but does increase the
+  // critical path for each operation.
+  for (auto [GEP, Offset] : Candidates) {
+    Value *NewBase = Phi;
+    LBuilder.SetInsertPoint(GEP);
+    if (Offset) {
+      Offset = PHBuilder.CreateMul(Offset, StructStride);
+      NewBase = LBuilder.CreateAdd(NewBase, Offset);
+    }
+    GEP->replaceAllUsesWith(LBuilder.CreateIntToPtr(NewBase, GEP->getType()));
+  }
+
+  return true;
+}
+
 bool CodeGenPrepare::optimizeInst(Instruction *I, ModifyDT &ModifiedDT) {
   bool AnyChange = false;
   AnyChange = fixupDbgVariableRecordsOnInst(*I);
@@ -8965,6 +9105,11 @@ bool CodeGenPrepare::optimizeInst(Instruction *I, ModifyDT &ModifiedDT) {
       ++NumPHIsElim;
       return true;
     }
+
+    // Look for simple GEPs on scalable vectors used as data which may
+    // introduce unnecessary multiplies in the loop.
+    if (P->getType()->isScalableTy())
+      AnyChange |= strengthReduceVectorPhiUsers(P, LI);
     return AnyChange;
   }
 
diff --git a/llvm/test/CodeGen/AArch64/strength-reduce-vector-as-data.ll b/llvm/test/CodeGen/AArch64/strength-reduce-vector-as-data.ll
index 853c94e1307c8..e49c276ba38dd 100644
--- a/llvm/test/CodeGen/AArch64/strength-reduce-vector-as-data.ll
+++ b/llvm/test/CodeGen/AArch64/strength-reduce-vector-as-data.ll
@@ -5,21 +5,19 @@ target triple = "aarch64-unknown-linux-gnu"
 define void @init_array_of_ptrs_to_structs(ptr noalias %arc_ptrs, ptr %arc_new, i64 %num_arcs) #0 {
 ; CHECK-LABEL: init_array_of_ptrs_to_structs:
 ; CHECK:       // %bb.0: // %entry
-; CHECK-NEXT:    cntd x8
-; CHECK-NEXT:    index z0.d, #0, #1
-; CHECK-NEXT:    mov z2.d, x1
+; CHECK-NEXT:    rdvl x8, #1
+; CHECK-NEXT:    mov w9, #72 // =0x48
+; CHECK-NEXT:    lsr x8, x8, #4
+; CHECK-NEXT:    index z0.d, x1, x9
+; CHECK-NEXT:    cntd x9
 ; CHECK-NEXT:    mov z1.d, x8
-; CHECK-NEXT:    mov z3.d, #72 // =0x48
-; CHECK-NEXT:    neg x8, x8
-; CHECK-NEXT:    ptrue p0.d
+; CHECK-NEXT:    neg x8, x9
 ; CHECK-NEXT:    and x8, x8, x2
 ; CHECK-NEXT:  .LBB0_1: // %vector.body
 ; CHECK-NEXT:    // =>This Inner Loop Header: Depth=1
-; CHECK-NEXT:    movprfx z4, z2
-; CHECK-NEXT:    mla z4.d, p0/m, z0.d, z3.d
 ; CHECK-NEXT:    decd x8
+; CHECK-NEXT:    str z0, [x0]
 ; CHECK-NEXT:    add z0.d, z0.d, z1.d
-; CHECK-NEXT:    str z4, [x0]
 ; CHECK-NEXT:    incb x0
 ; CHECK-NEXT:    cbnz x8, .LBB0_1
 ; CHECK-NEXT:  // %bb.2: // %middle.block

>From e1bfc8271ee8f5283cbf64a41fe19f2546bb2d53 Mon Sep 17 00:00:00 2001
From: Graham Hunter <graham.hunter at arm.com>
Date: Fri, 25 Sep 2026 13:36:33 +0000
Subject: [PATCH 2/6] update new IR test

---
 .../AArch64/strength-reduce-vector-as-data.ll   | 17 ++++++++++++++---
 1 file changed, 14 insertions(+), 3 deletions(-)

diff --git a/llvm/test/Transforms/CodeGenPrepare/AArch64/strength-reduce-vector-as-data.ll b/llvm/test/Transforms/CodeGenPrepare/AArch64/strength-reduce-vector-as-data.ll
index 11cb04cde3c96..1bf5a3d2d8477 100644
--- a/llvm/test/Transforms/CodeGenPrepare/AArch64/strength-reduce-vector-as-data.ll
+++ b/llvm/test/Transforms/CodeGenPrepare/AArch64/strength-reduce-vector-as-data.ll
@@ -13,17 +13,28 @@ define void @init_array_of_ptrs_to_structs(ptr noalias %arc_ptrs, ptr %arc_new,
 ; CHECK-NEXT:    [[TMP0:%.*]] = tail call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
 ; CHECK-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 2 x i64> poison, i64 [[STEP]], i64 0
 ; CHECK-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <vscale x 2 x i64> [[BROADCAST_SPLATINSERT]], <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+; CHECK-NEXT:    [[TMP7:%.*]] = mul <vscale x 2 x i64> [[TMP0]], splat (i64 72)
+; CHECK-NEXT:    [[TMP8:%.*]] = ptrtoint ptr [[ARC_NEW]] to i64
+; CHECK-NEXT:    [[DOTSPLATINSERT:%.*]] = insertelement <vscale x 2 x i64> poison, i64 [[TMP8]], i64 0
+; CHECK-NEXT:    [[DOTSPLAT:%.*]] = shufflevector <vscale x 2 x i64> [[DOTSPLATINSERT]], <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+; CHECK-NEXT:    [[TMP9:%.*]] = add <vscale x 2 x i64> [[TMP7]], [[DOTSPLAT]]
+; CHECK-NEXT:    [[TMP11:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-NEXT:    [[TMP5:%.*]] = mul i64 1, [[TMP11]]
+; CHECK-NEXT:    [[DOTSPLATINSERT1:%.*]] = insertelement <vscale x 2 x i64> poison, i64 [[TMP5]], i64 0
+; CHECK-NEXT:    [[DOTSPLAT2:%.*]] = shufflevector <vscale x 2 x i64> [[DOTSPLATINSERT1]], <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
 ; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
 ; CHECK:       [[VECTOR_BODY]]:
 ; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT:    [[VEC_IND:%.*]] = phi <vscale x 2 x i64> [ [[TMP0]], %[[ENTRY]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[VEC_IND:%.*]] = phi <vscale x 2 x i64> [ [[TMP9]], %[[ENTRY]] ], [ [[TMP10:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP6:%.*]] = inttoptr <vscale x 2 x i64> [[VEC_IND]] to <vscale x 2 x ptr>
 ; CHECK-NEXT:    [[WIDE_GEP:%.*]] = getelementptr inbounds nuw [72 x i8], ptr [[ARC_NEW]], <vscale x 2 x i64> [[VEC_IND]]
 ; CHECK-NEXT:    [[TMP1:%.*]] = getelementptr inbounds nuw [8 x i8], ptr [[ARC_PTRS]], i64 [[INDEX]]
-; CHECK-NEXT:    store <vscale x 2 x ptr> [[WIDE_GEP]], ptr [[TMP1]], align 8
+; CHECK-NEXT:    store <vscale x 2 x ptr> [[TMP6]], ptr [[TMP1]], align 8
 ; CHECK-NEXT:    [[TMP2:%.*]] = tail call i64 @llvm.vscale.i64()
 ; CHECK-NEXT:    [[TMP3:%.*]] = shl nuw nsw i64 [[TMP2]], 1
 ; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP3]]
-; CHECK-NEXT:    [[VEC_IND_NEXT]] = add nuw nsw <vscale x 2 x i64> [[VEC_IND]], [[BROADCAST_SPLAT]]
+; CHECK-NEXT:    [[TMP10]] = add <vscale x 2 x i64> [[VEC_IND]], [[DOTSPLAT2]]
+; CHECK-NEXT:    [[VEC_IND_NEXT:%.*]] = add nuw nsw <vscale x 2 x i64> [[VEC_IND]], [[BROADCAST_SPLAT]]
 ; CHECK-NEXT:    [[TMP4:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
 ; CHECK-NEXT:    br i1 [[TMP4]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
 ; CHECK:       [[MIDDLE_BLOCK]]:

>From 548a9c1b114d081fff15aacb6d0b9c949acf01ef Mon Sep 17 00:00:00 2001
From: Graham Hunter <graham.hunter at arm.com>
Date: Tue, 29 Sep 2026 12:28:09 +0000
Subject: [PATCH 3/6] Remove interleave code that was left in from prototype

---
 llvm/lib/CodeGen/CodeGenPrepare.cpp | 85 ++++++++++++-----------------
 1 file changed, 34 insertions(+), 51 deletions(-)

diff --git a/llvm/lib/CodeGen/CodeGenPrepare.cpp b/llvm/lib/CodeGen/CodeGenPrepare.cpp
index 29983cfbef024..af50dc631a943 100644
--- a/llvm/lib/CodeGen/CodeGenPrepare.cpp
+++ b/llvm/lib/CodeGen/CodeGenPrepare.cpp
@@ -8992,66 +8992,56 @@ static bool strengthReduceVectorPhiUsers(PHINode *Phi, LoopInfo *LI) {
   if (!match(LoopStride, m_Splat(m_Shl(m_VScale(), m_APInt(ShiftAmt)))))
     return false;
 
-  // Record users of interest.
-  struct OffsetGEP {
-    GetElementPtrInst *GEP;
-    Value *Offset;
-  };
-  SmallVector<OffsetGEP, 4> Candidates;
-  Value *CommonBase = nullptr;
-  unsigned CommonSize = 0;
+  // Check that the elements are integers equal in size to pointers.
   DataLayout DL = Phi->getFunction()->getDataLayout();
+  VectorType *VTy = cast<VectorType>(Start->getType());
+  Type *ITy = VTy->getElementType();
+  if (DL.getTypeStoreSize(ITy) != DL.getPointerSize())
+    return false;
+
+  // For the simplest case, we're only interested if the phi has two users;
+  // The Step forming the incoming value for the backedge, and a GEP.
+  GetElementPtrInst *GEP = nullptr;
   for (User *U : Phi->users()) {
     Instruction *I = cast<Instruction>(U);
     // Skip over the step, handled above.
     if (Step == I)
       continue;
 
-    // If we have an offset from the Phi values, record that and then look
-    // for a GEP.
-    Value *Offset = nullptr;
-    if (match(I, m_OneUse(m_c_Add(m_Specific(Phi), m_Value(Offset)))))
-      I = cast<Instruction>(I->getSingleUndroppableUse()->getUser());
-
-    // We're only interested in single index GEPs used only as the value
-    // operand in a store for now.
-    auto *GEP = dyn_cast<GetElementPtrInst>(I);
-    if (!GEP || GEP->getNumOperands() != 2)
+    // If we've already found a GEP, bail out.
+    // TODO: Support multiple GEPs.
+    if (GEP)
       return false;
 
-    Use *GEPUse = GEP->getSingleUndroppableUse();
-    if (!GEPUse || !isa<StoreInst>(GEPUse->getUser()) ||
-        GEPUse->getOperandNo() != 0)
-      return false;
-
-    // Reject anything with a loop-varying base or if we have different bases
-    // for different GEPs.
-    // TODO: Support multiple bases.
-    Value *Base = I->getOperand(0);
-    if (!L->isLoopInvariant(Base) || (CommonBase && CommonBase != Base))
+    // We're only interested in single index GEPs
+    GEP = dyn_cast<GetElementPtrInst>(I);
+    if (!GEP || GEP->getNumOperands() != 2)
       return false;
+  }
 
-    CommonBase = Base;
+  // Only continue processing if the GEP has a single user, with said user
+  // being a store using the result of the GEP as the data operand.
+  Use *GEPUse = GEP->getSingleUndroppableUse();
+  if (!GEPUse || !isa<StoreInst>(GEPUse->getUser()) ||
+      GEPUse->getOperandNo() != 0)
+    return false;
 
-    // Check that the indexed size is also the same.
-    unsigned Size = DL.getTypeStoreSize(GEP->getResultElementType());
-    if (!Size || (CommonSize && CommonSize != Size))
-      return false;
+  // Reject if the base isn't loop invariant.
+  Value *Base = GEP->getOperand(0);
+  if (!L->isLoopInvariant(Base))
+    return false;
 
-    CommonSize = Size;
-    Candidates.push_back({GEP, Offset});
-  }
+  // Check that the indexed size is also the same.
+  unsigned Size = DL.getTypeStoreSize(GEP->getResultElementType());
 
   // Multiply the start by the size of the struct, and add the base pointer.
   IRBuilder<> PHBuilder(PreHeader->getTerminator());
-  VectorType *VTy = cast<VectorType>(Start->getType());
-  Type *ITy = VTy->getElementType();
-  Value *StructSize = ConstantInt::get(ITy, APInt(64, CommonSize));
+  Value *StructSize = ConstantInt::get(ITy, APInt(64, Size));
   StructSize = PHBuilder.CreateVectorSplat(VTy->getElementCount(), StructSize);
   Value *NewStart = PHBuilder.CreateMul(Start, StructSize);
-  CommonBase = PHBuilder.CreatePtrToInt(CommonBase, ITy);
-  CommonBase = PHBuilder.CreateVectorSplat(VTy->getElementCount(), CommonBase);
-  NewStart = PHBuilder.CreateAdd(NewStart, CommonBase);
+  Base = PHBuilder.CreatePtrToInt(Base, ITy);
+  Base = PHBuilder.CreateVectorSplat(VTy->getElementCount(), Base);
+  NewStart = PHBuilder.CreateAdd(NewStart, Base);
 
   // Create a new step based on the total size of all struct addresses per
   // iteration.
@@ -9071,15 +9061,8 @@ static bool strengthReduceVectorPhiUsers(PHINode *Phi, LoopInfo *LI) {
   // TODO: An alternative would be to order by offset, and just add from the
   // previous term in the loop. Requires fewer registers, but does increase the
   // critical path for each operation.
-  for (auto [GEP, Offset] : Candidates) {
-    Value *NewBase = Phi;
-    LBuilder.SetInsertPoint(GEP);
-    if (Offset) {
-      Offset = PHBuilder.CreateMul(Offset, StructStride);
-      NewBase = LBuilder.CreateAdd(NewBase, Offset);
-    }
-    GEP->replaceAllUsesWith(LBuilder.CreateIntToPtr(NewBase, GEP->getType()));
-  }
+  LBuilder.SetInsertPoint(GEP);
+  GEP->replaceAllUsesWith(LBuilder.CreateIntToPtr(Phi, GEP->getType()));
 
   return true;
 }

>From 7c29b8347aee553132c35d50ddf7ba3cb3c55a8e Mon Sep 17 00:00:00 2001
From: Graham Hunter <graham.hunter at arm.com>
Date: Tue, 29 Sep 2026 13:23:26 +0000
Subject: [PATCH 4/6] Move shift amount check from interleaved PR

---
 llvm/lib/CodeGen/CodeGenPrepare.cpp | 5 +++++
 1 file changed, 5 insertions(+)

diff --git a/llvm/lib/CodeGen/CodeGenPrepare.cpp b/llvm/lib/CodeGen/CodeGenPrepare.cpp
index af50dc631a943..c53a75a8fa5f3 100644
--- a/llvm/lib/CodeGen/CodeGenPrepare.cpp
+++ b/llvm/lib/CodeGen/CodeGenPrepare.cpp
@@ -8992,6 +8992,11 @@ static bool strengthReduceVectorPhiUsers(PHINode *Phi, LoopInfo *LI) {
   if (!match(LoopStride, m_Splat(m_Shl(m_VScale(), m_APInt(ShiftAmt)))))
     return false;
 
+  // Make sure the shift amount matches the minimum element count.
+  auto EltCnt = cast<VectorType>(Phi->getType())->getElementCount();
+  if (1 << ShiftAmt->getZExtValue() != EltCnt.getKnownMinValue())
+    return false;
+
   // Check that the elements are integers equal in size to pointers.
   DataLayout DL = Phi->getFunction()->getDataLayout();
   VectorType *VTy = cast<VectorType>(Start->getType());

>From be9ca0c6a4e91b33250590fdd533b003b7d025f5 Mon Sep 17 00:00:00 2001
From: Graham Hunter <graham.hunter at arm.com>
Date: Tue, 29 Sep 2026 13:38:28 +0000
Subject: [PATCH 5/6] Add pseudo-IR for result to top comment

---
 llvm/lib/CodeGen/CodeGenPrepare.cpp | 16 +++++++++++++---
 1 file changed, 13 insertions(+), 3 deletions(-)

diff --git a/llvm/lib/CodeGen/CodeGenPrepare.cpp b/llvm/lib/CodeGen/CodeGenPrepare.cpp
index c53a75a8fa5f3..d20ff0fe8f407 100644
--- a/llvm/lib/CodeGen/CodeGenPrepare.cpp
+++ b/llvm/lib/CodeGen/CodeGenPrepare.cpp
@@ -8957,9 +8957,19 @@ static bool optimizeBranch(CondBrInst *Branch, const TargetLowering &TLI,
 //  store <vscale x 2 x ptr> %data.gep, ptr %addr.gep, align 8
 //  %vec.ind.next = add nuw nsw <vscale x 2 x i64> %vec.ind, %elt.cnt.splat
 //
-// This will end up with a multiply by a vscale-scaled term in the loop, as
-// well as adding to the base. Changing it to add a splat based on vscale *
-// min.elt.cnt * sizeof(ptrdiff) and remove the gep removes the multiply.
+// We want to change this to:
+//  %base = ptrtoint(%datap) + (<stepvector> * 72)
+//  %stride = splat(elementcount) * 72
+//  vector.body:
+//  %vec.ind = phi <vscale x 2 x i64> [ %base, %entry ],
+//                                    [ %vec.ind.next, %vector.body ]
+//  %data.val = inttoptr %vec.ind
+//  %addr.gep = getelementptr inbounds nuw [8 x i8], ptr %addrp, i64 %index
+//  store <vscale x 2 x ptr> %data.val, ptr %addr.gep, align 8
+//  %vec.ind.next = add nuw nsw <vscale x 2 x i64> %vec.ind, %stride
+//
+// Doing so will remove a multiply from the loop, and leave the update as just
+// an add.
 //
 // TODO: Support more cases, such as the address instead of value operand for
 //       strided memory operations.

>From da34b8ad84be1f70d8acb53a90d2b880a7db5314 Mon Sep 17 00:00:00 2001
From: Graham Hunter <graham.hunter at arm.com>
Date: Tue, 29 Sep 2026 16:26:46 +0000
Subject: [PATCH 6/6] Use vector of ptrs

---
 llvm/lib/CodeGen/CodeGenPrepare.cpp           | 58 +++++++++++--------
 .../AArch64/strength-reduce-vector-as-data.ll |  7 +--
 .../AArch64/strength-reduce-vector-as-data.ll | 35 ++++++-----
 3 files changed, 55 insertions(+), 45 deletions(-)

diff --git a/llvm/lib/CodeGen/CodeGenPrepare.cpp b/llvm/lib/CodeGen/CodeGenPrepare.cpp
index d20ff0fe8f407..f4d838754e6aa 100644
--- a/llvm/lib/CodeGen/CodeGenPrepare.cpp
+++ b/llvm/lib/CodeGen/CodeGenPrepare.cpp
@@ -8999,7 +8999,10 @@ static bool strengthReduceVectorPhiUsers(PHINode *Phi, LoopInfo *LI) {
     return false;
 
   const APInt *ShiftAmt = nullptr;
-  if (!match(LoopStride, m_Splat(m_Shl(m_VScale(), m_APInt(ShiftAmt)))))
+  Value *ShiftedVScale = nullptr;
+  if (!match(LoopStride,
+             m_Splat(
+                 m_Value(ShiftedVScale, m_Shl(m_VScale(), m_APInt(ShiftAmt))))))
     return false;
 
   // Make sure the shift amount matches the minimum element count.
@@ -9050,34 +9053,43 @@ static bool strengthReduceVectorPhiUsers(PHINode *Phi, LoopInfo *LI) {
   unsigned Size = DL.getTypeStoreSize(GEP->getResultElementType());
 
   // Multiply the start by the size of the struct, and add the base pointer.
-  IRBuilder<> PHBuilder(PreHeader->getTerminator());
+  IRBuilder<> Builder(PreHeader->getTerminator());
   Value *StructSize = ConstantInt::get(ITy, APInt(64, Size));
-  StructSize = PHBuilder.CreateVectorSplat(VTy->getElementCount(), StructSize);
-  Value *NewStart = PHBuilder.CreateMul(Start, StructSize);
-  Base = PHBuilder.CreatePtrToInt(Base, ITy);
-  Base = PHBuilder.CreateVectorSplat(VTy->getElementCount(), Base);
-  NewStart = PHBuilder.CreateAdd(NewStart, Base);
+  Value *SizeSplat =
+      Builder.CreateVectorSplat(VTy->getElementCount(), StructSize);
+  Value *NewStart = Builder.CreateMul(Start, SizeSplat);
+  Base = Builder.CreatePtrToInt(Base, ITy);
+  Base = Builder.CreateVectorSplat(VTy->getElementCount(), Base);
+  NewStart = Builder.CreateAdd(NewStart, Base);
+  NewStart = Builder.CreateIntToPtr(NewStart, GEP->getType());
 
   // Create a new step based on the total size of all struct addresses per
   // iteration.
-  Value *StructStride = ConstantInt::get(ITy, ShiftAmt->getZExtValue());
-  StructStride = PHBuilder.CreateMul(StructStride, PHBuilder.CreateVScale(ITy));
+  Value *StructStride = Builder.CreateMul(ShiftedVScale, StructSize);
   StructStride =
-      PHBuilder.CreateVectorSplat(VTy->getElementCount(), StructStride);
+      Builder.CreateVectorSplat(VTy->getElementCount(), StructStride);
 
-  IRBuilder<> LBuilder(cast<Instruction>(Step));
-  Value *NewStep = LBuilder.CreateAdd(Phi, StructStride);
-
-  // Update the phi to the new start and step.
-  Phi->setIncomingValueForBlock(PreHeader, NewStart);
-  Phi->setIncomingValueForBlock(Latch, NewStep);
-
-  // Replace the GEPs with casted adds to remove the multiplies from the loop.
-  // TODO: An alternative would be to order by offset, and just add from the
-  // previous term in the loop. Requires fewer registers, but does increase the
-  // critical path for each operation.
-  LBuilder.SetInsertPoint(GEP);
-  GEP->replaceAllUsesWith(LBuilder.CreateIntToPtr(Phi, GEP->getType()));
+  Builder.SetInsertPoint(Phi);
+  PHINode *NewPhi = Builder.CreatePHI(GEP->getType(), 2, "vec.lsr.phi");
+  Builder.SetInsertPoint(cast<Instruction>(Step));
+  LLVMContext &Ctx = ITy->getContext();
+  StructStride =
+      Builder.CreateGEP(IntegerType::getInt8Ty(Ctx), NewPhi, {StructStride});
+
+  // Set up the new PHI.
+  NewPhi->addIncoming(NewStart, PreHeader);
+  NewPhi->addIncoming(StructStride, Latch);
+
+  // Replace the GEP with the new phi.
+  Builder.SetInsertPoint(GEP);
+  GEP->replaceAllUsesWith(NewPhi);
+
+  // Remove incoming values for the old phi.
+  // TODO: It would be nice to erase it at this point, but removing the GEP
+  //       seems to cause crashes even if removeAllAssertingVHReferences is
+  //       called, so we're currently relying on the backend to remove it.
+  Phi->removeIncomingValue(1u);
+  Phi->removeIncomingValue(0u);
 
   return true;
 }
diff --git a/llvm/test/CodeGen/AArch64/strength-reduce-vector-as-data.ll b/llvm/test/CodeGen/AArch64/strength-reduce-vector-as-data.ll
index e49c276ba38dd..acf62cbe6cacd 100644
--- a/llvm/test/CodeGen/AArch64/strength-reduce-vector-as-data.ll
+++ b/llvm/test/CodeGen/AArch64/strength-reduce-vector-as-data.ll
@@ -5,11 +5,10 @@ target triple = "aarch64-unknown-linux-gnu"
 define void @init_array_of_ptrs_to_structs(ptr noalias %arc_ptrs, ptr %arc_new, i64 %num_arcs) #0 {
 ; CHECK-LABEL: init_array_of_ptrs_to_structs:
 ; CHECK:       // %bb.0: // %entry
-; CHECK-NEXT:    rdvl x8, #1
-; CHECK-NEXT:    mov w9, #72 // =0x48
-; CHECK-NEXT:    lsr x8, x8, #4
-; CHECK-NEXT:    index z0.d, x1, x9
+; CHECK-NEXT:    mov w8, #72 // =0x48
 ; CHECK-NEXT:    cntd x9
+; CHECK-NEXT:    index z0.d, x1, x8
+; CHECK-NEXT:    rdvl x8, #9
 ; CHECK-NEXT:    mov z1.d, x8
 ; CHECK-NEXT:    neg x8, x9
 ; CHECK-NEXT:    and x8, x8, x2
diff --git a/llvm/test/Transforms/CodeGenPrepare/AArch64/strength-reduce-vector-as-data.ll b/llvm/test/Transforms/CodeGenPrepare/AArch64/strength-reduce-vector-as-data.ll
index 1bf5a3d2d8477..6612a15a41d91 100644
--- a/llvm/test/Transforms/CodeGenPrepare/AArch64/strength-reduce-vector-as-data.ll
+++ b/llvm/test/Transforms/CodeGenPrepare/AArch64/strength-reduce-vector-as-data.ll
@@ -13,30 +13,29 @@ define void @init_array_of_ptrs_to_structs(ptr noalias %arc_ptrs, ptr %arc_new,
 ; CHECK-NEXT:    [[TMP0:%.*]] = tail call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
 ; CHECK-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 2 x i64> poison, i64 [[STEP]], i64 0
 ; CHECK-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <vscale x 2 x i64> [[BROADCAST_SPLATINSERT]], <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
-; CHECK-NEXT:    [[TMP7:%.*]] = mul <vscale x 2 x i64> [[TMP0]], splat (i64 72)
-; CHECK-NEXT:    [[TMP8:%.*]] = ptrtoint ptr [[ARC_NEW]] to i64
-; CHECK-NEXT:    [[DOTSPLATINSERT:%.*]] = insertelement <vscale x 2 x i64> poison, i64 [[TMP8]], i64 0
+; CHECK-NEXT:    [[TMP1:%.*]] = mul <vscale x 2 x i64> [[TMP0]], splat (i64 72)
+; CHECK-NEXT:    [[TMP2:%.*]] = ptrtoint ptr [[ARC_NEW]] to i64
+; CHECK-NEXT:    [[DOTSPLATINSERT:%.*]] = insertelement <vscale x 2 x i64> poison, i64 [[TMP2]], i64 0
 ; CHECK-NEXT:    [[DOTSPLAT:%.*]] = shufflevector <vscale x 2 x i64> [[DOTSPLATINSERT]], <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
-; CHECK-NEXT:    [[TMP9:%.*]] = add <vscale x 2 x i64> [[TMP7]], [[DOTSPLAT]]
-; CHECK-NEXT:    [[TMP11:%.*]] = call i64 @llvm.vscale.i64()
-; CHECK-NEXT:    [[TMP5:%.*]] = mul i64 1, [[TMP11]]
+; CHECK-NEXT:    [[TMP3:%.*]] = add <vscale x 2 x i64> [[TMP1]], [[DOTSPLAT]]
+; CHECK-NEXT:    [[TMP4:%.*]] = inttoptr <vscale x 2 x i64> [[TMP3]] to <vscale x 2 x ptr>
+; CHECK-NEXT:    [[TMP5:%.*]] = mul i64 [[STEP]], 72
 ; CHECK-NEXT:    [[DOTSPLATINSERT1:%.*]] = insertelement <vscale x 2 x i64> poison, i64 [[TMP5]], i64 0
 ; CHECK-NEXT:    [[DOTSPLAT2:%.*]] = shufflevector <vscale x 2 x i64> [[DOTSPLATINSERT1]], <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
 ; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
 ; CHECK:       [[VECTOR_BODY]]:
 ; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT:    [[VEC_IND:%.*]] = phi <vscale x 2 x i64> [ [[TMP9]], %[[ENTRY]] ], [ [[TMP10:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT:    [[TMP6:%.*]] = inttoptr <vscale x 2 x i64> [[VEC_IND]] to <vscale x 2 x ptr>
-; CHECK-NEXT:    [[WIDE_GEP:%.*]] = getelementptr inbounds nuw [72 x i8], ptr [[ARC_NEW]], <vscale x 2 x i64> [[VEC_IND]]
-; CHECK-NEXT:    [[TMP1:%.*]] = getelementptr inbounds nuw [8 x i8], ptr [[ARC_PTRS]], i64 [[INDEX]]
-; CHECK-NEXT:    store <vscale x 2 x ptr> [[TMP6]], ptr [[TMP1]], align 8
-; CHECK-NEXT:    [[TMP2:%.*]] = tail call i64 @llvm.vscale.i64()
-; CHECK-NEXT:    [[TMP3:%.*]] = shl nuw nsw i64 [[TMP2]], 1
-; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP3]]
-; CHECK-NEXT:    [[TMP10]] = add <vscale x 2 x i64> [[VEC_IND]], [[DOTSPLAT2]]
-; CHECK-NEXT:    [[VEC_IND_NEXT:%.*]] = add nuw nsw <vscale x 2 x i64> [[VEC_IND]], [[BROADCAST_SPLAT]]
-; CHECK-NEXT:    [[TMP4:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-NEXT:    br i1 [[TMP4]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
+; CHECK-NEXT:    [[VEC_LSR_PHI:%.*]] = phi <vscale x 2 x ptr> [ [[TMP4]], %[[ENTRY]] ], [ [[TMP9:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[WIDE_GEP:%.*]] = getelementptr inbounds nuw [72 x i8], ptr [[ARC_NEW]], <vscale x 2 x i64> poison
+; CHECK-NEXT:    [[TMP6:%.*]] = getelementptr inbounds nuw [8 x i8], ptr [[ARC_PTRS]], i64 [[INDEX]]
+; CHECK-NEXT:    store <vscale x 2 x ptr> [[VEC_LSR_PHI]], ptr [[TMP6]], align 8
+; CHECK-NEXT:    [[TMP7:%.*]] = tail call i64 @llvm.vscale.i64()
+; CHECK-NEXT:    [[TMP8:%.*]] = shl nuw nsw i64 [[TMP7]], 1
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP8]]
+; CHECK-NEXT:    [[TMP9]] = getelementptr i8, <vscale x 2 x ptr> [[VEC_LSR_PHI]], <vscale x 2 x i64> [[DOTSPLAT2]]
+; CHECK-NEXT:    [[VEC_IND_NEXT:%.*]] = add nuw nsw <vscale x 2 x i64> poison, [[BROADCAST_SPLAT]]
+; CHECK-NEXT:    [[TMP10:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP10]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
 ; CHECK:       [[MIDDLE_BLOCK]]:
 ; CHECK-NEXT:    ret void
 ;



More information about the llvm-branch-commits mailing list