[llvm-branch-commits] [llvm] [CGP] Simple loop strength reduction for vector values (PR #226197)
Graham Hunter via llvm-branch-commits
llvm-branch-commits at lists.llvm.org
Tue Sep 29 03:32:36 PDT 2026
https://github.com/huntergr-arm updated https://github.com/llvm/llvm-project/pull/226197
>From 807cb9998ef4e943e6166ba03c3f1e24724ebc33 Mon Sep 17 00:00:00 2001
From: Graham Hunter <graham.hunter at arm.com>
Date: Thu, 24 Sep 2026 12:40:37 +0000
Subject: [PATCH 1/2] Implement single gep case
---
llvm/lib/CodeGen/CodeGenPrepare.cpp | 145 ++++++++++++++++++
.../AArch64/strength-reduce-vector-as-data.ll | 16 +-
2 files changed, 152 insertions(+), 9 deletions(-)
diff --git a/llvm/lib/CodeGen/CodeGenPrepare.cpp b/llvm/lib/CodeGen/CodeGenPrepare.cpp
index ff40f210c5790..29983cfbef024 100644
--- a/llvm/lib/CodeGen/CodeGenPrepare.cpp
+++ b/llvm/lib/CodeGen/CodeGenPrepare.cpp
@@ -8944,6 +8944,146 @@ static bool optimizeBranch(CondBrInst *Branch, const TargetLowering &TLI,
return false;
}
+// Performs very basic loop strength reduction to vector values with a known
+// evolution from one iteration to the next that are used as the value operand
+// for a store.
+// A simple example to illustrate:
+//
+// %vec.ind = phi <vscale x 2 x i64> [ %start, %entry ],
+// [ %vec.ind.next, %vector.body ]
+// %data.gep = getelementptr inbounds nuw [72 x i8], ptr %datap,
+// <vscale x 2 x i64> %vec.ind
+// %addr.gep = getelementptr inbounds nuw [8 x i8], ptr %addrp, i64 %index
+// store <vscale x 2 x ptr> %data.gep, ptr %addr.gep, align 8
+// %vec.ind.next = add nuw nsw <vscale x 2 x i64> %vec.ind, %elt.cnt.splat
+//
+// This will end up with a multiply by a vscale-scaled term in the loop, as
+// well as adding to the base. Changing it to add a splat based on vscale *
+// min.elt.cnt * sizeof(ptrdiff) and remove the gep removes the multiply.
+//
+// TODO: Support more cases, such as the address instead of value operand for
+// strided memory operations.
+static bool strengthReduceVectorPhiUsers(PHINode *Phi, LoopInfo *LI) {
+ // We're only interested in header phis in innermost loops.
+ // We want a loop with an identifiable preheader and single latch.
+ Loop *L = LI->getLoopFor(Phi->getParent());
+ if (!L || !L->isInnermost())
+ return false;
+
+ BasicBlock *Header = L->getHeader();
+ BasicBlock *PreHeader = L->getLoopPreheader();
+ BasicBlock *Latch = L->getLoopLatch();
+ if (!PreHeader || !Latch || Phi->getParent() != Header)
+ return false;
+
+ // Check the progression of the phi.
+ Value *Start = Phi->getIncomingValueForBlock(PreHeader);
+ Value *Step = Phi->getIncomingValueForBlock(Latch);
+
+ // TODO: Handle offsets from stepvector.
+ if (!match(Start, m_Intrinsic<Intrinsic::stepvector>()))
+ return false;
+
+ Value *LoopStride = nullptr;
+ if (!match(Step, m_c_Add(m_Specific(Phi), m_Value(LoopStride))))
+ return false;
+
+ const APInt *ShiftAmt = nullptr;
+ if (!match(LoopStride, m_Splat(m_Shl(m_VScale(), m_APInt(ShiftAmt)))))
+ return false;
+
+ // Record users of interest.
+ struct OffsetGEP {
+ GetElementPtrInst *GEP;
+ Value *Offset;
+ };
+ SmallVector<OffsetGEP, 4> Candidates;
+ Value *CommonBase = nullptr;
+ unsigned CommonSize = 0;
+ DataLayout DL = Phi->getFunction()->getDataLayout();
+ for (User *U : Phi->users()) {
+ Instruction *I = cast<Instruction>(U);
+ // Skip over the step, handled above.
+ if (Step == I)
+ continue;
+
+ // If we have an offset from the Phi values, record that and then look
+ // for a GEP.
+ Value *Offset = nullptr;
+ if (match(I, m_OneUse(m_c_Add(m_Specific(Phi), m_Value(Offset)))))
+ I = cast<Instruction>(I->getSingleUndroppableUse()->getUser());
+
+ // We're only interested in single index GEPs used only as the value
+ // operand in a store for now.
+ auto *GEP = dyn_cast<GetElementPtrInst>(I);
+ if (!GEP || GEP->getNumOperands() != 2)
+ return false;
+
+ Use *GEPUse = GEP->getSingleUndroppableUse();
+ if (!GEPUse || !isa<StoreInst>(GEPUse->getUser()) ||
+ GEPUse->getOperandNo() != 0)
+ return false;
+
+ // Reject anything with a loop-varying base or if we have different bases
+ // for different GEPs.
+ // TODO: Support multiple bases.
+ Value *Base = I->getOperand(0);
+ if (!L->isLoopInvariant(Base) || (CommonBase && CommonBase != Base))
+ return false;
+
+ CommonBase = Base;
+
+ // Check that the indexed size is also the same.
+ unsigned Size = DL.getTypeStoreSize(GEP->getResultElementType());
+ if (!Size || (CommonSize && CommonSize != Size))
+ return false;
+
+ CommonSize = Size;
+ Candidates.push_back({GEP, Offset});
+ }
+
+ // Multiply the start by the size of the struct, and add the base pointer.
+ IRBuilder<> PHBuilder(PreHeader->getTerminator());
+ VectorType *VTy = cast<VectorType>(Start->getType());
+ Type *ITy = VTy->getElementType();
+ Value *StructSize = ConstantInt::get(ITy, APInt(64, CommonSize));
+ StructSize = PHBuilder.CreateVectorSplat(VTy->getElementCount(), StructSize);
+ Value *NewStart = PHBuilder.CreateMul(Start, StructSize);
+ CommonBase = PHBuilder.CreatePtrToInt(CommonBase, ITy);
+ CommonBase = PHBuilder.CreateVectorSplat(VTy->getElementCount(), CommonBase);
+ NewStart = PHBuilder.CreateAdd(NewStart, CommonBase);
+
+ // Create a new step based on the total size of all struct addresses per
+ // iteration.
+ Value *StructStride = ConstantInt::get(ITy, ShiftAmt->getZExtValue());
+ StructStride = PHBuilder.CreateMul(StructStride, PHBuilder.CreateVScale(ITy));
+ StructStride =
+ PHBuilder.CreateVectorSplat(VTy->getElementCount(), StructStride);
+
+ IRBuilder<> LBuilder(cast<Instruction>(Step));
+ Value *NewStep = LBuilder.CreateAdd(Phi, StructStride);
+
+ // Update the phi to the new start and step.
+ Phi->setIncomingValueForBlock(PreHeader, NewStart);
+ Phi->setIncomingValueForBlock(Latch, NewStep);
+
+ // Replace the GEPs with casted adds to remove the multiplies from the loop.
+ // TODO: An alternative would be to order by offset, and just add from the
+ // previous term in the loop. Requires fewer registers, but does increase the
+ // critical path for each operation.
+ for (auto [GEP, Offset] : Candidates) {
+ Value *NewBase = Phi;
+ LBuilder.SetInsertPoint(GEP);
+ if (Offset) {
+ Offset = PHBuilder.CreateMul(Offset, StructStride);
+ NewBase = LBuilder.CreateAdd(NewBase, Offset);
+ }
+ GEP->replaceAllUsesWith(LBuilder.CreateIntToPtr(NewBase, GEP->getType()));
+ }
+
+ return true;
+}
+
bool CodeGenPrepare::optimizeInst(Instruction *I, ModifyDT &ModifiedDT) {
bool AnyChange = false;
AnyChange = fixupDbgVariableRecordsOnInst(*I);
@@ -8965,6 +9105,11 @@ bool CodeGenPrepare::optimizeInst(Instruction *I, ModifyDT &ModifiedDT) {
++NumPHIsElim;
return true;
}
+
+ // Look for simple GEPs on scalable vectors used as data which may
+ // introduce unnecessary multiplies in the loop.
+ if (P->getType()->isScalableTy())
+ AnyChange |= strengthReduceVectorPhiUsers(P, LI);
return AnyChange;
}
diff --git a/llvm/test/CodeGen/AArch64/strength-reduce-vector-as-data.ll b/llvm/test/CodeGen/AArch64/strength-reduce-vector-as-data.ll
index 853c94e1307c8..e49c276ba38dd 100644
--- a/llvm/test/CodeGen/AArch64/strength-reduce-vector-as-data.ll
+++ b/llvm/test/CodeGen/AArch64/strength-reduce-vector-as-data.ll
@@ -5,21 +5,19 @@ target triple = "aarch64-unknown-linux-gnu"
define void @init_array_of_ptrs_to_structs(ptr noalias %arc_ptrs, ptr %arc_new, i64 %num_arcs) #0 {
; CHECK-LABEL: init_array_of_ptrs_to_structs:
; CHECK: // %bb.0: // %entry
-; CHECK-NEXT: cntd x8
-; CHECK-NEXT: index z0.d, #0, #1
-; CHECK-NEXT: mov z2.d, x1
+; CHECK-NEXT: rdvl x8, #1
+; CHECK-NEXT: mov w9, #72 // =0x48
+; CHECK-NEXT: lsr x8, x8, #4
+; CHECK-NEXT: index z0.d, x1, x9
+; CHECK-NEXT: cntd x9
; CHECK-NEXT: mov z1.d, x8
-; CHECK-NEXT: mov z3.d, #72 // =0x48
-; CHECK-NEXT: neg x8, x8
-; CHECK-NEXT: ptrue p0.d
+; CHECK-NEXT: neg x8, x9
; CHECK-NEXT: and x8, x8, x2
; CHECK-NEXT: .LBB0_1: // %vector.body
; CHECK-NEXT: // =>This Inner Loop Header: Depth=1
-; CHECK-NEXT: movprfx z4, z2
-; CHECK-NEXT: mla z4.d, p0/m, z0.d, z3.d
; CHECK-NEXT: decd x8
+; CHECK-NEXT: str z0, [x0]
; CHECK-NEXT: add z0.d, z0.d, z1.d
-; CHECK-NEXT: str z4, [x0]
; CHECK-NEXT: incb x0
; CHECK-NEXT: cbnz x8, .LBB0_1
; CHECK-NEXT: // %bb.2: // %middle.block
>From a0fbe55065a2766b58f37391d31cdbb01ae64a01 Mon Sep 17 00:00:00 2001
From: Graham Hunter <graham.hunter at arm.com>
Date: Fri, 25 Sep 2026 13:36:33 +0000
Subject: [PATCH 2/2] update new IR test
---
.../AArch64/strength-reduce-vector-as-data.ll | 17 ++++++++++++++---
1 file changed, 14 insertions(+), 3 deletions(-)
diff --git a/llvm/test/Transforms/CodeGenPrepare/AArch64/strength-reduce-vector-as-data.ll b/llvm/test/Transforms/CodeGenPrepare/AArch64/strength-reduce-vector-as-data.ll
index 11cb04cde3c96..1bf5a3d2d8477 100644
--- a/llvm/test/Transforms/CodeGenPrepare/AArch64/strength-reduce-vector-as-data.ll
+++ b/llvm/test/Transforms/CodeGenPrepare/AArch64/strength-reduce-vector-as-data.ll
@@ -13,17 +13,28 @@ define void @init_array_of_ptrs_to_structs(ptr noalias %arc_ptrs, ptr %arc_new,
; CHECK-NEXT: [[TMP0:%.*]] = tail call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
; CHECK-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 2 x i64> poison, i64 [[STEP]], i64 0
; CHECK-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <vscale x 2 x i64> [[BROADCAST_SPLATINSERT]], <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+; CHECK-NEXT: [[TMP7:%.*]] = mul <vscale x 2 x i64> [[TMP0]], splat (i64 72)
+; CHECK-NEXT: [[TMP8:%.*]] = ptrtoint ptr [[ARC_NEW]] to i64
+; CHECK-NEXT: [[DOTSPLATINSERT:%.*]] = insertelement <vscale x 2 x i64> poison, i64 [[TMP8]], i64 0
+; CHECK-NEXT: [[DOTSPLAT:%.*]] = shufflevector <vscale x 2 x i64> [[DOTSPLATINSERT]], <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+; CHECK-NEXT: [[TMP9:%.*]] = add <vscale x 2 x i64> [[TMP7]], [[DOTSPLAT]]
+; CHECK-NEXT: [[TMP11:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-NEXT: [[TMP5:%.*]] = mul i64 1, [[TMP11]]
+; CHECK-NEXT: [[DOTSPLATINSERT1:%.*]] = insertelement <vscale x 2 x i64> poison, i64 [[TMP5]], i64 0
+; CHECK-NEXT: [[DOTSPLAT2:%.*]] = shufflevector <vscale x 2 x i64> [[DOTSPLATINSERT1]], <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT: [[VEC_IND:%.*]] = phi <vscale x 2 x i64> [ [[TMP0]], %[[ENTRY]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[VEC_IND:%.*]] = phi <vscale x 2 x i64> [ [[TMP9]], %[[ENTRY]] ], [ [[TMP10:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP6:%.*]] = inttoptr <vscale x 2 x i64> [[VEC_IND]] to <vscale x 2 x ptr>
; CHECK-NEXT: [[WIDE_GEP:%.*]] = getelementptr inbounds nuw [72 x i8], ptr [[ARC_NEW]], <vscale x 2 x i64> [[VEC_IND]]
; CHECK-NEXT: [[TMP1:%.*]] = getelementptr inbounds nuw [8 x i8], ptr [[ARC_PTRS]], i64 [[INDEX]]
-; CHECK-NEXT: store <vscale x 2 x ptr> [[WIDE_GEP]], ptr [[TMP1]], align 8
+; CHECK-NEXT: store <vscale x 2 x ptr> [[TMP6]], ptr [[TMP1]], align 8
; CHECK-NEXT: [[TMP2:%.*]] = tail call i64 @llvm.vscale.i64()
; CHECK-NEXT: [[TMP3:%.*]] = shl nuw nsw i64 [[TMP2]], 1
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP3]]
-; CHECK-NEXT: [[VEC_IND_NEXT]] = add nuw nsw <vscale x 2 x i64> [[VEC_IND]], [[BROADCAST_SPLAT]]
+; CHECK-NEXT: [[TMP10]] = add <vscale x 2 x i64> [[VEC_IND]], [[DOTSPLAT2]]
+; CHECK-NEXT: [[VEC_IND_NEXT:%.*]] = add nuw nsw <vscale x 2 x i64> [[VEC_IND]], [[BROADCAST_SPLAT]]
; CHECK-NEXT: [[TMP4:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
; CHECK-NEXT: br i1 [[TMP4]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
; CHECK: [[MIDDLE_BLOCK]]:
More information about the llvm-branch-commits
mailing list