[llvm-branch-commits] [llvm] [CGP] Simple loop strength reduction for vector values (PR #226197)
Graham Hunter via llvm-branch-commits
llvm-branch-commits at lists.llvm.org
Wed Sep 30 02:58:17 PDT 2026
================
@@ -8944,6 +8944,146 @@ static bool optimizeBranch(CondBrInst *Branch, const TargetLowering &TLI,
return false;
}
+// Performs very basic loop strength reduction to vector values with a known
+// evolution from one iteration to the next that are used as the value operand
+// for a store.
+// A simple example to illustrate:
+//
+// %vec.ind = phi <vscale x 2 x i64> [ %start, %entry ],
+// [ %vec.ind.next, %vector.body ]
+// %data.gep = getelementptr inbounds nuw [72 x i8], ptr %datap,
+// <vscale x 2 x i64> %vec.ind
+// %addr.gep = getelementptr inbounds nuw [8 x i8], ptr %addrp, i64 %index
+// store <vscale x 2 x ptr> %data.gep, ptr %addr.gep, align 8
+// %vec.ind.next = add nuw nsw <vscale x 2 x i64> %vec.ind, %elt.cnt.splat
+//
+// This will end up with a multiply by a vscale-scaled term in the loop, as
+// well as adding to the base. Changing it to add a splat based on vscale *
+// min.elt.cnt * sizeof(ptrdiff) and remove the gep removes the multiply.
+//
+// TODO: Support more cases, such as the address instead of value operand for
+// strided memory operations.
+static bool strengthReduceVectorPhiUsers(PHINode *Phi, LoopInfo *LI) {
+ // We're only interested in header phis in innermost loops.
+ // We want a loop with an identifiable preheader and single latch.
+ Loop *L = LI->getLoopFor(Phi->getParent());
+ if (!L || !L->isInnermost())
+ return false;
+
+ BasicBlock *Header = L->getHeader();
+ BasicBlock *PreHeader = L->getLoopPreheader();
+ BasicBlock *Latch = L->getLoopLatch();
+ if (!PreHeader || !Latch || Phi->getParent() != Header)
+ return false;
+
+ // Check the progression of the phi.
+ Value *Start = Phi->getIncomingValueForBlock(PreHeader);
+ Value *Step = Phi->getIncomingValueForBlock(Latch);
+
+ // TODO: Handle offsets from stepvector.
+ if (!match(Start, m_Intrinsic<Intrinsic::stepvector>()))
+ return false;
+
+ Value *LoopStride = nullptr;
+ if (!match(Step, m_c_Add(m_Specific(Phi), m_Value(LoopStride))))
+ return false;
+
+ const APInt *ShiftAmt = nullptr;
+ if (!match(LoopStride, m_Splat(m_Shl(m_VScale(), m_APInt(ShiftAmt)))))
+ return false;
+
+ // Record users of interest.
+ struct OffsetGEP {
+ GetElementPtrInst *GEP;
+ Value *Offset;
+ };
+ SmallVector<OffsetGEP, 4> Candidates;
+ Value *CommonBase = nullptr;
+ unsigned CommonSize = 0;
+ DataLayout DL = Phi->getFunction()->getDataLayout();
+ for (User *U : Phi->users()) {
+ Instruction *I = cast<Instruction>(U);
+ // Skip over the step, handled above.
+ if (Step == I)
+ continue;
+
+ // If we have an offset from the Phi values, record that and then look
+ // for a GEP.
+ Value *Offset = nullptr;
+ if (match(I, m_OneUse(m_c_Add(m_Specific(Phi), m_Value(Offset)))))
+ I = cast<Instruction>(I->getSingleUndroppableUse()->getUser());
+
+ // We're only interested in single index GEPs used only as the value
+ // operand in a store for now.
+ auto *GEP = dyn_cast<GetElementPtrInst>(I);
+ if (!GEP || GEP->getNumOperands() != 2)
+ return false;
+
+ Use *GEPUse = GEP->getSingleUndroppableUse();
+ if (!GEPUse || !isa<StoreInst>(GEPUse->getUser()) ||
+ GEPUse->getOperandNo() != 0)
+ return false;
+
+ // Reject anything with a loop-varying base or if we have different bases
+ // for different GEPs.
+ // TODO: Support multiple bases.
+ Value *Base = I->getOperand(0);
+ if (!L->isLoopInvariant(Base) || (CommonBase && CommonBase != Base))
+ return false;
+
+ CommonBase = Base;
+
+ // Check that the indexed size is also the same.
+ unsigned Size = DL.getTypeStoreSize(GEP->getResultElementType());
+ if (!Size || (CommonSize && CommonSize != Size))
+ return false;
+
+ CommonSize = Size;
+ Candidates.push_back({GEP, Offset});
+ }
+
+ // Multiply the start by the size of the struct, and add the base pointer.
+ IRBuilder<> PHBuilder(PreHeader->getTerminator());
+ VectorType *VTy = cast<VectorType>(Start->getType());
+ Type *ITy = VTy->getElementType();
+ Value *StructSize = ConstantInt::get(ITy, APInt(64, CommonSize));
+ StructSize = PHBuilder.CreateVectorSplat(VTy->getElementCount(), StructSize);
+ Value *NewStart = PHBuilder.CreateMul(Start, StructSize);
+ CommonBase = PHBuilder.CreatePtrToInt(CommonBase, ITy);
+ CommonBase = PHBuilder.CreateVectorSplat(VTy->getElementCount(), CommonBase);
+ NewStart = PHBuilder.CreateAdd(NewStart, CommonBase);
+
+ // Create a new step based on the total size of all struct addresses per
+ // iteration.
+ Value *StructStride = ConstantInt::get(ITy, ShiftAmt->getZExtValue());
+ StructStride = PHBuilder.CreateMul(StructStride, PHBuilder.CreateVScale(ITy));
+ StructStride =
+ PHBuilder.CreateVectorSplat(VTy->getElementCount(), StructStride);
+
+ IRBuilder<> LBuilder(cast<Instruction>(Step));
+ Value *NewStep = LBuilder.CreateAdd(Phi, StructStride);
----------------
huntergr-arm wrote:
done.
https://github.com/llvm/llvm-project/pull/226197
More information about the llvm-branch-commits
mailing list