[llvm] [LV] Simple Linear Recurrence Vectorization (PR #219440)
Vladislav Tarasov via llvm-commits
llvm-commits at lists.llvm.org
Fri Aug 28 04:06:49 PDT 2026
https://github.com/tarvlad created https://github.com/llvm/llvm-project/pull/219440
None
>From babc567bda1ede963a9a5850a1469d0d3baae644 Mon Sep 17 00:00:00 2001
From: tarvlad <vladislav.tarasov at huawei.com>
Date: Fri, 28 Aug 2026 15:17:37 +0800
Subject: [PATCH 1/2] [IVDescriptors] Add RecurKind::IntLinear and
isLinearRecurrencePHI matcher
Add RecurKind::IntLinear describing an integer linear recurrence
h = C*h + x, where C is loop-invariant and x is a loop-varying value
that does not depend on the recurrence (e.g. a polynomial hash like
Rabin-Karp's h = 31*h + s[i]).
Add RecurrenceDescriptor::isLinearRecurrencePHI matching the pattern
add(mul(phi, C), x), restricting phi to a single in-loop use in the
chain and outside uses only through the exit value. The coefficient C
and per-lane value x can be retrieved via optional out-parameters.
Update the SLP horizontal-reduction switches for the new kind.
The matcher is not used yet (NFC); a follow-up patch will wire it into
the loop vectorizer to help address llvm/llvm-project#56999.
---
llvm/include/llvm/Analysis/IVDescriptors.h | 15 ++
llvm/lib/Analysis/IVDescriptors.cpp | 94 ++++++++++++
.../Transforms/Vectorize/SLPVectorizer.cpp | 3 +
llvm/unittests/Analysis/IVDescriptorsTest.cpp | 142 ++++++++++++++++++
4 files changed, 254 insertions(+)
diff --git a/llvm/include/llvm/Analysis/IVDescriptors.h b/llvm/include/llvm/Analysis/IVDescriptors.h
index bad372421dfae..db6153f4bc0b7 100644
--- a/llvm/include/llvm/Analysis/IVDescriptors.h
+++ b/llvm/include/llvm/Analysis/IVDescriptors.h
@@ -39,6 +39,8 @@ enum class RecurKind {
Sub, ///< Subtraction of integers
AddChainWithSubs, ///< A chain of adds and subs
Mul, ///< Product of integers.
+ IntLinear, ///< Linear recurrence of integers: h = C*h + x, where C is a
+ ///< loop-invariant coefficient and x is a loop-varying value.
Or, ///< Bitwise or logical OR of integers.
And, ///< Bitwise or logical AND of integers.
Xor, ///< Bitwise or logical XOR of integers.
@@ -225,6 +227,19 @@ class RecurrenceDescriptor {
LLVM_ABI static bool isFixedOrderRecurrence(PHINode *Phi, Loop *TheLoop,
DominatorTree *DT);
+ /// Returns true if Phi is a linear recurrence of the form
+ /// h = phi(start, h_next)
+ /// h_next = add(mul(h, C), x) (or with the add/mul operands swapped),
+ /// where C is loop-invariant and x is a loop-varying value that does not
+ /// depend on h. All in-loop uses of the recurrence must be part of this
+ /// chain; h itself may only be used outside the loop through h_next.
+ /// If non-null, \p Coeff and \p X are set to C and x respectively.
+ LLVM_ABI static bool isLinearRecurrencePHI(PHINode *Phi, Loop *TheLoop,
+ RecurrenceDescriptor &RedDes,
+ ScalarEvolution *SE,
+ Value **Coeff = nullptr,
+ Value **X = nullptr);
+
RecurKind getRecurrenceKind() const { return Kind; }
unsigned getOpcode() const { return getOpcode(getRecurrenceKind()); }
diff --git a/llvm/lib/Analysis/IVDescriptors.cpp b/llvm/lib/Analysis/IVDescriptors.cpp
index 800b64e9a29af..b37c0096f28a0 100644
--- a/llvm/lib/Analysis/IVDescriptors.cpp
+++ b/llvm/lib/Analysis/IVDescriptors.cpp
@@ -46,6 +46,7 @@ bool RecurrenceDescriptor::isIntegerRecurrenceKind(RecurKind Kind) {
case RecurKind::Sub:
case RecurKind::Add:
case RecurKind::Mul:
+ case RecurKind::IntLinear:
case RecurKind::Or:
case RecurKind::And:
case RecurKind::Xor:
@@ -1225,6 +1226,99 @@ bool RecurrenceDescriptor::isFixedOrderRecurrence(PHINode *Phi, Loop *TheLoop,
return true;
}
+bool RecurrenceDescriptor::isLinearRecurrencePHI(PHINode *Phi, Loop *TheLoop,
+ RecurrenceDescriptor &RedDes,
+ ScalarEvolution *SE,
+ Value **Coeff, Value **X) {
+ // Basic structural checks: the PHI must be in the loop header, with two
+ // incoming values, and of integer type.
+ if (Phi->getNumIncomingValues() != 2 ||
+ Phi->getParent() != TheLoop->getHeader() ||
+ !Phi->getType()->isIntegerTy())
+ return false;
+ auto *Preheader = TheLoop->getLoopPreheader();
+ auto *Latch = TheLoop->getLoopLatch();
+ if (!Preheader || !Latch)
+ return false;
+
+ Value *Start = Phi->getIncomingValueForBlock(Preheader);
+ auto *Exit = dyn_cast<Instruction>(Phi->getIncomingValueForBlock(Latch));
+ if (!Exit || !TheLoop->contains(Exit))
+ return false;
+
+ // The backedge value must be an add of the form add(mul(Phi, C), X) (the
+ // add and mul operands may be swapped), where C is loop-invariant and X is a
+ // loop-varying value that does not depend on Phi.
+ auto *Add = dyn_cast<BinaryOperator>(Exit);
+ if (!Add || Add->getOpcode() != Instruction::Add)
+ return false;
+
+ BinaryOperator *Mul = dyn_cast<BinaryOperator>(Add->getOperand(0));
+ Value *XVal = Add->getOperand(1);
+ if (!Mul || Mul->getOpcode() != Instruction::Mul) {
+ Mul = dyn_cast<BinaryOperator>(Add->getOperand(1));
+ XVal = Add->getOperand(0);
+ }
+ if (!Mul || Mul->getOpcode() != Instruction::Mul)
+ return false;
+
+ // The mul must use Phi and a loop-invariant coefficient.
+ Value *C = nullptr;
+ if (Mul->getOperand(0) == Phi)
+ C = Mul->getOperand(1);
+ else if (Mul->getOperand(1) == Phi)
+ C = Mul->getOperand(0);
+ else
+ return false;
+ if (!SE->isLoopInvariant(SE->getSCEV(C), TheLoop))
+ return false;
+
+ // X must be an in-loop value that does not depend on Phi.
+ auto *XI = dyn_cast<Instruction>(XVal);
+ if (!XI || !TheLoop->contains(XI))
+ return false;
+ SmallPtrSet<Value *, 8> Visited;
+ SmallVector<Value *, 4> Worklist({XVal});
+ while (!Worklist.empty()) {
+ Value *V = Worklist.pop_back_val();
+ if (V == Phi)
+ return false;
+ if (!Visited.insert(V).second)
+ continue;
+ if (auto *I = dyn_cast<Instruction>(V))
+ for (Value *Op : I->operands())
+ Worklist.push_back(Op);
+ }
+
+ // Phi may only be used in-loop by the mul and must not have outside uses.
+ // The mul may only feed the add, and in-loop uses of the add other than the
+ // PHI backedge are not allowed.
+ for (User *U : Phi->users()) {
+ Instruction *UI = cast<Instruction>(U);
+ if (!TheLoop->contains(UI) || UI != Mul)
+ return false;
+ }
+ if (!Mul->hasOneUse())
+ return false;
+ for (User *U : Add->users()) {
+ Instruction *UI = cast<Instruction>(U);
+ if (UI == Phi)
+ continue;
+ if (TheLoop->contains(UI))
+ return false;
+ }
+
+ RedDes = RecurrenceDescriptor(Start, Exit, /*Store=*/nullptr,
+ RecurKind::IntLinear, FastMathFlags(),
+ /*ExactFP=*/nullptr, Phi->getType());
+ if (Coeff)
+ *Coeff = C;
+ if (X)
+ *X = XVal;
+ LLVM_DEBUG(dbgs() << "Found a linear recurrence PHI." << *Phi << "\n");
+ return true;
+}
+
unsigned RecurrenceDescriptor::getOpcode(RecurKind Kind) {
switch (Kind) {
case RecurKind::Sub:
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 358901da42266..9b300f43712c3 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -32518,6 +32518,7 @@ class HorizontalReduction {
case RecurKind::AnyOf:
case RecurKind::FindIV:
case RecurKind::FindLast:
+ case RecurKind::IntLinear:
case RecurKind::FMaxNum:
case RecurKind::FMinNum:
case RecurKind::FMaximumNum:
@@ -32676,6 +32677,7 @@ class HorizontalReduction {
case RecurKind::AnyOf:
case RecurKind::FindIV:
case RecurKind::FindLast:
+ case RecurKind::IntLinear:
case RecurKind::FMaxNum:
case RecurKind::FMinNum:
case RecurKind::FMaximumNum:
@@ -32781,6 +32783,7 @@ class HorizontalReduction {
case RecurKind::AnyOf:
case RecurKind::FindIV:
case RecurKind::FindLast:
+ case RecurKind::IntLinear:
case RecurKind::FMaxNum:
case RecurKind::FMinNum:
case RecurKind::FMaximumNum:
diff --git a/llvm/unittests/Analysis/IVDescriptorsTest.cpp b/llvm/unittests/Analysis/IVDescriptorsTest.cpp
index faf30fd322c10..dd0fc2ae869db 100644
--- a/llvm/unittests/Analysis/IVDescriptorsTest.cpp
+++ b/llvm/unittests/Analysis/IVDescriptorsTest.cpp
@@ -386,6 +386,148 @@ for.end:
});
}
+// This tests that a linear recurrence h = C * h + x with a loop-invariant
+// coefficient C is recognized.
+TEST(IVDescriptorsTest, LinearRecurrence) {
+ // Parse the module.
+ LLVMContext Context;
+
+ std::unique_ptr<Module> M = parseIR(Context, R"(
+ define i32 @rabin_karp(ptr %s, i64 %n) {
+ entry:
+ br label %for.body
+
+ for.body:
+ %i = phi i64 [ 0, %entry ], [ %i.next, %for.body ]
+ %h = phi i32 [ 0, %entry ], [ %h.next, %for.body ]
+ %arrayidx = getelementptr inbounds i32, ptr %s, i64 %i
+ %ld = load i32, ptr %arrayidx
+ %mul = mul i32 %h, 31
+ %h.next = add i32 %ld, %mul
+ %i.next = add nsw i64 %i, 1
+ %cmp = icmp slt i64 %i.next, %n
+ br i1 %cmp, label %for.body, label %for.end
+
+ for.end:
+ %h.lcssa = phi i32 [ %h.next, %for.body ]
+ ret i32 %h.lcssa
+ })");
+
+ runWithLoopInfoAndSE(
+ *M, "rabin_karp", [&](Function &F, LoopInfo &LI, ScalarEvolution &SE) {
+ Function::iterator FI = F.begin();
+ // First basic block is entry - skip it.
+ BasicBlock *Header = &*(++FI);
+ assert(Header->getName() == "for.body");
+ Loop *L = LI.getLoopFor(Header);
+ EXPECT_NE(L, nullptr);
+ BasicBlock::iterator BBI = Header->begin();
+ assert((&*BBI)->getName() == "i");
+ ++BBI;
+ PHINode *Phi = dyn_cast<PHINode>(&*BBI);
+ assert(Phi->getName() == "h");
+ RecurrenceDescriptor Rdx;
+ bool IsRdxPhi =
+ RecurrenceDescriptor::isLinearRecurrencePHI(Phi, L, Rdx, &SE);
+ EXPECT_TRUE(IsRdxPhi);
+ EXPECT_EQ(Rdx.getRecurrenceKind(), RecurKind::IntLinear);
+ EXPECT_EQ(Rdx.getLoopExitInstr()->getName(), "h.next");
+ });
+}
+
+// Linear recurrences with a loop-varying coefficient, with in-loop uses of the
+// recurrence outside the chain, or with direct outside uses of the recurrence
+// must not be recognized.
+TEST(IVDescriptorsTest, UnsupportedLinearRecurrence) {
+ // Parse the module.
+ LLVMContext Context;
+
+ std::unique_ptr<Module> M = parseIR(Context, R"(
+ define i32 @varying_coeff(ptr %s, ptr %cptr, i64 %n) {
+ entry:
+ br label %for.body
+
+ for.body:
+ %i = phi i64 [ 0, %entry ], [ %i.next, %for.body ]
+ %h = phi i32 [ 0, %entry ], [ %h.next, %for.body ]
+ %c = load i32, ptr %cptr
+ %arrayidx = getelementptr inbounds i32, ptr %s, i64 %i
+ %ld = load i32, ptr %arrayidx
+ %mul = mul i32 %h, %c
+ %h.next = add i32 %ld, %mul
+ %i.next = add nsw i64 %i, 1
+ %cmp = icmp slt i64 %i.next, %n
+ br i1 %cmp, label %for.body, label %for.end
+
+ for.end:
+ %h.lcssa = phi i32 [ %h.next, %for.body ]
+ ret i32 %h.lcssa
+ }
+
+ define i32 @used_in_loop(ptr %s, ptr %dst, i64 %n) {
+ entry:
+ br label %for.body
+
+ for.body:
+ %i = phi i64 [ 0, %entry ], [ %i.next, %for.body ]
+ %h = phi i32 [ 0, %entry ], [ %h.next, %for.body ]
+ %arrayidx = getelementptr inbounds i32, ptr %s, i64 %i
+ %ld = load i32, ptr %arrayidx
+ %mul = mul i32 %h, 31
+ %h.next = add i32 %ld, %mul
+ store i32 %h, ptr %dst
+ %i.next = add nsw i64 %i, 1
+ %cmp = icmp slt i64 %i.next, %n
+ br i1 %cmp, label %for.body, label %for.end
+
+ for.end:
+ %h.lcssa = phi i32 [ %h.next, %for.body ]
+ ret i32 %h.lcssa
+ }
+
+ define i32 @used_outside(ptr %s, i64 %n) {
+ entry:
+ br label %for.body
+
+ for.body:
+ %i = phi i64 [ 0, %entry ], [ %i.next, %for.body ]
+ %h = phi i32 [ 0, %entry ], [ %h.next, %for.body ]
+ %arrayidx = getelementptr inbounds i32, ptr %s, i64 %i
+ %ld = load i32, ptr %arrayidx
+ %mul = mul i32 %h, 31
+ %h.next = add i32 %ld, %mul
+ %i.next = add nsw i64 %i, 1
+ %cmp = icmp slt i64 %i.next, %n
+ br i1 %cmp, label %for.body, label %for.end
+
+ for.end:
+ ret i32 %h
+ })");
+
+ auto Check = [&](StringRef FuncName) {
+ runWithLoopInfoAndSE(
+ *M, FuncName, [&](Function &F, LoopInfo &LI, ScalarEvolution &SE) {
+ Function::iterator FI = F.begin();
+ // First basic block is entry - skip it.
+ BasicBlock *Header = &*(++FI);
+ Loop *L = LI.getLoopFor(Header);
+ EXPECT_NE(L, nullptr);
+ BasicBlock::iterator BBI = Header->begin();
+ assert((&*BBI)->getName() == "i");
+ ++BBI;
+ PHINode *Phi = dyn_cast<PHINode>(&*BBI);
+ EXPECT_NE(Phi, nullptr);
+ RecurrenceDescriptor Rdx;
+ bool IsRdxPhi =
+ RecurrenceDescriptor::isLinearRecurrencePHI(Phi, L, Rdx, &SE);
+ EXPECT_FALSE(IsRdxPhi);
+ });
+ };
+ Check("varying_coeff");
+ Check("used_in_loop");
+ Check("used_outside");
+}
+
// Make sure isReductionPHI doesn't crash when SE is not passed to it.
TEST(IVDescriptorsTest, InvariantStoreNoSCEV) {
// Parse the module.
>From 6b059a6e35f0fddbad50aafe92697ef5530c3c2b Mon Sep 17 00:00:00 2001
From: tarvlad <vladislav.tarasov at huawei.com>
Date: Fri, 28 Aug 2026 15:17:48 +0800
Subject: [PATCH 2/2] [LV] Vectorize integer linear recurrences h = C*h + x
(chunked form)
Recognize linear recurrences of the form
h = phi(start, h_next)
h_next = add(mul(h, C), x) (operands may be swapped)
where C is a loop-invariant constant and x is a simple load (e.g.
polynomial hashes like Rabin-Karp's h = 31*h + s[i], or Java
String.hashCode), and vectorize them by rewriting the loop into the
equivalent chunked form
h_next = C^VF * h + sum_l C^(VF-1-l) * x_{i+l}
per vector iteration: a vector multiply by the constant weight vector
[C^(VF-1), ..., C^1, C^0], an in-loop add reduction and a scalar seed
update. For VF = 1 the recipe degenerates to the original computation.
The rewrite is exact for wrapping integer arithmetic, because
C*(a+b) == C*a + C*b modulo 2^N.
Implementation:
- LoopVectorizationLegality recognizes the pattern via
RecurrenceDescriptor::isLinearRecurrencePHI, behind
-enable-linear-recurrence-vectorization (off by default). The
vectorizer-specific restrictions (innermost loop, constant
coefficient, simple non-volatile/non-atomic load) are enforced here,
so loops it cannot handle fail legality cleanly.
- A new VPlanTransforms::createLinearRecurrenceRecipes pass replaces
the mul/add chain with a VPLinearRecurrenceChainRecipe; the
recurrence value is carried in a scalar register across the vector
loop by a VPLinearRecurrencePHIRecipe.
- The exit value of the vector loop is the chain recipe's result; the
scalar epilogue resumes from it.
Currently not supported: tail folding (bailed out up-front), scalable
VFs (rejected via an invalid cost) and interleaving (IC clamped to 1);
these are follow-ups.
Part of the fix for https://github.com/llvm/llvm-project/issues/56999.
---
.../Vectorize/LoopVectorizationLegality.h | 9 +
.../Vectorize/LoopVectorizationLegality.cpp | 18 ++
.../Vectorize/LoopVectorizationPlanner.h | 1 +
.../Transforms/Vectorize/LoopVectorize.cpp | 39 ++++-
llvm/lib/Transforms/Vectorize/VPlan.cpp | 7 +-
llvm/lib/Transforms/Vectorize/VPlan.h | 116 ++++++++++++-
.../Vectorize/VPlanConstruction.cpp | 55 +++++-
.../lib/Transforms/Vectorize/VPlanRecipes.cpp | 119 +++++++++++++
.../Transforms/Vectorize/VPlanTransforms.h | 12 +-
llvm/lib/Transforms/Vectorize/VPlanUtils.cpp | 2 +
.../AArch64/linear-recurrence.ll | 158 ++++++++++++++++++
.../LoopVectorize/X86/linear-recurrence.ll | 158 ++++++++++++++++++
12 files changed, 685 insertions(+), 9 deletions(-)
create mode 100644 llvm/test/Transforms/LoopVectorize/AArch64/linear-recurrence.ll
create mode 100644 llvm/test/Transforms/LoopVectorize/X86/linear-recurrence.ll
diff --git a/llvm/include/llvm/Transforms/Vectorize/LoopVectorizationLegality.h b/llvm/include/llvm/Transforms/Vectorize/LoopVectorizationLegality.h
index 7b8b27c6541e1..e32248ace3203 100644
--- a/llvm/include/llvm/Transforms/Vectorize/LoopVectorizationLegality.h
+++ b/llvm/include/llvm/Transforms/Vectorize/LoopVectorizationLegality.h
@@ -333,6 +333,11 @@ class LoopVectorizationLegality {
/// Return the fixed-order recurrences found in the loop.
RecurrenceSet &getFixedOrderRecurrences() { return FixedOrderRecurrences; }
+ /// Returns the linear recurrences found in the loop.
+ const ReductionList &getLinearRecurrences() const {
+ return LinearRecurrences;
+ }
+
/// Returns the widest induction type.
IntegerType *getWidestInductionType() { return WidestIndTy; }
@@ -698,6 +703,10 @@ class LoopVectorizationLegality {
/// Holds the phi nodes that are fixed-order recurrences.
RecurrenceSet FixedOrderRecurrences;
+ /// Holds the phi nodes that are linear recurrences (h = C*h + x) with their
+ /// recurrence descriptors.
+ ReductionList LinearRecurrences;
+
/// Holds the widest induction type encountered.
IntegerType *WidestIndTy = nullptr;
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp
index f1d785b571367..9e4be6e8c5786 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp
@@ -910,6 +910,24 @@ bool LoopVectorizationLegality::canVectorizeInstr(Instruction &I) {
return true;
}
+ // Linear recurrences h = C*h + x are legal if vectorization of them is
+ // enabled. To be able to vectorize them, the coefficient C must be a
+ // constant and the per-lane value x must be a simple (non-volatile,
+ // non-atomic) load; these restrictions are enforced here so that the
+ // VPlan transform can rely on them.
+ RecurrenceDescriptor LinRecDes;
+ Value *C = nullptr, *X = nullptr;
+ if (EnableLinearRecurrenceVectorization && TheLoop->isInnermost() &&
+ RecurrenceDescriptor::isLinearRecurrencePHI(Phi, TheLoop, LinRecDes,
+ PSE.getSE(), &C, &X)) {
+ auto *LI = dyn_cast<LoadInst>(X);
+ if (isa<ConstantInt>(C) && LI && LI->isSimple()) {
+ LinearRecurrences[Phi] = std::move(LinRecDes);
+ return true;
+ }
+ // Fall through to the unidentified PHI failure below.
+ }
+
// As a last resort, coerce the PHI to a AddRec expression
// and re-try classifying it a an induction PHI.
if (InductionDescriptor::isInductionPHI(Phi, TheLoop, PSE, ID, true) &&
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
index 9ca869f5ebe88..44154c6749489 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
@@ -53,6 +53,7 @@ struct VFRange;
extern cl::opt<bool> EnableVPlanNativePath;
extern cl::opt<unsigned> ForceTargetInstructionCost;
extern cl::opt<bool> PreferInLoopReductions;
+extern cl::opt<bool> EnableLinearRecurrenceVectorization;
/// \return An upper bound for vscale based on TTI or the vscale_range
/// attribute.
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 1f153afd6bedd..dc9c3e5c0bad3 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -347,6 +347,12 @@ cl::opt<bool> llvm::EnableVPlanNativePath(
cl::desc("Enable VPlan-native vectorization path with "
"support for outer loop vectorization."));
+cl::opt<bool> llvm::EnableLinearRecurrenceVectorization(
+ "enable-linear-recurrence-vectorization", cl::init(false), cl::Hidden,
+ cl::desc("Vectorize linear recurrences of the form h = C*h + x, where C "
+ "is a loop-invariant coefficient and x is a loop-varying value "
+ "(e.g. polynomial hashes like h = 31*h + s[i])."));
+
cl::opt<bool>
llvm::VerifyEachVPlan("vplan-verify-each",
#ifdef EXPENSIVE_CHECKS
@@ -3288,6 +3294,8 @@ static bool willGenerateVectors(VPlan &Plan, ElementCount VF,
case VPRecipeBase::VPWidenIntOrFpInductionSC:
case VPRecipeBase::VPWidenPointerInductionSC:
case VPRecipeBase::VPReductionPHISC:
+ case VPRecipeBase::VPLinearRecurrencePHISC:
+ case VPRecipeBase::VPLinearRecurrenceChainSC:
case VPRecipeBase::VPInterleaveEVLSC:
case VPRecipeBase::VPInterleaveSC:
case VPRecipeBase::VPWidenLoadEVLSC:
@@ -5405,6 +5413,14 @@ void LoopVectorizationCostModel::collectValuesToIgnore() {
}
void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
+ // Linear recurrences are currently only supported without interleaving, as
+ // the scalar recurrence value cannot be threaded through unrolled parts yet.
+ if (UserIC > 1 && !Legal->getLinearRecurrences().empty()) {
+ LLVM_DEBUG(dbgs() << "LV: Ignoring requested interleave count; linear "
+ "recurrences do not support interleaving yet.\n");
+ UserIC = 0;
+ }
+
CM.collectValuesToIgnore();
Config.collectElementTypesForWidening(&CM.ValuesToIgnore);
@@ -6494,10 +6510,15 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan1() {
VPlanTransforms::createHeaderPhiRecipes, *VPlan0, PSE, *OrigLoop,
VPDT, Legal->getInductionVars(), Legal->getReductionVars(),
Legal->getFixedOrderRecurrences(), Config.getInLoopReductions(),
- Config.getHints().allowReordering())) {
+ Legal->getLinearRecurrences(), Config.getHints().allowReordering())) {
return nullptr;
}
+ // Replace the mul/add chain of each linear recurrence with the chunked
+ // computation; bail out if a recurrence cannot be handled.
+ if (!RUN_VPLAN_PASS(VPlanTransforms::createLinearRecurrenceRecipes, *VPlan0))
+ return nullptr;
+
if (const LoopAccessInfo *LAI = Legal->getLAI())
RUN_VPLAN_PASS(VPlanTransforms::replaceSymbolicStrides, *VPlan0, PSE,
LAI->getSymbolicStrides(), VPDT);
@@ -6638,6 +6659,15 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(VPlanPtr Plan,
IVUpdateMayOverflow |= !isIndvarOverflowCheckKnownFalse(&CM, VF);
TailFoldingStyle Style = CM.getTailFoldingStyle();
+ // Linear recurrences cannot be vectorized when folding the tail: the
+ // chunked computation requires full chunks (see
+ // VPLinearRecurrenceChainRecipe).
+ if (Style != TailFoldingStyle::None &&
+ !Legal->getLinearRecurrences().empty()) {
+ LLVM_DEBUG(dbgs() << "LV: Can't vectorize linear recurrences when folding "
+ "the tail, bailing out.\n");
+ return nullptr;
+ }
// Use NUW for the induction increment if we proved that it won't overflow in
// the vector loop or when not folding the tail. In the later case, we know
// that the canonical induction increment will not overflow as the vector trip
@@ -6714,7 +6744,8 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(VPlanPtr Plan,
if (isa<VPWidenCanonicalIVRecipe, VPBlendRecipe, VPReductionRecipe,
VPReplicateRecipe, VPWidenLoadRecipe, VPWidenStoreRecipe,
VPWidenCallRecipe, VPWidenIntrinsicRecipe, VPVectorPointerRecipe,
- VPVectorEndPointerRecipe, VPHistogramRecipe>(&R) ||
+ VPVectorEndPointerRecipe, VPHistogramRecipe,
+ VPLinearRecurrenceChainRecipe>(&R) ||
(isa<VPInstructionWithType>(R) &&
Instruction::isCast(cast<VPInstructionWithType>(R).getOpcode()) &&
vputils::onlyFirstLaneUsed(R.getVPSingleValue())))
@@ -8173,6 +8204,10 @@ bool LoopVectorizePass::processLoop(Loop *L) {
// Override IC if user provided an interleave count.
IC = UserIC > 0 ? UserIC : IC;
+ // Linear recurrences are currently only supported without interleaving, as
+ // the scalar recurrence value cannot be threaded through unrolled parts yet.
+ if (!LVL.getLinearRecurrences().empty())
+ IC = 1;
if (CM.maskPartialAliasing()) {
LLVM_DEBUG(
diff --git a/llvm/lib/Transforms/Vectorize/VPlan.cpp b/llvm/lib/Transforms/Vectorize/VPlan.cpp
index 87d24ed9d8d4b..0e3dbd6f40fe1 100644
--- a/llvm/lib/Transforms/Vectorize/VPlan.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlan.cpp
@@ -403,9 +403,10 @@ void VPTransformState::fixupHeaderPhis() {
for (VPRecipeBase &R : Header->phis()) {
auto *PhiR = cast<VPSingleDefRecipe>(&R);
- bool NeedsScalar =
- isa<VPPhi>(PhiR) || (isa<VPReductionPHIRecipe>(PhiR) &&
- cast<VPReductionPHIRecipe>(PhiR)->isInLoop());
+ bool NeedsScalar = isa<VPPhi>(PhiR) ||
+ isa<VPLinearRecurrencePHIRecipe>(PhiR) ||
+ (isa<VPReductionPHIRecipe>(PhiR) &&
+ cast<VPReductionPHIRecipe>(PhiR)->isInLoop());
Value *Phi = get(PhiR, NeedsScalar);
Value *Val = get(PhiR->getOperand(1), NeedsScalar);
diff --git a/llvm/lib/Transforms/Vectorize/VPlan.h b/llvm/lib/Transforms/Vectorize/VPlan.h
index 7d2c2fa1bdd23..efd3d924e56d1 100644
--- a/llvm/lib/Transforms/Vectorize/VPlan.h
+++ b/llvm/lib/Transforms/Vectorize/VPlan.h
@@ -448,6 +448,7 @@ class LLVM_ABI_FOR_TEST VPRecipeBase
VPWidenStoreEVLSC,
VPWidenStoreSC,
VPWidenSC,
+ VPLinearRecurrenceChainSC,
VPBlendSC,
VPHistogramSC,
// START: Phi-like recipes. Need to be kept together.
@@ -461,12 +462,13 @@ class LLVM_ABI_FOR_TEST VPRecipeBase
VPWidenIntOrFpInductionSC,
VPWidenPointerInductionSC,
VPReductionPHISC,
+ VPLinearRecurrencePHISC,
// END: SubclassID for recipes that inherit VPHeaderPHIRecipe
// END: Phi-like recipes
VPFirstPHISC = VPWidenPHISC,
VPFirstHeaderPHISC = VPCurrentIterationPHISC,
- VPLastHeaderPHISC = VPReductionPHISC,
- VPLastPHISC = VPReductionPHISC,
+ VPLastHeaderPHISC = VPLinearRecurrencePHISC,
+ VPLastPHISC = VPLinearRecurrencePHISC,
};
VPRecipeBase(VPRecipeTy SC, ArrayRef<VPValue *> Operands,
@@ -657,6 +659,8 @@ class LLVM_ABI_FOR_TEST VPSingleDefRecipe : public VPRecipeBase,
case VPRecipeBase::VPWidenIntOrFpInductionSC:
case VPRecipeBase::VPWidenPointerInductionSC:
case VPRecipeBase::VPReductionPHISC:
+ case VPRecipeBase::VPLinearRecurrencePHISC:
+ case VPRecipeBase::VPLinearRecurrenceChainSC:
case VPRecipeBase::VPWidenLoadEVLSC:
case VPRecipeBase::VPWidenLoadSC:
return true;
@@ -2958,6 +2962,114 @@ class VPReductionPHIRecipe : public VPHeaderPHIRecipe, public VPIRFlags {
#endif
};
+/// A recipe modeling the scalar value of a linear recurrence
+/// h = C * h + x
+/// carried in a scalar register across iterations of the vector loop, where C
+/// is a loop-invariant coefficient and x is a loop-varying value.
+class LLVM_ABI_FOR_TEST VPLinearRecurrencePHIRecipe : public VPHeaderPHIRecipe {
+public:
+ VPLinearRecurrencePHIRecipe(PHINode *Phi, VPValue &Start, VPValue &Backedge)
+ : VPHeaderPHIRecipe(VPRecipeBase::VPLinearRecurrencePHISC, Phi, &Start) {
+ addOperand(&Backedge);
+ }
+
+ ~VPLinearRecurrencePHIRecipe() override = default;
+
+ VPLinearRecurrencePHIRecipe *clone() override {
+ return new VPLinearRecurrencePHIRecipe(
+ dyn_cast_or_null<PHINode>(getUnderlyingValue()), *getStartValue(),
+ *getBackedgeValue());
+ }
+
+ VP_CLASSOF_IMPL(VPRecipeBase::VPLinearRecurrencePHISC)
+
+ /// Generate the phi nodes.
+ void execute(VPTransformState &State) override;
+
+ /// Returns true if the recipe only uses the first lane of operand \p Op.
+ bool usesFirstLaneOnly(const VPValue *Op) const override {
+ assert(is_contained(operands(), Op) &&
+ "Op must be an operand of the recipe");
+ return true;
+ }
+
+ /// Return the cost of the recipe.
+ InstructionCost computeCost(ElementCount VF,
+ VPCostContext &Ctx) const override;
+
+#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
+ /// Print the recipe.
+ void printRecipe(raw_ostream &O, const Twine &Indent,
+ VPSlotTracker &SlotTracker) const override;
+#endif
+};
+
+/// A recipe computing the next value of a linear recurrence
+/// h = C * h + x
+/// in the vector loop, using the chunked formulation
+/// h_next = C^VF * h + sum_{l=0}^{VF-1} C^(VF-1-l) * x_{i+l}
+/// where \p ChainOp is the current recurrence value h, \p VecOp is the
+/// vectorized x and \p Coeff is the loop-invariant constant coefficient C.
+/// The recipe produces a scalar value.
+class LLVM_ABI_FOR_TEST VPLinearRecurrenceChainRecipe
+ : public VPSingleDefRecipe {
+ /// The loop-invariant constant coefficient of the recurrence.
+ ConstantInt *Coeff;
+
+public:
+ VPLinearRecurrenceChainRecipe(VPValue &ChainOp, VPValue &VecOp,
+ ConstantInt *Coeff, Instruction *I,
+ DebugLoc DL = DebugLoc::getUnknown())
+ : VPSingleDefRecipe(VPRecipeBase::VPLinearRecurrenceChainSC,
+ {&ChainOp, &VecOp}, ChainOp.getScalarType(), I, DL),
+ Coeff(Coeff) {
+ assert(VecOp.getScalarType() == ChainOp.getScalarType() &&
+ "the recurrence value and the per-lane value must have the same "
+ "type");
+ }
+
+ ~VPLinearRecurrenceChainRecipe() override = default;
+
+ VPLinearRecurrenceChainRecipe *clone() override {
+ return new VPLinearRecurrenceChainRecipe(
+ getChainOp(), getVecOp(), Coeff,
+ dyn_cast_or_null<Instruction>(getUnderlyingValue()), getDebugLoc());
+ }
+
+ VP_CLASSOF_IMPL(VPRecipeBase::VPLinearRecurrenceChainSC)
+
+ /// Generate the recipe.
+ void execute(VPTransformState &State) override;
+
+ /// Returns the current recurrence value operand.
+ VPValue &getChainOp() { return *getOperand(0); }
+ VPValue &getChainOp() const { return *getOperand(0); }
+
+ /// Returns the vectorized per-lane value operand.
+ VPValue &getVecOp() { return *getOperand(1); }
+ VPValue &getVecOp() const { return *getOperand(1); }
+
+ /// Returns the loop-invariant constant coefficient.
+ ConstantInt *getCoefficient() const { return Coeff; }
+
+ /// Return the cost of the recipe.
+ InstructionCost computeCost(ElementCount VF,
+ VPCostContext &Ctx) const override;
+
+ /// Returns true if the recipe only uses the first lane of operand \p Op.
+ bool usesFirstLaneOnly(const VPValue *Op) const override {
+ assert(is_contained(operands(), Op) &&
+ "Op must be an operand of the recipe");
+ return Op == getOperand(0);
+ }
+
+#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
+ /// Print the recipe.
+ void printRecipe(raw_ostream &O, const Twine &Indent,
+ VPSlotTracker &SlotTracker) const override;
+#endif
+};
+
/// A recipe for vectorizing a phi-node as a sequence of mask-based select
/// instructions.
class LLVM_ABI_FOR_TEST VPBlendRecipe : public VPRecipeWithIRFlags {
diff --git a/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp b/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
index 7f748960b1d8c..adb91266115df 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
@@ -911,7 +911,9 @@ bool VPlanTransforms::createHeaderPhiRecipes(
const MapVector<PHINode *, InductionDescriptor> &Inductions,
const MapVector<PHINode *, RecurrenceDescriptor> &Reductions,
const SmallPtrSetImpl<const PHINode *> &FixedOrderRecurrences,
- const SmallPtrSetImpl<PHINode *> &InLoopReductions, bool AllowReordering) {
+ const SmallPtrSetImpl<PHINode *> &InLoopReductions,
+ const MapVector<PHINode *, RecurrenceDescriptor> &LinearRecurrences,
+ bool AllowReordering) {
// Retrieve the header manually from the intial plain-CFG VPlan.
auto [HeaderVPBB, LatchVPBB] = VPBlockUtils::getPlainCFGHeaderAndLatch(Plan);
assert(VPDT.dominates(HeaderVPBB, LatchVPBB) &&
@@ -936,6 +938,9 @@ bool VPlanTransforms::createHeaderPhiRecipes(
return new VPFirstOrderRecurrencePHIRecipe(Phi, *Start, *BackedgeValue);
}
+ if (LinearRecurrences.contains(Phi))
+ return new VPLinearRecurrencePHIRecipe(Phi, *Start, *BackedgeValue);
+
auto InductionIt = Inductions.find(Phi);
if (InductionIt != Inductions.end())
return createWidenInductionRecipe(Phi, PhiR, Start, InductionIt->second,
@@ -1199,6 +1204,54 @@ void VPlanTransforms::createInLoopReductionRecipes(VPlan &Plan,
R->eraseFromParent();
}
+bool VPlanTransforms::createLinearRecurrenceRecipes(VPlan &Plan) {
+ auto [HeaderVPBB, _] = VPBlockUtils::getPlainCFGHeaderAndLatch(Plan);
+ for (VPRecipeBase &R : make_early_inc_range(HeaderVPBB->phis())) {
+ auto *PhiR = dyn_cast<VPLinearRecurrencePHIRecipe>(&R);
+ if (!PhiR)
+ continue;
+
+ // The phi must only be used by the multiply computing C * h, which in
+ // turn must only be used by the add computing C * h + x.
+ auto *MulVPI = dyn_cast<VPInstruction>(*PhiR->user_begin());
+ if (!PhiR->hasOneUse() || !MulVPI ||
+ MulVPI->getOpcode() != Instruction::Mul)
+ return false;
+ if (!MulVPI->hasOneUse())
+ return false;
+ auto *AddVPI = dyn_cast<VPInstruction>(*MulVPI->user_begin());
+ if (!AddVPI || AddVPI->getOpcode() != Instruction::Add)
+ return false;
+
+ // Determine the constant coefficient C and the per-lane value operand x.
+ auto *Phi = cast<PHINode>(PhiR->getUnderlyingValue());
+ auto *MulI = cast<Instruction>(MulVPI->getUnderlyingValue());
+ auto *C = dyn_cast<ConstantInt>(
+ MulI->getOperand(0) == Phi ? MulI->getOperand(1) : MulI->getOperand(0));
+ if (!C)
+ return false;
+ VPValue *X = AddVPI->getOperand(0) == MulVPI ? AddVPI->getOperand(1)
+ : AddVPI->getOperand(0);
+ // The per-lane value must be a load that will be widened to a vector load
+ // later; this also guarantees it dominates the chain recipe.
+ auto *LoadVPI = dyn_cast_or_null<VPInstruction>(X->getDefiningRecipe());
+ if (!LoadVPI || LoadVPI->getOpcode() != Instruction::Load ||
+ LoadVPI->isMasked())
+ return false;
+
+ // Replace the mul/add chain with the chunked computation
+ // h_next = C^VF * h + sum_l C^(VF-1-l) * x_{i+l}.
+ auto *AddI = cast<Instruction>(AddVPI->getUnderlyingValue());
+ auto *Chain = new VPLinearRecurrenceChainRecipe(*PhiR, *X, C, AddI,
+ AddI->getDebugLoc());
+ Chain->insertBefore(AddVPI);
+ AddVPI->replaceAllUsesWith(Chain);
+ AddVPI->eraseFromParent();
+ MulVPI->eraseFromParent();
+ }
+ return true;
+}
+
bool VPlanTransforms::areAllLoadsDereferenceable(VPBasicBlock *HeaderVPBB,
Loop *TheLoop,
PredicatedScalarEvolution &PSE,
diff --git a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
index 047bcb3a0abdb..207a939c82dc5 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
@@ -89,6 +89,8 @@ bool VPRecipeBase::mayWriteToMemory() const {
case VPDerivedIVSC:
case VPFirstOrderRecurrencePHISC:
case VPReductionPHISC:
+ case VPLinearRecurrencePHISC:
+ case VPLinearRecurrenceChainSC:
case VPScalarIVStepsSC:
case VPPredInstPHISC:
case VPExpandSCEVSC:
@@ -142,6 +144,8 @@ bool VPRecipeBase::mayReadFromMemory() const {
case VPCurrentIterationPHISC:
case VPFirstOrderRecurrencePHISC:
case VPReductionPHISC:
+ case VPLinearRecurrencePHISC:
+ case VPLinearRecurrenceChainSC:
case VPPredInstPHISC:
case VPScalarIVStepsSC:
case VPWidenStoreEVLSC:
@@ -181,6 +185,8 @@ bool VPRecipeBase::mayHaveSideEffects() const {
case VPCurrentIterationPHISC:
case VPFirstOrderRecurrencePHISC:
case VPReductionPHISC:
+ case VPLinearRecurrencePHISC:
+ case VPLinearRecurrenceChainSC:
case VPPredInstPHISC:
case VPVectorEndPointerSC:
case VPExpandSCEVSC:
@@ -2708,6 +2714,9 @@ static void printRecurrenceKind(raw_ostream &OS, const RecurKind &Kind) {
case RecurKind::FMulAdd:
OS << "fmuladd";
break;
+ case RecurKind::IntLinear:
+ OS << "int-linear";
+ break;
case RecurKind::AnyOf:
OS << "any-of";
break;
@@ -4976,6 +4985,116 @@ VPFirstOrderRecurrencePHIRecipe::computeCost(ElementCount VF,
return 0;
}
+void VPLinearRecurrencePHIRecipe::execute(VPTransformState &State) {
+ // The linear recurrence value is carried in a scalar register across the
+ // vector loop. Create a scalar PHI in the vector loop header, with the
+ // backedge added later by VPTransformState::fixupHeaderPhis.
+ BasicBlock *VectorPH =
+ State.CFG.VPBB2IRBB.at(getParent()->getCFGPredecessor(0));
+ Value *Start = State.get(getStartValue(), /*IsScalar=*/true);
+
+ BasicBlock *HeaderBB = State.CFG.PrevBB;
+ assert(State.CurrentParentLoop->getHeader() == HeaderBB &&
+ "recipe must be in the vector loop header");
+ PHINode *Phi = PHINode::Create(Start->getType(), 2, "vector.linear.phi");
+ Phi->insertBefore(HeaderBB->getFirstInsertionPt());
+ State.set(this, Phi, /*IsScalar=*/true);
+
+ Phi->addIncoming(Start, VectorPH);
+}
+
+InstructionCost
+VPLinearRecurrencePHIRecipe::computeCost(ElementCount VF,
+ VPCostContext &Ctx) const {
+ // The recurrence value is kept in a scalar register.
+ return Ctx.TTI.getCFInstrCost(Instruction::PHI, Ctx.CostKind);
+}
+
+void VPLinearRecurrenceChainRecipe::execute(VPTransformState &State) {
+ Value *X = State.get(&getVecOp());
+ Value *Prev = State.get(&getChainOp(), /*IsScalar=*/true);
+ Type *Ty = getScalarType();
+ ConstantInt *C = getCoefficient();
+
+ Value *Next;
+ if (State.VF.isVector()) {
+ unsigned VF = State.VF.getKnownMinValue();
+ assert(!State.VF.isScalable() &&
+ "scalable VFs are not supported for linear recurrences yet");
+ // Build the weight vector [C^(VF-1), ..., C^1, C^0].
+ SmallVector<Constant *, 8> Weights(VF);
+ APInt Power(C->getValue());
+ Weights[VF - 1] = ConstantInt::get(Ty, 1);
+ for (unsigned L = 1; L != VF; ++L) {
+ Weights[VF - 1 - L] = ConstantInt::get(Ty, Power);
+ Power *= C->getValue();
+ }
+ // Power now holds C^VF for the seed update.
+ auto *CVF = cast<ConstantInt>(ConstantInt::get(Ty, Power));
+
+ // h_next = h * C^VF + sum_l C^(VF-1-l) * x_{i+l}
+ Value *W = State.Builder.CreateMul(X, ConstantVector::get(Weights));
+ Value *R = createSimpleReduction(State.Builder, W, RecurKind::Add);
+ Value *Scaled = State.Builder.CreateMul(Prev, CVF);
+ Next = State.Builder.CreateAdd(Scaled, R);
+ } else {
+ // For scalar VFs the recipe degenerates to the original computation
+ // h_next = h * C + x.
+ Value *Scaled = State.Builder.CreateMul(Prev, C);
+ Next = State.Builder.CreateAdd(Scaled, X);
+ }
+ State.set(this, Next, /*IsScalar=*/true);
+}
+
+InstructionCost
+VPLinearRecurrenceChainRecipe::computeCost(ElementCount VF,
+ VPCostContext &Ctx) const {
+ // Scalable VFs are not supported yet; the invalid cost rejects them.
+ if (VF.isScalable())
+ return InstructionCost::getInvalid();
+
+ if (VF.isScalar()) {
+ // Matches the original scalar computation: h_next = h * C + x.
+ return Ctx.TTI.getArithmeticInstrCost(Instruction::Mul, getScalarType(),
+ Ctx.CostKind) +
+ Ctx.TTI.getArithmeticInstrCost(Instruction::Add, getScalarType(),
+ Ctx.CostKind);
+ }
+
+ // Cost of the chunked computation:
+ // w = x * weights (vector multiply)
+ // r = sum_l w_l (in-loop add reduction)
+ // h_next = h * C^VF + r (scalar mul + add)
+ VectorType *VecTy = VectorType::get(getScalarType(), VF);
+ InstructionCost Cost =
+ Ctx.TTI.getArithmeticInstrCost(Instruction::Mul, VecTy, Ctx.CostKind);
+ Cost += Ctx.TTI.getArithmeticReductionCost(Instruction::Add, VecTy,
+ std::nullopt, Ctx.CostKind);
+ Cost += Ctx.TTI.getArithmeticInstrCost(Instruction::Mul, getScalarType(),
+ Ctx.CostKind);
+ Cost += Ctx.TTI.getArithmeticInstrCost(Instruction::Add, getScalarType(),
+ Ctx.CostKind);
+ return Cost;
+}
+
+#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
+void VPLinearRecurrencePHIRecipe::printRecipe(
+ raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const {
+ O << Indent << "LINEAR-RECURRENCE-PHI ";
+ printAsOperand(O, SlotTracker);
+ O << " = phi ";
+ printOperands(O, SlotTracker);
+}
+
+void VPLinearRecurrenceChainRecipe::printRecipe(
+ raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const {
+ O << Indent << "EMIT ";
+ printAsOperand(O, SlotTracker);
+ O << " = LINEAR-RECURRENCE-CHAIN (coeff=" << *getCoefficient() << ")";
+ printOperands(O, SlotTracker);
+}
+#endif
+
#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
void VPFirstOrderRecurrencePHIRecipe::printRecipe(
raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const {
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.h b/llvm/lib/Transforms/Vectorize/VPlanTransforms.h
index a3922a858a3f5..fe1c689928ac1 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.h
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.h
@@ -171,7 +171,17 @@ struct VPlanTransforms {
const MapVector<PHINode *, InductionDescriptor> &Inductions,
const MapVector<PHINode *, RecurrenceDescriptor> &Reductions,
const SmallPtrSetImpl<const PHINode *> &FixedOrderRecurrences,
- const SmallPtrSetImpl<PHINode *> &InLoopReductions, bool AllowReordering);
+ const SmallPtrSetImpl<PHINode *> &InLoopReductions,
+ const MapVector<PHINode *, RecurrenceDescriptor> &LinearRecurrences,
+ bool AllowReordering);
+
+ /// Replace the chain of recipes computing the next value of each linear
+ /// recurrence h = C*h + x in \p Plan with a VPLinearRecurrenceChainRecipe,
+ /// which computes the chunked formulation
+ /// h_next = C^VF * h + sum_l C^(VF-1-l) * x_{i+l}
+ /// in the vector loop. Returns false if a linear recurrence cannot be
+ /// handled.
+ static bool createLinearRecurrenceRecipes(VPlan &Plan);
/// Finalize SCEV predicates by adding induction predicates from \p Plan to
/// \p PSE and checking constraints. Returns false if predicated IVs have
diff --git a/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp b/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
index b2a91b701fcaa..b97c6a4f4f0de 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
@@ -444,6 +444,8 @@ bool vputils::isSingleScalar(const VPValue *VPV) {
all_of(VPI->operands(), isSingleScalar));
if (auto *RR = dyn_cast<VPReductionRecipe>(VPV))
return !RR->isPartialReduction();
+ if (isa<VPLinearRecurrenceChainRecipe, VPLinearRecurrencePHIRecipe>(VPV))
+ return true;
if (isa<VPVectorPointerRecipe, VPVectorEndPointerRecipe, VPDerivedIVRecipe>(
VPV))
return true;
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/linear-recurrence.ll b/llvm/test/Transforms/LoopVectorize/AArch64/linear-recurrence.ll
new file mode 100644
index 0000000000000..46cebc55f8d8c
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/linear-recurrence.ll
@@ -0,0 +1,158 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 5
+; RUN: opt -passes=loop-vectorize -enable-linear-recurrence-vectorization -force-vector-width=4 -force-vector-interleave=1 -S %s | FileCheck %s
+
+target datalayout = "e-m:e-i64:64-i128:128-n32:64-S128"
+target triple = "aarch64-unknown-linux-gnu"
+
+; https://github.com/llvm/llvm-project/issues/56999
+define i32 @rabin_karp(ptr %s, i64 %n) {
+; CHECK-LABEL: define i32 @rabin_karp(
+; CHECK-SAME: ptr [[S:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 4
+; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: [[TMP1:%.*]] = and i64 [[TMP0]], 3
+; CHECK-NEXT: [[N_VEC_2:%.*]] = sub i64 [[TMP0]], [[TMP1]]
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[H:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[H_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[GEP:%.*]] = getelementptr inbounds i32, ptr [[S]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[GEP]], align 4
+; CHECK-NEXT: [[MUL:%.*]] = mul <4 x i32> [[WIDE_LOAD]], <i32 29791, i32 961, i32 31, i32 1>
+; CHECK-NEXT: [[RDX:%.*]] = call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> [[MUL]])
+; CHECK-NEXT: [[SCALED:%.*]] = mul i32 [[H]], 923521
+; CHECK-NEXT: [[H_NEXT]] = add i32 [[SCALED]], [[RDX]]
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT: [[CMP:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC_2]]
+; CHECK-NEXT: br i1 [[CMP]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC_2]]
+; CHECK-NEXT: br i1 [[CMP_N]], label %[[FOR_END:.*]], label %[[SCALAR_PH]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC_2]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT: [[BC_MERGE_RDX:%.*]] = phi i32 [ [[H_NEXT]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT: br label %[[FOR_BODY:.*]]
+; CHECK: [[FOR_BODY]]:
+; CHECK-NEXT: [[I:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[I_NEXT:%.*]], %[[FOR_BODY]] ]
+; CHECK-NEXT: [[H1:%.*]] = phi i32 [ [[BC_MERGE_RDX]], %[[SCALAR_PH]] ], [ [[H_NEXT1:%.*]], %[[FOR_BODY]] ]
+; CHECK-NEXT: [[GEP1:%.*]] = getelementptr inbounds i32, ptr [[S]], i64 [[I]]
+; CHECK-NEXT: [[LD:%.*]] = load i32, ptr [[GEP1]], align 4
+; CHECK-NEXT: [[MUL1:%.*]] = mul i32 [[H1]], 31
+; CHECK-NEXT: [[H_NEXT1]] = add i32 [[LD]], [[MUL1]]
+; CHECK-NEXT: [[I_NEXT]] = add nsw i64 [[I]], 1
+; CHECK-NEXT: [[CMP1:%.*]] = icmp slt i64 [[I_NEXT]], [[N]]
+; CHECK-NEXT: br i1 [[CMP1]], label %[[FOR_BODY]], label %[[FOR_END]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK: [[FOR_END]]:
+; CHECK-NEXT: [[H_LCSSA:%.*]] = phi i32 [ [[H_NEXT1]], %[[FOR_BODY]] ], [ [[H_NEXT]], %[[MIDDLE_BLOCK]] ]
+; CHECK-NEXT: ret i32 [[H_LCSSA]]
+;
+entry:
+ br label %for.body
+
+for.body:
+ %i = phi i64 [ 0, %entry ], [ %i.next, %for.body ]
+ %h = phi i32 [ 0, %entry ], [ %h.next, %for.body ]
+ %gep = getelementptr inbounds i32, ptr %s, i64 %i
+ %ld = load i32, ptr %gep, align 4
+ %mul = mul i32 %h, 31
+ %h.next = add i32 %ld, %mul
+ %i.next = add nsw i64 %i, 1
+ %cmp = icmp slt i64 %i.next, %n
+ br i1 %cmp, label %for.body, label %for.end
+
+for.end:
+ %h.lcssa = phi i32 [ %h.next, %for.body ]
+ ret i32 %h.lcssa
+}
+
+; The coefficient must be loop-invariant.
+define i32 @varying_coeff(ptr %s, ptr %cptr, i64 %n) {
+; CHECK-LABEL: define i32 @varying_coeff(
+; CHECK-SAME: ptr [[S:%.*]], ptr [[CPTR:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: br label %[[FOR_BODY:.*]]
+; CHECK: [[FOR_BODY]]:
+; CHECK-NEXT: [[I:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[I_NEXT:%.*]], %[[FOR_BODY]] ]
+; CHECK-NEXT: [[H:%.*]] = phi i32 [ 0, %[[ENTRY]] ], [ [[H_NEXT:%.*]], %[[FOR_BODY]] ]
+; CHECK-NEXT: [[C:%.*]] = load i32, ptr [[CPTR]], align 4
+; CHECK-NEXT: [[GEP:%.*]] = getelementptr inbounds i32, ptr [[S]], i64 [[I]]
+; CHECK-NEXT: [[LD:%.*]] = load i32, ptr [[GEP]], align 4
+; CHECK-NEXT: [[MUL:%.*]] = mul i32 [[H]], [[C]]
+; CHECK-NEXT: [[H_NEXT]] = add i32 [[LD]], [[MUL]]
+; CHECK-NEXT: [[I_NEXT]] = add nsw i64 [[I]], 1
+; CHECK-NEXT: [[CMP:%.*]] = icmp slt i64 [[I_NEXT]], [[N]]
+; CHECK-NEXT: br i1 [[CMP]], label %[[FOR_BODY]], label %[[FOR_END:.*]]
+; CHECK: [[FOR_END]]:
+; CHECK-NEXT: [[H_LCSSA:%.*]] = phi i32 [ [[H_NEXT]], %[[FOR_BODY]] ]
+; CHECK-NEXT: ret i32 [[H_LCSSA]]
+;
+entry:
+ br label %for.body
+
+for.body:
+ %i = phi i64 [ 0, %entry ], [ %i.next, %for.body ]
+ %h = phi i32 [ 0, %entry ], [ %h.next, %for.body ]
+ %c = load i32, ptr %cptr, align 4
+ %gep = getelementptr inbounds i32, ptr %s, i64 %i
+ %ld = load i32, ptr %gep, align 4
+ %mul = mul i32 %h, %c
+ %h.next = add i32 %ld, %mul
+ %i.next = add nsw i64 %i, 1
+ %cmp = icmp slt i64 %i.next, %n
+ br i1 %cmp, label %for.body, label %for.end
+
+for.end:
+ %h.lcssa = phi i32 [ %h.next, %for.body ]
+ ret i32 %h.lcssa
+}
+
+; Uses of the recurrence value inside the loop other than the mul/add chain
+; are not supported.
+define i32 @used_in_loop(ptr %s, ptr %dst, i64 %n) {
+; CHECK-LABEL: define i32 @used_in_loop(
+; CHECK-SAME: ptr [[S:%.*]], ptr [[DST:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: br label %[[FOR_BODY:.*]]
+; CHECK: [[FOR_BODY]]:
+; CHECK-NEXT: [[I:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[I_NEXT:%.*]], %[[FOR_BODY]] ]
+; CHECK-NEXT: [[H:%.*]] = phi i32 [ 0, %[[ENTRY]] ], [ [[H_NEXT:%.*]], %[[FOR_BODY]] ]
+; CHECK-NEXT: [[GEP:%.*]] = getelementptr inbounds i32, ptr [[S]], i64 [[I]]
+; CHECK-NEXT: [[LD:%.*]] = load i32, ptr [[GEP]], align 4
+; CHECK-NEXT: [[MUL:%.*]] = mul i32 [[H]], 31
+; CHECK-NEXT: [[H_NEXT]] = add i32 [[LD]], [[MUL]]
+; CHECK-NEXT: store i32 [[H]], ptr [[DST]], align 4
+; CHECK-NEXT: [[I_NEXT]] = add nsw i64 [[I]], 1
+; CHECK-NEXT: [[CMP:%.*]] = icmp slt i64 [[I_NEXT]], [[N]]
+; CHECK-NEXT: br i1 [[CMP]], label %[[FOR_BODY]], label %[[FOR_END:.*]]
+; CHECK: [[FOR_END]]:
+; CHECK-NEXT: [[H_LCSSA:%.*]] = phi i32 [ [[H_NEXT]], %[[FOR_BODY]] ]
+; CHECK-NEXT: ret i32 [[H_LCSSA]]
+;
+entry:
+ br label %for.body
+
+for.body:
+ %i = phi i64 [ 0, %entry ], [ %i.next, %for.body ]
+ %h = phi i32 [ 0, %entry ], [ %h.next, %for.body ]
+ %gep = getelementptr inbounds i32, ptr %s, i64 %i
+ %ld = load i32, ptr %gep, align 4
+ %mul = mul i32 %h, 31
+ %h.next = add i32 %ld, %mul
+ store i32 %h, ptr %dst, align 4
+ %i.next = add nsw i64 %i, 1
+ %cmp = icmp slt i64 %i.next, %n
+ br i1 %cmp, label %for.body, label %for.end
+
+for.end:
+ %h.lcssa = phi i32 [ %h.next, %for.body ]
+ ret i32 %h.lcssa
+}
+;.
+; CHECK: [[LOOP0]] = distinct !{[[LOOP0]], [[META1:![0-9]+]], [[META2:![0-9]+]]}
+; CHECK: [[META1]] = !{!"llvm.loop.isvectorized", i32 1}
+; CHECK: [[META2]] = !{!"llvm.loop.unroll.runtime.disable"}
+; CHECK: [[LOOP3]] = distinct !{[[LOOP3]], [[META2]], [[META1]]}
+;.
diff --git a/llvm/test/Transforms/LoopVectorize/X86/linear-recurrence.ll b/llvm/test/Transforms/LoopVectorize/X86/linear-recurrence.ll
new file mode 100644
index 0000000000000..3129950fba311
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/X86/linear-recurrence.ll
@@ -0,0 +1,158 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 5
+; RUN: opt -passes=loop-vectorize -enable-linear-recurrence-vectorization -force-vector-width=4 -force-vector-interleave=1 -S %s | FileCheck %s
+
+target datalayout = "e-m:e-p:64:64-i64:64-i128:128-n32:64-S128"
+target triple = "x86_64-unknown-linux-gnu"
+
+; https://github.com/llvm/llvm-project/issues/56999
+define i32 @rabin_karp(ptr %s, i64 %n) {
+; CHECK-LABEL: define i32 @rabin_karp(
+; CHECK-SAME: ptr [[S:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 4
+; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: [[TMP1:%.*]] = and i64 [[TMP0]], 3
+; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[H:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[H_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[GEP:%.*]] = getelementptr inbounds i32, ptr [[S]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[GEP]], align 4
+; CHECK-NEXT: [[MUL:%.*]] = mul <4 x i32> [[WIDE_LOAD]], <i32 29791, i32 961, i32 31, i32 1>
+; CHECK-NEXT: [[RDX:%.*]] = call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> [[MUL]])
+; CHECK-NEXT: [[SCALED:%.*]] = mul i32 [[H]], 923521
+; CHECK-NEXT: [[H_NEXT]] = add i32 [[SCALED]], [[RDX]]
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT: [[CMP:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[CMP]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[CMP_N]], label %[[FOR_END:.*]], label %[[SCALAR_PH]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT: [[BC_MERGE_RDX:%.*]] = phi i32 [ [[H_NEXT]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT: br label %[[FOR_BODY:.*]]
+; CHECK: [[FOR_BODY]]:
+; CHECK-NEXT: [[I:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[I_NEXT:%.*]], %[[FOR_BODY]] ]
+; CHECK-NEXT: [[H1:%.*]] = phi i32 [ [[BC_MERGE_RDX]], %[[SCALAR_PH]] ], [ [[H_NEXT1:%.*]], %[[FOR_BODY]] ]
+; CHECK-NEXT: [[GEP1:%.*]] = getelementptr inbounds i32, ptr [[S]], i64 [[I]]
+; CHECK-NEXT: [[LD:%.*]] = load i32, ptr [[GEP1]], align 4
+; CHECK-NEXT: [[MUL1:%.*]] = mul i32 [[H1]], 31
+; CHECK-NEXT: [[H_NEXT1]] = add i32 [[LD]], [[MUL1]]
+; CHECK-NEXT: [[I_NEXT]] = add nsw i64 [[I]], 1
+; CHECK-NEXT: [[CMP1:%.*]] = icmp slt i64 [[I_NEXT]], [[N]]
+; CHECK-NEXT: br i1 [[CMP1]], label %[[FOR_BODY]], label %[[FOR_END]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK: [[FOR_END]]:
+; CHECK-NEXT: [[H_LCSSA:%.*]] = phi i32 [ [[H_NEXT1]], %[[FOR_BODY]] ], [ [[H_NEXT]], %[[MIDDLE_BLOCK]] ]
+; CHECK-NEXT: ret i32 [[H_LCSSA]]
+;
+entry:
+ br label %for.body
+
+for.body:
+ %i = phi i64 [ 0, %entry ], [ %i.next, %for.body ]
+ %h = phi i32 [ 0, %entry ], [ %h.next, %for.body ]
+ %gep = getelementptr inbounds i32, ptr %s, i64 %i
+ %ld = load i32, ptr %gep, align 4
+ %mul = mul i32 %h, 31
+ %h.next = add i32 %ld, %mul
+ %i.next = add nsw i64 %i, 1
+ %cmp = icmp slt i64 %i.next, %n
+ br i1 %cmp, label %for.body, label %for.end
+
+for.end:
+ %h.lcssa = phi i32 [ %h.next, %for.body ]
+ ret i32 %h.lcssa
+}
+
+; The coefficient must be loop-invariant.
+define i32 @varying_coeff(ptr %s, ptr %cptr, i64 %n) {
+; CHECK-LABEL: define i32 @varying_coeff(
+; CHECK-SAME: ptr [[S:%.*]], ptr [[CPTR:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: br label %[[FOR_BODY:.*]]
+; CHECK: [[FOR_BODY]]:
+; CHECK-NEXT: [[I:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[I_NEXT:%.*]], %[[FOR_BODY]] ]
+; CHECK-NEXT: [[H:%.*]] = phi i32 [ 0, %[[ENTRY]] ], [ [[H_NEXT:%.*]], %[[FOR_BODY]] ]
+; CHECK-NEXT: [[C:%.*]] = load i32, ptr [[CPTR]], align 4
+; CHECK-NEXT: [[GEP:%.*]] = getelementptr inbounds i32, ptr [[S]], i64 [[I]]
+; CHECK-NEXT: [[LD:%.*]] = load i32, ptr [[GEP]], align 4
+; CHECK-NEXT: [[MUL:%.*]] = mul i32 [[H]], [[C]]
+; CHECK-NEXT: [[H_NEXT]] = add i32 [[LD]], [[MUL]]
+; CHECK-NEXT: [[I_NEXT]] = add nsw i64 [[I]], 1
+; CHECK-NEXT: [[CMP:%.*]] = icmp slt i64 [[I_NEXT]], [[N]]
+; CHECK-NEXT: br i1 [[CMP]], label %[[FOR_BODY]], label %[[FOR_END:.*]]
+; CHECK: [[FOR_END]]:
+; CHECK-NEXT: [[H_LCSSA:%.*]] = phi i32 [ [[H_NEXT]], %[[FOR_BODY]] ]
+; CHECK-NEXT: ret i32 [[H_LCSSA]]
+;
+entry:
+ br label %for.body
+
+for.body:
+ %i = phi i64 [ 0, %entry ], [ %i.next, %for.body ]
+ %h = phi i32 [ 0, %entry ], [ %h.next, %for.body ]
+ %c = load i32, ptr %cptr, align 4
+ %gep = getelementptr inbounds i32, ptr %s, i64 %i
+ %ld = load i32, ptr %gep, align 4
+ %mul = mul i32 %h, %c
+ %h.next = add i32 %ld, %mul
+ %i.next = add nsw i64 %i, 1
+ %cmp = icmp slt i64 %i.next, %n
+ br i1 %cmp, label %for.body, label %for.end
+
+for.end:
+ %h.lcssa = phi i32 [ %h.next, %for.body ]
+ ret i32 %h.lcssa
+}
+
+; Uses of the recurrence value inside the loop other than the mul/add chain
+; are not supported.
+define i32 @used_in_loop(ptr %s, ptr %dst, i64 %n) {
+; CHECK-LABEL: define i32 @used_in_loop(
+; CHECK-SAME: ptr [[S:%.*]], ptr [[DST:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: br label %[[FOR_BODY:.*]]
+; CHECK: [[FOR_BODY]]:
+; CHECK-NEXT: [[I:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[I_NEXT:%.*]], %[[FOR_BODY]] ]
+; CHECK-NEXT: [[H:%.*]] = phi i32 [ 0, %[[ENTRY]] ], [ [[H_NEXT:%.*]], %[[FOR_BODY]] ]
+; CHECK-NEXT: [[GEP:%.*]] = getelementptr inbounds i32, ptr [[S]], i64 [[I]]
+; CHECK-NEXT: [[LD:%.*]] = load i32, ptr [[GEP]], align 4
+; CHECK-NEXT: [[MUL:%.*]] = mul i32 [[H]], 31
+; CHECK-NEXT: [[H_NEXT]] = add i32 [[LD]], [[MUL]]
+; CHECK-NEXT: store i32 [[H]], ptr [[DST]], align 4
+; CHECK-NEXT: [[I_NEXT]] = add nsw i64 [[I]], 1
+; CHECK-NEXT: [[CMP:%.*]] = icmp slt i64 [[I_NEXT]], [[N]]
+; CHECK-NEXT: br i1 [[CMP]], label %[[FOR_BODY]], label %[[FOR_END:.*]]
+; CHECK: [[FOR_END]]:
+; CHECK-NEXT: [[H_LCSSA:%.*]] = phi i32 [ [[H_NEXT]], %[[FOR_BODY]] ]
+; CHECK-NEXT: ret i32 [[H_LCSSA]]
+;
+entry:
+ br label %for.body
+
+for.body:
+ %i = phi i64 [ 0, %entry ], [ %i.next, %for.body ]
+ %h = phi i32 [ 0, %entry ], [ %h.next, %for.body ]
+ %gep = getelementptr inbounds i32, ptr %s, i64 %i
+ %ld = load i32, ptr %gep, align 4
+ %mul = mul i32 %h, 31
+ %h.next = add i32 %ld, %mul
+ store i32 %h, ptr %dst, align 4
+ %i.next = add nsw i64 %i, 1
+ %cmp = icmp slt i64 %i.next, %n
+ br i1 %cmp, label %for.body, label %for.end
+
+for.end:
+ %h.lcssa = phi i32 [ %h.next, %for.body ]
+ ret i32 %h.lcssa
+}
+;.
+; CHECK: [[LOOP0]] = distinct !{[[LOOP0]], [[META1:![0-9]+]], [[META2:![0-9]+]]}
+; CHECK: [[META1]] = !{!"llvm.loop.isvectorized", i32 1}
+; CHECK: [[META2]] = !{!"llvm.loop.unroll.runtime.disable"}
+; CHECK: [[LOOP3]] = distinct !{[[LOOP3]], [[META2]], [[META1]]}
+;.
More information about the llvm-commits
mailing list