[llvm] [LV] Simple Linear Recurrence Vectorization (PR #219440)

Vladislav Tarasov via llvm-commits llvm-commits at lists.llvm.org
Fri Aug 28 04:06:49 PDT 2026


https://github.com/tarvlad created https://github.com/llvm/llvm-project/pull/219440

None

>From babc567bda1ede963a9a5850a1469d0d3baae644 Mon Sep 17 00:00:00 2001
From: tarvlad <vladislav.tarasov at huawei.com>
Date: Fri, 28 Aug 2026 15:17:37 +0800
Subject: [PATCH 1/2] [IVDescriptors] Add RecurKind::IntLinear and
 isLinearRecurrencePHI matcher

Add RecurKind::IntLinear describing an integer linear recurrence
h = C*h + x, where C is loop-invariant and x is a loop-varying value
that does not depend on the recurrence (e.g. a polynomial hash like
Rabin-Karp's h = 31*h + s[i]).

Add RecurrenceDescriptor::isLinearRecurrencePHI matching the pattern
add(mul(phi, C), x), restricting phi to a single in-loop use in the
chain and outside uses only through the exit value. The coefficient C
and per-lane value x can be retrieved via optional out-parameters.
Update the SLP horizontal-reduction switches for the new kind.

The matcher is not used yet (NFC); a follow-up patch will wire it into
the loop vectorizer to help address llvm/llvm-project#56999.
---
 llvm/include/llvm/Analysis/IVDescriptors.h    |  15 ++
 llvm/lib/Analysis/IVDescriptors.cpp           |  94 ++++++++++++
 .../Transforms/Vectorize/SLPVectorizer.cpp    |   3 +
 llvm/unittests/Analysis/IVDescriptorsTest.cpp | 142 ++++++++++++++++++
 4 files changed, 254 insertions(+)

diff --git a/llvm/include/llvm/Analysis/IVDescriptors.h b/llvm/include/llvm/Analysis/IVDescriptors.h
index bad372421dfae..db6153f4bc0b7 100644
--- a/llvm/include/llvm/Analysis/IVDescriptors.h
+++ b/llvm/include/llvm/Analysis/IVDescriptors.h
@@ -39,6 +39,8 @@ enum class RecurKind {
   Sub,      ///< Subtraction of integers
   AddChainWithSubs, ///< A chain of adds and subs
   Mul,      ///< Product of integers.
+  IntLinear, ///< Linear recurrence of integers: h = C*h + x, where C is a
+             ///< loop-invariant coefficient and x is a loop-varying value.
   Or,       ///< Bitwise or logical OR of integers.
   And,      ///< Bitwise or logical AND of integers.
   Xor,      ///< Bitwise or logical XOR of integers.
@@ -225,6 +227,19 @@ class RecurrenceDescriptor {
   LLVM_ABI static bool isFixedOrderRecurrence(PHINode *Phi, Loop *TheLoop,
                                               DominatorTree *DT);
 
+  /// Returns true if Phi is a linear recurrence of the form
+  ///   h = phi(start, h_next)
+  ///   h_next = add(mul(h, C), x)  (or with the add/mul operands swapped),
+  /// where C is loop-invariant and x is a loop-varying value that does not
+  /// depend on h. All in-loop uses of the recurrence must be part of this
+  /// chain; h itself may only be used outside the loop through h_next.
+  /// If non-null, \p Coeff and \p X are set to C and x respectively.
+  LLVM_ABI static bool isLinearRecurrencePHI(PHINode *Phi, Loop *TheLoop,
+                                             RecurrenceDescriptor &RedDes,
+                                             ScalarEvolution *SE,
+                                             Value **Coeff = nullptr,
+                                             Value **X = nullptr);
+
   RecurKind getRecurrenceKind() const { return Kind; }
 
   unsigned getOpcode() const { return getOpcode(getRecurrenceKind()); }
diff --git a/llvm/lib/Analysis/IVDescriptors.cpp b/llvm/lib/Analysis/IVDescriptors.cpp
index 800b64e9a29af..b37c0096f28a0 100644
--- a/llvm/lib/Analysis/IVDescriptors.cpp
+++ b/llvm/lib/Analysis/IVDescriptors.cpp
@@ -46,6 +46,7 @@ bool RecurrenceDescriptor::isIntegerRecurrenceKind(RecurKind Kind) {
   case RecurKind::Sub:
   case RecurKind::Add:
   case RecurKind::Mul:
+  case RecurKind::IntLinear:
   case RecurKind::Or:
   case RecurKind::And:
   case RecurKind::Xor:
@@ -1225,6 +1226,99 @@ bool RecurrenceDescriptor::isFixedOrderRecurrence(PHINode *Phi, Loop *TheLoop,
   return true;
 }
 
+bool RecurrenceDescriptor::isLinearRecurrencePHI(PHINode *Phi, Loop *TheLoop,
+                                                 RecurrenceDescriptor &RedDes,
+                                                 ScalarEvolution *SE,
+                                                 Value **Coeff, Value **X) {
+  // Basic structural checks: the PHI must be in the loop header, with two
+  // incoming values, and of integer type.
+  if (Phi->getNumIncomingValues() != 2 ||
+      Phi->getParent() != TheLoop->getHeader() ||
+      !Phi->getType()->isIntegerTy())
+    return false;
+  auto *Preheader = TheLoop->getLoopPreheader();
+  auto *Latch = TheLoop->getLoopLatch();
+  if (!Preheader || !Latch)
+    return false;
+
+  Value *Start = Phi->getIncomingValueForBlock(Preheader);
+  auto *Exit = dyn_cast<Instruction>(Phi->getIncomingValueForBlock(Latch));
+  if (!Exit || !TheLoop->contains(Exit))
+    return false;
+
+  // The backedge value must be an add of the form add(mul(Phi, C), X) (the
+  // add and mul operands may be swapped), where C is loop-invariant and X is a
+  // loop-varying value that does not depend on Phi.
+  auto *Add = dyn_cast<BinaryOperator>(Exit);
+  if (!Add || Add->getOpcode() != Instruction::Add)
+    return false;
+
+  BinaryOperator *Mul = dyn_cast<BinaryOperator>(Add->getOperand(0));
+  Value *XVal = Add->getOperand(1);
+  if (!Mul || Mul->getOpcode() != Instruction::Mul) {
+    Mul = dyn_cast<BinaryOperator>(Add->getOperand(1));
+    XVal = Add->getOperand(0);
+  }
+  if (!Mul || Mul->getOpcode() != Instruction::Mul)
+    return false;
+
+  // The mul must use Phi and a loop-invariant coefficient.
+  Value *C = nullptr;
+  if (Mul->getOperand(0) == Phi)
+    C = Mul->getOperand(1);
+  else if (Mul->getOperand(1) == Phi)
+    C = Mul->getOperand(0);
+  else
+    return false;
+  if (!SE->isLoopInvariant(SE->getSCEV(C), TheLoop))
+    return false;
+
+  // X must be an in-loop value that does not depend on Phi.
+  auto *XI = dyn_cast<Instruction>(XVal);
+  if (!XI || !TheLoop->contains(XI))
+    return false;
+  SmallPtrSet<Value *, 8> Visited;
+  SmallVector<Value *, 4> Worklist({XVal});
+  while (!Worklist.empty()) {
+    Value *V = Worklist.pop_back_val();
+    if (V == Phi)
+      return false;
+    if (!Visited.insert(V).second)
+      continue;
+    if (auto *I = dyn_cast<Instruction>(V))
+      for (Value *Op : I->operands())
+        Worklist.push_back(Op);
+  }
+
+  // Phi may only be used in-loop by the mul and must not have outside uses.
+  // The mul may only feed the add, and in-loop uses of the add other than the
+  // PHI backedge are not allowed.
+  for (User *U : Phi->users()) {
+    Instruction *UI = cast<Instruction>(U);
+    if (!TheLoop->contains(UI) || UI != Mul)
+      return false;
+  }
+  if (!Mul->hasOneUse())
+    return false;
+  for (User *U : Add->users()) {
+    Instruction *UI = cast<Instruction>(U);
+    if (UI == Phi)
+      continue;
+    if (TheLoop->contains(UI))
+      return false;
+  }
+
+  RedDes = RecurrenceDescriptor(Start, Exit, /*Store=*/nullptr,
+                                RecurKind::IntLinear, FastMathFlags(),
+                                /*ExactFP=*/nullptr, Phi->getType());
+  if (Coeff)
+    *Coeff = C;
+  if (X)
+    *X = XVal;
+  LLVM_DEBUG(dbgs() << "Found a linear recurrence PHI." << *Phi << "\n");
+  return true;
+}
+
 unsigned RecurrenceDescriptor::getOpcode(RecurKind Kind) {
   switch (Kind) {
   case RecurKind::Sub:
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 358901da42266..9b300f43712c3 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -32518,6 +32518,7 @@ class HorizontalReduction {
         case RecurKind::AnyOf:
         case RecurKind::FindIV:
         case RecurKind::FindLast:
+        case RecurKind::IntLinear:
         case RecurKind::FMaxNum:
         case RecurKind::FMinNum:
         case RecurKind::FMaximumNum:
@@ -32676,6 +32677,7 @@ class HorizontalReduction {
     case RecurKind::AnyOf:
     case RecurKind::FindIV:
     case RecurKind::FindLast:
+    case RecurKind::IntLinear:
     case RecurKind::FMaxNum:
     case RecurKind::FMinNum:
     case RecurKind::FMaximumNum:
@@ -32781,6 +32783,7 @@ class HorizontalReduction {
     case RecurKind::AnyOf:
     case RecurKind::FindIV:
     case RecurKind::FindLast:
+    case RecurKind::IntLinear:
     case RecurKind::FMaxNum:
     case RecurKind::FMinNum:
     case RecurKind::FMaximumNum:
diff --git a/llvm/unittests/Analysis/IVDescriptorsTest.cpp b/llvm/unittests/Analysis/IVDescriptorsTest.cpp
index faf30fd322c10..dd0fc2ae869db 100644
--- a/llvm/unittests/Analysis/IVDescriptorsTest.cpp
+++ b/llvm/unittests/Analysis/IVDescriptorsTest.cpp
@@ -386,6 +386,148 @@ for.end:
       });
 }
 
+// This tests that a linear recurrence h = C * h + x with a loop-invariant
+// coefficient C is recognized.
+TEST(IVDescriptorsTest, LinearRecurrence) {
+  // Parse the module.
+  LLVMContext Context;
+
+  std::unique_ptr<Module> M = parseIR(Context, R"(
+    define i32 @rabin_karp(ptr %s, i64 %n) {
+    entry:
+      br label %for.body
+
+    for.body:
+      %i = phi i64 [ 0, %entry ], [ %i.next, %for.body ]
+      %h = phi i32 [ 0, %entry ], [ %h.next, %for.body ]
+      %arrayidx = getelementptr inbounds i32, ptr %s, i64 %i
+      %ld = load i32, ptr %arrayidx
+      %mul = mul i32 %h, 31
+      %h.next = add i32 %ld, %mul
+      %i.next = add nsw i64 %i, 1
+      %cmp = icmp slt i64 %i.next, %n
+      br i1 %cmp, label %for.body, label %for.end
+
+    for.end:
+      %h.lcssa = phi i32 [ %h.next, %for.body ]
+      ret i32 %h.lcssa
+    })");
+
+  runWithLoopInfoAndSE(
+      *M, "rabin_karp", [&](Function &F, LoopInfo &LI, ScalarEvolution &SE) {
+        Function::iterator FI = F.begin();
+        // First basic block is entry - skip it.
+        BasicBlock *Header = &*(++FI);
+        assert(Header->getName() == "for.body");
+        Loop *L = LI.getLoopFor(Header);
+        EXPECT_NE(L, nullptr);
+        BasicBlock::iterator BBI = Header->begin();
+        assert((&*BBI)->getName() == "i");
+        ++BBI;
+        PHINode *Phi = dyn_cast<PHINode>(&*BBI);
+        assert(Phi->getName() == "h");
+        RecurrenceDescriptor Rdx;
+        bool IsRdxPhi =
+            RecurrenceDescriptor::isLinearRecurrencePHI(Phi, L, Rdx, &SE);
+        EXPECT_TRUE(IsRdxPhi);
+        EXPECT_EQ(Rdx.getRecurrenceKind(), RecurKind::IntLinear);
+        EXPECT_EQ(Rdx.getLoopExitInstr()->getName(), "h.next");
+      });
+}
+
+// Linear recurrences with a loop-varying coefficient, with in-loop uses of the
+// recurrence outside the chain, or with direct outside uses of the recurrence
+// must not be recognized.
+TEST(IVDescriptorsTest, UnsupportedLinearRecurrence) {
+  // Parse the module.
+  LLVMContext Context;
+
+  std::unique_ptr<Module> M = parseIR(Context, R"(
+    define i32 @varying_coeff(ptr %s, ptr %cptr, i64 %n) {
+    entry:
+      br label %for.body
+
+    for.body:
+      %i = phi i64 [ 0, %entry ], [ %i.next, %for.body ]
+      %h = phi i32 [ 0, %entry ], [ %h.next, %for.body ]
+      %c = load i32, ptr %cptr
+      %arrayidx = getelementptr inbounds i32, ptr %s, i64 %i
+      %ld = load i32, ptr %arrayidx
+      %mul = mul i32 %h, %c
+      %h.next = add i32 %ld, %mul
+      %i.next = add nsw i64 %i, 1
+      %cmp = icmp slt i64 %i.next, %n
+      br i1 %cmp, label %for.body, label %for.end
+
+    for.end:
+      %h.lcssa = phi i32 [ %h.next, %for.body ]
+      ret i32 %h.lcssa
+    }
+
+    define i32 @used_in_loop(ptr %s, ptr %dst, i64 %n) {
+    entry:
+      br label %for.body
+
+    for.body:
+      %i = phi i64 [ 0, %entry ], [ %i.next, %for.body ]
+      %h = phi i32 [ 0, %entry ], [ %h.next, %for.body ]
+      %arrayidx = getelementptr inbounds i32, ptr %s, i64 %i
+      %ld = load i32, ptr %arrayidx
+      %mul = mul i32 %h, 31
+      %h.next = add i32 %ld, %mul
+      store i32 %h, ptr %dst
+      %i.next = add nsw i64 %i, 1
+      %cmp = icmp slt i64 %i.next, %n
+      br i1 %cmp, label %for.body, label %for.end
+
+    for.end:
+      %h.lcssa = phi i32 [ %h.next, %for.body ]
+      ret i32 %h.lcssa
+    }
+
+    define i32 @used_outside(ptr %s, i64 %n) {
+    entry:
+      br label %for.body
+
+    for.body:
+      %i = phi i64 [ 0, %entry ], [ %i.next, %for.body ]
+      %h = phi i32 [ 0, %entry ], [ %h.next, %for.body ]
+      %arrayidx = getelementptr inbounds i32, ptr %s, i64 %i
+      %ld = load i32, ptr %arrayidx
+      %mul = mul i32 %h, 31
+      %h.next = add i32 %ld, %mul
+      %i.next = add nsw i64 %i, 1
+      %cmp = icmp slt i64 %i.next, %n
+      br i1 %cmp, label %for.body, label %for.end
+
+    for.end:
+      ret i32 %h
+    })");
+
+  auto Check = [&](StringRef FuncName) {
+    runWithLoopInfoAndSE(
+        *M, FuncName, [&](Function &F, LoopInfo &LI, ScalarEvolution &SE) {
+          Function::iterator FI = F.begin();
+          // First basic block is entry - skip it.
+          BasicBlock *Header = &*(++FI);
+          Loop *L = LI.getLoopFor(Header);
+          EXPECT_NE(L, nullptr);
+          BasicBlock::iterator BBI = Header->begin();
+          assert((&*BBI)->getName() == "i");
+          ++BBI;
+          PHINode *Phi = dyn_cast<PHINode>(&*BBI);
+          EXPECT_NE(Phi, nullptr);
+          RecurrenceDescriptor Rdx;
+          bool IsRdxPhi =
+              RecurrenceDescriptor::isLinearRecurrencePHI(Phi, L, Rdx, &SE);
+          EXPECT_FALSE(IsRdxPhi);
+        });
+  };
+  Check("varying_coeff");
+  Check("used_in_loop");
+  Check("used_outside");
+}
+
 // Make sure isReductionPHI doesn't crash when SE is not passed to it.
 TEST(IVDescriptorsTest, InvariantStoreNoSCEV) {
   // Parse the module.

>From 6b059a6e35f0fddbad50aafe92697ef5530c3c2b Mon Sep 17 00:00:00 2001
From: tarvlad <vladislav.tarasov at huawei.com>
Date: Fri, 28 Aug 2026 15:17:48 +0800
Subject: [PATCH 2/2] [LV] Vectorize integer linear recurrences h = C*h + x
 (chunked form)

Recognize linear recurrences of the form

  h = phi(start, h_next)
  h_next = add(mul(h, C), x)      (operands may be swapped)

where C is a loop-invariant constant and x is a simple load (e.g.
polynomial hashes like Rabin-Karp's h = 31*h + s[i], or Java
String.hashCode), and vectorize them by rewriting the loop into the
equivalent chunked form

  h_next = C^VF * h + sum_l C^(VF-1-l) * x_{i+l}

per vector iteration: a vector multiply by the constant weight vector
[C^(VF-1), ..., C^1, C^0], an in-loop add reduction and a scalar seed
update. For VF = 1 the recipe degenerates to the original computation.
The rewrite is exact for wrapping integer arithmetic, because
C*(a+b) == C*a + C*b modulo 2^N.

Implementation:
- LoopVectorizationLegality recognizes the pattern via
  RecurrenceDescriptor::isLinearRecurrencePHI, behind
  -enable-linear-recurrence-vectorization (off by default). The
  vectorizer-specific restrictions (innermost loop, constant
  coefficient, simple non-volatile/non-atomic load) are enforced here,
  so loops it cannot handle fail legality cleanly.
- A new VPlanTransforms::createLinearRecurrenceRecipes pass replaces
  the mul/add chain with a VPLinearRecurrenceChainRecipe; the
  recurrence value is carried in a scalar register across the vector
  loop by a VPLinearRecurrencePHIRecipe.
- The exit value of the vector loop is the chain recipe's result; the
  scalar epilogue resumes from it.

Currently not supported: tail folding (bailed out up-front), scalable
VFs (rejected via an invalid cost) and interleaving (IC clamped to 1);
these are follow-ups.

Part of the fix for https://github.com/llvm/llvm-project/issues/56999.
---
 .../Vectorize/LoopVectorizationLegality.h     |   9 +
 .../Vectorize/LoopVectorizationLegality.cpp   |  18 ++
 .../Vectorize/LoopVectorizationPlanner.h      |   1 +
 .../Transforms/Vectorize/LoopVectorize.cpp    |  39 ++++-
 llvm/lib/Transforms/Vectorize/VPlan.cpp       |   7 +-
 llvm/lib/Transforms/Vectorize/VPlan.h         | 116 ++++++++++++-
 .../Vectorize/VPlanConstruction.cpp           |  55 +++++-
 .../lib/Transforms/Vectorize/VPlanRecipes.cpp | 119 +++++++++++++
 .../Transforms/Vectorize/VPlanTransforms.h    |  12 +-
 llvm/lib/Transforms/Vectorize/VPlanUtils.cpp  |   2 +
 .../AArch64/linear-recurrence.ll              | 158 ++++++++++++++++++
 .../LoopVectorize/X86/linear-recurrence.ll    | 158 ++++++++++++++++++
 12 files changed, 685 insertions(+), 9 deletions(-)
 create mode 100644 llvm/test/Transforms/LoopVectorize/AArch64/linear-recurrence.ll
 create mode 100644 llvm/test/Transforms/LoopVectorize/X86/linear-recurrence.ll

diff --git a/llvm/include/llvm/Transforms/Vectorize/LoopVectorizationLegality.h b/llvm/include/llvm/Transforms/Vectorize/LoopVectorizationLegality.h
index 7b8b27c6541e1..e32248ace3203 100644
--- a/llvm/include/llvm/Transforms/Vectorize/LoopVectorizationLegality.h
+++ b/llvm/include/llvm/Transforms/Vectorize/LoopVectorizationLegality.h
@@ -333,6 +333,11 @@ class LoopVectorizationLegality {
   /// Return the fixed-order recurrences found in the loop.
   RecurrenceSet &getFixedOrderRecurrences() { return FixedOrderRecurrences; }
 
+  /// Returns the linear recurrences found in the loop.
+  const ReductionList &getLinearRecurrences() const {
+    return LinearRecurrences;
+  }
+
   /// Returns the widest induction type.
   IntegerType *getWidestInductionType() { return WidestIndTy; }
 
@@ -698,6 +703,10 @@ class LoopVectorizationLegality {
   /// Holds the phi nodes that are fixed-order recurrences.
   RecurrenceSet FixedOrderRecurrences;
 
+  /// Holds the phi nodes that are linear recurrences (h = C*h + x) with their
+  /// recurrence descriptors.
+  ReductionList LinearRecurrences;
+
   /// Holds the widest induction type encountered.
   IntegerType *WidestIndTy = nullptr;
 
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp
index f1d785b571367..9e4be6e8c5786 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp
@@ -910,6 +910,24 @@ bool LoopVectorizationLegality::canVectorizeInstr(Instruction &I) {
       return true;
     }
 
+    // Linear recurrences h = C*h + x are legal if vectorization of them is
+    // enabled. To be able to vectorize them, the coefficient C must be a
+    // constant and the per-lane value x must be a simple (non-volatile,
+    // non-atomic) load; these restrictions are enforced here so that the
+    // VPlan transform can rely on them.
+    RecurrenceDescriptor LinRecDes;
+    Value *C = nullptr, *X = nullptr;
+    if (EnableLinearRecurrenceVectorization && TheLoop->isInnermost() &&
+        RecurrenceDescriptor::isLinearRecurrencePHI(Phi, TheLoop, LinRecDes,
+                                                    PSE.getSE(), &C, &X)) {
+      auto *LI = dyn_cast<LoadInst>(X);
+      if (isa<ConstantInt>(C) && LI && LI->isSimple()) {
+        LinearRecurrences[Phi] = std::move(LinRecDes);
+        return true;
+      }
+      // Fall through to the unidentified PHI failure below.
+    }
+
     // As a last resort, coerce the PHI to a AddRec expression
     // and re-try classifying it a an induction PHI.
     if (InductionDescriptor::isInductionPHI(Phi, TheLoop, PSE, ID, true) &&
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
index 9ca869f5ebe88..44154c6749489 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
@@ -53,6 +53,7 @@ struct VFRange;
 extern cl::opt<bool> EnableVPlanNativePath;
 extern cl::opt<unsigned> ForceTargetInstructionCost;
 extern cl::opt<bool> PreferInLoopReductions;
+extern cl::opt<bool> EnableLinearRecurrenceVectorization;
 
 /// \return An upper bound for vscale based on TTI or the vscale_range
 /// attribute.
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 1f153afd6bedd..dc9c3e5c0bad3 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -347,6 +347,12 @@ cl::opt<bool> llvm::EnableVPlanNativePath(
     cl::desc("Enable VPlan-native vectorization path with "
              "support for outer loop vectorization."));
 
+cl::opt<bool> llvm::EnableLinearRecurrenceVectorization(
+    "enable-linear-recurrence-vectorization", cl::init(false), cl::Hidden,
+    cl::desc("Vectorize linear recurrences of the form h = C*h + x, where C "
+             "is a loop-invariant coefficient and x is a loop-varying value "
+             "(e.g. polynomial hashes like h = 31*h + s[i])."));
+
 cl::opt<bool>
     llvm::VerifyEachVPlan("vplan-verify-each",
 #ifdef EXPENSIVE_CHECKS
@@ -3288,6 +3294,8 @@ static bool willGenerateVectors(VPlan &Plan, ElementCount VF,
       case VPRecipeBase::VPWidenIntOrFpInductionSC:
       case VPRecipeBase::VPWidenPointerInductionSC:
       case VPRecipeBase::VPReductionPHISC:
+      case VPRecipeBase::VPLinearRecurrencePHISC:
+      case VPRecipeBase::VPLinearRecurrenceChainSC:
       case VPRecipeBase::VPInterleaveEVLSC:
       case VPRecipeBase::VPInterleaveSC:
       case VPRecipeBase::VPWidenLoadEVLSC:
@@ -5405,6 +5413,14 @@ void LoopVectorizationCostModel::collectValuesToIgnore() {
 }
 
 void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
+  // Linear recurrences are currently only supported without interleaving, as
+  // the scalar recurrence value cannot be threaded through unrolled parts yet.
+  if (UserIC > 1 && !Legal->getLinearRecurrences().empty()) {
+    LLVM_DEBUG(dbgs() << "LV: Ignoring requested interleave count; linear "
+                         "recurrences do not support interleaving yet.\n");
+    UserIC = 0;
+  }
+
   CM.collectValuesToIgnore();
   Config.collectElementTypesForWidening(&CM.ValuesToIgnore);
 
@@ -6494,10 +6510,15 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan1() {
           VPlanTransforms::createHeaderPhiRecipes, *VPlan0, PSE, *OrigLoop,
           VPDT, Legal->getInductionVars(), Legal->getReductionVars(),
           Legal->getFixedOrderRecurrences(), Config.getInLoopReductions(),
-          Config.getHints().allowReordering())) {
+          Legal->getLinearRecurrences(), Config.getHints().allowReordering())) {
     return nullptr;
   }
 
+  // Replace the mul/add chain of each linear recurrence with the chunked
+  // computation; bail out if a recurrence cannot be handled.
+  if (!RUN_VPLAN_PASS(VPlanTransforms::createLinearRecurrenceRecipes, *VPlan0))
+    return nullptr;
+
   if (const LoopAccessInfo *LAI = Legal->getLAI())
     RUN_VPLAN_PASS(VPlanTransforms::replaceSymbolicStrides, *VPlan0, PSE,
                    LAI->getSymbolicStrides(), VPDT);
@@ -6638,6 +6659,15 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(VPlanPtr Plan,
     IVUpdateMayOverflow |= !isIndvarOverflowCheckKnownFalse(&CM, VF);
 
   TailFoldingStyle Style = CM.getTailFoldingStyle();
+  // Linear recurrences cannot be vectorized when folding the tail: the
+  // chunked computation requires full chunks (see
+  // VPLinearRecurrenceChainRecipe).
+  if (Style != TailFoldingStyle::None &&
+      !Legal->getLinearRecurrences().empty()) {
+    LLVM_DEBUG(dbgs() << "LV: Can't vectorize linear recurrences when folding "
+                         "the tail, bailing out.\n");
+    return nullptr;
+  }
   // Use NUW for the induction increment if we proved that it won't overflow in
   // the vector loop or when not folding the tail. In the later case, we know
   // that the canonical induction increment will not overflow as the vector trip
@@ -6714,7 +6744,8 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(VPlanPtr Plan,
       if (isa<VPWidenCanonicalIVRecipe, VPBlendRecipe, VPReductionRecipe,
               VPReplicateRecipe, VPWidenLoadRecipe, VPWidenStoreRecipe,
               VPWidenCallRecipe, VPWidenIntrinsicRecipe, VPVectorPointerRecipe,
-              VPVectorEndPointerRecipe, VPHistogramRecipe>(&R) ||
+              VPVectorEndPointerRecipe, VPHistogramRecipe,
+              VPLinearRecurrenceChainRecipe>(&R) ||
           (isa<VPInstructionWithType>(R) &&
            Instruction::isCast(cast<VPInstructionWithType>(R).getOpcode()) &&
            vputils::onlyFirstLaneUsed(R.getVPSingleValue())))
@@ -8173,6 +8204,10 @@ bool LoopVectorizePass::processLoop(Loop *L) {
 
   // Override IC if user provided an interleave count.
   IC = UserIC > 0 ? UserIC : IC;
+  // Linear recurrences are currently only supported without interleaving, as
+  // the scalar recurrence value cannot be threaded through unrolled parts yet.
+  if (!LVL.getLinearRecurrences().empty())
+    IC = 1;
 
   if (CM.maskPartialAliasing()) {
     LLVM_DEBUG(
diff --git a/llvm/lib/Transforms/Vectorize/VPlan.cpp b/llvm/lib/Transforms/Vectorize/VPlan.cpp
index 87d24ed9d8d4b..0e3dbd6f40fe1 100644
--- a/llvm/lib/Transforms/Vectorize/VPlan.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlan.cpp
@@ -403,9 +403,10 @@ void VPTransformState::fixupHeaderPhis() {
 
     for (VPRecipeBase &R : Header->phis()) {
       auto *PhiR = cast<VPSingleDefRecipe>(&R);
-      bool NeedsScalar =
-          isa<VPPhi>(PhiR) || (isa<VPReductionPHIRecipe>(PhiR) &&
-                               cast<VPReductionPHIRecipe>(PhiR)->isInLoop());
+      bool NeedsScalar = isa<VPPhi>(PhiR) ||
+                         isa<VPLinearRecurrencePHIRecipe>(PhiR) ||
+                         (isa<VPReductionPHIRecipe>(PhiR) &&
+                          cast<VPReductionPHIRecipe>(PhiR)->isInLoop());
 
       Value *Phi = get(PhiR, NeedsScalar);
       Value *Val = get(PhiR->getOperand(1), NeedsScalar);
diff --git a/llvm/lib/Transforms/Vectorize/VPlan.h b/llvm/lib/Transforms/Vectorize/VPlan.h
index 7d2c2fa1bdd23..efd3d924e56d1 100644
--- a/llvm/lib/Transforms/Vectorize/VPlan.h
+++ b/llvm/lib/Transforms/Vectorize/VPlan.h
@@ -448,6 +448,7 @@ class LLVM_ABI_FOR_TEST VPRecipeBase
     VPWidenStoreEVLSC,
     VPWidenStoreSC,
     VPWidenSC,
+    VPLinearRecurrenceChainSC,
     VPBlendSC,
     VPHistogramSC,
     // START: Phi-like recipes. Need to be kept together.
@@ -461,12 +462,13 @@ class LLVM_ABI_FOR_TEST VPRecipeBase
     VPWidenIntOrFpInductionSC,
     VPWidenPointerInductionSC,
     VPReductionPHISC,
+    VPLinearRecurrencePHISC,
     // END: SubclassID for recipes that inherit VPHeaderPHIRecipe
     // END: Phi-like recipes
     VPFirstPHISC = VPWidenPHISC,
     VPFirstHeaderPHISC = VPCurrentIterationPHISC,
-    VPLastHeaderPHISC = VPReductionPHISC,
-    VPLastPHISC = VPReductionPHISC,
+    VPLastHeaderPHISC = VPLinearRecurrencePHISC,
+    VPLastPHISC = VPLinearRecurrencePHISC,
   };
 
   VPRecipeBase(VPRecipeTy SC, ArrayRef<VPValue *> Operands,
@@ -657,6 +659,8 @@ class LLVM_ABI_FOR_TEST VPSingleDefRecipe : public VPRecipeBase,
     case VPRecipeBase::VPWidenIntOrFpInductionSC:
     case VPRecipeBase::VPWidenPointerInductionSC:
     case VPRecipeBase::VPReductionPHISC:
+    case VPRecipeBase::VPLinearRecurrencePHISC:
+    case VPRecipeBase::VPLinearRecurrenceChainSC:
     case VPRecipeBase::VPWidenLoadEVLSC:
     case VPRecipeBase::VPWidenLoadSC:
       return true;
@@ -2958,6 +2962,114 @@ class VPReductionPHIRecipe : public VPHeaderPHIRecipe, public VPIRFlags {
 #endif
 };
 
+/// A recipe modeling the scalar value of a linear recurrence
+///   h = C * h + x
+/// carried in a scalar register across iterations of the vector loop, where C
+/// is a loop-invariant coefficient and x is a loop-varying value.
+class LLVM_ABI_FOR_TEST VPLinearRecurrencePHIRecipe : public VPHeaderPHIRecipe {
+public:
+  VPLinearRecurrencePHIRecipe(PHINode *Phi, VPValue &Start, VPValue &Backedge)
+      : VPHeaderPHIRecipe(VPRecipeBase::VPLinearRecurrencePHISC, Phi, &Start) {
+    addOperand(&Backedge);
+  }
+
+  ~VPLinearRecurrencePHIRecipe() override = default;
+
+  VPLinearRecurrencePHIRecipe *clone() override {
+    return new VPLinearRecurrencePHIRecipe(
+        dyn_cast_or_null<PHINode>(getUnderlyingValue()), *getStartValue(),
+        *getBackedgeValue());
+  }
+
+  VP_CLASSOF_IMPL(VPRecipeBase::VPLinearRecurrencePHISC)
+
+  /// Generate the phi nodes.
+  void execute(VPTransformState &State) override;
+
+  /// Returns true if the recipe only uses the first lane of operand \p Op.
+  bool usesFirstLaneOnly(const VPValue *Op) const override {
+    assert(is_contained(operands(), Op) &&
+           "Op must be an operand of the recipe");
+    return true;
+  }
+
+  /// Return the cost of the recipe.
+  InstructionCost computeCost(ElementCount VF,
+                              VPCostContext &Ctx) const override;
+
+#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
+  /// Print the recipe.
+  void printRecipe(raw_ostream &O, const Twine &Indent,
+                   VPSlotTracker &SlotTracker) const override;
+#endif
+};
+
+/// A recipe computing the next value of a linear recurrence
+///   h = C * h + x
+/// in the vector loop, using the chunked formulation
+///   h_next = C^VF * h + sum_{l=0}^{VF-1} C^(VF-1-l) * x_{i+l}
+/// where \p ChainOp is the current recurrence value h, \p VecOp is the
+/// vectorized x and \p Coeff is the loop-invariant constant coefficient C.
+/// The recipe produces a scalar value.
+class LLVM_ABI_FOR_TEST VPLinearRecurrenceChainRecipe
+    : public VPSingleDefRecipe {
+  /// The loop-invariant constant coefficient of the recurrence.
+  ConstantInt *Coeff;
+
+public:
+  VPLinearRecurrenceChainRecipe(VPValue &ChainOp, VPValue &VecOp,
+                                ConstantInt *Coeff, Instruction *I,
+                                DebugLoc DL = DebugLoc::getUnknown())
+      : VPSingleDefRecipe(VPRecipeBase::VPLinearRecurrenceChainSC,
+                          {&ChainOp, &VecOp}, ChainOp.getScalarType(), I, DL),
+        Coeff(Coeff) {
+    assert(VecOp.getScalarType() == ChainOp.getScalarType() &&
+           "the recurrence value and the per-lane value must have the same "
+           "type");
+  }
+
+  ~VPLinearRecurrenceChainRecipe() override = default;
+
+  VPLinearRecurrenceChainRecipe *clone() override {
+    return new VPLinearRecurrenceChainRecipe(
+        getChainOp(), getVecOp(), Coeff,
+        dyn_cast_or_null<Instruction>(getUnderlyingValue()), getDebugLoc());
+  }
+
+  VP_CLASSOF_IMPL(VPRecipeBase::VPLinearRecurrenceChainSC)
+
+  /// Generate the recipe.
+  void execute(VPTransformState &State) override;
+
+  /// Returns the current recurrence value operand.
+  VPValue &getChainOp() { return *getOperand(0); }
+  VPValue &getChainOp() const { return *getOperand(0); }
+
+  /// Returns the vectorized per-lane value operand.
+  VPValue &getVecOp() { return *getOperand(1); }
+  VPValue &getVecOp() const { return *getOperand(1); }
+
+  /// Returns the loop-invariant constant coefficient.
+  ConstantInt *getCoefficient() const { return Coeff; }
+
+  /// Return the cost of the recipe.
+  InstructionCost computeCost(ElementCount VF,
+                              VPCostContext &Ctx) const override;
+
+  /// Returns true if the recipe only uses the first lane of operand \p Op.
+  bool usesFirstLaneOnly(const VPValue *Op) const override {
+    assert(is_contained(operands(), Op) &&
+           "Op must be an operand of the recipe");
+    return Op == getOperand(0);
+  }
+
+#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
+  /// Print the recipe.
+  void printRecipe(raw_ostream &O, const Twine &Indent,
+                   VPSlotTracker &SlotTracker) const override;
+#endif
+};
+
 /// A recipe for vectorizing a phi-node as a sequence of mask-based select
 /// instructions.
 class LLVM_ABI_FOR_TEST VPBlendRecipe : public VPRecipeWithIRFlags {
diff --git a/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp b/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
index 7f748960b1d8c..adb91266115df 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
@@ -911,7 +911,9 @@ bool VPlanTransforms::createHeaderPhiRecipes(
     const MapVector<PHINode *, InductionDescriptor> &Inductions,
     const MapVector<PHINode *, RecurrenceDescriptor> &Reductions,
     const SmallPtrSetImpl<const PHINode *> &FixedOrderRecurrences,
-    const SmallPtrSetImpl<PHINode *> &InLoopReductions, bool AllowReordering) {
+    const SmallPtrSetImpl<PHINode *> &InLoopReductions,
+    const MapVector<PHINode *, RecurrenceDescriptor> &LinearRecurrences,
+    bool AllowReordering) {
   // Retrieve the header manually from the intial plain-CFG VPlan.
   auto [HeaderVPBB, LatchVPBB] = VPBlockUtils::getPlainCFGHeaderAndLatch(Plan);
   assert(VPDT.dominates(HeaderVPBB, LatchVPBB) &&
@@ -936,6 +938,9 @@ bool VPlanTransforms::createHeaderPhiRecipes(
       return new VPFirstOrderRecurrencePHIRecipe(Phi, *Start, *BackedgeValue);
     }
 
+    if (LinearRecurrences.contains(Phi))
+      return new VPLinearRecurrencePHIRecipe(Phi, *Start, *BackedgeValue);
+
     auto InductionIt = Inductions.find(Phi);
     if (InductionIt != Inductions.end())
       return createWidenInductionRecipe(Phi, PhiR, Start, InductionIt->second,
@@ -1199,6 +1204,54 @@ void VPlanTransforms::createInLoopReductionRecipes(VPlan &Plan,
     R->eraseFromParent();
 }
 
+bool VPlanTransforms::createLinearRecurrenceRecipes(VPlan &Plan) {
+  auto [HeaderVPBB, _] = VPBlockUtils::getPlainCFGHeaderAndLatch(Plan);
+  for (VPRecipeBase &R : make_early_inc_range(HeaderVPBB->phis())) {
+    auto *PhiR = dyn_cast<VPLinearRecurrencePHIRecipe>(&R);
+    if (!PhiR)
+      continue;
+
+    // The phi must only be used by the multiply computing C * h, which in
+    // turn must only be used by the add computing C * h + x.
+    auto *MulVPI = dyn_cast<VPInstruction>(*PhiR->user_begin());
+    if (!PhiR->hasOneUse() || !MulVPI ||
+        MulVPI->getOpcode() != Instruction::Mul)
+      return false;
+    if (!MulVPI->hasOneUse())
+      return false;
+    auto *AddVPI = dyn_cast<VPInstruction>(*MulVPI->user_begin());
+    if (!AddVPI || AddVPI->getOpcode() != Instruction::Add)
+      return false;
+
+    // Determine the constant coefficient C and the per-lane value operand x.
+    auto *Phi = cast<PHINode>(PhiR->getUnderlyingValue());
+    auto *MulI = cast<Instruction>(MulVPI->getUnderlyingValue());
+    auto *C = dyn_cast<ConstantInt>(
+        MulI->getOperand(0) == Phi ? MulI->getOperand(1) : MulI->getOperand(0));
+    if (!C)
+      return false;
+    VPValue *X = AddVPI->getOperand(0) == MulVPI ? AddVPI->getOperand(1)
+                                                 : AddVPI->getOperand(0);
+    // The per-lane value must be a load that will be widened to a vector load
+    // later; this also guarantees it dominates the chain recipe.
+    auto *LoadVPI = dyn_cast_or_null<VPInstruction>(X->getDefiningRecipe());
+    if (!LoadVPI || LoadVPI->getOpcode() != Instruction::Load ||
+        LoadVPI->isMasked())
+      return false;
+
+    // Replace the mul/add chain with the chunked computation
+    //   h_next = C^VF * h + sum_l C^(VF-1-l) * x_{i+l}.
+    auto *AddI = cast<Instruction>(AddVPI->getUnderlyingValue());
+    auto *Chain = new VPLinearRecurrenceChainRecipe(*PhiR, *X, C, AddI,
+                                                    AddI->getDebugLoc());
+    Chain->insertBefore(AddVPI);
+    AddVPI->replaceAllUsesWith(Chain);
+    AddVPI->eraseFromParent();
+    MulVPI->eraseFromParent();
+  }
+  return true;
+}
+
 bool VPlanTransforms::areAllLoadsDereferenceable(VPBasicBlock *HeaderVPBB,
                                                  Loop *TheLoop,
                                                  PredicatedScalarEvolution &PSE,
diff --git a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
index 047bcb3a0abdb..207a939c82dc5 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
@@ -89,6 +89,8 @@ bool VPRecipeBase::mayWriteToMemory() const {
   case VPDerivedIVSC:
   case VPFirstOrderRecurrencePHISC:
   case VPReductionPHISC:
+  case VPLinearRecurrencePHISC:
+  case VPLinearRecurrenceChainSC:
   case VPScalarIVStepsSC:
   case VPPredInstPHISC:
   case VPExpandSCEVSC:
@@ -142,6 +144,8 @@ bool VPRecipeBase::mayReadFromMemory() const {
   case VPCurrentIterationPHISC:
   case VPFirstOrderRecurrencePHISC:
   case VPReductionPHISC:
+  case VPLinearRecurrencePHISC:
+  case VPLinearRecurrenceChainSC:
   case VPPredInstPHISC:
   case VPScalarIVStepsSC:
   case VPWidenStoreEVLSC:
@@ -181,6 +185,8 @@ bool VPRecipeBase::mayHaveSideEffects() const {
   case VPCurrentIterationPHISC:
   case VPFirstOrderRecurrencePHISC:
   case VPReductionPHISC:
+  case VPLinearRecurrencePHISC:
+  case VPLinearRecurrenceChainSC:
   case VPPredInstPHISC:
   case VPVectorEndPointerSC:
   case VPExpandSCEVSC:
@@ -2708,6 +2714,9 @@ static void printRecurrenceKind(raw_ostream &OS, const RecurKind &Kind) {
   case RecurKind::FMulAdd:
     OS << "fmuladd";
     break;
+  case RecurKind::IntLinear:
+    OS << "int-linear";
+    break;
   case RecurKind::AnyOf:
     OS << "any-of";
     break;
@@ -4976,6 +4985,116 @@ VPFirstOrderRecurrencePHIRecipe::computeCost(ElementCount VF,
   return 0;
 }
 
+void VPLinearRecurrencePHIRecipe::execute(VPTransformState &State) {
+  // The linear recurrence value is carried in a scalar register across the
+  // vector loop. Create a scalar PHI in the vector loop header, with the
+  // backedge added later by VPTransformState::fixupHeaderPhis.
+  BasicBlock *VectorPH =
+      State.CFG.VPBB2IRBB.at(getParent()->getCFGPredecessor(0));
+  Value *Start = State.get(getStartValue(), /*IsScalar=*/true);
+
+  BasicBlock *HeaderBB = State.CFG.PrevBB;
+  assert(State.CurrentParentLoop->getHeader() == HeaderBB &&
+         "recipe must be in the vector loop header");
+  PHINode *Phi = PHINode::Create(Start->getType(), 2, "vector.linear.phi");
+  Phi->insertBefore(HeaderBB->getFirstInsertionPt());
+  State.set(this, Phi, /*IsScalar=*/true);
+
+  Phi->addIncoming(Start, VectorPH);
+}
+
+InstructionCost
+VPLinearRecurrencePHIRecipe::computeCost(ElementCount VF,
+                                         VPCostContext &Ctx) const {
+  // The recurrence value is kept in a scalar register.
+  return Ctx.TTI.getCFInstrCost(Instruction::PHI, Ctx.CostKind);
+}
+
+void VPLinearRecurrenceChainRecipe::execute(VPTransformState &State) {
+  Value *X = State.get(&getVecOp());
+  Value *Prev = State.get(&getChainOp(), /*IsScalar=*/true);
+  Type *Ty = getScalarType();
+  ConstantInt *C = getCoefficient();
+
+  Value *Next;
+  if (State.VF.isVector()) {
+    unsigned VF = State.VF.getKnownMinValue();
+    assert(!State.VF.isScalable() &&
+           "scalable VFs are not supported for linear recurrences yet");
+    // Build the weight vector [C^(VF-1), ..., C^1, C^0].
+    SmallVector<Constant *, 8> Weights(VF);
+    APInt Power(C->getValue());
+    Weights[VF - 1] = ConstantInt::get(Ty, 1);
+    for (unsigned L = 1; L != VF; ++L) {
+      Weights[VF - 1 - L] = ConstantInt::get(Ty, Power);
+      Power *= C->getValue();
+    }
+    // Power now holds C^VF for the seed update.
+    auto *CVF = cast<ConstantInt>(ConstantInt::get(Ty, Power));
+
+    // h_next = h * C^VF + sum_l C^(VF-1-l) * x_{i+l}
+    Value *W = State.Builder.CreateMul(X, ConstantVector::get(Weights));
+    Value *R = createSimpleReduction(State.Builder, W, RecurKind::Add);
+    Value *Scaled = State.Builder.CreateMul(Prev, CVF);
+    Next = State.Builder.CreateAdd(Scaled, R);
+  } else {
+    // For scalar VFs the recipe degenerates to the original computation
+    // h_next = h * C + x.
+    Value *Scaled = State.Builder.CreateMul(Prev, C);
+    Next = State.Builder.CreateAdd(Scaled, X);
+  }
+  State.set(this, Next, /*IsScalar=*/true);
+}
+
+InstructionCost
+VPLinearRecurrenceChainRecipe::computeCost(ElementCount VF,
+                                           VPCostContext &Ctx) const {
+  // Scalable VFs are not supported yet; the invalid cost rejects them.
+  if (VF.isScalable())
+    return InstructionCost::getInvalid();
+
+  if (VF.isScalar()) {
+    // Matches the original scalar computation: h_next = h * C + x.
+    return Ctx.TTI.getArithmeticInstrCost(Instruction::Mul, getScalarType(),
+                                          Ctx.CostKind) +
+           Ctx.TTI.getArithmeticInstrCost(Instruction::Add, getScalarType(),
+                                          Ctx.CostKind);
+  }
+
+  // Cost of the chunked computation:
+  //   w = x * weights                    (vector multiply)
+  //   r = sum_l w_l                      (in-loop add reduction)
+  //   h_next = h * C^VF + r              (scalar mul + add)
+  VectorType *VecTy = VectorType::get(getScalarType(), VF);
+  InstructionCost Cost =
+      Ctx.TTI.getArithmeticInstrCost(Instruction::Mul, VecTy, Ctx.CostKind);
+  Cost += Ctx.TTI.getArithmeticReductionCost(Instruction::Add, VecTy,
+                                             std::nullopt, Ctx.CostKind);
+  Cost += Ctx.TTI.getArithmeticInstrCost(Instruction::Mul, getScalarType(),
+                                         Ctx.CostKind);
+  Cost += Ctx.TTI.getArithmeticInstrCost(Instruction::Add, getScalarType(),
+                                         Ctx.CostKind);
+  return Cost;
+}
+
+#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
+void VPLinearRecurrencePHIRecipe::printRecipe(
+    raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const {
+  O << Indent << "LINEAR-RECURRENCE-PHI ";
+  printAsOperand(O, SlotTracker);
+  O << " = phi ";
+  printOperands(O, SlotTracker);
+}
+
+void VPLinearRecurrenceChainRecipe::printRecipe(
+    raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const {
+  O << Indent << "EMIT ";
+  printAsOperand(O, SlotTracker);
+  O << " = LINEAR-RECURRENCE-CHAIN (coeff=" << *getCoefficient() << ")";
+  printOperands(O, SlotTracker);
+}
+#endif
+
 #if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
 void VPFirstOrderRecurrencePHIRecipe::printRecipe(
     raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const {
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.h b/llvm/lib/Transforms/Vectorize/VPlanTransforms.h
index a3922a858a3f5..fe1c689928ac1 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.h
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.h
@@ -171,7 +171,17 @@ struct VPlanTransforms {
       const MapVector<PHINode *, InductionDescriptor> &Inductions,
       const MapVector<PHINode *, RecurrenceDescriptor> &Reductions,
       const SmallPtrSetImpl<const PHINode *> &FixedOrderRecurrences,
-      const SmallPtrSetImpl<PHINode *> &InLoopReductions, bool AllowReordering);
+      const SmallPtrSetImpl<PHINode *> &InLoopReductions,
+      const MapVector<PHINode *, RecurrenceDescriptor> &LinearRecurrences,
+      bool AllowReordering);
+
+  /// Replace the chain of recipes computing the next value of each linear
+  /// recurrence h = C*h + x in \p Plan with a VPLinearRecurrenceChainRecipe,
+  /// which computes the chunked formulation
+  ///   h_next = C^VF * h + sum_l C^(VF-1-l) * x_{i+l}
+  /// in the vector loop. Returns false if a linear recurrence cannot be
+  /// handled.
+  static bool createLinearRecurrenceRecipes(VPlan &Plan);
 
   /// Finalize SCEV predicates by adding induction predicates from \p Plan to
   /// \p PSE and checking constraints. Returns false if predicated IVs have
diff --git a/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp b/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
index b2a91b701fcaa..b97c6a4f4f0de 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
@@ -444,6 +444,8 @@ bool vputils::isSingleScalar(const VPValue *VPV) {
             all_of(VPI->operands(), isSingleScalar));
   if (auto *RR = dyn_cast<VPReductionRecipe>(VPV))
     return !RR->isPartialReduction();
+  if (isa<VPLinearRecurrenceChainRecipe, VPLinearRecurrencePHIRecipe>(VPV))
+    return true;
   if (isa<VPVectorPointerRecipe, VPVectorEndPointerRecipe, VPDerivedIVRecipe>(
           VPV))
     return true;
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/linear-recurrence.ll b/llvm/test/Transforms/LoopVectorize/AArch64/linear-recurrence.ll
new file mode 100644
index 0000000000000..46cebc55f8d8c
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/linear-recurrence.ll
@@ -0,0 +1,158 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 5
+; RUN: opt -passes=loop-vectorize -enable-linear-recurrence-vectorization -force-vector-width=4 -force-vector-interleave=1 -S %s | FileCheck %s
+
+target datalayout = "e-m:e-i64:64-i128:128-n32:64-S128"
+target triple = "aarch64-unknown-linux-gnu"
+
+; https://github.com/llvm/llvm-project/issues/56999
+define i32 @rabin_karp(ptr %s, i64 %n) {
+; CHECK-LABEL: define i32 @rabin_karp(
+; CHECK-SAME: ptr [[S:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 4
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP1:%.*]] = and i64 [[TMP0]], 3
+; CHECK-NEXT:    [[N_VEC_2:%.*]] = sub i64 [[TMP0]], [[TMP1]]
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[H:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[H_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[GEP:%.*]] = getelementptr inbounds i32, ptr [[S]], i64 [[INDEX]]
+; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[GEP]], align 4
+; CHECK-NEXT:    [[MUL:%.*]] = mul <4 x i32> [[WIDE_LOAD]], <i32 29791, i32 961, i32 31, i32 1>
+; CHECK-NEXT:    [[RDX:%.*]] = call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> [[MUL]])
+; CHECK-NEXT:    [[SCALED:%.*]] = mul i32 [[H]], 923521
+; CHECK-NEXT:    [[H_NEXT]] = add i32 [[SCALED]], [[RDX]]
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT:    [[CMP:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC_2]]
+; CHECK-NEXT:    br i1 [[CMP]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC_2]]
+; CHECK-NEXT:    br i1 [[CMP_N]], label %[[FOR_END:.*]], label %[[SCALAR_PH]]
+; CHECK:       [[SCALAR_PH]]:
+; CHECK-NEXT:    [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC_2]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i32 [ [[H_NEXT]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT:    br label %[[FOR_BODY:.*]]
+; CHECK:       [[FOR_BODY]]:
+; CHECK-NEXT:    [[I:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[I_NEXT:%.*]], %[[FOR_BODY]] ]
+; CHECK-NEXT:    [[H1:%.*]] = phi i32 [ [[BC_MERGE_RDX]], %[[SCALAR_PH]] ], [ [[H_NEXT1:%.*]], %[[FOR_BODY]] ]
+; CHECK-NEXT:    [[GEP1:%.*]] = getelementptr inbounds i32, ptr [[S]], i64 [[I]]
+; CHECK-NEXT:    [[LD:%.*]] = load i32, ptr [[GEP1]], align 4
+; CHECK-NEXT:    [[MUL1:%.*]] = mul i32 [[H1]], 31
+; CHECK-NEXT:    [[H_NEXT1]] = add i32 [[LD]], [[MUL1]]
+; CHECK-NEXT:    [[I_NEXT]] = add nsw i64 [[I]], 1
+; CHECK-NEXT:    [[CMP1:%.*]] = icmp slt i64 [[I_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[CMP1]], label %[[FOR_BODY]], label %[[FOR_END]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK:       [[FOR_END]]:
+; CHECK-NEXT:    [[H_LCSSA:%.*]] = phi i32 [ [[H_NEXT1]], %[[FOR_BODY]] ], [ [[H_NEXT]], %[[MIDDLE_BLOCK]] ]
+; CHECK-NEXT:    ret i32 [[H_LCSSA]]
+;
+entry:
+  br label %for.body
+
+for.body:
+  %i = phi i64 [ 0, %entry ], [ %i.next, %for.body ]
+  %h = phi i32 [ 0, %entry ], [ %h.next, %for.body ]
+  %gep = getelementptr inbounds i32, ptr %s, i64 %i
+  %ld = load i32, ptr %gep, align 4
+  %mul = mul i32 %h, 31
+  %h.next = add i32 %ld, %mul
+  %i.next = add nsw i64 %i, 1
+  %cmp = icmp slt i64 %i.next, %n
+  br i1 %cmp, label %for.body, label %for.end
+
+for.end:
+  %h.lcssa = phi i32 [ %h.next, %for.body ]
+  ret i32 %h.lcssa
+}
+
+; The coefficient must be loop-invariant.
+define i32 @varying_coeff(ptr %s, ptr %cptr, i64 %n) {
+; CHECK-LABEL: define i32 @varying_coeff(
+; CHECK-SAME: ptr [[S:%.*]], ptr [[CPTR:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    br label %[[FOR_BODY:.*]]
+; CHECK:       [[FOR_BODY]]:
+; CHECK-NEXT:    [[I:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[I_NEXT:%.*]], %[[FOR_BODY]] ]
+; CHECK-NEXT:    [[H:%.*]] = phi i32 [ 0, %[[ENTRY]] ], [ [[H_NEXT:%.*]], %[[FOR_BODY]] ]
+; CHECK-NEXT:    [[C:%.*]] = load i32, ptr [[CPTR]], align 4
+; CHECK-NEXT:    [[GEP:%.*]] = getelementptr inbounds i32, ptr [[S]], i64 [[I]]
+; CHECK-NEXT:    [[LD:%.*]] = load i32, ptr [[GEP]], align 4
+; CHECK-NEXT:    [[MUL:%.*]] = mul i32 [[H]], [[C]]
+; CHECK-NEXT:    [[H_NEXT]] = add i32 [[LD]], [[MUL]]
+; CHECK-NEXT:    [[I_NEXT]] = add nsw i64 [[I]], 1
+; CHECK-NEXT:    [[CMP:%.*]] = icmp slt i64 [[I_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[CMP]], label %[[FOR_BODY]], label %[[FOR_END:.*]]
+; CHECK:       [[FOR_END]]:
+; CHECK-NEXT:    [[H_LCSSA:%.*]] = phi i32 [ [[H_NEXT]], %[[FOR_BODY]] ]
+; CHECK-NEXT:    ret i32 [[H_LCSSA]]
+;
+entry:
+  br label %for.body
+
+for.body:
+  %i = phi i64 [ 0, %entry ], [ %i.next, %for.body ]
+  %h = phi i32 [ 0, %entry ], [ %h.next, %for.body ]
+  %c = load i32, ptr %cptr, align 4
+  %gep = getelementptr inbounds i32, ptr %s, i64 %i
+  %ld = load i32, ptr %gep, align 4
+  %mul = mul i32 %h, %c
+  %h.next = add i32 %ld, %mul
+  %i.next = add nsw i64 %i, 1
+  %cmp = icmp slt i64 %i.next, %n
+  br i1 %cmp, label %for.body, label %for.end
+
+for.end:
+  %h.lcssa = phi i32 [ %h.next, %for.body ]
+  ret i32 %h.lcssa
+}
+
+; Uses of the recurrence value inside the loop other than the mul/add chain
+; are not supported.
+define i32 @used_in_loop(ptr %s, ptr %dst, i64 %n) {
+; CHECK-LABEL: define i32 @used_in_loop(
+; CHECK-SAME: ptr [[S:%.*]], ptr [[DST:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    br label %[[FOR_BODY:.*]]
+; CHECK:       [[FOR_BODY]]:
+; CHECK-NEXT:    [[I:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[I_NEXT:%.*]], %[[FOR_BODY]] ]
+; CHECK-NEXT:    [[H:%.*]] = phi i32 [ 0, %[[ENTRY]] ], [ [[H_NEXT:%.*]], %[[FOR_BODY]] ]
+; CHECK-NEXT:    [[GEP:%.*]] = getelementptr inbounds i32, ptr [[S]], i64 [[I]]
+; CHECK-NEXT:    [[LD:%.*]] = load i32, ptr [[GEP]], align 4
+; CHECK-NEXT:    [[MUL:%.*]] = mul i32 [[H]], 31
+; CHECK-NEXT:    [[H_NEXT]] = add i32 [[LD]], [[MUL]]
+; CHECK-NEXT:    store i32 [[H]], ptr [[DST]], align 4
+; CHECK-NEXT:    [[I_NEXT]] = add nsw i64 [[I]], 1
+; CHECK-NEXT:    [[CMP:%.*]] = icmp slt i64 [[I_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[CMP]], label %[[FOR_BODY]], label %[[FOR_END:.*]]
+; CHECK:       [[FOR_END]]:
+; CHECK-NEXT:    [[H_LCSSA:%.*]] = phi i32 [ [[H_NEXT]], %[[FOR_BODY]] ]
+; CHECK-NEXT:    ret i32 [[H_LCSSA]]
+;
+entry:
+  br label %for.body
+
+for.body:
+  %i = phi i64 [ 0, %entry ], [ %i.next, %for.body ]
+  %h = phi i32 [ 0, %entry ], [ %h.next, %for.body ]
+  %gep = getelementptr inbounds i32, ptr %s, i64 %i
+  %ld = load i32, ptr %gep, align 4
+  %mul = mul i32 %h, 31
+  %h.next = add i32 %ld, %mul
+  store i32 %h, ptr %dst, align 4
+  %i.next = add nsw i64 %i, 1
+  %cmp = icmp slt i64 %i.next, %n
+  br i1 %cmp, label %for.body, label %for.end
+
+for.end:
+  %h.lcssa = phi i32 [ %h.next, %for.body ]
+  ret i32 %h.lcssa
+}
+;.
+; CHECK: [[LOOP0]] = distinct !{[[LOOP0]], [[META1:![0-9]+]], [[META2:![0-9]+]]}
+; CHECK: [[META1]] = !{!"llvm.loop.isvectorized", i32 1}
+; CHECK: [[META2]] = !{!"llvm.loop.unroll.runtime.disable"}
+; CHECK: [[LOOP3]] = distinct !{[[LOOP3]], [[META2]], [[META1]]}
+;.
diff --git a/llvm/test/Transforms/LoopVectorize/X86/linear-recurrence.ll b/llvm/test/Transforms/LoopVectorize/X86/linear-recurrence.ll
new file mode 100644
index 0000000000000..3129950fba311
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/X86/linear-recurrence.ll
@@ -0,0 +1,158 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 5
+; RUN: opt -passes=loop-vectorize -enable-linear-recurrence-vectorization -force-vector-width=4 -force-vector-interleave=1 -S %s | FileCheck %s
+
+target datalayout = "e-m:e-p:64:64-i64:64-i128:128-n32:64-S128"
+target triple = "x86_64-unknown-linux-gnu"
+
+; https://github.com/llvm/llvm-project/issues/56999
+define i32 @rabin_karp(ptr %s, i64 %n) {
+; CHECK-LABEL: define i32 @rabin_karp(
+; CHECK-SAME: ptr [[S:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 4
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP1:%.*]] = and i64 [[TMP0]], 3
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[H:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[H_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[GEP:%.*]] = getelementptr inbounds i32, ptr [[S]], i64 [[INDEX]]
+; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[GEP]], align 4
+; CHECK-NEXT:    [[MUL:%.*]] = mul <4 x i32> [[WIDE_LOAD]], <i32 29791, i32 961, i32 31, i32 1>
+; CHECK-NEXT:    [[RDX:%.*]] = call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> [[MUL]])
+; CHECK-NEXT:    [[SCALED:%.*]] = mul i32 [[H]], 923521
+; CHECK-NEXT:    [[H_NEXT]] = add i32 [[SCALED]], [[RDX]]
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT:    [[CMP:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], label %[[FOR_END:.*]], label %[[SCALAR_PH]]
+; CHECK:       [[SCALAR_PH]]:
+; CHECK-NEXT:    [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i32 [ [[H_NEXT]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT:    br label %[[FOR_BODY:.*]]
+; CHECK:       [[FOR_BODY]]:
+; CHECK-NEXT:    [[I:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[I_NEXT:%.*]], %[[FOR_BODY]] ]
+; CHECK-NEXT:    [[H1:%.*]] = phi i32 [ [[BC_MERGE_RDX]], %[[SCALAR_PH]] ], [ [[H_NEXT1:%.*]], %[[FOR_BODY]] ]
+; CHECK-NEXT:    [[GEP1:%.*]] = getelementptr inbounds i32, ptr [[S]], i64 [[I]]
+; CHECK-NEXT:    [[LD:%.*]] = load i32, ptr [[GEP1]], align 4
+; CHECK-NEXT:    [[MUL1:%.*]] = mul i32 [[H1]], 31
+; CHECK-NEXT:    [[H_NEXT1]] = add i32 [[LD]], [[MUL1]]
+; CHECK-NEXT:    [[I_NEXT]] = add nsw i64 [[I]], 1
+; CHECK-NEXT:    [[CMP1:%.*]] = icmp slt i64 [[I_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[CMP1]], label %[[FOR_BODY]], label %[[FOR_END]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK:       [[FOR_END]]:
+; CHECK-NEXT:    [[H_LCSSA:%.*]] = phi i32 [ [[H_NEXT1]], %[[FOR_BODY]] ], [ [[H_NEXT]], %[[MIDDLE_BLOCK]] ]
+; CHECK-NEXT:    ret i32 [[H_LCSSA]]
+;
+entry:
+  br label %for.body
+
+for.body:
+  %i = phi i64 [ 0, %entry ], [ %i.next, %for.body ]
+  %h = phi i32 [ 0, %entry ], [ %h.next, %for.body ]
+  %gep = getelementptr inbounds i32, ptr %s, i64 %i
+  %ld = load i32, ptr %gep, align 4
+  %mul = mul i32 %h, 31
+  %h.next = add i32 %ld, %mul
+  %i.next = add nsw i64 %i, 1
+  %cmp = icmp slt i64 %i.next, %n
+  br i1 %cmp, label %for.body, label %for.end
+
+for.end:
+  %h.lcssa = phi i32 [ %h.next, %for.body ]
+  ret i32 %h.lcssa
+}
+
+; The coefficient must be loop-invariant.
+define i32 @varying_coeff(ptr %s, ptr %cptr, i64 %n) {
+; CHECK-LABEL: define i32 @varying_coeff(
+; CHECK-SAME: ptr [[S:%.*]], ptr [[CPTR:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    br label %[[FOR_BODY:.*]]
+; CHECK:       [[FOR_BODY]]:
+; CHECK-NEXT:    [[I:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[I_NEXT:%.*]], %[[FOR_BODY]] ]
+; CHECK-NEXT:    [[H:%.*]] = phi i32 [ 0, %[[ENTRY]] ], [ [[H_NEXT:%.*]], %[[FOR_BODY]] ]
+; CHECK-NEXT:    [[C:%.*]] = load i32, ptr [[CPTR]], align 4
+; CHECK-NEXT:    [[GEP:%.*]] = getelementptr inbounds i32, ptr [[S]], i64 [[I]]
+; CHECK-NEXT:    [[LD:%.*]] = load i32, ptr [[GEP]], align 4
+; CHECK-NEXT:    [[MUL:%.*]] = mul i32 [[H]], [[C]]
+; CHECK-NEXT:    [[H_NEXT]] = add i32 [[LD]], [[MUL]]
+; CHECK-NEXT:    [[I_NEXT]] = add nsw i64 [[I]], 1
+; CHECK-NEXT:    [[CMP:%.*]] = icmp slt i64 [[I_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[CMP]], label %[[FOR_BODY]], label %[[FOR_END:.*]]
+; CHECK:       [[FOR_END]]:
+; CHECK-NEXT:    [[H_LCSSA:%.*]] = phi i32 [ [[H_NEXT]], %[[FOR_BODY]] ]
+; CHECK-NEXT:    ret i32 [[H_LCSSA]]
+;
+entry:
+  br label %for.body
+
+for.body:
+  %i = phi i64 [ 0, %entry ], [ %i.next, %for.body ]
+  %h = phi i32 [ 0, %entry ], [ %h.next, %for.body ]
+  %c = load i32, ptr %cptr, align 4
+  %gep = getelementptr inbounds i32, ptr %s, i64 %i
+  %ld = load i32, ptr %gep, align 4
+  %mul = mul i32 %h, %c
+  %h.next = add i32 %ld, %mul
+  %i.next = add nsw i64 %i, 1
+  %cmp = icmp slt i64 %i.next, %n
+  br i1 %cmp, label %for.body, label %for.end
+
+for.end:
+  %h.lcssa = phi i32 [ %h.next, %for.body ]
+  ret i32 %h.lcssa
+}
+
+; Uses of the recurrence value inside the loop other than the mul/add chain
+; are not supported.
+define i32 @used_in_loop(ptr %s, ptr %dst, i64 %n) {
+; CHECK-LABEL: define i32 @used_in_loop(
+; CHECK-SAME: ptr [[S:%.*]], ptr [[DST:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    br label %[[FOR_BODY:.*]]
+; CHECK:       [[FOR_BODY]]:
+; CHECK-NEXT:    [[I:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[I_NEXT:%.*]], %[[FOR_BODY]] ]
+; CHECK-NEXT:    [[H:%.*]] = phi i32 [ 0, %[[ENTRY]] ], [ [[H_NEXT:%.*]], %[[FOR_BODY]] ]
+; CHECK-NEXT:    [[GEP:%.*]] = getelementptr inbounds i32, ptr [[S]], i64 [[I]]
+; CHECK-NEXT:    [[LD:%.*]] = load i32, ptr [[GEP]], align 4
+; CHECK-NEXT:    [[MUL:%.*]] = mul i32 [[H]], 31
+; CHECK-NEXT:    [[H_NEXT]] = add i32 [[LD]], [[MUL]]
+; CHECK-NEXT:    store i32 [[H]], ptr [[DST]], align 4
+; CHECK-NEXT:    [[I_NEXT]] = add nsw i64 [[I]], 1
+; CHECK-NEXT:    [[CMP:%.*]] = icmp slt i64 [[I_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[CMP]], label %[[FOR_BODY]], label %[[FOR_END:.*]]
+; CHECK:       [[FOR_END]]:
+; CHECK-NEXT:    [[H_LCSSA:%.*]] = phi i32 [ [[H_NEXT]], %[[FOR_BODY]] ]
+; CHECK-NEXT:    ret i32 [[H_LCSSA]]
+;
+entry:
+  br label %for.body
+
+for.body:
+  %i = phi i64 [ 0, %entry ], [ %i.next, %for.body ]
+  %h = phi i32 [ 0, %entry ], [ %h.next, %for.body ]
+  %gep = getelementptr inbounds i32, ptr %s, i64 %i
+  %ld = load i32, ptr %gep, align 4
+  %mul = mul i32 %h, 31
+  %h.next = add i32 %ld, %mul
+  store i32 %h, ptr %dst, align 4
+  %i.next = add nsw i64 %i, 1
+  %cmp = icmp slt i64 %i.next, %n
+  br i1 %cmp, label %for.body, label %for.end
+
+for.end:
+  %h.lcssa = phi i32 [ %h.next, %for.body ]
+  ret i32 %h.lcssa
+}
+;.
+; CHECK: [[LOOP0]] = distinct !{[[LOOP0]], [[META1:![0-9]+]], [[META2:![0-9]+]]}
+; CHECK: [[META1]] = !{!"llvm.loop.isvectorized", i32 1}
+; CHECK: [[META2]] = !{!"llvm.loop.unroll.runtime.disable"}
+; CHECK: [[LOOP3]] = distinct !{[[LOOP3]], [[META2]], [[META1]]}
+;.



More information about the llvm-commits mailing list