[llvm] [LoopInterchange] Extract statically bounded outer epilogues (PR #224196)

via llvm-commits llvm-commits at lists.llvm.org
Wed Sep 16 22:40:33 PDT 2026


llvmorg-github-actions[bot] wrote:


<!--LLVM PR SUMMARY COMMENT-->

@llvm/pr-subscribers-llvm-transforms

Author: Matt (MattPD)

<details>
<summary>Changes</summary>

LoopInterchange requires a tightly nested pair, so an outer loop whose body ends with a small epilogue after the inner loop is rejected even when that epilogue could be moved into its own loop and the remaining nest interchanged. The motivating shape is a column reduction followed by a diagonal update of the same array:

```
for (i = 0; i < N; ++i) {
  for (j = 0; j < N; ++j)
    check += A[j][i];        // column-strided reduction
  A[i][i] = f(A[i][i]);      // per-outer-iteration epilogue
}
```

This patch adds that transformation behind a new option, `-loop-interchange-outer-epilogue-fission`, which is off by default. The pass discovers a single-path outer-loop epilogue, proves from the raw nest-to-epilogue dependences that moving it after the nest is legal, and runs the existing legality and profitability checks on the nest without the epilogue. After those checks succeed, the pass splits the epilogue into a sibling loop and interchanges the remaining nest.

A dependence whose direction DependenceAnalysis cannot determine is accepted only through an exact byte-offset proof, which bounds the outer trip count by the row stride recovered from the address arithmetic. Only plans whose bound is proved at compile time are applied. A plan that needs a runtime bound is declined with a missed-optimization remark and left to a follow-up. The checksum nest in SPEC CPU2000 171.swim is such a case. Moving its epilogue out and interchanging the nest would turn a 10,680-byte inner stride into unit stride, but the bound is a runtime value in the IR generated for the benchmark, so this patch leaves it unchanged.

Assisted-by: Claude Opus 5, GPT-5.6 Sol, GPT-6 Astra, Claude Fable 5.1.


---

Patch is 579.33 KiB, truncated to 20.00 KiB below, full version: https://github.com/llvm/llvm-project/pull/224196.diff


11 Files Affected:

- (modified) llvm/lib/Transforms/Scalar/LoopInterchange.cpp (+2594-214) 
- (added) llvm/lib/Transforms/Scalar/LoopInterchangeUtils.h (+91) 
- (added) llvm/test/Transforms/LoopInterchange/outer-epilogue-fission-bounds.ll (+1917) 
- (added) llvm/test/Transforms/LoopInterchange/outer-epilogue-fission-debug.ll (+123) 
- (added) llvm/test/Transforms/LoopInterchange/outer-epilogue-fission-deep-operand-chain.test (+210) 
- (added) llvm/test/Transforms/LoopInterchange/outer-epilogue-fission-rejections.ll (+2500) 
- (added) llvm/test/Transforms/LoopInterchange/outer-epilogue-fission-routing.ll (+1901) 
- (added) llvm/test/Transforms/LoopInterchange/outer-epilogue-fission-runtime.ll (+435) 
- (added) llvm/test/Transforms/LoopInterchange/outer-epilogue-fission.ll (+1300) 
- (modified) llvm/unittests/Transforms/Scalar/CMakeLists.txt (+1) 
- (added) llvm/unittests/Transforms/Scalar/LoopInterchangeTest.cpp (+175) 


``````````diff
diff --git a/llvm/lib/Transforms/Scalar/LoopInterchange.cpp b/llvm/lib/Transforms/Scalar/LoopInterchange.cpp
index 48a5ad97871b6..f17f066caf4a3 100644
--- a/llvm/lib/Transforms/Scalar/LoopInterchange.cpp
+++ b/llvm/lib/Transforms/Scalar/LoopInterchange.cpp
@@ -13,28 +13,41 @@
 //===----------------------------------------------------------------------===//
 
 #include "llvm/Transforms/Scalar/LoopInterchange.h"
+#include "LoopInterchangeUtils.h"
+#include "llvm/ADT/APInt.h"
+#include "llvm/ADT/DenseMap.h"
+#include "llvm/ADT/MapVector.h"
 #include "llvm/ADT/STLExtras.h"
+#include "llvm/ADT/SmallPtrSet.h"
 #include "llvm/ADT/SmallSet.h"
 #include "llvm/ADT/SmallVector.h"
 #include "llvm/ADT/Statistic.h"
 #include "llvm/ADT/StringMap.h"
 #include "llvm/ADT/StringRef.h"
 #include "llvm/Analysis/DependenceAnalysis.h"
+#include "llvm/Analysis/DomTreeUpdater.h"
 #include "llvm/Analysis/LoopCacheAnalysis.h"
 #include "llvm/Analysis/LoopInfo.h"
 #include "llvm/Analysis/LoopNestAnalysis.h"
 #include "llvm/Analysis/LoopPass.h"
+#include "llvm/Analysis/MemoryBuiltins.h"
 #include "llvm/Analysis/OptimizationRemarkEmitter.h"
 #include "llvm/Analysis/ScalarEvolution.h"
 #include "llvm/Analysis/ScalarEvolutionExpressions.h"
+#include "llvm/Analysis/TargetTransformInfo.h"
+#include "llvm/Analysis/ValueTracking.h"
 #include "llvm/IR/BasicBlock.h"
+#include "llvm/IR/DataLayout.h"
 #include "llvm/IR/DiagnosticInfo.h"
 #include "llvm/IR/Dominators.h"
 #include "llvm/IR/Function.h"
+#include "llvm/IR/GetElementPtrTypeIterator.h"
+#include "llvm/IR/GlobalVariable.h"
 #include "llvm/IR/IRBuilder.h"
 #include "llvm/IR/InstrTypes.h"
 #include "llvm/IR/Instruction.h"
 #include "llvm/IR/Instructions.h"
+#include "llvm/IR/Operator.h"
 #include "llvm/IR/User.h"
 #include "llvm/IR/Value.h"
 #include "llvm/IR/Verifier.h"
@@ -42,12 +55,20 @@
 #include "llvm/Support/CommandLine.h"
 #include "llvm/Support/Debug.h"
 #include "llvm/Support/ErrorHandling.h"
+#include "llvm/Support/MathExtras.h"
 #include "llvm/Support/raw_ostream.h"
 #include "llvm/Transforms/Scalar/LoopPassManager.h"
 #include "llvm/Transforms/Utils/BasicBlockUtils.h"
+#include "llvm/Transforms/Utils/Cloning.h"
 #include "llvm/Transforms/Utils/Local.h"
 #include "llvm/Transforms/Utils/LoopUtils.h"
+#include "llvm/Transforms/Utils/ScalarEvolutionExpander.h"
+#include <algorithm>
 #include <cassert>
+#include <cstdint>
+#include <limits>
+#include <memory>
+#include <optional>
 #include <utility>
 #include <vector>
 
@@ -56,6 +77,9 @@ using namespace llvm;
 #define DEBUG_TYPE "loop-interchange"
 
 STATISTIC(LoopsInterchanged, "Number of loops interchanged");
+STATISTIC(OuterEpiloguesDistributed,
+          "Number of outer-loop epilogues distributed into their own loop "
+          "before interchange");
 
 static cl::opt<int> LoopInterchangeCostThreshold(
     "loop-interchange-threshold", cl::init(0), cl::Hidden,
@@ -89,6 +113,11 @@ enum class RuleTy {
   Ignore
 };
 
+enum class DependenceColumns {
+  SelectedSubnest,
+  AbsoluteAncestors,
+};
+
 } // end anonymous namespace
 
 // Minimum loop depth supported.
@@ -124,6 +153,21 @@ static cl::opt<bool> EnableReduction2Memory(
     "loop-interchange-reduction-to-mem", cl::init(false), cl::Hidden,
     cl::desc("Support for the inner-loop reduction pattern."));
 
+static cl::opt<bool> EnableOuterEpilogueFission(
+    "loop-interchange-outer-epilogue-fission", cl::init(false), cl::Hidden,
+    cl::desc("Prepare a provably independent outer-loop epilogue for "
+             "distribution before interchange"));
+
+static cl::opt<unsigned int> MaxOuterEpilogueFissionCandidates(
+    "loop-interchange-max-outer-epilogue-fission-candidates", cl::init(10),
+    cl::Hidden,
+    cl::desc("Maximum number of eligible outer-epilogue fission candidates "
+             "prepared"));
+
+static cl::opt<bool> PrintPreparedEpiloguePlans(
+    "loop-interchange-print-prepared-plan", cl::init(false), cl::Hidden,
+    cl::desc("Print deterministic outer-epilogue preparation decisions"));
+
 #ifndef NDEBUG
 static bool noDuplicateRulesAndIgnore(ArrayRef<RuleTy> Rules) {
   SmallSet<RuleTy, 4> Set;
@@ -167,10 +211,65 @@ static bool inThisOrder(const Instruction *Src, const Instruction *Dst) {
 }
 #endif
 
-static bool populateDependencyMatrix(CharMatrix &DepMatrix, unsigned Level,
-                                     Loop *L, DependenceInfo *DI,
-                                     ScalarEvolution *SE,
-                                     OptimizationRemarkEmitter *ORE) {
+namespace {
+
+/// A borrowed view of the instructions omitted by virtual extraction.
+/// Ordinary interchange uses the empty view.
+class ExtractedEpilogueView {
+  const SmallPtrSetImpl<Instruction *> *Excluded = nullptr;
+
+public:
+  ExtractedEpilogueView() = default;
+  explicit ExtractedEpilogueView(const SmallPtrSetImpl<Instruction *> *Excluded)
+      : Excluded(Excluded) {}
+
+  bool excludes(const Instruction *I) const {
+    return Excluded && Excluded->contains(const_cast<Instruction *>(I));
+  }
+
+  explicit operator bool() const { return Excluded != nullptr; }
+
+  const SmallPtrSetImpl<Instruction *> *getExcludedInstructions() const {
+    return Excluded;
+  }
+};
+
+/// Follow the ordinary empty-block walk while treating the extracted slice as
+/// absent. Forwarding PHIs remain in the nest.
+static const BasicBlock &
+skipVirtuallyEmptyBlockUntil(const BasicBlock *From, const BasicBlock *End,
+                             ExtractedEpilogueView View) {
+  if (!View)
+    return LoopNest::skipEmptyBlockUntil(From, End);
+
+  assert(From && End && "expected valid path endpoints");
+  if (From == End || !From->getUniqueSuccessor())
+    return *From;
+
+  auto IsVirtuallyEmpty = [View](const BasicBlock *BB) {
+    return all_of(*BB, [View](const Instruction &I) {
+      return I.isTerminator() || View.excludes(&I);
+    });
+  };
+
+  SmallPtrSet<const BasicBlock *, 4> Visited;
+  const BasicBlock *BB = From->getUniqueSuccessor();
+  const BasicBlock *PredBB = From;
+  while (BB && BB != End && IsVirtuallyEmpty(BB) && !Visited.contains(BB)) {
+    Visited.insert(BB);
+    PredBB = BB;
+    BB = BB->getUniqueSuccessor();
+  }
+  return BB == End ? *End : *PredBB;
+}
+
+} // end anonymous namespace
+
+static bool populateDependencyMatrix(
+    CharMatrix &DepMatrix, unsigned Level, Loop *L, DependenceInfo *DI,
+    ScalarEvolution *SE, OptimizationRemarkEmitter *ORE,
+    DependenceColumns Columns = DependenceColumns::SelectedSubnest,
+    ExtractedEpilogueView View = {}) {
   using ValueVector = SmallVector<Value *, 16>;
 
   ValueVector MemInstr;
@@ -180,6 +279,8 @@ static bool populateDependencyMatrix(CharMatrix &DepMatrix, unsigned Level,
   for (BasicBlock *BB : L->blocks()) {
     // Scan the BB and collect legal loads and stores.
     for (Instruction &I : *BB) {
+      if (View.excludes(&I))
+        continue;
       NumInsts++;
       if (auto *Ld = dyn_cast<LoadInst>(&I)) {
         if (!Ld->isSimple())
@@ -204,12 +305,15 @@ static bool populateDependencyMatrix(CharMatrix &DepMatrix, unsigned Level,
   LLVM_DEBUG(dbgs() << "Found " << NumMemInstr
                     << " Loads and Stores to analyze\n");
   if (MaxMemInstrRatio * NumInsts < NumMemInstr * NumMemInstr) {
-    ORE->emit([&]() {
-      return OptimizationRemarkMissed(DEBUG_TYPE, "UnsupportedLoop",
-                                      L->getStartLoc(), L->getHeader())
-             << "Number of loads/stores exceeded, the supported maximum can be "
-                "increased with option -loop-interchange-max-mem-instr-ratio.";
-    });
+    if (ORE)
+      ORE->emit([&]() {
+        return OptimizationRemarkMissed(DEBUG_TYPE, "UnsupportedLoop",
+                                        L->getStartLoc(), L->getHeader())
+               << "Number of loads/stores exceeded, the supported maximum can "
+                  "be "
+                  "increased with option "
+                  "-loop-interchange-max-mem-instr-ratio.";
+      });
     return false;
   }
   ValueVector::iterator I, IE, J, JE;
@@ -261,29 +365,36 @@ static bool populateDependencyMatrix(CharMatrix &DepMatrix, unsigned Level,
 
         // If the Dependence object doesn't have any information, fill the
         // dependency vector with '*'.
+        unsigned AbsoluteDepth = L->getLoopDepth() + Level - 1;
         if (D->isConfused()) {
           assert(Dep.empty() && "Expected empty dependency vector");
-          Dep.assign(L->getLoopDepth() + Level - 1, '*');
+          Dep.assign(AbsoluteDepth, '*');
         }
 
-        while (Dep.size() < L->getLoopDepth() + Level - 1) {
+        // Absolute mode cannot represent levels below the selected inner loop.
+        if (Columns == DependenceColumns::AbsoluteAncestors &&
+            Dep.size() > AbsoluteDepth)
+          return false;
+
+        while (Dep.size() < AbsoluteDepth) {
           Dep.push_back('I');
         }
 
         // Dependence analysis reports levels for the full enclosing loop nest.
         // Keep only the suffix that corresponds to the selected perfect
         // subnest.
-        if (Dep.size() > Level)
+        if (Columns == DependenceColumns::SelectedSubnest && Dep.size() > Level)
           Dep.erase(Dep.begin(), Dep.end() - Level);
 
         // If all the elements of any direction vector have only '*', legality
         // can't be proven. Exit early to save compile time.
         if (all_of(Dep, equal_to('*'))) {
-          ORE->emit([&]() {
-            return OptimizationRemarkMissed(DEBUG_TYPE, "Dependence",
-                                            L->getStartLoc(), L->getHeader())
-                   << "All loops have dependencies in all directions.";
-          });
+          if (ORE)
+            ORE->emit([&]() {
+              return OptimizationRemarkMissed(DEBUG_TYPE, "Dependence",
+                                              L->getStartLoc(), L->getHeader())
+                     << "All loops have dependencies in all directions.";
+            });
           return false;
         }
 
@@ -391,6 +502,16 @@ static bool isLegalToInterChangeLoops(CharMatrix &DepMatrix,
   return true;
 }
 
+/// The absolute ancestor prefix must decide a row forward or leave it entirely
+/// equal/independent before the selected outer dimension.
+static bool hasDecisiveOrEqualAncestorPrefix(const CharMatrix &DepMatrix,
+                                             unsigned OuterLoopId) {
+  for (const std::vector<char> &Row : DepMatrix)
+    if (isLexicographicallyPositive(Row, 0, OuterLoopId) == false)
+      return false;
+  return true;
+}
+
 static void populateWorklist(Loop &L, LoopVector &LoopList) {
   LLVM_DEBUG(dbgs() << "Calling populateWorklist on Func: "
                     << L.getHeader()->getParent()->getName() << " Loop: %"
@@ -414,11 +535,15 @@ static void populateWorklist(Loop &L, LoopVector &LoopList) {
   LoopList.push_back(CurrentLoop);
 }
 
+static bool hasSupportedLoopDepth(ArrayRef<Loop *> LoopList) {
+  unsigned LoopNestDepth = LoopList.size();
+  return LoopNestDepth >= MinLoopNestDepth && LoopNestDepth <= MaxLoopNestDepth;
+}
+
 static bool hasSupportedLoopDepth(ArrayRef<Loop *> LoopList,
                                   OptimizationRemarkEmitter &ORE) {
-  unsigned LoopNestDepth = LoopList.size();
-  if (LoopNestDepth < MinLoopNestDepth || LoopNestDepth > MaxLoopNestDepth) {
-    LLVM_DEBUG(dbgs() << "Unsupported depth of loop nest " << LoopNestDepth
+  if (!hasSupportedLoopDepth(LoopList)) {
+    LLVM_DEBUG(dbgs() << "Unsupported depth of loop nest " << LoopList.size()
                       << ", the supported range is [" << MinLoopNestDepth
                       << ", " << MaxLoopNestDepth << "].\n");
     Loop *OuterLoop = LoopList.front();
@@ -461,8 +586,13 @@ namespace {
 class LoopInterchangeLegality {
 public:
   LoopInterchangeLegality(Loop *Outer, Loop *Inner, ScalarEvolution *SE,
-                          OptimizationRemarkEmitter *ORE, DominatorTree *DT)
-      : OuterLoop(Outer), InnerLoop(Inner), SE(SE), DT(DT), ORE(ORE) {}
+                          OptimizationRemarkEmitter *ORE, DominatorTree *DT,
+                          ExtractedEpilogueView View = {})
+      : OuterLoop(Outer), InnerLoop(Inner), SE(SE), DT(DT), ORE(ORE),
+        HasExtractedEpilogueView(bool(View)) {
+    if (const auto *Excluded = View.getExcludedInstructions())
+      ExcludedEpilogueInstructions.insert(Excluded->begin(), Excluded->end());
+  }
 
   /// Check if the loops can be interchanged.
   bool canInterchangeLoops(unsigned InnerLoopId, unsigned OuterLoopId,
@@ -488,6 +618,17 @@ class LoopInterchangeLegality {
 
   ArrayRef<Instruction *> getHasNoInfInsts() const { return HasNoInfInsts; }
 
+  bool isPreparedFor(
+      Loop *Outer, Loop *Inner,
+      const SmallPtrSetImpl<Instruction *> &ExcludedInstructions) const {
+    return OuterLoop == Outer && InnerLoop == Inner &&
+           HasExtractedEpilogueView &&
+           ExcludedEpilogueInstructions.size() == ExcludedInstructions.size() &&
+           all_of(ExcludedInstructions, [&](Instruction *I) {
+             return ExcludedEpilogueInstructions.contains(I);
+           });
+  }
+
   /// Record reductions in the inner loop. Currently supported reductions:
   /// - initialized from a constant.
   /// - reduction PHI node has only one user.
@@ -513,6 +654,15 @@ class LoopInterchangeLegality {
 private:
   bool tightlyNested(Loop *Outer, Loop *Inner);
   bool containsUnsafeInstructions(BasicBlock *BB, Instruction *Skip);
+  bool isVirtuallyExtracted(const Instruction *I) const {
+    return HasExtractedEpilogueView &&
+           ExcludedEpilogueInstructions.contains(const_cast<Instruction *>(I));
+  }
+  ExtractedEpilogueView getExtractedEpilogueView() {
+    if (!HasExtractedEpilogueView)
+      return {};
+    return ExtractedEpilogueView(&ExcludedEpilogueInstructions);
+  }
 
   /// Traverse all PHI nodes in the header of each loop in the loop nest
   /// starting from \p OuterLoop, and perform the following checks:
@@ -548,6 +698,10 @@ class LoopInterchangeLegality {
   /// Interface to emit optimization remarks.
   OptimizationRemarkEmitter *ORE;
 
+  /// Own the exclusion set so moving a prepared plan cannot invalidate it.
+  SmallPtrSet<Instruction *, 32> ExcludedEpilogueInstructions;
+  bool HasExtractedEpilogueView = false;
+
   /// Set of reduction PHIs taking part of a reduction across the inner and
   /// outer loop.
   SmallPtrSet<PHINode *, 4> OuterInnerReductions;
@@ -594,68 +748,1712 @@ class CacheCostManager {
   const DenseMap<const Loop *, unsigned> &getCostMap();
 };
 
-/// LoopInterchangeProfitability checks if it is profitable to interchange the
-/// loop.
-class LoopInterchangeProfitability {
-public:
-  LoopInterchangeProfitability(Loop *Outer, Loop *Inner, ScalarEvolution *SE,
-                               OptimizationRemarkEmitter *ORE)
-      : OuterLoop(Outer), InnerLoop(Inner), SE(SE), ORE(ORE) {}
+/// LoopInterchangeProfitability checks if it is profitable to interchange the
+/// loop.
+class LoopInterchangeProfitability {
+public:
+  LoopInterchangeProfitability(Loop *Outer, Loop *Inner, ScalarEvolution *SE,
+                               OptimizationRemarkEmitter *ORE)
+      : OuterLoop(Outer), InnerLoop(Inner), SE(SE), ORE(ORE) {}
+
+  /// Check if the loop interchange is profitable.
+  bool isProfitable(const Loop *InnerLoop, const Loop *OuterLoop,
+                    unsigned InnerLoopId, unsigned OuterLoopId,
+                    CharMatrix &DepMatrix, CacheCostManager &CCM);
+
+private:
+  int getInstrOrderCost();
+  std::optional<bool> isProfitablePerLoopCacheAnalysis(
+      const DenseMap<const Loop *, unsigned> &CostMap, CacheCost *CC);
+  std::optional<bool> isProfitablePerInstrOrderCost();
+  std::optional<bool> isProfitableForVectorization(unsigned InnerLoopId,
+                                                   unsigned OuterLoopId,
+                                                   CharMatrix &DepMatrix);
+  Loop *OuterLoop;
+  Loop *InnerLoop;
+
+  /// Scev analysis.
+  ScalarEvolution *SE;
+
+  /// Interface to emit optimization remarks.
+  OptimizationRemarkEmitter *ORE;
+};
+
+/// LoopInterchangeTransform interchanges the loop.
+class LoopInterchangeTransform {
+public:
+  LoopInterchangeTransform(Loop *Outer, Loop *Inner, ScalarEvolution *SE,
+                           LoopInfo *LI, DominatorTree *DT,
+                           const LoopInterchangeLegality &LIL)
+      : OuterLoop(Outer), InnerLoop(Inner), SE(SE), LI(LI), DT(DT), LIL(LIL) {}
+
+  /// Interchange OuterLoop and InnerLoop.
+  void transform(ArrayRef<Instruction *> DropNoWrapInsts,
+                 ArrayRef<Instruction *> DropNoInfInsts);
+  void reduction2Memory();
+  void restructureLoops(Loop *NewInner, Loop *NewOuter,
+                        BasicBlock *OrigInnerPreHeader,
+                        BasicBlock *OrigOuterPreHeader);
+  void removeChildLoop(Loop *OuterLoop, Loop *InnerLoop);
+
+private:
+  void adjustLoopBranches();
+
+  Loop *OuterLoop;
+  Loop *InnerLoop;
+
+  /// Scev analysis.
+  ScalarEvolution *SE;
+
+  LoopInfo *LI;
+  DominatorTree *DT;
+
+  const LoopInterchangeLegality &LIL;
+};
+
+struct PreparedOuterIVControl {
+  PHINode *Induction = nullptr;
+  Value *InitialValue = nullptr;
+  Instruction *NextValue = nullptr;
+  ICmpInst *LatchCompare = nullptr;
+  CondBrInst *LatchBranch = nullptr;
+  SmallVector<Instruction *, 4> LatchInstructions;
+};
+
+/// The closed post-inner slice is recorded in path/program order. Forwarding
+/// PHIs remain with the nest.
+struct DistributableOuterEpilogue {
+  Loop *Outer = nullptr;
+  Loop *Inner = nullptr;
+  SmallVector<BasicBlock *, 4> Path;
+  SmallVector<Instruction *, 16> Instructions;
+  SmallPtrSet<Instruction *, 32> InstructionSet;
+  SmallVector<Instruction *, 8> MemoryInstructions;
+  SmallVector<Instruction *, 8> OuterIVDerivedInstructions;
+  PreparedOuterIVControl OuterControl;
+};
+
+/// Dependence objects are query temporaries. Retain only the proof summary
+/// and the ID of any bound requirements that discharge it.
+struct PreparedCrossPartitionDependence {
+  bool UsedByteOffsetProof = false;
+  unsigned RequirementId = 0;
+};
+
+enum class PreparedBoundOutcome { StaticBound, RuntimeBound };
+enum class BoundRequirementKind { ModularOuterSpan, ObjectContainment };
+
+/// One requirement: trip(DomainLoop) u<= Limit.
+struct PreparedBoundRequirement {
+  Loop *DomainLoop = nullptr;
+  APInt Limit;
+  BoundRequirementKind Kind = BoundRequirementKind::ModularOuterSpan;
+  unsigned RequirementId = 0;
+  const SCEV *ExactTrip = nullptr;
+  bool Runtime = false;
+};
+
+struct PreparedTripBound {
+  PreparedBoundOutcome Outcome = PreparedBoundOutcome::StaticBound;
+  SmallVector<PreparedBoundRequirement, 2> Requirements;
+  const SCEV *RuntimeExactTrip = nullptr;
+  APInt Wmin;
+};
+
+/// All state needed to validate one collected-chain tail without mutating IR.
+struct PreparedInterchangePlan {
+  DistributableOuterEpilogue Epilogue;
+  SmallVector<Loop *, 8> RoutingChain;
+  SmallVector<Loop *, 8> AbsoluteAncestors;
+  unsigned AbsoluteOuterLoopId = 0;
+  unsigned AbsoluteInnerLoopId = 0;
+  unsigned RoutingOuterLoopId = 0;
+  unsigned RoutingInnerLoopId = 0;
+  SmallVector<Instruction *, 16> NestMemoryInstructions;
+  SmallVector<PreparedCrossPartitionDependence, 8> CrossDependences;
+  PreparedTripBound TripBound;
+  CharMatrix FissionContextMatrix;
+  CharMatrix RoutingMatrix;
+  Loop *FissionContextScanRoot = nullptr;
+  Loop *RoutingScanRoot = nullptr;
+  unsigned FissionContextLevel = 0;
+  unsigned RoutingLevel = 0;
+  DependenceColumns FissionContextColumns =
+      DependenceColumns::AbsoluteAncestors;
+  DependenceColumns RoutingColumns = DependenceColumns::SelectedSubnest;
+  SmallVector<Instruction *, 4> DropNoWrap;
+  SmallVector<Instruction *, 4> DropNoInf;
+  std::unique_ptr<LoopInterchangeLegality> Legality;
+  bool FissionLegal = false;
+  bool FissionContextMatrixComplete = false;
+  bool FissionContextLegal = false;
+  bool RoutingMatrixComplete = false;
+  bool InterchangeLegal = false;
+  bool Profitable = false;
+};
+
+static void debugPreparationReject(Loop *Outer, Loop *Inner, StringRef Reason) {
+  LLVM_DEBUG(dbgs() << "loop-interchange: outer-epilogue preparation rejected "
+                       "candidate in function '"
+                    << Outer->getHeader()->getParent()->getName()
+                    ...
[truncated]

``````````

</details>


https://github.com/llvm/llvm-project/pull/224196


More information about the llvm-commits mailing list