[llvm] 821ff7c - [LoopInterchange] Supported partially-perfect Loop Nests (#199511)

via llvm-commits llvm-commits at lists.llvm.org
Thu Sep 10 19:12:31 PDT 2026


Author: Rohit Garg
Date: 2026-09-11T02:12:25Z
New Revision: 821ff7cd6803d719515c2ae9ada665c4bc3c4c35

URL: https://github.com/llvm/llvm-project/commit/821ff7cd6803d719515c2ae9ada665c4bc3c4c35
DIFF: https://github.com/llvm/llvm-project/commit/821ff7cd6803d719515c2ae9ada665c4bc3c4c35.diff

LOG: [LoopInterchange] Supported partially-perfect Loop Nests (#199511)

In the current implementation when the outermost loop contains sibling
sub-loops (i.e. multiple independent loop nests at the same depth), the
pass silently skips the entire loop nest without attempting to
interchange any of the valid sub-nests (See Below Example).

Example:
```
for(int i=0; i<n; i++){  // Loop 1
    for(int j=0; j<m; j++){  // Loop 2
         for(int r=0; r<m; r++){  // Loop 3
         // Do something
         }
    }
    for(int k=0; k<p; k++){  // Loop 4
        for(int l=0; l<p; l++){  // Loop 5
            //Access A[l][k]
        }
    }
}
```

In the Example , GCC 16 performs loop interchange on (Loop2, Loop3) or
(Loop4, Loop5), provided the interchange is both legal and profitable.

Fixes: https://github.com/llvm/llvm-project/issues/196006

Added: 
    llvm/test/Transforms/LoopInterchange/confused-dependence-subnest.ll
    llvm/test/Transforms/LoopInterchange/dependency-matrix-padding.ll
    llvm/test/Transforms/LoopInterchange/missed-outer-prefix-subnest.ll
    llvm/test/Transforms/LoopInterchange/no-partially-perfect-subnest.ll
    llvm/test/Transforms/LoopInterchange/preserve-sibling-order.ll

Modified: 
    llvm/lib/Transforms/Scalar/LoopInterchange.cpp
    llvm/test/Transforms/LoopInterchange/bail-out-one-loop.ll
    llvm/test/Transforms/LoopInterchange/large-nested-6d.ll
    llvm/test/Transforms/LoopInterchange/partially-perfect-loop.ll

Removed: 
    


################################################################################
diff  --git a/llvm/lib/Transforms/Scalar/LoopInterchange.cpp b/llvm/lib/Transforms/Scalar/LoopInterchange.cpp
index b28d2af5cb7ff..48a5ad97871b6 100644
--- a/llvm/lib/Transforms/Scalar/LoopInterchange.cpp
+++ b/llvm/lib/Transforms/Scalar/LoopInterchange.cpp
@@ -263,13 +263,19 @@ static bool populateDependencyMatrix(CharMatrix &DepMatrix, unsigned Level,
         // dependency vector with '*'.
         if (D->isConfused()) {
           assert(Dep.empty() && "Expected empty dependency vector");
-          Dep.assign(Level, '*');
+          Dep.assign(L->getLoopDepth() + Level - 1, '*');
         }
 
-        while (Dep.size() != Level) {
+        while (Dep.size() < L->getLoopDepth() + Level - 1) {
           Dep.push_back('I');
         }
 
+        // Dependence analysis reports levels for the full enclosing loop nest.
+        // Keep only the suffix that corresponds to the selected perfect
+        // subnest.
+        if (Dep.size() > Level)
+          Dep.erase(Dep.begin(), Dep.end() - Level);
+
         // If all the elements of any direction vector have only '*', legality
         // can't be proven. Exit early to save compile time.
         if (all_of(Dep, equal_to('*'))) {
@@ -674,12 +680,64 @@ struct LoopInterchange {
     return processLoopList(LoopList);
   }
 
+  /// Consider below kernel:
+  /// for(int i=0; i<n; i++){  // Loop 1
+  ///     for(int j=0; j<m; j++){  // Loop 2
+  ///         for(int r=0; r<m; r++){  // Loop 3
+  ///         // Do something
+  ///         }
+  ///     }
+  ///     for(int k=0; k<p; k++){  // Loop 4
+  ///         for(int l=0; l<p; l++){  // Loop 5
+  ///             // Do something
+  ///         }
+  ///     }
+  /// }
+  /// Then collectPerfectNests() will return:
+  /// - [Loop2, Loop3]
+  /// - [Loop4, Loop5]
+  static SmallVector<SmallVector<Loop *, 8>, 4>
+  collectPerfectNests(LoopNest &LN) {
+    SmallVector<SmallVector<Loop *, 8>, 4> LoopLists;
+    for (Loop *L : LN.getLoops()) {
+      if (!L->isInnermost())
+        continue;
+
+      SmallVector<Loop *, 8> LoopList;
+      Loop *Current = L;
+      while (true) {
+        LoopList.push_back(Current);
+        Loop *Parent = Current->getParentLoop();
+        if (!Parent || Parent->getSubLoops().size() != 1)
+          break;
+        Current = Parent;
+      }
+      std::reverse(LoopList.begin(), LoopList.end());
+      if (LoopList.size() >= 2)
+        LoopLists.push_back(std::move(LoopList));
+    }
+    return LoopLists;
+  }
+
   bool run(LoopNest &LN) {
-    SmallVector<Loop *, 8> LoopList(LN.getLoops());
-    for (unsigned I = 1; I < LoopList.size(); ++I)
-      if (LoopList[I]->getParentLoop() != LoopList[I - 1])
-        return false;
-    return processLoopList(LoopList);
+    SmallVector<SmallVector<Loop *, 8>, 4> LoopLists = collectPerfectNests(LN);
+    if (LoopLists.empty()) {
+      LLVM_DEBUG(dbgs() << "No Valid candidates for loop interchange.\n");
+      return false;
+    }
+    bool Changed = false;
+    for (SmallVector<Loop *, 8> &LoopList : LoopLists) {
+      // Ensure minimum depth of the loop nest to do the interchange.
+      if (!hasSupportedLoopDepth(LoopList, *ORE))
+        continue;
+      // Ensure computable loop nest.
+      if (!isComputableLoopNest(&AR->SE, LoopList)) {
+        LLVM_DEBUG(dbgs() << "Not valid loop candidate for interchange\n");
+        continue;
+      }
+      Changed |= processLoopList(LoopList);
+    }
+    return Changed;
   }
 
   unsigned selectLoopForInterchange(ArrayRef<Loop *> LoopList) {
@@ -2072,15 +2130,10 @@ void LoopInterchangeTransform::restructureLoops(
   LI->changeLoopFor(OrigInnerPreHeader, OuterLoopParent);
 
   // Switch the loop levels.
-  if (OuterLoopParent) {
-    // Remove the loop from its parent loop.
-    removeChildLoop(OuterLoopParent, NewInner);
-    removeChildLoop(NewInner, NewOuter);
-    OuterLoopParent->addChildLoop(NewOuter);
-  } else {
-    removeChildLoop(NewInner, NewOuter);
-    LI->replaceLoop(NewInner, NewOuter);
-  }
+  removeChildLoop(NewInner, NewOuter);
+  // Replace NewInner with NewOuter in place, preserving sibling order.
+  LI->replaceLoop(NewInner, NewOuter);
+
   while (!NewOuter->isInnermost())
     NewInner->addChildLoop(NewOuter->removeChildLoop(NewOuter->begin()));
   NewOuter->addChildLoop(NewInner);
@@ -2672,19 +2725,9 @@ PreservedAnalyses LoopInterchangePass::run(LoopNest &LN,
                                            LoopStandardAnalysisResults &AR,
                                            LPMUpdater &U) {
   Function &F = *LN.getParent();
-  SmallVector<Loop *, 8> LoopList(LN.getLoops());
 
   OptimizationRemarkEmitter ORE(&F);
 
-  // Ensure minimum depth of the loop nest to do the interchange.
-  if (!hasSupportedLoopDepth(LoopList, ORE))
-    return PreservedAnalyses::all();
-  // Ensure computable loop nest.
-  if (!isComputableLoopNest(&AR.SE, LoopList)) {
-    LLVM_DEBUG(dbgs() << "Not valid loop candidate for interchange\n");
-    return PreservedAnalyses::all();
-  }
-
   ORE.emit([&]() {
     return OptimizationRemarkAnalysis(DEBUG_TYPE, "Dependence",
                                       LN.getOutermostLoop().getStartLoc(),

diff  --git a/llvm/test/Transforms/LoopInterchange/bail-out-one-loop.ll b/llvm/test/Transforms/LoopInterchange/bail-out-one-loop.ll
index d1cf33acd2831..d2d5889d2f964 100644
--- a/llvm/test/Transforms/LoopInterchange/bail-out-one-loop.ll
+++ b/llvm/test/Transforms/LoopInterchange/bail-out-one-loop.ll
@@ -15,7 +15,7 @@ target datalayout = "e-m:e-p270:32:32-p271:32:32-p272:64:64-i8:8:32-i16:16:32-i6
 ; CHECK-NOT: Delinearizing
 ; CHECK-NOT: Strides:
 ; CHECK-NOT: Terms:
-; CHECK: Unsupported depth of loop nest 1, the supported range is [2, 10].
+; CHECK: No Valid candidates for loop interchange.
 
 define void @foo() {
 entry:

diff  --git a/llvm/test/Transforms/LoopInterchange/confused-dependence-subnest.ll b/llvm/test/Transforms/LoopInterchange/confused-dependence-subnest.ll
new file mode 100644
index 0000000000000..57a9ce5a291fb
--- /dev/null
+++ b/llvm/test/Transforms/LoopInterchange/confused-dependence-subnest.ll
@@ -0,0 +1,119 @@
+; RUN: opt < %s -passes=loop-interchange -loop-interchange-profitabilities=ignore \
+; RUN:     -pass-remarks-missed=loop-interchange -pass-remarks=loop-interchange \
+; RUN:     -disable-output 2>&1 | FileCheck %s
+
+; Reproducer for the confused-dependence path through partially-perfect
+; subnests.
+; Here, confused [j,k] subnest at depth 3.
+; If the confused-dependence logic doesn't use the absolute depth,
+; the pass wrongly interchanges the potentially aliasing [j,k] pair.
+;
+;   void f(double *A, double *C) {
+;     for (int h = 0; h < 8; h++)
+;       for (int i = 0; i < 8; i++) {
+;         for (int j = 0; j < 8; j++)
+;           for (int k = 0; k < 8; k++)
+;             A[j*8 + k] = C[k] + 1.0;      // may alias -> confused, [* *]
+;         for (int x = 0; x < 8; x++)
+;           for (int y = 0; y < 8; y++)
+;             A[(long)y*8 + x] += 2.0;
+;       }
+;   }
+;
+; After fixing the confused-dependence width, the [j, k] subnest bails out
+; during dependency-matrix construction (its direction vector is all '*'),
+; so it cannot be interchanged. The disjoint [x, y] sibling subnest is still
+; interchanged.
+;
+; CHECK: remark: {{.*}}All loops have dependencies in all directions.
+; CHECK: remark: {{.*}}Loop interchanged with enclosing loop.
+
+
+define void @confused_subnest_depth3(ptr %A, ptr %C){
+entry:
+  br label %loop.h.header
+
+
+loop.h.header:
+  %h = phi i64 [ 0, %entry ], [ %h.next, %loop.h.latch ]
+  br label %loop.i2.header
+
+
+loop.i2.header:
+  %i = phi i64 [ 0, %loop.h.header ], [ %i.next, %loop.i2.latch ]
+  br label %loop.j2.header
+
+
+loop.j2.header:
+  %j = phi i64 [ 0, %loop.i2.header ], [ %j.next, %loop.j2.latch ]
+  br label %loop.k2.header
+
+
+loop.k2.header:
+  %k = phi i64 [ 0, %loop.j2.header ], [ %k.next, %loop.k2.latch ]
+  %c.ptr = getelementptr double, ptr %C, i64 %k
+  %c.val = load double, ptr %c.ptr, align 8
+  %sum = fadd double %c.val, 1.000000e+00
+  %jrow = mul nuw nsw i64 %j, 8
+  %aidx = add nuw nsw i64 %jrow, %k
+  %a.ptr = getelementptr double, ptr %A, i64 %aidx
+  store double %sum, ptr %a.ptr, align 8
+  br label %loop.k2.latch
+
+
+loop.k2.latch:
+  %k.next = add nuw nsw i64 %k, 1
+  %k.done = icmp eq i64 %k.next, 8
+  br i1 %k.done, label %loop.j2.latch, label %loop.k2.header
+
+
+loop.j2.latch:
+  %j.next = add nuw nsw i64 %j, 1
+  %j.done = icmp eq i64 %j.next, 8
+  br i1 %j.done, label %loop.x2.header, label %loop.j2.header
+
+
+loop.x2.header:
+  %x = phi i64 [ 0, %loop.j2.latch ], [ %x.next, %loop.x2.latch ]
+  br label %loop.y2.header
+
+
+loop.y2.header:
+  %y = phi i64 [ 0, %loop.x2.header ], [ %y.next, %loop.y2.latch ]
+  %row = mul nuw nsw i64 %y, 8
+  %idx = add nuw nsw i64 %row, %x
+  %axy.ptr = getelementptr double, ptr %A, i64 %idx
+  %old = load double, ptr %axy.ptr, align 8
+  %new = fadd double %old, 2.000000e+00
+  store double %new, ptr %axy.ptr, align 8
+  br label %loop.y2.latch
+
+
+loop.y2.latch:
+  %y.next = add nuw nsw i64 %y, 1
+  %y.done = icmp eq i64 %y.next, 8
+  br i1 %y.done, label %loop.x2.latch, label %loop.y2.header
+
+
+loop.x2.latch:
+  %x.next = add nuw nsw i64 %x, 1
+  %x.done = icmp eq i64 %x.next, 8
+  br i1 %x.done, label %loop.i2.latch, label %loop.x2.header
+
+
+loop.i2.latch:
+  %i.next = add nuw nsw i64 %i, 1
+  %i.done = icmp eq i64 %i.next, 8
+  br i1 %i.done, label %loop.h.latch, label %loop.i2.header
+
+
+loop.h.latch:
+  %h.next = add nuw nsw i64 %h, 1
+  %h.done = icmp eq i64 %h.next, 8
+  br i1 %h.done, label %exit, label %loop.h.header
+
+
+exit:
+  ret void
+}
+

diff  --git a/llvm/test/Transforms/LoopInterchange/dependency-matrix-padding.ll b/llvm/test/Transforms/LoopInterchange/dependency-matrix-padding.ll
new file mode 100644
index 0000000000000..9f790be2c4a65
--- /dev/null
+++ b/llvm/test/Transforms/LoopInterchange/dependency-matrix-padding.ll
@@ -0,0 +1,111 @@
+; RUN: opt < %s -passes=loop-interchange -loop-interchange-profitabilities=ignore -debug-only=loop-interchange,da -disable-output 2>&1 | FileCheck %s
+; REQUIRES: asserts
+;
+; This test focuses exclusively on validating the padding logic in the dependency
+; matrix construction and ensuring that matrix slicing preserves proper alignment 
+; with the corresponding loops.
+;
+; Corresponding C code:
+;
+;   for (int i = 0; i < 32; ++i) {
+;     for (int j = 0; j < 32; ++j) {
+;       int sum = 0;
+;       for (int k = 0; k < 32; ++k)
+;         sum += k;
+;       S[j] = sum + S[j-1]; 
+;     }
+;     for (int x = 0; x < 32; ++x)
+;       D[x] = 0;
+;   }
+;
+; For the first dependency, DA reports the subscript only lives at loop depth 2 (loop.j); confirming
+; that loop.i contributes '=' and loop.j contributes the non-trivial '>'.
+;
+; For the second dependency, Distance 0 at loop.j level gives '=' for both levels (loop.i, loop.j).
+; 
+; For both dependencies, Dep.size() = 2  after DA fill and before padding
+;   L->getLoopDepth() = 2  (loop.j is the outermost loop of the subnest)
+;   Level             = 2  (subnest [loop.j, loop.k] has 2 loops)
+;   L->getLoopDepth() + Level - 1 = 2 + 2 - 1 = 3
+;   2 < 3  =>  padding fires once, appending 'I'
+;
+;   Dep 1 after padding:  ['=', '>', 'I']  (size 3)
+;   Dep 2 after padding:  ['=', '=', 'I']  (size 3)
+;
+;   The first column of this matrix should be dropped. And the Final Dependecy Matrix before Interchange should be:
+;   Dep 1: ['>', 'I'] (Size 2)
+;   Dep 2: ['=', 'I'] (Size 3)
+;
+; CHECK: Found 2 Loads and Stores to analyze
+; CHECK: common nesting levels = 2
+; CHECK: loops = {2}
+; CHECK: Result = anti [S -1|<]!
+; CHECK: common nesting levels = 2
+; CHECK: loops = {2}
+; CHECK: Result = output [S 0]!
+; CHECK: Dependency matrix before interchange:
+; CHECK-NEXT: > I
+; CHECK-NEXT: = I
+; CHECK: Failed interchange InnerLoopId = 1 and OuterLoopId = 0 due to dependence
+;
+define void @test_padding_nontrivial_direction(ptr noalias %S, ptr noalias %D) {
+entry:
+  br label %loop.i.header
+
+loop.i.header:
+  %i = phi i64 [ 0, %entry ], [ %i.next, %loop.i.latch ]
+  br label %loop.j.header
+
+loop.j.header:
+  %j     = phi i64 [ 0, %loop.i.header ], [ %j.next, %loop.j.latch ]
+  %s.ptr = getelementptr i32, ptr %S, i64 %j
+  br label %loop.k.header
+
+loop.k.header:
+  %k        = phi i64 [ 0, %loop.j.header ], [ %k.next, %loop.k.latch ]
+  %sum      = phi i32 [ 0, %loop.j.header ], [ %sum.next, %loop.k.latch ]
+  %k.trunc  = trunc i64 %k to i32
+  %sum.next = add nsw i32 %sum, %k.trunc
+  br label %loop.k.latch
+
+loop.k.latch:
+  %k.next = add nuw nsw i64 %k, 1
+  %k.done = icmp eq i64 %k.next, 32
+  br i1 %k.done, label %loop.j.latch, label %loop.k.header
+
+loop.j.latch:
+  %sum.lcssa = phi i32 [ %sum.next, %loop.k.latch ]
+  ; Load S[j-1] — reads what the previous j iteration stored into S[j-1].
+  ; This load appears before the store in the same BB, so MemInstr order is
+  ; [load, store].  DA tests (load, store) and finds an anti-dependence with
+  ; distance -1 at the loop.j level; normalize() flips it to direction '>'.
+  ; After padding: ['>', 'I'].  The '>' makes interchange illegal.
+  %jm1    = add i64 %j, -1
+  %s.prev = getelementptr i32, ptr %S, i64 %jm1
+  %s.load = load i32, ptr %s.prev, align 4
+  ; Store S[j] = sum + S[j-1].
+  %val    = add nsw i32 %sum.lcssa, %s.load
+  store i32 %val, ptr %s.ptr, align 4
+  %j.next = add nuw nsw i64 %j, 1
+  %j.done = icmp eq i64 %j.next, 32
+  br i1 %j.done, label %loop.x.header, label %loop.j.header
+
+loop.x.header:
+  %x = phi i64 [ 0, %loop.j.latch ], [ %x.next, %loop.x.latch ]
+  %d.ptr = getelementptr i8, ptr %D, i64 %x
+  store i8 0, ptr %d.ptr, align 1
+  br label %loop.x.latch
+
+loop.x.latch:
+  %x.next = add nuw nsw i64 %x, 1
+  %x.done = icmp eq i64 %x.next, 32
+  br i1 %x.done, label %loop.i.latch, label %loop.x.header
+
+loop.i.latch:
+  %i.next = add nuw nsw i64 %i, 1
+  %i.done = icmp eq i64 %i.next, 32
+  br i1 %i.done, label %exit, label %loop.i.header
+
+exit:
+  ret void
+}

diff  --git a/llvm/test/Transforms/LoopInterchange/large-nested-6d.ll b/llvm/test/Transforms/LoopInterchange/large-nested-6d.ll
index 590c21fd5a1be..f2c4276f602de 100644
--- a/llvm/test/Transforms/LoopInterchange/large-nested-6d.ll
+++ b/llvm/test/Transforms/LoopInterchange/large-nested-6d.ll
@@ -55,16 +55,33 @@ target datalayout = "e-m:e-p270:32:32-p271:32:32-p272:64:64-i8:8:32-i16:16:32-i6
 ;      Dst:  store double %46, ptr %48, align 8
 ;
 ;
-; CHECK:       --- !Missed
+; CHECK:       --- !Analysis
 ; CHECK-NEXT:  Pass:            loop-interchange
-; CHECK-NEXT:  Name:            UnsupportedLoopNestDepth
+; CHECK-NEXT:  Name:            Dependence
+; CHECK-NEXT:  Function:        test
+; CHECK-NEXT:  Args:
+; CHECK-NEXT:    - String:          Computed dependence info, invoking the transform.
+; CHECK-NEXT:  ...
+; CHECK-NEXT:  --- !Missed
+; CHECK-NEXT:  Pass:            loop-interchange
+; CHECK-NEXT:  Name:            Dependence
 ; CHECK-NEXT:  Function:        test
 ; CHECK-NEXT:  Args:
-; CHECK-NEXT:    - String:          'Unsupported depth of loop nest, the supported range is ['
-; CHECK-NEXT:    - String:          '2'
-; CHECK-NEXT:    - String:          ', '
-; CHECK-NEXT:    - String:          '10'
-; CHECK-NEXT:    - String:          "].\n"
+; CHECK-NEXT:    - String:          Cannot interchange loops due to dependences.
+; CHECK-NEXT:  ...
+; CHECK-NEXT:  --- !Missed
+; CHECK-NEXT:  Pass:            loop-interchange
+; CHECK-NEXT:  Name:            Dependence
+; CHECK-NEXT:  Function:        test
+; CHECK-NEXT:  Args:
+; CHECK-NEXT:    - String:          Cannot interchange loops due to dependences.
+; CHECK-NEXT:  ...
+; CHECK-NEXT:  --- !Missed
+; CHECK-NEXT:  Pass:            loop-interchange
+; CHECK-NEXT:  Name:            Dependence
+; CHECK-NEXT:  Function:        test
+; CHECK-NEXT:  Args:
+; CHECK-NEXT:    - String:          All loops have dependencies in all directions.
 ; CHECK-NEXT:  ...
 ; CHECK-NEXT:  --- !Analysis
 ; CHECK-NEXT:  Pass:            loop-interchange
@@ -80,16 +97,12 @@ target datalayout = "e-m:e-p270:32:32-p271:32:32-p272:64:64-i8:8:32-i16:16:32-i6
 ; CHECK-NEXT:  Args:
 ; CHECK-NEXT:    - String:          Cannot interchange loops due to dependences.
 ; CHECK-NEXT:  ...
-; CHECK-NEXT:  --- !Missed
+; CHECK-NEXT:  --- !Analysis
 ; CHECK-NEXT:  Pass:            loop-interchange
-; CHECK-NEXT:  Name:            UnsupportedLoopNestDepth
+; CHECK-NEXT:  Name:            Dependence
 ; CHECK-NEXT:  Function:        test
 ; CHECK-NEXT:  Args:
-; CHECK-NEXT:    - String:          'Unsupported depth of loop nest, the supported range is ['
-; CHECK-NEXT:    - String:          '2'
-; CHECK-NEXT:    - String:          ', '
-; CHECK-NEXT:    - String:          '10'
-; CHECK-NEXT:    - String:          "].\n"
+; CHECK-NEXT:    - String:          Computed dependence info, invoking the transform.
 ; CHECK-NEXT:  ...
 ; CHECK-NEXT:  --- !Analysis
 ; CHECK-NEXT:  Pass:            loop-interchange

diff  --git a/llvm/test/Transforms/LoopInterchange/missed-outer-prefix-subnest.ll b/llvm/test/Transforms/LoopInterchange/missed-outer-prefix-subnest.ll
new file mode 100644
index 0000000000000..8af54c42e1d48
--- /dev/null
+++ b/llvm/test/Transforms/LoopInterchange/missed-outer-prefix-subnest.ll
@@ -0,0 +1,126 @@
+; REQUIRES: asserts
+; RUN: opt < %s -passes=loop-interchange -loop-interchange-profitabilities=ignore -debug-only=loop-interchange -disable-output -S 2>%t
+; RUN: FileCheck --input-file=%t %s
+
+; This test documents a currently missed optimization opportunity.
+;
+; collectPerfectNests() walks up from each innermost loop and stops as soon as
+; it reaches a loop with more than one subloop. But because [loop.i, loop.j] is never put in
+; any LoopList, the pass never analyses or attempts that interchange — the
+; opportunity is silently missed.
+;
+; Corresponding C code:
+;
+;   for (int i = 0; i < 8; ++i)
+;     for (int j = 0; j < 8; ++j) {
+;       A[i][j] = 0;                // missed: i/j interchange
+;       for (int k = 0; k < 8; ++k) {
+;         for (int l = 0; l < 8; ++l)
+;           for (int m = 0; m < 8; ++m)
+;             Left[m][l] = 0;       // interchanged: l/m swapped
+;         for (int n = 0; n < 8; ++n)
+;           for (int o = 0; o < 8; ++o)
+;             Right[o][n] = 0;      // interchanged: n/o swapped
+;       }
+;     }
+;
+;
+; CHECK:      Processing LoopList of size = 2 containing the following loops:
+; CHECK-NEXT:   - Loop at depth 4 containing: %loop.l.header<header>,%loop.m.header,%loop.m.latch,%loop.l.latch<latch><exiting>
+; CHECK-NEXT:     Loop at depth 5 containing: %loop.m.header<header>,%loop.m.latch<latch><exiting>
+; CHECK-NEXT:   - Loop at depth 5 containing: %loop.m.header<header>,%loop.m.latch<latch><exiting>
+; CHECK: Loops interchanged: outer loop 'loop.l.header' and inner loop 'loop.m.header'
+;
+; CHECK:      Processing LoopList of size = 2 containing the following loops:
+; CHECK-NEXT:   - Loop at depth 4 containing: %loop.n.header<header>,%loop.o.header,%loop.o.latch,%loop.n.latch<latch><exiting>
+; CHECK-NEXT:     Loop at depth 5 containing: %loop.o.header<header>,%loop.o.latch<latch><exiting>
+; CHECK-NEXT:   - Loop at depth 5 containing: %loop.o.header<header>,%loop.o.latch<latch><exiting>
+; CHECK: Loops interchanged: outer loop 'loop.n.header' and inner loop 'loop.o.header'
+;
+;
+; CHECK-NOT: loop.i.header
+; CHECK-NOT: loop.j.header
+
+define void @missed_outer_prefix_subnest(ptr noalias %Left, ptr noalias %Right,
+                                         ptr noalias %A) {
+entry:
+  br label %loop.i.header
+
+loop.i.header:
+  %i = phi i64 [ 0, %entry ], [ %i.next, %loop.i.latch ]
+  br label %loop.j.header
+
+loop.j.header:
+  %j = phi i64 [ 0, %loop.i.header ], [ %j.next, %loop.j.latch ]
+  %a.row = mul nuw nsw i64 %i, 8
+  %a.idx = add nuw nsw i64 %a.row, %j
+  %a.ptr = getelementptr i8, ptr %A, i64 %a.idx
+  store i8 0, ptr %a.ptr, align 1
+  br label %loop.k.header
+
+loop.k.header:
+  %k = phi i64 [ 0, %loop.j.header ], [ %k.next, %loop.k.latch ]
+  br label %loop.l.header
+
+loop.l.header:
+  %l = phi i64 [ 0, %loop.k.header ], [ %l.next, %loop.l.latch ]
+  br label %loop.m.header
+
+loop.m.header:
+  %m = phi i64 [ 0, %loop.l.header ], [ %m.next, %loop.m.latch ]
+  %left.row.base = mul nuw nsw i64 %m, 8
+  %left.index = add nuw nsw i64 %left.row.base, %l
+  %left.element.ptr = getelementptr i8, ptr %Left, i64 %left.index
+  store i8 0, ptr %left.element.ptr, align 1
+  br label %loop.m.latch
+
+loop.m.latch:
+  %m.next = add nuw nsw i64 %m, 1
+  %m.done = icmp eq i64 %m.next, 8
+  br i1 %m.done, label %loop.l.latch, label %loop.m.header
+
+loop.l.latch:
+  %l.next = add nuw nsw i64 %l, 1
+  %l.done = icmp eq i64 %l.next, 8
+  br i1 %l.done, label %loop.n.header, label %loop.l.header
+
+loop.n.header:
+  %n = phi i64 [ 0, %loop.l.latch ], [ %n.next, %loop.n.latch ]
+  br label %loop.o.header
+
+loop.o.header:
+  %o = phi i64 [ 0, %loop.n.header ], [ %o.next, %loop.o.latch ]
+  %right.row.base = mul nuw nsw i64 %o, 8
+  %right.index = add nuw nsw i64 %right.row.base, %n
+  %right.element.ptr = getelementptr i8, ptr %Right, i64 %right.index
+  store i8 0, ptr %right.element.ptr, align 1
+  br label %loop.o.latch
+
+loop.o.latch:
+  %o.next = add nuw nsw i64 %o, 1
+  %o.done = icmp eq i64 %o.next, 8
+  br i1 %o.done, label %loop.n.latch, label %loop.o.header
+
+loop.n.latch:
+  %n.next = add nuw nsw i64 %n, 1
+  %n.done = icmp eq i64 %n.next, 8
+  br i1 %n.done, label %loop.k.latch, label %loop.n.header
+
+loop.k.latch:
+  %k.next = add nuw nsw i64 %k, 1
+  %k.done = icmp eq i64 %k.next, 8
+  br i1 %k.done, label %loop.j.latch, label %loop.k.header
+
+loop.j.latch:
+  %j.next = add nuw nsw i64 %j, 1
+  %j.done = icmp eq i64 %j.next, 8
+  br i1 %j.done, label %loop.i.latch, label %loop.j.header
+
+loop.i.latch:
+  %i.next = add nuw nsw i64 %i, 1
+  %i.done = icmp eq i64 %i.next, 8
+  br i1 %i.done, label %exit, label %loop.i.header
+
+exit:
+  ret void
+}

diff  --git a/llvm/test/Transforms/LoopInterchange/no-partially-perfect-subnest.ll b/llvm/test/Transforms/LoopInterchange/no-partially-perfect-subnest.ll
new file mode 100644
index 0000000000000..96a2c15740bde
--- /dev/null
+++ b/llvm/test/Transforms/LoopInterchange/no-partially-perfect-subnest.ll
@@ -0,0 +1,63 @@
+; REQUIRES: asserts
+; RUN: opt < %s -passes=loop-interchange -loop-interchange-profitabilities=ignore -debug-only=loop-interchange -disable-output 2>&1 | FileCheck %s
+
+
+; There is no partially-perfect subnest here. Every innermost loop has a parent
+; with multiple child loops, so collectPerfectNests() should return an empty
+; list and loop-interchange should bail out immediately without attempting or
+; performing any interchange.
+;
+; Corresponding C code:
+;
+;   for (int i = 0; i < 16; ++i) {
+;     for (int j = 0; j < 16; ++j)
+;       A[i][j] = 0;
+;
+;     for (int k = 0; k < 16; ++k)
+;       B[i][k] = 0;
+;   }
+;
+; CHECK: No Valid candidates for loop interchange.
+
+define void @no_partially_perfect_subnest(ptr noalias %A, ptr noalias %B) {
+entry:
+  br label %loop.i.header
+
+loop.i.header:
+  %i = phi i64 [ 0, %entry ], [ %i.next, %loop.i.latch ]
+  br label %loop.j.header
+
+loop.j.header:
+  %j = phi i64 [ 0, %loop.i.header ], [ %j.next, %loop.j.latch ]
+  %a.row.base = mul nuw nsw i64 %i, 16
+  %a.index = add nuw nsw i64 %a.row.base, %j
+  %a.element.ptr = getelementptr i8, ptr %A, i64 %a.index
+  store i8 1, ptr %a.element.ptr, align 1
+  br label %loop.j.latch
+
+loop.j.latch:
+  %j.next = add nuw nsw i64 %j, 1
+  %j.done = icmp eq i64 %j.next, 16
+  br i1 %j.done, label %loop.k.header, label %loop.j.header
+
+loop.k.header:
+  %k = phi i64 [ 0, %loop.j.latch ], [ %k.next, %loop.k.latch ]
+  %b.row.base = mul nuw nsw i64 %i, 16
+  %b.index = add nuw nsw i64 %b.row.base, %k
+  %b.element.ptr = getelementptr i8, ptr %B, i64 %b.index
+  store i8 2, ptr %b.element.ptr, align 1
+  br label %loop.k.latch
+
+loop.k.latch:
+  %k.next = add nuw nsw i64 %k, 1
+  %k.done = icmp eq i64 %k.next, 16
+  br i1 %k.done, label %loop.i.latch, label %loop.k.header
+
+loop.i.latch:
+  %i.next = add nuw nsw i64 %i, 1
+  %i.done = icmp eq i64 %i.next, 16
+  br i1 %i.done, label %exit, label %loop.i.header
+
+exit:
+  ret void
+}

diff  --git a/llvm/test/Transforms/LoopInterchange/partially-perfect-loop.ll b/llvm/test/Transforms/LoopInterchange/partially-perfect-loop.ll
index 7d72b4a7f1804..43cdcce5f884d 100644
--- a/llvm/test/Transforms/LoopInterchange/partially-perfect-loop.ll
+++ b/llvm/test/Transforms/LoopInterchange/partially-perfect-loop.ll
@@ -17,39 +17,57 @@ define void @f(ptr noalias %A, ptr noalias %B) {
 ; CHECK-NEXT:  [[FOR_J_HEADER:.*]]:
 ; CHECK-NEXT:    br label %[[FOR_R_HEADER_SPLIT1:.*]]
 ; CHECK:       [[FOR_R_HEADER_SPLIT1]]:
-; CHECK-NEXT:    [[I:%.*]] = phi i64 [ 0, %[[FOR_J_HEADER]] ], [ [[I_NEXT:%.*]], %[[FOR_I_LATCH:.*]] ]
+; CHECK-NEXT:    [[I:%.*]] = phi i64 [ 0, %[[FOR_J_HEADER]] ], [ [[I_NEXT:%.*]], %[[FOR_I_LATCH1:.*]] ]
 ; CHECK-NEXT:    br label %[[FOR_R_HEADER:.*]]
-; CHECK:       [[FOR_R_HEADER]]:
-; CHECK-NEXT:    [[J:%.*]] = phi i64 [ 0, %[[FOR_R_HEADER_SPLIT1]] ], [ [[J_NEXT:%.*]], %[[FOR_J_LATCH:.*]] ]
+; CHECK:       [[FOR_J_HEADER_PREHEADER1:.*]]:
 ; CHECK-NEXT:    br label %[[FOR_J_HEADER_PREHEADER:.*]]
 ; CHECK:       [[FOR_J_HEADER_PREHEADER]]:
-; CHECK-NEXT:    [[R:%.*]] = phi i64 [ 0, %[[FOR_R_HEADER]] ], [ [[TMP0:%.*]], %[[FOR_J_HEADER_PREHEADER]] ]
+; CHECK-NEXT:    [[J:%.*]] = phi i64 [ [[J_NEXT:%.*]], %[[FOR_J_LATCH1:.*]] ], [ 0, %[[FOR_J_HEADER_PREHEADER1]] ]
+; CHECK-NEXT:    br label %[[FOR_R_HEADER_SPLIT2:.*]]
+; CHECK:       [[FOR_R_HEADER]]:
+; CHECK-NEXT:    br label %[[FOR_R_HEADER1:.*]]
+; CHECK:       [[FOR_R_HEADER1]]:
+; CHECK-NEXT:    [[R:%.*]] = phi i64 [ [[TMP4:%.*]], %[[FOR_J_LATCH:.*]] ], [ 0, %[[FOR_R_HEADER]] ]
+; CHECK-NEXT:    br label %[[FOR_J_HEADER_PREHEADER1]]
+; CHECK:       [[FOR_R_HEADER_SPLIT2]]:
 ; CHECK-NEXT:    [[A_ELEMENT:%.*]] = getelementptr [64 x i8], ptr [[A]], i64 [[J]], i64 [[R]]
 ; CHECK-NEXT:    store i8 0, ptr [[A_ELEMENT]], align 1
-; CHECK-NEXT:    [[TMP0]] = add i64 [[R]], 1
+; CHECK-NEXT:    [[TMP0:%.*]] = add i64 [[R]], 1
 ; CHECK-NEXT:    [[TMP1:%.*]] = icmp eq i64 [[TMP0]], 64
-; CHECK-NEXT:    br i1 [[TMP1]], label %[[FOR_J_LATCH]], label %[[FOR_J_HEADER_PREHEADER]]
+; CHECK-NEXT:    br label %[[FOR_J_LATCH1]]
 ; CHECK:       [[FOR_J_LATCH]]:
+; CHECK-NEXT:    [[TMP4]] = add i64 [[R]], 1
+; CHECK-NEXT:    [[TMP5:%.*]] = icmp eq i64 [[TMP4]], 64
+; CHECK-NEXT:    br i1 [[TMP5]], label %[[FOR_L_HEADER_PREHEADER1:.*]], label %[[FOR_R_HEADER1]]
+; CHECK:       [[FOR_J_LATCH1]]:
 ; CHECK-NEXT:    [[J_NEXT]] = add i64 [[J]], 1
 ; CHECK-NEXT:    [[J_DONE:%.*]] = icmp eq i64 [[J_NEXT]], 64
-; CHECK-NEXT:    br i1 [[J_DONE]], label %[[FOR_L_HEADER_PREHEADER:.*]], label %[[FOR_R_HEADER]]
-; CHECK:       [[FOR_L_HEADER_PREHEADER]]:
+; CHECK-NEXT:    br i1 [[J_DONE]], label %[[FOR_J_LATCH]], label %[[FOR_J_HEADER_PREHEADER]]
+; CHECK:       [[FOR_L_HEADER_PREHEADER:.*]]:
 ; CHECK-NEXT:    br label %[[FOR_L_HEADER:.*]]
 ; CHECK:       [[FOR_L_HEADER]]:
 ; CHECK-NEXT:    [[K:%.*]] = phi i64 [ [[K_NEXT:%.*]], %[[FOR_K_LATCH:.*]] ], [ 0, %[[FOR_L_HEADER_PREHEADER]] ]
 ; CHECK-NEXT:    br label %[[FOR_K_HEADER_PREHEADER:.*]]
+; CHECK:       [[FOR_L_HEADER_PREHEADER1]]:
+; CHECK-NEXT:    br label %[[FOR_L_HEADER1:.*]]
+; CHECK:       [[FOR_L_HEADER1]]:
+; CHECK-NEXT:    [[L:%.*]] = phi i64 [ [[TMP6:%.*]], %[[FOR_I_LATCH:.*]] ], [ 0, %[[FOR_L_HEADER_PREHEADER1]] ]
+; CHECK-NEXT:    br label %[[FOR_L_HEADER_PREHEADER]]
 ; CHECK:       [[FOR_K_HEADER_PREHEADER]]:
-; CHECK-NEXT:    [[L:%.*]] = phi i64 [ 0, %[[FOR_L_HEADER]] ], [ [[TMP2:%.*]], %[[FOR_K_HEADER_PREHEADER]] ]
 ; CHECK-NEXT:    [[B_ELEMENT:%.*]] = getelementptr [64 x i8], ptr [[B]], i64 [[L]], i64 [[K]]
 ; CHECK-NEXT:    store i8 0, ptr [[B_ELEMENT]], align 1
-; CHECK-NEXT:    [[TMP2]] = add i64 [[L]], 1
+; CHECK-NEXT:    [[TMP2:%.*]] = add i64 [[L]], 1
 ; CHECK-NEXT:    [[TMP3:%.*]] = icmp eq i64 [[TMP2]], 64
-; CHECK-NEXT:    br i1 [[TMP3]], label %[[FOR_K_LATCH]], label %[[FOR_K_HEADER_PREHEADER]]
+; CHECK-NEXT:    br label %[[FOR_K_LATCH]]
+; CHECK:       [[FOR_I_LATCH]]:
+; CHECK-NEXT:    [[TMP6]] = add i64 [[L]], 1
+; CHECK-NEXT:    [[TMP7:%.*]] = icmp eq i64 [[TMP6]], 64
+; CHECK-NEXT:    br i1 [[TMP7]], label %[[FOR_I_LATCH1]], label %[[FOR_L_HEADER1]]
 ; CHECK:       [[FOR_K_LATCH]]:
 ; CHECK-NEXT:    [[K_NEXT]] = add i64 [[K]], 1
 ; CHECK-NEXT:    [[K_DONE:%.*]] = icmp eq i64 [[K_NEXT]], 64
 ; CHECK-NEXT:    br i1 [[K_DONE]], label %[[FOR_I_LATCH]], label %[[FOR_L_HEADER]]
-; CHECK:       [[FOR_I_LATCH]]:
+; CHECK:       [[FOR_I_LATCH1]]:
 ; CHECK-NEXT:    [[I_NEXT]] = add i64 [[I]], 1
 ; CHECK-NEXT:    [[I_DONE:%.*]] = icmp eq i64 [[I_NEXT]], 64
 ; CHECK-NEXT:    br i1 [[I_DONE]], label %[[EXIT:.*]], label %[[FOR_R_HEADER_SPLIT1]]

diff  --git a/llvm/test/Transforms/LoopInterchange/preserve-sibling-order.ll b/llvm/test/Transforms/LoopInterchange/preserve-sibling-order.ll
new file mode 100644
index 0000000000000..7997fd71e6572
--- /dev/null
+++ b/llvm/test/Transforms/LoopInterchange/preserve-sibling-order.ll
@@ -0,0 +1,72 @@
+; RUN: opt -passes='function(loop(loop-interchange),print<loops>)' \
+; RUN:     -loop-interchange-profitabilities=ignore -disable-output < %s 2>&1 | \
+; RUN:     FileCheck %s
+;
+; A parent loop 'outer' contains an interchangeable perfect pair (pair.j/pair.k)
+; as its first subloop, followed by two sibling loops (sib1, sib2). After
+; interchanging the pair, LoopInfo must keep it in its original slot -- ahead of
+; the siblings -- so that the preserved LoopAnalysis matches a rebuilt one and
+; later loop passes see a consistent sibling traversal order.
+;
+;   for (i = 0; i < 64; i++) {
+;     for (j = 0; j < 64; j++)      // block %pair.j \  interchangeable pair
+;       for (k = 0; k < 64; k++)    // block %pair.k /  (first subloop of the i-loop)
+;         A[k][j] = 0;
+;     for (s1 = 0; s1 < 4; s1++);   // sibling
+;     for (s2 = 0; s2 < 4; s2++);   // sibling
+;   }
+
+; The interchanged pair (new outer header %pair.k, new inner header %pair.j)
+; stays first, ahead of sib1 and sib2.
+; CHECK-LABEL: Loop info for function 'f':
+; CHECK:         Loop at depth 1 containing: %outer.header<header>
+; CHECK-NEXT:      Loop at depth 2 containing: %pair.k<header>
+; CHECK-NEXT:        Loop at depth 3 containing: %pair.j<header>
+; CHECK-NEXT:      Loop at depth 2 containing: %sib1.header<header>
+; CHECK-NEXT:      Loop at depth 2 containing: %sib2.header<header>
+
+define void @f(ptr noalias %A) {
+entry:
+  br label %outer.header
+
+outer.header:
+  %i = phi i64 [ 0, %entry ], [ %i.next, %outer.latch ]
+  br label %pair.j
+
+pair.j:
+  %j = phi i64 [ 0, %outer.header ], [ %j.next, %pair.j.latch ]
+  br label %pair.k
+
+pair.k:
+  %k = phi i64 [ 0, %pair.j ], [ %k.next, %pair.k ]
+  %idx = getelementptr [64 x i8], ptr %A, i64 %k, i64 %j
+  store i8 0, ptr %idx, align 1
+  %k.next = add i64 %k, 1
+  %k.done = icmp eq i64 %k.next, 64
+  br i1 %k.done, label %pair.j.latch, label %pair.k
+
+pair.j.latch:
+  %j.next = add i64 %j, 1
+  %j.done = icmp eq i64 %j.next, 64
+  br i1 %j.done, label %sib1.header, label %pair.j
+
+sib1.header:
+  %s1 = phi i64 [ 0, %pair.j.latch ], [ %s1.next, %sib1.header ]
+  %s1.next = add i64 %s1, 1
+  %s1.done = icmp eq i64 %s1.next, 4
+  br i1 %s1.done, label %sib2.header, label %sib1.header
+
+sib2.header:
+  %s2 = phi i64 [ 0, %sib1.header ], [ %s2.next, %sib2.header ]
+  %s2.next = add i64 %s2, 1
+  %s2.done = icmp eq i64 %s2.next, 4
+  br i1 %s2.done, label %outer.latch, label %sib2.header
+
+outer.latch:
+  %i.next = add i64 %i, 1
+  %i.done = icmp eq i64 %i.next, 64
+  br i1 %i.done, label %exit, label %outer.header
+
+exit:
+  ret void
+}


        


More information about the llvm-commits mailing list