[llvm] [LoopInterchange] Swap preheader contents before rebuilding LCSSA (PR #218468)

via llvm-commits llvm-commits at lists.llvm.org
Tue Aug 25 00:16:00 PDT 2026


https://github.com/nishant-sachdeva-amd updated https://github.com/llvm/llvm-project/pull/218468

>From 4417f50312c0bfdceaf11f47783c10d8f1a1db6c Mon Sep 17 00:00:00 2001
From: nsachdev <nishant.sachdeva at amd.com>
Date: Mon, 24 Aug 2026 22:20:46 +0530
Subject: [PATCH 1/2] [LoopInterchange] Swap preheader contents before
 rebuilding LCSSA

adjustLoopLinks() swapped the inner/outer preheader bodies only after
adjustLoopBranches() had already moved the reduction PHIs and rebuilt
LCSSA via formLCSSAForInstructions(). A reduction init defined in the
inner preheader was therefore still stranded below its use on the seed
edge when LCSSA was rebuilt, so formLCSSAForInstructions() was handed
dominance-broken IR.

Move the swapBBContents() call into adjustLoopBranches(), after the
replacePhiUsesWith() relabeling and before the LCSSA rebuild, so the
reduction init is relocated into the new outer preheader and dominates
the seed edge. With the swap moved, adjustLoopLinks() only forwarded to
adjustLoopBranches(), so it is inlined into its sole caller and removed.

Fixes #215511
---
 .../lib/Transforms/Scalar/LoopInterchange.cpp |  22 ++--
 .../reduction-seed-relocation.ll              | 103 ++++++++++++++++++
 2 files changed, 110 insertions(+), 15 deletions(-)
 create mode 100644 llvm/test/Transforms/LoopInterchange/reduction-seed-relocation.ll

diff --git a/llvm/lib/Transforms/Scalar/LoopInterchange.cpp b/llvm/lib/Transforms/Scalar/LoopInterchange.cpp
index 53cd26b9b9011..708f26bace8d0 100644
--- a/llvm/lib/Transforms/Scalar/LoopInterchange.cpp
+++ b/llvm/lib/Transforms/Scalar/LoopInterchange.cpp
@@ -636,7 +636,6 @@ class LoopInterchangeTransform {
   void removeChildLoop(Loop *OuterLoop, Loop *InnerLoop);
 
 private:
-  void adjustLoopLinks();
   void adjustLoopBranches();
 
   Loop *OuterLoop;
@@ -815,7 +814,7 @@ bool LoopInterchangeLegality::containsUnsafeInstructions(BasicBlock *BB,
 
 static FreezeInst *findFreezeInReNestedBlocks(Loop *OuterLoop,
                                               Loop *InnerLoop) {
-  // adjustLoopLinks swaps the preheader bodies after changing their loop
+  // adjustLoopBranches swaps the preheader bodies after changing their loop
   // roles, so the original outer-preheader body remains outside the new outer
   // loop and retains its execution count.
   BasicBlock *Blocks[] = {
@@ -2281,7 +2280,7 @@ void LoopInterchangeTransform::transform(
       I.moveBeforePreserving(OuterLoopHeader->getTerminator()->getIterator());
   }
 
-  adjustLoopLinks();
+  adjustLoopBranches();
 
   // Finally, drop the nsw/nuw/ninf flags from the instructions for reduction
   // calculations.
@@ -2639,6 +2638,11 @@ void LoopInterchangeTransform::adjustLoopBranches() {
   InnerLoopHeader->replacePhiUsesWith(OuterLoopPreHeader, InnerLoopPreHeader);
   InnerLoopHeader->replacePhiUsesWith(OuterLoopLatch, InnerLoopLatch);
 
+  // Swap the preheader contents before rebuilding LCSSA below, so that a
+  // reduction init defined in the inner preheader is relocated into the new
+  // outer preheader and dominates its use on the reduction PHI's seed edge.
+  swapBBContents(OuterLoop->getLoopPreheader(), InnerLoop->getLoopPreheader());
+
   // Values defined in the outer loop header could be used in the inner loop
   // latch. In that case, we need to create LCSSA phis for them, because after
   // interchanging they will be defined in the new inner loop and used in the
@@ -2650,18 +2654,6 @@ void LoopInterchangeTransform::adjustLoopBranches() {
   formLCSSAForInstructions(MayNeedLCSSAPhis, *DT, *LI, SE);
 }
 
-void LoopInterchangeTransform::adjustLoopLinks() {
-  // Adjust all branches in the inner and outer loop.
-  adjustLoopBranches();
-
-  // We have interchanged the preheaders so we need to interchange the data in
-  // the preheaders as well. This is because the content of the inner
-  // preheader was previously executed inside the outer loop.
-  BasicBlock *OuterLoopPreHeader = OuterLoop->getLoopPreheader();
-  BasicBlock *InnerLoopPreHeader = InnerLoop->getLoopPreheader();
-  swapBBContents(OuterLoopPreHeader, InnerLoopPreHeader);
-}
-
 PreservedAnalyses LoopInterchangePass::run(LoopNest &LN,
                                            LoopAnalysisManager &AM,
                                            LoopStandardAnalysisResults &AR,
diff --git a/llvm/test/Transforms/LoopInterchange/reduction-seed-relocation.ll b/llvm/test/Transforms/LoopInterchange/reduction-seed-relocation.ll
new file mode 100644
index 0000000000000..cbc7c4e0fad8c
--- /dev/null
+++ b/llvm/test/Transforms/LoopInterchange/reduction-seed-relocation.ll
@@ -0,0 +1,103 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py
+; RUN: opt < %s -passes=loop-interchange -cache-line-size=64 -verify-dom-info -verify-loop-info -verify-scev -verify-loop-lcssa -S | FileCheck %s
+
+; A cross-inner/outer reduction over the (j,i) pair, nested inside an outer k
+; loop. The reduction seed %seed is defined in the pair's preheader. When j/i
+; interchange, the reduction PHI moves to the new outer header and its seed edge
+; references %seed. The preheader-content swap must run before the LCSSA rebuild
+; so that %seed is relocated into the new outer preheader and dominates the seed
+; edge; otherwise formLCSSAForInstructions is handed dominance-broken IR.
+
+target datalayout = "e-m:e-i64:64-f80:128-n8:16:32:64-S128"
+
+define i64 @adv2(ptr %Arr) {
+; CHECK-LABEL: @adv2(
+; CHECK-NEXT:  entry:
+; CHECK-NEXT:    br label [[K_HEADER:%.*]]
+; CHECK:       k.header:
+; CHECK-NEXT:    [[K:%.*]] = phi i64 [ 0, [[ENTRY:%.*]] ], [ [[K_NEXT:%.*]], [[K_LATCH:%.*]] ]
+; CHECK-NEXT:    [[TOTAL_OUTER:%.*]] = phi i64 [ 0, [[ENTRY]] ], [ [[TOTAL_NEXT:%.*]], [[K_LATCH]] ]
+; CHECK-NEXT:    br label [[I_BODY_PREHEADER:%.*]]
+; CHECK:       j.preheader:
+; CHECK-NEXT:    br label [[J_HEADER:%.*]]
+; CHECK:       j.header:
+; CHECK-NEXT:    [[J:%.*]] = phi i64 [ 0, [[J_PREHEADER:%.*]] ], [ [[J_NEXT:%.*]], [[J_INC:%.*]] ]
+; CHECK-NEXT:    [[ACC_INNER:%.*]] = phi i64 [ [[ACC_INNER_INC:%.*]], [[J_INC]] ], [ [[ACC_OUTER:%.*]], [[J_PREHEADER]] ]
+; CHECK-NEXT:    br label [[I_BODY_SPLIT1:%.*]]
+; CHECK:       i.body.preheader:
+; CHECK-NEXT:    [[SEED:%.*]] = mul i64 [[K]], 7
+; CHECK-NEXT:    br label [[I_BODY:%.*]]
+; CHECK:       i.body:
+; CHECK-NEXT:    [[I:%.*]] = phi i64 [ [[TMP0:%.*]], [[I_BODY_SPLIT:%.*]] ], [ 0, [[I_BODY_PREHEADER]] ]
+; CHECK-NEXT:    [[ACC_OUTER]] = phi i64 [ [[SEED]], [[I_BODY_PREHEADER]] ], [ [[ACC_INNER_LCSSA:%.*]], [[I_BODY_SPLIT]] ]
+; CHECK-NEXT:    br label [[J_PREHEADER]]
+; CHECK:       i.body.split1:
+; CHECK-NEXT:    [[IDX:%.*]] = getelementptr inbounds [100 x [100 x i64]], ptr [[ARR:%.*]], i64 0, i64 [[I]], i64 [[J]]
+; CHECK-NEXT:    [[V:%.*]] = load i64, ptr [[IDX]], align 4
+; CHECK-NEXT:    [[ACC_INNER_INC]] = add i64 [[ACC_INNER]], [[V]]
+; CHECK-NEXT:    [[I_NEXT:%.*]] = add nuw nsw i64 [[I]], 1
+; CHECK-NEXT:    [[I_EC:%.*]] = icmp eq i64 [[I_NEXT]], 100
+; CHECK-NEXT:    br label [[J_INC]]
+; CHECK:       i.body.split:
+; CHECK-NEXT:    [[ACC_INNER_LCSSA]] = phi i64 [ [[ACC_INNER_INC]], [[J_INC]] ]
+; CHECK-NEXT:    [[TMP0]] = add nuw nsw i64 [[I]], 1
+; CHECK-NEXT:    [[TMP1:%.*]] = icmp eq i64 [[TMP0]], 100
+; CHECK-NEXT:    br i1 [[TMP1]], label [[K_LATCH]], label [[I_BODY]]
+; CHECK:       j.inc:
+; CHECK-NEXT:    [[J_NEXT]] = add nuw nsw i64 [[J]], 1
+; CHECK-NEXT:    [[J_EC:%.*]] = icmp eq i64 [[J_NEXT]], 100
+; CHECK-NEXT:    br i1 [[J_EC]], label [[I_BODY_SPLIT]], label [[J_HEADER]]
+; CHECK:       k.latch:
+; CHECK-NEXT:    [[ACC_FINAL:%.*]] = phi i64 [ [[ACC_INNER_LCSSA]], [[I_BODY_SPLIT]] ]
+; CHECK-NEXT:    [[TOTAL_NEXT]] = add i64 [[TOTAL_OUTER]], [[ACC_FINAL]]
+; CHECK-NEXT:    [[K_NEXT]] = add nuw nsw i64 [[K]], 1
+; CHECK-NEXT:    [[K_EC:%.*]] = icmp eq i64 [[K_NEXT]], 100
+; CHECK-NEXT:    br i1 [[K_EC]], label [[EXIT:%.*]], label [[K_HEADER]]
+; CHECK:       exit:
+; CHECK-NEXT:    [[TOTAL_LCSSA:%.*]] = phi i64 [ [[TOTAL_NEXT]], [[K_LATCH]] ]
+; CHECK-NEXT:    ret i64 [[TOTAL_LCSSA]]
+;
+entry:
+  br label %k.header
+
+k.header:
+  %k = phi i64 [ 0, %entry ], [ %k.next, %k.latch ]
+  %total.outer = phi i64 [ 0, %entry ], [ %total.next, %k.latch ]
+  br label %j.preheader
+
+j.preheader:
+  %seed = mul i64 %k, 7
+  br label %j.header
+
+j.header:
+  %j = phi i64 [ 0, %j.preheader ], [ %j.next, %j.inc ]
+  %acc.outer = phi i64 [ %seed, %j.preheader ], [ %acc.inner.lcssa, %j.inc ]
+  br label %i.body
+
+i.body:
+  %acc.inner = phi i64 [ %acc.outer, %j.header ], [ %acc.inner.inc, %i.body ]
+  %i = phi i64 [ 0, %j.header ], [ %i.next, %i.body ]
+  %idx = getelementptr inbounds [100 x [100 x i64]], ptr %Arr, i64 0, i64 %i, i64 %j
+  %v = load i64, ptr %idx, align 4
+  %acc.inner.inc = add i64 %acc.inner, %v
+  %i.next = add nuw nsw i64 %i, 1
+  %i.ec = icmp eq i64 %i.next, 100
+  br i1 %i.ec, label %j.inc, label %i.body
+
+j.inc:
+  %acc.inner.lcssa = phi i64 [ %acc.inner.inc, %i.body ]
+  %j.next = add nuw nsw i64 %j, 1
+  %j.ec = icmp eq i64 %j.next, 100
+  br i1 %j.ec, label %k.latch, label %j.header
+
+k.latch:
+  %acc.final = phi i64 [ %acc.inner.lcssa, %j.inc ]
+  %total.next = add i64 %total.outer, %acc.final
+  %k.next = add nuw nsw i64 %k, 1
+  %k.ec = icmp eq i64 %k.next, 100
+  br i1 %k.ec, label %exit, label %k.header
+
+exit:
+  %total.lcssa = phi i64 [ %total.next, %k.latch ]
+  ret i64 %total.lcssa
+}

>From c0c784366e2e3b4cef014130bdaf4653c8719d3f Mon Sep 17 00:00:00 2001
From: nsachdev <nishant.sachdeva at amd.com>
Date: Tue, 25 Aug 2026 12:45:19 +0530
Subject: [PATCH 2/2] [DEBUG] LoopInterchange: assert dominance before LCSSA
 rebuild; trim test flags

Address review feedback:
- Add verifyFunction assert after relocating the reduction seed and before
  formLCSSAForInstructions, guarding against a future reordering that hands
  dominance-broken IR to the LCSSA rebuild.
- Trim reduction-seed-relocation.ll RUN flags to the minimal set
  (-verify-loop-lcssa); the assert now makes the test fail pre-fix.
---
 llvm/lib/Transforms/Scalar/LoopInterchange.cpp            | 8 ++++++++
 .../LoopInterchange/reduction-seed-relocation.ll          | 2 +-
 2 files changed, 9 insertions(+), 1 deletion(-)

diff --git a/llvm/lib/Transforms/Scalar/LoopInterchange.cpp b/llvm/lib/Transforms/Scalar/LoopInterchange.cpp
index 708f26bace8d0..3d2c600c901d6 100644
--- a/llvm/lib/Transforms/Scalar/LoopInterchange.cpp
+++ b/llvm/lib/Transforms/Scalar/LoopInterchange.cpp
@@ -37,6 +37,7 @@
 #include "llvm/IR/Instructions.h"
 #include "llvm/IR/User.h"
 #include "llvm/IR/Value.h"
+#include "llvm/IR/Verifier.h"
 #include "llvm/Support/Casting.h"
 #include "llvm/Support/CommandLine.h"
 #include "llvm/Support/Debug.h"
@@ -2651,6 +2652,13 @@ void LoopInterchangeTransform::adjustLoopBranches() {
   for (Instruction &I :
        make_range(OuterLoopHeader->begin(), std::prev(OuterLoopHeader->end())))
     MayNeedLCSSAPhis.push_back(&I);
+
+  // The preheader-content swap above relocates the reduction seed so the IR is
+  // already dominance-valid here. Guard against a future reordering that would
+  // hand dominance-broken IR to the LCSSA rebuild below.
+  assert(!verifyFunction(*OuterLoopHeader->getParent(), &errs()) &&
+         "LoopInterchange handed dominance-broken IR to LCSSA rebuild");
+
   formLCSSAForInstructions(MayNeedLCSSAPhis, *DT, *LI, SE);
 }
 
diff --git a/llvm/test/Transforms/LoopInterchange/reduction-seed-relocation.ll b/llvm/test/Transforms/LoopInterchange/reduction-seed-relocation.ll
index cbc7c4e0fad8c..e20ef7a904113 100644
--- a/llvm/test/Transforms/LoopInterchange/reduction-seed-relocation.ll
+++ b/llvm/test/Transforms/LoopInterchange/reduction-seed-relocation.ll
@@ -1,5 +1,5 @@
 ; NOTE: Assertions have been autogenerated by utils/update_test_checks.py
-; RUN: opt < %s -passes=loop-interchange -cache-line-size=64 -verify-dom-info -verify-loop-info -verify-scev -verify-loop-lcssa -S | FileCheck %s
+; RUN: opt < %s -passes=loop-interchange -verify-loop-lcssa -S | FileCheck %s
 
 ; A cross-inner/outer reduction over the (j,i) pair, nested inside an outer k
 ; loop. The reduction seed %seed is defined in the pair's preheader. When j/i



More information about the llvm-commits mailing list