[llvm] [LoopFusion] Cleanup control flow before Fusion (PR #226289)

Ehsan Amiri via llvm-commits llvm-commits at lists.llvm.org
Thu Sep 24 13:00:58 PDT 2026


https://github.com/amehsan created https://github.com/llvm/llvm-project/pull/226289

NOT READY FOR REVIEW YET

Non-loop successor of loop guard is sometimes modified by JumpThreading (and potentially some other passes) to jump to a new block, that doesn't post dominate the exit block of the loop:

   Guard:    br %c, %Preheader, %Skip
   Skip:     br %Merge
   ...
   Exit:     br %Merge
   Merge:    ...

Later on, this block becomes empty, so it can be removed. Its presence prevents Loop::getLoopGuardBranch() from detecting the loop guard which results in missed fusion opportunities. SimplifyCFG will run when "Skip" is still not optimzied, so it cannot resolve this issue.

>From d1a239a2eea98483ccc302b1bcba354a69b760ce Mon Sep 17 00:00:00 2001
From: Ehsan Amiri <ehsan.amiri at huawei.com>
Date: Thu, 24 Sep 2026 15:26:58 -0400
Subject: [PATCH] [LoopFusion] Cleanup control flow before Fusion

Non-loop successor of loop guard is sometimes modified by JumpThreading (and
potentially some other passes) to jump to a new block, that doesn't post dominate
the exit block of the loop:

   Guard:    br %c, %Preheader, %Skip
   Skip:     br %Merge
   ...
   Exit:     br %Merge
   Merge:    ...

Later on, this block becomes empty, so it can be removed. Its presence prevents
Loop::getLoopGuardBranch() from detecting the loop guard which results in missed
fusion opportunities. SimplifyCFG will run when "Skip" is still not optimzied,
so it cannot resolve this issue.
---
 llvm/lib/Transforms/Scalar/LoopFuse.cpp       |  88 ++++
 .../LoopFusion/guard_skip_empty_block.ll      | 389 ++++++++++++++++++
 2 files changed, 477 insertions(+)
 create mode 100644 llvm/test/Transforms/LoopFusion/guard_skip_empty_block.ll

diff --git a/llvm/lib/Transforms/Scalar/LoopFuse.cpp b/llvm/lib/Transforms/Scalar/LoopFuse.cpp
index fd2e84b2de5a7..5804aedc85b48 100644
--- a/llvm/lib/Transforms/Scalar/LoopFuse.cpp
+++ b/llvm/lib/Transforms/Scalar/LoopFuse.cpp
@@ -50,6 +50,7 @@
 #include "llvm/Analysis/DependenceAnalysis.h"
 #include "llvm/Analysis/DomTreeUpdater.h"
 #include "llvm/Analysis/LoopInfo.h"
+#include "llvm/Analysis/LoopNestAnalysis.h"
 #include "llvm/Analysis/OptimizationRemarkEmitter.h"
 #include "llvm/Analysis/PostDominators.h"
 #include "llvm/Analysis/ScalarEvolution.h"
@@ -63,6 +64,7 @@
 #include "llvm/Transforms/Utils/CodeMoverUtils.h"
 #include "llvm/Transforms/Utils/LoopPeel.h"
 #include "llvm/Transforms/Utils/LoopSimplify.h"
+#include "llvm/Transforms/Utils/LoopUtils.h"
 #include <list>
 
 using namespace llvm;
@@ -403,6 +405,87 @@ printFusionCandidates(const FusionCandidateCollection &FusionCandidates) {
 }
 #endif // NDEBUG
 
+/// Fold away an empty block on the "skip" edge of \p L's loop guard, if any.
+///
+/// Loop::getLoopGuardBranch() recognizes a guard only when the non-loop
+/// successor of the guard branch is the block that the loop exit flows into
+/// (looking through empty blocks on the exit side only). Passes such as
+/// JumpThreading can leave an empty forwarding block on the guard side
+/// instead:
+///
+///   Guard:    br %c, %Preheader, %Skip
+///   Skip:     br %Merge             ; empty, only reachable from Guard
+///   ...
+///   Exit:     br %Merge
+///   Merge:    ...
+///
+/// which makes getLoopGuardBranch() treat \p L as unguarded even though
+/// SimplifyCFG would fold %Skip away. This function performs that same
+/// fold: it redirects the guard branch to %Merge and deletes the empty
+/// %Skip block. Loop fusion calls this on every loop before collecting
+/// fusion candidates, so that a guarded loop left in this shape by an
+/// earlier pass is still recognized as guarded and as adjacent to its
+/// neighbor. Returns true if the CFG was changed.
+static bool simplifyLoopGuard(Loop *L, DomTreeUpdater &DTU, LoopInfo &LI,
+                              ScalarEvolution &SE) {
+  if (!L->isLoopSimplifyForm() || !L->isRotatedForm())
+    return false;
+
+  BasicBlock *Preheader = L->getLoopPreheader();
+  BasicBlock *ExitBlock = L->getUniqueExitBlock();
+  if (!ExitBlock)
+    return false;
+
+  BasicBlock *GuardBB = Preheader->getUniquePredecessor();
+  if (!GuardBB)
+    return false;
+
+  auto *GuardBI = dyn_cast<CondBrInst>(GuardBB->getTerminator());
+  if (!GuardBI)
+    return false;
+
+  BasicBlock *SkipBB = GuardBI->getSuccessor(0) == Preheader
+                           ? GuardBI->getSuccessor(1)
+                           : GuardBI->getSuccessor(0);
+  if (SkipBB == Preheader)
+    return false;
+
+  // The skip block must contain nothing but an unconditional branch and must
+  // be reachable only from the guard, so that removing it cannot change any
+  // other path.
+  if (SkipBB->size() != 1 || !isa<UncondBrInst>(SkipBB->getTerminator()) ||
+      SkipBB->hasAddressTaken() || SkipBB->getUniquePredecessor() != GuardBB)
+    return false;
+
+  BasicBlock *MergeBB = SkipBB->getUniqueSuccessor();
+  if (!MergeBB || MergeBB == SkipBB || MergeBB == GuardBB ||
+      LI.isLoopHeader(MergeBB))
+    return false;
+
+  // The loop exit must flow into the same block; otherwise the branch is
+  // not a loop guard.
+  if (&LoopNest::skipEmptyBlockUntil(ExitBlock, MergeBB,
+                                     /*CheckUniquePred=*/true) != MergeBB)
+    return false;
+
+  LLVM_DEBUG(dbgs() << "Removing empty guard skip block " << SkipBB->getName()
+                    << " of loop " << L->getHeader()->getName() << "\n");
+
+  MergeBB->replacePhiUsesWith(SkipBB, GuardBB);
+  GuardBI->replaceSuccessorWith(SkipBB, MergeBB);
+  SkipBB->getTerminator()->eraseFromParent();
+  new UnreachableInst(SkipBB->getContext(), SkipBB);
+
+  DTU.applyUpdates({{DominatorTree::Delete, GuardBB, SkipBB},
+                    {DominatorTree::Delete, SkipBB, MergeBB},
+                    {DominatorTree::Insert, GuardBB, MergeBB}});
+  LI.removeBlock(SkipBB);
+  DTU.deleteBB(SkipBB);
+  DTU.flush();
+
+  return true;
+}
+
 namespace {
 
 /// Collect all loops in function at the same nest level, starting at the
@@ -503,6 +586,11 @@ struct LoopFuser {
                       << "\n");
     bool Changed = false;
 
+    // Canonicalize the CFG around loop guards before looking for candidates,
+    // so that guarded loops are recognized as such and as adjacent.
+    for (Loop *L : LI.getLoopsInPreorder())
+      Changed |= simplifyLoopGuard(L, DTU, LI, SE);
+
     while (!LDT.empty()) {
       LLVM_DEBUG(dbgs() << "Got " << LDT.size() << " loop sets for depth "
                         << LDT.getDepth() << "\n";);
diff --git a/llvm/test/Transforms/LoopFusion/guard_skip_empty_block.ll b/llvm/test/Transforms/LoopFusion/guard_skip_empty_block.ll
new file mode 100644
index 0000000000000..d0eb161b86ca1
--- /dev/null
+++ b/llvm/test/Transforms/LoopFusion/guard_skip_empty_block.ll
@@ -0,0 +1,389 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -S -passes=loop-fusion < %s | FileCheck %s
+
+ at B = common global [1024 x i32] zeroinitializer, align 16
+
+; The skip edge of the first loop guard goes through an empty block (%skip)
+; that passes like JumpThreading can leave behind. Loop fusion must fold the
+; block away so that the guard is recognized and the two loops are fused.
+define void @guard_skip_empty_block(ptr noalias %A, i64 %N) {
+; CHECK-LABEL: define void @guard_skip_empty_block(
+; CHECK-SAME: ptr noalias [[A:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[CMP4:%.*]] = icmp slt i64 0, [[N]]
+; CHECK-NEXT:    [[CMP31:%.*]] = icmp slt i64 0, [[N]]
+; CHECK-NEXT:    br i1 [[CMP4]], label %[[BB3:.*]], label %[[BB12:.*]]
+; CHECK:       [[BB3]]:
+; CHECK-NEXT:    br label %[[BB5:.*]]
+; CHECK:       [[BB5]]:
+; CHECK-NEXT:    [[I_05:%.*]] = phi i64 [ [[INC:%.*]], %[[BB5]] ], [ 0, %[[BB3]] ]
+; CHECK-NEXT:    [[I1_02:%.*]] = phi i64 [ [[INC14:%.*]], %[[BB5]] ], [ 0, %[[BB3]] ]
+; CHECK-NEXT:    [[SUB:%.*]] = sub nsw i64 [[I_05]], 3
+; CHECK-NEXT:    [[ADD:%.*]] = add nsw i64 [[I_05]], 3
+; CHECK-NEXT:    [[MUL:%.*]] = mul nsw i64 [[SUB]], [[ADD]]
+; CHECK-NEXT:    [[REM:%.*]] = srem i64 [[MUL]], [[I_05]]
+; CHECK-NEXT:    [[CONV:%.*]] = trunc i64 [[REM]] to i32
+; CHECK-NEXT:    [[ARRAYIDX:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[I_05]]
+; CHECK-NEXT:    store i32 [[CONV]], ptr [[ARRAYIDX]], align 4
+; CHECK-NEXT:    [[INC]] = add nsw i64 [[I_05]], 1
+; CHECK-NEXT:    [[CMP:%.*]] = icmp slt i64 [[INC]], [[N]]
+; CHECK-NEXT:    [[SUB7:%.*]] = sub nsw i64 [[I1_02]], 3
+; CHECK-NEXT:    [[ADD8:%.*]] = add nsw i64 [[I1_02]], 3
+; CHECK-NEXT:    [[MUL9:%.*]] = mul nsw i64 [[SUB7]], [[ADD8]]
+; CHECK-NEXT:    [[REM10:%.*]] = srem i64 [[MUL9]], [[I1_02]]
+; CHECK-NEXT:    [[CONV11:%.*]] = trunc i64 [[REM10]] to i32
+; CHECK-NEXT:    [[ARRAYIDX12:%.*]] = getelementptr inbounds [1024 x i32], ptr @B, i64 0, i64 [[I1_02]]
+; CHECK-NEXT:    store i32 [[CONV11]], ptr [[ARRAYIDX12]], align 4
+; CHECK-NEXT:    [[INC14]] = add nsw i64 [[I1_02]], 1
+; CHECK-NEXT:    [[CMP3:%.*]] = icmp slt i64 [[INC14]], [[N]]
+; CHECK-NEXT:    br i1 [[CMP3]], label %[[BB5]], label %[[BB15:.*]]
+; CHECK:       [[BB15]]:
+; CHECK-NEXT:    br label %[[BB12]]
+; CHECK:       [[BB12]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  %cmp4 = icmp slt i64 0, %N
+  br i1 %cmp4, label %blk3, label %skip
+
+skip:
+  br label %blk14
+
+blk3:
+  br label %blk5
+
+blk5:
+  %i.05 = phi i64 [ %inc, %blk5 ], [ 0, %blk3 ]
+  %sub = sub nsw i64 %i.05, 3
+  %add = add nsw i64 %i.05, 3
+  %mul = mul nsw i64 %sub, %add
+  %rem = srem i64 %mul, %i.05
+  %conv = trunc i64 %rem to i32
+  %arrayidx = getelementptr inbounds i32, ptr %A, i64 %i.05
+  store i32 %conv, ptr %arrayidx, align 4
+  %inc = add nsw i64 %i.05, 1
+  %cmp = icmp slt i64 %inc, %N
+  br i1 %cmp, label %blk5, label %blk10
+
+blk10:
+  br label %blk14
+
+blk14:
+  %cmp31 = icmp slt i64 0, %N
+  br i1 %cmp31, label %blk8, label %blk12
+
+blk8:
+  br label %blk9
+
+blk9:
+  %i1.02 = phi i64 [ %inc14, %blk9 ], [ 0, %blk8 ]
+  %sub7 = sub nsw i64 %i1.02, 3
+  %add8 = add nsw i64 %i1.02, 3
+  %mul9 = mul nsw i64 %sub7, %add8
+  %rem10 = srem i64 %mul9, %i1.02
+  %conv11 = trunc i64 %rem10 to i32
+  %arrayidx12 = getelementptr inbounds [1024 x i32], ptr @B, i64 0, i64 %i1.02
+  store i32 %conv11, ptr %arrayidx12, align 4
+  %inc14 = add nsw i64 %i1.02, 1
+  %cmp3 = icmp slt i64 %inc14, %N
+  br i1 %cmp3, label %blk9, label %blk15
+
+blk15:
+  br label %blk12
+
+blk12:
+  ret void
+}
+
+; Same shape inside an outer loop: the skip block belongs to the outer loop
+; and must be removed from LoopInfo as well.
+define void @guard_skip_empty_block_nested(ptr noalias %A, i64 %N, i64 %M) {
+; CHECK-LABEL: define void @guard_skip_empty_block_nested(
+; CHECK-SAME: ptr noalias [[A:%.*]], i64 [[N:%.*]], i64 [[M:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[CMPM:%.*]] = icmp slt i64 0, [[M]]
+; CHECK-NEXT:    br i1 [[CMPM]], label %[[OUTER_PH:.*]], label %[[EXIT:.*]]
+; CHECK:       [[OUTER_PH]]:
+; CHECK-NEXT:    br label %[[OUTER:.*]]
+; CHECK:       [[OUTER]]:
+; CHECK-NEXT:    [[J:%.*]] = phi i64 [ 0, %[[OUTER_PH]] ], [ [[J_INC:%.*]], %[[OUTER_LATCH:.*]] ]
+; CHECK-NEXT:    [[CMP4:%.*]] = icmp slt i64 0, [[N]]
+; CHECK-NEXT:    [[CMP31:%.*]] = icmp slt i64 0, [[N]]
+; CHECK-NEXT:    br i1 [[CMP4]], label %[[BB3:.*]], label %[[OUTER_LATCH]]
+; CHECK:       [[BB3]]:
+; CHECK-NEXT:    br label %[[BB5:.*]]
+; CHECK:       [[BB5]]:
+; CHECK-NEXT:    [[I_05:%.*]] = phi i64 [ [[INC:%.*]], %[[BB5]] ], [ 0, %[[BB3]] ]
+; CHECK-NEXT:    [[I1_02:%.*]] = phi i64 [ [[INC14:%.*]], %[[BB5]] ], [ 0, %[[BB3]] ]
+; CHECK-NEXT:    [[SUB:%.*]] = sub nsw i64 [[I_05]], 3
+; CHECK-NEXT:    [[ADD:%.*]] = add nsw i64 [[I_05]], 3
+; CHECK-NEXT:    [[MUL:%.*]] = mul nsw i64 [[SUB]], [[ADD]]
+; CHECK-NEXT:    [[REM:%.*]] = srem i64 [[MUL]], [[I_05]]
+; CHECK-NEXT:    [[CONV:%.*]] = trunc i64 [[REM]] to i32
+; CHECK-NEXT:    [[IDX:%.*]] = add nsw i64 [[I_05]], [[J]]
+; CHECK-NEXT:    [[ARRAYIDX:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IDX]]
+; CHECK-NEXT:    store i32 [[CONV]], ptr [[ARRAYIDX]], align 4
+; CHECK-NEXT:    [[INC]] = add nsw i64 [[I_05]], 1
+; CHECK-NEXT:    [[CMP:%.*]] = icmp slt i64 [[INC]], [[N]]
+; CHECK-NEXT:    [[SUB7:%.*]] = sub nsw i64 [[I1_02]], 3
+; CHECK-NEXT:    [[ADD8:%.*]] = add nsw i64 [[I1_02]], 3
+; CHECK-NEXT:    [[MUL9:%.*]] = mul nsw i64 [[SUB7]], [[ADD8]]
+; CHECK-NEXT:    [[REM10:%.*]] = srem i64 [[MUL9]], [[I1_02]]
+; CHECK-NEXT:    [[CONV11:%.*]] = trunc i64 [[REM10]] to i32
+; CHECK-NEXT:    [[ARRAYIDX12:%.*]] = getelementptr inbounds [1024 x i32], ptr @B, i64 0, i64 [[I1_02]]
+; CHECK-NEXT:    store i32 [[CONV11]], ptr [[ARRAYIDX12]], align 4
+; CHECK-NEXT:    [[INC14]] = add nsw i64 [[I1_02]], 1
+; CHECK-NEXT:    [[CMP3:%.*]] = icmp slt i64 [[INC14]], [[N]]
+; CHECK-NEXT:    br i1 [[CMP3]], label %[[BB5]], label %[[BB15:.*]]
+; CHECK:       [[BB15]]:
+; CHECK-NEXT:    br label %[[OUTER_LATCH]]
+; CHECK:       [[OUTER_LATCH]]:
+; CHECK-NEXT:    [[J_INC]] = add nsw i64 [[J]], 1
+; CHECK-NEXT:    [[CMPJ:%.*]] = icmp slt i64 [[J_INC]], [[M]]
+; CHECK-NEXT:    br i1 [[CMPJ]], label %[[OUTER]], label %[[EXIT_LOOPEXIT:.*]]
+; CHECK:       [[EXIT_LOOPEXIT]]:
+; CHECK-NEXT:    br label %[[EXIT]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  %cmpM = icmp slt i64 0, %M
+  br i1 %cmpM, label %outer.ph, label %exit
+
+outer.ph:
+  br label %outer
+
+outer:
+  %j = phi i64 [ 0, %outer.ph ], [ %j.inc, %outer.latch ]
+  %cmp4 = icmp slt i64 0, %N
+  br i1 %cmp4, label %blk3, label %skip
+
+skip:
+  br label %blk14
+
+blk3:
+  br label %blk5
+
+blk5:
+  %i.05 = phi i64 [ %inc, %blk5 ], [ 0, %blk3 ]
+  %sub = sub nsw i64 %i.05, 3
+  %add = add nsw i64 %i.05, 3
+  %mul = mul nsw i64 %sub, %add
+  %rem = srem i64 %mul, %i.05
+  %conv = trunc i64 %rem to i32
+  %idx = add nsw i64 %i.05, %j
+  %arrayidx = getelementptr inbounds i32, ptr %A, i64 %idx
+  store i32 %conv, ptr %arrayidx, align 4
+  %inc = add nsw i64 %i.05, 1
+  %cmp = icmp slt i64 %inc, %N
+  br i1 %cmp, label %blk5, label %blk10
+
+blk10:
+  br label %blk14
+
+blk14:
+  %cmp31 = icmp slt i64 0, %N
+  br i1 %cmp31, label %blk8, label %outer.latch
+
+blk8:
+  br label %blk9
+
+blk9:
+  %i1.02 = phi i64 [ %inc14, %blk9 ], [ 0, %blk8 ]
+  %sub7 = sub nsw i64 %i1.02, 3
+  %add8 = add nsw i64 %i1.02, 3
+  %mul9 = mul nsw i64 %sub7, %add8
+  %rem10 = srem i64 %mul9, %i1.02
+  %conv11 = trunc i64 %rem10 to i32
+  %arrayidx12 = getelementptr inbounds [1024 x i32], ptr @B, i64 0, i64 %i1.02
+  store i32 %conv11, ptr %arrayidx12, align 4
+  %inc14 = add nsw i64 %i1.02, 1
+  %cmp3 = icmp slt i64 %inc14, %N
+  br i1 %cmp3, label %blk9, label %blk15
+
+blk15:
+  br label %outer.latch
+
+outer.latch:
+  %j.inc = add nsw i64 %j, 1
+  %cmpj = icmp slt i64 %j.inc, %M
+  br i1 %cmpj, label %outer, label %exit.loopexit
+
+exit.loopexit:
+  br label %exit
+
+exit:
+  ret void
+}
+
+; The loops have different trip counts and are not fused, but the empty skip
+; block is still folded and the PHI in the merge block is updated.
+define i32 @guard_skip_empty_block_phi(ptr noalias %A, i64 %N, i64 %M) {
+; CHECK-LABEL: define i32 @guard_skip_empty_block_phi(
+; CHECK-SAME: ptr noalias [[A:%.*]], i64 [[N:%.*]], i64 [[M:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[CMP4:%.*]] = icmp slt i64 0, [[N]]
+; CHECK-NEXT:    br i1 [[CMP4]], label %[[BB3:.*]], label %[[BB14:.*]]
+; CHECK:       [[BB3]]:
+; CHECK-NEXT:    br label %[[BB5:.*]]
+; CHECK:       [[BB5]]:
+; CHECK-NEXT:    [[I_05:%.*]] = phi i64 [ [[INC:%.*]], %[[BB5]] ], [ 0, %[[BB3]] ]
+; CHECK-NEXT:    [[CONV:%.*]] = trunc i64 [[I_05]] to i32
+; CHECK-NEXT:    [[ARRAYIDX:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[I_05]]
+; CHECK-NEXT:    store i32 [[CONV]], ptr [[ARRAYIDX]], align 4
+; CHECK-NEXT:    [[INC]] = add nsw i64 [[I_05]], 1
+; CHECK-NEXT:    [[CMP:%.*]] = icmp slt i64 [[INC]], [[N]]
+; CHECK-NEXT:    br i1 [[CMP]], label %[[BB5]], label %[[BB10:.*]]
+; CHECK:       [[BB10]]:
+; CHECK-NEXT:    br label %[[BB14]]
+; CHECK:       [[BB14]]:
+; CHECK-NEXT:    [[P:%.*]] = phi i32 [ 0, %[[ENTRY]] ], [ 1, %[[BB10]] ]
+; CHECK-NEXT:    [[CMP31:%.*]] = icmp slt i64 0, [[M]]
+; CHECK-NEXT:    br i1 [[CMP31]], label %[[BB8:.*]], label %[[BB12:.*]]
+; CHECK:       [[BB8]]:
+; CHECK-NEXT:    br label %[[BB9:.*]]
+; CHECK:       [[BB9]]:
+; CHECK-NEXT:    [[I1_02:%.*]] = phi i64 [ [[INC14:%.*]], %[[BB9]] ], [ 0, %[[BB8]] ]
+; CHECK-NEXT:    [[CONV11:%.*]] = trunc i64 [[I1_02]] to i32
+; CHECK-NEXT:    [[ARRAYIDX12:%.*]] = getelementptr inbounds [1024 x i32], ptr @B, i64 0, i64 [[I1_02]]
+; CHECK-NEXT:    store i32 [[CONV11]], ptr [[ARRAYIDX12]], align 4
+; CHECK-NEXT:    [[INC14]] = add nsw i64 [[I1_02]], 1
+; CHECK-NEXT:    [[CMP3:%.*]] = icmp slt i64 [[INC14]], [[M]]
+; CHECK-NEXT:    br i1 [[CMP3]], label %[[BB9]], label %[[BB15:.*]]
+; CHECK:       [[BB15]]:
+; CHECK-NEXT:    br label %[[BB12]]
+; CHECK:       [[BB12]]:
+; CHECK-NEXT:    ret i32 [[P]]
+;
+entry:
+  %cmp4 = icmp slt i64 0, %N
+  br i1 %cmp4, label %blk3, label %skip
+
+skip:
+  br label %blk14
+
+blk3:
+  br label %blk5
+
+blk5:
+  %i.05 = phi i64 [ %inc, %blk5 ], [ 0, %blk3 ]
+  %conv = trunc i64 %i.05 to i32
+  %arrayidx = getelementptr inbounds i32, ptr %A, i64 %i.05
+  store i32 %conv, ptr %arrayidx, align 4
+  %inc = add nsw i64 %i.05, 1
+  %cmp = icmp slt i64 %inc, %N
+  br i1 %cmp, label %blk5, label %blk10
+
+blk10:
+  br label %blk14
+
+blk14:
+  %p = phi i32 [ 0, %skip ], [ 1, %blk10 ]
+  %cmp31 = icmp slt i64 0, %M
+  br i1 %cmp31, label %blk8, label %blk12
+
+blk8:
+  br label %blk9
+
+blk9:
+  %i1.02 = phi i64 [ %inc14, %blk9 ], [ 0, %blk8 ]
+  %conv11 = trunc i64 %i1.02 to i32
+  %arrayidx12 = getelementptr inbounds [1024 x i32], ptr @B, i64 0, i64 %i1.02
+  store i32 %conv11, ptr %arrayidx12, align 4
+  %inc14 = add nsw i64 %i1.02, 1
+  %cmp3 = icmp slt i64 %inc14, %M
+  br i1 %cmp3, label %blk9, label %blk15
+
+blk15:
+  br label %blk12
+
+blk12:
+  ret i32 %p
+}
+
+; The skip block is not empty: it must be left alone, so the first loop stays
+; unguarded and the loops are not fused.
+define void @guard_skip_block_not_empty(ptr noalias %A, i64 %N) {
+; CHECK-LABEL: define void @guard_skip_block_not_empty(
+; CHECK-SAME: ptr noalias [[A:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[CMP4:%.*]] = icmp slt i64 0, [[N]]
+; CHECK-NEXT:    br i1 [[CMP4]], label %[[BB3:.*]], label %[[SKIP:.*]]
+; CHECK:       [[SKIP]]:
+; CHECK-NEXT:    store i32 0, ptr [[A]], align 4
+; CHECK-NEXT:    br label %[[BB14:.*]]
+; CHECK:       [[BB3]]:
+; CHECK-NEXT:    br label %[[BB5:.*]]
+; CHECK:       [[BB5]]:
+; CHECK-NEXT:    [[I_05:%.*]] = phi i64 [ [[INC:%.*]], %[[BB5]] ], [ 0, %[[BB3]] ]
+; CHECK-NEXT:    [[CONV:%.*]] = trunc i64 [[I_05]] to i32
+; CHECK-NEXT:    [[ARRAYIDX:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[I_05]]
+; CHECK-NEXT:    store i32 [[CONV]], ptr [[ARRAYIDX]], align 4
+; CHECK-NEXT:    [[INC]] = add nsw i64 [[I_05]], 1
+; CHECK-NEXT:    [[CMP:%.*]] = icmp slt i64 [[INC]], [[N]]
+; CHECK-NEXT:    br i1 [[CMP]], label %[[BB5]], label %[[BB10:.*]]
+; CHECK:       [[BB10]]:
+; CHECK-NEXT:    br label %[[BB14]]
+; CHECK:       [[BB14]]:
+; CHECK-NEXT:    [[CMP31:%.*]] = icmp slt i64 0, [[N]]
+; CHECK-NEXT:    br i1 [[CMP31]], label %[[BB8:.*]], label %[[BB12:.*]]
+; CHECK:       [[BB8]]:
+; CHECK-NEXT:    br label %[[BB9:.*]]
+; CHECK:       [[BB9]]:
+; CHECK-NEXT:    [[I1_02:%.*]] = phi i64 [ [[INC14:%.*]], %[[BB9]] ], [ 0, %[[BB8]] ]
+; CHECK-NEXT:    [[CONV11:%.*]] = trunc i64 [[I1_02]] to i32
+; CHECK-NEXT:    [[ARRAYIDX12:%.*]] = getelementptr inbounds [1024 x i32], ptr @B, i64 0, i64 [[I1_02]]
+; CHECK-NEXT:    store i32 [[CONV11]], ptr [[ARRAYIDX12]], align 4
+; CHECK-NEXT:    [[INC14]] = add nsw i64 [[I1_02]], 1
+; CHECK-NEXT:    [[CMP3:%.*]] = icmp slt i64 [[INC14]], [[N]]
+; CHECK-NEXT:    br i1 [[CMP3]], label %[[BB9]], label %[[BB15:.*]]
+; CHECK:       [[BB15]]:
+; CHECK-NEXT:    br label %[[BB12]]
+; CHECK:       [[BB12]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  %cmp4 = icmp slt i64 0, %N
+  br i1 %cmp4, label %blk3, label %skip
+
+skip:
+  store i32 0, ptr %A, align 4
+  br label %blk14
+
+blk3:
+  br label %blk5
+
+blk5:
+  %i.05 = phi i64 [ %inc, %blk5 ], [ 0, %blk3 ]
+  %conv = trunc i64 %i.05 to i32
+  %arrayidx = getelementptr inbounds i32, ptr %A, i64 %i.05
+  store i32 %conv, ptr %arrayidx, align 4
+  %inc = add nsw i64 %i.05, 1
+  %cmp = icmp slt i64 %inc, %N
+  br i1 %cmp, label %blk5, label %blk10
+
+blk10:
+  br label %blk14
+
+blk14:
+  %cmp31 = icmp slt i64 0, %N
+  br i1 %cmp31, label %blk8, label %blk12
+
+blk8:
+  br label %blk9
+
+blk9:
+  %i1.02 = phi i64 [ %inc14, %blk9 ], [ 0, %blk8 ]
+  %conv11 = trunc i64 %i1.02 to i32
+  %arrayidx12 = getelementptr inbounds [1024 x i32], ptr @B, i64 0, i64 %i1.02
+  store i32 %conv11, ptr %arrayidx12, align 4
+  %inc14 = add nsw i64 %i1.02, 1
+  %cmp3 = icmp slt i64 %inc14, %N
+  br i1 %cmp3, label %blk9, label %blk15
+
+blk15:
+  br label %blk12
+
+blk12:
+  ret void
+}



More information about the llvm-commits mailing list