[llvm] [GVN] Allow loop-load PRE when the blocker is in an inner loop (PR #227983)
via llvm-commits
llvm-commits at lists.llvm.org
Thu Oct 1 00:22:18 PDT 2026
llvmorg-github-actions[bot] wrote:
<!--LLVM PR SUMMARY COMMENT-->
@llvm/pr-subscribers-llvm-analysis
Author: anupkum-sgs
<details>
<summary>Changes</summary>
performLoopLoadPRE does not perform PRE when the blocker is inside an
inner loop. This patch allows the inner-loop blocker, and the outermost
exit block of the inner loop is selected for reload insertion, so that
the reload runs once each time the inner loop is left.
canBeFreed used by GVN is very conservative. Even if there is no free
between LoadPtr's load and the point where the reload would be inserted
(inside the outermost exit block), it can still return true. If the
address can be freed, the reload is still legal when willNotFreeBetween
finds no deallocation between the header load and that reload.
The non-linear walk in willNotFreeBetween landed in #<!-- -->223580. This patch
uses it along with canBeFreed and raises MaxInstrsToCheckForFree from
32 to 200, so that nested and larger loops can be covered and more
opportunities are found.
RFC: https://discourse.llvm.org/t/rfc-extending-gvns-loop-load-pre/91740
---
Patch is 37.64 KiB, truncated to 20.00 KiB below, full version: https://github.com/llvm/llvm-project/pull/227983.diff
6 Files Affected:
- (modified) llvm/lib/Analysis/ValueTracking.cpp (+1-1)
- (modified) llvm/lib/Transforms/Scalar/GVN.cpp (+30-4)
- (modified) llvm/test/Transforms/GVN/PRE/pre-aliasning-path.ll (+14-6)
- (modified) llvm/test/Transforms/GVN/PRE/pre-loop-load.ll (+343-13)
- (modified) llvm/test/Transforms/LoopUnroll/unroll-cleanup.ll (+15-10)
- (modified) llvm/test/Transforms/PhaseOrdering/X86/ptrtoaddr-ptrtoint.ll (+15-61)
``````````diff
diff --git a/llvm/lib/Analysis/ValueTracking.cpp b/llvm/lib/Analysis/ValueTracking.cpp
index f56375e4dbe73bf..71e283c086246f8 100644
--- a/llvm/lib/Analysis/ValueTracking.cpp
+++ b/llvm/lib/Analysis/ValueTracking.cpp
@@ -93,7 +93,7 @@ static cl::opt<unsigned> DomConditionsMaxUses("dom-conditions-max-uses",
/// Maximum number of instructions to check between assume and context
/// instruction.
-static constexpr unsigned MaxInstrsToCheckForFree = 32;
+static constexpr unsigned MaxInstrsToCheckForFree = 200;
template <typename InstTy>
static bool matchTwoInputRecurrence(const PHINode *PN, InstTy *&Inst,
diff --git a/llvm/lib/Transforms/Scalar/GVN.cpp b/llvm/lib/Transforms/Scalar/GVN.cpp
index b8a44882af86712..18ac54555f24e74 100644
--- a/llvm/lib/Transforms/Scalar/GVN.cpp
+++ b/llvm/lib/Transforms/Scalar/GVN.cpp
@@ -2008,9 +2008,29 @@ bool GVNPass::performLoopLoadPRE(LoadInst *Load,
if (LoopBlock)
return false;
- // Do not sink into inner loops. This may be non-profitable.
- if (L != LI->getLoopFor(Blocker))
- return false;
+ // A reload at the end of a nested blocker runs once per inner iteration,
+ // which is more often than the header. Put it on the exit of the inner
+ // loop that is directly inside L instead. That block is on every path
+ // from the blocker back to L's header, and it runs once each time that inner
+ // loop is left.
+ if (const Loop *Inner = LI->getLoopFor(Blocker); Inner != L) {
+ // The blocker is inside Inner, which may itself be nested further down.
+ // Walk parents until Inner is the loop directly inside L that contains
+ // it.
+ while (Inner->getParentLoop() != L)
+ Inner = Inner->getParentLoop();
+
+ BasicBlock *Exit = Inner->getExitBlock();
+
+ // The exit has to be in L. If it lands in another loop, such as a sibling
+ // of Inner, the reload runs once per iteration of that loop, which can
+ // be more often than L's header. getExitBlock() is null when Inner has
+ // more than one exit, and PRE is aborted in that case.
+ if (!Exit || LI->getLoopFor(Exit) != L)
+ return false;
+
+ Blocker = Exit;
+ }
// Blocks that dominate the latch execute on every single iteration, maybe
// except the last one. So PREing into these blocks doesn't make much sense
@@ -2032,7 +2052,13 @@ bool GVNPass::performLoopLoadPRE(LoadInst *Load,
// Make sure the memory at this pointer cannot be freed, therefore we can
// safely reload from it after clobber.
- if (LoadPtr->canBeFreed())
+ //
+ // The header load has already dereferenced LoadPtr on this iteration, so
+ // only a deallocation between that load and the reload in LoopBlock can make
+ // the same address unsafe to read again. Check every path between these two
+ // points for an instruction that may deallocate the memory.
+ if (LoadPtr->canBeFreed() &&
+ !willNotFreeBetween(Load, LoopBlock->getTerminator(), DT))
return false;
// TODO: Support critical edge splitting if blocker has more than 1 successor.
diff --git a/llvm/test/Transforms/GVN/PRE/pre-aliasning-path.ll b/llvm/test/Transforms/GVN/PRE/pre-aliasning-path.ll
index 7dc884f87e3605a..3d4feb76aaa44a6 100644
--- a/llvm/test/Transforms/GVN/PRE/pre-aliasning-path.ll
+++ b/llvm/test/Transforms/GVN/PRE/pre-aliasning-path.ll
@@ -8,22 +8,26 @@ declare void @side_effect_1(i32 %x) nofree
declare void @no_side_effect() readonly
-; TODO: We can PRE the load into the cold path, removing it from the hot path.
+; We can PRE the load into the cold path, removing it from the hot path: the
+; callee is nofree, so %p cannot be deallocated between the loads.
define i32 @test_01(ptr %p) {
; CHECK-LABEL: @test_01(
; CHECK-NEXT: entry:
+; CHECK-NEXT: [[X_PRE1:%.*]] = load i32, ptr [[P:%.*]], align 4
; CHECK-NEXT: br label [[LOOP:%.*]]
; CHECK: loop:
-; CHECK-NEXT: [[IV:%.*]] = phi i32 [ 0, [[ENTRY:%.*]] ], [ [[IV_NEXT:%.*]], [[BACKEDGE:%.*]] ]
-; CHECK-NEXT: [[X:%.*]] = load i32, ptr [[P:%.*]], align 4
+; CHECK-NEXT: [[X:%.*]] = phi i32 [ [[X_PRE1]], [[ENTRY:%.*]] ], [ [[X2:%.*]], [[BACKEDGE:%.*]] ]
+; CHECK-NEXT: [[IV:%.*]] = phi i32 [ 0, [[ENTRY]] ], [ [[IV_NEXT:%.*]], [[BACKEDGE]] ]
; CHECK-NEXT: [[COND:%.*]] = icmp ult i32 [[X]], 100
; CHECK-NEXT: br i1 [[COND]], label [[HOT_PATH:%.*]], label [[COLD_PATH:%.*]]
; CHECK: hot_path:
; CHECK-NEXT: br label [[BACKEDGE]]
; CHECK: cold_path:
; CHECK-NEXT: call void @side_effect_0() #[[ATTR0:[0-9]+]]
+; CHECK-NEXT: [[X_PRE:%.*]] = load i32, ptr [[P]], align 4
; CHECK-NEXT: br label [[BACKEDGE]]
; CHECK: backedge:
+; CHECK-NEXT: [[X2]] = phi i32 [ [[X_PRE]], [[COLD_PATH]] ], [ [[X]], [[HOT_PATH]] ]
; CHECK-NEXT: [[IV_NEXT]] = add i32 [[IV]], [[X]]
; CHECK-NEXT: [[LOOP_COND:%.*]] = icmp ult i32 [[IV_NEXT]], 1000
; CHECK-NEXT: br i1 [[LOOP_COND]], label [[LOOP]], label [[EXIT:%.*]]
@@ -55,22 +59,26 @@ exit:
ret i32 %x
}
-; TODO: We can PRE the load into the cold path, removing it from the hot path.
+; We can PRE the load into the cold path, removing it from the hot path: the
+; callee is nofree, so %p cannot be deallocated between the loads.
define i32 @test_02(ptr %p) {
; CHECK-LABEL: @test_02(
; CHECK-NEXT: entry:
+; CHECK-NEXT: [[X_PRE1:%.*]] = load i32, ptr [[P:%.*]], align 4
; CHECK-NEXT: br label [[LOOP:%.*]]
; CHECK: loop:
-; CHECK-NEXT: [[IV:%.*]] = phi i32 [ 0, [[ENTRY:%.*]] ], [ [[IV_NEXT:%.*]], [[BACKEDGE:%.*]] ]
-; CHECK-NEXT: [[X:%.*]] = load i32, ptr [[P:%.*]], align 4
+; CHECK-NEXT: [[X:%.*]] = phi i32 [ [[X_PRE1]], [[ENTRY:%.*]] ], [ [[X2:%.*]], [[BACKEDGE:%.*]] ]
+; CHECK-NEXT: [[IV:%.*]] = phi i32 [ 0, [[ENTRY]] ], [ [[IV_NEXT:%.*]], [[BACKEDGE]] ]
; CHECK-NEXT: [[COND:%.*]] = icmp ult i32 [[X]], 100
; CHECK-NEXT: br i1 [[COND]], label [[HOT_PATH:%.*]], label [[COLD_PATH:%.*]]
; CHECK: hot_path:
; CHECK-NEXT: br label [[BACKEDGE]]
; CHECK: cold_path:
; CHECK-NEXT: call void @side_effect_1(i32 [[X]]) #[[ATTR0]]
+; CHECK-NEXT: [[X_PRE:%.*]] = load i32, ptr [[P]], align 4
; CHECK-NEXT: br label [[BACKEDGE]]
; CHECK: backedge:
+; CHECK-NEXT: [[X2]] = phi i32 [ [[X_PRE]], [[COLD_PATH]] ], [ [[X]], [[HOT_PATH]] ]
; CHECK-NEXT: [[IV_NEXT]] = add i32 [[IV]], [[X]]
; CHECK-NEXT: [[LOOP_COND:%.*]] = icmp ult i32 [[IV_NEXT]], 1000
; CHECK-NEXT: br i1 [[LOOP_COND]], label [[LOOP]], label [[EXIT:%.*]]
diff --git a/llvm/test/Transforms/GVN/PRE/pre-loop-load.ll b/llvm/test/Transforms/GVN/PRE/pre-loop-load.ll
index 8f1b73ad8383d92..ce22a95bbfc20c9 100644
--- a/llvm/test/Transforms/GVN/PRE/pre-loop-load.ll
+++ b/llvm/test/Transforms/GVN/PRE/pre-loop-load.ll
@@ -107,22 +107,26 @@ exit:
}
-; TODO: We can PRE the load away from the hot path.
+; We can PRE the load away from the hot path: @side_effect is nofree, so the
+; object under %p cannot be deallocated between the loads.
define i32 @test_load_on_cold_path(ptr %p) {
; CHECK-LABEL: @test_load_on_cold_path(
; CHECK-NEXT: entry:
+; CHECK-NEXT: [[X_PRE1:%.*]] = load i32, ptr [[P:%.*]], align 4
; CHECK-NEXT: br label [[LOOP:%.*]]
; CHECK: loop:
-; CHECK-NEXT: [[IV:%.*]] = phi i32 [ 0, [[ENTRY:%.*]] ], [ [[IV_NEXT:%.*]], [[BACKEDGE:%.*]] ]
-; CHECK-NEXT: [[X:%.*]] = load i32, ptr [[P:%.*]], align 4
+; CHECK-NEXT: [[X:%.*]] = phi i32 [ [[X_PRE1]], [[ENTRY:%.*]] ], [ [[X2:%.*]], [[BACKEDGE:%.*]] ]
+; CHECK-NEXT: [[IV:%.*]] = phi i32 [ 0, [[ENTRY]] ], [ [[IV_NEXT:%.*]], [[BACKEDGE]] ]
; CHECK-NEXT: [[COND:%.*]] = icmp ne i32 [[X]], 0
; CHECK-NEXT: br i1 [[COND]], label [[HOT_PATH:%.*]], label [[COLD_PATH:%.*]]
; CHECK: hot_path:
; CHECK-NEXT: br label [[BACKEDGE]]
; CHECK: cold_path:
; CHECK-NEXT: call void @side_effect() #[[ATTR0:[0-9]+]]
+; CHECK-NEXT: [[X_PRE:%.*]] = load i32, ptr [[P]], align 4
; CHECK-NEXT: br label [[BACKEDGE]]
; CHECK: backedge:
+; CHECK-NEXT: [[X2]] = phi i32 [ [[X_PRE]], [[COLD_PATH]] ], [ [[X]], [[HOT_PATH]] ]
; CHECK-NEXT: [[IV_NEXT]] = add i32 [[IV]], [[X]]
; CHECK-NEXT: [[LOOP_COND:%.*]] = icmp ult i32 [[IV_NEXT]], 1000
; CHECK-NEXT: br i1 [[LOOP_COND]], label [[LOOP]], label [[EXIT:%.*]]
@@ -355,22 +359,26 @@ exit:
ret i32 %x
}
-; TODO: We can PRE via splitting of the critical edge in the cold path.
+; We PRE into the cold path. TODO: split the critical edge so that the reload
+; does not also run on the path that leaves the loop.
define i32 @test_load_on_exiting_cold_path_01(ptr %p) {
; CHECK-LABEL: @test_load_on_exiting_cold_path_01(
; CHECK-NEXT: entry:
+; CHECK-NEXT: [[X_PRE2:%.*]] = load i32, ptr [[P:%.*]], align 4
; CHECK-NEXT: br label [[LOOP:%.*]]
; CHECK: loop:
-; CHECK-NEXT: [[IV:%.*]] = phi i32 [ 0, [[ENTRY:%.*]] ], [ [[IV_NEXT:%.*]], [[BACKEDGE:%.*]] ]
-; CHECK-NEXT: [[X:%.*]] = load i32, ptr [[P:%.*]], align 4
+; CHECK-NEXT: [[X:%.*]] = phi i32 [ [[X_PRE2]], [[ENTRY:%.*]] ], [ [[X3:%.*]], [[BACKEDGE:%.*]] ]
+; CHECK-NEXT: [[IV:%.*]] = phi i32 [ 0, [[ENTRY]] ], [ [[IV_NEXT:%.*]], [[BACKEDGE]] ]
; CHECK-NEXT: [[COND:%.*]] = icmp ne i32 [[X]], 0
; CHECK-NEXT: br i1 [[COND]], label [[HOT_PATH:%.*]], label [[COLD_PATH:%.*]]
; CHECK: hot_path:
; CHECK-NEXT: br label [[BACKEDGE]]
; CHECK: cold_path:
; CHECK-NEXT: [[SIDE_COND:%.*]] = call i1 @side_effect_cond() #[[ATTR0]]
+; CHECK-NEXT: [[X_PRE:%.*]] = load i32, ptr [[P]], align 4
; CHECK-NEXT: br i1 [[SIDE_COND]], label [[BACKEDGE]], label [[COLD_EXIT:%.*]]
; CHECK: backedge:
+; CHECK-NEXT: [[X3]] = phi i32 [ [[X_PRE]], [[COLD_PATH]] ], [ [[X]], [[HOT_PATH]] ]
; CHECK-NEXT: [[IV_NEXT]] = add i32 [[IV]], [[X]]
; CHECK-NEXT: [[LOOP_COND:%.*]] = icmp ult i32 [[IV_NEXT]], 1000
; CHECK-NEXT: br i1 [[LOOP_COND]], label [[LOOP]], label [[EXIT:%.*]]
@@ -570,14 +578,17 @@ exit:
ret i32 %x
}
-; TODO: We can PRE via splitting of the critical edge in the cold path. Make sure we only insert 1 load.
+; We PRE into the last cold block, inserting exactly 1 load. TODO: split the
+; critical edge so that the reload does not also run on the path that leaves the
+; loop.
define i32 @test_load_on_multi_exiting_cold_path(ptr %p) {
; CHECK-LABEL: @test_load_on_multi_exiting_cold_path(
; CHECK-NEXT: entry:
+; CHECK-NEXT: [[X_PRE2:%.*]] = load i32, ptr [[P:%.*]], align 4
; CHECK-NEXT: br label [[LOOP:%.*]]
; CHECK: loop:
-; CHECK-NEXT: [[IV:%.*]] = phi i32 [ 0, [[ENTRY:%.*]] ], [ [[IV_NEXT:%.*]], [[BACKEDGE:%.*]] ]
-; CHECK-NEXT: [[X:%.*]] = load i32, ptr [[P:%.*]], align 4
+; CHECK-NEXT: [[X:%.*]] = phi i32 [ [[X_PRE2]], [[ENTRY:%.*]] ], [ [[X3:%.*]], [[BACKEDGE:%.*]] ]
+; CHECK-NEXT: [[IV:%.*]] = phi i32 [ 0, [[ENTRY]] ], [ [[IV_NEXT:%.*]], [[BACKEDGE]] ]
; CHECK-NEXT: [[COND:%.*]] = icmp ne i32 [[X]], 0
; CHECK-NEXT: br i1 [[COND]], label [[HOT_PATH:%.*]], label [[COLD_PATH_1:%.*]]
; CHECK: hot_path:
@@ -590,8 +601,10 @@ define i32 @test_load_on_multi_exiting_cold_path(ptr %p) {
; CHECK-NEXT: br i1 [[SIDE_COND_2]], label [[COLD_PATH_3:%.*]], label [[COLD_EXIT]]
; CHECK: cold_path.3:
; CHECK-NEXT: [[SIDE_COND_3:%.*]] = call i1 @side_effect_cond() #[[ATTR0]]
+; CHECK-NEXT: [[X_PRE:%.*]] = load i32, ptr [[P]], align 4
; CHECK-NEXT: br i1 [[SIDE_COND_3]], label [[BACKEDGE]], label [[COLD_EXIT]]
; CHECK: backedge:
+; CHECK-NEXT: [[X3]] = phi i32 [ [[X_PRE]], [[COLD_PATH_3]] ], [ [[X]], [[HOT_PATH]] ]
; CHECK-NEXT: [[IV_NEXT]] = add i32 [[IV]], [[X]]
; CHECK-NEXT: [[LOOP_COND:%.*]] = icmp ult i32 [[IV_NEXT]], 1000
; CHECK-NEXT: br i1 [[LOOP_COND]], label [[LOOP]], label [[EXIT:%.*]]
@@ -636,7 +649,8 @@ cold_exit:
ret i32 -1
}
-; TODO: PRE via splittinga backedge in the cold loop. Make sure we don't insert a load into an inner loop.
+; Do not PRE. The inner loop's unique exit is the outer latch %backedge, so the
+; remapped reload site fails the latch-dominance profitability check.
define i32 @test_inner_loop(ptr %p, i1 %arg) {
; CHECK-LABEL: @test_inner_loop(
; CHECK-NEXT: entry:
@@ -773,14 +787,16 @@ exit:
ret i32 %x
}
-; TODO: We can PRE via split of critical edge.
+; We PRE into the call block. TODO: split the critical edge so that the reload
+; does not also run on the path that leaves the loop.
define i32 @test_side_exit_after_merge(ptr %p) {
; CHECK-LABEL: @test_side_exit_after_merge(
; CHECK-NEXT: entry:
+; CHECK-NEXT: [[X_PRE2:%.*]] = load i32, ptr [[P:%.*]], align 4
; CHECK-NEXT: br label [[LOOP:%.*]]
; CHECK: loop:
-; CHECK-NEXT: [[IV:%.*]] = phi i32 [ 0, [[ENTRY:%.*]] ], [ [[IV_NEXT:%.*]], [[BACKEDGE:%.*]] ]
-; CHECK-NEXT: [[X:%.*]] = load i32, ptr [[P:%.*]], align 4
+; CHECK-NEXT: [[X:%.*]] = phi i32 [ [[X_PRE2]], [[ENTRY:%.*]] ], [ [[X3:%.*]], [[BACKEDGE:%.*]] ]
+; CHECK-NEXT: [[IV:%.*]] = phi i32 [ 0, [[ENTRY]] ], [ [[IV_NEXT:%.*]], [[BACKEDGE]] ]
; CHECK-NEXT: [[COND:%.*]] = icmp ne i32 [[X]], 0
; CHECK-NEXT: br i1 [[COND]], label [[HOT_PATH:%.*]], label [[COLD_PATH:%.*]]
; CHECK: hot_path:
@@ -790,11 +806,14 @@ define i32 @test_side_exit_after_merge(ptr %p) {
; CHECK-NEXT: br i1 [[COND_1]], label [[DO_CALL:%.*]], label [[SIDE_EXITING:%.*]]
; CHECK: do_call:
; CHECK-NEXT: [[SIDE_COND:%.*]] = call i1 @side_effect_cond() #[[ATTR0]]
+; CHECK-NEXT: [[X_PRE:%.*]] = load i32, ptr [[P]], align 4
; CHECK-NEXT: br label [[SIDE_EXITING]]
; CHECK: side_exiting:
+; CHECK-NEXT: [[X4:%.*]] = phi i32 [ [[X_PRE]], [[DO_CALL]] ], [ 0, [[COLD_PATH]] ]
; CHECK-NEXT: [[SIDE_COND_PHI:%.*]] = phi i1 [ [[SIDE_COND]], [[DO_CALL]] ], [ true, [[COLD_PATH]] ]
; CHECK-NEXT: br i1 [[SIDE_COND_PHI]], label [[BACKEDGE]], label [[COLD_EXIT:%.*]]
; CHECK: backedge:
+; CHECK-NEXT: [[X3]] = phi i32 [ [[X4]], [[SIDE_EXITING]] ], [ [[X]], [[HOT_PATH]] ]
; CHECK-NEXT: [[IV_NEXT]] = add i32 [[IV]], [[X]]
; CHECK-NEXT: [[LOOP_COND:%.*]] = icmp ult i32 [[IV_NEXT]], 1000
; CHECK-NEXT: br i1 [[LOOP_COND]], label [[LOOP]], label [[EXIT:%.*]]
@@ -940,3 +959,314 @@ backedge:
exit:
ret i32 %x
}
+
+; The blocker sits inside an inner loop. Reloading there would run once per
+; inner iteration, which is more often than the header we are relieving, so the
+; reload goes to the inner loop's exit block instead.
+define i32 @test_blocker_in_inner_loop(ptr %p, i1 %arg) {
+; CHECK-LABEL: @test_blocker_in_inner_loop(
+; CHECK-NEXT: entry:
+; CHECK-NEXT: [[X_PRE1:%.*]] = load i32, ptr [[P:%.*]], align 4
+; CHECK-NEXT: br label [[LOOP:%.*]]
+; CHECK: loop:
+; CHECK-NEXT: [[X:%.*]] = phi i32 [ [[X_PRE1]], [[ENTRY:%.*]] ], [ [[X2:%.*]], [[BACKEDGE:%.*]] ]
+; CHECK-NEXT: [[IV:%.*]] = phi i32 [ 0, [[ENTRY]] ], [ [[IV_NEXT:%.*]], [[BACKEDGE]] ]
+; CHECK-NEXT: [[COND:%.*]] = icmp ne i32 [[X]], 0
+; CHECK-NEXT: br i1 [[COND]], label [[HOT_PATH:%.*]], label [[COLD_PATH:%.*]]
+; CHECK: hot_path:
+; CHECK-NEXT: br label [[BACKEDGE]]
+; CHECK: cold_path:
+; CHECK-NEXT: br label [[INNER_LOOP:%.*]]
+; CHECK: inner_loop:
+; CHECK-NEXT: call void @side_effect() #[[ATTR0]]
+; CHECK-NEXT: br i1 [[ARG:%.*]], label [[INNER_LOOP]], label [[INNER_EXIT:%.*]]
+; CHECK: inner_exit:
+; CHECK-NEXT: [[X_PRE:%.*]] = load i32, ptr [[P]], align 4
+; CHECK-NEXT: br label [[BACKEDGE]]
+; CHECK: backedge:
+; CHECK-NEXT: [[X2]] = phi i32 [ [[X_PRE]], [[INNER_EXIT]] ], [ [[X]], [[HOT_PATH]] ]
+; CHECK-NEXT: [[IV_NEXT]] = add i32 [[IV]], [[X]]
+; CHECK-NEXT: [[LOOP_COND:%.*]] = icmp ult i32 [[IV_NEXT]], 1000
+; CHECK-NEXT: br i1 [[LOOP_COND]], label [[LOOP]], label [[EXIT:%.*]]
+; CHECK: exit:
+; CHECK-NEXT: ret i32 [[X]]
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i32 [ 0, %entry ], [ %iv.next, %backedge ]
+ %x = load i32, ptr %p
+ %cond = icmp ne i32 %x, 0
+ br i1 %cond, label %hot_path, label %cold_path
+
+hot_path:
+ br label %backedge
+
+cold_path:
+ br label %inner_loop
+
+inner_loop:
+ call void @side_effect() nofree
+ br i1 %arg, label %inner_loop, label %inner_exit
+
+inner_exit:
+ br label %backedge
+
+backedge:
+ %iv.next = add i32 %iv, %x
+ %loop.cond = icmp ult i32 %iv.next, 1000
+ br i1 %loop.cond, label %loop, label %exit
+
+exit:
+ ret i32 %x
+}
+
+; Same, with the blocker two levels deep. The reload goes to the exit block of
+; the loop directly nested in the loop being optimized, not to the exit of the
+; loop immediately containing the blocker.
+define i32 @test_blocker_in_inner_loop_nested(ptr %p, i1 %inner.back, i1 %mid.back) {
+; CHECK-LABEL: @test_blocker_in_inner_loop_nested(
+; CHECK-NEXT: entry:
+; CHECK-NEXT: [[X_PRE1:%.*]] = load i32, ptr [[P:%.*]], align 4
+; CHECK-NEXT: br label [[LOOP:%.*]]
+; CHECK: loop:
+; CHECK-NEXT: [[X:%.*]] = phi i32 [ [[X_PRE1]], [[ENTRY:%.*]] ], [ [[X2:%.*]], [[BACKEDGE:%.*]] ]
+; CHECK-NEXT: [[IV:%.*]] = phi i32 [ 0, [[ENTRY]] ], [ [[IV_NEXT:%.*]], [[BACKEDGE]] ]
+; CHECK-NEXT: [[COND:%.*]] = icmp ne i32 [[X]], 0
+; CHECK-NEXT: br i1 [[COND]], label [[HOT_PATH:%.*]], label [[COLD_PATH:%.*]]
+; CHECK: hot_path:
+; CHECK-NEXT: br label [[BACKEDGE]]
+; CHECK: cold_path:
+; CHECK-NEXT: br label [[MID_LOOP:%.*]]
+; CHECK: mid_loop:
+; CHECK-NEXT: br label [[INNER_LOOP:%.*]]
+; CHECK: inner_loop:
+; CHECK-NEXT: call void @side_effect() #[[ATTR0]]
+; CHECK-NEXT: br i1 [[ARG:%.*]], label [[INNER_LOOP]], label [[INNER_EXIT:%.*]]
+; CHECK: inner_exit:
+; CHECK-NEXT: br i1 [[MID_BACK:%.*]], label [[MID_LOOP]], label [[MID_EXIT:%.*]]
+; CHECK: mid_exit:
+; CHECK-NEXT: [[X_PRE:%.*]] = load i32, ptr [[P]], align 4
+; CHECK-NEXT: br label [[BACKEDGE]]
+; CHECK: backedge:
+; CHECK-NEXT: [[X2]] = phi i32 [ [[X_PRE]], [[MID_EXIT]] ], [ [[X]], [[HOT_PATH]] ]
+; CHECK-NEXT: [[IV_NEXT]] = add i32 [[IV]], [[X]]
+; CHECK-NEXT: [[LOOP_COND:%.*]] = icmp ult i32 [[IV_NEXT]], 1000
+; CHECK-NEXT: br i1 [[LOOP_COND]], label [[LOOP]], label [[EXIT:%.*]]
+; CHECK: exit:
+; CHECK-NEXT: ret i32 [[X]]
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i32 [ 0, %entry ], [ %iv.next, %backedge ]
+ %x = load i32, ptr %p
+ %cond = icmp ne i32 %x, 0
+ br i1 %cond, label %hot_path, label %cold_path
+
+hot_path:
+ br label %backedge
+
+cold_path:
+ br label %mid_loop
+
+mid_loop:
+ br label %inner_loop
+
+inner_loop:
+ call void @side_effect() nofree
+ br i1 %inner.back, label %inner_loop, label %inner_exit
+
+inner_exit:
+ br i1 %mid.back, label %mid_loop, label %mid_exit
+
+mid_exit:
+ br label %backedge
+
+backedge:
+ %iv.next = add i32 %iv, %x
+ %loop.cond = icmp ult i32 %iv.next, 1000
+ br i1 %loop.cond, label %loop, label %exit
+
+exit:
+ ret i32 %x
+}
+
+; Do not PRE when the inner loop exits into a sibling loop: placing the reload
+; in the sibling header could execute it more often than the outer header.
+define i32 @test_blocker_exits_to_sibling_loop_neg(ptr %p, i1 %inner.back, i1 %sibling.back) {
+; CHECK-LABEL: @test_blocker_exits_to_sibling_loop_neg(
+; CHECK-NEXT: entry:
+; CHECK-NEXT: br label [[LOOP:%.*]]
+; CHECK: loop:
+; CHECK-NEXT: [[IV:%.*]] = phi i32 [ 0, [[ENTRY:%.*]] ], [ [[IV_NEXT:%.*]], [[BACKEDGE:%.*]] ]
+; CHECK-NEXT: [[X:%.*]] = load i32, ptr [[P:%.*]], align 4
+; CHECK-NEXT: [[COND:%.*]] = icmp ne i32 [[X]], 0
+; CHECK-NEXT: br i1 [[COND]], label [[HOT_PATH:%.*]], label [[INNER_LOOP:%.*]]
+; CHECK: hot.path:
+; CHECK-NEXT: br label [[BACKEDGE]]
+; CHECK: inner.loop:
+; CHECK-NEXT: call void @side_effect() #[[ATTR0]]
+; CHECK-NEXT: br i1 [[INNER_BACK:%.*]], label [[INNER_LOOP]], label [[SIBLING_LOOP:%.*]]
+; CHECK: sibling.loop:
+; CHECK-NEXT: br i1 [[SIBLING_BACK:%.*]], label [[SIBLING_LOOP]], label [[BACKEDGE]]
+; CHECK: backedge:
+; CHECK-NEXT: [[IV_NEXT]] = add i32 [[IV]], [[X]]
+; CHECK-NEXT: [[LOOP_COND:%.*]] = icmp ult i32 [[IV_NEXT]], 1000
+; CHECK-NEXT: br i1 [[LOOP_COND]], label [[LOOP]], label [[EXIT:%.*]]
+; CHECK: exit:
+; CHECK-NEXT: ret i32 [[X]]
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i32 [ 0, %entry ], [ %iv.next, %backedge ]
+ %x = load i32, ptr %p
+ %cond = icmp ne i32 %x, 0
+ br i1 %cond, label %hot.path, label %inner.loop
+
+hot.path:
+ br label %backedge
+
+inner.loop:
+ call...
[truncated]
``````````
</details>
https://github.com/llvm/llvm-project/pull/227983
More information about the llvm-commits
mailing list