[llvm] [LoopRotate] Fix branch weights for rotated multi-exit loops (PR #202219)
Alok Kumar Sharma via llvm-commits
llvm-commits at lists.llvm.org
Wed Aug 19 11:01:26 PDT 2026
https://github.com/alokkrsharma updated https://github.com/llvm/llvm-project/pull/202219
>From 22a21d2def90415d296974554194c231be774ef8 Mon Sep 17 00:00:00 2001
From: Alok Kumar Sharma <AlokKumar.Sharma at amd.com>
Date: Wed, 19 Aug 2026 23:28:52 +0530
Subject: [PATCH 1/3] [LoopRotate] Preserve header weights when the preheader
guard stays conditional
When loop rotation leaves a conditional preheader guard, the guard and
latch are copies of the original header branch. Stop redistributing the
header weights across the guard and latch; keep the original {exit,
backedge} ratio on both branches instead.
This preserves block frequencies for loops whose guard remains
conditional after rotation, including multi-exit loops with a conditional
guard. Remove the zero-trip-count heuristics that drove the old
redistribution.
When the preheader guard folds away, keep the existing single-exit latch
adjustment unchanged.
---
.../Transforms/Utils/LoopRotationUtils.cpp | 131 ++++-------
.../LoopRotate/multi-exit-branch-weights.ll | 87 ++++++++
.../never-entered-branch-weights.ll | 78 +++++++
.../predecessor-no-prof-branch-weights.ll | 84 ++++++++
.../saturating-backedge-branch-weights.ll | 96 +++++++++
.../scaled-preheader-branch-weights.ll | 60 ++++++
.../LoopRotate/update-branch-weights.ll | 204 ++++++++++++------
7 files changed, 581 insertions(+), 159 deletions(-)
create mode 100644 llvm/test/Transforms/LoopRotate/multi-exit-branch-weights.ll
create mode 100644 llvm/test/Transforms/LoopRotate/never-entered-branch-weights.ll
create mode 100644 llvm/test/Transforms/LoopRotate/predecessor-no-prof-branch-weights.ll
create mode 100644 llvm/test/Transforms/LoopRotate/saturating-backedge-branch-weights.ll
create mode 100644 llvm/test/Transforms/LoopRotate/scaled-preheader-branch-weights.ll
diff --git a/llvm/lib/Transforms/Utils/LoopRotationUtils.cpp b/llvm/lib/Transforms/Utils/LoopRotationUtils.cpp
index c8bc5e4daeff3..7950e223bfb4d 100644
--- a/llvm/lib/Transforms/Utils/LoopRotationUtils.cpp
+++ b/llvm/lib/Transforms/Utils/LoopRotationUtils.cpp
@@ -46,9 +46,6 @@ STATISTIC(NumInstrsHoisted,
STATISTIC(NumInstrsDuplicated,
"Number of instructions cloned into loop preheader");
-// Probability that a rotated loop has zero trip count / is never entered.
-static constexpr uint32_t ZeroTripCountWeights[] = {1, 127};
-
namespace {
/// A simple loop rotation transformation.
class LoopRotate {
@@ -222,6 +219,26 @@ static void updateBranchWeights(CondBrInst &PreHeaderBI, CondBrInst &LoopBI,
if (WeightMD != getBranchWeightMDNode(LoopBI))
return;
+ // A conditional copied guard and latch use the same weights {x, y}:
+ //
+ // | |-------- |
+ // V V | V
+ // Br {x, y} | Br {x, y}
+ // | | | | |
+ // x| y| | becomes: x| y| |------
+ // V V | | V V |
+ // Exit Loop | | Loop |
+ // | | | Br {x, y} |
+ // ----- | | | |
+ // x | x | y| |
+ // V V -----
+ // Exit
+ //
+ // This preserves block frequencies, including multi-exit loops when the
+ // guard stays conditional.
+ if (HasConditionalPreHeader)
+ return;
+
SmallVector<uint32_t, 2> Weights;
extractFromBranchWeightMD32(WeightMD, Weights);
if (Weights.size() != 2)
@@ -232,108 +249,32 @@ static void updateBranchWeights(CondBrInst &PreHeaderBI, CondBrInst &LoopBI,
if (SuccsSwapped)
std::swap(OrigLoopExitWeight, OrigLoopBackedgeWeight);
- // Update branch weights. Consider the following edge-counts:
- //
- // | |-------- |
- // V V | V
- // Br i1 ... | Br i1 ...
- // | | | | |
- // x| y| | becomes: | y0| |-----
- // V V | | V V |
- // Exit Loop | | Loop |
- // | | | Br i1 ... |
- // ----- | | | |
- // x0| x1| y1 | |
- // V V ----
- // Exit
- //
- // The following must hold:
- // - x == x0 + x1 # counts to "exit" must stay the same.
- // - y0 == x - x0 == x1 # how often loop was entered at all.
- // - y1 == y - y0 # How often loop was repeated (after first iter.).
- //
- // We cannot generally deduce how often we had a zero-trip count loop so we
- // have to make a guess for how to distribute x among the new x0 and x1.
-
- uint32_t ExitWeight0; // aka x0
- uint32_t ExitWeight1; // aka x1
- uint32_t EnterWeight; // aka y0
- uint32_t LoopBackWeight; // aka y1
+ // The first iteration is now unconditional. Keep the historical single-exit
+ // latch adjustment. This does not account for side exits on multi-exit
+ // loops when the guard folds away.
+ uint32_t ExitWeight;
+ uint32_t LoopBackWeight;
if (OrigLoopExitWeight > 0 && OrigLoopBackedgeWeight > 0) {
- ExitWeight0 = 0;
- if (HasConditionalPreHeader) {
- // Here we cannot know how many 0-trip count loops we have, so we guess:
- if (OrigLoopBackedgeWeight >= OrigLoopExitWeight) {
- // If the loop count is bigger than the exit count then we set
- // probabilities as if 0-trip count nearly never happens.
- ExitWeight0 = ZeroTripCountWeights[0];
- // Scale up counts if necessary so we can match `ZeroTripCountWeights`
- // for the `ExitWeight0`:`ExitWeight1` (aka `x0`:`x1` ratio`) ratio.
- while (OrigLoopExitWeight < ZeroTripCountWeights[1] + ExitWeight0) {
- // ... but don't overflow.
- uint32_t const HighBit = uint32_t{1} << (sizeof(uint32_t) * 8 - 1);
- if ((OrigLoopBackedgeWeight & HighBit) != 0 ||
- (OrigLoopExitWeight & HighBit) != 0)
- break;
- OrigLoopBackedgeWeight <<= 1;
- OrigLoopExitWeight <<= 1;
- }
- } else {
- // If there's a higher exit-count than backedge-count then we set
- // probabilities as if there are only 0-trip and 1-trip cases.
- ExitWeight0 = OrigLoopExitWeight - OrigLoopBackedgeWeight;
- }
- } else {
- // Theoretically, if the loop body must be executed at least once, the
- // backedge count must be not less than exit count. However the branch
- // weight collected by sampling-based PGO may be not very accurate due to
- // sampling. Therefore this workaround is required here to avoid underflow
- // of unsigned in following update of branch weight.
- if (OrigLoopExitWeight > OrigLoopBackedgeWeight)
- OrigLoopBackedgeWeight = OrigLoopExitWeight;
- }
- assert(OrigLoopExitWeight >= ExitWeight0 && "Bad branch weight");
- ExitWeight1 = OrigLoopExitWeight - ExitWeight0;
- EnterWeight = ExitWeight1;
- assert(OrigLoopBackedgeWeight >= EnterWeight && "Bad branch weight");
- LoopBackWeight = OrigLoopBackedgeWeight - EnterWeight;
+ // Sampling can report fewer backedges than exits. Clamp to avoid underflow.
+ if (OrigLoopExitWeight > OrigLoopBackedgeWeight)
+ OrigLoopBackedgeWeight = OrigLoopExitWeight;
+ ExitWeight = OrigLoopExitWeight;
+ LoopBackWeight = OrigLoopBackedgeWeight - OrigLoopExitWeight;
} else if (OrigLoopExitWeight == 0) {
- if (OrigLoopBackedgeWeight == 0) {
- // degenerate case... keep everything zero...
- ExitWeight0 = 0;
- ExitWeight1 = 0;
- EnterWeight = 0;
- LoopBackWeight = 0;
- } else {
- // Special case "LoopExitWeight == 0" weights which behaves like an
- // endless where we don't want loop-enttry (y0) to be the same as
- // loop-exit (x1).
- ExitWeight0 = 0;
- ExitWeight1 = 0;
- EnterWeight = 1;
- LoopBackWeight = OrigLoopBackedgeWeight;
- }
+ ExitWeight = 0;
+ LoopBackWeight = OrigLoopBackedgeWeight;
} else {
- // loop is never entered.
+ // Loop is never entered.
assert(OrigLoopBackedgeWeight == 0 && "remaining case is backedge zero");
- ExitWeight0 = 1;
- ExitWeight1 = 1;
- EnterWeight = 0;
+ ExitWeight = 1;
LoopBackWeight = 0;
}
const uint32_t LoopBIWeights[] = {
- SuccsSwapped ? LoopBackWeight : ExitWeight1,
- SuccsSwapped ? ExitWeight1 : LoopBackWeight,
+ SuccsSwapped ? LoopBackWeight : ExitWeight,
+ SuccsSwapped ? ExitWeight : LoopBackWeight,
};
setBranchWeights(LoopBI, LoopBIWeights, /*IsExpected=*/false);
- if (HasConditionalPreHeader) {
- const uint32_t PreHeaderBIWeights[] = {
- SuccsSwapped ? EnterWeight : ExitWeight0,
- SuccsSwapped ? ExitWeight0 : EnterWeight,
- };
- setBranchWeights(PreHeaderBI, PreHeaderBIWeights, /*IsExpected=*/false);
- }
}
/// Rotate loop LP. Return true if the loop is rotated.
diff --git a/llvm/test/Transforms/LoopRotate/multi-exit-branch-weights.ll b/llvm/test/Transforms/LoopRotate/multi-exit-branch-weights.ll
new file mode 100644
index 0000000000000..ac47fc94b4bf5
--- /dev/null
+++ b/llvm/test/Transforms/LoopRotate/multi-exit-branch-weights.ll
@@ -0,0 +1,87 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --no-generate-body-for-unused-prefixes --version 6
+; RUN: opt < %s -passes=loop-rotate -S | FileCheck %s
+;
+; The copied guard and latch run the same test. Keep the original weights on
+; both, including loops with another body exit.
+;
+; Entry weights use another scale because each branch stores a local ratio.
+;
+; entry --(c0)--> ph !prof { 42, 2 }
+; ph --------> header
+; header --(hcmp)--> exit1 / body !prof { 200, 9800 }
+; body --(bcmp)--> exit2 / latch !prof { 10, 9790 }
+; latch ---------> header
+
+define void @f(ptr %p, i32 %start, i32 %limit, i32 %cond) {
+; CHECK-LABEL: define void @f(
+; CHECK-SAME: ptr [[P:%.*]], i32 [[START:%.*]], i32 [[LIMIT:%.*]], i32 [[COND:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[C0:%.*]] = icmp ne i32 [[COND]], 0
+; CHECK-NEXT: br i1 [[C0]], label %[[PH:.*]], label %[[RET:.*]], !prof [[PROF0:![0-9]+]]
+; CHECK: [[PH]]:
+; CHECK-NEXT: [[HCMP1:%.*]] = icmp eq i32 [[START]], [[LIMIT]]
+; CHECK-NEXT: br i1 [[HCMP1]], label %[[EXIT1:.*]], label %[[BODY_LR_PH:.*]], !prof [[PROF1:![0-9]+]]
+; CHECK: [[BODY_LR_PH]]:
+; CHECK-NEXT: br label %[[BODY:.*]]
+; CHECK: [[HEADER:.*]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i32 [ [[IV_NEXT:%.*]], %[[BODY]] ]
+; CHECK-NEXT: [[HCMP:%.*]] = icmp eq i32 [[IV]], [[LIMIT]]
+; CHECK-NEXT: br i1 [[HCMP]], label %[[HEADER_EXIT1_CRIT_EDGE:.*]], label %[[BODY]], !prof [[PROF1]]
+; CHECK: [[BODY]]:
+; CHECK-NEXT: [[IV2:%.*]] = phi i32 [ [[START]], %[[BODY_LR_PH]] ], [ [[IV]], %[[HEADER]] ]
+; CHECK-NEXT: [[ADDR:%.*]] = getelementptr i32, ptr [[P]], i32 [[IV2]]
+; CHECK-NEXT: [[V:%.*]] = load i32, ptr [[ADDR]], align 4
+; CHECK-NEXT: [[BCMP:%.*]] = icmp slt i32 [[V]], 0
+; CHECK-NEXT: [[IV_NEXT]] = add i32 [[IV2]], 1
+; CHECK-NEXT: br i1 [[BCMP]], label %[[EXIT2:.*]], label %[[HEADER]], !prof [[PROF2:![0-9]+]]
+; CHECK: [[HEADER_EXIT1_CRIT_EDGE]]:
+; CHECK-NEXT: br label %[[EXIT1]]
+; CHECK: [[EXIT1]]:
+; CHECK-NEXT: ret void
+; CHECK: [[EXIT2]]:
+; CHECK-NEXT: ret void
+; CHECK: [[RET]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ %c0 = icmp ne i32 %cond, 0
+ br i1 %c0, label %ph, label %ret, !prof !0
+
+ph: ; preds = %entry
+ br label %header
+
+header: ; preds = %latch, %ph
+ %iv = phi i32 [ %start, %ph ], [ %iv.next, %latch ]
+ ; Keep the guard conditional.
+ %hcmp = icmp eq i32 %iv, %limit
+ br i1 %hcmp, label %exit1, label %body, !prof !1
+
+body: ; preds = %header
+ %addr = getelementptr i32, ptr %p, i32 %iv
+ %v = load i32, ptr %addr, align 4
+ %bcmp = icmp slt i32 %v, 0
+ br i1 %bcmp, label %exit2, label %latch, !prof !2
+
+latch: ; preds = %body
+ %iv.next = add i32 %iv, 1
+ br label %header
+
+exit1: ; preds = %header
+ ret void
+
+exit2: ; preds = %body
+ ret void
+
+ret: ; preds = %entry
+ ret void
+}
+
+!0 = !{!"branch_weights", i32 42, i32 2}
+!1 = !{!"branch_weights", i32 200, i32 9800}
+!2 = !{!"branch_weights", i32 10, i32 9790}
+
+;.
+; CHECK: [[PROF0]] = !{!"branch_weights", i32 42, i32 2}
+; CHECK: [[PROF1]] = !{!"branch_weights", i32 200, i32 9800}
+; CHECK: [[PROF2]] = !{!"branch_weights", i32 10, i32 9790}
+;.
diff --git a/llvm/test/Transforms/LoopRotate/never-entered-branch-weights.ll b/llvm/test/Transforms/LoopRotate/never-entered-branch-weights.ll
new file mode 100644
index 0000000000000..27676b7408368
--- /dev/null
+++ b/llvm/test/Transforms/LoopRotate/never-entered-branch-weights.ll
@@ -0,0 +1,78 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --no-generate-body-for-unused-prefixes --version 6
+; RUN: opt < %s -passes=loop-rotate -S | FileCheck %s
+;
+; Header weights are {5, 0} (exit taken, backedge never taken). Rotation
+; leaves a conditional preheader guard, so updateBranchWeights() preserves
+; the cloned header weights instead of applying the folded-guard adjustment
+; (which would produce {1, 0} for this profile).
+;
+; Expected weights on the rotated guard and latch branches: {5, 0}.
+
+define void @g(i32 %n, i32 %cond) !prof !14 {
+; CHECK-LABEL: define void @g(
+; CHECK-SAME: i32 [[N:%.*]], i32 [[COND:%.*]]) !prof [[PROF14:![0-9]+]] {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[C0:%.*]] = icmp ne i32 [[COND]], 0
+; CHECK-NEXT: br i1 [[C0]], label %[[PH:.*]], label %[[RET:.*]], !prof [[PROF15:![0-9]+]]
+; CHECK: [[PH]]:
+; CHECK-NEXT: br label %[[HEADER:.*]]
+; CHECK: [[HEADER]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i32 [ 0, %[[PH]] ], [ [[IV_NEXT:%.*]], %[[HEADER]] ]
+; CHECK-NEXT: [[HCMP:%.*]] = icmp sge i32 [[IV]], [[N]]
+; CHECK-NEXT: [[IV_NEXT]] = add i32 [[IV]], 1
+; CHECK-NEXT: br i1 [[HCMP]], label %[[EXIT:.*]], label %[[HEADER]], !prof [[PROF16:![0-9]+]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+; CHECK: [[RET]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ %c0 = icmp ne i32 %cond, 0
+ br i1 %c0, label %ph, label %ret, !prof !15
+
+ph: ; preds = %entry
+ br label %header
+
+header: ; preds = %latch, %ph
+ %iv = phi i32 [ 0, %ph ], [ %iv.next, %latch ]
+ %hcmp = icmp sge i32 %iv, %n
+ br i1 %hcmp, label %exit, label %latch, !prof !16
+
+latch: ; preds = %header
+ %iv.next = add i32 %iv, 1
+ br label %header
+
+exit: ; preds = %header
+ ret void
+
+ret: ; preds = %entry
+ ret void
+}
+
+!llvm.module.flags = !{!0}
+
+!0 = !{i32 1, !"ProfileSummary", !1}
+!1 = !{!2, !3, !4, !5, !6, !7, !8, !9}
+!2 = !{!"ProfileFormat", !"InstrProf"}
+!3 = !{!"TotalCount", i64 6}
+!4 = !{!"MaxCount", i64 5}
+!5 = !{!"MaxInternalCount", i64 5}
+!6 = !{!"MaxFunctionCount", i64 6}
+!7 = !{!"NumCounts", i64 3}
+!8 = !{!"NumFunctions", i64 1}
+!9 = !{!"DetailedSummary", !10}
+!10 = !{!11, !12, !13}
+!11 = !{i32 10000, i64 5, i32 1}
+!12 = !{i32 999000, i64 5, i32 1}
+!13 = !{i32 999999, i64 5, i32 1}
+!14 = !{!"function_entry_count", i64 6}
+!15 = !{!"branch_weights", i32 5, i32 1}
+!16 = !{!"branch_weights", i32 5, i32 0}
+
+
+
+;.
+; CHECK: [[PROF14]] = !{!"function_entry_count", i64 6}
+; CHECK: [[PROF15]] = !{!"branch_weights", i32 5, i32 1}
+; CHECK: [[PROF16]] = !{!"branch_weights", i32 5, i32 0}
+;.
diff --git a/llvm/test/Transforms/LoopRotate/predecessor-no-prof-branch-weights.ll b/llvm/test/Transforms/LoopRotate/predecessor-no-prof-branch-weights.ll
new file mode 100644
index 0000000000000..b9104114369eb
--- /dev/null
+++ b/llvm/test/Transforms/LoopRotate/predecessor-no-prof-branch-weights.ll
@@ -0,0 +1,84 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --no-generate-body-for-unused-prefixes --version 6
+; RUN: opt < %s -passes=loop-rotate -S | FileCheck %s
+;
+; The preheader predecessor has no profile metadata. This must not affect
+; preservation of the original header weights on the guard and latch.
+
+define void @no_pred_prof(ptr %p, i32 %n, i32 %cond) !prof !14 {
+; CHECK-LABEL: define void @no_pred_prof(
+; CHECK-SAME: ptr [[P:%.*]], i32 [[N:%.*]], i32 [[COND:%.*]]) !prof [[PROF14:![0-9]+]] {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[C0:%.*]] = icmp ne i32 [[COND]], 0
+; CHECK-NEXT: br i1 [[C0]], label %[[PH:.*]], label %[[RET:.*]]
+; CHECK: [[PH]]:
+; CHECK-NEXT: [[HCMP1:%.*]] = icmp sge i32 0, [[N]]
+; CHECK-NEXT: br i1 [[HCMP1]], label %[[EXIT:.*]], label %[[BODY_LR_PH:.*]], !prof [[PROF15:![0-9]+]]
+; CHECK: [[BODY_LR_PH]]:
+; CHECK-NEXT: br label %[[BODY:.*]]
+; CHECK: [[BODY]]:
+; CHECK-NEXT: [[IV2:%.*]] = phi i32 [ 0, %[[BODY_LR_PH]] ], [ [[IV_NEXT:%.*]], %[[LATCH:.*]] ]
+; CHECK-NEXT: [[ADDR:%.*]] = getelementptr i32, ptr [[P]], i32 [[IV2]]
+; CHECK-NEXT: store i32 [[IV2]], ptr [[ADDR]], align 4
+; CHECK-NEXT: br label %[[LATCH]]
+; CHECK: [[LATCH]]:
+; CHECK-NEXT: [[IV_NEXT]] = add i32 [[IV2]], 1
+; CHECK-NEXT: [[HCMP:%.*]] = icmp sge i32 [[IV_NEXT]], [[N]]
+; CHECK-NEXT: br i1 [[HCMP]], label %[[HEADER_EXIT_CRIT_EDGE:.*]], label %[[BODY]], !prof [[PROF15]]
+; CHECK: [[HEADER_EXIT_CRIT_EDGE]]:
+; CHECK-NEXT: br label %[[EXIT]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+; CHECK: [[RET]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ %c0 = icmp ne i32 %cond, 0
+ ; Intentionally no !prof.
+ br i1 %c0, label %ph, label %ret
+
+ph: ; preds = %entry
+ br label %header
+
+header: ; preds = %latch, %ph
+ %iv = phi i32 [ 0, %ph ], [ %iv.next, %latch ]
+ %hcmp = icmp sge i32 %iv, %n
+ br i1 %hcmp, label %exit, label %body, !prof !15
+
+body: ; preds = %header
+ %addr = getelementptr i32, ptr %p, i32 %iv
+ store i32 %iv, ptr %addr, align 4
+ br label %latch
+
+latch: ; preds = %body
+ %iv.next = add i32 %iv, 1
+ br label %header
+
+exit: ; preds = %header
+ ret void
+
+ret: ; preds = %entry
+ ret void
+}
+
+!llvm.module.flags = !{!0}
+
+!0 = !{i32 1, !"ProfileSummary", !1}
+!1 = !{!2, !3, !4, !5, !6, !7, !8, !9}
+!2 = !{!"ProfileFormat", !"InstrProf"}
+!3 = !{!"TotalCount", i64 10000}
+!4 = !{!"MaxCount", i64 9800}
+!5 = !{!"MaxInternalCount", i64 9800}
+!6 = !{!"MaxFunctionCount", i64 200}
+!7 = !{!"NumCounts", i64 3}
+!8 = !{!"NumFunctions", i64 1}
+!9 = !{!"DetailedSummary", !10}
+!10 = !{!11, !12, !13}
+!11 = !{i32 10000, i64 9800, i32 1}
+!12 = !{i32 999000, i64 200, i32 1}
+!13 = !{i32 999999, i64 200, i32 1}
+!14 = !{!"function_entry_count", i64 201}
+!15 = !{!"branch_weights", i32 200, i32 9800}
+;.
+; CHECK: [[PROF14]] = !{!"function_entry_count", i64 201}
+; CHECK: [[PROF15]] = !{!"branch_weights", i32 200, i32 9800}
+;.
diff --git a/llvm/test/Transforms/LoopRotate/saturating-backedge-branch-weights.ll b/llvm/test/Transforms/LoopRotate/saturating-backedge-branch-weights.ll
new file mode 100644
index 0000000000000..145caa56e6462
--- /dev/null
+++ b/llvm/test/Transforms/LoopRotate/saturating-backedge-branch-weights.ll
@@ -0,0 +1,96 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --no-generate-body-for-unused-prefixes --version 6
+; RUN: opt < %s -passes=loop-rotate -S | FileCheck %s
+;
+; Multi-exit loop with large header and body weights. Rotation leaves a
+; conditional preheader guard, so updateBranchWeights() must preserve the
+; original header and body branch ratios on the guard and latch.
+
+define void @sat_backedge(ptr %p, i32 %n, i32 %cond) !prof !14 {
+; CHECK-LABEL: define void @sat_backedge(
+; CHECK-SAME: ptr [[P:%.*]], i32 [[N:%.*]], i32 [[COND:%.*]]) !prof [[PROF14:![0-9]+]] {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[C0:%.*]] = icmp ne i32 [[COND]], 0
+; CHECK-NEXT: br i1 [[C0]], label %[[PH:.*]], label %[[RET:.*]], !prof [[PROF15:![0-9]+]]
+; CHECK: [[PH]]:
+; CHECK-NEXT: [[HCMP1:%.*]] = icmp sge i32 0, [[N]]
+; CHECK-NEXT: br i1 [[HCMP1]], label %[[EXIT1:.*]], label %[[BODY_LR_PH:.*]], !prof [[PROF16:![0-9]+]]
+; CHECK: [[BODY_LR_PH]]:
+; CHECK-NEXT: br label %[[BODY:.*]]
+; CHECK: [[HEADER:.*]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i32 [ [[IV_NEXT:%.*]], %[[BODY]] ]
+; CHECK-NEXT: [[HCMP:%.*]] = icmp sge i32 [[IV]], [[N]]
+; CHECK-NEXT: br i1 [[HCMP]], label %[[HEADER_EXIT1_CRIT_EDGE:.*]], label %[[BODY]], !prof [[PROF16]]
+; CHECK: [[BODY]]:
+; CHECK-NEXT: [[IV2:%.*]] = phi i32 [ 0, %[[BODY_LR_PH]] ], [ [[IV]], %[[HEADER]] ]
+; CHECK-NEXT: [[ADDR:%.*]] = getelementptr i32, ptr [[P]], i32 [[IV2]]
+; CHECK-NEXT: [[V:%.*]] = load i32, ptr [[ADDR]], align 4
+; CHECK-NEXT: [[BCMP:%.*]] = icmp slt i32 [[V]], 0
+; CHECK-NEXT: [[IV_NEXT]] = add i32 [[IV2]], 1
+; CHECK-NEXT: br i1 [[BCMP]], label %[[EXIT2:.*]], label %[[HEADER]], !prof [[PROF17:![0-9]+]]
+; CHECK: [[HEADER_EXIT1_CRIT_EDGE]]:
+; CHECK-NEXT: br label %[[EXIT1]]
+; CHECK: [[EXIT1]]:
+; CHECK-NEXT: ret void
+; CHECK: [[EXIT2]]:
+; CHECK-NEXT: ret void
+; CHECK: [[RET]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ %c0 = icmp ne i32 %cond, 0
+ br i1 %c0, label %ph, label %ret, !prof !15
+
+ph: ; preds = %entry
+ br label %header
+
+header: ; preds = %latch, %ph
+ %iv = phi i32 [ 0, %ph ], [ %iv.next, %latch ]
+ %hcmp = icmp sge i32 %iv, %n
+ br i1 %hcmp, label %exit1, label %body, !prof !16
+
+body: ; preds = %header
+ %addr = getelementptr i32, ptr %p, i32 %iv
+ %v = load i32, ptr %addr, align 4
+ %bcmp = icmp slt i32 %v, 0
+ br i1 %bcmp, label %exit2, label %latch, !prof !17
+
+latch: ; preds = %body
+ %iv.next = add i32 %iv, 1
+ br label %header
+
+exit1: ; preds = %header
+ ret void
+
+exit2: ; preds = %body
+ ret void
+
+ret: ; preds = %entry
+ ret void
+}
+
+!llvm.module.flags = !{!0}
+
+!0 = !{i32 1, !"ProfileSummary", !1}
+!1 = !{!2, !3, !4, !5, !6, !7, !8, !9}
+!2 = !{!"ProfileFormat", !"InstrProf"}
+!3 = !{!"TotalCount", i64 11000}
+!4 = !{!"MaxCount", i64 990}
+!5 = !{!"MaxInternalCount", i64 990}
+!6 = !{!"MaxFunctionCount", i64 1000}
+!7 = !{!"NumCounts", i64 4}
+!8 = !{!"NumFunctions", i64 1}
+!9 = !{!"DetailedSummary", !10}
+!10 = !{!11, !12, !13}
+!11 = !{i32 10000, i64 990, i32 1}
+!12 = !{i32 999000, i64 1000, i32 1}
+!13 = !{i32 999999, i64 990, i32 1}
+!14 = !{!"function_entry_count", i64 1000}
+!15 = !{!"branch_weights", i32 999, i32 1}
+!16 = !{!"branch_weights", i32 990, i32 10}
+!17 = !{!"branch_weights", i32 9, i32 1}
+;.
+; CHECK: [[PROF14]] = !{!"function_entry_count", i64 1000}
+; CHECK: [[PROF15]] = !{!"branch_weights", i32 999, i32 1}
+; CHECK: [[PROF16]] = !{!"branch_weights", i32 990, i32 10}
+; CHECK: [[PROF17]] = !{!"branch_weights", i32 9, i32 1}
+;.
diff --git a/llvm/test/Transforms/LoopRotate/scaled-preheader-branch-weights.ll b/llvm/test/Transforms/LoopRotate/scaled-preheader-branch-weights.ll
new file mode 100644
index 0000000000000..3dba06fec2d74
--- /dev/null
+++ b/llvm/test/Transforms/LoopRotate/scaled-preheader-branch-weights.ll
@@ -0,0 +1,60 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --no-generate-body-for-unused-prefixes --version 6
+; RUN: opt < %s -passes=loop-rotate -S | FileCheck %s
+;
+; The preheader and header use different weight scales. Rotation must preserve
+; each branch's local ratio without mixing the scales.
+
+define void @small_scaled(ptr %p, i32 %n, i1 %c) !prof !0 {
+; CHECK-LABEL: define void @small_scaled(
+; CHECK-SAME: ptr [[P:%.*]], i32 [[N:%.*]], i1 [[C:%.*]]) !prof [[PROF0:![0-9]+]] {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: br i1 [[C]], label %[[PH:.*]], label %[[RET:.*]], !prof [[PROF1:![0-9]+]]
+; CHECK: [[PH]]:
+; CHECK-NEXT: [[CMP1:%.*]] = icmp slt i32 0, [[N]]
+; CHECK-NEXT: br i1 [[CMP1]], label %[[BODY_LR_PH:.*]], label %[[EXIT:.*]], !prof [[PROF2:![0-9]+]]
+; CHECK: [[BODY_LR_PH]]:
+; CHECK-NEXT: br label %[[BODY:.*]]
+; CHECK: [[BODY]]:
+; CHECK-NEXT: [[IV2:%.*]] = phi i32 [ 0, %[[BODY_LR_PH]] ], [ [[NEXT:%.*]], %[[BODY]] ]
+; CHECK-NEXT: store volatile i32 [[IV2]], ptr [[P]], align 4
+; CHECK-NEXT: [[NEXT]] = add i32 [[IV2]], 1
+; CHECK-NEXT: [[CMP:%.*]] = icmp slt i32 [[NEXT]], [[N]]
+; CHECK-NEXT: br i1 [[CMP]], label %[[BODY]], label %[[HEADER_EXIT_CRIT_EDGE:.*]], !prof [[PROF2]]
+; CHECK: [[HEADER_EXIT_CRIT_EDGE]]:
+; CHECK-NEXT: br label %[[EXIT]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+; CHECK: [[RET]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br i1 %c, label %ph, label %ret, !prof !1
+
+ph:
+ br label %header
+
+header:
+ %iv = phi i32 [ 0, %ph ], [ %next, %body ]
+ %cmp = icmp slt i32 %iv, %n
+ br i1 %cmp, label %body, label %exit, !prof !2
+
+body:
+ store volatile i32 %iv, ptr %p, align 4
+ %next = add i32 %iv, 1
+ br label %header
+
+exit:
+ ret void
+
+ret:
+ ret void
+}
+
+!0 = !{!"function_entry_count", i64 2}
+!1 = !{!"branch_weights", i32 1, i32 1}
+!2 = !{!"branch_weights", i32 100, i32 1}
+;.
+; CHECK: [[PROF0]] = !{!"function_entry_count", i64 2}
+; CHECK: [[PROF1]] = !{!"branch_weights", i32 1, i32 1}
+; CHECK: [[PROF2]] = !{!"branch_weights", i32 100, i32 1}
+;.
diff --git a/llvm/test/Transforms/LoopRotate/update-branch-weights.ll b/llvm/test/Transforms/LoopRotate/update-branch-weights.ll
index 9a1f36ec5ff2b..912f70b34bb82 100644
--- a/llvm/test/Transforms/LoopRotate/update-branch-weights.ll
+++ b/llvm/test/Transforms/LoopRotate/update-branch-weights.ll
@@ -1,3 +1,4 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --no-generate-body-for-unused-prefixes --version 6
; RUN: opt < %s -passes='print<block-freq>' -disable-output 2>&1 | FileCheck %s --check-prefixes=BFI_BEFORE
; RUN: opt < %s -passes='loop(loop-rotate),print<block-freq>' -disable-output 2>&1 | FileCheck %s --check-prefixes=BFI_AFTER
; RUN: opt < %s -passes='loop(loop-rotate)' -S | FileCheck %s --check-prefixes=IR
@@ -23,15 +24,31 @@
; BFI_AFTER: - inner_loop_exit: {{.*}} count = 1000
; BFI_AFTER: - outer_loop_exit: {{.*}} count = 1
-; IR-LABEL: define void @func0
-; IR: inner_loop_body:
-; IR: br i1 %cmp1, label %inner_loop_body, label %inner_loop_exit, !prof [[PROF_FUNC0_0:![0-9]+]]
-; IR: inner_loop_exit:
-; IR: br i1 %cmp0, label %outer_loop_body, label %outer_loop_exit, !prof [[PROF_FUNC0_1:![0-9]+]]
;
; A function with known loop-bounds where after loop-rotation we end with an
; unconditional branch in the pre-header.
define void @func0() !prof !0 {
+; IR-LABEL: define void @func0(
+; IR-SAME: ) !prof [[PROF0:![0-9]+]] {
+; IR-NEXT: [[ENTRY:.*]]:
+; IR-NEXT: br label %[[OUTER_LOOP_BODY:.*]]
+; IR: [[OUTER_LOOP_BODY]]:
+; IR-NEXT: [[I02:%.*]] = phi i32 [ 0, %[[ENTRY]] ], [ [[I0_INC:%.*]], %[[INNER_LOOP_EXIT:.*]] ]
+; IR-NEXT: store volatile i32 [[I02]], ptr @g, align 4
+; IR-NEXT: br label %[[INNER_LOOP_BODY:.*]]
+; IR: [[INNER_LOOP_BODY]]:
+; IR-NEXT: [[I11:%.*]] = phi i32 [ 0, %[[OUTER_LOOP_BODY]] ], [ [[I1_INC:%.*]], %[[INNER_LOOP_BODY]] ]
+; IR-NEXT: store volatile i32 [[I11]], ptr @g, align 4
+; IR-NEXT: [[I1_INC]] = add i32 [[I11]], 1
+; IR-NEXT: [[CMP1:%.*]] = icmp slt i32 [[I1_INC]], 3
+; IR-NEXT: br i1 [[CMP1]], label %[[INNER_LOOP_BODY]], label %[[INNER_LOOP_EXIT]], !prof [[PROF1:![0-9]+]]
+; IR: [[INNER_LOOP_EXIT]]:
+; IR-NEXT: [[I0_INC]] = add i32 [[I02]], 1
+; IR-NEXT: [[CMP0:%.*]] = icmp slt i32 [[I0_INC]], 1000
+; IR-NEXT: br i1 [[CMP0]], label %[[OUTER_LOOP_BODY]], label %[[OUTER_LOOP_EXIT:.*]], !prof [[PROF2:![0-9]+]]
+; IR: [[OUTER_LOOP_EXIT]]:
+; IR-NEXT: ret void
+;
entry:
br label %outer_loop_header
@@ -70,22 +87,31 @@ outer_loop_exit:
; BFI_AFTER-LABEL: block-frequency-info: func1
; BFI_AFTER: - entry: {{.*}} count = 1024
-; BFI_AFTER: - loop_body.lr.ph: {{.*}} count = 1016
; BFI_AFTER: - loop_body: {{.*}} count = 20480
-; BFI_AFTER: - loop_header.loop_exit_crit_edge: {{.*}} count = 1016
; BFI_AFTER: - loop_exit: {{.*}} count = 1024
-; IR-LABEL: define void @func1
-; IR: entry:
-; IR: br i1 %cmp1, label %loop_body.lr.ph, label %loop_exit, !prof [[PROF_FUNC1_0:![0-9]+]]
-
-; IR: loop_body:
-; IR: br i1 %cmp, label %loop_body, label %loop_header.loop_exit_crit_edge, !prof [[PROF_FUNC1_1:![0-9]+]]
-
; A function with unknown loop-bounds so loop-rotation ends up with a
; condition jump in pre-header and loop body. branch_weight shows body is
; executed more often than header.
define void @func1(i32 %n) !prof !3 {
+; IR-LABEL: define void @func1(
+; IR-SAME: i32 [[N:%.*]]) !prof [[PROF3:![0-9]+]] {
+; IR-NEXT: [[ENTRY:.*:]]
+; IR-NEXT: [[CMP1:%.*]] = icmp slt i32 0, [[N]]
+; IR-NEXT: br i1 [[CMP1]], label %[[LOOP_BODY_LR_PH:.*]], label %[[LOOP_EXIT:.*]], !prof [[PROF4:![0-9]+]]
+; IR: [[LOOP_BODY_LR_PH]]:
+; IR-NEXT: br label %[[LOOP_BODY:.*]]
+; IR: [[LOOP_BODY]]:
+; IR-NEXT: [[I2:%.*]] = phi i32 [ 0, %[[LOOP_BODY_LR_PH]] ], [ [[I_INC:%.*]], %[[LOOP_BODY]] ]
+; IR-NEXT: store volatile i32 [[I2]], ptr @g, align 4
+; IR-NEXT: [[I_INC]] = add i32 [[I2]], 1
+; IR-NEXT: [[CMP:%.*]] = icmp slt i32 [[I_INC]], [[N]]
+; IR-NEXT: br i1 [[CMP]], label %[[LOOP_BODY]], label %[[LOOP_HEADER_LOOP_EXIT_CRIT_EDGE:.*]], !prof [[PROF4]]
+; IR: [[LOOP_HEADER_LOOP_EXIT_CRIT_EDGE]]:
+; IR-NEXT: br label %[[LOOP_EXIT]]
+; IR: [[LOOP_EXIT]]:
+; IR-NEXT: ret void
+;
entry:
br label %loop_header
@@ -110,23 +136,32 @@ loop_exit:
; BFI_BEFORE: - loop_exit: {{.*}} count = 1024
; BFI_AFTER-LABEL: block-frequency-info: func2
-; - entry: {{.*}} count = 1024
-; - loop_body.lr.ph: {{.*}} count = 32
-; - loop_body: {{.*}} count = 32
-; - loop_header.loop_exit_crit_edge: {{.*}} count = 32
-; - loop_exit: {{.*}} count = 1024
-
-; IR-LABEL: define void @func2
-; IR: entry:
-; IR: br i1 %cmp1, label %loop_exit, label %loop_body.lr.ph, !prof [[PROF_FUNC2_0:![0-9]+]]
-
-; IR: loop_body:
-; IR: br i1 %cmp, label %loop_header.loop_exit_crit_edge, label %loop_body, !prof [[PROF_FUNC2_1:![0-9]+]]
+; BFI_AFTER: - entry: {{.*}} count = 1024
+; BFI_AFTER: - loop_body: {{.*}} count = 32
+; BFI_AFTER: - loop_exit: {{.*}} count = 1024
; A function with unknown loop-bounds so loop-rotation ends up with a
; condition jump in pre-header and loop body. Similar to `func1` but here
; loop-exit count is higher than backedge count.
define void @func2(i32 %n) !prof !3 {
+; IR-LABEL: define void @func2(
+; IR-SAME: i32 [[N:%.*]]) !prof [[PROF3]] {
+; IR-NEXT: [[ENTRY:.*:]]
+; IR-NEXT: [[CMP1:%.*]] = icmp slt i32 0, [[N]]
+; IR-NEXT: br i1 [[CMP1]], label %[[LOOP_EXIT:.*]], label %[[LOOP_BODY_LR_PH:.*]], !prof [[PROF5:![0-9]+]]
+; IR: [[LOOP_BODY_LR_PH]]:
+; IR-NEXT: br label %[[LOOP_BODY:.*]]
+; IR: [[LOOP_BODY]]:
+; IR-NEXT: [[I2:%.*]] = phi i32 [ 0, %[[LOOP_BODY_LR_PH]] ], [ [[I_INC:%.*]], %[[LOOP_BODY]] ]
+; IR-NEXT: store volatile i32 [[I2]], ptr @g, align 4
+; IR-NEXT: [[I_INC]] = add i32 [[I2]], 1
+; IR-NEXT: [[CMP:%.*]] = icmp slt i32 [[I_INC]], [[N]]
+; IR-NEXT: br i1 [[CMP]], label %[[LOOP_HEADER_LOOP_EXIT_CRIT_EDGE:.*]], label %[[LOOP_BODY]], !prof [[PROF5]]
+; IR: [[LOOP_HEADER_LOOP_EXIT_CRIT_EDGE]]:
+; IR-NEXT: br label %[[LOOP_EXIT]]
+; IR: [[LOOP_EXIT]]:
+; IR-NEXT: ret void
+;
entry:
br label %loop_header
@@ -157,14 +192,25 @@ loop_exit:
; BFI_AFTER: - loop_header.loop_exit_crit_edge: {{.*}} count = 1024
; BFI_AFTER: - loop_exit: {{.*}} count = 1024
-; IR-LABEL: define void @func3_zero_branch_weight
-; IR: entry:
-; IR: br i1 %cmp1, label %loop_exit, label %loop_body.lr.ph, !prof [[PROF_FUNC3_0:![0-9]+]]
-
-; IR: loop_body:
-; IR: br i1 %cmp, label %loop_header.loop_exit_crit_edge, label %loop_body, !prof [[PROF_FUNC3_0]]
-
define void @func3_zero_branch_weight(i32 %n) !prof !3 {
+; IR-LABEL: define void @func3_zero_branch_weight(
+; IR-SAME: i32 [[N:%.*]]) !prof [[PROF3]] {
+; IR-NEXT: [[ENTRY:.*:]]
+; IR-NEXT: [[CMP1:%.*]] = icmp slt i32 0, [[N]]
+; IR-NEXT: br i1 [[CMP1]], label %[[LOOP_EXIT:.*]], label %[[LOOP_BODY_LR_PH:.*]], !prof [[PROF6:![0-9]+]]
+; IR: [[LOOP_BODY_LR_PH]]:
+; IR-NEXT: br label %[[LOOP_BODY:.*]]
+; IR: [[LOOP_BODY]]:
+; IR-NEXT: [[I2:%.*]] = phi i32 [ 0, %[[LOOP_BODY_LR_PH]] ], [ [[I_INC:%.*]], %[[LOOP_BODY]] ]
+; IR-NEXT: store volatile i32 [[I2]], ptr @g, align 4
+; IR-NEXT: [[I_INC]] = add i32 [[I2]], 1
+; IR-NEXT: [[CMP:%.*]] = icmp slt i32 [[I_INC]], [[N]]
+; IR-NEXT: br i1 [[CMP]], label %[[LOOP_HEADER_LOOP_EXIT_CRIT_EDGE:.*]], label %[[LOOP_BODY]], !prof [[PROF6]]
+; IR: [[LOOP_HEADER_LOOP_EXIT_CRIT_EDGE]]:
+; IR-NEXT: br label %[[LOOP_EXIT]]
+; IR: [[LOOP_EXIT]]:
+; IR-NEXT: ret void
+;
entry:
br label %loop_header
@@ -182,14 +228,25 @@ loop_exit:
ret void
}
-; IR-LABEL: define void @func4_zero_branch_weight
-; IR: entry:
-; IR: br i1 %cmp1, label %loop_exit, label %loop_body.lr.ph, !prof [[PROF_FUNC4_0:![0-9]+]]
-
-; IR: loop_body:
-; IR: br i1 %cmp, label %loop_header.loop_exit_crit_edge, label %loop_body, !prof [[PROF_FUNC4_0]]
-
define void @func4_zero_branch_weight(i32 %n) !prof !3 {
+; IR-LABEL: define void @func4_zero_branch_weight(
+; IR-SAME: i32 [[N:%.*]]) !prof [[PROF3]] {
+; IR-NEXT: [[ENTRY:.*:]]
+; IR-NEXT: [[CMP1:%.*]] = icmp slt i32 0, [[N]]
+; IR-NEXT: br i1 [[CMP1]], label %[[LOOP_EXIT:.*]], label %[[LOOP_BODY_LR_PH:.*]], !prof [[PROF7:![0-9]+]]
+; IR: [[LOOP_BODY_LR_PH]]:
+; IR-NEXT: br label %[[LOOP_BODY:.*]]
+; IR: [[LOOP_BODY]]:
+; IR-NEXT: [[I2:%.*]] = phi i32 [ 0, %[[LOOP_BODY_LR_PH]] ], [ [[I_INC:%.*]], %[[LOOP_BODY]] ]
+; IR-NEXT: store volatile i32 [[I2]], ptr @g, align 4
+; IR-NEXT: [[I_INC]] = add i32 [[I2]], 1
+; IR-NEXT: [[CMP:%.*]] = icmp slt i32 [[I_INC]], [[N]]
+; IR-NEXT: br i1 [[CMP]], label %[[LOOP_HEADER_LOOP_EXIT_CRIT_EDGE:.*]], label %[[LOOP_BODY]], !prof [[PROF7]]
+; IR: [[LOOP_HEADER_LOOP_EXIT_CRIT_EDGE]]:
+; IR-NEXT: br label %[[LOOP_EXIT]]
+; IR: [[LOOP_EXIT]]:
+; IR-NEXT: ret void
+;
entry:
br label %loop_header
@@ -207,14 +264,25 @@ loop_exit:
ret void
}
-; IR-LABEL: define void @func5_zero_branch_weight
-; IR: entry:
-; IR: br i1 %cmp1, label %loop_exit, label %loop_body.lr.ph, !prof [[PROF_FUNC5_0:![0-9]+]]
-
-; IR: loop_body:
-; IR: br i1 %cmp, label %loop_header.loop_exit_crit_edge, label %loop_body, !prof [[PROF_FUNC5_0]]
-
define void @func5_zero_branch_weight(i32 %n) !prof !3 {
+; IR-LABEL: define void @func5_zero_branch_weight(
+; IR-SAME: i32 [[N:%.*]]) !prof [[PROF3]] {
+; IR-NEXT: [[ENTRY:.*:]]
+; IR-NEXT: [[CMP1:%.*]] = icmp slt i32 0, [[N]]
+; IR-NEXT: br i1 [[CMP1]], label %[[LOOP_EXIT:.*]], label %[[LOOP_BODY_LR_PH:.*]], !prof [[PROF8:![0-9]+]]
+; IR: [[LOOP_BODY_LR_PH]]:
+; IR-NEXT: br label %[[LOOP_BODY:.*]]
+; IR: [[LOOP_BODY]]:
+; IR-NEXT: [[I2:%.*]] = phi i32 [ 0, %[[LOOP_BODY_LR_PH]] ], [ [[I_INC:%.*]], %[[LOOP_BODY]] ]
+; IR-NEXT: store volatile i32 [[I2]], ptr @g, align 4
+; IR-NEXT: [[I_INC]] = add i32 [[I2]], 1
+; IR-NEXT: [[CMP:%.*]] = icmp slt i32 [[I_INC]], [[N]]
+; IR-NEXT: br i1 [[CMP]], label %[[LOOP_HEADER_LOOP_EXIT_CRIT_EDGE:.*]], label %[[LOOP_BODY]], !prof [[PROF8]]
+; IR: [[LOOP_HEADER_LOOP_EXIT_CRIT_EDGE]]:
+; IR-NEXT: br label %[[LOOP_EXIT]]
+; IR: [[LOOP_EXIT]]:
+; IR-NEXT: ret void
+;
entry:
br label %loop_header
@@ -243,18 +311,24 @@ loop_exit:
; BFI_AFTER: - loop_body: {{.*}} count = 1024
; BFI_AFTER: - loop_exit: {{.*}} count = 1024
-; IR-LABEL: define void @func6_inaccurate_branch_weight(
-; IR: entry:
-; IR: br label %loop_body
-; IR: loop_body:
-; IR: br i1 %cmp, label %loop_body, label %loop_exit, !prof [[PROF_FUNC6_0:![0-9]+]]
-; IR: loop_exit:
-; IR: ret void
; Branch weight from sample-based PGO may be inaccurate due to sampling.
; Count for loop_body in following case should be not less than loop_exit.
; However this may not hold for Sample-based PGO.
define void @func6_inaccurate_branch_weight() !prof !3 {
+; IR-LABEL: define void @func6_inaccurate_branch_weight(
+; IR-SAME: ) !prof [[PROF3]] {
+; IR-NEXT: [[ENTRY:.*]]:
+; IR-NEXT: br label %[[LOOP_BODY:.*]]
+; IR: [[LOOP_BODY]]:
+; IR-NEXT: [[I1:%.*]] = phi i32 [ 0, %[[ENTRY]] ], [ [[I_INC:%.*]], %[[LOOP_BODY]] ]
+; IR-NEXT: store volatile i32 [[I1]], ptr @g, align 4
+; IR-NEXT: [[I_INC]] = add i32 [[I1]], 1
+; IR-NEXT: [[CMP:%.*]] = icmp slt i32 [[I_INC]], 2
+; IR-NEXT: br i1 [[CMP]], label %[[LOOP_BODY]], label %[[LOOP_EXIT:.*]], !prof [[PROF9:![0-9]+]]
+; IR: [[LOOP_EXIT]]:
+; IR-NEXT: ret void
+;
entry:
br label %loop_header
@@ -283,13 +357,15 @@ loop_exit:
!8 = !{!"branch_weights", i32 0, i32 0}
!9 = !{!"branch_weights", i32 1023, i32 1024}
-; IR: [[PROF_FUNC0_0]] = !{!"branch_weights", i32 2000, i32 1000}
-; IR: [[PROF_FUNC0_1]] = !{!"branch_weights", i32 999, i32 1}
-; IR: [[PROF_FUNC1_0]] = !{!"branch_weights", i32 127, i32 1}
-; IR: [[PROF_FUNC1_1]] = !{!"branch_weights", i32 2433, i32 127}
-; IR: [[PROF_FUNC2_0]] = !{!"branch_weights", i32 9920, i32 320}
-; IR: [[PROF_FUNC2_1]] = !{!"branch_weights", i32 320, i32 0}
-; IR: [[PROF_FUNC3_0]] = !{!"branch_weights", i32 0, i32 1}
-; IR: [[PROF_FUNC4_0]] = !{!"branch_weights", i32 1, i32 0}
-; IR: [[PROF_FUNC5_0]] = !{!"branch_weights", i32 0, i32 0}
-; IR: [[PROF_FUNC6_0]] = !{!"branch_weights", i32 0, i32 1024}
+;.
+; IR: [[PROF0]] = !{!"function_entry_count", i64 1}
+; IR: [[PROF1]] = !{!"branch_weights", i32 2000, i32 1000}
+; IR: [[PROF2]] = !{!"branch_weights", i32 999, i32 1}
+; IR: [[PROF3]] = !{!"function_entry_count", i64 1024}
+; IR: [[PROF4]] = !{!"branch_weights", i32 40, i32 2}
+; IR: [[PROF5]] = !{!"branch_weights", i32 10240, i32 320}
+; IR: [[PROF6]] = !{!"branch_weights", i32 0, i32 1}
+; IR: [[PROF7]] = !{!"branch_weights", i32 1, i32 0}
+; IR: [[PROF8]] = !{!"branch_weights", i32 0, i32 0}
+; IR: [[PROF9]] = !{!"branch_weights", i32 0, i32 1024}
+;.
>From 4235672b7dec02d30ea135607c8a4d8388042b8c Mon Sep 17 00:00:00 2001
From: Alok Kumar Sharma <AlokKumar.Sharma at amd.com>
Date: Wed, 19 Aug 2026 23:29:26 +0530
Subject: [PATCH 2/3] [LoopRotate] Move post-rotation trip-count adjustment to
loop metadata
After rotation, decrement llvm.loop.estimated_trip_count instead of
rewriting branch weights. Infer from weights only on single-exit loops,
and keep explicit multi-exit estimates unchanged aside from the
decrement.
---
.../Transforms/Utils/LoopRotationUtils.cpp | 25 +++
.../LoopRotate/multi-exit-branch-weights.ll | 94 ++++++----
.../LoopRotate/update-branch-weights.ll | 169 +++++++++++++++---
3 files changed, 224 insertions(+), 64 deletions(-)
diff --git a/llvm/lib/Transforms/Utils/LoopRotationUtils.cpp b/llvm/lib/Transforms/Utils/LoopRotationUtils.cpp
index 7950e223bfb4d..17fb32a1536df 100644
--- a/llvm/lib/Transforms/Utils/LoopRotationUtils.cpp
+++ b/llvm/lib/Transforms/Utils/LoopRotationUtils.cpp
@@ -10,6 +10,8 @@
//
//===----------------------------------------------------------------------===//
+#include <optional>
+
#include "llvm/Transforms/Utils/LoopRotationUtils.h"
#include "llvm/ADT/Statistic.h"
#include "llvm/Analysis/AssumptionCache.h"
@@ -33,8 +35,10 @@
#include "llvm/Transforms/Utils/BasicBlockUtils.h"
#include "llvm/Transforms/Utils/Cloning.h"
#include "llvm/Transforms/Utils/Local.h"
+#include "llvm/Transforms/Utils/LoopUtils.h"
#include "llvm/Transforms/Utils/SSAUpdater.h"
#include "llvm/Transforms/Utils/ValueMapper.h"
+
using namespace llvm;
#define DEBUG_TYPE "loop-rotate"
@@ -697,6 +701,17 @@ bool LoopRotate::rotateLoop(Loop *L, bool SimplifiedLatch) {
!isa<ConstantInt>(Cond) ||
PHBI->getSuccessor(cast<ConstantInt>(Cond)->isZero()) != NewHeader;
+ // Save the trip count before updating branch weights. Only infer it from
+ // profile weights for single-exit loops, but still read and decrement
+ // explicit llvm.loop.estimated_trip_count on any loop.
+ // getExitingBlock() is null when there is no unique dedicated exit block.
+ bool IsMultiExitLoop = !L->getExitingBlock();
+ bool HasExplicitEstimatedTripCount =
+ getOptionalIntLoopAttribute(L, LLVMLoopEstimatedTripCount).has_value();
+ std::optional<unsigned> EstimatedTripCount;
+ if (!IsMultiExitLoop || HasExplicitEstimatedTripCount)
+ EstimatedTripCount = getLoopEstimatedTripCount(L);
+
updateBranchWeights(*PHBI, *BI, HasConditionalPreHeader, BISuccsSwapped);
if (HasConditionalPreHeader) {
@@ -749,6 +764,16 @@ bool LoopRotate::rotateLoop(Loop *L, bool SimplifiedLatch) {
MSSAU->removeEdge(OrigPreheader, Exit);
}
+ if (EstimatedTripCount) {
+ // One header test moved to the preheader and no longer counts as a trip.
+ // Record the decremented count in loop metadata only; updateBranchWeights()
+ // already handled branch weights above.
+ if (!setLoopEstimatedTripCount(
+ L, *EstimatedTripCount == 0 ? 0 : *EstimatedTripCount - 1))
+ LLVM_DEBUG(dbgs() << "LoopRotation: failed to set estimated trip "
+ << "count after rotation\n");
+ }
+
assert(L->getLoopPreheader() && "Invalid loop preheader after loop rotation");
assert(L->getLoopLatch() && "Invalid loop latch after loop rotation");
diff --git a/llvm/test/Transforms/LoopRotate/multi-exit-branch-weights.ll b/llvm/test/Transforms/LoopRotate/multi-exit-branch-weights.ll
index ac47fc94b4bf5..5e24888abbdaa 100644
--- a/llvm/test/Transforms/LoopRotate/multi-exit-branch-weights.ll
+++ b/llvm/test/Transforms/LoopRotate/multi-exit-branch-weights.ll
@@ -1,5 +1,6 @@
-; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --no-generate-body-for-unused-prefixes --version 6
; RUN: opt < %s -passes=loop-rotate -S | FileCheck %s
+; RUN: opt < %s -passes='print<block-freq>' -disable-output 2>&1 | FileCheck %s --check-prefix=BFI-BEFORE
+; RUN: opt < %s -passes='loop(loop-rotate),print<block-freq>' -disable-output 2>&1 | FileCheck %s --check-prefix=BFI-AFTER
;
; The copied guard and latch run the same test. Keep the original weights on
; both, including loops with another body exit.
@@ -12,37 +13,7 @@
; body --(bcmp)--> exit2 / latch !prof { 10, 9790 }
; latch ---------> header
-define void @f(ptr %p, i32 %start, i32 %limit, i32 %cond) {
-; CHECK-LABEL: define void @f(
-; CHECK-SAME: ptr [[P:%.*]], i32 [[START:%.*]], i32 [[LIMIT:%.*]], i32 [[COND:%.*]]) {
-; CHECK-NEXT: [[ENTRY:.*:]]
-; CHECK-NEXT: [[C0:%.*]] = icmp ne i32 [[COND]], 0
-; CHECK-NEXT: br i1 [[C0]], label %[[PH:.*]], label %[[RET:.*]], !prof [[PROF0:![0-9]+]]
-; CHECK: [[PH]]:
-; CHECK-NEXT: [[HCMP1:%.*]] = icmp eq i32 [[START]], [[LIMIT]]
-; CHECK-NEXT: br i1 [[HCMP1]], label %[[EXIT1:.*]], label %[[BODY_LR_PH:.*]], !prof [[PROF1:![0-9]+]]
-; CHECK: [[BODY_LR_PH]]:
-; CHECK-NEXT: br label %[[BODY:.*]]
-; CHECK: [[HEADER:.*]]:
-; CHECK-NEXT: [[IV:%.*]] = phi i32 [ [[IV_NEXT:%.*]], %[[BODY]] ]
-; CHECK-NEXT: [[HCMP:%.*]] = icmp eq i32 [[IV]], [[LIMIT]]
-; CHECK-NEXT: br i1 [[HCMP]], label %[[HEADER_EXIT1_CRIT_EDGE:.*]], label %[[BODY]], !prof [[PROF1]]
-; CHECK: [[BODY]]:
-; CHECK-NEXT: [[IV2:%.*]] = phi i32 [ [[START]], %[[BODY_LR_PH]] ], [ [[IV]], %[[HEADER]] ]
-; CHECK-NEXT: [[ADDR:%.*]] = getelementptr i32, ptr [[P]], i32 [[IV2]]
-; CHECK-NEXT: [[V:%.*]] = load i32, ptr [[ADDR]], align 4
-; CHECK-NEXT: [[BCMP:%.*]] = icmp slt i32 [[V]], 0
-; CHECK-NEXT: [[IV_NEXT]] = add i32 [[IV2]], 1
-; CHECK-NEXT: br i1 [[BCMP]], label %[[EXIT2:.*]], label %[[HEADER]], !prof [[PROF2:![0-9]+]]
-; CHECK: [[HEADER_EXIT1_CRIT_EDGE]]:
-; CHECK-NEXT: br label %[[EXIT1]]
-; CHECK: [[EXIT1]]:
-; CHECK-NEXT: ret void
-; CHECK: [[EXIT2]]:
-; CHECK-NEXT: ret void
-; CHECK: [[RET]]:
-; CHECK-NEXT: ret void
-;
+define void @f(ptr %p, i32 %start, i32 %limit, i32 %cond) !prof !3 {
entry:
%c0 = icmp ne i32 %cond, 0
br i1 %c0, label %ph, label %ret, !prof !0
@@ -76,12 +47,61 @@ ret: ; preds = %entry
ret void
}
+; Keep explicit trip-count metadata for multi-exit loops. Rotation decrements
+; the attribute but does not infer a trip count from branch weights without it.
+define void @explicit_trip_count(ptr %p, i32 %start, i32 %limit) {
+entry:
+ br label %header
+
+header:
+ %iv = phi i32 [ %start, %entry ], [ %iv.next, %latch ]
+ %hcmp = icmp eq i32 %iv, %limit
+ br i1 %hcmp, label %exit1, label %body, !prof !1, !llvm.loop !4
+
+body:
+ %v = load i32, ptr %p, align 4
+ %bcmp = icmp slt i32 %v, 0
+ br i1 %bcmp, label %exit2, label %latch, !prof !2
+
+latch:
+ %iv.next = add i32 %iv, 1
+ br label %header
+
+exit1:
+ ret void
+
+exit2:
+ ret void
+}
+
!0 = !{!"branch_weights", i32 42, i32 2}
!1 = !{!"branch_weights", i32 200, i32 9800}
!2 = !{!"branch_weights", i32 10, i32 9790}
+!3 = !{!"function_entry_count", i64 44}
+!4 = distinct !{!4, !5}
+!5 = !{!"llvm.loop.estimated_trip_count", i32 10}
+
+; BFI-BEFORE-LABEL: block-frequency-info: f
+; BFI-BEFORE: - body: {{.*}} count = 1960
+; BFI-BEFORE: - exit1: {{.*}} count = 40
+; BFI-BEFORE: - exit2: {{.*}} count = 2
+; BFI-BEFORE: - ret: {{.*}} count = 2
+
+; BFI-AFTER-LABEL: block-frequency-info: f
+; BFI-AFTER: - body: {{.*}} count = 1960
+; BFI-AFTER: - exit1: {{.*}} count = 40
+; BFI-AFTER: - exit2: {{.*}} count = 2
+; BFI-AFTER: - ret: {{.*}} count = 2
+
+; CHECK-LABEL: define void @f(
+
+; CHECK: ph:
+; CHECK: br i1 %{{.*}}, label %{{.*}}, label %{{.*}}.lr.ph, !prof [[WEIGHTS:![0-9]+]]
+
+; CHECK: br i1 %{{.*}}, label %{{.*}}, label %body, !prof [[WEIGHTS]]{{$}}
-;.
-; CHECK: [[PROF0]] = !{!"branch_weights", i32 42, i32 2}
-; CHECK: [[PROF1]] = !{!"branch_weights", i32 200, i32 9800}
-; CHECK: [[PROF2]] = !{!"branch_weights", i32 10, i32 9790}
-;.
+; CHECK-LABEL: define void @explicit_trip_count(
+; CHECK: br i1 %{{.*}}, label %{{.*}}, label %body, !prof {{![0-9]+}}, !llvm.loop [[EXPLICIT_LOOP:![0-9]+]]
+; CHECK-DAG: [[WEIGHTS]] = !{!"branch_weights", i32 200, i32 9800}
+; CHECK-DAG: [[EXPLICIT_LOOP]] = distinct !{[[EXPLICIT_LOOP]], [[EXPLICIT_TC:![0-9]+]]}
+; CHECK-DAG: [[EXPLICIT_TC]] = !{!"llvm.loop.estimated_trip_count", i32 9}
diff --git a/llvm/test/Transforms/LoopRotate/update-branch-weights.ll b/llvm/test/Transforms/LoopRotate/update-branch-weights.ll
index 912f70b34bb82..df5471788ae0d 100644
--- a/llvm/test/Transforms/LoopRotate/update-branch-weights.ll
+++ b/llvm/test/Transforms/LoopRotate/update-branch-weights.ll
@@ -41,11 +41,11 @@ define void @func0() !prof !0 {
; IR-NEXT: store volatile i32 [[I11]], ptr @g, align 4
; IR-NEXT: [[I1_INC]] = add i32 [[I11]], 1
; IR-NEXT: [[CMP1:%.*]] = icmp slt i32 [[I1_INC]], 3
-; IR-NEXT: br i1 [[CMP1]], label %[[INNER_LOOP_BODY]], label %[[INNER_LOOP_EXIT]], !prof [[PROF1:![0-9]+]]
+; IR-NEXT: br i1 [[CMP1]], label %[[INNER_LOOP_BODY]], label %[[INNER_LOOP_EXIT]], !prof [[PROF1:![0-9]+]], !llvm.loop [[LOOP2:![0-9]+]]
; IR: [[INNER_LOOP_EXIT]]:
; IR-NEXT: [[I0_INC]] = add i32 [[I02]], 1
; IR-NEXT: [[CMP0:%.*]] = icmp slt i32 [[I0_INC]], 1000
-; IR-NEXT: br i1 [[CMP0]], label %[[OUTER_LOOP_BODY]], label %[[OUTER_LOOP_EXIT:.*]], !prof [[PROF2:![0-9]+]]
+; IR-NEXT: br i1 [[CMP0]], label %[[OUTER_LOOP_BODY]], label %[[OUTER_LOOP_EXIT:.*]], !prof [[PROF4:![0-9]+]], !llvm.loop [[LOOP5:![0-9]+]]
; IR: [[OUTER_LOOP_EXIT]]:
; IR-NEXT: ret void
;
@@ -95,10 +95,10 @@ outer_loop_exit:
; executed more often than header.
define void @func1(i32 %n) !prof !3 {
; IR-LABEL: define void @func1(
-; IR-SAME: i32 [[N:%.*]]) !prof [[PROF3:![0-9]+]] {
+; IR-SAME: i32 [[N:%.*]]) !prof [[PROF7:![0-9]+]] {
; IR-NEXT: [[ENTRY:.*:]]
; IR-NEXT: [[CMP1:%.*]] = icmp slt i32 0, [[N]]
-; IR-NEXT: br i1 [[CMP1]], label %[[LOOP_BODY_LR_PH:.*]], label %[[LOOP_EXIT:.*]], !prof [[PROF4:![0-9]+]]
+; IR-NEXT: br i1 [[CMP1]], label %[[LOOP_BODY_LR_PH:.*]], label %[[LOOP_EXIT:.*]], !prof [[PROF8:![0-9]+]]
; IR: [[LOOP_BODY_LR_PH]]:
; IR-NEXT: br label %[[LOOP_BODY:.*]]
; IR: [[LOOP_BODY]]:
@@ -106,7 +106,7 @@ define void @func1(i32 %n) !prof !3 {
; IR-NEXT: store volatile i32 [[I2]], ptr @g, align 4
; IR-NEXT: [[I_INC]] = add i32 [[I2]], 1
; IR-NEXT: [[CMP:%.*]] = icmp slt i32 [[I_INC]], [[N]]
-; IR-NEXT: br i1 [[CMP]], label %[[LOOP_BODY]], label %[[LOOP_HEADER_LOOP_EXIT_CRIT_EDGE:.*]], !prof [[PROF4]]
+; IR-NEXT: br i1 [[CMP]], label %[[LOOP_BODY]], label %[[LOOP_HEADER_LOOP_EXIT_CRIT_EDGE:.*]], !prof [[PROF8]], !llvm.loop [[LOOP9:![0-9]+]]
; IR: [[LOOP_HEADER_LOOP_EXIT_CRIT_EDGE]]:
; IR-NEXT: br label %[[LOOP_EXIT]]
; IR: [[LOOP_EXIT]]:
@@ -145,10 +145,10 @@ loop_exit:
; loop-exit count is higher than backedge count.
define void @func2(i32 %n) !prof !3 {
; IR-LABEL: define void @func2(
-; IR-SAME: i32 [[N:%.*]]) !prof [[PROF3]] {
+; IR-SAME: i32 [[N:%.*]]) !prof [[PROF7]] {
; IR-NEXT: [[ENTRY:.*:]]
; IR-NEXT: [[CMP1:%.*]] = icmp slt i32 0, [[N]]
-; IR-NEXT: br i1 [[CMP1]], label %[[LOOP_EXIT:.*]], label %[[LOOP_BODY_LR_PH:.*]], !prof [[PROF5:![0-9]+]]
+; IR-NEXT: br i1 [[CMP1]], label %[[LOOP_EXIT:.*]], label %[[LOOP_BODY_LR_PH:.*]], !prof [[PROF11:![0-9]+]]
; IR: [[LOOP_BODY_LR_PH]]:
; IR-NEXT: br label %[[LOOP_BODY:.*]]
; IR: [[LOOP_BODY]]:
@@ -156,7 +156,7 @@ define void @func2(i32 %n) !prof !3 {
; IR-NEXT: store volatile i32 [[I2]], ptr @g, align 4
; IR-NEXT: [[I_INC]] = add i32 [[I2]], 1
; IR-NEXT: [[CMP:%.*]] = icmp slt i32 [[I_INC]], [[N]]
-; IR-NEXT: br i1 [[CMP]], label %[[LOOP_HEADER_LOOP_EXIT_CRIT_EDGE:.*]], label %[[LOOP_BODY]], !prof [[PROF5]]
+; IR-NEXT: br i1 [[CMP]], label %[[LOOP_HEADER_LOOP_EXIT_CRIT_EDGE:.*]], label %[[LOOP_BODY]], !prof [[PROF11]], !llvm.loop [[LOOP12:![0-9]+]]
; IR: [[LOOP_HEADER_LOOP_EXIT_CRIT_EDGE]]:
; IR-NEXT: br label %[[LOOP_EXIT]]
; IR: [[LOOP_EXIT]]:
@@ -194,10 +194,10 @@ loop_exit:
define void @func3_zero_branch_weight(i32 %n) !prof !3 {
; IR-LABEL: define void @func3_zero_branch_weight(
-; IR-SAME: i32 [[N:%.*]]) !prof [[PROF3]] {
+; IR-SAME: i32 [[N:%.*]]) !prof [[PROF7]] {
; IR-NEXT: [[ENTRY:.*:]]
; IR-NEXT: [[CMP1:%.*]] = icmp slt i32 0, [[N]]
-; IR-NEXT: br i1 [[CMP1]], label %[[LOOP_EXIT:.*]], label %[[LOOP_BODY_LR_PH:.*]], !prof [[PROF6:![0-9]+]]
+; IR-NEXT: br i1 [[CMP1]], label %[[LOOP_EXIT:.*]], label %[[LOOP_BODY_LR_PH:.*]], !prof [[PROF14:![0-9]+]]
; IR: [[LOOP_BODY_LR_PH]]:
; IR-NEXT: br label %[[LOOP_BODY:.*]]
; IR: [[LOOP_BODY]]:
@@ -205,7 +205,7 @@ define void @func3_zero_branch_weight(i32 %n) !prof !3 {
; IR-NEXT: store volatile i32 [[I2]], ptr @g, align 4
; IR-NEXT: [[I_INC]] = add i32 [[I2]], 1
; IR-NEXT: [[CMP:%.*]] = icmp slt i32 [[I_INC]], [[N]]
-; IR-NEXT: br i1 [[CMP]], label %[[LOOP_HEADER_LOOP_EXIT_CRIT_EDGE:.*]], label %[[LOOP_BODY]], !prof [[PROF6]]
+; IR-NEXT: br i1 [[CMP]], label %[[LOOP_HEADER_LOOP_EXIT_CRIT_EDGE:.*]], label %[[LOOP_BODY]], !prof [[PROF14]]
; IR: [[LOOP_HEADER_LOOP_EXIT_CRIT_EDGE]]:
; IR-NEXT: br label %[[LOOP_EXIT]]
; IR: [[LOOP_EXIT]]:
@@ -230,10 +230,10 @@ loop_exit:
define void @func4_zero_branch_weight(i32 %n) !prof !3 {
; IR-LABEL: define void @func4_zero_branch_weight(
-; IR-SAME: i32 [[N:%.*]]) !prof [[PROF3]] {
+; IR-SAME: i32 [[N:%.*]]) !prof [[PROF7]] {
; IR-NEXT: [[ENTRY:.*:]]
; IR-NEXT: [[CMP1:%.*]] = icmp slt i32 0, [[N]]
-; IR-NEXT: br i1 [[CMP1]], label %[[LOOP_EXIT:.*]], label %[[LOOP_BODY_LR_PH:.*]], !prof [[PROF7:![0-9]+]]
+; IR-NEXT: br i1 [[CMP1]], label %[[LOOP_EXIT:.*]], label %[[LOOP_BODY_LR_PH:.*]], !prof [[PROF15:![0-9]+]]
; IR: [[LOOP_BODY_LR_PH]]:
; IR-NEXT: br label %[[LOOP_BODY:.*]]
; IR: [[LOOP_BODY]]:
@@ -241,7 +241,7 @@ define void @func4_zero_branch_weight(i32 %n) !prof !3 {
; IR-NEXT: store volatile i32 [[I2]], ptr @g, align 4
; IR-NEXT: [[I_INC]] = add i32 [[I2]], 1
; IR-NEXT: [[CMP:%.*]] = icmp slt i32 [[I_INC]], [[N]]
-; IR-NEXT: br i1 [[CMP]], label %[[LOOP_HEADER_LOOP_EXIT_CRIT_EDGE:.*]], label %[[LOOP_BODY]], !prof [[PROF7]]
+; IR-NEXT: br i1 [[CMP]], label %[[LOOP_HEADER_LOOP_EXIT_CRIT_EDGE:.*]], label %[[LOOP_BODY]], !prof [[PROF15]], !llvm.loop [[LOOP16:![0-9]+]]
; IR: [[LOOP_HEADER_LOOP_EXIT_CRIT_EDGE]]:
; IR-NEXT: br label %[[LOOP_EXIT]]
; IR: [[LOOP_EXIT]]:
@@ -266,10 +266,10 @@ loop_exit:
define void @func5_zero_branch_weight(i32 %n) !prof !3 {
; IR-LABEL: define void @func5_zero_branch_weight(
-; IR-SAME: i32 [[N:%.*]]) !prof [[PROF3]] {
+; IR-SAME: i32 [[N:%.*]]) !prof [[PROF7]] {
; IR-NEXT: [[ENTRY:.*:]]
; IR-NEXT: [[CMP1:%.*]] = icmp slt i32 0, [[N]]
-; IR-NEXT: br i1 [[CMP1]], label %[[LOOP_EXIT:.*]], label %[[LOOP_BODY_LR_PH:.*]], !prof [[PROF8:![0-9]+]]
+; IR-NEXT: br i1 [[CMP1]], label %[[LOOP_EXIT:.*]], label %[[LOOP_BODY_LR_PH:.*]], !prof [[PROF17:![0-9]+]]
; IR: [[LOOP_BODY_LR_PH]]:
; IR-NEXT: br label %[[LOOP_BODY:.*]]
; IR: [[LOOP_BODY]]:
@@ -277,7 +277,7 @@ define void @func5_zero_branch_weight(i32 %n) !prof !3 {
; IR-NEXT: store volatile i32 [[I2]], ptr @g, align 4
; IR-NEXT: [[I_INC]] = add i32 [[I2]], 1
; IR-NEXT: [[CMP:%.*]] = icmp slt i32 [[I_INC]], [[N]]
-; IR-NEXT: br i1 [[CMP]], label %[[LOOP_HEADER_LOOP_EXIT_CRIT_EDGE:.*]], label %[[LOOP_BODY]], !prof [[PROF8]]
+; IR-NEXT: br i1 [[CMP]], label %[[LOOP_HEADER_LOOP_EXIT_CRIT_EDGE:.*]], label %[[LOOP_BODY]], !prof [[PROF17]]
; IR: [[LOOP_HEADER_LOOP_EXIT_CRIT_EDGE]]:
; IR-NEXT: br label %[[LOOP_EXIT]]
; IR: [[LOOP_EXIT]]:
@@ -317,7 +317,7 @@ loop_exit:
; However this may not hold for Sample-based PGO.
define void @func6_inaccurate_branch_weight() !prof !3 {
; IR-LABEL: define void @func6_inaccurate_branch_weight(
-; IR-SAME: ) !prof [[PROF3]] {
+; IR-SAME: ) !prof [[PROF7]] {
; IR-NEXT: [[ENTRY:.*]]:
; IR-NEXT: br label %[[LOOP_BODY:.*]]
; IR: [[LOOP_BODY]]:
@@ -325,7 +325,7 @@ define void @func6_inaccurate_branch_weight() !prof !3 {
; IR-NEXT: store volatile i32 [[I1]], ptr @g, align 4
; IR-NEXT: [[I_INC]] = add i32 [[I1]], 1
; IR-NEXT: [[CMP:%.*]] = icmp slt i32 [[I_INC]], 2
-; IR-NEXT: br i1 [[CMP]], label %[[LOOP_BODY]], label %[[LOOP_EXIT:.*]], !prof [[PROF9:![0-9]+]]
+; IR-NEXT: br i1 [[CMP]], label %[[LOOP_BODY]], label %[[LOOP_EXIT:.*]], !prof [[PROF18:![0-9]+]], !llvm.loop [[LOOP19:![0-9]+]]
; IR: [[LOOP_EXIT]]:
; IR-NEXT: ret void
;
@@ -346,6 +346,102 @@ loop_exit:
ret void
}
+; Record the reduced trip count when rotation folds a single-exit guard.
+
+define void @folded_single_exit_inferred_trip_count() {
+; IR-LABEL: define void @folded_single_exit_inferred_trip_count() {
+; IR-NEXT: [[ENTRY:.*]]:
+; IR-NEXT: br label %[[LOOP_BODY:.*]]
+; IR: [[LOOP_BODY]]:
+; IR-NEXT: [[I1:%.*]] = phi i32 [ 0, %[[ENTRY]] ], [ [[I_INC:%.*]], %[[LOOP_BODY]] ]
+; IR-NEXT: store volatile i32 [[I1]], ptr @g, align 4
+; IR-NEXT: [[I_INC]] = add i32 [[I1]], 1
+; IR-NEXT: [[CMP:%.*]] = icmp slt i32 [[I_INC]], 10
+; IR-NEXT: br i1 [[CMP]], label %[[LOOP_BODY]], label %[[LOOP_EXIT:.*]], !prof [[PROF4]], !llvm.loop [[LOOP21:![0-9]+]]
+; IR: [[LOOP_EXIT]]:
+; IR-NEXT: ret void
+;
+entry:
+ br label %loop_header
+
+loop_header:
+ %i = phi i32 [ 0, %entry ], [ %i.inc, %loop_body ]
+ %cmp = icmp slt i32 %i, 10
+ br i1 %cmp, label %loop_body, label %loop_exit, !prof !1
+
+loop_body:
+ store volatile i32 %i, ptr @g, align 4
+ %i.inc = add i32 %i, 1
+ br label %loop_header
+
+loop_exit:
+ ret void
+}
+
+; Decrement an explicit trip count when rotation folds the guard.
+
+define void @folded_single_exit_explicit_trip_count() {
+; IR-LABEL: define void @folded_single_exit_explicit_trip_count() {
+; IR-NEXT: [[ENTRY:.*]]:
+; IR-NEXT: br label %[[LOOP_BODY:.*]]
+; IR: [[LOOP_BODY]]:
+; IR-NEXT: [[I1:%.*]] = phi i32 [ 0, %[[ENTRY]] ], [ [[I_INC:%.*]], %[[LOOP_BODY]] ]
+; IR-NEXT: store volatile i32 [[I1]], ptr @g, align 4
+; IR-NEXT: [[I_INC]] = add i32 [[I1]], 1
+; IR-NEXT: [[CMP:%.*]] = icmp slt i32 [[I_INC]], 10
+; IR-NEXT: br i1 [[CMP]], label %[[LOOP_BODY]], label %[[LOOP_EXIT:.*]], !prof [[PROF4]], !llvm.loop [[LOOP22:![0-9]+]]
+; IR: [[LOOP_EXIT]]:
+; IR-NEXT: ret void
+;
+entry:
+ br label %loop_header
+
+loop_header:
+ %i = phi i32 [ 0, %entry ], [ %i.inc, %loop_body ]
+ %cmp = icmp slt i32 %i, 10
+ br i1 %cmp, label %loop_body, label %loop_exit, !prof !1, !llvm.loop !10
+
+loop_body:
+ store volatile i32 %i, ptr @g, align 4
+ %i.inc = add i32 %i, 1
+ br label %loop_header
+
+loop_exit:
+ ret void
+}
+
+; Do not underflow an explicit zero trip count.
+
+define void @folded_single_exit_zero_trip_count() {
+; IR-LABEL: define void @folded_single_exit_zero_trip_count() {
+; IR-NEXT: [[ENTRY:.*]]:
+; IR-NEXT: br label %[[LOOP_BODY:.*]]
+; IR: [[LOOP_BODY]]:
+; IR-NEXT: [[I1:%.*]] = phi i32 [ 0, %[[ENTRY]] ], [ [[I_INC:%.*]], %[[LOOP_BODY]] ]
+; IR-NEXT: store volatile i32 [[I1]], ptr @g, align 4
+; IR-NEXT: [[I_INC]] = add i32 [[I1]], 1
+; IR-NEXT: [[CMP:%.*]] = icmp slt i32 [[I_INC]], 10
+; IR-NEXT: br i1 [[CMP]], label %[[LOOP_BODY]], label %[[LOOP_EXIT:.*]], !prof [[PROF4]], !llvm.loop [[LOOP24:![0-9]+]]
+; IR: [[LOOP_EXIT]]:
+; IR-NEXT: ret void
+;
+entry:
+ br label %loop_header
+
+loop_header:
+ %i = phi i32 [ 0, %entry ], [ %i.inc, %loop_body ]
+ %cmp = icmp slt i32 %i, 10
+ br i1 %cmp, label %loop_body, label %loop_exit, !prof !1, !llvm.loop !12
+
+loop_body:
+ store volatile i32 %i, ptr @g, align 4
+ %i.inc = add i32 %i, 1
+ br label %loop_header
+
+loop_exit:
+ ret void
+}
+
!0 = !{!"function_entry_count", i64 1}
!1 = !{!"branch_weights", i32 1000, i32 1}
!2 = !{!"branch_weights", i32 3000, i32 1000}
@@ -356,16 +452,35 @@ loop_exit:
!7 = !{!"branch_weights", i32 1, i32 0}
!8 = !{!"branch_weights", i32 0, i32 0}
!9 = !{!"branch_weights", i32 1023, i32 1024}
+!10 = distinct !{!10, !11}
+!11 = !{!"llvm.loop.estimated_trip_count", i32 10}
+!12 = distinct !{!12, !13}
+!13 = !{!"llvm.loop.estimated_trip_count", i32 0}
;.
; IR: [[PROF0]] = !{!"function_entry_count", i64 1}
; IR: [[PROF1]] = !{!"branch_weights", i32 2000, i32 1000}
-; IR: [[PROF2]] = !{!"branch_weights", i32 999, i32 1}
-; IR: [[PROF3]] = !{!"function_entry_count", i64 1024}
-; IR: [[PROF4]] = !{!"branch_weights", i32 40, i32 2}
-; IR: [[PROF5]] = !{!"branch_weights", i32 10240, i32 320}
-; IR: [[PROF6]] = !{!"branch_weights", i32 0, i32 1}
-; IR: [[PROF7]] = !{!"branch_weights", i32 1, i32 0}
-; IR: [[PROF8]] = !{!"branch_weights", i32 0, i32 0}
-; IR: [[PROF9]] = !{!"branch_weights", i32 0, i32 1024}
+; IR: [[LOOP2]] = distinct !{[[LOOP2]], [[META3:![0-9]+]]}
+; IR: [[META3]] = !{!"llvm.loop.estimated_trip_count", i32 3}
+; IR: [[PROF4]] = !{!"branch_weights", i32 999, i32 1}
+; IR: [[LOOP5]] = distinct !{[[LOOP5]], [[META6:![0-9]+]]}
+; IR: [[META6]] = !{!"llvm.loop.estimated_trip_count", i32 1000}
+; IR: [[PROF7]] = !{!"function_entry_count", i64 1024}
+; IR: [[PROF8]] = !{!"branch_weights", i32 40, i32 2}
+; IR: [[LOOP9]] = distinct !{[[LOOP9]], [[META10:![0-9]+]]}
+; IR: [[META10]] = !{!"llvm.loop.estimated_trip_count", i32 20}
+; IR: [[PROF11]] = !{!"branch_weights", i32 10240, i32 320}
+; IR: [[LOOP12]] = distinct !{[[LOOP12]], [[META13:![0-9]+]]}
+; IR: [[META13]] = !{!"llvm.loop.estimated_trip_count", i32 0}
+; IR: [[PROF14]] = !{!"branch_weights", i32 0, i32 1}
+; IR: [[PROF15]] = !{!"branch_weights", i32 1, i32 0}
+; IR: [[LOOP16]] = distinct !{[[LOOP16]], [[META13]]}
+; IR: [[PROF17]] = !{!"branch_weights", i32 0, i32 0}
+; IR: [[PROF18]] = !{!"branch_weights", i32 0, i32 1024}
+; IR: [[LOOP19]] = distinct !{[[LOOP19]], [[META20:![0-9]+]]}
+; IR: [[META20]] = !{!"llvm.loop.estimated_trip_count", i32 1}
+; IR: [[LOOP21]] = distinct !{[[LOOP21]], [[META6]]}
+; IR: [[LOOP22]] = distinct !{[[LOOP22]], [[META23:![0-9]+]]}
+; IR: [[META23]] = !{!"llvm.loop.estimated_trip_count", i32 9}
+; IR: [[LOOP24]] = distinct !{[[LOOP24]], [[META13]]}
;.
>From c7d982012a9fbaec9af59d959030f1ea5ea43bc9 Mon Sep 17 00:00:00 2001
From: Alok Kumar Sharma <AlokKumar.Sharma at amd.com>
Date: Wed, 19 Aug 2026 23:29:36 +0530
Subject: [PATCH 3/3] [LoopRotate] Derive folded-guard latch weights for
multi-exit loops from BFI
When rotation folds the first-iteration guard, the single-exit
latch-weight adjustment overestimates the backedge for multi-exit loops.
While block frequencies still reflect the pre-rotation layout, derive
latch weights from BFI:
exit = header frequency - body frequency
backedge = body frequency - preheader frequency
Fit those values onto the latch branch metadata so aggregate BFI is
preserved. If the frequencies are inconsistent, keep the original latch
weights instead of applying the single-exit fallback.
---
.../Transforms/Utils/LoopRotationUtils.cpp | 100 ++++++--
.../folded-guard-multi-exit-branch-weights.ll | 237 ++++++++++++++++++
...olded-guard-multi-exit-inconsistent-bfi.ll | 60 +++++
.../LoopRotate/multi-exit-branch-weights.ll | 77 +++++-
4 files changed, 447 insertions(+), 27 deletions(-)
create mode 100644 llvm/test/Transforms/LoopRotate/folded-guard-multi-exit-branch-weights.ll
create mode 100644 llvm/test/Transforms/LoopRotate/folded-guard-multi-exit-inconsistent-bfi.ll
diff --git a/llvm/lib/Transforms/Utils/LoopRotationUtils.cpp b/llvm/lib/Transforms/Utils/LoopRotationUtils.cpp
index 17fb32a1536df..86ee153e5f679 100644
--- a/llvm/lib/Transforms/Utils/LoopRotationUtils.cpp
+++ b/llvm/lib/Transforms/Utils/LoopRotationUtils.cpp
@@ -12,10 +12,12 @@
#include <optional>
-#include "llvm/Transforms/Utils/LoopRotationUtils.h"
#include "llvm/ADT/Statistic.h"
#include "llvm/Analysis/AssumptionCache.h"
+#include "llvm/Analysis/BlockFrequencyInfo.h"
+#include "llvm/Analysis/BranchProbabilityInfo.h"
#include "llvm/Analysis/CodeMetrics.h"
+#include "llvm/Analysis/CycleAnalysis.h"
#include "llvm/Analysis/DomTreeUpdater.h"
#include "llvm/Analysis/InstructionSimplify.h"
#include "llvm/Analysis/LoopInfo.h"
@@ -35,6 +37,7 @@
#include "llvm/Transforms/Utils/BasicBlockUtils.h"
#include "llvm/Transforms/Utils/Cloning.h"
#include "llvm/Transforms/Utils/Local.h"
+#include "llvm/Transforms/Utils/LoopRotationUtils.h"
#include "llvm/Transforms/Utils/LoopUtils.h"
#include "llvm/Transforms/Utils/SSAUpdater.h"
#include "llvm/Transforms/Utils/ValueMapper.h"
@@ -51,6 +54,11 @@ STATISTIC(NumInstrsDuplicated,
"Number of instructions cloned into loop preheader");
namespace {
+struct RotatedLatchWeights {
+ uint64_t Exit;
+ uint64_t Backedge;
+};
+
/// A simple loop rotation transformation.
class LoopRotate {
const unsigned MaxHeaderSize;
@@ -210,9 +218,38 @@ static bool profitableToRotateLoopExitingLatch(Loop *L, ScalarEvolution *SE) {
return false;
}
-static void updateBranchWeights(CondBrInst &PreHeaderBI, CondBrInst &LoopBI,
- bool HasConditionalPreHeader,
- bool SuccsSwapped) {
+// Rebuild local BFI before moveToHeader/edge splitting, while block
+// frequencies still reflect the pre-rotation layout. Only used for folded-guard
+// multi-exit loops, so the extra analysis cost is limited to that case.
+static std::optional<RotatedLatchWeights> getFoldedMultiExitLatchWeightsFromBFI(
+ Function &F, const SimplifyQuery &SQ, DominatorTree *DT,
+ BasicBlock *OrigPreheader, BasicBlock *OrigHeader, BasicBlock *NewHeader) {
+ CycleInfo CI;
+ CI.compute(F);
+ BranchProbabilityInfo BPI(F, CI, SQ.TLI, DT);
+ BlockFrequencyInfo BFI(F, BPI, CI);
+ uint64_t PreheaderFreq = BFI.getBlockFreq(OrigPreheader).getFrequency();
+ uint64_t HeaderFreq = BFI.getBlockFreq(OrigHeader).getFrequency();
+ uint64_t NewHeaderFreq = BFI.getBlockFreq(NewHeader).getFrequency();
+ if (HeaderFreq < NewHeaderFreq || NewHeaderFreq < PreheaderFreq) {
+ LLVM_DEBUG(dbgs() << "LoopRotation: inconsistent BFI for folded "
+ << "multi-exit latch weights\n");
+ return std::nullopt;
+ }
+ uint64_t ExitFreq = HeaderFreq - NewHeaderFreq;
+ uint64_t BackedgeFreq = NewHeaderFreq - PreheaderFreq;
+ if (!ExitFreq && !BackedgeFreq) {
+ LLVM_DEBUG(dbgs() << "LoopRotation: zero BFI-derived latch weights for "
+ << "folded multi-exit loop\n");
+ return std::nullopt;
+ }
+ return RotatedLatchWeights{ExitFreq, BackedgeFreq};
+}
+
+static void updateBranchWeights(
+ CondBrInst &PreHeaderBI, CondBrInst &LoopBI, bool HasConditionalPreHeader,
+ bool SuccsSwapped, bool IsMultiExitLoop,
+ std::optional<RotatedLatchWeights> MultiExitFoldedGuardWeights) {
MDNode *WeightMD = getBranchWeightMDNode(PreHeaderBI);
if (WeightMD == nullptr)
return;
@@ -243,6 +280,25 @@ static void updateBranchWeights(CondBrInst &PreHeaderBI, CondBrInst &LoopBI,
if (HasConditionalPreHeader)
return;
+ // Use BFI because header weights miss side exits.
+ if (MultiExitFoldedGuardWeights) {
+ const uint64_t LoopBIWeights[] = {
+ SuccsSwapped ? MultiExitFoldedGuardWeights->Backedge
+ : MultiExitFoldedGuardWeights->Exit,
+ SuccsSwapped ? MultiExitFoldedGuardWeights->Exit
+ : MultiExitFoldedGuardWeights->Backedge,
+ };
+ setFittedBranchWeights(LoopBI, LoopBIWeights, /*IsExpected=*/false);
+ return;
+ }
+
+ // Keep the original weights for multi-exit loops without usable BFI.
+ if (IsMultiExitLoop) {
+ LLVM_DEBUG(dbgs() << "LoopRotation: keeping original latch weights for "
+ << "multi-exit loop without usable BFI\n");
+ return;
+ }
+
SmallVector<uint32_t, 2> Weights;
extractFromBranchWeightMD32(WeightMD, Weights);
if (Weights.size() != 2)
@@ -402,6 +458,10 @@ bool LoopRotate::rotateLoop(Loop *L, bool SimplifiedLatch) {
assert(L->contains(NewHeader) && !L->contains(Exit) &&
"Unable to determine loop header and exit blocks");
+ // getExitingBlock() is null when there is no unique dedicated exit block.
+ bool IsMultiExitLoop = !L->getExitingBlock();
+ std::optional<RotatedLatchWeights> MultiExitFoldedGuardWeights;
+
// This code assumes that the new header has exactly one predecessor.
// Remove any single-entry PHI nodes in it.
assert(NewHeader->getSinglePredecessor() &&
@@ -522,7 +582,6 @@ bool LoopRotate::rotateLoop(Loop *L, bool SimplifiedLatch) {
mapAtomInstance(DL, ValueMap);
C->insertBefore(LoopEntryBranch->getIterator());
-
++NumInstrsDuplicated;
if (!NextDbgInsts.empty()) {
@@ -568,6 +627,21 @@ bool LoopRotate::rotateLoop(Loop *L, bool SimplifiedLatch) {
}
}
+ auto *PHBI = cast<CondBrInst>(ValueMap.lookup(BI));
+ const Value *Cond = PHBI->getCondition();
+ const bool HasConditionalPreHeader =
+ !isa<ConstantInt>(Cond) ||
+ PHBI->getSuccessor(cast<ConstantInt>(Cond)->isZero()) != NewHeader;
+
+ // Derive latch weights before moveToHeader/edge splitting, while block
+ // frequencies still reflect the pre-rotation layout.
+ // exit = header - body, backedge = body - preheader.
+ if (!HasConditionalPreHeader && IsMultiExitLoop &&
+ getBranchWeightMDNode(*BI)) {
+ MultiExitFoldedGuardWeights = getFoldedMultiExitLatchWeightsFromBFI(
+ *OrigHeader->getParent(), SQ, DT, OrigPreheader, OrigHeader, NewHeader);
+ }
+
if (!NoAliasDeclInstructions.empty()) {
// There are noalias scope declarations:
// (general):
@@ -695,24 +769,20 @@ bool LoopRotate::rotateLoop(Loop *L, bool SimplifiedLatch) {
// then we fold away the cond branch to an uncond branch. This simplifies the
// loop in cases important for nested loops, and it also means we don't have
// to split as many edges.
- CondBrInst *PHBI = cast<CondBrInst>(OrigPreheader->getTerminator());
- const Value *Cond = PHBI->getCondition();
- const bool HasConditionalPreHeader =
- !isa<ConstantInt>(Cond) ||
- PHBI->getSuccessor(cast<ConstantInt>(Cond)->isZero()) != NewHeader;
+ assert(PHBI == OrigPreheader->getTerminator() &&
+ "Unexpected cloned preheader branch");
// Save the trip count before updating branch weights. Only infer it from
- // profile weights for single-exit loops, but still read and decrement
- // explicit llvm.loop.estimated_trip_count on any loop.
- // getExitingBlock() is null when there is no unique dedicated exit block.
- bool IsMultiExitLoop = !L->getExitingBlock();
+ // profile weights for single-exit loops (see IsMultiExitLoop above), but
+ // still read and decrement explicit llvm.loop.estimated_trip_count.
bool HasExplicitEstimatedTripCount =
getOptionalIntLoopAttribute(L, LLVMLoopEstimatedTripCount).has_value();
std::optional<unsigned> EstimatedTripCount;
if (!IsMultiExitLoop || HasExplicitEstimatedTripCount)
EstimatedTripCount = getLoopEstimatedTripCount(L);
- updateBranchWeights(*PHBI, *BI, HasConditionalPreHeader, BISuccsSwapped);
+ updateBranchWeights(*PHBI, *BI, HasConditionalPreHeader, BISuccsSwapped,
+ IsMultiExitLoop, MultiExitFoldedGuardWeights);
if (HasConditionalPreHeader) {
// The conditional branch can't be folded, handle the general case.
diff --git a/llvm/test/Transforms/LoopRotate/folded-guard-multi-exit-branch-weights.ll b/llvm/test/Transforms/LoopRotate/folded-guard-multi-exit-branch-weights.ll
new file mode 100644
index 0000000000000..eafdd5df62829
--- /dev/null
+++ b/llvm/test/Transforms/LoopRotate/folded-guard-multi-exit-branch-weights.ll
@@ -0,0 +1,237 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --no-generate-body-for-unused-prefixes --version 6
+; RUN: opt < %s -passes=loop-rotate -S | FileCheck %s
+; RUN: opt < %s -passes='print<block-freq>' -disable-output 2>&1 | FileCheck %s --check-prefix=BFI-BEFORE
+; RUN: opt < %s -passes='loop(loop-rotate),print<block-freq>' -disable-output 2>&1 | FileCheck %s --check-prefix=BFI-AFTER
+;
+; The guard folds, making the first body iteration unconditional. Since %body
+; has another exit, use BFI instead of the single-exit adjustment.
+;
+; preheader frequency = 1200
+; header frequency = 10000
+; body frequency = 9800
+;
+; exit = header - body = 10000 - 9800 = 200
+; backedge = body - preheader = 9800 - 1200 = 8600
+; setFittedBranchWeights preserves the 200:8600 ratio in uint32 metadata;
+; IR may print the scaled values as large or negative i32 constants.
+;
+; Preserve BFI. Do not infer a trip count from weights that miss a side exit.
+
+define void @folded_guard_multi_exit(ptr %p, i32 %cond) !prof !0 {
+; CHECK-LABEL: define void @folded_guard_multi_exit(
+; CHECK-SAME: ptr [[P:%.*]], i32 [[COND:%.*]]) !prof [[PROF0:![0-9]+]] {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[ENTER:%.*]] = icmp ne i32 [[COND]], 0
+; CHECK-NEXT: br i1 [[ENTER]], label %[[PH:.*]], label %[[RET:.*]], !prof [[PROF1:![0-9]+]]
+; CHECK: [[PH]]:
+; CHECK-NEXT: br label %[[BODY:.*]]
+; CHECK: [[HEADER:.*]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i32 [ [[NEXT:%.*]], %[[BODY]] ]
+; CHECK-NEXT: [[DONE:%.*]] = icmp eq i32 [[IV]], -1
+; CHECK-NEXT: br i1 [[DONE]], label %[[EXIT1:.*]], label %[[BODY]], !prof [[PROF2:![0-9]+]]
+; CHECK: [[BODY]]:
+; CHECK-NEXT: [[NEXT]] = load i32, ptr [[P]], align 4
+; CHECK-NEXT: [[LEAVE:%.*]] = icmp eq i32 [[NEXT]], -2
+; CHECK-NEXT: br i1 [[LEAVE]], label %[[EXIT2:.*]], label %[[HEADER]], !prof [[PROF3:![0-9]+]]
+; CHECK: [[EXIT1]]:
+; CHECK-NEXT: ret void
+; CHECK: [[EXIT2]]:
+; CHECK-NEXT: ret void
+; CHECK: [[RET]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ %enter = icmp ne i32 %cond, 0
+ br i1 %enter, label %ph, label %ret, !prof !1
+
+ph:
+ br label %header
+
+header:
+ %iv = phi i32 [ 0, %ph ], [ %next, %latch ]
+ %done = icmp eq i32 %iv, -1
+ br i1 %done, label %exit1, label %body, !prof !2
+
+body:
+ %next = load i32, ptr %p, align 4
+ %leave = icmp eq i32 %next, -2
+ br i1 %leave, label %exit2, label %latch, !prof !3
+
+latch:
+ br label %header
+
+exit1:
+ ret void
+
+exit2:
+ ret void
+
+ret:
+ ret void
+}
+
+; An explicit trip-count estimate remains authoritative when the guard folds.
+define void @folded_guard_multi_exit_explicit_tc(ptr %p, i32 %cond) !prof !0 {
+; CHECK-LABEL: define void @folded_guard_multi_exit_explicit_tc(
+; CHECK-SAME: ptr [[P:%.*]], i32 [[COND:%.*]]) !prof [[PROF0]] {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[ENTER:%.*]] = icmp ne i32 [[COND]], 0
+; CHECK-NEXT: br i1 [[ENTER]], label %[[PH:.*]], label %[[RET:.*]], !prof [[PROF1]]
+; CHECK: [[PH]]:
+; CHECK-NEXT: br label %[[BODY:.*]]
+; CHECK: [[HEADER:.*]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i32 [ [[NEXT:%.*]], %[[BODY]] ]
+; CHECK-NEXT: [[DONE:%.*]] = icmp eq i32 [[IV]], -1
+; CHECK-NEXT: br i1 [[DONE]], label %[[EXIT1:.*]], label %[[BODY]], !prof [[PROF2]], !llvm.loop [[LOOP4:![0-9]+]]
+; CHECK: [[BODY]]:
+; CHECK-NEXT: [[NEXT]] = load i32, ptr [[P]], align 4
+; CHECK-NEXT: [[LEAVE:%.*]] = icmp eq i32 [[NEXT]], -2
+; CHECK-NEXT: br i1 [[LEAVE]], label %[[EXIT2:.*]], label %[[HEADER]], !prof [[PROF3]]
+; CHECK: [[EXIT1]]:
+; CHECK-NEXT: ret void
+; CHECK: [[EXIT2]]:
+; CHECK-NEXT: ret void
+; CHECK: [[RET]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ %enter = icmp ne i32 %cond, 0
+ br i1 %enter, label %ph, label %ret, !prof !1
+
+ph:
+ br label %header
+
+header:
+ %iv = phi i32 [ 0, %ph ], [ %next, %latch ]
+ %done = icmp eq i32 %iv, -1
+ br i1 %done, label %exit1, label %body, !prof !2, !llvm.loop !4
+
+body:
+ %next = load i32, ptr %p, align 4
+ %leave = icmp eq i32 %next, -2
+ br i1 %leave, label %exit2, label %latch, !prof !3
+
+latch:
+ br label %header
+
+exit1:
+ ret void
+
+exit2:
+ ret void
+
+ret:
+ ret void
+}
+
+; Exercise the same folded multi-exit case with the header successors and
+; weights reversed.
+define void @folded_guard_multi_exit_swapped(ptr %p, i32 %cond) !prof !0 {
+; CHECK-LABEL: define void @folded_guard_multi_exit_swapped(
+; CHECK-SAME: ptr [[P:%.*]], i32 [[COND:%.*]]) !prof [[PROF0]] {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[ENTER:%.*]] = icmp ne i32 [[COND]], 0
+; CHECK-NEXT: br i1 [[ENTER]], label %[[PH:.*]], label %[[RET:.*]], !prof [[PROF1]]
+; CHECK: [[PH]]:
+; CHECK-NEXT: br label %[[BODY:.*]]
+; CHECK: [[HEADER:.*]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i32 [ [[NEXT:%.*]], %[[BODY]] ]
+; CHECK-NEXT: [[CONTINUE:%.*]] = icmp ne i32 [[IV]], -1
+; CHECK-NEXT: br i1 [[CONTINUE]], label %[[BODY]], label %[[EXIT1:.*]], !prof [[PROF6:![0-9]+]]
+; CHECK: [[BODY]]:
+; CHECK-NEXT: [[NEXT]] = load i32, ptr [[P]], align 4
+; CHECK-NEXT: [[LEAVE:%.*]] = icmp eq i32 [[NEXT]], -2
+; CHECK-NEXT: br i1 [[LEAVE]], label %[[EXIT2:.*]], label %[[HEADER]], !prof [[PROF3]]
+; CHECK: [[EXIT1]]:
+; CHECK-NEXT: ret void
+; CHECK: [[EXIT2]]:
+; CHECK-NEXT: ret void
+; CHECK: [[RET]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ %enter = icmp ne i32 %cond, 0
+ br i1 %enter, label %ph, label %ret, !prof !1
+
+ph:
+ br label %header
+
+header:
+ %iv = phi i32 [ 0, %ph ], [ %next, %latch ]
+ %continue = icmp ne i32 %iv, -1
+ br i1 %continue, label %body, label %exit1, !prof !6
+
+body:
+ %next = load i32, ptr %p, align 4
+ %leave = icmp eq i32 %next, -2
+ br i1 %leave, label %exit2, label %latch, !prof !3
+
+latch:
+ br label %header
+
+exit1:
+ ret void
+
+exit2:
+ ret void
+
+ret:
+ ret void
+}
+
+!0 = !{!"function_entry_count", i64 1201}
+!1 = !{!"branch_weights", i32 1200, i32 1}
+!2 = !{!"branch_weights", i32 200, i32 9800}
+!3 = !{!"branch_weights", i32 1000, i32 8800}
+!4 = distinct !{!4, !5}
+!5 = !{!"llvm.loop.estimated_trip_count", i32 10}
+!6 = !{!"branch_weights", i32 9800, i32 200}
+
+; BFI-BEFORE-LABEL: block-frequency-info: folded_guard_multi_exit
+; BFI-BEFORE: - body: {{.*}} count = 9800
+; BFI-BEFORE: - exit1: {{.*}} count = 200
+; BFI-BEFORE: - exit2: {{.*}} count = 1000
+; BFI-BEFORE: - ret: {{.*}} count = 1
+
+; BFI-BEFORE-LABEL: block-frequency-info: folded_guard_multi_exit_explicit_tc
+; BFI-BEFORE: - body: {{.*}} count = 9800
+; BFI-BEFORE: - exit1: {{.*}} count = 200
+; BFI-BEFORE: - exit2: {{.*}} count = 1000
+; BFI-BEFORE: - ret: {{.*}} count = 1
+
+; BFI-BEFORE-LABEL: block-frequency-info: folded_guard_multi_exit_swapped
+; BFI-BEFORE: - body: {{.*}} count = 9800
+; BFI-BEFORE: - exit1: {{.*}} count = 200
+; BFI-BEFORE: - exit2: {{.*}} count = 1000
+; BFI-BEFORE: - ret: {{.*}} count = 1
+
+; BFI-AFTER-LABEL: block-frequency-info: folded_guard_multi_exit
+; BFI-AFTER: - body: {{.*}} count = 9800
+; BFI-AFTER: - exit1: {{.*}} count = 200
+; BFI-AFTER: - exit2: {{.*}} count = 1000
+; BFI-AFTER: - ret: {{.*}} count = 1
+
+; BFI-AFTER-LABEL: block-frequency-info: folded_guard_multi_exit_explicit_tc
+; BFI-AFTER: - body: {{.*}} count = 9800
+; BFI-AFTER: - exit1: {{.*}} count = 200
+; BFI-AFTER: - exit2: {{.*}} count = 1000
+; BFI-AFTER: - ret: {{.*}} count = 1
+
+; BFI-AFTER-LABEL: block-frequency-info: folded_guard_multi_exit_swapped
+; BFI-AFTER: - body: {{.*}} count = 9800
+; BFI-AFTER: - exit1: {{.*}} count = 200
+; BFI-AFTER: - exit2: {{.*}} count = 1000
+; BFI-AFTER: - ret: {{.*}} count = 1
+
+
+
+
+;.
+; CHECK: [[PROF0]] = !{!"function_entry_count", i64 1201}
+; CHECK: [[PROF1]] = !{!"branch_weights", i32 1200, i32 1}
+; CHECK: [[PROF2]] = !{!"branch_weights", i32 85899346, i32 -601295421}
+; CHECK: [[PROF3]] = !{!"branch_weights", i32 1000, i32 8800}
+; CHECK: [[LOOP4]] = distinct !{[[LOOP4]], [[META5:![0-9]+]]}
+; CHECK: [[META5]] = !{!"llvm.loop.estimated_trip_count", i32 9}
+; CHECK: [[PROF6]] = !{!"branch_weights", i32 -601295421, i32 85899346}
+;.
diff --git a/llvm/test/Transforms/LoopRotate/folded-guard-multi-exit-inconsistent-bfi.ll b/llvm/test/Transforms/LoopRotate/folded-guard-multi-exit-inconsistent-bfi.ll
new file mode 100644
index 0000000000000..7d9ed1c579d4e
--- /dev/null
+++ b/llvm/test/Transforms/LoopRotate/folded-guard-multi-exit-inconsistent-bfi.ll
@@ -0,0 +1,60 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --no-generate-body-for-unused-prefixes --version 6
+; RUN: opt < %s -passes=loop-rotate -S | FileCheck %s
+; RUN: opt < %s -passes=loop-rotate -S | not grep llvm.loop.estimated_trip_count
+;
+; The profile says %body runs less often than the preheader. Keep the original
+; weights because BFI is inconsistent.
+
+define void @inconsistent_multi_exit(ptr %p) {
+; CHECK-LABEL: define void @inconsistent_multi_exit(
+; CHECK-SAME: ptr [[P:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: br label %[[PH:.*]]
+; CHECK: [[PH]]:
+; CHECK-NEXT: br label %[[BODY:.*]]
+; CHECK: [[HEADER:.*]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i32 [ [[NEXT:%.*]], %[[BODY]] ]
+; CHECK-NEXT: [[DONE:%.*]] = icmp eq i32 [[IV]], -1
+; CHECK-NEXT: br i1 [[DONE]], label %[[EXIT1:.*]], label %[[BODY]], !prof [[PROF0:![0-9]+]]
+; CHECK: [[BODY]]:
+; CHECK-NEXT: [[NEXT]] = load i32, ptr [[P]], align 4
+; CHECK-NEXT: [[LEAVE:%.*]] = icmp eq i32 [[NEXT]], -2
+; CHECK-NEXT: br i1 [[LEAVE]], label %[[EXIT2:.*]], label %[[HEADER]], !prof [[PROF1:![0-9]+]]
+; CHECK: [[EXIT1]]:
+; CHECK-NEXT: ret void
+; CHECK: [[EXIT2]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %ph
+
+ph:
+ br label %header
+
+header:
+ %iv = phi i32 [ 0, %ph ], [ %next, %latch ]
+ %done = icmp eq i32 %iv, -1
+ br i1 %done, label %exit1, label %body, !prof !0
+
+body:
+ %next = load i32, ptr %p, align 4
+ %leave = icmp eq i32 %next, -2
+ br i1 %leave, label %exit2, label %latch, !prof !1
+
+latch:
+ br label %header
+
+exit1:
+ ret void
+
+exit2:
+ ret void
+}
+
+!0 = !{!"branch_weights", i32 90, i32 10}
+!1 = !{!"branch_weights", i32 1, i32 9}
+
+;.
+; CHECK: [[PROF0]] = !{!"branch_weights", i32 90, i32 10}
+; CHECK: [[PROF1]] = !{!"branch_weights", i32 1, i32 9}
+;.
diff --git a/llvm/test/Transforms/LoopRotate/multi-exit-branch-weights.ll b/llvm/test/Transforms/LoopRotate/multi-exit-branch-weights.ll
index 5e24888abbdaa..dd967fcba20b5 100644
--- a/llvm/test/Transforms/LoopRotate/multi-exit-branch-weights.ll
+++ b/llvm/test/Transforms/LoopRotate/multi-exit-branch-weights.ll
@@ -1,3 +1,4 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --no-generate-body-for-unused-prefixes --version 6
; RUN: opt < %s -passes=loop-rotate -S | FileCheck %s
; RUN: opt < %s -passes='print<block-freq>' -disable-output 2>&1 | FileCheck %s --check-prefix=BFI-BEFORE
; RUN: opt < %s -passes='loop(loop-rotate),print<block-freq>' -disable-output 2>&1 | FileCheck %s --check-prefix=BFI-AFTER
@@ -14,6 +15,36 @@
; latch ---------> header
define void @f(ptr %p, i32 %start, i32 %limit, i32 %cond) !prof !3 {
+; CHECK-LABEL: define void @f(
+; CHECK-SAME: ptr [[P:%.*]], i32 [[START:%.*]], i32 [[LIMIT:%.*]], i32 [[COND:%.*]]) !prof [[PROF0:![0-9]+]] {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[C0:%.*]] = icmp ne i32 [[COND]], 0
+; CHECK-NEXT: br i1 [[C0]], label %[[PH:.*]], label %[[RET:.*]], !prof [[PROF1:![0-9]+]]
+; CHECK: [[PH]]:
+; CHECK-NEXT: [[HCMP1:%.*]] = icmp eq i32 [[START]], [[LIMIT]]
+; CHECK-NEXT: br i1 [[HCMP1]], label %[[EXIT1:.*]], label %[[BODY_LR_PH:.*]], !prof [[PROF2:![0-9]+]]
+; CHECK: [[BODY_LR_PH]]:
+; CHECK-NEXT: br label %[[BODY:.*]]
+; CHECK: [[HEADER:.*]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i32 [ [[IV_NEXT:%.*]], %[[BODY]] ]
+; CHECK-NEXT: [[HCMP:%.*]] = icmp eq i32 [[IV]], [[LIMIT]]
+; CHECK-NEXT: br i1 [[HCMP]], label %[[HEADER_EXIT1_CRIT_EDGE:.*]], label %[[BODY]], !prof [[PROF2]]
+; CHECK: [[BODY]]:
+; CHECK-NEXT: [[IV2:%.*]] = phi i32 [ [[START]], %[[BODY_LR_PH]] ], [ [[IV]], %[[HEADER]] ]
+; CHECK-NEXT: [[ADDR:%.*]] = getelementptr i32, ptr [[P]], i32 [[IV2]]
+; CHECK-NEXT: [[V:%.*]] = load i32, ptr [[ADDR]], align 4
+; CHECK-NEXT: [[BCMP:%.*]] = icmp slt i32 [[V]], 0
+; CHECK-NEXT: [[IV_NEXT]] = add i32 [[IV2]], 1
+; CHECK-NEXT: br i1 [[BCMP]], label %[[EXIT2:.*]], label %[[HEADER]], !prof [[PROF3:![0-9]+]]
+; CHECK: [[HEADER_EXIT1_CRIT_EDGE]]:
+; CHECK-NEXT: br label %[[EXIT1]]
+; CHECK: [[EXIT1]]:
+; CHECK-NEXT: ret void
+; CHECK: [[EXIT2]]:
+; CHECK-NEXT: ret void
+; CHECK: [[RET]]:
+; CHECK-NEXT: ret void
+;
entry:
%c0 = icmp ne i32 %cond, 0
br i1 %c0, label %ph, label %ret, !prof !0
@@ -50,6 +81,30 @@ ret: ; preds = %entry
; Keep explicit trip-count metadata for multi-exit loops. Rotation decrements
; the attribute but does not infer a trip count from branch weights without it.
define void @explicit_trip_count(ptr %p, i32 %start, i32 %limit) {
+; CHECK-LABEL: define void @explicit_trip_count(
+; CHECK-SAME: ptr [[P:%.*]], i32 [[START:%.*]], i32 [[LIMIT:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[HCMP1:%.*]] = icmp eq i32 [[START]], [[LIMIT]]
+; CHECK-NEXT: br i1 [[HCMP1]], label %[[EXIT1:.*]], label %[[BODY_LR_PH:.*]], !prof [[PROF2]], !llvm.loop [[LOOP4:![0-9]+]]
+; CHECK: [[BODY_LR_PH]]:
+; CHECK-NEXT: br label %[[BODY:.*]], !llvm.loop [[LOOP4]]
+; CHECK: [[HEADER:.*]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i32 [ [[IV_NEXT:%.*]], %[[BODY]] ]
+; CHECK-NEXT: [[HCMP:%.*]] = icmp eq i32 [[IV]], [[LIMIT]]
+; CHECK-NEXT: br i1 [[HCMP]], label %[[HEADER_EXIT1_CRIT_EDGE:.*]], label %[[BODY]], !prof [[PROF2]], !llvm.loop [[LOOP6:![0-9]+]]
+; CHECK: [[BODY]]:
+; CHECK-NEXT: [[IV2:%.*]] = phi i32 [ [[START]], %[[BODY_LR_PH]] ], [ [[IV]], %[[HEADER]] ]
+; CHECK-NEXT: [[V:%.*]] = load i32, ptr [[P]], align 4
+; CHECK-NEXT: [[BCMP:%.*]] = icmp slt i32 [[V]], 0
+; CHECK-NEXT: [[IV_NEXT]] = add i32 [[IV2]], 1
+; CHECK-NEXT: br i1 [[BCMP]], label %[[EXIT2:.*]], label %[[HEADER]], !prof [[PROF3]]
+; CHECK: [[HEADER_EXIT1_CRIT_EDGE]]:
+; CHECK-NEXT: br label %[[EXIT1]], !llvm.loop [[LOOP4]]
+; CHECK: [[EXIT1]]:
+; CHECK-NEXT: ret void
+; CHECK: [[EXIT2]]:
+; CHECK-NEXT: ret void
+;
entry:
br label %header
@@ -93,15 +148,13 @@ exit2:
; BFI-AFTER: - exit2: {{.*}} count = 2
; BFI-AFTER: - ret: {{.*}} count = 2
-; CHECK-LABEL: define void @f(
-
-; CHECK: ph:
-; CHECK: br i1 %{{.*}}, label %{{.*}}, label %{{.*}}.lr.ph, !prof [[WEIGHTS:![0-9]+]]
-
-; CHECK: br i1 %{{.*}}, label %{{.*}}, label %body, !prof [[WEIGHTS]]{{$}}
-
-; CHECK-LABEL: define void @explicit_trip_count(
-; CHECK: br i1 %{{.*}}, label %{{.*}}, label %body, !prof {{![0-9]+}}, !llvm.loop [[EXPLICIT_LOOP:![0-9]+]]
-; CHECK-DAG: [[WEIGHTS]] = !{!"branch_weights", i32 200, i32 9800}
-; CHECK-DAG: [[EXPLICIT_LOOP]] = distinct !{[[EXPLICIT_LOOP]], [[EXPLICIT_TC:![0-9]+]]}
-; CHECK-DAG: [[EXPLICIT_TC]] = !{!"llvm.loop.estimated_trip_count", i32 9}
+;.
+; CHECK: [[PROF0]] = !{!"function_entry_count", i64 44}
+; CHECK: [[PROF1]] = !{!"branch_weights", i32 42, i32 2}
+; CHECK: [[PROF2]] = !{!"branch_weights", i32 200, i32 9800}
+; CHECK: [[PROF3]] = !{!"branch_weights", i32 10, i32 9790}
+; CHECK: [[LOOP4]] = distinct !{[[LOOP4]], [[META5:![0-9]+]]}
+; CHECK: [[META5]] = !{!"llvm.loop.estimated_trip_count", i32 10}
+; CHECK: [[LOOP6]] = distinct !{[[LOOP6]], [[META7:![0-9]+]]}
+; CHECK: [[META7]] = !{!"llvm.loop.estimated_trip_count", i32 9}
+;.
More information about the llvm-commits
mailing list