[llvm] [LoopRotate] Fix branch weights for rotated multi-exit loops (PR #202219)

Alok Kumar Sharma via llvm-commits llvm-commits at lists.llvm.org
Wed Aug 19 11:01:26 PDT 2026


https://github.com/alokkrsharma updated https://github.com/llvm/llvm-project/pull/202219

>From 22a21d2def90415d296974554194c231be774ef8 Mon Sep 17 00:00:00 2001
From: Alok Kumar Sharma <AlokKumar.Sharma at amd.com>
Date: Wed, 19 Aug 2026 23:28:52 +0530
Subject: [PATCH 1/3] [LoopRotate] Preserve header weights when the preheader
 guard stays conditional

When loop rotation leaves a conditional preheader guard, the guard and
latch are copies of the original header branch. Stop redistributing the
header weights across the guard and latch; keep the original {exit,
backedge} ratio on both branches instead.

This preserves block frequencies for loops whose guard remains
conditional after rotation, including multi-exit loops with a conditional
guard. Remove the zero-trip-count heuristics that drove the old
redistribution.

When the preheader guard folds away, keep the existing single-exit latch
adjustment unchanged.
---
 .../Transforms/Utils/LoopRotationUtils.cpp    | 131 ++++-------
 .../LoopRotate/multi-exit-branch-weights.ll   |  87 ++++++++
 .../never-entered-branch-weights.ll           |  78 +++++++
 .../predecessor-no-prof-branch-weights.ll     |  84 ++++++++
 .../saturating-backedge-branch-weights.ll     |  96 +++++++++
 .../scaled-preheader-branch-weights.ll        |  60 ++++++
 .../LoopRotate/update-branch-weights.ll       | 204 ++++++++++++------
 7 files changed, 581 insertions(+), 159 deletions(-)
 create mode 100644 llvm/test/Transforms/LoopRotate/multi-exit-branch-weights.ll
 create mode 100644 llvm/test/Transforms/LoopRotate/never-entered-branch-weights.ll
 create mode 100644 llvm/test/Transforms/LoopRotate/predecessor-no-prof-branch-weights.ll
 create mode 100644 llvm/test/Transforms/LoopRotate/saturating-backedge-branch-weights.ll
 create mode 100644 llvm/test/Transforms/LoopRotate/scaled-preheader-branch-weights.ll

diff --git a/llvm/lib/Transforms/Utils/LoopRotationUtils.cpp b/llvm/lib/Transforms/Utils/LoopRotationUtils.cpp
index c8bc5e4daeff3..7950e223bfb4d 100644
--- a/llvm/lib/Transforms/Utils/LoopRotationUtils.cpp
+++ b/llvm/lib/Transforms/Utils/LoopRotationUtils.cpp
@@ -46,9 +46,6 @@ STATISTIC(NumInstrsHoisted,
 STATISTIC(NumInstrsDuplicated,
           "Number of instructions cloned into loop preheader");
 
-// Probability that a rotated loop has zero trip count / is never entered.
-static constexpr uint32_t ZeroTripCountWeights[] = {1, 127};
-
 namespace {
 /// A simple loop rotation transformation.
 class LoopRotate {
@@ -222,6 +219,26 @@ static void updateBranchWeights(CondBrInst &PreHeaderBI, CondBrInst &LoopBI,
   if (WeightMD != getBranchWeightMDNode(LoopBI))
     return;
 
+  // A conditional copied guard and latch use the same weights {x, y}:
+  //
+  //    |  |--------             |
+  //    V  V       |             V
+  //   Br {x, y}   |            Br {x, y}
+  //   |       |   |            |     |
+  //  x|      y|   |  becomes: x|    y|  |------
+  //   V       V   |            |     V  V     |
+  // Exit    Loop  |            |     Loop     |
+  //           |   |            |    Br {x, y} |
+  //           -----            |    |     |   |
+  //                          x |  x |    y|   |
+  //                            V    V     -----
+  //                             Exit
+  //
+  // This preserves block frequencies, including multi-exit loops when the
+  // guard stays conditional.
+  if (HasConditionalPreHeader)
+    return;
+
   SmallVector<uint32_t, 2> Weights;
   extractFromBranchWeightMD32(WeightMD, Weights);
   if (Weights.size() != 2)
@@ -232,108 +249,32 @@ static void updateBranchWeights(CondBrInst &PreHeaderBI, CondBrInst &LoopBI,
   if (SuccsSwapped)
     std::swap(OrigLoopExitWeight, OrigLoopBackedgeWeight);
 
-  // Update branch weights. Consider the following edge-counts:
-  //
-  //    |  |--------             |
-  //    V  V       |             V
-  //   Br i1 ...   |            Br i1 ...
-  //   |       |   |            |     |
-  //  x|      y|   |  becomes:  |   y0|  |-----
-  //   V       V   |            |     V  V    |
-  // Exit    Loop  |            |    Loop     |
-  //           |   |            |   Br i1 ... |
-  //           -----            |   |      |  |
-  //                          x0| x1|   y1 |  |
-  //                            V   V      ----
-  //                            Exit
-  //
-  // The following must hold:
-  //  -  x == x0 + x1        # counts to "exit" must stay the same.
-  //  - y0 == x - x0 == x1   # how often loop was entered at all.
-  //  - y1 == y - y0         # How often loop was repeated (after first iter.).
-  //
-  // We cannot generally deduce how often we had a zero-trip count loop so we
-  // have to make a guess for how to distribute x among the new x0 and x1.
-
-  uint32_t ExitWeight0;    // aka x0
-  uint32_t ExitWeight1;    // aka x1
-  uint32_t EnterWeight;    // aka y0
-  uint32_t LoopBackWeight; // aka y1
+  // The first iteration is now unconditional. Keep the historical single-exit
+  // latch adjustment. This does not account for side exits on multi-exit
+  // loops when the guard folds away.
+  uint32_t ExitWeight;
+  uint32_t LoopBackWeight;
   if (OrigLoopExitWeight > 0 && OrigLoopBackedgeWeight > 0) {
-    ExitWeight0 = 0;
-    if (HasConditionalPreHeader) {
-      // Here we cannot know how many 0-trip count loops we have, so we guess:
-      if (OrigLoopBackedgeWeight >= OrigLoopExitWeight) {
-        // If the loop count is bigger than the exit count then we set
-        // probabilities as if 0-trip count nearly never happens.
-        ExitWeight0 = ZeroTripCountWeights[0];
-        // Scale up counts if necessary so we can match `ZeroTripCountWeights`
-        // for the `ExitWeight0`:`ExitWeight1` (aka `x0`:`x1` ratio`) ratio.
-        while (OrigLoopExitWeight < ZeroTripCountWeights[1] + ExitWeight0) {
-          // ... but don't overflow.
-          uint32_t const HighBit = uint32_t{1} << (sizeof(uint32_t) * 8 - 1);
-          if ((OrigLoopBackedgeWeight & HighBit) != 0 ||
-              (OrigLoopExitWeight & HighBit) != 0)
-            break;
-          OrigLoopBackedgeWeight <<= 1;
-          OrigLoopExitWeight <<= 1;
-        }
-      } else {
-        // If there's a higher exit-count than backedge-count then we set
-        // probabilities as if there are only 0-trip and 1-trip cases.
-        ExitWeight0 = OrigLoopExitWeight - OrigLoopBackedgeWeight;
-      }
-    } else {
-      // Theoretically, if the loop body must be executed at least once, the
-      // backedge count must be not less than exit count. However the branch
-      // weight collected by sampling-based PGO may be not very accurate due to
-      // sampling. Therefore this workaround is required here to avoid underflow
-      // of unsigned in following update of branch weight.
-      if (OrigLoopExitWeight > OrigLoopBackedgeWeight)
-        OrigLoopBackedgeWeight = OrigLoopExitWeight;
-    }
-    assert(OrigLoopExitWeight >= ExitWeight0 && "Bad branch weight");
-    ExitWeight1 = OrigLoopExitWeight - ExitWeight0;
-    EnterWeight = ExitWeight1;
-    assert(OrigLoopBackedgeWeight >= EnterWeight && "Bad branch weight");
-    LoopBackWeight = OrigLoopBackedgeWeight - EnterWeight;
+    // Sampling can report fewer backedges than exits. Clamp to avoid underflow.
+    if (OrigLoopExitWeight > OrigLoopBackedgeWeight)
+      OrigLoopBackedgeWeight = OrigLoopExitWeight;
+    ExitWeight = OrigLoopExitWeight;
+    LoopBackWeight = OrigLoopBackedgeWeight - OrigLoopExitWeight;
   } else if (OrigLoopExitWeight == 0) {
-    if (OrigLoopBackedgeWeight == 0) {
-      // degenerate case... keep everything zero...
-      ExitWeight0 = 0;
-      ExitWeight1 = 0;
-      EnterWeight = 0;
-      LoopBackWeight = 0;
-    } else {
-      // Special case "LoopExitWeight == 0" weights which behaves like an
-      // endless where we don't want loop-enttry (y0) to be the same as
-      // loop-exit (x1).
-      ExitWeight0 = 0;
-      ExitWeight1 = 0;
-      EnterWeight = 1;
-      LoopBackWeight = OrigLoopBackedgeWeight;
-    }
+    ExitWeight = 0;
+    LoopBackWeight = OrigLoopBackedgeWeight;
   } else {
-    // loop is never entered.
+    // Loop is never entered.
     assert(OrigLoopBackedgeWeight == 0 && "remaining case is backedge zero");
-    ExitWeight0 = 1;
-    ExitWeight1 = 1;
-    EnterWeight = 0;
+    ExitWeight = 1;
     LoopBackWeight = 0;
   }
 
   const uint32_t LoopBIWeights[] = {
-      SuccsSwapped ? LoopBackWeight : ExitWeight1,
-      SuccsSwapped ? ExitWeight1 : LoopBackWeight,
+      SuccsSwapped ? LoopBackWeight : ExitWeight,
+      SuccsSwapped ? ExitWeight : LoopBackWeight,
   };
   setBranchWeights(LoopBI, LoopBIWeights, /*IsExpected=*/false);
-  if (HasConditionalPreHeader) {
-    const uint32_t PreHeaderBIWeights[] = {
-        SuccsSwapped ? EnterWeight : ExitWeight0,
-        SuccsSwapped ? ExitWeight0 : EnterWeight,
-    };
-    setBranchWeights(PreHeaderBI, PreHeaderBIWeights, /*IsExpected=*/false);
-  }
 }
 
 /// Rotate loop LP. Return true if the loop is rotated.
diff --git a/llvm/test/Transforms/LoopRotate/multi-exit-branch-weights.ll b/llvm/test/Transforms/LoopRotate/multi-exit-branch-weights.ll
new file mode 100644
index 0000000000000..ac47fc94b4bf5
--- /dev/null
+++ b/llvm/test/Transforms/LoopRotate/multi-exit-branch-weights.ll
@@ -0,0 +1,87 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --no-generate-body-for-unused-prefixes --version 6
+; RUN: opt < %s -passes=loop-rotate -S | FileCheck %s
+;
+; The copied guard and latch run the same test. Keep the original weights on
+; both, including loops with another body exit.
+;
+; Entry weights use another scale because each branch stores a local ratio.
+;
+;   entry --(c0)--> ph        !prof { 42, 2 }
+;   ph    --------> header
+;   header --(hcmp)--> exit1 / body   !prof { 200, 9800 }
+;   body   --(bcmp)--> exit2 / latch  !prof { 10, 9790 }
+;   latch  ---------> header
+
+define void @f(ptr %p, i32 %start, i32 %limit, i32 %cond) {
+; CHECK-LABEL: define void @f(
+; CHECK-SAME: ptr [[P:%.*]], i32 [[START:%.*]], i32 [[LIMIT:%.*]], i32 [[COND:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[C0:%.*]] = icmp ne i32 [[COND]], 0
+; CHECK-NEXT:    br i1 [[C0]], label %[[PH:.*]], label %[[RET:.*]], !prof [[PROF0:![0-9]+]]
+; CHECK:       [[PH]]:
+; CHECK-NEXT:    [[HCMP1:%.*]] = icmp eq i32 [[START]], [[LIMIT]]
+; CHECK-NEXT:    br i1 [[HCMP1]], label %[[EXIT1:.*]], label %[[BODY_LR_PH:.*]], !prof [[PROF1:![0-9]+]]
+; CHECK:       [[BODY_LR_PH]]:
+; CHECK-NEXT:    br label %[[BODY:.*]]
+; CHECK:       [[HEADER:.*]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i32 [ [[IV_NEXT:%.*]], %[[BODY]] ]
+; CHECK-NEXT:    [[HCMP:%.*]] = icmp eq i32 [[IV]], [[LIMIT]]
+; CHECK-NEXT:    br i1 [[HCMP]], label %[[HEADER_EXIT1_CRIT_EDGE:.*]], label %[[BODY]], !prof [[PROF1]]
+; CHECK:       [[BODY]]:
+; CHECK-NEXT:    [[IV2:%.*]] = phi i32 [ [[START]], %[[BODY_LR_PH]] ], [ [[IV]], %[[HEADER]] ]
+; CHECK-NEXT:    [[ADDR:%.*]] = getelementptr i32, ptr [[P]], i32 [[IV2]]
+; CHECK-NEXT:    [[V:%.*]] = load i32, ptr [[ADDR]], align 4
+; CHECK-NEXT:    [[BCMP:%.*]] = icmp slt i32 [[V]], 0
+; CHECK-NEXT:    [[IV_NEXT]] = add i32 [[IV2]], 1
+; CHECK-NEXT:    br i1 [[BCMP]], label %[[EXIT2:.*]], label %[[HEADER]], !prof [[PROF2:![0-9]+]]
+; CHECK:       [[HEADER_EXIT1_CRIT_EDGE]]:
+; CHECK-NEXT:    br label %[[EXIT1]]
+; CHECK:       [[EXIT1]]:
+; CHECK-NEXT:    ret void
+; CHECK:       [[EXIT2]]:
+; CHECK-NEXT:    ret void
+; CHECK:       [[RET]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  %c0 = icmp ne i32 %cond, 0
+  br i1 %c0, label %ph, label %ret, !prof !0
+
+ph:                                               ; preds = %entry
+  br label %header
+
+header:                                           ; preds = %latch, %ph
+  %iv = phi i32 [ %start, %ph ], [ %iv.next, %latch ]
+  ; Keep the guard conditional.
+  %hcmp = icmp eq i32 %iv, %limit
+  br i1 %hcmp, label %exit1, label %body, !prof !1
+
+body:                                             ; preds = %header
+  %addr = getelementptr i32, ptr %p, i32 %iv
+  %v = load i32, ptr %addr, align 4
+  %bcmp = icmp slt i32 %v, 0
+  br i1 %bcmp, label %exit2, label %latch, !prof !2
+
+latch:                                            ; preds = %body
+  %iv.next = add i32 %iv, 1
+  br label %header
+
+exit1:                                            ; preds = %header
+  ret void
+
+exit2:                                            ; preds = %body
+  ret void
+
+ret:                                              ; preds = %entry
+  ret void
+}
+
+!0 = !{!"branch_weights", i32 42, i32 2}
+!1 = !{!"branch_weights", i32 200, i32 9800}
+!2 = !{!"branch_weights", i32 10, i32 9790}
+
+;.
+; CHECK: [[PROF0]] = !{!"branch_weights", i32 42, i32 2}
+; CHECK: [[PROF1]] = !{!"branch_weights", i32 200, i32 9800}
+; CHECK: [[PROF2]] = !{!"branch_weights", i32 10, i32 9790}
+;.
diff --git a/llvm/test/Transforms/LoopRotate/never-entered-branch-weights.ll b/llvm/test/Transforms/LoopRotate/never-entered-branch-weights.ll
new file mode 100644
index 0000000000000..27676b7408368
--- /dev/null
+++ b/llvm/test/Transforms/LoopRotate/never-entered-branch-weights.ll
@@ -0,0 +1,78 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --no-generate-body-for-unused-prefixes --version 6
+; RUN: opt < %s -passes=loop-rotate -S | FileCheck %s
+;
+; Header weights are {5, 0} (exit taken, backedge never taken). Rotation
+; leaves a conditional preheader guard, so updateBranchWeights() preserves
+; the cloned header weights instead of applying the folded-guard adjustment
+; (which would produce {1, 0} for this profile).
+;
+; Expected weights on the rotated guard and latch branches: {5, 0}.
+
+define void @g(i32 %n, i32 %cond) !prof !14 {
+; CHECK-LABEL: define void @g(
+; CHECK-SAME: i32 [[N:%.*]], i32 [[COND:%.*]]) !prof [[PROF14:![0-9]+]] {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[C0:%.*]] = icmp ne i32 [[COND]], 0
+; CHECK-NEXT:    br i1 [[C0]], label %[[PH:.*]], label %[[RET:.*]], !prof [[PROF15:![0-9]+]]
+; CHECK:       [[PH]]:
+; CHECK-NEXT:    br label %[[HEADER:.*]]
+; CHECK:       [[HEADER]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i32 [ 0, %[[PH]] ], [ [[IV_NEXT:%.*]], %[[HEADER]] ]
+; CHECK-NEXT:    [[HCMP:%.*]] = icmp sge i32 [[IV]], [[N]]
+; CHECK-NEXT:    [[IV_NEXT]] = add i32 [[IV]], 1
+; CHECK-NEXT:    br i1 [[HCMP]], label %[[EXIT:.*]], label %[[HEADER]], !prof [[PROF16:![0-9]+]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+; CHECK:       [[RET]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  %c0 = icmp ne i32 %cond, 0
+  br i1 %c0, label %ph, label %ret, !prof !15
+
+ph:                                               ; preds = %entry
+  br label %header
+
+header:                                           ; preds = %latch, %ph
+  %iv = phi i32 [ 0, %ph ], [ %iv.next, %latch ]
+  %hcmp = icmp sge i32 %iv, %n
+  br i1 %hcmp, label %exit, label %latch, !prof !16
+
+latch:                                            ; preds = %header
+  %iv.next = add i32 %iv, 1
+  br label %header
+
+exit:                                             ; preds = %header
+  ret void
+
+ret:                                              ; preds = %entry
+  ret void
+}
+
+!llvm.module.flags = !{!0}
+
+!0 = !{i32 1, !"ProfileSummary", !1}
+!1 = !{!2, !3, !4, !5, !6, !7, !8, !9}
+!2 = !{!"ProfileFormat", !"InstrProf"}
+!3 = !{!"TotalCount", i64 6}
+!4 = !{!"MaxCount", i64 5}
+!5 = !{!"MaxInternalCount", i64 5}
+!6 = !{!"MaxFunctionCount", i64 6}
+!7 = !{!"NumCounts", i64 3}
+!8 = !{!"NumFunctions", i64 1}
+!9 = !{!"DetailedSummary", !10}
+!10 = !{!11, !12, !13}
+!11 = !{i32 10000, i64 5, i32 1}
+!12 = !{i32 999000, i64 5, i32 1}
+!13 = !{i32 999999, i64 5, i32 1}
+!14 = !{!"function_entry_count", i64 6}
+!15 = !{!"branch_weights", i32 5, i32 1}
+!16 = !{!"branch_weights", i32 5, i32 0}
+
+
+
+;.
+; CHECK: [[PROF14]] = !{!"function_entry_count", i64 6}
+; CHECK: [[PROF15]] = !{!"branch_weights", i32 5, i32 1}
+; CHECK: [[PROF16]] = !{!"branch_weights", i32 5, i32 0}
+;.
diff --git a/llvm/test/Transforms/LoopRotate/predecessor-no-prof-branch-weights.ll b/llvm/test/Transforms/LoopRotate/predecessor-no-prof-branch-weights.ll
new file mode 100644
index 0000000000000..b9104114369eb
--- /dev/null
+++ b/llvm/test/Transforms/LoopRotate/predecessor-no-prof-branch-weights.ll
@@ -0,0 +1,84 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --no-generate-body-for-unused-prefixes --version 6
+; RUN: opt < %s -passes=loop-rotate -S | FileCheck %s
+;
+; The preheader predecessor has no profile metadata. This must not affect
+; preservation of the original header weights on the guard and latch.
+
+define void @no_pred_prof(ptr %p, i32 %n, i32 %cond) !prof !14 {
+; CHECK-LABEL: define void @no_pred_prof(
+; CHECK-SAME: ptr [[P:%.*]], i32 [[N:%.*]], i32 [[COND:%.*]]) !prof [[PROF14:![0-9]+]] {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[C0:%.*]] = icmp ne i32 [[COND]], 0
+; CHECK-NEXT:    br i1 [[C0]], label %[[PH:.*]], label %[[RET:.*]]
+; CHECK:       [[PH]]:
+; CHECK-NEXT:    [[HCMP1:%.*]] = icmp sge i32 0, [[N]]
+; CHECK-NEXT:    br i1 [[HCMP1]], label %[[EXIT:.*]], label %[[BODY_LR_PH:.*]], !prof [[PROF15:![0-9]+]]
+; CHECK:       [[BODY_LR_PH]]:
+; CHECK-NEXT:    br label %[[BODY:.*]]
+; CHECK:       [[BODY]]:
+; CHECK-NEXT:    [[IV2:%.*]] = phi i32 [ 0, %[[BODY_LR_PH]] ], [ [[IV_NEXT:%.*]], %[[LATCH:.*]] ]
+; CHECK-NEXT:    [[ADDR:%.*]] = getelementptr i32, ptr [[P]], i32 [[IV2]]
+; CHECK-NEXT:    store i32 [[IV2]], ptr [[ADDR]], align 4
+; CHECK-NEXT:    br label %[[LATCH]]
+; CHECK:       [[LATCH]]:
+; CHECK-NEXT:    [[IV_NEXT]] = add i32 [[IV2]], 1
+; CHECK-NEXT:    [[HCMP:%.*]] = icmp sge i32 [[IV_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[HCMP]], label %[[HEADER_EXIT_CRIT_EDGE:.*]], label %[[BODY]], !prof [[PROF15]]
+; CHECK:       [[HEADER_EXIT_CRIT_EDGE]]:
+; CHECK-NEXT:    br label %[[EXIT]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+; CHECK:       [[RET]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  %c0 = icmp ne i32 %cond, 0
+  ; Intentionally no !prof.
+  br i1 %c0, label %ph, label %ret
+
+ph:                                               ; preds = %entry
+  br label %header
+
+header:                                           ; preds = %latch, %ph
+  %iv = phi i32 [ 0, %ph ], [ %iv.next, %latch ]
+  %hcmp = icmp sge i32 %iv, %n
+  br i1 %hcmp, label %exit, label %body, !prof !15
+
+body:                                             ; preds = %header
+  %addr = getelementptr i32, ptr %p, i32 %iv
+  store i32 %iv, ptr %addr, align 4
+  br label %latch
+
+latch:                                            ; preds = %body
+  %iv.next = add i32 %iv, 1
+  br label %header
+
+exit:                                             ; preds = %header
+  ret void
+
+ret:                                              ; preds = %entry
+  ret void
+}
+
+!llvm.module.flags = !{!0}
+
+!0 = !{i32 1, !"ProfileSummary", !1}
+!1 = !{!2, !3, !4, !5, !6, !7, !8, !9}
+!2 = !{!"ProfileFormat", !"InstrProf"}
+!3 = !{!"TotalCount", i64 10000}
+!4 = !{!"MaxCount", i64 9800}
+!5 = !{!"MaxInternalCount", i64 9800}
+!6 = !{!"MaxFunctionCount", i64 200}
+!7 = !{!"NumCounts", i64 3}
+!8 = !{!"NumFunctions", i64 1}
+!9 = !{!"DetailedSummary", !10}
+!10 = !{!11, !12, !13}
+!11 = !{i32 10000, i64 9800, i32 1}
+!12 = !{i32 999000, i64 200, i32 1}
+!13 = !{i32 999999, i64 200, i32 1}
+!14 = !{!"function_entry_count", i64 201}
+!15 = !{!"branch_weights", i32 200, i32 9800}
+;.
+; CHECK: [[PROF14]] = !{!"function_entry_count", i64 201}
+; CHECK: [[PROF15]] = !{!"branch_weights", i32 200, i32 9800}
+;.
diff --git a/llvm/test/Transforms/LoopRotate/saturating-backedge-branch-weights.ll b/llvm/test/Transforms/LoopRotate/saturating-backedge-branch-weights.ll
new file mode 100644
index 0000000000000..145caa56e6462
--- /dev/null
+++ b/llvm/test/Transforms/LoopRotate/saturating-backedge-branch-weights.ll
@@ -0,0 +1,96 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --no-generate-body-for-unused-prefixes --version 6
+; RUN: opt < %s -passes=loop-rotate -S | FileCheck %s
+;
+; Multi-exit loop with large header and body weights. Rotation leaves a
+; conditional preheader guard, so updateBranchWeights() must preserve the
+; original header and body branch ratios on the guard and latch.
+
+define void @sat_backedge(ptr %p, i32 %n, i32 %cond) !prof !14 {
+; CHECK-LABEL: define void @sat_backedge(
+; CHECK-SAME: ptr [[P:%.*]], i32 [[N:%.*]], i32 [[COND:%.*]]) !prof [[PROF14:![0-9]+]] {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[C0:%.*]] = icmp ne i32 [[COND]], 0
+; CHECK-NEXT:    br i1 [[C0]], label %[[PH:.*]], label %[[RET:.*]], !prof [[PROF15:![0-9]+]]
+; CHECK:       [[PH]]:
+; CHECK-NEXT:    [[HCMP1:%.*]] = icmp sge i32 0, [[N]]
+; CHECK-NEXT:    br i1 [[HCMP1]], label %[[EXIT1:.*]], label %[[BODY_LR_PH:.*]], !prof [[PROF16:![0-9]+]]
+; CHECK:       [[BODY_LR_PH]]:
+; CHECK-NEXT:    br label %[[BODY:.*]]
+; CHECK:       [[HEADER:.*]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i32 [ [[IV_NEXT:%.*]], %[[BODY]] ]
+; CHECK-NEXT:    [[HCMP:%.*]] = icmp sge i32 [[IV]], [[N]]
+; CHECK-NEXT:    br i1 [[HCMP]], label %[[HEADER_EXIT1_CRIT_EDGE:.*]], label %[[BODY]], !prof [[PROF16]]
+; CHECK:       [[BODY]]:
+; CHECK-NEXT:    [[IV2:%.*]] = phi i32 [ 0, %[[BODY_LR_PH]] ], [ [[IV]], %[[HEADER]] ]
+; CHECK-NEXT:    [[ADDR:%.*]] = getelementptr i32, ptr [[P]], i32 [[IV2]]
+; CHECK-NEXT:    [[V:%.*]] = load i32, ptr [[ADDR]], align 4
+; CHECK-NEXT:    [[BCMP:%.*]] = icmp slt i32 [[V]], 0
+; CHECK-NEXT:    [[IV_NEXT]] = add i32 [[IV2]], 1
+; CHECK-NEXT:    br i1 [[BCMP]], label %[[EXIT2:.*]], label %[[HEADER]], !prof [[PROF17:![0-9]+]]
+; CHECK:       [[HEADER_EXIT1_CRIT_EDGE]]:
+; CHECK-NEXT:    br label %[[EXIT1]]
+; CHECK:       [[EXIT1]]:
+; CHECK-NEXT:    ret void
+; CHECK:       [[EXIT2]]:
+; CHECK-NEXT:    ret void
+; CHECK:       [[RET]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  %c0 = icmp ne i32 %cond, 0
+  br i1 %c0, label %ph, label %ret, !prof !15
+
+ph:                                               ; preds = %entry
+  br label %header
+
+header:                                           ; preds = %latch, %ph
+  %iv = phi i32 [ 0, %ph ], [ %iv.next, %latch ]
+  %hcmp = icmp sge i32 %iv, %n
+  br i1 %hcmp, label %exit1, label %body, !prof !16
+
+body:                                             ; preds = %header
+  %addr = getelementptr i32, ptr %p, i32 %iv
+  %v = load i32, ptr %addr, align 4
+  %bcmp = icmp slt i32 %v, 0
+  br i1 %bcmp, label %exit2, label %latch, !prof !17
+
+latch:                                            ; preds = %body
+  %iv.next = add i32 %iv, 1
+  br label %header
+
+exit1:                                            ; preds = %header
+  ret void
+
+exit2:                                            ; preds = %body
+  ret void
+
+ret:                                              ; preds = %entry
+  ret void
+}
+
+!llvm.module.flags = !{!0}
+
+!0 = !{i32 1, !"ProfileSummary", !1}
+!1 = !{!2, !3, !4, !5, !6, !7, !8, !9}
+!2 = !{!"ProfileFormat", !"InstrProf"}
+!3 = !{!"TotalCount", i64 11000}
+!4 = !{!"MaxCount", i64 990}
+!5 = !{!"MaxInternalCount", i64 990}
+!6 = !{!"MaxFunctionCount", i64 1000}
+!7 = !{!"NumCounts", i64 4}
+!8 = !{!"NumFunctions", i64 1}
+!9 = !{!"DetailedSummary", !10}
+!10 = !{!11, !12, !13}
+!11 = !{i32 10000, i64 990, i32 1}
+!12 = !{i32 999000, i64 1000, i32 1}
+!13 = !{i32 999999, i64 990, i32 1}
+!14 = !{!"function_entry_count", i64 1000}
+!15 = !{!"branch_weights", i32 999, i32 1}
+!16 = !{!"branch_weights", i32 990, i32 10}
+!17 = !{!"branch_weights", i32 9,   i32 1}
+;.
+; CHECK: [[PROF14]] = !{!"function_entry_count", i64 1000}
+; CHECK: [[PROF15]] = !{!"branch_weights", i32 999, i32 1}
+; CHECK: [[PROF16]] = !{!"branch_weights", i32 990, i32 10}
+; CHECK: [[PROF17]] = !{!"branch_weights", i32 9, i32 1}
+;.
diff --git a/llvm/test/Transforms/LoopRotate/scaled-preheader-branch-weights.ll b/llvm/test/Transforms/LoopRotate/scaled-preheader-branch-weights.ll
new file mode 100644
index 0000000000000..3dba06fec2d74
--- /dev/null
+++ b/llvm/test/Transforms/LoopRotate/scaled-preheader-branch-weights.ll
@@ -0,0 +1,60 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --no-generate-body-for-unused-prefixes --version 6
+; RUN: opt < %s -passes=loop-rotate -S | FileCheck %s
+;
+; The preheader and header use different weight scales. Rotation must preserve
+; each branch's local ratio without mixing the scales.
+
+define void @small_scaled(ptr %p, i32 %n, i1 %c) !prof !0 {
+; CHECK-LABEL: define void @small_scaled(
+; CHECK-SAME: ptr [[P:%.*]], i32 [[N:%.*]], i1 [[C:%.*]]) !prof [[PROF0:![0-9]+]] {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    br i1 [[C]], label %[[PH:.*]], label %[[RET:.*]], !prof [[PROF1:![0-9]+]]
+; CHECK:       [[PH]]:
+; CHECK-NEXT:    [[CMP1:%.*]] = icmp slt i32 0, [[N]]
+; CHECK-NEXT:    br i1 [[CMP1]], label %[[BODY_LR_PH:.*]], label %[[EXIT:.*]], !prof [[PROF2:![0-9]+]]
+; CHECK:       [[BODY_LR_PH]]:
+; CHECK-NEXT:    br label %[[BODY:.*]]
+; CHECK:       [[BODY]]:
+; CHECK-NEXT:    [[IV2:%.*]] = phi i32 [ 0, %[[BODY_LR_PH]] ], [ [[NEXT:%.*]], %[[BODY]] ]
+; CHECK-NEXT:    store volatile i32 [[IV2]], ptr [[P]], align 4
+; CHECK-NEXT:    [[NEXT]] = add i32 [[IV2]], 1
+; CHECK-NEXT:    [[CMP:%.*]] = icmp slt i32 [[NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[CMP]], label %[[BODY]], label %[[HEADER_EXIT_CRIT_EDGE:.*]], !prof [[PROF2]]
+; CHECK:       [[HEADER_EXIT_CRIT_EDGE]]:
+; CHECK-NEXT:    br label %[[EXIT]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+; CHECK:       [[RET]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br i1 %c, label %ph, label %ret, !prof !1
+
+ph:
+  br label %header
+
+header:
+  %iv = phi i32 [ 0, %ph ], [ %next, %body ]
+  %cmp = icmp slt i32 %iv, %n
+  br i1 %cmp, label %body, label %exit, !prof !2
+
+body:
+  store volatile i32 %iv, ptr %p, align 4
+  %next = add i32 %iv, 1
+  br label %header
+
+exit:
+  ret void
+
+ret:
+  ret void
+}
+
+!0 = !{!"function_entry_count", i64 2}
+!1 = !{!"branch_weights", i32 1, i32 1}
+!2 = !{!"branch_weights", i32 100, i32 1}
+;.
+; CHECK: [[PROF0]] = !{!"function_entry_count", i64 2}
+; CHECK: [[PROF1]] = !{!"branch_weights", i32 1, i32 1}
+; CHECK: [[PROF2]] = !{!"branch_weights", i32 100, i32 1}
+;.
diff --git a/llvm/test/Transforms/LoopRotate/update-branch-weights.ll b/llvm/test/Transforms/LoopRotate/update-branch-weights.ll
index 9a1f36ec5ff2b..912f70b34bb82 100644
--- a/llvm/test/Transforms/LoopRotate/update-branch-weights.ll
+++ b/llvm/test/Transforms/LoopRotate/update-branch-weights.ll
@@ -1,3 +1,4 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --no-generate-body-for-unused-prefixes --version 6
 ; RUN: opt < %s -passes='print<block-freq>' -disable-output 2>&1 | FileCheck %s --check-prefixes=BFI_BEFORE
 ; RUN: opt < %s -passes='loop(loop-rotate),print<block-freq>' -disable-output 2>&1 | FileCheck %s --check-prefixes=BFI_AFTER
 ; RUN: opt < %s -passes='loop(loop-rotate)' -S | FileCheck %s --check-prefixes=IR
@@ -23,15 +24,31 @@
 ; BFI_AFTER: - inner_loop_exit: {{.*}} count = 1000
 ; BFI_AFTER: - outer_loop_exit: {{.*}} count = 1
 
-; IR-LABEL: define void @func0
-; IR: inner_loop_body:
-; IR:   br i1 %cmp1, label %inner_loop_body, label %inner_loop_exit, !prof [[PROF_FUNC0_0:![0-9]+]]
-; IR: inner_loop_exit:
-; IR:   br i1 %cmp0, label %outer_loop_body, label %outer_loop_exit, !prof [[PROF_FUNC0_1:![0-9]+]]
 ;
 ; A function with known loop-bounds where after loop-rotation we end with an
 ; unconditional branch in the pre-header.
 define void @func0() !prof !0 {
+; IR-LABEL: define void @func0(
+; IR-SAME: ) !prof [[PROF0:![0-9]+]] {
+; IR-NEXT:  [[ENTRY:.*]]:
+; IR-NEXT:    br label %[[OUTER_LOOP_BODY:.*]]
+; IR:       [[OUTER_LOOP_BODY]]:
+; IR-NEXT:    [[I02:%.*]] = phi i32 [ 0, %[[ENTRY]] ], [ [[I0_INC:%.*]], %[[INNER_LOOP_EXIT:.*]] ]
+; IR-NEXT:    store volatile i32 [[I02]], ptr @g, align 4
+; IR-NEXT:    br label %[[INNER_LOOP_BODY:.*]]
+; IR:       [[INNER_LOOP_BODY]]:
+; IR-NEXT:    [[I11:%.*]] = phi i32 [ 0, %[[OUTER_LOOP_BODY]] ], [ [[I1_INC:%.*]], %[[INNER_LOOP_BODY]] ]
+; IR-NEXT:    store volatile i32 [[I11]], ptr @g, align 4
+; IR-NEXT:    [[I1_INC]] = add i32 [[I11]], 1
+; IR-NEXT:    [[CMP1:%.*]] = icmp slt i32 [[I1_INC]], 3
+; IR-NEXT:    br i1 [[CMP1]], label %[[INNER_LOOP_BODY]], label %[[INNER_LOOP_EXIT]], !prof [[PROF1:![0-9]+]]
+; IR:       [[INNER_LOOP_EXIT]]:
+; IR-NEXT:    [[I0_INC]] = add i32 [[I02]], 1
+; IR-NEXT:    [[CMP0:%.*]] = icmp slt i32 [[I0_INC]], 1000
+; IR-NEXT:    br i1 [[CMP0]], label %[[OUTER_LOOP_BODY]], label %[[OUTER_LOOP_EXIT:.*]], !prof [[PROF2:![0-9]+]]
+; IR:       [[OUTER_LOOP_EXIT]]:
+; IR-NEXT:    ret void
+;
 entry:
   br label %outer_loop_header
 
@@ -70,22 +87,31 @@ outer_loop_exit:
 
 ; BFI_AFTER-LABEL: block-frequency-info: func1
 ; BFI_AFTER: - entry: {{.*}} count = 1024
-; BFI_AFTER: - loop_body.lr.ph: {{.*}} count = 1016
 ; BFI_AFTER: - loop_body: {{.*}} count = 20480
-; BFI_AFTER: - loop_header.loop_exit_crit_edge: {{.*}} count = 1016
 ; BFI_AFTER: - loop_exit: {{.*}} count = 1024
 
-; IR-LABEL: define void @func1
-; IR: entry:
-; IR:   br i1 %cmp1, label %loop_body.lr.ph, label %loop_exit, !prof [[PROF_FUNC1_0:![0-9]+]]
-
-; IR: loop_body:
-; IR:   br i1 %cmp, label %loop_body, label %loop_header.loop_exit_crit_edge, !prof [[PROF_FUNC1_1:![0-9]+]]
-
 ; A function with unknown loop-bounds so loop-rotation ends up with a
 ; condition jump in pre-header and loop body. branch_weight shows body is
 ; executed more often than header.
 define void @func1(i32 %n) !prof !3 {
+; IR-LABEL: define void @func1(
+; IR-SAME: i32 [[N:%.*]]) !prof [[PROF3:![0-9]+]] {
+; IR-NEXT:  [[ENTRY:.*:]]
+; IR-NEXT:    [[CMP1:%.*]] = icmp slt i32 0, [[N]]
+; IR-NEXT:    br i1 [[CMP1]], label %[[LOOP_BODY_LR_PH:.*]], label %[[LOOP_EXIT:.*]], !prof [[PROF4:![0-9]+]]
+; IR:       [[LOOP_BODY_LR_PH]]:
+; IR-NEXT:    br label %[[LOOP_BODY:.*]]
+; IR:       [[LOOP_BODY]]:
+; IR-NEXT:    [[I2:%.*]] = phi i32 [ 0, %[[LOOP_BODY_LR_PH]] ], [ [[I_INC:%.*]], %[[LOOP_BODY]] ]
+; IR-NEXT:    store volatile i32 [[I2]], ptr @g, align 4
+; IR-NEXT:    [[I_INC]] = add i32 [[I2]], 1
+; IR-NEXT:    [[CMP:%.*]] = icmp slt i32 [[I_INC]], [[N]]
+; IR-NEXT:    br i1 [[CMP]], label %[[LOOP_BODY]], label %[[LOOP_HEADER_LOOP_EXIT_CRIT_EDGE:.*]], !prof [[PROF4]]
+; IR:       [[LOOP_HEADER_LOOP_EXIT_CRIT_EDGE]]:
+; IR-NEXT:    br label %[[LOOP_EXIT]]
+; IR:       [[LOOP_EXIT]]:
+; IR-NEXT:    ret void
+;
 entry:
   br label %loop_header
 
@@ -110,23 +136,32 @@ loop_exit:
 ; BFI_BEFORE: - loop_exit: {{.*}} count = 1024
 
 ; BFI_AFTER-LABEL: block-frequency-info: func2
-; - entry: {{.*}} count = 1024
-; - loop_body.lr.ph: {{.*}} count = 32
-; - loop_body: {{.*}} count = 32
-; - loop_header.loop_exit_crit_edge: {{.*}} count = 32
-; - loop_exit: {{.*}} count = 1024
-
-; IR-LABEL: define void @func2
-; IR: entry:
-; IR:   br i1 %cmp1, label %loop_exit, label %loop_body.lr.ph, !prof [[PROF_FUNC2_0:![0-9]+]]
-
-; IR: loop_body:
-; IR:   br i1 %cmp, label %loop_header.loop_exit_crit_edge, label %loop_body, !prof [[PROF_FUNC2_1:![0-9]+]]
+; BFI_AFTER: - entry: {{.*}} count = 1024
+; BFI_AFTER: - loop_body: {{.*}} count = 32
+; BFI_AFTER: - loop_exit: {{.*}} count = 1024
 
 ; A function with unknown loop-bounds so loop-rotation ends up with a
 ; condition jump in pre-header and loop body. Similar to `func1` but here
 ; loop-exit count is higher than backedge count.
 define void @func2(i32 %n) !prof !3 {
+; IR-LABEL: define void @func2(
+; IR-SAME: i32 [[N:%.*]]) !prof [[PROF3]] {
+; IR-NEXT:  [[ENTRY:.*:]]
+; IR-NEXT:    [[CMP1:%.*]] = icmp slt i32 0, [[N]]
+; IR-NEXT:    br i1 [[CMP1]], label %[[LOOP_EXIT:.*]], label %[[LOOP_BODY_LR_PH:.*]], !prof [[PROF5:![0-9]+]]
+; IR:       [[LOOP_BODY_LR_PH]]:
+; IR-NEXT:    br label %[[LOOP_BODY:.*]]
+; IR:       [[LOOP_BODY]]:
+; IR-NEXT:    [[I2:%.*]] = phi i32 [ 0, %[[LOOP_BODY_LR_PH]] ], [ [[I_INC:%.*]], %[[LOOP_BODY]] ]
+; IR-NEXT:    store volatile i32 [[I2]], ptr @g, align 4
+; IR-NEXT:    [[I_INC]] = add i32 [[I2]], 1
+; IR-NEXT:    [[CMP:%.*]] = icmp slt i32 [[I_INC]], [[N]]
+; IR-NEXT:    br i1 [[CMP]], label %[[LOOP_HEADER_LOOP_EXIT_CRIT_EDGE:.*]], label %[[LOOP_BODY]], !prof [[PROF5]]
+; IR:       [[LOOP_HEADER_LOOP_EXIT_CRIT_EDGE]]:
+; IR-NEXT:    br label %[[LOOP_EXIT]]
+; IR:       [[LOOP_EXIT]]:
+; IR-NEXT:    ret void
+;
 entry:
   br label %loop_header
 
@@ -157,14 +192,25 @@ loop_exit:
 ; BFI_AFTER: - loop_header.loop_exit_crit_edge: {{.*}} count = 1024
 ; BFI_AFTER: - loop_exit: {{.*}} count = 1024
 
-; IR-LABEL: define void @func3_zero_branch_weight
-; IR: entry:
-; IR:   br i1 %cmp1, label %loop_exit, label %loop_body.lr.ph, !prof [[PROF_FUNC3_0:![0-9]+]]
-
-; IR: loop_body:
-; IR:   br i1 %cmp, label %loop_header.loop_exit_crit_edge, label %loop_body, !prof [[PROF_FUNC3_0]]
-
 define void @func3_zero_branch_weight(i32 %n) !prof !3 {
+; IR-LABEL: define void @func3_zero_branch_weight(
+; IR-SAME: i32 [[N:%.*]]) !prof [[PROF3]] {
+; IR-NEXT:  [[ENTRY:.*:]]
+; IR-NEXT:    [[CMP1:%.*]] = icmp slt i32 0, [[N]]
+; IR-NEXT:    br i1 [[CMP1]], label %[[LOOP_EXIT:.*]], label %[[LOOP_BODY_LR_PH:.*]], !prof [[PROF6:![0-9]+]]
+; IR:       [[LOOP_BODY_LR_PH]]:
+; IR-NEXT:    br label %[[LOOP_BODY:.*]]
+; IR:       [[LOOP_BODY]]:
+; IR-NEXT:    [[I2:%.*]] = phi i32 [ 0, %[[LOOP_BODY_LR_PH]] ], [ [[I_INC:%.*]], %[[LOOP_BODY]] ]
+; IR-NEXT:    store volatile i32 [[I2]], ptr @g, align 4
+; IR-NEXT:    [[I_INC]] = add i32 [[I2]], 1
+; IR-NEXT:    [[CMP:%.*]] = icmp slt i32 [[I_INC]], [[N]]
+; IR-NEXT:    br i1 [[CMP]], label %[[LOOP_HEADER_LOOP_EXIT_CRIT_EDGE:.*]], label %[[LOOP_BODY]], !prof [[PROF6]]
+; IR:       [[LOOP_HEADER_LOOP_EXIT_CRIT_EDGE]]:
+; IR-NEXT:    br label %[[LOOP_EXIT]]
+; IR:       [[LOOP_EXIT]]:
+; IR-NEXT:    ret void
+;
 entry:
   br label %loop_header
 
@@ -182,14 +228,25 @@ loop_exit:
   ret void
 }
 
-; IR-LABEL: define void @func4_zero_branch_weight
-; IR: entry:
-; IR:   br i1 %cmp1, label %loop_exit, label %loop_body.lr.ph, !prof [[PROF_FUNC4_0:![0-9]+]]
-
-; IR: loop_body:
-; IR:   br i1 %cmp, label %loop_header.loop_exit_crit_edge, label %loop_body, !prof [[PROF_FUNC4_0]]
-
 define void @func4_zero_branch_weight(i32 %n) !prof !3 {
+; IR-LABEL: define void @func4_zero_branch_weight(
+; IR-SAME: i32 [[N:%.*]]) !prof [[PROF3]] {
+; IR-NEXT:  [[ENTRY:.*:]]
+; IR-NEXT:    [[CMP1:%.*]] = icmp slt i32 0, [[N]]
+; IR-NEXT:    br i1 [[CMP1]], label %[[LOOP_EXIT:.*]], label %[[LOOP_BODY_LR_PH:.*]], !prof [[PROF7:![0-9]+]]
+; IR:       [[LOOP_BODY_LR_PH]]:
+; IR-NEXT:    br label %[[LOOP_BODY:.*]]
+; IR:       [[LOOP_BODY]]:
+; IR-NEXT:    [[I2:%.*]] = phi i32 [ 0, %[[LOOP_BODY_LR_PH]] ], [ [[I_INC:%.*]], %[[LOOP_BODY]] ]
+; IR-NEXT:    store volatile i32 [[I2]], ptr @g, align 4
+; IR-NEXT:    [[I_INC]] = add i32 [[I2]], 1
+; IR-NEXT:    [[CMP:%.*]] = icmp slt i32 [[I_INC]], [[N]]
+; IR-NEXT:    br i1 [[CMP]], label %[[LOOP_HEADER_LOOP_EXIT_CRIT_EDGE:.*]], label %[[LOOP_BODY]], !prof [[PROF7]]
+; IR:       [[LOOP_HEADER_LOOP_EXIT_CRIT_EDGE]]:
+; IR-NEXT:    br label %[[LOOP_EXIT]]
+; IR:       [[LOOP_EXIT]]:
+; IR-NEXT:    ret void
+;
 entry:
   br label %loop_header
 
@@ -207,14 +264,25 @@ loop_exit:
   ret void
 }
 
-; IR-LABEL: define void @func5_zero_branch_weight
-; IR: entry:
-; IR:   br i1 %cmp1, label %loop_exit, label %loop_body.lr.ph, !prof [[PROF_FUNC5_0:![0-9]+]]
-
-; IR: loop_body:
-; IR:   br i1 %cmp, label %loop_header.loop_exit_crit_edge, label %loop_body, !prof [[PROF_FUNC5_0]]
-
 define void @func5_zero_branch_weight(i32 %n) !prof !3 {
+; IR-LABEL: define void @func5_zero_branch_weight(
+; IR-SAME: i32 [[N:%.*]]) !prof [[PROF3]] {
+; IR-NEXT:  [[ENTRY:.*:]]
+; IR-NEXT:    [[CMP1:%.*]] = icmp slt i32 0, [[N]]
+; IR-NEXT:    br i1 [[CMP1]], label %[[LOOP_EXIT:.*]], label %[[LOOP_BODY_LR_PH:.*]], !prof [[PROF8:![0-9]+]]
+; IR:       [[LOOP_BODY_LR_PH]]:
+; IR-NEXT:    br label %[[LOOP_BODY:.*]]
+; IR:       [[LOOP_BODY]]:
+; IR-NEXT:    [[I2:%.*]] = phi i32 [ 0, %[[LOOP_BODY_LR_PH]] ], [ [[I_INC:%.*]], %[[LOOP_BODY]] ]
+; IR-NEXT:    store volatile i32 [[I2]], ptr @g, align 4
+; IR-NEXT:    [[I_INC]] = add i32 [[I2]], 1
+; IR-NEXT:    [[CMP:%.*]] = icmp slt i32 [[I_INC]], [[N]]
+; IR-NEXT:    br i1 [[CMP]], label %[[LOOP_HEADER_LOOP_EXIT_CRIT_EDGE:.*]], label %[[LOOP_BODY]], !prof [[PROF8]]
+; IR:       [[LOOP_HEADER_LOOP_EXIT_CRIT_EDGE]]:
+; IR-NEXT:    br label %[[LOOP_EXIT]]
+; IR:       [[LOOP_EXIT]]:
+; IR-NEXT:    ret void
+;
 entry:
   br label %loop_header
 
@@ -243,18 +311,24 @@ loop_exit:
 ; BFI_AFTER: - loop_body: {{.*}} count = 1024
 ; BFI_AFTER: - loop_exit: {{.*}} count = 1024
 
-; IR-LABEL: define void @func6_inaccurate_branch_weight(
-; IR: entry:
-; IR:   br label %loop_body
-; IR: loop_body:
-; IR:   br i1 %cmp, label %loop_body, label %loop_exit, !prof [[PROF_FUNC6_0:![0-9]+]]
-; IR: loop_exit:
-; IR:   ret void
 
 ; Branch weight from sample-based PGO may be inaccurate due to sampling.
 ; Count for loop_body in following case should be not less than loop_exit.
 ; However this may not hold for Sample-based PGO.
 define void @func6_inaccurate_branch_weight() !prof !3 {
+; IR-LABEL: define void @func6_inaccurate_branch_weight(
+; IR-SAME: ) !prof [[PROF3]] {
+; IR-NEXT:  [[ENTRY:.*]]:
+; IR-NEXT:    br label %[[LOOP_BODY:.*]]
+; IR:       [[LOOP_BODY]]:
+; IR-NEXT:    [[I1:%.*]] = phi i32 [ 0, %[[ENTRY]] ], [ [[I_INC:%.*]], %[[LOOP_BODY]] ]
+; IR-NEXT:    store volatile i32 [[I1]], ptr @g, align 4
+; IR-NEXT:    [[I_INC]] = add i32 [[I1]], 1
+; IR-NEXT:    [[CMP:%.*]] = icmp slt i32 [[I_INC]], 2
+; IR-NEXT:    br i1 [[CMP]], label %[[LOOP_BODY]], label %[[LOOP_EXIT:.*]], !prof [[PROF9:![0-9]+]]
+; IR:       [[LOOP_EXIT]]:
+; IR-NEXT:    ret void
+;
 entry:
   br label %loop_header
 
@@ -283,13 +357,15 @@ loop_exit:
 !8 = !{!"branch_weights", i32 0, i32 0}
 !9 = !{!"branch_weights", i32 1023, i32 1024}
 
-; IR: [[PROF_FUNC0_0]] = !{!"branch_weights", i32 2000, i32 1000}
-; IR: [[PROF_FUNC0_1]] = !{!"branch_weights", i32 999, i32 1}
-; IR: [[PROF_FUNC1_0]] = !{!"branch_weights", i32 127, i32 1}
-; IR: [[PROF_FUNC1_1]] = !{!"branch_weights", i32 2433, i32 127}
-; IR: [[PROF_FUNC2_0]] = !{!"branch_weights", i32 9920, i32 320}
-; IR: [[PROF_FUNC2_1]] = !{!"branch_weights", i32 320, i32 0}
-; IR: [[PROF_FUNC3_0]] = !{!"branch_weights", i32 0, i32 1}
-; IR: [[PROF_FUNC4_0]] = !{!"branch_weights", i32 1, i32 0}
-; IR: [[PROF_FUNC5_0]] = !{!"branch_weights", i32 0, i32 0}
-; IR: [[PROF_FUNC6_0]] = !{!"branch_weights", i32 0, i32 1024}
+;.
+; IR: [[PROF0]] = !{!"function_entry_count", i64 1}
+; IR: [[PROF1]] = !{!"branch_weights", i32 2000, i32 1000}
+; IR: [[PROF2]] = !{!"branch_weights", i32 999, i32 1}
+; IR: [[PROF3]] = !{!"function_entry_count", i64 1024}
+; IR: [[PROF4]] = !{!"branch_weights", i32 40, i32 2}
+; IR: [[PROF5]] = !{!"branch_weights", i32 10240, i32 320}
+; IR: [[PROF6]] = !{!"branch_weights", i32 0, i32 1}
+; IR: [[PROF7]] = !{!"branch_weights", i32 1, i32 0}
+; IR: [[PROF8]] = !{!"branch_weights", i32 0, i32 0}
+; IR: [[PROF9]] = !{!"branch_weights", i32 0, i32 1024}
+;.

>From 4235672b7dec02d30ea135607c8a4d8388042b8c Mon Sep 17 00:00:00 2001
From: Alok Kumar Sharma <AlokKumar.Sharma at amd.com>
Date: Wed, 19 Aug 2026 23:29:26 +0530
Subject: [PATCH 2/3] [LoopRotate] Move post-rotation trip-count adjustment to
 loop metadata

After rotation, decrement llvm.loop.estimated_trip_count instead of
rewriting branch weights. Infer from weights only on single-exit loops,
and keep explicit multi-exit estimates unchanged aside from the
decrement.
---
 .../Transforms/Utils/LoopRotationUtils.cpp    |  25 +++
 .../LoopRotate/multi-exit-branch-weights.ll   |  94 ++++++----
 .../LoopRotate/update-branch-weights.ll       | 169 +++++++++++++++---
 3 files changed, 224 insertions(+), 64 deletions(-)

diff --git a/llvm/lib/Transforms/Utils/LoopRotationUtils.cpp b/llvm/lib/Transforms/Utils/LoopRotationUtils.cpp
index 7950e223bfb4d..17fb32a1536df 100644
--- a/llvm/lib/Transforms/Utils/LoopRotationUtils.cpp
+++ b/llvm/lib/Transforms/Utils/LoopRotationUtils.cpp
@@ -10,6 +10,8 @@
 //
 //===----------------------------------------------------------------------===//
 
+#include <optional>
+
 #include "llvm/Transforms/Utils/LoopRotationUtils.h"
 #include "llvm/ADT/Statistic.h"
 #include "llvm/Analysis/AssumptionCache.h"
@@ -33,8 +35,10 @@
 #include "llvm/Transforms/Utils/BasicBlockUtils.h"
 #include "llvm/Transforms/Utils/Cloning.h"
 #include "llvm/Transforms/Utils/Local.h"
+#include "llvm/Transforms/Utils/LoopUtils.h"
 #include "llvm/Transforms/Utils/SSAUpdater.h"
 #include "llvm/Transforms/Utils/ValueMapper.h"
+
 using namespace llvm;
 
 #define DEBUG_TYPE "loop-rotate"
@@ -697,6 +701,17 @@ bool LoopRotate::rotateLoop(Loop *L, bool SimplifiedLatch) {
       !isa<ConstantInt>(Cond) ||
       PHBI->getSuccessor(cast<ConstantInt>(Cond)->isZero()) != NewHeader;
 
+  // Save the trip count before updating branch weights. Only infer it from
+  // profile weights for single-exit loops, but still read and decrement
+  // explicit llvm.loop.estimated_trip_count on any loop.
+  // getExitingBlock() is null when there is no unique dedicated exit block.
+  bool IsMultiExitLoop = !L->getExitingBlock();
+  bool HasExplicitEstimatedTripCount =
+      getOptionalIntLoopAttribute(L, LLVMLoopEstimatedTripCount).has_value();
+  std::optional<unsigned> EstimatedTripCount;
+  if (!IsMultiExitLoop || HasExplicitEstimatedTripCount)
+    EstimatedTripCount = getLoopEstimatedTripCount(L);
+
   updateBranchWeights(*PHBI, *BI, HasConditionalPreHeader, BISuccsSwapped);
 
   if (HasConditionalPreHeader) {
@@ -749,6 +764,16 @@ bool LoopRotate::rotateLoop(Loop *L, bool SimplifiedLatch) {
       MSSAU->removeEdge(OrigPreheader, Exit);
   }
 
+  if (EstimatedTripCount) {
+    // One header test moved to the preheader and no longer counts as a trip.
+    // Record the decremented count in loop metadata only; updateBranchWeights()
+    // already handled branch weights above.
+    if (!setLoopEstimatedTripCount(
+            L, *EstimatedTripCount == 0 ? 0 : *EstimatedTripCount - 1))
+      LLVM_DEBUG(dbgs() << "LoopRotation: failed to set estimated trip "
+                        << "count after rotation\n");
+  }
+
   assert(L->getLoopPreheader() && "Invalid loop preheader after loop rotation");
   assert(L->getLoopLatch() && "Invalid loop latch after loop rotation");
 
diff --git a/llvm/test/Transforms/LoopRotate/multi-exit-branch-weights.ll b/llvm/test/Transforms/LoopRotate/multi-exit-branch-weights.ll
index ac47fc94b4bf5..5e24888abbdaa 100644
--- a/llvm/test/Transforms/LoopRotate/multi-exit-branch-weights.ll
+++ b/llvm/test/Transforms/LoopRotate/multi-exit-branch-weights.ll
@@ -1,5 +1,6 @@
-; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --no-generate-body-for-unused-prefixes --version 6
 ; RUN: opt < %s -passes=loop-rotate -S | FileCheck %s
+; RUN: opt < %s -passes='print<block-freq>' -disable-output 2>&1 | FileCheck %s --check-prefix=BFI-BEFORE
+; RUN: opt < %s -passes='loop(loop-rotate),print<block-freq>' -disable-output 2>&1 | FileCheck %s --check-prefix=BFI-AFTER
 ;
 ; The copied guard and latch run the same test. Keep the original weights on
 ; both, including loops with another body exit.
@@ -12,37 +13,7 @@
 ;   body   --(bcmp)--> exit2 / latch  !prof { 10, 9790 }
 ;   latch  ---------> header
 
-define void @f(ptr %p, i32 %start, i32 %limit, i32 %cond) {
-; CHECK-LABEL: define void @f(
-; CHECK-SAME: ptr [[P:%.*]], i32 [[START:%.*]], i32 [[LIMIT:%.*]], i32 [[COND:%.*]]) {
-; CHECK-NEXT:  [[ENTRY:.*:]]
-; CHECK-NEXT:    [[C0:%.*]] = icmp ne i32 [[COND]], 0
-; CHECK-NEXT:    br i1 [[C0]], label %[[PH:.*]], label %[[RET:.*]], !prof [[PROF0:![0-9]+]]
-; CHECK:       [[PH]]:
-; CHECK-NEXT:    [[HCMP1:%.*]] = icmp eq i32 [[START]], [[LIMIT]]
-; CHECK-NEXT:    br i1 [[HCMP1]], label %[[EXIT1:.*]], label %[[BODY_LR_PH:.*]], !prof [[PROF1:![0-9]+]]
-; CHECK:       [[BODY_LR_PH]]:
-; CHECK-NEXT:    br label %[[BODY:.*]]
-; CHECK:       [[HEADER:.*]]:
-; CHECK-NEXT:    [[IV:%.*]] = phi i32 [ [[IV_NEXT:%.*]], %[[BODY]] ]
-; CHECK-NEXT:    [[HCMP:%.*]] = icmp eq i32 [[IV]], [[LIMIT]]
-; CHECK-NEXT:    br i1 [[HCMP]], label %[[HEADER_EXIT1_CRIT_EDGE:.*]], label %[[BODY]], !prof [[PROF1]]
-; CHECK:       [[BODY]]:
-; CHECK-NEXT:    [[IV2:%.*]] = phi i32 [ [[START]], %[[BODY_LR_PH]] ], [ [[IV]], %[[HEADER]] ]
-; CHECK-NEXT:    [[ADDR:%.*]] = getelementptr i32, ptr [[P]], i32 [[IV2]]
-; CHECK-NEXT:    [[V:%.*]] = load i32, ptr [[ADDR]], align 4
-; CHECK-NEXT:    [[BCMP:%.*]] = icmp slt i32 [[V]], 0
-; CHECK-NEXT:    [[IV_NEXT]] = add i32 [[IV2]], 1
-; CHECK-NEXT:    br i1 [[BCMP]], label %[[EXIT2:.*]], label %[[HEADER]], !prof [[PROF2:![0-9]+]]
-; CHECK:       [[HEADER_EXIT1_CRIT_EDGE]]:
-; CHECK-NEXT:    br label %[[EXIT1]]
-; CHECK:       [[EXIT1]]:
-; CHECK-NEXT:    ret void
-; CHECK:       [[EXIT2]]:
-; CHECK-NEXT:    ret void
-; CHECK:       [[RET]]:
-; CHECK-NEXT:    ret void
-;
+define void @f(ptr %p, i32 %start, i32 %limit, i32 %cond) !prof !3 {
 entry:
   %c0 = icmp ne i32 %cond, 0
   br i1 %c0, label %ph, label %ret, !prof !0
@@ -76,12 +47,61 @@ ret:                                              ; preds = %entry
   ret void
 }
 
+; Keep explicit trip-count metadata for multi-exit loops. Rotation decrements
+; the attribute but does not infer a trip count from branch weights without it.
+define void @explicit_trip_count(ptr %p, i32 %start, i32 %limit) {
+entry:
+  br label %header
+
+header:
+  %iv = phi i32 [ %start, %entry ], [ %iv.next, %latch ]
+  %hcmp = icmp eq i32 %iv, %limit
+  br i1 %hcmp, label %exit1, label %body, !prof !1, !llvm.loop !4
+
+body:
+  %v = load i32, ptr %p, align 4
+  %bcmp = icmp slt i32 %v, 0
+  br i1 %bcmp, label %exit2, label %latch, !prof !2
+
+latch:
+  %iv.next = add i32 %iv, 1
+  br label %header
+
+exit1:
+  ret void
+
+exit2:
+  ret void
+}
+
 !0 = !{!"branch_weights", i32 42, i32 2}
 !1 = !{!"branch_weights", i32 200, i32 9800}
 !2 = !{!"branch_weights", i32 10, i32 9790}
+!3 = !{!"function_entry_count", i64 44}
+!4 = distinct !{!4, !5}
+!5 = !{!"llvm.loop.estimated_trip_count", i32 10}
+
+; BFI-BEFORE-LABEL: block-frequency-info: f
+; BFI-BEFORE: - body: {{.*}} count = 1960
+; BFI-BEFORE: - exit1: {{.*}} count = 40
+; BFI-BEFORE: - exit2: {{.*}} count = 2
+; BFI-BEFORE: - ret: {{.*}} count = 2
+
+; BFI-AFTER-LABEL: block-frequency-info: f
+; BFI-AFTER: - body: {{.*}} count = 1960
+; BFI-AFTER: - exit1: {{.*}} count = 40
+; BFI-AFTER: - exit2: {{.*}} count = 2
+; BFI-AFTER: - ret: {{.*}} count = 2
+
+; CHECK-LABEL: define void @f(
+
+; CHECK:      ph:
+; CHECK:        br i1 %{{.*}}, label %{{.*}}, label %{{.*}}.lr.ph, !prof [[WEIGHTS:![0-9]+]]
+
+; CHECK:        br i1 %{{.*}}, label %{{.*}}, label %body, !prof [[WEIGHTS]]{{$}}
 
-;.
-; CHECK: [[PROF0]] = !{!"branch_weights", i32 42, i32 2}
-; CHECK: [[PROF1]] = !{!"branch_weights", i32 200, i32 9800}
-; CHECK: [[PROF2]] = !{!"branch_weights", i32 10, i32 9790}
-;.
+; CHECK-LABEL: define void @explicit_trip_count(
+; CHECK: br i1 %{{.*}}, label %{{.*}}, label %body, !prof {{![0-9]+}}, !llvm.loop [[EXPLICIT_LOOP:![0-9]+]]
+; CHECK-DAG: [[WEIGHTS]] = !{!"branch_weights", i32 200, i32 9800}
+; CHECK-DAG: [[EXPLICIT_LOOP]] = distinct !{[[EXPLICIT_LOOP]], [[EXPLICIT_TC:![0-9]+]]}
+; CHECK-DAG: [[EXPLICIT_TC]] = !{!"llvm.loop.estimated_trip_count", i32 9}
diff --git a/llvm/test/Transforms/LoopRotate/update-branch-weights.ll b/llvm/test/Transforms/LoopRotate/update-branch-weights.ll
index 912f70b34bb82..df5471788ae0d 100644
--- a/llvm/test/Transforms/LoopRotate/update-branch-weights.ll
+++ b/llvm/test/Transforms/LoopRotate/update-branch-weights.ll
@@ -41,11 +41,11 @@ define void @func0() !prof !0 {
 ; IR-NEXT:    store volatile i32 [[I11]], ptr @g, align 4
 ; IR-NEXT:    [[I1_INC]] = add i32 [[I11]], 1
 ; IR-NEXT:    [[CMP1:%.*]] = icmp slt i32 [[I1_INC]], 3
-; IR-NEXT:    br i1 [[CMP1]], label %[[INNER_LOOP_BODY]], label %[[INNER_LOOP_EXIT]], !prof [[PROF1:![0-9]+]]
+; IR-NEXT:    br i1 [[CMP1]], label %[[INNER_LOOP_BODY]], label %[[INNER_LOOP_EXIT]], !prof [[PROF1:![0-9]+]], !llvm.loop [[LOOP2:![0-9]+]]
 ; IR:       [[INNER_LOOP_EXIT]]:
 ; IR-NEXT:    [[I0_INC]] = add i32 [[I02]], 1
 ; IR-NEXT:    [[CMP0:%.*]] = icmp slt i32 [[I0_INC]], 1000
-; IR-NEXT:    br i1 [[CMP0]], label %[[OUTER_LOOP_BODY]], label %[[OUTER_LOOP_EXIT:.*]], !prof [[PROF2:![0-9]+]]
+; IR-NEXT:    br i1 [[CMP0]], label %[[OUTER_LOOP_BODY]], label %[[OUTER_LOOP_EXIT:.*]], !prof [[PROF4:![0-9]+]], !llvm.loop [[LOOP5:![0-9]+]]
 ; IR:       [[OUTER_LOOP_EXIT]]:
 ; IR-NEXT:    ret void
 ;
@@ -95,10 +95,10 @@ outer_loop_exit:
 ; executed more often than header.
 define void @func1(i32 %n) !prof !3 {
 ; IR-LABEL: define void @func1(
-; IR-SAME: i32 [[N:%.*]]) !prof [[PROF3:![0-9]+]] {
+; IR-SAME: i32 [[N:%.*]]) !prof [[PROF7:![0-9]+]] {
 ; IR-NEXT:  [[ENTRY:.*:]]
 ; IR-NEXT:    [[CMP1:%.*]] = icmp slt i32 0, [[N]]
-; IR-NEXT:    br i1 [[CMP1]], label %[[LOOP_BODY_LR_PH:.*]], label %[[LOOP_EXIT:.*]], !prof [[PROF4:![0-9]+]]
+; IR-NEXT:    br i1 [[CMP1]], label %[[LOOP_BODY_LR_PH:.*]], label %[[LOOP_EXIT:.*]], !prof [[PROF8:![0-9]+]]
 ; IR:       [[LOOP_BODY_LR_PH]]:
 ; IR-NEXT:    br label %[[LOOP_BODY:.*]]
 ; IR:       [[LOOP_BODY]]:
@@ -106,7 +106,7 @@ define void @func1(i32 %n) !prof !3 {
 ; IR-NEXT:    store volatile i32 [[I2]], ptr @g, align 4
 ; IR-NEXT:    [[I_INC]] = add i32 [[I2]], 1
 ; IR-NEXT:    [[CMP:%.*]] = icmp slt i32 [[I_INC]], [[N]]
-; IR-NEXT:    br i1 [[CMP]], label %[[LOOP_BODY]], label %[[LOOP_HEADER_LOOP_EXIT_CRIT_EDGE:.*]], !prof [[PROF4]]
+; IR-NEXT:    br i1 [[CMP]], label %[[LOOP_BODY]], label %[[LOOP_HEADER_LOOP_EXIT_CRIT_EDGE:.*]], !prof [[PROF8]], !llvm.loop [[LOOP9:![0-9]+]]
 ; IR:       [[LOOP_HEADER_LOOP_EXIT_CRIT_EDGE]]:
 ; IR-NEXT:    br label %[[LOOP_EXIT]]
 ; IR:       [[LOOP_EXIT]]:
@@ -145,10 +145,10 @@ loop_exit:
 ; loop-exit count is higher than backedge count.
 define void @func2(i32 %n) !prof !3 {
 ; IR-LABEL: define void @func2(
-; IR-SAME: i32 [[N:%.*]]) !prof [[PROF3]] {
+; IR-SAME: i32 [[N:%.*]]) !prof [[PROF7]] {
 ; IR-NEXT:  [[ENTRY:.*:]]
 ; IR-NEXT:    [[CMP1:%.*]] = icmp slt i32 0, [[N]]
-; IR-NEXT:    br i1 [[CMP1]], label %[[LOOP_EXIT:.*]], label %[[LOOP_BODY_LR_PH:.*]], !prof [[PROF5:![0-9]+]]
+; IR-NEXT:    br i1 [[CMP1]], label %[[LOOP_EXIT:.*]], label %[[LOOP_BODY_LR_PH:.*]], !prof [[PROF11:![0-9]+]]
 ; IR:       [[LOOP_BODY_LR_PH]]:
 ; IR-NEXT:    br label %[[LOOP_BODY:.*]]
 ; IR:       [[LOOP_BODY]]:
@@ -156,7 +156,7 @@ define void @func2(i32 %n) !prof !3 {
 ; IR-NEXT:    store volatile i32 [[I2]], ptr @g, align 4
 ; IR-NEXT:    [[I_INC]] = add i32 [[I2]], 1
 ; IR-NEXT:    [[CMP:%.*]] = icmp slt i32 [[I_INC]], [[N]]
-; IR-NEXT:    br i1 [[CMP]], label %[[LOOP_HEADER_LOOP_EXIT_CRIT_EDGE:.*]], label %[[LOOP_BODY]], !prof [[PROF5]]
+; IR-NEXT:    br i1 [[CMP]], label %[[LOOP_HEADER_LOOP_EXIT_CRIT_EDGE:.*]], label %[[LOOP_BODY]], !prof [[PROF11]], !llvm.loop [[LOOP12:![0-9]+]]
 ; IR:       [[LOOP_HEADER_LOOP_EXIT_CRIT_EDGE]]:
 ; IR-NEXT:    br label %[[LOOP_EXIT]]
 ; IR:       [[LOOP_EXIT]]:
@@ -194,10 +194,10 @@ loop_exit:
 
 define void @func3_zero_branch_weight(i32 %n) !prof !3 {
 ; IR-LABEL: define void @func3_zero_branch_weight(
-; IR-SAME: i32 [[N:%.*]]) !prof [[PROF3]] {
+; IR-SAME: i32 [[N:%.*]]) !prof [[PROF7]] {
 ; IR-NEXT:  [[ENTRY:.*:]]
 ; IR-NEXT:    [[CMP1:%.*]] = icmp slt i32 0, [[N]]
-; IR-NEXT:    br i1 [[CMP1]], label %[[LOOP_EXIT:.*]], label %[[LOOP_BODY_LR_PH:.*]], !prof [[PROF6:![0-9]+]]
+; IR-NEXT:    br i1 [[CMP1]], label %[[LOOP_EXIT:.*]], label %[[LOOP_BODY_LR_PH:.*]], !prof [[PROF14:![0-9]+]]
 ; IR:       [[LOOP_BODY_LR_PH]]:
 ; IR-NEXT:    br label %[[LOOP_BODY:.*]]
 ; IR:       [[LOOP_BODY]]:
@@ -205,7 +205,7 @@ define void @func3_zero_branch_weight(i32 %n) !prof !3 {
 ; IR-NEXT:    store volatile i32 [[I2]], ptr @g, align 4
 ; IR-NEXT:    [[I_INC]] = add i32 [[I2]], 1
 ; IR-NEXT:    [[CMP:%.*]] = icmp slt i32 [[I_INC]], [[N]]
-; IR-NEXT:    br i1 [[CMP]], label %[[LOOP_HEADER_LOOP_EXIT_CRIT_EDGE:.*]], label %[[LOOP_BODY]], !prof [[PROF6]]
+; IR-NEXT:    br i1 [[CMP]], label %[[LOOP_HEADER_LOOP_EXIT_CRIT_EDGE:.*]], label %[[LOOP_BODY]], !prof [[PROF14]]
 ; IR:       [[LOOP_HEADER_LOOP_EXIT_CRIT_EDGE]]:
 ; IR-NEXT:    br label %[[LOOP_EXIT]]
 ; IR:       [[LOOP_EXIT]]:
@@ -230,10 +230,10 @@ loop_exit:
 
 define void @func4_zero_branch_weight(i32 %n) !prof !3 {
 ; IR-LABEL: define void @func4_zero_branch_weight(
-; IR-SAME: i32 [[N:%.*]]) !prof [[PROF3]] {
+; IR-SAME: i32 [[N:%.*]]) !prof [[PROF7]] {
 ; IR-NEXT:  [[ENTRY:.*:]]
 ; IR-NEXT:    [[CMP1:%.*]] = icmp slt i32 0, [[N]]
-; IR-NEXT:    br i1 [[CMP1]], label %[[LOOP_EXIT:.*]], label %[[LOOP_BODY_LR_PH:.*]], !prof [[PROF7:![0-9]+]]
+; IR-NEXT:    br i1 [[CMP1]], label %[[LOOP_EXIT:.*]], label %[[LOOP_BODY_LR_PH:.*]], !prof [[PROF15:![0-9]+]]
 ; IR:       [[LOOP_BODY_LR_PH]]:
 ; IR-NEXT:    br label %[[LOOP_BODY:.*]]
 ; IR:       [[LOOP_BODY]]:
@@ -241,7 +241,7 @@ define void @func4_zero_branch_weight(i32 %n) !prof !3 {
 ; IR-NEXT:    store volatile i32 [[I2]], ptr @g, align 4
 ; IR-NEXT:    [[I_INC]] = add i32 [[I2]], 1
 ; IR-NEXT:    [[CMP:%.*]] = icmp slt i32 [[I_INC]], [[N]]
-; IR-NEXT:    br i1 [[CMP]], label %[[LOOP_HEADER_LOOP_EXIT_CRIT_EDGE:.*]], label %[[LOOP_BODY]], !prof [[PROF7]]
+; IR-NEXT:    br i1 [[CMP]], label %[[LOOP_HEADER_LOOP_EXIT_CRIT_EDGE:.*]], label %[[LOOP_BODY]], !prof [[PROF15]], !llvm.loop [[LOOP16:![0-9]+]]
 ; IR:       [[LOOP_HEADER_LOOP_EXIT_CRIT_EDGE]]:
 ; IR-NEXT:    br label %[[LOOP_EXIT]]
 ; IR:       [[LOOP_EXIT]]:
@@ -266,10 +266,10 @@ loop_exit:
 
 define void @func5_zero_branch_weight(i32 %n) !prof !3 {
 ; IR-LABEL: define void @func5_zero_branch_weight(
-; IR-SAME: i32 [[N:%.*]]) !prof [[PROF3]] {
+; IR-SAME: i32 [[N:%.*]]) !prof [[PROF7]] {
 ; IR-NEXT:  [[ENTRY:.*:]]
 ; IR-NEXT:    [[CMP1:%.*]] = icmp slt i32 0, [[N]]
-; IR-NEXT:    br i1 [[CMP1]], label %[[LOOP_EXIT:.*]], label %[[LOOP_BODY_LR_PH:.*]], !prof [[PROF8:![0-9]+]]
+; IR-NEXT:    br i1 [[CMP1]], label %[[LOOP_EXIT:.*]], label %[[LOOP_BODY_LR_PH:.*]], !prof [[PROF17:![0-9]+]]
 ; IR:       [[LOOP_BODY_LR_PH]]:
 ; IR-NEXT:    br label %[[LOOP_BODY:.*]]
 ; IR:       [[LOOP_BODY]]:
@@ -277,7 +277,7 @@ define void @func5_zero_branch_weight(i32 %n) !prof !3 {
 ; IR-NEXT:    store volatile i32 [[I2]], ptr @g, align 4
 ; IR-NEXT:    [[I_INC]] = add i32 [[I2]], 1
 ; IR-NEXT:    [[CMP:%.*]] = icmp slt i32 [[I_INC]], [[N]]
-; IR-NEXT:    br i1 [[CMP]], label %[[LOOP_HEADER_LOOP_EXIT_CRIT_EDGE:.*]], label %[[LOOP_BODY]], !prof [[PROF8]]
+; IR-NEXT:    br i1 [[CMP]], label %[[LOOP_HEADER_LOOP_EXIT_CRIT_EDGE:.*]], label %[[LOOP_BODY]], !prof [[PROF17]]
 ; IR:       [[LOOP_HEADER_LOOP_EXIT_CRIT_EDGE]]:
 ; IR-NEXT:    br label %[[LOOP_EXIT]]
 ; IR:       [[LOOP_EXIT]]:
@@ -317,7 +317,7 @@ loop_exit:
 ; However this may not hold for Sample-based PGO.
 define void @func6_inaccurate_branch_weight() !prof !3 {
 ; IR-LABEL: define void @func6_inaccurate_branch_weight(
-; IR-SAME: ) !prof [[PROF3]] {
+; IR-SAME: ) !prof [[PROF7]] {
 ; IR-NEXT:  [[ENTRY:.*]]:
 ; IR-NEXT:    br label %[[LOOP_BODY:.*]]
 ; IR:       [[LOOP_BODY]]:
@@ -325,7 +325,7 @@ define void @func6_inaccurate_branch_weight() !prof !3 {
 ; IR-NEXT:    store volatile i32 [[I1]], ptr @g, align 4
 ; IR-NEXT:    [[I_INC]] = add i32 [[I1]], 1
 ; IR-NEXT:    [[CMP:%.*]] = icmp slt i32 [[I_INC]], 2
-; IR-NEXT:    br i1 [[CMP]], label %[[LOOP_BODY]], label %[[LOOP_EXIT:.*]], !prof [[PROF9:![0-9]+]]
+; IR-NEXT:    br i1 [[CMP]], label %[[LOOP_BODY]], label %[[LOOP_EXIT:.*]], !prof [[PROF18:![0-9]+]], !llvm.loop [[LOOP19:![0-9]+]]
 ; IR:       [[LOOP_EXIT]]:
 ; IR-NEXT:    ret void
 ;
@@ -346,6 +346,102 @@ loop_exit:
   ret void
 }
 
+; Record the reduced trip count when rotation folds a single-exit guard.
+
+define void @folded_single_exit_inferred_trip_count() {
+; IR-LABEL: define void @folded_single_exit_inferred_trip_count() {
+; IR-NEXT:  [[ENTRY:.*]]:
+; IR-NEXT:    br label %[[LOOP_BODY:.*]]
+; IR:       [[LOOP_BODY]]:
+; IR-NEXT:    [[I1:%.*]] = phi i32 [ 0, %[[ENTRY]] ], [ [[I_INC:%.*]], %[[LOOP_BODY]] ]
+; IR-NEXT:    store volatile i32 [[I1]], ptr @g, align 4
+; IR-NEXT:    [[I_INC]] = add i32 [[I1]], 1
+; IR-NEXT:    [[CMP:%.*]] = icmp slt i32 [[I_INC]], 10
+; IR-NEXT:    br i1 [[CMP]], label %[[LOOP_BODY]], label %[[LOOP_EXIT:.*]], !prof [[PROF4]], !llvm.loop [[LOOP21:![0-9]+]]
+; IR:       [[LOOP_EXIT]]:
+; IR-NEXT:    ret void
+;
+entry:
+  br label %loop_header
+
+loop_header:
+  %i = phi i32 [ 0, %entry ], [ %i.inc, %loop_body ]
+  %cmp = icmp slt i32 %i, 10
+  br i1 %cmp, label %loop_body, label %loop_exit, !prof !1
+
+loop_body:
+  store volatile i32 %i, ptr @g, align 4
+  %i.inc = add i32 %i, 1
+  br label %loop_header
+
+loop_exit:
+  ret void
+}
+
+; Decrement an explicit trip count when rotation folds the guard.
+
+define void @folded_single_exit_explicit_trip_count() {
+; IR-LABEL: define void @folded_single_exit_explicit_trip_count() {
+; IR-NEXT:  [[ENTRY:.*]]:
+; IR-NEXT:    br label %[[LOOP_BODY:.*]]
+; IR:       [[LOOP_BODY]]:
+; IR-NEXT:    [[I1:%.*]] = phi i32 [ 0, %[[ENTRY]] ], [ [[I_INC:%.*]], %[[LOOP_BODY]] ]
+; IR-NEXT:    store volatile i32 [[I1]], ptr @g, align 4
+; IR-NEXT:    [[I_INC]] = add i32 [[I1]], 1
+; IR-NEXT:    [[CMP:%.*]] = icmp slt i32 [[I_INC]], 10
+; IR-NEXT:    br i1 [[CMP]], label %[[LOOP_BODY]], label %[[LOOP_EXIT:.*]], !prof [[PROF4]], !llvm.loop [[LOOP22:![0-9]+]]
+; IR:       [[LOOP_EXIT]]:
+; IR-NEXT:    ret void
+;
+entry:
+  br label %loop_header
+
+loop_header:
+  %i = phi i32 [ 0, %entry ], [ %i.inc, %loop_body ]
+  %cmp = icmp slt i32 %i, 10
+  br i1 %cmp, label %loop_body, label %loop_exit, !prof !1, !llvm.loop !10
+
+loop_body:
+  store volatile i32 %i, ptr @g, align 4
+  %i.inc = add i32 %i, 1
+  br label %loop_header
+
+loop_exit:
+  ret void
+}
+
+; Do not underflow an explicit zero trip count.
+
+define void @folded_single_exit_zero_trip_count() {
+; IR-LABEL: define void @folded_single_exit_zero_trip_count() {
+; IR-NEXT:  [[ENTRY:.*]]:
+; IR-NEXT:    br label %[[LOOP_BODY:.*]]
+; IR:       [[LOOP_BODY]]:
+; IR-NEXT:    [[I1:%.*]] = phi i32 [ 0, %[[ENTRY]] ], [ [[I_INC:%.*]], %[[LOOP_BODY]] ]
+; IR-NEXT:    store volatile i32 [[I1]], ptr @g, align 4
+; IR-NEXT:    [[I_INC]] = add i32 [[I1]], 1
+; IR-NEXT:    [[CMP:%.*]] = icmp slt i32 [[I_INC]], 10
+; IR-NEXT:    br i1 [[CMP]], label %[[LOOP_BODY]], label %[[LOOP_EXIT:.*]], !prof [[PROF4]], !llvm.loop [[LOOP24:![0-9]+]]
+; IR:       [[LOOP_EXIT]]:
+; IR-NEXT:    ret void
+;
+entry:
+  br label %loop_header
+
+loop_header:
+  %i = phi i32 [ 0, %entry ], [ %i.inc, %loop_body ]
+  %cmp = icmp slt i32 %i, 10
+  br i1 %cmp, label %loop_body, label %loop_exit, !prof !1, !llvm.loop !12
+
+loop_body:
+  store volatile i32 %i, ptr @g, align 4
+  %i.inc = add i32 %i, 1
+  br label %loop_header
+
+loop_exit:
+  ret void
+}
+
 !0 = !{!"function_entry_count", i64 1}
 !1 = !{!"branch_weights", i32 1000, i32 1}
 !2 = !{!"branch_weights", i32 3000, i32 1000}
@@ -356,16 +452,35 @@ loop_exit:
 !7 = !{!"branch_weights", i32 1, i32 0}
 !8 = !{!"branch_weights", i32 0, i32 0}
 !9 = !{!"branch_weights", i32 1023, i32 1024}
+!10 = distinct !{!10, !11}
+!11 = !{!"llvm.loop.estimated_trip_count", i32 10}
+!12 = distinct !{!12, !13}
+!13 = !{!"llvm.loop.estimated_trip_count", i32 0}
 
 ;.
 ; IR: [[PROF0]] = !{!"function_entry_count", i64 1}
 ; IR: [[PROF1]] = !{!"branch_weights", i32 2000, i32 1000}
-; IR: [[PROF2]] = !{!"branch_weights", i32 999, i32 1}
-; IR: [[PROF3]] = !{!"function_entry_count", i64 1024}
-; IR: [[PROF4]] = !{!"branch_weights", i32 40, i32 2}
-; IR: [[PROF5]] = !{!"branch_weights", i32 10240, i32 320}
-; IR: [[PROF6]] = !{!"branch_weights", i32 0, i32 1}
-; IR: [[PROF7]] = !{!"branch_weights", i32 1, i32 0}
-; IR: [[PROF8]] = !{!"branch_weights", i32 0, i32 0}
-; IR: [[PROF9]] = !{!"branch_weights", i32 0, i32 1024}
+; IR: [[LOOP2]] = distinct !{[[LOOP2]], [[META3:![0-9]+]]}
+; IR: [[META3]] = !{!"llvm.loop.estimated_trip_count", i32 3}
+; IR: [[PROF4]] = !{!"branch_weights", i32 999, i32 1}
+; IR: [[LOOP5]] = distinct !{[[LOOP5]], [[META6:![0-9]+]]}
+; IR: [[META6]] = !{!"llvm.loop.estimated_trip_count", i32 1000}
+; IR: [[PROF7]] = !{!"function_entry_count", i64 1024}
+; IR: [[PROF8]] = !{!"branch_weights", i32 40, i32 2}
+; IR: [[LOOP9]] = distinct !{[[LOOP9]], [[META10:![0-9]+]]}
+; IR: [[META10]] = !{!"llvm.loop.estimated_trip_count", i32 20}
+; IR: [[PROF11]] = !{!"branch_weights", i32 10240, i32 320}
+; IR: [[LOOP12]] = distinct !{[[LOOP12]], [[META13:![0-9]+]]}
+; IR: [[META13]] = !{!"llvm.loop.estimated_trip_count", i32 0}
+; IR: [[PROF14]] = !{!"branch_weights", i32 0, i32 1}
+; IR: [[PROF15]] = !{!"branch_weights", i32 1, i32 0}
+; IR: [[LOOP16]] = distinct !{[[LOOP16]], [[META13]]}
+; IR: [[PROF17]] = !{!"branch_weights", i32 0, i32 0}
+; IR: [[PROF18]] = !{!"branch_weights", i32 0, i32 1024}
+; IR: [[LOOP19]] = distinct !{[[LOOP19]], [[META20:![0-9]+]]}
+; IR: [[META20]] = !{!"llvm.loop.estimated_trip_count", i32 1}
+; IR: [[LOOP21]] = distinct !{[[LOOP21]], [[META6]]}
+; IR: [[LOOP22]] = distinct !{[[LOOP22]], [[META23:![0-9]+]]}
+; IR: [[META23]] = !{!"llvm.loop.estimated_trip_count", i32 9}
+; IR: [[LOOP24]] = distinct !{[[LOOP24]], [[META13]]}
 ;.

>From c7d982012a9fbaec9af59d959030f1ea5ea43bc9 Mon Sep 17 00:00:00 2001
From: Alok Kumar Sharma <AlokKumar.Sharma at amd.com>
Date: Wed, 19 Aug 2026 23:29:36 +0530
Subject: [PATCH 3/3] [LoopRotate] Derive folded-guard latch weights for
 multi-exit loops from BFI

When rotation folds the first-iteration guard, the single-exit
latch-weight adjustment overestimates the backedge for multi-exit loops.
While block frequencies still reflect the pre-rotation layout, derive
latch weights from BFI:
  exit     = header frequency - body frequency
  backedge = body frequency - preheader frequency
Fit those values onto the latch branch metadata so aggregate BFI is
preserved. If the frequencies are inconsistent, keep the original latch
weights instead of applying the single-exit fallback.
---
 .../Transforms/Utils/LoopRotationUtils.cpp    | 100 ++++++--
 .../folded-guard-multi-exit-branch-weights.ll | 237 ++++++++++++++++++
 ...olded-guard-multi-exit-inconsistent-bfi.ll |  60 +++++
 .../LoopRotate/multi-exit-branch-weights.ll   |  77 +++++-
 4 files changed, 447 insertions(+), 27 deletions(-)
 create mode 100644 llvm/test/Transforms/LoopRotate/folded-guard-multi-exit-branch-weights.ll
 create mode 100644 llvm/test/Transforms/LoopRotate/folded-guard-multi-exit-inconsistent-bfi.ll

diff --git a/llvm/lib/Transforms/Utils/LoopRotationUtils.cpp b/llvm/lib/Transforms/Utils/LoopRotationUtils.cpp
index 17fb32a1536df..86ee153e5f679 100644
--- a/llvm/lib/Transforms/Utils/LoopRotationUtils.cpp
+++ b/llvm/lib/Transforms/Utils/LoopRotationUtils.cpp
@@ -12,10 +12,12 @@
 
 #include <optional>
 
-#include "llvm/Transforms/Utils/LoopRotationUtils.h"
 #include "llvm/ADT/Statistic.h"
 #include "llvm/Analysis/AssumptionCache.h"
+#include "llvm/Analysis/BlockFrequencyInfo.h"
+#include "llvm/Analysis/BranchProbabilityInfo.h"
 #include "llvm/Analysis/CodeMetrics.h"
+#include "llvm/Analysis/CycleAnalysis.h"
 #include "llvm/Analysis/DomTreeUpdater.h"
 #include "llvm/Analysis/InstructionSimplify.h"
 #include "llvm/Analysis/LoopInfo.h"
@@ -35,6 +37,7 @@
 #include "llvm/Transforms/Utils/BasicBlockUtils.h"
 #include "llvm/Transforms/Utils/Cloning.h"
 #include "llvm/Transforms/Utils/Local.h"
+#include "llvm/Transforms/Utils/LoopRotationUtils.h"
 #include "llvm/Transforms/Utils/LoopUtils.h"
 #include "llvm/Transforms/Utils/SSAUpdater.h"
 #include "llvm/Transforms/Utils/ValueMapper.h"
@@ -51,6 +54,11 @@ STATISTIC(NumInstrsDuplicated,
           "Number of instructions cloned into loop preheader");
 
 namespace {
+struct RotatedLatchWeights {
+  uint64_t Exit;
+  uint64_t Backedge;
+};
+
 /// A simple loop rotation transformation.
 class LoopRotate {
   const unsigned MaxHeaderSize;
@@ -210,9 +218,38 @@ static bool profitableToRotateLoopExitingLatch(Loop *L, ScalarEvolution *SE) {
   return false;
 }
 
-static void updateBranchWeights(CondBrInst &PreHeaderBI, CondBrInst &LoopBI,
-                                bool HasConditionalPreHeader,
-                                bool SuccsSwapped) {
+// Rebuild local BFI before moveToHeader/edge splitting, while block
+// frequencies still reflect the pre-rotation layout. Only used for folded-guard
+// multi-exit loops, so the extra analysis cost is limited to that case.
+static std::optional<RotatedLatchWeights> getFoldedMultiExitLatchWeightsFromBFI(
+    Function &F, const SimplifyQuery &SQ, DominatorTree *DT,
+    BasicBlock *OrigPreheader, BasicBlock *OrigHeader, BasicBlock *NewHeader) {
+  CycleInfo CI;
+  CI.compute(F);
+  BranchProbabilityInfo BPI(F, CI, SQ.TLI, DT);
+  BlockFrequencyInfo BFI(F, BPI, CI);
+  uint64_t PreheaderFreq = BFI.getBlockFreq(OrigPreheader).getFrequency();
+  uint64_t HeaderFreq = BFI.getBlockFreq(OrigHeader).getFrequency();
+  uint64_t NewHeaderFreq = BFI.getBlockFreq(NewHeader).getFrequency();
+  if (HeaderFreq < NewHeaderFreq || NewHeaderFreq < PreheaderFreq) {
+    LLVM_DEBUG(dbgs() << "LoopRotation: inconsistent BFI for folded "
+                      << "multi-exit latch weights\n");
+    return std::nullopt;
+  }
+  uint64_t ExitFreq = HeaderFreq - NewHeaderFreq;
+  uint64_t BackedgeFreq = NewHeaderFreq - PreheaderFreq;
+  if (!ExitFreq && !BackedgeFreq) {
+    LLVM_DEBUG(dbgs() << "LoopRotation: zero BFI-derived latch weights for "
+                      << "folded multi-exit loop\n");
+    return std::nullopt;
+  }
+  return RotatedLatchWeights{ExitFreq, BackedgeFreq};
+}
+
+static void updateBranchWeights(
+    CondBrInst &PreHeaderBI, CondBrInst &LoopBI, bool HasConditionalPreHeader,
+    bool SuccsSwapped, bool IsMultiExitLoop,
+    std::optional<RotatedLatchWeights> MultiExitFoldedGuardWeights) {
   MDNode *WeightMD = getBranchWeightMDNode(PreHeaderBI);
   if (WeightMD == nullptr)
     return;
@@ -243,6 +280,25 @@ static void updateBranchWeights(CondBrInst &PreHeaderBI, CondBrInst &LoopBI,
   if (HasConditionalPreHeader)
     return;
 
+  // Use BFI because header weights miss side exits.
+  if (MultiExitFoldedGuardWeights) {
+    const uint64_t LoopBIWeights[] = {
+        SuccsSwapped ? MultiExitFoldedGuardWeights->Backedge
+                     : MultiExitFoldedGuardWeights->Exit,
+        SuccsSwapped ? MultiExitFoldedGuardWeights->Exit
+                     : MultiExitFoldedGuardWeights->Backedge,
+    };
+    setFittedBranchWeights(LoopBI, LoopBIWeights, /*IsExpected=*/false);
+    return;
+  }
+
+  // Keep the original weights for multi-exit loops without usable BFI.
+  if (IsMultiExitLoop) {
+    LLVM_DEBUG(dbgs() << "LoopRotation: keeping original latch weights for "
+                      << "multi-exit loop without usable BFI\n");
+    return;
+  }
+
   SmallVector<uint32_t, 2> Weights;
   extractFromBranchWeightMD32(WeightMD, Weights);
   if (Weights.size() != 2)
@@ -402,6 +458,10 @@ bool LoopRotate::rotateLoop(Loop *L, bool SimplifiedLatch) {
   assert(L->contains(NewHeader) && !L->contains(Exit) &&
          "Unable to determine loop header and exit blocks");
 
+  // getExitingBlock() is null when there is no unique dedicated exit block.
+  bool IsMultiExitLoop = !L->getExitingBlock();
+  std::optional<RotatedLatchWeights> MultiExitFoldedGuardWeights;
+
   // This code assumes that the new header has exactly one predecessor.
   // Remove any single-entry PHI nodes in it.
   assert(NewHeader->getSinglePredecessor() &&
@@ -522,7 +582,6 @@ bool LoopRotate::rotateLoop(Loop *L, bool SimplifiedLatch) {
       mapAtomInstance(DL, ValueMap);
 
     C->insertBefore(LoopEntryBranch->getIterator());
-
     ++NumInstrsDuplicated;
 
     if (!NextDbgInsts.empty()) {
@@ -568,6 +627,21 @@ bool LoopRotate::rotateLoop(Loop *L, bool SimplifiedLatch) {
     }
   }
 
+  auto *PHBI = cast<CondBrInst>(ValueMap.lookup(BI));
+  const Value *Cond = PHBI->getCondition();
+  const bool HasConditionalPreHeader =
+      !isa<ConstantInt>(Cond) ||
+      PHBI->getSuccessor(cast<ConstantInt>(Cond)->isZero()) != NewHeader;
+
+  // Derive latch weights before moveToHeader/edge splitting, while block
+  // frequencies still reflect the pre-rotation layout.
+  // exit = header - body, backedge = body - preheader.
+  if (!HasConditionalPreHeader && IsMultiExitLoop &&
+      getBranchWeightMDNode(*BI)) {
+    MultiExitFoldedGuardWeights = getFoldedMultiExitLatchWeightsFromBFI(
+        *OrigHeader->getParent(), SQ, DT, OrigPreheader, OrigHeader, NewHeader);
+  }
+
   if (!NoAliasDeclInstructions.empty()) {
     // There are noalias scope declarations:
     // (general):
@@ -695,24 +769,20 @@ bool LoopRotate::rotateLoop(Loop *L, bool SimplifiedLatch) {
   // then we fold away the cond branch to an uncond branch.  This simplifies the
   // loop in cases important for nested loops, and it also means we don't have
   // to split as many edges.
-  CondBrInst *PHBI = cast<CondBrInst>(OrigPreheader->getTerminator());
-  const Value *Cond = PHBI->getCondition();
-  const bool HasConditionalPreHeader =
-      !isa<ConstantInt>(Cond) ||
-      PHBI->getSuccessor(cast<ConstantInt>(Cond)->isZero()) != NewHeader;
+  assert(PHBI == OrigPreheader->getTerminator() &&
+         "Unexpected cloned preheader branch");
 
   // Save the trip count before updating branch weights. Only infer it from
-  // profile weights for single-exit loops, but still read and decrement
-  // explicit llvm.loop.estimated_trip_count on any loop.
-  // getExitingBlock() is null when there is no unique dedicated exit block.
-  bool IsMultiExitLoop = !L->getExitingBlock();
+  // profile weights for single-exit loops (see IsMultiExitLoop above), but
+  // still read and decrement explicit llvm.loop.estimated_trip_count.
   bool HasExplicitEstimatedTripCount =
       getOptionalIntLoopAttribute(L, LLVMLoopEstimatedTripCount).has_value();
   std::optional<unsigned> EstimatedTripCount;
   if (!IsMultiExitLoop || HasExplicitEstimatedTripCount)
     EstimatedTripCount = getLoopEstimatedTripCount(L);
 
-  updateBranchWeights(*PHBI, *BI, HasConditionalPreHeader, BISuccsSwapped);
+  updateBranchWeights(*PHBI, *BI, HasConditionalPreHeader, BISuccsSwapped,
+                      IsMultiExitLoop, MultiExitFoldedGuardWeights);
 
   if (HasConditionalPreHeader) {
     // The conditional branch can't be folded, handle the general case.
diff --git a/llvm/test/Transforms/LoopRotate/folded-guard-multi-exit-branch-weights.ll b/llvm/test/Transforms/LoopRotate/folded-guard-multi-exit-branch-weights.ll
new file mode 100644
index 0000000000000..eafdd5df62829
--- /dev/null
+++ b/llvm/test/Transforms/LoopRotate/folded-guard-multi-exit-branch-weights.ll
@@ -0,0 +1,237 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --no-generate-body-for-unused-prefixes --version 6
+; RUN: opt < %s -passes=loop-rotate -S | FileCheck %s
+; RUN: opt < %s -passes='print<block-freq>' -disable-output 2>&1 | FileCheck %s --check-prefix=BFI-BEFORE
+; RUN: opt < %s -passes='loop(loop-rotate),print<block-freq>' -disable-output 2>&1 | FileCheck %s --check-prefix=BFI-AFTER
+;
+; The guard folds, making the first body iteration unconditional. Since %body
+; has another exit, use BFI instead of the single-exit adjustment.
+;
+;   preheader frequency = 1200
+;   header frequency    = 10000
+;   body frequency      = 9800
+;
+;   exit     = header - body      = 10000 - 9800 = 200
+;   backedge = body - preheader   = 9800 - 1200  = 8600
+;   setFittedBranchWeights preserves the 200:8600 ratio in uint32 metadata;
+;   IR may print the scaled values as large or negative i32 constants.
+;
+; Preserve BFI. Do not infer a trip count from weights that miss a side exit.
+
+define void @folded_guard_multi_exit(ptr %p, i32 %cond) !prof !0 {
+; CHECK-LABEL: define void @folded_guard_multi_exit(
+; CHECK-SAME: ptr [[P:%.*]], i32 [[COND:%.*]]) !prof [[PROF0:![0-9]+]] {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[ENTER:%.*]] = icmp ne i32 [[COND]], 0
+; CHECK-NEXT:    br i1 [[ENTER]], label %[[PH:.*]], label %[[RET:.*]], !prof [[PROF1:![0-9]+]]
+; CHECK:       [[PH]]:
+; CHECK-NEXT:    br label %[[BODY:.*]]
+; CHECK:       [[HEADER:.*]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i32 [ [[NEXT:%.*]], %[[BODY]] ]
+; CHECK-NEXT:    [[DONE:%.*]] = icmp eq i32 [[IV]], -1
+; CHECK-NEXT:    br i1 [[DONE]], label %[[EXIT1:.*]], label %[[BODY]], !prof [[PROF2:![0-9]+]]
+; CHECK:       [[BODY]]:
+; CHECK-NEXT:    [[NEXT]] = load i32, ptr [[P]], align 4
+; CHECK-NEXT:    [[LEAVE:%.*]] = icmp eq i32 [[NEXT]], -2
+; CHECK-NEXT:    br i1 [[LEAVE]], label %[[EXIT2:.*]], label %[[HEADER]], !prof [[PROF3:![0-9]+]]
+; CHECK:       [[EXIT1]]:
+; CHECK-NEXT:    ret void
+; CHECK:       [[EXIT2]]:
+; CHECK-NEXT:    ret void
+; CHECK:       [[RET]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  %enter = icmp ne i32 %cond, 0
+  br i1 %enter, label %ph, label %ret, !prof !1
+
+ph:
+  br label %header
+
+header:
+  %iv = phi i32 [ 0, %ph ], [ %next, %latch ]
+  %done = icmp eq i32 %iv, -1
+  br i1 %done, label %exit1, label %body, !prof !2
+
+body:
+  %next = load i32, ptr %p, align 4
+  %leave = icmp eq i32 %next, -2
+  br i1 %leave, label %exit2, label %latch, !prof !3
+
+latch:
+  br label %header
+
+exit1:
+  ret void
+
+exit2:
+  ret void
+
+ret:
+  ret void
+}
+
+; An explicit trip-count estimate remains authoritative when the guard folds.
+define void @folded_guard_multi_exit_explicit_tc(ptr %p, i32 %cond) !prof !0 {
+; CHECK-LABEL: define void @folded_guard_multi_exit_explicit_tc(
+; CHECK-SAME: ptr [[P:%.*]], i32 [[COND:%.*]]) !prof [[PROF0]] {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[ENTER:%.*]] = icmp ne i32 [[COND]], 0
+; CHECK-NEXT:    br i1 [[ENTER]], label %[[PH:.*]], label %[[RET:.*]], !prof [[PROF1]]
+; CHECK:       [[PH]]:
+; CHECK-NEXT:    br label %[[BODY:.*]]
+; CHECK:       [[HEADER:.*]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i32 [ [[NEXT:%.*]], %[[BODY]] ]
+; CHECK-NEXT:    [[DONE:%.*]] = icmp eq i32 [[IV]], -1
+; CHECK-NEXT:    br i1 [[DONE]], label %[[EXIT1:.*]], label %[[BODY]], !prof [[PROF2]], !llvm.loop [[LOOP4:![0-9]+]]
+; CHECK:       [[BODY]]:
+; CHECK-NEXT:    [[NEXT]] = load i32, ptr [[P]], align 4
+; CHECK-NEXT:    [[LEAVE:%.*]] = icmp eq i32 [[NEXT]], -2
+; CHECK-NEXT:    br i1 [[LEAVE]], label %[[EXIT2:.*]], label %[[HEADER]], !prof [[PROF3]]
+; CHECK:       [[EXIT1]]:
+; CHECK-NEXT:    ret void
+; CHECK:       [[EXIT2]]:
+; CHECK-NEXT:    ret void
+; CHECK:       [[RET]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  %enter = icmp ne i32 %cond, 0
+  br i1 %enter, label %ph, label %ret, !prof !1
+
+ph:
+  br label %header
+
+header:
+  %iv = phi i32 [ 0, %ph ], [ %next, %latch ]
+  %done = icmp eq i32 %iv, -1
+  br i1 %done, label %exit1, label %body, !prof !2, !llvm.loop !4
+
+body:
+  %next = load i32, ptr %p, align 4
+  %leave = icmp eq i32 %next, -2
+  br i1 %leave, label %exit2, label %latch, !prof !3
+
+latch:
+  br label %header
+
+exit1:
+  ret void
+
+exit2:
+  ret void
+
+ret:
+  ret void
+}
+
+; Exercise the same folded multi-exit case with the header successors and
+; weights reversed.
+define void @folded_guard_multi_exit_swapped(ptr %p, i32 %cond) !prof !0 {
+; CHECK-LABEL: define void @folded_guard_multi_exit_swapped(
+; CHECK-SAME: ptr [[P:%.*]], i32 [[COND:%.*]]) !prof [[PROF0]] {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[ENTER:%.*]] = icmp ne i32 [[COND]], 0
+; CHECK-NEXT:    br i1 [[ENTER]], label %[[PH:.*]], label %[[RET:.*]], !prof [[PROF1]]
+; CHECK:       [[PH]]:
+; CHECK-NEXT:    br label %[[BODY:.*]]
+; CHECK:       [[HEADER:.*]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i32 [ [[NEXT:%.*]], %[[BODY]] ]
+; CHECK-NEXT:    [[CONTINUE:%.*]] = icmp ne i32 [[IV]], -1
+; CHECK-NEXT:    br i1 [[CONTINUE]], label %[[BODY]], label %[[EXIT1:.*]], !prof [[PROF6:![0-9]+]]
+; CHECK:       [[BODY]]:
+; CHECK-NEXT:    [[NEXT]] = load i32, ptr [[P]], align 4
+; CHECK-NEXT:    [[LEAVE:%.*]] = icmp eq i32 [[NEXT]], -2
+; CHECK-NEXT:    br i1 [[LEAVE]], label %[[EXIT2:.*]], label %[[HEADER]], !prof [[PROF3]]
+; CHECK:       [[EXIT1]]:
+; CHECK-NEXT:    ret void
+; CHECK:       [[EXIT2]]:
+; CHECK-NEXT:    ret void
+; CHECK:       [[RET]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  %enter = icmp ne i32 %cond, 0
+  br i1 %enter, label %ph, label %ret, !prof !1
+
+ph:
+  br label %header
+
+header:
+  %iv = phi i32 [ 0, %ph ], [ %next, %latch ]
+  %continue = icmp ne i32 %iv, -1
+  br i1 %continue, label %body, label %exit1, !prof !6
+
+body:
+  %next = load i32, ptr %p, align 4
+  %leave = icmp eq i32 %next, -2
+  br i1 %leave, label %exit2, label %latch, !prof !3
+
+latch:
+  br label %header
+
+exit1:
+  ret void
+
+exit2:
+  ret void
+
+ret:
+  ret void
+}
+
+!0 = !{!"function_entry_count", i64 1201}
+!1 = !{!"branch_weights", i32 1200, i32 1}
+!2 = !{!"branch_weights", i32 200, i32 9800}
+!3 = !{!"branch_weights", i32 1000, i32 8800}
+!4 = distinct !{!4, !5}
+!5 = !{!"llvm.loop.estimated_trip_count", i32 10}
+!6 = !{!"branch_weights", i32 9800, i32 200}
+
+; BFI-BEFORE-LABEL: block-frequency-info: folded_guard_multi_exit
+; BFI-BEFORE: - body: {{.*}} count = 9800
+; BFI-BEFORE: - exit1: {{.*}} count = 200
+; BFI-BEFORE: - exit2: {{.*}} count = 1000
+; BFI-BEFORE: - ret: {{.*}} count = 1
+
+; BFI-BEFORE-LABEL: block-frequency-info: folded_guard_multi_exit_explicit_tc
+; BFI-BEFORE: - body: {{.*}} count = 9800
+; BFI-BEFORE: - exit1: {{.*}} count = 200
+; BFI-BEFORE: - exit2: {{.*}} count = 1000
+; BFI-BEFORE: - ret: {{.*}} count = 1
+
+; BFI-BEFORE-LABEL: block-frequency-info: folded_guard_multi_exit_swapped
+; BFI-BEFORE: - body: {{.*}} count = 9800
+; BFI-BEFORE: - exit1: {{.*}} count = 200
+; BFI-BEFORE: - exit2: {{.*}} count = 1000
+; BFI-BEFORE: - ret: {{.*}} count = 1
+
+; BFI-AFTER-LABEL: block-frequency-info: folded_guard_multi_exit
+; BFI-AFTER: - body: {{.*}} count = 9800
+; BFI-AFTER: - exit1: {{.*}} count = 200
+; BFI-AFTER: - exit2: {{.*}} count = 1000
+; BFI-AFTER: - ret: {{.*}} count = 1
+
+; BFI-AFTER-LABEL: block-frequency-info: folded_guard_multi_exit_explicit_tc
+; BFI-AFTER: - body: {{.*}} count = 9800
+; BFI-AFTER: - exit1: {{.*}} count = 200
+; BFI-AFTER: - exit2: {{.*}} count = 1000
+; BFI-AFTER: - ret: {{.*}} count = 1
+
+; BFI-AFTER-LABEL: block-frequency-info: folded_guard_multi_exit_swapped
+; BFI-AFTER: - body: {{.*}} count = 9800
+; BFI-AFTER: - exit1: {{.*}} count = 200
+; BFI-AFTER: - exit2: {{.*}} count = 1000
+; BFI-AFTER: - ret: {{.*}} count = 1
+
+
+
+
+;.
+; CHECK: [[PROF0]] = !{!"function_entry_count", i64 1201}
+; CHECK: [[PROF1]] = !{!"branch_weights", i32 1200, i32 1}
+; CHECK: [[PROF2]] = !{!"branch_weights", i32 85899346, i32 -601295421}
+; CHECK: [[PROF3]] = !{!"branch_weights", i32 1000, i32 8800}
+; CHECK: [[LOOP4]] = distinct !{[[LOOP4]], [[META5:![0-9]+]]}
+; CHECK: [[META5]] = !{!"llvm.loop.estimated_trip_count", i32 9}
+; CHECK: [[PROF6]] = !{!"branch_weights", i32 -601295421, i32 85899346}
+;.
diff --git a/llvm/test/Transforms/LoopRotate/folded-guard-multi-exit-inconsistent-bfi.ll b/llvm/test/Transforms/LoopRotate/folded-guard-multi-exit-inconsistent-bfi.ll
new file mode 100644
index 0000000000000..7d9ed1c579d4e
--- /dev/null
+++ b/llvm/test/Transforms/LoopRotate/folded-guard-multi-exit-inconsistent-bfi.ll
@@ -0,0 +1,60 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --no-generate-body-for-unused-prefixes --version 6
+; RUN: opt < %s -passes=loop-rotate -S | FileCheck %s
+; RUN: opt < %s -passes=loop-rotate -S | not grep llvm.loop.estimated_trip_count
+;
+; The profile says %body runs less often than the preheader. Keep the original
+; weights because BFI is inconsistent.
+
+define void @inconsistent_multi_exit(ptr %p) {
+; CHECK-LABEL: define void @inconsistent_multi_exit(
+; CHECK-SAME: ptr [[P:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    br label %[[PH:.*]]
+; CHECK:       [[PH]]:
+; CHECK-NEXT:    br label %[[BODY:.*]]
+; CHECK:       [[HEADER:.*]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i32 [ [[NEXT:%.*]], %[[BODY]] ]
+; CHECK-NEXT:    [[DONE:%.*]] = icmp eq i32 [[IV]], -1
+; CHECK-NEXT:    br i1 [[DONE]], label %[[EXIT1:.*]], label %[[BODY]], !prof [[PROF0:![0-9]+]]
+; CHECK:       [[BODY]]:
+; CHECK-NEXT:    [[NEXT]] = load i32, ptr [[P]], align 4
+; CHECK-NEXT:    [[LEAVE:%.*]] = icmp eq i32 [[NEXT]], -2
+; CHECK-NEXT:    br i1 [[LEAVE]], label %[[EXIT2:.*]], label %[[HEADER]], !prof [[PROF1:![0-9]+]]
+; CHECK:       [[EXIT1]]:
+; CHECK-NEXT:    ret void
+; CHECK:       [[EXIT2]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %ph
+
+ph:
+  br label %header
+
+header:
+  %iv = phi i32 [ 0, %ph ], [ %next, %latch ]
+  %done = icmp eq i32 %iv, -1
+  br i1 %done, label %exit1, label %body, !prof !0
+
+body:
+  %next = load i32, ptr %p, align 4
+  %leave = icmp eq i32 %next, -2
+  br i1 %leave, label %exit2, label %latch, !prof !1
+
+latch:
+  br label %header
+
+exit1:
+  ret void
+
+exit2:
+  ret void
+}
+
+!0 = !{!"branch_weights", i32 90, i32 10}
+!1 = !{!"branch_weights", i32 1, i32 9}
+
+;.
+; CHECK: [[PROF0]] = !{!"branch_weights", i32 90, i32 10}
+; CHECK: [[PROF1]] = !{!"branch_weights", i32 1, i32 9}
+;.
diff --git a/llvm/test/Transforms/LoopRotate/multi-exit-branch-weights.ll b/llvm/test/Transforms/LoopRotate/multi-exit-branch-weights.ll
index 5e24888abbdaa..dd967fcba20b5 100644
--- a/llvm/test/Transforms/LoopRotate/multi-exit-branch-weights.ll
+++ b/llvm/test/Transforms/LoopRotate/multi-exit-branch-weights.ll
@@ -1,3 +1,4 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --no-generate-body-for-unused-prefixes --version 6
 ; RUN: opt < %s -passes=loop-rotate -S | FileCheck %s
 ; RUN: opt < %s -passes='print<block-freq>' -disable-output 2>&1 | FileCheck %s --check-prefix=BFI-BEFORE
 ; RUN: opt < %s -passes='loop(loop-rotate),print<block-freq>' -disable-output 2>&1 | FileCheck %s --check-prefix=BFI-AFTER
@@ -14,6 +15,36 @@
 ;   latch  ---------> header
 
 define void @f(ptr %p, i32 %start, i32 %limit, i32 %cond) !prof !3 {
+; CHECK-LABEL: define void @f(
+; CHECK-SAME: ptr [[P:%.*]], i32 [[START:%.*]], i32 [[LIMIT:%.*]], i32 [[COND:%.*]]) !prof [[PROF0:![0-9]+]] {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[C0:%.*]] = icmp ne i32 [[COND]], 0
+; CHECK-NEXT:    br i1 [[C0]], label %[[PH:.*]], label %[[RET:.*]], !prof [[PROF1:![0-9]+]]
+; CHECK:       [[PH]]:
+; CHECK-NEXT:    [[HCMP1:%.*]] = icmp eq i32 [[START]], [[LIMIT]]
+; CHECK-NEXT:    br i1 [[HCMP1]], label %[[EXIT1:.*]], label %[[BODY_LR_PH:.*]], !prof [[PROF2:![0-9]+]]
+; CHECK:       [[BODY_LR_PH]]:
+; CHECK-NEXT:    br label %[[BODY:.*]]
+; CHECK:       [[HEADER:.*]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i32 [ [[IV_NEXT:%.*]], %[[BODY]] ]
+; CHECK-NEXT:    [[HCMP:%.*]] = icmp eq i32 [[IV]], [[LIMIT]]
+; CHECK-NEXT:    br i1 [[HCMP]], label %[[HEADER_EXIT1_CRIT_EDGE:.*]], label %[[BODY]], !prof [[PROF2]]
+; CHECK:       [[BODY]]:
+; CHECK-NEXT:    [[IV2:%.*]] = phi i32 [ [[START]], %[[BODY_LR_PH]] ], [ [[IV]], %[[HEADER]] ]
+; CHECK-NEXT:    [[ADDR:%.*]] = getelementptr i32, ptr [[P]], i32 [[IV2]]
+; CHECK-NEXT:    [[V:%.*]] = load i32, ptr [[ADDR]], align 4
+; CHECK-NEXT:    [[BCMP:%.*]] = icmp slt i32 [[V]], 0
+; CHECK-NEXT:    [[IV_NEXT]] = add i32 [[IV2]], 1
+; CHECK-NEXT:    br i1 [[BCMP]], label %[[EXIT2:.*]], label %[[HEADER]], !prof [[PROF3:![0-9]+]]
+; CHECK:       [[HEADER_EXIT1_CRIT_EDGE]]:
+; CHECK-NEXT:    br label %[[EXIT1]]
+; CHECK:       [[EXIT1]]:
+; CHECK-NEXT:    ret void
+; CHECK:       [[EXIT2]]:
+; CHECK-NEXT:    ret void
+; CHECK:       [[RET]]:
+; CHECK-NEXT:    ret void
+;
 entry:
   %c0 = icmp ne i32 %cond, 0
   br i1 %c0, label %ph, label %ret, !prof !0
@@ -50,6 +81,30 @@ ret:                                              ; preds = %entry
 ; Keep explicit trip-count metadata for multi-exit loops. Rotation decrements
 ; the attribute but does not infer a trip count from branch weights without it.
 define void @explicit_trip_count(ptr %p, i32 %start, i32 %limit) {
+; CHECK-LABEL: define void @explicit_trip_count(
+; CHECK-SAME: ptr [[P:%.*]], i32 [[START:%.*]], i32 [[LIMIT:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[HCMP1:%.*]] = icmp eq i32 [[START]], [[LIMIT]]
+; CHECK-NEXT:    br i1 [[HCMP1]], label %[[EXIT1:.*]], label %[[BODY_LR_PH:.*]], !prof [[PROF2]], !llvm.loop [[LOOP4:![0-9]+]]
+; CHECK:       [[BODY_LR_PH]]:
+; CHECK-NEXT:    br label %[[BODY:.*]], !llvm.loop [[LOOP4]]
+; CHECK:       [[HEADER:.*]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i32 [ [[IV_NEXT:%.*]], %[[BODY]] ]
+; CHECK-NEXT:    [[HCMP:%.*]] = icmp eq i32 [[IV]], [[LIMIT]]
+; CHECK-NEXT:    br i1 [[HCMP]], label %[[HEADER_EXIT1_CRIT_EDGE:.*]], label %[[BODY]], !prof [[PROF2]], !llvm.loop [[LOOP6:![0-9]+]]
+; CHECK:       [[BODY]]:
+; CHECK-NEXT:    [[IV2:%.*]] = phi i32 [ [[START]], %[[BODY_LR_PH]] ], [ [[IV]], %[[HEADER]] ]
+; CHECK-NEXT:    [[V:%.*]] = load i32, ptr [[P]], align 4
+; CHECK-NEXT:    [[BCMP:%.*]] = icmp slt i32 [[V]], 0
+; CHECK-NEXT:    [[IV_NEXT]] = add i32 [[IV2]], 1
+; CHECK-NEXT:    br i1 [[BCMP]], label %[[EXIT2:.*]], label %[[HEADER]], !prof [[PROF3]]
+; CHECK:       [[HEADER_EXIT1_CRIT_EDGE]]:
+; CHECK-NEXT:    br label %[[EXIT1]], !llvm.loop [[LOOP4]]
+; CHECK:       [[EXIT1]]:
+; CHECK-NEXT:    ret void
+; CHECK:       [[EXIT2]]:
+; CHECK-NEXT:    ret void
+;
 entry:
   br label %header
 
@@ -93,15 +148,13 @@ exit2:
 ; BFI-AFTER: - exit2: {{.*}} count = 2
 ; BFI-AFTER: - ret: {{.*}} count = 2
 
-; CHECK-LABEL: define void @f(
-
-; CHECK:      ph:
-; CHECK:        br i1 %{{.*}}, label %{{.*}}, label %{{.*}}.lr.ph, !prof [[WEIGHTS:![0-9]+]]
-
-; CHECK:        br i1 %{{.*}}, label %{{.*}}, label %body, !prof [[WEIGHTS]]{{$}}
-
-; CHECK-LABEL: define void @explicit_trip_count(
-; CHECK: br i1 %{{.*}}, label %{{.*}}, label %body, !prof {{![0-9]+}}, !llvm.loop [[EXPLICIT_LOOP:![0-9]+]]
-; CHECK-DAG: [[WEIGHTS]] = !{!"branch_weights", i32 200, i32 9800}
-; CHECK-DAG: [[EXPLICIT_LOOP]] = distinct !{[[EXPLICIT_LOOP]], [[EXPLICIT_TC:![0-9]+]]}
-; CHECK-DAG: [[EXPLICIT_TC]] = !{!"llvm.loop.estimated_trip_count", i32 9}
+;.
+; CHECK: [[PROF0]] = !{!"function_entry_count", i64 44}
+; CHECK: [[PROF1]] = !{!"branch_weights", i32 42, i32 2}
+; CHECK: [[PROF2]] = !{!"branch_weights", i32 200, i32 9800}
+; CHECK: [[PROF3]] = !{!"branch_weights", i32 10, i32 9790}
+; CHECK: [[LOOP4]] = distinct !{[[LOOP4]], [[META5:![0-9]+]]}
+; CHECK: [[META5]] = !{!"llvm.loop.estimated_trip_count", i32 10}
+; CHECK: [[LOOP6]] = distinct !{[[LOOP6]], [[META7:![0-9]+]]}
+; CHECK: [[META7]] = !{!"llvm.loop.estimated_trip_count", i32 9}
+;.



More information about the llvm-commits mailing list