[llvm] [VPlan] Distribute execution frequencies with BlockFrequencyInfo's code. (PR #226841)

Florian Hahn via llvm-commits llvm-commits at lists.llvm.org
Sun Sep 27 14:54:36 PDT 2026


https://github.com/fhahn created https://github.com/llvm/llvm-project/pull/226841

Instead of hand-rolling frequency propagating, manually construct BFI and use BFI::distributeMass.

Adds a final step where frequencies are rounded up to at least one, to distinguish between unreachable and almost never executed blocks. This matches BFI behavior.

Note that we need to still track whether frequencies are from edges with unknown profile info.

>From 7cdee828750ab7e4e639103051694d1d65f8c212 Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Sat, 26 Sep 2026 07:28:34 +0100
Subject: [PATCH] [VPlan] Distribute execution frequencies with
 BlockFrequencyInfo's code.

Instead of hand-rolling frequency propagating, manually construct BFI
and use BFI::distributeMass.

Adds a final step where frequencies are rounded up to at least one, to
distinguish between unreachable and almost never executed blocks. This
matches BFI behavior.

Note that we need to still track whether frequencies are from edges with
unknown profile info.
---
 llvm/lib/Transforms/Vectorize/VPlanUtils.cpp  | 93 ++++++++++---------
 .../VPlan/execution-frequencies-match-bfi.ll  | 74 +++++++--------
 .../X86/predicated-instruction-cost.ll        | 70 ++++++++++++--
 .../replicate-region-branch-weights.ll        | 36 +++----
 4 files changed, 160 insertions(+), 113 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp b/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
index a56689848efdf..c7d551aa332f8 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
@@ -16,6 +16,7 @@
 #include "llvm/ADT/SetVector.h"
 #include "llvm/ADT/SmallVectorExtras.h"
 #include "llvm/ADT/TypeSwitch.h"
+#include "llvm/Analysis/BlockFrequencyInfoImpl.h"
 #include "llvm/Analysis/InstSimplifyFolder.h"
 #include "llvm/Analysis/LoopAccessAnalysis.h"
 #include "llvm/Analysis/LoopInfo.h"
@@ -1250,59 +1251,59 @@ getSuccessorProbabilities(const VPBasicBlock *VPBB) {
   });
 }
 
-/// Returns \p Freq scaled by \p Prob, rounding up to 1 instead of 0 to keep a
-/// rarely executed block distinguishable from an unreachable one.
-static BlockFrequency scaleKeepingNonZero(BlockFrequency Freq,
-                                          BranchProbability Prob) {
-  BlockFrequency Scaled = Freq * Prob;
-  if (Scaled == BlockFrequency() && Freq != BlockFrequency() && !Prob.isZero())
-    return BlockFrequency(1);
-  return Scaled;
-}
-
 DenseMap<const VPBasicBlock *, std::optional<VPExecutionFrequency>>
 vputils::computeExecutionFrequencies(ArrayRef<VPBasicBlock *> Blocks) {
+  using BFIBase = BlockFrequencyInfoImplBase;
   assert(!Blocks.empty() && "expected at least the header block");
-  // Push each block's frequency along its outgoing edges. Blocks is in reverse
-  // post-order and forms a DAG with the backedge from the latch (the last
-  // block) ignored, so a block's frequency is final by the time it is visited.
-  DenseMap<const VPBasicBlock *, std::optional<VPExecutionFrequency>>
-      Frequencies;
-  Frequencies.reserve(Blocks.size());
-  // The header (first block) always executes, the others start out unreachable.
-  Frequencies[Blocks.front()].emplace(BlockFrequency(AlwaysExecutesFreq),
-                                      false);
-  for (VPBasicBlock *VPBB : Blocks.drop_front())
-    Frequencies[VPBB].emplace(BlockFrequency(), false);
-
-  for (VPBasicBlock *VPBB : Blocks) {
-    std::optional<VPExecutionFrequency> Src = Frequencies.at(VPBB);
+  // Distribute the header's frequency using BFI. Nodes for blocks are numbered
+  // in reverse post-order. Edges leaving Blocks, i.e. a plain CFG's edges to
+  // the middle block or to an exit block, exit to a node outside the loop.
+  BFIBase BFI;
+  BFIBase::BlockNode Header(0), Outside(Blocks.size());
+  BFIBase::LoopData &Loop = BFI.Loops.emplace_back(nullptr, Header);
+  DenseMap<const VPBlockBase *, BFIBase::BlockNode> Nodes;
+  for (auto [Idx, VPBB] : enumerate(Blocks)) {
+    Nodes[VPBB] = BFIBase::BlockNode(Idx);
+    BFI.Working.emplace_back(BFIBase::BlockNode(Idx)).Loop = &Loop;
+  }
+  BFI.Working.emplace_back(Outside);
+  BFI.Working[Header.Index].getMass() = BFIBase::BlockMass(AlwaysExecutesFreq);
+
+  // Keep track nodes reached via an edge without branch weighs or with
+  // estimated ones
+  SmallVector<bool> IsUnknown(Blocks.size()), IsEstimated(Blocks.size());
+  for (auto [Idx, VPBB] : enumerate(Blocks)) {
+    BFIBase::BlockNode Node(Idx);
     auto *Term = dyn_cast_if_present<VPInstruction>(VPBB->getTerminator());
     bool TermIsEstimated = Term && Term->hasEstimatedBranchWeights();
-    for (const auto &[Succ, EdgeProb] : getSuccessorProbabilities(VPBB)) {
-      // Ignore the backedge to the header (already treated as always
-      // executing).
-      if (Succ == Blocks.front())
-        continue;
-      // Ignore edges leaving Blocks, i.e. a plain CFG's edges to the middle
-      // block or to an exit block.
-      auto It = Frequencies.find(Succ);
-      if (It == Frequencies.end())
-        continue;
-      std::optional<VPExecutionFrequency> &SuccFreq = It->second;
-      // An unknown edge or predecessor poisons the successor.
-      if (!Src || EdgeProb.isUnknown() || !SuccFreq) {
-        SuccFreq = std::nullopt;
-        continue;
+    BFIBase::Distribution Dist;
+    bool HasProbs = true;
+    for (const auto &[Succ, Prob] : getSuccessorProbabilities(VPBB)) {
+      BFIBase::BlockNode SuccNode = Nodes.lookup_or(Succ, Outside);
+      if (SuccNode != Header && SuccNode != Outside) {
+        IsUnknown[SuccNode.Index] |= IsUnknown[Idx] || Prob.isUnknown();
+        IsEstimated[SuccNode.Index] |= IsEstimated[Idx] || TermIsEstimated;
       }
-      // The sum can only exceed AlwaysExecutesFreq by rounding.
-      BlockFrequency NewFreq =
-          std::min(BlockFrequency(AlwaysExecutesFreq),
-                   SuccFreq->Freq + scaleKeepingNonZero(Src->Freq, EdgeProb));
-      bool NewIsEstimated =
-          SuccFreq->IsEstimated || Src->IsEstimated || TermIsEstimated;
-      SuccFreq.emplace(NewFreq, NewIsEstimated);
+      HasProbs &= !Prob.isUnknown();
+      if (!Prob.isUnknown())
+        BFI.addToDist(Dist, &Loop, Node, SuccNode,
+                      getWeightFromBranchProb(Prob));
     }
+    if (HasProbs)
+      BFI.distributeMass(Node, &Loop, Dist);
+  }
+
+  // Round frequencies up to at least 1, so all edges are reached with a
+  // non-zero frequency, to distinguish rarely executed blocks from unreachable
+  // ones. blocks distinguishable from unreachable ones.
+  DenseMap<const VPBasicBlock *, std::optional<VPExecutionFrequency>>
+      Frequencies;
+  for (auto [Idx, VPBB] : enumerate(Blocks)) {
+    std::optional<VPExecutionFrequency> &Freq = Frequencies[VPBB];
+    if (IsUnknown[Idx])
+      continue;
+    uint64_t Mass = BFI.Working[Idx].getMass().getMass();
+    Freq.emplace(BlockFrequency(std::max<uint64_t>(Mass, 1)), IsEstimated[Idx]);
   }
   return Frequencies;
 }
diff --git a/llvm/test/Transforms/LoopVectorize/VPlan/execution-frequencies-match-bfi.ll b/llvm/test/Transforms/LoopVectorize/VPlan/execution-frequencies-match-bfi.ll
index a99a536578d6f..27a86f1123160 100644
--- a/llvm/test/Transforms/LoopVectorize/VPlan/execution-frequencies-match-bfi.ll
+++ b/llvm/test/Transforms/LoopVectorize/VPlan/execution-frequencies-match-bfi.ll
@@ -78,8 +78,6 @@ define void @single_pred_zero_weight(ptr noalias %a, ptr noalias %b, ptr noalias
 ;   %if.then 2^-31 ~ 4.66e-10
 ;   %latch      1 =        1
 ;
-; TODO: VPlan currently records a frequency of 0 for %if.then.
-;
 ; BFI-LABEL: block-frequency-info: single_pred_zero_weight
 ; BFI-NEXT:   - entry: float = 1.0,
 ; BFI-NEXT:   - loop: float = 1000.0,
@@ -99,8 +97,8 @@ define void @single_pred_zero_weight(ptr noalias %a, ptr noalias %b, ptr noalias
 ; VPLAN-NEXT:  Successor(s): if.then, latch
 ; VPLAN-EMPTY:
 ; VPLAN-NEXT:  if.then:
-; VPLAN-NEXT:    EMIT ir<%gep.a> = getelementptr inbounds ir<%a>, ir<%iv>{{$}}
-; VPLAN-NEXT:    EMIT store ir<%i>, ir<%gep.a>{{$}}
+; VPLAN-NEXT:    EMIT ir<%gep.a> = getelementptr inbounds ir<%a>, ir<%iv> (!vplan.execution.frequency 4294967296 (4.657E-8%))
+; VPLAN-NEXT:    EMIT store ir<%i>, ir<%gep.a> (!vplan.execution.frequency 4294967296 (4.657E-8%))
 ; VPLAN-NEXT:  Successor(s): latch
 ; VPLAN-EMPTY:
 ; VPLAN-NEXT:  latch:
@@ -143,8 +141,6 @@ define void @single_pred_zero_weight_sibling(ptr noalias %a, ptr noalias %b, ptr
 ;   %if.then 1 - 2^-31 ~ 1 - 4.66e-10
 ;   %latch           1 =          1
 ;
-; TODO: VPlan currently records %if.then as always executing.
-;
 ; BFI-LABEL: block-frequency-info: single_pred_zero_weight_sibling
 ; BFI-NEXT:   - entry: float = 1.0,
 ; BFI-NEXT:   - loop: float = 1000.0, int = 18014398509481984
@@ -164,8 +160,8 @@ define void @single_pred_zero_weight_sibling(ptr noalias %a, ptr noalias %b, ptr
 ; VPLAN-NEXT:  Successor(s): if.then, latch
 ; VPLAN-EMPTY:
 ; VPLAN-NEXT:  if.then:
-; VPLAN-NEXT:    EMIT ir<%gep.a> = getelementptr inbounds ir<%a>, ir<%iv>{{$}}
-; VPLAN-NEXT:    EMIT store ir<%i>, ir<%gep.a>{{$}}
+; VPLAN-NEXT:    EMIT ir<%gep.a> = getelementptr inbounds ir<%a>, ir<%iv> (!vplan.execution.frequency 9223372032559808512 (100%))
+; VPLAN-NEXT:    EMIT store ir<%i>, ir<%gep.a> (!vplan.execution.frequency 9223372032559808512 (100%))
 ; VPLAN-NEXT:  Successor(s): latch
 ; VPLAN-EMPTY:
 ; VPLAN-NEXT:  latch:
@@ -459,13 +455,13 @@ define void @switch_common_dest(ptr noalias %a, ptr noalias %b, ptr noalias %c,
 ; VPLAN-NEXT:  Successor(s): latch
 ; VPLAN-EMPTY:
 ; VPLAN-NEXT:  if.then:
-; VPLAN-NEXT:    EMIT ir<%gep.a> = getelementptr inbounds ir<%a>, ir<%iv> (!vplan.execution.frequency 3458764513820540928 (37.5%))
-; VPLAN-NEXT:    EMIT store ir<1>, ir<%gep.a> (!vplan.execution.frequency 3458764513820540928 (37.5%))
+; VPLAN-NEXT:    EMIT ir<%gep.a> = getelementptr inbounds ir<%a>, ir<%iv> (!vplan.execution.frequency 3458764514357411840 (37.5%))
+; VPLAN-NEXT:    EMIT store ir<1>, ir<%gep.a> (!vplan.execution.frequency 3458764514357411840 (37.5%))
 ; VPLAN-NEXT:  Successor(s): latch
 ; VPLAN-EMPTY:
 ; VPLAN-NEXT:  default:
-; VPLAN-NEXT:    EMIT ir<%gep.c> = getelementptr inbounds ir<%c>, ir<%iv> (!vplan.execution.frequency 4611686018427387904 (50%))
-; VPLAN-NEXT:    EMIT store ir<0>, ir<%gep.c> (!vplan.execution.frequency 4611686018427387904 (50%))
+; VPLAN-NEXT:    EMIT ir<%gep.c> = getelementptr inbounds ir<%c>, ir<%iv> (!vplan.execution.frequency 4611686017890516992 (50%))
+; VPLAN-NEXT:    EMIT store ir<0>, ir<%gep.c> (!vplan.execution.frequency 4611686017890516992 (50%))
 ; VPLAN-NEXT:  Successor(s): latch
 ; VPLAN-EMPTY:
 ; VPLAN-NEXT:  latch:
@@ -1093,9 +1089,6 @@ define void @switch_join_always(ptr noalias %a, ptr noalias %idx) {
 ;   %join     1 =   1
 ;   %latch    1 =   1
 ;
-; TODO: %join and %latch are currently recorded as executing slightly less
-; often, as the rounded probabilities do not add up to 1.
-;
 ; BFI-LABEL: block-frequency-info: switch_join_always
 ; BFI-NEXT:   - entry: float = 1.0,
 ; BFI-NEXT:   - loop: float = 1000.0,
@@ -1111,18 +1104,18 @@ define void @switch_join_always(ptr noalias %a, ptr noalias %idx) {
 ; VPLAN-NEXT:  Successor(s): join
 ; VPLAN-EMPTY:
 ; VPLAN-NEXT:  case.1:
-; VPLAN-NEXT:    EMIT store ir<1>, ir<%gep.a> (!vplan.execution.frequency 1317624575466405888 (14.29%))
+; VPLAN-NEXT:    EMIT store ir<1>, ir<%gep.a> (!vplan.execution.frequency 1317624575670928140 (14.29%))
 ; VPLAN-NEXT:  Successor(s): join
 ; VPLAN-EMPTY:
 ; VPLAN-NEXT:  join:
-; VPLAN-NEXT:    EMIT ir<%add> = add ir<%i>, ir<10> (!vplan.execution.frequency 9223372032559808512 (100%))
-; VPLAN-NEXT:    EMIT store ir<%add>, ir<%gep.a> (!vplan.execution.frequency 9223372032559808512 (100%))
+; VPLAN-NEXT:    EMIT ir<%add> = add ir<%i>, ir<10>{{$}}
+; VPLAN-NEXT:    EMIT store ir<%add>, ir<%gep.a>{{$}}
 ; VPLAN-NEXT:  Successor(s): latch
 ; VPLAN-EMPTY:
 ; VPLAN-NEXT:  latch:
-; VPLAN-NEXT:    EMIT ir<%iv.next> = add ir<%iv>, ir<1> (!vplan.execution.frequency 9223372032559808512 (100%))
-; VPLAN-NEXT:    EMIT ir<%ec> = icmp eq ir<%iv.next>, ir<1024> (!vplan.execution.frequency 9223372032559808512 (100%))
-; VPLAN-NEXT:    EMIT branch-on-cond ir<%ec> (!prof {1, 999}, !vplan.execution.frequency 9223372032559808512 (100%))
+; VPLAN-NEXT:    EMIT ir<%iv.next> = add ir<%iv>, ir<1>{{$}}
+; VPLAN-NEXT:    EMIT ir<%ec> = icmp eq ir<%iv.next>, ir<1024>{{$}}
+; VPLAN-NEXT:    EMIT branch-on-cond ir<%ec> (!prof {1, 999}){{$}}
 ; VPLAN-NEXT:  Successor(s): middle.block, loop
 ;
 entry:
@@ -1306,9 +1299,6 @@ define void @rarely_executed_chain(ptr noalias %a, ptr noalias %idx) {
 ;   %if.e 2^-64 ~ 5.42e-20  (clamped up from 2^-65)
 ;   %latch    1 =        1
 ;
-; TODO: %if.a and the blocks it reaches are currently recorded as never
-; executing.
-;
 ; BFI-LABEL: block-frequency-info: rarely_executed_chain
 ; BFI-NEXT:   - entry: float = 1.0,
 ; BFI-NEXT:   - loop: float = 1000.0,
@@ -1322,28 +1312,28 @@ define void @rarely_executed_chain(ptr noalias %a, ptr noalias %idx) {
 ;
 ; VPLAN-LABEL: VPlan for loop in 'rarely_executed_chain'
 ; VPLAN:       if.a:
-; VPLAN-NEXT:    EMIT ir<%c.1> = icmp sgt ir<%i>, ir<10>{{$}}
-; VPLAN-NEXT:    EMIT branch-on-cond ir<%c.1> (!prof {1000, 0}){{$}}
+; VPLAN-NEXT:    EMIT ir<%c.1> = icmp sgt ir<%i>, ir<10> (!vplan.execution.frequency 4294967296 (4.657E-8%))
+; VPLAN-NEXT:    EMIT branch-on-cond ir<%c.1> (!prof {1000, 0}, !vplan.execution.frequency 4294967296 (4.657E-8%))
 ; VPLAN-NEXT:  Successor(s): latch, if.b
 ; VPLAN-EMPTY:
 ; VPLAN-NEXT:  if.b:
-; VPLAN-NEXT:    EMIT ir<%c.2> = icmp sgt ir<%i>, ir<20>{{$}}
-; VPLAN-NEXT:    EMIT branch-on-cond ir<%c.2> (!prof {1, 1}){{$}}
+; VPLAN-NEXT:    EMIT ir<%c.2> = icmp sgt ir<%i>, ir<20> (!vplan.execution.frequency 2 (2.168E-17%))
+; VPLAN-NEXT:    EMIT branch-on-cond ir<%c.2> (!prof {1, 1}, !vplan.execution.frequency 2 (2.168E-17%))
 ; VPLAN-NEXT:  Successor(s): if.c, latch
 ; VPLAN-EMPTY:
 ; VPLAN-NEXT:  if.c:
-; VPLAN-NEXT:    EMIT ir<%c.3> = icmp sgt ir<%i>, ir<30>{{$}}
-; VPLAN-NEXT:    EMIT branch-on-cond ir<%c.3> (!prof {1, 1}){{$}}
+; VPLAN-NEXT:    EMIT ir<%c.3> = icmp sgt ir<%i>, ir<30> (!vplan.execution.frequency 1 (1.084E-17%))
+; VPLAN-NEXT:    EMIT branch-on-cond ir<%c.3> (!prof {1, 1}, !vplan.execution.frequency 1 (1.084E-17%))
 ; VPLAN-NEXT:  Successor(s): latch, if.d
 ; VPLAN-EMPTY:
 ; VPLAN-NEXT:  if.d:
-; VPLAN-NEXT:    EMIT store ir<1>, ir<%gep.a>{{$}}
-; VPLAN-NEXT:    EMIT ir<%c.4> = icmp sgt ir<%i>, ir<40>{{$}}
-; VPLAN-NEXT:    EMIT branch-on-cond ir<%c.4> (!prof {1, 1}){{$}}
+; VPLAN-NEXT:    EMIT store ir<1>, ir<%gep.a> (!vplan.execution.frequency 1 (1.084E-17%))
+; VPLAN-NEXT:    EMIT ir<%c.4> = icmp sgt ir<%i>, ir<40> (!vplan.execution.frequency 1 (1.084E-17%))
+; VPLAN-NEXT:    EMIT branch-on-cond ir<%c.4> (!prof {1, 1}, !vplan.execution.frequency 1 (1.084E-17%))
 ; VPLAN-NEXT:  Successor(s): if.e, latch
 ; VPLAN-EMPTY:
 ; VPLAN-NEXT:  if.e:
-; VPLAN-NEXT:    EMIT store ir<2>, ir<%gep.a>{{$}}
+; VPLAN-NEXT:    EMIT store ir<2>, ir<%gep.a> (!vplan.execution.frequency 1 (1.084E-17%))
 ; VPLAN-NEXT:  Successor(s): latch
 ;
 entry:
@@ -1397,8 +1387,6 @@ define void @nested_zero_weight_siblings(ptr noalias %a, ptr noalias %idx) {
 ;   %then.2 (1 - 2^-31)^3 ~ 1 - 1.40e-9
 ;   %latch              1 =             1
 ;
-; TODO: %then.0, %then.1 and %then.2 are currently recorded as always executing.
-;
 ; BFI-LABEL: block-frequency-info: nested_zero_weight_siblings
 ; BFI-NEXT:   - entry: float = 1.0,
 ; BFI-NEXT:   - loop: float = 1000.0, int = 18014398509481984
@@ -1410,19 +1398,19 @@ define void @nested_zero_weight_siblings(ptr noalias %a, ptr noalias %idx) {
 ;
 ; VPLAN-LABEL: VPlan for loop in 'nested_zero_weight_siblings'
 ; VPLAN:       then.0:
-; VPLAN-NEXT:    EMIT store ir<0>, ir<%gep.a>{{$}}
-; VPLAN-NEXT:    EMIT ir<%c.1> = icmp sgt ir<%i>, ir<10>{{$}}
-; VPLAN-NEXT:    EMIT branch-on-cond ir<%c.1> (!prof {1000, 0}){{$}}
+; VPLAN-NEXT:    EMIT store ir<0>, ir<%gep.a> (!vplan.execution.frequency 9223372032559808512 (100%))
+; VPLAN-NEXT:    EMIT ir<%c.1> = icmp sgt ir<%i>, ir<10> (!vplan.execution.frequency 9223372032559808512 (100%))
+; VPLAN-NEXT:    EMIT branch-on-cond ir<%c.1> (!prof {1000, 0}, !vplan.execution.frequency 9223372032559808512 (100%))
 ; VPLAN-NEXT:  Successor(s): then.1, latch
 ; VPLAN-EMPTY:
 ; VPLAN-NEXT:  then.1:
-; VPLAN-NEXT:    EMIT store ir<1>, ir<%gep.a>{{$}}
-; VPLAN-NEXT:    EMIT ir<%c.2> = icmp sgt ir<%i>, ir<20>{{$}}
-; VPLAN-NEXT:    EMIT branch-on-cond ir<%c.2> (!prof {1000, 0}){{$}}
+; VPLAN-NEXT:    EMIT store ir<1>, ir<%gep.a> (!vplan.execution.frequency 9223372028264841218 (100%))
+; VPLAN-NEXT:    EMIT ir<%c.2> = icmp sgt ir<%i>, ir<20> (!vplan.execution.frequency 9223372028264841218 (100%))
+; VPLAN-NEXT:    EMIT branch-on-cond ir<%c.2> (!prof {1000, 0}, !vplan.execution.frequency 9223372028264841218 (100%))
 ; VPLAN-NEXT:  Successor(s): then.2, latch
 ; VPLAN-EMPTY:
 ; VPLAN-NEXT:  then.2:
-; VPLAN-NEXT:    EMIT store ir<2>, ir<%gep.a>{{$}}
+; VPLAN-NEXT:    EMIT store ir<2>, ir<%gep.a> (!vplan.execution.frequency 9223372023969873925 (100%))
 ; VPLAN-NEXT:  Successor(s): latch
 ; VPLAN-EMPTY:
 ; VPLAN-NEXT:  latch:
diff --git a/llvm/test/Transforms/LoopVectorize/X86/predicated-instruction-cost.ll b/llvm/test/Transforms/LoopVectorize/X86/predicated-instruction-cost.ll
index 730b7fa2404bb..44997f8eeb540 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/predicated-instruction-cost.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/predicated-instruction-cost.ll
@@ -305,22 +305,80 @@ define void @predicated_sdiv_zero_weight(ptr noalias %a, ptr noalias %b, i64 %n)
 ; CHECK-LABEL: define void @predicated_sdiv_zero_weight(
 ; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], i64 [[N:%.*]]) {
 ; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 4
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = and i64 [[N]], 3
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
 ; CHECK-NEXT:    br label %[[LOOP:.*]]
 ; CHECK:       [[LOOP]]:
-; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LATCH:.*]] ]
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[PRED_STORE_CONTINUE6:.*]] ]
 ; CHECK-NEXT:    [[GEP_A:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
-; CHECK-NEXT:    [[L:%.*]] = load i32, ptr [[GEP_A]], align 4
+; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[GEP_A]], align 4
+; CHECK-NEXT:    [[TMP2:%.*]] = icmp sgt <4 x i32> [[WIDE_LOAD]], zeroinitializer
+; CHECK-NEXT:    [[TMP3:%.*]] = extractelement <4 x i1> [[TMP2]], i64 0
+; CHECK-NEXT:    br i1 [[TMP3]], label %[[PRED_STORE_IF:.*]], label %[[PRED_STORE_CONTINUE:.*]], !prof [[PROF4:![0-9]+]]
+; CHECK:       [[PRED_STORE_IF]]:
+; CHECK-NEXT:    [[TMP4:%.*]] = extractelement <4 x i32> [[WIDE_LOAD]], i64 0
+; CHECK-NEXT:    [[TMP5:%.*]] = sdiv i32 1000, [[TMP4]]
+; CHECK-NEXT:    [[TMP6:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 [[IV]]
+; CHECK-NEXT:    store i32 [[TMP5]], ptr [[TMP6]], align 4
+; CHECK-NEXT:    br label %[[PRED_STORE_CONTINUE]]
+; CHECK:       [[PRED_STORE_CONTINUE]]:
+; CHECK-NEXT:    [[TMP7:%.*]] = extractelement <4 x i1> [[TMP2]], i64 1
+; CHECK-NEXT:    br i1 [[TMP7]], label %[[PRED_STORE_IF1:.*]], label %[[PRED_STORE_CONTINUE2:.*]], !prof [[PROF4]]
+; CHECK:       [[PRED_STORE_IF1]]:
+; CHECK-NEXT:    [[TMP8:%.*]] = extractelement <4 x i32> [[WIDE_LOAD]], i64 1
+; CHECK-NEXT:    [[TMP9:%.*]] = sdiv i32 1000, [[TMP8]]
+; CHECK-NEXT:    [[TMP10:%.*]] = add i64 [[IV]], 1
+; CHECK-NEXT:    [[TMP11:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 [[TMP10]]
+; CHECK-NEXT:    store i32 [[TMP9]], ptr [[TMP11]], align 4
+; CHECK-NEXT:    br label %[[PRED_STORE_CONTINUE2]]
+; CHECK:       [[PRED_STORE_CONTINUE2]]:
+; CHECK-NEXT:    [[TMP12:%.*]] = extractelement <4 x i1> [[TMP2]], i64 2
+; CHECK-NEXT:    br i1 [[TMP12]], label %[[PRED_STORE_IF3:.*]], label %[[PRED_STORE_CONTINUE4:.*]], !prof [[PROF4]]
+; CHECK:       [[PRED_STORE_IF3]]:
+; CHECK-NEXT:    [[TMP13:%.*]] = extractelement <4 x i32> [[WIDE_LOAD]], i64 2
+; CHECK-NEXT:    [[TMP14:%.*]] = sdiv i32 1000, [[TMP13]]
+; CHECK-NEXT:    [[TMP15:%.*]] = add i64 [[IV]], 2
+; CHECK-NEXT:    [[TMP16:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 [[TMP15]]
+; CHECK-NEXT:    store i32 [[TMP14]], ptr [[TMP16]], align 4
+; CHECK-NEXT:    br label %[[PRED_STORE_CONTINUE4]]
+; CHECK:       [[PRED_STORE_CONTINUE4]]:
+; CHECK-NEXT:    [[TMP17:%.*]] = extractelement <4 x i1> [[TMP2]], i64 3
+; CHECK-NEXT:    br i1 [[TMP17]], label %[[PRED_STORE_IF5:.*]], label %[[PRED_STORE_CONTINUE6]], !prof [[PROF4]]
+; CHECK:       [[PRED_STORE_IF5]]:
+; CHECK-NEXT:    [[TMP18:%.*]] = extractelement <4 x i32> [[WIDE_LOAD]], i64 3
+; CHECK-NEXT:    [[TMP19:%.*]] = sdiv i32 1000, [[TMP18]]
+; CHECK-NEXT:    [[TMP20:%.*]] = add i64 [[IV]], 3
+; CHECK-NEXT:    [[TMP21:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 [[TMP20]]
+; CHECK-NEXT:    store i32 [[TMP19]], ptr [[TMP21]], align 4
+; CHECK-NEXT:    br label %[[PRED_STORE_CONTINUE6]]
+; CHECK:       [[PRED_STORE_CONTINUE6]]:
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[IV]], 4
+; CHECK-NEXT:    [[TMP22:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP22]], label %[[MIDDLE_BLOCK:.*]], label %[[LOOP]], !llvm.loop [[LOOP5:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; CHECK:       [[SCALAR_PH]]:
+; CHECK-NEXT:    [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT:    br label %[[LOOP1:.*]]
+; CHECK:       [[LOOP1]]:
+; CHECK-NEXT:    [[IV1:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LATCH:.*]] ]
+; CHECK-NEXT:    [[GEP_A1:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV1]]
+; CHECK-NEXT:    [[L:%.*]] = load i32, ptr [[GEP_A1]], align 4
 ; CHECK-NEXT:    [[C:%.*]] = icmp sgt i32 [[L]], 0
-; CHECK-NEXT:    br i1 [[C]], label %[[THEN:.*]], label %[[LATCH]], !prof [[PROF4:![0-9]+]]
+; CHECK-NEXT:    br i1 [[C]], label %[[THEN:.*]], label %[[LATCH]], !prof [[PROF6:![0-9]+]]
 ; CHECK:       [[THEN]]:
 ; CHECK-NEXT:    [[D:%.*]] = sdiv i32 1000, [[L]]
-; CHECK-NEXT:    [[GEP_B:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 [[IV]]
+; CHECK-NEXT:    [[GEP_B:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 [[IV1]]
 ; CHECK-NEXT:    store i32 [[D]], ptr [[GEP_B]], align 4
 ; CHECK-NEXT:    br label %[[LATCH]]
 ; CHECK:       [[LATCH]]:
-; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV1]], 1
 ; CHECK-NEXT:    [[EC:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
-; CHECK-NEXT:    br i1 [[EC]], label %[[EXIT:.*]], label %[[LOOP]], !llvm.loop [[LOOP5:![0-9]+]]
+; CHECK-NEXT:    br i1 [[EC]], label %[[EXIT]], label %[[LOOP1]], !llvm.loop [[LOOP7:![0-9]+]]
 ; CHECK:       [[EXIT]]:
 ; CHECK-NEXT:    ret void
 ;
diff --git a/llvm/test/Transforms/LoopVectorize/replicate-region-branch-weights.ll b/llvm/test/Transforms/LoopVectorize/replicate-region-branch-weights.ll
index 2f7fdeda4c08f..80cedfd2a5704 100644
--- a/llvm/test/Transforms/LoopVectorize/replicate-region-branch-weights.ll
+++ b/llvm/test/Transforms/LoopVectorize/replicate-region-branch-weights.ll
@@ -1205,9 +1205,9 @@ exit:
 }
 
 ; Two predicated stores guarded by the same condition. The first branch is
-; always taken (so it needs no weights of its own), while the second is
+; almost always taken (its other edge has weight 0), while the second is
 ; rarely taken. The merged region fires whenever either original region did,
-; i.e. (near) always, so it must not keep the second region's low frequency.
+; i.e. almost always, so it must not keep the second region's low frequency.
 define void @merged_replicate_regions_first_always_taken(ptr noalias %a, ptr noalias %b, i32 %n) {
 ; VF4IC1-LABEL: define void @merged_replicate_regions_first_always_taken(
 ; VF4IC1-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], i32 [[N:%.*]]) {
@@ -1215,16 +1215,16 @@ define void @merged_replicate_regions_first_always_taken(ptr noalias %a, ptr noa
 ; VF4IC1:    br i1 [[MIN_ITERS_CHECK:%.*]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]], !prof [[PROF0]]
 ; VF4IC1:  [[VECTOR_PH]]:
 ; VF4IC1:  [[VECTOR_BODY:.*]]:
-; VF4IC1:    br i1 [[TMP3:%.*]], label %[[PRED_STORE_IF:.*]], label %[[PRED_STORE_CONTINUE:.*]]
+; VF4IC1:    br i1 [[TMP3:%.*]], label %[[PRED_STORE_IF:.*]], label %[[PRED_STORE_CONTINUE:.*]], !prof [[PROF44]]
 ; VF4IC1:  [[PRED_STORE_IF]]:
 ; VF4IC1:  [[PRED_STORE_CONTINUE]]:
-; VF4IC1:    br i1 [[TMP5:%.*]], label %[[PRED_STORE_IF1:.*]], label %[[PRED_STORE_CONTINUE2:.*]]
+; VF4IC1:    br i1 [[TMP5:%.*]], label %[[PRED_STORE_IF1:.*]], label %[[PRED_STORE_CONTINUE2:.*]], !prof [[PROF44]]
 ; VF4IC1:  [[PRED_STORE_IF1]]:
 ; VF4IC1:  [[PRED_STORE_CONTINUE2]]:
-; VF4IC1:    br i1 [[TMP9:%.*]], label %[[PRED_STORE_IF3:.*]], label %[[PRED_STORE_CONTINUE4:.*]]
+; VF4IC1:    br i1 [[TMP9:%.*]], label %[[PRED_STORE_IF3:.*]], label %[[PRED_STORE_CONTINUE4:.*]], !prof [[PROF44]]
 ; VF4IC1:  [[PRED_STORE_IF3]]:
 ; VF4IC1:  [[PRED_STORE_CONTINUE4]]:
-; VF4IC1:    br i1 [[TMP13:%.*]], label %[[PRED_STORE_IF5:.*]], label %[[PRED_STORE_CONTINUE6:.*]]
+; VF4IC1:    br i1 [[TMP13:%.*]], label %[[PRED_STORE_IF5:.*]], label %[[PRED_STORE_CONTINUE6:.*]], !prof [[PROF44]]
 ; VF4IC1:  [[PRED_STORE_IF5]]:
 ; VF4IC1:  [[PRED_STORE_CONTINUE6]]:
 ; VF4IC1:    br i1 [[TMP17:%.*]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !prof [[PROF2]], !llvm.loop [[LOOP48:![0-9]+]]
@@ -1247,16 +1247,16 @@ define void @merged_replicate_regions_first_always_taken(ptr noalias %a, ptr noa
 ; VF2IC2:    br i1 [[MIN_ITERS_CHECK:%.*]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]], !prof [[PROF0]]
 ; VF2IC2:  [[VECTOR_PH]]:
 ; VF2IC2:  [[VECTOR_BODY:.*]]:
-; VF2IC2:    br i1 [[TMP5:%.*]], label %[[PRED_STORE_IF:.*]], label %[[PRED_STORE_CONTINUE:.*]]
+; VF2IC2:    br i1 [[TMP5:%.*]], label %[[PRED_STORE_IF:.*]], label %[[PRED_STORE_CONTINUE:.*]], !prof [[PROF44]]
 ; VF2IC2:  [[PRED_STORE_IF]]:
 ; VF2IC2:  [[PRED_STORE_CONTINUE]]:
-; VF2IC2:    br i1 [[TMP7:%.*]], label %[[PRED_STORE_IF2:.*]], label %[[PRED_STORE_CONTINUE3:.*]]
+; VF2IC2:    br i1 [[TMP7:%.*]], label %[[PRED_STORE_IF2:.*]], label %[[PRED_STORE_CONTINUE3:.*]], !prof [[PROF44]]
 ; VF2IC2:  [[PRED_STORE_IF2]]:
 ; VF2IC2:  [[PRED_STORE_CONTINUE3]]:
-; VF2IC2:    br i1 [[TMP11:%.*]], label %[[PRED_STORE_IF4:.*]], label %[[PRED_STORE_CONTINUE5:.*]]
+; VF2IC2:    br i1 [[TMP11:%.*]], label %[[PRED_STORE_IF4:.*]], label %[[PRED_STORE_CONTINUE5:.*]], !prof [[PROF44]]
 ; VF2IC2:  [[PRED_STORE_IF4]]:
 ; VF2IC2:  [[PRED_STORE_CONTINUE5]]:
-; VF2IC2:    br i1 [[TMP15:%.*]], label %[[PRED_STORE_IF6:.*]], label %[[PRED_STORE_CONTINUE7:.*]]
+; VF2IC2:    br i1 [[TMP15:%.*]], label %[[PRED_STORE_IF6:.*]], label %[[PRED_STORE_CONTINUE7:.*]], !prof [[PROF44]]
 ; VF2IC2:  [[PRED_STORE_IF6]]:
 ; VF2IC2:  [[PRED_STORE_CONTINUE7]]:
 ; VF2IC2:    br i1 [[TMP19:%.*]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !prof [[PROF2]], !llvm.loop [[LOOP48:![0-9]+]]
@@ -1314,16 +1314,16 @@ define void @predicated_store_zero_taken_weight(ptr %a, i32 %n) {
 ; VF4IC1:    br i1 [[MIN_ITERS_CHECK:%.*]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]], !prof [[PROF0]]
 ; VF4IC1:  [[VECTOR_PH]]:
 ; VF4IC1:  [[VECTOR_BODY:.*]]:
-; VF4IC1:    br i1 [[TMP3:%.*]], label %[[PRED_STORE_IF:.*]], label %[[PRED_STORE_CONTINUE:.*]]
+; VF4IC1:    br i1 [[TMP3:%.*]], label %[[PRED_STORE_IF:.*]], label %[[PRED_STORE_CONTINUE:.*]], !prof [[PROF26]]
 ; VF4IC1:  [[PRED_STORE_IF]]:
 ; VF4IC1:  [[PRED_STORE_CONTINUE]]:
-; VF4IC1:    br i1 [[TMP4:%.*]], label %[[PRED_STORE_IF1:.*]], label %[[PRED_STORE_CONTINUE2:.*]]
+; VF4IC1:    br i1 [[TMP4:%.*]], label %[[PRED_STORE_IF1:.*]], label %[[PRED_STORE_CONTINUE2:.*]], !prof [[PROF26]]
 ; VF4IC1:  [[PRED_STORE_IF1]]:
 ; VF4IC1:  [[PRED_STORE_CONTINUE2]]:
-; VF4IC1:    br i1 [[TMP7:%.*]], label %[[PRED_STORE_IF3:.*]], label %[[PRED_STORE_CONTINUE4:.*]]
+; VF4IC1:    br i1 [[TMP7:%.*]], label %[[PRED_STORE_IF3:.*]], label %[[PRED_STORE_CONTINUE4:.*]], !prof [[PROF26]]
 ; VF4IC1:  [[PRED_STORE_IF3]]:
 ; VF4IC1:  [[PRED_STORE_CONTINUE4]]:
-; VF4IC1:    br i1 [[TMP10:%.*]], label %[[PRED_STORE_IF5:.*]], label %[[PRED_STORE_CONTINUE6:.*]]
+; VF4IC1:    br i1 [[TMP10:%.*]], label %[[PRED_STORE_IF5:.*]], label %[[PRED_STORE_CONTINUE6:.*]], !prof [[PROF26]]
 ; VF4IC1:  [[PRED_STORE_IF5]]:
 ; VF4IC1:  [[PRED_STORE_CONTINUE6]]:
 ; VF4IC1:    br i1 [[TMP13:%.*]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !prof [[PROF2]], !llvm.loop [[LOOP50:![0-9]+]]
@@ -1343,16 +1343,16 @@ define void @predicated_store_zero_taken_weight(ptr %a, i32 %n) {
 ; VF2IC2:    br i1 [[MIN_ITERS_CHECK:%.*]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]], !prof [[PROF0]]
 ; VF2IC2:  [[VECTOR_PH]]:
 ; VF2IC2:  [[VECTOR_BODY:.*]]:
-; VF2IC2:    br i1 [[TMP5:%.*]], label %[[PRED_STORE_IF:.*]], label %[[PRED_STORE_CONTINUE:.*]]
+; VF2IC2:    br i1 [[TMP5:%.*]], label %[[PRED_STORE_IF:.*]], label %[[PRED_STORE_CONTINUE:.*]], !prof [[PROF26]]
 ; VF2IC2:  [[PRED_STORE_IF]]:
 ; VF2IC2:  [[PRED_STORE_CONTINUE]]:
-; VF2IC2:    br i1 [[TMP6:%.*]], label %[[PRED_STORE_IF2:.*]], label %[[PRED_STORE_CONTINUE3:.*]]
+; VF2IC2:    br i1 [[TMP6:%.*]], label %[[PRED_STORE_IF2:.*]], label %[[PRED_STORE_CONTINUE3:.*]], !prof [[PROF26]]
 ; VF2IC2:  [[PRED_STORE_IF2]]:
 ; VF2IC2:  [[PRED_STORE_CONTINUE3]]:
-; VF2IC2:    br i1 [[TMP9:%.*]], label %[[PRED_STORE_IF4:.*]], label %[[PRED_STORE_CONTINUE5:.*]]
+; VF2IC2:    br i1 [[TMP9:%.*]], label %[[PRED_STORE_IF4:.*]], label %[[PRED_STORE_CONTINUE5:.*]], !prof [[PROF26]]
 ; VF2IC2:  [[PRED_STORE_IF4]]:
 ; VF2IC2:  [[PRED_STORE_CONTINUE5]]:
-; VF2IC2:    br i1 [[TMP12:%.*]], label %[[PRED_STORE_IF6:.*]], label %[[PRED_STORE_CONTINUE7:.*]]
+; VF2IC2:    br i1 [[TMP12:%.*]], label %[[PRED_STORE_IF6:.*]], label %[[PRED_STORE_CONTINUE7:.*]], !prof [[PROF26]]
 ; VF2IC2:  [[PRED_STORE_IF6]]:
 ; VF2IC2:  [[PRED_STORE_CONTINUE7]]:
 ; VF2IC2:    br i1 [[TMP15:%.*]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !prof [[PROF2]], !llvm.loop [[LOOP50:![0-9]+]]



More information about the llvm-commits mailing list