[llvm] [VPlan] Give zero-weight edges minimum probability. (PR #226150)

Florian Hahn via llvm-commits llvm-commits at lists.llvm.org
Thu Sep 24 05:51:38 PDT 2026


https://github.com/fhahn created https://github.com/llvm/llvm-project/pull/226150

A weight of zero is treated as no information. Previously we dropped
edges that had an explicit weight of 0.

Update getSuccessorProbabilities to match BFI's behavior and treat them
as having minimum probability.

>From 6a1bd43179e62b9ed0b4af82537453d98c0f6f24 Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Thu, 24 Sep 2026 13:28:53 +0100
Subject: [PATCH 1/2] [LV] Precommit test

---
 .../VPlan/execution-frequencies-match-bfi.ll  | 64 +++++++++++++
 .../X86/predicated-instruction-cost.ll        | 54 +++++++++++
 .../replicate-region-branch-weights.ll        | 92 +++++++++++++++++++
 3 files changed, 210 insertions(+)

diff --git a/llvm/test/Transforms/LoopVectorize/VPlan/execution-frequencies-match-bfi.ll b/llvm/test/Transforms/LoopVectorize/VPlan/execution-frequencies-match-bfi.ll
index 244c9add3b2ddd..90bb956d2fa97b 100644
--- a/llvm/test/Transforms/LoopVectorize/VPlan/execution-frequencies-match-bfi.ll
+++ b/llvm/test/Transforms/LoopVectorize/VPlan/execution-frequencies-match-bfi.ll
@@ -67,6 +67,69 @@ exit:
   ret void
 }
 
+define void @single_pred_zero_weight(ptr noalias %a, ptr noalias %b, ptr noalias %idx) {
+; %if.then is only entered via an edge with zero branch weight. BFI treats the
+; edge as cold, not as never taken, and gives %if.then the minimum non-zero
+; frequency of 2^-31.
+;
+;   %loop       1 =        1
+;   %if.then 2^-31 ~ 4.66e-10
+;   %latch      1 =        1
+;
+; TODO: VPlan currently records a frequency of 0 for %if.then.
+;
+; BFI-LABEL: block-frequency-info: single_pred_zero_weight
+; BFI-NEXT:   - entry: float = 1.0,
+; BFI-NEXT:   - loop: float = 1000.0,
+; BFI-NEXT:   - if.then: float = 0.00000046566,
+; BFI-NEXT:   - latch: float = 1000.0,
+; BFI-NEXT:   - exit: float = 1.0,
+;
+; VPLAN-LABEL: VPlan for loop in 'single_pred_zero_weight'
+; VPLAN:         vector.body:
+; VPLAN-NEXT:      ir<%iv> = WIDEN-INDUCTION ir<0>, ir<1>, vp<%{{.+}}>
+; VPLAN-NEXT:      EMIT ir<%gep.idx> = getelementptr inbounds ir<%idx>, ir<%iv>
+; VPLAN-NEXT:      EMIT-SCALAR ir<%i> = load ir<%gep.idx>
+; VPLAN-NEXT:      EMIT ir<%gep.b> = getelementptr inbounds ir<%b>, ir<%iv>
+; VPLAN-NEXT:      EMIT store ir<%i>, ir<%gep.b>{{$}}
+; VPLAN-NEXT:      EMIT ir<%c.0> = icmp sgt ir<%i>, ir<0>
+; VPLAN-NEXT:    Successor(s): if.then
+; VPLAN-EMPTY:
+; VPLAN-NEXT:    if.then:
+; VPLAN-NEXT:      EMIT ir<%gep.a> = getelementptr inbounds ir<%a>, ir<%iv>
+; VPLAN-NEXT:      EMIT store ir<%i>, ir<%gep.a>, ir<%c.0>{{$}}
+; VPLAN-NEXT:    Successor(s): latch
+; VPLAN-EMPTY:
+; VPLAN-NEXT:    latch:
+; VPLAN-NEXT:      EMIT ir<%iv.next> = add ir<%iv>, ir<1>
+; VPLAN-NEXT:      EMIT ir<%ec> = icmp eq ir<%iv.next>, ir<1024>
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %latch ]
+  %gep.idx = getelementptr inbounds i32, ptr %idx, i64 %iv
+  %i = load i32, ptr %gep.idx, align 4
+  %gep.b = getelementptr inbounds i32, ptr %b, i64 %iv
+  store i32 %i, ptr %gep.b, align 4
+  %c.0 = icmp sgt i32 %i, 0
+  br i1 %c.0, label %if.then, label %latch, !prof !13
+
+if.then:
+  %gep.a = getelementptr inbounds i32, ptr %a, i64 %iv
+  store i32 %i, ptr %gep.a, align 4
+  br label %latch
+
+latch:
+  %iv.next = add i64 %iv, 1
+  %ec = icmp eq i64 %iv.next, 1024
+  br i1 %ec, label %exit, label %loop, !prof !3
+
+exit:
+  ret void
+}
+
 define void @two_preds(ptr noalias %a, ptr noalias %b, ptr noalias %c, ptr noalias %idx) {
 ; Execution frequency of each block of the loop. %merge is reached from both
 ; %then (1/4) and %else (3/4 * 1/3 = 1/4).
@@ -975,3 +1038,4 @@ exit:
 !8 = !{!"branch_weights", i32 1, i32 1, i32 1, i32 1, i32 1}
 !9 = !{!"branch_weights", i32 1, i32 1, i32 1, i32 1, i32 1, i32 1, i32 1, i32 1, i32 1}
 !12 = !{!"branch_weights", i32 2, i32 2863311531, i32 2863311531, i32 2863311531}
+!13 = !{!"branch_weights", i32 0, i32 1000}
diff --git a/llvm/test/Transforms/LoopVectorize/X86/predicated-instruction-cost.ll b/llvm/test/Transforms/LoopVectorize/X86/predicated-instruction-cost.ll
index bf996014137e9c..a73c31bfd58155 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/predicated-instruction-cost.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/predicated-instruction-cost.ll
@@ -299,4 +299,58 @@ exit:
   ret i32 0
 }
 
+; %then is only entered via an edge with zero branch weight. The cost of its
+; replicate region must be scaled by the minimal possible execution probability.
+define void @predicated_sdiv_zero_weight(ptr noalias %a, ptr noalias %b, i64 %n) {
+; CHECK-LABEL: define void @predicated_sdiv_zero_weight(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT:  [[SCALAR_PH:.*]]:
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LATCH:.*]] ]
+; CHECK-NEXT:    [[GEP_A:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT:    [[L:%.*]] = load i32, ptr [[GEP_A]], align 4
+; CHECK-NEXT:    [[C:%.*]] = icmp sgt i32 [[L]], 0
+; CHECK-NEXT:    br i1 [[C]], label %[[THEN:.*]], label %[[LATCH]], !prof [[PROF4:![0-9]+]]
+; CHECK:       [[THEN]]:
+; CHECK-NEXT:    [[D:%.*]] = sdiv i32 1000, [[L]]
+; CHECK-NEXT:    [[GEP_B:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 [[IV]]
+; CHECK-NEXT:    store i32 [[D]], ptr [[GEP_B]], align 4
+; CHECK-NEXT:    br label %[[LATCH]]
+; CHECK:       [[LATCH]]:
+; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT:    [[EC:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[EC]], label %[[EXIT:.*]], label %[[LOOP]], !llvm.loop [[LOOP5:![0-9]+]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %latch ]
+  %gep.a = getelementptr inbounds i32, ptr %a, i64 %iv
+  %l = load i32, ptr %gep.a, align 4
+  %c = icmp sgt i32 %l, 0
+  br i1 %c, label %then, label %latch, !prof !0
+
+then:
+  %d = sdiv i32 1000, %l
+  %gep.b = getelementptr inbounds i32, ptr %b, i64 %iv
+  store i32 %d, ptr %gep.b, align 4
+  br label %latch
+
+latch:
+  %iv.next = add nuw nsw i64 %iv, 1
+  %ec = icmp eq i64 %iv.next, %n
+  br i1 %ec, label %exit, label %loop, !llvm.loop !1
+
+exit:
+  ret void
+}
+
 attributes #0 = { "target-cpu"="skylake-avx512" }
+
+!0 = !{!"branch_weights", i32 0, i32 1000}
+!1 = distinct !{!1, !2}
+!2 = !{!"llvm.loop.interleave.count", i32 1}
diff --git a/llvm/test/Transforms/LoopVectorize/replicate-region-branch-weights.ll b/llvm/test/Transforms/LoopVectorize/replicate-region-branch-weights.ll
index b8861ee4590bb5..2f7fdeda4c08f2 100644
--- a/llvm/test/Transforms/LoopVectorize/replicate-region-branch-weights.ll
+++ b/llvm/test/Transforms/LoopVectorize/replicate-region-branch-weights.ll
@@ -1304,6 +1304,91 @@ exit:
   ret void
 }
 
+; Predicated store where the condition has a zero taken weight. As in
+; BlockFrequencyInfo, a zero weight marks the edge as cold, not as never taken,
+; so the replicate region branches must still get (very unlikely) weights.
+define void @predicated_store_zero_taken_weight(ptr %a, i32 %n) {
+; VF4IC1-LABEL: define void @predicated_store_zero_taken_weight(
+; VF4IC1-SAME: ptr [[A:%.*]], i32 [[N:%.*]]) {
+; VF4IC1:  [[ENTRY:.*:]]
+; VF4IC1:    br i1 [[MIN_ITERS_CHECK:%.*]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]], !prof [[PROF0]]
+; VF4IC1:  [[VECTOR_PH]]:
+; VF4IC1:  [[VECTOR_BODY:.*]]:
+; VF4IC1:    br i1 [[TMP3:%.*]], label %[[PRED_STORE_IF:.*]], label %[[PRED_STORE_CONTINUE:.*]]
+; VF4IC1:  [[PRED_STORE_IF]]:
+; VF4IC1:  [[PRED_STORE_CONTINUE]]:
+; VF4IC1:    br i1 [[TMP4:%.*]], label %[[PRED_STORE_IF1:.*]], label %[[PRED_STORE_CONTINUE2:.*]]
+; VF4IC1:  [[PRED_STORE_IF1]]:
+; VF4IC1:  [[PRED_STORE_CONTINUE2]]:
+; VF4IC1:    br i1 [[TMP7:%.*]], label %[[PRED_STORE_IF3:.*]], label %[[PRED_STORE_CONTINUE4:.*]]
+; VF4IC1:  [[PRED_STORE_IF3]]:
+; VF4IC1:  [[PRED_STORE_CONTINUE4]]:
+; VF4IC1:    br i1 [[TMP10:%.*]], label %[[PRED_STORE_IF5:.*]], label %[[PRED_STORE_CONTINUE6:.*]]
+; VF4IC1:  [[PRED_STORE_IF5]]:
+; VF4IC1:  [[PRED_STORE_CONTINUE6]]:
+; VF4IC1:    br i1 [[TMP13:%.*]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !prof [[PROF2]], !llvm.loop [[LOOP50:![0-9]+]]
+; VF4IC1:  [[MIDDLE_BLOCK]]:
+; VF4IC1:    br i1 [[CMP_N:%.*]], label %[[EXIT:.*]], label %[[SCALAR_PH]], !prof [[PROF7]]
+; VF4IC1:  [[SCALAR_PH]]:
+; VF4IC1:  [[LOOP:.*]]:
+; VF4IC1:    br i1 [[C:%.*]], label %[[IF_THEN:.*]], label %[[LATCH:.*]], !prof [[PROF51:![0-9]+]]
+; VF4IC1:  [[IF_THEN]]:
+; VF4IC1:  [[LATCH]]:
+; VF4IC1:    br i1 [[EXITCOND:%.*]], label %[[EXIT]], label %[[LOOP]], !prof [[PROF8]], !llvm.loop [[LOOP52:![0-9]+]]
+; VF4IC1:  [[EXIT]]:
+;
+; VF2IC2-LABEL: define void @predicated_store_zero_taken_weight(
+; VF2IC2-SAME: ptr [[A:%.*]], i32 [[N:%.*]]) {
+; VF2IC2:  [[ENTRY:.*:]]
+; VF2IC2:    br i1 [[MIN_ITERS_CHECK:%.*]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]], !prof [[PROF0]]
+; VF2IC2:  [[VECTOR_PH]]:
+; VF2IC2:  [[VECTOR_BODY:.*]]:
+; VF2IC2:    br i1 [[TMP5:%.*]], label %[[PRED_STORE_IF:.*]], label %[[PRED_STORE_CONTINUE:.*]]
+; VF2IC2:  [[PRED_STORE_IF]]:
+; VF2IC2:  [[PRED_STORE_CONTINUE]]:
+; VF2IC2:    br i1 [[TMP6:%.*]], label %[[PRED_STORE_IF2:.*]], label %[[PRED_STORE_CONTINUE3:.*]]
+; VF2IC2:  [[PRED_STORE_IF2]]:
+; VF2IC2:  [[PRED_STORE_CONTINUE3]]:
+; VF2IC2:    br i1 [[TMP9:%.*]], label %[[PRED_STORE_IF4:.*]], label %[[PRED_STORE_CONTINUE5:.*]]
+; VF2IC2:  [[PRED_STORE_IF4]]:
+; VF2IC2:  [[PRED_STORE_CONTINUE5]]:
+; VF2IC2:    br i1 [[TMP12:%.*]], label %[[PRED_STORE_IF6:.*]], label %[[PRED_STORE_CONTINUE7:.*]]
+; VF2IC2:  [[PRED_STORE_IF6]]:
+; VF2IC2:  [[PRED_STORE_CONTINUE7]]:
+; VF2IC2:    br i1 [[TMP15:%.*]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !prof [[PROF2]], !llvm.loop [[LOOP50:![0-9]+]]
+; VF2IC2:  [[MIDDLE_BLOCK]]:
+; VF2IC2:    br i1 [[CMP_N:%.*]], label %[[EXIT:.*]], label %[[SCALAR_PH]], !prof [[PROF7]]
+; VF2IC2:  [[SCALAR_PH]]:
+; VF2IC2:  [[LOOP:.*]]:
+; VF2IC2:    br i1 [[C:%.*]], label %[[IF_THEN:.*]], label %[[LATCH:.*]], !prof [[PROF51:![0-9]+]]
+; VF2IC2:  [[IF_THEN]]:
+; VF2IC2:  [[LATCH]]:
+; VF2IC2:    br i1 [[EXITCOND:%.*]], label %[[EXIT]], label %[[LOOP]], !prof [[PROF8]], !llvm.loop [[LOOP52:![0-9]+]]
+; VF2IC2:  [[EXIT]]:
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i32 [ 0, %entry ], [ %iv.next, %latch ]
+  %gep = getelementptr inbounds i32, ptr %a, i32 %iv
+  %val = load i32, ptr %gep, align 4
+  %c = icmp sgt i32 %val, 0
+  br i1 %c, label %if.then, label %latch, !prof !10
+
+if.then:
+  store i32 0, ptr %gep, align 4
+  br label %latch
+
+latch:
+  %iv.next = add nuw nsw i32 %iv, 1
+  %exitcond = icmp eq i32 %iv.next, %n
+  br i1 %exitcond, label %exit, label %loop, !prof !0
+
+exit:
+  ret void
+}
+
 !0 = !{!"branch_weights", i32 1, i32 1000}
 !1 = !{!"branch_weights", i32 1, i32 7}
 !2 = !{!"branch_weights", i32 1, i32 1}
@@ -1314,6 +1399,7 @@ exit:
 !7 = !{!"branch_weights", i32 4, i32 1, i32 2, i32 1}
 !8 = !{!"branch_weights", i32 4294967295, i32 1}
 !9 = !{!"branch_weights", i32 1, i32 0}
+!10 = !{!"branch_weights", i32 0, i32 1000}
 ;.
 ; VF4IC1: [[PROF0]] = !{!"branch_weights", i32 1, i32 127}
 ; VF4IC1: [[PROF1]] = !{!"branch_weights", i32 1, i32 7}
@@ -1365,6 +1451,9 @@ exit:
 ; VF4IC1: [[LOOP47]] = distinct !{[[LOOP47]], [[META5]], [[META4]], [[META10]]}
 ; VF4IC1: [[LOOP48]] = distinct !{[[LOOP48]], [[META4]], [[META5]], [[META6]]}
 ; VF4IC1: [[LOOP49]] = distinct !{[[LOOP49]], [[META5]], [[META4]], [[META10]]}
+; VF4IC1: [[LOOP50]] = distinct !{[[LOOP50]], [[META4]], [[META5]], [[META6]]}
+; VF4IC1: [[PROF51]] = !{!"branch_weights", i32 0, i32 1000}
+; VF4IC1: [[LOOP52]] = distinct !{[[LOOP52]], [[META5]], [[META4]], [[META10]]}
 ;.
 ; VF2IC2: [[PROF0]] = !{!"branch_weights", i32 1, i32 127}
 ; VF2IC2: [[PROF1]] = !{!"branch_weights", i32 1, i32 7}
@@ -1416,4 +1505,7 @@ exit:
 ; VF2IC2: [[LOOP47]] = distinct !{[[LOOP47]], [[META5]], [[META4]], [[META10]]}
 ; VF2IC2: [[LOOP48]] = distinct !{[[LOOP48]], [[META4]], [[META5]], [[META6]]}
 ; VF2IC2: [[LOOP49]] = distinct !{[[LOOP49]], [[META5]], [[META4]], [[META10]]}
+; VF2IC2: [[LOOP50]] = distinct !{[[LOOP50]], [[META4]], [[META5]], [[META6]]}
+; VF2IC2: [[PROF51]] = !{!"branch_weights", i32 0, i32 1000}
+; VF2IC2: [[LOOP52]] = distinct !{[[LOOP52]], [[META5]], [[META4]], [[META10]]}
 ;.

>From 85b6c7b07339045d1fecd931acf94c6efaa4ac25 Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Thu, 24 Sep 2026 09:19:50 +0100
Subject: [PATCH 2/2] [VPlan] Give zero-weight edges minimum probability.

A weight of zero is treated as no information. Previously we dropped
edges that had an explicit weight of 0.

Update getSuccessorProbabilities to match BFI's behavior and treat them
as having minimum probability.
---
 llvm/lib/Transforms/Vectorize/VPlanUtils.cpp  |  4 ++
 .../VPlan/execution-frequencies-match-bfi.ll  |  4 +-
 .../X86/predicated-instruction-cost.ll        | 70 +++++++++++++++++--
 .../replicate-region-branch-weights.ll        | 16 ++---
 4 files changed, 77 insertions(+), 17 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp b/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
index 9a07697f66762f..aa3ee234c0b91a 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
@@ -1222,6 +1222,10 @@ getSuccessorProbabilities(const VPBasicBlock *VPBB) {
     auto [Succ, Weight] = SuccWeight;
     if (Total == 0)
       return std::make_pair(Succ, BranchProbability::getUnknown());
+    // Like BlockFrequencyInfo, treat zero branch weights cold, not as never
+    // taken.
+    if (Weight == 0)
+      return std::make_pair(Succ, BranchProbability::getRaw(1));
     return std::make_pair(Succ,
                           getBranchProbabilityKeepingPartial(Weight, Total));
   });
diff --git a/llvm/test/Transforms/LoopVectorize/VPlan/execution-frequencies-match-bfi.ll b/llvm/test/Transforms/LoopVectorize/VPlan/execution-frequencies-match-bfi.ll
index 90bb956d2fa97b..d175f9fed5191f 100644
--- a/llvm/test/Transforms/LoopVectorize/VPlan/execution-frequencies-match-bfi.ll
+++ b/llvm/test/Transforms/LoopVectorize/VPlan/execution-frequencies-match-bfi.ll
@@ -76,8 +76,6 @@ define void @single_pred_zero_weight(ptr noalias %a, ptr noalias %b, ptr noalias
 ;   %if.then 2^-31 ~ 4.66e-10
 ;   %latch      1 =        1
 ;
-; TODO: VPlan currently records a frequency of 0 for %if.then.
-;
 ; BFI-LABEL: block-frequency-info: single_pred_zero_weight
 ; BFI-NEXT:   - entry: float = 1.0,
 ; BFI-NEXT:   - loop: float = 1000.0,
@@ -97,7 +95,7 @@ define void @single_pred_zero_weight(ptr noalias %a, ptr noalias %b, ptr noalias
 ; VPLAN-EMPTY:
 ; VPLAN-NEXT:    if.then:
 ; VPLAN-NEXT:      EMIT ir<%gep.a> = getelementptr inbounds ir<%a>, ir<%iv>
-; VPLAN-NEXT:      EMIT store ir<%i>, ir<%gep.a>, ir<%c.0>{{$}}
+; VPLAN-NEXT:      EMIT store ir<%i>, ir<%gep.a>, ir<%c.0> (!vplan.execution.frequency 4294967296 (4.657e-08%))
 ; VPLAN-NEXT:    Successor(s): latch
 ; VPLAN-EMPTY:
 ; VPLAN-NEXT:    latch:
diff --git a/llvm/test/Transforms/LoopVectorize/X86/predicated-instruction-cost.ll b/llvm/test/Transforms/LoopVectorize/X86/predicated-instruction-cost.ll
index a73c31bfd58155..5890766b37238b 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/predicated-instruction-cost.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/predicated-instruction-cost.ll
@@ -305,22 +305,80 @@ define void @predicated_sdiv_zero_weight(ptr noalias %a, ptr noalias %b, i64 %n)
 ; CHECK-LABEL: define void @predicated_sdiv_zero_weight(
 ; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], i64 [[N:%.*]]) {
 ; CHECK-NEXT:  [[SCALAR_PH:.*]]:
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 4
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH1:.*]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = and i64 [[N]], 3
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
 ; CHECK-NEXT:    br label %[[LOOP:.*]]
 ; CHECK:       [[LOOP]]:
-; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LATCH:.*]] ]
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[PRED_STORE_CONTINUE6:.*]] ]
 ; CHECK-NEXT:    [[GEP_A:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
-; CHECK-NEXT:    [[L:%.*]] = load i32, ptr [[GEP_A]], align 4
+; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[GEP_A]], align 4
+; CHECK-NEXT:    [[TMP2:%.*]] = icmp sgt <4 x i32> [[WIDE_LOAD]], zeroinitializer
+; CHECK-NEXT:    [[TMP3:%.*]] = extractelement <4 x i1> [[TMP2]], i64 0
+; CHECK-NEXT:    br i1 [[TMP3]], label %[[PRED_STORE_IF:.*]], label %[[PRED_STORE_CONTINUE:.*]], !prof [[PROF4:![0-9]+]]
+; CHECK:       [[PRED_STORE_IF]]:
+; CHECK-NEXT:    [[TMP4:%.*]] = extractelement <4 x i32> [[WIDE_LOAD]], i64 0
+; CHECK-NEXT:    [[TMP5:%.*]] = sdiv i32 1000, [[TMP4]]
+; CHECK-NEXT:    [[TMP6:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 [[IV]]
+; CHECK-NEXT:    store i32 [[TMP5]], ptr [[TMP6]], align 4
+; CHECK-NEXT:    br label %[[PRED_STORE_CONTINUE]]
+; CHECK:       [[PRED_STORE_CONTINUE]]:
+; CHECK-NEXT:    [[TMP7:%.*]] = extractelement <4 x i1> [[TMP2]], i64 1
+; CHECK-NEXT:    br i1 [[TMP7]], label %[[PRED_STORE_IF1:.*]], label %[[PRED_STORE_CONTINUE2:.*]], !prof [[PROF4]]
+; CHECK:       [[PRED_STORE_IF1]]:
+; CHECK-NEXT:    [[TMP8:%.*]] = extractelement <4 x i32> [[WIDE_LOAD]], i64 1
+; CHECK-NEXT:    [[TMP9:%.*]] = sdiv i32 1000, [[TMP8]]
+; CHECK-NEXT:    [[TMP10:%.*]] = add i64 [[IV]], 1
+; CHECK-NEXT:    [[TMP11:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 [[TMP10]]
+; CHECK-NEXT:    store i32 [[TMP9]], ptr [[TMP11]], align 4
+; CHECK-NEXT:    br label %[[PRED_STORE_CONTINUE2]]
+; CHECK:       [[PRED_STORE_CONTINUE2]]:
+; CHECK-NEXT:    [[TMP12:%.*]] = extractelement <4 x i1> [[TMP2]], i64 2
+; CHECK-NEXT:    br i1 [[TMP12]], label %[[PRED_STORE_IF3:.*]], label %[[PRED_STORE_CONTINUE4:.*]], !prof [[PROF4]]
+; CHECK:       [[PRED_STORE_IF3]]:
+; CHECK-NEXT:    [[TMP13:%.*]] = extractelement <4 x i32> [[WIDE_LOAD]], i64 2
+; CHECK-NEXT:    [[TMP14:%.*]] = sdiv i32 1000, [[TMP13]]
+; CHECK-NEXT:    [[TMP15:%.*]] = add i64 [[IV]], 2
+; CHECK-NEXT:    [[TMP16:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 [[TMP15]]
+; CHECK-NEXT:    store i32 [[TMP14]], ptr [[TMP16]], align 4
+; CHECK-NEXT:    br label %[[PRED_STORE_CONTINUE4]]
+; CHECK:       [[PRED_STORE_CONTINUE4]]:
+; CHECK-NEXT:    [[TMP17:%.*]] = extractelement <4 x i1> [[TMP2]], i64 3
+; CHECK-NEXT:    br i1 [[TMP17]], label %[[PRED_STORE_IF5:.*]], label %[[PRED_STORE_CONTINUE6]], !prof [[PROF4]]
+; CHECK:       [[PRED_STORE_IF5]]:
+; CHECK-NEXT:    [[TMP18:%.*]] = extractelement <4 x i32> [[WIDE_LOAD]], i64 3
+; CHECK-NEXT:    [[TMP19:%.*]] = sdiv i32 1000, [[TMP18]]
+; CHECK-NEXT:    [[TMP20:%.*]] = add i64 [[IV]], 3
+; CHECK-NEXT:    [[TMP21:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 [[TMP20]]
+; CHECK-NEXT:    store i32 [[TMP19]], ptr [[TMP21]], align 4
+; CHECK-NEXT:    br label %[[PRED_STORE_CONTINUE6]]
+; CHECK:       [[PRED_STORE_CONTINUE6]]:
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[IV]], 4
+; CHECK-NEXT:    [[TMP22:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP22]], label %[[MIDDLE_BLOCK:.*]], label %[[LOOP]], !llvm.loop [[LOOP5:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH1]]
+; CHECK:       [[SCALAR_PH1]]:
+; CHECK-NEXT:    [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[SCALAR_PH]] ]
+; CHECK-NEXT:    br label %[[LOOP1:.*]]
+; CHECK:       [[LOOP1]]:
+; CHECK-NEXT:    [[IV1:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH1]] ], [ [[IV_NEXT:%.*]], %[[LATCH:.*]] ]
+; CHECK-NEXT:    [[GEP_A1:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV1]]
+; CHECK-NEXT:    [[L:%.*]] = load i32, ptr [[GEP_A1]], align 4
 ; CHECK-NEXT:    [[C:%.*]] = icmp sgt i32 [[L]], 0
-; CHECK-NEXT:    br i1 [[C]], label %[[THEN:.*]], label %[[LATCH]], !prof [[PROF4:![0-9]+]]
+; CHECK-NEXT:    br i1 [[C]], label %[[THEN:.*]], label %[[LATCH]], !prof [[PROF6:![0-9]+]]
 ; CHECK:       [[THEN]]:
 ; CHECK-NEXT:    [[D:%.*]] = sdiv i32 1000, [[L]]
-; CHECK-NEXT:    [[GEP_B:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 [[IV]]
+; CHECK-NEXT:    [[GEP_B:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 [[IV1]]
 ; CHECK-NEXT:    store i32 [[D]], ptr [[GEP_B]], align 4
 ; CHECK-NEXT:    br label %[[LATCH]]
 ; CHECK:       [[LATCH]]:
-; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV1]], 1
 ; CHECK-NEXT:    [[EC:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
-; CHECK-NEXT:    br i1 [[EC]], label %[[EXIT:.*]], label %[[LOOP]], !llvm.loop [[LOOP5:![0-9]+]]
+; CHECK-NEXT:    br i1 [[EC]], label %[[EXIT]], label %[[LOOP1]], !llvm.loop [[LOOP7:![0-9]+]]
 ; CHECK:       [[EXIT]]:
 ; CHECK-NEXT:    ret void
 ;
diff --git a/llvm/test/Transforms/LoopVectorize/replicate-region-branch-weights.ll b/llvm/test/Transforms/LoopVectorize/replicate-region-branch-weights.ll
index 2f7fdeda4c08f2..f33fac88f579dd 100644
--- a/llvm/test/Transforms/LoopVectorize/replicate-region-branch-weights.ll
+++ b/llvm/test/Transforms/LoopVectorize/replicate-region-branch-weights.ll
@@ -1314,16 +1314,16 @@ define void @predicated_store_zero_taken_weight(ptr %a, i32 %n) {
 ; VF4IC1:    br i1 [[MIN_ITERS_CHECK:%.*]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]], !prof [[PROF0]]
 ; VF4IC1:  [[VECTOR_PH]]:
 ; VF4IC1:  [[VECTOR_BODY:.*]]:
-; VF4IC1:    br i1 [[TMP3:%.*]], label %[[PRED_STORE_IF:.*]], label %[[PRED_STORE_CONTINUE:.*]]
+; VF4IC1:    br i1 [[TMP3:%.*]], label %[[PRED_STORE_IF:.*]], label %[[PRED_STORE_CONTINUE:.*]], !prof [[PROF26]]
 ; VF4IC1:  [[PRED_STORE_IF]]:
 ; VF4IC1:  [[PRED_STORE_CONTINUE]]:
-; VF4IC1:    br i1 [[TMP4:%.*]], label %[[PRED_STORE_IF1:.*]], label %[[PRED_STORE_CONTINUE2:.*]]
+; VF4IC1:    br i1 [[TMP4:%.*]], label %[[PRED_STORE_IF1:.*]], label %[[PRED_STORE_CONTINUE2:.*]], !prof [[PROF26]]
 ; VF4IC1:  [[PRED_STORE_IF1]]:
 ; VF4IC1:  [[PRED_STORE_CONTINUE2]]:
-; VF4IC1:    br i1 [[TMP7:%.*]], label %[[PRED_STORE_IF3:.*]], label %[[PRED_STORE_CONTINUE4:.*]]
+; VF4IC1:    br i1 [[TMP7:%.*]], label %[[PRED_STORE_IF3:.*]], label %[[PRED_STORE_CONTINUE4:.*]], !prof [[PROF26]]
 ; VF4IC1:  [[PRED_STORE_IF3]]:
 ; VF4IC1:  [[PRED_STORE_CONTINUE4]]:
-; VF4IC1:    br i1 [[TMP10:%.*]], label %[[PRED_STORE_IF5:.*]], label %[[PRED_STORE_CONTINUE6:.*]]
+; VF4IC1:    br i1 [[TMP10:%.*]], label %[[PRED_STORE_IF5:.*]], label %[[PRED_STORE_CONTINUE6:.*]], !prof [[PROF26]]
 ; VF4IC1:  [[PRED_STORE_IF5]]:
 ; VF4IC1:  [[PRED_STORE_CONTINUE6]]:
 ; VF4IC1:    br i1 [[TMP13:%.*]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !prof [[PROF2]], !llvm.loop [[LOOP50:![0-9]+]]
@@ -1343,16 +1343,16 @@ define void @predicated_store_zero_taken_weight(ptr %a, i32 %n) {
 ; VF2IC2:    br i1 [[MIN_ITERS_CHECK:%.*]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]], !prof [[PROF0]]
 ; VF2IC2:  [[VECTOR_PH]]:
 ; VF2IC2:  [[VECTOR_BODY:.*]]:
-; VF2IC2:    br i1 [[TMP5:%.*]], label %[[PRED_STORE_IF:.*]], label %[[PRED_STORE_CONTINUE:.*]]
+; VF2IC2:    br i1 [[TMP5:%.*]], label %[[PRED_STORE_IF:.*]], label %[[PRED_STORE_CONTINUE:.*]], !prof [[PROF26]]
 ; VF2IC2:  [[PRED_STORE_IF]]:
 ; VF2IC2:  [[PRED_STORE_CONTINUE]]:
-; VF2IC2:    br i1 [[TMP6:%.*]], label %[[PRED_STORE_IF2:.*]], label %[[PRED_STORE_CONTINUE3:.*]]
+; VF2IC2:    br i1 [[TMP6:%.*]], label %[[PRED_STORE_IF2:.*]], label %[[PRED_STORE_CONTINUE3:.*]], !prof [[PROF26]]
 ; VF2IC2:  [[PRED_STORE_IF2]]:
 ; VF2IC2:  [[PRED_STORE_CONTINUE3]]:
-; VF2IC2:    br i1 [[TMP9:%.*]], label %[[PRED_STORE_IF4:.*]], label %[[PRED_STORE_CONTINUE5:.*]]
+; VF2IC2:    br i1 [[TMP9:%.*]], label %[[PRED_STORE_IF4:.*]], label %[[PRED_STORE_CONTINUE5:.*]], !prof [[PROF26]]
 ; VF2IC2:  [[PRED_STORE_IF4]]:
 ; VF2IC2:  [[PRED_STORE_CONTINUE5]]:
-; VF2IC2:    br i1 [[TMP12:%.*]], label %[[PRED_STORE_IF6:.*]], label %[[PRED_STORE_CONTINUE7:.*]]
+; VF2IC2:    br i1 [[TMP12:%.*]], label %[[PRED_STORE_IF6:.*]], label %[[PRED_STORE_CONTINUE7:.*]], !prof [[PROF26]]
 ; VF2IC2:  [[PRED_STORE_IF6]]:
 ; VF2IC2:  [[PRED_STORE_CONTINUE7]]:
 ; VF2IC2:    br i1 [[TMP15:%.*]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !prof [[PROF2]], !llvm.loop [[LOOP50:![0-9]+]]



More information about the llvm-commits mailing list