[llvm] [VPlan] Preserve the branch weights of scalarized selects. (PR #224968)

Florian Hahn via llvm-commits llvm-commits at lists.llvm.org
Thu Oct 1 06:51:58 PDT 2026


https://github.com/fhahn updated https://github.com/llvm/llvm-project/pull/224968

>From 05f84a841ae96fb36ff6f0e87c8dc038f0974d65 Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Sun, 20 Sep 2026 15:02:14 +0100
Subject: [PATCH 1/3] [VPlan] Add VPlan printing test coverage for selects with
 branch weights (NFC).

The branch weights of the selects are currently not carried by their
recipes.
---
 .../VPlan/vplan-printing-branch-weights.ll    | 177 +++++++++++++++++-
 1 file changed, 172 insertions(+), 5 deletions(-)

diff --git a/llvm/test/Transforms/LoopVectorize/VPlan/vplan-printing-branch-weights.ll b/llvm/test/Transforms/LoopVectorize/VPlan/vplan-printing-branch-weights.ll
index e89172c1286bb..276cbc025e3fd 100644
--- a/llvm/test/Transforms/LoopVectorize/VPlan/vplan-printing-branch-weights.ll
+++ b/llvm/test/Transforms/LoopVectorize/VPlan/vplan-printing-branch-weights.ll
@@ -12,10 +12,10 @@
 ; RUN:     -vplan-print-after=dissolveLoopRegions -disable-output %s 2>&1 \
 ; RUN:   | FileCheck --strict-whitespace --check-prefix=DISSOLVE %s
 
-; Track the execution frequency of a predicated block through VPlan, from the
-; branch weights of the original loop to the branch weights of the branch
-; guarding the predicated block.
+; Track the branch weights of the original loop through VPlan.
 
+; The branch weights of a conditional branch become the execution frequency of
+; the predicated block and the branch weights of the branch guarding it.
 define void @predicated_block(ptr noalias %a, ptr noalias %idx) {
 ; PREDICATE-LABEL: VPlan for loop in 'predicated_block'
 ; PREDICATE:  VPlan ' for UF>=1' {
@@ -124,7 +124,7 @@ define void @predicated_block(ptr noalias %a, ptr noalias %idx) {
 ; REGION-NEXT:      vp<[[VP5:%[0-9]+]]> = vector-pointer inbounds i32, ir<%gep.idx>, ir<1>
 ; REGION-NEXT:      WIDEN ir<%i> = load vp<[[VP5]]>
 ; REGION-NEXT:      WIDEN ir<%cmp> = icmp sgt ir<%i>, ir<0>
-; REGION-NEXT:      WIDEN ir<%add> = add ir<%i>, ir<1>
+; REGION-NEXT:      WIDEN ir<%add> = add ir<%i>, ir<1> (!vplan.execution.frequency 2305843009213693952 (25%))
 ; REGION-NEXT:      WIDEN-CAST ir<%t> = trunc ir<%add> to i16
 ; REGION-NEXT:      WIDEN-CAST ir<%ext> = sext ir<%t> to i64
 ; REGION-NEXT:    Successor(s): pred.store
@@ -169,7 +169,7 @@ define void @predicated_block(ptr noalias %a, ptr noalias %idx) {
 ; DISSOLVE-NEXT:    CLONE ir<%gep.idx> = getelementptr inbounds ir<%idx>, vp<%index>
 ; DISSOLVE-NEXT:    WIDEN ir<%i> = load ir<%gep.idx>
 ; DISSOLVE-NEXT:    WIDEN ir<%cmp> = icmp sgt ir<%i>, ir<0>
-; DISSOLVE-NEXT:    WIDEN ir<%add> = add ir<%i>, ir<1>
+; DISSOLVE-NEXT:    WIDEN ir<%add> = add ir<%i>, ir<1> (!vplan.execution.frequency 2305843009213693952 (25%))
 ; DISSOLVE-NEXT:    WIDEN-CAST ir<%t> = trunc ir<%add> to i16
 ; DISSOLVE-NEXT:    WIDEN-CAST ir<%ext> = sext ir<%t> to i64
 ; DISSOLVE-NEXT:    EMIT vp<[[VP1:%[0-9]+]]> = extractelement ir<%cmp>, ir<0>
@@ -230,5 +230,172 @@ exit:
   ret void
 }
 
+define void @selects(ptr noalias %a, ptr noalias %b, i1 %c, i64 %n) {
+; PREDICATE-LABEL: VPlan for loop in 'selects'
+; PREDICATE:  VPlan ' for UF>=1' {
+; PREDICATE-NEXT:  Live-in vp<[[VP0:%[0-9]+]]> = VF
+; PREDICATE-NEXT:  Live-in vp<[[VP1:%[0-9]+]]> = VF * UF
+; PREDICATE-NEXT:  Live-in vp<[[VP2:%[0-9]+]]> = vector-trip-count
+; PREDICATE-NEXT:  Live-in ir<%n> = original trip-count
+; PREDICATE-EMPTY:
+; PREDICATE-NEXT:  ir-bb<entry>:
+; PREDICATE-NEXT:  Successor(s): scalar.ph, vector.ph
+; PREDICATE-EMPTY:
+; PREDICATE-NEXT:  vector.ph:
+; PREDICATE-NEXT:  Successor(s): vector loop
+; PREDICATE-EMPTY:
+; PREDICATE-NEXT:  <x1> vector loop: {
+; PREDICATE-NEXT:  vp<[[VP3:%[0-9]+]]> = CANONICAL-IV
+; PREDICATE-EMPTY:
+; PREDICATE-NEXT:    vector.body:
+; PREDICATE-NEXT:      ir<%iv> = WIDEN-INDUCTION ir<0>, ir<1>, vp<[[VP0]]>
+; PREDICATE-NEXT:      EMIT ir<%gep.a> = getelementptr inbounds ir<%a>, ir<%iv>
+; PREDICATE-NEXT:      EMIT-SCALAR ir<%l> = load ir<%gep.a>
+; PREDICATE-NEXT:      EMIT ir<%cmp> = icmp sgt ir<%l>, ir<0>
+; PREDICATE-NEXT:      EMIT ir<%sel.varying> = select ir<%cmp>, ir<%l>, ir<0>
+; PREDICATE-NEXT:      EMIT store ir<%sel.varying>, ir<%gep.a>
+; PREDICATE-NEXT:      EMIT ir<%sel.uniform> = select ir<%c>, ir<20>, ir<10>
+; PREDICATE-NEXT:      EMIT ir<%gep.b> = getelementptr inbounds ir<%b>, ir<%iv>
+; PREDICATE-NEXT:      EMIT store ir<%sel.uniform>, ir<%gep.b>
+; PREDICATE-NEXT:      EMIT ir<%iv.next> = add ir<%iv>, ir<1>
+; PREDICATE-NEXT:      EMIT ir<%ec> = icmp eq ir<%iv.next>, ir<%n>
+; PREDICATE-NEXT:      EMIT vp<%index.next> = add nuw vp<[[VP3]]>, vp<[[VP1]]>
+; PREDICATE-NEXT:      EMIT branch-on-count vp<%index.next>, vp<[[VP2]]>
+; PREDICATE-NEXT:    No successors
+; PREDICATE-NEXT:  }
+; PREDICATE-NEXT:  Successor(s): middle.block
+; PREDICATE-EMPTY:
+; PREDICATE-NEXT:  middle.block:
+;
+; CONSTRUCT-LABEL: VPlan for loop in 'selects'
+; CONSTRUCT:  VPlan ' for UF>=1' {
+; CONSTRUCT-NEXT:  Live-in vp<[[VP0:%[0-9]+]]> = VF
+; CONSTRUCT-NEXT:  Live-in vp<[[VP1:%[0-9]+]]> = VF * UF
+; CONSTRUCT-NEXT:  Live-in vp<[[VP2:%[0-9]+]]> = vector-trip-count
+; CONSTRUCT-NEXT:  Live-in ir<%n> = original trip-count
+; CONSTRUCT-EMPTY:
+; CONSTRUCT-NEXT:  ir-bb<entry>:
+; CONSTRUCT-NEXT:  Successor(s): scalar.ph, vector.ph
+; CONSTRUCT-EMPTY:
+; CONSTRUCT-NEXT:  vector.ph:
+; CONSTRUCT-NEXT:  Successor(s): vector loop
+; CONSTRUCT-EMPTY:
+; CONSTRUCT-NEXT:  <x1> vector loop: {
+; CONSTRUCT-NEXT:  vp<[[VP3:%[0-9]+]]> = CANONICAL-IV
+; CONSTRUCT-EMPTY:
+; CONSTRUCT-NEXT:    vector.body:
+; CONSTRUCT-NEXT:      ir<%iv> = WIDEN-INDUCTION ir<0>, ir<1>, vp<[[VP0]]>
+; CONSTRUCT-NEXT:      CLONE ir<%gep.a> = getelementptr inbounds ir<%a>, ir<%iv>
+; CONSTRUCT-NEXT:      vp<[[VP4:%[0-9]+]]> = vector-pointer inbounds i32, ir<%gep.a>, ir<1>
+; CONSTRUCT-NEXT:      WIDEN ir<%l> = load vp<[[VP4]]>
+; CONSTRUCT-NEXT:      EMIT ir<%cmp> = icmp sgt ir<%l>, ir<0>
+; CONSTRUCT-NEXT:      EMIT ir<%sel.varying> = select ir<%cmp>, ir<%l>, ir<0>
+; CONSTRUCT-NEXT:      vp<[[VP5:%[0-9]+]]> = vector-pointer inbounds i32, ir<%gep.a>, ir<1>
+; CONSTRUCT-NEXT:      WIDEN store vp<[[VP5]]>, ir<%sel.varying>
+; CONSTRUCT-NEXT:      EMIT ir<%sel.uniform> = select ir<%c>, ir<20>, ir<10>
+; CONSTRUCT-NEXT:      CLONE ir<%gep.b> = getelementptr inbounds ir<%b>, ir<%iv>
+; CONSTRUCT-NEXT:      vp<[[VP6:%[0-9]+]]> = vector-pointer inbounds i32, ir<%gep.b>, ir<1>
+; CONSTRUCT-NEXT:      WIDEN store vp<[[VP6]]>, ir<%sel.uniform>
+; CONSTRUCT-NEXT:      EMIT ir<%iv.next> = add ir<%iv>, ir<1>
+; CONSTRUCT-NEXT:      CLONE ir<%ec> = icmp eq ir<%iv.next>, ir<%n>
+; CONSTRUCT-NEXT:      EMIT vp<%index.next> = add nuw vp<[[VP3]]>, vp<[[VP1]]>
+; CONSTRUCT-NEXT:      EMIT branch-on-count vp<%index.next>, vp<[[VP2]]>
+; CONSTRUCT-NEXT:    No successors
+; CONSTRUCT-NEXT:  }
+; CONSTRUCT-NEXT:  Successor(s): middle.block
+; CONSTRUCT-EMPTY:
+; CONSTRUCT-NEXT:  middle.block:
+;
+; REGION-LABEL: VPlan for loop in 'selects'
+; REGION:  VPlan 'Initial VPlan for VF={2},UF>=1' {
+; REGION-NEXT:  Live-in vp<[[VP0:%[0-9]+]]> = VF
+; REGION-NEXT:  Live-in vp<[[VP1:%[0-9]+]]> = VF * UF
+; REGION-NEXT:  Live-in vp<[[VP2:%[0-9]+]]> = vector-trip-count
+; REGION-NEXT:  Live-in ir<%n> = original trip-count
+; REGION-EMPTY:
+; REGION-NEXT:  ir-bb<entry>:
+; REGION-NEXT:  Successor(s): scalar.ph, vector.ph
+; REGION-EMPTY:
+; REGION-NEXT:  vector.ph:
+; REGION-NEXT:  Successor(s): vector loop
+; REGION-EMPTY:
+; REGION-NEXT:  <x1> vector loop: {
+; REGION-NEXT:  vp<[[VP3:%[0-9]+]]> = CANONICAL-IV
+; REGION-EMPTY:
+; REGION-NEXT:    vector.body:
+; REGION-NEXT:      vp<[[VP4:%[0-9]+]]> = SCALAR-STEPS vp<[[VP3]]>, ir<1>, vp<[[VP0]]>
+; REGION-NEXT:      CLONE ir<%gep.a> = getelementptr inbounds ir<%a>, vp<[[VP4]]>
+; REGION-NEXT:      vp<[[VP5:%[0-9]+]]> = vector-pointer inbounds i32, ir<%gep.a>, ir<1>
+; REGION-NEXT:      WIDEN ir<%l> = load vp<[[VP5]]>
+; REGION-NEXT:      WIDEN ir<%cmp> = icmp sgt ir<%l>, ir<0>
+; REGION-NEXT:      WIDEN ir<%sel.varying> = select ir<%cmp>, ir<%l>, ir<0>
+; REGION-NEXT:      vp<[[VP6:%[0-9]+]]> = vector-pointer inbounds i32, ir<%gep.a>, ir<1>
+; REGION-NEXT:      WIDEN store vp<[[VP6]]>, ir<%sel.varying>
+; REGION-NEXT:      CLONE ir<%sel.uniform> = select ir<%c>, ir<20>, ir<10>
+; REGION-NEXT:      CLONE ir<%gep.b> = getelementptr inbounds ir<%b>, vp<[[VP4]]>
+; REGION-NEXT:      vp<[[VP7:%[0-9]+]]> = vector-pointer inbounds i32, ir<%gep.b>, ir<1>
+; REGION-NEXT:      WIDEN store vp<[[VP7]]>, ir<%sel.uniform>
+; REGION-NEXT:      EMIT vp<%index.next> = add nuw vp<[[VP3]]>, vp<[[VP1]]>
+; REGION-NEXT:      EMIT branch-on-count vp<%index.next>, vp<[[VP2]]>
+; REGION-NEXT:    No successors
+; REGION-NEXT:  }
+; REGION-NEXT:  Successor(s): middle.block
+; REGION-EMPTY:
+; REGION-NEXT:  middle.block:
+;
+; DISSOLVE-LABEL: VPlan for loop in 'selects'
+; DISSOLVE:  VPlan 'Initial VPlan for VF={2},UF={1}' {
+; DISSOLVE-NEXT:  Live-in vp<[[VP0:%[0-9]+]]> = VF * UF
+; DISSOLVE-NEXT:  Live-in vp<[[VP1:%[0-9]+]]> = vector-trip-count
+; DISSOLVE-NEXT:  Live-in ir<%n> = original trip-count
+; DISSOLVE-EMPTY:
+; DISSOLVE-NEXT:  ir-bb<entry>:
+; DISSOLVE-NEXT:    EMIT vp<%min.iters.check> = icmp ult ir<%n>, ir<2>
+; DISSOLVE-NEXT:    EMIT branch-on-cond vp<%min.iters.check>
+; DISSOLVE-NEXT:  Successor(s): scalar.ph, vector.ph
+; DISSOLVE-EMPTY:
+; DISSOLVE-NEXT:  vector.ph:
+; DISSOLVE-NEXT:    CLONE ir<%sel.uniform> = select ir<%c>, ir<20>, ir<10>
+; DISSOLVE-NEXT:  Successor(s): vector.body
+; DISSOLVE-EMPTY:
+; DISSOLVE-NEXT:  vector.body:
+; DISSOLVE-NEXT:    EMIT-SCALAR vp<%index> = phi [ ir<0>, vector.ph ], [ vp<%index.next>, vector.body ]
+; DISSOLVE-NEXT:    CLONE ir<%gep.a> = getelementptr inbounds ir<%a>, vp<%index>
+; DISSOLVE-NEXT:    WIDEN ir<%l> = load ir<%gep.a>
+; DISSOLVE-NEXT:    WIDEN ir<%cmp> = icmp sgt ir<%l>, ir<0>
+; DISSOLVE-NEXT:    WIDEN ir<%sel.varying> = select ir<%cmp>, ir<%l>, ir<0>
+; DISSOLVE-NEXT:    WIDEN store ir<%gep.a>, ir<%sel.varying>
+; DISSOLVE-NEXT:    CLONE ir<%gep.b> = getelementptr inbounds ir<%b>, vp<%index>
+; DISSOLVE-NEXT:    WIDEN store ir<%gep.b>, ir<%sel.uniform>
+; DISSOLVE-NEXT:    EMIT vp<%index.next> = add nuw vp<%index>, vp<[[VP0]]>
+; DISSOLVE-NEXT:    EMIT vp<[[VP3:%[0-9]+]]> = icmp eq vp<%index.next>, vp<[[VP1]]>
+; DISSOLVE-NEXT:    EMIT branch-on-cond vp<[[VP3]]>
+; DISSOLVE-NEXT:  Successor(s): middle.block, vector.body
+; DISSOLVE-EMPTY:
+; DISSOLVE-NEXT:  middle.block:
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %gep.a = getelementptr inbounds i32, ptr %a, i64 %iv
+  %l = load i32, ptr %gep.a, align 4
+  %cmp = icmp sgt i32 %l, 0
+  %sel.varying = select i1 %cmp, i32 %l, i32 0, !prof !2
+  store i32 %sel.varying, ptr %gep.a, align 4
+  %nc = xor i1 %c, true
+  %sel.uniform = select i1 %nc, i32 10, i32 20, !prof !2
+  %gep.b = getelementptr inbounds i32, ptr %b, i64 %iv
+  store i32 %sel.uniform, ptr %gep.b, align 4
+  %iv.next = add i64 %iv, 1
+  %ec = icmp eq i64 %iv.next, %n
+  br i1 %ec, label %exit, label %loop
+
+exit:
+  ret void
+}
+
 !0 = !{!"branch_weights", i32 1, i32 3}
 !1 = !{!"branch_weights", i32 1, i32 999}
+!2 = !{!"branch_weights", i32 3, i32 5}

>From 93b67b683432c5db49bd5958716d4825c482ff99 Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Sun, 13 Sep 2026 20:06:28 +0100
Subject: [PATCH 2/3] [VPlan] Preserve the branch weights of scalarized
 selects.

Update VPlan to also carry !prof for select instructions from original
IR. They are threaded through like for branches, and dropped for wide
selects.

This is part of addressing the remaining profcheck failures in
LoopVectorize: #161235.
---
 llvm/lib/Transforms/Vectorize/VPlan.h         | 10 +-
 .../Transforms/Vectorize/VPlanLowering.cpp    |  7 ++
 .../Transforms/Vectorize/VPlanTransforms.cpp  | 21 ++++
 .../VPlan/vplan-printing-branch-weights.ll    | 14 +--
 .../LoopVectorize/select-branch-weights.ll    | 96 ++++++++++---------
 5 files changed, 92 insertions(+), 56 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/VPlan.h b/llvm/lib/Transforms/Vectorize/VPlan.h
index daabca902724b..1b2aebbae102f 100644
--- a/llvm/lib/Transforms/Vectorize/VPlan.h
+++ b/llvm/lib/Transforms/Vectorize/VPlan.h
@@ -1221,8 +1221,9 @@ class LLVM_ABI_FOR_TEST VPIRMetadata {
   VPIRMetadata(Instruction &I) {
     getMetadataToPropagate(&I, Metadata);
     // Retain the branch weights of terminators. They are used to compute the
-    // frequencies with which the blocks of the original loop execute.
-    if (I.isTerminator())
+    // frequencies with which the blocks of the original loop execute. Also
+    // retain !prof on selects.
+    if (I.isTerminator() || isa<SelectInst>(&I))
       if (MDNode *BW = I.getMetadata(LLVMContext::MD_prof))
         Metadata.emplace_back(LLVMContext::MD_prof, BW);
   }
@@ -1248,6 +1249,11 @@ class LLVM_ABI_FOR_TEST VPIRMetadata {
       Metadata.emplace_back(Kind, Node);
   }
 
+  /// Remove the metadata of kind \p Kind, if present.
+  void eraseMetadata(unsigned Kind) {
+    erase_if(Metadata, [Kind](const auto &P) { return P.first == Kind; });
+  }
+
   /// Intersect this VPIRMetadata object with \p MD, keeping only metadata
   /// nodes that are common to both.
   void intersect(const VPIRMetadata &MD);
diff --git a/llvm/lib/Transforms/Vectorize/VPlanLowering.cpp b/llvm/lib/Transforms/Vectorize/VPlanLowering.cpp
index 230fb5f39d33f..32af9efdc889e 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanLowering.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanLowering.cpp
@@ -527,6 +527,13 @@ void VPlanTransforms::convertToConcreteRecipes(VPlan &Plan) {
            vp_depth_first_deep(Plan.getEntry()))) {
     for (VPRecipeBase &R : make_early_inc_range(*VPBB)) {
       VPBuilder Builder(&R);
+      // !prof is only supported on scalar selects.
+      if (auto *Widen = dyn_cast<VPWidenRecipe>(&R)) {
+        if (Widen->getOpcode() == Instruction::Select &&
+            !vputils::isSingleScalar(Widen->getOperand(0)))
+          Widen->eraseMetadata(LLVMContext::MD_prof);
+      }
+
       if (auto *WidenIVR = dyn_cast<VPWidenIntOrFpInductionRecipe>(&R)) {
         expandVPWidenIntOrFpInduction(WidenIVR);
         WidenIVR->eraseFromParent();
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index 3515a791292da..d2bf6a400b132 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -34,7 +34,9 @@
 #include "llvm/Analysis/ScopedNoAliasAA.h"
 #include "llvm/Analysis/VectorUtils.h"
 #include "llvm/IR/Intrinsics.h"
+#include "llvm/IR/MDBuilder.h"
 #include "llvm/IR/Metadata.h"
+#include "llvm/IR/ProfDataUtils.h"
 #include "llvm/Support/Casting.h"
 #include "llvm/Support/CommandLine.h"
 #include "llvm/Support/TypeSize.h"
@@ -1239,6 +1241,23 @@ static VPValue *simplifyLogicalRecipe(VPlan &Plan, VPSingleDefRecipe *Def) {
   return nullptr;
 }
 
+/// Swap the branch weights recorded for \p R, a select whose two selected
+/// operands are being swapped, so that they keep describing the probability of
+/// its condition. Does nothing if there are none, or if they are the marker for
+/// an explicitly unknown profile, which is symmetric.
+static void swapSelectBranchWeights(VPRecipeBase &R, VPlan &Plan) {
+  auto *MD = dyn_cast<VPIRMetadata>(&R);
+  if (!MD)
+    return;
+  SmallVector<uint32_t, 2> Weights;
+  if (!extractBranchWeights(MD->getMetadata(LLVMContext::MD_prof), Weights) ||
+      Weights.size() != 2)
+    return;
+  MD->setMetadata(
+      LLVMContext::MD_prof,
+      MDBuilder(Plan.getContext()).createBranchWeights(Weights[1], Weights[0]));
+}
+
 /// Return an existing value or a live in for VPSingleDefRecipe \p Def if
 /// possible. This shouldn't create or modify recipes.
 static VPValue *simplifyRecipe(VPlan &Plan, VPSingleDefRecipe *Def) {
@@ -1467,6 +1486,7 @@ static VPSingleDefRecipe *combineRecipe(VPlan &Plan, VPSingleDefRecipe *Def) {
     Def->setOperand(0, C);
     Def->setOperand(1, Y);
     Def->setOperand(2, X);
+    swapSelectBranchWeights(*Def, Plan);
     return Def;
   }
 
@@ -1577,6 +1597,7 @@ static VPSingleDefRecipe *combineRecipe(VPlan &Plan, VPSingleDefRecipe *Def) {
             // select (cmp pred), X, Y -> select (cmp inv_pred), Y, X
             R->setOperand(1, Y);
             R->setOperand(2, X);
+            swapSelectBranchWeights(*R, Plan);
           } else {
             // not (cmp pred) -> cmp inv_pred
             assert(match(R, m_Not(m_Specific(Cmp))) && "Unexpected user");
diff --git a/llvm/test/Transforms/LoopVectorize/VPlan/vplan-printing-branch-weights.ll b/llvm/test/Transforms/LoopVectorize/VPlan/vplan-printing-branch-weights.ll
index 276cbc025e3fd..3485a335e1b46 100644
--- a/llvm/test/Transforms/LoopVectorize/VPlan/vplan-printing-branch-weights.ll
+++ b/llvm/test/Transforms/LoopVectorize/VPlan/vplan-printing-branch-weights.ll
@@ -252,9 +252,9 @@ define void @selects(ptr noalias %a, ptr noalias %b, i1 %c, i64 %n) {
 ; PREDICATE-NEXT:      EMIT ir<%gep.a> = getelementptr inbounds ir<%a>, ir<%iv>
 ; PREDICATE-NEXT:      EMIT-SCALAR ir<%l> = load ir<%gep.a>
 ; PREDICATE-NEXT:      EMIT ir<%cmp> = icmp sgt ir<%l>, ir<0>
-; PREDICATE-NEXT:      EMIT ir<%sel.varying> = select ir<%cmp>, ir<%l>, ir<0>
+; PREDICATE-NEXT:      EMIT ir<%sel.varying> = select ir<%cmp>, ir<%l>, ir<0> (!prof {3, 5})
 ; PREDICATE-NEXT:      EMIT store ir<%sel.varying>, ir<%gep.a>
-; PREDICATE-NEXT:      EMIT ir<%sel.uniform> = select ir<%c>, ir<20>, ir<10>
+; PREDICATE-NEXT:      EMIT ir<%sel.uniform> = select ir<%c>, ir<20>, ir<10> (!prof {5, 3})
 ; PREDICATE-NEXT:      EMIT ir<%gep.b> = getelementptr inbounds ir<%b>, ir<%iv>
 ; PREDICATE-NEXT:      EMIT store ir<%sel.uniform>, ir<%gep.b>
 ; PREDICATE-NEXT:      EMIT ir<%iv.next> = add ir<%iv>, ir<1>
@@ -289,10 +289,10 @@ define void @selects(ptr noalias %a, ptr noalias %b, i1 %c, i64 %n) {
 ; CONSTRUCT-NEXT:      vp<[[VP4:%[0-9]+]]> = vector-pointer inbounds i32, ir<%gep.a>, ir<1>
 ; CONSTRUCT-NEXT:      WIDEN ir<%l> = load vp<[[VP4]]>
 ; CONSTRUCT-NEXT:      EMIT ir<%cmp> = icmp sgt ir<%l>, ir<0>
-; CONSTRUCT-NEXT:      EMIT ir<%sel.varying> = select ir<%cmp>, ir<%l>, ir<0>
+; CONSTRUCT-NEXT:      EMIT ir<%sel.varying> = select ir<%cmp>, ir<%l>, ir<0> (!prof {3, 5})
 ; CONSTRUCT-NEXT:      vp<[[VP5:%[0-9]+]]> = vector-pointer inbounds i32, ir<%gep.a>, ir<1>
 ; CONSTRUCT-NEXT:      WIDEN store vp<[[VP5]]>, ir<%sel.varying>
-; CONSTRUCT-NEXT:      EMIT ir<%sel.uniform> = select ir<%c>, ir<20>, ir<10>
+; CONSTRUCT-NEXT:      EMIT ir<%sel.uniform> = select ir<%c>, ir<20>, ir<10> (!prof {5, 3})
 ; CONSTRUCT-NEXT:      CLONE ir<%gep.b> = getelementptr inbounds ir<%b>, ir<%iv>
 ; CONSTRUCT-NEXT:      vp<[[VP6:%[0-9]+]]> = vector-pointer inbounds i32, ir<%gep.b>, ir<1>
 ; CONSTRUCT-NEXT:      WIDEN store vp<[[VP6]]>, ir<%sel.uniform>
@@ -328,10 +328,10 @@ define void @selects(ptr noalias %a, ptr noalias %b, i1 %c, i64 %n) {
 ; REGION-NEXT:      vp<[[VP5:%[0-9]+]]> = vector-pointer inbounds i32, ir<%gep.a>, ir<1>
 ; REGION-NEXT:      WIDEN ir<%l> = load vp<[[VP5]]>
 ; REGION-NEXT:      WIDEN ir<%cmp> = icmp sgt ir<%l>, ir<0>
-; REGION-NEXT:      WIDEN ir<%sel.varying> = select ir<%cmp>, ir<%l>, ir<0>
+; REGION-NEXT:      WIDEN ir<%sel.varying> = select ir<%cmp>, ir<%l>, ir<0> (!prof {3, 5})
 ; REGION-NEXT:      vp<[[VP6:%[0-9]+]]> = vector-pointer inbounds i32, ir<%gep.a>, ir<1>
 ; REGION-NEXT:      WIDEN store vp<[[VP6]]>, ir<%sel.varying>
-; REGION-NEXT:      CLONE ir<%sel.uniform> = select ir<%c>, ir<20>, ir<10>
+; REGION-NEXT:      CLONE ir<%sel.uniform> = select ir<%c>, ir<20>, ir<10> (!prof {5, 3})
 ; REGION-NEXT:      CLONE ir<%gep.b> = getelementptr inbounds ir<%b>, vp<[[VP4]]>
 ; REGION-NEXT:      vp<[[VP7:%[0-9]+]]> = vector-pointer inbounds i32, ir<%gep.b>, ir<1>
 ; REGION-NEXT:      WIDEN store vp<[[VP7]]>, ir<%sel.uniform>
@@ -355,7 +355,7 @@ define void @selects(ptr noalias %a, ptr noalias %b, i1 %c, i64 %n) {
 ; DISSOLVE-NEXT:  Successor(s): scalar.ph, vector.ph
 ; DISSOLVE-EMPTY:
 ; DISSOLVE-NEXT:  vector.ph:
-; DISSOLVE-NEXT:    CLONE ir<%sel.uniform> = select ir<%c>, ir<20>, ir<10>
+; DISSOLVE-NEXT:    CLONE ir<%sel.uniform> = select ir<%c>, ir<20>, ir<10> (!prof {5, 3})
 ; DISSOLVE-NEXT:  Successor(s): vector.body
 ; DISSOLVE-EMPTY:
 ; DISSOLVE-NEXT:  vector.body:
diff --git a/llvm/test/Transforms/LoopVectorize/select-branch-weights.ll b/llvm/test/Transforms/LoopVectorize/select-branch-weights.ll
index d80bef29e3278..c776a8a54e434 100644
--- a/llvm/test/Transforms/LoopVectorize/select-branch-weights.ll
+++ b/llvm/test/Transforms/LoopVectorize/select-branch-weights.ll
@@ -10,13 +10,13 @@ define void @widened_select_uniform_cond(ptr %a, i1 %c, i64 %n) !prof !0 {
 ; VF4:    br i1 [[MIN_ITERS_CHECK:%.*]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
 ; VF4:  [[VECTOR_PH]]:
 ; VF4:  [[VECTOR_BODY:.*]]:
-; VF4:    [[TMP2:%.*]] = select i1 [[C]], <4 x i32> [[WIDE_LOAD:%.*]], <4 x i32> zeroinitializer
-; VF4:    br i1 [[TMP3:%.*]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP1:![0-9]+]]
+; VF4:    [[TMP2:%.*]] = select i1 [[C]], <4 x i32> [[WIDE_LOAD:%.*]], <4 x i32> zeroinitializer, !prof [[PROF1:![0-9]+]]
+; VF4:    br i1 [[TMP3:%.*]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP2:![0-9]+]]
 ; VF4:  [[MIDDLE_BLOCK]]:
 ; VF4:    br i1 [[CMP_N:%.*]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
 ; VF4:  [[SCALAR_PH]]:
 ; VF4:  [[LOOP:.*]]:
-; VF4:    [[SEL:%.*]] = select i1 [[C]], i32 [[L:%.*]], i32 0, !prof [[PROF4:![0-9]+]]
+; VF4:    [[SEL:%.*]] = select i1 [[C]], i32 [[L:%.*]], i32 0, !prof [[PROF1]]
 ; VF4:    br i1 [[EC:%.*]], label %[[EXIT]], label %[[LOOP]], !llvm.loop [[LOOP5:![0-9]+]]
 ; VF4:  [[EXIT]]:
 ;
@@ -61,13 +61,13 @@ define void @widened_select_varying_cond(ptr %a, i64 %n) !prof !0 {
 ; VF4:    br i1 [[MIN_ITERS_CHECK:%.*]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
 ; VF4:  [[VECTOR_PH]]:
 ; VF4:  [[VECTOR_BODY:.*]]:
-; VF4:    [[TMP3:%.*]] = select <4 x i1> [[TMP2:%.*]], <4 x i32> [[WIDE_LOAD:%.*]], <4 x i32> zeroinitializer
+; VF4:    [[TMP3:%.*]] = select <4 x i1> [[TMP2:%.*]], <4 x i32> [[WIDE_LOAD:%.*]], <4 x i32> zeroinitializer{{$}}
 ; VF4:    br i1 [[TMP4:%.*]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP6:![0-9]+]]
 ; VF4:  [[MIDDLE_BLOCK]]:
 ; VF4:    br i1 [[CMP_N:%.*]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
 ; VF4:  [[SCALAR_PH]]:
 ; VF4:  [[LOOP:.*]]:
-; VF4:    [[SEL:%.*]] = select i1 [[C:%.*]], i32 [[L:%.*]], i32 0, !prof [[PROF4]]
+; VF4:    [[SEL:%.*]] = select i1 [[C:%.*]], i32 [[L:%.*]], i32 0, !prof [[PROF1]]
 ; VF4:    br i1 [[EC:%.*]], label %[[EXIT]], label %[[LOOP]], !llvm.loop [[LOOP7:![0-9]+]]
 ; VF4:  [[EXIT]]:
 ;
@@ -115,15 +115,15 @@ define void @swapped_by_folding_not_into_cmp(ptr %a, ptr %b, i32 %x, i32 %y, i64
 ; VF4:  [[VECTOR_MEMCHECK]]:
 ; VF4:    br i1 [[DIFF_CHECK:%.*]], label %[[SCALAR_PH]], label %[[VECTOR_PH:.*]]
 ; VF4:  [[VECTOR_PH]]:
-; VF4:    [[TMP4:%.*]] = select i1 [[TMP3:%.*]], <4 x i32> splat (i32 20), <4 x i32> splat (i32 10)
+; VF4:    [[TMP4:%.*]] = select i1 [[TMP3:%.*]], <4 x i32> splat (i32 20), <4 x i32> splat (i32 10), !prof [[PROF8:![0-9]+]]
 ; VF4:  [[VECTOR_BODY:.*]]:
-; VF4:    br i1 [[TMP8:%.*]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP8:![0-9]+]]
+; VF4:    br i1 [[TMP8:%.*]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP9:![0-9]+]]
 ; VF4:  [[MIDDLE_BLOCK]]:
 ; VF4:    br i1 [[CMP_N:%.*]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
 ; VF4:  [[SCALAR_PH]]:
 ; VF4:  [[LOOP:.*]]:
-; VF4:    [[SEL:%.*]] = select i1 [[C:%.*]], i32 10, i32 20, !prof [[PROF4]]
-; VF4:    br i1 [[EC:%.*]], label %[[EXIT]], label %[[LOOP]], !llvm.loop [[LOOP9:![0-9]+]]
+; VF4:    [[SEL:%.*]] = select i1 [[C:%.*]], i32 10, i32 20, !prof [[PROF1]]
+; VF4:    br i1 [[EC:%.*]], label %[[EXIT]], label %[[LOOP]], !llvm.loop [[LOOP10:![0-9]+]]
 ; VF4:  [[EXIT]]:
 ;
 ; VF1IC2-LABEL: define void @swapped_by_folding_not_into_cmp(
@@ -135,18 +135,18 @@ define void @swapped_by_folding_not_into_cmp(ptr %a, ptr %b, i32 %x, i32 %y, i64
 ; VF1IC2:  [[VECTOR_BODY:.*]]:
 ; VF1IC2:    br i1 [[TMP5:%.*]], label %[[PRED_STORE_IF:.*]], label %[[PRED_STORE_CONTINUE:.*]]
 ; VF1IC2:  [[PRED_STORE_IF]]:
-; VF1IC2:    [[TMP8:%.*]] = select i1 [[TMP3:%.*]], i32 20, i32 10, !prof [[PROF1]]
+; VF1IC2:    [[TMP8:%.*]] = select i1 [[TMP3:%.*]], i32 20, i32 10, !prof [[PROF6:![0-9]+]]
 ; VF1IC2:  [[PRED_STORE_CONTINUE]]:
 ; VF1IC2:    br i1 [[TMP6:%.*]], label %[[PRED_STORE_IF3:.*]], label %[[PRED_STORE_CONTINUE4:.*]]
 ; VF1IC2:  [[PRED_STORE_IF3]]:
-; VF1IC2:    [[TMP12:%.*]] = select i1 [[TMP3]], i32 20, i32 10, !prof [[PROF1]]
+; VF1IC2:    [[TMP12:%.*]] = select i1 [[TMP3]], i32 20, i32 10, !prof [[PROF6]]
 ; VF1IC2:  [[PRED_STORE_CONTINUE4]]:
-; VF1IC2:    br i1 [[TMP15:%.*]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP6:![0-9]+]]
+; VF1IC2:    br i1 [[TMP15:%.*]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP7:![0-9]+]]
 ; VF1IC2:  [[MIDDLE_BLOCK]]:
 ; VF1IC2:  [[SCALAR_PH]]:
 ; VF1IC2:  [[LOOP:.*]]:
 ; VF1IC2:    [[SEL:%.*]] = select i1 [[C:%.*]], i32 10, i32 20, !prof [[PROF1]]
-; VF1IC2:    br i1 [[EC:%.*]], label %[[EXIT:.*]], label %[[LOOP]], !llvm.loop [[LOOP7:![0-9]+]]
+; VF1IC2:    br i1 [[EC:%.*]], label %[[EXIT:.*]], label %[[LOOP]], !llvm.loop [[LOOP8:![0-9]+]]
 ; VF1IC2:  [[EXIT]]:
 ;
 entry:
@@ -177,22 +177,22 @@ define void @swapped_by_folding_not_into_select(ptr %a, i32 %x, i64 %n) !prof !0
 ; VF4:  [[ENTRY:.*:]]
 ; VF4:    br i1 [[MIN_ITERS_CHECK:%.*]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
 ; VF4:  [[VECTOR_PH]]:
-; VF4:    [[TMP1:%.*]] = select i1 [[C:%.*]], i32 20, i32 10, !prof [[PROF4]]
+; VF4:    [[TMP1:%.*]] = select i1 [[C:%.*]], i32 20, i32 10, !prof [[PROF8]]
 ; VF4:  [[VECTOR_BODY:.*]]:
-; VF4:    br i1 [[TMP3:%.*]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP10:![0-9]+]]
+; VF4:    br i1 [[TMP3:%.*]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP11:![0-9]+]]
 ; VF4:  [[MIDDLE_BLOCK]]:
 ; VF4:    br i1 [[CMP_N:%.*]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
 ; VF4:  [[SCALAR_PH]]:
 ; VF4:  [[LOOP:.*]]:
-; VF4:    [[SEL:%.*]] = select i1 [[NC:%.*]], i32 10, i32 20, !prof [[PROF4]]
-; VF4:    br i1 [[EC:%.*]], label %[[EXIT]], label %[[LOOP]], !llvm.loop [[LOOP11:![0-9]+]]
+; VF4:    [[SEL:%.*]] = select i1 [[NC:%.*]], i32 10, i32 20, !prof [[PROF1]]
+; VF4:    br i1 [[EC:%.*]], label %[[EXIT]], label %[[LOOP]], !llvm.loop [[LOOP12:![0-9]+]]
 ; VF4:  [[EXIT]]:
 ;
 ; VF1IC2-LABEL: define void @swapped_by_folding_not_into_select(
 ; VF1IC2-SAME: ptr [[A:%.*]], i32 [[X:%.*]], i64 [[N:%.*]]) !prof [[PROF0]] {
 ; VF1IC2:  [[ENTRY:.*:]]
 ; VF1IC2:  [[VECTOR_PH:.*:]]
-; VF1IC2:    [[TMP1:%.*]] = select i1 [[C:%.*]], i32 20, i32 10, !prof [[PROF1]]
+; VF1IC2:    [[TMP1:%.*]] = select i1 [[C:%.*]], i32 20, i32 10, !prof [[PROF6]]
 ; VF1IC2:  [[VECTOR_BODY:.*]]:
 ; VF1IC2:    br i1 [[TMP3:%.*]], label %[[PRED_STORE_IF:.*]], label %[[PRED_STORE_CONTINUE:.*]]
 ; VF1IC2:  [[PRED_STORE_IF]]:
@@ -200,7 +200,7 @@ define void @swapped_by_folding_not_into_select(ptr %a, i32 %x, i64 %n) !prof !0
 ; VF1IC2:    br i1 [[TMP4:%.*]], label %[[PRED_STORE_IF1:.*]], label %[[PRED_STORE_CONTINUE2:.*]]
 ; VF1IC2:  [[PRED_STORE_IF1]]:
 ; VF1IC2:  [[PRED_STORE_CONTINUE2]]:
-; VF1IC2:    br i1 [[TMP7:%.*]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP8:![0-9]+]]
+; VF1IC2:    br i1 [[TMP7:%.*]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP9:![0-9]+]]
 ; VF1IC2:  [[MIDDLE_BLOCK]]:
 ; VF1IC2:  [[EXIT:.*:]]
 ;
@@ -240,7 +240,7 @@ define float @ordered_reduction_tail_folded(ptr %src) !prof !0 {
 ; VF4:    br i1 [[TMP18:%.*]], label %[[PRED_LOAD_IF5:.*]], label %[[PRED_LOAD_CONTINUE6:.*]]
 ; VF4:  [[PRED_LOAD_IF5]]:
 ; VF4:  [[PRED_LOAD_CONTINUE6]]:
-; VF4:    br i1 [[TMP25:%.*]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !prof [[PROF12:![0-9]+]], !llvm.loop [[LOOP13:![0-9]+]]
+; VF4:    br i1 [[TMP25:%.*]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !prof [[PROF13:![0-9]+]], !llvm.loop [[LOOP14:![0-9]+]]
 ; VF4:  [[MIDDLE_BLOCK]]:
 ; VF4:    [[TMP26:%.*]] = select contract <4 x i1> [[TMP0:%.*]], <4 x float> [[TMP24:%.*]], <4 x float> [[VEC_PHI:%.*]]
 ; VF4:  [[EXIT:.*:]]
@@ -258,7 +258,7 @@ define float @ordered_reduction_tail_folded(ptr %src) !prof !0 {
 ; VF1IC2:  [[PRED_LOAD_CONTINUE2]]:
 ; VF1IC2:    [[TMP10:%.*]] = select contract i1 [[TMP2]], float [[TMP6:%.*]], float -0.000000e+00
 ; VF1IC2:    [[TMP12:%.*]] = select contract i1 [[TMP3]], float [[TMP9:%.*]], float -0.000000e+00
-; VF1IC2:    br i1 [[TMP14:%.*]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !prof [[PROF9:![0-9]+]], !llvm.loop [[LOOP10:![0-9]+]]
+; VF1IC2:    br i1 [[TMP14:%.*]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !prof [[PROF10:![0-9]+]], !llvm.loop [[LOOP11:![0-9]+]]
 ; VF1IC2:  [[MIDDLE_BLOCK]]:
 ; VF1IC2:  [[EXIT:.*:]]
 ;
@@ -301,7 +301,7 @@ define i32 @logical_and_of_mask_and_condition(ptr noalias %src1, ptr noalias %sr
 ; VF4:  [[PRED_LOAD_IF5]]:
 ; VF4:  [[PRED_LOAD_CONTINUE6]]:
 ; VF4:    [[TMP27:%.*]] = select <4 x i1> [[TMP2:%.*]], <4 x i1> [[TMP26:%.*]], <4 x i1> zeroinitializer
-; VF4:    br i1 [[TMP29:%.*]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP15:![0-9]+]]
+; VF4:    br i1 [[TMP29:%.*]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP16:![0-9]+]]
 ; VF4:  [[MIDDLE_BLOCK]]:
 ; VF4:    [[RDX_SELECT:%.*]] = select i1 [[TMP31:%.*]], i32 1, i32 0
 ; VF4:    br i1 [[CMP_N:%.*]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
@@ -311,7 +311,7 @@ define i32 @logical_and_of_mask_and_condition(ptr noalias %src1, ptr noalias %sr
 ; VF4:  [[THEN]]:
 ; VF4:    [[SEL:%.*]] = select i1 [[C_2:%.*]], i32 1, i32 [[RES:%.*]]
 ; VF4:  [[LOOP_LATCH]]:
-; VF4:    br i1 [[EC:%.*]], label %[[EXIT]], label %[[LOOP_HEADER]], !llvm.loop [[LOOP16:![0-9]+]]
+; VF4:    br i1 [[EC:%.*]], label %[[EXIT]], label %[[LOOP_HEADER]], !llvm.loop [[LOOP17:![0-9]+]]
 ; VF4:  [[EXIT]]:
 ;
 ; VF1IC2-LABEL: define i32 @logical_and_of_mask_and_condition(
@@ -335,7 +335,7 @@ define i32 @logical_and_of_mask_and_condition(ptr noalias %src1, ptr noalias %sr
 ; VF1IC2:  [[PRED_LOAD_CONTINUE7]]:
 ; VF1IC2:    [[TMP23:%.*]] = select i1 [[TMP13]], i1 [[TMP21:%.*]], i1 false
 ; VF1IC2:    [[TMP25:%.*]] = select i1 [[TMP14]], i1 [[TMP22:%.*]], i1 false
-; VF1IC2:    br i1 [[TMP27:%.*]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP12:![0-9]+]]
+; VF1IC2:    br i1 [[TMP27:%.*]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP13:![0-9]+]]
 ; VF1IC2:  [[MIDDLE_BLOCK]]:
 ; VF1IC2:    [[RDX_SELECT:%.*]] = select i1 [[TMP28:%.*]], i32 1, i32 0
 ; VF1IC2:  [[EXIT:.*:]]
@@ -372,22 +372,23 @@ exit:
 !1 = !{!"branch_weights", i32 3, i32 5}
 ;.
 ; VF4: [[PROF0]] = !{!"function_entry_count", i64 1000}
-; VF4: [[LOOP1]] = distinct !{[[LOOP1]], [[META2:![0-9]+]], [[META3:![0-9]+]]}
-; VF4: [[META2]] = !{!"llvm.loop.isvectorized", i32 1}
-; VF4: [[META3]] = !{!"llvm.loop.unroll.runtime.disable"}
-; VF4: [[PROF4]] = !{!"branch_weights", i32 3, i32 5}
-; VF4: [[LOOP5]] = distinct !{[[LOOP5]], [[META3]], [[META2]]}
-; VF4: [[LOOP6]] = distinct !{[[LOOP6]], [[META2]], [[META3]]}
-; VF4: [[LOOP7]] = distinct !{[[LOOP7]], [[META3]], [[META2]]}
-; VF4: [[LOOP8]] = distinct !{[[LOOP8]], [[META2]], [[META3]]}
-; VF4: [[LOOP9]] = distinct !{[[LOOP9]], [[META2]]}
-; VF4: [[LOOP10]] = distinct !{[[LOOP10]], [[META2]], [[META3]]}
-; VF4: [[LOOP11]] = distinct !{[[LOOP11]], [[META3]], [[META2]]}
-; VF4: [[PROF12]] = !{!"branch_weights", i32 1, i32 3}
-; VF4: [[LOOP13]] = distinct !{[[LOOP13]], [[META2]], [[META3]], [[META14:![0-9]+]]}
-; VF4: [[META14]] = !{!"llvm.loop.estimated_trip_count", i32 4}
-; VF4: [[LOOP15]] = distinct !{[[LOOP15]], [[META2]], [[META3]]}
-; VF4: [[LOOP16]] = distinct !{[[LOOP16]], [[META3]], [[META2]]}
+; VF4: [[PROF1]] = !{!"branch_weights", i32 3, i32 5}
+; VF4: [[LOOP2]] = distinct !{[[LOOP2]], [[META3:![0-9]+]], [[META4:![0-9]+]]}
+; VF4: [[META3]] = !{!"llvm.loop.isvectorized", i32 1}
+; VF4: [[META4]] = !{!"llvm.loop.unroll.runtime.disable"}
+; VF4: [[LOOP5]] = distinct !{[[LOOP5]], [[META4]], [[META3]]}
+; VF4: [[LOOP6]] = distinct !{[[LOOP6]], [[META3]], [[META4]]}
+; VF4: [[LOOP7]] = distinct !{[[LOOP7]], [[META4]], [[META3]]}
+; VF4: [[PROF8]] = !{!"branch_weights", i32 5, i32 3}
+; VF4: [[LOOP9]] = distinct !{[[LOOP9]], [[META3]], [[META4]]}
+; VF4: [[LOOP10]] = distinct !{[[LOOP10]], [[META3]]}
+; VF4: [[LOOP11]] = distinct !{[[LOOP11]], [[META3]], [[META4]]}
+; VF4: [[LOOP12]] = distinct !{[[LOOP12]], [[META4]], [[META3]]}
+; VF4: [[PROF13]] = !{!"branch_weights", i32 1, i32 3}
+; VF4: [[LOOP14]] = distinct !{[[LOOP14]], [[META3]], [[META4]], [[META15:![0-9]+]]}
+; VF4: [[META15]] = !{!"llvm.loop.estimated_trip_count", i32 4}
+; VF4: [[LOOP16]] = distinct !{[[LOOP16]], [[META3]], [[META4]]}
+; VF4: [[LOOP17]] = distinct !{[[LOOP17]], [[META4]], [[META3]]}
 ;.
 ; VF1IC2: [[PROF0]] = !{!"function_entry_count", i64 1000}
 ; VF1IC2: [[PROF1]] = !{!"branch_weights", i32 3, i32 5}
@@ -395,11 +396,12 @@ exit:
 ; VF1IC2: [[META3]] = !{!"llvm.loop.isvectorized", i32 1}
 ; VF1IC2: [[META4]] = !{!"llvm.loop.unroll.runtime.disable"}
 ; VF1IC2: [[LOOP5]] = distinct !{[[LOOP5]], [[META3]], [[META4]]}
-; VF1IC2: [[LOOP6]] = distinct !{[[LOOP6]], [[META3]], [[META4]]}
-; VF1IC2: [[LOOP7]] = distinct !{[[LOOP7]], [[META3]]}
-; VF1IC2: [[LOOP8]] = distinct !{[[LOOP8]], [[META3]], [[META4]]}
-; VF1IC2: [[PROF9]] = !{!"branch_weights", i32 1, i32 7}
-; VF1IC2: [[LOOP10]] = distinct !{[[LOOP10]], [[META3]], [[META4]], [[META11:![0-9]+]]}
-; VF1IC2: [[META11]] = !{!"llvm.loop.estimated_trip_count", i32 8}
-; VF1IC2: [[LOOP12]] = distinct !{[[LOOP12]], [[META3]], [[META4]]}
+; VF1IC2: [[PROF6]] = !{!"branch_weights", i32 5, i32 3}
+; VF1IC2: [[LOOP7]] = distinct !{[[LOOP7]], [[META3]], [[META4]]}
+; VF1IC2: [[LOOP8]] = distinct !{[[LOOP8]], [[META3]]}
+; VF1IC2: [[LOOP9]] = distinct !{[[LOOP9]], [[META3]], [[META4]]}
+; VF1IC2: [[PROF10]] = !{!"branch_weights", i32 1, i32 7}
+; VF1IC2: [[LOOP11]] = distinct !{[[LOOP11]], [[META3]], [[META4]], [[META12:![0-9]+]]}
+; VF1IC2: [[META12]] = !{!"llvm.loop.estimated_trip_count", i32 8}
+; VF1IC2: [[LOOP13]] = distinct !{[[LOOP13]], [[META3]], [[META4]]}
 ;.

>From 7cf01405563785f127504ff8c5a20481737a0170 Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Thu, 1 Oct 2026 14:51:16 +0100
Subject: [PATCH 3/3] !fixup turn check into assert

---
 llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp         | 8 +++-----
 .../LoopVectorize/VPlan/vplan-printing-branch-weights.ll  | 4 ++--
 2 files changed, 5 insertions(+), 7 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index ec7be761247ec..909ef9b2b7788 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -1215,17 +1215,15 @@ static VPValue *simplifyLogicalRecipe(VPlan &Plan, VPSingleDefRecipe *Def) {
 }
 
 /// Swap the branch weights recorded for \p R, a select whose two selected
-/// operands are being swapped, so that they keep describing the probability of
-/// its condition. Does nothing if there are none, or if they are the marker for
-/// an explicitly unknown profile, which is symmetric.
+/// operands are being swapped,
 static void swapSelectBranchWeights(VPRecipeBase &R, VPlan &Plan) {
   auto *MD = dyn_cast<VPIRMetadata>(&R);
   if (!MD)
     return;
   SmallVector<uint32_t, 2> Weights;
-  if (!extractBranchWeights(MD->getMetadata(LLVMContext::MD_prof), Weights) ||
-      Weights.size() != 2)
+  if (!extractBranchWeights(MD->getMetadata(LLVMContext::MD_prof), Weights))
     return;
+  assert(Weights.size() == 2 && "unexpected branch weights");
   MD->setMetadata(
       LLVMContext::MD_prof,
       MDBuilder(Plan.getContext()).createBranchWeights(Weights[1], Weights[0]));
diff --git a/llvm/test/Transforms/LoopVectorize/VPlan/vplan-printing-branch-weights.ll b/llvm/test/Transforms/LoopVectorize/VPlan/vplan-printing-branch-weights.ll
index a785befd682d6..0118f87f4b7b8 100644
--- a/llvm/test/Transforms/LoopVectorize/VPlan/vplan-printing-branch-weights.ll
+++ b/llvm/test/Transforms/LoopVectorize/VPlan/vplan-printing-branch-weights.ll
@@ -124,7 +124,7 @@ define void @predicated_block(ptr noalias %a, ptr noalias %idx) {
 ; REGION-NEXT:      vp<[[VP5:%[0-9]+]]> = vector-pointer inbounds i32, ir<%gep.idx>, ir<1>
 ; REGION-NEXT:      WIDEN ir<%i> = load vp<[[VP5]]>
 ; REGION-NEXT:      WIDEN ir<%cmp> = icmp sgt ir<%i>, ir<0>
-; REGION-NEXT:      WIDEN ir<%add> = add ir<%i>, ir<1> (!vplan.execution.frequency 2305843009213693952 (25%))
+; REGION-NEXT:      WIDEN ir<%add> = add ir<%i>, ir<1> (!vplan.execution.frequency 4611686018427387903 (25%))
 ; REGION-NEXT:      WIDEN-CAST ir<%t> = trunc ir<%add> to i16
 ; REGION-NEXT:      WIDEN-CAST ir<%ext> = sext ir<%t> to i64
 ; REGION-NEXT:    Successor(s): pred.store
@@ -169,7 +169,7 @@ define void @predicated_block(ptr noalias %a, ptr noalias %idx) {
 ; DISSOLVE-NEXT:    CLONE ir<%gep.idx> = getelementptr inbounds ir<%idx>, vp<%index>
 ; DISSOLVE-NEXT:    WIDEN ir<%i> = load ir<%gep.idx>
 ; DISSOLVE-NEXT:    WIDEN ir<%cmp> = icmp sgt ir<%i>, ir<0>
-; DISSOLVE-NEXT:    WIDEN ir<%add> = add ir<%i>, ir<1> (!vplan.execution.frequency 2305843009213693952 (25%))
+; DISSOLVE-NEXT:    WIDEN ir<%add> = add ir<%i>, ir<1> (!vplan.execution.frequency 4611686018427387903 (25%))
 ; DISSOLVE-NEXT:    WIDEN-CAST ir<%t> = trunc ir<%add> to i16
 ; DISSOLVE-NEXT:    WIDEN-CAST ir<%ext> = sext ir<%t> to i64
 ; DISSOLVE-NEXT:    EMIT vp<[[VP1:%[0-9]+]]> = extractelement ir<%cmp>, ir<0>



More information about the llvm-commits mailing list