[llvm] 66eb006 - [VPlan] Preserve the branch weights of scalarized selects. (#224968)

via llvm-commits llvm-commits at lists.llvm.org
Fri Oct 2 06:32:20 PDT 2026


Author: Florian Hahn
Date: 2026-10-02T13:32:06Z
New Revision: 66eb006e3a0fb7b3fa8a7ae403750054d3715642

URL: https://github.com/llvm/llvm-project/commit/66eb006e3a0fb7b3fa8a7ae403750054d3715642
DIFF: https://github.com/llvm/llvm-project/commit/66eb006e3a0fb7b3fa8a7ae403750054d3715642.diff

LOG: [VPlan] Preserve the branch weights of scalarized selects. (#224968)

Update VPlan to also carry !prof for select instructions from original
IR. They are threaded through like for branches, and dropped for wide
selects.

This is part of addressing the remaining profcheck failures in
LoopVectorize: https://github.com/llvm/llvm-project/issues/161235.

PR: https://github.com/llvm/llvm-project/pull/224968

Added: 
    

Modified: 
    llvm/lib/Transforms/Vectorize/VPlan.h
    llvm/lib/Transforms/Vectorize/VPlanLowering.cpp
    llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
    llvm/test/Transforms/LoopVectorize/VPlan/vplan-printing-branch-weights.ll
    llvm/test/Transforms/LoopVectorize/select-branch-weights.ll

Removed: 
    


################################################################################
diff  --git a/llvm/lib/Transforms/Vectorize/VPlan.h b/llvm/lib/Transforms/Vectorize/VPlan.h
index 55741c2ac0e42..2edb692e6e6e0 100644
--- a/llvm/lib/Transforms/Vectorize/VPlan.h
+++ b/llvm/lib/Transforms/Vectorize/VPlan.h
@@ -1210,8 +1210,9 @@ class LLVM_ABI_FOR_TEST VPIRMetadata {
   VPIRMetadata(Instruction &I) {
     getMetadataToPropagate(&I, Metadata);
     // Retain the branch weights of terminators. They are used to compute the
-    // frequencies with which the blocks of the original loop execute.
-    if (I.isTerminator())
+    // frequencies with which the blocks of the original loop execute. Also
+    // retain !prof on selects.
+    if (I.isTerminator() || isa<SelectInst>(&I))
       if (MDNode *BW = I.getMetadata(LLVMContext::MD_prof))
         Metadata.emplace_back(LLVMContext::MD_prof, BW);
   }
@@ -1237,6 +1238,11 @@ class LLVM_ABI_FOR_TEST VPIRMetadata {
       Metadata.emplace_back(Kind, Node);
   }
 
+  /// Remove the metadata of kind \p Kind, if present.
+  void eraseMetadata(unsigned Kind) {
+    erase_if(Metadata, [Kind](const auto &P) { return P.first == Kind; });
+  }
+
   /// Intersect this VPIRMetadata object with \p MD, keeping only metadata
   /// nodes that are common to both.
   void intersect(const VPIRMetadata &MD);

diff  --git a/llvm/lib/Transforms/Vectorize/VPlanLowering.cpp b/llvm/lib/Transforms/Vectorize/VPlanLowering.cpp
index 230fb5f39d33f..32af9efdc889e 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanLowering.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanLowering.cpp
@@ -527,6 +527,13 @@ void VPlanTransforms::convertToConcreteRecipes(VPlan &Plan) {
            vp_depth_first_deep(Plan.getEntry()))) {
     for (VPRecipeBase &R : make_early_inc_range(*VPBB)) {
       VPBuilder Builder(&R);
+      // !prof is only supported on scalar selects.
+      if (auto *Widen = dyn_cast<VPWidenRecipe>(&R)) {
+        if (Widen->getOpcode() == Instruction::Select &&
+            !vputils::isSingleScalar(Widen->getOperand(0)))
+          Widen->eraseMetadata(LLVMContext::MD_prof);
+      }
+
       if (auto *WidenIVR = dyn_cast<VPWidenIntOrFpInductionRecipe>(&R)) {
         expandVPWidenIntOrFpInduction(WidenIVR);
         WidenIVR->eraseFromParent();

diff  --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index cc1c9a683e404..909ef9b2b7788 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -34,7 +34,9 @@
 #include "llvm/Analysis/ScopedNoAliasAA.h"
 #include "llvm/Analysis/VectorUtils.h"
 #include "llvm/IR/Intrinsics.h"
+#include "llvm/IR/MDBuilder.h"
 #include "llvm/IR/Metadata.h"
+#include "llvm/IR/ProfDataUtils.h"
 #include "llvm/Support/Casting.h"
 #include "llvm/Support/CommandLine.h"
 #include "llvm/Support/TypeSize.h"
@@ -1212,6 +1214,21 @@ static VPValue *simplifyLogicalRecipe(VPlan &Plan, VPSingleDefRecipe *Def) {
   return nullptr;
 }
 
+/// Swap the branch weights recorded for \p R, a select whose two selected
+/// operands are being swapped,
+static void swapSelectBranchWeights(VPRecipeBase &R, VPlan &Plan) {
+  auto *MD = dyn_cast<VPIRMetadata>(&R);
+  if (!MD)
+    return;
+  SmallVector<uint32_t, 2> Weights;
+  if (!extractBranchWeights(MD->getMetadata(LLVMContext::MD_prof), Weights))
+    return;
+  assert(Weights.size() == 2 && "unexpected branch weights");
+  MD->setMetadata(
+      LLVMContext::MD_prof,
+      MDBuilder(Plan.getContext()).createBranchWeights(Weights[1], Weights[0]));
+}
+
 /// Return an existing value or a live in for VPSingleDefRecipe \p Def if
 /// possible. This shouldn't create or modify recipes.
 static VPValue *simplifyRecipe(VPlan &Plan, VPSingleDefRecipe *Def) {
@@ -1453,6 +1470,7 @@ static VPSingleDefRecipe *combineRecipe(VPlan &Plan, VPSingleDefRecipe *Def,
     Def->setOperand(0, C);
     Def->setOperand(1, Y);
     Def->setOperand(2, X);
+    swapSelectBranchWeights(*Def, Plan);
     return Def;
   }
 
@@ -1571,6 +1589,7 @@ static VPSingleDefRecipe *combineRecipe(VPlan &Plan, VPSingleDefRecipe *Def,
             // select (cmp pred), X, Y -> select (cmp inv_pred), Y, X
             R->setOperand(1, Y);
             R->setOperand(2, X);
+            swapSelectBranchWeights(*R, Plan);
           } else {
             // not (cmp pred) -> cmp inv_pred
             assert(match(R, m_Not(m_Specific(Cmp))) && "Unexpected user");

diff  --git a/llvm/test/Transforms/LoopVectorize/VPlan/vplan-printing-branch-weights.ll b/llvm/test/Transforms/LoopVectorize/VPlan/vplan-printing-branch-weights.ll
index 2a5022b2c229b..0118f87f4b7b8 100644
--- a/llvm/test/Transforms/LoopVectorize/VPlan/vplan-printing-branch-weights.ll
+++ b/llvm/test/Transforms/LoopVectorize/VPlan/vplan-printing-branch-weights.ll
@@ -12,10 +12,10 @@
 ; RUN:     -vplan-print-after=dissolveLoopRegions -disable-output %s 2>&1 \
 ; RUN:   | FileCheck --strict-whitespace --check-prefix=DISSOLVE %s
 
-; Track the execution frequency of a predicated block through VPlan, from the
-; branch weights of the original loop to the branch weights of the branch
-; guarding the predicated block.
+; Track the branch weights of the original loop through VPlan.
 
+; The branch weights of a conditional branch become the execution frequency of
+; the predicated block and the branch weights of the branch guarding it.
 define void @predicated_block(ptr noalias %a, ptr noalias %idx) {
 ; PREDICATE-LABEL: VPlan for loop in 'predicated_block'
 ; PREDICATE:  VPlan ' for UF>=1' {
@@ -124,7 +124,7 @@ define void @predicated_block(ptr noalias %a, ptr noalias %idx) {
 ; REGION-NEXT:      vp<[[VP5:%[0-9]+]]> = vector-pointer inbounds i32, ir<%gep.idx>, ir<1>
 ; REGION-NEXT:      WIDEN ir<%i> = load vp<[[VP5]]>
 ; REGION-NEXT:      WIDEN ir<%cmp> = icmp sgt ir<%i>, ir<0>
-; REGION-NEXT:      WIDEN ir<%add> = add ir<%i>, ir<1>
+; REGION-NEXT:      WIDEN ir<%add> = add ir<%i>, ir<1> (!vplan.execution.frequency 4611686018427387903 (25%))
 ; REGION-NEXT:      WIDEN-CAST ir<%t> = trunc ir<%add> to i16
 ; REGION-NEXT:      WIDEN-CAST ir<%ext> = sext ir<%t> to i64
 ; REGION-NEXT:    Successor(s): pred.store
@@ -169,7 +169,7 @@ define void @predicated_block(ptr noalias %a, ptr noalias %idx) {
 ; DISSOLVE-NEXT:    CLONE ir<%gep.idx> = getelementptr inbounds ir<%idx>, vp<%index>
 ; DISSOLVE-NEXT:    WIDEN ir<%i> = load ir<%gep.idx>
 ; DISSOLVE-NEXT:    WIDEN ir<%cmp> = icmp sgt ir<%i>, ir<0>
-; DISSOLVE-NEXT:    WIDEN ir<%add> = add ir<%i>, ir<1>
+; DISSOLVE-NEXT:    WIDEN ir<%add> = add ir<%i>, ir<1> (!vplan.execution.frequency 4611686018427387903 (25%))
 ; DISSOLVE-NEXT:    WIDEN-CAST ir<%t> = trunc ir<%add> to i16
 ; DISSOLVE-NEXT:    WIDEN-CAST ir<%ext> = sext ir<%t> to i64
 ; DISSOLVE-NEXT:    EMIT vp<[[VP1:%[0-9]+]]> = extractelement ir<%cmp>, ir<0>
@@ -230,5 +230,172 @@ exit:
   ret void
 }
 
+define void @selects(ptr noalias %a, ptr noalias %b, i1 %c, i64 %n) {
+; PREDICATE-LABEL: VPlan for loop in 'selects'
+; PREDICATE:  VPlan ' for UF>=1' {
+; PREDICATE-NEXT:  Live-in vp<[[VP0:%[0-9]+]]> = VF
+; PREDICATE-NEXT:  Live-in vp<[[VP1:%[0-9]+]]> = VF * UF
+; PREDICATE-NEXT:  Live-in vp<[[VP2:%[0-9]+]]> = vector-trip-count
+; PREDICATE-NEXT:  Live-in ir<%n> = original trip-count
+; PREDICATE-EMPTY:
+; PREDICATE-NEXT:  ir-bb<entry>:
+; PREDICATE-NEXT:  Successor(s): scalar.ph, vector.ph
+; PREDICATE-EMPTY:
+; PREDICATE-NEXT:  vector.ph:
+; PREDICATE-NEXT:  Successor(s): vector loop
+; PREDICATE-EMPTY:
+; PREDICATE-NEXT:  <x1> vector loop: {
+; PREDICATE-NEXT:  vp<[[VP3:%[0-9]+]]> = CANONICAL-IV
+; PREDICATE-EMPTY:
+; PREDICATE-NEXT:    vector.body:
+; PREDICATE-NEXT:      ir<%iv> = WIDEN-INDUCTION ir<0>, ir<1>, vp<[[VP0]]>
+; PREDICATE-NEXT:      EMIT ir<%gep.a> = getelementptr inbounds ir<%a>, ir<%iv>
+; PREDICATE-NEXT:      EMIT-SCALAR ir<%l> = load ir<%gep.a>
+; PREDICATE-NEXT:      EMIT ir<%cmp> = icmp sgt ir<%l>, ir<0>
+; PREDICATE-NEXT:      EMIT ir<%sel.varying> = select ir<%cmp>, ir<%l>, ir<0> (!prof {3, 5})
+; PREDICATE-NEXT:      EMIT store ir<%sel.varying>, ir<%gep.a>
+; PREDICATE-NEXT:      EMIT ir<%sel.uniform> = select ir<%c>, ir<20>, ir<10> (!prof {5, 3})
+; PREDICATE-NEXT:      EMIT ir<%gep.b> = getelementptr inbounds ir<%b>, ir<%iv>
+; PREDICATE-NEXT:      EMIT store ir<%sel.uniform>, ir<%gep.b>
+; PREDICATE-NEXT:      EMIT ir<%iv.next> = add ir<%iv>, ir<1>
+; PREDICATE-NEXT:      EMIT ir<%ec> = icmp eq ir<%iv.next>, ir<%n>
+; PREDICATE-NEXT:      EMIT vp<%index.next> = add nuw vp<[[VP3]]>, vp<[[VP1]]>
+; PREDICATE-NEXT:      EMIT branch-on-count vp<%index.next>, vp<[[VP2]]>
+; PREDICATE-NEXT:    No successors
+; PREDICATE-NEXT:  }
+; PREDICATE-NEXT:  Successor(s): middle.block
+; PREDICATE-EMPTY:
+; PREDICATE-NEXT:  middle.block:
+;
+; CONSTRUCT-LABEL: VPlan for loop in 'selects'
+; CONSTRUCT:  VPlan ' for UF>=1' {
+; CONSTRUCT-NEXT:  Live-in vp<[[VP0:%[0-9]+]]> = VF
+; CONSTRUCT-NEXT:  Live-in vp<[[VP1:%[0-9]+]]> = VF * UF
+; CONSTRUCT-NEXT:  Live-in vp<[[VP2:%[0-9]+]]> = vector-trip-count
+; CONSTRUCT-NEXT:  Live-in ir<%n> = original trip-count
+; CONSTRUCT-EMPTY:
+; CONSTRUCT-NEXT:  ir-bb<entry>:
+; CONSTRUCT-NEXT:  Successor(s): scalar.ph, vector.ph
+; CONSTRUCT-EMPTY:
+; CONSTRUCT-NEXT:  vector.ph:
+; CONSTRUCT-NEXT:  Successor(s): vector loop
+; CONSTRUCT-EMPTY:
+; CONSTRUCT-NEXT:  <x1> vector loop: {
+; CONSTRUCT-NEXT:  vp<[[VP3:%[0-9]+]]> = CANONICAL-IV
+; CONSTRUCT-EMPTY:
+; CONSTRUCT-NEXT:    vector.body:
+; CONSTRUCT-NEXT:      ir<%iv> = WIDEN-INDUCTION ir<0>, ir<1>, vp<[[VP0]]>
+; CONSTRUCT-NEXT:      CLONE ir<%gep.a> = getelementptr inbounds ir<%a>, ir<%iv>
+; CONSTRUCT-NEXT:      vp<[[VP4:%[0-9]+]]> = vector-pointer inbounds i32, ir<%gep.a>, ir<1>
+; CONSTRUCT-NEXT:      WIDEN ir<%l> = load vp<[[VP4]]>
+; CONSTRUCT-NEXT:      EMIT ir<%cmp> = icmp sgt ir<%l>, ir<0>
+; CONSTRUCT-NEXT:      EMIT ir<%sel.varying> = select ir<%cmp>, ir<%l>, ir<0> (!prof {3, 5})
+; CONSTRUCT-NEXT:      vp<[[VP5:%[0-9]+]]> = vector-pointer inbounds i32, ir<%gep.a>, ir<1>
+; CONSTRUCT-NEXT:      WIDEN store vp<[[VP5]]>, ir<%sel.varying>
+; CONSTRUCT-NEXT:      EMIT ir<%sel.uniform> = select ir<%c>, ir<20>, ir<10> (!prof {5, 3})
+; CONSTRUCT-NEXT:      CLONE ir<%gep.b> = getelementptr inbounds ir<%b>, ir<%iv>
+; CONSTRUCT-NEXT:      vp<[[VP6:%[0-9]+]]> = vector-pointer inbounds i32, ir<%gep.b>, ir<1>
+; CONSTRUCT-NEXT:      WIDEN store vp<[[VP6]]>, ir<%sel.uniform>
+; CONSTRUCT-NEXT:      EMIT ir<%iv.next> = add ir<%iv>, ir<1>
+; CONSTRUCT-NEXT:      CLONE ir<%ec> = icmp eq ir<%iv.next>, ir<%n>
+; CONSTRUCT-NEXT:      EMIT vp<%index.next> = add nuw vp<[[VP3]]>, vp<[[VP1]]>
+; CONSTRUCT-NEXT:      EMIT branch-on-count vp<%index.next>, vp<[[VP2]]>
+; CONSTRUCT-NEXT:    No successors
+; CONSTRUCT-NEXT:  }
+; CONSTRUCT-NEXT:  Successor(s): middle.block
+; CONSTRUCT-EMPTY:
+; CONSTRUCT-NEXT:  middle.block:
+;
+; REGION-LABEL: VPlan for loop in 'selects'
+; REGION:  VPlan 'Initial VPlan for VF={2},UF>=1' {
+; REGION-NEXT:  Live-in vp<[[VP0:%[0-9]+]]> = VF
+; REGION-NEXT:  Live-in vp<[[VP1:%[0-9]+]]> = VF * UF
+; REGION-NEXT:  Live-in vp<[[VP2:%[0-9]+]]> = vector-trip-count
+; REGION-NEXT:  Live-in ir<%n> = original trip-count
+; REGION-EMPTY:
+; REGION-NEXT:  ir-bb<entry>:
+; REGION-NEXT:  Successor(s): scalar.ph, vector.ph
+; REGION-EMPTY:
+; REGION-NEXT:  vector.ph:
+; REGION-NEXT:  Successor(s): vector loop
+; REGION-EMPTY:
+; REGION-NEXT:  <x1> vector loop: {
+; REGION-NEXT:  vp<[[VP3:%[0-9]+]]> = CANONICAL-IV
+; REGION-EMPTY:
+; REGION-NEXT:    vector.body:
+; REGION-NEXT:      vp<[[VP4:%[0-9]+]]> = SCALAR-STEPS vp<[[VP3]]>, ir<1>, vp<[[VP0]]>
+; REGION-NEXT:      CLONE ir<%gep.a> = getelementptr inbounds ir<%a>, vp<[[VP4]]>
+; REGION-NEXT:      vp<[[VP5:%[0-9]+]]> = vector-pointer inbounds i32, ir<%gep.a>, ir<1>
+; REGION-NEXT:      WIDEN ir<%l> = load vp<[[VP5]]>
+; REGION-NEXT:      WIDEN ir<%cmp> = icmp sgt ir<%l>, ir<0>
+; REGION-NEXT:      WIDEN ir<%sel.varying> = select ir<%cmp>, ir<%l>, ir<0> (!prof {3, 5})
+; REGION-NEXT:      vp<[[VP6:%[0-9]+]]> = vector-pointer inbounds i32, ir<%gep.a>, ir<1>
+; REGION-NEXT:      WIDEN store vp<[[VP6]]>, ir<%sel.varying>
+; REGION-NEXT:      CLONE ir<%sel.uniform> = select ir<%c>, ir<20>, ir<10> (!prof {5, 3})
+; REGION-NEXT:      CLONE ir<%gep.b> = getelementptr inbounds ir<%b>, vp<[[VP4]]>
+; REGION-NEXT:      vp<[[VP7:%[0-9]+]]> = vector-pointer inbounds i32, ir<%gep.b>, ir<1>
+; REGION-NEXT:      WIDEN store vp<[[VP7]]>, ir<%sel.uniform>
+; REGION-NEXT:      EMIT vp<%index.next> = add nuw vp<[[VP3]]>, vp<[[VP1]]>
+; REGION-NEXT:      EMIT branch-on-count vp<%index.next>, vp<[[VP2]]>
+; REGION-NEXT:    No successors
+; REGION-NEXT:  }
+; REGION-NEXT:  Successor(s): middle.block
+; REGION-EMPTY:
+; REGION-NEXT:  middle.block:
+;
+; DISSOLVE-LABEL: VPlan for loop in 'selects'
+; DISSOLVE:  VPlan 'Initial VPlan for VF={2},UF={1}' {
+; DISSOLVE-NEXT:  Live-in vp<[[VP0:%[0-9]+]]> = VF * UF
+; DISSOLVE-NEXT:  Live-in vp<[[VP1:%[0-9]+]]> = vector-trip-count
+; DISSOLVE-NEXT:  Live-in ir<%n> = original trip-count
+; DISSOLVE-EMPTY:
+; DISSOLVE-NEXT:  ir-bb<entry>:
+; DISSOLVE-NEXT:    EMIT vp<%min.iters.check> = icmp ult ir<%n>, ir<2>
+; DISSOLVE-NEXT:    EMIT branch-on-cond vp<%min.iters.check>
+; DISSOLVE-NEXT:  Successor(s): scalar.ph, vector.ph
+; DISSOLVE-EMPTY:
+; DISSOLVE-NEXT:  vector.ph:
+; DISSOLVE-NEXT:    CLONE ir<%sel.uniform> = select ir<%c>, ir<20>, ir<10> (!prof {5, 3})
+; DISSOLVE-NEXT:  Successor(s): vector.body
+; DISSOLVE-EMPTY:
+; DISSOLVE-NEXT:  vector.body:
+; DISSOLVE-NEXT:    EMIT-SCALAR vp<%index> = phi [ ir<0>, vector.ph ], [ vp<%index.next>, vector.body ]
+; DISSOLVE-NEXT:    CLONE ir<%gep.a> = getelementptr inbounds ir<%a>, vp<%index>
+; DISSOLVE-NEXT:    WIDEN ir<%l> = load ir<%gep.a>
+; DISSOLVE-NEXT:    WIDEN ir<%cmp> = icmp sgt ir<%l>, ir<0>
+; DISSOLVE-NEXT:    WIDEN ir<%sel.varying> = select ir<%cmp>, ir<%l>, ir<0>
+; DISSOLVE-NEXT:    WIDEN store ir<%gep.a>, ir<%sel.varying>
+; DISSOLVE-NEXT:    CLONE ir<%gep.b> = getelementptr inbounds ir<%b>, vp<%index>
+; DISSOLVE-NEXT:    WIDEN store ir<%gep.b>, ir<%sel.uniform>
+; DISSOLVE-NEXT:    EMIT vp<%index.next> = add nuw vp<%index>, vp<[[VP0]]>
+; DISSOLVE-NEXT:    EMIT vp<[[VP3:%[0-9]+]]> = icmp eq vp<%index.next>, vp<[[VP1]]>
+; DISSOLVE-NEXT:    EMIT branch-on-cond vp<[[VP3]]>
+; DISSOLVE-NEXT:  Successor(s): middle.block, vector.body
+; DISSOLVE-EMPTY:
+; DISSOLVE-NEXT:  middle.block:
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %gep.a = getelementptr inbounds i32, ptr %a, i64 %iv
+  %l = load i32, ptr %gep.a, align 4
+  %cmp = icmp sgt i32 %l, 0
+  %sel.varying = select i1 %cmp, i32 %l, i32 0, !prof !2
+  store i32 %sel.varying, ptr %gep.a, align 4
+  %nc = xor i1 %c, true
+  %sel.uniform = select i1 %nc, i32 10, i32 20, !prof !2
+  %gep.b = getelementptr inbounds i32, ptr %b, i64 %iv
+  store i32 %sel.uniform, ptr %gep.b, align 4
+  %iv.next = add i64 %iv, 1
+  %ec = icmp eq i64 %iv.next, %n
+  br i1 %ec, label %exit, label %loop
+
+exit:
+  ret void
+}
+
 !0 = !{!"branch_weights", i32 1, i32 3}
 !1 = !{!"branch_weights", i32 1, i32 999}
+!2 = !{!"branch_weights", i32 3, i32 5}

diff  --git a/llvm/test/Transforms/LoopVectorize/select-branch-weights.ll b/llvm/test/Transforms/LoopVectorize/select-branch-weights.ll
index d80bef29e3278..3dd09afb69649 100644
--- a/llvm/test/Transforms/LoopVectorize/select-branch-weights.ll
+++ b/llvm/test/Transforms/LoopVectorize/select-branch-weights.ll
@@ -10,13 +10,13 @@ define void @widened_select_uniform_cond(ptr %a, i1 %c, i64 %n) !prof !0 {
 ; VF4:    br i1 [[MIN_ITERS_CHECK:%.*]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
 ; VF4:  [[VECTOR_PH]]:
 ; VF4:  [[VECTOR_BODY:.*]]:
-; VF4:    [[TMP2:%.*]] = select i1 [[C]], <4 x i32> [[WIDE_LOAD:%.*]], <4 x i32> zeroinitializer
+; VF4:    [[TMP2:%.*]] = select i1 [[C]], <4 x i32> [[WIDE_LOAD:%.*]], <4 x i32> zeroinitializer, !prof [[PROF4:![0-9]+]]
 ; VF4:    br i1 [[TMP3:%.*]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP1:![0-9]+]]
 ; VF4:  [[MIDDLE_BLOCK]]:
 ; VF4:    br i1 [[CMP_N:%.*]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
 ; VF4:  [[SCALAR_PH]]:
 ; VF4:  [[LOOP:.*]]:
-; VF4:    [[SEL:%.*]] = select i1 [[C]], i32 [[L:%.*]], i32 0, !prof [[PROF4:![0-9]+]]
+; VF4:    [[SEL:%.*]] = select i1 [[C]], i32 [[L:%.*]], i32 0, !prof [[PROF4]]
 ; VF4:    br i1 [[EC:%.*]], label %[[EXIT]], label %[[LOOP]], !llvm.loop [[LOOP5:![0-9]+]]
 ; VF4:  [[EXIT]]:
 ;
@@ -61,7 +61,7 @@ define void @widened_select_varying_cond(ptr %a, i64 %n) !prof !0 {
 ; VF4:    br i1 [[MIN_ITERS_CHECK:%.*]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
 ; VF4:  [[VECTOR_PH]]:
 ; VF4:  [[VECTOR_BODY:.*]]:
-; VF4:    [[TMP3:%.*]] = select <4 x i1> [[TMP2:%.*]], <4 x i32> [[WIDE_LOAD:%.*]], <4 x i32> zeroinitializer
+; VF4:    [[TMP3:%.*]] = select <4 x i1> [[TMP2:%.*]], <4 x i32> [[WIDE_LOAD:%.*]], <4 x i32> zeroinitializer{{$}}
 ; VF4:    br i1 [[TMP4:%.*]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP6:![0-9]+]]
 ; VF4:  [[MIDDLE_BLOCK]]:
 ; VF4:    br i1 [[CMP_N:%.*]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
@@ -115,7 +115,7 @@ define void @swapped_by_folding_not_into_cmp(ptr %a, ptr %b, i32 %x, i32 %y, i64
 ; VF4:  [[VECTOR_MEMCHECK]]:
 ; VF4:    br i1 [[DIFF_CHECK:%.*]], label %[[SCALAR_PH]], label %[[VECTOR_PH:.*]]
 ; VF4:  [[VECTOR_PH]]:
-; VF4:    [[TMP4:%.*]] = select i1 [[TMP3:%.*]], <4 x i32> splat (i32 20), <4 x i32> splat (i32 10)
+; VF4:    [[TMP4:%.*]] = select i1 [[TMP3:%.*]], <4 x i32> splat (i32 20), <4 x i32> splat (i32 10), !prof [[PROF8:![0-9]+]]
 ; VF4:  [[VECTOR_BODY:.*]]:
 ; VF4:    br i1 [[TMP8:%.*]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP8:![0-9]+]]
 ; VF4:  [[MIDDLE_BLOCK]]:
@@ -135,11 +135,11 @@ define void @swapped_by_folding_not_into_cmp(ptr %a, ptr %b, i32 %x, i32 %y, i64
 ; VF1IC2:  [[VECTOR_BODY:.*]]:
 ; VF1IC2:    br i1 [[TMP5:%.*]], label %[[PRED_STORE_IF:.*]], label %[[PRED_STORE_CONTINUE:.*]]
 ; VF1IC2:  [[PRED_STORE_IF]]:
-; VF1IC2:    [[TMP8:%.*]] = select i1 [[TMP3:%.*]], i32 20, i32 10, !prof [[PROF1]]
+; VF1IC2:    [[TMP8:%.*]] = select i1 [[TMP3:%.*]], i32 20, i32 10, !prof [[PROF6:![0-9]+]]
 ; VF1IC2:  [[PRED_STORE_CONTINUE]]:
 ; VF1IC2:    br i1 [[TMP6:%.*]], label %[[PRED_STORE_IF3:.*]], label %[[PRED_STORE_CONTINUE4:.*]]
 ; VF1IC2:  [[PRED_STORE_IF3]]:
-; VF1IC2:    [[TMP12:%.*]] = select i1 [[TMP3]], i32 20, i32 10, !prof [[PROF1]]
+; VF1IC2:    [[TMP12:%.*]] = select i1 [[TMP3]], i32 20, i32 10, !prof [[PROF6]]
 ; VF1IC2:  [[PRED_STORE_CONTINUE4]]:
 ; VF1IC2:    br i1 [[TMP15:%.*]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP6:![0-9]+]]
 ; VF1IC2:  [[MIDDLE_BLOCK]]:
@@ -177,7 +177,7 @@ define void @swapped_by_folding_not_into_select(ptr %a, i32 %x, i64 %n) !prof !0
 ; VF4:  [[ENTRY:.*:]]
 ; VF4:    br i1 [[MIN_ITERS_CHECK:%.*]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
 ; VF4:  [[VECTOR_PH]]:
-; VF4:    [[TMP1:%.*]] = select i1 [[C:%.*]], i32 20, i32 10, !prof [[PROF4]]
+; VF4:    [[TMP1:%.*]] = select i1 [[C:%.*]], i32 20, i32 10, !prof [[PROF8]]
 ; VF4:  [[VECTOR_BODY:.*]]:
 ; VF4:    br i1 [[TMP3:%.*]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP10:![0-9]+]]
 ; VF4:  [[MIDDLE_BLOCK]]:
@@ -192,7 +192,7 @@ define void @swapped_by_folding_not_into_select(ptr %a, i32 %x, i64 %n) !prof !0
 ; VF1IC2-SAME: ptr [[A:%.*]], i32 [[X:%.*]], i64 [[N:%.*]]) !prof [[PROF0]] {
 ; VF1IC2:  [[ENTRY:.*:]]
 ; VF1IC2:  [[VECTOR_PH:.*:]]
-; VF1IC2:    [[TMP1:%.*]] = select i1 [[C:%.*]], i32 20, i32 10, !prof [[PROF1]]
+; VF1IC2:    [[TMP1:%.*]] = select i1 [[C:%.*]], i32 20, i32 10, !prof [[PROF6]]
 ; VF1IC2:  [[VECTOR_BODY:.*]]:
 ; VF1IC2:    br i1 [[TMP3:%.*]], label %[[PRED_STORE_IF:.*]], label %[[PRED_STORE_CONTINUE:.*]]
 ; VF1IC2:  [[PRED_STORE_IF]]:
@@ -372,13 +372,14 @@ exit:
 !1 = !{!"branch_weights", i32 3, i32 5}
 ;.
 ; VF4: [[PROF0]] = !{!"function_entry_count", i64 1000}
+; VF4: [[PROF4]] = !{!"branch_weights", i32 3, i32 5}
 ; VF4: [[LOOP1]] = distinct !{[[LOOP1]], [[META2:![0-9]+]], [[META3:![0-9]+]]}
 ; VF4: [[META2]] = !{!"llvm.loop.isvectorized", i32 1}
 ; VF4: [[META3]] = !{!"llvm.loop.unroll.runtime.disable"}
-; VF4: [[PROF4]] = !{!"branch_weights", i32 3, i32 5}
 ; VF4: [[LOOP5]] = distinct !{[[LOOP5]], [[META3]], [[META2]]}
 ; VF4: [[LOOP6]] = distinct !{[[LOOP6]], [[META2]], [[META3]]}
 ; VF4: [[LOOP7]] = distinct !{[[LOOP7]], [[META3]], [[META2]]}
+; VF4: [[PROF8]] = !{!"branch_weights", i32 5, i32 3}
 ; VF4: [[LOOP8]] = distinct !{[[LOOP8]], [[META2]], [[META3]]}
 ; VF4: [[LOOP9]] = distinct !{[[LOOP9]], [[META2]]}
 ; VF4: [[LOOP10]] = distinct !{[[LOOP10]], [[META2]], [[META3]]}
@@ -395,6 +396,7 @@ exit:
 ; VF1IC2: [[META3]] = !{!"llvm.loop.isvectorized", i32 1}
 ; VF1IC2: [[META4]] = !{!"llvm.loop.unroll.runtime.disable"}
 ; VF1IC2: [[LOOP5]] = distinct !{[[LOOP5]], [[META3]], [[META4]]}
+; VF1IC2: [[PROF6]] = !{!"branch_weights", i32 5, i32 3}
 ; VF1IC2: [[LOOP6]] = distinct !{[[LOOP6]], [[META3]], [[META4]]}
 ; VF1IC2: [[LOOP7]] = distinct !{[[LOOP7]], [[META3]]}
 ; VF1IC2: [[LOOP8]] = distinct !{[[LOOP8]], [[META3]], [[META4]]}


        


More information about the llvm-commits mailing list