[llvm] [VPlan] Compute scalar cost based on VPlan0 instead of legacy CM. (PR #196845)

Florian Hahn via llvm-commits llvm-commits at lists.llvm.org
Wed Sep 23 07:26:38 PDT 2026


https://github.com/fhahn updated https://github.com/llvm/llvm-project/pull/196845

>From e50eb1045f2130a5d3e4643351aee22fe3a5ca1a Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Fri, 29 May 2026 11:53:44 +0100
Subject: [PATCH 1/7] [VPlan] Compute scalar cost based on VPlan0 instead of
 legacy CM.

Replace legacy cost computation for the scalar loop with a VPlan-based
computation on VPlan0. This requires keeping around a copy of the
initial VPlan0, before introducing vectorization specific concepts (like
predication etc), to accurately reflect the cost of the scalar loop.
---
 .../Vectorize/LoopVectorizationPlanner.h      |   6 +
 .../Transforms/Vectorize/LoopVectorize.cpp    |  74 +++++-------
 .../lib/Transforms/Vectorize/VPlanRecipes.cpp |  68 ++++++++++--
 .../LoopVectorize/AArch64/arith-costs.ll      |   6 +-
 .../AArch64/arith-fp-frem-costs.ll            |  32 +++---
 .../LoopVectorize/AArch64/intrinsiccost.ll    |   9 --
 .../AArch64/multiple-result-intrinsics.ll     |  32 +++---
 .../AArch64/struct-return-cost.ll             |  22 ++--
 .../AArch64/type-shrinkage-zext-costs.ll      |   2 +-
 .../LoopVectorize/ARM/mve-icmpcost.ll         | 103 +++++++++--------
 .../LoopVectorize/ARM/mve-selectandorcost.ll  |   2 +-
 .../LoopVectorize/ARM/mve-shiftcost.ll        |   4 +-
 .../LoopVectorize/ARM/scalar-block-cost.ll    |  61 +++++-----
 .../LoopVectorize/SystemZ/pr47665.ll          |  56 +++-------
 .../WebAssembly/int-mac-reduction-costs.ll    |  76 ++++++-------
 .../X86/CostModel/gather-i16-with-i8-index.ll |  10 +-
 .../X86/CostModel/gather-i32-with-i8-index.ll |  12 +-
 .../X86/CostModel/gather-i64-with-i8-index.ll |  12 +-
 .../X86/CostModel/gather-i8-with-i8-index.ll  |  12 +-
 ...dle-iptr-with-data-layout-to-not-assert.ll |   1 -
 .../interleaved-load-f32-stride-2.ll          |   4 -
 .../interleaved-load-f32-stride-3.ll          |   4 -
 .../interleaved-load-f32-stride-4.ll          |   4 -
 .../interleaved-load-f32-stride-5.ll          |   4 -
 .../interleaved-load-f32-stride-6.ll          |   4 -
 .../interleaved-load-f32-stride-7.ll          |   4 -
 .../interleaved-load-f32-stride-8.ll          |   4 -
 .../interleaved-load-f64-stride-2.ll          |   4 -
 .../interleaved-load-f64-stride-3.ll          |   4 -
 .../interleaved-load-f64-stride-4.ll          |   4 -
 .../interleaved-load-f64-stride-5.ll          |   4 -
 .../interleaved-load-f64-stride-6.ll          |   4 -
 .../interleaved-load-f64-stride-7.ll          |   4 -
 .../interleaved-load-f64-stride-8.ll          |   4 -
 .../interleaved-load-i16-stride-2.ll          |   5 -
 .../interleaved-load-i16-stride-3.ll          |   5 -
 .../interleaved-load-i16-stride-4.ll          |   5 -
 .../interleaved-load-i16-stride-5.ll          |   5 -
 .../interleaved-load-i16-stride-6.ll          |   5 -
 .../interleaved-load-i16-stride-7.ll          |   5 -
 .../interleaved-load-i16-stride-8.ll          |   5 -
 ...nterleaved-load-i32-stride-2-indices-0u.ll |   4 -
 .../interleaved-load-i32-stride-2.ll          |   4 -
 ...terleaved-load-i32-stride-3-indices-01u.ll |   4 -
 ...terleaved-load-i32-stride-3-indices-0uu.ll |   4 -
 .../interleaved-load-i32-stride-3.ll          |   4 -
 ...erleaved-load-i32-stride-4-indices-012u.ll |   4 -
 ...erleaved-load-i32-stride-4-indices-01uu.ll |   4 -
 ...erleaved-load-i32-stride-4-indices-0uuu.ll |   4 -
 .../interleaved-load-i32-stride-4.ll          |   4 -
 .../interleaved-load-i32-stride-5.ll          |   4 -
 .../interleaved-load-i32-stride-6.ll          |   4 -
 .../interleaved-load-i32-stride-7.ll          |   4 -
 .../interleaved-load-i32-stride-8.ll          |   4 -
 .../interleaved-load-i64-stride-2.ll          |   4 -
 .../interleaved-load-i64-stride-3.ll          |   4 -
 .../interleaved-load-i64-stride-4.ll          |   4 -
 .../interleaved-load-i64-stride-5.ll          |   4 -
 .../interleaved-load-i64-stride-6.ll          |   4 -
 .../interleaved-load-i64-stride-7.ll          |   4 -
 .../interleaved-load-i64-stride-8.ll          |   4 -
 .../CostModel/interleaved-load-i8-stride-2.ll |   5 -
 .../CostModel/interleaved-load-i8-stride-3.ll |   5 -
 .../CostModel/interleaved-load-i8-stride-4.ll |   5 -
 .../CostModel/interleaved-load-i8-stride-5.ll |   5 -
 .../CostModel/interleaved-load-i8-stride-6.ll |   5 -
 .../CostModel/interleaved-load-i8-stride-7.ll |   5 -
 .../CostModel/interleaved-load-i8-stride-8.ll |   5 -
 .../masked-gather-i32-with-i8-index.ll        |  32 +++---
 .../masked-gather-i64-with-i8-index.ll        |  32 +++---
 .../CostModel/masked-interleaved-load-i16.ll  |  20 +---
 .../CostModel/masked-interleaved-store-i16.ll |  10 --
 .../X86/CostModel/masked-load-i16.ll          |   8 +-
 .../X86/CostModel/masked-load-i32.ll          |   8 +-
 .../X86/CostModel/masked-load-i64.ll          |   8 +-
 .../X86/CostModel/masked-load-i8.ll           |   8 +-
 .../masked-scatter-i32-with-i8-index.ll       |  13 +--
 .../masked-scatter-i64-with-i8-index.ll       |  13 +--
 .../X86/CostModel/masked-store-i16.ll         |   4 -
 .../X86/CostModel/masked-store-i32.ll         |   5 -
 .../X86/CostModel/masked-store-i64.ll         |   5 -
 .../X86/CostModel/masked-store-i8.ll          |   5 -
 .../CostModel/scatter-i16-with-i8-index.ll    |   5 -
 .../CostModel/scatter-i32-with-i8-index.ll    |   5 -
 .../CostModel/scatter-i64-with-i8-index.ll    |   5 -
 .../X86/CostModel/scatter-i8-with-i8-index.ll |   5 -
 .../X86/CostModel/strided-load-i16.ll         |   4 -
 .../X86/CostModel/strided-load-i32.ll         |   4 -
 .../X86/CostModel/strided-load-i64.ll         |   3 -
 .../X86/CostModel/strided-load-i8.ll          |   4 -
 .../X86/CostModel/vpinstruction-cost.ll       | 105 +++++++++---------
 .../Transforms/LoopVectorize/X86/fneg-cost.ll |  41 ++++++-
 .../X86/fp_to_sint8-cost-model.ll             |   2 +-
 .../LoopVectorize/X86/reduction-small-size.ll |  30 ++---
 .../X86/uint64_to_fp64-cost-model.ll          |   2 +-
 .../LoopVectorize/X86/uniformshift.ll         |   2 +-
 .../X86/vector-scalar-select-cost.ll          |   4 +-
 97 files changed, 471 insertions(+), 727 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
index cb38b0be1808aa..75623179a53347 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
@@ -889,6 +889,9 @@ class LoopVectorizationPlanner {
 
   SmallVector<VPlanPtr, 4> VPlans;
 
+  /// Copy of scalar VPlan0; used for scalar cost computation.
+  VPlanPtr InitialVPlan0;
+
   /// Profitable vector factors.
   SmallVector<VectorizationFactor, 8> ProfitableVFs;
 
@@ -905,6 +908,9 @@ class LoopVectorizationPlanner {
   /// been retired.
   InstructionCost cost(VPlan &Plan, ElementCount VF, VPRegisterUsage *RU) const;
 
+  /// Compute the scalar loop cost of InitialVPlan0.
+  InstructionCost computeScalarCost() const;
+
   /// Precompute costs for certain instructions using the legacy cost model. The
   /// function is used to bring up the VPlan-based cost model to initially avoid
   /// taking different decisions due to inaccuracies in the legacy cost model.
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index d929af8afbd1da..7248923bf84761 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -1282,12 +1282,6 @@ class LoopVectorizationCostModel {
     Scalars.clear();
   }
 
-  /// Returns the expected execution cost. The unit of the cost does
-  /// not matter because we use the 'cost' units to compare different
-  /// vector widths. The cost that is returned is *not* normalized by
-  /// the factor width.
-  InstructionCost expectedCost(ElementCount VF);
-
   /// Returns the execution time cost of an instruction for a given vector
   /// width. Vector width of one means scalar.
   InstructionCost getInstructionCost(Instruction *I, ElementCount VF);
@@ -3755,7 +3749,7 @@ LoopVectorizationPlanner::selectInterleaveCount(VPlan &Plan, ElementCount VF,
   // then we calculate the cost of VF here.
   if (LoopCost == 0) {
     if (VF.isScalar())
-      LoopCost = CM->expectedCost(VF);
+      LoopCost = computeScalarCost();
     else
       LoopCost = cost(Plan, VF, &R);
     assert(LoopCost.isValid() && "Expected to have chosen a VF with valid cost");
@@ -4233,42 +4227,6 @@ InstructionCost LoopVectorizationCostModel::computePredInstDiscount(
   return Discount;
 }
 
-InstructionCost LoopVectorizationCostModel::expectedCost(ElementCount VF) {
-  InstructionCost Cost;
-  assert(VF.isScalar() && "must only be called for scalar VFs");
-
-  // For each block.
-  for (BasicBlock *BB : TheLoop->blocks()) {
-    InstructionCost BlockCost;
-
-    // For each instruction in the old loop.
-    for (Instruction &I : *BB) {
-      // Skip ignored values.
-      if (ValuesToIgnore.count(&I) ||
-          (VF.isVector() && VecValuesToIgnore.count(&I)))
-        continue;
-
-      InstructionCost C = getInstructionCost(&I, VF);
-
-      // Check if we should override the cost.
-      if (C.isValid() && ForceTargetInstructionCost.getNumOccurrences() > 0)
-        C = InstructionCost(ForceTargetInstructionCost);
-
-      BlockCost += C;
-      LLVM_DEBUG(dbgs() << "LV: Found an estimated cost of " << C << " for VF "
-                        << VF << " For instruction: " << I << '\n');
-    }
-
-    // In the scalar loop, we may not always execute the predicated block, if it
-    // is an if-else block. Thus, scale the block's cost by the probability of
-    // executing it. getPredBlockCostDivisor will return 1 for blocks that are
-    // only predicated by the header mask when folding the tail.
-    Cost += BlockCost / getPredBlockCostDivisor(Config.CostKind, BB);
-  }
-
-  return Cost;
-}
-
 /// Gets the address access SCEV for Ptr, if it should be used for cost modeling
 /// according to isAddressSCEVForCost.
 ///
@@ -5599,6 +5557,30 @@ getRecordedExecutionFrequency(const VPBasicBlock *VPBB) {
 }
 #endif
 
+InstructionCost LoopVectorizationPlanner::computeScalarCost() const {
+  ElementCount ScalarVF = ElementCount::getFixed(1);
+  VPCostContext CostCtx(*TLI, *InitialVPlan0, *CM, Config);
+  VPBasicBlock *Header =
+      VPBlockUtils::getPlainCFGHeaderAndLatch(*InitialVPlan0).first;
+  InstructionCost Cost = 0;
+
+  for (VPBasicBlock *VPBB : vp_rpo_plain_cfg_loop_body(Header)) {
+    // Look up the divisor via the first underlying IR instruction in the loop.
+    uint64_t Divisor = 1;
+    for (const VPRecipeBase &R : *VPBB) {
+      auto *UI = dyn_cast_if_present<Instruction>(
+          cast<VPSingleDefRecipe>(&R)->getUnderlyingValue());
+      if (!UI)
+        continue;
+      Divisor = CostCtx.CM.getPredBlockCostDivisor(CostCtx.CostKind,
+                                                   UI->getParent());
+      break;
+    }
+    Cost += VPBB->cost(ScalarVF, CostCtx) / Divisor;
+  }
+  return Cost;
+}
+
 InstructionCost LoopVectorizationPlanner::cost(VPlan &Plan, ElementCount VF,
                                                VPRegisterUsage *RU) const {
   VPCostContext CostCtx(*TLI, Plan, *CM, Config,
@@ -5675,8 +5657,8 @@ LoopVectorizationPlanner::computeBestVF() {
   assert(FirstPlan.hasVF(ScalarVF) &&
          "More than a single plan/VF w/o any plan having scalar VF");
 
-  // TODO: Compute scalar cost using VPlan-based cost model.
-  InstructionCost ScalarCost = CM->expectedCost(ScalarVF);
+  // Compute the scalar cost from VPlan0.
+  InstructionCost ScalarCost = computeScalarCost();
   LLVM_DEBUG(dbgs() << "LV: Scalar loop costs: " << ScalarCost << ".\n");
   VectorizationFactor ScalarFactor(ScalarVF, ScalarCost, ScalarCost);
   VectorizationFactor BestFactor = ScalarFactor;
@@ -6456,6 +6438,8 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan1() {
     assert(verifyExecutionFrequenciesMatchBFI(*VPlan0, OrigLoop, LI, *CM) &&
            "execution frequencies do not match the loop's block frequencies");
   }
+  // Save copy of VPlan0 for scalar cost computation.
+  InitialVPlan0 = VPlanPtr(VPlan0->duplicate());
 
   // Create recipes for header phis. For outer loops, reductions, recurrences
   // and in-loop reductions are empty since legality doesn't detect them.
diff --git a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
index 38c94512f05464..0304127b9907c0 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
@@ -1158,7 +1158,6 @@ InstructionCost VPRecipeWithIRFlags::getCostForRecipeWithOpcode(
   Type *ResultTy = VF.isVector() ? toVectorTy(ScalarTy, VF) : ScalarTy;
   switch (Opcode) {
   case Instruction::FNeg:
-    return Ctx.TTI.getArithmeticInstrCost(Opcode, ResultTy, Ctx.CostKind);
   case Instruction::UDiv:
   case Instruction::SDiv:
   case Instruction::SRem:
@@ -1179,12 +1178,14 @@ InstructionCost VPRecipeWithIRFlags::getCostForRecipeWithOpcode(
   case Instruction::Xor: {
     // Certain instructions can be cheaper if they have a constant second
     // operand. One example of this are shifts on x86.
-    VPValue *RHS = getOperand(1);
-    TargetTransformInfo::OperandValueInfo RHSInfo = Ctx.getOperandInfo(RHS);
-
-    if (RHSInfo.Kind == TargetTransformInfo::OK_AnyValue &&
-        getOperand(1)->isDefinedOutsideLoopRegions())
-      RHSInfo.Kind = TargetTransformInfo::OK_UniformValue;
+    TargetTransformInfo::OperandValueInfo RHSInfo = {
+        TargetTransformInfo::OK_AnyValue, TargetTransformInfo::OP_None};
+    if (Opcode != Instruction::FNeg) {
+      RHSInfo = Ctx.getOperandInfo(getOperand(1));
+      if (RHSInfo.Kind == TargetTransformInfo::OK_AnyValue &&
+          getOperand(1)->isDefinedOutsideLoopRegions())
+        RHSInfo.Kind = TargetTransformInfo::OK_UniformValue;
+    }
 
     Instruction *CtxI = dyn_cast_or_null<Instruction>(getUnderlyingValue());
     SmallVector<const Value *, 4> Operands;
@@ -1205,6 +1206,19 @@ InstructionCost VPRecipeWithIRFlags::getCostForRecipeWithOpcode(
   case Instruction::ExtractValue:
     return Ctx.TTI.getInsertExtractValueCost(Instruction::ExtractValue,
                                              Ctx.CostKind);
+  case Instruction::Load:
+  case Instruction::Store: {
+    bool IsLoad = Opcode == Instruction::Load;
+    const Instruction *UI = getUnderlyingInstr();
+    Type *ValTy = (IsLoad ? this : getOperand(0))->getScalarType();
+    Type *PtrTy = getOperand(!IsLoad)->getScalarType();
+    return Ctx.TTI.getAddressComputationCost(PtrTy, nullptr, nullptr,
+                                             Ctx.CostKind) +
+           Ctx.TTI.getMemoryOpCost(Opcode, ValTy, getLoadStoreAlignment(UI),
+                                   cast<PointerType>(PtrTy)->getAddressSpace(),
+                                   Ctx.CostKind,
+                                   TTI::getOperandInfo(UI->getOperand(0)), UI);
+  }
   case Instruction::ICmp:
   case Instruction::FCmp: {
     Type *ScalarOpTy = getOperand(0)->getScalarType();
@@ -1247,6 +1261,12 @@ InstructionCost VPRecipeWithIRFlags::getCostForRecipeWithOpcode(
         return ReplicateRecipe->isPredicated() ? TTI::CastContextHint::Masked
                                                : TTI::CastContextHint::Normal;
       }
+      // Loads/stores in pre-predication VPlan0 are represented as
+      // VPInstructions; treat them like an unmasked memory access.
+      if (const auto *VPI = dyn_cast<VPInstruction>(R))
+        if (VPI->getOpcode() == Instruction::Load ||
+            VPI->getOpcode() == Instruction::Store)
+          return TTI::CastContextHint::Normal;
       const auto *WidenMemoryRecipe = dyn_cast<VPWidenMemoryRecipe>(R);
       if (WidenMemoryRecipe == nullptr)
         return TTI::CastContextHint::None;
@@ -1357,6 +1377,12 @@ InstructionCost VPRecipeWithIRFlags::getCostForRecipeWithOpcode(
 
 InstructionCost VPInstruction::computeCost(ElementCount VF,
                                            VPCostContext &Ctx) const {
+  // Vector-only opcodes have zero cost at scalar VF.
+  if (VF.isScalar() &&
+      (isVectorToScalar() ||
+       getOpcode() == VPInstruction::FirstOrderRecurrenceSplice))
+    return 0;
+
   // NOTE: At the moment it seems only possible to expose this path for
   // the trunc, zext and sext opcodes.
   // TODO: Update VF arg to use onlyFirstLaneUsed once WidenCast is unified.
@@ -1552,6 +1578,32 @@ InstructionCost VPInstruction::computeCost(ElementCount VF,
     return getCostForRecipeWithOpcode(
         getOpcode(),
         vputils::onlyFirstLaneUsed(this) ? ElementCount::getFixed(1) : VF, Ctx);
+  case Instruction::ExtractValue:
+  case Instruction::FNeg:
+  case Instruction::Freeze:
+    if (!VF.isScalar() || !getUnderlyingValue())
+      return 0;
+    return getCostForRecipeWithOpcode(getOpcode(), VF, Ctx);
+  case Instruction::Load:
+  case Instruction::Store:
+    assert(VF.isScalar() && "only scalar VF expected");
+    return getCostForRecipeWithOpcode(getOpcode(), VF, Ctx);
+  case Instruction::Call: {
+    assert(VF.isScalar() && "only scalar VF expected");
+    auto *CalledFn =
+        cast<Function>(getOperand(getNumOperands() - 1)->getLiveInIRValue());
+    SmallVector<const VPValue *> ArgOps(drop_end(operands()));
+    return VPReplicateRecipe::computeCallCost(CalledFn, getScalarType(), ArgOps,
+                                              /*IsSingleScalar=*/true, VF, Ctx);
+  }
+  case VPInstruction::BranchOnCond:
+  case Instruction::PHI:
+    if (!getUnderlyingValue())
+      return 0;
+    return Ctx.TTI.getCFInstrCost(getOpcode() == Instruction::PHI
+                                      ? Instruction::PHI
+                                      : Instruction::CondBr,
+                                  Ctx.CostKind);
   case VPInstruction::ExtractPenultimateElement:
     if (VF == ElementCount::getScalable(1))
       return InstructionCost::getInvalid();
@@ -1559,8 +1611,6 @@ InstructionCost VPInstruction::computeCost(ElementCount VF,
   default:
     // TODO: Compute cost other VPInstructions once the legacy cost model has
     // been retired.
-    assert(!getUnderlyingValue() &&
-           "unexpected VPInstruction witht underlying value");
     return 0;
   }
 }
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/arith-costs.ll b/llvm/test/Transforms/LoopVectorize/AArch64/arith-costs.ll
index 9b02fd24c828d2..e8b055a97804c5 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/arith-costs.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/arith-costs.ll
@@ -7,7 +7,7 @@ target triple = "arm64-apple-macosx"
 
 define void @udiv_rhs_opt_cost(ptr %dst) #0 {
 ; CHECK-LABEL: 'udiv_rhs_opt_cost'
-; CHECK:  LV: Found an estimated cost of 5 for VF 1 For instruction: %div = udiv i8 %iv.trunc, 3
+; CHECK:  Cost of 5 for VF 1: EMIT ir<%div> = udiv ir<%iv.trunc>, ir<3>
 ; CHECK:  Cost of 5 for VF 2: CLONE ir<%div> = udiv ir<%iv.trunc>, ir<3>
 ; CHECK:  Cost of 0 for VF 2: IR %div = udiv i8 %iv.trunc, 3
 ; CHECK:  Cost of 5 for VF 4: CLONE ir<%div> = udiv ir<%iv.trunc>, ir<3>
@@ -38,8 +38,8 @@ exit:
 
 define void @fneg_used_by_fmul_scalar_cost_is_zero(ptr %dst) #0 {
 ; CHECK-LABEL: 'fneg_used_by_fmul_scalar_cost_is_zero'
-; CHECK:  LV: Found an estimated cost of 0 for VF 1 For instruction: %neg = fneg double %conv
-; CHECK:  LV: Found an estimated cost of 2 for VF 1 For instruction: %mul = fmul double %neg, 2.500000e-04
+; CHECK:  Cost of 0 for VF 1: EMIT ir<%neg> = fneg ir<%conv>
+; CHECK:  Cost of 2 for VF 1: EMIT ir<%mul> = fmul ir<%neg>, ir<2.500000e-04>
 ; CHECK:  Cost of 1 for VF 2: WIDEN ir<%neg> = fneg ir<%conv>
 ; CHECK:  Cost of 2 for VF 2: WIDEN ir<%mul> = fmul ir<%neg>, ir<2.500000e-04>
 ; CHECK:  Cost of 0 for VF 2: IR %neg = fneg double %conv
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/arith-fp-frem-costs.ll b/llvm/test/Transforms/LoopVectorize/AArch64/arith-fp-frem-costs.ll
index 511244b46af148..06adbe718b9979 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/arith-fp-frem-costs.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/arith-fp-frem-costs.ll
@@ -12,43 +12,43 @@ target triple = "aarch64-unknown-linux-gnu"
 
 define void @frem_f64(ptr noalias %in.ptr, ptr noalias %out.ptr) {
 ; NEON-NO-VECLIB-LABEL: 'frem_f64'
-; NEON-NO-VECLIB:  LV: Found an estimated cost of 10 for VF 1 For instruction: %res = frem double %in, %in
+; NEON-NO-VECLIB:  Cost of 10 for VF 1: EMIT ir<%res> = frem ir<%in>, ir<%in>
 ; NEON-NO-VECLIB:  Cost of 24 for VF 2: WIDEN ir<%res> = frem ir<%in>, ir<%in>
 ;
 ; SVE-NO-VECLIB-LABEL: 'frem_f64'
-; SVE-NO-VECLIB:  LV: Found an estimated cost of 10 for VF 1 For instruction: %res = frem double %in, %in
+; SVE-NO-VECLIB:  Cost of 10 for VF 1: EMIT ir<%res> = frem ir<%in>, ir<%in>
 ; SVE-NO-VECLIB:  Cost of 24 for VF 2: WIDEN ir<%res> = frem ir<%in>, ir<%in>
 ; SVE-NO-VECLIB:  Cost of Invalid for VF vscale x 1: WIDEN ir<%res> = frem ir<%in>, ir<%in>
 ; SVE-NO-VECLIB:  Cost of Invalid for VF vscale x 2: WIDEN ir<%res> = frem ir<%in>, ir<%in>
 ;
 ; NEON-ARMPL-LABEL: 'frem_f64'
-; NEON-ARMPL:  LV: Found an estimated cost of 10 for VF 1 For instruction: %res = frem double %in, %in
+; NEON-ARMPL:  Cost of 10 for VF 1: EMIT ir<%res> = frem ir<%in>, ir<%in>
 ; NEON-ARMPL:  Cost of 10 for VF 2: WIDEN ir<%res> = frem ir<%in>, ir<%in>
 ;
 ; NEON-SLEEF-LABEL: 'frem_f64'
-; NEON-SLEEF:  LV: Found an estimated cost of 10 for VF 1 For instruction: %res = frem double %in, %in
+; NEON-SLEEF:  Cost of 10 for VF 1: EMIT ir<%res> = frem ir<%in>, ir<%in>
 ; NEON-SLEEF:  Cost of 10 for VF 2: WIDEN ir<%res> = frem ir<%in>, ir<%in>
 ;
 ; SVE-ARMPL-LABEL: 'frem_f64'
-; SVE-ARMPL:  LV: Found an estimated cost of 10 for VF 1 For instruction: %res = frem double %in, %in
+; SVE-ARMPL:  Cost of 10 for VF 1: EMIT ir<%res> = frem ir<%in>, ir<%in>
 ; SVE-ARMPL:  Cost of 10 for VF 2: WIDEN ir<%res> = frem ir<%in>, ir<%in>
 ; SVE-ARMPL:  Cost of Invalid for VF vscale x 1: WIDEN ir<%res> = frem ir<%in>, ir<%in>
 ; SVE-ARMPL:  Cost of 10 for VF vscale x 2: WIDEN ir<%res> = frem ir<%in>, ir<%in>
 ;
 ; SVE-SLEEF-LABEL: 'frem_f64'
-; SVE-SLEEF:  LV: Found an estimated cost of 10 for VF 1 For instruction: %res = frem double %in, %in
+; SVE-SLEEF:  Cost of 10 for VF 1: EMIT ir<%res> = frem ir<%in>, ir<%in>
 ; SVE-SLEEF:  Cost of 10 for VF 2: WIDEN ir<%res> = frem ir<%in>, ir<%in>
 ; SVE-SLEEF:  Cost of Invalid for VF vscale x 1: WIDEN ir<%res> = frem ir<%in>, ir<%in>
 ; SVE-SLEEF:  Cost of 10 for VF vscale x 2: WIDEN ir<%res> = frem ir<%in>, ir<%in>
 ;
 ; SVE-ARMPL-TAILFOLD-LABEL: 'frem_f64'
-; SVE-ARMPL-TAILFOLD:  LV: Found an estimated cost of 10 for VF 1 For instruction: %res = frem double %in, %in
+; SVE-ARMPL-TAILFOLD:  Cost of 10 for VF 1: EMIT ir<%res> = frem ir<%in>, ir<%in>
 ; SVE-ARMPL-TAILFOLD:  Cost of 10 for VF 2: WIDEN ir<%res> = frem ir<%in>, ir<%in>
 ; SVE-ARMPL-TAILFOLD:  Cost of Invalid for VF vscale x 1: WIDEN ir<%res> = frem ir<%in>, ir<%in>
 ; SVE-ARMPL-TAILFOLD:  Cost of 10 for VF vscale x 2: WIDEN ir<%res> = frem ir<%in>, ir<%in>
 ;
 ; SVE-SLEEF-TAILFOLD-LABEL: 'frem_f64'
-; SVE-SLEEF-TAILFOLD:  LV: Found an estimated cost of 10 for VF 1 For instruction: %res = frem double %in, %in
+; SVE-SLEEF-TAILFOLD:  Cost of 10 for VF 1: EMIT ir<%res> = frem ir<%in>, ir<%in>
 ; SVE-SLEEF-TAILFOLD:  Cost of 10 for VF 2: WIDEN ir<%res> = frem ir<%in>, ir<%in>
 ; SVE-SLEEF-TAILFOLD:  Cost of Invalid for VF vscale x 1: WIDEN ir<%res> = frem ir<%in>, ir<%in>
 ; SVE-SLEEF-TAILFOLD:  Cost of 10 for VF vscale x 2: WIDEN ir<%res> = frem ir<%in>, ir<%in>
@@ -73,12 +73,12 @@ define void @frem_f64(ptr noalias %in.ptr, ptr noalias %out.ptr) {
 
 define void @frem_f32(ptr noalias %in.ptr, ptr noalias %out.ptr) {
 ; NEON-NO-VECLIB-LABEL: 'frem_f32'
-; NEON-NO-VECLIB:  LV: Found an estimated cost of 10 for VF 1 For instruction: %res = frem float %in, %in
+; NEON-NO-VECLIB:  Cost of 10 for VF 1: EMIT ir<%res> = frem ir<%in>, ir<%in>
 ; NEON-NO-VECLIB:  Cost of 24 for VF 2: WIDEN ir<%res> = frem ir<%in>, ir<%in>
 ; NEON-NO-VECLIB:  Cost of 52 for VF 4: WIDEN ir<%res> = frem ir<%in>, ir<%in>
 ;
 ; SVE-NO-VECLIB-LABEL: 'frem_f32'
-; SVE-NO-VECLIB:  LV: Found an estimated cost of 10 for VF 1 For instruction: %res = frem float %in, %in
+; SVE-NO-VECLIB:  Cost of 10 for VF 1: EMIT ir<%res> = frem ir<%in>, ir<%in>
 ; SVE-NO-VECLIB:  Cost of 24 for VF 2: WIDEN ir<%res> = frem ir<%in>, ir<%in>
 ; SVE-NO-VECLIB:  Cost of 52 for VF 4: WIDEN ir<%res> = frem ir<%in>, ir<%in>
 ; SVE-NO-VECLIB:  Cost of Invalid for VF vscale x 1: WIDEN ir<%res> = frem ir<%in>, ir<%in>
@@ -86,17 +86,17 @@ define void @frem_f32(ptr noalias %in.ptr, ptr noalias %out.ptr) {
 ; SVE-NO-VECLIB:  Cost of Invalid for VF vscale x 4: WIDEN ir<%res> = frem ir<%in>, ir<%in>
 ;
 ; NEON-ARMPL-LABEL: 'frem_f32'
-; NEON-ARMPL:  LV: Found an estimated cost of 10 for VF 1 For instruction: %res = frem float %in, %in
+; NEON-ARMPL:  Cost of 10 for VF 1: EMIT ir<%res> = frem ir<%in>, ir<%in>
 ; NEON-ARMPL:  Cost of 24 for VF 2: WIDEN ir<%res> = frem ir<%in>, ir<%in>
 ; NEON-ARMPL:  Cost of 10 for VF 4: WIDEN ir<%res> = frem ir<%in>, ir<%in>
 ;
 ; NEON-SLEEF-LABEL: 'frem_f32'
-; NEON-SLEEF:  LV: Found an estimated cost of 10 for VF 1 For instruction: %res = frem float %in, %in
+; NEON-SLEEF:  Cost of 10 for VF 1: EMIT ir<%res> = frem ir<%in>, ir<%in>
 ; NEON-SLEEF:  Cost of 24 for VF 2: WIDEN ir<%res> = frem ir<%in>, ir<%in>
 ; NEON-SLEEF:  Cost of 10 for VF 4: WIDEN ir<%res> = frem ir<%in>, ir<%in>
 ;
 ; SVE-ARMPL-LABEL: 'frem_f32'
-; SVE-ARMPL:  LV: Found an estimated cost of 10 for VF 1 For instruction: %res = frem float %in, %in
+; SVE-ARMPL:  Cost of 10 for VF 1: EMIT ir<%res> = frem ir<%in>, ir<%in>
 ; SVE-ARMPL:  Cost of 24 for VF 2: WIDEN ir<%res> = frem ir<%in>, ir<%in>
 ; SVE-ARMPL:  Cost of 10 for VF 4: WIDEN ir<%res> = frem ir<%in>, ir<%in>
 ; SVE-ARMPL:  Cost of Invalid for VF vscale x 1: WIDEN ir<%res> = frem ir<%in>, ir<%in>
@@ -104,7 +104,7 @@ define void @frem_f32(ptr noalias %in.ptr, ptr noalias %out.ptr) {
 ; SVE-ARMPL:  Cost of 10 for VF vscale x 4: WIDEN ir<%res> = frem ir<%in>, ir<%in>
 ;
 ; SVE-SLEEF-LABEL: 'frem_f32'
-; SVE-SLEEF:  LV: Found an estimated cost of 10 for VF 1 For instruction: %res = frem float %in, %in
+; SVE-SLEEF:  Cost of 10 for VF 1: EMIT ir<%res> = frem ir<%in>, ir<%in>
 ; SVE-SLEEF:  Cost of 24 for VF 2: WIDEN ir<%res> = frem ir<%in>, ir<%in>
 ; SVE-SLEEF:  Cost of 10 for VF 4: WIDEN ir<%res> = frem ir<%in>, ir<%in>
 ; SVE-SLEEF:  Cost of Invalid for VF vscale x 1: WIDEN ir<%res> = frem ir<%in>, ir<%in>
@@ -112,7 +112,7 @@ define void @frem_f32(ptr noalias %in.ptr, ptr noalias %out.ptr) {
 ; SVE-SLEEF:  Cost of 10 for VF vscale x 4: WIDEN ir<%res> = frem ir<%in>, ir<%in>
 ;
 ; SVE-ARMPL-TAILFOLD-LABEL: 'frem_f32'
-; SVE-ARMPL-TAILFOLD:  LV: Found an estimated cost of 10 for VF 1 For instruction: %res = frem float %in, %in
+; SVE-ARMPL-TAILFOLD:  Cost of 10 for VF 1: EMIT ir<%res> = frem ir<%in>, ir<%in>
 ; SVE-ARMPL-TAILFOLD:  Cost of 24 for VF 2: WIDEN ir<%res> = frem ir<%in>, ir<%in>
 ; SVE-ARMPL-TAILFOLD:  Cost of 10 for VF 4: WIDEN ir<%res> = frem ir<%in>, ir<%in>
 ; SVE-ARMPL-TAILFOLD:  Cost of Invalid for VF vscale x 1: WIDEN ir<%res> = frem ir<%in>, ir<%in>
@@ -120,7 +120,7 @@ define void @frem_f32(ptr noalias %in.ptr, ptr noalias %out.ptr) {
 ; SVE-ARMPL-TAILFOLD:  Cost of 10 for VF vscale x 4: WIDEN ir<%res> = frem ir<%in>, ir<%in>
 ;
 ; SVE-SLEEF-TAILFOLD-LABEL: 'frem_f32'
-; SVE-SLEEF-TAILFOLD:  LV: Found an estimated cost of 10 for VF 1 For instruction: %res = frem float %in, %in
+; SVE-SLEEF-TAILFOLD:  Cost of 10 for VF 1: EMIT ir<%res> = frem ir<%in>, ir<%in>
 ; SVE-SLEEF-TAILFOLD:  Cost of 24 for VF 2: WIDEN ir<%res> = frem ir<%in>, ir<%in>
 ; SVE-SLEEF-TAILFOLD:  Cost of 10 for VF 4: WIDEN ir<%res> = frem ir<%in>, ir<%in>
 ; SVE-SLEEF-TAILFOLD:  Cost of Invalid for VF vscale x 1: WIDEN ir<%res> = frem ir<%in>, ir<%in>
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/intrinsiccost.ll b/llvm/test/Transforms/LoopVectorize/AArch64/intrinsiccost.ll
index 07d6caa861f9da..958e7763aad3d6 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/intrinsiccost.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/intrinsiccost.ll
@@ -7,10 +7,6 @@ target datalayout = "e-m:e-i8:8:32-i16:16:32-i64:64-i128:128-n32:64-S128"
 target triple = "aarch64--linux-gnu"
 
 ; CHECK-COST-LABEL: sadd
-; CHECK-COST: Found an estimated cost of 6 for VF 1 For instruction:   %1 = tail call i16 @llvm.sadd.sat.i16(i16 %0, i16 %offset)
-; CHECK-COST: Cost of 4 for VF 2: WIDEN-INTRINSIC ir<%1> = call llvm.sadd.sat(ir<%0>, ir<%offset>)
-; CHECK-COST: Cost of 1 for VF 4: WIDEN-INTRINSIC ir<%1> = call llvm.sadd.sat(ir<%0>, ir<%offset>)
-; CHECK-COST: Cost of 1 for VF 8: WIDEN-INTRINSIC ir<%1> = call llvm.sadd.sat(ir<%0>, ir<%offset>)
 
 define void @saddsat(ptr nocapture readonly %pSrc, i16 signext %offset, ptr nocapture noalias %pDst, i32 %blockSize) #0 {
 ; CHECK-LABEL: @saddsat(
@@ -127,11 +123,6 @@ while.end:
 }
 
 ; CHECK-COST-LABEL: umin
-; CHECK-COST: Found an estimated cost of 2 for VF 1 For instruction:   %1 = tail call i8 @llvm.umin.i8(i8 %0, i8 %offset)
-; CHECK-COST: Cost of 3 for VF 2: WIDEN-INTRINSIC ir<%1> = call llvm.umin(ir<%0>, ir<%offset>)
-; CHECK-COST: Cost of 3 for VF 4: WIDEN-INTRINSIC ir<%1> = call llvm.umin(ir<%0>, ir<%offset>)
-; CHECK-COST: Cost of 1 for VF 8: WIDEN-INTRINSIC ir<%1> = call llvm.umin(ir<%0>, ir<%offset>)
-; CHECK-COST: Cost of 1 for VF 16: WIDEN-INTRINSIC ir<%1> = call llvm.umin(ir<%0>, ir<%offset>)
 
 define void @umin(ptr nocapture readonly %pSrc, i8 signext %offset, ptr nocapture noalias %pDst, i32 %blockSize) #0 {
 ; CHECK-LABEL: @umin(
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/multiple-result-intrinsics.ll b/llvm/test/Transforms/LoopVectorize/AArch64/multiple-result-intrinsics.ll
index 55994ad9a98f8d..2a7c73bfa8ed30 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/multiple-result-intrinsics.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/multiple-result-intrinsics.ll
@@ -6,7 +6,7 @@
 ; REQUIRES: asserts
 
 ; CHECK-COST-LABEL: sincos_f32
-; CHECK-COST: LV: Found an estimated cost of 10 for VF 1 For instruction:   %call = tail call { float, float } @llvm.sincos.f32(float %in_val)
+; CHECK-COST: Cost of 10 for VF 1: EMIT ir<%call> = call ir<%in_val>, ir<@llvm.sincos.f32>
 ; CHECK-COST: Cost of 26 for VF 2: WIDEN-INTRINSIC ir<%call> = call llvm.sincos(ir<%in_val>)
 ; CHECK-COST: Cost of 58 for VF 4: WIDEN-INTRINSIC ir<%call> = call llvm.sincos(ir<%in_val>)
 ; CHECK-COST: Cost of Invalid for VF vscale x 1: REPLICATE ir<%call> = call @llvm.sincos.f32(ir<%in_val>)
@@ -14,7 +14,7 @@
 ; CHECK-COST: Cost of Invalid for VF vscale x 4: REPLICATE ir<%call> = call @llvm.sincos.f32(ir<%in_val>)
 
 ; CHECK-COST-ARMPL-LABEL: sincos_f32
-; CHECK-COST-ARMPL: LV: Found an estimated cost of 10 for VF 1 For instruction:   %call = tail call { float, float } @llvm.sincos.f32(float %in_val)
+; CHECK-COST-ARMPL: Cost of 10 for VF 1: EMIT ir<%call> = call ir<%in_val>, ir<@llvm.sincos.f32>
 ; CHECK-COST-ARMPL: Cost of 26 for VF 2: WIDEN-INTRINSIC ir<%call> = call llvm.sincos(ir<%in_val>)
 ; CHECK-COST-ARMPL: Cost of 12 for VF 4: WIDEN-INTRINSIC ir<%call> = call llvm.sincos(ir<%in_val>)
 ; CHECK-COST-ARMPL: Cost of Invalid for VF vscale x 1: REPLICATE ir<%call> = call @llvm.sincos.f32(ir<%in_val>)
@@ -83,13 +83,13 @@ exit:
 }
 
 ; CHECK-COST-LABEL: sincos_f64
-; CHECK-COST: LV: Found an estimated cost of 10 for VF 1 For instruction:   %call = tail call { double, double } @llvm.sincos.f64(double %in_val)
+; CHECK-COST: Cost of 10 for VF 1: EMIT ir<%call> = call ir<%in_val>, ir<@llvm.sincos.f64>
 ; CHECK-COST: Cost of 26 for VF 2: WIDEN-INTRINSIC ir<%call> = call llvm.sincos(ir<%in_val>)
 ; CHECK-COST: Cost of Invalid for VF vscale x 1: REPLICATE ir<%call> = call @llvm.sincos.f64(ir<%in_val>)
 ; CHECK-COST: Cost of Invalid for VF vscale x 2: REPLICATE ir<%call> = call @llvm.sincos.f64(ir<%in_val>)
 
 ; CHECK-COST-ARMPL-LABEL: sincos_f64
-; CHECK-COST-ARMPL: LV: Found an estimated cost of 10 for VF 1 For instruction:   %call = tail call { double, double } @llvm.sincos.f64(double %in_val)
+; CHECK-COST-ARMPL: Cost of 10 for VF 1: EMIT ir<%call> = call ir<%in_val>, ir<@llvm.sincos.f64>
 ; CHECK-COST-ARMPL: Cost of 12 for VF 2: WIDEN-INTRINSIC ir<%call> = call llvm.sincos(ir<%in_val>)
 ; CHECK-COST-ARMPL: Cost of Invalid for VF vscale x 1: REPLICATE ir<%call> = call @llvm.sincos.f64(ir<%in_val>)
 ; CHECK-COST-ARMPL: Cost of 13 for VF vscale x 2: WIDEN-INTRINSIC ir<%call> = call llvm.sincos(ir<%in_val>)
@@ -156,7 +156,7 @@ exit:
 }
 
 ; CHECK-COST-LABEL: predicated_sincos
-; CHECK-COST: LV: Found an estimated cost of 10 for VF 1 For instruction:   %call = tail call { float, float } @llvm.sincos.f32(float %in_val)
+; CHECK-COST: Cost of 10 for VF 1: EMIT ir<%call> = call ir<%in_val>, ir<@llvm.sincos.f32>
 ; CHECK-COST: Cost of 26 for VF 2: WIDEN-INTRINSIC ir<%call> = call llvm.sincos(ir<%in_val>)
 ; CHECK-COST: Cost of 58 for VF 4: WIDEN-INTRINSIC ir<%call> = call llvm.sincos(ir<%in_val>)
 ; CHECK-COST: Cost of Invalid for VF vscale x 1: REPLICATE ir<%call> = call @llvm.sincos.f32(ir<%in_val>)
@@ -164,7 +164,7 @@ exit:
 ; CHECK-COST: Cost of Invalid for VF vscale x 4: REPLICATE ir<%call> = call @llvm.sincos.f32(ir<%in_val>)
 
 ; CHECK-COST-ARMPL-LABEL: predicated_sincos
-; CHECK-COST-ARMPL: LV: Found an estimated cost of 10 for VF 1 For instruction:   %call = tail call { float, float } @llvm.sincos.f32(float %in_val)
+; CHECK-COST-ARMPL: Cost of 10 for VF 1: EMIT ir<%call> = call ir<%in_val>, ir<@llvm.sincos.f32>
 ; CHECK-COST-ARMPL: Cost of 26 for VF 2: WIDEN-INTRINSIC ir<%call> = call llvm.sincos(ir<%in_val>)
 ; CHECK-COST-ARMPL: Cost of 12 for VF 4: WIDEN-INTRINSIC ir<%call> = call llvm.sincos(ir<%in_val>)
 ; CHECK-COST-ARMPL: Cost of Invalid for VF vscale x 1: REPLICATE ir<%call> = call @llvm.sincos.f32(ir<%in_val>)
@@ -228,7 +228,7 @@ for.end:
 }
 
 ; CHECK-COST-LABEL: modf_f32
-; CHECK-COST: LV: Found an estimated cost of 10 for VF 1 For instruction:   %call = tail call { float, float } @llvm.modf.f32(float %in_val)
+; CHECK-COST: Cost of 10 for VF 1: EMIT ir<%call> = call ir<%in_val>, ir<@llvm.modf.f32>
 ; CHECK-COST: Cost of 26 for VF 2: WIDEN-INTRINSIC ir<%call> = call llvm.modf(ir<%in_val>)
 ; CHECK-COST: Cost of 58 for VF 4: WIDEN-INTRINSIC ir<%call> = call llvm.modf(ir<%in_val>)
 ; CHECK-COST: Cost of Invalid for VF vscale x 1: REPLICATE ir<%call> = call @llvm.modf.f32(ir<%in_val>)
@@ -236,7 +236,7 @@ for.end:
 ; CHECK-COST: Cost of Invalid for VF vscale x 4: REPLICATE ir<%call> = call @llvm.modf.f32(ir<%in_val>)
 
 ; CHECK-COST-ARMPL-LABEL: modf_f32
-; CHECK-COST-ARMPL: LV: Found an estimated cost of 10 for VF 1 For instruction:   %call = tail call { float, float } @llvm.modf.f32(float %in_val)
+; CHECK-COST-ARMPL: Cost of 10 for VF 1: EMIT ir<%call> = call ir<%in_val>, ir<@llvm.modf.f32>
 ; CHECK-COST-ARMPL: Cost of 26 for VF 2: WIDEN-INTRINSIC ir<%call> = call llvm.modf(ir<%in_val>)
 ; CHECK-COST-ARMPL: Cost of 11 for VF 4: WIDEN-INTRINSIC ir<%call> = call llvm.modf(ir<%in_val>)
 ; CHECK-COST-ARMPL: Cost of Invalid for VF vscale x 1: REPLICATE ir<%call> = call @llvm.modf.f32(ir<%in_val>)
@@ -305,13 +305,13 @@ exit:
 }
 
 ; CHECK-COST-LABEL: modf_f64
-; CHECK-COST: LV: Found an estimated cost of 10 for VF 1 For instruction:   %call = tail call { double, double } @llvm.modf.f64(double %in_val)
+; CHECK-COST: Cost of 10 for VF 1: EMIT ir<%call> = call ir<%in_val>, ir<@llvm.modf.f64>
 ; CHECK-COST: Cost of 26 for VF 2: WIDEN-INTRINSIC ir<%call> = call llvm.modf(ir<%in_val>)
 ; CHECK-COST: Cost of Invalid for VF vscale x 1: REPLICATE ir<%call> = call @llvm.modf.f64(ir<%in_val>)
 ; CHECK-COST: Cost of Invalid for VF vscale x 2: REPLICATE ir<%call> = call @llvm.modf.f64(ir<%in_val>)
 
 ; CHECK-COST-ARMPL-LABEL: modf_f64
-; CHECK-COST-ARMPL: LV: Found an estimated cost of 10 for VF 1 For instruction:   %call = tail call { double, double } @llvm.modf.f64(double %in_val)
+; CHECK-COST-ARMPL: Cost of 10 for VF 1: EMIT ir<%call> = call ir<%in_val>, ir<@llvm.modf.f64>
 ; CHECK-COST-ARMPL: Cost of 11 for VF 2: WIDEN-INTRINSIC ir<%call> = call llvm.modf(ir<%in_val>)
 ; CHECK-COST-ARMPL: Cost of Invalid for VF vscale x 1: REPLICATE ir<%call> = call @llvm.modf.f64(ir<%in_val>)
 ; CHECK-COST-ARMPL: Cost of 12 for VF vscale x 2: WIDEN-INTRINSIC ir<%call> = call llvm.modf(ir<%in_val>)
@@ -378,7 +378,7 @@ exit:
 }
 
 ; CHECK-COST-LABEL: sincospi_f32
-; CHECK-COST: LV: Found an estimated cost of 10 for VF 1 For instruction:   %call = tail call { float, float } @llvm.sincospi.f32(float %in_val)
+; CHECK-COST: Cost of 10 for VF 1: EMIT ir<%call> = call ir<%in_val>, ir<@llvm.sincospi.f32>
 ; CHECK-COST: Cost of 26 for VF 2: WIDEN-INTRINSIC ir<%call> = call llvm.sincospi(ir<%in_val>)
 ; CHECK-COST: Cost of 58 for VF 4: WIDEN-INTRINSIC ir<%call> = call llvm.sincospi(ir<%in_val>)
 ; CHECK-COST: Cost of Invalid for VF vscale x 1: REPLICATE ir<%call> = call @llvm.sincospi.f32(ir<%in_val>)
@@ -386,7 +386,7 @@ exit:
 ; CHECK-COST: Cost of Invalid for VF vscale x 4: REPLICATE ir<%call> = call @llvm.sincospi.f32(ir<%in_val>)
 
 ; CHECK-COST-ARMPL-LABEL: sincospi_f32
-; CHECK-COST-ARMPL: LV: Found an estimated cost of 10 for VF 1 For instruction:   %call = tail call { float, float } @llvm.sincospi.f32(float %in_val)
+; CHECK-COST-ARMPL: Cost of 10 for VF 1: EMIT ir<%call> = call ir<%in_val>, ir<@llvm.sincospi.f32>
 ; CHECK-COST-ARMPL: Cost of 26 for VF 2: WIDEN-INTRINSIC ir<%call> = call llvm.sincospi(ir<%in_val>)
 ; CHECK-COST-ARMPL: Cost of 12 for VF 4: WIDEN-INTRINSIC ir<%call> = call llvm.sincospi(ir<%in_val>)
 ; CHECK-COST-ARMPL: Cost of Invalid for VF vscale x 1: REPLICATE ir<%call> = call @llvm.sincospi.f32(ir<%in_val>)
@@ -455,13 +455,13 @@ exit:
 }
 
 ; CHECK-COST-LABEL: sincospi_f64
-; CHECK-COST: LV: Found an estimated cost of 10 for VF 1 For instruction:   %call = tail call { double, double } @llvm.sincospi.f64(double %in_val)
+; CHECK-COST: Cost of 10 for VF 1: EMIT ir<%call> = call ir<%in_val>, ir<@llvm.sincospi.f64>
 ; CHECK-COST: Cost of 26 for VF 2: WIDEN-INTRINSIC ir<%call> = call llvm.sincospi(ir<%in_val>)
 ; CHECK-COST: Cost of Invalid for VF vscale x 1: REPLICATE ir<%call> = call @llvm.sincospi.f64(ir<%in_val>)
 ; CHECK-COST: Cost of Invalid for VF vscale x 2: REPLICATE ir<%call> = call @llvm.sincospi.f64(ir<%in_val>)
 
 ; CHECK-COST-ARMPL-LABEL: sincospi_f64
-; CHECK-COST-ARMPL: LV: Found an estimated cost of 10 for VF 1 For instruction:   %call = tail call { double, double } @llvm.sincospi.f64(double %in_val)
+; CHECK-COST-ARMPL: Cost of 10 for VF 1: EMIT ir<%call> = call ir<%in_val>, ir<@llvm.sincospi.f64>
 ; CHECK-COST-ARMPL: Cost of 12 for VF 2: WIDEN-INTRINSIC ir<%call> = call llvm.sincospi(ir<%in_val>)
 ; CHECK-COST-ARMPL: Cost of Invalid for VF vscale x 1: REPLICATE ir<%call> = call @llvm.sincospi.f64(ir<%in_val>)
 ; CHECK-COST-ARMPL: Cost of 13 for VF vscale x 2: WIDEN-INTRINSIC ir<%call> = call llvm.sincospi(ir<%in_val>)
@@ -528,7 +528,7 @@ exit:
 }
 
 ; CHECK-COST-LABEL: sadd_with_overflow_i32
-; CHECK-COST: LV: Found an estimated cost of 1 for VF 1 For instruction:   %call = tail call { i32, i1 } @llvm.sadd.with.overflow.i32(i32 %val_a, i32 %val_b)
+; CHECK-COST: Cost of 1 for VF 1: EMIT ir<%call> = call ir<%val_a>, ir<%val_b>, ir<@llvm.sadd.with.overflow.i32>
 ; CHECK-COST: Cost of 4 for VF 2: WIDEN-INTRINSIC ir<%call> = call llvm.sadd.with.overflow(ir<%val_a>, ir<%val_b>)
 ; CHECK-COST: Cost of 4 for VF 4: WIDEN-INTRINSIC ir<%call> = call llvm.sadd.with.overflow(ir<%val_a>, ir<%val_b>)
 ; CHECK-COST: Cost of 7 for VF 8: WIDEN-INTRINSIC ir<%call> = call llvm.sadd.with.overflow(ir<%val_a>, ir<%val_b>)
@@ -538,7 +538,7 @@ exit:
 ; CHECK-COST: Cost of 4 for VF vscale x 4: WIDEN-INTRINSIC ir<%call> = call llvm.sadd.with.overflow(ir<%val_a>, ir<%val_b>)
 
 ; CHECK-COST-ARMPL-LABEL: sadd_with_overflow_i32
-; CHECK-COST-ARMPL: LV: Found an estimated cost of 1 for VF 1 For instruction:   %call = tail call { i32, i1 } @llvm.sadd.with.overflow.i32(i32 %val_a, i32 %val_b)
+; CHECK-COST-ARMPL: Cost of 1 for VF 1: EMIT ir<%call> = call ir<%val_a>, ir<%val_b>, ir<@llvm.sadd.with.overflow.i32>
 ; CHECK-COST-ARMPL: Cost of 4 for VF 2: WIDEN-INTRINSIC ir<%call> = call llvm.sadd.with.overflow(ir<%val_a>, ir<%val_b>)
 ; CHECK-COST-ARMPL: Cost of 4 for VF 4: WIDEN-INTRINSIC ir<%call> = call llvm.sadd.with.overflow(ir<%val_a>, ir<%val_b>)
 ; CHECK-COST-ARMPL: Cost of 7 for VF 8: WIDEN-INTRINSIC ir<%call> = call llvm.sadd.with.overflow(ir<%val_a>, ir<%val_b>)
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/struct-return-cost.ll b/llvm/test/Transforms/LoopVectorize/AArch64/struct-return-cost.ll
index e2e69eb4ca147f..96cd2887c4bd4e 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/struct-return-cost.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/struct-return-cost.ll
@@ -7,9 +7,9 @@ target datalayout = "e-m:e-i8:8:32-i16:16:32-i64:64-i128:128-n32:64-S128"
 target triple = "aarch64--linux-gnu"
 
 ; CHECK-COST-LABEL: struct_return_widen
-; CHECK-COST: LV: Found an estimated cost of 10 for VF 1 For instruction:   %call = tail call { half, half } @foo(half %in_val)
-; CHECK-COST: LV: Found an estimated cost of 0 for VF 1 For instruction:   %extract_a = extractvalue { half, half } %call, 0
-; CHECK-COST: LV: Found an estimated cost of 0 for VF 1 For instruction:   %extract_b = extractvalue { half, half } %call, 1
+; CHECK-COST: Cost of 10 for VF 1: EMIT ir<%call> = call ir<%in_val>, ir<@foo>
+; CHECK-COST: Cost of 0 for VF 1: EMIT ir<%extract_a> = extractvalue ir<%call>
+; CHECK-COST: Cost of 0 for VF 1: EMIT ir<%extract_b> = extractvalue ir<%call>
 ;
 ; CHECK-COST: Cost of 10 for VF 2: WIDEN-CALL ir<%call> = call @foo(ir<%in_val>) (using library function: fixed_vec_foo)
 ; CHECK-COST: Cost of 0 for VF 2: WIDEN ir<%extract_a> = extractvalue ir<%call>, ir<0>
@@ -57,9 +57,9 @@ exit:
 }
 
 ; CHECK-COST-LABEL: struct_return_replicate
-; CHECK-COST: LV: Found an estimated cost of 10 for VF 1 For instruction:   %call = tail call { half, half } @foo(half %in_val)
-; CHECK-COST: LV: Found an estimated cost of 0 for VF 1 For instruction:   %extract_a = extractvalue { half, half } %call, 0
-; CHECK-COST: LV: Found an estimated cost of 0 for VF 1 For instruction:   %extract_b = extractvalue { half, half } %call, 1
+; CHECK-COST: Cost of 10 for VF 1: EMIT ir<%call> = call ir<%in_val>, ir<@foo>
+; CHECK-COST: Cost of 0 for VF 1: EMIT ir<%extract_a> = extractvalue ir<%call>
+; CHECK-COST: Cost of 0 for VF 1: EMIT ir<%extract_b> = extractvalue ir<%call>
 ;
 ; CHECK-COST: Cost of 26 for VF 2: REPLICATE ir<%call> = call @foo(ir<%in_val>)
 ; CHECK-COST: Cost of 0 for VF 2: WIDEN ir<%extract_a> = extractvalue ir<%call>, ir<0>
@@ -79,8 +79,8 @@ define void @struct_return_replicate(ptr noalias %in, ptr noalias writeonly %out
 ; CHECK:  [[ENTRY:.*:]]
 ; CHECK:  [[VECTOR_PH:.*:]]
 ; CHECK:  [[VECTOR_BODY:.*:]]
-; CHECK:    [[TMP3:%.*]] = tail call { half, half } @foo(half [[TMP1:%.*]]) #[[ATTR2:[0-9]+]]
-; CHECK:    [[TMP4:%.*]] = tail call { half, half } @foo(half [[TMP2:%.*]]) #[[ATTR2]]
+; CHECK:    [[TMP2:%.*]] = tail call { half, half } @foo(half [[TMP1:%.*]]) #[[ATTR2:[0-9]+]]
+; CHECK:    [[TMP4:%.*]] = tail call { half, half } @foo(half [[TMP3:%.*]]) #[[ATTR2]]
 ; CHECK:  [[MIDDLE_BLOCK:.*:]]
 ; CHECK:  [[EXIT:.*:]]
 ;
@@ -108,9 +108,9 @@ exit:
 }
 
 ; CHECK-COST-LABEL: struct_return_scalable
-; CHECK-COST: LV: Found an estimated cost of 10 for VF 1 For instruction:   %call = tail call { half, half } @foo(half %in_val)
-; CHECK-COST: LV: Found an estimated cost of 0 for VF 1 For instruction:   %extract_a = extractvalue { half, half } %call, 0
-; CHECK-COST: LV: Found an estimated cost of 0 for VF 1 For instruction:   %extract_b = extractvalue { half, half } %call, 1
+; CHECK-COST: Cost of 10 for VF 1: EMIT ir<%call> = call ir<%in_val>, ir<@foo>
+; CHECK-COST: Cost of 0 for VF 1: EMIT ir<%extract_a> = extractvalue ir<%call>
+; CHECK-COST: Cost of 0 for VF 1: EMIT ir<%extract_b> = extractvalue ir<%call>
 ;
 ; CHECK-COST: Cost of 26 for VF 2: REPLICATE ir<%call> = call @foo(ir<%in_val>)
 ; CHECK-COST: Cost of 0 for VF 2: WIDEN ir<%extract_a> = extractvalue ir<%call>, ir<0>
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/type-shrinkage-zext-costs.ll b/llvm/test/Transforms/LoopVectorize/AArch64/type-shrinkage-zext-costs.ll
index 09c262b1adb482..11ec71fea57f9e 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/type-shrinkage-zext-costs.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/type-shrinkage-zext-costs.ll
@@ -8,7 +8,7 @@ target triple = "aarch64-unknown-linux-gnu"
 
 define void @zext_i8_i16(ptr noalias nocapture readonly %p, ptr noalias nocapture %q, i32 %len) #0 {
 ; CHECK-COST-LABEL: LV: Checking a loop in 'zext_i8_i16'
-; CHECK-COST: LV: Found an estimated cost of 0 for VF 1 For instruction:   %conv = zext i8 %0 to i32
+; CHECK-COST: Cost of 0 for VF 1: EMIT-SCALAR ir<%conv> = zext ir<%0> to i32
 ; CHECK-COST: Cost of 1 for VF 2: WIDEN-CAST ir<%conv> = zext ir<%0> to i16
 ; CHECK-COST: Cost of 1 for VF 4: WIDEN-CAST ir<%conv> = zext ir<%0> to i16
 ; CHECK-COST: Cost of 1 for VF 8: WIDEN-CAST ir<%conv> = zext ir<%0> to i16
diff --git a/llvm/test/Transforms/LoopVectorize/ARM/mve-icmpcost.ll b/llvm/test/Transforms/LoopVectorize/ARM/mve-icmpcost.ll
index 7369ef688f5e25..aa6ec753d92764 100644
--- a/llvm/test/Transforms/LoopVectorize/ARM/mve-icmpcost.ll
+++ b/llvm/test/Transforms/LoopVectorize/ARM/mve-icmpcost.ll
@@ -7,29 +7,28 @@ target triple = "thumbv8.1m.main-arm-none-eabi"
 
 define void @expensive_icmp(ptr noalias nocapture %d, ptr nocapture readonly %s, i32 %n, i16 zeroext %m) #0 {
 ; CHECK-LABEL: 'expensive_icmp'
-; CHECK:  LV: Found an estimated cost of 0 for VF 1 For instruction: %i.016 = phi i32 [ 0, %for.body.lr.ph ], [ %inc, %for.inc ]
-; CHECK:  LV: Found an estimated cost of 0 for VF 1 For instruction: %arrayidx = getelementptr inbounds i16, ptr %s, i32 %i.016
-; CHECK:  LV: Found an estimated cost of 1 for VF 1 For instruction: %1 = load i16, ptr %arrayidx, align 2
-; CHECK:  LV: Found an estimated cost of 0 for VF 1 For instruction: %conv = sext i16 %1 to i32
-; CHECK:  LV: Found an estimated cost of 1 for VF 1 For instruction: %cmp2 = icmp sgt i32 %conv, %conv1
-; CHECK:  LV: Found an estimated cost of 0 for VF 1 For instruction: br i1 %cmp2, label %if.then, label %for.inc
-; CHECK:  LV: Found an estimated cost of 1 for VF 1 For instruction: %conv6 = add i16 %1, %0
-; CHECK:  LV: Found an estimated cost of 0 for VF 1 For instruction: %arrayidx7 = getelementptr inbounds i16, ptr %d, i32 %i.016
-; CHECK:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i16 %conv6, ptr %arrayidx7, align 2
-; CHECK:  LV: Found an estimated cost of 0 for VF 1 For instruction: br label %for.inc
-; CHECK:  LV: Found an estimated cost of 1 for VF 1 For instruction: %inc = add nuw nsw i32 %i.016, 1
-; CHECK:  LV: Found an estimated cost of 1 for VF 1 For instruction: %exitcond.not = icmp eq i32 %inc, %n
-; CHECK:  LV: Found an estimated cost of 0 for VF 1 For instruction: br i1 %exitcond.not, label %for.cond.cleanup.loopexit, label %for.body
+; CHECK:  Cost of 0 for VF 1: EMIT-SCALAR ir<%i.016> = phi [ ir<0>, vector.ph ], [ ir<%inc>, for.inc ]
+; CHECK:  Cost of 0 for VF 1: EMIT ir<%arrayidx> = getelementptr inbounds ir<%s>, ir<%i.016>
+; CHECK:  Cost of 1 for VF 1: EMIT-SCALAR ir<%1> = load ir<%arrayidx>
+; CHECK:  Cost of 0 for VF 1: EMIT-SCALAR ir<%conv> = sext ir<%1> to i32
+; CHECK:  Cost of 1 for VF 1: EMIT ir<%cmp2> = icmp sgt ir<%conv>, ir<%conv1>
+; CHECK:  Cost of 0 for VF 1: EMIT branch-on-cond ir<%cmp2> (!vplan.prof.estimated estimated {1073741824, 1073741824})
+; CHECK:  Cost of 1 for VF 1: EMIT ir<%conv6> = add ir<%1>, ir<%0> (!vplan.execution.frequency 4611686018427387904 (50%, estimated))
+; CHECK:  Cost of 0 for VF 1: EMIT ir<%arrayidx7> = getelementptr inbounds ir<%d>, ir<%i.016> (!vplan.execution.frequency 4611686018427387904 (50%, estimated))
+; CHECK:  Cost of 1 for VF 1: EMIT store ir<%conv6>, ir<%arrayidx7> (!vplan.execution.frequency 4611686018427387904 (50%, estimated))
+; CHECK:  Cost of 1 for VF 1: EMIT ir<%inc> = add nuw nsw ir<%i.016>, ir<1>
+; CHECK:  Cost of 1 for VF 1: EMIT ir<%exitcond.not> = icmp eq ir<%inc>, ir<%n>
+; CHECK:  Cost of 0 for VF 1: EMIT branch-on-cond ir<%exitcond.not>
 ; CHECK:  Cost of 0 for VF 2: vp<[[VP4:%[0-9]+]]> = SCALAR-STEPS vp<[[VP3:%[0-9]+]]>, ir<1>, vp<[[VP0:%[0-9]+]]>
 ; CHECK:  Cost of 0 for VF 2: CLONE ir<%arrayidx> = getelementptr inbounds ir<%s>, vp<[[VP4]]>
 ; CHECK:  Cost of 0 for VF 2: vp<[[VP5:%[0-9]+]]> = vector-pointer inbounds i16, ir<%arrayidx>, ir<1>
 ; CHECK:  Cost of 18 for VF 2: WIDEN ir<%1> = load vp<[[VP5]]>
 ; CHECK:  Cost of 4 for VF 2: WIDEN-CAST ir<%conv> = sext ir<%1> to i32
 ; CHECK:  Cost of 20 for VF 2: WIDEN ir<%cmp2> = icmp sgt ir<%conv>, ir<%conv1>
-; CHECK:  Cost of 26 for VF 2: WIDEN ir<%conv6> = add ir<%1>, ir<%0>
+; CHECK:  Cost of 26 for VF 2: WIDEN ir<%conv6> = add ir<%1>, ir<%0> (!vplan.execution.frequency 4611686018427387904 (50%, estimated))
 ; CHECK:  Cost of 0 for VF 2: CLONE ir<%arrayidx7> = getelementptr ir<%d>, vp<[[VP4]]>
 ; CHECK:  Cost of 0 for VF 2: vp<[[VP6:%[0-9]+]]> = vector-pointer i16, ir<%arrayidx7>, ir<1>
-; CHECK:  Cost of 16 for VF 2: WIDEN store vp<[[VP6]]>, ir<%conv6>, ir<%cmp2>
+; CHECK:  Cost of 16 for VF 2: WIDEN store vp<[[VP6]]>, ir<%conv6>, ir<%cmp2> (!vplan.execution.frequency 4611686018427387904 (50%, estimated))
 ; CHECK:  Cost of 0 for VF 2: EMIT vp<%index.next> = add nuw vp<[[VP3]]>, vp<[[VP1:%[0-9]+]]>
 ; CHECK:  Cost of 1 for VF 2: EMIT branch-on-count vp<%index.next>, vp<[[VP2:%[0-9]+]]>
 ; CHECK:  Cost of 0 for VF 2: vector loop backedge
@@ -51,10 +50,10 @@ define void @expensive_icmp(ptr noalias nocapture %d, ptr nocapture readonly %s,
 ; CHECK:  Cost of 2 for VF 4: WIDEN ir<%1> = load vp<[[VP5]]>
 ; CHECK:  Cost of 0 for VF 4: WIDEN-CAST ir<%conv> = sext ir<%1> to i32
 ; CHECK:  Cost of 2 for VF 4: WIDEN ir<%cmp2> = icmp sgt ir<%conv>, ir<%conv1>
-; CHECK:  Cost of 2 for VF 4: WIDEN ir<%conv6> = add ir<%1>, ir<%0>
+; CHECK:  Cost of 2 for VF 4: WIDEN ir<%conv6> = add ir<%1>, ir<%0> (!vplan.execution.frequency 4611686018427387904 (50%, estimated))
 ; CHECK:  Cost of 0 for VF 4: CLONE ir<%arrayidx7> = getelementptr ir<%d>, vp<[[VP4]]>
 ; CHECK:  Cost of 0 for VF 4: vp<[[VP6]]> = vector-pointer i16, ir<%arrayidx7>, ir<1>
-; CHECK:  Cost of 2 for VF 4: WIDEN store vp<[[VP6]]>, ir<%conv6>, ir<%cmp2>
+; CHECK:  Cost of 2 for VF 4: WIDEN store vp<[[VP6]]>, ir<%conv6>, ir<%cmp2> (!vplan.execution.frequency 4611686018427387904 (50%, estimated))
 ; CHECK:  Cost of 0 for VF 4: EMIT vp<%index.next> = add nuw vp<[[VP3]]>, vp<[[VP1]]>
 ; CHECK:  Cost of 1 for VF 4: EMIT branch-on-count vp<%index.next>, vp<[[VP2]]>
 ; CHECK:  Cost of 0 for VF 4: vector loop backedge
@@ -76,10 +75,10 @@ define void @expensive_icmp(ptr noalias nocapture %d, ptr nocapture readonly %s,
 ; CHECK:  Cost of 2 for VF 8: WIDEN ir<%1> = load vp<[[VP5]]>
 ; CHECK:  Cost of 2 for VF 8: WIDEN-CAST ir<%conv> = sext ir<%1> to i32
 ; CHECK:  Cost of 36 for VF 8: WIDEN ir<%cmp2> = icmp sgt ir<%conv>, ir<%conv1>
-; CHECK:  Cost of 2 for VF 8: WIDEN ir<%conv6> = add ir<%1>, ir<%0>
+; CHECK:  Cost of 2 for VF 8: WIDEN ir<%conv6> = add ir<%1>, ir<%0> (!vplan.execution.frequency 4611686018427387904 (50%, estimated))
 ; CHECK:  Cost of 0 for VF 8: CLONE ir<%arrayidx7> = getelementptr ir<%d>, vp<[[VP4]]>
 ; CHECK:  Cost of 0 for VF 8: vp<[[VP6]]> = vector-pointer i16, ir<%arrayidx7>, ir<1>
-; CHECK:  Cost of 2 for VF 8: WIDEN store vp<[[VP6]]>, ir<%conv6>, ir<%cmp2>
+; CHECK:  Cost of 2 for VF 8: WIDEN store vp<[[VP6]]>, ir<%conv6>, ir<%cmp2> (!vplan.execution.frequency 4611686018427387904 (50%, estimated))
 ; CHECK:  Cost of 0 for VF 8: EMIT vp<%index.next> = add nuw vp<[[VP3]]>, vp<[[VP1]]>
 ; CHECK:  Cost of 1 for VF 8: EMIT branch-on-count vp<%index.next>, vp<[[VP2]]>
 ; CHECK:  Cost of 0 for VF 8: vector loop backedge
@@ -133,26 +132,26 @@ for.inc:
 
 define void @cheap_icmp(ptr nocapture readonly %pSrcA, ptr nocapture readonly %pSrcB, ptr nocapture %pDst, i32 %blockSize) #0 {
 ; CHECK-LABEL: 'cheap_icmp'
-; CHECK:  LV: Found an estimated cost of 0 for VF 1 For instruction: %blkCnt.012 = phi i32 [ %dec, %while.body ], [ %blockSize, %while.body.preheader ]
-; CHECK:  LV: Found an estimated cost of 0 for VF 1 For instruction: %pSrcA.addr.011 = phi ptr [ %incdec.ptr, %while.body ], [ %pSrcA, %while.body.preheader ]
-; CHECK:  LV: Found an estimated cost of 0 for VF 1 For instruction: %pDst.addr.010 = phi ptr [ %incdec.ptr5, %while.body ], [ %pDst, %while.body.preheader ]
-; CHECK:  LV: Found an estimated cost of 0 for VF 1 For instruction: %pSrcB.addr.09 = phi ptr [ %incdec.ptr2, %while.body ], [ %pSrcB, %while.body.preheader ]
-; CHECK:  LV: Found an estimated cost of 0 for VF 1 For instruction: %incdec.ptr = getelementptr inbounds i8, ptr %pSrcA.addr.011, i32 1
-; CHECK:  LV: Found an estimated cost of 1 for VF 1 For instruction: %0 = load i8, ptr %pSrcA.addr.011, align 1
-; CHECK:  LV: Found an estimated cost of 0 for VF 1 For instruction: %conv1 = sext i8 %0 to i32
-; CHECK:  LV: Found an estimated cost of 0 for VF 1 For instruction: %incdec.ptr2 = getelementptr inbounds i8, ptr %pSrcB.addr.09, i32 1
-; CHECK:  LV: Found an estimated cost of 1 for VF 1 For instruction: %1 = load i8, ptr %pSrcB.addr.09, align 1
-; CHECK:  LV: Found an estimated cost of 0 for VF 1 For instruction: %conv3 = sext i8 %1 to i32
-; CHECK:  LV: Found an estimated cost of 1 for VF 1 For instruction: %mul = mul nsw i32 %conv3, %conv1
-; CHECK:  LV: Found an estimated cost of 1 for VF 1 For instruction: %shr = ashr i32 %mul, 7
-; CHECK:  LV: Found an estimated cost of 1 for VF 1 For instruction: %2 = icmp slt i32 %shr, 127
-; CHECK:  LV: Found an estimated cost of 1 for VF 1 For instruction: %spec.select.i = select i1 %2, i32 %shr, i32 127
-; CHECK:  LV: Found an estimated cost of 0 for VF 1 For instruction: %conv4 = trunc i32 %spec.select.i to i8
-; CHECK:  LV: Found an estimated cost of 0 for VF 1 For instruction: %incdec.ptr5 = getelementptr inbounds i8, ptr %pDst.addr.010, i32 1
-; CHECK:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i8 %conv4, ptr %pDst.addr.010, align 1
-; CHECK:  LV: Found an estimated cost of 1 for VF 1 For instruction: %dec = add i32 %blkCnt.012, -1
-; CHECK:  LV: Found an estimated cost of 1 for VF 1 For instruction: %cmp.not = icmp eq i32 %dec, 0
-; CHECK:  LV: Found an estimated cost of 0 for VF 1 For instruction: br i1 %cmp.not, label %while.end.loopexit, label %while.body
+; CHECK:  Cost of 0 for VF 1: EMIT-SCALAR ir<%blkCnt.012> = phi [ ir<%blockSize>, vector.ph ], [ ir<%dec>, while.body ]
+; CHECK:  Cost of 0 for VF 1: EMIT-SCALAR ir<%pSrcA.addr.011> = phi [ ir<%pSrcA>, vector.ph ], [ ir<%incdec.ptr>, while.body ]
+; CHECK:  Cost of 0 for VF 1: EMIT-SCALAR ir<%pDst.addr.010> = phi [ ir<%pDst>, vector.ph ], [ ir<%incdec.ptr5>, while.body ]
+; CHECK:  Cost of 0 for VF 1: EMIT-SCALAR ir<%pSrcB.addr.09> = phi [ ir<%pSrcB>, vector.ph ], [ ir<%incdec.ptr2>, while.body ]
+; CHECK:  Cost of 0 for VF 1: EMIT ir<%incdec.ptr> = getelementptr inbounds ir<%pSrcA.addr.011>, ir<1>
+; CHECK:  Cost of 1 for VF 1: EMIT-SCALAR ir<%0> = load ir<%pSrcA.addr.011>
+; CHECK:  Cost of 0 for VF 1: EMIT-SCALAR ir<%conv1> = sext ir<%0> to i32
+; CHECK:  Cost of 0 for VF 1: EMIT ir<%incdec.ptr2> = getelementptr inbounds ir<%pSrcB.addr.09>, ir<1>
+; CHECK:  Cost of 1 for VF 1: EMIT-SCALAR ir<%1> = load ir<%pSrcB.addr.09>
+; CHECK:  Cost of 0 for VF 1: EMIT-SCALAR ir<%conv3> = sext ir<%1> to i32
+; CHECK:  Cost of 1 for VF 1: EMIT ir<%mul> = mul nsw ir<%conv3>, ir<%conv1>
+; CHECK:  Cost of 1 for VF 1: EMIT ir<%shr> = ashr ir<%mul>, ir<7>
+; CHECK:  Cost of 1 for VF 1: EMIT ir<%2> = icmp slt ir<%shr>, ir<127>
+; CHECK:  Cost of 1 for VF 1: EMIT ir<%spec.select.i> = select ir<%2>, ir<%shr>, ir<127>
+; CHECK:  Cost of 0 for VF 1: EMIT-SCALAR ir<%conv4> = trunc ir<%spec.select.i> to i8
+; CHECK:  Cost of 0 for VF 1: EMIT ir<%incdec.ptr5> = getelementptr inbounds ir<%pDst.addr.010>, ir<1>
+; CHECK:  Cost of 1 for VF 1: EMIT store ir<%conv4>, ir<%pDst.addr.010>
+; CHECK:  Cost of 1 for VF 1: EMIT ir<%dec> = add ir<%blkCnt.012>, ir<-1>
+; CHECK:  Cost of 1 for VF 1: EMIT ir<%cmp.not> = icmp eq ir<%dec>, ir<0>
+; CHECK:  Cost of 0 for VF 1: EMIT branch-on-cond ir<%cmp.not>
 ; CHECK:  Cost of 0 for VF 2: vp<[[VP8:%[0-9]+]]> = SCALAR-STEPS vp<[[VP7:%[0-9]+]]>, ir<1>, vp<[[VP0:%[0-9]+]]>
 ; CHECK:  Cost of 0 for VF 2: EMIT vp<%next.gep> = ptradd ir<%pSrcA>, vp<[[VP8]]>
 ; CHECK:  Cost of 0 for VF 2: vp<[[VP9:%[0-9]+]]> = SCALAR-STEPS vp<[[VP7]]>, ir<1>, vp<[[VP0]]>
@@ -407,19 +406,19 @@ while.end:
 
 define void @floatcmp(ptr nocapture readonly %pSrc, ptr nocapture %pDst, i32 %blockSize) #0 {
 ; CHECK-LABEL: 'floatcmp'
-; CHECK:  LV: Found an estimated cost of 0 for VF 1 For instruction: %pSrc.addr.010 = phi ptr [ %incdec.ptr2, %while.body ], [ %pSrc, %while.body.preheader ]
-; CHECK:  LV: Found an estimated cost of 0 for VF 1 For instruction: %blockSize.addr.09 = phi i32 [ %dec, %while.body ], [ %blockSize, %while.body.preheader ]
-; CHECK:  LV: Found an estimated cost of 0 for VF 1 For instruction: %pDst.addr.08 = phi ptr [ %incdec.ptr, %while.body ], [ %pDst, %while.body.preheader ]
-; CHECK:  LV: Found an estimated cost of 1 for VF 1 For instruction: %0 = load float, ptr %pSrc.addr.010, align 4
-; CHECK:  LV: Found an estimated cost of 1 for VF 1 For instruction: %cmp1 = fcmp nnan ninf nsz olt float %0, 0.000000e+00
-; CHECK:  LV: Found an estimated cost of 1 for VF 1 For instruction: %cond = select nnan ninf nsz i1 %cmp1, float 1.000000e+01, float %0
-; CHECK:  LV: Found an estimated cost of 1 for VF 1 For instruction: %conv = fptosi float %cond to i32
-; CHECK:  LV: Found an estimated cost of 0 for VF 1 For instruction: %incdec.ptr = getelementptr inbounds i32, ptr %pDst.addr.08, i32 1
-; CHECK:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i32 %conv, ptr %pDst.addr.08, align 4
-; CHECK:  LV: Found an estimated cost of 0 for VF 1 For instruction: %incdec.ptr2 = getelementptr inbounds float, ptr %pSrc.addr.010, i32 1
-; CHECK:  LV: Found an estimated cost of 1 for VF 1 For instruction: %dec = add i32 %blockSize.addr.09, -1
-; CHECK:  LV: Found an estimated cost of 1 for VF 1 For instruction: %cmp.not = icmp eq i32 %dec, 0
-; CHECK:  LV: Found an estimated cost of 0 for VF 1 For instruction: br i1 %cmp.not, label %while.end.loopexit, label %while.body
+; CHECK:  Cost of 0 for VF 1: EMIT-SCALAR ir<%pSrc.addr.010> = phi [ ir<%pSrc>, vector.ph ], [ ir<%incdec.ptr2>, while.body ]
+; CHECK:  Cost of 0 for VF 1: EMIT-SCALAR ir<%blockSize.addr.09> = phi [ ir<%blockSize>, vector.ph ], [ ir<%dec>, while.body ]
+; CHECK:  Cost of 0 for VF 1: EMIT-SCALAR ir<%pDst.addr.08> = phi [ ir<%pDst>, vector.ph ], [ ir<%incdec.ptr>, while.body ]
+; CHECK:  Cost of 1 for VF 1: EMIT-SCALAR ir<%0> = load ir<%pSrc.addr.010>
+; CHECK:  Cost of 1 for VF 1: EMIT ir<%cmp1> = fcmp olt nnan ninf nsz ir<%0>, ir<0.000000e+00>
+; CHECK:  Cost of 1 for VF 1: EMIT ir<%cond> = select nnan ninf nsz ir<%cmp1>, ir<1.000000e+01>, ir<%0>
+; CHECK:  Cost of 1 for VF 1: EMIT-SCALAR ir<%conv> = fptosi ir<%cond> to i32
+; CHECK:  Cost of 0 for VF 1: EMIT ir<%incdec.ptr> = getelementptr inbounds ir<%pDst.addr.08>, ir<1>
+; CHECK:  Cost of 1 for VF 1: EMIT store ir<%conv>, ir<%pDst.addr.08>
+; CHECK:  Cost of 0 for VF 1: EMIT ir<%incdec.ptr2> = getelementptr inbounds ir<%pSrc.addr.010>, ir<1>
+; CHECK:  Cost of 1 for VF 1: EMIT ir<%dec> = add ir<%blockSize.addr.09>, ir<-1>
+; CHECK:  Cost of 1 for VF 1: EMIT ir<%cmp.not> = icmp eq ir<%dec>, ir<0>
+; CHECK:  Cost of 0 for VF 1: EMIT branch-on-cond ir<%cmp.not>
 ; CHECK:  Cost of 1 for VF 2: vp<[[VP7:%[0-9]+]]> = DERIVED-IV ir<0> + vp<[[VP6:%[0-9]+]]> * ir<4>
 ; CHECK:  Cost of 0 for VF 2: vp<[[VP8:%[0-9]+]]> = SCALAR-STEPS vp<[[VP7]]>, ir<4>, vp<[[VP0:%[0-9]+]]>
 ; CHECK:  Cost of 0 for VF 2: EMIT vp<%next.gep> = ptradd ir<%pSrc>, vp<[[VP8]]>
diff --git a/llvm/test/Transforms/LoopVectorize/ARM/mve-selectandorcost.ll b/llvm/test/Transforms/LoopVectorize/ARM/mve-selectandorcost.ll
index a814515225dc81..03b4d9a7a72c0a 100644
--- a/llvm/test/Transforms/LoopVectorize/ARM/mve-selectandorcost.ll
+++ b/llvm/test/Transforms/LoopVectorize/ARM/mve-selectandorcost.ll
@@ -7,7 +7,7 @@ target datalayout = "e-m:e-p:32:32-Fi8-i64:64-v128:64:128-a:0:32-n32-S64"
 target triple = "thumbv8.1m.main-arm-none-eabi"
 
 ; CHECK-COST-LABEL: test
-; CHECK-COST: LV: Found an estimated cost of 1 for VF 1 For instruction:   %or.cond = select i1 %cmp2, i1 true, i1 %cmp3
+; CHECK-COST: Cost of 1 for VF 1: EMIT ir<%or.cond> = select ir<%cmp2>, ir<true>, ir<%cmp3>
 ; CHECK-COST: Cost of 26 for VF 2: WIDEN ir<%or.cond> = select ir<%cmp2>, ir<true>, ir<%cmp3>
 ; CHECK-COST: Cost of 2 for VF 4: WIDEN ir<%or.cond> = select ir<%cmp2>, ir<true>, ir<%cmp3>
 
diff --git a/llvm/test/Transforms/LoopVectorize/ARM/mve-shiftcost.ll b/llvm/test/Transforms/LoopVectorize/ARM/mve-shiftcost.ll
index 22913b4dae0db1..663f54600d3f47 100644
--- a/llvm/test/Transforms/LoopVectorize/ARM/mve-shiftcost.ll
+++ b/llvm/test/Transforms/LoopVectorize/ARM/mve-shiftcost.ll
@@ -6,8 +6,8 @@ target datalayout = "e-m:e-p:32:32-Fi8-i64:64-v128:64:128-a:0:32-n32-S64"
 target triple = "thumbv8.1m.main-none-none-eabi"
 
 ; CHECK-LABEL: test
-; CHECK-COST: LV: Found an estimated cost of 0 for VF 1 For instruction:   %and515 = shl i32 %l41, 3
-; CHECK-COST: LV: Found an estimated cost of 1 for VF 1 For instruction:   %l45 = and i32 %and515, 131072
+; CHECK-COST: Cost of 0 for VF 1: EMIT ir<%and515> = shl ir<%l41>, ir<3>
+; CHECK-COST: Cost of 1 for VF 1: EMIT ir<%l45> = and ir<%and515>, ir<131072>
 ; CHECK-COST: Cost of 2 for VF 4: WIDEN ir<%and515> = shl ir<%l41>, ir<3>
 ; CHECK-COST: Cost of 2 for VF 4: WIDEN ir<%l45> = and ir<%and515>, ir<131072>
 ; CHECK-NOT: vector.body
diff --git a/llvm/test/Transforms/LoopVectorize/ARM/scalar-block-cost.ll b/llvm/test/Transforms/LoopVectorize/ARM/scalar-block-cost.ll
index 8ebb04dd671dca..929503d556059e 100644
--- a/llvm/test/Transforms/LoopVectorize/ARM/scalar-block-cost.ll
+++ b/llvm/test/Transforms/LoopVectorize/ARM/scalar-block-cost.ll
@@ -6,16 +6,17 @@ target triple = "thumbv8.1m.main-none-none-eabi"
 
 define void @pred_loop(ptr %off, ptr %data, ptr %dst, i32 %n) #0 {
 
-; CHECK-COST: LV: Found an estimated cost of 0 for VF 1 For instruction:   %i.09 = phi i32 [ %add, %for.body ], [ 0, %for.body.preheader ]
-; CHECK-COST-NEXT: LV: Found an estimated cost of 1 for VF 1 For instruction:   %add = add nuw nsw i32 %i.09, 1
-; CHECK-COST-NEXT: LV: Found an estimated cost of 0 for VF 1 For instruction:   %arrayidx = getelementptr inbounds i32, ptr %data, i32 %add
-; CHECK-COST-NEXT: LV: Found an estimated cost of 1 for VF 1 For instruction:   %0 = load i32, ptr %arrayidx, align 4
-; CHECK-COST-NEXT: LV: Found an estimated cost of 1 for VF 1 For instruction:   %add1 = add nsw i32 %0, 5
-; CHECK-COST-NEXT: LV: Found an estimated cost of 0 for VF 1 For instruction:   %arrayidx2 = getelementptr inbounds i32, ptr %dst, i32 %i.09
-; CHECK-COST-NEXT: LV: Found an estimated cost of 1 for VF 1 For instruction:   store i32 %add1, ptr %arrayidx2, align 4
-; CHECK-COST-NEXT: LV: Found an estimated cost of 1 for VF 1 For instruction:   %exitcond.not = icmp eq i32 %add, %n
-; CHECK-COST-NEXT: LV: Found an estimated cost of 0 for VF 1 For instruction:   br i1 %exitcond.not, label %exit.loopexit, label %for.body
-; CHECK-COST-NEXT: LV: Scalar loop costs: 5.
+; CHECK-COST-LABEL: LV: Checking a loop in 'pred_loop'
+; CHECK-COST:      Cost of 0 for VF 1: EMIT-SCALAR ir<%i.09> = phi [ ir<0>, vector.ph ], [ ir<%add>, for.body ]
+; CHECK-COST-NEXT: Cost of 1 for VF 1: EMIT ir<%add> = add nuw nsw ir<%i.09>, ir<1>
+; CHECK-COST-NEXT: Cost of 0 for VF 1: EMIT ir<%arrayidx> = getelementptr inbounds ir<%data>, ir<%add>
+; CHECK-COST-NEXT: Cost of 1 for VF 1: EMIT-SCALAR ir<%0> = load ir<%arrayidx>
+; CHECK-COST-NEXT: Cost of 1 for VF 1: EMIT ir<%add1> = add nsw ir<%0>, ir<5>
+; CHECK-COST-NEXT: Cost of 0 for VF 1: EMIT ir<%arrayidx2> = getelementptr inbounds ir<%dst>, ir<%i.09>
+; CHECK-COST-NEXT: Cost of 1 for VF 1: EMIT store ir<%add1>, ir<%arrayidx2>
+; CHECK-COST-NEXT: Cost of 1 for VF 1: EMIT ir<%exitcond.not> = icmp eq ir<%add>, ir<%n>
+; CHECK-COST-NEXT: Cost of 0 for VF 1: EMIT branch-on-cond ir<%exitcond.not>
+; CHECK-COST:      LV: Scalar loop costs: 5.
 
 entry:
   %cmp8 = icmp sgt i32 %n, 0
@@ -38,26 +39,26 @@ for.body:
 
 define void @if_convert(ptr %a, ptr %b, i32 %start, i32 %end) #0 {
 
-; CHECK-COST-2: LV: Found an estimated cost of 0 for VF 1 For instruction:   %i.032 = phi i32 [ %inc, %if.end ], [ %start, %for.body.preheader ]
-; CHECK-COST-2-NEXT: LV: Found an estimated cost of 0 for VF 1 For instruction:   %arrayidx = getelementptr inbounds i32, ptr %a, i32 %i.032
-; CHECK-COST-2-NEXT: LV: Found an estimated cost of 1 for VF 1 For instruction:   %0 = load i32, ptr %arrayidx, align 4
-; CHECK-COST-2-NEXT: LV: Found an estimated cost of 0 for VF 1 For instruction:   %arrayidx2 = getelementptr inbounds i32, ptr %b, i32 %i.032
-; CHECK-COST-2-NEXT: LV: Found an estimated cost of 1 for VF 1 For instruction:   %1 = load i32, ptr %arrayidx2, align 4
-; CHECK-COST-2-NEXT: LV: Found an estimated cost of 1 for VF 1 For instruction:   %cmp3 = icmp sgt i32 %0, %1
-; CHECK-COST-2-NEXT: LV: Found an estimated cost of 0 for VF 1 For instruction:   br i1 %cmp3, label %if.then, label %if.end
-; CHECK-COST-2-NEXT: LV: Found an estimated cost of 1 for VF 1 For instruction:   %mul = mul nsw i32 %0, 5
-; CHECK-COST-2-NEXT: LV: Found an estimated cost of 1 for VF 1 For instruction:   %add = add nsw i32 %mul, 3
-; CHECK-COST-2-NEXT: LV: Found an estimated cost of 0 for VF 1 For instruction:   %factor = shl i32 %add, 1
-; CHECK-COST-2-NEXT: LV: Found an estimated cost of 1 for VF 1 For instruction:   %sub = sub i32 %0, %1
-; CHECK-COST-2-NEXT: LV: Found an estimated cost of 1 for VF 1 For instruction:   %add7 = add i32 %sub, %factor
-; CHECK-COST-2-NEXT: LV: Found an estimated cost of 1 for VF 1 For instruction:   store i32 %add7, ptr %arrayidx2, align 4
-; CHECK-COST-2-NEXT: LV: Found an estimated cost of 0 for VF 1 For instruction:   br label %if.end
-; CHECK-COST-2-NEXT: LV: Found an estimated cost of 0 for VF 1 For instruction:   %k.0 = phi i32 [ %add, %if.then ], [ %0, %for.body ]
-; CHECK-COST-2-NEXT: LV: Found an estimated cost of 1 for VF 1 For instruction:   store i32 %k.0, ptr %arrayidx, align 4
-; CHECK-COST-2-NEXT: LV: Found an estimated cost of 1 for VF 1 For instruction:   %inc = add nsw i32 %i.032, 1
-; CHECK-COST-2-NEXT: LV: Found an estimated cost of 1 for VF 1 For instruction:   %exitcond.not = icmp eq i32 %inc, %end
-; CHECK-COST-2-NEXT: LV: Found an estimated cost of 0 for VF 1 For instruction:   br i1 %exitcond.not, label %for.cond.cleanup.loopexit, label %for.body
-; CHECK-COST-2-NEXT: LV: Scalar loop costs: 8.5.
+; CHECK-COST-2-LABEL: LV: Checking a loop in 'if_convert'
+; CHECK-COST-2:      Cost of 0 for VF 1: EMIT-SCALAR ir<%i.032> = phi [ ir<%start>, vector.ph ], [ ir<%inc>, if.end ]
+; CHECK-COST-2-NEXT: Cost of 0 for VF 1: EMIT ir<%arrayidx> = getelementptr inbounds ir<%a>, ir<%i.032>
+; CHECK-COST-2-NEXT: Cost of 1 for VF 1: EMIT-SCALAR ir<%0> = load ir<%arrayidx>
+; CHECK-COST-2-NEXT: Cost of 0 for VF 1: EMIT ir<%arrayidx2> = getelementptr inbounds ir<%b>, ir<%i.032>
+; CHECK-COST-2-NEXT: Cost of 1 for VF 1: EMIT-SCALAR ir<%1> = load ir<%arrayidx2>
+; CHECK-COST-2-NEXT: Cost of 1 for VF 1: EMIT ir<%cmp3> = icmp sgt ir<%0>, ir<%1>
+; CHECK-COST-2-NEXT: Cost of 0 for VF 1: EMIT branch-on-cond ir<%cmp3>
+; CHECK-COST-2-NEXT: Cost of 1 for VF 1: EMIT ir<%mul> = mul nsw ir<%0>, ir<5>
+; CHECK-COST-2-NEXT: Cost of 1 for VF 1: EMIT ir<%add> = add nsw ir<%mul>, ir<3>
+; CHECK-COST-2-NEXT: Cost of 0 for VF 1: EMIT ir<%factor> = shl ir<%add>, ir<1>
+; CHECK-COST-2-NEXT: Cost of 1 for VF 1: EMIT ir<%sub> = sub ir<%0>, ir<%1>
+; CHECK-COST-2-NEXT: Cost of 1 for VF 1: EMIT ir<%add7> = add ir<%sub>, ir<%factor>
+; CHECK-COST-2-NEXT: Cost of 1 for VF 1: EMIT store ir<%add7>, ir<%arrayidx2>
+; CHECK-COST-2-NEXT: Cost of 0 for VF 1: EMIT-SCALAR ir<%k.0> = phi [ ir<%add>, if.then ], [ ir<%0>, for.body ]
+; CHECK-COST-2-NEXT: Cost of 1 for VF 1: EMIT store ir<%k.0>, ir<%arrayidx>
+; CHECK-COST-2-NEXT: Cost of 1 for VF 1: EMIT ir<%inc> = add nsw ir<%i.032>, ir<1>
+; CHECK-COST-2-NEXT: Cost of 1 for VF 1: EMIT ir<%exitcond.not> = icmp eq ir<%inc>, ir<%end>
+; CHECK-COST-2-NEXT: Cost of 0 for VF 1: EMIT branch-on-cond ir<%exitcond.not>
+; CHECK-COST-2:      LV: Scalar loop costs: 8.5.
 
 entry:
   %cmp31 = icmp slt i32 %start, %end
diff --git a/llvm/test/Transforms/LoopVectorize/SystemZ/pr47665.ll b/llvm/test/Transforms/LoopVectorize/SystemZ/pr47665.ll
index 20a2604b2a9bab..392a5c545bf844 100644
--- a/llvm/test/Transforms/LoopVectorize/SystemZ/pr47665.ll
+++ b/llvm/test/Transforms/LoopVectorize/SystemZ/pr47665.ll
@@ -4,52 +4,24 @@
 define void @test(ptr noalias %p, ptr noalias %q, i40 %a) {
 ; CHECK-LABEL: define void @test(
 ; CHECK-SAME: ptr noalias [[P:%.*]], ptr noalias [[Q:%.*]], i40 [[A:%.*]]) #[[ATTR0:[0-9]+]] {
-; CHECK-NEXT:  [[ENTRY:.*:]]
-; CHECK-NEXT:    br label %[[FOR_BODY:.*]]
-; CHECK:       [[FOR_BODY]]:
+; CHECK-NEXT:  [[FOR_BODY:.*]]:
 ; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
 ; CHECK:       [[VECTOR_BODY]]:
-; CHECK-NEXT:    [[IV:%.*]] = phi i32 [ 0, %[[FOR_BODY]] ], [ [[INDEX_NEXT:%.*]], %[[PRED_STORE_CONTINUE6:.*]] ]
-; CHECK-NEXT:    [[VEC_IND:%.*]] = phi <4 x i8> [ <i8 0, i8 1, i8 2, i8 3>, %[[FOR_BODY]] ], [ [[VEC_IND_NEXT:%.*]], %[[PRED_STORE_CONTINUE6]] ]
-; CHECK-NEXT:    [[TMP0:%.*]] = icmp ule <4 x i8> [[VEC_IND]], splat (i8 9)
-; CHECK-NEXT:    store i1 false, ptr [[P]], align 1
-; CHECK-NEXT:    [[TMP1:%.*]] = extractelement <4 x i1> [[TMP0]], i64 0
-; CHECK-NEXT:    br i1 [[TMP1]], label %[[PRED_STORE_IF:.*]], label %[[PRED_STORE_CONTINUE:.*]]
-; CHECK:       [[PRED_STORE_IF]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i32 [ 0, %[[FOR_BODY]] ], [ [[IV_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[SHL:%.*]] = shl i40 [[A]], 24
+; CHECK-NEXT:    [[ASHR:%.*]] = ashr i40 [[SHL]], 28
+; CHECK-NEXT:    [[TRUNC:%.*]] = trunc i40 [[ASHR]] to i32
+; CHECK-NEXT:    [[ICMP_EQ:%.*]] = icmp eq i32 [[TRUNC]], 0
+; CHECK-NEXT:    [[ZEXT:%.*]] = zext i1 [[ICMP_EQ]] to i32
+; CHECK-NEXT:    [[ICMP_ULT:%.*]] = icmp ult i32 0, [[ZEXT]]
+; CHECK-NEXT:    [[OR:%.*]] = or i1 [[ICMP_ULT]], true
+; CHECK-NEXT:    [[ICMP_SGT:%.*]] = icmp sgt i1 [[OR]], false
+; CHECK-NEXT:    store i1 [[ICMP_SGT]], ptr [[P]], align 1
 ; CHECK-NEXT:    [[GEP:%.*]] = getelementptr inbounds i8, ptr [[Q]], i32 [[IV]]
 ; CHECK-NEXT:    store i8 0, ptr [[GEP]], align 1
-; CHECK-NEXT:    br label %[[PRED_STORE_CONTINUE]]
-; CHECK:       [[PRED_STORE_CONTINUE]]:
-; CHECK-NEXT:    [[TMP3:%.*]] = extractelement <4 x i1> [[TMP0]], i64 1
-; CHECK-NEXT:    br i1 [[TMP3]], label %[[PRED_STORE_IF1:.*]], label %[[PRED_STORE_CONTINUE2:.*]]
-; CHECK:       [[PRED_STORE_IF1]]:
-; CHECK-NEXT:    [[IV_NEXT:%.*]] = add i32 [[IV]], 1
-; CHECK-NEXT:    [[TMP5:%.*]] = getelementptr inbounds i8, ptr [[Q]], i32 [[IV_NEXT]]
-; CHECK-NEXT:    store i8 0, ptr [[TMP5]], align 1
-; CHECK-NEXT:    br label %[[PRED_STORE_CONTINUE2]]
-; CHECK:       [[PRED_STORE_CONTINUE2]]:
-; CHECK-NEXT:    [[TMP6:%.*]] = extractelement <4 x i1> [[TMP0]], i64 2
-; CHECK-NEXT:    br i1 [[TMP6]], label %[[EXIT:.*]], label %[[PRED_STORE_CONTINUE4:.*]]
-; CHECK:       [[EXIT]]:
-; CHECK-NEXT:    [[TMP7:%.*]] = add i32 [[IV]], 2
-; CHECK-NEXT:    [[TMP8:%.*]] = getelementptr inbounds i8, ptr [[Q]], i32 [[TMP7]]
-; CHECK-NEXT:    store i8 0, ptr [[TMP8]], align 1
-; CHECK-NEXT:    br label %[[PRED_STORE_CONTINUE4]]
-; CHECK:       [[PRED_STORE_CONTINUE4]]:
-; CHECK-NEXT:    [[TMP9:%.*]] = extractelement <4 x i1> [[TMP0]], i64 3
-; CHECK-NEXT:    br i1 [[TMP9]], label %[[PRED_STORE_IF5:.*]], label %[[PRED_STORE_CONTINUE6]]
-; CHECK:       [[PRED_STORE_IF5]]:
-; CHECK-NEXT:    [[TMP10:%.*]] = add i32 [[IV]], 3
-; CHECK-NEXT:    [[TMP11:%.*]] = getelementptr inbounds i8, ptr [[Q]], i32 [[TMP10]]
-; CHECK-NEXT:    store i8 0, ptr [[TMP11]], align 1
-; CHECK-NEXT:    br label %[[PRED_STORE_CONTINUE6]]
-; CHECK:       [[PRED_STORE_CONTINUE6]]:
-; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i32 [[IV]], 4
-; CHECK-NEXT:    [[VEC_IND_NEXT]] = add nuw <4 x i8> [[VEC_IND]], splat (i8 4)
-; CHECK-NEXT:    [[TMP12:%.*]] = icmp eq i32 [[INDEX_NEXT]], 12
-; CHECK-NEXT:    br i1 [[TMP12]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
-; CHECK:       [[MIDDLE_BLOCK]]:
-; CHECK-NEXT:    br label %[[EXIT1:.*]]
+; CHECK-NEXT:    [[IV_NEXT]] = add i32 [[IV]], 1
+; CHECK-NEXT:    [[COND:%.*]] = icmp ult i32 [[IV_NEXT]], 10
+; CHECK-NEXT:    br i1 [[COND]], label %[[VECTOR_BODY]], label %[[EXIT1:.*]]
 ; CHECK:       [[EXIT1]]:
 ; CHECK-NEXT:    ret void
 ;
diff --git a/llvm/test/Transforms/LoopVectorize/WebAssembly/int-mac-reduction-costs.ll b/llvm/test/Transforms/LoopVectorize/WebAssembly/int-mac-reduction-costs.ll
index 6580a3dacc21c2..8e12d55c928873 100644
--- a/llvm/test/Transforms/LoopVectorize/WebAssembly/int-mac-reduction-costs.ll
+++ b/llvm/test/Transforms/LoopVectorize/WebAssembly/int-mac-reduction-costs.ll
@@ -5,11 +5,11 @@ target triple = "wasm32"
 
 define hidden i32 @i32_mac_s8(ptr nocapture noundef readonly %a, ptr nocapture noundef readonly %b, i32 noundef %N) {
 ; CHECK-LABEL: 'i32_mac_s8'
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction:   %0 = load i8, ptr %arrayidx, align 1
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction:   %conv = sext i8 %0 to i32
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction:   %1 = load i8, ptr %arrayidx1, align 1
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction:   %conv2 = sext i8 %1 to i32
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction:   %mul = mul nsw i32 %conv2, %conv
+; CHECK: Cost of 1 for VF 1: EMIT-SCALAR ir<%0> = load
+; CHECK: Cost of 0 for VF 1: EMIT-SCALAR ir<%conv> = sext ir<%0> to i32
+; CHECK: Cost of 1 for VF 1: EMIT-SCALAR ir<%1> = load
+; CHECK: Cost of 0 for VF 1: EMIT-SCALAR ir<%conv2> = sext ir<%1> to i32
+; CHECK: Cost of 1 for VF 1: EMIT ir<%mul> = mul nsw ir<%conv2>, ir<%conv>
 
 ; CHECK: Cost of 3 for VF 2: WIDEN ir<%0> = load
 ; CHECK: Cost of 0 for VF 2: WIDEN-CAST ir<%conv> = sext ir<%0> to i32
@@ -49,11 +49,11 @@ for.body:
 
 define hidden i32 @i32_mac_s16(ptr nocapture noundef readonly %a, ptr nocapture noundef readonly %b, i32 noundef %N) {
 ; CHECK-LABEL: 'i32_mac_s16'
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction:   %0 = load i16, ptr %arrayidx, align 2
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction:   %conv = sext i16 %0 to i32
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction:   %1 = load i16, ptr %arrayidx1, align 2
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction:   %conv2 = sext i16 %1 to i32
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction:   %mul = mul nsw i32 %conv2, %conv
+; CHECK: Cost of 1 for VF 1: EMIT-SCALAR ir<%0> = load
+; CHECK: Cost of 0 for VF 1: EMIT-SCALAR ir<%conv> = sext ir<%0> to i32
+; CHECK: Cost of 1 for VF 1: EMIT-SCALAR ir<%1> = load
+; CHECK: Cost of 0 for VF 1: EMIT-SCALAR ir<%conv2> = sext ir<%1> to i32
+; CHECK: Cost of 1 for VF 1: EMIT ir<%mul> = mul nsw ir<%conv2>, ir<%conv>
 
 ; CHECK: Cost of 2 for VF 2: WIDEN ir<%0> = load
 ; CHECK: Cost of 0 for VF 2: WIDEN-CAST ir<%conv> = sext ir<%0> to i32
@@ -93,11 +93,11 @@ for.body:
 
 define hidden i64 @i64_mac_s16(ptr nocapture noundef readonly %a, ptr nocapture noundef readonly %b, i32 noundef %N) {
 ; CHECK-LABEL: 'i64_mac_s16'
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction:   %0 = load i16, ptr %arrayidx, align 2
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction:   %conv = sext i16 %0 to i64
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction:   %1 = load i16, ptr %arrayidx1, align 2
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction:   %conv2 = sext i16 %1 to i64
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction:   %mul = mul nsw i64 %conv2, %conv
+; CHECK: Cost of 1 for VF 1: EMIT-SCALAR ir<%0> = load
+; CHECK: Cost of 0 for VF 1: EMIT-SCALAR ir<%conv> = sext ir<%0> to i64
+; CHECK: Cost of 1 for VF 1: EMIT-SCALAR ir<%1> = load
+; CHECK: Cost of 0 for VF 1: EMIT-SCALAR ir<%conv2> = sext ir<%1> to i64
+; CHECK: Cost of 1 for VF 1: EMIT ir<%mul> = mul nsw ir<%conv2>, ir<%conv>
 
 ; CHECK: Cost of 2 for VF 2: WIDEN ir<%0> = load
 ; CHECK: Cost of 1 for VF 2: WIDEN-CAST ir<%conv> = sext ir<%0> to i64
@@ -131,10 +131,10 @@ for.body:
 
 define hidden i64 @i64_mac_s32(ptr nocapture noundef readonly %a, ptr nocapture noundef readonly %b, i32 noundef %N) {
 ; CHECK-LABEL: 'i64_mac_s32'
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction:   %0 = load i32, ptr %arrayidx, align 4
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction:   %1 = load i32, ptr %arrayidx1, align 4
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction:   %mul = mul i32 %1, %0
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction:   %conv = sext i32 %mul to i64
+; CHECK: Cost of 1 for VF 1: EMIT-SCALAR ir<%0> = load
+; CHECK: Cost of 1 for VF 1: EMIT-SCALAR ir<%1> = load
+; CHECK: Cost of 1 for VF 1: EMIT ir<%mul> = mul ir<%1>, ir<%0>
+; CHECK: Cost of 1 for VF 1: EMIT-SCALAR ir<%conv> = sext ir<%mul> to i64
 
 ; CHECK: Cost of 2 for VF 2: WIDEN ir<%0> = load
 ; CHECK: Cost of 2 for VF 2: WIDEN ir<%1> = load
@@ -166,11 +166,11 @@ for.body:
 
 define hidden i32 @i32_mac_u8(ptr nocapture noundef readonly %a, ptr nocapture noundef readonly %b, i32 noundef %N) {
 ; CHECK-LABEL: 'i32_mac_u8'
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction:   %0 = load i8, ptr %arrayidx, align 1
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction:   %conv = zext i8 %0 to i32
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction:   %1 = load i8, ptr %arrayidx1, align 1
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction:   %conv2 = zext i8 %1 to i32
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction:   %mul = mul nuw nsw i32 %conv2, %conv
+; CHECK: Cost of 1 for VF 1: EMIT-SCALAR ir<%0> = load
+; CHECK: Cost of 0 for VF 1: EMIT-SCALAR ir<%conv> = zext ir<%0> to i32
+; CHECK: Cost of 1 for VF 1: EMIT-SCALAR ir<%1> = load
+; CHECK: Cost of 0 for VF 1: EMIT-SCALAR ir<%conv2> = zext ir<%1> to i32
+; CHECK: Cost of 1 for VF 1: EMIT ir<%mul> = mul nuw nsw ir<%conv2>, ir<%conv>
 
 ; CHECK: Cost of 3 for VF 2: WIDEN ir<%0> = load
 ; CHECK: Cost of 0 for VF 2: WIDEN-CAST ir<%conv> = zext ir<%0> to i32
@@ -210,11 +210,11 @@ for.body:
 
 define hidden i32 @i32_mac_u16(ptr nocapture noundef readonly %a, ptr nocapture noundef readonly %b, i32 noundef %N) {
 ; CHECK-LABEL: 'i32_mac_u16'
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction:   %0 = load i16, ptr %arrayidx, align 2
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction:   %conv = zext i16 %0 to i32
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction:   %1 = load i16, ptr %arrayidx1, align 2
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction:   %conv2 = zext i16 %1 to i32
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction:   %mul = mul nuw nsw i32 %conv2, %conv
+; CHECK: Cost of 1 for VF 1: EMIT-SCALAR ir<%0> = load
+; CHECK: Cost of 0 for VF 1: EMIT-SCALAR ir<%conv> = zext ir<%0> to i32
+; CHECK: Cost of 1 for VF 1: EMIT-SCALAR ir<%1> = load
+; CHECK: Cost of 0 for VF 1: EMIT-SCALAR ir<%conv2> = zext ir<%1> to i32
+; CHECK: Cost of 1 for VF 1: EMIT ir<%mul> = mul nuw nsw ir<%conv2>, ir<%conv>
 
 ; CHECK: Cost of 2 for VF 2: WIDEN ir<%0> = load
 ; CHECK: Cost of 0 for VF 2: WIDEN-CAST ir<%conv> = zext ir<%0> to i32
@@ -254,11 +254,11 @@ for.body:
 
 define hidden i64 @i64_mac_u16(ptr nocapture noundef readonly %a, ptr nocapture noundef readonly %b, i32 noundef %N) {
 ; CHECK-LABEL: 'i64_mac_u16'
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction:   %0 = load i16, ptr %arrayidx, align 2
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction:   %conv = zext i16 %0 to i64
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction:   %1 = load i16, ptr %arrayidx1, align 2
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction:   %conv2 = zext i16 %1 to i64
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction:   %mul = mul nuw nsw i64 %conv2, %conv
+; CHECK: Cost of 1 for VF 1: EMIT-SCALAR ir<%0> = load
+; CHECK: Cost of 0 for VF 1: EMIT-SCALAR ir<%conv> = zext ir<%0> to i64
+; CHECK: Cost of 1 for VF 1: EMIT-SCALAR ir<%1> = load
+; CHECK: Cost of 0 for VF 1: EMIT-SCALAR ir<%conv2> = zext ir<%1> to i64
+; CHECK: Cost of 1 for VF 1: EMIT ir<%mul> = mul nuw nsw ir<%conv2>, ir<%conv>
 
 ; CHECK: Cost of 2 for VF 2: WIDEN ir<%0> = load
 ; CHECK: Cost of 1 for VF 2: WIDEN-CAST ir<%conv> = zext ir<%0> to i64
@@ -292,10 +292,10 @@ for.body:
 
 define hidden i64 @i64_mac_u32(ptr nocapture noundef readonly %a, ptr nocapture noundef readonly %b, i32 noundef %N) {
 ; CHECK-LABEL: 'i64_mac_u32'
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction:   %0 = load i32, ptr %arrayidx, align 4
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction:   %1 = load i32, ptr %arrayidx1, align 4
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction:   %mul = mul i32 %1, %0
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction:   %conv = zext i32 %mul to i64
+; CHECK: Cost of 1 for VF 1: EMIT-SCALAR ir<%0> = load
+; CHECK: Cost of 1 for VF 1: EMIT-SCALAR ir<%1> = load
+; CHECK: Cost of 1 for VF 1: EMIT ir<%mul> = mul ir<%1>, ir<%0>
+; CHECK: Cost of 1 for VF 1: EMIT-SCALAR ir<%conv> = zext ir<%mul> to i64
 
 ; CHECK: Cost of 2 for VF 2: WIDEN ir<%0> = load
 ; CHECK: Cost of 2 for VF 2: WIDEN ir<%1> = load
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i16-with-i8-index.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i16-with-i8-index.ll
index 1ac8f11262064c..2770dd4801aebc 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i16-with-i8-index.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i16-with-i8-index.ll
@@ -17,14 +17,14 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE-LABEL: 'test'
-; SSE:  LV: Found an estimated cost of 1 for VF 1 For instruction: %valB = load i16, ptr %inB, align 2
+; SSE:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB> = load ir<%inB>
 ; SSE:  Cost of 24 for VF 2: REPLICATE ir<%valB> = load ir<%inB>
 ; SSE:  Cost of 48 for VF 4: REPLICATE ir<%valB> = load ir<%inB>
 ; SSE:  Cost of 96 for VF 8: REPLICATE ir<%valB> = load ir<%inB>
 ; SSE:  Cost of 192 for VF 16: REPLICATE ir<%valB> = load ir<%inB>
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: %valB = load i16, ptr %inB, align 2
+; AVX1:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB> = load ir<%inB>
 ; AVX1:  Cost of 24 for VF 2: REPLICATE ir<%valB> = load ir<%inB>
 ; AVX1:  Cost of 48 for VF 4: REPLICATE ir<%valB> = load ir<%inB>
 ; AVX1:  Cost of 96 for VF 8: REPLICATE ir<%valB> = load ir<%inB>
@@ -32,7 +32,7 @@ define void @test() {
 ; AVX1:  Cost of 386 for VF 32: REPLICATE ir<%valB> = load ir<%inB>
 ;
 ; AVX2-SLOWGATHER-LABEL: 'test'
-; AVX2-SLOWGATHER:  LV: Found an estimated cost of 1 for VF 1 For instruction: %valB = load i16, ptr %inB, align 2
+; AVX2-SLOWGATHER:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB> = load ir<%inB>
 ; AVX2-SLOWGATHER:  Cost of 4 for VF 2: REPLICATE ir<%valB> = load ir<%inB>
 ; AVX2-SLOWGATHER:  Cost of 8 for VF 4: REPLICATE ir<%valB> = load ir<%inB>
 ; AVX2-SLOWGATHER:  Cost of 16 for VF 8: REPLICATE ir<%valB> = load ir<%inB>
@@ -40,7 +40,7 @@ define void @test() {
 ; AVX2-SLOWGATHER:  Cost of 66 for VF 32: REPLICATE ir<%valB> = load ir<%inB>
 ;
 ; AVX2-FASTGATHER-LABEL: 'test'
-; AVX2-FASTGATHER:  LV: Found an estimated cost of 1 for VF 1 For instruction: %valB = load i16, ptr %inB, align 2
+; AVX2-FASTGATHER:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB> = load ir<%inB>
 ; AVX2-FASTGATHER:  Cost of 6 for VF 2: REPLICATE ir<%valB> = load ir<%inB>
 ; AVX2-FASTGATHER:  Cost of 13 for VF 4: REPLICATE ir<%valB> = load ir<%inB>
 ; AVX2-FASTGATHER:  Cost of 26 for VF 8: REPLICATE ir<%valB> = load ir<%inB>
@@ -48,7 +48,7 @@ define void @test() {
 ; AVX2-FASTGATHER:  Cost of 106 for VF 32: REPLICATE ir<%valB> = load ir<%inB>
 ;
 ; AVX512-LABEL: 'test'
-; AVX512:  LV: Found an estimated cost of 1 for VF 1 For instruction: %valB = load i16, ptr %inB, align 2
+; AVX512:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB> = load ir<%inB>
 ; AVX512:  Cost of 6 for VF 2: REPLICATE ir<%valB> = load ir<%inB>
 ; AVX512:  Cost of 13 for VF 4: REPLICATE ir<%valB> = load ir<%inB>
 ; AVX512:  Cost of 27 for VF 8: REPLICATE ir<%valB> = load ir<%inB>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i32-with-i8-index.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i32-with-i8-index.ll
index 2d5a30019bacd6..58b96771aedbea 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i32-with-i8-index.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i32-with-i8-index.ll
@@ -17,21 +17,21 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
-; SSE2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %valB = load i32, ptr %inB, align 4
+; SSE2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB> = load ir<%inB>
 ; SSE2:  Cost of 25 for VF 2: REPLICATE ir<%valB> = load ir<%inB>
 ; SSE2:  Cost of 51 for VF 4: REPLICATE ir<%valB> = load ir<%inB>
 ; SSE2:  Cost of 102 for VF 8: REPLICATE ir<%valB> = load ir<%inB>
 ; SSE2:  Cost of 204 for VF 16: REPLICATE ir<%valB> = load ir<%inB>
 ;
 ; SSE42-LABEL: 'test'
-; SSE42:  LV: Found an estimated cost of 1 for VF 1 For instruction: %valB = load i32, ptr %inB, align 4
+; SSE42:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB> = load ir<%inB>
 ; SSE42:  Cost of 24 for VF 2: REPLICATE ir<%valB> = load ir<%inB>
 ; SSE42:  Cost of 48 for VF 4: REPLICATE ir<%valB> = load ir<%inB>
 ; SSE42:  Cost of 96 for VF 8: REPLICATE ir<%valB> = load ir<%inB>
 ; SSE42:  Cost of 192 for VF 16: REPLICATE ir<%valB> = load ir<%inB>
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: %valB = load i32, ptr %inB, align 4
+; AVX1:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB> = load ir<%inB>
 ; AVX1:  Cost of 24 for VF 2: REPLICATE ir<%valB> = load ir<%inB>
 ; AVX1:  Cost of 48 for VF 4: REPLICATE ir<%valB> = load ir<%inB>
 ; AVX1:  Cost of 97 for VF 8: REPLICATE ir<%valB> = load ir<%inB>
@@ -39,7 +39,7 @@ define void @test() {
 ; AVX1:  Cost of 388 for VF 32: REPLICATE ir<%valB> = load ir<%inB>
 ;
 ; AVX2-SLOWGATHER-LABEL: 'test'
-; AVX2-SLOWGATHER:  LV: Found an estimated cost of 1 for VF 1 For instruction: %valB = load i32, ptr %inB, align 4
+; AVX2-SLOWGATHER:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB> = load ir<%inB>
 ; AVX2-SLOWGATHER:  Cost of 4 for VF 2: REPLICATE ir<%valB> = load ir<%inB>
 ; AVX2-SLOWGATHER:  Cost of 8 for VF 4: REPLICATE ir<%valB> = load ir<%inB>
 ; AVX2-SLOWGATHER:  Cost of 17 for VF 8: REPLICATE ir<%valB> = load ir<%inB>
@@ -47,7 +47,7 @@ define void @test() {
 ; AVX2-SLOWGATHER:  Cost of 68 for VF 32: REPLICATE ir<%valB> = load ir<%inB>
 ;
 ; AVX2-FASTGATHER-LABEL: 'test'
-; AVX2-FASTGATHER:  LV: Found an estimated cost of 1 for VF 1 For instruction: %valB = load i32, ptr %inB, align 4
+; AVX2-FASTGATHER:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB> = load ir<%inB>
 ; AVX2-FASTGATHER:  Cost of 4 for VF 2: WIDEN ir<%valB> = load ir<%inB>
 ; AVX2-FASTGATHER:  Cost of 6 for VF 4: WIDEN ir<%valB> = load ir<%inB>
 ; AVX2-FASTGATHER:  Cost of 12 for VF 8: WIDEN ir<%valB> = load ir<%inB>
@@ -55,7 +55,7 @@ define void @test() {
 ; AVX2-FASTGATHER:  Cost of 48 for VF 32: WIDEN ir<%valB> = load ir<%inB>
 ;
 ; AVX512-LABEL: 'test'
-; AVX512:  LV: Found an estimated cost of 1 for VF 1 For instruction: %valB = load i32, ptr %inB, align 4
+; AVX512:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB> = load ir<%inB>
 ; AVX512:  Cost of 6 for VF 2: REPLICATE ir<%valB> = load ir<%inB>
 ; AVX512:  Cost of 13 for VF 4: REPLICATE ir<%valB> = load ir<%inB>
 ; AVX512:  Cost of 10 for VF 8: WIDEN ir<%valB> = load ir<%inB>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i64-with-i8-index.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i64-with-i8-index.ll
index ce5828a46eea74..7876551023ee67 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i64-with-i8-index.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i64-with-i8-index.ll
@@ -17,21 +17,21 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
-; SSE2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %valB = load i64, ptr %inB, align 8
+; SSE2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB> = load ir<%inB>
 ; SSE2:  Cost of 25 for VF 2: REPLICATE ir<%valB> = load ir<%inB>
 ; SSE2:  Cost of 50 for VF 4: REPLICATE ir<%valB> = load ir<%inB>
 ; SSE2:  Cost of 100 for VF 8: REPLICATE ir<%valB> = load ir<%inB>
 ; SSE2:  Cost of 200 for VF 16: REPLICATE ir<%valB> = load ir<%inB>
 ;
 ; SSE42-LABEL: 'test'
-; SSE42:  LV: Found an estimated cost of 1 for VF 1 For instruction: %valB = load i64, ptr %inB, align 8
+; SSE42:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB> = load ir<%inB>
 ; SSE42:  Cost of 24 for VF 2: REPLICATE ir<%valB> = load ir<%inB>
 ; SSE42:  Cost of 48 for VF 4: REPLICATE ir<%valB> = load ir<%inB>
 ; SSE42:  Cost of 96 for VF 8: REPLICATE ir<%valB> = load ir<%inB>
 ; SSE42:  Cost of 192 for VF 16: REPLICATE ir<%valB> = load ir<%inB>
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: %valB = load i64, ptr %inB, align 8
+; AVX1:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB> = load ir<%inB>
 ; AVX1:  Cost of 24 for VF 2: REPLICATE ir<%valB> = load ir<%inB>
 ; AVX1:  Cost of 49 for VF 4: REPLICATE ir<%valB> = load ir<%inB>
 ; AVX1:  Cost of 98 for VF 8: REPLICATE ir<%valB> = load ir<%inB>
@@ -39,7 +39,7 @@ define void @test() {
 ; AVX1:  Cost of 392 for VF 32: REPLICATE ir<%valB> = load ir<%inB>
 ;
 ; AVX2-SLOWGATHER-LABEL: 'test'
-; AVX2-SLOWGATHER:  LV: Found an estimated cost of 1 for VF 1 For instruction: %valB = load i64, ptr %inB, align 8
+; AVX2-SLOWGATHER:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB> = load ir<%inB>
 ; AVX2-SLOWGATHER:  Cost of 4 for VF 2: REPLICATE ir<%valB> = load ir<%inB>
 ; AVX2-SLOWGATHER:  Cost of 9 for VF 4: REPLICATE ir<%valB> = load ir<%inB>
 ; AVX2-SLOWGATHER:  Cost of 18 for VF 8: REPLICATE ir<%valB> = load ir<%inB>
@@ -47,7 +47,7 @@ define void @test() {
 ; AVX2-SLOWGATHER:  Cost of 72 for VF 32: REPLICATE ir<%valB> = load ir<%inB>
 ;
 ; AVX2-FASTGATHER-LABEL: 'test'
-; AVX2-FASTGATHER:  LV: Found an estimated cost of 1 for VF 1 For instruction: %valB = load i64, ptr %inB, align 8
+; AVX2-FASTGATHER:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB> = load ir<%inB>
 ; AVX2-FASTGATHER:  Cost of 4 for VF 2: WIDEN ir<%valB> = load ir<%inB>
 ; AVX2-FASTGATHER:  Cost of 6 for VF 4: WIDEN ir<%valB> = load ir<%inB>
 ; AVX2-FASTGATHER:  Cost of 12 for VF 8: WIDEN ir<%valB> = load ir<%inB>
@@ -55,7 +55,7 @@ define void @test() {
 ; AVX2-FASTGATHER:  Cost of 48 for VF 32: WIDEN ir<%valB> = load ir<%inB>
 ;
 ; AVX512-LABEL: 'test'
-; AVX512:  LV: Found an estimated cost of 1 for VF 1 For instruction: %valB = load i64, ptr %inB, align 8
+; AVX512:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB> = load ir<%inB>
 ; AVX512:  Cost of 6 for VF 2: REPLICATE ir<%valB> = load ir<%inB>
 ; AVX512:  Cost of 14 for VF 4: REPLICATE ir<%valB> = load ir<%inB>
 ; AVX512:  Cost of 10 for VF 8: WIDEN ir<%valB> = load ir<%inB>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i8-with-i8-index.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i8-with-i8-index.ll
index d894cdb753bcd6..435f7c3becd730 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i8-with-i8-index.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i8-with-i8-index.ll
@@ -17,21 +17,21 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
-; SSE2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %valB = load i8, ptr %inB, align 1
+; SSE2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB> = load ir<%inB>
 ; SSE2:  Cost of 25 for VF 2: REPLICATE ir<%valB> = load ir<%inB>
 ; SSE2:  Cost of 51 for VF 4: REPLICATE ir<%valB> = load ir<%inB>
 ; SSE2:  Cost of 103 for VF 8: REPLICATE ir<%valB> = load ir<%inB>
 ; SSE2:  Cost of 207 for VF 16: REPLICATE ir<%valB> = load ir<%inB>
 ;
 ; SSE42-LABEL: 'test'
-; SSE42:  LV: Found an estimated cost of 1 for VF 1 For instruction: %valB = load i8, ptr %inB, align 1
+; SSE42:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB> = load ir<%inB>
 ; SSE42:  Cost of 24 for VF 2: REPLICATE ir<%valB> = load ir<%inB>
 ; SSE42:  Cost of 48 for VF 4: REPLICATE ir<%valB> = load ir<%inB>
 ; SSE42:  Cost of 96 for VF 8: REPLICATE ir<%valB> = load ir<%inB>
 ; SSE42:  Cost of 192 for VF 16: REPLICATE ir<%valB> = load ir<%inB>
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: %valB = load i8, ptr %inB, align 1
+; AVX1:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB> = load ir<%inB>
 ; AVX1:  Cost of 24 for VF 2: REPLICATE ir<%valB> = load ir<%inB>
 ; AVX1:  Cost of 48 for VF 4: REPLICATE ir<%valB> = load ir<%inB>
 ; AVX1:  Cost of 96 for VF 8: REPLICATE ir<%valB> = load ir<%inB>
@@ -39,7 +39,7 @@ define void @test() {
 ; AVX1:  Cost of 385 for VF 32: REPLICATE ir<%valB> = load ir<%inB>
 ;
 ; AVX2-SLOWGATHER-LABEL: 'test'
-; AVX2-SLOWGATHER:  LV: Found an estimated cost of 1 for VF 1 For instruction: %valB = load i8, ptr %inB, align 1
+; AVX2-SLOWGATHER:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB> = load ir<%inB>
 ; AVX2-SLOWGATHER:  Cost of 4 for VF 2: REPLICATE ir<%valB> = load ir<%inB>
 ; AVX2-SLOWGATHER:  Cost of 8 for VF 4: REPLICATE ir<%valB> = load ir<%inB>
 ; AVX2-SLOWGATHER:  Cost of 16 for VF 8: REPLICATE ir<%valB> = load ir<%inB>
@@ -47,7 +47,7 @@ define void @test() {
 ; AVX2-SLOWGATHER:  Cost of 65 for VF 32: REPLICATE ir<%valB> = load ir<%inB>
 ;
 ; AVX2-FASTGATHER-LABEL: 'test'
-; AVX2-FASTGATHER:  LV: Found an estimated cost of 1 for VF 1 For instruction: %valB = load i8, ptr %inB, align 1
+; AVX2-FASTGATHER:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB> = load ir<%inB>
 ; AVX2-FASTGATHER:  Cost of 6 for VF 2: REPLICATE ir<%valB> = load ir<%inB>
 ; AVX2-FASTGATHER:  Cost of 13 for VF 4: REPLICATE ir<%valB> = load ir<%inB>
 ; AVX2-FASTGATHER:  Cost of 26 for VF 8: REPLICATE ir<%valB> = load ir<%inB>
@@ -55,7 +55,7 @@ define void @test() {
 ; AVX2-FASTGATHER:  Cost of 105 for VF 32: REPLICATE ir<%valB> = load ir<%inB>
 ;
 ; AVX512-LABEL: 'test'
-; AVX512:  LV: Found an estimated cost of 1 for VF 1 For instruction: %valB = load i8, ptr %inB, align 1
+; AVX512:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB> = load ir<%inB>
 ; AVX512:  Cost of 6 for VF 2: REPLICATE ir<%valB> = load ir<%inB>
 ; AVX512:  Cost of 13 for VF 4: REPLICATE ir<%valB> = load ir<%inB>
 ; AVX512:  Cost of 27 for VF 8: REPLICATE ir<%valB> = load ir<%inB>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/handle-iptr-with-data-layout-to-not-assert.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/handle-iptr-with-data-layout-to-not-assert.ll
index 1147497d9982d4..1c5e5d5aa21a3d 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/handle-iptr-with-data-layout-to-not-assert.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/handle-iptr-with-data-layout-to-not-assert.ll
@@ -5,7 +5,6 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define ptr @foo(ptr %__first, ptr %__last) #0 {
 ; CHECK-LABEL: 'foo'
-; CHECK:  LV: Found an estimated cost of 1 for VF 1 For instruction: store ptr %0, ptr %__last, align 8
 ; CHECK:  Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 2, vp<%next.gep>
 ; CHECK:  Cost of 1 for VF 4: INTERLEAVE-GROUP with factor 2, vp<%next.gep>
 ; CHECK:  Cost of 3 for VF 8: INTERLEAVE-GROUP with factor 2, vp<%next.gep>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-2.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-2.ll
index 3b3c3dcb780416..fe3fbb9c4bfee8 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-2.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-2.ll
@@ -13,7 +13,6 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
-; SSE2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load float, ptr %in0, align 4
 ; SSE2:  Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
 ; SSE2:    ir<%v0> = load from index 0
 ; SSE2:    ir<%v1> = load from index 1
@@ -24,7 +23,6 @@ define void @test() {
 ; SSE2:  Cost of 28 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load float, ptr %in0, align 4
 ; AVX1:  Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
 ; AVX1:    ir<%v0> = load from index 0
 ; AVX1:    ir<%v1> = load from index 1
@@ -36,7 +34,6 @@ define void @test() {
 ; AVX1:  Cost of 60 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
-; AVX2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load float, ptr %in0, align 4
 ; AVX2:  Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
 ; AVX2:    ir<%v0> = load from index 0
 ; AVX2:    ir<%v1> = load from index 1
@@ -54,7 +51,6 @@ define void @test() {
 ; AVX2:    ir<%v1> = load from index 1
 ;
 ; AVX512-LABEL: 'test'
-; AVX512:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load float, ptr %in0, align 4
 ; AVX512:  Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-3.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-3.ll
index 2218048e476813..24ac329df20d63 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-3.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-3.ll
@@ -13,14 +13,12 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
-; SSE2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load float, ptr %in0, align 4
 ; SSE2:  Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 14 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 28 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load float, ptr %in0, align 4
 ; AVX1:  Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 15 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -28,7 +26,6 @@ define void @test() {
 ; AVX1:  Cost of 60 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
-; AVX2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load float, ptr %in0, align 4
 ; AVX2:  Cost of 6 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
 ; AVX2:    ir<%v0> = load from index 0
 ; AVX2:    ir<%v1> = load from index 1
@@ -51,7 +48,6 @@ define void @test() {
 ; AVX2:    ir<%v2> = load from index 2
 ;
 ; AVX512-LABEL: 'test'
-; AVX512:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load float, ptr %in0, align 4
 ; AVX512:  Cost of 4 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-4.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-4.ll
index 679ff5ecec952e..b2cd0273efd8de 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-4.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-4.ll
@@ -13,14 +13,12 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
-; SSE2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load float, ptr %in0, align 4
 ; SSE2:  Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 14 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 28 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load float, ptr %in0, align 4
 ; AVX1:  Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 15 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -28,7 +26,6 @@ define void @test() {
 ; AVX1:  Cost of 60 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
-; AVX2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load float, ptr %in0, align 4
 ; AVX2:  Cost of 5 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
 ; AVX2:    ir<%v0> = load from index 0
 ; AVX2:    ir<%v1> = load from index 1
@@ -56,7 +53,6 @@ define void @test() {
 ; AVX2:    ir<%v3> = load from index 3
 ;
 ; AVX512-LABEL: 'test'
-; AVX512:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load float, ptr %in0, align 4
 ; AVX512:  Cost of 5 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-5.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-5.ll
index 31a1a91bdac59a..f8980b2fb0f30d 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-5.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-5.ll
@@ -13,14 +13,12 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
-; SSE2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load float, ptr %in0, align 4
 ; SSE2:  Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 14 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 28 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load float, ptr %in0, align 4
 ; AVX1:  Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 15 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -28,7 +26,6 @@ define void @test() {
 ; AVX1:  Cost of 60 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
-; AVX2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load float, ptr %in0, align 4
 ; AVX2:  Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 15 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -36,7 +33,6 @@ define void @test() {
 ; AVX2:  Cost of 60 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX512-LABEL: 'test'
-; AVX512:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load float, ptr %in0, align 4
 ; AVX512:  Cost of 6 for VF 2: INTERLEAVE-GROUP with factor 5, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-6.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-6.ll
index fd19eecd004459..c62d8eea92143c 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-6.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-6.ll
@@ -13,14 +13,12 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
-; SSE2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load float, ptr %in0, align 4
 ; SSE2:  Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 14 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 28 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load float, ptr %in0, align 4
 ; AVX1:  Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 15 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -28,7 +26,6 @@ define void @test() {
 ; AVX1:  Cost of 60 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
-; AVX2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load float, ptr %in0, align 4
 ; AVX2:  Cost of 8 for VF 2: INTERLEAVE-GROUP with factor 6, ir<%in0>
 ; AVX2:    ir<%v0> = load from index 0
 ; AVX2:    ir<%v1> = load from index 1
@@ -60,7 +57,6 @@ define void @test() {
 ; AVX2:  Cost of 60 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX512-LABEL: 'test'
-; AVX512:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load float, ptr %in0, align 4
 ; AVX512:  Cost of 7 for VF 2: INTERLEAVE-GROUP with factor 6, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-7.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-7.ll
index f0395dbb3bb891..68f7f233a8a86e 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-7.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-7.ll
@@ -13,14 +13,12 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
-; SSE2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load float, ptr %in0, align 4
 ; SSE2:  Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 14 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 28 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load float, ptr %in0, align 4
 ; AVX1:  Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 15 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -28,7 +26,6 @@ define void @test() {
 ; AVX1:  Cost of 60 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
-; AVX2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load float, ptr %in0, align 4
 ; AVX2:  Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 15 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -36,7 +33,6 @@ define void @test() {
 ; AVX2:  Cost of 60 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX512-LABEL: 'test'
-; AVX512:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load float, ptr %in0, align 4
 ; AVX512:  Cost of 8 for VF 2: INTERLEAVE-GROUP with factor 7, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-8.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-8.ll
index eafe9cfd912c4b..154c14dd895e0d 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-8.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-8.ll
@@ -13,14 +13,12 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
-; SSE2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load float, ptr %in0, align 4
 ; SSE2:  Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 14 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 28 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load float, ptr %in0, align 4
 ; AVX1:  Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 15 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -28,7 +26,6 @@ define void @test() {
 ; AVX1:  Cost of 60 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
-; AVX2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load float, ptr %in0, align 4
 ; AVX2:  Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 48 for VF 8: INTERLEAVE-GROUP with factor 8, ir<%in0>
@@ -44,7 +41,6 @@ define void @test() {
 ; AVX2:  Cost of 60 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX512-LABEL: 'test'
-; AVX512:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load float, ptr %in0, align 4
 ; AVX512:  Cost of 9 for VF 2: INTERLEAVE-GROUP with factor 8, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-2.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-2.ll
index 38fbf409100e8c..c9ddd0a38f1038 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-2.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-2.ll
@@ -13,7 +13,6 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
-; SSE2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load double, ptr %in0, align 8
 ; SSE2:  Cost of 4 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
 ; SSE2:    ir<%v0> = load from index 0
 ; SSE2:    ir<%v1> = load from index 1
@@ -22,7 +21,6 @@ define void @test() {
 ; SSE2:  Cost of 24 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load double, ptr %in0, align 8
 ; AVX1:  Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
 ; AVX1:    ir<%v0> = load from index 0
 ; AVX1:    ir<%v1> = load from index 1
@@ -32,7 +30,6 @@ define void @test() {
 ; AVX1:  Cost of 56 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
-; AVX2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load double, ptr %in0, align 8
 ; AVX2:  Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
 ; AVX2:    ir<%v0> = load from index 0
 ; AVX2:    ir<%v1> = load from index 1
@@ -50,7 +47,6 @@ define void @test() {
 ; AVX2:    ir<%v1> = load from index 1
 ;
 ; AVX512-LABEL: 'test'
-; AVX512:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load double, ptr %in0, align 8
 ; AVX512:  Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-3.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-3.ll
index 51bc25a559fdae..d42188eaa22996 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-3.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-3.ll
@@ -13,14 +13,12 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
-; SSE2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load double, ptr %in0, align 8
 ; SSE2:  Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 6 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 12 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 24 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load double, ptr %in0, align 8
 ; AVX1:  Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 14 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -28,7 +26,6 @@ define void @test() {
 ; AVX1:  Cost of 56 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
-; AVX2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load double, ptr %in0, align 8
 ; AVX2:  Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
 ; AVX2:    ir<%v0> = load from index 0
 ; AVX2:    ir<%v1> = load from index 1
@@ -48,7 +45,6 @@ define void @test() {
 ; AVX2:  Cost of 56 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX512-LABEL: 'test'
-; AVX512:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load double, ptr %in0, align 8
 ; AVX512:  Cost of 4 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-4.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-4.ll
index 5f3c8a3f925e85..1bc77d8d3ff3c4 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-4.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-4.ll
@@ -13,14 +13,12 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
-; SSE2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load double, ptr %in0, align 8
 ; SSE2:  Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 6 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 12 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 24 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load double, ptr %in0, align 8
 ; AVX1:  Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 14 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -28,7 +26,6 @@ define void @test() {
 ; AVX1:  Cost of 56 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
-; AVX2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load double, ptr %in0, align 8
 ; AVX2:  Cost of 8 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
 ; AVX2:    ir<%v0> = load from index 0
 ; AVX2:    ir<%v1> = load from index 1
@@ -52,7 +49,6 @@ define void @test() {
 ; AVX2:  Cost of 56 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX512-LABEL: 'test'
-; AVX512:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load double, ptr %in0, align 8
 ; AVX512:  Cost of 5 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-5.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-5.ll
index 7d18d05a344a84..9992f5ad71ab42 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-5.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-5.ll
@@ -13,14 +13,12 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
-; SSE2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load double, ptr %in0, align 8
 ; SSE2:  Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 6 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 12 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 24 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load double, ptr %in0, align 8
 ; AVX1:  Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 14 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -28,7 +26,6 @@ define void @test() {
 ; AVX1:  Cost of 56 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
-; AVX2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load double, ptr %in0, align 8
 ; AVX2:  Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 14 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -36,7 +33,6 @@ define void @test() {
 ; AVX2:  Cost of 56 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX512-LABEL: 'test'
-; AVX512:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load double, ptr %in0, align 8
 ; AVX512:  Cost of 14.5 for VF 2: INTERLEAVE-GROUP with factor 5, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-6.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-6.ll
index 909e0a9c8ed182..4676e641166dd2 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-6.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-6.ll
@@ -13,14 +13,12 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
-; SSE2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load double, ptr %in0, align 8
 ; SSE2:  Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 6 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 12 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 24 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load double, ptr %in0, align 8
 ; AVX1:  Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 14 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -28,7 +26,6 @@ define void @test() {
 ; AVX1:  Cost of 56 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
-; AVX2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load double, ptr %in0, align 8
 ; AVX2:  Cost of 9 for VF 2: INTERLEAVE-GROUP with factor 6, ir<%in0>
 ; AVX2:    ir<%v0> = load from index 0
 ; AVX2:    ir<%v1> = load from index 1
@@ -54,7 +51,6 @@ define void @test() {
 ; AVX2:  Cost of 56 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX512-LABEL: 'test'
-; AVX512:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load double, ptr %in0, align 8
 ; AVX512:  Cost of 17 for VF 2: INTERLEAVE-GROUP with factor 6, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-7.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-7.ll
index a9cef6bd260a1a..fbbc94de0c1a91 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-7.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-7.ll
@@ -13,14 +13,12 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
-; SSE2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load double, ptr %in0, align 8
 ; SSE2:  Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 6 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 12 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 24 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load double, ptr %in0, align 8
 ; AVX1:  Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 14 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -28,7 +26,6 @@ define void @test() {
 ; AVX1:  Cost of 56 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
-; AVX2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load double, ptr %in0, align 8
 ; AVX2:  Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 14 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -36,7 +33,6 @@ define void @test() {
 ; AVX2:  Cost of 56 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX512-LABEL: 'test'
-; AVX512:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load double, ptr %in0, align 8
 ; AVX512:  Cost of 19.5 for VF 2: INTERLEAVE-GROUP with factor 7, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-8.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-8.ll
index ba511ad467b1aa..7c22d3e51be763 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-8.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-8.ll
@@ -13,14 +13,12 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
-; SSE2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load double, ptr %in0, align 8
 ; SSE2:  Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 6 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 12 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 24 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load double, ptr %in0, align 8
 ; AVX1:  Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 14 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -28,7 +26,6 @@ define void @test() {
 ; AVX1:  Cost of 56 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
-; AVX2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load double, ptr %in0, align 8
 ; AVX2:  Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 14 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -36,7 +33,6 @@ define void @test() {
 ; AVX2:  Cost of 56 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX512-LABEL: 'test'
-; AVX512:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load double, ptr %in0, align 8
 ; AVX512:  Cost of 22 for VF 2: INTERLEAVE-GROUP with factor 8, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-2.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-2.ll
index acf9c541d9edf2..4fe1ae7f140c47 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-2.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-2.ll
@@ -14,7 +14,6 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
-; SSE2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
 ; SSE2:  Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
 ; SSE2:    ir<%v0> = load from index 0
 ; SSE2:    ir<%v1> = load from index 1
@@ -25,7 +24,6 @@ define void @test() {
 ; SSE2:  Cost of 32 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
 ; AVX1:  Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
 ; AVX1:    ir<%v0> = load from index 0
 ; AVX1:    ir<%v1> = load from index 1
@@ -37,7 +35,6 @@ define void @test() {
 ; AVX1:  Cost of 66 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
-; AVX2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
 ; AVX2:  Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
 ; AVX2:    ir<%v0> = load from index 0
 ; AVX2:    ir<%v1> = load from index 1
@@ -55,7 +52,6 @@ define void @test() {
 ; AVX2:    ir<%v1> = load from index 1
 ;
 ; AVX512DQ-LABEL: 'test'
-; AVX512DQ:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
 ; AVX512DQ:  Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
 ; AVX512DQ:    ir<%v0> = load from index 0
 ; AVX512DQ:    ir<%v1> = load from index 1
@@ -76,7 +72,6 @@ define void @test() {
 ; AVX512DQ:    ir<%v1> = load from index 1
 ;
 ; AVX512BW-LABEL: 'test'
-; AVX512BW:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
 ; AVX512BW:  Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
 ; AVX512BW:    ir<%v0> = load from index 0
 ; AVX512BW:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-3.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-3.ll
index 156bd432a34477..a7c0f2516d5ed9 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-3.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-3.ll
@@ -14,14 +14,12 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
-; SSE2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
 ; SSE2:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 32 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
 ; AVX1:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -29,7 +27,6 @@ define void @test() {
 ; AVX1:  Cost of 66 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
-; AVX2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
 ; AVX2:  Cost of 7 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
 ; AVX2:    ir<%v0> = load from index 0
 ; AVX2:    ir<%v1> = load from index 1
@@ -52,7 +49,6 @@ define void @test() {
 ; AVX2:    ir<%v2> = load from index 2
 ;
 ; AVX512DQ-LABEL: 'test'
-; AVX512DQ:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
 ; AVX512DQ:  Cost of 7 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
 ; AVX512DQ:    ir<%v0> = load from index 0
 ; AVX512DQ:    ir<%v1> = load from index 1
@@ -79,7 +75,6 @@ define void @test() {
 ; AVX512DQ:    ir<%v2> = load from index 2
 ;
 ; AVX512BW-LABEL: 'test'
-; AVX512BW:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
 ; AVX512BW:  Cost of 4 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
 ; AVX512BW:    ir<%v0> = load from index 0
 ; AVX512BW:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-4.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-4.ll
index 8ceb2529fed177..71bae49cbd9061 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-4.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-4.ll
@@ -14,14 +14,12 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
-; SSE2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
 ; SSE2:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 32 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
 ; AVX1:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -29,7 +27,6 @@ define void @test() {
 ; AVX1:  Cost of 66 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
-; AVX2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
 ; AVX2:  Cost of 7 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
 ; AVX2:    ir<%v0> = load from index 0
 ; AVX2:    ir<%v1> = load from index 1
@@ -57,7 +54,6 @@ define void @test() {
 ; AVX2:    ir<%v3> = load from index 3
 ;
 ; AVX512DQ-LABEL: 'test'
-; AVX512DQ:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
 ; AVX512DQ:  Cost of 7 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
 ; AVX512DQ:    ir<%v0> = load from index 0
 ; AVX512DQ:    ir<%v1> = load from index 1
@@ -90,7 +86,6 @@ define void @test() {
 ; AVX512DQ:    ir<%v3> = load from index 3
 ;
 ; AVX512BW-LABEL: 'test'
-; AVX512BW:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
 ; AVX512BW:  Cost of 5 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
 ; AVX512BW:    ir<%v0> = load from index 0
 ; AVX512BW:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-5.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-5.ll
index a7b6bf588af099..14c75eba680e9b 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-5.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-5.ll
@@ -14,14 +14,12 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
-; SSE2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
 ; SSE2:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 32 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
 ; AVX1:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -29,7 +27,6 @@ define void @test() {
 ; AVX1:  Cost of 66 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
-; AVX2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
 ; AVX2:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -37,7 +34,6 @@ define void @test() {
 ; AVX2:  Cost of 66 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX512DQ-LABEL: 'test'
-; AVX512DQ:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
 ; AVX512DQ:  Cost of 24 for VF 2: INTERLEAVE-GROUP with factor 5, ir<%in0>
 ; AVX512DQ:    ir<%v0> = load from index 0
 ; AVX512DQ:    ir<%v1> = load from index 1
@@ -76,7 +72,6 @@ define void @test() {
 ; AVX512DQ:    ir<%v4> = load from index 4
 ;
 ; AVX512BW-LABEL: 'test'
-; AVX512BW:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
 ; AVX512BW:  Cost of 11 for VF 2: INTERLEAVE-GROUP with factor 5, ir<%in0>
 ; AVX512BW:    ir<%v0> = load from index 0
 ; AVX512BW:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-6.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-6.ll
index 65da1d5767c42c..727eec70f83873 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-6.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-6.ll
@@ -14,14 +14,12 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
-; SSE2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
 ; SSE2:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 32 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
 ; AVX1:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -29,7 +27,6 @@ define void @test() {
 ; AVX1:  Cost of 66 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
-; AVX2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
 ; AVX2:  Cost of 16 for VF 2: INTERLEAVE-GROUP with factor 6, ir<%in0>
 ; AVX2:    ir<%v0> = load from index 0
 ; AVX2:    ir<%v1> = load from index 1
@@ -67,7 +64,6 @@ define void @test() {
 ; AVX2:    ir<%v5> = load from index 5
 ;
 ; AVX512DQ-LABEL: 'test'
-; AVX512DQ:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
 ; AVX512DQ:  Cost of 16 for VF 2: INTERLEAVE-GROUP with factor 6, ir<%in0>
 ; AVX512DQ:    ir<%v0> = load from index 0
 ; AVX512DQ:    ir<%v1> = load from index 1
@@ -112,7 +108,6 @@ define void @test() {
 ; AVX512DQ:    ir<%v5> = load from index 5
 ;
 ; AVX512BW-LABEL: 'test'
-; AVX512BW:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
 ; AVX512BW:  Cost of 13 for VF 2: INTERLEAVE-GROUP with factor 6, ir<%in0>
 ; AVX512BW:    ir<%v0> = load from index 0
 ; AVX512BW:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-7.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-7.ll
index 26e82797634251..d593db1853fe8f 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-7.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-7.ll
@@ -14,14 +14,12 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
-; SSE2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
 ; SSE2:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 32 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
 ; AVX1:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -29,7 +27,6 @@ define void @test() {
 ; AVX1:  Cost of 66 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
-; AVX2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
 ; AVX2:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -37,7 +34,6 @@ define void @test() {
 ; AVX2:  Cost of 66 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX512DQ-LABEL: 'test'
-; AVX512DQ:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
 ; AVX512DQ:  Cost of 33 for VF 2: INTERLEAVE-GROUP with factor 7, ir<%in0>
 ; AVX512DQ:    ir<%v0> = load from index 0
 ; AVX512DQ:    ir<%v1> = load from index 1
@@ -88,7 +84,6 @@ define void @test() {
 ; AVX512DQ:    ir<%v6> = load from index 6
 ;
 ; AVX512BW-LABEL: 'test'
-; AVX512BW:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
 ; AVX512BW:  Cost of 15 for VF 2: INTERLEAVE-GROUP with factor 7, ir<%in0>
 ; AVX512BW:    ir<%v0> = load from index 0
 ; AVX512BW:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-8.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-8.ll
index 759f45629adf1d..5ef380a3032da3 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-8.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-8.ll
@@ -14,14 +14,12 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
-; SSE2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
 ; SSE2:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 32 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
 ; AVX1:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -29,7 +27,6 @@ define void @test() {
 ; AVX1:  Cost of 66 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
-; AVX2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
 ; AVX2:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -37,7 +34,6 @@ define void @test() {
 ; AVX2:  Cost of 66 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX512DQ-LABEL: 'test'
-; AVX512DQ:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
 ; AVX512DQ:  Cost of 34 for VF 2: INTERLEAVE-GROUP with factor 8, ir<%in0>
 ; AVX512DQ:    ir<%v0> = load from index 0
 ; AVX512DQ:    ir<%v1> = load from index 1
@@ -94,7 +90,6 @@ define void @test() {
 ; AVX512DQ:    ir<%v7> = load from index 7
 ;
 ; AVX512BW-LABEL: 'test'
-; AVX512BW:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
 ; AVX512BW:  Cost of 17 for VF 2: INTERLEAVE-GROUP with factor 8, ir<%in0>
 ; AVX512BW:    ir<%v0> = load from index 0
 ; AVX512BW:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-2-indices-0u.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-2-indices-0u.ll
index 3d0e795d653a46..85043e46d2023f 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-2-indices-0u.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-2-indices-0u.ll
@@ -13,7 +13,6 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
-; SSE2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
 ; SSE2:  Cost of 2 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
 ; SSE2:    ir<%v0> = load from index 0
 ; SSE2:  Cost of 3 for VF 4: INTERLEAVE-GROUP with factor 2, ir<%in0>
@@ -22,7 +21,6 @@ define void @test() {
 ; SSE2:  Cost of 44 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
 ; AVX1:  Cost of 2 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
 ; AVX1:    ir<%v0> = load from index 0
 ; AVX1:  Cost of 2 for VF 4: INTERLEAVE-GROUP with factor 2, ir<%in0>
@@ -32,7 +30,6 @@ define void @test() {
 ; AVX1:  Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
-; AVX2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
 ; AVX2:  Cost of 2 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
 ; AVX2:    ir<%v0> = load from index 0
 ; AVX2:  Cost of 2 for VF 4: INTERLEAVE-GROUP with factor 2, ir<%in0>
@@ -45,7 +42,6 @@ define void @test() {
 ; AVX2:    ir<%v0> = load from index 0
 ;
 ; AVX512-LABEL: 'test'
-; AVX512:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
 ; AVX512:  Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:  Cost of 1 for VF 4: INTERLEAVE-GROUP with factor 2, ir<%in0>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-2.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-2.ll
index 302e0cb447d653..3d8916969eade1 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-2.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-2.ll
@@ -13,7 +13,6 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
-; SSE2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
 ; SSE2:  Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
 ; SSE2:    ir<%v0> = load from index 0
 ; SSE2:    ir<%v1> = load from index 1
@@ -24,7 +23,6 @@ define void @test() {
 ; SSE2:  Cost of 44 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
 ; AVX1:  Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
 ; AVX1:    ir<%v0> = load from index 0
 ; AVX1:    ir<%v1> = load from index 1
@@ -36,7 +34,6 @@ define void @test() {
 ; AVX1:  Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
-; AVX2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
 ; AVX2:  Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
 ; AVX2:    ir<%v0> = load from index 0
 ; AVX2:    ir<%v1> = load from index 1
@@ -54,7 +51,6 @@ define void @test() {
 ; AVX2:    ir<%v1> = load from index 1
 ;
 ; AVX512-LABEL: 'test'
-; AVX512:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
 ; AVX512:  Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-3-indices-01u.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-3-indices-01u.ll
index e843b716aa01b2..02d90c5ffe2fbb 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-3-indices-01u.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-3-indices-01u.ll
@@ -13,14 +13,12 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
-; SSE2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
 ; SSE2:  Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 22 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 44 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
 ; AVX1:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 17 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -28,7 +26,6 @@ define void @test() {
 ; AVX1:  Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
-; AVX2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
 ; AVX2:  Cost of 5 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
 ; AVX2:    ir<%v0> = load from index 0
 ; AVX2:    ir<%v1> = load from index 1
@@ -46,7 +43,6 @@ define void @test() {
 ; AVX2:    ir<%v1> = load from index 1
 ;
 ; AVX512-LABEL: 'test'
-; AVX512:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
 ; AVX512:  Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-3-indices-0uu.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-3-indices-0uu.ll
index f40ad60dc6f3c8..e92c161c8ff960 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-3-indices-0uu.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-3-indices-0uu.ll
@@ -13,14 +13,12 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
-; SSE2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
 ; SSE2:  Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 22 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 44 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
 ; AVX1:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 17 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -28,7 +26,6 @@ define void @test() {
 ; AVX1:  Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
-; AVX2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
 ; AVX2:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 3 for VF 4: INTERLEAVE-GROUP with factor 3, ir<%in0>
 ; AVX2:    ir<%v0> = load from index 0
@@ -40,7 +37,6 @@ define void @test() {
 ; AVX2:    ir<%v0> = load from index 0
 ;
 ; AVX512-LABEL: 'test'
-; AVX512:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
 ; AVX512:  Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:  Cost of 1 for VF 4: INTERLEAVE-GROUP with factor 3, ir<%in0>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-3.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-3.ll
index c0de2fcacd4123..b68f44855dc3ed 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-3.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-3.ll
@@ -13,14 +13,12 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
-; SSE2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
 ; SSE2:  Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 22 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 44 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
 ; AVX1:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 17 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -28,7 +26,6 @@ define void @test() {
 ; AVX1:  Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
-; AVX2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
 ; AVX2:  Cost of 6 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
 ; AVX2:    ir<%v0> = load from index 0
 ; AVX2:    ir<%v1> = load from index 1
@@ -51,7 +48,6 @@ define void @test() {
 ; AVX2:    ir<%v2> = load from index 2
 ;
 ; AVX512-LABEL: 'test'
-; AVX512:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
 ; AVX512:  Cost of 4 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-4-indices-012u.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-4-indices-012u.ll
index 41bd92713c3861..69baf083073e40 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-4-indices-012u.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-4-indices-012u.ll
@@ -13,14 +13,12 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
-; SSE2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
 ; SSE2:  Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 22 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 44 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
 ; AVX1:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 17 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -28,7 +26,6 @@ define void @test() {
 ; AVX1:  Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
-; AVX2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
 ; AVX2:  Cost of 4 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
 ; AVX2:    ir<%v0> = load from index 0
 ; AVX2:    ir<%v1> = load from index 1
@@ -51,7 +48,6 @@ define void @test() {
 ; AVX2:    ir<%v2> = load from index 2
 ;
 ; AVX512-LABEL: 'test'
-; AVX512:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
 ; AVX512:  Cost of 4 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-4-indices-01uu.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-4-indices-01uu.ll
index 9a8ae47d519071..43e097a4ffeb01 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-4-indices-01uu.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-4-indices-01uu.ll
@@ -13,14 +13,12 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
-; SSE2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
 ; SSE2:  Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 22 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 44 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
 ; AVX1:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 17 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -28,7 +26,6 @@ define void @test() {
 ; AVX1:  Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
-; AVX2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
 ; AVX2:  Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
 ; AVX2:    ir<%v0> = load from index 0
 ; AVX2:    ir<%v1> = load from index 1
@@ -46,7 +43,6 @@ define void @test() {
 ; AVX2:    ir<%v1> = load from index 1
 ;
 ; AVX512-LABEL: 'test'
-; AVX512:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
 ; AVX512:  Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-4-indices-0uuu.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-4-indices-0uuu.ll
index d2e518a981de51..e3838488d3a967 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-4-indices-0uuu.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-4-indices-0uuu.ll
@@ -13,14 +13,12 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
-; SSE2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
 ; SSE2:  Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 22 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 44 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
 ; AVX1:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 17 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -28,7 +26,6 @@ define void @test() {
 ; AVX1:  Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
-; AVX2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
 ; AVX2:  Cost of 2 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
 ; AVX2:    ir<%v0> = load from index 0
 ; AVX2:  Cost of 4 for VF 4: INTERLEAVE-GROUP with factor 4, ir<%in0>
@@ -41,7 +38,6 @@ define void @test() {
 ; AVX2:    ir<%v0> = load from index 0
 ;
 ; AVX512-LABEL: 'test'
-; AVX512:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
 ; AVX512:  Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:  Cost of 1 for VF 4: INTERLEAVE-GROUP with factor 4, ir<%in0>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-4.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-4.ll
index bd7a63609acde1..bdb68363cdc2d8 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-4.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-4.ll
@@ -13,14 +13,12 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
-; SSE2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
 ; SSE2:  Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 22 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 44 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
 ; AVX1:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 17 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -28,7 +26,6 @@ define void @test() {
 ; AVX1:  Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
-; AVX2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
 ; AVX2:  Cost of 5 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
 ; AVX2:    ir<%v0> = load from index 0
 ; AVX2:    ir<%v1> = load from index 1
@@ -56,7 +53,6 @@ define void @test() {
 ; AVX2:    ir<%v3> = load from index 3
 ;
 ; AVX512-LABEL: 'test'
-; AVX512:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
 ; AVX512:  Cost of 5 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-5.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-5.ll
index cb315e61e57a2c..7f5b1687b59780 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-5.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-5.ll
@@ -13,14 +13,12 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
-; SSE2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
 ; SSE2:  Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 22 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 44 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
 ; AVX1:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 17 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -28,7 +26,6 @@ define void @test() {
 ; AVX1:  Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
-; AVX2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
 ; AVX2:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 17 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -36,7 +33,6 @@ define void @test() {
 ; AVX2:  Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX512-LABEL: 'test'
-; AVX512:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
 ; AVX512:  Cost of 6 for VF 2: INTERLEAVE-GROUP with factor 5, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-6.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-6.ll
index c916394a0c611e..aec171a16f12ba 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-6.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-6.ll
@@ -13,14 +13,12 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
-; SSE2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
 ; SSE2:  Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 22 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 44 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
 ; AVX1:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 17 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -28,7 +26,6 @@ define void @test() {
 ; AVX1:  Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
-; AVX2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
 ; AVX2:  Cost of 8 for VF 2: INTERLEAVE-GROUP with factor 6, ir<%in0>
 ; AVX2:    ir<%v0> = load from index 0
 ; AVX2:    ir<%v1> = load from index 1
@@ -60,7 +57,6 @@ define void @test() {
 ; AVX2:  Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX512-LABEL: 'test'
-; AVX512:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
 ; AVX512:  Cost of 7 for VF 2: INTERLEAVE-GROUP with factor 6, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-7.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-7.ll
index 0c2cc5f663ef3e..5d4904aec87eb7 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-7.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-7.ll
@@ -13,14 +13,12 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
-; SSE2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
 ; SSE2:  Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 22 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 44 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
 ; AVX1:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 17 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -28,7 +26,6 @@ define void @test() {
 ; AVX1:  Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
-; AVX2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
 ; AVX2:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 17 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -36,7 +33,6 @@ define void @test() {
 ; AVX2:  Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX512-LABEL: 'test'
-; AVX512:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
 ; AVX512:  Cost of 8 for VF 2: INTERLEAVE-GROUP with factor 7, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-8.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-8.ll
index c64076fca0d62d..e1af00bcbe1fa2 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-8.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-8.ll
@@ -13,14 +13,12 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
-; SSE2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
 ; SSE2:  Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 22 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 44 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
 ; AVX1:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 17 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -28,7 +26,6 @@ define void @test() {
 ; AVX1:  Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
-; AVX2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
 ; AVX2:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 48 for VF 8: INTERLEAVE-GROUP with factor 8, ir<%in0>
@@ -44,7 +41,6 @@ define void @test() {
 ; AVX2:  Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX512-LABEL: 'test'
-; AVX512:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
 ; AVX512:  Cost of 9 for VF 2: INTERLEAVE-GROUP with factor 8, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-2.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-2.ll
index d0a99efab706d9..a664eb308630b7 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-2.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-2.ll
@@ -13,7 +13,6 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
-; SSE2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i64, ptr %in0, align 8
 ; SSE2:  Cost of 4 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
 ; SSE2:    ir<%v0> = load from index 0
 ; SSE2:    ir<%v1> = load from index 1
@@ -22,7 +21,6 @@ define void @test() {
 ; SSE2:  Cost of 40 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i64, ptr %in0, align 8
 ; AVX1:  Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
 ; AVX1:    ir<%v0> = load from index 0
 ; AVX1:    ir<%v1> = load from index 1
@@ -32,7 +30,6 @@ define void @test() {
 ; AVX1:  Cost of 72 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
-; AVX2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i64, ptr %in0, align 8
 ; AVX2:  Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
 ; AVX2:    ir<%v0> = load from index 0
 ; AVX2:    ir<%v1> = load from index 1
@@ -50,7 +47,6 @@ define void @test() {
 ; AVX2:    ir<%v1> = load from index 1
 ;
 ; AVX512-LABEL: 'test'
-; AVX512:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i64, ptr %in0, align 8
 ; AVX512:  Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-3.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-3.ll
index 17cc63d11eb402..1fbcf7559aad64 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-3.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-3.ll
@@ -13,14 +13,12 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
-; SSE2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i64, ptr %in0, align 8
 ; SSE2:  Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 10 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 20 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 40 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i64, ptr %in0, align 8
 ; AVX1:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 9 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 18 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -28,7 +26,6 @@ define void @test() {
 ; AVX1:  Cost of 72 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
-; AVX2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i64, ptr %in0, align 8
 ; AVX2:  Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
 ; AVX2:    ir<%v0> = load from index 0
 ; AVX2:    ir<%v1> = load from index 1
@@ -48,7 +45,6 @@ define void @test() {
 ; AVX2:  Cost of 72 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX512-LABEL: 'test'
-; AVX512:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i64, ptr %in0, align 8
 ; AVX512:  Cost of 4 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-4.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-4.ll
index 180ef142675f52..43adad539acd4d 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-4.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-4.ll
@@ -13,14 +13,12 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
-; SSE2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i64, ptr %in0, align 8
 ; SSE2:  Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 10 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 20 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 40 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i64, ptr %in0, align 8
 ; AVX1:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 9 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 18 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -28,7 +26,6 @@ define void @test() {
 ; AVX1:  Cost of 72 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
-; AVX2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i64, ptr %in0, align 8
 ; AVX2:  Cost of 8 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
 ; AVX2:    ir<%v0> = load from index 0
 ; AVX2:    ir<%v1> = load from index 1
@@ -52,7 +49,6 @@ define void @test() {
 ; AVX2:  Cost of 72 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX512-LABEL: 'test'
-; AVX512:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i64, ptr %in0, align 8
 ; AVX512:  Cost of 5 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-5.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-5.ll
index 65a27fc96f2234..2c6c071b77be2f 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-5.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-5.ll
@@ -13,14 +13,12 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
-; SSE2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i64, ptr %in0, align 8
 ; SSE2:  Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 10 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 20 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 40 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i64, ptr %in0, align 8
 ; AVX1:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 9 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 18 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -28,7 +26,6 @@ define void @test() {
 ; AVX1:  Cost of 72 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
-; AVX2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i64, ptr %in0, align 8
 ; AVX2:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 9 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 18 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -36,7 +33,6 @@ define void @test() {
 ; AVX2:  Cost of 72 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX512-LABEL: 'test'
-; AVX512:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i64, ptr %in0, align 8
 ; AVX512:  Cost of 14.5 for VF 2: INTERLEAVE-GROUP with factor 5, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-6.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-6.ll
index d261e5242f8d86..18c252a8a3aaca 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-6.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-6.ll
@@ -13,14 +13,12 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
-; SSE2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i64, ptr %in0, align 8
 ; SSE2:  Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 10 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 20 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 40 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i64, ptr %in0, align 8
 ; AVX1:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 9 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 18 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -28,7 +26,6 @@ define void @test() {
 ; AVX1:  Cost of 72 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
-; AVX2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i64, ptr %in0, align 8
 ; AVX2:  Cost of 9 for VF 2: INTERLEAVE-GROUP with factor 6, ir<%in0>
 ; AVX2:    ir<%v0> = load from index 0
 ; AVX2:    ir<%v1> = load from index 1
@@ -54,7 +51,6 @@ define void @test() {
 ; AVX2:  Cost of 72 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX512-LABEL: 'test'
-; AVX512:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i64, ptr %in0, align 8
 ; AVX512:  Cost of 17 for VF 2: INTERLEAVE-GROUP with factor 6, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-7.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-7.ll
index 04c8db7a583572..6460df78c20c02 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-7.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-7.ll
@@ -13,14 +13,12 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
-; SSE2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i64, ptr %in0, align 8
 ; SSE2:  Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 10 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 20 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 40 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i64, ptr %in0, align 8
 ; AVX1:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 9 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 18 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -28,7 +26,6 @@ define void @test() {
 ; AVX1:  Cost of 72 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
-; AVX2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i64, ptr %in0, align 8
 ; AVX2:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 9 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 18 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -36,7 +33,6 @@ define void @test() {
 ; AVX2:  Cost of 72 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX512-LABEL: 'test'
-; AVX512:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i64, ptr %in0, align 8
 ; AVX512:  Cost of 19.5 for VF 2: INTERLEAVE-GROUP with factor 7, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-8.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-8.ll
index a2b5d1c4df2f83..d4d99b530f0108 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-8.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-8.ll
@@ -13,14 +13,12 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
-; SSE2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i64, ptr %in0, align 8
 ; SSE2:  Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 10 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 20 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 40 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i64, ptr %in0, align 8
 ; AVX1:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 9 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 18 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -28,7 +26,6 @@ define void @test() {
 ; AVX1:  Cost of 72 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
-; AVX2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i64, ptr %in0, align 8
 ; AVX2:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 9 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 18 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -36,7 +33,6 @@ define void @test() {
 ; AVX2:  Cost of 72 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX512-LABEL: 'test'
-; AVX512:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i64, ptr %in0, align 8
 ; AVX512:  Cost of 22 for VF 2: INTERLEAVE-GROUP with factor 8, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-2.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-2.ll
index f6f057ca88ca63..4686fbc32153a9 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-2.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-2.ll
@@ -14,14 +14,12 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
-; SSE2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
 ; SSE2:  Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 23 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 47 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
 ; AVX1:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -29,7 +27,6 @@ define void @test() {
 ; AVX1:  Cost of 65 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
-; AVX2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
 ; AVX2:  Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
 ; AVX2:    ir<%v0> = load from index 0
 ; AVX2:    ir<%v1> = load from index 1
@@ -47,7 +44,6 @@ define void @test() {
 ; AVX2:    ir<%v1> = load from index 1
 ;
 ; AVX512DQ-LABEL: 'test'
-; AVX512DQ:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
 ; AVX512DQ:  Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
 ; AVX512DQ:    ir<%v0> = load from index 0
 ; AVX512DQ:    ir<%v1> = load from index 1
@@ -68,7 +64,6 @@ define void @test() {
 ; AVX512DQ:    ir<%v1> = load from index 1
 ;
 ; AVX512BW-LABEL: 'test'
-; AVX512BW:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
 ; AVX512BW:  Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
 ; AVX512BW:    ir<%v0> = load from index 0
 ; AVX512BW:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-3.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-3.ll
index f385a46789ecc2..d5bdfeacdb30e6 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-3.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-3.ll
@@ -14,14 +14,12 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
-; SSE2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
 ; SSE2:  Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 23 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 47 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
 ; AVX1:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -29,7 +27,6 @@ define void @test() {
 ; AVX1:  Cost of 65 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
-; AVX2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
 ; AVX2:  Cost of 5 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
 ; AVX2:    ir<%v0> = load from index 0
 ; AVX2:    ir<%v1> = load from index 1
@@ -52,7 +49,6 @@ define void @test() {
 ; AVX2:    ir<%v2> = load from index 2
 ;
 ; AVX512DQ-LABEL: 'test'
-; AVX512DQ:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
 ; AVX512DQ:  Cost of 5 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
 ; AVX512DQ:    ir<%v0> = load from index 0
 ; AVX512DQ:    ir<%v1> = load from index 1
@@ -79,7 +75,6 @@ define void @test() {
 ; AVX512DQ:    ir<%v2> = load from index 2
 ;
 ; AVX512BW-LABEL: 'test'
-; AVX512BW:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
 ; AVX512BW:  Cost of 4 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
 ; AVX512BW:    ir<%v0> = load from index 0
 ; AVX512BW:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-4.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-4.ll
index 13f02ae479c417..7390b2f56010f4 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-4.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-4.ll
@@ -14,14 +14,12 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
-; SSE2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
 ; SSE2:  Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 23 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 47 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
 ; AVX1:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -29,7 +27,6 @@ define void @test() {
 ; AVX1:  Cost of 65 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
-; AVX2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
 ; AVX2:  Cost of 5 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
 ; AVX2:    ir<%v0> = load from index 0
 ; AVX2:    ir<%v1> = load from index 1
@@ -57,7 +54,6 @@ define void @test() {
 ; AVX2:    ir<%v3> = load from index 3
 ;
 ; AVX512DQ-LABEL: 'test'
-; AVX512DQ:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
 ; AVX512DQ:  Cost of 5 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
 ; AVX512DQ:    ir<%v0> = load from index 0
 ; AVX512DQ:    ir<%v1> = load from index 1
@@ -90,7 +86,6 @@ define void @test() {
 ; AVX512DQ:    ir<%v3> = load from index 3
 ;
 ; AVX512BW-LABEL: 'test'
-; AVX512BW:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
 ; AVX512BW:  Cost of 5 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
 ; AVX512BW:    ir<%v0> = load from index 0
 ; AVX512BW:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-5.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-5.ll
index 490ac46b0b4ac9..3fd3b9e11017d6 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-5.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-5.ll
@@ -14,14 +14,12 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
-; SSE2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
 ; SSE2:  Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 23 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 47 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
 ; AVX1:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -29,7 +27,6 @@ define void @test() {
 ; AVX1:  Cost of 65 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
-; AVX2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
 ; AVX2:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -37,7 +34,6 @@ define void @test() {
 ; AVX2:  Cost of 65 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX512DQ-LABEL: 'test'
-; AVX512DQ:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
 ; AVX512DQ:  Cost of 22 for VF 2: INTERLEAVE-GROUP with factor 5, ir<%in0>
 ; AVX512DQ:    ir<%v0> = load from index 0
 ; AVX512DQ:    ir<%v1> = load from index 1
@@ -76,7 +72,6 @@ define void @test() {
 ; AVX512DQ:    ir<%v4> = load from index 4
 ;
 ; AVX512BW-LABEL: 'test'
-; AVX512BW:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
 ; AVX512BW:  Cost of 6 for VF 2: INTERLEAVE-GROUP with factor 5, ir<%in0>
 ; AVX512BW:    ir<%v0> = load from index 0
 ; AVX512BW:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-6.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-6.ll
index 08c5f31fbd5d03..d8b77269cda803 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-6.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-6.ll
@@ -14,14 +14,12 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
-; SSE2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
 ; SSE2:  Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 23 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 47 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
 ; AVX1:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -29,7 +27,6 @@ define void @test() {
 ; AVX1:  Cost of 65 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
-; AVX2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
 ; AVX2:  Cost of 8 for VF 2: INTERLEAVE-GROUP with factor 6, ir<%in0>
 ; AVX2:    ir<%v0> = load from index 0
 ; AVX2:    ir<%v1> = load from index 1
@@ -67,7 +64,6 @@ define void @test() {
 ; AVX2:    ir<%v5> = load from index 5
 ;
 ; AVX512DQ-LABEL: 'test'
-; AVX512DQ:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
 ; AVX512DQ:  Cost of 8 for VF 2: INTERLEAVE-GROUP with factor 6, ir<%in0>
 ; AVX512DQ:    ir<%v0> = load from index 0
 ; AVX512DQ:    ir<%v1> = load from index 1
@@ -112,7 +108,6 @@ define void @test() {
 ; AVX512DQ:    ir<%v5> = load from index 5
 ;
 ; AVX512BW-LABEL: 'test'
-; AVX512BW:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
 ; AVX512BW:  Cost of 7 for VF 2: INTERLEAVE-GROUP with factor 6, ir<%in0>
 ; AVX512BW:    ir<%v0> = load from index 0
 ; AVX512BW:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-7.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-7.ll
index afbe3c05bfb9e4..e2e3dedca27c8d 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-7.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-7.ll
@@ -14,14 +14,12 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
-; SSE2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
 ; SSE2:  Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 23 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 47 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
 ; AVX1:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -29,7 +27,6 @@ define void @test() {
 ; AVX1:  Cost of 65 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
-; AVX2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
 ; AVX2:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -37,7 +34,6 @@ define void @test() {
 ; AVX2:  Cost of 65 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX512DQ-LABEL: 'test'
-; AVX512DQ:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
 ; AVX512DQ:  Cost of 31 for VF 2: INTERLEAVE-GROUP with factor 7, ir<%in0>
 ; AVX512DQ:    ir<%v0> = load from index 0
 ; AVX512DQ:    ir<%v1> = load from index 1
@@ -88,7 +84,6 @@ define void @test() {
 ; AVX512DQ:    ir<%v6> = load from index 6
 ;
 ; AVX512BW-LABEL: 'test'
-; AVX512BW:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
 ; AVX512BW:  Cost of 8 for VF 2: INTERLEAVE-GROUP with factor 7, ir<%in0>
 ; AVX512BW:    ir<%v0> = load from index 0
 ; AVX512BW:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-8.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-8.ll
index 11bcbd083c59f6..1fd2c4898762d9 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-8.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-8.ll
@@ -14,14 +14,12 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
-; SSE2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
 ; SSE2:  Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 23 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 47 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
 ; AVX1:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -29,7 +27,6 @@ define void @test() {
 ; AVX1:  Cost of 65 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
-; AVX2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
 ; AVX2:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -37,7 +34,6 @@ define void @test() {
 ; AVX2:  Cost of 65 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX512DQ-LABEL: 'test'
-; AVX512DQ:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
 ; AVX512DQ:  Cost of 33 for VF 2: INTERLEAVE-GROUP with factor 8, ir<%in0>
 ; AVX512DQ:    ir<%v0> = load from index 0
 ; AVX512DQ:    ir<%v1> = load from index 1
@@ -94,7 +90,6 @@ define void @test() {
 ; AVX512DQ:    ir<%v7> = load from index 7
 ;
 ; AVX512BW-LABEL: 'test'
-; AVX512BW:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
 ; AVX512BW:  Cost of 9 for VF 2: INTERLEAVE-GROUP with factor 8, ir<%in0>
 ; AVX512BW:    ir<%v0> = load from index 0
 ; AVX512BW:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-gather-i32-with-i8-index.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-gather-i32-with-i8-index.ll
index 8f5da77027970f..3d33bc34dda674 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-gather-i32-with-i8-index.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-gather-i32-with-i8-index.ll
@@ -17,14 +17,14 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE-LABEL: 'test'
-; SSE:  LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i32, ptr %inB, align 4
+; SSE:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
 ; SSE:  Cost of 3000000 for VF 2: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ; SSE:  Cost of 3000000 for VF 4: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ; SSE:  Cost of 3000000 for VF 8: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ; SSE:  Cost of 3000000 for VF 16: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i32, ptr %inB, align 4
+; AVX1:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
 ; AVX1:  Cost of 3000000 for VF 2: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ; AVX1:  Cost of 3000000 for VF 4: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ; AVX1:  Cost of 3000000 for VF 8: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
@@ -32,7 +32,7 @@ define void @test() {
 ; AVX1:  Cost of 3000000 for VF 32: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ;
 ; AVX2-SLOWGATHER-LABEL: 'test'
-; AVX2-SLOWGATHER:  LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i32, ptr %inB, align 4
+; AVX2-SLOWGATHER:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
 ; AVX2-SLOWGATHER:  Cost of 3000000 for VF 2: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ; AVX2-SLOWGATHER:  Cost of 3000000 for VF 4: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ; AVX2-SLOWGATHER:  Cost of 3000000 for VF 8: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
@@ -40,21 +40,21 @@ define void @test() {
 ; AVX2-SLOWGATHER:  Cost of 3000000 for VF 32: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ;
 ; AVX2-FASTGATHER-LABEL: 'test'
-; AVX2-FASTGATHER:  LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i32, ptr %inB, align 4
-; AVX2-FASTGATHER:  Cost of 4 for VF 2: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
-; AVX2-FASTGATHER:  Cost of 6 for VF 4: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
-; AVX2-FASTGATHER:  Cost of 12 for VF 8: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
-; AVX2-FASTGATHER:  Cost of 24 for VF 16: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
-; AVX2-FASTGATHER:  Cost of 48 for VF 32: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
+; AVX2-FASTGATHER:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX2-FASTGATHER:  Cost of 4 for VF 2: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX2-FASTGATHER:  Cost of 6 for VF 4: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX2-FASTGATHER:  Cost of 12 for VF 8: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX2-FASTGATHER:  Cost of 24 for VF 16: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX2-FASTGATHER:  Cost of 48 for VF 32: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
 ;
 ; AVX512-LABEL: 'test'
-; AVX512:  LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i32, ptr %inB, align 4
-; AVX512:  Cost of 8 for VF 2: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
-; AVX512:  Cost of 17 for VF 4: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
-; AVX512:  Cost of 10 for VF 8: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
-; AVX512:  Cost of 18 for VF 16: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
-; AVX512:  Cost of 36 for VF 32: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
-; AVX512:  Cost of 72 for VF 64: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
+; AVX512:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX512:  Cost of 8 for VF 2: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX512:  Cost of 17 for VF 4: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX512:  Cost of 10 for VF 8: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX512:  Cost of 18 for VF 16: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX512:  Cost of 36 for VF 32: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX512:  Cost of 72 for VF 64: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
 ;
 entry:
   br label %for.body
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-gather-i64-with-i8-index.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-gather-i64-with-i8-index.ll
index 6801a549e19c4f..b0fec98fcdf51d 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-gather-i64-with-i8-index.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-gather-i64-with-i8-index.ll
@@ -17,14 +17,14 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE-LABEL: 'test'
-; SSE:  LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i64, ptr %inB, align 8
+; SSE:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
 ; SSE:  Cost of 3000000 for VF 2: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ; SSE:  Cost of 3000000 for VF 4: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ; SSE:  Cost of 3000000 for VF 8: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ; SSE:  Cost of 3000000 for VF 16: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i64, ptr %inB, align 8
+; AVX1:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
 ; AVX1:  Cost of 3000000 for VF 2: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ; AVX1:  Cost of 3000000 for VF 4: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ; AVX1:  Cost of 3000000 for VF 8: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
@@ -32,7 +32,7 @@ define void @test() {
 ; AVX1:  Cost of 3000000 for VF 32: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ;
 ; AVX2-SLOWGATHER-LABEL: 'test'
-; AVX2-SLOWGATHER:  LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i64, ptr %inB, align 8
+; AVX2-SLOWGATHER:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
 ; AVX2-SLOWGATHER:  Cost of 3000000 for VF 2: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ; AVX2-SLOWGATHER:  Cost of 3000000 for VF 4: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ; AVX2-SLOWGATHER:  Cost of 3000000 for VF 8: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
@@ -40,21 +40,21 @@ define void @test() {
 ; AVX2-SLOWGATHER:  Cost of 3000000 for VF 32: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ;
 ; AVX2-FASTGATHER-LABEL: 'test'
-; AVX2-FASTGATHER:  LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i64, ptr %inB, align 8
-; AVX2-FASTGATHER:  Cost of 4 for VF 2: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
-; AVX2-FASTGATHER:  Cost of 6 for VF 4: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
-; AVX2-FASTGATHER:  Cost of 12 for VF 8: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
-; AVX2-FASTGATHER:  Cost of 24 for VF 16: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
-; AVX2-FASTGATHER:  Cost of 48 for VF 32: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
+; AVX2-FASTGATHER:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX2-FASTGATHER:  Cost of 4 for VF 2: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX2-FASTGATHER:  Cost of 6 for VF 4: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX2-FASTGATHER:  Cost of 12 for VF 8: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX2-FASTGATHER:  Cost of 24 for VF 16: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX2-FASTGATHER:  Cost of 48 for VF 32: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
 ;
 ; AVX512-LABEL: 'test'
-; AVX512:  LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i64, ptr %inB, align 8
-; AVX512:  Cost of 8 for VF 2: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
-; AVX512:  Cost of 18 for VF 4: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
-; AVX512:  Cost of 10 for VF 8: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
-; AVX512:  Cost of 20 for VF 16: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
-; AVX512:  Cost of 40 for VF 32: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
-; AVX512:  Cost of 80 for VF 64: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
+; AVX512:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX512:  Cost of 8 for VF 2: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX512:  Cost of 18 for VF 4: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX512:  Cost of 10 for VF 8: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX512:  Cost of 20 for VF 16: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX512:  Cost of 40 for VF 32: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX512:  Cost of 80 for VF 64: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
 ;
 entry:
   br label %for.body
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-interleaved-load-i16.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-interleaved-load-i16.ll
index d14260279e17f9..52d05b807ba827 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-interleaved-load-i16.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-interleaved-load-i16.ll
@@ -20,8 +20,6 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test1(ptr noalias nocapture %points, ptr noalias nocapture readonly %x, ptr noalias nocapture readonly %y) {
 ; DISABLED_MASKED_STRIDED-LABEL: 'test1'
-; DISABLED_MASKED_STRIDED:  LV: Found an estimated cost of 1 for VF 1 For instruction: %i2 = load i16, ptr %arrayidx2, align 2
-; DISABLED_MASKED_STRIDED:  LV: Found an estimated cost of 1 for VF 1 For instruction: %i4 = load i16, ptr %arrayidx7, align 2
 ; DISABLED_MASKED_STRIDED:  Cost of 6 for VF 2: REPLICATE ir<%i2> = load ir<%arrayidx2>
 ; DISABLED_MASKED_STRIDED:  Cost of 6 for VF 2: REPLICATE ir<%i4> = load ir<%arrayidx7>
 ; DISABLED_MASKED_STRIDED:  Cost of 13 for VF 4: REPLICATE ir<%i2> = load ir<%arrayidx2>
@@ -32,8 +30,6 @@ define void @test1(ptr noalias nocapture %points, ptr noalias nocapture readonly
 ; DISABLED_MASKED_STRIDED:  Cost of 55 for VF 16: REPLICATE ir<%i4> = load ir<%arrayidx7>
 ;
 ; ENABLED_MASKED_STRIDED-LABEL: 'test1'
-; ENABLED_MASKED_STRIDED:  LV: Found an estimated cost of 1 for VF 1 For instruction: %i2 = load i16, ptr %arrayidx2, align 2
-; ENABLED_MASKED_STRIDED:  LV: Found an estimated cost of 1 for VF 1 For instruction: %i4 = load i16, ptr %arrayidx7, align 2
 ; ENABLED_MASKED_STRIDED:  Cost of 8 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%arrayidx2>
 ; ENABLED_MASKED_STRIDED:    ir<%i2> = load from index 0
 ; ENABLED_MASKED_STRIDED:    ir<%i4> = load from index 1
@@ -81,8 +77,6 @@ for.end:
 
 define void @test2(ptr noalias nocapture %points, i32 %numPoints, ptr noalias nocapture readonly %x, ptr noalias nocapture readonly %y) {
 ; DISABLED_MASKED_STRIDED-LABEL: 'test2'
-; DISABLED_MASKED_STRIDED:  LV: Found an estimated cost of 1 for VF 1 For instruction: %i2 = load i16, ptr %arrayidx2, align 2
-; DISABLED_MASKED_STRIDED:  LV: Found an estimated cost of 1 for VF 1 For instruction: %i4 = load i16, ptr %arrayidx7, align 2
 ; DISABLED_MASKED_STRIDED:  Cost of 3000000 for VF 2: REPLICATE ir<%i2> = load ir<%arrayidx2> (S->V)
 ; DISABLED_MASKED_STRIDED:  Cost of 3000000 for VF 2: REPLICATE ir<%i4> = load ir<%arrayidx7> (S->V)
 ; DISABLED_MASKED_STRIDED:  Cost of 3000000 for VF 4: REPLICATE ir<%i2> = load ir<%arrayidx2> (S->V)
@@ -93,8 +87,6 @@ define void @test2(ptr noalias nocapture %points, i32 %numPoints, ptr noalias no
 ; DISABLED_MASKED_STRIDED:  Cost of 3000000 for VF 16: REPLICATE ir<%i4> = load ir<%arrayidx7> (S->V)
 ;
 ; ENABLED_MASKED_STRIDED-LABEL: 'test2'
-; ENABLED_MASKED_STRIDED:  LV: Found an estimated cost of 1 for VF 1 For instruction: %i2 = load i16, ptr %arrayidx2, align 2
-; ENABLED_MASKED_STRIDED:  LV: Found an estimated cost of 1 for VF 1 For instruction: %i4 = load i16, ptr %arrayidx7, align 2
 ; ENABLED_MASKED_STRIDED:  Cost of 8 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%arrayidx2>, vp<[[VP8:%[0-9]+]]>
 ; ENABLED_MASKED_STRIDED:    ir<%i2> = load from index 0
 ; ENABLED_MASKED_STRIDED:    ir<%i4> = load from index 1
@@ -152,24 +144,20 @@ for.end:
 
 define void @test(ptr noalias nocapture %points, ptr noalias nocapture readonly %x, ptr noalias nocapture readnone %y) {
 ; DISABLED_MASKED_STRIDED-LABEL: 'test'
-; DISABLED_MASKED_STRIDED:  LV: Found an estimated cost of 1 for VF 1 For instruction: %i2 = load i16, ptr %arrayidx, align 2
-; DISABLED_MASKED_STRIDED:  LV: Found an estimated cost of 1 for VF 1 For instruction: %i4 = load i16, ptr %arrayidx6, align 2
 ; DISABLED_MASKED_STRIDED:  Cost of 3000000 for VF 2: REPLICATE ir<%i4> = load ir<%arrayidx6> (S->V)
 ; DISABLED_MASKED_STRIDED:  Cost of 3000000 for VF 4: REPLICATE ir<%i4> = load ir<%arrayidx6> (S->V)
 ; DISABLED_MASKED_STRIDED:  Cost of 3000000 for VF 8: REPLICATE ir<%i4> = load ir<%arrayidx6> (S->V)
 ; DISABLED_MASKED_STRIDED:  Cost of 3000000 for VF 16: REPLICATE ir<%i4> = load ir<%arrayidx6> (S->V)
 ;
 ; ENABLED_MASKED_STRIDED-LABEL: 'test'
-; ENABLED_MASKED_STRIDED:  LV: Found an estimated cost of 1 for VF 1 For instruction: %i2 = load i16, ptr %arrayidx, align 2
-; ENABLED_MASKED_STRIDED:  LV: Found an estimated cost of 1 for VF 1 For instruction: %i4 = load i16, ptr %arrayidx6, align 2
 ; ENABLED_MASKED_STRIDED:  Cost of 7 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%arrayidx6>, ir<%cmp1>
-; ENABLED_MASKED_STRIDED:    ir<%i4> = load from index 0
+; ENABLED_MASKED_STRIDED:    ir<%i4> = load from index 0 (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
 ; ENABLED_MASKED_STRIDED:  Cost of 9 for VF 4: INTERLEAVE-GROUP with factor 3, ir<%arrayidx6>, ir<%cmp1>
-; ENABLED_MASKED_STRIDED:    ir<%i4> = load from index 0
+; ENABLED_MASKED_STRIDED:    ir<%i4> = load from index 0 (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
 ; ENABLED_MASKED_STRIDED:  Cost of 9 for VF 8: INTERLEAVE-GROUP with factor 3, ir<%arrayidx6>, ir<%cmp1>
-; ENABLED_MASKED_STRIDED:    ir<%i4> = load from index 0
+; ENABLED_MASKED_STRIDED:    ir<%i4> = load from index 0 (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
 ; ENABLED_MASKED_STRIDED:  Cost of 14 for VF 16: INTERLEAVE-GROUP with factor 3, ir<%arrayidx6>, ir<%cmp1>
-; ENABLED_MASKED_STRIDED:    ir<%i4> = load from index 0
+; ENABLED_MASKED_STRIDED:    ir<%i4> = load from index 0 (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
 ;
 entry:
   br label %for.body
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-interleaved-store-i16.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-interleaved-store-i16.ll
index cd59ce8482f603..6b7b2919a9899e 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-interleaved-store-i16.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-interleaved-store-i16.ll
@@ -20,8 +20,6 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test1(ptr noalias nocapture %points, ptr noalias nocapture readonly %x, ptr noalias nocapture readonly %y) {
 ; DISABLED_MASKED_STRIDED-LABEL: 'test1'
-; DISABLED_MASKED_STRIDED:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i16 %0, ptr %arrayidx2, align 2
-; DISABLED_MASKED_STRIDED:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i16 %2, ptr %arrayidx7, align 2
 ; DISABLED_MASKED_STRIDED:  Cost of 6 for VF 2: REPLICATE store ir<%0>, ir<%arrayidx2>
 ; DISABLED_MASKED_STRIDED:  Cost of 6 for VF 2: REPLICATE store ir<%2>, ir<%arrayidx7>
 ; DISABLED_MASKED_STRIDED:  Cost of 13 for VF 4: REPLICATE store ir<%0>, ir<%arrayidx2>
@@ -32,8 +30,6 @@ define void @test1(ptr noalias nocapture %points, ptr noalias nocapture readonly
 ; DISABLED_MASKED_STRIDED:  Cost of 55 for VF 16: REPLICATE store ir<%2>, ir<%arrayidx7>
 ;
 ; ENABLED_MASKED_STRIDED-LABEL: 'test1'
-; ENABLED_MASKED_STRIDED:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i16 %0, ptr %arrayidx2, align 2
-; ENABLED_MASKED_STRIDED:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i16 %2, ptr %arrayidx7, align 2
 ; ENABLED_MASKED_STRIDED:  Cost of 6 for VF 2: REPLICATE store ir<%0>, ir<%arrayidx2>
 ; ENABLED_MASKED_STRIDED:  Cost of 6 for VF 2: REPLICATE store ir<%2>, ir<%arrayidx7>
 ; ENABLED_MASKED_STRIDED:  Cost of 14 for VF 4: INTERLEAVE-GROUP with factor 4, ir<%arrayidx2>
@@ -74,8 +70,6 @@ for.end:
 
 define void @test2(ptr noalias nocapture %points, i32 %numPoints, ptr noalias nocapture readonly %x, ptr noalias nocapture readonly %y) {
 ; DISABLED_MASKED_STRIDED-LABEL: 'test2'
-; DISABLED_MASKED_STRIDED:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i16 %0, ptr %arrayidx2, align 2
-; DISABLED_MASKED_STRIDED:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i16 %2, ptr %arrayidx7, align 2
 ; DISABLED_MASKED_STRIDED:  Cost of 3000000 for VF 2: REPLICATE store ir<%0>, ir<%arrayidx2>
 ; DISABLED_MASKED_STRIDED:  Cost of 3000000 for VF 2: REPLICATE store ir<%2>, ir<%arrayidx7>
 ; DISABLED_MASKED_STRIDED:  Cost of 3000000 for VF 4: REPLICATE store ir<%0>, ir<%arrayidx2>
@@ -86,8 +80,6 @@ define void @test2(ptr noalias nocapture %points, i32 %numPoints, ptr noalias no
 ; DISABLED_MASKED_STRIDED:  Cost of 3000000 for VF 16: REPLICATE store ir<%2>, ir<%arrayidx7>
 ;
 ; ENABLED_MASKED_STRIDED-LABEL: 'test2'
-; ENABLED_MASKED_STRIDED:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i16 %0, ptr %arrayidx2, align 2
-; ENABLED_MASKED_STRIDED:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i16 %2, ptr %arrayidx7, align 2
 ; ENABLED_MASKED_STRIDED:  Cost of 13 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%arrayidx2>, vp<[[VP8:%[0-9]+]]>
 ; ENABLED_MASKED_STRIDED:  Cost of 14 for VF 4: INTERLEAVE-GROUP with factor 4, ir<%arrayidx2>, vp<[[VP8]]>
 ; ENABLED_MASKED_STRIDED:  Cost of 14 for VF 8: INTERLEAVE-GROUP with factor 4, ir<%arrayidx2>, vp<[[VP8]]>
@@ -137,14 +129,12 @@ for.end:
 
 define void @test(ptr noalias nocapture %points, ptr noalias nocapture readonly %x, ptr noalias nocapture readnone %y) {
 ; DISABLED_MASKED_STRIDED-LABEL: 'test'
-; DISABLED_MASKED_STRIDED:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i16 %0, ptr %arrayidx6, align 2
 ; DISABLED_MASKED_STRIDED:  Cost of 2 for VF 2: profitable to scalarize store i16 %0, ptr %arrayidx6, align 2
 ; DISABLED_MASKED_STRIDED:  Cost of 4 for VF 4: profitable to scalarize store i16 %0, ptr %arrayidx6, align 2
 ; DISABLED_MASKED_STRIDED:  Cost of 8 for VF 8: profitable to scalarize store i16 %0, ptr %arrayidx6, align 2
 ; DISABLED_MASKED_STRIDED:  Cost of 16.5 for VF 16: profitable to scalarize store i16 %0, ptr %arrayidx6, align 2
 ;
 ; ENABLED_MASKED_STRIDED-LABEL: 'test'
-; ENABLED_MASKED_STRIDED:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i16 %0, ptr %arrayidx6, align 2
 ; ENABLED_MASKED_STRIDED:  Cost of 2 for VF 2: profitable to scalarize store i16 %0, ptr %arrayidx6, align 2
 ; ENABLED_MASKED_STRIDED:  Cost of 4 for VF 4: profitable to scalarize store i16 %0, ptr %arrayidx6, align 2
 ; ENABLED_MASKED_STRIDED:  Cost of 12 for VF 8: INTERLEAVE-GROUP with factor 3, ir<%arrayidx6>, ir<%cmp1>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i16.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i16.ll
index 9cb0e482da47c9..a4a86cb40b3ea0 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i16.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i16.ll
@@ -16,14 +16,14 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test(ptr %B) {
 ; SSE-LABEL: 'test'
-; SSE:  LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i16, ptr %inB, align 2
+; SSE:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB>
 ; SSE:  Cost of 3000000 for VF 2: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ; SSE:  Cost of 3000000 for VF 4: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ; SSE:  Cost of 3000000 for VF 8: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ; SSE:  Cost of 3000000 for VF 16: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i16, ptr %inB, align 2
+; AVX1:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB>
 ; AVX1:  Cost of 3000000 for VF 2: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ; AVX1:  Cost of 3000000 for VF 4: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ; AVX1:  Cost of 3000000 for VF 8: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
@@ -31,7 +31,7 @@ define void @test(ptr %B) {
 ; AVX1:  Cost of 3000000 for VF 32: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ;
 ; AVX2-LABEL: 'test'
-; AVX2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i16, ptr %inB, align 2
+; AVX2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB>
 ; AVX2:  Cost of 3000000 for VF 2: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ; AVX2:  Cost of 3000000 for VF 4: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ; AVX2:  Cost of 3000000 for VF 8: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
@@ -39,7 +39,7 @@ define void @test(ptr %B) {
 ; AVX2:  Cost of 3000000 for VF 32: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ;
 ; AVX512-LABEL: 'test'
-; AVX512:  LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i16, ptr %inB, align 2
+; AVX512:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB>
 ; AVX512:  Cost of 2 for VF 2: WIDEN ir<%valB.loaded> = load vp<[[VP6:%[0-9]+]]>, ir<%canLoad>
 ; AVX512:  Cost of 2 for VF 4: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
 ; AVX512:  Cost of 1 for VF 8: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i32.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i32.ll
index fc42ce6e6f73ff..be448f4b9359f3 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i32.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i32.ll
@@ -16,14 +16,14 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test(ptr %B) {
 ; SSE-LABEL: 'test'
-; SSE:  LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i32, ptr %inB, align 4
+; SSE:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB>
 ; SSE:  Cost of 3000000 for VF 2: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ; SSE:  Cost of 3000000 for VF 4: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ; SSE:  Cost of 3000000 for VF 8: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ; SSE:  Cost of 3000000 for VF 16: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i32, ptr %inB, align 4
+; AVX1:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB>
 ; AVX1:  Cost of 3 for VF 2: WIDEN ir<%valB.loaded> = load vp<[[VP6:%[0-9]+]]>, ir<%canLoad>
 ; AVX1:  Cost of 2 for VF 4: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
 ; AVX1:  Cost of 2 for VF 8: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
@@ -31,7 +31,7 @@ define void @test(ptr %B) {
 ; AVX1:  Cost of 8 for VF 32: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
 ;
 ; AVX2-LABEL: 'test'
-; AVX2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i32, ptr %inB, align 4
+; AVX2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB>
 ; AVX2:  Cost of 3 for VF 2: WIDEN ir<%valB.loaded> = load vp<[[VP6:%[0-9]+]]>, ir<%canLoad>
 ; AVX2:  Cost of 2 for VF 4: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
 ; AVX2:  Cost of 2 for VF 8: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
@@ -39,7 +39,7 @@ define void @test(ptr %B) {
 ; AVX2:  Cost of 8 for VF 32: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
 ;
 ; AVX512-LABEL: 'test'
-; AVX512:  LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i32, ptr %inB, align 4
+; AVX512:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB>
 ; AVX512:  Cost of 2 for VF 2: WIDEN ir<%valB.loaded> = load vp<[[VP6:%[0-9]+]]>, ir<%canLoad>
 ; AVX512:  Cost of 1 for VF 4: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
 ; AVX512:  Cost of 1 for VF 8: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i64.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i64.ll
index 48c9b01beb8881..9900b8f2637def 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i64.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i64.ll
@@ -16,14 +16,14 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test(ptr %B) {
 ; SSE-LABEL: 'test'
-; SSE:  LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i64, ptr %inB, align 8
+; SSE:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB>
 ; SSE:  Cost of 3000000 for VF 2: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ; SSE:  Cost of 3000000 for VF 4: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ; SSE:  Cost of 3000000 for VF 8: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ; SSE:  Cost of 3000000 for VF 16: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i64, ptr %inB, align 8
+; AVX1:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB>
 ; AVX1:  Cost of 2 for VF 2: WIDEN ir<%valB.loaded> = load vp<[[VP6:%[0-9]+]]>, ir<%canLoad>
 ; AVX1:  Cost of 2 for VF 4: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
 ; AVX1:  Cost of 4 for VF 8: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
@@ -31,7 +31,7 @@ define void @test(ptr %B) {
 ; AVX1:  Cost of 16 for VF 32: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
 ;
 ; AVX2-LABEL: 'test'
-; AVX2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i64, ptr %inB, align 8
+; AVX2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB>
 ; AVX2:  Cost of 2 for VF 2: WIDEN ir<%valB.loaded> = load vp<[[VP6:%[0-9]+]]>, ir<%canLoad>
 ; AVX2:  Cost of 2 for VF 4: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
 ; AVX2:  Cost of 4 for VF 8: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
@@ -39,7 +39,7 @@ define void @test(ptr %B) {
 ; AVX2:  Cost of 16 for VF 32: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
 ;
 ; AVX512-LABEL: 'test'
-; AVX512:  LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i64, ptr %inB, align 8
+; AVX512:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB>
 ; AVX512:  Cost of 1 for VF 2: WIDEN ir<%valB.loaded> = load vp<[[VP6:%[0-9]+]]>, ir<%canLoad>
 ; AVX512:  Cost of 1 for VF 4: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
 ; AVX512:  Cost of 1 for VF 8: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i8.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i8.ll
index c8225989777042..594d4766b43a4d 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i8.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i8.ll
@@ -16,14 +16,14 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test(ptr %B) {
 ; SSE-LABEL: 'test'
-; SSE:  LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i8, ptr %inB, align 1
+; SSE:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB>
 ; SSE:  Cost of 3000000 for VF 2: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ; SSE:  Cost of 3000000 for VF 4: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ; SSE:  Cost of 3000000 for VF 8: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ; SSE:  Cost of 3000000 for VF 16: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i8, ptr %inB, align 1
+; AVX1:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB>
 ; AVX1:  Cost of 3000000 for VF 2: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ; AVX1:  Cost of 3000000 for VF 4: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ; AVX1:  Cost of 3000000 for VF 8: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
@@ -31,7 +31,7 @@ define void @test(ptr %B) {
 ; AVX1:  Cost of 3000000 for VF 32: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ;
 ; AVX2-LABEL: 'test'
-; AVX2:  LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i8, ptr %inB, align 1
+; AVX2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB>
 ; AVX2:  Cost of 3000000 for VF 2: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ; AVX2:  Cost of 3000000 for VF 4: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ; AVX2:  Cost of 3000000 for VF 8: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
@@ -39,7 +39,7 @@ define void @test(ptr %B) {
 ; AVX2:  Cost of 3000000 for VF 32: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ;
 ; AVX512-LABEL: 'test'
-; AVX512:  LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i8, ptr %inB, align 1
+; AVX512:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB>
 ; AVX512:  Cost of 2 for VF 2: WIDEN ir<%valB.loaded> = load vp<[[VP6:%[0-9]+]]>, ir<%canLoad>
 ; AVX512:  Cost of 2 for VF 4: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
 ; AVX512:  Cost of 2 for VF 8: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-scatter-i32-with-i8-index.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-scatter-i32-with-i8-index.ll
index 6959fea2d512bc..b026016d1331bd 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-scatter-i32-with-i8-index.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-scatter-i32-with-i8-index.ll
@@ -17,21 +17,18 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
-; SSE2:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i32 %valB, ptr %out, align 4
 ; SSE2:  Cost of 2.5 for VF 2: profitable to scalarize store i32 %valB, ptr %out, align 4
 ; SSE2:  Cost of 5.5 for VF 4: profitable to scalarize store i32 %valB, ptr %out, align 4
 ; SSE2:  Cost of 11 for VF 8: profitable to scalarize store i32 %valB, ptr %out, align 4
 ; SSE2:  Cost of 22 for VF 16: profitable to scalarize store i32 %valB, ptr %out, align 4
 ;
 ; SSE42-LABEL: 'test'
-; SSE42:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i32 %valB, ptr %out, align 4
 ; SSE42:  Cost of 2 for VF 2: profitable to scalarize store i32 %valB, ptr %out, align 4
 ; SSE42:  Cost of 4 for VF 4: profitable to scalarize store i32 %valB, ptr %out, align 4
 ; SSE42:  Cost of 8 for VF 8: profitable to scalarize store i32 %valB, ptr %out, align 4
 ; SSE42:  Cost of 16 for VF 16: profitable to scalarize store i32 %valB, ptr %out, align 4
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i32 %valB, ptr %out, align 4
 ; AVX1:  Cost of 2 for VF 2: profitable to scalarize store i32 %valB, ptr %out, align 4
 ; AVX1:  Cost of 4 for VF 4: profitable to scalarize store i32 %valB, ptr %out, align 4
 ; AVX1:  Cost of 8.5 for VF 8: profitable to scalarize store i32 %valB, ptr %out, align 4
@@ -39,7 +36,6 @@ define void @test() {
 ; AVX1:  Cost of 34 for VF 32: profitable to scalarize store i32 %valB, ptr %out, align 4
 ;
 ; AVX2-LABEL: 'test'
-; AVX2:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i32 %valB, ptr %out, align 4
 ; AVX2:  Cost of 2 for VF 2: profitable to scalarize store i32 %valB, ptr %out, align 4
 ; AVX2:  Cost of 4 for VF 4: profitable to scalarize store i32 %valB, ptr %out, align 4
 ; AVX2:  Cost of 8.5 for VF 8: profitable to scalarize store i32 %valB, ptr %out, align 4
@@ -47,13 +43,12 @@ define void @test() {
 ; AVX2:  Cost of 34 for VF 32: profitable to scalarize store i32 %valB, ptr %out, align 4
 ;
 ; AVX512-LABEL: 'test'
-; AVX512:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i32 %valB, ptr %out, align 4
 ; AVX512:  Cost of 5 for VF 2: REPLICATE store ir<%valB>, ir<%out>
 ; AVX512:  Cost of 10.5 for VF 4: REPLICATE store ir<%valB>, ir<%out>
-; AVX512:  Cost of 10 for VF 8: WIDEN store ir<%out>, ir<%valB>, ir<%canStore>
-; AVX512:  Cost of 18 for VF 16: WIDEN store ir<%out>, ir<%valB>, ir<%canStore>
-; AVX512:  Cost of 36 for VF 32: WIDEN store ir<%out>, ir<%valB>, ir<%canStore>
-; AVX512:  Cost of 72 for VF 64: WIDEN store ir<%out>, ir<%valB>, ir<%canStore>
+; AVX512:  Cost of 10 for VF 8: WIDEN store ir<%out>, ir<%valB>, ir<%canStore> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX512:  Cost of 18 for VF 16: WIDEN store ir<%out>, ir<%valB>, ir<%canStore> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX512:  Cost of 36 for VF 32: WIDEN store ir<%out>, ir<%valB>, ir<%canStore> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX512:  Cost of 72 for VF 64: WIDEN store ir<%out>, ir<%valB>, ir<%canStore> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
 ;
 entry:
   br label %for.body
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-scatter-i64-with-i8-index.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-scatter-i64-with-i8-index.ll
index 41ae89933204e4..f9c6a5b14c8024 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-scatter-i64-with-i8-index.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-scatter-i64-with-i8-index.ll
@@ -17,21 +17,18 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
-; SSE2:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i64 %valB, ptr %out, align 8
 ; SSE2:  Cost of 2.5 for VF 2: profitable to scalarize store i64 %valB, ptr %out, align 8
 ; SSE2:  Cost of 5 for VF 4: profitable to scalarize store i64 %valB, ptr %out, align 8
 ; SSE2:  Cost of 10 for VF 8: profitable to scalarize store i64 %valB, ptr %out, align 8
 ; SSE2:  Cost of 20 for VF 16: profitable to scalarize store i64 %valB, ptr %out, align 8
 ;
 ; SSE42-LABEL: 'test'
-; SSE42:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i64 %valB, ptr %out, align 8
 ; SSE42:  Cost of 2 for VF 2: profitable to scalarize store i64 %valB, ptr %out, align 8
 ; SSE42:  Cost of 4 for VF 4: profitable to scalarize store i64 %valB, ptr %out, align 8
 ; SSE42:  Cost of 8 for VF 8: profitable to scalarize store i64 %valB, ptr %out, align 8
 ; SSE42:  Cost of 16 for VF 16: profitable to scalarize store i64 %valB, ptr %out, align 8
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i64 %valB, ptr %out, align 8
 ; AVX1:  Cost of 2 for VF 2: profitable to scalarize store i64 %valB, ptr %out, align 8
 ; AVX1:  Cost of 4.5 for VF 4: profitable to scalarize store i64 %valB, ptr %out, align 8
 ; AVX1:  Cost of 9 for VF 8: profitable to scalarize store i64 %valB, ptr %out, align 8
@@ -39,7 +36,6 @@ define void @test() {
 ; AVX1:  Cost of 36 for VF 32: profitable to scalarize store i64 %valB, ptr %out, align 8
 ;
 ; AVX2-LABEL: 'test'
-; AVX2:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i64 %valB, ptr %out, align 8
 ; AVX2:  Cost of 2 for VF 2: profitable to scalarize store i64 %valB, ptr %out, align 8
 ; AVX2:  Cost of 4.5 for VF 4: profitable to scalarize store i64 %valB, ptr %out, align 8
 ; AVX2:  Cost of 9 for VF 8: profitable to scalarize store i64 %valB, ptr %out, align 8
@@ -47,13 +43,12 @@ define void @test() {
 ; AVX2:  Cost of 36 for VF 32: profitable to scalarize store i64 %valB, ptr %out, align 8
 ;
 ; AVX512-LABEL: 'test'
-; AVX512:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i64 %valB, ptr %out, align 8
 ; AVX512:  Cost of 5 for VF 2: REPLICATE store ir<%valB>, ir<%out>
 ; AVX512:  Cost of 11 for VF 4: REPLICATE store ir<%valB>, ir<%out>
-; AVX512:  Cost of 10 for VF 8: WIDEN store ir<%out>, ir<%valB>, ir<%canStore>
-; AVX512:  Cost of 20 for VF 16: WIDEN store ir<%out>, ir<%valB>, ir<%canStore>
-; AVX512:  Cost of 40 for VF 32: WIDEN store ir<%out>, ir<%valB>, ir<%canStore>
-; AVX512:  Cost of 80 for VF 64: WIDEN store ir<%out>, ir<%valB>, ir<%canStore>
+; AVX512:  Cost of 10 for VF 8: WIDEN store ir<%out>, ir<%valB>, ir<%canStore> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX512:  Cost of 20 for VF 16: WIDEN store ir<%out>, ir<%valB>, ir<%canStore> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX512:  Cost of 40 for VF 32: WIDEN store ir<%out>, ir<%valB>, ir<%canStore> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX512:  Cost of 80 for VF 64: WIDEN store ir<%out>, ir<%valB>, ir<%canStore> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
 ;
 entry:
   br label %for.body
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i16.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i16.ll
index 61436a61dba509..886e593362383e 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i16.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i16.ll
@@ -16,14 +16,12 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test(ptr %C) {
 ; SSE-LABEL: 'test'
-; SSE:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i16 %valB, ptr %out, align 2
 ; SSE:  Cost of 2 for VF 2: profitable to scalarize store i16 %valB, ptr %out, align 2
 ; SSE:  Cost of 4 for VF 4: profitable to scalarize store i16 %valB, ptr %out, align 2
 ; SSE:  Cost of 8 for VF 8: profitable to scalarize store i16 %valB, ptr %out, align 2
 ; SSE:  Cost of 16 for VF 16: profitable to scalarize store i16 %valB, ptr %out, align 2
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i16 %valB, ptr %out, align 2
 ; AVX1:  Cost of 2 for VF 2: profitable to scalarize store i16 %valB, ptr %out, align 2
 ; AVX1:  Cost of 4 for VF 4: profitable to scalarize store i16 %valB, ptr %out, align 2
 ; AVX1:  Cost of 8 for VF 8: profitable to scalarize store i16 %valB, ptr %out, align 2
@@ -31,7 +29,6 @@ define void @test(ptr %C) {
 ; AVX1:  Cost of 33 for VF 32: profitable to scalarize store i16 %valB, ptr %out, align 2
 ;
 ; AVX2-LABEL: 'test'
-; AVX2:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i16 %valB, ptr %out, align 2
 ; AVX2:  Cost of 2 for VF 2: profitable to scalarize store i16 %valB, ptr %out, align 2
 ; AVX2:  Cost of 4 for VF 4: profitable to scalarize store i16 %valB, ptr %out, align 2
 ; AVX2:  Cost of 8 for VF 8: profitable to scalarize store i16 %valB, ptr %out, align 2
@@ -39,7 +36,6 @@ define void @test(ptr %C) {
 ; AVX2:  Cost of 33 for VF 32: profitable to scalarize store i16 %valB, ptr %out, align 2
 ;
 ; AVX512-LABEL: 'test'
-; AVX512:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i16 %valB, ptr %out, align 2
 ; AVX512:  Cost of 2 for VF 2: WIDEN store vp<[[VP7:%[0-9]+]]>, ir<%valB>, ir<%canStore>
 ; AVX512:  Cost of 2 for VF 4: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
 ; AVX512:  Cost of 1 for VF 8: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i32.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i32.ll
index 0afea8d1664d58..9be2bd10f40fad 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i32.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i32.ll
@@ -16,21 +16,18 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test(ptr %C) {
 ; SSE2-LABEL: 'test'
-; SSE2:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i32 %valB, ptr %out, align 4
 ; SSE2:  Cost of 2.5 for VF 2: profitable to scalarize store i32 %valB, ptr %out, align 4
 ; SSE2:  Cost of 5.5 for VF 4: profitable to scalarize store i32 %valB, ptr %out, align 4
 ; SSE2:  Cost of 11 for VF 8: profitable to scalarize store i32 %valB, ptr %out, align 4
 ; SSE2:  Cost of 22 for VF 16: profitable to scalarize store i32 %valB, ptr %out, align 4
 ;
 ; SSE42-LABEL: 'test'
-; SSE42:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i32 %valB, ptr %out, align 4
 ; SSE42:  Cost of 2 for VF 2: profitable to scalarize store i32 %valB, ptr %out, align 4
 ; SSE42:  Cost of 4 for VF 4: profitable to scalarize store i32 %valB, ptr %out, align 4
 ; SSE42:  Cost of 8 for VF 8: profitable to scalarize store i32 %valB, ptr %out, align 4
 ; SSE42:  Cost of 16 for VF 16: profitable to scalarize store i32 %valB, ptr %out, align 4
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i32 %valB, ptr %out, align 4
 ; AVX1:  Cost of 9 for VF 2: WIDEN store vp<[[VP7:%[0-9]+]]>, ir<%valB>, ir<%canStore>
 ; AVX1:  Cost of 8 for VF 4: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
 ; AVX1:  Cost of 8 for VF 8: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
@@ -38,7 +35,6 @@ define void @test(ptr %C) {
 ; AVX1:  Cost of 32 for VF 32: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
 ;
 ; AVX2-LABEL: 'test'
-; AVX2:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i32 %valB, ptr %out, align 4
 ; AVX2:  Cost of 9 for VF 2: WIDEN store vp<[[VP7:%[0-9]+]]>, ir<%valB>, ir<%canStore>
 ; AVX2:  Cost of 8 for VF 4: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
 ; AVX2:  Cost of 8 for VF 8: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
@@ -46,7 +42,6 @@ define void @test(ptr %C) {
 ; AVX2:  Cost of 32 for VF 32: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
 ;
 ; AVX512-LABEL: 'test'
-; AVX512:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i32 %valB, ptr %out, align 4
 ; AVX512:  Cost of 2 for VF 2: WIDEN store vp<[[VP7:%[0-9]+]]>, ir<%valB>, ir<%canStore>
 ; AVX512:  Cost of 1 for VF 4: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
 ; AVX512:  Cost of 1 for VF 8: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i64.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i64.ll
index ce2d69fca6a3b5..acc0158917568f 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i64.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i64.ll
@@ -16,21 +16,18 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test(ptr %C) {
 ; SSE2-LABEL: 'test'
-; SSE2:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i64 %valB, ptr %out, align 8
 ; SSE2:  Cost of 2.5 for VF 2: profitable to scalarize store i64 %valB, ptr %out, align 8
 ; SSE2:  Cost of 5 for VF 4: profitable to scalarize store i64 %valB, ptr %out, align 8
 ; SSE2:  Cost of 10 for VF 8: profitable to scalarize store i64 %valB, ptr %out, align 8
 ; SSE2:  Cost of 20 for VF 16: profitable to scalarize store i64 %valB, ptr %out, align 8
 ;
 ; SSE42-LABEL: 'test'
-; SSE42:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i64 %valB, ptr %out, align 8
 ; SSE42:  Cost of 2 for VF 2: profitable to scalarize store i64 %valB, ptr %out, align 8
 ; SSE42:  Cost of 4 for VF 4: profitable to scalarize store i64 %valB, ptr %out, align 8
 ; SSE42:  Cost of 8 for VF 8: profitable to scalarize store i64 %valB, ptr %out, align 8
 ; SSE42:  Cost of 16 for VF 16: profitable to scalarize store i64 %valB, ptr %out, align 8
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i64 %valB, ptr %out, align 8
 ; AVX1:  Cost of 8 for VF 2: WIDEN store vp<[[VP7:%[0-9]+]]>, ir<%valB>, ir<%canStore>
 ; AVX1:  Cost of 8 for VF 4: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
 ; AVX1:  Cost of 16 for VF 8: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
@@ -38,7 +35,6 @@ define void @test(ptr %C) {
 ; AVX1:  Cost of 64 for VF 32: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
 ;
 ; AVX2-LABEL: 'test'
-; AVX2:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i64 %valB, ptr %out, align 8
 ; AVX2:  Cost of 8 for VF 2: WIDEN store vp<[[VP7:%[0-9]+]]>, ir<%valB>, ir<%canStore>
 ; AVX2:  Cost of 8 for VF 4: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
 ; AVX2:  Cost of 16 for VF 8: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
@@ -46,7 +42,6 @@ define void @test(ptr %C) {
 ; AVX2:  Cost of 64 for VF 32: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
 ;
 ; AVX512-LABEL: 'test'
-; AVX512:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i64 %valB, ptr %out, align 8
 ; AVX512:  Cost of 1 for VF 2: WIDEN store vp<[[VP7:%[0-9]+]]>, ir<%valB>, ir<%canStore>
 ; AVX512:  Cost of 1 for VF 4: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
 ; AVX512:  Cost of 1 for VF 8: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i8.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i8.ll
index 9028e1c5525a0f..721aa7ba285b18 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i8.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i8.ll
@@ -16,21 +16,18 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test(ptr %C) {
 ; SSE2-LABEL: 'test'
-; SSE2:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i8 %valB, ptr %out, align 1
 ; SSE2:  Cost of 2.5 for VF 2: profitable to scalarize store i8 %valB, ptr %out, align 1
 ; SSE2:  Cost of 5.5 for VF 4: profitable to scalarize store i8 %valB, ptr %out, align 1
 ; SSE2:  Cost of 11.5 for VF 8: profitable to scalarize store i8 %valB, ptr %out, align 1
 ; SSE2:  Cost of 23.5 for VF 16: profitable to scalarize store i8 %valB, ptr %out, align 1
 ;
 ; SSE42-LABEL: 'test'
-; SSE42:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i8 %valB, ptr %out, align 1
 ; SSE42:  Cost of 2 for VF 2: profitable to scalarize store i8 %valB, ptr %out, align 1
 ; SSE42:  Cost of 4 for VF 4: profitable to scalarize store i8 %valB, ptr %out, align 1
 ; SSE42:  Cost of 8 for VF 8: profitable to scalarize store i8 %valB, ptr %out, align 1
 ; SSE42:  Cost of 16 for VF 16: profitable to scalarize store i8 %valB, ptr %out, align 1
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i8 %valB, ptr %out, align 1
 ; AVX1:  Cost of 2 for VF 2: profitable to scalarize store i8 %valB, ptr %out, align 1
 ; AVX1:  Cost of 4 for VF 4: profitable to scalarize store i8 %valB, ptr %out, align 1
 ; AVX1:  Cost of 8 for VF 8: profitable to scalarize store i8 %valB, ptr %out, align 1
@@ -38,7 +35,6 @@ define void @test(ptr %C) {
 ; AVX1:  Cost of 32.5 for VF 32: profitable to scalarize store i8 %valB, ptr %out, align 1
 ;
 ; AVX2-LABEL: 'test'
-; AVX2:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i8 %valB, ptr %out, align 1
 ; AVX2:  Cost of 2 for VF 2: profitable to scalarize store i8 %valB, ptr %out, align 1
 ; AVX2:  Cost of 4 for VF 4: profitable to scalarize store i8 %valB, ptr %out, align 1
 ; AVX2:  Cost of 8 for VF 8: profitable to scalarize store i8 %valB, ptr %out, align 1
@@ -46,7 +42,6 @@ define void @test(ptr %C) {
 ; AVX2:  Cost of 32.5 for VF 32: profitable to scalarize store i8 %valB, ptr %out, align 1
 ;
 ; AVX512-LABEL: 'test'
-; AVX512:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i8 %valB, ptr %out, align 1
 ; AVX512:  Cost of 2 for VF 2: WIDEN store vp<[[VP7:%[0-9]+]]>, ir<%valB>, ir<%canStore>
 ; AVX512:  Cost of 2 for VF 4: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
 ; AVX512:  Cost of 2 for VF 8: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i16-with-i8-index.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i16-with-i8-index.ll
index 6782bbfeb53b31..9daa35ed531659 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i16-with-i8-index.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i16-with-i8-index.ll
@@ -17,21 +17,18 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
-; SSE2:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i16 %valB, ptr %out, align 2
 ; SSE2:  Cost of 28 for VF 2: REPLICATE store ir<%valB>, ir<%out>
 ; SSE2:  Cost of 56 for VF 4: REPLICATE store ir<%valB>, ir<%out>
 ; SSE2:  Cost of 112 for VF 8: REPLICATE store ir<%valB>, ir<%out>
 ; SSE2:  Cost of 224 for VF 16: REPLICATE store ir<%valB>, ir<%out>
 ;
 ; SSE42-LABEL: 'test'
-; SSE42:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i16 %valB, ptr %out, align 2
 ; SSE42:  Cost of 26 for VF 2: REPLICATE store ir<%valB>, ir<%out>
 ; SSE42:  Cost of 52 for VF 4: REPLICATE store ir<%valB>, ir<%out>
 ; SSE42:  Cost of 104 for VF 8: REPLICATE store ir<%valB>, ir<%out>
 ; SSE42:  Cost of 208 for VF 16: REPLICATE store ir<%valB>, ir<%out>
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i16 %valB, ptr %out, align 2
 ; AVX1:  Cost of 26 for VF 2: REPLICATE store ir<%valB>, ir<%out>
 ; AVX1:  Cost of 53 for VF 4: REPLICATE store ir<%valB>, ir<%out>
 ; AVX1:  Cost of 106 for VF 8: REPLICATE store ir<%valB>, ir<%out>
@@ -39,7 +36,6 @@ define void @test() {
 ; AVX1:  Cost of 426 for VF 32: REPLICATE store ir<%valB>, ir<%out>
 ;
 ; AVX2-LABEL: 'test'
-; AVX2:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i16 %valB, ptr %out, align 2
 ; AVX2:  Cost of 6 for VF 2: REPLICATE store ir<%valB>, ir<%out>
 ; AVX2:  Cost of 13 for VF 4: REPLICATE store ir<%valB>, ir<%out>
 ; AVX2:  Cost of 26 for VF 8: REPLICATE store ir<%valB>, ir<%out>
@@ -47,7 +43,6 @@ define void @test() {
 ; AVX2:  Cost of 106 for VF 32: REPLICATE store ir<%valB>, ir<%out>
 ;
 ; AVX512-LABEL: 'test'
-; AVX512:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i16 %valB, ptr %out, align 2
 ; AVX512:  Cost of 6 for VF 2: REPLICATE store ir<%valB>, ir<%out>
 ; AVX512:  Cost of 13 for VF 4: REPLICATE store ir<%valB>, ir<%out>
 ; AVX512:  Cost of 27 for VF 8: REPLICATE store ir<%valB>, ir<%out>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i32-with-i8-index.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i32-with-i8-index.ll
index 5e0b3277dd5e9e..671842c2645cf2 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i32-with-i8-index.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i32-with-i8-index.ll
@@ -17,21 +17,18 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
-; SSE2:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i32 %valB, ptr %out, align 4
 ; SSE2:  Cost of 29 for VF 2: REPLICATE store ir<%valB>, ir<%out>
 ; SSE2:  Cost of 59 for VF 4: REPLICATE store ir<%valB>, ir<%out>
 ; SSE2:  Cost of 118 for VF 8: REPLICATE store ir<%valB>, ir<%out>
 ; SSE2:  Cost of 236 for VF 16: REPLICATE store ir<%valB>, ir<%out>
 ;
 ; SSE42-LABEL: 'test'
-; SSE42:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i32 %valB, ptr %out, align 4
 ; SSE42:  Cost of 26 for VF 2: REPLICATE store ir<%valB>, ir<%out>
 ; SSE42:  Cost of 52 for VF 4: REPLICATE store ir<%valB>, ir<%out>
 ; SSE42:  Cost of 104 for VF 8: REPLICATE store ir<%valB>, ir<%out>
 ; SSE42:  Cost of 208 for VF 16: REPLICATE store ir<%valB>, ir<%out>
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i32 %valB, ptr %out, align 4
 ; AVX1:  Cost of 26 for VF 2: REPLICATE store ir<%valB>, ir<%out>
 ; AVX1:  Cost of 53 for VF 4: REPLICATE store ir<%valB>, ir<%out>
 ; AVX1:  Cost of 107 for VF 8: REPLICATE store ir<%valB>, ir<%out>
@@ -39,7 +36,6 @@ define void @test() {
 ; AVX1:  Cost of 428 for VF 32: REPLICATE store ir<%valB>, ir<%out>
 ;
 ; AVX2-LABEL: 'test'
-; AVX2:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i32 %valB, ptr %out, align 4
 ; AVX2:  Cost of 6 for VF 2: REPLICATE store ir<%valB>, ir<%out>
 ; AVX2:  Cost of 13 for VF 4: REPLICATE store ir<%valB>, ir<%out>
 ; AVX2:  Cost of 27 for VF 8: REPLICATE store ir<%valB>, ir<%out>
@@ -47,7 +43,6 @@ define void @test() {
 ; AVX2:  Cost of 108 for VF 32: REPLICATE store ir<%valB>, ir<%out>
 ;
 ; AVX512-LABEL: 'test'
-; AVX512:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i32 %valB, ptr %out, align 4
 ; AVX512:  Cost of 6 for VF 2: REPLICATE store ir<%valB>, ir<%out>
 ; AVX512:  Cost of 13 for VF 4: REPLICATE store ir<%valB>, ir<%out>
 ; AVX512:  Cost of 10 for VF 8: WIDEN store ir<%out>, ir<%valB>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i64-with-i8-index.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i64-with-i8-index.ll
index fcbf6042dec148..dc775779060122 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i64-with-i8-index.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i64-with-i8-index.ll
@@ -17,21 +17,18 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
-; SSE2:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i64 %valB, ptr %out, align 8
 ; SSE2:  Cost of 29 for VF 2: REPLICATE store ir<%valB>, ir<%out>
 ; SSE2:  Cost of 58 for VF 4: REPLICATE store ir<%valB>, ir<%out>
 ; SSE2:  Cost of 116 for VF 8: REPLICATE store ir<%valB>, ir<%out>
 ; SSE2:  Cost of 232 for VF 16: REPLICATE store ir<%valB>, ir<%out>
 ;
 ; SSE42-LABEL: 'test'
-; SSE42:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i64 %valB, ptr %out, align 8
 ; SSE42:  Cost of 26 for VF 2: REPLICATE store ir<%valB>, ir<%out>
 ; SSE42:  Cost of 52 for VF 4: REPLICATE store ir<%valB>, ir<%out>
 ; SSE42:  Cost of 104 for VF 8: REPLICATE store ir<%valB>, ir<%out>
 ; SSE42:  Cost of 208 for VF 16: REPLICATE store ir<%valB>, ir<%out>
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i64 %valB, ptr %out, align 8
 ; AVX1:  Cost of 26 for VF 2: REPLICATE store ir<%valB>, ir<%out>
 ; AVX1:  Cost of 54 for VF 4: REPLICATE store ir<%valB>, ir<%out>
 ; AVX1:  Cost of 108 for VF 8: REPLICATE store ir<%valB>, ir<%out>
@@ -39,7 +36,6 @@ define void @test() {
 ; AVX1:  Cost of 432 for VF 32: REPLICATE store ir<%valB>, ir<%out>
 ;
 ; AVX2-LABEL: 'test'
-; AVX2:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i64 %valB, ptr %out, align 8
 ; AVX2:  Cost of 6 for VF 2: REPLICATE store ir<%valB>, ir<%out>
 ; AVX2:  Cost of 14 for VF 4: REPLICATE store ir<%valB>, ir<%out>
 ; AVX2:  Cost of 28 for VF 8: REPLICATE store ir<%valB>, ir<%out>
@@ -47,7 +43,6 @@ define void @test() {
 ; AVX2:  Cost of 112 for VF 32: REPLICATE store ir<%valB>, ir<%out>
 ;
 ; AVX512-LABEL: 'test'
-; AVX512:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i64 %valB, ptr %out, align 8
 ; AVX512:  Cost of 6 for VF 2: REPLICATE store ir<%valB>, ir<%out>
 ; AVX512:  Cost of 14 for VF 4: REPLICATE store ir<%valB>, ir<%out>
 ; AVX512:  Cost of 10 for VF 8: WIDEN store ir<%out>, ir<%valB>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i8-with-i8-index.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i8-with-i8-index.ll
index 2946cd291d7fa1..c484b526ac1cb3 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i8-with-i8-index.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i8-with-i8-index.ll
@@ -17,21 +17,18 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
-; SSE2:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i8 %valB, ptr %out, align 1
 ; SSE2:  Cost of 29 for VF 2: REPLICATE store ir<%valB>, ir<%out>
 ; SSE2:  Cost of 59 for VF 4: REPLICATE store ir<%valB>, ir<%out>
 ; SSE2:  Cost of 119 for VF 8: REPLICATE store ir<%valB>, ir<%out>
 ; SSE2:  Cost of 239 for VF 16: REPLICATE store ir<%valB>, ir<%out>
 ;
 ; SSE42-LABEL: 'test'
-; SSE42:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i8 %valB, ptr %out, align 1
 ; SSE42:  Cost of 26 for VF 2: REPLICATE store ir<%valB>, ir<%out>
 ; SSE42:  Cost of 52 for VF 4: REPLICATE store ir<%valB>, ir<%out>
 ; SSE42:  Cost of 104 for VF 8: REPLICATE store ir<%valB>, ir<%out>
 ; SSE42:  Cost of 208 for VF 16: REPLICATE store ir<%valB>, ir<%out>
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i8 %valB, ptr %out, align 1
 ; AVX1:  Cost of 26 for VF 2: REPLICATE store ir<%valB>, ir<%out>
 ; AVX1:  Cost of 53 for VF 4: REPLICATE store ir<%valB>, ir<%out>
 ; AVX1:  Cost of 106 for VF 8: REPLICATE store ir<%valB>, ir<%out>
@@ -39,7 +36,6 @@ define void @test() {
 ; AVX1:  Cost of 425 for VF 32: REPLICATE store ir<%valB>, ir<%out>
 ;
 ; AVX2-LABEL: 'test'
-; AVX2:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i8 %valB, ptr %out, align 1
 ; AVX2:  Cost of 6 for VF 2: REPLICATE store ir<%valB>, ir<%out>
 ; AVX2:  Cost of 13 for VF 4: REPLICATE store ir<%valB>, ir<%out>
 ; AVX2:  Cost of 26 for VF 8: REPLICATE store ir<%valB>, ir<%out>
@@ -47,7 +43,6 @@ define void @test() {
 ; AVX2:  Cost of 105 for VF 32: REPLICATE store ir<%valB>, ir<%out>
 ;
 ; AVX512-LABEL: 'test'
-; AVX512:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i8 %valB, ptr %out, align 1
 ; AVX512:  Cost of 6 for VF 2: REPLICATE store ir<%valB>, ir<%out>
 ; AVX512:  Cost of 13 for VF 4: REPLICATE store ir<%valB>, ir<%out>
 ; AVX512:  Cost of 27 for VF 8: REPLICATE store ir<%valB>, ir<%out>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/strided-load-i16.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/strided-load-i16.ll
index b77b8ac294163d..3c27644f4ebcf0 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/strided-load-i16.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/strided-load-i16.ll
@@ -10,7 +10,6 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @load_i16_stride2() {
 ; CHECK-LABEL: 'load_i16_stride2'
-; CHECK:  LV: Found an estimated cost of 1 for VF 1 For instruction: %1 = load i16, ptr %arrayidx, align 4
 ; CHECK:  Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%arrayidx>
 ; CHECK:  Cost of 1 for VF 4: INTERLEAVE-GROUP with factor 2, ir<%arrayidx>
 ; CHECK:  Cost of 2 for VF 8: INTERLEAVE-GROUP with factor 2, ir<%arrayidx>
@@ -37,7 +36,6 @@ for.end:
 
 define void @load_i16_stride3() {
 ; CHECK-LABEL: 'load_i16_stride3'
-; CHECK:  LV: Found an estimated cost of 1 for VF 1 For instruction: %1 = load i16, ptr %arrayidx, align 4
 ; CHECK:  Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%arrayidx>
 ; CHECK:  Cost of 2 for VF 4: INTERLEAVE-GROUP with factor 3, ir<%arrayidx>
 ; CHECK:  Cost of 2 for VF 8: INTERLEAVE-GROUP with factor 3, ir<%arrayidx>
@@ -64,7 +62,6 @@ for.end:
 
 define void @load_i16_stride4() {
 ; CHECK-LABEL: 'load_i16_stride4'
-; CHECK:  LV: Found an estimated cost of 1 for VF 1 For instruction: %1 = load i16, ptr %arrayidx, align 4
 ; CHECK:  Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%arrayidx>
 ; CHECK:  Cost of 2 for VF 4: INTERLEAVE-GROUP with factor 4, ir<%arrayidx>
 ; CHECK:  Cost of 2 for VF 8: INTERLEAVE-GROUP with factor 4, ir<%arrayidx>
@@ -91,7 +88,6 @@ for.end:
 
 define void @load_i16_stride5() {
 ; CHECK-LABEL: 'load_i16_stride5'
-; CHECK:  LV: Found an estimated cost of 1 for VF 1 For instruction: %1 = load i16, ptr %arrayidx, align 4
 ; CHECK:  Cost of 2 for VF 2: INTERLEAVE-GROUP with factor 5, ir<%arrayidx>
 ; CHECK:  Cost of 2 for VF 4: INTERLEAVE-GROUP with factor 5, ir<%arrayidx>
 ; CHECK:  Cost of 3 for VF 8: INTERLEAVE-GROUP with factor 5, ir<%arrayidx>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/strided-load-i32.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/strided-load-i32.ll
index da9722b65ff06a..8bc141496b822b 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/strided-load-i32.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/strided-load-i32.ll
@@ -10,7 +10,6 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @load_int_stride2() {
 ; CHECK-LABEL: 'load_int_stride2'
-; CHECK:  LV: Found an estimated cost of 1 for VF 1 For instruction: %1 = load i32, ptr %arrayidx, align 4
 ; CHECK:  Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%arrayidx>
 ; CHECK:  Cost of 1 for VF 4: INTERLEAVE-GROUP with factor 2, ir<%arrayidx>
 ; CHECK:  Cost of 1 for VF 8: INTERLEAVE-GROUP with factor 2, ir<%arrayidx>
@@ -36,7 +35,6 @@ for.end:
 
 define void @load_int_stride3() {
 ; CHECK-LABEL: 'load_int_stride3'
-; CHECK:  LV: Found an estimated cost of 1 for VF 1 For instruction: %1 = load i32, ptr %arrayidx, align 4
 ; CHECK:  Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%arrayidx>
 ; CHECK:  Cost of 1 for VF 4: INTERLEAVE-GROUP with factor 3, ir<%arrayidx>
 ; CHECK:  Cost of 3 for VF 8: INTERLEAVE-GROUP with factor 3, ir<%arrayidx>
@@ -62,7 +60,6 @@ for.end:
 
 define void @load_int_stride4() {
 ; CHECK-LABEL: 'load_int_stride4'
-; CHECK:  LV: Found an estimated cost of 1 for VF 1 For instruction: %1 = load i32, ptr %arrayidx, align 4
 ; CHECK:  Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%arrayidx>
 ; CHECK:  Cost of 1 for VF 4: INTERLEAVE-GROUP with factor 4, ir<%arrayidx>
 ; CHECK:  Cost of 3 for VF 8: INTERLEAVE-GROUP with factor 4, ir<%arrayidx>
@@ -88,7 +85,6 @@ for.end:
 
 define void @load_int_stride5() {
 ; CHECK-LABEL: 'load_int_stride5'
-; CHECK:  LV: Found an estimated cost of 1 for VF 1 For instruction: %1 = load i32, ptr %arrayidx, align 4
 ; CHECK:  Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 5, ir<%arrayidx>
 ; CHECK:  Cost of 3 for VF 4: INTERLEAVE-GROUP with factor 5, ir<%arrayidx>
 ; CHECK:  Cost of 5 for VF 8: INTERLEAVE-GROUP with factor 5, ir<%arrayidx>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/strided-load-i64.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/strided-load-i64.ll
index 49a5fc47944a8d..1c91d01340a4b9 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/strided-load-i64.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/strided-load-i64.ll
@@ -10,7 +10,6 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @load_i64_stride2() {
 ; CHECK-LABEL: 'load_i64_stride2'
-; CHECK:  LV: Found an estimated cost of 1 for VF 1 For instruction: %1 = load i64, ptr %arrayidx, align 16
 ; CHECK:  Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%arrayidx>
 ; CHECK:  Cost of 1 for VF 4: INTERLEAVE-GROUP with factor 2, ir<%arrayidx>
 ; CHECK:  Cost of 3 for VF 8: INTERLEAVE-GROUP with factor 2, ir<%arrayidx>
@@ -35,7 +34,6 @@ for.end:
 
 define void @load_i64_stride3() {
 ; CHECK-LABEL: 'load_i64_stride3'
-; CHECK:  LV: Found an estimated cost of 1 for VF 1 For instruction: %1 = load i64, ptr %arrayidx, align 16
 ; CHECK:  Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%arrayidx>
 ; CHECK:  Cost of 3 for VF 4: INTERLEAVE-GROUP with factor 3, ir<%arrayidx>
 ; CHECK:  Cost of 5 for VF 8: INTERLEAVE-GROUP with factor 3, ir<%arrayidx>
@@ -60,7 +58,6 @@ for.end:
 
 define void @load_i64_stride4() {
 ; CHECK-LABEL: 'load_i64_stride4'
-; CHECK:  LV: Found an estimated cost of 1 for VF 1 For instruction: %1 = load i64, ptr %arrayidx, align 16
 ; CHECK:  Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%arrayidx>
 ; CHECK:  Cost of 3 for VF 4: INTERLEAVE-GROUP with factor 4, ir<%arrayidx>
 ; CHECK:  Cost of 8 for VF 8: INTERLEAVE-GROUP with factor 4, ir<%arrayidx>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/strided-load-i8.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/strided-load-i8.ll
index 9ac785187cc1aa..b87607872fdccf 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/strided-load-i8.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/strided-load-i8.ll
@@ -10,7 +10,6 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @load_i8_stride2() {
 ; CHECK-LABEL: 'load_i8_stride2'
-; CHECK:  LV: Found an estimated cost of 1 for VF 1 For instruction: %1 = load i8, ptr %arrayidx, align 2
 ; CHECK:  Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%arrayidx>
 ; CHECK:  Cost of 1 for VF 4: INTERLEAVE-GROUP with factor 2, ir<%arrayidx>
 ; CHECK:  Cost of 1 for VF 8: INTERLEAVE-GROUP with factor 2, ir<%arrayidx>
@@ -38,7 +37,6 @@ for.end:
 
 define void @load_i8_stride3() {
 ; CHECK-LABEL: 'load_i8_stride3'
-; CHECK:  LV: Found an estimated cost of 1 for VF 1 For instruction: %1 = load i8, ptr %arrayidx, align 2
 ; CHECK:  Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%arrayidx>
 ; CHECK:  Cost of 1 for VF 4: INTERLEAVE-GROUP with factor 3, ir<%arrayidx>
 ; CHECK:  Cost of 4 for VF 8: INTERLEAVE-GROUP with factor 3, ir<%arrayidx>
@@ -66,7 +64,6 @@ for.end:
 
 define void @load_i8_stride4() {
 ; CHECK-LABEL: 'load_i8_stride4'
-; CHECK:  LV: Found an estimated cost of 1 for VF 1 For instruction: %1 = load i8, ptr %arrayidx, align 2
 ; CHECK:  Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%arrayidx>
 ; CHECK:  Cost of 1 for VF 4: INTERLEAVE-GROUP with factor 4, ir<%arrayidx>
 ; CHECK:  Cost of 4 for VF 8: INTERLEAVE-GROUP with factor 4, ir<%arrayidx>
@@ -94,7 +91,6 @@ for.end:
 
 define void @load_i8_stride5() {
 ; CHECK-LABEL: 'load_i8_stride5'
-; CHECK:  LV: Found an estimated cost of 1 for VF 1 For instruction: %1 = load i8, ptr %arrayidx, align 2
 ; CHECK:  Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 5, ir<%arrayidx>
 ; CHECK:  Cost of 4 for VF 4: INTERLEAVE-GROUP with factor 5, ir<%arrayidx>
 ; CHECK:  Cost of 8 for VF 8: INTERLEAVE-GROUP with factor 5, ir<%arrayidx>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/vpinstruction-cost.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/vpinstruction-cost.ll
index be1b6b42496b1a..3e9ae12df16cba 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/vpinstruction-cost.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/vpinstruction-cost.ll
@@ -7,19 +7,18 @@ target datalayout = "e-m:e-p270:32:32-p271:32:32-p272:64:64-i64:64-i128:128-f80:
 
 define void @wide_or_replaced_with_add_vpinstruction(ptr %src, ptr noalias %dst) {
 ; CHECK-LABEL: 'wide_or_replaced_with_add_vpinstruction'
-; CHECK:  LV: Found an estimated cost of 0 for VF 1 For instruction: %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop.latch ]
-; CHECK:  LV: Found an estimated cost of 0 for VF 1 For instruction: %g.src = getelementptr inbounds i64, ptr %src, i64 %iv
-; CHECK:  LV: Found an estimated cost of 1 for VF 1 For instruction: %l = load i64, ptr %g.src, align 8
-; CHECK:  LV: Found an estimated cost of 1 for VF 1 For instruction: %iv.4 = add nuw nsw i64 %iv, 4
-; CHECK:  LV: Found an estimated cost of 1 for VF 1 For instruction: %c = icmp ule i64 %l, 128
-; CHECK:  LV: Found an estimated cost of 0 for VF 1 For instruction: br i1 %c, label %loop.then, label %loop.latch
-; CHECK:  LV: Found an estimated cost of 1 for VF 1 For instruction: %or = or disjoint i64 %iv.4, 1
-; CHECK:  LV: Found an estimated cost of 0 for VF 1 For instruction: %g.dst = getelementptr inbounds i64, ptr %dst, i64 %or
-; CHECK:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i64 %iv.4, ptr %g.dst, align 4
-; CHECK:  LV: Found an estimated cost of 0 for VF 1 For instruction: br label %loop.latch
-; CHECK:  LV: Found an estimated cost of 1 for VF 1 For instruction: %iv.next = add nuw nsw i64 %iv, 1
-; CHECK:  LV: Found an estimated cost of 1 for VF 1 For instruction: %exitcond = icmp eq i64 %iv.next, 32
-; CHECK:  LV: Found an estimated cost of 0 for VF 1 For instruction: br i1 %exitcond, label %exit, label %loop.header
+; CHECK:  Cost of 0 for VF 1: EMIT-SCALAR ir<%iv> = phi [ ir<0>, vector.ph ], [ ir<%iv.next>, loop.latch ]
+; CHECK:  Cost of 0 for VF 1: EMIT ir<%g.src> = getelementptr inbounds ir<%src>, ir<%iv>
+; CHECK:  Cost of 1 for VF 1: EMIT-SCALAR ir<%l> = load ir<%g.src>
+; CHECK:  Cost of 1 for VF 1: EMIT ir<%iv.4> = add nuw nsw ir<%iv>, ir<4>
+; CHECK:  Cost of 1 for VF 1: EMIT ir<%c> = icmp ule ir<%l>, ir<128>
+; CHECK:  Cost of 0 for VF 1: EMIT branch-on-cond ir<%c> (!vplan.prof.estimated estimated {1073741824, 1073741824})
+; CHECK:  Cost of 1 for VF 1: EMIT ir<%or> = or disjoint ir<%iv.4>, ir<1> (!vplan.execution.frequency 4611686018427387904 (50%, estimated))
+; CHECK:  Cost of 0 for VF 1: EMIT ir<%g.dst> = getelementptr inbounds ir<%dst>, ir<%or> (!vplan.execution.frequency 4611686018427387904 (50%, estimated))
+; CHECK:  Cost of 1 for VF 1: EMIT store ir<%iv.4>, ir<%g.dst> (!vplan.execution.frequency 4611686018427387904 (50%, estimated))
+; CHECK:  Cost of 1 for VF 1: EMIT ir<%iv.next> = add nuw nsw ir<%iv>, ir<1>
+; CHECK:  Cost of 1 for VF 1: EMIT ir<%exitcond> = icmp eq ir<%iv.next>, ir<32>
+; CHECK:  Cost of 0 for VF 1: EMIT branch-on-cond ir<%exitcond>
 ; CHECK:  Cost of 1 for VF 2: ir<%iv> = WIDEN-INDUCTION nuw nsw ir<0>, ir<1>, vp<[[VP0:%[0-9]+]]>
 ; CHECK:  Cost of 0 for VF 2: vp<[[VP4:%[0-9]+]]> = SCALAR-STEPS vp<[[VP3:%[0-9]+]]>, ir<1>, vp<[[VP0]]>
 ; CHECK:  Cost of 0 for VF 2: CLONE ir<%g.src> = getelementptr inbounds ir<%src>, vp<[[VP4]]>
@@ -30,7 +29,7 @@ define void @wide_or_replaced_with_add_vpinstruction(ptr %src, ptr noalias %dst)
 ; CHECK:  Cost of 1 for VF 2: EMIT ir<%or> = add ir<%iv.4>, ir<1>
 ; CHECK:  Cost of 0 for VF 2: CLONE ir<%g.dst> = getelementptr ir<%dst>, ir<%or>
 ; CHECK:  Cost of 0 for VF 2: vp<[[VP6:%[0-9]+]]> = vector-pointer i64, ir<%g.dst>, ir<1>
-; CHECK:  Cost of 1 for VF 2: WIDEN store vp<[[VP6]]>, ir<%iv.4>, ir<%c>
+; CHECK:  Cost of 1 for VF 2: WIDEN store vp<[[VP6]]>, ir<%iv.4>, ir<%c> (!vplan.execution.frequency 4611686018427387904 (50%, estimated))
 ; CHECK:  Cost of 0 for VF 2: EMIT vp<%index.next> = add nuw vp<[[VP3]]>, vp<[[VP1:%[0-9]+]]>
 ; CHECK:  Cost of 1 for VF 2: EMIT branch-on-count vp<%index.next>, vp<[[VP2:%[0-9]+]]>
 ; CHECK:  Cost of 0 for VF 2: vector loop backedge
@@ -53,7 +52,7 @@ define void @wide_or_replaced_with_add_vpinstruction(ptr %src, ptr noalias %dst)
 ; CHECK:  Cost of 1 for VF 4: EMIT ir<%or> = add ir<%iv.4>, ir<1>
 ; CHECK:  Cost of 0 for VF 4: CLONE ir<%g.dst> = getelementptr ir<%dst>, ir<%or>
 ; CHECK:  Cost of 0 for VF 4: vp<[[VP6]]> = vector-pointer i64, ir<%g.dst>, ir<1>
-; CHECK:  Cost of 1 for VF 4: WIDEN store vp<[[VP6]]>, ir<%iv.4>, ir<%c>
+; CHECK:  Cost of 1 for VF 4: WIDEN store vp<[[VP6]]>, ir<%iv.4>, ir<%c> (!vplan.execution.frequency 4611686018427387904 (50%, estimated))
 ; CHECK:  Cost of 0 for VF 4: EMIT vp<%index.next> = add nuw vp<[[VP3]]>, vp<[[VP1]]>
 ; CHECK:  Cost of 1 for VF 4: EMIT branch-on-count vp<%index.next>, vp<[[VP2]]>
 ; CHECK:  Cost of 0 for VF 4: vector loop backedge
@@ -97,15 +96,15 @@ exit:
 
 define void @test_vpinstruction_freeze_cost(ptr %src, ptr noalias %dst) {
 ; CHECK-LABEL: 'test_vpinstruction_freeze_cost'
-; CHECK:  LV: Found an estimated cost of 0 for VF 1 For instruction: %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
-; CHECK:  LV: Found an estimated cost of 0 for VF 1 For instruction: %g.src = getelementptr inbounds i64, ptr %src, i64 %iv
-; CHECK:  LV: Found an estimated cost of 1 for VF 1 For instruction: %l = load i64, ptr %g.src, align 8
-; CHECK:  LV: Found an estimated cost of 0 for VF 1 For instruction: %fr = freeze i64 %l
-; CHECK:  LV: Found an estimated cost of 0 for VF 1 For instruction: %g.dst = getelementptr inbounds i64, ptr %dst, i64 %iv
-; CHECK:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i64 %fr, ptr %g.dst, align 8
-; CHECK:  LV: Found an estimated cost of 1 for VF 1 For instruction: %iv.next = add nuw nsw i64 %iv, 1
-; CHECK:  LV: Found an estimated cost of 1 for VF 1 For instruction: %ec = icmp eq i64 %iv.next, 32
-; CHECK:  LV: Found an estimated cost of 0 for VF 1 For instruction: br i1 %ec, label %exit, label %loop
+; CHECK:  Cost of 0 for VF 1: EMIT-SCALAR ir<%iv> = phi [ ir<0>, vector.ph ], [ ir<%iv.next>, loop ]
+; CHECK:  Cost of 0 for VF 1: EMIT ir<%g.src> = getelementptr inbounds ir<%src>, ir<%iv>
+; CHECK:  Cost of 1 for VF 1: EMIT-SCALAR ir<%l> = load ir<%g.src>
+; CHECK:  Cost of 0 for VF 1: EMIT ir<%fr> = freeze ir<%l>
+; CHECK:  Cost of 0 for VF 1: EMIT ir<%g.dst> = getelementptr inbounds ir<%dst>, ir<%iv>
+; CHECK:  Cost of 1 for VF 1: EMIT store ir<%fr>, ir<%g.dst>
+; CHECK:  Cost of 1 for VF 1: EMIT ir<%iv.next> = add nuw nsw ir<%iv>, ir<1>
+; CHECK:  Cost of 1 for VF 1: EMIT ir<%ec> = icmp eq ir<%iv.next>, ir<32>
+; CHECK:  Cost of 0 for VF 1: EMIT branch-on-cond ir<%ec>
 ; CHECK:  Cost of 0 for VF 2: vp<[[VP4:%[0-9]+]]> = SCALAR-STEPS vp<[[VP3:%[0-9]+]]>, ir<1>, vp<[[VP0:%[0-9]+]]>
 ; CHECK:  Cost of 0 for VF 2: CLONE ir<%g.src> = getelementptr inbounds ir<%src>, vp<[[VP4]]>
 ; CHECK:  Cost of 0 for VF 2: vp<[[VP5:%[0-9]+]]> = vector-pointer inbounds i64, ir<%g.src>, ir<1>
@@ -175,20 +174,16 @@ exit:
 
 define void @test_vpinstruction_switch_cost(ptr %start, ptr %end) {
 ; CHECK-LABEL: 'test_vpinstruction_switch_cost'
-; CHECK:  LV: Found an estimated cost of 0 for VF 1 For instruction: %ptr.iv = phi ptr [ %start, %entry ], [ %ptr.iv.next, %loop.latch ]
-; CHECK:  LV: Found an estimated cost of 1 for VF 1 For instruction: %l = load i64, ptr %ptr.iv, align 1
-; CHECK:  LV: Found an estimated cost of 0 for VF 1 For instruction: switch i64 %l, label %default [
-; CHECK:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i64 1, ptr %ptr.iv, align 1
-; CHECK:  LV: Found an estimated cost of 0 for VF 1 For instruction: br label %loop.latch
-; CHECK:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i64 0, ptr %ptr.iv, align 1
-; CHECK:  LV: Found an estimated cost of 0 for VF 1 For instruction: br label %loop.latch
-; CHECK:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i64 42, ptr %ptr.iv, align 1
-; CHECK:  LV: Found an estimated cost of 0 for VF 1 For instruction: br label %loop.latch
-; CHECK:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i64 2, ptr %ptr.iv, align 1
-; CHECK:  LV: Found an estimated cost of 0 for VF 1 For instruction: br label %loop.latch
-; CHECK:  LV: Found an estimated cost of 0 for VF 1 For instruction: %ptr.iv.next = getelementptr inbounds i64, ptr %ptr.iv, i64 1
-; CHECK:  LV: Found an estimated cost of 1 for VF 1 For instruction: %ec = icmp eq ptr %ptr.iv.next, %end
-; CHECK:  LV: Found an estimated cost of 0 for VF 1 For instruction: br i1 %ec, label %exit, label %loop.header
+; CHECK:  Cost of 0 for VF 1: EMIT-SCALAR ir<%ptr.iv> = phi [ ir<%start>, vector.ph ], [ ir<%ptr.iv.next>, loop.latch ]
+; CHECK:  Cost of 1 for VF 1: EMIT-SCALAR ir<%l> = load ir<%ptr.iv>
+; CHECK:  Cost of 0 for VF 1: EMIT switch ir<%l>, ir<-12>, ir<13>, ir<0> (!vplan.prof.estimated estimated {536870912, 536870912, 536870912, 536870912})
+; CHECK:  Cost of 1 for VF 1: EMIT store ir<1>, ir<%ptr.iv> (!vplan.execution.frequency 2305843009213693952 (25%, estimated))
+; CHECK:  Cost of 1 for VF 1: EMIT store ir<0>, ir<%ptr.iv> (!vplan.execution.frequency 2305843009213693952 (25%, estimated))
+; CHECK:  Cost of 1 for VF 1: EMIT store ir<42>, ir<%ptr.iv> (!vplan.execution.frequency 2305843009213693952 (25%, estimated))
+; CHECK:  Cost of 1 for VF 1: EMIT store ir<2>, ir<%ptr.iv> (!vplan.execution.frequency 2305843009213693952 (25%, estimated))
+; CHECK:  Cost of 0 for VF 1: EMIT ir<%ptr.iv.next> = getelementptr inbounds ir<%ptr.iv>, ir<1>
+; CHECK:  Cost of 1 for VF 1: EMIT ir<%ec> = icmp eq ir<%ptr.iv.next>, ir<%end>
+; CHECK:  Cost of 0 for VF 1: EMIT branch-on-cond ir<%ec>
 ; CHECK:  Cost of 1 for VF 2: vp<[[VP6:%[0-9]+]]> = DERIVED-IV ir<0> + vp<[[VP5:%[0-9]+]]> * ir<8>
 ; CHECK:  Cost of 0 for VF 2: vp<[[VP7:%[0-9]+]]> = SCALAR-STEPS vp<[[VP6]]>, ir<8>, vp<[[VP0:%[0-9]+]]>
 ; CHECK:  Cost of 0 for VF 2: EMIT vp<%next.gep> = ptradd ir<%start>, vp<[[VP7]]>
@@ -201,13 +196,13 @@ define void @test_vpinstruction_switch_cost(ptr %start, ptr %end) {
 ; CHECK:  Cost of 0 for VF 2: EMIT vp<[[VP13:%[0-9]+]]> = or vp<[[VP12]]>, vp<[[VP11]]>
 ; CHECK:  Cost of 1 for VF 2: EMIT vp<[[VP14:%[0-9]+]]> = not vp<[[VP13]]>
 ; CHECK:  Cost of 0 for VF 2: vp<[[VP15:%[0-9]+]]> = vector-pointer i64, vp<%next.gep>, ir<1>
-; CHECK:  Cost of 1 for VF 2: WIDEN store vp<[[VP15]]>, ir<1>, vp<[[VP11]]>
+; CHECK:  Cost of 1 for VF 2: WIDEN store vp<[[VP15]]>, ir<1>, vp<[[VP11]]> (!vplan.execution.frequency 2305843009213693952 (25%, estimated))
 ; CHECK:  Cost of 0 for VF 2: vp<[[VP16:%[0-9]+]]> = vector-pointer i64, vp<%next.gep>, ir<1>
-; CHECK:  Cost of 1 for VF 2: WIDEN store vp<[[VP16]]>, ir<0>, vp<[[VP10]]>
+; CHECK:  Cost of 1 for VF 2: WIDEN store vp<[[VP16]]>, ir<0>, vp<[[VP10]]> (!vplan.execution.frequency 2305843009213693952 (25%, estimated))
 ; CHECK:  Cost of 0 for VF 2: vp<[[VP17:%[0-9]+]]> = vector-pointer i64, vp<%next.gep>, ir<1>
-; CHECK:  Cost of 1 for VF 2: WIDEN store vp<[[VP17]]>, ir<42>, vp<[[VP9]]>
+; CHECK:  Cost of 1 for VF 2: WIDEN store vp<[[VP17]]>, ir<42>, vp<[[VP9]]> (!vplan.execution.frequency 2305843009213693952 (25%, estimated))
 ; CHECK:  Cost of 0 for VF 2: vp<[[VP18:%[0-9]+]]> = vector-pointer i64, vp<%next.gep>, ir<1>
-; CHECK:  Cost of 1 for VF 2: WIDEN store vp<[[VP18]]>, ir<2>, vp<[[VP14]]>
+; CHECK:  Cost of 1 for VF 2: WIDEN store vp<[[VP18]]>, ir<2>, vp<[[VP14]]> (!vplan.execution.frequency 2305843009213693952 (25%, estimated))
 ; CHECK:  Cost of 0 for VF 2: EMIT vp<%index.next> = add nuw vp<[[VP5]]>, vp<[[VP1:%[0-9]+]]>
 ; CHECK:  Cost of 1 for VF 2: EMIT branch-on-count vp<%index.next>, vp<[[VP2:%[0-9]+]]>
 ; CHECK:  Cost of 0 for VF 2: vector loop backedge
@@ -231,13 +226,13 @@ define void @test_vpinstruction_switch_cost(ptr %start, ptr %end) {
 ; CHECK:  Cost of 0 for VF 4: EMIT vp<[[VP13]]> = or vp<[[VP12]]>, vp<[[VP11]]>
 ; CHECK:  Cost of 1 for VF 4: EMIT vp<[[VP14]]> = not vp<[[VP13]]>
 ; CHECK:  Cost of 0 for VF 4: vp<[[VP15]]> = vector-pointer i64, vp<%next.gep>, ir<1>
-; CHECK:  Cost of 1 for VF 4: WIDEN store vp<[[VP15]]>, ir<1>, vp<[[VP11]]>
+; CHECK:  Cost of 1 for VF 4: WIDEN store vp<[[VP15]]>, ir<1>, vp<[[VP11]]> (!vplan.execution.frequency 2305843009213693952 (25%, estimated))
 ; CHECK:  Cost of 0 for VF 4: vp<[[VP16]]> = vector-pointer i64, vp<%next.gep>, ir<1>
-; CHECK:  Cost of 1 for VF 4: WIDEN store vp<[[VP16]]>, ir<0>, vp<[[VP10]]>
+; CHECK:  Cost of 1 for VF 4: WIDEN store vp<[[VP16]]>, ir<0>, vp<[[VP10]]> (!vplan.execution.frequency 2305843009213693952 (25%, estimated))
 ; CHECK:  Cost of 0 for VF 4: vp<[[VP17]]> = vector-pointer i64, vp<%next.gep>, ir<1>
-; CHECK:  Cost of 1 for VF 4: WIDEN store vp<[[VP17]]>, ir<42>, vp<[[VP9]]>
+; CHECK:  Cost of 1 for VF 4: WIDEN store vp<[[VP17]]>, ir<42>, vp<[[VP9]]> (!vplan.execution.frequency 2305843009213693952 (25%, estimated))
 ; CHECK:  Cost of 0 for VF 4: vp<[[VP18]]> = vector-pointer i64, vp<%next.gep>, ir<1>
-; CHECK:  Cost of 1 for VF 4: WIDEN store vp<[[VP18]]>, ir<2>, vp<[[VP14]]>
+; CHECK:  Cost of 1 for VF 4: WIDEN store vp<[[VP18]]>, ir<2>, vp<[[VP14]]> (!vplan.execution.frequency 2305843009213693952 (25%, estimated))
 ; CHECK:  Cost of 0 for VF 4: EMIT vp<%index.next> = add nuw vp<[[VP5]]>, vp<[[VP1]]>
 ; CHECK:  Cost of 1 for VF 4: EMIT branch-on-count vp<%index.next>, vp<[[VP2]]>
 ; CHECK:  Cost of 0 for VF 4: vector loop backedge
@@ -289,15 +284,15 @@ exit:
 
 define void @test_vpinstruction_extractvalue_cost(ptr noalias %dst, {i64, i64} %sv) {
 ; CHECK-LABEL: 'test_vpinstruction_extractvalue_cost'
-; CHECK:  LV: Found an estimated cost of 0 for VF 1 For instruction: %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
-; CHECK:  LV: Found an estimated cost of 0 for VF 1 For instruction: %a = extractvalue { i64, i64 } %sv, 0
-; CHECK:  LV: Found an estimated cost of 0 for VF 1 For instruction: %b = extractvalue { i64, i64 } %sv, 1
-; CHECK:  LV: Found an estimated cost of 1 for VF 1 For instruction: %add = add i64 %a, %b
-; CHECK:  LV: Found an estimated cost of 0 for VF 1 For instruction: %g.dst = getelementptr inbounds i64, ptr %dst, i64 %iv
-; CHECK:  LV: Found an estimated cost of 1 for VF 1 For instruction: store i64 %add, ptr %g.dst, align 8
-; CHECK:  LV: Found an estimated cost of 1 for VF 1 For instruction: %iv.next = add nuw nsw i64 %iv, 1
-; CHECK:  LV: Found an estimated cost of 1 for VF 1 For instruction: %ec = icmp eq i64 %iv.next, 1000
-; CHECK:  LV: Found an estimated cost of 0 for VF 1 For instruction: br i1 %ec, label %exit, label %loop
+; CHECK:  Cost of 0 for VF 1: EMIT-SCALAR ir<%iv> = phi [ ir<0>, vector.ph ], [ ir<%iv.next>, loop ]
+; CHECK:  Cost of 0 for VF 1: EMIT ir<%a> = extractvalue ir<%sv>
+; CHECK:  Cost of 0 for VF 1: EMIT ir<%b> = extractvalue ir<%sv>
+; CHECK:  Cost of 1 for VF 1: EMIT ir<%add> = add ir<%a>, ir<%b>
+; CHECK:  Cost of 0 for VF 1: EMIT ir<%g.dst> = getelementptr inbounds ir<%dst>, ir<%iv>
+; CHECK:  Cost of 1 for VF 1: EMIT store ir<%add>, ir<%g.dst>
+; CHECK:  Cost of 1 for VF 1: EMIT ir<%iv.next> = add nuw nsw ir<%iv>, ir<1>
+; CHECK:  Cost of 1 for VF 1: EMIT ir<%ec> = icmp eq ir<%iv.next>, ir<1000>
+; CHECK:  Cost of 0 for VF 1: EMIT branch-on-cond ir<%ec>
 ; CHECK:  Cost of 0 for VF 2: vp<[[VP4:%[0-9]+]]> = SCALAR-STEPS vp<[[VP3:%[0-9]+]]>, ir<1>, vp<[[VP0:%[0-9]+]]>
 ; CHECK:  Cost of 0 for VF 2: CLONE ir<%g.dst> = getelementptr inbounds ir<%dst>, vp<[[VP4]]>
 ; CHECK:  Cost of 0 for VF 2: vp<[[VP5:%[0-9]+]]> = vector-pointer inbounds i64, ir<%g.dst>, ir<1>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/fneg-cost.ll b/llvm/test/Transforms/LoopVectorize/X86/fneg-cost.ll
index 693c6e5732d376..81fade6412464c 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/fneg-cost.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/fneg-cost.ll
@@ -5,10 +5,49 @@
 target datalayout = "e-p:64:64:64-i1:8:8-i8:8:8-i16:16:16-i32:32:32-i64:64:64-f32:32:32-f64:64:64-v64:64:64-v128:128:128-a0:0:64-s0:64:64-f80:128:128-n8:16:32:64-S128"
 target triple = "x86_64-apple-macosx10.8.0"
 
-; CHECK: Found an estimated cost of 1 for VF 1 For instruction:   %neg = fneg float %{{.*}}
+; CHECK: Cost of 1 for VF 1: EMIT ir<%neg> = fneg ir<%0>
 ; CHECK: Cost of 1 for VF 2: WIDEN ir<%neg> = fneg ir<%0>
 ; CHECK: Cost of 1 for VF 4: WIDEN ir<%neg> = fneg ir<%0>
 define void @fneg_cost(ptr %a, i64 %n) {
+; CHECK-LABEL: @fneg_cost(
+; CHECK-NEXT:  entry:
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N:%.*]], 8
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label [[SCALAR_PH:%.*]], label [[VECTOR_PH:%.*]]
+; CHECK:       vector.ph:
+; CHECK-NEXT:    [[TMP0:%.*]] = and i64 [[N]], 7
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
+; CHECK-NEXT:    br label [[VECTOR_BODY:%.*]]
+; CHECK:       vector.body:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, [[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], [[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP1:%.*]] = getelementptr inbounds float, ptr [[A:%.*]], i64 [[INDEX]]
+; CHECK-NEXT:    [[TMP2:%.*]] = getelementptr inbounds float, ptr [[TMP1]], i64 4
+; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <4 x float>, ptr [[TMP1]], align 4
+; CHECK-NEXT:    [[WIDE_LOAD1:%.*]] = load <4 x float>, ptr [[TMP2]], align 4
+; CHECK-NEXT:    [[TMP3:%.*]] = fneg <4 x float> [[WIDE_LOAD]]
+; CHECK-NEXT:    [[TMP4:%.*]] = fneg <4 x float> [[WIDE_LOAD1]]
+; CHECK-NEXT:    store <4 x float> [[TMP3]], ptr [[TMP1]], align 4
+; CHECK-NEXT:    store <4 x float> [[TMP4]], ptr [[TMP2]], align 4
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
+; CHECK-NEXT:    [[TMP5:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP5]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK:       middle.block:
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], label [[FOR_END:%.*]], label [[SCALAR_PH]]
+; CHECK:       scalar.ph:
+; CHECK-NEXT:    [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], [[MIDDLE_BLOCK]] ], [ 0, [[ENTRY:%.*]] ]
+; CHECK-NEXT:    br label [[FOR_BODY:%.*]]
+; CHECK:       for.body:
+; CHECK-NEXT:    [[INDVARS_IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], [[SCALAR_PH]] ], [ [[INDVARS_IV_NEXT:%.*]], [[FOR_BODY]] ]
+; CHECK-NEXT:    [[ARRAYIDX:%.*]] = getelementptr inbounds float, ptr [[A]], i64 [[INDVARS_IV]]
+; CHECK-NEXT:    [[TMP6:%.*]] = load float, ptr [[ARRAYIDX]], align 4
+; CHECK-NEXT:    [[NEG:%.*]] = fneg float [[TMP6]]
+; CHECK-NEXT:    store float [[NEG]], ptr [[ARRAYIDX]], align 4
+; CHECK-NEXT:    [[INDVARS_IV_NEXT]] = add nuw nsw i64 [[INDVARS_IV]], 1
+; CHECK-NEXT:    [[CMP:%.*]] = icmp eq i64 [[INDVARS_IV_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[CMP]], label [[FOR_END]], label [[FOR_BODY]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK:       for.end:
+; CHECK-NEXT:    ret void
+;
 entry:
   br label %for.body
 for.body:
diff --git a/llvm/test/Transforms/LoopVectorize/X86/fp_to_sint8-cost-model.ll b/llvm/test/Transforms/LoopVectorize/X86/fp_to_sint8-cost-model.ll
index 38662644e26123..7140ed86f15dc9 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/fp_to_sint8-cost-model.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/fp_to_sint8-cost-model.ll
@@ -5,7 +5,7 @@ target datalayout = "e-p:64:64:64-i1:8:8-i8:8:8-i16:16:16-i32:32:32-i64:64:64-f3
 target triple = "x86_64-apple-macosx10.8.0"
 
 
-; CHECK: cost of 1 for VF 1 For instruction:   %conv = fptosi float %tmp to i8
+; CHECK: Cost of 1 for VF 1: EMIT-SCALAR ir<%conv> = fptosi ir<%tmp> to i8
 define void @float_to_sint8_cost(ptr noalias nocapture %a, ptr noalias nocapture readonly %b) {
 entry:
   br label %for.body
diff --git a/llvm/test/Transforms/LoopVectorize/X86/reduction-small-size.ll b/llvm/test/Transforms/LoopVectorize/X86/reduction-small-size.ll
index c51b7ff07e19ec..4b4560f144ffae 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/reduction-small-size.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/reduction-small-size.ll
@@ -13,21 +13,21 @@ target datalayout = "e-m:e-i64:64-f80:128-n8:16:32:64-S128"
 ;
 
 ; CHECK-LABEL: reduction_i8
-; CHECK: LV: Found an estimated cost of {{[0-9]+}} for VF 1 For instruction:   %{{.*}} = phi
-; CHECK: LV: Found an estimated cost of {{[0-9]+}} for VF 1 For instruction:   %{{.*}} = phi
-; CHECK: LV: Found an estimated cost of {{[0-9]+}} for VF 1 For instruction:   %{{.*}} = getelementptr
-; CHECK: LV: Found an estimated cost of {{[0-9]+}} for VF 1 For instruction:   %{{.*}} = load
-; CHECK: LV: Found an estimated cost of {{[0-9]+}} for VF 1 For instruction:   %{{.*}} = zext i8 %{{.*}} to i32
-; CHECK: LV: Found an estimated cost of {{[0-9]+}} for VF 1 For instruction:   %{{.*}} = getelementptr
-; CHECK: LV: Found an estimated cost of {{[0-9]+}} for VF 1 For instruction:   %{{.*}} = load
-; CHECK: LV: Found an estimated cost of {{[0-9]+}} for VF 1 For instruction:   %{{.*}} = zext i8 %{{.*}} to i32
-; CHECK: LV: Found an estimated cost of {{[0-9]+}} for VF 1 For instruction:   %{{.*}} = and i32 %{{.*}}, 255
-; CHECK: LV: Found an estimated cost of {{[0-9]+}} for VF 1 For instruction:   %{{.*}} = add
-; CHECK: LV: Found an estimated cost of {{[0-9]+}} for VF 1 For instruction:   %{{.*}} = add
-; CHECK: LV: Found an estimated cost of {{[0-9]+}} for VF 1 For instruction:   %{{.*}} = add
-; CHECK: LV: Found an estimated cost of {{[0-9]+}} for VF 1 For instruction:   %{{.*}} = trunc
-; CHECK: LV: Found an estimated cost of {{[0-9]+}} for VF 1 For instruction:   %{{.*}} = icmp
-; CHECK: LV: Found an estimated cost of {{[0-9]+}} for VF 1 For instruction:   br
+; CHECK: Cost of {{[0-9]+}} for VF 1: EMIT-SCALAR ir<%indvars.iv> = phi
+; CHECK: Cost of {{[0-9]+}} for VF 1: EMIT-SCALAR ir<%sum.013> = phi
+; CHECK: Cost of {{[0-9]+}} for VF 1: EMIT ir<%arrayidx> = getelementptr
+; CHECK: Cost of {{[0-9]+}} for VF 1: EMIT-SCALAR ir<%0> = load
+; CHECK: Cost of {{[0-9]+}} for VF 1: EMIT-SCALAR ir<%conv> = zext ir<%0> to i32
+; CHECK: Cost of {{[0-9]+}} for VF 1: EMIT ir<%arrayidx2> = getelementptr
+; CHECK: Cost of {{[0-9]+}} for VF 1: EMIT-SCALAR ir<%1> = load
+; CHECK: Cost of {{[0-9]+}} for VF 1: EMIT-SCALAR ir<%conv3> = zext ir<%1> to i32
+; CHECK: Cost of {{[0-9]+}} for VF 1: EMIT ir<%conv4> = and ir<%sum.013>, ir<255>
+; CHECK: Cost of {{[0-9]+}} for VF 1: EMIT ir<%add> = add
+; CHECK: Cost of {{[0-9]+}} for VF 1: EMIT ir<%add5> = add
+; CHECK: Cost of {{[0-9]+}} for VF 1: EMIT ir<%indvars.iv.next> = add
+; CHECK: Cost of {{[0-9]+}} for VF 1: EMIT-SCALAR ir<%lftr.wideiv> = trunc
+; CHECK: Cost of {{[0-9]+}} for VF 1: EMIT ir<%exitcond> = icmp
+; CHECK: Cost of {{[0-9]+}} for VF 1: EMIT branch-on-cond
 ; CHECK: Cost of 1 for VF 2: WIDEN-REDUCTION-PHI ir<%sum.013> = phi (add) vp<{{.+}}>, vp<[[EXT:%.+]]>
 ; CHECK: Cost of 0 for VF 2: vp<[[STEPS:%.+]]> = SCALAR-STEPS vp<[[CAN_IV:%.+]]>, ir<1>
 ; CHECK: Cost of 0 for VF 2: CLONE ir<%arrayidx> = getelementptr inbounds ir<%a>, vp<[[STEPS]]>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/uint64_to_fp64-cost-model.ll b/llvm/test/Transforms/LoopVectorize/X86/uint64_to_fp64-cost-model.ll
index 0edb89af5bc545..f124389e7eaa4d 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/uint64_to_fp64-cost-model.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/uint64_to_fp64-cost-model.ll
@@ -5,7 +5,7 @@ target datalayout = "e-p:64:64:64-i1:8:8-i8:8:8-i16:16:16-i32:32:32-i64:64:64-f3
 target triple = "x86_64-apple-macosx10.8.0"
 
 
-; CHECK: cost of 4 for VF 1 For instruction:   %conv = uitofp i64 %tmp to double
+; CHECK: Cost of 4 for VF 1: EMIT-SCALAR ir<%conv> = uitofp ir<%tmp> to double
 ; CHECK: Cost of 5 for VF 2: WIDEN-CAST ir<%conv> = uitofp ir<%tmp> to double
 ; CHECK: Cost of 10 for VF 4: WIDEN-CAST ir<%conv> = uitofp ir<%tmp> to double
 define void @uint64_to_double_cost(ptr noalias nocapture %a, ptr noalias nocapture readonly %b) {
diff --git a/llvm/test/Transforms/LoopVectorize/X86/uniformshift.ll b/llvm/test/Transforms/LoopVectorize/X86/uniformshift.ll
index 61eaa3524fa569..e38ba170eb6029 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/uniformshift.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/uniformshift.ll
@@ -2,7 +2,7 @@
 ; REQUIRES: asserts
 
 ; CHECK: 'foo'
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction:   %shift = ashr i32 %val, %k
+; CHECK: Cost of 1 for VF 1: EMIT ir<%shift> = ashr ir<%val>, ir<%k>
 ; CHECK: Cost of 2 for VF 2: WIDEN ir<%shift> = ashr ir<%val>, ir<%k>
 ; CHECK: Cost of 2 for VF 4: WIDEN ir<%shift> = ashr ir<%val>, ir<%k>
 define void @foo(ptr nocapture %p, i32 %k) {
diff --git a/llvm/test/Transforms/LoopVectorize/X86/vector-scalar-select-cost.ll b/llvm/test/Transforms/LoopVectorize/X86/vector-scalar-select-cost.ll
index 8acdabd6d675e3..bfa3e59df03c72 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/vector-scalar-select-cost.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/vector-scalar-select-cost.ll
@@ -22,7 +22,7 @@ define void @scalarselect(i1 %cond) {
   %6 = add nsw i32 %5, %3
   %7 = getelementptr inbounds [2048 x i32], ptr @a, i64 0, i64 %indvars.iv
 
-; CHECK: cost of 1 for VF 1 {{.*}}  select i1 %cond, i32 %6, i32 0
+; CHECK: Cost of 1 for VF 1: EMIT ir<%sel> = select ir<%cond>, ir<%6>, ir<0>
 ; CHECK: Cost of 2 for VF 2: WIDEN ir<%sel> = select ir<%cond>, ir<%6>, ir<0>
 ; CHECK: Cost of 2 for VF 4: WIDEN ir<%sel> = select ir<%cond>, ir<%6>, ir<0>
 
@@ -51,7 +51,7 @@ define void @vectorselect(i1 %cond) {
   %7 = getelementptr inbounds [2048 x i32], ptr @a, i64 0, i64 %indvars.iv
   %8 = icmp ult i64 %indvars.iv, 8
 
-; CHECK: cost of 1 for VF 1 {{.*}}  select i1 %8, i32 %6, i32 0
+; CHECK: Cost of 1 for VF 1: EMIT ir<%sel> = select ir<%8>, ir<%6>, ir<0>
 ; CHECK: Cost of 2 for VF 2: WIDEN ir<%sel> = select ir<%8>, ir<%6>, ir<0>
 ; CHECK: Cost of 2 for VF 4: WIDEN ir<%sel> = select ir<%8>, ir<%6>, ir<0>
 

>From 1de4f00f3de5bf43171bcc49a3053ec3b42058cb Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Tue, 16 Jun 2026 22:08:31 +0200
Subject: [PATCH 2/7] !fixup address comments, thanks

---
 llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp | 5 ++---
 1 file changed, 2 insertions(+), 3 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
index 0304127b9907c0..b752072c71f5fb 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
@@ -1178,9 +1178,8 @@ InstructionCost VPRecipeWithIRFlags::getCostForRecipeWithOpcode(
   case Instruction::Xor: {
     // Certain instructions can be cheaper if they have a constant second
     // operand. One example of this are shifts on x86.
-    TargetTransformInfo::OperandValueInfo RHSInfo = {
-        TargetTransformInfo::OK_AnyValue, TargetTransformInfo::OP_None};
-    if (Opcode != Instruction::FNeg) {
+    TargetTransformInfo::OperandValueInfo RHSInfo;
+    if (getNumOperands() == 2) {
       RHSInfo = Ctx.getOperandInfo(getOperand(1));
       if (RHSInfo.Kind == TargetTransformInfo::OK_AnyValue &&
           getOperand(1)->isDefinedOutsideLoopRegions())

>From 2e6276438d4a418de63b05a9ce2c5f3f875101b7 Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Wed, 9 Sep 2026 14:01:30 +0100
Subject: [PATCH 3/7] !fixup update after merge

---
 .../Transforms/Vectorize/LoopVectorize.cpp    | 18 ++++--------
 .../lib/Transforms/Vectorize/VPlanRecipes.cpp | 29 ++++++++++---------
 2 files changed, 21 insertions(+), 26 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 7248923bf84761..9fbb03482e6dcd 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -5565,18 +5565,11 @@ InstructionCost LoopVectorizationPlanner::computeScalarCost() const {
   InstructionCost Cost = 0;
 
   for (VPBasicBlock *VPBB : vp_rpo_plain_cfg_loop_body(Header)) {
-    // Look up the divisor via the first underlying IR instruction in the loop.
-    uint64_t Divisor = 1;
-    for (const VPRecipeBase &R : *VPBB) {
-      auto *UI = dyn_cast_if_present<Instruction>(
-          cast<VPSingleDefRecipe>(&R)->getUnderlyingValue());
-      if (!UI)
-        continue;
-      Divisor = CostCtx.CM.getPredBlockCostDivisor(CostCtx.CostKind,
-                                                   UI->getParent());
-      break;
-    }
-    Cost += VPBB->cost(ScalarVF, CostCtx) / Divisor;
+    // In the scalar loop, we may not always execute the predicated block, if
+    // it is an if-else block. Thus, scale the block's cost by the probability
+    // of executing it.
+    Cost += VPBB->cost(ScalarVF, CostCtx) /
+            CostCtx.getCostDivisor(getRecordedExecutionFrequency(VPBB));
   }
   return Cost;
 }
@@ -6379,7 +6372,6 @@ static bool verifyExecutionFrequenciesMatchBFI(VPlan &Plan, Loop *OrigLoop,
 
   for (const auto &[VPBB, BB] :
        zip_equal(drop_begin(Blocks), drop_begin(OrigRPO))) {
-    // Nothing to check for blocks without a recorded frequency.
     std::optional<VPExecutionFrequency> Freq =
         getRecordedExecutionFrequency(VPBB);
     if (!Freq)
diff --git a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
index b752072c71f5fb..e1e55111ddd2a3 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
@@ -1177,9 +1177,10 @@ InstructionCost VPRecipeWithIRFlags::getCostForRecipeWithOpcode(
   case Instruction::Or:
   case Instruction::Xor: {
     // Certain instructions can be cheaper if they have a constant second
-    // operand. One example of this are shifts on x86.
+    // operand. One example of this are shifts on x86. FNeg is the only unary
+    // opcode handled here and has no second operand.
     TargetTransformInfo::OperandValueInfo RHSInfo;
-    if (getNumOperands() == 2) {
+    if (Opcode != Instruction::FNeg) {
       RHSInfo = Ctx.getOperandInfo(getOperand(1));
       if (RHSInfo.Kind == TargetTransformInfo::OK_AnyValue &&
           getOperand(1)->isDefinedOutsideLoopRegions())
@@ -1214,8 +1215,7 @@ InstructionCost VPRecipeWithIRFlags::getCostForRecipeWithOpcode(
     return Ctx.TTI.getAddressComputationCost(PtrTy, nullptr, nullptr,
                                              Ctx.CostKind) +
            Ctx.TTI.getMemoryOpCost(Opcode, ValTy, getLoadStoreAlignment(UI),
-                                   cast<PointerType>(PtrTy)->getAddressSpace(),
-                                   Ctx.CostKind,
+                                   getLoadStoreAddressSpace(UI), Ctx.CostKind,
                                    TTI::getOperandInfo(UI->getOperand(0)), UI);
   }
   case Instruction::ICmp:
@@ -1262,10 +1262,10 @@ InstructionCost VPRecipeWithIRFlags::getCostForRecipeWithOpcode(
       }
       // Loads/stores in pre-predication VPlan0 are represented as
       // VPInstructions; treat them like an unmasked memory access.
-      if (const auto *VPI = dyn_cast<VPInstruction>(R))
-        if (VPI->getOpcode() == Instruction::Load ||
-            VPI->getOpcode() == Instruction::Store)
-          return TTI::CastContextHint::Normal;
+      const auto *VPI = dyn_cast<VPInstruction>(R);
+      if (VPI && (VPI->getOpcode() == Instruction::Load ||
+                  VPI->getOpcode() == Instruction::Store))
+        return TTI::CastContextHint::Normal;
       const auto *WidenMemoryRecipe = dyn_cast<VPWidenMemoryRecipe>(R);
       if (WidenMemoryRecipe == nullptr)
         return TTI::CastContextHint::None;
@@ -1376,11 +1376,12 @@ InstructionCost VPRecipeWithIRFlags::getCostForRecipeWithOpcode(
 
 InstructionCost VPInstruction::computeCost(ElementCount VF,
                                            VPCostContext &Ctx) const {
-  // Vector-only opcodes have zero cost at scalar VF.
-  if (VF.isScalar() &&
-      (isVectorToScalar() ||
-       getOpcode() == VPInstruction::FirstOrderRecurrenceSplice))
-    return 0;
+  // A scalar cost is only computed for VPlan0, which has no vector-only
+  // opcodes.
+  assert(!(VF.isScalar() &&
+           (isVectorToScalar() ||
+            getOpcode() == VPInstruction::FirstOrderRecurrenceSplice)) &&
+         "unexpected vector-only opcode at scalar VF");
 
   // NOTE: At the moment it seems only possible to expose this path for
   // the trunc, zext and sext opcodes.
@@ -1610,6 +1611,8 @@ InstructionCost VPInstruction::computeCost(ElementCount VF,
   default:
     // TODO: Compute cost other VPInstructions once the legacy cost model has
     // been retired.
+    assert((VF.isScalar() || !getUnderlyingValue()) &&
+           "unexpected VPInstruction with underlying value");
     return 0;
   }
 }

>From 218a98db94163d555b0935f333e46a4a8bd82b83 Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Sun, 20 Sep 2026 22:58:21 +0100
Subject: [PATCH 4/7] !fixup update UpdateTestChecks expected output

---
 .../Inputs/x86-loopvectorize-costmodel.ll.expected               | 1 -
 1 file changed, 1 deletion(-)

diff --git a/llvm/test/tools/UpdateTestChecks/update_analyze_test_checks/Inputs/x86-loopvectorize-costmodel.ll.expected b/llvm/test/tools/UpdateTestChecks/update_analyze_test_checks/Inputs/x86-loopvectorize-costmodel.ll.expected
index 3abf35ff0feba7..88911d7440d382 100644
--- a/llvm/test/tools/UpdateTestChecks/update_analyze_test_checks/Inputs/x86-loopvectorize-costmodel.ll.expected
+++ b/llvm/test/tools/UpdateTestChecks/update_analyze_test_checks/Inputs/x86-loopvectorize-costmodel.ll.expected
@@ -10,7 +10,6 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; CHECK-LABEL: 'test'
-; CHECK:  LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load float, ptr %in0, align 4
 ; CHECK:  Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
 ; CHECK:  Cost of 3 for VF 4: INTERLEAVE-GROUP with factor 2, ir<%in0>
 ; CHECK:  Cost of 3 for VF 8: INTERLEAVE-GROUP with factor 2, ir<%in0>

>From 8fe9f65d4b7c24b5c4bd26cdf87df70f3073f90f Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Mon, 21 Sep 2026 11:18:04 +0100
Subject: [PATCH 5/7] !fixup restore some VF = 1 check lines

---
 .../LoopVectorize/AArch64/intrinsiccost.ll    |  9 +++++
 .../LoopVectorize/ARM/mve-icmpcost.ll         | 22 +++++-----
 .../X86/CostModel/gather-i16-with-i8-index.ll |  2 +-
 .../X86/CostModel/gather-i32-with-i8-index.ll |  2 +-
 .../X86/CostModel/gather-i64-with-i8-index.ll |  2 +-
 .../X86/CostModel/gather-i8-with-i8-index.ll  |  2 +-
 ...dle-iptr-with-data-layout-to-not-assert.ll |  3 +-
 .../interleaved-load-f32-stride-2.ll          |  6 ++-
 .../interleaved-load-f32-stride-3.ll          |  6 ++-
 .../interleaved-load-f32-stride-4.ll          |  6 ++-
 .../interleaved-load-f32-stride-5.ll          |  6 ++-
 .../interleaved-load-f32-stride-6.ll          |  6 ++-
 .../interleaved-load-f32-stride-7.ll          |  6 ++-
 .../interleaved-load-f32-stride-8.ll          |  6 ++-
 .../interleaved-load-f64-stride-2.ll          |  6 ++-
 .../interleaved-load-f64-stride-3.ll          |  6 ++-
 .../interleaved-load-f64-stride-4.ll          |  6 ++-
 .../interleaved-load-f64-stride-5.ll          |  6 ++-
 .../interleaved-load-f64-stride-6.ll          |  6 ++-
 .../interleaved-load-f64-stride-7.ll          |  6 ++-
 .../interleaved-load-f64-stride-8.ll          |  6 ++-
 .../interleaved-load-i16-stride-2.ll          |  7 +++-
 .../interleaved-load-i16-stride-3.ll          |  7 +++-
 .../interleaved-load-i16-stride-4.ll          |  7 +++-
 .../interleaved-load-i16-stride-5.ll          |  7 +++-
 .../interleaved-load-i16-stride-6.ll          |  7 +++-
 .../interleaved-load-i16-stride-7.ll          |  7 +++-
 .../interleaved-load-i16-stride-8.ll          |  7 +++-
 ...nterleaved-load-i32-stride-2-indices-0u.ll |  6 ++-
 .../interleaved-load-i32-stride-2.ll          |  6 ++-
 ...terleaved-load-i32-stride-3-indices-01u.ll |  6 ++-
 ...terleaved-load-i32-stride-3-indices-0uu.ll |  6 ++-
 .../interleaved-load-i32-stride-3.ll          |  6 ++-
 ...erleaved-load-i32-stride-4-indices-012u.ll |  6 ++-
 ...erleaved-load-i32-stride-4-indices-01uu.ll |  6 ++-
 ...erleaved-load-i32-stride-4-indices-0uuu.ll |  6 ++-
 .../interleaved-load-i32-stride-4.ll          |  6 ++-
 .../interleaved-load-i32-stride-5.ll          |  6 ++-
 .../interleaved-load-i32-stride-6.ll          |  6 ++-
 .../interleaved-load-i32-stride-7.ll          |  6 ++-
 .../interleaved-load-i32-stride-8.ll          |  6 ++-
 .../interleaved-load-i64-stride-2.ll          |  6 ++-
 .../interleaved-load-i64-stride-3.ll          |  6 ++-
 .../interleaved-load-i64-stride-4.ll          |  6 ++-
 .../interleaved-load-i64-stride-5.ll          |  6 ++-
 .../interleaved-load-i64-stride-6.ll          |  6 ++-
 .../interleaved-load-i64-stride-7.ll          |  6 ++-
 .../interleaved-load-i64-stride-8.ll          |  6 ++-
 .../CostModel/interleaved-load-i8-stride-2.ll |  7 +++-
 .../CostModel/interleaved-load-i8-stride-3.ll |  7 +++-
 .../CostModel/interleaved-load-i8-stride-4.ll |  7 +++-
 .../CostModel/interleaved-load-i8-stride-5.ll |  7 +++-
 .../CostModel/interleaved-load-i8-stride-6.ll |  7 +++-
 .../CostModel/interleaved-load-i8-stride-7.ll |  7 +++-
 .../CostModel/interleaved-load-i8-stride-8.ll |  7 +++-
 .../masked-gather-i32-with-i8-index.ll        | 34 ++++++++--------
 .../masked-gather-i64-with-i8-index.ll        | 34 ++++++++--------
 .../CostModel/masked-interleaved-load-i16.ll  | 22 +++++++---
 .../CostModel/masked-interleaved-store-i16.ll | 12 +++++-
 .../X86/CostModel/masked-load-i16.ll          |  2 +-
 .../X86/CostModel/masked-load-i32.ll          |  2 +-
 .../X86/CostModel/masked-load-i64.ll          |  2 +-
 .../X86/CostModel/masked-load-i8.ll           |  2 +-
 .../masked-scatter-i32-with-i8-index.ll       | 15 ++++---
 .../masked-scatter-i64-with-i8-index.ll       | 15 ++++---
 .../X86/CostModel/masked-store-i16.ll         |  6 ++-
 .../X86/CostModel/masked-store-i32.ll         |  7 +++-
 .../X86/CostModel/masked-store-i64.ll         |  7 +++-
 .../X86/CostModel/masked-store-i8.ll          |  7 +++-
 .../CostModel/scatter-i16-with-i8-index.ll    |  7 +++-
 .../CostModel/scatter-i32-with-i8-index.ll    |  7 +++-
 .../CostModel/scatter-i64-with-i8-index.ll    |  7 +++-
 .../X86/CostModel/scatter-i8-with-i8-index.ll |  7 +++-
 .../X86/CostModel/strided-load-i16.ll         |  6 ++-
 .../X86/CostModel/strided-load-i32.ll         |  6 ++-
 .../X86/CostModel/strided-load-i64.ll         |  5 ++-
 .../X86/CostModel/strided-load-i8.ll          |  6 ++-
 .../X86/CostModel/vpinstruction-cost.ll       | 40 +++++++++----------
 .../x86-loopvectorize-costmodel.ll.expected   |  3 +-
 .../loopvectorize-costmodel.test              |  6 +--
 80 files changed, 457 insertions(+), 154 deletions(-)

diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/intrinsiccost.ll b/llvm/test/Transforms/LoopVectorize/AArch64/intrinsiccost.ll
index 958e7763aad3d6..69409fc55e03d0 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/intrinsiccost.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/intrinsiccost.ll
@@ -7,6 +7,10 @@ target datalayout = "e-m:e-i8:8:32-i16:16:32-i64:64-i128:128-n32:64-S128"
 target triple = "aarch64--linux-gnu"
 
 ; CHECK-COST-LABEL: sadd
+; CHECK-COST: Cost of 6 for VF 1: EMIT ir<%1> = call ir<%0>, ir<%offset>, ir<@llvm.sadd.sat.i16>
+; CHECK-COST: Cost of 4 for VF 2: WIDEN-INTRINSIC ir<%1> = call llvm.sadd.sat(ir<%0>, ir<%offset>)
+; CHECK-COST: Cost of 1 for VF 4: WIDEN-INTRINSIC ir<%1> = call llvm.sadd.sat(ir<%0>, ir<%offset>)
+; CHECK-COST: Cost of 1 for VF 8: WIDEN-INTRINSIC ir<%1> = call llvm.sadd.sat(ir<%0>, ir<%offset>)
 
 define void @saddsat(ptr nocapture readonly %pSrc, i16 signext %offset, ptr nocapture noalias %pDst, i32 %blockSize) #0 {
 ; CHECK-LABEL: @saddsat(
@@ -123,6 +127,11 @@ while.end:
 }
 
 ; CHECK-COST-LABEL: umin
+; CHECK-COST: Cost of 2 for VF 1: EMIT ir<%1> = call ir<%0>, ir<%offset>, ir<@llvm.umin.i8>
+; CHECK-COST: Cost of 3 for VF 2: WIDEN-INTRINSIC ir<%1> = call llvm.umin(ir<%0>, ir<%offset>)
+; CHECK-COST: Cost of 3 for VF 4: WIDEN-INTRINSIC ir<%1> = call llvm.umin(ir<%0>, ir<%offset>)
+; CHECK-COST: Cost of 1 for VF 8: WIDEN-INTRINSIC ir<%1> = call llvm.umin(ir<%0>, ir<%offset>)
+; CHECK-COST: Cost of 1 for VF 16: WIDEN-INTRINSIC ir<%1> = call llvm.umin(ir<%0>, ir<%offset>)
 
 define void @umin(ptr nocapture readonly %pSrc, i8 signext %offset, ptr nocapture noalias %pDst, i32 %blockSize) #0 {
 ; CHECK-LABEL: @umin(
diff --git a/llvm/test/Transforms/LoopVectorize/ARM/mve-icmpcost.ll b/llvm/test/Transforms/LoopVectorize/ARM/mve-icmpcost.ll
index aa6ec753d92764..a503617f6f47ad 100644
--- a/llvm/test/Transforms/LoopVectorize/ARM/mve-icmpcost.ll
+++ b/llvm/test/Transforms/LoopVectorize/ARM/mve-icmpcost.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost for VF" --filter "LV: Selecting VF" --filter "Found an estimated cost of .* for VF 1" --filter "Cost of .* for VF" --version 6
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost for VF" --filter "LV: Selecting VF" --filter "Cost of .* for VF" --version 6
 ; RUN: opt -passes=loop-vectorize -debug-only=loop-vectorize -disable-output < %s 2>&1 | FileCheck %s
 ; REQUIRES: asserts
 
@@ -12,10 +12,10 @@ define void @expensive_icmp(ptr noalias nocapture %d, ptr nocapture readonly %s,
 ; CHECK:  Cost of 1 for VF 1: EMIT-SCALAR ir<%1> = load ir<%arrayidx>
 ; CHECK:  Cost of 0 for VF 1: EMIT-SCALAR ir<%conv> = sext ir<%1> to i32
 ; CHECK:  Cost of 1 for VF 1: EMIT ir<%cmp2> = icmp sgt ir<%conv>, ir<%conv1>
-; CHECK:  Cost of 0 for VF 1: EMIT branch-on-cond ir<%cmp2> (!vplan.prof.estimated estimated {1073741824, 1073741824})
-; CHECK:  Cost of 1 for VF 1: EMIT ir<%conv6> = add ir<%1>, ir<%0> (!vplan.execution.frequency 4611686018427387904 (50%, estimated))
-; CHECK:  Cost of 0 for VF 1: EMIT ir<%arrayidx7> = getelementptr inbounds ir<%d>, ir<%i.016> (!vplan.execution.frequency 4611686018427387904 (50%, estimated))
-; CHECK:  Cost of 1 for VF 1: EMIT store ir<%conv6>, ir<%arrayidx7> (!vplan.execution.frequency 4611686018427387904 (50%, estimated))
+; CHECK:  Cost of 0 for VF 1: EMIT branch-on-cond ir<%cmp2>
+; CHECK:  Cost of 1 for VF 1: EMIT ir<%conv6> = add ir<%1>, ir<%0>
+; CHECK:  Cost of 0 for VF 1: EMIT ir<%arrayidx7> = getelementptr inbounds ir<%d>, ir<%i.016>
+; CHECK:  Cost of 1 for VF 1: EMIT store ir<%conv6>, ir<%arrayidx7>
 ; CHECK:  Cost of 1 for VF 1: EMIT ir<%inc> = add nuw nsw ir<%i.016>, ir<1>
 ; CHECK:  Cost of 1 for VF 1: EMIT ir<%exitcond.not> = icmp eq ir<%inc>, ir<%n>
 ; CHECK:  Cost of 0 for VF 1: EMIT branch-on-cond ir<%exitcond.not>
@@ -25,10 +25,10 @@ define void @expensive_icmp(ptr noalias nocapture %d, ptr nocapture readonly %s,
 ; CHECK:  Cost of 18 for VF 2: WIDEN ir<%1> = load vp<[[VP5]]>
 ; CHECK:  Cost of 4 for VF 2: WIDEN-CAST ir<%conv> = sext ir<%1> to i32
 ; CHECK:  Cost of 20 for VF 2: WIDEN ir<%cmp2> = icmp sgt ir<%conv>, ir<%conv1>
-; CHECK:  Cost of 26 for VF 2: WIDEN ir<%conv6> = add ir<%1>, ir<%0> (!vplan.execution.frequency 4611686018427387904 (50%, estimated))
+; CHECK:  Cost of 26 for VF 2: WIDEN ir<%conv6> = add ir<%1>, ir<%0>
 ; CHECK:  Cost of 0 for VF 2: CLONE ir<%arrayidx7> = getelementptr ir<%d>, vp<[[VP4]]>
 ; CHECK:  Cost of 0 for VF 2: vp<[[VP6:%[0-9]+]]> = vector-pointer i16, ir<%arrayidx7>, ir<1>
-; CHECK:  Cost of 16 for VF 2: WIDEN store vp<[[VP6]]>, ir<%conv6>, ir<%cmp2> (!vplan.execution.frequency 4611686018427387904 (50%, estimated))
+; CHECK:  Cost of 16 for VF 2: WIDEN store vp<[[VP6]]>, ir<%conv6>, ir<%cmp2>
 ; CHECK:  Cost of 0 for VF 2: EMIT vp<%index.next> = add nuw vp<[[VP3]]>, vp<[[VP1:%[0-9]+]]>
 ; CHECK:  Cost of 1 for VF 2: EMIT branch-on-count vp<%index.next>, vp<[[VP2:%[0-9]+]]>
 ; CHECK:  Cost of 0 for VF 2: vector loop backedge
@@ -50,10 +50,10 @@ define void @expensive_icmp(ptr noalias nocapture %d, ptr nocapture readonly %s,
 ; CHECK:  Cost of 2 for VF 4: WIDEN ir<%1> = load vp<[[VP5]]>
 ; CHECK:  Cost of 0 for VF 4: WIDEN-CAST ir<%conv> = sext ir<%1> to i32
 ; CHECK:  Cost of 2 for VF 4: WIDEN ir<%cmp2> = icmp sgt ir<%conv>, ir<%conv1>
-; CHECK:  Cost of 2 for VF 4: WIDEN ir<%conv6> = add ir<%1>, ir<%0> (!vplan.execution.frequency 4611686018427387904 (50%, estimated))
+; CHECK:  Cost of 2 for VF 4: WIDEN ir<%conv6> = add ir<%1>, ir<%0>
 ; CHECK:  Cost of 0 for VF 4: CLONE ir<%arrayidx7> = getelementptr ir<%d>, vp<[[VP4]]>
 ; CHECK:  Cost of 0 for VF 4: vp<[[VP6]]> = vector-pointer i16, ir<%arrayidx7>, ir<1>
-; CHECK:  Cost of 2 for VF 4: WIDEN store vp<[[VP6]]>, ir<%conv6>, ir<%cmp2> (!vplan.execution.frequency 4611686018427387904 (50%, estimated))
+; CHECK:  Cost of 2 for VF 4: WIDEN store vp<[[VP6]]>, ir<%conv6>, ir<%cmp2>
 ; CHECK:  Cost of 0 for VF 4: EMIT vp<%index.next> = add nuw vp<[[VP3]]>, vp<[[VP1]]>
 ; CHECK:  Cost of 1 for VF 4: EMIT branch-on-count vp<%index.next>, vp<[[VP2]]>
 ; CHECK:  Cost of 0 for VF 4: vector loop backedge
@@ -75,10 +75,10 @@ define void @expensive_icmp(ptr noalias nocapture %d, ptr nocapture readonly %s,
 ; CHECK:  Cost of 2 for VF 8: WIDEN ir<%1> = load vp<[[VP5]]>
 ; CHECK:  Cost of 2 for VF 8: WIDEN-CAST ir<%conv> = sext ir<%1> to i32
 ; CHECK:  Cost of 36 for VF 8: WIDEN ir<%cmp2> = icmp sgt ir<%conv>, ir<%conv1>
-; CHECK:  Cost of 2 for VF 8: WIDEN ir<%conv6> = add ir<%1>, ir<%0> (!vplan.execution.frequency 4611686018427387904 (50%, estimated))
+; CHECK:  Cost of 2 for VF 8: WIDEN ir<%conv6> = add ir<%1>, ir<%0>
 ; CHECK:  Cost of 0 for VF 8: CLONE ir<%arrayidx7> = getelementptr ir<%d>, vp<[[VP4]]>
 ; CHECK:  Cost of 0 for VF 8: vp<[[VP6]]> = vector-pointer i16, ir<%arrayidx7>, ir<1>
-; CHECK:  Cost of 2 for VF 8: WIDEN store vp<[[VP6]]>, ir<%conv6>, ir<%cmp2> (!vplan.execution.frequency 4611686018427387904 (50%, estimated))
+; CHECK:  Cost of 2 for VF 8: WIDEN store vp<[[VP6]]>, ir<%conv6>, ir<%cmp2>
 ; CHECK:  Cost of 0 for VF 8: EMIT vp<%index.next> = add nuw vp<[[VP3]]>, vp<[[VP1]]>
 ; CHECK:  Cost of 1 for VF 8: EMIT branch-on-count vp<%index.next>, vp<[[VP2]]>
 ; CHECK:  Cost of 0 for VF 8: vector loop backedge
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i16-with-i8-index.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i16-with-i8-index.ll
index 2770dd4801aebc..0af968f8b946a8 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i16-with-i8-index.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i16-with-i8-index.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF [0-9]+ For instruction:\s*%valB = load i16, ptr %inB, align 2" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: .* ir<%valB> = load"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: .* ir<%valB> = load"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=SSE
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse4.2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=SSE
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=AVX1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i32-with-i8-index.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i32-with-i8-index.ll
index 58b96771aedbea..b14c2b701eba7c 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i32-with-i8-index.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i32-with-i8-index.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF [0-9]+ For instruction:\s*%valB = load i32, ptr %inB, align 4" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: .* ir<%valB> = load"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: .* ir<%valB> = load"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=SSE2
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse4.2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=SSE42
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=AVX1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i64-with-i8-index.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i64-with-i8-index.ll
index 7876551023ee67..c4b9254e4abae2 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i64-with-i8-index.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i64-with-i8-index.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF [0-9]+ For instruction:\s*%valB = load i64, ptr %inB, align 8" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: .* ir<%valB> = load"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: .* ir<%valB> = load"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=SSE2
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse4.2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=SSE42
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=AVX1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i8-with-i8-index.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i8-with-i8-index.ll
index 435f7c3becd730..e3b5dc06878597 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i8-with-i8-index.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i8-with-i8-index.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF [0-9]+ For instruction:\s*%valB = load i8, ptr %inB, align 1" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: .* ir<%valB> = load"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: .* ir<%valB> = load"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=SSE2
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse4.2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=SSE42
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=AVX1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/handle-iptr-with-data-layout-to-not-assert.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/handle-iptr-with-data-layout-to-not-assert.ll
index 1c5e5d5aa21a3d..0c8b59417645cc 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/handle-iptr-with-data-layout-to-not-assert.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/handle-iptr-with-data-layout-to-not-assert.ll
@@ -1,10 +1,11 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*store ptr" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: INTERLEAVE-GROUP with factor [0-9]+" --version 5
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT store ir<%0>" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: INTERLEAVE-GROUP with factor [0-9]+" --version 5
 ; REQUIRES: asserts
 ; RUN: opt -passes=loop-vectorize -debug-only=loop-vectorize -S < %s 2>&1 | FileCheck %s
 target triple = "x86_64-unknown-linux-gnu"
 
 define ptr @foo(ptr %__first, ptr %__last) #0 {
 ; CHECK-LABEL: 'foo'
+; CHECK:  Cost of 1 for VF 1: EMIT store ir<%0>, ir<%__last>
 ; CHECK:  Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 2, vp<%next.gep>
 ; CHECK:  Cost of 1 for VF 4: INTERLEAVE-GROUP with factor 2, vp<%next.gep>
 ; CHECK:  Cost of 3 for VF 8: INTERLEAVE-GROUP with factor 2, vp<%next.gep>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-2.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-2.ll
index fe3fbb9c4bfee8..306975a7db84de 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-2.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-2.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,6 +13,7 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
+; SSE2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
 ; SSE2:    ir<%v0> = load from index 0
 ; SSE2:    ir<%v1> = load from index 1
@@ -23,6 +24,7 @@ define void @test() {
 ; SSE2:  Cost of 28 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
+; AVX1:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
 ; AVX1:    ir<%v0> = load from index 0
 ; AVX1:    ir<%v1> = load from index 1
@@ -34,6 +36,7 @@ define void @test() {
 ; AVX1:  Cost of 60 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
+; AVX2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
 ; AVX2:    ir<%v0> = load from index 0
 ; AVX2:    ir<%v1> = load from index 1
@@ -51,6 +54,7 @@ define void @test() {
 ; AVX2:    ir<%v1> = load from index 1
 ;
 ; AVX512-LABEL: 'test'
+; AVX512:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX512:  Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-3.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-3.ll
index 24ac329df20d63..f96e7096dd8012 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-3.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-3.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,12 +13,14 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
+; SSE2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 14 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 28 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
+; AVX1:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 15 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -26,6 +28,7 @@ define void @test() {
 ; AVX1:  Cost of 60 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
+; AVX2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 6 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
 ; AVX2:    ir<%v0> = load from index 0
 ; AVX2:    ir<%v1> = load from index 1
@@ -48,6 +51,7 @@ define void @test() {
 ; AVX2:    ir<%v2> = load from index 2
 ;
 ; AVX512-LABEL: 'test'
+; AVX512:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX512:  Cost of 4 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-4.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-4.ll
index b2cd0273efd8de..e88f49360ea3d9 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-4.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-4.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,12 +13,14 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
+; SSE2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 14 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 28 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
+; AVX1:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 15 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -26,6 +28,7 @@ define void @test() {
 ; AVX1:  Cost of 60 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
+; AVX2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 5 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
 ; AVX2:    ir<%v0> = load from index 0
 ; AVX2:    ir<%v1> = load from index 1
@@ -53,6 +56,7 @@ define void @test() {
 ; AVX2:    ir<%v3> = load from index 3
 ;
 ; AVX512-LABEL: 'test'
+; AVX512:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX512:  Cost of 5 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-5.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-5.ll
index f8980b2fb0f30d..3dc3b9cfd8f47e 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-5.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-5.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,12 +13,14 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
+; SSE2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 14 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 28 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
+; AVX1:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 15 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -26,6 +28,7 @@ define void @test() {
 ; AVX1:  Cost of 60 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
+; AVX2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 15 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -33,6 +36,7 @@ define void @test() {
 ; AVX2:  Cost of 60 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX512-LABEL: 'test'
+; AVX512:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX512:  Cost of 6 for VF 2: INTERLEAVE-GROUP with factor 5, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-6.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-6.ll
index c62d8eea92143c..5af85931fcd35b 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-6.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-6.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,12 +13,14 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
+; SSE2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 14 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 28 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
+; AVX1:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 15 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -26,6 +28,7 @@ define void @test() {
 ; AVX1:  Cost of 60 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
+; AVX2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 8 for VF 2: INTERLEAVE-GROUP with factor 6, ir<%in0>
 ; AVX2:    ir<%v0> = load from index 0
 ; AVX2:    ir<%v1> = load from index 1
@@ -57,6 +60,7 @@ define void @test() {
 ; AVX2:  Cost of 60 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX512-LABEL: 'test'
+; AVX512:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX512:  Cost of 7 for VF 2: INTERLEAVE-GROUP with factor 6, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-7.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-7.ll
index 68f7f233a8a86e..b9975cc07631d9 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-7.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-7.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,12 +13,14 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
+; SSE2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 14 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 28 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
+; AVX1:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 15 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -26,6 +28,7 @@ define void @test() {
 ; AVX1:  Cost of 60 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
+; AVX2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 15 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -33,6 +36,7 @@ define void @test() {
 ; AVX2:  Cost of 60 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX512-LABEL: 'test'
+; AVX512:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX512:  Cost of 8 for VF 2: INTERLEAVE-GROUP with factor 7, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-8.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-8.ll
index 154c14dd895e0d..8348f83928791c 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-8.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-8.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,12 +13,14 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
+; SSE2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 14 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 28 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
+; AVX1:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 15 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -26,6 +28,7 @@ define void @test() {
 ; AVX1:  Cost of 60 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
+; AVX2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 48 for VF 8: INTERLEAVE-GROUP with factor 8, ir<%in0>
@@ -41,6 +44,7 @@ define void @test() {
 ; AVX2:  Cost of 60 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX512-LABEL: 'test'
+; AVX512:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX512:  Cost of 9 for VF 2: INTERLEAVE-GROUP with factor 8, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-2.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-2.ll
index c9ddd0a38f1038..8de4225d48e78b 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-2.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-2.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,6 +13,7 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
+; SSE2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 4 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
 ; SSE2:    ir<%v0> = load from index 0
 ; SSE2:    ir<%v1> = load from index 1
@@ -21,6 +22,7 @@ define void @test() {
 ; SSE2:  Cost of 24 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
+; AVX1:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
 ; AVX1:    ir<%v0> = load from index 0
 ; AVX1:    ir<%v1> = load from index 1
@@ -30,6 +32,7 @@ define void @test() {
 ; AVX1:  Cost of 56 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
+; AVX2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
 ; AVX2:    ir<%v0> = load from index 0
 ; AVX2:    ir<%v1> = load from index 1
@@ -47,6 +50,7 @@ define void @test() {
 ; AVX2:    ir<%v1> = load from index 1
 ;
 ; AVX512-LABEL: 'test'
+; AVX512:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX512:  Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-3.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-3.ll
index d42188eaa22996..02b0676ff5156a 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-3.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-3.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,12 +13,14 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
+; SSE2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 6 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 12 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 24 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
+; AVX1:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 14 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -26,6 +28,7 @@ define void @test() {
 ; AVX1:  Cost of 56 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
+; AVX2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
 ; AVX2:    ir<%v0> = load from index 0
 ; AVX2:    ir<%v1> = load from index 1
@@ -45,6 +48,7 @@ define void @test() {
 ; AVX2:  Cost of 56 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX512-LABEL: 'test'
+; AVX512:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX512:  Cost of 4 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-4.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-4.ll
index 1bc77d8d3ff3c4..17ea3b4d11649d 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-4.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-4.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,12 +13,14 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
+; SSE2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 6 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 12 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 24 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
+; AVX1:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 14 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -26,6 +28,7 @@ define void @test() {
 ; AVX1:  Cost of 56 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
+; AVX2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 8 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
 ; AVX2:    ir<%v0> = load from index 0
 ; AVX2:    ir<%v1> = load from index 1
@@ -49,6 +52,7 @@ define void @test() {
 ; AVX2:  Cost of 56 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX512-LABEL: 'test'
+; AVX512:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX512:  Cost of 5 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-5.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-5.ll
index 9992f5ad71ab42..ff4e1335914a9c 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-5.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-5.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load|WIDEN ir<%v[0-9]> = load)" --filter "^  ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load|WIDEN ir<%v[0-9]> = load)" --filter "^  ir<.* = load from index"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,12 +13,14 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
+; SSE2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 6 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 12 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 24 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
+; AVX1:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 14 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -26,6 +28,7 @@ define void @test() {
 ; AVX1:  Cost of 56 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
+; AVX2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 14 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -33,6 +36,7 @@ define void @test() {
 ; AVX2:  Cost of 56 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX512-LABEL: 'test'
+; AVX512:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX512:  Cost of 14.5 for VF 2: INTERLEAVE-GROUP with factor 5, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-6.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-6.ll
index 4676e641166dd2..516a374ddd49a8 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-6.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-6.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load|WIDEN ir<%v[0-9]> = load)" --filter "^  ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load|WIDEN ir<%v[0-9]> = load)" --filter "^  ir<.* = load from index"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,12 +13,14 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
+; SSE2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 6 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 12 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 24 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
+; AVX1:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 14 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -26,6 +28,7 @@ define void @test() {
 ; AVX1:  Cost of 56 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
+; AVX2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 9 for VF 2: INTERLEAVE-GROUP with factor 6, ir<%in0>
 ; AVX2:    ir<%v0> = load from index 0
 ; AVX2:    ir<%v1> = load from index 1
@@ -51,6 +54,7 @@ define void @test() {
 ; AVX2:  Cost of 56 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX512-LABEL: 'test'
+; AVX512:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX512:  Cost of 17 for VF 2: INTERLEAVE-GROUP with factor 6, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-7.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-7.ll
index fbbc94de0c1a91..27bc9169c7e115 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-7.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-7.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load|WIDEN ir<%v[0-9]> = load)" --filter "^  ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load|WIDEN ir<%v[0-9]> = load)" --filter "^  ir<.* = load from index"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,12 +13,14 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
+; SSE2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 6 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 12 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 24 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
+; AVX1:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 14 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -26,6 +28,7 @@ define void @test() {
 ; AVX1:  Cost of 56 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
+; AVX2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 14 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -33,6 +36,7 @@ define void @test() {
 ; AVX2:  Cost of 56 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX512-LABEL: 'test'
+; AVX512:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX512:  Cost of 19.5 for VF 2: INTERLEAVE-GROUP with factor 7, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-8.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-8.ll
index 7c22d3e51be763..0a0c5695b5808b 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-8.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-8.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load|WIDEN ir<%v[0-9]> = load)" --filter "^  ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load|WIDEN ir<%v[0-9]> = load)" --filter "^  ir<.* = load from index"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,12 +13,14 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
+; SSE2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 6 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 12 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 24 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
+; AVX1:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 14 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -26,6 +28,7 @@ define void @test() {
 ; AVX1:  Cost of 56 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
+; AVX2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 14 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -33,6 +36,7 @@ define void @test() {
 ; AVX2:  Cost of 56 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX512-LABEL: 'test'
+; AVX512:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX512:  Cost of 22 for VF 2: INTERLEAVE-GROUP with factor 8, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-2.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-2.ll
index 4fe1ae7f140c47..b7e8ce35dbc630 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-2.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-2.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -14,6 +14,7 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
+; SSE2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
 ; SSE2:    ir<%v0> = load from index 0
 ; SSE2:    ir<%v1> = load from index 1
@@ -24,6 +25,7 @@ define void @test() {
 ; SSE2:  Cost of 32 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
+; AVX1:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
 ; AVX1:    ir<%v0> = load from index 0
 ; AVX1:    ir<%v1> = load from index 1
@@ -35,6 +37,7 @@ define void @test() {
 ; AVX1:  Cost of 66 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
+; AVX2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
 ; AVX2:    ir<%v0> = load from index 0
 ; AVX2:    ir<%v1> = load from index 1
@@ -52,6 +55,7 @@ define void @test() {
 ; AVX2:    ir<%v1> = load from index 1
 ;
 ; AVX512DQ-LABEL: 'test'
+; AVX512DQ:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX512DQ:  Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
 ; AVX512DQ:    ir<%v0> = load from index 0
 ; AVX512DQ:    ir<%v1> = load from index 1
@@ -72,6 +76,7 @@ define void @test() {
 ; AVX512DQ:    ir<%v1> = load from index 1
 ;
 ; AVX512BW-LABEL: 'test'
+; AVX512BW:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX512BW:  Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
 ; AVX512BW:    ir<%v0> = load from index 0
 ; AVX512BW:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-3.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-3.ll
index a7c0f2516d5ed9..25e0b8dae93e6c 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-3.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-3.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -14,12 +14,14 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
+; SSE2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 32 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
+; AVX1:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -27,6 +29,7 @@ define void @test() {
 ; AVX1:  Cost of 66 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
+; AVX2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 7 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
 ; AVX2:    ir<%v0> = load from index 0
 ; AVX2:    ir<%v1> = load from index 1
@@ -49,6 +52,7 @@ define void @test() {
 ; AVX2:    ir<%v2> = load from index 2
 ;
 ; AVX512DQ-LABEL: 'test'
+; AVX512DQ:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX512DQ:  Cost of 7 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
 ; AVX512DQ:    ir<%v0> = load from index 0
 ; AVX512DQ:    ir<%v1> = load from index 1
@@ -75,6 +79,7 @@ define void @test() {
 ; AVX512DQ:    ir<%v2> = load from index 2
 ;
 ; AVX512BW-LABEL: 'test'
+; AVX512BW:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX512BW:  Cost of 4 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
 ; AVX512BW:    ir<%v0> = load from index 0
 ; AVX512BW:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-4.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-4.ll
index 71bae49cbd9061..007f423e7a463a 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-4.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-4.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -14,12 +14,14 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
+; SSE2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 32 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
+; AVX1:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -27,6 +29,7 @@ define void @test() {
 ; AVX1:  Cost of 66 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
+; AVX2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 7 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
 ; AVX2:    ir<%v0> = load from index 0
 ; AVX2:    ir<%v1> = load from index 1
@@ -54,6 +57,7 @@ define void @test() {
 ; AVX2:    ir<%v3> = load from index 3
 ;
 ; AVX512DQ-LABEL: 'test'
+; AVX512DQ:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX512DQ:  Cost of 7 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
 ; AVX512DQ:    ir<%v0> = load from index 0
 ; AVX512DQ:    ir<%v1> = load from index 1
@@ -86,6 +90,7 @@ define void @test() {
 ; AVX512DQ:    ir<%v3> = load from index 3
 ;
 ; AVX512BW-LABEL: 'test'
+; AVX512BW:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX512BW:  Cost of 5 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
 ; AVX512BW:    ir<%v0> = load from index 0
 ; AVX512BW:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-5.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-5.ll
index 14c75eba680e9b..adba6d8145fe8e 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-5.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-5.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -14,12 +14,14 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
+; SSE2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 32 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
+; AVX1:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -27,6 +29,7 @@ define void @test() {
 ; AVX1:  Cost of 66 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
+; AVX2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -34,6 +37,7 @@ define void @test() {
 ; AVX2:  Cost of 66 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX512DQ-LABEL: 'test'
+; AVX512DQ:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX512DQ:  Cost of 24 for VF 2: INTERLEAVE-GROUP with factor 5, ir<%in0>
 ; AVX512DQ:    ir<%v0> = load from index 0
 ; AVX512DQ:    ir<%v1> = load from index 1
@@ -72,6 +76,7 @@ define void @test() {
 ; AVX512DQ:    ir<%v4> = load from index 4
 ;
 ; AVX512BW-LABEL: 'test'
+; AVX512BW:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX512BW:  Cost of 11 for VF 2: INTERLEAVE-GROUP with factor 5, ir<%in0>
 ; AVX512BW:    ir<%v0> = load from index 0
 ; AVX512BW:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-6.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-6.ll
index 727eec70f83873..019eaf1648c745 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-6.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-6.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -14,12 +14,14 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
+; SSE2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 32 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
+; AVX1:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -27,6 +29,7 @@ define void @test() {
 ; AVX1:  Cost of 66 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
+; AVX2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 16 for VF 2: INTERLEAVE-GROUP with factor 6, ir<%in0>
 ; AVX2:    ir<%v0> = load from index 0
 ; AVX2:    ir<%v1> = load from index 1
@@ -64,6 +67,7 @@ define void @test() {
 ; AVX2:    ir<%v5> = load from index 5
 ;
 ; AVX512DQ-LABEL: 'test'
+; AVX512DQ:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX512DQ:  Cost of 16 for VF 2: INTERLEAVE-GROUP with factor 6, ir<%in0>
 ; AVX512DQ:    ir<%v0> = load from index 0
 ; AVX512DQ:    ir<%v1> = load from index 1
@@ -108,6 +112,7 @@ define void @test() {
 ; AVX512DQ:    ir<%v5> = load from index 5
 ;
 ; AVX512BW-LABEL: 'test'
+; AVX512BW:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX512BW:  Cost of 13 for VF 2: INTERLEAVE-GROUP with factor 6, ir<%in0>
 ; AVX512BW:    ir<%v0> = load from index 0
 ; AVX512BW:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-7.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-7.ll
index d593db1853fe8f..869916d5ecb10f 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-7.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-7.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -14,12 +14,14 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
+; SSE2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 32 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
+; AVX1:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -27,6 +29,7 @@ define void @test() {
 ; AVX1:  Cost of 66 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
+; AVX2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -34,6 +37,7 @@ define void @test() {
 ; AVX2:  Cost of 66 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX512DQ-LABEL: 'test'
+; AVX512DQ:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX512DQ:  Cost of 33 for VF 2: INTERLEAVE-GROUP with factor 7, ir<%in0>
 ; AVX512DQ:    ir<%v0> = load from index 0
 ; AVX512DQ:    ir<%v1> = load from index 1
@@ -84,6 +88,7 @@ define void @test() {
 ; AVX512DQ:    ir<%v6> = load from index 6
 ;
 ; AVX512BW-LABEL: 'test'
+; AVX512BW:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX512BW:  Cost of 15 for VF 2: INTERLEAVE-GROUP with factor 7, ir<%in0>
 ; AVX512BW:    ir<%v0> = load from index 0
 ; AVX512BW:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-8.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-8.ll
index 5ef380a3032da3..cd583a46242919 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-8.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-8.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -14,12 +14,14 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
+; SSE2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 32 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
+; AVX1:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -27,6 +29,7 @@ define void @test() {
 ; AVX1:  Cost of 66 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
+; AVX2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -34,6 +37,7 @@ define void @test() {
 ; AVX2:  Cost of 66 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX512DQ-LABEL: 'test'
+; AVX512DQ:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX512DQ:  Cost of 34 for VF 2: INTERLEAVE-GROUP with factor 8, ir<%in0>
 ; AVX512DQ:    ir<%v0> = load from index 0
 ; AVX512DQ:    ir<%v1> = load from index 1
@@ -90,6 +94,7 @@ define void @test() {
 ; AVX512DQ:    ir<%v7> = load from index 7
 ;
 ; AVX512BW-LABEL: 'test'
+; AVX512BW:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX512BW:  Cost of 17 for VF 2: INTERLEAVE-GROUP with factor 8, ir<%in0>
 ; AVX512BW:    ir<%v0> = load from index 0
 ; AVX512BW:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-2-indices-0u.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-2-indices-0u.ll
index 85043e46d2023f..dc0ca2ec955ad2 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-2-indices-0u.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-2-indices-0u.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,6 +13,7 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
+; SSE2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 2 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
 ; SSE2:    ir<%v0> = load from index 0
 ; SSE2:  Cost of 3 for VF 4: INTERLEAVE-GROUP with factor 2, ir<%in0>
@@ -21,6 +22,7 @@ define void @test() {
 ; SSE2:  Cost of 44 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
+; AVX1:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 2 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
 ; AVX1:    ir<%v0> = load from index 0
 ; AVX1:  Cost of 2 for VF 4: INTERLEAVE-GROUP with factor 2, ir<%in0>
@@ -30,6 +32,7 @@ define void @test() {
 ; AVX1:  Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
+; AVX2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 2 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
 ; AVX2:    ir<%v0> = load from index 0
 ; AVX2:  Cost of 2 for VF 4: INTERLEAVE-GROUP with factor 2, ir<%in0>
@@ -42,6 +45,7 @@ define void @test() {
 ; AVX2:    ir<%v0> = load from index 0
 ;
 ; AVX512-LABEL: 'test'
+; AVX512:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX512:  Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:  Cost of 1 for VF 4: INTERLEAVE-GROUP with factor 2, ir<%in0>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-2.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-2.ll
index 3d8916969eade1..4eb6a98f2e53d9 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-2.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-2.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,6 +13,7 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
+; SSE2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
 ; SSE2:    ir<%v0> = load from index 0
 ; SSE2:    ir<%v1> = load from index 1
@@ -23,6 +24,7 @@ define void @test() {
 ; SSE2:  Cost of 44 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
+; AVX1:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
 ; AVX1:    ir<%v0> = load from index 0
 ; AVX1:    ir<%v1> = load from index 1
@@ -34,6 +36,7 @@ define void @test() {
 ; AVX1:  Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
+; AVX2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
 ; AVX2:    ir<%v0> = load from index 0
 ; AVX2:    ir<%v1> = load from index 1
@@ -51,6 +54,7 @@ define void @test() {
 ; AVX2:    ir<%v1> = load from index 1
 ;
 ; AVX512-LABEL: 'test'
+; AVX512:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX512:  Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-3-indices-01u.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-3-indices-01u.ll
index 02d90c5ffe2fbb..488f2e6a7928f1 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-3-indices-01u.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-3-indices-01u.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,12 +13,14 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
+; SSE2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 22 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 44 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
+; AVX1:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 17 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -26,6 +28,7 @@ define void @test() {
 ; AVX1:  Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
+; AVX2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 5 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
 ; AVX2:    ir<%v0> = load from index 0
 ; AVX2:    ir<%v1> = load from index 1
@@ -43,6 +46,7 @@ define void @test() {
 ; AVX2:    ir<%v1> = load from index 1
 ;
 ; AVX512-LABEL: 'test'
+; AVX512:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX512:  Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-3-indices-0uu.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-3-indices-0uu.ll
index e92c161c8ff960..56d22c59784990 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-3-indices-0uu.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-3-indices-0uu.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,12 +13,14 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
+; SSE2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 22 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 44 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
+; AVX1:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 17 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -26,6 +28,7 @@ define void @test() {
 ; AVX1:  Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
+; AVX2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 3 for VF 4: INTERLEAVE-GROUP with factor 3, ir<%in0>
 ; AVX2:    ir<%v0> = load from index 0
@@ -37,6 +40,7 @@ define void @test() {
 ; AVX2:    ir<%v0> = load from index 0
 ;
 ; AVX512-LABEL: 'test'
+; AVX512:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX512:  Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:  Cost of 1 for VF 4: INTERLEAVE-GROUP with factor 3, ir<%in0>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-3.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-3.ll
index b68f44855dc3ed..f68113ce79dd26 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-3.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-3.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,12 +13,14 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
+; SSE2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 22 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 44 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
+; AVX1:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 17 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -26,6 +28,7 @@ define void @test() {
 ; AVX1:  Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
+; AVX2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 6 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
 ; AVX2:    ir<%v0> = load from index 0
 ; AVX2:    ir<%v1> = load from index 1
@@ -48,6 +51,7 @@ define void @test() {
 ; AVX2:    ir<%v2> = load from index 2
 ;
 ; AVX512-LABEL: 'test'
+; AVX512:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX512:  Cost of 4 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-4-indices-012u.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-4-indices-012u.ll
index 69baf083073e40..8253f18260ab10 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-4-indices-012u.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-4-indices-012u.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,12 +13,14 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
+; SSE2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 22 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 44 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
+; AVX1:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 17 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -26,6 +28,7 @@ define void @test() {
 ; AVX1:  Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
+; AVX2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 4 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
 ; AVX2:    ir<%v0> = load from index 0
 ; AVX2:    ir<%v1> = load from index 1
@@ -48,6 +51,7 @@ define void @test() {
 ; AVX2:    ir<%v2> = load from index 2
 ;
 ; AVX512-LABEL: 'test'
+; AVX512:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX512:  Cost of 4 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-4-indices-01uu.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-4-indices-01uu.ll
index 43e097a4ffeb01..0ffde89ebbba10 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-4-indices-01uu.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-4-indices-01uu.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,12 +13,14 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
+; SSE2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 22 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 44 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
+; AVX1:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 17 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -26,6 +28,7 @@ define void @test() {
 ; AVX1:  Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
+; AVX2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
 ; AVX2:    ir<%v0> = load from index 0
 ; AVX2:    ir<%v1> = load from index 1
@@ -43,6 +46,7 @@ define void @test() {
 ; AVX2:    ir<%v1> = load from index 1
 ;
 ; AVX512-LABEL: 'test'
+; AVX512:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX512:  Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-4-indices-0uuu.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-4-indices-0uuu.ll
index e3838488d3a967..4225c0b8e7e715 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-4-indices-0uuu.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-4-indices-0uuu.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load|WIDEN ir<%v[0-9]> = load)" --filter "^  ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load|WIDEN ir<%v[0-9]> = load)" --filter "^  ir<.* = load from index"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,12 +13,14 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
+; SSE2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 22 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 44 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
+; AVX1:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 17 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -26,6 +28,7 @@ define void @test() {
 ; AVX1:  Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
+; AVX2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 2 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
 ; AVX2:    ir<%v0> = load from index 0
 ; AVX2:  Cost of 4 for VF 4: INTERLEAVE-GROUP with factor 4, ir<%in0>
@@ -38,6 +41,7 @@ define void @test() {
 ; AVX2:    ir<%v0> = load from index 0
 ;
 ; AVX512-LABEL: 'test'
+; AVX512:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX512:  Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:  Cost of 1 for VF 4: INTERLEAVE-GROUP with factor 4, ir<%in0>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-4.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-4.ll
index bdb68363cdc2d8..f1eebd79c9a620 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-4.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-4.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,12 +13,14 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
+; SSE2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 22 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 44 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
+; AVX1:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 17 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -26,6 +28,7 @@ define void @test() {
 ; AVX1:  Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
+; AVX2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 5 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
 ; AVX2:    ir<%v0> = load from index 0
 ; AVX2:    ir<%v1> = load from index 1
@@ -53,6 +56,7 @@ define void @test() {
 ; AVX2:    ir<%v3> = load from index 3
 ;
 ; AVX512-LABEL: 'test'
+; AVX512:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX512:  Cost of 5 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-5.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-5.ll
index 7f5b1687b59780..1ded6db5282eff 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-5.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-5.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,12 +13,14 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
+; SSE2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 22 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 44 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
+; AVX1:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 17 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -26,6 +28,7 @@ define void @test() {
 ; AVX1:  Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
+; AVX2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 17 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -33,6 +36,7 @@ define void @test() {
 ; AVX2:  Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX512-LABEL: 'test'
+; AVX512:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX512:  Cost of 6 for VF 2: INTERLEAVE-GROUP with factor 5, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-6.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-6.ll
index aec171a16f12ba..354b3c2b9050e5 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-6.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-6.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,12 +13,14 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
+; SSE2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 22 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 44 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
+; AVX1:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 17 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -26,6 +28,7 @@ define void @test() {
 ; AVX1:  Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
+; AVX2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 8 for VF 2: INTERLEAVE-GROUP with factor 6, ir<%in0>
 ; AVX2:    ir<%v0> = load from index 0
 ; AVX2:    ir<%v1> = load from index 1
@@ -57,6 +60,7 @@ define void @test() {
 ; AVX2:  Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX512-LABEL: 'test'
+; AVX512:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX512:  Cost of 7 for VF 2: INTERLEAVE-GROUP with factor 6, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-7.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-7.ll
index 5d4904aec87eb7..1144007cbe60fc 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-7.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-7.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,12 +13,14 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
+; SSE2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 22 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 44 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
+; AVX1:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 17 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -26,6 +28,7 @@ define void @test() {
 ; AVX1:  Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
+; AVX2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 17 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -33,6 +36,7 @@ define void @test() {
 ; AVX2:  Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX512-LABEL: 'test'
+; AVX512:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX512:  Cost of 8 for VF 2: INTERLEAVE-GROUP with factor 7, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-8.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-8.ll
index e1af00bcbe1fa2..71934416070dff 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-8.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-8.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,12 +13,14 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
+; SSE2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 22 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 44 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
+; AVX1:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 17 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -26,6 +28,7 @@ define void @test() {
 ; AVX1:  Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
+; AVX2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 48 for VF 8: INTERLEAVE-GROUP with factor 8, ir<%in0>
@@ -41,6 +44,7 @@ define void @test() {
 ; AVX2:  Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX512-LABEL: 'test'
+; AVX512:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX512:  Cost of 9 for VF 2: INTERLEAVE-GROUP with factor 8, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-2.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-2.ll
index a664eb308630b7..4dd2c42f2a228b 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-2.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-2.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load|WIDEN ir<%v[0-9]> = load)" --filter "^  ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load|WIDEN ir<%v[0-9]> = load)" --filter "^  ir<.* = load from index"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,6 +13,7 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
+; SSE2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 4 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
 ; SSE2:    ir<%v0> = load from index 0
 ; SSE2:    ir<%v1> = load from index 1
@@ -21,6 +22,7 @@ define void @test() {
 ; SSE2:  Cost of 40 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
+; AVX1:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
 ; AVX1:    ir<%v0> = load from index 0
 ; AVX1:    ir<%v1> = load from index 1
@@ -30,6 +32,7 @@ define void @test() {
 ; AVX1:  Cost of 72 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
+; AVX2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
 ; AVX2:    ir<%v0> = load from index 0
 ; AVX2:    ir<%v1> = load from index 1
@@ -47,6 +50,7 @@ define void @test() {
 ; AVX2:    ir<%v1> = load from index 1
 ;
 ; AVX512-LABEL: 'test'
+; AVX512:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX512:  Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-3.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-3.ll
index 1fbcf7559aad64..b740b83b74bea1 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-3.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-3.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,12 +13,14 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
+; SSE2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 10 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 20 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 40 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
+; AVX1:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 9 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 18 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -26,6 +28,7 @@ define void @test() {
 ; AVX1:  Cost of 72 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
+; AVX2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
 ; AVX2:    ir<%v0> = load from index 0
 ; AVX2:    ir<%v1> = load from index 1
@@ -45,6 +48,7 @@ define void @test() {
 ; AVX2:  Cost of 72 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX512-LABEL: 'test'
+; AVX512:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX512:  Cost of 4 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-4.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-4.ll
index 43adad539acd4d..12578c2293af60 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-4.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-4.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load|WIDEN ir<%v[0-9]> = load)" --filter "^  ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load|WIDEN ir<%v[0-9]> = load)" --filter "^  ir<.* = load from index"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,12 +13,14 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
+; SSE2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 10 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 20 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 40 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
+; AVX1:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 9 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 18 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -26,6 +28,7 @@ define void @test() {
 ; AVX1:  Cost of 72 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
+; AVX2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 8 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
 ; AVX2:    ir<%v0> = load from index 0
 ; AVX2:    ir<%v1> = load from index 1
@@ -49,6 +52,7 @@ define void @test() {
 ; AVX2:  Cost of 72 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX512-LABEL: 'test'
+; AVX512:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX512:  Cost of 5 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-5.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-5.ll
index 2c6c071b77be2f..0ac7b710e1a05a 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-5.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-5.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load|WIDEN ir<%v[0-9]> = load)" --filter "^  ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load|WIDEN ir<%v[0-9]> = load)" --filter "^  ir<.* = load from index"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,12 +13,14 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
+; SSE2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 10 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 20 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 40 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
+; AVX1:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 9 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 18 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -26,6 +28,7 @@ define void @test() {
 ; AVX1:  Cost of 72 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
+; AVX2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 9 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 18 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -33,6 +36,7 @@ define void @test() {
 ; AVX2:  Cost of 72 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX512-LABEL: 'test'
+; AVX512:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX512:  Cost of 14.5 for VF 2: INTERLEAVE-GROUP with factor 5, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-6.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-6.ll
index 18c252a8a3aaca..2985c0c232aaa9 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-6.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-6.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load|WIDEN ir<%v[0-9]> = load)" --filter "^  ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load|WIDEN ir<%v[0-9]> = load)" --filter "^  ir<.* = load from index"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,12 +13,14 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
+; SSE2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 10 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 20 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 40 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
+; AVX1:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 9 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 18 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -26,6 +28,7 @@ define void @test() {
 ; AVX1:  Cost of 72 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
+; AVX2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 9 for VF 2: INTERLEAVE-GROUP with factor 6, ir<%in0>
 ; AVX2:    ir<%v0> = load from index 0
 ; AVX2:    ir<%v1> = load from index 1
@@ -51,6 +54,7 @@ define void @test() {
 ; AVX2:  Cost of 72 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX512-LABEL: 'test'
+; AVX512:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX512:  Cost of 17 for VF 2: INTERLEAVE-GROUP with factor 6, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-7.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-7.ll
index 6460df78c20c02..0b0778efb01160 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-7.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-7.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load|WIDEN ir<%v[0-9]> = load)" --filter "^  ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load|WIDEN ir<%v[0-9]> = load)" --filter "^  ir<.* = load from index"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,12 +13,14 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
+; SSE2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 10 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 20 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 40 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
+; AVX1:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 9 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 18 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -26,6 +28,7 @@ define void @test() {
 ; AVX1:  Cost of 72 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
+; AVX2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 9 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 18 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -33,6 +36,7 @@ define void @test() {
 ; AVX2:  Cost of 72 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX512-LABEL: 'test'
+; AVX512:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX512:  Cost of 19.5 for VF 2: INTERLEAVE-GROUP with factor 7, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-8.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-8.ll
index d4d99b530f0108..81066a4364609b 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-8.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-8.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load|WIDEN ir<%v[0-9]> = load)" --filter "^  ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load|WIDEN ir<%v[0-9]> = load)" --filter "^  ir<.* = load from index"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,12 +13,14 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
+; SSE2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 10 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 20 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 40 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
+; AVX1:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 9 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 18 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -26,6 +28,7 @@ define void @test() {
 ; AVX1:  Cost of 72 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
+; AVX2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 9 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 18 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -33,6 +36,7 @@ define void @test() {
 ; AVX2:  Cost of 72 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX512-LABEL: 'test'
+; AVX512:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX512:  Cost of 22 for VF 2: INTERLEAVE-GROUP with factor 8, ir<%in0>
 ; AVX512:    ir<%v0> = load from index 0
 ; AVX512:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-2.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-2.ll
index 4686fbc32153a9..e61a54fa6d30d3 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-2.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-2.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -14,12 +14,14 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
+; SSE2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 23 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 47 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
+; AVX1:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -27,6 +29,7 @@ define void @test() {
 ; AVX1:  Cost of 65 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
+; AVX2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
 ; AVX2:    ir<%v0> = load from index 0
 ; AVX2:    ir<%v1> = load from index 1
@@ -44,6 +47,7 @@ define void @test() {
 ; AVX2:    ir<%v1> = load from index 1
 ;
 ; AVX512DQ-LABEL: 'test'
+; AVX512DQ:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX512DQ:  Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
 ; AVX512DQ:    ir<%v0> = load from index 0
 ; AVX512DQ:    ir<%v1> = load from index 1
@@ -64,6 +68,7 @@ define void @test() {
 ; AVX512DQ:    ir<%v1> = load from index 1
 ;
 ; AVX512BW-LABEL: 'test'
+; AVX512BW:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX512BW:  Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
 ; AVX512BW:    ir<%v0> = load from index 0
 ; AVX512BW:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-3.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-3.ll
index d5bdfeacdb30e6..b68dd3eaf9c40e 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-3.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-3.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -14,12 +14,14 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
+; SSE2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 23 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 47 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
+; AVX1:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -27,6 +29,7 @@ define void @test() {
 ; AVX1:  Cost of 65 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
+; AVX2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 5 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
 ; AVX2:    ir<%v0> = load from index 0
 ; AVX2:    ir<%v1> = load from index 1
@@ -49,6 +52,7 @@ define void @test() {
 ; AVX2:    ir<%v2> = load from index 2
 ;
 ; AVX512DQ-LABEL: 'test'
+; AVX512DQ:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX512DQ:  Cost of 5 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
 ; AVX512DQ:    ir<%v0> = load from index 0
 ; AVX512DQ:    ir<%v1> = load from index 1
@@ -75,6 +79,7 @@ define void @test() {
 ; AVX512DQ:    ir<%v2> = load from index 2
 ;
 ; AVX512BW-LABEL: 'test'
+; AVX512BW:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX512BW:  Cost of 4 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
 ; AVX512BW:    ir<%v0> = load from index 0
 ; AVX512BW:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-4.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-4.ll
index 7390b2f56010f4..fe4904678c69a6 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-4.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-4.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -14,12 +14,14 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
+; SSE2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 23 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 47 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
+; AVX1:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -27,6 +29,7 @@ define void @test() {
 ; AVX1:  Cost of 65 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
+; AVX2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 5 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
 ; AVX2:    ir<%v0> = load from index 0
 ; AVX2:    ir<%v1> = load from index 1
@@ -54,6 +57,7 @@ define void @test() {
 ; AVX2:    ir<%v3> = load from index 3
 ;
 ; AVX512DQ-LABEL: 'test'
+; AVX512DQ:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX512DQ:  Cost of 5 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
 ; AVX512DQ:    ir<%v0> = load from index 0
 ; AVX512DQ:    ir<%v1> = load from index 1
@@ -86,6 +90,7 @@ define void @test() {
 ; AVX512DQ:    ir<%v3> = load from index 3
 ;
 ; AVX512BW-LABEL: 'test'
+; AVX512BW:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX512BW:  Cost of 5 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
 ; AVX512BW:    ir<%v0> = load from index 0
 ; AVX512BW:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-5.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-5.ll
index 3fd3b9e11017d6..5240bfbdfc0eb6 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-5.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-5.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -14,12 +14,14 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
+; SSE2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 23 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 47 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
+; AVX1:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -27,6 +29,7 @@ define void @test() {
 ; AVX1:  Cost of 65 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
+; AVX2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -34,6 +37,7 @@ define void @test() {
 ; AVX2:  Cost of 65 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX512DQ-LABEL: 'test'
+; AVX512DQ:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX512DQ:  Cost of 22 for VF 2: INTERLEAVE-GROUP with factor 5, ir<%in0>
 ; AVX512DQ:    ir<%v0> = load from index 0
 ; AVX512DQ:    ir<%v1> = load from index 1
@@ -72,6 +76,7 @@ define void @test() {
 ; AVX512DQ:    ir<%v4> = load from index 4
 ;
 ; AVX512BW-LABEL: 'test'
+; AVX512BW:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX512BW:  Cost of 6 for VF 2: INTERLEAVE-GROUP with factor 5, ir<%in0>
 ; AVX512BW:    ir<%v0> = load from index 0
 ; AVX512BW:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-6.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-6.ll
index d8b77269cda803..1f1e568b42f290 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-6.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-6.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -14,12 +14,14 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
+; SSE2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 23 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 47 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
+; AVX1:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -27,6 +29,7 @@ define void @test() {
 ; AVX1:  Cost of 65 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
+; AVX2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 8 for VF 2: INTERLEAVE-GROUP with factor 6, ir<%in0>
 ; AVX2:    ir<%v0> = load from index 0
 ; AVX2:    ir<%v1> = load from index 1
@@ -64,6 +67,7 @@ define void @test() {
 ; AVX2:    ir<%v5> = load from index 5
 ;
 ; AVX512DQ-LABEL: 'test'
+; AVX512DQ:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX512DQ:  Cost of 8 for VF 2: INTERLEAVE-GROUP with factor 6, ir<%in0>
 ; AVX512DQ:    ir<%v0> = load from index 0
 ; AVX512DQ:    ir<%v1> = load from index 1
@@ -108,6 +112,7 @@ define void @test() {
 ; AVX512DQ:    ir<%v5> = load from index 5
 ;
 ; AVX512BW-LABEL: 'test'
+; AVX512BW:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX512BW:  Cost of 7 for VF 2: INTERLEAVE-GROUP with factor 6, ir<%in0>
 ; AVX512BW:    ir<%v0> = load from index 0
 ; AVX512BW:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-7.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-7.ll
index e2e3dedca27c8d..2565a4646120c7 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-7.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-7.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -14,12 +14,14 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
+; SSE2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 23 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 47 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
+; AVX1:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -27,6 +29,7 @@ define void @test() {
 ; AVX1:  Cost of 65 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
+; AVX2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -34,6 +37,7 @@ define void @test() {
 ; AVX2:  Cost of 65 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX512DQ-LABEL: 'test'
+; AVX512DQ:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX512DQ:  Cost of 31 for VF 2: INTERLEAVE-GROUP with factor 7, ir<%in0>
 ; AVX512DQ:    ir<%v0> = load from index 0
 ; AVX512DQ:    ir<%v1> = load from index 1
@@ -84,6 +88,7 @@ define void @test() {
 ; AVX512DQ:    ir<%v6> = load from index 6
 ;
 ; AVX512BW-LABEL: 'test'
+; AVX512BW:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX512BW:  Cost of 8 for VF 2: INTERLEAVE-GROUP with factor 7, ir<%in0>
 ; AVX512BW:    ir<%v0> = load from index 0
 ; AVX512BW:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-8.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-8.ll
index 1fd2c4898762d9..375c4271249eb3 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-8.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-8.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^  ir<.* = load from index"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -14,12 +14,14 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
+; SSE2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 23 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
 ; SSE2:  Cost of 47 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX1-LABEL: 'test'
+; AVX1:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX1:  Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -27,6 +29,7 @@ define void @test() {
 ; AVX1:  Cost of 65 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX2-LABEL: 'test'
+; AVX2:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
 ; AVX2:  Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -34,6 +37,7 @@ define void @test() {
 ; AVX2:  Cost of 65 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
 ;
 ; AVX512DQ-LABEL: 'test'
+; AVX512DQ:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX512DQ:  Cost of 33 for VF 2: INTERLEAVE-GROUP with factor 8, ir<%in0>
 ; AVX512DQ:    ir<%v0> = load from index 0
 ; AVX512DQ:    ir<%v1> = load from index 1
@@ -90,6 +94,7 @@ define void @test() {
 ; AVX512DQ:    ir<%v7> = load from index 7
 ;
 ; AVX512BW-LABEL: 'test'
+; AVX512BW:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; AVX512BW:  Cost of 9 for VF 2: INTERLEAVE-GROUP with factor 8, ir<%in0>
 ; AVX512BW:    ir<%v0> = load from index 0
 ; AVX512BW:    ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-gather-i32-with-i8-index.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-gather-i32-with-i8-index.ll
index 3d33bc34dda674..a109a3acd3bc4d 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-gather-i32-with-i8-index.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-gather-i32-with-i8-index.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF [0-9]+ For instruction:\s*%valB.loaded = load i32, ptr %inB, align 4" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: .* ir<%valB.loaded> = load"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: .* ir<%valB.loaded> = load"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=SSE
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse4.2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=SSE
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=AVX1
@@ -17,14 +17,14 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE-LABEL: 'test'
-; SSE:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; SSE:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB>
 ; SSE:  Cost of 3000000 for VF 2: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ; SSE:  Cost of 3000000 for VF 4: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ; SSE:  Cost of 3000000 for VF 8: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ; SSE:  Cost of 3000000 for VF 16: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX1:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB>
 ; AVX1:  Cost of 3000000 for VF 2: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ; AVX1:  Cost of 3000000 for VF 4: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ; AVX1:  Cost of 3000000 for VF 8: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
@@ -32,7 +32,7 @@ define void @test() {
 ; AVX1:  Cost of 3000000 for VF 32: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ;
 ; AVX2-SLOWGATHER-LABEL: 'test'
-; AVX2-SLOWGATHER:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX2-SLOWGATHER:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB>
 ; AVX2-SLOWGATHER:  Cost of 3000000 for VF 2: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ; AVX2-SLOWGATHER:  Cost of 3000000 for VF 4: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ; AVX2-SLOWGATHER:  Cost of 3000000 for VF 8: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
@@ -40,21 +40,21 @@ define void @test() {
 ; AVX2-SLOWGATHER:  Cost of 3000000 for VF 32: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ;
 ; AVX2-FASTGATHER-LABEL: 'test'
-; AVX2-FASTGATHER:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
-; AVX2-FASTGATHER:  Cost of 4 for VF 2: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
-; AVX2-FASTGATHER:  Cost of 6 for VF 4: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
-; AVX2-FASTGATHER:  Cost of 12 for VF 8: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
-; AVX2-FASTGATHER:  Cost of 24 for VF 16: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
-; AVX2-FASTGATHER:  Cost of 48 for VF 32: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX2-FASTGATHER:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB>
+; AVX2-FASTGATHER:  Cost of 4 for VF 2: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
+; AVX2-FASTGATHER:  Cost of 6 for VF 4: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
+; AVX2-FASTGATHER:  Cost of 12 for VF 8: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
+; AVX2-FASTGATHER:  Cost of 24 for VF 16: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
+; AVX2-FASTGATHER:  Cost of 48 for VF 32: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
 ;
 ; AVX512-LABEL: 'test'
-; AVX512:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
-; AVX512:  Cost of 8 for VF 2: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
-; AVX512:  Cost of 17 for VF 4: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
-; AVX512:  Cost of 10 for VF 8: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
-; AVX512:  Cost of 18 for VF 16: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
-; AVX512:  Cost of 36 for VF 32: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
-; AVX512:  Cost of 72 for VF 64: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX512:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB>
+; AVX512:  Cost of 8 for VF 2: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
+; AVX512:  Cost of 17 for VF 4: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
+; AVX512:  Cost of 10 for VF 8: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
+; AVX512:  Cost of 18 for VF 16: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
+; AVX512:  Cost of 36 for VF 32: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
+; AVX512:  Cost of 72 for VF 64: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
 ;
 entry:
   br label %for.body
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-gather-i64-with-i8-index.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-gather-i64-with-i8-index.ll
index b0fec98fcdf51d..a957d896529711 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-gather-i64-with-i8-index.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-gather-i64-with-i8-index.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF [0-9]+ For instruction:\s*%valB.loaded = load i64, ptr %inB, align 8" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: .* ir<%valB.loaded> = load"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: .* ir<%valB.loaded> = load"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=SSE
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse4.2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=SSE
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=AVX1
@@ -17,14 +17,14 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE-LABEL: 'test'
-; SSE:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; SSE:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB>
 ; SSE:  Cost of 3000000 for VF 2: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ; SSE:  Cost of 3000000 for VF 4: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ; SSE:  Cost of 3000000 for VF 8: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ; SSE:  Cost of 3000000 for VF 16: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ;
 ; AVX1-LABEL: 'test'
-; AVX1:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX1:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB>
 ; AVX1:  Cost of 3000000 for VF 2: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ; AVX1:  Cost of 3000000 for VF 4: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ; AVX1:  Cost of 3000000 for VF 8: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
@@ -32,7 +32,7 @@ define void @test() {
 ; AVX1:  Cost of 3000000 for VF 32: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ;
 ; AVX2-SLOWGATHER-LABEL: 'test'
-; AVX2-SLOWGATHER:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX2-SLOWGATHER:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB>
 ; AVX2-SLOWGATHER:  Cost of 3000000 for VF 2: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ; AVX2-SLOWGATHER:  Cost of 3000000 for VF 4: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ; AVX2-SLOWGATHER:  Cost of 3000000 for VF 8: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
@@ -40,21 +40,21 @@ define void @test() {
 ; AVX2-SLOWGATHER:  Cost of 3000000 for VF 32: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
 ;
 ; AVX2-FASTGATHER-LABEL: 'test'
-; AVX2-FASTGATHER:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
-; AVX2-FASTGATHER:  Cost of 4 for VF 2: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
-; AVX2-FASTGATHER:  Cost of 6 for VF 4: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
-; AVX2-FASTGATHER:  Cost of 12 for VF 8: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
-; AVX2-FASTGATHER:  Cost of 24 for VF 16: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
-; AVX2-FASTGATHER:  Cost of 48 for VF 32: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX2-FASTGATHER:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB>
+; AVX2-FASTGATHER:  Cost of 4 for VF 2: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
+; AVX2-FASTGATHER:  Cost of 6 for VF 4: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
+; AVX2-FASTGATHER:  Cost of 12 for VF 8: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
+; AVX2-FASTGATHER:  Cost of 24 for VF 16: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
+; AVX2-FASTGATHER:  Cost of 48 for VF 32: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
 ;
 ; AVX512-LABEL: 'test'
-; AVX512:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
-; AVX512:  Cost of 8 for VF 2: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
-; AVX512:  Cost of 18 for VF 4: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
-; AVX512:  Cost of 10 for VF 8: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
-; AVX512:  Cost of 20 for VF 16: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
-; AVX512:  Cost of 40 for VF 32: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
-; AVX512:  Cost of 80 for VF 64: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX512:  Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB>
+; AVX512:  Cost of 8 for VF 2: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
+; AVX512:  Cost of 18 for VF 4: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
+; AVX512:  Cost of 10 for VF 8: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
+; AVX512:  Cost of 20 for VF 16: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
+; AVX512:  Cost of 40 for VF 32: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
+; AVX512:  Cost of 80 for VF 64: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
 ;
 entry:
   br label %for.body
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-interleaved-load-i16.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-interleaved-load-i16.ll
index 52d05b807ba827..fb730224a01aa0 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-interleaved-load-i16.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-interleaved-load-i16.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%i[2,4] = load i16, ptr %[a-zA-Z0-7]+, align 2" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (REPLICATE ir<%i[24]> = load|INTERLEAVE-GROUP with factor [0-9]+)" --filter "^  ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%i[24]> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (REPLICATE ir<%i[24]> = load|INTERLEAVE-GROUP with factor [0-9]+)" --filter "^  ir<.* = load from index"
 ; RUN: opt -passes=loop-vectorize -enable-interleaved-mem-accesses -tail-folding-policy=must-fold-tail -S -mcpu=skx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=DISABLED_MASKED_STRIDED
 ; RUN: opt -passes=loop-vectorize -enable-interleaved-mem-accesses -enable-masked-interleaved-mem-accesses -tail-folding-policy=must-fold-tail -S -mcpu=skx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=ENABLED_MASKED_STRIDED
 ; REQUIRES: asserts
@@ -20,6 +20,8 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test1(ptr noalias nocapture %points, ptr noalias nocapture readonly %x, ptr noalias nocapture readonly %y) {
 ; DISABLED_MASKED_STRIDED-LABEL: 'test1'
+; DISABLED_MASKED_STRIDED:  Cost of 1 for VF 1: EMIT-SCALAR ir<%i2> = load ir<%arrayidx2>
+; DISABLED_MASKED_STRIDED:  Cost of 1 for VF 1: EMIT-SCALAR ir<%i4> = load ir<%arrayidx7>
 ; DISABLED_MASKED_STRIDED:  Cost of 6 for VF 2: REPLICATE ir<%i2> = load ir<%arrayidx2>
 ; DISABLED_MASKED_STRIDED:  Cost of 6 for VF 2: REPLICATE ir<%i4> = load ir<%arrayidx7>
 ; DISABLED_MASKED_STRIDED:  Cost of 13 for VF 4: REPLICATE ir<%i2> = load ir<%arrayidx2>
@@ -30,6 +32,8 @@ define void @test1(ptr noalias nocapture %points, ptr noalias nocapture readonly
 ; DISABLED_MASKED_STRIDED:  Cost of 55 for VF 16: REPLICATE ir<%i4> = load ir<%arrayidx7>
 ;
 ; ENABLED_MASKED_STRIDED-LABEL: 'test1'
+; ENABLED_MASKED_STRIDED:  Cost of 1 for VF 1: EMIT-SCALAR ir<%i2> = load ir<%arrayidx2>
+; ENABLED_MASKED_STRIDED:  Cost of 1 for VF 1: EMIT-SCALAR ir<%i4> = load ir<%arrayidx7>
 ; ENABLED_MASKED_STRIDED:  Cost of 8 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%arrayidx2>
 ; ENABLED_MASKED_STRIDED:    ir<%i2> = load from index 0
 ; ENABLED_MASKED_STRIDED:    ir<%i4> = load from index 1
@@ -77,6 +81,8 @@ for.end:
 
 define void @test2(ptr noalias nocapture %points, i32 %numPoints, ptr noalias nocapture readonly %x, ptr noalias nocapture readonly %y) {
 ; DISABLED_MASKED_STRIDED-LABEL: 'test2'
+; DISABLED_MASKED_STRIDED:  Cost of 1 for VF 1: EMIT-SCALAR ir<%i2> = load ir<%arrayidx2>
+; DISABLED_MASKED_STRIDED:  Cost of 1 for VF 1: EMIT-SCALAR ir<%i4> = load ir<%arrayidx7>
 ; DISABLED_MASKED_STRIDED:  Cost of 3000000 for VF 2: REPLICATE ir<%i2> = load ir<%arrayidx2> (S->V)
 ; DISABLED_MASKED_STRIDED:  Cost of 3000000 for VF 2: REPLICATE ir<%i4> = load ir<%arrayidx7> (S->V)
 ; DISABLED_MASKED_STRIDED:  Cost of 3000000 for VF 4: REPLICATE ir<%i2> = load ir<%arrayidx2> (S->V)
@@ -87,6 +93,8 @@ define void @test2(ptr noalias nocapture %points, i32 %numPoints, ptr noalias no
 ; DISABLED_MASKED_STRIDED:  Cost of 3000000 for VF 16: REPLICATE ir<%i4> = load ir<%arrayidx7> (S->V)
 ;
 ; ENABLED_MASKED_STRIDED-LABEL: 'test2'
+; ENABLED_MASKED_STRIDED:  Cost of 1 for VF 1: EMIT-SCALAR ir<%i2> = load ir<%arrayidx2>
+; ENABLED_MASKED_STRIDED:  Cost of 1 for VF 1: EMIT-SCALAR ir<%i4> = load ir<%arrayidx7>
 ; ENABLED_MASKED_STRIDED:  Cost of 8 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%arrayidx2>, vp<[[VP8:%[0-9]+]]>
 ; ENABLED_MASKED_STRIDED:    ir<%i2> = load from index 0
 ; ENABLED_MASKED_STRIDED:    ir<%i4> = load from index 1
@@ -144,20 +152,24 @@ for.end:
 
 define void @test(ptr noalias nocapture %points, ptr noalias nocapture readonly %x, ptr noalias nocapture readnone %y) {
 ; DISABLED_MASKED_STRIDED-LABEL: 'test'
+; DISABLED_MASKED_STRIDED:  Cost of 1 for VF 1: EMIT-SCALAR ir<%i2> = load ir<%arrayidx>
+; DISABLED_MASKED_STRIDED:  Cost of 1 for VF 1: EMIT-SCALAR ir<%i4> = load ir<%arrayidx6>
 ; DISABLED_MASKED_STRIDED:  Cost of 3000000 for VF 2: REPLICATE ir<%i4> = load ir<%arrayidx6> (S->V)
 ; DISABLED_MASKED_STRIDED:  Cost of 3000000 for VF 4: REPLICATE ir<%i4> = load ir<%arrayidx6> (S->V)
 ; DISABLED_MASKED_STRIDED:  Cost of 3000000 for VF 8: REPLICATE ir<%i4> = load ir<%arrayidx6> (S->V)
 ; DISABLED_MASKED_STRIDED:  Cost of 3000000 for VF 16: REPLICATE ir<%i4> = load ir<%arrayidx6> (S->V)
 ;
 ; ENABLED_MASKED_STRIDED-LABEL: 'test'
+; ENABLED_MASKED_STRIDED:  Cost of 1 for VF 1: EMIT-SCALAR ir<%i2> = load ir<%arrayidx>
+; ENABLED_MASKED_STRIDED:  Cost of 1 for VF 1: EMIT-SCALAR ir<%i4> = load ir<%arrayidx6>
 ; ENABLED_MASKED_STRIDED:  Cost of 7 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%arrayidx6>, ir<%cmp1>
-; ENABLED_MASKED_STRIDED:    ir<%i4> = load from index 0 (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; ENABLED_MASKED_STRIDED:    ir<%i4> = load from index 0
 ; ENABLED_MASKED_STRIDED:  Cost of 9 for VF 4: INTERLEAVE-GROUP with factor 3, ir<%arrayidx6>, ir<%cmp1>
-; ENABLED_MASKED_STRIDED:    ir<%i4> = load from index 0 (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; ENABLED_MASKED_STRIDED:    ir<%i4> = load from index 0
 ; ENABLED_MASKED_STRIDED:  Cost of 9 for VF 8: INTERLEAVE-GROUP with factor 3, ir<%arrayidx6>, ir<%cmp1>
-; ENABLED_MASKED_STRIDED:    ir<%i4> = load from index 0 (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; ENABLED_MASKED_STRIDED:    ir<%i4> = load from index 0
 ; ENABLED_MASKED_STRIDED:  Cost of 14 for VF 16: INTERLEAVE-GROUP with factor 3, ir<%arrayidx6>, ir<%cmp1>
-; ENABLED_MASKED_STRIDED:    ir<%i4> = load from index 0 (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; ENABLED_MASKED_STRIDED:    ir<%i4> = load from index 0
 ;
 entry:
   br label %for.body
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-interleaved-store-i16.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-interleaved-store-i16.ll
index 6b7b2919a9899e..ed9e7a03aec67a 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-interleaved-store-i16.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-interleaved-store-i16.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*store i16 %[0,2], ptr %[a-zA-Z0-7]+, align 2" --filter "Cost of [1-9][0-9]*(.[0-9]+)? for VF [0-9]+: (profitable to scalarize\s+store i16 %[02]|REPLICATE store ir<%[02]>|INTERLEAVE-GROUP with factor [0-9]+)"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT store ir<%[02]>" --filter "Cost of [1-9][0-9]*(.[0-9]+)? for VF [0-9]+: (profitable to scalarize\s+store i16 %[02]|REPLICATE store ir<%[02]>|INTERLEAVE-GROUP with factor [0-9]+)"
 ; RUN: opt -passes=loop-vectorize -enable-interleaved-mem-accesses -tail-folding-policy=must-fold-tail -S -mcpu=skx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=DISABLED_MASKED_STRIDED
 ; RUN: opt -passes=loop-vectorize -enable-interleaved-mem-accesses -enable-masked-interleaved-mem-accesses -tail-folding-policy=must-fold-tail -S -mcpu=skx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=ENABLED_MASKED_STRIDED
 ; REQUIRES: asserts
@@ -20,6 +20,8 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test1(ptr noalias nocapture %points, ptr noalias nocapture readonly %x, ptr noalias nocapture readonly %y) {
 ; DISABLED_MASKED_STRIDED-LABEL: 'test1'
+; DISABLED_MASKED_STRIDED:  Cost of 1 for VF 1: EMIT store ir<%0>, ir<%arrayidx2>
+; DISABLED_MASKED_STRIDED:  Cost of 1 for VF 1: EMIT store ir<%2>, ir<%arrayidx7>
 ; DISABLED_MASKED_STRIDED:  Cost of 6 for VF 2: REPLICATE store ir<%0>, ir<%arrayidx2>
 ; DISABLED_MASKED_STRIDED:  Cost of 6 for VF 2: REPLICATE store ir<%2>, ir<%arrayidx7>
 ; DISABLED_MASKED_STRIDED:  Cost of 13 for VF 4: REPLICATE store ir<%0>, ir<%arrayidx2>
@@ -30,6 +32,8 @@ define void @test1(ptr noalias nocapture %points, ptr noalias nocapture readonly
 ; DISABLED_MASKED_STRIDED:  Cost of 55 for VF 16: REPLICATE store ir<%2>, ir<%arrayidx7>
 ;
 ; ENABLED_MASKED_STRIDED-LABEL: 'test1'
+; ENABLED_MASKED_STRIDED:  Cost of 1 for VF 1: EMIT store ir<%0>, ir<%arrayidx2>
+; ENABLED_MASKED_STRIDED:  Cost of 1 for VF 1: EMIT store ir<%2>, ir<%arrayidx7>
 ; ENABLED_MASKED_STRIDED:  Cost of 6 for VF 2: REPLICATE store ir<%0>, ir<%arrayidx2>
 ; ENABLED_MASKED_STRIDED:  Cost of 6 for VF 2: REPLICATE store ir<%2>, ir<%arrayidx7>
 ; ENABLED_MASKED_STRIDED:  Cost of 14 for VF 4: INTERLEAVE-GROUP with factor 4, ir<%arrayidx2>
@@ -70,6 +74,8 @@ for.end:
 
 define void @test2(ptr noalias nocapture %points, i32 %numPoints, ptr noalias nocapture readonly %x, ptr noalias nocapture readonly %y) {
 ; DISABLED_MASKED_STRIDED-LABEL: 'test2'
+; DISABLED_MASKED_STRIDED:  Cost of 1 for VF 1: EMIT store ir<%0>, ir<%arrayidx2>
+; DISABLED_MASKED_STRIDED:  Cost of 1 for VF 1: EMIT store ir<%2>, ir<%arrayidx7>
 ; DISABLED_MASKED_STRIDED:  Cost of 3000000 for VF 2: REPLICATE store ir<%0>, ir<%arrayidx2>
 ; DISABLED_MASKED_STRIDED:  Cost of 3000000 for VF 2: REPLICATE store ir<%2>, ir<%arrayidx7>
 ; DISABLED_MASKED_STRIDED:  Cost of 3000000 for VF 4: REPLICATE store ir<%0>, ir<%arrayidx2>
@@ -80,6 +86,8 @@ define void @test2(ptr noalias nocapture %points, i32 %numPoints, ptr noalias no
 ; DISABLED_MASKED_STRIDED:  Cost of 3000000 for VF 16: REPLICATE store ir<%2>, ir<%arrayidx7>
 ;
 ; ENABLED_MASKED_STRIDED-LABEL: 'test2'
+; ENABLED_MASKED_STRIDED:  Cost of 1 for VF 1: EMIT store ir<%0>, ir<%arrayidx2>
+; ENABLED_MASKED_STRIDED:  Cost of 1 for VF 1: EMIT store ir<%2>, ir<%arrayidx7>
 ; ENABLED_MASKED_STRIDED:  Cost of 13 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%arrayidx2>, vp<[[VP8:%[0-9]+]]>
 ; ENABLED_MASKED_STRIDED:  Cost of 14 for VF 4: INTERLEAVE-GROUP with factor 4, ir<%arrayidx2>, vp<[[VP8]]>
 ; ENABLED_MASKED_STRIDED:  Cost of 14 for VF 8: INTERLEAVE-GROUP with factor 4, ir<%arrayidx2>, vp<[[VP8]]>
@@ -129,12 +137,14 @@ for.end:
 
 define void @test(ptr noalias nocapture %points, ptr noalias nocapture readonly %x, ptr noalias nocapture readnone %y) {
 ; DISABLED_MASKED_STRIDED-LABEL: 'test'
+; DISABLED_MASKED_STRIDED:  Cost of 1 for VF 1: EMIT store ir<%0>, ir<%arrayidx6>
 ; DISABLED_MASKED_STRIDED:  Cost of 2 for VF 2: profitable to scalarize store i16 %0, ptr %arrayidx6, align 2
 ; DISABLED_MASKED_STRIDED:  Cost of 4 for VF 4: profitable to scalarize store i16 %0, ptr %arrayidx6, align 2
 ; DISABLED_MASKED_STRIDED:  Cost of 8 for VF 8: profitable to scalarize store i16 %0, ptr %arrayidx6, align 2
 ; DISABLED_MASKED_STRIDED:  Cost of 16.5 for VF 16: profitable to scalarize store i16 %0, ptr %arrayidx6, align 2
 ;
 ; ENABLED_MASKED_STRIDED-LABEL: 'test'
+; ENABLED_MASKED_STRIDED:  Cost of 1 for VF 1: EMIT store ir<%0>, ir<%arrayidx6>
 ; ENABLED_MASKED_STRIDED:  Cost of 2 for VF 2: profitable to scalarize store i16 %0, ptr %arrayidx6, align 2
 ; ENABLED_MASKED_STRIDED:  Cost of 4 for VF 4: profitable to scalarize store i16 %0, ptr %arrayidx6, align 2
 ; ENABLED_MASKED_STRIDED:  Cost of 12 for VF 8: INTERLEAVE-GROUP with factor 3, ir<%arrayidx6>, ir<%cmp1>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i16.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i16.ll
index a4a86cb40b3ea0..1f03201e56e285 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i16.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i16.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF [0-9]+ For instruction:\s*%valB.loaded = load i16, ptr %inB, align 2" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: .* ir<%valB.loaded> = load"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: .* ir<%valB.loaded> = load"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output -vplan-print-metadata=false < %s 2>&1 | FileCheck %s --check-prefixes=SSE
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse4.2 --debug-only=loop-vectorize --disable-output -vplan-print-metadata=false < %s 2>&1 | FileCheck %s --check-prefixes=SSE
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output -vplan-print-metadata=false < %s 2>&1 | FileCheck %s --check-prefixes=AVX1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i32.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i32.ll
index be448f4b9359f3..51908f4fad8b09 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i32.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i32.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF [0-9]+ For instruction:\s*%valB.loaded = load i32, ptr %inB, align 4" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: .* ir<%valB.loaded> = load"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: .* ir<%valB.loaded> = load"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output -vplan-print-metadata=false < %s 2>&1 | FileCheck %s --check-prefixes=SSE
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse4.2 --debug-only=loop-vectorize --disable-output -vplan-print-metadata=false < %s 2>&1 | FileCheck %s --check-prefixes=SSE
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output -vplan-print-metadata=false < %s 2>&1 | FileCheck %s --check-prefixes=AVX1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i64.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i64.ll
index 9900b8f2637def..8df28a0b608f1d 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i64.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i64.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF [0-9]+ For instruction:\s*%valB.loaded = load i64, ptr %inB, align 8" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: .* ir<%valB.loaded> = load"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: .* ir<%valB.loaded> = load"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output -vplan-print-metadata=false < %s 2>&1 | FileCheck %s --check-prefixes=SSE
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse4.2 --debug-only=loop-vectorize --disable-output -vplan-print-metadata=false < %s 2>&1 | FileCheck %s --check-prefixes=SSE
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output -vplan-print-metadata=false < %s 2>&1 | FileCheck %s --check-prefixes=AVX1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i8.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i8.ll
index 594d4766b43a4d..1118023a958f06 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i8.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i8.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF [0-9]+ For instruction:\s*%valB.loaded = load i8, ptr %inB, align 1" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: .* ir<%valB.loaded> = load"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: .* ir<%valB.loaded> = load"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output -vplan-print-metadata=false < %s 2>&1 | FileCheck %s --check-prefixes=SSE
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse4.2 --debug-only=loop-vectorize --disable-output -vplan-print-metadata=false < %s 2>&1 | FileCheck %s --check-prefixes=SSE
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output -vplan-print-metadata=false < %s 2>&1 | FileCheck %s --check-prefixes=AVX1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-scatter-i32-with-i8-index.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-scatter-i32-with-i8-index.ll
index b026016d1331bd..86d317982d1ea5 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-scatter-i32-with-i8-index.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-scatter-i32-with-i8-index.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*store i32 %valB, ptr %out" --filter "Cost of [1-9][0-9]*(.[0-9]+)? for VF [0-9]+: (profitable to scalarize\s+store i32 %valB|WIDEN store .*, ir<%valB>|REPLICATE store ir<%valB>)"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT store ir<%valB>" --filter "Cost of [1-9][0-9]*(.[0-9]+)? for VF [0-9]+: (profitable to scalarize\s+store i32 %valB|WIDEN store .*, ir<%valB>|REPLICATE store ir<%valB>)"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=SSE2
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse4.2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=SSE42
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=AVX1
@@ -17,18 +17,21 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
+; SSE2:  Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
 ; SSE2:  Cost of 2.5 for VF 2: profitable to scalarize store i32 %valB, ptr %out, align 4
 ; SSE2:  Cost of 5.5 for VF 4: profitable to scalarize store i32 %valB, ptr %out, align 4
 ; SSE2:  Cost of 11 for VF 8: profitable to scalarize store i32 %valB, ptr %out, align 4
 ; SSE2:  Cost of 22 for VF 16: profitable to scalarize store i32 %valB, ptr %out, align 4
 ;
 ; SSE42-LABEL: 'test'
+; SSE42:  Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
 ; SSE42:  Cost of 2 for VF 2: profitable to scalarize store i32 %valB, ptr %out, align 4
 ; SSE42:  Cost of 4 for VF 4: profitable to scalarize store i32 %valB, ptr %out, align 4
 ; SSE42:  Cost of 8 for VF 8: profitable to scalarize store i32 %valB, ptr %out, align 4
 ; SSE42:  Cost of 16 for VF 16: profitable to scalarize store i32 %valB, ptr %out, align 4
 ;
 ; AVX1-LABEL: 'test'
+; AVX1:  Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
 ; AVX1:  Cost of 2 for VF 2: profitable to scalarize store i32 %valB, ptr %out, align 4
 ; AVX1:  Cost of 4 for VF 4: profitable to scalarize store i32 %valB, ptr %out, align 4
 ; AVX1:  Cost of 8.5 for VF 8: profitable to scalarize store i32 %valB, ptr %out, align 4
@@ -36,6 +39,7 @@ define void @test() {
 ; AVX1:  Cost of 34 for VF 32: profitable to scalarize store i32 %valB, ptr %out, align 4
 ;
 ; AVX2-LABEL: 'test'
+; AVX2:  Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
 ; AVX2:  Cost of 2 for VF 2: profitable to scalarize store i32 %valB, ptr %out, align 4
 ; AVX2:  Cost of 4 for VF 4: profitable to scalarize store i32 %valB, ptr %out, align 4
 ; AVX2:  Cost of 8.5 for VF 8: profitable to scalarize store i32 %valB, ptr %out, align 4
@@ -43,12 +47,13 @@ define void @test() {
 ; AVX2:  Cost of 34 for VF 32: profitable to scalarize store i32 %valB, ptr %out, align 4
 ;
 ; AVX512-LABEL: 'test'
+; AVX512:  Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
 ; AVX512:  Cost of 5 for VF 2: REPLICATE store ir<%valB>, ir<%out>
 ; AVX512:  Cost of 10.5 for VF 4: REPLICATE store ir<%valB>, ir<%out>
-; AVX512:  Cost of 10 for VF 8: WIDEN store ir<%out>, ir<%valB>, ir<%canStore> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
-; AVX512:  Cost of 18 for VF 16: WIDEN store ir<%out>, ir<%valB>, ir<%canStore> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
-; AVX512:  Cost of 36 for VF 32: WIDEN store ir<%out>, ir<%valB>, ir<%canStore> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
-; AVX512:  Cost of 72 for VF 64: WIDEN store ir<%out>, ir<%valB>, ir<%canStore> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX512:  Cost of 10 for VF 8: WIDEN store ir<%out>, ir<%valB>, ir<%canStore>
+; AVX512:  Cost of 18 for VF 16: WIDEN store ir<%out>, ir<%valB>, ir<%canStore>
+; AVX512:  Cost of 36 for VF 32: WIDEN store ir<%out>, ir<%valB>, ir<%canStore>
+; AVX512:  Cost of 72 for VF 64: WIDEN store ir<%out>, ir<%valB>, ir<%canStore>
 ;
 entry:
   br label %for.body
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-scatter-i64-with-i8-index.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-scatter-i64-with-i8-index.ll
index f9c6a5b14c8024..216a24086891b9 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-scatter-i64-with-i8-index.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-scatter-i64-with-i8-index.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*store i64 %valB, ptr %out" --filter "Cost of [1-9][0-9]*(.[0-9]+)? for VF [0-9]+: (profitable to scalarize\s+store i64 %valB|WIDEN store .*, ir<%valB>|REPLICATE store ir<%valB>)"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT store ir<%valB>" --filter "Cost of [1-9][0-9]*(.[0-9]+)? for VF [0-9]+: (profitable to scalarize\s+store i64 %valB|WIDEN store .*, ir<%valB>|REPLICATE store ir<%valB>)"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=SSE2
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse4.2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=SSE42
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=AVX1
@@ -17,18 +17,21 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
+; SSE2:  Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
 ; SSE2:  Cost of 2.5 for VF 2: profitable to scalarize store i64 %valB, ptr %out, align 8
 ; SSE2:  Cost of 5 for VF 4: profitable to scalarize store i64 %valB, ptr %out, align 8
 ; SSE2:  Cost of 10 for VF 8: profitable to scalarize store i64 %valB, ptr %out, align 8
 ; SSE2:  Cost of 20 for VF 16: profitable to scalarize store i64 %valB, ptr %out, align 8
 ;
 ; SSE42-LABEL: 'test'
+; SSE42:  Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
 ; SSE42:  Cost of 2 for VF 2: profitable to scalarize store i64 %valB, ptr %out, align 8
 ; SSE42:  Cost of 4 for VF 4: profitable to scalarize store i64 %valB, ptr %out, align 8
 ; SSE42:  Cost of 8 for VF 8: profitable to scalarize store i64 %valB, ptr %out, align 8
 ; SSE42:  Cost of 16 for VF 16: profitable to scalarize store i64 %valB, ptr %out, align 8
 ;
 ; AVX1-LABEL: 'test'
+; AVX1:  Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
 ; AVX1:  Cost of 2 for VF 2: profitable to scalarize store i64 %valB, ptr %out, align 8
 ; AVX1:  Cost of 4.5 for VF 4: profitable to scalarize store i64 %valB, ptr %out, align 8
 ; AVX1:  Cost of 9 for VF 8: profitable to scalarize store i64 %valB, ptr %out, align 8
@@ -36,6 +39,7 @@ define void @test() {
 ; AVX1:  Cost of 36 for VF 32: profitable to scalarize store i64 %valB, ptr %out, align 8
 ;
 ; AVX2-LABEL: 'test'
+; AVX2:  Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
 ; AVX2:  Cost of 2 for VF 2: profitable to scalarize store i64 %valB, ptr %out, align 8
 ; AVX2:  Cost of 4.5 for VF 4: profitable to scalarize store i64 %valB, ptr %out, align 8
 ; AVX2:  Cost of 9 for VF 8: profitable to scalarize store i64 %valB, ptr %out, align 8
@@ -43,12 +47,13 @@ define void @test() {
 ; AVX2:  Cost of 36 for VF 32: profitable to scalarize store i64 %valB, ptr %out, align 8
 ;
 ; AVX512-LABEL: 'test'
+; AVX512:  Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
 ; AVX512:  Cost of 5 for VF 2: REPLICATE store ir<%valB>, ir<%out>
 ; AVX512:  Cost of 11 for VF 4: REPLICATE store ir<%valB>, ir<%out>
-; AVX512:  Cost of 10 for VF 8: WIDEN store ir<%out>, ir<%valB>, ir<%canStore> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
-; AVX512:  Cost of 20 for VF 16: WIDEN store ir<%out>, ir<%valB>, ir<%canStore> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
-; AVX512:  Cost of 40 for VF 32: WIDEN store ir<%out>, ir<%valB>, ir<%canStore> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
-; AVX512:  Cost of 80 for VF 64: WIDEN store ir<%out>, ir<%valB>, ir<%canStore> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX512:  Cost of 10 for VF 8: WIDEN store ir<%out>, ir<%valB>, ir<%canStore>
+; AVX512:  Cost of 20 for VF 16: WIDEN store ir<%out>, ir<%valB>, ir<%canStore>
+; AVX512:  Cost of 40 for VF 32: WIDEN store ir<%out>, ir<%valB>, ir<%canStore>
+; AVX512:  Cost of 80 for VF 64: WIDEN store ir<%out>, ir<%valB>, ir<%canStore>
 ;
 entry:
   br label %for.body
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i16.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i16.ll
index 886e593362383e..a00dba4f1330e2 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i16.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i16.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*store i16 %valB, ptr %out" --filter "Cost of [1-9][0-9]*(.[0-9]+)? for VF [0-9]+: (profitable to scalarize\s+store i16 %valB|WIDEN store .*, ir<%valB>|REPLICATE store ir<%valB>)"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT store ir<%valB>" --filter "Cost of [1-9][0-9]*(.[0-9]+)? for VF [0-9]+: (profitable to scalarize\s+store i16 %valB|WIDEN store .*, ir<%valB>|REPLICATE store ir<%valB>)"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output -vplan-print-metadata=false < %s 2>&1 | FileCheck %s --check-prefixes=SSE
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse4.2 --debug-only=loop-vectorize --disable-output -vplan-print-metadata=false < %s 2>&1 | FileCheck %s --check-prefixes=SSE
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output -vplan-print-metadata=false < %s 2>&1 | FileCheck %s --check-prefixes=AVX1
@@ -16,12 +16,14 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test(ptr %C) {
 ; SSE-LABEL: 'test'
+; SSE:  Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
 ; SSE:  Cost of 2 for VF 2: profitable to scalarize store i16 %valB, ptr %out, align 2
 ; SSE:  Cost of 4 for VF 4: profitable to scalarize store i16 %valB, ptr %out, align 2
 ; SSE:  Cost of 8 for VF 8: profitable to scalarize store i16 %valB, ptr %out, align 2
 ; SSE:  Cost of 16 for VF 16: profitable to scalarize store i16 %valB, ptr %out, align 2
 ;
 ; AVX1-LABEL: 'test'
+; AVX1:  Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
 ; AVX1:  Cost of 2 for VF 2: profitable to scalarize store i16 %valB, ptr %out, align 2
 ; AVX1:  Cost of 4 for VF 4: profitable to scalarize store i16 %valB, ptr %out, align 2
 ; AVX1:  Cost of 8 for VF 8: profitable to scalarize store i16 %valB, ptr %out, align 2
@@ -29,6 +31,7 @@ define void @test(ptr %C) {
 ; AVX1:  Cost of 33 for VF 32: profitable to scalarize store i16 %valB, ptr %out, align 2
 ;
 ; AVX2-LABEL: 'test'
+; AVX2:  Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
 ; AVX2:  Cost of 2 for VF 2: profitable to scalarize store i16 %valB, ptr %out, align 2
 ; AVX2:  Cost of 4 for VF 4: profitable to scalarize store i16 %valB, ptr %out, align 2
 ; AVX2:  Cost of 8 for VF 8: profitable to scalarize store i16 %valB, ptr %out, align 2
@@ -36,6 +39,7 @@ define void @test(ptr %C) {
 ; AVX2:  Cost of 33 for VF 32: profitable to scalarize store i16 %valB, ptr %out, align 2
 ;
 ; AVX512-LABEL: 'test'
+; AVX512:  Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
 ; AVX512:  Cost of 2 for VF 2: WIDEN store vp<[[VP7:%[0-9]+]]>, ir<%valB>, ir<%canStore>
 ; AVX512:  Cost of 2 for VF 4: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
 ; AVX512:  Cost of 1 for VF 8: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i32.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i32.ll
index 9be2bd10f40fad..f65258131e69f6 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i32.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i32.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*store i32 %valB, ptr %out" --filter "Cost of [1-9][0-9]*(.[0-9]+)? for VF [0-9]+: (profitable to scalarize\s+store i32 %valB|WIDEN store .*, ir<%valB>|REPLICATE store ir<%valB>)"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT store ir<%valB>" --filter "Cost of [1-9][0-9]*(.[0-9]+)? for VF [0-9]+: (profitable to scalarize\s+store i32 %valB|WIDEN store .*, ir<%valB>|REPLICATE store ir<%valB>)"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output -vplan-print-metadata=false < %s 2>&1 | FileCheck %s --check-prefixes=SSE2
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse4.2 --debug-only=loop-vectorize --disable-output -vplan-print-metadata=false < %s 2>&1 | FileCheck %s --check-prefixes=SSE42
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output -vplan-print-metadata=false < %s 2>&1 | FileCheck %s --check-prefixes=AVX1
@@ -16,18 +16,21 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test(ptr %C) {
 ; SSE2-LABEL: 'test'
+; SSE2:  Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
 ; SSE2:  Cost of 2.5 for VF 2: profitable to scalarize store i32 %valB, ptr %out, align 4
 ; SSE2:  Cost of 5.5 for VF 4: profitable to scalarize store i32 %valB, ptr %out, align 4
 ; SSE2:  Cost of 11 for VF 8: profitable to scalarize store i32 %valB, ptr %out, align 4
 ; SSE2:  Cost of 22 for VF 16: profitable to scalarize store i32 %valB, ptr %out, align 4
 ;
 ; SSE42-LABEL: 'test'
+; SSE42:  Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
 ; SSE42:  Cost of 2 for VF 2: profitable to scalarize store i32 %valB, ptr %out, align 4
 ; SSE42:  Cost of 4 for VF 4: profitable to scalarize store i32 %valB, ptr %out, align 4
 ; SSE42:  Cost of 8 for VF 8: profitable to scalarize store i32 %valB, ptr %out, align 4
 ; SSE42:  Cost of 16 for VF 16: profitable to scalarize store i32 %valB, ptr %out, align 4
 ;
 ; AVX1-LABEL: 'test'
+; AVX1:  Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
 ; AVX1:  Cost of 9 for VF 2: WIDEN store vp<[[VP7:%[0-9]+]]>, ir<%valB>, ir<%canStore>
 ; AVX1:  Cost of 8 for VF 4: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
 ; AVX1:  Cost of 8 for VF 8: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
@@ -35,6 +38,7 @@ define void @test(ptr %C) {
 ; AVX1:  Cost of 32 for VF 32: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
 ;
 ; AVX2-LABEL: 'test'
+; AVX2:  Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
 ; AVX2:  Cost of 9 for VF 2: WIDEN store vp<[[VP7:%[0-9]+]]>, ir<%valB>, ir<%canStore>
 ; AVX2:  Cost of 8 for VF 4: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
 ; AVX2:  Cost of 8 for VF 8: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
@@ -42,6 +46,7 @@ define void @test(ptr %C) {
 ; AVX2:  Cost of 32 for VF 32: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
 ;
 ; AVX512-LABEL: 'test'
+; AVX512:  Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
 ; AVX512:  Cost of 2 for VF 2: WIDEN store vp<[[VP7:%[0-9]+]]>, ir<%valB>, ir<%canStore>
 ; AVX512:  Cost of 1 for VF 4: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
 ; AVX512:  Cost of 1 for VF 8: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i64.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i64.ll
index acc0158917568f..1c1692ea164a8f 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i64.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i64.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*store i64 %valB, ptr %out" --filter "Cost of [1-9][0-9]*(.[0-9]+)? for VF [0-9]+: (profitable to scalarize\s+store i64 %valB|WIDEN store .*, ir<%valB>|REPLICATE store ir<%valB>)"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT store ir<%valB>" --filter "Cost of [1-9][0-9]*(.[0-9]+)? for VF [0-9]+: (profitable to scalarize\s+store i64 %valB|WIDEN store .*, ir<%valB>|REPLICATE store ir<%valB>)"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output -vplan-print-metadata=false < %s 2>&1 | FileCheck %s --check-prefixes=SSE2
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse4.2 --debug-only=loop-vectorize --disable-output -vplan-print-metadata=false < %s 2>&1 | FileCheck %s --check-prefixes=SSE42
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output -vplan-print-metadata=false < %s 2>&1 | FileCheck %s --check-prefixes=AVX1
@@ -16,18 +16,21 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test(ptr %C) {
 ; SSE2-LABEL: 'test'
+; SSE2:  Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
 ; SSE2:  Cost of 2.5 for VF 2: profitable to scalarize store i64 %valB, ptr %out, align 8
 ; SSE2:  Cost of 5 for VF 4: profitable to scalarize store i64 %valB, ptr %out, align 8
 ; SSE2:  Cost of 10 for VF 8: profitable to scalarize store i64 %valB, ptr %out, align 8
 ; SSE2:  Cost of 20 for VF 16: profitable to scalarize store i64 %valB, ptr %out, align 8
 ;
 ; SSE42-LABEL: 'test'
+; SSE42:  Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
 ; SSE42:  Cost of 2 for VF 2: profitable to scalarize store i64 %valB, ptr %out, align 8
 ; SSE42:  Cost of 4 for VF 4: profitable to scalarize store i64 %valB, ptr %out, align 8
 ; SSE42:  Cost of 8 for VF 8: profitable to scalarize store i64 %valB, ptr %out, align 8
 ; SSE42:  Cost of 16 for VF 16: profitable to scalarize store i64 %valB, ptr %out, align 8
 ;
 ; AVX1-LABEL: 'test'
+; AVX1:  Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
 ; AVX1:  Cost of 8 for VF 2: WIDEN store vp<[[VP7:%[0-9]+]]>, ir<%valB>, ir<%canStore>
 ; AVX1:  Cost of 8 for VF 4: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
 ; AVX1:  Cost of 16 for VF 8: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
@@ -35,6 +38,7 @@ define void @test(ptr %C) {
 ; AVX1:  Cost of 64 for VF 32: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
 ;
 ; AVX2-LABEL: 'test'
+; AVX2:  Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
 ; AVX2:  Cost of 8 for VF 2: WIDEN store vp<[[VP7:%[0-9]+]]>, ir<%valB>, ir<%canStore>
 ; AVX2:  Cost of 8 for VF 4: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
 ; AVX2:  Cost of 16 for VF 8: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
@@ -42,6 +46,7 @@ define void @test(ptr %C) {
 ; AVX2:  Cost of 64 for VF 32: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
 ;
 ; AVX512-LABEL: 'test'
+; AVX512:  Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
 ; AVX512:  Cost of 1 for VF 2: WIDEN store vp<[[VP7:%[0-9]+]]>, ir<%valB>, ir<%canStore>
 ; AVX512:  Cost of 1 for VF 4: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
 ; AVX512:  Cost of 1 for VF 8: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i8.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i8.ll
index 721aa7ba285b18..a8ccf6c2e6639e 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i8.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i8.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*store i8 %valB, ptr %out" --filter "Cost of [1-9][0-9]*(.[0-9]+)? for VF [0-9]+: (profitable to scalarize\s+store i8 %valB|WIDEN store .*, ir<%valB>|REPLICATE store ir<%valB>)"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT store ir<%valB>" --filter "Cost of [1-9][0-9]*(.[0-9]+)? for VF [0-9]+: (profitable to scalarize\s+store i8 %valB|WIDEN store .*, ir<%valB>|REPLICATE store ir<%valB>)"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output -vplan-print-metadata=false < %s 2>&1 | FileCheck %s --check-prefixes=SSE2
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse4.2 --debug-only=loop-vectorize --disable-output -vplan-print-metadata=false < %s 2>&1 | FileCheck %s --check-prefixes=SSE42
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output -vplan-print-metadata=false < %s 2>&1 | FileCheck %s --check-prefixes=AVX1
@@ -16,18 +16,21 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test(ptr %C) {
 ; SSE2-LABEL: 'test'
+; SSE2:  Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
 ; SSE2:  Cost of 2.5 for VF 2: profitable to scalarize store i8 %valB, ptr %out, align 1
 ; SSE2:  Cost of 5.5 for VF 4: profitable to scalarize store i8 %valB, ptr %out, align 1
 ; SSE2:  Cost of 11.5 for VF 8: profitable to scalarize store i8 %valB, ptr %out, align 1
 ; SSE2:  Cost of 23.5 for VF 16: profitable to scalarize store i8 %valB, ptr %out, align 1
 ;
 ; SSE42-LABEL: 'test'
+; SSE42:  Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
 ; SSE42:  Cost of 2 for VF 2: profitable to scalarize store i8 %valB, ptr %out, align 1
 ; SSE42:  Cost of 4 for VF 4: profitable to scalarize store i8 %valB, ptr %out, align 1
 ; SSE42:  Cost of 8 for VF 8: profitable to scalarize store i8 %valB, ptr %out, align 1
 ; SSE42:  Cost of 16 for VF 16: profitable to scalarize store i8 %valB, ptr %out, align 1
 ;
 ; AVX1-LABEL: 'test'
+; AVX1:  Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
 ; AVX1:  Cost of 2 for VF 2: profitable to scalarize store i8 %valB, ptr %out, align 1
 ; AVX1:  Cost of 4 for VF 4: profitable to scalarize store i8 %valB, ptr %out, align 1
 ; AVX1:  Cost of 8 for VF 8: profitable to scalarize store i8 %valB, ptr %out, align 1
@@ -35,6 +38,7 @@ define void @test(ptr %C) {
 ; AVX1:  Cost of 32.5 for VF 32: profitable to scalarize store i8 %valB, ptr %out, align 1
 ;
 ; AVX2-LABEL: 'test'
+; AVX2:  Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
 ; AVX2:  Cost of 2 for VF 2: profitable to scalarize store i8 %valB, ptr %out, align 1
 ; AVX2:  Cost of 4 for VF 4: profitable to scalarize store i8 %valB, ptr %out, align 1
 ; AVX2:  Cost of 8 for VF 8: profitable to scalarize store i8 %valB, ptr %out, align 1
@@ -42,6 +46,7 @@ define void @test(ptr %C) {
 ; AVX2:  Cost of 32.5 for VF 32: profitable to scalarize store i8 %valB, ptr %out, align 1
 ;
 ; AVX512-LABEL: 'test'
+; AVX512:  Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
 ; AVX512:  Cost of 2 for VF 2: WIDEN store vp<[[VP7:%[0-9]+]]>, ir<%valB>, ir<%canStore>
 ; AVX512:  Cost of 2 for VF 4: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
 ; AVX512:  Cost of 2 for VF 8: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i16-with-i8-index.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i16-with-i8-index.ll
index 9daa35ed531659..b40147e8ac3553 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i16-with-i8-index.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i16-with-i8-index.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*store i16 %valB, ptr %out" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (WIDEN store|REPLICATE store ir<%valB>)"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT store ir<%valB>" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (WIDEN store|REPLICATE store ir<%valB>)"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=SSE2
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse4.2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=SSE42
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=AVX1
@@ -17,18 +17,21 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
+; SSE2:  Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
 ; SSE2:  Cost of 28 for VF 2: REPLICATE store ir<%valB>, ir<%out>
 ; SSE2:  Cost of 56 for VF 4: REPLICATE store ir<%valB>, ir<%out>
 ; SSE2:  Cost of 112 for VF 8: REPLICATE store ir<%valB>, ir<%out>
 ; SSE2:  Cost of 224 for VF 16: REPLICATE store ir<%valB>, ir<%out>
 ;
 ; SSE42-LABEL: 'test'
+; SSE42:  Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
 ; SSE42:  Cost of 26 for VF 2: REPLICATE store ir<%valB>, ir<%out>
 ; SSE42:  Cost of 52 for VF 4: REPLICATE store ir<%valB>, ir<%out>
 ; SSE42:  Cost of 104 for VF 8: REPLICATE store ir<%valB>, ir<%out>
 ; SSE42:  Cost of 208 for VF 16: REPLICATE store ir<%valB>, ir<%out>
 ;
 ; AVX1-LABEL: 'test'
+; AVX1:  Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
 ; AVX1:  Cost of 26 for VF 2: REPLICATE store ir<%valB>, ir<%out>
 ; AVX1:  Cost of 53 for VF 4: REPLICATE store ir<%valB>, ir<%out>
 ; AVX1:  Cost of 106 for VF 8: REPLICATE store ir<%valB>, ir<%out>
@@ -36,6 +39,7 @@ define void @test() {
 ; AVX1:  Cost of 426 for VF 32: REPLICATE store ir<%valB>, ir<%out>
 ;
 ; AVX2-LABEL: 'test'
+; AVX2:  Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
 ; AVX2:  Cost of 6 for VF 2: REPLICATE store ir<%valB>, ir<%out>
 ; AVX2:  Cost of 13 for VF 4: REPLICATE store ir<%valB>, ir<%out>
 ; AVX2:  Cost of 26 for VF 8: REPLICATE store ir<%valB>, ir<%out>
@@ -43,6 +47,7 @@ define void @test() {
 ; AVX2:  Cost of 106 for VF 32: REPLICATE store ir<%valB>, ir<%out>
 ;
 ; AVX512-LABEL: 'test'
+; AVX512:  Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
 ; AVX512:  Cost of 6 for VF 2: REPLICATE store ir<%valB>, ir<%out>
 ; AVX512:  Cost of 13 for VF 4: REPLICATE store ir<%valB>, ir<%out>
 ; AVX512:  Cost of 27 for VF 8: REPLICATE store ir<%valB>, ir<%out>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i32-with-i8-index.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i32-with-i8-index.ll
index 671842c2645cf2..a8cf9c1ffa5942 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i32-with-i8-index.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i32-with-i8-index.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*store i32 %valB, ptr %out" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (WIDEN store|REPLICATE store ir<%valB>)"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT store ir<%valB>" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (WIDEN store|REPLICATE store ir<%valB>)"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=SSE2
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse4.2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=SSE42
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=AVX1
@@ -17,18 +17,21 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
+; SSE2:  Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
 ; SSE2:  Cost of 29 for VF 2: REPLICATE store ir<%valB>, ir<%out>
 ; SSE2:  Cost of 59 for VF 4: REPLICATE store ir<%valB>, ir<%out>
 ; SSE2:  Cost of 118 for VF 8: REPLICATE store ir<%valB>, ir<%out>
 ; SSE2:  Cost of 236 for VF 16: REPLICATE store ir<%valB>, ir<%out>
 ;
 ; SSE42-LABEL: 'test'
+; SSE42:  Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
 ; SSE42:  Cost of 26 for VF 2: REPLICATE store ir<%valB>, ir<%out>
 ; SSE42:  Cost of 52 for VF 4: REPLICATE store ir<%valB>, ir<%out>
 ; SSE42:  Cost of 104 for VF 8: REPLICATE store ir<%valB>, ir<%out>
 ; SSE42:  Cost of 208 for VF 16: REPLICATE store ir<%valB>, ir<%out>
 ;
 ; AVX1-LABEL: 'test'
+; AVX1:  Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
 ; AVX1:  Cost of 26 for VF 2: REPLICATE store ir<%valB>, ir<%out>
 ; AVX1:  Cost of 53 for VF 4: REPLICATE store ir<%valB>, ir<%out>
 ; AVX1:  Cost of 107 for VF 8: REPLICATE store ir<%valB>, ir<%out>
@@ -36,6 +39,7 @@ define void @test() {
 ; AVX1:  Cost of 428 for VF 32: REPLICATE store ir<%valB>, ir<%out>
 ;
 ; AVX2-LABEL: 'test'
+; AVX2:  Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
 ; AVX2:  Cost of 6 for VF 2: REPLICATE store ir<%valB>, ir<%out>
 ; AVX2:  Cost of 13 for VF 4: REPLICATE store ir<%valB>, ir<%out>
 ; AVX2:  Cost of 27 for VF 8: REPLICATE store ir<%valB>, ir<%out>
@@ -43,6 +47,7 @@ define void @test() {
 ; AVX2:  Cost of 108 for VF 32: REPLICATE store ir<%valB>, ir<%out>
 ;
 ; AVX512-LABEL: 'test'
+; AVX512:  Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
 ; AVX512:  Cost of 6 for VF 2: REPLICATE store ir<%valB>, ir<%out>
 ; AVX512:  Cost of 13 for VF 4: REPLICATE store ir<%valB>, ir<%out>
 ; AVX512:  Cost of 10 for VF 8: WIDEN store ir<%out>, ir<%valB>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i64-with-i8-index.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i64-with-i8-index.ll
index dc775779060122..960bd5e8ae2026 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i64-with-i8-index.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i64-with-i8-index.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*store i64 %valB, ptr %out" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (WIDEN store|REPLICATE store ir<%valB>)"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT store ir<%valB>" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (WIDEN store|REPLICATE store ir<%valB>)"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=SSE2
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse4.2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=SSE42
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=AVX1
@@ -17,18 +17,21 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
+; SSE2:  Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
 ; SSE2:  Cost of 29 for VF 2: REPLICATE store ir<%valB>, ir<%out>
 ; SSE2:  Cost of 58 for VF 4: REPLICATE store ir<%valB>, ir<%out>
 ; SSE2:  Cost of 116 for VF 8: REPLICATE store ir<%valB>, ir<%out>
 ; SSE2:  Cost of 232 for VF 16: REPLICATE store ir<%valB>, ir<%out>
 ;
 ; SSE42-LABEL: 'test'
+; SSE42:  Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
 ; SSE42:  Cost of 26 for VF 2: REPLICATE store ir<%valB>, ir<%out>
 ; SSE42:  Cost of 52 for VF 4: REPLICATE store ir<%valB>, ir<%out>
 ; SSE42:  Cost of 104 for VF 8: REPLICATE store ir<%valB>, ir<%out>
 ; SSE42:  Cost of 208 for VF 16: REPLICATE store ir<%valB>, ir<%out>
 ;
 ; AVX1-LABEL: 'test'
+; AVX1:  Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
 ; AVX1:  Cost of 26 for VF 2: REPLICATE store ir<%valB>, ir<%out>
 ; AVX1:  Cost of 54 for VF 4: REPLICATE store ir<%valB>, ir<%out>
 ; AVX1:  Cost of 108 for VF 8: REPLICATE store ir<%valB>, ir<%out>
@@ -36,6 +39,7 @@ define void @test() {
 ; AVX1:  Cost of 432 for VF 32: REPLICATE store ir<%valB>, ir<%out>
 ;
 ; AVX2-LABEL: 'test'
+; AVX2:  Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
 ; AVX2:  Cost of 6 for VF 2: REPLICATE store ir<%valB>, ir<%out>
 ; AVX2:  Cost of 14 for VF 4: REPLICATE store ir<%valB>, ir<%out>
 ; AVX2:  Cost of 28 for VF 8: REPLICATE store ir<%valB>, ir<%out>
@@ -43,6 +47,7 @@ define void @test() {
 ; AVX2:  Cost of 112 for VF 32: REPLICATE store ir<%valB>, ir<%out>
 ;
 ; AVX512-LABEL: 'test'
+; AVX512:  Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
 ; AVX512:  Cost of 6 for VF 2: REPLICATE store ir<%valB>, ir<%out>
 ; AVX512:  Cost of 14 for VF 4: REPLICATE store ir<%valB>, ir<%out>
 ; AVX512:  Cost of 10 for VF 8: WIDEN store ir<%out>, ir<%valB>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i8-with-i8-index.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i8-with-i8-index.ll
index c484b526ac1cb3..840c699af0be5a 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i8-with-i8-index.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i8-with-i8-index.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*store i8 %valB, ptr %out" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (WIDEN store|REPLICATE store ir<%valB>)"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT store ir<%valB>" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (WIDEN store|REPLICATE store ir<%valB>)"
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=SSE2
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse4.2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=SSE42
 ; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx  --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=AVX1
@@ -17,18 +17,21 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; SSE2-LABEL: 'test'
+; SSE2:  Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
 ; SSE2:  Cost of 29 for VF 2: REPLICATE store ir<%valB>, ir<%out>
 ; SSE2:  Cost of 59 for VF 4: REPLICATE store ir<%valB>, ir<%out>
 ; SSE2:  Cost of 119 for VF 8: REPLICATE store ir<%valB>, ir<%out>
 ; SSE2:  Cost of 239 for VF 16: REPLICATE store ir<%valB>, ir<%out>
 ;
 ; SSE42-LABEL: 'test'
+; SSE42:  Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
 ; SSE42:  Cost of 26 for VF 2: REPLICATE store ir<%valB>, ir<%out>
 ; SSE42:  Cost of 52 for VF 4: REPLICATE store ir<%valB>, ir<%out>
 ; SSE42:  Cost of 104 for VF 8: REPLICATE store ir<%valB>, ir<%out>
 ; SSE42:  Cost of 208 for VF 16: REPLICATE store ir<%valB>, ir<%out>
 ;
 ; AVX1-LABEL: 'test'
+; AVX1:  Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
 ; AVX1:  Cost of 26 for VF 2: REPLICATE store ir<%valB>, ir<%out>
 ; AVX1:  Cost of 53 for VF 4: REPLICATE store ir<%valB>, ir<%out>
 ; AVX1:  Cost of 106 for VF 8: REPLICATE store ir<%valB>, ir<%out>
@@ -36,6 +39,7 @@ define void @test() {
 ; AVX1:  Cost of 425 for VF 32: REPLICATE store ir<%valB>, ir<%out>
 ;
 ; AVX2-LABEL: 'test'
+; AVX2:  Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
 ; AVX2:  Cost of 6 for VF 2: REPLICATE store ir<%valB>, ir<%out>
 ; AVX2:  Cost of 13 for VF 4: REPLICATE store ir<%valB>, ir<%out>
 ; AVX2:  Cost of 26 for VF 8: REPLICATE store ir<%valB>, ir<%out>
@@ -43,6 +47,7 @@ define void @test() {
 ; AVX2:  Cost of 105 for VF 32: REPLICATE store ir<%valB>, ir<%out>
 ;
 ; AVX512-LABEL: 'test'
+; AVX512:  Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
 ; AVX512:  Cost of 6 for VF 2: REPLICATE store ir<%valB>, ir<%out>
 ; AVX512:  Cost of 13 for VF 4: REPLICATE store ir<%valB>, ir<%out>
 ; AVX512:  Cost of 27 for VF 8: REPLICATE store ir<%valB>, ir<%out>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/strided-load-i16.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/strided-load-i16.ll
index 3c27644f4ebcf0..c0c810e4bfaf91 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/strided-load-i16.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/strided-load-i16.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Found an estimated cost of [0-9]+(.[0-9]+)? for VF [0-9]+ For instruction:\s*%1" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: INTERLEAVE-GROUP" --version 6
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%1> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: INTERLEAVE-GROUP" --version 6
 ; REQUIRES: asserts
 ; RUN: opt -passes=loop-vectorize -S -mattr=avx512bw --debug-only=loop-vectorize --disable-output < %s 2>&1| FileCheck %s
 
@@ -10,6 +10,7 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @load_i16_stride2() {
 ; CHECK-LABEL: 'load_i16_stride2'
+; CHECK:  Cost of 1 for VF 1: EMIT-SCALAR ir<%1> = load ir<%arrayidx>
 ; CHECK:  Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%arrayidx>
 ; CHECK:  Cost of 1 for VF 4: INTERLEAVE-GROUP with factor 2, ir<%arrayidx>
 ; CHECK:  Cost of 2 for VF 8: INTERLEAVE-GROUP with factor 2, ir<%arrayidx>
@@ -36,6 +37,7 @@ for.end:
 
 define void @load_i16_stride3() {
 ; CHECK-LABEL: 'load_i16_stride3'
+; CHECK:  Cost of 1 for VF 1: EMIT-SCALAR ir<%1> = load ir<%arrayidx>
 ; CHECK:  Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%arrayidx>
 ; CHECK:  Cost of 2 for VF 4: INTERLEAVE-GROUP with factor 3, ir<%arrayidx>
 ; CHECK:  Cost of 2 for VF 8: INTERLEAVE-GROUP with factor 3, ir<%arrayidx>
@@ -62,6 +64,7 @@ for.end:
 
 define void @load_i16_stride4() {
 ; CHECK-LABEL: 'load_i16_stride4'
+; CHECK:  Cost of 1 for VF 1: EMIT-SCALAR ir<%1> = load ir<%arrayidx>
 ; CHECK:  Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%arrayidx>
 ; CHECK:  Cost of 2 for VF 4: INTERLEAVE-GROUP with factor 4, ir<%arrayidx>
 ; CHECK:  Cost of 2 for VF 8: INTERLEAVE-GROUP with factor 4, ir<%arrayidx>
@@ -88,6 +91,7 @@ for.end:
 
 define void @load_i16_stride5() {
 ; CHECK-LABEL: 'load_i16_stride5'
+; CHECK:  Cost of 1 for VF 1: EMIT-SCALAR ir<%1> = load ir<%arrayidx>
 ; CHECK:  Cost of 2 for VF 2: INTERLEAVE-GROUP with factor 5, ir<%arrayidx>
 ; CHECK:  Cost of 2 for VF 4: INTERLEAVE-GROUP with factor 5, ir<%arrayidx>
 ; CHECK:  Cost of 3 for VF 8: INTERLEAVE-GROUP with factor 5, ir<%arrayidx>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/strided-load-i32.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/strided-load-i32.ll
index 8bc141496b822b..afa24bb18fdc54 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/strided-load-i32.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/strided-load-i32.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Found an estimated cost of [0-9]+(.[0-9]+)? for VF [0-9]+ For instruction:\s*%1" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: INTERLEAVE-GROUP" --version 6
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%1> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: INTERLEAVE-GROUP" --version 6
 ; REQUIRES: asserts
 ; RUN: opt -passes=loop-vectorize -S -mattr=avx512f --debug-only=loop-vectorize --disable-output < %s 2>&1| FileCheck %s
 
@@ -10,6 +10,7 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @load_int_stride2() {
 ; CHECK-LABEL: 'load_int_stride2'
+; CHECK:  Cost of 1 for VF 1: EMIT-SCALAR ir<%1> = load ir<%arrayidx>
 ; CHECK:  Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%arrayidx>
 ; CHECK:  Cost of 1 for VF 4: INTERLEAVE-GROUP with factor 2, ir<%arrayidx>
 ; CHECK:  Cost of 1 for VF 8: INTERLEAVE-GROUP with factor 2, ir<%arrayidx>
@@ -35,6 +36,7 @@ for.end:
 
 define void @load_int_stride3() {
 ; CHECK-LABEL: 'load_int_stride3'
+; CHECK:  Cost of 1 for VF 1: EMIT-SCALAR ir<%1> = load ir<%arrayidx>
 ; CHECK:  Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%arrayidx>
 ; CHECK:  Cost of 1 for VF 4: INTERLEAVE-GROUP with factor 3, ir<%arrayidx>
 ; CHECK:  Cost of 3 for VF 8: INTERLEAVE-GROUP with factor 3, ir<%arrayidx>
@@ -60,6 +62,7 @@ for.end:
 
 define void @load_int_stride4() {
 ; CHECK-LABEL: 'load_int_stride4'
+; CHECK:  Cost of 1 for VF 1: EMIT-SCALAR ir<%1> = load ir<%arrayidx>
 ; CHECK:  Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%arrayidx>
 ; CHECK:  Cost of 1 for VF 4: INTERLEAVE-GROUP with factor 4, ir<%arrayidx>
 ; CHECK:  Cost of 3 for VF 8: INTERLEAVE-GROUP with factor 4, ir<%arrayidx>
@@ -85,6 +88,7 @@ for.end:
 
 define void @load_int_stride5() {
 ; CHECK-LABEL: 'load_int_stride5'
+; CHECK:  Cost of 1 for VF 1: EMIT-SCALAR ir<%1> = load ir<%arrayidx>
 ; CHECK:  Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 5, ir<%arrayidx>
 ; CHECK:  Cost of 3 for VF 4: INTERLEAVE-GROUP with factor 5, ir<%arrayidx>
 ; CHECK:  Cost of 5 for VF 8: INTERLEAVE-GROUP with factor 5, ir<%arrayidx>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/strided-load-i64.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/strided-load-i64.ll
index 1c91d01340a4b9..ca7865ce4f307c 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/strided-load-i64.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/strided-load-i64.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Found an estimated cost of [0-9]+(.[0-9]+)? for VF [0-9]+ For instruction:\s*%1" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: INTERLEAVE-GROUP" --version 6
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%1> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: INTERLEAVE-GROUP" --version 6
 ; REQUIRES: asserts
 ; RUN: opt -passes=loop-vectorize -S -mattr=avx512f --debug-only=loop-vectorize --disable-output < %s 2>&1| FileCheck %s
 
@@ -10,6 +10,7 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @load_i64_stride2() {
 ; CHECK-LABEL: 'load_i64_stride2'
+; CHECK:  Cost of 1 for VF 1: EMIT-SCALAR ir<%1> = load ir<%arrayidx>
 ; CHECK:  Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%arrayidx>
 ; CHECK:  Cost of 1 for VF 4: INTERLEAVE-GROUP with factor 2, ir<%arrayidx>
 ; CHECK:  Cost of 3 for VF 8: INTERLEAVE-GROUP with factor 2, ir<%arrayidx>
@@ -34,6 +35,7 @@ for.end:
 
 define void @load_i64_stride3() {
 ; CHECK-LABEL: 'load_i64_stride3'
+; CHECK:  Cost of 1 for VF 1: EMIT-SCALAR ir<%1> = load ir<%arrayidx>
 ; CHECK:  Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%arrayidx>
 ; CHECK:  Cost of 3 for VF 4: INTERLEAVE-GROUP with factor 3, ir<%arrayidx>
 ; CHECK:  Cost of 5 for VF 8: INTERLEAVE-GROUP with factor 3, ir<%arrayidx>
@@ -58,6 +60,7 @@ for.end:
 
 define void @load_i64_stride4() {
 ; CHECK-LABEL: 'load_i64_stride4'
+; CHECK:  Cost of 1 for VF 1: EMIT-SCALAR ir<%1> = load ir<%arrayidx>
 ; CHECK:  Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%arrayidx>
 ; CHECK:  Cost of 3 for VF 4: INTERLEAVE-GROUP with factor 4, ir<%arrayidx>
 ; CHECK:  Cost of 8 for VF 8: INTERLEAVE-GROUP with factor 4, ir<%arrayidx>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/strided-load-i8.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/strided-load-i8.ll
index b87607872fdccf..2548fbc4adeb64 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/strided-load-i8.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/strided-load-i8.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Found an estimated cost of [0-9]+(.[0-9]+)? for VF [0-9]+ For instruction:\s*%1" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: INTERLEAVE-GROUP" --version 6
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%1> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: INTERLEAVE-GROUP" --version 6
 ; REQUIRES: asserts
 ; RUN: opt -passes=loop-vectorize -S -mattr=avx512bw --debug-only=loop-vectorize --disable-output < %s 2>&1| FileCheck %s
 
@@ -10,6 +10,7 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @load_i8_stride2() {
 ; CHECK-LABEL: 'load_i8_stride2'
+; CHECK:  Cost of 1 for VF 1: EMIT-SCALAR ir<%1> = load ir<%arrayidx>
 ; CHECK:  Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%arrayidx>
 ; CHECK:  Cost of 1 for VF 4: INTERLEAVE-GROUP with factor 2, ir<%arrayidx>
 ; CHECK:  Cost of 1 for VF 8: INTERLEAVE-GROUP with factor 2, ir<%arrayidx>
@@ -37,6 +38,7 @@ for.end:
 
 define void @load_i8_stride3() {
 ; CHECK-LABEL: 'load_i8_stride3'
+; CHECK:  Cost of 1 for VF 1: EMIT-SCALAR ir<%1> = load ir<%arrayidx>
 ; CHECK:  Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%arrayidx>
 ; CHECK:  Cost of 1 for VF 4: INTERLEAVE-GROUP with factor 3, ir<%arrayidx>
 ; CHECK:  Cost of 4 for VF 8: INTERLEAVE-GROUP with factor 3, ir<%arrayidx>
@@ -64,6 +66,7 @@ for.end:
 
 define void @load_i8_stride4() {
 ; CHECK-LABEL: 'load_i8_stride4'
+; CHECK:  Cost of 1 for VF 1: EMIT-SCALAR ir<%1> = load ir<%arrayidx>
 ; CHECK:  Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%arrayidx>
 ; CHECK:  Cost of 1 for VF 4: INTERLEAVE-GROUP with factor 4, ir<%arrayidx>
 ; CHECK:  Cost of 4 for VF 8: INTERLEAVE-GROUP with factor 4, ir<%arrayidx>
@@ -91,6 +94,7 @@ for.end:
 
 define void @load_i8_stride5() {
 ; CHECK-LABEL: 'load_i8_stride5'
+; CHECK:  Cost of 1 for VF 1: EMIT-SCALAR ir<%1> = load ir<%arrayidx>
 ; CHECK:  Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 5, ir<%arrayidx>
 ; CHECK:  Cost of 4 for VF 4: INTERLEAVE-GROUP with factor 5, ir<%arrayidx>
 ; CHECK:  Cost of 8 for VF 8: INTERLEAVE-GROUP with factor 5, ir<%arrayidx>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/vpinstruction-cost.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/vpinstruction-cost.ll
index 3e9ae12df16cba..6247b3910a609a 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/vpinstruction-cost.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/vpinstruction-cost.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+ for VF 1 For instruction" --filter "Cost of"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of"
 ; RUN: opt -S -passes=loop-vectorize -mcpu=skylake-avx512 -mtriple=x86_64-apple-macosx -debug -disable-output -S %s 2>&1 | FileCheck %s
 
 ; REQUIRES: asserts
@@ -12,10 +12,10 @@ define void @wide_or_replaced_with_add_vpinstruction(ptr %src, ptr noalias %dst)
 ; CHECK:  Cost of 1 for VF 1: EMIT-SCALAR ir<%l> = load ir<%g.src>
 ; CHECK:  Cost of 1 for VF 1: EMIT ir<%iv.4> = add nuw nsw ir<%iv>, ir<4>
 ; CHECK:  Cost of 1 for VF 1: EMIT ir<%c> = icmp ule ir<%l>, ir<128>
-; CHECK:  Cost of 0 for VF 1: EMIT branch-on-cond ir<%c> (!vplan.prof.estimated estimated {1073741824, 1073741824})
-; CHECK:  Cost of 1 for VF 1: EMIT ir<%or> = or disjoint ir<%iv.4>, ir<1> (!vplan.execution.frequency 4611686018427387904 (50%, estimated))
-; CHECK:  Cost of 0 for VF 1: EMIT ir<%g.dst> = getelementptr inbounds ir<%dst>, ir<%or> (!vplan.execution.frequency 4611686018427387904 (50%, estimated))
-; CHECK:  Cost of 1 for VF 1: EMIT store ir<%iv.4>, ir<%g.dst> (!vplan.execution.frequency 4611686018427387904 (50%, estimated))
+; CHECK:  Cost of 0 for VF 1: EMIT branch-on-cond ir<%c>
+; CHECK:  Cost of 1 for VF 1: EMIT ir<%or> = or disjoint ir<%iv.4>, ir<1>
+; CHECK:  Cost of 0 for VF 1: EMIT ir<%g.dst> = getelementptr inbounds ir<%dst>, ir<%or>
+; CHECK:  Cost of 1 for VF 1: EMIT store ir<%iv.4>, ir<%g.dst>
 ; CHECK:  Cost of 1 for VF 1: EMIT ir<%iv.next> = add nuw nsw ir<%iv>, ir<1>
 ; CHECK:  Cost of 1 for VF 1: EMIT ir<%exitcond> = icmp eq ir<%iv.next>, ir<32>
 ; CHECK:  Cost of 0 for VF 1: EMIT branch-on-cond ir<%exitcond>
@@ -29,7 +29,7 @@ define void @wide_or_replaced_with_add_vpinstruction(ptr %src, ptr noalias %dst)
 ; CHECK:  Cost of 1 for VF 2: EMIT ir<%or> = add ir<%iv.4>, ir<1>
 ; CHECK:  Cost of 0 for VF 2: CLONE ir<%g.dst> = getelementptr ir<%dst>, ir<%or>
 ; CHECK:  Cost of 0 for VF 2: vp<[[VP6:%[0-9]+]]> = vector-pointer i64, ir<%g.dst>, ir<1>
-; CHECK:  Cost of 1 for VF 2: WIDEN store vp<[[VP6]]>, ir<%iv.4>, ir<%c> (!vplan.execution.frequency 4611686018427387904 (50%, estimated))
+; CHECK:  Cost of 1 for VF 2: WIDEN store vp<[[VP6]]>, ir<%iv.4>, ir<%c>
 ; CHECK:  Cost of 0 for VF 2: EMIT vp<%index.next> = add nuw vp<[[VP3]]>, vp<[[VP1:%[0-9]+]]>
 ; CHECK:  Cost of 1 for VF 2: EMIT branch-on-count vp<%index.next>, vp<[[VP2:%[0-9]+]]>
 ; CHECK:  Cost of 0 for VF 2: vector loop backedge
@@ -52,7 +52,7 @@ define void @wide_or_replaced_with_add_vpinstruction(ptr %src, ptr noalias %dst)
 ; CHECK:  Cost of 1 for VF 4: EMIT ir<%or> = add ir<%iv.4>, ir<1>
 ; CHECK:  Cost of 0 for VF 4: CLONE ir<%g.dst> = getelementptr ir<%dst>, ir<%or>
 ; CHECK:  Cost of 0 for VF 4: vp<[[VP6]]> = vector-pointer i64, ir<%g.dst>, ir<1>
-; CHECK:  Cost of 1 for VF 4: WIDEN store vp<[[VP6]]>, ir<%iv.4>, ir<%c> (!vplan.execution.frequency 4611686018427387904 (50%, estimated))
+; CHECK:  Cost of 1 for VF 4: WIDEN store vp<[[VP6]]>, ir<%iv.4>, ir<%c>
 ; CHECK:  Cost of 0 for VF 4: EMIT vp<%index.next> = add nuw vp<[[VP3]]>, vp<[[VP1]]>
 ; CHECK:  Cost of 1 for VF 4: EMIT branch-on-count vp<%index.next>, vp<[[VP2]]>
 ; CHECK:  Cost of 0 for VF 4: vector loop backedge
@@ -176,11 +176,11 @@ define void @test_vpinstruction_switch_cost(ptr %start, ptr %end) {
 ; CHECK-LABEL: 'test_vpinstruction_switch_cost'
 ; CHECK:  Cost of 0 for VF 1: EMIT-SCALAR ir<%ptr.iv> = phi [ ir<%start>, vector.ph ], [ ir<%ptr.iv.next>, loop.latch ]
 ; CHECK:  Cost of 1 for VF 1: EMIT-SCALAR ir<%l> = load ir<%ptr.iv>
-; CHECK:  Cost of 0 for VF 1: EMIT switch ir<%l>, ir<-12>, ir<13>, ir<0> (!vplan.prof.estimated estimated {536870912, 536870912, 536870912, 536870912})
-; CHECK:  Cost of 1 for VF 1: EMIT store ir<1>, ir<%ptr.iv> (!vplan.execution.frequency 2305843009213693952 (25%, estimated))
-; CHECK:  Cost of 1 for VF 1: EMIT store ir<0>, ir<%ptr.iv> (!vplan.execution.frequency 2305843009213693952 (25%, estimated))
-; CHECK:  Cost of 1 for VF 1: EMIT store ir<42>, ir<%ptr.iv> (!vplan.execution.frequency 2305843009213693952 (25%, estimated))
-; CHECK:  Cost of 1 for VF 1: EMIT store ir<2>, ir<%ptr.iv> (!vplan.execution.frequency 2305843009213693952 (25%, estimated))
+; CHECK:  Cost of 0 for VF 1: EMIT switch ir<%l>, ir<-12>, ir<13>, ir<0>
+; CHECK:  Cost of 1 for VF 1: EMIT store ir<1>, ir<%ptr.iv>
+; CHECK:  Cost of 1 for VF 1: EMIT store ir<0>, ir<%ptr.iv>
+; CHECK:  Cost of 1 for VF 1: EMIT store ir<42>, ir<%ptr.iv>
+; CHECK:  Cost of 1 for VF 1: EMIT store ir<2>, ir<%ptr.iv>
 ; CHECK:  Cost of 0 for VF 1: EMIT ir<%ptr.iv.next> = getelementptr inbounds ir<%ptr.iv>, ir<1>
 ; CHECK:  Cost of 1 for VF 1: EMIT ir<%ec> = icmp eq ir<%ptr.iv.next>, ir<%end>
 ; CHECK:  Cost of 0 for VF 1: EMIT branch-on-cond ir<%ec>
@@ -196,13 +196,13 @@ define void @test_vpinstruction_switch_cost(ptr %start, ptr %end) {
 ; CHECK:  Cost of 0 for VF 2: EMIT vp<[[VP13:%[0-9]+]]> = or vp<[[VP12]]>, vp<[[VP11]]>
 ; CHECK:  Cost of 1 for VF 2: EMIT vp<[[VP14:%[0-9]+]]> = not vp<[[VP13]]>
 ; CHECK:  Cost of 0 for VF 2: vp<[[VP15:%[0-9]+]]> = vector-pointer i64, vp<%next.gep>, ir<1>
-; CHECK:  Cost of 1 for VF 2: WIDEN store vp<[[VP15]]>, ir<1>, vp<[[VP11]]> (!vplan.execution.frequency 2305843009213693952 (25%, estimated))
+; CHECK:  Cost of 1 for VF 2: WIDEN store vp<[[VP15]]>, ir<1>, vp<[[VP11]]>
 ; CHECK:  Cost of 0 for VF 2: vp<[[VP16:%[0-9]+]]> = vector-pointer i64, vp<%next.gep>, ir<1>
-; CHECK:  Cost of 1 for VF 2: WIDEN store vp<[[VP16]]>, ir<0>, vp<[[VP10]]> (!vplan.execution.frequency 2305843009213693952 (25%, estimated))
+; CHECK:  Cost of 1 for VF 2: WIDEN store vp<[[VP16]]>, ir<0>, vp<[[VP10]]>
 ; CHECK:  Cost of 0 for VF 2: vp<[[VP17:%[0-9]+]]> = vector-pointer i64, vp<%next.gep>, ir<1>
-; CHECK:  Cost of 1 for VF 2: WIDEN store vp<[[VP17]]>, ir<42>, vp<[[VP9]]> (!vplan.execution.frequency 2305843009213693952 (25%, estimated))
+; CHECK:  Cost of 1 for VF 2: WIDEN store vp<[[VP17]]>, ir<42>, vp<[[VP9]]>
 ; CHECK:  Cost of 0 for VF 2: vp<[[VP18:%[0-9]+]]> = vector-pointer i64, vp<%next.gep>, ir<1>
-; CHECK:  Cost of 1 for VF 2: WIDEN store vp<[[VP18]]>, ir<2>, vp<[[VP14]]> (!vplan.execution.frequency 2305843009213693952 (25%, estimated))
+; CHECK:  Cost of 1 for VF 2: WIDEN store vp<[[VP18]]>, ir<2>, vp<[[VP14]]>
 ; CHECK:  Cost of 0 for VF 2: EMIT vp<%index.next> = add nuw vp<[[VP5]]>, vp<[[VP1:%[0-9]+]]>
 ; CHECK:  Cost of 1 for VF 2: EMIT branch-on-count vp<%index.next>, vp<[[VP2:%[0-9]+]]>
 ; CHECK:  Cost of 0 for VF 2: vector loop backedge
@@ -226,13 +226,13 @@ define void @test_vpinstruction_switch_cost(ptr %start, ptr %end) {
 ; CHECK:  Cost of 0 for VF 4: EMIT vp<[[VP13]]> = or vp<[[VP12]]>, vp<[[VP11]]>
 ; CHECK:  Cost of 1 for VF 4: EMIT vp<[[VP14]]> = not vp<[[VP13]]>
 ; CHECK:  Cost of 0 for VF 4: vp<[[VP15]]> = vector-pointer i64, vp<%next.gep>, ir<1>
-; CHECK:  Cost of 1 for VF 4: WIDEN store vp<[[VP15]]>, ir<1>, vp<[[VP11]]> (!vplan.execution.frequency 2305843009213693952 (25%, estimated))
+; CHECK:  Cost of 1 for VF 4: WIDEN store vp<[[VP15]]>, ir<1>, vp<[[VP11]]>
 ; CHECK:  Cost of 0 for VF 4: vp<[[VP16]]> = vector-pointer i64, vp<%next.gep>, ir<1>
-; CHECK:  Cost of 1 for VF 4: WIDEN store vp<[[VP16]]>, ir<0>, vp<[[VP10]]> (!vplan.execution.frequency 2305843009213693952 (25%, estimated))
+; CHECK:  Cost of 1 for VF 4: WIDEN store vp<[[VP16]]>, ir<0>, vp<[[VP10]]>
 ; CHECK:  Cost of 0 for VF 4: vp<[[VP17]]> = vector-pointer i64, vp<%next.gep>, ir<1>
-; CHECK:  Cost of 1 for VF 4: WIDEN store vp<[[VP17]]>, ir<42>, vp<[[VP9]]> (!vplan.execution.frequency 2305843009213693952 (25%, estimated))
+; CHECK:  Cost of 1 for VF 4: WIDEN store vp<[[VP17]]>, ir<42>, vp<[[VP9]]>
 ; CHECK:  Cost of 0 for VF 4: vp<[[VP18]]> = vector-pointer i64, vp<%next.gep>, ir<1>
-; CHECK:  Cost of 1 for VF 4: WIDEN store vp<[[VP18]]>, ir<2>, vp<[[VP14]]> (!vplan.execution.frequency 2305843009213693952 (25%, estimated))
+; CHECK:  Cost of 1 for VF 4: WIDEN store vp<[[VP18]]>, ir<2>, vp<[[VP14]]>
 ; CHECK:  Cost of 0 for VF 4: EMIT vp<%index.next> = add nuw vp<[[VP5]]>, vp<[[VP1]]>
 ; CHECK:  Cost of 1 for VF 4: EMIT branch-on-count vp<%index.next>, vp<[[VP2]]>
 ; CHECK:  Cost of 0 for VF 4: vector loop backedge
diff --git a/llvm/test/tools/UpdateTestChecks/update_analyze_test_checks/Inputs/x86-loopvectorize-costmodel.ll.expected b/llvm/test/tools/UpdateTestChecks/update_analyze_test_checks/Inputs/x86-loopvectorize-costmodel.ll.expected
index 88911d7440d382..d2951e89b50e40 100644
--- a/llvm/test/tools/UpdateTestChecks/update_analyze_test_checks/Inputs/x86-loopvectorize-costmodel.ll.expected
+++ b/llvm/test/tools/UpdateTestChecks/update_analyze_test_checks/Inputs/x86-loopvectorize-costmodel.ll.expected
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+ for VF [0-9]+ For instruction:\s*%v0 = load float, ptr %in0, align 4" --filter "Cost of [0-9]+ for VF [0-9]+: INTERLEAVE-GROUP with factor 2, ir<%in0>" --filter "LV: Found an estimated cost of [0-9]+ for VF [0-9]+ For instruction:\s*%v0 = load float, float\* %in0, align 4"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+ for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+ for VF [0-9]+: INTERLEAVE-GROUP with factor 2, ir<%in0>" --filter "LV: Found an estimated cost of [0-9]+ for VF [0-9]+ For instruction:\s*%v0 = load float, float\* %in0, align 4"
 ; RUN: opt  -passes=loop-vectorize  -vectorizer-maximize-bandwidth -S -mattr=+avx512bw --debug-only=loop-vectorize < %s 2>&1 | FileCheck %s --check-prefixes=CHECK,AVX512
 ; REQUIRES: asserts
 
@@ -10,6 +10,7 @@ target triple = "x86_64-unknown-linux-gnu"
 
 define void @test() {
 ; CHECK-LABEL: 'test'
+; CHECK:  Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
 ; CHECK:  Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
 ; CHECK:  Cost of 3 for VF 4: INTERLEAVE-GROUP with factor 2, ir<%in0>
 ; CHECK:  Cost of 3 for VF 8: INTERLEAVE-GROUP with factor 2, ir<%in0>
diff --git a/llvm/test/tools/UpdateTestChecks/update_analyze_test_checks/loopvectorize-costmodel.test b/llvm/test/tools/UpdateTestChecks/update_analyze_test_checks/loopvectorize-costmodel.test
index c82ea022edd6f9..53f25d3f34d9ad 100644
--- a/llvm/test/tools/UpdateTestChecks/update_analyze_test_checks/loopvectorize-costmodel.test
+++ b/llvm/test/tools/UpdateTestChecks/update_analyze_test_checks/loopvectorize-costmodel.test
@@ -1,11 +1,11 @@
 # REQUIRES: x86-registered-target, asserts
 
-## Check that --filter works properly with both legacy and VPlan cost model output.
-# RUN: cp -f %S/Inputs/x86-loopvectorize-costmodel.ll %t.ll && %update_analyze_test_checks --filter "LV: Found an estimated cost of [0-9]+ for VF [0-9]+ For instruction:\s*%v0 = load float, ptr %in0, align 4" --filter "Cost of [0-9]+ for VF [0-9]+: INTERLEAVE-GROUP with factor 2, ir<%%in0>" %t.ll
+## Check that --filter works properly for both scalar and vector VF cost model output.
+# RUN: cp -f %S/Inputs/x86-loopvectorize-costmodel.ll %t.ll && %update_analyze_test_checks --filter "Cost of [0-9]+ for VF 1: EMIT-SCALAR ir<%%v0> = load" --filter "Cost of [0-9]+ for VF [0-9]+: INTERLEAVE-GROUP with factor 2, ir<%%in0>" %t.ll
 # RUN: diff -u %t.ll %S/Inputs/x86-loopvectorize-costmodel.ll.expected
 
 ## Check that running the script again does not change the result:
-# RUN: %update_analyze_test_checks --filter "LV: Found an estimated cost of [0-9]+ for VF [0-9]+ For instruction:\s*%v0 = load float, ptr %in0, align 4" --filter "Cost of [0-9]+ for VF [0-9]+: INTERLEAVE-GROUP with factor 2, ir<%%in0>" %t.ll
+# RUN: %update_analyze_test_checks --filter "Cost of [0-9]+ for VF 1: EMIT-SCALAR ir<%%v0> = load" --filter "Cost of [0-9]+ for VF [0-9]+: INTERLEAVE-GROUP with factor 2, ir<%%in0>" %t.ll
 # RUN: diff -u %t.ll %S/Inputs/x86-loopvectorize-costmodel.ll.expected
 
 ## Check that running the script again, without arguments, does not change the result:

>From 8c93049f8d4d20d94b7b77ff590b775e5ce6d5b4 Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Tue, 22 Sep 2026 13:57:26 +0100
Subject: [PATCH 6/7] !fixup add -vplan-scalar-cost flag

---
 .../Vectorize/LoopVectorizationPlanner.h      |  3 +-
 .../Transforms/Vectorize/LoopVectorize.cpp    | 52 +++++++++++++++++++
 .../LoopVectorize/X86/uniformshift.ll         |  7 ++-
 3 files changed, 59 insertions(+), 3 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
index 75623179a53347..2bca17c0e59865 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
@@ -908,7 +908,8 @@ class LoopVectorizationPlanner {
   /// been retired.
   InstructionCost cost(VPlan &Plan, ElementCount VF, VPRegisterUsage *RU) const;
 
-  /// Compute the scalar loop cost of InitialVPlan0.
+  /// Compute the scalar loop cost of InitialVPlan0, or using the legacy cost
+  /// model if -vplan-scalar-cost is disabled.
   InstructionCost computeScalarCost() const;
 
   /// Precompute costs for certain instructions using the legacy cost model. The
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 9fbb03482e6dcd..a07956f614596c 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -404,6 +404,13 @@ static cl::opt<cl::boolOrDefault>
                       cl::desc("Override cost based masked intrinsic widening "
                                "for div/rem instructions"));
 
+// TODO: This is a temporary option to disable the VPlan code path in case any
+// regressions surface. Will be removed after a transition period.
+static cl::opt<bool> UseVPlanScalarCost(
+    "vplan-scalar-cost", cl::init(true), cl::Hidden,
+    cl::desc("Compute the cost of the scalar loop using the VPlan-based cost "
+             "model. If disabled, fall back to the legacy cost model."));
+
 static cl::opt<bool> EnableEarlyExitVectorization(
     "enable-early-exit-vectorization", cl::init(true), cl::Hidden,
     cl::desc(
@@ -1282,6 +1289,12 @@ class LoopVectorizationCostModel {
     Scalars.clear();
   }
 
+  /// Returns the expected execution cost. The unit of the cost does
+  /// not matter because we use the 'cost' units to compare different
+  /// vector widths. The cost that is returned is *not* normalized by
+  /// the factor width.
+  InstructionCost expectedCost(ElementCount VF);
+
   /// Returns the execution time cost of an instruction for a given vector
   /// width. Vector width of one means scalar.
   InstructionCost getInstructionCost(Instruction *I, ElementCount VF);
@@ -4227,6 +4240,42 @@ InstructionCost LoopVectorizationCostModel::computePredInstDiscount(
   return Discount;
 }
 
+InstructionCost LoopVectorizationCostModel::expectedCost(ElementCount VF) {
+  InstructionCost Cost;
+  assert(VF.isScalar() && "must only be called for scalar VFs");
+
+  // For each block.
+  for (BasicBlock *BB : TheLoop->blocks()) {
+    InstructionCost BlockCost;
+
+    // For each instruction in the old loop.
+    for (Instruction &I : *BB) {
+      // Skip ignored values.
+      if (ValuesToIgnore.count(&I) ||
+          (VF.isVector() && VecValuesToIgnore.count(&I)))
+        continue;
+
+      InstructionCost C = getInstructionCost(&I, VF);
+
+      // Check if we should override the cost.
+      if (C.isValid() && ForceTargetInstructionCost.getNumOccurrences() > 0)
+        C = InstructionCost(ForceTargetInstructionCost);
+
+      BlockCost += C;
+      LLVM_DEBUG(dbgs() << "LV: Found an estimated cost of " << C << " for VF "
+                        << VF << " For instruction: " << I << '\n');
+    }
+
+    // In the scalar loop, we may not always execute the predicated block, if it
+    // is an if-else block. Thus, scale the block's cost by the probability of
+    // executing it. getPredBlockCostDivisor will return 1 for blocks that are
+    // only predicated by the header mask when folding the tail.
+    Cost += BlockCost / getPredBlockCostDivisor(Config.CostKind, BB);
+  }
+
+  return Cost;
+}
+
 /// Gets the address access SCEV for Ptr, if it should be used for cost modeling
 /// according to isAddressSCEVForCost.
 ///
@@ -5559,6 +5608,9 @@ getRecordedExecutionFrequency(const VPBasicBlock *VPBB) {
 
 InstructionCost LoopVectorizationPlanner::computeScalarCost() const {
   ElementCount ScalarVF = ElementCount::getFixed(1);
+  if (!UseVPlanScalarCost)
+    return CM->expectedCost(ScalarVF);
+
   VPCostContext CostCtx(*TLI, *InitialVPlan0, *CM, Config);
   VPBasicBlock *Header =
       VPBlockUtils::getPlainCFGHeaderAndLatch(*InitialVPlan0).first;
diff --git a/llvm/test/Transforms/LoopVectorize/X86/uniformshift.ll b/llvm/test/Transforms/LoopVectorize/X86/uniformshift.ll
index e38ba170eb6029..7cb076889bd366 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/uniformshift.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/uniformshift.ll
@@ -1,8 +1,11 @@
-; RUN: opt -mtriple=x86_64-apple-darwin -mattr=+sse2 -passes=loop-vectorize -debug-only=loop-vectorize -S < %s 2>&1 | FileCheck %s
+; RUN: opt -mtriple=x86_64-apple-darwin -mattr=+sse2 -passes=loop-vectorize -debug-only=loop-vectorize -S < %s 2>&1 | FileCheck --check-prefixes=CHECK,VPLAN %s
+; Check the scalar costs computed by the legacy cost model.
+; RUN: opt -mtriple=x86_64-apple-darwin -mattr=+sse2 -passes=loop-vectorize -debug-only=loop-vectorize -vplan-scalar-cost=false -S < %s 2>&1 | FileCheck --check-prefixes=CHECK,LEGACY %s
 ; REQUIRES: asserts
 
 ; CHECK: 'foo'
-; CHECK: Cost of 1 for VF 1: EMIT ir<%shift> = ashr ir<%val>, ir<%k>
+; VPLAN: Cost of 1 for VF 1: EMIT ir<%shift> = ashr ir<%val>, ir<%k>
+; LEGACY: LV: Found an estimated cost of 1 for VF 1 For instruction:   %shift = ashr i32 %val, %k
 ; CHECK: Cost of 2 for VF 2: WIDEN ir<%shift> = ashr ir<%val>, ir<%k>
 ; CHECK: Cost of 2 for VF 4: WIDEN ir<%shift> = ashr ir<%val>, ir<%k>
 define void @foo(ptr nocapture %p, i32 %k) {

>From 7cc3d0f5159d196280f3946af32b0a99eac48dd4 Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Wed, 23 Sep 2026 15:23:39 +0100
Subject: [PATCH 7/7] !fxiup aaddress missed comments

---
 .../lib/Transforms/Vectorize/VPlanRecipes.cpp |  36 ++--
 .../X86/CostModel/vpinstruction-cost.ll       | 160 ++++++++++++++++++
 2 files changed, 181 insertions(+), 15 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
index e1e55111ddd2a3..ccff7dee10925e 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
@@ -1581,9 +1581,13 @@ InstructionCost VPInstruction::computeCost(ElementCount VF,
   case Instruction::ExtractValue:
   case Instruction::FNeg:
   case Instruction::Freeze:
-    if (!VF.isScalar() || !getUnderlyingValue())
-      return 0;
-    return getCostForRecipeWithOpcode(getOpcode(), VF, Ctx);
+    if (VF.isScalar())
+      return getCostForRecipeWithOpcode(getOpcode(), VF, Ctx);
+    break;
+  case Instruction::Alloca:
+    assert(VF.isScalar() && "only scalar VF expected");
+    return Ctx.TTI.getArithmeticInstrCost(Instruction::Mul, getScalarType(),
+                                          Ctx.CostKind);
   case Instruction::Load:
   case Instruction::Store:
     assert(VF.isScalar() && "only scalar VF expected");
@@ -1598,23 +1602,25 @@ InstructionCost VPInstruction::computeCost(ElementCount VF,
   }
   case VPInstruction::BranchOnCond:
   case Instruction::PHI:
-    if (!getUnderlyingValue())
-      return 0;
-    return Ctx.TTI.getCFInstrCost(getOpcode() == Instruction::PHI
-                                      ? Instruction::PHI
-                                      : Instruction::CondBr,
-                                  Ctx.CostKind);
+  case Instruction::Switch:
+    if (VF.isScalar())
+      return Ctx.TTI.getCFInstrCost(getOpcode() == VPInstruction::BranchOnCond
+                                        ? Instruction::CondBr
+                                        : getOpcode(),
+                                    Ctx.CostKind);
+    break;
   case VPInstruction::ExtractPenultimateElement:
     if (VF == ElementCount::getScalable(1))
       return InstructionCost::getInvalid();
-    [[fallthrough]];
+    break;
   default:
-    // TODO: Compute cost other VPInstructions once the legacy cost model has
-    // been retired.
-    assert((VF.isScalar() || !getUnderlyingValue()) &&
-           "unexpected VPInstruction with underlying value");
-    return 0;
+    break;
   }
+  // TODO: Compute cost other VPInstructions once the legacy cost model has
+  // been retired.
+  assert((VF.isScalar() || !getUnderlyingValue()) &&
+         "unexpected VPInstruction with underlying value");
+  return 0;
 }
 
 bool VPInstruction::isVectorToScalar() const {
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/vpinstruction-cost.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/vpinstruction-cost.ll
index 6247b3910a609a..f0960b5c0f4c84 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/vpinstruction-cost.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/vpinstruction-cost.ll
@@ -357,3 +357,163 @@ loop:
 exit:
   ret void
 }
+
+define void @test_vpinstruction_alloca_cost(ptr noalias %dst) {
+; CHECK-LABEL: 'test_vpinstruction_alloca_cost'
+; CHECK:  Cost of 0 for VF 1: EMIT-SCALAR ir<%iv> = phi [ ir<0>, vector.ph ], [ ir<%iv.next>, loop ]
+; CHECK:  Cost of 1 for VF 1: EMIT ir<%a> = alloca ir<1>
+; CHECK:  Cost of 0 for VF 1: EMIT ir<%g.dst> = getelementptr inbounds ir<%dst>, ir<%iv>
+; CHECK:  Cost of 1 for VF 1: EMIT store ir<%a>, ir<%g.dst>
+; CHECK:  Cost of 1 for VF 1: EMIT ir<%iv.next> = add nuw nsw ir<%iv>, ir<1>
+; CHECK:  Cost of 1 for VF 1: EMIT ir<%ec> = icmp eq ir<%iv.next>, ir<32>
+; CHECK:  Cost of 0 for VF 1: EMIT branch-on-cond ir<%ec>
+; CHECK:  Cost of 0 for VF 2: vp<[[VP4:%[0-9]+]]> = SCALAR-STEPS vp<[[VP3:%[0-9]+]]>, ir<1>, vp<[[VP0:%[0-9]+]]>
+; CHECK:  Cost of 1 for VF 2: REPLICATE ir<%a> = alloca ir<1>
+; CHECK:  Cost of 0 for VF 2: CLONE ir<%g.dst> = getelementptr inbounds ir<%dst>, vp<[[VP4]]>
+; CHECK:  Cost of 0 for VF 2: vp<[[VP5:%[0-9]+]]> = vector-pointer inbounds ptr, ir<%g.dst>, ir<1>
+; CHECK:  Cost of 1 for VF 2: WIDEN store vp<[[VP5]]>, ir<%a>
+; CHECK:  Cost of 0 for VF 2: EMIT vp<%index.next> = add nuw vp<[[VP3]]>, vp<[[VP1:%[0-9]+]]>
+; CHECK:  Cost of 1 for VF 2: EMIT branch-on-count vp<%index.next>, vp<[[VP2:%[0-9]+]]>
+; CHECK:  Cost of 0 for VF 2: vector loop backedge
+; CHECK:  Cost of 1 for VF 2: canonical IV increment
+; CHECK:  Cost of 0 for VF 2: EMIT-SCALAR vp<%bc.resume.val> = phi [ vp<[[VP2]]>, middle.block ], [ ir<0>, ir-bb<entry> ]
+; CHECK:  Cost of 0 for VF 2: IR %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ] (extra operand: vp<%bc.resume.val> from scalar.ph)
+; CHECK:  Cost of 0 for VF 2: IR %a = alloca i8, align 16
+; CHECK:  Cost of 0 for VF 2: IR %g.dst = getelementptr inbounds ptr, ptr %dst, i64 %iv
+; CHECK:  Cost of 0 for VF 2: IR store ptr %a, ptr %g.dst, align 8
+; CHECK:  Cost of 0 for VF 2: IR %iv.next = add nuw nsw i64 %iv, 1
+; CHECK:  Cost of 0 for VF 2: IR %ec = icmp eq i64 %iv.next, 32
+; CHECK:  Cost of 1 for VF 2: EMIT vp<%cmp.n> = icmp eq ir<32>, vp<[[VP2]]>
+; CHECK:  Cost of 0 for VF 2: EMIT branch-on-cond vp<%cmp.n>
+; CHECK:  Cost of 0 for VF 4: vp<[[VP4]]> = SCALAR-STEPS vp<[[VP3]]>, ir<1>, vp<[[VP0]]>
+; CHECK:  Cost of 1 for VF 4: REPLICATE ir<%a> = alloca ir<1>
+; CHECK:  Cost of 0 for VF 4: CLONE ir<%g.dst> = getelementptr inbounds ir<%dst>, vp<[[VP4]]>
+; CHECK:  Cost of 0 for VF 4: vp<[[VP5]]> = vector-pointer inbounds ptr, ir<%g.dst>, ir<1>
+; CHECK:  Cost of 1 for VF 4: WIDEN store vp<[[VP5]]>, ir<%a>
+; CHECK:  Cost of 0 for VF 4: EMIT vp<%index.next> = add nuw vp<[[VP3]]>, vp<[[VP1]]>
+; CHECK:  Cost of 1 for VF 4: EMIT branch-on-count vp<%index.next>, vp<[[VP2]]>
+; CHECK:  Cost of 0 for VF 4: vector loop backedge
+; CHECK:  Cost of 1 for VF 4: canonical IV increment
+; CHECK:  Cost of 0 for VF 4: EMIT-SCALAR vp<%bc.resume.val> = phi [ vp<[[VP2]]>, middle.block ], [ ir<0>, ir-bb<entry> ]
+; CHECK:  Cost of 0 for VF 4: IR %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ] (extra operand: vp<%bc.resume.val> from scalar.ph)
+; CHECK:  Cost of 0 for VF 4: IR %a = alloca i8, align 16
+; CHECK:  Cost of 0 for VF 4: IR %g.dst = getelementptr inbounds ptr, ptr %dst, i64 %iv
+; CHECK:  Cost of 0 for VF 4: IR store ptr %a, ptr %g.dst, align 8
+; CHECK:  Cost of 0 for VF 4: IR %iv.next = add nuw nsw i64 %iv, 1
+; CHECK:  Cost of 0 for VF 4: IR %ec = icmp eq i64 %iv.next, 32
+; CHECK:  Cost of 1 for VF 4: EMIT vp<%cmp.n> = icmp eq ir<32>, vp<[[VP2]]>
+; CHECK:  Cost of 0 for VF 4: EMIT branch-on-cond vp<%cmp.n>
+; CHECK:  Cost of 1 for VF 4: EMIT vp<%cmp.n> = icmp eq ir<32>, vp<[[VP2]]>
+; CHECK:  Cost of 0 for VF 4: EMIT branch-on-cond vp<%cmp.n>
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %a = alloca i8, align 16
+  %g.dst = getelementptr inbounds ptr, ptr %dst, i64 %iv
+  store ptr %a, ptr %g.dst, align 8
+  %iv.next = add nuw nsw i64 %iv, 1
+  %ec = icmp eq i64 %iv.next, 32
+  br i1 %ec, label %exit, label %loop
+
+exit:
+  ret void
+}
+
+; Switch is free for TCK_RecipThroughput on X86, use minsize (TCK_CodeSize) to
+; expose its cost.
+define void @test_vpinstruction_switch_cost_minsize(ptr noalias %dst) minsize {
+; CHECK-LABEL: 'test_vpinstruction_switch_cost_minsize'
+; CHECK:  Cost of 0 for VF 1: EMIT-SCALAR ir<%iv> = phi [ ir<0>, vector.ph ], [ ir<%iv.next>, loop.latch ]
+; CHECK:  Cost of 0 for VF 1: EMIT ir<%g.dst> = getelementptr inbounds ir<%dst>, ir<%iv>
+; CHECK:  Cost of 1 for VF 1: EMIT-SCALAR ir<%l> = load ir<%g.dst>
+; CHECK:  Cost of 1 for VF 1: EMIT switch ir<%l>, ir<-12>, ir<13> (!vplan.prof.estimated estimated {715827883, 715827883, 715827883})
+; CHECK:  Cost of 2 for VF 1: EMIT store ir<0>, ir<%g.dst> (!vplan.execution.frequency 3074457347049914368 (33.33%, estimated))
+; CHECK:  Cost of 2 for VF 1: EMIT store ir<42>, ir<%g.dst> (!vplan.execution.frequency 3074457347049914368 (33.33%, estimated))
+; CHECK:  Cost of 2 for VF 1: EMIT store ir<2>, ir<%g.dst> (!vplan.execution.frequency 3074457347049914368 (33.33%, estimated))
+; CHECK:  Cost of 1 for VF 1: EMIT ir<%iv.next> = add nuw nsw ir<%iv>, ir<1>
+; CHECK:  Cost of 1 for VF 1: EMIT ir<%ec> = icmp eq ir<%iv.next>, ir<32>
+; CHECK:  Cost of 1 for VF 1: EMIT branch-on-cond ir<%ec>
+; CHECK:  Cost of 0 for VF 2: vp<[[VP4:%[0-9]+]]> = SCALAR-STEPS vp<[[VP3:%[0-9]+]]>, ir<1>, vp<[[VP0:%[0-9]+]]>
+; CHECK:  Cost of 0 for VF 2: CLONE ir<%g.dst> = getelementptr ir<%dst>, vp<[[VP4]]>
+; CHECK:  Cost of 0 for VF 2: vp<[[VP5:%[0-9]+]]> = vector-pointer inbounds i64, ir<%g.dst>, ir<1>
+; CHECK:  Cost of 1 for VF 2: WIDEN ir<%l> = load vp<[[VP5]]>
+; CHECK:  Cost of 1 for VF 2: EMIT vp<[[VP6:%[0-9]+]]> = icmp eq ir<%l>, ir<-12>
+; CHECK:  Cost of 1 for VF 2: EMIT vp<[[VP7:%[0-9]+]]> = icmp eq ir<%l>, ir<13>
+; CHECK:  Cost of 0 for VF 2: EMIT vp<[[VP8:%[0-9]+]]> = or vp<[[VP6]]>, vp<[[VP7]]>
+; CHECK:  Cost of 1 for VF 2: EMIT vp<[[VP9:%[0-9]+]]> = not vp<[[VP8]]>
+; CHECK:  Cost of 0 for VF 2: vp<[[VP10:%[0-9]+]]> = vector-pointer i64, ir<%g.dst>, ir<1>
+; CHECK:  Cost of 1 for VF 2: WIDEN store vp<[[VP10]]>, ir<0>, vp<[[VP7]]> (!vplan.execution.frequency 3074457347049914368 (33.33%, estimated))
+; CHECK:  Cost of 0 for VF 2: vp<[[VP11:%[0-9]+]]> = vector-pointer i64, ir<%g.dst>, ir<1>
+; CHECK:  Cost of 1 for VF 2: WIDEN store vp<[[VP11]]>, ir<42>, vp<[[VP6]]> (!vplan.execution.frequency 3074457347049914368 (33.33%, estimated))
+; CHECK:  Cost of 0 for VF 2: vp<[[VP12:%[0-9]+]]> = vector-pointer i64, ir<%g.dst>, ir<1>
+; CHECK:  Cost of 1 for VF 2: WIDEN store vp<[[VP12]]>, ir<2>, vp<[[VP9]]> (!vplan.execution.frequency 3074457347049914368 (33.33%, estimated))
+; CHECK:  Cost of 0 for VF 2: EMIT vp<%index.next> = add nuw vp<[[VP3]]>, vp<[[VP1:%[0-9]+]]>
+; CHECK:  Cost of 1 for VF 2: EMIT branch-on-count vp<%index.next>, vp<[[VP2:%[0-9]+]]>
+; CHECK:  Cost of 1 for VF 2: vector loop backedge
+; CHECK:  Cost of 1 for VF 2: canonical IV increment
+; CHECK:  Cost of 0 for VF 2: EMIT-SCALAR vp<%bc.resume.val> = phi [ vp<[[VP2]]>, middle.block ], [ ir<0>, ir-bb<entry> ]
+; CHECK:  Cost of 0 for VF 2: IR %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop.latch ] (extra operand: vp<%bc.resume.val> from scalar.ph)
+; CHECK:  Cost of 0 for VF 2: IR %g.dst = getelementptr inbounds i64, ptr %dst, i64 %iv
+; CHECK:  Cost of 0 for VF 2: IR %l = load i64, ptr %g.dst, align 8
+; CHECK:  Cost of 1 for VF 2: EMIT vp<%cmp.n> = icmp eq ir<32>, vp<[[VP2]]>
+; CHECK:  Cost of 0 for VF 2: EMIT branch-on-cond vp<%cmp.n>
+; CHECK:  Cost of 0 for VF 4: vp<[[VP4]]> = SCALAR-STEPS vp<[[VP3]]>, ir<1>, vp<[[VP0]]>
+; CHECK:  Cost of 0 for VF 4: CLONE ir<%g.dst> = getelementptr ir<%dst>, vp<[[VP4]]>
+; CHECK:  Cost of 0 for VF 4: vp<[[VP5]]> = vector-pointer inbounds i64, ir<%g.dst>, ir<1>
+; CHECK:  Cost of 1 for VF 4: WIDEN ir<%l> = load vp<[[VP5]]>
+; CHECK:  Cost of 1 for VF 4: EMIT vp<[[VP6]]> = icmp eq ir<%l>, ir<-12>
+; CHECK:  Cost of 1 for VF 4: EMIT vp<[[VP7]]> = icmp eq ir<%l>, ir<13>
+; CHECK:  Cost of 0 for VF 4: EMIT vp<[[VP8]]> = or vp<[[VP6]]>, vp<[[VP7]]>
+; CHECK:  Cost of 1 for VF 4: EMIT vp<[[VP9]]> = not vp<[[VP8]]>
+; CHECK:  Cost of 0 for VF 4: vp<[[VP10]]> = vector-pointer i64, ir<%g.dst>, ir<1>
+; CHECK:  Cost of 1 for VF 4: WIDEN store vp<[[VP10]]>, ir<0>, vp<[[VP7]]> (!vplan.execution.frequency 3074457347049914368 (33.33%, estimated))
+; CHECK:  Cost of 0 for VF 4: vp<[[VP11]]> = vector-pointer i64, ir<%g.dst>, ir<1>
+; CHECK:  Cost of 1 for VF 4: WIDEN store vp<[[VP11]]>, ir<42>, vp<[[VP6]]> (!vplan.execution.frequency 3074457347049914368 (33.33%, estimated))
+; CHECK:  Cost of 0 for VF 4: vp<[[VP12]]> = vector-pointer i64, ir<%g.dst>, ir<1>
+; CHECK:  Cost of 1 for VF 4: WIDEN store vp<[[VP12]]>, ir<2>, vp<[[VP9]]> (!vplan.execution.frequency 3074457347049914368 (33.33%, estimated))
+; CHECK:  Cost of 0 for VF 4: EMIT vp<%index.next> = add nuw vp<[[VP3]]>, vp<[[VP1]]>
+; CHECK:  Cost of 1 for VF 4: EMIT branch-on-count vp<%index.next>, vp<[[VP2]]>
+; CHECK:  Cost of 1 for VF 4: vector loop backedge
+; CHECK:  Cost of 1 for VF 4: canonical IV increment
+; CHECK:  Cost of 0 for VF 4: EMIT-SCALAR vp<%bc.resume.val> = phi [ vp<[[VP2]]>, middle.block ], [ ir<0>, ir-bb<entry> ]
+; CHECK:  Cost of 0 for VF 4: IR %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop.latch ] (extra operand: vp<%bc.resume.val> from scalar.ph)
+; CHECK:  Cost of 0 for VF 4: IR %g.dst = getelementptr inbounds i64, ptr %dst, i64 %iv
+; CHECK:  Cost of 0 for VF 4: IR %l = load i64, ptr %g.dst, align 8
+; CHECK:  Cost of 1 for VF 4: EMIT vp<%cmp.n> = icmp eq ir<32>, vp<[[VP2]]>
+; CHECK:  Cost of 0 for VF 4: EMIT branch-on-cond vp<%cmp.n>
+;
+entry:
+  br label %loop.header
+
+loop.header:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop.latch ]
+  %g.dst = getelementptr inbounds i64, ptr %dst, i64 %iv
+  %l = load i64, ptr %g.dst, align 8
+  switch i64 %l, label %default [
+    i64 -12, label %case1
+    i64 13, label %case2
+  ]
+
+case1:
+  store i64 42, ptr %g.dst, align 8
+  br label %loop.latch
+
+case2:
+  store i64 0, ptr %g.dst, align 8
+  br label %loop.latch
+
+default:
+  store i64 2, ptr %g.dst, align 8
+  br label %loop.latch
+
+loop.latch:
+  %iv.next = add nuw nsw i64 %iv, 1
+  %ec = icmp eq i64 %iv.next, 32
+  br i1 %ec, label %exit, label %loop.header
+
+exit:
+  ret void
+}



More information about the llvm-commits mailing list