[llvm] [VPlan] Compute scalar cost based on VPlan0 instead of legacy CM. (PR #196845)
Florian Hahn via llvm-commits
llvm-commits at lists.llvm.org
Wed Sep 23 07:26:38 PDT 2026
https://github.com/fhahn updated https://github.com/llvm/llvm-project/pull/196845
>From e50eb1045f2130a5d3e4643351aee22fe3a5ca1a Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Fri, 29 May 2026 11:53:44 +0100
Subject: [PATCH 1/7] [VPlan] Compute scalar cost based on VPlan0 instead of
legacy CM.
Replace legacy cost computation for the scalar loop with a VPlan-based
computation on VPlan0. This requires keeping around a copy of the
initial VPlan0, before introducing vectorization specific concepts (like
predication etc), to accurately reflect the cost of the scalar loop.
---
.../Vectorize/LoopVectorizationPlanner.h | 6 +
.../Transforms/Vectorize/LoopVectorize.cpp | 74 +++++-------
.../lib/Transforms/Vectorize/VPlanRecipes.cpp | 68 ++++++++++--
.../LoopVectorize/AArch64/arith-costs.ll | 6 +-
.../AArch64/arith-fp-frem-costs.ll | 32 +++---
.../LoopVectorize/AArch64/intrinsiccost.ll | 9 --
.../AArch64/multiple-result-intrinsics.ll | 32 +++---
.../AArch64/struct-return-cost.ll | 22 ++--
.../AArch64/type-shrinkage-zext-costs.ll | 2 +-
.../LoopVectorize/ARM/mve-icmpcost.ll | 103 +++++++++--------
.../LoopVectorize/ARM/mve-selectandorcost.ll | 2 +-
.../LoopVectorize/ARM/mve-shiftcost.ll | 4 +-
.../LoopVectorize/ARM/scalar-block-cost.ll | 61 +++++-----
.../LoopVectorize/SystemZ/pr47665.ll | 56 +++-------
.../WebAssembly/int-mac-reduction-costs.ll | 76 ++++++-------
.../X86/CostModel/gather-i16-with-i8-index.ll | 10 +-
.../X86/CostModel/gather-i32-with-i8-index.ll | 12 +-
.../X86/CostModel/gather-i64-with-i8-index.ll | 12 +-
.../X86/CostModel/gather-i8-with-i8-index.ll | 12 +-
...dle-iptr-with-data-layout-to-not-assert.ll | 1 -
.../interleaved-load-f32-stride-2.ll | 4 -
.../interleaved-load-f32-stride-3.ll | 4 -
.../interleaved-load-f32-stride-4.ll | 4 -
.../interleaved-load-f32-stride-5.ll | 4 -
.../interleaved-load-f32-stride-6.ll | 4 -
.../interleaved-load-f32-stride-7.ll | 4 -
.../interleaved-load-f32-stride-8.ll | 4 -
.../interleaved-load-f64-stride-2.ll | 4 -
.../interleaved-load-f64-stride-3.ll | 4 -
.../interleaved-load-f64-stride-4.ll | 4 -
.../interleaved-load-f64-stride-5.ll | 4 -
.../interleaved-load-f64-stride-6.ll | 4 -
.../interleaved-load-f64-stride-7.ll | 4 -
.../interleaved-load-f64-stride-8.ll | 4 -
.../interleaved-load-i16-stride-2.ll | 5 -
.../interleaved-load-i16-stride-3.ll | 5 -
.../interleaved-load-i16-stride-4.ll | 5 -
.../interleaved-load-i16-stride-5.ll | 5 -
.../interleaved-load-i16-stride-6.ll | 5 -
.../interleaved-load-i16-stride-7.ll | 5 -
.../interleaved-load-i16-stride-8.ll | 5 -
...nterleaved-load-i32-stride-2-indices-0u.ll | 4 -
.../interleaved-load-i32-stride-2.ll | 4 -
...terleaved-load-i32-stride-3-indices-01u.ll | 4 -
...terleaved-load-i32-stride-3-indices-0uu.ll | 4 -
.../interleaved-load-i32-stride-3.ll | 4 -
...erleaved-load-i32-stride-4-indices-012u.ll | 4 -
...erleaved-load-i32-stride-4-indices-01uu.ll | 4 -
...erleaved-load-i32-stride-4-indices-0uuu.ll | 4 -
.../interleaved-load-i32-stride-4.ll | 4 -
.../interleaved-load-i32-stride-5.ll | 4 -
.../interleaved-load-i32-stride-6.ll | 4 -
.../interleaved-load-i32-stride-7.ll | 4 -
.../interleaved-load-i32-stride-8.ll | 4 -
.../interleaved-load-i64-stride-2.ll | 4 -
.../interleaved-load-i64-stride-3.ll | 4 -
.../interleaved-load-i64-stride-4.ll | 4 -
.../interleaved-load-i64-stride-5.ll | 4 -
.../interleaved-load-i64-stride-6.ll | 4 -
.../interleaved-load-i64-stride-7.ll | 4 -
.../interleaved-load-i64-stride-8.ll | 4 -
.../CostModel/interleaved-load-i8-stride-2.ll | 5 -
.../CostModel/interleaved-load-i8-stride-3.ll | 5 -
.../CostModel/interleaved-load-i8-stride-4.ll | 5 -
.../CostModel/interleaved-load-i8-stride-5.ll | 5 -
.../CostModel/interleaved-load-i8-stride-6.ll | 5 -
.../CostModel/interleaved-load-i8-stride-7.ll | 5 -
.../CostModel/interleaved-load-i8-stride-8.ll | 5 -
.../masked-gather-i32-with-i8-index.ll | 32 +++---
.../masked-gather-i64-with-i8-index.ll | 32 +++---
.../CostModel/masked-interleaved-load-i16.ll | 20 +---
.../CostModel/masked-interleaved-store-i16.ll | 10 --
.../X86/CostModel/masked-load-i16.ll | 8 +-
.../X86/CostModel/masked-load-i32.ll | 8 +-
.../X86/CostModel/masked-load-i64.ll | 8 +-
.../X86/CostModel/masked-load-i8.ll | 8 +-
.../masked-scatter-i32-with-i8-index.ll | 13 +--
.../masked-scatter-i64-with-i8-index.ll | 13 +--
.../X86/CostModel/masked-store-i16.ll | 4 -
.../X86/CostModel/masked-store-i32.ll | 5 -
.../X86/CostModel/masked-store-i64.ll | 5 -
.../X86/CostModel/masked-store-i8.ll | 5 -
.../CostModel/scatter-i16-with-i8-index.ll | 5 -
.../CostModel/scatter-i32-with-i8-index.ll | 5 -
.../CostModel/scatter-i64-with-i8-index.ll | 5 -
.../X86/CostModel/scatter-i8-with-i8-index.ll | 5 -
.../X86/CostModel/strided-load-i16.ll | 4 -
.../X86/CostModel/strided-load-i32.ll | 4 -
.../X86/CostModel/strided-load-i64.ll | 3 -
.../X86/CostModel/strided-load-i8.ll | 4 -
.../X86/CostModel/vpinstruction-cost.ll | 105 +++++++++---------
.../Transforms/LoopVectorize/X86/fneg-cost.ll | 41 ++++++-
.../X86/fp_to_sint8-cost-model.ll | 2 +-
.../LoopVectorize/X86/reduction-small-size.ll | 30 ++---
.../X86/uint64_to_fp64-cost-model.ll | 2 +-
.../LoopVectorize/X86/uniformshift.ll | 2 +-
.../X86/vector-scalar-select-cost.ll | 4 +-
97 files changed, 471 insertions(+), 727 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
index cb38b0be1808aa..75623179a53347 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
@@ -889,6 +889,9 @@ class LoopVectorizationPlanner {
SmallVector<VPlanPtr, 4> VPlans;
+ /// Copy of scalar VPlan0; used for scalar cost computation.
+ VPlanPtr InitialVPlan0;
+
/// Profitable vector factors.
SmallVector<VectorizationFactor, 8> ProfitableVFs;
@@ -905,6 +908,9 @@ class LoopVectorizationPlanner {
/// been retired.
InstructionCost cost(VPlan &Plan, ElementCount VF, VPRegisterUsage *RU) const;
+ /// Compute the scalar loop cost of InitialVPlan0.
+ InstructionCost computeScalarCost() const;
+
/// Precompute costs for certain instructions using the legacy cost model. The
/// function is used to bring up the VPlan-based cost model to initially avoid
/// taking different decisions due to inaccuracies in the legacy cost model.
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index d929af8afbd1da..7248923bf84761 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -1282,12 +1282,6 @@ class LoopVectorizationCostModel {
Scalars.clear();
}
- /// Returns the expected execution cost. The unit of the cost does
- /// not matter because we use the 'cost' units to compare different
- /// vector widths. The cost that is returned is *not* normalized by
- /// the factor width.
- InstructionCost expectedCost(ElementCount VF);
-
/// Returns the execution time cost of an instruction for a given vector
/// width. Vector width of one means scalar.
InstructionCost getInstructionCost(Instruction *I, ElementCount VF);
@@ -3755,7 +3749,7 @@ LoopVectorizationPlanner::selectInterleaveCount(VPlan &Plan, ElementCount VF,
// then we calculate the cost of VF here.
if (LoopCost == 0) {
if (VF.isScalar())
- LoopCost = CM->expectedCost(VF);
+ LoopCost = computeScalarCost();
else
LoopCost = cost(Plan, VF, &R);
assert(LoopCost.isValid() && "Expected to have chosen a VF with valid cost");
@@ -4233,42 +4227,6 @@ InstructionCost LoopVectorizationCostModel::computePredInstDiscount(
return Discount;
}
-InstructionCost LoopVectorizationCostModel::expectedCost(ElementCount VF) {
- InstructionCost Cost;
- assert(VF.isScalar() && "must only be called for scalar VFs");
-
- // For each block.
- for (BasicBlock *BB : TheLoop->blocks()) {
- InstructionCost BlockCost;
-
- // For each instruction in the old loop.
- for (Instruction &I : *BB) {
- // Skip ignored values.
- if (ValuesToIgnore.count(&I) ||
- (VF.isVector() && VecValuesToIgnore.count(&I)))
- continue;
-
- InstructionCost C = getInstructionCost(&I, VF);
-
- // Check if we should override the cost.
- if (C.isValid() && ForceTargetInstructionCost.getNumOccurrences() > 0)
- C = InstructionCost(ForceTargetInstructionCost);
-
- BlockCost += C;
- LLVM_DEBUG(dbgs() << "LV: Found an estimated cost of " << C << " for VF "
- << VF << " For instruction: " << I << '\n');
- }
-
- // In the scalar loop, we may not always execute the predicated block, if it
- // is an if-else block. Thus, scale the block's cost by the probability of
- // executing it. getPredBlockCostDivisor will return 1 for blocks that are
- // only predicated by the header mask when folding the tail.
- Cost += BlockCost / getPredBlockCostDivisor(Config.CostKind, BB);
- }
-
- return Cost;
-}
-
/// Gets the address access SCEV for Ptr, if it should be used for cost modeling
/// according to isAddressSCEVForCost.
///
@@ -5599,6 +5557,30 @@ getRecordedExecutionFrequency(const VPBasicBlock *VPBB) {
}
#endif
+InstructionCost LoopVectorizationPlanner::computeScalarCost() const {
+ ElementCount ScalarVF = ElementCount::getFixed(1);
+ VPCostContext CostCtx(*TLI, *InitialVPlan0, *CM, Config);
+ VPBasicBlock *Header =
+ VPBlockUtils::getPlainCFGHeaderAndLatch(*InitialVPlan0).first;
+ InstructionCost Cost = 0;
+
+ for (VPBasicBlock *VPBB : vp_rpo_plain_cfg_loop_body(Header)) {
+ // Look up the divisor via the first underlying IR instruction in the loop.
+ uint64_t Divisor = 1;
+ for (const VPRecipeBase &R : *VPBB) {
+ auto *UI = dyn_cast_if_present<Instruction>(
+ cast<VPSingleDefRecipe>(&R)->getUnderlyingValue());
+ if (!UI)
+ continue;
+ Divisor = CostCtx.CM.getPredBlockCostDivisor(CostCtx.CostKind,
+ UI->getParent());
+ break;
+ }
+ Cost += VPBB->cost(ScalarVF, CostCtx) / Divisor;
+ }
+ return Cost;
+}
+
InstructionCost LoopVectorizationPlanner::cost(VPlan &Plan, ElementCount VF,
VPRegisterUsage *RU) const {
VPCostContext CostCtx(*TLI, Plan, *CM, Config,
@@ -5675,8 +5657,8 @@ LoopVectorizationPlanner::computeBestVF() {
assert(FirstPlan.hasVF(ScalarVF) &&
"More than a single plan/VF w/o any plan having scalar VF");
- // TODO: Compute scalar cost using VPlan-based cost model.
- InstructionCost ScalarCost = CM->expectedCost(ScalarVF);
+ // Compute the scalar cost from VPlan0.
+ InstructionCost ScalarCost = computeScalarCost();
LLVM_DEBUG(dbgs() << "LV: Scalar loop costs: " << ScalarCost << ".\n");
VectorizationFactor ScalarFactor(ScalarVF, ScalarCost, ScalarCost);
VectorizationFactor BestFactor = ScalarFactor;
@@ -6456,6 +6438,8 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan1() {
assert(verifyExecutionFrequenciesMatchBFI(*VPlan0, OrigLoop, LI, *CM) &&
"execution frequencies do not match the loop's block frequencies");
}
+ // Save copy of VPlan0 for scalar cost computation.
+ InitialVPlan0 = VPlanPtr(VPlan0->duplicate());
// Create recipes for header phis. For outer loops, reductions, recurrences
// and in-loop reductions are empty since legality doesn't detect them.
diff --git a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
index 38c94512f05464..0304127b9907c0 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
@@ -1158,7 +1158,6 @@ InstructionCost VPRecipeWithIRFlags::getCostForRecipeWithOpcode(
Type *ResultTy = VF.isVector() ? toVectorTy(ScalarTy, VF) : ScalarTy;
switch (Opcode) {
case Instruction::FNeg:
- return Ctx.TTI.getArithmeticInstrCost(Opcode, ResultTy, Ctx.CostKind);
case Instruction::UDiv:
case Instruction::SDiv:
case Instruction::SRem:
@@ -1179,12 +1178,14 @@ InstructionCost VPRecipeWithIRFlags::getCostForRecipeWithOpcode(
case Instruction::Xor: {
// Certain instructions can be cheaper if they have a constant second
// operand. One example of this are shifts on x86.
- VPValue *RHS = getOperand(1);
- TargetTransformInfo::OperandValueInfo RHSInfo = Ctx.getOperandInfo(RHS);
-
- if (RHSInfo.Kind == TargetTransformInfo::OK_AnyValue &&
- getOperand(1)->isDefinedOutsideLoopRegions())
- RHSInfo.Kind = TargetTransformInfo::OK_UniformValue;
+ TargetTransformInfo::OperandValueInfo RHSInfo = {
+ TargetTransformInfo::OK_AnyValue, TargetTransformInfo::OP_None};
+ if (Opcode != Instruction::FNeg) {
+ RHSInfo = Ctx.getOperandInfo(getOperand(1));
+ if (RHSInfo.Kind == TargetTransformInfo::OK_AnyValue &&
+ getOperand(1)->isDefinedOutsideLoopRegions())
+ RHSInfo.Kind = TargetTransformInfo::OK_UniformValue;
+ }
Instruction *CtxI = dyn_cast_or_null<Instruction>(getUnderlyingValue());
SmallVector<const Value *, 4> Operands;
@@ -1205,6 +1206,19 @@ InstructionCost VPRecipeWithIRFlags::getCostForRecipeWithOpcode(
case Instruction::ExtractValue:
return Ctx.TTI.getInsertExtractValueCost(Instruction::ExtractValue,
Ctx.CostKind);
+ case Instruction::Load:
+ case Instruction::Store: {
+ bool IsLoad = Opcode == Instruction::Load;
+ const Instruction *UI = getUnderlyingInstr();
+ Type *ValTy = (IsLoad ? this : getOperand(0))->getScalarType();
+ Type *PtrTy = getOperand(!IsLoad)->getScalarType();
+ return Ctx.TTI.getAddressComputationCost(PtrTy, nullptr, nullptr,
+ Ctx.CostKind) +
+ Ctx.TTI.getMemoryOpCost(Opcode, ValTy, getLoadStoreAlignment(UI),
+ cast<PointerType>(PtrTy)->getAddressSpace(),
+ Ctx.CostKind,
+ TTI::getOperandInfo(UI->getOperand(0)), UI);
+ }
case Instruction::ICmp:
case Instruction::FCmp: {
Type *ScalarOpTy = getOperand(0)->getScalarType();
@@ -1247,6 +1261,12 @@ InstructionCost VPRecipeWithIRFlags::getCostForRecipeWithOpcode(
return ReplicateRecipe->isPredicated() ? TTI::CastContextHint::Masked
: TTI::CastContextHint::Normal;
}
+ // Loads/stores in pre-predication VPlan0 are represented as
+ // VPInstructions; treat them like an unmasked memory access.
+ if (const auto *VPI = dyn_cast<VPInstruction>(R))
+ if (VPI->getOpcode() == Instruction::Load ||
+ VPI->getOpcode() == Instruction::Store)
+ return TTI::CastContextHint::Normal;
const auto *WidenMemoryRecipe = dyn_cast<VPWidenMemoryRecipe>(R);
if (WidenMemoryRecipe == nullptr)
return TTI::CastContextHint::None;
@@ -1357,6 +1377,12 @@ InstructionCost VPRecipeWithIRFlags::getCostForRecipeWithOpcode(
InstructionCost VPInstruction::computeCost(ElementCount VF,
VPCostContext &Ctx) const {
+ // Vector-only opcodes have zero cost at scalar VF.
+ if (VF.isScalar() &&
+ (isVectorToScalar() ||
+ getOpcode() == VPInstruction::FirstOrderRecurrenceSplice))
+ return 0;
+
// NOTE: At the moment it seems only possible to expose this path for
// the trunc, zext and sext opcodes.
// TODO: Update VF arg to use onlyFirstLaneUsed once WidenCast is unified.
@@ -1552,6 +1578,32 @@ InstructionCost VPInstruction::computeCost(ElementCount VF,
return getCostForRecipeWithOpcode(
getOpcode(),
vputils::onlyFirstLaneUsed(this) ? ElementCount::getFixed(1) : VF, Ctx);
+ case Instruction::ExtractValue:
+ case Instruction::FNeg:
+ case Instruction::Freeze:
+ if (!VF.isScalar() || !getUnderlyingValue())
+ return 0;
+ return getCostForRecipeWithOpcode(getOpcode(), VF, Ctx);
+ case Instruction::Load:
+ case Instruction::Store:
+ assert(VF.isScalar() && "only scalar VF expected");
+ return getCostForRecipeWithOpcode(getOpcode(), VF, Ctx);
+ case Instruction::Call: {
+ assert(VF.isScalar() && "only scalar VF expected");
+ auto *CalledFn =
+ cast<Function>(getOperand(getNumOperands() - 1)->getLiveInIRValue());
+ SmallVector<const VPValue *> ArgOps(drop_end(operands()));
+ return VPReplicateRecipe::computeCallCost(CalledFn, getScalarType(), ArgOps,
+ /*IsSingleScalar=*/true, VF, Ctx);
+ }
+ case VPInstruction::BranchOnCond:
+ case Instruction::PHI:
+ if (!getUnderlyingValue())
+ return 0;
+ return Ctx.TTI.getCFInstrCost(getOpcode() == Instruction::PHI
+ ? Instruction::PHI
+ : Instruction::CondBr,
+ Ctx.CostKind);
case VPInstruction::ExtractPenultimateElement:
if (VF == ElementCount::getScalable(1))
return InstructionCost::getInvalid();
@@ -1559,8 +1611,6 @@ InstructionCost VPInstruction::computeCost(ElementCount VF,
default:
// TODO: Compute cost other VPInstructions once the legacy cost model has
// been retired.
- assert(!getUnderlyingValue() &&
- "unexpected VPInstruction witht underlying value");
return 0;
}
}
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/arith-costs.ll b/llvm/test/Transforms/LoopVectorize/AArch64/arith-costs.ll
index 9b02fd24c828d2..e8b055a97804c5 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/arith-costs.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/arith-costs.ll
@@ -7,7 +7,7 @@ target triple = "arm64-apple-macosx"
define void @udiv_rhs_opt_cost(ptr %dst) #0 {
; CHECK-LABEL: 'udiv_rhs_opt_cost'
-; CHECK: LV: Found an estimated cost of 5 for VF 1 For instruction: %div = udiv i8 %iv.trunc, 3
+; CHECK: Cost of 5 for VF 1: EMIT ir<%div> = udiv ir<%iv.trunc>, ir<3>
; CHECK: Cost of 5 for VF 2: CLONE ir<%div> = udiv ir<%iv.trunc>, ir<3>
; CHECK: Cost of 0 for VF 2: IR %div = udiv i8 %iv.trunc, 3
; CHECK: Cost of 5 for VF 4: CLONE ir<%div> = udiv ir<%iv.trunc>, ir<3>
@@ -38,8 +38,8 @@ exit:
define void @fneg_used_by_fmul_scalar_cost_is_zero(ptr %dst) #0 {
; CHECK-LABEL: 'fneg_used_by_fmul_scalar_cost_is_zero'
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction: %neg = fneg double %conv
-; CHECK: LV: Found an estimated cost of 2 for VF 1 For instruction: %mul = fmul double %neg, 2.500000e-04
+; CHECK: Cost of 0 for VF 1: EMIT ir<%neg> = fneg ir<%conv>
+; CHECK: Cost of 2 for VF 1: EMIT ir<%mul> = fmul ir<%neg>, ir<2.500000e-04>
; CHECK: Cost of 1 for VF 2: WIDEN ir<%neg> = fneg ir<%conv>
; CHECK: Cost of 2 for VF 2: WIDEN ir<%mul> = fmul ir<%neg>, ir<2.500000e-04>
; CHECK: Cost of 0 for VF 2: IR %neg = fneg double %conv
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/arith-fp-frem-costs.ll b/llvm/test/Transforms/LoopVectorize/AArch64/arith-fp-frem-costs.ll
index 511244b46af148..06adbe718b9979 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/arith-fp-frem-costs.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/arith-fp-frem-costs.ll
@@ -12,43 +12,43 @@ target triple = "aarch64-unknown-linux-gnu"
define void @frem_f64(ptr noalias %in.ptr, ptr noalias %out.ptr) {
; NEON-NO-VECLIB-LABEL: 'frem_f64'
-; NEON-NO-VECLIB: LV: Found an estimated cost of 10 for VF 1 For instruction: %res = frem double %in, %in
+; NEON-NO-VECLIB: Cost of 10 for VF 1: EMIT ir<%res> = frem ir<%in>, ir<%in>
; NEON-NO-VECLIB: Cost of 24 for VF 2: WIDEN ir<%res> = frem ir<%in>, ir<%in>
;
; SVE-NO-VECLIB-LABEL: 'frem_f64'
-; SVE-NO-VECLIB: LV: Found an estimated cost of 10 for VF 1 For instruction: %res = frem double %in, %in
+; SVE-NO-VECLIB: Cost of 10 for VF 1: EMIT ir<%res> = frem ir<%in>, ir<%in>
; SVE-NO-VECLIB: Cost of 24 for VF 2: WIDEN ir<%res> = frem ir<%in>, ir<%in>
; SVE-NO-VECLIB: Cost of Invalid for VF vscale x 1: WIDEN ir<%res> = frem ir<%in>, ir<%in>
; SVE-NO-VECLIB: Cost of Invalid for VF vscale x 2: WIDEN ir<%res> = frem ir<%in>, ir<%in>
;
; NEON-ARMPL-LABEL: 'frem_f64'
-; NEON-ARMPL: LV: Found an estimated cost of 10 for VF 1 For instruction: %res = frem double %in, %in
+; NEON-ARMPL: Cost of 10 for VF 1: EMIT ir<%res> = frem ir<%in>, ir<%in>
; NEON-ARMPL: Cost of 10 for VF 2: WIDEN ir<%res> = frem ir<%in>, ir<%in>
;
; NEON-SLEEF-LABEL: 'frem_f64'
-; NEON-SLEEF: LV: Found an estimated cost of 10 for VF 1 For instruction: %res = frem double %in, %in
+; NEON-SLEEF: Cost of 10 for VF 1: EMIT ir<%res> = frem ir<%in>, ir<%in>
; NEON-SLEEF: Cost of 10 for VF 2: WIDEN ir<%res> = frem ir<%in>, ir<%in>
;
; SVE-ARMPL-LABEL: 'frem_f64'
-; SVE-ARMPL: LV: Found an estimated cost of 10 for VF 1 For instruction: %res = frem double %in, %in
+; SVE-ARMPL: Cost of 10 for VF 1: EMIT ir<%res> = frem ir<%in>, ir<%in>
; SVE-ARMPL: Cost of 10 for VF 2: WIDEN ir<%res> = frem ir<%in>, ir<%in>
; SVE-ARMPL: Cost of Invalid for VF vscale x 1: WIDEN ir<%res> = frem ir<%in>, ir<%in>
; SVE-ARMPL: Cost of 10 for VF vscale x 2: WIDEN ir<%res> = frem ir<%in>, ir<%in>
;
; SVE-SLEEF-LABEL: 'frem_f64'
-; SVE-SLEEF: LV: Found an estimated cost of 10 for VF 1 For instruction: %res = frem double %in, %in
+; SVE-SLEEF: Cost of 10 for VF 1: EMIT ir<%res> = frem ir<%in>, ir<%in>
; SVE-SLEEF: Cost of 10 for VF 2: WIDEN ir<%res> = frem ir<%in>, ir<%in>
; SVE-SLEEF: Cost of Invalid for VF vscale x 1: WIDEN ir<%res> = frem ir<%in>, ir<%in>
; SVE-SLEEF: Cost of 10 for VF vscale x 2: WIDEN ir<%res> = frem ir<%in>, ir<%in>
;
; SVE-ARMPL-TAILFOLD-LABEL: 'frem_f64'
-; SVE-ARMPL-TAILFOLD: LV: Found an estimated cost of 10 for VF 1 For instruction: %res = frem double %in, %in
+; SVE-ARMPL-TAILFOLD: Cost of 10 for VF 1: EMIT ir<%res> = frem ir<%in>, ir<%in>
; SVE-ARMPL-TAILFOLD: Cost of 10 for VF 2: WIDEN ir<%res> = frem ir<%in>, ir<%in>
; SVE-ARMPL-TAILFOLD: Cost of Invalid for VF vscale x 1: WIDEN ir<%res> = frem ir<%in>, ir<%in>
; SVE-ARMPL-TAILFOLD: Cost of 10 for VF vscale x 2: WIDEN ir<%res> = frem ir<%in>, ir<%in>
;
; SVE-SLEEF-TAILFOLD-LABEL: 'frem_f64'
-; SVE-SLEEF-TAILFOLD: LV: Found an estimated cost of 10 for VF 1 For instruction: %res = frem double %in, %in
+; SVE-SLEEF-TAILFOLD: Cost of 10 for VF 1: EMIT ir<%res> = frem ir<%in>, ir<%in>
; SVE-SLEEF-TAILFOLD: Cost of 10 for VF 2: WIDEN ir<%res> = frem ir<%in>, ir<%in>
; SVE-SLEEF-TAILFOLD: Cost of Invalid for VF vscale x 1: WIDEN ir<%res> = frem ir<%in>, ir<%in>
; SVE-SLEEF-TAILFOLD: Cost of 10 for VF vscale x 2: WIDEN ir<%res> = frem ir<%in>, ir<%in>
@@ -73,12 +73,12 @@ define void @frem_f64(ptr noalias %in.ptr, ptr noalias %out.ptr) {
define void @frem_f32(ptr noalias %in.ptr, ptr noalias %out.ptr) {
; NEON-NO-VECLIB-LABEL: 'frem_f32'
-; NEON-NO-VECLIB: LV: Found an estimated cost of 10 for VF 1 For instruction: %res = frem float %in, %in
+; NEON-NO-VECLIB: Cost of 10 for VF 1: EMIT ir<%res> = frem ir<%in>, ir<%in>
; NEON-NO-VECLIB: Cost of 24 for VF 2: WIDEN ir<%res> = frem ir<%in>, ir<%in>
; NEON-NO-VECLIB: Cost of 52 for VF 4: WIDEN ir<%res> = frem ir<%in>, ir<%in>
;
; SVE-NO-VECLIB-LABEL: 'frem_f32'
-; SVE-NO-VECLIB: LV: Found an estimated cost of 10 for VF 1 For instruction: %res = frem float %in, %in
+; SVE-NO-VECLIB: Cost of 10 for VF 1: EMIT ir<%res> = frem ir<%in>, ir<%in>
; SVE-NO-VECLIB: Cost of 24 for VF 2: WIDEN ir<%res> = frem ir<%in>, ir<%in>
; SVE-NO-VECLIB: Cost of 52 for VF 4: WIDEN ir<%res> = frem ir<%in>, ir<%in>
; SVE-NO-VECLIB: Cost of Invalid for VF vscale x 1: WIDEN ir<%res> = frem ir<%in>, ir<%in>
@@ -86,17 +86,17 @@ define void @frem_f32(ptr noalias %in.ptr, ptr noalias %out.ptr) {
; SVE-NO-VECLIB: Cost of Invalid for VF vscale x 4: WIDEN ir<%res> = frem ir<%in>, ir<%in>
;
; NEON-ARMPL-LABEL: 'frem_f32'
-; NEON-ARMPL: LV: Found an estimated cost of 10 for VF 1 For instruction: %res = frem float %in, %in
+; NEON-ARMPL: Cost of 10 for VF 1: EMIT ir<%res> = frem ir<%in>, ir<%in>
; NEON-ARMPL: Cost of 24 for VF 2: WIDEN ir<%res> = frem ir<%in>, ir<%in>
; NEON-ARMPL: Cost of 10 for VF 4: WIDEN ir<%res> = frem ir<%in>, ir<%in>
;
; NEON-SLEEF-LABEL: 'frem_f32'
-; NEON-SLEEF: LV: Found an estimated cost of 10 for VF 1 For instruction: %res = frem float %in, %in
+; NEON-SLEEF: Cost of 10 for VF 1: EMIT ir<%res> = frem ir<%in>, ir<%in>
; NEON-SLEEF: Cost of 24 for VF 2: WIDEN ir<%res> = frem ir<%in>, ir<%in>
; NEON-SLEEF: Cost of 10 for VF 4: WIDEN ir<%res> = frem ir<%in>, ir<%in>
;
; SVE-ARMPL-LABEL: 'frem_f32'
-; SVE-ARMPL: LV: Found an estimated cost of 10 for VF 1 For instruction: %res = frem float %in, %in
+; SVE-ARMPL: Cost of 10 for VF 1: EMIT ir<%res> = frem ir<%in>, ir<%in>
; SVE-ARMPL: Cost of 24 for VF 2: WIDEN ir<%res> = frem ir<%in>, ir<%in>
; SVE-ARMPL: Cost of 10 for VF 4: WIDEN ir<%res> = frem ir<%in>, ir<%in>
; SVE-ARMPL: Cost of Invalid for VF vscale x 1: WIDEN ir<%res> = frem ir<%in>, ir<%in>
@@ -104,7 +104,7 @@ define void @frem_f32(ptr noalias %in.ptr, ptr noalias %out.ptr) {
; SVE-ARMPL: Cost of 10 for VF vscale x 4: WIDEN ir<%res> = frem ir<%in>, ir<%in>
;
; SVE-SLEEF-LABEL: 'frem_f32'
-; SVE-SLEEF: LV: Found an estimated cost of 10 for VF 1 For instruction: %res = frem float %in, %in
+; SVE-SLEEF: Cost of 10 for VF 1: EMIT ir<%res> = frem ir<%in>, ir<%in>
; SVE-SLEEF: Cost of 24 for VF 2: WIDEN ir<%res> = frem ir<%in>, ir<%in>
; SVE-SLEEF: Cost of 10 for VF 4: WIDEN ir<%res> = frem ir<%in>, ir<%in>
; SVE-SLEEF: Cost of Invalid for VF vscale x 1: WIDEN ir<%res> = frem ir<%in>, ir<%in>
@@ -112,7 +112,7 @@ define void @frem_f32(ptr noalias %in.ptr, ptr noalias %out.ptr) {
; SVE-SLEEF: Cost of 10 for VF vscale x 4: WIDEN ir<%res> = frem ir<%in>, ir<%in>
;
; SVE-ARMPL-TAILFOLD-LABEL: 'frem_f32'
-; SVE-ARMPL-TAILFOLD: LV: Found an estimated cost of 10 for VF 1 For instruction: %res = frem float %in, %in
+; SVE-ARMPL-TAILFOLD: Cost of 10 for VF 1: EMIT ir<%res> = frem ir<%in>, ir<%in>
; SVE-ARMPL-TAILFOLD: Cost of 24 for VF 2: WIDEN ir<%res> = frem ir<%in>, ir<%in>
; SVE-ARMPL-TAILFOLD: Cost of 10 for VF 4: WIDEN ir<%res> = frem ir<%in>, ir<%in>
; SVE-ARMPL-TAILFOLD: Cost of Invalid for VF vscale x 1: WIDEN ir<%res> = frem ir<%in>, ir<%in>
@@ -120,7 +120,7 @@ define void @frem_f32(ptr noalias %in.ptr, ptr noalias %out.ptr) {
; SVE-ARMPL-TAILFOLD: Cost of 10 for VF vscale x 4: WIDEN ir<%res> = frem ir<%in>, ir<%in>
;
; SVE-SLEEF-TAILFOLD-LABEL: 'frem_f32'
-; SVE-SLEEF-TAILFOLD: LV: Found an estimated cost of 10 for VF 1 For instruction: %res = frem float %in, %in
+; SVE-SLEEF-TAILFOLD: Cost of 10 for VF 1: EMIT ir<%res> = frem ir<%in>, ir<%in>
; SVE-SLEEF-TAILFOLD: Cost of 24 for VF 2: WIDEN ir<%res> = frem ir<%in>, ir<%in>
; SVE-SLEEF-TAILFOLD: Cost of 10 for VF 4: WIDEN ir<%res> = frem ir<%in>, ir<%in>
; SVE-SLEEF-TAILFOLD: Cost of Invalid for VF vscale x 1: WIDEN ir<%res> = frem ir<%in>, ir<%in>
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/intrinsiccost.ll b/llvm/test/Transforms/LoopVectorize/AArch64/intrinsiccost.ll
index 07d6caa861f9da..958e7763aad3d6 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/intrinsiccost.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/intrinsiccost.ll
@@ -7,10 +7,6 @@ target datalayout = "e-m:e-i8:8:32-i16:16:32-i64:64-i128:128-n32:64-S128"
target triple = "aarch64--linux-gnu"
; CHECK-COST-LABEL: sadd
-; CHECK-COST: Found an estimated cost of 6 for VF 1 For instruction: %1 = tail call i16 @llvm.sadd.sat.i16(i16 %0, i16 %offset)
-; CHECK-COST: Cost of 4 for VF 2: WIDEN-INTRINSIC ir<%1> = call llvm.sadd.sat(ir<%0>, ir<%offset>)
-; CHECK-COST: Cost of 1 for VF 4: WIDEN-INTRINSIC ir<%1> = call llvm.sadd.sat(ir<%0>, ir<%offset>)
-; CHECK-COST: Cost of 1 for VF 8: WIDEN-INTRINSIC ir<%1> = call llvm.sadd.sat(ir<%0>, ir<%offset>)
define void @saddsat(ptr nocapture readonly %pSrc, i16 signext %offset, ptr nocapture noalias %pDst, i32 %blockSize) #0 {
; CHECK-LABEL: @saddsat(
@@ -127,11 +123,6 @@ while.end:
}
; CHECK-COST-LABEL: umin
-; CHECK-COST: Found an estimated cost of 2 for VF 1 For instruction: %1 = tail call i8 @llvm.umin.i8(i8 %0, i8 %offset)
-; CHECK-COST: Cost of 3 for VF 2: WIDEN-INTRINSIC ir<%1> = call llvm.umin(ir<%0>, ir<%offset>)
-; CHECK-COST: Cost of 3 for VF 4: WIDEN-INTRINSIC ir<%1> = call llvm.umin(ir<%0>, ir<%offset>)
-; CHECK-COST: Cost of 1 for VF 8: WIDEN-INTRINSIC ir<%1> = call llvm.umin(ir<%0>, ir<%offset>)
-; CHECK-COST: Cost of 1 for VF 16: WIDEN-INTRINSIC ir<%1> = call llvm.umin(ir<%0>, ir<%offset>)
define void @umin(ptr nocapture readonly %pSrc, i8 signext %offset, ptr nocapture noalias %pDst, i32 %blockSize) #0 {
; CHECK-LABEL: @umin(
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/multiple-result-intrinsics.ll b/llvm/test/Transforms/LoopVectorize/AArch64/multiple-result-intrinsics.ll
index 55994ad9a98f8d..2a7c73bfa8ed30 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/multiple-result-intrinsics.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/multiple-result-intrinsics.ll
@@ -6,7 +6,7 @@
; REQUIRES: asserts
; CHECK-COST-LABEL: sincos_f32
-; CHECK-COST: LV: Found an estimated cost of 10 for VF 1 For instruction: %call = tail call { float, float } @llvm.sincos.f32(float %in_val)
+; CHECK-COST: Cost of 10 for VF 1: EMIT ir<%call> = call ir<%in_val>, ir<@llvm.sincos.f32>
; CHECK-COST: Cost of 26 for VF 2: WIDEN-INTRINSIC ir<%call> = call llvm.sincos(ir<%in_val>)
; CHECK-COST: Cost of 58 for VF 4: WIDEN-INTRINSIC ir<%call> = call llvm.sincos(ir<%in_val>)
; CHECK-COST: Cost of Invalid for VF vscale x 1: REPLICATE ir<%call> = call @llvm.sincos.f32(ir<%in_val>)
@@ -14,7 +14,7 @@
; CHECK-COST: Cost of Invalid for VF vscale x 4: REPLICATE ir<%call> = call @llvm.sincos.f32(ir<%in_val>)
; CHECK-COST-ARMPL-LABEL: sincos_f32
-; CHECK-COST-ARMPL: LV: Found an estimated cost of 10 for VF 1 For instruction: %call = tail call { float, float } @llvm.sincos.f32(float %in_val)
+; CHECK-COST-ARMPL: Cost of 10 for VF 1: EMIT ir<%call> = call ir<%in_val>, ir<@llvm.sincos.f32>
; CHECK-COST-ARMPL: Cost of 26 for VF 2: WIDEN-INTRINSIC ir<%call> = call llvm.sincos(ir<%in_val>)
; CHECK-COST-ARMPL: Cost of 12 for VF 4: WIDEN-INTRINSIC ir<%call> = call llvm.sincos(ir<%in_val>)
; CHECK-COST-ARMPL: Cost of Invalid for VF vscale x 1: REPLICATE ir<%call> = call @llvm.sincos.f32(ir<%in_val>)
@@ -83,13 +83,13 @@ exit:
}
; CHECK-COST-LABEL: sincos_f64
-; CHECK-COST: LV: Found an estimated cost of 10 for VF 1 For instruction: %call = tail call { double, double } @llvm.sincos.f64(double %in_val)
+; CHECK-COST: Cost of 10 for VF 1: EMIT ir<%call> = call ir<%in_val>, ir<@llvm.sincos.f64>
; CHECK-COST: Cost of 26 for VF 2: WIDEN-INTRINSIC ir<%call> = call llvm.sincos(ir<%in_val>)
; CHECK-COST: Cost of Invalid for VF vscale x 1: REPLICATE ir<%call> = call @llvm.sincos.f64(ir<%in_val>)
; CHECK-COST: Cost of Invalid for VF vscale x 2: REPLICATE ir<%call> = call @llvm.sincos.f64(ir<%in_val>)
; CHECK-COST-ARMPL-LABEL: sincos_f64
-; CHECK-COST-ARMPL: LV: Found an estimated cost of 10 for VF 1 For instruction: %call = tail call { double, double } @llvm.sincos.f64(double %in_val)
+; CHECK-COST-ARMPL: Cost of 10 for VF 1: EMIT ir<%call> = call ir<%in_val>, ir<@llvm.sincos.f64>
; CHECK-COST-ARMPL: Cost of 12 for VF 2: WIDEN-INTRINSIC ir<%call> = call llvm.sincos(ir<%in_val>)
; CHECK-COST-ARMPL: Cost of Invalid for VF vscale x 1: REPLICATE ir<%call> = call @llvm.sincos.f64(ir<%in_val>)
; CHECK-COST-ARMPL: Cost of 13 for VF vscale x 2: WIDEN-INTRINSIC ir<%call> = call llvm.sincos(ir<%in_val>)
@@ -156,7 +156,7 @@ exit:
}
; CHECK-COST-LABEL: predicated_sincos
-; CHECK-COST: LV: Found an estimated cost of 10 for VF 1 For instruction: %call = tail call { float, float } @llvm.sincos.f32(float %in_val)
+; CHECK-COST: Cost of 10 for VF 1: EMIT ir<%call> = call ir<%in_val>, ir<@llvm.sincos.f32>
; CHECK-COST: Cost of 26 for VF 2: WIDEN-INTRINSIC ir<%call> = call llvm.sincos(ir<%in_val>)
; CHECK-COST: Cost of 58 for VF 4: WIDEN-INTRINSIC ir<%call> = call llvm.sincos(ir<%in_val>)
; CHECK-COST: Cost of Invalid for VF vscale x 1: REPLICATE ir<%call> = call @llvm.sincos.f32(ir<%in_val>)
@@ -164,7 +164,7 @@ exit:
; CHECK-COST: Cost of Invalid for VF vscale x 4: REPLICATE ir<%call> = call @llvm.sincos.f32(ir<%in_val>)
; CHECK-COST-ARMPL-LABEL: predicated_sincos
-; CHECK-COST-ARMPL: LV: Found an estimated cost of 10 for VF 1 For instruction: %call = tail call { float, float } @llvm.sincos.f32(float %in_val)
+; CHECK-COST-ARMPL: Cost of 10 for VF 1: EMIT ir<%call> = call ir<%in_val>, ir<@llvm.sincos.f32>
; CHECK-COST-ARMPL: Cost of 26 for VF 2: WIDEN-INTRINSIC ir<%call> = call llvm.sincos(ir<%in_val>)
; CHECK-COST-ARMPL: Cost of 12 for VF 4: WIDEN-INTRINSIC ir<%call> = call llvm.sincos(ir<%in_val>)
; CHECK-COST-ARMPL: Cost of Invalid for VF vscale x 1: REPLICATE ir<%call> = call @llvm.sincos.f32(ir<%in_val>)
@@ -228,7 +228,7 @@ for.end:
}
; CHECK-COST-LABEL: modf_f32
-; CHECK-COST: LV: Found an estimated cost of 10 for VF 1 For instruction: %call = tail call { float, float } @llvm.modf.f32(float %in_val)
+; CHECK-COST: Cost of 10 for VF 1: EMIT ir<%call> = call ir<%in_val>, ir<@llvm.modf.f32>
; CHECK-COST: Cost of 26 for VF 2: WIDEN-INTRINSIC ir<%call> = call llvm.modf(ir<%in_val>)
; CHECK-COST: Cost of 58 for VF 4: WIDEN-INTRINSIC ir<%call> = call llvm.modf(ir<%in_val>)
; CHECK-COST: Cost of Invalid for VF vscale x 1: REPLICATE ir<%call> = call @llvm.modf.f32(ir<%in_val>)
@@ -236,7 +236,7 @@ for.end:
; CHECK-COST: Cost of Invalid for VF vscale x 4: REPLICATE ir<%call> = call @llvm.modf.f32(ir<%in_val>)
; CHECK-COST-ARMPL-LABEL: modf_f32
-; CHECK-COST-ARMPL: LV: Found an estimated cost of 10 for VF 1 For instruction: %call = tail call { float, float } @llvm.modf.f32(float %in_val)
+; CHECK-COST-ARMPL: Cost of 10 for VF 1: EMIT ir<%call> = call ir<%in_val>, ir<@llvm.modf.f32>
; CHECK-COST-ARMPL: Cost of 26 for VF 2: WIDEN-INTRINSIC ir<%call> = call llvm.modf(ir<%in_val>)
; CHECK-COST-ARMPL: Cost of 11 for VF 4: WIDEN-INTRINSIC ir<%call> = call llvm.modf(ir<%in_val>)
; CHECK-COST-ARMPL: Cost of Invalid for VF vscale x 1: REPLICATE ir<%call> = call @llvm.modf.f32(ir<%in_val>)
@@ -305,13 +305,13 @@ exit:
}
; CHECK-COST-LABEL: modf_f64
-; CHECK-COST: LV: Found an estimated cost of 10 for VF 1 For instruction: %call = tail call { double, double } @llvm.modf.f64(double %in_val)
+; CHECK-COST: Cost of 10 for VF 1: EMIT ir<%call> = call ir<%in_val>, ir<@llvm.modf.f64>
; CHECK-COST: Cost of 26 for VF 2: WIDEN-INTRINSIC ir<%call> = call llvm.modf(ir<%in_val>)
; CHECK-COST: Cost of Invalid for VF vscale x 1: REPLICATE ir<%call> = call @llvm.modf.f64(ir<%in_val>)
; CHECK-COST: Cost of Invalid for VF vscale x 2: REPLICATE ir<%call> = call @llvm.modf.f64(ir<%in_val>)
; CHECK-COST-ARMPL-LABEL: modf_f64
-; CHECK-COST-ARMPL: LV: Found an estimated cost of 10 for VF 1 For instruction: %call = tail call { double, double } @llvm.modf.f64(double %in_val)
+; CHECK-COST-ARMPL: Cost of 10 for VF 1: EMIT ir<%call> = call ir<%in_val>, ir<@llvm.modf.f64>
; CHECK-COST-ARMPL: Cost of 11 for VF 2: WIDEN-INTRINSIC ir<%call> = call llvm.modf(ir<%in_val>)
; CHECK-COST-ARMPL: Cost of Invalid for VF vscale x 1: REPLICATE ir<%call> = call @llvm.modf.f64(ir<%in_val>)
; CHECK-COST-ARMPL: Cost of 12 for VF vscale x 2: WIDEN-INTRINSIC ir<%call> = call llvm.modf(ir<%in_val>)
@@ -378,7 +378,7 @@ exit:
}
; CHECK-COST-LABEL: sincospi_f32
-; CHECK-COST: LV: Found an estimated cost of 10 for VF 1 For instruction: %call = tail call { float, float } @llvm.sincospi.f32(float %in_val)
+; CHECK-COST: Cost of 10 for VF 1: EMIT ir<%call> = call ir<%in_val>, ir<@llvm.sincospi.f32>
; CHECK-COST: Cost of 26 for VF 2: WIDEN-INTRINSIC ir<%call> = call llvm.sincospi(ir<%in_val>)
; CHECK-COST: Cost of 58 for VF 4: WIDEN-INTRINSIC ir<%call> = call llvm.sincospi(ir<%in_val>)
; CHECK-COST: Cost of Invalid for VF vscale x 1: REPLICATE ir<%call> = call @llvm.sincospi.f32(ir<%in_val>)
@@ -386,7 +386,7 @@ exit:
; CHECK-COST: Cost of Invalid for VF vscale x 4: REPLICATE ir<%call> = call @llvm.sincospi.f32(ir<%in_val>)
; CHECK-COST-ARMPL-LABEL: sincospi_f32
-; CHECK-COST-ARMPL: LV: Found an estimated cost of 10 for VF 1 For instruction: %call = tail call { float, float } @llvm.sincospi.f32(float %in_val)
+; CHECK-COST-ARMPL: Cost of 10 for VF 1: EMIT ir<%call> = call ir<%in_val>, ir<@llvm.sincospi.f32>
; CHECK-COST-ARMPL: Cost of 26 for VF 2: WIDEN-INTRINSIC ir<%call> = call llvm.sincospi(ir<%in_val>)
; CHECK-COST-ARMPL: Cost of 12 for VF 4: WIDEN-INTRINSIC ir<%call> = call llvm.sincospi(ir<%in_val>)
; CHECK-COST-ARMPL: Cost of Invalid for VF vscale x 1: REPLICATE ir<%call> = call @llvm.sincospi.f32(ir<%in_val>)
@@ -455,13 +455,13 @@ exit:
}
; CHECK-COST-LABEL: sincospi_f64
-; CHECK-COST: LV: Found an estimated cost of 10 for VF 1 For instruction: %call = tail call { double, double } @llvm.sincospi.f64(double %in_val)
+; CHECK-COST: Cost of 10 for VF 1: EMIT ir<%call> = call ir<%in_val>, ir<@llvm.sincospi.f64>
; CHECK-COST: Cost of 26 for VF 2: WIDEN-INTRINSIC ir<%call> = call llvm.sincospi(ir<%in_val>)
; CHECK-COST: Cost of Invalid for VF vscale x 1: REPLICATE ir<%call> = call @llvm.sincospi.f64(ir<%in_val>)
; CHECK-COST: Cost of Invalid for VF vscale x 2: REPLICATE ir<%call> = call @llvm.sincospi.f64(ir<%in_val>)
; CHECK-COST-ARMPL-LABEL: sincospi_f64
-; CHECK-COST-ARMPL: LV: Found an estimated cost of 10 for VF 1 For instruction: %call = tail call { double, double } @llvm.sincospi.f64(double %in_val)
+; CHECK-COST-ARMPL: Cost of 10 for VF 1: EMIT ir<%call> = call ir<%in_val>, ir<@llvm.sincospi.f64>
; CHECK-COST-ARMPL: Cost of 12 for VF 2: WIDEN-INTRINSIC ir<%call> = call llvm.sincospi(ir<%in_val>)
; CHECK-COST-ARMPL: Cost of Invalid for VF vscale x 1: REPLICATE ir<%call> = call @llvm.sincospi.f64(ir<%in_val>)
; CHECK-COST-ARMPL: Cost of 13 for VF vscale x 2: WIDEN-INTRINSIC ir<%call> = call llvm.sincospi(ir<%in_val>)
@@ -528,7 +528,7 @@ exit:
}
; CHECK-COST-LABEL: sadd_with_overflow_i32
-; CHECK-COST: LV: Found an estimated cost of 1 for VF 1 For instruction: %call = tail call { i32, i1 } @llvm.sadd.with.overflow.i32(i32 %val_a, i32 %val_b)
+; CHECK-COST: Cost of 1 for VF 1: EMIT ir<%call> = call ir<%val_a>, ir<%val_b>, ir<@llvm.sadd.with.overflow.i32>
; CHECK-COST: Cost of 4 for VF 2: WIDEN-INTRINSIC ir<%call> = call llvm.sadd.with.overflow(ir<%val_a>, ir<%val_b>)
; CHECK-COST: Cost of 4 for VF 4: WIDEN-INTRINSIC ir<%call> = call llvm.sadd.with.overflow(ir<%val_a>, ir<%val_b>)
; CHECK-COST: Cost of 7 for VF 8: WIDEN-INTRINSIC ir<%call> = call llvm.sadd.with.overflow(ir<%val_a>, ir<%val_b>)
@@ -538,7 +538,7 @@ exit:
; CHECK-COST: Cost of 4 for VF vscale x 4: WIDEN-INTRINSIC ir<%call> = call llvm.sadd.with.overflow(ir<%val_a>, ir<%val_b>)
; CHECK-COST-ARMPL-LABEL: sadd_with_overflow_i32
-; CHECK-COST-ARMPL: LV: Found an estimated cost of 1 for VF 1 For instruction: %call = tail call { i32, i1 } @llvm.sadd.with.overflow.i32(i32 %val_a, i32 %val_b)
+; CHECK-COST-ARMPL: Cost of 1 for VF 1: EMIT ir<%call> = call ir<%val_a>, ir<%val_b>, ir<@llvm.sadd.with.overflow.i32>
; CHECK-COST-ARMPL: Cost of 4 for VF 2: WIDEN-INTRINSIC ir<%call> = call llvm.sadd.with.overflow(ir<%val_a>, ir<%val_b>)
; CHECK-COST-ARMPL: Cost of 4 for VF 4: WIDEN-INTRINSIC ir<%call> = call llvm.sadd.with.overflow(ir<%val_a>, ir<%val_b>)
; CHECK-COST-ARMPL: Cost of 7 for VF 8: WIDEN-INTRINSIC ir<%call> = call llvm.sadd.with.overflow(ir<%val_a>, ir<%val_b>)
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/struct-return-cost.ll b/llvm/test/Transforms/LoopVectorize/AArch64/struct-return-cost.ll
index e2e69eb4ca147f..96cd2887c4bd4e 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/struct-return-cost.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/struct-return-cost.ll
@@ -7,9 +7,9 @@ target datalayout = "e-m:e-i8:8:32-i16:16:32-i64:64-i128:128-n32:64-S128"
target triple = "aarch64--linux-gnu"
; CHECK-COST-LABEL: struct_return_widen
-; CHECK-COST: LV: Found an estimated cost of 10 for VF 1 For instruction: %call = tail call { half, half } @foo(half %in_val)
-; CHECK-COST: LV: Found an estimated cost of 0 for VF 1 For instruction: %extract_a = extractvalue { half, half } %call, 0
-; CHECK-COST: LV: Found an estimated cost of 0 for VF 1 For instruction: %extract_b = extractvalue { half, half } %call, 1
+; CHECK-COST: Cost of 10 for VF 1: EMIT ir<%call> = call ir<%in_val>, ir<@foo>
+; CHECK-COST: Cost of 0 for VF 1: EMIT ir<%extract_a> = extractvalue ir<%call>
+; CHECK-COST: Cost of 0 for VF 1: EMIT ir<%extract_b> = extractvalue ir<%call>
;
; CHECK-COST: Cost of 10 for VF 2: WIDEN-CALL ir<%call> = call @foo(ir<%in_val>) (using library function: fixed_vec_foo)
; CHECK-COST: Cost of 0 for VF 2: WIDEN ir<%extract_a> = extractvalue ir<%call>, ir<0>
@@ -57,9 +57,9 @@ exit:
}
; CHECK-COST-LABEL: struct_return_replicate
-; CHECK-COST: LV: Found an estimated cost of 10 for VF 1 For instruction: %call = tail call { half, half } @foo(half %in_val)
-; CHECK-COST: LV: Found an estimated cost of 0 for VF 1 For instruction: %extract_a = extractvalue { half, half } %call, 0
-; CHECK-COST: LV: Found an estimated cost of 0 for VF 1 For instruction: %extract_b = extractvalue { half, half } %call, 1
+; CHECK-COST: Cost of 10 for VF 1: EMIT ir<%call> = call ir<%in_val>, ir<@foo>
+; CHECK-COST: Cost of 0 for VF 1: EMIT ir<%extract_a> = extractvalue ir<%call>
+; CHECK-COST: Cost of 0 for VF 1: EMIT ir<%extract_b> = extractvalue ir<%call>
;
; CHECK-COST: Cost of 26 for VF 2: REPLICATE ir<%call> = call @foo(ir<%in_val>)
; CHECK-COST: Cost of 0 for VF 2: WIDEN ir<%extract_a> = extractvalue ir<%call>, ir<0>
@@ -79,8 +79,8 @@ define void @struct_return_replicate(ptr noalias %in, ptr noalias writeonly %out
; CHECK: [[ENTRY:.*:]]
; CHECK: [[VECTOR_PH:.*:]]
; CHECK: [[VECTOR_BODY:.*:]]
-; CHECK: [[TMP3:%.*]] = tail call { half, half } @foo(half [[TMP1:%.*]]) #[[ATTR2:[0-9]+]]
-; CHECK: [[TMP4:%.*]] = tail call { half, half } @foo(half [[TMP2:%.*]]) #[[ATTR2]]
+; CHECK: [[TMP2:%.*]] = tail call { half, half } @foo(half [[TMP1:%.*]]) #[[ATTR2:[0-9]+]]
+; CHECK: [[TMP4:%.*]] = tail call { half, half } @foo(half [[TMP3:%.*]]) #[[ATTR2]]
; CHECK: [[MIDDLE_BLOCK:.*:]]
; CHECK: [[EXIT:.*:]]
;
@@ -108,9 +108,9 @@ exit:
}
; CHECK-COST-LABEL: struct_return_scalable
-; CHECK-COST: LV: Found an estimated cost of 10 for VF 1 For instruction: %call = tail call { half, half } @foo(half %in_val)
-; CHECK-COST: LV: Found an estimated cost of 0 for VF 1 For instruction: %extract_a = extractvalue { half, half } %call, 0
-; CHECK-COST: LV: Found an estimated cost of 0 for VF 1 For instruction: %extract_b = extractvalue { half, half } %call, 1
+; CHECK-COST: Cost of 10 for VF 1: EMIT ir<%call> = call ir<%in_val>, ir<@foo>
+; CHECK-COST: Cost of 0 for VF 1: EMIT ir<%extract_a> = extractvalue ir<%call>
+; CHECK-COST: Cost of 0 for VF 1: EMIT ir<%extract_b> = extractvalue ir<%call>
;
; CHECK-COST: Cost of 26 for VF 2: REPLICATE ir<%call> = call @foo(ir<%in_val>)
; CHECK-COST: Cost of 0 for VF 2: WIDEN ir<%extract_a> = extractvalue ir<%call>, ir<0>
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/type-shrinkage-zext-costs.ll b/llvm/test/Transforms/LoopVectorize/AArch64/type-shrinkage-zext-costs.ll
index 09c262b1adb482..11ec71fea57f9e 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/type-shrinkage-zext-costs.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/type-shrinkage-zext-costs.ll
@@ -8,7 +8,7 @@ target triple = "aarch64-unknown-linux-gnu"
define void @zext_i8_i16(ptr noalias nocapture readonly %p, ptr noalias nocapture %q, i32 %len) #0 {
; CHECK-COST-LABEL: LV: Checking a loop in 'zext_i8_i16'
-; CHECK-COST: LV: Found an estimated cost of 0 for VF 1 For instruction: %conv = zext i8 %0 to i32
+; CHECK-COST: Cost of 0 for VF 1: EMIT-SCALAR ir<%conv> = zext ir<%0> to i32
; CHECK-COST: Cost of 1 for VF 2: WIDEN-CAST ir<%conv> = zext ir<%0> to i16
; CHECK-COST: Cost of 1 for VF 4: WIDEN-CAST ir<%conv> = zext ir<%0> to i16
; CHECK-COST: Cost of 1 for VF 8: WIDEN-CAST ir<%conv> = zext ir<%0> to i16
diff --git a/llvm/test/Transforms/LoopVectorize/ARM/mve-icmpcost.ll b/llvm/test/Transforms/LoopVectorize/ARM/mve-icmpcost.ll
index 7369ef688f5e25..aa6ec753d92764 100644
--- a/llvm/test/Transforms/LoopVectorize/ARM/mve-icmpcost.ll
+++ b/llvm/test/Transforms/LoopVectorize/ARM/mve-icmpcost.ll
@@ -7,29 +7,28 @@ target triple = "thumbv8.1m.main-arm-none-eabi"
define void @expensive_icmp(ptr noalias nocapture %d, ptr nocapture readonly %s, i32 %n, i16 zeroext %m) #0 {
; CHECK-LABEL: 'expensive_icmp'
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction: %i.016 = phi i32 [ 0, %for.body.lr.ph ], [ %inc, %for.inc ]
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction: %arrayidx = getelementptr inbounds i16, ptr %s, i32 %i.016
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %1 = load i16, ptr %arrayidx, align 2
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction: %conv = sext i16 %1 to i32
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %cmp2 = icmp sgt i32 %conv, %conv1
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction: br i1 %cmp2, label %if.then, label %for.inc
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %conv6 = add i16 %1, %0
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction: %arrayidx7 = getelementptr inbounds i16, ptr %d, i32 %i.016
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: store i16 %conv6, ptr %arrayidx7, align 2
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction: br label %for.inc
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %inc = add nuw nsw i32 %i.016, 1
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %exitcond.not = icmp eq i32 %inc, %n
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction: br i1 %exitcond.not, label %for.cond.cleanup.loopexit, label %for.body
+; CHECK: Cost of 0 for VF 1: EMIT-SCALAR ir<%i.016> = phi [ ir<0>, vector.ph ], [ ir<%inc>, for.inc ]
+; CHECK: Cost of 0 for VF 1: EMIT ir<%arrayidx> = getelementptr inbounds ir<%s>, ir<%i.016>
+; CHECK: Cost of 1 for VF 1: EMIT-SCALAR ir<%1> = load ir<%arrayidx>
+; CHECK: Cost of 0 for VF 1: EMIT-SCALAR ir<%conv> = sext ir<%1> to i32
+; CHECK: Cost of 1 for VF 1: EMIT ir<%cmp2> = icmp sgt ir<%conv>, ir<%conv1>
+; CHECK: Cost of 0 for VF 1: EMIT branch-on-cond ir<%cmp2> (!vplan.prof.estimated estimated {1073741824, 1073741824})
+; CHECK: Cost of 1 for VF 1: EMIT ir<%conv6> = add ir<%1>, ir<%0> (!vplan.execution.frequency 4611686018427387904 (50%, estimated))
+; CHECK: Cost of 0 for VF 1: EMIT ir<%arrayidx7> = getelementptr inbounds ir<%d>, ir<%i.016> (!vplan.execution.frequency 4611686018427387904 (50%, estimated))
+; CHECK: Cost of 1 for VF 1: EMIT store ir<%conv6>, ir<%arrayidx7> (!vplan.execution.frequency 4611686018427387904 (50%, estimated))
+; CHECK: Cost of 1 for VF 1: EMIT ir<%inc> = add nuw nsw ir<%i.016>, ir<1>
+; CHECK: Cost of 1 for VF 1: EMIT ir<%exitcond.not> = icmp eq ir<%inc>, ir<%n>
+; CHECK: Cost of 0 for VF 1: EMIT branch-on-cond ir<%exitcond.not>
; CHECK: Cost of 0 for VF 2: vp<[[VP4:%[0-9]+]]> = SCALAR-STEPS vp<[[VP3:%[0-9]+]]>, ir<1>, vp<[[VP0:%[0-9]+]]>
; CHECK: Cost of 0 for VF 2: CLONE ir<%arrayidx> = getelementptr inbounds ir<%s>, vp<[[VP4]]>
; CHECK: Cost of 0 for VF 2: vp<[[VP5:%[0-9]+]]> = vector-pointer inbounds i16, ir<%arrayidx>, ir<1>
; CHECK: Cost of 18 for VF 2: WIDEN ir<%1> = load vp<[[VP5]]>
; CHECK: Cost of 4 for VF 2: WIDEN-CAST ir<%conv> = sext ir<%1> to i32
; CHECK: Cost of 20 for VF 2: WIDEN ir<%cmp2> = icmp sgt ir<%conv>, ir<%conv1>
-; CHECK: Cost of 26 for VF 2: WIDEN ir<%conv6> = add ir<%1>, ir<%0>
+; CHECK: Cost of 26 for VF 2: WIDEN ir<%conv6> = add ir<%1>, ir<%0> (!vplan.execution.frequency 4611686018427387904 (50%, estimated))
; CHECK: Cost of 0 for VF 2: CLONE ir<%arrayidx7> = getelementptr ir<%d>, vp<[[VP4]]>
; CHECK: Cost of 0 for VF 2: vp<[[VP6:%[0-9]+]]> = vector-pointer i16, ir<%arrayidx7>, ir<1>
-; CHECK: Cost of 16 for VF 2: WIDEN store vp<[[VP6]]>, ir<%conv6>, ir<%cmp2>
+; CHECK: Cost of 16 for VF 2: WIDEN store vp<[[VP6]]>, ir<%conv6>, ir<%cmp2> (!vplan.execution.frequency 4611686018427387904 (50%, estimated))
; CHECK: Cost of 0 for VF 2: EMIT vp<%index.next> = add nuw vp<[[VP3]]>, vp<[[VP1:%[0-9]+]]>
; CHECK: Cost of 1 for VF 2: EMIT branch-on-count vp<%index.next>, vp<[[VP2:%[0-9]+]]>
; CHECK: Cost of 0 for VF 2: vector loop backedge
@@ -51,10 +50,10 @@ define void @expensive_icmp(ptr noalias nocapture %d, ptr nocapture readonly %s,
; CHECK: Cost of 2 for VF 4: WIDEN ir<%1> = load vp<[[VP5]]>
; CHECK: Cost of 0 for VF 4: WIDEN-CAST ir<%conv> = sext ir<%1> to i32
; CHECK: Cost of 2 for VF 4: WIDEN ir<%cmp2> = icmp sgt ir<%conv>, ir<%conv1>
-; CHECK: Cost of 2 for VF 4: WIDEN ir<%conv6> = add ir<%1>, ir<%0>
+; CHECK: Cost of 2 for VF 4: WIDEN ir<%conv6> = add ir<%1>, ir<%0> (!vplan.execution.frequency 4611686018427387904 (50%, estimated))
; CHECK: Cost of 0 for VF 4: CLONE ir<%arrayidx7> = getelementptr ir<%d>, vp<[[VP4]]>
; CHECK: Cost of 0 for VF 4: vp<[[VP6]]> = vector-pointer i16, ir<%arrayidx7>, ir<1>
-; CHECK: Cost of 2 for VF 4: WIDEN store vp<[[VP6]]>, ir<%conv6>, ir<%cmp2>
+; CHECK: Cost of 2 for VF 4: WIDEN store vp<[[VP6]]>, ir<%conv6>, ir<%cmp2> (!vplan.execution.frequency 4611686018427387904 (50%, estimated))
; CHECK: Cost of 0 for VF 4: EMIT vp<%index.next> = add nuw vp<[[VP3]]>, vp<[[VP1]]>
; CHECK: Cost of 1 for VF 4: EMIT branch-on-count vp<%index.next>, vp<[[VP2]]>
; CHECK: Cost of 0 for VF 4: vector loop backedge
@@ -76,10 +75,10 @@ define void @expensive_icmp(ptr noalias nocapture %d, ptr nocapture readonly %s,
; CHECK: Cost of 2 for VF 8: WIDEN ir<%1> = load vp<[[VP5]]>
; CHECK: Cost of 2 for VF 8: WIDEN-CAST ir<%conv> = sext ir<%1> to i32
; CHECK: Cost of 36 for VF 8: WIDEN ir<%cmp2> = icmp sgt ir<%conv>, ir<%conv1>
-; CHECK: Cost of 2 for VF 8: WIDEN ir<%conv6> = add ir<%1>, ir<%0>
+; CHECK: Cost of 2 for VF 8: WIDEN ir<%conv6> = add ir<%1>, ir<%0> (!vplan.execution.frequency 4611686018427387904 (50%, estimated))
; CHECK: Cost of 0 for VF 8: CLONE ir<%arrayidx7> = getelementptr ir<%d>, vp<[[VP4]]>
; CHECK: Cost of 0 for VF 8: vp<[[VP6]]> = vector-pointer i16, ir<%arrayidx7>, ir<1>
-; CHECK: Cost of 2 for VF 8: WIDEN store vp<[[VP6]]>, ir<%conv6>, ir<%cmp2>
+; CHECK: Cost of 2 for VF 8: WIDEN store vp<[[VP6]]>, ir<%conv6>, ir<%cmp2> (!vplan.execution.frequency 4611686018427387904 (50%, estimated))
; CHECK: Cost of 0 for VF 8: EMIT vp<%index.next> = add nuw vp<[[VP3]]>, vp<[[VP1]]>
; CHECK: Cost of 1 for VF 8: EMIT branch-on-count vp<%index.next>, vp<[[VP2]]>
; CHECK: Cost of 0 for VF 8: vector loop backedge
@@ -133,26 +132,26 @@ for.inc:
define void @cheap_icmp(ptr nocapture readonly %pSrcA, ptr nocapture readonly %pSrcB, ptr nocapture %pDst, i32 %blockSize) #0 {
; CHECK-LABEL: 'cheap_icmp'
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction: %blkCnt.012 = phi i32 [ %dec, %while.body ], [ %blockSize, %while.body.preheader ]
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction: %pSrcA.addr.011 = phi ptr [ %incdec.ptr, %while.body ], [ %pSrcA, %while.body.preheader ]
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction: %pDst.addr.010 = phi ptr [ %incdec.ptr5, %while.body ], [ %pDst, %while.body.preheader ]
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction: %pSrcB.addr.09 = phi ptr [ %incdec.ptr2, %while.body ], [ %pSrcB, %while.body.preheader ]
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction: %incdec.ptr = getelementptr inbounds i8, ptr %pSrcA.addr.011, i32 1
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %0 = load i8, ptr %pSrcA.addr.011, align 1
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction: %conv1 = sext i8 %0 to i32
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction: %incdec.ptr2 = getelementptr inbounds i8, ptr %pSrcB.addr.09, i32 1
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %1 = load i8, ptr %pSrcB.addr.09, align 1
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction: %conv3 = sext i8 %1 to i32
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %mul = mul nsw i32 %conv3, %conv1
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %shr = ashr i32 %mul, 7
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %2 = icmp slt i32 %shr, 127
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %spec.select.i = select i1 %2, i32 %shr, i32 127
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction: %conv4 = trunc i32 %spec.select.i to i8
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction: %incdec.ptr5 = getelementptr inbounds i8, ptr %pDst.addr.010, i32 1
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: store i8 %conv4, ptr %pDst.addr.010, align 1
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %dec = add i32 %blkCnt.012, -1
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %cmp.not = icmp eq i32 %dec, 0
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction: br i1 %cmp.not, label %while.end.loopexit, label %while.body
+; CHECK: Cost of 0 for VF 1: EMIT-SCALAR ir<%blkCnt.012> = phi [ ir<%blockSize>, vector.ph ], [ ir<%dec>, while.body ]
+; CHECK: Cost of 0 for VF 1: EMIT-SCALAR ir<%pSrcA.addr.011> = phi [ ir<%pSrcA>, vector.ph ], [ ir<%incdec.ptr>, while.body ]
+; CHECK: Cost of 0 for VF 1: EMIT-SCALAR ir<%pDst.addr.010> = phi [ ir<%pDst>, vector.ph ], [ ir<%incdec.ptr5>, while.body ]
+; CHECK: Cost of 0 for VF 1: EMIT-SCALAR ir<%pSrcB.addr.09> = phi [ ir<%pSrcB>, vector.ph ], [ ir<%incdec.ptr2>, while.body ]
+; CHECK: Cost of 0 for VF 1: EMIT ir<%incdec.ptr> = getelementptr inbounds ir<%pSrcA.addr.011>, ir<1>
+; CHECK: Cost of 1 for VF 1: EMIT-SCALAR ir<%0> = load ir<%pSrcA.addr.011>
+; CHECK: Cost of 0 for VF 1: EMIT-SCALAR ir<%conv1> = sext ir<%0> to i32
+; CHECK: Cost of 0 for VF 1: EMIT ir<%incdec.ptr2> = getelementptr inbounds ir<%pSrcB.addr.09>, ir<1>
+; CHECK: Cost of 1 for VF 1: EMIT-SCALAR ir<%1> = load ir<%pSrcB.addr.09>
+; CHECK: Cost of 0 for VF 1: EMIT-SCALAR ir<%conv3> = sext ir<%1> to i32
+; CHECK: Cost of 1 for VF 1: EMIT ir<%mul> = mul nsw ir<%conv3>, ir<%conv1>
+; CHECK: Cost of 1 for VF 1: EMIT ir<%shr> = ashr ir<%mul>, ir<7>
+; CHECK: Cost of 1 for VF 1: EMIT ir<%2> = icmp slt ir<%shr>, ir<127>
+; CHECK: Cost of 1 for VF 1: EMIT ir<%spec.select.i> = select ir<%2>, ir<%shr>, ir<127>
+; CHECK: Cost of 0 for VF 1: EMIT-SCALAR ir<%conv4> = trunc ir<%spec.select.i> to i8
+; CHECK: Cost of 0 for VF 1: EMIT ir<%incdec.ptr5> = getelementptr inbounds ir<%pDst.addr.010>, ir<1>
+; CHECK: Cost of 1 for VF 1: EMIT store ir<%conv4>, ir<%pDst.addr.010>
+; CHECK: Cost of 1 for VF 1: EMIT ir<%dec> = add ir<%blkCnt.012>, ir<-1>
+; CHECK: Cost of 1 for VF 1: EMIT ir<%cmp.not> = icmp eq ir<%dec>, ir<0>
+; CHECK: Cost of 0 for VF 1: EMIT branch-on-cond ir<%cmp.not>
; CHECK: Cost of 0 for VF 2: vp<[[VP8:%[0-9]+]]> = SCALAR-STEPS vp<[[VP7:%[0-9]+]]>, ir<1>, vp<[[VP0:%[0-9]+]]>
; CHECK: Cost of 0 for VF 2: EMIT vp<%next.gep> = ptradd ir<%pSrcA>, vp<[[VP8]]>
; CHECK: Cost of 0 for VF 2: vp<[[VP9:%[0-9]+]]> = SCALAR-STEPS vp<[[VP7]]>, ir<1>, vp<[[VP0]]>
@@ -407,19 +406,19 @@ while.end:
define void @floatcmp(ptr nocapture readonly %pSrc, ptr nocapture %pDst, i32 %blockSize) #0 {
; CHECK-LABEL: 'floatcmp'
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction: %pSrc.addr.010 = phi ptr [ %incdec.ptr2, %while.body ], [ %pSrc, %while.body.preheader ]
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction: %blockSize.addr.09 = phi i32 [ %dec, %while.body ], [ %blockSize, %while.body.preheader ]
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction: %pDst.addr.08 = phi ptr [ %incdec.ptr, %while.body ], [ %pDst, %while.body.preheader ]
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %0 = load float, ptr %pSrc.addr.010, align 4
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %cmp1 = fcmp nnan ninf nsz olt float %0, 0.000000e+00
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %cond = select nnan ninf nsz i1 %cmp1, float 1.000000e+01, float %0
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %conv = fptosi float %cond to i32
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction: %incdec.ptr = getelementptr inbounds i32, ptr %pDst.addr.08, i32 1
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: store i32 %conv, ptr %pDst.addr.08, align 4
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction: %incdec.ptr2 = getelementptr inbounds float, ptr %pSrc.addr.010, i32 1
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %dec = add i32 %blockSize.addr.09, -1
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %cmp.not = icmp eq i32 %dec, 0
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction: br i1 %cmp.not, label %while.end.loopexit, label %while.body
+; CHECK: Cost of 0 for VF 1: EMIT-SCALAR ir<%pSrc.addr.010> = phi [ ir<%pSrc>, vector.ph ], [ ir<%incdec.ptr2>, while.body ]
+; CHECK: Cost of 0 for VF 1: EMIT-SCALAR ir<%blockSize.addr.09> = phi [ ir<%blockSize>, vector.ph ], [ ir<%dec>, while.body ]
+; CHECK: Cost of 0 for VF 1: EMIT-SCALAR ir<%pDst.addr.08> = phi [ ir<%pDst>, vector.ph ], [ ir<%incdec.ptr>, while.body ]
+; CHECK: Cost of 1 for VF 1: EMIT-SCALAR ir<%0> = load ir<%pSrc.addr.010>
+; CHECK: Cost of 1 for VF 1: EMIT ir<%cmp1> = fcmp olt nnan ninf nsz ir<%0>, ir<0.000000e+00>
+; CHECK: Cost of 1 for VF 1: EMIT ir<%cond> = select nnan ninf nsz ir<%cmp1>, ir<1.000000e+01>, ir<%0>
+; CHECK: Cost of 1 for VF 1: EMIT-SCALAR ir<%conv> = fptosi ir<%cond> to i32
+; CHECK: Cost of 0 for VF 1: EMIT ir<%incdec.ptr> = getelementptr inbounds ir<%pDst.addr.08>, ir<1>
+; CHECK: Cost of 1 for VF 1: EMIT store ir<%conv>, ir<%pDst.addr.08>
+; CHECK: Cost of 0 for VF 1: EMIT ir<%incdec.ptr2> = getelementptr inbounds ir<%pSrc.addr.010>, ir<1>
+; CHECK: Cost of 1 for VF 1: EMIT ir<%dec> = add ir<%blockSize.addr.09>, ir<-1>
+; CHECK: Cost of 1 for VF 1: EMIT ir<%cmp.not> = icmp eq ir<%dec>, ir<0>
+; CHECK: Cost of 0 for VF 1: EMIT branch-on-cond ir<%cmp.not>
; CHECK: Cost of 1 for VF 2: vp<[[VP7:%[0-9]+]]> = DERIVED-IV ir<0> + vp<[[VP6:%[0-9]+]]> * ir<4>
; CHECK: Cost of 0 for VF 2: vp<[[VP8:%[0-9]+]]> = SCALAR-STEPS vp<[[VP7]]>, ir<4>, vp<[[VP0:%[0-9]+]]>
; CHECK: Cost of 0 for VF 2: EMIT vp<%next.gep> = ptradd ir<%pSrc>, vp<[[VP8]]>
diff --git a/llvm/test/Transforms/LoopVectorize/ARM/mve-selectandorcost.ll b/llvm/test/Transforms/LoopVectorize/ARM/mve-selectandorcost.ll
index a814515225dc81..03b4d9a7a72c0a 100644
--- a/llvm/test/Transforms/LoopVectorize/ARM/mve-selectandorcost.ll
+++ b/llvm/test/Transforms/LoopVectorize/ARM/mve-selectandorcost.ll
@@ -7,7 +7,7 @@ target datalayout = "e-m:e-p:32:32-Fi8-i64:64-v128:64:128-a:0:32-n32-S64"
target triple = "thumbv8.1m.main-arm-none-eabi"
; CHECK-COST-LABEL: test
-; CHECK-COST: LV: Found an estimated cost of 1 for VF 1 For instruction: %or.cond = select i1 %cmp2, i1 true, i1 %cmp3
+; CHECK-COST: Cost of 1 for VF 1: EMIT ir<%or.cond> = select ir<%cmp2>, ir<true>, ir<%cmp3>
; CHECK-COST: Cost of 26 for VF 2: WIDEN ir<%or.cond> = select ir<%cmp2>, ir<true>, ir<%cmp3>
; CHECK-COST: Cost of 2 for VF 4: WIDEN ir<%or.cond> = select ir<%cmp2>, ir<true>, ir<%cmp3>
diff --git a/llvm/test/Transforms/LoopVectorize/ARM/mve-shiftcost.ll b/llvm/test/Transforms/LoopVectorize/ARM/mve-shiftcost.ll
index 22913b4dae0db1..663f54600d3f47 100644
--- a/llvm/test/Transforms/LoopVectorize/ARM/mve-shiftcost.ll
+++ b/llvm/test/Transforms/LoopVectorize/ARM/mve-shiftcost.ll
@@ -6,8 +6,8 @@ target datalayout = "e-m:e-p:32:32-Fi8-i64:64-v128:64:128-a:0:32-n32-S64"
target triple = "thumbv8.1m.main-none-none-eabi"
; CHECK-LABEL: test
-; CHECK-COST: LV: Found an estimated cost of 0 for VF 1 For instruction: %and515 = shl i32 %l41, 3
-; CHECK-COST: LV: Found an estimated cost of 1 for VF 1 For instruction: %l45 = and i32 %and515, 131072
+; CHECK-COST: Cost of 0 for VF 1: EMIT ir<%and515> = shl ir<%l41>, ir<3>
+; CHECK-COST: Cost of 1 for VF 1: EMIT ir<%l45> = and ir<%and515>, ir<131072>
; CHECK-COST: Cost of 2 for VF 4: WIDEN ir<%and515> = shl ir<%l41>, ir<3>
; CHECK-COST: Cost of 2 for VF 4: WIDEN ir<%l45> = and ir<%and515>, ir<131072>
; CHECK-NOT: vector.body
diff --git a/llvm/test/Transforms/LoopVectorize/ARM/scalar-block-cost.ll b/llvm/test/Transforms/LoopVectorize/ARM/scalar-block-cost.ll
index 8ebb04dd671dca..929503d556059e 100644
--- a/llvm/test/Transforms/LoopVectorize/ARM/scalar-block-cost.ll
+++ b/llvm/test/Transforms/LoopVectorize/ARM/scalar-block-cost.ll
@@ -6,16 +6,17 @@ target triple = "thumbv8.1m.main-none-none-eabi"
define void @pred_loop(ptr %off, ptr %data, ptr %dst, i32 %n) #0 {
-; CHECK-COST: LV: Found an estimated cost of 0 for VF 1 For instruction: %i.09 = phi i32 [ %add, %for.body ], [ 0, %for.body.preheader ]
-; CHECK-COST-NEXT: LV: Found an estimated cost of 1 for VF 1 For instruction: %add = add nuw nsw i32 %i.09, 1
-; CHECK-COST-NEXT: LV: Found an estimated cost of 0 for VF 1 For instruction: %arrayidx = getelementptr inbounds i32, ptr %data, i32 %add
-; CHECK-COST-NEXT: LV: Found an estimated cost of 1 for VF 1 For instruction: %0 = load i32, ptr %arrayidx, align 4
-; CHECK-COST-NEXT: LV: Found an estimated cost of 1 for VF 1 For instruction: %add1 = add nsw i32 %0, 5
-; CHECK-COST-NEXT: LV: Found an estimated cost of 0 for VF 1 For instruction: %arrayidx2 = getelementptr inbounds i32, ptr %dst, i32 %i.09
-; CHECK-COST-NEXT: LV: Found an estimated cost of 1 for VF 1 For instruction: store i32 %add1, ptr %arrayidx2, align 4
-; CHECK-COST-NEXT: LV: Found an estimated cost of 1 for VF 1 For instruction: %exitcond.not = icmp eq i32 %add, %n
-; CHECK-COST-NEXT: LV: Found an estimated cost of 0 for VF 1 For instruction: br i1 %exitcond.not, label %exit.loopexit, label %for.body
-; CHECK-COST-NEXT: LV: Scalar loop costs: 5.
+; CHECK-COST-LABEL: LV: Checking a loop in 'pred_loop'
+; CHECK-COST: Cost of 0 for VF 1: EMIT-SCALAR ir<%i.09> = phi [ ir<0>, vector.ph ], [ ir<%add>, for.body ]
+; CHECK-COST-NEXT: Cost of 1 for VF 1: EMIT ir<%add> = add nuw nsw ir<%i.09>, ir<1>
+; CHECK-COST-NEXT: Cost of 0 for VF 1: EMIT ir<%arrayidx> = getelementptr inbounds ir<%data>, ir<%add>
+; CHECK-COST-NEXT: Cost of 1 for VF 1: EMIT-SCALAR ir<%0> = load ir<%arrayidx>
+; CHECK-COST-NEXT: Cost of 1 for VF 1: EMIT ir<%add1> = add nsw ir<%0>, ir<5>
+; CHECK-COST-NEXT: Cost of 0 for VF 1: EMIT ir<%arrayidx2> = getelementptr inbounds ir<%dst>, ir<%i.09>
+; CHECK-COST-NEXT: Cost of 1 for VF 1: EMIT store ir<%add1>, ir<%arrayidx2>
+; CHECK-COST-NEXT: Cost of 1 for VF 1: EMIT ir<%exitcond.not> = icmp eq ir<%add>, ir<%n>
+; CHECK-COST-NEXT: Cost of 0 for VF 1: EMIT branch-on-cond ir<%exitcond.not>
+; CHECK-COST: LV: Scalar loop costs: 5.
entry:
%cmp8 = icmp sgt i32 %n, 0
@@ -38,26 +39,26 @@ for.body:
define void @if_convert(ptr %a, ptr %b, i32 %start, i32 %end) #0 {
-; CHECK-COST-2: LV: Found an estimated cost of 0 for VF 1 For instruction: %i.032 = phi i32 [ %inc, %if.end ], [ %start, %for.body.preheader ]
-; CHECK-COST-2-NEXT: LV: Found an estimated cost of 0 for VF 1 For instruction: %arrayidx = getelementptr inbounds i32, ptr %a, i32 %i.032
-; CHECK-COST-2-NEXT: LV: Found an estimated cost of 1 for VF 1 For instruction: %0 = load i32, ptr %arrayidx, align 4
-; CHECK-COST-2-NEXT: LV: Found an estimated cost of 0 for VF 1 For instruction: %arrayidx2 = getelementptr inbounds i32, ptr %b, i32 %i.032
-; CHECK-COST-2-NEXT: LV: Found an estimated cost of 1 for VF 1 For instruction: %1 = load i32, ptr %arrayidx2, align 4
-; CHECK-COST-2-NEXT: LV: Found an estimated cost of 1 for VF 1 For instruction: %cmp3 = icmp sgt i32 %0, %1
-; CHECK-COST-2-NEXT: LV: Found an estimated cost of 0 for VF 1 For instruction: br i1 %cmp3, label %if.then, label %if.end
-; CHECK-COST-2-NEXT: LV: Found an estimated cost of 1 for VF 1 For instruction: %mul = mul nsw i32 %0, 5
-; CHECK-COST-2-NEXT: LV: Found an estimated cost of 1 for VF 1 For instruction: %add = add nsw i32 %mul, 3
-; CHECK-COST-2-NEXT: LV: Found an estimated cost of 0 for VF 1 For instruction: %factor = shl i32 %add, 1
-; CHECK-COST-2-NEXT: LV: Found an estimated cost of 1 for VF 1 For instruction: %sub = sub i32 %0, %1
-; CHECK-COST-2-NEXT: LV: Found an estimated cost of 1 for VF 1 For instruction: %add7 = add i32 %sub, %factor
-; CHECK-COST-2-NEXT: LV: Found an estimated cost of 1 for VF 1 For instruction: store i32 %add7, ptr %arrayidx2, align 4
-; CHECK-COST-2-NEXT: LV: Found an estimated cost of 0 for VF 1 For instruction: br label %if.end
-; CHECK-COST-2-NEXT: LV: Found an estimated cost of 0 for VF 1 For instruction: %k.0 = phi i32 [ %add, %if.then ], [ %0, %for.body ]
-; CHECK-COST-2-NEXT: LV: Found an estimated cost of 1 for VF 1 For instruction: store i32 %k.0, ptr %arrayidx, align 4
-; CHECK-COST-2-NEXT: LV: Found an estimated cost of 1 for VF 1 For instruction: %inc = add nsw i32 %i.032, 1
-; CHECK-COST-2-NEXT: LV: Found an estimated cost of 1 for VF 1 For instruction: %exitcond.not = icmp eq i32 %inc, %end
-; CHECK-COST-2-NEXT: LV: Found an estimated cost of 0 for VF 1 For instruction: br i1 %exitcond.not, label %for.cond.cleanup.loopexit, label %for.body
-; CHECK-COST-2-NEXT: LV: Scalar loop costs: 8.5.
+; CHECK-COST-2-LABEL: LV: Checking a loop in 'if_convert'
+; CHECK-COST-2: Cost of 0 for VF 1: EMIT-SCALAR ir<%i.032> = phi [ ir<%start>, vector.ph ], [ ir<%inc>, if.end ]
+; CHECK-COST-2-NEXT: Cost of 0 for VF 1: EMIT ir<%arrayidx> = getelementptr inbounds ir<%a>, ir<%i.032>
+; CHECK-COST-2-NEXT: Cost of 1 for VF 1: EMIT-SCALAR ir<%0> = load ir<%arrayidx>
+; CHECK-COST-2-NEXT: Cost of 0 for VF 1: EMIT ir<%arrayidx2> = getelementptr inbounds ir<%b>, ir<%i.032>
+; CHECK-COST-2-NEXT: Cost of 1 for VF 1: EMIT-SCALAR ir<%1> = load ir<%arrayidx2>
+; CHECK-COST-2-NEXT: Cost of 1 for VF 1: EMIT ir<%cmp3> = icmp sgt ir<%0>, ir<%1>
+; CHECK-COST-2-NEXT: Cost of 0 for VF 1: EMIT branch-on-cond ir<%cmp3>
+; CHECK-COST-2-NEXT: Cost of 1 for VF 1: EMIT ir<%mul> = mul nsw ir<%0>, ir<5>
+; CHECK-COST-2-NEXT: Cost of 1 for VF 1: EMIT ir<%add> = add nsw ir<%mul>, ir<3>
+; CHECK-COST-2-NEXT: Cost of 0 for VF 1: EMIT ir<%factor> = shl ir<%add>, ir<1>
+; CHECK-COST-2-NEXT: Cost of 1 for VF 1: EMIT ir<%sub> = sub ir<%0>, ir<%1>
+; CHECK-COST-2-NEXT: Cost of 1 for VF 1: EMIT ir<%add7> = add ir<%sub>, ir<%factor>
+; CHECK-COST-2-NEXT: Cost of 1 for VF 1: EMIT store ir<%add7>, ir<%arrayidx2>
+; CHECK-COST-2-NEXT: Cost of 0 for VF 1: EMIT-SCALAR ir<%k.0> = phi [ ir<%add>, if.then ], [ ir<%0>, for.body ]
+; CHECK-COST-2-NEXT: Cost of 1 for VF 1: EMIT store ir<%k.0>, ir<%arrayidx>
+; CHECK-COST-2-NEXT: Cost of 1 for VF 1: EMIT ir<%inc> = add nsw ir<%i.032>, ir<1>
+; CHECK-COST-2-NEXT: Cost of 1 for VF 1: EMIT ir<%exitcond.not> = icmp eq ir<%inc>, ir<%end>
+; CHECK-COST-2-NEXT: Cost of 0 for VF 1: EMIT branch-on-cond ir<%exitcond.not>
+; CHECK-COST-2: LV: Scalar loop costs: 8.5.
entry:
%cmp31 = icmp slt i32 %start, %end
diff --git a/llvm/test/Transforms/LoopVectorize/SystemZ/pr47665.ll b/llvm/test/Transforms/LoopVectorize/SystemZ/pr47665.ll
index 20a2604b2a9bab..392a5c545bf844 100644
--- a/llvm/test/Transforms/LoopVectorize/SystemZ/pr47665.ll
+++ b/llvm/test/Transforms/LoopVectorize/SystemZ/pr47665.ll
@@ -4,52 +4,24 @@
define void @test(ptr noalias %p, ptr noalias %q, i40 %a) {
; CHECK-LABEL: define void @test(
; CHECK-SAME: ptr noalias [[P:%.*]], ptr noalias [[Q:%.*]], i40 [[A:%.*]]) #[[ATTR0:[0-9]+]] {
-; CHECK-NEXT: [[ENTRY:.*:]]
-; CHECK-NEXT: br label %[[FOR_BODY:.*]]
-; CHECK: [[FOR_BODY]]:
+; CHECK-NEXT: [[FOR_BODY:.*]]:
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
-; CHECK-NEXT: [[IV:%.*]] = phi i32 [ 0, %[[FOR_BODY]] ], [ [[INDEX_NEXT:%.*]], %[[PRED_STORE_CONTINUE6:.*]] ]
-; CHECK-NEXT: [[VEC_IND:%.*]] = phi <4 x i8> [ <i8 0, i8 1, i8 2, i8 3>, %[[FOR_BODY]] ], [ [[VEC_IND_NEXT:%.*]], %[[PRED_STORE_CONTINUE6]] ]
-; CHECK-NEXT: [[TMP0:%.*]] = icmp ule <4 x i8> [[VEC_IND]], splat (i8 9)
-; CHECK-NEXT: store i1 false, ptr [[P]], align 1
-; CHECK-NEXT: [[TMP1:%.*]] = extractelement <4 x i1> [[TMP0]], i64 0
-; CHECK-NEXT: br i1 [[TMP1]], label %[[PRED_STORE_IF:.*]], label %[[PRED_STORE_CONTINUE:.*]]
-; CHECK: [[PRED_STORE_IF]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i32 [ 0, %[[FOR_BODY]] ], [ [[IV_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[SHL:%.*]] = shl i40 [[A]], 24
+; CHECK-NEXT: [[ASHR:%.*]] = ashr i40 [[SHL]], 28
+; CHECK-NEXT: [[TRUNC:%.*]] = trunc i40 [[ASHR]] to i32
+; CHECK-NEXT: [[ICMP_EQ:%.*]] = icmp eq i32 [[TRUNC]], 0
+; CHECK-NEXT: [[ZEXT:%.*]] = zext i1 [[ICMP_EQ]] to i32
+; CHECK-NEXT: [[ICMP_ULT:%.*]] = icmp ult i32 0, [[ZEXT]]
+; CHECK-NEXT: [[OR:%.*]] = or i1 [[ICMP_ULT]], true
+; CHECK-NEXT: [[ICMP_SGT:%.*]] = icmp sgt i1 [[OR]], false
+; CHECK-NEXT: store i1 [[ICMP_SGT]], ptr [[P]], align 1
; CHECK-NEXT: [[GEP:%.*]] = getelementptr inbounds i8, ptr [[Q]], i32 [[IV]]
; CHECK-NEXT: store i8 0, ptr [[GEP]], align 1
-; CHECK-NEXT: br label %[[PRED_STORE_CONTINUE]]
-; CHECK: [[PRED_STORE_CONTINUE]]:
-; CHECK-NEXT: [[TMP3:%.*]] = extractelement <4 x i1> [[TMP0]], i64 1
-; CHECK-NEXT: br i1 [[TMP3]], label %[[PRED_STORE_IF1:.*]], label %[[PRED_STORE_CONTINUE2:.*]]
-; CHECK: [[PRED_STORE_IF1]]:
-; CHECK-NEXT: [[IV_NEXT:%.*]] = add i32 [[IV]], 1
-; CHECK-NEXT: [[TMP5:%.*]] = getelementptr inbounds i8, ptr [[Q]], i32 [[IV_NEXT]]
-; CHECK-NEXT: store i8 0, ptr [[TMP5]], align 1
-; CHECK-NEXT: br label %[[PRED_STORE_CONTINUE2]]
-; CHECK: [[PRED_STORE_CONTINUE2]]:
-; CHECK-NEXT: [[TMP6:%.*]] = extractelement <4 x i1> [[TMP0]], i64 2
-; CHECK-NEXT: br i1 [[TMP6]], label %[[EXIT:.*]], label %[[PRED_STORE_CONTINUE4:.*]]
-; CHECK: [[EXIT]]:
-; CHECK-NEXT: [[TMP7:%.*]] = add i32 [[IV]], 2
-; CHECK-NEXT: [[TMP8:%.*]] = getelementptr inbounds i8, ptr [[Q]], i32 [[TMP7]]
-; CHECK-NEXT: store i8 0, ptr [[TMP8]], align 1
-; CHECK-NEXT: br label %[[PRED_STORE_CONTINUE4]]
-; CHECK: [[PRED_STORE_CONTINUE4]]:
-; CHECK-NEXT: [[TMP9:%.*]] = extractelement <4 x i1> [[TMP0]], i64 3
-; CHECK-NEXT: br i1 [[TMP9]], label %[[PRED_STORE_IF5:.*]], label %[[PRED_STORE_CONTINUE6]]
-; CHECK: [[PRED_STORE_IF5]]:
-; CHECK-NEXT: [[TMP10:%.*]] = add i32 [[IV]], 3
-; CHECK-NEXT: [[TMP11:%.*]] = getelementptr inbounds i8, ptr [[Q]], i32 [[TMP10]]
-; CHECK-NEXT: store i8 0, ptr [[TMP11]], align 1
-; CHECK-NEXT: br label %[[PRED_STORE_CONTINUE6]]
-; CHECK: [[PRED_STORE_CONTINUE6]]:
-; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i32 [[IV]], 4
-; CHECK-NEXT: [[VEC_IND_NEXT]] = add nuw <4 x i8> [[VEC_IND]], splat (i8 4)
-; CHECK-NEXT: [[TMP12:%.*]] = icmp eq i32 [[INDEX_NEXT]], 12
-; CHECK-NEXT: br i1 [[TMP12]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
-; CHECK: [[MIDDLE_BLOCK]]:
-; CHECK-NEXT: br label %[[EXIT1:.*]]
+; CHECK-NEXT: [[IV_NEXT]] = add i32 [[IV]], 1
+; CHECK-NEXT: [[COND:%.*]] = icmp ult i32 [[IV_NEXT]], 10
+; CHECK-NEXT: br i1 [[COND]], label %[[VECTOR_BODY]], label %[[EXIT1:.*]]
; CHECK: [[EXIT1]]:
; CHECK-NEXT: ret void
;
diff --git a/llvm/test/Transforms/LoopVectorize/WebAssembly/int-mac-reduction-costs.ll b/llvm/test/Transforms/LoopVectorize/WebAssembly/int-mac-reduction-costs.ll
index 6580a3dacc21c2..8e12d55c928873 100644
--- a/llvm/test/Transforms/LoopVectorize/WebAssembly/int-mac-reduction-costs.ll
+++ b/llvm/test/Transforms/LoopVectorize/WebAssembly/int-mac-reduction-costs.ll
@@ -5,11 +5,11 @@ target triple = "wasm32"
define hidden i32 @i32_mac_s8(ptr nocapture noundef readonly %a, ptr nocapture noundef readonly %b, i32 noundef %N) {
; CHECK-LABEL: 'i32_mac_s8'
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %0 = load i8, ptr %arrayidx, align 1
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction: %conv = sext i8 %0 to i32
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %1 = load i8, ptr %arrayidx1, align 1
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction: %conv2 = sext i8 %1 to i32
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %mul = mul nsw i32 %conv2, %conv
+; CHECK: Cost of 1 for VF 1: EMIT-SCALAR ir<%0> = load
+; CHECK: Cost of 0 for VF 1: EMIT-SCALAR ir<%conv> = sext ir<%0> to i32
+; CHECK: Cost of 1 for VF 1: EMIT-SCALAR ir<%1> = load
+; CHECK: Cost of 0 for VF 1: EMIT-SCALAR ir<%conv2> = sext ir<%1> to i32
+; CHECK: Cost of 1 for VF 1: EMIT ir<%mul> = mul nsw ir<%conv2>, ir<%conv>
; CHECK: Cost of 3 for VF 2: WIDEN ir<%0> = load
; CHECK: Cost of 0 for VF 2: WIDEN-CAST ir<%conv> = sext ir<%0> to i32
@@ -49,11 +49,11 @@ for.body:
define hidden i32 @i32_mac_s16(ptr nocapture noundef readonly %a, ptr nocapture noundef readonly %b, i32 noundef %N) {
; CHECK-LABEL: 'i32_mac_s16'
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %0 = load i16, ptr %arrayidx, align 2
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction: %conv = sext i16 %0 to i32
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %1 = load i16, ptr %arrayidx1, align 2
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction: %conv2 = sext i16 %1 to i32
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %mul = mul nsw i32 %conv2, %conv
+; CHECK: Cost of 1 for VF 1: EMIT-SCALAR ir<%0> = load
+; CHECK: Cost of 0 for VF 1: EMIT-SCALAR ir<%conv> = sext ir<%0> to i32
+; CHECK: Cost of 1 for VF 1: EMIT-SCALAR ir<%1> = load
+; CHECK: Cost of 0 for VF 1: EMIT-SCALAR ir<%conv2> = sext ir<%1> to i32
+; CHECK: Cost of 1 for VF 1: EMIT ir<%mul> = mul nsw ir<%conv2>, ir<%conv>
; CHECK: Cost of 2 for VF 2: WIDEN ir<%0> = load
; CHECK: Cost of 0 for VF 2: WIDEN-CAST ir<%conv> = sext ir<%0> to i32
@@ -93,11 +93,11 @@ for.body:
define hidden i64 @i64_mac_s16(ptr nocapture noundef readonly %a, ptr nocapture noundef readonly %b, i32 noundef %N) {
; CHECK-LABEL: 'i64_mac_s16'
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %0 = load i16, ptr %arrayidx, align 2
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction: %conv = sext i16 %0 to i64
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %1 = load i16, ptr %arrayidx1, align 2
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction: %conv2 = sext i16 %1 to i64
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %mul = mul nsw i64 %conv2, %conv
+; CHECK: Cost of 1 for VF 1: EMIT-SCALAR ir<%0> = load
+; CHECK: Cost of 0 for VF 1: EMIT-SCALAR ir<%conv> = sext ir<%0> to i64
+; CHECK: Cost of 1 for VF 1: EMIT-SCALAR ir<%1> = load
+; CHECK: Cost of 0 for VF 1: EMIT-SCALAR ir<%conv2> = sext ir<%1> to i64
+; CHECK: Cost of 1 for VF 1: EMIT ir<%mul> = mul nsw ir<%conv2>, ir<%conv>
; CHECK: Cost of 2 for VF 2: WIDEN ir<%0> = load
; CHECK: Cost of 1 for VF 2: WIDEN-CAST ir<%conv> = sext ir<%0> to i64
@@ -131,10 +131,10 @@ for.body:
define hidden i64 @i64_mac_s32(ptr nocapture noundef readonly %a, ptr nocapture noundef readonly %b, i32 noundef %N) {
; CHECK-LABEL: 'i64_mac_s32'
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %0 = load i32, ptr %arrayidx, align 4
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %1 = load i32, ptr %arrayidx1, align 4
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %mul = mul i32 %1, %0
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %conv = sext i32 %mul to i64
+; CHECK: Cost of 1 for VF 1: EMIT-SCALAR ir<%0> = load
+; CHECK: Cost of 1 for VF 1: EMIT-SCALAR ir<%1> = load
+; CHECK: Cost of 1 for VF 1: EMIT ir<%mul> = mul ir<%1>, ir<%0>
+; CHECK: Cost of 1 for VF 1: EMIT-SCALAR ir<%conv> = sext ir<%mul> to i64
; CHECK: Cost of 2 for VF 2: WIDEN ir<%0> = load
; CHECK: Cost of 2 for VF 2: WIDEN ir<%1> = load
@@ -166,11 +166,11 @@ for.body:
define hidden i32 @i32_mac_u8(ptr nocapture noundef readonly %a, ptr nocapture noundef readonly %b, i32 noundef %N) {
; CHECK-LABEL: 'i32_mac_u8'
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %0 = load i8, ptr %arrayidx, align 1
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction: %conv = zext i8 %0 to i32
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %1 = load i8, ptr %arrayidx1, align 1
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction: %conv2 = zext i8 %1 to i32
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %mul = mul nuw nsw i32 %conv2, %conv
+; CHECK: Cost of 1 for VF 1: EMIT-SCALAR ir<%0> = load
+; CHECK: Cost of 0 for VF 1: EMIT-SCALAR ir<%conv> = zext ir<%0> to i32
+; CHECK: Cost of 1 for VF 1: EMIT-SCALAR ir<%1> = load
+; CHECK: Cost of 0 for VF 1: EMIT-SCALAR ir<%conv2> = zext ir<%1> to i32
+; CHECK: Cost of 1 for VF 1: EMIT ir<%mul> = mul nuw nsw ir<%conv2>, ir<%conv>
; CHECK: Cost of 3 for VF 2: WIDEN ir<%0> = load
; CHECK: Cost of 0 for VF 2: WIDEN-CAST ir<%conv> = zext ir<%0> to i32
@@ -210,11 +210,11 @@ for.body:
define hidden i32 @i32_mac_u16(ptr nocapture noundef readonly %a, ptr nocapture noundef readonly %b, i32 noundef %N) {
; CHECK-LABEL: 'i32_mac_u16'
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %0 = load i16, ptr %arrayidx, align 2
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction: %conv = zext i16 %0 to i32
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %1 = load i16, ptr %arrayidx1, align 2
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction: %conv2 = zext i16 %1 to i32
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %mul = mul nuw nsw i32 %conv2, %conv
+; CHECK: Cost of 1 for VF 1: EMIT-SCALAR ir<%0> = load
+; CHECK: Cost of 0 for VF 1: EMIT-SCALAR ir<%conv> = zext ir<%0> to i32
+; CHECK: Cost of 1 for VF 1: EMIT-SCALAR ir<%1> = load
+; CHECK: Cost of 0 for VF 1: EMIT-SCALAR ir<%conv2> = zext ir<%1> to i32
+; CHECK: Cost of 1 for VF 1: EMIT ir<%mul> = mul nuw nsw ir<%conv2>, ir<%conv>
; CHECK: Cost of 2 for VF 2: WIDEN ir<%0> = load
; CHECK: Cost of 0 for VF 2: WIDEN-CAST ir<%conv> = zext ir<%0> to i32
@@ -254,11 +254,11 @@ for.body:
define hidden i64 @i64_mac_u16(ptr nocapture noundef readonly %a, ptr nocapture noundef readonly %b, i32 noundef %N) {
; CHECK-LABEL: 'i64_mac_u16'
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %0 = load i16, ptr %arrayidx, align 2
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction: %conv = zext i16 %0 to i64
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %1 = load i16, ptr %arrayidx1, align 2
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction: %conv2 = zext i16 %1 to i64
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %mul = mul nuw nsw i64 %conv2, %conv
+; CHECK: Cost of 1 for VF 1: EMIT-SCALAR ir<%0> = load
+; CHECK: Cost of 0 for VF 1: EMIT-SCALAR ir<%conv> = zext ir<%0> to i64
+; CHECK: Cost of 1 for VF 1: EMIT-SCALAR ir<%1> = load
+; CHECK: Cost of 0 for VF 1: EMIT-SCALAR ir<%conv2> = zext ir<%1> to i64
+; CHECK: Cost of 1 for VF 1: EMIT ir<%mul> = mul nuw nsw ir<%conv2>, ir<%conv>
; CHECK: Cost of 2 for VF 2: WIDEN ir<%0> = load
; CHECK: Cost of 1 for VF 2: WIDEN-CAST ir<%conv> = zext ir<%0> to i64
@@ -292,10 +292,10 @@ for.body:
define hidden i64 @i64_mac_u32(ptr nocapture noundef readonly %a, ptr nocapture noundef readonly %b, i32 noundef %N) {
; CHECK-LABEL: 'i64_mac_u32'
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %0 = load i32, ptr %arrayidx, align 4
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %1 = load i32, ptr %arrayidx1, align 4
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %mul = mul i32 %1, %0
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %conv = zext i32 %mul to i64
+; CHECK: Cost of 1 for VF 1: EMIT-SCALAR ir<%0> = load
+; CHECK: Cost of 1 for VF 1: EMIT-SCALAR ir<%1> = load
+; CHECK: Cost of 1 for VF 1: EMIT ir<%mul> = mul ir<%1>, ir<%0>
+; CHECK: Cost of 1 for VF 1: EMIT-SCALAR ir<%conv> = zext ir<%mul> to i64
; CHECK: Cost of 2 for VF 2: WIDEN ir<%0> = load
; CHECK: Cost of 2 for VF 2: WIDEN ir<%1> = load
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i16-with-i8-index.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i16-with-i8-index.ll
index 1ac8f11262064c..2770dd4801aebc 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i16-with-i8-index.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i16-with-i8-index.ll
@@ -17,14 +17,14 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE-LABEL: 'test'
-; SSE: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB = load i16, ptr %inB, align 2
+; SSE: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB> = load ir<%inB>
; SSE: Cost of 24 for VF 2: REPLICATE ir<%valB> = load ir<%inB>
; SSE: Cost of 48 for VF 4: REPLICATE ir<%valB> = load ir<%inB>
; SSE: Cost of 96 for VF 8: REPLICATE ir<%valB> = load ir<%inB>
; SSE: Cost of 192 for VF 16: REPLICATE ir<%valB> = load ir<%inB>
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB = load i16, ptr %inB, align 2
+; AVX1: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB> = load ir<%inB>
; AVX1: Cost of 24 for VF 2: REPLICATE ir<%valB> = load ir<%inB>
; AVX1: Cost of 48 for VF 4: REPLICATE ir<%valB> = load ir<%inB>
; AVX1: Cost of 96 for VF 8: REPLICATE ir<%valB> = load ir<%inB>
@@ -32,7 +32,7 @@ define void @test() {
; AVX1: Cost of 386 for VF 32: REPLICATE ir<%valB> = load ir<%inB>
;
; AVX2-SLOWGATHER-LABEL: 'test'
-; AVX2-SLOWGATHER: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB = load i16, ptr %inB, align 2
+; AVX2-SLOWGATHER: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB> = load ir<%inB>
; AVX2-SLOWGATHER: Cost of 4 for VF 2: REPLICATE ir<%valB> = load ir<%inB>
; AVX2-SLOWGATHER: Cost of 8 for VF 4: REPLICATE ir<%valB> = load ir<%inB>
; AVX2-SLOWGATHER: Cost of 16 for VF 8: REPLICATE ir<%valB> = load ir<%inB>
@@ -40,7 +40,7 @@ define void @test() {
; AVX2-SLOWGATHER: Cost of 66 for VF 32: REPLICATE ir<%valB> = load ir<%inB>
;
; AVX2-FASTGATHER-LABEL: 'test'
-; AVX2-FASTGATHER: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB = load i16, ptr %inB, align 2
+; AVX2-FASTGATHER: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB> = load ir<%inB>
; AVX2-FASTGATHER: Cost of 6 for VF 2: REPLICATE ir<%valB> = load ir<%inB>
; AVX2-FASTGATHER: Cost of 13 for VF 4: REPLICATE ir<%valB> = load ir<%inB>
; AVX2-FASTGATHER: Cost of 26 for VF 8: REPLICATE ir<%valB> = load ir<%inB>
@@ -48,7 +48,7 @@ define void @test() {
; AVX2-FASTGATHER: Cost of 106 for VF 32: REPLICATE ir<%valB> = load ir<%inB>
;
; AVX512-LABEL: 'test'
-; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB = load i16, ptr %inB, align 2
+; AVX512: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB> = load ir<%inB>
; AVX512: Cost of 6 for VF 2: REPLICATE ir<%valB> = load ir<%inB>
; AVX512: Cost of 13 for VF 4: REPLICATE ir<%valB> = load ir<%inB>
; AVX512: Cost of 27 for VF 8: REPLICATE ir<%valB> = load ir<%inB>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i32-with-i8-index.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i32-with-i8-index.ll
index 2d5a30019bacd6..58b96771aedbea 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i32-with-i8-index.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i32-with-i8-index.ll
@@ -17,21 +17,21 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
-; SSE2: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB = load i32, ptr %inB, align 4
+; SSE2: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB> = load ir<%inB>
; SSE2: Cost of 25 for VF 2: REPLICATE ir<%valB> = load ir<%inB>
; SSE2: Cost of 51 for VF 4: REPLICATE ir<%valB> = load ir<%inB>
; SSE2: Cost of 102 for VF 8: REPLICATE ir<%valB> = load ir<%inB>
; SSE2: Cost of 204 for VF 16: REPLICATE ir<%valB> = load ir<%inB>
;
; SSE42-LABEL: 'test'
-; SSE42: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB = load i32, ptr %inB, align 4
+; SSE42: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB> = load ir<%inB>
; SSE42: Cost of 24 for VF 2: REPLICATE ir<%valB> = load ir<%inB>
; SSE42: Cost of 48 for VF 4: REPLICATE ir<%valB> = load ir<%inB>
; SSE42: Cost of 96 for VF 8: REPLICATE ir<%valB> = load ir<%inB>
; SSE42: Cost of 192 for VF 16: REPLICATE ir<%valB> = load ir<%inB>
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB = load i32, ptr %inB, align 4
+; AVX1: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB> = load ir<%inB>
; AVX1: Cost of 24 for VF 2: REPLICATE ir<%valB> = load ir<%inB>
; AVX1: Cost of 48 for VF 4: REPLICATE ir<%valB> = load ir<%inB>
; AVX1: Cost of 97 for VF 8: REPLICATE ir<%valB> = load ir<%inB>
@@ -39,7 +39,7 @@ define void @test() {
; AVX1: Cost of 388 for VF 32: REPLICATE ir<%valB> = load ir<%inB>
;
; AVX2-SLOWGATHER-LABEL: 'test'
-; AVX2-SLOWGATHER: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB = load i32, ptr %inB, align 4
+; AVX2-SLOWGATHER: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB> = load ir<%inB>
; AVX2-SLOWGATHER: Cost of 4 for VF 2: REPLICATE ir<%valB> = load ir<%inB>
; AVX2-SLOWGATHER: Cost of 8 for VF 4: REPLICATE ir<%valB> = load ir<%inB>
; AVX2-SLOWGATHER: Cost of 17 for VF 8: REPLICATE ir<%valB> = load ir<%inB>
@@ -47,7 +47,7 @@ define void @test() {
; AVX2-SLOWGATHER: Cost of 68 for VF 32: REPLICATE ir<%valB> = load ir<%inB>
;
; AVX2-FASTGATHER-LABEL: 'test'
-; AVX2-FASTGATHER: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB = load i32, ptr %inB, align 4
+; AVX2-FASTGATHER: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB> = load ir<%inB>
; AVX2-FASTGATHER: Cost of 4 for VF 2: WIDEN ir<%valB> = load ir<%inB>
; AVX2-FASTGATHER: Cost of 6 for VF 4: WIDEN ir<%valB> = load ir<%inB>
; AVX2-FASTGATHER: Cost of 12 for VF 8: WIDEN ir<%valB> = load ir<%inB>
@@ -55,7 +55,7 @@ define void @test() {
; AVX2-FASTGATHER: Cost of 48 for VF 32: WIDEN ir<%valB> = load ir<%inB>
;
; AVX512-LABEL: 'test'
-; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB = load i32, ptr %inB, align 4
+; AVX512: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB> = load ir<%inB>
; AVX512: Cost of 6 for VF 2: REPLICATE ir<%valB> = load ir<%inB>
; AVX512: Cost of 13 for VF 4: REPLICATE ir<%valB> = load ir<%inB>
; AVX512: Cost of 10 for VF 8: WIDEN ir<%valB> = load ir<%inB>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i64-with-i8-index.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i64-with-i8-index.ll
index ce5828a46eea74..7876551023ee67 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i64-with-i8-index.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i64-with-i8-index.ll
@@ -17,21 +17,21 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
-; SSE2: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB = load i64, ptr %inB, align 8
+; SSE2: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB> = load ir<%inB>
; SSE2: Cost of 25 for VF 2: REPLICATE ir<%valB> = load ir<%inB>
; SSE2: Cost of 50 for VF 4: REPLICATE ir<%valB> = load ir<%inB>
; SSE2: Cost of 100 for VF 8: REPLICATE ir<%valB> = load ir<%inB>
; SSE2: Cost of 200 for VF 16: REPLICATE ir<%valB> = load ir<%inB>
;
; SSE42-LABEL: 'test'
-; SSE42: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB = load i64, ptr %inB, align 8
+; SSE42: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB> = load ir<%inB>
; SSE42: Cost of 24 for VF 2: REPLICATE ir<%valB> = load ir<%inB>
; SSE42: Cost of 48 for VF 4: REPLICATE ir<%valB> = load ir<%inB>
; SSE42: Cost of 96 for VF 8: REPLICATE ir<%valB> = load ir<%inB>
; SSE42: Cost of 192 for VF 16: REPLICATE ir<%valB> = load ir<%inB>
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB = load i64, ptr %inB, align 8
+; AVX1: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB> = load ir<%inB>
; AVX1: Cost of 24 for VF 2: REPLICATE ir<%valB> = load ir<%inB>
; AVX1: Cost of 49 for VF 4: REPLICATE ir<%valB> = load ir<%inB>
; AVX1: Cost of 98 for VF 8: REPLICATE ir<%valB> = load ir<%inB>
@@ -39,7 +39,7 @@ define void @test() {
; AVX1: Cost of 392 for VF 32: REPLICATE ir<%valB> = load ir<%inB>
;
; AVX2-SLOWGATHER-LABEL: 'test'
-; AVX2-SLOWGATHER: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB = load i64, ptr %inB, align 8
+; AVX2-SLOWGATHER: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB> = load ir<%inB>
; AVX2-SLOWGATHER: Cost of 4 for VF 2: REPLICATE ir<%valB> = load ir<%inB>
; AVX2-SLOWGATHER: Cost of 9 for VF 4: REPLICATE ir<%valB> = load ir<%inB>
; AVX2-SLOWGATHER: Cost of 18 for VF 8: REPLICATE ir<%valB> = load ir<%inB>
@@ -47,7 +47,7 @@ define void @test() {
; AVX2-SLOWGATHER: Cost of 72 for VF 32: REPLICATE ir<%valB> = load ir<%inB>
;
; AVX2-FASTGATHER-LABEL: 'test'
-; AVX2-FASTGATHER: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB = load i64, ptr %inB, align 8
+; AVX2-FASTGATHER: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB> = load ir<%inB>
; AVX2-FASTGATHER: Cost of 4 for VF 2: WIDEN ir<%valB> = load ir<%inB>
; AVX2-FASTGATHER: Cost of 6 for VF 4: WIDEN ir<%valB> = load ir<%inB>
; AVX2-FASTGATHER: Cost of 12 for VF 8: WIDEN ir<%valB> = load ir<%inB>
@@ -55,7 +55,7 @@ define void @test() {
; AVX2-FASTGATHER: Cost of 48 for VF 32: WIDEN ir<%valB> = load ir<%inB>
;
; AVX512-LABEL: 'test'
-; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB = load i64, ptr %inB, align 8
+; AVX512: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB> = load ir<%inB>
; AVX512: Cost of 6 for VF 2: REPLICATE ir<%valB> = load ir<%inB>
; AVX512: Cost of 14 for VF 4: REPLICATE ir<%valB> = load ir<%inB>
; AVX512: Cost of 10 for VF 8: WIDEN ir<%valB> = load ir<%inB>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i8-with-i8-index.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i8-with-i8-index.ll
index d894cdb753bcd6..435f7c3becd730 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i8-with-i8-index.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i8-with-i8-index.ll
@@ -17,21 +17,21 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
-; SSE2: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB = load i8, ptr %inB, align 1
+; SSE2: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB> = load ir<%inB>
; SSE2: Cost of 25 for VF 2: REPLICATE ir<%valB> = load ir<%inB>
; SSE2: Cost of 51 for VF 4: REPLICATE ir<%valB> = load ir<%inB>
; SSE2: Cost of 103 for VF 8: REPLICATE ir<%valB> = load ir<%inB>
; SSE2: Cost of 207 for VF 16: REPLICATE ir<%valB> = load ir<%inB>
;
; SSE42-LABEL: 'test'
-; SSE42: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB = load i8, ptr %inB, align 1
+; SSE42: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB> = load ir<%inB>
; SSE42: Cost of 24 for VF 2: REPLICATE ir<%valB> = load ir<%inB>
; SSE42: Cost of 48 for VF 4: REPLICATE ir<%valB> = load ir<%inB>
; SSE42: Cost of 96 for VF 8: REPLICATE ir<%valB> = load ir<%inB>
; SSE42: Cost of 192 for VF 16: REPLICATE ir<%valB> = load ir<%inB>
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB = load i8, ptr %inB, align 1
+; AVX1: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB> = load ir<%inB>
; AVX1: Cost of 24 for VF 2: REPLICATE ir<%valB> = load ir<%inB>
; AVX1: Cost of 48 for VF 4: REPLICATE ir<%valB> = load ir<%inB>
; AVX1: Cost of 96 for VF 8: REPLICATE ir<%valB> = load ir<%inB>
@@ -39,7 +39,7 @@ define void @test() {
; AVX1: Cost of 385 for VF 32: REPLICATE ir<%valB> = load ir<%inB>
;
; AVX2-SLOWGATHER-LABEL: 'test'
-; AVX2-SLOWGATHER: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB = load i8, ptr %inB, align 1
+; AVX2-SLOWGATHER: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB> = load ir<%inB>
; AVX2-SLOWGATHER: Cost of 4 for VF 2: REPLICATE ir<%valB> = load ir<%inB>
; AVX2-SLOWGATHER: Cost of 8 for VF 4: REPLICATE ir<%valB> = load ir<%inB>
; AVX2-SLOWGATHER: Cost of 16 for VF 8: REPLICATE ir<%valB> = load ir<%inB>
@@ -47,7 +47,7 @@ define void @test() {
; AVX2-SLOWGATHER: Cost of 65 for VF 32: REPLICATE ir<%valB> = load ir<%inB>
;
; AVX2-FASTGATHER-LABEL: 'test'
-; AVX2-FASTGATHER: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB = load i8, ptr %inB, align 1
+; AVX2-FASTGATHER: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB> = load ir<%inB>
; AVX2-FASTGATHER: Cost of 6 for VF 2: REPLICATE ir<%valB> = load ir<%inB>
; AVX2-FASTGATHER: Cost of 13 for VF 4: REPLICATE ir<%valB> = load ir<%inB>
; AVX2-FASTGATHER: Cost of 26 for VF 8: REPLICATE ir<%valB> = load ir<%inB>
@@ -55,7 +55,7 @@ define void @test() {
; AVX2-FASTGATHER: Cost of 105 for VF 32: REPLICATE ir<%valB> = load ir<%inB>
;
; AVX512-LABEL: 'test'
-; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB = load i8, ptr %inB, align 1
+; AVX512: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB> = load ir<%inB>
; AVX512: Cost of 6 for VF 2: REPLICATE ir<%valB> = load ir<%inB>
; AVX512: Cost of 13 for VF 4: REPLICATE ir<%valB> = load ir<%inB>
; AVX512: Cost of 27 for VF 8: REPLICATE ir<%valB> = load ir<%inB>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/handle-iptr-with-data-layout-to-not-assert.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/handle-iptr-with-data-layout-to-not-assert.ll
index 1147497d9982d4..1c5e5d5aa21a3d 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/handle-iptr-with-data-layout-to-not-assert.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/handle-iptr-with-data-layout-to-not-assert.ll
@@ -5,7 +5,6 @@ target triple = "x86_64-unknown-linux-gnu"
define ptr @foo(ptr %__first, ptr %__last) #0 {
; CHECK-LABEL: 'foo'
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: store ptr %0, ptr %__last, align 8
; CHECK: Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 2, vp<%next.gep>
; CHECK: Cost of 1 for VF 4: INTERLEAVE-GROUP with factor 2, vp<%next.gep>
; CHECK: Cost of 3 for VF 8: INTERLEAVE-GROUP with factor 2, vp<%next.gep>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-2.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-2.ll
index 3b3c3dcb780416..fe3fbb9c4bfee8 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-2.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-2.ll
@@ -13,7 +13,6 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
-; SSE2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load float, ptr %in0, align 4
; SSE2: Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
; SSE2: ir<%v0> = load from index 0
; SSE2: ir<%v1> = load from index 1
@@ -24,7 +23,6 @@ define void @test() {
; SSE2: Cost of 28 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load float, ptr %in0, align 4
; AVX1: Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
; AVX1: ir<%v0> = load from index 0
; AVX1: ir<%v1> = load from index 1
@@ -36,7 +34,6 @@ define void @test() {
; AVX1: Cost of 60 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
-; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load float, ptr %in0, align 4
; AVX2: Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
; AVX2: ir<%v0> = load from index 0
; AVX2: ir<%v1> = load from index 1
@@ -54,7 +51,6 @@ define void @test() {
; AVX2: ir<%v1> = load from index 1
;
; AVX512-LABEL: 'test'
-; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load float, ptr %in0, align 4
; AVX512: Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-3.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-3.ll
index 2218048e476813..24ac329df20d63 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-3.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-3.ll
@@ -13,14 +13,12 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
-; SSE2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load float, ptr %in0, align 4
; SSE2: Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 14 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 28 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load float, ptr %in0, align 4
; AVX1: Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 15 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -28,7 +26,6 @@ define void @test() {
; AVX1: Cost of 60 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
-; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load float, ptr %in0, align 4
; AVX2: Cost of 6 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
; AVX2: ir<%v0> = load from index 0
; AVX2: ir<%v1> = load from index 1
@@ -51,7 +48,6 @@ define void @test() {
; AVX2: ir<%v2> = load from index 2
;
; AVX512-LABEL: 'test'
-; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load float, ptr %in0, align 4
; AVX512: Cost of 4 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-4.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-4.ll
index 679ff5ecec952e..b2cd0273efd8de 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-4.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-4.ll
@@ -13,14 +13,12 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
-; SSE2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load float, ptr %in0, align 4
; SSE2: Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 14 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 28 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load float, ptr %in0, align 4
; AVX1: Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 15 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -28,7 +26,6 @@ define void @test() {
; AVX1: Cost of 60 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
-; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load float, ptr %in0, align 4
; AVX2: Cost of 5 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
; AVX2: ir<%v0> = load from index 0
; AVX2: ir<%v1> = load from index 1
@@ -56,7 +53,6 @@ define void @test() {
; AVX2: ir<%v3> = load from index 3
;
; AVX512-LABEL: 'test'
-; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load float, ptr %in0, align 4
; AVX512: Cost of 5 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-5.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-5.ll
index 31a1a91bdac59a..f8980b2fb0f30d 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-5.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-5.ll
@@ -13,14 +13,12 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
-; SSE2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load float, ptr %in0, align 4
; SSE2: Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 14 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 28 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load float, ptr %in0, align 4
; AVX1: Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 15 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -28,7 +26,6 @@ define void @test() {
; AVX1: Cost of 60 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
-; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load float, ptr %in0, align 4
; AVX2: Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 15 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -36,7 +33,6 @@ define void @test() {
; AVX2: Cost of 60 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX512-LABEL: 'test'
-; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load float, ptr %in0, align 4
; AVX512: Cost of 6 for VF 2: INTERLEAVE-GROUP with factor 5, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-6.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-6.ll
index fd19eecd004459..c62d8eea92143c 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-6.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-6.ll
@@ -13,14 +13,12 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
-; SSE2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load float, ptr %in0, align 4
; SSE2: Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 14 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 28 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load float, ptr %in0, align 4
; AVX1: Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 15 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -28,7 +26,6 @@ define void @test() {
; AVX1: Cost of 60 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
-; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load float, ptr %in0, align 4
; AVX2: Cost of 8 for VF 2: INTERLEAVE-GROUP with factor 6, ir<%in0>
; AVX2: ir<%v0> = load from index 0
; AVX2: ir<%v1> = load from index 1
@@ -60,7 +57,6 @@ define void @test() {
; AVX2: Cost of 60 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX512-LABEL: 'test'
-; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load float, ptr %in0, align 4
; AVX512: Cost of 7 for VF 2: INTERLEAVE-GROUP with factor 6, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-7.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-7.ll
index f0395dbb3bb891..68f7f233a8a86e 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-7.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-7.ll
@@ -13,14 +13,12 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
-; SSE2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load float, ptr %in0, align 4
; SSE2: Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 14 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 28 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load float, ptr %in0, align 4
; AVX1: Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 15 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -28,7 +26,6 @@ define void @test() {
; AVX1: Cost of 60 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
-; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load float, ptr %in0, align 4
; AVX2: Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 15 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -36,7 +33,6 @@ define void @test() {
; AVX2: Cost of 60 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX512-LABEL: 'test'
-; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load float, ptr %in0, align 4
; AVX512: Cost of 8 for VF 2: INTERLEAVE-GROUP with factor 7, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-8.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-8.ll
index eafe9cfd912c4b..154c14dd895e0d 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-8.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-8.ll
@@ -13,14 +13,12 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
-; SSE2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load float, ptr %in0, align 4
; SSE2: Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 14 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 28 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load float, ptr %in0, align 4
; AVX1: Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 15 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -28,7 +26,6 @@ define void @test() {
; AVX1: Cost of 60 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
-; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load float, ptr %in0, align 4
; AVX2: Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 48 for VF 8: INTERLEAVE-GROUP with factor 8, ir<%in0>
@@ -44,7 +41,6 @@ define void @test() {
; AVX2: Cost of 60 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX512-LABEL: 'test'
-; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load float, ptr %in0, align 4
; AVX512: Cost of 9 for VF 2: INTERLEAVE-GROUP with factor 8, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-2.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-2.ll
index 38fbf409100e8c..c9ddd0a38f1038 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-2.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-2.ll
@@ -13,7 +13,6 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
-; SSE2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load double, ptr %in0, align 8
; SSE2: Cost of 4 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
; SSE2: ir<%v0> = load from index 0
; SSE2: ir<%v1> = load from index 1
@@ -22,7 +21,6 @@ define void @test() {
; SSE2: Cost of 24 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load double, ptr %in0, align 8
; AVX1: Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
; AVX1: ir<%v0> = load from index 0
; AVX1: ir<%v1> = load from index 1
@@ -32,7 +30,6 @@ define void @test() {
; AVX1: Cost of 56 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
-; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load double, ptr %in0, align 8
; AVX2: Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
; AVX2: ir<%v0> = load from index 0
; AVX2: ir<%v1> = load from index 1
@@ -50,7 +47,6 @@ define void @test() {
; AVX2: ir<%v1> = load from index 1
;
; AVX512-LABEL: 'test'
-; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load double, ptr %in0, align 8
; AVX512: Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-3.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-3.ll
index 51bc25a559fdae..d42188eaa22996 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-3.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-3.ll
@@ -13,14 +13,12 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
-; SSE2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load double, ptr %in0, align 8
; SSE2: Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 6 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 12 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 24 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load double, ptr %in0, align 8
; AVX1: Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 14 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -28,7 +26,6 @@ define void @test() {
; AVX1: Cost of 56 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
-; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load double, ptr %in0, align 8
; AVX2: Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
; AVX2: ir<%v0> = load from index 0
; AVX2: ir<%v1> = load from index 1
@@ -48,7 +45,6 @@ define void @test() {
; AVX2: Cost of 56 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX512-LABEL: 'test'
-; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load double, ptr %in0, align 8
; AVX512: Cost of 4 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-4.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-4.ll
index 5f3c8a3f925e85..1bc77d8d3ff3c4 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-4.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-4.ll
@@ -13,14 +13,12 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
-; SSE2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load double, ptr %in0, align 8
; SSE2: Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 6 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 12 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 24 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load double, ptr %in0, align 8
; AVX1: Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 14 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -28,7 +26,6 @@ define void @test() {
; AVX1: Cost of 56 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
-; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load double, ptr %in0, align 8
; AVX2: Cost of 8 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
; AVX2: ir<%v0> = load from index 0
; AVX2: ir<%v1> = load from index 1
@@ -52,7 +49,6 @@ define void @test() {
; AVX2: Cost of 56 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX512-LABEL: 'test'
-; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load double, ptr %in0, align 8
; AVX512: Cost of 5 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-5.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-5.ll
index 7d18d05a344a84..9992f5ad71ab42 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-5.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-5.ll
@@ -13,14 +13,12 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
-; SSE2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load double, ptr %in0, align 8
; SSE2: Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 6 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 12 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 24 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load double, ptr %in0, align 8
; AVX1: Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 14 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -28,7 +26,6 @@ define void @test() {
; AVX1: Cost of 56 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
-; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load double, ptr %in0, align 8
; AVX2: Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 14 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -36,7 +33,6 @@ define void @test() {
; AVX2: Cost of 56 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX512-LABEL: 'test'
-; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load double, ptr %in0, align 8
; AVX512: Cost of 14.5 for VF 2: INTERLEAVE-GROUP with factor 5, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-6.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-6.ll
index 909e0a9c8ed182..4676e641166dd2 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-6.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-6.ll
@@ -13,14 +13,12 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
-; SSE2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load double, ptr %in0, align 8
; SSE2: Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 6 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 12 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 24 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load double, ptr %in0, align 8
; AVX1: Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 14 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -28,7 +26,6 @@ define void @test() {
; AVX1: Cost of 56 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
-; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load double, ptr %in0, align 8
; AVX2: Cost of 9 for VF 2: INTERLEAVE-GROUP with factor 6, ir<%in0>
; AVX2: ir<%v0> = load from index 0
; AVX2: ir<%v1> = load from index 1
@@ -54,7 +51,6 @@ define void @test() {
; AVX2: Cost of 56 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX512-LABEL: 'test'
-; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load double, ptr %in0, align 8
; AVX512: Cost of 17 for VF 2: INTERLEAVE-GROUP with factor 6, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-7.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-7.ll
index a9cef6bd260a1a..fbbc94de0c1a91 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-7.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-7.ll
@@ -13,14 +13,12 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
-; SSE2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load double, ptr %in0, align 8
; SSE2: Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 6 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 12 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 24 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load double, ptr %in0, align 8
; AVX1: Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 14 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -28,7 +26,6 @@ define void @test() {
; AVX1: Cost of 56 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
-; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load double, ptr %in0, align 8
; AVX2: Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 14 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -36,7 +33,6 @@ define void @test() {
; AVX2: Cost of 56 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX512-LABEL: 'test'
-; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load double, ptr %in0, align 8
; AVX512: Cost of 19.5 for VF 2: INTERLEAVE-GROUP with factor 7, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-8.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-8.ll
index ba511ad467b1aa..7c22d3e51be763 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-8.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-8.ll
@@ -13,14 +13,12 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
-; SSE2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load double, ptr %in0, align 8
; SSE2: Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 6 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 12 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 24 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load double, ptr %in0, align 8
; AVX1: Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 14 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -28,7 +26,6 @@ define void @test() {
; AVX1: Cost of 56 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
-; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load double, ptr %in0, align 8
; AVX2: Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 14 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -36,7 +33,6 @@ define void @test() {
; AVX2: Cost of 56 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX512-LABEL: 'test'
-; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load double, ptr %in0, align 8
; AVX512: Cost of 22 for VF 2: INTERLEAVE-GROUP with factor 8, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-2.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-2.ll
index acf9c541d9edf2..4fe1ae7f140c47 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-2.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-2.ll
@@ -14,7 +14,6 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
-; SSE2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
; SSE2: Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
; SSE2: ir<%v0> = load from index 0
; SSE2: ir<%v1> = load from index 1
@@ -25,7 +24,6 @@ define void @test() {
; SSE2: Cost of 32 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
; AVX1: Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
; AVX1: ir<%v0> = load from index 0
; AVX1: ir<%v1> = load from index 1
@@ -37,7 +35,6 @@ define void @test() {
; AVX1: Cost of 66 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
-; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
; AVX2: Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
; AVX2: ir<%v0> = load from index 0
; AVX2: ir<%v1> = load from index 1
@@ -55,7 +52,6 @@ define void @test() {
; AVX2: ir<%v1> = load from index 1
;
; AVX512DQ-LABEL: 'test'
-; AVX512DQ: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
; AVX512DQ: Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
; AVX512DQ: ir<%v0> = load from index 0
; AVX512DQ: ir<%v1> = load from index 1
@@ -76,7 +72,6 @@ define void @test() {
; AVX512DQ: ir<%v1> = load from index 1
;
; AVX512BW-LABEL: 'test'
-; AVX512BW: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
; AVX512BW: Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
; AVX512BW: ir<%v0> = load from index 0
; AVX512BW: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-3.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-3.ll
index 156bd432a34477..a7c0f2516d5ed9 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-3.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-3.ll
@@ -14,14 +14,12 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
-; SSE2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
; SSE2: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 32 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
; AVX1: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -29,7 +27,6 @@ define void @test() {
; AVX1: Cost of 66 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
-; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
; AVX2: Cost of 7 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
; AVX2: ir<%v0> = load from index 0
; AVX2: ir<%v1> = load from index 1
@@ -52,7 +49,6 @@ define void @test() {
; AVX2: ir<%v2> = load from index 2
;
; AVX512DQ-LABEL: 'test'
-; AVX512DQ: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
; AVX512DQ: Cost of 7 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
; AVX512DQ: ir<%v0> = load from index 0
; AVX512DQ: ir<%v1> = load from index 1
@@ -79,7 +75,6 @@ define void @test() {
; AVX512DQ: ir<%v2> = load from index 2
;
; AVX512BW-LABEL: 'test'
-; AVX512BW: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
; AVX512BW: Cost of 4 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
; AVX512BW: ir<%v0> = load from index 0
; AVX512BW: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-4.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-4.ll
index 8ceb2529fed177..71bae49cbd9061 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-4.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-4.ll
@@ -14,14 +14,12 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
-; SSE2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
; SSE2: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 32 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
; AVX1: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -29,7 +27,6 @@ define void @test() {
; AVX1: Cost of 66 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
-; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
; AVX2: Cost of 7 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
; AVX2: ir<%v0> = load from index 0
; AVX2: ir<%v1> = load from index 1
@@ -57,7 +54,6 @@ define void @test() {
; AVX2: ir<%v3> = load from index 3
;
; AVX512DQ-LABEL: 'test'
-; AVX512DQ: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
; AVX512DQ: Cost of 7 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
; AVX512DQ: ir<%v0> = load from index 0
; AVX512DQ: ir<%v1> = load from index 1
@@ -90,7 +86,6 @@ define void @test() {
; AVX512DQ: ir<%v3> = load from index 3
;
; AVX512BW-LABEL: 'test'
-; AVX512BW: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
; AVX512BW: Cost of 5 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
; AVX512BW: ir<%v0> = load from index 0
; AVX512BW: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-5.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-5.ll
index a7b6bf588af099..14c75eba680e9b 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-5.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-5.ll
@@ -14,14 +14,12 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
-; SSE2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
; SSE2: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 32 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
; AVX1: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -29,7 +27,6 @@ define void @test() {
; AVX1: Cost of 66 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
-; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
; AVX2: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -37,7 +34,6 @@ define void @test() {
; AVX2: Cost of 66 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX512DQ-LABEL: 'test'
-; AVX512DQ: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
; AVX512DQ: Cost of 24 for VF 2: INTERLEAVE-GROUP with factor 5, ir<%in0>
; AVX512DQ: ir<%v0> = load from index 0
; AVX512DQ: ir<%v1> = load from index 1
@@ -76,7 +72,6 @@ define void @test() {
; AVX512DQ: ir<%v4> = load from index 4
;
; AVX512BW-LABEL: 'test'
-; AVX512BW: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
; AVX512BW: Cost of 11 for VF 2: INTERLEAVE-GROUP with factor 5, ir<%in0>
; AVX512BW: ir<%v0> = load from index 0
; AVX512BW: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-6.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-6.ll
index 65da1d5767c42c..727eec70f83873 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-6.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-6.ll
@@ -14,14 +14,12 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
-; SSE2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
; SSE2: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 32 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
; AVX1: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -29,7 +27,6 @@ define void @test() {
; AVX1: Cost of 66 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
-; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
; AVX2: Cost of 16 for VF 2: INTERLEAVE-GROUP with factor 6, ir<%in0>
; AVX2: ir<%v0> = load from index 0
; AVX2: ir<%v1> = load from index 1
@@ -67,7 +64,6 @@ define void @test() {
; AVX2: ir<%v5> = load from index 5
;
; AVX512DQ-LABEL: 'test'
-; AVX512DQ: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
; AVX512DQ: Cost of 16 for VF 2: INTERLEAVE-GROUP with factor 6, ir<%in0>
; AVX512DQ: ir<%v0> = load from index 0
; AVX512DQ: ir<%v1> = load from index 1
@@ -112,7 +108,6 @@ define void @test() {
; AVX512DQ: ir<%v5> = load from index 5
;
; AVX512BW-LABEL: 'test'
-; AVX512BW: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
; AVX512BW: Cost of 13 for VF 2: INTERLEAVE-GROUP with factor 6, ir<%in0>
; AVX512BW: ir<%v0> = load from index 0
; AVX512BW: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-7.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-7.ll
index 26e82797634251..d593db1853fe8f 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-7.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-7.ll
@@ -14,14 +14,12 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
-; SSE2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
; SSE2: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 32 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
; AVX1: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -29,7 +27,6 @@ define void @test() {
; AVX1: Cost of 66 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
-; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
; AVX2: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -37,7 +34,6 @@ define void @test() {
; AVX2: Cost of 66 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX512DQ-LABEL: 'test'
-; AVX512DQ: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
; AVX512DQ: Cost of 33 for VF 2: INTERLEAVE-GROUP with factor 7, ir<%in0>
; AVX512DQ: ir<%v0> = load from index 0
; AVX512DQ: ir<%v1> = load from index 1
@@ -88,7 +84,6 @@ define void @test() {
; AVX512DQ: ir<%v6> = load from index 6
;
; AVX512BW-LABEL: 'test'
-; AVX512BW: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
; AVX512BW: Cost of 15 for VF 2: INTERLEAVE-GROUP with factor 7, ir<%in0>
; AVX512BW: ir<%v0> = load from index 0
; AVX512BW: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-8.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-8.ll
index 759f45629adf1d..5ef380a3032da3 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-8.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-8.ll
@@ -14,14 +14,12 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
-; SSE2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
; SSE2: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 32 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
; AVX1: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -29,7 +27,6 @@ define void @test() {
; AVX1: Cost of 66 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
-; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
; AVX2: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -37,7 +34,6 @@ define void @test() {
; AVX2: Cost of 66 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX512DQ-LABEL: 'test'
-; AVX512DQ: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
; AVX512DQ: Cost of 34 for VF 2: INTERLEAVE-GROUP with factor 8, ir<%in0>
; AVX512DQ: ir<%v0> = load from index 0
; AVX512DQ: ir<%v1> = load from index 1
@@ -94,7 +90,6 @@ define void @test() {
; AVX512DQ: ir<%v7> = load from index 7
;
; AVX512BW-LABEL: 'test'
-; AVX512BW: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i16, ptr %in0, align 2
; AVX512BW: Cost of 17 for VF 2: INTERLEAVE-GROUP with factor 8, ir<%in0>
; AVX512BW: ir<%v0> = load from index 0
; AVX512BW: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-2-indices-0u.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-2-indices-0u.ll
index 3d0e795d653a46..85043e46d2023f 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-2-indices-0u.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-2-indices-0u.ll
@@ -13,7 +13,6 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
-; SSE2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
; SSE2: Cost of 2 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
; SSE2: ir<%v0> = load from index 0
; SSE2: Cost of 3 for VF 4: INTERLEAVE-GROUP with factor 2, ir<%in0>
@@ -22,7 +21,6 @@ define void @test() {
; SSE2: Cost of 44 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
; AVX1: Cost of 2 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
; AVX1: ir<%v0> = load from index 0
; AVX1: Cost of 2 for VF 4: INTERLEAVE-GROUP with factor 2, ir<%in0>
@@ -32,7 +30,6 @@ define void @test() {
; AVX1: Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
-; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
; AVX2: Cost of 2 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
; AVX2: ir<%v0> = load from index 0
; AVX2: Cost of 2 for VF 4: INTERLEAVE-GROUP with factor 2, ir<%in0>
@@ -45,7 +42,6 @@ define void @test() {
; AVX2: ir<%v0> = load from index 0
;
; AVX512-LABEL: 'test'
-; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
; AVX512: Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: Cost of 1 for VF 4: INTERLEAVE-GROUP with factor 2, ir<%in0>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-2.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-2.ll
index 302e0cb447d653..3d8916969eade1 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-2.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-2.ll
@@ -13,7 +13,6 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
-; SSE2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
; SSE2: Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
; SSE2: ir<%v0> = load from index 0
; SSE2: ir<%v1> = load from index 1
@@ -24,7 +23,6 @@ define void @test() {
; SSE2: Cost of 44 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
; AVX1: Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
; AVX1: ir<%v0> = load from index 0
; AVX1: ir<%v1> = load from index 1
@@ -36,7 +34,6 @@ define void @test() {
; AVX1: Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
-; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
; AVX2: Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
; AVX2: ir<%v0> = load from index 0
; AVX2: ir<%v1> = load from index 1
@@ -54,7 +51,6 @@ define void @test() {
; AVX2: ir<%v1> = load from index 1
;
; AVX512-LABEL: 'test'
-; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
; AVX512: Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-3-indices-01u.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-3-indices-01u.ll
index e843b716aa01b2..02d90c5ffe2fbb 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-3-indices-01u.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-3-indices-01u.ll
@@ -13,14 +13,12 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
-; SSE2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
; SSE2: Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 22 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 44 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
; AVX1: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 17 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -28,7 +26,6 @@ define void @test() {
; AVX1: Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
-; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
; AVX2: Cost of 5 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
; AVX2: ir<%v0> = load from index 0
; AVX2: ir<%v1> = load from index 1
@@ -46,7 +43,6 @@ define void @test() {
; AVX2: ir<%v1> = load from index 1
;
; AVX512-LABEL: 'test'
-; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
; AVX512: Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-3-indices-0uu.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-3-indices-0uu.ll
index f40ad60dc6f3c8..e92c161c8ff960 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-3-indices-0uu.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-3-indices-0uu.ll
@@ -13,14 +13,12 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
-; SSE2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
; SSE2: Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 22 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 44 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
; AVX1: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 17 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -28,7 +26,6 @@ define void @test() {
; AVX1: Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
-; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
; AVX2: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 3 for VF 4: INTERLEAVE-GROUP with factor 3, ir<%in0>
; AVX2: ir<%v0> = load from index 0
@@ -40,7 +37,6 @@ define void @test() {
; AVX2: ir<%v0> = load from index 0
;
; AVX512-LABEL: 'test'
-; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
; AVX512: Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: Cost of 1 for VF 4: INTERLEAVE-GROUP with factor 3, ir<%in0>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-3.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-3.ll
index c0de2fcacd4123..b68f44855dc3ed 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-3.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-3.ll
@@ -13,14 +13,12 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
-; SSE2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
; SSE2: Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 22 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 44 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
; AVX1: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 17 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -28,7 +26,6 @@ define void @test() {
; AVX1: Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
-; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
; AVX2: Cost of 6 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
; AVX2: ir<%v0> = load from index 0
; AVX2: ir<%v1> = load from index 1
@@ -51,7 +48,6 @@ define void @test() {
; AVX2: ir<%v2> = load from index 2
;
; AVX512-LABEL: 'test'
-; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
; AVX512: Cost of 4 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-4-indices-012u.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-4-indices-012u.ll
index 41bd92713c3861..69baf083073e40 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-4-indices-012u.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-4-indices-012u.ll
@@ -13,14 +13,12 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
-; SSE2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
; SSE2: Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 22 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 44 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
; AVX1: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 17 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -28,7 +26,6 @@ define void @test() {
; AVX1: Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
-; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
; AVX2: Cost of 4 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
; AVX2: ir<%v0> = load from index 0
; AVX2: ir<%v1> = load from index 1
@@ -51,7 +48,6 @@ define void @test() {
; AVX2: ir<%v2> = load from index 2
;
; AVX512-LABEL: 'test'
-; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
; AVX512: Cost of 4 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-4-indices-01uu.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-4-indices-01uu.ll
index 9a8ae47d519071..43e097a4ffeb01 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-4-indices-01uu.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-4-indices-01uu.ll
@@ -13,14 +13,12 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
-; SSE2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
; SSE2: Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 22 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 44 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
; AVX1: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 17 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -28,7 +26,6 @@ define void @test() {
; AVX1: Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
-; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
; AVX2: Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
; AVX2: ir<%v0> = load from index 0
; AVX2: ir<%v1> = load from index 1
@@ -46,7 +43,6 @@ define void @test() {
; AVX2: ir<%v1> = load from index 1
;
; AVX512-LABEL: 'test'
-; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
; AVX512: Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-4-indices-0uuu.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-4-indices-0uuu.ll
index d2e518a981de51..e3838488d3a967 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-4-indices-0uuu.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-4-indices-0uuu.ll
@@ -13,14 +13,12 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
-; SSE2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
; SSE2: Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 22 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 44 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
; AVX1: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 17 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -28,7 +26,6 @@ define void @test() {
; AVX1: Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
-; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
; AVX2: Cost of 2 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
; AVX2: ir<%v0> = load from index 0
; AVX2: Cost of 4 for VF 4: INTERLEAVE-GROUP with factor 4, ir<%in0>
@@ -41,7 +38,6 @@ define void @test() {
; AVX2: ir<%v0> = load from index 0
;
; AVX512-LABEL: 'test'
-; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
; AVX512: Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: Cost of 1 for VF 4: INTERLEAVE-GROUP with factor 4, ir<%in0>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-4.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-4.ll
index bd7a63609acde1..bdb68363cdc2d8 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-4.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-4.ll
@@ -13,14 +13,12 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
-; SSE2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
; SSE2: Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 22 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 44 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
; AVX1: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 17 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -28,7 +26,6 @@ define void @test() {
; AVX1: Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
-; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
; AVX2: Cost of 5 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
; AVX2: ir<%v0> = load from index 0
; AVX2: ir<%v1> = load from index 1
@@ -56,7 +53,6 @@ define void @test() {
; AVX2: ir<%v3> = load from index 3
;
; AVX512-LABEL: 'test'
-; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
; AVX512: Cost of 5 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-5.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-5.ll
index cb315e61e57a2c..7f5b1687b59780 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-5.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-5.ll
@@ -13,14 +13,12 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
-; SSE2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
; SSE2: Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 22 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 44 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
; AVX1: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 17 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -28,7 +26,6 @@ define void @test() {
; AVX1: Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
-; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
; AVX2: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 17 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -36,7 +33,6 @@ define void @test() {
; AVX2: Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX512-LABEL: 'test'
-; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
; AVX512: Cost of 6 for VF 2: INTERLEAVE-GROUP with factor 5, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-6.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-6.ll
index c916394a0c611e..aec171a16f12ba 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-6.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-6.ll
@@ -13,14 +13,12 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
-; SSE2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
; SSE2: Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 22 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 44 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
; AVX1: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 17 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -28,7 +26,6 @@ define void @test() {
; AVX1: Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
-; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
; AVX2: Cost of 8 for VF 2: INTERLEAVE-GROUP with factor 6, ir<%in0>
; AVX2: ir<%v0> = load from index 0
; AVX2: ir<%v1> = load from index 1
@@ -60,7 +57,6 @@ define void @test() {
; AVX2: Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX512-LABEL: 'test'
-; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
; AVX512: Cost of 7 for VF 2: INTERLEAVE-GROUP with factor 6, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-7.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-7.ll
index 0c2cc5f663ef3e..5d4904aec87eb7 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-7.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-7.ll
@@ -13,14 +13,12 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
-; SSE2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
; SSE2: Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 22 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 44 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
; AVX1: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 17 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -28,7 +26,6 @@ define void @test() {
; AVX1: Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
-; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
; AVX2: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 17 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -36,7 +33,6 @@ define void @test() {
; AVX2: Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX512-LABEL: 'test'
-; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
; AVX512: Cost of 8 for VF 2: INTERLEAVE-GROUP with factor 7, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-8.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-8.ll
index c64076fca0d62d..e1af00bcbe1fa2 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-8.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-8.ll
@@ -13,14 +13,12 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
-; SSE2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
; SSE2: Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 22 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 44 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
; AVX1: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 17 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -28,7 +26,6 @@ define void @test() {
; AVX1: Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
-; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
; AVX2: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 48 for VF 8: INTERLEAVE-GROUP with factor 8, ir<%in0>
@@ -44,7 +41,6 @@ define void @test() {
; AVX2: Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX512-LABEL: 'test'
-; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i32, ptr %in0, align 4
; AVX512: Cost of 9 for VF 2: INTERLEAVE-GROUP with factor 8, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-2.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-2.ll
index d0a99efab706d9..a664eb308630b7 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-2.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-2.ll
@@ -13,7 +13,6 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
-; SSE2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i64, ptr %in0, align 8
; SSE2: Cost of 4 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
; SSE2: ir<%v0> = load from index 0
; SSE2: ir<%v1> = load from index 1
@@ -22,7 +21,6 @@ define void @test() {
; SSE2: Cost of 40 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i64, ptr %in0, align 8
; AVX1: Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
; AVX1: ir<%v0> = load from index 0
; AVX1: ir<%v1> = load from index 1
@@ -32,7 +30,6 @@ define void @test() {
; AVX1: Cost of 72 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
-; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i64, ptr %in0, align 8
; AVX2: Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
; AVX2: ir<%v0> = load from index 0
; AVX2: ir<%v1> = load from index 1
@@ -50,7 +47,6 @@ define void @test() {
; AVX2: ir<%v1> = load from index 1
;
; AVX512-LABEL: 'test'
-; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i64, ptr %in0, align 8
; AVX512: Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-3.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-3.ll
index 17cc63d11eb402..1fbcf7559aad64 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-3.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-3.ll
@@ -13,14 +13,12 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
-; SSE2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i64, ptr %in0, align 8
; SSE2: Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 10 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 20 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 40 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i64, ptr %in0, align 8
; AVX1: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 9 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 18 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -28,7 +26,6 @@ define void @test() {
; AVX1: Cost of 72 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
-; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i64, ptr %in0, align 8
; AVX2: Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
; AVX2: ir<%v0> = load from index 0
; AVX2: ir<%v1> = load from index 1
@@ -48,7 +45,6 @@ define void @test() {
; AVX2: Cost of 72 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX512-LABEL: 'test'
-; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i64, ptr %in0, align 8
; AVX512: Cost of 4 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-4.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-4.ll
index 180ef142675f52..43adad539acd4d 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-4.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-4.ll
@@ -13,14 +13,12 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
-; SSE2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i64, ptr %in0, align 8
; SSE2: Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 10 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 20 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 40 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i64, ptr %in0, align 8
; AVX1: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 9 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 18 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -28,7 +26,6 @@ define void @test() {
; AVX1: Cost of 72 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
-; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i64, ptr %in0, align 8
; AVX2: Cost of 8 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
; AVX2: ir<%v0> = load from index 0
; AVX2: ir<%v1> = load from index 1
@@ -52,7 +49,6 @@ define void @test() {
; AVX2: Cost of 72 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX512-LABEL: 'test'
-; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i64, ptr %in0, align 8
; AVX512: Cost of 5 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-5.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-5.ll
index 65a27fc96f2234..2c6c071b77be2f 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-5.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-5.ll
@@ -13,14 +13,12 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
-; SSE2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i64, ptr %in0, align 8
; SSE2: Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 10 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 20 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 40 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i64, ptr %in0, align 8
; AVX1: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 9 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 18 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -28,7 +26,6 @@ define void @test() {
; AVX1: Cost of 72 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
-; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i64, ptr %in0, align 8
; AVX2: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 9 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 18 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -36,7 +33,6 @@ define void @test() {
; AVX2: Cost of 72 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX512-LABEL: 'test'
-; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i64, ptr %in0, align 8
; AVX512: Cost of 14.5 for VF 2: INTERLEAVE-GROUP with factor 5, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-6.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-6.ll
index d261e5242f8d86..18c252a8a3aaca 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-6.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-6.ll
@@ -13,14 +13,12 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
-; SSE2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i64, ptr %in0, align 8
; SSE2: Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 10 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 20 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 40 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i64, ptr %in0, align 8
; AVX1: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 9 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 18 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -28,7 +26,6 @@ define void @test() {
; AVX1: Cost of 72 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
-; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i64, ptr %in0, align 8
; AVX2: Cost of 9 for VF 2: INTERLEAVE-GROUP with factor 6, ir<%in0>
; AVX2: ir<%v0> = load from index 0
; AVX2: ir<%v1> = load from index 1
@@ -54,7 +51,6 @@ define void @test() {
; AVX2: Cost of 72 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX512-LABEL: 'test'
-; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i64, ptr %in0, align 8
; AVX512: Cost of 17 for VF 2: INTERLEAVE-GROUP with factor 6, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-7.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-7.ll
index 04c8db7a583572..6460df78c20c02 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-7.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-7.ll
@@ -13,14 +13,12 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
-; SSE2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i64, ptr %in0, align 8
; SSE2: Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 10 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 20 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 40 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i64, ptr %in0, align 8
; AVX1: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 9 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 18 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -28,7 +26,6 @@ define void @test() {
; AVX1: Cost of 72 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
-; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i64, ptr %in0, align 8
; AVX2: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 9 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 18 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -36,7 +33,6 @@ define void @test() {
; AVX2: Cost of 72 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX512-LABEL: 'test'
-; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i64, ptr %in0, align 8
; AVX512: Cost of 19.5 for VF 2: INTERLEAVE-GROUP with factor 7, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-8.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-8.ll
index a2b5d1c4df2f83..d4d99b530f0108 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-8.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-8.ll
@@ -13,14 +13,12 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
-; SSE2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i64, ptr %in0, align 8
; SSE2: Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 10 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 20 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 40 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i64, ptr %in0, align 8
; AVX1: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 9 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 18 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -28,7 +26,6 @@ define void @test() {
; AVX1: Cost of 72 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
-; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i64, ptr %in0, align 8
; AVX2: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 9 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 18 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -36,7 +33,6 @@ define void @test() {
; AVX2: Cost of 72 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX512-LABEL: 'test'
-; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i64, ptr %in0, align 8
; AVX512: Cost of 22 for VF 2: INTERLEAVE-GROUP with factor 8, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-2.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-2.ll
index f6f057ca88ca63..4686fbc32153a9 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-2.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-2.ll
@@ -14,14 +14,12 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
-; SSE2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
; SSE2: Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 23 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 47 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
; AVX1: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -29,7 +27,6 @@ define void @test() {
; AVX1: Cost of 65 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
-; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
; AVX2: Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
; AVX2: ir<%v0> = load from index 0
; AVX2: ir<%v1> = load from index 1
@@ -47,7 +44,6 @@ define void @test() {
; AVX2: ir<%v1> = load from index 1
;
; AVX512DQ-LABEL: 'test'
-; AVX512DQ: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
; AVX512DQ: Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
; AVX512DQ: ir<%v0> = load from index 0
; AVX512DQ: ir<%v1> = load from index 1
@@ -68,7 +64,6 @@ define void @test() {
; AVX512DQ: ir<%v1> = load from index 1
;
; AVX512BW-LABEL: 'test'
-; AVX512BW: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
; AVX512BW: Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
; AVX512BW: ir<%v0> = load from index 0
; AVX512BW: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-3.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-3.ll
index f385a46789ecc2..d5bdfeacdb30e6 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-3.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-3.ll
@@ -14,14 +14,12 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
-; SSE2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
; SSE2: Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 23 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 47 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
; AVX1: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -29,7 +27,6 @@ define void @test() {
; AVX1: Cost of 65 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
-; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
; AVX2: Cost of 5 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
; AVX2: ir<%v0> = load from index 0
; AVX2: ir<%v1> = load from index 1
@@ -52,7 +49,6 @@ define void @test() {
; AVX2: ir<%v2> = load from index 2
;
; AVX512DQ-LABEL: 'test'
-; AVX512DQ: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
; AVX512DQ: Cost of 5 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
; AVX512DQ: ir<%v0> = load from index 0
; AVX512DQ: ir<%v1> = load from index 1
@@ -79,7 +75,6 @@ define void @test() {
; AVX512DQ: ir<%v2> = load from index 2
;
; AVX512BW-LABEL: 'test'
-; AVX512BW: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
; AVX512BW: Cost of 4 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
; AVX512BW: ir<%v0> = load from index 0
; AVX512BW: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-4.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-4.ll
index 13f02ae479c417..7390b2f56010f4 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-4.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-4.ll
@@ -14,14 +14,12 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
-; SSE2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
; SSE2: Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 23 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 47 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
; AVX1: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -29,7 +27,6 @@ define void @test() {
; AVX1: Cost of 65 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
-; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
; AVX2: Cost of 5 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
; AVX2: ir<%v0> = load from index 0
; AVX2: ir<%v1> = load from index 1
@@ -57,7 +54,6 @@ define void @test() {
; AVX2: ir<%v3> = load from index 3
;
; AVX512DQ-LABEL: 'test'
-; AVX512DQ: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
; AVX512DQ: Cost of 5 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
; AVX512DQ: ir<%v0> = load from index 0
; AVX512DQ: ir<%v1> = load from index 1
@@ -90,7 +86,6 @@ define void @test() {
; AVX512DQ: ir<%v3> = load from index 3
;
; AVX512BW-LABEL: 'test'
-; AVX512BW: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
; AVX512BW: Cost of 5 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
; AVX512BW: ir<%v0> = load from index 0
; AVX512BW: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-5.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-5.ll
index 490ac46b0b4ac9..3fd3b9e11017d6 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-5.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-5.ll
@@ -14,14 +14,12 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
-; SSE2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
; SSE2: Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 23 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 47 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
; AVX1: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -29,7 +27,6 @@ define void @test() {
; AVX1: Cost of 65 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
-; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
; AVX2: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -37,7 +34,6 @@ define void @test() {
; AVX2: Cost of 65 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX512DQ-LABEL: 'test'
-; AVX512DQ: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
; AVX512DQ: Cost of 22 for VF 2: INTERLEAVE-GROUP with factor 5, ir<%in0>
; AVX512DQ: ir<%v0> = load from index 0
; AVX512DQ: ir<%v1> = load from index 1
@@ -76,7 +72,6 @@ define void @test() {
; AVX512DQ: ir<%v4> = load from index 4
;
; AVX512BW-LABEL: 'test'
-; AVX512BW: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
; AVX512BW: Cost of 6 for VF 2: INTERLEAVE-GROUP with factor 5, ir<%in0>
; AVX512BW: ir<%v0> = load from index 0
; AVX512BW: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-6.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-6.ll
index 08c5f31fbd5d03..d8b77269cda803 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-6.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-6.ll
@@ -14,14 +14,12 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
-; SSE2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
; SSE2: Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 23 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 47 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
; AVX1: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -29,7 +27,6 @@ define void @test() {
; AVX1: Cost of 65 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
-; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
; AVX2: Cost of 8 for VF 2: INTERLEAVE-GROUP with factor 6, ir<%in0>
; AVX2: ir<%v0> = load from index 0
; AVX2: ir<%v1> = load from index 1
@@ -67,7 +64,6 @@ define void @test() {
; AVX2: ir<%v5> = load from index 5
;
; AVX512DQ-LABEL: 'test'
-; AVX512DQ: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
; AVX512DQ: Cost of 8 for VF 2: INTERLEAVE-GROUP with factor 6, ir<%in0>
; AVX512DQ: ir<%v0> = load from index 0
; AVX512DQ: ir<%v1> = load from index 1
@@ -112,7 +108,6 @@ define void @test() {
; AVX512DQ: ir<%v5> = load from index 5
;
; AVX512BW-LABEL: 'test'
-; AVX512BW: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
; AVX512BW: Cost of 7 for VF 2: INTERLEAVE-GROUP with factor 6, ir<%in0>
; AVX512BW: ir<%v0> = load from index 0
; AVX512BW: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-7.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-7.ll
index afbe3c05bfb9e4..e2e3dedca27c8d 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-7.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-7.ll
@@ -14,14 +14,12 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
-; SSE2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
; SSE2: Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 23 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 47 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
; AVX1: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -29,7 +27,6 @@ define void @test() {
; AVX1: Cost of 65 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
-; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
; AVX2: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -37,7 +34,6 @@ define void @test() {
; AVX2: Cost of 65 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX512DQ-LABEL: 'test'
-; AVX512DQ: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
; AVX512DQ: Cost of 31 for VF 2: INTERLEAVE-GROUP with factor 7, ir<%in0>
; AVX512DQ: ir<%v0> = load from index 0
; AVX512DQ: ir<%v1> = load from index 1
@@ -88,7 +84,6 @@ define void @test() {
; AVX512DQ: ir<%v6> = load from index 6
;
; AVX512BW-LABEL: 'test'
-; AVX512BW: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
; AVX512BW: Cost of 8 for VF 2: INTERLEAVE-GROUP with factor 7, ir<%in0>
; AVX512BW: ir<%v0> = load from index 0
; AVX512BW: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-8.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-8.ll
index 11bcbd083c59f6..1fd2c4898762d9 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-8.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-8.ll
@@ -14,14 +14,12 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
-; SSE2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
; SSE2: Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 23 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 47 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
; AVX1: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -29,7 +27,6 @@ define void @test() {
; AVX1: Cost of 65 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
-; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
; AVX2: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -37,7 +34,6 @@ define void @test() {
; AVX2: Cost of 65 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX512DQ-LABEL: 'test'
-; AVX512DQ: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
; AVX512DQ: Cost of 33 for VF 2: INTERLEAVE-GROUP with factor 8, ir<%in0>
; AVX512DQ: ir<%v0> = load from index 0
; AVX512DQ: ir<%v1> = load from index 1
@@ -94,7 +90,6 @@ define void @test() {
; AVX512DQ: ir<%v7> = load from index 7
;
; AVX512BW-LABEL: 'test'
-; AVX512BW: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load i8, ptr %in0, align 1
; AVX512BW: Cost of 9 for VF 2: INTERLEAVE-GROUP with factor 8, ir<%in0>
; AVX512BW: ir<%v0> = load from index 0
; AVX512BW: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-gather-i32-with-i8-index.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-gather-i32-with-i8-index.ll
index 8f5da77027970f..3d33bc34dda674 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-gather-i32-with-i8-index.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-gather-i32-with-i8-index.ll
@@ -17,14 +17,14 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE-LABEL: 'test'
-; SSE: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i32, ptr %inB, align 4
+; SSE: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
; SSE: Cost of 3000000 for VF 2: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
; SSE: Cost of 3000000 for VF 4: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
; SSE: Cost of 3000000 for VF 8: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
; SSE: Cost of 3000000 for VF 16: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i32, ptr %inB, align 4
+; AVX1: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
; AVX1: Cost of 3000000 for VF 2: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
; AVX1: Cost of 3000000 for VF 4: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
; AVX1: Cost of 3000000 for VF 8: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
@@ -32,7 +32,7 @@ define void @test() {
; AVX1: Cost of 3000000 for VF 32: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
;
; AVX2-SLOWGATHER-LABEL: 'test'
-; AVX2-SLOWGATHER: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i32, ptr %inB, align 4
+; AVX2-SLOWGATHER: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
; AVX2-SLOWGATHER: Cost of 3000000 for VF 2: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
; AVX2-SLOWGATHER: Cost of 3000000 for VF 4: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
; AVX2-SLOWGATHER: Cost of 3000000 for VF 8: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
@@ -40,21 +40,21 @@ define void @test() {
; AVX2-SLOWGATHER: Cost of 3000000 for VF 32: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
;
; AVX2-FASTGATHER-LABEL: 'test'
-; AVX2-FASTGATHER: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i32, ptr %inB, align 4
-; AVX2-FASTGATHER: Cost of 4 for VF 2: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
-; AVX2-FASTGATHER: Cost of 6 for VF 4: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
-; AVX2-FASTGATHER: Cost of 12 for VF 8: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
-; AVX2-FASTGATHER: Cost of 24 for VF 16: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
-; AVX2-FASTGATHER: Cost of 48 for VF 32: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
+; AVX2-FASTGATHER: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX2-FASTGATHER: Cost of 4 for VF 2: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX2-FASTGATHER: Cost of 6 for VF 4: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX2-FASTGATHER: Cost of 12 for VF 8: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX2-FASTGATHER: Cost of 24 for VF 16: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX2-FASTGATHER: Cost of 48 for VF 32: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
;
; AVX512-LABEL: 'test'
-; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i32, ptr %inB, align 4
-; AVX512: Cost of 8 for VF 2: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
-; AVX512: Cost of 17 for VF 4: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
-; AVX512: Cost of 10 for VF 8: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
-; AVX512: Cost of 18 for VF 16: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
-; AVX512: Cost of 36 for VF 32: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
-; AVX512: Cost of 72 for VF 64: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
+; AVX512: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX512: Cost of 8 for VF 2: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX512: Cost of 17 for VF 4: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX512: Cost of 10 for VF 8: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX512: Cost of 18 for VF 16: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX512: Cost of 36 for VF 32: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX512: Cost of 72 for VF 64: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
;
entry:
br label %for.body
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-gather-i64-with-i8-index.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-gather-i64-with-i8-index.ll
index 6801a549e19c4f..b0fec98fcdf51d 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-gather-i64-with-i8-index.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-gather-i64-with-i8-index.ll
@@ -17,14 +17,14 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE-LABEL: 'test'
-; SSE: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i64, ptr %inB, align 8
+; SSE: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
; SSE: Cost of 3000000 for VF 2: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
; SSE: Cost of 3000000 for VF 4: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
; SSE: Cost of 3000000 for VF 8: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
; SSE: Cost of 3000000 for VF 16: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i64, ptr %inB, align 8
+; AVX1: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
; AVX1: Cost of 3000000 for VF 2: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
; AVX1: Cost of 3000000 for VF 4: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
; AVX1: Cost of 3000000 for VF 8: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
@@ -32,7 +32,7 @@ define void @test() {
; AVX1: Cost of 3000000 for VF 32: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
;
; AVX2-SLOWGATHER-LABEL: 'test'
-; AVX2-SLOWGATHER: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i64, ptr %inB, align 8
+; AVX2-SLOWGATHER: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
; AVX2-SLOWGATHER: Cost of 3000000 for VF 2: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
; AVX2-SLOWGATHER: Cost of 3000000 for VF 4: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
; AVX2-SLOWGATHER: Cost of 3000000 for VF 8: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
@@ -40,21 +40,21 @@ define void @test() {
; AVX2-SLOWGATHER: Cost of 3000000 for VF 32: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
;
; AVX2-FASTGATHER-LABEL: 'test'
-; AVX2-FASTGATHER: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i64, ptr %inB, align 8
-; AVX2-FASTGATHER: Cost of 4 for VF 2: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
-; AVX2-FASTGATHER: Cost of 6 for VF 4: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
-; AVX2-FASTGATHER: Cost of 12 for VF 8: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
-; AVX2-FASTGATHER: Cost of 24 for VF 16: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
-; AVX2-FASTGATHER: Cost of 48 for VF 32: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
+; AVX2-FASTGATHER: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX2-FASTGATHER: Cost of 4 for VF 2: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX2-FASTGATHER: Cost of 6 for VF 4: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX2-FASTGATHER: Cost of 12 for VF 8: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX2-FASTGATHER: Cost of 24 for VF 16: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX2-FASTGATHER: Cost of 48 for VF 32: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
;
; AVX512-LABEL: 'test'
-; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i64, ptr %inB, align 8
-; AVX512: Cost of 8 for VF 2: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
-; AVX512: Cost of 18 for VF 4: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
-; AVX512: Cost of 10 for VF 8: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
-; AVX512: Cost of 20 for VF 16: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
-; AVX512: Cost of 40 for VF 32: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
-; AVX512: Cost of 80 for VF 64: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
+; AVX512: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX512: Cost of 8 for VF 2: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX512: Cost of 18 for VF 4: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX512: Cost of 10 for VF 8: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX512: Cost of 20 for VF 16: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX512: Cost of 40 for VF 32: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX512: Cost of 80 for VF 64: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
;
entry:
br label %for.body
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-interleaved-load-i16.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-interleaved-load-i16.ll
index d14260279e17f9..52d05b807ba827 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-interleaved-load-i16.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-interleaved-load-i16.ll
@@ -20,8 +20,6 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test1(ptr noalias nocapture %points, ptr noalias nocapture readonly %x, ptr noalias nocapture readonly %y) {
; DISABLED_MASKED_STRIDED-LABEL: 'test1'
-; DISABLED_MASKED_STRIDED: LV: Found an estimated cost of 1 for VF 1 For instruction: %i2 = load i16, ptr %arrayidx2, align 2
-; DISABLED_MASKED_STRIDED: LV: Found an estimated cost of 1 for VF 1 For instruction: %i4 = load i16, ptr %arrayidx7, align 2
; DISABLED_MASKED_STRIDED: Cost of 6 for VF 2: REPLICATE ir<%i2> = load ir<%arrayidx2>
; DISABLED_MASKED_STRIDED: Cost of 6 for VF 2: REPLICATE ir<%i4> = load ir<%arrayidx7>
; DISABLED_MASKED_STRIDED: Cost of 13 for VF 4: REPLICATE ir<%i2> = load ir<%arrayidx2>
@@ -32,8 +30,6 @@ define void @test1(ptr noalias nocapture %points, ptr noalias nocapture readonly
; DISABLED_MASKED_STRIDED: Cost of 55 for VF 16: REPLICATE ir<%i4> = load ir<%arrayidx7>
;
; ENABLED_MASKED_STRIDED-LABEL: 'test1'
-; ENABLED_MASKED_STRIDED: LV: Found an estimated cost of 1 for VF 1 For instruction: %i2 = load i16, ptr %arrayidx2, align 2
-; ENABLED_MASKED_STRIDED: LV: Found an estimated cost of 1 for VF 1 For instruction: %i4 = load i16, ptr %arrayidx7, align 2
; ENABLED_MASKED_STRIDED: Cost of 8 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%arrayidx2>
; ENABLED_MASKED_STRIDED: ir<%i2> = load from index 0
; ENABLED_MASKED_STRIDED: ir<%i4> = load from index 1
@@ -81,8 +77,6 @@ for.end:
define void @test2(ptr noalias nocapture %points, i32 %numPoints, ptr noalias nocapture readonly %x, ptr noalias nocapture readonly %y) {
; DISABLED_MASKED_STRIDED-LABEL: 'test2'
-; DISABLED_MASKED_STRIDED: LV: Found an estimated cost of 1 for VF 1 For instruction: %i2 = load i16, ptr %arrayidx2, align 2
-; DISABLED_MASKED_STRIDED: LV: Found an estimated cost of 1 for VF 1 For instruction: %i4 = load i16, ptr %arrayidx7, align 2
; DISABLED_MASKED_STRIDED: Cost of 3000000 for VF 2: REPLICATE ir<%i2> = load ir<%arrayidx2> (S->V)
; DISABLED_MASKED_STRIDED: Cost of 3000000 for VF 2: REPLICATE ir<%i4> = load ir<%arrayidx7> (S->V)
; DISABLED_MASKED_STRIDED: Cost of 3000000 for VF 4: REPLICATE ir<%i2> = load ir<%arrayidx2> (S->V)
@@ -93,8 +87,6 @@ define void @test2(ptr noalias nocapture %points, i32 %numPoints, ptr noalias no
; DISABLED_MASKED_STRIDED: Cost of 3000000 for VF 16: REPLICATE ir<%i4> = load ir<%arrayidx7> (S->V)
;
; ENABLED_MASKED_STRIDED-LABEL: 'test2'
-; ENABLED_MASKED_STRIDED: LV: Found an estimated cost of 1 for VF 1 For instruction: %i2 = load i16, ptr %arrayidx2, align 2
-; ENABLED_MASKED_STRIDED: LV: Found an estimated cost of 1 for VF 1 For instruction: %i4 = load i16, ptr %arrayidx7, align 2
; ENABLED_MASKED_STRIDED: Cost of 8 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%arrayidx2>, vp<[[VP8:%[0-9]+]]>
; ENABLED_MASKED_STRIDED: ir<%i2> = load from index 0
; ENABLED_MASKED_STRIDED: ir<%i4> = load from index 1
@@ -152,24 +144,20 @@ for.end:
define void @test(ptr noalias nocapture %points, ptr noalias nocapture readonly %x, ptr noalias nocapture readnone %y) {
; DISABLED_MASKED_STRIDED-LABEL: 'test'
-; DISABLED_MASKED_STRIDED: LV: Found an estimated cost of 1 for VF 1 For instruction: %i2 = load i16, ptr %arrayidx, align 2
-; DISABLED_MASKED_STRIDED: LV: Found an estimated cost of 1 for VF 1 For instruction: %i4 = load i16, ptr %arrayidx6, align 2
; DISABLED_MASKED_STRIDED: Cost of 3000000 for VF 2: REPLICATE ir<%i4> = load ir<%arrayidx6> (S->V)
; DISABLED_MASKED_STRIDED: Cost of 3000000 for VF 4: REPLICATE ir<%i4> = load ir<%arrayidx6> (S->V)
; DISABLED_MASKED_STRIDED: Cost of 3000000 for VF 8: REPLICATE ir<%i4> = load ir<%arrayidx6> (S->V)
; DISABLED_MASKED_STRIDED: Cost of 3000000 for VF 16: REPLICATE ir<%i4> = load ir<%arrayidx6> (S->V)
;
; ENABLED_MASKED_STRIDED-LABEL: 'test'
-; ENABLED_MASKED_STRIDED: LV: Found an estimated cost of 1 for VF 1 For instruction: %i2 = load i16, ptr %arrayidx, align 2
-; ENABLED_MASKED_STRIDED: LV: Found an estimated cost of 1 for VF 1 For instruction: %i4 = load i16, ptr %arrayidx6, align 2
; ENABLED_MASKED_STRIDED: Cost of 7 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%arrayidx6>, ir<%cmp1>
-; ENABLED_MASKED_STRIDED: ir<%i4> = load from index 0
+; ENABLED_MASKED_STRIDED: ir<%i4> = load from index 0 (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
; ENABLED_MASKED_STRIDED: Cost of 9 for VF 4: INTERLEAVE-GROUP with factor 3, ir<%arrayidx6>, ir<%cmp1>
-; ENABLED_MASKED_STRIDED: ir<%i4> = load from index 0
+; ENABLED_MASKED_STRIDED: ir<%i4> = load from index 0 (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
; ENABLED_MASKED_STRIDED: Cost of 9 for VF 8: INTERLEAVE-GROUP with factor 3, ir<%arrayidx6>, ir<%cmp1>
-; ENABLED_MASKED_STRIDED: ir<%i4> = load from index 0
+; ENABLED_MASKED_STRIDED: ir<%i4> = load from index 0 (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
; ENABLED_MASKED_STRIDED: Cost of 14 for VF 16: INTERLEAVE-GROUP with factor 3, ir<%arrayidx6>, ir<%cmp1>
-; ENABLED_MASKED_STRIDED: ir<%i4> = load from index 0
+; ENABLED_MASKED_STRIDED: ir<%i4> = load from index 0 (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
;
entry:
br label %for.body
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-interleaved-store-i16.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-interleaved-store-i16.ll
index cd59ce8482f603..6b7b2919a9899e 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-interleaved-store-i16.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-interleaved-store-i16.ll
@@ -20,8 +20,6 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test1(ptr noalias nocapture %points, ptr noalias nocapture readonly %x, ptr noalias nocapture readonly %y) {
; DISABLED_MASKED_STRIDED-LABEL: 'test1'
-; DISABLED_MASKED_STRIDED: LV: Found an estimated cost of 1 for VF 1 For instruction: store i16 %0, ptr %arrayidx2, align 2
-; DISABLED_MASKED_STRIDED: LV: Found an estimated cost of 1 for VF 1 For instruction: store i16 %2, ptr %arrayidx7, align 2
; DISABLED_MASKED_STRIDED: Cost of 6 for VF 2: REPLICATE store ir<%0>, ir<%arrayidx2>
; DISABLED_MASKED_STRIDED: Cost of 6 for VF 2: REPLICATE store ir<%2>, ir<%arrayidx7>
; DISABLED_MASKED_STRIDED: Cost of 13 for VF 4: REPLICATE store ir<%0>, ir<%arrayidx2>
@@ -32,8 +30,6 @@ define void @test1(ptr noalias nocapture %points, ptr noalias nocapture readonly
; DISABLED_MASKED_STRIDED: Cost of 55 for VF 16: REPLICATE store ir<%2>, ir<%arrayidx7>
;
; ENABLED_MASKED_STRIDED-LABEL: 'test1'
-; ENABLED_MASKED_STRIDED: LV: Found an estimated cost of 1 for VF 1 For instruction: store i16 %0, ptr %arrayidx2, align 2
-; ENABLED_MASKED_STRIDED: LV: Found an estimated cost of 1 for VF 1 For instruction: store i16 %2, ptr %arrayidx7, align 2
; ENABLED_MASKED_STRIDED: Cost of 6 for VF 2: REPLICATE store ir<%0>, ir<%arrayidx2>
; ENABLED_MASKED_STRIDED: Cost of 6 for VF 2: REPLICATE store ir<%2>, ir<%arrayidx7>
; ENABLED_MASKED_STRIDED: Cost of 14 for VF 4: INTERLEAVE-GROUP with factor 4, ir<%arrayidx2>
@@ -74,8 +70,6 @@ for.end:
define void @test2(ptr noalias nocapture %points, i32 %numPoints, ptr noalias nocapture readonly %x, ptr noalias nocapture readonly %y) {
; DISABLED_MASKED_STRIDED-LABEL: 'test2'
-; DISABLED_MASKED_STRIDED: LV: Found an estimated cost of 1 for VF 1 For instruction: store i16 %0, ptr %arrayidx2, align 2
-; DISABLED_MASKED_STRIDED: LV: Found an estimated cost of 1 for VF 1 For instruction: store i16 %2, ptr %arrayidx7, align 2
; DISABLED_MASKED_STRIDED: Cost of 3000000 for VF 2: REPLICATE store ir<%0>, ir<%arrayidx2>
; DISABLED_MASKED_STRIDED: Cost of 3000000 for VF 2: REPLICATE store ir<%2>, ir<%arrayidx7>
; DISABLED_MASKED_STRIDED: Cost of 3000000 for VF 4: REPLICATE store ir<%0>, ir<%arrayidx2>
@@ -86,8 +80,6 @@ define void @test2(ptr noalias nocapture %points, i32 %numPoints, ptr noalias no
; DISABLED_MASKED_STRIDED: Cost of 3000000 for VF 16: REPLICATE store ir<%2>, ir<%arrayidx7>
;
; ENABLED_MASKED_STRIDED-LABEL: 'test2'
-; ENABLED_MASKED_STRIDED: LV: Found an estimated cost of 1 for VF 1 For instruction: store i16 %0, ptr %arrayidx2, align 2
-; ENABLED_MASKED_STRIDED: LV: Found an estimated cost of 1 for VF 1 For instruction: store i16 %2, ptr %arrayidx7, align 2
; ENABLED_MASKED_STRIDED: Cost of 13 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%arrayidx2>, vp<[[VP8:%[0-9]+]]>
; ENABLED_MASKED_STRIDED: Cost of 14 for VF 4: INTERLEAVE-GROUP with factor 4, ir<%arrayidx2>, vp<[[VP8]]>
; ENABLED_MASKED_STRIDED: Cost of 14 for VF 8: INTERLEAVE-GROUP with factor 4, ir<%arrayidx2>, vp<[[VP8]]>
@@ -137,14 +129,12 @@ for.end:
define void @test(ptr noalias nocapture %points, ptr noalias nocapture readonly %x, ptr noalias nocapture readnone %y) {
; DISABLED_MASKED_STRIDED-LABEL: 'test'
-; DISABLED_MASKED_STRIDED: LV: Found an estimated cost of 1 for VF 1 For instruction: store i16 %0, ptr %arrayidx6, align 2
; DISABLED_MASKED_STRIDED: Cost of 2 for VF 2: profitable to scalarize store i16 %0, ptr %arrayidx6, align 2
; DISABLED_MASKED_STRIDED: Cost of 4 for VF 4: profitable to scalarize store i16 %0, ptr %arrayidx6, align 2
; DISABLED_MASKED_STRIDED: Cost of 8 for VF 8: profitable to scalarize store i16 %0, ptr %arrayidx6, align 2
; DISABLED_MASKED_STRIDED: Cost of 16.5 for VF 16: profitable to scalarize store i16 %0, ptr %arrayidx6, align 2
;
; ENABLED_MASKED_STRIDED-LABEL: 'test'
-; ENABLED_MASKED_STRIDED: LV: Found an estimated cost of 1 for VF 1 For instruction: store i16 %0, ptr %arrayidx6, align 2
; ENABLED_MASKED_STRIDED: Cost of 2 for VF 2: profitable to scalarize store i16 %0, ptr %arrayidx6, align 2
; ENABLED_MASKED_STRIDED: Cost of 4 for VF 4: profitable to scalarize store i16 %0, ptr %arrayidx6, align 2
; ENABLED_MASKED_STRIDED: Cost of 12 for VF 8: INTERLEAVE-GROUP with factor 3, ir<%arrayidx6>, ir<%cmp1>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i16.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i16.ll
index 9cb0e482da47c9..a4a86cb40b3ea0 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i16.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i16.ll
@@ -16,14 +16,14 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test(ptr %B) {
; SSE-LABEL: 'test'
-; SSE: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i16, ptr %inB, align 2
+; SSE: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB>
; SSE: Cost of 3000000 for VF 2: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
; SSE: Cost of 3000000 for VF 4: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
; SSE: Cost of 3000000 for VF 8: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
; SSE: Cost of 3000000 for VF 16: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i16, ptr %inB, align 2
+; AVX1: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB>
; AVX1: Cost of 3000000 for VF 2: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
; AVX1: Cost of 3000000 for VF 4: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
; AVX1: Cost of 3000000 for VF 8: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
@@ -31,7 +31,7 @@ define void @test(ptr %B) {
; AVX1: Cost of 3000000 for VF 32: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
;
; AVX2-LABEL: 'test'
-; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i16, ptr %inB, align 2
+; AVX2: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB>
; AVX2: Cost of 3000000 for VF 2: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
; AVX2: Cost of 3000000 for VF 4: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
; AVX2: Cost of 3000000 for VF 8: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
@@ -39,7 +39,7 @@ define void @test(ptr %B) {
; AVX2: Cost of 3000000 for VF 32: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
;
; AVX512-LABEL: 'test'
-; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i16, ptr %inB, align 2
+; AVX512: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB>
; AVX512: Cost of 2 for VF 2: WIDEN ir<%valB.loaded> = load vp<[[VP6:%[0-9]+]]>, ir<%canLoad>
; AVX512: Cost of 2 for VF 4: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
; AVX512: Cost of 1 for VF 8: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i32.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i32.ll
index fc42ce6e6f73ff..be448f4b9359f3 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i32.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i32.ll
@@ -16,14 +16,14 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test(ptr %B) {
; SSE-LABEL: 'test'
-; SSE: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i32, ptr %inB, align 4
+; SSE: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB>
; SSE: Cost of 3000000 for VF 2: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
; SSE: Cost of 3000000 for VF 4: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
; SSE: Cost of 3000000 for VF 8: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
; SSE: Cost of 3000000 for VF 16: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i32, ptr %inB, align 4
+; AVX1: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB>
; AVX1: Cost of 3 for VF 2: WIDEN ir<%valB.loaded> = load vp<[[VP6:%[0-9]+]]>, ir<%canLoad>
; AVX1: Cost of 2 for VF 4: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
; AVX1: Cost of 2 for VF 8: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
@@ -31,7 +31,7 @@ define void @test(ptr %B) {
; AVX1: Cost of 8 for VF 32: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
;
; AVX2-LABEL: 'test'
-; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i32, ptr %inB, align 4
+; AVX2: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB>
; AVX2: Cost of 3 for VF 2: WIDEN ir<%valB.loaded> = load vp<[[VP6:%[0-9]+]]>, ir<%canLoad>
; AVX2: Cost of 2 for VF 4: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
; AVX2: Cost of 2 for VF 8: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
@@ -39,7 +39,7 @@ define void @test(ptr %B) {
; AVX2: Cost of 8 for VF 32: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
;
; AVX512-LABEL: 'test'
-; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i32, ptr %inB, align 4
+; AVX512: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB>
; AVX512: Cost of 2 for VF 2: WIDEN ir<%valB.loaded> = load vp<[[VP6:%[0-9]+]]>, ir<%canLoad>
; AVX512: Cost of 1 for VF 4: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
; AVX512: Cost of 1 for VF 8: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i64.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i64.ll
index 48c9b01beb8881..9900b8f2637def 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i64.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i64.ll
@@ -16,14 +16,14 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test(ptr %B) {
; SSE-LABEL: 'test'
-; SSE: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i64, ptr %inB, align 8
+; SSE: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB>
; SSE: Cost of 3000000 for VF 2: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
; SSE: Cost of 3000000 for VF 4: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
; SSE: Cost of 3000000 for VF 8: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
; SSE: Cost of 3000000 for VF 16: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i64, ptr %inB, align 8
+; AVX1: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB>
; AVX1: Cost of 2 for VF 2: WIDEN ir<%valB.loaded> = load vp<[[VP6:%[0-9]+]]>, ir<%canLoad>
; AVX1: Cost of 2 for VF 4: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
; AVX1: Cost of 4 for VF 8: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
@@ -31,7 +31,7 @@ define void @test(ptr %B) {
; AVX1: Cost of 16 for VF 32: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
;
; AVX2-LABEL: 'test'
-; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i64, ptr %inB, align 8
+; AVX2: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB>
; AVX2: Cost of 2 for VF 2: WIDEN ir<%valB.loaded> = load vp<[[VP6:%[0-9]+]]>, ir<%canLoad>
; AVX2: Cost of 2 for VF 4: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
; AVX2: Cost of 4 for VF 8: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
@@ -39,7 +39,7 @@ define void @test(ptr %B) {
; AVX2: Cost of 16 for VF 32: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
;
; AVX512-LABEL: 'test'
-; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i64, ptr %inB, align 8
+; AVX512: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB>
; AVX512: Cost of 1 for VF 2: WIDEN ir<%valB.loaded> = load vp<[[VP6:%[0-9]+]]>, ir<%canLoad>
; AVX512: Cost of 1 for VF 4: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
; AVX512: Cost of 1 for VF 8: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i8.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i8.ll
index c8225989777042..594d4766b43a4d 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i8.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i8.ll
@@ -16,14 +16,14 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test(ptr %B) {
; SSE-LABEL: 'test'
-; SSE: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i8, ptr %inB, align 1
+; SSE: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB>
; SSE: Cost of 3000000 for VF 2: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
; SSE: Cost of 3000000 for VF 4: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
; SSE: Cost of 3000000 for VF 8: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
; SSE: Cost of 3000000 for VF 16: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i8, ptr %inB, align 1
+; AVX1: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB>
; AVX1: Cost of 3000000 for VF 2: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
; AVX1: Cost of 3000000 for VF 4: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
; AVX1: Cost of 3000000 for VF 8: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
@@ -31,7 +31,7 @@ define void @test(ptr %B) {
; AVX1: Cost of 3000000 for VF 32: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
;
; AVX2-LABEL: 'test'
-; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i8, ptr %inB, align 1
+; AVX2: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB>
; AVX2: Cost of 3000000 for VF 2: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
; AVX2: Cost of 3000000 for VF 4: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
; AVX2: Cost of 3000000 for VF 8: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
@@ -39,7 +39,7 @@ define void @test(ptr %B) {
; AVX2: Cost of 3000000 for VF 32: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
;
; AVX512-LABEL: 'test'
-; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i8, ptr %inB, align 1
+; AVX512: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB>
; AVX512: Cost of 2 for VF 2: WIDEN ir<%valB.loaded> = load vp<[[VP6:%[0-9]+]]>, ir<%canLoad>
; AVX512: Cost of 2 for VF 4: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
; AVX512: Cost of 2 for VF 8: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-scatter-i32-with-i8-index.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-scatter-i32-with-i8-index.ll
index 6959fea2d512bc..b026016d1331bd 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-scatter-i32-with-i8-index.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-scatter-i32-with-i8-index.ll
@@ -17,21 +17,18 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
-; SSE2: LV: Found an estimated cost of 1 for VF 1 For instruction: store i32 %valB, ptr %out, align 4
; SSE2: Cost of 2.5 for VF 2: profitable to scalarize store i32 %valB, ptr %out, align 4
; SSE2: Cost of 5.5 for VF 4: profitable to scalarize store i32 %valB, ptr %out, align 4
; SSE2: Cost of 11 for VF 8: profitable to scalarize store i32 %valB, ptr %out, align 4
; SSE2: Cost of 22 for VF 16: profitable to scalarize store i32 %valB, ptr %out, align 4
;
; SSE42-LABEL: 'test'
-; SSE42: LV: Found an estimated cost of 1 for VF 1 For instruction: store i32 %valB, ptr %out, align 4
; SSE42: Cost of 2 for VF 2: profitable to scalarize store i32 %valB, ptr %out, align 4
; SSE42: Cost of 4 for VF 4: profitable to scalarize store i32 %valB, ptr %out, align 4
; SSE42: Cost of 8 for VF 8: profitable to scalarize store i32 %valB, ptr %out, align 4
; SSE42: Cost of 16 for VF 16: profitable to scalarize store i32 %valB, ptr %out, align 4
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: store i32 %valB, ptr %out, align 4
; AVX1: Cost of 2 for VF 2: profitable to scalarize store i32 %valB, ptr %out, align 4
; AVX1: Cost of 4 for VF 4: profitable to scalarize store i32 %valB, ptr %out, align 4
; AVX1: Cost of 8.5 for VF 8: profitable to scalarize store i32 %valB, ptr %out, align 4
@@ -39,7 +36,6 @@ define void @test() {
; AVX1: Cost of 34 for VF 32: profitable to scalarize store i32 %valB, ptr %out, align 4
;
; AVX2-LABEL: 'test'
-; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: store i32 %valB, ptr %out, align 4
; AVX2: Cost of 2 for VF 2: profitable to scalarize store i32 %valB, ptr %out, align 4
; AVX2: Cost of 4 for VF 4: profitable to scalarize store i32 %valB, ptr %out, align 4
; AVX2: Cost of 8.5 for VF 8: profitable to scalarize store i32 %valB, ptr %out, align 4
@@ -47,13 +43,12 @@ define void @test() {
; AVX2: Cost of 34 for VF 32: profitable to scalarize store i32 %valB, ptr %out, align 4
;
; AVX512-LABEL: 'test'
-; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: store i32 %valB, ptr %out, align 4
; AVX512: Cost of 5 for VF 2: REPLICATE store ir<%valB>, ir<%out>
; AVX512: Cost of 10.5 for VF 4: REPLICATE store ir<%valB>, ir<%out>
-; AVX512: Cost of 10 for VF 8: WIDEN store ir<%out>, ir<%valB>, ir<%canStore>
-; AVX512: Cost of 18 for VF 16: WIDEN store ir<%out>, ir<%valB>, ir<%canStore>
-; AVX512: Cost of 36 for VF 32: WIDEN store ir<%out>, ir<%valB>, ir<%canStore>
-; AVX512: Cost of 72 for VF 64: WIDEN store ir<%out>, ir<%valB>, ir<%canStore>
+; AVX512: Cost of 10 for VF 8: WIDEN store ir<%out>, ir<%valB>, ir<%canStore> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX512: Cost of 18 for VF 16: WIDEN store ir<%out>, ir<%valB>, ir<%canStore> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX512: Cost of 36 for VF 32: WIDEN store ir<%out>, ir<%valB>, ir<%canStore> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX512: Cost of 72 for VF 64: WIDEN store ir<%out>, ir<%valB>, ir<%canStore> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
;
entry:
br label %for.body
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-scatter-i64-with-i8-index.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-scatter-i64-with-i8-index.ll
index 41ae89933204e4..f9c6a5b14c8024 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-scatter-i64-with-i8-index.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-scatter-i64-with-i8-index.ll
@@ -17,21 +17,18 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
-; SSE2: LV: Found an estimated cost of 1 for VF 1 For instruction: store i64 %valB, ptr %out, align 8
; SSE2: Cost of 2.5 for VF 2: profitable to scalarize store i64 %valB, ptr %out, align 8
; SSE2: Cost of 5 for VF 4: profitable to scalarize store i64 %valB, ptr %out, align 8
; SSE2: Cost of 10 for VF 8: profitable to scalarize store i64 %valB, ptr %out, align 8
; SSE2: Cost of 20 for VF 16: profitable to scalarize store i64 %valB, ptr %out, align 8
;
; SSE42-LABEL: 'test'
-; SSE42: LV: Found an estimated cost of 1 for VF 1 For instruction: store i64 %valB, ptr %out, align 8
; SSE42: Cost of 2 for VF 2: profitable to scalarize store i64 %valB, ptr %out, align 8
; SSE42: Cost of 4 for VF 4: profitable to scalarize store i64 %valB, ptr %out, align 8
; SSE42: Cost of 8 for VF 8: profitable to scalarize store i64 %valB, ptr %out, align 8
; SSE42: Cost of 16 for VF 16: profitable to scalarize store i64 %valB, ptr %out, align 8
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: store i64 %valB, ptr %out, align 8
; AVX1: Cost of 2 for VF 2: profitable to scalarize store i64 %valB, ptr %out, align 8
; AVX1: Cost of 4.5 for VF 4: profitable to scalarize store i64 %valB, ptr %out, align 8
; AVX1: Cost of 9 for VF 8: profitable to scalarize store i64 %valB, ptr %out, align 8
@@ -39,7 +36,6 @@ define void @test() {
; AVX1: Cost of 36 for VF 32: profitable to scalarize store i64 %valB, ptr %out, align 8
;
; AVX2-LABEL: 'test'
-; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: store i64 %valB, ptr %out, align 8
; AVX2: Cost of 2 for VF 2: profitable to scalarize store i64 %valB, ptr %out, align 8
; AVX2: Cost of 4.5 for VF 4: profitable to scalarize store i64 %valB, ptr %out, align 8
; AVX2: Cost of 9 for VF 8: profitable to scalarize store i64 %valB, ptr %out, align 8
@@ -47,13 +43,12 @@ define void @test() {
; AVX2: Cost of 36 for VF 32: profitable to scalarize store i64 %valB, ptr %out, align 8
;
; AVX512-LABEL: 'test'
-; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: store i64 %valB, ptr %out, align 8
; AVX512: Cost of 5 for VF 2: REPLICATE store ir<%valB>, ir<%out>
; AVX512: Cost of 11 for VF 4: REPLICATE store ir<%valB>, ir<%out>
-; AVX512: Cost of 10 for VF 8: WIDEN store ir<%out>, ir<%valB>, ir<%canStore>
-; AVX512: Cost of 20 for VF 16: WIDEN store ir<%out>, ir<%valB>, ir<%canStore>
-; AVX512: Cost of 40 for VF 32: WIDEN store ir<%out>, ir<%valB>, ir<%canStore>
-; AVX512: Cost of 80 for VF 64: WIDEN store ir<%out>, ir<%valB>, ir<%canStore>
+; AVX512: Cost of 10 for VF 8: WIDEN store ir<%out>, ir<%valB>, ir<%canStore> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX512: Cost of 20 for VF 16: WIDEN store ir<%out>, ir<%valB>, ir<%canStore> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX512: Cost of 40 for VF 32: WIDEN store ir<%out>, ir<%valB>, ir<%canStore> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX512: Cost of 80 for VF 64: WIDEN store ir<%out>, ir<%valB>, ir<%canStore> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
;
entry:
br label %for.body
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i16.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i16.ll
index 61436a61dba509..886e593362383e 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i16.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i16.ll
@@ -16,14 +16,12 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test(ptr %C) {
; SSE-LABEL: 'test'
-; SSE: LV: Found an estimated cost of 1 for VF 1 For instruction: store i16 %valB, ptr %out, align 2
; SSE: Cost of 2 for VF 2: profitable to scalarize store i16 %valB, ptr %out, align 2
; SSE: Cost of 4 for VF 4: profitable to scalarize store i16 %valB, ptr %out, align 2
; SSE: Cost of 8 for VF 8: profitable to scalarize store i16 %valB, ptr %out, align 2
; SSE: Cost of 16 for VF 16: profitable to scalarize store i16 %valB, ptr %out, align 2
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: store i16 %valB, ptr %out, align 2
; AVX1: Cost of 2 for VF 2: profitable to scalarize store i16 %valB, ptr %out, align 2
; AVX1: Cost of 4 for VF 4: profitable to scalarize store i16 %valB, ptr %out, align 2
; AVX1: Cost of 8 for VF 8: profitable to scalarize store i16 %valB, ptr %out, align 2
@@ -31,7 +29,6 @@ define void @test(ptr %C) {
; AVX1: Cost of 33 for VF 32: profitable to scalarize store i16 %valB, ptr %out, align 2
;
; AVX2-LABEL: 'test'
-; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: store i16 %valB, ptr %out, align 2
; AVX2: Cost of 2 for VF 2: profitable to scalarize store i16 %valB, ptr %out, align 2
; AVX2: Cost of 4 for VF 4: profitable to scalarize store i16 %valB, ptr %out, align 2
; AVX2: Cost of 8 for VF 8: profitable to scalarize store i16 %valB, ptr %out, align 2
@@ -39,7 +36,6 @@ define void @test(ptr %C) {
; AVX2: Cost of 33 for VF 32: profitable to scalarize store i16 %valB, ptr %out, align 2
;
; AVX512-LABEL: 'test'
-; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: store i16 %valB, ptr %out, align 2
; AVX512: Cost of 2 for VF 2: WIDEN store vp<[[VP7:%[0-9]+]]>, ir<%valB>, ir<%canStore>
; AVX512: Cost of 2 for VF 4: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
; AVX512: Cost of 1 for VF 8: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i32.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i32.ll
index 0afea8d1664d58..9be2bd10f40fad 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i32.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i32.ll
@@ -16,21 +16,18 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test(ptr %C) {
; SSE2-LABEL: 'test'
-; SSE2: LV: Found an estimated cost of 1 for VF 1 For instruction: store i32 %valB, ptr %out, align 4
; SSE2: Cost of 2.5 for VF 2: profitable to scalarize store i32 %valB, ptr %out, align 4
; SSE2: Cost of 5.5 for VF 4: profitable to scalarize store i32 %valB, ptr %out, align 4
; SSE2: Cost of 11 for VF 8: profitable to scalarize store i32 %valB, ptr %out, align 4
; SSE2: Cost of 22 for VF 16: profitable to scalarize store i32 %valB, ptr %out, align 4
;
; SSE42-LABEL: 'test'
-; SSE42: LV: Found an estimated cost of 1 for VF 1 For instruction: store i32 %valB, ptr %out, align 4
; SSE42: Cost of 2 for VF 2: profitable to scalarize store i32 %valB, ptr %out, align 4
; SSE42: Cost of 4 for VF 4: profitable to scalarize store i32 %valB, ptr %out, align 4
; SSE42: Cost of 8 for VF 8: profitable to scalarize store i32 %valB, ptr %out, align 4
; SSE42: Cost of 16 for VF 16: profitable to scalarize store i32 %valB, ptr %out, align 4
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: store i32 %valB, ptr %out, align 4
; AVX1: Cost of 9 for VF 2: WIDEN store vp<[[VP7:%[0-9]+]]>, ir<%valB>, ir<%canStore>
; AVX1: Cost of 8 for VF 4: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
; AVX1: Cost of 8 for VF 8: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
@@ -38,7 +35,6 @@ define void @test(ptr %C) {
; AVX1: Cost of 32 for VF 32: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
;
; AVX2-LABEL: 'test'
-; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: store i32 %valB, ptr %out, align 4
; AVX2: Cost of 9 for VF 2: WIDEN store vp<[[VP7:%[0-9]+]]>, ir<%valB>, ir<%canStore>
; AVX2: Cost of 8 for VF 4: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
; AVX2: Cost of 8 for VF 8: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
@@ -46,7 +42,6 @@ define void @test(ptr %C) {
; AVX2: Cost of 32 for VF 32: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
;
; AVX512-LABEL: 'test'
-; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: store i32 %valB, ptr %out, align 4
; AVX512: Cost of 2 for VF 2: WIDEN store vp<[[VP7:%[0-9]+]]>, ir<%valB>, ir<%canStore>
; AVX512: Cost of 1 for VF 4: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
; AVX512: Cost of 1 for VF 8: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i64.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i64.ll
index ce2d69fca6a3b5..acc0158917568f 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i64.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i64.ll
@@ -16,21 +16,18 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test(ptr %C) {
; SSE2-LABEL: 'test'
-; SSE2: LV: Found an estimated cost of 1 for VF 1 For instruction: store i64 %valB, ptr %out, align 8
; SSE2: Cost of 2.5 for VF 2: profitable to scalarize store i64 %valB, ptr %out, align 8
; SSE2: Cost of 5 for VF 4: profitable to scalarize store i64 %valB, ptr %out, align 8
; SSE2: Cost of 10 for VF 8: profitable to scalarize store i64 %valB, ptr %out, align 8
; SSE2: Cost of 20 for VF 16: profitable to scalarize store i64 %valB, ptr %out, align 8
;
; SSE42-LABEL: 'test'
-; SSE42: LV: Found an estimated cost of 1 for VF 1 For instruction: store i64 %valB, ptr %out, align 8
; SSE42: Cost of 2 for VF 2: profitable to scalarize store i64 %valB, ptr %out, align 8
; SSE42: Cost of 4 for VF 4: profitable to scalarize store i64 %valB, ptr %out, align 8
; SSE42: Cost of 8 for VF 8: profitable to scalarize store i64 %valB, ptr %out, align 8
; SSE42: Cost of 16 for VF 16: profitable to scalarize store i64 %valB, ptr %out, align 8
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: store i64 %valB, ptr %out, align 8
; AVX1: Cost of 8 for VF 2: WIDEN store vp<[[VP7:%[0-9]+]]>, ir<%valB>, ir<%canStore>
; AVX1: Cost of 8 for VF 4: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
; AVX1: Cost of 16 for VF 8: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
@@ -38,7 +35,6 @@ define void @test(ptr %C) {
; AVX1: Cost of 64 for VF 32: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
;
; AVX2-LABEL: 'test'
-; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: store i64 %valB, ptr %out, align 8
; AVX2: Cost of 8 for VF 2: WIDEN store vp<[[VP7:%[0-9]+]]>, ir<%valB>, ir<%canStore>
; AVX2: Cost of 8 for VF 4: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
; AVX2: Cost of 16 for VF 8: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
@@ -46,7 +42,6 @@ define void @test(ptr %C) {
; AVX2: Cost of 64 for VF 32: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
;
; AVX512-LABEL: 'test'
-; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: store i64 %valB, ptr %out, align 8
; AVX512: Cost of 1 for VF 2: WIDEN store vp<[[VP7:%[0-9]+]]>, ir<%valB>, ir<%canStore>
; AVX512: Cost of 1 for VF 4: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
; AVX512: Cost of 1 for VF 8: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i8.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i8.ll
index 9028e1c5525a0f..721aa7ba285b18 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i8.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i8.ll
@@ -16,21 +16,18 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test(ptr %C) {
; SSE2-LABEL: 'test'
-; SSE2: LV: Found an estimated cost of 1 for VF 1 For instruction: store i8 %valB, ptr %out, align 1
; SSE2: Cost of 2.5 for VF 2: profitable to scalarize store i8 %valB, ptr %out, align 1
; SSE2: Cost of 5.5 for VF 4: profitable to scalarize store i8 %valB, ptr %out, align 1
; SSE2: Cost of 11.5 for VF 8: profitable to scalarize store i8 %valB, ptr %out, align 1
; SSE2: Cost of 23.5 for VF 16: profitable to scalarize store i8 %valB, ptr %out, align 1
;
; SSE42-LABEL: 'test'
-; SSE42: LV: Found an estimated cost of 1 for VF 1 For instruction: store i8 %valB, ptr %out, align 1
; SSE42: Cost of 2 for VF 2: profitable to scalarize store i8 %valB, ptr %out, align 1
; SSE42: Cost of 4 for VF 4: profitable to scalarize store i8 %valB, ptr %out, align 1
; SSE42: Cost of 8 for VF 8: profitable to scalarize store i8 %valB, ptr %out, align 1
; SSE42: Cost of 16 for VF 16: profitable to scalarize store i8 %valB, ptr %out, align 1
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: store i8 %valB, ptr %out, align 1
; AVX1: Cost of 2 for VF 2: profitable to scalarize store i8 %valB, ptr %out, align 1
; AVX1: Cost of 4 for VF 4: profitable to scalarize store i8 %valB, ptr %out, align 1
; AVX1: Cost of 8 for VF 8: profitable to scalarize store i8 %valB, ptr %out, align 1
@@ -38,7 +35,6 @@ define void @test(ptr %C) {
; AVX1: Cost of 32.5 for VF 32: profitable to scalarize store i8 %valB, ptr %out, align 1
;
; AVX2-LABEL: 'test'
-; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: store i8 %valB, ptr %out, align 1
; AVX2: Cost of 2 for VF 2: profitable to scalarize store i8 %valB, ptr %out, align 1
; AVX2: Cost of 4 for VF 4: profitable to scalarize store i8 %valB, ptr %out, align 1
; AVX2: Cost of 8 for VF 8: profitable to scalarize store i8 %valB, ptr %out, align 1
@@ -46,7 +42,6 @@ define void @test(ptr %C) {
; AVX2: Cost of 32.5 for VF 32: profitable to scalarize store i8 %valB, ptr %out, align 1
;
; AVX512-LABEL: 'test'
-; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: store i8 %valB, ptr %out, align 1
; AVX512: Cost of 2 for VF 2: WIDEN store vp<[[VP7:%[0-9]+]]>, ir<%valB>, ir<%canStore>
; AVX512: Cost of 2 for VF 4: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
; AVX512: Cost of 2 for VF 8: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i16-with-i8-index.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i16-with-i8-index.ll
index 6782bbfeb53b31..9daa35ed531659 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i16-with-i8-index.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i16-with-i8-index.ll
@@ -17,21 +17,18 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
-; SSE2: LV: Found an estimated cost of 1 for VF 1 For instruction: store i16 %valB, ptr %out, align 2
; SSE2: Cost of 28 for VF 2: REPLICATE store ir<%valB>, ir<%out>
; SSE2: Cost of 56 for VF 4: REPLICATE store ir<%valB>, ir<%out>
; SSE2: Cost of 112 for VF 8: REPLICATE store ir<%valB>, ir<%out>
; SSE2: Cost of 224 for VF 16: REPLICATE store ir<%valB>, ir<%out>
;
; SSE42-LABEL: 'test'
-; SSE42: LV: Found an estimated cost of 1 for VF 1 For instruction: store i16 %valB, ptr %out, align 2
; SSE42: Cost of 26 for VF 2: REPLICATE store ir<%valB>, ir<%out>
; SSE42: Cost of 52 for VF 4: REPLICATE store ir<%valB>, ir<%out>
; SSE42: Cost of 104 for VF 8: REPLICATE store ir<%valB>, ir<%out>
; SSE42: Cost of 208 for VF 16: REPLICATE store ir<%valB>, ir<%out>
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: store i16 %valB, ptr %out, align 2
; AVX1: Cost of 26 for VF 2: REPLICATE store ir<%valB>, ir<%out>
; AVX1: Cost of 53 for VF 4: REPLICATE store ir<%valB>, ir<%out>
; AVX1: Cost of 106 for VF 8: REPLICATE store ir<%valB>, ir<%out>
@@ -39,7 +36,6 @@ define void @test() {
; AVX1: Cost of 426 for VF 32: REPLICATE store ir<%valB>, ir<%out>
;
; AVX2-LABEL: 'test'
-; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: store i16 %valB, ptr %out, align 2
; AVX2: Cost of 6 for VF 2: REPLICATE store ir<%valB>, ir<%out>
; AVX2: Cost of 13 for VF 4: REPLICATE store ir<%valB>, ir<%out>
; AVX2: Cost of 26 for VF 8: REPLICATE store ir<%valB>, ir<%out>
@@ -47,7 +43,6 @@ define void @test() {
; AVX2: Cost of 106 for VF 32: REPLICATE store ir<%valB>, ir<%out>
;
; AVX512-LABEL: 'test'
-; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: store i16 %valB, ptr %out, align 2
; AVX512: Cost of 6 for VF 2: REPLICATE store ir<%valB>, ir<%out>
; AVX512: Cost of 13 for VF 4: REPLICATE store ir<%valB>, ir<%out>
; AVX512: Cost of 27 for VF 8: REPLICATE store ir<%valB>, ir<%out>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i32-with-i8-index.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i32-with-i8-index.ll
index 5e0b3277dd5e9e..671842c2645cf2 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i32-with-i8-index.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i32-with-i8-index.ll
@@ -17,21 +17,18 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
-; SSE2: LV: Found an estimated cost of 1 for VF 1 For instruction: store i32 %valB, ptr %out, align 4
; SSE2: Cost of 29 for VF 2: REPLICATE store ir<%valB>, ir<%out>
; SSE2: Cost of 59 for VF 4: REPLICATE store ir<%valB>, ir<%out>
; SSE2: Cost of 118 for VF 8: REPLICATE store ir<%valB>, ir<%out>
; SSE2: Cost of 236 for VF 16: REPLICATE store ir<%valB>, ir<%out>
;
; SSE42-LABEL: 'test'
-; SSE42: LV: Found an estimated cost of 1 for VF 1 For instruction: store i32 %valB, ptr %out, align 4
; SSE42: Cost of 26 for VF 2: REPLICATE store ir<%valB>, ir<%out>
; SSE42: Cost of 52 for VF 4: REPLICATE store ir<%valB>, ir<%out>
; SSE42: Cost of 104 for VF 8: REPLICATE store ir<%valB>, ir<%out>
; SSE42: Cost of 208 for VF 16: REPLICATE store ir<%valB>, ir<%out>
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: store i32 %valB, ptr %out, align 4
; AVX1: Cost of 26 for VF 2: REPLICATE store ir<%valB>, ir<%out>
; AVX1: Cost of 53 for VF 4: REPLICATE store ir<%valB>, ir<%out>
; AVX1: Cost of 107 for VF 8: REPLICATE store ir<%valB>, ir<%out>
@@ -39,7 +36,6 @@ define void @test() {
; AVX1: Cost of 428 for VF 32: REPLICATE store ir<%valB>, ir<%out>
;
; AVX2-LABEL: 'test'
-; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: store i32 %valB, ptr %out, align 4
; AVX2: Cost of 6 for VF 2: REPLICATE store ir<%valB>, ir<%out>
; AVX2: Cost of 13 for VF 4: REPLICATE store ir<%valB>, ir<%out>
; AVX2: Cost of 27 for VF 8: REPLICATE store ir<%valB>, ir<%out>
@@ -47,7 +43,6 @@ define void @test() {
; AVX2: Cost of 108 for VF 32: REPLICATE store ir<%valB>, ir<%out>
;
; AVX512-LABEL: 'test'
-; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: store i32 %valB, ptr %out, align 4
; AVX512: Cost of 6 for VF 2: REPLICATE store ir<%valB>, ir<%out>
; AVX512: Cost of 13 for VF 4: REPLICATE store ir<%valB>, ir<%out>
; AVX512: Cost of 10 for VF 8: WIDEN store ir<%out>, ir<%valB>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i64-with-i8-index.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i64-with-i8-index.ll
index fcbf6042dec148..dc775779060122 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i64-with-i8-index.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i64-with-i8-index.ll
@@ -17,21 +17,18 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
-; SSE2: LV: Found an estimated cost of 1 for VF 1 For instruction: store i64 %valB, ptr %out, align 8
; SSE2: Cost of 29 for VF 2: REPLICATE store ir<%valB>, ir<%out>
; SSE2: Cost of 58 for VF 4: REPLICATE store ir<%valB>, ir<%out>
; SSE2: Cost of 116 for VF 8: REPLICATE store ir<%valB>, ir<%out>
; SSE2: Cost of 232 for VF 16: REPLICATE store ir<%valB>, ir<%out>
;
; SSE42-LABEL: 'test'
-; SSE42: LV: Found an estimated cost of 1 for VF 1 For instruction: store i64 %valB, ptr %out, align 8
; SSE42: Cost of 26 for VF 2: REPLICATE store ir<%valB>, ir<%out>
; SSE42: Cost of 52 for VF 4: REPLICATE store ir<%valB>, ir<%out>
; SSE42: Cost of 104 for VF 8: REPLICATE store ir<%valB>, ir<%out>
; SSE42: Cost of 208 for VF 16: REPLICATE store ir<%valB>, ir<%out>
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: store i64 %valB, ptr %out, align 8
; AVX1: Cost of 26 for VF 2: REPLICATE store ir<%valB>, ir<%out>
; AVX1: Cost of 54 for VF 4: REPLICATE store ir<%valB>, ir<%out>
; AVX1: Cost of 108 for VF 8: REPLICATE store ir<%valB>, ir<%out>
@@ -39,7 +36,6 @@ define void @test() {
; AVX1: Cost of 432 for VF 32: REPLICATE store ir<%valB>, ir<%out>
;
; AVX2-LABEL: 'test'
-; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: store i64 %valB, ptr %out, align 8
; AVX2: Cost of 6 for VF 2: REPLICATE store ir<%valB>, ir<%out>
; AVX2: Cost of 14 for VF 4: REPLICATE store ir<%valB>, ir<%out>
; AVX2: Cost of 28 for VF 8: REPLICATE store ir<%valB>, ir<%out>
@@ -47,7 +43,6 @@ define void @test() {
; AVX2: Cost of 112 for VF 32: REPLICATE store ir<%valB>, ir<%out>
;
; AVX512-LABEL: 'test'
-; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: store i64 %valB, ptr %out, align 8
; AVX512: Cost of 6 for VF 2: REPLICATE store ir<%valB>, ir<%out>
; AVX512: Cost of 14 for VF 4: REPLICATE store ir<%valB>, ir<%out>
; AVX512: Cost of 10 for VF 8: WIDEN store ir<%out>, ir<%valB>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i8-with-i8-index.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i8-with-i8-index.ll
index 2946cd291d7fa1..c484b526ac1cb3 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i8-with-i8-index.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i8-with-i8-index.ll
@@ -17,21 +17,18 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
-; SSE2: LV: Found an estimated cost of 1 for VF 1 For instruction: store i8 %valB, ptr %out, align 1
; SSE2: Cost of 29 for VF 2: REPLICATE store ir<%valB>, ir<%out>
; SSE2: Cost of 59 for VF 4: REPLICATE store ir<%valB>, ir<%out>
; SSE2: Cost of 119 for VF 8: REPLICATE store ir<%valB>, ir<%out>
; SSE2: Cost of 239 for VF 16: REPLICATE store ir<%valB>, ir<%out>
;
; SSE42-LABEL: 'test'
-; SSE42: LV: Found an estimated cost of 1 for VF 1 For instruction: store i8 %valB, ptr %out, align 1
; SSE42: Cost of 26 for VF 2: REPLICATE store ir<%valB>, ir<%out>
; SSE42: Cost of 52 for VF 4: REPLICATE store ir<%valB>, ir<%out>
; SSE42: Cost of 104 for VF 8: REPLICATE store ir<%valB>, ir<%out>
; SSE42: Cost of 208 for VF 16: REPLICATE store ir<%valB>, ir<%out>
;
; AVX1-LABEL: 'test'
-; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: store i8 %valB, ptr %out, align 1
; AVX1: Cost of 26 for VF 2: REPLICATE store ir<%valB>, ir<%out>
; AVX1: Cost of 53 for VF 4: REPLICATE store ir<%valB>, ir<%out>
; AVX1: Cost of 106 for VF 8: REPLICATE store ir<%valB>, ir<%out>
@@ -39,7 +36,6 @@ define void @test() {
; AVX1: Cost of 425 for VF 32: REPLICATE store ir<%valB>, ir<%out>
;
; AVX2-LABEL: 'test'
-; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: store i8 %valB, ptr %out, align 1
; AVX2: Cost of 6 for VF 2: REPLICATE store ir<%valB>, ir<%out>
; AVX2: Cost of 13 for VF 4: REPLICATE store ir<%valB>, ir<%out>
; AVX2: Cost of 26 for VF 8: REPLICATE store ir<%valB>, ir<%out>
@@ -47,7 +43,6 @@ define void @test() {
; AVX2: Cost of 105 for VF 32: REPLICATE store ir<%valB>, ir<%out>
;
; AVX512-LABEL: 'test'
-; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: store i8 %valB, ptr %out, align 1
; AVX512: Cost of 6 for VF 2: REPLICATE store ir<%valB>, ir<%out>
; AVX512: Cost of 13 for VF 4: REPLICATE store ir<%valB>, ir<%out>
; AVX512: Cost of 27 for VF 8: REPLICATE store ir<%valB>, ir<%out>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/strided-load-i16.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/strided-load-i16.ll
index b77b8ac294163d..3c27644f4ebcf0 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/strided-load-i16.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/strided-load-i16.ll
@@ -10,7 +10,6 @@ target triple = "x86_64-unknown-linux-gnu"
define void @load_i16_stride2() {
; CHECK-LABEL: 'load_i16_stride2'
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %1 = load i16, ptr %arrayidx, align 4
; CHECK: Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%arrayidx>
; CHECK: Cost of 1 for VF 4: INTERLEAVE-GROUP with factor 2, ir<%arrayidx>
; CHECK: Cost of 2 for VF 8: INTERLEAVE-GROUP with factor 2, ir<%arrayidx>
@@ -37,7 +36,6 @@ for.end:
define void @load_i16_stride3() {
; CHECK-LABEL: 'load_i16_stride3'
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %1 = load i16, ptr %arrayidx, align 4
; CHECK: Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%arrayidx>
; CHECK: Cost of 2 for VF 4: INTERLEAVE-GROUP with factor 3, ir<%arrayidx>
; CHECK: Cost of 2 for VF 8: INTERLEAVE-GROUP with factor 3, ir<%arrayidx>
@@ -64,7 +62,6 @@ for.end:
define void @load_i16_stride4() {
; CHECK-LABEL: 'load_i16_stride4'
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %1 = load i16, ptr %arrayidx, align 4
; CHECK: Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%arrayidx>
; CHECK: Cost of 2 for VF 4: INTERLEAVE-GROUP with factor 4, ir<%arrayidx>
; CHECK: Cost of 2 for VF 8: INTERLEAVE-GROUP with factor 4, ir<%arrayidx>
@@ -91,7 +88,6 @@ for.end:
define void @load_i16_stride5() {
; CHECK-LABEL: 'load_i16_stride5'
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %1 = load i16, ptr %arrayidx, align 4
; CHECK: Cost of 2 for VF 2: INTERLEAVE-GROUP with factor 5, ir<%arrayidx>
; CHECK: Cost of 2 for VF 4: INTERLEAVE-GROUP with factor 5, ir<%arrayidx>
; CHECK: Cost of 3 for VF 8: INTERLEAVE-GROUP with factor 5, ir<%arrayidx>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/strided-load-i32.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/strided-load-i32.ll
index da9722b65ff06a..8bc141496b822b 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/strided-load-i32.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/strided-load-i32.ll
@@ -10,7 +10,6 @@ target triple = "x86_64-unknown-linux-gnu"
define void @load_int_stride2() {
; CHECK-LABEL: 'load_int_stride2'
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %1 = load i32, ptr %arrayidx, align 4
; CHECK: Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%arrayidx>
; CHECK: Cost of 1 for VF 4: INTERLEAVE-GROUP with factor 2, ir<%arrayidx>
; CHECK: Cost of 1 for VF 8: INTERLEAVE-GROUP with factor 2, ir<%arrayidx>
@@ -36,7 +35,6 @@ for.end:
define void @load_int_stride3() {
; CHECK-LABEL: 'load_int_stride3'
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %1 = load i32, ptr %arrayidx, align 4
; CHECK: Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%arrayidx>
; CHECK: Cost of 1 for VF 4: INTERLEAVE-GROUP with factor 3, ir<%arrayidx>
; CHECK: Cost of 3 for VF 8: INTERLEAVE-GROUP with factor 3, ir<%arrayidx>
@@ -62,7 +60,6 @@ for.end:
define void @load_int_stride4() {
; CHECK-LABEL: 'load_int_stride4'
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %1 = load i32, ptr %arrayidx, align 4
; CHECK: Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%arrayidx>
; CHECK: Cost of 1 for VF 4: INTERLEAVE-GROUP with factor 4, ir<%arrayidx>
; CHECK: Cost of 3 for VF 8: INTERLEAVE-GROUP with factor 4, ir<%arrayidx>
@@ -88,7 +85,6 @@ for.end:
define void @load_int_stride5() {
; CHECK-LABEL: 'load_int_stride5'
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %1 = load i32, ptr %arrayidx, align 4
; CHECK: Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 5, ir<%arrayidx>
; CHECK: Cost of 3 for VF 4: INTERLEAVE-GROUP with factor 5, ir<%arrayidx>
; CHECK: Cost of 5 for VF 8: INTERLEAVE-GROUP with factor 5, ir<%arrayidx>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/strided-load-i64.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/strided-load-i64.ll
index 49a5fc47944a8d..1c91d01340a4b9 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/strided-load-i64.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/strided-load-i64.ll
@@ -10,7 +10,6 @@ target triple = "x86_64-unknown-linux-gnu"
define void @load_i64_stride2() {
; CHECK-LABEL: 'load_i64_stride2'
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %1 = load i64, ptr %arrayidx, align 16
; CHECK: Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%arrayidx>
; CHECK: Cost of 1 for VF 4: INTERLEAVE-GROUP with factor 2, ir<%arrayidx>
; CHECK: Cost of 3 for VF 8: INTERLEAVE-GROUP with factor 2, ir<%arrayidx>
@@ -35,7 +34,6 @@ for.end:
define void @load_i64_stride3() {
; CHECK-LABEL: 'load_i64_stride3'
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %1 = load i64, ptr %arrayidx, align 16
; CHECK: Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%arrayidx>
; CHECK: Cost of 3 for VF 4: INTERLEAVE-GROUP with factor 3, ir<%arrayidx>
; CHECK: Cost of 5 for VF 8: INTERLEAVE-GROUP with factor 3, ir<%arrayidx>
@@ -60,7 +58,6 @@ for.end:
define void @load_i64_stride4() {
; CHECK-LABEL: 'load_i64_stride4'
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %1 = load i64, ptr %arrayidx, align 16
; CHECK: Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%arrayidx>
; CHECK: Cost of 3 for VF 4: INTERLEAVE-GROUP with factor 4, ir<%arrayidx>
; CHECK: Cost of 8 for VF 8: INTERLEAVE-GROUP with factor 4, ir<%arrayidx>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/strided-load-i8.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/strided-load-i8.ll
index 9ac785187cc1aa..b87607872fdccf 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/strided-load-i8.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/strided-load-i8.ll
@@ -10,7 +10,6 @@ target triple = "x86_64-unknown-linux-gnu"
define void @load_i8_stride2() {
; CHECK-LABEL: 'load_i8_stride2'
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %1 = load i8, ptr %arrayidx, align 2
; CHECK: Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%arrayidx>
; CHECK: Cost of 1 for VF 4: INTERLEAVE-GROUP with factor 2, ir<%arrayidx>
; CHECK: Cost of 1 for VF 8: INTERLEAVE-GROUP with factor 2, ir<%arrayidx>
@@ -38,7 +37,6 @@ for.end:
define void @load_i8_stride3() {
; CHECK-LABEL: 'load_i8_stride3'
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %1 = load i8, ptr %arrayidx, align 2
; CHECK: Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%arrayidx>
; CHECK: Cost of 1 for VF 4: INTERLEAVE-GROUP with factor 3, ir<%arrayidx>
; CHECK: Cost of 4 for VF 8: INTERLEAVE-GROUP with factor 3, ir<%arrayidx>
@@ -66,7 +64,6 @@ for.end:
define void @load_i8_stride4() {
; CHECK-LABEL: 'load_i8_stride4'
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %1 = load i8, ptr %arrayidx, align 2
; CHECK: Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%arrayidx>
; CHECK: Cost of 1 for VF 4: INTERLEAVE-GROUP with factor 4, ir<%arrayidx>
; CHECK: Cost of 4 for VF 8: INTERLEAVE-GROUP with factor 4, ir<%arrayidx>
@@ -94,7 +91,6 @@ for.end:
define void @load_i8_stride5() {
; CHECK-LABEL: 'load_i8_stride5'
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %1 = load i8, ptr %arrayidx, align 2
; CHECK: Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 5, ir<%arrayidx>
; CHECK: Cost of 4 for VF 4: INTERLEAVE-GROUP with factor 5, ir<%arrayidx>
; CHECK: Cost of 8 for VF 8: INTERLEAVE-GROUP with factor 5, ir<%arrayidx>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/vpinstruction-cost.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/vpinstruction-cost.ll
index be1b6b42496b1a..3e9ae12df16cba 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/vpinstruction-cost.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/vpinstruction-cost.ll
@@ -7,19 +7,18 @@ target datalayout = "e-m:e-p270:32:32-p271:32:32-p272:64:64-i64:64-i128:128-f80:
define void @wide_or_replaced_with_add_vpinstruction(ptr %src, ptr noalias %dst) {
; CHECK-LABEL: 'wide_or_replaced_with_add_vpinstruction'
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction: %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop.latch ]
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction: %g.src = getelementptr inbounds i64, ptr %src, i64 %iv
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %l = load i64, ptr %g.src, align 8
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %iv.4 = add nuw nsw i64 %iv, 4
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %c = icmp ule i64 %l, 128
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction: br i1 %c, label %loop.then, label %loop.latch
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %or = or disjoint i64 %iv.4, 1
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction: %g.dst = getelementptr inbounds i64, ptr %dst, i64 %or
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: store i64 %iv.4, ptr %g.dst, align 4
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction: br label %loop.latch
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %iv.next = add nuw nsw i64 %iv, 1
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %exitcond = icmp eq i64 %iv.next, 32
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction: br i1 %exitcond, label %exit, label %loop.header
+; CHECK: Cost of 0 for VF 1: EMIT-SCALAR ir<%iv> = phi [ ir<0>, vector.ph ], [ ir<%iv.next>, loop.latch ]
+; CHECK: Cost of 0 for VF 1: EMIT ir<%g.src> = getelementptr inbounds ir<%src>, ir<%iv>
+; CHECK: Cost of 1 for VF 1: EMIT-SCALAR ir<%l> = load ir<%g.src>
+; CHECK: Cost of 1 for VF 1: EMIT ir<%iv.4> = add nuw nsw ir<%iv>, ir<4>
+; CHECK: Cost of 1 for VF 1: EMIT ir<%c> = icmp ule ir<%l>, ir<128>
+; CHECK: Cost of 0 for VF 1: EMIT branch-on-cond ir<%c> (!vplan.prof.estimated estimated {1073741824, 1073741824})
+; CHECK: Cost of 1 for VF 1: EMIT ir<%or> = or disjoint ir<%iv.4>, ir<1> (!vplan.execution.frequency 4611686018427387904 (50%, estimated))
+; CHECK: Cost of 0 for VF 1: EMIT ir<%g.dst> = getelementptr inbounds ir<%dst>, ir<%or> (!vplan.execution.frequency 4611686018427387904 (50%, estimated))
+; CHECK: Cost of 1 for VF 1: EMIT store ir<%iv.4>, ir<%g.dst> (!vplan.execution.frequency 4611686018427387904 (50%, estimated))
+; CHECK: Cost of 1 for VF 1: EMIT ir<%iv.next> = add nuw nsw ir<%iv>, ir<1>
+; CHECK: Cost of 1 for VF 1: EMIT ir<%exitcond> = icmp eq ir<%iv.next>, ir<32>
+; CHECK: Cost of 0 for VF 1: EMIT branch-on-cond ir<%exitcond>
; CHECK: Cost of 1 for VF 2: ir<%iv> = WIDEN-INDUCTION nuw nsw ir<0>, ir<1>, vp<[[VP0:%[0-9]+]]>
; CHECK: Cost of 0 for VF 2: vp<[[VP4:%[0-9]+]]> = SCALAR-STEPS vp<[[VP3:%[0-9]+]]>, ir<1>, vp<[[VP0]]>
; CHECK: Cost of 0 for VF 2: CLONE ir<%g.src> = getelementptr inbounds ir<%src>, vp<[[VP4]]>
@@ -30,7 +29,7 @@ define void @wide_or_replaced_with_add_vpinstruction(ptr %src, ptr noalias %dst)
; CHECK: Cost of 1 for VF 2: EMIT ir<%or> = add ir<%iv.4>, ir<1>
; CHECK: Cost of 0 for VF 2: CLONE ir<%g.dst> = getelementptr ir<%dst>, ir<%or>
; CHECK: Cost of 0 for VF 2: vp<[[VP6:%[0-9]+]]> = vector-pointer i64, ir<%g.dst>, ir<1>
-; CHECK: Cost of 1 for VF 2: WIDEN store vp<[[VP6]]>, ir<%iv.4>, ir<%c>
+; CHECK: Cost of 1 for VF 2: WIDEN store vp<[[VP6]]>, ir<%iv.4>, ir<%c> (!vplan.execution.frequency 4611686018427387904 (50%, estimated))
; CHECK: Cost of 0 for VF 2: EMIT vp<%index.next> = add nuw vp<[[VP3]]>, vp<[[VP1:%[0-9]+]]>
; CHECK: Cost of 1 for VF 2: EMIT branch-on-count vp<%index.next>, vp<[[VP2:%[0-9]+]]>
; CHECK: Cost of 0 for VF 2: vector loop backedge
@@ -53,7 +52,7 @@ define void @wide_or_replaced_with_add_vpinstruction(ptr %src, ptr noalias %dst)
; CHECK: Cost of 1 for VF 4: EMIT ir<%or> = add ir<%iv.4>, ir<1>
; CHECK: Cost of 0 for VF 4: CLONE ir<%g.dst> = getelementptr ir<%dst>, ir<%or>
; CHECK: Cost of 0 for VF 4: vp<[[VP6]]> = vector-pointer i64, ir<%g.dst>, ir<1>
-; CHECK: Cost of 1 for VF 4: WIDEN store vp<[[VP6]]>, ir<%iv.4>, ir<%c>
+; CHECK: Cost of 1 for VF 4: WIDEN store vp<[[VP6]]>, ir<%iv.4>, ir<%c> (!vplan.execution.frequency 4611686018427387904 (50%, estimated))
; CHECK: Cost of 0 for VF 4: EMIT vp<%index.next> = add nuw vp<[[VP3]]>, vp<[[VP1]]>
; CHECK: Cost of 1 for VF 4: EMIT branch-on-count vp<%index.next>, vp<[[VP2]]>
; CHECK: Cost of 0 for VF 4: vector loop backedge
@@ -97,15 +96,15 @@ exit:
define void @test_vpinstruction_freeze_cost(ptr %src, ptr noalias %dst) {
; CHECK-LABEL: 'test_vpinstruction_freeze_cost'
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction: %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction: %g.src = getelementptr inbounds i64, ptr %src, i64 %iv
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %l = load i64, ptr %g.src, align 8
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction: %fr = freeze i64 %l
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction: %g.dst = getelementptr inbounds i64, ptr %dst, i64 %iv
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: store i64 %fr, ptr %g.dst, align 8
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %iv.next = add nuw nsw i64 %iv, 1
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %ec = icmp eq i64 %iv.next, 32
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction: br i1 %ec, label %exit, label %loop
+; CHECK: Cost of 0 for VF 1: EMIT-SCALAR ir<%iv> = phi [ ir<0>, vector.ph ], [ ir<%iv.next>, loop ]
+; CHECK: Cost of 0 for VF 1: EMIT ir<%g.src> = getelementptr inbounds ir<%src>, ir<%iv>
+; CHECK: Cost of 1 for VF 1: EMIT-SCALAR ir<%l> = load ir<%g.src>
+; CHECK: Cost of 0 for VF 1: EMIT ir<%fr> = freeze ir<%l>
+; CHECK: Cost of 0 for VF 1: EMIT ir<%g.dst> = getelementptr inbounds ir<%dst>, ir<%iv>
+; CHECK: Cost of 1 for VF 1: EMIT store ir<%fr>, ir<%g.dst>
+; CHECK: Cost of 1 for VF 1: EMIT ir<%iv.next> = add nuw nsw ir<%iv>, ir<1>
+; CHECK: Cost of 1 for VF 1: EMIT ir<%ec> = icmp eq ir<%iv.next>, ir<32>
+; CHECK: Cost of 0 for VF 1: EMIT branch-on-cond ir<%ec>
; CHECK: Cost of 0 for VF 2: vp<[[VP4:%[0-9]+]]> = SCALAR-STEPS vp<[[VP3:%[0-9]+]]>, ir<1>, vp<[[VP0:%[0-9]+]]>
; CHECK: Cost of 0 for VF 2: CLONE ir<%g.src> = getelementptr inbounds ir<%src>, vp<[[VP4]]>
; CHECK: Cost of 0 for VF 2: vp<[[VP5:%[0-9]+]]> = vector-pointer inbounds i64, ir<%g.src>, ir<1>
@@ -175,20 +174,16 @@ exit:
define void @test_vpinstruction_switch_cost(ptr %start, ptr %end) {
; CHECK-LABEL: 'test_vpinstruction_switch_cost'
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction: %ptr.iv = phi ptr [ %start, %entry ], [ %ptr.iv.next, %loop.latch ]
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %l = load i64, ptr %ptr.iv, align 1
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction: switch i64 %l, label %default [
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: store i64 1, ptr %ptr.iv, align 1
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction: br label %loop.latch
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: store i64 0, ptr %ptr.iv, align 1
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction: br label %loop.latch
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: store i64 42, ptr %ptr.iv, align 1
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction: br label %loop.latch
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: store i64 2, ptr %ptr.iv, align 1
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction: br label %loop.latch
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction: %ptr.iv.next = getelementptr inbounds i64, ptr %ptr.iv, i64 1
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %ec = icmp eq ptr %ptr.iv.next, %end
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction: br i1 %ec, label %exit, label %loop.header
+; CHECK: Cost of 0 for VF 1: EMIT-SCALAR ir<%ptr.iv> = phi [ ir<%start>, vector.ph ], [ ir<%ptr.iv.next>, loop.latch ]
+; CHECK: Cost of 1 for VF 1: EMIT-SCALAR ir<%l> = load ir<%ptr.iv>
+; CHECK: Cost of 0 for VF 1: EMIT switch ir<%l>, ir<-12>, ir<13>, ir<0> (!vplan.prof.estimated estimated {536870912, 536870912, 536870912, 536870912})
+; CHECK: Cost of 1 for VF 1: EMIT store ir<1>, ir<%ptr.iv> (!vplan.execution.frequency 2305843009213693952 (25%, estimated))
+; CHECK: Cost of 1 for VF 1: EMIT store ir<0>, ir<%ptr.iv> (!vplan.execution.frequency 2305843009213693952 (25%, estimated))
+; CHECK: Cost of 1 for VF 1: EMIT store ir<42>, ir<%ptr.iv> (!vplan.execution.frequency 2305843009213693952 (25%, estimated))
+; CHECK: Cost of 1 for VF 1: EMIT store ir<2>, ir<%ptr.iv> (!vplan.execution.frequency 2305843009213693952 (25%, estimated))
+; CHECK: Cost of 0 for VF 1: EMIT ir<%ptr.iv.next> = getelementptr inbounds ir<%ptr.iv>, ir<1>
+; CHECK: Cost of 1 for VF 1: EMIT ir<%ec> = icmp eq ir<%ptr.iv.next>, ir<%end>
+; CHECK: Cost of 0 for VF 1: EMIT branch-on-cond ir<%ec>
; CHECK: Cost of 1 for VF 2: vp<[[VP6:%[0-9]+]]> = DERIVED-IV ir<0> + vp<[[VP5:%[0-9]+]]> * ir<8>
; CHECK: Cost of 0 for VF 2: vp<[[VP7:%[0-9]+]]> = SCALAR-STEPS vp<[[VP6]]>, ir<8>, vp<[[VP0:%[0-9]+]]>
; CHECK: Cost of 0 for VF 2: EMIT vp<%next.gep> = ptradd ir<%start>, vp<[[VP7]]>
@@ -201,13 +196,13 @@ define void @test_vpinstruction_switch_cost(ptr %start, ptr %end) {
; CHECK: Cost of 0 for VF 2: EMIT vp<[[VP13:%[0-9]+]]> = or vp<[[VP12]]>, vp<[[VP11]]>
; CHECK: Cost of 1 for VF 2: EMIT vp<[[VP14:%[0-9]+]]> = not vp<[[VP13]]>
; CHECK: Cost of 0 for VF 2: vp<[[VP15:%[0-9]+]]> = vector-pointer i64, vp<%next.gep>, ir<1>
-; CHECK: Cost of 1 for VF 2: WIDEN store vp<[[VP15]]>, ir<1>, vp<[[VP11]]>
+; CHECK: Cost of 1 for VF 2: WIDEN store vp<[[VP15]]>, ir<1>, vp<[[VP11]]> (!vplan.execution.frequency 2305843009213693952 (25%, estimated))
; CHECK: Cost of 0 for VF 2: vp<[[VP16:%[0-9]+]]> = vector-pointer i64, vp<%next.gep>, ir<1>
-; CHECK: Cost of 1 for VF 2: WIDEN store vp<[[VP16]]>, ir<0>, vp<[[VP10]]>
+; CHECK: Cost of 1 for VF 2: WIDEN store vp<[[VP16]]>, ir<0>, vp<[[VP10]]> (!vplan.execution.frequency 2305843009213693952 (25%, estimated))
; CHECK: Cost of 0 for VF 2: vp<[[VP17:%[0-9]+]]> = vector-pointer i64, vp<%next.gep>, ir<1>
-; CHECK: Cost of 1 for VF 2: WIDEN store vp<[[VP17]]>, ir<42>, vp<[[VP9]]>
+; CHECK: Cost of 1 for VF 2: WIDEN store vp<[[VP17]]>, ir<42>, vp<[[VP9]]> (!vplan.execution.frequency 2305843009213693952 (25%, estimated))
; CHECK: Cost of 0 for VF 2: vp<[[VP18:%[0-9]+]]> = vector-pointer i64, vp<%next.gep>, ir<1>
-; CHECK: Cost of 1 for VF 2: WIDEN store vp<[[VP18]]>, ir<2>, vp<[[VP14]]>
+; CHECK: Cost of 1 for VF 2: WIDEN store vp<[[VP18]]>, ir<2>, vp<[[VP14]]> (!vplan.execution.frequency 2305843009213693952 (25%, estimated))
; CHECK: Cost of 0 for VF 2: EMIT vp<%index.next> = add nuw vp<[[VP5]]>, vp<[[VP1:%[0-9]+]]>
; CHECK: Cost of 1 for VF 2: EMIT branch-on-count vp<%index.next>, vp<[[VP2:%[0-9]+]]>
; CHECK: Cost of 0 for VF 2: vector loop backedge
@@ -231,13 +226,13 @@ define void @test_vpinstruction_switch_cost(ptr %start, ptr %end) {
; CHECK: Cost of 0 for VF 4: EMIT vp<[[VP13]]> = or vp<[[VP12]]>, vp<[[VP11]]>
; CHECK: Cost of 1 for VF 4: EMIT vp<[[VP14]]> = not vp<[[VP13]]>
; CHECK: Cost of 0 for VF 4: vp<[[VP15]]> = vector-pointer i64, vp<%next.gep>, ir<1>
-; CHECK: Cost of 1 for VF 4: WIDEN store vp<[[VP15]]>, ir<1>, vp<[[VP11]]>
+; CHECK: Cost of 1 for VF 4: WIDEN store vp<[[VP15]]>, ir<1>, vp<[[VP11]]> (!vplan.execution.frequency 2305843009213693952 (25%, estimated))
; CHECK: Cost of 0 for VF 4: vp<[[VP16]]> = vector-pointer i64, vp<%next.gep>, ir<1>
-; CHECK: Cost of 1 for VF 4: WIDEN store vp<[[VP16]]>, ir<0>, vp<[[VP10]]>
+; CHECK: Cost of 1 for VF 4: WIDEN store vp<[[VP16]]>, ir<0>, vp<[[VP10]]> (!vplan.execution.frequency 2305843009213693952 (25%, estimated))
; CHECK: Cost of 0 for VF 4: vp<[[VP17]]> = vector-pointer i64, vp<%next.gep>, ir<1>
-; CHECK: Cost of 1 for VF 4: WIDEN store vp<[[VP17]]>, ir<42>, vp<[[VP9]]>
+; CHECK: Cost of 1 for VF 4: WIDEN store vp<[[VP17]]>, ir<42>, vp<[[VP9]]> (!vplan.execution.frequency 2305843009213693952 (25%, estimated))
; CHECK: Cost of 0 for VF 4: vp<[[VP18]]> = vector-pointer i64, vp<%next.gep>, ir<1>
-; CHECK: Cost of 1 for VF 4: WIDEN store vp<[[VP18]]>, ir<2>, vp<[[VP14]]>
+; CHECK: Cost of 1 for VF 4: WIDEN store vp<[[VP18]]>, ir<2>, vp<[[VP14]]> (!vplan.execution.frequency 2305843009213693952 (25%, estimated))
; CHECK: Cost of 0 for VF 4: EMIT vp<%index.next> = add nuw vp<[[VP5]]>, vp<[[VP1]]>
; CHECK: Cost of 1 for VF 4: EMIT branch-on-count vp<%index.next>, vp<[[VP2]]>
; CHECK: Cost of 0 for VF 4: vector loop backedge
@@ -289,15 +284,15 @@ exit:
define void @test_vpinstruction_extractvalue_cost(ptr noalias %dst, {i64, i64} %sv) {
; CHECK-LABEL: 'test_vpinstruction_extractvalue_cost'
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction: %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction: %a = extractvalue { i64, i64 } %sv, 0
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction: %b = extractvalue { i64, i64 } %sv, 1
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %add = add i64 %a, %b
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction: %g.dst = getelementptr inbounds i64, ptr %dst, i64 %iv
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: store i64 %add, ptr %g.dst, align 8
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %iv.next = add nuw nsw i64 %iv, 1
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %ec = icmp eq i64 %iv.next, 1000
-; CHECK: LV: Found an estimated cost of 0 for VF 1 For instruction: br i1 %ec, label %exit, label %loop
+; CHECK: Cost of 0 for VF 1: EMIT-SCALAR ir<%iv> = phi [ ir<0>, vector.ph ], [ ir<%iv.next>, loop ]
+; CHECK: Cost of 0 for VF 1: EMIT ir<%a> = extractvalue ir<%sv>
+; CHECK: Cost of 0 for VF 1: EMIT ir<%b> = extractvalue ir<%sv>
+; CHECK: Cost of 1 for VF 1: EMIT ir<%add> = add ir<%a>, ir<%b>
+; CHECK: Cost of 0 for VF 1: EMIT ir<%g.dst> = getelementptr inbounds ir<%dst>, ir<%iv>
+; CHECK: Cost of 1 for VF 1: EMIT store ir<%add>, ir<%g.dst>
+; CHECK: Cost of 1 for VF 1: EMIT ir<%iv.next> = add nuw nsw ir<%iv>, ir<1>
+; CHECK: Cost of 1 for VF 1: EMIT ir<%ec> = icmp eq ir<%iv.next>, ir<1000>
+; CHECK: Cost of 0 for VF 1: EMIT branch-on-cond ir<%ec>
; CHECK: Cost of 0 for VF 2: vp<[[VP4:%[0-9]+]]> = SCALAR-STEPS vp<[[VP3:%[0-9]+]]>, ir<1>, vp<[[VP0:%[0-9]+]]>
; CHECK: Cost of 0 for VF 2: CLONE ir<%g.dst> = getelementptr inbounds ir<%dst>, vp<[[VP4]]>
; CHECK: Cost of 0 for VF 2: vp<[[VP5:%[0-9]+]]> = vector-pointer inbounds i64, ir<%g.dst>, ir<1>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/fneg-cost.ll b/llvm/test/Transforms/LoopVectorize/X86/fneg-cost.ll
index 693c6e5732d376..81fade6412464c 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/fneg-cost.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/fneg-cost.ll
@@ -5,10 +5,49 @@
target datalayout = "e-p:64:64:64-i1:8:8-i8:8:8-i16:16:16-i32:32:32-i64:64:64-f32:32:32-f64:64:64-v64:64:64-v128:128:128-a0:0:64-s0:64:64-f80:128:128-n8:16:32:64-S128"
target triple = "x86_64-apple-macosx10.8.0"
-; CHECK: Found an estimated cost of 1 for VF 1 For instruction: %neg = fneg float %{{.*}}
+; CHECK: Cost of 1 for VF 1: EMIT ir<%neg> = fneg ir<%0>
; CHECK: Cost of 1 for VF 2: WIDEN ir<%neg> = fneg ir<%0>
; CHECK: Cost of 1 for VF 4: WIDEN ir<%neg> = fneg ir<%0>
define void @fneg_cost(ptr %a, i64 %n) {
+; CHECK-LABEL: @fneg_cost(
+; CHECK-NEXT: entry:
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N:%.*]], 8
+; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label [[SCALAR_PH:%.*]], label [[VECTOR_PH:%.*]]
+; CHECK: vector.ph:
+; CHECK-NEXT: [[TMP0:%.*]] = and i64 [[N]], 7
+; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
+; CHECK-NEXT: br label [[VECTOR_BODY:%.*]]
+; CHECK: vector.body:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, [[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], [[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP1:%.*]] = getelementptr inbounds float, ptr [[A:%.*]], i64 [[INDEX]]
+; CHECK-NEXT: [[TMP2:%.*]] = getelementptr inbounds float, ptr [[TMP1]], i64 4
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x float>, ptr [[TMP1]], align 4
+; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <4 x float>, ptr [[TMP2]], align 4
+; CHECK-NEXT: [[TMP3:%.*]] = fneg <4 x float> [[WIDE_LOAD]]
+; CHECK-NEXT: [[TMP4:%.*]] = fneg <4 x float> [[WIDE_LOAD1]]
+; CHECK-NEXT: store <4 x float> [[TMP3]], ptr [[TMP1]], align 4
+; CHECK-NEXT: store <4 x float> [[TMP4]], ptr [[TMP2]], align 4
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
+; CHECK-NEXT: [[TMP5:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[TMP5]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK: middle.block:
+; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[CMP_N]], label [[FOR_END:%.*]], label [[SCALAR_PH]]
+; CHECK: scalar.ph:
+; CHECK-NEXT: [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], [[MIDDLE_BLOCK]] ], [ 0, [[ENTRY:%.*]] ]
+; CHECK-NEXT: br label [[FOR_BODY:%.*]]
+; CHECK: for.body:
+; CHECK-NEXT: [[INDVARS_IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], [[SCALAR_PH]] ], [ [[INDVARS_IV_NEXT:%.*]], [[FOR_BODY]] ]
+; CHECK-NEXT: [[ARRAYIDX:%.*]] = getelementptr inbounds float, ptr [[A]], i64 [[INDVARS_IV]]
+; CHECK-NEXT: [[TMP6:%.*]] = load float, ptr [[ARRAYIDX]], align 4
+; CHECK-NEXT: [[NEG:%.*]] = fneg float [[TMP6]]
+; CHECK-NEXT: store float [[NEG]], ptr [[ARRAYIDX]], align 4
+; CHECK-NEXT: [[INDVARS_IV_NEXT]] = add nuw nsw i64 [[INDVARS_IV]], 1
+; CHECK-NEXT: [[CMP:%.*]] = icmp eq i64 [[INDVARS_IV_NEXT]], [[N]]
+; CHECK-NEXT: br i1 [[CMP]], label [[FOR_END]], label [[FOR_BODY]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK: for.end:
+; CHECK-NEXT: ret void
+;
entry:
br label %for.body
for.body:
diff --git a/llvm/test/Transforms/LoopVectorize/X86/fp_to_sint8-cost-model.ll b/llvm/test/Transforms/LoopVectorize/X86/fp_to_sint8-cost-model.ll
index 38662644e26123..7140ed86f15dc9 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/fp_to_sint8-cost-model.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/fp_to_sint8-cost-model.ll
@@ -5,7 +5,7 @@ target datalayout = "e-p:64:64:64-i1:8:8-i8:8:8-i16:16:16-i32:32:32-i64:64:64-f3
target triple = "x86_64-apple-macosx10.8.0"
-; CHECK: cost of 1 for VF 1 For instruction: %conv = fptosi float %tmp to i8
+; CHECK: Cost of 1 for VF 1: EMIT-SCALAR ir<%conv> = fptosi ir<%tmp> to i8
define void @float_to_sint8_cost(ptr noalias nocapture %a, ptr noalias nocapture readonly %b) {
entry:
br label %for.body
diff --git a/llvm/test/Transforms/LoopVectorize/X86/reduction-small-size.ll b/llvm/test/Transforms/LoopVectorize/X86/reduction-small-size.ll
index c51b7ff07e19ec..4b4560f144ffae 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/reduction-small-size.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/reduction-small-size.ll
@@ -13,21 +13,21 @@ target datalayout = "e-m:e-i64:64-f80:128-n8:16:32:64-S128"
;
; CHECK-LABEL: reduction_i8
-; CHECK: LV: Found an estimated cost of {{[0-9]+}} for VF 1 For instruction: %{{.*}} = phi
-; CHECK: LV: Found an estimated cost of {{[0-9]+}} for VF 1 For instruction: %{{.*}} = phi
-; CHECK: LV: Found an estimated cost of {{[0-9]+}} for VF 1 For instruction: %{{.*}} = getelementptr
-; CHECK: LV: Found an estimated cost of {{[0-9]+}} for VF 1 For instruction: %{{.*}} = load
-; CHECK: LV: Found an estimated cost of {{[0-9]+}} for VF 1 For instruction: %{{.*}} = zext i8 %{{.*}} to i32
-; CHECK: LV: Found an estimated cost of {{[0-9]+}} for VF 1 For instruction: %{{.*}} = getelementptr
-; CHECK: LV: Found an estimated cost of {{[0-9]+}} for VF 1 For instruction: %{{.*}} = load
-; CHECK: LV: Found an estimated cost of {{[0-9]+}} for VF 1 For instruction: %{{.*}} = zext i8 %{{.*}} to i32
-; CHECK: LV: Found an estimated cost of {{[0-9]+}} for VF 1 For instruction: %{{.*}} = and i32 %{{.*}}, 255
-; CHECK: LV: Found an estimated cost of {{[0-9]+}} for VF 1 For instruction: %{{.*}} = add
-; CHECK: LV: Found an estimated cost of {{[0-9]+}} for VF 1 For instruction: %{{.*}} = add
-; CHECK: LV: Found an estimated cost of {{[0-9]+}} for VF 1 For instruction: %{{.*}} = add
-; CHECK: LV: Found an estimated cost of {{[0-9]+}} for VF 1 For instruction: %{{.*}} = trunc
-; CHECK: LV: Found an estimated cost of {{[0-9]+}} for VF 1 For instruction: %{{.*}} = icmp
-; CHECK: LV: Found an estimated cost of {{[0-9]+}} for VF 1 For instruction: br
+; CHECK: Cost of {{[0-9]+}} for VF 1: EMIT-SCALAR ir<%indvars.iv> = phi
+; CHECK: Cost of {{[0-9]+}} for VF 1: EMIT-SCALAR ir<%sum.013> = phi
+; CHECK: Cost of {{[0-9]+}} for VF 1: EMIT ir<%arrayidx> = getelementptr
+; CHECK: Cost of {{[0-9]+}} for VF 1: EMIT-SCALAR ir<%0> = load
+; CHECK: Cost of {{[0-9]+}} for VF 1: EMIT-SCALAR ir<%conv> = zext ir<%0> to i32
+; CHECK: Cost of {{[0-9]+}} for VF 1: EMIT ir<%arrayidx2> = getelementptr
+; CHECK: Cost of {{[0-9]+}} for VF 1: EMIT-SCALAR ir<%1> = load
+; CHECK: Cost of {{[0-9]+}} for VF 1: EMIT-SCALAR ir<%conv3> = zext ir<%1> to i32
+; CHECK: Cost of {{[0-9]+}} for VF 1: EMIT ir<%conv4> = and ir<%sum.013>, ir<255>
+; CHECK: Cost of {{[0-9]+}} for VF 1: EMIT ir<%add> = add
+; CHECK: Cost of {{[0-9]+}} for VF 1: EMIT ir<%add5> = add
+; CHECK: Cost of {{[0-9]+}} for VF 1: EMIT ir<%indvars.iv.next> = add
+; CHECK: Cost of {{[0-9]+}} for VF 1: EMIT-SCALAR ir<%lftr.wideiv> = trunc
+; CHECK: Cost of {{[0-9]+}} for VF 1: EMIT ir<%exitcond> = icmp
+; CHECK: Cost of {{[0-9]+}} for VF 1: EMIT branch-on-cond
; CHECK: Cost of 1 for VF 2: WIDEN-REDUCTION-PHI ir<%sum.013> = phi (add) vp<{{.+}}>, vp<[[EXT:%.+]]>
; CHECK: Cost of 0 for VF 2: vp<[[STEPS:%.+]]> = SCALAR-STEPS vp<[[CAN_IV:%.+]]>, ir<1>
; CHECK: Cost of 0 for VF 2: CLONE ir<%arrayidx> = getelementptr inbounds ir<%a>, vp<[[STEPS]]>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/uint64_to_fp64-cost-model.ll b/llvm/test/Transforms/LoopVectorize/X86/uint64_to_fp64-cost-model.ll
index 0edb89af5bc545..f124389e7eaa4d 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/uint64_to_fp64-cost-model.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/uint64_to_fp64-cost-model.ll
@@ -5,7 +5,7 @@ target datalayout = "e-p:64:64:64-i1:8:8-i8:8:8-i16:16:16-i32:32:32-i64:64:64-f3
target triple = "x86_64-apple-macosx10.8.0"
-; CHECK: cost of 4 for VF 1 For instruction: %conv = uitofp i64 %tmp to double
+; CHECK: Cost of 4 for VF 1: EMIT-SCALAR ir<%conv> = uitofp ir<%tmp> to double
; CHECK: Cost of 5 for VF 2: WIDEN-CAST ir<%conv> = uitofp ir<%tmp> to double
; CHECK: Cost of 10 for VF 4: WIDEN-CAST ir<%conv> = uitofp ir<%tmp> to double
define void @uint64_to_double_cost(ptr noalias nocapture %a, ptr noalias nocapture readonly %b) {
diff --git a/llvm/test/Transforms/LoopVectorize/X86/uniformshift.ll b/llvm/test/Transforms/LoopVectorize/X86/uniformshift.ll
index 61eaa3524fa569..e38ba170eb6029 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/uniformshift.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/uniformshift.ll
@@ -2,7 +2,7 @@
; REQUIRES: asserts
; CHECK: 'foo'
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %shift = ashr i32 %val, %k
+; CHECK: Cost of 1 for VF 1: EMIT ir<%shift> = ashr ir<%val>, ir<%k>
; CHECK: Cost of 2 for VF 2: WIDEN ir<%shift> = ashr ir<%val>, ir<%k>
; CHECK: Cost of 2 for VF 4: WIDEN ir<%shift> = ashr ir<%val>, ir<%k>
define void @foo(ptr nocapture %p, i32 %k) {
diff --git a/llvm/test/Transforms/LoopVectorize/X86/vector-scalar-select-cost.ll b/llvm/test/Transforms/LoopVectorize/X86/vector-scalar-select-cost.ll
index 8acdabd6d675e3..bfa3e59df03c72 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/vector-scalar-select-cost.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/vector-scalar-select-cost.ll
@@ -22,7 +22,7 @@ define void @scalarselect(i1 %cond) {
%6 = add nsw i32 %5, %3
%7 = getelementptr inbounds [2048 x i32], ptr @a, i64 0, i64 %indvars.iv
-; CHECK: cost of 1 for VF 1 {{.*}} select i1 %cond, i32 %6, i32 0
+; CHECK: Cost of 1 for VF 1: EMIT ir<%sel> = select ir<%cond>, ir<%6>, ir<0>
; CHECK: Cost of 2 for VF 2: WIDEN ir<%sel> = select ir<%cond>, ir<%6>, ir<0>
; CHECK: Cost of 2 for VF 4: WIDEN ir<%sel> = select ir<%cond>, ir<%6>, ir<0>
@@ -51,7 +51,7 @@ define void @vectorselect(i1 %cond) {
%7 = getelementptr inbounds [2048 x i32], ptr @a, i64 0, i64 %indvars.iv
%8 = icmp ult i64 %indvars.iv, 8
-; CHECK: cost of 1 for VF 1 {{.*}} select i1 %8, i32 %6, i32 0
+; CHECK: Cost of 1 for VF 1: EMIT ir<%sel> = select ir<%8>, ir<%6>, ir<0>
; CHECK: Cost of 2 for VF 2: WIDEN ir<%sel> = select ir<%8>, ir<%6>, ir<0>
; CHECK: Cost of 2 for VF 4: WIDEN ir<%sel> = select ir<%8>, ir<%6>, ir<0>
>From 1de4f00f3de5bf43171bcc49a3053ec3b42058cb Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Tue, 16 Jun 2026 22:08:31 +0200
Subject: [PATCH 2/7] !fixup address comments, thanks
---
llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp | 5 ++---
1 file changed, 2 insertions(+), 3 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
index 0304127b9907c0..b752072c71f5fb 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
@@ -1178,9 +1178,8 @@ InstructionCost VPRecipeWithIRFlags::getCostForRecipeWithOpcode(
case Instruction::Xor: {
// Certain instructions can be cheaper if they have a constant second
// operand. One example of this are shifts on x86.
- TargetTransformInfo::OperandValueInfo RHSInfo = {
- TargetTransformInfo::OK_AnyValue, TargetTransformInfo::OP_None};
- if (Opcode != Instruction::FNeg) {
+ TargetTransformInfo::OperandValueInfo RHSInfo;
+ if (getNumOperands() == 2) {
RHSInfo = Ctx.getOperandInfo(getOperand(1));
if (RHSInfo.Kind == TargetTransformInfo::OK_AnyValue &&
getOperand(1)->isDefinedOutsideLoopRegions())
>From 2e6276438d4a418de63b05a9ce2c5f3f875101b7 Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Wed, 9 Sep 2026 14:01:30 +0100
Subject: [PATCH 3/7] !fixup update after merge
---
.../Transforms/Vectorize/LoopVectorize.cpp | 18 ++++--------
.../lib/Transforms/Vectorize/VPlanRecipes.cpp | 29 ++++++++++---------
2 files changed, 21 insertions(+), 26 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 7248923bf84761..9fbb03482e6dcd 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -5565,18 +5565,11 @@ InstructionCost LoopVectorizationPlanner::computeScalarCost() const {
InstructionCost Cost = 0;
for (VPBasicBlock *VPBB : vp_rpo_plain_cfg_loop_body(Header)) {
- // Look up the divisor via the first underlying IR instruction in the loop.
- uint64_t Divisor = 1;
- for (const VPRecipeBase &R : *VPBB) {
- auto *UI = dyn_cast_if_present<Instruction>(
- cast<VPSingleDefRecipe>(&R)->getUnderlyingValue());
- if (!UI)
- continue;
- Divisor = CostCtx.CM.getPredBlockCostDivisor(CostCtx.CostKind,
- UI->getParent());
- break;
- }
- Cost += VPBB->cost(ScalarVF, CostCtx) / Divisor;
+ // In the scalar loop, we may not always execute the predicated block, if
+ // it is an if-else block. Thus, scale the block's cost by the probability
+ // of executing it.
+ Cost += VPBB->cost(ScalarVF, CostCtx) /
+ CostCtx.getCostDivisor(getRecordedExecutionFrequency(VPBB));
}
return Cost;
}
@@ -6379,7 +6372,6 @@ static bool verifyExecutionFrequenciesMatchBFI(VPlan &Plan, Loop *OrigLoop,
for (const auto &[VPBB, BB] :
zip_equal(drop_begin(Blocks), drop_begin(OrigRPO))) {
- // Nothing to check for blocks without a recorded frequency.
std::optional<VPExecutionFrequency> Freq =
getRecordedExecutionFrequency(VPBB);
if (!Freq)
diff --git a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
index b752072c71f5fb..e1e55111ddd2a3 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
@@ -1177,9 +1177,10 @@ InstructionCost VPRecipeWithIRFlags::getCostForRecipeWithOpcode(
case Instruction::Or:
case Instruction::Xor: {
// Certain instructions can be cheaper if they have a constant second
- // operand. One example of this are shifts on x86.
+ // operand. One example of this are shifts on x86. FNeg is the only unary
+ // opcode handled here and has no second operand.
TargetTransformInfo::OperandValueInfo RHSInfo;
- if (getNumOperands() == 2) {
+ if (Opcode != Instruction::FNeg) {
RHSInfo = Ctx.getOperandInfo(getOperand(1));
if (RHSInfo.Kind == TargetTransformInfo::OK_AnyValue &&
getOperand(1)->isDefinedOutsideLoopRegions())
@@ -1214,8 +1215,7 @@ InstructionCost VPRecipeWithIRFlags::getCostForRecipeWithOpcode(
return Ctx.TTI.getAddressComputationCost(PtrTy, nullptr, nullptr,
Ctx.CostKind) +
Ctx.TTI.getMemoryOpCost(Opcode, ValTy, getLoadStoreAlignment(UI),
- cast<PointerType>(PtrTy)->getAddressSpace(),
- Ctx.CostKind,
+ getLoadStoreAddressSpace(UI), Ctx.CostKind,
TTI::getOperandInfo(UI->getOperand(0)), UI);
}
case Instruction::ICmp:
@@ -1262,10 +1262,10 @@ InstructionCost VPRecipeWithIRFlags::getCostForRecipeWithOpcode(
}
// Loads/stores in pre-predication VPlan0 are represented as
// VPInstructions; treat them like an unmasked memory access.
- if (const auto *VPI = dyn_cast<VPInstruction>(R))
- if (VPI->getOpcode() == Instruction::Load ||
- VPI->getOpcode() == Instruction::Store)
- return TTI::CastContextHint::Normal;
+ const auto *VPI = dyn_cast<VPInstruction>(R);
+ if (VPI && (VPI->getOpcode() == Instruction::Load ||
+ VPI->getOpcode() == Instruction::Store))
+ return TTI::CastContextHint::Normal;
const auto *WidenMemoryRecipe = dyn_cast<VPWidenMemoryRecipe>(R);
if (WidenMemoryRecipe == nullptr)
return TTI::CastContextHint::None;
@@ -1376,11 +1376,12 @@ InstructionCost VPRecipeWithIRFlags::getCostForRecipeWithOpcode(
InstructionCost VPInstruction::computeCost(ElementCount VF,
VPCostContext &Ctx) const {
- // Vector-only opcodes have zero cost at scalar VF.
- if (VF.isScalar() &&
- (isVectorToScalar() ||
- getOpcode() == VPInstruction::FirstOrderRecurrenceSplice))
- return 0;
+ // A scalar cost is only computed for VPlan0, which has no vector-only
+ // opcodes.
+ assert(!(VF.isScalar() &&
+ (isVectorToScalar() ||
+ getOpcode() == VPInstruction::FirstOrderRecurrenceSplice)) &&
+ "unexpected vector-only opcode at scalar VF");
// NOTE: At the moment it seems only possible to expose this path for
// the trunc, zext and sext opcodes.
@@ -1610,6 +1611,8 @@ InstructionCost VPInstruction::computeCost(ElementCount VF,
default:
// TODO: Compute cost other VPInstructions once the legacy cost model has
// been retired.
+ assert((VF.isScalar() || !getUnderlyingValue()) &&
+ "unexpected VPInstruction with underlying value");
return 0;
}
}
>From 218a98db94163d555b0935f333e46a4a8bd82b83 Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Sun, 20 Sep 2026 22:58:21 +0100
Subject: [PATCH 4/7] !fixup update UpdateTestChecks expected output
---
.../Inputs/x86-loopvectorize-costmodel.ll.expected | 1 -
1 file changed, 1 deletion(-)
diff --git a/llvm/test/tools/UpdateTestChecks/update_analyze_test_checks/Inputs/x86-loopvectorize-costmodel.ll.expected b/llvm/test/tools/UpdateTestChecks/update_analyze_test_checks/Inputs/x86-loopvectorize-costmodel.ll.expected
index 3abf35ff0feba7..88911d7440d382 100644
--- a/llvm/test/tools/UpdateTestChecks/update_analyze_test_checks/Inputs/x86-loopvectorize-costmodel.ll.expected
+++ b/llvm/test/tools/UpdateTestChecks/update_analyze_test_checks/Inputs/x86-loopvectorize-costmodel.ll.expected
@@ -10,7 +10,6 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; CHECK-LABEL: 'test'
-; CHECK: LV: Found an estimated cost of 1 for VF 1 For instruction: %v0 = load float, ptr %in0, align 4
; CHECK: Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
; CHECK: Cost of 3 for VF 4: INTERLEAVE-GROUP with factor 2, ir<%in0>
; CHECK: Cost of 3 for VF 8: INTERLEAVE-GROUP with factor 2, ir<%in0>
>From 8fe9f65d4b7c24b5c4bd26cdf87df70f3073f90f Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Mon, 21 Sep 2026 11:18:04 +0100
Subject: [PATCH 5/7] !fixup restore some VF = 1 check lines
---
.../LoopVectorize/AArch64/intrinsiccost.ll | 9 +++++
.../LoopVectorize/ARM/mve-icmpcost.ll | 22 +++++-----
.../X86/CostModel/gather-i16-with-i8-index.ll | 2 +-
.../X86/CostModel/gather-i32-with-i8-index.ll | 2 +-
.../X86/CostModel/gather-i64-with-i8-index.ll | 2 +-
.../X86/CostModel/gather-i8-with-i8-index.ll | 2 +-
...dle-iptr-with-data-layout-to-not-assert.ll | 3 +-
.../interleaved-load-f32-stride-2.ll | 6 ++-
.../interleaved-load-f32-stride-3.ll | 6 ++-
.../interleaved-load-f32-stride-4.ll | 6 ++-
.../interleaved-load-f32-stride-5.ll | 6 ++-
.../interleaved-load-f32-stride-6.ll | 6 ++-
.../interleaved-load-f32-stride-7.ll | 6 ++-
.../interleaved-load-f32-stride-8.ll | 6 ++-
.../interleaved-load-f64-stride-2.ll | 6 ++-
.../interleaved-load-f64-stride-3.ll | 6 ++-
.../interleaved-load-f64-stride-4.ll | 6 ++-
.../interleaved-load-f64-stride-5.ll | 6 ++-
.../interleaved-load-f64-stride-6.ll | 6 ++-
.../interleaved-load-f64-stride-7.ll | 6 ++-
.../interleaved-load-f64-stride-8.ll | 6 ++-
.../interleaved-load-i16-stride-2.ll | 7 +++-
.../interleaved-load-i16-stride-3.ll | 7 +++-
.../interleaved-load-i16-stride-4.ll | 7 +++-
.../interleaved-load-i16-stride-5.ll | 7 +++-
.../interleaved-load-i16-stride-6.ll | 7 +++-
.../interleaved-load-i16-stride-7.ll | 7 +++-
.../interleaved-load-i16-stride-8.ll | 7 +++-
...nterleaved-load-i32-stride-2-indices-0u.ll | 6 ++-
.../interleaved-load-i32-stride-2.ll | 6 ++-
...terleaved-load-i32-stride-3-indices-01u.ll | 6 ++-
...terleaved-load-i32-stride-3-indices-0uu.ll | 6 ++-
.../interleaved-load-i32-stride-3.ll | 6 ++-
...erleaved-load-i32-stride-4-indices-012u.ll | 6 ++-
...erleaved-load-i32-stride-4-indices-01uu.ll | 6 ++-
...erleaved-load-i32-stride-4-indices-0uuu.ll | 6 ++-
.../interleaved-load-i32-stride-4.ll | 6 ++-
.../interleaved-load-i32-stride-5.ll | 6 ++-
.../interleaved-load-i32-stride-6.ll | 6 ++-
.../interleaved-load-i32-stride-7.ll | 6 ++-
.../interleaved-load-i32-stride-8.ll | 6 ++-
.../interleaved-load-i64-stride-2.ll | 6 ++-
.../interleaved-load-i64-stride-3.ll | 6 ++-
.../interleaved-load-i64-stride-4.ll | 6 ++-
.../interleaved-load-i64-stride-5.ll | 6 ++-
.../interleaved-load-i64-stride-6.ll | 6 ++-
.../interleaved-load-i64-stride-7.ll | 6 ++-
.../interleaved-load-i64-stride-8.ll | 6 ++-
.../CostModel/interleaved-load-i8-stride-2.ll | 7 +++-
.../CostModel/interleaved-load-i8-stride-3.ll | 7 +++-
.../CostModel/interleaved-load-i8-stride-4.ll | 7 +++-
.../CostModel/interleaved-load-i8-stride-5.ll | 7 +++-
.../CostModel/interleaved-load-i8-stride-6.ll | 7 +++-
.../CostModel/interleaved-load-i8-stride-7.ll | 7 +++-
.../CostModel/interleaved-load-i8-stride-8.ll | 7 +++-
.../masked-gather-i32-with-i8-index.ll | 34 ++++++++--------
.../masked-gather-i64-with-i8-index.ll | 34 ++++++++--------
.../CostModel/masked-interleaved-load-i16.ll | 22 +++++++---
.../CostModel/masked-interleaved-store-i16.ll | 12 +++++-
.../X86/CostModel/masked-load-i16.ll | 2 +-
.../X86/CostModel/masked-load-i32.ll | 2 +-
.../X86/CostModel/masked-load-i64.ll | 2 +-
.../X86/CostModel/masked-load-i8.ll | 2 +-
.../masked-scatter-i32-with-i8-index.ll | 15 ++++---
.../masked-scatter-i64-with-i8-index.ll | 15 ++++---
.../X86/CostModel/masked-store-i16.ll | 6 ++-
.../X86/CostModel/masked-store-i32.ll | 7 +++-
.../X86/CostModel/masked-store-i64.ll | 7 +++-
.../X86/CostModel/masked-store-i8.ll | 7 +++-
.../CostModel/scatter-i16-with-i8-index.ll | 7 +++-
.../CostModel/scatter-i32-with-i8-index.ll | 7 +++-
.../CostModel/scatter-i64-with-i8-index.ll | 7 +++-
.../X86/CostModel/scatter-i8-with-i8-index.ll | 7 +++-
.../X86/CostModel/strided-load-i16.ll | 6 ++-
.../X86/CostModel/strided-load-i32.ll | 6 ++-
.../X86/CostModel/strided-load-i64.ll | 5 ++-
.../X86/CostModel/strided-load-i8.ll | 6 ++-
.../X86/CostModel/vpinstruction-cost.ll | 40 +++++++++----------
.../x86-loopvectorize-costmodel.ll.expected | 3 +-
.../loopvectorize-costmodel.test | 6 +--
80 files changed, 457 insertions(+), 154 deletions(-)
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/intrinsiccost.ll b/llvm/test/Transforms/LoopVectorize/AArch64/intrinsiccost.ll
index 958e7763aad3d6..69409fc55e03d0 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/intrinsiccost.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/intrinsiccost.ll
@@ -7,6 +7,10 @@ target datalayout = "e-m:e-i8:8:32-i16:16:32-i64:64-i128:128-n32:64-S128"
target triple = "aarch64--linux-gnu"
; CHECK-COST-LABEL: sadd
+; CHECK-COST: Cost of 6 for VF 1: EMIT ir<%1> = call ir<%0>, ir<%offset>, ir<@llvm.sadd.sat.i16>
+; CHECK-COST: Cost of 4 for VF 2: WIDEN-INTRINSIC ir<%1> = call llvm.sadd.sat(ir<%0>, ir<%offset>)
+; CHECK-COST: Cost of 1 for VF 4: WIDEN-INTRINSIC ir<%1> = call llvm.sadd.sat(ir<%0>, ir<%offset>)
+; CHECK-COST: Cost of 1 for VF 8: WIDEN-INTRINSIC ir<%1> = call llvm.sadd.sat(ir<%0>, ir<%offset>)
define void @saddsat(ptr nocapture readonly %pSrc, i16 signext %offset, ptr nocapture noalias %pDst, i32 %blockSize) #0 {
; CHECK-LABEL: @saddsat(
@@ -123,6 +127,11 @@ while.end:
}
; CHECK-COST-LABEL: umin
+; CHECK-COST: Cost of 2 for VF 1: EMIT ir<%1> = call ir<%0>, ir<%offset>, ir<@llvm.umin.i8>
+; CHECK-COST: Cost of 3 for VF 2: WIDEN-INTRINSIC ir<%1> = call llvm.umin(ir<%0>, ir<%offset>)
+; CHECK-COST: Cost of 3 for VF 4: WIDEN-INTRINSIC ir<%1> = call llvm.umin(ir<%0>, ir<%offset>)
+; CHECK-COST: Cost of 1 for VF 8: WIDEN-INTRINSIC ir<%1> = call llvm.umin(ir<%0>, ir<%offset>)
+; CHECK-COST: Cost of 1 for VF 16: WIDEN-INTRINSIC ir<%1> = call llvm.umin(ir<%0>, ir<%offset>)
define void @umin(ptr nocapture readonly %pSrc, i8 signext %offset, ptr nocapture noalias %pDst, i32 %blockSize) #0 {
; CHECK-LABEL: @umin(
diff --git a/llvm/test/Transforms/LoopVectorize/ARM/mve-icmpcost.ll b/llvm/test/Transforms/LoopVectorize/ARM/mve-icmpcost.ll
index aa6ec753d92764..a503617f6f47ad 100644
--- a/llvm/test/Transforms/LoopVectorize/ARM/mve-icmpcost.ll
+++ b/llvm/test/Transforms/LoopVectorize/ARM/mve-icmpcost.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost for VF" --filter "LV: Selecting VF" --filter "Found an estimated cost of .* for VF 1" --filter "Cost of .* for VF" --version 6
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost for VF" --filter "LV: Selecting VF" --filter "Cost of .* for VF" --version 6
; RUN: opt -passes=loop-vectorize -debug-only=loop-vectorize -disable-output < %s 2>&1 | FileCheck %s
; REQUIRES: asserts
@@ -12,10 +12,10 @@ define void @expensive_icmp(ptr noalias nocapture %d, ptr nocapture readonly %s,
; CHECK: Cost of 1 for VF 1: EMIT-SCALAR ir<%1> = load ir<%arrayidx>
; CHECK: Cost of 0 for VF 1: EMIT-SCALAR ir<%conv> = sext ir<%1> to i32
; CHECK: Cost of 1 for VF 1: EMIT ir<%cmp2> = icmp sgt ir<%conv>, ir<%conv1>
-; CHECK: Cost of 0 for VF 1: EMIT branch-on-cond ir<%cmp2> (!vplan.prof.estimated estimated {1073741824, 1073741824})
-; CHECK: Cost of 1 for VF 1: EMIT ir<%conv6> = add ir<%1>, ir<%0> (!vplan.execution.frequency 4611686018427387904 (50%, estimated))
-; CHECK: Cost of 0 for VF 1: EMIT ir<%arrayidx7> = getelementptr inbounds ir<%d>, ir<%i.016> (!vplan.execution.frequency 4611686018427387904 (50%, estimated))
-; CHECK: Cost of 1 for VF 1: EMIT store ir<%conv6>, ir<%arrayidx7> (!vplan.execution.frequency 4611686018427387904 (50%, estimated))
+; CHECK: Cost of 0 for VF 1: EMIT branch-on-cond ir<%cmp2>
+; CHECK: Cost of 1 for VF 1: EMIT ir<%conv6> = add ir<%1>, ir<%0>
+; CHECK: Cost of 0 for VF 1: EMIT ir<%arrayidx7> = getelementptr inbounds ir<%d>, ir<%i.016>
+; CHECK: Cost of 1 for VF 1: EMIT store ir<%conv6>, ir<%arrayidx7>
; CHECK: Cost of 1 for VF 1: EMIT ir<%inc> = add nuw nsw ir<%i.016>, ir<1>
; CHECK: Cost of 1 for VF 1: EMIT ir<%exitcond.not> = icmp eq ir<%inc>, ir<%n>
; CHECK: Cost of 0 for VF 1: EMIT branch-on-cond ir<%exitcond.not>
@@ -25,10 +25,10 @@ define void @expensive_icmp(ptr noalias nocapture %d, ptr nocapture readonly %s,
; CHECK: Cost of 18 for VF 2: WIDEN ir<%1> = load vp<[[VP5]]>
; CHECK: Cost of 4 for VF 2: WIDEN-CAST ir<%conv> = sext ir<%1> to i32
; CHECK: Cost of 20 for VF 2: WIDEN ir<%cmp2> = icmp sgt ir<%conv>, ir<%conv1>
-; CHECK: Cost of 26 for VF 2: WIDEN ir<%conv6> = add ir<%1>, ir<%0> (!vplan.execution.frequency 4611686018427387904 (50%, estimated))
+; CHECK: Cost of 26 for VF 2: WIDEN ir<%conv6> = add ir<%1>, ir<%0>
; CHECK: Cost of 0 for VF 2: CLONE ir<%arrayidx7> = getelementptr ir<%d>, vp<[[VP4]]>
; CHECK: Cost of 0 for VF 2: vp<[[VP6:%[0-9]+]]> = vector-pointer i16, ir<%arrayidx7>, ir<1>
-; CHECK: Cost of 16 for VF 2: WIDEN store vp<[[VP6]]>, ir<%conv6>, ir<%cmp2> (!vplan.execution.frequency 4611686018427387904 (50%, estimated))
+; CHECK: Cost of 16 for VF 2: WIDEN store vp<[[VP6]]>, ir<%conv6>, ir<%cmp2>
; CHECK: Cost of 0 for VF 2: EMIT vp<%index.next> = add nuw vp<[[VP3]]>, vp<[[VP1:%[0-9]+]]>
; CHECK: Cost of 1 for VF 2: EMIT branch-on-count vp<%index.next>, vp<[[VP2:%[0-9]+]]>
; CHECK: Cost of 0 for VF 2: vector loop backedge
@@ -50,10 +50,10 @@ define void @expensive_icmp(ptr noalias nocapture %d, ptr nocapture readonly %s,
; CHECK: Cost of 2 for VF 4: WIDEN ir<%1> = load vp<[[VP5]]>
; CHECK: Cost of 0 for VF 4: WIDEN-CAST ir<%conv> = sext ir<%1> to i32
; CHECK: Cost of 2 for VF 4: WIDEN ir<%cmp2> = icmp sgt ir<%conv>, ir<%conv1>
-; CHECK: Cost of 2 for VF 4: WIDEN ir<%conv6> = add ir<%1>, ir<%0> (!vplan.execution.frequency 4611686018427387904 (50%, estimated))
+; CHECK: Cost of 2 for VF 4: WIDEN ir<%conv6> = add ir<%1>, ir<%0>
; CHECK: Cost of 0 for VF 4: CLONE ir<%arrayidx7> = getelementptr ir<%d>, vp<[[VP4]]>
; CHECK: Cost of 0 for VF 4: vp<[[VP6]]> = vector-pointer i16, ir<%arrayidx7>, ir<1>
-; CHECK: Cost of 2 for VF 4: WIDEN store vp<[[VP6]]>, ir<%conv6>, ir<%cmp2> (!vplan.execution.frequency 4611686018427387904 (50%, estimated))
+; CHECK: Cost of 2 for VF 4: WIDEN store vp<[[VP6]]>, ir<%conv6>, ir<%cmp2>
; CHECK: Cost of 0 for VF 4: EMIT vp<%index.next> = add nuw vp<[[VP3]]>, vp<[[VP1]]>
; CHECK: Cost of 1 for VF 4: EMIT branch-on-count vp<%index.next>, vp<[[VP2]]>
; CHECK: Cost of 0 for VF 4: vector loop backedge
@@ -75,10 +75,10 @@ define void @expensive_icmp(ptr noalias nocapture %d, ptr nocapture readonly %s,
; CHECK: Cost of 2 for VF 8: WIDEN ir<%1> = load vp<[[VP5]]>
; CHECK: Cost of 2 for VF 8: WIDEN-CAST ir<%conv> = sext ir<%1> to i32
; CHECK: Cost of 36 for VF 8: WIDEN ir<%cmp2> = icmp sgt ir<%conv>, ir<%conv1>
-; CHECK: Cost of 2 for VF 8: WIDEN ir<%conv6> = add ir<%1>, ir<%0> (!vplan.execution.frequency 4611686018427387904 (50%, estimated))
+; CHECK: Cost of 2 for VF 8: WIDEN ir<%conv6> = add ir<%1>, ir<%0>
; CHECK: Cost of 0 for VF 8: CLONE ir<%arrayidx7> = getelementptr ir<%d>, vp<[[VP4]]>
; CHECK: Cost of 0 for VF 8: vp<[[VP6]]> = vector-pointer i16, ir<%arrayidx7>, ir<1>
-; CHECK: Cost of 2 for VF 8: WIDEN store vp<[[VP6]]>, ir<%conv6>, ir<%cmp2> (!vplan.execution.frequency 4611686018427387904 (50%, estimated))
+; CHECK: Cost of 2 for VF 8: WIDEN store vp<[[VP6]]>, ir<%conv6>, ir<%cmp2>
; CHECK: Cost of 0 for VF 8: EMIT vp<%index.next> = add nuw vp<[[VP3]]>, vp<[[VP1]]>
; CHECK: Cost of 1 for VF 8: EMIT branch-on-count vp<%index.next>, vp<[[VP2]]>
; CHECK: Cost of 0 for VF 8: vector loop backedge
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i16-with-i8-index.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i16-with-i8-index.ll
index 2770dd4801aebc..0af968f8b946a8 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i16-with-i8-index.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i16-with-i8-index.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF [0-9]+ For instruction:\s*%valB = load i16, ptr %inB, align 2" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: .* ir<%valB> = load"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: .* ir<%valB> = load"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=SSE
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse4.2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=SSE
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=AVX1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i32-with-i8-index.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i32-with-i8-index.ll
index 58b96771aedbea..b14c2b701eba7c 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i32-with-i8-index.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i32-with-i8-index.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF [0-9]+ For instruction:\s*%valB = load i32, ptr %inB, align 4" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: .* ir<%valB> = load"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: .* ir<%valB> = load"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=SSE2
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse4.2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=SSE42
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=AVX1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i64-with-i8-index.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i64-with-i8-index.ll
index 7876551023ee67..c4b9254e4abae2 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i64-with-i8-index.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i64-with-i8-index.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF [0-9]+ For instruction:\s*%valB = load i64, ptr %inB, align 8" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: .* ir<%valB> = load"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: .* ir<%valB> = load"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=SSE2
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse4.2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=SSE42
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=AVX1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i8-with-i8-index.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i8-with-i8-index.ll
index 435f7c3becd730..e3b5dc06878597 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i8-with-i8-index.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i8-with-i8-index.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF [0-9]+ For instruction:\s*%valB = load i8, ptr %inB, align 1" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: .* ir<%valB> = load"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: .* ir<%valB> = load"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=SSE2
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse4.2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=SSE42
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=AVX1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/handle-iptr-with-data-layout-to-not-assert.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/handle-iptr-with-data-layout-to-not-assert.ll
index 1c5e5d5aa21a3d..0c8b59417645cc 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/handle-iptr-with-data-layout-to-not-assert.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/handle-iptr-with-data-layout-to-not-assert.ll
@@ -1,10 +1,11 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*store ptr" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: INTERLEAVE-GROUP with factor [0-9]+" --version 5
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT store ir<%0>" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: INTERLEAVE-GROUP with factor [0-9]+" --version 5
; REQUIRES: asserts
; RUN: opt -passes=loop-vectorize -debug-only=loop-vectorize -S < %s 2>&1 | FileCheck %s
target triple = "x86_64-unknown-linux-gnu"
define ptr @foo(ptr %__first, ptr %__last) #0 {
; CHECK-LABEL: 'foo'
+; CHECK: Cost of 1 for VF 1: EMIT store ir<%0>, ir<%__last>
; CHECK: Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 2, vp<%next.gep>
; CHECK: Cost of 1 for VF 4: INTERLEAVE-GROUP with factor 2, vp<%next.gep>
; CHECK: Cost of 3 for VF 8: INTERLEAVE-GROUP with factor 2, vp<%next.gep>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-2.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-2.ll
index fe3fbb9c4bfee8..306975a7db84de 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-2.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-2.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,6 +13,7 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
+; SSE2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; SSE2: Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
; SSE2: ir<%v0> = load from index 0
; SSE2: ir<%v1> = load from index 1
@@ -23,6 +24,7 @@ define void @test() {
; SSE2: Cost of 28 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
+; AVX1: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX1: Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
; AVX1: ir<%v0> = load from index 0
; AVX1: ir<%v1> = load from index 1
@@ -34,6 +36,7 @@ define void @test() {
; AVX1: Cost of 60 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
+; AVX2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX2: Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
; AVX2: ir<%v0> = load from index 0
; AVX2: ir<%v1> = load from index 1
@@ -51,6 +54,7 @@ define void @test() {
; AVX2: ir<%v1> = load from index 1
;
; AVX512-LABEL: 'test'
+; AVX512: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX512: Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-3.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-3.ll
index 24ac329df20d63..f96e7096dd8012 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-3.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-3.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,12 +13,14 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
+; SSE2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; SSE2: Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 14 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 28 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
+; AVX1: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX1: Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 15 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -26,6 +28,7 @@ define void @test() {
; AVX1: Cost of 60 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
+; AVX2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX2: Cost of 6 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
; AVX2: ir<%v0> = load from index 0
; AVX2: ir<%v1> = load from index 1
@@ -48,6 +51,7 @@ define void @test() {
; AVX2: ir<%v2> = load from index 2
;
; AVX512-LABEL: 'test'
+; AVX512: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX512: Cost of 4 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-4.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-4.ll
index b2cd0273efd8de..e88f49360ea3d9 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-4.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-4.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,12 +13,14 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
+; SSE2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; SSE2: Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 14 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 28 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
+; AVX1: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX1: Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 15 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -26,6 +28,7 @@ define void @test() {
; AVX1: Cost of 60 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
+; AVX2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX2: Cost of 5 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
; AVX2: ir<%v0> = load from index 0
; AVX2: ir<%v1> = load from index 1
@@ -53,6 +56,7 @@ define void @test() {
; AVX2: ir<%v3> = load from index 3
;
; AVX512-LABEL: 'test'
+; AVX512: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX512: Cost of 5 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-5.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-5.ll
index f8980b2fb0f30d..3dc3b9cfd8f47e 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-5.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-5.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,12 +13,14 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
+; SSE2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; SSE2: Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 14 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 28 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
+; AVX1: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX1: Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 15 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -26,6 +28,7 @@ define void @test() {
; AVX1: Cost of 60 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
+; AVX2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX2: Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 15 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -33,6 +36,7 @@ define void @test() {
; AVX2: Cost of 60 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX512-LABEL: 'test'
+; AVX512: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX512: Cost of 6 for VF 2: INTERLEAVE-GROUP with factor 5, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-6.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-6.ll
index c62d8eea92143c..5af85931fcd35b 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-6.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-6.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,12 +13,14 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
+; SSE2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; SSE2: Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 14 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 28 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
+; AVX1: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX1: Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 15 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -26,6 +28,7 @@ define void @test() {
; AVX1: Cost of 60 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
+; AVX2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX2: Cost of 8 for VF 2: INTERLEAVE-GROUP with factor 6, ir<%in0>
; AVX2: ir<%v0> = load from index 0
; AVX2: ir<%v1> = load from index 1
@@ -57,6 +60,7 @@ define void @test() {
; AVX2: Cost of 60 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX512-LABEL: 'test'
+; AVX512: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX512: Cost of 7 for VF 2: INTERLEAVE-GROUP with factor 6, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-7.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-7.ll
index 68f7f233a8a86e..b9975cc07631d9 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-7.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-7.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,12 +13,14 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
+; SSE2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; SSE2: Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 14 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 28 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
+; AVX1: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX1: Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 15 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -26,6 +28,7 @@ define void @test() {
; AVX1: Cost of 60 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
+; AVX2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX2: Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 15 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -33,6 +36,7 @@ define void @test() {
; AVX2: Cost of 60 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX512-LABEL: 'test'
+; AVX512: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX512: Cost of 8 for VF 2: INTERLEAVE-GROUP with factor 7, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-8.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-8.ll
index 154c14dd895e0d..8348f83928791c 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-8.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f32-stride-8.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,12 +13,14 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
+; SSE2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; SSE2: Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 14 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 28 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
+; AVX1: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX1: Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 15 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -26,6 +28,7 @@ define void @test() {
; AVX1: Cost of 60 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
+; AVX2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX2: Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 48 for VF 8: INTERLEAVE-GROUP with factor 8, ir<%in0>
@@ -41,6 +44,7 @@ define void @test() {
; AVX2: Cost of 60 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX512-LABEL: 'test'
+; AVX512: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX512: Cost of 9 for VF 2: INTERLEAVE-GROUP with factor 8, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-2.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-2.ll
index c9ddd0a38f1038..8de4225d48e78b 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-2.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-2.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,6 +13,7 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
+; SSE2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; SSE2: Cost of 4 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
; SSE2: ir<%v0> = load from index 0
; SSE2: ir<%v1> = load from index 1
@@ -21,6 +22,7 @@ define void @test() {
; SSE2: Cost of 24 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
+; AVX1: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX1: Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
; AVX1: ir<%v0> = load from index 0
; AVX1: ir<%v1> = load from index 1
@@ -30,6 +32,7 @@ define void @test() {
; AVX1: Cost of 56 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
+; AVX2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX2: Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
; AVX2: ir<%v0> = load from index 0
; AVX2: ir<%v1> = load from index 1
@@ -47,6 +50,7 @@ define void @test() {
; AVX2: ir<%v1> = load from index 1
;
; AVX512-LABEL: 'test'
+; AVX512: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX512: Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-3.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-3.ll
index d42188eaa22996..02b0676ff5156a 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-3.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-3.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,12 +13,14 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
+; SSE2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; SSE2: Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 6 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 12 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 24 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
+; AVX1: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX1: Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 14 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -26,6 +28,7 @@ define void @test() {
; AVX1: Cost of 56 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
+; AVX2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX2: Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
; AVX2: ir<%v0> = load from index 0
; AVX2: ir<%v1> = load from index 1
@@ -45,6 +48,7 @@ define void @test() {
; AVX2: Cost of 56 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX512-LABEL: 'test'
+; AVX512: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX512: Cost of 4 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-4.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-4.ll
index 1bc77d8d3ff3c4..17ea3b4d11649d 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-4.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-4.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,12 +13,14 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
+; SSE2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; SSE2: Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 6 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 12 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 24 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
+; AVX1: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX1: Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 14 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -26,6 +28,7 @@ define void @test() {
; AVX1: Cost of 56 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
+; AVX2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX2: Cost of 8 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
; AVX2: ir<%v0> = load from index 0
; AVX2: ir<%v1> = load from index 1
@@ -49,6 +52,7 @@ define void @test() {
; AVX2: Cost of 56 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX512-LABEL: 'test'
+; AVX512: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX512: Cost of 5 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-5.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-5.ll
index 9992f5ad71ab42..ff4e1335914a9c 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-5.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-5.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load|WIDEN ir<%v[0-9]> = load)" --filter "^ ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load|WIDEN ir<%v[0-9]> = load)" --filter "^ ir<.* = load from index"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,12 +13,14 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
+; SSE2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; SSE2: Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 6 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 12 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 24 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
+; AVX1: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX1: Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 14 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -26,6 +28,7 @@ define void @test() {
; AVX1: Cost of 56 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
+; AVX2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX2: Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 14 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -33,6 +36,7 @@ define void @test() {
; AVX2: Cost of 56 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX512-LABEL: 'test'
+; AVX512: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX512: Cost of 14.5 for VF 2: INTERLEAVE-GROUP with factor 5, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-6.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-6.ll
index 4676e641166dd2..516a374ddd49a8 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-6.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-6.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load|WIDEN ir<%v[0-9]> = load)" --filter "^ ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load|WIDEN ir<%v[0-9]> = load)" --filter "^ ir<.* = load from index"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,12 +13,14 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
+; SSE2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; SSE2: Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 6 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 12 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 24 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
+; AVX1: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX1: Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 14 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -26,6 +28,7 @@ define void @test() {
; AVX1: Cost of 56 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
+; AVX2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX2: Cost of 9 for VF 2: INTERLEAVE-GROUP with factor 6, ir<%in0>
; AVX2: ir<%v0> = load from index 0
; AVX2: ir<%v1> = load from index 1
@@ -51,6 +54,7 @@ define void @test() {
; AVX2: Cost of 56 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX512-LABEL: 'test'
+; AVX512: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX512: Cost of 17 for VF 2: INTERLEAVE-GROUP with factor 6, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-7.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-7.ll
index fbbc94de0c1a91..27bc9169c7e115 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-7.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-7.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load|WIDEN ir<%v[0-9]> = load)" --filter "^ ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load|WIDEN ir<%v[0-9]> = load)" --filter "^ ir<.* = load from index"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,12 +13,14 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
+; SSE2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; SSE2: Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 6 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 12 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 24 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
+; AVX1: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX1: Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 14 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -26,6 +28,7 @@ define void @test() {
; AVX1: Cost of 56 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
+; AVX2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX2: Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 14 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -33,6 +36,7 @@ define void @test() {
; AVX2: Cost of 56 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX512-LABEL: 'test'
+; AVX512: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX512: Cost of 19.5 for VF 2: INTERLEAVE-GROUP with factor 7, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-8.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-8.ll
index 7c22d3e51be763..0a0c5695b5808b 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-8.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-8.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load|WIDEN ir<%v[0-9]> = load)" --filter "^ ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load|WIDEN ir<%v[0-9]> = load)" --filter "^ ir<.* = load from index"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,12 +13,14 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
+; SSE2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; SSE2: Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 6 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 12 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 24 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
+; AVX1: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX1: Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 14 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -26,6 +28,7 @@ define void @test() {
; AVX1: Cost of 56 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
+; AVX2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX2: Cost of 3 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 7 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 14 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -33,6 +36,7 @@ define void @test() {
; AVX2: Cost of 56 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX512-LABEL: 'test'
+; AVX512: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX512: Cost of 22 for VF 2: INTERLEAVE-GROUP with factor 8, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-2.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-2.ll
index 4fe1ae7f140c47..b7e8ce35dbc630 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-2.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-2.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -14,6 +14,7 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
+; SSE2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; SSE2: Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
; SSE2: ir<%v0> = load from index 0
; SSE2: ir<%v1> = load from index 1
@@ -24,6 +25,7 @@ define void @test() {
; SSE2: Cost of 32 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
+; AVX1: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX1: Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
; AVX1: ir<%v0> = load from index 0
; AVX1: ir<%v1> = load from index 1
@@ -35,6 +37,7 @@ define void @test() {
; AVX1: Cost of 66 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
+; AVX2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX2: Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
; AVX2: ir<%v0> = load from index 0
; AVX2: ir<%v1> = load from index 1
@@ -52,6 +55,7 @@ define void @test() {
; AVX2: ir<%v1> = load from index 1
;
; AVX512DQ-LABEL: 'test'
+; AVX512DQ: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX512DQ: Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
; AVX512DQ: ir<%v0> = load from index 0
; AVX512DQ: ir<%v1> = load from index 1
@@ -72,6 +76,7 @@ define void @test() {
; AVX512DQ: ir<%v1> = load from index 1
;
; AVX512BW-LABEL: 'test'
+; AVX512BW: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX512BW: Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
; AVX512BW: ir<%v0> = load from index 0
; AVX512BW: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-3.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-3.ll
index a7c0f2516d5ed9..25e0b8dae93e6c 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-3.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-3.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -14,12 +14,14 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
+; SSE2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; SSE2: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 32 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
+; AVX1: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX1: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -27,6 +29,7 @@ define void @test() {
; AVX1: Cost of 66 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
+; AVX2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX2: Cost of 7 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
; AVX2: ir<%v0> = load from index 0
; AVX2: ir<%v1> = load from index 1
@@ -49,6 +52,7 @@ define void @test() {
; AVX2: ir<%v2> = load from index 2
;
; AVX512DQ-LABEL: 'test'
+; AVX512DQ: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX512DQ: Cost of 7 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
; AVX512DQ: ir<%v0> = load from index 0
; AVX512DQ: ir<%v1> = load from index 1
@@ -75,6 +79,7 @@ define void @test() {
; AVX512DQ: ir<%v2> = load from index 2
;
; AVX512BW-LABEL: 'test'
+; AVX512BW: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX512BW: Cost of 4 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
; AVX512BW: ir<%v0> = load from index 0
; AVX512BW: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-4.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-4.ll
index 71bae49cbd9061..007f423e7a463a 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-4.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-4.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -14,12 +14,14 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
+; SSE2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; SSE2: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 32 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
+; AVX1: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX1: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -27,6 +29,7 @@ define void @test() {
; AVX1: Cost of 66 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
+; AVX2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX2: Cost of 7 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
; AVX2: ir<%v0> = load from index 0
; AVX2: ir<%v1> = load from index 1
@@ -54,6 +57,7 @@ define void @test() {
; AVX2: ir<%v3> = load from index 3
;
; AVX512DQ-LABEL: 'test'
+; AVX512DQ: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX512DQ: Cost of 7 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
; AVX512DQ: ir<%v0> = load from index 0
; AVX512DQ: ir<%v1> = load from index 1
@@ -86,6 +90,7 @@ define void @test() {
; AVX512DQ: ir<%v3> = load from index 3
;
; AVX512BW-LABEL: 'test'
+; AVX512BW: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX512BW: Cost of 5 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
; AVX512BW: ir<%v0> = load from index 0
; AVX512BW: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-5.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-5.ll
index 14c75eba680e9b..adba6d8145fe8e 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-5.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-5.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -14,12 +14,14 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
+; SSE2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; SSE2: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 32 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
+; AVX1: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX1: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -27,6 +29,7 @@ define void @test() {
; AVX1: Cost of 66 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
+; AVX2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX2: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -34,6 +37,7 @@ define void @test() {
; AVX2: Cost of 66 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX512DQ-LABEL: 'test'
+; AVX512DQ: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX512DQ: Cost of 24 for VF 2: INTERLEAVE-GROUP with factor 5, ir<%in0>
; AVX512DQ: ir<%v0> = load from index 0
; AVX512DQ: ir<%v1> = load from index 1
@@ -72,6 +76,7 @@ define void @test() {
; AVX512DQ: ir<%v4> = load from index 4
;
; AVX512BW-LABEL: 'test'
+; AVX512BW: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX512BW: Cost of 11 for VF 2: INTERLEAVE-GROUP with factor 5, ir<%in0>
; AVX512BW: ir<%v0> = load from index 0
; AVX512BW: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-6.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-6.ll
index 727eec70f83873..019eaf1648c745 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-6.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-6.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -14,12 +14,14 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
+; SSE2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; SSE2: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 32 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
+; AVX1: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX1: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -27,6 +29,7 @@ define void @test() {
; AVX1: Cost of 66 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
+; AVX2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX2: Cost of 16 for VF 2: INTERLEAVE-GROUP with factor 6, ir<%in0>
; AVX2: ir<%v0> = load from index 0
; AVX2: ir<%v1> = load from index 1
@@ -64,6 +67,7 @@ define void @test() {
; AVX2: ir<%v5> = load from index 5
;
; AVX512DQ-LABEL: 'test'
+; AVX512DQ: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX512DQ: Cost of 16 for VF 2: INTERLEAVE-GROUP with factor 6, ir<%in0>
; AVX512DQ: ir<%v0> = load from index 0
; AVX512DQ: ir<%v1> = load from index 1
@@ -108,6 +112,7 @@ define void @test() {
; AVX512DQ: ir<%v5> = load from index 5
;
; AVX512BW-LABEL: 'test'
+; AVX512BW: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX512BW: Cost of 13 for VF 2: INTERLEAVE-GROUP with factor 6, ir<%in0>
; AVX512BW: ir<%v0> = load from index 0
; AVX512BW: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-7.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-7.ll
index d593db1853fe8f..869916d5ecb10f 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-7.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-7.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -14,12 +14,14 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
+; SSE2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; SSE2: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 32 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
+; AVX1: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX1: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -27,6 +29,7 @@ define void @test() {
; AVX1: Cost of 66 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
+; AVX2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX2: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -34,6 +37,7 @@ define void @test() {
; AVX2: Cost of 66 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX512DQ-LABEL: 'test'
+; AVX512DQ: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX512DQ: Cost of 33 for VF 2: INTERLEAVE-GROUP with factor 7, ir<%in0>
; AVX512DQ: ir<%v0> = load from index 0
; AVX512DQ: ir<%v1> = load from index 1
@@ -84,6 +88,7 @@ define void @test() {
; AVX512DQ: ir<%v6> = load from index 6
;
; AVX512BW-LABEL: 'test'
+; AVX512BW: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX512BW: Cost of 15 for VF 2: INTERLEAVE-GROUP with factor 7, ir<%in0>
; AVX512BW: ir<%v0> = load from index 0
; AVX512BW: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-8.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-8.ll
index 5ef380a3032da3..cd583a46242919 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-8.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i16-stride-8.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -14,12 +14,14 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
+; SSE2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; SSE2: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 32 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
+; AVX1: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX1: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -27,6 +29,7 @@ define void @test() {
; AVX1: Cost of 66 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
+; AVX2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX2: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -34,6 +37,7 @@ define void @test() {
; AVX2: Cost of 66 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX512DQ-LABEL: 'test'
+; AVX512DQ: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX512DQ: Cost of 34 for VF 2: INTERLEAVE-GROUP with factor 8, ir<%in0>
; AVX512DQ: ir<%v0> = load from index 0
; AVX512DQ: ir<%v1> = load from index 1
@@ -90,6 +94,7 @@ define void @test() {
; AVX512DQ: ir<%v7> = load from index 7
;
; AVX512BW-LABEL: 'test'
+; AVX512BW: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX512BW: Cost of 17 for VF 2: INTERLEAVE-GROUP with factor 8, ir<%in0>
; AVX512BW: ir<%v0> = load from index 0
; AVX512BW: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-2-indices-0u.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-2-indices-0u.ll
index 85043e46d2023f..dc0ca2ec955ad2 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-2-indices-0u.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-2-indices-0u.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,6 +13,7 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
+; SSE2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; SSE2: Cost of 2 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
; SSE2: ir<%v0> = load from index 0
; SSE2: Cost of 3 for VF 4: INTERLEAVE-GROUP with factor 2, ir<%in0>
@@ -21,6 +22,7 @@ define void @test() {
; SSE2: Cost of 44 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
+; AVX1: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX1: Cost of 2 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
; AVX1: ir<%v0> = load from index 0
; AVX1: Cost of 2 for VF 4: INTERLEAVE-GROUP with factor 2, ir<%in0>
@@ -30,6 +32,7 @@ define void @test() {
; AVX1: Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
+; AVX2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX2: Cost of 2 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
; AVX2: ir<%v0> = load from index 0
; AVX2: Cost of 2 for VF 4: INTERLEAVE-GROUP with factor 2, ir<%in0>
@@ -42,6 +45,7 @@ define void @test() {
; AVX2: ir<%v0> = load from index 0
;
; AVX512-LABEL: 'test'
+; AVX512: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX512: Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: Cost of 1 for VF 4: INTERLEAVE-GROUP with factor 2, ir<%in0>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-2.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-2.ll
index 3d8916969eade1..4eb6a98f2e53d9 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-2.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-2.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,6 +13,7 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
+; SSE2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; SSE2: Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
; SSE2: ir<%v0> = load from index 0
; SSE2: ir<%v1> = load from index 1
@@ -23,6 +24,7 @@ define void @test() {
; SSE2: Cost of 44 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
+; AVX1: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX1: Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
; AVX1: ir<%v0> = load from index 0
; AVX1: ir<%v1> = load from index 1
@@ -34,6 +36,7 @@ define void @test() {
; AVX1: Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
+; AVX2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX2: Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
; AVX2: ir<%v0> = load from index 0
; AVX2: ir<%v1> = load from index 1
@@ -51,6 +54,7 @@ define void @test() {
; AVX2: ir<%v1> = load from index 1
;
; AVX512-LABEL: 'test'
+; AVX512: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX512: Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-3-indices-01u.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-3-indices-01u.ll
index 02d90c5ffe2fbb..488f2e6a7928f1 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-3-indices-01u.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-3-indices-01u.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,12 +13,14 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
+; SSE2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; SSE2: Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 22 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 44 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
+; AVX1: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX1: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 17 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -26,6 +28,7 @@ define void @test() {
; AVX1: Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
+; AVX2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX2: Cost of 5 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
; AVX2: ir<%v0> = load from index 0
; AVX2: ir<%v1> = load from index 1
@@ -43,6 +46,7 @@ define void @test() {
; AVX2: ir<%v1> = load from index 1
;
; AVX512-LABEL: 'test'
+; AVX512: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX512: Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-3-indices-0uu.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-3-indices-0uu.ll
index e92c161c8ff960..56d22c59784990 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-3-indices-0uu.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-3-indices-0uu.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,12 +13,14 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
+; SSE2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; SSE2: Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 22 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 44 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
+; AVX1: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX1: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 17 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -26,6 +28,7 @@ define void @test() {
; AVX1: Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
+; AVX2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX2: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 3 for VF 4: INTERLEAVE-GROUP with factor 3, ir<%in0>
; AVX2: ir<%v0> = load from index 0
@@ -37,6 +40,7 @@ define void @test() {
; AVX2: ir<%v0> = load from index 0
;
; AVX512-LABEL: 'test'
+; AVX512: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX512: Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: Cost of 1 for VF 4: INTERLEAVE-GROUP with factor 3, ir<%in0>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-3.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-3.ll
index b68f44855dc3ed..f68113ce79dd26 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-3.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-3.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,12 +13,14 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
+; SSE2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; SSE2: Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 22 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 44 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
+; AVX1: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX1: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 17 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -26,6 +28,7 @@ define void @test() {
; AVX1: Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
+; AVX2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX2: Cost of 6 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
; AVX2: ir<%v0> = load from index 0
; AVX2: ir<%v1> = load from index 1
@@ -48,6 +51,7 @@ define void @test() {
; AVX2: ir<%v2> = load from index 2
;
; AVX512-LABEL: 'test'
+; AVX512: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX512: Cost of 4 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-4-indices-012u.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-4-indices-012u.ll
index 69baf083073e40..8253f18260ab10 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-4-indices-012u.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-4-indices-012u.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,12 +13,14 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
+; SSE2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; SSE2: Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 22 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 44 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
+; AVX1: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX1: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 17 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -26,6 +28,7 @@ define void @test() {
; AVX1: Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
+; AVX2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX2: Cost of 4 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
; AVX2: ir<%v0> = load from index 0
; AVX2: ir<%v1> = load from index 1
@@ -48,6 +51,7 @@ define void @test() {
; AVX2: ir<%v2> = load from index 2
;
; AVX512-LABEL: 'test'
+; AVX512: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX512: Cost of 4 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-4-indices-01uu.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-4-indices-01uu.ll
index 43e097a4ffeb01..0ffde89ebbba10 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-4-indices-01uu.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-4-indices-01uu.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,12 +13,14 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
+; SSE2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; SSE2: Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 22 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 44 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
+; AVX1: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX1: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 17 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -26,6 +28,7 @@ define void @test() {
; AVX1: Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
+; AVX2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX2: Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
; AVX2: ir<%v0> = load from index 0
; AVX2: ir<%v1> = load from index 1
@@ -43,6 +46,7 @@ define void @test() {
; AVX2: ir<%v1> = load from index 1
;
; AVX512-LABEL: 'test'
+; AVX512: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX512: Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-4-indices-0uuu.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-4-indices-0uuu.ll
index e3838488d3a967..4225c0b8e7e715 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-4-indices-0uuu.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-4-indices-0uuu.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load|WIDEN ir<%v[0-9]> = load)" --filter "^ ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load|WIDEN ir<%v[0-9]> = load)" --filter "^ ir<.* = load from index"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,12 +13,14 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
+; SSE2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; SSE2: Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 22 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 44 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
+; AVX1: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX1: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 17 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -26,6 +28,7 @@ define void @test() {
; AVX1: Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
+; AVX2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX2: Cost of 2 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
; AVX2: ir<%v0> = load from index 0
; AVX2: Cost of 4 for VF 4: INTERLEAVE-GROUP with factor 4, ir<%in0>
@@ -38,6 +41,7 @@ define void @test() {
; AVX2: ir<%v0> = load from index 0
;
; AVX512-LABEL: 'test'
+; AVX512: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX512: Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: Cost of 1 for VF 4: INTERLEAVE-GROUP with factor 4, ir<%in0>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-4.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-4.ll
index bdb68363cdc2d8..f1eebd79c9a620 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-4.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-4.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,12 +13,14 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
+; SSE2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; SSE2: Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 22 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 44 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
+; AVX1: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX1: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 17 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -26,6 +28,7 @@ define void @test() {
; AVX1: Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
+; AVX2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX2: Cost of 5 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
; AVX2: ir<%v0> = load from index 0
; AVX2: ir<%v1> = load from index 1
@@ -53,6 +56,7 @@ define void @test() {
; AVX2: ir<%v3> = load from index 3
;
; AVX512-LABEL: 'test'
+; AVX512: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX512: Cost of 5 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-5.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-5.ll
index 7f5b1687b59780..1ded6db5282eff 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-5.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-5.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,12 +13,14 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
+; SSE2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; SSE2: Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 22 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 44 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
+; AVX1: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX1: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 17 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -26,6 +28,7 @@ define void @test() {
; AVX1: Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
+; AVX2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX2: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 17 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -33,6 +36,7 @@ define void @test() {
; AVX2: Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX512-LABEL: 'test'
+; AVX512: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX512: Cost of 6 for VF 2: INTERLEAVE-GROUP with factor 5, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-6.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-6.ll
index aec171a16f12ba..354b3c2b9050e5 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-6.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-6.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,12 +13,14 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
+; SSE2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; SSE2: Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 22 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 44 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
+; AVX1: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX1: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 17 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -26,6 +28,7 @@ define void @test() {
; AVX1: Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
+; AVX2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX2: Cost of 8 for VF 2: INTERLEAVE-GROUP with factor 6, ir<%in0>
; AVX2: ir<%v0> = load from index 0
; AVX2: ir<%v1> = load from index 1
@@ -57,6 +60,7 @@ define void @test() {
; AVX2: Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX512-LABEL: 'test'
+; AVX512: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX512: Cost of 7 for VF 2: INTERLEAVE-GROUP with factor 6, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-7.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-7.ll
index 5d4904aec87eb7..1144007cbe60fc 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-7.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-7.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,12 +13,14 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
+; SSE2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; SSE2: Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 22 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 44 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
+; AVX1: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX1: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 17 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -26,6 +28,7 @@ define void @test() {
; AVX1: Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
+; AVX2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX2: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 17 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -33,6 +36,7 @@ define void @test() {
; AVX2: Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX512-LABEL: 'test'
+; AVX512: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX512: Cost of 8 for VF 2: INTERLEAVE-GROUP with factor 7, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-8.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-8.ll
index e1af00bcbe1fa2..71934416070dff 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-8.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i32-stride-8.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,12 +13,14 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
+; SSE2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; SSE2: Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 22 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 44 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
+; AVX1: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX1: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 17 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -26,6 +28,7 @@ define void @test() {
; AVX1: Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
+; AVX2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX2: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 48 for VF 8: INTERLEAVE-GROUP with factor 8, ir<%in0>
@@ -41,6 +44,7 @@ define void @test() {
; AVX2: Cost of 68 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX512-LABEL: 'test'
+; AVX512: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX512: Cost of 9 for VF 2: INTERLEAVE-GROUP with factor 8, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-2.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-2.ll
index a664eb308630b7..4dd2c42f2a228b 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-2.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-2.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load|WIDEN ir<%v[0-9]> = load)" --filter "^ ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load|WIDEN ir<%v[0-9]> = load)" --filter "^ ir<.* = load from index"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,6 +13,7 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
+; SSE2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; SSE2: Cost of 4 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
; SSE2: ir<%v0> = load from index 0
; SSE2: ir<%v1> = load from index 1
@@ -21,6 +22,7 @@ define void @test() {
; SSE2: Cost of 40 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
+; AVX1: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX1: Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
; AVX1: ir<%v0> = load from index 0
; AVX1: ir<%v1> = load from index 1
@@ -30,6 +32,7 @@ define void @test() {
; AVX1: Cost of 72 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
+; AVX2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX2: Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
; AVX2: ir<%v0> = load from index 0
; AVX2: ir<%v1> = load from index 1
@@ -47,6 +50,7 @@ define void @test() {
; AVX2: ir<%v1> = load from index 1
;
; AVX512-LABEL: 'test'
+; AVX512: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX512: Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-3.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-3.ll
index 1fbcf7559aad64..b740b83b74bea1 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-3.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-3.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,12 +13,14 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
+; SSE2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; SSE2: Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 10 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 20 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 40 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
+; AVX1: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX1: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 9 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 18 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -26,6 +28,7 @@ define void @test() {
; AVX1: Cost of 72 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
+; AVX2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX2: Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
; AVX2: ir<%v0> = load from index 0
; AVX2: ir<%v1> = load from index 1
@@ -45,6 +48,7 @@ define void @test() {
; AVX2: Cost of 72 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX512-LABEL: 'test'
+; AVX512: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX512: Cost of 4 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-4.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-4.ll
index 43adad539acd4d..12578c2293af60 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-4.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-4.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load|WIDEN ir<%v[0-9]> = load)" --filter "^ ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load|WIDEN ir<%v[0-9]> = load)" --filter "^ ir<.* = load from index"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,12 +13,14 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
+; SSE2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; SSE2: Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 10 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 20 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 40 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
+; AVX1: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX1: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 9 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 18 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -26,6 +28,7 @@ define void @test() {
; AVX1: Cost of 72 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
+; AVX2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX2: Cost of 8 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
; AVX2: ir<%v0> = load from index 0
; AVX2: ir<%v1> = load from index 1
@@ -49,6 +52,7 @@ define void @test() {
; AVX2: Cost of 72 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX512-LABEL: 'test'
+; AVX512: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX512: Cost of 5 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-5.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-5.ll
index 2c6c071b77be2f..0ac7b710e1a05a 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-5.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-5.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load|WIDEN ir<%v[0-9]> = load)" --filter "^ ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load|WIDEN ir<%v[0-9]> = load)" --filter "^ ir<.* = load from index"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,12 +13,14 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
+; SSE2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; SSE2: Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 10 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 20 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 40 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
+; AVX1: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX1: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 9 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 18 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -26,6 +28,7 @@ define void @test() {
; AVX1: Cost of 72 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
+; AVX2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX2: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 9 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 18 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -33,6 +36,7 @@ define void @test() {
; AVX2: Cost of 72 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX512-LABEL: 'test'
+; AVX512: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX512: Cost of 14.5 for VF 2: INTERLEAVE-GROUP with factor 5, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-6.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-6.ll
index 18c252a8a3aaca..2985c0c232aaa9 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-6.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-6.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load|WIDEN ir<%v[0-9]> = load)" --filter "^ ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load|WIDEN ir<%v[0-9]> = load)" --filter "^ ir<.* = load from index"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,12 +13,14 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
+; SSE2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; SSE2: Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 10 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 20 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 40 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
+; AVX1: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX1: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 9 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 18 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -26,6 +28,7 @@ define void @test() {
; AVX1: Cost of 72 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
+; AVX2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX2: Cost of 9 for VF 2: INTERLEAVE-GROUP with factor 6, ir<%in0>
; AVX2: ir<%v0> = load from index 0
; AVX2: ir<%v1> = load from index 1
@@ -51,6 +54,7 @@ define void @test() {
; AVX2: Cost of 72 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX512-LABEL: 'test'
+; AVX512: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX512: Cost of 17 for VF 2: INTERLEAVE-GROUP with factor 6, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-7.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-7.ll
index 6460df78c20c02..0b0778efb01160 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-7.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-7.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load|WIDEN ir<%v[0-9]> = load)" --filter "^ ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load|WIDEN ir<%v[0-9]> = load)" --filter "^ ir<.* = load from index"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,12 +13,14 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
+; SSE2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; SSE2: Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 10 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 20 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 40 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
+; AVX1: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX1: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 9 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 18 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -26,6 +28,7 @@ define void @test() {
; AVX1: Cost of 72 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
+; AVX2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX2: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 9 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 18 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -33,6 +36,7 @@ define void @test() {
; AVX2: Cost of 72 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX512-LABEL: 'test'
+; AVX512: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX512: Cost of 19.5 for VF 2: INTERLEAVE-GROUP with factor 7, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-8.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-8.ll
index d4d99b530f0108..81066a4364609b 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-8.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-8.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load|WIDEN ir<%v[0-9]> = load)" --filter "^ ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load|WIDEN ir<%v[0-9]> = load)" --filter "^ ir<.* = load from index"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -13,12 +13,14 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
+; SSE2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; SSE2: Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 10 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 20 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 40 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
+; AVX1: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX1: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 9 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 18 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -26,6 +28,7 @@ define void @test() {
; AVX1: Cost of 72 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
+; AVX2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX2: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 9 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 18 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -33,6 +36,7 @@ define void @test() {
; AVX2: Cost of 72 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX512-LABEL: 'test'
+; AVX512: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX512: Cost of 22 for VF 2: INTERLEAVE-GROUP with factor 8, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-2.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-2.ll
index 4686fbc32153a9..e61a54fa6d30d3 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-2.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-2.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -14,12 +14,14 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
+; SSE2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; SSE2: Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 23 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 47 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
+; AVX1: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX1: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -27,6 +29,7 @@ define void @test() {
; AVX1: Cost of 65 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
+; AVX2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX2: Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
; AVX2: ir<%v0> = load from index 0
; AVX2: ir<%v1> = load from index 1
@@ -44,6 +47,7 @@ define void @test() {
; AVX2: ir<%v1> = load from index 1
;
; AVX512DQ-LABEL: 'test'
+; AVX512DQ: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX512DQ: Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
; AVX512DQ: ir<%v0> = load from index 0
; AVX512DQ: ir<%v1> = load from index 1
@@ -64,6 +68,7 @@ define void @test() {
; AVX512DQ: ir<%v1> = load from index 1
;
; AVX512BW-LABEL: 'test'
+; AVX512BW: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX512BW: Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
; AVX512BW: ir<%v0> = load from index 0
; AVX512BW: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-3.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-3.ll
index d5bdfeacdb30e6..b68dd3eaf9c40e 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-3.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-3.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -14,12 +14,14 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
+; SSE2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; SSE2: Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 23 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 47 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
+; AVX1: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX1: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -27,6 +29,7 @@ define void @test() {
; AVX1: Cost of 65 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
+; AVX2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX2: Cost of 5 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
; AVX2: ir<%v0> = load from index 0
; AVX2: ir<%v1> = load from index 1
@@ -49,6 +52,7 @@ define void @test() {
; AVX2: ir<%v2> = load from index 2
;
; AVX512DQ-LABEL: 'test'
+; AVX512DQ: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX512DQ: Cost of 5 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
; AVX512DQ: ir<%v0> = load from index 0
; AVX512DQ: ir<%v1> = load from index 1
@@ -75,6 +79,7 @@ define void @test() {
; AVX512DQ: ir<%v2> = load from index 2
;
; AVX512BW-LABEL: 'test'
+; AVX512BW: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX512BW: Cost of 4 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%in0>
; AVX512BW: ir<%v0> = load from index 0
; AVX512BW: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-4.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-4.ll
index 7390b2f56010f4..fe4904678c69a6 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-4.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-4.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -14,12 +14,14 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
+; SSE2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; SSE2: Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 23 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 47 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
+; AVX1: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX1: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -27,6 +29,7 @@ define void @test() {
; AVX1: Cost of 65 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
+; AVX2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX2: Cost of 5 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
; AVX2: ir<%v0> = load from index 0
; AVX2: ir<%v1> = load from index 1
@@ -54,6 +57,7 @@ define void @test() {
; AVX2: ir<%v3> = load from index 3
;
; AVX512DQ-LABEL: 'test'
+; AVX512DQ: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX512DQ: Cost of 5 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
; AVX512DQ: ir<%v0> = load from index 0
; AVX512DQ: ir<%v1> = load from index 1
@@ -86,6 +90,7 @@ define void @test() {
; AVX512DQ: ir<%v3> = load from index 3
;
; AVX512BW-LABEL: 'test'
+; AVX512BW: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX512BW: Cost of 5 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%in0>
; AVX512BW: ir<%v0> = load from index 0
; AVX512BW: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-5.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-5.ll
index 3fd3b9e11017d6..5240bfbdfc0eb6 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-5.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-5.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -14,12 +14,14 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
+; SSE2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; SSE2: Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 23 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 47 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
+; AVX1: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX1: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -27,6 +29,7 @@ define void @test() {
; AVX1: Cost of 65 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
+; AVX2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX2: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -34,6 +37,7 @@ define void @test() {
; AVX2: Cost of 65 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX512DQ-LABEL: 'test'
+; AVX512DQ: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX512DQ: Cost of 22 for VF 2: INTERLEAVE-GROUP with factor 5, ir<%in0>
; AVX512DQ: ir<%v0> = load from index 0
; AVX512DQ: ir<%v1> = load from index 1
@@ -72,6 +76,7 @@ define void @test() {
; AVX512DQ: ir<%v4> = load from index 4
;
; AVX512BW-LABEL: 'test'
+; AVX512BW: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX512BW: Cost of 6 for VF 2: INTERLEAVE-GROUP with factor 5, ir<%in0>
; AVX512BW: ir<%v0> = load from index 0
; AVX512BW: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-6.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-6.ll
index d8b77269cda803..1f1e568b42f290 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-6.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-6.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -14,12 +14,14 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
+; SSE2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; SSE2: Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 23 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 47 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
+; AVX1: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX1: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -27,6 +29,7 @@ define void @test() {
; AVX1: Cost of 65 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
+; AVX2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX2: Cost of 8 for VF 2: INTERLEAVE-GROUP with factor 6, ir<%in0>
; AVX2: ir<%v0> = load from index 0
; AVX2: ir<%v1> = load from index 1
@@ -64,6 +67,7 @@ define void @test() {
; AVX2: ir<%v5> = load from index 5
;
; AVX512DQ-LABEL: 'test'
+; AVX512DQ: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX512DQ: Cost of 8 for VF 2: INTERLEAVE-GROUP with factor 6, ir<%in0>
; AVX512DQ: ir<%v0> = load from index 0
; AVX512DQ: ir<%v1> = load from index 1
@@ -108,6 +112,7 @@ define void @test() {
; AVX512DQ: ir<%v5> = load from index 5
;
; AVX512BW-LABEL: 'test'
+; AVX512BW: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX512BW: Cost of 7 for VF 2: INTERLEAVE-GROUP with factor 6, ir<%in0>
; AVX512BW: ir<%v0> = load from index 0
; AVX512BW: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-7.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-7.ll
index e2e3dedca27c8d..2565a4646120c7 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-7.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-7.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -14,12 +14,14 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
+; SSE2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; SSE2: Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 23 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 47 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
+; AVX1: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX1: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -27,6 +29,7 @@ define void @test() {
; AVX1: Cost of 65 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
+; AVX2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX2: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -34,6 +37,7 @@ define void @test() {
; AVX2: Cost of 65 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX512DQ-LABEL: 'test'
+; AVX512DQ: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX512DQ: Cost of 31 for VF 2: INTERLEAVE-GROUP with factor 7, ir<%in0>
; AVX512DQ: ir<%v0> = load from index 0
; AVX512DQ: ir<%v1> = load from index 1
@@ -84,6 +88,7 @@ define void @test() {
; AVX512DQ: ir<%v6> = load from index 6
;
; AVX512BW-LABEL: 'test'
+; AVX512BW: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX512BW: Cost of 8 for VF 2: INTERLEAVE-GROUP with factor 7, ir<%in0>
; AVX512BW: ir<%v0> = load from index 0
; AVX512BW: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-8.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-8.ll
index 1fd2c4898762d9..375c4271249eb3 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-8.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i8-stride-8.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%v0 = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (INTERLEAVE-GROUP with factor [0-9]+,|REPLICATE ir<%v0> = load)" --filter "^ ir<.* = load from index"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=SSE2
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX1
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=AVX2
@@ -14,12 +14,14 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
+; SSE2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; SSE2: Cost of 5 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 11 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 23 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
; SSE2: Cost of 47 for VF 16: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX1-LABEL: 'test'
+; AVX1: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX1: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX1: Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -27,6 +29,7 @@ define void @test() {
; AVX1: Cost of 65 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX2-LABEL: 'test'
+; AVX2: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX2: Cost of 4 for VF 2: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 8 for VF 4: REPLICATE ir<%v0> = load ir<%in0>
; AVX2: Cost of 16 for VF 8: REPLICATE ir<%v0> = load ir<%in0>
@@ -34,6 +37,7 @@ define void @test() {
; AVX2: Cost of 65 for VF 32: REPLICATE ir<%v0> = load ir<%in0>
;
; AVX512DQ-LABEL: 'test'
+; AVX512DQ: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX512DQ: Cost of 33 for VF 2: INTERLEAVE-GROUP with factor 8, ir<%in0>
; AVX512DQ: ir<%v0> = load from index 0
; AVX512DQ: ir<%v1> = load from index 1
@@ -90,6 +94,7 @@ define void @test() {
; AVX512DQ: ir<%v7> = load from index 7
;
; AVX512BW-LABEL: 'test'
+; AVX512BW: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; AVX512BW: Cost of 9 for VF 2: INTERLEAVE-GROUP with factor 8, ir<%in0>
; AVX512BW: ir<%v0> = load from index 0
; AVX512BW: ir<%v1> = load from index 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-gather-i32-with-i8-index.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-gather-i32-with-i8-index.ll
index 3d33bc34dda674..a109a3acd3bc4d 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-gather-i32-with-i8-index.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-gather-i32-with-i8-index.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF [0-9]+ For instruction:\s*%valB.loaded = load i32, ptr %inB, align 4" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: .* ir<%valB.loaded> = load"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: .* ir<%valB.loaded> = load"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=SSE
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse4.2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=SSE
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=AVX1
@@ -17,14 +17,14 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE-LABEL: 'test'
-; SSE: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; SSE: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB>
; SSE: Cost of 3000000 for VF 2: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
; SSE: Cost of 3000000 for VF 4: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
; SSE: Cost of 3000000 for VF 8: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
; SSE: Cost of 3000000 for VF 16: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
;
; AVX1-LABEL: 'test'
-; AVX1: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX1: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB>
; AVX1: Cost of 3000000 for VF 2: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
; AVX1: Cost of 3000000 for VF 4: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
; AVX1: Cost of 3000000 for VF 8: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
@@ -32,7 +32,7 @@ define void @test() {
; AVX1: Cost of 3000000 for VF 32: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
;
; AVX2-SLOWGATHER-LABEL: 'test'
-; AVX2-SLOWGATHER: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX2-SLOWGATHER: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB>
; AVX2-SLOWGATHER: Cost of 3000000 for VF 2: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
; AVX2-SLOWGATHER: Cost of 3000000 for VF 4: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
; AVX2-SLOWGATHER: Cost of 3000000 for VF 8: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
@@ -40,21 +40,21 @@ define void @test() {
; AVX2-SLOWGATHER: Cost of 3000000 for VF 32: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
;
; AVX2-FASTGATHER-LABEL: 'test'
-; AVX2-FASTGATHER: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
-; AVX2-FASTGATHER: Cost of 4 for VF 2: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
-; AVX2-FASTGATHER: Cost of 6 for VF 4: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
-; AVX2-FASTGATHER: Cost of 12 for VF 8: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
-; AVX2-FASTGATHER: Cost of 24 for VF 16: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
-; AVX2-FASTGATHER: Cost of 48 for VF 32: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX2-FASTGATHER: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB>
+; AVX2-FASTGATHER: Cost of 4 for VF 2: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
+; AVX2-FASTGATHER: Cost of 6 for VF 4: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
+; AVX2-FASTGATHER: Cost of 12 for VF 8: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
+; AVX2-FASTGATHER: Cost of 24 for VF 16: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
+; AVX2-FASTGATHER: Cost of 48 for VF 32: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
;
; AVX512-LABEL: 'test'
-; AVX512: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
-; AVX512: Cost of 8 for VF 2: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
-; AVX512: Cost of 17 for VF 4: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
-; AVX512: Cost of 10 for VF 8: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
-; AVX512: Cost of 18 for VF 16: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
-; AVX512: Cost of 36 for VF 32: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
-; AVX512: Cost of 72 for VF 64: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX512: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB>
+; AVX512: Cost of 8 for VF 2: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
+; AVX512: Cost of 17 for VF 4: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
+; AVX512: Cost of 10 for VF 8: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
+; AVX512: Cost of 18 for VF 16: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
+; AVX512: Cost of 36 for VF 32: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
+; AVX512: Cost of 72 for VF 64: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
;
entry:
br label %for.body
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-gather-i64-with-i8-index.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-gather-i64-with-i8-index.ll
index b0fec98fcdf51d..a957d896529711 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-gather-i64-with-i8-index.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-gather-i64-with-i8-index.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF [0-9]+ For instruction:\s*%valB.loaded = load i64, ptr %inB, align 8" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: .* ir<%valB.loaded> = load"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: .* ir<%valB.loaded> = load"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=SSE
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse4.2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=SSE
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=AVX1
@@ -17,14 +17,14 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE-LABEL: 'test'
-; SSE: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; SSE: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB>
; SSE: Cost of 3000000 for VF 2: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
; SSE: Cost of 3000000 for VF 4: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
; SSE: Cost of 3000000 for VF 8: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
; SSE: Cost of 3000000 for VF 16: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
;
; AVX1-LABEL: 'test'
-; AVX1: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX1: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB>
; AVX1: Cost of 3000000 for VF 2: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
; AVX1: Cost of 3000000 for VF 4: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
; AVX1: Cost of 3000000 for VF 8: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
@@ -32,7 +32,7 @@ define void @test() {
; AVX1: Cost of 3000000 for VF 32: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
;
; AVX2-SLOWGATHER-LABEL: 'test'
-; AVX2-SLOWGATHER: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX2-SLOWGATHER: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB>
; AVX2-SLOWGATHER: Cost of 3000000 for VF 2: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
; AVX2-SLOWGATHER: Cost of 3000000 for VF 4: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
; AVX2-SLOWGATHER: Cost of 3000000 for VF 8: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
@@ -40,21 +40,21 @@ define void @test() {
; AVX2-SLOWGATHER: Cost of 3000000 for VF 32: REPLICATE ir<%valB.loaded> = load ir<%inB> (S->V)
;
; AVX2-FASTGATHER-LABEL: 'test'
-; AVX2-FASTGATHER: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
-; AVX2-FASTGATHER: Cost of 4 for VF 2: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
-; AVX2-FASTGATHER: Cost of 6 for VF 4: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
-; AVX2-FASTGATHER: Cost of 12 for VF 8: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
-; AVX2-FASTGATHER: Cost of 24 for VF 16: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
-; AVX2-FASTGATHER: Cost of 48 for VF 32: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX2-FASTGATHER: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB>
+; AVX2-FASTGATHER: Cost of 4 for VF 2: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
+; AVX2-FASTGATHER: Cost of 6 for VF 4: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
+; AVX2-FASTGATHER: Cost of 12 for VF 8: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
+; AVX2-FASTGATHER: Cost of 24 for VF 16: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
+; AVX2-FASTGATHER: Cost of 48 for VF 32: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
;
; AVX512-LABEL: 'test'
-; AVX512: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
-; AVX512: Cost of 8 for VF 2: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
-; AVX512: Cost of 18 for VF 4: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
-; AVX512: Cost of 10 for VF 8: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
-; AVX512: Cost of 20 for VF 16: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
-; AVX512: Cost of 40 for VF 32: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
-; AVX512: Cost of 80 for VF 64: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX512: Cost of 1 for VF 1: EMIT-SCALAR ir<%valB.loaded> = load ir<%inB>
+; AVX512: Cost of 8 for VF 2: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
+; AVX512: Cost of 18 for VF 4: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
+; AVX512: Cost of 10 for VF 8: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
+; AVX512: Cost of 20 for VF 16: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
+; AVX512: Cost of 40 for VF 32: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
+; AVX512: Cost of 80 for VF 64: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
;
entry:
br label %for.body
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-interleaved-load-i16.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-interleaved-load-i16.ll
index 52d05b807ba827..fb730224a01aa0 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-interleaved-load-i16.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-interleaved-load-i16.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*%i[2,4] = load i16, ptr %[a-zA-Z0-7]+, align 2" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (REPLICATE ir<%i[24]> = load|INTERLEAVE-GROUP with factor [0-9]+)" --filter "^ ir<.* = load from index"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%i[24]> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (REPLICATE ir<%i[24]> = load|INTERLEAVE-GROUP with factor [0-9]+)" --filter "^ ir<.* = load from index"
; RUN: opt -passes=loop-vectorize -enable-interleaved-mem-accesses -tail-folding-policy=must-fold-tail -S -mcpu=skx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=DISABLED_MASKED_STRIDED
; RUN: opt -passes=loop-vectorize -enable-interleaved-mem-accesses -enable-masked-interleaved-mem-accesses -tail-folding-policy=must-fold-tail -S -mcpu=skx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=ENABLED_MASKED_STRIDED
; REQUIRES: asserts
@@ -20,6 +20,8 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test1(ptr noalias nocapture %points, ptr noalias nocapture readonly %x, ptr noalias nocapture readonly %y) {
; DISABLED_MASKED_STRIDED-LABEL: 'test1'
+; DISABLED_MASKED_STRIDED: Cost of 1 for VF 1: EMIT-SCALAR ir<%i2> = load ir<%arrayidx2>
+; DISABLED_MASKED_STRIDED: Cost of 1 for VF 1: EMIT-SCALAR ir<%i4> = load ir<%arrayidx7>
; DISABLED_MASKED_STRIDED: Cost of 6 for VF 2: REPLICATE ir<%i2> = load ir<%arrayidx2>
; DISABLED_MASKED_STRIDED: Cost of 6 for VF 2: REPLICATE ir<%i4> = load ir<%arrayidx7>
; DISABLED_MASKED_STRIDED: Cost of 13 for VF 4: REPLICATE ir<%i2> = load ir<%arrayidx2>
@@ -30,6 +32,8 @@ define void @test1(ptr noalias nocapture %points, ptr noalias nocapture readonly
; DISABLED_MASKED_STRIDED: Cost of 55 for VF 16: REPLICATE ir<%i4> = load ir<%arrayidx7>
;
; ENABLED_MASKED_STRIDED-LABEL: 'test1'
+; ENABLED_MASKED_STRIDED: Cost of 1 for VF 1: EMIT-SCALAR ir<%i2> = load ir<%arrayidx2>
+; ENABLED_MASKED_STRIDED: Cost of 1 for VF 1: EMIT-SCALAR ir<%i4> = load ir<%arrayidx7>
; ENABLED_MASKED_STRIDED: Cost of 8 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%arrayidx2>
; ENABLED_MASKED_STRIDED: ir<%i2> = load from index 0
; ENABLED_MASKED_STRIDED: ir<%i4> = load from index 1
@@ -77,6 +81,8 @@ for.end:
define void @test2(ptr noalias nocapture %points, i32 %numPoints, ptr noalias nocapture readonly %x, ptr noalias nocapture readonly %y) {
; DISABLED_MASKED_STRIDED-LABEL: 'test2'
+; DISABLED_MASKED_STRIDED: Cost of 1 for VF 1: EMIT-SCALAR ir<%i2> = load ir<%arrayidx2>
+; DISABLED_MASKED_STRIDED: Cost of 1 for VF 1: EMIT-SCALAR ir<%i4> = load ir<%arrayidx7>
; DISABLED_MASKED_STRIDED: Cost of 3000000 for VF 2: REPLICATE ir<%i2> = load ir<%arrayidx2> (S->V)
; DISABLED_MASKED_STRIDED: Cost of 3000000 for VF 2: REPLICATE ir<%i4> = load ir<%arrayidx7> (S->V)
; DISABLED_MASKED_STRIDED: Cost of 3000000 for VF 4: REPLICATE ir<%i2> = load ir<%arrayidx2> (S->V)
@@ -87,6 +93,8 @@ define void @test2(ptr noalias nocapture %points, i32 %numPoints, ptr noalias no
; DISABLED_MASKED_STRIDED: Cost of 3000000 for VF 16: REPLICATE ir<%i4> = load ir<%arrayidx7> (S->V)
;
; ENABLED_MASKED_STRIDED-LABEL: 'test2'
+; ENABLED_MASKED_STRIDED: Cost of 1 for VF 1: EMIT-SCALAR ir<%i2> = load ir<%arrayidx2>
+; ENABLED_MASKED_STRIDED: Cost of 1 for VF 1: EMIT-SCALAR ir<%i4> = load ir<%arrayidx7>
; ENABLED_MASKED_STRIDED: Cost of 8 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%arrayidx2>, vp<[[VP8:%[0-9]+]]>
; ENABLED_MASKED_STRIDED: ir<%i2> = load from index 0
; ENABLED_MASKED_STRIDED: ir<%i4> = load from index 1
@@ -144,20 +152,24 @@ for.end:
define void @test(ptr noalias nocapture %points, ptr noalias nocapture readonly %x, ptr noalias nocapture readnone %y) {
; DISABLED_MASKED_STRIDED-LABEL: 'test'
+; DISABLED_MASKED_STRIDED: Cost of 1 for VF 1: EMIT-SCALAR ir<%i2> = load ir<%arrayidx>
+; DISABLED_MASKED_STRIDED: Cost of 1 for VF 1: EMIT-SCALAR ir<%i4> = load ir<%arrayidx6>
; DISABLED_MASKED_STRIDED: Cost of 3000000 for VF 2: REPLICATE ir<%i4> = load ir<%arrayidx6> (S->V)
; DISABLED_MASKED_STRIDED: Cost of 3000000 for VF 4: REPLICATE ir<%i4> = load ir<%arrayidx6> (S->V)
; DISABLED_MASKED_STRIDED: Cost of 3000000 for VF 8: REPLICATE ir<%i4> = load ir<%arrayidx6> (S->V)
; DISABLED_MASKED_STRIDED: Cost of 3000000 for VF 16: REPLICATE ir<%i4> = load ir<%arrayidx6> (S->V)
;
; ENABLED_MASKED_STRIDED-LABEL: 'test'
+; ENABLED_MASKED_STRIDED: Cost of 1 for VF 1: EMIT-SCALAR ir<%i2> = load ir<%arrayidx>
+; ENABLED_MASKED_STRIDED: Cost of 1 for VF 1: EMIT-SCALAR ir<%i4> = load ir<%arrayidx6>
; ENABLED_MASKED_STRIDED: Cost of 7 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%arrayidx6>, ir<%cmp1>
-; ENABLED_MASKED_STRIDED: ir<%i4> = load from index 0 (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; ENABLED_MASKED_STRIDED: ir<%i4> = load from index 0
; ENABLED_MASKED_STRIDED: Cost of 9 for VF 4: INTERLEAVE-GROUP with factor 3, ir<%arrayidx6>, ir<%cmp1>
-; ENABLED_MASKED_STRIDED: ir<%i4> = load from index 0 (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; ENABLED_MASKED_STRIDED: ir<%i4> = load from index 0
; ENABLED_MASKED_STRIDED: Cost of 9 for VF 8: INTERLEAVE-GROUP with factor 3, ir<%arrayidx6>, ir<%cmp1>
-; ENABLED_MASKED_STRIDED: ir<%i4> = load from index 0 (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; ENABLED_MASKED_STRIDED: ir<%i4> = load from index 0
; ENABLED_MASKED_STRIDED: Cost of 14 for VF 16: INTERLEAVE-GROUP with factor 3, ir<%arrayidx6>, ir<%cmp1>
-; ENABLED_MASKED_STRIDED: ir<%i4> = load from index 0 (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; ENABLED_MASKED_STRIDED: ir<%i4> = load from index 0
;
entry:
br label %for.body
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-interleaved-store-i16.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-interleaved-store-i16.ll
index 6b7b2919a9899e..ed9e7a03aec67a 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-interleaved-store-i16.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-interleaved-store-i16.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*store i16 %[0,2], ptr %[a-zA-Z0-7]+, align 2" --filter "Cost of [1-9][0-9]*(.[0-9]+)? for VF [0-9]+: (profitable to scalarize\s+store i16 %[02]|REPLICATE store ir<%[02]>|INTERLEAVE-GROUP with factor [0-9]+)"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT store ir<%[02]>" --filter "Cost of [1-9][0-9]*(.[0-9]+)? for VF [0-9]+: (profitable to scalarize\s+store i16 %[02]|REPLICATE store ir<%[02]>|INTERLEAVE-GROUP with factor [0-9]+)"
; RUN: opt -passes=loop-vectorize -enable-interleaved-mem-accesses -tail-folding-policy=must-fold-tail -S -mcpu=skx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=DISABLED_MASKED_STRIDED
; RUN: opt -passes=loop-vectorize -enable-interleaved-mem-accesses -enable-masked-interleaved-mem-accesses -tail-folding-policy=must-fold-tail -S -mcpu=skx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefix=ENABLED_MASKED_STRIDED
; REQUIRES: asserts
@@ -20,6 +20,8 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test1(ptr noalias nocapture %points, ptr noalias nocapture readonly %x, ptr noalias nocapture readonly %y) {
; DISABLED_MASKED_STRIDED-LABEL: 'test1'
+; DISABLED_MASKED_STRIDED: Cost of 1 for VF 1: EMIT store ir<%0>, ir<%arrayidx2>
+; DISABLED_MASKED_STRIDED: Cost of 1 for VF 1: EMIT store ir<%2>, ir<%arrayidx7>
; DISABLED_MASKED_STRIDED: Cost of 6 for VF 2: REPLICATE store ir<%0>, ir<%arrayidx2>
; DISABLED_MASKED_STRIDED: Cost of 6 for VF 2: REPLICATE store ir<%2>, ir<%arrayidx7>
; DISABLED_MASKED_STRIDED: Cost of 13 for VF 4: REPLICATE store ir<%0>, ir<%arrayidx2>
@@ -30,6 +32,8 @@ define void @test1(ptr noalias nocapture %points, ptr noalias nocapture readonly
; DISABLED_MASKED_STRIDED: Cost of 55 for VF 16: REPLICATE store ir<%2>, ir<%arrayidx7>
;
; ENABLED_MASKED_STRIDED-LABEL: 'test1'
+; ENABLED_MASKED_STRIDED: Cost of 1 for VF 1: EMIT store ir<%0>, ir<%arrayidx2>
+; ENABLED_MASKED_STRIDED: Cost of 1 for VF 1: EMIT store ir<%2>, ir<%arrayidx7>
; ENABLED_MASKED_STRIDED: Cost of 6 for VF 2: REPLICATE store ir<%0>, ir<%arrayidx2>
; ENABLED_MASKED_STRIDED: Cost of 6 for VF 2: REPLICATE store ir<%2>, ir<%arrayidx7>
; ENABLED_MASKED_STRIDED: Cost of 14 for VF 4: INTERLEAVE-GROUP with factor 4, ir<%arrayidx2>
@@ -70,6 +74,8 @@ for.end:
define void @test2(ptr noalias nocapture %points, i32 %numPoints, ptr noalias nocapture readonly %x, ptr noalias nocapture readonly %y) {
; DISABLED_MASKED_STRIDED-LABEL: 'test2'
+; DISABLED_MASKED_STRIDED: Cost of 1 for VF 1: EMIT store ir<%0>, ir<%arrayidx2>
+; DISABLED_MASKED_STRIDED: Cost of 1 for VF 1: EMIT store ir<%2>, ir<%arrayidx7>
; DISABLED_MASKED_STRIDED: Cost of 3000000 for VF 2: REPLICATE store ir<%0>, ir<%arrayidx2>
; DISABLED_MASKED_STRIDED: Cost of 3000000 for VF 2: REPLICATE store ir<%2>, ir<%arrayidx7>
; DISABLED_MASKED_STRIDED: Cost of 3000000 for VF 4: REPLICATE store ir<%0>, ir<%arrayidx2>
@@ -80,6 +86,8 @@ define void @test2(ptr noalias nocapture %points, i32 %numPoints, ptr noalias no
; DISABLED_MASKED_STRIDED: Cost of 3000000 for VF 16: REPLICATE store ir<%2>, ir<%arrayidx7>
;
; ENABLED_MASKED_STRIDED-LABEL: 'test2'
+; ENABLED_MASKED_STRIDED: Cost of 1 for VF 1: EMIT store ir<%0>, ir<%arrayidx2>
+; ENABLED_MASKED_STRIDED: Cost of 1 for VF 1: EMIT store ir<%2>, ir<%arrayidx7>
; ENABLED_MASKED_STRIDED: Cost of 13 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%arrayidx2>, vp<[[VP8:%[0-9]+]]>
; ENABLED_MASKED_STRIDED: Cost of 14 for VF 4: INTERLEAVE-GROUP with factor 4, ir<%arrayidx2>, vp<[[VP8]]>
; ENABLED_MASKED_STRIDED: Cost of 14 for VF 8: INTERLEAVE-GROUP with factor 4, ir<%arrayidx2>, vp<[[VP8]]>
@@ -129,12 +137,14 @@ for.end:
define void @test(ptr noalias nocapture %points, ptr noalias nocapture readonly %x, ptr noalias nocapture readnone %y) {
; DISABLED_MASKED_STRIDED-LABEL: 'test'
+; DISABLED_MASKED_STRIDED: Cost of 1 for VF 1: EMIT store ir<%0>, ir<%arrayidx6>
; DISABLED_MASKED_STRIDED: Cost of 2 for VF 2: profitable to scalarize store i16 %0, ptr %arrayidx6, align 2
; DISABLED_MASKED_STRIDED: Cost of 4 for VF 4: profitable to scalarize store i16 %0, ptr %arrayidx6, align 2
; DISABLED_MASKED_STRIDED: Cost of 8 for VF 8: profitable to scalarize store i16 %0, ptr %arrayidx6, align 2
; DISABLED_MASKED_STRIDED: Cost of 16.5 for VF 16: profitable to scalarize store i16 %0, ptr %arrayidx6, align 2
;
; ENABLED_MASKED_STRIDED-LABEL: 'test'
+; ENABLED_MASKED_STRIDED: Cost of 1 for VF 1: EMIT store ir<%0>, ir<%arrayidx6>
; ENABLED_MASKED_STRIDED: Cost of 2 for VF 2: profitable to scalarize store i16 %0, ptr %arrayidx6, align 2
; ENABLED_MASKED_STRIDED: Cost of 4 for VF 4: profitable to scalarize store i16 %0, ptr %arrayidx6, align 2
; ENABLED_MASKED_STRIDED: Cost of 12 for VF 8: INTERLEAVE-GROUP with factor 3, ir<%arrayidx6>, ir<%cmp1>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i16.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i16.ll
index a4a86cb40b3ea0..1f03201e56e285 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i16.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i16.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF [0-9]+ For instruction:\s*%valB.loaded = load i16, ptr %inB, align 2" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: .* ir<%valB.loaded> = load"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: .* ir<%valB.loaded> = load"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output -vplan-print-metadata=false < %s 2>&1 | FileCheck %s --check-prefixes=SSE
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse4.2 --debug-only=loop-vectorize --disable-output -vplan-print-metadata=false < %s 2>&1 | FileCheck %s --check-prefixes=SSE
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output -vplan-print-metadata=false < %s 2>&1 | FileCheck %s --check-prefixes=AVX1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i32.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i32.ll
index be448f4b9359f3..51908f4fad8b09 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i32.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i32.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF [0-9]+ For instruction:\s*%valB.loaded = load i32, ptr %inB, align 4" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: .* ir<%valB.loaded> = load"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: .* ir<%valB.loaded> = load"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output -vplan-print-metadata=false < %s 2>&1 | FileCheck %s --check-prefixes=SSE
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse4.2 --debug-only=loop-vectorize --disable-output -vplan-print-metadata=false < %s 2>&1 | FileCheck %s --check-prefixes=SSE
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output -vplan-print-metadata=false < %s 2>&1 | FileCheck %s --check-prefixes=AVX1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i64.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i64.ll
index 9900b8f2637def..8df28a0b608f1d 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i64.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i64.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF [0-9]+ For instruction:\s*%valB.loaded = load i64, ptr %inB, align 8" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: .* ir<%valB.loaded> = load"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: .* ir<%valB.loaded> = load"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output -vplan-print-metadata=false < %s 2>&1 | FileCheck %s --check-prefixes=SSE
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse4.2 --debug-only=loop-vectorize --disable-output -vplan-print-metadata=false < %s 2>&1 | FileCheck %s --check-prefixes=SSE
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output -vplan-print-metadata=false < %s 2>&1 | FileCheck %s --check-prefixes=AVX1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i8.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i8.ll
index 594d4766b43a4d..1118023a958f06 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i8.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i8.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF [0-9]+ For instruction:\s*%valB.loaded = load i8, ptr %inB, align 1" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: .* ir<%valB.loaded> = load"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: .* ir<%valB.loaded> = load"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output -vplan-print-metadata=false < %s 2>&1 | FileCheck %s --check-prefixes=SSE
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse4.2 --debug-only=loop-vectorize --disable-output -vplan-print-metadata=false < %s 2>&1 | FileCheck %s --check-prefixes=SSE
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output -vplan-print-metadata=false < %s 2>&1 | FileCheck %s --check-prefixes=AVX1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-scatter-i32-with-i8-index.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-scatter-i32-with-i8-index.ll
index b026016d1331bd..86d317982d1ea5 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-scatter-i32-with-i8-index.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-scatter-i32-with-i8-index.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*store i32 %valB, ptr %out" --filter "Cost of [1-9][0-9]*(.[0-9]+)? for VF [0-9]+: (profitable to scalarize\s+store i32 %valB|WIDEN store .*, ir<%valB>|REPLICATE store ir<%valB>)"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT store ir<%valB>" --filter "Cost of [1-9][0-9]*(.[0-9]+)? for VF [0-9]+: (profitable to scalarize\s+store i32 %valB|WIDEN store .*, ir<%valB>|REPLICATE store ir<%valB>)"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=SSE2
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse4.2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=SSE42
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=AVX1
@@ -17,18 +17,21 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
+; SSE2: Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
; SSE2: Cost of 2.5 for VF 2: profitable to scalarize store i32 %valB, ptr %out, align 4
; SSE2: Cost of 5.5 for VF 4: profitable to scalarize store i32 %valB, ptr %out, align 4
; SSE2: Cost of 11 for VF 8: profitable to scalarize store i32 %valB, ptr %out, align 4
; SSE2: Cost of 22 for VF 16: profitable to scalarize store i32 %valB, ptr %out, align 4
;
; SSE42-LABEL: 'test'
+; SSE42: Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
; SSE42: Cost of 2 for VF 2: profitable to scalarize store i32 %valB, ptr %out, align 4
; SSE42: Cost of 4 for VF 4: profitable to scalarize store i32 %valB, ptr %out, align 4
; SSE42: Cost of 8 for VF 8: profitable to scalarize store i32 %valB, ptr %out, align 4
; SSE42: Cost of 16 for VF 16: profitable to scalarize store i32 %valB, ptr %out, align 4
;
; AVX1-LABEL: 'test'
+; AVX1: Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
; AVX1: Cost of 2 for VF 2: profitable to scalarize store i32 %valB, ptr %out, align 4
; AVX1: Cost of 4 for VF 4: profitable to scalarize store i32 %valB, ptr %out, align 4
; AVX1: Cost of 8.5 for VF 8: profitable to scalarize store i32 %valB, ptr %out, align 4
@@ -36,6 +39,7 @@ define void @test() {
; AVX1: Cost of 34 for VF 32: profitable to scalarize store i32 %valB, ptr %out, align 4
;
; AVX2-LABEL: 'test'
+; AVX2: Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
; AVX2: Cost of 2 for VF 2: profitable to scalarize store i32 %valB, ptr %out, align 4
; AVX2: Cost of 4 for VF 4: profitable to scalarize store i32 %valB, ptr %out, align 4
; AVX2: Cost of 8.5 for VF 8: profitable to scalarize store i32 %valB, ptr %out, align 4
@@ -43,12 +47,13 @@ define void @test() {
; AVX2: Cost of 34 for VF 32: profitable to scalarize store i32 %valB, ptr %out, align 4
;
; AVX512-LABEL: 'test'
+; AVX512: Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
; AVX512: Cost of 5 for VF 2: REPLICATE store ir<%valB>, ir<%out>
; AVX512: Cost of 10.5 for VF 4: REPLICATE store ir<%valB>, ir<%out>
-; AVX512: Cost of 10 for VF 8: WIDEN store ir<%out>, ir<%valB>, ir<%canStore> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
-; AVX512: Cost of 18 for VF 16: WIDEN store ir<%out>, ir<%valB>, ir<%canStore> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
-; AVX512: Cost of 36 for VF 32: WIDEN store ir<%out>, ir<%valB>, ir<%canStore> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
-; AVX512: Cost of 72 for VF 64: WIDEN store ir<%out>, ir<%valB>, ir<%canStore> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX512: Cost of 10 for VF 8: WIDEN store ir<%out>, ir<%valB>, ir<%canStore>
+; AVX512: Cost of 18 for VF 16: WIDEN store ir<%out>, ir<%valB>, ir<%canStore>
+; AVX512: Cost of 36 for VF 32: WIDEN store ir<%out>, ir<%valB>, ir<%canStore>
+; AVX512: Cost of 72 for VF 64: WIDEN store ir<%out>, ir<%valB>, ir<%canStore>
;
entry:
br label %for.body
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-scatter-i64-with-i8-index.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-scatter-i64-with-i8-index.ll
index f9c6a5b14c8024..216a24086891b9 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-scatter-i64-with-i8-index.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-scatter-i64-with-i8-index.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*store i64 %valB, ptr %out" --filter "Cost of [1-9][0-9]*(.[0-9]+)? for VF [0-9]+: (profitable to scalarize\s+store i64 %valB|WIDEN store .*, ir<%valB>|REPLICATE store ir<%valB>)"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT store ir<%valB>" --filter "Cost of [1-9][0-9]*(.[0-9]+)? for VF [0-9]+: (profitable to scalarize\s+store i64 %valB|WIDEN store .*, ir<%valB>|REPLICATE store ir<%valB>)"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=SSE2
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse4.2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=SSE42
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=AVX1
@@ -17,18 +17,21 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
+; SSE2: Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
; SSE2: Cost of 2.5 for VF 2: profitable to scalarize store i64 %valB, ptr %out, align 8
; SSE2: Cost of 5 for VF 4: profitable to scalarize store i64 %valB, ptr %out, align 8
; SSE2: Cost of 10 for VF 8: profitable to scalarize store i64 %valB, ptr %out, align 8
; SSE2: Cost of 20 for VF 16: profitable to scalarize store i64 %valB, ptr %out, align 8
;
; SSE42-LABEL: 'test'
+; SSE42: Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
; SSE42: Cost of 2 for VF 2: profitable to scalarize store i64 %valB, ptr %out, align 8
; SSE42: Cost of 4 for VF 4: profitable to scalarize store i64 %valB, ptr %out, align 8
; SSE42: Cost of 8 for VF 8: profitable to scalarize store i64 %valB, ptr %out, align 8
; SSE42: Cost of 16 for VF 16: profitable to scalarize store i64 %valB, ptr %out, align 8
;
; AVX1-LABEL: 'test'
+; AVX1: Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
; AVX1: Cost of 2 for VF 2: profitable to scalarize store i64 %valB, ptr %out, align 8
; AVX1: Cost of 4.5 for VF 4: profitable to scalarize store i64 %valB, ptr %out, align 8
; AVX1: Cost of 9 for VF 8: profitable to scalarize store i64 %valB, ptr %out, align 8
@@ -36,6 +39,7 @@ define void @test() {
; AVX1: Cost of 36 for VF 32: profitable to scalarize store i64 %valB, ptr %out, align 8
;
; AVX2-LABEL: 'test'
+; AVX2: Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
; AVX2: Cost of 2 for VF 2: profitable to scalarize store i64 %valB, ptr %out, align 8
; AVX2: Cost of 4.5 for VF 4: profitable to scalarize store i64 %valB, ptr %out, align 8
; AVX2: Cost of 9 for VF 8: profitable to scalarize store i64 %valB, ptr %out, align 8
@@ -43,12 +47,13 @@ define void @test() {
; AVX2: Cost of 36 for VF 32: profitable to scalarize store i64 %valB, ptr %out, align 8
;
; AVX512-LABEL: 'test'
+; AVX512: Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
; AVX512: Cost of 5 for VF 2: REPLICATE store ir<%valB>, ir<%out>
; AVX512: Cost of 11 for VF 4: REPLICATE store ir<%valB>, ir<%out>
-; AVX512: Cost of 10 for VF 8: WIDEN store ir<%out>, ir<%valB>, ir<%canStore> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
-; AVX512: Cost of 20 for VF 16: WIDEN store ir<%out>, ir<%valB>, ir<%canStore> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
-; AVX512: Cost of 40 for VF 32: WIDEN store ir<%out>, ir<%valB>, ir<%canStore> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
-; AVX512: Cost of 80 for VF 64: WIDEN store ir<%out>, ir<%valB>, ir<%canStore> (!vplan.execution.frequency 5764607523034234880 (62.5%, estimated))
+; AVX512: Cost of 10 for VF 8: WIDEN store ir<%out>, ir<%valB>, ir<%canStore>
+; AVX512: Cost of 20 for VF 16: WIDEN store ir<%out>, ir<%valB>, ir<%canStore>
+; AVX512: Cost of 40 for VF 32: WIDEN store ir<%out>, ir<%valB>, ir<%canStore>
+; AVX512: Cost of 80 for VF 64: WIDEN store ir<%out>, ir<%valB>, ir<%canStore>
;
entry:
br label %for.body
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i16.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i16.ll
index 886e593362383e..a00dba4f1330e2 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i16.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i16.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*store i16 %valB, ptr %out" --filter "Cost of [1-9][0-9]*(.[0-9]+)? for VF [0-9]+: (profitable to scalarize\s+store i16 %valB|WIDEN store .*, ir<%valB>|REPLICATE store ir<%valB>)"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT store ir<%valB>" --filter "Cost of [1-9][0-9]*(.[0-9]+)? for VF [0-9]+: (profitable to scalarize\s+store i16 %valB|WIDEN store .*, ir<%valB>|REPLICATE store ir<%valB>)"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output -vplan-print-metadata=false < %s 2>&1 | FileCheck %s --check-prefixes=SSE
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse4.2 --debug-only=loop-vectorize --disable-output -vplan-print-metadata=false < %s 2>&1 | FileCheck %s --check-prefixes=SSE
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output -vplan-print-metadata=false < %s 2>&1 | FileCheck %s --check-prefixes=AVX1
@@ -16,12 +16,14 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test(ptr %C) {
; SSE-LABEL: 'test'
+; SSE: Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
; SSE: Cost of 2 for VF 2: profitable to scalarize store i16 %valB, ptr %out, align 2
; SSE: Cost of 4 for VF 4: profitable to scalarize store i16 %valB, ptr %out, align 2
; SSE: Cost of 8 for VF 8: profitable to scalarize store i16 %valB, ptr %out, align 2
; SSE: Cost of 16 for VF 16: profitable to scalarize store i16 %valB, ptr %out, align 2
;
; AVX1-LABEL: 'test'
+; AVX1: Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
; AVX1: Cost of 2 for VF 2: profitable to scalarize store i16 %valB, ptr %out, align 2
; AVX1: Cost of 4 for VF 4: profitable to scalarize store i16 %valB, ptr %out, align 2
; AVX1: Cost of 8 for VF 8: profitable to scalarize store i16 %valB, ptr %out, align 2
@@ -29,6 +31,7 @@ define void @test(ptr %C) {
; AVX1: Cost of 33 for VF 32: profitable to scalarize store i16 %valB, ptr %out, align 2
;
; AVX2-LABEL: 'test'
+; AVX2: Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
; AVX2: Cost of 2 for VF 2: profitable to scalarize store i16 %valB, ptr %out, align 2
; AVX2: Cost of 4 for VF 4: profitable to scalarize store i16 %valB, ptr %out, align 2
; AVX2: Cost of 8 for VF 8: profitable to scalarize store i16 %valB, ptr %out, align 2
@@ -36,6 +39,7 @@ define void @test(ptr %C) {
; AVX2: Cost of 33 for VF 32: profitable to scalarize store i16 %valB, ptr %out, align 2
;
; AVX512-LABEL: 'test'
+; AVX512: Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
; AVX512: Cost of 2 for VF 2: WIDEN store vp<[[VP7:%[0-9]+]]>, ir<%valB>, ir<%canStore>
; AVX512: Cost of 2 for VF 4: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
; AVX512: Cost of 1 for VF 8: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i32.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i32.ll
index 9be2bd10f40fad..f65258131e69f6 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i32.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i32.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*store i32 %valB, ptr %out" --filter "Cost of [1-9][0-9]*(.[0-9]+)? for VF [0-9]+: (profitable to scalarize\s+store i32 %valB|WIDEN store .*, ir<%valB>|REPLICATE store ir<%valB>)"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT store ir<%valB>" --filter "Cost of [1-9][0-9]*(.[0-9]+)? for VF [0-9]+: (profitable to scalarize\s+store i32 %valB|WIDEN store .*, ir<%valB>|REPLICATE store ir<%valB>)"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output -vplan-print-metadata=false < %s 2>&1 | FileCheck %s --check-prefixes=SSE2
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse4.2 --debug-only=loop-vectorize --disable-output -vplan-print-metadata=false < %s 2>&1 | FileCheck %s --check-prefixes=SSE42
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output -vplan-print-metadata=false < %s 2>&1 | FileCheck %s --check-prefixes=AVX1
@@ -16,18 +16,21 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test(ptr %C) {
; SSE2-LABEL: 'test'
+; SSE2: Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
; SSE2: Cost of 2.5 for VF 2: profitable to scalarize store i32 %valB, ptr %out, align 4
; SSE2: Cost of 5.5 for VF 4: profitable to scalarize store i32 %valB, ptr %out, align 4
; SSE2: Cost of 11 for VF 8: profitable to scalarize store i32 %valB, ptr %out, align 4
; SSE2: Cost of 22 for VF 16: profitable to scalarize store i32 %valB, ptr %out, align 4
;
; SSE42-LABEL: 'test'
+; SSE42: Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
; SSE42: Cost of 2 for VF 2: profitable to scalarize store i32 %valB, ptr %out, align 4
; SSE42: Cost of 4 for VF 4: profitable to scalarize store i32 %valB, ptr %out, align 4
; SSE42: Cost of 8 for VF 8: profitable to scalarize store i32 %valB, ptr %out, align 4
; SSE42: Cost of 16 for VF 16: profitable to scalarize store i32 %valB, ptr %out, align 4
;
; AVX1-LABEL: 'test'
+; AVX1: Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
; AVX1: Cost of 9 for VF 2: WIDEN store vp<[[VP7:%[0-9]+]]>, ir<%valB>, ir<%canStore>
; AVX1: Cost of 8 for VF 4: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
; AVX1: Cost of 8 for VF 8: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
@@ -35,6 +38,7 @@ define void @test(ptr %C) {
; AVX1: Cost of 32 for VF 32: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
;
; AVX2-LABEL: 'test'
+; AVX2: Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
; AVX2: Cost of 9 for VF 2: WIDEN store vp<[[VP7:%[0-9]+]]>, ir<%valB>, ir<%canStore>
; AVX2: Cost of 8 for VF 4: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
; AVX2: Cost of 8 for VF 8: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
@@ -42,6 +46,7 @@ define void @test(ptr %C) {
; AVX2: Cost of 32 for VF 32: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
;
; AVX512-LABEL: 'test'
+; AVX512: Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
; AVX512: Cost of 2 for VF 2: WIDEN store vp<[[VP7:%[0-9]+]]>, ir<%valB>, ir<%canStore>
; AVX512: Cost of 1 for VF 4: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
; AVX512: Cost of 1 for VF 8: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i64.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i64.ll
index acc0158917568f..1c1692ea164a8f 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i64.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i64.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*store i64 %valB, ptr %out" --filter "Cost of [1-9][0-9]*(.[0-9]+)? for VF [0-9]+: (profitable to scalarize\s+store i64 %valB|WIDEN store .*, ir<%valB>|REPLICATE store ir<%valB>)"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT store ir<%valB>" --filter "Cost of [1-9][0-9]*(.[0-9]+)? for VF [0-9]+: (profitable to scalarize\s+store i64 %valB|WIDEN store .*, ir<%valB>|REPLICATE store ir<%valB>)"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output -vplan-print-metadata=false < %s 2>&1 | FileCheck %s --check-prefixes=SSE2
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse4.2 --debug-only=loop-vectorize --disable-output -vplan-print-metadata=false < %s 2>&1 | FileCheck %s --check-prefixes=SSE42
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output -vplan-print-metadata=false < %s 2>&1 | FileCheck %s --check-prefixes=AVX1
@@ -16,18 +16,21 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test(ptr %C) {
; SSE2-LABEL: 'test'
+; SSE2: Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
; SSE2: Cost of 2.5 for VF 2: profitable to scalarize store i64 %valB, ptr %out, align 8
; SSE2: Cost of 5 for VF 4: profitable to scalarize store i64 %valB, ptr %out, align 8
; SSE2: Cost of 10 for VF 8: profitable to scalarize store i64 %valB, ptr %out, align 8
; SSE2: Cost of 20 for VF 16: profitable to scalarize store i64 %valB, ptr %out, align 8
;
; SSE42-LABEL: 'test'
+; SSE42: Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
; SSE42: Cost of 2 for VF 2: profitable to scalarize store i64 %valB, ptr %out, align 8
; SSE42: Cost of 4 for VF 4: profitable to scalarize store i64 %valB, ptr %out, align 8
; SSE42: Cost of 8 for VF 8: profitable to scalarize store i64 %valB, ptr %out, align 8
; SSE42: Cost of 16 for VF 16: profitable to scalarize store i64 %valB, ptr %out, align 8
;
; AVX1-LABEL: 'test'
+; AVX1: Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
; AVX1: Cost of 8 for VF 2: WIDEN store vp<[[VP7:%[0-9]+]]>, ir<%valB>, ir<%canStore>
; AVX1: Cost of 8 for VF 4: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
; AVX1: Cost of 16 for VF 8: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
@@ -35,6 +38,7 @@ define void @test(ptr %C) {
; AVX1: Cost of 64 for VF 32: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
;
; AVX2-LABEL: 'test'
+; AVX2: Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
; AVX2: Cost of 8 for VF 2: WIDEN store vp<[[VP7:%[0-9]+]]>, ir<%valB>, ir<%canStore>
; AVX2: Cost of 8 for VF 4: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
; AVX2: Cost of 16 for VF 8: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
@@ -42,6 +46,7 @@ define void @test(ptr %C) {
; AVX2: Cost of 64 for VF 32: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
;
; AVX512-LABEL: 'test'
+; AVX512: Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
; AVX512: Cost of 1 for VF 2: WIDEN store vp<[[VP7:%[0-9]+]]>, ir<%valB>, ir<%canStore>
; AVX512: Cost of 1 for VF 4: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
; AVX512: Cost of 1 for VF 8: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i8.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i8.ll
index 721aa7ba285b18..a8ccf6c2e6639e 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i8.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i8.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*store i8 %valB, ptr %out" --filter "Cost of [1-9][0-9]*(.[0-9]+)? for VF [0-9]+: (profitable to scalarize\s+store i8 %valB|WIDEN store .*, ir<%valB>|REPLICATE store ir<%valB>)"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT store ir<%valB>" --filter "Cost of [1-9][0-9]*(.[0-9]+)? for VF [0-9]+: (profitable to scalarize\s+store i8 %valB|WIDEN store .*, ir<%valB>|REPLICATE store ir<%valB>)"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output -vplan-print-metadata=false < %s 2>&1 | FileCheck %s --check-prefixes=SSE2
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse4.2 --debug-only=loop-vectorize --disable-output -vplan-print-metadata=false < %s 2>&1 | FileCheck %s --check-prefixes=SSE42
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output -vplan-print-metadata=false < %s 2>&1 | FileCheck %s --check-prefixes=AVX1
@@ -16,18 +16,21 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test(ptr %C) {
; SSE2-LABEL: 'test'
+; SSE2: Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
; SSE2: Cost of 2.5 for VF 2: profitable to scalarize store i8 %valB, ptr %out, align 1
; SSE2: Cost of 5.5 for VF 4: profitable to scalarize store i8 %valB, ptr %out, align 1
; SSE2: Cost of 11.5 for VF 8: profitable to scalarize store i8 %valB, ptr %out, align 1
; SSE2: Cost of 23.5 for VF 16: profitable to scalarize store i8 %valB, ptr %out, align 1
;
; SSE42-LABEL: 'test'
+; SSE42: Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
; SSE42: Cost of 2 for VF 2: profitable to scalarize store i8 %valB, ptr %out, align 1
; SSE42: Cost of 4 for VF 4: profitable to scalarize store i8 %valB, ptr %out, align 1
; SSE42: Cost of 8 for VF 8: profitable to scalarize store i8 %valB, ptr %out, align 1
; SSE42: Cost of 16 for VF 16: profitable to scalarize store i8 %valB, ptr %out, align 1
;
; AVX1-LABEL: 'test'
+; AVX1: Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
; AVX1: Cost of 2 for VF 2: profitable to scalarize store i8 %valB, ptr %out, align 1
; AVX1: Cost of 4 for VF 4: profitable to scalarize store i8 %valB, ptr %out, align 1
; AVX1: Cost of 8 for VF 8: profitable to scalarize store i8 %valB, ptr %out, align 1
@@ -35,6 +38,7 @@ define void @test(ptr %C) {
; AVX1: Cost of 32.5 for VF 32: profitable to scalarize store i8 %valB, ptr %out, align 1
;
; AVX2-LABEL: 'test'
+; AVX2: Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
; AVX2: Cost of 2 for VF 2: profitable to scalarize store i8 %valB, ptr %out, align 1
; AVX2: Cost of 4 for VF 4: profitable to scalarize store i8 %valB, ptr %out, align 1
; AVX2: Cost of 8 for VF 8: profitable to scalarize store i8 %valB, ptr %out, align 1
@@ -42,6 +46,7 @@ define void @test(ptr %C) {
; AVX2: Cost of 32.5 for VF 32: profitable to scalarize store i8 %valB, ptr %out, align 1
;
; AVX512-LABEL: 'test'
+; AVX512: Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
; AVX512: Cost of 2 for VF 2: WIDEN store vp<[[VP7:%[0-9]+]]>, ir<%valB>, ir<%canStore>
; AVX512: Cost of 2 for VF 4: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
; AVX512: Cost of 2 for VF 8: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i16-with-i8-index.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i16-with-i8-index.ll
index 9daa35ed531659..b40147e8ac3553 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i16-with-i8-index.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i16-with-i8-index.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*store i16 %valB, ptr %out" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (WIDEN store|REPLICATE store ir<%valB>)"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT store ir<%valB>" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (WIDEN store|REPLICATE store ir<%valB>)"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=SSE2
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse4.2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=SSE42
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=AVX1
@@ -17,18 +17,21 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
+; SSE2: Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
; SSE2: Cost of 28 for VF 2: REPLICATE store ir<%valB>, ir<%out>
; SSE2: Cost of 56 for VF 4: REPLICATE store ir<%valB>, ir<%out>
; SSE2: Cost of 112 for VF 8: REPLICATE store ir<%valB>, ir<%out>
; SSE2: Cost of 224 for VF 16: REPLICATE store ir<%valB>, ir<%out>
;
; SSE42-LABEL: 'test'
+; SSE42: Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
; SSE42: Cost of 26 for VF 2: REPLICATE store ir<%valB>, ir<%out>
; SSE42: Cost of 52 for VF 4: REPLICATE store ir<%valB>, ir<%out>
; SSE42: Cost of 104 for VF 8: REPLICATE store ir<%valB>, ir<%out>
; SSE42: Cost of 208 for VF 16: REPLICATE store ir<%valB>, ir<%out>
;
; AVX1-LABEL: 'test'
+; AVX1: Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
; AVX1: Cost of 26 for VF 2: REPLICATE store ir<%valB>, ir<%out>
; AVX1: Cost of 53 for VF 4: REPLICATE store ir<%valB>, ir<%out>
; AVX1: Cost of 106 for VF 8: REPLICATE store ir<%valB>, ir<%out>
@@ -36,6 +39,7 @@ define void @test() {
; AVX1: Cost of 426 for VF 32: REPLICATE store ir<%valB>, ir<%out>
;
; AVX2-LABEL: 'test'
+; AVX2: Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
; AVX2: Cost of 6 for VF 2: REPLICATE store ir<%valB>, ir<%out>
; AVX2: Cost of 13 for VF 4: REPLICATE store ir<%valB>, ir<%out>
; AVX2: Cost of 26 for VF 8: REPLICATE store ir<%valB>, ir<%out>
@@ -43,6 +47,7 @@ define void @test() {
; AVX2: Cost of 106 for VF 32: REPLICATE store ir<%valB>, ir<%out>
;
; AVX512-LABEL: 'test'
+; AVX512: Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
; AVX512: Cost of 6 for VF 2: REPLICATE store ir<%valB>, ir<%out>
; AVX512: Cost of 13 for VF 4: REPLICATE store ir<%valB>, ir<%out>
; AVX512: Cost of 27 for VF 8: REPLICATE store ir<%valB>, ir<%out>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i32-with-i8-index.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i32-with-i8-index.ll
index 671842c2645cf2..a8cf9c1ffa5942 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i32-with-i8-index.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i32-with-i8-index.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*store i32 %valB, ptr %out" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (WIDEN store|REPLICATE store ir<%valB>)"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT store ir<%valB>" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (WIDEN store|REPLICATE store ir<%valB>)"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=SSE2
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse4.2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=SSE42
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=AVX1
@@ -17,18 +17,21 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
+; SSE2: Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
; SSE2: Cost of 29 for VF 2: REPLICATE store ir<%valB>, ir<%out>
; SSE2: Cost of 59 for VF 4: REPLICATE store ir<%valB>, ir<%out>
; SSE2: Cost of 118 for VF 8: REPLICATE store ir<%valB>, ir<%out>
; SSE2: Cost of 236 for VF 16: REPLICATE store ir<%valB>, ir<%out>
;
; SSE42-LABEL: 'test'
+; SSE42: Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
; SSE42: Cost of 26 for VF 2: REPLICATE store ir<%valB>, ir<%out>
; SSE42: Cost of 52 for VF 4: REPLICATE store ir<%valB>, ir<%out>
; SSE42: Cost of 104 for VF 8: REPLICATE store ir<%valB>, ir<%out>
; SSE42: Cost of 208 for VF 16: REPLICATE store ir<%valB>, ir<%out>
;
; AVX1-LABEL: 'test'
+; AVX1: Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
; AVX1: Cost of 26 for VF 2: REPLICATE store ir<%valB>, ir<%out>
; AVX1: Cost of 53 for VF 4: REPLICATE store ir<%valB>, ir<%out>
; AVX1: Cost of 107 for VF 8: REPLICATE store ir<%valB>, ir<%out>
@@ -36,6 +39,7 @@ define void @test() {
; AVX1: Cost of 428 for VF 32: REPLICATE store ir<%valB>, ir<%out>
;
; AVX2-LABEL: 'test'
+; AVX2: Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
; AVX2: Cost of 6 for VF 2: REPLICATE store ir<%valB>, ir<%out>
; AVX2: Cost of 13 for VF 4: REPLICATE store ir<%valB>, ir<%out>
; AVX2: Cost of 27 for VF 8: REPLICATE store ir<%valB>, ir<%out>
@@ -43,6 +47,7 @@ define void @test() {
; AVX2: Cost of 108 for VF 32: REPLICATE store ir<%valB>, ir<%out>
;
; AVX512-LABEL: 'test'
+; AVX512: Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
; AVX512: Cost of 6 for VF 2: REPLICATE store ir<%valB>, ir<%out>
; AVX512: Cost of 13 for VF 4: REPLICATE store ir<%valB>, ir<%out>
; AVX512: Cost of 10 for VF 8: WIDEN store ir<%out>, ir<%valB>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i64-with-i8-index.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i64-with-i8-index.ll
index dc775779060122..960bd5e8ae2026 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i64-with-i8-index.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i64-with-i8-index.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*store i64 %valB, ptr %out" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (WIDEN store|REPLICATE store ir<%valB>)"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT store ir<%valB>" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (WIDEN store|REPLICATE store ir<%valB>)"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=SSE2
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse4.2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=SSE42
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=AVX1
@@ -17,18 +17,21 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
+; SSE2: Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
; SSE2: Cost of 29 for VF 2: REPLICATE store ir<%valB>, ir<%out>
; SSE2: Cost of 58 for VF 4: REPLICATE store ir<%valB>, ir<%out>
; SSE2: Cost of 116 for VF 8: REPLICATE store ir<%valB>, ir<%out>
; SSE2: Cost of 232 for VF 16: REPLICATE store ir<%valB>, ir<%out>
;
; SSE42-LABEL: 'test'
+; SSE42: Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
; SSE42: Cost of 26 for VF 2: REPLICATE store ir<%valB>, ir<%out>
; SSE42: Cost of 52 for VF 4: REPLICATE store ir<%valB>, ir<%out>
; SSE42: Cost of 104 for VF 8: REPLICATE store ir<%valB>, ir<%out>
; SSE42: Cost of 208 for VF 16: REPLICATE store ir<%valB>, ir<%out>
;
; AVX1-LABEL: 'test'
+; AVX1: Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
; AVX1: Cost of 26 for VF 2: REPLICATE store ir<%valB>, ir<%out>
; AVX1: Cost of 54 for VF 4: REPLICATE store ir<%valB>, ir<%out>
; AVX1: Cost of 108 for VF 8: REPLICATE store ir<%valB>, ir<%out>
@@ -36,6 +39,7 @@ define void @test() {
; AVX1: Cost of 432 for VF 32: REPLICATE store ir<%valB>, ir<%out>
;
; AVX2-LABEL: 'test'
+; AVX2: Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
; AVX2: Cost of 6 for VF 2: REPLICATE store ir<%valB>, ir<%out>
; AVX2: Cost of 14 for VF 4: REPLICATE store ir<%valB>, ir<%out>
; AVX2: Cost of 28 for VF 8: REPLICATE store ir<%valB>, ir<%out>
@@ -43,6 +47,7 @@ define void @test() {
; AVX2: Cost of 112 for VF 32: REPLICATE store ir<%valB>, ir<%out>
;
; AVX512-LABEL: 'test'
+; AVX512: Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
; AVX512: Cost of 6 for VF 2: REPLICATE store ir<%valB>, ir<%out>
; AVX512: Cost of 14 for VF 4: REPLICATE store ir<%valB>, ir<%out>
; AVX512: Cost of 10 for VF 8: WIDEN store ir<%out>, ir<%valB>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i8-with-i8-index.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i8-with-i8-index.ll
index c484b526ac1cb3..840c699af0be5a 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i8-with-i8-index.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i8-with-i8-index.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+(.[0-9]+)? for VF 1 For instruction:\s*store i8 %valB, ptr %out" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (WIDEN store|REPLICATE store ir<%valB>)"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT store ir<%valB>" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: (WIDEN store|REPLICATE store ir<%valB>)"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=SSE2
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+sse4.2 --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=SSE42
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx --debug-only=loop-vectorize --disable-output < %s 2>&1 | FileCheck %s --check-prefixes=AVX1
@@ -17,18 +17,21 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; SSE2-LABEL: 'test'
+; SSE2: Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
; SSE2: Cost of 29 for VF 2: REPLICATE store ir<%valB>, ir<%out>
; SSE2: Cost of 59 for VF 4: REPLICATE store ir<%valB>, ir<%out>
; SSE2: Cost of 119 for VF 8: REPLICATE store ir<%valB>, ir<%out>
; SSE2: Cost of 239 for VF 16: REPLICATE store ir<%valB>, ir<%out>
;
; SSE42-LABEL: 'test'
+; SSE42: Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
; SSE42: Cost of 26 for VF 2: REPLICATE store ir<%valB>, ir<%out>
; SSE42: Cost of 52 for VF 4: REPLICATE store ir<%valB>, ir<%out>
; SSE42: Cost of 104 for VF 8: REPLICATE store ir<%valB>, ir<%out>
; SSE42: Cost of 208 for VF 16: REPLICATE store ir<%valB>, ir<%out>
;
; AVX1-LABEL: 'test'
+; AVX1: Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
; AVX1: Cost of 26 for VF 2: REPLICATE store ir<%valB>, ir<%out>
; AVX1: Cost of 53 for VF 4: REPLICATE store ir<%valB>, ir<%out>
; AVX1: Cost of 106 for VF 8: REPLICATE store ir<%valB>, ir<%out>
@@ -36,6 +39,7 @@ define void @test() {
; AVX1: Cost of 425 for VF 32: REPLICATE store ir<%valB>, ir<%out>
;
; AVX2-LABEL: 'test'
+; AVX2: Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
; AVX2: Cost of 6 for VF 2: REPLICATE store ir<%valB>, ir<%out>
; AVX2: Cost of 13 for VF 4: REPLICATE store ir<%valB>, ir<%out>
; AVX2: Cost of 26 for VF 8: REPLICATE store ir<%valB>, ir<%out>
@@ -43,6 +47,7 @@ define void @test() {
; AVX2: Cost of 105 for VF 32: REPLICATE store ir<%valB>, ir<%out>
;
; AVX512-LABEL: 'test'
+; AVX512: Cost of 1 for VF 1: EMIT store ir<%valB>, ir<%out>
; AVX512: Cost of 6 for VF 2: REPLICATE store ir<%valB>, ir<%out>
; AVX512: Cost of 13 for VF 4: REPLICATE store ir<%valB>, ir<%out>
; AVX512: Cost of 27 for VF 8: REPLICATE store ir<%valB>, ir<%out>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/strided-load-i16.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/strided-load-i16.ll
index 3c27644f4ebcf0..c0c810e4bfaf91 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/strided-load-i16.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/strided-load-i16.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Found an estimated cost of [0-9]+(.[0-9]+)? for VF [0-9]+ For instruction:\s*%1" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: INTERLEAVE-GROUP" --version 6
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%1> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: INTERLEAVE-GROUP" --version 6
; REQUIRES: asserts
; RUN: opt -passes=loop-vectorize -S -mattr=avx512bw --debug-only=loop-vectorize --disable-output < %s 2>&1| FileCheck %s
@@ -10,6 +10,7 @@ target triple = "x86_64-unknown-linux-gnu"
define void @load_i16_stride2() {
; CHECK-LABEL: 'load_i16_stride2'
+; CHECK: Cost of 1 for VF 1: EMIT-SCALAR ir<%1> = load ir<%arrayidx>
; CHECK: Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%arrayidx>
; CHECK: Cost of 1 for VF 4: INTERLEAVE-GROUP with factor 2, ir<%arrayidx>
; CHECK: Cost of 2 for VF 8: INTERLEAVE-GROUP with factor 2, ir<%arrayidx>
@@ -36,6 +37,7 @@ for.end:
define void @load_i16_stride3() {
; CHECK-LABEL: 'load_i16_stride3'
+; CHECK: Cost of 1 for VF 1: EMIT-SCALAR ir<%1> = load ir<%arrayidx>
; CHECK: Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%arrayidx>
; CHECK: Cost of 2 for VF 4: INTERLEAVE-GROUP with factor 3, ir<%arrayidx>
; CHECK: Cost of 2 for VF 8: INTERLEAVE-GROUP with factor 3, ir<%arrayidx>
@@ -62,6 +64,7 @@ for.end:
define void @load_i16_stride4() {
; CHECK-LABEL: 'load_i16_stride4'
+; CHECK: Cost of 1 for VF 1: EMIT-SCALAR ir<%1> = load ir<%arrayidx>
; CHECK: Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%arrayidx>
; CHECK: Cost of 2 for VF 4: INTERLEAVE-GROUP with factor 4, ir<%arrayidx>
; CHECK: Cost of 2 for VF 8: INTERLEAVE-GROUP with factor 4, ir<%arrayidx>
@@ -88,6 +91,7 @@ for.end:
define void @load_i16_stride5() {
; CHECK-LABEL: 'load_i16_stride5'
+; CHECK: Cost of 1 for VF 1: EMIT-SCALAR ir<%1> = load ir<%arrayidx>
; CHECK: Cost of 2 for VF 2: INTERLEAVE-GROUP with factor 5, ir<%arrayidx>
; CHECK: Cost of 2 for VF 4: INTERLEAVE-GROUP with factor 5, ir<%arrayidx>
; CHECK: Cost of 3 for VF 8: INTERLEAVE-GROUP with factor 5, ir<%arrayidx>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/strided-load-i32.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/strided-load-i32.ll
index 8bc141496b822b..afa24bb18fdc54 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/strided-load-i32.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/strided-load-i32.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Found an estimated cost of [0-9]+(.[0-9]+)? for VF [0-9]+ For instruction:\s*%1" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: INTERLEAVE-GROUP" --version 6
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%1> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: INTERLEAVE-GROUP" --version 6
; REQUIRES: asserts
; RUN: opt -passes=loop-vectorize -S -mattr=avx512f --debug-only=loop-vectorize --disable-output < %s 2>&1| FileCheck %s
@@ -10,6 +10,7 @@ target triple = "x86_64-unknown-linux-gnu"
define void @load_int_stride2() {
; CHECK-LABEL: 'load_int_stride2'
+; CHECK: Cost of 1 for VF 1: EMIT-SCALAR ir<%1> = load ir<%arrayidx>
; CHECK: Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%arrayidx>
; CHECK: Cost of 1 for VF 4: INTERLEAVE-GROUP with factor 2, ir<%arrayidx>
; CHECK: Cost of 1 for VF 8: INTERLEAVE-GROUP with factor 2, ir<%arrayidx>
@@ -35,6 +36,7 @@ for.end:
define void @load_int_stride3() {
; CHECK-LABEL: 'load_int_stride3'
+; CHECK: Cost of 1 for VF 1: EMIT-SCALAR ir<%1> = load ir<%arrayidx>
; CHECK: Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%arrayidx>
; CHECK: Cost of 1 for VF 4: INTERLEAVE-GROUP with factor 3, ir<%arrayidx>
; CHECK: Cost of 3 for VF 8: INTERLEAVE-GROUP with factor 3, ir<%arrayidx>
@@ -60,6 +62,7 @@ for.end:
define void @load_int_stride4() {
; CHECK-LABEL: 'load_int_stride4'
+; CHECK: Cost of 1 for VF 1: EMIT-SCALAR ir<%1> = load ir<%arrayidx>
; CHECK: Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%arrayidx>
; CHECK: Cost of 1 for VF 4: INTERLEAVE-GROUP with factor 4, ir<%arrayidx>
; CHECK: Cost of 3 for VF 8: INTERLEAVE-GROUP with factor 4, ir<%arrayidx>
@@ -85,6 +88,7 @@ for.end:
define void @load_int_stride5() {
; CHECK-LABEL: 'load_int_stride5'
+; CHECK: Cost of 1 for VF 1: EMIT-SCALAR ir<%1> = load ir<%arrayidx>
; CHECK: Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 5, ir<%arrayidx>
; CHECK: Cost of 3 for VF 4: INTERLEAVE-GROUP with factor 5, ir<%arrayidx>
; CHECK: Cost of 5 for VF 8: INTERLEAVE-GROUP with factor 5, ir<%arrayidx>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/strided-load-i64.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/strided-load-i64.ll
index 1c91d01340a4b9..ca7865ce4f307c 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/strided-load-i64.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/strided-load-i64.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Found an estimated cost of [0-9]+(.[0-9]+)? for VF [0-9]+ For instruction:\s*%1" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: INTERLEAVE-GROUP" --version 6
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%1> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: INTERLEAVE-GROUP" --version 6
; REQUIRES: asserts
; RUN: opt -passes=loop-vectorize -S -mattr=avx512f --debug-only=loop-vectorize --disable-output < %s 2>&1| FileCheck %s
@@ -10,6 +10,7 @@ target triple = "x86_64-unknown-linux-gnu"
define void @load_i64_stride2() {
; CHECK-LABEL: 'load_i64_stride2'
+; CHECK: Cost of 1 for VF 1: EMIT-SCALAR ir<%1> = load ir<%arrayidx>
; CHECK: Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%arrayidx>
; CHECK: Cost of 1 for VF 4: INTERLEAVE-GROUP with factor 2, ir<%arrayidx>
; CHECK: Cost of 3 for VF 8: INTERLEAVE-GROUP with factor 2, ir<%arrayidx>
@@ -34,6 +35,7 @@ for.end:
define void @load_i64_stride3() {
; CHECK-LABEL: 'load_i64_stride3'
+; CHECK: Cost of 1 for VF 1: EMIT-SCALAR ir<%1> = load ir<%arrayidx>
; CHECK: Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%arrayidx>
; CHECK: Cost of 3 for VF 4: INTERLEAVE-GROUP with factor 3, ir<%arrayidx>
; CHECK: Cost of 5 for VF 8: INTERLEAVE-GROUP with factor 3, ir<%arrayidx>
@@ -58,6 +60,7 @@ for.end:
define void @load_i64_stride4() {
; CHECK-LABEL: 'load_i64_stride4'
+; CHECK: Cost of 1 for VF 1: EMIT-SCALAR ir<%1> = load ir<%arrayidx>
; CHECK: Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%arrayidx>
; CHECK: Cost of 3 for VF 4: INTERLEAVE-GROUP with factor 4, ir<%arrayidx>
; CHECK: Cost of 8 for VF 8: INTERLEAVE-GROUP with factor 4, ir<%arrayidx>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/strided-load-i8.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/strided-load-i8.ll
index b87607872fdccf..2548fbc4adeb64 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/strided-load-i8.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/strided-load-i8.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Found an estimated cost of [0-9]+(.[0-9]+)? for VF [0-9]+ For instruction:\s*%1" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: INTERLEAVE-GROUP" --version 6
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+(.[0-9]+)? for VF 1: EMIT-SCALAR ir<%1> = load" --filter "Cost of [0-9]+(.[0-9]+)? for VF [0-9]+: INTERLEAVE-GROUP" --version 6
; REQUIRES: asserts
; RUN: opt -passes=loop-vectorize -S -mattr=avx512bw --debug-only=loop-vectorize --disable-output < %s 2>&1| FileCheck %s
@@ -10,6 +10,7 @@ target triple = "x86_64-unknown-linux-gnu"
define void @load_i8_stride2() {
; CHECK-LABEL: 'load_i8_stride2'
+; CHECK: Cost of 1 for VF 1: EMIT-SCALAR ir<%1> = load ir<%arrayidx>
; CHECK: Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%arrayidx>
; CHECK: Cost of 1 for VF 4: INTERLEAVE-GROUP with factor 2, ir<%arrayidx>
; CHECK: Cost of 1 for VF 8: INTERLEAVE-GROUP with factor 2, ir<%arrayidx>
@@ -37,6 +38,7 @@ for.end:
define void @load_i8_stride3() {
; CHECK-LABEL: 'load_i8_stride3'
+; CHECK: Cost of 1 for VF 1: EMIT-SCALAR ir<%1> = load ir<%arrayidx>
; CHECK: Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 3, ir<%arrayidx>
; CHECK: Cost of 1 for VF 4: INTERLEAVE-GROUP with factor 3, ir<%arrayidx>
; CHECK: Cost of 4 for VF 8: INTERLEAVE-GROUP with factor 3, ir<%arrayidx>
@@ -64,6 +66,7 @@ for.end:
define void @load_i8_stride4() {
; CHECK-LABEL: 'load_i8_stride4'
+; CHECK: Cost of 1 for VF 1: EMIT-SCALAR ir<%1> = load ir<%arrayidx>
; CHECK: Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 4, ir<%arrayidx>
; CHECK: Cost of 1 for VF 4: INTERLEAVE-GROUP with factor 4, ir<%arrayidx>
; CHECK: Cost of 4 for VF 8: INTERLEAVE-GROUP with factor 4, ir<%arrayidx>
@@ -91,6 +94,7 @@ for.end:
define void @load_i8_stride5() {
; CHECK-LABEL: 'load_i8_stride5'
+; CHECK: Cost of 1 for VF 1: EMIT-SCALAR ir<%1> = load ir<%arrayidx>
; CHECK: Cost of 1 for VF 2: INTERLEAVE-GROUP with factor 5, ir<%arrayidx>
; CHECK: Cost of 4 for VF 4: INTERLEAVE-GROUP with factor 5, ir<%arrayidx>
; CHECK: Cost of 8 for VF 8: INTERLEAVE-GROUP with factor 5, ir<%arrayidx>
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/vpinstruction-cost.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/vpinstruction-cost.ll
index 3e9ae12df16cba..6247b3910a609a 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/vpinstruction-cost.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/vpinstruction-cost.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+ for VF 1 For instruction" --filter "Cost of"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of"
; RUN: opt -S -passes=loop-vectorize -mcpu=skylake-avx512 -mtriple=x86_64-apple-macosx -debug -disable-output -S %s 2>&1 | FileCheck %s
; REQUIRES: asserts
@@ -12,10 +12,10 @@ define void @wide_or_replaced_with_add_vpinstruction(ptr %src, ptr noalias %dst)
; CHECK: Cost of 1 for VF 1: EMIT-SCALAR ir<%l> = load ir<%g.src>
; CHECK: Cost of 1 for VF 1: EMIT ir<%iv.4> = add nuw nsw ir<%iv>, ir<4>
; CHECK: Cost of 1 for VF 1: EMIT ir<%c> = icmp ule ir<%l>, ir<128>
-; CHECK: Cost of 0 for VF 1: EMIT branch-on-cond ir<%c> (!vplan.prof.estimated estimated {1073741824, 1073741824})
-; CHECK: Cost of 1 for VF 1: EMIT ir<%or> = or disjoint ir<%iv.4>, ir<1> (!vplan.execution.frequency 4611686018427387904 (50%, estimated))
-; CHECK: Cost of 0 for VF 1: EMIT ir<%g.dst> = getelementptr inbounds ir<%dst>, ir<%or> (!vplan.execution.frequency 4611686018427387904 (50%, estimated))
-; CHECK: Cost of 1 for VF 1: EMIT store ir<%iv.4>, ir<%g.dst> (!vplan.execution.frequency 4611686018427387904 (50%, estimated))
+; CHECK: Cost of 0 for VF 1: EMIT branch-on-cond ir<%c>
+; CHECK: Cost of 1 for VF 1: EMIT ir<%or> = or disjoint ir<%iv.4>, ir<1>
+; CHECK: Cost of 0 for VF 1: EMIT ir<%g.dst> = getelementptr inbounds ir<%dst>, ir<%or>
+; CHECK: Cost of 1 for VF 1: EMIT store ir<%iv.4>, ir<%g.dst>
; CHECK: Cost of 1 for VF 1: EMIT ir<%iv.next> = add nuw nsw ir<%iv>, ir<1>
; CHECK: Cost of 1 for VF 1: EMIT ir<%exitcond> = icmp eq ir<%iv.next>, ir<32>
; CHECK: Cost of 0 for VF 1: EMIT branch-on-cond ir<%exitcond>
@@ -29,7 +29,7 @@ define void @wide_or_replaced_with_add_vpinstruction(ptr %src, ptr noalias %dst)
; CHECK: Cost of 1 for VF 2: EMIT ir<%or> = add ir<%iv.4>, ir<1>
; CHECK: Cost of 0 for VF 2: CLONE ir<%g.dst> = getelementptr ir<%dst>, ir<%or>
; CHECK: Cost of 0 for VF 2: vp<[[VP6:%[0-9]+]]> = vector-pointer i64, ir<%g.dst>, ir<1>
-; CHECK: Cost of 1 for VF 2: WIDEN store vp<[[VP6]]>, ir<%iv.4>, ir<%c> (!vplan.execution.frequency 4611686018427387904 (50%, estimated))
+; CHECK: Cost of 1 for VF 2: WIDEN store vp<[[VP6]]>, ir<%iv.4>, ir<%c>
; CHECK: Cost of 0 for VF 2: EMIT vp<%index.next> = add nuw vp<[[VP3]]>, vp<[[VP1:%[0-9]+]]>
; CHECK: Cost of 1 for VF 2: EMIT branch-on-count vp<%index.next>, vp<[[VP2:%[0-9]+]]>
; CHECK: Cost of 0 for VF 2: vector loop backedge
@@ -52,7 +52,7 @@ define void @wide_or_replaced_with_add_vpinstruction(ptr %src, ptr noalias %dst)
; CHECK: Cost of 1 for VF 4: EMIT ir<%or> = add ir<%iv.4>, ir<1>
; CHECK: Cost of 0 for VF 4: CLONE ir<%g.dst> = getelementptr ir<%dst>, ir<%or>
; CHECK: Cost of 0 for VF 4: vp<[[VP6]]> = vector-pointer i64, ir<%g.dst>, ir<1>
-; CHECK: Cost of 1 for VF 4: WIDEN store vp<[[VP6]]>, ir<%iv.4>, ir<%c> (!vplan.execution.frequency 4611686018427387904 (50%, estimated))
+; CHECK: Cost of 1 for VF 4: WIDEN store vp<[[VP6]]>, ir<%iv.4>, ir<%c>
; CHECK: Cost of 0 for VF 4: EMIT vp<%index.next> = add nuw vp<[[VP3]]>, vp<[[VP1]]>
; CHECK: Cost of 1 for VF 4: EMIT branch-on-count vp<%index.next>, vp<[[VP2]]>
; CHECK: Cost of 0 for VF 4: vector loop backedge
@@ -176,11 +176,11 @@ define void @test_vpinstruction_switch_cost(ptr %start, ptr %end) {
; CHECK-LABEL: 'test_vpinstruction_switch_cost'
; CHECK: Cost of 0 for VF 1: EMIT-SCALAR ir<%ptr.iv> = phi [ ir<%start>, vector.ph ], [ ir<%ptr.iv.next>, loop.latch ]
; CHECK: Cost of 1 for VF 1: EMIT-SCALAR ir<%l> = load ir<%ptr.iv>
-; CHECK: Cost of 0 for VF 1: EMIT switch ir<%l>, ir<-12>, ir<13>, ir<0> (!vplan.prof.estimated estimated {536870912, 536870912, 536870912, 536870912})
-; CHECK: Cost of 1 for VF 1: EMIT store ir<1>, ir<%ptr.iv> (!vplan.execution.frequency 2305843009213693952 (25%, estimated))
-; CHECK: Cost of 1 for VF 1: EMIT store ir<0>, ir<%ptr.iv> (!vplan.execution.frequency 2305843009213693952 (25%, estimated))
-; CHECK: Cost of 1 for VF 1: EMIT store ir<42>, ir<%ptr.iv> (!vplan.execution.frequency 2305843009213693952 (25%, estimated))
-; CHECK: Cost of 1 for VF 1: EMIT store ir<2>, ir<%ptr.iv> (!vplan.execution.frequency 2305843009213693952 (25%, estimated))
+; CHECK: Cost of 0 for VF 1: EMIT switch ir<%l>, ir<-12>, ir<13>, ir<0>
+; CHECK: Cost of 1 for VF 1: EMIT store ir<1>, ir<%ptr.iv>
+; CHECK: Cost of 1 for VF 1: EMIT store ir<0>, ir<%ptr.iv>
+; CHECK: Cost of 1 for VF 1: EMIT store ir<42>, ir<%ptr.iv>
+; CHECK: Cost of 1 for VF 1: EMIT store ir<2>, ir<%ptr.iv>
; CHECK: Cost of 0 for VF 1: EMIT ir<%ptr.iv.next> = getelementptr inbounds ir<%ptr.iv>, ir<1>
; CHECK: Cost of 1 for VF 1: EMIT ir<%ec> = icmp eq ir<%ptr.iv.next>, ir<%end>
; CHECK: Cost of 0 for VF 1: EMIT branch-on-cond ir<%ec>
@@ -196,13 +196,13 @@ define void @test_vpinstruction_switch_cost(ptr %start, ptr %end) {
; CHECK: Cost of 0 for VF 2: EMIT vp<[[VP13:%[0-9]+]]> = or vp<[[VP12]]>, vp<[[VP11]]>
; CHECK: Cost of 1 for VF 2: EMIT vp<[[VP14:%[0-9]+]]> = not vp<[[VP13]]>
; CHECK: Cost of 0 for VF 2: vp<[[VP15:%[0-9]+]]> = vector-pointer i64, vp<%next.gep>, ir<1>
-; CHECK: Cost of 1 for VF 2: WIDEN store vp<[[VP15]]>, ir<1>, vp<[[VP11]]> (!vplan.execution.frequency 2305843009213693952 (25%, estimated))
+; CHECK: Cost of 1 for VF 2: WIDEN store vp<[[VP15]]>, ir<1>, vp<[[VP11]]>
; CHECK: Cost of 0 for VF 2: vp<[[VP16:%[0-9]+]]> = vector-pointer i64, vp<%next.gep>, ir<1>
-; CHECK: Cost of 1 for VF 2: WIDEN store vp<[[VP16]]>, ir<0>, vp<[[VP10]]> (!vplan.execution.frequency 2305843009213693952 (25%, estimated))
+; CHECK: Cost of 1 for VF 2: WIDEN store vp<[[VP16]]>, ir<0>, vp<[[VP10]]>
; CHECK: Cost of 0 for VF 2: vp<[[VP17:%[0-9]+]]> = vector-pointer i64, vp<%next.gep>, ir<1>
-; CHECK: Cost of 1 for VF 2: WIDEN store vp<[[VP17]]>, ir<42>, vp<[[VP9]]> (!vplan.execution.frequency 2305843009213693952 (25%, estimated))
+; CHECK: Cost of 1 for VF 2: WIDEN store vp<[[VP17]]>, ir<42>, vp<[[VP9]]>
; CHECK: Cost of 0 for VF 2: vp<[[VP18:%[0-9]+]]> = vector-pointer i64, vp<%next.gep>, ir<1>
-; CHECK: Cost of 1 for VF 2: WIDEN store vp<[[VP18]]>, ir<2>, vp<[[VP14]]> (!vplan.execution.frequency 2305843009213693952 (25%, estimated))
+; CHECK: Cost of 1 for VF 2: WIDEN store vp<[[VP18]]>, ir<2>, vp<[[VP14]]>
; CHECK: Cost of 0 for VF 2: EMIT vp<%index.next> = add nuw vp<[[VP5]]>, vp<[[VP1:%[0-9]+]]>
; CHECK: Cost of 1 for VF 2: EMIT branch-on-count vp<%index.next>, vp<[[VP2:%[0-9]+]]>
; CHECK: Cost of 0 for VF 2: vector loop backedge
@@ -226,13 +226,13 @@ define void @test_vpinstruction_switch_cost(ptr %start, ptr %end) {
; CHECK: Cost of 0 for VF 4: EMIT vp<[[VP13]]> = or vp<[[VP12]]>, vp<[[VP11]]>
; CHECK: Cost of 1 for VF 4: EMIT vp<[[VP14]]> = not vp<[[VP13]]>
; CHECK: Cost of 0 for VF 4: vp<[[VP15]]> = vector-pointer i64, vp<%next.gep>, ir<1>
-; CHECK: Cost of 1 for VF 4: WIDEN store vp<[[VP15]]>, ir<1>, vp<[[VP11]]> (!vplan.execution.frequency 2305843009213693952 (25%, estimated))
+; CHECK: Cost of 1 for VF 4: WIDEN store vp<[[VP15]]>, ir<1>, vp<[[VP11]]>
; CHECK: Cost of 0 for VF 4: vp<[[VP16]]> = vector-pointer i64, vp<%next.gep>, ir<1>
-; CHECK: Cost of 1 for VF 4: WIDEN store vp<[[VP16]]>, ir<0>, vp<[[VP10]]> (!vplan.execution.frequency 2305843009213693952 (25%, estimated))
+; CHECK: Cost of 1 for VF 4: WIDEN store vp<[[VP16]]>, ir<0>, vp<[[VP10]]>
; CHECK: Cost of 0 for VF 4: vp<[[VP17]]> = vector-pointer i64, vp<%next.gep>, ir<1>
-; CHECK: Cost of 1 for VF 4: WIDEN store vp<[[VP17]]>, ir<42>, vp<[[VP9]]> (!vplan.execution.frequency 2305843009213693952 (25%, estimated))
+; CHECK: Cost of 1 for VF 4: WIDEN store vp<[[VP17]]>, ir<42>, vp<[[VP9]]>
; CHECK: Cost of 0 for VF 4: vp<[[VP18]]> = vector-pointer i64, vp<%next.gep>, ir<1>
-; CHECK: Cost of 1 for VF 4: WIDEN store vp<[[VP18]]>, ir<2>, vp<[[VP14]]> (!vplan.execution.frequency 2305843009213693952 (25%, estimated))
+; CHECK: Cost of 1 for VF 4: WIDEN store vp<[[VP18]]>, ir<2>, vp<[[VP14]]>
; CHECK: Cost of 0 for VF 4: EMIT vp<%index.next> = add nuw vp<[[VP5]]>, vp<[[VP1]]>
; CHECK: Cost of 1 for VF 4: EMIT branch-on-count vp<%index.next>, vp<[[VP2]]>
; CHECK: Cost of 0 for VF 4: vector loop backedge
diff --git a/llvm/test/tools/UpdateTestChecks/update_analyze_test_checks/Inputs/x86-loopvectorize-costmodel.ll.expected b/llvm/test/tools/UpdateTestChecks/update_analyze_test_checks/Inputs/x86-loopvectorize-costmodel.ll.expected
index 88911d7440d382..d2951e89b50e40 100644
--- a/llvm/test/tools/UpdateTestChecks/update_analyze_test_checks/Inputs/x86-loopvectorize-costmodel.ll.expected
+++ b/llvm/test/tools/UpdateTestChecks/update_analyze_test_checks/Inputs/x86-loopvectorize-costmodel.ll.expected
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "LV: Found an estimated cost of [0-9]+ for VF [0-9]+ For instruction:\s*%v0 = load float, ptr %in0, align 4" --filter "Cost of [0-9]+ for VF [0-9]+: INTERLEAVE-GROUP with factor 2, ir<%in0>" --filter "LV: Found an estimated cost of [0-9]+ for VF [0-9]+ For instruction:\s*%v0 = load float, float\* %in0, align 4"
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "Cost of [0-9]+ for VF 1: EMIT-SCALAR ir<%v0> = load" --filter "Cost of [0-9]+ for VF [0-9]+: INTERLEAVE-GROUP with factor 2, ir<%in0>" --filter "LV: Found an estimated cost of [0-9]+ for VF [0-9]+ For instruction:\s*%v0 = load float, float\* %in0, align 4"
; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -S -mattr=+avx512bw --debug-only=loop-vectorize < %s 2>&1 | FileCheck %s --check-prefixes=CHECK,AVX512
; REQUIRES: asserts
@@ -10,6 +10,7 @@ target triple = "x86_64-unknown-linux-gnu"
define void @test() {
; CHECK-LABEL: 'test'
+; CHECK: Cost of 1 for VF 1: EMIT-SCALAR ir<%v0> = load ir<%in0>
; CHECK: Cost of 3 for VF 2: INTERLEAVE-GROUP with factor 2, ir<%in0>
; CHECK: Cost of 3 for VF 4: INTERLEAVE-GROUP with factor 2, ir<%in0>
; CHECK: Cost of 3 for VF 8: INTERLEAVE-GROUP with factor 2, ir<%in0>
diff --git a/llvm/test/tools/UpdateTestChecks/update_analyze_test_checks/loopvectorize-costmodel.test b/llvm/test/tools/UpdateTestChecks/update_analyze_test_checks/loopvectorize-costmodel.test
index c82ea022edd6f9..53f25d3f34d9ad 100644
--- a/llvm/test/tools/UpdateTestChecks/update_analyze_test_checks/loopvectorize-costmodel.test
+++ b/llvm/test/tools/UpdateTestChecks/update_analyze_test_checks/loopvectorize-costmodel.test
@@ -1,11 +1,11 @@
# REQUIRES: x86-registered-target, asserts
-## Check that --filter works properly with both legacy and VPlan cost model output.
-# RUN: cp -f %S/Inputs/x86-loopvectorize-costmodel.ll %t.ll && %update_analyze_test_checks --filter "LV: Found an estimated cost of [0-9]+ for VF [0-9]+ For instruction:\s*%v0 = load float, ptr %in0, align 4" --filter "Cost of [0-9]+ for VF [0-9]+: INTERLEAVE-GROUP with factor 2, ir<%%in0>" %t.ll
+## Check that --filter works properly for both scalar and vector VF cost model output.
+# RUN: cp -f %S/Inputs/x86-loopvectorize-costmodel.ll %t.ll && %update_analyze_test_checks --filter "Cost of [0-9]+ for VF 1: EMIT-SCALAR ir<%%v0> = load" --filter "Cost of [0-9]+ for VF [0-9]+: INTERLEAVE-GROUP with factor 2, ir<%%in0>" %t.ll
# RUN: diff -u %t.ll %S/Inputs/x86-loopvectorize-costmodel.ll.expected
## Check that running the script again does not change the result:
-# RUN: %update_analyze_test_checks --filter "LV: Found an estimated cost of [0-9]+ for VF [0-9]+ For instruction:\s*%v0 = load float, ptr %in0, align 4" --filter "Cost of [0-9]+ for VF [0-9]+: INTERLEAVE-GROUP with factor 2, ir<%%in0>" %t.ll
+# RUN: %update_analyze_test_checks --filter "Cost of [0-9]+ for VF 1: EMIT-SCALAR ir<%%v0> = load" --filter "Cost of [0-9]+ for VF [0-9]+: INTERLEAVE-GROUP with factor 2, ir<%%in0>" %t.ll
# RUN: diff -u %t.ll %S/Inputs/x86-loopvectorize-costmodel.ll.expected
## Check that running the script again, without arguments, does not change the result:
>From 8c93049f8d4d20d94b7b77ff590b775e5ce6d5b4 Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Tue, 22 Sep 2026 13:57:26 +0100
Subject: [PATCH 6/7] !fixup add -vplan-scalar-cost flag
---
.../Vectorize/LoopVectorizationPlanner.h | 3 +-
.../Transforms/Vectorize/LoopVectorize.cpp | 52 +++++++++++++++++++
.../LoopVectorize/X86/uniformshift.ll | 7 ++-
3 files changed, 59 insertions(+), 3 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
index 75623179a53347..2bca17c0e59865 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
@@ -908,7 +908,8 @@ class LoopVectorizationPlanner {
/// been retired.
InstructionCost cost(VPlan &Plan, ElementCount VF, VPRegisterUsage *RU) const;
- /// Compute the scalar loop cost of InitialVPlan0.
+ /// Compute the scalar loop cost of InitialVPlan0, or using the legacy cost
+ /// model if -vplan-scalar-cost is disabled.
InstructionCost computeScalarCost() const;
/// Precompute costs for certain instructions using the legacy cost model. The
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 9fbb03482e6dcd..a07956f614596c 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -404,6 +404,13 @@ static cl::opt<cl::boolOrDefault>
cl::desc("Override cost based masked intrinsic widening "
"for div/rem instructions"));
+// TODO: This is a temporary option to disable the VPlan code path in case any
+// regressions surface. Will be removed after a transition period.
+static cl::opt<bool> UseVPlanScalarCost(
+ "vplan-scalar-cost", cl::init(true), cl::Hidden,
+ cl::desc("Compute the cost of the scalar loop using the VPlan-based cost "
+ "model. If disabled, fall back to the legacy cost model."));
+
static cl::opt<bool> EnableEarlyExitVectorization(
"enable-early-exit-vectorization", cl::init(true), cl::Hidden,
cl::desc(
@@ -1282,6 +1289,12 @@ class LoopVectorizationCostModel {
Scalars.clear();
}
+ /// Returns the expected execution cost. The unit of the cost does
+ /// not matter because we use the 'cost' units to compare different
+ /// vector widths. The cost that is returned is *not* normalized by
+ /// the factor width.
+ InstructionCost expectedCost(ElementCount VF);
+
/// Returns the execution time cost of an instruction for a given vector
/// width. Vector width of one means scalar.
InstructionCost getInstructionCost(Instruction *I, ElementCount VF);
@@ -4227,6 +4240,42 @@ InstructionCost LoopVectorizationCostModel::computePredInstDiscount(
return Discount;
}
+InstructionCost LoopVectorizationCostModel::expectedCost(ElementCount VF) {
+ InstructionCost Cost;
+ assert(VF.isScalar() && "must only be called for scalar VFs");
+
+ // For each block.
+ for (BasicBlock *BB : TheLoop->blocks()) {
+ InstructionCost BlockCost;
+
+ // For each instruction in the old loop.
+ for (Instruction &I : *BB) {
+ // Skip ignored values.
+ if (ValuesToIgnore.count(&I) ||
+ (VF.isVector() && VecValuesToIgnore.count(&I)))
+ continue;
+
+ InstructionCost C = getInstructionCost(&I, VF);
+
+ // Check if we should override the cost.
+ if (C.isValid() && ForceTargetInstructionCost.getNumOccurrences() > 0)
+ C = InstructionCost(ForceTargetInstructionCost);
+
+ BlockCost += C;
+ LLVM_DEBUG(dbgs() << "LV: Found an estimated cost of " << C << " for VF "
+ << VF << " For instruction: " << I << '\n');
+ }
+
+ // In the scalar loop, we may not always execute the predicated block, if it
+ // is an if-else block. Thus, scale the block's cost by the probability of
+ // executing it. getPredBlockCostDivisor will return 1 for blocks that are
+ // only predicated by the header mask when folding the tail.
+ Cost += BlockCost / getPredBlockCostDivisor(Config.CostKind, BB);
+ }
+
+ return Cost;
+}
+
/// Gets the address access SCEV for Ptr, if it should be used for cost modeling
/// according to isAddressSCEVForCost.
///
@@ -5559,6 +5608,9 @@ getRecordedExecutionFrequency(const VPBasicBlock *VPBB) {
InstructionCost LoopVectorizationPlanner::computeScalarCost() const {
ElementCount ScalarVF = ElementCount::getFixed(1);
+ if (!UseVPlanScalarCost)
+ return CM->expectedCost(ScalarVF);
+
VPCostContext CostCtx(*TLI, *InitialVPlan0, *CM, Config);
VPBasicBlock *Header =
VPBlockUtils::getPlainCFGHeaderAndLatch(*InitialVPlan0).first;
diff --git a/llvm/test/Transforms/LoopVectorize/X86/uniformshift.ll b/llvm/test/Transforms/LoopVectorize/X86/uniformshift.ll
index e38ba170eb6029..7cb076889bd366 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/uniformshift.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/uniformshift.ll
@@ -1,8 +1,11 @@
-; RUN: opt -mtriple=x86_64-apple-darwin -mattr=+sse2 -passes=loop-vectorize -debug-only=loop-vectorize -S < %s 2>&1 | FileCheck %s
+; RUN: opt -mtriple=x86_64-apple-darwin -mattr=+sse2 -passes=loop-vectorize -debug-only=loop-vectorize -S < %s 2>&1 | FileCheck --check-prefixes=CHECK,VPLAN %s
+; Check the scalar costs computed by the legacy cost model.
+; RUN: opt -mtriple=x86_64-apple-darwin -mattr=+sse2 -passes=loop-vectorize -debug-only=loop-vectorize -vplan-scalar-cost=false -S < %s 2>&1 | FileCheck --check-prefixes=CHECK,LEGACY %s
; REQUIRES: asserts
; CHECK: 'foo'
-; CHECK: Cost of 1 for VF 1: EMIT ir<%shift> = ashr ir<%val>, ir<%k>
+; VPLAN: Cost of 1 for VF 1: EMIT ir<%shift> = ashr ir<%val>, ir<%k>
+; LEGACY: LV: Found an estimated cost of 1 for VF 1 For instruction: %shift = ashr i32 %val, %k
; CHECK: Cost of 2 for VF 2: WIDEN ir<%shift> = ashr ir<%val>, ir<%k>
; CHECK: Cost of 2 for VF 4: WIDEN ir<%shift> = ashr ir<%val>, ir<%k>
define void @foo(ptr nocapture %p, i32 %k) {
>From 7cc3d0f5159d196280f3946af32b0a99eac48dd4 Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Wed, 23 Sep 2026 15:23:39 +0100
Subject: [PATCH 7/7] !fxiup aaddress missed comments
---
.../lib/Transforms/Vectorize/VPlanRecipes.cpp | 36 ++--
.../X86/CostModel/vpinstruction-cost.ll | 160 ++++++++++++++++++
2 files changed, 181 insertions(+), 15 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
index e1e55111ddd2a3..ccff7dee10925e 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
@@ -1581,9 +1581,13 @@ InstructionCost VPInstruction::computeCost(ElementCount VF,
case Instruction::ExtractValue:
case Instruction::FNeg:
case Instruction::Freeze:
- if (!VF.isScalar() || !getUnderlyingValue())
- return 0;
- return getCostForRecipeWithOpcode(getOpcode(), VF, Ctx);
+ if (VF.isScalar())
+ return getCostForRecipeWithOpcode(getOpcode(), VF, Ctx);
+ break;
+ case Instruction::Alloca:
+ assert(VF.isScalar() && "only scalar VF expected");
+ return Ctx.TTI.getArithmeticInstrCost(Instruction::Mul, getScalarType(),
+ Ctx.CostKind);
case Instruction::Load:
case Instruction::Store:
assert(VF.isScalar() && "only scalar VF expected");
@@ -1598,23 +1602,25 @@ InstructionCost VPInstruction::computeCost(ElementCount VF,
}
case VPInstruction::BranchOnCond:
case Instruction::PHI:
- if (!getUnderlyingValue())
- return 0;
- return Ctx.TTI.getCFInstrCost(getOpcode() == Instruction::PHI
- ? Instruction::PHI
- : Instruction::CondBr,
- Ctx.CostKind);
+ case Instruction::Switch:
+ if (VF.isScalar())
+ return Ctx.TTI.getCFInstrCost(getOpcode() == VPInstruction::BranchOnCond
+ ? Instruction::CondBr
+ : getOpcode(),
+ Ctx.CostKind);
+ break;
case VPInstruction::ExtractPenultimateElement:
if (VF == ElementCount::getScalable(1))
return InstructionCost::getInvalid();
- [[fallthrough]];
+ break;
default:
- // TODO: Compute cost other VPInstructions once the legacy cost model has
- // been retired.
- assert((VF.isScalar() || !getUnderlyingValue()) &&
- "unexpected VPInstruction with underlying value");
- return 0;
+ break;
}
+ // TODO: Compute cost other VPInstructions once the legacy cost model has
+ // been retired.
+ assert((VF.isScalar() || !getUnderlyingValue()) &&
+ "unexpected VPInstruction with underlying value");
+ return 0;
}
bool VPInstruction::isVectorToScalar() const {
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/vpinstruction-cost.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/vpinstruction-cost.ll
index 6247b3910a609a..f0960b5c0f4c84 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/vpinstruction-cost.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/vpinstruction-cost.ll
@@ -357,3 +357,163 @@ loop:
exit:
ret void
}
+
+define void @test_vpinstruction_alloca_cost(ptr noalias %dst) {
+; CHECK-LABEL: 'test_vpinstruction_alloca_cost'
+; CHECK: Cost of 0 for VF 1: EMIT-SCALAR ir<%iv> = phi [ ir<0>, vector.ph ], [ ir<%iv.next>, loop ]
+; CHECK: Cost of 1 for VF 1: EMIT ir<%a> = alloca ir<1>
+; CHECK: Cost of 0 for VF 1: EMIT ir<%g.dst> = getelementptr inbounds ir<%dst>, ir<%iv>
+; CHECK: Cost of 1 for VF 1: EMIT store ir<%a>, ir<%g.dst>
+; CHECK: Cost of 1 for VF 1: EMIT ir<%iv.next> = add nuw nsw ir<%iv>, ir<1>
+; CHECK: Cost of 1 for VF 1: EMIT ir<%ec> = icmp eq ir<%iv.next>, ir<32>
+; CHECK: Cost of 0 for VF 1: EMIT branch-on-cond ir<%ec>
+; CHECK: Cost of 0 for VF 2: vp<[[VP4:%[0-9]+]]> = SCALAR-STEPS vp<[[VP3:%[0-9]+]]>, ir<1>, vp<[[VP0:%[0-9]+]]>
+; CHECK: Cost of 1 for VF 2: REPLICATE ir<%a> = alloca ir<1>
+; CHECK: Cost of 0 for VF 2: CLONE ir<%g.dst> = getelementptr inbounds ir<%dst>, vp<[[VP4]]>
+; CHECK: Cost of 0 for VF 2: vp<[[VP5:%[0-9]+]]> = vector-pointer inbounds ptr, ir<%g.dst>, ir<1>
+; CHECK: Cost of 1 for VF 2: WIDEN store vp<[[VP5]]>, ir<%a>
+; CHECK: Cost of 0 for VF 2: EMIT vp<%index.next> = add nuw vp<[[VP3]]>, vp<[[VP1:%[0-9]+]]>
+; CHECK: Cost of 1 for VF 2: EMIT branch-on-count vp<%index.next>, vp<[[VP2:%[0-9]+]]>
+; CHECK: Cost of 0 for VF 2: vector loop backedge
+; CHECK: Cost of 1 for VF 2: canonical IV increment
+; CHECK: Cost of 0 for VF 2: EMIT-SCALAR vp<%bc.resume.val> = phi [ vp<[[VP2]]>, middle.block ], [ ir<0>, ir-bb<entry> ]
+; CHECK: Cost of 0 for VF 2: IR %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ] (extra operand: vp<%bc.resume.val> from scalar.ph)
+; CHECK: Cost of 0 for VF 2: IR %a = alloca i8, align 16
+; CHECK: Cost of 0 for VF 2: IR %g.dst = getelementptr inbounds ptr, ptr %dst, i64 %iv
+; CHECK: Cost of 0 for VF 2: IR store ptr %a, ptr %g.dst, align 8
+; CHECK: Cost of 0 for VF 2: IR %iv.next = add nuw nsw i64 %iv, 1
+; CHECK: Cost of 0 for VF 2: IR %ec = icmp eq i64 %iv.next, 32
+; CHECK: Cost of 1 for VF 2: EMIT vp<%cmp.n> = icmp eq ir<32>, vp<[[VP2]]>
+; CHECK: Cost of 0 for VF 2: EMIT branch-on-cond vp<%cmp.n>
+; CHECK: Cost of 0 for VF 4: vp<[[VP4]]> = SCALAR-STEPS vp<[[VP3]]>, ir<1>, vp<[[VP0]]>
+; CHECK: Cost of 1 for VF 4: REPLICATE ir<%a> = alloca ir<1>
+; CHECK: Cost of 0 for VF 4: CLONE ir<%g.dst> = getelementptr inbounds ir<%dst>, vp<[[VP4]]>
+; CHECK: Cost of 0 for VF 4: vp<[[VP5]]> = vector-pointer inbounds ptr, ir<%g.dst>, ir<1>
+; CHECK: Cost of 1 for VF 4: WIDEN store vp<[[VP5]]>, ir<%a>
+; CHECK: Cost of 0 for VF 4: EMIT vp<%index.next> = add nuw vp<[[VP3]]>, vp<[[VP1]]>
+; CHECK: Cost of 1 for VF 4: EMIT branch-on-count vp<%index.next>, vp<[[VP2]]>
+; CHECK: Cost of 0 for VF 4: vector loop backedge
+; CHECK: Cost of 1 for VF 4: canonical IV increment
+; CHECK: Cost of 0 for VF 4: EMIT-SCALAR vp<%bc.resume.val> = phi [ vp<[[VP2]]>, middle.block ], [ ir<0>, ir-bb<entry> ]
+; CHECK: Cost of 0 for VF 4: IR %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ] (extra operand: vp<%bc.resume.val> from scalar.ph)
+; CHECK: Cost of 0 for VF 4: IR %a = alloca i8, align 16
+; CHECK: Cost of 0 for VF 4: IR %g.dst = getelementptr inbounds ptr, ptr %dst, i64 %iv
+; CHECK: Cost of 0 for VF 4: IR store ptr %a, ptr %g.dst, align 8
+; CHECK: Cost of 0 for VF 4: IR %iv.next = add nuw nsw i64 %iv, 1
+; CHECK: Cost of 0 for VF 4: IR %ec = icmp eq i64 %iv.next, 32
+; CHECK: Cost of 1 for VF 4: EMIT vp<%cmp.n> = icmp eq ir<32>, vp<[[VP2]]>
+; CHECK: Cost of 0 for VF 4: EMIT branch-on-cond vp<%cmp.n>
+; CHECK: Cost of 1 for VF 4: EMIT vp<%cmp.n> = icmp eq ir<32>, vp<[[VP2]]>
+; CHECK: Cost of 0 for VF 4: EMIT branch-on-cond vp<%cmp.n>
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %a = alloca i8, align 16
+ %g.dst = getelementptr inbounds ptr, ptr %dst, i64 %iv
+ store ptr %a, ptr %g.dst, align 8
+ %iv.next = add nuw nsw i64 %iv, 1
+ %ec = icmp eq i64 %iv.next, 32
+ br i1 %ec, label %exit, label %loop
+
+exit:
+ ret void
+}
+
+; Switch is free for TCK_RecipThroughput on X86, use minsize (TCK_CodeSize) to
+; expose its cost.
+define void @test_vpinstruction_switch_cost_minsize(ptr noalias %dst) minsize {
+; CHECK-LABEL: 'test_vpinstruction_switch_cost_minsize'
+; CHECK: Cost of 0 for VF 1: EMIT-SCALAR ir<%iv> = phi [ ir<0>, vector.ph ], [ ir<%iv.next>, loop.latch ]
+; CHECK: Cost of 0 for VF 1: EMIT ir<%g.dst> = getelementptr inbounds ir<%dst>, ir<%iv>
+; CHECK: Cost of 1 for VF 1: EMIT-SCALAR ir<%l> = load ir<%g.dst>
+; CHECK: Cost of 1 for VF 1: EMIT switch ir<%l>, ir<-12>, ir<13> (!vplan.prof.estimated estimated {715827883, 715827883, 715827883})
+; CHECK: Cost of 2 for VF 1: EMIT store ir<0>, ir<%g.dst> (!vplan.execution.frequency 3074457347049914368 (33.33%, estimated))
+; CHECK: Cost of 2 for VF 1: EMIT store ir<42>, ir<%g.dst> (!vplan.execution.frequency 3074457347049914368 (33.33%, estimated))
+; CHECK: Cost of 2 for VF 1: EMIT store ir<2>, ir<%g.dst> (!vplan.execution.frequency 3074457347049914368 (33.33%, estimated))
+; CHECK: Cost of 1 for VF 1: EMIT ir<%iv.next> = add nuw nsw ir<%iv>, ir<1>
+; CHECK: Cost of 1 for VF 1: EMIT ir<%ec> = icmp eq ir<%iv.next>, ir<32>
+; CHECK: Cost of 1 for VF 1: EMIT branch-on-cond ir<%ec>
+; CHECK: Cost of 0 for VF 2: vp<[[VP4:%[0-9]+]]> = SCALAR-STEPS vp<[[VP3:%[0-9]+]]>, ir<1>, vp<[[VP0:%[0-9]+]]>
+; CHECK: Cost of 0 for VF 2: CLONE ir<%g.dst> = getelementptr ir<%dst>, vp<[[VP4]]>
+; CHECK: Cost of 0 for VF 2: vp<[[VP5:%[0-9]+]]> = vector-pointer inbounds i64, ir<%g.dst>, ir<1>
+; CHECK: Cost of 1 for VF 2: WIDEN ir<%l> = load vp<[[VP5]]>
+; CHECK: Cost of 1 for VF 2: EMIT vp<[[VP6:%[0-9]+]]> = icmp eq ir<%l>, ir<-12>
+; CHECK: Cost of 1 for VF 2: EMIT vp<[[VP7:%[0-9]+]]> = icmp eq ir<%l>, ir<13>
+; CHECK: Cost of 0 for VF 2: EMIT vp<[[VP8:%[0-9]+]]> = or vp<[[VP6]]>, vp<[[VP7]]>
+; CHECK: Cost of 1 for VF 2: EMIT vp<[[VP9:%[0-9]+]]> = not vp<[[VP8]]>
+; CHECK: Cost of 0 for VF 2: vp<[[VP10:%[0-9]+]]> = vector-pointer i64, ir<%g.dst>, ir<1>
+; CHECK: Cost of 1 for VF 2: WIDEN store vp<[[VP10]]>, ir<0>, vp<[[VP7]]> (!vplan.execution.frequency 3074457347049914368 (33.33%, estimated))
+; CHECK: Cost of 0 for VF 2: vp<[[VP11:%[0-9]+]]> = vector-pointer i64, ir<%g.dst>, ir<1>
+; CHECK: Cost of 1 for VF 2: WIDEN store vp<[[VP11]]>, ir<42>, vp<[[VP6]]> (!vplan.execution.frequency 3074457347049914368 (33.33%, estimated))
+; CHECK: Cost of 0 for VF 2: vp<[[VP12:%[0-9]+]]> = vector-pointer i64, ir<%g.dst>, ir<1>
+; CHECK: Cost of 1 for VF 2: WIDEN store vp<[[VP12]]>, ir<2>, vp<[[VP9]]> (!vplan.execution.frequency 3074457347049914368 (33.33%, estimated))
+; CHECK: Cost of 0 for VF 2: EMIT vp<%index.next> = add nuw vp<[[VP3]]>, vp<[[VP1:%[0-9]+]]>
+; CHECK: Cost of 1 for VF 2: EMIT branch-on-count vp<%index.next>, vp<[[VP2:%[0-9]+]]>
+; CHECK: Cost of 1 for VF 2: vector loop backedge
+; CHECK: Cost of 1 for VF 2: canonical IV increment
+; CHECK: Cost of 0 for VF 2: EMIT-SCALAR vp<%bc.resume.val> = phi [ vp<[[VP2]]>, middle.block ], [ ir<0>, ir-bb<entry> ]
+; CHECK: Cost of 0 for VF 2: IR %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop.latch ] (extra operand: vp<%bc.resume.val> from scalar.ph)
+; CHECK: Cost of 0 for VF 2: IR %g.dst = getelementptr inbounds i64, ptr %dst, i64 %iv
+; CHECK: Cost of 0 for VF 2: IR %l = load i64, ptr %g.dst, align 8
+; CHECK: Cost of 1 for VF 2: EMIT vp<%cmp.n> = icmp eq ir<32>, vp<[[VP2]]>
+; CHECK: Cost of 0 for VF 2: EMIT branch-on-cond vp<%cmp.n>
+; CHECK: Cost of 0 for VF 4: vp<[[VP4]]> = SCALAR-STEPS vp<[[VP3]]>, ir<1>, vp<[[VP0]]>
+; CHECK: Cost of 0 for VF 4: CLONE ir<%g.dst> = getelementptr ir<%dst>, vp<[[VP4]]>
+; CHECK: Cost of 0 for VF 4: vp<[[VP5]]> = vector-pointer inbounds i64, ir<%g.dst>, ir<1>
+; CHECK: Cost of 1 for VF 4: WIDEN ir<%l> = load vp<[[VP5]]>
+; CHECK: Cost of 1 for VF 4: EMIT vp<[[VP6]]> = icmp eq ir<%l>, ir<-12>
+; CHECK: Cost of 1 for VF 4: EMIT vp<[[VP7]]> = icmp eq ir<%l>, ir<13>
+; CHECK: Cost of 0 for VF 4: EMIT vp<[[VP8]]> = or vp<[[VP6]]>, vp<[[VP7]]>
+; CHECK: Cost of 1 for VF 4: EMIT vp<[[VP9]]> = not vp<[[VP8]]>
+; CHECK: Cost of 0 for VF 4: vp<[[VP10]]> = vector-pointer i64, ir<%g.dst>, ir<1>
+; CHECK: Cost of 1 for VF 4: WIDEN store vp<[[VP10]]>, ir<0>, vp<[[VP7]]> (!vplan.execution.frequency 3074457347049914368 (33.33%, estimated))
+; CHECK: Cost of 0 for VF 4: vp<[[VP11]]> = vector-pointer i64, ir<%g.dst>, ir<1>
+; CHECK: Cost of 1 for VF 4: WIDEN store vp<[[VP11]]>, ir<42>, vp<[[VP6]]> (!vplan.execution.frequency 3074457347049914368 (33.33%, estimated))
+; CHECK: Cost of 0 for VF 4: vp<[[VP12]]> = vector-pointer i64, ir<%g.dst>, ir<1>
+; CHECK: Cost of 1 for VF 4: WIDEN store vp<[[VP12]]>, ir<2>, vp<[[VP9]]> (!vplan.execution.frequency 3074457347049914368 (33.33%, estimated))
+; CHECK: Cost of 0 for VF 4: EMIT vp<%index.next> = add nuw vp<[[VP3]]>, vp<[[VP1]]>
+; CHECK: Cost of 1 for VF 4: EMIT branch-on-count vp<%index.next>, vp<[[VP2]]>
+; CHECK: Cost of 1 for VF 4: vector loop backedge
+; CHECK: Cost of 1 for VF 4: canonical IV increment
+; CHECK: Cost of 0 for VF 4: EMIT-SCALAR vp<%bc.resume.val> = phi [ vp<[[VP2]]>, middle.block ], [ ir<0>, ir-bb<entry> ]
+; CHECK: Cost of 0 for VF 4: IR %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop.latch ] (extra operand: vp<%bc.resume.val> from scalar.ph)
+; CHECK: Cost of 0 for VF 4: IR %g.dst = getelementptr inbounds i64, ptr %dst, i64 %iv
+; CHECK: Cost of 0 for VF 4: IR %l = load i64, ptr %g.dst, align 8
+; CHECK: Cost of 1 for VF 4: EMIT vp<%cmp.n> = icmp eq ir<32>, vp<[[VP2]]>
+; CHECK: Cost of 0 for VF 4: EMIT branch-on-cond vp<%cmp.n>
+;
+entry:
+ br label %loop.header
+
+loop.header:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop.latch ]
+ %g.dst = getelementptr inbounds i64, ptr %dst, i64 %iv
+ %l = load i64, ptr %g.dst, align 8
+ switch i64 %l, label %default [
+ i64 -12, label %case1
+ i64 13, label %case2
+ ]
+
+case1:
+ store i64 42, ptr %g.dst, align 8
+ br label %loop.latch
+
+case2:
+ store i64 0, ptr %g.dst, align 8
+ br label %loop.latch
+
+default:
+ store i64 2, ptr %g.dst, align 8
+ br label %loop.latch
+
+loop.latch:
+ %iv.next = add nuw nsw i64 %iv, 1
+ %ec = icmp eq i64 %iv.next, 32
+ br i1 %ec, label %exit, label %loop.header
+
+exit:
+ ret void
+}
More information about the llvm-commits
mailing list