[llvm] [SLP]Pass operand info and context to TTI cost queries (PR #224931)
Alexey Bataev via llvm-commits
llvm-commits at lists.llvm.org
Sun Sep 20 06:54:55 PDT 2026
https://github.com/alexey-bataev created https://github.com/llvm/llvm-project/pull/224931
Feed operand value info, the context instruction, cast context hints,
and the actual select condition predicate to the TTI cost queries, so
target discounts that depend on them (fused fmul/fma pricing, folded
read-modify-write, constant or uniform operands) apply to both scalar
and vector sides of the SLP cost model. Also make the reversed-store
query describe the stored value instead of the pointer, per the
getMemoryOpCost contract.
Assisted-by: Cursor
>From e11e18def061b20e53f35a58e199c7c8a0aef243 Mon Sep 17 00:00:00 2001
From: Alexey Bataev <a.bataev at outlook.com>
Date: Sun, 20 Sep 2026 06:54:43 -0700
Subject: [PATCH] =?UTF-8?q?[=F0=9D=98=80=F0=9D=97=BD=F0=9D=97=BF]=20initia?=
=?UTF-8?q?l=20version?=
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
Created using spr 1.3.7
---
.../Transforms/Vectorize/SLPVectorizer.cpp | 198 +++++++++++-------
.../SLPVectorizer/SLPMemoryUtils.cpp | 3 +-
.../AArch64/externally-used-copyables.ll | 13 +-
.../AMDGPU/elementwise-fma-operand1.ll | 129 ++++++++++--
...on-commutative-second-arg-only-copyable.ll | 10 +-
.../RISCV/reduced-copyable-element.ll | 6 +-
.../strided-loads-with-external-indices.ll | 12 +-
.../SystemZ/non-power-2-subvector-extract.ll | 36 +++-
.../SLPVectorizer/X86/phi-multi-same-nodes.ll | 5 +-
9 files changed, 290 insertions(+), 122 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 205311d10e76c..26075b1ef2a87 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -6391,11 +6391,11 @@ BoUpSLP::LoadsState BoUpSLP::canVectorizeLoads(
}
switch (LS) {
case LoadsState::Vectorize:
- VecLdCost +=
- TTI.getMemoryOpCost(Instruction::Load, SubVecTy, LI0->getAlign(),
- LI0->getPointerAddressSpace(), CostKind,
- TTI::OperandValueInfo()) +
- VectorGEPCost;
+ VecLdCost += TTI.getMemoryOpCost(
+ Instruction::Load, SubVecTy, LI0->getAlign(),
+ LI0->getPointerAddressSpace(), CostKind,
+ TTI::getOperandInfo(LI0->getPointerOperand())) +
+ VectorGEPCost;
break;
case LoadsState::StridedVectorize:
VecLdCost += TTI.getMemIntrinsicInstrCost(
@@ -9086,7 +9086,10 @@ static InstructionCost getVectorOpCost(Instruction *I, unsigned VF,
getVectorCallCosts(CI, VecTy, &TTI, &TLI, ArgTys, CostKind);
return std::min(IntrCost, LibCost);
}
- return TTI.getArithmeticInstrCost(I->getOpcode(), VecTy, CostKind);
+ SmallVector<const Value *, 2> Args{I->getOperand(0), I->getOperand(1)};
+ return TTI.getArithmeticInstrCost(
+ I->getOpcode(), VecTy, CostKind, TTI::getOperandInfo(I->getOperand(0)),
+ TTI::getOperandInfo(I->getOperand(1)), Args, I);
}
/// Packs a type's kind and scalar width into one key, so an opcode/intrinsic
@@ -10322,9 +10325,21 @@ bool BoUpSLP::canBuildSplitNode(ArrayRef<Value *> VL,
(LocalState.getMainOp()->isCast() && LocalState.getAltOp()->isCast()) ||
(LocalState.getMainOp()->isUnaryOp() &&
LocalState.getAltOp()->isUnaryOp())) {
+ SmallVector<Value *> Ops0, Ops1;
+ for (Value *V : VL) {
+ auto *I = dyn_cast<Instruction>(V);
+ if (!I)
+ continue;
+ Ops0.push_back(I->getOperand(0));
+ Ops1.push_back(I->getOperand(I->isBinaryOp() ? 1 : 0));
+ }
+ TTI::OperandValueInfo Op1Info = getOperandInfo(Ops0);
+ TTI::OperandValueInfo Op2Info = getOperandInfo(Ops1);
InstructionCost OriginalVecOpsCost =
- TTI->getArithmeticInstrCost(Opcode0, VecTy, CostKind) +
- TTI->getArithmeticInstrCost(Opcode1, VecTy, CostKind);
+ TTI->getArithmeticInstrCost(Opcode0, VecTy, CostKind, Op1Info, Op2Info,
+ {}, LocalState.getMainOp()) +
+ TTI->getArithmeticInstrCost(Opcode1, VecTy, CostKind, Op1Info, Op2Info,
+ {}, LocalState.getAltOp());
SmallVector<int> OriginalMask(VL.size(), PoisonMaskElem);
for (unsigned Idx : seq<unsigned>(VL.size())) {
if (isa<PoisonValue>(VL[Idx]))
@@ -10335,8 +10350,10 @@ bool BoUpSLP::canBuildSplitNode(ArrayRef<Value *> VL,
OriginalVecOpsCost + getShuffleCost(*TTI, TTI::SK_PermuteTwoSrc, VecTy,
CostKind, OriginalMask);
InstructionCost NewVecOpsCost =
- TTI->getArithmeticInstrCost(Opcode0, Op1VecTy, CostKind) +
- TTI->getArithmeticInstrCost(Opcode1, Op2VecTy, CostKind);
+ TTI->getArithmeticInstrCost(Opcode0, Op1VecTy, CostKind, Op1Info,
+ Op2Info, {}, LocalState.getMainOp()) +
+ TTI->getArithmeticInstrCost(Opcode1, Op2VecTy, CostKind, Op1Info,
+ Op2Info, {}, LocalState.getAltOp());
InstructionCost NewCost =
NewVecOpsCost + InsertCost +
(!VectorizableTree.empty() && getRootNode().hasState() &&
@@ -11138,7 +11155,11 @@ class InstructionsCompatibilityAnalysis {
case Instruction::FMul:
case Instruction::FSub:
case Instruction::FDiv:
- VectorCost = TTI.getArithmeticInstrCost(MainOpcode, VecTy, Kind);
+ VectorCost = TTI.getArithmeticInstrCost(
+ MainOpcode, VecTy, Kind,
+ TTI::commonOperandInfo(Operands[0][0], Operands[0][1]),
+ TTI::commonOperandInfo(Operands[1][0], Operands[1][1]), {},
+ S.getMainOp());
break;
default:
// Calls (min/max, fmuladd) return above before reaching this switch.
@@ -13688,12 +13709,13 @@ bool BoUpSLP::matchesShlZExt(const TreeEntry &TE, OrdersType &Order,
getCastContextHint(*getOperandEntry(LhsTE, /*Idx=*/0));
InstructionCost VecCost =
TTI->getArithmeticReductionCost(Instruction::Or, VecTy, FMF, CostKind) +
- TTI->getArithmeticInstrCost(Instruction::Shl, VecTy, CostKind,
- getOperandInfo(LhsTE->Scalars)) +
+ TTI->getArithmeticInstrCost(
+ Instruction::Shl, VecTy, CostKind, getOperandInfo(LhsTE->Scalars),
+ getOperandInfo(RhsTE->Scalars), {}, TE.getMainOp()) +
TTI->getCastInstrCost(
Instruction::ZExt, VecTy,
getWidenedType(SrcScalarTy, LhsTE->getVectorFactor()), CastCtx,
- CostKind);
+ CostKind, LhsTE->getMainOp());
InstructionCost BitcastCost = TTI->getCastInstrCost(
Instruction::BitCast, SrcType, SrcVecTy, CastCtx, CostKind);
if (!Order.empty()) {
@@ -13726,12 +13748,14 @@ bool BoUpSLP::matchesShlZExt(const TreeEntry &TE, OrdersType &Order,
IntrinsicCostAttributes CostAttrs(Intrinsic::bswap, SrcType, {SrcType});
InstructionCost BSwapCost =
TTI->getMemoryOpCost(Instruction::Load, SrcType, LI->getAlign(),
- LI->getPointerAddressSpace(), CostKind) +
+ LI->getPointerAddressSpace(), CostKind,
+ TTI::getOperandInfo(LI->getPointerOperand())) +
TTI->getIntrinsicInstrCost(CostAttrs, CostKind);
if (BSwapCost <= BitcastCost) {
- VecCost +=
- TTI->getMemoryOpCost(Instruction::Load, SrcVecTy, LI->getAlign(),
- LI->getPointerAddressSpace(), CostKind);
+ VecCost += TTI->getMemoryOpCost(
+ Instruction::Load, SrcVecTy, LI->getAlign(),
+ LI->getPointerAddressSpace(), CostKind,
+ TTI::getOperandInfo(LI->getPointerOperand()));
BitcastCost = BSwapCost;
ForLoads = true;
}
@@ -13747,16 +13771,18 @@ bool BoUpSLP::matchesShlZExt(const TreeEntry &TE, OrdersType &Order,
auto *LI = cast<LoadInst>(SrcTE->getMainOp());
BitcastCost =
TTI->getMemoryOpCost(Instruction::Load, SrcType, LI->getAlign(),
- LI->getPointerAddressSpace(), CostKind);
+ LI->getPointerAddressSpace(), CostKind,
+ TTI::getOperandInfo(LI->getPointerOperand()));
VecCost +=
TTI->getMemoryOpCost(Instruction::Load, SrcVecTy, LI->getAlign(),
- LI->getPointerAddressSpace(), CostKind);
+ LI->getPointerAddressSpace(), CostKind,
+ TTI::getOperandInfo(LI->getPointerOperand()));
ForLoads = true;
}
}
if (SrcType != ScalarTy) {
BitcastCost += TTI->getCastInstrCost(Instruction::ZExt, ScalarTy, SrcType,
- TTI::CastContextHint::None, CostKind);
+ CastCtx, CostKind);
}
return BitcastCost < VecCost;
}
@@ -13941,10 +13967,10 @@ bool BoUpSLP::matchesInversedZExtSelect(
getWidenedType(Cmp->getOperand(0)->getType(), CmpTE->getVectorFactor());
Type *CmpTy = CmpInst::makeCmpResultType(VecTy);
- InstructionCost VecCost =
- TTI->getCmpSelInstrCost(CmpTE->getOpcode(), VecTy, CmpTy, MainPred,
- CostKind, getOperandInfo(CmpTE->getOperand(0)),
- getOperandInfo(CmpTE->getOperand(1)));
+ InstructionCost VecCost = TTI->getCmpSelInstrCost(
+ CmpTE->getOpcode(), VecTy, CmpTy, MainPred, CostKind,
+ getOperandInfo(CmpTE->getOperand(0)),
+ getOperandInfo(CmpTE->getOperand(1)), CmpTE->getMainOp());
InstructionCost BVCost = getScalarizationOverhead(
*TTI, SLPReVec, Cmp->getType(), cast<VectorType>(CmpTy),
APInt::getAllOnes(CmpTE->getVectorFactor()),
@@ -14014,12 +14040,25 @@ bool BoUpSLP::matchesSelectOfBits(const TreeEntry &SelectTE) const {
BitcastCost += TTI->getCastInstrCost(Instruction::ZExt, ScalarTy, DstTy,
TTI::CastContextHint::None, CostKind);
}
+ // Use the condition predicate shared by all lanes, if there is one.
+ CmpPredicate SelPred = CmpInst::BAD_ICMP_PREDICATE;
+ CmpPredicate P;
+ if (match(SelectTE.getMainOp(),
+ m_Select(m_ICmp(P, m_Value(), m_Value()), m_Value(), m_Value())) &&
+ all_of(SelectTE.Scalars, [&](Value *V) {
+ CmpPredicate LaneP;
+ return match(V, m_Select(m_ICmp(LaneP, m_Value(), m_Value()), m_Value(),
+ m_Value())) &&
+ LaneP == P.dropSameSign();
+ }))
+ SelPred = P;
+
FastMathFlags FMF;
InstructionCost SelectCost =
- TTI->getCmpSelInstrCost(Instruction::Select, VecTy, CmpTy,
- CmpInst::BAD_ICMP_PREDICATE, CostKind,
- getOperandInfo(Op1TE->Scalars),
- getOperandInfo(Op2TE->Scalars)) +
+ TTI->getCmpSelInstrCost(Instruction::Select, VecTy, CmpTy, SelPred,
+ CostKind, getOperandInfo(Op1TE->Scalars),
+ getOperandInfo(Op2TE->Scalars),
+ SelectTE.getMainOp()) +
TTI->getArithmeticReductionCost(Instruction::Or, VecTy, FMF, CostKind);
return BitcastCost <= SelectCost;
}
@@ -14330,9 +14369,10 @@ void BoUpSLP::transformNodes() {
inversePermutation(E.ReorderIndices, Mask);
auto *BaseLI = cast<LoadInst>(E.Scalars.back());
InstructionCost OriginalVecCost =
- TTI->getMemoryOpCost(Instruction::Load, VecTy, BaseLI->getAlign(),
- BaseLI->getPointerAddressSpace(), CostKind,
- TTI::OperandValueInfo()) +
+ TTI->getMemoryOpCost(
+ Instruction::Load, VecTy, BaseLI->getAlign(),
+ BaseLI->getPointerAddressSpace(), CostKind,
+ TTI::getOperandInfo(BaseLI->getPointerOperand())) +
getShuffleCost(*TTI, TTI::SK_Reverse, VecTy, CostKind, Mask);
InstructionCost StridedCost = TTI->getMemIntrinsicInstrCost(
MemIntrinsicCostAttributes(Intrinsic::experimental_vp_strided_load,
@@ -14373,7 +14413,7 @@ void BoUpSLP::transformNodes() {
InstructionCost OriginalVecCost =
TTI->getMemoryOpCost(Instruction::Store, VecTy, BaseSI->getAlign(),
BaseSI->getPointerAddressSpace(), CostKind,
- TTI::OperandValueInfo()) +
+ getOperandInfo(E.getOperand(0))) +
getShuffleCost(*TTI, TTI::SK_Reverse, VecTy, CostKind, Mask);
InstructionCost StridedCost = TTI->getMemIntrinsicInstrCost(
MemIntrinsicCostAttributes(Intrinsic::experimental_vp_strided_store,
@@ -15043,7 +15083,7 @@ class BoUpSLP::ShuffleCostEstimator : public BaseShuffleAnalysis {
CastOpcode = IsSigned ? Instruction::SExt : Instruction::ZExt;
return TTI.getCastInstrCost(CastOpcode, getWidenedType(ScalarTy, VF),
getWidenedType(EScalarTy, VF),
- TTI::CastContextHint::None, CostKind);
+ R.getCastContextHint(E), CostKind);
}
return TTI::TCC_Free;
};
@@ -15059,9 +15099,12 @@ class BoUpSLP::ShuffleCostEstimator : public BaseShuffleAnalysis {
unsigned SrcSz = R.DL->getTypeSizeInBits(EScalarTy);
if (DstSz > SrcSz)
CastOpcode = IsSigned ? Instruction::SExt : Instruction::ZExt;
+ TTI::CastContextHint CCH = TTI::CastContextHint::None;
+ if (ArrayRef<TreeEntry *> TEs = R.getTreeEntries(V); TEs.size() == 1)
+ CCH = R.getCastContextHint(*TEs.front());
return TTI.getCastInstrCost(
CastOpcode, VectorType::get(ScalarTy, VecTy->getElementCount()),
- VecTy, TTI::CastContextHint::None, CostKind);
+ VecTy, CCH, CostKind);
}
return TTI::TCC_Free;
};
@@ -16212,7 +16255,7 @@ BoUpSLP::getEntryCost(const TreeEntry *E, ArrayRef<Value *> VectorizedVals,
TTI::OperandValueInfo Op1Info = TTI::getOperandInfo(I->getOperand(0));
TTI::OperandValueInfo Op2Info = TTI::getOperandInfo(I->getOperand(1));
PeeledScalarCost += TTI->getArithmeticInstrCost(
- I->getOpcode(), OrigScalarTy, CostKind, Op1Info, Op2Info);
+ I->getOpcode(), OrigScalarTy, CostKind, Op1Info, Op2Info, {}, I);
}
bool PeeledCostAdded = false;
InstructionCost CostDiff = GetCostDiff(
@@ -16273,8 +16316,8 @@ BoUpSLP::getEntryCost(const TreeEntry *E, ArrayRef<Value *> VectorizedVals,
auto *CI = cast<CmpInst>(VI->getOperand(0));
IntrinsicCost -= TTI->getCmpSelInstrCost(
CI->getOpcode(), Ty, Builder.getInt1Ty(), CI->getPredicate(),
- CostKind, {TTI::OK_AnyValue, TTI::OP_None},
- {TTI::OK_AnyValue, TTI::OP_None}, CI);
+ CostKind, TTI::getOperandInfo(CI->getOperand(0)),
+ TTI::getOperandInfo(CI->getOperand(1)), CI);
}
return IntrinsicCost;
};
@@ -16505,6 +16548,8 @@ BoUpSLP::getEntryCost(const TreeEntry *E, ArrayRef<Value *> VectorizedVals,
// vector and the original aggregate.
Align VecAlign = std::max(DL->getPrefTypeAlign(SrcVecTy),
DL->getPrefTypeAlign(VL0->getType()));
+ // The stack slot roundtrip is synthesized by codegen; there is no IR
+ // value or pointer to take operand info from.
Cost += TTI->getMemoryOpCost(Instruction::Store, SrcVecTy, VecAlign,
/*AddressSpace=*/0, CostKind) +
TTI->getMemoryOpCost(Instruction::Load, VL0->getType(), VecAlign,
@@ -16805,9 +16850,8 @@ BoUpSLP::getEntryCost(const TreeEntry *E, ArrayRef<Value *> VectorizedVals,
TTI.getIntrinsicInstrCost(CostAttrs, CostKind);
BitcastCost += IntrinsicCost;
if (SrcType != ScalarTy) {
- BitcastCost +=
- TTI.getCastInstrCost(Instruction::ZExt, ScalarTy, SrcType,
- TTI::CastContextHint::None, CostKind);
+ BitcastCost += TTI.getCastInstrCost(Instruction::ZExt, ScalarTy,
+ SrcType, CastCtx, CostKind);
}
}
return BitcastCost + CommonCost;
@@ -16851,7 +16895,7 @@ BoUpSLP::getEntryCost(const TreeEntry *E, ArrayRef<Value *> VectorizedVals,
if (SrcType != ScalarTy) {
LoadCost +=
TTI.getCastInstrCost(Instruction::ZExt, ScalarTy, SrcType,
- TTI::CastContextHint::None, CostKind);
+ getCastContextHint(*LoadTE), CostKind);
}
}
return LoadCost + CommonCost;
@@ -16966,10 +17010,11 @@ BoUpSLP::getEntryCost(const TreeEntry *E, ArrayRef<Value *> VectorizedVals,
}
TTI::OperandValueInfo Op1Info = TTI::getOperandInfo(Op1);
TTI::OperandValueInfo Op2Info = TTI::getOperandInfo(Op2);
+ auto *I = dyn_cast<Instruction>(UniqueValues[Idx]);
InstructionCost ScalarCost = TTI->getArithmeticInstrCost(
- ShuffleOrOp, OrigScalarTy, CostKind, Op1Info, Op2Info, Operands);
- if (auto *I = dyn_cast<Instruction>(UniqueValues[Idx]);
- I && (ShuffleOrOp == Instruction::FAdd ||
+ ShuffleOrOp, OrigScalarTy, CostKind, Op1Info, Op2Info, Operands,
+ I && I->getOpcode() == ShuffleOrOp ? I : nullptr);
+ if (I && (ShuffleOrOp == Instruction::FAdd ||
ShuffleOrOp == Instruction::FSub)) {
InstructionCost IntrinsicCost = GetFMulAddCost(E->getOperations(), I);
if (IntrinsicCost.isValid())
@@ -17000,7 +17045,8 @@ BoUpSLP::getEntryCost(const TreeEntry *E, ArrayRef<Value *> VectorizedVals,
TTI::OperandValueInfo Op1Info = getOperandInfo(E->getOperand(0));
TTI::OperandValueInfo Op2Info = getOperandInfo(E->getOperand(OpIdx));
InstructionCost Cost = TTI->getArithmeticInstrCost(
- ShuffleOrOp, VecTy, CostKind, Op1Info, Op2Info, {}, nullptr, TLI);
+ ShuffleOrOp, VecTy, CostKind, Op1Info, Op2Info, {},
+ VL0->getOpcode() == ShuffleOrOp ? VL0 : nullptr, TLI);
// N columns need N-1 vector combines; price extra columns
// conservatively, skipping identity-only columns (not combined by
// codegen).
@@ -17011,9 +17057,12 @@ BoUpSLP::getEntryCost(const TreeEntry *E, ArrayRef<Value *> VectorizedVals,
return isBinOpIdentityConstant(V, CombineOpcode);
}))
continue;
+ // The first operand is the running fold, a computed value with no
+ // static properties to model.
Cost += TTI->getArithmeticInstrCost(
ShuffleOrOp, VecTy, CostKind, {},
- getOperandInfo(E->getOperand(Idx)), {}, nullptr, TLI);
+ getOperandInfo(E->getOperand(Idx)), {},
+ VL0->getOpcode() == ShuffleOrOp ? VL0 : nullptr, TLI);
}
}
return Cost + CommonCost;
@@ -17026,9 +17075,10 @@ BoUpSLP::getEntryCost(const TreeEntry *E, ArrayRef<Value *> VectorizedVals,
case Instruction::Load: {
auto GetScalarCost = [&](unsigned Idx) {
auto *VI = cast<LoadInst>(UniqueValues[Idx]);
- return TTI->getMemoryOpCost(Instruction::Load, OrigScalarTy,
- VI->getAlign(), VI->getPointerAddressSpace(),
- CostKind, TTI::OperandValueInfo(), VI);
+ return TTI->getMemoryOpCost(
+ Instruction::Load, OrigScalarTy, VI->getAlign(),
+ VI->getPointerAddressSpace(), CostKind,
+ TTI::getOperandInfo(VI->getPointerOperand()), VI);
};
auto *LI0 = cast<LoadInst>(VL0);
auto GetVectorCost = [&](InstructionCost CommonCost) {
@@ -17043,7 +17093,8 @@ BoUpSLP::getEntryCost(const TreeEntry *E, ArrayRef<Value *> VectorizedVals,
} else {
VecLdCost = TTI->getMemoryOpCost(
Instruction::Load, VecTy, LI0->getAlign(),
- LI0->getPointerAddressSpace(), CostKind, TTI::OperandValueInfo());
+ LI0->getPointerAddressSpace(), CostKind,
+ TTI::getOperandInfo(LI0->getPointerOperand()));
}
break;
case TreeEntry::StridedVectorize: {
@@ -17101,7 +17152,8 @@ BoUpSLP::getEntryCost(const TreeEntry *E, ArrayRef<Value *> VectorizedVals,
} else {
VecLdCost = TTI->getMemoryOpCost(
Instruction::Load, LoadVecTy, CommonAlignment,
- LI0->getPointerAddressSpace(), CostKind, TTI::OperandValueInfo());
+ LI0->getPointerAddressSpace(), CostKind,
+ TTI::getOperandInfo(LI0->getPointerOperand()));
// TODO: include this cost into CommonCost.
VecLdCost += getShuffleCost(*TTI, TTI::SK_PermuteSingleSrc, LoadVecTy,
CostKind, CompressMask);
@@ -17311,9 +17363,10 @@ BoUpSLP::getEntryCost(const TreeEntry *E, ArrayRef<Value *> VectorizedVals,
for (unsigned Idx : seq<unsigned>(1, E->getNumOperands())) {
TTI::OperandValueInfo ColInfo = getOperandInfo(E->getOperand(Idx));
if (!RunningInfo.isConstant() || !ColInfo.isConstant())
- Cost += TTIRef.getArithmeticInstrCost(Opcode, VecTy, CostKind,
- RunningInfo, ColInfo, {},
- nullptr, TLI);
+ Cost += TTIRef.getArithmeticInstrCost(
+ Opcode, VecTy, CostKind, RunningInfo, ColInfo, {},
+ Opcode == E->getOpcode() ? E->getMainOp() : E->getAltOp(),
+ TLI);
TTI::OperandValueKind Kind = TTI::OK_AnyValue;
if (RunningInfo.isConstant() && ColInfo.isConstant())
Kind = RunningInfo.Kind == TTI::OK_UniformConstantValue &&
@@ -17329,15 +17382,15 @@ BoUpSLP::getEntryCost(const TreeEntry *E, ArrayRef<Value *> VectorizedVals,
VecCost = ChainCost(E->getOpcode()) + ChainCost(E->getAltOpcode());
} else if (auto *CI0 = dyn_cast<CmpInst>(VL0)) {
auto *MaskTy = getWidenedType(Builder.getInt1Ty(), VL.size());
- VecCost = TTIRef.getCmpSelInstrCost(
- E->getOpcode(), VecTy, MaskTy, CI0->getPredicate(), CostKind,
- {TTI::OK_AnyValue, TTI::OP_None}, {TTI::OK_AnyValue, TTI::OP_None},
- VL0);
+ TTI::OperandValueInfo Op1Info = getOperandInfo(E->getOperand(0));
+ TTI::OperandValueInfo Op2Info = getOperandInfo(E->getOperand(1));
+ VecCost = TTIRef.getCmpSelInstrCost(E->getOpcode(), VecTy, MaskTy,
+ CI0->getPredicate(), CostKind,
+ Op1Info, Op2Info, VL0);
VecCost += TTIRef.getCmpSelInstrCost(
E->getOpcode(), VecTy, MaskTy,
- cast<CmpInst>(E->getAltOp())->getPredicate(), CostKind,
- {TTI::OK_AnyValue, TTI::OP_None}, {TTI::OK_AnyValue, TTI::OP_None},
- E->getAltOp());
+ cast<CmpInst>(E->getAltOp())->getPredicate(), CostKind, Op1Info,
+ Op2Info, E->getAltOp());
} else {
Type *SrcSclTy = E->getMainOp()->getOperand(0)->getType();
auto *SrcTy = getWidenedType(SrcSclTy, VL.size());
@@ -17353,9 +17406,9 @@ BoUpSLP::getEntryCost(const TreeEntry *E, ArrayRef<Value *> VectorizedVals,
}
if (BWSz <= SrcBWSz) {
if (BWSz < SrcBWSz)
- VecCost =
- TTIRef.getCastInstrCost(Instruction::Trunc, VecTy, SrcTy,
- TTI::CastContextHint::None, CostKind);
+ VecCost = TTIRef.getCastInstrCost(
+ Instruction::Trunc, VecTy, SrcTy,
+ GetCastContextHint(E->getMainOp()->getOperand(0)), CostKind);
LLVM_DEBUG({
dbgs()
<< "SLP: alternate extension, which should be truncated.\n";
@@ -17364,11 +17417,14 @@ BoUpSLP::getEntryCost(const TreeEntry *E, ArrayRef<Value *> VectorizedVals,
return VecCost;
}
}
- VecCost = TTIRef.getCastInstrCost(E->getOpcode(), VecTy, SrcTy,
- TTI::CastContextHint::None, CostKind);
- VecCost +=
- TTIRef.getCastInstrCost(E->getAltOpcode(), VecTy, SrcTy,
- TTI::CastContextHint::None, CostKind);
+ VecCost = TTIRef.getCastInstrCost(
+ E->getOpcode(), VecTy, SrcTy,
+ GetCastContextHint(E->getMainOp()->getOperand(0)), CostKind,
+ E->getMainOp());
+ VecCost += TTIRef.getCastInstrCost(
+ E->getAltOpcode(), VecTy, SrcTy,
+ GetCastContextHint(E->getMainOp()->getOperand(0)), CostKind,
+ E->getAltOp());
}
SmallVector<int> Mask;
E->buildAltOpShuffleMask(
@@ -19842,7 +19898,7 @@ InstructionCost BoUpSLP::getTreeCost(InstructionCost TreeCost,
VecOpcode, FTy,
getWidenedType(IntegerType::get(FTy->getContext(), BWSz),
FTy->getNumElements()),
- TTI::CastContextHint::None, CostKind);
+ getCastContextHint(*ScalarTE), CostKind);
LLVM_DEBUG(dbgs() << "SLP: Adding cost " << C
<< " for extending externally used vector with "
"non-equal minimum bitwidth.\n");
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPMemoryUtils.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPMemoryUtils.cpp
index 382adb71cede2..8279c277a1f04 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPMemoryUtils.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPMemoryUtils.cpp
@@ -297,7 +297,8 @@ bool isMaskedLoadCompress(
} else {
LoadCost =
TTI.getMemoryOpCost(Instruction::Load, LoadVecTy, CommonAlignment,
- LI->getPointerAddressSpace(), CostKind);
+ LI->getPointerAddressSpace(), CostKind,
+ TTI::getOperandInfo(LI->getPointerOperand()));
}
if (IsStrided && !IsMasked && Order.empty()) {
// Check for potential segmented(interleaved) loads.
diff --git a/llvm/test/Transforms/SLPVectorizer/AArch64/externally-used-copyables.ll b/llvm/test/Transforms/SLPVectorizer/AArch64/externally-used-copyables.ll
index c98a4d1f2c182..ab1f6ce5aa544 100644
--- a/llvm/test/Transforms/SLPVectorizer/AArch64/externally-used-copyables.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AArch64/externally-used-copyables.ll
@@ -4,7 +4,7 @@
define void @test(i64 %0, i64 %1, i64 %2, i64 %3, i64 %.sroa.3341.0.copyload, i64 %.sroa.3308.0.copyload, i64 %.neg1, i64 %indvar3788, i64 %4, i64 %5, i64 %6, i64 %7, i64 %8) {
; CHECK-LABEL: define void @test(
; CHECK-SAME: i64 [[TMP0:%.*]], i64 [[TMP1:%.*]], i64 [[TMP2:%.*]], i64 [[TMP3:%.*]], i64 [[DOTSROA_3341_0_COPYLOAD:%.*]], i64 [[DOTSROA_3308_0_COPYLOAD:%.*]], i64 [[DOTNEG1:%.*]], i64 [[INDVAR3788:%.*]], i64 [[TMP4:%.*]], i64 [[TMP5:%.*]], i64 [[TMP6:%.*]], i64 [[TMP7:%.*]], i64 [[TMP8:%.*]]) #[[ATTR0:[0-9]+]] {
-; CHECK-NEXT: [[_LR_PH_PREHEADER:.*:]]
+; CHECK-NEXT: [[_LR_PH_PREHEADER:.*]]:
; CHECK-NEXT: [[TMP9:%.*]] = insertelement <2 x i64> poison, i64 [[TMP1]], i64 0
; CHECK-NEXT: [[TMP10:%.*]] = insertelement <2 x i64> [[TMP9]], i64 [[TMP0]], i64 1
; CHECK-NEXT: [[TMP13:%.*]] = mul <2 x i64> [[TMP10]], <i64 1, i64 24>
@@ -14,10 +14,9 @@ define void @test(i64 %0, i64 %1, i64 %2, i64 %3, i64 %.sroa.3341.0.copyload, i6
; CHECK-NEXT: [[TMP15:%.*]] = shufflevector <2 x i64> [[TMP14]], <2 x i64> <i64 -1, i64 poison>, <2 x i32> <i32 2, i32 0>
; CHECK-NEXT: [[TMP16:%.*]] = sub <2 x i64> [[TMP14]], [[TMP15]]
; CHECK-NEXT: [[TMP20:%.*]] = shl i64 [[TMP0]], 11
-; CHECK-NEXT: [[TMP35:%.*]] = insertelement <2 x i64> poison, i64 [[TMP0]], i64 0
-; CHECK-NEXT: [[TMP12:%.*]] = shufflevector <2 x i64> [[TMP35]], <2 x i64> poison, <2 x i32> zeroinitializer
-; CHECK-NEXT: [[TMP25:%.*]] = shl <2 x i64> [[TMP12]], <i64 11, i64 0>
; CHECK-NEXT: [[TMP21:%.*]] = sub i64 1, [[TMP20]]
+; CHECK-NEXT: [[TMP28:%.*]] = insertelement <2 x i64> poison, i64 [[TMP0]], i64 1
+; CHECK-NEXT: [[TMP25:%.*]] = insertelement <2 x i64> [[TMP28]], i64 [[TMP20]], i64 0
; CHECK-NEXT: [[TMP22:%.*]] = add <2 x i64> [[TMP25]], <i64 8, i64 1>
; CHECK-NEXT: [[TMP23:%.*]] = or <2 x i64> [[TMP25]], <i64 8, i64 1>
; CHECK-NEXT: [[TMP33:%.*]] = shufflevector <2 x i64> [[TMP22]], <2 x i64> [[TMP23]], <2 x i32> <i32 0, i32 3>
@@ -37,6 +36,8 @@ define void @test(i64 %0, i64 %1, i64 %2, i64 %3, i64 %.sroa.3341.0.copyload, i6
; CHECK-NEXT: [[TMP38:%.*]] = insertelement <4 x i64> [[TMP37]], i64 [[TMP8]], i64 1
; CHECK-NEXT: [[TMP39:%.*]] = insertelement <4 x i64> [[TMP38]], i64 [[DOTSROA_3308_0_COPYLOAD]], i64 2
; CHECK-NEXT: [[TMP40:%.*]] = insertelement <4 x i64> [[TMP39]], i64 [[TMP0]], i64 3
+; CHECK-NEXT: [[TMP64:%.*]] = insertelement <2 x i64> poison, i64 [[TMP0]], i64 0
+; CHECK-NEXT: [[TMP12:%.*]] = shufflevector <2 x i64> [[TMP64]], <2 x i64> poison, <2 x i32> zeroinitializer
; CHECK-NEXT: [[TMP41:%.*]] = shufflevector <2 x i64> [[TMP17]], <2 x i64> poison, <8 x i32> <i32 0, i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
; CHECK-NEXT: [[TMP63:%.*]] = insertelement <4 x i64> poison, i64 [[TMP0]], i64 1
; CHECK-NEXT: [[TMP43:%.*]] = mul i64 [[TMP0]], [[TMP0]]
@@ -46,8 +47,8 @@ define void @test(i64 %0, i64 %1, i64 %2, i64 %3, i64 %.sroa.3341.0.copyload, i6
; CHECK-NEXT: [[TMP47:%.*]] = shufflevector <6 x i64> [[TMP31]], <6 x i64> [[TMP46]], <6 x i32> <i32 0, i32 1, i32 2, i32 3, i32 6, i32 7>
; CHECK-NEXT: [[TMP48:%.*]] = shufflevector <2 x i64> [[TMP16]], <2 x i64> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
; CHECK-NEXT: br label %[[DOTLR_PH1977_US:.*]]
-; CHECK: [[_LR_PH1977_US:.*:]]
-; CHECK-NEXT: [[INDVAR37888:%.*]] = phi i64 [ 0, [[DOTLR_PH_PREHEADER:%.*]] ], [ 1, %[[DOTLR_PH1977_US]] ]
+; CHECK: [[DOTLR_PH1977_US]]:
+; CHECK-NEXT: [[INDVAR37888:%.*]] = phi i64 [ 0, %[[_LR_PH_PREHEADER]] ], [ 1, %[[DOTLR_PH1977_US]] ]
; CHECK-NEXT: [[TMP49:%.*]] = shufflevector <6 x i64> [[TMP47]], <6 x i64> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 5, i32 5>
; CHECK-NEXT: [[TMP50:%.*]] = mul <8 x i64> [[TMP30]], [[TMP49]]
; CHECK-NEXT: [[TMP51:%.*]] = or <8 x i64> [[TMP30]], [[TMP49]]
diff --git a/llvm/test/Transforms/SLPVectorizer/AMDGPU/elementwise-fma-operand1.ll b/llvm/test/Transforms/SLPVectorizer/AMDGPU/elementwise-fma-operand1.ll
index c4c97622ac726..05e5e5f29d51b 100644
--- a/llvm/test/Transforms/SLPVectorizer/AMDGPU/elementwise-fma-operand1.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AMDGPU/elementwise-fma-operand1.ll
@@ -8,7 +8,8 @@
; Elementwise d = c + a * b, where the fmul is operand 1 of the fadd. These
; targets halve the cost of a packed fmul, so SLP is tempted to vectorize and
-; break the scalar fma chain. The 14 runs sit at the cost boundary. The 12 runs
+; break the scalar fma chain. The scalar fmul is costed as free once the fma
+; fusion context is visible, so the 14 runs stay scalar. The 12 runs
; vectorize either way and guard against the fmuladd marking landing on the load
; at operand 0 after the fma detection picked the fmul at operand 1, which
; asserts. axpy4_mixed_reassoc carries reassoc on one lane only, so the whole
@@ -18,12 +19,42 @@ define void @axpy4_contract(ptr noalias %d, ptr noalias %a, ptr noalias %b, ptr
; CHECK-LABEL: define void @axpy4_contract(
; CHECK-SAME: ptr noalias [[D:%.*]], ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], ptr noalias [[C:%.*]]) #[[ATTR0:[0-9]+]] {
; CHECK-NEXT: [[ENTRY:.*:]]
-; CHECK-NEXT: [[TMP0:%.*]] = load <4 x float>, ptr [[C]], align 4
-; CHECK-NEXT: [[TMP1:%.*]] = load <4 x float>, ptr [[A]], align 4
-; CHECK-NEXT: [[TMP2:%.*]] = load <4 x float>, ptr [[B]], align 4
-; CHECK-NEXT: [[TMP3:%.*]] = fmul contract <4 x float> [[TMP1]], [[TMP2]]
-; CHECK-NEXT: [[TMP4:%.*]] = fadd contract <4 x float> [[TMP0]], [[TMP3]]
-; CHECK-NEXT: store <4 x float> [[TMP4]], ptr [[D]], align 4
+; CHECK-NEXT: [[C0:%.*]] = load float, ptr [[C]], align 4
+; CHECK-NEXT: [[A0:%.*]] = load float, ptr [[A]], align 4
+; CHECK-NEXT: [[B0:%.*]] = load float, ptr [[B]], align 4
+; CHECK-NEXT: [[M0:%.*]] = fmul contract float [[A0]], [[B0]]
+; CHECK-NEXT: [[R0:%.*]] = fadd contract float [[C0]], [[M0]]
+; CHECK-NEXT: store float [[R0]], ptr [[D]], align 4
+; CHECK-NEXT: [[CP1:%.*]] = getelementptr inbounds float, ptr [[C]], i64 1
+; CHECK-NEXT: [[C1:%.*]] = load float, ptr [[CP1]], align 4
+; CHECK-NEXT: [[AP1:%.*]] = getelementptr inbounds float, ptr [[A]], i64 1
+; CHECK-NEXT: [[A1:%.*]] = load float, ptr [[AP1]], align 4
+; CHECK-NEXT: [[BP1:%.*]] = getelementptr inbounds float, ptr [[B]], i64 1
+; CHECK-NEXT: [[B1:%.*]] = load float, ptr [[BP1]], align 4
+; CHECK-NEXT: [[M1:%.*]] = fmul contract float [[A1]], [[B1]]
+; CHECK-NEXT: [[R1:%.*]] = fadd contract float [[C1]], [[M1]]
+; CHECK-NEXT: [[DP1:%.*]] = getelementptr inbounds float, ptr [[D]], i64 1
+; CHECK-NEXT: store float [[R1]], ptr [[DP1]], align 4
+; CHECK-NEXT: [[CP2:%.*]] = getelementptr inbounds float, ptr [[C]], i64 2
+; CHECK-NEXT: [[C2:%.*]] = load float, ptr [[CP2]], align 4
+; CHECK-NEXT: [[AP2:%.*]] = getelementptr inbounds float, ptr [[A]], i64 2
+; CHECK-NEXT: [[A2:%.*]] = load float, ptr [[AP2]], align 4
+; CHECK-NEXT: [[BP2:%.*]] = getelementptr inbounds float, ptr [[B]], i64 2
+; CHECK-NEXT: [[B2:%.*]] = load float, ptr [[BP2]], align 4
+; CHECK-NEXT: [[M2:%.*]] = fmul contract float [[A2]], [[B2]]
+; CHECK-NEXT: [[R2:%.*]] = fadd contract float [[C2]], [[M2]]
+; CHECK-NEXT: [[DP2:%.*]] = getelementptr inbounds float, ptr [[D]], i64 2
+; CHECK-NEXT: store float [[R2]], ptr [[DP2]], align 4
+; CHECK-NEXT: [[CP3:%.*]] = getelementptr inbounds float, ptr [[C]], i64 3
+; CHECK-NEXT: [[C3:%.*]] = load float, ptr [[CP3]], align 4
+; CHECK-NEXT: [[AP3:%.*]] = getelementptr inbounds float, ptr [[A]], i64 3
+; CHECK-NEXT: [[A3:%.*]] = load float, ptr [[AP3]], align 4
+; CHECK-NEXT: [[BP3:%.*]] = getelementptr inbounds float, ptr [[B]], i64 3
+; CHECK-NEXT: [[B3:%.*]] = load float, ptr [[BP3]], align 4
+; CHECK-NEXT: [[M3:%.*]] = fmul contract float [[A3]], [[B3]]
+; CHECK-NEXT: [[R3:%.*]] = fadd contract float [[C3]], [[M3]]
+; CHECK-NEXT: [[DP3:%.*]] = getelementptr inbounds float, ptr [[D]], i64 3
+; CHECK-NEXT: store float [[R3]], ptr [[DP3]], align 4
; CHECK-NEXT: ret void
;
; THR12-LABEL: define void @axpy4_contract(
@@ -81,12 +112,42 @@ define void @axpy4_reassoc(ptr noalias %d, ptr noalias %a, ptr noalias %b, ptr n
; CHECK-LABEL: define void @axpy4_reassoc(
; CHECK-SAME: ptr noalias [[D:%.*]], ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], ptr noalias [[C:%.*]]) #[[ATTR0]] {
; CHECK-NEXT: [[ENTRY:.*:]]
-; CHECK-NEXT: [[TMP0:%.*]] = load <4 x float>, ptr [[C]], align 4
-; CHECK-NEXT: [[TMP1:%.*]] = load <4 x float>, ptr [[A]], align 4
-; CHECK-NEXT: [[TMP2:%.*]] = load <4 x float>, ptr [[B]], align 4
-; CHECK-NEXT: [[TMP3:%.*]] = fmul reassoc contract <4 x float> [[TMP1]], [[TMP2]]
-; CHECK-NEXT: [[TMP4:%.*]] = fadd reassoc contract <4 x float> [[TMP0]], [[TMP3]]
-; CHECK-NEXT: store <4 x float> [[TMP4]], ptr [[D]], align 4
+; CHECK-NEXT: [[C0:%.*]] = load float, ptr [[C]], align 4
+; CHECK-NEXT: [[A0:%.*]] = load float, ptr [[A]], align 4
+; CHECK-NEXT: [[B0:%.*]] = load float, ptr [[B]], align 4
+; CHECK-NEXT: [[M0:%.*]] = fmul reassoc contract float [[A0]], [[B0]]
+; CHECK-NEXT: [[R0:%.*]] = fadd reassoc contract float [[C0]], [[M0]]
+; CHECK-NEXT: store float [[R0]], ptr [[D]], align 4
+; CHECK-NEXT: [[CP1:%.*]] = getelementptr inbounds float, ptr [[C]], i64 1
+; CHECK-NEXT: [[C1:%.*]] = load float, ptr [[CP1]], align 4
+; CHECK-NEXT: [[AP1:%.*]] = getelementptr inbounds float, ptr [[A]], i64 1
+; CHECK-NEXT: [[A1:%.*]] = load float, ptr [[AP1]], align 4
+; CHECK-NEXT: [[BP1:%.*]] = getelementptr inbounds float, ptr [[B]], i64 1
+; CHECK-NEXT: [[B1:%.*]] = load float, ptr [[BP1]], align 4
+; CHECK-NEXT: [[M1:%.*]] = fmul reassoc contract float [[A1]], [[B1]]
+; CHECK-NEXT: [[R1:%.*]] = fadd reassoc contract float [[C1]], [[M1]]
+; CHECK-NEXT: [[DP1:%.*]] = getelementptr inbounds float, ptr [[D]], i64 1
+; CHECK-NEXT: store float [[R1]], ptr [[DP1]], align 4
+; CHECK-NEXT: [[CP2:%.*]] = getelementptr inbounds float, ptr [[C]], i64 2
+; CHECK-NEXT: [[C2:%.*]] = load float, ptr [[CP2]], align 4
+; CHECK-NEXT: [[AP2:%.*]] = getelementptr inbounds float, ptr [[A]], i64 2
+; CHECK-NEXT: [[A2:%.*]] = load float, ptr [[AP2]], align 4
+; CHECK-NEXT: [[BP2:%.*]] = getelementptr inbounds float, ptr [[B]], i64 2
+; CHECK-NEXT: [[B2:%.*]] = load float, ptr [[BP2]], align 4
+; CHECK-NEXT: [[M2:%.*]] = fmul reassoc contract float [[A2]], [[B2]]
+; CHECK-NEXT: [[R2:%.*]] = fadd reassoc contract float [[C2]], [[M2]]
+; CHECK-NEXT: [[DP2:%.*]] = getelementptr inbounds float, ptr [[D]], i64 2
+; CHECK-NEXT: store float [[R2]], ptr [[DP2]], align 4
+; CHECK-NEXT: [[CP3:%.*]] = getelementptr inbounds float, ptr [[C]], i64 3
+; CHECK-NEXT: [[C3:%.*]] = load float, ptr [[CP3]], align 4
+; CHECK-NEXT: [[AP3:%.*]] = getelementptr inbounds float, ptr [[A]], i64 3
+; CHECK-NEXT: [[A3:%.*]] = load float, ptr [[AP3]], align 4
+; CHECK-NEXT: [[BP3:%.*]] = getelementptr inbounds float, ptr [[B]], i64 3
+; CHECK-NEXT: [[B3:%.*]] = load float, ptr [[BP3]], align 4
+; CHECK-NEXT: [[M3:%.*]] = fmul reassoc contract float [[A3]], [[B3]]
+; CHECK-NEXT: [[R3:%.*]] = fadd reassoc contract float [[C3]], [[M3]]
+; CHECK-NEXT: [[DP3:%.*]] = getelementptr inbounds float, ptr [[D]], i64 3
+; CHECK-NEXT: store float [[R3]], ptr [[DP3]], align 4
; CHECK-NEXT: ret void
;
; THR12-LABEL: define void @axpy4_reassoc(
@@ -144,12 +205,42 @@ define void @axpy4_mixed_reassoc(ptr noalias %d, ptr noalias %a, ptr noalias %b,
; CHECK-LABEL: define void @axpy4_mixed_reassoc(
; CHECK-SAME: ptr noalias [[D:%.*]], ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], ptr noalias [[C:%.*]]) #[[ATTR0]] {
; CHECK-NEXT: [[ENTRY:.*:]]
-; CHECK-NEXT: [[TMP0:%.*]] = load <4 x float>, ptr [[C]], align 4
-; CHECK-NEXT: [[TMP1:%.*]] = load <4 x float>, ptr [[A]], align 4
-; CHECK-NEXT: [[TMP2:%.*]] = load <4 x float>, ptr [[B]], align 4
-; CHECK-NEXT: [[TMP3:%.*]] = fmul contract <4 x float> [[TMP1]], [[TMP2]]
-; CHECK-NEXT: [[TMP4:%.*]] = fadd contract <4 x float> [[TMP0]], [[TMP3]]
-; CHECK-NEXT: store <4 x float> [[TMP4]], ptr [[D]], align 4
+; CHECK-NEXT: [[C0:%.*]] = load float, ptr [[C]], align 4
+; CHECK-NEXT: [[A0:%.*]] = load float, ptr [[A]], align 4
+; CHECK-NEXT: [[B0:%.*]] = load float, ptr [[B]], align 4
+; CHECK-NEXT: [[M0:%.*]] = fmul contract float [[A0]], [[B0]]
+; CHECK-NEXT: [[R0:%.*]] = fadd reassoc contract float [[C0]], [[M0]]
+; CHECK-NEXT: store float [[R0]], ptr [[D]], align 4
+; CHECK-NEXT: [[CP1:%.*]] = getelementptr inbounds float, ptr [[C]], i64 1
+; CHECK-NEXT: [[C1:%.*]] = load float, ptr [[CP1]], align 4
+; CHECK-NEXT: [[AP1:%.*]] = getelementptr inbounds float, ptr [[A]], i64 1
+; CHECK-NEXT: [[A1:%.*]] = load float, ptr [[AP1]], align 4
+; CHECK-NEXT: [[BP1:%.*]] = getelementptr inbounds float, ptr [[B]], i64 1
+; CHECK-NEXT: [[B1:%.*]] = load float, ptr [[BP1]], align 4
+; CHECK-NEXT: [[M1:%.*]] = fmul contract float [[A1]], [[B1]]
+; CHECK-NEXT: [[R1:%.*]] = fadd contract float [[C1]], [[M1]]
+; CHECK-NEXT: [[DP1:%.*]] = getelementptr inbounds float, ptr [[D]], i64 1
+; CHECK-NEXT: store float [[R1]], ptr [[DP1]], align 4
+; CHECK-NEXT: [[CP2:%.*]] = getelementptr inbounds float, ptr [[C]], i64 2
+; CHECK-NEXT: [[C2:%.*]] = load float, ptr [[CP2]], align 4
+; CHECK-NEXT: [[AP2:%.*]] = getelementptr inbounds float, ptr [[A]], i64 2
+; CHECK-NEXT: [[A2:%.*]] = load float, ptr [[AP2]], align 4
+; CHECK-NEXT: [[BP2:%.*]] = getelementptr inbounds float, ptr [[B]], i64 2
+; CHECK-NEXT: [[B2:%.*]] = load float, ptr [[BP2]], align 4
+; CHECK-NEXT: [[M2:%.*]] = fmul contract float [[A2]], [[B2]]
+; CHECK-NEXT: [[R2:%.*]] = fadd contract float [[C2]], [[M2]]
+; CHECK-NEXT: [[DP2:%.*]] = getelementptr inbounds float, ptr [[D]], i64 2
+; CHECK-NEXT: store float [[R2]], ptr [[DP2]], align 4
+; CHECK-NEXT: [[CP3:%.*]] = getelementptr inbounds float, ptr [[C]], i64 3
+; CHECK-NEXT: [[C3:%.*]] = load float, ptr [[CP3]], align 4
+; CHECK-NEXT: [[AP3:%.*]] = getelementptr inbounds float, ptr [[A]], i64 3
+; CHECK-NEXT: [[A3:%.*]] = load float, ptr [[AP3]], align 4
+; CHECK-NEXT: [[BP3:%.*]] = getelementptr inbounds float, ptr [[B]], i64 3
+; CHECK-NEXT: [[B3:%.*]] = load float, ptr [[BP3]], align 4
+; CHECK-NEXT: [[M3:%.*]] = fmul contract float [[A3]], [[B3]]
+; CHECK-NEXT: [[R3:%.*]] = fadd contract float [[C3]], [[M3]]
+; CHECK-NEXT: [[DP3:%.*]] = getelementptr inbounds float, ptr [[D]], i64 3
+; CHECK-NEXT: store float [[R3]], ptr [[DP3]], align 4
; CHECK-NEXT: ret void
;
; THR12-LABEL: define void @axpy4_mixed_reassoc(
diff --git a/llvm/test/Transforms/SLPVectorizer/RISCV/non-commutative-second-arg-only-copyable.ll b/llvm/test/Transforms/SLPVectorizer/RISCV/non-commutative-second-arg-only-copyable.ll
index 442b86733fe96..dae264f6c96cd 100644
--- a/llvm/test/Transforms/SLPVectorizer/RISCV/non-commutative-second-arg-only-copyable.ll
+++ b/llvm/test/Transforms/SLPVectorizer/RISCV/non-commutative-second-arg-only-copyable.ll
@@ -6,11 +6,11 @@ define i32 @main(ptr %q, ptr %a, i8 %.pre) {
; CHECK-SAME: ptr [[Q:%.*]], ptr [[A:%.*]], i8 [[DOTPRE:%.*]]) #[[ATTR0:[0-9]+]] {
; CHECK-NEXT: [[ENTRY:.*:]]
; CHECK-NEXT: [[DOTPRE1:%.*]] = load i8, ptr [[Q]], align 1
-; CHECK-NEXT: [[TMP0:%.*]] = insertelement <2 x i8> poison, i8 [[DOTPRE]], i64 0
-; CHECK-NEXT: [[TMP1:%.*]] = insertelement <2 x i8> [[TMP0]], i8 [[DOTPRE1]], i64 1
-; CHECK-NEXT: [[TMP2:%.*]] = sext <2 x i8> [[TMP1]] to <2 x i32>
-; CHECK-NEXT: [[TMP3:%.*]] = add <2 x i32> [[TMP2]], <i32 0, i32 1>
-; CHECK-NEXT: [[TMP4:%.*]] = shufflevector <2 x i32> [[TMP2]], <2 x i32> <i32 poison, i32 1>, <2 x i32> <i32 0, i32 3>
+; CHECK-NEXT: [[CONV11_I:%.*]] = sext i8 [[DOTPRE]] to i32
+; CHECK-NEXT: [[TMP0:%.*]] = sext i8 [[DOTPRE1]] to i32
+; CHECK-NEXT: [[TMP1:%.*]] = add i32 [[TMP0]], 1
+; CHECK-NEXT: [[TMP4:%.*]] = insertelement <2 x i32> <i32 poison, i32 1>, i32 [[CONV11_I]], i64 0
+; CHECK-NEXT: [[TMP3:%.*]] = insertelement <2 x i32> [[TMP4]], i32 [[TMP1]], i64 1
; CHECK-NEXT: [[TMP5:%.*]] = shl <2 x i32> [[TMP4]], [[TMP3]]
; CHECK-NEXT: [[TMP6:%.*]] = trunc <2 x i32> [[TMP5]] to <2 x i16>
; CHECK-NEXT: store <2 x i16> [[TMP6]], ptr [[A]], align 2
diff --git a/llvm/test/Transforms/SLPVectorizer/RISCV/reduced-copyable-element.ll b/llvm/test/Transforms/SLPVectorizer/RISCV/reduced-copyable-element.ll
index 7c35c888fe901..066abb629e8ff 100644
--- a/llvm/test/Transforms/SLPVectorizer/RISCV/reduced-copyable-element.ll
+++ b/llvm/test/Transforms/SLPVectorizer/RISCV/reduced-copyable-element.ll
@@ -13,8 +13,10 @@ define i32 @main() {
; CHECK-NEXT: [[TMP3:%.*]] = sext <2 x i32> [[TMP2]] to <2 x i64>
; CHECK-NEXT: [[TMP4:%.*]] = call <2 x i64> @llvm.umin.v2i64(<2 x i64> [[TMP3]], <2 x i64> splat (i64 17179869184))
; CHECK-NEXT: [[TMP5:%.*]] = trunc <2 x i64> [[TMP4]] to <2 x i32>
-; CHECK-NEXT: [[TMP6:%.*]] = add <2 x i32> [[TMP5]], <i32 0, i32 1>
-; CHECK-NEXT: [[TMP7:%.*]] = call i32 @llvm.vector.reduce.or.v2i32(<2 x i32> [[TMP6]])
+; CHECK-NEXT: [[TMP6:%.*]] = extractelement <2 x i32> [[TMP5]], i64 1
+; CHECK-NEXT: [[TMP9:%.*]] = add i32 [[TMP6]], 1
+; CHECK-NEXT: [[TMP8:%.*]] = extractelement <2 x i32> [[TMP5]], i64 0
+; CHECK-NEXT: [[TMP7:%.*]] = or i32 [[TMP9]], [[TMP8]]
; CHECK-NEXT: ret i32 [[TMP7]]
;
entry:
diff --git a/llvm/test/Transforms/SLPVectorizer/RISCV/strided-loads-with-external-indices.ll b/llvm/test/Transforms/SLPVectorizer/RISCV/strided-loads-with-external-indices.ll
index b1d045b22a730..604be4b570998 100644
--- a/llvm/test/Transforms/SLPVectorizer/RISCV/strided-loads-with-external-indices.ll
+++ b/llvm/test/Transforms/SLPVectorizer/RISCV/strided-loads-with-external-indices.ll
@@ -10,10 +10,14 @@ define void @test() {
; CHECK-NEXT: [[SUB4_I_I65_US:%.*]] = or i64 0, 1
; CHECK-NEXT: br label [[BODY:%.*]]
; CHECK: body:
-; CHECK-NEXT: [[TMP0:%.*]] = call <2 x i32> @llvm.masked.gather.v2i32.v2p0(<2 x ptr> align 4 getelementptr ([[CLASS_A:%.*]], <2 x ptr> splat (ptr null), <2 x i64> <i64 0, i64 1>), <2 x i1> splat (i1 true), <2 x i32> poison)
-; CHECK-NEXT: [[TMP1:%.*]] = extractelement <2 x i32> [[TMP0]], i64 0
-; CHECK-NEXT: [[TMP2:%.*]] = extractelement <2 x i32> [[TMP0]], i64 1
-; CHECK-NEXT: [[CMP_I_I_I_I67_US:%.*]] = icmp slt i32 [[TMP1]], [[TMP2]]
+; CHECK-NEXT: [[ADD_I_I62_US:%.*]] = shl i64 0, 0
+; CHECK-NEXT: [[TMP0:%.*]] = insertelement <2 x i64> <i64 poison, i64 1>, i64 [[ADD_I_I62_US]], i64 0
+; CHECK-NEXT: [[TMP1:%.*]] = or <2 x i64> zeroinitializer, [[TMP0]]
+; CHECK-NEXT: [[TMP2:%.*]] = getelementptr [[CLASS_A:%.*]], <2 x ptr> splat (ptr null), <2 x i64> [[TMP1]]
+; CHECK-NEXT: [[TMP3:%.*]] = call <2 x i32> @llvm.masked.gather.v2i32.v2p0(<2 x ptr> align 4 [[TMP2]], <2 x i1> splat (i1 true), <2 x i32> poison)
+; CHECK-NEXT: [[TMP4:%.*]] = extractelement <2 x i32> [[TMP3]], i64 0
+; CHECK-NEXT: [[TMP5:%.*]] = extractelement <2 x i32> [[TMP3]], i64 1
+; CHECK-NEXT: [[CMP_I_I_I_I67_US:%.*]] = icmp slt i32 [[TMP4]], [[TMP5]]
; CHECK-NEXT: [[SPEC_SELECT_I_I68_US:%.*]] = select i1 false, i64 [[SUB4_I_I65_US]], i64 0
; CHECK-NEXT: br label [[BODY]]
;
diff --git a/llvm/test/Transforms/SLPVectorizer/SystemZ/non-power-2-subvector-extract.ll b/llvm/test/Transforms/SLPVectorizer/SystemZ/non-power-2-subvector-extract.ll
index bde56da998c0b..d6cd8b6b12d27 100644
--- a/llvm/test/Transforms/SLPVectorizer/SystemZ/non-power-2-subvector-extract.ll
+++ b/llvm/test/Transforms/SLPVectorizer/SystemZ/non-power-2-subvector-extract.ll
@@ -8,18 +8,34 @@ define void @p() {
; CHECK-LABEL: define void @p(
; CHECK-SAME: ) #[[ATTR0:[0-9]+]] {
; CHECK-NEXT: [[ENTRY:.*:]]
-; CHECK-NEXT: [[TMP0:%.*]] = load <4 x i32>, ptr getelementptr inbounds nuw (i8, ptr @c, i64 52), align 4
-; CHECK-NEXT: [[TMP1:%.*]] = extractelement <4 x i32> [[TMP0]], i64 0
-; CHECK-NEXT: [[TMP2:%.*]] = extractelement <4 x i32> [[TMP0]], i64 2
-; CHECK-NEXT: [[TMP3:%.*]] = xor <4 x i32> [[TMP0]], splat (i32 1)
-; CHECK-NEXT: store <4 x i32> [[TMP3]], ptr getelementptr inbounds nuw (i8, ptr @c, i64 52), align 4
-; CHECK-NEXT: [[TMP4:%.*]] = load <7 x i32>, ptr getelementptr inbounds nuw (i8, ptr @c, i64 200), align 4
-; CHECK-NEXT: [[TMP5:%.*]] = extractelement <7 x i32> [[TMP4]], i64 3
+; CHECK-NEXT: [[ARRAYIDX12_PROMOTED_5_I:%.*]] = load i32, ptr getelementptr inbounds nuw (i8, ptr @c, i64 200), align 4
+; CHECK-NEXT: [[CONV14_5_I:%.*]] = xor i32 [[ARRAYIDX12_PROMOTED_5_I]], 1
+; CHECK-NEXT: store i32 [[CONV14_5_I]], ptr getelementptr inbounds nuw (i8, ptr @c, i64 200), align 4
+; CHECK-NEXT: [[ARRAYIDX12_PROMOTED_5_I_1:%.*]] = load i32, ptr getelementptr inbounds nuw (i8, ptr @c, i64 204), align 4
+; CHECK-NEXT: [[CONV14_5_I_1:%.*]] = xor i32 [[ARRAYIDX12_PROMOTED_5_I_1]], 1
+; CHECK-NEXT: store i32 [[CONV14_5_I_1]], ptr getelementptr inbounds nuw (i8, ptr @c, i64 204), align 4
+; CHECK-NEXT: [[TMP1:%.*]] = load i32, ptr getelementptr inbounds nuw (i8, ptr @c, i64 52), align 4
+; CHECK-NEXT: [[CONV14_1_I_3:%.*]] = xor i32 [[TMP1]], 1
+; CHECK-NEXT: store i32 [[CONV14_1_I_3]], ptr getelementptr inbounds nuw (i8, ptr @c, i64 52), align 4
+; CHECK-NEXT: [[ARRAYIDX12_PROMOTED_1_I_4:%.*]] = load i32, ptr getelementptr inbounds nuw (i8, ptr @c, i64 56), align 4
+; CHECK-NEXT: [[CONV14_1_I_4:%.*]] = xor i32 [[ARRAYIDX12_PROMOTED_1_I_4]], 1
+; CHECK-NEXT: store i32 [[CONV14_1_I_4]], ptr getelementptr inbounds nuw (i8, ptr @c, i64 56), align 4
+; CHECK-NEXT: [[TMP2:%.*]] = load i32, ptr getelementptr inbounds nuw (i8, ptr @c, i64 60), align 4
+; CHECK-NEXT: [[CONV14_1_I_5:%.*]] = xor i32 [[TMP2]], 1
+; CHECK-NEXT: store i32 [[CONV14_1_I_5]], ptr getelementptr inbounds nuw (i8, ptr @c, i64 60), align 4
+; CHECK-NEXT: [[TMP0:%.*]] = load <4 x i32>, ptr getelementptr inbounds nuw (i8, ptr @c, i64 208), align 4
+; CHECK-NEXT: [[TMP5:%.*]] = extractelement <4 x i32> [[TMP0]], i64 1
; CHECK-NEXT: [[OR_1_5_I_3:%.*]] = or i32 [[TMP1]], [[TMP5]]
; CHECK-NEXT: store i32 [[OR_1_5_I_3]], ptr @j.0, align 4
-; CHECK-NEXT: [[TMP6:%.*]] = extractelement <7 x i32> [[TMP4]], i64 5
-; CHECK-NEXT: [[TMP7:%.*]] = xor <7 x i32> [[TMP4]], splat (i32 1)
-; CHECK-NEXT: store <7 x i32> [[TMP7]], ptr getelementptr inbounds nuw (i8, ptr @c, i64 200), align 4
+; CHECK-NEXT: [[TMP3:%.*]] = xor <4 x i32> [[TMP0]], splat (i32 1)
+; CHECK-NEXT: store <4 x i32> [[TMP3]], ptr getelementptr inbounds nuw (i8, ptr @c, i64 208), align 4
+; CHECK-NEXT: [[TMP6:%.*]] = extractelement <4 x i32> [[TMP0]], i64 3
+; CHECK-NEXT: [[ARRAYIDX12_PROMOTED_1_I_6:%.*]] = load i32, ptr getelementptr inbounds nuw (i8, ptr @c, i64 64), align 4
+; CHECK-NEXT: [[CONV14_1_I_6:%.*]] = xor i32 [[ARRAYIDX12_PROMOTED_1_I_6]], 1
+; CHECK-NEXT: store i32 [[CONV14_1_I_6]], ptr getelementptr inbounds nuw (i8, ptr @c, i64 64), align 4
+; CHECK-NEXT: [[ARRAYIDX12_PROMOTED_5_I_6:%.*]] = load i32, ptr getelementptr inbounds nuw (i8, ptr @c, i64 224), align 4
+; CHECK-NEXT: [[CONV14_5_I_6:%.*]] = xor i32 [[ARRAYIDX12_PROMOTED_5_I_6]], 1
+; CHECK-NEXT: store i32 [[CONV14_5_I_6]], ptr getelementptr inbounds nuw (i8, ptr @c, i64 224), align 4
; CHECK-NEXT: [[TMP8:%.*]] = load <4 x i32>, ptr getelementptr inbounds nuw (i8, ptr @c, i64 252), align 4
; CHECK-NEXT: [[TMP9:%.*]] = extractelement <4 x i32> [[TMP8]], i64 1
; CHECK-NEXT: [[TMP10:%.*]] = or i32 [[TMP9]], [[TMP2]]
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/phi-multi-same-nodes.ll b/llvm/test/Transforms/SLPVectorizer/X86/phi-multi-same-nodes.ll
index 681d91df53b91..0c73c985bf079 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/phi-multi-same-nodes.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/phi-multi-same-nodes.ll
@@ -19,9 +19,6 @@ define void @test() {
; CHECK-NEXT: i32 19, label %[[BB4]]
; CHECK-NEXT: ]
; CHECK: [[BB1]]:
-; CHECK-NEXT: [[TMP0:%.*]] = ashr <2 x i32> zeroinitializer, <i32 1, i32 0>
-; CHECK-NEXT: [[TMP1:%.*]] = or <2 x i32> zeroinitializer, <i32 1, i32 0>
-; CHECK-NEXT: [[TMP2:%.*]] = shufflevector <2 x i32> [[TMP0]], <2 x i32> [[TMP1]], <2 x i32> <i32 0, i32 3>
; CHECK-NEXT: switch i32 0, label %[[BB4]] [
; CHECK-NEXT: i32 -4, label %[[BB4]]
; CHECK-NEXT: i32 -1, label %[[BB4]]
@@ -37,7 +34,7 @@ define void @test() {
; CHECK-NEXT: i32 19, label %[[BB4]]
; CHECK-NEXT: ]
; CHECK: [[BB4]]:
-; CHECK-NEXT: [[TMP3:%.*]] = phi <2 x i32> [ [[TMP2]], %[[BB1]] ], [ [[TMP2]], %[[BB1]] ], [ [[TMP2]], %[[BB1]] ], [ [[TMP2]], %[[BB1]] ], [ [[TMP2]], %[[BB1]] ], [ [[TMP2]], %[[BB1]] ], [ [[TMP2]], %[[BB1]] ], [ [[TMP2]], %[[BB1]] ], [ [[TMP2]], %[[BB1]] ], [ [[TMP2]], %[[BB1]] ], [ [[TMP2]], %[[BB1]] ], [ zeroinitializer, %[[BB]] ], [ zeroinitializer, %[[BB]] ], [ zeroinitializer, %[[BB]] ], [ zeroinitializer, %[[BB]] ], [ zeroinitializer, %[[BB]] ], [ zeroinitializer, %[[BB]] ], [ zeroinitializer, %[[BB]] ], [ zeroinitializer, %[[BB]] ], [ zeroinitializer, %[[BB]] ], [ zeroinitializer, %[[BB]] ], [ zeroinitializer, %[[BB]] ], [ zeroinitializer, %[[BB]] ], [ [[TMP2]], %[[BB1]] ], [ [[TMP2]], %[[BB1]] ]
+; CHECK-NEXT: [[TMP0:%.*]] = phi <2 x i32> [ zeroinitializer, %[[BB1]] ], [ zeroinitializer, %[[BB1]] ], [ zeroinitializer, %[[BB1]] ], [ zeroinitializer, %[[BB1]] ], [ zeroinitializer, %[[BB1]] ], [ zeroinitializer, %[[BB1]] ], [ zeroinitializer, %[[BB1]] ], [ zeroinitializer, %[[BB1]] ], [ zeroinitializer, %[[BB1]] ], [ zeroinitializer, %[[BB1]] ], [ zeroinitializer, %[[BB1]] ], [ zeroinitializer, %[[BB]] ], [ zeroinitializer, %[[BB]] ], [ zeroinitializer, %[[BB]] ], [ zeroinitializer, %[[BB]] ], [ zeroinitializer, %[[BB]] ], [ zeroinitializer, %[[BB]] ], [ zeroinitializer, %[[BB]] ], [ zeroinitializer, %[[BB]] ], [ zeroinitializer, %[[BB]] ], [ zeroinitializer, %[[BB]] ], [ zeroinitializer, %[[BB]] ], [ zeroinitializer, %[[BB]] ], [ zeroinitializer, %[[BB1]] ], [ zeroinitializer, %[[BB1]] ]
; CHECK-NEXT: ret void
;
bb:
More information about the llvm-commits
mailing list