[llvm] [SLP]Initial support for non-power-of-2 vectorization (PR #151530)
Ryan Buchner via llvm-commits
llvm-commits at lists.llvm.org
Fri Apr 10 13:08:09 PDT 2026
================
@@ -10961,72 +10949,178 @@ static bool tryToFindDuplicates(SmallVectorImpl<Value *> &VL,
UniqueValues.emplace_back(V);
}
+ bool AreAllValuesNonConst = UniquePositions.size() == UniqueValues.size();
+
+ // Check if we need to schedule the scalars. If no, can keep original scalars
+ // and avoid extra shuffles.
+ bool RequireScheduling = S && S.getOpcode() != Instruction::PHI &&
+ !isVectorLikeInstWithConstOps(S.getMainOp()) &&
+ (S.areInstructionsWithCopyableElements() ||
+ !doesNotNeedToSchedule(UniqueValues));
+ // Drop tail poisons, if the values can be vectorized.
+ if (RequireScheduling) {
+ const auto EndIt =
+ find_if_not(make_range(UniqueValues.rbegin(), UniqueValues.rend()),
+ IsaPred<PoisonValue>);
+ assert(EndIt != UniqueValues.rend() && "Expected at least one non-poison.");
+ UniqueValues.erase(EndIt.base(), UniqueValues.end());
+ }
// Easy case: VL has unique values and a "natural" size
size_t NumUniqueScalarValues = UniqueValues.size();
- bool IsFullVectors = hasFullVectorsOrPowerOf2(
- TTI, getValueType(UniqueValues.front()), NumUniqueScalarValues);
- if (NumUniqueScalarValues == VL.size() &&
- (VectorizeNonPowerOf2 || IsFullVectors)) {
+ if (NumUniqueScalarValues == VL.size()) {
ReuseShuffleIndices.clear();
return true;
}
- // FIXME: Reshuffing scalars is not supported yet for non-power-of-2 ops.
- if ((UserTreeIdx.UserTE &&
- UserTreeIdx.UserTE->hasNonWholeRegisterOrNonPowerOf2Vec(TTI)) ||
- !hasFullVectorsOrPowerOf2(TTI, getValueType(VL.front()), VL.size())) {
- LLVM_DEBUG(dbgs() << "SLP: Reshuffling scalars not yet supported "
- "for nodes with padding.\n");
+ // For VL=4 with 3 unique values: keep originals. A <3 x T> vector is
+ // always widened to <4 x T> on hardware, so the packing just adds an
+ // extra expand shuffle. Does not apply to loads (a <3 x T> load is a
+ // single memory access) or PHIs (benefit from compact packing in loops).
+ constexpr unsigned SmallVecWidth = 4;
+ constexpr unsigned SmallVecUniqueThreshold = 3;
+ if (VL.size() == SmallVecWidth &&
+ NumUniqueScalarValues == SmallVecUniqueThreshold && !BuildGatherOnly &&
+ !(S && (S.getOpcode() == Instruction::Load ||
+ S.getOpcode() == Instruction::PHI))) {
+ // Keep originals with identity reuse — no packing, no extra shuffle.
ReuseShuffleIndices.clear();
- return false;
+ return true;
}
- LLVM_DEBUG(dbgs() << "SLP: Shuffle for reused scalars.\n");
- if (NumUniqueScalarValues <= 1 || !IsFullVectors ||
- (UniquePositions.size() == 1 && all_of(UniqueValues, [](Value *V) {
- return isa<UndefValue>(V) || !isConstant(V);
- }))) {
- if (TryPad && UniquePositions.size() > 1 && NumUniqueScalarValues > 1 &&
- S.getMainOp()->isSafeToRemove() &&
- (S.areInstructionsWithCopyableElements() ||
- all_of(UniqueValues, IsaPred<Instruction, PoisonValue>))) {
- // Find the number of elements, which forms full vectors.
- unsigned PWSz = getFullVectorNumberOfElements(
- TTI, UniqueValues.front()->getType(), UniqueValues.size());
- PWSz = std::min<unsigned>(PWSz, VL.size());
- if (PWSz == VL.size()) {
- // We ended up with the same size after removing duplicates and
- // upgrading the resulting vector size to a "nice size". Just keep
- // the initial VL then.
- ReuseShuffleIndices.clear();
- } else {
- // Pad unique values with poison to grow the vector to a "nice" size
- SmallVector<Value *> PaddedUniqueValues(UniqueValues.begin(),
- UniqueValues.end());
- PaddedUniqueValues.append(
- PWSz - UniqueValues.size(),
- PoisonValue::get(UniqueValues.front()->getType()));
- // Check that extended with poisons/copyable operations are still valid
- // for vectorization (div/rem are not allowed).
- if ((!S.areInstructionsWithCopyableElements() &&
- !getSameOpcode(PaddedUniqueValues, TLI).valid()) ||
- (S.areInstructionsWithCopyableElements() && S.isMulDivLikeOp() &&
- (S.getMainOp()->isIntDivRem() || S.getMainOp()->isFPDivRem() ||
- isa<CallInst>(S.getMainOp())))) {
- LLVM_DEBUG(dbgs() << "SLP: Scalar used twice in bundle.\n");
- ReuseShuffleIndices.clear();
- return false;
- }
- VL = std::move(PaddedUniqueValues);
- }
- return true;
+ // Checks if unique inserts + shuffle is more profitable than just inserts or
+ // vectorized values.
+ auto EstimatePackPlusShuffleVsInserts = [&]() {
+ // Single instruction/argument insert - no shuffle.
+ if (UniquePositions.size() == 1 &&
+ (NumUniqueScalarValues == 1 ||
+ all_of(UniqueValues, IsaPred<UndefValue, Instruction, Argument>)))
+ return std::make_pair(false, false);
+ // For large gathers with power-of-2 VL where packing would produce
+ // non-power-of-2, reject if most scalars are constants — the packing
+ // overhead (non-power-of-2 split + shuffles) outweighs the benefit.
+ constexpr unsigned MinVLForConstGatherCheck = 4;
+ if (BuildGatherOnly && VL.size() > MinVLForConstGatherCheck &&
+ has_single_bit(static_cast<unsigned>(VL.size())) &&
+ !has_single_bit(static_cast<unsigned>(NumUniqueScalarValues)) &&
+ UniquePositions.size() * 2 < NumUniqueScalarValues)
+ return std::make_pair(false, false);
+ // Check if the given list of loads can be effectively vectorized.
+ auto CheckLoads = [&](ArrayRef<Value *> VL, bool IncludeGather) {
+ assert(S && S.getOpcode() == Instruction::Load && "Expected load.");
+ BoUpSLP::OrdersType Order;
+ SmallVector<Value *> PointerOps;
+ BoUpSLP::StridedPtrInfo SPtrInfo;
+ BoUpSLP::LoadsState Res =
+ R.canVectorizeLoads(VL, S.getMainOp(), Order, PointerOps, SPtrInfo);
+ return (IncludeGather && Res == BoUpSLP::LoadsState::Gather) ||
+ Res == BoUpSLP::LoadsState::ScatterVectorize ||
+ Res == BoUpSLP::LoadsState::CompressVectorize;
+ };
+ bool IsRootOperand =
+ UserTreeIdx.UserTE && UserTreeIdx.UserTE->Idx == 0 && !BuildGatherOnly;
+ if (IsRootOperand) {
+ if (S && S.getOpcode() == Instruction::Load) {
+ bool UseOrig = (CheckLoads(UniqueValues, /*IncludeGather=*/true) &&
+ CheckLoads(VL, /*IncludeGather=*/false)) ||
+ ShuffleVectorInst::isIdentityMask(
+ ReuseShuffleIndices, ReuseShuffleIndices.size());
+ return std::make_pair(true, UseOrig);
+ }
+ return std::make_pair(true, !RequireScheduling);
+ }
+ APInt DemandedElts = APInt::getZero(VL.size());
+ for_each(enumerate(ReuseShuffleIndices), [&](const auto &P) {
+ if (P.value() != PoisonMaskElem &&
+ UniquePositions.contains(UniqueValues[P.value()]))
+ DemandedElts.setBit(P.index());
+ });
+ Type *ScalarTy = ::getValueType(UniqueValues.front());
+ auto *VecTy = getWidenedType(ScalarTy, VL.size());
+ auto *UniquesVecTy = getWidenedType(ScalarTy, NumUniqueScalarValues);
+ const unsigned NumParts = ::getNumberOfParts(TTI, VecTy);
+ const unsigned UniquesNumParts = ::getNumberOfParts(TTI, UniquesVecTy);
+ // No need to schedule scalars and only single register used? Use original
+ // scalars, do not pack.
+ if (!RequireScheduling) {
+ if (VL.size() / NumUniqueScalarValues == 1 &&
+ (NumParts <= 1 || UniquesNumParts >= NumParts))
+ return std::make_pair(true, true);
+ // For PHI operands, prefer packing with reuse shuffle — the PHI
+ // carries the vector through the loop cheaply.
+ if (S && S.getOpcode() == Instruction::PHI && NumUniqueScalarValues > 1 &&
+ UniquesNumParts <= NumParts)
+ return std::make_pair(true, false);
}
- LLVM_DEBUG(dbgs() << "SLP: Scalar used twice in bundle.\n");
- ReuseShuffleIndices.clear();
- return false;
+ constexpr TTI::TargetCostKind CostKind = TTI::TCK_RecipThroughput;
+ InstructionCost ReusesCost = ::getShuffleCost(
+ TTI, TTI::SK_PermuteSingleSrc, VecTy,
+ NumUniqueScalarValues > VL.size() / 2 ? ArrayRef<int>()
+ : ArrayRef(ReuseShuffleIndices),
+ CostKind, /*Index=*/0, UniquesVecTy);
+ if (S && !BuildGatherOnly &&
+ ((S.getOpcode() != Instruction::Load &&
+ NumUniqueScalarValues + 1 == VL.size()) ||
+ ReusesCost > NumParts * (VL.size() > SmallVecWidth ? 1 : 2)))
+ return std::make_pair(true, true);
+ // Check if unique loads more profitable than repeated loads.
+ if (S && S.getOpcode() == Instruction::Load) {
+ bool UniquesVectorized =
+ CheckLoads(UniqueValues, /*IncludeGather=*/false);
+ if (UniquesVectorized || CheckLoads(VL, /*IncludeGather=*/false))
+ return std::make_pair(true, !UniquesVectorized);
+ }
+ bool CanSkipBVCost =
+ (!BuildGatherOnly && !RequireScheduling) || R.hasSameNode(S, VL);
+ InstructionCost InsertsCost =
+ CanSkipBVCost
+ ? InstructionCost(TTI::TCC_Free)
+ : ::getScalarizationOverhead(TTI, ScalarTy, VecTy, DemandedElts,
+ /*Insert=*/true, /*Extract=*/false,
+ CostKind, AreAllValuesNonConst, VL);
+ APInt UniquesDemandedElts = APInt::getAllOnes(NumUniqueScalarValues);
+ for (unsigned Idx : seq<unsigned>(NumUniqueScalarValues))
+ if (isConstant(UniqueValues[Idx]))
+ UniquesDemandedElts.clearBit(Idx);
+ InstructionCost UniquesCost =
+ (!BuildGatherOnly || R.hasSameNode(S, UniqueValues))
+ ? InstructionCost(TTI::TCC_Free)
+ : ::getScalarizationOverhead(TTI, ScalarTy, UniquesVecTy,
+ UniquesDemandedElts, /*Insert=*/true,
+ /*Extract=*/false, CostKind,
+ AreAllValuesNonConst, UniqueValues);
+ UniquesCost += ReusesCost;
+ if (UniquesCost <= InsertsCost)
+ return std::make_pair(true, false);
+ InstructionCost CostDiff = UniquesCost - InsertsCost;
+ if (CostDiff < TTI::TCC_Expensive ||
+ (R.getTreeSize() == 0 && R.isReductionTree() &&
+ CostDiff == TTI::TCC_Expensive))
+ return std::make_pair(S && (!S.isAltShuffle() || !BuildGatherOnly),
+ false);
+ // Otherwise, use original values, if values do not require scheduling and
+ // pass still try to vectorize them.
+ bool KeepOriginal = !BuildGatherOnly && !RequireScheduling;
+ return std::make_pair(KeepOriginal, KeepOriginal);
+ };
+
+ const auto [DoPack, UseOriginal] = EstimatePackPlusShuffleVsInserts();
+
+ if (DoPack) {
----------------
bababuck wrote:
Might just be misunderstanding terminology, but `DoPack` seems like a confusing name choice to me since packing is only the case when we use unique's shuffles.
Above, the comment says:
```
// No need to schedule scalars and only single register used? Use original
// scalars, do not pack.
if (!RequireScheduling) {
if (VL.size() / NumUniqueScalarValues == 1 &&
(NumParts <= 1 || UniquesNumParts >= NumParts))
return std::make_pair(true, true);
```
In that case, `DoPack` is set to true even though the comment describes the case as not being packed.
https://github.com/llvm/llvm-project/pull/151530
More information about the llvm-commits
mailing list