[llvm] [SLP]Flatten alternate associative chains into one reassociated node (PR #215098)
Alexey Bataev via llvm-commits
llvm-commits at lists.llvm.org
Sun Aug 9 07:01:36 PDT 2026
https://github.com/alexey-bataev created https://github.com/llvm/llvm-project/pull/215098
Each lane peels only chain links with its own opcode, keeping the
per-lane opcode on every combine level; emission linearizes into
main-opcode chain, alt-opcode chain, and a single select shuffle.
>From 6b7830c167187cd832458d8030f035ee7bc3fde1 Mon Sep 17 00:00:00 2001
From: Alexey Bataev <a.bataev at outlook.com>
Date: Sun, 9 Aug 2026 07:01:24 -0700
Subject: [PATCH] =?UTF-8?q?[=F0=9D=98=80=F0=9D=97=BD=F0=9D=97=BF]=20initia?=
=?UTF-8?q?l=20version?=
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
Created using spr 1.3.7
---
.../Transforms/Vectorize/SLPVectorizer.cpp | 308 +++++++++++++-----
.../SLPCompatibilityAnalysis.cpp | 59 ++++
.../SLPVectorizer/SLPCompatibilityAnalysis.h | 13 +
.../X86/BinOpSameOpcodeHelper.ll | 12 +-
.../Transforms/SLPVectorizer/X86/supernode.ll | 11 +-
.../vectorize-reorder-alt-shuffle.ll | 23 +-
6 files changed, 302 insertions(+), 124 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index a04877d183516..cd26abc89fcf9 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -12347,11 +12347,13 @@ static void scanAssociativeOperands(
/// family of their first operand where available, so e.g. shifts fed by the
/// same load family land in one column instead of pairing by encounter order.
/// Values move between columns only within the same sign: a subtracted leaf
-/// never lands in an added column.
-static SmallVector<BoUpSLP::ValueList>
-alignReassociatedOperandsByKey(ArrayRef<BoUpSLP::ValueList> Operands,
- const SmallBitVector &NegatedColumns,
- const TargetLibraryInfo &TLI) {
+/// never lands in an added column. The sign is queried per lane and column
+/// with \p IsNegated: alternate add/sub nodes negate only the non-leading
+/// columns of their subtract lanes.
+static SmallVector<BoUpSLP::ValueList> alignReassociatedOperandsByKey(
+ ArrayRef<BoUpSLP::ValueList> Operands,
+ function_ref<bool(unsigned Lane, unsigned Col)> IsNegated,
+ const TargetLibraryInfo &TLI) {
const unsigned NumCols = Operands.size();
const unsigned NumLanes = Operands.front().size();
auto LoadsSubkey = [](size_t /*Key*/, LoadInst *LI) {
@@ -12384,7 +12386,7 @@ alignReassociatedOperandsByKey(ArrayRef<BoUpSLP::ValueList> Operands,
for (unsigned Lane : seq<unsigned>(1, NumLanes)) {
// Buckets are keyed by the value key and the column sign.
using Key = std::pair<std::pair<size_t, size_t>, unsigned>;
- auto Sign = [&](unsigned Col) { return NegatedColumns[Col] ? 1U : 0U; };
+ auto Sign = [&](unsigned Col) { return IsNegated(Lane, Col) ? 1U : 0U; };
SmallDenseMap<Key, SmallVector<unsigned, 2>, 8> Buckets;
for (unsigned Col : seq<unsigned>(NumCols))
Buckets[{GetKey(Operands[Col][Lane]), Sign(Col)}].push_back(Col);
@@ -12420,7 +12422,7 @@ alignReassociatedOperandsByKey(ArrayRef<BoUpSLP::ValueList> Operands,
if (SlotSrcCol[Slot] != NumCols)
continue;
for (unsigned Col : seq<unsigned>(NumCols)) {
- if (!ColClaimed[Col] && NegatedColumns[Col] == NegatedColumns[Slot]) {
+ if (!ColClaimed[Col] && IsNegated(Lane, Col) == IsNegated(Lane, Slot)) {
SlotSrcCol[Slot] = Col;
ColClaimed[Col] = true;
break;
@@ -12719,8 +12721,9 @@ void BoUpSLP::buildTreeRec(ArrayRef<Value *> VLRef, unsigned Depth,
SmallVector<ValueList> Operands = Analysis.buildOperands(S, VL);
// Flatten associative binary chains into operand columns. Only the peeled
// chain links are required to be single-use (they are erased); the root
- // being flattened may have other uses. Skip alt-shuffle nodes and lanes
- // that are neither chain links nor copyable identity leaves. Restricted to
+ // being flattened may have other uses. Skip lanes that are neither chain
+ // links nor copyable identity leaves. Alternate nodes flatten too, but
+ // each lane keeps its own opcode on every combine level. Restricted to
// BinaryOperator: isAssociative() is also true for associative intrinsics
// (e.g. smax/smin/umax/umin), which are CallInst, not BinaryOperator, and
// are not supported by the copyable-identity machinery used below
@@ -12729,27 +12732,51 @@ void BoUpSLP::buildTreeRec(ArrayRef<Value *> VLRef, unsigned Depth,
SmallVector<Value *> ReassocScalars;
// Sign of each flattened operand column (a subtracted leaf is negated).
SmallBitVector NegatedColumns;
+ // Per-lane subtract markers for flattened alternate nodes.
+ SmallBitVector SubLanes;
// Cached below (when the peel is kept) so the reorder step further down
// does not need to redo the aligning/scoring work.
SmallVector<ValueList> ReassocAlignedOperands;
// Snapshot of the pre-flatten operand columns, used by both revert points.
SmallVector<ValueList> NaturalTwoColumns;
std::tuple<unsigned, unsigned, unsigned, int> ReassocPeeledQuality;
- if (VectorizeReassociatedOps && !S.isAltShuffle() && Operands.size() == 2 &&
+ if (VectorizeReassociatedOps && Operands.size() == 2 &&
all_of(VL, [&](Value *V) {
auto *I = dyn_cast<BinaryOperator>(V);
+ if (S.isAltShuffle())
+ return I && isReassocChainLink(I);
return S.isCopyableElement(V) || (I && isReassocChainLink(I));
})) {
NaturalTwoColumns = Operands;
- scanAssociativeOperands(S, *DT, *DL, *TTI, *TLI, *this, Operands,
- NegatedColumns, ReassocScalars);
+ if (S.isAltShuffle()) {
+ if (SmallVector<SmallVector<Value *>> Flattened =
+ scanAltAssociativeOperands(S, *TLI, VL, Operands[0], Operands[1],
+ ReassocScalars, SubLanes);
+ !Flattened.empty()) {
+ Operands.clear();
+ for (auto &Col : Flattened)
+ Operands.emplace_back(std::move(Col));
+ }
+ } else {
+ scanAssociativeOperands(S, *DT, *DL, *TTI, *TLI, *this, Operands,
+ NegatedColumns, ReassocScalars);
+ }
// Drop flattening unless realigning improves load or broadcast column
// structure; an unimproved peel ties and reverts to natural columns.
if (!ReassocScalars.empty()) {
+ auto IsNegated = [&](unsigned Lane, unsigned Col) {
+ return S.isAltShuffle() ? Col > 0 && SubLanes.test(Lane)
+ : NegatedColumns.test(Col);
+ };
ReassocAlignedOperands =
- alignReassociatedOperandsByKey(Operands, NegatedColumns, *TLI);
- ReassocPeeledQuality =
- getReassocColumnsQuality(Operands, *this, S.getOpcode());
+ alignReassociatedOperandsByKey(Operands, IsNegated, *TLI);
+ // Alternate nodes compare against the natural two-column form rather
+ // than the raw peel: the natural form pays one lane-select shuffle per
+ // level, so the flatten is worth keeping whenever the realigned
+ // columns improve on the natural ones.
+ ReassocPeeledQuality = getReassocColumnsQuality(
+ S.isAltShuffle() ? NaturalTwoColumns : Operands, *this,
+ S.getOpcode());
// The unique-value count (4th field) is only a tie-break for the
// later reorder-or-not decision, not for this one.
auto DropUniqueCount = [](const auto &Quality) {
@@ -12764,6 +12791,28 @@ void BoUpSLP::buildTreeRec(ArrayRef<Value *> VLRef, unsigned Depth,
}
}
}
+ // Registers peeled chain links on the node so they are erased with it and
+ // their scheduling deps are released as reassociated operands.
+ auto RegisterReassocScalars = [&](TreeEntry *TE) {
+ for (Value *V : ReassocScalars) {
+ TE->addReassocScalar(V);
+ SmallVectorImpl<const TreeEntry *> &Owners =
+ ReassocScalarToTreeEntries.try_emplace(V).first->second;
+ if (!is_contained(Owners, TE))
+ Owners.push_back(TE);
+ }
+ };
+ // A value routed into several columns is consumed by several combines;
+ // that distorts both the cost and the vector structure (duplicated loads,
+ // partial masked columns), so prefer the natural two-column shape.
+ auto HasDupColumnValues = [&]() {
+ SmallPtrSet<const Value *, 16> ColumnValues;
+ return any_of(Operands, [&](const ValueList &Col) {
+ return any_of(Col, [&](const Value *V) {
+ return !isa<Constant>(V) && !ColumnValues.insert(V).second;
+ });
+ });
+ };
ScheduleBundle Empty;
ScheduleBundle &Bundle = BundlePtr.value() ? *BundlePtr.value() : Empty;
LLVM_DEBUG(dbgs() << "SLP: We are able to schedule this bundle.\n");
@@ -13109,29 +13158,14 @@ void BoUpSLP::buildTreeRec(ArrayRef<Value *> VLRef, unsigned Depth,
} else {
Operands = std::move(ReassocAlignedOperands);
}
- // A value routed into several columns is consumed by several
- // combines; that distorts both the cost and the vector structure
- // (duplicated loads, partial masked columns), so prefer the natural
- // two-column shape.
- SmallPtrSet<const Value *, 16> ColumnValues;
- if (any_of(Operands, [&](const ValueList &Col) {
- return any_of(Col, [&](const Value *V) {
- return !isa<Constant>(V) && !ColumnValues.insert(V).second;
- });
- })) {
+ if (HasDupColumnValues()) {
Operands = std::move(NaturalTwoColumns);
ReassocScalars.clear();
}
}
- if (!ReassocScalars.empty()) {
- for (Value *V : ReassocScalars) {
- TE->addReassocScalar(V);
- SmallVectorImpl<const TreeEntry *> &Owners =
- ReassocScalarToTreeEntries.try_emplace(V).first->second;
- if (!is_contained(Owners, TE))
- Owners.push_back(TE);
- }
- } else if (isa<BinaryOperator>(VL0) && isCommutative(VL0)) {
+ if (!ReassocScalars.empty())
+ RegisterReassocScalars(TE);
+ else if (isa<BinaryOperator>(VL0) && isCommutative(VL0)) {
VLOperands Ops(VL, Operands, S, *this);
Ops.reorder();
Operands[0] = Ops.getVL(0);
@@ -13264,14 +13298,25 @@ void BoUpSLP::buildTreeRec(ArrayRef<Value *> VLRef, unsigned Depth,
return;
}
- if (isa<BinaryOperator>(VL0) || CI) {
+ if (!ReassocScalars.empty()) {
+ // Peeled alternate chains take the realigned columns; the operand
+ // reorder below does not preserve the per-lane sign constraints.
+ Operands = std::move(ReassocAlignedOperands);
+ if (HasDupColumnValues()) {
+ Operands = std::move(NaturalTwoColumns);
+ ReassocScalars.clear();
+ }
+ }
+ if (ReassocScalars.empty() && (isa<BinaryOperator>(VL0) || CI)) {
VLOperands Ops(VL, Operands, S, *this);
Ops.reorder();
Operands[0] = Ops.getVL(0);
Operands[1] = Ops.getVL(1);
}
+ if (!ReassocScalars.empty())
+ RegisterReassocScalars(TE);
TE->setOperands(Operands);
- for (unsigned I : seq<unsigned>(VL0->getNumOperands()))
+ for (unsigned I : seq<unsigned>(TE->getNumOperands()))
buildTreeRec(TE->getOperand(I), Depth + 1, {TE, I});
return;
}
@@ -16760,6 +16805,39 @@ BoUpSLP::getEntryCost(const TreeEntry *E, ArrayRef<Value *> VectorizedVals,
ScalarCost, "Calculated costs for Tree"));
return VecCost - ScalarCost;
};
+ // Price peeled intermediate instructions on the scalar side: they are
+ // erased when the node vectorizes. The peeled cost is folded into the
+ // first scalar-cost query so the cost dump reports the full scalar cost.
+ // Peeled chain links are always 2-operand associative binops, priced per
+ // instruction so the operand properties (constants, uniformity) apply.
+ auto GetCostDiffWithPeeled =
+ [&](function_ref<InstructionCost(unsigned)> ScalarEltCost,
+ function_ref<InstructionCost(InstructionCost)> VectorCost) {
+ InstructionCost PeeledScalarCost = 0;
+ for (Value *V : E->getReassocScalars()) {
+ auto *I = cast<Instruction>(V);
+ TTI::OperandValueInfo Op1Info = TTI::getOperandInfo(I->getOperand(0));
+ TTI::OperandValueInfo Op2Info = TTI::getOperandInfo(I->getOperand(1));
+ PeeledScalarCost += TTI->getArithmeticInstrCost(
+ I->getOpcode(), OrigScalarTy, CostKind, Op1Info, Op2Info);
+ }
+ bool PeeledCostAdded = false;
+ InstructionCost CostDiff = GetCostDiff(
+ [&](unsigned Idx) {
+ InstructionCost Cost = ScalarEltCost(Idx);
+ if (!PeeledCostAdded) {
+ PeeledCostAdded = true;
+ Cost += PeeledScalarCost;
+ }
+ return Cost;
+ },
+ VectorCost);
+ // Every scalar may be marked as used elsewhere, leaving the
+ // scalar-cost query uncalled and the peeled cost unapplied.
+ if (!PeeledCostAdded)
+ CostDiff -= PeeledScalarCost;
+ return CostDiff;
+ };
// Calculate cost difference from vectorizing set of GEPs.
// Negative value means vectorizing is profitable.
auto GetGEPCostDiff = [=](ArrayRef<Value *> Ptrs, Value *BasePtr) {
@@ -17474,35 +17552,7 @@ BoUpSLP::getEntryCost(const TreeEntry *E, ArrayRef<Value *> VectorizedVals,
}
return Cost + CommonCost;
};
- // Price peeled intermediate instructions on the scalar side: they are
- // erased when the node vectorizes. Folded into the first scalar-cost
- // query so the cost dump reports the full scalar cost. These are always
- // 2-operand chain links (adds, subtracts, or other associative binops),
- // so operand 1 is always the second operand.
- InstructionCost PeeledScalarCost = 0;
- for (Value *V : E->getReassocScalars()) {
- auto *I = cast<Instruction>(V);
- TTI::OperandValueInfo Op1Info = TTI::getOperandInfo(I->getOperand(0));
- TTI::OperandValueInfo Op2Info = TTI::getOperandInfo(I->getOperand(1));
- PeeledScalarCost += TTI->getArithmeticInstrCost(
- I->getOpcode(), OrigScalarTy, CostKind, Op1Info, Op2Info);
- }
- bool PeeledCostAdded = false;
- InstructionCost CostDiff = GetCostDiff(
- [&](unsigned Idx) {
- InstructionCost Cost = GetScalarCost(Idx);
- if (!PeeledCostAdded) {
- PeeledCostAdded = true;
- Cost += PeeledScalarCost;
- }
- return Cost;
- },
- GetVectorCost);
- // Every scalar may be marked as used elsewhere, leaving the scalar-cost
- // query uncalled and the peeled cost unapplied.
- if (!PeeledCostAdded)
- CostDiff -= PeeledScalarCost;
- return CostDiff;
+ return GetCostDiffWithPeeled(GetScalarCost, GetVectorCost);
}
case Instruction::GetElementPtr: {
return CommonCost + GetGEPCostDiff(VL, VL0);
@@ -17780,10 +17830,36 @@ BoUpSLP::getEntryCost(const TreeEntry *E, ArrayRef<Value *> VectorizedVals,
// No need to add new vector costs here since we're going to reuse
// same main/alternate vector ops, just do different shuffling.
} else if (Instruction::isBinaryOp(E->getOpcode())) {
- VecCost =
- TTIRef.getArithmeticInstrCost(E->getOpcode(), VecTy, CostKind);
- VecCost +=
- TTIRef.getArithmeticInstrCost(E->getAltOpcode(), VecTy, CostKind);
+ // Peeled alternate chains fold the operand columns into one pure
+ // main-opcode chain and one pure alt-opcode chain, followed by a
+ // single lane-select shuffle; a plain alternate node is a single
+ // combine. Each combine is priced with the properties of the column
+ // it folds in; the other operand is the running fold, which stays
+ // constant while every folded column is constant (such combines
+ // constant-fold away in codegen) and stays uniform while every
+ // folded column is uniform.
+ auto ChainCost = [&](unsigned Opcode) {
+ InstructionCost Cost = 0;
+ TTI::OperandValueInfo RunningInfo = getOperandInfo(E->getOperand(0));
+ for (unsigned Idx : seq<unsigned>(1, E->getNumOperands())) {
+ TTI::OperandValueInfo ColInfo = getOperandInfo(E->getOperand(Idx));
+ if (!RunningInfo.isConstant() || !ColInfo.isConstant())
+ Cost += TTIRef.getArithmeticInstrCost(Opcode, VecTy, CostKind,
+ RunningInfo, ColInfo, {},
+ nullptr, TLI);
+ TTI::OperandValueKind Kind = TTI::OK_AnyValue;
+ if (RunningInfo.isConstant() && ColInfo.isConstant())
+ Kind = RunningInfo.Kind == TTI::OK_UniformConstantValue &&
+ ColInfo.Kind == TTI::OK_UniformConstantValue
+ ? TTI::OK_UniformConstantValue
+ : TTI::OK_NonUniformConstantValue;
+ else if (RunningInfo.isUniform() && ColInfo.isUniform())
+ Kind = TTI::OK_UniformValue;
+ RunningInfo = {Kind, TTI::OP_None};
+ }
+ return Cost;
+ };
+ VecCost = ChainCost(E->getOpcode()) + ChainCost(E->getAltOpcode());
} else if (auto *CI0 = dyn_cast<CmpInst>(VL0)) {
auto *MaskTy = getWidenedType(Builder.getInt1Ty(), VL.size());
VecCost = TTIRef.getCmpSelInstrCost(
@@ -17841,7 +17917,9 @@ BoUpSLP::getEntryCost(const TreeEntry *E, ArrayRef<Value *> VectorizedVals,
// Patterns like [fadd,fsub] can be combined into a single instruction
// in x86. Reordering them into [fsub,fadd] blocks this pattern. So we
// need to take into account their order when looking for the most used
- // order.
+ // order. Linearized chains emit no alternate-ops pattern.
+ if (E->hasReassocScalars())
+ return VecCost;
unsigned Opcode0 = E->getOpcode();
unsigned Opcode1 = E->getAltOpcode();
SmallBitVector OpcodeMask(
@@ -17893,7 +17971,7 @@ BoUpSLP::getEntryCost(const TreeEntry *E, ArrayRef<Value *> VectorizedVals,
}
return TTI::TCC_Free;
});
- return GetCostDiff(GetScalarCost, GetVectorCost);
+ return GetCostDiffWithPeeled(GetScalarCost, GetVectorCost);
}
case Instruction::Freeze:
return CommonCost;
@@ -24341,6 +24419,75 @@ Value *BoUpSLP::vectorizeTree(TreeEntry *E) {
(isa<CmpInst>(VL0) && isa<CmpInst>(E->getAltOp()))) &&
"Invalid Shuffle Vector Operand");
+ // Gather up main and alt scalar ops to propagate IR flags to each
+ // vector operation and build the lane-select mask.
+ ValueList OpScalars, AltScalars;
+ SmallVector<int> Mask;
+ E->buildAltOpShuffleMask(
+ [E, this](Instruction *I) {
+ assert(E->getMatchingMainOpOrAltOp(I) &&
+ "Unexpected main/alternate opcode");
+ return isAlternateInstruction(I, E->getMainOp(), E->getAltOp(),
+ *TLI);
+ },
+ Mask, &OpScalars, &AltScalars);
+ if (auto *VecTy = dyn_cast<FixedVectorType>(ScalarTy)) {
+ assert(SLPReVec && "FixedVectorType is not expected.");
+ transformScalarShuffleIndiciesToVector(VecTy->getNumElements(), Mask);
+ }
+
+ if (E->hasReassocScalars() && Instruction::isBinaryOp(E->getOpcode())) {
+ // Peeled alternate chains are linearized completely: the operand
+ // columns fold into one pure main-opcode chain and one pure
+ // alt-opcode chain, followed by a single lane-select shuffle. Every
+ // lane keeps one opcode on all levels, so its value is exact in one
+ // of the chains.
+ setInsertPointAfterBundle(E);
+ for (Value *V : E->getReassocScalars()) {
+ auto *I = cast<Instruction>(V);
+ (isAlternateInstruction(I, E->getMainOp(), E->getAltOp(), *TLI)
+ ? AltScalars
+ : OpScalars)
+ .push_back(I);
+ }
+ auto FoldColumns = [&](unsigned Opcode,
+ const ValueList &FlagScalars) {
+ Value *R = nullptr;
+ for (unsigned Idx : seq<unsigned>(E->getNumOperands())) {
+ Value *Column = vectorizeOperand(E, Idx);
+ if (Column->getType() != VecTy)
+ Column = Builder.CreateIntCast(Column, VecTy,
+ GetOperandSignedness(Idx));
+ if (!R) {
+ R = Column;
+ continue;
+ }
+ R = Builder.CreateBinOp(
+ static_cast<Instruction::BinaryOps>(Opcode), R, Column);
+ PropagateIRFlags(R, Opcode, FlagScalars);
+ auto *I = dyn_cast<Instruction>(R);
+ if (!I)
+ continue;
+ // Regrouping can invalidate flags even when each original
+ // step was safe.
+ I->dropPoisonGeneratingFlags();
+ GatherShuffleExtractSeq.insert(I);
+ CSEBlocks.insert(I->getParent());
+ }
+ return R;
+ };
+ Value *V0 = FoldColumns(E->getOpcode(), OpScalars);
+ Value *V1 = FoldColumns(E->getAltOpcode(), AltScalars);
+ V = Builder.CreateShuffleVector(V0, V1, Mask);
+ if (auto *I = dyn_cast<Instruction>(V)) {
+ GatherShuffleExtractSeq.insert(I);
+ CSEBlocks.insert(I->getParent());
+ }
+ E->VectorizedValue = V;
+ ++NumVectorInstructions;
+ return V;
+ }
+
Value *LHS = nullptr, *RHS = nullptr;
if (Instruction::isBinaryOp(E->getOpcode()) || isa<CmpInst>(VL0)) {
setInsertPointAfterBundle(E);
@@ -24419,27 +24566,10 @@ Value *BoUpSLP::vectorizeTree(TreeEntry *E) {
}
}
- // Create shuffle to take alternate operations from the vector.
- // Also, gather up main and alt scalar ops to propagate IR flags to
- // each vector operation.
- ValueList OpScalars, AltScalars;
- SmallVector<int> Mask;
- E->buildAltOpShuffleMask(
- [E, this](Instruction *I) {
- assert(E->getMatchingMainOpOrAltOp(I) &&
- "Unexpected main/alternate opcode");
- return isAlternateInstruction(I, E->getMainOp(), E->getAltOp(),
- *TLI);
- },
- Mask, &OpScalars, &AltScalars);
-
PropagateIRFlags(V0, E->getOpcode(), OpScalars);
PropagateIRFlags(V1, E->getAltOpcode(), AltScalars);
- if (auto *VecTy = dyn_cast<FixedVectorType>(ScalarTy)) {
- assert(SLPReVec && "FixedVectorType is not expected.");
- transformScalarShuffleIndiciesToVector(VecTy->getNumElements(), Mask);
- }
+ // Create shuffle to take alternate operations from the vector.
V = Builder.CreateShuffleVector(V0, V1, Mask);
if (auto *I = dyn_cast<Instruction>(V)) {
GatherShuffleExtractSeq.insert(I);
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCompatibilityAnalysis.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCompatibilityAnalysis.cpp
index fd31f8e8d01fc..0dcd867952839 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCompatibilityAnalysis.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCompatibilityAnalysis.cpp
@@ -14,6 +14,7 @@
#include "llvm/ADT/STLExtras.h"
#include "llvm/ADT/SetVector.h"
#include "llvm/ADT/SmallVector.h"
+#include "llvm/ADT/SmallVectorExtras.h"
#include "llvm/Analysis/VectorUtils.h"
#include "llvm/IR/Constants.h"
#include "llvm/IR/InstrTypes.h"
@@ -808,4 +809,62 @@ bool isAlternateInstruction(Instruction *I, Instruction *MainOp,
}
return InstructionsState(MainOp, AltOp).getMatchingMainOpOrAltOp(I) == AltOp;
}
+
+SmallVector<SmallVector<Value *>> scanAltAssociativeOperands(
+ const InstructionsState &S, const TargetLibraryInfo &TLI,
+ ArrayRef<Value *> VL, ArrayRef<Value *> Op0, ArrayRef<Value *> Op1,
+ SmallVectorImpl<Value *> &ReassocScalars, SmallBitVector &SubLanes) {
+ assert(S.isAltShuffle() && "Expected an alternate node.");
+ const unsigned NumLanes = VL.size();
+ SmallVector<unsigned> LaneOpcodes =
+ map_to_vector(seq<unsigned>(NumLanes), [&](unsigned Lane) {
+ return isAlternateInstruction(cast<Instruction>(VL[Lane]),
+ S.getMainOp(), S.getAltOp(), TLI)
+ ? S.getAltOpcode()
+ : S.getOpcode();
+ });
+ // A lane value peels only as a single-use chain link with the lane's own
+ // opcode, keeping every combine level on the same main/alt pattern.
+ auto GetChainLink = [&](unsigned Lane, Value *V) -> Instruction * {
+ auto *I = dyn_cast<Instruction>(V);
+ if (!I || !I->hasOneUse() || I->getOpcode() != LaneOpcodes[Lane] ||
+ !isReassocChainLink(I))
+ return nullptr;
+ return I;
+ };
+ SmallVector<SmallVector<Value *>> Columns;
+ Columns.emplace_back(Op0.begin(), Op0.end());
+ Columns.emplace_back(Op1.begin(), Op1.end());
+ // The chain link of a commutative lane may sit in the second column;
+ // normalize so every lane's link leads.
+ for (unsigned Lane : seq<unsigned>(NumLanes)) {
+ if (GetChainLink(Lane, Columns[0][Lane]))
+ continue;
+ Instruction *Link = GetChainLink(Lane, Columns[1][Lane]);
+ if (!Link || !Link->isCommutative())
+ return {};
+ std::swap(Columns[0][Lane], Columns[1][Lane]);
+ }
+ // Peel the leading column while every lane stays a matching chain link.
+ while (all_of(seq<unsigned>(NumLanes), [&](unsigned Lane) {
+ return GetChainLink(Lane, Columns[0][Lane]) != nullptr;
+ })) {
+ SmallVector<Value *> NewColumn(NumLanes);
+ for (unsigned Lane : seq<unsigned>(NumLanes)) {
+ Instruction *Link = GetChainLink(Lane, Columns[0][Lane]);
+ ReassocScalars.push_back(Link);
+ NewColumn[Lane] = Link->getOperand(1);
+ Columns[0][Lane] = Link->getOperand(0);
+ }
+ Columns.insert(std::next(Columns.begin()), std::move(NewColumn));
+ }
+ if (ReassocScalars.empty())
+ return {};
+ SubLanes.resize(NumLanes);
+ for (unsigned Lane : seq<unsigned>(NumLanes))
+ if (LaneOpcodes[Lane] == Instruction::Sub ||
+ LaneOpcodes[Lane] == Instruction::FSub)
+ SubLanes.set(Lane);
+ return Columns;
+}
} // namespace llvm::slpvectorizer
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCompatibilityAnalysis.h b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCompatibilityAnalysis.h
index e91a96129bf6c..62c8fe985c4a7 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCompatibilityAnalysis.h
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCompatibilityAnalysis.h
@@ -19,6 +19,7 @@
#include "llvm/ADT/ArrayRef.h"
#include "llvm/ADT/BitmaskEnum.h"
#include "llvm/ADT/STLExtras.h"
+#include "llvm/ADT/SmallBitVector.h"
#include "llvm/ADT/SmallVector.h"
#include "llvm/Analysis/IVDescriptors.h"
#include "llvm/IR/Instruction.h"
@@ -306,6 +307,18 @@ convertTo(Instruction *I, const InstructionsState &S);
/// the given \p MainOp and \p AltOp instructions.
bool isAlternateInstruction(Instruction *I, Instruction *MainOp,
Instruction *AltOp, const TargetLibraryInfo &TLI);
+
+/// Peel the per-lane associative chains of an alternate node into operand
+/// columns. Lanes peel in lockstep and only chain links with the lane's own
+/// opcode, so every combine level keeps the root's main/alt opcode pattern
+/// and a subtract lane never becomes an add of a negated leaf. Only the
+/// leading (running) column peels: peeling a subtracted subtract would flip
+/// signs. \p SubLanes records the subtract lanes for the realignment sign
+/// query. Returns the flattened columns, empty when no level peels.
+SmallVector<SmallVector<Value *>> scanAltAssociativeOperands(
+ const InstructionsState &S, const TargetLibraryInfo &TLI,
+ ArrayRef<Value *> VL, ArrayRef<Value *> Op0, ArrayRef<Value *> Op1,
+ SmallVectorImpl<Value *> &ReassocScalars, SmallBitVector &SubLanes);
} // namespace llvm::slpvectorizer
#endif // LLVM_LIB_TRANSFORMS_VECTORIZE_SLPVECTORIZER_SLPCOMPATIBILITYANALYSIS_H
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/BinOpSameOpcodeHelper.ll b/llvm/test/Transforms/SLPVectorizer/X86/BinOpSameOpcodeHelper.ll
index 6f27555aeb3f1..1b97a85bfadf3 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/BinOpSameOpcodeHelper.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/BinOpSameOpcodeHelper.ll
@@ -4,17 +4,7 @@
define void @test() {
; CHECK-LABEL: @test(
; CHECK-NEXT: entry:
-; CHECK-NEXT: [[TMP0:%.*]] = lshr i64 0, 0
-; CHECK-NEXT: [[TMP1:%.*]] = sub i64 0, 1
-; CHECK-NEXT: [[TMP2:%.*]] = lshr i64 [[TMP1]], 0
-; CHECK-NEXT: [[UMIN120:%.*]] = call i64 @llvm.umin.i64(i64 [[TMP0]], i64 [[TMP2]])
-; CHECK-NEXT: [[TMP3:%.*]] = sub i64 0, 0
-; CHECK-NEXT: [[TMP4:%.*]] = lshr i64 [[TMP3]], 0
-; CHECK-NEXT: [[UMIN122:%.*]] = call i64 @llvm.umin.i64(i64 [[UMIN120]], i64 [[TMP4]])
-; CHECK-NEXT: [[TMP5:%.*]] = add i64 0, 1
-; CHECK-NEXT: [[TMP6:%.*]] = lshr i64 [[TMP5]], 0
-; CHECK-NEXT: [[UMIN123:%.*]] = call i64 @llvm.umin.i64(i64 [[UMIN122]], i64 [[TMP6]])
-; CHECK-NEXT: [[UMIN124:%.*]] = call i64 @llvm.umin.i64(i64 [[UMIN123]], i64 0)
+; CHECK-NEXT: [[UMIN124:%.*]] = call i64 @llvm.umin.i64(i64 0, i64 0)
; CHECK-NEXT: ret void
;
entry:
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/supernode.ll b/llvm/test/Transforms/SLPVectorizer/X86/supernode.ll
index 8c3aac2d59c11..fedadf5a2d7d6 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/supernode.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/supernode.ll
@@ -89,13 +89,10 @@ define void @test_supernode_addsub_alt(ptr %Aarray, ptr %Barray, ptr %Carray, pt
; ENABLED-NEXT: [[TMP0:%.*]] = load <2 x double>, ptr [[AARRAY:%.*]], align 8
; ENABLED-NEXT: [[TMP1:%.*]] = load <2 x double>, ptr [[BARRAY:%.*]], align 8
; ENABLED-NEXT: [[TMP2:%.*]] = load <2 x double>, ptr [[CARRAY:%.*]], align 8
-; ENABLED-NEXT: [[TMP3:%.*]] = shufflevector <2 x double> [[TMP0]], <2 x double> [[TMP2]], <2 x i32> <i32 0, i32 3>
-; ENABLED-NEXT: [[TMP4:%.*]] = fsub fast <2 x double> [[TMP3]], [[TMP1]]
-; ENABLED-NEXT: [[TMP5:%.*]] = fadd fast <2 x double> [[TMP3]], [[TMP1]]
-; ENABLED-NEXT: [[TMP6:%.*]] = shufflevector <2 x double> [[TMP4]], <2 x double> [[TMP5]], <2 x i32> <i32 0, i32 3>
-; ENABLED-NEXT: [[TMP7:%.*]] = shufflevector <2 x double> [[TMP2]], <2 x double> [[TMP0]], <2 x i32> <i32 0, i32 3>
-; ENABLED-NEXT: [[TMP8:%.*]] = fsub fast <2 x double> [[TMP6]], [[TMP7]]
-; ENABLED-NEXT: [[TMP9:%.*]] = fadd fast <2 x double> [[TMP6]], [[TMP7]]
+; ENABLED-NEXT: [[TMP4:%.*]] = fsub reassoc nsz arcp contract afn <2 x double> [[TMP0]], [[TMP1]]
+; ENABLED-NEXT: [[TMP8:%.*]] = fsub reassoc nsz arcp contract afn <2 x double> [[TMP4]], [[TMP2]]
+; ENABLED-NEXT: [[TMP5:%.*]] = fadd reassoc nsz arcp contract afn <2 x double> [[TMP0]], [[TMP1]]
+; ENABLED-NEXT: [[TMP9:%.*]] = fadd reassoc nsz arcp contract afn <2 x double> [[TMP5]], [[TMP2]]
; ENABLED-NEXT: [[TMP10:%.*]] = shufflevector <2 x double> [[TMP8]], <2 x double> [[TMP9]], <2 x i32> <i32 0, i32 3>
; ENABLED-NEXT: store <2 x double> [[TMP10]], ptr [[SARRAY:%.*]], align 8
; ENABLED-NEXT: ret void
diff --git a/llvm/test/Transforms/SLPVectorizer/vectorize-reorder-alt-shuffle.ll b/llvm/test/Transforms/SLPVectorizer/vectorize-reorder-alt-shuffle.ll
index c2f07de9fc6b5..dc78d1ba2a654 100644
--- a/llvm/test/Transforms/SLPVectorizer/vectorize-reorder-alt-shuffle.ll
+++ b/llvm/test/Transforms/SLPVectorizer/vectorize-reorder-alt-shuffle.ll
@@ -5,24 +5,13 @@
define void @foo(ptr %c, ptr %d) {
; X86-LABEL: @foo(
; X86-NEXT: entry:
-; X86-NEXT: [[ARRAYIDX1:%.*]] = getelementptr inbounds i8, ptr [[C:%.*]], i64 4
-; X86-NEXT: [[ARRAYIDX4:%.*]] = getelementptr inbounds i8, ptr [[C]], i64 1
-; X86-NEXT: [[ARRAYIDX12:%.*]] = getelementptr inbounds i8, ptr [[C]], i64 2
+; X86-NEXT: [[ARRAYIDX4:%.*]] = getelementptr inbounds i8, ptr [[C:%.*]], i64 1
; X86-NEXT: [[ADD_PTR53:%.*]] = getelementptr inbounds float, ptr [[D:%.*]], i64 -4
-; X86-NEXT: [[TMP0:%.*]] = load i8, ptr [[ARRAYIDX4]], align 1
-; X86-NEXT: [[TMP1:%.*]] = load i8, ptr [[ARRAYIDX1]], align 1
-; X86-NEXT: [[CONV5:%.*]] = zext i8 [[TMP0]] to i32
-; X86-NEXT: [[CONV2:%.*]] = zext i8 [[TMP1]] to i32
-; X86-NEXT: [[SHL6:%.*]] = shl nuw nsw i32 [[CONV5]], 2
-; X86-NEXT: [[AND:%.*]] = and i32 [[CONV2]], 3
-; X86-NEXT: [[TMP2:%.*]] = load <2 x i8>, ptr [[ARRAYIDX12]], align 1
-; X86-NEXT: [[TMP3:%.*]] = zext <2 x i8> [[TMP2]] to <2 x i16>
-; X86-NEXT: [[TMP4:%.*]] = shl <2 x i16> [[TMP3]], splat (i16 2)
-; X86-NEXT: [[TMP5:%.*]] = insertelement <4 x i32> poison, i32 [[SHL6]], i64 0
-; X86-NEXT: [[TMP6:%.*]] = zext <2 x i16> [[TMP4]] to <2 x i32>
-; X86-NEXT: [[TMP7:%.*]] = shufflevector <2 x i32> [[TMP6]], <2 x i32> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
-; X86-NEXT: [[TMP8:%.*]] = shufflevector <4 x i32> [[TMP5]], <4 x i32> [[TMP7]], <4 x i32> <i32 0, i32 4, i32 5, i32 poison>
-; X86-NEXT: [[TMP9:%.*]] = insertelement <4 x i32> [[TMP8]], i32 [[AND]], i64 3
+; X86-NEXT: [[TMP0:%.*]] = load <4 x i8>, ptr [[ARRAYIDX4]], align 1
+; X86-NEXT: [[TMP1:%.*]] = zext <4 x i8> [[TMP0]] to <4 x i32>
+; X86-NEXT: [[TMP2:%.*]] = shl nuw nsw <4 x i32> [[TMP1]], <i32 2, i32 2, i32 2, i32 3>
+; X86-NEXT: [[TMP3:%.*]] = and <4 x i32> [[TMP1]], <i32 2, i32 2, i32 2, i32 3>
+; X86-NEXT: [[TMP9:%.*]] = shufflevector <4 x i32> [[TMP2]], <4 x i32> [[TMP3]], <4 x i32> <i32 0, i32 1, i32 2, i32 7>
; X86-NEXT: [[TMP10:%.*]] = add nsw <4 x i32> undef, [[TMP9]]
; X86-NEXT: [[TMP11:%.*]] = sitofp <4 x i32> [[TMP10]] to <4 x float>
; X86-NEXT: [[TMP12:%.*]] = fdiv <4 x float> [[TMP11]], undef
More information about the llvm-commits
mailing list