[llvm] [SLP]Drop unprofitable splat subtrees before the profitability trim (PR #224470)
Alexey Bataev via llvm-commits
llvm-commits at lists.llvm.org
Mon Sep 21 05:00:51 PDT 2026
https://github.com/alexey-bataev updated https://github.com/llvm/llvm-project/pull/224470
>From a4391cb54ec0d8291c8ef4567b2d2f32ea4c7fd9 Mon Sep 17 00:00:00 2001
From: Alexey Bataev <a.bataev at outlook.com>
Date: Thu, 17 Sep 2026 16:06:40 -0700
Subject: [PATCH] =?UTF-8?q?[=F0=9D=98=80=F0=9D=97=BD=F0=9D=97=BF]=20initia?=
=?UTF-8?q?l=20version?=
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
Created using spr 1.3.7
---
.../Transforms/Vectorize/SLPVectorizer.cpp | 573 +++++++++---------
.../splat-gather-subtree-loop-extracts.ll | 15 +-
.../AArch64/splat-gather-subtree-satd.ll | 293 +++++----
.../splat-gather-subtree-trim-revert-cost.ll | 26 +-
4 files changed, 463 insertions(+), 444 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 40952c66faf1a..a88f4813e8235 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -18946,47 +18946,249 @@ BoUpSLP::calculateTreeCostAndTrimNonProfitable(ArrayRef<Value *> VectorizedVals,
SmallVector<
std::tuple<InstructionCost, InstructionCost, SmallVector<unsigned>>>
SubtreeCosts(VectorizableTree.size());
- auto UpdateParentNodes =
- [&](const TreeEntry *UserTE, const TreeEntry *TE,
- InstructionCost TotalCost, InstructionCost Cost,
- SmallDenseSet<std::pair<const TreeEntry *, const TreeEntry *>, 4>
- &VisitedUser,
- bool AddToList = true) {
- while (UserTE &&
- VisitedUser.insert(std::make_pair(TE, UserTE)).second) {
- std::get<0>(SubtreeCosts[UserTE->Idx]) += TotalCost;
- std::get<1>(SubtreeCosts[UserTE->Idx]) += Cost;
- if (AddToList)
- std::get<2>(SubtreeCosts[UserTE->Idx]).push_back(TE->Idx);
- UserTE = UserTE->UserTreeIndex.UserTE;
+ // (Re)computes the subtree costs for the nodes that are not deleted.
+ auto ComputeSubtreeCosts = [&]() {
+ SubtreeCosts.assign(
+ VectorizableTree.size(),
+ std::tuple<InstructionCost, InstructionCost, SmallVector<unsigned>>());
+ auto UpdateParentNodes =
+ [&](const TreeEntry *UserTE, const TreeEntry *TE,
+ InstructionCost TotalCost, InstructionCost Cost,
+ SmallDenseSet<std::pair<const TreeEntry *, const TreeEntry *>, 4>
+ &VisitedUser,
+ bool AddToList = true) {
+ while (UserTE &&
+ VisitedUser.insert(std::make_pair(TE, UserTE)).second) {
+ std::get<0>(SubtreeCosts[UserTE->Idx]) += TotalCost;
+ std::get<1>(SubtreeCosts[UserTE->Idx]) += Cost;
+ if (AddToList)
+ std::get<2>(SubtreeCosts[UserTE->Idx]).push_back(TE->Idx);
+ UserTE = UserTE->UserTreeIndex.UserTE;
+ }
+ };
+ for (const std::unique_ptr<TreeEntry> &Ptr : VectorizableTree) {
+ TreeEntry &TE = *Ptr;
+ if (DeletedNodes.contains(&TE))
+ continue;
+ // Combined subnodes are not costed on their own (their cost is 0), but
+ // must be included into the ancestors' subtree node lists, so that they
+ // get deleted together with the trimmed combined root.
+ InstructionCost C = NodesCosts.at(&TE);
+ InstructionCost ExtractCost = ExtractCosts.lookup(&TE);
+ std::get<0>(SubtreeCosts[TE.Idx]) += C + ExtractCost;
+ std::get<1>(SubtreeCosts[TE.Idx]) += C;
+ if (const TreeEntry *UserTE = TE.UserTreeIndex.UserTE) {
+ SmallDenseSet<std::pair<const TreeEntry *, const TreeEntry *>, 4>
+ VisitedUser;
+ UpdateParentNodes(UserTE, &TE, C + ExtractCost, C, VisitedUser);
+ }
+ }
+ SmallDenseSet<std::pair<const TreeEntry *, const TreeEntry *>, 4> Visited;
+ for (TreeEntry *TE : GatheredLoadsNodes) {
+ if (DeletedNodes.contains(TE))
+ continue;
+ InstructionCost TotalCost = std::get<0>(SubtreeCosts[TE->Idx]);
+ InstructionCost Cost = std::get<1>(SubtreeCosts[TE->Idx]);
+ for (Value *V : TE->Scalars) {
+ for (const TreeEntry *BVTE : ValueToGatherNodes.lookup(V)) {
+ if (DeletedNodes.contains(BVTE))
+ continue;
+ UpdateParentNodes(BVTE, TE, TotalCost, Cost, Visited,
+ /*AddToList=*/false);
}
- };
- for (const std::unique_ptr<TreeEntry> &Ptr : VectorizableTree) {
- TreeEntry &TE = *Ptr;
- // Combined subnodes are not costed on their own (their cost is 0), but
- // must be included into the ancestors' subtree node lists, so that they
- // get deleted together with the trimmed combined root.
- InstructionCost C = NodesCosts.at(&TE);
- InstructionCost ExtractCost = ExtractCosts.lookup(&TE);
- std::get<0>(SubtreeCosts[TE.Idx]) += C + ExtractCost;
- std::get<1>(SubtreeCosts[TE.Idx]) += C;
- if (const TreeEntry *UserTE = TE.UserTreeIndex.UserTE) {
- SmallDenseSet<std::pair<const TreeEntry *, const TreeEntry *>, 4>
- VisitedUser;
- UpdateParentNodes(UserTE, &TE, C + ExtractCost, C, VisitedUser);
- }
- }
- SmallDenseSet<std::pair<const TreeEntry *, const TreeEntry *>, 4> Visited;
- for (TreeEntry *TE : GatheredLoadsNodes) {
- InstructionCost TotalCost = std::get<0>(SubtreeCosts[TE->Idx]);
- InstructionCost Cost = std::get<1>(SubtreeCosts[TE->Idx]);
+ }
+ }
+ };
+ ComputeSubtreeCosts();
+ using ValuesToInsertTy =
+ SmallDenseMap<const TreeEntry *, SmallVector<Value *>>;
+ auto GetScalarTy = [&](const TreeEntry *TE) {
+ Type *ScalarTy = TE->Scalars.front()->getType();
+ auto It = MinBWs.find(TE);
+ if (It != MinBWs.end())
+ ScalarTy = IntegerType::get(ScalarTy->getContext(), It->second.first);
+ return ScalarTy;
+ };
+ // Lanes of the subtree scalars used by the surviving gather nodes, and the
+ // values to materialize in those gathers if the subtree is deleted.
+ auto FindDemandedElts = [&](TreeEntry *TE, ValuesToInsertTy &ValuesToInsert) {
+ APInt DemandedElts = APInt::getZero(TE->getVectorFactor());
+ for (Value *V : TE->Scalars) {
+ unsigned Pos = TE->findLaneForValue(V);
+ for (const TreeEntry *BVE : ValueToGatherNodes.lookup(V)) {
+ if (DeletedNodes.contains(BVE))
+ continue;
+ DemandedElts.setBit(Pos);
+ ValuesToInsert.try_emplace(BVE).first->second.push_back(V);
+ }
+ }
+ return DemandedElts;
+ };
+ // Cost of materializing the values directly in the surviving gather nodes
+ // that use them.
+ auto GetGatherInsertCost = [&](Type *ScalarTy,
+ const ValuesToInsertTy &ValuesToInsert) {
+ InstructionCost BVCost = 0;
+ for (const auto &[BVE, Values] : ValuesToInsert) {
+ APInt BVDemandedElts = APInt::getZero(BVE->getVectorFactor());
+ SmallVector<Value *> BVValues(BVE->getVectorFactor(),
+ PoisonValue::get(ScalarTy));
+ for (Value *V : Values) {
+ unsigned Pos = BVE->findLaneForValue(V);
+ BVValues[Pos] = V;
+ BVDemandedElts.setBit(Pos);
+ }
+ BVCost += getScalarizationOverhead(
+ *TTI, SLPReVec, ScalarTy,
+ cast<VectorType>(getWidenedType(ScalarTy, BVE->getVectorFactor())),
+ BVDemandedElts, /*Insert=*/true, /*Extract=*/false, CostKind,
+ BVDemandedElts.isAllOnes(), BVValues);
+ }
+ return BVCost;
+ };
+ auto RecostEntry = [&](const TreeEntry *TE) {
+ InstructionCost C = getEntryCost(TE, VectorizedVals, CheckedExtracts);
+ if (!C.isValid() || C == 0)
+ return C;
+ uint64_t Scale = EntryToScale.lookup(TE);
+ if (!Scale)
+ Scale = getEntryEffectiveScale(*TE);
+ return C * Scale;
+ };
+ // A splat subtree pays off only if some surviving gather node reuses it and
+ // its price plus the extracts of its scalars used by the remaining scalar
+ // code plus the current cost of the gather nodes that reuse it is not worse
+ // than re-emitting those gathers without the subtree.
+ auto IsSplatSubtreeProfitable = [&](TreeEntry *TE,
+ ValuesToInsertTy &ValuesToInsert,
+ InstructionCost &CurrentGathersCost,
+ InstructionCost &DroppedGathersCost) {
+ if (FindDemandedElts(TE, ValuesToInsert).isZero())
+ return false;
+ APInt ExtractElts = APInt::getZero(TE->getVectorFactor());
for (Value *V : TE->Scalars) {
- for (const TreeEntry *BVTE : ValueToGatherNodes.lookup(V))
- UpdateParentNodes(BVTE, TE, TotalCost, Cost, Visited,
- /*AddToList=*/false);
+ if (!isa<Instruction>(V) || TE->isCopyableElement(V))
+ continue;
+ // Too many users - the scalar is extracted anyway.
+ if (V->hasNUsesOrMore(UsesLimit) || any_of(V->users(), [&](User *U) {
+ return none_of(getTreeEntries(U), [&](const TreeEntry *UseTE) {
+ return !DeletedNodes.contains(UseTE) &&
+ !TransformedToGatherNodes.contains(UseTE);
+ });
+ }))
+ ExtractElts.setBit(TE->findLaneForValue(V));
+ }
+ Type *ScalarTy = GetScalarTy(TE);
+ InstructionCost KeepCost = getScalarizationOverhead(
+ *TTI, SLPReVec, ScalarTy,
+ cast<VectorType>(getWidenedType(ScalarTy, TE->getVectorFactor())),
+ ExtractElts, /*Insert=*/false, /*Extract=*/true, CostKind);
+ // Scale the extract cost to the subtree's execution frequency: the
+ // subtree and gather costs it is compared against are already loop-scaled.
+ if (KeepCost.isValid() && KeepCost != 0)
+ KeepCost *= getEntryEffectiveScale(*TE);
+ // Add the cost of the subtree itself, computed before any trimming:
+ // trimming of the subtree's own nodes would otherwise make it look
+ // artificially cheap. The inner nodes of the subtree keep their external
+ // scalar users too, so their extracts are part of the price as well.
+ KeepCost += std::get<1>(SubtreeCosts[TE->Idx]);
+ for (unsigned Idx : std::get<2>(SubtreeCosts[TE->Idx]))
+ KeepCost += ExtractCosts.lookup(VectorizableTree[Idx].get());
+ // The reusing gather nodes currently pay the broadcast cost; without
+ // the subtree they fall back to plain insertion sequences. Gather
+ // nodes already erased from NodesCosts are being deleted and do not
+ // count on either side.
+ CurrentGathersCost = 0;
+ for (const auto &[BVE, _] : ValuesToInsert)
+ CurrentGathersCost += NodesCosts.lookup(BVE);
+ KeepCost += CurrentGathersCost;
+ // Re-cost the gather nodes with the subtree tentatively deleted.
+ DeletedNodes.insert(TE);
+ SmallVector<TreeEntry *> TempDeleted;
+ for (unsigned Idx : std::get<2>(SubtreeCosts[TE->Idx])) {
+ TreeEntry *Child = VectorizableTree[Idx].get();
+ if (DeletedNodes.insert(Child).second)
+ TempDeleted.push_back(Child);
+ }
+ DroppedGathersCost = 0;
+ for (const auto &[BVE, _] : ValuesToInsert) {
+ if (!NodesCosts.contains(BVE))
+ continue;
+ DroppedGathersCost += RecostEntry(BVE);
+ }
+ DeletedNodes.erase(TE);
+ for (TreeEntry *Child : TempDeleted)
+ DeletedNodes.erase(Child);
+ // On a cost tie prefer dropping the subtree: the reusing gathers
+ // materialize the scalars at the same price with fewer instructions.
+ return KeepCost < DroppedGathersCost;
+ };
+ // Re-price the gathers that reused a dropped splat subtree, so the
+ // remaining subtrees are checked against the costs with it deleted.
+ auto SyncGatherCosts = [&](const ValuesToInsertTy &ValuesToInsert) {
+ for (const auto &[BVE, _] : ValuesToInsert)
+ if (!DeletedNodes.contains(BVE))
+ NodesCosts[BVE] = RecostEntry(BVE);
+ };
+ // Deletes the subtree. The per-node costs are erased eagerly only where the
+ // total is recomputed from them right after the deletion.
+ auto DeleteSubtree = [&](TreeEntry *TE, bool EraseCosts) {
+ DeletedNodes.insert(TE);
+ if (EraseCosts)
+ NodesCosts.erase(TE);
+ for (unsigned Idx : std::get<2>(SubtreeCosts[TE->Idx])) {
+ TreeEntry *Child = VectorizableTree[Idx].get();
+ DeletedNodes.insert(Child);
+ if (EraseCosts)
+ NodesCosts.erase(Child);
+ }
+ };
+ // Gathered loads subtrees left without surviving gather users are dead.
+ auto DropDeadGatheredLoads = [&](bool EraseCosts) {
+ for (TreeEntry *TE : GatheredLoadsNodes) {
+ if (DeletedNodes.contains(TE))
+ continue;
+ ValuesToInsertTy ValuesToInsert;
+ if (!FindDemandedElts(TE, ValuesToInsert).isZero())
+ continue;
+ DeleteSubtree(TE, EraseCosts);
}
+ };
+ // Drops the splat subtrees that are unused by the surviving gathers or do
+ // not pay off; returns true if anything was dropped.
+ auto DropUnprofitableSplatSubtrees = [&](bool EraseCosts) {
+ bool Dropped = false;
+ for (TreeEntry *TE : SplatGatheredScalarsRoots) {
+ if (DeletedNodes.contains(TE))
+ continue;
+ ValuesToInsertTy ValuesToInsert;
+ InstructionCost CurrentGathersCost = 0, DroppedGathersCost = 0;
+ if (IsSplatSubtreeProfitable(TE, ValuesToInsert, CurrentGathersCost,
+ DroppedGathersCost))
+ continue;
+ DeleteSubtree(TE, EraseCosts);
+ SyncGatherCosts(ValuesToInsert);
+ Dropped = true;
+ }
+ return Dropped;
+ };
+ auto SumNodesCosts = [&]() {
+ InstructionCost C = 0;
+ for (const auto &P : NodesCosts)
+ C += P.second;
+ return C;
+ };
+ // The splat subtrees are speculative. The ones that do not pay off even in
+ // the full tree are dropped before the trimming: the reusing gathers are
+ // priced for the broadcast from the subtree vector while the subtree is
+ // around, which distorts the trimming decisions of the main tree, and fewer
+ // surviving gathers after the trimming can only make keeping a subtree less
+ // profitable.
+ if (DropUnprofitableSplatSubtrees(/*EraseCosts=*/true)) {
+ DropDeadGatheredLoads(/*EraseCosts=*/true);
+ ComputeSubtreeCosts();
+ Cost = SumNodesCosts();
}
- Visited.clear();
using CostIndicesTy =
std::pair<TreeEntry *, std::tuple<InstructionCost, InstructionCost,
SmallVector<unsigned>>>;
@@ -19014,13 +19216,11 @@ BoUpSLP::calculateTreeCostAndTrimNonProfitable(ArrayRef<Value *> VectorizedVals,
(Worklist.top().first->Idx == 0 || Worklist.top().first->Idx == 1))
return Cost;
- // Original gather costs before any trimming; used to rebase the reference
- // cost on the recosted gathers if the trimming is reverted.
- SmallDenseMap<const TreeEntry *, InstructionCost> OrigGatherCosts;
- if (!SplatGatheredScalarsRoots.empty())
- for (const std::unique_ptr<TreeEntry> &TE : VectorizableTree)
- if (TE->isGather())
- OrigGatherCosts.try_emplace(TE.get(), NodesCosts.lookup(TE.get()));
+ // Node costs and deleted nodes before any trimming; the restored tree is
+ // priced from them if the trimming is reverted.
+ const SmallDenseMap<const TreeEntry *, InstructionCost> OrigNodesCosts =
+ NodesCosts;
+ const SmallPtrSet<const TreeEntry *, 8> OrigDeletedNodes = DeletedNodes;
bool Changed = false;
bool PreferTrimmedTree = false;
while (!Worklist.empty() && std::get<0>(Worklist.top().second) > 0) {
@@ -19194,131 +19394,6 @@ BoUpSLP::calculateTreeCostAndTrimNonProfitable(ArrayRef<Value *> VectorizedVals,
}
Worklist.pop();
}
- using ValuesToInsertTy =
- SmallDenseMap<const TreeEntry *, SmallVector<Value *>>;
- auto GetScalarTy = [&](const TreeEntry *TE) {
- Type *ScalarTy = TE->Scalars.front()->getType();
- auto It = MinBWs.find(TE);
- if (It != MinBWs.end())
- ScalarTy = IntegerType::get(ScalarTy->getContext(), It->second.first);
- return ScalarTy;
- };
- // Lanes of the subtree scalars used by the surviving gather nodes, and the
- // values to materialize in those gathers if the subtree is deleted.
- auto FindDemandedElts = [&](TreeEntry *TE, ValuesToInsertTy &ValuesToInsert) {
- APInt DemandedElts = APInt::getZero(TE->getVectorFactor());
- for (Value *V : TE->Scalars) {
- unsigned Pos = TE->findLaneForValue(V);
- for (const TreeEntry *BVE : ValueToGatherNodes.lookup(V)) {
- if (DeletedNodes.contains(BVE))
- continue;
- DemandedElts.setBit(Pos);
- ValuesToInsert.try_emplace(BVE).first->second.push_back(V);
- }
- }
- return DemandedElts;
- };
- // Cost of materializing the values directly in the surviving gather nodes
- // that use them.
- auto GetGatherInsertCost = [&](Type *ScalarTy,
- const ValuesToInsertTy &ValuesToInsert) {
- InstructionCost BVCost = 0;
- for (const auto &[BVE, Values] : ValuesToInsert) {
- APInt BVDemandedElts = APInt::getZero(BVE->getVectorFactor());
- SmallVector<Value *> BVValues(BVE->getVectorFactor(),
- PoisonValue::get(ScalarTy));
- for (Value *V : Values) {
- unsigned Pos = BVE->findLaneForValue(V);
- BVValues[Pos] = V;
- BVDemandedElts.setBit(Pos);
- }
- BVCost += getScalarizationOverhead(
- *TTI, SLPReVec, ScalarTy,
- cast<VectorType>(getWidenedType(ScalarTy, BVE->getVectorFactor())),
- BVDemandedElts, /*Insert=*/true, /*Extract=*/false, CostKind,
- BVDemandedElts.isAllOnes(), BVValues);
- }
- return BVCost;
- };
- auto RecostEntry = [&](const TreeEntry *TE) {
- InstructionCost C = getEntryCost(TE, VectorizedVals, CheckedExtracts);
- if (!C.isValid() || C == 0)
- return C;
- uint64_t Scale = EntryToScale.lookup(TE);
- if (!Scale)
- Scale = getEntryEffectiveScale(*TE);
- return C * Scale;
- };
- // A splat subtree pays off only if its price plus the extracts of its
- // scalars used by the remaining scalar code plus the current cost of the
- // gather nodes that reuse it is not worse than re-emitting those gathers
- // without the subtree.
- auto IsSplatSubtreeProfitable = [&](TreeEntry *TE,
- const ValuesToInsertTy &ValuesToInsert,
- InstructionCost &CurrentGathersCost,
- InstructionCost &DroppedGathersCost) {
- APInt ExtractElts = APInt::getZero(TE->getVectorFactor());
- for (Value *V : TE->Scalars) {
- if (!isa<Instruction>(V) || TE->isCopyableElement(V))
- continue;
- // Too many users - the scalar is extracted anyway.
- if (V->hasNUsesOrMore(UsesLimit) || any_of(V->users(), [&](User *U) {
- return none_of(getTreeEntries(U), [&](const TreeEntry *UseTE) {
- return !DeletedNodes.contains(UseTE) &&
- !TransformedToGatherNodes.contains(UseTE);
- });
- }))
- ExtractElts.setBit(TE->findLaneForValue(V));
- }
- Type *ScalarTy = GetScalarTy(TE);
- InstructionCost KeepCost = getScalarizationOverhead(
- *TTI, SLPReVec, ScalarTy,
- cast<VectorType>(getWidenedType(ScalarTy, TE->getVectorFactor())),
- ExtractElts, /*Insert=*/false, /*Extract=*/true, CostKind);
- // Scale the extract cost to the subtree's execution frequency: the
- // subtree and gather costs it is compared against are already loop-scaled.
- if (KeepCost.isValid() && KeepCost != 0)
- KeepCost *= getEntryEffectiveScale(*TE);
- // Add the cost of the subtree itself, computed before any trimming:
- // trimming of the subtree's own nodes would otherwise make it look
- // artificially cheap.
- KeepCost += std::get<1>(SubtreeCosts[TE->Idx]);
- // The reusing gather nodes currently pay the broadcast cost; without
- // the subtree they fall back to plain insertion sequences. Gather
- // nodes already erased from NodesCosts are being deleted and do not
- // count on either side.
- CurrentGathersCost = 0;
- for (const auto &[BVE, _] : ValuesToInsert)
- CurrentGathersCost += NodesCosts.lookup(BVE);
- KeepCost += CurrentGathersCost;
- // Re-cost the gather nodes with the subtree tentatively deleted.
- DeletedNodes.insert(TE);
- SmallVector<TreeEntry *> TempDeleted;
- for (unsigned Idx : std::get<2>(SubtreeCosts[TE->Idx])) {
- TreeEntry *Child = VectorizableTree[Idx].get();
- if (DeletedNodes.insert(Child).second)
- TempDeleted.push_back(Child);
- }
- DroppedGathersCost = 0;
- for (const auto &[BVE, _] : ValuesToInsert) {
- if (!NodesCosts.contains(BVE))
- continue;
- DroppedGathersCost += RecostEntry(BVE);
- }
- DeletedNodes.erase(TE);
- for (TreeEntry *Child : TempDeleted)
- DeletedNodes.erase(Child);
- // On a cost tie prefer dropping the subtree: the reusing gathers
- // materialize the scalars at the same price with fewer instructions.
- return KeepCost < DroppedGathersCost;
- };
- // Re-price the gathers that reused a dropped splat subtree, so the
- // remaining subtrees are checked against the costs with it deleted.
- auto SyncGatherCosts = [&](const ValuesToInsertTy &ValuesToInsert) {
- for (const auto &[BVE, _] : ValuesToInsert)
- if (!DeletedNodes.contains(BVE))
- NodesCosts[BVE] = RecostEntry(BVE);
- };
// The VF=2 instruction-count veto rejects the whole tree when the vector code
// has more instructions than the scalar code. The auxiliary subtrees - splat
// gather roots and gathered loads - are optional: the surviving gathers can
@@ -19355,9 +19430,7 @@ BoUpSLP::calculateTreeCostAndTrimNonProfitable(ArrayRef<Value *> VectorizedVals,
InstructionCost CurrentGathersCost = 0;
for (const auto &[BVE, _] : ValuesToInsert)
CurrentGathersCost += NodesCosts.lookup(BVE);
- DeletedNodes.insert(TE);
- for (unsigned Idx : std::get<2>(SubtreeCosts[TE->Idx]))
- DeletedNodes.insert(VectorizableTree[Idx].get());
+ DeleteSubtree(TE, /*EraseCosts=*/false);
// All reusers were dropped: the subtree is dead. Without current node
// costs the total conservatively keeps the subtree cost.
if (IsDead) {
@@ -19404,39 +19477,45 @@ BoUpSLP::calculateTreeCostAndTrimNonProfitable(ArrayRef<Value *> VectorizedVals,
// unprofitable ones instead of letting them reject the whole tree.
InstructionCost TotalCost = std::get<1>(SubtreeCosts.front());
for (TreeEntry *TE : SplatGatheredScalarsRoots) {
+ if (DeletedNodes.contains(TE))
+ continue;
ValuesToInsertTy ValuesToInsert;
InstructionCost CurrentGathersCost = 0, DroppedGathersCost = 0;
- if (!FindDemandedElts(TE, ValuesToInsert).isZero() &&
- IsSplatSubtreeProfitable(TE, ValuesToInsert, CurrentGathersCost,
+ if (IsSplatSubtreeProfitable(TE, ValuesToInsert, CurrentGathersCost,
DroppedGathersCost)) {
TotalCost += std::get<1>(SubtreeCosts[TE->Idx]);
continue;
}
- DeletedNodes.insert(TE);
- for (unsigned Idx : std::get<2>(SubtreeCosts[TE->Idx]))
- DeletedNodes.insert(VectorizableTree[Idx].get());
+ DeleteSubtree(TE, /*EraseCosts=*/false);
// The gather nodes that reused the subtree are re-emitted without it.
TotalCost += DroppedGathersCost - CurrentGathersCost;
SyncGatherCosts(ValuesToInsert);
}
- // Gathered loads subtrees left without surviving gather users are dead.
- for (TreeEntry *TE : GatheredLoadsNodes) {
- if (DeletedNodes.contains(TE))
- continue;
- ValuesToInsertTy ValuesToInsert;
- if (!FindDemandedElts(TE, ValuesToInsert).isZero())
- continue;
- DeletedNodes.insert(TE);
- for (unsigned Idx : std::get<2>(SubtreeCosts[TE->Idx]))
- DeletedNodes.insert(VectorizableTree[Idx].get());
- }
+ DropDeadGatheredLoads(/*EraseCosts=*/false);
TrimAuxSubtreesForInstCount(TotalCost, /*UseCurrentNodeCosts=*/false);
return TotalCost;
}
SmallPtrSet<TreeEntry *, 4> SubtreesToDelete;
- SmallPtrSet<TreeEntry *, 4> DroppedSplatSubtrees;
InstructionCost LoadsExtractsCost = 0;
+ // Check if all gather nodes that reuse the splat subtrees are marked for
+ // deletion. In this case the whole splat subtree must be deleted. If only
+ // some of the gathers are trimmed, keeping the subtree still costs its full
+ // price plus the extracts of the scalars used by the remaining scalar code,
+ // while the surviving gathers can materialize the splatted scalars
+ // directly. Drop the subtree if it does not pay off.
+ for (TreeEntry *TE : SplatGatheredScalarsRoots) {
+ if (DeletedNodes.contains(TE))
+ continue;
+ ValuesToInsertTy ValuesToInsert;
+ InstructionCost CurrentGathersCost, DroppedGathersCost;
+ if (IsSplatSubtreeProfitable(TE, ValuesToInsert, CurrentGathersCost,
+ DroppedGathersCost))
+ continue;
+ SubtreesToDelete.insert(TE);
+ NodesCosts.erase(TE);
+ }
+
// Check if all loads of gathered loads nodes are marked for deletion. In this
// case the whole gathered loads subtree must be deleted.
// Also, try to account for extracts, which might be required, if only part of
@@ -19466,35 +19545,6 @@ BoUpSLP::calculateTreeCostAndTrimNonProfitable(ArrayRef<Value *> VectorizedVals,
NodesCosts.erase(TE);
}
- // Check if all gather nodes that reuse the splat subtrees are marked for
- // deletion. In this case the whole splat subtree must be deleted. If only
- // some of the gathers are trimmed, keeping the subtree still costs its full
- // price plus the extracts of the scalars used by the remaining scalar code,
- // while the surviving gathers can materialize the splatted scalars
- // directly. Drop the subtree if it does not pay off.
- for (TreeEntry *TE : SplatGatheredScalarsRoots) {
- if (DeletedNodes.contains(TE))
- continue;
- ValuesToInsertTy ValuesToInsert;
- APInt DemandedElts = FindDemandedElts(TE, ValuesToInsert);
- if (!DemandedElts.isZero()) {
- InstructionCost CurrentGathersCost, DroppedGathersCost;
- if (IsSplatSubtreeProfitable(TE, ValuesToInsert, CurrentGathersCost,
- DroppedGathersCost))
- continue;
- // Dropped as unprofitable: exclude its cost from the reference cost, so
- // the trimming of the remaining tree is not reverted because of it, and
- // keep it deleted even if the trimming is reverted.
- DroppedSplatSubtrees.insert(TE);
- for (unsigned Idx : std::get<2>(SubtreeCosts[TE->Idx]))
- DroppedSplatSubtrees.insert(VectorizableTree[Idx].get());
- Cost -= std::get<1>(SubtreeCosts[TE->Idx]);
- }
- // Not used by the surviving gathers or not profitable to keep.
- SubtreesToDelete.insert(TE);
- NodesCosts.erase(TE);
- }
-
// Deleted all subtrees rooted at gathered loads nodes or splat subtrees.
for (std::unique_ptr<TreeEntry> &TE : VectorizableTree) {
if (TE->UserTreeIndex &&
@@ -19530,63 +19580,44 @@ BoUpSLP::calculateTreeCostAndTrimNonProfitable(ArrayRef<Value *> VectorizedVals,
<< ".\n"
<< "SLP: Current total cost = " << NewCost << "\n");
}
- const bool Reverted =
- NewCost + LoadsExtractsCost > Cost ||
- (!PreferTrimmedTree && NewCost + LoadsExtractsCost == Cost);
- if (Reverted) {
- DeletedNodes.clear();
+ // The trimming is reverted if the restored tree is not more expensive. The
+ // restored tree is priced the same way as the trimmed one, from the per-node
+ // costs: adjusting the pre-trimming total incrementally is not exact with
+ // the auxiliary subtrees around, since the subtree costs recorded before the
+ // trimming fold in the gathered loads they reuse, and the gathers inside a
+ // subtree change their price once their reuse sources are dropped.
+ auto ComputeRestoredCost = [&]() {
+ DeletedNodes = OrigDeletedNodes;
TransformedToGatherNodes.clear();
- // The dropped splat subtrees stay deleted: they were excluded from the
- // reference cost and must not be resurrected by the revert.
- DeletedNodes.insert(DroppedSplatSubtrees.begin(),
- DroppedSplatSubtrees.end());
- NewCost = Cost;
+ NodesCosts = OrigNodesCosts;
if (!SplatGatheredScalarsRoots.empty()) {
- // The revert restores the pre-trimming tree, including the splat
- // subtrees that were deleted as unused while their reusing gathers were
- // trimmed. Recost the gather nodes for the restored set of vectorized
- // nodes, then drop the splat subtrees that do not pay off in the
- // restored tree. The reference cost still holds the gather costs of the
- // full tree, where the already dropped splat subtrees served as reuse
- // sources; rebase it on the recosted costs.
- for (const std::unique_ptr<TreeEntry> &TE : VectorizableTree) {
- if (!TE->isGather() || DeletedNodes.contains(TE.get()))
- continue;
- InstructionCost C = RecostEntry(TE.get());
- NewCost += C - OrigGatherCosts.lookup(TE.get());
- NodesCosts[TE.get()] = C;
- }
- for (TreeEntry *TE : SplatGatheredScalarsRoots) {
- if (DeletedNodes.contains(TE))
- continue;
- ValuesToInsertTy ValuesToInsert;
- InstructionCost CurrentGathersCost = 0, DroppedGathersCost = 0;
- if (!FindDemandedElts(TE, ValuesToInsert).isZero() &&
- IsSplatSubtreeProfitable(TE, ValuesToInsert, CurrentGathersCost,
- DroppedGathersCost))
- continue;
- DeletedNodes.insert(TE);
- for (unsigned Idx : std::get<2>(SubtreeCosts[TE->Idx]))
- DeletedNodes.insert(VectorizableTree[Idx].get());
- NewCost += DroppedGathersCost - CurrentGathersCost -
- std::get<1>(SubtreeCosts[TE->Idx]);
- SyncGatherCosts(ValuesToInsert);
- }
+ for (const std::unique_ptr<TreeEntry> &TE : VectorizableTree)
+ if (TE->isGather() && !DeletedNodes.contains(TE.get()))
+ NodesCosts[TE.get()] = RecostEntry(TE.get());
+ DropUnprofitableSplatSubtrees(/*EraseCosts=*/false);
// Dropping the splat subtrees may leave the gathered loads subtrees
- // built for them without surviving gather users; delete those too.
- for (TreeEntry *TE : GatheredLoadsNodes) {
- if (DeletedNodes.contains(TE))
- continue;
- ValuesToInsertTy ValuesToInsert;
- if (!FindDemandedElts(TE, ValuesToInsert).isZero())
- continue;
- DeletedNodes.insert(TE);
- for (unsigned Idx : std::get<2>(SubtreeCosts[TE->Idx]))
- DeletedNodes.insert(VectorizableTree[Idx].get());
- NewCost -= std::get<1>(SubtreeCosts[TE->Idx]);
- }
+ // built for them without surviving gather users.
+ DropDeadGatheredLoads(/*EraseCosts=*/false);
+ for (const TreeEntry *TE : DeletedNodes)
+ NodesCosts.erase(TE);
}
+ return SumNodesCosts();
+ };
+ const InstructionCost TrimmedCost = NewCost + LoadsExtractsCost;
+ auto TrimmedDeletedNodes = DeletedNodes;
+ auto TrimmedTransformedNodes = TransformedToGatherNodes;
+ auto TrimmedNodesCosts = NodesCosts;
+ const InstructionCost RestoredCost = ComputeRestoredCost();
+ LLVM_DEBUG(dbgs() << "SLP: Trimmed tree cost = " << TrimmedCost
+ << ", restored tree cost = " << RestoredCost << ".\n");
+ const bool Reverted = TrimmedCost > RestoredCost ||
+ (!PreferTrimmedTree && TrimmedCost == RestoredCost);
+ if (Reverted) {
+ NewCost = RestoredCost;
} else {
+ DeletedNodes = std::move(TrimmedDeletedNodes);
+ TransformedToGatherNodes = std::move(TrimmedTransformedNodes);
+ NodesCosts = std::move(TrimmedNodesCosts);
// If the remaining tree is just a buildvector - exit, it will cause
// endless attempts to vectorize.
if (VectorizableTree.size() >= 2 && getRootNode().hasState() &&
@@ -19603,7 +19634,7 @@ BoUpSLP::calculateTreeCostAndTrimNonProfitable(ArrayRef<Value *> VectorizedVals,
TransformedToGatherNodes.contains(VectorizableTree[2].get()))
return InstructionCost::getInvalid();
}
- TrimAuxSubtreesForInstCount(NewCost, /*UseCurrentNodeCosts=*/!Reverted);
+ TrimAuxSubtreesForInstCount(NewCost, /*UseCurrentNodeCosts=*/true);
return NewCost;
}
diff --git a/llvm/test/Transforms/SLPVectorizer/AArch64/splat-gather-subtree-loop-extracts.ll b/llvm/test/Transforms/SLPVectorizer/AArch64/splat-gather-subtree-loop-extracts.ll
index c2ca451ba0fb3..572dab05408bd 100644
--- a/llvm/test/Transforms/SLPVectorizer/AArch64/splat-gather-subtree-loop-extracts.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AArch64/splat-gather-subtree-loop-extracts.ll
@@ -9,18 +9,21 @@
define i32 @splat_subtree_loop_extracts(double %div.i) {
; CHECK-LABEL: @splat_subtree_loop_extracts(
; CHECK-NEXT: entry:
-; CHECK-NEXT: [[TMP0:%.*]] = tail call double @llvm.fmuladd.f64(double 0.000000e+00, double 0.000000e+00, double 0.000000e+00)
+; CHECK-NEXT: [[TMP4:%.*]] = call <2 x double> @llvm.fmuladd.v2f64(<2 x double> zeroinitializer, <2 x double> <double 0.000000e+00, double -0.000000e+00>, <2 x double> zeroinitializer)
+; CHECK-NEXT: [[TMP14:%.*]] = insertelement <2 x double> <double poison, double 0.000000e+00>, double [[DIV_I1:%.*]], i64 0
; CHECK-NEXT: br label [[FOR_BODY46_I:%.*]]
; CHECK: for.body46.i:
+; CHECK-NEXT: [[TMP0:%.*]] = tail call double @llvm.fmuladd.f64(double 0.000000e+00, double 0.000000e+00, double 0.000000e+00)
+; CHECK-NEXT: [[DIV_I:%.*]] = fdiv double 0.000000e+00, 0.000000e+00
; CHECK-NEXT: [[ARRAYIDX19_US63_I_3_1:%.*]] = getelementptr i8, ptr poison, i64 328
-; CHECK-NEXT: [[MUL5_I471:%.*]] = fmul double [[TMP0]], [[DIV_I:%.*]]
-; CHECK-NEXT: [[TMP1:%.*]] = insertelement <2 x double> poison, double [[MUL5_I471]], i64 0
+; CHECK-NEXT: [[TMP1:%.*]] = fmul <2 x double> [[TMP14]], [[TMP4]]
+; CHECK-NEXT: [[MUL5_I471:%.*]] = fmul double [[TMP0]], [[DIV_I]]
; CHECK-NEXT: [[TMP2:%.*]] = shufflevector <2 x double> [[TMP1]], <2 x double> poison, <2 x i32> zeroinitializer
; CHECK-NEXT: [[TMP3:%.*]] = call <2 x double> @llvm.fmuladd.v2f64(<2 x double> [[TMP2]], <2 x double> zeroinitializer, <2 x double> zeroinitializer)
-; CHECK-NEXT: [[TMP4:%.*]] = call <2 x double> @llvm.fmuladd.v2f64(<2 x double> zeroinitializer, <2 x double> <double 0.000000e+00, double -0.000000e+00>, <2 x double> zeroinitializer)
-; CHECK-NEXT: [[TMP5:%.*]] = fmul <2 x double> [[TMP4]], <double +qnan, double 0.000000e+00>
+; CHECK-NEXT: [[TMP5:%.*]] = insertelement <2 x double> poison, double [[MUL5_I471]], i64 0
; CHECK-NEXT: [[TMP6:%.*]] = shufflevector <2 x double> [[TMP5]], <2 x double> poison, <2 x i32> zeroinitializer
-; CHECK-NEXT: [[TMP7:%.*]] = shufflevector <2 x double> [[TMP5]], <2 x double> poison, <2 x i32> <i32 1, i32 1>
+; CHECK-NEXT: [[TMP15:%.*]] = shufflevector <2 x double> [[TMP1]], <2 x double> poison, <2 x i32> <i32 1, i32 poison>
+; CHECK-NEXT: [[TMP7:%.*]] = shufflevector <2 x double> [[TMP15]], <2 x double> poison, <2 x i32> zeroinitializer
; CHECK-NEXT: [[TMP8:%.*]] = call <2 x double> @llvm.fmuladd.v2f64(<2 x double> [[TMP6]], <2 x double> zeroinitializer, <2 x double> [[TMP7]])
; CHECK-NEXT: [[TMP9:%.*]] = call <2 x double> @llvm.fmuladd.v2f64(<2 x double> zeroinitializer, <2 x double> zeroinitializer, <2 x double> [[TMP8]])
; CHECK-NEXT: [[TMP10:%.*]] = fmul <2 x double> zeroinitializer, [[TMP3]]
diff --git a/llvm/test/Transforms/SLPVectorizer/AArch64/splat-gather-subtree-satd.ll b/llvm/test/Transforms/SLPVectorizer/AArch64/splat-gather-subtree-satd.ll
index 93891801e9f59..f14406a87317d 100644
--- a/llvm/test/Transforms/SLPVectorizer/AArch64/splat-gather-subtree-satd.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AArch64/splat-gather-subtree-satd.ll
@@ -39,199 +39,186 @@ define i32 @satd_4x4(ptr %pix1, i64 %i_pix1, ptr %pix2, i64 %i_pix2) {
; CHECK-NEXT: [[ARRAYIDX12_3:%.*]] = getelementptr inbounds nuw i8, ptr [[ADD_PTR31_2]], i64 2
; CHECK-NEXT: [[ARRAYIDX15_3:%.*]] = getelementptr inbounds nuw i8, ptr [[ADD_PTR_2]], i64 3
; CHECK-NEXT: [[ARRAYIDX17_3:%.*]] = getelementptr inbounds nuw i8, ptr [[ADD_PTR31_2]], i64 3
+; CHECK-NEXT: [[TMP6:%.*]] = load i8, ptr [[ARRAYIDX15]], align 1
+; CHECK-NEXT: [[TMP7:%.*]] = load i8, ptr [[ARRAYIDX10]], align 1
; CHECK-NEXT: [[TMP0:%.*]] = load i8, ptr [[ARRAYIDX3]], align 1
; CHECK-NEXT: [[TMP1:%.*]] = load i8, ptr [[PIX1]], align 1
+; CHECK-NEXT: [[TMP4:%.*]] = load i8, ptr [[ARRAYIDX17]], align 1
+; CHECK-NEXT: [[TMP5:%.*]] = load i8, ptr [[ARRAYIDX12]], align 1
; CHECK-NEXT: [[TMP2:%.*]] = load i8, ptr [[ARRAYIDX5]], align 1
; CHECK-NEXT: [[TMP3:%.*]] = load i8, ptr [[PIX2]], align 1
-; CHECK-NEXT: [[TMP4:%.*]] = load i8, ptr [[ARRAYIDX15]], align 1
-; CHECK-NEXT: [[TMP5:%.*]] = load i8, ptr [[ARRAYIDX10]], align 1
-; CHECK-NEXT: [[TMP6:%.*]] = load i8, ptr [[ARRAYIDX17]], align 1
-; CHECK-NEXT: [[TMP7:%.*]] = load i8, ptr [[ARRAYIDX12]], align 1
+; CHECK-NEXT: [[TMP18:%.*]] = load i8, ptr [[ARRAYIDX15_1]], align 1
+; CHECK-NEXT: [[TMP9:%.*]] = load i8, ptr [[ARRAYIDX10_1]], align 1
+; CHECK-NEXT: [[TMP19:%.*]] = load i8, ptr [[ARRAYIDX3_1]], align 1
+; CHECK-NEXT: [[TMP22:%.*]] = load i8, ptr [[ADD_PTR]], align 1
+; CHECK-NEXT: [[TMP12:%.*]] = load i8, ptr [[ARRAYIDX17_1]], align 1
+; CHECK-NEXT: [[TMP13:%.*]] = load i8, ptr [[ARRAYIDX12_1]], align 1
+; CHECK-NEXT: [[TMP14:%.*]] = load i8, ptr [[ARRAYIDX5_1]], align 1
+; CHECK-NEXT: [[TMP15:%.*]] = load i8, ptr [[ADD_PTR31]], align 1
+; CHECK-NEXT: [[TMP16:%.*]] = load i8, ptr [[ARRAYIDX15_2]], align 1
+; CHECK-NEXT: [[TMP17:%.*]] = load i8, ptr [[ARRAYIDX10_2]], align 1
; CHECK-NEXT: [[TMP8:%.*]] = load i8, ptr [[ARRAYIDX3_2]], align 1
; CHECK-NEXT: [[TMP31:%.*]] = load i8, ptr [[ADD_PTR_1]], align 1
-; CHECK-NEXT: [[CONV18_3:%.*]] = zext i8 [[TMP31]] to i32
-; CHECK-NEXT: [[CONV:%.*]] = zext i8 [[TMP1]] to i32
+; CHECK-NEXT: [[TMP20:%.*]] = load i8, ptr [[ARRAYIDX17_2]], align 1
+; CHECK-NEXT: [[TMP21:%.*]] = load i8, ptr [[ARRAYIDX12_2]], align 1
; CHECK-NEXT: [[TMP10:%.*]] = load i8, ptr [[ARRAYIDX5_2]], align 1
; CHECK-NEXT: [[TMP11:%.*]] = load i8, ptr [[ADD_PTR31_1]], align 1
-; CHECK-NEXT: [[CONV2_2:%.*]] = zext i8 [[TMP11]] to i32
+; CHECK-NEXT: [[TMP24:%.*]] = load i8, ptr [[ARRAYIDX15_3]], align 1
+; CHECK-NEXT: [[TMP25:%.*]] = load i8, ptr [[ARRAYIDX10_3]], align 1
+; CHECK-NEXT: [[TMP26:%.*]] = load i8, ptr [[ARRAYIDX3_3]], align 1
+; CHECK-NEXT: [[TMP27:%.*]] = load i8, ptr [[ADD_PTR_2]], align 1
+; CHECK-NEXT: [[CONV11:%.*]] = zext i8 [[TMP7]] to i32
+; CHECK-NEXT: [[CONV:%.*]] = zext i8 [[TMP1]] to i32
+; CHECK-NEXT: [[CONV11_1:%.*]] = zext i8 [[TMP9]] to i32
+; CHECK-NEXT: [[CONV_1:%.*]] = zext i8 [[TMP22]] to i32
+; CHECK-NEXT: [[CONV18_3:%.*]] = zext i8 [[TMP17]] to i32
+; CHECK-NEXT: [[CONV_2:%.*]] = zext i8 [[TMP31]] to i32
+; CHECK-NEXT: [[CONV11_3:%.*]] = zext i8 [[TMP25]] to i32
+; CHECK-NEXT: [[CONV_3:%.*]] = zext i8 [[TMP27]] to i32
+; CHECK-NEXT: [[TMP28:%.*]] = load i8, ptr [[ARRAYIDX17_3]], align 1
+; CHECK-NEXT: [[TMP29:%.*]] = load i8, ptr [[ARRAYIDX12_3]], align 1
+; CHECK-NEXT: [[TMP30:%.*]] = load i8, ptr [[ARRAYIDX5_3]], align 1
+; CHECK-NEXT: [[TMP37:%.*]] = load i8, ptr [[ADD_PTR31_2]], align 1
+; CHECK-NEXT: [[CONV13:%.*]] = zext i8 [[TMP5]] to i32
; CHECK-NEXT: [[CONV2:%.*]] = zext i8 [[TMP3]] to i32
+; CHECK-NEXT: [[CONV13_1:%.*]] = zext i8 [[TMP13]] to i32
+; CHECK-NEXT: [[CONV2_1:%.*]] = zext i8 [[TMP15]] to i32
+; CHECK-NEXT: [[CONV2_2:%.*]] = zext i8 [[TMP21]] to i32
+; CHECK-NEXT: [[CONV2_4:%.*]] = zext i8 [[TMP11]] to i32
+; CHECK-NEXT: [[CONV13_3:%.*]] = zext i8 [[TMP29]] to i32
+; CHECK-NEXT: [[CONV2_3:%.*]] = zext i8 [[TMP37]] to i32
+; CHECK-NEXT: [[SHL22_3:%.*]] = sub nsw i32 [[CONV11]], [[CONV13]]
+; CHECK-NEXT: [[ADD9_3:%.*]] = sub nsw i32 [[CONV]], [[CONV2]]
+; CHECK-NEXT: [[SUB14_2:%.*]] = sub nsw i32 [[CONV11_1]], [[CONV13_1]]
+; CHECK-NEXT: [[SUB14:%.*]] = sub nsw i32 [[CONV_1]], [[CONV2_1]]
; CHECK-NEXT: [[SUB14_3:%.*]] = sub nsw i32 [[CONV18_3]], [[CONV2_2]]
-; CHECK-NEXT: [[SUB:%.*]] = sub nsw i32 [[CONV]], [[CONV2]]
-; CHECK-NEXT: [[CONV4_2:%.*]] = zext i8 [[TMP8]] to i32
+; CHECK-NEXT: [[SUB_2:%.*]] = sub nsw i32 [[CONV_2]], [[CONV2_4]]
+; CHECK-NEXT: [[ADD44:%.*]] = sub nsw i32 [[CONV11_3]], [[CONV13_3]]
+; CHECK-NEXT: [[SUB_1:%.*]] = sub nsw i32 [[CONV_3]], [[CONV2_3]]
+; CHECK-NEXT: [[CONV16:%.*]] = zext i8 [[TMP6]] to i32
; CHECK-NEXT: [[CONV4:%.*]] = zext i8 [[TMP0]] to i32
-; CHECK-NEXT: [[CONV6_2:%.*]] = zext i8 [[TMP10]] to i32
+; CHECK-NEXT: [[CONV16_1:%.*]] = zext i8 [[TMP18]] to i32
+; CHECK-NEXT: [[CONV4_1:%.*]] = zext i8 [[TMP19]] to i32
+; CHECK-NEXT: [[CONV4_2:%.*]] = zext i8 [[TMP16]] to i32
+; CHECK-NEXT: [[SUB:%.*]] = zext i8 [[TMP8]] to i32
+; CHECK-NEXT: [[CONV16_3:%.*]] = zext i8 [[TMP24]] to i32
+; CHECK-NEXT: [[CONV4_3:%.*]] = zext i8 [[TMP26]] to i32
+; CHECK-NEXT: [[CONV18:%.*]] = zext i8 [[TMP4]] to i32
; CHECK-NEXT: [[CONV6:%.*]] = zext i8 [[TMP2]] to i32
+; CHECK-NEXT: [[CONV18_1:%.*]] = zext i8 [[TMP12]] to i32
+; CHECK-NEXT: [[CONV6_1:%.*]] = zext i8 [[TMP14]] to i32
+; CHECK-NEXT: [[CONV6_2:%.*]] = zext i8 [[TMP20]] to i32
+; CHECK-NEXT: [[SUB7:%.*]] = zext i8 [[TMP10]] to i32
+; CHECK-NEXT: [[CONV18_4:%.*]] = zext i8 [[TMP28]] to i32
+; CHECK-NEXT: [[CONV6_3:%.*]] = zext i8 [[TMP30]] to i32
+; CHECK-NEXT: [[ADD20_3:%.*]] = sub nsw i32 [[CONV16]], [[CONV18]]
+; CHECK-NEXT: [[ADD23_3:%.*]] = sub nsw i32 [[CONV4]], [[CONV6]]
+; CHECK-NEXT: [[SUB19_2:%.*]] = sub nsw i32 [[CONV16_1]], [[CONV18_1]]
+; CHECK-NEXT: [[SUB19:%.*]] = sub nsw i32 [[CONV4_1]], [[CONV6_1]]
; CHECK-NEXT: [[SUB19_3:%.*]] = sub nsw i32 [[CONV4_2]], [[CONV6_2]]
-; CHECK-NEXT: [[SUB7:%.*]] = sub nsw i32 [[CONV4]], [[CONV6]]
-; CHECK-NEXT: [[ADD20_3:%.*]] = add nsw i32 [[SUB19_3]], [[SUB14_3]]
-; CHECK-NEXT: [[ADD23_3:%.*]] = add nsw i32 [[SUB7]], [[SUB]]
-; CHECK-NEXT: [[SUB21_3:%.*]] = sub nsw i32 [[SUB14_3]], [[SUB19_3]]
; CHECK-NEXT: [[SUB8:%.*]] = sub nsw i32 [[SUB]], [[SUB7]]
-; CHECK-NEXT: [[SHL22_3:%.*]] = shl nsw i32 [[SUB21_3]], 16
-; CHECK-NEXT: [[ADD9_3:%.*]] = shl nsw i32 [[SUB8]], 16
+; CHECK-NEXT: [[ADD58:%.*]] = sub nsw i32 [[CONV16_3]], [[CONV18_4]]
+; CHECK-NEXT: [[SUB7_1:%.*]] = sub nsw i32 [[CONV4_3]], [[CONV6_3]]
; CHECK-NEXT: [[ADD9_2:%.*]] = add nsw i32 [[ADD20_3]], [[SHL22_3]]
; CHECK-NEXT: [[ADD24_3:%.*]] = add nsw i32 [[ADD23_3]], [[ADD9_3]]
-; CHECK-NEXT: [[TMP12:%.*]] = load i8, ptr [[ARRAYIDX15_2]], align 1
-; CHECK-NEXT: [[TMP13:%.*]] = load i8, ptr [[ARRAYIDX10_2]], align 1
-; CHECK-NEXT: [[CONV11_2:%.*]] = zext i8 [[TMP13]] to i32
-; CHECK-NEXT: [[CONV11:%.*]] = zext i8 [[TMP5]] to i32
-; CHECK-NEXT: [[TMP14:%.*]] = load i8, ptr [[ARRAYIDX17_2]], align 1
-; CHECK-NEXT: [[TMP15:%.*]] = load i8, ptr [[ARRAYIDX12_2]], align 1
-; CHECK-NEXT: [[CONV13_2:%.*]] = zext i8 [[TMP15]] to i32
-; CHECK-NEXT: [[CONV13:%.*]] = zext i8 [[TMP7]] to i32
-; CHECK-NEXT: [[SUB14_2:%.*]] = sub nsw i32 [[CONV11_2]], [[CONV13_2]]
-; CHECK-NEXT: [[SUB14:%.*]] = sub nsw i32 [[CONV11]], [[CONV13]]
-; CHECK-NEXT: [[CONV16_2:%.*]] = zext i8 [[TMP12]] to i32
-; CHECK-NEXT: [[CONV16:%.*]] = zext i8 [[TMP4]] to i32
-; CHECK-NEXT: [[CONV18_2:%.*]] = zext i8 [[TMP14]] to i32
-; CHECK-NEXT: [[CONV18:%.*]] = zext i8 [[TMP6]] to i32
-; CHECK-NEXT: [[SUB19_2:%.*]] = sub nsw i32 [[CONV16_2]], [[CONV18_2]]
-; CHECK-NEXT: [[SUB19:%.*]] = sub nsw i32 [[CONV16]], [[CONV18]]
; CHECK-NEXT: [[ADD20_2:%.*]] = add nsw i32 [[SUB19_2]], [[SUB14_2]]
; CHECK-NEXT: [[ADD20:%.*]] = add nsw i32 [[SUB19]], [[SUB14]]
-; CHECK-NEXT: [[SUB21_2:%.*]] = sub nsw i32 [[SUB14_2]], [[SUB19_2]]
-; CHECK-NEXT: [[SUB21:%.*]] = sub nsw i32 [[SUB14]], [[SUB19]]
-; CHECK-NEXT: [[SHL22_2:%.*]] = shl nsw i32 [[SUB21_2]], 16
-; CHECK-NEXT: [[SHL22:%.*]] = shl nsw i32 [[SUB21]], 16
-; CHECK-NEXT: [[ADD23_2:%.*]] = add nsw i32 [[ADD20_2]], [[SHL22_2]]
-; CHECK-NEXT: [[ADD23:%.*]] = add nsw i32 [[ADD20]], [[SHL22]]
-; CHECK-NEXT: [[TMP16:%.*]] = load i8, ptr [[ARRAYIDX3_1]], align 1
-; CHECK-NEXT: [[TMP17:%.*]] = load i8, ptr [[ADD_PTR]], align 1
-; CHECK-NEXT: [[TMP18:%.*]] = load i8, ptr [[ARRAYIDX5_1]], align 1
-; CHECK-NEXT: [[TMP19:%.*]] = load i8, ptr [[ADD_PTR31]], align 1
-; CHECK-NEXT: [[TMP20:%.*]] = load i8, ptr [[ARRAYIDX15_1]], align 1
-; CHECK-NEXT: [[TMP21:%.*]] = load i8, ptr [[ARRAYIDX10_1]], align 1
-; CHECK-NEXT: [[TMP22:%.*]] = load i8, ptr [[ARRAYIDX17_1]], align 1
-; CHECK-NEXT: [[TMP23:%.*]] = load i8, ptr [[ARRAYIDX12_1]], align 1
-; CHECK-NEXT: [[TMP24:%.*]] = load i8, ptr [[ARRAYIDX3_3]], align 1
-; CHECK-NEXT: [[TMP25:%.*]] = load i8, ptr [[ADD_PTR_2]], align 1
-; CHECK-NEXT: [[CONV_3:%.*]] = zext i8 [[TMP25]] to i32
-; CHECK-NEXT: [[CONV_1:%.*]] = zext i8 [[TMP17]] to i32
-; CHECK-NEXT: [[TMP26:%.*]] = load i8, ptr [[ARRAYIDX5_3]], align 1
-; CHECK-NEXT: [[TMP27:%.*]] = load i8, ptr [[ADD_PTR31_2]], align 1
-; CHECK-NEXT: [[CONV2_3:%.*]] = zext i8 [[TMP27]] to i32
-; CHECK-NEXT: [[CONV2_1:%.*]] = zext i8 [[TMP19]] to i32
-; CHECK-NEXT: [[ADD44:%.*]] = sub nsw i32 [[CONV_3]], [[CONV2_3]]
-; CHECK-NEXT: [[SUB_1:%.*]] = sub nsw i32 [[CONV_1]], [[CONV2_1]]
-; CHECK-NEXT: [[CONV4_3:%.*]] = zext i8 [[TMP24]] to i32
-; CHECK-NEXT: [[CONV4_1:%.*]] = zext i8 [[TMP16]] to i32
-; CHECK-NEXT: [[CONV6_3:%.*]] = zext i8 [[TMP26]] to i32
-; CHECK-NEXT: [[CONV6_1:%.*]] = zext i8 [[TMP18]] to i32
-; CHECK-NEXT: [[ADD58:%.*]] = sub nsw i32 [[CONV4_3]], [[CONV6_3]]
-; CHECK-NEXT: [[SUB7_1:%.*]] = sub nsw i32 [[CONV4_1]], [[CONV6_1]]
+; CHECK-NEXT: [[ADD23_4:%.*]] = add nsw i32 [[SUB19_3]], [[SUB14_3]]
+; CHECK-NEXT: [[ADD23_1:%.*]] = add nsw i32 [[SUB8]], [[SUB_2]]
; CHECK-NEXT: [[ADD66:%.*]] = add nsw i32 [[ADD58]], [[ADD44]]
; CHECK-NEXT: [[ADD_1:%.*]] = add nsw i32 [[SUB7_1]], [[SUB_1]]
-; CHECK-NEXT: [[SUB67:%.*]] = sub nsw i32 [[ADD44]], [[ADD58]]
-; CHECK-NEXT: [[SUB8_1:%.*]] = sub nsw i32 [[SUB_1]], [[SUB7_1]]
+; CHECK-NEXT: [[SUB67:%.*]] = sub nsw i32 [[SHL22_3]], [[ADD20_3]]
+; CHECK-NEXT: [[SUB8_1:%.*]] = sub nsw i32 [[ADD9_3]], [[ADD23_3]]
+; CHECK-NEXT: [[SUB21_4:%.*]] = sub nsw i32 [[SUB14_2]], [[SUB19_2]]
+; CHECK-NEXT: [[SUB21_1:%.*]] = sub nsw i32 [[SUB14]], [[SUB19]]
+; CHECK-NEXT: [[SUB21_2:%.*]] = sub nsw i32 [[SUB14_3]], [[SUB19_3]]
+; CHECK-NEXT: [[SUB8_2:%.*]] = sub nsw i32 [[SUB_2]], [[SUB8]]
+; CHECK-NEXT: [[SUB21_3:%.*]] = sub nsw i32 [[ADD44]], [[ADD58]]
+; CHECK-NEXT: [[SUB8_3:%.*]] = sub nsw i32 [[SUB_1]], [[SUB7_1]]
; CHECK-NEXT: [[SHL_3:%.*]] = shl nsw i32 [[SUB67]], 16
; CHECK-NEXT: [[SHL_1:%.*]] = shl nsw i32 [[SUB8_1]], 16
-; CHECK-NEXT: [[ADD9_4:%.*]] = add nsw i32 [[ADD66]], [[SHL_3]]
-; CHECK-NEXT: [[ADD9_1:%.*]] = add nsw i32 [[ADD_1]], [[SHL_1]]
-; CHECK-NEXT: [[TMP28:%.*]] = load i8, ptr [[ARRAYIDX15_3]], align 1
-; CHECK-NEXT: [[TMP29:%.*]] = load i8, ptr [[ARRAYIDX10_3]], align 1
-; CHECK-NEXT: [[CONV11_3:%.*]] = zext i8 [[TMP29]] to i32
-; CHECK-NEXT: [[CONV11_1:%.*]] = zext i8 [[TMP21]] to i32
-; CHECK-NEXT: [[TMP30:%.*]] = load i8, ptr [[ARRAYIDX17_3]], align 1
-; CHECK-NEXT: [[TMP44:%.*]] = load i8, ptr [[ARRAYIDX12_3]], align 1
-; CHECK-NEXT: [[CONV13_3:%.*]] = zext i8 [[TMP44]] to i32
-; CHECK-NEXT: [[CONV13_1:%.*]] = zext i8 [[TMP23]] to i32
-; CHECK-NEXT: [[SUB14_4:%.*]] = sub nsw i32 [[CONV11_3]], [[CONV13_3]]
-; CHECK-NEXT: [[SUB14_1:%.*]] = sub nsw i32 [[CONV11_1]], [[CONV13_1]]
-; CHECK-NEXT: [[CONV16_3:%.*]] = zext i8 [[TMP28]] to i32
-; CHECK-NEXT: [[CONV16_1:%.*]] = zext i8 [[TMP20]] to i32
-; CHECK-NEXT: [[CONV18_4:%.*]] = zext i8 [[TMP30]] to i32
-; CHECK-NEXT: [[CONV18_1:%.*]] = zext i8 [[TMP22]] to i32
-; CHECK-NEXT: [[SUB19_4:%.*]] = sub nsw i32 [[CONV16_3]], [[CONV18_4]]
-; CHECK-NEXT: [[SUB19_1:%.*]] = sub nsw i32 [[CONV16_1]], [[CONV18_1]]
-; CHECK-NEXT: [[ADD20_4:%.*]] = add nsw i32 [[SUB19_4]], [[SUB14_4]]
-; CHECK-NEXT: [[ADD20_1:%.*]] = add nsw i32 [[SUB19_1]], [[SUB14_1]]
-; CHECK-NEXT: [[SUB21_4:%.*]] = sub nsw i32 [[SUB14_4]], [[SUB19_4]]
-; CHECK-NEXT: [[SUB21_1:%.*]] = sub nsw i32 [[SUB14_1]], [[SUB19_1]]
; CHECK-NEXT: [[SHL22_4:%.*]] = shl nsw i32 [[SUB21_4]], 16
; CHECK-NEXT: [[SHL22_1:%.*]] = shl nsw i32 [[SUB21_1]], 16
-; CHECK-NEXT: [[ADD23_4:%.*]] = add nsw i32 [[ADD20_4]], [[SHL22_4]]
-; CHECK-NEXT: [[ADD23_1:%.*]] = add nsw i32 [[ADD20_1]], [[SHL22_1]]
-; CHECK-NEXT: [[ADD24_2:%.*]] = add nsw i32 [[ADD23_2]], [[ADD9_2]]
-; CHECK-NEXT: [[ADD24:%.*]] = add nsw i32 [[ADD23]], [[ADD24_3]]
-; CHECK-NEXT: [[SUB27_2:%.*]] = sub nsw i32 [[ADD9_2]], [[ADD23_2]]
-; CHECK-NEXT: [[SUB27:%.*]] = sub nsw i32 [[ADD24_3]], [[ADD23]]
+; CHECK-NEXT: [[ADD9_4:%.*]] = shl nsw i32 [[SUB21_2]], 16
+; CHECK-NEXT: [[ADD9_1:%.*]] = shl nsw i32 [[SUB8_2]], 16
+; CHECK-NEXT: [[SHL22_5:%.*]] = shl nsw i32 [[SUB21_3]], 16
+; CHECK-NEXT: [[SHL_4:%.*]] = shl nsw i32 [[SUB8_3]], 16
+; CHECK-NEXT: [[ADD59:%.*]] = add nsw i32 [[ADD9_2]], [[SHL_3]]
+; CHECK-NEXT: [[ADD45:%.*]] = add nsw i32 [[ADD24_3]], [[SHL_1]]
+; CHECK-NEXT: [[ADD23_2:%.*]] = add nsw i32 [[ADD20_2]], [[SHL22_4]]
+; CHECK-NEXT: [[ADD9_5:%.*]] = add nsw i32 [[ADD20]], [[SHL22_1]]
; CHECK-NEXT: [[ADD24_4:%.*]] = add nsw i32 [[ADD23_4]], [[ADD9_4]]
; CHECK-NEXT: [[ADD24_1:%.*]] = add nsw i32 [[ADD23_1]], [[ADD9_1]]
-; CHECK-NEXT: [[SUB27_3:%.*]] = sub nsw i32 [[ADD9_4]], [[ADD23_4]]
-; CHECK-NEXT: [[SUB27_1:%.*]] = sub nsw i32 [[ADD9_1]], [[ADD23_1]]
-; CHECK-NEXT: [[SUB51:%.*]] = sub nsw i32 [[ADD24]], [[ADD24_1]]
-; CHECK-NEXT: [[ADD59:%.*]] = add nsw i32 [[ADD24_4]], [[ADD24_2]]
-; CHECK-NEXT: [[ADD45:%.*]] = add nsw i32 [[ADD24_1]], [[ADD24]]
-; CHECK-NEXT: [[SUB65:%.*]] = sub nsw i32 [[ADD24_2]], [[ADD24_4]]
+; CHECK-NEXT: [[ADD23_5:%.*]] = add nsw i32 [[ADD66]], [[SHL22_5]]
+; CHECK-NEXT: [[ADD9_6:%.*]] = add nsw i32 [[ADD_1]], [[SHL_4]]
; CHECK-NEXT: [[TMP32:%.*]] = insertelement <2 x i32> poison, i32 [[ADD45]], i64 0
; CHECK-NEXT: [[TMP33:%.*]] = shufflevector <2 x i32> [[TMP32]], <2 x i32> poison, <2 x i32> zeroinitializer
; CHECK-NEXT: [[TMP34:%.*]] = insertelement <2 x i32> poison, i32 [[ADD59]], i64 0
; CHECK-NEXT: [[TMP35:%.*]] = shufflevector <2 x i32> [[TMP34]], <2 x i32> poison, <2 x i32> zeroinitializer
+; CHECK-NEXT: [[TMP51:%.*]] = sub nsw <2 x i32> [[TMP33]], [[TMP35]]
; CHECK-NEXT: [[TMP36:%.*]] = add nsw <2 x i32> [[TMP33]], [[TMP35]]
-; CHECK-NEXT: [[TMP37:%.*]] = sub nsw <2 x i32> [[TMP33]], [[TMP35]]
-; CHECK-NEXT: [[TMP38:%.*]] = shufflevector <2 x i32> [[TMP36]], <2 x i32> [[TMP37]], <2 x i32> <i32 0, i32 3>
-; CHECK-NEXT: [[ADD68:%.*]] = add nsw i32 [[SUB65]], [[SUB51]]
-; CHECK-NEXT: [[SUB69:%.*]] = sub nsw i32 [[SUB51]], [[SUB65]]
-; CHECK-NEXT: [[SHR_I126:%.*]] = lshr i32 [[ADD68]], 15
-; CHECK-NEXT: [[AND_I127:%.*]] = and i32 [[SHR_I126]], 65537
-; CHECK-NEXT: [[MUL_I128:%.*]] = mul nuw i32 [[AND_I127]], 65535
-; CHECK-NEXT: [[ADD_I129:%.*]] = add i32 [[MUL_I128]], [[ADD68]]
-; CHECK-NEXT: [[XOR_I130:%.*]] = xor i32 [[ADD_I129]], [[MUL_I128]]
-; CHECK-NEXT: [[TMP39:%.*]] = lshr <2 x i32> [[TMP38]], splat (i32 15)
-; CHECK-NEXT: [[TMP40:%.*]] = and <2 x i32> [[TMP39]], splat (i32 65537)
-; CHECK-NEXT: [[TMP41:%.*]] = mul nuw <2 x i32> [[TMP40]], splat (i32 65535)
-; CHECK-NEXT: [[TMP42:%.*]] = add <2 x i32> [[TMP41]], [[TMP38]]
-; CHECK-NEXT: [[TMP43:%.*]] = xor <2 x i32> [[TMP42]], [[TMP41]]
-; CHECK-NEXT: [[XOR_I135:%.*]] = extractelement <2 x i32> [[TMP43]], i64 0
-; CHECK-NEXT: [[ADD71:%.*]] = add i32 [[XOR_I135]], [[XOR_I130]]
-; CHECK-NEXT: [[XOR_I125:%.*]] = extractelement <2 x i32> [[TMP43]], i64 1
-; CHECK-NEXT: [[ADD73:%.*]] = add i32 [[ADD71]], [[XOR_I125]]
-; CHECK-NEXT: [[SHR_I:%.*]] = lshr i32 [[SUB69]], 15
-; CHECK-NEXT: [[AND_I:%.*]] = and i32 [[SHR_I]], 65537
-; CHECK-NEXT: [[MUL_I:%.*]] = mul nuw i32 [[AND_I]], 65535
-; CHECK-NEXT: [[ADD_I:%.*]] = add i32 [[MUL_I]], [[SUB69]]
-; CHECK-NEXT: [[XOR_I:%.*]] = xor i32 [[ADD_I]], [[MUL_I]]
-; CHECK-NEXT: [[ADD75:%.*]] = add i32 [[ADD73]], [[XOR_I]]
-; CHECK-NEXT: [[CONV77:%.*]] = and i32 [[ADD75]], 65535
-; CHECK-NEXT: [[SHR:%.*]] = lshr i32 [[ADD75]], 16
-; CHECK-NEXT: [[ADD79:%.*]] = add nuw nsw i32 [[SHR]], [[CONV77]]
-; CHECK-NEXT: [[SUB51_1:%.*]] = sub nsw i32 [[SUB27]], [[SUB27_1]]
-; CHECK-NEXT: [[ADD58_1:%.*]] = add nsw i32 [[SUB27_3]], [[SUB27_2]]
-; CHECK-NEXT: [[ADD44_1:%.*]] = add nsw i32 [[SUB27_1]], [[SUB27]]
-; CHECK-NEXT: [[SUB65_1:%.*]] = sub nsw i32 [[SUB27_2]], [[SUB27_3]]
-; CHECK-NEXT: [[TMP46:%.*]] = insertelement <2 x i32> poison, i32 [[ADD44_1]], i64 0
+; CHECK-NEXT: [[TMP38:%.*]] = shufflevector <2 x i32> [[TMP51]], <2 x i32> [[TMP36]], <2 x i32> <i32 0, i32 3>
+; CHECK-NEXT: [[TMP39:%.*]] = insertelement <2 x i32> poison, i32 [[ADD9_5]], i64 0
+; CHECK-NEXT: [[TMP40:%.*]] = shufflevector <2 x i32> [[TMP39]], <2 x i32> poison, <2 x i32> zeroinitializer
+; CHECK-NEXT: [[TMP41:%.*]] = insertelement <2 x i32> poison, i32 [[ADD23_2]], i64 0
+; CHECK-NEXT: [[TMP42:%.*]] = shufflevector <2 x i32> [[TMP41]], <2 x i32> poison, <2 x i32> zeroinitializer
+; CHECK-NEXT: [[TMP43:%.*]] = sub nsw <2 x i32> [[TMP40]], [[TMP42]]
+; CHECK-NEXT: [[TMP44:%.*]] = add nsw <2 x i32> [[TMP40]], [[TMP42]]
+; CHECK-NEXT: [[TMP45:%.*]] = shufflevector <2 x i32> [[TMP43]], <2 x i32> [[TMP44]], <2 x i32> <i32 0, i32 3>
+; CHECK-NEXT: [[TMP46:%.*]] = insertelement <2 x i32> poison, i32 [[ADD24_1]], i64 0
; CHECK-NEXT: [[TMP47:%.*]] = shufflevector <2 x i32> [[TMP46]], <2 x i32> poison, <2 x i32> zeroinitializer
-; CHECK-NEXT: [[TMP48:%.*]] = insertelement <2 x i32> poison, i32 [[ADD58_1]], i64 0
+; CHECK-NEXT: [[TMP48:%.*]] = insertelement <2 x i32> poison, i32 [[ADD24_4]], i64 0
; CHECK-NEXT: [[TMP49:%.*]] = shufflevector <2 x i32> [[TMP48]], <2 x i32> poison, <2 x i32> zeroinitializer
+; CHECK-NEXT: [[TMP64:%.*]] = sub nsw <2 x i32> [[TMP47]], [[TMP49]]
; CHECK-NEXT: [[TMP50:%.*]] = add nsw <2 x i32> [[TMP47]], [[TMP49]]
-; CHECK-NEXT: [[TMP51:%.*]] = sub nsw <2 x i32> [[TMP47]], [[TMP49]]
-; CHECK-NEXT: [[TMP52:%.*]] = shufflevector <2 x i32> [[TMP50]], <2 x i32> [[TMP51]], <2 x i32> <i32 0, i32 3>
-; CHECK-NEXT: [[ADD68_1:%.*]] = add nsw i32 [[SUB65_1]], [[SUB51_1]]
-; CHECK-NEXT: [[SUB69_1:%.*]] = sub nsw i32 [[SUB51_1]], [[SUB65_1]]
-; CHECK-NEXT: [[SHR_I126_1:%.*]] = lshr i32 [[ADD68_1]], 15
-; CHECK-NEXT: [[AND_I127_1:%.*]] = and i32 [[SHR_I126_1]], 65537
-; CHECK-NEXT: [[MUL_I128_1:%.*]] = mul nuw i32 [[AND_I127_1]], 65535
-; CHECK-NEXT: [[ADD_I129_1:%.*]] = add i32 [[MUL_I128_1]], [[ADD68_1]]
-; CHECK-NEXT: [[XOR_I130_1:%.*]] = xor i32 [[ADD_I129_1]], [[MUL_I128_1]]
+; CHECK-NEXT: [[TMP68:%.*]] = shufflevector <2 x i32> [[TMP64]], <2 x i32> [[TMP50]], <2 x i32> <i32 0, i32 3>
+; CHECK-NEXT: [[TMP69:%.*]] = insertelement <2 x i32> poison, i32 [[ADD9_6]], i64 0
+; CHECK-NEXT: [[TMP70:%.*]] = shufflevector <2 x i32> [[TMP69]], <2 x i32> poison, <2 x i32> zeroinitializer
+; CHECK-NEXT: [[TMP71:%.*]] = insertelement <2 x i32> poison, i32 [[ADD23_5]], i64 0
+; CHECK-NEXT: [[TMP72:%.*]] = shufflevector <2 x i32> [[TMP71]], <2 x i32> poison, <2 x i32> zeroinitializer
+; CHECK-NEXT: [[TMP92:%.*]] = sub nsw <2 x i32> [[TMP70]], [[TMP72]]
+; CHECK-NEXT: [[TMP58:%.*]] = add nsw <2 x i32> [[TMP70]], [[TMP72]]
+; CHECK-NEXT: [[TMP59:%.*]] = shufflevector <2 x i32> [[TMP92]], <2 x i32> [[TMP58]], <2 x i32> <i32 0, i32 3>
+; CHECK-NEXT: [[TMP60:%.*]] = add nsw <2 x i32> [[TMP45]], [[TMP38]]
+; CHECK-NEXT: [[TMP61:%.*]] = sub nsw <2 x i32> [[TMP38]], [[TMP45]]
+; CHECK-NEXT: [[TMP62:%.*]] = add nsw <2 x i32> [[TMP59]], [[TMP68]]
+; CHECK-NEXT: [[TMP63:%.*]] = sub nsw <2 x i32> [[TMP68]], [[TMP59]]
+; CHECK-NEXT: [[TMP52:%.*]] = add nsw <2 x i32> [[TMP62]], [[TMP60]]
+; CHECK-NEXT: [[TMP65:%.*]] = sub nsw <2 x i32> [[TMP60]], [[TMP62]]
+; CHECK-NEXT: [[TMP66:%.*]] = add nsw <2 x i32> [[TMP63]], [[TMP61]]
+; CHECK-NEXT: [[TMP67:%.*]] = sub nsw <2 x i32> [[TMP61]], [[TMP63]]
; CHECK-NEXT: [[TMP53:%.*]] = lshr <2 x i32> [[TMP52]], splat (i32 15)
; CHECK-NEXT: [[TMP54:%.*]] = and <2 x i32> [[TMP53]], splat (i32 65537)
; CHECK-NEXT: [[TMP55:%.*]] = mul nuw <2 x i32> [[TMP54]], splat (i32 65535)
; CHECK-NEXT: [[TMP56:%.*]] = add <2 x i32> [[TMP55]], [[TMP52]]
; CHECK-NEXT: [[TMP57:%.*]] = xor <2 x i32> [[TMP56]], [[TMP55]]
-; CHECK-NEXT: [[XOR_I135_1:%.*]] = extractelement <2 x i32> [[TMP57]], i64 0
-; CHECK-NEXT: [[ADD71_1:%.*]] = add i32 [[XOR_I135_1]], [[XOR_I130_1]]
-; CHECK-NEXT: [[XOR_I125_1:%.*]] = extractelement <2 x i32> [[TMP57]], i64 1
-; CHECK-NEXT: [[ADD73_1:%.*]] = add i32 [[ADD71_1]], [[XOR_I125_1]]
-; CHECK-NEXT: [[SHR_I_1:%.*]] = lshr i32 [[SUB69_1]], 15
-; CHECK-NEXT: [[AND_I_1:%.*]] = and i32 [[SHR_I_1]], 65537
-; CHECK-NEXT: [[MUL_I_1:%.*]] = mul nuw i32 [[AND_I_1]], 65535
-; CHECK-NEXT: [[ADD_I_1:%.*]] = add i32 [[MUL_I_1]], [[SUB69_1]]
-; CHECK-NEXT: [[XOR_I_1:%.*]] = xor i32 [[ADD_I_1]], [[MUL_I_1]]
-; CHECK-NEXT: [[ADD75_1:%.*]] = add i32 [[ADD73_1]], [[XOR_I_1]]
-; CHECK-NEXT: [[CONV77_1:%.*]] = and i32 [[ADD75_1]], 65535
-; CHECK-NEXT: [[SHR_1:%.*]] = lshr i32 [[ADD75_1]], 16
+; CHECK-NEXT: [[TMP73:%.*]] = lshr <2 x i32> [[TMP66]], splat (i32 15)
+; CHECK-NEXT: [[TMP74:%.*]] = and <2 x i32> [[TMP73]], splat (i32 65537)
+; CHECK-NEXT: [[TMP75:%.*]] = mul nuw <2 x i32> [[TMP74]], splat (i32 65535)
+; CHECK-NEXT: [[TMP76:%.*]] = add <2 x i32> [[TMP75]], [[TMP66]]
+; CHECK-NEXT: [[TMP77:%.*]] = xor <2 x i32> [[TMP76]], [[TMP75]]
+; CHECK-NEXT: [[TMP78:%.*]] = add <2 x i32> [[TMP57]], [[TMP77]]
+; CHECK-NEXT: [[TMP79:%.*]] = lshr <2 x i32> [[TMP65]], splat (i32 15)
+; CHECK-NEXT: [[TMP80:%.*]] = and <2 x i32> [[TMP79]], splat (i32 65537)
+; CHECK-NEXT: [[TMP81:%.*]] = mul nuw <2 x i32> [[TMP80]], splat (i32 65535)
+; CHECK-NEXT: [[TMP82:%.*]] = add <2 x i32> [[TMP81]], [[TMP65]]
+; CHECK-NEXT: [[TMP83:%.*]] = xor <2 x i32> [[TMP82]], [[TMP81]]
+; CHECK-NEXT: [[TMP84:%.*]] = add <2 x i32> [[TMP78]], [[TMP83]]
+; CHECK-NEXT: [[TMP85:%.*]] = lshr <2 x i32> [[TMP67]], splat (i32 15)
+; CHECK-NEXT: [[TMP86:%.*]] = and <2 x i32> [[TMP85]], splat (i32 65537)
+; CHECK-NEXT: [[TMP87:%.*]] = mul nuw <2 x i32> [[TMP86]], splat (i32 65535)
+; CHECK-NEXT: [[TMP88:%.*]] = add <2 x i32> [[TMP87]], [[TMP67]]
+; CHECK-NEXT: [[TMP89:%.*]] = xor <2 x i32> [[TMP88]], [[TMP87]]
+; CHECK-NEXT: [[TMP90:%.*]] = add <2 x i32> [[TMP84]], [[TMP89]]
+; CHECK-NEXT: [[TMP91:%.*]] = lshr <2 x i32> [[TMP90]], splat (i32 16)
+; CHECK-NEXT: [[SHR_1:%.*]] = extractelement <2 x i32> [[TMP91]], i64 1
+; CHECK-NEXT: [[TMP93:%.*]] = and <2 x i32> [[TMP90]], splat (i32 65535)
+; CHECK-NEXT: [[ADD79:%.*]] = extractelement <2 x i32> [[TMP93]], i64 1
; CHECK-NEXT: [[ADD78_1:%.*]] = add nuw nsw i32 [[SHR_1]], [[ADD79]]
-; CHECK-NEXT: [[ADD79_1:%.*]] = add nuw nsw i32 [[ADD78_1]], [[CONV77_1]]
+; CHECK-NEXT: [[TMP95:%.*]] = extractelement <2 x i32> [[TMP91]], i64 0
+; CHECK-NEXT: [[ADD78_2:%.*]] = add nuw nsw i32 [[TMP95]], [[ADD78_1]]
+; CHECK-NEXT: [[TMP96:%.*]] = extractelement <2 x i32> [[TMP93]], i64 0
+; CHECK-NEXT: [[ADD79_1:%.*]] = add nuw nsw i32 [[ADD78_2]], [[TMP96]]
; CHECK-NEXT: [[SHR83:%.*]] = lshr i32 [[ADD79_1]], 1
; CHECK-NEXT: ret i32 [[SHR83]]
;
diff --git a/llvm/test/Transforms/SLPVectorizer/AArch64/splat-gather-subtree-trim-revert-cost.ll b/llvm/test/Transforms/SLPVectorizer/AArch64/splat-gather-subtree-trim-revert-cost.ll
index 3ee07e04acc84..b08bae2711238 100644
--- a/llvm/test/Transforms/SLPVectorizer/AArch64/splat-gather-subtree-trim-revert-cost.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AArch64/splat-gather-subtree-trim-revert-cost.ll
@@ -12,24 +12,22 @@ define void @splat_subtree_trim_revert_cost(ptr %p, double %x, double %y) {
; CHECK-LABEL: define void @splat_subtree_trim_revert_cost(
; CHECK-SAME: ptr [[P:%.*]], double [[X:%.*]], double [[Y:%.*]]) {
; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[TMP0:%.*]] = insertelement <2 x double> <double 0.000000e+00, double poison>, double [[Y]], i64 1
+; CHECK-NEXT: [[TMP1:%.*]] = insertelement <2 x double> poison, double [[X]], i64 1
; CHECK-NEXT: br label %[[LOOP:.*]]
; CHECK: [[LOOP]]:
-; CHECK-NEXT: [[A0:%.*]] = tail call double @llvm.fmuladd.f64(double 0.000000e+00, double 0.000000e+00, double 0.000000e+00)
-; CHECK-NEXT: [[A1:%.*]] = tail call double @llvm.fmuladd.f64(double 0.000000e+00, double 0.000000e+00, double 0.000000e+00)
-; CHECK-NEXT: [[B0:%.*]] = tail call double @llvm.fmuladd.f64(double 0.000000e+00, double 0.000000e+00, double [[A1]])
; CHECK-NEXT: [[A2:%.*]] = tail call double @llvm.fmuladd.f64(double 0.000000e+00, double 0.000000e+00, double 0.000000e+00)
-; CHECK-NEXT: [[M0:%.*]] = fmul double 0.000000e+00, [[A0]]
-; CHECK-NEXT: [[C0:%.*]] = tail call double @llvm.fmuladd.f64(double [[A2]], double 0.000000e+00, double [[M0]])
-; CHECK-NEXT: [[D0:%.*]] = tail call double @llvm.fmuladd.f64(double 0.000000e+00, double [[B0]], double [[C0]])
-; CHECK-NEXT: [[E0:%.*]] = tail call double @llvm.fmuladd.f64(double 0.000000e+00, double [[D0]], double 0.000000e+00)
+; CHECK-NEXT: [[TMP2:%.*]] = call <2 x double> @llvm.fmuladd.v2f64(<2 x double> zeroinitializer, <2 x double> zeroinitializer, <2 x double> zeroinitializer)
+; CHECK-NEXT: [[M0:%.*]] = fmul double 0.000000e+00, [[A2]]
; CHECK-NEXT: [[GEP0:%.*]] = getelementptr i8, ptr [[P]], i64 736
-; CHECK-NEXT: store double [[E0]], ptr [[GEP0]], align 8
-; CHECK-NEXT: [[B1:%.*]] = tail call double @llvm.fmuladd.f64(double 0.000000e+00, double 0.000000e+00, double [[A1]])
-; CHECK-NEXT: [[C1:%.*]] = tail call double @llvm.fmuladd.f64(double [[A2]], double [[Y]], double [[X]])
-; CHECK-NEXT: [[D1:%.*]] = tail call double @llvm.fmuladd.f64(double 0.000000e+00, double [[B1]], double [[C1]])
-; CHECK-NEXT: [[E1:%.*]] = tail call double @llvm.fmuladd.f64(double 0.000000e+00, double [[D1]], double 0.000000e+00)
-; CHECK-NEXT: [[GEP1:%.*]] = getelementptr i8, ptr [[P]], i64 744
-; CHECK-NEXT: store double [[E1]], ptr [[GEP1]], align 8
+; CHECK-NEXT: [[TMP3:%.*]] = shufflevector <2 x double> [[TMP2]], <2 x double> poison, <2 x i32> zeroinitializer
+; CHECK-NEXT: [[TMP4:%.*]] = call <2 x double> @llvm.fmuladd.v2f64(<2 x double> zeroinitializer, <2 x double> zeroinitializer, <2 x double> [[TMP3]])
+; CHECK-NEXT: [[TMP5:%.*]] = shufflevector <2 x double> [[TMP2]], <2 x double> poison, <2 x i32> <i32 1, i32 1>
+; CHECK-NEXT: [[TMP6:%.*]] = insertelement <2 x double> [[TMP1]], double [[M0]], i64 0
+; CHECK-NEXT: [[TMP7:%.*]] = call <2 x double> @llvm.fmuladd.v2f64(<2 x double> [[TMP5]], <2 x double> [[TMP0]], <2 x double> [[TMP6]])
+; CHECK-NEXT: [[TMP8:%.*]] = call <2 x double> @llvm.fmuladd.v2f64(<2 x double> zeroinitializer, <2 x double> [[TMP4]], <2 x double> [[TMP7]])
+; CHECK-NEXT: [[TMP9:%.*]] = call <2 x double> @llvm.fmuladd.v2f64(<2 x double> zeroinitializer, <2 x double> [[TMP8]], <2 x double> zeroinitializer)
+; CHECK-NEXT: store <2 x double> [[TMP9]], ptr [[GEP0]], align 8
; CHECK-NEXT: br label %[[LOOP]]
;
entry:
More information about the llvm-commits
mailing list