[llvm] [SLP]Drop unprofitable splat subtrees before the profitability trim (PR #224470)

Alexey Bataev via llvm-commits llvm-commits at lists.llvm.org
Mon Sep 21 05:00:51 PDT 2026


https://github.com/alexey-bataev updated https://github.com/llvm/llvm-project/pull/224470

>From a4391cb54ec0d8291c8ef4567b2d2f32ea4c7fd9 Mon Sep 17 00:00:00 2001
From: Alexey Bataev <a.bataev at outlook.com>
Date: Thu, 17 Sep 2026 16:06:40 -0700
Subject: [PATCH] =?UTF-8?q?[=F0=9D=98=80=F0=9D=97=BD=F0=9D=97=BF]=20initia?=
 =?UTF-8?q?l=20version?=
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit

Created using spr 1.3.7
---
 .../Transforms/Vectorize/SLPVectorizer.cpp    | 573 +++++++++---------
 .../splat-gather-subtree-loop-extracts.ll     |  15 +-
 .../AArch64/splat-gather-subtree-satd.ll      | 293 +++++----
 .../splat-gather-subtree-trim-revert-cost.ll  |  26 +-
 4 files changed, 463 insertions(+), 444 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 40952c66faf1a..a88f4813e8235 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -18946,47 +18946,249 @@ BoUpSLP::calculateTreeCostAndTrimNonProfitable(ArrayRef<Value *> VectorizedVals,
   SmallVector<
       std::tuple<InstructionCost, InstructionCost, SmallVector<unsigned>>>
       SubtreeCosts(VectorizableTree.size());
-  auto UpdateParentNodes =
-      [&](const TreeEntry *UserTE, const TreeEntry *TE,
-          InstructionCost TotalCost, InstructionCost Cost,
-          SmallDenseSet<std::pair<const TreeEntry *, const TreeEntry *>, 4>
-              &VisitedUser,
-          bool AddToList = true) {
-        while (UserTE &&
-               VisitedUser.insert(std::make_pair(TE, UserTE)).second) {
-          std::get<0>(SubtreeCosts[UserTE->Idx]) += TotalCost;
-          std::get<1>(SubtreeCosts[UserTE->Idx]) += Cost;
-          if (AddToList)
-            std::get<2>(SubtreeCosts[UserTE->Idx]).push_back(TE->Idx);
-          UserTE = UserTE->UserTreeIndex.UserTE;
+  // (Re)computes the subtree costs for the nodes that are not deleted.
+  auto ComputeSubtreeCosts = [&]() {
+    SubtreeCosts.assign(
+        VectorizableTree.size(),
+        std::tuple<InstructionCost, InstructionCost, SmallVector<unsigned>>());
+    auto UpdateParentNodes =
+        [&](const TreeEntry *UserTE, const TreeEntry *TE,
+            InstructionCost TotalCost, InstructionCost Cost,
+            SmallDenseSet<std::pair<const TreeEntry *, const TreeEntry *>, 4>
+                &VisitedUser,
+            bool AddToList = true) {
+          while (UserTE &&
+                 VisitedUser.insert(std::make_pair(TE, UserTE)).second) {
+            std::get<0>(SubtreeCosts[UserTE->Idx]) += TotalCost;
+            std::get<1>(SubtreeCosts[UserTE->Idx]) += Cost;
+            if (AddToList)
+              std::get<2>(SubtreeCosts[UserTE->Idx]).push_back(TE->Idx);
+            UserTE = UserTE->UserTreeIndex.UserTE;
+          }
+        };
+    for (const std::unique_ptr<TreeEntry> &Ptr : VectorizableTree) {
+      TreeEntry &TE = *Ptr;
+      if (DeletedNodes.contains(&TE))
+        continue;
+      // Combined subnodes are not costed on their own (their cost is 0), but
+      // must be included into the ancestors' subtree node lists, so that they
+      // get deleted together with the trimmed combined root.
+      InstructionCost C = NodesCosts.at(&TE);
+      InstructionCost ExtractCost = ExtractCosts.lookup(&TE);
+      std::get<0>(SubtreeCosts[TE.Idx]) += C + ExtractCost;
+      std::get<1>(SubtreeCosts[TE.Idx]) += C;
+      if (const TreeEntry *UserTE = TE.UserTreeIndex.UserTE) {
+        SmallDenseSet<std::pair<const TreeEntry *, const TreeEntry *>, 4>
+            VisitedUser;
+        UpdateParentNodes(UserTE, &TE, C + ExtractCost, C, VisitedUser);
+      }
+    }
+    SmallDenseSet<std::pair<const TreeEntry *, const TreeEntry *>, 4> Visited;
+    for (TreeEntry *TE : GatheredLoadsNodes) {
+      if (DeletedNodes.contains(TE))
+        continue;
+      InstructionCost TotalCost = std::get<0>(SubtreeCosts[TE->Idx]);
+      InstructionCost Cost = std::get<1>(SubtreeCosts[TE->Idx]);
+      for (Value *V : TE->Scalars) {
+        for (const TreeEntry *BVTE : ValueToGatherNodes.lookup(V)) {
+          if (DeletedNodes.contains(BVTE))
+            continue;
+          UpdateParentNodes(BVTE, TE, TotalCost, Cost, Visited,
+                            /*AddToList=*/false);
         }
-      };
-  for (const std::unique_ptr<TreeEntry> &Ptr : VectorizableTree) {
-    TreeEntry &TE = *Ptr;
-    // Combined subnodes are not costed on their own (their cost is 0), but
-    // must be included into the ancestors' subtree node lists, so that they
-    // get deleted together with the trimmed combined root.
-    InstructionCost C = NodesCosts.at(&TE);
-    InstructionCost ExtractCost = ExtractCosts.lookup(&TE);
-    std::get<0>(SubtreeCosts[TE.Idx]) += C + ExtractCost;
-    std::get<1>(SubtreeCosts[TE.Idx]) += C;
-    if (const TreeEntry *UserTE = TE.UserTreeIndex.UserTE) {
-      SmallDenseSet<std::pair<const TreeEntry *, const TreeEntry *>, 4>
-          VisitedUser;
-      UpdateParentNodes(UserTE, &TE, C + ExtractCost, C, VisitedUser);
-    }
-  }
-  SmallDenseSet<std::pair<const TreeEntry *, const TreeEntry *>, 4> Visited;
-  for (TreeEntry *TE : GatheredLoadsNodes) {
-    InstructionCost TotalCost = std::get<0>(SubtreeCosts[TE->Idx]);
-    InstructionCost Cost = std::get<1>(SubtreeCosts[TE->Idx]);
+      }
+    }
+  };
+  ComputeSubtreeCosts();
+  using ValuesToInsertTy =
+      SmallDenseMap<const TreeEntry *, SmallVector<Value *>>;
+  auto GetScalarTy = [&](const TreeEntry *TE) {
+    Type *ScalarTy = TE->Scalars.front()->getType();
+    auto It = MinBWs.find(TE);
+    if (It != MinBWs.end())
+      ScalarTy = IntegerType::get(ScalarTy->getContext(), It->second.first);
+    return ScalarTy;
+  };
+  // Lanes of the subtree scalars used by the surviving gather nodes, and the
+  // values to materialize in those gathers if the subtree is deleted.
+  auto FindDemandedElts = [&](TreeEntry *TE, ValuesToInsertTy &ValuesToInsert) {
+    APInt DemandedElts = APInt::getZero(TE->getVectorFactor());
+    for (Value *V : TE->Scalars) {
+      unsigned Pos = TE->findLaneForValue(V);
+      for (const TreeEntry *BVE : ValueToGatherNodes.lookup(V)) {
+        if (DeletedNodes.contains(BVE))
+          continue;
+        DemandedElts.setBit(Pos);
+        ValuesToInsert.try_emplace(BVE).first->second.push_back(V);
+      }
+    }
+    return DemandedElts;
+  };
+  // Cost of materializing the values directly in the surviving gather nodes
+  // that use them.
+  auto GetGatherInsertCost = [&](Type *ScalarTy,
+                                 const ValuesToInsertTy &ValuesToInsert) {
+    InstructionCost BVCost = 0;
+    for (const auto &[BVE, Values] : ValuesToInsert) {
+      APInt BVDemandedElts = APInt::getZero(BVE->getVectorFactor());
+      SmallVector<Value *> BVValues(BVE->getVectorFactor(),
+                                    PoisonValue::get(ScalarTy));
+      for (Value *V : Values) {
+        unsigned Pos = BVE->findLaneForValue(V);
+        BVValues[Pos] = V;
+        BVDemandedElts.setBit(Pos);
+      }
+      BVCost += getScalarizationOverhead(
+          *TTI, SLPReVec, ScalarTy,
+          cast<VectorType>(getWidenedType(ScalarTy, BVE->getVectorFactor())),
+          BVDemandedElts, /*Insert=*/true, /*Extract=*/false, CostKind,
+          BVDemandedElts.isAllOnes(), BVValues);
+    }
+    return BVCost;
+  };
+  auto RecostEntry = [&](const TreeEntry *TE) {
+    InstructionCost C = getEntryCost(TE, VectorizedVals, CheckedExtracts);
+    if (!C.isValid() || C == 0)
+      return C;
+    uint64_t Scale = EntryToScale.lookup(TE);
+    if (!Scale)
+      Scale = getEntryEffectiveScale(*TE);
+    return C * Scale;
+  };
+  // A splat subtree pays off only if some surviving gather node reuses it and
+  // its price plus the extracts of its scalars used by the remaining scalar
+  // code plus the current cost of the gather nodes that reuse it is not worse
+  // than re-emitting those gathers without the subtree.
+  auto IsSplatSubtreeProfitable = [&](TreeEntry *TE,
+                                      ValuesToInsertTy &ValuesToInsert,
+                                      InstructionCost &CurrentGathersCost,
+                                      InstructionCost &DroppedGathersCost) {
+    if (FindDemandedElts(TE, ValuesToInsert).isZero())
+      return false;
+    APInt ExtractElts = APInt::getZero(TE->getVectorFactor());
     for (Value *V : TE->Scalars) {
-      for (const TreeEntry *BVTE : ValueToGatherNodes.lookup(V))
-        UpdateParentNodes(BVTE, TE, TotalCost, Cost, Visited,
-                          /*AddToList=*/false);
+      if (!isa<Instruction>(V) || TE->isCopyableElement(V))
+        continue;
+      // Too many users - the scalar is extracted anyway.
+      if (V->hasNUsesOrMore(UsesLimit) || any_of(V->users(), [&](User *U) {
+            return none_of(getTreeEntries(U), [&](const TreeEntry *UseTE) {
+              return !DeletedNodes.contains(UseTE) &&
+                     !TransformedToGatherNodes.contains(UseTE);
+            });
+          }))
+        ExtractElts.setBit(TE->findLaneForValue(V));
+    }
+    Type *ScalarTy = GetScalarTy(TE);
+    InstructionCost KeepCost = getScalarizationOverhead(
+        *TTI, SLPReVec, ScalarTy,
+        cast<VectorType>(getWidenedType(ScalarTy, TE->getVectorFactor())),
+        ExtractElts, /*Insert=*/false, /*Extract=*/true, CostKind);
+    // Scale the extract cost to the subtree's execution frequency: the
+    // subtree and gather costs it is compared against are already loop-scaled.
+    if (KeepCost.isValid() && KeepCost != 0)
+      KeepCost *= getEntryEffectiveScale(*TE);
+    // Add the cost of the subtree itself, computed before any trimming:
+    // trimming of the subtree's own nodes would otherwise make it look
+    // artificially cheap. The inner nodes of the subtree keep their external
+    // scalar users too, so their extracts are part of the price as well.
+    KeepCost += std::get<1>(SubtreeCosts[TE->Idx]);
+    for (unsigned Idx : std::get<2>(SubtreeCosts[TE->Idx]))
+      KeepCost += ExtractCosts.lookup(VectorizableTree[Idx].get());
+    // The reusing gather nodes currently pay the broadcast cost; without
+    // the subtree they fall back to plain insertion sequences. Gather
+    // nodes already erased from NodesCosts are being deleted and do not
+    // count on either side.
+    CurrentGathersCost = 0;
+    for (const auto &[BVE, _] : ValuesToInsert)
+      CurrentGathersCost += NodesCosts.lookup(BVE);
+    KeepCost += CurrentGathersCost;
+    // Re-cost the gather nodes with the subtree tentatively deleted.
+    DeletedNodes.insert(TE);
+    SmallVector<TreeEntry *> TempDeleted;
+    for (unsigned Idx : std::get<2>(SubtreeCosts[TE->Idx])) {
+      TreeEntry *Child = VectorizableTree[Idx].get();
+      if (DeletedNodes.insert(Child).second)
+        TempDeleted.push_back(Child);
+    }
+    DroppedGathersCost = 0;
+    for (const auto &[BVE, _] : ValuesToInsert) {
+      if (!NodesCosts.contains(BVE))
+        continue;
+      DroppedGathersCost += RecostEntry(BVE);
+    }
+    DeletedNodes.erase(TE);
+    for (TreeEntry *Child : TempDeleted)
+      DeletedNodes.erase(Child);
+    // On a cost tie prefer dropping the subtree: the reusing gathers
+    // materialize the scalars at the same price with fewer instructions.
+    return KeepCost < DroppedGathersCost;
+  };
+  // Re-price the gathers that reused a dropped splat subtree, so the
+  // remaining subtrees are checked against the costs with it deleted.
+  auto SyncGatherCosts = [&](const ValuesToInsertTy &ValuesToInsert) {
+    for (const auto &[BVE, _] : ValuesToInsert)
+      if (!DeletedNodes.contains(BVE))
+        NodesCosts[BVE] = RecostEntry(BVE);
+  };
+  // Deletes the subtree. The per-node costs are erased eagerly only where the
+  // total is recomputed from them right after the deletion.
+  auto DeleteSubtree = [&](TreeEntry *TE, bool EraseCosts) {
+    DeletedNodes.insert(TE);
+    if (EraseCosts)
+      NodesCosts.erase(TE);
+    for (unsigned Idx : std::get<2>(SubtreeCosts[TE->Idx])) {
+      TreeEntry *Child = VectorizableTree[Idx].get();
+      DeletedNodes.insert(Child);
+      if (EraseCosts)
+        NodesCosts.erase(Child);
+    }
+  };
+  // Gathered loads subtrees left without surviving gather users are dead.
+  auto DropDeadGatheredLoads = [&](bool EraseCosts) {
+    for (TreeEntry *TE : GatheredLoadsNodes) {
+      if (DeletedNodes.contains(TE))
+        continue;
+      ValuesToInsertTy ValuesToInsert;
+      if (!FindDemandedElts(TE, ValuesToInsert).isZero())
+        continue;
+      DeleteSubtree(TE, EraseCosts);
     }
+  };
+  // Drops the splat subtrees that are unused by the surviving gathers or do
+  // not pay off; returns true if anything was dropped.
+  auto DropUnprofitableSplatSubtrees = [&](bool EraseCosts) {
+    bool Dropped = false;
+    for (TreeEntry *TE : SplatGatheredScalarsRoots) {
+      if (DeletedNodes.contains(TE))
+        continue;
+      ValuesToInsertTy ValuesToInsert;
+      InstructionCost CurrentGathersCost = 0, DroppedGathersCost = 0;
+      if (IsSplatSubtreeProfitable(TE, ValuesToInsert, CurrentGathersCost,
+                                   DroppedGathersCost))
+        continue;
+      DeleteSubtree(TE, EraseCosts);
+      SyncGatherCosts(ValuesToInsert);
+      Dropped = true;
+    }
+    return Dropped;
+  };
+  auto SumNodesCosts = [&]() {
+    InstructionCost C = 0;
+    for (const auto &P : NodesCosts)
+      C += P.second;
+    return C;
+  };
+  // The splat subtrees are speculative. The ones that do not pay off even in
+  // the full tree are dropped before the trimming: the reusing gathers are
+  // priced for the broadcast from the subtree vector while the subtree is
+  // around, which distorts the trimming decisions of the main tree, and fewer
+  // surviving gathers after the trimming can only make keeping a subtree less
+  // profitable.
+  if (DropUnprofitableSplatSubtrees(/*EraseCosts=*/true)) {
+    DropDeadGatheredLoads(/*EraseCosts=*/true);
+    ComputeSubtreeCosts();
+    Cost = SumNodesCosts();
   }
-  Visited.clear();
   using CostIndicesTy =
       std::pair<TreeEntry *, std::tuple<InstructionCost, InstructionCost,
                                         SmallVector<unsigned>>>;
@@ -19014,13 +19216,11 @@ BoUpSLP::calculateTreeCostAndTrimNonProfitable(ArrayRef<Value *> VectorizedVals,
       (Worklist.top().first->Idx == 0 || Worklist.top().first->Idx == 1))
     return Cost;
 
-  // Original gather costs before any trimming; used to rebase the reference
-  // cost on the recosted gathers if the trimming is reverted.
-  SmallDenseMap<const TreeEntry *, InstructionCost> OrigGatherCosts;
-  if (!SplatGatheredScalarsRoots.empty())
-    for (const std::unique_ptr<TreeEntry> &TE : VectorizableTree)
-      if (TE->isGather())
-        OrigGatherCosts.try_emplace(TE.get(), NodesCosts.lookup(TE.get()));
+  // Node costs and deleted nodes before any trimming; the restored tree is
+  // priced from them if the trimming is reverted.
+  const SmallDenseMap<const TreeEntry *, InstructionCost> OrigNodesCosts =
+      NodesCosts;
+  const SmallPtrSet<const TreeEntry *, 8> OrigDeletedNodes = DeletedNodes;
   bool Changed = false;
   bool PreferTrimmedTree = false;
   while (!Worklist.empty() && std::get<0>(Worklist.top().second) > 0) {
@@ -19194,131 +19394,6 @@ BoUpSLP::calculateTreeCostAndTrimNonProfitable(ArrayRef<Value *> VectorizedVals,
     }
     Worklist.pop();
   }
-  using ValuesToInsertTy =
-      SmallDenseMap<const TreeEntry *, SmallVector<Value *>>;
-  auto GetScalarTy = [&](const TreeEntry *TE) {
-    Type *ScalarTy = TE->Scalars.front()->getType();
-    auto It = MinBWs.find(TE);
-    if (It != MinBWs.end())
-      ScalarTy = IntegerType::get(ScalarTy->getContext(), It->second.first);
-    return ScalarTy;
-  };
-  // Lanes of the subtree scalars used by the surviving gather nodes, and the
-  // values to materialize in those gathers if the subtree is deleted.
-  auto FindDemandedElts = [&](TreeEntry *TE, ValuesToInsertTy &ValuesToInsert) {
-    APInt DemandedElts = APInt::getZero(TE->getVectorFactor());
-    for (Value *V : TE->Scalars) {
-      unsigned Pos = TE->findLaneForValue(V);
-      for (const TreeEntry *BVE : ValueToGatherNodes.lookup(V)) {
-        if (DeletedNodes.contains(BVE))
-          continue;
-        DemandedElts.setBit(Pos);
-        ValuesToInsert.try_emplace(BVE).first->second.push_back(V);
-      }
-    }
-    return DemandedElts;
-  };
-  // Cost of materializing the values directly in the surviving gather nodes
-  // that use them.
-  auto GetGatherInsertCost = [&](Type *ScalarTy,
-                                 const ValuesToInsertTy &ValuesToInsert) {
-    InstructionCost BVCost = 0;
-    for (const auto &[BVE, Values] : ValuesToInsert) {
-      APInt BVDemandedElts = APInt::getZero(BVE->getVectorFactor());
-      SmallVector<Value *> BVValues(BVE->getVectorFactor(),
-                                    PoisonValue::get(ScalarTy));
-      for (Value *V : Values) {
-        unsigned Pos = BVE->findLaneForValue(V);
-        BVValues[Pos] = V;
-        BVDemandedElts.setBit(Pos);
-      }
-      BVCost += getScalarizationOverhead(
-          *TTI, SLPReVec, ScalarTy,
-          cast<VectorType>(getWidenedType(ScalarTy, BVE->getVectorFactor())),
-          BVDemandedElts, /*Insert=*/true, /*Extract=*/false, CostKind,
-          BVDemandedElts.isAllOnes(), BVValues);
-    }
-    return BVCost;
-  };
-  auto RecostEntry = [&](const TreeEntry *TE) {
-    InstructionCost C = getEntryCost(TE, VectorizedVals, CheckedExtracts);
-    if (!C.isValid() || C == 0)
-      return C;
-    uint64_t Scale = EntryToScale.lookup(TE);
-    if (!Scale)
-      Scale = getEntryEffectiveScale(*TE);
-    return C * Scale;
-  };
-  // A splat subtree pays off only if its price plus the extracts of its
-  // scalars used by the remaining scalar code plus the current cost of the
-  // gather nodes that reuse it is not worse than re-emitting those gathers
-  // without the subtree.
-  auto IsSplatSubtreeProfitable = [&](TreeEntry *TE,
-                                      const ValuesToInsertTy &ValuesToInsert,
-                                      InstructionCost &CurrentGathersCost,
-                                      InstructionCost &DroppedGathersCost) {
-    APInt ExtractElts = APInt::getZero(TE->getVectorFactor());
-    for (Value *V : TE->Scalars) {
-      if (!isa<Instruction>(V) || TE->isCopyableElement(V))
-        continue;
-      // Too many users - the scalar is extracted anyway.
-      if (V->hasNUsesOrMore(UsesLimit) || any_of(V->users(), [&](User *U) {
-            return none_of(getTreeEntries(U), [&](const TreeEntry *UseTE) {
-              return !DeletedNodes.contains(UseTE) &&
-                     !TransformedToGatherNodes.contains(UseTE);
-            });
-          }))
-        ExtractElts.setBit(TE->findLaneForValue(V));
-    }
-    Type *ScalarTy = GetScalarTy(TE);
-    InstructionCost KeepCost = getScalarizationOverhead(
-        *TTI, SLPReVec, ScalarTy,
-        cast<VectorType>(getWidenedType(ScalarTy, TE->getVectorFactor())),
-        ExtractElts, /*Insert=*/false, /*Extract=*/true, CostKind);
-    // Scale the extract cost to the subtree's execution frequency: the
-    // subtree and gather costs it is compared against are already loop-scaled.
-    if (KeepCost.isValid() && KeepCost != 0)
-      KeepCost *= getEntryEffectiveScale(*TE);
-    // Add the cost of the subtree itself, computed before any trimming:
-    // trimming of the subtree's own nodes would otherwise make it look
-    // artificially cheap.
-    KeepCost += std::get<1>(SubtreeCosts[TE->Idx]);
-    // The reusing gather nodes currently pay the broadcast cost; without
-    // the subtree they fall back to plain insertion sequences. Gather
-    // nodes already erased from NodesCosts are being deleted and do not
-    // count on either side.
-    CurrentGathersCost = 0;
-    for (const auto &[BVE, _] : ValuesToInsert)
-      CurrentGathersCost += NodesCosts.lookup(BVE);
-    KeepCost += CurrentGathersCost;
-    // Re-cost the gather nodes with the subtree tentatively deleted.
-    DeletedNodes.insert(TE);
-    SmallVector<TreeEntry *> TempDeleted;
-    for (unsigned Idx : std::get<2>(SubtreeCosts[TE->Idx])) {
-      TreeEntry *Child = VectorizableTree[Idx].get();
-      if (DeletedNodes.insert(Child).second)
-        TempDeleted.push_back(Child);
-    }
-    DroppedGathersCost = 0;
-    for (const auto &[BVE, _] : ValuesToInsert) {
-      if (!NodesCosts.contains(BVE))
-        continue;
-      DroppedGathersCost += RecostEntry(BVE);
-    }
-    DeletedNodes.erase(TE);
-    for (TreeEntry *Child : TempDeleted)
-      DeletedNodes.erase(Child);
-    // On a cost tie prefer dropping the subtree: the reusing gathers
-    // materialize the scalars at the same price with fewer instructions.
-    return KeepCost < DroppedGathersCost;
-  };
-  // Re-price the gathers that reused a dropped splat subtree, so the
-  // remaining subtrees are checked against the costs with it deleted.
-  auto SyncGatherCosts = [&](const ValuesToInsertTy &ValuesToInsert) {
-    for (const auto &[BVE, _] : ValuesToInsert)
-      if (!DeletedNodes.contains(BVE))
-        NodesCosts[BVE] = RecostEntry(BVE);
-  };
   // The VF=2 instruction-count veto rejects the whole tree when the vector code
   // has more instructions than the scalar code. The auxiliary subtrees - splat
   // gather roots and gathered loads - are optional: the surviving gathers can
@@ -19355,9 +19430,7 @@ BoUpSLP::calculateTreeCostAndTrimNonProfitable(ArrayRef<Value *> VectorizedVals,
       InstructionCost CurrentGathersCost = 0;
       for (const auto &[BVE, _] : ValuesToInsert)
         CurrentGathersCost += NodesCosts.lookup(BVE);
-      DeletedNodes.insert(TE);
-      for (unsigned Idx : std::get<2>(SubtreeCosts[TE->Idx]))
-        DeletedNodes.insert(VectorizableTree[Idx].get());
+      DeleteSubtree(TE, /*EraseCosts=*/false);
       // All reusers were dropped: the subtree is dead. Without current node
       // costs the total conservatively keeps the subtree cost.
       if (IsDead) {
@@ -19404,39 +19477,45 @@ BoUpSLP::calculateTreeCostAndTrimNonProfitable(ArrayRef<Value *> VectorizedVals,
     // unprofitable ones instead of letting them reject the whole tree.
     InstructionCost TotalCost = std::get<1>(SubtreeCosts.front());
     for (TreeEntry *TE : SplatGatheredScalarsRoots) {
+      if (DeletedNodes.contains(TE))
+        continue;
       ValuesToInsertTy ValuesToInsert;
       InstructionCost CurrentGathersCost = 0, DroppedGathersCost = 0;
-      if (!FindDemandedElts(TE, ValuesToInsert).isZero() &&
-          IsSplatSubtreeProfitable(TE, ValuesToInsert, CurrentGathersCost,
+      if (IsSplatSubtreeProfitable(TE, ValuesToInsert, CurrentGathersCost,
                                    DroppedGathersCost)) {
         TotalCost += std::get<1>(SubtreeCosts[TE->Idx]);
         continue;
       }
-      DeletedNodes.insert(TE);
-      for (unsigned Idx : std::get<2>(SubtreeCosts[TE->Idx]))
-        DeletedNodes.insert(VectorizableTree[Idx].get());
+      DeleteSubtree(TE, /*EraseCosts=*/false);
       // The gather nodes that reused the subtree are re-emitted without it.
       TotalCost += DroppedGathersCost - CurrentGathersCost;
       SyncGatherCosts(ValuesToInsert);
     }
-    // Gathered loads subtrees left without surviving gather users are dead.
-    for (TreeEntry *TE : GatheredLoadsNodes) {
-      if (DeletedNodes.contains(TE))
-        continue;
-      ValuesToInsertTy ValuesToInsert;
-      if (!FindDemandedElts(TE, ValuesToInsert).isZero())
-        continue;
-      DeletedNodes.insert(TE);
-      for (unsigned Idx : std::get<2>(SubtreeCosts[TE->Idx]))
-        DeletedNodes.insert(VectorizableTree[Idx].get());
-    }
+    DropDeadGatheredLoads(/*EraseCosts=*/false);
     TrimAuxSubtreesForInstCount(TotalCost, /*UseCurrentNodeCosts=*/false);
     return TotalCost;
   }
 
   SmallPtrSet<TreeEntry *, 4> SubtreesToDelete;
-  SmallPtrSet<TreeEntry *, 4> DroppedSplatSubtrees;
   InstructionCost LoadsExtractsCost = 0;
+  // Check if all gather nodes that reuse the splat subtrees are marked for
+  // deletion. In this case the whole splat subtree must be deleted. If only
+  // some of the gathers are trimmed, keeping the subtree still costs its full
+  // price plus the extracts of the scalars used by the remaining scalar code,
+  // while the surviving gathers can materialize the splatted scalars
+  // directly. Drop the subtree if it does not pay off.
+  for (TreeEntry *TE : SplatGatheredScalarsRoots) {
+    if (DeletedNodes.contains(TE))
+      continue;
+    ValuesToInsertTy ValuesToInsert;
+    InstructionCost CurrentGathersCost, DroppedGathersCost;
+    if (IsSplatSubtreeProfitable(TE, ValuesToInsert, CurrentGathersCost,
+                                 DroppedGathersCost))
+      continue;
+    SubtreesToDelete.insert(TE);
+    NodesCosts.erase(TE);
+  }
+
   // Check if all loads of gathered loads nodes are marked for deletion. In this
   // case the whole gathered loads subtree must be deleted.
   // Also, try to account for extracts, which might be required, if only part of
@@ -19466,35 +19545,6 @@ BoUpSLP::calculateTreeCostAndTrimNonProfitable(ArrayRef<Value *> VectorizedVals,
     NodesCosts.erase(TE);
   }
 
-  // Check if all gather nodes that reuse the splat subtrees are marked for
-  // deletion. In this case the whole splat subtree must be deleted. If only
-  // some of the gathers are trimmed, keeping the subtree still costs its full
-  // price plus the extracts of the scalars used by the remaining scalar code,
-  // while the surviving gathers can materialize the splatted scalars
-  // directly. Drop the subtree if it does not pay off.
-  for (TreeEntry *TE : SplatGatheredScalarsRoots) {
-    if (DeletedNodes.contains(TE))
-      continue;
-    ValuesToInsertTy ValuesToInsert;
-    APInt DemandedElts = FindDemandedElts(TE, ValuesToInsert);
-    if (!DemandedElts.isZero()) {
-      InstructionCost CurrentGathersCost, DroppedGathersCost;
-      if (IsSplatSubtreeProfitable(TE, ValuesToInsert, CurrentGathersCost,
-                                   DroppedGathersCost))
-        continue;
-      // Dropped as unprofitable: exclude its cost from the reference cost, so
-      // the trimming of the remaining tree is not reverted because of it, and
-      // keep it deleted even if the trimming is reverted.
-      DroppedSplatSubtrees.insert(TE);
-      for (unsigned Idx : std::get<2>(SubtreeCosts[TE->Idx]))
-        DroppedSplatSubtrees.insert(VectorizableTree[Idx].get());
-      Cost -= std::get<1>(SubtreeCosts[TE->Idx]);
-    }
-    // Not used by the surviving gathers or not profitable to keep.
-    SubtreesToDelete.insert(TE);
-    NodesCosts.erase(TE);
-  }
-
   // Deleted all subtrees rooted at gathered loads nodes or splat subtrees.
   for (std::unique_ptr<TreeEntry> &TE : VectorizableTree) {
     if (TE->UserTreeIndex &&
@@ -19530,63 +19580,44 @@ BoUpSLP::calculateTreeCostAndTrimNonProfitable(ArrayRef<Value *> VectorizedVals,
                       << ".\n"
                       << "SLP: Current total cost = " << NewCost << "\n");
   }
-  const bool Reverted =
-      NewCost + LoadsExtractsCost > Cost ||
-      (!PreferTrimmedTree && NewCost + LoadsExtractsCost == Cost);
-  if (Reverted) {
-    DeletedNodes.clear();
+  // The trimming is reverted if the restored tree is not more expensive. The
+  // restored tree is priced the same way as the trimmed one, from the per-node
+  // costs: adjusting the pre-trimming total incrementally is not exact with
+  // the auxiliary subtrees around, since the subtree costs recorded before the
+  // trimming fold in the gathered loads they reuse, and the gathers inside a
+  // subtree change their price once their reuse sources are dropped.
+  auto ComputeRestoredCost = [&]() {
+    DeletedNodes = OrigDeletedNodes;
     TransformedToGatherNodes.clear();
-    // The dropped splat subtrees stay deleted: they were excluded from the
-    // reference cost and must not be resurrected by the revert.
-    DeletedNodes.insert(DroppedSplatSubtrees.begin(),
-                        DroppedSplatSubtrees.end());
-    NewCost = Cost;
+    NodesCosts = OrigNodesCosts;
     if (!SplatGatheredScalarsRoots.empty()) {
-      // The revert restores the pre-trimming tree, including the splat
-      // subtrees that were deleted as unused while their reusing gathers were
-      // trimmed. Recost the gather nodes for the restored set of vectorized
-      // nodes, then drop the splat subtrees that do not pay off in the
-      // restored tree. The reference cost still holds the gather costs of the
-      // full tree, where the already dropped splat subtrees served as reuse
-      // sources; rebase it on the recosted costs.
-      for (const std::unique_ptr<TreeEntry> &TE : VectorizableTree) {
-        if (!TE->isGather() || DeletedNodes.contains(TE.get()))
-          continue;
-        InstructionCost C = RecostEntry(TE.get());
-        NewCost += C - OrigGatherCosts.lookup(TE.get());
-        NodesCosts[TE.get()] = C;
-      }
-      for (TreeEntry *TE : SplatGatheredScalarsRoots) {
-        if (DeletedNodes.contains(TE))
-          continue;
-        ValuesToInsertTy ValuesToInsert;
-        InstructionCost CurrentGathersCost = 0, DroppedGathersCost = 0;
-        if (!FindDemandedElts(TE, ValuesToInsert).isZero() &&
-            IsSplatSubtreeProfitable(TE, ValuesToInsert, CurrentGathersCost,
-                                     DroppedGathersCost))
-          continue;
-        DeletedNodes.insert(TE);
-        for (unsigned Idx : std::get<2>(SubtreeCosts[TE->Idx]))
-          DeletedNodes.insert(VectorizableTree[Idx].get());
-        NewCost += DroppedGathersCost - CurrentGathersCost -
-                   std::get<1>(SubtreeCosts[TE->Idx]);
-        SyncGatherCosts(ValuesToInsert);
-      }
+      for (const std::unique_ptr<TreeEntry> &TE : VectorizableTree)
+        if (TE->isGather() && !DeletedNodes.contains(TE.get()))
+          NodesCosts[TE.get()] = RecostEntry(TE.get());
+      DropUnprofitableSplatSubtrees(/*EraseCosts=*/false);
       // Dropping the splat subtrees may leave the gathered loads subtrees
-      // built for them without surviving gather users; delete those too.
-      for (TreeEntry *TE : GatheredLoadsNodes) {
-        if (DeletedNodes.contains(TE))
-          continue;
-        ValuesToInsertTy ValuesToInsert;
-        if (!FindDemandedElts(TE, ValuesToInsert).isZero())
-          continue;
-        DeletedNodes.insert(TE);
-        for (unsigned Idx : std::get<2>(SubtreeCosts[TE->Idx]))
-          DeletedNodes.insert(VectorizableTree[Idx].get());
-        NewCost -= std::get<1>(SubtreeCosts[TE->Idx]);
-      }
+      // built for them without surviving gather users.
+      DropDeadGatheredLoads(/*EraseCosts=*/false);
+      for (const TreeEntry *TE : DeletedNodes)
+        NodesCosts.erase(TE);
     }
+    return SumNodesCosts();
+  };
+  const InstructionCost TrimmedCost = NewCost + LoadsExtractsCost;
+  auto TrimmedDeletedNodes = DeletedNodes;
+  auto TrimmedTransformedNodes = TransformedToGatherNodes;
+  auto TrimmedNodesCosts = NodesCosts;
+  const InstructionCost RestoredCost = ComputeRestoredCost();
+  LLVM_DEBUG(dbgs() << "SLP: Trimmed tree cost = " << TrimmedCost
+                    << ", restored tree cost = " << RestoredCost << ".\n");
+  const bool Reverted = TrimmedCost > RestoredCost ||
+                        (!PreferTrimmedTree && TrimmedCost == RestoredCost);
+  if (Reverted) {
+    NewCost = RestoredCost;
   } else {
+    DeletedNodes = std::move(TrimmedDeletedNodes);
+    TransformedToGatherNodes = std::move(TrimmedTransformedNodes);
+    NodesCosts = std::move(TrimmedNodesCosts);
     // If the remaining tree is just a buildvector - exit, it will cause
     // endless attempts to vectorize.
     if (VectorizableTree.size() >= 2 && getRootNode().hasState() &&
@@ -19603,7 +19634,7 @@ BoUpSLP::calculateTreeCostAndTrimNonProfitable(ArrayRef<Value *> VectorizedVals,
         TransformedToGatherNodes.contains(VectorizableTree[2].get()))
       return InstructionCost::getInvalid();
   }
-  TrimAuxSubtreesForInstCount(NewCost, /*UseCurrentNodeCosts=*/!Reverted);
+  TrimAuxSubtreesForInstCount(NewCost, /*UseCurrentNodeCosts=*/true);
   return NewCost;
 }
 
diff --git a/llvm/test/Transforms/SLPVectorizer/AArch64/splat-gather-subtree-loop-extracts.ll b/llvm/test/Transforms/SLPVectorizer/AArch64/splat-gather-subtree-loop-extracts.ll
index c2ca451ba0fb3..572dab05408bd 100644
--- a/llvm/test/Transforms/SLPVectorizer/AArch64/splat-gather-subtree-loop-extracts.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AArch64/splat-gather-subtree-loop-extracts.ll
@@ -9,18 +9,21 @@
 define i32 @splat_subtree_loop_extracts(double %div.i) {
 ; CHECK-LABEL: @splat_subtree_loop_extracts(
 ; CHECK-NEXT:  entry:
-; CHECK-NEXT:    [[TMP0:%.*]] = tail call double @llvm.fmuladd.f64(double 0.000000e+00, double 0.000000e+00, double 0.000000e+00)
+; CHECK-NEXT:    [[TMP4:%.*]] = call <2 x double> @llvm.fmuladd.v2f64(<2 x double> zeroinitializer, <2 x double> <double 0.000000e+00, double -0.000000e+00>, <2 x double> zeroinitializer)
+; CHECK-NEXT:    [[TMP14:%.*]] = insertelement <2 x double> <double poison, double 0.000000e+00>, double [[DIV_I1:%.*]], i64 0
 ; CHECK-NEXT:    br label [[FOR_BODY46_I:%.*]]
 ; CHECK:       for.body46.i:
+; CHECK-NEXT:    [[TMP0:%.*]] = tail call double @llvm.fmuladd.f64(double 0.000000e+00, double 0.000000e+00, double 0.000000e+00)
+; CHECK-NEXT:    [[DIV_I:%.*]] = fdiv double 0.000000e+00, 0.000000e+00
 ; CHECK-NEXT:    [[ARRAYIDX19_US63_I_3_1:%.*]] = getelementptr i8, ptr poison, i64 328
-; CHECK-NEXT:    [[MUL5_I471:%.*]] = fmul double [[TMP0]], [[DIV_I:%.*]]
-; CHECK-NEXT:    [[TMP1:%.*]] = insertelement <2 x double> poison, double [[MUL5_I471]], i64 0
+; CHECK-NEXT:    [[TMP1:%.*]] = fmul <2 x double> [[TMP14]], [[TMP4]]
+; CHECK-NEXT:    [[MUL5_I471:%.*]] = fmul double [[TMP0]], [[DIV_I]]
 ; CHECK-NEXT:    [[TMP2:%.*]] = shufflevector <2 x double> [[TMP1]], <2 x double> poison, <2 x i32> zeroinitializer
 ; CHECK-NEXT:    [[TMP3:%.*]] = call <2 x double> @llvm.fmuladd.v2f64(<2 x double> [[TMP2]], <2 x double> zeroinitializer, <2 x double> zeroinitializer)
-; CHECK-NEXT:    [[TMP4:%.*]] = call <2 x double> @llvm.fmuladd.v2f64(<2 x double> zeroinitializer, <2 x double> <double 0.000000e+00, double -0.000000e+00>, <2 x double> zeroinitializer)
-; CHECK-NEXT:    [[TMP5:%.*]] = fmul <2 x double> [[TMP4]], <double +qnan, double 0.000000e+00>
+; CHECK-NEXT:    [[TMP5:%.*]] = insertelement <2 x double> poison, double [[MUL5_I471]], i64 0
 ; CHECK-NEXT:    [[TMP6:%.*]] = shufflevector <2 x double> [[TMP5]], <2 x double> poison, <2 x i32> zeroinitializer
-; CHECK-NEXT:    [[TMP7:%.*]] = shufflevector <2 x double> [[TMP5]], <2 x double> poison, <2 x i32> <i32 1, i32 1>
+; CHECK-NEXT:    [[TMP15:%.*]] = shufflevector <2 x double> [[TMP1]], <2 x double> poison, <2 x i32> <i32 1, i32 poison>
+; CHECK-NEXT:    [[TMP7:%.*]] = shufflevector <2 x double> [[TMP15]], <2 x double> poison, <2 x i32> zeroinitializer
 ; CHECK-NEXT:    [[TMP8:%.*]] = call <2 x double> @llvm.fmuladd.v2f64(<2 x double> [[TMP6]], <2 x double> zeroinitializer, <2 x double> [[TMP7]])
 ; CHECK-NEXT:    [[TMP9:%.*]] = call <2 x double> @llvm.fmuladd.v2f64(<2 x double> zeroinitializer, <2 x double> zeroinitializer, <2 x double> [[TMP8]])
 ; CHECK-NEXT:    [[TMP10:%.*]] = fmul <2 x double> zeroinitializer, [[TMP3]]
diff --git a/llvm/test/Transforms/SLPVectorizer/AArch64/splat-gather-subtree-satd.ll b/llvm/test/Transforms/SLPVectorizer/AArch64/splat-gather-subtree-satd.ll
index 93891801e9f59..f14406a87317d 100644
--- a/llvm/test/Transforms/SLPVectorizer/AArch64/splat-gather-subtree-satd.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AArch64/splat-gather-subtree-satd.ll
@@ -39,199 +39,186 @@ define i32 @satd_4x4(ptr %pix1, i64 %i_pix1, ptr %pix2, i64 %i_pix2) {
 ; CHECK-NEXT:    [[ARRAYIDX12_3:%.*]] = getelementptr inbounds nuw i8, ptr [[ADD_PTR31_2]], i64 2
 ; CHECK-NEXT:    [[ARRAYIDX15_3:%.*]] = getelementptr inbounds nuw i8, ptr [[ADD_PTR_2]], i64 3
 ; CHECK-NEXT:    [[ARRAYIDX17_3:%.*]] = getelementptr inbounds nuw i8, ptr [[ADD_PTR31_2]], i64 3
+; CHECK-NEXT:    [[TMP6:%.*]] = load i8, ptr [[ARRAYIDX15]], align 1
+; CHECK-NEXT:    [[TMP7:%.*]] = load i8, ptr [[ARRAYIDX10]], align 1
 ; CHECK-NEXT:    [[TMP0:%.*]] = load i8, ptr [[ARRAYIDX3]], align 1
 ; CHECK-NEXT:    [[TMP1:%.*]] = load i8, ptr [[PIX1]], align 1
+; CHECK-NEXT:    [[TMP4:%.*]] = load i8, ptr [[ARRAYIDX17]], align 1
+; CHECK-NEXT:    [[TMP5:%.*]] = load i8, ptr [[ARRAYIDX12]], align 1
 ; CHECK-NEXT:    [[TMP2:%.*]] = load i8, ptr [[ARRAYIDX5]], align 1
 ; CHECK-NEXT:    [[TMP3:%.*]] = load i8, ptr [[PIX2]], align 1
-; CHECK-NEXT:    [[TMP4:%.*]] = load i8, ptr [[ARRAYIDX15]], align 1
-; CHECK-NEXT:    [[TMP5:%.*]] = load i8, ptr [[ARRAYIDX10]], align 1
-; CHECK-NEXT:    [[TMP6:%.*]] = load i8, ptr [[ARRAYIDX17]], align 1
-; CHECK-NEXT:    [[TMP7:%.*]] = load i8, ptr [[ARRAYIDX12]], align 1
+; CHECK-NEXT:    [[TMP18:%.*]] = load i8, ptr [[ARRAYIDX15_1]], align 1
+; CHECK-NEXT:    [[TMP9:%.*]] = load i8, ptr [[ARRAYIDX10_1]], align 1
+; CHECK-NEXT:    [[TMP19:%.*]] = load i8, ptr [[ARRAYIDX3_1]], align 1
+; CHECK-NEXT:    [[TMP22:%.*]] = load i8, ptr [[ADD_PTR]], align 1
+; CHECK-NEXT:    [[TMP12:%.*]] = load i8, ptr [[ARRAYIDX17_1]], align 1
+; CHECK-NEXT:    [[TMP13:%.*]] = load i8, ptr [[ARRAYIDX12_1]], align 1
+; CHECK-NEXT:    [[TMP14:%.*]] = load i8, ptr [[ARRAYIDX5_1]], align 1
+; CHECK-NEXT:    [[TMP15:%.*]] = load i8, ptr [[ADD_PTR31]], align 1
+; CHECK-NEXT:    [[TMP16:%.*]] = load i8, ptr [[ARRAYIDX15_2]], align 1
+; CHECK-NEXT:    [[TMP17:%.*]] = load i8, ptr [[ARRAYIDX10_2]], align 1
 ; CHECK-NEXT:    [[TMP8:%.*]] = load i8, ptr [[ARRAYIDX3_2]], align 1
 ; CHECK-NEXT:    [[TMP31:%.*]] = load i8, ptr [[ADD_PTR_1]], align 1
-; CHECK-NEXT:    [[CONV18_3:%.*]] = zext i8 [[TMP31]] to i32
-; CHECK-NEXT:    [[CONV:%.*]] = zext i8 [[TMP1]] to i32
+; CHECK-NEXT:    [[TMP20:%.*]] = load i8, ptr [[ARRAYIDX17_2]], align 1
+; CHECK-NEXT:    [[TMP21:%.*]] = load i8, ptr [[ARRAYIDX12_2]], align 1
 ; CHECK-NEXT:    [[TMP10:%.*]] = load i8, ptr [[ARRAYIDX5_2]], align 1
 ; CHECK-NEXT:    [[TMP11:%.*]] = load i8, ptr [[ADD_PTR31_1]], align 1
-; CHECK-NEXT:    [[CONV2_2:%.*]] = zext i8 [[TMP11]] to i32
+; CHECK-NEXT:    [[TMP24:%.*]] = load i8, ptr [[ARRAYIDX15_3]], align 1
+; CHECK-NEXT:    [[TMP25:%.*]] = load i8, ptr [[ARRAYIDX10_3]], align 1
+; CHECK-NEXT:    [[TMP26:%.*]] = load i8, ptr [[ARRAYIDX3_3]], align 1
+; CHECK-NEXT:    [[TMP27:%.*]] = load i8, ptr [[ADD_PTR_2]], align 1
+; CHECK-NEXT:    [[CONV11:%.*]] = zext i8 [[TMP7]] to i32
+; CHECK-NEXT:    [[CONV:%.*]] = zext i8 [[TMP1]] to i32
+; CHECK-NEXT:    [[CONV11_1:%.*]] = zext i8 [[TMP9]] to i32
+; CHECK-NEXT:    [[CONV_1:%.*]] = zext i8 [[TMP22]] to i32
+; CHECK-NEXT:    [[CONV18_3:%.*]] = zext i8 [[TMP17]] to i32
+; CHECK-NEXT:    [[CONV_2:%.*]] = zext i8 [[TMP31]] to i32
+; CHECK-NEXT:    [[CONV11_3:%.*]] = zext i8 [[TMP25]] to i32
+; CHECK-NEXT:    [[CONV_3:%.*]] = zext i8 [[TMP27]] to i32
+; CHECK-NEXT:    [[TMP28:%.*]] = load i8, ptr [[ARRAYIDX17_3]], align 1
+; CHECK-NEXT:    [[TMP29:%.*]] = load i8, ptr [[ARRAYIDX12_3]], align 1
+; CHECK-NEXT:    [[TMP30:%.*]] = load i8, ptr [[ARRAYIDX5_3]], align 1
+; CHECK-NEXT:    [[TMP37:%.*]] = load i8, ptr [[ADD_PTR31_2]], align 1
+; CHECK-NEXT:    [[CONV13:%.*]] = zext i8 [[TMP5]] to i32
 ; CHECK-NEXT:    [[CONV2:%.*]] = zext i8 [[TMP3]] to i32
+; CHECK-NEXT:    [[CONV13_1:%.*]] = zext i8 [[TMP13]] to i32
+; CHECK-NEXT:    [[CONV2_1:%.*]] = zext i8 [[TMP15]] to i32
+; CHECK-NEXT:    [[CONV2_2:%.*]] = zext i8 [[TMP21]] to i32
+; CHECK-NEXT:    [[CONV2_4:%.*]] = zext i8 [[TMP11]] to i32
+; CHECK-NEXT:    [[CONV13_3:%.*]] = zext i8 [[TMP29]] to i32
+; CHECK-NEXT:    [[CONV2_3:%.*]] = zext i8 [[TMP37]] to i32
+; CHECK-NEXT:    [[SHL22_3:%.*]] = sub nsw i32 [[CONV11]], [[CONV13]]
+; CHECK-NEXT:    [[ADD9_3:%.*]] = sub nsw i32 [[CONV]], [[CONV2]]
+; CHECK-NEXT:    [[SUB14_2:%.*]] = sub nsw i32 [[CONV11_1]], [[CONV13_1]]
+; CHECK-NEXT:    [[SUB14:%.*]] = sub nsw i32 [[CONV_1]], [[CONV2_1]]
 ; CHECK-NEXT:    [[SUB14_3:%.*]] = sub nsw i32 [[CONV18_3]], [[CONV2_2]]
-; CHECK-NEXT:    [[SUB:%.*]] = sub nsw i32 [[CONV]], [[CONV2]]
-; CHECK-NEXT:    [[CONV4_2:%.*]] = zext i8 [[TMP8]] to i32
+; CHECK-NEXT:    [[SUB_2:%.*]] = sub nsw i32 [[CONV_2]], [[CONV2_4]]
+; CHECK-NEXT:    [[ADD44:%.*]] = sub nsw i32 [[CONV11_3]], [[CONV13_3]]
+; CHECK-NEXT:    [[SUB_1:%.*]] = sub nsw i32 [[CONV_3]], [[CONV2_3]]
+; CHECK-NEXT:    [[CONV16:%.*]] = zext i8 [[TMP6]] to i32
 ; CHECK-NEXT:    [[CONV4:%.*]] = zext i8 [[TMP0]] to i32
-; CHECK-NEXT:    [[CONV6_2:%.*]] = zext i8 [[TMP10]] to i32
+; CHECK-NEXT:    [[CONV16_1:%.*]] = zext i8 [[TMP18]] to i32
+; CHECK-NEXT:    [[CONV4_1:%.*]] = zext i8 [[TMP19]] to i32
+; CHECK-NEXT:    [[CONV4_2:%.*]] = zext i8 [[TMP16]] to i32
+; CHECK-NEXT:    [[SUB:%.*]] = zext i8 [[TMP8]] to i32
+; CHECK-NEXT:    [[CONV16_3:%.*]] = zext i8 [[TMP24]] to i32
+; CHECK-NEXT:    [[CONV4_3:%.*]] = zext i8 [[TMP26]] to i32
+; CHECK-NEXT:    [[CONV18:%.*]] = zext i8 [[TMP4]] to i32
 ; CHECK-NEXT:    [[CONV6:%.*]] = zext i8 [[TMP2]] to i32
+; CHECK-NEXT:    [[CONV18_1:%.*]] = zext i8 [[TMP12]] to i32
+; CHECK-NEXT:    [[CONV6_1:%.*]] = zext i8 [[TMP14]] to i32
+; CHECK-NEXT:    [[CONV6_2:%.*]] = zext i8 [[TMP20]] to i32
+; CHECK-NEXT:    [[SUB7:%.*]] = zext i8 [[TMP10]] to i32
+; CHECK-NEXT:    [[CONV18_4:%.*]] = zext i8 [[TMP28]] to i32
+; CHECK-NEXT:    [[CONV6_3:%.*]] = zext i8 [[TMP30]] to i32
+; CHECK-NEXT:    [[ADD20_3:%.*]] = sub nsw i32 [[CONV16]], [[CONV18]]
+; CHECK-NEXT:    [[ADD23_3:%.*]] = sub nsw i32 [[CONV4]], [[CONV6]]
+; CHECK-NEXT:    [[SUB19_2:%.*]] = sub nsw i32 [[CONV16_1]], [[CONV18_1]]
+; CHECK-NEXT:    [[SUB19:%.*]] = sub nsw i32 [[CONV4_1]], [[CONV6_1]]
 ; CHECK-NEXT:    [[SUB19_3:%.*]] = sub nsw i32 [[CONV4_2]], [[CONV6_2]]
-; CHECK-NEXT:    [[SUB7:%.*]] = sub nsw i32 [[CONV4]], [[CONV6]]
-; CHECK-NEXT:    [[ADD20_3:%.*]] = add nsw i32 [[SUB19_3]], [[SUB14_3]]
-; CHECK-NEXT:    [[ADD23_3:%.*]] = add nsw i32 [[SUB7]], [[SUB]]
-; CHECK-NEXT:    [[SUB21_3:%.*]] = sub nsw i32 [[SUB14_3]], [[SUB19_3]]
 ; CHECK-NEXT:    [[SUB8:%.*]] = sub nsw i32 [[SUB]], [[SUB7]]
-; CHECK-NEXT:    [[SHL22_3:%.*]] = shl nsw i32 [[SUB21_3]], 16
-; CHECK-NEXT:    [[ADD9_3:%.*]] = shl nsw i32 [[SUB8]], 16
+; CHECK-NEXT:    [[ADD58:%.*]] = sub nsw i32 [[CONV16_3]], [[CONV18_4]]
+; CHECK-NEXT:    [[SUB7_1:%.*]] = sub nsw i32 [[CONV4_3]], [[CONV6_3]]
 ; CHECK-NEXT:    [[ADD9_2:%.*]] = add nsw i32 [[ADD20_3]], [[SHL22_3]]
 ; CHECK-NEXT:    [[ADD24_3:%.*]] = add nsw i32 [[ADD23_3]], [[ADD9_3]]
-; CHECK-NEXT:    [[TMP12:%.*]] = load i8, ptr [[ARRAYIDX15_2]], align 1
-; CHECK-NEXT:    [[TMP13:%.*]] = load i8, ptr [[ARRAYIDX10_2]], align 1
-; CHECK-NEXT:    [[CONV11_2:%.*]] = zext i8 [[TMP13]] to i32
-; CHECK-NEXT:    [[CONV11:%.*]] = zext i8 [[TMP5]] to i32
-; CHECK-NEXT:    [[TMP14:%.*]] = load i8, ptr [[ARRAYIDX17_2]], align 1
-; CHECK-NEXT:    [[TMP15:%.*]] = load i8, ptr [[ARRAYIDX12_2]], align 1
-; CHECK-NEXT:    [[CONV13_2:%.*]] = zext i8 [[TMP15]] to i32
-; CHECK-NEXT:    [[CONV13:%.*]] = zext i8 [[TMP7]] to i32
-; CHECK-NEXT:    [[SUB14_2:%.*]] = sub nsw i32 [[CONV11_2]], [[CONV13_2]]
-; CHECK-NEXT:    [[SUB14:%.*]] = sub nsw i32 [[CONV11]], [[CONV13]]
-; CHECK-NEXT:    [[CONV16_2:%.*]] = zext i8 [[TMP12]] to i32
-; CHECK-NEXT:    [[CONV16:%.*]] = zext i8 [[TMP4]] to i32
-; CHECK-NEXT:    [[CONV18_2:%.*]] = zext i8 [[TMP14]] to i32
-; CHECK-NEXT:    [[CONV18:%.*]] = zext i8 [[TMP6]] to i32
-; CHECK-NEXT:    [[SUB19_2:%.*]] = sub nsw i32 [[CONV16_2]], [[CONV18_2]]
-; CHECK-NEXT:    [[SUB19:%.*]] = sub nsw i32 [[CONV16]], [[CONV18]]
 ; CHECK-NEXT:    [[ADD20_2:%.*]] = add nsw i32 [[SUB19_2]], [[SUB14_2]]
 ; CHECK-NEXT:    [[ADD20:%.*]] = add nsw i32 [[SUB19]], [[SUB14]]
-; CHECK-NEXT:    [[SUB21_2:%.*]] = sub nsw i32 [[SUB14_2]], [[SUB19_2]]
-; CHECK-NEXT:    [[SUB21:%.*]] = sub nsw i32 [[SUB14]], [[SUB19]]
-; CHECK-NEXT:    [[SHL22_2:%.*]] = shl nsw i32 [[SUB21_2]], 16
-; CHECK-NEXT:    [[SHL22:%.*]] = shl nsw i32 [[SUB21]], 16
-; CHECK-NEXT:    [[ADD23_2:%.*]] = add nsw i32 [[ADD20_2]], [[SHL22_2]]
-; CHECK-NEXT:    [[ADD23:%.*]] = add nsw i32 [[ADD20]], [[SHL22]]
-; CHECK-NEXT:    [[TMP16:%.*]] = load i8, ptr [[ARRAYIDX3_1]], align 1
-; CHECK-NEXT:    [[TMP17:%.*]] = load i8, ptr [[ADD_PTR]], align 1
-; CHECK-NEXT:    [[TMP18:%.*]] = load i8, ptr [[ARRAYIDX5_1]], align 1
-; CHECK-NEXT:    [[TMP19:%.*]] = load i8, ptr [[ADD_PTR31]], align 1
-; CHECK-NEXT:    [[TMP20:%.*]] = load i8, ptr [[ARRAYIDX15_1]], align 1
-; CHECK-NEXT:    [[TMP21:%.*]] = load i8, ptr [[ARRAYIDX10_1]], align 1
-; CHECK-NEXT:    [[TMP22:%.*]] = load i8, ptr [[ARRAYIDX17_1]], align 1
-; CHECK-NEXT:    [[TMP23:%.*]] = load i8, ptr [[ARRAYIDX12_1]], align 1
-; CHECK-NEXT:    [[TMP24:%.*]] = load i8, ptr [[ARRAYIDX3_3]], align 1
-; CHECK-NEXT:    [[TMP25:%.*]] = load i8, ptr [[ADD_PTR_2]], align 1
-; CHECK-NEXT:    [[CONV_3:%.*]] = zext i8 [[TMP25]] to i32
-; CHECK-NEXT:    [[CONV_1:%.*]] = zext i8 [[TMP17]] to i32
-; CHECK-NEXT:    [[TMP26:%.*]] = load i8, ptr [[ARRAYIDX5_3]], align 1
-; CHECK-NEXT:    [[TMP27:%.*]] = load i8, ptr [[ADD_PTR31_2]], align 1
-; CHECK-NEXT:    [[CONV2_3:%.*]] = zext i8 [[TMP27]] to i32
-; CHECK-NEXT:    [[CONV2_1:%.*]] = zext i8 [[TMP19]] to i32
-; CHECK-NEXT:    [[ADD44:%.*]] = sub nsw i32 [[CONV_3]], [[CONV2_3]]
-; CHECK-NEXT:    [[SUB_1:%.*]] = sub nsw i32 [[CONV_1]], [[CONV2_1]]
-; CHECK-NEXT:    [[CONV4_3:%.*]] = zext i8 [[TMP24]] to i32
-; CHECK-NEXT:    [[CONV4_1:%.*]] = zext i8 [[TMP16]] to i32
-; CHECK-NEXT:    [[CONV6_3:%.*]] = zext i8 [[TMP26]] to i32
-; CHECK-NEXT:    [[CONV6_1:%.*]] = zext i8 [[TMP18]] to i32
-; CHECK-NEXT:    [[ADD58:%.*]] = sub nsw i32 [[CONV4_3]], [[CONV6_3]]
-; CHECK-NEXT:    [[SUB7_1:%.*]] = sub nsw i32 [[CONV4_1]], [[CONV6_1]]
+; CHECK-NEXT:    [[ADD23_4:%.*]] = add nsw i32 [[SUB19_3]], [[SUB14_3]]
+; CHECK-NEXT:    [[ADD23_1:%.*]] = add nsw i32 [[SUB8]], [[SUB_2]]
 ; CHECK-NEXT:    [[ADD66:%.*]] = add nsw i32 [[ADD58]], [[ADD44]]
 ; CHECK-NEXT:    [[ADD_1:%.*]] = add nsw i32 [[SUB7_1]], [[SUB_1]]
-; CHECK-NEXT:    [[SUB67:%.*]] = sub nsw i32 [[ADD44]], [[ADD58]]
-; CHECK-NEXT:    [[SUB8_1:%.*]] = sub nsw i32 [[SUB_1]], [[SUB7_1]]
+; CHECK-NEXT:    [[SUB67:%.*]] = sub nsw i32 [[SHL22_3]], [[ADD20_3]]
+; CHECK-NEXT:    [[SUB8_1:%.*]] = sub nsw i32 [[ADD9_3]], [[ADD23_3]]
+; CHECK-NEXT:    [[SUB21_4:%.*]] = sub nsw i32 [[SUB14_2]], [[SUB19_2]]
+; CHECK-NEXT:    [[SUB21_1:%.*]] = sub nsw i32 [[SUB14]], [[SUB19]]
+; CHECK-NEXT:    [[SUB21_2:%.*]] = sub nsw i32 [[SUB14_3]], [[SUB19_3]]
+; CHECK-NEXT:    [[SUB8_2:%.*]] = sub nsw i32 [[SUB_2]], [[SUB8]]
+; CHECK-NEXT:    [[SUB21_3:%.*]] = sub nsw i32 [[ADD44]], [[ADD58]]
+; CHECK-NEXT:    [[SUB8_3:%.*]] = sub nsw i32 [[SUB_1]], [[SUB7_1]]
 ; CHECK-NEXT:    [[SHL_3:%.*]] = shl nsw i32 [[SUB67]], 16
 ; CHECK-NEXT:    [[SHL_1:%.*]] = shl nsw i32 [[SUB8_1]], 16
-; CHECK-NEXT:    [[ADD9_4:%.*]] = add nsw i32 [[ADD66]], [[SHL_3]]
-; CHECK-NEXT:    [[ADD9_1:%.*]] = add nsw i32 [[ADD_1]], [[SHL_1]]
-; CHECK-NEXT:    [[TMP28:%.*]] = load i8, ptr [[ARRAYIDX15_3]], align 1
-; CHECK-NEXT:    [[TMP29:%.*]] = load i8, ptr [[ARRAYIDX10_3]], align 1
-; CHECK-NEXT:    [[CONV11_3:%.*]] = zext i8 [[TMP29]] to i32
-; CHECK-NEXT:    [[CONV11_1:%.*]] = zext i8 [[TMP21]] to i32
-; CHECK-NEXT:    [[TMP30:%.*]] = load i8, ptr [[ARRAYIDX17_3]], align 1
-; CHECK-NEXT:    [[TMP44:%.*]] = load i8, ptr [[ARRAYIDX12_3]], align 1
-; CHECK-NEXT:    [[CONV13_3:%.*]] = zext i8 [[TMP44]] to i32
-; CHECK-NEXT:    [[CONV13_1:%.*]] = zext i8 [[TMP23]] to i32
-; CHECK-NEXT:    [[SUB14_4:%.*]] = sub nsw i32 [[CONV11_3]], [[CONV13_3]]
-; CHECK-NEXT:    [[SUB14_1:%.*]] = sub nsw i32 [[CONV11_1]], [[CONV13_1]]
-; CHECK-NEXT:    [[CONV16_3:%.*]] = zext i8 [[TMP28]] to i32
-; CHECK-NEXT:    [[CONV16_1:%.*]] = zext i8 [[TMP20]] to i32
-; CHECK-NEXT:    [[CONV18_4:%.*]] = zext i8 [[TMP30]] to i32
-; CHECK-NEXT:    [[CONV18_1:%.*]] = zext i8 [[TMP22]] to i32
-; CHECK-NEXT:    [[SUB19_4:%.*]] = sub nsw i32 [[CONV16_3]], [[CONV18_4]]
-; CHECK-NEXT:    [[SUB19_1:%.*]] = sub nsw i32 [[CONV16_1]], [[CONV18_1]]
-; CHECK-NEXT:    [[ADD20_4:%.*]] = add nsw i32 [[SUB19_4]], [[SUB14_4]]
-; CHECK-NEXT:    [[ADD20_1:%.*]] = add nsw i32 [[SUB19_1]], [[SUB14_1]]
-; CHECK-NEXT:    [[SUB21_4:%.*]] = sub nsw i32 [[SUB14_4]], [[SUB19_4]]
-; CHECK-NEXT:    [[SUB21_1:%.*]] = sub nsw i32 [[SUB14_1]], [[SUB19_1]]
 ; CHECK-NEXT:    [[SHL22_4:%.*]] = shl nsw i32 [[SUB21_4]], 16
 ; CHECK-NEXT:    [[SHL22_1:%.*]] = shl nsw i32 [[SUB21_1]], 16
-; CHECK-NEXT:    [[ADD23_4:%.*]] = add nsw i32 [[ADD20_4]], [[SHL22_4]]
-; CHECK-NEXT:    [[ADD23_1:%.*]] = add nsw i32 [[ADD20_1]], [[SHL22_1]]
-; CHECK-NEXT:    [[ADD24_2:%.*]] = add nsw i32 [[ADD23_2]], [[ADD9_2]]
-; CHECK-NEXT:    [[ADD24:%.*]] = add nsw i32 [[ADD23]], [[ADD24_3]]
-; CHECK-NEXT:    [[SUB27_2:%.*]] = sub nsw i32 [[ADD9_2]], [[ADD23_2]]
-; CHECK-NEXT:    [[SUB27:%.*]] = sub nsw i32 [[ADD24_3]], [[ADD23]]
+; CHECK-NEXT:    [[ADD9_4:%.*]] = shl nsw i32 [[SUB21_2]], 16
+; CHECK-NEXT:    [[ADD9_1:%.*]] = shl nsw i32 [[SUB8_2]], 16
+; CHECK-NEXT:    [[SHL22_5:%.*]] = shl nsw i32 [[SUB21_3]], 16
+; CHECK-NEXT:    [[SHL_4:%.*]] = shl nsw i32 [[SUB8_3]], 16
+; CHECK-NEXT:    [[ADD59:%.*]] = add nsw i32 [[ADD9_2]], [[SHL_3]]
+; CHECK-NEXT:    [[ADD45:%.*]] = add nsw i32 [[ADD24_3]], [[SHL_1]]
+; CHECK-NEXT:    [[ADD23_2:%.*]] = add nsw i32 [[ADD20_2]], [[SHL22_4]]
+; CHECK-NEXT:    [[ADD9_5:%.*]] = add nsw i32 [[ADD20]], [[SHL22_1]]
 ; CHECK-NEXT:    [[ADD24_4:%.*]] = add nsw i32 [[ADD23_4]], [[ADD9_4]]
 ; CHECK-NEXT:    [[ADD24_1:%.*]] = add nsw i32 [[ADD23_1]], [[ADD9_1]]
-; CHECK-NEXT:    [[SUB27_3:%.*]] = sub nsw i32 [[ADD9_4]], [[ADD23_4]]
-; CHECK-NEXT:    [[SUB27_1:%.*]] = sub nsw i32 [[ADD9_1]], [[ADD23_1]]
-; CHECK-NEXT:    [[SUB51:%.*]] = sub nsw i32 [[ADD24]], [[ADD24_1]]
-; CHECK-NEXT:    [[ADD59:%.*]] = add nsw i32 [[ADD24_4]], [[ADD24_2]]
-; CHECK-NEXT:    [[ADD45:%.*]] = add nsw i32 [[ADD24_1]], [[ADD24]]
-; CHECK-NEXT:    [[SUB65:%.*]] = sub nsw i32 [[ADD24_2]], [[ADD24_4]]
+; CHECK-NEXT:    [[ADD23_5:%.*]] = add nsw i32 [[ADD66]], [[SHL22_5]]
+; CHECK-NEXT:    [[ADD9_6:%.*]] = add nsw i32 [[ADD_1]], [[SHL_4]]
 ; CHECK-NEXT:    [[TMP32:%.*]] = insertelement <2 x i32> poison, i32 [[ADD45]], i64 0
 ; CHECK-NEXT:    [[TMP33:%.*]] = shufflevector <2 x i32> [[TMP32]], <2 x i32> poison, <2 x i32> zeroinitializer
 ; CHECK-NEXT:    [[TMP34:%.*]] = insertelement <2 x i32> poison, i32 [[ADD59]], i64 0
 ; CHECK-NEXT:    [[TMP35:%.*]] = shufflevector <2 x i32> [[TMP34]], <2 x i32> poison, <2 x i32> zeroinitializer
+; CHECK-NEXT:    [[TMP51:%.*]] = sub nsw <2 x i32> [[TMP33]], [[TMP35]]
 ; CHECK-NEXT:    [[TMP36:%.*]] = add nsw <2 x i32> [[TMP33]], [[TMP35]]
-; CHECK-NEXT:    [[TMP37:%.*]] = sub nsw <2 x i32> [[TMP33]], [[TMP35]]
-; CHECK-NEXT:    [[TMP38:%.*]] = shufflevector <2 x i32> [[TMP36]], <2 x i32> [[TMP37]], <2 x i32> <i32 0, i32 3>
-; CHECK-NEXT:    [[ADD68:%.*]] = add nsw i32 [[SUB65]], [[SUB51]]
-; CHECK-NEXT:    [[SUB69:%.*]] = sub nsw i32 [[SUB51]], [[SUB65]]
-; CHECK-NEXT:    [[SHR_I126:%.*]] = lshr i32 [[ADD68]], 15
-; CHECK-NEXT:    [[AND_I127:%.*]] = and i32 [[SHR_I126]], 65537
-; CHECK-NEXT:    [[MUL_I128:%.*]] = mul nuw i32 [[AND_I127]], 65535
-; CHECK-NEXT:    [[ADD_I129:%.*]] = add i32 [[MUL_I128]], [[ADD68]]
-; CHECK-NEXT:    [[XOR_I130:%.*]] = xor i32 [[ADD_I129]], [[MUL_I128]]
-; CHECK-NEXT:    [[TMP39:%.*]] = lshr <2 x i32> [[TMP38]], splat (i32 15)
-; CHECK-NEXT:    [[TMP40:%.*]] = and <2 x i32> [[TMP39]], splat (i32 65537)
-; CHECK-NEXT:    [[TMP41:%.*]] = mul nuw <2 x i32> [[TMP40]], splat (i32 65535)
-; CHECK-NEXT:    [[TMP42:%.*]] = add <2 x i32> [[TMP41]], [[TMP38]]
-; CHECK-NEXT:    [[TMP43:%.*]] = xor <2 x i32> [[TMP42]], [[TMP41]]
-; CHECK-NEXT:    [[XOR_I135:%.*]] = extractelement <2 x i32> [[TMP43]], i64 0
-; CHECK-NEXT:    [[ADD71:%.*]] = add i32 [[XOR_I135]], [[XOR_I130]]
-; CHECK-NEXT:    [[XOR_I125:%.*]] = extractelement <2 x i32> [[TMP43]], i64 1
-; CHECK-NEXT:    [[ADD73:%.*]] = add i32 [[ADD71]], [[XOR_I125]]
-; CHECK-NEXT:    [[SHR_I:%.*]] = lshr i32 [[SUB69]], 15
-; CHECK-NEXT:    [[AND_I:%.*]] = and i32 [[SHR_I]], 65537
-; CHECK-NEXT:    [[MUL_I:%.*]] = mul nuw i32 [[AND_I]], 65535
-; CHECK-NEXT:    [[ADD_I:%.*]] = add i32 [[MUL_I]], [[SUB69]]
-; CHECK-NEXT:    [[XOR_I:%.*]] = xor i32 [[ADD_I]], [[MUL_I]]
-; CHECK-NEXT:    [[ADD75:%.*]] = add i32 [[ADD73]], [[XOR_I]]
-; CHECK-NEXT:    [[CONV77:%.*]] = and i32 [[ADD75]], 65535
-; CHECK-NEXT:    [[SHR:%.*]] = lshr i32 [[ADD75]], 16
-; CHECK-NEXT:    [[ADD79:%.*]] = add nuw nsw i32 [[SHR]], [[CONV77]]
-; CHECK-NEXT:    [[SUB51_1:%.*]] = sub nsw i32 [[SUB27]], [[SUB27_1]]
-; CHECK-NEXT:    [[ADD58_1:%.*]] = add nsw i32 [[SUB27_3]], [[SUB27_2]]
-; CHECK-NEXT:    [[ADD44_1:%.*]] = add nsw i32 [[SUB27_1]], [[SUB27]]
-; CHECK-NEXT:    [[SUB65_1:%.*]] = sub nsw i32 [[SUB27_2]], [[SUB27_3]]
-; CHECK-NEXT:    [[TMP46:%.*]] = insertelement <2 x i32> poison, i32 [[ADD44_1]], i64 0
+; CHECK-NEXT:    [[TMP38:%.*]] = shufflevector <2 x i32> [[TMP51]], <2 x i32> [[TMP36]], <2 x i32> <i32 0, i32 3>
+; CHECK-NEXT:    [[TMP39:%.*]] = insertelement <2 x i32> poison, i32 [[ADD9_5]], i64 0
+; CHECK-NEXT:    [[TMP40:%.*]] = shufflevector <2 x i32> [[TMP39]], <2 x i32> poison, <2 x i32> zeroinitializer
+; CHECK-NEXT:    [[TMP41:%.*]] = insertelement <2 x i32> poison, i32 [[ADD23_2]], i64 0
+; CHECK-NEXT:    [[TMP42:%.*]] = shufflevector <2 x i32> [[TMP41]], <2 x i32> poison, <2 x i32> zeroinitializer
+; CHECK-NEXT:    [[TMP43:%.*]] = sub nsw <2 x i32> [[TMP40]], [[TMP42]]
+; CHECK-NEXT:    [[TMP44:%.*]] = add nsw <2 x i32> [[TMP40]], [[TMP42]]
+; CHECK-NEXT:    [[TMP45:%.*]] = shufflevector <2 x i32> [[TMP43]], <2 x i32> [[TMP44]], <2 x i32> <i32 0, i32 3>
+; CHECK-NEXT:    [[TMP46:%.*]] = insertelement <2 x i32> poison, i32 [[ADD24_1]], i64 0
 ; CHECK-NEXT:    [[TMP47:%.*]] = shufflevector <2 x i32> [[TMP46]], <2 x i32> poison, <2 x i32> zeroinitializer
-; CHECK-NEXT:    [[TMP48:%.*]] = insertelement <2 x i32> poison, i32 [[ADD58_1]], i64 0
+; CHECK-NEXT:    [[TMP48:%.*]] = insertelement <2 x i32> poison, i32 [[ADD24_4]], i64 0
 ; CHECK-NEXT:    [[TMP49:%.*]] = shufflevector <2 x i32> [[TMP48]], <2 x i32> poison, <2 x i32> zeroinitializer
+; CHECK-NEXT:    [[TMP64:%.*]] = sub nsw <2 x i32> [[TMP47]], [[TMP49]]
 ; CHECK-NEXT:    [[TMP50:%.*]] = add nsw <2 x i32> [[TMP47]], [[TMP49]]
-; CHECK-NEXT:    [[TMP51:%.*]] = sub nsw <2 x i32> [[TMP47]], [[TMP49]]
-; CHECK-NEXT:    [[TMP52:%.*]] = shufflevector <2 x i32> [[TMP50]], <2 x i32> [[TMP51]], <2 x i32> <i32 0, i32 3>
-; CHECK-NEXT:    [[ADD68_1:%.*]] = add nsw i32 [[SUB65_1]], [[SUB51_1]]
-; CHECK-NEXT:    [[SUB69_1:%.*]] = sub nsw i32 [[SUB51_1]], [[SUB65_1]]
-; CHECK-NEXT:    [[SHR_I126_1:%.*]] = lshr i32 [[ADD68_1]], 15
-; CHECK-NEXT:    [[AND_I127_1:%.*]] = and i32 [[SHR_I126_1]], 65537
-; CHECK-NEXT:    [[MUL_I128_1:%.*]] = mul nuw i32 [[AND_I127_1]], 65535
-; CHECK-NEXT:    [[ADD_I129_1:%.*]] = add i32 [[MUL_I128_1]], [[ADD68_1]]
-; CHECK-NEXT:    [[XOR_I130_1:%.*]] = xor i32 [[ADD_I129_1]], [[MUL_I128_1]]
+; CHECK-NEXT:    [[TMP68:%.*]] = shufflevector <2 x i32> [[TMP64]], <2 x i32> [[TMP50]], <2 x i32> <i32 0, i32 3>
+; CHECK-NEXT:    [[TMP69:%.*]] = insertelement <2 x i32> poison, i32 [[ADD9_6]], i64 0
+; CHECK-NEXT:    [[TMP70:%.*]] = shufflevector <2 x i32> [[TMP69]], <2 x i32> poison, <2 x i32> zeroinitializer
+; CHECK-NEXT:    [[TMP71:%.*]] = insertelement <2 x i32> poison, i32 [[ADD23_5]], i64 0
+; CHECK-NEXT:    [[TMP72:%.*]] = shufflevector <2 x i32> [[TMP71]], <2 x i32> poison, <2 x i32> zeroinitializer
+; CHECK-NEXT:    [[TMP92:%.*]] = sub nsw <2 x i32> [[TMP70]], [[TMP72]]
+; CHECK-NEXT:    [[TMP58:%.*]] = add nsw <2 x i32> [[TMP70]], [[TMP72]]
+; CHECK-NEXT:    [[TMP59:%.*]] = shufflevector <2 x i32> [[TMP92]], <2 x i32> [[TMP58]], <2 x i32> <i32 0, i32 3>
+; CHECK-NEXT:    [[TMP60:%.*]] = add nsw <2 x i32> [[TMP45]], [[TMP38]]
+; CHECK-NEXT:    [[TMP61:%.*]] = sub nsw <2 x i32> [[TMP38]], [[TMP45]]
+; CHECK-NEXT:    [[TMP62:%.*]] = add nsw <2 x i32> [[TMP59]], [[TMP68]]
+; CHECK-NEXT:    [[TMP63:%.*]] = sub nsw <2 x i32> [[TMP68]], [[TMP59]]
+; CHECK-NEXT:    [[TMP52:%.*]] = add nsw <2 x i32> [[TMP62]], [[TMP60]]
+; CHECK-NEXT:    [[TMP65:%.*]] = sub nsw <2 x i32> [[TMP60]], [[TMP62]]
+; CHECK-NEXT:    [[TMP66:%.*]] = add nsw <2 x i32> [[TMP63]], [[TMP61]]
+; CHECK-NEXT:    [[TMP67:%.*]] = sub nsw <2 x i32> [[TMP61]], [[TMP63]]
 ; CHECK-NEXT:    [[TMP53:%.*]] = lshr <2 x i32> [[TMP52]], splat (i32 15)
 ; CHECK-NEXT:    [[TMP54:%.*]] = and <2 x i32> [[TMP53]], splat (i32 65537)
 ; CHECK-NEXT:    [[TMP55:%.*]] = mul nuw <2 x i32> [[TMP54]], splat (i32 65535)
 ; CHECK-NEXT:    [[TMP56:%.*]] = add <2 x i32> [[TMP55]], [[TMP52]]
 ; CHECK-NEXT:    [[TMP57:%.*]] = xor <2 x i32> [[TMP56]], [[TMP55]]
-; CHECK-NEXT:    [[XOR_I135_1:%.*]] = extractelement <2 x i32> [[TMP57]], i64 0
-; CHECK-NEXT:    [[ADD71_1:%.*]] = add i32 [[XOR_I135_1]], [[XOR_I130_1]]
-; CHECK-NEXT:    [[XOR_I125_1:%.*]] = extractelement <2 x i32> [[TMP57]], i64 1
-; CHECK-NEXT:    [[ADD73_1:%.*]] = add i32 [[ADD71_1]], [[XOR_I125_1]]
-; CHECK-NEXT:    [[SHR_I_1:%.*]] = lshr i32 [[SUB69_1]], 15
-; CHECK-NEXT:    [[AND_I_1:%.*]] = and i32 [[SHR_I_1]], 65537
-; CHECK-NEXT:    [[MUL_I_1:%.*]] = mul nuw i32 [[AND_I_1]], 65535
-; CHECK-NEXT:    [[ADD_I_1:%.*]] = add i32 [[MUL_I_1]], [[SUB69_1]]
-; CHECK-NEXT:    [[XOR_I_1:%.*]] = xor i32 [[ADD_I_1]], [[MUL_I_1]]
-; CHECK-NEXT:    [[ADD75_1:%.*]] = add i32 [[ADD73_1]], [[XOR_I_1]]
-; CHECK-NEXT:    [[CONV77_1:%.*]] = and i32 [[ADD75_1]], 65535
-; CHECK-NEXT:    [[SHR_1:%.*]] = lshr i32 [[ADD75_1]], 16
+; CHECK-NEXT:    [[TMP73:%.*]] = lshr <2 x i32> [[TMP66]], splat (i32 15)
+; CHECK-NEXT:    [[TMP74:%.*]] = and <2 x i32> [[TMP73]], splat (i32 65537)
+; CHECK-NEXT:    [[TMP75:%.*]] = mul nuw <2 x i32> [[TMP74]], splat (i32 65535)
+; CHECK-NEXT:    [[TMP76:%.*]] = add <2 x i32> [[TMP75]], [[TMP66]]
+; CHECK-NEXT:    [[TMP77:%.*]] = xor <2 x i32> [[TMP76]], [[TMP75]]
+; CHECK-NEXT:    [[TMP78:%.*]] = add <2 x i32> [[TMP57]], [[TMP77]]
+; CHECK-NEXT:    [[TMP79:%.*]] = lshr <2 x i32> [[TMP65]], splat (i32 15)
+; CHECK-NEXT:    [[TMP80:%.*]] = and <2 x i32> [[TMP79]], splat (i32 65537)
+; CHECK-NEXT:    [[TMP81:%.*]] = mul nuw <2 x i32> [[TMP80]], splat (i32 65535)
+; CHECK-NEXT:    [[TMP82:%.*]] = add <2 x i32> [[TMP81]], [[TMP65]]
+; CHECK-NEXT:    [[TMP83:%.*]] = xor <2 x i32> [[TMP82]], [[TMP81]]
+; CHECK-NEXT:    [[TMP84:%.*]] = add <2 x i32> [[TMP78]], [[TMP83]]
+; CHECK-NEXT:    [[TMP85:%.*]] = lshr <2 x i32> [[TMP67]], splat (i32 15)
+; CHECK-NEXT:    [[TMP86:%.*]] = and <2 x i32> [[TMP85]], splat (i32 65537)
+; CHECK-NEXT:    [[TMP87:%.*]] = mul nuw <2 x i32> [[TMP86]], splat (i32 65535)
+; CHECK-NEXT:    [[TMP88:%.*]] = add <2 x i32> [[TMP87]], [[TMP67]]
+; CHECK-NEXT:    [[TMP89:%.*]] = xor <2 x i32> [[TMP88]], [[TMP87]]
+; CHECK-NEXT:    [[TMP90:%.*]] = add <2 x i32> [[TMP84]], [[TMP89]]
+; CHECK-NEXT:    [[TMP91:%.*]] = lshr <2 x i32> [[TMP90]], splat (i32 16)
+; CHECK-NEXT:    [[SHR_1:%.*]] = extractelement <2 x i32> [[TMP91]], i64 1
+; CHECK-NEXT:    [[TMP93:%.*]] = and <2 x i32> [[TMP90]], splat (i32 65535)
+; CHECK-NEXT:    [[ADD79:%.*]] = extractelement <2 x i32> [[TMP93]], i64 1
 ; CHECK-NEXT:    [[ADD78_1:%.*]] = add nuw nsw i32 [[SHR_1]], [[ADD79]]
-; CHECK-NEXT:    [[ADD79_1:%.*]] = add nuw nsw i32 [[ADD78_1]], [[CONV77_1]]
+; CHECK-NEXT:    [[TMP95:%.*]] = extractelement <2 x i32> [[TMP91]], i64 0
+; CHECK-NEXT:    [[ADD78_2:%.*]] = add nuw nsw i32 [[TMP95]], [[ADD78_1]]
+; CHECK-NEXT:    [[TMP96:%.*]] = extractelement <2 x i32> [[TMP93]], i64 0
+; CHECK-NEXT:    [[ADD79_1:%.*]] = add nuw nsw i32 [[ADD78_2]], [[TMP96]]
 ; CHECK-NEXT:    [[SHR83:%.*]] = lshr i32 [[ADD79_1]], 1
 ; CHECK-NEXT:    ret i32 [[SHR83]]
 ;
diff --git a/llvm/test/Transforms/SLPVectorizer/AArch64/splat-gather-subtree-trim-revert-cost.ll b/llvm/test/Transforms/SLPVectorizer/AArch64/splat-gather-subtree-trim-revert-cost.ll
index 3ee07e04acc84..b08bae2711238 100644
--- a/llvm/test/Transforms/SLPVectorizer/AArch64/splat-gather-subtree-trim-revert-cost.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AArch64/splat-gather-subtree-trim-revert-cost.ll
@@ -12,24 +12,22 @@ define void @splat_subtree_trim_revert_cost(ptr %p, double %x, double %y) {
 ; CHECK-LABEL: define void @splat_subtree_trim_revert_cost(
 ; CHECK-SAME: ptr [[P:%.*]], double [[X:%.*]], double [[Y:%.*]]) {
 ; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[TMP0:%.*]] = insertelement <2 x double> <double 0.000000e+00, double poison>, double [[Y]], i64 1
+; CHECK-NEXT:    [[TMP1:%.*]] = insertelement <2 x double> poison, double [[X]], i64 1
 ; CHECK-NEXT:    br label %[[LOOP:.*]]
 ; CHECK:       [[LOOP]]:
-; CHECK-NEXT:    [[A0:%.*]] = tail call double @llvm.fmuladd.f64(double 0.000000e+00, double 0.000000e+00, double 0.000000e+00)
-; CHECK-NEXT:    [[A1:%.*]] = tail call double @llvm.fmuladd.f64(double 0.000000e+00, double 0.000000e+00, double 0.000000e+00)
-; CHECK-NEXT:    [[B0:%.*]] = tail call double @llvm.fmuladd.f64(double 0.000000e+00, double 0.000000e+00, double [[A1]])
 ; CHECK-NEXT:    [[A2:%.*]] = tail call double @llvm.fmuladd.f64(double 0.000000e+00, double 0.000000e+00, double 0.000000e+00)
-; CHECK-NEXT:    [[M0:%.*]] = fmul double 0.000000e+00, [[A0]]
-; CHECK-NEXT:    [[C0:%.*]] = tail call double @llvm.fmuladd.f64(double [[A2]], double 0.000000e+00, double [[M0]])
-; CHECK-NEXT:    [[D0:%.*]] = tail call double @llvm.fmuladd.f64(double 0.000000e+00, double [[B0]], double [[C0]])
-; CHECK-NEXT:    [[E0:%.*]] = tail call double @llvm.fmuladd.f64(double 0.000000e+00, double [[D0]], double 0.000000e+00)
+; CHECK-NEXT:    [[TMP2:%.*]] = call <2 x double> @llvm.fmuladd.v2f64(<2 x double> zeroinitializer, <2 x double> zeroinitializer, <2 x double> zeroinitializer)
+; CHECK-NEXT:    [[M0:%.*]] = fmul double 0.000000e+00, [[A2]]
 ; CHECK-NEXT:    [[GEP0:%.*]] = getelementptr i8, ptr [[P]], i64 736
-; CHECK-NEXT:    store double [[E0]], ptr [[GEP0]], align 8
-; CHECK-NEXT:    [[B1:%.*]] = tail call double @llvm.fmuladd.f64(double 0.000000e+00, double 0.000000e+00, double [[A1]])
-; CHECK-NEXT:    [[C1:%.*]] = tail call double @llvm.fmuladd.f64(double [[A2]], double [[Y]], double [[X]])
-; CHECK-NEXT:    [[D1:%.*]] = tail call double @llvm.fmuladd.f64(double 0.000000e+00, double [[B1]], double [[C1]])
-; CHECK-NEXT:    [[E1:%.*]] = tail call double @llvm.fmuladd.f64(double 0.000000e+00, double [[D1]], double 0.000000e+00)
-; CHECK-NEXT:    [[GEP1:%.*]] = getelementptr i8, ptr [[P]], i64 744
-; CHECK-NEXT:    store double [[E1]], ptr [[GEP1]], align 8
+; CHECK-NEXT:    [[TMP3:%.*]] = shufflevector <2 x double> [[TMP2]], <2 x double> poison, <2 x i32> zeroinitializer
+; CHECK-NEXT:    [[TMP4:%.*]] = call <2 x double> @llvm.fmuladd.v2f64(<2 x double> zeroinitializer, <2 x double> zeroinitializer, <2 x double> [[TMP3]])
+; CHECK-NEXT:    [[TMP5:%.*]] = shufflevector <2 x double> [[TMP2]], <2 x double> poison, <2 x i32> <i32 1, i32 1>
+; CHECK-NEXT:    [[TMP6:%.*]] = insertelement <2 x double> [[TMP1]], double [[M0]], i64 0
+; CHECK-NEXT:    [[TMP7:%.*]] = call <2 x double> @llvm.fmuladd.v2f64(<2 x double> [[TMP5]], <2 x double> [[TMP0]], <2 x double> [[TMP6]])
+; CHECK-NEXT:    [[TMP8:%.*]] = call <2 x double> @llvm.fmuladd.v2f64(<2 x double> zeroinitializer, <2 x double> [[TMP4]], <2 x double> [[TMP7]])
+; CHECK-NEXT:    [[TMP9:%.*]] = call <2 x double> @llvm.fmuladd.v2f64(<2 x double> zeroinitializer, <2 x double> [[TMP8]], <2 x double> zeroinitializer)
+; CHECK-NEXT:    store <2 x double> [[TMP9]], ptr [[GEP0]], align 8
 ; CHECK-NEXT:    br label %[[LOOP]]
 ;
 entry:



More information about the llvm-commits mailing list