[llvm] bb3c162 - [SLP]Drop unprofitable splat gather subtrees
via llvm-commits
llvm-commits at lists.llvm.org
Wed Sep 9 05:20:20 PDT 2026
Author: Alexey Bataev
Date: 2026-09-09T08:20:14-04:00
New Revision: bb3c1628511349afbc0a0a82ec690a2da7020b63
URL: https://github.com/llvm/llvm-project/commit/bb3c1628511349afbc0a0a82ec690a2da7020b63
DIFF: https://github.com/llvm/llvm-project/commit/bb3c1628511349afbc0a0a82ec690a2da7020b63.diff
LOG: [SLP]Drop unprofitable splat gather subtrees
The keep/drop check ran only when trimming changed something and
ignored the splat emission cost on both sides. Run it in both paths
with the actual costs: keep the subtree if its price plus extracts plus
the reusing gathers' current costs is not worse than re-emitting those
gathers without it. Exclude splat roots from the main trimming loop.
Fixes the perf regression from #220250, reported in https://github.com/llvm/llvm-project/pull/220250?email_source=notifications&email_token=ABI45DSHUYRFN4NA4DILB5T5NZTZTA5CNFSNUABFM5UWIORPF5TWS5BNNB2WEL2JONZXKZKDN5WW2ZLOOQXTKNJWG4ZTSMBZGUYKM4TFMFZW63VMON2GC5DFL5RWQYLOM5S2KZLWMVXHJLDGN5XXIZLSL5RWY2LDNM#issuecomment-5567390950.
Reviewers: RKSimon, bababuck
Pull Request: https://github.com/llvm/llvm-project/pull/221717
Added:
Modified:
llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
llvm/test/Transforms/SLPVectorizer/AArch64/splat-gather-subtree-drop.ll
llvm/test/Transforms/SLPVectorizer/AArch64/splat-gather-subtree-store-chain.ll
llvm/test/Transforms/SLPVectorizer/X86/lookahead.ll
llvm/test/Transforms/SLPVectorizer/X86/reorder_phi.ll
Removed:
################################################################################
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 3d1acd87f37b9..174841cb8666d 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -18898,6 +18898,10 @@ BoUpSLP::calculateTreeCostAndTrimNonProfitable(ArrayRef<Value *> VectorizedVals,
TreeEntry *TE = Worklist.top().first;
if (TE->isGather() || TE->Idx == 0 || DeletedNodes.contains(TE) ||
isa<StructType>(getValueType(TE->Scalars.front(), SLPReVec)) ||
+ // Splat subtree roots are decided by the keep/drop check below: their
+ // scalars are materialized in the reusing splat gathers, not in a
+ // gather of the root's own scalars.
+ is_contained(SplatGatheredScalarsRoots, TE) ||
// Exit early if the parent node is split node and any of scalars is
// used in other split nodes.
(TE->UserTreeIndex &&
@@ -19053,18 +19057,6 @@ BoUpSLP::calculateTreeCostAndTrimNonProfitable(ArrayRef<Value *> VectorizedVals,
}
Worklist.pop();
}
- if (!Changed) {
- // The splat subtrees are not linked to the tree root, so their cost is
- // not included in the root's subtree cost; add it explicitly.
- InstructionCost TotalCost = std::get<1>(SubtreeCosts.front());
- for (const TreeEntry *TE : SplatGatheredScalarsRoots)
- TotalCost += std::get<1>(SubtreeCosts[TE->Idx]);
- return TotalCost;
- }
-
- SmallPtrSet<TreeEntry *, 4> SubtreesToDelete;
- SmallPtrSet<TreeEntry *, 4> DroppedSplatSubtrees;
- InstructionCost LoadsExtractsCost = 0;
using ValuesToInsertTy =
SmallDenseMap<const TreeEntry *, SmallVector<Value *>>;
auto GetScalarTy = [&](const TreeEntry *TE) {
@@ -19111,6 +19103,109 @@ BoUpSLP::calculateTreeCostAndTrimNonProfitable(ArrayRef<Value *> VectorizedVals,
}
return BVCost;
};
+ auto RecostEntry = [&](const TreeEntry *TE) {
+ InstructionCost C = getEntryCost(TE, VectorizedVals, CheckedExtracts);
+ if (!C.isValid() || C == 0)
+ return C;
+ uint64_t Scale = EntryToScale.lookup(TE);
+ if (!Scale)
+ Scale = getEntryEffectiveScale(*TE);
+ return C * Scale;
+ };
+ // A splat subtree pays off only if its price plus the extracts of its
+ // scalars used by the remaining scalar code plus the current cost of the
+ // gather nodes that reuse it is not worse than re-emitting those gathers
+ // without the subtree.
+ auto IsSplatSubtreeProfitable = [&](TreeEntry *TE,
+ const ValuesToInsertTy &ValuesToInsert,
+ InstructionCost &CurrentGathersCost,
+ InstructionCost &DroppedGathersCost) {
+ APInt ExtractElts = APInt::getZero(TE->getVectorFactor());
+ for (Value *V : TE->Scalars) {
+ if (!isa<Instruction>(V) || TE->isCopyableElement(V))
+ continue;
+ // Too many users - the scalar is extracted anyway.
+ if (V->hasNUsesOrMore(UsesLimit) || any_of(V->users(), [&](User *U) {
+ return none_of(getTreeEntries(U), [&](const TreeEntry *UseTE) {
+ return !DeletedNodes.contains(UseTE) &&
+ !TransformedToGatherNodes.contains(UseTE);
+ });
+ }))
+ ExtractElts.setBit(TE->findLaneForValue(V));
+ }
+ Type *ScalarTy = GetScalarTy(TE);
+ InstructionCost KeepCost = getScalarizationOverhead(
+ *TTI, SLPReVec, ScalarTy,
+ cast<VectorType>(getWidenedType(ScalarTy, TE->getVectorFactor())),
+ ExtractElts, /*Insert=*/false, /*Extract=*/true, CostKind);
+ // Add the cost of the subtree itself, computed before any trimming:
+ // trimming of the subtree's own nodes would otherwise make it look
+ // artificially cheap.
+ KeepCost += std::get<1>(SubtreeCosts[TE->Idx]);
+ // The reusing gather nodes currently pay the broadcast cost; without
+ // the subtree they fall back to plain insertion sequences. Gather
+ // nodes already erased from NodesCosts are being deleted and do not
+ // count on either side.
+ CurrentGathersCost = 0;
+ for (const auto &[BVE, _] : ValuesToInsert)
+ CurrentGathersCost += NodesCosts.lookup(BVE);
+ KeepCost += CurrentGathersCost;
+ // Re-cost the gather nodes with the subtree tentatively deleted.
+ DeletedNodes.insert(TE);
+ SmallVector<TreeEntry *> TempDeleted;
+ for (unsigned Idx : std::get<2>(SubtreeCosts[TE->Idx])) {
+ TreeEntry *Child = VectorizableTree[Idx].get();
+ if (DeletedNodes.insert(Child).second)
+ TempDeleted.push_back(Child);
+ }
+ DroppedGathersCost = 0;
+ for (const auto &[BVE, _] : ValuesToInsert) {
+ if (!NodesCosts.contains(BVE))
+ continue;
+ DroppedGathersCost += RecostEntry(BVE);
+ }
+ DeletedNodes.erase(TE);
+ for (TreeEntry *Child : TempDeleted)
+ DeletedNodes.erase(Child);
+ return KeepCost <= DroppedGathersCost;
+ };
+ if (!Changed) {
+ // The splat subtrees are not linked to the tree root, so their cost is
+ // not included in the root's subtree cost; add it explicitly. Drop the
+ // unprofitable ones instead of letting them reject the whole tree.
+ InstructionCost TotalCost = std::get<1>(SubtreeCosts.front());
+ for (TreeEntry *TE : SplatGatheredScalarsRoots) {
+ ValuesToInsertTy ValuesToInsert;
+ InstructionCost CurrentGathersCost = 0, DroppedGathersCost = 0;
+ if (!FindDemandedElts(TE, ValuesToInsert).isZero() &&
+ IsSplatSubtreeProfitable(TE, ValuesToInsert, CurrentGathersCost,
+ DroppedGathersCost)) {
+ TotalCost += std::get<1>(SubtreeCosts[TE->Idx]);
+ continue;
+ }
+ DeletedNodes.insert(TE);
+ for (unsigned Idx : std::get<2>(SubtreeCosts[TE->Idx]))
+ DeletedNodes.insert(VectorizableTree[Idx].get());
+ // The gather nodes that reused the subtree are re-emitted without it.
+ TotalCost += DroppedGathersCost - CurrentGathersCost;
+ }
+ // Gathered loads subtrees left without surviving gather users are dead.
+ for (TreeEntry *TE : GatheredLoadsNodes) {
+ if (DeletedNodes.contains(TE))
+ continue;
+ ValuesToInsertTy ValuesToInsert;
+ if (!FindDemandedElts(TE, ValuesToInsert).isZero())
+ continue;
+ DeletedNodes.insert(TE);
+ for (unsigned Idx : std::get<2>(SubtreeCosts[TE->Idx]))
+ DeletedNodes.insert(VectorizableTree[Idx].get());
+ }
+ return TotalCost;
+ }
+
+ SmallPtrSet<TreeEntry *, 4> SubtreesToDelete;
+ SmallPtrSet<TreeEntry *, 4> DroppedSplatSubtrees;
+ InstructionCost LoadsExtractsCost = 0;
// Check if all loads of gathered loads nodes are marked for deletion. In this
// case the whole gathered loads subtree must be deleted.
// Also, try to account for extracts, which might be required, if only part of
@@ -19152,32 +19247,9 @@ BoUpSLP::calculateTreeCostAndTrimNonProfitable(ArrayRef<Value *> VectorizedVals,
ValuesToInsertTy ValuesToInsert;
APInt DemandedElts = FindDemandedElts(TE, ValuesToInsert);
if (!DemandedElts.isZero()) {
- Type *ScalarTy = GetScalarTy(TE);
- // Lanes of the subtree scalars still used by the remaining scalar code
- // must be extracted if the subtree is kept.
- APInt ExtractElts = APInt::getZero(TE->getVectorFactor());
- for (Value *V : TE->Scalars) {
- if (!isa<Instruction>(V) || TE->isCopyableElement(V))
- continue;
- // Too many users - the scalar is extracted anyway.
- if (V->hasNUsesOrMore(UsesLimit) || any_of(V->users(), [&](User *U) {
- return none_of(getTreeEntries(U), [&](const TreeEntry *UseTE) {
- return !DeletedNodes.contains(UseTE) &&
- !TransformedToGatherNodes.contains(UseTE);
- });
- }))
- ExtractElts.setBit(TE->findLaneForValue(V));
- }
- InstructionCost KeepCost = getScalarizationOverhead(
- *TTI, SLPReVec, ScalarTy,
- cast<VectorType>(getWidenedType(ScalarTy, TE->getVectorFactor())),
- ExtractElts, /*Insert=*/false, /*Extract=*/true, CostKind);
- // Add the cost of the subtree itself, computed before any trimming:
- // trimming of the subtree's own nodes would otherwise make it look
- // artificially cheap.
- KeepCost += std::get<1>(SubtreeCosts[TE->Idx]);
- InstructionCost DropCost = GetGatherInsertCost(ScalarTy, ValuesToInsert);
- if (KeepCost <= DropCost)
+ InstructionCost CurrentGathersCost, DroppedGathersCost;
+ if (IsSplatSubtreeProfitable(TE, ValuesToInsert, CurrentGathersCost,
+ DroppedGathersCost))
continue;
// Dropped as unprofitable: exclude its cost from the reference cost, so
// the trimming of the remaining tree is not reverted because of it, and
@@ -19214,19 +19286,8 @@ BoUpSLP::calculateTreeCostAndTrimNonProfitable(ArrayRef<Value *> VectorizedVals,
// Gather costs depend on the set of vectorized nodes available for
// reuse, which changes during trimming, so recalculate them for all
// gather nodes, not just for the transformed ones.
- if (TE->isGather() || !NodesCosts.contains(TE.get())) {
- InstructionCost C =
- getEntryCost(TE.get(), VectorizedVals, CheckedExtracts);
- if (!C.isValid() || C == 0) {
- NodesCosts[TE.get()] = C;
- continue;
- }
- uint64_t Scale = EntryToScale.lookup(TE.get());
- if (!Scale)
- Scale = getEntryEffectiveScale(*TE);
- C *= Scale;
- NodesCosts[TE.get()] = C;
- }
+ if (TE->isGather() || !NodesCosts.contains(TE.get()))
+ NodesCosts[TE.get()] = RecostEntry(TE.get());
}
LLVM_DEBUG(dbgs() << "SLP: Recalculate costs after tree trimming.\n");
diff --git a/llvm/test/Transforms/SLPVectorizer/AArch64/splat-gather-subtree-drop.ll b/llvm/test/Transforms/SLPVectorizer/AArch64/splat-gather-subtree-drop.ll
index c6677935dacda..65dbbd74e8938 100644
--- a/llvm/test/Transforms/SLPVectorizer/AArch64/splat-gather-subtree-drop.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AArch64/splat-gather-subtree-drop.ll
@@ -10,27 +10,25 @@ define i32 @test1(ptr %p, ptr %q, i32 %seed) {
; CHECK-SAME: ptr [[P:%.*]], ptr [[Q:%.*]], i32 [[SEED:%.*]]) #[[ATTR0:[0-9]+]] {
; CHECK-NEXT: [[ENTRY:.*:]]
; CHECK-NEXT: [[V0:%.*]] = load i8, ptr [[P]], align 1
-; CHECK-NEXT: [[V1:%.*]] = zext i8 [[V0]] to i32
-; CHECK-NEXT: [[V2:%.*]] = add nuw nsw i32 [[V1]], 1
; CHECK-NEXT: [[V3:%.*]] = load i8, ptr [[Q]], align 1
; CHECK-NEXT: [[V4:%.*]] = zext i8 [[V3]] to i32
-; CHECK-NEXT: [[V5:%.*]] = or i32 [[V2]], [[V4]]
-; CHECK-NEXT: [[V6:%.*]] = and i32 [[V4]], 1
-; CHECK-NEXT: [[V7:%.*]] = add nuw nsw i32 [[V6]], 1
-; CHECK-NEXT: [[V8:%.*]] = xor i32 [[V7]], 1
-; CHECK-NEXT: [[V9:%.*]] = add nuw nsw i32 [[V5]], 1
-; CHECK-NEXT: [[V10:%.*]] = and i32 [[V5]], 1
-; CHECK-NEXT: [[V11:%.*]] = xor i32 [[V9]], [[V10]]
-; CHECK-NEXT: [[V12:%.*]] = or i32 [[V8]], [[V11]]
-; CHECK-NEXT: [[V13:%.*]] = add nsw i32 [[V1]], -2
-; CHECK-NEXT: [[V14:%.*]] = or i32 [[V13]], [[V4]]
-; CHECK-NEXT: [[V15:%.*]] = and i32 [[SEED]], 1
-; CHECK-NEXT: [[V16:%.*]] = add nuw nsw i32 [[V15]], 1
-; CHECK-NEXT: [[V17:%.*]] = xor i32 [[V16]], 1
-; CHECK-NEXT: [[V18:%.*]] = add nsw i32 [[V14]], 1
-; CHECK-NEXT: [[V19:%.*]] = and i32 [[V14]], 1
-; CHECK-NEXT: [[V20:%.*]] = xor i32 [[V18]], [[V19]]
-; CHECK-NEXT: [[V21:%.*]] = or i32 [[V17]], [[V20]]
+; CHECK-NEXT: [[V1:%.*]] = zext i8 [[V0]] to i32
+; CHECK-NEXT: [[TMP0:%.*]] = insertelement <2 x i32> poison, i32 [[V1]], i64 0
+; CHECK-NEXT: [[TMP1:%.*]] = shufflevector <2 x i32> [[TMP0]], <2 x i32> poison, <2 x i32> zeroinitializer
+; CHECK-NEXT: [[TMP2:%.*]] = add nsw <2 x i32> [[TMP1]], <i32 1, i32 -2>
+; CHECK-NEXT: [[TMP3:%.*]] = insertelement <2 x i32> poison, i32 [[V4]], i64 0
+; CHECK-NEXT: [[TMP4:%.*]] = shufflevector <2 x i32> [[TMP3]], <2 x i32> poison, <2 x i32> zeroinitializer
+; CHECK-NEXT: [[TMP5:%.*]] = or <2 x i32> [[TMP2]], [[TMP4]]
+; CHECK-NEXT: [[TMP6:%.*]] = insertelement <2 x i32> [[TMP4]], i32 [[SEED]], i64 1
+; CHECK-NEXT: [[TMP7:%.*]] = and <2 x i32> [[TMP6]], splat (i32 1)
+; CHECK-NEXT: [[TMP8:%.*]] = add nuw nsw <2 x i32> [[TMP7]], splat (i32 1)
+; CHECK-NEXT: [[TMP9:%.*]] = xor <2 x i32> [[TMP8]], splat (i32 1)
+; CHECK-NEXT: [[TMP10:%.*]] = add nsw <2 x i32> [[TMP5]], splat (i32 1)
+; CHECK-NEXT: [[TMP11:%.*]] = and <2 x i32> [[TMP5]], splat (i32 1)
+; CHECK-NEXT: [[TMP12:%.*]] = xor <2 x i32> [[TMP10]], [[TMP11]]
+; CHECK-NEXT: [[TMP13:%.*]] = or <2 x i32> [[TMP9]], [[TMP12]]
+; CHECK-NEXT: [[V12:%.*]] = extractelement <2 x i32> [[TMP13]], i64 0
+; CHECK-NEXT: [[V21:%.*]] = extractelement <2 x i32> [[TMP13]], i64 1
; CHECK-NEXT: [[V22:%.*]] = or i32 [[V12]], [[V21]]
; CHECK-NEXT: ret i32 [[V22]]
;
@@ -69,35 +67,30 @@ define void @test2(ptr %out, ptr %in, i64 %n, double %a0, double %a1, double %a2
; CHECK-LABEL: define void @test2(
; CHECK-SAME: ptr [[OUT:%.*]], ptr [[IN:%.*]], i64 [[N:%.*]], double [[A0:%.*]], double [[A1:%.*]], double [[A2:%.*]], double [[A3:%.*]], double [[A10:%.*]], double [[A11:%.*]]) #[[ATTR0]] {
; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: [[TMP0:%.*]] = insertelement <2 x double> poison, double [[A10]], i64 0
+; CHECK-NEXT: [[TMP1:%.*]] = insertelement <2 x double> [[TMP0]], double [[A11]], i64 1
; CHECK-NEXT: br label %[[BODY:.*]]
; CHECK: [[BODY]]:
; CHECK-NEXT: [[I:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[NEXT:%.*]], %[[BODY]] ]
-; CHECK-NEXT: [[X0:%.*]] = load double, ptr [[IN]], align 8
-; CHECK-NEXT: [[P1:%.*]] = getelementptr double, ptr [[IN]], i64 1
-; CHECK-NEXT: [[X1:%.*]] = load double, ptr [[P1]], align 8
; CHECK-NEXT: [[P2:%.*]] = getelementptr double, ptr [[IN]], i64 2
-; CHECK-NEXT: [[X2:%.*]] = load double, ptr [[P2]], align 8
-; CHECK-NEXT: [[P3:%.*]] = getelementptr double, ptr [[IN]], i64 3
-; CHECK-NEXT: [[X3:%.*]] = load double, ptr [[P3]], align 8
; CHECK-NEXT: [[S0:%.*]] = fdiv double [[A0]], [[A1]]
; CHECK-NEXT: [[S1:%.*]] = fsub double [[A2]], [[A3]]
-; CHECK-NEXT: [[V0_0:%.*]] = fadd double [[X0]], [[A10]]
-; CHECK-NEXT: [[V1_0:%.*]] = fsub double [[X1]], [[A11]]
-; CHECK-NEXT: [[V0_1:%.*]] = fsub double [[V0_0]], [[S1]]
-; CHECK-NEXT: [[V1_1:%.*]] = fsub double [[V1_0]], [[S1]]
-; CHECK-NEXT: [[V0_2:%.*]] = fsub double [[V0_1]], [[S0]]
-; CHECK-NEXT: [[V1_2:%.*]] = fsub double [[V1_1]], [[S0]]
-; CHECK-NEXT: [[V0_3:%.*]] = fadd double [[V0_2]], [[A10]]
-; CHECK-NEXT: [[V1_3:%.*]] = fadd double [[V1_2]], [[A11]]
-; CHECK-NEXT: [[V0_4:%.*]] = fsub double [[V0_3]], [[S0]]
-; CHECK-NEXT: [[V1_4:%.*]] = fsub double [[V1_3]], [[S0]]
-; CHECK-NEXT: [[V0_5:%.*]] = fadd double [[V0_4]], [[S0]]
-; CHECK-NEXT: [[V1_5:%.*]] = fadd double [[V1_4]], [[S0]]
-; CHECK-NEXT: [[V0_6:%.*]] = fsub double [[V0_5]], [[X2]]
-; CHECK-NEXT: [[V1_6:%.*]] = fsub double [[V1_5]], [[X3]]
-; CHECK-NEXT: store double [[V0_6]], ptr [[OUT]], align 8
-; CHECK-NEXT: [[O1:%.*]] = getelementptr double, ptr [[OUT]], i64 1
-; CHECK-NEXT: store double [[V1_6]], ptr [[O1]], align 8
+; CHECK-NEXT: [[TMP2:%.*]] = load <2 x double>, ptr [[IN]], align 8
+; CHECK-NEXT: [[TMP3:%.*]] = load <2 x double>, ptr [[P2]], align 8
+; CHECK-NEXT: [[TMP4:%.*]] = fadd <2 x double> [[TMP2]], [[TMP1]]
+; CHECK-NEXT: [[TMP5:%.*]] = fsub <2 x double> [[TMP2]], [[TMP1]]
+; CHECK-NEXT: [[TMP6:%.*]] = shufflevector <2 x double> [[TMP4]], <2 x double> [[TMP5]], <2 x i32> <i32 0, i32 3>
+; CHECK-NEXT: [[TMP7:%.*]] = insertelement <2 x double> poison, double [[S1]], i64 0
+; CHECK-NEXT: [[TMP8:%.*]] = shufflevector <2 x double> [[TMP7]], <2 x double> poison, <2 x i32> zeroinitializer
+; CHECK-NEXT: [[TMP9:%.*]] = fsub <2 x double> [[TMP6]], [[TMP8]]
+; CHECK-NEXT: [[TMP10:%.*]] = insertelement <2 x double> poison, double [[S0]], i64 0
+; CHECK-NEXT: [[TMP11:%.*]] = shufflevector <2 x double> [[TMP10]], <2 x double> poison, <2 x i32> zeroinitializer
+; CHECK-NEXT: [[TMP12:%.*]] = fsub <2 x double> [[TMP9]], [[TMP11]]
+; CHECK-NEXT: [[TMP13:%.*]] = fadd <2 x double> [[TMP12]], [[TMP1]]
+; CHECK-NEXT: [[TMP14:%.*]] = fsub <2 x double> [[TMP13]], [[TMP11]]
+; CHECK-NEXT: [[TMP15:%.*]] = fadd <2 x double> [[TMP14]], [[TMP11]]
+; CHECK-NEXT: [[TMP16:%.*]] = fsub <2 x double> [[TMP15]], [[TMP3]]
+; CHECK-NEXT: store <2 x double> [[TMP16]], ptr [[OUT]], align 8
; CHECK-NEXT: [[NEXT]] = add i64 [[I]], 1
; CHECK-NEXT: [[MORE:%.*]] = icmp ult i64 [[NEXT]], [[N]]
; CHECK-NEXT: br i1 [[MORE]], label %[[BODY]], label %[[EXIT:.*]]
diff --git a/llvm/test/Transforms/SLPVectorizer/AArch64/splat-gather-subtree-store-chain.ll b/llvm/test/Transforms/SLPVectorizer/AArch64/splat-gather-subtree-store-chain.ll
index dd0427692a9f4..a70ee00f98b96 100644
--- a/llvm/test/Transforms/SLPVectorizer/AArch64/splat-gather-subtree-store-chain.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AArch64/splat-gather-subtree-store-chain.ll
@@ -64,53 +64,38 @@ define void @splat_subtree_with_scalar_uses(ptr noalias %out, ptr noalias %in) {
; CHECK-LABEL: define void @splat_subtree_with_scalar_uses(
; CHECK-SAME: ptr noalias [[OUT:%.*]], ptr noalias [[IN:%.*]]) {
; CHECK-NEXT: [[ENTRY_RTVEC:.*:]]
-; CHECK-NEXT: [[TMP0:%.*]] = load i32, ptr [[IN]], align 4
; CHECK-NEXT: [[ARRAYIDX1:%.*]] = getelementptr inbounds nuw i8, ptr [[IN]], i64 4
-; CHECK-NEXT: [[TMP1:%.*]] = load i32, ptr [[ARRAYIDX1]], align 4
-; CHECK-NEXT: [[TMP5:%.*]] = add i32 [[TMP1]], [[TMP0]]
; CHECK-NEXT: [[ARRAYIDX2:%.*]] = getelementptr inbounds nuw i8, ptr [[IN]], i64 8
-; CHECK-NEXT: [[TMP7:%.*]] = load i32, ptr [[ARRAYIDX2]], align 4
; CHECK-NEXT: [[ARRAYIDX3:%.*]] = getelementptr inbounds nuw i8, ptr [[IN]], i64 12
-; CHECK-NEXT: [[TMP9:%.*]] = load i32, ptr [[ARRAYIDX3]], align 4
-; CHECK-NEXT: [[ADD5:%.*]] = add i32 [[TMP9]], [[TMP7]]
; CHECK-NEXT: [[ARRAYIDX5:%.*]] = getelementptr inbounds nuw i8, ptr [[IN]], i64 16
-; CHECK-NEXT: [[TMP10:%.*]] = load i32, ptr [[ARRAYIDX5]], align 4
; CHECK-NEXT: [[ARRAYIDX7:%.*]] = getelementptr inbounds nuw i8, ptr [[IN]], i64 20
-; CHECK-NEXT: [[TMP11:%.*]] = load i32, ptr [[ARRAYIDX7]], align 4
-; CHECK-NEXT: [[TMP3:%.*]] = add i32 [[TMP11]], [[TMP10]]
; CHECK-NEXT: [[ARRAYIDX8:%.*]] = getelementptr inbounds nuw i8, ptr [[IN]], i64 24
-; CHECK-NEXT: [[TMP12:%.*]] = load i32, ptr [[ARRAYIDX8]], align 4
-; CHECK-NEXT: [[ADD9:%.*]] = add i32 [[TMP12]], [[TMP5]]
-; CHECK-NEXT: [[XOR:%.*]] = xor i32 [[ADD9]], [[ADD5]]
-; CHECK-NEXT: [[ADD10:%.*]] = add i32 [[XOR]], [[TMP3]]
-; CHECK-NEXT: store i32 [[ADD10]], ptr [[OUT]], align 4
-; CHECK-NEXT: [[ARRAYIDX6:%.*]] = getelementptr inbounds nuw i8, ptr [[IN]], i64 28
-; CHECK-NEXT: [[TMP6:%.*]] = load i32, ptr [[ARRAYIDX6]], align 4
-; CHECK-NEXT: [[ADD13:%.*]] = add i32 [[TMP6]], [[TMP5]]
-; CHECK-NEXT: [[XOR14:%.*]] = xor i32 [[ADD13]], [[ADD5]]
-; CHECK-NEXT: [[ADD15:%.*]] = add i32 [[XOR14]], [[TMP3]]
-; CHECK-NEXT: [[ARRAYIDX16:%.*]] = getelementptr inbounds nuw i8, ptr [[OUT]], i64 4
-; CHECK-NEXT: store i32 [[ADD15]], ptr [[ARRAYIDX16]], align 4
-; CHECK-NEXT: [[ARRAYIDX17:%.*]] = getelementptr inbounds nuw i8, ptr [[IN]], i64 32
-; CHECK-NEXT: [[TMP8:%.*]] = load i32, ptr [[ARRAYIDX17]], align 4
-; CHECK-NEXT: [[ADD18:%.*]] = add i32 [[TMP8]], [[TMP5]]
-; CHECK-NEXT: [[TMP2:%.*]] = xor i32 [[ADD18]], [[ADD5]]
-; CHECK-NEXT: [[ADD4:%.*]] = add i32 [[TMP2]], [[TMP3]]
-; CHECK-NEXT: [[ARRAYIDX21:%.*]] = getelementptr inbounds nuw i8, ptr [[OUT]], i64 8
-; CHECK-NEXT: store i32 [[ADD4]], ptr [[ARRAYIDX21]], align 4
-; CHECK-NEXT: [[ARRAYIDX22:%.*]] = getelementptr inbounds nuw i8, ptr [[IN]], i64 36
-; CHECK-NEXT: [[TMP4:%.*]] = load i32, ptr [[ARRAYIDX22]], align 4
+; CHECK-NEXT: [[XOR24:%.*]] = load i32, ptr [[ARRAYIDX3]], align 4
+; CHECK-NEXT: [[TMP3:%.*]] = load i32, ptr [[ARRAYIDX2]], align 4
+; CHECK-NEXT: [[TMP2:%.*]] = load i32, ptr [[ARRAYIDX1]], align 4
+; CHECK-NEXT: [[TMP16:%.*]] = load i32, ptr [[IN]], align 4
+; CHECK-NEXT: [[TMP4:%.*]] = load i32, ptr [[ARRAYIDX7]], align 4
+; CHECK-NEXT: [[TMP5:%.*]] = load i32, ptr [[ARRAYIDX5]], align 4
; CHECK-NEXT: [[ADD:%.*]] = add i32 [[TMP4]], [[TMP5]]
-; CHECK-NEXT: [[XOR24:%.*]] = xor i32 [[ADD]], [[ADD5]]
; CHECK-NEXT: [[ADD25:%.*]] = add i32 [[XOR24]], [[TMP3]]
-; CHECK-NEXT: [[ARRAYIDX26:%.*]] = getelementptr inbounds nuw i8, ptr [[OUT]], i64 12
-; CHECK-NEXT: store i32 [[ADD25]], ptr [[ARRAYIDX26]], align 4
+; CHECK-NEXT: [[ADD1:%.*]] = add i32 [[TMP2]], [[TMP16]]
+; CHECK-NEXT: [[TMP6:%.*]] = load <4 x i32>, ptr [[ARRAYIDX8]], align 4
+; CHECK-NEXT: [[TMP7:%.*]] = insertelement <4 x i32> poison, i32 [[ADD1]], i64 0
+; CHECK-NEXT: [[TMP8:%.*]] = shufflevector <4 x i32> [[TMP7]], <4 x i32> poison, <4 x i32> zeroinitializer
+; CHECK-NEXT: [[TMP9:%.*]] = add <4 x i32> [[TMP6]], [[TMP8]]
+; CHECK-NEXT: [[TMP10:%.*]] = insertelement <4 x i32> poison, i32 [[ADD25]], i64 0
+; CHECK-NEXT: [[TMP11:%.*]] = shufflevector <4 x i32> [[TMP10]], <4 x i32> poison, <4 x i32> zeroinitializer
+; CHECK-NEXT: [[TMP12:%.*]] = xor <4 x i32> [[TMP9]], [[TMP11]]
+; CHECK-NEXT: [[TMP13:%.*]] = insertelement <4 x i32> poison, i32 [[ADD]], i64 0
+; CHECK-NEXT: [[TMP14:%.*]] = shufflevector <4 x i32> [[TMP13]], <4 x i32> poison, <4 x i32> zeroinitializer
+; CHECK-NEXT: [[TMP15:%.*]] = add <4 x i32> [[TMP12]], [[TMP14]]
+; CHECK-NEXT: store <4 x i32> [[TMP15]], ptr [[OUT]], align 4
; CHECK-NEXT: [[ARRAYIDX27_SCALAR:%.*]] = getelementptr inbounds nuw i8, ptr [[OUT]], i64 16
-; CHECK-NEXT: store i32 [[TMP5]], ptr [[ARRAYIDX27_SCALAR]], align 4
+; CHECK-NEXT: store i32 [[ADD1]], ptr [[ARRAYIDX27_SCALAR]], align 4
; CHECK-NEXT: [[ARRAYIDX28_SCALAR:%.*]] = getelementptr inbounds nuw i8, ptr [[OUT]], i64 24
-; CHECK-NEXT: store i32 [[ADD5]], ptr [[ARRAYIDX28_SCALAR]], align 4
+; CHECK-NEXT: store i32 [[ADD25]], ptr [[ARRAYIDX28_SCALAR]], align 4
; CHECK-NEXT: [[ARRAYIDX29_SCALAR:%.*]] = getelementptr inbounds nuw i8, ptr [[OUT]], i64 32
-; CHECK-NEXT: store i32 [[TMP3]], ptr [[ARRAYIDX29_SCALAR]], align 4
+; CHECK-NEXT: store i32 [[ADD]], ptr [[ARRAYIDX29_SCALAR]], align 4
; CHECK-NEXT: ret void
;
entry:
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/lookahead.ll b/llvm/test/Transforms/SLPVectorizer/X86/lookahead.ll
index e96c0c32272a7..1c85515164032 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/lookahead.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/lookahead.ll
@@ -386,17 +386,33 @@ define void @lookahead_crash(ptr %A, ptr %S, ptr %Arg0) {
; This checks that we choose to group consecutive extracts from the same vectors.
define void @ChecksExtractScores(ptr %storeArray, ptr %array, ptr %vecPtr1, ptr %vecPtr2) {
-; CHECK-LABEL: @ChecksExtractScores(
-; CHECK-NEXT: [[LOADVEC:%.*]] = load <2 x double>, ptr [[VECPTR1:%.*]], align 4
-; CHECK-NEXT: [[LOADVEC2:%.*]] = load <2 x double>, ptr [[VECPTR2:%.*]], align 4
-; CHECK-NEXT: [[TMP1:%.*]] = load <2 x double>, ptr [[ARRAY:%.*]], align 4
-; CHECK-NEXT: [[TMP2:%.*]] = shufflevector <2 x double> [[TMP1]], <2 x double> poison, <2 x i32> zeroinitializer
-; CHECK-NEXT: [[TMP3:%.*]] = fmul <2 x double> [[LOADVEC]], [[TMP2]]
-; CHECK-NEXT: [[TMP5:%.*]] = shufflevector <2 x double> [[TMP1]], <2 x double> poison, <2 x i32> <i32 1, i32 1>
-; CHECK-NEXT: [[TMP6:%.*]] = fmul <2 x double> [[LOADVEC2]], [[TMP5]]
-; CHECK-NEXT: [[TMP7:%.*]] = fadd <2 x double> [[TMP3]], [[TMP6]]
-; CHECK-NEXT: store <2 x double> [[TMP7]], ptr [[STOREARRAY:%.*]], align 8
-; CHECK-NEXT: ret void
+; SSE-LABEL: @ChecksExtractScores(
+; SSE-NEXT: [[LOADVEC:%.*]] = load <2 x double>, ptr [[VECPTR1:%.*]], align 4
+; SSE-NEXT: [[LOADVEC2:%.*]] = load <2 x double>, ptr [[VECPTR2:%.*]], align 4
+; SSE-NEXT: [[TMP1:%.*]] = load <2 x double>, ptr [[ARRAY:%.*]], align 4
+; SSE-NEXT: [[TMP2:%.*]] = shufflevector <2 x double> [[TMP1]], <2 x double> poison, <2 x i32> zeroinitializer
+; SSE-NEXT: [[TMP3:%.*]] = fmul <2 x double> [[LOADVEC]], [[TMP2]]
+; SSE-NEXT: [[TMP4:%.*]] = shufflevector <2 x double> [[TMP1]], <2 x double> poison, <2 x i32> <i32 1, i32 1>
+; SSE-NEXT: [[TMP5:%.*]] = fmul <2 x double> [[LOADVEC2]], [[TMP4]]
+; SSE-NEXT: [[TMP6:%.*]] = fadd <2 x double> [[TMP3]], [[TMP5]]
+; SSE-NEXT: store <2 x double> [[TMP6]], ptr [[STOREARRAY:%.*]], align 8
+; SSE-NEXT: ret void
+;
+; AVX-LABEL: @ChecksExtractScores(
+; AVX-NEXT: [[IDX1:%.*]] = getelementptr inbounds double, ptr [[ARRAY:%.*]], i64 1
+; AVX-NEXT: [[LOADVEC:%.*]] = load <2 x double>, ptr [[VECPTR1:%.*]], align 4
+; AVX-NEXT: [[LOADVEC2:%.*]] = load <2 x double>, ptr [[VECPTR2:%.*]], align 4
+; AVX-NEXT: [[LOADA1:%.*]] = load double, ptr [[IDX1]], align 4
+; AVX-NEXT: [[LOADA0:%.*]] = load double, ptr [[ARRAY]], align 4
+; AVX-NEXT: [[TMP1:%.*]] = insertelement <2 x double> poison, double [[LOADA0]], i64 0
+; AVX-NEXT: [[TMP2:%.*]] = shufflevector <2 x double> [[TMP1]], <2 x double> poison, <2 x i32> zeroinitializer
+; AVX-NEXT: [[TMP3:%.*]] = fmul <2 x double> [[LOADVEC]], [[TMP2]]
+; AVX-NEXT: [[TMP4:%.*]] = insertelement <2 x double> poison, double [[LOADA1]], i64 0
+; AVX-NEXT: [[TMP5:%.*]] = shufflevector <2 x double> [[TMP4]], <2 x double> poison, <2 x i32> zeroinitializer
+; AVX-NEXT: [[TMP6:%.*]] = fmul <2 x double> [[LOADVEC2]], [[TMP5]]
+; AVX-NEXT: [[TMP7:%.*]] = fadd <2 x double> [[TMP3]], [[TMP6]]
+; AVX-NEXT: store <2 x double> [[TMP7]], ptr [[STOREARRAY:%.*]], align 8
+; AVX-NEXT: ret void
;
%idx1 = getelementptr inbounds double, ptr %array, i64 1
%loadA0 = load double, ptr %array, align 4
@@ -531,16 +547,20 @@ define void @ChecksExtractScores_
diff erent_vectors(ptr %storeArray, ptr %array,
; SSE-NEXT: ret void
;
; AVX-LABEL: @ChecksExtractScores_
diff erent_vectors(
-; AVX-NEXT: [[LOADVEC:%.*]] = load <2 x double>, ptr [[VECPTR1:%.*]], align 4
+; AVX-NEXT: [[IDX1:%.*]] = getelementptr inbounds double, ptr [[ARRAY1:%.*]], i64 1
; AVX-NEXT: [[LOADVEC2:%.*]] = load <2 x double>, ptr [[VECPTR2:%.*]], align 4
; AVX-NEXT: [[LOADVEC3:%.*]] = load <2 x double>, ptr [[VECPTR3:%.*]], align 4
; AVX-NEXT: [[LOADVEC4:%.*]] = load <2 x double>, ptr [[VECPTR4:%.*]], align 4
; AVX-NEXT: [[TMP2:%.*]] = load <2 x double>, ptr [[ARRAY:%.*]], align 4
-; AVX-NEXT: [[TMP1:%.*]] = shufflevector <2 x double> [[LOADVEC]], <2 x double> [[LOADVEC2]], <2 x i32> <i32 0, i32 3>
-; AVX-NEXT: [[TMP3:%.*]] = shufflevector <2 x double> [[TMP2]], <2 x double> poison, <2 x i32> zeroinitializer
+; AVX-NEXT: [[LOADA1:%.*]] = load double, ptr [[IDX1]], align 4
+; AVX-NEXT: [[LOADA0:%.*]] = load double, ptr [[ARRAY1]], align 4
+; AVX-NEXT: [[TMP1:%.*]] = shufflevector <2 x double> [[LOADVEC2]], <2 x double> [[LOADVEC3]], <2 x i32> <i32 0, i32 3>
+; AVX-NEXT: [[TMP10:%.*]] = insertelement <2 x double> poison, double [[LOADA0]], i64 0
+; AVX-NEXT: [[TMP3:%.*]] = shufflevector <2 x double> [[TMP10]], <2 x double> poison, <2 x i32> zeroinitializer
; AVX-NEXT: [[TMP4:%.*]] = fmul <2 x double> [[TMP1]], [[TMP3]]
-; AVX-NEXT: [[TMP5:%.*]] = shufflevector <2 x double> [[LOADVEC3]], <2 x double> [[LOADVEC4]], <2 x i32> <i32 0, i32 3>
-; AVX-NEXT: [[TMP7:%.*]] = shufflevector <2 x double> [[TMP2]], <2 x double> poison, <2 x i32> <i32 1, i32 1>
+; AVX-NEXT: [[TMP5:%.*]] = shufflevector <2 x double> [[LOADVEC4]], <2 x double> [[TMP2]], <2 x i32> <i32 0, i32 3>
+; AVX-NEXT: [[TMP6:%.*]] = insertelement <2 x double> poison, double [[LOADA1]], i64 0
+; AVX-NEXT: [[TMP7:%.*]] = shufflevector <2 x double> [[TMP6]], <2 x double> poison, <2 x i32> zeroinitializer
; AVX-NEXT: [[TMP8:%.*]] = fmul <2 x double> [[TMP5]], [[TMP7]]
; AVX-NEXT: [[TMP9:%.*]] = fadd <2 x double> [[TMP4]], [[TMP8]]
; AVX-NEXT: store <2 x double> [[TMP9]], ptr [[STOREARRAY:%.*]], align 8
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/reorder_phi.ll b/llvm/test/Transforms/SLPVectorizer/X86/reorder_phi.ll
index 4a71954781342..a0b653f42e7bf 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/reorder_phi.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/reorder_phi.ll
@@ -13,11 +13,15 @@ define void @foo (ptr %A, ptr %B, ptr %Result) {
; CHECK-NEXT: [[TMP2:%.*]] = phi <2 x float> [ zeroinitializer, [[ENTRY]] ], [ [[TMP20:%.*]], [[LOOP]] ]
; CHECK-NEXT: [[TMP3:%.*]] = getelementptr inbounds [[STRUCT_COMPLEX:%.*]], ptr [[A:%.*]], i64 [[TMP1]], i32 0
; CHECK-NEXT: [[TMP4:%.*]] = getelementptr inbounds [[STRUCT_COMPLEX]], ptr [[B:%.*]], i64 [[TMP1]], i32 0
+; CHECK-NEXT: [[TMP5:%.*]] = getelementptr inbounds [[STRUCT_COMPLEX]], ptr [[B]], i64 [[TMP1]], i32 1
; CHECK-NEXT: [[TMP8:%.*]] = load <2 x float>, ptr [[TMP3]], align 4
-; CHECK-NEXT: [[TMP9:%.*]] = load <2 x float>, ptr [[TMP4]], align 4
+; CHECK-NEXT: [[TMP7:%.*]] = load float, ptr [[TMP5]], align 4
+; CHECK-NEXT: [[TMP22:%.*]] = load float, ptr [[TMP4]], align 4
+; CHECK-NEXT: [[TMP9:%.*]] = insertelement <2 x float> poison, float [[TMP22]], i64 0
; CHECK-NEXT: [[TMP10:%.*]] = shufflevector <2 x float> [[TMP9]], <2 x float> poison, <2 x i32> zeroinitializer
; CHECK-NEXT: [[TMP11:%.*]] = fmul <2 x float> [[TMP8]], [[TMP10]]
-; CHECK-NEXT: [[TMP13:%.*]] = shufflevector <2 x float> [[TMP9]], <2 x float> poison, <2 x i32> <i32 1, i32 1>
+; CHECK-NEXT: [[TMP12:%.*]] = insertelement <2 x float> poison, float [[TMP7]], i64 0
+; CHECK-NEXT: [[TMP13:%.*]] = shufflevector <2 x float> [[TMP12]], <2 x float> poison, <2 x i32> zeroinitializer
; CHECK-NEXT: [[TMP14:%.*]] = fmul <2 x float> [[TMP8]], [[TMP13]]
; CHECK-NEXT: [[TMP15:%.*]] = shufflevector <2 x float> [[TMP14]], <2 x float> poison, <2 x i32> <i32 1, i32 0>
; CHECK-NEXT: [[TMP16:%.*]] = fsub <2 x float> [[TMP11]], [[TMP15]]
More information about the llvm-commits
mailing list