[llvm] [SLP]Allow min-VF vectorization of seed-level reduction groups (PR #222757)
Alexey Bataev via llvm-commits
llvm-commits at lists.llvm.org
Fri Sep 11 10:56:07 PDT 2026
https://github.com/alexey-bataev updated https://github.com/llvm/llvm-project/pull/222757
>From 97edbfbbdb121cf15387bd8a5bcc17a2fb59c14b Mon Sep 17 00:00:00 2001
From: Alexey Bataev <a.bataev at outlook.com>
Date: Thu, 10 Sep 2026 13:14:39 -0700
Subject: [PATCH] =?UTF-8?q?[=F0=9D=98=80=F0=9D=97=BD=F0=9D=97=BF]=20initia?=
=?UTF-8?q?l=20version?=
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
Created using spr 1.3.7
---
.../Transforms/Vectorize/SLPVectorizer.cpp | 81 ++++++++++++++-----
.../X86/fma-reassociate-pairs.ll | 2 +-
.../SLPVectorizer/X86/fma-operand-index.ll | 42 ++++------
.../X86/horizontal-fadd-with-sub.ll | 2 +-
.../SLPVectorizer/X86/horizontal-list.ll | 21 +++--
.../X86/reassociated-fma-reduction.ll | 20 ++---
.../X86/revectorized_rdx_crash.ll | 20 ++---
.../SLPVectorizer/X86/vectorize-pair-path.ll | 12 +--
8 files changed, 112 insertions(+), 88 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 28e665c4c07cf..75b4bda8796d7 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -29846,13 +29846,15 @@ class HorizontalReduction {
/// Try to find a reduction tree. If \p FlattenNegations is false, the
/// reassociable fsub/fneg links of an fadd chain are not flattened (they are
- /// leaves of the reduction).
+ /// leaves of the reduction). \p IsSeedRoot enables the profitability
+ /// ordering of the same-size groups for the seed-level fadd reductions.
bool matchAssociativeReduction(BoUpSLP &R, Instruction *Root,
ScalarEvolution &SE, DominatorTree &DT,
const DataLayout &DL,
const TargetTransformInfo &TTI,
const TargetLibraryInfo &TLI,
- bool FlattenNegations = true) {
+ bool FlattenNegations = true,
+ bool IsSeedRoot = false) {
RdxKind = HorizontalReduction::getRdxKind(Root);
// A reassociable fsub root is a flattened link of an fadd reduction: its
// subtracted operand enters with a flipped sign. Without the flattening
@@ -30238,17 +30240,18 @@ class HorizontalReduction {
// sort them by size.
optimizeReducedVals(R, DT, DL, TTI, TLI);
// Sort the reduced values by number of same/alternate opcode and/or
- // pointer operand. For the sign-aware reductions, same-size groups are
- // ordered by their expected profitability: the first vectorized group
- // is charged the cost of the reduction operation itself, which a group
- // of loads or non-instructions rarely amortizes on its own, while it is
- // a cheap addition (a single vector operation) to an already
- // vectorized group - such groups go last. Among the rest, non-negated
- // groups go first.
+ // pointer operand. For the sign-aware and the seed-level fadd
+ // reductions, same-size groups are ordered by their expected
+ // profitability: the first vectorized group is charged the cost of the
+ // reduction operation itself, which a group of loads or
+ // non-instructions rarely amortizes on its own, while it is a cheap
+ // addition (a single vector operation) to an already vectorized group -
+ // such groups go last. Among the rest, non-negated groups go first.
stable_sort(ReducedVals, [&](ArrayRef<Value *> P1, ArrayRef<Value *> P2) {
if (P1.size() != P2.size())
return P1.size() > P2.size();
- if (NegatedReducedVals.empty())
+ if (NegatedReducedVals.empty() &&
+ !(IsSeedRoot && RdxKind == RecurKind::FAdd))
return false;
auto IsCheapGroup = [](ArrayRef<Value *> P) {
return !isa<Instruction>(P.front()) || isa<LoadInst>(P.front());
@@ -30303,9 +30306,11 @@ class HorizontalReduction {
}
/// Attempt to vectorize the tree found by matchAssociativeReduction.
+ /// \p IsSeedRoot allows vectorizing the groups at the minimum vector
+ /// factor. Set for single-use seed-level roots only.
Value *tryToReduce(BoUpSLP &V, const DataLayout &DL, TargetTransformInfo *TTI,
const TargetLibraryInfo &TLI, AssumptionCache *AC,
- DominatorTree &DT) {
+ DominatorTree &DT, bool IsSeedRoot = false) {
constexpr unsigned RegMaxNumber = 4;
const unsigned RedValsMaxNumber =
(RK == ReductionOrdering::Ordered &&
@@ -30552,6 +30557,14 @@ class HorizontalReduction {
States.push_back(getSameOpcode(RV, TLI));
}
ReducedVals.swap(LocalReducedVals);
+ // The minimum vector factor pays off only when it covers the whole
+ // reduction: every group must be either a pair of distinct values or
+ // large enough for the regular vector factor.
+ const bool NoScalarLeftovers =
+ all_of(ReducedVals, [this](ArrayRef<Value *> Vals) {
+ return Vals.size() >= ReductionLimit ||
+ (Vals.size() == 2 && Vals.front() != Vals.back());
+ });
for (unsigned I = 0, E = ReducedVals.size(); I < E; ++I) {
ArrayRef<Value *> OrigReducedVals = ReducedVals[I];
InstructionsState S = States[I];
@@ -30644,11 +30657,24 @@ class HorizontalReduction {
}
unsigned NumReducedVals = Candidates.size();
+ auto UsedByReductionOnly = [&](Value *V) {
+ if (!V->hasUseList())
+ return true;
+ // Bail out if we have too many uses to save compilation time.
+ if (V->hasNUsesOrMore(UsesLimit))
+ return false;
+ return all_of(V->users(),
+ [&](User *U) { return IgnoreList.contains(U); });
+ };
// Sign-aware reductions pair small positive/negative groups: allow
- // non-splat groups down to 2 elements.
+ // non-splat groups down to 2 elements. Seed-level reductions get the
+ // same for groups whose values are used by the reduction operations
+ // only, if the minimum vector factor covers the whole reduction.
+ const bool MinVFAllowed = IsSeedRoot && NoScalarLeftovers &&
+ all_of(Candidates, UsedByReductionOnly);
if (NumReducedVals < ReductionLimit &&
- (NumReducedVals < 2 ||
- (!isSplat(Candidates) && NegatedReducedVals.empty())))
+ (NumReducedVals < 2 || (!isSplat(Candidates) &&
+ NegatedReducedVals.empty() && !MinVFAllowed)))
continue;
// Check if we support repeated scalar values processing (optimization of
@@ -30782,9 +30808,12 @@ class HorizontalReduction {
};
bool AnyVectorized = false;
SmallDenseSet<std::pair<unsigned, unsigned>, 8> IgnoredCandidates;
- // Same small-group allowance for the vector width.
+ // Same small-group allowance for the vector width, if it covers the
+ // whole group.
const unsigned MinReduxWidth =
- NegatedReducedVals.empty() ? ReductionLimit : 2;
+ !NegatedReducedVals.empty() || (MinVFAllowed && NumReducedVals == 2)
+ ? 2
+ : ReductionLimit;
while (Pos < NumReducedVals - ReduxWidth + 1 &&
ReduxWidth >= MinReduxWidth) {
// Dependency in tree of the reduction ops - drop this attempt, try
@@ -32568,23 +32597,28 @@ bool SLPVectorizerPass::vectorizeHorReduction(
Stack.emplace(SelectRoot(), 0);
SmallPtrSet<Value *, 8> VisitedInstrs;
bool Res = false;
- auto TryToReduce = [this, &R, TTI = TTI](Instruction *Inst) -> Value * {
+ auto TryToReduce = [this, &R, TTI = TTI](Instruction *Inst,
+ bool IsSeedRoot) -> Value * {
if (R.isAnalyzedReductionRoot(Inst))
return nullptr;
if (!isReductionCandidate(Inst))
return nullptr;
HorizontalReduction HorRdx;
Value *Res = nullptr;
- if (HorRdx.matchAssociativeReduction(R, Inst, *SE, *DT, *DL, *TTI, *TLI)) {
- Value *Red = HorRdx.tryToReduce(R, *DL, TTI, *TLI, AC, *DT);
+ if (HorRdx.matchAssociativeReduction(R, Inst, *SE, *DT, *DL, *TTI, *TLI,
+ /*FlattenNegations=*/true,
+ IsSeedRoot)) {
+ Value *Red = HorRdx.tryToReduce(R, *DL, TTI, *TLI, AC, *DT, IsSeedRoot);
// The chain with the flattened fsub/fneg links produced no vector part:
// retry with the fsub/fneg links as leaves, they may be vectorizable on
// their own.
if (!Red && HorRdx.hasFlattenedNegations()) {
HorizontalReduction UnflattenedHorRdx;
if (UnflattenedHorRdx.matchAssociativeReduction(
- R, Inst, *SE, *DT, *DL, *TTI, *TLI, /*FlattenNegations=*/false))
- Red = UnflattenedHorRdx.tryToReduce(R, *DL, TTI, *TLI, AC, *DT);
+ R, Inst, *SE, *DT, *DL, *TTI, *TLI, /*FlattenNegations=*/false,
+ IsSeedRoot))
+ Red = UnflattenedHorRdx.tryToReduce(R, *DL, TTI, *TLI, AC, *DT,
+ IsSeedRoot);
}
if (Red) {
if (Red != Inst)
@@ -32629,7 +32663,10 @@ bool SLPVectorizerPass::vectorizeHorReduction(
// iteration while stack was populated before that happened.
if (R.isDeleted(Inst))
continue;
- if (Value *VectorizedV = TryToReduce(Inst)) {
+ // The minimum-vector-factor allowance applies to single-use seed-level
+ // roots only.
+ if (Value *VectorizedV =
+ TryToReduce(Inst, /*IsSeedRoot=*/Level == 0 && Inst->hasOneUse())) {
Res = true;
if (auto *I = dyn_cast<Instruction>(VectorizedV); I && I != Inst) {
// Try to find another reduction.
diff --git a/llvm/test/Transforms/PhaseOrdering/X86/fma-reassociate-pairs.ll b/llvm/test/Transforms/PhaseOrdering/X86/fma-reassociate-pairs.ll
index 1d2985733a698..f0d5eee0c7283 100644
--- a/llvm/test/Transforms/PhaseOrdering/X86/fma-reassociate-pairs.ll
+++ b/llvm/test/Transforms/PhaseOrdering/X86/fma-reassociate-pairs.ll
@@ -64,7 +64,7 @@ define double @fadd_fmul_2_right(ptr %x, ptr %y, ptr %z) {
; CHECK-NEXT: [[TMP2:%.*]] = load <2 x double>, ptr [[Y]], align 8
; CHECK-NEXT: [[TMP3:%.*]] = fmul reassoc nsz contract <2 x double> [[TMP2]], [[TMP1]]
; CHECK-NEXT: [[TMP4:%.*]] = load <2 x double>, ptr [[Z]], align 8
-; CHECK-NEXT: [[TMP5:%.*]] = fadd reassoc nsz contract <2 x double> [[TMP4]], [[TMP3]]
+; CHECK-NEXT: [[TMP5:%.*]] = fadd reassoc nsz contract <2 x double> [[TMP3]], [[TMP4]]
; CHECK-NEXT: [[R:%.*]] = tail call reassoc nsz contract double @llvm.vector.reduce.fadd.v2f64(double 0.000000e+00, <2 x double> [[TMP5]])
; CHECK-NEXT: ret double [[R]]
;
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/fma-operand-index.ll b/llvm/test/Transforms/SLPVectorizer/X86/fma-operand-index.ll
index 0fa4058f874d2..154d15f4d2dad 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/fma-operand-index.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/fma-operand-index.ll
@@ -1,26 +1,22 @@
; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
; RUN: opt -S --passes=slp-vectorizer -mtriple=x86_64-unknown-linux-gnu -mcpu=znver4 < %s | FileCheck %s
-; A one-use fmul is left scalar so the backend can fuse it, but only operand 0
-; of the fadd/fsub is checked. The same fmul is kept or gathered depending on
-; which side it sits on. To be addressed in a follow-up.
+; Seed-level reassociable fadd/fsub chains vectorize as unordered reductions
+; when the reduced values split into 2-element groups: the groups share the
+; reduction operation cost. The vectorized fmul still fuses into a vector fma
+; in the backend, whichever operand of the fadd/fsub it feeds.
-; Both fmuls are operand 1, so they gather and cannot fuse.
+; Both fmuls are operand 1 of the fadds.
define double @fmul_rhs_fadd(ptr %x, ptr %y, ptr %z) {
; CHECK-LABEL: define double @fmul_rhs_fadd(
; CHECK-SAME: ptr [[X:%.*]], ptr [[Y:%.*]], ptr [[Z:%.*]]) #[[ATTR0:[0-9]+]] {
; CHECK-NEXT: [[ENTRY:.*:]]
-; CHECK-NEXT: [[Z8:%.*]] = getelementptr inbounds nuw i8, ptr [[Z]], i64 8
-; CHECK-NEXT: [[Z0:%.*]] = load double, ptr [[Z]], align 8
; CHECK-NEXT: [[TMP0:%.*]] = load <2 x double>, ptr [[X]], align 8
; CHECK-NEXT: [[TMP1:%.*]] = load <2 x double>, ptr [[Y]], align 8
; CHECK-NEXT: [[TMP2:%.*]] = fmul reassoc nsz contract <2 x double> [[TMP1]], [[TMP0]]
-; CHECK-NEXT: [[Z1:%.*]] = load double, ptr [[Z8]], align 8
-; CHECK-NEXT: [[ZSUM:%.*]] = fadd reassoc nsz contract double [[Z0]], [[Z1]]
-; CHECK-NEXT: [[MUL0:%.*]] = extractelement <2 x double> [[TMP2]], i64 0
-; CHECK-NEXT: [[ADD0:%.*]] = fadd reassoc nsz contract double [[ZSUM]], [[MUL0]]
-; CHECK-NEXT: [[MUL1:%.*]] = extractelement <2 x double> [[TMP2]], i64 1
-; CHECK-NEXT: [[ADD1:%.*]] = fadd reassoc nsz contract double [[ADD0]], [[MUL1]]
+; CHECK-NEXT: [[TMP3:%.*]] = load <2 x double>, ptr [[Z]], align 8
+; CHECK-NEXT: [[RDX_OP:%.*]] = fadd reassoc nsz contract <2 x double> [[TMP2]], [[TMP3]]
+; CHECK-NEXT: [[ADD1:%.*]] = call reassoc nsz contract double @llvm.vector.reduce.fadd.v2f64(double 0.000000e+00, <2 x double> [[RDX_OP]])
; CHECK-NEXT: ret double [[ADD1]]
;
entry:
@@ -72,25 +68,17 @@ entry:
ret double %sub1
}
-; Operand 0 is the one checked, so these stay scalar and fuse.
+; Both fmuls are operand 0 of the fadds.
define double @fmul_lhs_fadd(ptr %x, ptr %y, ptr %z) {
; CHECK-LABEL: define double @fmul_lhs_fadd(
; CHECK-SAME: ptr [[X:%.*]], ptr [[Y:%.*]], ptr [[Z:%.*]]) #[[ATTR0]] {
; CHECK-NEXT: [[ENTRY:.*:]]
-; CHECK-NEXT: [[X8:%.*]] = getelementptr inbounds nuw i8, ptr [[X]], i64 8
-; CHECK-NEXT: [[Y8:%.*]] = getelementptr inbounds nuw i8, ptr [[Y]], i64 8
-; CHECK-NEXT: [[Z8:%.*]] = getelementptr inbounds nuw i8, ptr [[Z]], i64 8
-; CHECK-NEXT: [[X0:%.*]] = load double, ptr [[X]], align 8
-; CHECK-NEXT: [[Y0:%.*]] = load double, ptr [[Y]], align 8
-; CHECK-NEXT: [[MUL0:%.*]] = fmul reassoc nsz contract double [[Y0]], [[X0]]
-; CHECK-NEXT: [[Z0:%.*]] = load double, ptr [[Z]], align 8
-; CHECK-NEXT: [[X1:%.*]] = load double, ptr [[X8]], align 8
-; CHECK-NEXT: [[Y1:%.*]] = load double, ptr [[Y8]], align 8
-; CHECK-NEXT: [[MUL1:%.*]] = fmul reassoc nsz contract double [[Y1]], [[X1]]
-; CHECK-NEXT: [[Z1:%.*]] = load double, ptr [[Z8]], align 8
-; CHECK-NEXT: [[ZSUM:%.*]] = fadd reassoc nsz contract double [[Z0]], [[Z1]]
-; CHECK-NEXT: [[ADD0:%.*]] = fadd reassoc nsz contract double [[MUL0]], [[ZSUM]]
-; CHECK-NEXT: [[ADD1:%.*]] = fadd reassoc nsz contract double [[MUL1]], [[ADD0]]
+; CHECK-NEXT: [[TMP0:%.*]] = load <2 x double>, ptr [[X]], align 8
+; CHECK-NEXT: [[TMP1:%.*]] = load <2 x double>, ptr [[Y]], align 8
+; CHECK-NEXT: [[TMP2:%.*]] = fmul reassoc nsz contract <2 x double> [[TMP1]], [[TMP0]]
+; CHECK-NEXT: [[TMP3:%.*]] = load <2 x double>, ptr [[Z]], align 8
+; CHECK-NEXT: [[RDX_OP:%.*]] = fadd reassoc nsz contract <2 x double> [[TMP2]], [[TMP3]]
+; CHECK-NEXT: [[ADD1:%.*]] = call reassoc nsz contract double @llvm.vector.reduce.fadd.v2f64(double 0.000000e+00, <2 x double> [[RDX_OP]])
; CHECK-NEXT: ret double [[ADD1]]
;
entry:
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/horizontal-fadd-with-sub.ll b/llvm/test/Transforms/SLPVectorizer/X86/horizontal-fadd-with-sub.ll
index e8c566ecdaabc..dc1a2ccfb0c2c 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/horizontal-fadd-with-sub.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/horizontal-fadd-with-sub.ll
@@ -1092,7 +1092,7 @@ define double @opaque_fneg_leaves_mixed(ptr %x, ptr %y) {
; CHECK-NEXT: [[TMP0:%.*]] = load <4 x double>, ptr [[X]], align 8
; CHECK-NEXT: [[TMP1:%.*]] = load <4 x double>, ptr [[Y]], align 8
; CHECK-NEXT: [[TMP2:%.*]] = fneg <4 x double> [[TMP1]]
-; CHECK-NEXT: [[RDX_OP:%.*]] = fadd reassoc nsz contract <4 x double> [[TMP0]], [[TMP2]]
+; CHECK-NEXT: [[RDX_OP:%.*]] = fadd reassoc nsz contract <4 x double> [[TMP2]], [[TMP0]]
; CHECK-NEXT: [[TMP3:%.*]] = call reassoc nsz contract double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[RDX_OP]])
; CHECK-NEXT: ret double [[TMP3]]
;
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/horizontal-list.ll b/llvm/test/Transforms/SLPVectorizer/X86/horizontal-list.ll
index 69822351ffd57..994ff81c9f633 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/horizontal-list.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/horizontal-list.ll
@@ -842,11 +842,11 @@ define float @extra_args_no_replace(ptr nocapture readonly %x, i32 %a, i32 %b, i
; CHECK-NEXT: [[CONV:%.*]] = sitofp i32 [[MUL]] to float
; CHECK-NEXT: [[CONVC:%.*]] = sitofp i32 [[C:%.*]] to float
; CHECK-NEXT: [[TMP0:%.*]] = load <8 x float>, ptr [[X:%.*]], align 4
-; CHECK-NEXT: [[TMP2:%.*]] = fmul reassoc nnan ninf nsz arcp afn float [[CONV]], 2.000000e+00
; CHECK-NEXT: [[TMP3:%.*]] = call fast float @llvm.vector.reduce.fadd.v8f32(float 0.000000e+00, <8 x float> [[TMP0]])
-; CHECK-NEXT: [[OP_RDX:%.*]] = fadd fast float [[TMP2]], [[TMP3]]
-; CHECK-NEXT: [[OP_RDX1:%.*]] = fadd fast float [[OP_RDX]], 3.000000e+00
-; CHECK-NEXT: [[OP_RDX2:%.*]] = fadd fast float [[OP_RDX1]], [[CONVC]]
+; CHECK-NEXT: [[OP_RDX:%.*]] = fadd fast float [[TMP3]], [[CONV]]
+; CHECK-NEXT: [[OP_RDX1:%.*]] = fadd fast float [[CONV]], [[CONVC]]
+; CHECK-NEXT: [[OP_RDX3:%.*]] = fadd fast float [[OP_RDX]], [[OP_RDX1]]
+; CHECK-NEXT: [[OP_RDX2:%.*]] = fadd fast float [[OP_RDX3]], 3.000000e+00
; CHECK-NEXT: ret float [[OP_RDX2]]
;
; THRESHOLD-LABEL: @extra_args_no_replace(
@@ -855,12 +855,17 @@ define float @extra_args_no_replace(ptr nocapture readonly %x, i32 %a, i32 %b, i
; THRESHOLD-NEXT: [[CONV:%.*]] = sitofp i32 [[MUL]] to float
; THRESHOLD-NEXT: [[CONVC:%.*]] = sitofp i32 [[C:%.*]] to float
; THRESHOLD-NEXT: [[TMP0:%.*]] = load <8 x float>, ptr [[X:%.*]], align 4
-; THRESHOLD-NEXT: [[TMP2:%.*]] = fmul reassoc nnan ninf nsz arcp afn float [[CONV]], 2.000000e+00
; THRESHOLD-NEXT: [[TMP3:%.*]] = call fast float @llvm.vector.reduce.fadd.v8f32(float 0.000000e+00, <8 x float> [[TMP0]])
-; THRESHOLD-NEXT: [[OP_RDX:%.*]] = fadd fast float [[TMP2]], [[TMP3]]
+; THRESHOLD-NEXT: [[TMP2:%.*]] = insertelement <2 x float> poison, float [[TMP3]], i64 0
+; THRESHOLD-NEXT: [[TMP9:%.*]] = insertelement <2 x float> [[TMP2]], float [[CONVC]], i64 1
+; THRESHOLD-NEXT: [[TMP4:%.*]] = insertelement <2 x float> poison, float [[CONV]], i64 0
+; THRESHOLD-NEXT: [[TMP5:%.*]] = shufflevector <2 x float> [[TMP4]], <2 x float> poison, <2 x i32> zeroinitializer
+; THRESHOLD-NEXT: [[TMP6:%.*]] = fadd fast <2 x float> [[TMP9]], [[TMP5]]
+; THRESHOLD-NEXT: [[TMP7:%.*]] = extractelement <2 x float> [[TMP6]], i64 0
+; THRESHOLD-NEXT: [[TMP8:%.*]] = extractelement <2 x float> [[TMP6]], i64 1
+; THRESHOLD-NEXT: [[OP_RDX:%.*]] = fadd fast float [[TMP7]], [[TMP8]]
; THRESHOLD-NEXT: [[OP_RDX1:%.*]] = fadd fast float [[OP_RDX]], 3.000000e+00
-; THRESHOLD-NEXT: [[OP_RDX2:%.*]] = fadd fast float [[OP_RDX1]], [[CONVC]]
-; THRESHOLD-NEXT: ret float [[OP_RDX2]]
+; THRESHOLD-NEXT: ret float [[OP_RDX1]]
;
entry:
%mul = mul nsw i32 %b, %a
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/reassociated-fma-reduction.ll b/llvm/test/Transforms/SLPVectorizer/X86/reassociated-fma-reduction.ll
index 179d94b02307c..1fd49b8e58101 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/reassociated-fma-reduction.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/reassociated-fma-reduction.ll
@@ -5,20 +5,12 @@ define double @test(ptr %x, ptr %y, ptr %z) {
; CHECK-LABEL: define double @test(
; CHECK-SAME: ptr [[X:%.*]], ptr [[Y:%.*]], ptr [[Z:%.*]]) #[[ATTR0:[0-9]+]] {
; CHECK-NEXT: [[ENTRY:.*:]]
-; CHECK-NEXT: [[X0:%.*]] = load double, ptr [[X]], align 8
-; CHECK-NEXT: [[Y0:%.*]] = load double, ptr [[Y]], align 8
-; CHECK-NEXT: [[MUL0:%.*]] = fmul reassoc nsz contract double [[X0]], [[Y0]]
-; CHECK-NEXT: [[Z0:%.*]] = load double, ptr [[Z]], align 8
-; CHECK-NEXT: [[ADD0:%.*]] = fadd reassoc nsz contract double [[MUL0]], [[Z0]]
-; CHECK-NEXT: [[X1P:%.*]] = getelementptr inbounds i8, ptr [[X]], i64 8
-; CHECK-NEXT: [[X1:%.*]] = load double, ptr [[X1P]], align 8
-; CHECK-NEXT: [[Y1P:%.*]] = getelementptr inbounds i8, ptr [[Y]], i64 8
-; CHECK-NEXT: [[Y1:%.*]] = load double, ptr [[Y1P]], align 8
-; CHECK-NEXT: [[MUL1:%.*]] = fmul reassoc nsz contract double [[X1]], [[Y1]]
-; CHECK-NEXT: [[Z1P:%.*]] = getelementptr inbounds i8, ptr [[Z]], i64 8
-; CHECK-NEXT: [[Z1:%.*]] = load double, ptr [[Z1P]], align 8
-; CHECK-NEXT: [[ADD1:%.*]] = fadd reassoc nsz contract double [[ADD0]], [[Z1]]
-; CHECK-NEXT: [[ADD2:%.*]] = fadd reassoc nsz contract double [[ADD1]], [[MUL1]]
+; CHECK-NEXT: [[TMP0:%.*]] = load <2 x double>, ptr [[X]], align 8
+; CHECK-NEXT: [[TMP1:%.*]] = load <2 x double>, ptr [[Y]], align 8
+; CHECK-NEXT: [[TMP2:%.*]] = fmul reassoc nsz contract <2 x double> [[TMP0]], [[TMP1]]
+; CHECK-NEXT: [[TMP3:%.*]] = load <2 x double>, ptr [[Z]], align 8
+; CHECK-NEXT: [[RDX_OP:%.*]] = fadd reassoc nsz contract <2 x double> [[TMP2]], [[TMP3]]
+; CHECK-NEXT: [[ADD2:%.*]] = call reassoc nsz contract double @llvm.vector.reduce.fadd.v2f64(double 0.000000e+00, <2 x double> [[RDX_OP]])
; CHECK-NEXT: ret double [[ADD2]]
;
; OFF-LABEL: @reassoc_fma_reduction(
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/revectorized_rdx_crash.ll b/llvm/test/Transforms/SLPVectorizer/X86/revectorized_rdx_crash.ll
index 48b2174cac688..c5bdff3bf59d1 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/revectorized_rdx_crash.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/revectorized_rdx_crash.ll
@@ -19,19 +19,21 @@ define void @test(i1 %arg, ptr %p) {
; CHECK: for.cond.preheader:
; CHECK-NEXT: [[I:%.*]] = getelementptr inbounds [100 x i32], ptr [[P:%.*]], i64 0, i64 2
; CHECK-NEXT: [[I1:%.*]] = getelementptr inbounds [100 x i32], ptr [[P]], i64 0, i64 3
-; CHECK-NEXT: [[TMP0:%.*]] = load <4 x i32>, ptr [[I]], align 8
-; CHECK-NEXT: [[TMP1:%.*]] = call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> [[TMP0]])
-; CHECK-NEXT: [[OP_RDX3:%.*]] = add i32 0, [[TMP1]]
-; CHECK-NEXT: [[TMP2:%.*]] = load <4 x i32>, ptr [[I1]], align 4
-; CHECK-NEXT: [[TMP3:%.*]] = call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> [[TMP2]])
-; CHECK-NEXT: [[OP_RDX2:%.*]] = add i32 0, [[TMP3]]
+; CHECK-NEXT: [[I2:%.*]] = getelementptr inbounds [100 x i32], ptr [[P]], i64 0, i64 4
+; CHECK-NEXT: [[I3:%.*]] = getelementptr inbounds [100 x i32], ptr [[P]], i64 0, i64 5
+; CHECK-NEXT: [[TMP0:%.*]] = load <2 x i32>, ptr [[I3]], align 4
+; CHECK-NEXT: [[TMP1:%.*]] = load <2 x i32>, ptr [[I2]], align 16
+; CHECK-NEXT: [[TMP2:%.*]] = load <2 x i32>, ptr [[I1]], align 4
+; CHECK-NEXT: [[TMP3:%.*]] = load <2 x i32>, ptr [[I]], align 8
+; CHECK-NEXT: [[TMP7:%.*]] = add <2 x i32> [[TMP0]], [[TMP1]]
+; CHECK-NEXT: [[TMP5:%.*]] = add <2 x i32> [[TMP2]], [[TMP3]]
+; CHECK-NEXT: [[TMP6:%.*]] = add <2 x i32> [[TMP7]], [[TMP5]]
+; CHECK-NEXT: [[OP_RDX3:%.*]] = call i32 @llvm.vector.reduce.add.v2i32(<2 x i32> [[TMP6]])
; CHECK-NEXT: [[TMP4:%.*]] = mul i32 [[OP_RDX3]], 2
; CHECK-NEXT: [[OP_RDX:%.*]] = add i32 0, [[TMP4]]
-; CHECK-NEXT: [[TMP5:%.*]] = mul i32 [[OP_RDX2]], 2
-; CHECK-NEXT: [[OP_RDX1:%.*]] = add i32 [[OP_RDX]], [[TMP5]]
; CHECK-NEXT: br label [[IF_END]]
; CHECK: if.end:
-; CHECK-NEXT: [[R:%.*]] = phi i32 [ [[OP_RDX1]], [[FOR_COND_PREHEADER]] ], [ 0, [[ENTRY:%.*]] ]
+; CHECK-NEXT: [[R:%.*]] = phi i32 [ [[OP_RDX]], [[FOR_COND_PREHEADER]] ], [ 0, [[ENTRY:%.*]] ]
; CHECK-NEXT: ret void
;
entry:
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/vectorize-pair-path.ll b/llvm/test/Transforms/SLPVectorizer/X86/vectorize-pair-path.ll
index 7a280ab2b1ec8..16cec296ed862 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/vectorize-pair-path.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/vectorize-pair-path.ll
@@ -22,13 +22,13 @@ define double @root_selection(double %a, double %b, double %c, double %d) local_
; CHECK-NEXT: [[TMP5:%.*]] = fmul fast <2 x double> [[TMP3]], <double 3.000000e+00, double undef>
; CHECK-NEXT: [[TMP6:%.*]] = insertelement <2 x double> <double undef, double poison>, double [[I10]], i64 1
; CHECK-NEXT: [[TMP7:%.*]] = fmul fast <2 x double> [[TMP6]], [[TMP5]]
-; CHECK-NEXT: [[TMP8:%.*]] = insertelement <4 x double> poison, double [[C:%.*]], i64 0
-; CHECK-NEXT: [[TMP9:%.*]] = insertelement <4 x double> [[TMP8]], double [[D:%.*]], i64 1
+; CHECK-NEXT: [[TMP8:%.*]] = insertelement <4 x double> poison, double [[C:%.*]], i64 2
+; CHECK-NEXT: [[TMP9:%.*]] = insertelement <4 x double> [[TMP8]], double [[D:%.*]], i64 3
; CHECK-NEXT: [[TMP10:%.*]] = shufflevector <2 x double> [[TMP7]], <2 x double> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
-; CHECK-NEXT: [[TMP11:%.*]] = shufflevector <4 x double> [[TMP9]], <4 x double> [[TMP10]], <4 x i32> <i32 0, i32 1, i32 4, i32 5>
-; CHECK-NEXT: [[TMP12:%.*]] = fsub reassoc nsz arcp contract afn <4 x double> [[TMP11]], <double 0.000000e+00, double 0.000000e+00, double undef, double 1.100000e+01>
-; CHECK-NEXT: [[TMP13:%.*]] = fmul reassoc nsz arcp contract afn <4 x double> [[TMP12]], <double 1.000000e+00, double 1.000000e+00, double 4.000000e+00, double 1.200000e+01>
-; CHECK-NEXT: [[TMP14:%.*]] = fdiv reassoc nsz arcp contract afn <4 x double> [[TMP13]], <double 1.000000e+00, double 1.000000e+00, double 1.400000e+00, double 1.400000e+00>
+; CHECK-NEXT: [[TMP11:%.*]] = shufflevector <4 x double> [[TMP9]], <4 x double> [[TMP10]], <4 x i32> <i32 4, i32 5, i32 2, i32 3>
+; CHECK-NEXT: [[TMP12:%.*]] = fsub reassoc nsz arcp contract afn <4 x double> [[TMP11]], <double undef, double 1.100000e+01, double 0.000000e+00, double 0.000000e+00>
+; CHECK-NEXT: [[TMP13:%.*]] = fmul reassoc nsz arcp contract afn <4 x double> [[TMP12]], <double 4.000000e+00, double 1.200000e+01, double 1.000000e+00, double 1.000000e+00>
+; CHECK-NEXT: [[TMP14:%.*]] = fdiv reassoc nsz arcp contract afn <4 x double> [[TMP13]], <double 1.400000e+00, double 1.400000e+00, double 1.000000e+00, double 1.000000e+00>
; CHECK-NEXT: [[TMP15:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[TMP14]])
; CHECK-NEXT: [[I18:%.*]] = fadd fast double [[TMP15]], undef
; CHECK-NEXT: ret double [[I18]]
More information about the llvm-commits
mailing list