[llvm] [SLP]Allow min-VF vectorization of seed-level reduction groups (PR #222757)

Alexey Bataev via llvm-commits llvm-commits at lists.llvm.org
Fri Sep 11 10:56:07 PDT 2026


https://github.com/alexey-bataev updated https://github.com/llvm/llvm-project/pull/222757

>From 97edbfbbdb121cf15387bd8a5bcc17a2fb59c14b Mon Sep 17 00:00:00 2001
From: Alexey Bataev <a.bataev at outlook.com>
Date: Thu, 10 Sep 2026 13:14:39 -0700
Subject: [PATCH] =?UTF-8?q?[=F0=9D=98=80=F0=9D=97=BD=F0=9D=97=BF]=20initia?=
 =?UTF-8?q?l=20version?=
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit

Created using spr 1.3.7
---
 .../Transforms/Vectorize/SLPVectorizer.cpp    | 81 ++++++++++++++-----
 .../X86/fma-reassociate-pairs.ll              |  2 +-
 .../SLPVectorizer/X86/fma-operand-index.ll    | 42 ++++------
 .../X86/horizontal-fadd-with-sub.ll           |  2 +-
 .../SLPVectorizer/X86/horizontal-list.ll      | 21 +++--
 .../X86/reassociated-fma-reduction.ll         | 20 ++---
 .../X86/revectorized_rdx_crash.ll             | 20 ++---
 .../SLPVectorizer/X86/vectorize-pair-path.ll  | 12 +--
 8 files changed, 112 insertions(+), 88 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 28e665c4c07cf..75b4bda8796d7 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -29846,13 +29846,15 @@ class HorizontalReduction {
 
   /// Try to find a reduction tree. If \p FlattenNegations is false, the
   /// reassociable fsub/fneg links of an fadd chain are not flattened (they are
-  /// leaves of the reduction).
+  /// leaves of the reduction). \p IsSeedRoot enables the profitability
+  /// ordering of the same-size groups for the seed-level fadd reductions.
   bool matchAssociativeReduction(BoUpSLP &R, Instruction *Root,
                                  ScalarEvolution &SE, DominatorTree &DT,
                                  const DataLayout &DL,
                                  const TargetTransformInfo &TTI,
                                  const TargetLibraryInfo &TLI,
-                                 bool FlattenNegations = true) {
+                                 bool FlattenNegations = true,
+                                 bool IsSeedRoot = false) {
     RdxKind = HorizontalReduction::getRdxKind(Root);
     // A reassociable fsub root is a flattened link of an fadd reduction: its
     // subtracted operand enters with a flipped sign. Without the flattening
@@ -30238,17 +30240,18 @@ class HorizontalReduction {
       // sort them by size.
       optimizeReducedVals(R, DT, DL, TTI, TLI);
       // Sort the reduced values by number of same/alternate opcode and/or
-      // pointer operand. For the sign-aware reductions, same-size groups are
-      // ordered by their expected profitability: the first vectorized group
-      // is charged the cost of the reduction operation itself, which a group
-      // of loads or non-instructions rarely amortizes on its own, while it is
-      // a cheap addition (a single vector operation) to an already
-      // vectorized group - such groups go last. Among the rest, non-negated
-      // groups go first.
+      // pointer operand. For the sign-aware and the seed-level fadd
+      // reductions, same-size groups are ordered by their expected
+      // profitability: the first vectorized group is charged the cost of the
+      // reduction operation itself, which a group of loads or
+      // non-instructions rarely amortizes on its own, while it is a cheap
+      // addition (a single vector operation) to an already vectorized group -
+      // such groups go last. Among the rest, non-negated groups go first.
       stable_sort(ReducedVals, [&](ArrayRef<Value *> P1, ArrayRef<Value *> P2) {
         if (P1.size() != P2.size())
           return P1.size() > P2.size();
-        if (NegatedReducedVals.empty())
+        if (NegatedReducedVals.empty() &&
+            !(IsSeedRoot && RdxKind == RecurKind::FAdd))
           return false;
         auto IsCheapGroup = [](ArrayRef<Value *> P) {
           return !isa<Instruction>(P.front()) || isa<LoadInst>(P.front());
@@ -30303,9 +30306,11 @@ class HorizontalReduction {
   }
 
   /// Attempt to vectorize the tree found by matchAssociativeReduction.
+  /// \p IsSeedRoot allows vectorizing the groups at the minimum vector
+  /// factor. Set for single-use seed-level roots only.
   Value *tryToReduce(BoUpSLP &V, const DataLayout &DL, TargetTransformInfo *TTI,
                      const TargetLibraryInfo &TLI, AssumptionCache *AC,
-                     DominatorTree &DT) {
+                     DominatorTree &DT, bool IsSeedRoot = false) {
     constexpr unsigned RegMaxNumber = 4;
     const unsigned RedValsMaxNumber =
         (RK == ReductionOrdering::Ordered &&
@@ -30552,6 +30557,14 @@ class HorizontalReduction {
       States.push_back(getSameOpcode(RV, TLI));
     }
     ReducedVals.swap(LocalReducedVals);
+    // The minimum vector factor pays off only when it covers the whole
+    // reduction: every group must be either a pair of distinct values or
+    // large enough for the regular vector factor.
+    const bool NoScalarLeftovers =
+        all_of(ReducedVals, [this](ArrayRef<Value *> Vals) {
+          return Vals.size() >= ReductionLimit ||
+                 (Vals.size() == 2 && Vals.front() != Vals.back());
+        });
     for (unsigned I = 0, E = ReducedVals.size(); I < E; ++I) {
       ArrayRef<Value *> OrigReducedVals = ReducedVals[I];
       InstructionsState S = States[I];
@@ -30644,11 +30657,24 @@ class HorizontalReduction {
       }
 
       unsigned NumReducedVals = Candidates.size();
+      auto UsedByReductionOnly = [&](Value *V) {
+        if (!V->hasUseList())
+          return true;
+        // Bail out if we have too many uses to save compilation time.
+        if (V->hasNUsesOrMore(UsesLimit))
+          return false;
+        return all_of(V->users(),
+                      [&](User *U) { return IgnoreList.contains(U); });
+      };
       // Sign-aware reductions pair small positive/negative groups: allow
-      // non-splat groups down to 2 elements.
+      // non-splat groups down to 2 elements. Seed-level reductions get the
+      // same for groups whose values are used by the reduction operations
+      // only, if the minimum vector factor covers the whole reduction.
+      const bool MinVFAllowed = IsSeedRoot && NoScalarLeftovers &&
+                                all_of(Candidates, UsedByReductionOnly);
       if (NumReducedVals < ReductionLimit &&
-          (NumReducedVals < 2 ||
-           (!isSplat(Candidates) && NegatedReducedVals.empty())))
+          (NumReducedVals < 2 || (!isSplat(Candidates) &&
+                                  NegatedReducedVals.empty() && !MinVFAllowed)))
         continue;
 
       // Check if we support repeated scalar values processing (optimization of
@@ -30782,9 +30808,12 @@ class HorizontalReduction {
       };
       bool AnyVectorized = false;
       SmallDenseSet<std::pair<unsigned, unsigned>, 8> IgnoredCandidates;
-      // Same small-group allowance for the vector width.
+      // Same small-group allowance for the vector width, if it covers the
+      // whole group.
       const unsigned MinReduxWidth =
-          NegatedReducedVals.empty() ? ReductionLimit : 2;
+          !NegatedReducedVals.empty() || (MinVFAllowed && NumReducedVals == 2)
+              ? 2
+              : ReductionLimit;
       while (Pos < NumReducedVals - ReduxWidth + 1 &&
              ReduxWidth >= MinReduxWidth) {
         // Dependency in tree of the reduction ops - drop this attempt, try
@@ -32568,23 +32597,28 @@ bool SLPVectorizerPass::vectorizeHorReduction(
   Stack.emplace(SelectRoot(), 0);
   SmallPtrSet<Value *, 8> VisitedInstrs;
   bool Res = false;
-  auto TryToReduce = [this, &R, TTI = TTI](Instruction *Inst) -> Value * {
+  auto TryToReduce = [this, &R, TTI = TTI](Instruction *Inst,
+                                           bool IsSeedRoot) -> Value * {
     if (R.isAnalyzedReductionRoot(Inst))
       return nullptr;
     if (!isReductionCandidate(Inst))
       return nullptr;
     HorizontalReduction HorRdx;
     Value *Res = nullptr;
-    if (HorRdx.matchAssociativeReduction(R, Inst, *SE, *DT, *DL, *TTI, *TLI)) {
-      Value *Red = HorRdx.tryToReduce(R, *DL, TTI, *TLI, AC, *DT);
+    if (HorRdx.matchAssociativeReduction(R, Inst, *SE, *DT, *DL, *TTI, *TLI,
+                                         /*FlattenNegations=*/true,
+                                         IsSeedRoot)) {
+      Value *Red = HorRdx.tryToReduce(R, *DL, TTI, *TLI, AC, *DT, IsSeedRoot);
       // The chain with the flattened fsub/fneg links produced no vector part:
       // retry with the fsub/fneg links as leaves, they may be vectorizable on
       // their own.
       if (!Red && HorRdx.hasFlattenedNegations()) {
         HorizontalReduction UnflattenedHorRdx;
         if (UnflattenedHorRdx.matchAssociativeReduction(
-                R, Inst, *SE, *DT, *DL, *TTI, *TLI, /*FlattenNegations=*/false))
-          Red = UnflattenedHorRdx.tryToReduce(R, *DL, TTI, *TLI, AC, *DT);
+                R, Inst, *SE, *DT, *DL, *TTI, *TLI, /*FlattenNegations=*/false,
+                IsSeedRoot))
+          Red = UnflattenedHorRdx.tryToReduce(R, *DL, TTI, *TLI, AC, *DT,
+                                              IsSeedRoot);
       }
       if (Red) {
         if (Red != Inst)
@@ -32629,7 +32663,10 @@ bool SLPVectorizerPass::vectorizeHorReduction(
     // iteration while stack was populated before that happened.
     if (R.isDeleted(Inst))
       continue;
-    if (Value *VectorizedV = TryToReduce(Inst)) {
+    // The minimum-vector-factor allowance applies to single-use seed-level
+    // roots only.
+    if (Value *VectorizedV =
+            TryToReduce(Inst, /*IsSeedRoot=*/Level == 0 && Inst->hasOneUse())) {
       Res = true;
       if (auto *I = dyn_cast<Instruction>(VectorizedV); I && I != Inst) {
         // Try to find another reduction.
diff --git a/llvm/test/Transforms/PhaseOrdering/X86/fma-reassociate-pairs.ll b/llvm/test/Transforms/PhaseOrdering/X86/fma-reassociate-pairs.ll
index 1d2985733a698..f0d5eee0c7283 100644
--- a/llvm/test/Transforms/PhaseOrdering/X86/fma-reassociate-pairs.ll
+++ b/llvm/test/Transforms/PhaseOrdering/X86/fma-reassociate-pairs.ll
@@ -64,7 +64,7 @@ define double @fadd_fmul_2_right(ptr %x, ptr %y, ptr %z) {
 ; CHECK-NEXT:    [[TMP2:%.*]] = load <2 x double>, ptr [[Y]], align 8
 ; CHECK-NEXT:    [[TMP3:%.*]] = fmul reassoc nsz contract <2 x double> [[TMP2]], [[TMP1]]
 ; CHECK-NEXT:    [[TMP4:%.*]] = load <2 x double>, ptr [[Z]], align 8
-; CHECK-NEXT:    [[TMP5:%.*]] = fadd reassoc nsz contract <2 x double> [[TMP4]], [[TMP3]]
+; CHECK-NEXT:    [[TMP5:%.*]] = fadd reassoc nsz contract <2 x double> [[TMP3]], [[TMP4]]
 ; CHECK-NEXT:    [[R:%.*]] = tail call reassoc nsz contract double @llvm.vector.reduce.fadd.v2f64(double 0.000000e+00, <2 x double> [[TMP5]])
 ; CHECK-NEXT:    ret double [[R]]
 ;
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/fma-operand-index.ll b/llvm/test/Transforms/SLPVectorizer/X86/fma-operand-index.ll
index 0fa4058f874d2..154d15f4d2dad 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/fma-operand-index.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/fma-operand-index.ll
@@ -1,26 +1,22 @@
 ; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
 ; RUN: opt -S --passes=slp-vectorizer -mtriple=x86_64-unknown-linux-gnu -mcpu=znver4 < %s | FileCheck %s
 
-; A one-use fmul is left scalar so the backend can fuse it, but only operand 0
-; of the fadd/fsub is checked. The same fmul is kept or gathered depending on
-; which side it sits on. To be addressed in a follow-up.
+; Seed-level reassociable fadd/fsub chains vectorize as unordered reductions
+; when the reduced values split into 2-element groups: the groups share the
+; reduction operation cost. The vectorized fmul still fuses into a vector fma
+; in the backend, whichever operand of the fadd/fsub it feeds.
 
-; Both fmuls are operand 1, so they gather and cannot fuse.
+; Both fmuls are operand 1 of the fadds.
 define double @fmul_rhs_fadd(ptr %x, ptr %y, ptr %z) {
 ; CHECK-LABEL: define double @fmul_rhs_fadd(
 ; CHECK-SAME: ptr [[X:%.*]], ptr [[Y:%.*]], ptr [[Z:%.*]]) #[[ATTR0:[0-9]+]] {
 ; CHECK-NEXT:  [[ENTRY:.*:]]
-; CHECK-NEXT:    [[Z8:%.*]] = getelementptr inbounds nuw i8, ptr [[Z]], i64 8
-; CHECK-NEXT:    [[Z0:%.*]] = load double, ptr [[Z]], align 8
 ; CHECK-NEXT:    [[TMP0:%.*]] = load <2 x double>, ptr [[X]], align 8
 ; CHECK-NEXT:    [[TMP1:%.*]] = load <2 x double>, ptr [[Y]], align 8
 ; CHECK-NEXT:    [[TMP2:%.*]] = fmul reassoc nsz contract <2 x double> [[TMP1]], [[TMP0]]
-; CHECK-NEXT:    [[Z1:%.*]] = load double, ptr [[Z8]], align 8
-; CHECK-NEXT:    [[ZSUM:%.*]] = fadd reassoc nsz contract double [[Z0]], [[Z1]]
-; CHECK-NEXT:    [[MUL0:%.*]] = extractelement <2 x double> [[TMP2]], i64 0
-; CHECK-NEXT:    [[ADD0:%.*]] = fadd reassoc nsz contract double [[ZSUM]], [[MUL0]]
-; CHECK-NEXT:    [[MUL1:%.*]] = extractelement <2 x double> [[TMP2]], i64 1
-; CHECK-NEXT:    [[ADD1:%.*]] = fadd reassoc nsz contract double [[ADD0]], [[MUL1]]
+; CHECK-NEXT:    [[TMP3:%.*]] = load <2 x double>, ptr [[Z]], align 8
+; CHECK-NEXT:    [[RDX_OP:%.*]] = fadd reassoc nsz contract <2 x double> [[TMP2]], [[TMP3]]
+; CHECK-NEXT:    [[ADD1:%.*]] = call reassoc nsz contract double @llvm.vector.reduce.fadd.v2f64(double 0.000000e+00, <2 x double> [[RDX_OP]])
 ; CHECK-NEXT:    ret double [[ADD1]]
 ;
 entry:
@@ -72,25 +68,17 @@ entry:
   ret double %sub1
 }
 
-; Operand 0 is the one checked, so these stay scalar and fuse.
+; Both fmuls are operand 0 of the fadds.
 define double @fmul_lhs_fadd(ptr %x, ptr %y, ptr %z) {
 ; CHECK-LABEL: define double @fmul_lhs_fadd(
 ; CHECK-SAME: ptr [[X:%.*]], ptr [[Y:%.*]], ptr [[Z:%.*]]) #[[ATTR0]] {
 ; CHECK-NEXT:  [[ENTRY:.*:]]
-; CHECK-NEXT:    [[X8:%.*]] = getelementptr inbounds nuw i8, ptr [[X]], i64 8
-; CHECK-NEXT:    [[Y8:%.*]] = getelementptr inbounds nuw i8, ptr [[Y]], i64 8
-; CHECK-NEXT:    [[Z8:%.*]] = getelementptr inbounds nuw i8, ptr [[Z]], i64 8
-; CHECK-NEXT:    [[X0:%.*]] = load double, ptr [[X]], align 8
-; CHECK-NEXT:    [[Y0:%.*]] = load double, ptr [[Y]], align 8
-; CHECK-NEXT:    [[MUL0:%.*]] = fmul reassoc nsz contract double [[Y0]], [[X0]]
-; CHECK-NEXT:    [[Z0:%.*]] = load double, ptr [[Z]], align 8
-; CHECK-NEXT:    [[X1:%.*]] = load double, ptr [[X8]], align 8
-; CHECK-NEXT:    [[Y1:%.*]] = load double, ptr [[Y8]], align 8
-; CHECK-NEXT:    [[MUL1:%.*]] = fmul reassoc nsz contract double [[Y1]], [[X1]]
-; CHECK-NEXT:    [[Z1:%.*]] = load double, ptr [[Z8]], align 8
-; CHECK-NEXT:    [[ZSUM:%.*]] = fadd reassoc nsz contract double [[Z0]], [[Z1]]
-; CHECK-NEXT:    [[ADD0:%.*]] = fadd reassoc nsz contract double [[MUL0]], [[ZSUM]]
-; CHECK-NEXT:    [[ADD1:%.*]] = fadd reassoc nsz contract double [[MUL1]], [[ADD0]]
+; CHECK-NEXT:    [[TMP0:%.*]] = load <2 x double>, ptr [[X]], align 8
+; CHECK-NEXT:    [[TMP1:%.*]] = load <2 x double>, ptr [[Y]], align 8
+; CHECK-NEXT:    [[TMP2:%.*]] = fmul reassoc nsz contract <2 x double> [[TMP1]], [[TMP0]]
+; CHECK-NEXT:    [[TMP3:%.*]] = load <2 x double>, ptr [[Z]], align 8
+; CHECK-NEXT:    [[RDX_OP:%.*]] = fadd reassoc nsz contract <2 x double> [[TMP2]], [[TMP3]]
+; CHECK-NEXT:    [[ADD1:%.*]] = call reassoc nsz contract double @llvm.vector.reduce.fadd.v2f64(double 0.000000e+00, <2 x double> [[RDX_OP]])
 ; CHECK-NEXT:    ret double [[ADD1]]
 ;
 entry:
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/horizontal-fadd-with-sub.ll b/llvm/test/Transforms/SLPVectorizer/X86/horizontal-fadd-with-sub.ll
index e8c566ecdaabc..dc1a2ccfb0c2c 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/horizontal-fadd-with-sub.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/horizontal-fadd-with-sub.ll
@@ -1092,7 +1092,7 @@ define double @opaque_fneg_leaves_mixed(ptr %x, ptr %y) {
 ; CHECK-NEXT:    [[TMP0:%.*]] = load <4 x double>, ptr [[X]], align 8
 ; CHECK-NEXT:    [[TMP1:%.*]] = load <4 x double>, ptr [[Y]], align 8
 ; CHECK-NEXT:    [[TMP2:%.*]] = fneg <4 x double> [[TMP1]]
-; CHECK-NEXT:    [[RDX_OP:%.*]] = fadd reassoc nsz contract <4 x double> [[TMP0]], [[TMP2]]
+; CHECK-NEXT:    [[RDX_OP:%.*]] = fadd reassoc nsz contract <4 x double> [[TMP2]], [[TMP0]]
 ; CHECK-NEXT:    [[TMP3:%.*]] = call reassoc nsz contract double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[RDX_OP]])
 ; CHECK-NEXT:    ret double [[TMP3]]
 ;
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/horizontal-list.ll b/llvm/test/Transforms/SLPVectorizer/X86/horizontal-list.ll
index 69822351ffd57..994ff81c9f633 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/horizontal-list.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/horizontal-list.ll
@@ -842,11 +842,11 @@ define float @extra_args_no_replace(ptr nocapture readonly %x, i32 %a, i32 %b, i
 ; CHECK-NEXT:    [[CONV:%.*]] = sitofp i32 [[MUL]] to float
 ; CHECK-NEXT:    [[CONVC:%.*]] = sitofp i32 [[C:%.*]] to float
 ; CHECK-NEXT:    [[TMP0:%.*]] = load <8 x float>, ptr [[X:%.*]], align 4
-; CHECK-NEXT:    [[TMP2:%.*]] = fmul reassoc nnan ninf nsz arcp afn float [[CONV]], 2.000000e+00
 ; CHECK-NEXT:    [[TMP3:%.*]] = call fast float @llvm.vector.reduce.fadd.v8f32(float 0.000000e+00, <8 x float> [[TMP0]])
-; CHECK-NEXT:    [[OP_RDX:%.*]] = fadd fast float [[TMP2]], [[TMP3]]
-; CHECK-NEXT:    [[OP_RDX1:%.*]] = fadd fast float [[OP_RDX]], 3.000000e+00
-; CHECK-NEXT:    [[OP_RDX2:%.*]] = fadd fast float [[OP_RDX1]], [[CONVC]]
+; CHECK-NEXT:    [[OP_RDX:%.*]] = fadd fast float [[TMP3]], [[CONV]]
+; CHECK-NEXT:    [[OP_RDX1:%.*]] = fadd fast float [[CONV]], [[CONVC]]
+; CHECK-NEXT:    [[OP_RDX3:%.*]] = fadd fast float [[OP_RDX]], [[OP_RDX1]]
+; CHECK-NEXT:    [[OP_RDX2:%.*]] = fadd fast float [[OP_RDX3]], 3.000000e+00
 ; CHECK-NEXT:    ret float [[OP_RDX2]]
 ;
 ; THRESHOLD-LABEL: @extra_args_no_replace(
@@ -855,12 +855,17 @@ define float @extra_args_no_replace(ptr nocapture readonly %x, i32 %a, i32 %b, i
 ; THRESHOLD-NEXT:    [[CONV:%.*]] = sitofp i32 [[MUL]] to float
 ; THRESHOLD-NEXT:    [[CONVC:%.*]] = sitofp i32 [[C:%.*]] to float
 ; THRESHOLD-NEXT:    [[TMP0:%.*]] = load <8 x float>, ptr [[X:%.*]], align 4
-; THRESHOLD-NEXT:    [[TMP2:%.*]] = fmul reassoc nnan ninf nsz arcp afn float [[CONV]], 2.000000e+00
 ; THRESHOLD-NEXT:    [[TMP3:%.*]] = call fast float @llvm.vector.reduce.fadd.v8f32(float 0.000000e+00, <8 x float> [[TMP0]])
-; THRESHOLD-NEXT:    [[OP_RDX:%.*]] = fadd fast float [[TMP2]], [[TMP3]]
+; THRESHOLD-NEXT:    [[TMP2:%.*]] = insertelement <2 x float> poison, float [[TMP3]], i64 0
+; THRESHOLD-NEXT:    [[TMP9:%.*]] = insertelement <2 x float> [[TMP2]], float [[CONVC]], i64 1
+; THRESHOLD-NEXT:    [[TMP4:%.*]] = insertelement <2 x float> poison, float [[CONV]], i64 0
+; THRESHOLD-NEXT:    [[TMP5:%.*]] = shufflevector <2 x float> [[TMP4]], <2 x float> poison, <2 x i32> zeroinitializer
+; THRESHOLD-NEXT:    [[TMP6:%.*]] = fadd fast <2 x float> [[TMP9]], [[TMP5]]
+; THRESHOLD-NEXT:    [[TMP7:%.*]] = extractelement <2 x float> [[TMP6]], i64 0
+; THRESHOLD-NEXT:    [[TMP8:%.*]] = extractelement <2 x float> [[TMP6]], i64 1
+; THRESHOLD-NEXT:    [[OP_RDX:%.*]] = fadd fast float [[TMP7]], [[TMP8]]
 ; THRESHOLD-NEXT:    [[OP_RDX1:%.*]] = fadd fast float [[OP_RDX]], 3.000000e+00
-; THRESHOLD-NEXT:    [[OP_RDX2:%.*]] = fadd fast float [[OP_RDX1]], [[CONVC]]
-; THRESHOLD-NEXT:    ret float [[OP_RDX2]]
+; THRESHOLD-NEXT:    ret float [[OP_RDX1]]
 ;
   entry:
   %mul = mul nsw i32 %b, %a
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/reassociated-fma-reduction.ll b/llvm/test/Transforms/SLPVectorizer/X86/reassociated-fma-reduction.ll
index 179d94b02307c..1fd49b8e58101 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/reassociated-fma-reduction.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/reassociated-fma-reduction.ll
@@ -5,20 +5,12 @@ define double @test(ptr %x, ptr %y, ptr %z) {
 ; CHECK-LABEL: define double @test(
 ; CHECK-SAME: ptr [[X:%.*]], ptr [[Y:%.*]], ptr [[Z:%.*]]) #[[ATTR0:[0-9]+]] {
 ; CHECK-NEXT:  [[ENTRY:.*:]]
-; CHECK-NEXT:    [[X0:%.*]] = load double, ptr [[X]], align 8
-; CHECK-NEXT:    [[Y0:%.*]] = load double, ptr [[Y]], align 8
-; CHECK-NEXT:    [[MUL0:%.*]] = fmul reassoc nsz contract double [[X0]], [[Y0]]
-; CHECK-NEXT:    [[Z0:%.*]] = load double, ptr [[Z]], align 8
-; CHECK-NEXT:    [[ADD0:%.*]] = fadd reassoc nsz contract double [[MUL0]], [[Z0]]
-; CHECK-NEXT:    [[X1P:%.*]] = getelementptr inbounds i8, ptr [[X]], i64 8
-; CHECK-NEXT:    [[X1:%.*]] = load double, ptr [[X1P]], align 8
-; CHECK-NEXT:    [[Y1P:%.*]] = getelementptr inbounds i8, ptr [[Y]], i64 8
-; CHECK-NEXT:    [[Y1:%.*]] = load double, ptr [[Y1P]], align 8
-; CHECK-NEXT:    [[MUL1:%.*]] = fmul reassoc nsz contract double [[X1]], [[Y1]]
-; CHECK-NEXT:    [[Z1P:%.*]] = getelementptr inbounds i8, ptr [[Z]], i64 8
-; CHECK-NEXT:    [[Z1:%.*]] = load double, ptr [[Z1P]], align 8
-; CHECK-NEXT:    [[ADD1:%.*]] = fadd reassoc nsz contract double [[ADD0]], [[Z1]]
-; CHECK-NEXT:    [[ADD2:%.*]] = fadd reassoc nsz contract double [[ADD1]], [[MUL1]]
+; CHECK-NEXT:    [[TMP0:%.*]] = load <2 x double>, ptr [[X]], align 8
+; CHECK-NEXT:    [[TMP1:%.*]] = load <2 x double>, ptr [[Y]], align 8
+; CHECK-NEXT:    [[TMP2:%.*]] = fmul reassoc nsz contract <2 x double> [[TMP0]], [[TMP1]]
+; CHECK-NEXT:    [[TMP3:%.*]] = load <2 x double>, ptr [[Z]], align 8
+; CHECK-NEXT:    [[RDX_OP:%.*]] = fadd reassoc nsz contract <2 x double> [[TMP2]], [[TMP3]]
+; CHECK-NEXT:    [[ADD2:%.*]] = call reassoc nsz contract double @llvm.vector.reduce.fadd.v2f64(double 0.000000e+00, <2 x double> [[RDX_OP]])
 ; CHECK-NEXT:    ret double [[ADD2]]
 ;
 ; OFF-LABEL: @reassoc_fma_reduction(
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/revectorized_rdx_crash.ll b/llvm/test/Transforms/SLPVectorizer/X86/revectorized_rdx_crash.ll
index 48b2174cac688..c5bdff3bf59d1 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/revectorized_rdx_crash.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/revectorized_rdx_crash.ll
@@ -19,19 +19,21 @@ define void @test(i1 %arg, ptr %p) {
 ; CHECK:       for.cond.preheader:
 ; CHECK-NEXT:    [[I:%.*]] = getelementptr inbounds [100 x i32], ptr [[P:%.*]], i64 0, i64 2
 ; CHECK-NEXT:    [[I1:%.*]] = getelementptr inbounds [100 x i32], ptr [[P]], i64 0, i64 3
-; CHECK-NEXT:    [[TMP0:%.*]] = load <4 x i32>, ptr [[I]], align 8
-; CHECK-NEXT:    [[TMP1:%.*]] = call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> [[TMP0]])
-; CHECK-NEXT:    [[OP_RDX3:%.*]] = add i32 0, [[TMP1]]
-; CHECK-NEXT:    [[TMP2:%.*]] = load <4 x i32>, ptr [[I1]], align 4
-; CHECK-NEXT:    [[TMP3:%.*]] = call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> [[TMP2]])
-; CHECK-NEXT:    [[OP_RDX2:%.*]] = add i32 0, [[TMP3]]
+; CHECK-NEXT:    [[I2:%.*]] = getelementptr inbounds [100 x i32], ptr [[P]], i64 0, i64 4
+; CHECK-NEXT:    [[I3:%.*]] = getelementptr inbounds [100 x i32], ptr [[P]], i64 0, i64 5
+; CHECK-NEXT:    [[TMP0:%.*]] = load <2 x i32>, ptr [[I3]], align 4
+; CHECK-NEXT:    [[TMP1:%.*]] = load <2 x i32>, ptr [[I2]], align 16
+; CHECK-NEXT:    [[TMP2:%.*]] = load <2 x i32>, ptr [[I1]], align 4
+; CHECK-NEXT:    [[TMP3:%.*]] = load <2 x i32>, ptr [[I]], align 8
+; CHECK-NEXT:    [[TMP7:%.*]] = add <2 x i32> [[TMP0]], [[TMP1]]
+; CHECK-NEXT:    [[TMP5:%.*]] = add <2 x i32> [[TMP2]], [[TMP3]]
+; CHECK-NEXT:    [[TMP6:%.*]] = add <2 x i32> [[TMP7]], [[TMP5]]
+; CHECK-NEXT:    [[OP_RDX3:%.*]] = call i32 @llvm.vector.reduce.add.v2i32(<2 x i32> [[TMP6]])
 ; CHECK-NEXT:    [[TMP4:%.*]] = mul i32 [[OP_RDX3]], 2
 ; CHECK-NEXT:    [[OP_RDX:%.*]] = add i32 0, [[TMP4]]
-; CHECK-NEXT:    [[TMP5:%.*]] = mul i32 [[OP_RDX2]], 2
-; CHECK-NEXT:    [[OP_RDX1:%.*]] = add i32 [[OP_RDX]], [[TMP5]]
 ; CHECK-NEXT:    br label [[IF_END]]
 ; CHECK:       if.end:
-; CHECK-NEXT:    [[R:%.*]] = phi i32 [ [[OP_RDX1]], [[FOR_COND_PREHEADER]] ], [ 0, [[ENTRY:%.*]] ]
+; CHECK-NEXT:    [[R:%.*]] = phi i32 [ [[OP_RDX]], [[FOR_COND_PREHEADER]] ], [ 0, [[ENTRY:%.*]] ]
 ; CHECK-NEXT:    ret void
 ;
 entry:
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/vectorize-pair-path.ll b/llvm/test/Transforms/SLPVectorizer/X86/vectorize-pair-path.ll
index 7a280ab2b1ec8..16cec296ed862 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/vectorize-pair-path.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/vectorize-pair-path.ll
@@ -22,13 +22,13 @@ define double @root_selection(double %a, double %b, double %c, double %d) local_
 ; CHECK-NEXT:    [[TMP5:%.*]] = fmul fast <2 x double> [[TMP3]], <double 3.000000e+00, double undef>
 ; CHECK-NEXT:    [[TMP6:%.*]] = insertelement <2 x double> <double undef, double poison>, double [[I10]], i64 1
 ; CHECK-NEXT:    [[TMP7:%.*]] = fmul fast <2 x double> [[TMP6]], [[TMP5]]
-; CHECK-NEXT:    [[TMP8:%.*]] = insertelement <4 x double> poison, double [[C:%.*]], i64 0
-; CHECK-NEXT:    [[TMP9:%.*]] = insertelement <4 x double> [[TMP8]], double [[D:%.*]], i64 1
+; CHECK-NEXT:    [[TMP8:%.*]] = insertelement <4 x double> poison, double [[C:%.*]], i64 2
+; CHECK-NEXT:    [[TMP9:%.*]] = insertelement <4 x double> [[TMP8]], double [[D:%.*]], i64 3
 ; CHECK-NEXT:    [[TMP10:%.*]] = shufflevector <2 x double> [[TMP7]], <2 x double> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
-; CHECK-NEXT:    [[TMP11:%.*]] = shufflevector <4 x double> [[TMP9]], <4 x double> [[TMP10]], <4 x i32> <i32 0, i32 1, i32 4, i32 5>
-; CHECK-NEXT:    [[TMP12:%.*]] = fsub reassoc nsz arcp contract afn <4 x double> [[TMP11]], <double 0.000000e+00, double 0.000000e+00, double undef, double 1.100000e+01>
-; CHECK-NEXT:    [[TMP13:%.*]] = fmul reassoc nsz arcp contract afn <4 x double> [[TMP12]], <double 1.000000e+00, double 1.000000e+00, double 4.000000e+00, double 1.200000e+01>
-; CHECK-NEXT:    [[TMP14:%.*]] = fdiv reassoc nsz arcp contract afn <4 x double> [[TMP13]], <double 1.000000e+00, double 1.000000e+00, double 1.400000e+00, double 1.400000e+00>
+; CHECK-NEXT:    [[TMP11:%.*]] = shufflevector <4 x double> [[TMP9]], <4 x double> [[TMP10]], <4 x i32> <i32 4, i32 5, i32 2, i32 3>
+; CHECK-NEXT:    [[TMP12:%.*]] = fsub reassoc nsz arcp contract afn <4 x double> [[TMP11]], <double undef, double 1.100000e+01, double 0.000000e+00, double 0.000000e+00>
+; CHECK-NEXT:    [[TMP13:%.*]] = fmul reassoc nsz arcp contract afn <4 x double> [[TMP12]], <double 4.000000e+00, double 1.200000e+01, double 1.000000e+00, double 1.000000e+00>
+; CHECK-NEXT:    [[TMP14:%.*]] = fdiv reassoc nsz arcp contract afn <4 x double> [[TMP13]], <double 1.400000e+00, double 1.400000e+00, double 1.000000e+00, double 1.000000e+00>
 ; CHECK-NEXT:    [[TMP15:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[TMP14]])
 ; CHECK-NEXT:    [[I18:%.*]] = fadd fast double [[TMP15]], undef
 ; CHECK-NEXT:    ret double [[I18]]



More information about the llvm-commits mailing list