[llvm] [SLP]Initial support for ordreded reductions (PR #182644)

Alexey Bataev via llvm-commits llvm-commits at lists.llvm.org
Fri Feb 20 17:51:19 PST 2026


https://github.com/alexey-bataev updated https://github.com/llvm/llvm-project/pull/182644

>From c9d5e477b4e05d483db4921c5c566aeeb3a82897 Mon Sep 17 00:00:00 2001
From: Alexey Bataev <a.bataev at outlook.com>
Date: Fri, 20 Feb 2026 17:46:54 -0800
Subject: [PATCH 1/2] =?UTF-8?q?[=F0=9D=98=80=F0=9D=97=BD=F0=9D=97=BF]=20in?=
 =?UTF-8?q?itial=20version?=
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit

Created using spr 1.3.7
---
 .../Transforms/Vectorize/SLPVectorizer.cpp    |  83 ++++--
 .../SLPVectorizer/AArch64/tsc-s352.ll         |  20 +-
 .../SLPVectorizer/X86/dot-product.ll          |  92 ++++--
 .../Transforms/SLPVectorizer/X86/fmaxnum.ll   | 282 +++++++++++++++---
 .../Transforms/SLPVectorizer/X86/fminnum.ll   | 262 +++++++++++++---
 .../SLPVectorizer/X86/horizontal-list.ll      |  12 +-
 llvm/test/Transforms/SLPVectorizer/X86/phi.ll |  50 ++--
 .../scatter-vectorize-reorder-non-empty.ll    |  17 +-
 8 files changed, 633 insertions(+), 185 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 4caa1707f0f27..b1df83e021cb7 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -25349,31 +25349,37 @@ class HorizontalReduction {
   }
 
   /// Checks if instruction is associative and can be vectorized.
-  static bool isVectorizable(RecurKind Kind, Instruction *I,
-                             bool TwoElementReduction = false) {
+  enum class ReductionKind {Unordered, Ordered, None};
+  ReductionKind RK = ReductionKind::None;
+  static ReductionKind isVectorizable(RecurKind Kind, Instruction *I,
+                                      bool TwoElementReduction = false) {
     if (Kind == RecurKind::None)
-      return false;
+      return ReductionKind::None;
 
     // Integer ops that map to select instructions or intrinsics are fine.
     if (RecurrenceDescriptor::isIntMinMaxRecurrenceKind(Kind) ||
         isBoolLogicOp(I))
-      return true;
+      return ReductionKind::Unordered;
 
     // No need to check for associativity, if 2 reduced values.
     if (TwoElementReduction)
-      return true;
+      return ReductionKind::Unordered;
 
     if (Kind == RecurKind::FMax || Kind == RecurKind::FMin) {
       // FP min/max are associative except for NaN and -0.0. We do not
       // have to rule out -0.0 here because the intrinsic semantics do not
       // specify a fixed result for it.
-      return I->getFastMathFlags().noNaNs();
+      return I->getFastMathFlags().noNaNs() ? ReductionKind::Unordered
+                                            : ReductionKind::Ordered;
     }
 
     if (Kind == RecurKind::FMaximum || Kind == RecurKind::FMinimum)
-      return true;
+      return ReductionKind::Unordered;
+
+    if (I->isAssociative())
+      return ReductionKind::Unordered;
 
-    return I->isAssociative();
+    return ::isCommutative(I) ? ReductionKind::Ordered : ReductionKind::None;
   }
 
   static Value *getRdxOperand(Instruction *I, unsigned Index) {
@@ -25675,13 +25681,10 @@ class HorizontalReduction {
     // Analyze "regular" integer/FP types for reductions - no target-specific
     // types or pointers.
     assert(ReductionRoot && "Reduction root is not set!");
-    if (!isVectorizable(RdxKind, cast<Instruction>(ReductionRoot),
-                        all_of(ReducedVals, [](ArrayRef<Value *> Ops) {
-                          return Ops.size() == 2;
-                        })))
-      return false;
-
-    return true;
+    return isVectorizable(RdxKind, cast<Instruction>(ReductionRoot),
+                          all_of(ReducedVals, [](ArrayRef<Value *> Ops) {
+                            return Ops.size() == 2;
+                          })) != ReductionKind::None;
   }
 
   /// Try to find a reduction tree.
@@ -25689,7 +25692,8 @@ class HorizontalReduction {
                                  ScalarEvolution &SE, const DataLayout &DL,
                                  const TargetLibraryInfo &TLI) {
     RdxKind = HorizontalReduction::getRdxKind(Root);
-    if (!isVectorizable(RdxKind, Root))
+    RK = isVectorizable(RdxKind, Root);
+    if (RK == ReductionKind::None)
       return false;
 
     // Analyze "regular" integer/FP types for reductions - no target-specific
@@ -25728,16 +25732,21 @@ class HorizontalReduction {
         // reduction opcode or has too many uses - possible reduced value.
         // Also, do not try to reduce const values, if the operation is not
         // foldable.
-        if (!EdgeInst || Level > RecursionMaxDepth ||
+        bool IsReducedVal = !EdgeInst || Level > RecursionMaxDepth ||
             getRdxKind(EdgeInst) != RdxKind ||
             IsCmpSelMinMax != isCmpSelMinMax(EdgeInst) ||
-            !hasRequiredNumberOfUses(IsCmpSelMinMax, EdgeInst) ||
-            !isVectorizable(RdxKind, EdgeInst) ||
+            !hasRequiredNumberOfUses(IsCmpSelMinMax, EdgeInst);
+        ReductionKind CurrentRK = IsReducedVal
+                                      ? ReductionKind::None
+                                      : isVectorizable(RdxKind, EdgeInst);
+        if (CurrentRK == ReductionKind::None ||
             (R.isAnalyzedReductionRoot(EdgeInst) &&
              all_of(EdgeInst->operands(), IsaPred<Constant>))) {
           PossibleReducedVals.push_back(EdgeVal);
           continue;
         }
+        if (CurrentRK == ReductionKind::Ordered)
+          RK = ReductionKind::Ordered;
         ReductionOps.push_back(EdgeInst);
       }
     };
@@ -25943,6 +25952,10 @@ class HorizontalReduction {
     for (Value *U : IgnoreList)
       if (auto *FPMO = dyn_cast<FPMathOperator>(U))
         RdxFMF &= FPMO->getFastMathFlags();
+    // For ordered reductions here we need to generate extractelement
+    // instructions, so clear IgnoreList.
+    if (RK == ReductionKind::Ordered)
+      IgnoreList.clear();
     bool IsCmpSelMinMax = isCmpSelMinMax(cast<Instruction>(ReductionRoot));
 
     // Need to track reduced vals, they may be changed during vectorization of
@@ -26054,6 +26067,8 @@ class HorizontalReduction {
 
       // Emit code for constant values.
       if (Candidates.size() > 1 && allConstant(Candidates)) {
+        if (RK == ReductionKind::Ordered)
+          continue;
         Value *Res = Candidates.front();
         Value *OrigV = TrackedToOrig.at(Candidates.front());
         ++VectorizedVals.try_emplace(OrigV).first->getSecond();
@@ -26075,9 +26090,9 @@ class HorizontalReduction {
 
       // Check if we support repeated scalar values processing (optimization of
       // original scalar identity operations on matched horizontal reductions).
-      IsSupportedHorRdxIdentityOp = RdxKind != RecurKind::Mul &&
-                                    RdxKind != RecurKind::FMul &&
-                                    RdxKind != RecurKind::FMulAdd;
+      IsSupportedHorRdxIdentityOp =
+          RK == ReductionKind::Unordered && RdxKind != RecurKind::Mul &&
+          RdxKind != RecurKind::FMul && RdxKind != RecurKind::FMulAdd;
       // Gather same values.
       SmallMapVector<Value *, unsigned, 16> SameValuesCounter;
       if (IsSupportedHorRdxIdentityOp)
@@ -26346,6 +26361,23 @@ class HorizontalReduction {
         // Vectorize a tree.
         Value *VectorizedRoot = V.vectorizeTree(
             LocalExternallyUsedValues, InsertPt, VectorValuesAndScales);
+        if (RK == ReductionKind::Ordered) {
+          // No need to generate reduction here, eit extractelements instead in
+          // the tree vectorizer.
+          assert(VectorizedRoot && "Expected vectorized tree");
+          // Count vectorized reduced values to exclude them from final
+          // reduction.
+          for (Value *RdxVal : VL)
+            ++VectorizedVals.try_emplace(RdxVal).first->getSecond();
+          Pos += ReduxWidth;
+          Start = Pos;
+          ReduxWidth = NumReducedVals - Pos;
+          if (ReduxWidth > 1)
+            ReduxWidth = GetVectorFactor(NumReducedVals - Pos);
+          AnyVectorized = true;
+          VectorizedTree = ReductionRoot;
+          continue;
+        }
         // Update TrackedToOrig mapping, since the tracked values might be
         // updated.
         for (Value *RdxVal : Candidates) {
@@ -26410,6 +26442,11 @@ class HorizontalReduction {
         continue;
       }
     }
+    // Early exit for the ordered reductions.
+    // No need to do anything else here, so we can just exit.
+    if (RK == ReductionKind::Ordered)
+      return VectorizedTree;
+
     if (!VectorValuesAndScales.empty())
       VectorizedTree = GetNewVectorizedTree(
           VectorizedTree,
@@ -27458,7 +27495,7 @@ bool SLPVectorizerPass::vectorizeHorReduction(
       continue;
     if (Value *VectorizedV = TryToReduce(Inst)) {
       Res = true;
-      if (auto *I = dyn_cast<Instruction>(VectorizedV)) {
+      if (auto *I = dyn_cast<Instruction>(VectorizedV); I && I != Inst) {
         // Try to find another reduction.
         Stack.emplace(I, Level);
         continue;
diff --git a/llvm/test/Transforms/SLPVectorizer/AArch64/tsc-s352.ll b/llvm/test/Transforms/SLPVectorizer/AArch64/tsc-s352.ll
index 70b7adfd6456e..c9a2219b12a8a 100644
--- a/llvm/test/Transforms/SLPVectorizer/AArch64/tsc-s352.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AArch64/tsc-s352.ll
@@ -31,22 +31,16 @@ define i32 @s352() {
 ; CHECK-NEXT:    [[DOT_115:%.*]] = phi float [ 0.000000e+00, [[PREHEADER]] ], [ [[ADD39:%.*]], [[FOR_BODY]] ]
 ; CHECK-NEXT:    [[ARRAYIDX:%.*]] = getelementptr inbounds [[STRUCT_GLOBALDATA:%.*]], ptr @global_data, i64 0, i32 0, i64 [[INDVARS_IV]]
 ; CHECK-NEXT:    [[ARRAYIDX6:%.*]] = getelementptr inbounds [[STRUCT_GLOBALDATA]], ptr @global_data, i64 0, i32 3, i64 [[INDVARS_IV]]
-; CHECK-NEXT:    [[TMP1:%.*]] = load <2 x float>, ptr [[ARRAYIDX]], align 4
-; CHECK-NEXT:    [[TMP3:%.*]] = load <2 x float>, ptr [[ARRAYIDX6]], align 4
-; CHECK-NEXT:    [[TMP4:%.*]] = fmul <2 x float> [[TMP1]], [[TMP3]]
-; CHECK-NEXT:    [[TMP5:%.*]] = extractelement <2 x float> [[TMP4]], i32 0
+; CHECK-NEXT:    [[TMP0:%.*]] = load <4 x float>, ptr [[ARRAYIDX]], align 4
+; CHECK-NEXT:    [[TMP1:%.*]] = load <4 x float>, ptr [[ARRAYIDX6]], align 4
+; CHECK-NEXT:    [[TMP2:%.*]] = fmul <4 x float> [[TMP0]], [[TMP1]]
+; CHECK-NEXT:    [[TMP5:%.*]] = extractelement <4 x float> [[TMP2]], i32 0
 ; CHECK-NEXT:    [[ADD:%.*]] = fadd float [[DOT_115]], [[TMP5]]
-; CHECK-NEXT:    [[TMP6:%.*]] = extractelement <2 x float> [[TMP4]], i32 1
+; CHECK-NEXT:    [[TMP6:%.*]] = extractelement <4 x float> [[TMP2]], i32 1
 ; CHECK-NEXT:    [[ADD15:%.*]] = fadd float [[ADD]], [[TMP6]]
-; CHECK-NEXT:    [[TMP7:%.*]] = add nuw nsw i64 [[INDVARS_IV]], 2
-; CHECK-NEXT:    [[ARRAYIDX18:%.*]] = getelementptr inbounds [[STRUCT_GLOBALDATA]], ptr @global_data, i64 0, i32 0, i64 [[TMP7]]
-; CHECK-NEXT:    [[ARRAYIDX21:%.*]] = getelementptr inbounds [[STRUCT_GLOBALDATA]], ptr @global_data, i64 0, i32 3, i64 [[TMP7]]
-; CHECK-NEXT:    [[TMP9:%.*]] = load <2 x float>, ptr [[ARRAYIDX18]], align 4
-; CHECK-NEXT:    [[TMP11:%.*]] = load <2 x float>, ptr [[ARRAYIDX21]], align 4
-; CHECK-NEXT:    [[TMP12:%.*]] = fmul <2 x float> [[TMP9]], [[TMP11]]
-; CHECK-NEXT:    [[TMP13:%.*]] = extractelement <2 x float> [[TMP12]], i32 0
+; CHECK-NEXT:    [[TMP13:%.*]] = extractelement <4 x float> [[TMP2]], i32 2
 ; CHECK-NEXT:    [[ADD23:%.*]] = fadd float [[ADD15]], [[TMP13]]
-; CHECK-NEXT:    [[TMP14:%.*]] = extractelement <2 x float> [[TMP12]], i32 1
+; CHECK-NEXT:    [[TMP14:%.*]] = extractelement <4 x float> [[TMP2]], i32 3
 ; CHECK-NEXT:    [[ADD31:%.*]] = fadd float [[ADD23]], [[TMP14]]
 ; CHECK-NEXT:    [[TMP15:%.*]] = add nuw nsw i64 [[INDVARS_IV]], 4
 ; CHECK-NEXT:    [[ARRAYIDX34:%.*]] = getelementptr inbounds [[STRUCT_GLOBALDATA]], ptr @global_data, i64 0, i32 0, i64 [[TMP15]]
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/dot-product.ll b/llvm/test/Transforms/SLPVectorizer/X86/dot-product.ll
index a333d162297bc..1b9dbaca0a34d 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/dot-product.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/dot-product.ll
@@ -9,23 +9,62 @@
 ;
 
 define double @dot4f64(ptr dereferenceable(32) %ptrx, ptr dereferenceable(32) %ptry) {
-; CHECK-LABEL: @dot4f64(
-; CHECK-NEXT:    [[PTRX2:%.*]] = getelementptr inbounds double, ptr [[PTRX:%.*]], i64 2
-; CHECK-NEXT:    [[PTRY2:%.*]] = getelementptr inbounds double, ptr [[PTRY:%.*]], i64 2
-; CHECK-NEXT:    [[TMP1:%.*]] = load <2 x double>, ptr [[PTRX]], align 4
-; CHECK-NEXT:    [[TMP2:%.*]] = load <2 x double>, ptr [[PTRY]], align 4
-; CHECK-NEXT:    [[TMP3:%.*]] = fmul <2 x double> [[TMP1]], [[TMP2]]
-; CHECK-NEXT:    [[TMP4:%.*]] = load <2 x double>, ptr [[PTRX2]], align 4
-; CHECK-NEXT:    [[TMP5:%.*]] = load <2 x double>, ptr [[PTRY2]], align 4
-; CHECK-NEXT:    [[TMP6:%.*]] = fmul <2 x double> [[TMP4]], [[TMP5]]
-; CHECK-NEXT:    [[TMP7:%.*]] = extractelement <2 x double> [[TMP3]], i32 0
-; CHECK-NEXT:    [[TMP8:%.*]] = extractelement <2 x double> [[TMP3]], i32 1
-; CHECK-NEXT:    [[DOT01:%.*]] = fadd double [[TMP7]], [[TMP8]]
-; CHECK-NEXT:    [[TMP9:%.*]] = extractelement <2 x double> [[TMP6]], i32 0
-; CHECK-NEXT:    [[DOT012:%.*]] = fadd double [[DOT01]], [[TMP9]]
-; CHECK-NEXT:    [[TMP10:%.*]] = extractelement <2 x double> [[TMP6]], i32 1
-; CHECK-NEXT:    [[DOT0123:%.*]] = fadd double [[DOT012]], [[TMP10]]
-; CHECK-NEXT:    ret double [[DOT0123]]
+; SSE2-LABEL: @dot4f64(
+; SSE2-NEXT:    [[TMP1:%.*]] = load <4 x double>, ptr [[PTRX:%.*]], align 4
+; SSE2-NEXT:    [[TMP2:%.*]] = load <4 x double>, ptr [[PTRY:%.*]], align 4
+; SSE2-NEXT:    [[TMP3:%.*]] = fmul <4 x double> [[TMP1]], [[TMP2]]
+; SSE2-NEXT:    [[TMP4:%.*]] = extractelement <4 x double> [[TMP3]], i32 0
+; SSE2-NEXT:    [[TMP5:%.*]] = extractelement <4 x double> [[TMP3]], i32 1
+; SSE2-NEXT:    [[DOT01:%.*]] = fadd double [[TMP4]], [[TMP5]]
+; SSE2-NEXT:    [[TMP6:%.*]] = extractelement <4 x double> [[TMP3]], i32 2
+; SSE2-NEXT:    [[DOT012:%.*]] = fadd double [[DOT01]], [[TMP6]]
+; SSE2-NEXT:    [[TMP7:%.*]] = extractelement <4 x double> [[TMP3]], i32 3
+; SSE2-NEXT:    [[DOT0123:%.*]] = fadd double [[DOT012]], [[TMP7]]
+; SSE2-NEXT:    ret double [[DOT0123]]
+;
+; SSE4-LABEL: @dot4f64(
+; SSE4-NEXT:    [[TMP1:%.*]] = load <4 x double>, ptr [[PTRX:%.*]], align 4
+; SSE4-NEXT:    [[TMP2:%.*]] = load <4 x double>, ptr [[PTRY:%.*]], align 4
+; SSE4-NEXT:    [[TMP3:%.*]] = fmul <4 x double> [[TMP1]], [[TMP2]]
+; SSE4-NEXT:    [[TMP4:%.*]] = extractelement <4 x double> [[TMP3]], i32 0
+; SSE4-NEXT:    [[TMP5:%.*]] = extractelement <4 x double> [[TMP3]], i32 1
+; SSE4-NEXT:    [[DOT01:%.*]] = fadd double [[TMP4]], [[TMP5]]
+; SSE4-NEXT:    [[TMP6:%.*]] = extractelement <4 x double> [[TMP3]], i32 2
+; SSE4-NEXT:    [[DOT012:%.*]] = fadd double [[DOT01]], [[TMP6]]
+; SSE4-NEXT:    [[TMP7:%.*]] = extractelement <4 x double> [[TMP3]], i32 3
+; SSE4-NEXT:    [[DOT0123:%.*]] = fadd double [[DOT012]], [[TMP7]]
+; SSE4-NEXT:    ret double [[DOT0123]]
+;
+; AVX-LABEL: @dot4f64(
+; AVX-NEXT:    [[PTRX2:%.*]] = getelementptr inbounds double, ptr [[PTRX:%.*]], i64 2
+; AVX-NEXT:    [[PTRY2:%.*]] = getelementptr inbounds double, ptr [[PTRY:%.*]], i64 2
+; AVX-NEXT:    [[TMP1:%.*]] = load <2 x double>, ptr [[PTRX]], align 4
+; AVX-NEXT:    [[TMP2:%.*]] = load <2 x double>, ptr [[PTRY]], align 4
+; AVX-NEXT:    [[TMP3:%.*]] = fmul <2 x double> [[TMP1]], [[TMP2]]
+; AVX-NEXT:    [[TMP4:%.*]] = load <2 x double>, ptr [[PTRX2]], align 4
+; AVX-NEXT:    [[TMP5:%.*]] = load <2 x double>, ptr [[PTRY2]], align 4
+; AVX-NEXT:    [[TMP6:%.*]] = fmul <2 x double> [[TMP4]], [[TMP5]]
+; AVX-NEXT:    [[TMP7:%.*]] = extractelement <2 x double> [[TMP3]], i32 0
+; AVX-NEXT:    [[TMP8:%.*]] = extractelement <2 x double> [[TMP3]], i32 1
+; AVX-NEXT:    [[DOT01:%.*]] = fadd double [[TMP7]], [[TMP8]]
+; AVX-NEXT:    [[TMP9:%.*]] = extractelement <2 x double> [[TMP6]], i32 0
+; AVX-NEXT:    [[DOT012:%.*]] = fadd double [[DOT01]], [[TMP9]]
+; AVX-NEXT:    [[TMP10:%.*]] = extractelement <2 x double> [[TMP6]], i32 1
+; AVX-NEXT:    [[DOT0123:%.*]] = fadd double [[DOT012]], [[TMP10]]
+; AVX-NEXT:    ret double [[DOT0123]]
+;
+; AVX2-LABEL: @dot4f64(
+; AVX2-NEXT:    [[TMP1:%.*]] = load <4 x double>, ptr [[PTRX:%.*]], align 4
+; AVX2-NEXT:    [[TMP2:%.*]] = load <4 x double>, ptr [[PTRY:%.*]], align 4
+; AVX2-NEXT:    [[TMP3:%.*]] = fmul <4 x double> [[TMP1]], [[TMP2]]
+; AVX2-NEXT:    [[TMP4:%.*]] = extractelement <4 x double> [[TMP3]], i32 0
+; AVX2-NEXT:    [[TMP5:%.*]] = extractelement <4 x double> [[TMP3]], i32 1
+; AVX2-NEXT:    [[DOT01:%.*]] = fadd double [[TMP4]], [[TMP5]]
+; AVX2-NEXT:    [[TMP6:%.*]] = extractelement <4 x double> [[TMP3]], i32 2
+; AVX2-NEXT:    [[DOT012:%.*]] = fadd double [[DOT01]], [[TMP6]]
+; AVX2-NEXT:    [[TMP7:%.*]] = extractelement <4 x double> [[TMP3]], i32 3
+; AVX2-NEXT:    [[DOT0123:%.*]] = fadd double [[DOT012]], [[TMP7]]
+; AVX2-NEXT:    ret double [[DOT0123]]
 ;
   %ptrx1 = getelementptr inbounds double, ptr %ptrx, i64 1
   %ptry1 = getelementptr inbounds double, ptr %ptry, i64 1
@@ -53,20 +92,15 @@ define double @dot4f64(ptr dereferenceable(32) %ptrx, ptr dereferenceable(32) %p
 
 define float @dot4f32(ptr dereferenceable(16) %ptrx, ptr dereferenceable(16) %ptry) {
 ; CHECK-LABEL: @dot4f32(
-; CHECK-NEXT:    [[PTRX2:%.*]] = getelementptr inbounds float, ptr [[PTRX:%.*]], i64 2
-; CHECK-NEXT:    [[PTRY2:%.*]] = getelementptr inbounds float, ptr [[PTRY:%.*]], i64 2
-; CHECK-NEXT:    [[TMP1:%.*]] = load <2 x float>, ptr [[PTRX]], align 4
-; CHECK-NEXT:    [[TMP2:%.*]] = load <2 x float>, ptr [[PTRY]], align 4
-; CHECK-NEXT:    [[TMP3:%.*]] = fmul <2 x float> [[TMP1]], [[TMP2]]
-; CHECK-NEXT:    [[TMP4:%.*]] = load <2 x float>, ptr [[PTRX2]], align 4
-; CHECK-NEXT:    [[TMP5:%.*]] = load <2 x float>, ptr [[PTRY2]], align 4
-; CHECK-NEXT:    [[TMP6:%.*]] = fmul <2 x float> [[TMP4]], [[TMP5]]
-; CHECK-NEXT:    [[TMP7:%.*]] = extractelement <2 x float> [[TMP3]], i32 0
-; CHECK-NEXT:    [[TMP8:%.*]] = extractelement <2 x float> [[TMP3]], i32 1
+; CHECK-NEXT:    [[TMP1:%.*]] = load <4 x float>, ptr [[PTRX:%.*]], align 4
+; CHECK-NEXT:    [[TMP2:%.*]] = load <4 x float>, ptr [[PTRY:%.*]], align 4
+; CHECK-NEXT:    [[TMP3:%.*]] = fmul <4 x float> [[TMP1]], [[TMP2]]
+; CHECK-NEXT:    [[TMP7:%.*]] = extractelement <4 x float> [[TMP3]], i32 0
+; CHECK-NEXT:    [[TMP8:%.*]] = extractelement <4 x float> [[TMP3]], i32 1
 ; CHECK-NEXT:    [[DOT01:%.*]] = fadd float [[TMP7]], [[TMP8]]
-; CHECK-NEXT:    [[TMP9:%.*]] = extractelement <2 x float> [[TMP6]], i32 0
+; CHECK-NEXT:    [[TMP9:%.*]] = extractelement <4 x float> [[TMP3]], i32 2
 ; CHECK-NEXT:    [[DOT012:%.*]] = fadd float [[DOT01]], [[TMP9]]
-; CHECK-NEXT:    [[TMP10:%.*]] = extractelement <2 x float> [[TMP6]], i32 1
+; CHECK-NEXT:    [[TMP10:%.*]] = extractelement <4 x float> [[TMP3]], i32 3
 ; CHECK-NEXT:    [[DOT0123:%.*]] = fadd float [[DOT012]], [[TMP10]]
 ; CHECK-NEXT:    ret float [[DOT0123]]
 ;
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/fmaxnum.ll b/llvm/test/Transforms/SLPVectorizer/X86/fmaxnum.ll
index a42567c5e2e46..822f1051d45d6 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/fmaxnum.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/fmaxnum.ll
@@ -1,8 +1,8 @@
 ; NOTE: Assertions have been autogenerated by utils/update_test_checks.py
 ; RUN: opt < %s -mtriple=x86_64-unknown -passes=slp-vectorizer -S | FileCheck %s --check-prefixes=CHECK,SSE
-; RUN: opt < %s -mtriple=x86_64-unknown -mcpu=corei7-avx -passes=slp-vectorizer -S | FileCheck %s --check-prefixes=CHECK,AVX,AVX256
-; RUN: opt < %s -mtriple=x86_64-unknown -mcpu=bdver1 -passes=slp-vectorizer -S | FileCheck %s --check-prefixes=CHECK,AVX,AVX256
-; RUN: opt < %s -mtriple=x86_64-unknown -mcpu=core-avx2 -passes=slp-vectorizer -S | FileCheck %s --check-prefixes=CHECK,AVX,AVX256
+; RUN: opt < %s -mtriple=x86_64-unknown -mcpu=corei7-avx -passes=slp-vectorizer -S | FileCheck %s --check-prefixes=CHECK,AVX,COREI7
+; RUN: opt < %s -mtriple=x86_64-unknown -mcpu=bdver1 -passes=slp-vectorizer -S | FileCheck %s --check-prefixes=CHECK,AVX,BDVER1
+; RUN: opt < %s -mtriple=x86_64-unknown -mcpu=core-avx2 -passes=slp-vectorizer -S | FileCheck %s --check-prefixes=CHECK,AVX,AVX2
 ; RUN: opt < %s -mtriple=x86_64-unknown -mcpu=skylake-avx512 -mattr=-prefer-256-bit -passes=slp-vectorizer -S | FileCheck %s --check-prefixes=CHECK,AVX,AVX512
 ; RUN: opt < %s -mtriple=x86_64-unknown -mcpu=skylake-avx512 -mattr=+prefer-256-bit -passes=slp-vectorizer -S | FileCheck %s --check-prefixes=CHECK,AVX,AVX256
 
@@ -99,6 +99,46 @@ define void @fmaxnum_8f64() #0 {
 ; SSE-NEXT:    store <2 x double> [[TMP12]], ptr getelementptr inbounds ([8 x double], ptr @dst64, i32 0, i64 6), align 4
 ; SSE-NEXT:    ret void
 ;
+; COREI7-LABEL: @fmaxnum_8f64(
+; COREI7-NEXT:    [[TMP1:%.*]] = load <4 x double>, ptr @srcA64, align 4
+; COREI7-NEXT:    [[TMP2:%.*]] = load <4 x double>, ptr @srcB64, align 4
+; COREI7-NEXT:    [[TMP3:%.*]] = call <4 x double> @llvm.maxnum.v4f64(<4 x double> [[TMP1]], <4 x double> [[TMP2]])
+; COREI7-NEXT:    store <4 x double> [[TMP3]], ptr @dst64, align 4
+; COREI7-NEXT:    [[TMP4:%.*]] = load <4 x double>, ptr getelementptr inbounds ([8 x double], ptr @srcA64, i32 0, i64 4), align 4
+; COREI7-NEXT:    [[TMP5:%.*]] = load <4 x double>, ptr getelementptr inbounds ([8 x double], ptr @srcB64, i32 0, i64 4), align 4
+; COREI7-NEXT:    [[TMP6:%.*]] = call <4 x double> @llvm.maxnum.v4f64(<4 x double> [[TMP4]], <4 x double> [[TMP5]])
+; COREI7-NEXT:    store <4 x double> [[TMP6]], ptr getelementptr inbounds ([8 x double], ptr @dst64, i32 0, i64 4), align 4
+; COREI7-NEXT:    ret void
+;
+; BDVER1-LABEL: @fmaxnum_8f64(
+; BDVER1-NEXT:    [[TMP1:%.*]] = load <4 x double>, ptr @srcA64, align 4
+; BDVER1-NEXT:    [[TMP2:%.*]] = load <4 x double>, ptr @srcB64, align 4
+; BDVER1-NEXT:    [[TMP3:%.*]] = call <4 x double> @llvm.maxnum.v4f64(<4 x double> [[TMP1]], <4 x double> [[TMP2]])
+; BDVER1-NEXT:    store <4 x double> [[TMP3]], ptr @dst64, align 4
+; BDVER1-NEXT:    [[TMP4:%.*]] = load <4 x double>, ptr getelementptr inbounds ([8 x double], ptr @srcA64, i32 0, i64 4), align 4
+; BDVER1-NEXT:    [[TMP5:%.*]] = load <4 x double>, ptr getelementptr inbounds ([8 x double], ptr @srcB64, i32 0, i64 4), align 4
+; BDVER1-NEXT:    [[TMP6:%.*]] = call <4 x double> @llvm.maxnum.v4f64(<4 x double> [[TMP4]], <4 x double> [[TMP5]])
+; BDVER1-NEXT:    store <4 x double> [[TMP6]], ptr getelementptr inbounds ([8 x double], ptr @dst64, i32 0, i64 4), align 4
+; BDVER1-NEXT:    ret void
+;
+; AVX2-LABEL: @fmaxnum_8f64(
+; AVX2-NEXT:    [[TMP1:%.*]] = load <4 x double>, ptr @srcA64, align 4
+; AVX2-NEXT:    [[TMP2:%.*]] = load <4 x double>, ptr @srcB64, align 4
+; AVX2-NEXT:    [[TMP3:%.*]] = call <4 x double> @llvm.maxnum.v4f64(<4 x double> [[TMP1]], <4 x double> [[TMP2]])
+; AVX2-NEXT:    store <4 x double> [[TMP3]], ptr @dst64, align 4
+; AVX2-NEXT:    [[TMP4:%.*]] = load <4 x double>, ptr getelementptr inbounds ([8 x double], ptr @srcA64, i32 0, i64 4), align 4
+; AVX2-NEXT:    [[TMP5:%.*]] = load <4 x double>, ptr getelementptr inbounds ([8 x double], ptr @srcB64, i32 0, i64 4), align 4
+; AVX2-NEXT:    [[TMP6:%.*]] = call <4 x double> @llvm.maxnum.v4f64(<4 x double> [[TMP4]], <4 x double> [[TMP5]])
+; AVX2-NEXT:    store <4 x double> [[TMP6]], ptr getelementptr inbounds ([8 x double], ptr @dst64, i32 0, i64 4), align 4
+; AVX2-NEXT:    ret void
+;
+; AVX512-LABEL: @fmaxnum_8f64(
+; AVX512-NEXT:    [[TMP1:%.*]] = load <8 x double>, ptr @srcA64, align 4
+; AVX512-NEXT:    [[TMP2:%.*]] = load <8 x double>, ptr @srcB64, align 4
+; AVX512-NEXT:    [[TMP3:%.*]] = call <8 x double> @llvm.maxnum.v8f64(<8 x double> [[TMP1]], <8 x double> [[TMP2]])
+; AVX512-NEXT:    store <8 x double> [[TMP3]], ptr @dst64, align 4
+; AVX512-NEXT:    ret void
+;
 ; AVX256-LABEL: @fmaxnum_8f64(
 ; AVX256-NEXT:    [[TMP1:%.*]] = load <4 x double>, ptr @srcA64, align 4
 ; AVX256-NEXT:    [[TMP2:%.*]] = load <4 x double>, ptr @srcB64, align 4
@@ -109,13 +149,6 @@ define void @fmaxnum_8f64() #0 {
 ; AVX256-NEXT:    [[TMP6:%.*]] = call <4 x double> @llvm.maxnum.v4f64(<4 x double> [[TMP4]], <4 x double> [[TMP5]])
 ; AVX256-NEXT:    store <4 x double> [[TMP6]], ptr getelementptr inbounds ([8 x double], ptr @dst64, i32 0, i64 4), align 4
 ; AVX256-NEXT:    ret void
-;
-; AVX512-LABEL: @fmaxnum_8f64(
-; AVX512-NEXT:    [[TMP1:%.*]] = load <8 x double>, ptr @srcA64, align 4
-; AVX512-NEXT:    [[TMP2:%.*]] = load <8 x double>, ptr @srcB64, align 4
-; AVX512-NEXT:    [[TMP3:%.*]] = call <8 x double> @llvm.maxnum.v8f64(<8 x double> [[TMP1]], <8 x double> [[TMP2]])
-; AVX512-NEXT:    store <8 x double> [[TMP3]], ptr @dst64, align 4
-; AVX512-NEXT:    ret void
 ;
   %a0 = load double, ptr @srcA64, align 4
   %a1 = load double, ptr getelementptr inbounds ([8 x double], ptr @srcA64, i32 0, i64 1), align 4
@@ -253,6 +286,46 @@ define void @fmaxnum_16f32() #0 {
 ; SSE-NEXT:    store <4 x float> [[TMP12]], ptr getelementptr inbounds ([16 x float], ptr @dst32, i32 0, i64 12), align 4
 ; SSE-NEXT:    ret void
 ;
+; COREI7-LABEL: @fmaxnum_16f32(
+; COREI7-NEXT:    [[TMP1:%.*]] = load <8 x float>, ptr @srcA32, align 4
+; COREI7-NEXT:    [[TMP2:%.*]] = load <8 x float>, ptr @srcB32, align 4
+; COREI7-NEXT:    [[TMP3:%.*]] = call <8 x float> @llvm.maxnum.v8f32(<8 x float> [[TMP1]], <8 x float> [[TMP2]])
+; COREI7-NEXT:    store <8 x float> [[TMP3]], ptr @dst32, align 4
+; COREI7-NEXT:    [[TMP4:%.*]] = load <8 x float>, ptr getelementptr inbounds ([16 x float], ptr @srcA32, i32 0, i64 8), align 4
+; COREI7-NEXT:    [[TMP5:%.*]] = load <8 x float>, ptr getelementptr inbounds ([16 x float], ptr @srcB32, i32 0, i64 8), align 4
+; COREI7-NEXT:    [[TMP6:%.*]] = call <8 x float> @llvm.maxnum.v8f32(<8 x float> [[TMP4]], <8 x float> [[TMP5]])
+; COREI7-NEXT:    store <8 x float> [[TMP6]], ptr getelementptr inbounds ([16 x float], ptr @dst32, i32 0, i64 8), align 4
+; COREI7-NEXT:    ret void
+;
+; BDVER1-LABEL: @fmaxnum_16f32(
+; BDVER1-NEXT:    [[TMP1:%.*]] = load <8 x float>, ptr @srcA32, align 4
+; BDVER1-NEXT:    [[TMP2:%.*]] = load <8 x float>, ptr @srcB32, align 4
+; BDVER1-NEXT:    [[TMP3:%.*]] = call <8 x float> @llvm.maxnum.v8f32(<8 x float> [[TMP1]], <8 x float> [[TMP2]])
+; BDVER1-NEXT:    store <8 x float> [[TMP3]], ptr @dst32, align 4
+; BDVER1-NEXT:    [[TMP4:%.*]] = load <8 x float>, ptr getelementptr inbounds ([16 x float], ptr @srcA32, i32 0, i64 8), align 4
+; BDVER1-NEXT:    [[TMP5:%.*]] = load <8 x float>, ptr getelementptr inbounds ([16 x float], ptr @srcB32, i32 0, i64 8), align 4
+; BDVER1-NEXT:    [[TMP6:%.*]] = call <8 x float> @llvm.maxnum.v8f32(<8 x float> [[TMP4]], <8 x float> [[TMP5]])
+; BDVER1-NEXT:    store <8 x float> [[TMP6]], ptr getelementptr inbounds ([16 x float], ptr @dst32, i32 0, i64 8), align 4
+; BDVER1-NEXT:    ret void
+;
+; AVX2-LABEL: @fmaxnum_16f32(
+; AVX2-NEXT:    [[TMP1:%.*]] = load <8 x float>, ptr @srcA32, align 4
+; AVX2-NEXT:    [[TMP2:%.*]] = load <8 x float>, ptr @srcB32, align 4
+; AVX2-NEXT:    [[TMP3:%.*]] = call <8 x float> @llvm.maxnum.v8f32(<8 x float> [[TMP1]], <8 x float> [[TMP2]])
+; AVX2-NEXT:    store <8 x float> [[TMP3]], ptr @dst32, align 4
+; AVX2-NEXT:    [[TMP4:%.*]] = load <8 x float>, ptr getelementptr inbounds ([16 x float], ptr @srcA32, i32 0, i64 8), align 4
+; AVX2-NEXT:    [[TMP5:%.*]] = load <8 x float>, ptr getelementptr inbounds ([16 x float], ptr @srcB32, i32 0, i64 8), align 4
+; AVX2-NEXT:    [[TMP6:%.*]] = call <8 x float> @llvm.maxnum.v8f32(<8 x float> [[TMP4]], <8 x float> [[TMP5]])
+; AVX2-NEXT:    store <8 x float> [[TMP6]], ptr getelementptr inbounds ([16 x float], ptr @dst32, i32 0, i64 8), align 4
+; AVX2-NEXT:    ret void
+;
+; AVX512-LABEL: @fmaxnum_16f32(
+; AVX512-NEXT:    [[TMP1:%.*]] = load <16 x float>, ptr @srcA32, align 4
+; AVX512-NEXT:    [[TMP2:%.*]] = load <16 x float>, ptr @srcB32, align 4
+; AVX512-NEXT:    [[TMP3:%.*]] = call <16 x float> @llvm.maxnum.v16f32(<16 x float> [[TMP1]], <16 x float> [[TMP2]])
+; AVX512-NEXT:    store <16 x float> [[TMP3]], ptr @dst32, align 4
+; AVX512-NEXT:    ret void
+;
 ; AVX256-LABEL: @fmaxnum_16f32(
 ; AVX256-NEXT:    [[TMP1:%.*]] = load <8 x float>, ptr @srcA32, align 4
 ; AVX256-NEXT:    [[TMP2:%.*]] = load <8 x float>, ptr @srcB32, align 4
@@ -263,13 +336,6 @@ define void @fmaxnum_16f32() #0 {
 ; AVX256-NEXT:    [[TMP6:%.*]] = call <8 x float> @llvm.maxnum.v8f32(<8 x float> [[TMP4]], <8 x float> [[TMP5]])
 ; AVX256-NEXT:    store <8 x float> [[TMP6]], ptr getelementptr inbounds ([16 x float], ptr @dst32, i32 0, i64 8), align 4
 ; AVX256-NEXT:    ret void
-;
-; AVX512-LABEL: @fmaxnum_16f32(
-; AVX512-NEXT:    [[TMP1:%.*]] = load <16 x float>, ptr @srcA32, align 4
-; AVX512-NEXT:    [[TMP2:%.*]] = load <16 x float>, ptr @srcB32, align 4
-; AVX512-NEXT:    [[TMP3:%.*]] = call <16 x float> @llvm.maxnum.v16f32(<16 x float> [[TMP1]], <16 x float> [[TMP2]])
-; AVX512-NEXT:    store <16 x float> [[TMP3]], ptr @dst32, align 4
-; AVX512-NEXT:    ret void
 ;
   %a0  = load float, ptr @srcA32, align 4
   %a1  = load float, ptr getelementptr inbounds ([16 x float], ptr @srcA32, i32 0, i64  1), align 4
@@ -379,19 +445,84 @@ define float @reduction_v4f32_nnan(ptr %p) {
 ; Negative test - must have nnan.
 
 define float @reduction_v4f32_not_fast(ptr %p) {
-; CHECK-LABEL: @reduction_v4f32_not_fast(
-; CHECK-NEXT:    [[G1:%.*]] = getelementptr inbounds float, ptr [[P:%.*]], i64 1
-; CHECK-NEXT:    [[G2:%.*]] = getelementptr inbounds float, ptr [[P]], i64 2
-; CHECK-NEXT:    [[G3:%.*]] = getelementptr inbounds float, ptr [[P]], i64 3
-; CHECK-NEXT:    [[T0:%.*]] = load float, ptr [[P]], align 4
-; CHECK-NEXT:    [[T1:%.*]] = load float, ptr [[G1]], align 4
-; CHECK-NEXT:    [[T2:%.*]] = load float, ptr [[G2]], align 4
-; CHECK-NEXT:    [[T3:%.*]] = load float, ptr [[G3]], align 4
-; CHECK-NEXT:    [[M1:%.*]] = tail call float @llvm.maxnum.f32(float [[T1]], float [[T0]])
-; CHECK-NEXT:    [[M2:%.*]] = tail call float @llvm.maxnum.f32(float [[T2]], float [[M1]])
-; CHECK-NEXT:    [[M3:%.*]] = tail call float @llvm.maxnum.f32(float [[T3]], float [[M2]])
-; CHECK-NEXT:    ret float [[M3]]
+; SSE-LABEL: @reduction_v4f32_not_fast(
+; SSE-NEXT:    [[TMP1:%.*]] = load <4 x float>, ptr [[P:%.*]], align 4
+; SSE-NEXT:    [[TMP2:%.*]] = extractelement <4 x float> [[TMP1]], i32 0
+; SSE-NEXT:    [[TMP3:%.*]] = extractelement <4 x float> [[TMP1]], i32 1
+; SSE-NEXT:    [[M1:%.*]] = tail call float @llvm.maxnum.f32(float [[TMP3]], float [[TMP2]])
+; SSE-NEXT:    [[TMP4:%.*]] = extractelement <4 x float> [[TMP1]], i32 2
+; SSE-NEXT:    [[M2:%.*]] = tail call float @llvm.maxnum.f32(float [[TMP4]], float [[M1]])
+; SSE-NEXT:    [[TMP5:%.*]] = extractelement <4 x float> [[TMP1]], i32 3
+; SSE-NEXT:    [[M3:%.*]] = tail call float @llvm.maxnum.f32(float [[TMP5]], float [[M2]])
+; SSE-NEXT:    ret float [[M3]]
+;
+; COREI7-LABEL: @reduction_v4f32_not_fast(
+; COREI7-NEXT:    [[TMP1:%.*]] = load <4 x float>, ptr [[P:%.*]], align 4
+; COREI7-NEXT:    [[TMP2:%.*]] = extractelement <4 x float> [[TMP1]], i32 0
+; COREI7-NEXT:    [[TMP3:%.*]] = extractelement <4 x float> [[TMP1]], i32 1
+; COREI7-NEXT:    [[M1:%.*]] = tail call float @llvm.maxnum.f32(float [[TMP3]], float [[TMP2]])
+; COREI7-NEXT:    [[TMP4:%.*]] = extractelement <4 x float> [[TMP1]], i32 2
+; COREI7-NEXT:    [[M2:%.*]] = tail call float @llvm.maxnum.f32(float [[TMP4]], float [[M1]])
+; COREI7-NEXT:    [[TMP5:%.*]] = extractelement <4 x float> [[TMP1]], i32 3
+; COREI7-NEXT:    [[M3:%.*]] = tail call float @llvm.maxnum.f32(float [[TMP5]], float [[M2]])
+; COREI7-NEXT:    ret float [[M3]]
+;
+; BDVER1-LABEL: @reduction_v4f32_not_fast(
+; BDVER1-NEXT:    [[TMP1:%.*]] = load <4 x float>, ptr [[P:%.*]], align 4
+; BDVER1-NEXT:    [[TMP2:%.*]] = extractelement <4 x float> [[TMP1]], i32 0
+; BDVER1-NEXT:    [[TMP3:%.*]] = extractelement <4 x float> [[TMP1]], i32 1
+; BDVER1-NEXT:    [[M1:%.*]] = tail call float @llvm.maxnum.f32(float [[TMP3]], float [[TMP2]])
+; BDVER1-NEXT:    [[TMP4:%.*]] = extractelement <4 x float> [[TMP1]], i32 2
+; BDVER1-NEXT:    [[M2:%.*]] = tail call float @llvm.maxnum.f32(float [[TMP4]], float [[M1]])
+; BDVER1-NEXT:    [[TMP5:%.*]] = extractelement <4 x float> [[TMP1]], i32 3
+; BDVER1-NEXT:    [[M3:%.*]] = tail call float @llvm.maxnum.f32(float [[TMP5]], float [[M2]])
+; BDVER1-NEXT:    ret float [[M3]]
 ;
+; AVX2-LABEL: @reduction_v4f32_not_fast(
+; AVX2-NEXT:    [[G1:%.*]] = getelementptr inbounds float, ptr [[P:%.*]], i64 1
+; AVX2-NEXT:    [[G2:%.*]] = getelementptr inbounds float, ptr [[P]], i64 2
+; AVX2-NEXT:    [[G3:%.*]] = getelementptr inbounds float, ptr [[P]], i64 3
+; AVX2-NEXT:    [[T0:%.*]] = load float, ptr [[P]], align 4
+; AVX2-NEXT:    [[T1:%.*]] = load float, ptr [[G1]], align 4
+; AVX2-NEXT:    [[T2:%.*]] = load float, ptr [[G2]], align 4
+; AVX2-NEXT:    [[T3:%.*]] = load float, ptr [[G3]], align 4
+; AVX2-NEXT:    [[M1:%.*]] = tail call float @llvm.maxnum.f32(float [[T1]], float [[T0]])
+; AVX2-NEXT:    [[M2:%.*]] = tail call float @llvm.maxnum.f32(float [[T2]], float [[M1]])
+; AVX2-NEXT:    [[M3:%.*]] = tail call float @llvm.maxnum.f32(float [[T3]], float [[M2]])
+; AVX2-NEXT:    ret float [[M3]]
+;
+; AVX512-LABEL: @reduction_v4f32_not_fast(
+; AVX512-NEXT:    [[TMP1:%.*]] = load <4 x float>, ptr [[P:%.*]], align 4
+; AVX512-NEXT:    [[TMP2:%.*]] = extractelement <4 x float> [[TMP1]], i32 0
+; AVX512-NEXT:    [[TMP3:%.*]] = extractelement <4 x float> [[TMP1]], i32 1
+; AVX512-NEXT:    [[M1:%.*]] = tail call float @llvm.maxnum.f32(float [[TMP3]], float [[TMP2]])
+; AVX512-NEXT:    [[TMP4:%.*]] = extractelement <4 x float> [[TMP1]], i32 2
+; AVX512-NEXT:    [[M2:%.*]] = tail call float @llvm.maxnum.f32(float [[TMP4]], float [[M1]])
+; AVX512-NEXT:    [[TMP5:%.*]] = extractelement <4 x float> [[TMP1]], i32 3
+; AVX512-NEXT:    [[M3:%.*]] = tail call float @llvm.maxnum.f32(float [[TMP5]], float [[M2]])
+; AVX512-NEXT:    ret float [[M3]]
+;
+; AVX256-LABEL: @reduction_v4f32_not_fast(
+; AVX256-NEXT:    [[TMP1:%.*]] = load <4 x float>, ptr [[P:%.*]], align 4
+; AVX256-NEXT:    [[TMP2:%.*]] = extractelement <4 x float> [[TMP1]], i32 0
+; AVX256-NEXT:    [[TMP3:%.*]] = extractelement <4 x float> [[TMP1]], i32 1
+; AVX256-NEXT:    [[M1:%.*]] = tail call float @llvm.maxnum.f32(float [[TMP3]], float [[TMP2]])
+; AVX256-NEXT:    [[TMP4:%.*]] = extractelement <4 x float> [[TMP1]], i32 2
+; AVX256-NEXT:    [[M2:%.*]] = tail call float @llvm.maxnum.f32(float [[TMP4]], float [[M1]])
+; AVX256-NEXT:    [[TMP5:%.*]] = extractelement <4 x float> [[TMP1]], i32 3
+; AVX256-NEXT:    [[M3:%.*]] = tail call float @llvm.maxnum.f32(float [[TMP5]], float [[M2]])
+; AVX256-NEXT:    ret float [[M3]]
+;
+; PREF-AVX256-LABEL: @reduction_v4f32_not_fast(
+; PREF-AVX256-NEXT:    [[TMP1:%.*]] = load <4 x float>, ptr [[P:%.*]], align 4
+; PREF-AVX256-NEXT:    [[TMP2:%.*]] = extractelement <4 x float> [[TMP1]], i32 0
+; PREF-AVX256-NEXT:    [[TMP3:%.*]] = extractelement <4 x float> [[TMP1]], i32 1
+; PREF-AVX256-NEXT:    [[M1:%.*]] = tail call float @llvm.maxnum.f32(float [[TMP3]], float [[TMP2]])
+; PREF-AVX256-NEXT:    [[TMP4:%.*]] = extractelement <4 x float> [[TMP1]], i32 2
+; PREF-AVX256-NEXT:    [[M2:%.*]] = tail call float @llvm.maxnum.f32(float [[TMP4]], float [[M1]])
+; PREF-AVX256-NEXT:    [[TMP5:%.*]] = extractelement <4 x float> [[TMP1]], i32 3
+; PREF-AVX256-NEXT:    [[M3:%.*]] = tail call float @llvm.maxnum.f32(float [[TMP5]], float [[M2]])
+; PREF-AVX256-NEXT:    ret float [[M3]]
   %g1 = getelementptr inbounds float, ptr %p, i64 1
   %g2 = getelementptr inbounds float, ptr %p, i64 2
   %g3 = getelementptr inbounds float, ptr %p, i64 3
@@ -473,19 +604,88 @@ define double @reduction_v4f64_fast(ptr %p) {
 ; Negative test - must have nnan.
 
 define double @reduction_v4f64_wrong_fmf(ptr %p) {
-; CHECK-LABEL: @reduction_v4f64_wrong_fmf(
-; CHECK-NEXT:    [[G1:%.*]] = getelementptr inbounds double, ptr [[P:%.*]], i64 1
-; CHECK-NEXT:    [[G2:%.*]] = getelementptr inbounds double, ptr [[P]], i64 2
-; CHECK-NEXT:    [[G3:%.*]] = getelementptr inbounds double, ptr [[P]], i64 3
-; CHECK-NEXT:    [[T0:%.*]] = load double, ptr [[P]], align 4
-; CHECK-NEXT:    [[T1:%.*]] = load double, ptr [[G1]], align 4
-; CHECK-NEXT:    [[T2:%.*]] = load double, ptr [[G2]], align 4
-; CHECK-NEXT:    [[T3:%.*]] = load double, ptr [[G3]], align 4
-; CHECK-NEXT:    [[M1:%.*]] = tail call ninf nsz double @llvm.maxnum.f64(double [[T1]], double [[T0]])
-; CHECK-NEXT:    [[M2:%.*]] = tail call ninf nsz double @llvm.maxnum.f64(double [[T2]], double [[M1]])
-; CHECK-NEXT:    [[M3:%.*]] = tail call ninf nsz double @llvm.maxnum.f64(double [[T3]], double [[M2]])
-; CHECK-NEXT:    ret double [[M3]]
+; SSE-LABEL: @reduction_v4f64_wrong_fmf(
+; SSE-NEXT:    [[TMP1:%.*]] = load <4 x double>, ptr [[P:%.*]], align 4
+; SSE-NEXT:    [[TMP2:%.*]] = extractelement <4 x double> [[TMP1]], i32 0
+; SSE-NEXT:    [[TMP3:%.*]] = extractelement <4 x double> [[TMP1]], i32 1
+; SSE-NEXT:    [[M1:%.*]] = tail call ninf nsz double @llvm.maxnum.f64(double [[TMP3]], double [[TMP2]])
+; SSE-NEXT:    [[TMP4:%.*]] = extractelement <4 x double> [[TMP1]], i32 2
+; SSE-NEXT:    [[M2:%.*]] = tail call ninf nsz double @llvm.maxnum.f64(double [[TMP4]], double [[M1]])
+; SSE-NEXT:    [[TMP5:%.*]] = extractelement <4 x double> [[TMP1]], i32 3
+; SSE-NEXT:    [[M3:%.*]] = tail call ninf nsz double @llvm.maxnum.f64(double [[TMP5]], double [[M2]])
+; SSE-NEXT:    ret double [[M3]]
+;
+; COREI7-LABEL: @reduction_v4f64_wrong_fmf(
+; COREI7-NEXT:    [[G1:%.*]] = getelementptr inbounds double, ptr [[P:%.*]], i64 1
+; COREI7-NEXT:    [[G2:%.*]] = getelementptr inbounds double, ptr [[P]], i64 2
+; COREI7-NEXT:    [[G3:%.*]] = getelementptr inbounds double, ptr [[P]], i64 3
+; COREI7-NEXT:    [[T0:%.*]] = load double, ptr [[P]], align 4
+; COREI7-NEXT:    [[T1:%.*]] = load double, ptr [[G1]], align 4
+; COREI7-NEXT:    [[T2:%.*]] = load double, ptr [[G2]], align 4
+; COREI7-NEXT:    [[T3:%.*]] = load double, ptr [[G3]], align 4
+; COREI7-NEXT:    [[M1:%.*]] = tail call ninf nsz double @llvm.maxnum.f64(double [[T1]], double [[T0]])
+; COREI7-NEXT:    [[M2:%.*]] = tail call ninf nsz double @llvm.maxnum.f64(double [[T2]], double [[M1]])
+; COREI7-NEXT:    [[M3:%.*]] = tail call ninf nsz double @llvm.maxnum.f64(double [[T3]], double [[M2]])
+; COREI7-NEXT:    ret double [[M3]]
+;
+; BDVER1-LABEL: @reduction_v4f64_wrong_fmf(
+; BDVER1-NEXT:    [[G1:%.*]] = getelementptr inbounds double, ptr [[P:%.*]], i64 1
+; BDVER1-NEXT:    [[G2:%.*]] = getelementptr inbounds double, ptr [[P]], i64 2
+; BDVER1-NEXT:    [[G3:%.*]] = getelementptr inbounds double, ptr [[P]], i64 3
+; BDVER1-NEXT:    [[T0:%.*]] = load double, ptr [[P]], align 4
+; BDVER1-NEXT:    [[T1:%.*]] = load double, ptr [[G1]], align 4
+; BDVER1-NEXT:    [[T2:%.*]] = load double, ptr [[G2]], align 4
+; BDVER1-NEXT:    [[T3:%.*]] = load double, ptr [[G3]], align 4
+; BDVER1-NEXT:    [[M1:%.*]] = tail call ninf nsz double @llvm.maxnum.f64(double [[T1]], double [[T0]])
+; BDVER1-NEXT:    [[M2:%.*]] = tail call ninf nsz double @llvm.maxnum.f64(double [[T2]], double [[M1]])
+; BDVER1-NEXT:    [[M3:%.*]] = tail call ninf nsz double @llvm.maxnum.f64(double [[T3]], double [[M2]])
+; BDVER1-NEXT:    ret double [[M3]]
+;
+; AVX2-LABEL: @reduction_v4f64_wrong_fmf(
+; AVX2-NEXT:    [[G1:%.*]] = getelementptr inbounds double, ptr [[P:%.*]], i64 1
+; AVX2-NEXT:    [[G2:%.*]] = getelementptr inbounds double, ptr [[P]], i64 2
+; AVX2-NEXT:    [[G3:%.*]] = getelementptr inbounds double, ptr [[P]], i64 3
+; AVX2-NEXT:    [[T0:%.*]] = load double, ptr [[P]], align 4
+; AVX2-NEXT:    [[T1:%.*]] = load double, ptr [[G1]], align 4
+; AVX2-NEXT:    [[T2:%.*]] = load double, ptr [[G2]], align 4
+; AVX2-NEXT:    [[T3:%.*]] = load double, ptr [[G3]], align 4
+; AVX2-NEXT:    [[M1:%.*]] = tail call ninf nsz double @llvm.maxnum.f64(double [[T1]], double [[T0]])
+; AVX2-NEXT:    [[M2:%.*]] = tail call ninf nsz double @llvm.maxnum.f64(double [[T2]], double [[M1]])
+; AVX2-NEXT:    [[M3:%.*]] = tail call ninf nsz double @llvm.maxnum.f64(double [[T3]], double [[M2]])
+; AVX2-NEXT:    ret double [[M3]]
+;
+; AVX512-LABEL: @reduction_v4f64_wrong_fmf(
+; AVX512-NEXT:    [[TMP1:%.*]] = load <4 x double>, ptr [[P:%.*]], align 4
+; AVX512-NEXT:    [[TMP2:%.*]] = extractelement <4 x double> [[TMP1]], i32 0
+; AVX512-NEXT:    [[TMP3:%.*]] = extractelement <4 x double> [[TMP1]], i32 1
+; AVX512-NEXT:    [[M1:%.*]] = tail call ninf nsz double @llvm.maxnum.f64(double [[TMP3]], double [[TMP2]])
+; AVX512-NEXT:    [[TMP4:%.*]] = extractelement <4 x double> [[TMP1]], i32 2
+; AVX512-NEXT:    [[M2:%.*]] = tail call ninf nsz double @llvm.maxnum.f64(double [[TMP4]], double [[M1]])
+; AVX512-NEXT:    [[TMP5:%.*]] = extractelement <4 x double> [[TMP1]], i32 3
+; AVX512-NEXT:    [[M3:%.*]] = tail call ninf nsz double @llvm.maxnum.f64(double [[TMP5]], double [[M2]])
+; AVX512-NEXT:    ret double [[M3]]
+;
+; AVX256-LABEL: @reduction_v4f64_wrong_fmf(
+; AVX256-NEXT:    [[TMP1:%.*]] = load <4 x double>, ptr [[P:%.*]], align 4
+; AVX256-NEXT:    [[TMP2:%.*]] = extractelement <4 x double> [[TMP1]], i32 0
+; AVX256-NEXT:    [[TMP3:%.*]] = extractelement <4 x double> [[TMP1]], i32 1
+; AVX256-NEXT:    [[M1:%.*]] = tail call ninf nsz double @llvm.maxnum.f64(double [[TMP3]], double [[TMP2]])
+; AVX256-NEXT:    [[TMP4:%.*]] = extractelement <4 x double> [[TMP1]], i32 2
+; AVX256-NEXT:    [[M2:%.*]] = tail call ninf nsz double @llvm.maxnum.f64(double [[TMP4]], double [[M1]])
+; AVX256-NEXT:    [[TMP5:%.*]] = extractelement <4 x double> [[TMP1]], i32 3
+; AVX256-NEXT:    [[M3:%.*]] = tail call ninf nsz double @llvm.maxnum.f64(double [[TMP5]], double [[M2]])
+; AVX256-NEXT:    ret double [[M3]]
 ;
+; PREF-AVX256-LABEL: @reduction_v4f64_wrong_fmf(
+; PREF-AVX256-NEXT:    [[TMP1:%.*]] = load <4 x double>, ptr [[P:%.*]], align 4
+; PREF-AVX256-NEXT:    [[TMP2:%.*]] = extractelement <4 x double> [[TMP1]], i32 0
+; PREF-AVX256-NEXT:    [[TMP3:%.*]] = extractelement <4 x double> [[TMP1]], i32 1
+; PREF-AVX256-NEXT:    [[M1:%.*]] = tail call ninf nsz double @llvm.maxnum.f64(double [[TMP3]], double [[TMP2]])
+; PREF-AVX256-NEXT:    [[TMP4:%.*]] = extractelement <4 x double> [[TMP1]], i32 2
+; PREF-AVX256-NEXT:    [[M2:%.*]] = tail call ninf nsz double @llvm.maxnum.f64(double [[TMP4]], double [[M1]])
+; PREF-AVX256-NEXT:    [[TMP5:%.*]] = extractelement <4 x double> [[TMP1]], i32 3
+; PREF-AVX256-NEXT:    [[M3:%.*]] = tail call ninf nsz double @llvm.maxnum.f64(double [[TMP5]], double [[M2]])
+; PREF-AVX256-NEXT:    ret double [[M3]]
   %g1 = getelementptr inbounds double, ptr %p, i64 1
   %g2 = getelementptr inbounds double, ptr %p, i64 2
   %g3 = getelementptr inbounds double, ptr %p, i64 3
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/fminnum.ll b/llvm/test/Transforms/SLPVectorizer/X86/fminnum.ll
index 434fa13e880bb..5eb2328a3dc8f 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/fminnum.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/fminnum.ll
@@ -1,8 +1,8 @@
 ; NOTE: Assertions have been autogenerated by utils/update_test_checks.py
 ; RUN: opt < %s -mtriple=x86_64-unknown -passes=slp-vectorizer -S | FileCheck %s --check-prefixes=CHECK,SSE
-; RUN: opt < %s -mtriple=x86_64-unknown -mcpu=corei7-avx -passes=slp-vectorizer -S | FileCheck %s --check-prefixes=CHECK,AVX,AVX256
-; RUN: opt < %s -mtriple=x86_64-unknown -mcpu=bdver1 -passes=slp-vectorizer -S | FileCheck %s --check-prefixes=CHECK,AVX,AVX256
-; RUN: opt < %s -mtriple=x86_64-unknown -mcpu=core-avx2 -passes=slp-vectorizer -S | FileCheck %s --check-prefixes=CHECK,AVX,AVX256
+; RUN: opt < %s -mtriple=x86_64-unknown -mcpu=corei7-avx -passes=slp-vectorizer -S | FileCheck %s --check-prefixes=CHECK,AVX,COREI7
+; RUN: opt < %s -mtriple=x86_64-unknown -mcpu=bdver1 -passes=slp-vectorizer -S | FileCheck %s --check-prefixes=CHECK,AVX,BDVER1
+; RUN: opt < %s -mtriple=x86_64-unknown -mcpu=core-avx2 -passes=slp-vectorizer -S | FileCheck %s --check-prefixes=CHECK,AVX,AVX2
 ; RUN: opt < %s -mtriple=x86_64-unknown -mcpu=skylake-avx512 -mattr=-prefer-256-bit -passes=slp-vectorizer -S | FileCheck %s --check-prefixes=CHECK,AVX,AVX512
 ; RUN: opt < %s -mtriple=x86_64-unknown -mcpu=skylake-avx512 -mattr=+prefer-256-bit -passes=slp-vectorizer -S | FileCheck %s --check-prefixes=CHECK,AVX,AVX256
 
@@ -99,6 +99,46 @@ define void @fminnum_8f64() #0 {
 ; SSE-NEXT:    store <2 x double> [[TMP12]], ptr getelementptr inbounds ([8 x double], ptr @dst64, i32 0, i64 6), align 4
 ; SSE-NEXT:    ret void
 ;
+; COREI7-LABEL: @fminnum_8f64(
+; COREI7-NEXT:    [[TMP1:%.*]] = load <4 x double>, ptr @srcA64, align 4
+; COREI7-NEXT:    [[TMP2:%.*]] = load <4 x double>, ptr @srcB64, align 4
+; COREI7-NEXT:    [[TMP3:%.*]] = call <4 x double> @llvm.minnum.v4f64(<4 x double> [[TMP1]], <4 x double> [[TMP2]])
+; COREI7-NEXT:    store <4 x double> [[TMP3]], ptr @dst64, align 4
+; COREI7-NEXT:    [[TMP4:%.*]] = load <4 x double>, ptr getelementptr inbounds ([8 x double], ptr @srcA64, i32 0, i64 4), align 4
+; COREI7-NEXT:    [[TMP5:%.*]] = load <4 x double>, ptr getelementptr inbounds ([8 x double], ptr @srcB64, i32 0, i64 4), align 4
+; COREI7-NEXT:    [[TMP6:%.*]] = call <4 x double> @llvm.minnum.v4f64(<4 x double> [[TMP4]], <4 x double> [[TMP5]])
+; COREI7-NEXT:    store <4 x double> [[TMP6]], ptr getelementptr inbounds ([8 x double], ptr @dst64, i32 0, i64 4), align 4
+; COREI7-NEXT:    ret void
+;
+; BDVER1-LABEL: @fminnum_8f64(
+; BDVER1-NEXT:    [[TMP1:%.*]] = load <4 x double>, ptr @srcA64, align 4
+; BDVER1-NEXT:    [[TMP2:%.*]] = load <4 x double>, ptr @srcB64, align 4
+; BDVER1-NEXT:    [[TMP3:%.*]] = call <4 x double> @llvm.minnum.v4f64(<4 x double> [[TMP1]], <4 x double> [[TMP2]])
+; BDVER1-NEXT:    store <4 x double> [[TMP3]], ptr @dst64, align 4
+; BDVER1-NEXT:    [[TMP4:%.*]] = load <4 x double>, ptr getelementptr inbounds ([8 x double], ptr @srcA64, i32 0, i64 4), align 4
+; BDVER1-NEXT:    [[TMP5:%.*]] = load <4 x double>, ptr getelementptr inbounds ([8 x double], ptr @srcB64, i32 0, i64 4), align 4
+; BDVER1-NEXT:    [[TMP6:%.*]] = call <4 x double> @llvm.minnum.v4f64(<4 x double> [[TMP4]], <4 x double> [[TMP5]])
+; BDVER1-NEXT:    store <4 x double> [[TMP6]], ptr getelementptr inbounds ([8 x double], ptr @dst64, i32 0, i64 4), align 4
+; BDVER1-NEXT:    ret void
+;
+; AVX2-LABEL: @fminnum_8f64(
+; AVX2-NEXT:    [[TMP1:%.*]] = load <4 x double>, ptr @srcA64, align 4
+; AVX2-NEXT:    [[TMP2:%.*]] = load <4 x double>, ptr @srcB64, align 4
+; AVX2-NEXT:    [[TMP3:%.*]] = call <4 x double> @llvm.minnum.v4f64(<4 x double> [[TMP1]], <4 x double> [[TMP2]])
+; AVX2-NEXT:    store <4 x double> [[TMP3]], ptr @dst64, align 4
+; AVX2-NEXT:    [[TMP4:%.*]] = load <4 x double>, ptr getelementptr inbounds ([8 x double], ptr @srcA64, i32 0, i64 4), align 4
+; AVX2-NEXT:    [[TMP5:%.*]] = load <4 x double>, ptr getelementptr inbounds ([8 x double], ptr @srcB64, i32 0, i64 4), align 4
+; AVX2-NEXT:    [[TMP6:%.*]] = call <4 x double> @llvm.minnum.v4f64(<4 x double> [[TMP4]], <4 x double> [[TMP5]])
+; AVX2-NEXT:    store <4 x double> [[TMP6]], ptr getelementptr inbounds ([8 x double], ptr @dst64, i32 0, i64 4), align 4
+; AVX2-NEXT:    ret void
+;
+; AVX512-LABEL: @fminnum_8f64(
+; AVX512-NEXT:    [[TMP1:%.*]] = load <8 x double>, ptr @srcA64, align 4
+; AVX512-NEXT:    [[TMP2:%.*]] = load <8 x double>, ptr @srcB64, align 4
+; AVX512-NEXT:    [[TMP3:%.*]] = call <8 x double> @llvm.minnum.v8f64(<8 x double> [[TMP1]], <8 x double> [[TMP2]])
+; AVX512-NEXT:    store <8 x double> [[TMP3]], ptr @dst64, align 4
+; AVX512-NEXT:    ret void
+;
 ; AVX256-LABEL: @fminnum_8f64(
 ; AVX256-NEXT:    [[TMP1:%.*]] = load <4 x double>, ptr @srcA64, align 4
 ; AVX256-NEXT:    [[TMP2:%.*]] = load <4 x double>, ptr @srcB64, align 4
@@ -109,13 +149,6 @@ define void @fminnum_8f64() #0 {
 ; AVX256-NEXT:    [[TMP6:%.*]] = call <4 x double> @llvm.minnum.v4f64(<4 x double> [[TMP4]], <4 x double> [[TMP5]])
 ; AVX256-NEXT:    store <4 x double> [[TMP6]], ptr getelementptr inbounds ([8 x double], ptr @dst64, i32 0, i64 4), align 4
 ; AVX256-NEXT:    ret void
-;
-; AVX512-LABEL: @fminnum_8f64(
-; AVX512-NEXT:    [[TMP1:%.*]] = load <8 x double>, ptr @srcA64, align 4
-; AVX512-NEXT:    [[TMP2:%.*]] = load <8 x double>, ptr @srcB64, align 4
-; AVX512-NEXT:    [[TMP3:%.*]] = call <8 x double> @llvm.minnum.v8f64(<8 x double> [[TMP1]], <8 x double> [[TMP2]])
-; AVX512-NEXT:    store <8 x double> [[TMP3]], ptr @dst64, align 4
-; AVX512-NEXT:    ret void
 ;
   %a0 = load double, ptr @srcA64, align 4
   %a1 = load double, ptr getelementptr inbounds ([8 x double], ptr @srcA64, i32 0, i64 1), align 4
@@ -253,6 +286,46 @@ define void @fminnum_16f32() #0 {
 ; SSE-NEXT:    store <4 x float> [[TMP12]], ptr getelementptr inbounds ([16 x float], ptr @dst32, i32 0, i64 12), align 4
 ; SSE-NEXT:    ret void
 ;
+; COREI7-LABEL: @fminnum_16f32(
+; COREI7-NEXT:    [[TMP1:%.*]] = load <8 x float>, ptr @srcA32, align 4
+; COREI7-NEXT:    [[TMP2:%.*]] = load <8 x float>, ptr @srcB32, align 4
+; COREI7-NEXT:    [[TMP3:%.*]] = call <8 x float> @llvm.minnum.v8f32(<8 x float> [[TMP1]], <8 x float> [[TMP2]])
+; COREI7-NEXT:    store <8 x float> [[TMP3]], ptr @dst32, align 4
+; COREI7-NEXT:    [[TMP4:%.*]] = load <8 x float>, ptr getelementptr inbounds ([16 x float], ptr @srcA32, i32 0, i64 8), align 4
+; COREI7-NEXT:    [[TMP5:%.*]] = load <8 x float>, ptr getelementptr inbounds ([16 x float], ptr @srcB32, i32 0, i64 8), align 4
+; COREI7-NEXT:    [[TMP6:%.*]] = call <8 x float> @llvm.minnum.v8f32(<8 x float> [[TMP4]], <8 x float> [[TMP5]])
+; COREI7-NEXT:    store <8 x float> [[TMP6]], ptr getelementptr inbounds ([16 x float], ptr @dst32, i32 0, i64 8), align 4
+; COREI7-NEXT:    ret void
+;
+; BDVER1-LABEL: @fminnum_16f32(
+; BDVER1-NEXT:    [[TMP1:%.*]] = load <8 x float>, ptr @srcA32, align 4
+; BDVER1-NEXT:    [[TMP2:%.*]] = load <8 x float>, ptr @srcB32, align 4
+; BDVER1-NEXT:    [[TMP3:%.*]] = call <8 x float> @llvm.minnum.v8f32(<8 x float> [[TMP1]], <8 x float> [[TMP2]])
+; BDVER1-NEXT:    store <8 x float> [[TMP3]], ptr @dst32, align 4
+; BDVER1-NEXT:    [[TMP4:%.*]] = load <8 x float>, ptr getelementptr inbounds ([16 x float], ptr @srcA32, i32 0, i64 8), align 4
+; BDVER1-NEXT:    [[TMP5:%.*]] = load <8 x float>, ptr getelementptr inbounds ([16 x float], ptr @srcB32, i32 0, i64 8), align 4
+; BDVER1-NEXT:    [[TMP6:%.*]] = call <8 x float> @llvm.minnum.v8f32(<8 x float> [[TMP4]], <8 x float> [[TMP5]])
+; BDVER1-NEXT:    store <8 x float> [[TMP6]], ptr getelementptr inbounds ([16 x float], ptr @dst32, i32 0, i64 8), align 4
+; BDVER1-NEXT:    ret void
+;
+; AVX2-LABEL: @fminnum_16f32(
+; AVX2-NEXT:    [[TMP1:%.*]] = load <8 x float>, ptr @srcA32, align 4
+; AVX2-NEXT:    [[TMP2:%.*]] = load <8 x float>, ptr @srcB32, align 4
+; AVX2-NEXT:    [[TMP3:%.*]] = call <8 x float> @llvm.minnum.v8f32(<8 x float> [[TMP1]], <8 x float> [[TMP2]])
+; AVX2-NEXT:    store <8 x float> [[TMP3]], ptr @dst32, align 4
+; AVX2-NEXT:    [[TMP4:%.*]] = load <8 x float>, ptr getelementptr inbounds ([16 x float], ptr @srcA32, i32 0, i64 8), align 4
+; AVX2-NEXT:    [[TMP5:%.*]] = load <8 x float>, ptr getelementptr inbounds ([16 x float], ptr @srcB32, i32 0, i64 8), align 4
+; AVX2-NEXT:    [[TMP6:%.*]] = call <8 x float> @llvm.minnum.v8f32(<8 x float> [[TMP4]], <8 x float> [[TMP5]])
+; AVX2-NEXT:    store <8 x float> [[TMP6]], ptr getelementptr inbounds ([16 x float], ptr @dst32, i32 0, i64 8), align 4
+; AVX2-NEXT:    ret void
+;
+; AVX512-LABEL: @fminnum_16f32(
+; AVX512-NEXT:    [[TMP1:%.*]] = load <16 x float>, ptr @srcA32, align 4
+; AVX512-NEXT:    [[TMP2:%.*]] = load <16 x float>, ptr @srcB32, align 4
+; AVX512-NEXT:    [[TMP3:%.*]] = call <16 x float> @llvm.minnum.v16f32(<16 x float> [[TMP1]], <16 x float> [[TMP2]])
+; AVX512-NEXT:    store <16 x float> [[TMP3]], ptr @dst32, align 4
+; AVX512-NEXT:    ret void
+;
 ; AVX256-LABEL: @fminnum_16f32(
 ; AVX256-NEXT:    [[TMP1:%.*]] = load <8 x float>, ptr @srcA32, align 4
 ; AVX256-NEXT:    [[TMP2:%.*]] = load <8 x float>, ptr @srcB32, align 4
@@ -263,13 +336,6 @@ define void @fminnum_16f32() #0 {
 ; AVX256-NEXT:    [[TMP6:%.*]] = call <8 x float> @llvm.minnum.v8f32(<8 x float> [[TMP4]], <8 x float> [[TMP5]])
 ; AVX256-NEXT:    store <8 x float> [[TMP6]], ptr getelementptr inbounds ([16 x float], ptr @dst32, i32 0, i64 8), align 4
 ; AVX256-NEXT:    ret void
-;
-; AVX512-LABEL: @fminnum_16f32(
-; AVX512-NEXT:    [[TMP1:%.*]] = load <16 x float>, ptr @srcA32, align 4
-; AVX512-NEXT:    [[TMP2:%.*]] = load <16 x float>, ptr @srcB32, align 4
-; AVX512-NEXT:    [[TMP3:%.*]] = call <16 x float> @llvm.minnum.v16f32(<16 x float> [[TMP1]], <16 x float> [[TMP2]])
-; AVX512-NEXT:    store <16 x float> [[TMP3]], ptr @dst32, align 4
-; AVX512-NEXT:    ret void
 ;
   %a0  = load float, ptr @srcA32, align 4
   %a1  = load float, ptr getelementptr inbounds ([16 x float], ptr @srcA32, i32 0, i64  1), align 4
@@ -379,18 +445,73 @@ define float @reduction_v4f32_nnan(ptr %p) {
 ; Negative test - must have nnan.
 
 define float @reduction_v4f32_wrong_fmf(ptr %p) {
-; CHECK-LABEL: @reduction_v4f32_wrong_fmf(
-; CHECK-NEXT:    [[G1:%.*]] = getelementptr inbounds float, ptr [[P:%.*]], i64 1
-; CHECK-NEXT:    [[G2:%.*]] = getelementptr inbounds float, ptr [[P]], i64 2
-; CHECK-NEXT:    [[G3:%.*]] = getelementptr inbounds float, ptr [[P]], i64 3
-; CHECK-NEXT:    [[T0:%.*]] = load float, ptr [[P]], align 4
-; CHECK-NEXT:    [[T1:%.*]] = load float, ptr [[G1]], align 4
-; CHECK-NEXT:    [[T2:%.*]] = load float, ptr [[G2]], align 4
-; CHECK-NEXT:    [[T3:%.*]] = load float, ptr [[G3]], align 4
-; CHECK-NEXT:    [[M1:%.*]] = tail call reassoc nsz float @llvm.minnum.f32(float [[T1]], float [[T0]])
-; CHECK-NEXT:    [[M2:%.*]] = tail call reassoc nsz float @llvm.minnum.f32(float [[T2]], float [[M1]])
-; CHECK-NEXT:    [[M3:%.*]] = tail call reassoc nsz float @llvm.minnum.f32(float [[T3]], float [[M2]])
-; CHECK-NEXT:    ret float [[M3]]
+; SSE-LABEL: @reduction_v4f32_wrong_fmf(
+; SSE-NEXT:    [[TMP1:%.*]] = load <4 x float>, ptr [[P:%.*]], align 4
+; SSE-NEXT:    [[TMP2:%.*]] = extractelement <4 x float> [[TMP1]], i32 0
+; SSE-NEXT:    [[TMP3:%.*]] = extractelement <4 x float> [[TMP1]], i32 1
+; SSE-NEXT:    [[M1:%.*]] = tail call reassoc nsz float @llvm.minnum.f32(float [[TMP3]], float [[TMP2]])
+; SSE-NEXT:    [[TMP4:%.*]] = extractelement <4 x float> [[TMP1]], i32 2
+; SSE-NEXT:    [[M2:%.*]] = tail call reassoc nsz float @llvm.minnum.f32(float [[TMP4]], float [[M1]])
+; SSE-NEXT:    [[TMP5:%.*]] = extractelement <4 x float> [[TMP1]], i32 3
+; SSE-NEXT:    [[M3:%.*]] = tail call reassoc nsz float @llvm.minnum.f32(float [[TMP5]], float [[M2]])
+; SSE-NEXT:    ret float [[M3]]
+;
+; COREI7-LABEL: @reduction_v4f32_wrong_fmf(
+; COREI7-NEXT:    [[TMP1:%.*]] = load <4 x float>, ptr [[P:%.*]], align 4
+; COREI7-NEXT:    [[TMP2:%.*]] = extractelement <4 x float> [[TMP1]], i32 0
+; COREI7-NEXT:    [[TMP3:%.*]] = extractelement <4 x float> [[TMP1]], i32 1
+; COREI7-NEXT:    [[M1:%.*]] = tail call reassoc nsz float @llvm.minnum.f32(float [[TMP3]], float [[TMP2]])
+; COREI7-NEXT:    [[TMP4:%.*]] = extractelement <4 x float> [[TMP1]], i32 2
+; COREI7-NEXT:    [[M2:%.*]] = tail call reassoc nsz float @llvm.minnum.f32(float [[TMP4]], float [[M1]])
+; COREI7-NEXT:    [[TMP5:%.*]] = extractelement <4 x float> [[TMP1]], i32 3
+; COREI7-NEXT:    [[M3:%.*]] = tail call reassoc nsz float @llvm.minnum.f32(float [[TMP5]], float [[M2]])
+; COREI7-NEXT:    ret float [[M3]]
+;
+; BDVER1-LABEL: @reduction_v4f32_wrong_fmf(
+; BDVER1-NEXT:    [[TMP1:%.*]] = load <4 x float>, ptr [[P:%.*]], align 4
+; BDVER1-NEXT:    [[TMP2:%.*]] = extractelement <4 x float> [[TMP1]], i32 0
+; BDVER1-NEXT:    [[TMP3:%.*]] = extractelement <4 x float> [[TMP1]], i32 1
+; BDVER1-NEXT:    [[M1:%.*]] = tail call reassoc nsz float @llvm.minnum.f32(float [[TMP3]], float [[TMP2]])
+; BDVER1-NEXT:    [[TMP4:%.*]] = extractelement <4 x float> [[TMP1]], i32 2
+; BDVER1-NEXT:    [[M2:%.*]] = tail call reassoc nsz float @llvm.minnum.f32(float [[TMP4]], float [[M1]])
+; BDVER1-NEXT:    [[TMP5:%.*]] = extractelement <4 x float> [[TMP1]], i32 3
+; BDVER1-NEXT:    [[M3:%.*]] = tail call reassoc nsz float @llvm.minnum.f32(float [[TMP5]], float [[M2]])
+; BDVER1-NEXT:    ret float [[M3]]
+;
+; AVX2-LABEL: @reduction_v4f32_wrong_fmf(
+; AVX2-NEXT:    [[G1:%.*]] = getelementptr inbounds float, ptr [[P:%.*]], i64 1
+; AVX2-NEXT:    [[G2:%.*]] = getelementptr inbounds float, ptr [[P]], i64 2
+; AVX2-NEXT:    [[G3:%.*]] = getelementptr inbounds float, ptr [[P]], i64 3
+; AVX2-NEXT:    [[T0:%.*]] = load float, ptr [[P]], align 4
+; AVX2-NEXT:    [[T1:%.*]] = load float, ptr [[G1]], align 4
+; AVX2-NEXT:    [[T2:%.*]] = load float, ptr [[G2]], align 4
+; AVX2-NEXT:    [[T3:%.*]] = load float, ptr [[G3]], align 4
+; AVX2-NEXT:    [[M1:%.*]] = tail call reassoc nsz float @llvm.minnum.f32(float [[T1]], float [[T0]])
+; AVX2-NEXT:    [[M2:%.*]] = tail call reassoc nsz float @llvm.minnum.f32(float [[T2]], float [[M1]])
+; AVX2-NEXT:    [[M3:%.*]] = tail call reassoc nsz float @llvm.minnum.f32(float [[T3]], float [[M2]])
+; AVX2-NEXT:    ret float [[M3]]
+;
+; AVX512-LABEL: @reduction_v4f32_wrong_fmf(
+; AVX512-NEXT:    [[TMP1:%.*]] = load <4 x float>, ptr [[P:%.*]], align 4
+; AVX512-NEXT:    [[TMP2:%.*]] = extractelement <4 x float> [[TMP1]], i32 0
+; AVX512-NEXT:    [[TMP3:%.*]] = extractelement <4 x float> [[TMP1]], i32 1
+; AVX512-NEXT:    [[M1:%.*]] = tail call reassoc nsz float @llvm.minnum.f32(float [[TMP3]], float [[TMP2]])
+; AVX512-NEXT:    [[TMP4:%.*]] = extractelement <4 x float> [[TMP1]], i32 2
+; AVX512-NEXT:    [[M2:%.*]] = tail call reassoc nsz float @llvm.minnum.f32(float [[TMP4]], float [[M1]])
+; AVX512-NEXT:    [[TMP5:%.*]] = extractelement <4 x float> [[TMP1]], i32 3
+; AVX512-NEXT:    [[M3:%.*]] = tail call reassoc nsz float @llvm.minnum.f32(float [[TMP5]], float [[M2]])
+; AVX512-NEXT:    ret float [[M3]]
+;
+; AVX256-LABEL: @reduction_v4f32_wrong_fmf(
+; AVX256-NEXT:    [[TMP1:%.*]] = load <4 x float>, ptr [[P:%.*]], align 4
+; AVX256-NEXT:    [[TMP2:%.*]] = extractelement <4 x float> [[TMP1]], i32 0
+; AVX256-NEXT:    [[TMP3:%.*]] = extractelement <4 x float> [[TMP1]], i32 1
+; AVX256-NEXT:    [[M1:%.*]] = tail call reassoc nsz float @llvm.minnum.f32(float [[TMP3]], float [[TMP2]])
+; AVX256-NEXT:    [[TMP4:%.*]] = extractelement <4 x float> [[TMP1]], i32 2
+; AVX256-NEXT:    [[M2:%.*]] = tail call reassoc nsz float @llvm.minnum.f32(float [[TMP4]], float [[M1]])
+; AVX256-NEXT:    [[TMP5:%.*]] = extractelement <4 x float> [[TMP1]], i32 3
+; AVX256-NEXT:    [[M3:%.*]] = tail call reassoc nsz float @llvm.minnum.f32(float [[TMP5]], float [[M2]])
+; AVX256-NEXT:    ret float [[M3]]
 ;
   %g1 = getelementptr inbounds float, ptr %p, i64 1
   %g2 = getelementptr inbounds float, ptr %p, i64 2
@@ -473,18 +594,77 @@ define double @reduction_v4f64_fast(ptr %p) {
 ; Negative test - must have nnan.
 
 define double @reduction_v4f64_not_fast(ptr %p) {
-; CHECK-LABEL: @reduction_v4f64_not_fast(
-; CHECK-NEXT:    [[G1:%.*]] = getelementptr inbounds double, ptr [[P:%.*]], i64 1
-; CHECK-NEXT:    [[G2:%.*]] = getelementptr inbounds double, ptr [[P]], i64 2
-; CHECK-NEXT:    [[G3:%.*]] = getelementptr inbounds double, ptr [[P]], i64 3
-; CHECK-NEXT:    [[T0:%.*]] = load double, ptr [[P]], align 4
-; CHECK-NEXT:    [[T1:%.*]] = load double, ptr [[G1]], align 4
-; CHECK-NEXT:    [[T2:%.*]] = load double, ptr [[G2]], align 4
-; CHECK-NEXT:    [[T3:%.*]] = load double, ptr [[G3]], align 4
-; CHECK-NEXT:    [[M1:%.*]] = tail call double @llvm.minnum.f64(double [[T1]], double [[T0]])
-; CHECK-NEXT:    [[M2:%.*]] = tail call double @llvm.minnum.f64(double [[T2]], double [[M1]])
-; CHECK-NEXT:    [[M3:%.*]] = tail call double @llvm.minnum.f64(double [[T3]], double [[M2]])
-; CHECK-NEXT:    ret double [[M3]]
+; SSE-LABEL: @reduction_v4f64_not_fast(
+; SSE-NEXT:    [[TMP1:%.*]] = load <4 x double>, ptr [[P:%.*]], align 4
+; SSE-NEXT:    [[TMP2:%.*]] = extractelement <4 x double> [[TMP1]], i32 0
+; SSE-NEXT:    [[TMP3:%.*]] = extractelement <4 x double> [[TMP1]], i32 1
+; SSE-NEXT:    [[M1:%.*]] = tail call double @llvm.minnum.f64(double [[TMP3]], double [[TMP2]])
+; SSE-NEXT:    [[TMP4:%.*]] = extractelement <4 x double> [[TMP1]], i32 2
+; SSE-NEXT:    [[M2:%.*]] = tail call double @llvm.minnum.f64(double [[TMP4]], double [[M1]])
+; SSE-NEXT:    [[TMP5:%.*]] = extractelement <4 x double> [[TMP1]], i32 3
+; SSE-NEXT:    [[M3:%.*]] = tail call double @llvm.minnum.f64(double [[TMP5]], double [[M2]])
+; SSE-NEXT:    ret double [[M3]]
+;
+; COREI7-LABEL: @reduction_v4f64_not_fast(
+; COREI7-NEXT:    [[G1:%.*]] = getelementptr inbounds double, ptr [[P:%.*]], i64 1
+; COREI7-NEXT:    [[G2:%.*]] = getelementptr inbounds double, ptr [[P]], i64 2
+; COREI7-NEXT:    [[G3:%.*]] = getelementptr inbounds double, ptr [[P]], i64 3
+; COREI7-NEXT:    [[T0:%.*]] = load double, ptr [[P]], align 4
+; COREI7-NEXT:    [[T1:%.*]] = load double, ptr [[G1]], align 4
+; COREI7-NEXT:    [[T2:%.*]] = load double, ptr [[G2]], align 4
+; COREI7-NEXT:    [[T3:%.*]] = load double, ptr [[G3]], align 4
+; COREI7-NEXT:    [[M1:%.*]] = tail call double @llvm.minnum.f64(double [[T1]], double [[T0]])
+; COREI7-NEXT:    [[M2:%.*]] = tail call double @llvm.minnum.f64(double [[T2]], double [[M1]])
+; COREI7-NEXT:    [[M3:%.*]] = tail call double @llvm.minnum.f64(double [[T3]], double [[M2]])
+; COREI7-NEXT:    ret double [[M3]]
+;
+; BDVER1-LABEL: @reduction_v4f64_not_fast(
+; BDVER1-NEXT:    [[G1:%.*]] = getelementptr inbounds double, ptr [[P:%.*]], i64 1
+; BDVER1-NEXT:    [[G2:%.*]] = getelementptr inbounds double, ptr [[P]], i64 2
+; BDVER1-NEXT:    [[G3:%.*]] = getelementptr inbounds double, ptr [[P]], i64 3
+; BDVER1-NEXT:    [[T0:%.*]] = load double, ptr [[P]], align 4
+; BDVER1-NEXT:    [[T1:%.*]] = load double, ptr [[G1]], align 4
+; BDVER1-NEXT:    [[T2:%.*]] = load double, ptr [[G2]], align 4
+; BDVER1-NEXT:    [[T3:%.*]] = load double, ptr [[G3]], align 4
+; BDVER1-NEXT:    [[M1:%.*]] = tail call double @llvm.minnum.f64(double [[T1]], double [[T0]])
+; BDVER1-NEXT:    [[M2:%.*]] = tail call double @llvm.minnum.f64(double [[T2]], double [[M1]])
+; BDVER1-NEXT:    [[M3:%.*]] = tail call double @llvm.minnum.f64(double [[T3]], double [[M2]])
+; BDVER1-NEXT:    ret double [[M3]]
+;
+; AVX2-LABEL: @reduction_v4f64_not_fast(
+; AVX2-NEXT:    [[G1:%.*]] = getelementptr inbounds double, ptr [[P:%.*]], i64 1
+; AVX2-NEXT:    [[G2:%.*]] = getelementptr inbounds double, ptr [[P]], i64 2
+; AVX2-NEXT:    [[G3:%.*]] = getelementptr inbounds double, ptr [[P]], i64 3
+; AVX2-NEXT:    [[T0:%.*]] = load double, ptr [[P]], align 4
+; AVX2-NEXT:    [[T1:%.*]] = load double, ptr [[G1]], align 4
+; AVX2-NEXT:    [[T2:%.*]] = load double, ptr [[G2]], align 4
+; AVX2-NEXT:    [[T3:%.*]] = load double, ptr [[G3]], align 4
+; AVX2-NEXT:    [[M1:%.*]] = tail call double @llvm.minnum.f64(double [[T1]], double [[T0]])
+; AVX2-NEXT:    [[M2:%.*]] = tail call double @llvm.minnum.f64(double [[T2]], double [[M1]])
+; AVX2-NEXT:    [[M3:%.*]] = tail call double @llvm.minnum.f64(double [[T3]], double [[M2]])
+; AVX2-NEXT:    ret double [[M3]]
+;
+; AVX512-LABEL: @reduction_v4f64_not_fast(
+; AVX512-NEXT:    [[TMP1:%.*]] = load <4 x double>, ptr [[P:%.*]], align 4
+; AVX512-NEXT:    [[TMP2:%.*]] = extractelement <4 x double> [[TMP1]], i32 0
+; AVX512-NEXT:    [[TMP3:%.*]] = extractelement <4 x double> [[TMP1]], i32 1
+; AVX512-NEXT:    [[M1:%.*]] = tail call double @llvm.minnum.f64(double [[TMP3]], double [[TMP2]])
+; AVX512-NEXT:    [[TMP4:%.*]] = extractelement <4 x double> [[TMP1]], i32 2
+; AVX512-NEXT:    [[M2:%.*]] = tail call double @llvm.minnum.f64(double [[TMP4]], double [[M1]])
+; AVX512-NEXT:    [[TMP5:%.*]] = extractelement <4 x double> [[TMP1]], i32 3
+; AVX512-NEXT:    [[M3:%.*]] = tail call double @llvm.minnum.f64(double [[TMP5]], double [[M2]])
+; AVX512-NEXT:    ret double [[M3]]
+;
+; AVX256-LABEL: @reduction_v4f64_not_fast(
+; AVX256-NEXT:    [[TMP1:%.*]] = load <4 x double>, ptr [[P:%.*]], align 4
+; AVX256-NEXT:    [[TMP2:%.*]] = extractelement <4 x double> [[TMP1]], i32 0
+; AVX256-NEXT:    [[TMP3:%.*]] = extractelement <4 x double> [[TMP1]], i32 1
+; AVX256-NEXT:    [[M1:%.*]] = tail call double @llvm.minnum.f64(double [[TMP3]], double [[TMP2]])
+; AVX256-NEXT:    [[TMP4:%.*]] = extractelement <4 x double> [[TMP1]], i32 2
+; AVX256-NEXT:    [[M2:%.*]] = tail call double @llvm.minnum.f64(double [[TMP4]], double [[M1]])
+; AVX256-NEXT:    [[TMP5:%.*]] = extractelement <4 x double> [[TMP1]], i32 3
+; AVX256-NEXT:    [[M3:%.*]] = tail call double @llvm.minnum.f64(double [[TMP5]], double [[M2]])
+; AVX256-NEXT:    ret double [[M3]]
 ;
   %g1 = getelementptr inbounds double, ptr %p, i64 1
   %g2 = getelementptr inbounds double, ptr %p, i64 2
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/horizontal-list.ll b/llvm/test/Transforms/SLPVectorizer/X86/horizontal-list.ll
index 5cdbedb7c6dad..4e434a61e1f1c 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/horizontal-list.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/horizontal-list.ll
@@ -914,16 +914,14 @@ define float @extra_args_no_fast(ptr %x, float %a, float %b) {
 ; THRESHOLD-LABEL: @extra_args_no_fast(
 ; THRESHOLD-NEXT:    [[ADDC:%.*]] = fadd fast float [[B:%.*]], 3.000000e+00
 ; THRESHOLD-NEXT:    [[ADD:%.*]] = fadd fast float [[A:%.*]], [[ADDC]]
-; THRESHOLD-NEXT:    [[ARRAYIDX3:%.*]] = getelementptr inbounds float, ptr [[X:%.*]], i64 1
-; THRESHOLD-NEXT:    [[ARRAYIDX3_1:%.*]] = getelementptr inbounds float, ptr [[X]], i64 2
-; THRESHOLD-NEXT:    [[ARRAYIDX3_2:%.*]] = getelementptr inbounds float, ptr [[X]], i64 3
-; THRESHOLD-NEXT:    [[T0:%.*]] = load float, ptr [[X]], align 4
-; THRESHOLD-NEXT:    [[T1:%.*]] = load float, ptr [[ARRAYIDX3]], align 4
-; THRESHOLD-NEXT:    [[T2:%.*]] = load float, ptr [[ARRAYIDX3_1]], align 4
-; THRESHOLD-NEXT:    [[T3:%.*]] = load float, ptr [[ARRAYIDX3_2]], align 4
+; THRESHOLD-NEXT:    [[TMP1:%.*]] = load <4 x float>, ptr [[X:%.*]], align 4
+; THRESHOLD-NEXT:    [[T0:%.*]] = extractelement <4 x float> [[TMP1]], i32 0
 ; THRESHOLD-NEXT:    [[ADD1:%.*]] = fadd fast float [[T0]], [[ADD]]
+; THRESHOLD-NEXT:    [[T1:%.*]] = extractelement <4 x float> [[TMP1]], i32 1
 ; THRESHOLD-NEXT:    [[ADD4:%.*]] = fadd fast float [[T1]], [[ADD1]]
+; THRESHOLD-NEXT:    [[T2:%.*]] = extractelement <4 x float> [[TMP1]], i32 2
 ; THRESHOLD-NEXT:    [[ADD4_1:%.*]] = fadd float [[T2]], [[ADD4]]
+; THRESHOLD-NEXT:    [[T3:%.*]] = extractelement <4 x float> [[TMP1]], i32 3
 ; THRESHOLD-NEXT:    [[ADD4_2:%.*]] = fadd fast float [[T3]], [[ADD4_1]]
 ; THRESHOLD-NEXT:    [[ADD5:%.*]] = fadd fast float [[ADD4_2]], [[A]]
 ; THRESHOLD-NEXT:    ret float [[ADD5]]
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/phi.ll b/llvm/test/Transforms/SLPVectorizer/X86/phi.ll
index 17ae33652b6d8..caae1e3dc7da8 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/phi.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/phi.ll
@@ -136,45 +136,47 @@ for.end:                                          ; preds = %for.body
 define float @foo3(ptr nocapture readonly %A) #0 {
 ; CHECK-LABEL: @foo3(
 ; CHECK-NEXT:  entry:
-; CHECK-NEXT:    [[ARRAYIDX1:%.*]] = getelementptr inbounds float, ptr [[A:%.*]], i64 1
-; CHECK-NEXT:    [[TMP0:%.*]] = load <2 x float>, ptr [[A]], align 4
-; CHECK-NEXT:    [[TMP1:%.*]] = load <4 x float>, ptr [[ARRAYIDX1]], align 4
-; CHECK-NEXT:    [[TMP2:%.*]] = extractelement <2 x float> [[TMP0]], i32 0
+; CHECK-NEXT:    [[TMP1:%.*]] = load <4 x float>, ptr [[ARRAYIDX1:%.*]], align 4
+; CHECK-NEXT:    [[ARRAYIDX4:%.*]] = getelementptr inbounds float, ptr [[ARRAYIDX1]], i64 4
+; CHECK-NEXT:    [[TMP2:%.*]] = load float, ptr [[ARRAYIDX4]], align 4
+; CHECK-NEXT:    [[TMP0:%.*]] = shufflevector <4 x float> [[TMP1]], <4 x float> poison, <2 x i32> <i32 0, i32 1>
 ; CHECK-NEXT:    br label [[FOR_BODY:%.*]]
 ; CHECK:       for.body:
 ; CHECK-NEXT:    [[INDVARS_IV:%.*]] = phi i64 [ 0, [[ENTRY:%.*]] ], [ [[INDVARS_IV_NEXT:%.*]], [[FOR_BODY]] ]
 ; CHECK-NEXT:    [[R_052:%.*]] = phi float [ [[TMP2]], [[ENTRY]] ], [ [[ADD6:%.*]], [[FOR_BODY]] ]
 ; CHECK-NEXT:    [[TMP3:%.*]] = phi <4 x float> [ [[TMP1]], [[ENTRY]] ], [ [[TMP15:%.*]], [[FOR_BODY]] ]
 ; CHECK-NEXT:    [[TMP4:%.*]] = phi <2 x float> [ [[TMP0]], [[ENTRY]] ], [ [[TMP7:%.*]], [[FOR_BODY]] ]
-; CHECK-NEXT:    [[TMP5:%.*]] = extractelement <2 x float> [[TMP4]], i32 0
-; CHECK-NEXT:    [[MUL:%.*]] = fmul float [[TMP5]], 7.000000e+00
-; CHECK-NEXT:    [[ADD6]] = fadd float [[R_052]], [[MUL]]
 ; CHECK-NEXT:    [[TMP6:%.*]] = add nsw i64 [[INDVARS_IV]], 2
-; CHECK-NEXT:    [[ARRAYIDX14:%.*]] = getelementptr inbounds float, ptr [[A]], i64 [[TMP6]]
+; CHECK-NEXT:    [[ARRAYIDX14:%.*]] = getelementptr inbounds float, ptr [[ARRAYIDX1]], i64 [[TMP6]]
+; CHECK-NEXT:    [[TMP9:%.*]] = load float, ptr [[ARRAYIDX14]], align 4
 ; CHECK-NEXT:    [[INDVARS_IV_NEXT]] = add nuw nsw i64 [[INDVARS_IV]], 3
-; CHECK-NEXT:    [[ARRAYIDX19:%.*]] = getelementptr inbounds float, ptr [[A]], i64 [[INDVARS_IV_NEXT]]
-; CHECK-NEXT:    [[TMP8:%.*]] = load <2 x float>, ptr [[ARRAYIDX14]], align 4
+; CHECK-NEXT:    [[ARRAYIDX19:%.*]] = getelementptr inbounds float, ptr [[ARRAYIDX1]], i64 [[INDVARS_IV_NEXT]]
+; CHECK-NEXT:    [[TMP11:%.*]] = add nsw i64 [[INDVARS_IV]], 4
+; CHECK-NEXT:    [[ARRAYIDX24:%.*]] = getelementptr inbounds float, ptr [[ARRAYIDX1]], i64 [[TMP11]]
+; CHECK-NEXT:    [[TMP8:%.*]] = load float, ptr [[ARRAYIDX24]], align 4
 ; CHECK-NEXT:    [[TMP7]] = load <2 x float>, ptr [[ARRAYIDX19]], align 4
-; CHECK-NEXT:    [[TMP9:%.*]] = shufflevector <2 x float> [[TMP8]], <2 x float> poison, <4 x i32> <i32 poison, i32 0, i32 1, i32 poison>
-; CHECK-NEXT:    [[TMP10:%.*]] = shufflevector <2 x float> [[TMP4]], <2 x float> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
-; CHECK-NEXT:    [[TMP11:%.*]] = shufflevector <4 x float> [[TMP9]], <4 x float> [[TMP10]], <4 x i32> <i32 5, i32 1, i32 2, i32 poison>
+; CHECK-NEXT:    [[TMP10:%.*]] = insertelement <4 x float> poison, float [[TMP9]], i32 2
 ; CHECK-NEXT:    [[TMP12:%.*]] = shufflevector <2 x float> [[TMP7]], <2 x float> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
-; CHECK-NEXT:    [[TMP13:%.*]] = shufflevector <4 x float> [[TMP11]], <4 x float> [[TMP12]], <4 x i32> <i32 0, i32 1, i32 2, i32 5>
-; CHECK-NEXT:    [[TMP14:%.*]] = fmul <4 x float> [[TMP13]], <float 8.000000e+00, float 9.000000e+00, float 1.000000e+01, float 1.100000e+01>
+; CHECK-NEXT:    [[TMP21:%.*]] = shufflevector <4 x float> [[TMP10]], <4 x float> [[TMP12]], <4 x i32> <i32 poison, i32 poison, i32 2, i32 4>
+; CHECK-NEXT:    [[TMP13:%.*]] = shufflevector <2 x float> [[TMP4]], <2 x float> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
+; CHECK-NEXT:    [[TMP22:%.*]] = shufflevector <4 x float> [[TMP21]], <4 x float> [[TMP13]], <4 x i32> <i32 4, i32 5, i32 2, i32 3>
+; CHECK-NEXT:    [[TMP14:%.*]] = fmul <4 x float> [[TMP22]], <float 7.000000e+00, float 8.000000e+00, float 9.000000e+00, float 1.000000e+01>
 ; CHECK-NEXT:    [[TMP15]] = fadd <4 x float> [[TMP3]], [[TMP14]]
+; CHECK-NEXT:    [[MUL25:%.*]] = fmul float [[TMP8]], 1.100000e+01
+; CHECK-NEXT:    [[ADD6]] = fadd float [[R_052]], [[MUL25]]
 ; CHECK-NEXT:    [[TMP16:%.*]] = trunc i64 [[INDVARS_IV_NEXT]] to i32
 ; CHECK-NEXT:    [[CMP:%.*]] = icmp slt i32 [[TMP16]], 121
 ; CHECK-NEXT:    br i1 [[CMP]], label [[FOR_BODY]], label [[FOR_END:%.*]]
 ; CHECK:       for.end:
 ; CHECK-NEXT:    [[TMP17:%.*]] = extractelement <4 x float> [[TMP15]], i32 0
-; CHECK-NEXT:    [[ADD28:%.*]] = fadd float [[ADD6]], [[TMP17]]
 ; CHECK-NEXT:    [[TMP18:%.*]] = extractelement <4 x float> [[TMP15]], i32 1
-; CHECK-NEXT:    [[ADD29:%.*]] = fadd float [[ADD28]], [[TMP18]]
+; CHECK-NEXT:    [[ADD29:%.*]] = fadd float [[TMP17]], [[TMP18]]
 ; CHECK-NEXT:    [[TMP19:%.*]] = extractelement <4 x float> [[TMP15]], i32 2
 ; CHECK-NEXT:    [[ADD30:%.*]] = fadd float [[ADD29]], [[TMP19]]
 ; CHECK-NEXT:    [[TMP20:%.*]] = extractelement <4 x float> [[TMP15]], i32 3
 ; CHECK-NEXT:    [[ADD31:%.*]] = fadd float [[ADD30]], [[TMP20]]
-; CHECK-NEXT:    ret float [[ADD31]]
+; CHECK-NEXT:    [[ADD32:%.*]] = fadd float [[ADD31]], [[ADD6]]
+; CHECK-NEXT:    ret float [[ADD32]]
 ;
 entry:
   %0 = load float, ptr %A, align 4
@@ -237,18 +239,18 @@ define float @sort_phi_type(ptr nocapture readonly %A) {
 ; CHECK:       for.body:
 ; CHECK-NEXT:    [[INDVARS_IV:%.*]] = phi i64 [ 0, [[ENTRY:%.*]] ], [ [[INDVARS_IV_NEXT:%.*]], [[FOR_BODY]] ]
 ; CHECK-NEXT:    [[TMP0:%.*]] = phi <4 x float> [ splat (float 1.000000e+01), [[ENTRY]] ], [ [[TMP2:%.*]], [[FOR_BODY]] ]
-; CHECK-NEXT:    [[TMP1:%.*]] = shufflevector <4 x float> [[TMP0]], <4 x float> poison, <4 x i32> <i32 0, i32 1, i32 3, i32 2>
-; CHECK-NEXT:    [[TMP2]] = fmul <4 x float> [[TMP1]], <float 8.000000e+00, float 9.000000e+00, float 1.000000e+02, float 1.110000e+02>
+; CHECK-NEXT:    [[TMP1:%.*]] = fmul <4 x float> [[TMP0]], <float 8.000000e+00, float 9.000000e+00, float 1.000000e+02, float 1.110000e+02>
 ; CHECK-NEXT:    [[INDVARS_IV_NEXT]] = add nsw i64 [[INDVARS_IV]], 4
 ; CHECK-NEXT:    [[CMP:%.*]] = icmp slt i64 [[INDVARS_IV_NEXT]], 128
+; CHECK-NEXT:    [[TMP2]] = shufflevector <4 x float> [[TMP1]], <4 x float> poison, <4 x i32> <i32 0, i32 1, i32 3, i32 2>
 ; CHECK-NEXT:    br i1 [[CMP]], label [[FOR_BODY]], label [[FOR_END:%.*]]
 ; CHECK:       for.end:
-; CHECK-NEXT:    [[TMP3:%.*]] = extractelement <4 x float> [[TMP2]], i32 0
-; CHECK-NEXT:    [[TMP4:%.*]] = extractelement <4 x float> [[TMP2]], i32 1
+; CHECK-NEXT:    [[TMP3:%.*]] = extractelement <4 x float> [[TMP1]], i32 0
+; CHECK-NEXT:    [[TMP4:%.*]] = extractelement <4 x float> [[TMP1]], i32 1
 ; CHECK-NEXT:    [[ADD29:%.*]] = fadd float [[TMP3]], [[TMP4]]
-; CHECK-NEXT:    [[TMP5:%.*]] = extractelement <4 x float> [[TMP2]], i32 2
+; CHECK-NEXT:    [[TMP5:%.*]] = extractelement <4 x float> [[TMP1]], i32 2
 ; CHECK-NEXT:    [[ADD30:%.*]] = fadd float [[ADD29]], [[TMP5]]
-; CHECK-NEXT:    [[TMP6:%.*]] = extractelement <4 x float> [[TMP2]], i32 3
+; CHECK-NEXT:    [[TMP6:%.*]] = extractelement <4 x float> [[TMP1]], i32 3
 ; CHECK-NEXT:    [[ADD31:%.*]] = fadd float [[ADD30]], [[TMP6]]
 ; CHECK-NEXT:    ret float [[ADD31]]
 ;
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/scatter-vectorize-reorder-non-empty.ll b/llvm/test/Transforms/SLPVectorizer/X86/scatter-vectorize-reorder-non-empty.ll
index 94172cffb0295..80bd8ae07e2e2 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/scatter-vectorize-reorder-non-empty.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/scatter-vectorize-reorder-non-empty.ll
@@ -7,13 +7,16 @@ define double @test01() {
 ; CHECK-NEXT:    [[TMP1:%.*]] = load <2 x i32>, ptr null, align 8
 ; CHECK-NEXT:    [[TMP2:%.*]] = getelementptr double, <2 x ptr> zeroinitializer, <2 x i32> [[TMP1]]
 ; CHECK-NEXT:    [[TMP3:%.*]] = call <2 x double> @llvm.masked.gather.v2f64.v2p0(<2 x ptr> align 8 [[TMP2]], <2 x i1> splat (i1 true), <2 x double> poison)
-; CHECK-NEXT:    [[TMP4:%.*]] = shufflevector <2 x double> [[TMP3]], <2 x double> <double 0.000000e+00, double poison>, <2 x i32> <i32 2, i32 0>
-; CHECK-NEXT:    [[TMP5:%.*]] = fadd <2 x double> [[TMP4]], [[TMP4]]
-; CHECK-NEXT:    [[TMP6:%.*]] = fadd <2 x double> [[TMP3]], [[TMP5]]
-; CHECK-NEXT:    [[TMP7:%.*]] = extractelement <2 x double> [[TMP6]], i32 0
-; CHECK-NEXT:    [[TMP8:%.*]] = extractelement <2 x double> [[TMP6]], i32 1
-; CHECK-NEXT:    [[TMP9:%.*]] = fadd double [[TMP7]], [[TMP8]]
-; CHECK-NEXT:    ret double [[TMP9]]
+; CHECK-NEXT:    [[TMP4:%.*]] = shufflevector <2 x double> [[TMP3]], <2 x double> poison, <4 x i32> <i32 0, i32 0, i32 0, i32 1>
+; CHECK-NEXT:    [[TMP5:%.*]] = extractelement <4 x double> [[TMP4]], i32 0
+; CHECK-NEXT:    [[TMP6:%.*]] = shufflevector <2 x double> [[TMP3]], <2 x double> poison, <2 x i32> zeroinitializer
+; CHECK-NEXT:    [[TMP7:%.*]] = fadd double [[TMP5]], [[TMP5]]
+; CHECK-NEXT:    [[TMP8:%.*]] = extractelement <4 x double> [[TMP4]], i32 3
+; CHECK-NEXT:    [[TMP9:%.*]] = fadd double [[TMP8]], [[TMP7]]
+; CHECK-NEXT:    [[TMP10:%.*]] = fadd double 0.000000e+00, 0.000000e+00
+; CHECK-NEXT:    [[TMP11:%.*]] = fadd double [[TMP5]], [[TMP10]]
+; CHECK-NEXT:    [[TMP12:%.*]] = fadd double [[TMP11]], [[TMP9]]
+; CHECK-NEXT:    ret double [[TMP12]]
 ;
   %1 = load i32, ptr null, align 8
   %2 = load i32, ptr getelementptr inbounds (i32, ptr null, i32 1), align 4

>From c3ddc3f74593c6cd31043c976583f7557edba661 Mon Sep 17 00:00:00 2001
From: Alexey Bataev <a.bataev at outlook.com>
Date: Fri, 20 Feb 2026 17:51:09 -0800
Subject: [PATCH 2/2] Fix formatting

Created using spr 1.3.7
---
 llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp | 8 ++++----
 1 file changed, 4 insertions(+), 4 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index b1df83e021cb7..044d0d4579c04 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -25349,7 +25349,7 @@ class HorizontalReduction {
   }
 
   /// Checks if instruction is associative and can be vectorized.
-  enum class ReductionKind {Unordered, Ordered, None};
+  enum class ReductionKind { Unordered, Ordered, None };
   ReductionKind RK = ReductionKind::None;
   static ReductionKind isVectorizable(RecurKind Kind, Instruction *I,
                                       bool TwoElementReduction = false) {
@@ -25733,9 +25733,9 @@ class HorizontalReduction {
         // Also, do not try to reduce const values, if the operation is not
         // foldable.
         bool IsReducedVal = !EdgeInst || Level > RecursionMaxDepth ||
-            getRdxKind(EdgeInst) != RdxKind ||
-            IsCmpSelMinMax != isCmpSelMinMax(EdgeInst) ||
-            !hasRequiredNumberOfUses(IsCmpSelMinMax, EdgeInst);
+                            getRdxKind(EdgeInst) != RdxKind ||
+                            IsCmpSelMinMax != isCmpSelMinMax(EdgeInst) ||
+                            !hasRequiredNumberOfUses(IsCmpSelMinMax, EdgeInst);
         ReductionKind CurrentRK = IsReducedVal
                                       ? ReductionKind::None
                                       : isVectorizable(RdxKind, EdgeInst);



More information about the llvm-commits mailing list