[llvm] [SLP]Initial support for ordreded reductions (PR #182644)
Alexey Bataev via llvm-commits
llvm-commits at lists.llvm.org
Fri Feb 20 17:51:19 PST 2026
https://github.com/alexey-bataev updated https://github.com/llvm/llvm-project/pull/182644
>From c9d5e477b4e05d483db4921c5c566aeeb3a82897 Mon Sep 17 00:00:00 2001
From: Alexey Bataev <a.bataev at outlook.com>
Date: Fri, 20 Feb 2026 17:46:54 -0800
Subject: [PATCH 1/2] =?UTF-8?q?[=F0=9D=98=80=F0=9D=97=BD=F0=9D=97=BF]=20in?=
=?UTF-8?q?itial=20version?=
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
Created using spr 1.3.7
---
.../Transforms/Vectorize/SLPVectorizer.cpp | 83 ++++--
.../SLPVectorizer/AArch64/tsc-s352.ll | 20 +-
.../SLPVectorizer/X86/dot-product.ll | 92 ++++--
.../Transforms/SLPVectorizer/X86/fmaxnum.ll | 282 +++++++++++++++---
.../Transforms/SLPVectorizer/X86/fminnum.ll | 262 +++++++++++++---
.../SLPVectorizer/X86/horizontal-list.ll | 12 +-
llvm/test/Transforms/SLPVectorizer/X86/phi.ll | 50 ++--
.../scatter-vectorize-reorder-non-empty.ll | 17 +-
8 files changed, 633 insertions(+), 185 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 4caa1707f0f27..b1df83e021cb7 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -25349,31 +25349,37 @@ class HorizontalReduction {
}
/// Checks if instruction is associative and can be vectorized.
- static bool isVectorizable(RecurKind Kind, Instruction *I,
- bool TwoElementReduction = false) {
+ enum class ReductionKind {Unordered, Ordered, None};
+ ReductionKind RK = ReductionKind::None;
+ static ReductionKind isVectorizable(RecurKind Kind, Instruction *I,
+ bool TwoElementReduction = false) {
if (Kind == RecurKind::None)
- return false;
+ return ReductionKind::None;
// Integer ops that map to select instructions or intrinsics are fine.
if (RecurrenceDescriptor::isIntMinMaxRecurrenceKind(Kind) ||
isBoolLogicOp(I))
- return true;
+ return ReductionKind::Unordered;
// No need to check for associativity, if 2 reduced values.
if (TwoElementReduction)
- return true;
+ return ReductionKind::Unordered;
if (Kind == RecurKind::FMax || Kind == RecurKind::FMin) {
// FP min/max are associative except for NaN and -0.0. We do not
// have to rule out -0.0 here because the intrinsic semantics do not
// specify a fixed result for it.
- return I->getFastMathFlags().noNaNs();
+ return I->getFastMathFlags().noNaNs() ? ReductionKind::Unordered
+ : ReductionKind::Ordered;
}
if (Kind == RecurKind::FMaximum || Kind == RecurKind::FMinimum)
- return true;
+ return ReductionKind::Unordered;
+
+ if (I->isAssociative())
+ return ReductionKind::Unordered;
- return I->isAssociative();
+ return ::isCommutative(I) ? ReductionKind::Ordered : ReductionKind::None;
}
static Value *getRdxOperand(Instruction *I, unsigned Index) {
@@ -25675,13 +25681,10 @@ class HorizontalReduction {
// Analyze "regular" integer/FP types for reductions - no target-specific
// types or pointers.
assert(ReductionRoot && "Reduction root is not set!");
- if (!isVectorizable(RdxKind, cast<Instruction>(ReductionRoot),
- all_of(ReducedVals, [](ArrayRef<Value *> Ops) {
- return Ops.size() == 2;
- })))
- return false;
-
- return true;
+ return isVectorizable(RdxKind, cast<Instruction>(ReductionRoot),
+ all_of(ReducedVals, [](ArrayRef<Value *> Ops) {
+ return Ops.size() == 2;
+ })) != ReductionKind::None;
}
/// Try to find a reduction tree.
@@ -25689,7 +25692,8 @@ class HorizontalReduction {
ScalarEvolution &SE, const DataLayout &DL,
const TargetLibraryInfo &TLI) {
RdxKind = HorizontalReduction::getRdxKind(Root);
- if (!isVectorizable(RdxKind, Root))
+ RK = isVectorizable(RdxKind, Root);
+ if (RK == ReductionKind::None)
return false;
// Analyze "regular" integer/FP types for reductions - no target-specific
@@ -25728,16 +25732,21 @@ class HorizontalReduction {
// reduction opcode or has too many uses - possible reduced value.
// Also, do not try to reduce const values, if the operation is not
// foldable.
- if (!EdgeInst || Level > RecursionMaxDepth ||
+ bool IsReducedVal = !EdgeInst || Level > RecursionMaxDepth ||
getRdxKind(EdgeInst) != RdxKind ||
IsCmpSelMinMax != isCmpSelMinMax(EdgeInst) ||
- !hasRequiredNumberOfUses(IsCmpSelMinMax, EdgeInst) ||
- !isVectorizable(RdxKind, EdgeInst) ||
+ !hasRequiredNumberOfUses(IsCmpSelMinMax, EdgeInst);
+ ReductionKind CurrentRK = IsReducedVal
+ ? ReductionKind::None
+ : isVectorizable(RdxKind, EdgeInst);
+ if (CurrentRK == ReductionKind::None ||
(R.isAnalyzedReductionRoot(EdgeInst) &&
all_of(EdgeInst->operands(), IsaPred<Constant>))) {
PossibleReducedVals.push_back(EdgeVal);
continue;
}
+ if (CurrentRK == ReductionKind::Ordered)
+ RK = ReductionKind::Ordered;
ReductionOps.push_back(EdgeInst);
}
};
@@ -25943,6 +25952,10 @@ class HorizontalReduction {
for (Value *U : IgnoreList)
if (auto *FPMO = dyn_cast<FPMathOperator>(U))
RdxFMF &= FPMO->getFastMathFlags();
+ // For ordered reductions here we need to generate extractelement
+ // instructions, so clear IgnoreList.
+ if (RK == ReductionKind::Ordered)
+ IgnoreList.clear();
bool IsCmpSelMinMax = isCmpSelMinMax(cast<Instruction>(ReductionRoot));
// Need to track reduced vals, they may be changed during vectorization of
@@ -26054,6 +26067,8 @@ class HorizontalReduction {
// Emit code for constant values.
if (Candidates.size() > 1 && allConstant(Candidates)) {
+ if (RK == ReductionKind::Ordered)
+ continue;
Value *Res = Candidates.front();
Value *OrigV = TrackedToOrig.at(Candidates.front());
++VectorizedVals.try_emplace(OrigV).first->getSecond();
@@ -26075,9 +26090,9 @@ class HorizontalReduction {
// Check if we support repeated scalar values processing (optimization of
// original scalar identity operations on matched horizontal reductions).
- IsSupportedHorRdxIdentityOp = RdxKind != RecurKind::Mul &&
- RdxKind != RecurKind::FMul &&
- RdxKind != RecurKind::FMulAdd;
+ IsSupportedHorRdxIdentityOp =
+ RK == ReductionKind::Unordered && RdxKind != RecurKind::Mul &&
+ RdxKind != RecurKind::FMul && RdxKind != RecurKind::FMulAdd;
// Gather same values.
SmallMapVector<Value *, unsigned, 16> SameValuesCounter;
if (IsSupportedHorRdxIdentityOp)
@@ -26346,6 +26361,23 @@ class HorizontalReduction {
// Vectorize a tree.
Value *VectorizedRoot = V.vectorizeTree(
LocalExternallyUsedValues, InsertPt, VectorValuesAndScales);
+ if (RK == ReductionKind::Ordered) {
+ // No need to generate reduction here, eit extractelements instead in
+ // the tree vectorizer.
+ assert(VectorizedRoot && "Expected vectorized tree");
+ // Count vectorized reduced values to exclude them from final
+ // reduction.
+ for (Value *RdxVal : VL)
+ ++VectorizedVals.try_emplace(RdxVal).first->getSecond();
+ Pos += ReduxWidth;
+ Start = Pos;
+ ReduxWidth = NumReducedVals - Pos;
+ if (ReduxWidth > 1)
+ ReduxWidth = GetVectorFactor(NumReducedVals - Pos);
+ AnyVectorized = true;
+ VectorizedTree = ReductionRoot;
+ continue;
+ }
// Update TrackedToOrig mapping, since the tracked values might be
// updated.
for (Value *RdxVal : Candidates) {
@@ -26410,6 +26442,11 @@ class HorizontalReduction {
continue;
}
}
+ // Early exit for the ordered reductions.
+ // No need to do anything else here, so we can just exit.
+ if (RK == ReductionKind::Ordered)
+ return VectorizedTree;
+
if (!VectorValuesAndScales.empty())
VectorizedTree = GetNewVectorizedTree(
VectorizedTree,
@@ -27458,7 +27495,7 @@ bool SLPVectorizerPass::vectorizeHorReduction(
continue;
if (Value *VectorizedV = TryToReduce(Inst)) {
Res = true;
- if (auto *I = dyn_cast<Instruction>(VectorizedV)) {
+ if (auto *I = dyn_cast<Instruction>(VectorizedV); I && I != Inst) {
// Try to find another reduction.
Stack.emplace(I, Level);
continue;
diff --git a/llvm/test/Transforms/SLPVectorizer/AArch64/tsc-s352.ll b/llvm/test/Transforms/SLPVectorizer/AArch64/tsc-s352.ll
index 70b7adfd6456e..c9a2219b12a8a 100644
--- a/llvm/test/Transforms/SLPVectorizer/AArch64/tsc-s352.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AArch64/tsc-s352.ll
@@ -31,22 +31,16 @@ define i32 @s352() {
; CHECK-NEXT: [[DOT_115:%.*]] = phi float [ 0.000000e+00, [[PREHEADER]] ], [ [[ADD39:%.*]], [[FOR_BODY]] ]
; CHECK-NEXT: [[ARRAYIDX:%.*]] = getelementptr inbounds [[STRUCT_GLOBALDATA:%.*]], ptr @global_data, i64 0, i32 0, i64 [[INDVARS_IV]]
; CHECK-NEXT: [[ARRAYIDX6:%.*]] = getelementptr inbounds [[STRUCT_GLOBALDATA]], ptr @global_data, i64 0, i32 3, i64 [[INDVARS_IV]]
-; CHECK-NEXT: [[TMP1:%.*]] = load <2 x float>, ptr [[ARRAYIDX]], align 4
-; CHECK-NEXT: [[TMP3:%.*]] = load <2 x float>, ptr [[ARRAYIDX6]], align 4
-; CHECK-NEXT: [[TMP4:%.*]] = fmul <2 x float> [[TMP1]], [[TMP3]]
-; CHECK-NEXT: [[TMP5:%.*]] = extractelement <2 x float> [[TMP4]], i32 0
+; CHECK-NEXT: [[TMP0:%.*]] = load <4 x float>, ptr [[ARRAYIDX]], align 4
+; CHECK-NEXT: [[TMP1:%.*]] = load <4 x float>, ptr [[ARRAYIDX6]], align 4
+; CHECK-NEXT: [[TMP2:%.*]] = fmul <4 x float> [[TMP0]], [[TMP1]]
+; CHECK-NEXT: [[TMP5:%.*]] = extractelement <4 x float> [[TMP2]], i32 0
; CHECK-NEXT: [[ADD:%.*]] = fadd float [[DOT_115]], [[TMP5]]
-; CHECK-NEXT: [[TMP6:%.*]] = extractelement <2 x float> [[TMP4]], i32 1
+; CHECK-NEXT: [[TMP6:%.*]] = extractelement <4 x float> [[TMP2]], i32 1
; CHECK-NEXT: [[ADD15:%.*]] = fadd float [[ADD]], [[TMP6]]
-; CHECK-NEXT: [[TMP7:%.*]] = add nuw nsw i64 [[INDVARS_IV]], 2
-; CHECK-NEXT: [[ARRAYIDX18:%.*]] = getelementptr inbounds [[STRUCT_GLOBALDATA]], ptr @global_data, i64 0, i32 0, i64 [[TMP7]]
-; CHECK-NEXT: [[ARRAYIDX21:%.*]] = getelementptr inbounds [[STRUCT_GLOBALDATA]], ptr @global_data, i64 0, i32 3, i64 [[TMP7]]
-; CHECK-NEXT: [[TMP9:%.*]] = load <2 x float>, ptr [[ARRAYIDX18]], align 4
-; CHECK-NEXT: [[TMP11:%.*]] = load <2 x float>, ptr [[ARRAYIDX21]], align 4
-; CHECK-NEXT: [[TMP12:%.*]] = fmul <2 x float> [[TMP9]], [[TMP11]]
-; CHECK-NEXT: [[TMP13:%.*]] = extractelement <2 x float> [[TMP12]], i32 0
+; CHECK-NEXT: [[TMP13:%.*]] = extractelement <4 x float> [[TMP2]], i32 2
; CHECK-NEXT: [[ADD23:%.*]] = fadd float [[ADD15]], [[TMP13]]
-; CHECK-NEXT: [[TMP14:%.*]] = extractelement <2 x float> [[TMP12]], i32 1
+; CHECK-NEXT: [[TMP14:%.*]] = extractelement <4 x float> [[TMP2]], i32 3
; CHECK-NEXT: [[ADD31:%.*]] = fadd float [[ADD23]], [[TMP14]]
; CHECK-NEXT: [[TMP15:%.*]] = add nuw nsw i64 [[INDVARS_IV]], 4
; CHECK-NEXT: [[ARRAYIDX34:%.*]] = getelementptr inbounds [[STRUCT_GLOBALDATA]], ptr @global_data, i64 0, i32 0, i64 [[TMP15]]
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/dot-product.ll b/llvm/test/Transforms/SLPVectorizer/X86/dot-product.ll
index a333d162297bc..1b9dbaca0a34d 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/dot-product.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/dot-product.ll
@@ -9,23 +9,62 @@
;
define double @dot4f64(ptr dereferenceable(32) %ptrx, ptr dereferenceable(32) %ptry) {
-; CHECK-LABEL: @dot4f64(
-; CHECK-NEXT: [[PTRX2:%.*]] = getelementptr inbounds double, ptr [[PTRX:%.*]], i64 2
-; CHECK-NEXT: [[PTRY2:%.*]] = getelementptr inbounds double, ptr [[PTRY:%.*]], i64 2
-; CHECK-NEXT: [[TMP1:%.*]] = load <2 x double>, ptr [[PTRX]], align 4
-; CHECK-NEXT: [[TMP2:%.*]] = load <2 x double>, ptr [[PTRY]], align 4
-; CHECK-NEXT: [[TMP3:%.*]] = fmul <2 x double> [[TMP1]], [[TMP2]]
-; CHECK-NEXT: [[TMP4:%.*]] = load <2 x double>, ptr [[PTRX2]], align 4
-; CHECK-NEXT: [[TMP5:%.*]] = load <2 x double>, ptr [[PTRY2]], align 4
-; CHECK-NEXT: [[TMP6:%.*]] = fmul <2 x double> [[TMP4]], [[TMP5]]
-; CHECK-NEXT: [[TMP7:%.*]] = extractelement <2 x double> [[TMP3]], i32 0
-; CHECK-NEXT: [[TMP8:%.*]] = extractelement <2 x double> [[TMP3]], i32 1
-; CHECK-NEXT: [[DOT01:%.*]] = fadd double [[TMP7]], [[TMP8]]
-; CHECK-NEXT: [[TMP9:%.*]] = extractelement <2 x double> [[TMP6]], i32 0
-; CHECK-NEXT: [[DOT012:%.*]] = fadd double [[DOT01]], [[TMP9]]
-; CHECK-NEXT: [[TMP10:%.*]] = extractelement <2 x double> [[TMP6]], i32 1
-; CHECK-NEXT: [[DOT0123:%.*]] = fadd double [[DOT012]], [[TMP10]]
-; CHECK-NEXT: ret double [[DOT0123]]
+; SSE2-LABEL: @dot4f64(
+; SSE2-NEXT: [[TMP1:%.*]] = load <4 x double>, ptr [[PTRX:%.*]], align 4
+; SSE2-NEXT: [[TMP2:%.*]] = load <4 x double>, ptr [[PTRY:%.*]], align 4
+; SSE2-NEXT: [[TMP3:%.*]] = fmul <4 x double> [[TMP1]], [[TMP2]]
+; SSE2-NEXT: [[TMP4:%.*]] = extractelement <4 x double> [[TMP3]], i32 0
+; SSE2-NEXT: [[TMP5:%.*]] = extractelement <4 x double> [[TMP3]], i32 1
+; SSE2-NEXT: [[DOT01:%.*]] = fadd double [[TMP4]], [[TMP5]]
+; SSE2-NEXT: [[TMP6:%.*]] = extractelement <4 x double> [[TMP3]], i32 2
+; SSE2-NEXT: [[DOT012:%.*]] = fadd double [[DOT01]], [[TMP6]]
+; SSE2-NEXT: [[TMP7:%.*]] = extractelement <4 x double> [[TMP3]], i32 3
+; SSE2-NEXT: [[DOT0123:%.*]] = fadd double [[DOT012]], [[TMP7]]
+; SSE2-NEXT: ret double [[DOT0123]]
+;
+; SSE4-LABEL: @dot4f64(
+; SSE4-NEXT: [[TMP1:%.*]] = load <4 x double>, ptr [[PTRX:%.*]], align 4
+; SSE4-NEXT: [[TMP2:%.*]] = load <4 x double>, ptr [[PTRY:%.*]], align 4
+; SSE4-NEXT: [[TMP3:%.*]] = fmul <4 x double> [[TMP1]], [[TMP2]]
+; SSE4-NEXT: [[TMP4:%.*]] = extractelement <4 x double> [[TMP3]], i32 0
+; SSE4-NEXT: [[TMP5:%.*]] = extractelement <4 x double> [[TMP3]], i32 1
+; SSE4-NEXT: [[DOT01:%.*]] = fadd double [[TMP4]], [[TMP5]]
+; SSE4-NEXT: [[TMP6:%.*]] = extractelement <4 x double> [[TMP3]], i32 2
+; SSE4-NEXT: [[DOT012:%.*]] = fadd double [[DOT01]], [[TMP6]]
+; SSE4-NEXT: [[TMP7:%.*]] = extractelement <4 x double> [[TMP3]], i32 3
+; SSE4-NEXT: [[DOT0123:%.*]] = fadd double [[DOT012]], [[TMP7]]
+; SSE4-NEXT: ret double [[DOT0123]]
+;
+; AVX-LABEL: @dot4f64(
+; AVX-NEXT: [[PTRX2:%.*]] = getelementptr inbounds double, ptr [[PTRX:%.*]], i64 2
+; AVX-NEXT: [[PTRY2:%.*]] = getelementptr inbounds double, ptr [[PTRY:%.*]], i64 2
+; AVX-NEXT: [[TMP1:%.*]] = load <2 x double>, ptr [[PTRX]], align 4
+; AVX-NEXT: [[TMP2:%.*]] = load <2 x double>, ptr [[PTRY]], align 4
+; AVX-NEXT: [[TMP3:%.*]] = fmul <2 x double> [[TMP1]], [[TMP2]]
+; AVX-NEXT: [[TMP4:%.*]] = load <2 x double>, ptr [[PTRX2]], align 4
+; AVX-NEXT: [[TMP5:%.*]] = load <2 x double>, ptr [[PTRY2]], align 4
+; AVX-NEXT: [[TMP6:%.*]] = fmul <2 x double> [[TMP4]], [[TMP5]]
+; AVX-NEXT: [[TMP7:%.*]] = extractelement <2 x double> [[TMP3]], i32 0
+; AVX-NEXT: [[TMP8:%.*]] = extractelement <2 x double> [[TMP3]], i32 1
+; AVX-NEXT: [[DOT01:%.*]] = fadd double [[TMP7]], [[TMP8]]
+; AVX-NEXT: [[TMP9:%.*]] = extractelement <2 x double> [[TMP6]], i32 0
+; AVX-NEXT: [[DOT012:%.*]] = fadd double [[DOT01]], [[TMP9]]
+; AVX-NEXT: [[TMP10:%.*]] = extractelement <2 x double> [[TMP6]], i32 1
+; AVX-NEXT: [[DOT0123:%.*]] = fadd double [[DOT012]], [[TMP10]]
+; AVX-NEXT: ret double [[DOT0123]]
+;
+; AVX2-LABEL: @dot4f64(
+; AVX2-NEXT: [[TMP1:%.*]] = load <4 x double>, ptr [[PTRX:%.*]], align 4
+; AVX2-NEXT: [[TMP2:%.*]] = load <4 x double>, ptr [[PTRY:%.*]], align 4
+; AVX2-NEXT: [[TMP3:%.*]] = fmul <4 x double> [[TMP1]], [[TMP2]]
+; AVX2-NEXT: [[TMP4:%.*]] = extractelement <4 x double> [[TMP3]], i32 0
+; AVX2-NEXT: [[TMP5:%.*]] = extractelement <4 x double> [[TMP3]], i32 1
+; AVX2-NEXT: [[DOT01:%.*]] = fadd double [[TMP4]], [[TMP5]]
+; AVX2-NEXT: [[TMP6:%.*]] = extractelement <4 x double> [[TMP3]], i32 2
+; AVX2-NEXT: [[DOT012:%.*]] = fadd double [[DOT01]], [[TMP6]]
+; AVX2-NEXT: [[TMP7:%.*]] = extractelement <4 x double> [[TMP3]], i32 3
+; AVX2-NEXT: [[DOT0123:%.*]] = fadd double [[DOT012]], [[TMP7]]
+; AVX2-NEXT: ret double [[DOT0123]]
;
%ptrx1 = getelementptr inbounds double, ptr %ptrx, i64 1
%ptry1 = getelementptr inbounds double, ptr %ptry, i64 1
@@ -53,20 +92,15 @@ define double @dot4f64(ptr dereferenceable(32) %ptrx, ptr dereferenceable(32) %p
define float @dot4f32(ptr dereferenceable(16) %ptrx, ptr dereferenceable(16) %ptry) {
; CHECK-LABEL: @dot4f32(
-; CHECK-NEXT: [[PTRX2:%.*]] = getelementptr inbounds float, ptr [[PTRX:%.*]], i64 2
-; CHECK-NEXT: [[PTRY2:%.*]] = getelementptr inbounds float, ptr [[PTRY:%.*]], i64 2
-; CHECK-NEXT: [[TMP1:%.*]] = load <2 x float>, ptr [[PTRX]], align 4
-; CHECK-NEXT: [[TMP2:%.*]] = load <2 x float>, ptr [[PTRY]], align 4
-; CHECK-NEXT: [[TMP3:%.*]] = fmul <2 x float> [[TMP1]], [[TMP2]]
-; CHECK-NEXT: [[TMP4:%.*]] = load <2 x float>, ptr [[PTRX2]], align 4
-; CHECK-NEXT: [[TMP5:%.*]] = load <2 x float>, ptr [[PTRY2]], align 4
-; CHECK-NEXT: [[TMP6:%.*]] = fmul <2 x float> [[TMP4]], [[TMP5]]
-; CHECK-NEXT: [[TMP7:%.*]] = extractelement <2 x float> [[TMP3]], i32 0
-; CHECK-NEXT: [[TMP8:%.*]] = extractelement <2 x float> [[TMP3]], i32 1
+; CHECK-NEXT: [[TMP1:%.*]] = load <4 x float>, ptr [[PTRX:%.*]], align 4
+; CHECK-NEXT: [[TMP2:%.*]] = load <4 x float>, ptr [[PTRY:%.*]], align 4
+; CHECK-NEXT: [[TMP3:%.*]] = fmul <4 x float> [[TMP1]], [[TMP2]]
+; CHECK-NEXT: [[TMP7:%.*]] = extractelement <4 x float> [[TMP3]], i32 0
+; CHECK-NEXT: [[TMP8:%.*]] = extractelement <4 x float> [[TMP3]], i32 1
; CHECK-NEXT: [[DOT01:%.*]] = fadd float [[TMP7]], [[TMP8]]
-; CHECK-NEXT: [[TMP9:%.*]] = extractelement <2 x float> [[TMP6]], i32 0
+; CHECK-NEXT: [[TMP9:%.*]] = extractelement <4 x float> [[TMP3]], i32 2
; CHECK-NEXT: [[DOT012:%.*]] = fadd float [[DOT01]], [[TMP9]]
-; CHECK-NEXT: [[TMP10:%.*]] = extractelement <2 x float> [[TMP6]], i32 1
+; CHECK-NEXT: [[TMP10:%.*]] = extractelement <4 x float> [[TMP3]], i32 3
; CHECK-NEXT: [[DOT0123:%.*]] = fadd float [[DOT012]], [[TMP10]]
; CHECK-NEXT: ret float [[DOT0123]]
;
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/fmaxnum.ll b/llvm/test/Transforms/SLPVectorizer/X86/fmaxnum.ll
index a42567c5e2e46..822f1051d45d6 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/fmaxnum.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/fmaxnum.ll
@@ -1,8 +1,8 @@
; NOTE: Assertions have been autogenerated by utils/update_test_checks.py
; RUN: opt < %s -mtriple=x86_64-unknown -passes=slp-vectorizer -S | FileCheck %s --check-prefixes=CHECK,SSE
-; RUN: opt < %s -mtriple=x86_64-unknown -mcpu=corei7-avx -passes=slp-vectorizer -S | FileCheck %s --check-prefixes=CHECK,AVX,AVX256
-; RUN: opt < %s -mtriple=x86_64-unknown -mcpu=bdver1 -passes=slp-vectorizer -S | FileCheck %s --check-prefixes=CHECK,AVX,AVX256
-; RUN: opt < %s -mtriple=x86_64-unknown -mcpu=core-avx2 -passes=slp-vectorizer -S | FileCheck %s --check-prefixes=CHECK,AVX,AVX256
+; RUN: opt < %s -mtriple=x86_64-unknown -mcpu=corei7-avx -passes=slp-vectorizer -S | FileCheck %s --check-prefixes=CHECK,AVX,COREI7
+; RUN: opt < %s -mtriple=x86_64-unknown -mcpu=bdver1 -passes=slp-vectorizer -S | FileCheck %s --check-prefixes=CHECK,AVX,BDVER1
+; RUN: opt < %s -mtriple=x86_64-unknown -mcpu=core-avx2 -passes=slp-vectorizer -S | FileCheck %s --check-prefixes=CHECK,AVX,AVX2
; RUN: opt < %s -mtriple=x86_64-unknown -mcpu=skylake-avx512 -mattr=-prefer-256-bit -passes=slp-vectorizer -S | FileCheck %s --check-prefixes=CHECK,AVX,AVX512
; RUN: opt < %s -mtriple=x86_64-unknown -mcpu=skylake-avx512 -mattr=+prefer-256-bit -passes=slp-vectorizer -S | FileCheck %s --check-prefixes=CHECK,AVX,AVX256
@@ -99,6 +99,46 @@ define void @fmaxnum_8f64() #0 {
; SSE-NEXT: store <2 x double> [[TMP12]], ptr getelementptr inbounds ([8 x double], ptr @dst64, i32 0, i64 6), align 4
; SSE-NEXT: ret void
;
+; COREI7-LABEL: @fmaxnum_8f64(
+; COREI7-NEXT: [[TMP1:%.*]] = load <4 x double>, ptr @srcA64, align 4
+; COREI7-NEXT: [[TMP2:%.*]] = load <4 x double>, ptr @srcB64, align 4
+; COREI7-NEXT: [[TMP3:%.*]] = call <4 x double> @llvm.maxnum.v4f64(<4 x double> [[TMP1]], <4 x double> [[TMP2]])
+; COREI7-NEXT: store <4 x double> [[TMP3]], ptr @dst64, align 4
+; COREI7-NEXT: [[TMP4:%.*]] = load <4 x double>, ptr getelementptr inbounds ([8 x double], ptr @srcA64, i32 0, i64 4), align 4
+; COREI7-NEXT: [[TMP5:%.*]] = load <4 x double>, ptr getelementptr inbounds ([8 x double], ptr @srcB64, i32 0, i64 4), align 4
+; COREI7-NEXT: [[TMP6:%.*]] = call <4 x double> @llvm.maxnum.v4f64(<4 x double> [[TMP4]], <4 x double> [[TMP5]])
+; COREI7-NEXT: store <4 x double> [[TMP6]], ptr getelementptr inbounds ([8 x double], ptr @dst64, i32 0, i64 4), align 4
+; COREI7-NEXT: ret void
+;
+; BDVER1-LABEL: @fmaxnum_8f64(
+; BDVER1-NEXT: [[TMP1:%.*]] = load <4 x double>, ptr @srcA64, align 4
+; BDVER1-NEXT: [[TMP2:%.*]] = load <4 x double>, ptr @srcB64, align 4
+; BDVER1-NEXT: [[TMP3:%.*]] = call <4 x double> @llvm.maxnum.v4f64(<4 x double> [[TMP1]], <4 x double> [[TMP2]])
+; BDVER1-NEXT: store <4 x double> [[TMP3]], ptr @dst64, align 4
+; BDVER1-NEXT: [[TMP4:%.*]] = load <4 x double>, ptr getelementptr inbounds ([8 x double], ptr @srcA64, i32 0, i64 4), align 4
+; BDVER1-NEXT: [[TMP5:%.*]] = load <4 x double>, ptr getelementptr inbounds ([8 x double], ptr @srcB64, i32 0, i64 4), align 4
+; BDVER1-NEXT: [[TMP6:%.*]] = call <4 x double> @llvm.maxnum.v4f64(<4 x double> [[TMP4]], <4 x double> [[TMP5]])
+; BDVER1-NEXT: store <4 x double> [[TMP6]], ptr getelementptr inbounds ([8 x double], ptr @dst64, i32 0, i64 4), align 4
+; BDVER1-NEXT: ret void
+;
+; AVX2-LABEL: @fmaxnum_8f64(
+; AVX2-NEXT: [[TMP1:%.*]] = load <4 x double>, ptr @srcA64, align 4
+; AVX2-NEXT: [[TMP2:%.*]] = load <4 x double>, ptr @srcB64, align 4
+; AVX2-NEXT: [[TMP3:%.*]] = call <4 x double> @llvm.maxnum.v4f64(<4 x double> [[TMP1]], <4 x double> [[TMP2]])
+; AVX2-NEXT: store <4 x double> [[TMP3]], ptr @dst64, align 4
+; AVX2-NEXT: [[TMP4:%.*]] = load <4 x double>, ptr getelementptr inbounds ([8 x double], ptr @srcA64, i32 0, i64 4), align 4
+; AVX2-NEXT: [[TMP5:%.*]] = load <4 x double>, ptr getelementptr inbounds ([8 x double], ptr @srcB64, i32 0, i64 4), align 4
+; AVX2-NEXT: [[TMP6:%.*]] = call <4 x double> @llvm.maxnum.v4f64(<4 x double> [[TMP4]], <4 x double> [[TMP5]])
+; AVX2-NEXT: store <4 x double> [[TMP6]], ptr getelementptr inbounds ([8 x double], ptr @dst64, i32 0, i64 4), align 4
+; AVX2-NEXT: ret void
+;
+; AVX512-LABEL: @fmaxnum_8f64(
+; AVX512-NEXT: [[TMP1:%.*]] = load <8 x double>, ptr @srcA64, align 4
+; AVX512-NEXT: [[TMP2:%.*]] = load <8 x double>, ptr @srcB64, align 4
+; AVX512-NEXT: [[TMP3:%.*]] = call <8 x double> @llvm.maxnum.v8f64(<8 x double> [[TMP1]], <8 x double> [[TMP2]])
+; AVX512-NEXT: store <8 x double> [[TMP3]], ptr @dst64, align 4
+; AVX512-NEXT: ret void
+;
; AVX256-LABEL: @fmaxnum_8f64(
; AVX256-NEXT: [[TMP1:%.*]] = load <4 x double>, ptr @srcA64, align 4
; AVX256-NEXT: [[TMP2:%.*]] = load <4 x double>, ptr @srcB64, align 4
@@ -109,13 +149,6 @@ define void @fmaxnum_8f64() #0 {
; AVX256-NEXT: [[TMP6:%.*]] = call <4 x double> @llvm.maxnum.v4f64(<4 x double> [[TMP4]], <4 x double> [[TMP5]])
; AVX256-NEXT: store <4 x double> [[TMP6]], ptr getelementptr inbounds ([8 x double], ptr @dst64, i32 0, i64 4), align 4
; AVX256-NEXT: ret void
-;
-; AVX512-LABEL: @fmaxnum_8f64(
-; AVX512-NEXT: [[TMP1:%.*]] = load <8 x double>, ptr @srcA64, align 4
-; AVX512-NEXT: [[TMP2:%.*]] = load <8 x double>, ptr @srcB64, align 4
-; AVX512-NEXT: [[TMP3:%.*]] = call <8 x double> @llvm.maxnum.v8f64(<8 x double> [[TMP1]], <8 x double> [[TMP2]])
-; AVX512-NEXT: store <8 x double> [[TMP3]], ptr @dst64, align 4
-; AVX512-NEXT: ret void
;
%a0 = load double, ptr @srcA64, align 4
%a1 = load double, ptr getelementptr inbounds ([8 x double], ptr @srcA64, i32 0, i64 1), align 4
@@ -253,6 +286,46 @@ define void @fmaxnum_16f32() #0 {
; SSE-NEXT: store <4 x float> [[TMP12]], ptr getelementptr inbounds ([16 x float], ptr @dst32, i32 0, i64 12), align 4
; SSE-NEXT: ret void
;
+; COREI7-LABEL: @fmaxnum_16f32(
+; COREI7-NEXT: [[TMP1:%.*]] = load <8 x float>, ptr @srcA32, align 4
+; COREI7-NEXT: [[TMP2:%.*]] = load <8 x float>, ptr @srcB32, align 4
+; COREI7-NEXT: [[TMP3:%.*]] = call <8 x float> @llvm.maxnum.v8f32(<8 x float> [[TMP1]], <8 x float> [[TMP2]])
+; COREI7-NEXT: store <8 x float> [[TMP3]], ptr @dst32, align 4
+; COREI7-NEXT: [[TMP4:%.*]] = load <8 x float>, ptr getelementptr inbounds ([16 x float], ptr @srcA32, i32 0, i64 8), align 4
+; COREI7-NEXT: [[TMP5:%.*]] = load <8 x float>, ptr getelementptr inbounds ([16 x float], ptr @srcB32, i32 0, i64 8), align 4
+; COREI7-NEXT: [[TMP6:%.*]] = call <8 x float> @llvm.maxnum.v8f32(<8 x float> [[TMP4]], <8 x float> [[TMP5]])
+; COREI7-NEXT: store <8 x float> [[TMP6]], ptr getelementptr inbounds ([16 x float], ptr @dst32, i32 0, i64 8), align 4
+; COREI7-NEXT: ret void
+;
+; BDVER1-LABEL: @fmaxnum_16f32(
+; BDVER1-NEXT: [[TMP1:%.*]] = load <8 x float>, ptr @srcA32, align 4
+; BDVER1-NEXT: [[TMP2:%.*]] = load <8 x float>, ptr @srcB32, align 4
+; BDVER1-NEXT: [[TMP3:%.*]] = call <8 x float> @llvm.maxnum.v8f32(<8 x float> [[TMP1]], <8 x float> [[TMP2]])
+; BDVER1-NEXT: store <8 x float> [[TMP3]], ptr @dst32, align 4
+; BDVER1-NEXT: [[TMP4:%.*]] = load <8 x float>, ptr getelementptr inbounds ([16 x float], ptr @srcA32, i32 0, i64 8), align 4
+; BDVER1-NEXT: [[TMP5:%.*]] = load <8 x float>, ptr getelementptr inbounds ([16 x float], ptr @srcB32, i32 0, i64 8), align 4
+; BDVER1-NEXT: [[TMP6:%.*]] = call <8 x float> @llvm.maxnum.v8f32(<8 x float> [[TMP4]], <8 x float> [[TMP5]])
+; BDVER1-NEXT: store <8 x float> [[TMP6]], ptr getelementptr inbounds ([16 x float], ptr @dst32, i32 0, i64 8), align 4
+; BDVER1-NEXT: ret void
+;
+; AVX2-LABEL: @fmaxnum_16f32(
+; AVX2-NEXT: [[TMP1:%.*]] = load <8 x float>, ptr @srcA32, align 4
+; AVX2-NEXT: [[TMP2:%.*]] = load <8 x float>, ptr @srcB32, align 4
+; AVX2-NEXT: [[TMP3:%.*]] = call <8 x float> @llvm.maxnum.v8f32(<8 x float> [[TMP1]], <8 x float> [[TMP2]])
+; AVX2-NEXT: store <8 x float> [[TMP3]], ptr @dst32, align 4
+; AVX2-NEXT: [[TMP4:%.*]] = load <8 x float>, ptr getelementptr inbounds ([16 x float], ptr @srcA32, i32 0, i64 8), align 4
+; AVX2-NEXT: [[TMP5:%.*]] = load <8 x float>, ptr getelementptr inbounds ([16 x float], ptr @srcB32, i32 0, i64 8), align 4
+; AVX2-NEXT: [[TMP6:%.*]] = call <8 x float> @llvm.maxnum.v8f32(<8 x float> [[TMP4]], <8 x float> [[TMP5]])
+; AVX2-NEXT: store <8 x float> [[TMP6]], ptr getelementptr inbounds ([16 x float], ptr @dst32, i32 0, i64 8), align 4
+; AVX2-NEXT: ret void
+;
+; AVX512-LABEL: @fmaxnum_16f32(
+; AVX512-NEXT: [[TMP1:%.*]] = load <16 x float>, ptr @srcA32, align 4
+; AVX512-NEXT: [[TMP2:%.*]] = load <16 x float>, ptr @srcB32, align 4
+; AVX512-NEXT: [[TMP3:%.*]] = call <16 x float> @llvm.maxnum.v16f32(<16 x float> [[TMP1]], <16 x float> [[TMP2]])
+; AVX512-NEXT: store <16 x float> [[TMP3]], ptr @dst32, align 4
+; AVX512-NEXT: ret void
+;
; AVX256-LABEL: @fmaxnum_16f32(
; AVX256-NEXT: [[TMP1:%.*]] = load <8 x float>, ptr @srcA32, align 4
; AVX256-NEXT: [[TMP2:%.*]] = load <8 x float>, ptr @srcB32, align 4
@@ -263,13 +336,6 @@ define void @fmaxnum_16f32() #0 {
; AVX256-NEXT: [[TMP6:%.*]] = call <8 x float> @llvm.maxnum.v8f32(<8 x float> [[TMP4]], <8 x float> [[TMP5]])
; AVX256-NEXT: store <8 x float> [[TMP6]], ptr getelementptr inbounds ([16 x float], ptr @dst32, i32 0, i64 8), align 4
; AVX256-NEXT: ret void
-;
-; AVX512-LABEL: @fmaxnum_16f32(
-; AVX512-NEXT: [[TMP1:%.*]] = load <16 x float>, ptr @srcA32, align 4
-; AVX512-NEXT: [[TMP2:%.*]] = load <16 x float>, ptr @srcB32, align 4
-; AVX512-NEXT: [[TMP3:%.*]] = call <16 x float> @llvm.maxnum.v16f32(<16 x float> [[TMP1]], <16 x float> [[TMP2]])
-; AVX512-NEXT: store <16 x float> [[TMP3]], ptr @dst32, align 4
-; AVX512-NEXT: ret void
;
%a0 = load float, ptr @srcA32, align 4
%a1 = load float, ptr getelementptr inbounds ([16 x float], ptr @srcA32, i32 0, i64 1), align 4
@@ -379,19 +445,84 @@ define float @reduction_v4f32_nnan(ptr %p) {
; Negative test - must have nnan.
define float @reduction_v4f32_not_fast(ptr %p) {
-; CHECK-LABEL: @reduction_v4f32_not_fast(
-; CHECK-NEXT: [[G1:%.*]] = getelementptr inbounds float, ptr [[P:%.*]], i64 1
-; CHECK-NEXT: [[G2:%.*]] = getelementptr inbounds float, ptr [[P]], i64 2
-; CHECK-NEXT: [[G3:%.*]] = getelementptr inbounds float, ptr [[P]], i64 3
-; CHECK-NEXT: [[T0:%.*]] = load float, ptr [[P]], align 4
-; CHECK-NEXT: [[T1:%.*]] = load float, ptr [[G1]], align 4
-; CHECK-NEXT: [[T2:%.*]] = load float, ptr [[G2]], align 4
-; CHECK-NEXT: [[T3:%.*]] = load float, ptr [[G3]], align 4
-; CHECK-NEXT: [[M1:%.*]] = tail call float @llvm.maxnum.f32(float [[T1]], float [[T0]])
-; CHECK-NEXT: [[M2:%.*]] = tail call float @llvm.maxnum.f32(float [[T2]], float [[M1]])
-; CHECK-NEXT: [[M3:%.*]] = tail call float @llvm.maxnum.f32(float [[T3]], float [[M2]])
-; CHECK-NEXT: ret float [[M3]]
+; SSE-LABEL: @reduction_v4f32_not_fast(
+; SSE-NEXT: [[TMP1:%.*]] = load <4 x float>, ptr [[P:%.*]], align 4
+; SSE-NEXT: [[TMP2:%.*]] = extractelement <4 x float> [[TMP1]], i32 0
+; SSE-NEXT: [[TMP3:%.*]] = extractelement <4 x float> [[TMP1]], i32 1
+; SSE-NEXT: [[M1:%.*]] = tail call float @llvm.maxnum.f32(float [[TMP3]], float [[TMP2]])
+; SSE-NEXT: [[TMP4:%.*]] = extractelement <4 x float> [[TMP1]], i32 2
+; SSE-NEXT: [[M2:%.*]] = tail call float @llvm.maxnum.f32(float [[TMP4]], float [[M1]])
+; SSE-NEXT: [[TMP5:%.*]] = extractelement <4 x float> [[TMP1]], i32 3
+; SSE-NEXT: [[M3:%.*]] = tail call float @llvm.maxnum.f32(float [[TMP5]], float [[M2]])
+; SSE-NEXT: ret float [[M3]]
+;
+; COREI7-LABEL: @reduction_v4f32_not_fast(
+; COREI7-NEXT: [[TMP1:%.*]] = load <4 x float>, ptr [[P:%.*]], align 4
+; COREI7-NEXT: [[TMP2:%.*]] = extractelement <4 x float> [[TMP1]], i32 0
+; COREI7-NEXT: [[TMP3:%.*]] = extractelement <4 x float> [[TMP1]], i32 1
+; COREI7-NEXT: [[M1:%.*]] = tail call float @llvm.maxnum.f32(float [[TMP3]], float [[TMP2]])
+; COREI7-NEXT: [[TMP4:%.*]] = extractelement <4 x float> [[TMP1]], i32 2
+; COREI7-NEXT: [[M2:%.*]] = tail call float @llvm.maxnum.f32(float [[TMP4]], float [[M1]])
+; COREI7-NEXT: [[TMP5:%.*]] = extractelement <4 x float> [[TMP1]], i32 3
+; COREI7-NEXT: [[M3:%.*]] = tail call float @llvm.maxnum.f32(float [[TMP5]], float [[M2]])
+; COREI7-NEXT: ret float [[M3]]
+;
+; BDVER1-LABEL: @reduction_v4f32_not_fast(
+; BDVER1-NEXT: [[TMP1:%.*]] = load <4 x float>, ptr [[P:%.*]], align 4
+; BDVER1-NEXT: [[TMP2:%.*]] = extractelement <4 x float> [[TMP1]], i32 0
+; BDVER1-NEXT: [[TMP3:%.*]] = extractelement <4 x float> [[TMP1]], i32 1
+; BDVER1-NEXT: [[M1:%.*]] = tail call float @llvm.maxnum.f32(float [[TMP3]], float [[TMP2]])
+; BDVER1-NEXT: [[TMP4:%.*]] = extractelement <4 x float> [[TMP1]], i32 2
+; BDVER1-NEXT: [[M2:%.*]] = tail call float @llvm.maxnum.f32(float [[TMP4]], float [[M1]])
+; BDVER1-NEXT: [[TMP5:%.*]] = extractelement <4 x float> [[TMP1]], i32 3
+; BDVER1-NEXT: [[M3:%.*]] = tail call float @llvm.maxnum.f32(float [[TMP5]], float [[M2]])
+; BDVER1-NEXT: ret float [[M3]]
;
+; AVX2-LABEL: @reduction_v4f32_not_fast(
+; AVX2-NEXT: [[G1:%.*]] = getelementptr inbounds float, ptr [[P:%.*]], i64 1
+; AVX2-NEXT: [[G2:%.*]] = getelementptr inbounds float, ptr [[P]], i64 2
+; AVX2-NEXT: [[G3:%.*]] = getelementptr inbounds float, ptr [[P]], i64 3
+; AVX2-NEXT: [[T0:%.*]] = load float, ptr [[P]], align 4
+; AVX2-NEXT: [[T1:%.*]] = load float, ptr [[G1]], align 4
+; AVX2-NEXT: [[T2:%.*]] = load float, ptr [[G2]], align 4
+; AVX2-NEXT: [[T3:%.*]] = load float, ptr [[G3]], align 4
+; AVX2-NEXT: [[M1:%.*]] = tail call float @llvm.maxnum.f32(float [[T1]], float [[T0]])
+; AVX2-NEXT: [[M2:%.*]] = tail call float @llvm.maxnum.f32(float [[T2]], float [[M1]])
+; AVX2-NEXT: [[M3:%.*]] = tail call float @llvm.maxnum.f32(float [[T3]], float [[M2]])
+; AVX2-NEXT: ret float [[M3]]
+;
+; AVX512-LABEL: @reduction_v4f32_not_fast(
+; AVX512-NEXT: [[TMP1:%.*]] = load <4 x float>, ptr [[P:%.*]], align 4
+; AVX512-NEXT: [[TMP2:%.*]] = extractelement <4 x float> [[TMP1]], i32 0
+; AVX512-NEXT: [[TMP3:%.*]] = extractelement <4 x float> [[TMP1]], i32 1
+; AVX512-NEXT: [[M1:%.*]] = tail call float @llvm.maxnum.f32(float [[TMP3]], float [[TMP2]])
+; AVX512-NEXT: [[TMP4:%.*]] = extractelement <4 x float> [[TMP1]], i32 2
+; AVX512-NEXT: [[M2:%.*]] = tail call float @llvm.maxnum.f32(float [[TMP4]], float [[M1]])
+; AVX512-NEXT: [[TMP5:%.*]] = extractelement <4 x float> [[TMP1]], i32 3
+; AVX512-NEXT: [[M3:%.*]] = tail call float @llvm.maxnum.f32(float [[TMP5]], float [[M2]])
+; AVX512-NEXT: ret float [[M3]]
+;
+; AVX256-LABEL: @reduction_v4f32_not_fast(
+; AVX256-NEXT: [[TMP1:%.*]] = load <4 x float>, ptr [[P:%.*]], align 4
+; AVX256-NEXT: [[TMP2:%.*]] = extractelement <4 x float> [[TMP1]], i32 0
+; AVX256-NEXT: [[TMP3:%.*]] = extractelement <4 x float> [[TMP1]], i32 1
+; AVX256-NEXT: [[M1:%.*]] = tail call float @llvm.maxnum.f32(float [[TMP3]], float [[TMP2]])
+; AVX256-NEXT: [[TMP4:%.*]] = extractelement <4 x float> [[TMP1]], i32 2
+; AVX256-NEXT: [[M2:%.*]] = tail call float @llvm.maxnum.f32(float [[TMP4]], float [[M1]])
+; AVX256-NEXT: [[TMP5:%.*]] = extractelement <4 x float> [[TMP1]], i32 3
+; AVX256-NEXT: [[M3:%.*]] = tail call float @llvm.maxnum.f32(float [[TMP5]], float [[M2]])
+; AVX256-NEXT: ret float [[M3]]
+;
+; PREF-AVX256-LABEL: @reduction_v4f32_not_fast(
+; PREF-AVX256-NEXT: [[TMP1:%.*]] = load <4 x float>, ptr [[P:%.*]], align 4
+; PREF-AVX256-NEXT: [[TMP2:%.*]] = extractelement <4 x float> [[TMP1]], i32 0
+; PREF-AVX256-NEXT: [[TMP3:%.*]] = extractelement <4 x float> [[TMP1]], i32 1
+; PREF-AVX256-NEXT: [[M1:%.*]] = tail call float @llvm.maxnum.f32(float [[TMP3]], float [[TMP2]])
+; PREF-AVX256-NEXT: [[TMP4:%.*]] = extractelement <4 x float> [[TMP1]], i32 2
+; PREF-AVX256-NEXT: [[M2:%.*]] = tail call float @llvm.maxnum.f32(float [[TMP4]], float [[M1]])
+; PREF-AVX256-NEXT: [[TMP5:%.*]] = extractelement <4 x float> [[TMP1]], i32 3
+; PREF-AVX256-NEXT: [[M3:%.*]] = tail call float @llvm.maxnum.f32(float [[TMP5]], float [[M2]])
+; PREF-AVX256-NEXT: ret float [[M3]]
%g1 = getelementptr inbounds float, ptr %p, i64 1
%g2 = getelementptr inbounds float, ptr %p, i64 2
%g3 = getelementptr inbounds float, ptr %p, i64 3
@@ -473,19 +604,88 @@ define double @reduction_v4f64_fast(ptr %p) {
; Negative test - must have nnan.
define double @reduction_v4f64_wrong_fmf(ptr %p) {
-; CHECK-LABEL: @reduction_v4f64_wrong_fmf(
-; CHECK-NEXT: [[G1:%.*]] = getelementptr inbounds double, ptr [[P:%.*]], i64 1
-; CHECK-NEXT: [[G2:%.*]] = getelementptr inbounds double, ptr [[P]], i64 2
-; CHECK-NEXT: [[G3:%.*]] = getelementptr inbounds double, ptr [[P]], i64 3
-; CHECK-NEXT: [[T0:%.*]] = load double, ptr [[P]], align 4
-; CHECK-NEXT: [[T1:%.*]] = load double, ptr [[G1]], align 4
-; CHECK-NEXT: [[T2:%.*]] = load double, ptr [[G2]], align 4
-; CHECK-NEXT: [[T3:%.*]] = load double, ptr [[G3]], align 4
-; CHECK-NEXT: [[M1:%.*]] = tail call ninf nsz double @llvm.maxnum.f64(double [[T1]], double [[T0]])
-; CHECK-NEXT: [[M2:%.*]] = tail call ninf nsz double @llvm.maxnum.f64(double [[T2]], double [[M1]])
-; CHECK-NEXT: [[M3:%.*]] = tail call ninf nsz double @llvm.maxnum.f64(double [[T3]], double [[M2]])
-; CHECK-NEXT: ret double [[M3]]
+; SSE-LABEL: @reduction_v4f64_wrong_fmf(
+; SSE-NEXT: [[TMP1:%.*]] = load <4 x double>, ptr [[P:%.*]], align 4
+; SSE-NEXT: [[TMP2:%.*]] = extractelement <4 x double> [[TMP1]], i32 0
+; SSE-NEXT: [[TMP3:%.*]] = extractelement <4 x double> [[TMP1]], i32 1
+; SSE-NEXT: [[M1:%.*]] = tail call ninf nsz double @llvm.maxnum.f64(double [[TMP3]], double [[TMP2]])
+; SSE-NEXT: [[TMP4:%.*]] = extractelement <4 x double> [[TMP1]], i32 2
+; SSE-NEXT: [[M2:%.*]] = tail call ninf nsz double @llvm.maxnum.f64(double [[TMP4]], double [[M1]])
+; SSE-NEXT: [[TMP5:%.*]] = extractelement <4 x double> [[TMP1]], i32 3
+; SSE-NEXT: [[M3:%.*]] = tail call ninf nsz double @llvm.maxnum.f64(double [[TMP5]], double [[M2]])
+; SSE-NEXT: ret double [[M3]]
+;
+; COREI7-LABEL: @reduction_v4f64_wrong_fmf(
+; COREI7-NEXT: [[G1:%.*]] = getelementptr inbounds double, ptr [[P:%.*]], i64 1
+; COREI7-NEXT: [[G2:%.*]] = getelementptr inbounds double, ptr [[P]], i64 2
+; COREI7-NEXT: [[G3:%.*]] = getelementptr inbounds double, ptr [[P]], i64 3
+; COREI7-NEXT: [[T0:%.*]] = load double, ptr [[P]], align 4
+; COREI7-NEXT: [[T1:%.*]] = load double, ptr [[G1]], align 4
+; COREI7-NEXT: [[T2:%.*]] = load double, ptr [[G2]], align 4
+; COREI7-NEXT: [[T3:%.*]] = load double, ptr [[G3]], align 4
+; COREI7-NEXT: [[M1:%.*]] = tail call ninf nsz double @llvm.maxnum.f64(double [[T1]], double [[T0]])
+; COREI7-NEXT: [[M2:%.*]] = tail call ninf nsz double @llvm.maxnum.f64(double [[T2]], double [[M1]])
+; COREI7-NEXT: [[M3:%.*]] = tail call ninf nsz double @llvm.maxnum.f64(double [[T3]], double [[M2]])
+; COREI7-NEXT: ret double [[M3]]
+;
+; BDVER1-LABEL: @reduction_v4f64_wrong_fmf(
+; BDVER1-NEXT: [[G1:%.*]] = getelementptr inbounds double, ptr [[P:%.*]], i64 1
+; BDVER1-NEXT: [[G2:%.*]] = getelementptr inbounds double, ptr [[P]], i64 2
+; BDVER1-NEXT: [[G3:%.*]] = getelementptr inbounds double, ptr [[P]], i64 3
+; BDVER1-NEXT: [[T0:%.*]] = load double, ptr [[P]], align 4
+; BDVER1-NEXT: [[T1:%.*]] = load double, ptr [[G1]], align 4
+; BDVER1-NEXT: [[T2:%.*]] = load double, ptr [[G2]], align 4
+; BDVER1-NEXT: [[T3:%.*]] = load double, ptr [[G3]], align 4
+; BDVER1-NEXT: [[M1:%.*]] = tail call ninf nsz double @llvm.maxnum.f64(double [[T1]], double [[T0]])
+; BDVER1-NEXT: [[M2:%.*]] = tail call ninf nsz double @llvm.maxnum.f64(double [[T2]], double [[M1]])
+; BDVER1-NEXT: [[M3:%.*]] = tail call ninf nsz double @llvm.maxnum.f64(double [[T3]], double [[M2]])
+; BDVER1-NEXT: ret double [[M3]]
+;
+; AVX2-LABEL: @reduction_v4f64_wrong_fmf(
+; AVX2-NEXT: [[G1:%.*]] = getelementptr inbounds double, ptr [[P:%.*]], i64 1
+; AVX2-NEXT: [[G2:%.*]] = getelementptr inbounds double, ptr [[P]], i64 2
+; AVX2-NEXT: [[G3:%.*]] = getelementptr inbounds double, ptr [[P]], i64 3
+; AVX2-NEXT: [[T0:%.*]] = load double, ptr [[P]], align 4
+; AVX2-NEXT: [[T1:%.*]] = load double, ptr [[G1]], align 4
+; AVX2-NEXT: [[T2:%.*]] = load double, ptr [[G2]], align 4
+; AVX2-NEXT: [[T3:%.*]] = load double, ptr [[G3]], align 4
+; AVX2-NEXT: [[M1:%.*]] = tail call ninf nsz double @llvm.maxnum.f64(double [[T1]], double [[T0]])
+; AVX2-NEXT: [[M2:%.*]] = tail call ninf nsz double @llvm.maxnum.f64(double [[T2]], double [[M1]])
+; AVX2-NEXT: [[M3:%.*]] = tail call ninf nsz double @llvm.maxnum.f64(double [[T3]], double [[M2]])
+; AVX2-NEXT: ret double [[M3]]
+;
+; AVX512-LABEL: @reduction_v4f64_wrong_fmf(
+; AVX512-NEXT: [[TMP1:%.*]] = load <4 x double>, ptr [[P:%.*]], align 4
+; AVX512-NEXT: [[TMP2:%.*]] = extractelement <4 x double> [[TMP1]], i32 0
+; AVX512-NEXT: [[TMP3:%.*]] = extractelement <4 x double> [[TMP1]], i32 1
+; AVX512-NEXT: [[M1:%.*]] = tail call ninf nsz double @llvm.maxnum.f64(double [[TMP3]], double [[TMP2]])
+; AVX512-NEXT: [[TMP4:%.*]] = extractelement <4 x double> [[TMP1]], i32 2
+; AVX512-NEXT: [[M2:%.*]] = tail call ninf nsz double @llvm.maxnum.f64(double [[TMP4]], double [[M1]])
+; AVX512-NEXT: [[TMP5:%.*]] = extractelement <4 x double> [[TMP1]], i32 3
+; AVX512-NEXT: [[M3:%.*]] = tail call ninf nsz double @llvm.maxnum.f64(double [[TMP5]], double [[M2]])
+; AVX512-NEXT: ret double [[M3]]
+;
+; AVX256-LABEL: @reduction_v4f64_wrong_fmf(
+; AVX256-NEXT: [[TMP1:%.*]] = load <4 x double>, ptr [[P:%.*]], align 4
+; AVX256-NEXT: [[TMP2:%.*]] = extractelement <4 x double> [[TMP1]], i32 0
+; AVX256-NEXT: [[TMP3:%.*]] = extractelement <4 x double> [[TMP1]], i32 1
+; AVX256-NEXT: [[M1:%.*]] = tail call ninf nsz double @llvm.maxnum.f64(double [[TMP3]], double [[TMP2]])
+; AVX256-NEXT: [[TMP4:%.*]] = extractelement <4 x double> [[TMP1]], i32 2
+; AVX256-NEXT: [[M2:%.*]] = tail call ninf nsz double @llvm.maxnum.f64(double [[TMP4]], double [[M1]])
+; AVX256-NEXT: [[TMP5:%.*]] = extractelement <4 x double> [[TMP1]], i32 3
+; AVX256-NEXT: [[M3:%.*]] = tail call ninf nsz double @llvm.maxnum.f64(double [[TMP5]], double [[M2]])
+; AVX256-NEXT: ret double [[M3]]
;
+; PREF-AVX256-LABEL: @reduction_v4f64_wrong_fmf(
+; PREF-AVX256-NEXT: [[TMP1:%.*]] = load <4 x double>, ptr [[P:%.*]], align 4
+; PREF-AVX256-NEXT: [[TMP2:%.*]] = extractelement <4 x double> [[TMP1]], i32 0
+; PREF-AVX256-NEXT: [[TMP3:%.*]] = extractelement <4 x double> [[TMP1]], i32 1
+; PREF-AVX256-NEXT: [[M1:%.*]] = tail call ninf nsz double @llvm.maxnum.f64(double [[TMP3]], double [[TMP2]])
+; PREF-AVX256-NEXT: [[TMP4:%.*]] = extractelement <4 x double> [[TMP1]], i32 2
+; PREF-AVX256-NEXT: [[M2:%.*]] = tail call ninf nsz double @llvm.maxnum.f64(double [[TMP4]], double [[M1]])
+; PREF-AVX256-NEXT: [[TMP5:%.*]] = extractelement <4 x double> [[TMP1]], i32 3
+; PREF-AVX256-NEXT: [[M3:%.*]] = tail call ninf nsz double @llvm.maxnum.f64(double [[TMP5]], double [[M2]])
+; PREF-AVX256-NEXT: ret double [[M3]]
%g1 = getelementptr inbounds double, ptr %p, i64 1
%g2 = getelementptr inbounds double, ptr %p, i64 2
%g3 = getelementptr inbounds double, ptr %p, i64 3
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/fminnum.ll b/llvm/test/Transforms/SLPVectorizer/X86/fminnum.ll
index 434fa13e880bb..5eb2328a3dc8f 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/fminnum.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/fminnum.ll
@@ -1,8 +1,8 @@
; NOTE: Assertions have been autogenerated by utils/update_test_checks.py
; RUN: opt < %s -mtriple=x86_64-unknown -passes=slp-vectorizer -S | FileCheck %s --check-prefixes=CHECK,SSE
-; RUN: opt < %s -mtriple=x86_64-unknown -mcpu=corei7-avx -passes=slp-vectorizer -S | FileCheck %s --check-prefixes=CHECK,AVX,AVX256
-; RUN: opt < %s -mtriple=x86_64-unknown -mcpu=bdver1 -passes=slp-vectorizer -S | FileCheck %s --check-prefixes=CHECK,AVX,AVX256
-; RUN: opt < %s -mtriple=x86_64-unknown -mcpu=core-avx2 -passes=slp-vectorizer -S | FileCheck %s --check-prefixes=CHECK,AVX,AVX256
+; RUN: opt < %s -mtriple=x86_64-unknown -mcpu=corei7-avx -passes=slp-vectorizer -S | FileCheck %s --check-prefixes=CHECK,AVX,COREI7
+; RUN: opt < %s -mtriple=x86_64-unknown -mcpu=bdver1 -passes=slp-vectorizer -S | FileCheck %s --check-prefixes=CHECK,AVX,BDVER1
+; RUN: opt < %s -mtriple=x86_64-unknown -mcpu=core-avx2 -passes=slp-vectorizer -S | FileCheck %s --check-prefixes=CHECK,AVX,AVX2
; RUN: opt < %s -mtriple=x86_64-unknown -mcpu=skylake-avx512 -mattr=-prefer-256-bit -passes=slp-vectorizer -S | FileCheck %s --check-prefixes=CHECK,AVX,AVX512
; RUN: opt < %s -mtriple=x86_64-unknown -mcpu=skylake-avx512 -mattr=+prefer-256-bit -passes=slp-vectorizer -S | FileCheck %s --check-prefixes=CHECK,AVX,AVX256
@@ -99,6 +99,46 @@ define void @fminnum_8f64() #0 {
; SSE-NEXT: store <2 x double> [[TMP12]], ptr getelementptr inbounds ([8 x double], ptr @dst64, i32 0, i64 6), align 4
; SSE-NEXT: ret void
;
+; COREI7-LABEL: @fminnum_8f64(
+; COREI7-NEXT: [[TMP1:%.*]] = load <4 x double>, ptr @srcA64, align 4
+; COREI7-NEXT: [[TMP2:%.*]] = load <4 x double>, ptr @srcB64, align 4
+; COREI7-NEXT: [[TMP3:%.*]] = call <4 x double> @llvm.minnum.v4f64(<4 x double> [[TMP1]], <4 x double> [[TMP2]])
+; COREI7-NEXT: store <4 x double> [[TMP3]], ptr @dst64, align 4
+; COREI7-NEXT: [[TMP4:%.*]] = load <4 x double>, ptr getelementptr inbounds ([8 x double], ptr @srcA64, i32 0, i64 4), align 4
+; COREI7-NEXT: [[TMP5:%.*]] = load <4 x double>, ptr getelementptr inbounds ([8 x double], ptr @srcB64, i32 0, i64 4), align 4
+; COREI7-NEXT: [[TMP6:%.*]] = call <4 x double> @llvm.minnum.v4f64(<4 x double> [[TMP4]], <4 x double> [[TMP5]])
+; COREI7-NEXT: store <4 x double> [[TMP6]], ptr getelementptr inbounds ([8 x double], ptr @dst64, i32 0, i64 4), align 4
+; COREI7-NEXT: ret void
+;
+; BDVER1-LABEL: @fminnum_8f64(
+; BDVER1-NEXT: [[TMP1:%.*]] = load <4 x double>, ptr @srcA64, align 4
+; BDVER1-NEXT: [[TMP2:%.*]] = load <4 x double>, ptr @srcB64, align 4
+; BDVER1-NEXT: [[TMP3:%.*]] = call <4 x double> @llvm.minnum.v4f64(<4 x double> [[TMP1]], <4 x double> [[TMP2]])
+; BDVER1-NEXT: store <4 x double> [[TMP3]], ptr @dst64, align 4
+; BDVER1-NEXT: [[TMP4:%.*]] = load <4 x double>, ptr getelementptr inbounds ([8 x double], ptr @srcA64, i32 0, i64 4), align 4
+; BDVER1-NEXT: [[TMP5:%.*]] = load <4 x double>, ptr getelementptr inbounds ([8 x double], ptr @srcB64, i32 0, i64 4), align 4
+; BDVER1-NEXT: [[TMP6:%.*]] = call <4 x double> @llvm.minnum.v4f64(<4 x double> [[TMP4]], <4 x double> [[TMP5]])
+; BDVER1-NEXT: store <4 x double> [[TMP6]], ptr getelementptr inbounds ([8 x double], ptr @dst64, i32 0, i64 4), align 4
+; BDVER1-NEXT: ret void
+;
+; AVX2-LABEL: @fminnum_8f64(
+; AVX2-NEXT: [[TMP1:%.*]] = load <4 x double>, ptr @srcA64, align 4
+; AVX2-NEXT: [[TMP2:%.*]] = load <4 x double>, ptr @srcB64, align 4
+; AVX2-NEXT: [[TMP3:%.*]] = call <4 x double> @llvm.minnum.v4f64(<4 x double> [[TMP1]], <4 x double> [[TMP2]])
+; AVX2-NEXT: store <4 x double> [[TMP3]], ptr @dst64, align 4
+; AVX2-NEXT: [[TMP4:%.*]] = load <4 x double>, ptr getelementptr inbounds ([8 x double], ptr @srcA64, i32 0, i64 4), align 4
+; AVX2-NEXT: [[TMP5:%.*]] = load <4 x double>, ptr getelementptr inbounds ([8 x double], ptr @srcB64, i32 0, i64 4), align 4
+; AVX2-NEXT: [[TMP6:%.*]] = call <4 x double> @llvm.minnum.v4f64(<4 x double> [[TMP4]], <4 x double> [[TMP5]])
+; AVX2-NEXT: store <4 x double> [[TMP6]], ptr getelementptr inbounds ([8 x double], ptr @dst64, i32 0, i64 4), align 4
+; AVX2-NEXT: ret void
+;
+; AVX512-LABEL: @fminnum_8f64(
+; AVX512-NEXT: [[TMP1:%.*]] = load <8 x double>, ptr @srcA64, align 4
+; AVX512-NEXT: [[TMP2:%.*]] = load <8 x double>, ptr @srcB64, align 4
+; AVX512-NEXT: [[TMP3:%.*]] = call <8 x double> @llvm.minnum.v8f64(<8 x double> [[TMP1]], <8 x double> [[TMP2]])
+; AVX512-NEXT: store <8 x double> [[TMP3]], ptr @dst64, align 4
+; AVX512-NEXT: ret void
+;
; AVX256-LABEL: @fminnum_8f64(
; AVX256-NEXT: [[TMP1:%.*]] = load <4 x double>, ptr @srcA64, align 4
; AVX256-NEXT: [[TMP2:%.*]] = load <4 x double>, ptr @srcB64, align 4
@@ -109,13 +149,6 @@ define void @fminnum_8f64() #0 {
; AVX256-NEXT: [[TMP6:%.*]] = call <4 x double> @llvm.minnum.v4f64(<4 x double> [[TMP4]], <4 x double> [[TMP5]])
; AVX256-NEXT: store <4 x double> [[TMP6]], ptr getelementptr inbounds ([8 x double], ptr @dst64, i32 0, i64 4), align 4
; AVX256-NEXT: ret void
-;
-; AVX512-LABEL: @fminnum_8f64(
-; AVX512-NEXT: [[TMP1:%.*]] = load <8 x double>, ptr @srcA64, align 4
-; AVX512-NEXT: [[TMP2:%.*]] = load <8 x double>, ptr @srcB64, align 4
-; AVX512-NEXT: [[TMP3:%.*]] = call <8 x double> @llvm.minnum.v8f64(<8 x double> [[TMP1]], <8 x double> [[TMP2]])
-; AVX512-NEXT: store <8 x double> [[TMP3]], ptr @dst64, align 4
-; AVX512-NEXT: ret void
;
%a0 = load double, ptr @srcA64, align 4
%a1 = load double, ptr getelementptr inbounds ([8 x double], ptr @srcA64, i32 0, i64 1), align 4
@@ -253,6 +286,46 @@ define void @fminnum_16f32() #0 {
; SSE-NEXT: store <4 x float> [[TMP12]], ptr getelementptr inbounds ([16 x float], ptr @dst32, i32 0, i64 12), align 4
; SSE-NEXT: ret void
;
+; COREI7-LABEL: @fminnum_16f32(
+; COREI7-NEXT: [[TMP1:%.*]] = load <8 x float>, ptr @srcA32, align 4
+; COREI7-NEXT: [[TMP2:%.*]] = load <8 x float>, ptr @srcB32, align 4
+; COREI7-NEXT: [[TMP3:%.*]] = call <8 x float> @llvm.minnum.v8f32(<8 x float> [[TMP1]], <8 x float> [[TMP2]])
+; COREI7-NEXT: store <8 x float> [[TMP3]], ptr @dst32, align 4
+; COREI7-NEXT: [[TMP4:%.*]] = load <8 x float>, ptr getelementptr inbounds ([16 x float], ptr @srcA32, i32 0, i64 8), align 4
+; COREI7-NEXT: [[TMP5:%.*]] = load <8 x float>, ptr getelementptr inbounds ([16 x float], ptr @srcB32, i32 0, i64 8), align 4
+; COREI7-NEXT: [[TMP6:%.*]] = call <8 x float> @llvm.minnum.v8f32(<8 x float> [[TMP4]], <8 x float> [[TMP5]])
+; COREI7-NEXT: store <8 x float> [[TMP6]], ptr getelementptr inbounds ([16 x float], ptr @dst32, i32 0, i64 8), align 4
+; COREI7-NEXT: ret void
+;
+; BDVER1-LABEL: @fminnum_16f32(
+; BDVER1-NEXT: [[TMP1:%.*]] = load <8 x float>, ptr @srcA32, align 4
+; BDVER1-NEXT: [[TMP2:%.*]] = load <8 x float>, ptr @srcB32, align 4
+; BDVER1-NEXT: [[TMP3:%.*]] = call <8 x float> @llvm.minnum.v8f32(<8 x float> [[TMP1]], <8 x float> [[TMP2]])
+; BDVER1-NEXT: store <8 x float> [[TMP3]], ptr @dst32, align 4
+; BDVER1-NEXT: [[TMP4:%.*]] = load <8 x float>, ptr getelementptr inbounds ([16 x float], ptr @srcA32, i32 0, i64 8), align 4
+; BDVER1-NEXT: [[TMP5:%.*]] = load <8 x float>, ptr getelementptr inbounds ([16 x float], ptr @srcB32, i32 0, i64 8), align 4
+; BDVER1-NEXT: [[TMP6:%.*]] = call <8 x float> @llvm.minnum.v8f32(<8 x float> [[TMP4]], <8 x float> [[TMP5]])
+; BDVER1-NEXT: store <8 x float> [[TMP6]], ptr getelementptr inbounds ([16 x float], ptr @dst32, i32 0, i64 8), align 4
+; BDVER1-NEXT: ret void
+;
+; AVX2-LABEL: @fminnum_16f32(
+; AVX2-NEXT: [[TMP1:%.*]] = load <8 x float>, ptr @srcA32, align 4
+; AVX2-NEXT: [[TMP2:%.*]] = load <8 x float>, ptr @srcB32, align 4
+; AVX2-NEXT: [[TMP3:%.*]] = call <8 x float> @llvm.minnum.v8f32(<8 x float> [[TMP1]], <8 x float> [[TMP2]])
+; AVX2-NEXT: store <8 x float> [[TMP3]], ptr @dst32, align 4
+; AVX2-NEXT: [[TMP4:%.*]] = load <8 x float>, ptr getelementptr inbounds ([16 x float], ptr @srcA32, i32 0, i64 8), align 4
+; AVX2-NEXT: [[TMP5:%.*]] = load <8 x float>, ptr getelementptr inbounds ([16 x float], ptr @srcB32, i32 0, i64 8), align 4
+; AVX2-NEXT: [[TMP6:%.*]] = call <8 x float> @llvm.minnum.v8f32(<8 x float> [[TMP4]], <8 x float> [[TMP5]])
+; AVX2-NEXT: store <8 x float> [[TMP6]], ptr getelementptr inbounds ([16 x float], ptr @dst32, i32 0, i64 8), align 4
+; AVX2-NEXT: ret void
+;
+; AVX512-LABEL: @fminnum_16f32(
+; AVX512-NEXT: [[TMP1:%.*]] = load <16 x float>, ptr @srcA32, align 4
+; AVX512-NEXT: [[TMP2:%.*]] = load <16 x float>, ptr @srcB32, align 4
+; AVX512-NEXT: [[TMP3:%.*]] = call <16 x float> @llvm.minnum.v16f32(<16 x float> [[TMP1]], <16 x float> [[TMP2]])
+; AVX512-NEXT: store <16 x float> [[TMP3]], ptr @dst32, align 4
+; AVX512-NEXT: ret void
+;
; AVX256-LABEL: @fminnum_16f32(
; AVX256-NEXT: [[TMP1:%.*]] = load <8 x float>, ptr @srcA32, align 4
; AVX256-NEXT: [[TMP2:%.*]] = load <8 x float>, ptr @srcB32, align 4
@@ -263,13 +336,6 @@ define void @fminnum_16f32() #0 {
; AVX256-NEXT: [[TMP6:%.*]] = call <8 x float> @llvm.minnum.v8f32(<8 x float> [[TMP4]], <8 x float> [[TMP5]])
; AVX256-NEXT: store <8 x float> [[TMP6]], ptr getelementptr inbounds ([16 x float], ptr @dst32, i32 0, i64 8), align 4
; AVX256-NEXT: ret void
-;
-; AVX512-LABEL: @fminnum_16f32(
-; AVX512-NEXT: [[TMP1:%.*]] = load <16 x float>, ptr @srcA32, align 4
-; AVX512-NEXT: [[TMP2:%.*]] = load <16 x float>, ptr @srcB32, align 4
-; AVX512-NEXT: [[TMP3:%.*]] = call <16 x float> @llvm.minnum.v16f32(<16 x float> [[TMP1]], <16 x float> [[TMP2]])
-; AVX512-NEXT: store <16 x float> [[TMP3]], ptr @dst32, align 4
-; AVX512-NEXT: ret void
;
%a0 = load float, ptr @srcA32, align 4
%a1 = load float, ptr getelementptr inbounds ([16 x float], ptr @srcA32, i32 0, i64 1), align 4
@@ -379,18 +445,73 @@ define float @reduction_v4f32_nnan(ptr %p) {
; Negative test - must have nnan.
define float @reduction_v4f32_wrong_fmf(ptr %p) {
-; CHECK-LABEL: @reduction_v4f32_wrong_fmf(
-; CHECK-NEXT: [[G1:%.*]] = getelementptr inbounds float, ptr [[P:%.*]], i64 1
-; CHECK-NEXT: [[G2:%.*]] = getelementptr inbounds float, ptr [[P]], i64 2
-; CHECK-NEXT: [[G3:%.*]] = getelementptr inbounds float, ptr [[P]], i64 3
-; CHECK-NEXT: [[T0:%.*]] = load float, ptr [[P]], align 4
-; CHECK-NEXT: [[T1:%.*]] = load float, ptr [[G1]], align 4
-; CHECK-NEXT: [[T2:%.*]] = load float, ptr [[G2]], align 4
-; CHECK-NEXT: [[T3:%.*]] = load float, ptr [[G3]], align 4
-; CHECK-NEXT: [[M1:%.*]] = tail call reassoc nsz float @llvm.minnum.f32(float [[T1]], float [[T0]])
-; CHECK-NEXT: [[M2:%.*]] = tail call reassoc nsz float @llvm.minnum.f32(float [[T2]], float [[M1]])
-; CHECK-NEXT: [[M3:%.*]] = tail call reassoc nsz float @llvm.minnum.f32(float [[T3]], float [[M2]])
-; CHECK-NEXT: ret float [[M3]]
+; SSE-LABEL: @reduction_v4f32_wrong_fmf(
+; SSE-NEXT: [[TMP1:%.*]] = load <4 x float>, ptr [[P:%.*]], align 4
+; SSE-NEXT: [[TMP2:%.*]] = extractelement <4 x float> [[TMP1]], i32 0
+; SSE-NEXT: [[TMP3:%.*]] = extractelement <4 x float> [[TMP1]], i32 1
+; SSE-NEXT: [[M1:%.*]] = tail call reassoc nsz float @llvm.minnum.f32(float [[TMP3]], float [[TMP2]])
+; SSE-NEXT: [[TMP4:%.*]] = extractelement <4 x float> [[TMP1]], i32 2
+; SSE-NEXT: [[M2:%.*]] = tail call reassoc nsz float @llvm.minnum.f32(float [[TMP4]], float [[M1]])
+; SSE-NEXT: [[TMP5:%.*]] = extractelement <4 x float> [[TMP1]], i32 3
+; SSE-NEXT: [[M3:%.*]] = tail call reassoc nsz float @llvm.minnum.f32(float [[TMP5]], float [[M2]])
+; SSE-NEXT: ret float [[M3]]
+;
+; COREI7-LABEL: @reduction_v4f32_wrong_fmf(
+; COREI7-NEXT: [[TMP1:%.*]] = load <4 x float>, ptr [[P:%.*]], align 4
+; COREI7-NEXT: [[TMP2:%.*]] = extractelement <4 x float> [[TMP1]], i32 0
+; COREI7-NEXT: [[TMP3:%.*]] = extractelement <4 x float> [[TMP1]], i32 1
+; COREI7-NEXT: [[M1:%.*]] = tail call reassoc nsz float @llvm.minnum.f32(float [[TMP3]], float [[TMP2]])
+; COREI7-NEXT: [[TMP4:%.*]] = extractelement <4 x float> [[TMP1]], i32 2
+; COREI7-NEXT: [[M2:%.*]] = tail call reassoc nsz float @llvm.minnum.f32(float [[TMP4]], float [[M1]])
+; COREI7-NEXT: [[TMP5:%.*]] = extractelement <4 x float> [[TMP1]], i32 3
+; COREI7-NEXT: [[M3:%.*]] = tail call reassoc nsz float @llvm.minnum.f32(float [[TMP5]], float [[M2]])
+; COREI7-NEXT: ret float [[M3]]
+;
+; BDVER1-LABEL: @reduction_v4f32_wrong_fmf(
+; BDVER1-NEXT: [[TMP1:%.*]] = load <4 x float>, ptr [[P:%.*]], align 4
+; BDVER1-NEXT: [[TMP2:%.*]] = extractelement <4 x float> [[TMP1]], i32 0
+; BDVER1-NEXT: [[TMP3:%.*]] = extractelement <4 x float> [[TMP1]], i32 1
+; BDVER1-NEXT: [[M1:%.*]] = tail call reassoc nsz float @llvm.minnum.f32(float [[TMP3]], float [[TMP2]])
+; BDVER1-NEXT: [[TMP4:%.*]] = extractelement <4 x float> [[TMP1]], i32 2
+; BDVER1-NEXT: [[M2:%.*]] = tail call reassoc nsz float @llvm.minnum.f32(float [[TMP4]], float [[M1]])
+; BDVER1-NEXT: [[TMP5:%.*]] = extractelement <4 x float> [[TMP1]], i32 3
+; BDVER1-NEXT: [[M3:%.*]] = tail call reassoc nsz float @llvm.minnum.f32(float [[TMP5]], float [[M2]])
+; BDVER1-NEXT: ret float [[M3]]
+;
+; AVX2-LABEL: @reduction_v4f32_wrong_fmf(
+; AVX2-NEXT: [[G1:%.*]] = getelementptr inbounds float, ptr [[P:%.*]], i64 1
+; AVX2-NEXT: [[G2:%.*]] = getelementptr inbounds float, ptr [[P]], i64 2
+; AVX2-NEXT: [[G3:%.*]] = getelementptr inbounds float, ptr [[P]], i64 3
+; AVX2-NEXT: [[T0:%.*]] = load float, ptr [[P]], align 4
+; AVX2-NEXT: [[T1:%.*]] = load float, ptr [[G1]], align 4
+; AVX2-NEXT: [[T2:%.*]] = load float, ptr [[G2]], align 4
+; AVX2-NEXT: [[T3:%.*]] = load float, ptr [[G3]], align 4
+; AVX2-NEXT: [[M1:%.*]] = tail call reassoc nsz float @llvm.minnum.f32(float [[T1]], float [[T0]])
+; AVX2-NEXT: [[M2:%.*]] = tail call reassoc nsz float @llvm.minnum.f32(float [[T2]], float [[M1]])
+; AVX2-NEXT: [[M3:%.*]] = tail call reassoc nsz float @llvm.minnum.f32(float [[T3]], float [[M2]])
+; AVX2-NEXT: ret float [[M3]]
+;
+; AVX512-LABEL: @reduction_v4f32_wrong_fmf(
+; AVX512-NEXT: [[TMP1:%.*]] = load <4 x float>, ptr [[P:%.*]], align 4
+; AVX512-NEXT: [[TMP2:%.*]] = extractelement <4 x float> [[TMP1]], i32 0
+; AVX512-NEXT: [[TMP3:%.*]] = extractelement <4 x float> [[TMP1]], i32 1
+; AVX512-NEXT: [[M1:%.*]] = tail call reassoc nsz float @llvm.minnum.f32(float [[TMP3]], float [[TMP2]])
+; AVX512-NEXT: [[TMP4:%.*]] = extractelement <4 x float> [[TMP1]], i32 2
+; AVX512-NEXT: [[M2:%.*]] = tail call reassoc nsz float @llvm.minnum.f32(float [[TMP4]], float [[M1]])
+; AVX512-NEXT: [[TMP5:%.*]] = extractelement <4 x float> [[TMP1]], i32 3
+; AVX512-NEXT: [[M3:%.*]] = tail call reassoc nsz float @llvm.minnum.f32(float [[TMP5]], float [[M2]])
+; AVX512-NEXT: ret float [[M3]]
+;
+; AVX256-LABEL: @reduction_v4f32_wrong_fmf(
+; AVX256-NEXT: [[TMP1:%.*]] = load <4 x float>, ptr [[P:%.*]], align 4
+; AVX256-NEXT: [[TMP2:%.*]] = extractelement <4 x float> [[TMP1]], i32 0
+; AVX256-NEXT: [[TMP3:%.*]] = extractelement <4 x float> [[TMP1]], i32 1
+; AVX256-NEXT: [[M1:%.*]] = tail call reassoc nsz float @llvm.minnum.f32(float [[TMP3]], float [[TMP2]])
+; AVX256-NEXT: [[TMP4:%.*]] = extractelement <4 x float> [[TMP1]], i32 2
+; AVX256-NEXT: [[M2:%.*]] = tail call reassoc nsz float @llvm.minnum.f32(float [[TMP4]], float [[M1]])
+; AVX256-NEXT: [[TMP5:%.*]] = extractelement <4 x float> [[TMP1]], i32 3
+; AVX256-NEXT: [[M3:%.*]] = tail call reassoc nsz float @llvm.minnum.f32(float [[TMP5]], float [[M2]])
+; AVX256-NEXT: ret float [[M3]]
;
%g1 = getelementptr inbounds float, ptr %p, i64 1
%g2 = getelementptr inbounds float, ptr %p, i64 2
@@ -473,18 +594,77 @@ define double @reduction_v4f64_fast(ptr %p) {
; Negative test - must have nnan.
define double @reduction_v4f64_not_fast(ptr %p) {
-; CHECK-LABEL: @reduction_v4f64_not_fast(
-; CHECK-NEXT: [[G1:%.*]] = getelementptr inbounds double, ptr [[P:%.*]], i64 1
-; CHECK-NEXT: [[G2:%.*]] = getelementptr inbounds double, ptr [[P]], i64 2
-; CHECK-NEXT: [[G3:%.*]] = getelementptr inbounds double, ptr [[P]], i64 3
-; CHECK-NEXT: [[T0:%.*]] = load double, ptr [[P]], align 4
-; CHECK-NEXT: [[T1:%.*]] = load double, ptr [[G1]], align 4
-; CHECK-NEXT: [[T2:%.*]] = load double, ptr [[G2]], align 4
-; CHECK-NEXT: [[T3:%.*]] = load double, ptr [[G3]], align 4
-; CHECK-NEXT: [[M1:%.*]] = tail call double @llvm.minnum.f64(double [[T1]], double [[T0]])
-; CHECK-NEXT: [[M2:%.*]] = tail call double @llvm.minnum.f64(double [[T2]], double [[M1]])
-; CHECK-NEXT: [[M3:%.*]] = tail call double @llvm.minnum.f64(double [[T3]], double [[M2]])
-; CHECK-NEXT: ret double [[M3]]
+; SSE-LABEL: @reduction_v4f64_not_fast(
+; SSE-NEXT: [[TMP1:%.*]] = load <4 x double>, ptr [[P:%.*]], align 4
+; SSE-NEXT: [[TMP2:%.*]] = extractelement <4 x double> [[TMP1]], i32 0
+; SSE-NEXT: [[TMP3:%.*]] = extractelement <4 x double> [[TMP1]], i32 1
+; SSE-NEXT: [[M1:%.*]] = tail call double @llvm.minnum.f64(double [[TMP3]], double [[TMP2]])
+; SSE-NEXT: [[TMP4:%.*]] = extractelement <4 x double> [[TMP1]], i32 2
+; SSE-NEXT: [[M2:%.*]] = tail call double @llvm.minnum.f64(double [[TMP4]], double [[M1]])
+; SSE-NEXT: [[TMP5:%.*]] = extractelement <4 x double> [[TMP1]], i32 3
+; SSE-NEXT: [[M3:%.*]] = tail call double @llvm.minnum.f64(double [[TMP5]], double [[M2]])
+; SSE-NEXT: ret double [[M3]]
+;
+; COREI7-LABEL: @reduction_v4f64_not_fast(
+; COREI7-NEXT: [[G1:%.*]] = getelementptr inbounds double, ptr [[P:%.*]], i64 1
+; COREI7-NEXT: [[G2:%.*]] = getelementptr inbounds double, ptr [[P]], i64 2
+; COREI7-NEXT: [[G3:%.*]] = getelementptr inbounds double, ptr [[P]], i64 3
+; COREI7-NEXT: [[T0:%.*]] = load double, ptr [[P]], align 4
+; COREI7-NEXT: [[T1:%.*]] = load double, ptr [[G1]], align 4
+; COREI7-NEXT: [[T2:%.*]] = load double, ptr [[G2]], align 4
+; COREI7-NEXT: [[T3:%.*]] = load double, ptr [[G3]], align 4
+; COREI7-NEXT: [[M1:%.*]] = tail call double @llvm.minnum.f64(double [[T1]], double [[T0]])
+; COREI7-NEXT: [[M2:%.*]] = tail call double @llvm.minnum.f64(double [[T2]], double [[M1]])
+; COREI7-NEXT: [[M3:%.*]] = tail call double @llvm.minnum.f64(double [[T3]], double [[M2]])
+; COREI7-NEXT: ret double [[M3]]
+;
+; BDVER1-LABEL: @reduction_v4f64_not_fast(
+; BDVER1-NEXT: [[G1:%.*]] = getelementptr inbounds double, ptr [[P:%.*]], i64 1
+; BDVER1-NEXT: [[G2:%.*]] = getelementptr inbounds double, ptr [[P]], i64 2
+; BDVER1-NEXT: [[G3:%.*]] = getelementptr inbounds double, ptr [[P]], i64 3
+; BDVER1-NEXT: [[T0:%.*]] = load double, ptr [[P]], align 4
+; BDVER1-NEXT: [[T1:%.*]] = load double, ptr [[G1]], align 4
+; BDVER1-NEXT: [[T2:%.*]] = load double, ptr [[G2]], align 4
+; BDVER1-NEXT: [[T3:%.*]] = load double, ptr [[G3]], align 4
+; BDVER1-NEXT: [[M1:%.*]] = tail call double @llvm.minnum.f64(double [[T1]], double [[T0]])
+; BDVER1-NEXT: [[M2:%.*]] = tail call double @llvm.minnum.f64(double [[T2]], double [[M1]])
+; BDVER1-NEXT: [[M3:%.*]] = tail call double @llvm.minnum.f64(double [[T3]], double [[M2]])
+; BDVER1-NEXT: ret double [[M3]]
+;
+; AVX2-LABEL: @reduction_v4f64_not_fast(
+; AVX2-NEXT: [[G1:%.*]] = getelementptr inbounds double, ptr [[P:%.*]], i64 1
+; AVX2-NEXT: [[G2:%.*]] = getelementptr inbounds double, ptr [[P]], i64 2
+; AVX2-NEXT: [[G3:%.*]] = getelementptr inbounds double, ptr [[P]], i64 3
+; AVX2-NEXT: [[T0:%.*]] = load double, ptr [[P]], align 4
+; AVX2-NEXT: [[T1:%.*]] = load double, ptr [[G1]], align 4
+; AVX2-NEXT: [[T2:%.*]] = load double, ptr [[G2]], align 4
+; AVX2-NEXT: [[T3:%.*]] = load double, ptr [[G3]], align 4
+; AVX2-NEXT: [[M1:%.*]] = tail call double @llvm.minnum.f64(double [[T1]], double [[T0]])
+; AVX2-NEXT: [[M2:%.*]] = tail call double @llvm.minnum.f64(double [[T2]], double [[M1]])
+; AVX2-NEXT: [[M3:%.*]] = tail call double @llvm.minnum.f64(double [[T3]], double [[M2]])
+; AVX2-NEXT: ret double [[M3]]
+;
+; AVX512-LABEL: @reduction_v4f64_not_fast(
+; AVX512-NEXT: [[TMP1:%.*]] = load <4 x double>, ptr [[P:%.*]], align 4
+; AVX512-NEXT: [[TMP2:%.*]] = extractelement <4 x double> [[TMP1]], i32 0
+; AVX512-NEXT: [[TMP3:%.*]] = extractelement <4 x double> [[TMP1]], i32 1
+; AVX512-NEXT: [[M1:%.*]] = tail call double @llvm.minnum.f64(double [[TMP3]], double [[TMP2]])
+; AVX512-NEXT: [[TMP4:%.*]] = extractelement <4 x double> [[TMP1]], i32 2
+; AVX512-NEXT: [[M2:%.*]] = tail call double @llvm.minnum.f64(double [[TMP4]], double [[M1]])
+; AVX512-NEXT: [[TMP5:%.*]] = extractelement <4 x double> [[TMP1]], i32 3
+; AVX512-NEXT: [[M3:%.*]] = tail call double @llvm.minnum.f64(double [[TMP5]], double [[M2]])
+; AVX512-NEXT: ret double [[M3]]
+;
+; AVX256-LABEL: @reduction_v4f64_not_fast(
+; AVX256-NEXT: [[TMP1:%.*]] = load <4 x double>, ptr [[P:%.*]], align 4
+; AVX256-NEXT: [[TMP2:%.*]] = extractelement <4 x double> [[TMP1]], i32 0
+; AVX256-NEXT: [[TMP3:%.*]] = extractelement <4 x double> [[TMP1]], i32 1
+; AVX256-NEXT: [[M1:%.*]] = tail call double @llvm.minnum.f64(double [[TMP3]], double [[TMP2]])
+; AVX256-NEXT: [[TMP4:%.*]] = extractelement <4 x double> [[TMP1]], i32 2
+; AVX256-NEXT: [[M2:%.*]] = tail call double @llvm.minnum.f64(double [[TMP4]], double [[M1]])
+; AVX256-NEXT: [[TMP5:%.*]] = extractelement <4 x double> [[TMP1]], i32 3
+; AVX256-NEXT: [[M3:%.*]] = tail call double @llvm.minnum.f64(double [[TMP5]], double [[M2]])
+; AVX256-NEXT: ret double [[M3]]
;
%g1 = getelementptr inbounds double, ptr %p, i64 1
%g2 = getelementptr inbounds double, ptr %p, i64 2
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/horizontal-list.ll b/llvm/test/Transforms/SLPVectorizer/X86/horizontal-list.ll
index 5cdbedb7c6dad..4e434a61e1f1c 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/horizontal-list.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/horizontal-list.ll
@@ -914,16 +914,14 @@ define float @extra_args_no_fast(ptr %x, float %a, float %b) {
; THRESHOLD-LABEL: @extra_args_no_fast(
; THRESHOLD-NEXT: [[ADDC:%.*]] = fadd fast float [[B:%.*]], 3.000000e+00
; THRESHOLD-NEXT: [[ADD:%.*]] = fadd fast float [[A:%.*]], [[ADDC]]
-; THRESHOLD-NEXT: [[ARRAYIDX3:%.*]] = getelementptr inbounds float, ptr [[X:%.*]], i64 1
-; THRESHOLD-NEXT: [[ARRAYIDX3_1:%.*]] = getelementptr inbounds float, ptr [[X]], i64 2
-; THRESHOLD-NEXT: [[ARRAYIDX3_2:%.*]] = getelementptr inbounds float, ptr [[X]], i64 3
-; THRESHOLD-NEXT: [[T0:%.*]] = load float, ptr [[X]], align 4
-; THRESHOLD-NEXT: [[T1:%.*]] = load float, ptr [[ARRAYIDX3]], align 4
-; THRESHOLD-NEXT: [[T2:%.*]] = load float, ptr [[ARRAYIDX3_1]], align 4
-; THRESHOLD-NEXT: [[T3:%.*]] = load float, ptr [[ARRAYIDX3_2]], align 4
+; THRESHOLD-NEXT: [[TMP1:%.*]] = load <4 x float>, ptr [[X:%.*]], align 4
+; THRESHOLD-NEXT: [[T0:%.*]] = extractelement <4 x float> [[TMP1]], i32 0
; THRESHOLD-NEXT: [[ADD1:%.*]] = fadd fast float [[T0]], [[ADD]]
+; THRESHOLD-NEXT: [[T1:%.*]] = extractelement <4 x float> [[TMP1]], i32 1
; THRESHOLD-NEXT: [[ADD4:%.*]] = fadd fast float [[T1]], [[ADD1]]
+; THRESHOLD-NEXT: [[T2:%.*]] = extractelement <4 x float> [[TMP1]], i32 2
; THRESHOLD-NEXT: [[ADD4_1:%.*]] = fadd float [[T2]], [[ADD4]]
+; THRESHOLD-NEXT: [[T3:%.*]] = extractelement <4 x float> [[TMP1]], i32 3
; THRESHOLD-NEXT: [[ADD4_2:%.*]] = fadd fast float [[T3]], [[ADD4_1]]
; THRESHOLD-NEXT: [[ADD5:%.*]] = fadd fast float [[ADD4_2]], [[A]]
; THRESHOLD-NEXT: ret float [[ADD5]]
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/phi.ll b/llvm/test/Transforms/SLPVectorizer/X86/phi.ll
index 17ae33652b6d8..caae1e3dc7da8 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/phi.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/phi.ll
@@ -136,45 +136,47 @@ for.end: ; preds = %for.body
define float @foo3(ptr nocapture readonly %A) #0 {
; CHECK-LABEL: @foo3(
; CHECK-NEXT: entry:
-; CHECK-NEXT: [[ARRAYIDX1:%.*]] = getelementptr inbounds float, ptr [[A:%.*]], i64 1
-; CHECK-NEXT: [[TMP0:%.*]] = load <2 x float>, ptr [[A]], align 4
-; CHECK-NEXT: [[TMP1:%.*]] = load <4 x float>, ptr [[ARRAYIDX1]], align 4
-; CHECK-NEXT: [[TMP2:%.*]] = extractelement <2 x float> [[TMP0]], i32 0
+; CHECK-NEXT: [[TMP1:%.*]] = load <4 x float>, ptr [[ARRAYIDX1:%.*]], align 4
+; CHECK-NEXT: [[ARRAYIDX4:%.*]] = getelementptr inbounds float, ptr [[ARRAYIDX1]], i64 4
+; CHECK-NEXT: [[TMP2:%.*]] = load float, ptr [[ARRAYIDX4]], align 4
+; CHECK-NEXT: [[TMP0:%.*]] = shufflevector <4 x float> [[TMP1]], <4 x float> poison, <2 x i32> <i32 0, i32 1>
; CHECK-NEXT: br label [[FOR_BODY:%.*]]
; CHECK: for.body:
; CHECK-NEXT: [[INDVARS_IV:%.*]] = phi i64 [ 0, [[ENTRY:%.*]] ], [ [[INDVARS_IV_NEXT:%.*]], [[FOR_BODY]] ]
; CHECK-NEXT: [[R_052:%.*]] = phi float [ [[TMP2]], [[ENTRY]] ], [ [[ADD6:%.*]], [[FOR_BODY]] ]
; CHECK-NEXT: [[TMP3:%.*]] = phi <4 x float> [ [[TMP1]], [[ENTRY]] ], [ [[TMP15:%.*]], [[FOR_BODY]] ]
; CHECK-NEXT: [[TMP4:%.*]] = phi <2 x float> [ [[TMP0]], [[ENTRY]] ], [ [[TMP7:%.*]], [[FOR_BODY]] ]
-; CHECK-NEXT: [[TMP5:%.*]] = extractelement <2 x float> [[TMP4]], i32 0
-; CHECK-NEXT: [[MUL:%.*]] = fmul float [[TMP5]], 7.000000e+00
-; CHECK-NEXT: [[ADD6]] = fadd float [[R_052]], [[MUL]]
; CHECK-NEXT: [[TMP6:%.*]] = add nsw i64 [[INDVARS_IV]], 2
-; CHECK-NEXT: [[ARRAYIDX14:%.*]] = getelementptr inbounds float, ptr [[A]], i64 [[TMP6]]
+; CHECK-NEXT: [[ARRAYIDX14:%.*]] = getelementptr inbounds float, ptr [[ARRAYIDX1]], i64 [[TMP6]]
+; CHECK-NEXT: [[TMP9:%.*]] = load float, ptr [[ARRAYIDX14]], align 4
; CHECK-NEXT: [[INDVARS_IV_NEXT]] = add nuw nsw i64 [[INDVARS_IV]], 3
-; CHECK-NEXT: [[ARRAYIDX19:%.*]] = getelementptr inbounds float, ptr [[A]], i64 [[INDVARS_IV_NEXT]]
-; CHECK-NEXT: [[TMP8:%.*]] = load <2 x float>, ptr [[ARRAYIDX14]], align 4
+; CHECK-NEXT: [[ARRAYIDX19:%.*]] = getelementptr inbounds float, ptr [[ARRAYIDX1]], i64 [[INDVARS_IV_NEXT]]
+; CHECK-NEXT: [[TMP11:%.*]] = add nsw i64 [[INDVARS_IV]], 4
+; CHECK-NEXT: [[ARRAYIDX24:%.*]] = getelementptr inbounds float, ptr [[ARRAYIDX1]], i64 [[TMP11]]
+; CHECK-NEXT: [[TMP8:%.*]] = load float, ptr [[ARRAYIDX24]], align 4
; CHECK-NEXT: [[TMP7]] = load <2 x float>, ptr [[ARRAYIDX19]], align 4
-; CHECK-NEXT: [[TMP9:%.*]] = shufflevector <2 x float> [[TMP8]], <2 x float> poison, <4 x i32> <i32 poison, i32 0, i32 1, i32 poison>
-; CHECK-NEXT: [[TMP10:%.*]] = shufflevector <2 x float> [[TMP4]], <2 x float> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
-; CHECK-NEXT: [[TMP11:%.*]] = shufflevector <4 x float> [[TMP9]], <4 x float> [[TMP10]], <4 x i32> <i32 5, i32 1, i32 2, i32 poison>
+; CHECK-NEXT: [[TMP10:%.*]] = insertelement <4 x float> poison, float [[TMP9]], i32 2
; CHECK-NEXT: [[TMP12:%.*]] = shufflevector <2 x float> [[TMP7]], <2 x float> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
-; CHECK-NEXT: [[TMP13:%.*]] = shufflevector <4 x float> [[TMP11]], <4 x float> [[TMP12]], <4 x i32> <i32 0, i32 1, i32 2, i32 5>
-; CHECK-NEXT: [[TMP14:%.*]] = fmul <4 x float> [[TMP13]], <float 8.000000e+00, float 9.000000e+00, float 1.000000e+01, float 1.100000e+01>
+; CHECK-NEXT: [[TMP21:%.*]] = shufflevector <4 x float> [[TMP10]], <4 x float> [[TMP12]], <4 x i32> <i32 poison, i32 poison, i32 2, i32 4>
+; CHECK-NEXT: [[TMP13:%.*]] = shufflevector <2 x float> [[TMP4]], <2 x float> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
+; CHECK-NEXT: [[TMP22:%.*]] = shufflevector <4 x float> [[TMP21]], <4 x float> [[TMP13]], <4 x i32> <i32 4, i32 5, i32 2, i32 3>
+; CHECK-NEXT: [[TMP14:%.*]] = fmul <4 x float> [[TMP22]], <float 7.000000e+00, float 8.000000e+00, float 9.000000e+00, float 1.000000e+01>
; CHECK-NEXT: [[TMP15]] = fadd <4 x float> [[TMP3]], [[TMP14]]
+; CHECK-NEXT: [[MUL25:%.*]] = fmul float [[TMP8]], 1.100000e+01
+; CHECK-NEXT: [[ADD6]] = fadd float [[R_052]], [[MUL25]]
; CHECK-NEXT: [[TMP16:%.*]] = trunc i64 [[INDVARS_IV_NEXT]] to i32
; CHECK-NEXT: [[CMP:%.*]] = icmp slt i32 [[TMP16]], 121
; CHECK-NEXT: br i1 [[CMP]], label [[FOR_BODY]], label [[FOR_END:%.*]]
; CHECK: for.end:
; CHECK-NEXT: [[TMP17:%.*]] = extractelement <4 x float> [[TMP15]], i32 0
-; CHECK-NEXT: [[ADD28:%.*]] = fadd float [[ADD6]], [[TMP17]]
; CHECK-NEXT: [[TMP18:%.*]] = extractelement <4 x float> [[TMP15]], i32 1
-; CHECK-NEXT: [[ADD29:%.*]] = fadd float [[ADD28]], [[TMP18]]
+; CHECK-NEXT: [[ADD29:%.*]] = fadd float [[TMP17]], [[TMP18]]
; CHECK-NEXT: [[TMP19:%.*]] = extractelement <4 x float> [[TMP15]], i32 2
; CHECK-NEXT: [[ADD30:%.*]] = fadd float [[ADD29]], [[TMP19]]
; CHECK-NEXT: [[TMP20:%.*]] = extractelement <4 x float> [[TMP15]], i32 3
; CHECK-NEXT: [[ADD31:%.*]] = fadd float [[ADD30]], [[TMP20]]
-; CHECK-NEXT: ret float [[ADD31]]
+; CHECK-NEXT: [[ADD32:%.*]] = fadd float [[ADD31]], [[ADD6]]
+; CHECK-NEXT: ret float [[ADD32]]
;
entry:
%0 = load float, ptr %A, align 4
@@ -237,18 +239,18 @@ define float @sort_phi_type(ptr nocapture readonly %A) {
; CHECK: for.body:
; CHECK-NEXT: [[INDVARS_IV:%.*]] = phi i64 [ 0, [[ENTRY:%.*]] ], [ [[INDVARS_IV_NEXT:%.*]], [[FOR_BODY]] ]
; CHECK-NEXT: [[TMP0:%.*]] = phi <4 x float> [ splat (float 1.000000e+01), [[ENTRY]] ], [ [[TMP2:%.*]], [[FOR_BODY]] ]
-; CHECK-NEXT: [[TMP1:%.*]] = shufflevector <4 x float> [[TMP0]], <4 x float> poison, <4 x i32> <i32 0, i32 1, i32 3, i32 2>
-; CHECK-NEXT: [[TMP2]] = fmul <4 x float> [[TMP1]], <float 8.000000e+00, float 9.000000e+00, float 1.000000e+02, float 1.110000e+02>
+; CHECK-NEXT: [[TMP1:%.*]] = fmul <4 x float> [[TMP0]], <float 8.000000e+00, float 9.000000e+00, float 1.000000e+02, float 1.110000e+02>
; CHECK-NEXT: [[INDVARS_IV_NEXT]] = add nsw i64 [[INDVARS_IV]], 4
; CHECK-NEXT: [[CMP:%.*]] = icmp slt i64 [[INDVARS_IV_NEXT]], 128
+; CHECK-NEXT: [[TMP2]] = shufflevector <4 x float> [[TMP1]], <4 x float> poison, <4 x i32> <i32 0, i32 1, i32 3, i32 2>
; CHECK-NEXT: br i1 [[CMP]], label [[FOR_BODY]], label [[FOR_END:%.*]]
; CHECK: for.end:
-; CHECK-NEXT: [[TMP3:%.*]] = extractelement <4 x float> [[TMP2]], i32 0
-; CHECK-NEXT: [[TMP4:%.*]] = extractelement <4 x float> [[TMP2]], i32 1
+; CHECK-NEXT: [[TMP3:%.*]] = extractelement <4 x float> [[TMP1]], i32 0
+; CHECK-NEXT: [[TMP4:%.*]] = extractelement <4 x float> [[TMP1]], i32 1
; CHECK-NEXT: [[ADD29:%.*]] = fadd float [[TMP3]], [[TMP4]]
-; CHECK-NEXT: [[TMP5:%.*]] = extractelement <4 x float> [[TMP2]], i32 2
+; CHECK-NEXT: [[TMP5:%.*]] = extractelement <4 x float> [[TMP1]], i32 2
; CHECK-NEXT: [[ADD30:%.*]] = fadd float [[ADD29]], [[TMP5]]
-; CHECK-NEXT: [[TMP6:%.*]] = extractelement <4 x float> [[TMP2]], i32 3
+; CHECK-NEXT: [[TMP6:%.*]] = extractelement <4 x float> [[TMP1]], i32 3
; CHECK-NEXT: [[ADD31:%.*]] = fadd float [[ADD30]], [[TMP6]]
; CHECK-NEXT: ret float [[ADD31]]
;
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/scatter-vectorize-reorder-non-empty.ll b/llvm/test/Transforms/SLPVectorizer/X86/scatter-vectorize-reorder-non-empty.ll
index 94172cffb0295..80bd8ae07e2e2 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/scatter-vectorize-reorder-non-empty.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/scatter-vectorize-reorder-non-empty.ll
@@ -7,13 +7,16 @@ define double @test01() {
; CHECK-NEXT: [[TMP1:%.*]] = load <2 x i32>, ptr null, align 8
; CHECK-NEXT: [[TMP2:%.*]] = getelementptr double, <2 x ptr> zeroinitializer, <2 x i32> [[TMP1]]
; CHECK-NEXT: [[TMP3:%.*]] = call <2 x double> @llvm.masked.gather.v2f64.v2p0(<2 x ptr> align 8 [[TMP2]], <2 x i1> splat (i1 true), <2 x double> poison)
-; CHECK-NEXT: [[TMP4:%.*]] = shufflevector <2 x double> [[TMP3]], <2 x double> <double 0.000000e+00, double poison>, <2 x i32> <i32 2, i32 0>
-; CHECK-NEXT: [[TMP5:%.*]] = fadd <2 x double> [[TMP4]], [[TMP4]]
-; CHECK-NEXT: [[TMP6:%.*]] = fadd <2 x double> [[TMP3]], [[TMP5]]
-; CHECK-NEXT: [[TMP7:%.*]] = extractelement <2 x double> [[TMP6]], i32 0
-; CHECK-NEXT: [[TMP8:%.*]] = extractelement <2 x double> [[TMP6]], i32 1
-; CHECK-NEXT: [[TMP9:%.*]] = fadd double [[TMP7]], [[TMP8]]
-; CHECK-NEXT: ret double [[TMP9]]
+; CHECK-NEXT: [[TMP4:%.*]] = shufflevector <2 x double> [[TMP3]], <2 x double> poison, <4 x i32> <i32 0, i32 0, i32 0, i32 1>
+; CHECK-NEXT: [[TMP5:%.*]] = extractelement <4 x double> [[TMP4]], i32 0
+; CHECK-NEXT: [[TMP6:%.*]] = shufflevector <2 x double> [[TMP3]], <2 x double> poison, <2 x i32> zeroinitializer
+; CHECK-NEXT: [[TMP7:%.*]] = fadd double [[TMP5]], [[TMP5]]
+; CHECK-NEXT: [[TMP8:%.*]] = extractelement <4 x double> [[TMP4]], i32 3
+; CHECK-NEXT: [[TMP9:%.*]] = fadd double [[TMP8]], [[TMP7]]
+; CHECK-NEXT: [[TMP10:%.*]] = fadd double 0.000000e+00, 0.000000e+00
+; CHECK-NEXT: [[TMP11:%.*]] = fadd double [[TMP5]], [[TMP10]]
+; CHECK-NEXT: [[TMP12:%.*]] = fadd double [[TMP11]], [[TMP9]]
+; CHECK-NEXT: ret double [[TMP12]]
;
%1 = load i32, ptr null, align 8
%2 = load i32, ptr getelementptr inbounds (i32, ptr null, i32 1), align 4
>From c3ddc3f74593c6cd31043c976583f7557edba661 Mon Sep 17 00:00:00 2001
From: Alexey Bataev <a.bataev at outlook.com>
Date: Fri, 20 Feb 2026 17:51:09 -0800
Subject: [PATCH 2/2] Fix formatting
Created using spr 1.3.7
---
llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp | 8 ++++----
1 file changed, 4 insertions(+), 4 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index b1df83e021cb7..044d0d4579c04 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -25349,7 +25349,7 @@ class HorizontalReduction {
}
/// Checks if instruction is associative and can be vectorized.
- enum class ReductionKind {Unordered, Ordered, None};
+ enum class ReductionKind { Unordered, Ordered, None };
ReductionKind RK = ReductionKind::None;
static ReductionKind isVectorizable(RecurKind Kind, Instruction *I,
bool TwoElementReduction = false) {
@@ -25733,9 +25733,9 @@ class HorizontalReduction {
// Also, do not try to reduce const values, if the operation is not
// foldable.
bool IsReducedVal = !EdgeInst || Level > RecursionMaxDepth ||
- getRdxKind(EdgeInst) != RdxKind ||
- IsCmpSelMinMax != isCmpSelMinMax(EdgeInst) ||
- !hasRequiredNumberOfUses(IsCmpSelMinMax, EdgeInst);
+ getRdxKind(EdgeInst) != RdxKind ||
+ IsCmpSelMinMax != isCmpSelMinMax(EdgeInst) ||
+ !hasRequiredNumberOfUses(IsCmpSelMinMax, EdgeInst);
ReductionKind CurrentRK = IsReducedVal
? ReductionKind::None
: isVectorizable(RdxKind, EdgeInst);
More information about the llvm-commits
mailing list