[llvm] 038f596 - [SLP]Cost-based choice of the i1 reduction form

via llvm-commits llvm-commits at lists.llvm.org
Mon Sep 21 05:15:48 PDT 2026


Author: Alexey Bataev
Date: 2026-09-21T08:15:41-04:00
New Revision: 038f5968616a3b466e790084bc72eab62d6a6743

URL: https://github.com/llvm/llvm-project/commit/038f5968616a3b466e790084bc72eab62d6a6743
DIFF: https://github.com/llvm/llvm-project/commit/038f5968616a3b466e790084bc72eab62d6a6743.diff

LOG: [SLP]Cost-based choice of the i1 reduction form

i1 reductions (and/or of <n x i1>, add of zexted <n x i1>) can be
emitted as the plain target reduction or in the bitcast-based form
(bitcast to the scalar int type plus a compare for and/or, plus ctpop
for add). Choose the cheaper form by cost, estimating the bitcast-based
form from its emitted components; the ctpop form is estimated as the
cheaper of the scalar bitcast+ctpop and the vector ctpop on the mask
type.

Fixes #40657

Reviewers: michaelselehov, RKSimon, bababuck

Pull Request: https://github.com/llvm/llvm-project/pull/223163

Added: 
    

Modified: 
    llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
    llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.cpp
    llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.h
    llvm/test/Transforms/PhaseOrdering/X86/vector-reductions.ll
    llvm/test/Transforms/SLPVectorizer/AArch64/externally-used-copyables.ll
    llvm/test/Transforms/SLPVectorizer/AArch64/non-power-of-2-with-adjusted-gathers.ll
    llvm/test/Transforms/SLPVectorizer/AArch64/revec-non-pow2.ll
    llvm/test/Transforms/SLPVectorizer/AMDGPU/reduction-i1-mask.ll
    llvm/test/Transforms/SLPVectorizer/RISCV/remark-zext-incoming-for-neg-icmp.ll
    llvm/test/Transforms/SLPVectorizer/SystemZ/cmp-ptr-minmax.ll
    llvm/test/Transforms/SLPVectorizer/X86/alternate-cmp-swapped-pred.ll
    llvm/test/Transforms/SLPVectorizer/X86/entries-different-vf.ll
    llvm/test/Transforms/SLPVectorizer/X86/fabs-cost-softfp.ll
    llvm/test/Transforms/SLPVectorizer/X86/reduction-logical.ll
    llvm/test/Transforms/SLPVectorizer/X86/reduction2.ll
    llvm/test/Transforms/SLPVectorizer/X86/reorder-vf-to-resize.ll
    llvm/test/Transforms/SLPVectorizer/zext-incoming-for-neg-icmp.ll

Removed: 
    


################################################################################
diff  --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 5d2bf0cd7df31..d35337d4bae44 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -2405,6 +2405,10 @@ class slpvectorizer::BoUpSLP {
                          SmallVectorImpl<Value *> &Op2,
                          OrdersType &ReorderIndices) const;
 
+  /// \returns Cast context for the given graph node.
+  TargetTransformInfo::CastContextHint
+  getCastContextHint(const TreeEntry &TE) const;
+
   ~BoUpSLP();
 
 private:
@@ -2472,10 +2476,6 @@ class slpvectorizer::BoUpSLP {
   /// one.
   Instruction *getRootEntryInstruction(const TreeEntry &Entry) const;
 
-  /// \returns Cast context for the given graph node.
-  TargetTransformInfo::CastContextHint
-  getCastContextHint(const TreeEntry &TE) const;
-
   /// \returns the scale of the given tree entry to the loop iteration.
   /// \p Scalar is the scalar value from the entry, if using the parent for the
   /// external use.
@@ -32004,7 +32004,7 @@ class HorizontalReduction {
                                          V.getCostKind());
       if (!Res)
         std::tie(Res, ResNegated) = emitReduction(
-            Builder, *TTI,
+            Builder, *TTI, V.getCastContextHint(V.getRootNode()),
             BoolReduxWideTy ? BoolReduxWideTy : ReductionRoot->getType());
       Builder.setFastMathFlags(RdxFMF);
       // The reduction result of the all-negated parts is subtracted in the
@@ -32466,9 +32466,11 @@ class HorizontalReduction {
              "Expected floating point types for ordered reduction");
       Builder.SetCurrentDebugLocation(
           cast<Instruction>(ReductionRoot)->getDebugLoc());
-      VectorizedTree = createSingleOp(Builder, *TTI, SuccessRoot, /*Scale=*/1,
-                                      /*IsSigned=*/false, DestTy,
-                                      /*ReducedInTree=*/false, VectorizedTree);
+      VectorizedTree =
+          createSingleOp(Builder, *TTI, V.getCastContextHint(V.getRootNode()),
+                         SuccessRoot, /*Scale=*/1,
+                         /*IsSigned=*/false, DestTy,
+                         /*ReducedInTree=*/false, VectorizedTree);
 
       // Fold trailing scalars [SuccessStart+SuccessWidth, N).
       for (Value *RdxVal :
@@ -32512,8 +32514,9 @@ class HorizontalReduction {
   /// Creates the reduction from the given \p Vec vector value with the given
   /// scale \p Scale and signedness \p IsSigned.
   Value *createSingleOp(IRBuilderBase &Builder, const TargetTransformInfo &TTI,
-                        Value *Vec, unsigned Scale, bool IsSigned, Type *DestTy,
-                        bool ReducedInTree, Value *Start = nullptr) {
+                        TTI::CastContextHint Ctx, Value *Vec, unsigned Scale,
+                        bool IsSigned, Type *DestTy, bool ReducedInTree,
+                        Value *Start = nullptr) {
     Value *Rdx;
     if (ReducedInTree) {
       Rdx = Vec;
@@ -32552,7 +32555,7 @@ class HorizontalReduction {
           Rdx = createOp(Builder, RdxKind, Rdx, SubVec, "rdx.op", ReductionOps);
       }
     } else {
-      Rdx = emitReduction(Vec, Builder, &TTI, DestTy, Start);
+      Rdx = emitReduction(Vec, Builder, &TTI, Ctx, DestTy, Start);
     }
     if (Rdx->getType() != DestTy)
       Rdx = Builder.CreateIntCast(Rdx, DestTy, IsSigned);
@@ -32736,8 +32739,19 @@ class HorizontalReduction {
                                     *TTI, RdxKind, WideVecTy, ReductionRoot,
                                     NarrowedChainInsts, CostKind));
             } else if (RType == RedTy) {
-              VectorCost = TTI->getArithmeticReductionCost(RdxOpcode, VectorTy,
-                                                           FMF, CostKind);
+              if (VectorTy->getElementType()->isIntegerTy(1) &&
+                  ((ScalarTy->isIntegerTy(1) &&
+                    (RdxKind == RecurKind::And || RdxKind == RecurKind::Or)) ||
+                   (RdxKind == RecurKind::Add && !ScalarTy->isIntegerTy(1)))) {
+                VectorCost =
+                    getI1ReductionCost(RdxKind, *TTI, VectorTy, ScalarTy,
+                                       R.getCastContextHint(R.getRootNode()),
+                                       CostKind)
+                        .first;
+              } else {
+                VectorCost = TTI->getArithmeticReductionCost(
+                    RdxOpcode, VectorTy, FMF, CostKind);
+              }
             } else {
               VectorCost = TTI->getExtendedReductionCost(
                   RdxOpcode, !IsSigned, RedTy,
@@ -32880,6 +32894,7 @@ class HorizontalReduction {
   /// combine.
   std::pair<Value *, bool> emitReduction(IRBuilderBase &Builder,
                                          const TargetTransformInfo &TTI,
+                                         TTI::CastContextHint Ctx,
                                          Type *DestTy) {
     Value *ReducedSubTree = nullptr;
     bool ResNegated = false;
@@ -32907,8 +32922,8 @@ class HorizontalReduction {
     // the signs of the operands.
     auto CreateSingleOp = [&](Value *Vec, unsigned Scale, bool IsSigned,
                               bool ReducedInTree, bool Negated) {
-      Value *Rdx = createSingleOp(Builder, TTI, Vec, Scale, IsSigned, DestTy,
-                                  ReducedInTree);
+      Value *Rdx = createSingleOp(Builder, TTI, Ctx, Vec, Scale, IsSigned,
+                                  DestTy, ReducedInTree);
       if (!ReducedSubTree) {
         ReducedSubTree = Rdx;
         ResNegated = Negated;
@@ -33092,27 +33107,66 @@ class HorizontalReduction {
     return {ReducedSubTree, ResNegated};
   }
 
+  /// Emits an i1 reduction in the cheaper of the plain and the bitcast-based
+  /// form. Returns nullptr if not an i1 reduction handled here.
+  Value *emitI1Reduction(Value *VectorizedValue, IRBuilderBase &Builder,
+                         const TargetTransformInfo &TTI,
+                         TTI::CastContextHint Ctx, Type *DestTy, Value *Start) {
+    auto *FTy = cast<FixedVectorType>(VectorizedValue->getType());
+    bool IsAdd = RdxKind == RecurKind::Add &&
+                 DestTy->getScalarType() != FTy->getScalarType();
+    bool IsBoolLogic =
+        (RdxKind == RecurKind::And || RdxKind == RecurKind::Or) &&
+        DestTy->getScalarType() == FTy->getScalarType();
+    if (Start || !FTy->getScalarType()->isIntegerTy(1) ||
+        (!IsAdd && !IsBoolLogic))
+      return nullptr;
+    if (getI1ReductionCost(
+            RdxKind, TTI, FTy, DestTy->getScalarType(), Ctx,
+            getSLPCostKind(Builder.GetInsertBlock()->getParent()))
+            .second) {
+      // Convert vector_reduce_and/or(<n x i1>) to icmp eq/ne(bitcast <n x i1>
+      // to in) and vector_reduce_add(ZExt(<n x i1>)) to
+      // ZExtOrTrunc(ctpop(bitcast <n x i1> to in)).
+      Value *V = Builder.CreateBitCast(
+          VectorizedValue, Builder.getIntNTy(FTy->getNumElements()));
+      ++NumVectorInstructions;
+      switch (RdxKind) {
+      case RecurKind::And:
+        return Builder.CreateICmpEQ(V,
+                                    ConstantInt::getAllOnesValue(V->getType()));
+      case RecurKind::Or:
+        return Builder.CreateIsNotNull(V);
+      case RecurKind::Add:
+        return Builder.CreateUnaryIntrinsic(Intrinsic::ctpop, V);
+      default:
+        llvm_unreachable("Unexpected reduction kind for the i1 bitcast form");
+      }
+    }
+    if (IsAdd) {
+      // Keep the extended vector_reduce_add form.
+      VectorizedValue = Builder.CreateZExt(
+          VectorizedValue,
+          getWidenedType(DestTy->getScalarType(), FTy->getNumElements()));
+      ++NumVectorInstructions;
+    }
+    ++NumVectorInstructions;
+    return createSimpleReduction(Builder, VectorizedValue, RdxKind);
+  }
+
   /// Emit a horizontal reduction of the vectorized value.
   /// If \p Start is non-null, emit an ordered reduction intrinsic that
   /// sequentially accumulates into \p Start (only valid for FAdd/FMulAdd).
   Value *emitReduction(Value *VectorizedValue, IRBuilderBase &Builder,
-                       const TargetTransformInfo *TTI, Type *DestTy,
-                       Value *Start = nullptr) {
+                       const TargetTransformInfo *TTI, TTI::CastContextHint Ctx,
+                       Type *DestTy, Value *Start = nullptr) {
     assert(VectorizedValue && "Need to have a vectorized tree node");
     assert(RdxKind != RecurKind::FMulAdd &&
            "A call to the llvm.fmuladd intrinsic is not handled yet");
 
-    auto *FTy = cast<FixedVectorType>(VectorizedValue->getType());
-    if (!Start && FTy->getScalarType() == Builder.getInt1Ty() &&
-        RdxKind == RecurKind::Add &&
-        DestTy->getScalarType() != FTy->getScalarType()) {
-      // Convert vector_reduce_add(ZExt(<n x i1>)) to
-      // ZExtOrTrunc(ctpop(bitcast <n x i1> to in)).
-      Value *V = Builder.CreateBitCast(
-          VectorizedValue, Builder.getIntNTy(FTy->getNumElements()));
-      ++NumVectorInstructions;
-      return Builder.CreateUnaryIntrinsic(Intrinsic::ctpop, V);
-    }
+    if (Value *Rdx =
+            emitI1Reduction(VectorizedValue, Builder, *TTI, Ctx, DestTy, Start))
+      return Rdx;
     ++NumVectorInstructions;
     if (Start)
       return createOrderedReduction(Builder, RdxKind, VectorizedValue, Start);
@@ -33686,6 +33740,20 @@ bool SLPVectorizerPass::tryToVectorize(
             VecTy, APInt::getAllOnes(getNumElements(VecTy)), /*Insert=*/false,
             /*Extract=*/true, CostKind) +
         TTI.getInstructionCost(Inst, CostKind);
+    // The reduction-only cost cannot price the compared-values tree of i1
+    // and/or reductions, so ties are left to the full analysis.
+    if (Ty->isIntegerTy(1) &&
+        (Kind == RecurKind::And || Kind == RecurKind::Or)) {
+      TTI::CastContextHint Ctx =
+          all_of(Ops, [](Value *Op) { return isa<LoadInst>(Op); })
+              ? TTI::CastContextHint::Normal
+              : TTI::CastContextHint::None;
+      if (getI1ReductionCost(Kind, TTI, cast<FixedVectorType>(VecTy), Ty, Ctx,
+                             CostKind)
+              .first > ScalarCost)
+        return false;
+      return HorRdx.tryToReduce(R, *DL, &TTI, *TLI, AC, *DT) != nullptr;
+    }
     InstructionCost RedCost;
     switch (::getRdxKind(Inst)) {
     case RecurKind::Add:

diff  --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.cpp
index a8259d158cf19..16f9b1adf893f 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.cpp
@@ -19,6 +19,7 @@
 #include "llvm/IR/DerivedTypes.h"
 #include "llvm/IR/Instructions.h"
 #include "llvm/IR/IntrinsicInst.h"
+#include "llvm/IR/Intrinsics.h"
 #include "llvm/IR/Operator.h"
 #include "llvm/IR/PatternMatch.h"
 #include "llvm/IR/Type.h"
@@ -298,11 +299,74 @@ InstructionCost getBoolReduxBitcastCmpCost(const TargetTransformInfo &TTI,
                               TruncI) +
          TTI.getCastInstrCost(Instruction::BitCast, IntTy, I1VecTy,
                               TTI.getCastContextHint(TruncI), CostKind) +
-         TTI.getCmpSelInstrCost(Instruction::ICmp, IntTy, /*CondTy=*/nullptr,
-                                RdxKind == RecurKind::And ? CmpInst::ICMP_EQ
-                                                          : CmpInst::ICMP_NE,
-                                CostKind, TTI.getOperandInfo(Root),
-                                TTI.getOperandInfo(CmpRHS), CmpI);
+         TTI.getCmpSelInstrCost(
+             Instruction::ICmp, IntTy, CmpInst::makeCmpResultType(IntTy),
+             RdxKind == RecurKind::And ? CmpInst::ICMP_EQ : CmpInst::ICMP_NE,
+             CostKind, TTI.getOperandInfo(Root), TTI.getOperandInfo(CmpRHS),
+             CmpI);
+}
+
+static InstructionCost
+getBoolLogicRdxBitcastCost(RecurKind Kind, const TargetTransformInfo &TTI,
+                           FixedVectorType *VectorTy, TTI::CastContextHint Ctx,
+                           TTI::TargetCostKind CostKind) {
+  assert((Kind == RecurKind::And || Kind == RecurKind::Or) &&
+         VectorTy->getElementType()->isIntegerTy(1) &&
+         "Expected and/or reduction of i1");
+  auto *IntTy =
+      IntegerType::get(VectorTy->getContext(), getNumElements(VectorTy));
+  CmpInst::Predicate Pred =
+      Kind == RecurKind::And ? CmpInst::ICMP_EQ : CmpInst::ICMP_NE;
+  // The compare is against the all-ones (and) or zero (or) constant.
+  return TTI.getCastInstrCost(Instruction::BitCast, IntTy, VectorTy, Ctx,
+                              CostKind) +
+         TTI.getCmpSelInstrCost(Instruction::ICmp, IntTy,
+                                CmpInst::makeCmpResultType(IntTy), Pred,
+                                CostKind, /*Op1Info=*/{},
+                                {TTI::OK_UniformConstantValue, TTI::OP_None});
+}
+
+std::pair<InstructionCost, bool>
+getI1ReductionCost(RecurKind Kind, const TargetTransformInfo &TTI,
+                   FixedVectorType *VectorTy, Type *ScalarTy,
+                   TTI::CastContextHint Ctx, TTI::TargetCostKind CostKind) {
+  unsigned RdxOpcode = RecurrenceDescriptor::getOpcode(Kind);
+  if (Kind == RecurKind::And || Kind == RecurKind::Or) {
+    InstructionCost RdxCost = TTI.getArithmeticReductionCost(
+        RdxOpcode, VectorTy, std::nullopt, CostKind);
+    InstructionCost BitcastCost =
+        getBoolLogicRdxBitcastCost(Kind, TTI, VectorTy, Ctx, CostKind);
+    return {std::min(RdxCost, BitcastCost), BitcastCost < RdxCost};
+  }
+  assert(Kind == RecurKind::Add && !ScalarTy->isIntegerTy(1) &&
+         "Expected add reduction of zexted i1 values");
+  // The bitcast+ctpop form is estimated as the cheaper of the extended
+  // reduction cost, which models it for the zexted i1 add reduction, and the
+  // explicitly priced components, including the cast of the ctpop result to
+  // the destination type.
+  auto *IntTy =
+      IntegerType::get(VectorTy->getContext(), getNumElements(VectorTy));
+  InstructionCost ExplicitCost =
+      TTI.getCastInstrCost(Instruction::BitCast, IntTy, VectorTy, Ctx,
+                           CostKind) +
+      TTI.getIntrinsicInstrCost(
+          IntrinsicCostAttributes(Intrinsic::ctpop, IntTy, {IntTy}), CostKind);
+  if (IntTy != ScalarTy)
+    ExplicitCost += TTI.getCastInstrCost(IntTy->getBitWidth() <
+                                                 ScalarTy->getIntegerBitWidth()
+                                             ? Instruction::ZExt
+                                             : Instruction::Trunc,
+                                         ScalarTy, IntTy, Ctx, CostKind);
+  InstructionCost CtpopCost = std::min(
+      TTI.getExtendedReductionCost(RdxOpcode, /*IsUnsigned=*/true, ScalarTy,
+                                   VectorTy, std::nullopt, CostKind),
+      ExplicitCost);
+  // The plain form is the zext to the wide vector type plus the reduction.
+  auto *ExtTy = VectorType::get(ScalarTy, VectorTy);
+  InstructionCost ExtRdxCost =
+      TTI.getCastInstrCost(Instruction::ZExt, ExtTy, VectorTy, Ctx, CostKind) +
+      TTI.getArithmeticReductionCost(RdxOpcode, ExtTy, std::nullopt, CostKind);
+  return {std::min(ExtRdxCost, CtpopCost), CtpopCost <= ExtRdxCost};
 }
 
 InstructionCost getBitPackCost(const TargetTransformInfo &TTI,

diff  --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.h b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.h
index a2956043070de..681204b8ae76a 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.h
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.h
@@ -135,6 +135,17 @@ InstructionCost getBitPackCost(const TargetTransformInfo &TTI,
                                const TargetLibraryInfo *TLI,
                                const Instruction *CxtI, unsigned &ShiftWidth);
 
+/// i1 reductions can be emitted as the plain target reduction or in the
+/// bitcast-based form (bitcast to a scalar integer type plus a compare for
+/// and/or, plus ctpop for add). Returns the cost of the cheaper form and
+/// whether it is the bitcast-based one. Ties keep the historically default
+/// form: plain for and/or, bitcast-based for add.
+std::pair<InstructionCost, bool>
+getI1ReductionCost(RecurKind Kind, const TargetTransformInfo &TTI,
+                   FixedVectorType *VectorTy, Type *ScalarTy,
+                   TargetTransformInfo::CastContextHint Ctx,
+                   TargetTransformInfo::TargetCostKind CostKind);
+
 } // namespace llvm::slpvectorizer
 
 #endif // LLVM_LIB_TRANSFORMS_VECTORIZE_SLPVECTORIZER_SLPCOSTANALYSIS_H

diff  --git a/llvm/test/Transforms/PhaseOrdering/X86/vector-reductions.ll b/llvm/test/Transforms/PhaseOrdering/X86/vector-reductions.ll
index 78f68c420f6ab..fe21d4039f240 100644
--- a/llvm/test/Transforms/PhaseOrdering/X86/vector-reductions.ll
+++ b/llvm/test/Transforms/PhaseOrdering/X86/vector-reductions.ll
@@ -277,13 +277,12 @@ define i1 @cmp_lt_gt(double %a, double %b, double %c) {
 ; CHECK-NEXT:    [[TMP6:%.*]] = shufflevector <2 x double> [[TMP5]], <2 x double> poison, <2 x i32> zeroinitializer
 ; CHECK-NEXT:    [[TMP7:%.*]] = fdiv <2 x double> [[TMP3]], [[TMP6]]
 ; CHECK-NEXT:    [[TMP8:%.*]] = fcmp uge <2 x double> [[TMP7]], splat (double f0x3EB0C6F7A0B5ED8D)
-; CHECK-NEXT:    [[SHIFT:%.*]] = shufflevector <2 x i1> [[TMP8]], <2 x i1> poison, <2 x i32> <i32 1, i32 poison>
-; CHECK-NEXT:    [[FOLDEXTEXTBINOP:%.*]] = or <2 x i1> [[TMP8]], [[SHIFT]]
+; CHECK-NEXT:    [[TMP11:%.*]] = bitcast <2 x i1> [[TMP8]] to i2
+; CHECK-NEXT:    [[TMP12:%.*]] = icmp ne i2 [[TMP11]], 0
 ; CHECK-NEXT:    [[TMP9:%.*]] = fcmp ule <2 x double> [[TMP7]], splat (double 1.000000e+00)
-; CHECK-NEXT:    [[SHIFT3:%.*]] = shufflevector <2 x i1> [[TMP9]], <2 x i1> poison, <2 x i32> <i32 1, i32 poison>
-; CHECK-NEXT:    [[TMP10:%.*]] = or <2 x i1> [[TMP9]], [[SHIFT3]]
-; CHECK-NEXT:    [[FOLDEXTEXTBINOP4:%.*]] = and <2 x i1> [[FOLDEXTEXTBINOP]], [[TMP10]]
-; CHECK-NEXT:    [[RETVAL_0:%.*]] = extractelement <2 x i1> [[FOLDEXTEXTBINOP4]], i64 0
+; CHECK-NEXT:    [[TMP13:%.*]] = bitcast <2 x i1> [[TMP9]] to i2
+; CHECK-NEXT:    [[TMP10:%.*]] = icmp ne i2 [[TMP13]], 0
+; CHECK-NEXT:    [[RETVAL_0:%.*]] = and i1 [[TMP12]], [[TMP10]]
 ; CHECK-NEXT:    ret i1 [[RETVAL_0]]
 ;
 entry:

diff  --git a/llvm/test/Transforms/SLPVectorizer/AArch64/externally-used-copyables.ll b/llvm/test/Transforms/SLPVectorizer/AArch64/externally-used-copyables.ll
index ab1f6ce5aa544..4df6673208ff9 100644
--- a/llvm/test/Transforms/SLPVectorizer/AArch64/externally-used-copyables.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AArch64/externally-used-copyables.ll
@@ -7,139 +7,141 @@ define void @test(i64 %0, i64 %1, i64 %2, i64 %3, i64 %.sroa.3341.0.copyload, i6
 ; CHECK-NEXT:  [[_LR_PH_PREHEADER:.*]]:
 ; CHECK-NEXT:    [[TMP9:%.*]] = insertelement <2 x i64> poison, i64 [[TMP1]], i64 0
 ; CHECK-NEXT:    [[TMP10:%.*]] = insertelement <2 x i64> [[TMP9]], i64 [[TMP0]], i64 1
-; CHECK-NEXT:    [[TMP13:%.*]] = mul <2 x i64> [[TMP10]], <i64 1, i64 24>
+; CHECK-NEXT:    [[TMP12:%.*]] = mul <2 x i64> [[TMP10]], <i64 1, i64 24>
 ; CHECK-NEXT:    [[TMP17:%.*]] = mul <2 x i64> [[TMP10]], <i64 1, i64 40>
 ; CHECK-NEXT:    [[TMP11:%.*]] = mul i64 [[TMP0]], 48
 ; CHECK-NEXT:    [[TMP14:%.*]] = insertelement <2 x i64> <i64 poison, i64 1>, i64 [[TMP0]], i64 0
-; CHECK-NEXT:    [[TMP15:%.*]] = shufflevector <2 x i64> [[TMP14]], <2 x i64> <i64 -1, i64 poison>, <2 x i32> <i32 2, i32 0>
-; CHECK-NEXT:    [[TMP16:%.*]] = sub <2 x i64> [[TMP14]], [[TMP15]]
-; CHECK-NEXT:    [[TMP20:%.*]] = shl i64 [[TMP0]], 11
-; CHECK-NEXT:    [[TMP21:%.*]] = sub i64 1, [[TMP20]]
-; CHECK-NEXT:    [[TMP28:%.*]] = insertelement <2 x i64> poison, i64 [[TMP0]], i64 1
-; CHECK-NEXT:    [[TMP25:%.*]] = insertelement <2 x i64> [[TMP28]], i64 [[TMP20]], i64 0
-; CHECK-NEXT:    [[TMP22:%.*]] = add <2 x i64> [[TMP25]], <i64 8, i64 1>
-; CHECK-NEXT:    [[TMP23:%.*]] = or <2 x i64> [[TMP25]], <i64 8, i64 1>
-; CHECK-NEXT:    [[TMP33:%.*]] = shufflevector <2 x i64> [[TMP22]], <2 x i64> [[TMP23]], <2 x i32> <i32 0, i32 3>
+; CHECK-NEXT:    [[TMP22:%.*]] = shufflevector <2 x i64> [[TMP14]], <2 x i64> <i64 -1, i64 poison>, <2 x i32> <i32 2, i32 0>
+; CHECK-NEXT:    [[TMP16:%.*]] = sub <2 x i64> [[TMP14]], [[TMP22]]
+; CHECK-NEXT:    [[TMP25:%.*]] = shl i64 [[TMP0]], 11
+; CHECK-NEXT:    [[TMP21:%.*]] = sub i64 1, [[TMP25]]
+; CHECK-NEXT:    [[TMP66:%.*]] = insertelement <2 x i64> poison, i64 [[TMP0]], i64 1
+; CHECK-NEXT:    [[TMP20:%.*]] = insertelement <2 x i64> [[TMP66]], i64 [[TMP25]], i64 0
+; CHECK-NEXT:    [[TMP41:%.*]] = add <2 x i64> [[TMP20]], <i64 8, i64 1>
+; CHECK-NEXT:    [[TMP23:%.*]] = or <2 x i64> [[TMP20]], <i64 8, i64 1>
+; CHECK-NEXT:    [[TMP44:%.*]] = shufflevector <2 x i64> [[TMP41]], <2 x i64> [[TMP23]], <2 x i32> <i32 0, i32 3>
 ; CHECK-NEXT:    [[TMP18:%.*]] = shl i64 [[TMP0]], 1
 ; CHECK-NEXT:    [[TMP19:%.*]] = or i64 [[TMP18]], [[TMP0]]
-; CHECK-NEXT:    [[TMP34:%.*]] = or i64 [[TMP19]], 1
-; CHECK-NEXT:    [[TMP55:%.*]] = mul i64 [[TMP0]], [[TMP0]]
-; CHECK-NEXT:    [[TMP58:%.*]] = insertelement <8 x i64> poison, i64 [[TMP0]], i64 0
-; CHECK-NEXT:    [[TMP30:%.*]] = shufflevector <8 x i64> [[TMP58]], <8 x i64> poison, <8 x i32> zeroinitializer
+; CHECK-NEXT:    [[TMP27:%.*]] = or i64 [[TMP19]], 1
+; CHECK-NEXT:    [[TMP28:%.*]] = mul i64 [[TMP0]], [[TMP0]]
+; CHECK-NEXT:    [[TMP29:%.*]] = insertelement <8 x i64> poison, i64 [[TMP0]], i64 0
+; CHECK-NEXT:    [[TMP30:%.*]] = shufflevector <8 x i64> [[TMP29]], <8 x i64> poison, <8 x i32> zeroinitializer
 ; CHECK-NEXT:    [[TMP31:%.*]] = shufflevector <8 x i64> [[TMP30]], <8 x i64> <i64 poison, i64 1, i64 1, i64 1, i64 poison, i64 poison, i64 poison, i64 poison>, <6 x i32> <i32 0, i32 9, i32 10, i32 11, i32 poison, i32 poison>
 ; CHECK-NEXT:    [[TMP32:%.*]] = shufflevector <8 x i64> [[TMP30]], <8 x i64> <i64 0, i64 0, i64 0, i64 0, i64 poison, i64 1, i64 1, i64 1>, <8 x i32> <i32 8, i32 9, i32 10, i32 11, i32 0, i32 13, i32 14, i32 15>
-; CHECK-NEXT:    [[TMP27:%.*]] = shufflevector <8 x i64> [[TMP30]], <8 x i64> poison, <3 x i32> <i32 0, i32 poison, i32 poison>
-; CHECK-NEXT:    [[TMP62:%.*]] = insertelement <3 x i64> [[TMP27]], i64 [[TMP1]], i64 1
-; CHECK-NEXT:    [[TMP29:%.*]] = insertelement <3 x i64> [[TMP62]], i64 [[DOTNEG1]], i64 2
-; CHECK-NEXT:    [[TMP36:%.*]] = shufflevector <3 x i64> [[TMP29]], <3 x i64> poison, <8 x i32> <i32 0, i32 1, i32 1, i32 1, i32 2, i32 0, i32 0, i32 0>
+; CHECK-NEXT:    [[TMP55:%.*]] = shufflevector <8 x i64> [[TMP30]], <8 x i64> poison, <3 x i32> <i32 0, i32 poison, i32 poison>
+; CHECK-NEXT:    [[TMP34:%.*]] = insertelement <3 x i64> [[TMP55]], i64 [[TMP1]], i64 1
+; CHECK-NEXT:    [[TMP56:%.*]] = insertelement <3 x i64> [[TMP34]], i64 [[DOTNEG1]], i64 2
+; CHECK-NEXT:    [[TMP36:%.*]] = shufflevector <3 x i64> [[TMP56]], <3 x i64> poison, <8 x i32> <i32 0, i32 1, i32 1, i32 1, i32 2, i32 0, i32 0, i32 0>
 ; CHECK-NEXT:    [[TMP37:%.*]] = insertelement <4 x i64> poison, i64 [[TMP2]], i64 0
 ; CHECK-NEXT:    [[TMP38:%.*]] = insertelement <4 x i64> [[TMP37]], i64 [[TMP8]], i64 1
 ; CHECK-NEXT:    [[TMP39:%.*]] = insertelement <4 x i64> [[TMP38]], i64 [[DOTSROA_3308_0_COPYLOAD]], i64 2
 ; CHECK-NEXT:    [[TMP40:%.*]] = insertelement <4 x i64> [[TMP39]], i64 [[TMP0]], i64 3
-; CHECK-NEXT:    [[TMP64:%.*]] = insertelement <2 x i64> poison, i64 [[TMP0]], i64 0
-; CHECK-NEXT:    [[TMP12:%.*]] = shufflevector <2 x i64> [[TMP64]], <2 x i64> poison, <2 x i32> zeroinitializer
-; CHECK-NEXT:    [[TMP41:%.*]] = shufflevector <2 x i64> [[TMP17]], <2 x i64> poison, <8 x i32> <i32 0, i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT:    [[TMP63:%.*]] = insertelement <4 x i64> poison, i64 [[TMP0]], i64 1
-; CHECK-NEXT:    [[TMP43:%.*]] = mul i64 [[TMP0]], [[TMP0]]
-; CHECK-NEXT:    [[TMP44:%.*]] = shufflevector <2 x i64> [[TMP13]], <2 x i64> poison, <32 x i32> <i32 0, i32 poison, i32 poison, i32 poison, i32 poison, i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT:    [[TMP45:%.*]] = shufflevector <32 x i64> [[TMP44]], <32 x i64> poison, <32 x i32> <i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 0, i32 poison, i32 poison, i32 poison, i32 poison, i32 5, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT:    [[TMP46:%.*]] = shufflevector <2 x i64> [[TMP33]], <2 x i64> poison, <6 x i32> <i32 0, i32 1, i32 poison, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT:    [[TMP47:%.*]] = shufflevector <6 x i64> [[TMP31]], <6 x i64> [[TMP46]], <6 x i32> <i32 0, i32 1, i32 2, i32 3, i32 6, i32 7>
-; CHECK-NEXT:    [[TMP48:%.*]] = shufflevector <2 x i64> [[TMP16]], <2 x i64> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
+; CHECK-NEXT:    [[TMP35:%.*]] = insertelement <2 x i64> poison, i64 [[TMP0]], i64 0
+; CHECK-NEXT:    [[TMP33:%.*]] = shufflevector <2 x i64> [[TMP35]], <2 x i64> poison, <2 x i32> zeroinitializer
+; CHECK-NEXT:    [[TMP48:%.*]] = insertelement <2 x i64> [[TMP35]], i64 [[TMP1]], i64 1
+; CHECK-NEXT:    [[TMP59:%.*]] = shufflevector <2 x i64> [[TMP48]], <2 x i64> <i64 poison, i64 1>, <2 x i32> <i32 0, i32 3>
+; CHECK-NEXT:    [[TMP115:%.*]] = insertelement <2 x i64> [[TMP48]], i64 [[TMP2]], i64 1
+; CHECK-NEXT:    [[TMP43:%.*]] = shufflevector <2 x i64> [[TMP14]], <2 x i64> poison, <2 x i32> zeroinitializer
+; CHECK-NEXT:    [[TMP45:%.*]] = shufflevector <2 x i64> [[TMP17]], <2 x i64> poison, <8 x i32> <i32 0, i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT:    [[TMP46:%.*]] = shufflevector <2 x i64> [[TMP12]], <2 x i64> poison, <32 x i32> <i32 0, i32 poison, i32 poison, i32 poison, i32 poison, i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT:    [[TMP47:%.*]] = shufflevector <32 x i64> [[TMP46]], <32 x i64> poison, <32 x i32> <i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 0, i32 poison, i32 poison, i32 poison, i32 poison, i32 5, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT:    [[TMP60:%.*]] = shufflevector <2 x i64> [[TMP44]], <2 x i64> poison, <6 x i32> <i32 0, i32 1, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT:    [[TMP49:%.*]] = shufflevector <6 x i64> [[TMP31]], <6 x i64> [[TMP60]], <6 x i32> <i32 0, i32 1, i32 2, i32 3, i32 6, i32 7>
 ; CHECK-NEXT:    br label %[[DOTLR_PH1977_US:.*]]
 ; CHECK:       [[DOTLR_PH1977_US]]:
 ; CHECK-NEXT:    [[INDVAR37888:%.*]] = phi i64 [ 0, %[[_LR_PH_PREHEADER]] ], [ 1, %[[DOTLR_PH1977_US]] ]
-; CHECK-NEXT:    [[TMP49:%.*]] = shufflevector <6 x i64> [[TMP47]], <6 x i64> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 5, i32 5>
-; CHECK-NEXT:    [[TMP50:%.*]] = mul <8 x i64> [[TMP30]], [[TMP49]]
-; CHECK-NEXT:    [[TMP51:%.*]] = or <8 x i64> [[TMP30]], [[TMP49]]
-; CHECK-NEXT:    [[TMP52:%.*]] = shufflevector <8 x i64> [[TMP50]], <8 x i64> [[TMP51]], <8 x i32> <i32 0, i32 9, i32 10, i32 11, i32 4, i32 5, i32 6, i32 7>
-; CHECK-NEXT:    [[TMP53:%.*]] = mul i64 [[TMP34]], [[TMP0]]
-; CHECK-NEXT:    [[TMP54:%.*]] = mul i64 [[TMP21]], [[TMP0]]
+; CHECK-NEXT:    [[TMP50:%.*]] = shufflevector <6 x i64> [[TMP49]], <6 x i64> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 5, i32 5>
+; CHECK-NEXT:    [[TMP51:%.*]] = mul <8 x i64> [[TMP30]], [[TMP50]]
+; CHECK-NEXT:    [[TMP52:%.*]] = or <8 x i64> [[TMP30]], [[TMP50]]
+; CHECK-NEXT:    [[TMP53:%.*]] = shufflevector <8 x i64> [[TMP51]], <8 x i64> [[TMP52]], <8 x i32> <i32 0, i32 9, i32 10, i32 11, i32 4, i32 5, i32 6, i32 7>
+; CHECK-NEXT:    [[TMP54:%.*]] = mul i64 [[TMP27]], [[TMP0]]
+; CHECK-NEXT:    [[TMP94:%.*]] = mul i64 [[TMP21]], [[TMP0]]
 ; CHECK-NEXT:    [[TMP26:%.*]] = call i64 @llvm.vscale.i64()
 ; CHECK-NEXT:    [[DIFF_CHECK3783:%.*]] = icmp ult i64 [[TMP11]], [[TMP26]]
-; CHECK-NEXT:    [[DIFF_CHECK3790:%.*]] = icmp ult i64 [[INDVAR37888]], [[TMP0]]
-; CHECK-NEXT:    [[DIFF_CHECK3805:%.*]] = icmp ugt i64 [[TMP26]], 1
-; CHECK-NEXT:    [[TMP56:%.*]] = add <8 x i64> [[TMP52]], [[TMP32]]
-; CHECK-NEXT:    [[TMP57:%.*]] = icmp ult <8 x i64> [[TMP56]], [[TMP36]]
-; CHECK-NEXT:    [[TMP24:%.*]] = extractelement <8 x i64> [[TMP52]], i64 5
-; CHECK-NEXT:    [[TMP42:%.*]] = add i64 [[TMP24]], 1
-; CHECK-NEXT:    [[DIFF_CHECK3842:%.*]] = icmp ult i64 [[TMP42]], [[TMP0]]
+; CHECK-NEXT:    [[OP_RDX52:%.*]] = icmp ugt i64 [[TMP26]], 1
+; CHECK-NEXT:    [[TMP57:%.*]] = add <8 x i64> [[TMP53]], [[TMP32]]
+; CHECK-NEXT:    [[TMP58:%.*]] = icmp ult <8 x i64> [[TMP57]], [[TMP36]]
+; CHECK-NEXT:    [[TMP62:%.*]] = extractelement <8 x i64> [[TMP53]], i64 5
+; CHECK-NEXT:    [[TMP42:%.*]] = add i64 [[TMP62]], 1
+; CHECK-NEXT:    [[TMP61:%.*]] = insertelement <2 x i64> poison, i64 [[INDVAR37888]], i64 0
+; CHECK-NEXT:    [[TMP64:%.*]] = insertelement <2 x i64> [[TMP61]], i64 [[TMP42]], i64 1
+; CHECK-NEXT:    [[TMP63:%.*]] = icmp ult <2 x i64> [[TMP64]], [[TMP33]]
 ; CHECK-NEXT:    [[DIFF_CHECK3817:%.*]] = icmp ult i64 [[TMP1]], [[TMP2]]
-; CHECK-NEXT:    [[TMP76:%.*]] = add i64 [[TMP53]], 1
-; CHECK-NEXT:    [[DIFF_CHECK3820:%.*]] = icmp ult i64 [[TMP76]], [[TMP0]]
-; CHECK-NEXT:    [[TMP77:%.*]] = insertelement <4 x i64> [[TMP63]], i64 [[INDVAR37888]], i64 0
-; CHECK-NEXT:    [[TMP78:%.*]] = shufflevector <4 x i64> [[TMP77]], <4 x i64> poison, <4 x i32> <i32 0, i32 1, i32 1, i32 1>
-; CHECK-NEXT:    [[TMP80:%.*]] = shufflevector <4 x i64> [[TMP77]], <4 x i64> <i64 1, i64 poison, i64 poison, i64 poison>, <4 x i32> <i32 4, i32 1, i32 poison, i32 poison>
-; CHECK-NEXT:    [[TMP81:%.*]] = shufflevector <4 x i64> [[TMP80]], <4 x i64> [[TMP48]], <4 x i32> <i32 0, i32 1, i32 4, i32 5>
-; CHECK-NEXT:    [[TMP82:%.*]] = mul <4 x i64> [[TMP78]], [[TMP81]]
-; CHECK-NEXT:    [[TMP84:%.*]] = add i64 [[TMP54]], 1
-; CHECK-NEXT:    [[TMP85:%.*]] = insertelement <32 x i64> [[TMP45]], i64 [[TMP55]], i64 11
-; CHECK-NEXT:    [[TMP86:%.*]] = shufflevector <4 x i64> [[TMP82]], <4 x i64> poison, <32 x i32> <i32 0, i32 1, i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT:    [[TMP89:%.*]] = shufflevector <32 x i64> [[TMP85]], <32 x i64> [[TMP86]], <32 x i32> <i32 32, i32 33, i32 34, i32 35, i32 poison, i32 5, i32 poison, i32 poison, i32 poison, i32 poison, i32 10, i32 11, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT:    [[TMP90:%.*]] = insertelement <32 x i64> [[TMP89]], i64 [[TMP84]], i64 4
-; CHECK-NEXT:    [[TMP100:%.*]] = shufflevector <8 x i64> [[TMP52]], <8 x i64> poison, <32 x i32> <i32 poison, i32 poison, i32 poison, i32 poison, i32 4, i32 5, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT:    [[TMP73:%.*]] = shufflevector <8 x i64> [[TMP52]], <8 x i64> poison, <32 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT:    [[TMP87:%.*]] = shufflevector <32 x i64> [[TMP90]], <32 x i64> [[TMP73]], <32 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 36, i32 37, i32 poison, i32 poison, i32 10, i32 11, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT:    [[TMP98:%.*]] = shl <2 x i64> [[TMP12]], splat (i64 1)
-; CHECK-NEXT:    [[TMP125:%.*]] = shufflevector <2 x i64> [[TMP98]], <2 x i64> poison, <32 x i32> <i32 0, i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT:    [[TMP88:%.*]] = shufflevector <32 x i64> [[TMP87]], <32 x i64> [[TMP125]], <32 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 32, i32 33, i32 10, i32 11, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT:    [[TMP59:%.*]] = insertelement <32 x i64> [[TMP88]], i64 [[TMP54]], i64 12
-; CHECK-NEXT:    [[TMP60:%.*]] = shufflevector <32 x i64> [[TMP59]], <32 x i64> poison, <32 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 5, i32 5, i32 7, i32 5, i32 8, i32 9, i32 10, i32 5, i32 0, i32 1, i32 11, i32 12, i32 5, i32 5, i32 6, i32 7, i32 5, i32 5, i32 7, i32 5, i32 9, i32 10, i32 0, i32 1>
-; CHECK-NEXT:    [[TMP61:%.*]] = shufflevector <32 x i64> [[TMP59]], <32 x i64> poison, <10 x i32> <i32 poison, i32 poison, i32 5, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT:    [[TMP109:%.*]] = insertelement <10 x i64> [[TMP61]], i64 [[TMP0]], i64 0
-; CHECK-NEXT:    [[TMP110:%.*]] = insertelement <10 x i64> [[TMP109]], i64 [[TMP3]], i64 1
-; CHECK-NEXT:    [[TMP111:%.*]] = insertelement <10 x i64> [[TMP110]], i64 [[TMP4]], i64 3
-; CHECK-NEXT:    [[TMP112:%.*]] = insertelement <10 x i64> [[TMP111]], i64 [[INDVAR3788]], i64 4
-; CHECK-NEXT:    [[TMP113:%.*]] = insertelement <10 x i64> [[TMP112]], i64 [[TMP2]], i64 5
-; CHECK-NEXT:    [[TMP114:%.*]] = insertelement <10 x i64> [[TMP113]], i64 [[TMP5]], i64 6
-; CHECK-NEXT:    [[TMP115:%.*]] = insertelement <10 x i64> [[TMP114]], i64 [[TMP6]], i64 7
-; CHECK-NEXT:    [[TMP116:%.*]] = insertelement <10 x i64> [[TMP115]], i64 [[TMP7]], i64 8
-; CHECK-NEXT:    [[TMP70:%.*]] = insertelement <10 x i64> [[TMP116]], i64 [[DOTSROA_3341_0_COPYLOAD]], i64 9
-; CHECK-NEXT:    [[TMP71:%.*]] = shufflevector <10 x i64> [[TMP70]], <10 x i64> poison, <32 x i32> <i32 0, i32 0, i32 0, i32 0, i32 0, i32 1, i32 0, i32 2, i32 3, i32 4, i32 0, i32 0, i32 2, i32 2, i32 0, i32 5, i32 0, i32 0, i32 0, i32 0, i32 6, i32 7, i32 0, i32 2, i32 8, i32 9, i32 0, i32 0, i32 2, i32 0, i32 0, i32 0>
-; CHECK-NEXT:    [[TMP72:%.*]] = icmp ult <32 x i64> [[TMP60]], [[TMP71]]
-; CHECK-NEXT:    [[TMP91:%.*]] = add i64 [[TMP54]], 1
-; CHECK-NEXT:    [[TMP92:%.*]] = shufflevector <8 x i64> [[TMP52]], <8 x i64> poison, <8 x i32> <i32 5, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 4, i32 4>
-; CHECK-NEXT:    [[TMP93:%.*]] = shufflevector <8 x i64> [[TMP92]], <8 x i64> [[TMP41]], <8 x i32> <i32 0, i32 1, i32 2, i32 8, i32 9, i32 5, i32 6, i32 7>
-; CHECK-NEXT:    [[TMP94:%.*]] = shufflevector <2 x i64> [[TMP98]], <2 x i64> poison, <8 x i32> <i32 0, i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT:    [[TMP95:%.*]] = shufflevector <8 x i64> [[TMP93]], <8 x i64> [[TMP94]], <8 x i32> <i32 0, i32 8, i32 9, i32 3, i32 4, i32 5, i32 6, i32 7>
-; CHECK-NEXT:    [[TMP96:%.*]] = insertelement <8 x i64> [[TMP95]], i64 [[TMP91]], i64 5
-; CHECK-NEXT:    [[TMP105:%.*]] = shufflevector <8 x i64> [[TMP96]], <8 x i64> poison, <2 x i32> <i32 poison, i32 3>
-; CHECK-NEXT:    [[TMP107:%.*]] = insertelement <2 x i64> [[TMP105]], i64 [[TMP0]], i64 0
-; CHECK-NEXT:    [[TMP99:%.*]] = shufflevector <2 x i64> [[TMP107]], <2 x i64> poison, <8 x i32> <i32 0, i32 1, i32 1, i32 0, i32 0, i32 0, i32 0, i32 1>
-; CHECK-NEXT:    [[TMP83:%.*]] = icmp ult <8 x i64> [[TMP96]], [[TMP99]]
-; CHECK-NEXT:    [[TMP101:%.*]] = shufflevector <8 x i64> [[TMP52]], <8 x i64> poison, <4 x i32> <i32 4, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT:    [[TMP102:%.*]] = insertelement <4 x i64> [[TMP101]], i64 [[TMP1]], i64 1
-; CHECK-NEXT:    [[TMP103:%.*]] = insertelement <4 x i64> [[TMP102]], i64 [[TMP53]], i64 3
-; CHECK-NEXT:    [[TMP104:%.*]] = shufflevector <4 x i64> [[TMP103]], <4 x i64> poison, <4 x i32> <i32 0, i32 1, i32 1, i32 3>
-; CHECK-NEXT:    [[TMP97:%.*]] = icmp ult <4 x i64> [[TMP104]], [[TMP40]]
+; CHECK-NEXT:    [[TMP24:%.*]] = add i64 [[TMP54]], 1
 ; CHECK-NEXT:    [[DIFF_CHECK3930:%.*]] = icmp ult i64 [[TMP24]], [[TMP0]]
-; CHECK-NEXT:    [[TMP106:%.*]] = extractelement <2 x i64> [[TMP98]], i64 0
-; CHECK-NEXT:    [[DIFF_CHECK3932:%.*]] = icmp ult i64 [[TMP106]], [[TMP1]]
-; CHECK-NEXT:    [[OP_RDX50:%.*]] = icmp ult i64 [[TMP1]], [[TMP0]]
-; CHECK-NEXT:    [[TMP126:%.*]] = icmp ult i64 [[INDVAR37888]], [[TMP0]]
-; CHECK-NEXT:    [[DIFF_CHECK3938:%.*]] = icmp ult i64 [[TMP43]], [[TMP0]]
-; CHECK-NEXT:    [[DIFF_CHECK3950:%.*]] = icmp ult i64 [[TMP1]], [[TMP2]]
-; CHECK-NEXT:    [[TMP75:%.*]] = call i1 @llvm.vector.reduce.or.v32i1(<32 x i1> [[TMP72]])
-; CHECK-NEXT:    [[TMP108:%.*]] = call i1 @llvm.vector.reduce.or.v8i1(<8 x i1> [[TMP57]])
-; CHECK-NEXT:    [[TMP127:%.*]] = shufflevector <8 x i1> [[TMP83]], <8 x i1> poison, <4 x i32> <i32 0, i32 1, i32 2, i32 3>
-; CHECK-NEXT:    [[RDX_OP:%.*]] = or <4 x i1> [[TMP127]], [[TMP97]]
-; CHECK-NEXT:    [[TMP128:%.*]] = shufflevector <4 x i1> [[RDX_OP]], <4 x i1> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT:    [[TMP129:%.*]] = shufflevector <8 x i1> [[TMP83]], <8 x i1> [[TMP128]], <8 x i32> <i32 8, i32 9, i32 10, i32 11, i32 4, i32 5, i32 6, i32 7>
-; CHECK-NEXT:    [[TMP130:%.*]] = call i1 @llvm.vector.reduce.or.v8i1(<8 x i1> [[TMP129]])
-; CHECK-NEXT:    [[OP_RDX26:%.*]] = or i1 [[TMP130]], [[DIFF_CHECK3783]]
-; CHECK-NEXT:    [[OP_RDX57:%.*]] = or i1 [[DIFF_CHECK3790]], [[DIFF_CHECK3842]]
-; CHECK-NEXT:    [[OP_RDX58:%.*]] = or i1 [[DIFF_CHECK3817]], [[DIFF_CHECK3820]]
-; CHECK-NEXT:    [[OP_RDX59:%.*]] = or i1 [[DIFF_CHECK3930]], [[DIFF_CHECK3932]]
-; CHECK-NEXT:    [[OP_RDX52:%.*]] = or i1 [[OP_RDX50]], [[TMP126]]
-; CHECK-NEXT:    [[OP_RDX61:%.*]] = or i1 [[DIFF_CHECK3938]], [[DIFF_CHECK3950]]
-; CHECK-NEXT:    [[OP_RDX62:%.*]] = or i1 [[DIFF_CHECK3805]], [[TMP75]]
+; CHECK-NEXT:    [[TMP65:%.*]] = mul <2 x i64> [[TMP16]], [[TMP43]]
+; CHECK-NEXT:    [[TMP78:%.*]] = add i64 [[TMP94]], 1
+; CHECK-NEXT:    [[TMP67:%.*]] = insertelement <32 x i64> [[TMP47]], i64 [[TMP28]], i64 11
+; CHECK-NEXT:    [[TMP68:%.*]] = insertelement <32 x i64> [[TMP67]], i64 [[INDVAR37888]], i64 0
+; CHECK-NEXT:    [[TMP69:%.*]] = shufflevector <8 x i64> [[TMP53]], <8 x i64> poison, <32 x i32> <i32 poison, i32 poison, i32 poison, i32 poison, i32 4, i32 5, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT:    [[TMP70:%.*]] = shufflevector <8 x i64> [[TMP53]], <8 x i64> poison, <32 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT:    [[TMP71:%.*]] = shl <2 x i64> [[TMP33]], splat (i64 1)
+; CHECK-NEXT:    [[TMP72:%.*]] = add i64 [[TMP94]], 1
+; CHECK-NEXT:    [[TMP73:%.*]] = shufflevector <8 x i64> [[TMP53]], <8 x i64> poison, <8 x i32> <i32 5, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 4, i32 4>
+; CHECK-NEXT:    [[TMP74:%.*]] = shufflevector <8 x i64> [[TMP73]], <8 x i64> [[TMP45]], <8 x i32> <i32 0, i32 1, i32 2, i32 8, i32 9, i32 5, i32 6, i32 7>
+; CHECK-NEXT:    [[TMP75:%.*]] = shufflevector <2 x i64> [[TMP71]], <2 x i64> poison, <8 x i32> <i32 0, i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT:    [[TMP76:%.*]] = shufflevector <8 x i64> [[TMP74]], <8 x i64> [[TMP75]], <8 x i32> <i32 0, i32 8, i32 9, i32 3, i32 4, i32 5, i32 6, i32 7>
+; CHECK-NEXT:    [[TMP77:%.*]] = insertelement <8 x i64> [[TMP76]], i64 [[TMP72]], i64 5
+; CHECK-NEXT:    [[TMP87:%.*]] = shufflevector <8 x i64> [[TMP77]], <8 x i64> poison, <2 x i32> <i32 poison, i32 3>
+; CHECK-NEXT:    [[TMP88:%.*]] = insertelement <2 x i64> [[TMP87]], i64 [[TMP0]], i64 0
+; CHECK-NEXT:    [[TMP80:%.*]] = shufflevector <2 x i64> [[TMP88]], <2 x i64> poison, <8 x i32> <i32 0, i32 1, i32 1, i32 0, i32 0, i32 0, i32 0, i32 1>
+; CHECK-NEXT:    [[TMP81:%.*]] = icmp ult <8 x i64> [[TMP77]], [[TMP80]]
+; CHECK-NEXT:    [[TMP82:%.*]] = shufflevector <8 x i64> [[TMP53]], <8 x i64> poison, <4 x i32> <i32 4, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT:    [[TMP83:%.*]] = insertelement <4 x i64> [[TMP82]], i64 [[TMP1]], i64 1
+; CHECK-NEXT:    [[TMP84:%.*]] = insertelement <4 x i64> [[TMP83]], i64 [[TMP54]], i64 3
+; CHECK-NEXT:    [[TMP85:%.*]] = shufflevector <4 x i64> [[TMP84]], <4 x i64> poison, <4 x i32> <i32 0, i32 1, i32 1, i32 3>
+; CHECK-NEXT:    [[TMP86:%.*]] = icmp ult <4 x i64> [[TMP85]], [[TMP40]]
+; CHECK-NEXT:    [[OP_RDX26:%.*]] = icmp ult i64 [[TMP62]], [[TMP0]]
+; CHECK-NEXT:    [[TMP104:%.*]] = extractelement <2 x i64> [[TMP71]], i64 0
+; CHECK-NEXT:    [[OP_RDX57:%.*]] = icmp ult i64 [[TMP104]], [[TMP1]]
+; CHECK-NEXT:    [[TMP107:%.*]] = insertelement <2 x i64> [[TMP9]], i64 [[INDVAR37888]], i64 1
+; CHECK-NEXT:    [[TMP112:%.*]] = icmp ult <2 x i64> [[TMP107]], [[TMP33]]
+; CHECK-NEXT:    [[TMP139:%.*]] = mul <2 x i64> [[TMP48]], [[TMP59]]
+; CHECK-NEXT:    [[TMP91:%.*]] = shufflevector <2 x i64> [[TMP139]], <2 x i64> poison, <32 x i32> <i32 0, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT:    [[TMP92:%.*]] = shufflevector <2 x i64> [[TMP139]], <2 x i64> poison, <32 x i32> <i32 0, i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT:    [[TMP93:%.*]] = shufflevector <32 x i64> [[TMP68]], <32 x i64> [[TMP92]], <32 x i32> <i32 0, i32 32, i32 poison, i32 poison, i32 poison, i32 5, i32 poison, i32 poison, i32 poison, i32 poison, i32 10, i32 11, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT:    [[TMP114:%.*]] = shufflevector <2 x i64> [[TMP65]], <2 x i64> poison, <32 x i32> <i32 0, i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT:    [[TMP95:%.*]] = shufflevector <32 x i64> [[TMP93]], <32 x i64> [[TMP114]], <32 x i32> <i32 0, i32 1, i32 32, i32 33, i32 4, i32 5, i32 6, i32 7, i32 8, i32 9, i32 10, i32 11, i32 12, i32 13, i32 14, i32 15, i32 16, i32 17, i32 18, i32 19, i32 20, i32 21, i32 22, i32 23, i32 24, i32 25, i32 26, i32 27, i32 28, i32 29, i32 30, i32 31>
+; CHECK-NEXT:    [[TMP96:%.*]] = insertelement <32 x i64> [[TMP95]], i64 [[TMP78]], i64 4
+; CHECK-NEXT:    [[TMP97:%.*]] = shufflevector <32 x i64> [[TMP96]], <32 x i64> [[TMP70]], <32 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 36, i32 37, i32 poison, i32 poison, i32 10, i32 11, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT:    [[TMP98:%.*]] = shufflevector <2 x i64> [[TMP71]], <2 x i64> poison, <32 x i32> <i32 0, i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT:    [[TMP99:%.*]] = shufflevector <32 x i64> [[TMP97]], <32 x i64> [[TMP98]], <32 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 32, i32 33, i32 10, i32 11, i32 12, i32 13, i32 14, i32 15, i32 16, i32 17, i32 18, i32 19, i32 20, i32 21, i32 22, i32 23, i32 24, i32 25, i32 26, i32 27, i32 28, i32 29, i32 30, i32 31>
+; CHECK-NEXT:    [[TMP100:%.*]] = insertelement <32 x i64> [[TMP99]], i64 [[TMP94]], i64 12
+; CHECK-NEXT:    [[TMP101:%.*]] = shufflevector <32 x i64> [[TMP100]], <32 x i64> poison, <32 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 5, i32 5, i32 7, i32 5, i32 8, i32 9, i32 10, i32 5, i32 0, i32 1, i32 11, i32 12, i32 5, i32 5, i32 6, i32 7, i32 5, i32 5, i32 7, i32 5, i32 9, i32 10, i32 0, i32 1>
+; CHECK-NEXT:    [[TMP102:%.*]] = shufflevector <32 x i64> [[TMP100]], <32 x i64> poison, <10 x i32> <i32 poison, i32 poison, i32 5, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT:    [[TMP103:%.*]] = insertelement <10 x i64> [[TMP102]], i64 [[TMP0]], i64 0
+; CHECK-NEXT:    [[TMP121:%.*]] = insertelement <10 x i64> [[TMP103]], i64 [[TMP3]], i64 1
+; CHECK-NEXT:    [[TMP105:%.*]] = insertelement <10 x i64> [[TMP121]], i64 [[TMP4]], i64 3
+; CHECK-NEXT:    [[TMP106:%.*]] = insertelement <10 x i64> [[TMP105]], i64 [[INDVAR3788]], i64 4
+; CHECK-NEXT:    [[TMP124:%.*]] = insertelement <10 x i64> [[TMP106]], i64 [[TMP2]], i64 5
+; CHECK-NEXT:    [[TMP108:%.*]] = insertelement <10 x i64> [[TMP124]], i64 [[TMP5]], i64 6
+; CHECK-NEXT:    [[TMP109:%.*]] = insertelement <10 x i64> [[TMP108]], i64 [[TMP6]], i64 7
+; CHECK-NEXT:    [[TMP110:%.*]] = insertelement <10 x i64> [[TMP109]], i64 [[TMP7]], i64 8
+; CHECK-NEXT:    [[TMP111:%.*]] = insertelement <10 x i64> [[TMP110]], i64 [[DOTSROA_3341_0_COPYLOAD]], i64 9
+; CHECK-NEXT:    [[TMP125:%.*]] = shufflevector <10 x i64> [[TMP111]], <10 x i64> poison, <32 x i32> <i32 0, i32 0, i32 0, i32 0, i32 0, i32 1, i32 0, i32 2, i32 3, i32 4, i32 0, i32 0, i32 2, i32 2, i32 0, i32 5, i32 0, i32 0, i32 0, i32 0, i32 6, i32 7, i32 0, i32 2, i32 8, i32 9, i32 0, i32 0, i32 2, i32 0, i32 0, i32 0>
+; CHECK-NEXT:    [[TMP113:%.*]] = icmp ult <32 x i64> [[TMP101]], [[TMP125]]
+; CHECK-NEXT:    [[TMP116:%.*]] = icmp ult <2 x i64> [[TMP139]], [[TMP115]]
+; CHECK-NEXT:    [[OP_RDX61:%.*]] = call i1 @llvm.vector.reduce.or.v32i1(<32 x i1> [[TMP113]])
+; CHECK-NEXT:    [[TMP160:%.*]] = call i1 @llvm.vector.reduce.or.v8i1(<8 x i1> [[TMP58]])
+; CHECK-NEXT:    [[TMP117:%.*]] = shufflevector <8 x i1> [[TMP81]], <8 x i1> poison, <4 x i32> <i32 0, i32 1, i32 2, i32 3>
+; CHECK-NEXT:    [[RDX_OP:%.*]] = or <4 x i1> [[TMP117]], [[TMP86]]
+; CHECK-NEXT:    [[TMP118:%.*]] = shufflevector <4 x i1> [[RDX_OP]], <4 x i1> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT:    [[TMP119:%.*]] = shufflevector <8 x i1> [[TMP81]], <8 x i1> [[TMP118]], <8 x i32> <i32 8, i32 9, i32 10, i32 11, i32 4, i32 5, i32 6, i32 7>
+; CHECK-NEXT:    [[TMP120:%.*]] = call i1 @llvm.vector.reduce.or.v8i1(<8 x i1> [[TMP119]])
+; CHECK-NEXT:    [[OP_RDX62:%.*]] = or i1 [[TMP120]], [[DIFF_CHECK3783]]
+; CHECK-NEXT:    [[TMP150:%.*]] = call i1 @llvm.vector.reduce.or.v2i1(<2 x i1> [[TMP63]])
+; CHECK-NEXT:    [[OP_RDX58:%.*]] = or i1 [[DIFF_CHECK3817]], [[DIFF_CHECK3930]]
 ; CHECK-NEXT:    [[OP_RDX63:%.*]] = or i1 [[OP_RDX26]], [[OP_RDX57]]
-; CHECK-NEXT:    [[OP_RDX64:%.*]] = or i1 [[OP_RDX58]], [[OP_RDX59]]
-; CHECK-NEXT:    [[OP_RDX65:%.*]] = or i1 [[OP_RDX52]], [[OP_RDX61]]
-; CHECK-NEXT:    [[OP_RDX66:%.*]] = or i1 [[OP_RDX62]], [[TMP108]]
-; CHECK-NEXT:    [[OP_RDX67:%.*]] = or i1 [[OP_RDX63]], [[OP_RDX64]]
-; CHECK-NEXT:    [[OP_RDX68:%.*]] = or i1 [[OP_RDX65]], [[OP_RDX66]]
-; CHECK-NEXT:    [[TMP79:%.*]] = or i1 [[OP_RDX67]], [[OP_RDX68]]
+; CHECK-NEXT:    [[TMP122:%.*]] = call i1 @llvm.vector.reduce.or.v2i1(<2 x i1> [[TMP112]])
+; CHECK-NEXT:    [[TMP123:%.*]] = call i1 @llvm.vector.reduce.or.v2i1(<2 x i1> [[TMP116]])
+; CHECK-NEXT:    [[OP_RDX38:%.*]] = or i1 [[OP_RDX52]], [[OP_RDX61]]
+; CHECK-NEXT:    [[OP_RDX65:%.*]] = or i1 [[OP_RDX62]], [[TMP150]]
+; CHECK-NEXT:    [[OP_RDX64:%.*]] = or i1 [[OP_RDX58]], [[OP_RDX63]]
+; CHECK-NEXT:    [[OP_RDX66:%.*]] = or i1 [[TMP122]], [[TMP123]]
+; CHECK-NEXT:    [[OP_RDX43:%.*]] = or i1 [[OP_RDX38]], [[TMP160]]
+; CHECK-NEXT:    [[OP_RDX45:%.*]] = or i1 [[OP_RDX65]], [[OP_RDX64]]
+; CHECK-NEXT:    [[OP_RDX46:%.*]] = or i1 [[OP_RDX66]], [[OP_RDX43]]
+; CHECK-NEXT:    [[TMP79:%.*]] = or i1 [[OP_RDX45]], [[OP_RDX46]]
 ; CHECK-NEXT:    br i1 [[TMP79]], label %[[SCALAR_PH4023:.*]], label %[[DOTLR_PH1977_US]]
 ; CHECK:       [[SCALAR_PH4023]]:
 ; CHECK-NEXT:    ret void

diff  --git a/llvm/test/Transforms/SLPVectorizer/AArch64/non-power-of-2-with-adjusted-gathers.ll b/llvm/test/Transforms/SLPVectorizer/AArch64/non-power-of-2-with-adjusted-gathers.ll
index 2e5259ad9bfa1..489d47eeb606d 100644
--- a/llvm/test/Transforms/SLPVectorizer/AArch64/non-power-of-2-with-adjusted-gathers.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AArch64/non-power-of-2-with-adjusted-gathers.ll
@@ -12,18 +12,9 @@ define i1 @test(ptr %arg, ptr %arg1, i1 %arg2, i1 %arg3, i1 %arg4) {
 ; CHECK-NEXT:    [[TMP2:%.*]] = getelementptr i8, <8 x ptr> [[TMP1]], <8 x i64> <i64 8484, i64 8484, i64 8484, i64 5284, i64 8484, i64 8484, i64 8484, i64 8484>
 ; CHECK-NEXT:    [[TMP3:%.*]] = extractelement <8 x ptr> [[TMP2]], i64 0
 ; CHECK-NEXT:    [[ICMP:%.*]] = icmp ult ptr [[TMP3]], null
-; CHECK-NEXT:    [[ICMP7:%.*]] = icmp ult ptr null, null
-; CHECK-NEXT:    [[AND:%.*]] = and i1 false, [[ICMP7]]
-; CHECK-NEXT:    [[ICMP11:%.*]] = icmp ult ptr null, null
-; CHECK-NEXT:    [[AND12:%.*]] = and i1 false, [[ICMP11]]
-; CHECK-NEXT:    [[ICMP14:%.*]] = icmp ult ptr null, null
-; CHECK-NEXT:    [[AND15:%.*]] = and i1 false, [[ICMP14]]
 ; CHECK-NEXT:    [[TMP4:%.*]] = icmp ult <8 x ptr> [[TMP2]], splat (ptr null)
 ; CHECK-NEXT:    [[TMP5:%.*]] = insertelement <8 x i1> <i1 false, i1 poison, i1 false, i1 false, i1 false, i1 false, i1 false, i1 false>, i1 [[ARG2]], i64 1
 ; CHECK-NEXT:    [[TMP6:%.*]] = and <8 x i1> [[TMP4]], [[TMP5]]
-; CHECK-NEXT:    [[TMP7:%.*]] = extractelement <8 x ptr> [[TMP2]], i64 3
-; CHECK-NEXT:    [[ICMP46:%.*]] = icmp ult ptr [[TMP7]], null
-; CHECK-NEXT:    [[AND47:%.*]] = and i1 false, [[ICMP46]]
 ; CHECK-NEXT:    [[ICMP49:%.*]] = icmp ult ptr [[ARG]], null
 ; CHECK-NEXT:    [[AND50:%.*]] = and i1 [[ICMP49]], false
 ; CHECK-NEXT:    [[ICMP52:%.*]] = icmp ult ptr [[ARG1]], null
@@ -37,20 +28,21 @@ define i1 @test(ptr %arg, ptr %arg1, i1 %arg2, i1 %arg3, i1 %arg4) {
 ; CHECK-NEXT:    [[ICMP65:%.*]] = icmp ult ptr [[GETELEMENTPTR]], null
 ; CHECK-NEXT:    [[AND66:%.*]] = and i1 [[ICMP65]], false
 ; CHECK-NEXT:    [[TMP15:%.*]] = call i1 @llvm.vector.reduce.or.v8i1(<8 x i1> [[TMP6]])
-; CHECK-NEXT:    [[OP_RDX30:%.*]] = or i1 [[TMP15]], [[AND]]
-; CHECK-NEXT:    [[OP_RDX31:%.*]] = or i1 false, [[AND12]]
-; CHECK-NEXT:    [[OP_RDX26:%.*]] = or i1 [[AND15]], false
-; CHECK-NEXT:    [[OP_RDX27:%.*]] = or i1 false, [[AND47]]
 ; CHECK-NEXT:    [[OP_RDX28:%.*]] = or i1 [[AND50]], [[AND53]]
 ; CHECK-NEXT:    [[OP_RDX29:%.*]] = or i1 [[AND56]], [[AND59]]
 ; CHECK-NEXT:    [[OP_RDX33:%.*]] = or i1 [[AND63]], [[AND66]]
 ; CHECK-NEXT:    [[OP_RDX34:%.*]] = or i1 [[ARG3]], [[ARG2]]
 ; CHECK-NEXT:    [[OP_RDX32:%.*]] = or i1 [[ARG4]], false
-; CHECK-NEXT:    [[OP_RDX36:%.*]] = or i1 [[OP_RDX30]], [[OP_RDX31]]
-; CHECK-NEXT:    [[OP_RDX38:%.*]] = or i1 [[OP_RDX26]], [[OP_RDX27]]
+; CHECK-NEXT:    [[TMP16:%.*]] = shufflevector <8 x ptr> [[TMP2]], <8 x ptr> poison, <2 x i32> <i32 poison, i32 3>
+; CHECK-NEXT:    [[TMP9:%.*]] = shufflevector <2 x ptr> [[TMP16]], <2 x ptr> <ptr null, ptr poison>, <2 x i32> <i32 2, i32 1>
+; CHECK-NEXT:    [[TMP10:%.*]] = icmp ult <2 x ptr> [[TMP9]], splat (ptr null)
+; CHECK-NEXT:    [[TMP11:%.*]] = and <2 x i1> zeroinitializer, [[TMP10]]
+; CHECK-NEXT:    [[TMP12:%.*]] = insertelement <2 x i1> <i1 poison, i1 false>, i1 [[TMP15]], i64 0
+; CHECK-NEXT:    [[TMP13:%.*]] = or <2 x i1> [[TMP12]], [[TMP11]]
+; CHECK-NEXT:    [[TMP14:%.*]] = or <2 x i1> zeroinitializer, [[TMP13]]
 ; CHECK-NEXT:    [[OP_RDX35:%.*]] = or i1 [[OP_RDX28]], [[OP_RDX29]]
 ; CHECK-NEXT:    [[OP_RDX37:%.*]] = or i1 [[OP_RDX33]], [[OP_RDX34]]
-; CHECK-NEXT:    [[OP_RDX46:%.*]] = or i1 [[OP_RDX36]], [[OP_RDX38]]
+; CHECK-NEXT:    [[OP_RDX46:%.*]] = call i1 @llvm.vector.reduce.or.v2i1(<2 x i1> [[TMP14]])
 ; CHECK-NEXT:    [[TMP8:%.*]] = or i1 [[OP_RDX35]], [[OP_RDX37]]
 ; CHECK-NEXT:    [[OP_RDX39:%.*]] = or i1 [[OP_RDX46]], [[TMP8]]
 ; CHECK-NEXT:    [[OP_RDX40:%.*]] = or i1 [[OP_RDX39]], [[OP_RDX32]]

diff  --git a/llvm/test/Transforms/SLPVectorizer/AArch64/revec-non-pow2.ll b/llvm/test/Transforms/SLPVectorizer/AArch64/revec-non-pow2.ll
index 1355224e94b71..93c8d0ee15d3e 100644
--- a/llvm/test/Transforms/SLPVectorizer/AArch64/revec-non-pow2.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AArch64/revec-non-pow2.ll
@@ -31,9 +31,8 @@ define i32 @test(ptr %0) {
 ; CHECK-NEXT:    [[TMP24:%.*]] = load <3 x i32>, ptr [[TMP9]], align 4
 ; CHECK-NEXT:    [[TMP25:%.*]] = add <3 x i32> [[TMP23]], [[TMP24]]
 ; CHECK-NEXT:    [[TMP26:%.*]] = icmp sgt <3 x i32> [[TMP25]], zeroinitializer
-; CHECK-NEXT:    [[TMP27:%.*]] = bitcast <3 x i1> [[TMP26]] to i3
-; CHECK-NEXT:    [[TMP28:%.*]] = call i3 @llvm.ctpop.i3(i3 [[TMP27]])
-; CHECK-NEXT:    [[TMP29:%.*]] = zext i3 [[TMP28]] to i32
+; CHECK-NEXT:    [[TMP27:%.*]] = zext <3 x i1> [[TMP26]] to <3 x i32>
+; CHECK-NEXT:    [[TMP29:%.*]] = call i32 @llvm.vector.reduce.add.v3i32(<3 x i32> [[TMP27]])
 ; CHECK-NEXT:    ret i32 [[TMP29]]
 ;
 .split:

diff  --git a/llvm/test/Transforms/SLPVectorizer/AMDGPU/reduction-i1-mask.ll b/llvm/test/Transforms/SLPVectorizer/AMDGPU/reduction-i1-mask.ll
index a23e972aefbac..e3387f2b5a640 100644
--- a/llvm/test/Transforms/SLPVectorizer/AMDGPU/reduction-i1-mask.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AMDGPU/reduction-i1-mask.ll
@@ -60,9 +60,8 @@ define i32 @count_smaller(ptr addrspace(3) %tab, i32 %key) {
 ; FORCED-NEXT:    [[TMP1:%.*]] = insertelement <8 x i32> poison, i32 [[KEY]], i64 0
 ; FORCED-NEXT:    [[TMP2:%.*]] = shufflevector <8 x i32> [[TMP1]], <8 x i32> poison, <8 x i32> zeroinitializer
 ; FORCED-NEXT:    [[TMP3:%.*]] = icmp slt <8 x i32> [[TMP0]], [[TMP2]]
-; FORCED-NEXT:    [[TMP4:%.*]] = bitcast <8 x i1> [[TMP3]] to i8
-; FORCED-NEXT:    [[TMP5:%.*]] = call i8 @llvm.ctpop.i8(i8 [[TMP4]])
-; FORCED-NEXT:    [[TMP6:%.*]] = zext i8 [[TMP5]] to i32
+; FORCED-NEXT:    [[TMP4:%.*]] = zext <8 x i1> [[TMP3]] to <8 x i32>
+; FORCED-NEXT:    [[TMP6:%.*]] = call i32 @llvm.vector.reduce.add.v8i32(<8 x i32> [[TMP4]])
 ; FORCED-NEXT:    ret i32 [[TMP6]]
 ;
 entry:

diff  --git a/llvm/test/Transforms/SLPVectorizer/RISCV/remark-zext-incoming-for-neg-icmp.ll b/llvm/test/Transforms/SLPVectorizer/RISCV/remark-zext-incoming-for-neg-icmp.ll
index 5ef40a3f6d2f7..0146718c2b0f4 100644
--- a/llvm/test/Transforms/SLPVectorizer/RISCV/remark-zext-incoming-for-neg-icmp.ll
+++ b/llvm/test/Transforms/SLPVectorizer/RISCV/remark-zext-incoming-for-neg-icmp.ll
@@ -8,7 +8,7 @@
 ; YAML-NEXT: Function:        test
 ; YAML-NEXT: Args:
 ; YAML-NEXT:   - String:          'Vectorized horizontal reduction with cost '
-; YAML-NEXT:   - Cost:            '-10'
+; YAML-NEXT:   - Cost:            '-11'
 ; YAML-NEXT:   - String:          ' and with tree size '
 ; YAML-NEXT:   - TreeSize:        '8'
 ; YAML-NEXT:...

diff  --git a/llvm/test/Transforms/SLPVectorizer/SystemZ/cmp-ptr-minmax.ll b/llvm/test/Transforms/SLPVectorizer/SystemZ/cmp-ptr-minmax.ll
index 08257a47ce06c..aa9be8c251fe6 100644
--- a/llvm/test/Transforms/SLPVectorizer/SystemZ/cmp-ptr-minmax.ll
+++ b/llvm/test/Transforms/SLPVectorizer/SystemZ/cmp-ptr-minmax.ll
@@ -18,9 +18,7 @@ define i1 @test(i64 %0, i64 %1, ptr %2) {
 ; CHECK-NEXT:    [[TMP9:%.*]] = insertelement <2 x ptr> poison, ptr [[TMP2]], i64 0
 ; CHECK-NEXT:    [[TMP10:%.*]] = shufflevector <2 x ptr> [[TMP9]], <2 x ptr> poison, <2 x i32> zeroinitializer
 ; CHECK-NEXT:    [[TMP11:%.*]] = icmp ult <2 x ptr> [[TMP8]], [[TMP10]]
-; CHECK-NEXT:    [[TMP12:%.*]] = extractelement <2 x i1> [[TMP11]], i64 0
-; CHECK-NEXT:    [[TMP13:%.*]] = extractelement <2 x i1> [[TMP11]], i64 1
-; CHECK-NEXT:    [[RES:%.*]] = and i1 [[TMP12]], [[TMP13]]
+; CHECK-NEXT:    [[RES:%.*]] = call i1 @llvm.vector.reduce.and.v2i1(<2 x i1> [[TMP11]])
 ; CHECK-NEXT:    ret i1 [[RES]]
 ;
 entry:

diff  --git a/llvm/test/Transforms/SLPVectorizer/X86/alternate-cmp-swapped-pred.ll b/llvm/test/Transforms/SLPVectorizer/X86/alternate-cmp-swapped-pred.ll
index 7a1d47f879ebf..ee4e57699a6ce 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/alternate-cmp-swapped-pred.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/alternate-cmp-swapped-pred.ll
@@ -11,9 +11,8 @@ define i16 @test(i16 %call37) {
 ; CHECK-NEXT:    [[TMP2:%.*]] = icmp slt <8 x i16> [[SHUFFLE]], zeroinitializer
 ; CHECK-NEXT:    [[TMP3:%.*]] = icmp sgt <8 x i16> [[SHUFFLE]], zeroinitializer
 ; CHECK-NEXT:    [[TMP4:%.*]] = shufflevector <8 x i1> [[TMP2]], <8 x i1> [[TMP3]], <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 12, i32 5, i32 6, i32 7>
-; CHECK-NEXT:    [[TMP8:%.*]] = bitcast <8 x i1> [[TMP4]] to i8
-; CHECK-NEXT:    [[TMP7:%.*]] = call i8 @llvm.ctpop.i8(i8 [[TMP8]])
-; CHECK-NEXT:    [[TMP6:%.*]] = zext i8 [[TMP7]] to i16
+; CHECK-NEXT:    [[TMP7:%.*]] = zext <8 x i1> [[TMP4]] to <8 x i16>
+; CHECK-NEXT:    [[TMP6:%.*]] = call i16 @llvm.vector.reduce.add.v8i16(<8 x i16> [[TMP7]])
 ; CHECK-NEXT:    [[OP_RDX:%.*]] = add i16 [[TMP6]], 0
 ; CHECK-NEXT:    ret i16 [[OP_RDX]]
 ;

diff  --git a/llvm/test/Transforms/SLPVectorizer/X86/entries-
diff erent-vf.ll b/llvm/test/Transforms/SLPVectorizer/X86/entries-
diff erent-vf.ll
index a46467563bb24..53cba0f957935 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/entries-
diff erent-vf.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/entries-
diff erent-vf.ll
@@ -17,8 +17,9 @@ define i1 @test(i64 %v) {
 ; CHECK-NEXT:    [[TMP9:%.*]] = sub <8 x i64> [[TMP7]], [[TMP5]]
 ; CHECK-NEXT:    [[TMP10:%.*]] = shufflevector <8 x i64> [[TMP8]], <8 x i64> [[TMP9]], <8 x i32> <i32 0, i32 1, i32 2, i32 11, i32 12, i32 5, i32 6, i32 7>
 ; CHECK-NEXT:    [[TMP11:%.*]] = icmp ult <8 x i64> [[TMP10]], zeroinitializer
-; CHECK-NEXT:    [[TMP12:%.*]] = call i1 @llvm.vector.reduce.or.v8i1(<8 x i1> [[TMP11]])
-; CHECK-NEXT:    ret i1 [[TMP12]]
+; CHECK-NEXT:    [[TMP12:%.*]] = bitcast <8 x i1> [[TMP11]] to i8
+; CHECK-NEXT:    [[TMP13:%.*]] = icmp ne i8 [[TMP12]], 0
+; CHECK-NEXT:    ret i1 [[TMP13]]
 ;
 entry:
   %0 = shl i64 %v, 1

diff  --git a/llvm/test/Transforms/SLPVectorizer/X86/fabs-cost-softfp.ll b/llvm/test/Transforms/SLPVectorizer/X86/fabs-cost-softfp.ll
index fdb43f683f644..28b4e8f2cc889 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/fabs-cost-softfp.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/fabs-cost-softfp.ll
@@ -15,9 +15,7 @@ define void @vectorize_fp128(fp128 %c, fp128 %d) #0 {
 ; CHECK-NEXT:    [[TMP1:%.*]] = insertelement <2 x fp128> [[TMP0]], fp128 [[D:%.*]], i64 1
 ; CHECK-NEXT:    [[TMP2:%.*]] = call <2 x fp128> @llvm.fabs.v2f128(<2 x fp128> [[TMP1]])
 ; CHECK-NEXT:    [[TMP3:%.*]] = fcmp oeq <2 x fp128> [[TMP2]], splat (fp128 +inf)
-; CHECK-NEXT:    [[TMP4:%.*]] = extractelement <2 x i1> [[TMP3]], i64 0
-; CHECK-NEXT:    [[TMP5:%.*]] = extractelement <2 x i1> [[TMP3]], i64 1
-; CHECK-NEXT:    [[OR_COND39:%.*]] = or i1 [[TMP4]], [[TMP5]]
+; CHECK-NEXT:    [[OR_COND39:%.*]] = call i1 @llvm.vector.reduce.or.v2i1(<2 x i1> [[TMP3]])
 ; CHECK-NEXT:    br i1 [[OR_COND39]], label [[IF_THEN13:%.*]], label [[IF_END24:%.*]]
 ; CHECK:       if.then13:
 ; CHECK-NEXT:    unreachable

diff  --git a/llvm/test/Transforms/SLPVectorizer/X86/reduction-logical.ll b/llvm/test/Transforms/SLPVectorizer/X86/reduction-logical.ll
index 6b073b10bac9b..8f160742246f4 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/reduction-logical.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/reduction-logical.ll
@@ -5,11 +5,18 @@
 declare void @use1(i1)
 
 define i1 @logical_and_icmp(<4 x i32> %x) {
-; CHECK-LABEL: @logical_and_icmp(
-; CHECK-NEXT:    [[TMP1:%.*]] = icmp slt <4 x i32> [[X:%.*]], zeroinitializer
-; CHECK-NEXT:    [[TMP2:%.*]] = freeze <4 x i1> [[TMP1]]
-; CHECK-NEXT:    [[TMP3:%.*]] = call i1 @llvm.vector.reduce.and.v4i1(<4 x i1> [[TMP2]])
-; CHECK-NEXT:    ret i1 [[TMP3]]
+; SSE-LABEL: @logical_and_icmp(
+; SSE-NEXT:    [[TMP1:%.*]] = icmp slt <4 x i32> [[X:%.*]], zeroinitializer
+; SSE-NEXT:    [[TMP2:%.*]] = freeze <4 x i1> [[TMP1]]
+; SSE-NEXT:    [[TMP3:%.*]] = call i1 @llvm.vector.reduce.and.v4i1(<4 x i1> [[TMP2]])
+; SSE-NEXT:    ret i1 [[TMP3]]
+;
+; AVX-LABEL: @logical_and_icmp(
+; AVX-NEXT:    [[TMP1:%.*]] = icmp slt <4 x i32> [[X:%.*]], zeroinitializer
+; AVX-NEXT:    [[TMP2:%.*]] = freeze <4 x i1> [[TMP1]]
+; AVX-NEXT:    [[TMP3:%.*]] = bitcast <4 x i1> [[TMP2]] to i4
+; AVX-NEXT:    [[TMP4:%.*]] = icmp eq i4 [[TMP3]], -1
+; AVX-NEXT:    ret i1 [[TMP4]]
 ;
   %x0 = extractelement <4 x i32> %x, i32 0
   %x1 = extractelement <4 x i32> %x, i32 1
@@ -26,11 +33,18 @@ define i1 @logical_and_icmp(<4 x i32> %x) {
 }
 
 define i1 @logical_or_icmp(<4 x i32> %x, <4 x i32> %y) {
-; CHECK-LABEL: @logical_or_icmp(
-; CHECK-NEXT:    [[TMP1:%.*]] = icmp slt <4 x i32> [[X:%.*]], [[Y:%.*]]
-; CHECK-NEXT:    [[TMP2:%.*]] = freeze <4 x i1> [[TMP1]]
-; CHECK-NEXT:    [[TMP3:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP2]])
-; CHECK-NEXT:    ret i1 [[TMP3]]
+; SSE-LABEL: @logical_or_icmp(
+; SSE-NEXT:    [[TMP1:%.*]] = icmp slt <4 x i32> [[X:%.*]], [[Y:%.*]]
+; SSE-NEXT:    [[TMP2:%.*]] = freeze <4 x i1> [[TMP1]]
+; SSE-NEXT:    [[TMP3:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP2]])
+; SSE-NEXT:    ret i1 [[TMP3]]
+;
+; AVX-LABEL: @logical_or_icmp(
+; AVX-NEXT:    [[TMP1:%.*]] = icmp slt <4 x i32> [[X:%.*]], [[Y:%.*]]
+; AVX-NEXT:    [[TMP2:%.*]] = freeze <4 x i1> [[TMP1]]
+; AVX-NEXT:    [[TMP3:%.*]] = bitcast <4 x i1> [[TMP2]] to i4
+; AVX-NEXT:    [[TMP4:%.*]] = icmp ne i4 [[TMP3]], 0
+; AVX-NEXT:    ret i1 [[TMP4]]
 ;
   %x0 = extractelement <4 x i32> %x, i32 0
   %x1 = extractelement <4 x i32> %x, i32 1
@@ -51,11 +65,18 @@ define i1 @logical_or_icmp(<4 x i32> %x, <4 x i32> %y) {
 }
 
 define i1 @logical_and_fcmp(<4 x float> %x) {
-; CHECK-LABEL: @logical_and_fcmp(
-; CHECK-NEXT:    [[TMP1:%.*]] = fcmp olt <4 x float> [[X:%.*]], zeroinitializer
-; CHECK-NEXT:    [[TMP2:%.*]] = freeze <4 x i1> [[TMP1]]
-; CHECK-NEXT:    [[TMP3:%.*]] = call i1 @llvm.vector.reduce.and.v4i1(<4 x i1> [[TMP2]])
-; CHECK-NEXT:    ret i1 [[TMP3]]
+; SSE-LABEL: @logical_and_fcmp(
+; SSE-NEXT:    [[TMP1:%.*]] = fcmp olt <4 x float> [[X:%.*]], zeroinitializer
+; SSE-NEXT:    [[TMP2:%.*]] = freeze <4 x i1> [[TMP1]]
+; SSE-NEXT:    [[TMP3:%.*]] = call i1 @llvm.vector.reduce.and.v4i1(<4 x i1> [[TMP2]])
+; SSE-NEXT:    ret i1 [[TMP3]]
+;
+; AVX-LABEL: @logical_and_fcmp(
+; AVX-NEXT:    [[TMP1:%.*]] = fcmp olt <4 x float> [[X:%.*]], zeroinitializer
+; AVX-NEXT:    [[TMP2:%.*]] = freeze <4 x i1> [[TMP1]]
+; AVX-NEXT:    [[TMP3:%.*]] = bitcast <4 x i1> [[TMP2]] to i4
+; AVX-NEXT:    [[TMP4:%.*]] = icmp eq i4 [[TMP3]], -1
+; AVX-NEXT:    ret i1 [[TMP4]]
 ;
   %x0 = extractelement <4 x float> %x, i32 0
   %x1 = extractelement <4 x float> %x, i32 1
@@ -72,11 +93,18 @@ define i1 @logical_and_fcmp(<4 x float> %x) {
 }
 
 define i1 @logical_or_fcmp(<4 x float> %x) {
-; CHECK-LABEL: @logical_or_fcmp(
-; CHECK-NEXT:    [[TMP1:%.*]] = fcmp olt <4 x float> [[X:%.*]], zeroinitializer
-; CHECK-NEXT:    [[TMP2:%.*]] = freeze <4 x i1> [[TMP1]]
-; CHECK-NEXT:    [[TMP3:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP2]])
-; CHECK-NEXT:    ret i1 [[TMP3]]
+; SSE-LABEL: @logical_or_fcmp(
+; SSE-NEXT:    [[TMP1:%.*]] = fcmp olt <4 x float> [[X:%.*]], zeroinitializer
+; SSE-NEXT:    [[TMP2:%.*]] = freeze <4 x i1> [[TMP1]]
+; SSE-NEXT:    [[TMP3:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP2]])
+; SSE-NEXT:    ret i1 [[TMP3]]
+;
+; AVX-LABEL: @logical_or_fcmp(
+; AVX-NEXT:    [[TMP1:%.*]] = fcmp olt <4 x float> [[X:%.*]], zeroinitializer
+; AVX-NEXT:    [[TMP2:%.*]] = freeze <4 x i1> [[TMP1]]
+; AVX-NEXT:    [[TMP3:%.*]] = bitcast <4 x i1> [[TMP2]] to i4
+; AVX-NEXT:    [[TMP4:%.*]] = icmp ne i4 [[TMP3]], 0
+; AVX-NEXT:    ret i1 [[TMP4]]
 ;
   %x0 = extractelement <4 x float> %x, i32 0
   %x1 = extractelement <4 x float> %x, i32 1
@@ -104,18 +132,15 @@ define i1 @logical_and_icmp_
diff _preds(<4 x i32> %x) {
 ; SSE-NEXT:    ret i1 [[TMP7]]
 ;
 ; AVX-LABEL: @logical_and_icmp_
diff _preds(
-; AVX-NEXT:    [[X0:%.*]] = extractelement <4 x i32> [[X:%.*]], i32 0
-; AVX-NEXT:    [[X1:%.*]] = extractelement <4 x i32> [[X]], i32 1
-; AVX-NEXT:    [[X2:%.*]] = extractelement <4 x i32> [[X]], i32 2
-; AVX-NEXT:    [[X3:%.*]] = extractelement <4 x i32> [[X]], i32 3
-; AVX-NEXT:    [[C0:%.*]] = icmp ult i32 [[X0]], 0
-; AVX-NEXT:    [[C1:%.*]] = icmp slt i32 [[X1]], 0
-; AVX-NEXT:    [[C2:%.*]] = icmp sgt i32 [[X2]], 0
-; AVX-NEXT:    [[C3:%.*]] = icmp slt i32 [[X3]], 0
-; AVX-NEXT:    [[S1:%.*]] = select i1 [[C0]], i1 [[C1]], i1 false
-; AVX-NEXT:    [[S2:%.*]] = select i1 [[S1]], i1 [[C2]], i1 false
-; AVX-NEXT:    [[S3:%.*]] = select i1 [[S2]], i1 [[C3]], i1 false
-; AVX-NEXT:    ret i1 [[S3]]
+; AVX-NEXT:    [[TMP1:%.*]] = shufflevector <4 x i32> [[X:%.*]], <4 x i32> <i32 poison, i32 poison, i32 0, i32 poison>, <4 x i32> <i32 1, i32 3, i32 6, i32 0>
+; AVX-NEXT:    [[TMP2:%.*]] = shufflevector <4 x i32> [[X]], <4 x i32> <i32 0, i32 0, i32 poison, i32 0>, <4 x i32> <i32 4, i32 5, i32 2, i32 7>
+; AVX-NEXT:    [[TMP3:%.*]] = icmp slt <4 x i32> [[TMP1]], [[TMP2]]
+; AVX-NEXT:    [[TMP4:%.*]] = icmp ult <4 x i32> [[TMP1]], [[TMP2]]
+; AVX-NEXT:    [[TMP5:%.*]] = shufflevector <4 x i1> [[TMP3]], <4 x i1> [[TMP4]], <4 x i32> <i32 0, i32 1, i32 2, i32 7>
+; AVX-NEXT:    [[TMP6:%.*]] = freeze <4 x i1> [[TMP5]]
+; AVX-NEXT:    [[TMP7:%.*]] = bitcast <4 x i1> [[TMP6]] to i4
+; AVX-NEXT:    [[TMP8:%.*]] = icmp eq i4 [[TMP7]], -1
+; AVX-NEXT:    ret i1 [[TMP8]]
 ;
   %x0 = extractelement <4 x i32> %x, i32 0
   %x1 = extractelement <4 x i32> %x, i32 1
@@ -132,11 +157,18 @@ define i1 @logical_and_icmp_
diff _preds(<4 x i32> %x) {
 }
 
 define i1 @logical_and_icmp_
diff _const(<4 x i32> %x) {
-; CHECK-LABEL: @logical_and_icmp_
diff _const(
-; CHECK-NEXT:    [[TMP1:%.*]] = icmp sgt <4 x i32> [[X:%.*]], <i32 0, i32 1, i32 2, i32 3>
-; CHECK-NEXT:    [[TMP2:%.*]] = freeze <4 x i1> [[TMP1]]
-; CHECK-NEXT:    [[TMP3:%.*]] = call i1 @llvm.vector.reduce.and.v4i1(<4 x i1> [[TMP2]])
-; CHECK-NEXT:    ret i1 [[TMP3]]
+; SSE-LABEL: @logical_and_icmp_
diff _const(
+; SSE-NEXT:    [[TMP1:%.*]] = icmp sgt <4 x i32> [[X:%.*]], <i32 0, i32 1, i32 2, i32 3>
+; SSE-NEXT:    [[TMP2:%.*]] = freeze <4 x i1> [[TMP1]]
+; SSE-NEXT:    [[TMP3:%.*]] = call i1 @llvm.vector.reduce.and.v4i1(<4 x i1> [[TMP2]])
+; SSE-NEXT:    ret i1 [[TMP3]]
+;
+; AVX-LABEL: @logical_and_icmp_
diff _const(
+; AVX-NEXT:    [[TMP1:%.*]] = icmp sgt <4 x i32> [[X:%.*]], <i32 0, i32 1, i32 2, i32 3>
+; AVX-NEXT:    [[TMP2:%.*]] = freeze <4 x i1> [[TMP1]]
+; AVX-NEXT:    [[TMP3:%.*]] = bitcast <4 x i1> [[TMP2]] to i4
+; AVX-NEXT:    [[TMP4:%.*]] = icmp eq i4 [[TMP3]], -1
+; AVX-NEXT:    ret i1 [[TMP4]]
 ;
   %x0 = extractelement <4 x i32> %x, i32 0
   %x1 = extractelement <4 x i32> %x, i32 1
@@ -206,15 +238,26 @@ define i1 @logical_and_icmp_subvec(<4 x i32> %x) {
 ;       logic...or a wide reduction?
 
 define i1 @logical_and_icmp_clamp(<4 x i32> %x) {
-; CHECK-LABEL: @logical_and_icmp_clamp(
-; CHECK-NEXT:    [[TMP1:%.*]] = icmp slt <4 x i32> [[X:%.*]], splat (i32 42)
-; CHECK-NEXT:    [[TMP2:%.*]] = icmp sgt <4 x i32> [[X]], splat (i32 17)
-; CHECK-NEXT:    [[TMP3:%.*]] = shufflevector <4 x i1> [[TMP2]], <4 x i1> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT:    [[TMP7:%.*]] = shufflevector <4 x i1> [[TMP1]], <4 x i1> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT:    [[TMP4:%.*]] = shufflevector <8 x i1> [[TMP3]], <8 x i1> [[TMP7]], <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 8, i32 9, i32 10, i32 11>
-; CHECK-NEXT:    [[TMP5:%.*]] = freeze <8 x i1> [[TMP4]]
-; CHECK-NEXT:    [[TMP6:%.*]] = call i1 @llvm.vector.reduce.and.v8i1(<8 x i1> [[TMP5]])
-; CHECK-NEXT:    ret i1 [[TMP6]]
+; SSE-LABEL: @logical_and_icmp_clamp(
+; SSE-NEXT:    [[TMP1:%.*]] = icmp slt <4 x i32> [[X:%.*]], splat (i32 42)
+; SSE-NEXT:    [[TMP2:%.*]] = icmp sgt <4 x i32> [[X]], splat (i32 17)
+; SSE-NEXT:    [[TMP3:%.*]] = shufflevector <4 x i1> [[TMP2]], <4 x i1> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison>
+; SSE-NEXT:    [[TMP4:%.*]] = shufflevector <4 x i1> [[TMP1]], <4 x i1> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison>
+; SSE-NEXT:    [[TMP5:%.*]] = shufflevector <8 x i1> [[TMP3]], <8 x i1> [[TMP4]], <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 8, i32 9, i32 10, i32 11>
+; SSE-NEXT:    [[TMP6:%.*]] = freeze <8 x i1> [[TMP5]]
+; SSE-NEXT:    [[TMP7:%.*]] = call i1 @llvm.vector.reduce.and.v8i1(<8 x i1> [[TMP6]])
+; SSE-NEXT:    ret i1 [[TMP7]]
+;
+; AVX-LABEL: @logical_and_icmp_clamp(
+; AVX-NEXT:    [[TMP1:%.*]] = icmp slt <4 x i32> [[X:%.*]], splat (i32 42)
+; AVX-NEXT:    [[TMP2:%.*]] = icmp sgt <4 x i32> [[X]], splat (i32 17)
+; AVX-NEXT:    [[TMP3:%.*]] = shufflevector <4 x i1> [[TMP2]], <4 x i1> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison>
+; AVX-NEXT:    [[TMP4:%.*]] = shufflevector <4 x i1> [[TMP1]], <4 x i1> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison>
+; AVX-NEXT:    [[TMP5:%.*]] = shufflevector <8 x i1> [[TMP3]], <8 x i1> [[TMP4]], <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 8, i32 9, i32 10, i32 11>
+; AVX-NEXT:    [[TMP6:%.*]] = freeze <8 x i1> [[TMP5]]
+; AVX-NEXT:    [[TMP7:%.*]] = bitcast <8 x i1> [[TMP6]] to i8
+; AVX-NEXT:    [[TMP8:%.*]] = icmp eq i8 [[TMP7]], -1
+; AVX-NEXT:    ret i1 [[TMP8]]
 ;
   %x0 = extractelement <4 x i32> %x, i32 0
   %x1 = extractelement <4 x i32> %x, i32 1
@@ -239,17 +282,30 @@ define i1 @logical_and_icmp_clamp(<4 x i32> %x) {
 }
 
 define i1 @logical_and_icmp_clamp_extra_use_cmp(<4 x i32> %x) {
-; CHECK-LABEL: @logical_and_icmp_clamp_extra_use_cmp(
-; CHECK-NEXT:    [[TMP1:%.*]] = icmp slt <4 x i32> [[X:%.*]], splat (i32 42)
-; CHECK-NEXT:    [[TMP5:%.*]] = extractelement <4 x i1> [[TMP1]], i64 2
-; CHECK-NEXT:    call void @use1(i1 [[TMP5]])
-; CHECK-NEXT:    [[TMP3:%.*]] = icmp sgt <4 x i32> [[X]], splat (i32 17)
-; CHECK-NEXT:    [[TMP8:%.*]] = shufflevector <4 x i1> [[TMP3]], <4 x i1> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT:    [[TMP9:%.*]] = shufflevector <4 x i1> [[TMP1]], <4 x i1> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT:    [[TMP4:%.*]] = shufflevector <8 x i1> [[TMP8]], <8 x i1> [[TMP9]], <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 8, i32 9, i32 10, i32 11>
-; CHECK-NEXT:    [[TMP6:%.*]] = freeze <8 x i1> [[TMP4]]
-; CHECK-NEXT:    [[TMP7:%.*]] = call i1 @llvm.vector.reduce.and.v8i1(<8 x i1> [[TMP6]])
-; CHECK-NEXT:    ret i1 [[TMP7]]
+; SSE-LABEL: @logical_and_icmp_clamp_extra_use_cmp(
+; SSE-NEXT:    [[TMP1:%.*]] = icmp slt <4 x i32> [[X:%.*]], splat (i32 42)
+; SSE-NEXT:    [[TMP2:%.*]] = extractelement <4 x i1> [[TMP1]], i64 2
+; SSE-NEXT:    call void @use1(i1 [[TMP2]])
+; SSE-NEXT:    [[TMP3:%.*]] = icmp sgt <4 x i32> [[X]], splat (i32 17)
+; SSE-NEXT:    [[TMP4:%.*]] = shufflevector <4 x i1> [[TMP3]], <4 x i1> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison>
+; SSE-NEXT:    [[TMP5:%.*]] = shufflevector <4 x i1> [[TMP1]], <4 x i1> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison>
+; SSE-NEXT:    [[TMP6:%.*]] = shufflevector <8 x i1> [[TMP4]], <8 x i1> [[TMP5]], <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 8, i32 9, i32 10, i32 11>
+; SSE-NEXT:    [[TMP7:%.*]] = freeze <8 x i1> [[TMP6]]
+; SSE-NEXT:    [[TMP8:%.*]] = call i1 @llvm.vector.reduce.and.v8i1(<8 x i1> [[TMP7]])
+; SSE-NEXT:    ret i1 [[TMP8]]
+;
+; AVX-LABEL: @logical_and_icmp_clamp_extra_use_cmp(
+; AVX-NEXT:    [[TMP1:%.*]] = icmp slt <4 x i32> [[X:%.*]], splat (i32 42)
+; AVX-NEXT:    [[TMP2:%.*]] = extractelement <4 x i1> [[TMP1]], i64 2
+; AVX-NEXT:    call void @use1(i1 [[TMP2]])
+; AVX-NEXT:    [[TMP3:%.*]] = icmp sgt <4 x i32> [[X]], splat (i32 17)
+; AVX-NEXT:    [[TMP4:%.*]] = shufflevector <4 x i1> [[TMP3]], <4 x i1> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison>
+; AVX-NEXT:    [[TMP5:%.*]] = shufflevector <4 x i1> [[TMP1]], <4 x i1> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison>
+; AVX-NEXT:    [[TMP6:%.*]] = shufflevector <8 x i1> [[TMP4]], <8 x i1> [[TMP5]], <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 8, i32 9, i32 10, i32 11>
+; AVX-NEXT:    [[TMP7:%.*]] = freeze <8 x i1> [[TMP6]]
+; AVX-NEXT:    [[TMP8:%.*]] = bitcast <8 x i1> [[TMP7]] to i8
+; AVX-NEXT:    [[TMP9:%.*]] = icmp eq i8 [[TMP8]], -1
+; AVX-NEXT:    ret i1 [[TMP9]]
 ;
   %x0 = extractelement <4 x i32> %x, i32 0
   %x1 = extractelement <4 x i32> %x, i32 1
@@ -275,21 +331,38 @@ define i1 @logical_and_icmp_clamp_extra_use_cmp(<4 x i32> %x) {
 }
 
 define i1 @logical_and_icmp_clamp_extra_use_select(<4 x i32> %x) {
-; CHECK-LABEL: @logical_and_icmp_clamp_extra_use_select(
-; CHECK-NEXT:    [[TMP1:%.*]] = icmp slt <4 x i32> [[X:%.*]], splat (i32 42)
-; CHECK-NEXT:    [[TMP2:%.*]] = icmp sgt <4 x i32> [[X]], splat (i32 17)
-; CHECK-NEXT:    [[TMP3:%.*]] = extractelement <4 x i1> [[TMP1]], i64 0
-; CHECK-NEXT:    [[TMP4:%.*]] = extractelement <4 x i1> [[TMP1]], i64 1
-; CHECK-NEXT:    [[S1:%.*]] = select i1 [[TMP3]], i1 [[TMP4]], i1 false
-; CHECK-NEXT:    [[TMP5:%.*]] = extractelement <4 x i1> [[TMP1]], i64 2
-; CHECK-NEXT:    [[S2:%.*]] = select i1 [[S1]], i1 [[TMP5]], i1 false
-; CHECK-NEXT:    call void @use1(i1 [[S2]])
-; CHECK-NEXT:    [[TMP6:%.*]] = freeze <4 x i1> [[TMP2]]
-; CHECK-NEXT:    [[TMP7:%.*]] = call i1 @llvm.vector.reduce.and.v4i1(<4 x i1> [[TMP6]])
-; CHECK-NEXT:    [[TMP8:%.*]] = extractelement <4 x i1> [[TMP1]], i64 3
-; CHECK-NEXT:    [[OP_RDX:%.*]] = select i1 [[TMP7]], i1 [[TMP8]], i1 false
-; CHECK-NEXT:    [[OP_RDX1:%.*]] = select i1 [[S2]], i1 [[OP_RDX]], i1 false
-; CHECK-NEXT:    ret i1 [[OP_RDX1]]
+; SSE-LABEL: @logical_and_icmp_clamp_extra_use_select(
+; SSE-NEXT:    [[TMP1:%.*]] = icmp slt <4 x i32> [[X:%.*]], splat (i32 42)
+; SSE-NEXT:    [[TMP2:%.*]] = icmp sgt <4 x i32> [[X]], splat (i32 17)
+; SSE-NEXT:    [[TMP3:%.*]] = extractelement <4 x i1> [[TMP1]], i64 0
+; SSE-NEXT:    [[TMP4:%.*]] = extractelement <4 x i1> [[TMP1]], i64 1
+; SSE-NEXT:    [[S1:%.*]] = select i1 [[TMP3]], i1 [[TMP4]], i1 false
+; SSE-NEXT:    [[TMP5:%.*]] = extractelement <4 x i1> [[TMP1]], i64 2
+; SSE-NEXT:    [[S2:%.*]] = select i1 [[S1]], i1 [[TMP5]], i1 false
+; SSE-NEXT:    call void @use1(i1 [[S2]])
+; SSE-NEXT:    [[TMP6:%.*]] = freeze <4 x i1> [[TMP2]]
+; SSE-NEXT:    [[TMP7:%.*]] = call i1 @llvm.vector.reduce.and.v4i1(<4 x i1> [[TMP6]])
+; SSE-NEXT:    [[TMP8:%.*]] = extractelement <4 x i1> [[TMP1]], i64 3
+; SSE-NEXT:    [[OP_RDX:%.*]] = select i1 [[TMP7]], i1 [[TMP8]], i1 false
+; SSE-NEXT:    [[OP_RDX1:%.*]] = select i1 [[S2]], i1 [[OP_RDX]], i1 false
+; SSE-NEXT:    ret i1 [[OP_RDX1]]
+;
+; AVX-LABEL: @logical_and_icmp_clamp_extra_use_select(
+; AVX-NEXT:    [[TMP1:%.*]] = icmp slt <4 x i32> [[X:%.*]], splat (i32 42)
+; AVX-NEXT:    [[TMP2:%.*]] = icmp sgt <4 x i32> [[X]], splat (i32 17)
+; AVX-NEXT:    [[TMP3:%.*]] = extractelement <4 x i1> [[TMP1]], i64 0
+; AVX-NEXT:    [[TMP4:%.*]] = extractelement <4 x i1> [[TMP1]], i64 1
+; AVX-NEXT:    [[S1:%.*]] = select i1 [[TMP3]], i1 [[TMP4]], i1 false
+; AVX-NEXT:    [[TMP5:%.*]] = extractelement <4 x i1> [[TMP1]], i64 2
+; AVX-NEXT:    [[S2:%.*]] = select i1 [[S1]], i1 [[TMP5]], i1 false
+; AVX-NEXT:    call void @use1(i1 [[S2]])
+; AVX-NEXT:    [[TMP6:%.*]] = freeze <4 x i1> [[TMP2]]
+; AVX-NEXT:    [[TMP7:%.*]] = bitcast <4 x i1> [[TMP6]] to i4
+; AVX-NEXT:    [[TMP8:%.*]] = icmp eq i4 [[TMP7]], -1
+; AVX-NEXT:    [[TMP9:%.*]] = extractelement <4 x i1> [[TMP1]], i64 3
+; AVX-NEXT:    [[OP_RDX:%.*]] = select i1 [[TMP8]], i1 [[TMP9]], i1 false
+; AVX-NEXT:    [[OP_RDX1:%.*]] = select i1 [[S2]], i1 [[OP_RDX]], i1 false
+; AVX-NEXT:    ret i1 [[OP_RDX1]]
 ;
   %x0 = extractelement <4 x i32> %x, i32 0
   %x1 = extractelement <4 x i32> %x, i32 1
@@ -315,13 +388,22 @@ define i1 @logical_and_icmp_clamp_extra_use_select(<4 x i32> %x) {
 }
 
 define i1 @logical_and_icmp_clamp_v8i32(<8 x i32> %x, <8 x i32> %y) {
-; CHECK-LABEL: @logical_and_icmp_clamp_v8i32(
-; CHECK-NEXT:    [[TMP1:%.*]] = shufflevector <8 x i32> [[X:%.*]], <8 x i32> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 0, i32 1, i32 2, i32 3>
-; CHECK-NEXT:    [[TMP3:%.*]] = shufflevector <8 x i32> [[Y:%.*]], <8 x i32> <i32 42, i32 42, i32 42, i32 42, i32 poison, i32 poison, i32 poison, i32 poison>, <8 x i32> <i32 8, i32 9, i32 10, i32 11, i32 0, i32 1, i32 2, i32 3>
-; CHECK-NEXT:    [[TMP4:%.*]] = icmp slt <8 x i32> [[TMP1]], [[TMP3]]
-; CHECK-NEXT:    [[TMP5:%.*]] = freeze <8 x i1> [[TMP4]]
-; CHECK-NEXT:    [[TMP6:%.*]] = call i1 @llvm.vector.reduce.and.v8i1(<8 x i1> [[TMP5]])
-; CHECK-NEXT:    ret i1 [[TMP6]]
+; SSE-LABEL: @logical_and_icmp_clamp_v8i32(
+; SSE-NEXT:    [[TMP1:%.*]] = shufflevector <8 x i32> [[X:%.*]], <8 x i32> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 0, i32 1, i32 2, i32 3>
+; SSE-NEXT:    [[TMP2:%.*]] = shufflevector <8 x i32> [[Y:%.*]], <8 x i32> <i32 42, i32 42, i32 42, i32 42, i32 poison, i32 poison, i32 poison, i32 poison>, <8 x i32> <i32 8, i32 9, i32 10, i32 11, i32 0, i32 1, i32 2, i32 3>
+; SSE-NEXT:    [[TMP3:%.*]] = icmp slt <8 x i32> [[TMP1]], [[TMP2]]
+; SSE-NEXT:    [[TMP4:%.*]] = freeze <8 x i1> [[TMP3]]
+; SSE-NEXT:    [[TMP5:%.*]] = call i1 @llvm.vector.reduce.and.v8i1(<8 x i1> [[TMP4]])
+; SSE-NEXT:    ret i1 [[TMP5]]
+;
+; AVX-LABEL: @logical_and_icmp_clamp_v8i32(
+; AVX-NEXT:    [[TMP1:%.*]] = shufflevector <8 x i32> [[X:%.*]], <8 x i32> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 0, i32 1, i32 2, i32 3>
+; AVX-NEXT:    [[TMP2:%.*]] = shufflevector <8 x i32> [[Y:%.*]], <8 x i32> <i32 42, i32 42, i32 42, i32 42, i32 poison, i32 poison, i32 poison, i32 poison>, <8 x i32> <i32 8, i32 9, i32 10, i32 11, i32 0, i32 1, i32 2, i32 3>
+; AVX-NEXT:    [[TMP3:%.*]] = icmp slt <8 x i32> [[TMP1]], [[TMP2]]
+; AVX-NEXT:    [[TMP4:%.*]] = freeze <8 x i1> [[TMP3]]
+; AVX-NEXT:    [[TMP5:%.*]] = bitcast <8 x i1> [[TMP4]] to i8
+; AVX-NEXT:    [[TMP6:%.*]] = icmp eq i8 [[TMP5]], -1
+; AVX-NEXT:    ret i1 [[TMP6]]
 ;
   %x0 = extractelement <8 x i32> %x, i32 0
   %x1 = extractelement <8 x i32> %x, i32 1
@@ -350,22 +432,40 @@ define i1 @logical_and_icmp_clamp_v8i32(<8 x i32> %x, <8 x i32> %y) {
 }
 
 define i1 @logical_and_icmp_clamp_partial(<4 x i32> %x) {
-; CHECK-LABEL: @logical_and_icmp_clamp_partial(
-; CHECK-NEXT:    [[TMP1:%.*]] = extractelement <4 x i32> [[X:%.*]], i64 2
-; CHECK-NEXT:    [[TMP2:%.*]] = shufflevector <4 x i32> [[X]], <4 x i32> poison, <2 x i32> <i32 0, i32 1>
-; CHECK-NEXT:    [[TMP3:%.*]] = icmp slt <2 x i32> [[TMP2]], splat (i32 42)
-; CHECK-NEXT:    [[C2:%.*]] = icmp slt i32 [[TMP1]], 42
-; CHECK-NEXT:    [[TMP4:%.*]] = icmp sgt <4 x i32> [[X]], splat (i32 17)
-; CHECK-NEXT:    [[TMP5:%.*]] = freeze <4 x i1> [[TMP4]]
-; CHECK-NEXT:    [[TMP6:%.*]] = call i1 @llvm.vector.reduce.and.v4i1(<4 x i1> [[TMP5]])
-; CHECK-NEXT:    [[TMP7:%.*]] = extractelement <2 x i1> [[TMP3]], i64 0
-; CHECK-NEXT:    [[OP_RDX:%.*]] = select i1 [[TMP6]], i1 [[TMP7]], i1 false
-; CHECK-NEXT:    [[TMP8:%.*]] = extractelement <2 x i1> [[TMP3]], i64 1
-; CHECK-NEXT:    [[TMP9:%.*]] = freeze i1 [[TMP8]]
-; CHECK-NEXT:    [[OP_RDX1:%.*]] = select i1 [[TMP9]], i1 [[C2]], i1 false
-; CHECK-NEXT:    [[TMP10:%.*]] = freeze i1 [[OP_RDX]]
-; CHECK-NEXT:    [[OP_RDX2:%.*]] = select i1 [[TMP10]], i1 [[OP_RDX1]], i1 false
-; CHECK-NEXT:    ret i1 [[OP_RDX2]]
+; SSE-LABEL: @logical_and_icmp_clamp_partial(
+; SSE-NEXT:    [[TMP1:%.*]] = extractelement <4 x i32> [[X:%.*]], i64 2
+; SSE-NEXT:    [[TMP2:%.*]] = shufflevector <4 x i32> [[X]], <4 x i32> poison, <2 x i32> <i32 0, i32 1>
+; SSE-NEXT:    [[TMP3:%.*]] = icmp slt <2 x i32> [[TMP2]], splat (i32 42)
+; SSE-NEXT:    [[C2:%.*]] = icmp slt i32 [[TMP1]], 42
+; SSE-NEXT:    [[TMP4:%.*]] = icmp sgt <4 x i32> [[X]], splat (i32 17)
+; SSE-NEXT:    [[TMP5:%.*]] = freeze <4 x i1> [[TMP4]]
+; SSE-NEXT:    [[TMP6:%.*]] = call i1 @llvm.vector.reduce.and.v4i1(<4 x i1> [[TMP5]])
+; SSE-NEXT:    [[TMP7:%.*]] = extractelement <2 x i1> [[TMP3]], i64 0
+; SSE-NEXT:    [[OP_RDX:%.*]] = select i1 [[TMP6]], i1 [[TMP7]], i1 false
+; SSE-NEXT:    [[TMP8:%.*]] = extractelement <2 x i1> [[TMP3]], i64 1
+; SSE-NEXT:    [[TMP9:%.*]] = freeze i1 [[TMP8]]
+; SSE-NEXT:    [[OP_RDX1:%.*]] = select i1 [[TMP9]], i1 [[C2]], i1 false
+; SSE-NEXT:    [[TMP10:%.*]] = freeze i1 [[OP_RDX]]
+; SSE-NEXT:    [[OP_RDX2:%.*]] = select i1 [[TMP10]], i1 [[OP_RDX1]], i1 false
+; SSE-NEXT:    ret i1 [[OP_RDX2]]
+;
+; AVX-LABEL: @logical_and_icmp_clamp_partial(
+; AVX-NEXT:    [[TMP1:%.*]] = extractelement <4 x i32> [[X:%.*]], i64 2
+; AVX-NEXT:    [[TMP2:%.*]] = shufflevector <4 x i32> [[X]], <4 x i32> poison, <2 x i32> <i32 0, i32 1>
+; AVX-NEXT:    [[TMP3:%.*]] = icmp slt <2 x i32> [[TMP2]], splat (i32 42)
+; AVX-NEXT:    [[C2:%.*]] = icmp slt i32 [[TMP1]], 42
+; AVX-NEXT:    [[TMP4:%.*]] = icmp sgt <4 x i32> [[X]], splat (i32 17)
+; AVX-NEXT:    [[TMP5:%.*]] = freeze <4 x i1> [[TMP4]]
+; AVX-NEXT:    [[TMP6:%.*]] = bitcast <4 x i1> [[TMP5]] to i4
+; AVX-NEXT:    [[TMP7:%.*]] = icmp eq i4 [[TMP6]], -1
+; AVX-NEXT:    [[TMP8:%.*]] = extractelement <2 x i1> [[TMP3]], i64 0
+; AVX-NEXT:    [[OP_RDX:%.*]] = select i1 [[TMP7]], i1 [[TMP8]], i1 false
+; AVX-NEXT:    [[TMP9:%.*]] = extractelement <2 x i1> [[TMP3]], i64 1
+; AVX-NEXT:    [[TMP10:%.*]] = freeze i1 [[TMP9]]
+; AVX-NEXT:    [[OP_RDX1:%.*]] = select i1 [[TMP10]], i1 [[C2]], i1 false
+; AVX-NEXT:    [[TMP11:%.*]] = freeze i1 [[OP_RDX]]
+; AVX-NEXT:    [[OP_RDX2:%.*]] = select i1 [[TMP11]], i1 [[OP_RDX1]], i1 false
+; AVX-NEXT:    ret i1 [[OP_RDX2]]
 ;
   %x0 = extractelement <4 x i32> %x, i32 0
   %x1 = extractelement <4 x i32> %x, i32 1
@@ -390,17 +490,30 @@ define i1 @logical_and_icmp_clamp_partial(<4 x i32> %x) {
 }
 
 define i1 @logical_and_icmp_clamp_pred_
diff (<4 x i32> %x) {
-; CHECK-LABEL: @logical_and_icmp_clamp_pred_
diff (
-; CHECK-NEXT:    [[TMP1:%.*]] = shufflevector <4 x i32> [[X:%.*]], <4 x i32> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 0, i32 1, i32 2, i32 3>
-; CHECK-NEXT:    [[TMP2:%.*]] = shufflevector <8 x i32> [[TMP1]], <8 x i32> <i32 poison, i32 poison, i32 poison, i32 poison, i32 42, i32 42, i32 42, i32 poison>, <8 x i32> <i32 poison, i32 poison, i32 poison, i32 poison, i32 12, i32 13, i32 14, i32 3>
-; CHECK-NEXT:    [[TMP3:%.*]] = shufflevector <8 x i32> [[TMP2]], <8 x i32> [[TMP1]], <8 x i32> <i32 8, i32 9, i32 10, i32 11, i32 4, i32 5, i32 6, i32 7>
-; CHECK-NEXT:    [[TMP4:%.*]] = shufflevector <8 x i32> [[TMP1]], <8 x i32> <i32 17, i32 17, i32 17, i32 17, i32 poison, i32 poison, i32 poison, i32 42>, <8 x i32> <i32 8, i32 9, i32 10, i32 11, i32 0, i32 1, i32 2, i32 15>
-; CHECK-NEXT:    [[TMP5:%.*]] = icmp sgt <8 x i32> [[TMP3]], [[TMP4]]
-; CHECK-NEXT:    [[TMP6:%.*]] = icmp ult <8 x i32> [[TMP3]], [[TMP4]]
-; CHECK-NEXT:    [[TMP7:%.*]] = shufflevector <8 x i1> [[TMP5]], <8 x i1> [[TMP6]], <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 15>
-; CHECK-NEXT:    [[TMP8:%.*]] = freeze <8 x i1> [[TMP7]]
-; CHECK-NEXT:    [[TMP9:%.*]] = call i1 @llvm.vector.reduce.and.v8i1(<8 x i1> [[TMP8]])
-; CHECK-NEXT:    ret i1 [[TMP9]]
+; SSE-LABEL: @logical_and_icmp_clamp_pred_
diff (
+; SSE-NEXT:    [[TMP1:%.*]] = shufflevector <4 x i32> [[X:%.*]], <4 x i32> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 0, i32 1, i32 2, i32 3>
+; SSE-NEXT:    [[TMP2:%.*]] = shufflevector <8 x i32> [[TMP1]], <8 x i32> <i32 poison, i32 poison, i32 poison, i32 poison, i32 42, i32 42, i32 42, i32 poison>, <8 x i32> <i32 poison, i32 poison, i32 poison, i32 poison, i32 12, i32 13, i32 14, i32 3>
+; SSE-NEXT:    [[TMP3:%.*]] = shufflevector <8 x i32> [[TMP2]], <8 x i32> [[TMP1]], <8 x i32> <i32 8, i32 9, i32 10, i32 11, i32 4, i32 5, i32 6, i32 7>
+; SSE-NEXT:    [[TMP4:%.*]] = shufflevector <8 x i32> [[TMP1]], <8 x i32> <i32 17, i32 17, i32 17, i32 17, i32 poison, i32 poison, i32 poison, i32 42>, <8 x i32> <i32 8, i32 9, i32 10, i32 11, i32 0, i32 1, i32 2, i32 15>
+; SSE-NEXT:    [[TMP5:%.*]] = icmp sgt <8 x i32> [[TMP3]], [[TMP4]]
+; SSE-NEXT:    [[TMP6:%.*]] = icmp ult <8 x i32> [[TMP3]], [[TMP4]]
+; SSE-NEXT:    [[TMP7:%.*]] = shufflevector <8 x i1> [[TMP5]], <8 x i1> [[TMP6]], <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 15>
+; SSE-NEXT:    [[TMP8:%.*]] = freeze <8 x i1> [[TMP7]]
+; SSE-NEXT:    [[TMP9:%.*]] = call i1 @llvm.vector.reduce.and.v8i1(<8 x i1> [[TMP8]])
+; SSE-NEXT:    ret i1 [[TMP9]]
+;
+; AVX-LABEL: @logical_and_icmp_clamp_pred_
diff (
+; AVX-NEXT:    [[TMP1:%.*]] = shufflevector <4 x i32> [[X:%.*]], <4 x i32> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 0, i32 1, i32 2, i32 3>
+; AVX-NEXT:    [[TMP2:%.*]] = shufflevector <8 x i32> [[TMP1]], <8 x i32> <i32 poison, i32 poison, i32 poison, i32 poison, i32 42, i32 42, i32 42, i32 poison>, <8 x i32> <i32 poison, i32 poison, i32 poison, i32 poison, i32 12, i32 13, i32 14, i32 3>
+; AVX-NEXT:    [[TMP3:%.*]] = shufflevector <8 x i32> [[TMP2]], <8 x i32> [[TMP1]], <8 x i32> <i32 8, i32 9, i32 10, i32 11, i32 4, i32 5, i32 6, i32 7>
+; AVX-NEXT:    [[TMP4:%.*]] = shufflevector <8 x i32> [[TMP1]], <8 x i32> <i32 17, i32 17, i32 17, i32 17, i32 poison, i32 poison, i32 poison, i32 42>, <8 x i32> <i32 8, i32 9, i32 10, i32 11, i32 0, i32 1, i32 2, i32 15>
+; AVX-NEXT:    [[TMP5:%.*]] = icmp sgt <8 x i32> [[TMP3]], [[TMP4]]
+; AVX-NEXT:    [[TMP6:%.*]] = icmp ult <8 x i32> [[TMP3]], [[TMP4]]
+; AVX-NEXT:    [[TMP7:%.*]] = shufflevector <8 x i1> [[TMP5]], <8 x i1> [[TMP6]], <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 15>
+; AVX-NEXT:    [[TMP8:%.*]] = freeze <8 x i1> [[TMP7]]
+; AVX-NEXT:    [[TMP9:%.*]] = bitcast <8 x i1> [[TMP8]] to i8
+; AVX-NEXT:    [[TMP10:%.*]] = icmp eq i8 [[TMP9]], -1
+; AVX-NEXT:    ret i1 [[TMP10]]
 ;
   %x0 = extractelement <4 x i32> %x, i32 0
   %x1 = extractelement <4 x i32> %x, i32 1
@@ -425,12 +538,20 @@ define i1 @logical_and_icmp_clamp_pred_
diff (<4 x i32> %x) {
 }
 
 define i1 @logical_and_icmp_extra_op(<4 x i32> %x, <4 x i32> %y, i1 %c) {
-; CHECK-LABEL: @logical_and_icmp_extra_op(
-; CHECK-NEXT:    [[TMP1:%.*]] = icmp slt <4 x i32> [[X:%.*]], [[Y:%.*]]
-; CHECK-NEXT:    [[TMP2:%.*]] = freeze <4 x i1> [[TMP1]]
-; CHECK-NEXT:    [[TMP3:%.*]] = call i1 @llvm.vector.reduce.and.v4i1(<4 x i1> [[TMP2]])
-; CHECK-NEXT:    [[OP_RDX:%.*]] = select i1 [[C:%.*]], i1 [[TMP3]], i1 false
-; CHECK-NEXT:    ret i1 [[OP_RDX]]
+; SSE-LABEL: @logical_and_icmp_extra_op(
+; SSE-NEXT:    [[TMP1:%.*]] = icmp slt <4 x i32> [[X:%.*]], [[Y:%.*]]
+; SSE-NEXT:    [[TMP2:%.*]] = freeze <4 x i1> [[TMP1]]
+; SSE-NEXT:    [[TMP3:%.*]] = call i1 @llvm.vector.reduce.and.v4i1(<4 x i1> [[TMP2]])
+; SSE-NEXT:    [[OP_RDX:%.*]] = select i1 [[C:%.*]], i1 [[TMP3]], i1 false
+; SSE-NEXT:    ret i1 [[OP_RDX]]
+;
+; AVX-LABEL: @logical_and_icmp_extra_op(
+; AVX-NEXT:    [[TMP1:%.*]] = icmp slt <4 x i32> [[X:%.*]], [[Y:%.*]]
+; AVX-NEXT:    [[TMP2:%.*]] = freeze <4 x i1> [[TMP1]]
+; AVX-NEXT:    [[TMP3:%.*]] = bitcast <4 x i1> [[TMP2]] to i4
+; AVX-NEXT:    [[TMP4:%.*]] = icmp eq i4 [[TMP3]], -1
+; AVX-NEXT:    [[OP_RDX:%.*]] = select i1 [[C:%.*]], i1 [[TMP4]], i1 false
+; AVX-NEXT:    ret i1 [[OP_RDX]]
 ;
   %x0 = extractelement <4 x i32> %x, i32 0
   %x1 = extractelement <4 x i32> %x, i32 1
@@ -453,12 +574,20 @@ define i1 @logical_and_icmp_extra_op(<4 x i32> %x, <4 x i32> %y, i1 %c) {
 }
 
 define i1 @logical_or_icmp_extra_op(<4 x i32> %x, <4 x i32> %y, i1 %c) {
-; CHECK-LABEL: @logical_or_icmp_extra_op(
-; CHECK-NEXT:    [[TMP1:%.*]] = icmp slt <4 x i32> [[X:%.*]], [[Y:%.*]]
-; CHECK-NEXT:    [[TMP2:%.*]] = freeze <4 x i1> [[TMP1]]
-; CHECK-NEXT:    [[TMP3:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP2]])
-; CHECK-NEXT:    [[OP_RDX:%.*]] = select i1 [[C:%.*]], i1 true, i1 [[TMP3]]
-; CHECK-NEXT:    ret i1 [[OP_RDX]]
+; SSE-LABEL: @logical_or_icmp_extra_op(
+; SSE-NEXT:    [[TMP1:%.*]] = icmp slt <4 x i32> [[X:%.*]], [[Y:%.*]]
+; SSE-NEXT:    [[TMP2:%.*]] = freeze <4 x i1> [[TMP1]]
+; SSE-NEXT:    [[TMP3:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP2]])
+; SSE-NEXT:    [[OP_RDX:%.*]] = select i1 [[C:%.*]], i1 true, i1 [[TMP3]]
+; SSE-NEXT:    ret i1 [[OP_RDX]]
+;
+; AVX-LABEL: @logical_or_icmp_extra_op(
+; AVX-NEXT:    [[TMP1:%.*]] = icmp slt <4 x i32> [[X:%.*]], [[Y:%.*]]
+; AVX-NEXT:    [[TMP2:%.*]] = freeze <4 x i1> [[TMP1]]
+; AVX-NEXT:    [[TMP3:%.*]] = bitcast <4 x i1> [[TMP2]] to i4
+; AVX-NEXT:    [[TMP4:%.*]] = icmp ne i4 [[TMP3]], 0
+; AVX-NEXT:    [[OP_RDX:%.*]] = select i1 [[C:%.*]], i1 true, i1 [[TMP4]]
+; AVX-NEXT:    ret i1 [[OP_RDX]]
 ;
   %x0 = extractelement <4 x i32> %x, i32 0
   %x1 = extractelement <4 x i32> %x, i32 1
@@ -481,16 +610,28 @@ define i1 @logical_or_icmp_extra_op(<4 x i32> %x, <4 x i32> %y, i1 %c) {
 }
 
 define i1 @logical_and_icmp_extra_args(<4 x i32> %x, i1 %c0, i1 %c1, i1 %c2) {
-; CHECK-LABEL: @logical_and_icmp_extra_args(
-; CHECK-NEXT:    [[TMP1:%.*]] = icmp sgt <4 x i32> [[X:%.*]], splat (i32 17)
-; CHECK-NEXT:    [[TMP2:%.*]] = freeze <4 x i1> [[TMP1]]
-; CHECK-NEXT:    [[TMP3:%.*]] = call i1 @llvm.vector.reduce.and.v4i1(<4 x i1> [[TMP2]])
-; CHECK-NEXT:    [[OP_RDX:%.*]] = select i1 [[TMP3]], i1 [[C0:%.*]], i1 false
-; CHECK-NEXT:    [[TMP4:%.*]] = freeze i1 [[C1:%.*]]
-; CHECK-NEXT:    [[OP_RDX1:%.*]] = select i1 [[TMP4]], i1 [[C2:%.*]], i1 false
-; CHECK-NEXT:    [[TMP5:%.*]] = freeze i1 [[OP_RDX]]
-; CHECK-NEXT:    [[OP_RDX2:%.*]] = select i1 [[TMP5]], i1 [[OP_RDX1]], i1 false
-; CHECK-NEXT:    ret i1 [[OP_RDX2]]
+; SSE-LABEL: @logical_and_icmp_extra_args(
+; SSE-NEXT:    [[TMP1:%.*]] = icmp sgt <4 x i32> [[X:%.*]], splat (i32 17)
+; SSE-NEXT:    [[TMP2:%.*]] = freeze <4 x i1> [[TMP1]]
+; SSE-NEXT:    [[TMP3:%.*]] = call i1 @llvm.vector.reduce.and.v4i1(<4 x i1> [[TMP2]])
+; SSE-NEXT:    [[OP_RDX:%.*]] = select i1 [[TMP3]], i1 [[C0:%.*]], i1 false
+; SSE-NEXT:    [[TMP4:%.*]] = freeze i1 [[C1:%.*]]
+; SSE-NEXT:    [[OP_RDX1:%.*]] = select i1 [[TMP4]], i1 [[C2:%.*]], i1 false
+; SSE-NEXT:    [[TMP5:%.*]] = freeze i1 [[OP_RDX]]
+; SSE-NEXT:    [[OP_RDX2:%.*]] = select i1 [[TMP5]], i1 [[OP_RDX1]], i1 false
+; SSE-NEXT:    ret i1 [[OP_RDX2]]
+;
+; AVX-LABEL: @logical_and_icmp_extra_args(
+; AVX-NEXT:    [[TMP1:%.*]] = icmp sgt <4 x i32> [[X:%.*]], splat (i32 17)
+; AVX-NEXT:    [[TMP2:%.*]] = freeze <4 x i1> [[TMP1]]
+; AVX-NEXT:    [[TMP3:%.*]] = bitcast <4 x i1> [[TMP2]] to i4
+; AVX-NEXT:    [[TMP4:%.*]] = icmp eq i4 [[TMP3]], -1
+; AVX-NEXT:    [[OP_RDX:%.*]] = select i1 [[TMP4]], i1 [[C0:%.*]], i1 false
+; AVX-NEXT:    [[TMP5:%.*]] = freeze i1 [[C1:%.*]]
+; AVX-NEXT:    [[OP_RDX1:%.*]] = select i1 [[TMP5]], i1 [[C2:%.*]], i1 false
+; AVX-NEXT:    [[TMP6:%.*]] = freeze i1 [[OP_RDX]]
+; AVX-NEXT:    [[OP_RDX2:%.*]] = select i1 [[TMP6]], i1 [[OP_RDX1]], i1 false
+; AVX-NEXT:    ret i1 [[OP_RDX2]]
 ;
   %x0 = extractelement <4 x i32> %x, i32 0
   %x1 = extractelement <4 x i32> %x, i32 1

diff  --git a/llvm/test/Transforms/SLPVectorizer/X86/reduction2.ll b/llvm/test/Transforms/SLPVectorizer/X86/reduction2.ll
index 03cdc54eb779f..28c09dececd5b 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/reduction2.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/reduction2.ll
@@ -94,17 +94,12 @@ define i1 @fcmp_lt_gt(double %a, double %b, double %c) {
 ; CHECK-NEXT:    [[TMP2:%.*]] = insertelement <2 x double> poison, double [[MUL]], i64 0
 ; CHECK-NEXT:    [[TMP3:%.*]] = shufflevector <2 x double> [[TMP2]], <2 x double> poison, <2 x i32> zeroinitializer
 ; CHECK-NEXT:    [[TMP4:%.*]] = fdiv <2 x double> [[TMP1]], [[TMP3]]
-; CHECK-NEXT:    [[TMP8:%.*]] = extractelement <2 x double> [[TMP4]], i64 1
-; CHECK-NEXT:    [[CMP:%.*]] = fcmp olt double [[TMP8]], f0x3EB0C6F7A0B5ED8D
-; CHECK-NEXT:    [[TMP9:%.*]] = extractelement <2 x double> [[TMP4]], i64 0
-; CHECK-NEXT:    [[CMP4:%.*]] = fcmp olt double [[TMP9]], f0x3EB0C6F7A0B5ED8D
-; CHECK-NEXT:    [[OR_COND:%.*]] = and i1 [[CMP]], [[CMP4]]
+; CHECK-NEXT:    [[TMP5:%.*]] = fcmp olt <2 x double> [[TMP4]], splat (double f0x3EB0C6F7A0B5ED8D)
+; CHECK-NEXT:    [[OR_COND:%.*]] = call i1 @llvm.vector.reduce.and.v2i1(<2 x i1> [[TMP5]])
 ; CHECK-NEXT:    br i1 [[OR_COND]], label [[CLEANUP:%.*]], label [[LOR_LHS_FALSE:%.*]]
 ; CHECK:       lor.lhs.false:
 ; CHECK-NEXT:    [[TMP7:%.*]] = fcmp ule <2 x double> [[TMP4]], splat (double 1.000000e+00)
-; CHECK-NEXT:    [[TMP11:%.*]] = extractelement <2 x i1> [[TMP7]], i64 0
-; CHECK-NEXT:    [[TMP12:%.*]] = extractelement <2 x i1> [[TMP7]], i64 1
-; CHECK-NEXT:    [[NOT_OR_COND9:%.*]] = or i1 [[TMP11]], [[TMP12]]
+; CHECK-NEXT:    [[NOT_OR_COND9:%.*]] = call i1 @llvm.vector.reduce.or.v2i1(<2 x i1> [[TMP7]])
 ; CHECK-NEXT:    ret i1 [[NOT_OR_COND9]]
 ; CHECK:       cleanup:
 ; CHECK-NEXT:    ret i1 false
@@ -143,9 +138,7 @@ define i1 @fcmp_lt(double %a, double %b, double %c) {
 ; CHECK-NEXT:    [[TMP4:%.*]] = shufflevector <2 x double> [[TMP3]], <2 x double> poison, <2 x i32> zeroinitializer
 ; CHECK-NEXT:    [[TMP5:%.*]] = fdiv <2 x double> [[TMP2]], [[TMP4]]
 ; CHECK-NEXT:    [[TMP6:%.*]] = fcmp uge <2 x double> [[TMP5]], splat (double f0x3EB0C6F7A0B5ED8D)
-; CHECK-NEXT:    [[TMP10:%.*]] = extractelement <2 x i1> [[TMP6]], i64 0
-; CHECK-NEXT:    [[TMP11:%.*]] = extractelement <2 x i1> [[TMP6]], i64 1
-; CHECK-NEXT:    [[NOT_OR_COND:%.*]] = or i1 [[TMP10]], [[TMP11]]
+; CHECK-NEXT:    [[NOT_OR_COND:%.*]] = call i1 @llvm.vector.reduce.or.v2i1(<2 x i1> [[TMP6]])
 ; CHECK-NEXT:    ret i1 [[NOT_OR_COND]]
 ;
   %fneg = fneg double %b

diff  --git a/llvm/test/Transforms/SLPVectorizer/X86/reorder-vf-to-resize.ll b/llvm/test/Transforms/SLPVectorizer/X86/reorder-vf-to-resize.ll
index 3f61fa3d44bc7..180284942db9c 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/reorder-vf-to-resize.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/reorder-vf-to-resize.ll
@@ -11,7 +11,8 @@ define void @main(ptr %0) {
 ; CHECK-NEXT:    [[TMP6:%.*]] = fmul <4 x double> [[TMP5]], zeroinitializer
 ; CHECK-NEXT:    [[TMP7:%.*]] = call <4 x double> @llvm.fabs.v4f64(<4 x double> [[TMP6]])
 ; CHECK-NEXT:    [[TMP8:%.*]] = fcmp oeq <4 x double> [[TMP7]], zeroinitializer
-; CHECK-NEXT:    [[TMP9:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP8]])
+; CHECK-NEXT:    [[TMP12:%.*]] = bitcast <4 x i1> [[TMP8]] to i4
+; CHECK-NEXT:    [[TMP9:%.*]] = icmp ne i4 [[TMP12]], 0
 ; CHECK-NEXT:    [[TMP10:%.*]] = select i1 [[TMP9]], double 0.000000e+00, double 0.000000e+00
 ; CHECK-NEXT:    store double [[TMP10]], ptr null, align 8
 ; CHECK-NEXT:    ret void

diff  --git a/llvm/test/Transforms/SLPVectorizer/zext-incoming-for-neg-icmp.ll b/llvm/test/Transforms/SLPVectorizer/zext-incoming-for-neg-icmp.ll
index 0e4b38a48d509..8562aa0682462 100644
--- a/llvm/test/Transforms/SLPVectorizer/zext-incoming-for-neg-icmp.ll
+++ b/llvm/test/Transforms/SLPVectorizer/zext-incoming-for-neg-icmp.ll
@@ -1,24 +1,39 @@
 ; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 4
-; RUN: %if x86-registered-target %{ opt -S --passes=slp-vectorizer -mtriple=x86_64-unknown-linux-gnu < %s | FileCheck %s %}
-; RUN: %if aarch64-registered-target %{ opt -S --passes=slp-vectorizer -mtriple=aarch64-unknown-linux-gnu < %s | FileCheck %s %}
+; RUN: %if x86-registered-target %{ opt -S --passes=slp-vectorizer -mtriple=x86_64-unknown-linux-gnu < %s | FileCheck %s --check-prefixes=X86 %}
+; RUN: %if aarch64-registered-target %{ opt -S --passes=slp-vectorizer -mtriple=aarch64-unknown-linux-gnu < %s | FileCheck %s --check-prefixes=AARCH64 %}
 
 define i32 @test(i32 %a, i8 %b, i8 %c) {
-; CHECK-LABEL: define i32 @test(
-; CHECK-SAME: i32 [[A:%.*]], i8 [[B:%.*]], i8 [[C:%.*]]) {
-; CHECK-NEXT:  entry:
-; CHECK-NEXT:    [[TMP0:%.*]] = insertelement <4 x i8> poison, i8 [[C]], i64 0
-; CHECK-NEXT:    [[TMP1:%.*]] = shufflevector <4 x i8> [[TMP0]], <4 x i8> poison, <4 x i32> zeroinitializer
-; CHECK-NEXT:    [[TMP2:%.*]] = add <4 x i8> [[TMP1]], <i8 -1, i8 -2, i8 -3, i8 -4>
-; CHECK-NEXT:    [[TMP3:%.*]] = insertelement <4 x i8> poison, i8 [[B]], i64 0
-; CHECK-NEXT:    [[TMP4:%.*]] = shufflevector <4 x i8> [[TMP3]], <4 x i8> poison, <4 x i32> zeroinitializer
-; CHECK-NEXT:    [[TMP8:%.*]] = zext <4 x i8> [[TMP2]] to <4 x i16>
-; CHECK-NEXT:    [[TMP9:%.*]] = sext <4 x i8> [[TMP4]] to <4 x i16>
-; CHECK-NEXT:    [[TMP5:%.*]] = icmp sle <4 x i16> [[TMP8]], [[TMP9]]
-; CHECK-NEXT:    [[TMP10:%.*]] = bitcast <4 x i1> [[TMP5]] to i4
-; CHECK-NEXT:    [[TMP11:%.*]] = call i4 @llvm.ctpop.i4(i4 [[TMP10]])
-; CHECK-NEXT:    [[TMP7:%.*]] = zext i4 [[TMP11]] to i32
-; CHECK-NEXT:    [[OP_RDX:%.*]] = add i32 [[TMP7]], [[A]]
-; CHECK-NEXT:    ret i32 [[OP_RDX]]
+; X86-LABEL: define i32 @test(
+; X86-SAME: i32 [[A:%.*]], i8 [[B:%.*]], i8 [[C:%.*]]) {
+; X86-NEXT:  entry:
+; X86-NEXT:    [[TMP0:%.*]] = insertelement <4 x i8> poison, i8 [[C]], i64 0
+; X86-NEXT:    [[TMP1:%.*]] = shufflevector <4 x i8> [[TMP0]], <4 x i8> poison, <4 x i32> zeroinitializer
+; X86-NEXT:    [[TMP2:%.*]] = add <4 x i8> [[TMP1]], <i8 -1, i8 -2, i8 -3, i8 -4>
+; X86-NEXT:    [[TMP3:%.*]] = insertelement <4 x i8> poison, i8 [[B]], i64 0
+; X86-NEXT:    [[TMP4:%.*]] = shufflevector <4 x i8> [[TMP3]], <4 x i8> poison, <4 x i32> zeroinitializer
+; X86-NEXT:    [[TMP5:%.*]] = zext <4 x i8> [[TMP2]] to <4 x i16>
+; X86-NEXT:    [[TMP6:%.*]] = sext <4 x i8> [[TMP4]] to <4 x i16>
+; X86-NEXT:    [[TMP7:%.*]] = icmp sle <4 x i16> [[TMP5]], [[TMP6]]
+; X86-NEXT:    [[TMP8:%.*]] = zext <4 x i1> [[TMP7]] to <4 x i32>
+; X86-NEXT:    [[TMP9:%.*]] = call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> [[TMP8]])
+; X86-NEXT:    [[OP_RDX:%.*]] = add i32 [[TMP9]], [[A]]
+; X86-NEXT:    ret i32 [[OP_RDX]]
+;
+; AARCH64-LABEL: define i32 @test(
+; AARCH64-SAME: i32 [[A:%.*]], i8 [[B:%.*]], i8 [[C:%.*]]) {
+; AARCH64-NEXT:  entry:
+; AARCH64-NEXT:    [[TMP0:%.*]] = insertelement <4 x i8> poison, i8 [[C]], i64 0
+; AARCH64-NEXT:    [[TMP1:%.*]] = shufflevector <4 x i8> [[TMP0]], <4 x i8> poison, <4 x i32> zeroinitializer
+; AARCH64-NEXT:    [[TMP2:%.*]] = add <4 x i8> [[TMP1]], <i8 -1, i8 -2, i8 -3, i8 -4>
+; AARCH64-NEXT:    [[TMP3:%.*]] = insertelement <4 x i8> poison, i8 [[B]], i64 0
+; AARCH64-NEXT:    [[TMP4:%.*]] = shufflevector <4 x i8> [[TMP3]], <4 x i8> poison, <4 x i32> zeroinitializer
+; AARCH64-NEXT:    [[TMP5:%.*]] = zext <4 x i8> [[TMP2]] to <4 x i16>
+; AARCH64-NEXT:    [[TMP6:%.*]] = sext <4 x i8> [[TMP4]] to <4 x i16>
+; AARCH64-NEXT:    [[TMP7:%.*]] = icmp sle <4 x i16> [[TMP5]], [[TMP6]]
+; AARCH64-NEXT:    [[TMP8:%.*]] = zext <4 x i1> [[TMP7]] to <4 x i32>
+; AARCH64-NEXT:    [[TMP10:%.*]] = call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> [[TMP8]])
+; AARCH64-NEXT:    [[OP_RDX:%.*]] = add i32 [[TMP10]], [[A]]
+; AARCH64-NEXT:    ret i32 [[OP_RDX]]
 ;
 entry:
   %0 = add i8 %c, -3


        


More information about the llvm-commits mailing list