[llvm] [SLP]Cost-based choice of the i1 reduction form (PR #223163)

Alexey Bataev via llvm-commits llvm-commits at lists.llvm.org
Sun Sep 13 08:35:25 PDT 2026


https://github.com/alexey-bataev updated https://github.com/llvm/llvm-project/pull/223163

>From d295bc8407ea549491a2300fde04efc1d81f5ee6 Mon Sep 17 00:00:00 2001
From: Alexey Bataev <a.bataev at outlook.com>
Date: Sat, 12 Sep 2026 11:00:45 -0700
Subject: [PATCH] =?UTF-8?q?[=F0=9D=98=80=F0=9D=97=BD=F0=9D=97=BF]=20initia?=
 =?UTF-8?q?l=20version?=
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit

Created using spr 1.3.7
---
 .../Transforms/Vectorize/SLPVectorizer.cpp    | 123 ++++--
 .../SLPVectorizer/SLPCostAnalysis.cpp         |  61 +++
 .../Vectorize/SLPVectorizer/SLPCostAnalysis.h |  11 +
 .../PhaseOrdering/X86/vector-reductions.ll    |  11 +-
 .../AArch64/externally-used-copyables.ll      | 232 +++++-----
 .../non-power-of-2-with-adjusted-gathers.ll   |  24 +-
 .../SLPVectorizer/AMDGPU/reduction-i1-mask.ll |  45 +-
 .../remark-zext-incoming-for-neg-icmp.ll      |   7 +-
 .../SLPVectorizer/SystemZ/cmp-ptr-minmax.ll   |   4 +-
 .../SLPVectorizer/X86/entries-different-vf.ll |   5 +-
 .../SLPVectorizer/X86/fabs-cost-softfp.ll     |   4 +-
 .../SLPVectorizer/X86/reduction-logical.ll    | 397 ++++++++++++------
 .../SLPVectorizer/X86/reduction2.ll           |  15 +-
 .../SLPVectorizer/X86/reorder-vf-to-resize.ll |   3 +-
 .../zext-incoming-for-neg-icmp.ll             |  53 ++-
 15 files changed, 622 insertions(+), 373 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 488bfb36b6ccc..cbcce81d94e52 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -2399,6 +2399,10 @@ class slpvectorizer::BoUpSLP {
                          SmallVectorImpl<Value *> &Op2,
                          OrdersType &ReorderIndices) const;
 
+  /// \returns Cast context for the given graph node.
+  TargetTransformInfo::CastContextHint
+  getCastContextHint(const TreeEntry &TE) const;
+
   ~BoUpSLP();
 
 private:
@@ -2461,10 +2465,6 @@ class slpvectorizer::BoUpSLP {
   /// one.
   Instruction *getRootEntryInstruction(const TreeEntry &Entry) const;
 
-  /// \returns Cast context for the given graph node.
-  TargetTransformInfo::CastContextHint
-  getCastContextHint(const TreeEntry &TE) const;
-
   /// \returns the scale of the given tree entry to the loop iteration.
   /// \p Scalar is the scalar value from the entry, if using the parent for the
   /// external use.
@@ -31155,7 +31155,8 @@ class HorizontalReduction {
     if (!VectorValuesAndScales.empty()) {
       Builder.setFastMathFlags(GroupRdxFMF);
       auto [Res, ResNegated] =
-          emitReduction(Builder, *TTI, ReductionRoot->getType());
+          emitReduction(Builder, *TTI, V.getCastContextHint(V.getRootNode()),
+                        ReductionRoot->getType());
       Builder.setFastMathFlags(RdxFMF);
       // The reduction result of the all-negated parts is subtracted in the
       // final combine.
@@ -31612,9 +31613,11 @@ class HorizontalReduction {
              "Expected floating point types for ordered reduction");
       Builder.SetCurrentDebugLocation(
           cast<Instruction>(ReductionRoot)->getDebugLoc());
-      VectorizedTree = createSingleOp(Builder, *TTI, SuccessRoot, /*Scale=*/1,
-                                      /*IsSigned=*/false, DestTy,
-                                      /*ReducedInTree=*/false, VectorizedTree);
+      VectorizedTree =
+          createSingleOp(Builder, *TTI, V.getCastContextHint(V.getRootNode()),
+                         SuccessRoot, /*Scale=*/1,
+                         /*IsSigned=*/false, DestTy,
+                         /*ReducedInTree=*/false, VectorizedTree);
 
       // Fold trailing scalars [SuccessStart+SuccessWidth, N).
       for (Value *RdxVal :
@@ -31658,8 +31661,9 @@ class HorizontalReduction {
   /// Creates the reduction from the given \p Vec vector value with the given
   /// scale \p Scale and signedness \p IsSigned.
   Value *createSingleOp(IRBuilderBase &Builder, const TargetTransformInfo &TTI,
-                        Value *Vec, unsigned Scale, bool IsSigned, Type *DestTy,
-                        bool ReducedInTree, Value *Start = nullptr) {
+                        TTI::CastContextHint Ctx, Value *Vec, unsigned Scale,
+                        bool IsSigned, Type *DestTy, bool ReducedInTree,
+                        Value *Start = nullptr) {
     Value *Rdx;
     if (ReducedInTree) {
       Rdx = Vec;
@@ -31698,7 +31702,7 @@ class HorizontalReduction {
           Rdx = createOp(Builder, RdxKind, Rdx, SubVec, "rdx.op", ReductionOps);
       }
     } else {
-      Rdx = emitReduction(Vec, Builder, &TTI, DestTy, Start);
+      Rdx = emitReduction(Vec, Builder, &TTI, Ctx, DestTy, Start);
     }
     if (Rdx->getType() != DestTy)
       Rdx = Builder.CreateIntCast(Rdx, DestTy, IsSigned);
@@ -31867,8 +31871,17 @@ class HorizontalReduction {
             auto [RType, IsSigned] = R.getRootNodeTypeWithNoCast().value_or(
                 std::make_pair(RedTy, true));
             if (RType == RedTy) {
-              VectorCost = TTI->getArithmeticReductionCost(RdxOpcode, VectorTy,
-                                                           FMF, CostKind);
+              if (VectorTy->getElementType()->isIntegerTy(1) &&
+                  (RdxKind == RecurKind::And || RdxKind == RecurKind::Or ||
+                   (RdxKind == RecurKind::Add && !ScalarTy->isIntegerTy(1))))
+                VectorCost =
+                    getI1ReductionCost(RdxKind, *TTI, VectorTy, ScalarTy,
+                                       R.getCastContextHint(R.getRootNode()),
+                                       CostKind)
+                        .first;
+              else
+                VectorCost = TTI->getArithmeticReductionCost(
+                    RdxOpcode, VectorTy, FMF, CostKind);
             } else {
               VectorCost = TTI->getExtendedReductionCost(
                   RdxOpcode, !IsSigned, RedTy,
@@ -32011,6 +32024,7 @@ class HorizontalReduction {
   /// combine.
   std::pair<Value *, bool> emitReduction(IRBuilderBase &Builder,
                                          const TargetTransformInfo &TTI,
+                                         TTI::CastContextHint Ctx,
                                          Type *DestTy) {
     Value *ReducedSubTree = nullptr;
     bool ResNegated = false;
@@ -32038,8 +32052,8 @@ class HorizontalReduction {
     // the signs of the operands.
     auto CreateSingleOp = [&](Value *Vec, unsigned Scale, bool IsSigned,
                               bool ReducedInTree, bool Negated) {
-      Value *Rdx = createSingleOp(Builder, TTI, Vec, Scale, IsSigned, DestTy,
-                                  ReducedInTree);
+      Value *Rdx = createSingleOp(Builder, TTI, Ctx, Vec, Scale, IsSigned,
+                                  DestTy, ReducedInTree);
       if (!ReducedSubTree) {
         ReducedSubTree = Rdx;
         ResNegated = Negated;
@@ -32220,27 +32234,66 @@ class HorizontalReduction {
     return {ReducedSubTree, ResNegated};
   }
 
+  /// Emits an i1 reduction in the cheaper of the plain and the bitcast-based
+  /// form. Returns nullptr if not an i1 reduction handled here.
+  Value *emitI1Reduction(Value *VectorizedValue, IRBuilderBase &Builder,
+                         const TargetTransformInfo &TTI,
+                         TTI::CastContextHint Ctx, Type *DestTy, Value *Start) {
+    auto *FTy = cast<FixedVectorType>(VectorizedValue->getType());
+    bool IsAdd = RdxKind == RecurKind::Add &&
+                 DestTy->getScalarType() != FTy->getScalarType();
+    bool IsBoolLogic =
+        (RdxKind == RecurKind::And || RdxKind == RecurKind::Or) &&
+        DestTy->getScalarType() == FTy->getScalarType();
+    if (Start || FTy->getScalarType() != Builder.getInt1Ty() ||
+        (!IsAdd && !IsBoolLogic))
+      return nullptr;
+    if (getI1ReductionCost(
+            RdxKind, TTI, FTy, DestTy->getScalarType(), Ctx,
+            getSLPCostKind(Builder.GetInsertBlock()->getParent()))
+            .second) {
+      // Convert vector_reduce_and/or(<n x i1>) to icmp eq/ne(bitcast <n x i1>
+      // to in) and vector_reduce_add(ZExt(<n x i1>)) to
+      // ZExtOrTrunc(ctpop(bitcast <n x i1> to in)).
+      Value *V = Builder.CreateBitCast(
+          VectorizedValue, Builder.getIntNTy(FTy->getNumElements()));
+      ++NumVectorInstructions;
+      switch (RdxKind) {
+      case RecurKind::And:
+        return Builder.CreateICmpEQ(V,
+                                    ConstantInt::getAllOnesValue(V->getType()));
+      case RecurKind::Or:
+        return Builder.CreateIsNotNull(V);
+      case RecurKind::Add:
+        return Builder.CreateUnaryIntrinsic(Intrinsic::ctpop, V);
+      default:
+        llvm_unreachable("Unexpected reduction kind for the i1 bitcast form");
+      }
+    }
+    if (IsAdd) {
+      // Keep the extended vector_reduce_add form.
+      VectorizedValue = Builder.CreateZExt(
+          VectorizedValue,
+          getWidenedType(DestTy->getScalarType(), FTy->getNumElements()));
+      ++NumVectorInstructions;
+    }
+    ++NumVectorInstructions;
+    return createSimpleReduction(Builder, VectorizedValue, RdxKind);
+  }
+
   /// Emit a horizontal reduction of the vectorized value.
   /// If \p Start is non-null, emit an ordered reduction intrinsic that
   /// sequentially accumulates into \p Start (only valid for FAdd/FMulAdd).
   Value *emitReduction(Value *VectorizedValue, IRBuilderBase &Builder,
-                       const TargetTransformInfo *TTI, Type *DestTy,
-                       Value *Start = nullptr) {
+                       const TargetTransformInfo *TTI, TTI::CastContextHint Ctx,
+                       Type *DestTy, Value *Start = nullptr) {
     assert(VectorizedValue && "Need to have a vectorized tree node");
     assert(RdxKind != RecurKind::FMulAdd &&
            "A call to the llvm.fmuladd intrinsic is not handled yet");
 
-    auto *FTy = cast<FixedVectorType>(VectorizedValue->getType());
-    if (!Start && FTy->getScalarType() == Builder.getInt1Ty() &&
-        RdxKind == RecurKind::Add &&
-        DestTy->getScalarType() != FTy->getScalarType()) {
-      // Convert vector_reduce_add(ZExt(<n x i1>)) to
-      // ZExtOrTrunc(ctpop(bitcast <n x i1> to in)).
-      Value *V = Builder.CreateBitCast(
-          VectorizedValue, Builder.getIntNTy(FTy->getNumElements()));
-      ++NumVectorInstructions;
-      return Builder.CreateUnaryIntrinsic(Intrinsic::ctpop, V);
-    }
+    if (Value *Rdx =
+            emitI1Reduction(VectorizedValue, Builder, *TTI, Ctx, DestTy, Start))
+      return Rdx;
     ++NumVectorInstructions;
     if (Start)
       return createOrderedReduction(Builder, RdxKind, VectorizedValue, Start);
@@ -32815,6 +32868,20 @@ bool SLPVectorizerPass::tryToVectorize(
             VecTy, APInt::getAllOnes(getNumElements(VecTy)), /*Insert=*/false,
             /*Extract=*/true, CostKind) +
         TTI.getInstructionCost(Inst, CostKind);
+    // The reduction-only cost cannot price the compared-values tree of i1
+    // and/or reductions, so ties are left to the full analysis.
+    if (Ty->isIntegerTy(1) &&
+        (Kind == RecurKind::And || Kind == RecurKind::Or)) {
+      TTI::CastContextHint Ctx =
+          all_of(Ops, [](Value *Op) { return isa<LoadInst>(Op); })
+              ? TTI::CastContextHint::Normal
+              : TTI::CastContextHint::None;
+      if (getI1ReductionCost(Kind, TTI, cast<FixedVectorType>(VecTy), Ty, Ctx,
+                             CostKind)
+              .first > ScalarCost)
+        return false;
+      return HorRdx.tryToReduce(R, *DL, &TTI, *TLI, AC, *DT) != nullptr;
+    }
     InstructionCost RedCost;
     switch (::getRdxKind(Inst)) {
     case RecurKind::Add:
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.cpp
index c5679c52c2a49..c18da825e7e2a 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.cpp
@@ -14,8 +14,11 @@
 #include "llvm/ADT/STLExtras.h"
 #include "llvm/ADT/Sequence.h"
 #include "llvm/ADT/SmallVector.h"
+#include "llvm/Analysis/IVDescriptors.h"
+#include "llvm/IR/Constants.h"
 #include "llvm/IR/DerivedTypes.h"
 #include "llvm/IR/Instructions.h"
+#include "llvm/IR/Intrinsics.h"
 #include "llvm/IR/Operator.h"
 #include "llvm/IR/Type.h"
 #include "llvm/IR/Value.h"
@@ -240,4 +243,62 @@ InstructionCost getExtractWithExtendCost(const TargetTransformInfo &TTI,
   return TTI.getExtractWithExtendCost(Opcode, Dst, VecTy, Index, CostKind);
 }
 
+static InstructionCost
+getBoolLogicRdxBitcastCost(RecurKind Kind, const TargetTransformInfo &TTI,
+                           FixedVectorType *VectorTy, TTI::CastContextHint Ctx,
+                           TTI::TargetCostKind CostKind) {
+  assert((Kind == RecurKind::And || Kind == RecurKind::Or) &&
+         VectorTy->getElementType()->isIntegerTy(1) &&
+         "Expected and/or reduction of i1");
+  auto *IntTy =
+      IntegerType::get(VectorTy->getContext(), getNumElements(VectorTy));
+  CmpInst::Predicate Pred =
+      Kind == RecurKind::And ? CmpInst::ICMP_EQ : CmpInst::ICMP_NE;
+  // The compare is against the all-ones (and) or zero (or) constant.
+  Constant *CmpConst = Kind == RecurKind::And
+                           ? ConstantInt::getAllOnesValue(IntTy)
+                           : ConstantInt::getNullValue(IntTy);
+  return TTI.getCastInstrCost(Instruction::BitCast, IntTy, VectorTy, Ctx,
+                              CostKind) +
+         TTI.getCmpSelInstrCost(
+             Instruction::ICmp, IntTy, CmpInst::makeCmpResultType(IntTy), Pred,
+             CostKind, /*Op1Info=*/{}, TTI::getOperandInfo(CmpConst));
+}
+
+std::pair<InstructionCost, bool>
+getI1ReductionCost(RecurKind Kind, const TargetTransformInfo &TTI,
+                   FixedVectorType *VectorTy, Type *ScalarTy,
+                   TTI::CastContextHint Ctx, TTI::TargetCostKind CostKind) {
+  unsigned RdxOpcode = RecurrenceDescriptor::getOpcode(Kind);
+  if (Kind == RecurKind::And || Kind == RecurKind::Or) {
+    InstructionCost RdxCost = TTI.getArithmeticReductionCost(
+        RdxOpcode, VectorTy, std::nullopt, CostKind);
+    InstructionCost BitcastCost =
+        getBoolLogicRdxBitcastCost(Kind, TTI, VectorTy, Ctx, CostKind);
+    return {std::min(RdxCost, BitcastCost), BitcastCost < RdxCost};
+  }
+  assert(Kind == RecurKind::Add && !ScalarTy->isIntegerTy(1) &&
+         "Expected add reduction of zexted i1 values");
+  // The extended reduction form.
+  InstructionCost ExtRdxCost =
+      TTI.getExtendedReductionCost(RdxOpcode, /*IsUnsigned=*/true, ScalarTy,
+                                   VectorTy, std::nullopt, CostKind);
+  // The ctpop form is estimated as the cheaper of the scalar bitcast+ctpop
+  // and the vector ctpop on the mask type; it is emitted as bitcast+ctpop.
+  // The zext/trunc of the ctpop result to the destination type is absorbed by
+  // the legalization of the ctpop itself.
+  auto *IntTy =
+      IntegerType::get(VectorTy->getContext(), getNumElements(VectorTy));
+  InstructionCost CtpopCost = std::min(
+      TTI.getCastInstrCost(Instruction::BitCast, IntTy, VectorTy, Ctx,
+                           CostKind) +
+          TTI.getIntrinsicInstrCost(
+              IntrinsicCostAttributes(Intrinsic::ctpop, IntTy, {IntTy}),
+              CostKind),
+      TTI.getIntrinsicInstrCost(
+          IntrinsicCostAttributes(Intrinsic::ctpop, VectorTy, {VectorTy}),
+          CostKind));
+  return {std::min(ExtRdxCost, CtpopCost), CtpopCost <= ExtRdxCost};
+}
+
 } // namespace llvm::slpvectorizer
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.h b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.h
index 75ad9fe88db80..c29782bd4de5b 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.h
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.h
@@ -30,6 +30,7 @@ class Type;
 class User;
 class Value;
 class VectorType;
+enum class RecurKind;
 } // namespace llvm
 
 namespace llvm::slpvectorizer {
@@ -99,6 +100,16 @@ getExtractWithExtendCost(const TargetTransformInfo &TTI, bool ReVec,
                          unsigned Index,
                          const TargetTransformInfo::TargetCostKind CostKind);
 
+/// i1 reductions can be emitted as the plain target reduction or in the
+/// bitcast-based form (bitcast to a scalar integer type plus a compare for
+/// and/or, plus ctpop for add). Returns the cost of the cheaper form and
+/// whether it is the bitcast-based one.
+std::pair<InstructionCost, bool>
+getI1ReductionCost(RecurKind Kind, const TargetTransformInfo &TTI,
+                   FixedVectorType *VectorTy, Type *ScalarTy,
+                   TargetTransformInfo::CastContextHint Ctx,
+                   TargetTransformInfo::TargetCostKind CostKind);
+
 } // namespace llvm::slpvectorizer
 
 #endif // LLVM_LIB_TRANSFORMS_VECTORIZE_SLPVECTORIZER_SLPCOSTANALYSIS_H
diff --git a/llvm/test/Transforms/PhaseOrdering/X86/vector-reductions.ll b/llvm/test/Transforms/PhaseOrdering/X86/vector-reductions.ll
index 78f68c420f6ab..fe21d4039f240 100644
--- a/llvm/test/Transforms/PhaseOrdering/X86/vector-reductions.ll
+++ b/llvm/test/Transforms/PhaseOrdering/X86/vector-reductions.ll
@@ -277,13 +277,12 @@ define i1 @cmp_lt_gt(double %a, double %b, double %c) {
 ; CHECK-NEXT:    [[TMP6:%.*]] = shufflevector <2 x double> [[TMP5]], <2 x double> poison, <2 x i32> zeroinitializer
 ; CHECK-NEXT:    [[TMP7:%.*]] = fdiv <2 x double> [[TMP3]], [[TMP6]]
 ; CHECK-NEXT:    [[TMP8:%.*]] = fcmp uge <2 x double> [[TMP7]], splat (double f0x3EB0C6F7A0B5ED8D)
-; CHECK-NEXT:    [[SHIFT:%.*]] = shufflevector <2 x i1> [[TMP8]], <2 x i1> poison, <2 x i32> <i32 1, i32 poison>
-; CHECK-NEXT:    [[FOLDEXTEXTBINOP:%.*]] = or <2 x i1> [[TMP8]], [[SHIFT]]
+; CHECK-NEXT:    [[TMP11:%.*]] = bitcast <2 x i1> [[TMP8]] to i2
+; CHECK-NEXT:    [[TMP12:%.*]] = icmp ne i2 [[TMP11]], 0
 ; CHECK-NEXT:    [[TMP9:%.*]] = fcmp ule <2 x double> [[TMP7]], splat (double 1.000000e+00)
-; CHECK-NEXT:    [[SHIFT3:%.*]] = shufflevector <2 x i1> [[TMP9]], <2 x i1> poison, <2 x i32> <i32 1, i32 poison>
-; CHECK-NEXT:    [[TMP10:%.*]] = or <2 x i1> [[TMP9]], [[SHIFT3]]
-; CHECK-NEXT:    [[FOLDEXTEXTBINOP4:%.*]] = and <2 x i1> [[FOLDEXTEXTBINOP]], [[TMP10]]
-; CHECK-NEXT:    [[RETVAL_0:%.*]] = extractelement <2 x i1> [[FOLDEXTEXTBINOP4]], i64 0
+; CHECK-NEXT:    [[TMP13:%.*]] = bitcast <2 x i1> [[TMP9]] to i2
+; CHECK-NEXT:    [[TMP10:%.*]] = icmp ne i2 [[TMP13]], 0
+; CHECK-NEXT:    [[RETVAL_0:%.*]] = and i1 [[TMP12]], [[TMP10]]
 ; CHECK-NEXT:    ret i1 [[RETVAL_0]]
 ;
 entry:
diff --git a/llvm/test/Transforms/SLPVectorizer/AArch64/externally-used-copyables.ll b/llvm/test/Transforms/SLPVectorizer/AArch64/externally-used-copyables.ll
index c98a4d1f2c182..35d804d927ff0 100644
--- a/llvm/test/Transforms/SLPVectorizer/AArch64/externally-used-copyables.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AArch64/externally-used-copyables.ll
@@ -4,141 +4,143 @@
 define void @test(i64 %0, i64 %1, i64 %2, i64 %3, i64 %.sroa.3341.0.copyload, i64 %.sroa.3308.0.copyload, i64 %.neg1, i64 %indvar3788, i64 %4, i64 %5, i64 %6, i64 %7, i64 %8) {
 ; CHECK-LABEL: define void @test(
 ; CHECK-SAME: i64 [[TMP0:%.*]], i64 [[TMP1:%.*]], i64 [[TMP2:%.*]], i64 [[TMP3:%.*]], i64 [[DOTSROA_3341_0_COPYLOAD:%.*]], i64 [[DOTSROA_3308_0_COPYLOAD:%.*]], i64 [[DOTNEG1:%.*]], i64 [[INDVAR3788:%.*]], i64 [[TMP4:%.*]], i64 [[TMP5:%.*]], i64 [[TMP6:%.*]], i64 [[TMP7:%.*]], i64 [[TMP8:%.*]]) #[[ATTR0:[0-9]+]] {
-; CHECK-NEXT:  [[_LR_PH_PREHEADER:.*:]]
+; CHECK-NEXT:  [[_LR_PH_PREHEADER:.*]]:
 ; CHECK-NEXT:    [[TMP9:%.*]] = insertelement <2 x i64> poison, i64 [[TMP1]], i64 0
 ; CHECK-NEXT:    [[TMP10:%.*]] = insertelement <2 x i64> [[TMP9]], i64 [[TMP0]], i64 1
-; CHECK-NEXT:    [[TMP13:%.*]] = mul <2 x i64> [[TMP10]], <i64 1, i64 24>
+; CHECK-NEXT:    [[TMP12:%.*]] = mul <2 x i64> [[TMP10]], <i64 1, i64 24>
 ; CHECK-NEXT:    [[TMP17:%.*]] = mul <2 x i64> [[TMP10]], <i64 1, i64 40>
 ; CHECK-NEXT:    [[TMP11:%.*]] = mul i64 [[TMP0]], 48
 ; CHECK-NEXT:    [[TMP14:%.*]] = insertelement <2 x i64> <i64 poison, i64 1>, i64 [[TMP0]], i64 0
-; CHECK-NEXT:    [[TMP15:%.*]] = shufflevector <2 x i64> [[TMP14]], <2 x i64> <i64 -1, i64 poison>, <2 x i32> <i32 2, i32 0>
-; CHECK-NEXT:    [[TMP16:%.*]] = sub <2 x i64> [[TMP14]], [[TMP15]]
-; CHECK-NEXT:    [[TMP20:%.*]] = shl i64 [[TMP0]], 11
+; CHECK-NEXT:    [[TMP22:%.*]] = shufflevector <2 x i64> [[TMP14]], <2 x i64> <i64 -1, i64 poison>, <2 x i32> <i32 2, i32 0>
+; CHECK-NEXT:    [[TMP16:%.*]] = sub <2 x i64> [[TMP14]], [[TMP22]]
+; CHECK-NEXT:    [[TMP25:%.*]] = shl i64 [[TMP0]], 11
 ; CHECK-NEXT:    [[TMP35:%.*]] = insertelement <2 x i64> poison, i64 [[TMP0]], i64 0
-; CHECK-NEXT:    [[TMP12:%.*]] = shufflevector <2 x i64> [[TMP35]], <2 x i64> poison, <2 x i32> zeroinitializer
-; CHECK-NEXT:    [[TMP25:%.*]] = shl <2 x i64> [[TMP12]], <i64 11, i64 0>
-; CHECK-NEXT:    [[TMP21:%.*]] = sub i64 1, [[TMP20]]
-; CHECK-NEXT:    [[TMP22:%.*]] = add <2 x i64> [[TMP25]], <i64 8, i64 1>
-; CHECK-NEXT:    [[TMP23:%.*]] = or <2 x i64> [[TMP25]], <i64 8, i64 1>
-; CHECK-NEXT:    [[TMP33:%.*]] = shufflevector <2 x i64> [[TMP22]], <2 x i64> [[TMP23]], <2 x i32> <i32 0, i32 3>
+; CHECK-NEXT:    [[TMP33:%.*]] = shufflevector <2 x i64> [[TMP35]], <2 x i64> poison, <2 x i32> zeroinitializer
+; CHECK-NEXT:    [[TMP20:%.*]] = shl <2 x i64> [[TMP33]], <i64 11, i64 0>
+; CHECK-NEXT:    [[TMP21:%.*]] = sub i64 1, [[TMP25]]
+; CHECK-NEXT:    [[TMP41:%.*]] = add <2 x i64> [[TMP20]], <i64 8, i64 1>
+; CHECK-NEXT:    [[TMP23:%.*]] = or <2 x i64> [[TMP20]], <i64 8, i64 1>
+; CHECK-NEXT:    [[TMP44:%.*]] = shufflevector <2 x i64> [[TMP41]], <2 x i64> [[TMP23]], <2 x i32> <i32 0, i32 3>
 ; CHECK-NEXT:    [[TMP18:%.*]] = shl i64 [[TMP0]], 1
 ; CHECK-NEXT:    [[TMP19:%.*]] = or i64 [[TMP18]], [[TMP0]]
-; CHECK-NEXT:    [[TMP34:%.*]] = or i64 [[TMP19]], 1
-; CHECK-NEXT:    [[TMP55:%.*]] = mul i64 [[TMP0]], [[TMP0]]
-; CHECK-NEXT:    [[TMP58:%.*]] = insertelement <8 x i64> poison, i64 [[TMP0]], i64 0
-; CHECK-NEXT:    [[TMP30:%.*]] = shufflevector <8 x i64> [[TMP58]], <8 x i64> poison, <8 x i32> zeroinitializer
+; CHECK-NEXT:    [[TMP27:%.*]] = or i64 [[TMP19]], 1
+; CHECK-NEXT:    [[TMP28:%.*]] = mul i64 [[TMP0]], [[TMP0]]
+; CHECK-NEXT:    [[TMP29:%.*]] = insertelement <8 x i64> poison, i64 [[TMP0]], i64 0
+; CHECK-NEXT:    [[TMP30:%.*]] = shufflevector <8 x i64> [[TMP29]], <8 x i64> poison, <8 x i32> zeroinitializer
 ; CHECK-NEXT:    [[TMP31:%.*]] = shufflevector <8 x i64> [[TMP30]], <8 x i64> <i64 poison, i64 1, i64 1, i64 1, i64 poison, i64 poison, i64 poison, i64 poison>, <6 x i32> <i32 0, i32 9, i32 10, i32 11, i32 poison, i32 poison>
 ; CHECK-NEXT:    [[TMP32:%.*]] = shufflevector <8 x i64> [[TMP30]], <8 x i64> <i64 0, i64 0, i64 0, i64 0, i64 poison, i64 1, i64 1, i64 1>, <8 x i32> <i32 8, i32 9, i32 10, i32 11, i32 0, i32 13, i32 14, i32 15>
-; CHECK-NEXT:    [[TMP27:%.*]] = shufflevector <8 x i64> [[TMP30]], <8 x i64> poison, <3 x i32> <i32 0, i32 poison, i32 poison>
-; CHECK-NEXT:    [[TMP62:%.*]] = insertelement <3 x i64> [[TMP27]], i64 [[TMP1]], i64 1
-; CHECK-NEXT:    [[TMP29:%.*]] = insertelement <3 x i64> [[TMP62]], i64 [[DOTNEG1]], i64 2
-; CHECK-NEXT:    [[TMP36:%.*]] = shufflevector <3 x i64> [[TMP29]], <3 x i64> poison, <8 x i32> <i32 0, i32 1, i32 1, i32 1, i32 2, i32 0, i32 0, i32 0>
+; CHECK-NEXT:    [[TMP55:%.*]] = shufflevector <8 x i64> [[TMP30]], <8 x i64> poison, <3 x i32> <i32 0, i32 poison, i32 poison>
+; CHECK-NEXT:    [[TMP34:%.*]] = insertelement <3 x i64> [[TMP55]], i64 [[TMP1]], i64 1
+; CHECK-NEXT:    [[TMP56:%.*]] = insertelement <3 x i64> [[TMP34]], i64 [[DOTNEG1]], i64 2
+; CHECK-NEXT:    [[TMP36:%.*]] = shufflevector <3 x i64> [[TMP56]], <3 x i64> poison, <8 x i32> <i32 0, i32 1, i32 1, i32 1, i32 2, i32 0, i32 0, i32 0>
 ; CHECK-NEXT:    [[TMP37:%.*]] = insertelement <4 x i64> poison, i64 [[TMP2]], i64 0
 ; CHECK-NEXT:    [[TMP38:%.*]] = insertelement <4 x i64> [[TMP37]], i64 [[TMP8]], i64 1
 ; CHECK-NEXT:    [[TMP39:%.*]] = insertelement <4 x i64> [[TMP38]], i64 [[DOTSROA_3308_0_COPYLOAD]], i64 2
 ; CHECK-NEXT:    [[TMP40:%.*]] = insertelement <4 x i64> [[TMP39]], i64 [[TMP0]], i64 3
-; CHECK-NEXT:    [[TMP41:%.*]] = shufflevector <2 x i64> [[TMP17]], <2 x i64> poison, <8 x i32> <i32 0, i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT:    [[TMP63:%.*]] = insertelement <4 x i64> poison, i64 [[TMP0]], i64 1
-; CHECK-NEXT:    [[TMP43:%.*]] = mul i64 [[TMP0]], [[TMP0]]
-; CHECK-NEXT:    [[TMP44:%.*]] = shufflevector <2 x i64> [[TMP13]], <2 x i64> poison, <32 x i32> <i32 0, i32 poison, i32 poison, i32 poison, i32 poison, i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT:    [[TMP45:%.*]] = shufflevector <32 x i64> [[TMP44]], <32 x i64> poison, <32 x i32> <i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 0, i32 poison, i32 poison, i32 poison, i32 poison, i32 5, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT:    [[TMP46:%.*]] = shufflevector <2 x i64> [[TMP33]], <2 x i64> poison, <6 x i32> <i32 0, i32 1, i32 poison, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT:    [[TMP47:%.*]] = shufflevector <6 x i64> [[TMP31]], <6 x i64> [[TMP46]], <6 x i32> <i32 0, i32 1, i32 2, i32 3, i32 6, i32 7>
-; CHECK-NEXT:    [[TMP48:%.*]] = shufflevector <2 x i64> [[TMP16]], <2 x i64> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
+; CHECK-NEXT:    [[TMP48:%.*]] = insertelement <2 x i64> [[TMP35]], i64 [[TMP1]], i64 1
+; CHECK-NEXT:    [[TMP59:%.*]] = shufflevector <2 x i64> [[TMP48]], <2 x i64> <i64 poison, i64 1>, <2 x i32> <i32 0, i32 3>
+; CHECK-NEXT:    [[TMP115:%.*]] = insertelement <2 x i64> [[TMP48]], i64 [[TMP2]], i64 1
+; CHECK-NEXT:    [[TMP43:%.*]] = shufflevector <2 x i64> [[TMP14]], <2 x i64> poison, <2 x i32> zeroinitializer
+; CHECK-NEXT:    [[TMP45:%.*]] = shufflevector <2 x i64> [[TMP17]], <2 x i64> poison, <8 x i32> <i32 0, i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT:    [[TMP46:%.*]] = shufflevector <2 x i64> [[TMP12]], <2 x i64> poison, <32 x i32> <i32 0, i32 poison, i32 poison, i32 poison, i32 poison, i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT:    [[TMP47:%.*]] = shufflevector <32 x i64> [[TMP46]], <32 x i64> poison, <32 x i32> <i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 0, i32 poison, i32 poison, i32 poison, i32 poison, i32 5, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT:    [[TMP60:%.*]] = shufflevector <2 x i64> [[TMP44]], <2 x i64> poison, <6 x i32> <i32 0, i32 1, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT:    [[TMP49:%.*]] = shufflevector <6 x i64> [[TMP31]], <6 x i64> [[TMP60]], <6 x i32> <i32 0, i32 1, i32 2, i32 3, i32 6, i32 7>
 ; CHECK-NEXT:    br label %[[DOTLR_PH1977_US:.*]]
-; CHECK:       [[_LR_PH1977_US:.*:]]
-; CHECK-NEXT:    [[INDVAR37888:%.*]] = phi i64 [ 0, [[DOTLR_PH_PREHEADER:%.*]] ], [ 1, %[[DOTLR_PH1977_US]] ]
-; CHECK-NEXT:    [[TMP49:%.*]] = shufflevector <6 x i64> [[TMP47]], <6 x i64> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 5, i32 5>
-; CHECK-NEXT:    [[TMP50:%.*]] = mul <8 x i64> [[TMP30]], [[TMP49]]
-; CHECK-NEXT:    [[TMP51:%.*]] = or <8 x i64> [[TMP30]], [[TMP49]]
-; CHECK-NEXT:    [[TMP52:%.*]] = shufflevector <8 x i64> [[TMP50]], <8 x i64> [[TMP51]], <8 x i32> <i32 0, i32 9, i32 10, i32 11, i32 4, i32 5, i32 6, i32 7>
-; CHECK-NEXT:    [[TMP53:%.*]] = mul i64 [[TMP34]], [[TMP0]]
-; CHECK-NEXT:    [[TMP54:%.*]] = mul i64 [[TMP21]], [[TMP0]]
+; CHECK:       [[DOTLR_PH1977_US]]:
+; CHECK-NEXT:    [[INDVAR37888:%.*]] = phi i64 [ 0, %[[_LR_PH_PREHEADER]] ], [ 1, %[[DOTLR_PH1977_US]] ]
+; CHECK-NEXT:    [[TMP50:%.*]] = shufflevector <6 x i64> [[TMP49]], <6 x i64> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 5, i32 5>
+; CHECK-NEXT:    [[TMP51:%.*]] = mul <8 x i64> [[TMP30]], [[TMP50]]
+; CHECK-NEXT:    [[TMP52:%.*]] = or <8 x i64> [[TMP30]], [[TMP50]]
+; CHECK-NEXT:    [[TMP53:%.*]] = shufflevector <8 x i64> [[TMP51]], <8 x i64> [[TMP52]], <8 x i32> <i32 0, i32 9, i32 10, i32 11, i32 4, i32 5, i32 6, i32 7>
+; CHECK-NEXT:    [[TMP54:%.*]] = mul i64 [[TMP27]], [[TMP0]]
+; CHECK-NEXT:    [[TMP94:%.*]] = mul i64 [[TMP21]], [[TMP0]]
 ; CHECK-NEXT:    [[TMP26:%.*]] = call i64 @llvm.vscale.i64()
 ; CHECK-NEXT:    [[DIFF_CHECK3783:%.*]] = icmp ult i64 [[TMP11]], [[TMP26]]
-; CHECK-NEXT:    [[DIFF_CHECK3790:%.*]] = icmp ult i64 [[INDVAR37888]], [[TMP0]]
-; CHECK-NEXT:    [[DIFF_CHECK3805:%.*]] = icmp ugt i64 [[TMP26]], 1
-; CHECK-NEXT:    [[TMP56:%.*]] = add <8 x i64> [[TMP52]], [[TMP32]]
-; CHECK-NEXT:    [[TMP57:%.*]] = icmp ult <8 x i64> [[TMP56]], [[TMP36]]
-; CHECK-NEXT:    [[TMP24:%.*]] = extractelement <8 x i64> [[TMP52]], i64 5
-; CHECK-NEXT:    [[TMP42:%.*]] = add i64 [[TMP24]], 1
-; CHECK-NEXT:    [[DIFF_CHECK3842:%.*]] = icmp ult i64 [[TMP42]], [[TMP0]]
+; CHECK-NEXT:    [[OP_RDX52:%.*]] = icmp ugt i64 [[TMP26]], 1
+; CHECK-NEXT:    [[TMP57:%.*]] = add <8 x i64> [[TMP53]], [[TMP32]]
+; CHECK-NEXT:    [[TMP58:%.*]] = icmp ult <8 x i64> [[TMP57]], [[TMP36]]
+; CHECK-NEXT:    [[TMP62:%.*]] = extractelement <8 x i64> [[TMP53]], i64 5
+; CHECK-NEXT:    [[TMP42:%.*]] = add i64 [[TMP62]], 1
+; CHECK-NEXT:    [[TMP61:%.*]] = insertelement <2 x i64> poison, i64 [[INDVAR37888]], i64 0
+; CHECK-NEXT:    [[TMP64:%.*]] = insertelement <2 x i64> [[TMP61]], i64 [[TMP42]], i64 1
+; CHECK-NEXT:    [[TMP63:%.*]] = icmp ult <2 x i64> [[TMP64]], [[TMP33]]
 ; CHECK-NEXT:    [[DIFF_CHECK3817:%.*]] = icmp ult i64 [[TMP1]], [[TMP2]]
-; CHECK-NEXT:    [[TMP76:%.*]] = add i64 [[TMP53]], 1
-; CHECK-NEXT:    [[DIFF_CHECK3820:%.*]] = icmp ult i64 [[TMP76]], [[TMP0]]
-; CHECK-NEXT:    [[TMP77:%.*]] = insertelement <4 x i64> [[TMP63]], i64 [[INDVAR37888]], i64 0
-; CHECK-NEXT:    [[TMP78:%.*]] = shufflevector <4 x i64> [[TMP77]], <4 x i64> poison, <4 x i32> <i32 0, i32 1, i32 1, i32 1>
-; CHECK-NEXT:    [[TMP80:%.*]] = shufflevector <4 x i64> [[TMP77]], <4 x i64> <i64 1, i64 poison, i64 poison, i64 poison>, <4 x i32> <i32 4, i32 1, i32 poison, i32 poison>
-; CHECK-NEXT:    [[TMP81:%.*]] = shufflevector <4 x i64> [[TMP80]], <4 x i64> [[TMP48]], <4 x i32> <i32 0, i32 1, i32 4, i32 5>
-; CHECK-NEXT:    [[TMP82:%.*]] = mul <4 x i64> [[TMP78]], [[TMP81]]
-; CHECK-NEXT:    [[TMP84:%.*]] = add i64 [[TMP54]], 1
-; CHECK-NEXT:    [[TMP85:%.*]] = insertelement <32 x i64> [[TMP45]], i64 [[TMP55]], i64 11
-; CHECK-NEXT:    [[TMP86:%.*]] = shufflevector <4 x i64> [[TMP82]], <4 x i64> poison, <32 x i32> <i32 0, i32 1, i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT:    [[TMP89:%.*]] = shufflevector <32 x i64> [[TMP85]], <32 x i64> [[TMP86]], <32 x i32> <i32 32, i32 33, i32 34, i32 35, i32 poison, i32 5, i32 poison, i32 poison, i32 poison, i32 poison, i32 10, i32 11, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT:    [[TMP90:%.*]] = insertelement <32 x i64> [[TMP89]], i64 [[TMP84]], i64 4
-; CHECK-NEXT:    [[TMP100:%.*]] = shufflevector <8 x i64> [[TMP52]], <8 x i64> poison, <32 x i32> <i32 poison, i32 poison, i32 poison, i32 poison, i32 4, i32 5, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT:    [[TMP73:%.*]] = shufflevector <8 x i64> [[TMP52]], <8 x i64> poison, <32 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT:    [[TMP87:%.*]] = shufflevector <32 x i64> [[TMP90]], <32 x i64> [[TMP73]], <32 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 36, i32 37, i32 poison, i32 poison, i32 10, i32 11, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT:    [[TMP98:%.*]] = shl <2 x i64> [[TMP12]], splat (i64 1)
-; CHECK-NEXT:    [[TMP125:%.*]] = shufflevector <2 x i64> [[TMP98]], <2 x i64> poison, <32 x i32> <i32 0, i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT:    [[TMP88:%.*]] = shufflevector <32 x i64> [[TMP87]], <32 x i64> [[TMP125]], <32 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 32, i32 33, i32 10, i32 11, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT:    [[TMP59:%.*]] = insertelement <32 x i64> [[TMP88]], i64 [[TMP54]], i64 12
-; CHECK-NEXT:    [[TMP60:%.*]] = shufflevector <32 x i64> [[TMP59]], <32 x i64> poison, <32 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 5, i32 5, i32 7, i32 5, i32 8, i32 9, i32 10, i32 5, i32 0, i32 1, i32 11, i32 12, i32 5, i32 5, i32 6, i32 7, i32 5, i32 5, i32 7, i32 5, i32 9, i32 10, i32 0, i32 1>
-; CHECK-NEXT:    [[TMP61:%.*]] = shufflevector <32 x i64> [[TMP59]], <32 x i64> poison, <10 x i32> <i32 poison, i32 poison, i32 5, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT:    [[TMP109:%.*]] = insertelement <10 x i64> [[TMP61]], i64 [[TMP0]], i64 0
-; CHECK-NEXT:    [[TMP110:%.*]] = insertelement <10 x i64> [[TMP109]], i64 [[TMP3]], i64 1
-; CHECK-NEXT:    [[TMP111:%.*]] = insertelement <10 x i64> [[TMP110]], i64 [[TMP4]], i64 3
-; CHECK-NEXT:    [[TMP112:%.*]] = insertelement <10 x i64> [[TMP111]], i64 [[INDVAR3788]], i64 4
-; CHECK-NEXT:    [[TMP113:%.*]] = insertelement <10 x i64> [[TMP112]], i64 [[TMP2]], i64 5
-; CHECK-NEXT:    [[TMP114:%.*]] = insertelement <10 x i64> [[TMP113]], i64 [[TMP5]], i64 6
-; CHECK-NEXT:    [[TMP115:%.*]] = insertelement <10 x i64> [[TMP114]], i64 [[TMP6]], i64 7
-; CHECK-NEXT:    [[TMP116:%.*]] = insertelement <10 x i64> [[TMP115]], i64 [[TMP7]], i64 8
-; CHECK-NEXT:    [[TMP70:%.*]] = insertelement <10 x i64> [[TMP116]], i64 [[DOTSROA_3341_0_COPYLOAD]], i64 9
-; CHECK-NEXT:    [[TMP71:%.*]] = shufflevector <10 x i64> [[TMP70]], <10 x i64> poison, <32 x i32> <i32 0, i32 0, i32 0, i32 0, i32 0, i32 1, i32 0, i32 2, i32 3, i32 4, i32 0, i32 0, i32 2, i32 2, i32 0, i32 5, i32 0, i32 0, i32 0, i32 0, i32 6, i32 7, i32 0, i32 2, i32 8, i32 9, i32 0, i32 0, i32 2, i32 0, i32 0, i32 0>
-; CHECK-NEXT:    [[TMP72:%.*]] = icmp ult <32 x i64> [[TMP60]], [[TMP71]]
-; CHECK-NEXT:    [[TMP91:%.*]] = add i64 [[TMP54]], 1
-; CHECK-NEXT:    [[TMP92:%.*]] = shufflevector <8 x i64> [[TMP52]], <8 x i64> poison, <8 x i32> <i32 5, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 4, i32 4>
-; CHECK-NEXT:    [[TMP93:%.*]] = shufflevector <8 x i64> [[TMP92]], <8 x i64> [[TMP41]], <8 x i32> <i32 0, i32 1, i32 2, i32 8, i32 9, i32 5, i32 6, i32 7>
-; CHECK-NEXT:    [[TMP94:%.*]] = shufflevector <2 x i64> [[TMP98]], <2 x i64> poison, <8 x i32> <i32 0, i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT:    [[TMP95:%.*]] = shufflevector <8 x i64> [[TMP93]], <8 x i64> [[TMP94]], <8 x i32> <i32 0, i32 8, i32 9, i32 3, i32 4, i32 5, i32 6, i32 7>
-; CHECK-NEXT:    [[TMP96:%.*]] = insertelement <8 x i64> [[TMP95]], i64 [[TMP91]], i64 5
-; CHECK-NEXT:    [[TMP105:%.*]] = shufflevector <8 x i64> [[TMP96]], <8 x i64> poison, <2 x i32> <i32 poison, i32 3>
-; CHECK-NEXT:    [[TMP107:%.*]] = insertelement <2 x i64> [[TMP105]], i64 [[TMP0]], i64 0
-; CHECK-NEXT:    [[TMP99:%.*]] = shufflevector <2 x i64> [[TMP107]], <2 x i64> poison, <8 x i32> <i32 0, i32 1, i32 1, i32 0, i32 0, i32 0, i32 0, i32 1>
-; CHECK-NEXT:    [[TMP83:%.*]] = icmp ult <8 x i64> [[TMP96]], [[TMP99]]
-; CHECK-NEXT:    [[TMP101:%.*]] = shufflevector <8 x i64> [[TMP52]], <8 x i64> poison, <4 x i32> <i32 4, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT:    [[TMP102:%.*]] = insertelement <4 x i64> [[TMP101]], i64 [[TMP1]], i64 1
-; CHECK-NEXT:    [[TMP103:%.*]] = insertelement <4 x i64> [[TMP102]], i64 [[TMP53]], i64 3
-; CHECK-NEXT:    [[TMP104:%.*]] = shufflevector <4 x i64> [[TMP103]], <4 x i64> poison, <4 x i32> <i32 0, i32 1, i32 1, i32 3>
-; CHECK-NEXT:    [[TMP97:%.*]] = icmp ult <4 x i64> [[TMP104]], [[TMP40]]
+; CHECK-NEXT:    [[TMP24:%.*]] = add i64 [[TMP54]], 1
 ; CHECK-NEXT:    [[DIFF_CHECK3930:%.*]] = icmp ult i64 [[TMP24]], [[TMP0]]
-; CHECK-NEXT:    [[TMP106:%.*]] = extractelement <2 x i64> [[TMP98]], i64 0
-; CHECK-NEXT:    [[DIFF_CHECK3932:%.*]] = icmp ult i64 [[TMP106]], [[TMP1]]
-; CHECK-NEXT:    [[OP_RDX50:%.*]] = icmp ult i64 [[TMP1]], [[TMP0]]
-; CHECK-NEXT:    [[TMP126:%.*]] = icmp ult i64 [[INDVAR37888]], [[TMP0]]
-; CHECK-NEXT:    [[DIFF_CHECK3938:%.*]] = icmp ult i64 [[TMP43]], [[TMP0]]
-; CHECK-NEXT:    [[DIFF_CHECK3950:%.*]] = icmp ult i64 [[TMP1]], [[TMP2]]
-; CHECK-NEXT:    [[TMP75:%.*]] = call i1 @llvm.vector.reduce.or.v32i1(<32 x i1> [[TMP72]])
-; CHECK-NEXT:    [[TMP108:%.*]] = call i1 @llvm.vector.reduce.or.v8i1(<8 x i1> [[TMP57]])
-; CHECK-NEXT:    [[TMP127:%.*]] = shufflevector <8 x i1> [[TMP83]], <8 x i1> poison, <4 x i32> <i32 0, i32 1, i32 2, i32 3>
-; CHECK-NEXT:    [[RDX_OP:%.*]] = or <4 x i1> [[TMP127]], [[TMP97]]
-; CHECK-NEXT:    [[TMP128:%.*]] = shufflevector <4 x i1> [[RDX_OP]], <4 x i1> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT:    [[TMP129:%.*]] = shufflevector <8 x i1> [[TMP83]], <8 x i1> [[TMP128]], <8 x i32> <i32 8, i32 9, i32 10, i32 11, i32 4, i32 5, i32 6, i32 7>
-; CHECK-NEXT:    [[TMP130:%.*]] = call i1 @llvm.vector.reduce.or.v8i1(<8 x i1> [[TMP129]])
-; CHECK-NEXT:    [[OP_RDX26:%.*]] = or i1 [[TMP130]], [[DIFF_CHECK3783]]
-; CHECK-NEXT:    [[OP_RDX57:%.*]] = or i1 [[DIFF_CHECK3790]], [[DIFF_CHECK3842]]
-; CHECK-NEXT:    [[OP_RDX58:%.*]] = or i1 [[DIFF_CHECK3817]], [[DIFF_CHECK3820]]
-; CHECK-NEXT:    [[OP_RDX59:%.*]] = or i1 [[DIFF_CHECK3930]], [[DIFF_CHECK3932]]
-; CHECK-NEXT:    [[OP_RDX52:%.*]] = or i1 [[OP_RDX50]], [[TMP126]]
-; CHECK-NEXT:    [[OP_RDX61:%.*]] = or i1 [[DIFF_CHECK3938]], [[DIFF_CHECK3950]]
-; CHECK-NEXT:    [[OP_RDX62:%.*]] = or i1 [[DIFF_CHECK3805]], [[TMP75]]
+; CHECK-NEXT:    [[TMP65:%.*]] = mul <2 x i64> [[TMP16]], [[TMP43]]
+; CHECK-NEXT:    [[TMP78:%.*]] = add i64 [[TMP94]], 1
+; CHECK-NEXT:    [[TMP67:%.*]] = insertelement <32 x i64> [[TMP47]], i64 [[TMP28]], i64 11
+; CHECK-NEXT:    [[TMP68:%.*]] = insertelement <32 x i64> [[TMP67]], i64 [[INDVAR37888]], i64 0
+; CHECK-NEXT:    [[TMP69:%.*]] = shufflevector <8 x i64> [[TMP53]], <8 x i64> poison, <32 x i32> <i32 poison, i32 poison, i32 poison, i32 poison, i32 4, i32 5, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT:    [[TMP70:%.*]] = shufflevector <8 x i64> [[TMP53]], <8 x i64> poison, <32 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT:    [[TMP71:%.*]] = shl <2 x i64> [[TMP33]], splat (i64 1)
+; CHECK-NEXT:    [[TMP72:%.*]] = add i64 [[TMP94]], 1
+; CHECK-NEXT:    [[TMP73:%.*]] = shufflevector <8 x i64> [[TMP53]], <8 x i64> poison, <8 x i32> <i32 5, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 4, i32 4>
+; CHECK-NEXT:    [[TMP74:%.*]] = shufflevector <8 x i64> [[TMP73]], <8 x i64> [[TMP45]], <8 x i32> <i32 0, i32 1, i32 2, i32 8, i32 9, i32 5, i32 6, i32 7>
+; CHECK-NEXT:    [[TMP75:%.*]] = shufflevector <2 x i64> [[TMP71]], <2 x i64> poison, <8 x i32> <i32 0, i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT:    [[TMP76:%.*]] = shufflevector <8 x i64> [[TMP74]], <8 x i64> [[TMP75]], <8 x i32> <i32 0, i32 8, i32 9, i32 3, i32 4, i32 5, i32 6, i32 7>
+; CHECK-NEXT:    [[TMP77:%.*]] = insertelement <8 x i64> [[TMP76]], i64 [[TMP72]], i64 5
+; CHECK-NEXT:    [[TMP87:%.*]] = shufflevector <8 x i64> [[TMP77]], <8 x i64> poison, <2 x i32> <i32 poison, i32 3>
+; CHECK-NEXT:    [[TMP88:%.*]] = insertelement <2 x i64> [[TMP87]], i64 [[TMP0]], i64 0
+; CHECK-NEXT:    [[TMP80:%.*]] = shufflevector <2 x i64> [[TMP88]], <2 x i64> poison, <8 x i32> <i32 0, i32 1, i32 1, i32 0, i32 0, i32 0, i32 0, i32 1>
+; CHECK-NEXT:    [[TMP81:%.*]] = icmp ult <8 x i64> [[TMP77]], [[TMP80]]
+; CHECK-NEXT:    [[TMP82:%.*]] = shufflevector <8 x i64> [[TMP53]], <8 x i64> poison, <4 x i32> <i32 4, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT:    [[TMP83:%.*]] = insertelement <4 x i64> [[TMP82]], i64 [[TMP1]], i64 1
+; CHECK-NEXT:    [[TMP84:%.*]] = insertelement <4 x i64> [[TMP83]], i64 [[TMP54]], i64 3
+; CHECK-NEXT:    [[TMP85:%.*]] = shufflevector <4 x i64> [[TMP84]], <4 x i64> poison, <4 x i32> <i32 0, i32 1, i32 1, i32 3>
+; CHECK-NEXT:    [[TMP86:%.*]] = icmp ult <4 x i64> [[TMP85]], [[TMP40]]
+; CHECK-NEXT:    [[OP_RDX26:%.*]] = icmp ult i64 [[TMP62]], [[TMP0]]
+; CHECK-NEXT:    [[TMP104:%.*]] = extractelement <2 x i64> [[TMP71]], i64 0
+; CHECK-NEXT:    [[OP_RDX57:%.*]] = icmp ult i64 [[TMP104]], [[TMP1]]
+; CHECK-NEXT:    [[TMP107:%.*]] = insertelement <2 x i64> [[TMP9]], i64 [[INDVAR37888]], i64 1
+; CHECK-NEXT:    [[TMP112:%.*]] = icmp ult <2 x i64> [[TMP107]], [[TMP33]]
+; CHECK-NEXT:    [[TMP139:%.*]] = mul <2 x i64> [[TMP48]], [[TMP59]]
+; CHECK-NEXT:    [[TMP91:%.*]] = shufflevector <2 x i64> [[TMP139]], <2 x i64> poison, <32 x i32> <i32 0, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT:    [[TMP92:%.*]] = shufflevector <2 x i64> [[TMP139]], <2 x i64> poison, <32 x i32> <i32 0, i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT:    [[TMP93:%.*]] = shufflevector <32 x i64> [[TMP68]], <32 x i64> [[TMP92]], <32 x i32> <i32 0, i32 32, i32 poison, i32 poison, i32 poison, i32 5, i32 poison, i32 poison, i32 poison, i32 poison, i32 10, i32 11, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT:    [[TMP114:%.*]] = shufflevector <2 x i64> [[TMP65]], <2 x i64> poison, <32 x i32> <i32 0, i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT:    [[TMP95:%.*]] = shufflevector <32 x i64> [[TMP93]], <32 x i64> [[TMP114]], <32 x i32> <i32 0, i32 1, i32 32, i32 33, i32 4, i32 5, i32 6, i32 7, i32 8, i32 9, i32 10, i32 11, i32 12, i32 13, i32 14, i32 15, i32 16, i32 17, i32 18, i32 19, i32 20, i32 21, i32 22, i32 23, i32 24, i32 25, i32 26, i32 27, i32 28, i32 29, i32 30, i32 31>
+; CHECK-NEXT:    [[TMP96:%.*]] = insertelement <32 x i64> [[TMP95]], i64 [[TMP78]], i64 4
+; CHECK-NEXT:    [[TMP97:%.*]] = shufflevector <32 x i64> [[TMP96]], <32 x i64> [[TMP70]], <32 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 36, i32 37, i32 poison, i32 poison, i32 10, i32 11, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT:    [[TMP98:%.*]] = shufflevector <2 x i64> [[TMP71]], <2 x i64> poison, <32 x i32> <i32 0, i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT:    [[TMP99:%.*]] = shufflevector <32 x i64> [[TMP97]], <32 x i64> [[TMP98]], <32 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 32, i32 33, i32 10, i32 11, i32 12, i32 13, i32 14, i32 15, i32 16, i32 17, i32 18, i32 19, i32 20, i32 21, i32 22, i32 23, i32 24, i32 25, i32 26, i32 27, i32 28, i32 29, i32 30, i32 31>
+; CHECK-NEXT:    [[TMP100:%.*]] = insertelement <32 x i64> [[TMP99]], i64 [[TMP94]], i64 12
+; CHECK-NEXT:    [[TMP101:%.*]] = shufflevector <32 x i64> [[TMP100]], <32 x i64> poison, <32 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 5, i32 5, i32 7, i32 5, i32 8, i32 9, i32 10, i32 5, i32 0, i32 1, i32 11, i32 12, i32 5, i32 5, i32 6, i32 7, i32 5, i32 5, i32 7, i32 5, i32 9, i32 10, i32 0, i32 1>
+; CHECK-NEXT:    [[TMP102:%.*]] = shufflevector <32 x i64> [[TMP100]], <32 x i64> poison, <10 x i32> <i32 poison, i32 poison, i32 5, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT:    [[TMP103:%.*]] = insertelement <10 x i64> [[TMP102]], i64 [[TMP0]], i64 0
+; CHECK-NEXT:    [[TMP121:%.*]] = insertelement <10 x i64> [[TMP103]], i64 [[TMP3]], i64 1
+; CHECK-NEXT:    [[TMP105:%.*]] = insertelement <10 x i64> [[TMP121]], i64 [[TMP4]], i64 3
+; CHECK-NEXT:    [[TMP106:%.*]] = insertelement <10 x i64> [[TMP105]], i64 [[INDVAR3788]], i64 4
+; CHECK-NEXT:    [[TMP124:%.*]] = insertelement <10 x i64> [[TMP106]], i64 [[TMP2]], i64 5
+; CHECK-NEXT:    [[TMP108:%.*]] = insertelement <10 x i64> [[TMP124]], i64 [[TMP5]], i64 6
+; CHECK-NEXT:    [[TMP109:%.*]] = insertelement <10 x i64> [[TMP108]], i64 [[TMP6]], i64 7
+; CHECK-NEXT:    [[TMP110:%.*]] = insertelement <10 x i64> [[TMP109]], i64 [[TMP7]], i64 8
+; CHECK-NEXT:    [[TMP111:%.*]] = insertelement <10 x i64> [[TMP110]], i64 [[DOTSROA_3341_0_COPYLOAD]], i64 9
+; CHECK-NEXT:    [[TMP125:%.*]] = shufflevector <10 x i64> [[TMP111]], <10 x i64> poison, <32 x i32> <i32 0, i32 0, i32 0, i32 0, i32 0, i32 1, i32 0, i32 2, i32 3, i32 4, i32 0, i32 0, i32 2, i32 2, i32 0, i32 5, i32 0, i32 0, i32 0, i32 0, i32 6, i32 7, i32 0, i32 2, i32 8, i32 9, i32 0, i32 0, i32 2, i32 0, i32 0, i32 0>
+; CHECK-NEXT:    [[TMP113:%.*]] = icmp ult <32 x i64> [[TMP101]], [[TMP125]]
+; CHECK-NEXT:    [[TMP116:%.*]] = icmp ult <2 x i64> [[TMP139]], [[TMP115]]
+; CHECK-NEXT:    [[OP_RDX61:%.*]] = call i1 @llvm.vector.reduce.or.v32i1(<32 x i1> [[TMP113]])
+; CHECK-NEXT:    [[TMP160:%.*]] = call i1 @llvm.vector.reduce.or.v8i1(<8 x i1> [[TMP58]])
+; CHECK-NEXT:    [[TMP117:%.*]] = shufflevector <8 x i1> [[TMP81]], <8 x i1> poison, <4 x i32> <i32 0, i32 1, i32 2, i32 3>
+; CHECK-NEXT:    [[RDX_OP:%.*]] = or <4 x i1> [[TMP117]], [[TMP86]]
+; CHECK-NEXT:    [[TMP118:%.*]] = shufflevector <4 x i1> [[RDX_OP]], <4 x i1> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT:    [[TMP119:%.*]] = shufflevector <8 x i1> [[TMP81]], <8 x i1> [[TMP118]], <8 x i32> <i32 8, i32 9, i32 10, i32 11, i32 4, i32 5, i32 6, i32 7>
+; CHECK-NEXT:    [[TMP120:%.*]] = call i1 @llvm.vector.reduce.or.v8i1(<8 x i1> [[TMP119]])
+; CHECK-NEXT:    [[OP_RDX62:%.*]] = or i1 [[TMP120]], [[DIFF_CHECK3783]]
+; CHECK-NEXT:    [[TMP150:%.*]] = call i1 @llvm.vector.reduce.or.v2i1(<2 x i1> [[TMP63]])
+; CHECK-NEXT:    [[OP_RDX58:%.*]] = or i1 [[DIFF_CHECK3817]], [[DIFF_CHECK3930]]
 ; CHECK-NEXT:    [[OP_RDX63:%.*]] = or i1 [[OP_RDX26]], [[OP_RDX57]]
-; CHECK-NEXT:    [[OP_RDX64:%.*]] = or i1 [[OP_RDX58]], [[OP_RDX59]]
-; CHECK-NEXT:    [[OP_RDX65:%.*]] = or i1 [[OP_RDX52]], [[OP_RDX61]]
-; CHECK-NEXT:    [[OP_RDX66:%.*]] = or i1 [[OP_RDX62]], [[TMP108]]
-; CHECK-NEXT:    [[OP_RDX67:%.*]] = or i1 [[OP_RDX63]], [[OP_RDX64]]
-; CHECK-NEXT:    [[OP_RDX68:%.*]] = or i1 [[OP_RDX65]], [[OP_RDX66]]
-; CHECK-NEXT:    [[TMP79:%.*]] = or i1 [[OP_RDX67]], [[OP_RDX68]]
+; CHECK-NEXT:    [[TMP122:%.*]] = call i1 @llvm.vector.reduce.or.v2i1(<2 x i1> [[TMP112]])
+; CHECK-NEXT:    [[TMP123:%.*]] = call i1 @llvm.vector.reduce.or.v2i1(<2 x i1> [[TMP116]])
+; CHECK-NEXT:    [[OP_RDX38:%.*]] = or i1 [[OP_RDX52]], [[OP_RDX61]]
+; CHECK-NEXT:    [[OP_RDX65:%.*]] = or i1 [[OP_RDX62]], [[TMP150]]
+; CHECK-NEXT:    [[OP_RDX64:%.*]] = or i1 [[OP_RDX58]], [[OP_RDX63]]
+; CHECK-NEXT:    [[OP_RDX66:%.*]] = or i1 [[TMP122]], [[TMP123]]
+; CHECK-NEXT:    [[OP_RDX43:%.*]] = or i1 [[OP_RDX38]], [[TMP160]]
+; CHECK-NEXT:    [[OP_RDX45:%.*]] = or i1 [[OP_RDX65]], [[OP_RDX64]]
+; CHECK-NEXT:    [[OP_RDX46:%.*]] = or i1 [[OP_RDX66]], [[OP_RDX43]]
+; CHECK-NEXT:    [[TMP79:%.*]] = or i1 [[OP_RDX45]], [[OP_RDX46]]
 ; CHECK-NEXT:    br i1 [[TMP79]], label %[[SCALAR_PH4023:.*]], label %[[DOTLR_PH1977_US]]
 ; CHECK:       [[SCALAR_PH4023]]:
 ; CHECK-NEXT:    ret void
diff --git a/llvm/test/Transforms/SLPVectorizer/AArch64/non-power-of-2-with-adjusted-gathers.ll b/llvm/test/Transforms/SLPVectorizer/AArch64/non-power-of-2-with-adjusted-gathers.ll
index 2e5259ad9bfa1..489d47eeb606d 100644
--- a/llvm/test/Transforms/SLPVectorizer/AArch64/non-power-of-2-with-adjusted-gathers.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AArch64/non-power-of-2-with-adjusted-gathers.ll
@@ -12,18 +12,9 @@ define i1 @test(ptr %arg, ptr %arg1, i1 %arg2, i1 %arg3, i1 %arg4) {
 ; CHECK-NEXT:    [[TMP2:%.*]] = getelementptr i8, <8 x ptr> [[TMP1]], <8 x i64> <i64 8484, i64 8484, i64 8484, i64 5284, i64 8484, i64 8484, i64 8484, i64 8484>
 ; CHECK-NEXT:    [[TMP3:%.*]] = extractelement <8 x ptr> [[TMP2]], i64 0
 ; CHECK-NEXT:    [[ICMP:%.*]] = icmp ult ptr [[TMP3]], null
-; CHECK-NEXT:    [[ICMP7:%.*]] = icmp ult ptr null, null
-; CHECK-NEXT:    [[AND:%.*]] = and i1 false, [[ICMP7]]
-; CHECK-NEXT:    [[ICMP11:%.*]] = icmp ult ptr null, null
-; CHECK-NEXT:    [[AND12:%.*]] = and i1 false, [[ICMP11]]
-; CHECK-NEXT:    [[ICMP14:%.*]] = icmp ult ptr null, null
-; CHECK-NEXT:    [[AND15:%.*]] = and i1 false, [[ICMP14]]
 ; CHECK-NEXT:    [[TMP4:%.*]] = icmp ult <8 x ptr> [[TMP2]], splat (ptr null)
 ; CHECK-NEXT:    [[TMP5:%.*]] = insertelement <8 x i1> <i1 false, i1 poison, i1 false, i1 false, i1 false, i1 false, i1 false, i1 false>, i1 [[ARG2]], i64 1
 ; CHECK-NEXT:    [[TMP6:%.*]] = and <8 x i1> [[TMP4]], [[TMP5]]
-; CHECK-NEXT:    [[TMP7:%.*]] = extractelement <8 x ptr> [[TMP2]], i64 3
-; CHECK-NEXT:    [[ICMP46:%.*]] = icmp ult ptr [[TMP7]], null
-; CHECK-NEXT:    [[AND47:%.*]] = and i1 false, [[ICMP46]]
 ; CHECK-NEXT:    [[ICMP49:%.*]] = icmp ult ptr [[ARG]], null
 ; CHECK-NEXT:    [[AND50:%.*]] = and i1 [[ICMP49]], false
 ; CHECK-NEXT:    [[ICMP52:%.*]] = icmp ult ptr [[ARG1]], null
@@ -37,20 +28,21 @@ define i1 @test(ptr %arg, ptr %arg1, i1 %arg2, i1 %arg3, i1 %arg4) {
 ; CHECK-NEXT:    [[ICMP65:%.*]] = icmp ult ptr [[GETELEMENTPTR]], null
 ; CHECK-NEXT:    [[AND66:%.*]] = and i1 [[ICMP65]], false
 ; CHECK-NEXT:    [[TMP15:%.*]] = call i1 @llvm.vector.reduce.or.v8i1(<8 x i1> [[TMP6]])
-; CHECK-NEXT:    [[OP_RDX30:%.*]] = or i1 [[TMP15]], [[AND]]
-; CHECK-NEXT:    [[OP_RDX31:%.*]] = or i1 false, [[AND12]]
-; CHECK-NEXT:    [[OP_RDX26:%.*]] = or i1 [[AND15]], false
-; CHECK-NEXT:    [[OP_RDX27:%.*]] = or i1 false, [[AND47]]
 ; CHECK-NEXT:    [[OP_RDX28:%.*]] = or i1 [[AND50]], [[AND53]]
 ; CHECK-NEXT:    [[OP_RDX29:%.*]] = or i1 [[AND56]], [[AND59]]
 ; CHECK-NEXT:    [[OP_RDX33:%.*]] = or i1 [[AND63]], [[AND66]]
 ; CHECK-NEXT:    [[OP_RDX34:%.*]] = or i1 [[ARG3]], [[ARG2]]
 ; CHECK-NEXT:    [[OP_RDX32:%.*]] = or i1 [[ARG4]], false
-; CHECK-NEXT:    [[OP_RDX36:%.*]] = or i1 [[OP_RDX30]], [[OP_RDX31]]
-; CHECK-NEXT:    [[OP_RDX38:%.*]] = or i1 [[OP_RDX26]], [[OP_RDX27]]
+; CHECK-NEXT:    [[TMP16:%.*]] = shufflevector <8 x ptr> [[TMP2]], <8 x ptr> poison, <2 x i32> <i32 poison, i32 3>
+; CHECK-NEXT:    [[TMP9:%.*]] = shufflevector <2 x ptr> [[TMP16]], <2 x ptr> <ptr null, ptr poison>, <2 x i32> <i32 2, i32 1>
+; CHECK-NEXT:    [[TMP10:%.*]] = icmp ult <2 x ptr> [[TMP9]], splat (ptr null)
+; CHECK-NEXT:    [[TMP11:%.*]] = and <2 x i1> zeroinitializer, [[TMP10]]
+; CHECK-NEXT:    [[TMP12:%.*]] = insertelement <2 x i1> <i1 poison, i1 false>, i1 [[TMP15]], i64 0
+; CHECK-NEXT:    [[TMP13:%.*]] = or <2 x i1> [[TMP12]], [[TMP11]]
+; CHECK-NEXT:    [[TMP14:%.*]] = or <2 x i1> zeroinitializer, [[TMP13]]
 ; CHECK-NEXT:    [[OP_RDX35:%.*]] = or i1 [[OP_RDX28]], [[OP_RDX29]]
 ; CHECK-NEXT:    [[OP_RDX37:%.*]] = or i1 [[OP_RDX33]], [[OP_RDX34]]
-; CHECK-NEXT:    [[OP_RDX46:%.*]] = or i1 [[OP_RDX36]], [[OP_RDX38]]
+; CHECK-NEXT:    [[OP_RDX46:%.*]] = call i1 @llvm.vector.reduce.or.v2i1(<2 x i1> [[TMP14]])
 ; CHECK-NEXT:    [[TMP8:%.*]] = or i1 [[OP_RDX35]], [[OP_RDX37]]
 ; CHECK-NEXT:    [[OP_RDX39:%.*]] = or i1 [[OP_RDX46]], [[TMP8]]
 ; CHECK-NEXT:    [[OP_RDX40:%.*]] = or i1 [[OP_RDX39]], [[OP_RDX32]]
diff --git a/llvm/test/Transforms/SLPVectorizer/AMDGPU/reduction-i1-mask.ll b/llvm/test/Transforms/SLPVectorizer/AMDGPU/reduction-i1-mask.ll
index a23e972aefbac..38f807f2ce824 100644
--- a/llvm/test/Transforms/SLPVectorizer/AMDGPU/reduction-i1-mask.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AMDGPU/reduction-i1-mask.ll
@@ -13,44 +13,13 @@ define i32 @count_smaller(ptr addrspace(3) %tab, i32 %key) {
 ; CHECK-LABEL: define i32 @count_smaller(
 ; CHECK-SAME: ptr addrspace(3) [[TAB:%.*]], i32 [[KEY:%.*]]) {
 ; CHECK-NEXT:  [[ENTRY:.*:]]
-; CHECK-NEXT:    [[P1:%.*]] = getelementptr inbounds i32, ptr addrspace(3) [[TAB]], i32 1
-; CHECK-NEXT:    [[P2:%.*]] = getelementptr inbounds i32, ptr addrspace(3) [[TAB]], i32 2
-; CHECK-NEXT:    [[P3:%.*]] = getelementptr inbounds i32, ptr addrspace(3) [[TAB]], i32 3
-; CHECK-NEXT:    [[P4:%.*]] = getelementptr inbounds i32, ptr addrspace(3) [[TAB]], i32 4
-; CHECK-NEXT:    [[P5:%.*]] = getelementptr inbounds i32, ptr addrspace(3) [[TAB]], i32 5
-; CHECK-NEXT:    [[P6:%.*]] = getelementptr inbounds i32, ptr addrspace(3) [[TAB]], i32 6
-; CHECK-NEXT:    [[P7:%.*]] = getelementptr inbounds i32, ptr addrspace(3) [[TAB]], i32 7
-; CHECK-NEXT:    [[V0:%.*]] = load i32, ptr addrspace(3) [[TAB]], align 4
-; CHECK-NEXT:    [[V1:%.*]] = load i32, ptr addrspace(3) [[P1]], align 4
-; CHECK-NEXT:    [[V2:%.*]] = load i32, ptr addrspace(3) [[P2]], align 4
-; CHECK-NEXT:    [[V3:%.*]] = load i32, ptr addrspace(3) [[P3]], align 4
-; CHECK-NEXT:    [[V4:%.*]] = load i32, ptr addrspace(3) [[P4]], align 4
-; CHECK-NEXT:    [[V5:%.*]] = load i32, ptr addrspace(3) [[P5]], align 4
-; CHECK-NEXT:    [[V6:%.*]] = load i32, ptr addrspace(3) [[P6]], align 4
-; CHECK-NEXT:    [[V7:%.*]] = load i32, ptr addrspace(3) [[P7]], align 4
-; CHECK-NEXT:    [[C0:%.*]] = icmp slt i32 [[V0]], [[KEY]]
-; CHECK-NEXT:    [[C1:%.*]] = icmp slt i32 [[V1]], [[KEY]]
-; CHECK-NEXT:    [[C2:%.*]] = icmp slt i32 [[V2]], [[KEY]]
-; CHECK-NEXT:    [[C3:%.*]] = icmp slt i32 [[V3]], [[KEY]]
-; CHECK-NEXT:    [[C4:%.*]] = icmp slt i32 [[V4]], [[KEY]]
-; CHECK-NEXT:    [[C5:%.*]] = icmp slt i32 [[V5]], [[KEY]]
-; CHECK-NEXT:    [[C6:%.*]] = icmp slt i32 [[V6]], [[KEY]]
-; CHECK-NEXT:    [[C7:%.*]] = icmp slt i32 [[V7]], [[KEY]]
-; CHECK-NEXT:    [[Z0:%.*]] = zext i1 [[C0]] to i32
-; CHECK-NEXT:    [[Z1:%.*]] = zext i1 [[C1]] to i32
-; CHECK-NEXT:    [[Z2:%.*]] = zext i1 [[C2]] to i32
-; CHECK-NEXT:    [[Z3:%.*]] = zext i1 [[C3]] to i32
-; CHECK-NEXT:    [[Z4:%.*]] = zext i1 [[C4]] to i32
-; CHECK-NEXT:    [[Z5:%.*]] = zext i1 [[C5]] to i32
-; CHECK-NEXT:    [[Z6:%.*]] = zext i1 [[C6]] to i32
-; CHECK-NEXT:    [[Z7:%.*]] = zext i1 [[C7]] to i32
-; CHECK-NEXT:    [[S1:%.*]] = add i32 [[Z0]], [[Z1]]
-; CHECK-NEXT:    [[S2:%.*]] = add i32 [[S1]], [[Z2]]
-; CHECK-NEXT:    [[S3:%.*]] = add i32 [[S2]], [[Z3]]
-; CHECK-NEXT:    [[S4:%.*]] = add i32 [[S3]], [[Z4]]
-; CHECK-NEXT:    [[S5:%.*]] = add i32 [[S4]], [[Z5]]
-; CHECK-NEXT:    [[S6:%.*]] = add i32 [[S5]], [[Z6]]
-; CHECK-NEXT:    [[S7:%.*]] = add i32 [[S6]], [[Z7]]
+; CHECK-NEXT:    [[TMP0:%.*]] = load <8 x i32>, ptr addrspace(3) [[TAB]], align 4
+; CHECK-NEXT:    [[TMP1:%.*]] = insertelement <8 x i32> poison, i32 [[KEY]], i64 0
+; CHECK-NEXT:    [[TMP2:%.*]] = shufflevector <8 x i32> [[TMP1]], <8 x i32> poison, <8 x i32> zeroinitializer
+; CHECK-NEXT:    [[TMP3:%.*]] = icmp slt <8 x i32> [[TMP0]], [[TMP2]]
+; CHECK-NEXT:    [[TMP4:%.*]] = bitcast <8 x i1> [[TMP3]] to i8
+; CHECK-NEXT:    [[TMP5:%.*]] = call i8 @llvm.ctpop.i8(i8 [[TMP4]])
+; CHECK-NEXT:    [[S7:%.*]] = zext i8 [[TMP5]] to i32
 ; CHECK-NEXT:    ret i32 [[S7]]
 ;
 ; FORCED-LABEL: define i32 @count_smaller(
diff --git a/llvm/test/Transforms/SLPVectorizer/RISCV/remark-zext-incoming-for-neg-icmp.ll b/llvm/test/Transforms/SLPVectorizer/RISCV/remark-zext-incoming-for-neg-icmp.ll
index 5ef40a3f6d2f7..7f3e3ab209a85 100644
--- a/llvm/test/Transforms/SLPVectorizer/RISCV/remark-zext-incoming-for-neg-icmp.ll
+++ b/llvm/test/Transforms/SLPVectorizer/RISCV/remark-zext-incoming-for-neg-icmp.ll
@@ -8,7 +8,7 @@
 ; YAML-NEXT: Function:        test
 ; YAML-NEXT: Args:
 ; YAML-NEXT:   - String:          'Vectorized horizontal reduction with cost '
-; YAML-NEXT:   - Cost:            '-10'
+; YAML-NEXT:   - Cost:            '-11'
 ; YAML-NEXT:   - String:          ' and with tree size '
 ; YAML-NEXT:   - TreeSize:        '8'
 ; YAML-NEXT:...
@@ -24,9 +24,8 @@ define i32 @test(i32 %a, i8 %b, i8 %c) {
 ; CHECK-NEXT:    [[TMP8:%.*]] = zext <4 x i8> [[TMP2]] to <4 x i16>
 ; CHECK-NEXT:    [[TMP9:%.*]] = sext <4 x i8> [[TMP4]] to <4 x i16>
 ; CHECK-NEXT:    [[TMP5:%.*]] = icmp sle <4 x i16> [[TMP8]], [[TMP9]]
-; CHECK-NEXT:    [[TMP10:%.*]] = bitcast <4 x i1> [[TMP5]] to i4
-; CHECK-NEXT:    [[TMP11:%.*]] = call i4 @llvm.ctpop.i4(i4 [[TMP10]])
-; CHECK-NEXT:    [[TMP7:%.*]] = zext i4 [[TMP11]] to i32
+; CHECK-NEXT:    [[TMP10:%.*]] = zext <4 x i1> [[TMP5]] to <4 x i32>
+; CHECK-NEXT:    [[TMP7:%.*]] = call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> [[TMP10]])
 ; CHECK-NEXT:    [[OP_RDX:%.*]] = add i32 [[TMP7]], [[A]]
 ; CHECK-NEXT:    ret i32 [[OP_RDX]]
 ;
diff --git a/llvm/test/Transforms/SLPVectorizer/SystemZ/cmp-ptr-minmax.ll b/llvm/test/Transforms/SLPVectorizer/SystemZ/cmp-ptr-minmax.ll
index 08257a47ce06c..aa9be8c251fe6 100644
--- a/llvm/test/Transforms/SLPVectorizer/SystemZ/cmp-ptr-minmax.ll
+++ b/llvm/test/Transforms/SLPVectorizer/SystemZ/cmp-ptr-minmax.ll
@@ -18,9 +18,7 @@ define i1 @test(i64 %0, i64 %1, ptr %2) {
 ; CHECK-NEXT:    [[TMP9:%.*]] = insertelement <2 x ptr> poison, ptr [[TMP2]], i64 0
 ; CHECK-NEXT:    [[TMP10:%.*]] = shufflevector <2 x ptr> [[TMP9]], <2 x ptr> poison, <2 x i32> zeroinitializer
 ; CHECK-NEXT:    [[TMP11:%.*]] = icmp ult <2 x ptr> [[TMP8]], [[TMP10]]
-; CHECK-NEXT:    [[TMP12:%.*]] = extractelement <2 x i1> [[TMP11]], i64 0
-; CHECK-NEXT:    [[TMP13:%.*]] = extractelement <2 x i1> [[TMP11]], i64 1
-; CHECK-NEXT:    [[RES:%.*]] = and i1 [[TMP12]], [[TMP13]]
+; CHECK-NEXT:    [[RES:%.*]] = call i1 @llvm.vector.reduce.and.v2i1(<2 x i1> [[TMP11]])
 ; CHECK-NEXT:    ret i1 [[RES]]
 ;
 entry:
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/entries-different-vf.ll b/llvm/test/Transforms/SLPVectorizer/X86/entries-different-vf.ll
index a46467563bb24..53cba0f957935 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/entries-different-vf.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/entries-different-vf.ll
@@ -17,8 +17,9 @@ define i1 @test(i64 %v) {
 ; CHECK-NEXT:    [[TMP9:%.*]] = sub <8 x i64> [[TMP7]], [[TMP5]]
 ; CHECK-NEXT:    [[TMP10:%.*]] = shufflevector <8 x i64> [[TMP8]], <8 x i64> [[TMP9]], <8 x i32> <i32 0, i32 1, i32 2, i32 11, i32 12, i32 5, i32 6, i32 7>
 ; CHECK-NEXT:    [[TMP11:%.*]] = icmp ult <8 x i64> [[TMP10]], zeroinitializer
-; CHECK-NEXT:    [[TMP12:%.*]] = call i1 @llvm.vector.reduce.or.v8i1(<8 x i1> [[TMP11]])
-; CHECK-NEXT:    ret i1 [[TMP12]]
+; CHECK-NEXT:    [[TMP12:%.*]] = bitcast <8 x i1> [[TMP11]] to i8
+; CHECK-NEXT:    [[TMP13:%.*]] = icmp ne i8 [[TMP12]], 0
+; CHECK-NEXT:    ret i1 [[TMP13]]
 ;
 entry:
   %0 = shl i64 %v, 1
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/fabs-cost-softfp.ll b/llvm/test/Transforms/SLPVectorizer/X86/fabs-cost-softfp.ll
index fdb43f683f644..28b4e8f2cc889 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/fabs-cost-softfp.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/fabs-cost-softfp.ll
@@ -15,9 +15,7 @@ define void @vectorize_fp128(fp128 %c, fp128 %d) #0 {
 ; CHECK-NEXT:    [[TMP1:%.*]] = insertelement <2 x fp128> [[TMP0]], fp128 [[D:%.*]], i64 1
 ; CHECK-NEXT:    [[TMP2:%.*]] = call <2 x fp128> @llvm.fabs.v2f128(<2 x fp128> [[TMP1]])
 ; CHECK-NEXT:    [[TMP3:%.*]] = fcmp oeq <2 x fp128> [[TMP2]], splat (fp128 +inf)
-; CHECK-NEXT:    [[TMP4:%.*]] = extractelement <2 x i1> [[TMP3]], i64 0
-; CHECK-NEXT:    [[TMP5:%.*]] = extractelement <2 x i1> [[TMP3]], i64 1
-; CHECK-NEXT:    [[OR_COND39:%.*]] = or i1 [[TMP4]], [[TMP5]]
+; CHECK-NEXT:    [[OR_COND39:%.*]] = call i1 @llvm.vector.reduce.or.v2i1(<2 x i1> [[TMP3]])
 ; CHECK-NEXT:    br i1 [[OR_COND39]], label [[IF_THEN13:%.*]], label [[IF_END24:%.*]]
 ; CHECK:       if.then13:
 ; CHECK-NEXT:    unreachable
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/reduction-logical.ll b/llvm/test/Transforms/SLPVectorizer/X86/reduction-logical.ll
index 6b073b10bac9b..8f160742246f4 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/reduction-logical.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/reduction-logical.ll
@@ -5,11 +5,18 @@
 declare void @use1(i1)
 
 define i1 @logical_and_icmp(<4 x i32> %x) {
-; CHECK-LABEL: @logical_and_icmp(
-; CHECK-NEXT:    [[TMP1:%.*]] = icmp slt <4 x i32> [[X:%.*]], zeroinitializer
-; CHECK-NEXT:    [[TMP2:%.*]] = freeze <4 x i1> [[TMP1]]
-; CHECK-NEXT:    [[TMP3:%.*]] = call i1 @llvm.vector.reduce.and.v4i1(<4 x i1> [[TMP2]])
-; CHECK-NEXT:    ret i1 [[TMP3]]
+; SSE-LABEL: @logical_and_icmp(
+; SSE-NEXT:    [[TMP1:%.*]] = icmp slt <4 x i32> [[X:%.*]], zeroinitializer
+; SSE-NEXT:    [[TMP2:%.*]] = freeze <4 x i1> [[TMP1]]
+; SSE-NEXT:    [[TMP3:%.*]] = call i1 @llvm.vector.reduce.and.v4i1(<4 x i1> [[TMP2]])
+; SSE-NEXT:    ret i1 [[TMP3]]
+;
+; AVX-LABEL: @logical_and_icmp(
+; AVX-NEXT:    [[TMP1:%.*]] = icmp slt <4 x i32> [[X:%.*]], zeroinitializer
+; AVX-NEXT:    [[TMP2:%.*]] = freeze <4 x i1> [[TMP1]]
+; AVX-NEXT:    [[TMP3:%.*]] = bitcast <4 x i1> [[TMP2]] to i4
+; AVX-NEXT:    [[TMP4:%.*]] = icmp eq i4 [[TMP3]], -1
+; AVX-NEXT:    ret i1 [[TMP4]]
 ;
   %x0 = extractelement <4 x i32> %x, i32 0
   %x1 = extractelement <4 x i32> %x, i32 1
@@ -26,11 +33,18 @@ define i1 @logical_and_icmp(<4 x i32> %x) {
 }
 
 define i1 @logical_or_icmp(<4 x i32> %x, <4 x i32> %y) {
-; CHECK-LABEL: @logical_or_icmp(
-; CHECK-NEXT:    [[TMP1:%.*]] = icmp slt <4 x i32> [[X:%.*]], [[Y:%.*]]
-; CHECK-NEXT:    [[TMP2:%.*]] = freeze <4 x i1> [[TMP1]]
-; CHECK-NEXT:    [[TMP3:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP2]])
-; CHECK-NEXT:    ret i1 [[TMP3]]
+; SSE-LABEL: @logical_or_icmp(
+; SSE-NEXT:    [[TMP1:%.*]] = icmp slt <4 x i32> [[X:%.*]], [[Y:%.*]]
+; SSE-NEXT:    [[TMP2:%.*]] = freeze <4 x i1> [[TMP1]]
+; SSE-NEXT:    [[TMP3:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP2]])
+; SSE-NEXT:    ret i1 [[TMP3]]
+;
+; AVX-LABEL: @logical_or_icmp(
+; AVX-NEXT:    [[TMP1:%.*]] = icmp slt <4 x i32> [[X:%.*]], [[Y:%.*]]
+; AVX-NEXT:    [[TMP2:%.*]] = freeze <4 x i1> [[TMP1]]
+; AVX-NEXT:    [[TMP3:%.*]] = bitcast <4 x i1> [[TMP2]] to i4
+; AVX-NEXT:    [[TMP4:%.*]] = icmp ne i4 [[TMP3]], 0
+; AVX-NEXT:    ret i1 [[TMP4]]
 ;
   %x0 = extractelement <4 x i32> %x, i32 0
   %x1 = extractelement <4 x i32> %x, i32 1
@@ -51,11 +65,18 @@ define i1 @logical_or_icmp(<4 x i32> %x, <4 x i32> %y) {
 }
 
 define i1 @logical_and_fcmp(<4 x float> %x) {
-; CHECK-LABEL: @logical_and_fcmp(
-; CHECK-NEXT:    [[TMP1:%.*]] = fcmp olt <4 x float> [[X:%.*]], zeroinitializer
-; CHECK-NEXT:    [[TMP2:%.*]] = freeze <4 x i1> [[TMP1]]
-; CHECK-NEXT:    [[TMP3:%.*]] = call i1 @llvm.vector.reduce.and.v4i1(<4 x i1> [[TMP2]])
-; CHECK-NEXT:    ret i1 [[TMP3]]
+; SSE-LABEL: @logical_and_fcmp(
+; SSE-NEXT:    [[TMP1:%.*]] = fcmp olt <4 x float> [[X:%.*]], zeroinitializer
+; SSE-NEXT:    [[TMP2:%.*]] = freeze <4 x i1> [[TMP1]]
+; SSE-NEXT:    [[TMP3:%.*]] = call i1 @llvm.vector.reduce.and.v4i1(<4 x i1> [[TMP2]])
+; SSE-NEXT:    ret i1 [[TMP3]]
+;
+; AVX-LABEL: @logical_and_fcmp(
+; AVX-NEXT:    [[TMP1:%.*]] = fcmp olt <4 x float> [[X:%.*]], zeroinitializer
+; AVX-NEXT:    [[TMP2:%.*]] = freeze <4 x i1> [[TMP1]]
+; AVX-NEXT:    [[TMP3:%.*]] = bitcast <4 x i1> [[TMP2]] to i4
+; AVX-NEXT:    [[TMP4:%.*]] = icmp eq i4 [[TMP3]], -1
+; AVX-NEXT:    ret i1 [[TMP4]]
 ;
   %x0 = extractelement <4 x float> %x, i32 0
   %x1 = extractelement <4 x float> %x, i32 1
@@ -72,11 +93,18 @@ define i1 @logical_and_fcmp(<4 x float> %x) {
 }
 
 define i1 @logical_or_fcmp(<4 x float> %x) {
-; CHECK-LABEL: @logical_or_fcmp(
-; CHECK-NEXT:    [[TMP1:%.*]] = fcmp olt <4 x float> [[X:%.*]], zeroinitializer
-; CHECK-NEXT:    [[TMP2:%.*]] = freeze <4 x i1> [[TMP1]]
-; CHECK-NEXT:    [[TMP3:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP2]])
-; CHECK-NEXT:    ret i1 [[TMP3]]
+; SSE-LABEL: @logical_or_fcmp(
+; SSE-NEXT:    [[TMP1:%.*]] = fcmp olt <4 x float> [[X:%.*]], zeroinitializer
+; SSE-NEXT:    [[TMP2:%.*]] = freeze <4 x i1> [[TMP1]]
+; SSE-NEXT:    [[TMP3:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP2]])
+; SSE-NEXT:    ret i1 [[TMP3]]
+;
+; AVX-LABEL: @logical_or_fcmp(
+; AVX-NEXT:    [[TMP1:%.*]] = fcmp olt <4 x float> [[X:%.*]], zeroinitializer
+; AVX-NEXT:    [[TMP2:%.*]] = freeze <4 x i1> [[TMP1]]
+; AVX-NEXT:    [[TMP3:%.*]] = bitcast <4 x i1> [[TMP2]] to i4
+; AVX-NEXT:    [[TMP4:%.*]] = icmp ne i4 [[TMP3]], 0
+; AVX-NEXT:    ret i1 [[TMP4]]
 ;
   %x0 = extractelement <4 x float> %x, i32 0
   %x1 = extractelement <4 x float> %x, i32 1
@@ -104,18 +132,15 @@ define i1 @logical_and_icmp_diff_preds(<4 x i32> %x) {
 ; SSE-NEXT:    ret i1 [[TMP7]]
 ;
 ; AVX-LABEL: @logical_and_icmp_diff_preds(
-; AVX-NEXT:    [[X0:%.*]] = extractelement <4 x i32> [[X:%.*]], i32 0
-; AVX-NEXT:    [[X1:%.*]] = extractelement <4 x i32> [[X]], i32 1
-; AVX-NEXT:    [[X2:%.*]] = extractelement <4 x i32> [[X]], i32 2
-; AVX-NEXT:    [[X3:%.*]] = extractelement <4 x i32> [[X]], i32 3
-; AVX-NEXT:    [[C0:%.*]] = icmp ult i32 [[X0]], 0
-; AVX-NEXT:    [[C1:%.*]] = icmp slt i32 [[X1]], 0
-; AVX-NEXT:    [[C2:%.*]] = icmp sgt i32 [[X2]], 0
-; AVX-NEXT:    [[C3:%.*]] = icmp slt i32 [[X3]], 0
-; AVX-NEXT:    [[S1:%.*]] = select i1 [[C0]], i1 [[C1]], i1 false
-; AVX-NEXT:    [[S2:%.*]] = select i1 [[S1]], i1 [[C2]], i1 false
-; AVX-NEXT:    [[S3:%.*]] = select i1 [[S2]], i1 [[C3]], i1 false
-; AVX-NEXT:    ret i1 [[S3]]
+; AVX-NEXT:    [[TMP1:%.*]] = shufflevector <4 x i32> [[X:%.*]], <4 x i32> <i32 poison, i32 poison, i32 0, i32 poison>, <4 x i32> <i32 1, i32 3, i32 6, i32 0>
+; AVX-NEXT:    [[TMP2:%.*]] = shufflevector <4 x i32> [[X]], <4 x i32> <i32 0, i32 0, i32 poison, i32 0>, <4 x i32> <i32 4, i32 5, i32 2, i32 7>
+; AVX-NEXT:    [[TMP3:%.*]] = icmp slt <4 x i32> [[TMP1]], [[TMP2]]
+; AVX-NEXT:    [[TMP4:%.*]] = icmp ult <4 x i32> [[TMP1]], [[TMP2]]
+; AVX-NEXT:    [[TMP5:%.*]] = shufflevector <4 x i1> [[TMP3]], <4 x i1> [[TMP4]], <4 x i32> <i32 0, i32 1, i32 2, i32 7>
+; AVX-NEXT:    [[TMP6:%.*]] = freeze <4 x i1> [[TMP5]]
+; AVX-NEXT:    [[TMP7:%.*]] = bitcast <4 x i1> [[TMP6]] to i4
+; AVX-NEXT:    [[TMP8:%.*]] = icmp eq i4 [[TMP7]], -1
+; AVX-NEXT:    ret i1 [[TMP8]]
 ;
   %x0 = extractelement <4 x i32> %x, i32 0
   %x1 = extractelement <4 x i32> %x, i32 1
@@ -132,11 +157,18 @@ define i1 @logical_and_icmp_diff_preds(<4 x i32> %x) {
 }
 
 define i1 @logical_and_icmp_diff_const(<4 x i32> %x) {
-; CHECK-LABEL: @logical_and_icmp_diff_const(
-; CHECK-NEXT:    [[TMP1:%.*]] = icmp sgt <4 x i32> [[X:%.*]], <i32 0, i32 1, i32 2, i32 3>
-; CHECK-NEXT:    [[TMP2:%.*]] = freeze <4 x i1> [[TMP1]]
-; CHECK-NEXT:    [[TMP3:%.*]] = call i1 @llvm.vector.reduce.and.v4i1(<4 x i1> [[TMP2]])
-; CHECK-NEXT:    ret i1 [[TMP3]]
+; SSE-LABEL: @logical_and_icmp_diff_const(
+; SSE-NEXT:    [[TMP1:%.*]] = icmp sgt <4 x i32> [[X:%.*]], <i32 0, i32 1, i32 2, i32 3>
+; SSE-NEXT:    [[TMP2:%.*]] = freeze <4 x i1> [[TMP1]]
+; SSE-NEXT:    [[TMP3:%.*]] = call i1 @llvm.vector.reduce.and.v4i1(<4 x i1> [[TMP2]])
+; SSE-NEXT:    ret i1 [[TMP3]]
+;
+; AVX-LABEL: @logical_and_icmp_diff_const(
+; AVX-NEXT:    [[TMP1:%.*]] = icmp sgt <4 x i32> [[X:%.*]], <i32 0, i32 1, i32 2, i32 3>
+; AVX-NEXT:    [[TMP2:%.*]] = freeze <4 x i1> [[TMP1]]
+; AVX-NEXT:    [[TMP3:%.*]] = bitcast <4 x i1> [[TMP2]] to i4
+; AVX-NEXT:    [[TMP4:%.*]] = icmp eq i4 [[TMP3]], -1
+; AVX-NEXT:    ret i1 [[TMP4]]
 ;
   %x0 = extractelement <4 x i32> %x, i32 0
   %x1 = extractelement <4 x i32> %x, i32 1
@@ -206,15 +238,26 @@ define i1 @logical_and_icmp_subvec(<4 x i32> %x) {
 ;       logic...or a wide reduction?
 
 define i1 @logical_and_icmp_clamp(<4 x i32> %x) {
-; CHECK-LABEL: @logical_and_icmp_clamp(
-; CHECK-NEXT:    [[TMP1:%.*]] = icmp slt <4 x i32> [[X:%.*]], splat (i32 42)
-; CHECK-NEXT:    [[TMP2:%.*]] = icmp sgt <4 x i32> [[X]], splat (i32 17)
-; CHECK-NEXT:    [[TMP3:%.*]] = shufflevector <4 x i1> [[TMP2]], <4 x i1> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT:    [[TMP7:%.*]] = shufflevector <4 x i1> [[TMP1]], <4 x i1> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT:    [[TMP4:%.*]] = shufflevector <8 x i1> [[TMP3]], <8 x i1> [[TMP7]], <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 8, i32 9, i32 10, i32 11>
-; CHECK-NEXT:    [[TMP5:%.*]] = freeze <8 x i1> [[TMP4]]
-; CHECK-NEXT:    [[TMP6:%.*]] = call i1 @llvm.vector.reduce.and.v8i1(<8 x i1> [[TMP5]])
-; CHECK-NEXT:    ret i1 [[TMP6]]
+; SSE-LABEL: @logical_and_icmp_clamp(
+; SSE-NEXT:    [[TMP1:%.*]] = icmp slt <4 x i32> [[X:%.*]], splat (i32 42)
+; SSE-NEXT:    [[TMP2:%.*]] = icmp sgt <4 x i32> [[X]], splat (i32 17)
+; SSE-NEXT:    [[TMP3:%.*]] = shufflevector <4 x i1> [[TMP2]], <4 x i1> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison>
+; SSE-NEXT:    [[TMP4:%.*]] = shufflevector <4 x i1> [[TMP1]], <4 x i1> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison>
+; SSE-NEXT:    [[TMP5:%.*]] = shufflevector <8 x i1> [[TMP3]], <8 x i1> [[TMP4]], <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 8, i32 9, i32 10, i32 11>
+; SSE-NEXT:    [[TMP6:%.*]] = freeze <8 x i1> [[TMP5]]
+; SSE-NEXT:    [[TMP7:%.*]] = call i1 @llvm.vector.reduce.and.v8i1(<8 x i1> [[TMP6]])
+; SSE-NEXT:    ret i1 [[TMP7]]
+;
+; AVX-LABEL: @logical_and_icmp_clamp(
+; AVX-NEXT:    [[TMP1:%.*]] = icmp slt <4 x i32> [[X:%.*]], splat (i32 42)
+; AVX-NEXT:    [[TMP2:%.*]] = icmp sgt <4 x i32> [[X]], splat (i32 17)
+; AVX-NEXT:    [[TMP3:%.*]] = shufflevector <4 x i1> [[TMP2]], <4 x i1> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison>
+; AVX-NEXT:    [[TMP4:%.*]] = shufflevector <4 x i1> [[TMP1]], <4 x i1> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison>
+; AVX-NEXT:    [[TMP5:%.*]] = shufflevector <8 x i1> [[TMP3]], <8 x i1> [[TMP4]], <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 8, i32 9, i32 10, i32 11>
+; AVX-NEXT:    [[TMP6:%.*]] = freeze <8 x i1> [[TMP5]]
+; AVX-NEXT:    [[TMP7:%.*]] = bitcast <8 x i1> [[TMP6]] to i8
+; AVX-NEXT:    [[TMP8:%.*]] = icmp eq i8 [[TMP7]], -1
+; AVX-NEXT:    ret i1 [[TMP8]]
 ;
   %x0 = extractelement <4 x i32> %x, i32 0
   %x1 = extractelement <4 x i32> %x, i32 1
@@ -239,17 +282,30 @@ define i1 @logical_and_icmp_clamp(<4 x i32> %x) {
 }
 
 define i1 @logical_and_icmp_clamp_extra_use_cmp(<4 x i32> %x) {
-; CHECK-LABEL: @logical_and_icmp_clamp_extra_use_cmp(
-; CHECK-NEXT:    [[TMP1:%.*]] = icmp slt <4 x i32> [[X:%.*]], splat (i32 42)
-; CHECK-NEXT:    [[TMP5:%.*]] = extractelement <4 x i1> [[TMP1]], i64 2
-; CHECK-NEXT:    call void @use1(i1 [[TMP5]])
-; CHECK-NEXT:    [[TMP3:%.*]] = icmp sgt <4 x i32> [[X]], splat (i32 17)
-; CHECK-NEXT:    [[TMP8:%.*]] = shufflevector <4 x i1> [[TMP3]], <4 x i1> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT:    [[TMP9:%.*]] = shufflevector <4 x i1> [[TMP1]], <4 x i1> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT:    [[TMP4:%.*]] = shufflevector <8 x i1> [[TMP8]], <8 x i1> [[TMP9]], <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 8, i32 9, i32 10, i32 11>
-; CHECK-NEXT:    [[TMP6:%.*]] = freeze <8 x i1> [[TMP4]]
-; CHECK-NEXT:    [[TMP7:%.*]] = call i1 @llvm.vector.reduce.and.v8i1(<8 x i1> [[TMP6]])
-; CHECK-NEXT:    ret i1 [[TMP7]]
+; SSE-LABEL: @logical_and_icmp_clamp_extra_use_cmp(
+; SSE-NEXT:    [[TMP1:%.*]] = icmp slt <4 x i32> [[X:%.*]], splat (i32 42)
+; SSE-NEXT:    [[TMP2:%.*]] = extractelement <4 x i1> [[TMP1]], i64 2
+; SSE-NEXT:    call void @use1(i1 [[TMP2]])
+; SSE-NEXT:    [[TMP3:%.*]] = icmp sgt <4 x i32> [[X]], splat (i32 17)
+; SSE-NEXT:    [[TMP4:%.*]] = shufflevector <4 x i1> [[TMP3]], <4 x i1> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison>
+; SSE-NEXT:    [[TMP5:%.*]] = shufflevector <4 x i1> [[TMP1]], <4 x i1> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison>
+; SSE-NEXT:    [[TMP6:%.*]] = shufflevector <8 x i1> [[TMP4]], <8 x i1> [[TMP5]], <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 8, i32 9, i32 10, i32 11>
+; SSE-NEXT:    [[TMP7:%.*]] = freeze <8 x i1> [[TMP6]]
+; SSE-NEXT:    [[TMP8:%.*]] = call i1 @llvm.vector.reduce.and.v8i1(<8 x i1> [[TMP7]])
+; SSE-NEXT:    ret i1 [[TMP8]]
+;
+; AVX-LABEL: @logical_and_icmp_clamp_extra_use_cmp(
+; AVX-NEXT:    [[TMP1:%.*]] = icmp slt <4 x i32> [[X:%.*]], splat (i32 42)
+; AVX-NEXT:    [[TMP2:%.*]] = extractelement <4 x i1> [[TMP1]], i64 2
+; AVX-NEXT:    call void @use1(i1 [[TMP2]])
+; AVX-NEXT:    [[TMP3:%.*]] = icmp sgt <4 x i32> [[X]], splat (i32 17)
+; AVX-NEXT:    [[TMP4:%.*]] = shufflevector <4 x i1> [[TMP3]], <4 x i1> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison>
+; AVX-NEXT:    [[TMP5:%.*]] = shufflevector <4 x i1> [[TMP1]], <4 x i1> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison>
+; AVX-NEXT:    [[TMP6:%.*]] = shufflevector <8 x i1> [[TMP4]], <8 x i1> [[TMP5]], <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 8, i32 9, i32 10, i32 11>
+; AVX-NEXT:    [[TMP7:%.*]] = freeze <8 x i1> [[TMP6]]
+; AVX-NEXT:    [[TMP8:%.*]] = bitcast <8 x i1> [[TMP7]] to i8
+; AVX-NEXT:    [[TMP9:%.*]] = icmp eq i8 [[TMP8]], -1
+; AVX-NEXT:    ret i1 [[TMP9]]
 ;
   %x0 = extractelement <4 x i32> %x, i32 0
   %x1 = extractelement <4 x i32> %x, i32 1
@@ -275,21 +331,38 @@ define i1 @logical_and_icmp_clamp_extra_use_cmp(<4 x i32> %x) {
 }
 
 define i1 @logical_and_icmp_clamp_extra_use_select(<4 x i32> %x) {
-; CHECK-LABEL: @logical_and_icmp_clamp_extra_use_select(
-; CHECK-NEXT:    [[TMP1:%.*]] = icmp slt <4 x i32> [[X:%.*]], splat (i32 42)
-; CHECK-NEXT:    [[TMP2:%.*]] = icmp sgt <4 x i32> [[X]], splat (i32 17)
-; CHECK-NEXT:    [[TMP3:%.*]] = extractelement <4 x i1> [[TMP1]], i64 0
-; CHECK-NEXT:    [[TMP4:%.*]] = extractelement <4 x i1> [[TMP1]], i64 1
-; CHECK-NEXT:    [[S1:%.*]] = select i1 [[TMP3]], i1 [[TMP4]], i1 false
-; CHECK-NEXT:    [[TMP5:%.*]] = extractelement <4 x i1> [[TMP1]], i64 2
-; CHECK-NEXT:    [[S2:%.*]] = select i1 [[S1]], i1 [[TMP5]], i1 false
-; CHECK-NEXT:    call void @use1(i1 [[S2]])
-; CHECK-NEXT:    [[TMP6:%.*]] = freeze <4 x i1> [[TMP2]]
-; CHECK-NEXT:    [[TMP7:%.*]] = call i1 @llvm.vector.reduce.and.v4i1(<4 x i1> [[TMP6]])
-; CHECK-NEXT:    [[TMP8:%.*]] = extractelement <4 x i1> [[TMP1]], i64 3
-; CHECK-NEXT:    [[OP_RDX:%.*]] = select i1 [[TMP7]], i1 [[TMP8]], i1 false
-; CHECK-NEXT:    [[OP_RDX1:%.*]] = select i1 [[S2]], i1 [[OP_RDX]], i1 false
-; CHECK-NEXT:    ret i1 [[OP_RDX1]]
+; SSE-LABEL: @logical_and_icmp_clamp_extra_use_select(
+; SSE-NEXT:    [[TMP1:%.*]] = icmp slt <4 x i32> [[X:%.*]], splat (i32 42)
+; SSE-NEXT:    [[TMP2:%.*]] = icmp sgt <4 x i32> [[X]], splat (i32 17)
+; SSE-NEXT:    [[TMP3:%.*]] = extractelement <4 x i1> [[TMP1]], i64 0
+; SSE-NEXT:    [[TMP4:%.*]] = extractelement <4 x i1> [[TMP1]], i64 1
+; SSE-NEXT:    [[S1:%.*]] = select i1 [[TMP3]], i1 [[TMP4]], i1 false
+; SSE-NEXT:    [[TMP5:%.*]] = extractelement <4 x i1> [[TMP1]], i64 2
+; SSE-NEXT:    [[S2:%.*]] = select i1 [[S1]], i1 [[TMP5]], i1 false
+; SSE-NEXT:    call void @use1(i1 [[S2]])
+; SSE-NEXT:    [[TMP6:%.*]] = freeze <4 x i1> [[TMP2]]
+; SSE-NEXT:    [[TMP7:%.*]] = call i1 @llvm.vector.reduce.and.v4i1(<4 x i1> [[TMP6]])
+; SSE-NEXT:    [[TMP8:%.*]] = extractelement <4 x i1> [[TMP1]], i64 3
+; SSE-NEXT:    [[OP_RDX:%.*]] = select i1 [[TMP7]], i1 [[TMP8]], i1 false
+; SSE-NEXT:    [[OP_RDX1:%.*]] = select i1 [[S2]], i1 [[OP_RDX]], i1 false
+; SSE-NEXT:    ret i1 [[OP_RDX1]]
+;
+; AVX-LABEL: @logical_and_icmp_clamp_extra_use_select(
+; AVX-NEXT:    [[TMP1:%.*]] = icmp slt <4 x i32> [[X:%.*]], splat (i32 42)
+; AVX-NEXT:    [[TMP2:%.*]] = icmp sgt <4 x i32> [[X]], splat (i32 17)
+; AVX-NEXT:    [[TMP3:%.*]] = extractelement <4 x i1> [[TMP1]], i64 0
+; AVX-NEXT:    [[TMP4:%.*]] = extractelement <4 x i1> [[TMP1]], i64 1
+; AVX-NEXT:    [[S1:%.*]] = select i1 [[TMP3]], i1 [[TMP4]], i1 false
+; AVX-NEXT:    [[TMP5:%.*]] = extractelement <4 x i1> [[TMP1]], i64 2
+; AVX-NEXT:    [[S2:%.*]] = select i1 [[S1]], i1 [[TMP5]], i1 false
+; AVX-NEXT:    call void @use1(i1 [[S2]])
+; AVX-NEXT:    [[TMP6:%.*]] = freeze <4 x i1> [[TMP2]]
+; AVX-NEXT:    [[TMP7:%.*]] = bitcast <4 x i1> [[TMP6]] to i4
+; AVX-NEXT:    [[TMP8:%.*]] = icmp eq i4 [[TMP7]], -1
+; AVX-NEXT:    [[TMP9:%.*]] = extractelement <4 x i1> [[TMP1]], i64 3
+; AVX-NEXT:    [[OP_RDX:%.*]] = select i1 [[TMP8]], i1 [[TMP9]], i1 false
+; AVX-NEXT:    [[OP_RDX1:%.*]] = select i1 [[S2]], i1 [[OP_RDX]], i1 false
+; AVX-NEXT:    ret i1 [[OP_RDX1]]
 ;
   %x0 = extractelement <4 x i32> %x, i32 0
   %x1 = extractelement <4 x i32> %x, i32 1
@@ -315,13 +388,22 @@ define i1 @logical_and_icmp_clamp_extra_use_select(<4 x i32> %x) {
 }
 
 define i1 @logical_and_icmp_clamp_v8i32(<8 x i32> %x, <8 x i32> %y) {
-; CHECK-LABEL: @logical_and_icmp_clamp_v8i32(
-; CHECK-NEXT:    [[TMP1:%.*]] = shufflevector <8 x i32> [[X:%.*]], <8 x i32> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 0, i32 1, i32 2, i32 3>
-; CHECK-NEXT:    [[TMP3:%.*]] = shufflevector <8 x i32> [[Y:%.*]], <8 x i32> <i32 42, i32 42, i32 42, i32 42, i32 poison, i32 poison, i32 poison, i32 poison>, <8 x i32> <i32 8, i32 9, i32 10, i32 11, i32 0, i32 1, i32 2, i32 3>
-; CHECK-NEXT:    [[TMP4:%.*]] = icmp slt <8 x i32> [[TMP1]], [[TMP3]]
-; CHECK-NEXT:    [[TMP5:%.*]] = freeze <8 x i1> [[TMP4]]
-; CHECK-NEXT:    [[TMP6:%.*]] = call i1 @llvm.vector.reduce.and.v8i1(<8 x i1> [[TMP5]])
-; CHECK-NEXT:    ret i1 [[TMP6]]
+; SSE-LABEL: @logical_and_icmp_clamp_v8i32(
+; SSE-NEXT:    [[TMP1:%.*]] = shufflevector <8 x i32> [[X:%.*]], <8 x i32> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 0, i32 1, i32 2, i32 3>
+; SSE-NEXT:    [[TMP2:%.*]] = shufflevector <8 x i32> [[Y:%.*]], <8 x i32> <i32 42, i32 42, i32 42, i32 42, i32 poison, i32 poison, i32 poison, i32 poison>, <8 x i32> <i32 8, i32 9, i32 10, i32 11, i32 0, i32 1, i32 2, i32 3>
+; SSE-NEXT:    [[TMP3:%.*]] = icmp slt <8 x i32> [[TMP1]], [[TMP2]]
+; SSE-NEXT:    [[TMP4:%.*]] = freeze <8 x i1> [[TMP3]]
+; SSE-NEXT:    [[TMP5:%.*]] = call i1 @llvm.vector.reduce.and.v8i1(<8 x i1> [[TMP4]])
+; SSE-NEXT:    ret i1 [[TMP5]]
+;
+; AVX-LABEL: @logical_and_icmp_clamp_v8i32(
+; AVX-NEXT:    [[TMP1:%.*]] = shufflevector <8 x i32> [[X:%.*]], <8 x i32> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 0, i32 1, i32 2, i32 3>
+; AVX-NEXT:    [[TMP2:%.*]] = shufflevector <8 x i32> [[Y:%.*]], <8 x i32> <i32 42, i32 42, i32 42, i32 42, i32 poison, i32 poison, i32 poison, i32 poison>, <8 x i32> <i32 8, i32 9, i32 10, i32 11, i32 0, i32 1, i32 2, i32 3>
+; AVX-NEXT:    [[TMP3:%.*]] = icmp slt <8 x i32> [[TMP1]], [[TMP2]]
+; AVX-NEXT:    [[TMP4:%.*]] = freeze <8 x i1> [[TMP3]]
+; AVX-NEXT:    [[TMP5:%.*]] = bitcast <8 x i1> [[TMP4]] to i8
+; AVX-NEXT:    [[TMP6:%.*]] = icmp eq i8 [[TMP5]], -1
+; AVX-NEXT:    ret i1 [[TMP6]]
 ;
   %x0 = extractelement <8 x i32> %x, i32 0
   %x1 = extractelement <8 x i32> %x, i32 1
@@ -350,22 +432,40 @@ define i1 @logical_and_icmp_clamp_v8i32(<8 x i32> %x, <8 x i32> %y) {
 }
 
 define i1 @logical_and_icmp_clamp_partial(<4 x i32> %x) {
-; CHECK-LABEL: @logical_and_icmp_clamp_partial(
-; CHECK-NEXT:    [[TMP1:%.*]] = extractelement <4 x i32> [[X:%.*]], i64 2
-; CHECK-NEXT:    [[TMP2:%.*]] = shufflevector <4 x i32> [[X]], <4 x i32> poison, <2 x i32> <i32 0, i32 1>
-; CHECK-NEXT:    [[TMP3:%.*]] = icmp slt <2 x i32> [[TMP2]], splat (i32 42)
-; CHECK-NEXT:    [[C2:%.*]] = icmp slt i32 [[TMP1]], 42
-; CHECK-NEXT:    [[TMP4:%.*]] = icmp sgt <4 x i32> [[X]], splat (i32 17)
-; CHECK-NEXT:    [[TMP5:%.*]] = freeze <4 x i1> [[TMP4]]
-; CHECK-NEXT:    [[TMP6:%.*]] = call i1 @llvm.vector.reduce.and.v4i1(<4 x i1> [[TMP5]])
-; CHECK-NEXT:    [[TMP7:%.*]] = extractelement <2 x i1> [[TMP3]], i64 0
-; CHECK-NEXT:    [[OP_RDX:%.*]] = select i1 [[TMP6]], i1 [[TMP7]], i1 false
-; CHECK-NEXT:    [[TMP8:%.*]] = extractelement <2 x i1> [[TMP3]], i64 1
-; CHECK-NEXT:    [[TMP9:%.*]] = freeze i1 [[TMP8]]
-; CHECK-NEXT:    [[OP_RDX1:%.*]] = select i1 [[TMP9]], i1 [[C2]], i1 false
-; CHECK-NEXT:    [[TMP10:%.*]] = freeze i1 [[OP_RDX]]
-; CHECK-NEXT:    [[OP_RDX2:%.*]] = select i1 [[TMP10]], i1 [[OP_RDX1]], i1 false
-; CHECK-NEXT:    ret i1 [[OP_RDX2]]
+; SSE-LABEL: @logical_and_icmp_clamp_partial(
+; SSE-NEXT:    [[TMP1:%.*]] = extractelement <4 x i32> [[X:%.*]], i64 2
+; SSE-NEXT:    [[TMP2:%.*]] = shufflevector <4 x i32> [[X]], <4 x i32> poison, <2 x i32> <i32 0, i32 1>
+; SSE-NEXT:    [[TMP3:%.*]] = icmp slt <2 x i32> [[TMP2]], splat (i32 42)
+; SSE-NEXT:    [[C2:%.*]] = icmp slt i32 [[TMP1]], 42
+; SSE-NEXT:    [[TMP4:%.*]] = icmp sgt <4 x i32> [[X]], splat (i32 17)
+; SSE-NEXT:    [[TMP5:%.*]] = freeze <4 x i1> [[TMP4]]
+; SSE-NEXT:    [[TMP6:%.*]] = call i1 @llvm.vector.reduce.and.v4i1(<4 x i1> [[TMP5]])
+; SSE-NEXT:    [[TMP7:%.*]] = extractelement <2 x i1> [[TMP3]], i64 0
+; SSE-NEXT:    [[OP_RDX:%.*]] = select i1 [[TMP6]], i1 [[TMP7]], i1 false
+; SSE-NEXT:    [[TMP8:%.*]] = extractelement <2 x i1> [[TMP3]], i64 1
+; SSE-NEXT:    [[TMP9:%.*]] = freeze i1 [[TMP8]]
+; SSE-NEXT:    [[OP_RDX1:%.*]] = select i1 [[TMP9]], i1 [[C2]], i1 false
+; SSE-NEXT:    [[TMP10:%.*]] = freeze i1 [[OP_RDX]]
+; SSE-NEXT:    [[OP_RDX2:%.*]] = select i1 [[TMP10]], i1 [[OP_RDX1]], i1 false
+; SSE-NEXT:    ret i1 [[OP_RDX2]]
+;
+; AVX-LABEL: @logical_and_icmp_clamp_partial(
+; AVX-NEXT:    [[TMP1:%.*]] = extractelement <4 x i32> [[X:%.*]], i64 2
+; AVX-NEXT:    [[TMP2:%.*]] = shufflevector <4 x i32> [[X]], <4 x i32> poison, <2 x i32> <i32 0, i32 1>
+; AVX-NEXT:    [[TMP3:%.*]] = icmp slt <2 x i32> [[TMP2]], splat (i32 42)
+; AVX-NEXT:    [[C2:%.*]] = icmp slt i32 [[TMP1]], 42
+; AVX-NEXT:    [[TMP4:%.*]] = icmp sgt <4 x i32> [[X]], splat (i32 17)
+; AVX-NEXT:    [[TMP5:%.*]] = freeze <4 x i1> [[TMP4]]
+; AVX-NEXT:    [[TMP6:%.*]] = bitcast <4 x i1> [[TMP5]] to i4
+; AVX-NEXT:    [[TMP7:%.*]] = icmp eq i4 [[TMP6]], -1
+; AVX-NEXT:    [[TMP8:%.*]] = extractelement <2 x i1> [[TMP3]], i64 0
+; AVX-NEXT:    [[OP_RDX:%.*]] = select i1 [[TMP7]], i1 [[TMP8]], i1 false
+; AVX-NEXT:    [[TMP9:%.*]] = extractelement <2 x i1> [[TMP3]], i64 1
+; AVX-NEXT:    [[TMP10:%.*]] = freeze i1 [[TMP9]]
+; AVX-NEXT:    [[OP_RDX1:%.*]] = select i1 [[TMP10]], i1 [[C2]], i1 false
+; AVX-NEXT:    [[TMP11:%.*]] = freeze i1 [[OP_RDX]]
+; AVX-NEXT:    [[OP_RDX2:%.*]] = select i1 [[TMP11]], i1 [[OP_RDX1]], i1 false
+; AVX-NEXT:    ret i1 [[OP_RDX2]]
 ;
   %x0 = extractelement <4 x i32> %x, i32 0
   %x1 = extractelement <4 x i32> %x, i32 1
@@ -390,17 +490,30 @@ define i1 @logical_and_icmp_clamp_partial(<4 x i32> %x) {
 }
 
 define i1 @logical_and_icmp_clamp_pred_diff(<4 x i32> %x) {
-; CHECK-LABEL: @logical_and_icmp_clamp_pred_diff(
-; CHECK-NEXT:    [[TMP1:%.*]] = shufflevector <4 x i32> [[X:%.*]], <4 x i32> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 0, i32 1, i32 2, i32 3>
-; CHECK-NEXT:    [[TMP2:%.*]] = shufflevector <8 x i32> [[TMP1]], <8 x i32> <i32 poison, i32 poison, i32 poison, i32 poison, i32 42, i32 42, i32 42, i32 poison>, <8 x i32> <i32 poison, i32 poison, i32 poison, i32 poison, i32 12, i32 13, i32 14, i32 3>
-; CHECK-NEXT:    [[TMP3:%.*]] = shufflevector <8 x i32> [[TMP2]], <8 x i32> [[TMP1]], <8 x i32> <i32 8, i32 9, i32 10, i32 11, i32 4, i32 5, i32 6, i32 7>
-; CHECK-NEXT:    [[TMP4:%.*]] = shufflevector <8 x i32> [[TMP1]], <8 x i32> <i32 17, i32 17, i32 17, i32 17, i32 poison, i32 poison, i32 poison, i32 42>, <8 x i32> <i32 8, i32 9, i32 10, i32 11, i32 0, i32 1, i32 2, i32 15>
-; CHECK-NEXT:    [[TMP5:%.*]] = icmp sgt <8 x i32> [[TMP3]], [[TMP4]]
-; CHECK-NEXT:    [[TMP6:%.*]] = icmp ult <8 x i32> [[TMP3]], [[TMP4]]
-; CHECK-NEXT:    [[TMP7:%.*]] = shufflevector <8 x i1> [[TMP5]], <8 x i1> [[TMP6]], <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 15>
-; CHECK-NEXT:    [[TMP8:%.*]] = freeze <8 x i1> [[TMP7]]
-; CHECK-NEXT:    [[TMP9:%.*]] = call i1 @llvm.vector.reduce.and.v8i1(<8 x i1> [[TMP8]])
-; CHECK-NEXT:    ret i1 [[TMP9]]
+; SSE-LABEL: @logical_and_icmp_clamp_pred_diff(
+; SSE-NEXT:    [[TMP1:%.*]] = shufflevector <4 x i32> [[X:%.*]], <4 x i32> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 0, i32 1, i32 2, i32 3>
+; SSE-NEXT:    [[TMP2:%.*]] = shufflevector <8 x i32> [[TMP1]], <8 x i32> <i32 poison, i32 poison, i32 poison, i32 poison, i32 42, i32 42, i32 42, i32 poison>, <8 x i32> <i32 poison, i32 poison, i32 poison, i32 poison, i32 12, i32 13, i32 14, i32 3>
+; SSE-NEXT:    [[TMP3:%.*]] = shufflevector <8 x i32> [[TMP2]], <8 x i32> [[TMP1]], <8 x i32> <i32 8, i32 9, i32 10, i32 11, i32 4, i32 5, i32 6, i32 7>
+; SSE-NEXT:    [[TMP4:%.*]] = shufflevector <8 x i32> [[TMP1]], <8 x i32> <i32 17, i32 17, i32 17, i32 17, i32 poison, i32 poison, i32 poison, i32 42>, <8 x i32> <i32 8, i32 9, i32 10, i32 11, i32 0, i32 1, i32 2, i32 15>
+; SSE-NEXT:    [[TMP5:%.*]] = icmp sgt <8 x i32> [[TMP3]], [[TMP4]]
+; SSE-NEXT:    [[TMP6:%.*]] = icmp ult <8 x i32> [[TMP3]], [[TMP4]]
+; SSE-NEXT:    [[TMP7:%.*]] = shufflevector <8 x i1> [[TMP5]], <8 x i1> [[TMP6]], <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 15>
+; SSE-NEXT:    [[TMP8:%.*]] = freeze <8 x i1> [[TMP7]]
+; SSE-NEXT:    [[TMP9:%.*]] = call i1 @llvm.vector.reduce.and.v8i1(<8 x i1> [[TMP8]])
+; SSE-NEXT:    ret i1 [[TMP9]]
+;
+; AVX-LABEL: @logical_and_icmp_clamp_pred_diff(
+; AVX-NEXT:    [[TMP1:%.*]] = shufflevector <4 x i32> [[X:%.*]], <4 x i32> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 0, i32 1, i32 2, i32 3>
+; AVX-NEXT:    [[TMP2:%.*]] = shufflevector <8 x i32> [[TMP1]], <8 x i32> <i32 poison, i32 poison, i32 poison, i32 poison, i32 42, i32 42, i32 42, i32 poison>, <8 x i32> <i32 poison, i32 poison, i32 poison, i32 poison, i32 12, i32 13, i32 14, i32 3>
+; AVX-NEXT:    [[TMP3:%.*]] = shufflevector <8 x i32> [[TMP2]], <8 x i32> [[TMP1]], <8 x i32> <i32 8, i32 9, i32 10, i32 11, i32 4, i32 5, i32 6, i32 7>
+; AVX-NEXT:    [[TMP4:%.*]] = shufflevector <8 x i32> [[TMP1]], <8 x i32> <i32 17, i32 17, i32 17, i32 17, i32 poison, i32 poison, i32 poison, i32 42>, <8 x i32> <i32 8, i32 9, i32 10, i32 11, i32 0, i32 1, i32 2, i32 15>
+; AVX-NEXT:    [[TMP5:%.*]] = icmp sgt <8 x i32> [[TMP3]], [[TMP4]]
+; AVX-NEXT:    [[TMP6:%.*]] = icmp ult <8 x i32> [[TMP3]], [[TMP4]]
+; AVX-NEXT:    [[TMP7:%.*]] = shufflevector <8 x i1> [[TMP5]], <8 x i1> [[TMP6]], <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 15>
+; AVX-NEXT:    [[TMP8:%.*]] = freeze <8 x i1> [[TMP7]]
+; AVX-NEXT:    [[TMP9:%.*]] = bitcast <8 x i1> [[TMP8]] to i8
+; AVX-NEXT:    [[TMP10:%.*]] = icmp eq i8 [[TMP9]], -1
+; AVX-NEXT:    ret i1 [[TMP10]]
 ;
   %x0 = extractelement <4 x i32> %x, i32 0
   %x1 = extractelement <4 x i32> %x, i32 1
@@ -425,12 +538,20 @@ define i1 @logical_and_icmp_clamp_pred_diff(<4 x i32> %x) {
 }
 
 define i1 @logical_and_icmp_extra_op(<4 x i32> %x, <4 x i32> %y, i1 %c) {
-; CHECK-LABEL: @logical_and_icmp_extra_op(
-; CHECK-NEXT:    [[TMP1:%.*]] = icmp slt <4 x i32> [[X:%.*]], [[Y:%.*]]
-; CHECK-NEXT:    [[TMP2:%.*]] = freeze <4 x i1> [[TMP1]]
-; CHECK-NEXT:    [[TMP3:%.*]] = call i1 @llvm.vector.reduce.and.v4i1(<4 x i1> [[TMP2]])
-; CHECK-NEXT:    [[OP_RDX:%.*]] = select i1 [[C:%.*]], i1 [[TMP3]], i1 false
-; CHECK-NEXT:    ret i1 [[OP_RDX]]
+; SSE-LABEL: @logical_and_icmp_extra_op(
+; SSE-NEXT:    [[TMP1:%.*]] = icmp slt <4 x i32> [[X:%.*]], [[Y:%.*]]
+; SSE-NEXT:    [[TMP2:%.*]] = freeze <4 x i1> [[TMP1]]
+; SSE-NEXT:    [[TMP3:%.*]] = call i1 @llvm.vector.reduce.and.v4i1(<4 x i1> [[TMP2]])
+; SSE-NEXT:    [[OP_RDX:%.*]] = select i1 [[C:%.*]], i1 [[TMP3]], i1 false
+; SSE-NEXT:    ret i1 [[OP_RDX]]
+;
+; AVX-LABEL: @logical_and_icmp_extra_op(
+; AVX-NEXT:    [[TMP1:%.*]] = icmp slt <4 x i32> [[X:%.*]], [[Y:%.*]]
+; AVX-NEXT:    [[TMP2:%.*]] = freeze <4 x i1> [[TMP1]]
+; AVX-NEXT:    [[TMP3:%.*]] = bitcast <4 x i1> [[TMP2]] to i4
+; AVX-NEXT:    [[TMP4:%.*]] = icmp eq i4 [[TMP3]], -1
+; AVX-NEXT:    [[OP_RDX:%.*]] = select i1 [[C:%.*]], i1 [[TMP4]], i1 false
+; AVX-NEXT:    ret i1 [[OP_RDX]]
 ;
   %x0 = extractelement <4 x i32> %x, i32 0
   %x1 = extractelement <4 x i32> %x, i32 1
@@ -453,12 +574,20 @@ define i1 @logical_and_icmp_extra_op(<4 x i32> %x, <4 x i32> %y, i1 %c) {
 }
 
 define i1 @logical_or_icmp_extra_op(<4 x i32> %x, <4 x i32> %y, i1 %c) {
-; CHECK-LABEL: @logical_or_icmp_extra_op(
-; CHECK-NEXT:    [[TMP1:%.*]] = icmp slt <4 x i32> [[X:%.*]], [[Y:%.*]]
-; CHECK-NEXT:    [[TMP2:%.*]] = freeze <4 x i1> [[TMP1]]
-; CHECK-NEXT:    [[TMP3:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP2]])
-; CHECK-NEXT:    [[OP_RDX:%.*]] = select i1 [[C:%.*]], i1 true, i1 [[TMP3]]
-; CHECK-NEXT:    ret i1 [[OP_RDX]]
+; SSE-LABEL: @logical_or_icmp_extra_op(
+; SSE-NEXT:    [[TMP1:%.*]] = icmp slt <4 x i32> [[X:%.*]], [[Y:%.*]]
+; SSE-NEXT:    [[TMP2:%.*]] = freeze <4 x i1> [[TMP1]]
+; SSE-NEXT:    [[TMP3:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP2]])
+; SSE-NEXT:    [[OP_RDX:%.*]] = select i1 [[C:%.*]], i1 true, i1 [[TMP3]]
+; SSE-NEXT:    ret i1 [[OP_RDX]]
+;
+; AVX-LABEL: @logical_or_icmp_extra_op(
+; AVX-NEXT:    [[TMP1:%.*]] = icmp slt <4 x i32> [[X:%.*]], [[Y:%.*]]
+; AVX-NEXT:    [[TMP2:%.*]] = freeze <4 x i1> [[TMP1]]
+; AVX-NEXT:    [[TMP3:%.*]] = bitcast <4 x i1> [[TMP2]] to i4
+; AVX-NEXT:    [[TMP4:%.*]] = icmp ne i4 [[TMP3]], 0
+; AVX-NEXT:    [[OP_RDX:%.*]] = select i1 [[C:%.*]], i1 true, i1 [[TMP4]]
+; AVX-NEXT:    ret i1 [[OP_RDX]]
 ;
   %x0 = extractelement <4 x i32> %x, i32 0
   %x1 = extractelement <4 x i32> %x, i32 1
@@ -481,16 +610,28 @@ define i1 @logical_or_icmp_extra_op(<4 x i32> %x, <4 x i32> %y, i1 %c) {
 }
 
 define i1 @logical_and_icmp_extra_args(<4 x i32> %x, i1 %c0, i1 %c1, i1 %c2) {
-; CHECK-LABEL: @logical_and_icmp_extra_args(
-; CHECK-NEXT:    [[TMP1:%.*]] = icmp sgt <4 x i32> [[X:%.*]], splat (i32 17)
-; CHECK-NEXT:    [[TMP2:%.*]] = freeze <4 x i1> [[TMP1]]
-; CHECK-NEXT:    [[TMP3:%.*]] = call i1 @llvm.vector.reduce.and.v4i1(<4 x i1> [[TMP2]])
-; CHECK-NEXT:    [[OP_RDX:%.*]] = select i1 [[TMP3]], i1 [[C0:%.*]], i1 false
-; CHECK-NEXT:    [[TMP4:%.*]] = freeze i1 [[C1:%.*]]
-; CHECK-NEXT:    [[OP_RDX1:%.*]] = select i1 [[TMP4]], i1 [[C2:%.*]], i1 false
-; CHECK-NEXT:    [[TMP5:%.*]] = freeze i1 [[OP_RDX]]
-; CHECK-NEXT:    [[OP_RDX2:%.*]] = select i1 [[TMP5]], i1 [[OP_RDX1]], i1 false
-; CHECK-NEXT:    ret i1 [[OP_RDX2]]
+; SSE-LABEL: @logical_and_icmp_extra_args(
+; SSE-NEXT:    [[TMP1:%.*]] = icmp sgt <4 x i32> [[X:%.*]], splat (i32 17)
+; SSE-NEXT:    [[TMP2:%.*]] = freeze <4 x i1> [[TMP1]]
+; SSE-NEXT:    [[TMP3:%.*]] = call i1 @llvm.vector.reduce.and.v4i1(<4 x i1> [[TMP2]])
+; SSE-NEXT:    [[OP_RDX:%.*]] = select i1 [[TMP3]], i1 [[C0:%.*]], i1 false
+; SSE-NEXT:    [[TMP4:%.*]] = freeze i1 [[C1:%.*]]
+; SSE-NEXT:    [[OP_RDX1:%.*]] = select i1 [[TMP4]], i1 [[C2:%.*]], i1 false
+; SSE-NEXT:    [[TMP5:%.*]] = freeze i1 [[OP_RDX]]
+; SSE-NEXT:    [[OP_RDX2:%.*]] = select i1 [[TMP5]], i1 [[OP_RDX1]], i1 false
+; SSE-NEXT:    ret i1 [[OP_RDX2]]
+;
+; AVX-LABEL: @logical_and_icmp_extra_args(
+; AVX-NEXT:    [[TMP1:%.*]] = icmp sgt <4 x i32> [[X:%.*]], splat (i32 17)
+; AVX-NEXT:    [[TMP2:%.*]] = freeze <4 x i1> [[TMP1]]
+; AVX-NEXT:    [[TMP3:%.*]] = bitcast <4 x i1> [[TMP2]] to i4
+; AVX-NEXT:    [[TMP4:%.*]] = icmp eq i4 [[TMP3]], -1
+; AVX-NEXT:    [[OP_RDX:%.*]] = select i1 [[TMP4]], i1 [[C0:%.*]], i1 false
+; AVX-NEXT:    [[TMP5:%.*]] = freeze i1 [[C1:%.*]]
+; AVX-NEXT:    [[OP_RDX1:%.*]] = select i1 [[TMP5]], i1 [[C2:%.*]], i1 false
+; AVX-NEXT:    [[TMP6:%.*]] = freeze i1 [[OP_RDX]]
+; AVX-NEXT:    [[OP_RDX2:%.*]] = select i1 [[TMP6]], i1 [[OP_RDX1]], i1 false
+; AVX-NEXT:    ret i1 [[OP_RDX2]]
 ;
   %x0 = extractelement <4 x i32> %x, i32 0
   %x1 = extractelement <4 x i32> %x, i32 1
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/reduction2.ll b/llvm/test/Transforms/SLPVectorizer/X86/reduction2.ll
index 03cdc54eb779f..28c09dececd5b 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/reduction2.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/reduction2.ll
@@ -94,17 +94,12 @@ define i1 @fcmp_lt_gt(double %a, double %b, double %c) {
 ; CHECK-NEXT:    [[TMP2:%.*]] = insertelement <2 x double> poison, double [[MUL]], i64 0
 ; CHECK-NEXT:    [[TMP3:%.*]] = shufflevector <2 x double> [[TMP2]], <2 x double> poison, <2 x i32> zeroinitializer
 ; CHECK-NEXT:    [[TMP4:%.*]] = fdiv <2 x double> [[TMP1]], [[TMP3]]
-; CHECK-NEXT:    [[TMP8:%.*]] = extractelement <2 x double> [[TMP4]], i64 1
-; CHECK-NEXT:    [[CMP:%.*]] = fcmp olt double [[TMP8]], f0x3EB0C6F7A0B5ED8D
-; CHECK-NEXT:    [[TMP9:%.*]] = extractelement <2 x double> [[TMP4]], i64 0
-; CHECK-NEXT:    [[CMP4:%.*]] = fcmp olt double [[TMP9]], f0x3EB0C6F7A0B5ED8D
-; CHECK-NEXT:    [[OR_COND:%.*]] = and i1 [[CMP]], [[CMP4]]
+; CHECK-NEXT:    [[TMP5:%.*]] = fcmp olt <2 x double> [[TMP4]], splat (double f0x3EB0C6F7A0B5ED8D)
+; CHECK-NEXT:    [[OR_COND:%.*]] = call i1 @llvm.vector.reduce.and.v2i1(<2 x i1> [[TMP5]])
 ; CHECK-NEXT:    br i1 [[OR_COND]], label [[CLEANUP:%.*]], label [[LOR_LHS_FALSE:%.*]]
 ; CHECK:       lor.lhs.false:
 ; CHECK-NEXT:    [[TMP7:%.*]] = fcmp ule <2 x double> [[TMP4]], splat (double 1.000000e+00)
-; CHECK-NEXT:    [[TMP11:%.*]] = extractelement <2 x i1> [[TMP7]], i64 0
-; CHECK-NEXT:    [[TMP12:%.*]] = extractelement <2 x i1> [[TMP7]], i64 1
-; CHECK-NEXT:    [[NOT_OR_COND9:%.*]] = or i1 [[TMP11]], [[TMP12]]
+; CHECK-NEXT:    [[NOT_OR_COND9:%.*]] = call i1 @llvm.vector.reduce.or.v2i1(<2 x i1> [[TMP7]])
 ; CHECK-NEXT:    ret i1 [[NOT_OR_COND9]]
 ; CHECK:       cleanup:
 ; CHECK-NEXT:    ret i1 false
@@ -143,9 +138,7 @@ define i1 @fcmp_lt(double %a, double %b, double %c) {
 ; CHECK-NEXT:    [[TMP4:%.*]] = shufflevector <2 x double> [[TMP3]], <2 x double> poison, <2 x i32> zeroinitializer
 ; CHECK-NEXT:    [[TMP5:%.*]] = fdiv <2 x double> [[TMP2]], [[TMP4]]
 ; CHECK-NEXT:    [[TMP6:%.*]] = fcmp uge <2 x double> [[TMP5]], splat (double f0x3EB0C6F7A0B5ED8D)
-; CHECK-NEXT:    [[TMP10:%.*]] = extractelement <2 x i1> [[TMP6]], i64 0
-; CHECK-NEXT:    [[TMP11:%.*]] = extractelement <2 x i1> [[TMP6]], i64 1
-; CHECK-NEXT:    [[NOT_OR_COND:%.*]] = or i1 [[TMP10]], [[TMP11]]
+; CHECK-NEXT:    [[NOT_OR_COND:%.*]] = call i1 @llvm.vector.reduce.or.v2i1(<2 x i1> [[TMP6]])
 ; CHECK-NEXT:    ret i1 [[NOT_OR_COND]]
 ;
   %fneg = fneg double %b
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/reorder-vf-to-resize.ll b/llvm/test/Transforms/SLPVectorizer/X86/reorder-vf-to-resize.ll
index 3f61fa3d44bc7..180284942db9c 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/reorder-vf-to-resize.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/reorder-vf-to-resize.ll
@@ -11,7 +11,8 @@ define void @main(ptr %0) {
 ; CHECK-NEXT:    [[TMP6:%.*]] = fmul <4 x double> [[TMP5]], zeroinitializer
 ; CHECK-NEXT:    [[TMP7:%.*]] = call <4 x double> @llvm.fabs.v4f64(<4 x double> [[TMP6]])
 ; CHECK-NEXT:    [[TMP8:%.*]] = fcmp oeq <4 x double> [[TMP7]], zeroinitializer
-; CHECK-NEXT:    [[TMP9:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP8]])
+; CHECK-NEXT:    [[TMP12:%.*]] = bitcast <4 x i1> [[TMP8]] to i4
+; CHECK-NEXT:    [[TMP9:%.*]] = icmp ne i4 [[TMP12]], 0
 ; CHECK-NEXT:    [[TMP10:%.*]] = select i1 [[TMP9]], double 0.000000e+00, double 0.000000e+00
 ; CHECK-NEXT:    store double [[TMP10]], ptr null, align 8
 ; CHECK-NEXT:    ret void
diff --git a/llvm/test/Transforms/SLPVectorizer/zext-incoming-for-neg-icmp.ll b/llvm/test/Transforms/SLPVectorizer/zext-incoming-for-neg-icmp.ll
index 0e4b38a48d509..5e598d634713f 100644
--- a/llvm/test/Transforms/SLPVectorizer/zext-incoming-for-neg-icmp.ll
+++ b/llvm/test/Transforms/SLPVectorizer/zext-incoming-for-neg-icmp.ll
@@ -1,24 +1,41 @@
 ; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 4
-; RUN: %if x86-registered-target %{ opt -S --passes=slp-vectorizer -mtriple=x86_64-unknown-linux-gnu < %s | FileCheck %s %}
-; RUN: %if aarch64-registered-target %{ opt -S --passes=slp-vectorizer -mtriple=aarch64-unknown-linux-gnu < %s | FileCheck %s %}
+; RUN: %if x86-registered-target %{ opt -S --passes=slp-vectorizer -mtriple=x86_64-unknown-linux-gnu < %s | FileCheck %s --check-prefixes=X86 %}
+; RUN: %if aarch64-registered-target %{ opt -S --passes=slp-vectorizer -mtriple=aarch64-unknown-linux-gnu < %s | FileCheck %s --check-prefixes=AARCH64 %}
 
 define i32 @test(i32 %a, i8 %b, i8 %c) {
-; CHECK-LABEL: define i32 @test(
-; CHECK-SAME: i32 [[A:%.*]], i8 [[B:%.*]], i8 [[C:%.*]]) {
-; CHECK-NEXT:  entry:
-; CHECK-NEXT:    [[TMP0:%.*]] = insertelement <4 x i8> poison, i8 [[C]], i64 0
-; CHECK-NEXT:    [[TMP1:%.*]] = shufflevector <4 x i8> [[TMP0]], <4 x i8> poison, <4 x i32> zeroinitializer
-; CHECK-NEXT:    [[TMP2:%.*]] = add <4 x i8> [[TMP1]], <i8 -1, i8 -2, i8 -3, i8 -4>
-; CHECK-NEXT:    [[TMP3:%.*]] = insertelement <4 x i8> poison, i8 [[B]], i64 0
-; CHECK-NEXT:    [[TMP4:%.*]] = shufflevector <4 x i8> [[TMP3]], <4 x i8> poison, <4 x i32> zeroinitializer
-; CHECK-NEXT:    [[TMP8:%.*]] = zext <4 x i8> [[TMP2]] to <4 x i16>
-; CHECK-NEXT:    [[TMP9:%.*]] = sext <4 x i8> [[TMP4]] to <4 x i16>
-; CHECK-NEXT:    [[TMP5:%.*]] = icmp sle <4 x i16> [[TMP8]], [[TMP9]]
-; CHECK-NEXT:    [[TMP10:%.*]] = bitcast <4 x i1> [[TMP5]] to i4
-; CHECK-NEXT:    [[TMP11:%.*]] = call i4 @llvm.ctpop.i4(i4 [[TMP10]])
-; CHECK-NEXT:    [[TMP7:%.*]] = zext i4 [[TMP11]] to i32
-; CHECK-NEXT:    [[OP_RDX:%.*]] = add i32 [[TMP7]], [[A]]
-; CHECK-NEXT:    ret i32 [[OP_RDX]]
+; X86-LABEL: define i32 @test(
+; X86-SAME: i32 [[A:%.*]], i8 [[B:%.*]], i8 [[C:%.*]]) {
+; X86-NEXT:  entry:
+; X86-NEXT:    [[TMP0:%.*]] = insertelement <4 x i8> poison, i8 [[C]], i64 0
+; X86-NEXT:    [[TMP1:%.*]] = shufflevector <4 x i8> [[TMP0]], <4 x i8> poison, <4 x i32> zeroinitializer
+; X86-NEXT:    [[TMP2:%.*]] = add <4 x i8> [[TMP1]], <i8 -1, i8 -2, i8 -3, i8 -4>
+; X86-NEXT:    [[TMP3:%.*]] = insertelement <4 x i8> poison, i8 [[B]], i64 0
+; X86-NEXT:    [[TMP4:%.*]] = shufflevector <4 x i8> [[TMP3]], <4 x i8> poison, <4 x i32> zeroinitializer
+; X86-NEXT:    [[TMP5:%.*]] = zext <4 x i8> [[TMP2]] to <4 x i16>
+; X86-NEXT:    [[TMP6:%.*]] = sext <4 x i8> [[TMP4]] to <4 x i16>
+; X86-NEXT:    [[TMP7:%.*]] = icmp sle <4 x i16> [[TMP5]], [[TMP6]]
+; X86-NEXT:    [[TMP8:%.*]] = bitcast <4 x i1> [[TMP7]] to i4
+; X86-NEXT:    [[TMP10:%.*]] = call i4 @llvm.ctpop.i4(i4 [[TMP8]])
+; X86-NEXT:    [[TMP9:%.*]] = zext i4 [[TMP10]] to i32
+; X86-NEXT:    [[OP_RDX:%.*]] = add i32 [[TMP9]], [[A]]
+; X86-NEXT:    ret i32 [[OP_RDX]]
+;
+; AARCH64-LABEL: define i32 @test(
+; AARCH64-SAME: i32 [[A:%.*]], i8 [[B:%.*]], i8 [[C:%.*]]) {
+; AARCH64-NEXT:  entry:
+; AARCH64-NEXT:    [[TMP0:%.*]] = insertelement <4 x i8> poison, i8 [[C]], i64 0
+; AARCH64-NEXT:    [[TMP1:%.*]] = shufflevector <4 x i8> [[TMP0]], <4 x i8> poison, <4 x i32> zeroinitializer
+; AARCH64-NEXT:    [[TMP2:%.*]] = add <4 x i8> [[TMP1]], <i8 -1, i8 -2, i8 -3, i8 -4>
+; AARCH64-NEXT:    [[TMP3:%.*]] = insertelement <4 x i8> poison, i8 [[B]], i64 0
+; AARCH64-NEXT:    [[TMP4:%.*]] = shufflevector <4 x i8> [[TMP3]], <4 x i8> poison, <4 x i32> zeroinitializer
+; AARCH64-NEXT:    [[TMP5:%.*]] = zext <4 x i8> [[TMP2]] to <4 x i16>
+; AARCH64-NEXT:    [[TMP6:%.*]] = sext <4 x i8> [[TMP4]] to <4 x i16>
+; AARCH64-NEXT:    [[TMP7:%.*]] = icmp sle <4 x i16> [[TMP5]], [[TMP6]]
+; AARCH64-NEXT:    [[TMP8:%.*]] = bitcast <4 x i1> [[TMP7]] to i4
+; AARCH64-NEXT:    [[TMP9:%.*]] = call i4 @llvm.ctpop.i4(i4 [[TMP8]])
+; AARCH64-NEXT:    [[TMP10:%.*]] = zext i4 [[TMP9]] to i32
+; AARCH64-NEXT:    [[OP_RDX:%.*]] = add i32 [[TMP10]], [[A]]
+; AARCH64-NEXT:    ret i32 [[OP_RDX]]
 ;
 entry:
   %0 = add i8 %c, -3



More information about the llvm-commits mailing list