[llvm] [X86][APX] Optimize usub.sat(X, 1) to cmp+adc with NDD (PR #208475)

via llvm-commits llvm-commits at lists.llvm.org
Fri Jul 10 00:07:30 PDT 2026


https://github.com/AntonyCJ30 updated https://github.com/llvm/llvm-project/pull/208475

>From 2726d7973219552a8a8180348412704e69ef7730 Mon Sep 17 00:00:00 2001
From: AntonyCJ30 <cj6186609 at gmail@gmail.com>
Date: Thu, 9 Jul 2026 19:46:47 +0530
Subject: [PATCH] [X86][APX] Optimize usub.sat(X,1) to cmp+adc with NDD

---
 llvm/lib/Target/X86/X86ISelLowering.cpp | 2823 +++++++++--------------
 llvm/test/CodeGen/X86/apx/sub.ll        |  383 ++-
 2 files changed, 1287 insertions(+), 1919 deletions(-)

diff --git a/llvm/lib/Target/X86/X86ISelLowering.cpp b/llvm/lib/Target/X86/X86ISelLowering.cpp
index e97b4e6d84a9f..91cf7d3abab30 100644
--- a/llvm/lib/Target/X86/X86ISelLowering.cpp
+++ b/llvm/lib/Target/X86/X86ISelLowering.cpp
@@ -321,10 +321,6 @@ X86TargetLowering::X86TargetLowering(const X86TargetMachine &TM,
     }
   }
   if (Subtarget.hasAVX10_2()) {
-    for (MVT VT : {MVT::v8i8, MVT::v16i8, MVT::v32i8}) {
-      setOperationAction(ISD::FP_TO_UINT_SAT, VT, Custom);
-      setOperationAction(ISD::FP_TO_SINT_SAT, VT, Custom);
-    }
     setOperationAction(ISD::FP_TO_UINT_SAT, MVT::v2i32, Custom);
     setOperationAction(ISD::FP_TO_SINT_SAT, MVT::v2i32, Custom);
     setOperationAction(ISD::FP_TO_UINT_SAT, MVT::v8i64, Legal);
@@ -403,34 +399,34 @@ X86TargetLowering::X86TargetLowering(const X86TargetMachine &TM,
 
   // Promote the i8 variants and force them on up to i32 which has a shorter
   // encoding.
-  setOperationPromotedToType(ISD::CTTZ, MVT::i8, MVT::i32);
-  setOperationPromotedToType(ISD::CTTZ_ZERO_POISON, MVT::i8, MVT::i32);
+  setOperationPromotedToType(ISD::CTTZ           , MVT::i8   , MVT::i32);
+  setOperationPromotedToType(ISD::CTTZ_ZERO_UNDEF, MVT::i8   , MVT::i32);
   // Promoted i16. tzcntw has a false dependency on Intel CPUs. For BSF, we emit
   // a REP prefix to encode it as TZCNT for modern CPUs so it makes sense to
   // promote that too.
-  setOperationPromotedToType(ISD::CTTZ, MVT::i16, MVT::i32);
-  setOperationPromotedToType(ISD::CTTZ_ZERO_POISON, MVT::i16, MVT::i32);
+  setOperationPromotedToType(ISD::CTTZ           , MVT::i16  , MVT::i32);
+  setOperationPromotedToType(ISD::CTTZ_ZERO_UNDEF, MVT::i16  , MVT::i32);
 
   if (!Subtarget.hasBMI()) {
-    setOperationAction(ISD::CTTZ, MVT::i32, Custom);
-    setOperationAction(ISD::CTTZ_ZERO_POISON, MVT::i32, Legal);
+    setOperationAction(ISD::CTTZ           , MVT::i32  , Custom);
+    setOperationAction(ISD::CTTZ_ZERO_UNDEF, MVT::i32  , Legal);
     if (Subtarget.is64Bit()) {
-      setOperationAction(ISD::CTTZ, MVT::i64, Custom);
-      setOperationAction(ISD::CTTZ_ZERO_POISON, MVT::i64, Legal);
+      setOperationAction(ISD::CTTZ         , MVT::i64  , Custom);
+      setOperationAction(ISD::CTTZ_ZERO_UNDEF, MVT::i64, Legal);
     }
   }
 
   if (Subtarget.hasLZCNT()) {
     // When promoting the i8 variants, force them to i32 for a shorter
     // encoding.
-    setOperationPromotedToType(ISD::CTLZ, MVT::i8, MVT::i32);
-    setOperationPromotedToType(ISD::CTLZ_ZERO_POISON, MVT::i8, MVT::i32);
+    setOperationPromotedToType(ISD::CTLZ           , MVT::i8   , MVT::i32);
+    setOperationPromotedToType(ISD::CTLZ_ZERO_UNDEF, MVT::i8   , MVT::i32);
   } else {
     for (auto VT : {MVT::i8, MVT::i16, MVT::i32, MVT::i64}) {
       if (VT == MVT::i64 && !Subtarget.is64Bit())
         continue;
-      setOperationAction(ISD::CTLZ, VT, Custom);
-      setOperationAction(ISD::CTLZ_ZERO_POISON, VT, Custom);
+      setOperationAction(ISD::CTLZ           , VT, Custom);
+      setOperationAction(ISD::CTLZ_ZERO_UNDEF, VT, Custom);
     }
   }
 
@@ -480,14 +476,6 @@ X86TargetLowering::X86TargetLowering(const X86TargetMachine &TM,
     setOperationAction(ISD::CTPOP          , MVT::i64  , Custom);
   }
 
-  if (Subtarget.hasBMI2()) {
-    setOperationAction({ISD::PEXT, ISD::PDEP}, MVT::i8, Promote);
-    setOperationAction({ISD::PEXT, ISD::PDEP}, MVT::i16, Promote);
-    setOperationAction({ISD::PEXT, ISD::PDEP}, MVT::i32, Legal);
-    if (Subtarget.is64Bit())
-      setOperationAction({ISD::PEXT, ISD::PDEP}, MVT::i64, Legal);
-  }
-
   setOperationAction(ISD::READCYCLECOUNTER , MVT::i64  , Custom);
 
   if (!Subtarget.hasMOVBE())
@@ -1060,7 +1048,14 @@ X86TargetLowering::X86TargetLowering(const X86TargetMachine &TM,
         setLoadExtAction(ISD::EXTLOAD, InnerVT, VT, Expand);
     }
   }
-
+  if (Subtarget.hasNDD()) {
+    // Enable custom lowering for scalar USUBSAT to optimize usub.sat(X,1)
+    // with cmp+adc when NDD is available.
+    setOperationAction(ISD::USUBSAT, MVT::i8, Custom);
+    setOperationAction(ISD::USUBSAT, MVT::i16, Custom);
+    setOperationAction(ISD::USUBSAT, MVT::i32, Custom);
+    setOperationAction(ISD::USUBSAT, MVT::i64, Custom);
+  }
   // FIXME: In order to prevent SSE instructions being expanded to MMX ones
   // with -msoft-float, disable use of MMX as well.
   if (!Subtarget.useSoftFloat() && Subtarget.hasMMX()) {
@@ -1171,20 +1166,6 @@ X86TargetLowering::X86TargetLowering(const X86TargetMachine &TM,
       setOperationAction(ISD::SMIN, VT, VT == MVT::v8i16 ? Legal : Custom);
       setOperationAction(ISD::UMAX, VT, VT == MVT::v16i8 ? Legal : Custom);
       setOperationAction(ISD::UMIN, VT, VT == MVT::v16i8 ? Legal : Custom);
-      setOperationAction(ISD::VECREDUCE_AND, VT, Custom);
-      setOperationAction(ISD::VECREDUCE_OR, VT, Custom);
-      setOperationAction(ISD::VECREDUCE_XOR, VT, Custom);
-    }
-
-    // SSE2 can use basic vector unrolling.
-    // SSE41 can use PHMINPOS to perform v16i8/v8i16 minmax reductions.
-    // Fallback to ReplaceNodeResults for vXi64 reductions on 32-bit targets.
-    for (auto VT : {MVT::v16i8, MVT::v8i16, MVT::v4i32, MVT::v2i64, MVT::i64}) {
-      setOperationAction(ISD::VECREDUCE_MUL,  VT, Custom);
-      setOperationAction(ISD::VECREDUCE_SMAX, VT, Custom);
-      setOperationAction(ISD::VECREDUCE_SMIN, VT, Custom);
-      setOperationAction(ISD::VECREDUCE_UMAX, VT, Custom);
-      setOperationAction(ISD::VECREDUCE_UMIN, VT, Custom);
     }
 
     setOperationAction(ISD::UADDSAT,            MVT::v16i8, Legal);
@@ -1569,14 +1550,6 @@ X86TargetLowering::X86TargetLowering(const X86TargetMachine &TM,
       setOperationAction(ISD::SRA,             VT, Custom);
       setOperationAction(ISD::ABDS,            VT, Custom);
       setOperationAction(ISD::ABDU,            VT, Custom);
-      setOperationAction(ISD::VECREDUCE_AND,   VT, Custom);
-      setOperationAction(ISD::VECREDUCE_OR,    VT, Custom);
-      setOperationAction(ISD::VECREDUCE_XOR,   VT, Custom);
-      setOperationAction(ISD::VECREDUCE_MUL,   VT, Custom);
-      setOperationAction(ISD::VECREDUCE_SMAX,  VT, Custom);
-      setOperationAction(ISD::VECREDUCE_SMIN,  VT, Custom);
-      setOperationAction(ISD::VECREDUCE_UMAX,  VT, Custom);
-      setOperationAction(ISD::VECREDUCE_UMIN,  VT, Custom);
       if (VT == MVT::v4i64) continue;
       setOperationAction(ISD::ROTL,            VT, Custom);
       setOperationAction(ISD::ROTR,            VT, Custom);
@@ -2046,14 +2019,6 @@ X86TargetLowering::X86TargetLowering(const X86TargetMachine &TM,
       setOperationAction(ISD::ABDS,             VT, Custom);
       setOperationAction(ISD::ABDU,             VT, Custom);
       setOperationAction(ISD::BITREVERSE,       VT, Custom);
-      setOperationAction(ISD::VECREDUCE_AND,    VT, Custom);
-      setOperationAction(ISD::VECREDUCE_OR,     VT, Custom);
-      setOperationAction(ISD::VECREDUCE_XOR,    VT, Custom);
-      setOperationAction(ISD::VECREDUCE_MUL,    VT, Custom);
-      setOperationAction(ISD::VECREDUCE_SMAX,   VT, Custom);
-      setOperationAction(ISD::VECREDUCE_SMIN,   VT, Custom);
-      setOperationAction(ISD::VECREDUCE_UMAX,   VT, Custom);
-      setOperationAction(ISD::VECREDUCE_UMIN,   VT, Custom);
 
       // The condition codes aren't legal in SSE/AVX and under AVX512 we use
       // setcc all the way to isel and prefer SETGT in some isel patterns.
@@ -2273,8 +2238,8 @@ X86TargetLowering::X86TargetLowering(const X86TargetMachine &TM,
           continue;
         setOperationAction(ISD::CTLZ, VT, Custom);
         setOperationAction(ISD::CTTZ, VT, Custom);
-        setOperationAction(ISD::CTLZ_ZERO_POISON, VT, Custom);
-        setOperationAction(ISD::CTTZ_ZERO_POISON, VT, Custom);
+        setOperationAction(ISD::CTLZ_ZERO_UNDEF, VT, Custom);
+        setOperationAction(ISD::CTTZ_ZERO_UNDEF, VT, Custom);
       }
       for (auto VT : { MVT::v4i32, MVT::v8i32, MVT::v2i64, MVT::v4i64 }) {
         setOperationAction(ISD::CTLZ,            VT, Legal);
@@ -2358,11 +2323,6 @@ X86TargetLowering::X86TargetLowering(const X86TargetMachine &TM,
       for (auto VT : { MVT::v16i8, MVT::v32i8, MVT::v8i16, MVT::v16i16 })
         setOperationAction(ISD::CTPOP, VT, Legal);
     }
-
-    if (Subtarget.hasBMM()) {
-      for (auto VT : {MVT::v16i8, MVT::v32i8, MVT::v64i8})
-        setOperationAction(ISD::BITREVERSE, VT, Legal);
-    }
   }
 
   if (!Subtarget.useSoftFloat() && Subtarget.hasFP16()) {
@@ -2765,10 +2725,6 @@ X86TargetLowering::X86TargetLowering(const X86TargetMachine &TM,
   setOperationPromotedToType(ISD::ATOMIC_LOAD, MVT::f32, MVT::i32);
   setOperationPromotedToType(ISD::ATOMIC_LOAD, MVT::f64, MVT::i64);
 
-  setOperationPromotedToType(ISD::ATOMIC_STORE, MVT::f16, MVT::i16);
-  setOperationPromotedToType(ISD::ATOMIC_STORE, MVT::f32, MVT::i32);
-  setOperationPromotedToType(ISD::ATOMIC_STORE, MVT::f64, MVT::i64);
-
   // We have target-specific dag combine patterns for the following nodes:
   setTargetDAGCombine({ISD::VECTOR_SHUFFLE,
                        ISD::SCALAR_TO_VECTOR,
@@ -2801,7 +2757,6 @@ X86TargetLowering::X86TargetLowering(const X86TargetMachine &TM,
                        ISD::FMINNUM,
                        ISD::FMAXNUM,
                        ISD::SUB,
-                       ISD::ATOMIC_LOAD,
                        ISD::LOAD,
                        ISD::LRINT,
                        ISD::LLRINT,
@@ -2816,7 +2771,6 @@ X86TargetLowering::X86TargetLowering(const X86TargetMachine &TM,
                        ISD::ANY_EXTEND_VECTOR_INREG,
                        ISD::SIGN_EXTEND_VECTOR_INREG,
                        ISD::ZERO_EXTEND_VECTOR_INREG,
-                       ISD::VECREDUCE_MUL,
                        ISD::SINT_TO_FP,
                        ISD::UINT_TO_FP,
                        ISD::FP_TO_SINT,
@@ -2877,12 +2831,12 @@ bool X86TargetLowering::useLoadStackGuardNode(const Module &M) const {
   return Subtarget.isTargetMachO() && Subtarget.is64Bit();
 }
 
-bool X86TargetLowering::useStackGuardMixFP() const {
-  // Currently only MSVC CRTs mix the frame pointer into the stack guard value.
+bool X86TargetLowering::useStackGuardXorFP() const {
+  // Currently only MSVC CRTs XOR the frame pointer into the stack guard value.
   return Subtarget.getTargetTriple().isOSMSVCRT() && !Subtarget.isTargetMachO();
 }
 
-SDValue X86TargetLowering::emitStackGuardMixFP(SelectionDAG &DAG, SDValue Val,
+SDValue X86TargetLowering::emitStackGuardXorFP(SelectionDAG &DAG, SDValue Val,
                                                const SDLoc &DL) const {
   EVT PtrTy = getPointerTy(DAG.getDataLayout());
   unsigned XorOp = Subtarget.is64Bit() ? X86::XOR64_FP : X86::XOR32_FP;
@@ -2963,7 +2917,7 @@ bool X86::mayFoldIntoStore(SDValue Op) {
       return false;
     User = *User->user_begin();
   }
-  return ISD::isNormalStore(User) || User->getOpcode() == ISD::ATOMIC_STORE;
+  return ISD::isNormalStore(User);
 }
 
 bool X86::mayFoldIntoZeroExtend(SDValue Op) {
@@ -3389,7 +3343,7 @@ void X86TargetLowering::getTgtMemIntrinsic(
     else if (IntrData->Type == TRUNCATE_TO_MEM_VI32)
       ScalarVT = MVT::i32;
 
-    Info.memVT = VT.changeElementType(ScalarVT);
+    Info.memVT = MVT::getVectorVT(ScalarVT, VT.getVectorNumElements());
     Info.align = Align(1);
     Info.flags |= MachineMemOperand::MOStore;
     Infos.push_back(Info);
@@ -3585,9 +3539,8 @@ bool X86TargetLowering::isExtractSubvectorCheap(EVT ResVT, EVT SrcVT,
   // Mask vectors support all subregister combinations and operations that
   // extract half of vector.
   if (ResVT.getVectorElementType() == MVT::i1)
-    return Index == 0 ||
-           ((ResVT.getSizeInBits() * 2 == SrcVT.getSizeInBits()) &&
-            (Index == ResVT.getVectorNumElements()));
+    return Index == 0 || ((ResVT.getSizeInBits() == SrcVT.getSizeInBits()*2) &&
+                          (Index == ResVT.getVectorNumElements()));
 
   return (Index % ResVT.getVectorNumElements()) == 0;
 }
@@ -4987,27 +4940,29 @@ static unsigned getTargetVShiftUniformOpcode(unsigned Opc, bool IsVariable) {
 static SDValue getTargetVShiftByConstNode(unsigned Opc, const SDLoc &dl, MVT VT,
                                           SDValue SrcOp, uint64_t ShiftAmt,
                                           SelectionDAG &DAG) {
-  assert(
-      (Opc == X86ISD::VSHLI || Opc == X86ISD::VSRLI || Opc == X86ISD::VSRAI) &&
-      "Unknown target vector shift-by-constant node");
+  MVT ElementType = VT.getVectorElementType();
 
   // Bitcast the source vector to the output type, this is mainly necessary for
   // vXi8/vXi64 shifts.
-  SrcOp = DAG.getBitcast(VT, SrcOp);
+  if (VT != SrcOp.getSimpleValueType())
+    SrcOp = DAG.getBitcast(VT, SrcOp);
 
   // Fold this packed shift into its first operand if ShiftAmt is 0.
   if (ShiftAmt == 0)
     return SrcOp;
 
   // Check for ShiftAmt >= element width
-  unsigned EltSizeInBits = VT.getScalarSizeInBits();
-  if (ShiftAmt >= EltSizeInBits) {
+  if (ShiftAmt >= ElementType.getSizeInBits()) {
     if (Opc == X86ISD::VSRAI)
-      ShiftAmt = EltSizeInBits - 1;
+      ShiftAmt = ElementType.getSizeInBits() - 1;
     else
       return DAG.getConstant(0, dl, VT);
   }
 
+  assert(
+      (Opc == X86ISD::VSHLI || Opc == X86ISD::VSRLI || Opc == X86ISD::VSRAI) &&
+      "Unknown target vector shift-by-constant node");
+
   // Fold this packed vector shift into a build vector if SrcOp is a
   // vector of Constants or UNDEFs.
   if (ISD::isBuildVectorOfConstantSDNodes(SrcOp.getNode())) {
@@ -5458,13 +5413,11 @@ static bool getTargetConstantBitsFromNode(SDValue Op, unsigned EltSizeInBits,
       return true;
     }
     if (auto *CInt = dyn_cast<ConstantInt>(Cst)) {
-      Mask = APInt::getSplat(CInt->getType()->getPrimitiveSizeInBits(),
-                             CInt->getValue());
+      Mask = CInt->getValue();
       return true;
     }
     if (auto *CFP = dyn_cast<ConstantFP>(Cst)) {
-      Mask = APInt::getSplat(CFP->getType()->getPrimitiveSizeInBits(),
-                             CFP->getValueAPF().bitcastToAPInt());
+      Mask = CFP->getValueAPF().bitcastToAPInt();
       return true;
     }
     if (auto *CDS = dyn_cast<ConstantDataSequential>(Cst)) {
@@ -7050,26 +7003,6 @@ static bool getFauxShuffleMask(SDValue N, const APInt &DemandedElts,
     }
     return true;
   }
-  case X86ISD::VSHLD:
-  case X86ISD::VSHRD: {
-    // We can only decode 'whole byte' bit funnel shifts as shuffles.
-    uint64_t ShiftVal = N.getConstantOperandAPInt(2).urem(NumBitsPerElt);
-    int Offset = ShiftVal / 8;
-    if ((ShiftVal % 8) != 0 || Offset == 0)
-      return false;
-    Ops.push_back(N.getOperand(X86ISD::VSHRD == Opcode ? 1 : 0));
-    Ops.push_back(N.getOperand(X86ISD::VSHRD == Opcode ? 0 : 1));
-    Offset = X86ISD::VSHRD == Opcode ? (NumBytesPerElt - Offset) : Offset;
-    for (int I = 0; I != (int)NumElts; ++I) {
-      int BaseIdx = (I * NumBytesPerElt) - Offset;
-      for (int J = 0; J != (int)NumBytesPerElt; ++J) {
-        int MaskIdx = BaseIdx + J;
-        MaskIdx += J < Offset ? (NumSizeInBytes + NumBytesPerElt) : 0;
-        Mask.push_back(MaskIdx);
-      }
-    }
-    return true;
-  }
   case X86ISD::VBROADCAST: {
     SDValue Src = N.getOperand(0);
     if (!Src.getSimpleValueType().isVector()) {
@@ -7157,7 +7090,7 @@ static void resolveTargetShuffleInputsAndMask(SmallVectorImpl<SDValue> &Inputs,
     // Check for repeated inputs.
     bool IsRepeat = false;
     for (int j = 0, ue = UsedInputs.size(); j != ue; ++j) {
-      if (peekThroughBitcasts(UsedInputs[j]) != peekThroughBitcasts(Inputs[i]))
+      if (UsedInputs[j] != Inputs[i])
         continue;
       for (int &M : Mask)
         if (lo <= M)
@@ -7762,22 +7695,6 @@ static SDValue EltsFromConsecutiveLoads(EVT VT, ArrayRef<SDValue> Elts,
   if ((VT.getScalarSizeInBits() % 8) != 0)
     return SDValue();
 
-  // If all of these are oneuse frozen loads, then attempt to create a frozen
-  // consecutive load.
-  if (all_of(Elts, [](SDValue Elt) {
-        return Elt.getOpcode() == ISD::FREEZE &&
-               ISD::isNormalLoad(Elt.getOperand(0).getNode()) &&
-               Elt.hasOneUse();
-      })) {
-    SmallVector<SDValue, 16> SrcElts;
-    for (SDValue Elt : Elts)
-      SrcElts.push_back(peekThroughFreeze(Elt));
-    if (SDValue LD = EltsFromConsecutiveLoads(VT, SrcElts, DL, DAG, Subtarget,
-                                              IsAfterLegalize, Depth + 1))
-      return DAG.getFreeze(LD);
-    return SDValue();
-  }
-
   unsigned NumElems = Elts.size();
 
   int LastLoadedElt = -1;
@@ -8545,7 +8462,7 @@ static SDValue LowerBUILD_VECTORvXbf16(SDValue Op, SelectionDAG &DAG,
   return DAG.getBitcast(VT, Res);
 }
 
-// Lower BUILD_VECTOR operation for vXi1 types.
+// Lower BUILD_VECTOR operation for v8i1 and v16i1 types.
 static SDValue LowerBUILD_VECTORvXi1(SDValue Op, const SDLoc &dl,
                                      SelectionDAG &DAG,
                                      const X86Subtarget &Subtarget) {
@@ -8557,64 +8474,51 @@ static SDValue LowerBUILD_VECTORvXi1(SDValue Op, const SDLoc &dl,
       ISD::isBuildVectorAllOnes(Op.getNode()))
     return Op;
 
-  uint64_t Undefs = 0;
   uint64_t Immediate = 0;
-  uint64_t NonConstMask = 0;
-  SmallSet<SDValue, 16> NonConstElts;
+  SmallVector<unsigned, 16> NonConstIdx;
+  bool IsSplat = true;
   bool HasConstElts = false;
+  int SplatIdx = -1;
   for (unsigned idx = 0, e = Op.getNumOperands(); idx < e; ++idx) {
     SDValue In = Op.getOperand(idx);
-    if (In.isUndef()) {
-      Undefs |= 1ULL << idx;
+    if (In.isUndef())
       continue;
-    }
     if (auto *InC = dyn_cast<ConstantSDNode>(In)) {
       Immediate |= (InC->getZExtValue() & 0x1) << idx;
       HasConstElts = true;
     } else {
-      NonConstMask |= 1ULL << idx;
-      NonConstElts.insert(In);
+      NonConstIdx.push_back(idx);
     }
+    if (SplatIdx < 0)
+      SplatIdx = idx;
+    else if (In != Op.getOperand(SplatIdx))
+      IsSplat = false;
   }
 
-  // for single non-const use " (select i1 elt, imm | elt_mask, imm)"
-  if (NonConstElts.size() == 1) {
+  // for splat use " (select i1 splat_elt, all-ones, all-zeroes)"
+  if (IsSplat) {
     // The build_vector allows the scalar element to be larger than the vector
     // element type. We need to mask it to use as a condition unless we know
     // the upper bits are zero.
     // FIXME: Use computeKnownBits instead of checking specific opcode?
-    SDValue Cond = *NonConstElts.begin();
+    SDValue Cond = Op.getOperand(SplatIdx);
     assert(Cond.getValueType() == MVT::i8 && "Unexpected VT!");
     if (Cond.getOpcode() != ISD::SETCC)
       Cond = DAG.getNode(ISD::AND, dl, MVT::i8, Cond,
                          DAG.getConstant(1, dl, MVT::i8));
 
-    uint64_t TrueImm = NonConstMask | Immediate;
-    uint64_t FalseImm = Immediate;
-
     // Perform the select in the scalar domain so we can use cmov.
     if (VT == MVT::v64i1 && !Subtarget.is64Bit()) {
-      uint64_t TrueLo = (unsigned)TrueImm;
-      uint64_t TrueHi = TrueImm >> 32;
-      uint64_t FalseLo = (unsigned)FalseImm;
-      uint64_t FalseHi = FalseImm >> 32;
-      SDValue Lo = DAG.getSelect(dl, MVT::i32, Cond,
-                                 DAG.getConstant(TrueLo, dl, MVT::i32),
-                                 DAG.getConstant(FalseLo, dl, MVT::i32));
-      SDValue Hi = DAG.getSelect(dl, MVT::i32, Cond,
-                                 DAG.getConstant(TrueHi, dl, MVT::i32),
-                                 DAG.getConstant(FalseHi, dl, MVT::i32));
-      Lo = DAG.getBitcast(MVT::v32i1, Lo);
-      Hi = DAG.getBitcast(MVT::v32i1, Hi);
-      return DAG.getNode(ISD::CONCAT_VECTORS, dl, MVT::v64i1, Lo, Hi);
+      SDValue Select = DAG.getSelect(dl, MVT::i32, Cond,
+                                     DAG.getAllOnesConstant(dl, MVT::i32),
+                                     DAG.getConstant(0, dl, MVT::i32));
+      Select = DAG.getBitcast(MVT::v32i1, Select);
+      return DAG.getNode(ISD::CONCAT_VECTORS, dl, MVT::v64i1, Select, Select);
     } else {
       MVT ImmVT = MVT::getIntegerVT(std::max((unsigned)VT.getSizeInBits(), 8U));
-      // Adjust extended value to -1 as it will improve folding.
-      if ((TrueImm | Undefs) == (~0ULL >> (64 - VT.getSizeInBits())))
-        TrueImm = ~0ULL >> (64 - ImmVT.getSizeInBits());
-      SDValue Select =
-          DAG.getSelect(dl, ImmVT, Cond, DAG.getConstant(TrueImm, dl, ImmVT),
-                        DAG.getConstant(FalseImm, dl, ImmVT));
+      SDValue Select = DAG.getSelect(dl, ImmVT, Cond,
+                                     DAG.getAllOnesConstant(dl, ImmVT),
+                                     DAG.getConstant(0, dl, ImmVT));
       MVT VecVT = VT.getSizeInBits() >= 8 ? VT : MVT::v8i1;
       Select = DAG.getBitcast(VecVT, Select);
       return DAG.getNode(ISD::EXTRACT_SUBVECTOR, dl, VT, Select,
@@ -8622,32 +8526,6 @@ static SDValue LowerBUILD_VECTORvXi1(SDValue Op, const SDLoc &dl,
     }
   }
 
-  // See if we can cheaply generate a vXi8 vector and convert to vXi1.
-  MVT OpVT = Op.getOperand(0).getSimpleValueType();
-  if (OpVT == MVT::i8 && NonConstMask != 0) {
-    // On pre-BWI targets, we must extend to vXi32 instead.
-    MVT ByteVT = VT.changeVectorElementType(MVT::i8);
-    MVT WideSVT = Subtarget.hasBWI() ? MVT::i8 : MVT::i32;
-    if (ByteVT.getSizeInBits() < 128) {
-      WideSVT = ByteVT == MVT::v4i8 ? MVT::i32 : MVT::i64;
-      ByteVT = MVT::v16i8;
-    }
-    MVT WideVT = VT.changeVectorElementType(WideSVT);
-    if (DAG.getTargetLoweringInfo().isTypeLegal(ByteVT) &&
-        DAG.getTargetLoweringInfo().isTypeLegal(WideVT)) {
-      SmallVector<SDValue, 16> Elts(Op->op_values());
-      Elts.append(ByteVT.getVectorNumElements() - Elts.size(),
-                  DAG.getPOISON(OpVT));
-      SDValue ByteBV = DAG.getBuildVector(ByteVT, dl, Elts);
-      SDValue WideBV =
-          getEXTEND_VECTOR_INREG(ISD::ANY_EXTEND, dl, WideVT, ByteBV, DAG);
-      WideBV = DAG.getNode(ISD::AND, dl, WideVT, WideBV,
-                           DAG.getConstant(1, dl, WideVT));
-      return DAG.getSetCC(dl, VT, WideBV, DAG.getConstant(0, dl, WideVT),
-                          ISD::SETNE);
-    }
-  }
-
   // insert elements one by one
   SDValue DstVec;
   if (HasConstElts) {
@@ -8668,10 +8546,11 @@ static SDValue LowerBUILD_VECTORvXi1(SDValue Op, const SDLoc &dl,
   } else
     DstVec = DAG.getUNDEF(VT);
 
-  for (unsigned Idx = 0, E = Op.getNumOperands(); Idx != E; ++Idx)
-    if (NonConstMask & (1ULL << Idx))
-      DstVec = DAG.getInsertVectorElt(dl, DstVec, Op.getOperand(Idx), Idx);
-
+  for (unsigned InsertIdx : NonConstIdx) {
+    DstVec = DAG.getNode(ISD::INSERT_VECTOR_ELT, dl, VT, DstVec,
+                         Op.getOperand(InsertIdx),
+                         DAG.getVectorIdxConstant(InsertIdx, dl));
+  }
   return DstVec;
 }
 
@@ -8690,6 +8569,176 @@ static SDValue LowerBUILD_VECTORvXi1(SDValue Op, const SDLoc &dl,
   return false;
 }
 
+/// This is a helper function of LowerToHorizontalOp().
+/// This function checks that the build_vector \p N in input implements a
+/// 128-bit partial horizontal operation on a 256-bit vector, but that operation
+/// may not match the layout of an x86 256-bit horizontal instruction.
+/// In other words, if this returns true, then some extraction/insertion will
+/// be required to produce a valid horizontal instruction.
+///
+/// Parameter \p Opcode defines the kind of horizontal operation to match.
+/// For example, if \p Opcode is equal to ISD::ADD, then this function
+/// checks if \p N implements a horizontal arithmetic add; if instead \p Opcode
+/// is equal to ISD::SUB, then this function checks if this is a horizontal
+/// arithmetic sub.
+///
+/// This function only analyzes elements of \p N whose indices are
+/// in range [BaseIdx, LastIdx).
+///
+/// TODO: This function was originally used to match both real and fake partial
+/// horizontal operations, but the index-matching logic is incorrect for that.
+/// See the corrected implementation in isHopBuildVector(). Can we reduce this
+/// code because it is only used for partial h-op matching now?
+static bool isHorizontalBinOpPart(const BuildVectorSDNode *N, unsigned Opcode,
+                                  const SDLoc &DL, SelectionDAG &DAG,
+                                  unsigned BaseIdx, unsigned LastIdx,
+                                  SDValue &V0, SDValue &V1) {
+  EVT VT = N->getValueType(0);
+  assert(VT.is256BitVector() && "Only use for matching partial 256-bit h-ops");
+  assert(BaseIdx * 2 <= LastIdx && "Invalid Indices in input!");
+  assert(VT.isVector() && VT.getVectorNumElements() >= LastIdx &&
+         "Invalid Vector in input!");
+
+  bool IsCommutable = (Opcode == ISD::ADD || Opcode == ISD::FADD);
+  bool CanFold = true;
+  unsigned ExpectedVExtractIdx = BaseIdx;
+  unsigned NumElts = LastIdx - BaseIdx;
+  V0 = DAG.getUNDEF(VT);
+  V1 = DAG.getUNDEF(VT);
+
+  // Check if N implements a horizontal binop.
+  for (unsigned i = 0, e = NumElts; i != e && CanFold; ++i) {
+    SDValue Op = N->getOperand(i + BaseIdx);
+
+    // Skip UNDEFs.
+    if (Op->isUndef()) {
+      // Update the expected vector extract index.
+      if (i * 2 == NumElts)
+        ExpectedVExtractIdx = BaseIdx;
+      ExpectedVExtractIdx += 2;
+      continue;
+    }
+
+    CanFold = Op->getOpcode() == Opcode && Op->hasOneUse();
+
+    if (!CanFold)
+      break;
+
+    SDValue Op0 = Op.getOperand(0);
+    SDValue Op1 = Op.getOperand(1);
+
+    // Try to match the following pattern:
+    // (BINOP (extract_vector_elt A, I), (extract_vector_elt A, I+1))
+    CanFold = (Op0.getOpcode() == ISD::EXTRACT_VECTOR_ELT &&
+        Op1.getOpcode() == ISD::EXTRACT_VECTOR_ELT &&
+        Op0.getOperand(0) == Op1.getOperand(0) &&
+        isa<ConstantSDNode>(Op0.getOperand(1)) &&
+        isa<ConstantSDNode>(Op1.getOperand(1)));
+    if (!CanFold)
+      break;
+
+    unsigned I0 = Op0.getConstantOperandVal(1);
+    unsigned I1 = Op1.getConstantOperandVal(1);
+
+    if (i * 2 < NumElts) {
+      if (V0.isUndef()) {
+        V0 = Op0.getOperand(0);
+        if (V0.getValueType() != VT)
+          return false;
+      }
+    } else {
+      if (V1.isUndef()) {
+        V1 = Op0.getOperand(0);
+        if (V1.getValueType() != VT)
+          return false;
+      }
+      if (i * 2 == NumElts)
+        ExpectedVExtractIdx = BaseIdx;
+    }
+
+    SDValue Expected = (i * 2 < NumElts) ? V0 : V1;
+    if (I0 == ExpectedVExtractIdx)
+      CanFold = I1 == I0 + 1 && Op0.getOperand(0) == Expected;
+    else if (IsCommutable && I1 == ExpectedVExtractIdx) {
+      // Try to match the following dag sequence:
+      // (BINOP (extract_vector_elt A, I+1), (extract_vector_elt A, I))
+      CanFold = I0 == I1 + 1 && Op1.getOperand(0) == Expected;
+    } else
+      CanFold = false;
+
+    ExpectedVExtractIdx += 2;
+  }
+
+  return CanFold;
+}
+
+/// Emit a sequence of two 128-bit horizontal add/sub followed by
+/// a concat_vector.
+///
+/// This is a helper function of LowerToHorizontalOp().
+/// This function expects two 256-bit vectors called V0 and V1.
+/// At first, each vector is split into two separate 128-bit vectors.
+/// Then, the resulting 128-bit vectors are used to implement two
+/// horizontal binary operations.
+///
+/// The kind of horizontal binary operation is defined by \p X86Opcode.
+///
+/// \p Mode specifies how the 128-bit parts of V0 and V1 are passed in input to
+/// the two new horizontal binop.
+/// When Mode is set, the first horizontal binop dag node would take as input
+/// the lower 128-bit of V0 and the upper 128-bit of V0. The second
+/// horizontal binop dag node would take as input the lower 128-bit of V1
+/// and the upper 128-bit of V1.
+///   Example:
+///     HADD V0_LO, V0_HI
+///     HADD V1_LO, V1_HI
+///
+/// Otherwise, the first horizontal binop dag node takes as input the lower
+/// 128-bit of V0 and the lower 128-bit of V1, and the second horizontal binop
+/// dag node takes the upper 128-bit of V0 and the upper 128-bit of V1.
+///   Example:
+///     HADD V0_LO, V1_LO
+///     HADD V0_HI, V1_HI
+///
+/// If \p isUndefLO is set, then the algorithm propagates UNDEF to the lower
+/// 128-bits of the result. If \p isUndefHI is set, then UNDEF is propagated to
+/// the upper 128-bits of the result.
+static SDValue ExpandHorizontalBinOp(const SDValue &V0, const SDValue &V1,
+                                     const SDLoc &DL, SelectionDAG &DAG,
+                                     unsigned X86Opcode, bool Mode,
+                                     bool isUndefLO, bool isUndefHI) {
+  MVT VT = V0.getSimpleValueType();
+  assert(VT.is256BitVector() && VT == V1.getSimpleValueType() &&
+         "Invalid nodes in input!");
+
+  unsigned NumElts = VT.getVectorNumElements();
+  SDValue V0_LO = extract128BitVector(V0, 0, DAG, DL);
+  SDValue V0_HI = extract128BitVector(V0, NumElts/2, DAG, DL);
+  SDValue V1_LO = extract128BitVector(V1, 0, DAG, DL);
+  SDValue V1_HI = extract128BitVector(V1, NumElts/2, DAG, DL);
+  MVT NewVT = V0_LO.getSimpleValueType();
+
+  SDValue LO = DAG.getUNDEF(NewVT);
+  SDValue HI = DAG.getUNDEF(NewVT);
+
+  if (Mode) {
+    // Don't emit a horizontal binop if the result is expected to be UNDEF.
+    if (!isUndefLO && !V0->isUndef())
+      LO = DAG.getNode(X86Opcode, DL, NewVT, V0_LO, V0_HI);
+    if (!isUndefHI && !V1->isUndef())
+      HI = DAG.getNode(X86Opcode, DL, NewVT, V1_LO, V1_HI);
+  } else {
+    // Don't emit a horizontal binop if the result is expected to be UNDEF.
+    if (!isUndefLO && (!V0_LO->isUndef() || !V1_LO->isUndef()))
+      LO = DAG.getNode(X86Opcode, DL, NewVT, V0_LO, V1_LO);
+
+    if (!isUndefHI && (!V0_HI->isUndef() || !V1_HI->isUndef()))
+      HI = DAG.getNode(X86Opcode, DL, NewVT, V0_HI, V1_HI);
+  }
+
+  return DAG.getNode(ISD::CONCAT_VECTORS, DL, VT, LO, HI);
+}
+
 /// Returns true iff \p BV builds a vector with the result equivalent to
 /// the result of ADDSUB/SUBADD operation.
 /// If true is returned then the operands of ADDSUB = Opnd0 +- Opnd1
@@ -8883,6 +8932,248 @@ static SDValue lowerToAddSubOrFMAddSub(const BuildVectorSDNode *BV,
   return DAG.getNode(X86ISD::ADDSUB, DL, VT, Opnd0, Opnd1);
 }
 
+static bool isHopBuildVector(const BuildVectorSDNode *BV, SelectionDAG &DAG,
+                             unsigned &HOpcode, SDValue &V0, SDValue &V1) {
+  // Initialize outputs to known values.
+  MVT VT = BV->getSimpleValueType(0);
+  HOpcode = ISD::DELETED_NODE;
+  V0 = DAG.getUNDEF(VT);
+  V1 = DAG.getUNDEF(VT);
+
+  // x86 256-bit horizontal ops are defined in a non-obvious way. Each 128-bit
+  // half of the result is calculated independently from the 128-bit halves of
+  // the inputs, so that makes the index-checking logic below more complicated.
+  unsigned NumElts = VT.getVectorNumElements();
+  unsigned GenericOpcode = ISD::DELETED_NODE;
+  unsigned Num128BitChunks = VT.is256BitVector() ? 2 : 1;
+  unsigned NumEltsIn128Bits = NumElts / Num128BitChunks;
+  unsigned NumEltsIn64Bits = NumEltsIn128Bits / 2;
+  for (unsigned i = 0; i != Num128BitChunks; ++i) {
+    for (unsigned j = 0; j != NumEltsIn128Bits; ++j) {
+      // Ignore undef elements.
+      SDValue Op = BV->getOperand(i * NumEltsIn128Bits + j);
+      if (Op.isUndef())
+        continue;
+
+      // If there's an opcode mismatch, we're done.
+      if (HOpcode != ISD::DELETED_NODE && Op.getOpcode() != GenericOpcode)
+        return false;
+
+      // Initialize horizontal opcode.
+      if (HOpcode == ISD::DELETED_NODE) {
+        GenericOpcode = Op.getOpcode();
+        switch (GenericOpcode) {
+        // clang-format off
+        case ISD::ADD: HOpcode = X86ISD::HADD; break;
+        case ISD::SUB: HOpcode = X86ISD::HSUB; break;
+        case ISD::FADD: HOpcode = X86ISD::FHADD; break;
+        case ISD::FSUB: HOpcode = X86ISD::FHSUB; break;
+        default: return false;
+        // clang-format on
+        }
+      }
+
+      SDValue Op0 = Op.getOperand(0);
+      SDValue Op1 = Op.getOperand(1);
+      if (Op0.getOpcode() != ISD::EXTRACT_VECTOR_ELT ||
+          Op1.getOpcode() != ISD::EXTRACT_VECTOR_ELT ||
+          Op0.getOperand(0) != Op1.getOperand(0) ||
+          !isa<ConstantSDNode>(Op0.getOperand(1)) ||
+          !isa<ConstantSDNode>(Op1.getOperand(1)) || !Op.hasOneUse())
+        return false;
+
+      // The source vector is chosen based on which 64-bit half of the
+      // destination vector is being calculated.
+      if (j < NumEltsIn64Bits) {
+        if (V0.isUndef())
+          V0 = Op0.getOperand(0);
+      } else {
+        if (V1.isUndef())
+          V1 = Op0.getOperand(0);
+      }
+
+      SDValue SourceVec = (j < NumEltsIn64Bits) ? V0 : V1;
+      if (SourceVec != Op0.getOperand(0))
+        return false;
+
+      // op (extract_vector_elt A, I), (extract_vector_elt A, I+1)
+      unsigned ExtIndex0 = Op0.getConstantOperandVal(1);
+      unsigned ExtIndex1 = Op1.getConstantOperandVal(1);
+      unsigned ExpectedIndex = i * NumEltsIn128Bits +
+                               (j % NumEltsIn64Bits) * 2;
+      if (ExpectedIndex == ExtIndex0 && ExtIndex1 == ExtIndex0 + 1)
+        continue;
+
+      // If this is not a commutative op, this does not match.
+      if (GenericOpcode != ISD::ADD && GenericOpcode != ISD::FADD)
+        return false;
+
+      // Addition is commutative, so try swapping the extract indexes.
+      // op (extract_vector_elt A, I+1), (extract_vector_elt A, I)
+      if (ExpectedIndex == ExtIndex1 && ExtIndex0 == ExtIndex1 + 1)
+        continue;
+
+      // Extract indexes do not match horizontal requirement.
+      return false;
+    }
+  }
+  // We matched. Opcode and operands are returned by reference as arguments.
+  return true;
+}
+
+static SDValue getHopForBuildVector(const BuildVectorSDNode *BV,
+                                    const SDLoc &DL, SelectionDAG &DAG,
+                                    unsigned HOpcode, SDValue V0, SDValue V1) {
+  // If either input vector is not the same size as the build vector,
+  // extract/insert the low bits to the correct size.
+  // This is free (examples: zmm --> xmm, xmm --> ymm).
+  MVT VT = BV->getSimpleValueType(0);
+  unsigned Width = VT.getSizeInBits();
+  if (V0.getValueSizeInBits() > Width)
+    V0 = extractSubVector(V0, 0, DAG, DL, Width);
+  else if (V0.getValueSizeInBits() < Width)
+    V0 = insertSubVector(DAG.getUNDEF(VT), V0, 0, DAG, DL, Width);
+
+  if (V1.getValueSizeInBits() > Width)
+    V1 = extractSubVector(V1, 0, DAG, DL, Width);
+  else if (V1.getValueSizeInBits() < Width)
+    V1 = insertSubVector(DAG.getUNDEF(VT), V1, 0, DAG, DL, Width);
+
+  unsigned NumElts = VT.getVectorNumElements();
+  APInt DemandedElts = APInt::getAllOnes(NumElts);
+  for (unsigned i = 0; i != NumElts; ++i)
+    if (BV->getOperand(i).isUndef())
+      DemandedElts.clearBit(i);
+
+  // If we don't need the upper xmm, then perform as a xmm hop.
+  unsigned HalfNumElts = NumElts / 2;
+  if (VT.is256BitVector() && DemandedElts.lshr(HalfNumElts) == 0) {
+    MVT HalfVT = VT.getHalfNumVectorElementsVT();
+    V0 = extractSubVector(V0, 0, DAG, DL, 128);
+    V1 = extractSubVector(V1, 0, DAG, DL, 128);
+    SDValue Half = DAG.getNode(HOpcode, DL, HalfVT, V0, V1);
+    return insertSubVector(DAG.getUNDEF(VT), Half, 0, DAG, DL, 256);
+  }
+
+  return DAG.getNode(HOpcode, DL, VT, V0, V1);
+}
+
+/// Lower BUILD_VECTOR to a horizontal add/sub operation if possible.
+static SDValue LowerToHorizontalOp(const BuildVectorSDNode *BV, const SDLoc &DL,
+                                   const X86Subtarget &Subtarget,
+                                   SelectionDAG &DAG) {
+  // We need at least 2 non-undef elements to make this worthwhile by default.
+  unsigned NumNonUndefs =
+      count_if(BV->op_values(), [](SDValue V) { return !V.isUndef(); });
+  if (NumNonUndefs < 2)
+    return SDValue();
+
+  // There are 4 sets of horizontal math operations distinguished by type:
+  // int/FP at 128-bit/256-bit. Each type was introduced with a different
+  // subtarget feature. Try to match those "native" patterns first.
+  MVT VT = BV->getSimpleValueType(0);
+  if (((VT == MVT::v4f32 || VT == MVT::v2f64) && Subtarget.hasSSE3()) ||
+      ((VT == MVT::v8i16 || VT == MVT::v4i32) && Subtarget.hasSSSE3()) ||
+      ((VT == MVT::v8f32 || VT == MVT::v4f64) && Subtarget.hasAVX()) ||
+      ((VT == MVT::v16i16 || VT == MVT::v8i32) && Subtarget.hasAVX2())) {
+    unsigned HOpcode;
+    SDValue V0, V1;
+    if (isHopBuildVector(BV, DAG, HOpcode, V0, V1))
+      return getHopForBuildVector(BV, DL, DAG, HOpcode, V0, V1);
+  }
+
+  // Try harder to match 256-bit ops by using extract/concat.
+  if (!Subtarget.hasAVX() || !VT.is256BitVector())
+    return SDValue();
+
+  // Count the number of UNDEF operands in the build_vector in input.
+  unsigned NumElts = VT.getVectorNumElements();
+  unsigned Half = NumElts / 2;
+  unsigned NumUndefsLO = 0;
+  unsigned NumUndefsHI = 0;
+  for (unsigned i = 0, e = Half; i != e; ++i)
+    if (BV->getOperand(i)->isUndef())
+      NumUndefsLO++;
+
+  for (unsigned i = Half, e = NumElts; i != e; ++i)
+    if (BV->getOperand(i)->isUndef())
+      NumUndefsHI++;
+
+  SDValue InVec0, InVec1;
+  if (VT == MVT::v8i32 || VT == MVT::v16i16) {
+    SDValue InVec2, InVec3;
+    unsigned X86Opcode;
+    bool CanFold = true;
+
+    if (isHorizontalBinOpPart(BV, ISD::ADD, DL, DAG, 0, Half, InVec0, InVec1) &&
+        isHorizontalBinOpPart(BV, ISD::ADD, DL, DAG, Half, NumElts, InVec2,
+                              InVec3) &&
+        ((InVec0.isUndef() || InVec2.isUndef()) || InVec0 == InVec2) &&
+        ((InVec1.isUndef() || InVec3.isUndef()) || InVec1 == InVec3))
+      X86Opcode = X86ISD::HADD;
+    else if (isHorizontalBinOpPart(BV, ISD::SUB, DL, DAG, 0, Half, InVec0,
+                                   InVec1) &&
+             isHorizontalBinOpPart(BV, ISD::SUB, DL, DAG, Half, NumElts, InVec2,
+                                   InVec3) &&
+             ((InVec0.isUndef() || InVec2.isUndef()) || InVec0 == InVec2) &&
+             ((InVec1.isUndef() || InVec3.isUndef()) || InVec1 == InVec3))
+      X86Opcode = X86ISD::HSUB;
+    else
+      CanFold = false;
+
+    if (CanFold) {
+      // Do not try to expand this build_vector into a pair of horizontal
+      // add/sub if we can emit a pair of scalar add/sub.
+      if (NumUndefsLO + 1 == Half || NumUndefsHI + 1 == Half)
+        return SDValue();
+
+      // Convert this build_vector into a pair of horizontal binops followed by
+      // a concat vector. We must adjust the outputs from the partial horizontal
+      // matching calls above to account for undefined vector halves.
+      SDValue V0 = InVec0.isUndef() ? InVec2 : InVec0;
+      SDValue V1 = InVec1.isUndef() ? InVec3 : InVec1;
+      assert((!V0.isUndef() || !V1.isUndef()) && "Horizontal-op of undefs?");
+      bool isUndefLO = NumUndefsLO == Half;
+      bool isUndefHI = NumUndefsHI == Half;
+      return ExpandHorizontalBinOp(V0, V1, DL, DAG, X86Opcode, false, isUndefLO,
+                                   isUndefHI);
+    }
+  }
+
+  if (VT == MVT::v8f32 || VT == MVT::v4f64 || VT == MVT::v8i32 ||
+      VT == MVT::v16i16) {
+    unsigned X86Opcode;
+    if (isHorizontalBinOpPart(BV, ISD::ADD, DL, DAG, 0, NumElts, InVec0,
+                              InVec1))
+      X86Opcode = X86ISD::HADD;
+    else if (isHorizontalBinOpPart(BV, ISD::SUB, DL, DAG, 0, NumElts, InVec0,
+                                   InVec1))
+      X86Opcode = X86ISD::HSUB;
+    else if (isHorizontalBinOpPart(BV, ISD::FADD, DL, DAG, 0, NumElts, InVec0,
+                                   InVec1))
+      X86Opcode = X86ISD::FHADD;
+    else if (isHorizontalBinOpPart(BV, ISD::FSUB, DL, DAG, 0, NumElts, InVec0,
+                                   InVec1))
+      X86Opcode = X86ISD::FHSUB;
+    else
+      return SDValue();
+
+    // Don't try to expand this build_vector into a pair of horizontal add/sub
+    // if we can simply emit a pair of scalar add/sub.
+    if (NumUndefsLO + 1 == Half || NumUndefsHI + 1 == Half)
+      return SDValue();
+
+    // Convert this build_vector into two horizontal add/sub followed by
+    // a concat vector.
+    bool isUndefLO = NumUndefsLO == Half;
+    bool isUndefHI = NumUndefsHI == Half;
+    return ExpandHorizontalBinOp(InVec0, InVec1, DL, DAG, X86Opcode, true,
+                                 isUndefLO, isUndefHI);
+  }
+
+  return SDValue();
+}
+
 static SDValue LowerShift(SDValue Op, const X86Subtarget &Subtarget,
                           SelectionDAG &DAG);
 
@@ -9385,11 +9676,16 @@ LowerBUILD_VECTORAsVariablePermute(SDValue V, const SDLoc &DL,
                                    const X86Subtarget &Subtarget) {
   SDValue SrcVec, IndicesVec;
 
+  auto PeekThroughFreeze = [](SDValue N) {
+    if (N->getOpcode() == ISD::FREEZE && N.hasOneUse())
+      return N->getOperand(0);
+    return N;
+  };
   // Check for a match of the permute source vector and permute index elements.
   // This is done by checking that the i-th build_vector operand is of the form:
   // (extract_elt SrcVec, (extract_elt IndicesVec, i)).
   for (unsigned Idx = 0, E = V.getNumOperands(); Idx != E; ++Idx) {
-    SDValue Op = peekThroughOneUseFreeze(V.getOperand(Idx));
+    SDValue Op = PeekThroughFreeze(V.getOperand(Idx));
     if (Op.getOpcode() != ISD::EXTRACT_VECTOR_ELT)
       return SDValue();
 
@@ -9540,6 +9836,8 @@ X86TargetLowering::LowerBUILD_VECTOR(SDValue Op, SelectionDAG &DAG) const {
 
   if (SDValue AddSub = lowerToAddSubOrFMAddSub(BV, dl, Subtarget, DAG))
     return AddSub;
+  if (SDValue HorizontalOp = LowerToHorizontalOp(BV, dl, Subtarget, DAG))
+    return HorizontalOp;
   if (SDValue Broadcast = lowerBuildVectorAsBroadcast(BV, dl, Subtarget, DAG))
     return Broadcast;
   if (SDValue BitOp = lowerBuildVectorToBitOp(BV, dl, Subtarget, DAG))
@@ -10602,16 +10900,19 @@ static SDValue lowerShuffleWithPSHUFB(const SDLoc &DL, MVT VT,
          (Subtarget.hasAVX2() && VT.is256BitVector()) ||
          (Subtarget.hasBWI() && VT.is512BitVector()));
 
-  SmallVector<int, 64> PSHUFBMask(NumBytes, -1);
+  SmallVector<SDValue, 64> PSHUFBMask(NumBytes);
+  // Sign bit set in i8 mask means zero element.
+  SDValue ZeroMask = DAG.getConstant(0x80, DL, MVT::i8);
+
   SDValue V;
   for (int i = 0; i < NumBytes; ++i) {
     int M = Mask[i / NumEltBytes];
-    if (M < 0)
+    if (M < 0) {
+      PSHUFBMask[i] = DAG.getUNDEF(MVT::i8);
       continue;
-
+    }
     if (Zeroable[i / NumEltBytes]) {
-      // Sign bit set in i8 mask means zero element.
-      PSHUFBMask[i] = 0x80;
+      PSHUFBMask[i] = ZeroMask;
       continue;
     }
 
@@ -10628,14 +10929,14 @@ static SDValue lowerShuffleWithPSHUFB(const SDLoc &DL, MVT VT,
 
     M = M % LaneSize;
     M = M * NumEltBytes + (i % NumEltBytes);
-    PSHUFBMask[i] = M;
+    PSHUFBMask[i] = DAG.getConstant(M, DL, MVT::i8);
   }
   assert(V && "Failed to find a source input");
 
   MVT I8VT = MVT::getVectorVT(MVT::i8, NumBytes);
-  SDValue R = getConstVector(PSHUFBMask, I8VT, DAG, DL, /*IsMask=*/true);
-  R = DAG.getNode(X86ISD::PSHUFB, DL, I8VT, DAG.getBitcast(I8VT, V), R);
-  return DAG.getBitcast(VT, R);
+  return DAG.getBitcast(
+      VT, DAG.getNode(X86ISD::PSHUFB, DL, I8VT, DAG.getBitcast(I8VT, V),
+                      DAG.getBuildVector(I8VT, DL, PSHUFBMask)));
 }
 
 /// Return Mask with the necessary casting or extending
@@ -11045,8 +11346,15 @@ static SDValue lowerShuffleAsVTRUNC(const SDLoc &DL, MVT VT, SDValue V1,
 
       MVT SrcSVT = MVT::getIntegerVT(SrcEltBits);
       MVT SrcVT = MVT::getVectorVT(SrcSVT, NumSrcElts);
-      Src = getTargetVShiftByConstNode(X86ISD::VSRLI, DL, SrcVT, Src,
-                                       Offset * EltSizeInBits, DAG);
+      Src = DAG.getBitcast(SrcVT, Src);
+
+      // Shift the offset'd elements into place for the truncation.
+      // TODO: Use getTargetVShiftByConstNode.
+      if (Offset)
+        Src = DAG.getNode(
+            X86ISD::VSRLI, DL, SrcVT, Src,
+            DAG.getTargetConstant(Offset * EltSizeInBits, DL, MVT::i8));
+
       return getAVX512TruncNode(DL, VT, Src, Subtarget, DAG, !UndefUppers);
     }
   }
@@ -11253,30 +11561,48 @@ static SDValue lowerShuffleWithPACK(const SDLoc &DL, MVT VT, SDValue V1,
 /// one of the inputs being zeroable.
 static SDValue lowerShuffleAsBitMask(const SDLoc &DL, MVT VT, SDValue V1,
                                      SDValue V2, ArrayRef<int> Mask,
-                                     const APInt &Zeroable, SelectionDAG &DAG) {
-  unsigned EltSizeInBIts = VT.getScalarSizeInBits();
-  APInt Zero = APInt::getZero(EltSizeInBIts);
-  APInt AllOnes = APInt::getAllOnes(EltSizeInBIts);
+                                     const APInt &Zeroable,
+                                     const X86Subtarget &Subtarget,
+                                     SelectionDAG &DAG) {
+  MVT MaskVT = VT;
+  MVT EltVT = VT.getVectorElementType();
+  SDValue Zero, AllOnes;
+  // Use f64 if i64 isn't legal.
+  if (EltVT == MVT::i64 && !Subtarget.is64Bit()) {
+    EltVT = MVT::f64;
+    MaskVT = MVT::getVectorVT(EltVT, Mask.size());
+  }
 
-  SmallVector<APInt, 16> VMaskOps(Mask.size(), Zero);
+  MVT LogicVT = VT;
+  if (EltVT.isFloatingPoint()) {
+    Zero = DAG.getConstantFP(0.0, DL, EltVT);
+    APFloat AllOnesValue = APFloat::getAllOnesValue(EltVT.getFltSemantics());
+    AllOnes = DAG.getConstantFP(AllOnesValue, DL, EltVT);
+    LogicVT = MVT::getVectorVT(EltVT.changeTypeToInteger(), Mask.size());
+  } else {
+    Zero = DAG.getConstant(0, DL, EltVT);
+    AllOnes = DAG.getAllOnesConstant(DL, EltVT);
+  }
+
+  SmallVector<SDValue, 16> VMaskOps(Mask.size(), Zero);
   SDValue V;
-  for (int I = 0, Size = Mask.size(); I != Size; ++I) {
-    if (Zeroable[I])
+  for (int i = 0, Size = Mask.size(); i < Size; ++i) {
+    if (Zeroable[i])
       continue;
-    if (Mask[I] % Size != I)
+    if (Mask[i] % Size != i)
       return SDValue(); // Not a blend.
     if (!V)
-      V = Mask[I] < Size ? V1 : V2;
-    else if (V != (Mask[I] < Size ? V1 : V2))
+      V = Mask[i] < Size ? V1 : V2;
+    else if (V != (Mask[i] < Size ? V1 : V2))
       return SDValue(); // Can only let one input through the mask.
 
-    VMaskOps[I] = AllOnes;
+    VMaskOps[i] = AllOnes;
   }
   if (!V)
     return SDValue(); // No non-zeroable elements!
 
-  MVT LogicVT = VT.changeTypeToInteger();
-  SDValue VMask = getConstVector(VMaskOps, LogicVT, DAG, DL);
+  SDValue VMask = DAG.getBuildVector(MaskVT, DL, VMaskOps);
+  VMask = DAG.getBitcast(LogicVT, VMask);
   V = DAG.getBitcast(LogicVT, V);
   SDValue And = DAG.getNode(ISD::AND, DL, LogicVT, V, VMask);
   return DAG.getBitcast(VT, And);
@@ -11291,16 +11617,17 @@ static SDValue lowerShuffleAsBitBlend(const SDLoc &DL, MVT VT, SDValue V1,
                                       SDValue V2, ArrayRef<int> Mask,
                                       SelectionDAG &DAG) {
   assert(VT.isInteger() && "Only supports integer vector types!");
-  unsigned EltSizeInBIts = VT.getScalarSizeInBits();
-  APInt Zero = APInt::getZero(EltSizeInBIts);
-  APInt AllOnes = APInt::getAllOnes(EltSizeInBIts);
-  SmallVector<APInt, 16> MaskOps;
+  MVT EltVT = VT.getVectorElementType();
+  SDValue Zero = DAG.getConstant(0, DL, EltVT);
+  SDValue AllOnes = DAG.getAllOnesConstant(DL, EltVT);
+  SmallVector<SDValue, 16> MaskOps;
   for (int i = 0, Size = Mask.size(); i < Size; ++i) {
     if (Mask[i] >= 0 && Mask[i] != i && Mask[i] != i + Size)
       return SDValue(); // Shuffled input!
     MaskOps.push_back(Mask[i] < Size ? AllOnes : Zero);
   }
-  SDValue V1Mask = getConstVector(MaskOps, VT, DAG, DL);
+
+  SDValue V1Mask = DAG.getBuildVector(VT, DL, MaskOps);
   return getBitSelect(DL, VT, V1, V2, V1Mask, DAG);
 }
 
@@ -11466,8 +11793,8 @@ static SDValue lowerShuffleAsBlend(const SDLoc &DL, MVT VT, SDValue V1,
     assert(Subtarget.hasSSE41() && "128-bit byte-blends require SSE41!");
 
     // Attempt to lower to a bitmask if we can. VPAND is faster than VPBLENDVB.
-    if (SDValue Masked =
-            lowerShuffleAsBitMask(DL, VT, V1, V2, Mask, Zeroable, DAG))
+    if (SDValue Masked = lowerShuffleAsBitMask(DL, VT, V1, V2, Mask, Zeroable,
+                                               Subtarget, DAG))
       return Masked;
 
     if (Subtarget.hasBWI() && Subtarget.hasVLX()) {
@@ -11533,8 +11860,8 @@ static SDValue lowerShuffleAsBlend(const SDLoc &DL, MVT VT, SDValue V1,
     // Attempt to lower to a bitmask if we can. Only if not optimizing for size.
     bool OptForSize = DAG.shouldOptForSize();
     if (!OptForSize) {
-      if (SDValue Masked =
-              lowerShuffleAsBitMask(DL, VT, V1, V2, Mask, Zeroable, DAG))
+      if (SDValue Masked = lowerShuffleAsBitMask(DL, VT, V1, V2, Mask, Zeroable,
+                                                 Subtarget, DAG))
         return Masked;
     }
 
@@ -12080,12 +12407,14 @@ static SDValue lowerShuffleAsBitRotate(const SDLoc &DL, MVT VT, SDValue V1,
   if (!IsLegal) {
     if ((RotateAmt % 16) == 0)
       return SDValue();
+    // TODO: Use getTargetVShiftByConstNode.
     unsigned ShlAmt = RotateAmt;
     unsigned SrlAmt = RotateVT.getScalarSizeInBits() - RotateAmt;
-    SDValue SHL = getTargetVShiftByConstNode(X86ISD::VSHLI, DL, RotateVT, V1,
-                                             ShlAmt, DAG);
-    SDValue SRL = getTargetVShiftByConstNode(X86ISD::VSRLI, DL, RotateVT, V1,
-                                             SrlAmt, DAG);
+    V1 = DAG.getBitcast(RotateVT, V1);
+    SDValue SHL = DAG.getNode(X86ISD::VSHLI, DL, RotateVT, V1,
+                              DAG.getTargetConstant(ShlAmt, DL, MVT::i8));
+    SDValue SRL = DAG.getNode(X86ISD::VSRLI, DL, RotateVT, V1,
+                              DAG.getTargetConstant(SrlAmt, DL, MVT::i8));
     SDValue Rot = DAG.getNode(ISD::OR, DL, RotateVT, SHL, SRL);
     return DAG.getBitcast(VT, Rot);
   }
@@ -12268,29 +12597,18 @@ static SDValue lowerShuffleAsVALIGN(const SDLoc &DL, MVT VT, SDValue V1,
                                     const APInt &Zeroable,
                                     const X86Subtarget &Subtarget,
                                     SelectionDAG &DAG) {
-  unsigned EltBits = VT.getScalarSizeInBits();
-  if (EltBits != 32 && EltBits != 64)
-    return SDValue();
-
-  MVT AlignVT = VT.changeVectorElementTypeToInteger();
+  assert((VT.getScalarType() == MVT::i32 || VT.getScalarType() == MVT::i64) &&
+         "Only 32-bit and 64-bit elements are supported!");
 
   // 128/256-bit vectors are only supported with VLX.
-  assert((Subtarget.hasVLX() ||
-          (!AlignVT.is128BitVector() && !AlignVT.is256BitVector())) &&
-         "VLX required for 128/256-bit vectors");
-
-  auto emitVALIGN = [&](SDValue Lo, SDValue Hi, unsigned Imm) -> SDValue {
-    SDValue AlignLo = VT.isFloatingPoint() ? DAG.getBitcast(AlignVT, Lo) : Lo;
-    SDValue AlignHi = VT.isFloatingPoint() ? DAG.getBitcast(AlignVT, Hi) : Hi;
-    SDValue Res = DAG.getNode(X86ISD::VALIGN, DL, AlignVT, AlignLo, AlignHi,
-                              DAG.getTargetConstant(Imm, DL, MVT::i8));
-    return VT.isFloatingPoint() ? DAG.getBitcast(VT, Res) : Res;
-  };
+  assert((Subtarget.hasVLX() || (!VT.is128BitVector() && !VT.is256BitVector()))
+         && "VLX required for 128/256-bit vectors");
 
   SDValue Lo = V1, Hi = V2;
   int Rotation = matchShuffleAsElementRotate(Lo, Hi, Mask);
   if (0 < Rotation)
-    return emitVALIGN(Lo, Hi, Rotation);
+    return DAG.getNode(X86ISD::VALIGN, DL, VT, Lo, Hi,
+                       DAG.getTargetConstant(Rotation, DL, MVT::i8));
 
   // See if we can use VALIGN as a cross-lane version of VSHLDQ/VSRLDQ.
   // TODO: Pull this out as a matchShuffleAsElementShift helper?
@@ -12307,16 +12625,18 @@ static SDValue lowerShuffleAsVALIGN(const SDLoc &DL, MVT VT, SDValue V1,
     SDValue Src = Mask[ZeroLo] < (int)NumElts ? V1 : V2;
     int Low = Mask[ZeroLo] < (int)NumElts ? 0 : NumElts;
     if (isSequentialOrUndefInRange(Mask, ZeroLo, NumElts - ZeroLo, Low))
-      return emitVALIGN(Src, getZeroVector(AlignVT, Subtarget, DAG, DL),
-                        NumElts - ZeroLo);
+      return DAG.getNode(X86ISD::VALIGN, DL, VT, Src,
+                         getZeroVector(VT, Subtarget, DAG, DL),
+                         DAG.getTargetConstant(NumElts - ZeroLo, DL, MVT::i8));
   }
 
   if (ZeroHi) {
     SDValue Src = Mask[0] < (int)NumElts ? V1 : V2;
     int Low = Mask[0] < (int)NumElts ? 0 : NumElts;
     if (isSequentialOrUndefInRange(Mask, 0, NumElts - ZeroHi, Low + ZeroHi))
-      return emitVALIGN(getZeroVector(AlignVT, Subtarget, DAG, DL), Src,
-                        ZeroHi);
+      return DAG.getNode(X86ISD::VALIGN, DL, VT,
+                         getZeroVector(VT, Subtarget, DAG, DL), Src,
+                         DAG.getTargetConstant(ZeroHi, DL, MVT::i8));
   }
 
   return SDValue();
@@ -12509,42 +12829,6 @@ static SDValue lowerShuffleAsShift(const SDLoc &DL, MVT VT, SDValue V1,
   return DAG.getBitcast(VT, V);
 }
 
-/// Try to match a vector shuffle as a X86ISD::VSHLD funnel shift.
-static int matchShuffleAsVSHLD(MVT &ShiftVT, SDValue &V1, SDValue &V2,
-                               unsigned ScalarSizeInBits, ArrayRef<int> Mask) {
-  assert(isPowerOf2_32(ScalarSizeInBits) && ScalarSizeInBits >= 8 &&
-         "Unexpected element size");
-  int Size = Mask.size();
-  if (llvm::is_contained(Mask, SM_SentinelZero))
-    return -1;
-
-  SmallVector<int, 32> FunnelMask(Size);
-  for (int Scale = 2; (Scale * ScalarSizeInBits) <= 64; Scale *= 2) {
-    for (int Shift = 1; Shift != Scale; ++Shift) {
-      for (int Elt = 0; Elt != Size; Elt += Scale) {
-        std::iota(FunnelMask.begin() + Elt, FunnelMask.begin() + Elt + Shift,
-                  Elt + Size + (Scale - Shift));
-        std::iota(FunnelMask.begin() + Elt + Shift,
-                  FunnelMask.begin() + Elt + Scale, Elt);
-      }
-      if (isShuffleEquivalent(Mask, FunnelMask)) {
-        MVT ShiftSVT = MVT::getIntegerVT(ScalarSizeInBits * Scale);
-        ShiftVT = MVT::getVectorVT(ShiftSVT, Size / Scale);
-        return Shift * ScalarSizeInBits;
-      }
-      ShuffleVectorSDNode::commuteMask(FunnelMask);
-      if (isShuffleEquivalent(Mask, FunnelMask)) {
-        MVT ShiftSVT = MVT::getIntegerVT(ScalarSizeInBits * Scale);
-        ShiftVT = MVT::getVectorVT(ShiftSVT, Size / Scale);
-        std::swap(V1, V2);
-        return Shift * ScalarSizeInBits;
-      }
-    }
-  }
-
-  return -1;
-}
-
 // EXTRQ: Extract Len elements from lower half of source, starting at Idx.
 // Remainder of lower half result is zero and upper half is all undef.
 static bool matchShuffleAsEXTRQ(MVT VT, SDValue &V1, SDValue &V2,
@@ -13028,12 +13312,6 @@ static bool isSoftF16(T VT, const X86Subtarget &Subtarget) {
          (EltVT == MVT::f16 && !Subtarget.hasFP16());
 }
 
-template<typename T>
-static bool isBF16orSoftF16(T VT, const X86Subtarget &Subtarget) {
-  T EltVT = VT.getScalarType();
-  return EltVT == MVT::bf16 || (EltVT == MVT::f16 && !Subtarget.hasFP16());
-}
-
 /// Try to lower insertion of a single element into a zero vector.
 ///
 /// This is a common pattern that we have especially efficient patterns to lower
@@ -14106,8 +14384,8 @@ static SDValue lowerV4I32Shuffle(const SDLoc &DL, ArrayRef<int> Mask,
                                             Zeroable, Subtarget, DAG))
       return Blend;
 
-  if (SDValue Masked =
-          lowerShuffleAsBitMask(DL, MVT::v4i32, V1, V2, Mask, Zeroable, DAG))
+  if (SDValue Masked = lowerShuffleAsBitMask(DL, MVT::v4i32, V1, V2, Mask,
+                                             Zeroable, Subtarget, DAG))
     return Masked;
 
   // Use dedicated unpack instructions for masks that match their pattern.
@@ -14822,8 +15100,8 @@ static SDValue lowerV8I16Shuffle(const SDLoc &DL, ArrayRef<int> Mask,
                                             Zeroable, Subtarget, DAG))
       return Blend;
 
-  if (SDValue Masked =
-          lowerShuffleAsBitMask(DL, MVT::v8i16, V1, V2, Mask, Zeroable, DAG))
+  if (SDValue Masked = lowerShuffleAsBitMask(DL, MVT::v8i16, V1, V2, Mask,
+                                             Zeroable, Subtarget, DAG))
     return Masked;
 
   // Use dedicated unpack instructions for masks that match their pattern.
@@ -15178,8 +15456,8 @@ static SDValue lowerV16I8Shuffle(const SDLoc &DL, ArrayRef<int> Mask,
       return V;
   }
 
-  if (SDValue Masked =
-          lowerShuffleAsBitMask(DL, MVT::v16i8, V1, V2, Mask, Zeroable, DAG))
+  if (SDValue Masked = lowerShuffleAsBitMask(DL, MVT::v16i8, V1, V2, Mask,
+                                             Zeroable, Subtarget, DAG))
     return Masked;
 
   // Use dedicated unpack instructions for masks that match their pattern.
@@ -17517,8 +17795,8 @@ static SDValue lower256BitShuffle(const SDLoc &DL, ArrayRef<int> Mask, MVT VT,
     if (ElementBits < 32) {
       // No floating point type available, if we can't use the bit operations
       // for masking/blending then decompose into 128-bit vectors.
-      if (SDValue V =
-              lowerShuffleAsBitMask(DL, VT, V1, V2, Mask, Zeroable, DAG))
+      if (SDValue V = lowerShuffleAsBitMask(DL, VT, V1, V2, Mask, Zeroable,
+                                            Subtarget, DAG))
         return V;
       if (SDValue V = lowerShuffleAsBitBlend(DL, VT, V1, V2, Mask, DAG))
         return V;
@@ -17715,12 +17993,6 @@ static SDValue lowerV8F64Shuffle(const SDLoc &DL, ArrayRef<int> Mask,
                                           Zeroable, Subtarget, DAG))
     return Blend;
 
-  // Try to use VALIGN via integer domain bitcast. Avoids VPERMPD which
-  // requires an extra register for the index vector; VALIGNQ uses an immediate.
-  if (SDValue Rotate = lowerShuffleAsVALIGN(DL, MVT::v8f64, V1, V2, Mask,
-                                            Zeroable, Subtarget, DAG))
-    return Rotate;
-
   return lowerShuffleWithPERMV(DL, MVT::v8f64, Mask, V1, V2, Subtarget, DAG);
 }
 
@@ -17788,12 +18060,6 @@ static SDValue lowerV16F32Shuffle(const SDLoc &DL, ArrayRef<int> Mask,
                                          Zeroable, Subtarget, DAG))
     return V;
 
-  // Try to use VALIGN via integer domain bitcast. Avoids VPERMPS which
-  // requires an extra register for the index vector; VALIGND uses an immediate.
-  if (SDValue Rotate = lowerShuffleAsVALIGN(DL, MVT::v16f32, V1, V2, Mask,
-                                            Zeroable, Subtarget, DAG))
-    return Rotate;
-
   return lowerShuffleWithPERMV(DL, MVT::v16f32, Mask, V1, V2, Subtarget, DAG);
 }
 
@@ -18081,8 +18347,8 @@ static SDValue lowerV64I8Shuffle(const SDLoc &DL, ArrayRef<int> Mask,
       return Rotate;
 
   // Lower as AND if possible.
-  if (SDValue Masked =
-          lowerShuffleAsBitMask(DL, MVT::v64i8, V1, V2, Mask, Zeroable, DAG))
+  if (SDValue Masked = lowerShuffleAsBitMask(DL, MVT::v64i8, V1, V2, Mask,
+                                             Zeroable, Subtarget, DAG))
     return Masked;
 
   if (SDValue PSHUFB = lowerShuffleWithPSHUFB(DL, MVT::v64i8, Mask, V1, V2,
@@ -18176,7 +18442,8 @@ static SDValue lower512BitShuffle(const SDLoc &DL, ArrayRef<int> Mask,
   if ((VT == MVT::v32i16 || VT == MVT::v64i8) && !Subtarget.hasBWI()) {
     // Try using bit ops for masking and blending before falling back to
     // splitting.
-    if (SDValue V = lowerShuffleAsBitMask(DL, VT, V1, V2, Mask, Zeroable, DAG))
+    if (SDValue V = lowerShuffleAsBitMask(DL, VT, V1, V2, Mask, Zeroable,
+                                          Subtarget, DAG))
       return V;
     if (SDValue V = lowerShuffleAsBitBlend(DL, VT, V1, V2, Mask, DAG))
       return V;
@@ -19520,18 +19787,10 @@ static SDValue LowerFLDEXP(SDValue Op, const X86Subtarget &Subtarget,
     return splitVectorOp(Op, DAG, DL);
   }
   SDValue WideX = widenSubVector(X, true, Subtarget, DAG, DL, 512);
-  // Widen Exp to the same *lane count* as WideX (not necessarily 512 bits) so
-  // SINT_TO_FP has matching vector lengths. For wide f64 the int exponent
-  // vector is narrower than 512 bits (e.g. v2i32 -> v8i32 to match v8f64).
-  MVT WideExpVT =
-      MVT::getVectorVT(Exp.getSimpleValueType().getVectorElementType(),
-                       WideX.getValueType().getVectorNumElements());
-  SDValue WideExp = widenSubVector(WideExpVT, Exp, /*ZeroNewElements=*/true,
-                                   Subtarget, DAG, DL);
-  SDValue WideExpFp =
-      DAG.getNode(ISD::SINT_TO_FP, DL, WideX.getValueType(), WideExp);
+  SDValue WideExp = widenSubVector(Exp, true, Subtarget, DAG, DL, 512);
+  Exp = DAG.getNode(ISD::SINT_TO_FP, DL, WideExp.getSimpleValueType(), Exp);
   SDValue Scalef =
-      DAG.getNode(X86ISD::SCALEF, DL, WideX.getValueType(), WideX, WideExpFp);
+      DAG.getNode(X86ISD::SCALEF, DL, WideX.getValueType(), WideX, WideExp);
   SDValue Final =
       DAG.getExtractSubvector(DL, X.getSimpleValueType(), Scalef, 0);
   return DAG.getFPExtendOrRound(Final, DL, XTy);
@@ -20487,7 +20746,7 @@ static SDValue promoteXINT_TO_FP(SDValue Op, const SDLoc &dl,
   SDValue Src = Op.getOperand(IsStrict ? 1 : 0);
   SDValue Chain = IsStrict ? Op->getOperand(0) : DAG.getEntryNode();
   MVT VT = Op.getSimpleValueType();
-  MVT NVT = VT.changeElementType(MVT::f32);
+  MVT NVT = VT.isVector() ? VT.changeVectorElementType(MVT::f32) : MVT::f32;
 
   SDValue Rnd = DAG.getIntPtrConstant(0, dl, /*isTarget=*/true);
   if (IsStrict)
@@ -20534,7 +20793,7 @@ SDValue X86TargetLowering::LowerSINT_TO_FP(SDValue Op,
   MVT VT = Op.getSimpleValueType();
   SDLoc dl(Op);
 
-  if (isBF16orSoftF16(VT, Subtarget))
+  if (isSoftF16(VT, Subtarget))
     return promoteXINT_TO_FP(Op, dl, DAG);
   else if (isLegalConversion(SrcVT, VT, true, Subtarget))
     return Op;
@@ -20666,7 +20925,7 @@ std::pair<SDValue, SDValue> X86TargetLowering::BuildFILD(
 /// Horizontal vector math instructions may be slower than normal math with
 /// shuffles. Limit horizontal op codegen based on size/speed trade-offs, uarch
 /// implementation, and likely shuffle complexity of the alternate sequence.
-static bool shouldUseHorizontalOp(bool IsSingleSource, const SelectionDAG &DAG,
+static bool shouldUseHorizontalOp(bool IsSingleSource, SelectionDAG &DAG,
                                   const X86Subtarget &Subtarget) {
   bool IsOptimizingSize = DAG.shouldOptForSize();
   bool HasFastHOps = Subtarget.hasFastHorizontalOps();
@@ -21034,32 +21293,32 @@ SDValue X86TargetLowering::LowerUINT_TO_FP(SDValue Op,
   bool IsStrict = Op->isStrictFPOpcode();
   unsigned OpNo = IsStrict ? 1 : 0;
   SDValue Src = Op.getOperand(OpNo);
-  SDValue Chain = IsStrict ? Op.getOperand(0) : DAG.getEntryNode();
+  SDLoc dl(Op);
   auto PtrVT = getPointerTy(DAG.getDataLayout());
   MVT SrcVT = Src.getSimpleValueType();
   MVT DstVT = Op->getSimpleValueType(0);
-  SDLoc dl(Op);
+  SDValue Chain = IsStrict ? Op.getOperand(0) : DAG.getEntryNode();
+
+  // Bail out when we don't have native conversion instructions.
+  if (DstVT == MVT::f128)
+    return SDValue();
 
-  if (isBF16orSoftF16(DstVT, Subtarget))
+  if (isSoftF16(DstVT, Subtarget))
     return promoteXINT_TO_FP(Op, dl, DAG);
   else if (isLegalConversion(SrcVT, DstVT, false, Subtarget))
     return Op;
 
-  if (Subtarget.isTargetWin64() && SrcVT == MVT::i128)
-    return LowerWin64_INT128_TO_FP(Op, DAG);
-
-  if (SDValue Extract = vectorizeExtractedCast(Op, dl, DAG, Subtarget))
-    return Extract;
-
   if (SDValue V = lowerFPToIntToFP(Op, dl, DAG, Subtarget))
     return V;
 
   if (DstVT.isVector())
     return lowerUINT_TO_FP_vec(Op, dl, DAG, Subtarget);
 
-  // Bail out when we don't have native conversion instructions.
-  if (DstVT == MVT::f128)
-    return SDValue();
+  if (Subtarget.isTargetWin64() && SrcVT == MVT::i128)
+    return LowerWin64_INT128_TO_FP(Op, DAG);
+
+  if (SDValue Extract = vectorizeExtractedCast(Op, dl, DAG, Subtarget))
+    return Extract;
 
   if (Subtarget.hasAVX512() && isScalarFPTypeInSSEReg(DstVT) &&
       (SrcVT == MVT::i32 || (SrcVT == MVT::i64 && Subtarget.is64Bit()))) {
@@ -22044,8 +22303,9 @@ static SDValue expandFP_TO_UINT_SSE(MVT VT, SDValue Src, const SDLoc &dl,
     return DAG.getNode(X86ISD::BLENDV, dl, VT, Small, Overflow, Small);
   }
 
-  SDValue IsOverflown = getTargetVShiftByConstNode(X86ISD::VSRAI, dl, VT, Small,
-                                                   DstBits - 1, DAG);
+  SDValue IsOverflown =
+      DAG.getNode(X86ISD::VSRAI, dl, VT, Small,
+                  DAG.getTargetConstant(DstBits - 1, dl, MVT::i8));
   return DAG.getNode(ISD::OR, dl, VT, Small,
                      DAG.getNode(ISD::AND, dl, VT, Big, IsOverflown));
 }
@@ -22062,8 +22322,8 @@ SDValue X86TargetLowering::LowerFP_TO_INT(SDValue Op, SelectionDAG &DAG) const {
   SDLoc dl(Op);
 
   SDValue Res;
-  if (isBF16orSoftF16(SrcVT, Subtarget)) {
-    MVT NVT = VT.changeElementType(MVT::f32);
+  if (isSoftF16(SrcVT, Subtarget)) {
+    MVT NVT = VT.isVector() ? VT.changeVectorElementType(MVT::f32) : MVT::f32;
     if (IsStrict)
       return DAG.getNode(Op.getOpcode(), dl, {VT, MVT::Other},
                          {Chain, DAG.getNode(ISD::STRICT_FP_EXTEND, dl,
@@ -22506,21 +22766,13 @@ X86TargetLowering::LowerFP_TO_INT_SAT(SDValue Op, SelectionDAG &DAG) const {
   EVT SrcVT = Src.getValueType();
   EVT DstVT = Node->getValueType(0);
   EVT TmpVT = DstVT;
-  EVT SatVT = cast<VTSDNode>(Node->getOperand(1))->getVT();
-
-  if (Subtarget.hasAVX10_2() && SrcVT.isVector() &&
-      SrcVT.getVectorElementType() == MVT::bf16 && SatVT == MVT::i8) {
-    MVT VecI16VT = SrcVT.getSimpleVT().changeVectorElementType(MVT::i16);
-    SDValue Res = DAG.getNode(IsSigned ? X86ISD::CVTTP2IBS : X86ISD::CVTTP2IUBS,
-                              dl, VecI16VT, Src);
-    return DAG.getNode(ISD::TRUNCATE, dl, DstVT, Res);
-  }
 
   // This code is only for floats and doubles. Fall back to generic code for
   // anything else.
-  if (!isScalarFPTypeInSSEReg(SrcVT) || isBF16orSoftF16(SrcVT, Subtarget))
+  if (!isScalarFPTypeInSSEReg(SrcVT) || isSoftF16(SrcVT, Subtarget))
     return SDValue();
 
+  EVT SatVT = cast<VTSDNode>(Node->getOperand(1))->getVT();
   unsigned SatWidth = SatVT.getScalarSizeInBits();
   unsigned DstWidth = DstVT.getScalarSizeInBits();
   unsigned TmpWidth = TmpVT.getScalarSizeInBits();
@@ -23488,7 +23740,7 @@ static bool matchScalarReduction(SDValue Op, ISD::NodeType BinOp,
       return false;
 
     SDValue Src = I->getOperand(0);
-    auto M = SrcOpMap.find(Src);
+    DenseMap<SDValue, APInt>::iterator M = SrcOpMap.find(Src);
     if (M == SrcOpMap.end()) {
       VT = Src.getValueType();
       // Quit if not the same type.
@@ -24415,7 +24667,8 @@ static SDValue incDecVectorConstant(SDValue V, SelectionDAG &DAG, bool IsInc,
   MVT VT = V.getSimpleValueType();
   MVT EltVT = VT.getVectorElementType();
   unsigned NumElts = VT.getVectorNumElements();
-  SmallVector<APInt, 8> NewVecC;
+  SmallVector<SDValue, 8> NewVecC;
+  SDLoc DL(V);
   for (unsigned i = 0; i < NumElts; ++i) {
     auto *Elt = dyn_cast<ConstantSDNode>(BV->getOperand(i));
     if (!Elt || Elt->isOpaque() || Elt->getSimpleValueType(0) != EltVT)
@@ -24429,9 +24682,10 @@ static SDValue incDecVectorConstant(SDValue V, SelectionDAG &DAG, bool IsInc,
                 (!IsInc && EltC.isMinSignedValue())))
       return SDValue();
 
-    NewVecC.push_back(EltC + (IsInc ? 1 : -1));
+    NewVecC.push_back(DAG.getConstant(EltC + (IsInc ? 1 : -1), DL, EltVT));
   }
-  return getConstVector(NewVecC, VT, DAG, SDLoc(V));
+
+  return DAG.getBuildVector(VT, DL, NewVecC);
 }
 
 /// As another special case, use PSUBUS[BW] when it's profitable. E.g. for
@@ -24568,12 +24822,6 @@ static SDValue LowerVSETCC(SDValue Op, const X86Subtarget &Subtarget,
     SDValue Cmp;
     bool IsAlwaysSignaling;
     unsigned SSECC = translateX86FSETCC(Cond, Op0, Op1, IsAlwaysSignaling);
-    if ((Cond == ISD::SETO || Cond == ISD::SETUO) && Op0 != Op1) {
-      if (DAG.isKnownNeverNaN(Op1))
-        Op1 = Op0;
-      else if (DAG.isKnownNeverNaN(Op0))
-        Op0 = Op1;
-    }
     if (!Subtarget.hasAVX()) {
       // TODO: We could use following steps to handle a quiet compare with
       // signaling encodings.
@@ -25565,12 +25813,12 @@ SDValue X86TargetLowering::LowerSELECT(SDValue Op, SelectionDAG &DAG) const {
     unsigned CondCode = Cond.getConstantOperandVal(0);
 
     // Special handling for __builtin_ffs(X) - 1 pattern which looks like
-    // (select (seteq X, 0), -1, (cttz_zero_poison X)). Disable the special
+    // (select (seteq X, 0), -1, (cttz_zero_undef X)). Disable the special
     // handle to keep the CMP with 0. This should be removed by
     // optimizeCompareInst by using the flags from the BSR/TZCNT used for the
-    // cttz_zero_poison.
+    // cttz_zero_undef.
     auto MatchFFSMinus1 = [&](SDValue Op1, SDValue Op2) {
-      return (Op1.getOpcode() == ISD::CTTZ_ZERO_POISON && Op1.hasOneUse() &&
+      return (Op1.getOpcode() == ISD::CTTZ_ZERO_UNDEF && Op1.hasOneUse() &&
               Op1.getOperand(0) == CmpOp0 && isAllOnesConstant(Op2));
     };
     if (Subtarget.canUseCMOV() && (VT == MVT::i32 || VT == MVT::i64) &&
@@ -25899,8 +26147,8 @@ static SDValue LowerEXTEND_VECTOR_INREG(SDValue Op,
     Curr = DAG.getBitcast(DestVT, Curr);
 
     unsigned SignExtShift = DestWidth - InSVT.getSizeInBits();
-    SignExt = getTargetVShiftByConstNode(X86ISD::VSRAI, dl, DestVT, Curr,
-                                         SignExtShift, DAG);
+    SignExt = DAG.getNode(X86ISD::VSRAI, dl, DestVT, Curr,
+                          DAG.getTargetConstant(SignExtShift, dl, MVT::i8));
   }
 
   if (VT == MVT::v2i64) {
@@ -26512,12 +26760,11 @@ static SDValue LowerVACOPY(SDValue Op, const X86Subtarget &Subtarget,
   const Value *DstSV = cast<SrcValueSDNode>(Op.getOperand(3))->getValue();
   const Value *SrcSV = cast<SrcValueSDNode>(Op.getOperand(4))->getValue();
   SDLoc DL(Op);
-  Align Alignment = Align(Subtarget.isTarget64BitLP64() ? 8 : 4);
 
   return DAG.getMemcpy(
       Chain, DL, DstPtr, SrcPtr,
       DAG.getIntPtrConstant(Subtarget.isTarget64BitLP64() ? 24 : 16, DL),
-      Alignment, Alignment, /*isVolatile*/ false, false,
+      Align(Subtarget.isTarget64BitLP64() ? 8 : 4), /*isVolatile*/ false, false,
       /*CI=*/nullptr, std::nullopt, MachinePointerInfo(DstSV),
       MachinePointerInfo(SrcSV));
 }
@@ -26530,7 +26777,7 @@ static SDValue getVectorMaskingNode(SDValue Op, SDValue Mask,
                                     const X86Subtarget &Subtarget,
                                     SelectionDAG &DAG) {
   MVT VT = Op.getSimpleValueType();
-  MVT MaskVT = VT.changeElementType(MVT::i1);
+  MVT MaskVT = MVT::getVectorVT(MVT::i1, VT.getVectorNumElements());
   unsigned OpcodeSelect = ISD::VSELECT;
   SDLoc dl(Op);
 
@@ -27308,7 +27555,7 @@ SDValue X86TargetLowering::LowerINTRINSIC_WO_CHAIN(SDValue Op,
         return DAG.getNode(IntrData->Opc0, dl, Op.getValueType(), Src);
 
       MVT SrcVT = Src.getSimpleValueType();
-      MVT MaskVT = SrcVT.changeElementType(MVT::i1);
+      MVT MaskVT = MVT::getVectorVT(MVT::i1, SrcVT.getVectorNumElements());
       Mask = getMaskNode(Mask, MaskVT, Subtarget, DAG, dl);
       return DAG.getNode(IntrData->Opc1, dl, Op.getValueType(),
                          {Src, PassThru, Mask});
@@ -27323,7 +27570,7 @@ SDValue X86TargetLowering::LowerINTRINSIC_WO_CHAIN(SDValue Op,
         return DAG.getNode(IntrData->Opc0, dl, Op.getValueType(), {Src, Src2});
 
       MVT Src2VT = Src2.getSimpleValueType();
-      MVT MaskVT = Src2VT.changeElementType(MVT::i1);
+      MVT MaskVT = MVT::getVectorVT(MVT::i1, Src2VT.getVectorNumElements());
       Mask = getMaskNode(Mask, MaskVT, Subtarget, DAG, dl);
       return DAG.getNode(IntrData->Opc1, dl, Op.getValueType(),
                          {Src, Src2, PassThru, Mask});
@@ -27351,7 +27598,7 @@ SDValue X86TargetLowering::LowerINTRINSIC_WO_CHAIN(SDValue Op,
       else
         Opc = IntrData->Opc1;
       MVT SrcVT = Src.getSimpleValueType();
-      MVT MaskVT = SrcVT.changeElementType(MVT::i1);
+      MVT MaskVT = MVT::getVectorVT(MVT::i1, SrcVT.getVectorNumElements());
       Mask = getMaskNode(Mask, MaskVT, Subtarget, DAG, dl);
       return DAG.getNode(Opc, dl, Op.getValueType(), Src, Rnd, PassThru, Mask);
     }
@@ -27983,7 +28230,7 @@ bool X86::isExtendedSwiftAsyncFrameSupported(const X86Subtarget &Subtarget,
     return false;
   // 64-bit targets support extended Swift async frame setup,
   // except for targets that use the windows 64 prologue.
-  return !MF.getTarget().getMCAsmInfo().usesWindowsCFI();
+  return !MF.getTarget().getMCAsmInfo()->usesWindowsCFI();
 }
 
 static SDValue LowerINTRINSIC_W_CHAIN(SDValue Op, const X86Subtarget &Subtarget,
@@ -28530,7 +28777,7 @@ SDValue X86TargetLowering::LowerFRAMEADDR(SDValue Op, SelectionDAG &DAG) const {
 
   MFI.setFrameAddressIsTaken(true);
 
-  if (MF.getTarget().getMCAsmInfo().usesWindowsCFI()) {
+  if (MF.getTarget().getMCAsmInfo()->usesWindowsCFI()) {
     // Depth > 0 makes no sense on targets which use Windows unwind codes.  It
     // is not possible to crawl up the stack without looking at the unwind codes
     // simultaneously.
@@ -29121,7 +29368,7 @@ SDValue X86TargetLowering::LowerRESET_FPENV(SDValue Op,
   MachinePointerInfo MPI =
       MachinePointerInfo::getConstantPool(DAG.getMachineFunction());
   MachineMemOperand *MMO = MF.getMachineMemOperand(
-      MPI, MachineMemOperand::MOLoad, X87StateSize, Align(4));
+      MPI, MachineMemOperand::MOStore, X87StateSize, Align(4));
 
   return createSetFPEnvNodes(Env, Chain, DL, MVT::i32, MMO, DAG, Subtarget);
 }
@@ -29166,25 +29413,6 @@ SDValue getGFNICtrlMask(unsigned Opcode, SelectionDAG &DAG, const SDLoc &DL,
   return DAG.getBuildVector(VT, DL, MaskBits);
 }
 
-static APInt getGFNIByteAffine(const APInt &ByteToAffine, const APInt &Matrix64,
-                               const APInt &Addend8) {
-  assert(ByteToAffine.getBitWidth() == 8 && "Byte input unexpected size!");
-  assert(Addend8.getBitWidth() == 8 && "8-bit addend input unexpected size!");
-  assert(Matrix64.getBitWidth() == 64 &&
-         "64-bit matrix input unexpected size!");
-
-  APInt ByteSplat = APInt::getSplat(64, ByteToAffine);
-  ByteSplat &= Matrix64.byteSwap();
-
-  // Cumulative parity
-  for (unsigned I = 0; I != 3; ++I)
-    ByteSplat ^= ByteSplat.lshr(1 << I);
-  ByteSplat &= 0x0101010101010101ull;
-
-  APInt Affined = APIntOps::ScaleBitMask(ByteSplat, 8);
-  return Affined ^ Addend8;
-}
-
 /// Lower a vector CTLZ using native supported vector CTLZ instruction.
 //
 // i8/i16 vector implemented using dword LZCNT vector instruction
@@ -29253,7 +29481,7 @@ static SDValue LowerVectorCTLZInRegLUT(SDValue Op, const SDLoc &DL,
   SDValue Hi = DAG.getNode(ISD::SRL, DL, CurrVT, Op0, NibbleShift);
   SDValue HiZ;
   if (CurrVT.is512BitVector()) {
-    MVT MaskVT = CurrVT.changeElementType(MVT::i1);
+    MVT MaskVT = MVT::getVectorVT(MVT::i1, CurrVT.getVectorNumElements());
     HiZ = DAG.getSetCC(DL, MaskVT, Hi, Zero, ISD::SETEQ);
     HiZ = DAG.getNode(ISD::SIGN_EXTEND, DL, CurrVT, HiZ);
   } else {
@@ -29279,7 +29507,7 @@ static SDValue LowerVectorCTLZInRegLUT(SDValue Op, const SDLoc &DL,
 
     // Check if the upper half of the input element is zero.
     if (CurrVT.is512BitVector()) {
-      MVT MaskVT = CurrVT.changeElementType(MVT::i1);
+      MVT MaskVT = MVT::getVectorVT(MVT::i1, CurrVT.getVectorNumElements());
       HiZ = DAG.getSetCC(DL, MaskVT, DAG.getBitcast(CurrVT, Op0),
                          DAG.getBitcast(CurrVT, Zero), ISD::SETEQ);
       HiZ = DAG.getNode(ISD::SIGN_EXTEND, DL, CurrVT, HiZ);
@@ -29385,11 +29613,6 @@ static SDValue LowerCTTZ(SDValue Op, const X86Subtarget &Subtarget,
   SDLoc dl(Op);
   bool NonZeroSrc = DAG.isKnownNeverZero(N0);
 
-  // Default to expansion if CTPOP is legal.
-  if (VT.isVector() &&
-      DAG.getTargetLoweringInfo().isOperationLegal(ISD::CTPOP, VT))
-    return SDValue();
-
   // GFNI - isolate LSB and perform GF2P8AFFINEQB lookup.
   if (Subtarget.hasGFNI() && VT.isVector() &&
       VT.getVectorElementType() == MVT::i8) {
@@ -29422,69 +29645,6 @@ static SDValue LowerCTTZ(SDValue Op, const X86Subtarget &Subtarget,
   return DAG.getNode(X86ISD::CMOV, dl, VT, Ops);
 }
 
-static SDValue peekThroughDemandedElts(SDValue V, const APInt &DemandedElts,
-                                       SelectionDAG &DAG) {
-  const TargetLowering &TLI = DAG.getTargetLoweringInfo();
-  while (SDValue NewV =
-             TLI.SimplifyMultipleUseDemandedVectorElts(V, DemandedElts, DAG))
-    V = NewV;
-  return V;
-}
-
-// Generic x86 vector reduction expansion.
-static SDValue LowerVECREDUCE(SDValue Op, const X86Subtarget &Subtarget,
-                              SelectionDAG &DAG, bool AllowScalarization) {
-  ISD::NodeType BinOp = ISD::getVecReduceBaseOpcode(Op.getOpcode());
-  assert(DAG.getTargetLoweringInfo().isBinOp(BinOp) &&
-         "Only binops expected to be used by reductions");
-
-  EVT ExtractVT = Op.getValueType();
-  SDValue Src = Op.getOperand(0);
-  EVT SrcVT = Src.getValueType();
-  EVT SrcSVT = SrcVT.getScalarType();
-  const TargetLowering &TLI = DAG.getTargetLoweringInfo();
-  SDLoc DL(Op);
-
-  // TODO: Pad non-pow2 vectors with identity constants.
-  if (SrcSVT != ExtractVT || SrcVT.getSizeInBits() < 128 ||
-      !isPowerOf2_32(SrcVT.getVectorNumElements()))
-    return SDValue();
-
-  // Split vector down to 128-bits, performing bin to lo/hi subvectors.
-  while (SrcVT.getSizeInBits() > 128) {
-    SDValue Lo, Hi;
-    std::tie(Lo, Hi) = splitVector(Src, DAG, DL);
-    SrcVT = Lo.getValueType();
-    Src = DAG.getNode(BinOp, DL, SrcVT, Lo, Hi);
-  }
-  assert(SrcVT.is128BitVector() && "Unexpected value type");
-
-  // Expand 128-bit shuffle tree + reduction binops.
-  unsigned NumSrcElts = SrcVT.getVectorNumElements();
-  for (unsigned NumElts = NumSrcElts; NumElts != 1; NumElts /= 2) {
-    // Scalarize the last 2 elements if the vector binop isn't legal.
-    if (NumElts == 2 && AllowScalarization &&
-        !TLI.isOperationLegal(BinOp, SrcVT) && TLI.isTypeLegal(ExtractVT)) {
-      return DAG.getNode(BinOp, DL, ExtractVT,
-                         DAG.getExtractVectorElt(DL, ExtractVT, Src, 0),
-                         DAG.getExtractVectorElt(DL, ExtractVT, Src, 1));
-    }
-
-    // Peek through identity elements added by legalisation padding.
-    APInt HiElts = APInt::getBitsSet(NumSrcElts, NumElts / 2, NumElts);
-    if (!DAG.isIdentityElement(BinOp, SDNodeFlags(), Src, HiElts, 1)) {
-      APInt LoElts = APInt::getLowBitsSet(NumSrcElts, NumElts / 2);
-      SDValue Lo = peekThroughDemandedElts(Src, LoElts, DAG);
-      SmallVector<int, 16> Mask(NumSrcElts, -1);
-      std::iota(Mask.begin(), Mask.begin() + (NumElts / 2), NumElts / 2);
-      SDValue Hi =
-          DAG.getVectorShuffle(SrcVT, DL, Src, DAG.getUNDEF(SrcVT), Mask);
-      Src = DAG.getNode(BinOp, DL, SrcVT, Lo, Hi);
-    }
-  }
-  return DAG.getExtractVectorElt(DL, ExtractVT, Src, 0);
-}
-
 static SDValue lowerAddSub(SDValue Op, SelectionDAG &DAG,
                            const X86Subtarget &Subtarget) {
   MVT VT = Op.getSimpleValueType();
@@ -29509,6 +29669,20 @@ static SDValue LowerADDSAT_SUBSAT(SDValue Op, SelectionDAG &DAG,
   unsigned Opcode = Op.getOpcode();
   SDLoc DL(Op);
 
+  if (Opcode == ISD::USUBSAT && !VT.isVector() && Subtarget.hasNDD()) {
+    if (isOneConstant(Y)) {
+      // usub.sat(X,1) == (X==0 ? 0 : X-1). Lower to cmp+adc with NDD.
+      SDValue Sub = DAG.getNode(X86ISD::SUB, DL, DAG.getVTList(VT, MVT::i32), X,
+                                DAG.getConstant(1, DL, VT));
+      SDValue EFLAGS = Sub.getValue(1);
+      SDValue MinusOne = DAG.getAllOnesConstant(DL, VT);
+      return DAG.getNode(X86ISD::ADC, DL, DAG.getVTList(VT, MVT::i32), X,
+                         MinusOne, EFLAGS);
+    }
+    // Scalar USUBSAT was previously Expand. Don't fall through to vector path.
+    return SDValue();
+  }
+
   if (VT == MVT::v32i16 || VT == MVT::v64i8 ||
       (VT.is256BitVector() && !Subtarget.hasInt256())) {
     assert(Op.getSimpleValueType().isInteger() &&
@@ -29672,80 +29846,6 @@ static SDValue LowerMINMAX(SDValue Op, const X86Subtarget &Subtarget,
   return SDValue();
 }
 
-// Attempt to replace an min/max v8i16/v16i8 horizontal reduction with
-// PHMINPOSUW.
-static SDValue LowerMINMAX_REDUCE(SDValue Op, const X86Subtarget &Subtarget,
-                                  SelectionDAG &DAG) {
-  EVT ExtractVT = Op.getValueType();
-  bool AllowScalarization = !Subtarget.hasAVX512();
-  if (!Subtarget.hasSSE41() || (ExtractVT != MVT::i16 && ExtractVT != MVT::i8))
-    return LowerVECREDUCE(Op, Subtarget, DAG, AllowScalarization);
-
-  SDValue Src = Op.getOperand(0);
-  EVT SrcVT = Src.getValueType();
-  EVT SrcSVT = SrcVT.getScalarType();
-  ISD::NodeType BinOp = ISD::getVecReduceBaseOpcode(Op.getOpcode());
-  if (SrcSVT != ExtractVT || SrcVT.getSizeInBits() < 128 ||
-      !isPowerOf2_32(SrcVT.getVectorNumElements()))
-    return SDValue();
-
-  // Bail if at least the upper half elements are identity.
-  APInt HiElts = APInt::getHighBitsSet(SrcVT.getVectorNumElements(),
-                                       SrcVT.getVectorNumElements() / 2);
-  if (DAG.isIdentityElement(BinOp, SDNodeFlags(), Src, HiElts, 1))
-    return LowerVECREDUCE(Op, Subtarget, DAG, AllowScalarization);
-
-  SDLoc DL(Op);
-  SDValue MinPos = Src;
-
-  // First, reduce the source down to 128-bit, applying BinOp to lo/hi.
-  while (SrcVT.getSizeInBits() > 128) {
-    SDValue Lo, Hi;
-    std::tie(Lo, Hi) = splitVector(MinPos, DAG, DL);
-    SrcVT = Lo.getValueType();
-    MinPos = DAG.getNode(BinOp, DL, SrcVT, Lo, Hi);
-  }
-  assert(((SrcVT == MVT::v8i16 && ExtractVT == MVT::i16) ||
-          (SrcVT == MVT::v16i8 && ExtractVT == MVT::i8)) &&
-         "Unexpected value type");
-
-  // PHMINPOSUW applies to UMIN(v8i16), for SMIN/SMAX/UMAX we must apply a mask
-  // to flip the value accordingly.
-  SDValue Mask;
-  unsigned MaskEltsBits = ExtractVT.getSizeInBits();
-  if (BinOp == ISD::SMAX)
-    Mask = DAG.getConstant(APInt::getSignedMaxValue(MaskEltsBits), DL, SrcVT);
-  else if (BinOp == ISD::SMIN)
-    Mask = DAG.getConstant(APInt::getSignedMinValue(MaskEltsBits), DL, SrcVT);
-  else if (BinOp == ISD::UMAX)
-    Mask = DAG.getAllOnesConstant(DL, SrcVT);
-
-  if (Mask)
-    MinPos = DAG.getNode(ISD::XOR, DL, SrcVT, Mask, MinPos);
-
-  // For v16i8 cases we need to perform UMIN on pairs of byte elements,
-  // shuffling each upper element down and insert zeros. This means that the
-  // v16i8 UMIN will leave the upper element as zero, performing zero-extension
-  // ready for the PHMINPOS.
-  if (ExtractVT == MVT::i8) {
-    SDValue Upper = DAG.getVectorShuffle(
-        SrcVT, DL, MinPos, DAG.getConstant(0, DL, MVT::v16i8),
-        {1, 16, 3, 16, 5, 16, 7, 16, 9, 16, 11, 16, 13, 16, 15, 16});
-    MinPos = DAG.getNode(ISD::UMIN, DL, SrcVT, MinPos, Upper);
-  }
-
-  // Perform the PHMINPOS on a v8i16 vector,
-  MinPos = DAG.getBitcast(MVT::v8i16, MinPos);
-  MinPos = DAG.getNode(X86ISD::PHMINPOS, DL, MVT::v8i16, MinPos);
-  MinPos = DAG.getBitcast(SrcVT, MinPos);
-
-  if (Mask)
-    MinPos = DAG.getNode(ISD::XOR, DL, SrcVT, Mask, MinPos);
-
-  return DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, ExtractVT, MinPos,
-                     DAG.getVectorIdxConstant(0, DL));
-}
-
 static SDValue LowerFMINIMUM_FMAXIMUM(SDValue Op, const X86Subtarget &Subtarget,
                                       SelectionDAG &DAG) {
   const TargetLowering &TLI = DAG.getTargetLoweringInfo();
@@ -29843,8 +29943,8 @@ static SDValue LowerFMINIMUM_FMAXIMUM(SDValue Op, const X86Subtarget &Subtarget,
   bool IsXNeverNaN = DAG.isKnownNeverNaN(X);
   bool IsYNeverNaN = DAG.isKnownNeverNaN(Y);
   bool IgnoreSignedZero = Op->getFlags().hasNoSignedZeros() ||
-                          DAG.isKnownNeverLogicalZero(X) ||
-                          DAG.isKnownNeverLogicalZero(Y);
+                          DAG.isKnownNeverZeroFloat(X) ||
+                          DAG.isKnownNeverZeroFloat(Y);
   bool ShouldHandleZeros = true;
   SDValue NewX = X;
   SDValue NewY = Y;
@@ -30821,16 +30921,6 @@ static SDValue LowerShiftByScalarImmediate(SDValue Op, SelectionDAG &DAG,
       return DAG.getNode(ISD::AND, dl, VT, SRL, DAG.getConstant(Mask, dl, VT));
     }
     if (Op.getOpcode() == ISD::SRA) {
-      // ashr(R, 1) === xor(avgceilu(R,-1),and(not(R),MIN_SIGNED))
-      if (ShiftAmt == 1) {
-        R = DAG.getFreeze(R);
-        SDValue AllOnes = DAG.getAllOnesConstant(dl, VT);
-        SDValue Avg = DAG.getNode(ISD::AVGCEILU, dl, VT, R, AllOnes);
-        SDValue Not = DAG.getNode(ISD::XOR, dl, VT, R, AllOnes);
-        SDValue Hi =
-            DAG.getNode(ISD::AND, dl, VT, Not, DAG.getConstant(0x80, dl, VT));
-        return DAG.getNode(ISD::XOR, dl, VT, Avg, Hi);
-      }
       // ashr(R, Amt) === sub(xor(lshr(R, Amt), Mask), Mask)
       SDValue Res = DAG.getNode(ISD::SRL, dl, VT, R, Amt);
 
@@ -31996,25 +32086,6 @@ static SDValue LowerRotate(SDValue Op, const X86Subtarget &Subtarget,
   if (VT.is256BitVector() && (Subtarget.hasXOP() || !Subtarget.hasAVX2()))
     return splitVectorIntBinary(Op, DAG, DL);
 
-  // rotl(x,7) -> pavgb(x, 0 - (x & 1))
-  if (EltSizeInBits == 8 && IsCstSplat &&
-      CstSplatValue.urem(EltSizeInBits) == 7 && IsROTL &&
-      !Subtarget.hasAVX512()) {
-    SDValue One = DAG.getConstant(1, DL, VT);
-    SDValue LSB = DAG.getNode(ISD::AND, DL, VT, R, One);
-    SDValue Neg = DAG.getNegative(LSB, DL, VT);
-    return DAG.getNode(ISD::AVGCEILU, DL, VT, R, Neg);
-  }
-
-  // rotl(x,1) -> sub(add(x, x), icmp_slt(x, 0))
-  if (IsROTL && EltSizeInBits == 8 && IsCstSplat &&
-      CstSplatValue.urem(EltSizeInBits) == 1 && !Subtarget.hasAVX512()) {
-    SDValue Double = DAG.getNode(ISD::ADD, DL, VT, R, R);
-    SDValue Zero = DAG.getConstant(0, DL, VT);
-    SDValue CmpNeg = DAG.getSetCC(DL, VT, R, Zero, ISD::SETLT);
-    return DAG.getNode(ISD::SUB, DL, VT, Double, CmpNeg);
-  }
-
   // Rotate by an uniform constant - expand back to shifts.
   // TODO: Can't use generic expansion as UNDEF amt elements can be converted
   // to other values when folded to shift amounts, losing the splat.
@@ -32434,12 +32505,8 @@ X86TargetLowering::shouldExpandLogicAtomicRMWInIR(
 }
 
 void X86TargetLowering::emitBitTestAtomicRMWIntrinsic(AtomicRMWInst *AI) const {
-  LLVMContext &Ctx = AI->getContext();
-  IRBuilder<ConstantFolder, IRBuilderCallbackInserter> Builder(
-      Ctx, ConstantFolder{}, IRBuilderCallbackInserter([&AI](Instruction *I) {
-        I->copyMetadata(*AI, LLVMContext::MD_pcsections);
-      }));
-  Builder.SetInsertPoint(AI);
+  IRBuilder<> Builder(AI);
+  Builder.CollectMetadataToCopy(AI, {LLVMContext::MD_pcsections});
   Intrinsic::ID IID_C = Intrinsic::not_intrinsic;
   Intrinsic::ID IID_I = Intrinsic::not_intrinsic;
   switch (AI->getOperation()) {
@@ -32459,6 +32526,7 @@ void X86TargetLowering::emitBitTestAtomicRMWIntrinsic(AtomicRMWInst *AI) const {
     break;
   }
   Instruction *I = AI->user_back();
+  LLVMContext &Ctx = AI->getContext();
   Value *Addr = Builder.CreatePointerCast(AI->getPointerOperand(),
                                           PointerType::getUnqual(Ctx));
   Value *Result = nullptr;
@@ -32515,188 +32583,100 @@ void X86TargetLowering::emitBitTestAtomicRMWIntrinsic(AtomicRMWInst *AI) const {
   AI->eraseFromParent();
 }
 
-static X86::CondCode matchSignedNewValueCC(const Instruction *I) {
-  using namespace llvm::PatternMatch;
-  if (match(I->user_back(),
-            m_SpecificICmp(CmpInst::ICMP_SLT, m_Value(), m_ZeroInt())))
-    return X86::COND_S;
-  if (match(I->user_back(),
-            m_SpecificICmp(CmpInst::ICMP_SGT, m_Value(), m_AllOnes())))
-    return X86::COND_NS;
-  return X86::COND_INVALID;
-}
-
-static X86::CondCode matchAddCC(const AtomicRMWInst *AI, const Instruction *I) {
-  using namespace llvm::PatternMatch;
-  Value *Op = AI->getOperand(1);
-  CmpPredicate Pred;
-
-  // Folded from icmp eq/ne (old + Op), 0 to icmp eq/ne old, -Op.
-  if (match(I, m_c_ICmp(Pred, m_Neg(m_Specific(Op)), m_Value()))) {
-    if (Pred == CmpInst::ICMP_EQ)
-      return X86::COND_E;
-    if (Pred == CmpInst::ICMP_NE)
-      return X86::COND_NE;
-  }
-
-  // Non-folded SF form: %new = add %old, Op; icmp slt/sgt %new, 0/-1
-  // lock add sets SF on the new value directly.
-  if (match(I, m_OneUse(m_c_Add(m_Specific(Op), m_Value()))))
-    return matchSignedNewValueCC(I);
-
-  return X86::COND_INVALID;
-}
-
-static X86::CondCode matchSubCC(const AtomicRMWInst *AI, const Instruction *I) {
+static bool shouldExpandCmpArithRMWInIR(const AtomicRMWInst *AI) {
   using namespace llvm::PatternMatch;
-  Value *Op = AI->getOperand(1);
-  CmpPredicate Pred;
-
-  // Folded from icmp eq/ne (old - Op), 0 to icmp eq/ne old, Op.
-  if (match(I, m_c_ICmp(Pred, m_Specific(Op), m_Value()))) {
-    if (Pred == CmpInst::ICMP_EQ)
-      return X86::COND_E;
-    if (Pred == CmpInst::ICMP_NE)
-      return X86::COND_NE;
-  }
-
-  // Non-folded SF form: %new = sub %old, Op; icmp slt/sgt %new, 0/-1
-  // lock sub sets SF on the new value directly.
-  if (match(I, m_OneUse(m_Sub(m_Value(), m_Specific(Op)))))
-    return matchSignedNewValueCC(I);
-
-  return X86::COND_INVALID;
-}
+  if (!AI->hasOneUse())
+    return false;
 
-static X86::CondCode matchOrCC(const AtomicRMWInst *AI, const Instruction *I) {
-  using namespace llvm::PatternMatch;
   Value *Op = AI->getOperand(1);
   CmpPredicate Pred;
-
-  // Non-folded form: %new = or %old, Op; icmp P %new, 0/-1
-  // lock or sets ZF/SF on the new value directly.
-  if (!match(I, m_OneUse(m_c_Or(m_Specific(Op), m_Value()))))
-    return X86::COND_INVALID;
-
-  if (match(I->user_back(), m_ICmp(Pred, m_Value(), m_ZeroInt()))) {
-    if (Pred == CmpInst::ICMP_EQ)
-      return X86::COND_E;
-    if (Pred == CmpInst::ICMP_NE)
-      return X86::COND_NE;
-    if (Pred == CmpInst::ICMP_SLT)
-      return X86::COND_S;
+  const Instruction *I = AI->user_back();
+  AtomicRMWInst::BinOp Opc = AI->getOperation();
+  if (Opc == AtomicRMWInst::Add) {
+    if (match(I, m_c_ICmp(Pred, m_Sub(m_ZeroInt(), m_Specific(Op)), m_Value())))
+      return Pred == CmpInst::ICMP_EQ || Pred == CmpInst::ICMP_NE;
+    if (match(I, m_OneUse(m_c_Add(m_Specific(Op), m_Value())))) {
+      if (match(I->user_back(),
+                m_SpecificICmp(CmpInst::ICMP_SLT, m_Value(), m_ZeroInt())))
+        return true;
+      if (match(I->user_back(),
+                m_SpecificICmp(CmpInst::ICMP_SGT, m_Value(), m_AllOnes())))
+        return true;
+    }
+    return false;
   }
-  if (match(I->user_back(),
-            m_SpecificICmp(CmpInst::ICMP_SGT, m_Value(), m_AllOnes())))
-    return X86::COND_NS;
-
-  return X86::COND_INVALID;
-}
-
-static X86::CondCode matchAndCC(const AtomicRMWInst *AI, const Instruction *I) {
-  using namespace llvm::PatternMatch;
-  Value *Op = AI->getOperand(1);
-  CmpPredicate Pred;
-
-  // Non-folded form: %new = and %old, Op; icmp P %new, 0/-1
-  // lock and sets ZF/SF on the new value directly.
-  if (match(I, m_OneUse(m_c_And(m_Specific(Op), m_Value())))) {
-    if (match(I->user_back(), m_ICmp(Pred, m_Value(), m_ZeroInt()))) {
-      if (Pred == CmpInst::ICMP_EQ)
-        return X86::COND_E;
-      if (Pred == CmpInst::ICMP_NE)
-        return X86::COND_NE;
-      if (Pred == CmpInst::ICMP_SLT)
-        return X86::COND_S;
+  if (Opc == AtomicRMWInst::Sub) {
+    if (match(I, m_c_ICmp(Pred, m_Specific(Op), m_Value())))
+      return Pred == CmpInst::ICMP_EQ || Pred == CmpInst::ICMP_NE;
+    if (match(I, m_OneUse(m_Sub(m_Value(), m_Specific(Op))))) {
+      if (match(I->user_back(),
+                m_SpecificICmp(CmpInst::ICMP_SLT, m_Value(), m_ZeroInt())))
+        return true;
+      if (match(I->user_back(),
+                m_SpecificICmp(CmpInst::ICMP_SGT, m_Value(), m_AllOnes())))
+        return true;
     }
+    return false;
+  }
+  if ((Opc == AtomicRMWInst::Or &&
+       match(I, m_OneUse(m_c_Or(m_Specific(Op), m_Value())))) ||
+      (Opc == AtomicRMWInst::And &&
+       match(I, m_OneUse(m_c_And(m_Specific(Op), m_Value()))))) {
+    if (match(I->user_back(), m_ICmp(Pred, m_Value(), m_ZeroInt())))
+      return Pred == CmpInst::ICMP_EQ || Pred == CmpInst::ICMP_NE ||
+             Pred == CmpInst::ICMP_SLT;
     if (match(I->user_back(),
               m_SpecificICmp(CmpInst::ICMP_SGT, m_Value(), m_AllOnes())))
-      return X86::COND_NS;
-    return X86::COND_INVALID;
-  }
-
-  // If -C is a power of 2:
-  //   (old & C) == 0 <=> old ult -C
-  //   (old & C) != 0 <=> old ugt ~C
-  auto *CI = dyn_cast<ConstantInt>(Op);
-  if (!CI)
-    return X86::COND_INVALID;
-  const APInt &C = CI->getValue();
-  const APInt *K;
-  if (!(-C).isPowerOf2() ||
-      !match(I, m_c_ICmp(Pred, m_Specific(AI), m_APInt(K))))
-    return X86::COND_INVALID;
-  if (Pred == ICmpInst::ICMP_ULT && *K == -C)
-    return X86::COND_E;
-  if (Pred == ICmpInst::ICMP_UGT && *K == ~C)
-    return X86::COND_NE;
-  return X86::COND_INVALID;
-}
-
-static X86::CondCode matchXorCC(const AtomicRMWInst *AI, const Instruction *I) {
-  using namespace llvm::PatternMatch;
-  Value *Op = AI->getOperand(1);
-  CmpPredicate Pred;
-
-  // Folded from icmp eq/ne (old ^ Op), 0 to icmp eq/ne old, Op.
-  if (match(I, m_c_ICmp(Pred, m_Specific(Op), m_Value()))) {
-    if (Pred == CmpInst::ICMP_EQ)
-      return X86::COND_E;
-    if (Pred == CmpInst::ICMP_NE)
-      return X86::COND_NE;
+      return true;
+    return false;
   }
-
-  // Non-folded SF form: %new = xor %old, Op; icmp slt/sgt %new, 0/-1
-  // lock xor sets SF on the new value directly.
-  if (match(I, m_OneUse(m_c_Xor(m_Specific(Op), m_Value()))))
-    return matchSignedNewValueCC(I);
-
-  return X86::COND_INVALID;
-}
-
-static X86::CondCode getCmpArithCC(const AtomicRMWInst *AI) {
-  if (!AI->hasOneUse())
-    return X86::COND_INVALID;
-
-  const Instruction *I = AI->user_back();
-  switch (AI->getOperation()) {
-  case AtomicRMWInst::Add:
-    return matchAddCC(AI, I);
-  case AtomicRMWInst::Sub:
-    return matchSubCC(AI, I);
-  case AtomicRMWInst::Or:
-    return matchOrCC(AI, I);
-  case AtomicRMWInst::And:
-    return matchAndCC(AI, I);
-  case AtomicRMWInst::Xor:
-    return matchXorCC(AI, I);
-  default:
-    return X86::COND_INVALID;
+  if (Opc == AtomicRMWInst::Xor) {
+    if (match(I, m_c_ICmp(Pred, m_Specific(Op), m_Value())))
+      return Pred == CmpInst::ICMP_EQ || Pred == CmpInst::ICMP_NE;
+    if (match(I, m_OneUse(m_c_Xor(m_Specific(Op), m_Value())))) {
+      if (match(I->user_back(),
+                m_SpecificICmp(CmpInst::ICMP_SLT, m_Value(), m_ZeroInt())))
+        return true;
+      if (match(I->user_back(),
+                m_SpecificICmp(CmpInst::ICMP_SGT, m_Value(), m_AllOnes())))
+        return true;
+    }
+    return false;
   }
-}
 
-static bool shouldExpandCmpArithRMWInIR(const AtomicRMWInst *AI) {
-  return getCmpArithCC(AI) != X86::COND_INVALID;
+  return false;
 }
 
 void X86TargetLowering::emitCmpArithAtomicRMWIntrinsic(
     AtomicRMWInst *AI) const {
-  LLVMContext &Ctx = AI->getContext();
-  IRBuilder<ConstantFolder, IRBuilderCallbackInserter> Builder(
-      Ctx, ConstantFolder{}, IRBuilderCallbackInserter([&AI](Instruction *I) {
-        I->copyMetadata(*AI, LLVMContext::MD_pcsections);
-      }));
-  Builder.SetInsertPoint(AI);
+  IRBuilder<> Builder(AI);
+  Builder.CollectMetadataToCopy(AI, {LLVMContext::MD_pcsections});
   Instruction *TempI = nullptr;
+  LLVMContext &Ctx = AI->getContext();
   ICmpInst *ICI = dyn_cast<ICmpInst>(AI->user_back());
   if (!ICI) {
     TempI = AI->user_back();
     assert(TempI->hasOneUse() && "Must have one use");
     ICI = cast<ICmpInst>(TempI->user_back());
   }
-  X86::CondCode CC = getCmpArithCC(AI);
-  assert(CC != X86::COND_INVALID && "emitCmpArithAtomicRMWIntrinsic called "
-                                    "without a recognised pattern");
+  X86::CondCode CC = X86::COND_INVALID;
+  ICmpInst::Predicate Pred = ICI->getPredicate();
+  switch (Pred) {
+  default:
+    llvm_unreachable("Not supported Pred");
+  case CmpInst::ICMP_EQ:
+    CC = X86::COND_E;
+    break;
+  case CmpInst::ICMP_NE:
+    CC = X86::COND_NE;
+    break;
+  case CmpInst::ICMP_SLT:
+    CC = X86::COND_S;
+    break;
+  case CmpInst::ICMP_SGT:
+    CC = X86::COND_NS;
+    break;
+  }
   Intrinsic::ID IID = Intrinsic::not_intrinsic;
   switch (AI->getOperation()) {
   default:
@@ -32796,12 +32776,8 @@ X86TargetLowering::lowerIdempotentRMWIntoFencedLoad(AtomicRMWInst *AI) const {
         AI->use_empty())
       return nullptr;
 
-  IRBuilder<ConstantFolder, IRBuilderCallbackInserter> Builder(
-      AI->getContext(), ConstantFolder{},
-      IRBuilderCallbackInserter([&AI](Instruction *I) {
-        I->copyMetadata(*AI, LLVMContext::MD_pcsections);
-      }));
-  Builder.SetInsertPoint(AI);
+  IRBuilder<> Builder(AI);
+  Builder.CollectMetadataToCopy(AI, {LLVMContext::MD_pcsections});
   auto SSID = AI->getSyncScopeID();
   // We must restrict the ordering to avoid generating loads with Release or
   // ReleaseAcquire orderings.
@@ -33395,15 +33371,11 @@ static SDValue LowerBITREVERSE(SDValue Op, const X86Subtarget &Subtarget,
     Res = DAG.getNode(ISD::BITREVERSE, DL, ByteVT, Res);
     return DAG.getBitcast(VT, Res);
   }
-  assert(VT.isVectorOf(MVT::i8) && "Only byte vector BITREVERSE supported");
+  assert(VT.isVector() && VT.getScalarType() == MVT::i8 &&
+         "Only byte vector BITREVERSE supported");
 
   unsigned NumElts = VT.getVectorNumElements();
 
-  // If we have BMM, BITREVERSE on vXi8 is marked Legal and will be handled
-  // by TableGen pattern matching to VPBITREVB instruction. We should not
-  // reach here in that case.
-  assert(!Subtarget.hasBMM() && "BMM should use Legal operation action");
-
   // If we have GFNI, we can use GF2P8AFFINEQB to reverse the bits.
   if (Subtarget.hasGFNI()) {
     SDValue Matrix = getGFNICtrlMask(ISD::BITREVERSE, DAG, DL, VT);
@@ -34315,9 +34287,9 @@ SDValue X86TargetLowering::LowerOperation(SDValue Op, SelectionDAG &DAG) const {
   case ISD::SET_FPENV_MEM:      return LowerSET_FPENV_MEM(Op, DAG);
   case ISD::RESET_FPENV:        return LowerRESET_FPENV(Op, DAG);
   case ISD::CTLZ:
-  case ISD::CTLZ_ZERO_POISON:   return LowerCTLZ(Op, Subtarget, DAG);
+  case ISD::CTLZ_ZERO_UNDEF:    return LowerCTLZ(Op, Subtarget, DAG);
   case ISD::CTTZ:
-  case ISD::CTTZ_ZERO_POISON:   return LowerCTTZ(Op, Subtarget, DAG);
+  case ISD::CTTZ_ZERO_UNDEF:    return LowerCTTZ(Op, Subtarget, DAG);
   case ISD::MUL:                return LowerMUL(Op, Subtarget, DAG);
   case ISD::MULHS:
   case ISD::MULHU:              return LowerMULH(Op, Subtarget, DAG);
@@ -34348,14 +34320,6 @@ SDValue X86TargetLowering::LowerOperation(SDValue Op, SelectionDAG &DAG) const {
   case ISD::SMIN:
   case ISD::UMAX:
   case ISD::UMIN:               return LowerMINMAX(Op, Subtarget, DAG);
-  case ISD::VECREDUCE_SMAX:
-  case ISD::VECREDUCE_SMIN:
-  case ISD::VECREDUCE_UMAX:
-  case ISD::VECREDUCE_UMIN:     return LowerMINMAX_REDUCE(Op, Subtarget, DAG);
-  case ISD::VECREDUCE_AND:
-  case ISD::VECREDUCE_OR:
-  case ISD::VECREDUCE_XOR:
-  case ISD::VECREDUCE_MUL:      return LowerVECREDUCE(Op, Subtarget, DAG, true);
   case ISD::FMINIMUM:
   case ISD::FMAXIMUM:
   case ISD::FMINIMUMNUM:
@@ -34795,8 +34759,8 @@ void X86TargetLowering::ReplaceNodeResults(SDNode *N,
   }
   case ISD::CTLZ:
   case ISD::CTTZ:
-  case ISD::CTLZ_ZERO_POISON:
-  case ISD::CTTZ_ZERO_POISON: {
+  case ISD::CTLZ_ZERO_UNDEF:
+  case ISD::CTTZ_ZERO_UNDEF: {
     // Fold i256/i512 CTLZ/CTTZ patterns to make use of AVX512
     // vXi64 CTLZ/CTTZ and VECTOR_COMPRESS.
     // Compute the CTLZ/CTTZ of each element, add the element's bit offset,
@@ -34825,7 +34789,7 @@ void X86TargetLowering::ReplaceNodeResults(SDNode *N,
     // CTLZ - reverse the elements as we want the top non-zero element at the
     // bottom for compression.
     unsigned VecOpc = ISD::CTTZ;
-    if (Opc == ISD::CTLZ || Opc == ISD::CTLZ_ZERO_POISON) {
+    if (Opc == ISD::CTLZ || Opc == ISD::CTLZ_ZERO_UNDEF) {
       VecOpc = ISD::CTLZ;
       Vec = DAG.getVectorShuffle(VecVT, dl, Vec, Vec, RevMask);
     }
@@ -35158,21 +35122,6 @@ void X86TargetLowering::ReplaceNodeResults(SDNode *N,
     }
     return;
   }
-  case ISD::VECREDUCE_MUL: {
-    assert(N->getValueType(0) == MVT::i64 && "Unexpected vector reduction");
-    if (SDValue Res = LowerVECREDUCE(SDValue(N, 0), Subtarget, DAG, false))
-      Results.push_back(Res);
-    return;
-  }
-  case ISD::VECREDUCE_SMAX:
-  case ISD::VECREDUCE_SMIN:
-  case ISD::VECREDUCE_UMAX:
-  case ISD::VECREDUCE_UMIN: {
-    assert(N->getValueType(0) == MVT::i64 && "Unexpected vector reduction");
-    if (SDValue Res = LowerMINMAX_REDUCE(SDValue(N, 0), Subtarget, DAG))
-      Results.push_back(Res);
-    return;
-  }
   case ISD::FP_TO_SINT_SAT:
   case ISD::FP_TO_UINT_SAT: {
     if (!Subtarget.hasAVX10_2())
@@ -35182,17 +35131,8 @@ void X86TargetLowering::ReplaceNodeResults(SDNode *N,
     EVT VT = N->getValueType(0);
     SDValue Op = N->getOperand(0);
     EVT OpVT = Op.getValueType();
-    EVT SatVT = cast<VTSDNode>(N->getOperand(1))->getVT();
     SDValue Res;
 
-    if (VT == MVT::v8i8 && OpVT == MVT::v8bf16 && SatVT == MVT::i8) {
-      Res = DAG.getNode(IsSigned ? X86ISD::CVTTP2IBS : X86ISD::CVTTP2IUBS, dl,
-                        MVT::v8i16, Op);
-      Res = DAG.getNode(ISD::TRUNCATE, dl, VT, Res);
-      Results.push_back(Res);
-      return;
-    }
-
     if (VT == MVT::v2i32 && OpVT == MVT::v2f64) {
       if (IsSigned)
         Res = DAG.getNode(X86ISD::FP_TO_SINT_SAT, dl, MVT::v4i32, Op);
@@ -35214,7 +35154,7 @@ void X86TargetLowering::ReplaceNodeResults(SDNode *N,
     EVT SrcVT = Src.getValueType();
 
     SDValue Res;
-    if (isBF16orSoftF16(SrcVT, Subtarget)) {
+    if (isSoftF16(SrcVT, Subtarget)) {
       EVT NVT = VT.changeElementType(*DAG.getContext(), MVT::f32);
       if (IsStrict) {
         Res =
@@ -37145,7 +37085,7 @@ X86TargetLowering::EmitLoweredProbedAlloca(MachineInstr &MI,
 
   BuildMI(testMBB, MIMD, TII->get(X86::JCC_1))
       .addMBB(tailMBB)
-      .addImm(X86::COND_AE);
+      .addImm(X86::COND_GE);
   testMBB->addSuccessor(blockMBB);
   testMBB->addSuccessor(tailMBB);
 
@@ -38730,7 +38670,8 @@ X86TargetLowering::EmitInstrWithCustomInserter(MachineInstr &MI,
   case X86::PTDPBF8PS:
   case X86::PTDPBHF8PS:
   case X86::PTDPHBF8PS:
-  case X86::PTDPHF8PS: {
+  case X86::PTDPHF8PS:
+  case X86::PTMMULTF32PS: {
     unsigned Opc;
     switch (MI.getOpcode()) {
     default: llvm_unreachable("illegal opcode!");
@@ -38747,6 +38688,7 @@ X86TargetLowering::EmitInstrWithCustomInserter(MachineInstr &MI,
     case X86::PTDPBHF8PS: Opc = X86::TDPBHF8PS; break;
     case X86::PTDPHBF8PS: Opc = X86::TDPHBF8PS; break;
     case X86::PTDPHF8PS: Opc = X86::TDPHF8PS; break;
+    case X86::PTMMULTF32PS: Opc = X86::TMMULTF32PS; break;
       // clang-format on
     }
 
@@ -39399,6 +39341,25 @@ void X86TargetLowering::computeKnownBitsForTargetNode(const SDValue Op,
     Known.One.clearAllBits();
     break;
   }
+  case X86ISD::PDEP: {
+    KnownBits Known2;
+    Known = DAG.computeKnownBits(Op.getOperand(1), DemandedElts, Depth + 1);
+    Known2 = DAG.computeKnownBits(Op.getOperand(0), DemandedElts, Depth + 1);
+    // Zeros are retained from the mask operand. But not ones.
+    Known.One.clearAllBits();
+    // The result will have at least as many trailing zeros as the non-mask
+    // operand since bits can only map to the same or higher bit position.
+    Known.Zero.setLowBits(Known2.countMinTrailingZeros());
+    break;
+  }
+  case X86ISD::PEXT: {
+    Known = DAG.computeKnownBits(Op.getOperand(1), DemandedElts, Depth + 1);
+    // The result has as many leading zeros as the number of zeroes in the mask.
+    unsigned Count = Known.Zero.popcount();
+    Known.Zero = APInt::getHighBitsSet(BitWidth, Count);
+    Known.One.clearAllBits();
+    break;
+  }
   case X86ISD::VTRUNC:
   case X86ISD::VTRUNCS:
   case X86ISD::VTRUNCUS:
@@ -39450,14 +39411,6 @@ void X86TargetLowering::computeKnownBitsForTargetNode(const SDValue Op,
       Known.setAllZero();
     break;
   }
-  case X86ISD::VZEXT_LOAD: {
-    // Upper elements are known zero.
-    auto *LN = cast<MemIntrinsicSDNode>(Op);
-    if (!DemandedElts[0] &&
-        LN->getMemoryVT().getSizeInBits() == VT.getScalarSizeInBits())
-      Known.setAllZero();
-    break;
-  }
   case X86ISD::VBROADCAST_LOAD: {
     APInt UndefElts;
     SmallVector<APInt, 16> EltBits;
@@ -39805,16 +39758,19 @@ static bool matchUnaryShuffle(MVT MaskVT, ArrayRef<int> Mask,
   unsigned NumMaskElts = Mask.size();
   unsigned MaskEltSize = MaskVT.getScalarSizeInBits();
 
-  // Match against a foldable vXi32/vXi16 VZEXT_MOVL zero-extending instruction.
-  if ((MaskEltSize == 32 || (MaskEltSize == 16 && Subtarget.hasFP16())) &&
-      (V1.getOpcode() == ISD::SCALAR_TO_VECTOR || isa<MemSDNode>(V1)) &&
-      Mask[0] == 0 && isUndefOrZeroInRange(Mask, 1, NumMaskElts - 1)) {
-    Shuffle = X86ISD::VZEXT_MOVL;
-    if (MaskEltSize == 16)
-      SrcVT = DstVT = MaskVT.changeVectorElementType(MVT::f16);
-    else
-      SrcVT = DstVT = !Subtarget.hasSSE2() ? MVT::v4f32 : MaskVT;
-    return true;
+  // Match against a VZEXT_MOVL vXi32 and vXi16 zero-extending instruction.
+  if (Mask[0] == 0 &&
+      (MaskEltSize == 32 || (MaskEltSize == 16 && Subtarget.hasFP16()))) {
+    if ((isUndefOrZero(Mask[1]) && isUndefInRange(Mask, 2, NumMaskElts - 2)) ||
+        (V1.getOpcode() == ISD::SCALAR_TO_VECTOR &&
+         isUndefOrZeroInRange(Mask, 1, NumMaskElts - 1))) {
+      Shuffle = X86ISD::VZEXT_MOVL;
+      if (MaskEltSize == 16)
+        SrcVT = DstVT = MaskVT.changeVectorElementType(MVT::f16);
+      else
+        SrcVT = DstVT = !Subtarget.hasSSE2() ? MVT::v4f32 : MaskVT;
+      return true;
+    }
   }
 
   // Match against a ANY/SIGN/ZERO_EXTEND_VECTOR_INREG instruction.
@@ -39869,7 +39825,8 @@ static bool matchUnaryShuffle(MVT MaskVT, ArrayRef<int> Mask,
   // Match against a VZEXT_MOVL instruction, SSE1 only supports 32-bits (MOVSS).
   if (((MaskEltSize == 32) || (MaskEltSize == 64 && Subtarget.hasSSE2()) ||
        (MaskEltSize == 16 && Subtarget.hasFP16())) &&
-      Mask[0] == 0 && isUndefOrZeroInRange(Mask, 1, NumMaskElts - 1)) {
+      isUndefOrEqual(Mask[0], 0) &&
+      isUndefOrZeroInRange(Mask, 1, NumMaskElts - 1)) {
     Shuffle = X86ISD::VZEXT_MOVL;
     if (MaskEltSize == 16)
       SrcVT = DstVT = MaskVT.changeVectorElementType(MVT::f16);
@@ -40363,16 +40320,6 @@ static bool matchBinaryPermuteShuffle(
     }
   }
 
-  // Attempt to match against VSHLD funnel shift.
-  if (AllowIntDomain && Subtarget.hasVBMI2()) {
-    int ShiftAmt = matchShuffleAsVSHLD(ShuffleVT, V1, V2, EltSizeInBits, Mask);
-    if (0 < ShiftAmt) {
-      Shuffle = X86ISD::VSHLD;
-      PermuteImm = (unsigned)ShiftAmt;
-      return true;
-    }
-  }
-
   // Attempt to combine to X86ISD::BLENDI.
   if ((NumMaskElts <= 8 && ((Subtarget.hasSSE41() && MaskVT.is128BitVector()) ||
                             (Subtarget.hasAVX() && MaskVT.is256BitVector()))) ||
@@ -42114,20 +42061,7 @@ static SDValue combineX86ShufflesRecursively(
             Op, OpScaledDemandedElts, DAG))
       Op = NewOp;
   }
-
-  // Reresolve - we might have repeated subvector sources.
-  resolveTargetShuffleInputsAndMask(Ops, Mask);
-
-  // Handle the all undef/zero/ones cases.
-  if (all_of(Mask, [](int Idx) { return Idx == SM_SentinelUndef; }))
-    return DAG.getUNDEF(RootVT);
-  if (all_of(Mask, [](int Idx) { return Idx < 0; }))
-    return getZeroVector(RootVT, Subtarget, DAG, DL);
-  if (Ops.size() == 1 && ISD::isBuildVectorAllOnes(Ops[0].getNode()) &&
-      !llvm::is_contained(Mask, SM_SentinelZero))
-    return getOnesVector(RootVT, DAG, DL);
-
-  assert(!Ops.empty() && "Shuffle with no inputs detected");
+  // FIXME: should we rerun resolveTargetShuffleInputsAndMask() now?
 
   // Widen any subvector shuffle inputs we've collected.
   // TODO: Remove this to avoid generating temporary nodes, we should only
@@ -42139,8 +42073,21 @@ static SDValue combineX86ShufflesRecursively(
       if (Op.getValueSizeInBits() < RootSizeInBits)
         Op = widenSubVector(Op, false, Subtarget, DAG, SDLoc(Op),
                             RootSizeInBits);
+    // Reresolve - we might have repeated subvector sources.
+    resolveTargetShuffleInputsAndMask(Ops, Mask);
   }
 
+  // Handle the all undef/zero/ones cases.
+  if (all_of(Mask, [](int Idx) { return Idx == SM_SentinelUndef; }))
+    return DAG.getUNDEF(RootVT);
+  if (all_of(Mask, [](int Idx) { return Idx < 0; }))
+    return getZeroVector(RootVT, Subtarget, DAG, DL);
+  if (Ops.size() == 1 && ISD::isBuildVectorAllOnes(Ops[0].getNode()) &&
+      !llvm::is_contained(Mask, SM_SentinelZero))
+    return getOnesVector(RootVT, DAG, DL);
+
+  assert(!Ops.empty() && "Shuffle with no inputs detected");
+
   // We can only combine unary and binary shuffle mask cases.
   if (Ops.size() <= 2) {
     // Minor canonicalization of the accumulated shuffle mask to make it easier
@@ -42904,22 +42851,6 @@ static SDValue combineTargetShuffle(SDValue N, const SDLoc &DL,
         TLI.isTypeLegal(Src.getOperand(0).getValueType()))
       return DAG.getNode(X86ISD::VBROADCAST, DL, VT, Src.getOperand(0));
 
-    // broadcast(truncate(extract_vector_elt(x, 0))) -> bitcast(broadcast(x)).
-    if (Src.getOpcode() == ISD::TRUNCATE &&
-        Src.getOperand(0).getOpcode() == ISD::EXTRACT_VECTOR_ELT &&
-        isNullConstant(Src.getOperand(0).getOperand(1))) {
-      SDValue NewSrc = Src.getOperand(0).getOperand(0);
-      if (Src.getOperand(0).getValueType() ==
-              NewSrc.getValueType().getScalarType() &&
-          TLI.isTypeLegal(NewSrc.getValueType())) {
-        MVT VecVT = MVT::getVectorVT(Src.getSimpleValueType(),
-                                     NewSrc.getValueSizeInBits() /
-                                         Src.getValueSizeInBits());
-        return DAG.getNode(X86ISD::VBROADCAST, DL, VT,
-                           DAG.getBitcast(VecVT, NewSrc));
-      }
-    }
-
     // Share broadcast with the longest vector and extract low subvector (free).
     // Ensure the same SDValue from the SDNode use is being used.
     for (SDNode *User : Src->users())
@@ -43292,25 +43223,11 @@ static SDValue combineTargetShuffle(SDValue N, const SDLoc &DL,
     // If we're permuting the upper 256-bits subvectors of a concatenation, then
     // see if we can peek through and access the subvector directly.
     if (VT.is512BitVector()) {
+      // 512-bit mask uses 4 x i2 indices - if the msb is always set then only
+      // the upper subvector is used.
       SDValue LHS = peekThroughBitcasts(N->getOperand(0));
       SDValue RHS = peekThroughBitcasts(N->getOperand(1));
       uint64_t Mask = N->getConstantOperandVal(2);
-      // Attempt to concat directly instead of shuffling source together.
-      // TODO: combineX86ShufflesRecursively should do this generically.
-      if (Mask == 0x44 && LHS.getValueType() == RHS.getValueType() &&
-          LHS.getValueType().isSimple()) {
-        SmallVector<SDValue> LHSOps, RHSOps;
-        if (collectConcatOps(LHS.getNode(), LHSOps, DAG) &&
-            collectConcatOps(RHS.getNode(), RHSOps, DAG) &&
-            LHSOps.size() == 2 && RHSOps.size() == 2) {
-          if (SDValue Concat = combineConcatVectorOps(
-                  DL, LHS.getSimpleValueType(), {LHSOps[0], RHSOps[0]}, DAG,
-                  Subtarget))
-            return DAG.getBitcast(VT, Concat);
-        }
-      }
-      // 512-bit mask uses 4 x i2 indices - if the msb is always set then only
-      // the upper subvector is used.
       SmallVector<SDValue> LHSOps, RHSOps;
       SDValue NewLHS, NewRHS;
       if ((Mask & 0x0A) == 0x0A &&
@@ -43669,26 +43586,20 @@ static SDValue combineTargetShuffle(SDValue N, const SDLoc &DL,
         return lowerShuffleWithPERMV(DL, VT, Mask, N.getOperand(0),
                                      DAG.getUNDEF(VT), Subtarget, DAG);
       }
-      // If sources are widened, then concat and use VPERMV with adjusted mask.
+      // If sources are half width, then concat and use VPERMV with adjusted
+      // mask.
       SDValue Ops[2];
+      MVT HalfVT = VT.getHalfNumVectorElementsVT();
       if (sd_match(V1,
                    m_InsertSubvector(m_Undef(), m_Value(Ops[0]), m_Zero())) &&
           sd_match(V2,
                    m_InsertSubvector(m_Undef(), m_Value(Ops[1]), m_Zero())) &&
-          (Ops[0].getValueSizeInBits() % VT.getScalarSizeInBits()) == 0 &&
-          Ops[0].getValueType() == Ops[1].getValueType() &&
-          Ops[0].getValueType().isSimple()) {
-        MVT SubVT = Ops[0].getSimpleValueType();
-        MVT ConcatVT = SubVT.getDoubleNumVectorElementsVT();
-        unsigned NumSubElts = SubVT.getSizeInBits() / VT.getScalarSizeInBits();
+          Ops[0].getValueType() == HalfVT && Ops[1].getValueType() == HalfVT) {
         if (SDValue ConcatSrc =
-                combineConcatVectorOps(DL, ConcatVT, Ops, DAG, Subtarget)) {
-          ConcatSrc = widenSubVector(ConcatSrc, false, Subtarget, DAG, DL,
-                                     VT.getSizeInBits());
+                combineConcatVectorOps(DL, VT, Ops, DAG, Subtarget)) {
           for (int &M : Mask)
-            M = (M < (int)NumElts ? M : (M - (NumElts - NumSubElts)));
-          return lowerShuffleWithPERMV(DL, VT, Mask,
-                                       DAG.getBitcast(VT, ConcatSrc),
+            M = (M < (int)NumElts ? M : (M - (NumElts / 2)));
+          return lowerShuffleWithPERMV(DL, VT, Mask, ConcatSrc,
                                        DAG.getUNDEF(VT), Subtarget, DAG);
         }
       }
@@ -45659,6 +45570,34 @@ bool X86TargetLowering::SimplifyDemandedBitsForTargetNode(
 
     break;
   }
+  case X86ISD::PDEP: {
+    SDValue Op0 = Op.getOperand(0);
+    SDValue Op1 = Op.getOperand(1);
+
+    unsigned DemandedBitsLZ = OriginalDemandedBits.countl_zero();
+    APInt LoMask = APInt::getLowBitsSet(BitWidth, BitWidth - DemandedBitsLZ);
+
+    // If the demanded bits has leading zeroes, we don't demand those from the
+    // mask.
+    if (SimplifyDemandedBits(Op1, LoMask, Known, TLO, Depth + 1))
+      return true;
+
+    // The number of possible 1s in the mask determines the number of LSBs of
+    // operand 0 used. Undemanded bits from the mask don't matter so filter
+    // them before counting.
+    KnownBits Known2;
+    uint64_t Count = (~Known.Zero & LoMask).popcount();
+    APInt DemandedMask(APInt::getLowBitsSet(BitWidth, Count));
+    if (SimplifyDemandedBits(Op0, DemandedMask, Known2, TLO, Depth + 1))
+      return true;
+
+    // Zeroes are retained from the mask, but not ones.
+    Known.One.clearAllBits();
+    // The result will have at least as many trailing zeros as the non-mask
+    // operand since bits can only map to the same or higher bit position.
+    Known.Zero.setLowBits(Known2.countMinTrailingZeros());
+    return false;
+  }
   case X86ISD::VPMADD52L:
   case X86ISD::VPMADD52H: {
     KnownBits KnownOp0, KnownOp1, KnownOp2;
@@ -45825,17 +45764,13 @@ SDValue X86TargetLowering::SimplifyMultipleUseDemandedBitsForTargetNode(
 
 bool X86TargetLowering::isGuaranteedNotToBeUndefOrPoisonForTargetNode(
     SDValue Op, const APInt &DemandedElts, const SelectionDAG &DAG,
-    UndefPoisonKind Kind, unsigned Depth) const {
+    bool PoisonOnly, unsigned Depth) const {
   unsigned NumElts = DemandedElts.getBitWidth();
 
   switch (Op.getOpcode()) {
   case X86ISD::GlobalBaseReg:
   case X86ISD::Wrapper:
   case X86ISD::WrapperRIP:
-  // SETCC/SETCC_CARRY always produces a well-defined result based on
-  // EFLAGS/carry flag.
-  case X86ISD::SETCC:
-  case X86ISD::SETCC_CARRY:
     return true;
   case X86ISD::PACKSS:
   case X86ISD::PACKUS: {
@@ -45844,10 +45779,10 @@ bool X86TargetLowering::isGuaranteedNotToBeUndefOrPoisonForTargetNode(
                         DemandedRHS);
     return (!DemandedLHS ||
             DAG.isGuaranteedNotToBeUndefOrPoison(Op.getOperand(0), DemandedLHS,
-                                                 Kind, Depth + 1)) &&
+                                                 PoisonOnly, Depth + 1)) &&
            (!DemandedRHS ||
             DAG.isGuaranteedNotToBeUndefOrPoison(Op.getOperand(1), DemandedRHS,
-                                                 Kind, Depth + 1));
+                                                 PoisonOnly, Depth + 1));
   }
   case X86ISD::INSERTPS:
   case X86ISD::BLENDI:
@@ -45881,7 +45816,7 @@ bool X86TargetLowering::isGuaranteedNotToBeUndefOrPoisonForTargetNode(
       for (auto Op : enumerate(Ops))
         if (!DemandedSrcElts[Op.index()].isZero() &&
             !DAG.isGuaranteedNotToBeUndefOrPoison(
-                Op.value(), DemandedSrcElts[Op.index()], Kind, Depth + 1))
+                Op.value(), DemandedSrcElts[Op.index()], PoisonOnly, Depth + 1))
           return false;
       return true;
     }
@@ -45892,19 +45827,19 @@ bool X86TargetLowering::isGuaranteedNotToBeUndefOrPoisonForTargetNode(
     MVT SrcVT = Src.getSimpleValueType();
     if (SrcVT.isVector()) {
       APInt DemandedSrc = APInt::getOneBitSet(SrcVT.getVectorNumElements(), 0);
-      return DAG.isGuaranteedNotToBeUndefOrPoison(Src, DemandedSrc, Kind,
+      return DAG.isGuaranteedNotToBeUndefOrPoison(Src, DemandedSrc, PoisonOnly,
                                                   Depth + 1);
     }
-    return DAG.isGuaranteedNotToBeUndefOrPoison(Src, Kind, Depth + 1);
+    return DAG.isGuaranteedNotToBeUndefOrPoison(Src, PoisonOnly, Depth + 1);
   }
   }
   return TargetLowering::isGuaranteedNotToBeUndefOrPoisonForTargetNode(
-      Op, DemandedElts, DAG, Kind, Depth);
+      Op, DemandedElts, DAG, PoisonOnly, Depth);
 }
 
 bool X86TargetLowering::canCreateUndefOrPoisonForTargetNode(
     SDValue Op, const APInt &DemandedElts, const SelectionDAG &DAG,
-    UndefPoisonKind Kind, bool ConsiderFlags, unsigned Depth) const {
+    bool PoisonOnly, bool ConsiderFlags, unsigned Depth) const {
 
   switch (Op.getOpcode()) {
   // SSE bit logic.
@@ -46000,7 +45935,7 @@ bool X86TargetLowering::canCreateUndefOrPoisonForTargetNode(
     }
   }
   return TargetLowering::canCreateUndefOrPoisonForTargetNode(
-      Op, DemandedElts, DAG, Kind, ConsiderFlags, Depth);
+      Op, DemandedElts, DAG, PoisonOnly, ConsiderFlags, Depth);
 }
 
 bool X86TargetLowering::isSplatValueForTargetNode(SDValue Op,
@@ -46332,8 +46267,10 @@ static SDValue combineCastedMaskArithmetic(SDNode *N, SelectionDAG &DAG,
   EVT SrcVT = Op.getValueType();
 
   // Make sure we have a bitcast between mask registers and a scalar type.
-  if (!(SrcVT.isVectorOf(MVT::i1) && DstVT.isScalarInteger()) &&
-      !(DstVT.isVectorOf(MVT::i1) && SrcVT.isScalarInteger()))
+  if (!(SrcVT.isVector() && SrcVT.getVectorElementType() == MVT::i1 &&
+        DstVT.isScalarInteger()) &&
+      !(DstVT.isVector() && DstVT.getVectorElementType() == MVT::i1 &&
+        SrcVT.isScalarInteger()))
     return SDValue();
 
   SDValue LHS, RHS;
@@ -46592,8 +46529,8 @@ static SDValue combineBitcast(SDNode *N, SelectionDAG &DAG,
   } else if (DCI.isAfterLegalizeDAG()) {
     // If we're bitcasting from iX to vXi1, see if the integer originally
     // began as a vXi1 and whether we can remove the bitcast entirely.
-    if (VT.isVectorOf(MVT::i1) && SrcVT.isScalarInteger() &&
-        TLI.isTypeLegal(VT)) {
+    if (VT.isVector() && VT.getScalarType() == MVT::i1 &&
+        SrcVT.isScalarInteger() && TLI.isTypeLegal(VT)) {
       if (SDValue V =
               combineBitcastToBoolVector(VT, N0, SDLoc(N), DAG, Subtarget))
         return V;
@@ -46722,7 +46659,7 @@ static SDValue combineBitcast(SDNode *N, SelectionDAG &DAG,
   // Try to remove a bitcast of constant vXi1 vector. We have to legalize
   // most of these to scalar anyway.
   if (Subtarget.hasAVX512() && VT.isScalarInteger() &&
-      SrcVT.isVectorOf(MVT::i1) &&
+      SrcVT.isVector() && SrcVT.getVectorElementType() == MVT::i1 &&
       ISD::isBuildVectorOfConstantSDNodes(N0.getNode())) {
     return combinevXi1ConstantToInteger(N0, DAG);
   }
@@ -46741,7 +46678,8 @@ static SDValue combineBitcast(SDNode *N, SelectionDAG &DAG,
   // Turn it into a sign bit compare that produces a k-register. This avoids
   // a trip through a GPR.
   if (Subtarget.hasAVX512() && SrcVT.isScalarInteger() &&
-      VT.isVectorOf(MVT::i1) && isPowerOf2_32(VT.getVectorNumElements())) {
+      VT.isVector() && VT.getVectorElementType() == MVT::i1 &&
+      isPowerOf2_32(VT.getVectorNumElements())) {
     unsigned NumElts = VT.getVectorNumElements();
     SDValue Src = N0;
 
@@ -46940,6 +46878,81 @@ static SDValue createPSADBW(SelectionDAG &DAG, SDValue N0, SDValue N1,
                           PSADBWBuilder);
 }
 
+// Attempt to replace an min/max v8i16/v16i8 horizontal reduction with
+// PHMINPOSUW.
+static SDValue combineMinMaxReduction(SDNode *Extract, SelectionDAG &DAG,
+                                      const X86Subtarget &Subtarget) {
+  // Bail without SSE41.
+  if (!Subtarget.hasSSE41())
+    return SDValue();
+
+  EVT ExtractVT = Extract->getValueType(0);
+  if (ExtractVT != MVT::i16 && ExtractVT != MVT::i8)
+    return SDValue();
+
+  // Check for SMAX/SMIN/UMAX/UMIN horizontal reduction patterns.
+  ISD::NodeType BinOp;
+  SDValue Src = DAG.matchBinOpReduction(
+      Extract, BinOp, {ISD::SMAX, ISD::SMIN, ISD::UMAX, ISD::UMIN}, true);
+  if (!Src)
+    return SDValue();
+
+  EVT SrcVT = Src.getValueType();
+  EVT SrcSVT = SrcVT.getScalarType();
+  if (SrcSVT != ExtractVT || (SrcVT.getSizeInBits() % 128) != 0)
+    return SDValue();
+
+  SDLoc DL(Extract);
+  SDValue MinPos = Src;
+
+  // First, reduce the source down to 128-bit, applying BinOp to lo/hi.
+  while (SrcVT.getSizeInBits() > 128) {
+    SDValue Lo, Hi;
+    std::tie(Lo, Hi) = splitVector(MinPos, DAG, DL);
+    SrcVT = Lo.getValueType();
+    MinPos = DAG.getNode(BinOp, DL, SrcVT, Lo, Hi);
+  }
+  assert(((SrcVT == MVT::v8i16 && ExtractVT == MVT::i16) ||
+          (SrcVT == MVT::v16i8 && ExtractVT == MVT::i8)) &&
+         "Unexpected value type");
+
+  // PHMINPOSUW applies to UMIN(v8i16), for SMIN/SMAX/UMAX we must apply a mask
+  // to flip the value accordingly.
+  SDValue Mask;
+  unsigned MaskEltsBits = ExtractVT.getSizeInBits();
+  if (BinOp == ISD::SMAX)
+    Mask = DAG.getConstant(APInt::getSignedMaxValue(MaskEltsBits), DL, SrcVT);
+  else if (BinOp == ISD::SMIN)
+    Mask = DAG.getConstant(APInt::getSignedMinValue(MaskEltsBits), DL, SrcVT);
+  else if (BinOp == ISD::UMAX)
+    Mask = DAG.getAllOnesConstant(DL, SrcVT);
+
+  if (Mask)
+    MinPos = DAG.getNode(ISD::XOR, DL, SrcVT, Mask, MinPos);
+
+  // For v16i8 cases we need to perform UMIN on pairs of byte elements,
+  // shuffling each upper element down and insert zeros. This means that the
+  // v16i8 UMIN will leave the upper element as zero, performing zero-extension
+  // ready for the PHMINPOS.
+  if (ExtractVT == MVT::i8) {
+    SDValue Upper = DAG.getVectorShuffle(
+        SrcVT, DL, MinPos, DAG.getConstant(0, DL, MVT::v16i8),
+        {1, 16, 3, 16, 5, 16, 7, 16, 9, 16, 11, 16, 13, 16, 15, 16});
+    MinPos = DAG.getNode(ISD::UMIN, DL, SrcVT, MinPos, Upper);
+  }
+
+  // Perform the PHMINPOS on a v8i16 vector,
+  MinPos = DAG.getBitcast(MVT::v8i16, MinPos);
+  MinPos = DAG.getNode(X86ISD::PHMINPOS, DL, MVT::v8i16, MinPos);
+  MinPos = DAG.getBitcast(SrcVT, MinPos);
+
+  if (Mask)
+    MinPos = DAG.getNode(ISD::XOR, DL, SrcVT, Mask, MinPos);
+
+  return DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, ExtractVT, MinPos,
+                     DAG.getVectorIdxConstant(0, DL));
+}
+
 // Attempt to replace an all_of/any_of/parity style horizontal reduction with a MOVMSK.
 static SDValue combinePredicateReduction(SDNode *Extract, SelectionDAG &DAG,
                                          const X86Subtarget &Subtarget) {
@@ -47597,8 +47610,8 @@ static SDValue combineArithReduction(SDNode *ExtElt, SelectionDAG &DAG,
     return SDValue();
 
   ISD::NodeType Opc;
-  SDValue Rdx =
-      DAG.matchBinOpReduction(ExtElt, Opc, {ISD::ADD, ISD::FADD}, true);
+  SDValue Rdx = DAG.matchBinOpReduction(ExtElt, Opc,
+                                        {ISD::ADD, ISD::MUL, ISD::FADD}, true);
   if (!Rdx)
     return SDValue();
 
@@ -47615,10 +47628,10 @@ static SDValue combineArithReduction(SDNode *ExtElt, SelectionDAG &DAG,
   unsigned NumElts = VecVT.getVectorNumElements();
   unsigned EltSizeInBits = VecVT.getScalarSizeInBits();
 
-  // ZeroExtend v4i8/v8i8 vector to v16i8, with undef upper 64-bits.
-  auto WidenToV16I8 = [&](SDValue V) {
+  // Extend v4i8/v8i8 vector to v16i8, with undef upper 64-bits.
+  auto WidenToV16I8 = [&](SDValue V, bool ZeroExtend) {
     if (V.getValueType() == MVT::v4i8) {
-      if (Subtarget.hasSSE41()) {
+      if (ZeroExtend && Subtarget.hasSSE41()) {
         V = DAG.getNode(ISD::INSERT_VECTOR_ELT, DL, MVT::v4i32,
                         DAG.getConstant(0, DL, MVT::v4i32),
                         DAG.getBitcast(MVT::i32, V),
@@ -47626,15 +47639,50 @@ static SDValue combineArithReduction(SDNode *ExtElt, SelectionDAG &DAG,
         return DAG.getBitcast(MVT::v16i8, V);
       }
       V = DAG.getNode(ISD::CONCAT_VECTORS, DL, MVT::v8i8, V,
-                      DAG.getConstant(0, DL, MVT::v4i8));
+                      ZeroExtend ? DAG.getConstant(0, DL, MVT::v4i8)
+                                 : DAG.getUNDEF(MVT::v4i8));
     }
     return DAG.getNode(ISD::CONCAT_VECTORS, DL, MVT::v16i8, V,
                        DAG.getUNDEF(MVT::v8i8));
   };
 
+  // vXi8 mul reduction - promote to vXi16 mul reduction.
+  if (Opc == ISD::MUL) {
+    if (VT != MVT::i8 || NumElts < 4 || !isPowerOf2_32(NumElts))
+      return SDValue();
+    if (VecVT.getSizeInBits() >= 128) {
+      EVT WideVT = EVT::getVectorVT(*DAG.getContext(), MVT::i16, NumElts / 2);
+      SDValue Lo = getUnpackl(DAG, DL, VecVT, Rdx, DAG.getUNDEF(VecVT));
+      SDValue Hi = getUnpackh(DAG, DL, VecVT, Rdx, DAG.getUNDEF(VecVT));
+      Lo = DAG.getBitcast(WideVT, Lo);
+      Hi = DAG.getBitcast(WideVT, Hi);
+      Rdx = DAG.getNode(Opc, DL, WideVT, Lo, Hi);
+      while (Rdx.getValueSizeInBits() > 128) {
+        std::tie(Lo, Hi) = splitVector(Rdx, DAG, DL);
+        Rdx = DAG.getNode(Opc, DL, Lo.getValueType(), Lo, Hi);
+      }
+    } else {
+      Rdx = WidenToV16I8(Rdx, false);
+      Rdx = getUnpackl(DAG, DL, MVT::v16i8, Rdx, DAG.getUNDEF(MVT::v16i8));
+      Rdx = DAG.getBitcast(MVT::v8i16, Rdx);
+    }
+    if (NumElts >= 8)
+      Rdx = DAG.getNode(Opc, DL, MVT::v8i16, Rdx,
+                        DAG.getVectorShuffle(MVT::v8i16, DL, Rdx, Rdx,
+                                             {4, 5, 6, 7, -1, -1, -1, -1}));
+    Rdx = DAG.getNode(Opc, DL, MVT::v8i16, Rdx,
+                      DAG.getVectorShuffle(MVT::v8i16, DL, Rdx, Rdx,
+                                           {2, 3, -1, -1, -1, -1, -1, -1}));
+    Rdx = DAG.getNode(Opc, DL, MVT::v8i16, Rdx,
+                      DAG.getVectorShuffle(MVT::v8i16, DL, Rdx, Rdx,
+                                           {1, -1, -1, -1, -1, -1, -1, -1}));
+    Rdx = DAG.getBitcast(MVT::v16i8, Rdx);
+    return DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, VT, Rdx, Index);
+  }
+
   // vXi8 add reduction - sub 128-bit vector.
   if (VecVT == MVT::v4i8 || VecVT == MVT::v8i8) {
-    Rdx = WidenToV16I8(Rdx);
+    Rdx = WidenToV16I8(Rdx, true);
     Rdx = DAG.getNode(X86ISD::PSADBW, DL, MVT::v2i64, Rdx,
                       DAG.getConstant(0, DL, MVT::v16i8));
     Rdx = DAG.getBitcast(MVT::v16i8, Rdx);
@@ -47680,7 +47728,7 @@ static SDValue combineArithReduction(SDNode *ExtElt, SelectionDAG &DAG,
       EVT ByteVT = VecVT.changeVectorElementType(*DAG.getContext(), MVT::i8);
       Rdx = DAG.getNode(ISD::TRUNCATE, DL, ByteVT, Rdx);
       if (ByteVT.getSizeInBits() < 128)
-        Rdx = WidenToV16I8(Rdx);
+        Rdx = WidenToV16I8(Rdx, true);
     }
 
     // Build the PSADBW, split as 128/256/512 bits for SSE/AVX2/AVX512BW.
@@ -47818,17 +47866,6 @@ static SDValue combineExtractVectorElt(SDNode *N, SelectionDAG &DAG,
     return SDValue();
   }
 
-  // Attempt to avoid multi-use src if we don't need anything from it.
-  // TODO: Generlize this and move to DAGCombine.
-  if (CIdx && ISD::isBitwiseLogicOp(InputVector.getOpcode())) {
-    unsigned Idx = CIdx->getZExtValue();
-    APInt DemandedElts = APInt::getOneBitSet(NumSrcElts, Idx);
-    if (SDValue NewVector = TLI.SimplifyMultipleUseDemandedVectorElts(
-            InputVector, DemandedElts, DAG))
-      if (NewVector.getOpcode() == ISD::BUILD_VECTOR)
-        return DAG.getNode(N->getOpcode(), dl, VT, NewVector, EltIdx);
-  }
-
   // Detect mmx extraction of all bits as a i64. It works better as a bitcast.
   if (VT == MVT::i64 && SrcVT == MVT::v1i64 &&
       InputVector.getOpcode() == ISD::BITCAST &&
@@ -47857,7 +47894,11 @@ static SDValue combineExtractVectorElt(SDNode *N, SelectionDAG &DAG,
   if (SDValue Cmp = combinePredicateReduction(N, DAG, Subtarget))
     return Cmp;
 
-  // Attempt to optimize ADD/FADD reductions with HADD, promotion etc..
+  // Attempt to replace min/max v8i16/v16i8 reductions with PHMINPOSUW.
+  if (SDValue MinMax = combineMinMaxReduction(N, DAG, Subtarget))
+    return MinMax;
+
+  // Attempt to optimize ADD/FADD/MUL reductions with HADD, promotion etc..
   if (SDValue V = combineArithReduction(N, DAG, Subtarget))
     return V;
 
@@ -47946,38 +47987,6 @@ static SDValue combineExtractVectorElt(SDNode *N, SelectionDAG &DAG,
   return SDValue();
 }
 
-static SDValue combineVECREDUCE_MUL(SDNode *N, SelectionDAG &DAG,
-                                    const X86Subtarget &Subtarget) {
-  SDValue Src = N->getOperand(0);
-  EVT SrcVT = Src.getValueType();
-  unsigned NumElts = SrcVT.getVectorNumElements();
-  SDLoc DL(N);
-
-  // vXi8 mul reduction - promote to vXi16 mul reduction.
-  if (!isPowerOf2_32(NumElts) || SrcVT.getScalarType() != MVT::i8)
-    return SDValue();
-
-  // Early out for v2i8 - no promotion necessary.
-  if (NumElts == 2) {
-    SDValue Hi = DAG.getVectorShuffle(SrcVT, DL, Src, Src, {1, -1});
-    SDValue Rdx = DAG.getNode(ISD::MUL, DL, SrcVT, Src, Hi);
-    return DAG.getExtractVectorElt(DL, N->getValueType(0), Rdx, 0);
-  }
-
-  if (SrcVT.getSizeInBits() >= 128) {
-    EVT WideVT = EVT::getVectorVT(*DAG.getContext(), MVT::i16, NumElts / 2);
-    SDValue Lo = getUnpackl(DAG, DL, SrcVT, Src, DAG.getUNDEF(SrcVT));
-    SDValue Hi = getUnpackh(DAG, DL, SrcVT, Src, DAG.getUNDEF(SrcVT));
-    Src = DAG.getNode(ISD::MUL, DL, WideVT, DAG.getBitcast(WideVT, Lo),
-                      DAG.getBitcast(WideVT, Hi));
-  } else {
-    EVT ExtVT = SrcVT.changeVectorElementType(*DAG.getContext(), MVT::i16);
-    Src = DAG.getNode(ISD::ANY_EXTEND, DL, ExtVT, Src);
-  }
-  Src = DAG.getNode(ISD::VECREDUCE_MUL, DL, MVT::i16, Src);
-  return DAG.getZExtOrTrunc(Src, DL, N->getValueType(0));
-}
-
 // Convert (vXiY *ext(vXi1 bitcast(iX))) to extend_in_reg(broadcast(iX)).
 // This is more or less the reverse of combineBitcastvxi1.
 static SDValue combineToExtendBoolVectorInReg(
@@ -48405,8 +48414,8 @@ static SDValue combineSelectToMinMax(SelectionDAG &DAG,
   case ISD::SETOLE:
     // Converting this to a min would handle comparisons between positive
     // and negative zero incorrectly.
-    if (!N->getFlags().hasNoSignedZeros() &&
-        !DAG.isKnownNeverLogicalZero(LHS) && !DAG.isKnownNeverLogicalZero(RHS))
+    if (!N->getFlags().hasNoSignedZeros() && !DAG.isKnownNeverZeroFloat(LHS) &&
+        !DAG.isKnownNeverZeroFloat(RHS))
       break;
     Opcode = X86ISD::FMIN;
     break;
@@ -48422,8 +48431,8 @@ static SDValue combineSelectToMinMax(SelectionDAG &DAG,
   case ISD::SETOGE:
     // Converting this to a max would handle comparisons between positive
     // and negative zero incorrectly.
-    if (!N->getFlags().hasNoSignedZeros() &&
-        !DAG.isKnownNeverLogicalZero(LHS) && !DAG.isKnownNeverLogicalZero(RHS))
+    if (!N->getFlags().hasNoSignedZeros() && !DAG.isKnownNeverZeroFloat(LHS) &&
+        !DAG.isKnownNeverZeroFloat(RHS))
       break;
     Opcode = X86ISD::FMAX;
     break;
@@ -48582,15 +48591,6 @@ static SDValue combineSelect(SDNode *N, SelectionDAG &DAG,
       CondVT.getVectorElementType() == MVT::i1 &&
       (VT.getVectorElementType() == MVT::i8 ||
        VT.getVectorElementType() == MVT::i16)) {
-    // Handle AVX512F masked trunc patterns, which do have vXi8/vXi16 selects.
-    if (LHS.getOpcode() == ISD::TRUNCATE) {
-      SDValue TruncSrc = LHS.getOperand(0);
-      EVT TruncSrcVT = TruncSrc.getValueType();
-      if ((VT == MVT::v8i16 && TruncSrcVT == MVT::v8i64) ||
-          (VT == MVT::v16i8 && TruncSrcVT == MVT::v16i32) ||
-          (VT == MVT::v16i16 && TruncSrcVT == MVT::v16i32))
-        return DAG.getNode(X86ISD::VMTRUNC, DL, VT, TruncSrc, RHS, Cond);
-    }
     Cond = DAG.getNode(ISD::SIGN_EXTEND, DL, VT, Cond);
     return DAG.getNode(N->getOpcode(), DL, VT, Cond, LHS, RHS);
   }
@@ -48778,9 +48778,7 @@ static SDValue combineSelect(SDNode *N, SelectionDAG &DAG,
     return V;
 
   // select(~Cond, X, Y) -> select(Cond, Y, X)
-  // This is only valid for vector selects which use an all-bits mask semantic.
-  // For scalar selects, ~Cond != 0 is not equivalent to Cond == 0.
-  if (CondVT.isVector() && CondVT.getScalarType() != MVT::i1) {
+  if (CondVT.getScalarType() != MVT::i1) {
     if (SDValue CondNot = IsNOT(Cond, DAG))
       return DAG.getNode(N->getOpcode(), DL, VT,
                          DAG.getBitcast(CondVT, CondNot), RHS, LHS);
@@ -50008,7 +50006,7 @@ static SDValue combineCMov(SDNode *N, SelectionDAG &DAG,
     // Ok, now make sure that Add is (add (cttz X), C2) and Const is a constant.
     if (isa<ConstantSDNode>(Const) && Add.getOpcode() == ISD::ADD &&
         Add.hasOneUse() && isa<ConstantSDNode>(Add.getOperand(1)) &&
-        (Add.getOperand(0).getOpcode() == ISD::CTTZ_ZERO_POISON ||
+        (Add.getOperand(0).getOpcode() == ISD::CTTZ_ZERO_UNDEF ||
          Add.getOperand(0).getOpcode() == ISD::CTTZ) &&
         Add.getOperand(0).getOperand(0) == Cond.getOperand(0)) {
       // This should constant fold.
@@ -50258,7 +50256,7 @@ static SDValue combineMulToPMADDWD(SDNode *N, const SDLoc &DL,
   EVT VT = N->getValueType(0);
 
   // Only support vXi32 vectors.
-  if (!VT.isVectorOf(MVT::i32))
+  if (!VT.isVector() || VT.getVectorElementType() != MVT::i32)
     return SDValue();
 
   // Make sure the type is legal or can split/widen to a legal type.
@@ -50313,7 +50311,7 @@ static SDValue combineMulToPMADDWD(SDNode *N, const SDLoc &DL,
     if (Op.getOpcode() == ISD::SIGN_EXTEND && N->isOnlyUserOf(Op.getNode())) {
       SDValue Src = Op.getOperand(0);
       // Convert sext(vXi16) to zext(vXi16).
-      if (Src.getScalarValueSizeInBits() == 16)
+      if (Src.getScalarValueSizeInBits() == 16 && VT.getSizeInBits() <= 128)
         return DAG.getNode(ISD::ZERO_EXTEND, DL, VT, Src);
       // Convert sext(vXi8) to zext(vXi16 sext(vXi8)) on pre-SSE41 targets
       // which will expand the extension.
@@ -50365,7 +50363,8 @@ static SDValue combineMulToPMULDQ(SDNode *N, const SDLoc &DL, SelectionDAG &DAG,
   EVT VT = N->getValueType(0);
 
   // Only support vXi64 vectors.
-  if (!VT.isVectorOf(MVT::i64) || VT.getVectorNumElements() < 2 ||
+  if (!VT.isVector() || VT.getVectorElementType() != MVT::i64 ||
+      VT.getVectorNumElements() < 2 ||
       !isPowerOf2_32(VT.getVectorNumElements()))
     return SDValue();
 
@@ -50410,8 +50409,9 @@ static SDValue combineMulToPMADD52(SDNode *N, const SDLoc &DL,
   // 128/256-bit vectors (v2i64/v4i64) require either AVX512-IFMA + VLX, or
   // AVX-IFMA.
   bool Supported512 = (VT == MVT::v8i64) && Subtarget.hasIFMA();
-  bool SupportedSmall = (VT == MVT::v2i64 || VT == MVT::v4i64) &&
-                        (Subtarget.hasIFMA() || Subtarget.hasAVXIFMA());
+  bool SupportedSmall =
+      (VT == MVT::v2i64 || VT == MVT::v4i64) &&
+      ((Subtarget.hasIFMA() && Subtarget.hasVLX()) || Subtarget.hasAVXIFMA());
 
   if (!Supported512 && !SupportedSmall)
     return SDValue();
@@ -51398,9 +51398,6 @@ static SDValue combineVectorInsert(SDNode *N, SelectionDAG &DAG,
                                    const X86Subtarget &Subtarget) {
   EVT VT = N->getValueType(0);
   unsigned Opcode = N->getOpcode();
-  unsigned NumElts = VT.getVectorNumElements();
-  unsigned NumBitsPerElt = VT.getScalarSizeInBits();
-  const TargetLowering &TLI = DAG.getTargetLoweringInfo();
   assert(((Opcode == X86ISD::PINSRB && VT == MVT::v16i8) ||
           (Opcode == X86ISD::PINSRW && VT == MVT::v8i16) ||
           Opcode == ISD::INSERT_VECTOR_ELT) &&
@@ -51410,48 +51407,13 @@ static SDValue combineVectorInsert(SDNode *N, SelectionDAG &DAG,
   SDValue Scl = N->getOperand(1);
   SDValue Idx = N->getOperand(2);
 
-  if (Opcode == ISD::INSERT_VECTOR_ELT) {
-    auto *CIdx = dyn_cast<ConstantSDNode>(Idx);
-    if (CIdx && CIdx->getAPIntValue().uge(NumElts))
-      return DAG.getPOISON(VT);
-
-    // Fold insert_vector_elt(undef, elt, 0) --> scalar_to_vector(elt).
-    if (Vec.isUndef() && isNullConstant(Idx))
-      return DAG.getNode(ISD::SCALAR_TO_VECTOR, SDLoc(N), VT, Scl);
-
-    // Attempt to fold neighboring pairs of inserted constants into a single
-    // large constant insertion.
-    // TODO: Consecutive loads might benefit as well?
-    if (!DCI.isBeforeLegalize() && VT.isInteger() && CIdx &&
-        (CIdx->getZExtValue() & 1) == 1 && TLI.isTypeLegal(VT)) {
-      using namespace SDPatternMatch;
-      auto *Cst = dyn_cast<ConstantSDNode>(Scl);
-      MVT WideSVT = MVT::getIntegerVT(2 * NumBitsPerElt);
-      MVT WideVT = MVT::getVectorVT(WideSVT, NumElts / 2);
-      SDValue InnerVec, InnerScl;
-      if (Cst && TLI.isTypeLegal(WideSVT) && TLI.isTypeLegal(WideVT) &&
-          sd_match(Vec, m_OneUse(m_InsertElt(
-                            m_Value(InnerVec), m_Value(InnerScl),
-                            m_SpecificInt(CIdx->getZExtValue() - 1))))) {
-        if (isa<ConstantSDNode>(InnerScl)) {
-          SDLoc DL(N);
-          SDValue Lo = DAG.getNode(ISD::ZERO_EXTEND, DL, WideSVT, InnerScl);
-          SDValue Hi = DAG.getNode(ISD::ZERO_EXTEND, DL, WideSVT, Scl);
-          Lo = DAG.getZeroExtendInReg(Lo, DL, VT.getScalarType());
-          Hi = DAG.getNode(
-              ISD::SHL, DL, WideSVT, Hi,
-              DAG.getShiftAmountConstant(NumBitsPerElt, WideSVT, DL));
-          unsigned NewIdx = (CIdx->getZExtValue() - 1) / 2;
-          SDValue NewInsert = DAG.getInsertVectorElt(
-              DL, DAG.getBitcast(WideVT, InnerVec),
-              DAG.getNode(ISD::OR, DL, WideSVT, Lo, Hi), NewIdx);
-          return DAG.getBitcast(VT, NewInsert);
-        }
-      }
-    }
-  }
+  // Fold insert_vector_elt(undef, elt, 0) --> scalar_to_vector(elt).
+  if (Opcode == ISD::INSERT_VECTOR_ELT && Vec.isUndef() && isNullConstant(Idx))
+    return DAG.getNode(ISD::SCALAR_TO_VECTOR, SDLoc(N), VT, Scl);
 
   if (Opcode == X86ISD::PINSRB || Opcode == X86ISD::PINSRW) {
+    unsigned NumBitsPerElt = VT.getScalarSizeInBits();
+    const TargetLowering &TLI = DAG.getTargetLoweringInfo();
     if (TLI.SimplifyDemandedBits(SDValue(N, 0),
                                  APInt::getAllOnes(NumBitsPerElt), DCI))
       return SDValue(N, 0);
@@ -52024,9 +51986,9 @@ static SDValue combineAndMaskToShift(SDNode *N, const SDLoc &DL,
   if (EltBitWidth != DAG.ComputeNumSignBits(Op0))
     return SDValue();
 
-  unsigned ShiftVal = EltBitWidth - SplatVal.countr_one();
-  SDValue Shift = getTargetVShiftByConstNode(
-      X86ISD::VSRLI, DL, VT.getSimpleVT(), Op0, ShiftVal, DAG);
+  unsigned ShiftVal = SplatVal.countr_one();
+  SDValue ShAmt = DAG.getTargetConstant(EltBitWidth - ShiftVal, DL, MVT::i8);
+  SDValue Shift = DAG.getNode(X86ISD::VSRLI, DL, VT, Op0, ShAmt);
   return DAG.getBitcast(N->getValueType(0), Shift);
 }
 
@@ -52076,41 +52038,6 @@ static SDValue combineAndNotOrIntoAndNotAnd(SDNode *N, const SDLoc &DL,
   return SDValue();
 }
 
-// Fold vXi1 logicop(truncate(N0),truncate(N1)) -> truncate(logicop(X,Y))
-// Generic vector logicops are always quicker than predicate equivalents.
-static SDValue combineMaskBitOp(SDNode *N, const SDLoc &DL, SelectionDAG &DAG) {
-  using namespace SDPatternMatch;
-  unsigned Opc = N->getOpcode();
-  assert(ISD::isBitwiseLogicOp(Opc) && "Unexpected opcode!");
-  EVT VT = N->getValueType(0);
-  const TargetLowering &TLI = DAG.getTargetLoweringInfo();
-
-  if (!VT.isVector() || VT.getScalarType() != MVT::i1 || !TLI.isTypeLegal(VT))
-    return SDValue();
-
-  SDValue Src0, Src1;
-  if (sd_match(
-          N, m_BitwiseLogic(m_Trunc(m_Value(Src0)), m_Trunc(m_Value(Src1))))) {
-    EVT SrcVT = Src0.getValueType();
-    if (SrcVT == Src1.getValueType() && TLI.isOperationLegal(Opc, SrcVT)) {
-      return DAG.getNode(ISD::TRUNCATE, DL, VT,
-                         DAG.getNode(Opc, DL, SrcVT, Src0, Src1));
-    }
-  }
-
-  // Attempt to match expanded ANDNOT pattern (if AND is legal then ANDNP is).
-  if (sd_match(N, m_And(m_OneUse(m_Not(m_Trunc(m_Value(Src0)))),
-                        m_Trunc(m_Value(Src1))))) {
-    EVT SrcVT = Src0.getValueType();
-    if (SrcVT == Src1.getValueType() && TLI.isOperationLegal(ISD::AND, SrcVT)) {
-      return DAG.getNode(ISD::TRUNCATE, DL, VT,
-                         DAG.getNode(X86ISD::ANDNP, DL, SrcVT, Src0, Src1));
-    }
-  }
-
-  return SDValue();
-}
-
 // This function recognizes cases where X86 bzhi instruction can replace and
 // 'and-load' sequence.
 // In case of loading integer value from an array of constants which is defined
@@ -52229,7 +52156,8 @@ static SDValue combineScalarAndWithMaskSetcc(SDNode *N, SelectionDAG &DAG,
   EVT SrcVT = Src.getValueType();
 
   const TargetLowering &TLI = DAG.getTargetLoweringInfo();
-  if (!SrcVT.isVectorOf(MVT::i1) || !TLI.isTypeLegal(SrcVT))
+  if (!SrcVT.isVector() || SrcVT.getVectorElementType() != MVT::i1 ||
+      !TLI.isTypeLegal(SrcVT))
     return SDValue();
 
   if (Src.getOpcode() != ISD::CONCAT_VECTORS)
@@ -52597,9 +52525,6 @@ static SDValue combineAnd(SDNode *N, SelectionDAG &DAG,
   if (SDValue V = combineScalarAndWithMaskSetcc(N, DAG, Subtarget))
     return V;
 
-  if (SDValue R = combineMaskBitOp(N, dl, DAG))
-    return R;
-
   if (SDValue R = combineBitOpWithMOVMSK(N->getOpcode(), dl, N0, N1, DAG))
     return R;
 
@@ -53267,70 +53192,6 @@ static SDValue combineAddOrSubToADCOrSBB(SDNode *N, const SDLoc &DL,
   return SDValue();
 }
 
-/// GF2P8AFFINEQB computes each output bit from one row of the
-/// 8x8 affine matrix, then xors the corresponding immediate bit.
-/// OR-ing the result with a constant byte splat forces selected output
-/// bits to 1, so the original affine result for those bits is irrelevant.
-///
-/// Fold:
-///   gf2p8affineqb(X, Matrix, Imm) | Mask
-/// into:
-///   gf2p8affineqb(X, NewMatrix, Imm | Mask)
-///
-static SDValue combineOrWithGF2P8AFFINEQB(SDNode *N, const SDLoc &DL,
-                                          SelectionDAG &DAG, EVT VT) {
-  using namespace SDPatternMatch;
-  assert(N->getOpcode() == ISD::OR && "Expected OR node");
-
-  SDValue LHS = N->getOperand(0), RHS = N->getOperand(1);
-
-  SDValue X, Matrix, SplatOp;
-  APInt Imm;
-  if (!sd_match(N, m_Or(m_OneUse(m_TernaryOp(X86ISD::GF2P8AFFINEQB, m_Value(X),
-                                             m_Value(Matrix), m_ConstInt(Imm))),
-                        m_Value(SplatOp))))
-    return SDValue();
-
-  APInt SplatVal;
-  if (!X86::isConstantSplat(SplatOp, SplatVal, /*AllowPartialUndefs=*/false))
-    return SDValue();
-
-  if (N->getFlags().hasDisjoint() || DAG.haveNoCommonBitsSet(LHS, RHS)) {
-    // Fold: (GF2P8AFFINEQB(X, Matrix, Imm) or_disjoint SplatVal)
-    //     -> GF2P8AFFINEQB(X, Matrix, Imm ^ SplatVal)
-    // When OR is disjoint (no common bits), the splat constant can be folded
-    // directly into the GF2P8AFFINEQB immediate via XOR.
-    uint64_t NewImm = (Imm.getZExtValue() ^ SplatVal.getZExtValue()) & 0xFF;
-    return DAG.getNode(X86ISD::GF2P8AFFINEQB, DL, VT, X, Matrix,
-                       DAG.getTargetConstant(NewImm, DL, MVT::i8));
-  }
-
-  APInt UndefElts;
-  SmallVector<APInt, 16> OldMatrix;
-  if (!getTargetConstantBitsFromNode(Matrix, 8, UndefElts, OldMatrix,
-                                     /*AllowWholeUndefs=*/false,
-                                     /*AllowPartialUndefs=*/false))
-    return SDValue();
-
-  uint8_t Mask8 = SplatVal.getZExtValue() & 0xFF;
-  uint8_t NewImm = Imm.getZExtValue() | Mask8;
-  // For each output bit selected by Mask, clear the corresponding matrix row
-  // and set the same bit in the immediate. Rows are encoded in reverse bit
-  // order within each byte: output bit B corresponds to row 7 - B.
-  SmallVector<SDValue, 64> NewMatrixOps;
-  for (unsigned I = 0, E = VT.getVectorNumElements(); I != E; ++I) {
-    unsigned OutBit = 7 - (I & 7);
-    uint8_t OldRow = OldMatrix[I].getZExtValue();
-    uint8_t NewRow = ((Mask8 >> OutBit) & 1) ? 0x00 : OldRow;
-    NewMatrixOps.push_back(DAG.getConstant(NewRow, DL, MVT::i8));
-  }
-
-  SDValue NewMatrix = DAG.getBuildVector(VT, DL, NewMatrixOps);
-  SDValue NewGF = DAG.getNode(X86ISD::GF2P8AFFINEQB, DL, VT, X, NewMatrix,
-                              DAG.getTargetConstant(NewImm, DL, MVT::i8));
-  return NewGF;
-}
-
 static SDValue combineOrXorWithSETCC(unsigned Opc, const SDLoc &DL, EVT VT,
                                      SDValue N0, SDValue N1,
                                      SelectionDAG &DAG) {
@@ -53410,9 +53271,6 @@ static SDValue combineOr(SDNode *N, SelectionDAG &DAG,
   if (SDValue SetCC = combineAndOrForCcmpCtest(N, DAG, DCI, Subtarget))
     return SetCC;
 
-  if (SDValue R = combineMaskBitOp(N, dl, DAG))
-    return R;
-
   if (SDValue R = combineBitOpWithMOVMSK(N->getOpcode(), dl, N0, N1, DAG))
     return R;
 
@@ -53426,9 +53284,6 @@ static SDValue combineOr(SDNode *N, SelectionDAG &DAG,
                                                  DAG, DCI, Subtarget))
     return FPLogic;
 
-  if (SDValue R = combineOrWithGF2P8AFFINEQB(N, dl, DAG, VT))
-    return R;
-
   if (DCI.isBeforeLegalizeOps())
     return SDValue();
 
@@ -53541,9 +53396,6 @@ static SDValue combineOr(SDNode *N, SelectionDAG &DAG,
   if (SDValue R = combineOrXorWithSETCC(N->getOpcode(), dl, VT, N0, N1, DAG))
     return R;
 
-  if (SDValue R = combineOrWithGF2P8AFFINEQB(N, dl, DAG, VT))
-    return R;
-
   return SDValue();
 }
 
@@ -53906,29 +53758,6 @@ static SDValue combineConstantPoolLoads(SDNode *N, const SDLoc &dl,
   return SDValue();
 }
 
-static SDValue combineAtomicLoad(SDNode *N, SelectionDAG &DAG,
-                                 TargetLowering::DAGCombinerInfo &DCI) {
-  if (!DCI.isBeforeLegalize())
-    return SDValue();
-
-  auto *AN = cast<AtomicSDNode>(N);
-  EVT VT = AN->getValueType(0);
-  if (!VT.getScalarType().isFloatingPoint())
-    return SDValue();
-
-  unsigned BitWidth = VT.getStoreSizeInBits();
-  if (BitWidth != VT.getSizeInBits())
-    return SDValue();
-
-  SDLoc DL(N);
-  EVT IntVT = EVT::getIntegerVT(*DAG.getContext(), BitWidth);
-  SDValue IntLoad = DAG.getAtomic(
-      ISD::ATOMIC_LOAD, DL, IntVT, DAG.getVTList(IntVT, MVT::Other),
-      {AN->getChain(), AN->getBasePtr()}, AN->getMemOperand());
-  SDValue Cast = DAG.getBitcast(VT, IntLoad);
-  return DAG.getMergeValues({Cast, IntLoad.getValue(1)}, DL);
-}
-
 static SDValue combineLoad(SDNode *N, SelectionDAG &DAG,
                            TargetLowering::DAGCombinerInfo &DCI,
                            const X86Subtarget &Subtarget) {
@@ -54916,8 +54745,8 @@ static SDValue combineVEXTRACT_STORE(SDNode *N, SelectionDAG &DAG,
 /// A horizontal-op B, for some already available A and B, and if so then LHS is
 /// set to A, RHS to B, and the routine returns 'true'.
 static bool isHorizontalBinOp(unsigned HOpcode, SDValue &LHS, SDValue &RHS,
-                              const SelectionDAG &DAG,
-                              const X86Subtarget &Subtarget, bool IsCommutative,
+                              SelectionDAG &DAG, const X86Subtarget &Subtarget,
+                              bool IsCommutative,
                               SmallVectorImpl<int> &PostShuffleMask,
                               bool ForceHorizOp) {
   // If either operand is undef, bail out. The binop should be simplified.
@@ -54962,11 +54791,8 @@ static bool isHorizontalBinOp(unsigned HOpcode, SDValue &LHS, SDValue &RHS,
         ShuffleMask.assign(ScaledMask.begin(), ScaledMask.end());
       }
       if (UseSubVector && SrcOps.size() == 1 &&
-          scaleShuffleElements(SrcMask, 2 * NumElts, ScaledMask) &&
-          SrcOps[0].getOpcode() == ISD::CONCAT_VECTORS &&
-          SrcOps[0].getNumOperands() == 2) {
-        N0 = SrcOps[0].getOperand(0);
-        N1 = SrcOps[0].getOperand(1);
+          scaleShuffleElements(SrcMask, 2 * NumElts, ScaledMask)) {
+        std::tie(N0, N1) = DAG.SplitVector(SrcOps[0], SDLoc(Op));
         ArrayRef<int> Mask = ArrayRef<int>(ScaledMask).slice(0, NumElts);
         ShuffleMask.assign(Mask.begin(), Mask.end());
       }
@@ -55075,9 +54901,12 @@ static bool isHorizontalBinOp(unsigned HOpcode, SDValue &LHS, SDValue &RHS,
   SDValue NewLHS = A.getNode() ? A : B; // If A is 'UNDEF', use B for it.
   SDValue NewRHS = B.getNode() ? B : A; // If B is 'UNDEF', use A for it.
 
-  // Avoid 128-bit multi lane shuffles if pre-AVX2 and FP (integer will split).
   bool IsIdentityPostShuffle =
       isSequentialOrUndefInRange(PostShuffleMask, 0, NumElts, 0);
+  if (IsIdentityPostShuffle)
+    PostShuffleMask.clear();
+
+  // Avoid 128-bit multi lane shuffles if pre-AVX2 and FP (integer will split).
   if (!IsIdentityPostShuffle && !Subtarget.hasAVX2() && VT.isFloatingPoint() &&
       isMultiLaneShuffleMask(128, VT.getScalarSizeInBits(), PostShuffleMask))
     return false;
@@ -55099,8 +54928,8 @@ static bool isHorizontalBinOp(unsigned HOpcode, SDValue &LHS, SDValue &RHS,
                              DAG, Subtarget))
     return false;
 
-  LHS = NewLHS;
-  RHS = NewRHS;
+  LHS = DAG.getBitcast(VT, NewLHS);
+  RHS = DAG.getBitcast(VT, NewRHS);
   return true;
 }
 
@@ -55121,21 +54950,6 @@ static SDValue combineToHorizontalAddSub(SDNode *N, SelectionDAG &DAG,
             N->user_begin()->getOperand(1).getOpcode() == HorizOpcode);
   };
 
-  auto CanonicalizeHorizOps = [&](const SDLoc &DL, SDValue &LHS, SDValue &RHS) {
-    assert(PostShuffleMask.size() == VT.getVectorNumElements() &&
-           "Illegal shuffle mask");
-    LHS = DAG.getBitcast(VT, LHS);
-    RHS = DAG.getBitcast(VT, RHS);
-    if (VT.is256BitVector() && isUndefUpperHalf(PostShuffleMask) &&
-        !isLaneCrossingShuffleMask(128, VT.getScalarSizeInBits(),
-                                   PostShuffleMask)) {
-      LHS = extract128BitVector(LHS, 0, DAG, DL);
-      RHS = extract128BitVector(RHS, 0, DAG, DL);
-      PostShuffleMask.truncate(PostShuffleMask.size() / 2);
-    }
-    return LHS.getSimpleValueType();
-  };
-
   switch (Opcode) {
   case ISD::FADD:
   case ISD::FSUB:
@@ -55146,12 +54960,11 @@ static SDValue combineToHorizontalAddSub(SDNode *N, SelectionDAG &DAG,
       auto HorizOpcode = IsAdd ? X86ISD::FHADD : X86ISD::FHSUB;
       if (isHorizontalBinOp(HorizOpcode, LHS, RHS, DAG, Subtarget, IsAdd,
                             PostShuffleMask, MergableHorizOp(HorizOpcode))) {
-        SDLoc DL(N);
-        MVT HorizVT = CanonicalizeHorizOps(DL, LHS, RHS);
-        SDValue HorizBinOp = DAG.getNode(HorizOpcode, DL, HorizVT, LHS, RHS);
-        HorizBinOp = DAG.getVectorShuffle(HorizVT, DL, HorizBinOp, HorizBinOp,
-                                          PostShuffleMask);
-        return DAG.getInsertSubvector(DL, DAG.getUNDEF(VT), HorizBinOp, 0);
+        SDValue HorizBinOp = DAG.getNode(HorizOpcode, SDLoc(N), VT, LHS, RHS);
+        if (!PostShuffleMask.empty())
+          HorizBinOp = DAG.getVectorShuffle(VT, SDLoc(HorizBinOp), HorizBinOp,
+                                            DAG.getUNDEF(VT), PostShuffleMask);
+        return HorizBinOp;
       }
     }
     break;
@@ -55163,6 +54976,7 @@ static SDValue combineToHorizontalAddSub(SDNode *N, SelectionDAG &DAG,
       break;
     if (VT == MVT::v8i16 || VT == MVT::v16i16 ||
         (!IsSat && (VT == MVT::v4i32 || VT == MVT::v8i32))) {
+
       SDValue LHS = N->getOperand(0);
       SDValue RHS = N->getOperand(1);
       auto HorizOpcode = IsSat ? (IsAdd ? X86ISD::HADDS : X86ISD::HSUBS)
@@ -55173,13 +54987,12 @@ static SDValue combineToHorizontalAddSub(SDNode *N, SelectionDAG &DAG,
                                         ArrayRef<SDValue> Ops) {
           return DAG.getNode(HorizOpcode, DL, Ops[0].getValueType(), Ops);
         };
-        SDLoc DL(N);
-        MVT HorizVT = CanonicalizeHorizOps(DL, LHS, RHS);
-        SDValue HorizBinOp = SplitOpsAndApply(DAG, Subtarget, DL, HorizVT,
+        SDValue HorizBinOp = SplitOpsAndApply(DAG, Subtarget, SDLoc(N), VT,
                                               {LHS, RHS}, HOpBuilder);
-        HorizBinOp = DAG.getVectorShuffle(HorizVT, DL, HorizBinOp, HorizBinOp,
-                                          PostShuffleMask);
-        return DAG.getInsertSubvector(DL, DAG.getUNDEF(VT), HorizBinOp, 0);
+        if (!PostShuffleMask.empty())
+          HorizBinOp = DAG.getVectorShuffle(VT, SDLoc(HorizBinOp), HorizBinOp,
+                                            DAG.getUNDEF(VT), PostShuffleMask);
+        return HorizBinOp;
       }
     }
     break;
@@ -55493,7 +55306,7 @@ static SDValue combinePMULH(SDValue Src, EVT VT, const SDLoc &DL,
 
   // Only handle vXi16 types that are at least 128-bits unless they will be
   // widened.
-  if (!VT.isVectorOf(MVT::i16))
+  if (!VT.isVector() || VT.getVectorElementType() != MVT::i16)
     return SDValue();
 
   // Input type should be at least vXi32.
@@ -55787,8 +55600,7 @@ static SDValue combineVTRUNC(SDNode *N, SelectionDAG &DAG,
 }
 
 static SDValue combineVTRUNCSAT(SDNode *N, SelectionDAG &DAG,
-                                TargetLowering::DAGCombinerInfo &DCI,
-                                const X86Subtarget &Subtarget) {
+                                TargetLowering::DAGCombinerInfo &DCI) {
   using namespace SDPatternMatch;
   unsigned Opc = N->getOpcode();
   EVT VT = N->getValueType(0);
@@ -55807,17 +55619,7 @@ static SDValue combineVTRUNCSAT(SDNode *N, SelectionDAG &DAG,
       (EltSizeInBits * 2) == Src.getScalarValueSizeInBits() &&
       isFreeToSplitVector(Src, DAG)) {
     SDLoc DL(N);
-    SDValue LHS, RHS;
-    if (Src.getValueSizeInBits() == VT.getSizeInBits()) {
-      assert(VT.is128BitVector() && "128-bit VTRUNC source expected");
-      LHS = Src;
-      RHS = getZeroVector(Src.getSimpleValueType(), Subtarget, DAG, DL);
-    } else {
-      std::tie(LHS, RHS) = splitVector(Src, DAG, DL);
-    }
-    assert(LHS.getValueSizeInBits() == VT.getSizeInBits() &&
-           RHS.getValueSizeInBits() == VT.getSizeInBits() &&
-           "PACK src/dst size mismatch");
+    auto [LHS, RHS] = splitVector(Src, DAG, DL);
     unsigned PackOpc = Opc == X86ISD::VTRUNCS ? X86ISD::PACKSS : X86ISD::PACKUS;
     SDValue Pack = DAG.getNode(PackOpc, DL, VT, LHS, RHS);
     if (VT.is128BitVector())
@@ -56166,44 +55968,21 @@ static SDValue combineXorWithGF2P8AFFINEQB(SDNode *N, const SDLoc &DL,
                                            SelectionDAG &DAG, EVT VT) {
   using namespace SDPatternMatch;
 
-  SDValue X, Y, XorOp;
-  APInt Imm, ConstUndef;
-
+  SDValue X, Y, SplatOp;
+  APInt Imm;
+  // Use sd_match for structure matching - m_Xor handles commutation
   if (!sd_match(N, m_Xor(m_OneUse(m_TernaryOp(X86ISD::GF2P8AFFINEQB, m_Value(X),
                                               m_Value(Y), m_ConstInt(Imm))),
-                         m_Value(XorOp))))
+                         m_Value(SplatOp))))
     return SDValue();
 
   // GF2P8AFFINEQB only operates on i8 vector types
   assert((VT == MVT::v16i8 || VT == MVT::v32i8 || VT == MVT::v64i8) &&
          "Unsupported GFNI type");
 
-  unsigned NumElts = VT.getVectorNumElements();
-
-  // Fold: GF2P8AFFINEQB(x, M) ^ x
-  //   =>  GF2P8AFFINEQB(x, M ^ IDENTITY)
-  // The "identity" is a noop. It adds a XOR by the unmodified input.
-  SmallVector<APInt> YEltBits;
-  if (X == XorOp && getTargetConstantBitsFromNode(Y, 8, ConstUndef, YEltBits,
-                                                  /*AllowWholeUndefs=*/false)) {
-    const uint8_t IdentityMatrix[] = {128, 64, 32, 16, 8, 4, 2, 1};
-
-    SmallVector<SDValue> BuildMatrix;
-    for (unsigned I = 0; I != NumElts; ++I) {
-      APInt MatrixRow = YEltBits[I] ^ IdentityMatrix[I % 8];
-      BuildMatrix.push_back(DAG.getConstant(MatrixRow, DL, MVT::i8));
-    }
-
-    SDValue NewMatrix = DAG.getBuildVector(VT, DL, BuildMatrix);
-
-    return DAG.getNode(X86ISD::GF2P8AFFINEQB, DL, VT, X, NewMatrix,
-                       DAG.getTargetConstant(Imm, DL, MVT::i8));
-  }
-
-  // Fold: GF2P8AFFINEQB(x, m, Imm) ^ Splat(C)
-  //   =>  GF2P8AFFINEQB(x, m, Imm ^ C)
+  // Use X86::isConstantSplat for robust splat constant extraction
   APInt SplatVal;
-  if (!X86::isConstantSplat(XorOp, SplatVal, /*AllowPartialUndefs=*/false))
+  if (!X86::isConstantSplat(SplatOp, SplatVal, /*AllowPartialUndefs=*/false))
     return SDValue();
 
   uint64_t NewImm = Imm.getZExtValue() ^ SplatVal.getZExtValue();
@@ -56211,49 +55990,32 @@ static SDValue combineXorWithGF2P8AFFINEQB(SDNode *N, const SDLoc &DL,
                      DAG.getTargetConstant(NewImm, DL, MVT::i8));
 }
 
-// Given that vgf2p8affineqb performs a XOR permutation, two affines that share
-// a operand can be reassociated through a standalone XOR.
+// Fold: vgf2p8affineqb(x, m1, i1) ^ vgf2p8affineqb(x, m2, i2)
+//   =>  vgf2p8affineqb(x, m1 ^ m2, i1 ^ i2)
+// The matrix in vgf2p8affineqb determines which bits of the input are XORed
+// together. XORing two affine transformations of the same input can be folded
+// by XORing both their matrices and immediates together.
 static SDValue combineXorWithTwoGF2P8AFFINEQB(SDNode *N, const SDLoc &DL,
                                               SelectionDAG &DAG, EVT VT) {
   using namespace SDPatternMatch;
 
-  SDValue X0, X1, M0, M1;
+  SDValue X0, Y0, Y1;
   APInt Imm0, Imm1;
   // Use sd_match for structure matching - m_Xor handles commutation
-  if (!sd_match(N,
-                m_Xor(m_OneUse(m_TernaryOp(X86ISD::GF2P8AFFINEQB, m_Value(X0),
-                                           m_Value(M0), m_ConstInt(Imm0))),
-                      m_OneUse(m_TernaryOp(X86ISD::GF2P8AFFINEQB, m_Value(X1),
-                                           m_Value(M1), m_ConstInt(Imm1))))))
+  // Match: GF2P8AFFINEQB(x, m1, i1) ^ GF2P8AFFINEQB(x, m2, i2)
+  if (!sd_match(
+          N, m_Xor(m_OneUse(m_TernaryOp(X86ISD::GF2P8AFFINEQB, m_Value(X0),
+                                        m_Value(Y0), m_ConstInt(Imm0))),
+                   m_OneUse(m_TernaryOp(X86ISD::GF2P8AFFINEQB, m_Deferred(X0),
+                                        m_Value(Y1), m_ConstInt(Imm1))))))
     return SDValue();
 
   assert((VT == MVT::v16i8 || VT == MVT::v32i8 || VT == MVT::v64i8) &&
          "Unsupported GFNI type");
 
-  // Fold: GF2P8AFFINEQB(x0, m, i1) ^ GF2P8AFFINEQB(x1, m, i2)
-  //   =>  GF2P8AFFINEQB(x0 ^ x1, m, i1 ^ i2)
-  // This instruction performs an XOR permutation of the input, which is
-  // associative. Therefore XORing before permuting is equivalent.
-  if (M0 == M1) {
-    uint64_t NewImm = Imm0.getZExtValue() ^ Imm1.getZExtValue();
-
-    SDValue NewSrc = DAG.getNode(ISD::XOR, DL, VT, X0, X1);
-
-    return DAG.getNode(X86ISD::GF2P8AFFINEQB, DL, VT, NewSrc, M0,
-                       DAG.getTargetConstant(NewImm, DL, MVT::i8));
-  }
-
-  // Fold: vgf2p8affineqb(x, m0, i1) ^ vgf2p8affineqb(x, m1, i2)
-  //   =>  vgf2p8affineqb(x, m0 ^ m1, i1 ^ i2)
-  // The matrix in vgf2p8affineqb determines which bits of the input are XORed
-  // together. XORing two affine transformations of the same input can be folded
-  // by XORing both their matrices and immediates together.
-  if (X0 != X1)
-    return SDValue();
-
   uint64_t NewImm = Imm0.getZExtValue() ^ Imm1.getZExtValue();
 
-  SDValue NewMatrix = DAG.getNode(ISD::XOR, DL, VT, M0, M1);
+  SDValue NewMatrix = DAG.getNode(ISD::XOR, DL, VT, Y0, Y1);
 
   return DAG.getNode(X86ISD::GF2P8AFFINEQB, DL, VT, X0, NewMatrix,
                      DAG.getTargetConstant(NewImm, DL, MVT::i8));
@@ -56274,14 +56036,14 @@ static SDValue combineXorSubCTLZ(SDNode *N, const SDLoc &DL, SelectionDAG &DAG,
   SDValue N0 = N->getOperand(0);
   SDValue N1 = N->getOperand(1);
 
-  if (N0.getOpcode() != ISD::CTLZ_ZERO_POISON &&
-      N1.getOpcode() != ISD::CTLZ_ZERO_POISON)
+  if (N0.getOpcode() != ISD::CTLZ_ZERO_UNDEF &&
+      N1.getOpcode() != ISD::CTLZ_ZERO_UNDEF)
     return SDValue();
 
   SDValue OpCTLZ;
   SDValue OpSizeTM1;
 
-  if (N1.getOpcode() == ISD::CTLZ_ZERO_POISON) {
+  if (N1.getOpcode() == ISD::CTLZ_ZERO_UNDEF) {
     OpCTLZ = N1;
     OpSizeTM1 = N0;
   } else if (N->getOpcode() == ISD::SUB) {
@@ -56334,9 +56096,6 @@ static SDValue combineXor(SDNode *N, SelectionDAG &DAG,
   if (SDValue Cmp = foldVectorXorShiftIntoCmp(N, DAG, Subtarget))
     return Cmp;
 
-  if (SDValue R = combineMaskBitOp(N, DL, DAG))
-    return R;
-
   if (SDValue R = combineBitOpWithMOVMSK(N->getOpcode(), DL, N0, N1, DAG))
     return R;
 
@@ -56424,7 +56183,7 @@ static SDValue combineBITREVERSE(SDNode *N, SelectionDAG &DAG,
   if (VT.isInteger() && N0.getOpcode() == ISD::BITCAST && N0.hasOneUse()) {
     SDValue Src = N0.getOperand(0);
     EVT SrcVT = Src.getValueType();
-    if (SrcVT.isVectorOf(MVT::i1) &&
+    if (SrcVT.isVector() && SrcVT.getScalarType() == MVT::i1 &&
         (DCI.isBeforeLegalize() ||
          DAG.getTargetLoweringInfo().isTypeLegal(SrcVT)) &&
         Subtarget.hasSSSE3()) {
@@ -56881,101 +56640,6 @@ static SDValue combineAndnp(SDNode *N, SelectionDAG &DAG,
   return SDValue();
 }
 
-// Strip ext/trunc/mask wrappers that do not change a bit index interpreted
-// modulo BW. Don't peek below log2(BW) bits (e.g. through a zext from i1):
-// a narrower value no longer determines the bit index on its own.
-static SDValue peekThroughBitPosExtTrunc(SDValue V, unsigned BW) {
-  assert(V.getScalarValueSizeInBits() >= Log2_32(BW) &&
-         "bit position type must cover the whole index range");
-  APInt LowBits =
-      APInt::getLowBitsSet(V.getScalarValueSizeInBits(), Log2_32(BW));
-  for (;;) {
-    unsigned Op = V.getOpcode();
-    if ((Op == ISD::TRUNCATE || Op == ISD::ZERO_EXTEND ||
-         Op == ISD::ANY_EXTEND) &&
-        V.getOperand(0).getScalarValueSizeInBits() >= Log2_32(BW)) {
-      V = V.getOperand(0);
-      LowBits = LowBits.zextOrTrunc(V.getScalarValueSizeInBits());
-      continue;
-    }
-    if (Op == ISD::AND) {
-      // The mask must keep all bits the bit-test cares about set.
-      auto *C = dyn_cast<ConstantSDNode>(V.getOperand(1));
-      if (C && LowBits.isSubsetOf(C->getAPIntValue())) {
-        V = V.getOperand(0);
-        continue;
-      }
-    }
-    return V;
-  }
-}
-
-// Try to merge a (X86ISD::BT Src, BitNo) with a sibling bit-modifying op on
-// Src (AND(Src, rotl -2, X), OR(Src, shl 1, X), XOR(Src, shl 1, X)) into a
-// single flag-producing X86ISD::{BTR,BTS,BTC} node. Both BT and BTR/BTS/BTC
-// set CF from the pre-op bit value, so one instruction subsumes the other.
-static SDValue combineBTToBitOpFlag(SDNode *N, SelectionDAG &DAG) {
-  using namespace SDPatternMatch;
-  // Src is the BT source; matching against it with m_Specific requires
-  // SDValue identity with the modify's operand, so no peek-through is needed
-  // here (only the bit *position* may differ by ext/trunc/and-mask).
-  SDValue Src = N->getOperand(0);
-  SDValue BitNo = N->getOperand(1);
-  EVT VT = Src.getValueType();
-  SDLoc DL(N);
-
-  // X86ISD::BT is only emitted for i32/i64 (smaller widths are promoted in
-  // getBT before the node is created).
-  assert((VT == MVT::i32 || VT == MVT::i64) &&
-         "X86ISD::BT is only emitted for i32/i64");
-
-  unsigned BW = VT.getScalarSizeInBits();
-  SDValue PeeledBitNo = peekThroughBitPosExtTrunc(BitNo, BW);
-
-  for (SDNode *User : Src->users()) {
-    if (User == N)
-      continue;
-
-    unsigned FlagOp = 0;
-    SDValue ShAmt;
-    if (sd_match(User,
-                 m_And(m_Specific(Src),
-                       m_OneUse(m_Rotl(m_SpecificInt(APInt::getAllOnes(BW) - 1),
-                                       m_Value(ShAmt)))))) {
-      // (and Src, (rotl -2, X)): clears bit X.
-      FlagOp = X86ISD::BTR;
-    } else if (sd_match(User, m_Or(m_Specific(Src),
-                                   m_OneUse(m_Shl(m_SpecificInt(1),
-                                                  m_Value(ShAmt)))))) {
-      // (or Src, (shl 1, X)): sets bit X.
-      FlagOp = X86ISD::BTS;
-    } else if (sd_match(User, m_Xor(m_Specific(Src),
-                                    m_OneUse(m_Shl(m_SpecificInt(1),
-                                                   m_Value(ShAmt)))))) {
-      // (xor Src, (shl 1, X)): flips bit X.
-      FlagOp = X86ISD::BTC;
-    } else {
-      continue;
-    }
-
-    // The BT and the bit-op must address the same bit. They can differ only
-    // by truncation/extension or an AND that preserves the low log2(BW) bits.
-    if (peekThroughBitPosExtTrunc(ShAmt, BW) != PeeledBitNo)
-      continue;
-
-    // BT's bit index is constrained to Src's type, so we can reuse it as-is
-    // for BTR/BTS/BTC's bit index operand.
-    assert(BitNo.getValueType() == VT && "BT bit index must match Src type");
-    SDValue New =
-        DAG.getNode(FlagOp, DL, DAG.getVTList(VT, MVT::i32), Src, BitNo);
-    // Reroute the value output through User's consumers.
-    DAG.ReplaceAllUsesOfValueWith(SDValue(User, 0), New.getValue(0));
-    // Return the flags output so combineBT installs it as N's replacement.
-    return New.getValue(1);
-  }
-  return SDValue();
-}
-
 static SDValue combineBT(SDNode *N, SelectionDAG &DAG,
                          TargetLowering::DAGCombinerInfo &DCI) {
   SDValue N1 = N->getOperand(1);
@@ -56989,9 +56653,6 @@ static SDValue combineBT(SDNode *N, SelectionDAG &DAG,
     return SDValue(N, 0);
   }
 
-  if (SDValue V = combineBTToBitOpFlag(N, DAG))
-    return V;
-
   return SDValue();
 }
 
@@ -57847,7 +57508,7 @@ static SDValue combineSetCC(SDNode *N, SelectionDAG &DAG,
     // Both of these patterns can be better optimized in
     // DAGCombiner::foldAndOrOfSETCC. Note this only applies for scalar
     // integers which is checked above.
-    if (ISD::isAbsOpcode(LHS.getOpcode()) && LHS.hasOneUse()) {
+    if (LHS.getOpcode() == ISD::ABS && LHS.hasOneUse()) {
       if (auto *C = dyn_cast<ConstantSDNode>(RHS)) {
         const APInt &CInt = C->getAPIntValue();
         // We can better optimize this case in DAGCombiner::foldAndOrOfSETCC.
@@ -57867,7 +57528,7 @@ static SDValue combineSetCC(SDNode *N, SelectionDAG &DAG,
       return V;
   }
 
-  if (VT.isVectorOf(MVT::i1) &&
+  if (VT.isVector() && VT.getVectorElementType() == MVT::i1 &&
       (CC == ISD::SETNE || CC == ISD::SETEQ || ISD::isSignedIntSetCC(CC))) {
     // Using temporaries to avoid messing up operand ordering for later
     // transformations if this doesn't work.
@@ -59231,7 +58892,8 @@ static SDValue matchPMADDWD(SelectionDAG &DAG, SDNode *N,
   if (!Subtarget.hasSSE2())
     return SDValue();
 
-  if (!VT.isVectorOf(MVT::i32) || VT.getVectorNumElements() < 4 ||
+  if (!VT.isVector() || VT.getVectorElementType() != MVT::i32 ||
+      VT.getVectorNumElements() < 4 ||
       !isPowerOf2_32(VT.getVectorNumElements()))
     return SDValue();
 
@@ -59283,12 +58945,12 @@ static SDValue matchPMADDWD(SelectionDAG &DAG, SDNode *N,
       return SDValue();
     if (!Mul) {
       // First time an extract_elt's source vector is visited. Must be a MUL
-      // with at least 2X number of vector elements than the BUILD_VECTOR.
+      // with 2X number of vector elements than the BUILD_VECTOR.
       // Both extracts must be from same MUL.
       Mul = Vec0L;
       if ((Mul.getOpcode() != ISD::MUL && Mul.getOpcode() != ISD::SHL &&
            Mul.getOpcode() != ISD::SIGN_EXTEND) ||
-          Mul.getValueType().getVectorNumElements() < (2 * e))
+          Mul.getValueType().getVectorNumElements() != 2 * e)
         return SDValue();
     }
     // Check that the extract is from the same MUL previously seen.
@@ -59298,7 +58960,6 @@ static SDValue matchPMADDWD(SelectionDAG &DAG, SDNode *N,
 
   EVT TruncVT = EVT::getVectorVT(*DAG.getContext(), MVT::i16,
                                  VT.getVectorNumElements() * 2);
-  EVT MulVT = TruncVT.changeVectorElementType(*DAG.getContext(), MVT::i32);
 
   SDValue N0, N1;
   if (Mul.getOpcode() == ISD::MUL) {
@@ -59308,37 +58969,34 @@ static SDValue matchPMADDWD(SelectionDAG &DAG, SDNode *N,
         Mode == ShrinkMode::MULU16)
       return SDValue();
 
-    N0 = DAG.getExtractSubvector(DL, MulVT, Mul.getOperand(0), 0);
-    N1 = DAG.getExtractSubvector(DL, MulVT, Mul.getOperand(1), 0);
-    N0 = DAG.getNode(ISD::TRUNCATE, DL, TruncVT, N0);
-    N1 = DAG.getNode(ISD::TRUNCATE, DL, TruncVT, N1);
+    N0 = DAG.getNode(ISD::TRUNCATE, DL, TruncVT, Mul.getOperand(0));
+    N1 = DAG.getNode(ISD::TRUNCATE, DL, TruncVT, Mul.getOperand(1));
   } else if (Mul.getOpcode() == ISD::SHL) {
     SDValue ShVal = Mul.getOperand(0);
     if (ShVal.getOpcode() != ISD::SIGN_EXTEND)
       return SDValue();
 
     N0 = ShVal.getOperand(0);
-    if (N0.getValueType().getScalarType() != MVT::i16)
+    if (N0.getValueType() != TruncVT)
       return SDValue();
 
-    // A shift by more 15 or more would overflow a signed i16.
+    // A shift by more than 15 would overflow an i16.
     if (!ISD::matchUnaryPredicate(Mul.getOperand(1), [](ConstantSDNode *C) {
-          return C->getAPIntValue().ult(15);
+          return C->getAPIntValue().ule(15);
         }))
       return SDValue();
 
-    N0 = DAG.getExtractSubvector(DL, TruncVT, N0, 0);
-    N1 = DAG.getExtractSubvector(DL, MulVT, Mul.getOperand(1), 0);
     N1 = DAG.getNode(ISD::SHL, DL, TruncVT, DAG.getConstant(1, DL, TruncVT),
-                     DAG.getZExtOrTrunc(N1, DL, TruncVT));
+                     DAG.getZExtOrTrunc(Mul.getOperand(1), DL, TruncVT));
   } else {
     assert(Mul.getOpcode() == ISD::SIGN_EXTEND);
 
-    if (Mul.getOperand(0).getValueType().getScalarType() != MVT::i16)
+    // Add a trivial multiplication with 1 so that we can make use of VPMADDWD.
+    N0 = Mul.getOperand(0);
+
+    if (N0.getValueType() != TruncVT)
       return SDValue();
 
-    // Add a trivial multiplication with 1 so that we can make use of VPMADDWD.
-    N0 = DAG.getExtractSubvector(DL, TruncVT, Mul.getOperand(0), 0);
     N1 = DAG.getConstant(1, DL, TruncVT);
   }
 
@@ -59367,7 +59025,8 @@ static SDValue matchPMADDWD_2(SelectionDAG &DAG, SDNode *N,
   if (!Subtarget.hasSSE2())
     return SDValue();
 
-  if (!VT.isVectorOf(MVT::i32) || VT.getVectorNumElements() < 4 ||
+  if (!VT.isVector() || VT.getVectorElementType() != MVT::i32 ||
+      VT.getVectorNumElements() < 4 ||
       !isPowerOf2_32(VT.getVectorNumElements()))
     return SDValue();
 
@@ -59484,12 +59143,15 @@ static SDValue combineAddOfPMADDWD(SelectionDAG &DAG, SDValue N0, SDValue N1,
 
   unsigned NumElts = VT.getVectorNumElements();
   MVT OpVT = N0.getOperand(0).getSimpleValueType();
+  APInt DemandedBits = APInt::getAllOnes(OpVT.getScalarSizeInBits());
   APInt DemandedHiElts = APInt::getSplat(2 * NumElts, APInt(2, 2));
 
-  bool Op0HiZero = DAG.MaskedVectorIsZero(N0.getOperand(0), DemandedHiElts) ||
-                   DAG.MaskedVectorIsZero(N0.getOperand(1), DemandedHiElts);
-  bool Op1HiZero = DAG.MaskedVectorIsZero(N1.getOperand(0), DemandedHiElts) ||
-                   DAG.MaskedVectorIsZero(N1.getOperand(1), DemandedHiElts);
+  bool Op0HiZero =
+      DAG.MaskedValueIsZero(N0.getOperand(0), DemandedBits, DemandedHiElts) ||
+      DAG.MaskedValueIsZero(N0.getOperand(1), DemandedBits, DemandedHiElts);
+  bool Op1HiZero =
+      DAG.MaskedValueIsZero(N1.getOperand(0), DemandedBits, DemandedHiElts) ||
+      DAG.MaskedValueIsZero(N1.getOperand(1), DemandedBits, DemandedHiElts);
 
   // TODO: Check for zero lower elements once we have actual codegen that
   // creates them.
@@ -59588,6 +59250,11 @@ static SDValue matchVPMADD52(SDNode *N, SelectionDAG &DAG, const SDLoc &DL,
       (!Subtarget.hasAVXIFMA() && !Subtarget.hasIFMA()))
     return SDValue();
 
+  // Need AVX-512VL vector length extensions if operating on XMM/YMM registers
+  if (!Subtarget.hasAVXIFMA() && !Subtarget.hasVLX() &&
+      VT.getSizeInBits() < 512)
+    return SDValue();
+
   const auto TotalSize = VT.getSizeInBits();
   if (TotalSize < 128 || !isPowerOf2_64(TotalSize))
     return SDValue();
@@ -60282,27 +59949,6 @@ static SDValue combineConcatVectorOps(const SDLoc &DL, MVT VT,
           return DAG.getVectorShuffle(VT, DL, Concat0, Concat1, NewMask);
         }
       }
-      // If we're concatenating 4 x 128-bit to 512-bit, see if we can
-      // concat either 2 x 128-bit pair instead.
-      // TODO: Can we do this generically?
-      if (!IsSplat && NumOps == 4 && VT.is512BitVector() &&
-          Subtarget.useAVX512Regs() &&
-          (EltSizeInBits >= 32 || Subtarget.useBWIRegs())) {
-        MVT HalfVT = VT.getHalfNumVectorElementsVT();
-        SDValue Concat0 = combineConcatVectorOps(DL, HalfVT, Ops.slice(0, 2),
-                                                 DAG, Subtarget, Depth + 1);
-        SDValue Concat1 = combineConcatVectorOps(DL, HalfVT, Ops.slice(2, 2),
-                                                 DAG, Subtarget, Depth + 1);
-        if (Concat0 || Concat1) {
-          Concat0 = Concat0 ? Concat0
-                            : DAG.getNode(ISD::CONCAT_VECTORS, DL, HalfVT,
-                                          Ops.slice(0, 2));
-          Concat1 = Concat1 ? Concat1
-                            : DAG.getNode(ISD::CONCAT_VECTORS, DL, HalfVT,
-                                          Ops.slice(2, 2));
-          return DAG.getNode(ISD::CONCAT_VECTORS, DL, VT, Concat0, Concat1);
-        }
-      }
       break;
     }
     case X86ISD::VBROADCAST: {
@@ -60647,65 +60293,22 @@ static SDValue combineConcatVectorOps(const SDLoc &DL, MVT VT,
                            Op0.getOperand(1));
       }
       break;
-    case X86ISD::VSHLDQ:
-    case X86ISD::VSRLDQ:
-      if (!IsSplat &&
-          ((VT.is256BitVector() && Subtarget.hasInt256()) ||
-           (VT.is512BitVector() && Subtarget.useBWIRegs())) &&
-          llvm::all_of(Ops, [Op0](SDValue Op) {
-            return Op0.getOperand(1) == Op.getOperand(1);
-          })) {
-        return DAG.getNode(Opcode, DL, VT, ConcatSubOperand(VT, Ops, 0),
-                           Op0.getOperand(1));
-      }
-      break;
+    case X86ISD::VPERMI:
     case X86ISD::VROTLI:
     case X86ISD::VROTRI:
       if (!IsSplat &&
-          ((VT.is256BitVector() && Subtarget.hasAVX512()) ||
+          ((VT.is256BitVector() && Subtarget.hasVLX()) ||
            (VT.is512BitVector() && Subtarget.useAVX512Regs())) &&
           llvm::all_of(Ops, [Op0](SDValue Op) {
             return Op0.getOperand(1) == Op.getOperand(1);
           })) {
+        assert(!(Opcode == X86ISD::VPERMI &&
+                 Op0.getValueType().is128BitVector()) &&
+               "Illegal 128-bit X86ISD::VPERMI nodes");
         return DAG.getNode(Opcode, DL, VT, ConcatSubOperand(VT, Ops, 0),
                            Op0.getOperand(1));
       }
       break;
-    case ISD::ROTL:
-    case ISD::ROTR:
-      if (!IsSplat && ((VT.is256BitVector() && Subtarget.hasAVX512()) ||
-                       (VT.is512BitVector() && Subtarget.useAVX512Regs()))) {
-        SDValue Concat0 = CombineSubOperand(VT, Ops, 0);
-        SDValue Concat1 = CombineSubOperand(VT, Ops, 1);
-        if (Concat0 || Concat1)
-          return DAG.getNode(Opcode, DL, VT,
-                             Concat0 ? Concat0 : ConcatSubOperand(VT, Ops, 0),
-                             Concat1 ? Concat1 : ConcatSubOperand(VT, Ops, 1));
-      }
-      break;
-    case X86ISD::VPERMI:
-      if (!IsSplat && NumOps == 2 &&
-          (VT.is512BitVector() && Subtarget.useAVX512Regs())) {
-        // If both halves share the mask - then concat as VPERMI.
-        if (Ops[0].getOperand(1) == Ops[1].getOperand(1))
-          return DAG.getNode(Opcode, DL, VT, ConcatSubOperand(VT, Ops, 0),
-                             Op0.getOperand(1));
-
-        // Fallback to a VPERMV3 variable mask shuffle.
-        SmallVector<int, 8> Mask, HiMask;
-        unsigned NumElts = VT.getVectorNumElements();
-        DecodeVPERMMask(NumElts / 2, Ops[0].getConstantOperandVal(1), Mask);
-        DecodeVPERMMask(NumElts / 2, Ops[1].getConstantOperandVal(1), HiMask);
-        for (int M : HiMask)
-          Mask.push_back(NumElts + M);
-
-        return lowerShuffleWithPERMV(
-            DL, VT, Mask,
-            widenSubVector(VT, Ops[0].getOperand(0), false, Subtarget, DAG, DL),
-            widenSubVector(VT, Ops[1].getOperand(0), false, Subtarget, DAG, DL),
-            Subtarget, DAG);
-      }
-      break;
     case ISD::AND:
     case ISD::OR:
     case ISD::XOR:
@@ -60732,6 +60335,7 @@ static SDValue combineConcatVectorOps(const SDLoc &DL, MVT VT,
       break;
     case X86ISD::PCMPEQ:
     case X86ISD::PCMPGT:
+      // TODO: 512-bit PCMPEQ/PCMPGT -> VPCMP+VPMOVM2 handling.
       if (!IsSplat && VT.is256BitVector() && Subtarget.hasInt256()) {
         SDValue Concat0 = CombineSubOperand(VT, Ops, 0);
         SDValue Concat1 = CombineSubOperand(VT, Ops, 1);
@@ -60741,19 +60345,6 @@ static SDValue combineConcatVectorOps(const SDLoc &DL, MVT VT,
                              Concat1 ? Concat1 : ConcatSubOperand(VT, Ops, 1));
         break;
       }
-      if (!IsSplat && VT.is512BitVector() && Subtarget.useAVX512Regs() &&
-          (EltSizeInBits >= 32 || Subtarget.useBWIRegs())) {
-        if (IsConcatFree(VT, Ops, 0) && IsConcatFree(VT, Ops, 1)) {
-          MVT BoolVT = VT.changeVectorElementType(MVT::i1);
-          SDValue Cmp =
-              DAG.getSetCC(DL, BoolVT, ConcatSubOperand(VT, Ops, 0),
-                           ConcatSubOperand(VT, Ops, 1),
-                           Opcode == X86ISD::PCMPEQ ? ISD::CondCode::SETEQ
-                                                    : ISD::CondCode::SETGT);
-          return DAG.getNode(ISD::SIGN_EXTEND, DL, VT, Cmp);
-        }
-        break;
-      }
 
       if (!IsSplat && VT == MVT::v8i32) {
         // Without AVX2, see if we can cast the values to v8f32 and use fcmp.
@@ -60825,8 +60416,8 @@ static SDValue combineConcatVectorOps(const SDLoc &DL, MVT VT,
     case ISD::CTPOP:
     case ISD::CTTZ:
     case ISD::CTLZ:
-    case ISD::CTTZ_ZERO_POISON:
-    case ISD::CTLZ_ZERO_POISON:
+    case ISD::CTTZ_ZERO_UNDEF:
+    case ISD::CTLZ_ZERO_UNDEF:
       if (!IsSplat && ((VT.is256BitVector() && Subtarget.hasInt256()) ||
                        (VT.is512BitVector() && Subtarget.useBWIRegs()))) {
         return DAG.getNode(Opcode, DL, VT, ConcatSubOperand(VT, Ops, 0));
@@ -61328,7 +60919,6 @@ static SDValue combineINSERT_SUBVECTOR(SDNode *N, SelectionDAG &DAG,
   MVT SubVecVT = SubVec.getSimpleValueType();
   int VecNumElts = OpVT.getVectorNumElements();
   int SubVecNumElts = SubVecVT.getVectorNumElements();
-  const TargetLowering &TLI = DAG.getTargetLoweringInfo();
 
   if (Vec.isUndef() && SubVec.isUndef())
     return DAG.getUNDEF(OpVT);
@@ -61366,19 +60956,6 @@ static SDValue combineINSERT_SUBVECTOR(SDNode *N, SelectionDAG &DAG,
                              getZeroVector(OpVT, Subtarget, DAG, dl),
                              Ins.getOperand(1), N->getOperand(2));
     }
-
-    // See if were inserting into a zero vXi1 vector and the subvector was
-    // bitcast from a gpr that could be zero-extended directly.
-    if (IsI1Vector && TLI.isTypeLegal(OpVT) && SubVec.hasOneUse()) {
-      SDValue SubInt = peekThroughOneUseBitcasts(SubVec);
-      EVT IntVT = EVT::getIntegerVT(*DAG.getContext(), VecNumElts);
-      if (TLI.isTypeLegal(IntVT) && SubInt.getValueType().isScalarInteger()) {
-        SubInt = DAG.getNode(ISD::ZERO_EXTEND, dl, IntVT, SubInt);
-        SubInt = DAG.getNode(ISD::SHL, dl, IntVT, SubInt,
-                             DAG.getShiftAmountConstant(IdxVal, IntVT, dl));
-        return DAG.getBitcast(OpVT, SubInt);
-      }
-    }
   }
 
   // Stop here if this is an i1 vector.
@@ -61510,16 +61087,9 @@ static SDValue combineINSERT_SUBVECTOR(SDNode *N, SelectionDAG &DAG,
     }
   }
 
-  auto peekThroughBitcastsAndExtracts = [](SDValue V) {
-    while (V.getOpcode() == ISD::BITCAST ||
-           V.getOpcode() == ISD::EXTRACT_SUBVECTOR)
-      V = V.getOperand(0);
-    return V;
-  };
-
   // Attempt to recursively combine to a shuffle.
   if (isTargetShuffle(peekThroughBitcasts(Vec).getOpcode()) &&
-      isTargetShuffle(peekThroughBitcastsAndExtracts(SubVec).getOpcode())) {
+      isTargetShuffle(peekThroughBitcasts(SubVec).getOpcode())) {
     SDValue Op(N, 0);
     if (SDValue Res = combineX86ShufflesRecursively(Op, DAG, Subtarget))
       return Res;
@@ -61653,6 +61223,14 @@ static SDValue combineEXTRACT_SUBVECTOR(SDNode *N, SelectionDAG &DAG,
   if (InVec.getOpcode() == ISD::BUILD_VECTOR)
     return DAG.getBuildVector(VT, DL, InVec->ops().slice(IdxVal, NumSubElts));
 
+  // EXTRACT_SUBVECTOR(EXTRACT_SUBVECTOR(V,C1)),C2) - EXTRACT_SUBVECTOR(V,C1+C2)
+  if (IdxVal != 0 && InVec.getOpcode() == ISD::EXTRACT_SUBVECTOR &&
+      InVec.hasOneUse() && TLI.isTypeLegal(VT) &&
+      TLI.isTypeLegal(InVec.getOperand(0).getValueType())) {
+    unsigned NewIdx = IdxVal + InVec.getConstantOperandVal(1);
+    return extractSubVector(InVec.getOperand(0), NewIdx, DAG, DL, SizeInBits);
+  }
+
   // EXTRACT_SUBVECTOR(INSERT_SUBVECTOR(SRC,SUB,C1),C2)
   // --> INSERT_SUBVECTOR(EXTRACT_SUBVECTOR(SRC,C2),SUB,C1-C2)
   // iff SUB is entirely contained in the extraction.
@@ -62237,62 +61815,28 @@ static SDValue combineEXTEND_VECTOR_INREG(SDNode *N, SelectionDAG &DAG,
 static SDValue combineKSHIFT(SDNode *N, SelectionDAG &DAG,
                              TargetLowering::DAGCombinerInfo &DCI) {
   EVT VT = N->getValueType(0);
-  SDValue Src = N->getOperand(0);
-  uint64_t Amt = N->getConstantOperandVal(1);
-  unsigned Opcode = N->getOpcode();
   const TargetLowering &TLI = DAG.getTargetLoweringInfo();
-  SDLoc DL(N);
-
-  if (ISD::isBuildVectorAllZeros(Src.getNode()))
-    return DAG.getConstant(0, DL, VT);
+  if (ISD::isBuildVectorAllZeros(N->getOperand(0).getNode()))
+    return DAG.getConstant(0, SDLoc(N), VT);
 
-  // Constant Fold.
-  if (auto *SrcC = dyn_cast<ConstantSDNode>(peekThroughBitcasts(Src))) {
-    APInt NewCst = Opcode == X86ISD::KSHIFTR ? SrcC->getAPIntValue().lshr(Amt)
-                                             : SrcC->getAPIntValue().shl(Amt);
-    return DAG.getBitcast(VT,
-                          DAG.getConstant(NewCst, DL, SrcC->getValueType(0)));
-  }
-
-  if (Opcode == X86ISD::KSHIFTR) {
-    // Fold kshiftr(extract_subvector(X,C1),C2)
-    //  --> extract_subvector(kshiftr(X,C1+C2),0)
-    // Fold kshiftr(kshiftr(X,C1),C2) --> kshiftr(X,C1+C2)
-    if (Src.getOpcode() == ISD::EXTRACT_SUBVECTOR ||
-        Src.getOpcode() == X86ISD::KSHIFTR) {
-      SDValue Inner = Src.getOperand(0);
-      EVT InnerVT = Inner.getValueType();
-      uint64_t NewAmt = Amt + Src.getConstantOperandVal(1);
-      if (TLI.isTypeLegal(InnerVT) && NewAmt < InnerVT.getVectorNumElements()) {
-        SDValue Shift = DAG.getNode(X86ISD::KSHIFTR, DL, InnerVT, Inner,
-                                    DAG.getTargetConstant(NewAmt, DL, MVT::i8));
+  // Fold kshiftr(extract_subvector(X,C1),C2)
+  //  --> extract_subvector(kshiftr(X,C1+C2),0)
+  // Fold kshiftr(kshiftr(X,C1),C2) --> kshiftr(X,C1+C2)
+  if (N->getOpcode() == X86ISD::KSHIFTR) {
+    SDLoc DL(N);
+    if (N->getOperand(0).getOpcode() == ISD::EXTRACT_SUBVECTOR ||
+        N->getOperand(0).getOpcode() == X86ISD::KSHIFTR) {
+      SDValue Src = N->getOperand(0).getOperand(0);
+      uint64_t Amt = N->getConstantOperandVal(1) +
+                     N->getOperand(0).getConstantOperandVal(1);
+      EVT SrcVT = Src.getValueType();
+      if (TLI.isTypeLegal(SrcVT) && Amt < SrcVT.getVectorNumElements()) {
+        SDValue Shift = DAG.getNode(X86ISD::KSHIFTR, DL, SrcVT, Src,
+                                    DAG.getTargetConstant(Amt, DL, MVT::i8));
         return DAG.getNode(ISD::EXTRACT_SUBVECTOR, DL, VT, Shift,
                            DAG.getVectorIdxConstant(0, DL));
       }
     }
-    // Fold kshiftr(concat_vectors(X,Y,Z,W),C)
-    //  --> concat_vectors(Z,W,0,0) iff amount is whole subvector shift.
-    if (Src.getOpcode() == ISD::CONCAT_VECTORS &&
-        (Amt % Src.getOperand(0).getValueType().getVectorNumElements()) == 0) {
-      unsigned NumSubs = Src.getNumOperands();
-      EVT SubVT = Src.getOperand(0).getValueType();
-      unsigned Ofs = Amt / SubVT.getVectorNumElements();
-      SmallVector<SDValue, 4> SubOps(NumSubs, DAG.getConstant(0, DL, SubVT));
-      for (unsigned I = Ofs; I != NumSubs; ++I)
-        SubOps[I - Ofs] = Src.getOperand(I);
-      return DAG.getNode(ISD::CONCAT_VECTORS, DL, VT, SubOps);
-    }
-  }
-
-  // Fold kshift(logicop(X,C1),C2)
-  //  --> logicop(kshift(X,C2),kshift(C1,C2))
-  if (ISD::isBitwiseLogicOp(Src.getOpcode()) &&
-      isa<ConstantSDNode>(peekThroughBitcasts(Src.getOperand(1)))) {
-    SDValue LHS =
-        DAG.getNode(Opcode, DL, VT, Src.getOperand(0), N->getOperand(1));
-    SDValue RHS =
-        DAG.getNode(Opcode, DL, VT, Src.getOperand(1), N->getOperand(1));
-    return DAG.getNode(Src.getOpcode(), DL, VT, LHS, RHS);
   }
 
   APInt DemandedElts = APInt::getAllOnes(VT.getVectorNumElements());
@@ -62302,175 +61846,6 @@ static SDValue combineKSHIFT(SDNode *N, SelectionDAG &DAG,
   return SDValue();
 }
 
-// Reassociates AND by splat to other operand when profitable.
-// Equivalent as removing the same bit within each matrix's row acts like the
-// corresponding source bit is zero.
-static SDValue combineAndOnGF2P8AFFINEQBOperand(SDNode *N, const SDLoc &DL,
-                                                SelectionDAG &DAG, EVT VT) {
-  using namespace SDPatternMatch;
-
-  SDValue X, Y, AndOp, SplatOp;
-  APInt Imm, SplatVal, ConstUndef;
-  SmallVector<APInt> ConstEltBits;
-
-  // TODO: Add reverse fold when X is constant
-  // Fold GF2P8AFFINEQB(x & Splat(C), M, Imm)
-  //  --> GF2P8AFFINEQB(x, M & Splat(C), Imm)
-  if (sd_match(N, m_TernaryOp(X86ISD::GF2P8AFFINEQB, m_Value(AndOp), m_Value(Y),
-                              m_ConstInt(Imm))) &&
-      sd_match(AndOp, m_And(m_Value(X), m_Value(SplatOp))) &&
-      getTargetConstantBitsFromNode(Y, Y.getScalarValueSizeInBits(), ConstUndef,
-                                    ConstEltBits, /*AllowWholeUndefs=*/false)) {
-    bool SplatIsConst =
-        X86::isConstantSplat(SplatOp, SplatVal, /*AllowPartialUndefs=*/false);
-
-    // Can still shorten the chain when constant folded with the matrix
-    if (!AndOp->hasOneUse() && !SplatIsConst)
-      return SDValue();
-
-    if (!(SplatIsConst || DAG.isSplatValue(SplatOp, /*AllowUndefs=*/false)) ||
-        SplatOp.getScalarValueSizeInBits() != 8)
-      return SDValue();
-
-    // ANDs with constants are not folded away this far into lowering
-    SDValue NewMatrix;
-    if (!SplatIsConst) {
-      NewMatrix = DAG.getNode(ISD::AND, DL, VT, SplatOp, Y);
-    } else {
-      SmallVector<SDValue> FoldedAnd;
-      for (APInt &Elt : ConstEltBits)
-        FoldedAnd.push_back(DAG.getConstant(SplatVal & Elt, DL, MVT::i8));
-
-      NewMatrix = DAG.getBuildVector(VT, DL, FoldedAnd);
-    }
-
-    return DAG.getNode(X86ISD::GF2P8AFFINEQB, DL, VT, X, NewMatrix,
-                       DAG.getTargetConstant(Imm, DL, MVT::i8));
-  }
-
-  return SDValue();
-}
-
-// Fold: GF2P8AFFINEQB(GF2P8AFFINEQB(X, YSub), YSup)
-//    => GF2P8AFFINEQB(X, YFolded)
-// Permuting the sub-matrix by the super-matrix at a byte, rather than bit,
-// granularity produces a matrix that performs both permutations at once.
-static SDValue combineNestedGF2P8AFFINEQB(SDNode *N, const SDLoc &DL,
-                                          SelectionDAG &DAG, EVT VT) {
-  using namespace SDPatternMatch;
-
-  unsigned VecWidth = VT.getSizeInBits();
-  unsigned NumElts = VT.getVectorNumElements();
-  unsigned EltWidth = VT.getScalarSizeInBits();
-
-  SDValue X, YSub, YSup;
-  APInt ImmSub, ImmSup, ConstUndef;
-  SmallVector<APInt> YSubEltBits, YSupEltBits;
-
-  if (!(sd_match(N, m_TernaryOp(X86ISD::GF2P8AFFINEQB,
-                                m_TernaryOp(X86ISD::GF2P8AFFINEQB, m_Value(X),
-                                            m_Value(YSub), m_ConstInt(ImmSub)),
-                                m_Value(YSup), m_ConstInt(ImmSup))) &&
-        getTargetConstantBitsFromNode(YSub, EltWidth, ConstUndef, YSubEltBits,
-                                      /*AllowWholeUndefs=*/false) &&
-        getTargetConstantBitsFromNode(YSup, EltWidth, ConstUndef, YSupEltBits,
-                                      /*AllowWholeUndefs=*/false)))
-    return SDValue();
-
-  APInt SubM(VecWidth, 0);
-  APInt SupM(VecWidth, 0);
-  for (unsigned I = 0; I != NumElts; ++I) {
-    SubM.insertBits(YSubEltBits[I], I * EltWidth);
-    SupM.insertBits(YSupEltBits[I], I * EltWidth);
-  }
-
-  // Immediate permute
-  APInt FoldedImm;
-  if (SupM.isSplat(64)) {
-    FoldedImm = getGFNIByteAffine(ImmSub, SupM.trunc(64), ImmSup);
-  } else {
-    // Immediate is shared and needs to be permuted in the same manner
-    if (ImmSub != 0)
-      return SDValue();
-    FoldedImm = ImmSup;
-  }
-
-  // Matrix permute
-  APInt FoldedMatrix = APInt(VecWidth, 0);
-  APInt LeastRowMask = APInt::getSplat(VecWidth, APInt(64, 0xFF));
-  APInt LeastBitInByte = APInt::getSplat(VecWidth, APInt(8, 0x01));
-  APInt RowSplatter = APInt(VecWidth, 0x0101010101010101ull);
-
-  for (unsigned Row = 0; Row != 8; ++Row) {
-    APInt RowSplat = (SubM & LeastRowMask) * RowSplatter;
-    SubM = SubM.lshr(EltWidth);
-
-    APInt ByteMaskIfSet = (SupM.lshr(7 - Row)) & LeastBitInByte;
-    ByteMaskIfSet *= 0xFF;
-
-    FoldedMatrix ^= RowSplat & ByteMaskIfSet;
-  }
-
-  SmallVector<SDValue> FoldedVector;
-  for (unsigned I = 0; I < NumElts; ++I) {
-    APInt FoldedElt = FoldedMatrix.extractBits(EltWidth, I * EltWidth);
-    FoldedVector.push_back(DAG.getConstant(FoldedElt, DL, MVT::i8));
-  }
-  SDValue NewMatrix = DAG.getBuildVector(VT, DL, FoldedVector);
-
-  return DAG.getNode(X86ISD::GF2P8AFFINEQB, DL, VT, X, NewMatrix,
-                     DAG.getTargetConstant(FoldedImm, DL, MVT::i8));
-}
-
-// Fold: GF2P8AFFINEQB(X ^ Splat(C), Y, Imm)
-//    => GF2P8AFFINEQB(X, Y, (u8)GF2P8AFFINEQB(C, Y, Imm))
-// Reassociating a XOR (by permuting it in the same manner as the input) allows
-// it to be applied for free using the immediate.
-static SDValue combineXorOnGF2P8AFFINEQBOperand(SDNode *N, const SDLoc &DL,
-                                                SelectionDAG &DAG, EVT VT) {
-  using namespace SDPatternMatch;
-
-  unsigned MatrixWidth = 64;
-
-  SDValue X, Y, SplatOp;
-  APInt Imm, SplatVal, ConstUndef;
-  SmallVector<APInt> MatEltBits;
-
-  if (sd_match(N, m_TernaryOp(X86ISD::GF2P8AFFINEQB,
-                              m_Xor(m_Value(X), m_Value(SplatOp)), m_Value(Y),
-                              m_ConstInt(Imm))) &&
-      X86::isConstantSplat(SplatOp, SplatVal, /*AllowPartialUndefs=*/false) &&
-      getTargetConstantBitsFromNode(Y, MatrixWidth, ConstUndef, MatEltBits,
-                                    /*AllowWholeUndefs=*/false)) {
-    // Immediate is shared, so all matrices need to permute it in the same way
-    if (!llvm::all_equal(MatEltBits))
-      return SDValue();
-
-    APInt NewImm = getGFNIByteAffine(SplatVal, MatEltBits[0], Imm);
-
-    return DAG.getNode(X86ISD::GF2P8AFFINEQB, DL, VT, X, Y,
-                       DAG.getTargetConstant(NewImm, DL, MVT::i8));
-  }
-
-  return SDValue();
-}
-
-static SDValue combineGF2P8AFFINEQB(SDNode *N, SelectionDAG &DAG) {
-  EVT VT = N->getValueType(0);
-  SDLoc dl(N);
-
-  if (SDValue R = combineAndOnGF2P8AFFINEQBOperand(N, dl, DAG, VT))
-    return R;
-
-  if (SDValue R = combineXorOnGF2P8AFFINEQBOperand(N, dl, DAG, VT))
-    return R;
-
-  if (SDValue R = combineNestedGF2P8AFFINEQB(N, dl, DAG, VT))
-    return R;
-
-  return SDValue();
-}
-
 // Optimize (fp16_to_fp (fp_to_fp16 X)) to VCVTPS2PH followed by VCVTPH2PS.
 // Done as a combine because the lowering for fp16_to_fp and fp_to_fp16 produce
 // extra instructions between the conversion due to going to scalar and back.
@@ -62534,7 +61909,7 @@ static SDValue combineFP_EXTEND(SDNode *N, SelectionDAG &DAG,
   if (Subtarget.hasFP16())
     return SDValue();
 
-  if (!SrcVT.isVectorOf(MVT::f16))
+  if (!SrcVT.isVector() || SrcVT.getVectorElementType() != MVT::f16)
     return SDValue();
 
   if (VT.getVectorElementType() != MVT::f32 &&
@@ -62648,7 +62023,8 @@ static SDValue combineFP_ROUND(SDNode *N, SelectionDAG &DAG,
 
   EVT SrcVT = Src.getValueType();
 
-  if (!VT.isVectorOf(MVT::f16) || SrcVT.getVectorElementType() != MVT::f32)
+  if (!VT.isVector() || VT.getVectorElementType() != MVT::f16 ||
+      SrcVT.getVectorElementType() != MVT::f32)
     return SDValue();
 
   SDValue Cvt, Chain;
@@ -62927,7 +62303,6 @@ SDValue X86TargetLowering::PerformDAGCombine(SDNode *N,
   case ISD::AVGCEILU:
   case ISD::AVGFLOORS:
   case ISD::AVGFLOORU:      return combineAVG(N, DAG, DCI, Subtarget);
-  case ISD::ATOMIC_LOAD:    return combineAtomicLoad(N, DAG, DCI);
   case ISD::LOAD:           return combineLoad(N, DAG, DCI, Subtarget);
   case ISD::MLOAD:          return combineMaskedLoad(N, DAG, DCI, Subtarget);
   case ISD::STORE:          return combineStore(N, DAG, DCI, Subtarget);
@@ -62948,11 +62323,10 @@ SDValue X86TargetLowering::PerformDAGCombine(SDNode *N,
   case X86ISD::VFCMULC:
   case X86ISD::VFMULC:      return combineFMulcFCMulc(N, DAG, Subtarget);
   case ISD::FNEG:           return combineFneg(N, DAG, DCI, Subtarget);
-  case ISD::VECREDUCE_MUL:  return combineVECREDUCE_MUL(N, DAG, Subtarget);
   case ISD::TRUNCATE:       return combineTruncate(N, DAG, Subtarget);
   case X86ISD::VTRUNC:      return combineVTRUNC(N, DAG, DCI);
   case X86ISD::VTRUNCS:
-  case X86ISD::VTRUNCUS:    return combineVTRUNCSAT(N, DAG, DCI, Subtarget);
+  case X86ISD::VTRUNCUS:    return combineVTRUNCSAT(N, DAG, DCI);
   case X86ISD::ANDNP:       return combineAndnp(N, DAG, DCI, Subtarget);
   case X86ISD::FAND:        return combineFAnd(N, DAG, Subtarget);
   case X86ISD::FANDN:       return combineFAndn(N, DAG, Subtarget);
@@ -63071,7 +62445,6 @@ SDValue X86TargetLowering::PerformDAGCombine(SDNode *N,
   case X86ISD::VPMADD52H:    return combineVPMADD52LH(N, DAG, DCI);
   case X86ISD::KSHIFTL:
   case X86ISD::KSHIFTR:     return combineKSHIFT(N, DAG, DCI);
-  case X86ISD::GF2P8AFFINEQB:  return combineGF2P8AFFINEQB(N, DAG);
   case ISD::FP16_TO_FP:     return combineFP16_TO_FP(N, DAG, Subtarget);
   case ISD::STRICT_FP_EXTEND:
   case ISD::FP_EXTEND:      return combineFP_EXTEND(N, DAG, DCI, Subtarget);
@@ -63082,7 +62455,8 @@ SDValue X86TargetLowering::PerformDAGCombine(SDNode *N,
   case X86ISD::MOVDQ2Q:     return combineMOVDQ2Q(N, DAG);
   case X86ISD::BEXTR:
   case X86ISD::BEXTRI:
-  case X86ISD::BZHI:        return combineBMI(N, DAG, DCI);
+  case X86ISD::BZHI:
+  case X86ISD::PDEP:        return combineBMI(N, DAG, DCI);
   case X86ISD::PCLMULQDQ:   return combinePCLMULQDQ(N, DAG, DCI);
   case ISD::INTRINSIC_WO_CHAIN:  return combineINTRINSIC_WO_CHAIN(N, DAG, DCI);
   case ISD::INTRINSIC_W_CHAIN:  return combineINTRINSIC_W_CHAIN(N, DAG, DCI);
@@ -63112,7 +62486,7 @@ bool X86TargetLowering::isTypeDesirableForOp(unsigned Opc, EVT VT) const {
     return false;
 
   // There are no vXi8 shifts.
-  if (Opc == ISD::SHL && VT.isVectorOf(MVT::i8))
+  if (Opc == ISD::SHL && VT.isVector() && VT.getVectorElementType() == MVT::i8)
     return false;
 
   // TODO: Almost no 8-bit ops are desirable because they have no actual
@@ -64453,7 +63827,7 @@ bool X86TargetLowering::hasStackProbeSymbol(const MachineFunction &MF) const {
 bool X86TargetLowering::hasInlineStackProbe(const MachineFunction &MF) const {
 
   // No inline stack probe for Windows, they have their own mechanism.
-  if (Subtarget.isOSWindowsOrUEFI() ||
+  if (Subtarget.isOSWindows() || Subtarget.isUEFI() ||
       MF.getFunction().hasFnAttribute("no-stack-arg-probe"))
     return false;
 
@@ -64479,7 +63853,8 @@ X86TargetLowering::getStackProbeSymbolName(const MachineFunction &MF) const {
 
   // Generally, if we aren't on Windows, the platform ABI does not include
   // support for stack probes, so don't emit them.
-  if (!Subtarget.isOSWindowsOrUEFI() || Subtarget.isTargetMachO() ||
+  if ((!Subtarget.isOSWindows() && !Subtarget.isUEFI()) ||
+      Subtarget.isTargetMachO() ||
       MF.getFunction().hasFnAttribute("no-stack-arg-probe"))
     return "";
 
diff --git a/llvm/test/CodeGen/X86/apx/sub.ll b/llvm/test/CodeGen/X86/apx/sub.ll
index 34af966465d93..bfc134ed1ed15 100644
--- a/llvm/test/CodeGen/X86/apx/sub.ll
+++ b/llvm/test/CodeGen/X86/apx/sub.ll
@@ -1,14 +1,19 @@
 ; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
 ; RUN: llc < %s -mtriple=x86_64-unknown -mattr=+ndd -verify-machineinstrs --show-mc-encoding | FileCheck %s --check-prefixes=CHECK,NDD
 ; RUN: llc < %s -mtriple=x86_64-unknown -mattr=+ndd,+prefer-ndd-mem -verify-machineinstrs --show-mc-encoding | FileCheck %s --check-prefixes=CHECK,MEM
-; RUN: llc < %s -mtriple=x86_64-unknown -mattr=+ndd,nf -verify-machineinstrs --show-mc-encoding | FileCheck --check-prefix=NF %s
-; RUN: llc < %s -mtriple=x86_64-unknown -mattr=+ndd,nf -x86-enable-apx-for-relocation=true -verify-machineinstrs --show-mc-encoding | FileCheck --check-prefix=NF %s
+; RUN: llc < %s -mtriple=x86_64-unknown -mattr=+ndd,nf -verify-machineinstrs --show-mc-encoding | FileCheck --check-prefixes=CHECK,NF %s
+; RUN: llc < %s -mtriple=x86_64-unknown -mattr=+ndd,nf -x86-enable-apx-for-relocation=true -verify-machineinstrs --show-mc-encoding | FileCheck --check-prefixes=CHECK,NF %s
 
 define i8 @sub8rr(i8 noundef %a, i8 noundef %b) {
-; CHECK-LABEL: sub8rr:
-; CHECK:       # %bb.0: # %entry
-; CHECK-NEXT:    subb %sil, %dil, %al # encoding: [0x62,0xf4,0x7c,0x18,0x28,0xf7]
-; CHECK-NEXT:    retq # encoding: [0xc3]
+; NDD-LABEL: sub8rr:
+; NDD:       # %bb.0: # %entry
+; NDD-NEXT:    subb %sil, %dil, %al # encoding: [0x62,0xf4,0x7c,0x18,0x28,0xf7]
+; NDD-NEXT:    retq # encoding: [0xc3]
+;
+; MEM-LABEL: sub8rr:
+; MEM:       # %bb.0: # %entry
+; MEM-NEXT:    subb %sil, %dil, %al # encoding: [0x62,0xf4,0x7c,0x18,0x28,0xf7]
+; MEM-NEXT:    retq # encoding: [0xc3]
 ;
 ; NF-LABEL: sub8rr:
 ; NF:       # %bb.0: # %entry
@@ -20,10 +25,15 @@ entry:
 }
 
 define i16 @sub16rr(i16 noundef %a, i16 noundef %b) {
-; CHECK-LABEL: sub16rr:
-; CHECK:       # %bb.0: # %entry
-; CHECK-NEXT:    subw %si, %di, %ax # encoding: [0x62,0xf4,0x7d,0x18,0x29,0xf7]
-; CHECK-NEXT:    retq # encoding: [0xc3]
+; NDD-LABEL: sub16rr:
+; NDD:       # %bb.0: # %entry
+; NDD-NEXT:    subw %si, %di, %ax # encoding: [0x62,0xf4,0x7d,0x18,0x29,0xf7]
+; NDD-NEXT:    retq # encoding: [0xc3]
+;
+; MEM-LABEL: sub16rr:
+; MEM:       # %bb.0: # %entry
+; MEM-NEXT:    subw %si, %di, %ax # encoding: [0x62,0xf4,0x7d,0x18,0x29,0xf7]
+; MEM-NEXT:    retq # encoding: [0xc3]
 ;
 ; NF-LABEL: sub16rr:
 ; NF:       # %bb.0: # %entry
@@ -35,10 +45,15 @@ entry:
 }
 
 define i32 @sub32rr(i32 noundef %a, i32 noundef %b) {
-; CHECK-LABEL: sub32rr:
-; CHECK:       # %bb.0: # %entry
-; CHECK-NEXT:    subl %esi, %edi, %eax # encoding: [0x62,0xf4,0x7c,0x18,0x29,0xf7]
-; CHECK-NEXT:    retq # encoding: [0xc3]
+; NDD-LABEL: sub32rr:
+; NDD:       # %bb.0: # %entry
+; NDD-NEXT:    subl %esi, %edi, %eax # encoding: [0x62,0xf4,0x7c,0x18,0x29,0xf7]
+; NDD-NEXT:    retq # encoding: [0xc3]
+;
+; MEM-LABEL: sub32rr:
+; MEM:       # %bb.0: # %entry
+; MEM-NEXT:    subl %esi, %edi, %eax # encoding: [0x62,0xf4,0x7c,0x18,0x29,0xf7]
+; MEM-NEXT:    retq # encoding: [0xc3]
 ;
 ; NF-LABEL: sub32rr:
 ; NF:       # %bb.0: # %entry
@@ -50,10 +65,15 @@ entry:
 }
 
 define i64 @sub64rr(i64 noundef %a, i64 noundef %b) {
-; CHECK-LABEL: sub64rr:
-; CHECK:       # %bb.0: # %entry
-; CHECK-NEXT:    subq %rsi, %rdi, %rax # encoding: [0x62,0xf4,0xfc,0x18,0x29,0xf7]
-; CHECK-NEXT:    retq # encoding: [0xc3]
+; NDD-LABEL: sub64rr:
+; NDD:       # %bb.0: # %entry
+; NDD-NEXT:    subq %rsi, %rdi, %rax # encoding: [0x62,0xf4,0xfc,0x18,0x29,0xf7]
+; NDD-NEXT:    retq # encoding: [0xc3]
+;
+; MEM-LABEL: sub64rr:
+; MEM:       # %bb.0: # %entry
+; MEM-NEXT:    subq %rsi, %rdi, %rax # encoding: [0x62,0xf4,0xfc,0x18,0x29,0xf7]
+; MEM-NEXT:    retq # encoding: [0xc3]
 ;
 ; NF-LABEL: sub64rr:
 ; NF:       # %bb.0: # %entry
@@ -161,10 +181,15 @@ entry:
 }
 
 define i16 @sub16ri8(i16 noundef %a) {
-; CHECK-LABEL: sub16ri8:
-; CHECK:       # %bb.0: # %entry
-; CHECK-NEXT:    subw $-128, %di, %ax # encoding: [0x62,0xf4,0x7d,0x18,0x83,0xef,0x80]
-; CHECK-NEXT:    retq # encoding: [0xc3]
+; NDD-LABEL: sub16ri8:
+; NDD:       # %bb.0: # %entry
+; NDD-NEXT:    subw $-128, %di, %ax # encoding: [0x62,0xf4,0x7d,0x18,0x83,0xef,0x80]
+; NDD-NEXT:    retq # encoding: [0xc3]
+;
+; MEM-LABEL: sub16ri8:
+; MEM:       # %bb.0: # %entry
+; MEM-NEXT:    subw $-128, %di, %ax # encoding: [0x62,0xf4,0x7d,0x18,0x83,0xef,0x80]
+; MEM-NEXT:    retq # encoding: [0xc3]
 ;
 ; NF-LABEL: sub16ri8:
 ; NF:       # %bb.0: # %entry
@@ -176,10 +201,15 @@ entry:
 }
 
 define i32 @sub32ri8(i32 noundef %a) {
-; CHECK-LABEL: sub32ri8:
-; CHECK:       # %bb.0: # %entry
-; CHECK-NEXT:    subl $-128, %edi, %eax # encoding: [0x62,0xf4,0x7c,0x18,0x83,0xef,0x80]
-; CHECK-NEXT:    retq # encoding: [0xc3]
+; NDD-LABEL: sub32ri8:
+; NDD:       # %bb.0: # %entry
+; NDD-NEXT:    subl $-128, %edi, %eax # encoding: [0x62,0xf4,0x7c,0x18,0x83,0xef,0x80]
+; NDD-NEXT:    retq # encoding: [0xc3]
+;
+; MEM-LABEL: sub32ri8:
+; MEM:       # %bb.0: # %entry
+; MEM-NEXT:    subl $-128, %edi, %eax # encoding: [0x62,0xf4,0x7c,0x18,0x83,0xef,0x80]
+; MEM-NEXT:    retq # encoding: [0xc3]
 ;
 ; NF-LABEL: sub32ri8:
 ; NF:       # %bb.0: # %entry
@@ -191,10 +221,15 @@ entry:
 }
 
 define i64 @sub64ri8(i64 noundef %a) {
-; CHECK-LABEL: sub64ri8:
-; CHECK:       # %bb.0: # %entry
-; CHECK-NEXT:    subq $-128, %rdi, %rax # encoding: [0x62,0xf4,0xfc,0x18,0x83,0xef,0x80]
-; CHECK-NEXT:    retq # encoding: [0xc3]
+; NDD-LABEL: sub64ri8:
+; NDD:       # %bb.0: # %entry
+; NDD-NEXT:    subq $-128, %rdi, %rax # encoding: [0x62,0xf4,0xfc,0x18,0x83,0xef,0x80]
+; NDD-NEXT:    retq # encoding: [0xc3]
+;
+; MEM-LABEL: sub64ri8:
+; MEM:       # %bb.0: # %entry
+; MEM-NEXT:    subq $-128, %rdi, %rax # encoding: [0x62,0xf4,0xfc,0x18,0x83,0xef,0x80]
+; MEM-NEXT:    retq # encoding: [0xc3]
 ;
 ; NF-LABEL: sub64ri8:
 ; NF:       # %bb.0: # %entry
@@ -206,10 +241,15 @@ entry:
 }
 
 define i8 @sub8ri(i8 noundef %a) {
-; CHECK-LABEL: sub8ri:
-; CHECK:       # %bb.0: # %entry
-; CHECK-NEXT:    addb $-123, %dil, %al # encoding: [0x62,0xf4,0x7c,0x18,0x80,0xc7,0x85]
-; CHECK-NEXT:    retq # encoding: [0xc3]
+; NDD-LABEL: sub8ri:
+; NDD:       # %bb.0: # %entry
+; NDD-NEXT:    addb $-123, %dil, %al # encoding: [0x62,0xf4,0x7c,0x18,0x80,0xc7,0x85]
+; NDD-NEXT:    retq # encoding: [0xc3]
+;
+; MEM-LABEL: sub8ri:
+; MEM:       # %bb.0: # %entry
+; MEM-NEXT:    addb $-123, %dil, %al # encoding: [0x62,0xf4,0x7c,0x18,0x80,0xc7,0x85]
+; MEM-NEXT:    retq # encoding: [0xc3]
 ;
 ; NF-LABEL: sub8ri:
 ; NF:       # %bb.0: # %entry
@@ -221,11 +261,17 @@ entry:
 }
 
 define i16 @sub16ri(i16 noundef %a) {
-; CHECK-LABEL: sub16ri:
-; CHECK:       # %bb.0: # %entry
-; CHECK-NEXT:    addw $-1234, %di, %ax # encoding: [0x62,0xf4,0x7d,0x18,0x81,0xc7,0x2e,0xfb]
-; CHECK-NEXT:    # imm = 0xFB2E
-; CHECK-NEXT:    retq # encoding: [0xc3]
+; NDD-LABEL: sub16ri:
+; NDD:       # %bb.0: # %entry
+; NDD-NEXT:    addw $-1234, %di, %ax # encoding: [0x62,0xf4,0x7d,0x18,0x81,0xc7,0x2e,0xfb]
+; NDD-NEXT:    # imm = 0xFB2E
+; NDD-NEXT:    retq # encoding: [0xc3]
+;
+; MEM-LABEL: sub16ri:
+; MEM:       # %bb.0: # %entry
+; MEM-NEXT:    addw $-1234, %di, %ax # encoding: [0x62,0xf4,0x7d,0x18,0x81,0xc7,0x2e,0xfb]
+; MEM-NEXT:    # imm = 0xFB2E
+; MEM-NEXT:    retq # encoding: [0xc3]
 ;
 ; NF-LABEL: sub16ri:
 ; NF:       # %bb.0: # %entry
@@ -242,22 +288,23 @@ define i32 @sub32ri(i32 noundef %a) {
 ; CHECK:       # %bb.0: # %entry
 ; CHECK-NEXT:    leal -123456(%rdi), %eax # encoding: [0x8d,0x87,0xc0,0x1d,0xfe,0xff]
 ; CHECK-NEXT:    retq # encoding: [0xc3]
-;
-; NF-LABEL: sub32ri:
-; NF:       # %bb.0: # %entry
-; NF-NEXT:    leal -123456(%rdi), %eax # encoding: [0x8d,0x87,0xc0,0x1d,0xfe,0xff]
-; NF-NEXT:    retq # encoding: [0xc3]
 entry:
     %sub = sub i32 %a, 123456
     ret i32 %sub
 }
 
 define i64 @sub64ri(i64 noundef %a) {
-; CHECK-LABEL: sub64ri:
-; CHECK:       # %bb.0: # %entry
-; CHECK-NEXT:    subq $-2147483648, %rdi, %rax # encoding: [0x62,0xf4,0xfc,0x18,0x81,0xef,0x00,0x00,0x00,0x80]
-; CHECK-NEXT:    # imm = 0x80000000
-; CHECK-NEXT:    retq # encoding: [0xc3]
+; NDD-LABEL: sub64ri:
+; NDD:       # %bb.0: # %entry
+; NDD-NEXT:    subq $-2147483648, %rdi, %rax # encoding: [0x62,0xf4,0xfc,0x18,0x81,0xef,0x00,0x00,0x00,0x80]
+; NDD-NEXT:    # imm = 0x80000000
+; NDD-NEXT:    retq # encoding: [0xc3]
+;
+; MEM-LABEL: sub64ri:
+; MEM:       # %bb.0: # %entry
+; MEM-NEXT:    subq $-2147483648, %rdi, %rax # encoding: [0x62,0xf4,0xfc,0x18,0x81,0xef,0x00,0x00,0x00,0x80]
+; MEM-NEXT:    # imm = 0x80000000
+; MEM-NEXT:    retq # encoding: [0xc3]
 ;
 ; NF-LABEL: sub64ri:
 ; NF:       # %bb.0: # %entry
@@ -545,15 +592,6 @@ define i8 @subflag8rr(i8 noundef %a, i8 noundef %b) {
 ; CHECK-NEXT:    cmovael %ecx, %eax # EVEX TO LEGACY Compression encoding: [0x0f,0x43,0xc1]
 ; CHECK-NEXT:    # kill: def $al killed $al killed $eax
 ; CHECK-NEXT:    retq # encoding: [0xc3]
-;
-; NF-LABEL: subflag8rr:
-; NF:       # %bb.0: # %entry
-; NF-NEXT:    xorl %eax, %eax # encoding: [0x31,0xc0]
-; NF-NEXT:    subb %sil, %dil, %cl # encoding: [0x62,0xf4,0x74,0x18,0x28,0xf7]
-; NF-NEXT:    movzbl %cl, %ecx # encoding: [0x0f,0xb6,0xc9]
-; NF-NEXT:    cmovael %ecx, %eax # EVEX TO LEGACY Compression encoding: [0x0f,0x43,0xc1]
-; NF-NEXT:    # kill: def $al killed $al killed $eax
-; NF-NEXT:    retq # encoding: [0xc3]
 entry:
     %sub = call i8 @llvm.usub.sat.i8(i8 %a, i8 %b)
     ret i8 %sub
@@ -567,14 +605,6 @@ define i16 @subflag16rr(i16 noundef %a, i16 noundef %b) {
 ; CHECK-NEXT:    cmovael %edi, %eax # EVEX TO LEGACY Compression encoding: [0x0f,0x43,0xc7]
 ; CHECK-NEXT:    # kill: def $ax killed $ax killed $eax
 ; CHECK-NEXT:    retq # encoding: [0xc3]
-;
-; NF-LABEL: subflag16rr:
-; NF:       # %bb.0: # %entry
-; NF-NEXT:    xorl %eax, %eax # encoding: [0x31,0xc0]
-; NF-NEXT:    subw %si, %di # EVEX TO LEGACY Compression encoding: [0x66,0x29,0xf7]
-; NF-NEXT:    cmovael %edi, %eax # EVEX TO LEGACY Compression encoding: [0x0f,0x43,0xc7]
-; NF-NEXT:    # kill: def $ax killed $ax killed $eax
-; NF-NEXT:    retq # encoding: [0xc3]
 entry:
     %sub = call i16 @llvm.usub.sat.i16(i16 %a, i16 %b)
     ret i16 %sub
@@ -587,13 +617,6 @@ define i32 @subflag32rr(i32 noundef %a, i32 noundef %b) {
 ; CHECK-NEXT:    subl %esi, %edi # EVEX TO LEGACY Compression encoding: [0x29,0xf7]
 ; CHECK-NEXT:    cmovael %edi, %eax # EVEX TO LEGACY Compression encoding: [0x0f,0x43,0xc7]
 ; CHECK-NEXT:    retq # encoding: [0xc3]
-;
-; NF-LABEL: subflag32rr:
-; NF:       # %bb.0: # %entry
-; NF-NEXT:    xorl %eax, %eax # encoding: [0x31,0xc0]
-; NF-NEXT:    subl %esi, %edi # EVEX TO LEGACY Compression encoding: [0x29,0xf7]
-; NF-NEXT:    cmovael %edi, %eax # EVEX TO LEGACY Compression encoding: [0x0f,0x43,0xc7]
-; NF-NEXT:    retq # encoding: [0xc3]
 entry:
     %sub = call i32 @llvm.usub.sat.i32(i32 %a, i32 %b)
     ret i32 %sub
@@ -606,13 +629,6 @@ define i64 @subflag64rr(i64 noundef %a, i64 noundef %b) {
 ; CHECK-NEXT:    subq %rsi, %rdi # EVEX TO LEGACY Compression encoding: [0x48,0x29,0xf7]
 ; CHECK-NEXT:    cmovaeq %rdi, %rax # EVEX TO LEGACY Compression encoding: [0x48,0x0f,0x43,0xc7]
 ; CHECK-NEXT:    retq # encoding: [0xc3]
-;
-; NF-LABEL: subflag64rr:
-; NF:       # %bb.0: # %entry
-; NF-NEXT:    xorl %eax, %eax # encoding: [0x31,0xc0]
-; NF-NEXT:    subq %rsi, %rdi # EVEX TO LEGACY Compression encoding: [0x48,0x29,0xf7]
-; NF-NEXT:    cmovaeq %rdi, %rax # EVEX TO LEGACY Compression encoding: [0x48,0x0f,0x43,0xc7]
-; NF-NEXT:    retq # encoding: [0xc3]
 entry:
     %sub = call i64 @llvm.usub.sat.i64(i64 %a, i64 %b)
     ret i64 %sub
@@ -743,14 +759,6 @@ define i16 @subflag16ri8(i16 noundef %a) {
 ; CHECK-NEXT:    cmovael %edi, %eax # EVEX TO LEGACY Compression encoding: [0x0f,0x43,0xc7]
 ; CHECK-NEXT:    # kill: def $ax killed $ax killed $eax
 ; CHECK-NEXT:    retq # encoding: [0xc3]
-;
-; NF-LABEL: subflag16ri8:
-; NF:       # %bb.0: # %entry
-; NF-NEXT:    xorl %eax, %eax # encoding: [0x31,0xc0]
-; NF-NEXT:    subw $123, %di # EVEX TO LEGACY Compression encoding: [0x66,0x83,0xef,0x7b]
-; NF-NEXT:    cmovael %edi, %eax # EVEX TO LEGACY Compression encoding: [0x0f,0x43,0xc7]
-; NF-NEXT:    # kill: def $ax killed $ax killed $eax
-; NF-NEXT:    retq # encoding: [0xc3]
 entry:
     %sub = call i16 @llvm.usub.sat.i16(i16 %a, i16 123)
     ret i16 %sub
@@ -763,13 +771,6 @@ define i32 @subflag32ri8(i32 noundef %a) {
 ; CHECK-NEXT:    subl $123, %edi # EVEX TO LEGACY Compression encoding: [0x83,0xef,0x7b]
 ; CHECK-NEXT:    cmovael %edi, %eax # EVEX TO LEGACY Compression encoding: [0x0f,0x43,0xc7]
 ; CHECK-NEXT:    retq # encoding: [0xc3]
-;
-; NF-LABEL: subflag32ri8:
-; NF:       # %bb.0: # %entry
-; NF-NEXT:    xorl %eax, %eax # encoding: [0x31,0xc0]
-; NF-NEXT:    subl $123, %edi # EVEX TO LEGACY Compression encoding: [0x83,0xef,0x7b]
-; NF-NEXT:    cmovael %edi, %eax # EVEX TO LEGACY Compression encoding: [0x0f,0x43,0xc7]
-; NF-NEXT:    retq # encoding: [0xc3]
 entry:
     %sub = call i32 @llvm.usub.sat.i32(i32 %a, i32 123)
     ret i32 %sub
@@ -782,13 +783,6 @@ define i64 @subflag64ri8(i64 noundef %a) {
 ; CHECK-NEXT:    subq $123, %rdi # EVEX TO LEGACY Compression encoding: [0x48,0x83,0xef,0x7b]
 ; CHECK-NEXT:    cmovaeq %rdi, %rax # EVEX TO LEGACY Compression encoding: [0x48,0x0f,0x43,0xc7]
 ; CHECK-NEXT:    retq # encoding: [0xc3]
-;
-; NF-LABEL: subflag64ri8:
-; NF:       # %bb.0: # %entry
-; NF-NEXT:    xorl %eax, %eax # encoding: [0x31,0xc0]
-; NF-NEXT:    subq $123, %rdi # EVEX TO LEGACY Compression encoding: [0x48,0x83,0xef,0x7b]
-; NF-NEXT:    cmovaeq %rdi, %rax # EVEX TO LEGACY Compression encoding: [0x48,0x0f,0x43,0xc7]
-; NF-NEXT:    retq # encoding: [0xc3]
 entry:
     %sub = call i64 @llvm.usub.sat.i64(i64 %a, i64 123)
     ret i64 %sub
@@ -803,15 +797,6 @@ define i8 @subflag8ri(i8 noundef %a) {
 ; CHECK-NEXT:    cmovael %ecx, %eax # EVEX TO LEGACY Compression encoding: [0x0f,0x43,0xc1]
 ; CHECK-NEXT:    # kill: def $al killed $al killed $eax
 ; CHECK-NEXT:    retq # encoding: [0xc3]
-;
-; NF-LABEL: subflag8ri:
-; NF:       # %bb.0: # %entry
-; NF-NEXT:    xorl %eax, %eax # encoding: [0x31,0xc0]
-; NF-NEXT:    subb $123, %dil, %cl # encoding: [0x62,0xf4,0x74,0x18,0x80,0xef,0x7b]
-; NF-NEXT:    movzbl %cl, %ecx # encoding: [0x0f,0xb6,0xc9]
-; NF-NEXT:    cmovael %ecx, %eax # EVEX TO LEGACY Compression encoding: [0x0f,0x43,0xc1]
-; NF-NEXT:    # kill: def $al killed $al killed $eax
-; NF-NEXT:    retq # encoding: [0xc3]
 entry:
     %sub = call i8 @llvm.usub.sat.i8(i8 %a, i8 123)
     ret i8 %sub
@@ -826,15 +811,6 @@ define i16 @subflag16ri(i16 noundef %a) {
 ; CHECK-NEXT:    cmovael %edi, %eax # EVEX TO LEGACY Compression encoding: [0x0f,0x43,0xc7]
 ; CHECK-NEXT:    # kill: def $ax killed $ax killed $eax
 ; CHECK-NEXT:    retq # encoding: [0xc3]
-;
-; NF-LABEL: subflag16ri:
-; NF:       # %bb.0: # %entry
-; NF-NEXT:    xorl %eax, %eax # encoding: [0x31,0xc0]
-; NF-NEXT:    subw $1234, %di # EVEX TO LEGACY Compression encoding: [0x66,0x81,0xef,0xd2,0x04]
-; NF-NEXT:    # imm = 0x4D2
-; NF-NEXT:    cmovael %edi, %eax # EVEX TO LEGACY Compression encoding: [0x0f,0x43,0xc7]
-; NF-NEXT:    # kill: def $ax killed $ax killed $eax
-; NF-NEXT:    retq # encoding: [0xc3]
 entry:
     %sub = call i16 @llvm.usub.sat.i16(i16 %a, i16 1234)
     ret i16 %sub
@@ -848,14 +824,6 @@ define i32 @subflag32ri(i32 noundef %a) {
 ; CHECK-NEXT:    # imm = 0x1E240
 ; CHECK-NEXT:    cmovael %edi, %eax # EVEX TO LEGACY Compression encoding: [0x0f,0x43,0xc7]
 ; CHECK-NEXT:    retq # encoding: [0xc3]
-;
-; NF-LABEL: subflag32ri:
-; NF:       # %bb.0: # %entry
-; NF-NEXT:    xorl %eax, %eax # encoding: [0x31,0xc0]
-; NF-NEXT:    subl $123456, %edi # EVEX TO LEGACY Compression encoding: [0x81,0xef,0x40,0xe2,0x01,0x00]
-; NF-NEXT:    # imm = 0x1E240
-; NF-NEXT:    cmovael %edi, %eax # EVEX TO LEGACY Compression encoding: [0x0f,0x43,0xc7]
-; NF-NEXT:    retq # encoding: [0xc3]
 entry:
     %sub = call i32 @llvm.usub.sat.i32(i32 %a, i32 123456)
     ret i32 %sub
@@ -869,14 +837,6 @@ define i64 @subflag64ri(i64 noundef %a) {
 ; CHECK-NEXT:    # imm = 0x1E240
 ; CHECK-NEXT:    cmovaeq %rdi, %rax # EVEX TO LEGACY Compression encoding: [0x48,0x0f,0x43,0xc7]
 ; CHECK-NEXT:    retq # encoding: [0xc3]
-;
-; NF-LABEL: subflag64ri:
-; NF:       # %bb.0: # %entry
-; NF-NEXT:    xorl %eax, %eax # encoding: [0x31,0xc0]
-; NF-NEXT:    subq $123456, %rdi # EVEX TO LEGACY Compression encoding: [0x48,0x81,0xef,0x40,0xe2,0x01,0x00]
-; NF-NEXT:    # imm = 0x1E240
-; NF-NEXT:    cmovaeq %rdi, %rax # EVEX TO LEGACY Compression encoding: [0x48,0x0f,0x43,0xc7]
-; NF-NEXT:    retq # encoding: [0xc3]
 entry:
     %sub = call i64 @llvm.usub.sat.i64(i64 %a, i64 123456)
     ret i64 %sub
@@ -902,22 +862,6 @@ define void @sub64ri_reloc(i64 %val) {
 ; CHECK-NEXT:    .cfi_def_cfa_offset 8
 ; CHECK-NEXT:  .LBB41_2: # %f
 ; CHECK-NEXT:    retq # encoding: [0xc3]
-;
-; NF-LABEL: sub64ri_reloc:
-; NF:       # %bb.0:
-; NF-NEXT:    cmpq $val, %rdi # encoding: [0x48,0x81,0xff,A,A,A,A]
-; NF-NEXT:    # fixup A - offset: 3, value: val, kind: reloc_signed_4byte
-; NF-NEXT:    jbe .LBB41_2 # encoding: [0x76,A]
-; NF-NEXT:    # fixup A - offset: 1, value: .LBB41_2, kind: FK_PCRel_1
-; NF-NEXT:  # %bb.1: # %t
-; NF-NEXT:    pushq %rax # encoding: [0x50]
-; NF-NEXT:    .cfi_def_cfa_offset 16
-; NF-NEXT:    callq f at PLT # encoding: [0xe8,A,A,A,A]
-; NF-NEXT:    # fixup A - offset: 1, value: f at PLT, kind: FK_PCRel_4
-; NF-NEXT:    popq %rax # encoding: [0x58]
-; NF-NEXT:    .cfi_def_cfa_offset 8
-; NF-NEXT:  .LBB41_2: # %f
-; NF-NEXT:    retq # encoding: [0xc3]
   %cmp = icmp ugt i64 %val, ptrtoint (ptr @val to i64)
   br i1 %cmp, label %t, label %f
 
@@ -934,11 +878,6 @@ define void @sub8mr_legacy(ptr %a, i8 noundef %b) {
 ; CHECK:       # %bb.0: # %entry
 ; CHECK-NEXT:    subb %sil, (%rdi) # encoding: [0x40,0x28,0x37]
 ; CHECK-NEXT:    retq # encoding: [0xc3]
-;
-; NF-LABEL: sub8mr_legacy:
-; NF:       # %bb.0: # %entry
-; NF-NEXT:    subb %sil, (%rdi) # encoding: [0x40,0x28,0x37]
-; NF-NEXT:    retq # encoding: [0xc3]
 entry:
   %t= load i8, ptr %a
   %sub = sub i8 %t, %b
@@ -951,11 +890,6 @@ define void @sub16mr_legacy(ptr %a, i16 noundef %b) {
 ; CHECK:       # %bb.0: # %entry
 ; CHECK-NEXT:    subw %si, (%rdi) # encoding: [0x66,0x29,0x37]
 ; CHECK-NEXT:    retq # encoding: [0xc3]
-;
-; NF-LABEL: sub16mr_legacy:
-; NF:       # %bb.0: # %entry
-; NF-NEXT:    subw %si, (%rdi) # encoding: [0x66,0x29,0x37]
-; NF-NEXT:    retq # encoding: [0xc3]
 entry:
   %t= load i16, ptr %a
   %sub = sub i16 %t, %b
@@ -968,11 +902,6 @@ define void @sub32mr_legacy(ptr %a, i32 noundef %b) {
 ; CHECK:       # %bb.0: # %entry
 ; CHECK-NEXT:    subl %esi, (%rdi) # encoding: [0x29,0x37]
 ; CHECK-NEXT:    retq # encoding: [0xc3]
-;
-; NF-LABEL: sub32mr_legacy:
-; NF:       # %bb.0: # %entry
-; NF-NEXT:    subl %esi, (%rdi) # encoding: [0x29,0x37]
-; NF-NEXT:    retq # encoding: [0xc3]
 entry:
   %t= load i32, ptr %a
   %sub = sub i32 %t, %b
@@ -985,11 +914,6 @@ define void @sub64mr_legacy(ptr %a, i64 noundef %b) {
 ; CHECK:       # %bb.0: # %entry
 ; CHECK-NEXT:    subq %rsi, (%rdi) # encoding: [0x48,0x29,0x37]
 ; CHECK-NEXT:    retq # encoding: [0xc3]
-;
-; NF-LABEL: sub64mr_legacy:
-; NF:       # %bb.0: # %entry
-; NF-NEXT:    subq %rsi, (%rdi) # encoding: [0x48,0x29,0x37]
-; NF-NEXT:    retq # encoding: [0xc3]
 entry:
   %t= load i64, ptr %a
   %sub = sub i64 %t, %b
@@ -1002,11 +926,6 @@ define void @sub8mi_legacy(ptr %a) {
 ; CHECK:       # %bb.0: # %entry
 ; CHECK-NEXT:    addb $-123, (%rdi) # encoding: [0x80,0x07,0x85]
 ; CHECK-NEXT:    retq # encoding: [0xc3]
-;
-; NF-LABEL: sub8mi_legacy:
-; NF:       # %bb.0: # %entry
-; NF-NEXT:    addb $-123, (%rdi) # encoding: [0x80,0x07,0x85]
-; NF-NEXT:    retq # encoding: [0xc3]
 entry:
   %t= load i8, ptr %a
   %sub = sub nsw i8 %t, 123
@@ -1020,12 +939,6 @@ define void @sub16mi_legacy(ptr %a) {
 ; CHECK-NEXT:    addw $-1234, (%rdi) # encoding: [0x66,0x81,0x07,0x2e,0xfb]
 ; CHECK-NEXT:    # imm = 0xFB2E
 ; CHECK-NEXT:    retq # encoding: [0xc3]
-;
-; NF-LABEL: sub16mi_legacy:
-; NF:       # %bb.0: # %entry
-; NF-NEXT:    addw $-1234, (%rdi) # encoding: [0x66,0x81,0x07,0x2e,0xfb]
-; NF-NEXT:    # imm = 0xFB2E
-; NF-NEXT:    retq # encoding: [0xc3]
 entry:
   %t= load i16, ptr %a
   %sub = sub nsw i16 %t, 1234
@@ -1039,12 +952,6 @@ define void @sub32mi_legacy(ptr %a) {
 ; CHECK-NEXT:    addl $-123456, (%rdi) # encoding: [0x81,0x07,0xc0,0x1d,0xfe,0xff]
 ; CHECK-NEXT:    # imm = 0xFFFE1DC0
 ; CHECK-NEXT:    retq # encoding: [0xc3]
-;
-; NF-LABEL: sub32mi_legacy:
-; NF:       # %bb.0: # %entry
-; NF-NEXT:    addl $-123456, (%rdi) # encoding: [0x81,0x07,0xc0,0x1d,0xfe,0xff]
-; NF-NEXT:    # imm = 0xFFFE1DC0
-; NF-NEXT:    retq # encoding: [0xc3]
 entry:
   %t= load i32, ptr %a
   %sub = sub nsw i32 %t, 123456
@@ -1058,12 +965,6 @@ define void @sub64mi_legacy(ptr %a) {
 ; CHECK-NEXT:    addq $-123456, (%rdi) # encoding: [0x48,0x81,0x07,0xc0,0x1d,0xfe,0xff]
 ; CHECK-NEXT:    # imm = 0xFFFE1DC0
 ; CHECK-NEXT:    retq # encoding: [0xc3]
-;
-; NF-LABEL: sub64mi_legacy:
-; NF:       # %bb.0: # %entry
-; NF-NEXT:    addq $-123456, (%rdi) # encoding: [0x48,0x81,0x07,0xc0,0x1d,0xfe,0xff]
-; NF-NEXT:    # imm = 0xFFFE1DC0
-; NF-NEXT:    retq # encoding: [0xc3]
 entry:
   %t= load i64, ptr %a
   %sub = sub nsw i64 %t, 123456
@@ -1238,3 +1139,95 @@ bb2:                                              ; preds = %bb2, %bb1
   store ptr null, ptr %arg2, align 8
   br i1 %arg3, label %bb1, label %bb2
 }
+;
+define i8 @usubsat8_const1(i8 noundef %a) {
+; CHECK-LABEL: usubsat8_const1:
+; CHECK:       # %bb.0: # %entry
+; CHECK-NEXT:    cmpb $1, %dil # encoding: [0x40,0x80,0xff,0x01]
+; CHECK-NEXT:    adcb $-1, %dil, %al # encoding: [0x62,0xf4,0x7c,0x18,0x80,0xd7,0xff]
+; CHECK-NEXT:    retq # encoding: [0xc3]
+entry:
+    %sub = call i8 @llvm.usub.sat.i8(i8 %a, i8 1)
+    ret i8 %sub
+}
+
+define i16 @usubsat16_const1(i16 noundef %a) {
+; CHECK-LABEL: usubsat16_const1:
+; CHECK:       # %bb.0: # %entry
+; CHECK-NEXT:    cmpw $1, %di # encoding: [0x66,0x83,0xff,0x01]
+; CHECK-NEXT:    adcw $-1, %di, %ax # encoding: [0x62,0xf4,0x7d,0x18,0x83,0xd7,0xff]
+; CHECK-NEXT:    retq # encoding: [0xc3]
+entry:
+    %sub = call i16 @llvm.usub.sat.i16(i16 %a, i16 1)
+    ret i16 %sub
+}
+
+define i32 @usubsat32_const1(i32 noundef %a) {
+; CHECK-LABEL: usubsat32_const1:
+; CHECK:       # %bb.0: # %entry
+; CHECK-NEXT:    cmpl $1, %edi # encoding: [0x83,0xff,0x01]
+; CHECK-NEXT:    adcl $-1, %edi, %eax # encoding: [0x62,0xf4,0x7c,0x18,0x83,0xd7,0xff]
+; CHECK-NEXT:    retq # encoding: [0xc3]
+entry:
+    %sub = call i32 @llvm.usub.sat.i32(i32 %a, i32 1)
+    ret i32 %sub
+}
+
+define i64 @usubsat64_const1(i64 noundef %a) {
+; CHECK-LABEL: usubsat64_const1:
+; CHECK:       # %bb.0: # %entry
+; CHECK-NEXT:    cmpq $1, %rdi # encoding: [0x48,0x83,0xff,0x01]
+; CHECK-NEXT:    adcq $-1, %rdi, %rax # encoding: [0x62,0xf4,0xfc,0x18,0x83,0xd7,0xff]
+; CHECK-NEXT:    retq # encoding: [0xc3]
+entry:
+    %sub = call i64 @llvm.usub.sat.i64(i64 %a, i64 1)
+    ret i64 %sub
+}
+
+define i32 @usubsat32_var(i32 noundef %a, i32 noundef %b) {
+; CHECK-LABEL: usubsat32_var:
+; CHECK:       # %bb.0: # %entry
+; CHECK-NEXT:    xorl %eax, %eax # encoding: [0x31,0xc0]
+; CHECK-NEXT:    subl %esi, %edi # EVEX TO LEGACY Compression encoding: [0x29,0xf7]
+; CHECK-NEXT:    cmovael %edi, %eax # EVEX TO LEGACY Compression encoding: [0x0f,0x43,0xc7]
+; CHECK-NEXT:    retq # encoding: [0xc3]
+entry:
+    %sub = call i32 @llvm.usub.sat.i32(i32 %a, i32 %b)
+    ret i32 %sub
+}
+
+define i32 @usubsat32_const123(i32 noundef %a) {
+; CHECK-LABEL: usubsat32_const123:
+; CHECK:       # %bb.0: # %entry
+; CHECK-NEXT:    xorl %eax, %eax # encoding: [0x31,0xc0]
+; CHECK-NEXT:    subl $123, %edi # EVEX TO LEGACY Compression encoding: [0x83,0xef,0x7b]
+; CHECK-NEXT:    cmovael %edi, %eax # EVEX TO LEGACY Compression encoding: [0x0f,0x43,0xc7]
+; CHECK-NEXT:    retq # encoding: [0xc3]
+entry:
+    %sub = call i32 @llvm.usub.sat.i32(i32 %a, i32 123)
+    ret i32 %sub
+}
+
+define i32 @uaddsat32_const1(i32 noundef %a) {
+; CHECK-LABEL: uaddsat32_const1:
+; CHECK:       # %bb.0: # %entry
+; CHECK-NEXT:    incl %edi, %eax # encoding: [0x62,0xf4,0x7c,0x18,0xff,0xc7]
+; CHECK-NEXT:    movl $-1, %ecx # encoding: [0xb9,0xff,0xff,0xff,0xff]
+; CHECK-NEXT:    cmovel %ecx, %eax # EVEX TO LEGACY Compression encoding: [0x0f,0x44,0xc1]
+; CHECK-NEXT:    retq # encoding: [0xc3]
+entry:
+    %add = call i32 @llvm.uadd.sat.i32(i32 %a, i32 1)
+    ret i32 %add
+}
+
+define i32 @uaddsat32_const_neg1(i32 noundef %a) {
+; CHECK-LABEL: uaddsat32_const_neg1:
+; CHECK:       # %bb.0: # %entry
+; CHECK-NEXT:    addl $-1, %edi, %eax # encoding: [0x62,0xf4,0x7c,0x18,0x83,0xc7,0xff]
+; CHECK-NEXT:    movl $-1, %ecx # encoding: [0xb9,0xff,0xff,0xff,0xff]
+; CHECK-NEXT:    cmovbl %ecx, %eax # EVEX TO LEGACY Compression encoding: [0x0f,0x42,0xc1]
+; CHECK-NEXT:    retq # encoding: [0xc3]
+entry:
+    %add = call i32 @llvm.uadd.sat.i32(i32 %a, i32 -1)
+    ret i32 %add
+}



More information about the llvm-commits mailing list