[llvm] [SLP]Model or-reduction of masked shifted lanes as a bitfield pack (PR #219731)

Alexey Bataev via llvm-commits llvm-commits at lists.llvm.org
Sun Sep 13 09:42:42 PDT 2026


https://github.com/alexey-bataev updated https://github.com/llvm/llvm-project/pull/219731

>From bdd550cda50721c234a809980c022944f01a1d1a Mon Sep 17 00:00:00 2001
From: Alexey Bataev <a.bataev at outlook.com>
Date: Sat, 29 Aug 2026 15:38:38 -0700
Subject: [PATCH] =?UTF-8?q?[=F0=9D=98=80=F0=9D=97=BD=F0=9D=97=BF]=20initia?=
 =?UTF-8?q?l=20version?=
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit

Created using spr 1.3.7
---
 .../Transforms/Vectorize/SLPVectorizer.cpp    | 316 +++++++++++++++---
 .../SLPVectorizer/SLPCostAnalysis.cpp         |  65 ++++
 .../Vectorize/SLPVectorizer/SLPCostAnalysis.h |  15 +
 .../Vectorize/SLPVectorizer/SLPUtils.cpp      | 150 +++++++++
 .../Vectorize/SLPVectorizer/SLPUtils.h        |  41 +++
 llvm/test/Transforms/PhaseOrdering/X86/avg.ll | 235 +++++++------
 .../SLPVectorizer/X86/bad-reduction.ll        |  18 +-
 .../X86/load-merge-inseltpoison.ll            |   4 +-
 .../SLPVectorizer/X86/load-merge.ll           |   4 +-
 .../SLPVectorizer/X86/pr48879-sroa.ll         |  66 ++--
 .../SLPVectorizer/X86/reduce-or-bitpack.ll    |  20 +-
 .../SLPVectorizer/non-power-of-2-bswap.ll     |   5 +-
 12 files changed, 753 insertions(+), 186 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 5490fe24d5529..2b2e3c5758e53 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -904,7 +904,8 @@ class slpvectorizer::BoUpSLP {
            (getRootNode().CombinedOp == TreeEntry::ReducedBitcast ||
             getRootNode().CombinedOp == TreeEntry::ReducedBitcastBSwap ||
             getRootNode().CombinedOp == TreeEntry::ReducedBitcastLoads ||
-            getRootNode().CombinedOp == TreeEntry::ReducedBitcastBSwapLoads) &&
+            getRootNode().CombinedOp == TreeEntry::ReducedBitcastBSwapLoads ||
+            getRootNode().CombinedOp == TreeEntry::ReducedBitPack) &&
            getRootNode().State == TreeEntry::Vectorize;
   }
 
@@ -2933,6 +2934,20 @@ class slpvectorizer::BoUpSLP {
   bool matchesShlZExt(const TreeEntry &TE, OrdersType &Order, bool &IsBSwap,
                       bool &ForLoads) const;
 
+  /// Computes the packing layout of an or reduction of masked shifted values
+  /// packing per-lane byte fields of the scalar result (and(shl(x, s), m) per
+  /// lane). The field analysis proves the disjointness of the fields, so the
+  /// reduction operations are not required to be marked disjoint.
+  std::optional<BitPackInfo> computeBitPackLayout(const TreeEntry &TE) const;
+
+  /// Checks if the tree entry is the root of such a reduction, the tree nodes
+  /// are in the expected state and it is profitable to model as a bitcast.
+  bool matchesBitPack(const TreeEntry &TE) const;
+
+  /// True if the node's lanes are a zext from i8, so compacting them to bytes
+  /// is free.
+  bool isByteZExtNode(const TreeEntry &TE) const;
+
   /// Checks if the \p SelectTE matches zext+selects, which can be inversed for
   /// better codegen in case like zext (icmp ne), select (icmp eq), ....
   bool matchesInversedZExtSelect(
@@ -3086,6 +3101,7 @@ class slpvectorizer::BoUpSLP {
       ReducedBitcastBSwap,
       ReducedBitcastLoads,
       ReducedBitcastBSwapLoads,
+      ReducedBitPack,
       ReducedCmpBitcast,
     };
     CombinedOpcode CombinedOp = NotCombinedOp;
@@ -14827,6 +14843,126 @@ bool BoUpSLP::matchesShlZExt(const TreeEntry &TE, OrdersType &Order,
   return BitcastCost < VecCost;
 }
 
+std::optional<BitPackInfo>
+BoUpSLP::computeBitPackLayout(const TreeEntry &TE) const {
+  assert(TE.hasState() &&
+         (TE.getOpcode() == Instruction::And ||
+          TE.getOpcode() == Instruction::Shl) &&
+         "Expected And or Shl node.");
+  Type *ScalarTy = TE.getMainOp()->getType();
+  const unsigned BitWidth = DL->getTypeSizeInBits(ScalarTy);
+  const TreeEntry *ShlTE = &TE;
+  const TreeEntry *MaskTE = nullptr;
+  if (TE.getOpcode() == Instruction::And) {
+    MaskTE = getOperandEntry(&TE, 1);
+    ShlTE = getOperandEntry(&TE, 0);
+  }
+  const TreeEntry *ShiftTE = getOperandEntry(ShlTE, 1);
+  const TreeEntry *LhsTE = getOperandEntry(ShlTE, 0);
+
+  const unsigned VF = TE.getVectorFactor();
+  SmallVector<APInt> PossibleBits(VF), Masks(VF);
+  SmallVector<uint64_t> ShlAmts(VF);
+  auto IsCopyable = [](const TreeEntry *TE, unsigned Idx) {
+    return TE->hasCopyableElements() && TE->isCopyableElement(TE->Scalars[Idx]);
+  };
+  for (unsigned Idx : seq(VF)) {
+    uint64_t ShlAmt = 0;
+    const APInt *C;
+    if (!IsCopyable(ShlTE, Idx)) {
+      if (!match(ShiftTE->Scalars[Idx], m_APInt(C)) || C->uge(BitWidth))
+        return std::nullopt;
+      ShlAmt = C->getZExtValue();
+    }
+    APInt Mask = APInt::getAllOnes(BitWidth);
+    if (MaskTE && !IsCopyable(&TE, Idx)) {
+      if (!match(MaskTE->Scalars[Idx], m_APInt(C)))
+        return std::nullopt;
+      Mask = *C;
+    }
+    Value *X = LhsTE->Scalars[Idx];
+    if (isa<PoisonValue>(X)) {
+      PossibleBits[Idx] = APInt(BitWidth, 0);
+    } else {
+      PossibleBits[Idx] =
+          APInt::getLowBitsSet(BitWidth, getScalarMaxValue(X).getActiveBits()) &
+          ~computeKnownBits(X, SimplifyQuery(*DL)).Zero;
+    }
+    ShlAmts[Idx] = ShlAmt;
+    Masks[Idx] = Mask;
+  }
+  return computeBitPackInfo(BitWidth, PossibleBits, ShlAmts, Masks);
+}
+
+bool BoUpSLP::isByteZExtNode(const TreeEntry &TE) const {
+  return TE.getOpcode() == Instruction::ZExt &&
+         cast<ZExtInst>(TE.getMainOp())->getSrcTy()->isIntegerTy(8);
+}
+
+bool BoUpSLP::matchesBitPack(const TreeEntry &TE) const {
+  assert(TE.hasState() &&
+         (TE.getOpcode() == Instruction::And ||
+          TE.getOpcode() == Instruction::Shl) &&
+         "Expected And or Shl node.");
+  auto IsVectorizedOneUseNode = [&](const TreeEntry *N) {
+    return N->State == TreeEntry::Vectorize && !N->isAltShuffle() &&
+           N->ReorderIndices.empty() && N->ReuseShuffleIndices.empty() &&
+           !MinBWs.contains(N) && all_of(N->Scalars, [&](Value *V) {
+             return (N->hasCopyableElements() && N->isCopyableElement(V)) ||
+                    V->hasOneUse();
+           });
+  };
+  if (!IsVectorizedOneUseNode(&TE))
+    return false;
+  Type *ScalarTy = TE.getMainOp()->getType();
+  if (!ScalarTy->isIntegerTy())
+    return false;
+  const unsigned BitWidth = DL->getTypeSizeInBits(ScalarTy);
+  if (!isPowerOf2_64(BitWidth))
+    return false;
+
+  const TreeEntry *ShlTE = &TE;
+  const TreeEntry *MaskTE = nullptr;
+  if (TE.getOpcode() == Instruction::And) {
+    MaskTE = getOperandEntry(&TE, 1);
+    ShlTE = getOperandEntry(&TE, 0);
+    if (!ShlTE->hasState() || ShlTE->getOpcode() != Instruction::Shl)
+      return false;
+  }
+  if (!IsVectorizedOneUseNode(ShlTE))
+    return false;
+  const TreeEntry *ShiftTE = getOperandEntry(ShlTE, 1);
+  const TreeEntry *LhsTE = getOperandEntry(ShlTE, 0);
+  if (!ShiftTE->isGather() || (MaskTE && !MaskTE->isGather()))
+    return false;
+  if (LhsTE->State != TreeEntry::Vectorize || LhsTE->isAltShuffle() ||
+      !LhsTE->ReorderIndices.empty() || !LhsTE->ReuseShuffleIndices.empty() ||
+      MinBWs.contains(LhsTE))
+    return false;
+  std::optional<BitPackInfo> Info = computeBitPackLayout(TE);
+  if (!Info)
+    return false;
+  auto *VecTy =
+      cast<VectorType>(getWidenedType(ScalarTy, TE.getVectorFactor()));
+  FastMathFlags FMF;
+  InstructionCost VecCost =
+      TTI->getArithmeticReductionCost(Instruction::Or, VecTy, FMF, CostKind) +
+      TTI->getArithmeticInstrCost(Instruction::Shl, VecTy, CostKind,
+                                  getOperandInfo(LhsTE->Scalars),
+                                  getOperandInfo(ShiftTE->Scalars), /*Args=*/{},
+                                  ShlTE->getMainOp(), TLI);
+  if (TE.getOpcode() == Instruction::And)
+    VecCost += TTI->getArithmeticInstrCost(
+        Instruction::And, VecTy, CostKind, getOperandInfo(ShlTE->Scalars),
+        getOperandInfo(MaskTE->Scalars), /*Args=*/{}, TE.getMainOp(), TLI);
+  unsigned ShiftWidth;
+  bool FreeByteTrunc = isByteZExtNode(*LhsTE);
+  InstructionCost PackCost =
+      getBitPackCost(*TTI, cast<FixedVectorType>(VecTy), ScalarTy, *Info,
+                     FreeByteTrunc, CostKind, TLI, TE.getMainOp(), ShiftWidth);
+  return PackCost.isValid() && PackCost <= VecCost;
+}
+
 bool BoUpSLP::matchesInversedZExtSelect(
     const TreeEntry &SelectTE,
     SmallVectorImpl<unsigned> &InversedCmpsIndices) const {
@@ -15464,8 +15600,10 @@ void BoUpSLP::transformNodes() {
       }
       break;
     }
-    case Instruction::Shl: {
-      // Shl is not reassociated; guard since this case indexes operands 0/1.
+    case Instruction::Shl:
+    case Instruction::And: {
+      // Shl/And are not reassociated; guard since this case indexes operands
+      // 0/1.
       if (E.getNumOperands() != 2)
         break;
       if (E.Idx != 0 || DL->isBigEndian())
@@ -15473,47 +15611,79 @@ void BoUpSLP::transformNodes() {
       if (!UserIgnoreList)
         break;
       // Check that all reduction operands are disjoint or instructions.
+      if (E.getOpcode() == Instruction::Shl &&
+          all_of(*UserIgnoreList, [](Value *V) {
+            return match(V, m_DisjointOr(m_Value(), m_Value()));
+          })) {
+        OrdersType Order;
+        bool IsBSwap;
+        bool ForLoads;
+        if (matchesShlZExt(E, Order, IsBSwap, ForLoads)) {
+          // This node is a (reduced disjoint or) bitcast node.
+          TreeEntry::CombinedOpcode Code =
+              IsBSwap ? (ForLoads ? TreeEntry::ReducedBitcastBSwapLoads
+                                  : TreeEntry::ReducedBitcastBSwap)
+                      : (ForLoads ? TreeEntry::ReducedBitcastLoads
+                                  : TreeEntry::ReducedBitcast);
+          E.CombinedOp = Code;
+          E.ReorderIndices = std::move(Order);
+          TreeEntry *ZExtEntry = getOperandEntry(&E, 0);
+          assert(ZExtEntry->UserTreeIndex &&
+                 ZExtEntry->State == TreeEntry::Vectorize &&
+                 ZExtEntry->getOpcode() == Instruction::ZExt &&
+                 "Expected ZExt node.");
+          // The ZExt node is part of the combined node.
+          ZExtEntry->State = TreeEntry::CombinedVectorize;
+          ZExtEntry->CombinedOp = Code;
+          if (ForLoads) {
+            TreeEntry *LoadsEntry = getOperandEntry(ZExtEntry, 0);
+            assert(LoadsEntry->UserTreeIndex &&
+                   LoadsEntry->State == TreeEntry::Vectorize &&
+                   LoadsEntry->getOpcode() == Instruction::Load &&
+                   "Expected Load node.");
+            // The Load node is part of the combined node.
+            LoadsEntry->State = TreeEntry::CombinedVectorize;
+            LoadsEntry->CombinedOp = Code;
+          }
+          TreeEntry *ConstEntry = getOperandEntry(&E, 1);
+          assert(ConstEntry->UserTreeIndex && ConstEntry->isGather() &&
+                 "Expected ZExt node.");
+          // The ConstNode node is part of the combined node.
+          ConstEntry->State = TreeEntry::CombinedVectorize;
+          ConstEntry->CombinedOp = Code;
+          break;
+        }
+      }
+      // Check that all reduction operands are or instructions.
       if (any_of(*UserIgnoreList, [](Value *V) {
-            return !match(V, m_DisjointOr(m_Value(), m_Value()));
+            return !match(V, m_Or(m_Value(), m_Value()));
           }))
         break;
-      OrdersType Order;
-      bool IsBSwap;
-      bool ForLoads;
-      if (!matchesShlZExt(E, Order, IsBSwap, ForLoads))
+      if (!matchesBitPack(E))
         break;
-      // This node is a (reduced disjoint or) bitcast node.
-      TreeEntry::CombinedOpcode Code =
-          IsBSwap ? (ForLoads ? TreeEntry::ReducedBitcastBSwapLoads
-                              : TreeEntry::ReducedBitcastBSwap)
-                  : (ForLoads ? TreeEntry::ReducedBitcastLoads
-                              : TreeEntry::ReducedBitcast);
-      E.CombinedOp = Code;
-      E.ReorderIndices = std::move(Order);
-      TreeEntry *ZExtEntry = getOperandEntry(&E, 0);
-      assert(ZExtEntry->UserTreeIndex &&
-             ZExtEntry->State == TreeEntry::Vectorize &&
-             ZExtEntry->getOpcode() == Instruction::ZExt &&
-             "Expected ZExt node.");
-      // The ZExt node is part of the combined node.
-      ZExtEntry->State = TreeEntry::CombinedVectorize;
-      ZExtEntry->CombinedOp = Code;
-      if (ForLoads) {
-        TreeEntry *LoadsEntry = getOperandEntry(ZExtEntry, 0);
-        assert(LoadsEntry->UserTreeIndex &&
-               LoadsEntry->State == TreeEntry::Vectorize &&
-               LoadsEntry->getOpcode() == Instruction::Load &&
-               "Expected Load node.");
-        // The Load node is part of the combined node.
-        LoadsEntry->State = TreeEntry::CombinedVectorize;
-        LoadsEntry->CombinedOp = Code;
-      }
-      TreeEntry *ConstEntry = getOperandEntry(&E, 1);
-      assert(ConstEntry->UserTreeIndex && ConstEntry->isGather() &&
-             "Expected ZExt node.");
-      // The ConstNode node is part of the combined node.
-      ConstEntry->State = TreeEntry::CombinedVectorize;
-      ConstEntry->CombinedOp = Code;
+      // This node is a (reduced or) bit pack node.
+      E.CombinedOp = TreeEntry::ReducedBitPack;
+      TreeEntry *ShlEntry = &E;
+      if (E.getOpcode() == Instruction::And) {
+        ShlEntry = getOperandEntry(&E, 0);
+        assert(ShlEntry->UserTreeIndex &&
+               ShlEntry->State == TreeEntry::Vectorize &&
+               ShlEntry->getOpcode() == Instruction::Shl &&
+               "Expected Shl node.");
+        // The Shl node is part of the combined node.
+        ShlEntry->State = TreeEntry::CombinedVectorize;
+        ShlEntry->CombinedOp = TreeEntry::ReducedBitPack;
+        TreeEntry *MaskEntry = getOperandEntry(&E, 1);
+        assert(MaskEntry->UserTreeIndex && MaskEntry->isGather() &&
+               "Expected constants only.");
+        MaskEntry->State = TreeEntry::CombinedVectorize;
+        MaskEntry->CombinedOp = TreeEntry::ReducedBitPack;
+      }
+      TreeEntry *ShiftEntry = getOperandEntry(ShlEntry, 1);
+      assert(ShiftEntry->UserTreeIndex && ShiftEntry->isGather() &&
+             "Expected constants only.");
+      ShiftEntry->State = TreeEntry::CombinedVectorize;
+      ShiftEntry->CombinedOp = TreeEntry::ReducedBitPack;
       break;
     }
     default:
@@ -16921,7 +17091,8 @@ BoUpSLP::getEntryCost(const TreeEntry *E, ArrayRef<Value *> VectorizedVals,
     ScalarTy = IntegerType::get(F->getContext(), It->second.first);
     if (VecTy)
       ScalarTy = getWidenedType(ScalarTy, VecTy->getNumElements());
-  } else if (E->Idx == 0 && isReducedBitcastRoot()) {
+  } else if (E->Idx == 0 && isReducedBitcastRoot() &&
+             getRootNode().CombinedOp != TreeEntry::ReducedBitPack) {
     const TreeEntry *ZExt = getOperandEntry(E, /*Idx=*/0);
     ScalarTy = cast<CastInst>(ZExt->getMainOp())->getSrcTy();
   }
@@ -17743,6 +17914,37 @@ BoUpSLP::getEntryCost(const TreeEntry *E, ArrayRef<Value *> VectorizedVals,
     };
     return GetCostDiff(GetScalarCost, GetVectorCost);
   }
+  case TreeEntry::ReducedBitPack: {
+    auto GetScalarCost = [&, &TTI = *TTI](unsigned Idx) {
+      Instruction *I;
+      if (!match(UniqueValues[Idx], m_Instruction(I)))
+        return InstructionCost(TTI::TCC_Free);
+      InstructionCost ScalarCost = TTI.getInstructionCost(I, CostKind);
+      if (match(I, m_And(m_Shl(m_Value(), m_Value()), m_Value())))
+        ScalarCost += TTI.getInstructionCost(
+            cast<Instruction>(I->getOperand(0)), CostKind);
+      return ScalarCost;
+    };
+    auto GetVectorCost = [&, &TTI = *TTI](InstructionCost CommonCost) {
+      const TreeEntry *ShlTE = E->getOpcode() == Instruction::And
+                                   ? getOperandEntry(E, /*Idx=*/0)
+                                   : E;
+      const TreeEntry *LhsTE = getOperandEntry(ShlTE, /*Idx=*/0);
+      std::optional<BitPackInfo> Info = computeBitPackLayout(*E);
+      if (!Info)
+        return InstructionCost::getInvalid();
+      unsigned ShiftWidth;
+      bool FreeByteTrunc = isByteZExtNode(*LhsTE);
+      return getBitPackCost(
+                 TTI,
+                 cast<FixedVectorType>(getWidenedType(
+                     LhsTE->getMainOp()->getType(), LhsTE->getVectorFactor())),
+                 ScalarTy, *Info, FreeByteTrunc, CostKind, TLI, E->getMainOp(),
+                 ShiftWidth) +
+             CommonCost;
+    };
+    return GetCostDiff(GetScalarCost, GetVectorCost);
+  }
   case TreeEntry::ReducedCmpBitcast: {
     auto GetScalarCost = [&, &TTI = *TTI](unsigned Idx) {
       if (isa<PoisonValue>(UniqueValues[Idx]))
@@ -18786,6 +18988,7 @@ InstructionCost BoUpSLP::getSpillCost() {
         TEPtr->CombinedOp == TreeEntry::ReducedBitcastBSwap ||
         TEPtr->CombinedOp == TreeEntry::ReducedBitcastLoads ||
         TEPtr->CombinedOp == TreeEntry::ReducedBitcastBSwapLoads ||
+        TEPtr->CombinedOp == TreeEntry::ReducedBitPack ||
         TEPtr->CombinedOp == TreeEntry::ReducedCmpBitcast) {
       ScalarOrPseudoEntries.insert(TEPtr.get());
       continue;
@@ -23478,6 +23681,7 @@ Value *BoUpSLP::vectorizeTree(TreeEntry *E) {
     case TreeEntry::ReducedBitcastBSwap:
     case TreeEntry::ReducedBitcastLoads:
     case TreeEntry::ReducedBitcastBSwapLoads:
+    case TreeEntry::ReducedBitPack:
     case TreeEntry::ReducedCmpBitcast:
       ShuffleOrOp = E->CombinedOp;
       break;
@@ -24993,6 +25197,34 @@ Value *BoUpSLP::vectorizeTree(TreeEntry *E) {
       E->VectorizedValue = V;
       return V;
     }
+    case TreeEntry::ReducedBitPack: {
+      assert(UserIgnoreList && "Expected reduction operations only.");
+      setInsertPointAfterBundle(E);
+      TreeEntry *ShlTE = E->getOpcode() == Instruction::And
+                             ? getOperandEntry(E, /*Idx=*/0)
+                             : E;
+      SmallVector<TreeEntry *> CombinedTEs;
+      if (ShlTE != E)
+        CombinedTEs.append({ShlTE, getOperandEntry(E, /*Idx=*/1)});
+      CombinedTEs.push_back(getOperandEntry(ShlTE, /*Idx=*/1));
+      for (TreeEntry *TE : CombinedTEs)
+        TE->VectorizedValue = PoisonValue::get(getWidenedType(
+            TE->Scalars.front()->getType(), TE->getVectorFactor()));
+      Value *X = vectorizeOperand(ShlTE, /*NodeIdx=*/0);
+      std::optional<BitPackInfo> Info = computeBitPackLayout(*E);
+      assert(Info && "Expected bit pack info.");
+      unsigned ShiftWidth;
+      const TreeEntry *LhsTE = getOperandEntry(ShlTE, /*Idx=*/0);
+      bool FreeByteTrunc = isByteZExtNode(*LhsTE);
+      InstructionCost PackCost = getBitPackCost(
+          *TTI, cast<FixedVectorType>(X->getType()), ScalarTy, *Info,
+          FreeByteTrunc, CostKind, TLI, E->getMainOp(), ShiftWidth);
+      assert(PackCost.isValid() && "Expected valid cost.");
+      Value *V = buildBitPack(Builder, X, *Info, ShiftWidth);
+      ++NumVectorInstructions;
+      E->VectorizedValue = V;
+      return V;
+    }
     case TreeEntry::ReducedCmpBitcast: {
       assert(UserIgnoreList && "Expected reduction operations only.");
       setInsertPointAfterBundle(E);
@@ -25511,6 +25743,7 @@ Value *BoUpSLP::vectorizeTree(
         (TE->State == TreeEntry::CombinedVectorize &&
          (TE->CombinedOp == TreeEntry::ReducedBitcast ||
           TE->CombinedOp == TreeEntry::ReducedBitcastBSwap ||
+          TE->CombinedOp == TreeEntry::ReducedBitPack ||
           ((TE->CombinedOp == TreeEntry::ReducedBitcastLoads ||
             TE->CombinedOp == TreeEntry::ReducedBitcastBSwapLoads ||
             TE->CombinedOp == TreeEntry::ReducedCmpBitcast) &&
@@ -26191,6 +26424,7 @@ Value *BoUpSLP::vectorizeTree(
         Entry->CombinedOp == TreeEntry::ReducedBitcastBSwap ||
         Entry->CombinedOp == TreeEntry::ReducedBitcastLoads ||
         Entry->CombinedOp == TreeEntry::ReducedBitcastBSwapLoads ||
+        Entry->CombinedOp == TreeEntry::ReducedBitPack ||
         Entry->CombinedOp == TreeEntry::ReducedCmpBitcast) {
       // Skip constant node
       if (!Entry->hasState()) {
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.cpp
index de4c6ef1c7d1f..de10ca07c1c13 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.cpp
@@ -141,4 +141,69 @@ InstructionCost getBlendedLoadCost(const TargetTransformInfo &TTI, Type *VecTy,
                                 CmpInst::BAD_ICMP_PREDICATE, CostKind);
 }
 
+InstructionCost getBitPackCost(const TargetTransformInfo &TTI,
+                               FixedVectorType *SrcTy, Type *ResultTy,
+                               const BitPackInfo &Info, bool FreeByteTrunc,
+                               TTI::TargetCostKind CostKind,
+                               const TargetLibraryInfo *TLI,
+                               const Instruction *CxtI, unsigned &ShiftWidth) {
+  unsigned BitWidth = SrcTy->getScalarSizeInBits();
+  unsigned NumElts = SrcTy->getNumElements();
+  uint64_t MaxAmt = *max_element(Info.LShrAmts);
+  // The shift amounts form a constant vector.
+  TTI::OperandValueInfo ShiftAmtInfo = {
+      all_of(Info.LShrAmts,
+             [&](uint64_t A) { return A == Info.LShrAmts.front(); })
+          ? TTI::OK_UniformConstantValue
+          : TTI::OK_NonUniformConstantValue,
+      all_of(Info.LShrAmts, isPowerOf2_64) ? TTI::OP_PowerOf2 : TTI::OP_None};
+  // After the shift the field content of each lane sits in the low bits of
+  // the lane, so the packing is a single byte shuffle of the shifted lanes.
+  // Pick the cheapest shift width: the narrowest type still holding the field
+  // content is not always the cheapest (e.g. missing narrow variable shifts).
+  Type *Int8Ty = IntegerType::get(SrcTy->getContext(), 8);
+  unsigned OutBytes = BitWidth / 8;
+  auto *PackTy = FixedVectorType::get(Int8Ty, OutBytes);
+  unsigned MinShiftWidth = 8;
+  while (MinShiftWidth < MaxAmt + Info.FieldWidth)
+    MinShiftWidth *= 2;
+  InstructionCost NewCost = InstructionCost::getInvalid();
+  ShiftWidth = 0;
+  for (unsigned W2 = MinShiftWidth; W2 <= BitWidth; W2 *= 2) {
+    auto *ShiftTy = FixedVectorType::get(
+        IntegerType::get(SrcTy->getContext(), W2), NumElts);
+    unsigned BytesPerLane = W2 / 8;
+    unsigned InBytes = NumElts * BytesPerLane;
+    SmallVector<int> Mask =
+        getBitPackMask(Info, OutBytes, NumElts, BytesPerLane);
+    InstructionCost C =
+        TTI.getCastInstrCost(Instruction::BitCast, ResultTy, PackTy,
+                             TTI::CastContextHint::None, CostKind);
+    // A plain byte reversal of the shifted lanes is a bswap, no shuffle.
+    if (ShuffleVectorInst::isReverseMask(Mask, InBytes)) {
+      IntrinsicCostAttributes CostAttrs(Intrinsic::bswap, ResultTy, {ResultTy});
+      C += TTI.getIntrinsicInstrCost(CostAttrs, CostKind);
+    } else if (!ShuffleVectorInst::isIdentityMask(Mask, InBytes)) {
+      C += TTI.getShuffleCost(
+          is_contained(Info.LaneOfField, BitPackInfo::NoLane)
+              ? TargetTransformInfo::SK_PermuteTwoSrc
+              : TargetTransformInfo::SK_PermuteSingleSrc,
+          PackTy, FixedVectorType::get(Int8Ty, InBytes), CostKind, Mask,
+          /*Index=*/0, /*SubTp=*/nullptr, /*Args=*/{}, CxtI);
+    }
+    if (W2 != BitWidth && !(W2 == 8 && FreeByteTrunc))
+      C += TTI.getCastInstrCost(Instruction::Trunc, ShiftTy, SrcTy,
+                                TTI::CastContextHint::None, CostKind);
+    if (Info.needsShift())
+      C += TTI.getArithmeticInstrCost(Instruction::LShr, ShiftTy, CostKind,
+                                      /*Opd1Info=*/{}, ShiftAmtInfo,
+                                      /*Args=*/{}, CxtI, TLI);
+    if (C.isValid() && (!NewCost.isValid() || C < NewCost)) {
+      NewCost = C;
+      ShiftWidth = W2;
+    }
+  }
+  return NewCost;
+}
+
 } // namespace llvm::slpvectorizer
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.h b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.h
index 09f5ad640859a..218ccaaedbe5a 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.h
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.h
@@ -16,6 +16,7 @@
 #ifndef LLVM_LIB_TRANSFORMS_VECTORIZE_SLPVECTORIZER_SLPCOSTANALYSIS_H
 #define LLVM_LIB_TRANSFORMS_VECTORIZE_SLPVECTORIZER_SLPCOSTANALYSIS_H
 
+#include "SLPUtils.h"
 #include "llvm/ADT/ArrayRef.h"
 #include "llvm/Analysis/TargetTransformInfo.h"
 #include "llvm/Support/InstructionCost.h"
@@ -23,6 +24,9 @@
 #include <utility>
 
 namespace llvm {
+class FixedVectorType;
+class Instruction;
+class TargetLibraryInfo;
 class Type;
 class Value;
 class VectorType;
@@ -56,6 +60,17 @@ getBlendedLoadCost(const TargetTransformInfo &TTI, Type *VecTy, Align Alignment,
                    unsigned AddressSpace,
                    const TargetTransformInfo::TargetCostKind CostKind);
 
+/// Returns the cost of the bitfield packing of \p SrcTy into \p ResultTy,
+/// picking the cheapest shift width. The packing is a trunc, an lshr, a byte
+/// shuffle and a bitcast. \p FreeByteTrunc marks the lanes as a zext from i8,
+/// so compacting them to bytes is free.
+InstructionCost getBitPackCost(const TargetTransformInfo &TTI,
+                               FixedVectorType *SrcTy, Type *ResultTy,
+                               const BitPackInfo &Info, bool FreeByteTrunc,
+                               TargetTransformInfo::TargetCostKind CostKind,
+                               const TargetLibraryInfo *TLI,
+                               const Instruction *CxtI, unsigned &ShiftWidth);
+
 } // namespace llvm::slpvectorizer
 
 #endif // LLVM_LIB_TRANSFORMS_VECTORIZE_SLPVECTORIZER_SLPCOSTANALYSIS_H
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp
index af6f8032c7b0b..5c0b4c2cf7ca5 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp
@@ -16,6 +16,7 @@
 #include "llvm/IR/Constants.h"
 #include "llvm/IR/DataLayout.h"
 #include "llvm/IR/DerivedTypes.h"
+#include "llvm/IR/IRBuilder.h"
 #include "llvm/IR/Instructions.h"
 #include "llvm/IR/IntrinsicInst.h"
 #include "llvm/IR/PatternMatch.h"
@@ -891,4 +892,153 @@ TargetTransformInfo::TargetCostKind getSLPCostKind(const Function *F) {
   return F->hasOptSize() ? TTI::TCK_CodeSize : TTI::TCK_RecipThroughput;
 }
 
+/// Deeper than the standard analysis recursion depth to keep the numeric
+/// bound precise through arithmetic carry chains.
+constexpr unsigned MaxBitPackAnalysisDepth = MaxAnalysisRecursionDepth + 2;
+
+APInt getScalarMaxValue(const Value *V, unsigned Depth) {
+  unsigned BitWidth = V->getType()->getScalarSizeInBits();
+  const APInt Unknown = APInt::getAllOnes(BitWidth);
+  if (Depth > MaxBitPackAnalysisDepth || !V->getType()->isIntegerTy())
+    return Unknown;
+  const APInt *C, *Amt;
+  if (match(V, m_APInt(C)))
+    return *C;
+  Value *L, *R;
+  if (match(V, m_Add(m_Value(L), m_Value(R))) ||
+      match(V, m_Or(m_Value(L), m_Value(R))) ||
+      match(V, m_Xor(m_Value(L), m_Value(R))))
+    return getScalarMaxValue(L, Depth + 1)
+        .uadd_sat(getScalarMaxValue(R, Depth + 1));
+  if (match(V, m_NUWSub(m_Value(L), m_Value(R))))
+    return getScalarMaxValue(L, Depth + 1);
+  if (match(V, m_Sub(m_Value(L), m_Value(R))))
+    return Unknown;
+  if (match(V, m_Mul(m_Value(L), m_Value(R))))
+    return getScalarMaxValue(L, Depth + 1)
+        .umul_sat(getScalarMaxValue(R, Depth + 1));
+  if (match(V, m_And(m_Value(L), m_Value(R))))
+    return APIntOps::umin(getScalarMaxValue(L, Depth + 1),
+                          getScalarMaxValue(R, Depth + 1));
+  if (match(V, m_LShr(m_Value(L), m_APInt(Amt))) && Amt->ult(BitWidth))
+    return getScalarMaxValue(L, Depth + 1).lshr(*Amt);
+  if (match(V, m_Shl(m_Value(L), m_APInt(Amt))) && Amt->ult(BitWidth)) {
+    APInt LMax = getScalarMaxValue(L, Depth + 1);
+    return LMax.getActiveBits() + Amt->getZExtValue() <= BitWidth
+               ? LMax.shl(*Amt)
+               : Unknown;
+  }
+  if (match(V, m_ZExt(m_Value(L))))
+    return getScalarMaxValue(L, Depth + 1).zext(BitWidth);
+  if (match(V, m_Trunc(m_Value(L)))) {
+    APInt Max = getScalarMaxValue(L, Depth + 1);
+    return Max.getActiveBits() <= BitWidth ? Max.trunc(BitWidth) : Unknown;
+  }
+  if (match(V, m_SExt(m_Value(L)))) {
+    APInt Max = getScalarMaxValue(L, Depth + 1);
+    return Max.isNonNegative() ? Max.zext(BitWidth) : Unknown;
+  }
+  Value *F;
+  if (match(V, m_Select(m_Value(), m_Value(L), m_Value(F))))
+    return APIntOps::umax(getScalarMaxValue(L, Depth + 1),
+                          getScalarMaxValue(F, Depth + 1));
+  return Unknown;
+}
+
+std::optional<BitPackInfo> computeBitPackInfo(unsigned BitWidth,
+                                              ArrayRef<APInt> PossibleBits,
+                                              ArrayRef<uint64_t> ShlAmts,
+                                              ArrayRef<APInt> Masks) {
+  unsigned NumElts = PossibleBits.size();
+  BitPackInfo Info;
+  Info.LShrAmts.assign(NumElts, 0);
+  for (unsigned Idx : seq(NumElts)) {
+    APInt Possible = PossibleBits[Idx];
+    Possible <<= ShlAmts[Idx];
+    Possible &= Masks[Idx];
+    if (Possible.isZero())
+      continue;
+    if (!Possible.isShiftedMask())
+      return std::nullopt;
+    unsigned Lo = Possible.countr_zero();
+    unsigned W = Possible.popcount();
+    if (Info.FieldWidth == 0) {
+      if (BitWidth % W != 0)
+        return std::nullopt;
+      Info.FieldWidth = W;
+      Info.LaneOfField.assign(BitWidth / W, BitPackInfo::NoLane);
+    }
+    if (W != Info.FieldWidth || Lo % W != 0 || Lo < ShlAmts[Idx])
+      return std::nullopt;
+    unsigned Field = Lo / W;
+    if (Info.LaneOfField[Field] != BitPackInfo::NoLane)
+      return std::nullopt;
+    Info.LaneOfField[Field] = Idx;
+    Info.LShrAmts[Idx] = Lo - ShlAmts[Idx];
+  }
+  if (Info.FieldWidth == 0 || Info.FieldWidth % 8 != 0)
+    return std::nullopt;
+  return Info;
+}
+
+SmallVector<int> getBitPackMask(const BitPackInfo &Info, unsigned NumBytes,
+                                unsigned NumElts, unsigned BytesPerLane) {
+  unsigned BytesPerField = Info.FieldWidth / 8;
+  SmallVector<int> Mask;
+  for (unsigned J : seq(NumBytes)) {
+    unsigned Lane = Info.LaneOfField[J / BytesPerField];
+    Mask.push_back(Lane == BitPackInfo::NoLane
+                       ? (int)(NumElts * BytesPerLane)
+                       : (int)(Lane * BytesPerLane + J % BytesPerField));
+  }
+  return Mask;
+}
+
+Value *buildBitPack(IRBuilderBase &Builder, Value *X, const BitPackInfo &Info,
+                    unsigned ShiftWidth) {
+  auto *VecTy = cast<FixedVectorType>(X->getType());
+  unsigned BitWidth = VecTy->getScalarSizeInBits();
+  unsigned NumElts = VecTy->getNumElements();
+  Value *Y = X;
+  if (ShiftWidth != BitWidth) {
+    // Compacting a byte zext is free, use its source directly.
+    if (auto *Z = dyn_cast<ZExtInst>(X);
+        Z && ShiftWidth == 8 &&
+        Z->getSrcTy() == FixedVectorType::get(Builder.getInt8Ty(), NumElts))
+      Y = Z->getOperand(0);
+    else
+      Y = Builder.CreateTrunc(
+          Y, FixedVectorType::get(IntegerType::get(X->getContext(), ShiftWidth),
+                                  NumElts));
+  }
+  if (Info.needsShift()) {
+    SmallVector<Constant *> Amts;
+    for (uint64_t A : Info.LShrAmts)
+      Amts.push_back(
+          ConstantInt::get(IntegerType::get(X->getContext(), ShiftWidth), A));
+    Y = Builder.CreateLShr(Y, ConstantVector::get(Amts));
+  }
+  unsigned InBytes = NumElts * (ShiftWidth / 8);
+  auto *ByteTy = FixedVectorType::get(Builder.getInt8Ty(), InBytes);
+  SmallVector<int> Mask =
+      getBitPackMask(Info, BitWidth / 8, NumElts, ShiftWidth / 8);
+  // A plain byte reversal of the shifted lanes is a bswap.
+  if (ShuffleVectorInst::isReverseMask(Mask, InBytes))
+    return Builder.CreateUnaryIntrinsic(
+        Intrinsic::bswap,
+        Builder.CreateBitCast(Y, IntegerType::get(X->getContext(), BitWidth)));
+  // An identity byte order needs no shuffle.
+  if (ShuffleVectorInst::isIdentityMask(Mask, InBytes))
+    return Builder.CreateBitCast(Y,
+                                 IntegerType::get(X->getContext(), BitWidth));
+  Value *Packed = Builder.CreateShuffleVector(
+      Builder.CreateBitCast(Y, ByteTy),
+      is_contained(Info.LaneOfField, BitPackInfo::NoLane)
+          ? Constant::getNullValue(ByteTy)
+          : PoisonValue::get(ByteTy),
+      Mask);
+  return Builder.CreateBitCast(Packed,
+                               IntegerType::get(X->getContext(), BitWidth));
+}
+
 } // namespace llvm::slpvectorizer
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.h b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.h
index e9aeff605eb91..d5a02a0e40c4b 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.h
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.h
@@ -18,18 +18,21 @@
 
 #include "llvm/ADT/APInt.h"
 #include "llvm/ADT/ArrayRef.h"
+#include "llvm/ADT/STLExtras.h"
 #include "llvm/ADT/SmallBitVector.h"
 #include "llvm/ADT/SmallVector.h"
 #include "llvm/Analysis/MemoryLocation.h"
 #include "llvm/Analysis/TargetTransformInfo.h"
 #include "llvm/IR/Intrinsics.h"
 
+#include <cstdint>
 #include <optional>
 #include <string>
 
 namespace llvm {
 class Constant;
 class DataLayout;
+class IRBuilderBase;
 class Instruction;
 class TargetLibraryInfo;
 class Type;
@@ -365,6 +368,44 @@ void collectNarrowedLeaves(Value *V, unsigned RdxOpcode, unsigned WideBW,
 
 TargetTransformInfo::TargetCostKind getSLPCostKind(const Function *F);
 
+/// Returns a saturating unsigned upper bound of the scalar V. The numeric
+/// bound keeps precision on arithmetic carries, where bit-wise analysis
+/// loses it.
+APInt getScalarMaxValue(const Value *V, unsigned Depth = 0);
+
+/// Description of a bitfield packing of vector lanes into a scalar value:
+/// every lane contributes a disjoint contiguous byte field of the result.
+struct BitPackInfo {
+  static constexpr unsigned NoLane = ~0u;
+  unsigned FieldWidth = 0;
+  /// Lane covering each field, NoLane if the field is always zero.
+  SmallVector<unsigned, 8> LaneOfField;
+  /// Per-lane right-shift amounts bringing the field content to the low bits.
+  SmallVector<uint64_t, 8> LShrAmts;
+
+  /// True if any lane needs a right shift to align its field content.
+  bool needsShift() const {
+    return any_of(LShrAmts, [](uint64_t A) { return A != 0; });
+  }
+};
+
+/// Computes the bitfield packing layout from the per-lane possibly set bits
+/// of the source values, the per-lane left-shift amounts and the per-lane
+/// masks (all-ones for unmasked lanes).
+std::optional<BitPackInfo> computeBitPackInfo(unsigned BitWidth,
+                                              ArrayRef<APInt> PossibleBits,
+                                              ArrayRef<uint64_t> ShlAmts,
+                                              ArrayRef<APInt> Masks);
+
+/// Returns the byte shuffle mask packing the per-lane fields of the shifted
+/// lanes (BytesPerLane bytes each) into the packed scalar of NumBytes bytes.
+SmallVector<int> getBitPackMask(const BitPackInfo &Info, unsigned NumBytes,
+                                unsigned NumElts, unsigned BytesPerLane);
+
+/// Builds the bitfield packing of X per the layout and the shift width.
+Value *buildBitPack(IRBuilderBase &Builder, Value *X, const BitPackInfo &Info,
+                    unsigned ShiftWidth);
+
 } // namespace llvm::slpvectorizer
 
 #endif // LLVM_LIB_TRANSFORMS_VECTORIZE_SLPVECTORIZER_SLPUTILS_H
diff --git a/llvm/test/Transforms/PhaseOrdering/X86/avg.ll b/llvm/test/Transforms/PhaseOrdering/X86/avg.ll
index e7d927156a9f1..8f8899d285ad3 100644
--- a/llvm/test/Transforms/PhaseOrdering/X86/avg.ll
+++ b/llvm/test/Transforms/PhaseOrdering/X86/avg.ll
@@ -95,80 +95,121 @@ define { i64, i64 } @avgr_16_u8(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerce0,
 ;
 ; SSE4-LABEL: @avgr_16_u8(
 ; SSE4-NEXT:  entry:
-; SSE4-NEXT:    [[TMP1:%.*]] = insertelement <2 x i64> poison, i64 [[A_COERCE0:%.*]], i64 0
-; SSE4-NEXT:    [[TMP2:%.*]] = insertelement <2 x i64> [[TMP1]], i64 [[A_COERCE1:%.*]], i64 1
-; SSE4-NEXT:    [[TMP19:%.*]] = trunc <2 x i64> [[TMP2]] to <2 x i16>
-; SSE4-NEXT:    [[TMP3:%.*]] = lshr <2 x i64> [[TMP2]], splat (i64 16)
-; SSE4-NEXT:    [[TMP4:%.*]] = lshr <2 x i64> [[TMP2]], splat (i64 24)
-; SSE4-NEXT:    [[TMP5:%.*]] = lshr <2 x i64> [[TMP2]], splat (i64 32)
-; SSE4-NEXT:    [[TMP6:%.*]] = lshr <2 x i64> [[TMP2]], splat (i64 40)
-; SSE4-NEXT:    [[TMP45:%.*]] = lshr <2 x i64> [[TMP2]], splat (i64 48)
-; SSE4-NEXT:    [[TMP46:%.*]] = lshr <2 x i64> [[TMP2]], splat (i64 56)
-; SSE4-NEXT:    [[TMP9:%.*]] = insertelement <2 x i64> poison, i64 [[B_COERCE0:%.*]], i64 0
-; SSE4-NEXT:    [[TMP10:%.*]] = insertelement <2 x i64> [[TMP9]], i64 [[B_COERCE1:%.*]], i64 1
-; SSE4-NEXT:    [[TMP22:%.*]] = trunc <2 x i64> [[TMP10]] to <2 x i16>
-; SSE4-NEXT:    [[TMP11:%.*]] = lshr <2 x i64> [[TMP10]], splat (i64 16)
-; SSE4-NEXT:    [[TMP12:%.*]] = lshr <2 x i64> [[TMP10]], splat (i64 24)
-; SSE4-NEXT:    [[TMP13:%.*]] = lshr <2 x i64> [[TMP10]], splat (i64 32)
-; SSE4-NEXT:    [[TMP14:%.*]] = lshr <2 x i64> [[TMP10]], splat (i64 40)
-; SSE4-NEXT:    [[TMP47:%.*]] = lshr <2 x i64> [[TMP10]], splat (i64 48)
-; SSE4-NEXT:    [[TMP48:%.*]] = lshr <2 x i64> [[TMP10]], splat (i64 56)
-; SSE4-NEXT:    [[TMP16:%.*]] = and <2 x i64> [[TMP2]], splat (i64 255)
-; SSE4-NEXT:    [[TMP17:%.*]] = and <2 x i64> [[TMP10]], splat (i64 255)
-; SSE4-NEXT:    [[TMP70:%.*]] = and <2 x i64> [[TMP45]], splat (i64 255)
-; SSE4-NEXT:    [[TMP20:%.*]] = lshr <2 x i16> [[TMP19]], splat (i16 8)
-; SSE4-NEXT:    [[TMP23:%.*]] = lshr <2 x i16> [[TMP22]], splat (i16 8)
-; SSE4-NEXT:    [[TMP24:%.*]] = add nuw nsw <2 x i64> [[TMP16]], splat (i64 1)
-; SSE4-NEXT:    [[TMP25:%.*]] = add nuw nsw <2 x i64> [[TMP24]], [[TMP17]]
-; SSE4-NEXT:    [[TMP26:%.*]] = lshr <2 x i64> [[TMP25]], splat (i64 1)
-; SSE4-NEXT:    [[TMP27:%.*]] = add nuw nsw <2 x i16> [[TMP20]], splat (i16 1)
-; SSE4-NEXT:    [[TMP28:%.*]] = add nuw nsw <2 x i16> [[TMP27]], [[TMP23]]
-; SSE4-NEXT:    [[TMP29:%.*]] = and <2 x i64> [[TMP3]], splat (i64 255)
-; SSE4-NEXT:    [[TMP30:%.*]] = and <2 x i64> [[TMP11]], splat (i64 255)
-; SSE4-NEXT:    [[TMP31:%.*]] = add nuw nsw <2 x i64> [[TMP29]], splat (i64 1)
-; SSE4-NEXT:    [[TMP32:%.*]] = add nuw nsw <2 x i64> [[TMP31]], [[TMP30]]
-; SSE4-NEXT:    [[TMP33:%.*]] = and <2 x i64> [[TMP4]], splat (i64 255)
-; SSE4-NEXT:    [[TMP34:%.*]] = and <2 x i64> [[TMP12]], splat (i64 255)
-; SSE4-NEXT:    [[TMP35:%.*]] = add nuw nsw <2 x i64> [[TMP33]], splat (i64 1)
-; SSE4-NEXT:    [[TMP36:%.*]] = add nuw nsw <2 x i64> [[TMP35]], [[TMP34]]
-; SSE4-NEXT:    [[TMP37:%.*]] = and <2 x i64> [[TMP5]], splat (i64 255)
-; SSE4-NEXT:    [[TMP38:%.*]] = and <2 x i64> [[TMP13]], splat (i64 255)
-; SSE4-NEXT:    [[TMP39:%.*]] = add nuw nsw <2 x i64> [[TMP37]], splat (i64 1)
-; SSE4-NEXT:    [[TMP40:%.*]] = add nuw nsw <2 x i64> [[TMP39]], [[TMP38]]
-; SSE4-NEXT:    [[TMP41:%.*]] = and <2 x i64> [[TMP6]], splat (i64 255)
-; SSE4-NEXT:    [[TMP42:%.*]] = and <2 x i64> [[TMP14]], splat (i64 255)
-; SSE4-NEXT:    [[TMP43:%.*]] = add nuw nsw <2 x i64> [[TMP41]], splat (i64 1)
-; SSE4-NEXT:    [[TMP44:%.*]] = add nuw nsw <2 x i64> [[TMP43]], [[TMP42]]
-; SSE4-NEXT:    [[TMP71:%.*]] = and <2 x i64> [[TMP47]], splat (i64 255)
-; SSE4-NEXT:    [[TMP51:%.*]] = add nuw nsw <2 x i64> [[TMP70]], splat (i64 1)
-; SSE4-NEXT:    [[TMP72:%.*]] = add nuw nsw <2 x i64> [[TMP51]], [[TMP71]]
-; SSE4-NEXT:    [[TMP73:%.*]] = add nuw nsw <2 x i64> [[TMP46]], splat (i64 1)
-; SSE4-NEXT:    [[TMP74:%.*]] = add nuw nsw <2 x i64> [[TMP73]], [[TMP48]]
-; SSE4-NEXT:    [[TMP75:%.*]] = shl nuw <2 x i64> [[TMP74]], splat (i64 55)
-; SSE4-NEXT:    [[TMP76:%.*]] = and <2 x i64> [[TMP75]], splat (i64 -72057594037927936)
-; SSE4-NEXT:    [[TMP77:%.*]] = shl nuw nsw <2 x i64> [[TMP72]], splat (i64 47)
-; SSE4-NEXT:    [[TMP78:%.*]] = and <2 x i64> [[TMP77]], splat (i64 71776119061217280)
-; SSE4-NEXT:    [[TMP52:%.*]] = or disjoint <2 x i64> [[TMP76]], [[TMP78]]
-; SSE4-NEXT:    [[TMP49:%.*]] = shl nuw nsw <2 x i64> [[TMP44]], splat (i64 39)
-; SSE4-NEXT:    [[TMP50:%.*]] = and <2 x i64> [[TMP49]], splat (i64 280375465082880)
-; SSE4-NEXT:    [[TMP53:%.*]] = or disjoint <2 x i64> [[TMP52]], [[TMP50]]
-; SSE4-NEXT:    [[TMP54:%.*]] = shl nuw nsw <2 x i64> [[TMP40]], splat (i64 31)
-; SSE4-NEXT:    [[TMP55:%.*]] = and <2 x i64> [[TMP54]], splat (i64 1095216660480)
-; SSE4-NEXT:    [[TMP56:%.*]] = or disjoint <2 x i64> [[TMP53]], [[TMP55]]
-; SSE4-NEXT:    [[TMP57:%.*]] = shl nuw nsw <2 x i64> [[TMP36]], splat (i64 23)
-; SSE4-NEXT:    [[TMP58:%.*]] = and <2 x i64> [[TMP57]], splat (i64 4278190080)
-; SSE4-NEXT:    [[TMP59:%.*]] = or disjoint <2 x i64> [[TMP56]], [[TMP58]]
-; SSE4-NEXT:    [[TMP60:%.*]] = shl nuw nsw <2 x i64> [[TMP32]], splat (i64 15)
-; SSE4-NEXT:    [[TMP61:%.*]] = and <2 x i64> [[TMP60]], splat (i64 16711680)
-; SSE4-NEXT:    [[TMP62:%.*]] = shl nuw <2 x i16> [[TMP28]], splat (i16 7)
-; SSE4-NEXT:    [[TMP63:%.*]] = or disjoint <2 x i64> [[TMP59]], [[TMP61]]
-; SSE4-NEXT:    [[TMP64:%.*]] = and <2 x i16> [[TMP62]], splat (i16 -256)
-; SSE4-NEXT:    [[TMP65:%.*]] = zext <2 x i16> [[TMP64]] to <2 x i64>
-; SSE4-NEXT:    [[TMP66:%.*]] = or <2 x i64> [[TMP63]], [[TMP65]]
-; SSE4-NEXT:    [[TMP67:%.*]] = or <2 x i64> [[TMP66]], [[TMP26]]
-; SSE4-NEXT:    [[TMP68:%.*]] = extractelement <2 x i64> [[TMP67]], i64 0
+; SSE4-NEXT:    [[TMP0:%.*]] = trunc i64 [[A_COERCE0:%.*]] to i16
+; SSE4-NEXT:    [[TMP1:%.*]] = lshr i16 [[TMP0]], 8
+; SSE4-NEXT:    [[A_SROA_3_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE0]], 16
+; SSE4-NEXT:    [[A_SROA_4_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE0]], 24
+; SSE4-NEXT:    [[A_SROA_5_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE0]], 32
+; SSE4-NEXT:    [[A_SROA_6_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE0]], 40
+; SSE4-NEXT:    [[A_SROA_7_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE0]], 48
+; SSE4-NEXT:    [[A_SROA_8_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE0]], 56
+; SSE4-NEXT:    [[TMP2:%.*]] = trunc i64 [[A_COERCE1:%.*]] to i16
+; SSE4-NEXT:    [[TMP3:%.*]] = lshr i16 [[TMP2]], 8
+; SSE4-NEXT:    [[A_SROA_12_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE1]], 16
+; SSE4-NEXT:    [[A_SROA_13_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE1]], 24
+; SSE4-NEXT:    [[A_SROA_14_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE1]], 32
+; SSE4-NEXT:    [[A_SROA_15_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE1]], 40
+; SSE4-NEXT:    [[A_SROA_16_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE1]], 48
+; SSE4-NEXT:    [[A_SROA_17_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE1]], 56
+; SSE4-NEXT:    [[TMP4:%.*]] = trunc i64 [[B_COERCE0:%.*]] to i16
+; SSE4-NEXT:    [[TMP5:%.*]] = lshr i16 [[TMP4]], 8
+; SSE4-NEXT:    [[B_SROA_3_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE0]], 16
+; SSE4-NEXT:    [[B_SROA_4_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE0]], 24
+; SSE4-NEXT:    [[B_SROA_5_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE0]], 32
+; SSE4-NEXT:    [[B_SROA_6_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE0]], 40
+; SSE4-NEXT:    [[B_SROA_7_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE0]], 48
+; SSE4-NEXT:    [[B_SROA_8_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE0]], 56
+; SSE4-NEXT:    [[TMP6:%.*]] = trunc i64 [[B_COERCE1:%.*]] to i16
+; SSE4-NEXT:    [[TMP7:%.*]] = lshr i16 [[TMP6]], 8
+; SSE4-NEXT:    [[B_SROA_12_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE1]], 16
+; SSE4-NEXT:    [[B_SROA_13_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE1]], 24
+; SSE4-NEXT:    [[B_SROA_14_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE1]], 32
+; SSE4-NEXT:    [[B_SROA_15_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE1]], 40
+; SSE4-NEXT:    [[B_SROA_16_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE1]], 48
+; SSE4-NEXT:    [[B_SROA_17_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE1]], 56
+; SSE4-NEXT:    [[CONV1:%.*]] = and i64 [[A_COERCE0]], 255
+; SSE4-NEXT:    [[CONV4:%.*]] = and i64 [[B_COERCE0]], 255
+; SSE4-NEXT:    [[ADD:%.*]] = add nuw nsw i64 [[CONV1]], 1
+; SSE4-NEXT:    [[ADD5:%.*]] = add nuw nsw i64 [[ADD]], [[CONV4]]
+; SSE4-NEXT:    [[SHR:%.*]] = lshr i64 [[ADD5]], 1
+; SSE4-NEXT:    [[ADD_1:%.*]] = add nuw nsw i16 [[TMP1]], 1
+; SSE4-NEXT:    [[ADD5_1:%.*]] = add nuw nsw i16 [[ADD_1]], [[TMP5]]
+; SSE4-NEXT:    [[CONV1_6:%.*]] = and i64 [[A_SROA_7_0_EXTRACT_SHIFT]], 255
+; SSE4-NEXT:    [[CONV4_6:%.*]] = and i64 [[B_SROA_7_0_EXTRACT_SHIFT]], 255
+; SSE4-NEXT:    [[ADD_6:%.*]] = add nuw nsw i64 [[CONV1_6]], 1
+; SSE4-NEXT:    [[ADD5_6:%.*]] = add nuw nsw i64 [[ADD_6]], [[CONV4_6]]
+; SSE4-NEXT:    [[ADD_7:%.*]] = add nuw nsw i64 [[A_SROA_8_0_EXTRACT_SHIFT]], 1
+; SSE4-NEXT:    [[ADD5_7:%.*]] = add nuw nsw i64 [[ADD_7]], [[B_SROA_8_0_EXTRACT_SHIFT]]
+; SSE4-NEXT:    [[CONV1_8:%.*]] = and i64 [[A_COERCE1]], 255
+; SSE4-NEXT:    [[CONV4_8:%.*]] = and i64 [[B_COERCE1]], 255
+; SSE4-NEXT:    [[ADD_8:%.*]] = add nuw nsw i64 [[CONV1_8]], 1
+; SSE4-NEXT:    [[ADD5_8:%.*]] = add nuw nsw i64 [[ADD_8]], [[CONV4_8]]
+; SSE4-NEXT:    [[SHR_8:%.*]] = lshr i64 [[ADD5_8]], 1
+; SSE4-NEXT:    [[ADD_9:%.*]] = add nuw nsw i16 [[TMP3]], 1
+; SSE4-NEXT:    [[ADD5_9:%.*]] = add nuw nsw i16 [[ADD_9]], [[TMP7]]
+; SSE4-NEXT:    [[CONV1_14:%.*]] = and i64 [[A_SROA_16_8_EXTRACT_SHIFT]], 255
+; SSE4-NEXT:    [[CONV4_14:%.*]] = and i64 [[B_SROA_16_8_EXTRACT_SHIFT]], 255
+; SSE4-NEXT:    [[ADD_14:%.*]] = add nuw nsw i64 [[CONV1_14]], 1
+; SSE4-NEXT:    [[ADD5_14:%.*]] = add nuw nsw i64 [[ADD_14]], [[CONV4_14]]
+; SSE4-NEXT:    [[ADD_15:%.*]] = add nuw nsw i64 [[A_SROA_17_8_EXTRACT_SHIFT]], 1
+; SSE4-NEXT:    [[ADD5_15:%.*]] = add nuw nsw i64 [[ADD_15]], [[B_SROA_17_8_EXTRACT_SHIFT]]
+; SSE4-NEXT:    [[TMP8:%.*]] = shl nuw i64 [[ADD5_7]], 55
+; SSE4-NEXT:    [[RETVAL_SROA_8_0_INSERT_EXT:%.*]] = and i64 [[TMP8]], -72057594037927936
+; SSE4-NEXT:    [[TMP9:%.*]] = shl nuw nsw i64 [[ADD5_6]], 47
+; SSE4-NEXT:    [[RETVAL_SROA_7_0_INSERT_SHIFT:%.*]] = and i64 [[TMP9]], 71776119061217280
+; SSE4-NEXT:    [[TMP10:%.*]] = insertelement <4 x i64> poison, i64 [[A_SROA_6_0_EXTRACT_SHIFT]], i64 0
+; SSE4-NEXT:    [[TMP11:%.*]] = insertelement <4 x i64> [[TMP10]], i64 [[A_SROA_5_0_EXTRACT_SHIFT]], i64 1
+; SSE4-NEXT:    [[TMP12:%.*]] = insertelement <4 x i64> [[TMP11]], i64 [[A_SROA_4_0_EXTRACT_SHIFT]], i64 2
+; SSE4-NEXT:    [[TMP13:%.*]] = insertelement <4 x i64> [[TMP12]], i64 [[A_SROA_3_0_EXTRACT_SHIFT]], i64 3
+; SSE4-NEXT:    [[TMP14:%.*]] = and <4 x i64> [[TMP13]], splat (i64 255)
+; SSE4-NEXT:    [[TMP15:%.*]] = insertelement <4 x i64> poison, i64 [[B_SROA_6_0_EXTRACT_SHIFT]], i64 0
+; SSE4-NEXT:    [[TMP16:%.*]] = insertelement <4 x i64> [[TMP15]], i64 [[B_SROA_5_0_EXTRACT_SHIFT]], i64 1
+; SSE4-NEXT:    [[TMP17:%.*]] = insertelement <4 x i64> [[TMP16]], i64 [[B_SROA_4_0_EXTRACT_SHIFT]], i64 2
+; SSE4-NEXT:    [[TMP18:%.*]] = insertelement <4 x i64> [[TMP17]], i64 [[B_SROA_3_0_EXTRACT_SHIFT]], i64 3
+; SSE4-NEXT:    [[TMP19:%.*]] = and <4 x i64> [[TMP18]], splat (i64 255)
+; SSE4-NEXT:    [[TMP20:%.*]] = add nuw nsw <4 x i64> [[TMP14]], splat (i64 1)
+; SSE4-NEXT:    [[TMP21:%.*]] = add nuw nsw <4 x i64> [[TMP20]], [[TMP19]]
+; SSE4-NEXT:    [[TMP22:%.*]] = trunc nuw nsw <4 x i64> [[TMP21]] to <4 x i32>
+; SSE4-NEXT:    [[TMP23:%.*]] = lshr <4 x i32> [[TMP22]], splat (i32 1)
+; SSE4-NEXT:    [[TMP24:%.*]] = bitcast <4 x i32> [[TMP23]] to <16 x i8>
+; SSE4-NEXT:    [[TMP25:%.*]] = shufflevector <16 x i8> [[TMP24]], <16 x i8> <i8 0, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison>, <8 x i32> <i32 16, i32 16, i32 12, i32 8, i32 4, i32 0, i32 16, i32 16>
+; SSE4-NEXT:    [[TMP26:%.*]] = bitcast <8 x i8> [[TMP25]] to i64
+; SSE4-NEXT:    [[TMP27:%.*]] = shl nuw i16 [[ADD5_1]], 7
+; SSE4-NEXT:    [[TMP28:%.*]] = and i16 [[TMP27]], -256
+; SSE4-NEXT:    [[RETVAL_SROA_2_0_INSERT_SHIFT_MASKED:%.*]] = zext i16 [[TMP28]] to i64
+; SSE4-NEXT:    [[OP_RDX5:%.*]] = or disjoint i64 [[RETVAL_SROA_8_0_INSERT_EXT]], [[TMP26]]
+; SSE4-NEXT:    [[OP_RDX6:%.*]] = or disjoint i64 [[RETVAL_SROA_7_0_INSERT_SHIFT]], [[SHR]]
+; SSE4-NEXT:    [[OP_RDX7:%.*]] = or disjoint i64 [[OP_RDX5]], [[OP_RDX6]]
+; SSE4-NEXT:    [[TMP68:%.*]] = or i64 [[OP_RDX7]], [[RETVAL_SROA_2_0_INSERT_SHIFT_MASKED]]
 ; SSE4-NEXT:    [[DOTFCA_0_INSERT:%.*]] = insertvalue { i64, i64 } poison, i64 [[TMP68]], 0
-; SSE4-NEXT:    [[TMP69:%.*]] = extractelement <2 x i64> [[TMP67]], i64 1
+; SSE4-NEXT:    [[TMP29:%.*]] = shl nuw i64 [[ADD5_15]], 55
+; SSE4-NEXT:    [[RETVAL_SROA_17_8_INSERT_EXT:%.*]] = and i64 [[TMP29]], -72057594037927936
+; SSE4-NEXT:    [[TMP30:%.*]] = shl nuw nsw i64 [[ADD5_14]], 47
+; SSE4-NEXT:    [[RETVAL_SROA_16_8_INSERT_SHIFT:%.*]] = and i64 [[TMP30]], 71776119061217280
+; SSE4-NEXT:    [[TMP31:%.*]] = insertelement <4 x i64> poison, i64 [[A_SROA_15_8_EXTRACT_SHIFT]], i64 0
+; SSE4-NEXT:    [[TMP32:%.*]] = insertelement <4 x i64> [[TMP31]], i64 [[A_SROA_14_8_EXTRACT_SHIFT]], i64 1
+; SSE4-NEXT:    [[TMP33:%.*]] = insertelement <4 x i64> [[TMP32]], i64 [[A_SROA_13_8_EXTRACT_SHIFT]], i64 2
+; SSE4-NEXT:    [[TMP34:%.*]] = insertelement <4 x i64> [[TMP33]], i64 [[A_SROA_12_8_EXTRACT_SHIFT]], i64 3
+; SSE4-NEXT:    [[TMP35:%.*]] = and <4 x i64> [[TMP34]], splat (i64 255)
+; SSE4-NEXT:    [[TMP36:%.*]] = insertelement <4 x i64> poison, i64 [[B_SROA_15_8_EXTRACT_SHIFT]], i64 0
+; SSE4-NEXT:    [[TMP37:%.*]] = insertelement <4 x i64> [[TMP36]], i64 [[B_SROA_14_8_EXTRACT_SHIFT]], i64 1
+; SSE4-NEXT:    [[TMP38:%.*]] = insertelement <4 x i64> [[TMP37]], i64 [[B_SROA_13_8_EXTRACT_SHIFT]], i64 2
+; SSE4-NEXT:    [[TMP39:%.*]] = insertelement <4 x i64> [[TMP38]], i64 [[B_SROA_12_8_EXTRACT_SHIFT]], i64 3
+; SSE4-NEXT:    [[TMP40:%.*]] = and <4 x i64> [[TMP39]], splat (i64 255)
+; SSE4-NEXT:    [[TMP41:%.*]] = add nuw nsw <4 x i64> [[TMP35]], splat (i64 1)
+; SSE4-NEXT:    [[TMP42:%.*]] = add nuw nsw <4 x i64> [[TMP41]], [[TMP40]]
+; SSE4-NEXT:    [[TMP43:%.*]] = trunc nuw nsw <4 x i64> [[TMP42]] to <4 x i32>
+; SSE4-NEXT:    [[TMP44:%.*]] = lshr <4 x i32> [[TMP43]], splat (i32 1)
+; SSE4-NEXT:    [[TMP45:%.*]] = bitcast <4 x i32> [[TMP44]] to <16 x i8>
+; SSE4-NEXT:    [[TMP46:%.*]] = shufflevector <16 x i8> [[TMP45]], <16 x i8> <i8 0, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison>, <8 x i32> <i32 16, i32 16, i32 12, i32 8, i32 4, i32 0, i32 16, i32 16>
+; SSE4-NEXT:    [[TMP47:%.*]] = bitcast <8 x i8> [[TMP46]] to i64
+; SSE4-NEXT:    [[TMP48:%.*]] = shl nuw i16 [[ADD5_9]], 7
+; SSE4-NEXT:    [[TMP49:%.*]] = and i16 [[TMP48]], -256
+; SSE4-NEXT:    [[RETVAL_SROA_11_8_INSERT_SHIFT_MASKED:%.*]] = zext i16 [[TMP49]] to i64
+; SSE4-NEXT:    [[OP_RDX:%.*]] = or disjoint i64 [[RETVAL_SROA_17_8_INSERT_EXT]], [[TMP47]]
+; SSE4-NEXT:    [[OP_RDX2:%.*]] = or disjoint i64 [[RETVAL_SROA_16_8_INSERT_SHIFT]], [[SHR_8]]
+; SSE4-NEXT:    [[OP_RDX3:%.*]] = or disjoint i64 [[OP_RDX]], [[OP_RDX2]]
+; SSE4-NEXT:    [[TMP69:%.*]] = or i64 [[OP_RDX3]], [[RETVAL_SROA_11_8_INSERT_SHIFT_MASKED]]
 ; SSE4-NEXT:    [[DOTFCA_1_INSERT:%.*]] = insertvalue { i64, i64 } [[DOTFCA_0_INSERT]], i64 [[TMP69]], 1
 ; SSE4-NEXT:    ret { i64, i64 } [[DOTFCA_1_INSERT]]
 ;
@@ -212,9 +253,11 @@ define { i64, i64 } @avgr_16_u8(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerce0,
 ; AVX2-NEXT:    [[TMP27:%.*]] = and <8 x i64> [[TMP26]], <i64 -1, i64 255, i64 255, i64 255, i64 255, i64 255, i64 -1, i64 -1>
 ; AVX2-NEXT:    [[TMP28:%.*]] = add nuw nsw <8 x i64> [[TMP27]], <i64 1, i64 1, i64 1, i64 1, i64 1, i64 1, i64 0, i64 0>
 ; AVX2-NEXT:    [[TMP29:%.*]] = add nuw nsw <8 x i64> [[TMP28]], [[TMP13]]
-; AVX2-NEXT:    [[TMP30:%.*]] = shl nuw <8 x i64> [[TMP29]], <i64 55, i64 47, i64 39, i64 31, i64 23, i64 15, i64 0, i64 0>
-; AVX2-NEXT:    [[TMP31:%.*]] = and <8 x i64> [[TMP30]], <i64 -72057594037927936, i64 71776119061217280, i64 280375465082880, i64 1095216660480, i64 4278190080, i64 16711680, i64 -1, i64 -1>
-; AVX2-NEXT:    [[TMP32:%.*]] = tail call i64 @llvm.vector.reduce.or.v8i64(<8 x i64> [[TMP31]])
+; AVX2-NEXT:    [[TMP30:%.*]] = trunc <8 x i64> [[TMP29]] to <8 x i32>
+; AVX2-NEXT:    [[TMP31:%.*]] = lshr <8 x i32> [[TMP30]], <i32 1, i32 1, i32 1, i32 1, i32 1, i32 1, i32 0, i32 8>
+; AVX2-NEXT:    [[TMP41:%.*]] = bitcast <8 x i32> [[TMP31]] to <32 x i8>
+; AVX2-NEXT:    [[TMP42:%.*]] = shufflevector <32 x i8> [[TMP41]], <32 x i8> poison, <8 x i32> <i32 24, i32 28, i32 20, i32 16, i32 12, i32 8, i32 4, i32 0>
+; AVX2-NEXT:    [[TMP32:%.*]] = bitcast <8 x i8> [[TMP42]] to i64
 ; AVX2-NEXT:    [[DOTFCA_0_INSERT:%.*]] = insertvalue { i64, i64 } poison, i64 [[TMP32]], 0
 ; AVX2-NEXT:    [[TMP33:%.*]] = insertelement <8 x i64> poison, i64 [[A_COERCE1]], i64 0
 ; AVX2-NEXT:    [[TMP34:%.*]] = insertelement <8 x i64> [[TMP33]], i64 [[ADD5_8]], i64 1
@@ -224,9 +267,11 @@ define { i64, i64 } @avgr_16_u8(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerce0,
 ; AVX2-NEXT:    [[TMP38:%.*]] = and <8 x i64> [[TMP16]], <i64 -1, i64 255, i64 255, i64 255, i64 255, i64 255, i64 0, i64 0>
 ; AVX2-NEXT:    [[TMP39:%.*]] = add nuw nsw <8 x i64> [[TMP37]], <i64 1, i64 1, i64 1, i64 1, i64 1, i64 1, i64 0, i64 0>
 ; AVX2-NEXT:    [[TMP40:%.*]] = add nuw nsw <8 x i64> [[TMP39]], [[TMP38]]
-; AVX2-NEXT:    [[TMP41:%.*]] = shl nuw <8 x i64> [[TMP40]], <i64 55, i64 47, i64 39, i64 31, i64 23, i64 15, i64 0, i64 0>
-; AVX2-NEXT:    [[TMP42:%.*]] = and <8 x i64> [[TMP41]], <i64 -72057594037927936, i64 71776119061217280, i64 280375465082880, i64 1095216660480, i64 4278190080, i64 16711680, i64 -1, i64 -1>
-; AVX2-NEXT:    [[TMP43:%.*]] = tail call i64 @llvm.vector.reduce.or.v8i64(<8 x i64> [[TMP42]])
+; AVX2-NEXT:    [[TMP47:%.*]] = trunc <8 x i64> [[TMP40]] to <8 x i32>
+; AVX2-NEXT:    [[TMP44:%.*]] = lshr <8 x i32> [[TMP47]], <i32 1, i32 1, i32 1, i32 1, i32 1, i32 1, i32 0, i32 8>
+; AVX2-NEXT:    [[TMP45:%.*]] = bitcast <8 x i32> [[TMP44]] to <32 x i8>
+; AVX2-NEXT:    [[TMP46:%.*]] = shufflevector <32 x i8> [[TMP45]], <32 x i8> poison, <8 x i32> <i32 24, i32 28, i32 20, i32 16, i32 12, i32 8, i32 4, i32 0>
+; AVX2-NEXT:    [[TMP43:%.*]] = bitcast <8 x i8> [[TMP46]] to i64
 ; AVX2-NEXT:    [[DOTFCA_1_INSERT:%.*]] = insertvalue { i64, i64 } [[DOTFCA_0_INSERT]], i64 [[TMP43]], 1
 ; AVX2-NEXT:    ret { i64, i64 } [[DOTFCA_1_INSERT]]
 ;
@@ -270,9 +315,11 @@ define { i64, i64 } @avgr_16_u8(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerce0,
 ; AVX512-NEXT:    [[TMP27:%.*]] = and <8 x i64> [[TMP26]], <i64 -1, i64 255, i64 255, i64 255, i64 255, i64 255, i64 -1, i64 -1>
 ; AVX512-NEXT:    [[TMP28:%.*]] = add nuw nsw <8 x i64> [[TMP27]], <i64 1, i64 1, i64 1, i64 1, i64 1, i64 1, i64 0, i64 0>
 ; AVX512-NEXT:    [[TMP29:%.*]] = add nuw nsw <8 x i64> [[TMP28]], [[TMP13]]
-; AVX512-NEXT:    [[TMP30:%.*]] = shl nuw <8 x i64> [[TMP29]], <i64 55, i64 47, i64 39, i64 31, i64 23, i64 15, i64 0, i64 0>
-; AVX512-NEXT:    [[TMP31:%.*]] = and <8 x i64> [[TMP30]], <i64 -72057594037927936, i64 71776119061217280, i64 280375465082880, i64 1095216660480, i64 4278190080, i64 16711680, i64 -1, i64 -1>
-; AVX512-NEXT:    [[TMP32:%.*]] = tail call i64 @llvm.vector.reduce.or.v8i64(<8 x i64> [[TMP31]])
+; AVX512-NEXT:    [[TMP30:%.*]] = trunc <8 x i64> [[TMP29]] to <8 x i16>
+; AVX512-NEXT:    [[TMP31:%.*]] = lshr <8 x i16> [[TMP30]], <i16 1, i16 1, i16 1, i16 1, i16 1, i16 1, i16 0, i16 8>
+; AVX512-NEXT:    [[TMP41:%.*]] = bitcast <8 x i16> [[TMP31]] to <16 x i8>
+; AVX512-NEXT:    [[TMP42:%.*]] = shufflevector <16 x i8> [[TMP41]], <16 x i8> poison, <8 x i32> <i32 12, i32 14, i32 10, i32 8, i32 6, i32 4, i32 2, i32 0>
+; AVX512-NEXT:    [[TMP32:%.*]] = bitcast <8 x i8> [[TMP42]] to i64
 ; AVX512-NEXT:    [[DOTFCA_0_INSERT:%.*]] = insertvalue { i64, i64 } poison, i64 [[TMP32]], 0
 ; AVX512-NEXT:    [[TMP33:%.*]] = insertelement <8 x i64> poison, i64 [[A_COERCE1]], i64 0
 ; AVX512-NEXT:    [[TMP34:%.*]] = insertelement <8 x i64> [[TMP33]], i64 [[ADD5_8]], i64 1
@@ -282,9 +329,11 @@ define { i64, i64 } @avgr_16_u8(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerce0,
 ; AVX512-NEXT:    [[TMP38:%.*]] = and <8 x i64> [[TMP16]], <i64 -1, i64 255, i64 255, i64 255, i64 255, i64 255, i64 0, i64 0>
 ; AVX512-NEXT:    [[TMP39:%.*]] = add nuw nsw <8 x i64> [[TMP37]], <i64 1, i64 1, i64 1, i64 1, i64 1, i64 1, i64 0, i64 0>
 ; AVX512-NEXT:    [[TMP40:%.*]] = add nuw nsw <8 x i64> [[TMP39]], [[TMP38]]
-; AVX512-NEXT:    [[TMP41:%.*]] = shl nuw <8 x i64> [[TMP40]], <i64 55, i64 47, i64 39, i64 31, i64 23, i64 15, i64 0, i64 0>
-; AVX512-NEXT:    [[TMP42:%.*]] = and <8 x i64> [[TMP41]], <i64 -72057594037927936, i64 71776119061217280, i64 280375465082880, i64 1095216660480, i64 4278190080, i64 16711680, i64 -1, i64 -1>
-; AVX512-NEXT:    [[TMP43:%.*]] = tail call i64 @llvm.vector.reduce.or.v8i64(<8 x i64> [[TMP42]])
+; AVX512-NEXT:    [[TMP47:%.*]] = trunc <8 x i64> [[TMP40]] to <8 x i16>
+; AVX512-NEXT:    [[TMP44:%.*]] = lshr <8 x i16> [[TMP47]], <i16 1, i16 1, i16 1, i16 1, i16 1, i16 1, i16 0, i16 8>
+; AVX512-NEXT:    [[TMP45:%.*]] = bitcast <8 x i16> [[TMP44]] to <16 x i8>
+; AVX512-NEXT:    [[TMP46:%.*]] = shufflevector <16 x i8> [[TMP45]], <16 x i8> poison, <8 x i32> <i32 12, i32 14, i32 10, i32 8, i32 6, i32 4, i32 2, i32 0>
+; AVX512-NEXT:    [[TMP43:%.*]] = bitcast <8 x i8> [[TMP46]] to i64
 ; AVX512-NEXT:    [[DOTFCA_1_INSERT:%.*]] = insertvalue { i64, i64 } [[DOTFCA_0_INSERT]], i64 [[TMP43]], 1
 ; AVX512-NEXT:    ret { i64, i64 } [[DOTFCA_1_INSERT]]
 ;
@@ -553,9 +602,7 @@ define { i64, i64 } @avgr_16_u8_alt(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerc
 ; SSE4-NEXT:    [[TMP11:%.*]] = or <8 x i8> [[TMP7]], [[TMP3]]
 ; SSE4-NEXT:    [[TMP12:%.*]] = and <8 x i8> [[TMP11]], splat (i8 1)
 ; SSE4-NEXT:    [[TMP13:%.*]] = add nuw <8 x i8> [[TMP10]], [[TMP12]]
-; SSE4-NEXT:    [[TMP14:%.*]] = zext <8 x i8> [[TMP13]] to <8 x i64>
-; SSE4-NEXT:    [[TMP15:%.*]] = shl nuw <8 x i64> [[TMP14]], <i64 0, i64 8, i64 16, i64 24, i64 32, i64 40, i64 48, i64 56>
-; SSE4-NEXT:    [[VEC2STRUCT_SLOT_SROA_0_0_VEC_EXTRACT:%.*]] = tail call i64 @llvm.vector.reduce.or.v8i64(<8 x i64> [[TMP15]])
+; SSE4-NEXT:    [[VEC2STRUCT_SLOT_SROA_0_0_VEC_EXTRACT:%.*]] = bitcast <8 x i8> [[TMP13]] to i64
 ; SSE4-NEXT:    [[DOTFCA_0_INSERT:%.*]] = insertvalue { i64, i64 } poison, i64 [[VEC2STRUCT_SLOT_SROA_0_0_VEC_EXTRACT]], 0
 ; SSE4-NEXT:    [[TMP17:%.*]] = insertelement <8 x i64> poison, i64 [[A_COERCE1:%.*]], i64 0
 ; SSE4-NEXT:    [[TMP18:%.*]] = shufflevector <8 x i64> [[TMP17]], <8 x i64> poison, <8 x i32> zeroinitializer
@@ -571,9 +618,7 @@ define { i64, i64 } @avgr_16_u8_alt(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerc
 ; SSE4-NEXT:    [[TMP28:%.*]] = or <8 x i8> [[TMP24]], [[TMP20]]
 ; SSE4-NEXT:    [[TMP29:%.*]] = and <8 x i8> [[TMP28]], splat (i8 1)
 ; SSE4-NEXT:    [[TMP30:%.*]] = add nuw <8 x i8> [[TMP27]], [[TMP29]]
-; SSE4-NEXT:    [[TMP31:%.*]] = zext <8 x i8> [[TMP30]] to <8 x i64>
-; SSE4-NEXT:    [[TMP32:%.*]] = shl nuw <8 x i64> [[TMP31]], <i64 0, i64 8, i64 16, i64 24, i64 32, i64 40, i64 48, i64 56>
-; SSE4-NEXT:    [[TMP49:%.*]] = tail call i64 @llvm.vector.reduce.or.v8i64(<8 x i64> [[TMP32]])
+; SSE4-NEXT:    [[TMP49:%.*]] = bitcast <8 x i8> [[TMP30]] to i64
 ; SSE4-NEXT:    [[DOTFCA_1_INSERT:%.*]] = insertvalue { i64, i64 } [[DOTFCA_0_INSERT]], i64 [[TMP49]], 1
 ; SSE4-NEXT:    ret { i64, i64 } [[DOTFCA_1_INSERT]]
 ;
@@ -593,9 +638,7 @@ define { i64, i64 } @avgr_16_u8_alt(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerc
 ; AVX-NEXT:    [[TMP11:%.*]] = or <8 x i8> [[TMP7]], [[TMP3]]
 ; AVX-NEXT:    [[TMP12:%.*]] = and <8 x i8> [[TMP11]], splat (i8 1)
 ; AVX-NEXT:    [[TMP13:%.*]] = add nuw <8 x i8> [[TMP10]], [[TMP12]]
-; AVX-NEXT:    [[TMP14:%.*]] = zext <8 x i8> [[TMP13]] to <8 x i64>
-; AVX-NEXT:    [[TMP15:%.*]] = shl nuw <8 x i64> [[TMP14]], <i64 0, i64 8, i64 16, i64 24, i64 32, i64 40, i64 48, i64 56>
-; AVX-NEXT:    [[TMP16:%.*]] = tail call i64 @llvm.vector.reduce.or.v8i64(<8 x i64> [[TMP15]])
+; AVX-NEXT:    [[TMP16:%.*]] = bitcast <8 x i8> [[TMP13]] to i64
 ; AVX-NEXT:    [[DOTFCA_0_INSERT:%.*]] = insertvalue { i64, i64 } poison, i64 [[TMP16]], 0
 ; AVX-NEXT:    [[TMP17:%.*]] = insertelement <8 x i64> poison, i64 [[A_COERCE1:%.*]], i64 0
 ; AVX-NEXT:    [[TMP18:%.*]] = shufflevector <8 x i64> [[TMP17]], <8 x i64> poison, <8 x i32> zeroinitializer
@@ -611,9 +654,7 @@ define { i64, i64 } @avgr_16_u8_alt(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerc
 ; AVX-NEXT:    [[TMP28:%.*]] = or <8 x i8> [[TMP24]], [[TMP20]]
 ; AVX-NEXT:    [[TMP29:%.*]] = and <8 x i8> [[TMP28]], splat (i8 1)
 ; AVX-NEXT:    [[TMP30:%.*]] = add nuw <8 x i8> [[TMP27]], [[TMP29]]
-; AVX-NEXT:    [[TMP31:%.*]] = zext <8 x i8> [[TMP30]] to <8 x i64>
-; AVX-NEXT:    [[TMP32:%.*]] = shl nuw <8 x i64> [[TMP31]], <i64 0, i64 8, i64 16, i64 24, i64 32, i64 40, i64 48, i64 56>
-; AVX-NEXT:    [[TMP33:%.*]] = tail call i64 @llvm.vector.reduce.or.v8i64(<8 x i64> [[TMP32]])
+; AVX-NEXT:    [[TMP33:%.*]] = bitcast <8 x i8> [[TMP30]] to i64
 ; AVX-NEXT:    [[DOTFCA_1_INSERT:%.*]] = insertvalue { i64, i64 } [[DOTFCA_0_INSERT]], i64 [[TMP33]], 1
 ; AVX-NEXT:    ret { i64, i64 } [[DOTFCA_1_INSERT]]
 ;
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/bad-reduction.ll b/llvm/test/Transforms/SLPVectorizer/X86/bad-reduction.ll
index 45d3405417b0e..77bb4a6b3ff27 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/bad-reduction.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/bad-reduction.ll
@@ -8,9 +8,8 @@
 define i64 @load_bswap(ptr %p) {
 ; CHECK-LABEL: @load_bswap(
 ; CHECK-NEXT:    [[TMP1:%.*]] = load <8 x i8>, ptr [[P:%.*]], align 1
-; CHECK-NEXT:    [[TMP2:%.*]] = zext <8 x i8> [[TMP1]] to <8 x i64>
-; CHECK-NEXT:    [[TMP3:%.*]] = shl nuw <8 x i64> [[TMP2]], <i64 56, i64 48, i64 40, i64 32, i64 24, i64 16, i64 8, i64 0>
-; CHECK-NEXT:    [[OR01234567:%.*]] = call i64 @llvm.vector.reduce.or.v8i64(<8 x i64> [[TMP3]])
+; CHECK-NEXT:    [[TMP2:%.*]] = bitcast <8 x i8> [[TMP1]] to i64
+; CHECK-NEXT:    [[OR01234567:%.*]] = call i64 @llvm.bswap.i64(i64 [[TMP2]])
 ; CHECK-NEXT:    ret i64 [[OR01234567]]
 ;
   %g1 = getelementptr inbounds %v8i8, ptr %p, i64 0, i32 1
@@ -61,9 +60,8 @@ define i64 @load_bswap(ptr %p) {
 define i64 @load_bswap_nop_shift(ptr %p) {
 ; CHECK-LABEL: @load_bswap_nop_shift(
 ; CHECK-NEXT:    [[TMP1:%.*]] = load <8 x i8>, ptr [[P:%.*]], align 1
-; CHECK-NEXT:    [[TMP2:%.*]] = zext <8 x i8> [[TMP1]] to <8 x i64>
-; CHECK-NEXT:    [[TMP3:%.*]] = shl nuw <8 x i64> [[TMP2]], <i64 56, i64 48, i64 40, i64 32, i64 24, i64 16, i64 8, i64 0>
-; CHECK-NEXT:    [[OR01234567:%.*]] = call i64 @llvm.vector.reduce.or.v8i64(<8 x i64> [[TMP3]])
+; CHECK-NEXT:    [[TMP2:%.*]] = bitcast <8 x i8> [[TMP1]] to i64
+; CHECK-NEXT:    [[OR01234567:%.*]] = call i64 @llvm.bswap.i64(i64 [[TMP2]])
 ; CHECK-NEXT:    ret i64 [[OR01234567]]
 ;
   %g1 = getelementptr inbounds %v8i8, ptr %p, i64 0, i32 1
@@ -116,9 +114,7 @@ define i64 @load_bswap_nop_shift(ptr %p) {
 define i64 @load64le(ptr %arg) {
 ; CHECK-LABEL: @load64le(
 ; CHECK-NEXT:    [[TMP1:%.*]] = load <8 x i8>, ptr [[ARG:%.*]], align 1
-; CHECK-NEXT:    [[TMP2:%.*]] = zext <8 x i8> [[TMP1]] to <8 x i64>
-; CHECK-NEXT:    [[TMP3:%.*]] = shl nuw <8 x i64> [[TMP2]], <i64 0, i64 8, i64 16, i64 24, i64 32, i64 40, i64 48, i64 56>
-; CHECK-NEXT:    [[O7:%.*]] = call i64 @llvm.vector.reduce.or.v8i64(<8 x i64> [[TMP3]])
+; CHECK-NEXT:    [[O7:%.*]] = bitcast <8 x i8> [[TMP1]] to i64
 ; CHECK-NEXT:    ret i64 [[O7]]
 ;
   %g1 = getelementptr inbounds i8, ptr %arg, i64 1
@@ -169,9 +165,7 @@ define i64 @load64le(ptr %arg) {
 define i64 @load64le_nop_shift(ptr %arg) {
 ; CHECK-LABEL: @load64le_nop_shift(
 ; CHECK-NEXT:    [[TMP1:%.*]] = load <8 x i8>, ptr [[ARG:%.*]], align 1
-; CHECK-NEXT:    [[TMP2:%.*]] = zext <8 x i8> [[TMP1]] to <8 x i64>
-; CHECK-NEXT:    [[TMP3:%.*]] = shl nuw <8 x i64> [[TMP2]], <i64 0, i64 8, i64 16, i64 24, i64 32, i64 40, i64 48, i64 56>
-; CHECK-NEXT:    [[O7:%.*]] = call i64 @llvm.vector.reduce.or.v8i64(<8 x i64> [[TMP3]])
+; CHECK-NEXT:    [[O7:%.*]] = bitcast <8 x i8> [[TMP1]] to i64
 ; CHECK-NEXT:    ret i64 [[O7]]
 ;
   %g1 = getelementptr inbounds i8, ptr %arg, i64 1
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/load-merge-inseltpoison.ll b/llvm/test/Transforms/SLPVectorizer/X86/load-merge-inseltpoison.ll
index daf93190cab6d..1c03d8ee6c582 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/load-merge-inseltpoison.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/load-merge-inseltpoison.ll
@@ -11,9 +11,7 @@ define i32 @_Z9load_le32Ph(ptr nocapture readonly %data) {
 ; CHECK-LABEL: @_Z9load_le32Ph(
 ; CHECK-NEXT:  entry:
 ; CHECK-NEXT:    [[TMP0:%.*]] = load <4 x i8>, ptr [[DATA:%.*]], align 1
-; CHECK-NEXT:    [[TMP1:%.*]] = zext <4 x i8> [[TMP0]] to <4 x i32>
-; CHECK-NEXT:    [[TMP2:%.*]] = shl nuw <4 x i32> [[TMP1]], <i32 0, i32 8, i32 16, i32 24>
-; CHECK-NEXT:    [[OR11:%.*]] = call i32 @llvm.vector.reduce.or.v4i32(<4 x i32> [[TMP2]])
+; CHECK-NEXT:    [[OR11:%.*]] = bitcast <4 x i8> [[TMP0]] to i32
 ; CHECK-NEXT:    ret i32 [[OR11]]
 ;
 entry:
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/load-merge.ll b/llvm/test/Transforms/SLPVectorizer/X86/load-merge.ll
index b407a83333502..74984d8703eb4 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/load-merge.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/load-merge.ll
@@ -11,9 +11,7 @@ define i32 @_Z9load_le32Ph(ptr nocapture readonly %data) {
 ; CHECK-LABEL: @_Z9load_le32Ph(
 ; CHECK-NEXT:  entry:
 ; CHECK-NEXT:    [[TMP0:%.*]] = load <4 x i8>, ptr [[DATA:%.*]], align 1
-; CHECK-NEXT:    [[TMP1:%.*]] = zext <4 x i8> [[TMP0]] to <4 x i32>
-; CHECK-NEXT:    [[TMP2:%.*]] = shl nuw <4 x i32> [[TMP1]], <i32 0, i32 8, i32 16, i32 24>
-; CHECK-NEXT:    [[OR11:%.*]] = call i32 @llvm.vector.reduce.or.v4i32(<4 x i32> [[TMP2]])
+; CHECK-NEXT:    [[OR11:%.*]] = bitcast <4 x i8> [[TMP0]] to i32
 ; CHECK-NEXT:    ret i32 [[OR11]]
 ;
 entry:
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/pr48879-sroa.ll b/llvm/test/Transforms/SLPVectorizer/X86/pr48879-sroa.ll
index 9d2ee4bafbbed..30877c242d71b 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/pr48879-sroa.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/pr48879-sroa.ll
@@ -2,7 +2,7 @@
 ; RUN: opt < %s -mtriple=x86_64-unknown -passes=slp-vectorizer -mcpu=x86-64    -S | FileCheck %s --check-prefixes=SSE
 ; RUN: opt < %s -mtriple=x86_64-unknown -passes=slp-vectorizer -mcpu=x86-64-v2 -S | FileCheck %s --check-prefixes=AVX
 ; RUN: opt < %s -mtriple=x86_64-unknown -passes=slp-vectorizer -mcpu=x86-64-v3 -S | FileCheck %s --check-prefixes=AVX2
-; RUN: opt < %s -mtriple=x86_64-unknown -passes=slp-vectorizer -mcpu=x86-64-v4 -S | FileCheck %s --check-prefixes=AVX2
+; RUN: opt < %s -mtriple=x86_64-unknown -passes=slp-vectorizer -mcpu=x86-64-v4 -S | FileCheck %s --check-prefixes=AVX512
 
 define { i64, i64 } @compute_min(ptr nocapture noundef nonnull readonly align 2 dereferenceable(16) %x, ptr nocapture noundef nonnull readonly align 2 dereferenceable(16) %y) {
 ; SSE-LABEL: @compute_min(
@@ -13,15 +13,15 @@ define { i64, i64 } @compute_min(ptr nocapture noundef nonnull readonly align 2
 ; SSE-NEXT:    [[TMP1:%.*]] = load <4 x i16>, ptr [[X]], align 2
 ; SSE-NEXT:    [[TMP2:%.*]] = call <4 x i16> @llvm.smin.v4i16(<4 x i16> [[TMP0]], <4 x i16> [[TMP1]])
 ; SSE-NEXT:    [[TMP3:%.*]] = zext <4 x i16> [[TMP2]] to <4 x i64>
-; SSE-NEXT:    [[TMP4:%.*]] = shl nuw <4 x i64> [[TMP3]], <i64 0, i64 16, i64 32, i64 48>
-; SSE-NEXT:    [[RETVAL_SROA_0_0_INSERT_INSERT:%.*]] = call i64 @llvm.vector.reduce.or.v4i64(<4 x i64> [[TMP4]])
+; SSE-NEXT:    [[TMP4:%.*]] = trunc <4 x i64> [[TMP3]] to <4 x i16>
+; SSE-NEXT:    [[RETVAL_SROA_0_0_INSERT_INSERT:%.*]] = bitcast <4 x i16> [[TMP4]] to i64
 ; SSE-NEXT:    [[DOTFCA_0_INSERT:%.*]] = insertvalue { i64, i64 } poison, i64 [[RETVAL_SROA_0_0_INSERT_INSERT]], 0
 ; SSE-NEXT:    [[TMP6:%.*]] = load <4 x i16>, ptr [[ARRAYIDX_I_I10_4]], align 2
 ; SSE-NEXT:    [[TMP7:%.*]] = load <4 x i16>, ptr [[ARRAYIDX_I_I_4]], align 2
 ; SSE-NEXT:    [[TMP8:%.*]] = call <4 x i16> @llvm.smin.v4i16(<4 x i16> [[TMP6]], <4 x i16> [[TMP7]])
 ; SSE-NEXT:    [[TMP9:%.*]] = zext <4 x i16> [[TMP8]] to <4 x i64>
-; SSE-NEXT:    [[TMP10:%.*]] = shl nuw <4 x i64> [[TMP9]], <i64 0, i64 16, i64 32, i64 48>
-; SSE-NEXT:    [[RETVAL_SROA_5_8_INSERT_INSERT:%.*]] = call i64 @llvm.vector.reduce.or.v4i64(<4 x i64> [[TMP10]])
+; SSE-NEXT:    [[TMP12:%.*]] = trunc <4 x i64> [[TMP9]] to <4 x i16>
+; SSE-NEXT:    [[RETVAL_SROA_5_8_INSERT_INSERT:%.*]] = bitcast <4 x i16> [[TMP12]] to i64
 ; SSE-NEXT:    [[DOTFCA_1_INSERT:%.*]] = insertvalue { i64, i64 } [[DOTFCA_0_INSERT]], i64 [[RETVAL_SROA_5_8_INSERT_INSERT]], 1
 ; SSE-NEXT:    ret { i64, i64 } [[DOTFCA_1_INSERT]]
 ;
@@ -33,15 +33,19 @@ define { i64, i64 } @compute_min(ptr nocapture noundef nonnull readonly align 2
 ; AVX-NEXT:    [[TMP1:%.*]] = load <4 x i16>, ptr [[X]], align 2
 ; AVX-NEXT:    [[TMP2:%.*]] = call <4 x i16> @llvm.smin.v4i16(<4 x i16> [[TMP0]], <4 x i16> [[TMP1]])
 ; AVX-NEXT:    [[TMP3:%.*]] = zext <4 x i16> [[TMP2]] to <4 x i64>
-; AVX-NEXT:    [[TMP4:%.*]] = shl nuw <4 x i64> [[TMP3]], <i64 0, i64 16, i64 32, i64 48>
-; AVX-NEXT:    [[TMP24:%.*]] = call i64 @llvm.vector.reduce.or.v4i64(<4 x i64> [[TMP4]])
+; AVX-NEXT:    [[TMP4:%.*]] = trunc <4 x i64> [[TMP3]] to <4 x i32>
+; AVX-NEXT:    [[TMP5:%.*]] = bitcast <4 x i32> [[TMP4]] to <16 x i8>
+; AVX-NEXT:    [[TMP10:%.*]] = shufflevector <16 x i8> [[TMP5]], <16 x i8> poison, <8 x i32> <i32 0, i32 1, i32 4, i32 5, i32 8, i32 9, i32 12, i32 13>
+; AVX-NEXT:    [[TMP24:%.*]] = bitcast <8 x i8> [[TMP10]] to i64
 ; AVX-NEXT:    [[DOTFCA_0_INSERT:%.*]] = insertvalue { i64, i64 } poison, i64 [[TMP24]], 0
 ; AVX-NEXT:    [[TMP6:%.*]] = load <4 x i16>, ptr [[ARRAYIDX_I_I10_4]], align 2
 ; AVX-NEXT:    [[TMP7:%.*]] = load <4 x i16>, ptr [[ARRAYIDX_I_I_4]], align 2
 ; AVX-NEXT:    [[TMP8:%.*]] = call <4 x i16> @llvm.smin.v4i16(<4 x i16> [[TMP6]], <4 x i16> [[TMP7]])
 ; AVX-NEXT:    [[TMP9:%.*]] = zext <4 x i16> [[TMP8]] to <4 x i64>
-; AVX-NEXT:    [[TMP10:%.*]] = shl nuw <4 x i64> [[TMP9]], <i64 0, i64 16, i64 32, i64 48>
-; AVX-NEXT:    [[TMP25:%.*]] = call i64 @llvm.vector.reduce.or.v4i64(<4 x i64> [[TMP10]])
+; AVX-NEXT:    [[TMP12:%.*]] = trunc <4 x i64> [[TMP9]] to <4 x i32>
+; AVX-NEXT:    [[TMP13:%.*]] = bitcast <4 x i32> [[TMP12]] to <16 x i8>
+; AVX-NEXT:    [[TMP14:%.*]] = shufflevector <16 x i8> [[TMP13]], <16 x i8> poison, <8 x i32> <i32 0, i32 1, i32 4, i32 5, i32 8, i32 9, i32 12, i32 13>
+; AVX-NEXT:    [[TMP25:%.*]] = bitcast <8 x i8> [[TMP14]] to i64
 ; AVX-NEXT:    [[DOTFCA_1_INSERT:%.*]] = insertvalue { i64, i64 } [[DOTFCA_0_INSERT]], i64 [[TMP25]], 1
 ; AVX-NEXT:    ret { i64, i64 } [[DOTFCA_1_INSERT]]
 ;
@@ -53,18 +57,42 @@ define { i64, i64 } @compute_min(ptr nocapture noundef nonnull readonly align 2
 ; AVX2-NEXT:    [[TMP1:%.*]] = load <4 x i16>, ptr [[X]], align 2
 ; AVX2-NEXT:    [[TMP2:%.*]] = call <4 x i16> @llvm.smin.v4i16(<4 x i16> [[TMP0]], <4 x i16> [[TMP1]])
 ; AVX2-NEXT:    [[TMP3:%.*]] = zext <4 x i16> [[TMP2]] to <4 x i64>
-; AVX2-NEXT:    [[TMP4:%.*]] = shl nuw <4 x i64> [[TMP3]], <i64 0, i64 16, i64 32, i64 48>
-; AVX2-NEXT:    [[TMP24:%.*]] = call i64 @llvm.vector.reduce.or.v4i64(<4 x i64> [[TMP4]])
-; AVX2-NEXT:    [[DOTFCA_0_INSERT:%.*]] = insertvalue { i64, i64 } poison, i64 [[TMP24]], 0
-; AVX2-NEXT:    [[TMP6:%.*]] = load <4 x i16>, ptr [[ARRAYIDX_I_I10_4]], align 2
-; AVX2-NEXT:    [[TMP7:%.*]] = load <4 x i16>, ptr [[ARRAYIDX_I_I_4]], align 2
-; AVX2-NEXT:    [[TMP8:%.*]] = call <4 x i16> @llvm.smin.v4i16(<4 x i16> [[TMP6]], <4 x i16> [[TMP7]])
-; AVX2-NEXT:    [[TMP9:%.*]] = zext <4 x i16> [[TMP8]] to <4 x i64>
-; AVX2-NEXT:    [[TMP10:%.*]] = shl nuw <4 x i64> [[TMP9]], <i64 0, i64 16, i64 32, i64 48>
-; AVX2-NEXT:    [[TMP25:%.*]] = call i64 @llvm.vector.reduce.or.v4i64(<4 x i64> [[TMP10]])
-; AVX2-NEXT:    [[DOTFCA_1_INSERT:%.*]] = insertvalue { i64, i64 } [[DOTFCA_0_INSERT]], i64 [[TMP25]], 1
+; AVX2-NEXT:    [[TMP4:%.*]] = trunc <4 x i64> [[TMP3]] to <4 x i32>
+; AVX2-NEXT:    [[TMP5:%.*]] = bitcast <4 x i32> [[TMP4]] to <16 x i8>
+; AVX2-NEXT:    [[TMP6:%.*]] = shufflevector <16 x i8> [[TMP5]], <16 x i8> poison, <8 x i32> <i32 0, i32 1, i32 4, i32 5, i32 8, i32 9, i32 12, i32 13>
+; AVX2-NEXT:    [[TMP7:%.*]] = bitcast <8 x i8> [[TMP6]] to i64
+; AVX2-NEXT:    [[DOTFCA_0_INSERT:%.*]] = insertvalue { i64, i64 } poison, i64 [[TMP7]], 0
+; AVX2-NEXT:    [[TMP8:%.*]] = load <4 x i16>, ptr [[ARRAYIDX_I_I10_4]], align 2
+; AVX2-NEXT:    [[TMP9:%.*]] = load <4 x i16>, ptr [[ARRAYIDX_I_I_4]], align 2
+; AVX2-NEXT:    [[TMP10:%.*]] = call <4 x i16> @llvm.smin.v4i16(<4 x i16> [[TMP8]], <4 x i16> [[TMP9]])
+; AVX2-NEXT:    [[TMP11:%.*]] = zext <4 x i16> [[TMP10]] to <4 x i64>
+; AVX2-NEXT:    [[TMP12:%.*]] = trunc <4 x i64> [[TMP11]] to <4 x i32>
+; AVX2-NEXT:    [[TMP13:%.*]] = bitcast <4 x i32> [[TMP12]] to <16 x i8>
+; AVX2-NEXT:    [[TMP14:%.*]] = shufflevector <16 x i8> [[TMP13]], <16 x i8> poison, <8 x i32> <i32 0, i32 1, i32 4, i32 5, i32 8, i32 9, i32 12, i32 13>
+; AVX2-NEXT:    [[TMP15:%.*]] = bitcast <8 x i8> [[TMP14]] to i64
+; AVX2-NEXT:    [[DOTFCA_1_INSERT:%.*]] = insertvalue { i64, i64 } [[DOTFCA_0_INSERT]], i64 [[TMP15]], 1
 ; AVX2-NEXT:    ret { i64, i64 } [[DOTFCA_1_INSERT]]
 ;
+; AVX512-LABEL: @compute_min(
+; AVX512-NEXT:  entry:
+; AVX512-NEXT:    [[ARRAYIDX_I_I_4:%.*]] = getelementptr inbounds [8 x i16], ptr [[X:%.*]], i64 0, i64 4
+; AVX512-NEXT:    [[ARRAYIDX_I_I10_4:%.*]] = getelementptr inbounds [8 x i16], ptr [[Y:%.*]], i64 0, i64 4
+; AVX512-NEXT:    [[TMP0:%.*]] = load <4 x i16>, ptr [[Y]], align 2
+; AVX512-NEXT:    [[TMP1:%.*]] = load <4 x i16>, ptr [[X]], align 2
+; AVX512-NEXT:    [[TMP2:%.*]] = call <4 x i16> @llvm.smin.v4i16(<4 x i16> [[TMP0]], <4 x i16> [[TMP1]])
+; AVX512-NEXT:    [[TMP3:%.*]] = zext <4 x i16> [[TMP2]] to <4 x i64>
+; AVX512-NEXT:    [[TMP4:%.*]] = trunc <4 x i64> [[TMP3]] to <4 x i16>
+; AVX512-NEXT:    [[TMP5:%.*]] = bitcast <4 x i16> [[TMP4]] to i64
+; AVX512-NEXT:    [[DOTFCA_0_INSERT:%.*]] = insertvalue { i64, i64 } poison, i64 [[TMP5]], 0
+; AVX512-NEXT:    [[TMP6:%.*]] = load <4 x i16>, ptr [[ARRAYIDX_I_I10_4]], align 2
+; AVX512-NEXT:    [[TMP7:%.*]] = load <4 x i16>, ptr [[ARRAYIDX_I_I_4]], align 2
+; AVX512-NEXT:    [[TMP8:%.*]] = call <4 x i16> @llvm.smin.v4i16(<4 x i16> [[TMP6]], <4 x i16> [[TMP7]])
+; AVX512-NEXT:    [[TMP9:%.*]] = zext <4 x i16> [[TMP8]] to <4 x i64>
+; AVX512-NEXT:    [[TMP10:%.*]] = trunc <4 x i64> [[TMP9]] to <4 x i16>
+; AVX512-NEXT:    [[TMP11:%.*]] = bitcast <4 x i16> [[TMP10]] to i64
+; AVX512-NEXT:    [[DOTFCA_1_INSERT:%.*]] = insertvalue { i64, i64 } [[DOTFCA_0_INSERT]], i64 [[TMP11]], 1
+; AVX512-NEXT:    ret { i64, i64 } [[DOTFCA_1_INSERT]]
+;
 entry:
   %0 = load i16, ptr %y, align 2
   %1 = load i16, ptr %x, align 2
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/reduce-or-bitpack.ll b/llvm/test/Transforms/SLPVectorizer/X86/reduce-or-bitpack.ll
index 6c4ec216e1b2f..a5c5177787190 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/reduce-or-bitpack.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/reduce-or-bitpack.ll
@@ -12,8 +12,8 @@ define i32 @shl_and_pack(i32 %x) {
 ; CHECK-NEXT:    [[TMP1:%.*]] = shufflevector <4 x i32> [[TMP0]], <4 x i32> poison, <4 x i32> zeroinitializer
 ; CHECK-NEXT:    [[TMP2:%.*]] = lshr <4 x i32> [[TMP1]], <i32 0, i32 8, i32 16, i32 24>
 ; CHECK-NEXT:    [[TMP3:%.*]] = and <4 x i32> [[TMP2]], <i32 255, i32 255, i32 255, i32 -1>
-; CHECK-NEXT:    [[TMP4:%.*]] = shl nuw <4 x i32> [[TMP3]], <i32 0, i32 8, i32 16, i32 24>
-; CHECK-NEXT:    [[TMP6:%.*]] = call i32 @llvm.vector.reduce.or.v4i32(<4 x i32> [[TMP4]])
+; CHECK-NEXT:    [[TMP4:%.*]] = trunc <4 x i32> [[TMP3]] to <4 x i8>
+; CHECK-NEXT:    [[TMP6:%.*]] = bitcast <4 x i8> [[TMP4]] to i32
 ; CHECK-NEXT:    ret i32 [[TMP6]]
 ;
 entry:
@@ -109,9 +109,11 @@ define dso_local { i64, i64 } @and_shl_add_pack(i64 %0, i64 %1, i64 %2, i64 %3)
 ; CHECK-NEXT:    [[TMP42:%.*]] = and <8 x i64> [[TMP41]], <i64 -1, i64 255, i64 255, i64 255, i64 255, i64 255, i64 -1, i64 -1>
 ; CHECK-NEXT:    [[TMP43:%.*]] = add nuw nsw <8 x i64> [[TMP42]], <i64 1, i64 1, i64 1, i64 1, i64 1, i64 1, i64 0, i64 0>
 ; CHECK-NEXT:    [[TMP44:%.*]] = add nuw nsw <8 x i64> [[TMP43]], [[TMP26]]
-; CHECK-NEXT:    [[TMP45:%.*]] = shl nuw <8 x i64> [[TMP44]], <i64 55, i64 47, i64 39, i64 31, i64 23, i64 15, i64 0, i64 0>
-; CHECK-NEXT:    [[TMP46:%.*]] = and <8 x i64> [[TMP45]], <i64 -72057594037927936, i64 71776119061217280, i64 280375465082880, i64 1095216660480, i64 4278190080, i64 16711680, i64 -1, i64 -1>
-; CHECK-NEXT:    [[TMP49:%.*]] = call i64 @llvm.vector.reduce.or.v8i64(<8 x i64> [[TMP46]])
+; CHECK-NEXT:    [[TMP45:%.*]] = trunc <8 x i64> [[TMP44]] to <8 x i32>
+; CHECK-NEXT:    [[TMP46:%.*]] = lshr <8 x i32> [[TMP45]], <i32 1, i32 1, i32 1, i32 1, i32 1, i32 1, i32 0, i32 8>
+; CHECK-NEXT:    [[TMP47:%.*]] = bitcast <8 x i32> [[TMP46]] to <32 x i8>
+; CHECK-NEXT:    [[TMP48:%.*]] = shufflevector <32 x i8> [[TMP47]], <32 x i8> poison, <8 x i32> <i32 24, i32 28, i32 20, i32 16, i32 12, i32 8, i32 4, i32 0>
+; CHECK-NEXT:    [[TMP49:%.*]] = bitcast <8 x i8> [[TMP48]] to i64
 ; CHECK-NEXT:    [[TMP50:%.*]] = insertvalue { i64, i64 } poison, i64 [[TMP49]], 0
 ; CHECK-NEXT:    [[TMP51:%.*]] = insertelement <8 x i64> poison, i64 [[TMP1]], i64 0
 ; CHECK-NEXT:    [[TMP52:%.*]] = insertelement <8 x i64> [[TMP51]], i64 [[TMP20]], i64 1
@@ -123,9 +125,11 @@ define dso_local { i64, i64 } @and_shl_add_pack(i64 %0, i64 %1, i64 %2, i64 %3)
 ; CHECK-NEXT:    [[TMP58:%.*]] = and <8 x i64> <i64 -1, i64 255, i64 255, i64 255, i64 255, i64 255, i64 0, i64 0>, [[TMP29]]
 ; CHECK-NEXT:    [[TMP59:%.*]] = add nuw nsw <8 x i64> [[TMP57]], <i64 1, i64 1, i64 1, i64 1, i64 1, i64 1, i64 0, i64 0>
 ; CHECK-NEXT:    [[TMP60:%.*]] = add nuw nsw <8 x i64> [[TMP59]], [[TMP58]]
-; CHECK-NEXT:    [[TMP61:%.*]] = shl nuw <8 x i64> [[TMP60]], <i64 55, i64 47, i64 39, i64 31, i64 23, i64 15, i64 0, i64 0>
-; CHECK-NEXT:    [[TMP62:%.*]] = and <8 x i64> [[TMP61]], <i64 -72057594037927936, i64 71776119061217280, i64 280375465082880, i64 1095216660480, i64 4278190080, i64 16711680, i64 -1, i64 -1>
-; CHECK-NEXT:    [[TMP65:%.*]] = call i64 @llvm.vector.reduce.or.v8i64(<8 x i64> [[TMP62]])
+; CHECK-NEXT:    [[TMP61:%.*]] = trunc <8 x i64> [[TMP60]] to <8 x i32>
+; CHECK-NEXT:    [[TMP62:%.*]] = lshr <8 x i32> [[TMP61]], <i32 1, i32 1, i32 1, i32 1, i32 1, i32 1, i32 0, i32 8>
+; CHECK-NEXT:    [[TMP63:%.*]] = bitcast <8 x i32> [[TMP62]] to <32 x i8>
+; CHECK-NEXT:    [[TMP64:%.*]] = shufflevector <32 x i8> [[TMP63]], <32 x i8> poison, <8 x i32> <i32 24, i32 28, i32 20, i32 16, i32 12, i32 8, i32 4, i32 0>
+; CHECK-NEXT:    [[TMP65:%.*]] = bitcast <8 x i8> [[TMP64]] to i64
 ; CHECK-NEXT:    [[TMP66:%.*]] = insertvalue { i64, i64 } [[TMP50]], i64 [[TMP65]], 1
 ; CHECK-NEXT:    ret { i64, i64 } [[TMP66]]
 ;
diff --git a/llvm/test/Transforms/SLPVectorizer/non-power-of-2-bswap.ll b/llvm/test/Transforms/SLPVectorizer/non-power-of-2-bswap.ll
index 923663c2adb22..7732167767b8a 100644
--- a/llvm/test/Transforms/SLPVectorizer/non-power-of-2-bswap.ll
+++ b/llvm/test/Transforms/SLPVectorizer/non-power-of-2-bswap.ll
@@ -7,9 +7,8 @@ define i64 @bswap_i24(ptr noalias %p, ptr noalias %p1) {
 ; CHECK-NEXT:    [[TMP1:%.*]] = load <3 x i8>, ptr [[P]], align 1
 ; CHECK-NEXT:    [[TMP2:%.*]] = load <3 x i8>, ptr [[P1]], align 1
 ; CHECK-NEXT:    [[TMP3:%.*]] = add <3 x i8> [[TMP1]], [[TMP2]]
-; CHECK-NEXT:    [[TMP4:%.*]] = zext <3 x i8> [[TMP3]] to <3 x i32>
-; CHECK-NEXT:    [[TMP5:%.*]] = shl <3 x i32> [[TMP4]], <i32 16, i32 8, i32 0>
-; CHECK-NEXT:    [[TMP8:%.*]] = call i32 @llvm.vector.reduce.or.v3i32(<3 x i32> [[TMP5]])
+; CHECK-NEXT:    [[TMP6:%.*]] = shufflevector <3 x i8> [[TMP3]], <3 x i8> zeroinitializer, <4 x i32> <i32 2, i32 1, i32 0, i32 3>
+; CHECK-NEXT:    [[TMP8:%.*]] = bitcast <4 x i8> [[TMP6]] to i32
 ; CHECK-NEXT:    [[TMP9:%.*]] = zext i32 [[TMP8]] to i64
 ; CHECK-NEXT:    ret i64 [[TMP9]]
 ;



More information about the llvm-commits mailing list