[llvm] [SLP]Model gathers of extracted integer sub-fields as bitcast+permute+ext (PR #224919)

Alexey Bataev via llvm-commits llvm-commits at lists.llvm.org
Thu Sep 24 06:30:18 PDT 2026


https://github.com/alexey-bataev updated https://github.com/llvm/llvm-project/pull/224919

>From 18397d8949537b75c58d58a25a94afedeaa4cfa0 Mon Sep 17 00:00:00 2001
From: Alexey Bataev <a.bataev at outlook.com>
Date: Sun, 20 Sep 2026 04:53:37 -0700
Subject: [PATCH] =?UTF-8?q?[=F0=9D=98=80=F0=9D=97=BD=F0=9D=97=BF]=20initia?=
 =?UTF-8?q?l=20version?=
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit

Created using spr 1.3.7
---
 .../Transforms/Vectorize/SLPVectorizer.cpp    | 130 +++-
 .../Vectorize/SLPVectorizer/SLPUtils.cpp      | 133 ++++
 .../Vectorize/SLPVectorizer/SLPUtils.h        |   9 +
 .../AArch64/scalarize-load-ext-extract.ll     |   7 +-
 llvm/test/Transforms/PhaseOrdering/X86/avg.ll | 722 +++---------------
 .../SLPVectorizer/X86/extracted-subfields.ll  | 156 +---
 .../SLPVectorizer/X86/reduce-or-bitpack.ll    |   6 +-
 7 files changed, 416 insertions(+), 747 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 205311d10e76c2..dfc2fe49b8cee6 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -2552,6 +2552,15 @@ class slpvectorizer::BoUpSLP {
   template <typename BVTy, typename ResTy, typename... Args>
   ResTy processBuildVector(const TreeEntry *E, Type *ScalarTy, Args &...Params);
 
+  /// Handles the gather of zero-extended sub-fields of the same wider integer
+  /// scalar, emitted as a bitcast to the field vector plus a permutation and
+  /// an extension to the lane type, if the gathered values match.
+  template <typename ResTy, typename BVTy>
+  std::optional<ResTy> processExtractedFieldsGather(
+      BVTy &ShuffleBuilder, const TreeEntry *E, Type *ScalarTy,
+      ArrayRef<Value *> VL,
+      ArrayRef<std::pair<const TreeEntry *, unsigned>> SubVectors);
+
   /// Create a new vector from a list of scalar values.  Produces a sequence
   /// which exploits values reused across lanes, and arranges the inserts
   /// for ease of later optimization.
@@ -12081,7 +12090,12 @@ void BoUpSLP::buildTreeRec(ArrayRef<Value *> VLRef, unsigned Depth,
   TreeEntry::EntryState State = getScalarsVectorizationState(
       S, VL, IsScatterVectorizeUserTE, CurrentOrder, PointerOps, SPtrInfo,
       ExpandShuffleMask);
-  if (State == TreeEntry::NeedToGather) {
+  // The zero-extended sub-fields of the same wider integer scalar are
+  // gathered and emitted as a bitcast to the field vector plus a permutation.
+  // For 2 lanes the emission is not cheaper than the insertelement chain.
+  if (State == TreeEntry::NeedToGather ||
+      (VL.size() > 2 && ReuseShuffleIndices.empty() &&
+       matchGatheredExtractedFields(VL, *DL))) {
     newGatherTreeEntry(VL, S, UserTreeIdx, ReuseShuffleIndices);
     return;
   }
@@ -14145,7 +14159,9 @@ void BoUpSLP::transformNodes() {
             // We use allSameOpcode instead of isAltShuffle because we don't
             // want to use interchangeable instruction here.
             !allSameOpcode(VL) || !allSameBlock(VL)) ||
-          allConstant(VL) || isSplat(VL))
+          allConstant(VL) || isSplat(VL) ||
+          (E.ReuseShuffleIndices.empty() &&
+           matchGatheredExtractedFields(VL, *DL)))
         continue;
       if (ForceLoadGather && E.hasState() && E.getOpcode() == Instruction::Load)
         continue;
@@ -15527,6 +15543,43 @@ class BoUpSLP::ShuffleCostEstimator : public BaseShuffleAnalysis {
             cast<FixedVectorType>(Root->getType())->getNumElements()),
         getAllOnesValue(*R.DL, ScalarTy->getScalarType()));
   }
+  /// Estimates the cost of the gather of zero-extended sub-fields of width
+  /// \p FieldWidth of the same wider integer scalar \p Src, permuted by
+  /// \p Mask, as a bitcast to the field vector plus a permutation and an
+  /// extension to the lane type.
+  InstructionCost createExtractedFieldsVector(Value *Src, unsigned FieldWidth,
+                                              ArrayRef<int> Mask,
+                                              const TreeEntry &E) {
+    auto *FieldTy = IntegerType::get(Src->getContext(), FieldWidth);
+    unsigned SrcWidth = Src->getType()->getIntegerBitWidth();
+    assert(SrcWidth % FieldWidth == 0 &&
+           "Expected the field width to divide the source width.");
+    unsigned NumFields = SrcWidth / FieldWidth;
+    auto *FieldVecTy = cast<VectorType>(getWidenedType(FieldTy, NumFields));
+    TTI::CastContextHint CCH = R.getCastContextHint(E);
+    InstructionCost Cost = TTI.getCastInstrCost(
+        Instruction::BitCast, FieldVecTy, Src->getType(), CCH, CostKind);
+    // The permutation is elided only for the full-length identity mask.
+    if (Mask.size() != NumFields ||
+        !ShuffleVectorInst::isIdentityMask(Mask, NumFields))
+      Cost += getShuffleCost(TTI, TTI::SK_PermuteSingleSrc, FieldVecTy,
+                             CostKind, Mask);
+    if (!ScalarTy->isIntegerTy(FieldWidth))
+      Cost += TTI.getCastInstrCost(
+          Instruction::ZExt, getWidenedType(ScalarTy, Mask.size()),
+          getWidenedType(FieldTy, Mask.size()), CCH, CostKind);
+    // The extraction instructions die once the fields are emitted from the
+    // source scalar; their scalar cost is credited back for the gathers
+    // directly feeding the reduction. Instructions vectorized elsewhere are
+    // skipped: their scalar cost is already a part of the scalar baseline.
+    if (R.UserIgnoreList &&
+        (!E.UserTreeIndex || E.UserTreeIndex.UserTE->Idx == 0))
+      for (Instruction *I : make_isa_range<Instruction>(E.Scalars))
+        if (CheckedExtracts.insert(I).second && !R.isVectorized(I) &&
+            R.areAllUsersVectorized(I, &VectorizedVals))
+          Cost -= TTI.getInstructionCost(I, CostKind);
+    return Cost;
+  }
   InstructionCost createFreeze(InstructionCost Cost) { return Cost; }
   /// Finalize emission of the shuffles.
   InstructionCost finalize(
@@ -15664,6 +15717,17 @@ TTI::CastContextHint BoUpSLP::getCastContextHint(const TreeEntry &TE) const {
     if (ShuffleVectorInst::isReverseMask(Mask, Mask.size()))
       return TTI::CastContextHint::Reversed;
   }
+  // A gather of extracted sub-fields inherits the context of the entry
+  // vectorizing the common source scalar, or of the source scalar load.
+  if (TE.isGather())
+    if (std::optional<std::tuple<Value *, unsigned, SmallVector<int>>> Fields =
+            matchGatheredExtractedFields(TE.Scalars, *DL)) {
+      Value *Src = std::get<0>(*Fields);
+      if (ArrayRef<TreeEntry *> TEs = getTreeEntries(Src); !TEs.empty())
+        return getCastContextHint(*TEs.front());
+      if (isa<LoadInst>(Src))
+        return TTI::CastContextHint::Normal;
+    }
   return TTI::CastContextHint::None;
 }
 
@@ -17458,6 +17522,7 @@ bool BoUpSLP::isFullyVectorizableTinyTree(bool ForReduction) const {
                    [this](Value *V) { return EphValues.contains(V); }) &&
            (allConstant(TE->Scalars) || isSplat(TE->Scalars) ||
             TE->Scalars.size() < Limit ||
+            matchGatheredExtractedFields(TE->Scalars, *DL) ||
             // Nodes with copyable lanes may mix in non-extract lanes, which
             // are not representable as a shuffle of the source vector.
             (((TE->hasState() &&
@@ -22275,6 +22340,26 @@ class BoUpSLP::ShuffleInstructionBuilder final : public BaseShuffleAnalysis {
                       return createShuffle(V1, V2, Mask);
                     });
   }
+  /// Emits the gather of zero-extended sub-fields of width \p FieldWidth of
+  /// the same wider integer scalar \p Src, permuted by \p Mask, as a bitcast
+  /// to the field vector plus a permutation and an extension to the lane type.
+  Value *createExtractedFieldsVector(Value *Src, unsigned FieldWidth,
+                                     ArrayRef<int> Mask, const TreeEntry &) {
+    auto *FieldTy = IntegerType::get(Src->getContext(), FieldWidth);
+    unsigned SrcWidth = Src->getType()->getIntegerBitWidth();
+    assert(SrcWidth % FieldWidth == 0 &&
+           "Expected the field width to divide the source width.");
+    Value *Vec = Builder.CreateBitCast(
+        Src, getWidenedType(FieldTy, SrcWidth / FieldWidth));
+    if (auto *I = dyn_cast<Instruction>(Vec)) {
+      R.GatherShuffleExtractSeq.insert(I);
+      R.CSEBlocks.insert(I->getParent());
+    }
+    Vec = createShuffle(Vec, /*V2=*/nullptr, Mask);
+    // The extracted fields are unsigned.
+    Vec = castToScalarTyElem(Vec, /*IsSigned=*/false);
+    return Vec;
+  }
   Value *createFreeze(Value *V) { return Builder.CreateFreeze(V); }
   /// Finalize emission of the shuffles.
   /// \param Action the action (if any) to be performed before final applying of
@@ -22394,6 +22479,44 @@ Value *BoUpSLP::vectorizeOperand(TreeEntry *E, unsigned NodeIdx) {
   return vectorizeTree(getOperandEntry(E, NodeIdx));
 }
 
+template <typename ResTy, typename BVTy>
+std::optional<ResTy> BoUpSLP::processExtractedFieldsGather(
+    BVTy &ShuffleBuilder, const TreeEntry *E, Type *ScalarTy,
+    ArrayRef<Value *> VL,
+    ArrayRef<std::pair<const TreeEntry *, unsigned>> SubVectors) {
+  if (!SubVectors.empty() || !E->ReuseShuffleIndices.empty() ||
+      !ScalarTy->isIntegerTy())
+    return std::nullopt;
+  std::optional<std::tuple<Value *, unsigned, SmallVector<int>>>
+      ExtractedFields = matchGatheredExtractedFields(VL, *DL);
+  if (!ExtractedFields ||
+      ScalarTy->getIntegerBitWidth() < std::get<1>(*ExtractedFields))
+    return std::nullopt;
+  Value *Src = std::get<0>(*ExtractedFields);
+  unsigned FieldWidth = std::get<1>(*ExtractedFields);
+  SmallVector<int> &Mask = std::get<2>(*ExtractedFields);
+  unsigned NumFields = Src->getType()->getIntegerBitWidth() / FieldWidth;
+  // The lane order of the reduction root is unobservable when every gathered
+  // scalar is used only by the reduction operations and each lane holds a
+  // distinct field: emit the fields in the natural order and skip the
+  // permutation.
+  if (!E->UserTreeIndex && UserIgnoreList && Mask.size() == NumFields &&
+      all_of(make_isa_range<Instruction>(VL), [&](Instruction *I) {
+        return !I->hasNUsesOrMore(UsesLimit) &&
+               all_of(I->users(),
+                      [&](User *U) { return UserIgnoreList->contains(U); });
+      })) {
+    SmallVector<int> SortedMask(Mask);
+    sort(SortedMask);
+    // Each field is held at most once; poison lanes are ignored.
+    if (adjacent_find(SortedMask, [](int A, int B) {
+          return A != PoisonMaskElem && A == B;
+        }) == SortedMask.end())
+      std::iota(Mask.begin(), Mask.end(), 0);
+  }
+  return ShuffleBuilder.createExtractedFieldsVector(Src, FieldWidth, Mask, *E);
+}
+
 template <typename BVTy, typename ResTy, typename... Args>
 ResTy BoUpSLP::processBuildVector(const TreeEntry *E, Type *ScalarTy,
                                   Args &...Params) {
@@ -22505,6 +22628,9 @@ ResTy BoUpSLP::processBuildVector(const TreeEntry *E, Type *ScalarTy,
   Type *OrigScalarTy = GatheredScalars.front()->getType();
   auto *VecTy = getWidenedType(ScalarTy, GatheredScalars.size());
   unsigned NumParts = getNumberOfParts(VecTy, ScalarTy, GatheredScalars.size());
+  if (std::optional<ResTy> Res = processExtractedFieldsGather<ResTy>(
+          ShuffleBuilder, E, ScalarTy, GatheredScalars, SubVectors))
+    return *Res;
   if (!all_of(GatheredScalars, IsaPred<UndefValue>)) {
     // Check for gathered extracts.
     bool Resized = false;
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp
index cd6319fd6cce50..85ee4a6c0c4ab8 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp
@@ -1139,6 +1139,139 @@ TargetTransformInfo::TargetCostKind getSLPCostKind(const Function *F) {
   return F->hasOptSize() ? TTI::TCK_CodeSize : TTI::TCK_RecipThroughput;
 }
 
+/// Checks if \p V is a zero-extended sub-field of a wider integer scalar.
+/// Returns the source scalar, the field width and the field offset.
+static std::optional<std::tuple<Value *, unsigned, unsigned>>
+matchExtractedField(Value *V) {
+  if (!V->getType()->isIntegerTy())
+    return std::nullopt;
+  // Field offset for the field-aligned shift amount, if the shifted value of
+  // the given bit width keeps at least one full field.
+  auto GetFieldOffset = [](const APInt *Amt, unsigned BitWidth,
+                           unsigned FieldWidth) -> std::optional<unsigned> {
+    uint64_t ShAmt = Amt->getLimitedValue(BitWidth);
+    if (ShAmt % FieldWidth != 0 || ShAmt + FieldWidth > BitWidth)
+      return std::nullopt;
+    return ShAmt / FieldWidth;
+  };
+  // Checks if the low bits of Val are a sub-field of the given width of a
+  // wider integer scalar. Val is a scalar integer, since V is one, and so is
+  // the matched source.
+  auto MatchLowField =
+      [&](Value *Val,
+          unsigned FieldWidth) -> std::optional<std::pair<Value *, unsigned>> {
+    Value *Src;
+    const APInt *Amt;
+    // Only the low bits of Val are observed, so lshr and ashr are equivalent.
+    if (match(Val, m_Trunc(m_Shr(m_Value(Src), m_APInt(Amt)))) ||
+        match(Val, m_Shr(m_Value(Src), m_APInt(Amt)))) {
+      if (std::optional<unsigned> Offset = GetFieldOffset(
+              Amt, Src->getType()->getIntegerBitWidth(), FieldWidth))
+        return std::make_pair(Src, *Offset);
+      return std::nullopt;
+    }
+    if (match(Val, m_Shr(m_Trunc(m_Value(Src)), m_APInt(Amt)))) {
+      if (std::optional<unsigned> Offset = GetFieldOffset(
+              Amt, Val->getType()->getIntegerBitWidth(), FieldWidth))
+        return std::make_pair(Src, *Offset);
+      return std::nullopt;
+    }
+    if (match(Val, m_Trunc(m_Value(Src))) &&
+        Src->getType()->getIntegerBitWidth() >= FieldWidth)
+      return std::make_pair(Src, 0u);
+    // Val itself is the source of its low field.
+    if (Val->getType()->getIntegerBitWidth() > FieldWidth)
+      return std::make_pair(Val, 0u);
+    return std::nullopt;
+  };
+  Value *Val;
+  const APInt *Mask;
+  // and Val, (1 << FieldWidth) - 1 or zext i<FieldWidth> Val - the low bits of
+  // Val.
+  unsigned FieldWidth = 0;
+  if (match(V, m_c_And(m_Value(Val), m_APInt(Mask))) && Mask->isMask())
+    FieldWidth = Mask->popcount();
+  else if (match(V, m_ZExt(m_Value(Val))))
+    FieldWidth = Val->getType()->getIntegerBitWidth();
+  if (FieldWidth != 0) {
+    if (std::optional<std::pair<Value *, unsigned>> Field =
+            MatchLowField(Val, FieldWidth))
+      return std::make_tuple(Field->first, FieldWidth, Field->second);
+    return std::nullopt;
+  }
+  unsigned LaneWidth = V->getType()->getIntegerBitWidth();
+  Value *Src;
+  const APInt *Amt;
+  if (match(V, m_Trunc(m_LShr(m_Value(Src), m_APInt(Amt))))) {
+    unsigned SrcWidth = Src->getType()->getIntegerBitWidth();
+    uint64_t ShAmt = Amt->getLimitedValue(SrcWidth);
+    // The field itself, if the lane width is the field width.
+    if (std::optional<unsigned> Offset =
+            GetFieldOffset(Amt, SrcWidth, LaneWidth))
+      return std::make_tuple(Src, LaneWidth, *Offset);
+    // The zero-extended top field of the source.
+    unsigned FieldWidth = SrcWidth - ShAmt;
+    if (FieldWidth > 0 && FieldWidth < LaneWidth && ShAmt % FieldWidth == 0)
+      return std::make_tuple(Src, FieldWidth, ShAmt / FieldWidth);
+    return std::nullopt;
+  }
+  if (match(V, m_LShr(m_Value(Src), m_APInt(Amt)))) {
+    // The zero-extended top field of the source, if the result keeps exactly
+    // one field. Look through a truncation of the shifted value.
+    unsigned ShfWidth = Src->getType()->getIntegerBitWidth();
+    uint64_t ShAmt = Amt->getLimitedValue(ShfWidth);
+    unsigned FieldWidth = ShfWidth - ShAmt;
+    if (FieldWidth > 0 && ShAmt % FieldWidth == 0) {
+      match(Src, m_Trunc(m_Value(Src)));
+      return std::make_tuple(Src, FieldWidth, ShAmt / FieldWidth);
+    }
+    return std::nullopt;
+  }
+  if (match(V, m_Trunc(m_Value(Src))))
+    return std::make_tuple(Src, LaneWidth, 0u);
+  return std::nullopt;
+}
+
+std::optional<std::tuple<Value *, unsigned, SmallVector<int>>>
+matchGatheredExtractedFields(ArrayRef<Value *> VL, const DataLayout &DL) {
+  // Splats are emitted as broadcasts, sub-fields of a constant are folded.
+  // The bitcast to the field vector maps lane 0 to the least significant
+  // field on little-endian targets only.
+  if (VL.size() < 2 || !VL.front()->getType()->isIntegerTy() || isSplat(VL) ||
+      DL.isBigEndian())
+    return std::nullopt;
+  Value *Src = nullptr;
+  unsigned FieldWidth = 0;
+  SmallVector<int> Mask(VL.size(), PoisonMaskElem);
+  for (auto [Idx, V] : enumerate(VL)) {
+    if (isa<UndefValue>(V))
+      continue;
+    if (V->getType() != VL.front()->getType())
+      return std::nullopt;
+    std::optional<std::tuple<Value *, unsigned, unsigned>> Field =
+        matchExtractedField(V);
+    if (!Field || (Src && (Src != std::get<0>(*Field) ||
+                           FieldWidth != std::get<1>(*Field))))
+      return std::nullopt;
+    Src = std::get<0>(*Field);
+    FieldWidth = std::get<1>(*Field);
+    Mask[Idx] = std::get<2>(*Field);
+  }
+  // The field width is a whole number of bytes and divides the source
+  // exactly, same as for the packing layout, so the source bitcasts to the
+  // field vector.
+  if (!Src || isa<Constant>(Src) || FieldWidth % 8 != 0 ||
+      Src->getType()->getIntegerBitWidth() % FieldWidth != 0)
+    return std::nullopt;
+  // The same field in every lane is a splat, emitted as a broadcast.
+  if (all_of(Mask, [First = *find_if(Mask, not_equal_to(PoisonMaskElem))](
+                       int MaskElt) {
+        return MaskElt == PoisonMaskElem || MaskElt == First;
+      }))
+    return std::nullopt;
+  return std::make_tuple(Src, FieldWidth, std::move(Mask));
+}
+
 /// Deeper than the standard analysis recursion depth to keep the numeric
 /// bound precise through arithmetic carry chains.
 constexpr unsigned MaxBitPackAnalysisDepth = MaxAnalysisRecursionDepth + 2;
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.h b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.h
index bffe539822341c..4ac6880d0f17d0 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.h
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.h
@@ -30,6 +30,7 @@
 #include <limits>
 #include <optional>
 #include <string>
+#include <tuple>
 
 namespace llvm {
 class AssumptionCache;
@@ -419,6 +420,14 @@ TargetTransformInfo::TargetCostKind getSLPCostKind(const Function *F);
 /// loses it.
 APInt getScalarMaxValue(const Value *V, unsigned Depth = 0);
 
+/// Checks if the values in \p VL are zero-extended sub-fields of the same
+/// wider integer scalar. Returns the source scalar, the field width and the
+/// field permutation mask. The extraction dual of the lane-packing layout.
+/// The field-to-lane mapping of the bitcast to the field vector is defined
+/// for little-endian targets only.
+std::optional<std::tuple<Value *, unsigned, SmallVector<int>>>
+matchGatheredExtractedFields(ArrayRef<Value *> VL, const DataLayout &DL);
+
 /// Description of a bitfield packing of vector lanes into a scalar value:
 /// every lane contributes a disjoint contiguous byte field of the result.
 struct BitPackInfo {
diff --git a/llvm/test/Transforms/PhaseOrdering/AArch64/scalarize-load-ext-extract.ll b/llvm/test/Transforms/PhaseOrdering/AArch64/scalarize-load-ext-extract.ll
index 8af7ae1b0ac77c..ba0974ad8d80c6 100644
--- a/llvm/test/Transforms/PhaseOrdering/AArch64/scalarize-load-ext-extract.ll
+++ b/llvm/test/Transforms/PhaseOrdering/AArch64/scalarize-load-ext-extract.ll
@@ -5,11 +5,8 @@ define noundef i32 @load_ext_extract(ptr %src) {
 ; CHECK-LABEL: define noundef range(i32 0, 1021) i32 @load_ext_extract(
 ; CHECK-SAME: ptr nofree readonly captures(none) [[SRC:%.*]]) local_unnamed_addr #[[ATTR0:[0-9]+]] {
 ; CHECK-NEXT:  [[ENTRY:.*:]]
-; CHECK-NEXT:    [[TMP14:%.*]] = load i32, ptr [[SRC]], align 4
-; CHECK-NEXT:    [[TMP0:%.*]] = insertelement <4 x i32> poison, i32 [[TMP14]], i64 0
-; CHECK-NEXT:    [[TMP1:%.*]] = shufflevector <4 x i32> [[TMP0]], <4 x i32> poison, <4 x i32> zeroinitializer
-; CHECK-NEXT:    [[TMP2:%.*]] = lshr <4 x i32> [[TMP1]], <i32 0, i32 8, i32 16, i32 24>
-; CHECK-NEXT:    [[TMP5:%.*]] = and <4 x i32> [[TMP2]], <i32 255, i32 255, i32 255, i32 -1>
+; CHECK-NEXT:    [[X12:%.*]] = load <4 x i8>, ptr [[SRC]], align 4
+; CHECK-NEXT:    [[TMP5:%.*]] = zext <4 x i8> [[X12]] to <4 x i32>
 ; CHECK-NEXT:    [[ADD3:%.*]] = tail call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> [[TMP5]])
 ; CHECK-NEXT:    ret i32 [[ADD3]]
 ;
diff --git a/llvm/test/Transforms/PhaseOrdering/X86/avg.ll b/llvm/test/Transforms/PhaseOrdering/X86/avg.ll
index 8f8899d285ad37..be56447572ab37 100644
--- a/llvm/test/Transforms/PhaseOrdering/X86/avg.ll
+++ b/llvm/test/Transforms/PhaseOrdering/X86/avg.ll
@@ -1,12 +1,12 @@
 ; NOTE: Assertions have been autogenerated by utils/update_test_checks.py
 ; RUN: opt < %s -O3 -S -mtriple=x86_64-- -mcpu=x86-64    | FileCheck %s --check-prefixes=CHECK,SSE2
 ; RUN: opt < %s -O3 -S -mtriple=x86_64-- -mcpu=x86-64-v2 | FileCheck %s --check-prefixes=CHECK,SSE4
-; RUN: opt < %s -O3 -S -mtriple=x86_64-- -mcpu=x86-64-v3 | FileCheck %s --check-prefixes=CHECK,AVX,AVX2
-; RUN: opt < %s -O3 -S -mtriple=x86_64-- -mcpu=x86-64-v4 | FileCheck %s --check-prefixes=CHECK,AVX,AVX512
+; RUN: opt < %s -O3 -S -mtriple=x86_64-- -mcpu=x86-64-v3 | FileCheck %s --check-prefixes=CHECK,AVX2
+; RUN: opt < %s -O3 -S -mtriple=x86_64-- -mcpu=x86-64-v4 | FileCheck %s --check-prefixes=CHECK,AVX512
 ; RUN: opt < %s -passes="default<O3>" -S -mtriple=x86_64-- -mcpu=x86-64    | FileCheck %s --check-prefixes=CHECK,SSE2
 ; RUN: opt < %s -passes="default<O3>" -S -mtriple=x86_64-- -mcpu=x86-64-v2 | FileCheck %s --check-prefixes=CHECK,SSE4
-; RUN: opt < %s -passes="default<O3>" -S -mtriple=x86_64-- -mcpu=x86-64-v3 | FileCheck %s --check-prefixes=CHECK,AVX,AVX2
-; RUN: opt < %s -passes="default<O3>" -S -mtriple=x86_64-- -mcpu=x86-64-v4 | FileCheck %s --check-prefixes=CHECK,AVX,AVX512
+; RUN: opt < %s -passes="default<O3>" -S -mtriple=x86_64-- -mcpu=x86-64-v3 | FileCheck %s --check-prefixes=CHECK,AVX2
+; RUN: opt < %s -passes="default<O3>" -S -mtriple=x86_64-- -mcpu=x86-64-v4 | FileCheck %s --check-prefixes=CHECK,AVX512
 
 ; PR128424
 
@@ -95,121 +95,80 @@ define { i64, i64 } @avgr_16_u8(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerce0,
 ;
 ; SSE4-LABEL: @avgr_16_u8(
 ; SSE4-NEXT:  entry:
-; SSE4-NEXT:    [[TMP0:%.*]] = trunc i64 [[A_COERCE0:%.*]] to i16
-; SSE4-NEXT:    [[TMP1:%.*]] = lshr i16 [[TMP0]], 8
-; SSE4-NEXT:    [[A_SROA_3_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE0]], 16
-; SSE4-NEXT:    [[A_SROA_4_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE0]], 24
-; SSE4-NEXT:    [[A_SROA_5_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE0]], 32
-; SSE4-NEXT:    [[A_SROA_6_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE0]], 40
-; SSE4-NEXT:    [[A_SROA_7_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE0]], 48
-; SSE4-NEXT:    [[A_SROA_8_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE0]], 56
-; SSE4-NEXT:    [[TMP2:%.*]] = trunc i64 [[A_COERCE1:%.*]] to i16
-; SSE4-NEXT:    [[TMP3:%.*]] = lshr i16 [[TMP2]], 8
-; SSE4-NEXT:    [[A_SROA_12_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE1]], 16
-; SSE4-NEXT:    [[A_SROA_13_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE1]], 24
-; SSE4-NEXT:    [[A_SROA_14_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE1]], 32
-; SSE4-NEXT:    [[A_SROA_15_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE1]], 40
-; SSE4-NEXT:    [[A_SROA_16_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE1]], 48
-; SSE4-NEXT:    [[A_SROA_17_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE1]], 56
-; SSE4-NEXT:    [[TMP4:%.*]] = trunc i64 [[B_COERCE0:%.*]] to i16
-; SSE4-NEXT:    [[TMP5:%.*]] = lshr i16 [[TMP4]], 8
-; SSE4-NEXT:    [[B_SROA_3_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE0]], 16
-; SSE4-NEXT:    [[B_SROA_4_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE0]], 24
-; SSE4-NEXT:    [[B_SROA_5_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE0]], 32
-; SSE4-NEXT:    [[B_SROA_6_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE0]], 40
-; SSE4-NEXT:    [[B_SROA_7_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE0]], 48
-; SSE4-NEXT:    [[B_SROA_8_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE0]], 56
-; SSE4-NEXT:    [[TMP6:%.*]] = trunc i64 [[B_COERCE1:%.*]] to i16
-; SSE4-NEXT:    [[TMP7:%.*]] = lshr i16 [[TMP6]], 8
-; SSE4-NEXT:    [[B_SROA_12_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE1]], 16
-; SSE4-NEXT:    [[B_SROA_13_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE1]], 24
-; SSE4-NEXT:    [[B_SROA_14_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE1]], 32
-; SSE4-NEXT:    [[B_SROA_15_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE1]], 40
-; SSE4-NEXT:    [[B_SROA_16_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE1]], 48
-; SSE4-NEXT:    [[B_SROA_17_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE1]], 56
-; SSE4-NEXT:    [[CONV1:%.*]] = and i64 [[A_COERCE0]], 255
-; SSE4-NEXT:    [[CONV4:%.*]] = and i64 [[B_COERCE0]], 255
-; SSE4-NEXT:    [[ADD:%.*]] = add nuw nsw i64 [[CONV1]], 1
-; SSE4-NEXT:    [[ADD5:%.*]] = add nuw nsw i64 [[ADD]], [[CONV4]]
-; SSE4-NEXT:    [[SHR:%.*]] = lshr i64 [[ADD5]], 1
-; SSE4-NEXT:    [[ADD_1:%.*]] = add nuw nsw i16 [[TMP1]], 1
-; SSE4-NEXT:    [[ADD5_1:%.*]] = add nuw nsw i16 [[ADD_1]], [[TMP5]]
-; SSE4-NEXT:    [[CONV1_6:%.*]] = and i64 [[A_SROA_7_0_EXTRACT_SHIFT]], 255
-; SSE4-NEXT:    [[CONV4_6:%.*]] = and i64 [[B_SROA_7_0_EXTRACT_SHIFT]], 255
-; SSE4-NEXT:    [[ADD_6:%.*]] = add nuw nsw i64 [[CONV1_6]], 1
-; SSE4-NEXT:    [[ADD5_6:%.*]] = add nuw nsw i64 [[ADD_6]], [[CONV4_6]]
-; SSE4-NEXT:    [[ADD_7:%.*]] = add nuw nsw i64 [[A_SROA_8_0_EXTRACT_SHIFT]], 1
-; SSE4-NEXT:    [[ADD5_7:%.*]] = add nuw nsw i64 [[ADD_7]], [[B_SROA_8_0_EXTRACT_SHIFT]]
-; SSE4-NEXT:    [[CONV1_8:%.*]] = and i64 [[A_COERCE1]], 255
-; SSE4-NEXT:    [[CONV4_8:%.*]] = and i64 [[B_COERCE1]], 255
-; SSE4-NEXT:    [[ADD_8:%.*]] = add nuw nsw i64 [[CONV1_8]], 1
-; SSE4-NEXT:    [[ADD5_8:%.*]] = add nuw nsw i64 [[ADD_8]], [[CONV4_8]]
-; SSE4-NEXT:    [[SHR_8:%.*]] = lshr i64 [[ADD5_8]], 1
-; SSE4-NEXT:    [[ADD_9:%.*]] = add nuw nsw i16 [[TMP3]], 1
-; SSE4-NEXT:    [[ADD5_9:%.*]] = add nuw nsw i16 [[ADD_9]], [[TMP7]]
-; SSE4-NEXT:    [[CONV1_14:%.*]] = and i64 [[A_SROA_16_8_EXTRACT_SHIFT]], 255
-; SSE4-NEXT:    [[CONV4_14:%.*]] = and i64 [[B_SROA_16_8_EXTRACT_SHIFT]], 255
-; SSE4-NEXT:    [[ADD_14:%.*]] = add nuw nsw i64 [[CONV1_14]], 1
-; SSE4-NEXT:    [[ADD5_14:%.*]] = add nuw nsw i64 [[ADD_14]], [[CONV4_14]]
-; SSE4-NEXT:    [[ADD_15:%.*]] = add nuw nsw i64 [[A_SROA_17_8_EXTRACT_SHIFT]], 1
-; SSE4-NEXT:    [[ADD5_15:%.*]] = add nuw nsw i64 [[ADD_15]], [[B_SROA_17_8_EXTRACT_SHIFT]]
-; SSE4-NEXT:    [[TMP8:%.*]] = shl nuw i64 [[ADD5_7]], 55
-; SSE4-NEXT:    [[RETVAL_SROA_8_0_INSERT_EXT:%.*]] = and i64 [[TMP8]], -72057594037927936
-; SSE4-NEXT:    [[TMP9:%.*]] = shl nuw nsw i64 [[ADD5_6]], 47
-; SSE4-NEXT:    [[RETVAL_SROA_7_0_INSERT_SHIFT:%.*]] = and i64 [[TMP9]], 71776119061217280
-; SSE4-NEXT:    [[TMP10:%.*]] = insertelement <4 x i64> poison, i64 [[A_SROA_6_0_EXTRACT_SHIFT]], i64 0
-; SSE4-NEXT:    [[TMP11:%.*]] = insertelement <4 x i64> [[TMP10]], i64 [[A_SROA_5_0_EXTRACT_SHIFT]], i64 1
-; SSE4-NEXT:    [[TMP12:%.*]] = insertelement <4 x i64> [[TMP11]], i64 [[A_SROA_4_0_EXTRACT_SHIFT]], i64 2
-; SSE4-NEXT:    [[TMP13:%.*]] = insertelement <4 x i64> [[TMP12]], i64 [[A_SROA_3_0_EXTRACT_SHIFT]], i64 3
-; SSE4-NEXT:    [[TMP14:%.*]] = and <4 x i64> [[TMP13]], splat (i64 255)
-; SSE4-NEXT:    [[TMP15:%.*]] = insertelement <4 x i64> poison, i64 [[B_SROA_6_0_EXTRACT_SHIFT]], i64 0
-; SSE4-NEXT:    [[TMP16:%.*]] = insertelement <4 x i64> [[TMP15]], i64 [[B_SROA_5_0_EXTRACT_SHIFT]], i64 1
-; SSE4-NEXT:    [[TMP17:%.*]] = insertelement <4 x i64> [[TMP16]], i64 [[B_SROA_4_0_EXTRACT_SHIFT]], i64 2
-; SSE4-NEXT:    [[TMP18:%.*]] = insertelement <4 x i64> [[TMP17]], i64 [[B_SROA_3_0_EXTRACT_SHIFT]], i64 3
-; SSE4-NEXT:    [[TMP19:%.*]] = and <4 x i64> [[TMP18]], splat (i64 255)
-; SSE4-NEXT:    [[TMP20:%.*]] = add nuw nsw <4 x i64> [[TMP14]], splat (i64 1)
-; SSE4-NEXT:    [[TMP21:%.*]] = add nuw nsw <4 x i64> [[TMP20]], [[TMP19]]
-; SSE4-NEXT:    [[TMP22:%.*]] = trunc nuw nsw <4 x i64> [[TMP21]] to <4 x i32>
-; SSE4-NEXT:    [[TMP23:%.*]] = lshr <4 x i32> [[TMP22]], splat (i32 1)
-; SSE4-NEXT:    [[TMP24:%.*]] = bitcast <4 x i32> [[TMP23]] to <16 x i8>
-; SSE4-NEXT:    [[TMP25:%.*]] = shufflevector <16 x i8> [[TMP24]], <16 x i8> <i8 0, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison>, <8 x i32> <i32 16, i32 16, i32 12, i32 8, i32 4, i32 0, i32 16, i32 16>
-; SSE4-NEXT:    [[TMP26:%.*]] = bitcast <8 x i8> [[TMP25]] to i64
-; SSE4-NEXT:    [[TMP27:%.*]] = shl nuw i16 [[ADD5_1]], 7
-; SSE4-NEXT:    [[TMP28:%.*]] = and i16 [[TMP27]], -256
-; SSE4-NEXT:    [[RETVAL_SROA_2_0_INSERT_SHIFT_MASKED:%.*]] = zext i16 [[TMP28]] to i64
-; SSE4-NEXT:    [[OP_RDX5:%.*]] = or disjoint i64 [[RETVAL_SROA_8_0_INSERT_EXT]], [[TMP26]]
-; SSE4-NEXT:    [[OP_RDX6:%.*]] = or disjoint i64 [[RETVAL_SROA_7_0_INSERT_SHIFT]], [[SHR]]
-; SSE4-NEXT:    [[OP_RDX7:%.*]] = or disjoint i64 [[OP_RDX5]], [[OP_RDX6]]
-; SSE4-NEXT:    [[TMP68:%.*]] = or i64 [[OP_RDX7]], [[RETVAL_SROA_2_0_INSERT_SHIFT_MASKED]]
+; SSE4-NEXT:    [[TMP0:%.*]] = insertelement <2 x i64> poison, i64 [[A_COERCE0:%.*]], i64 0
+; SSE4-NEXT:    [[TMP1:%.*]] = insertelement <2 x i64> [[TMP0]], i64 [[A_COERCE1:%.*]], i64 1
+; SSE4-NEXT:    [[TMP2:%.*]] = trunc <2 x i64> [[TMP1]] to <2 x i16>
+; SSE4-NEXT:    [[TMP3:%.*]] = lshr <2 x i64> [[TMP1]], splat (i64 16)
+; SSE4-NEXT:    [[TMP4:%.*]] = lshr <2 x i64> [[TMP1]], splat (i64 24)
+; SSE4-NEXT:    [[TMP5:%.*]] = lshr <2 x i64> [[TMP1]], splat (i64 32)
+; SSE4-NEXT:    [[TMP6:%.*]] = lshr <2 x i64> [[TMP1]], splat (i64 40)
+; SSE4-NEXT:    [[TMP7:%.*]] = lshr <2 x i64> [[TMP1]], splat (i64 48)
+; SSE4-NEXT:    [[TMP8:%.*]] = lshr <2 x i64> [[TMP1]], splat (i64 56)
+; SSE4-NEXT:    [[TMP9:%.*]] = insertelement <2 x i64> poison, i64 [[B_COERCE0:%.*]], i64 0
+; SSE4-NEXT:    [[TMP10:%.*]] = insertelement <2 x i64> [[TMP9]], i64 [[B_COERCE1:%.*]], i64 1
+; SSE4-NEXT:    [[TMP11:%.*]] = trunc <2 x i64> [[TMP10]] to <2 x i16>
+; SSE4-NEXT:    [[TMP12:%.*]] = lshr <2 x i64> [[TMP10]], splat (i64 16)
+; SSE4-NEXT:    [[TMP13:%.*]] = lshr <2 x i64> [[TMP10]], splat (i64 24)
+; SSE4-NEXT:    [[TMP14:%.*]] = lshr <2 x i64> [[TMP10]], splat (i64 32)
+; SSE4-NEXT:    [[TMP15:%.*]] = lshr <2 x i64> [[TMP10]], splat (i64 40)
+; SSE4-NEXT:    [[TMP16:%.*]] = lshr <2 x i64> [[TMP10]], splat (i64 48)
+; SSE4-NEXT:    [[TMP17:%.*]] = lshr <2 x i64> [[TMP10]], splat (i64 56)
+; SSE4-NEXT:    [[TMP18:%.*]] = and <2 x i64> [[TMP1]], splat (i64 255)
+; SSE4-NEXT:    [[TMP19:%.*]] = and <2 x i64> [[TMP10]], splat (i64 255)
+; SSE4-NEXT:    [[TMP20:%.*]] = and <2 x i64> [[TMP7]], splat (i64 255)
+; SSE4-NEXT:    [[TMP21:%.*]] = lshr <2 x i16> [[TMP2]], splat (i16 8)
+; SSE4-NEXT:    [[TMP22:%.*]] = lshr <2 x i16> [[TMP11]], splat (i16 8)
+; SSE4-NEXT:    [[TMP23:%.*]] = add nuw nsw <2 x i64> [[TMP18]], splat (i64 1)
+; SSE4-NEXT:    [[TMP24:%.*]] = add nuw nsw <2 x i64> [[TMP23]], [[TMP19]]
+; SSE4-NEXT:    [[TMP25:%.*]] = lshr <2 x i64> [[TMP24]], splat (i64 1)
+; SSE4-NEXT:    [[TMP26:%.*]] = add nuw nsw <2 x i16> [[TMP21]], splat (i16 1)
+; SSE4-NEXT:    [[TMP27:%.*]] = add nuw nsw <2 x i16> [[TMP26]], [[TMP22]]
+; SSE4-NEXT:    [[TMP28:%.*]] = and <2 x i64> [[TMP3]], splat (i64 255)
+; SSE4-NEXT:    [[TMP29:%.*]] = and <2 x i64> [[TMP12]], splat (i64 255)
+; SSE4-NEXT:    [[TMP30:%.*]] = add nuw nsw <2 x i64> [[TMP28]], splat (i64 1)
+; SSE4-NEXT:    [[TMP31:%.*]] = add nuw nsw <2 x i64> [[TMP30]], [[TMP29]]
+; SSE4-NEXT:    [[TMP32:%.*]] = and <2 x i64> [[TMP4]], splat (i64 255)
+; SSE4-NEXT:    [[TMP33:%.*]] = and <2 x i64> [[TMP13]], splat (i64 255)
+; SSE4-NEXT:    [[TMP34:%.*]] = add nuw nsw <2 x i64> [[TMP32]], splat (i64 1)
+; SSE4-NEXT:    [[TMP35:%.*]] = add nuw nsw <2 x i64> [[TMP34]], [[TMP33]]
+; SSE4-NEXT:    [[TMP36:%.*]] = and <2 x i64> [[TMP5]], splat (i64 255)
+; SSE4-NEXT:    [[TMP37:%.*]] = and <2 x i64> [[TMP14]], splat (i64 255)
+; SSE4-NEXT:    [[TMP38:%.*]] = add nuw nsw <2 x i64> [[TMP36]], splat (i64 1)
+; SSE4-NEXT:    [[TMP39:%.*]] = add nuw nsw <2 x i64> [[TMP38]], [[TMP37]]
+; SSE4-NEXT:    [[TMP40:%.*]] = and <2 x i64> [[TMP6]], splat (i64 255)
+; SSE4-NEXT:    [[TMP41:%.*]] = and <2 x i64> [[TMP15]], splat (i64 255)
+; SSE4-NEXT:    [[TMP42:%.*]] = add nuw nsw <2 x i64> [[TMP40]], splat (i64 1)
+; SSE4-NEXT:    [[TMP43:%.*]] = add nuw nsw <2 x i64> [[TMP42]], [[TMP41]]
+; SSE4-NEXT:    [[TMP44:%.*]] = and <2 x i64> [[TMP16]], splat (i64 255)
+; SSE4-NEXT:    [[TMP45:%.*]] = add nuw nsw <2 x i64> [[TMP20]], splat (i64 1)
+; SSE4-NEXT:    [[TMP46:%.*]] = add nuw nsw <2 x i64> [[TMP45]], [[TMP44]]
+; SSE4-NEXT:    [[TMP47:%.*]] = add nuw nsw <2 x i64> [[TMP8]], splat (i64 1)
+; SSE4-NEXT:    [[TMP48:%.*]] = add nuw nsw <2 x i64> [[TMP47]], [[TMP17]]
+; SSE4-NEXT:    [[TMP49:%.*]] = shl nuw <2 x i64> [[TMP48]], splat (i64 55)
+; SSE4-NEXT:    [[TMP50:%.*]] = and <2 x i64> [[TMP49]], splat (i64 -72057594037927936)
+; SSE4-NEXT:    [[TMP51:%.*]] = shl nuw nsw <2 x i64> [[TMP46]], splat (i64 47)
+; SSE4-NEXT:    [[TMP52:%.*]] = and <2 x i64> [[TMP51]], splat (i64 71776119061217280)
+; SSE4-NEXT:    [[TMP53:%.*]] = or disjoint <2 x i64> [[TMP50]], [[TMP52]]
+; SSE4-NEXT:    [[TMP54:%.*]] = shl nuw nsw <2 x i64> [[TMP43]], splat (i64 39)
+; SSE4-NEXT:    [[TMP55:%.*]] = and <2 x i64> [[TMP54]], splat (i64 280375465082880)
+; SSE4-NEXT:    [[TMP56:%.*]] = or disjoint <2 x i64> [[TMP53]], [[TMP55]]
+; SSE4-NEXT:    [[TMP57:%.*]] = shl nuw nsw <2 x i64> [[TMP39]], splat (i64 31)
+; SSE4-NEXT:    [[TMP58:%.*]] = and <2 x i64> [[TMP57]], splat (i64 1095216660480)
+; SSE4-NEXT:    [[TMP59:%.*]] = or disjoint <2 x i64> [[TMP56]], [[TMP58]]
+; SSE4-NEXT:    [[TMP60:%.*]] = shl nuw nsw <2 x i64> [[TMP35]], splat (i64 23)
+; SSE4-NEXT:    [[TMP61:%.*]] = and <2 x i64> [[TMP60]], splat (i64 4278190080)
+; SSE4-NEXT:    [[TMP62:%.*]] = or disjoint <2 x i64> [[TMP59]], [[TMP61]]
+; SSE4-NEXT:    [[TMP63:%.*]] = shl nuw nsw <2 x i64> [[TMP31]], splat (i64 15)
+; SSE4-NEXT:    [[TMP64:%.*]] = and <2 x i64> [[TMP63]], splat (i64 16711680)
+; SSE4-NEXT:    [[TMP65:%.*]] = shl nuw <2 x i16> [[TMP27]], splat (i16 7)
+; SSE4-NEXT:    [[TMP66:%.*]] = or disjoint <2 x i64> [[TMP62]], [[TMP64]]
+; SSE4-NEXT:    [[TMP67:%.*]] = and <2 x i16> [[TMP65]], splat (i16 -256)
+; SSE4-NEXT:    [[TMP71:%.*]] = zext <2 x i16> [[TMP67]] to <2 x i64>
+; SSE4-NEXT:    [[TMP72:%.*]] = or <2 x i64> [[TMP66]], [[TMP71]]
+; SSE4-NEXT:    [[TMP70:%.*]] = or <2 x i64> [[TMP72]], [[TMP25]]
+; SSE4-NEXT:    [[TMP68:%.*]] = extractelement <2 x i64> [[TMP70]], i64 0
 ; SSE4-NEXT:    [[DOTFCA_0_INSERT:%.*]] = insertvalue { i64, i64 } poison, i64 [[TMP68]], 0
-; SSE4-NEXT:    [[TMP29:%.*]] = shl nuw i64 [[ADD5_15]], 55
-; SSE4-NEXT:    [[RETVAL_SROA_17_8_INSERT_EXT:%.*]] = and i64 [[TMP29]], -72057594037927936
-; SSE4-NEXT:    [[TMP30:%.*]] = shl nuw nsw i64 [[ADD5_14]], 47
-; SSE4-NEXT:    [[RETVAL_SROA_16_8_INSERT_SHIFT:%.*]] = and i64 [[TMP30]], 71776119061217280
-; SSE4-NEXT:    [[TMP31:%.*]] = insertelement <4 x i64> poison, i64 [[A_SROA_15_8_EXTRACT_SHIFT]], i64 0
-; SSE4-NEXT:    [[TMP32:%.*]] = insertelement <4 x i64> [[TMP31]], i64 [[A_SROA_14_8_EXTRACT_SHIFT]], i64 1
-; SSE4-NEXT:    [[TMP33:%.*]] = insertelement <4 x i64> [[TMP32]], i64 [[A_SROA_13_8_EXTRACT_SHIFT]], i64 2
-; SSE4-NEXT:    [[TMP34:%.*]] = insertelement <4 x i64> [[TMP33]], i64 [[A_SROA_12_8_EXTRACT_SHIFT]], i64 3
-; SSE4-NEXT:    [[TMP35:%.*]] = and <4 x i64> [[TMP34]], splat (i64 255)
-; SSE4-NEXT:    [[TMP36:%.*]] = insertelement <4 x i64> poison, i64 [[B_SROA_15_8_EXTRACT_SHIFT]], i64 0
-; SSE4-NEXT:    [[TMP37:%.*]] = insertelement <4 x i64> [[TMP36]], i64 [[B_SROA_14_8_EXTRACT_SHIFT]], i64 1
-; SSE4-NEXT:    [[TMP38:%.*]] = insertelement <4 x i64> [[TMP37]], i64 [[B_SROA_13_8_EXTRACT_SHIFT]], i64 2
-; SSE4-NEXT:    [[TMP39:%.*]] = insertelement <4 x i64> [[TMP38]], i64 [[B_SROA_12_8_EXTRACT_SHIFT]], i64 3
-; SSE4-NEXT:    [[TMP40:%.*]] = and <4 x i64> [[TMP39]], splat (i64 255)
-; SSE4-NEXT:    [[TMP41:%.*]] = add nuw nsw <4 x i64> [[TMP35]], splat (i64 1)
-; SSE4-NEXT:    [[TMP42:%.*]] = add nuw nsw <4 x i64> [[TMP41]], [[TMP40]]
-; SSE4-NEXT:    [[TMP43:%.*]] = trunc nuw nsw <4 x i64> [[TMP42]] to <4 x i32>
-; SSE4-NEXT:    [[TMP44:%.*]] = lshr <4 x i32> [[TMP43]], splat (i32 1)
-; SSE4-NEXT:    [[TMP45:%.*]] = bitcast <4 x i32> [[TMP44]] to <16 x i8>
-; SSE4-NEXT:    [[TMP46:%.*]] = shufflevector <16 x i8> [[TMP45]], <16 x i8> <i8 0, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison>, <8 x i32> <i32 16, i32 16, i32 12, i32 8, i32 4, i32 0, i32 16, i32 16>
-; SSE4-NEXT:    [[TMP47:%.*]] = bitcast <8 x i8> [[TMP46]] to i64
-; SSE4-NEXT:    [[TMP48:%.*]] = shl nuw i16 [[ADD5_9]], 7
-; SSE4-NEXT:    [[TMP49:%.*]] = and i16 [[TMP48]], -256
-; SSE4-NEXT:    [[RETVAL_SROA_11_8_INSERT_SHIFT_MASKED:%.*]] = zext i16 [[TMP49]] to i64
-; SSE4-NEXT:    [[OP_RDX:%.*]] = or disjoint i64 [[RETVAL_SROA_17_8_INSERT_EXT]], [[TMP47]]
-; SSE4-NEXT:    [[OP_RDX2:%.*]] = or disjoint i64 [[RETVAL_SROA_16_8_INSERT_SHIFT]], [[SHR_8]]
-; SSE4-NEXT:    [[OP_RDX3:%.*]] = or disjoint i64 [[OP_RDX]], [[OP_RDX2]]
-; SSE4-NEXT:    [[TMP69:%.*]] = or i64 [[OP_RDX3]], [[RETVAL_SROA_11_8_INSERT_SHIFT_MASKED]]
+; SSE4-NEXT:    [[TMP69:%.*]] = extractelement <2 x i64> [[TMP70]], i64 1
 ; SSE4-NEXT:    [[DOTFCA_1_INSERT:%.*]] = insertvalue { i64, i64 } [[DOTFCA_0_INSERT]], i64 [[TMP69]], 1
 ; SSE4-NEXT:    ret { i64, i64 } [[DOTFCA_1_INSERT]]
 ;
@@ -380,283 +339,29 @@ for.body:                                         ; preds = %for.cond
 }
 
 define { i64, i64 } @avgr_16_u8_alt(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerce0, i64 %b.coerce1) {
-; SSE2-LABEL: @avgr_16_u8_alt(
-; SSE2-NEXT:  entry:
-; SSE2-NEXT:    [[A_SROA_0_0_EXTRACT_TRUNC:%.*]] = trunc i64 [[A_COERCE0:%.*]] to i8
-; SSE2-NEXT:    [[A_SROA_2_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE0]], 8
-; SSE2-NEXT:    [[A_SROA_2_0_EXTRACT_TRUNC:%.*]] = trunc i64 [[A_SROA_2_0_EXTRACT_SHIFT]] to i8
-; SSE2-NEXT:    [[A_SROA_3_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE0]], 16
-; SSE2-NEXT:    [[A_SROA_3_0_EXTRACT_TRUNC:%.*]] = trunc i64 [[A_SROA_3_0_EXTRACT_SHIFT]] to i8
-; SSE2-NEXT:    [[A_SROA_4_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE0]], 24
-; SSE2-NEXT:    [[A_SROA_4_0_EXTRACT_TRUNC:%.*]] = trunc i64 [[A_SROA_4_0_EXTRACT_SHIFT]] to i8
-; SSE2-NEXT:    [[A_SROA_5_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE0]], 32
-; SSE2-NEXT:    [[A_SROA_5_0_EXTRACT_TRUNC:%.*]] = trunc i64 [[A_SROA_5_0_EXTRACT_SHIFT]] to i8
-; SSE2-NEXT:    [[A_SROA_6_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE0]], 40
-; SSE2-NEXT:    [[A_SROA_6_0_EXTRACT_TRUNC:%.*]] = trunc i64 [[A_SROA_6_0_EXTRACT_SHIFT]] to i8
-; SSE2-NEXT:    [[A_SROA_7_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE0]], 48
-; SSE2-NEXT:    [[A_SROA_7_0_EXTRACT_TRUNC:%.*]] = trunc i64 [[A_SROA_7_0_EXTRACT_SHIFT]] to i8
-; SSE2-NEXT:    [[A_SROA_8_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE0]], 56
-; SSE2-NEXT:    [[A_SROA_8_0_EXTRACT_TRUNC:%.*]] = trunc nuw i64 [[A_SROA_8_0_EXTRACT_SHIFT]] to i8
-; SSE2-NEXT:    [[A_SROA_9_8_EXTRACT_TRUNC:%.*]] = trunc i64 [[A_COERCE1:%.*]] to i8
-; SSE2-NEXT:    [[A_SROA_11_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE1]], 8
-; SSE2-NEXT:    [[A_SROA_11_8_EXTRACT_TRUNC:%.*]] = trunc i64 [[A_SROA_11_8_EXTRACT_SHIFT]] to i8
-; SSE2-NEXT:    [[A_SROA_12_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE1]], 16
-; SSE2-NEXT:    [[A_SROA_12_8_EXTRACT_TRUNC:%.*]] = trunc i64 [[A_SROA_12_8_EXTRACT_SHIFT]] to i8
-; SSE2-NEXT:    [[A_SROA_13_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE1]], 24
-; SSE2-NEXT:    [[A_SROA_13_8_EXTRACT_TRUNC:%.*]] = trunc i64 [[A_SROA_13_8_EXTRACT_SHIFT]] to i8
-; SSE2-NEXT:    [[A_SROA_14_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE1]], 32
-; SSE2-NEXT:    [[A_SROA_14_8_EXTRACT_TRUNC:%.*]] = trunc i64 [[A_SROA_14_8_EXTRACT_SHIFT]] to i8
-; SSE2-NEXT:    [[A_SROA_15_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE1]], 40
-; SSE2-NEXT:    [[A_SROA_15_8_EXTRACT_TRUNC:%.*]] = trunc i64 [[A_SROA_15_8_EXTRACT_SHIFT]] to i8
-; SSE2-NEXT:    [[A_SROA_16_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE1]], 48
-; SSE2-NEXT:    [[A_SROA_16_8_EXTRACT_TRUNC:%.*]] = trunc i64 [[A_SROA_16_8_EXTRACT_SHIFT]] to i8
-; SSE2-NEXT:    [[A_SROA_17_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE1]], 56
-; SSE2-NEXT:    [[A_SROA_17_8_EXTRACT_TRUNC:%.*]] = trunc nuw i64 [[A_SROA_17_8_EXTRACT_SHIFT]] to i8
-; SSE2-NEXT:    [[B_SROA_0_0_EXTRACT_TRUNC:%.*]] = trunc i64 [[B_COERCE0:%.*]] to i8
-; SSE2-NEXT:    [[B_SROA_2_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE0]], 8
-; SSE2-NEXT:    [[B_SROA_2_0_EXTRACT_TRUNC:%.*]] = trunc i64 [[B_SROA_2_0_EXTRACT_SHIFT]] to i8
-; SSE2-NEXT:    [[B_SROA_3_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE0]], 16
-; SSE2-NEXT:    [[B_SROA_3_0_EXTRACT_TRUNC:%.*]] = trunc i64 [[B_SROA_3_0_EXTRACT_SHIFT]] to i8
-; SSE2-NEXT:    [[B_SROA_4_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE0]], 24
-; SSE2-NEXT:    [[B_SROA_4_0_EXTRACT_TRUNC:%.*]] = trunc i64 [[B_SROA_4_0_EXTRACT_SHIFT]] to i8
-; SSE2-NEXT:    [[B_SROA_5_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE0]], 32
-; SSE2-NEXT:    [[B_SROA_5_0_EXTRACT_TRUNC:%.*]] = trunc i64 [[B_SROA_5_0_EXTRACT_SHIFT]] to i8
-; SSE2-NEXT:    [[B_SROA_6_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE0]], 40
-; SSE2-NEXT:    [[B_SROA_6_0_EXTRACT_TRUNC:%.*]] = trunc i64 [[B_SROA_6_0_EXTRACT_SHIFT]] to i8
-; SSE2-NEXT:    [[B_SROA_7_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE0]], 48
-; SSE2-NEXT:    [[B_SROA_7_0_EXTRACT_TRUNC:%.*]] = trunc i64 [[B_SROA_7_0_EXTRACT_SHIFT]] to i8
-; SSE2-NEXT:    [[B_SROA_8_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE0]], 56
-; SSE2-NEXT:    [[B_SROA_8_0_EXTRACT_TRUNC:%.*]] = trunc nuw i64 [[B_SROA_8_0_EXTRACT_SHIFT]] to i8
-; SSE2-NEXT:    [[B_SROA_9_8_EXTRACT_TRUNC:%.*]] = trunc i64 [[B_COERCE1:%.*]] to i8
-; SSE2-NEXT:    [[B_SROA_11_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE1]], 8
-; SSE2-NEXT:    [[B_SROA_11_8_EXTRACT_TRUNC:%.*]] = trunc i64 [[B_SROA_11_8_EXTRACT_SHIFT]] to i8
-; SSE2-NEXT:    [[B_SROA_12_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE1]], 16
-; SSE2-NEXT:    [[B_SROA_12_8_EXTRACT_TRUNC:%.*]] = trunc i64 [[B_SROA_12_8_EXTRACT_SHIFT]] to i8
-; SSE2-NEXT:    [[B_SROA_13_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE1]], 24
-; SSE2-NEXT:    [[B_SROA_13_8_EXTRACT_TRUNC:%.*]] = trunc i64 [[B_SROA_13_8_EXTRACT_SHIFT]] to i8
-; SSE2-NEXT:    [[B_SROA_14_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE1]], 32
-; SSE2-NEXT:    [[B_SROA_14_8_EXTRACT_TRUNC:%.*]] = trunc i64 [[B_SROA_14_8_EXTRACT_SHIFT]] to i8
-; SSE2-NEXT:    [[B_SROA_15_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE1]], 40
-; SSE2-NEXT:    [[B_SROA_15_8_EXTRACT_TRUNC:%.*]] = trunc i64 [[B_SROA_15_8_EXTRACT_SHIFT]] to i8
-; SSE2-NEXT:    [[B_SROA_16_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE1]], 48
-; SSE2-NEXT:    [[B_SROA_16_8_EXTRACT_TRUNC:%.*]] = trunc i64 [[B_SROA_16_8_EXTRACT_SHIFT]] to i8
-; SSE2-NEXT:    [[B_SROA_17_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE1]], 56
-; SSE2-NEXT:    [[B_SROA_17_8_EXTRACT_TRUNC:%.*]] = trunc nuw i64 [[B_SROA_17_8_EXTRACT_SHIFT]] to i8
-; SSE2-NEXT:    [[SHR:%.*]] = lshr i8 [[A_SROA_0_0_EXTRACT_TRUNC]], 1
-; SSE2-NEXT:    [[SHR5:%.*]] = lshr i8 [[B_SROA_0_0_EXTRACT_TRUNC]], 1
-; SSE2-NEXT:    [[NARROW:%.*]] = add nuw i8 [[SHR5]], [[SHR]]
-; SSE2-NEXT:    [[OR21:%.*]] = or i8 [[B_SROA_0_0_EXTRACT_TRUNC]], [[A_SROA_0_0_EXTRACT_TRUNC]]
-; SSE2-NEXT:    [[TMP0:%.*]] = and i8 [[OR21]], 1
-; SSE2-NEXT:    [[ADD12:%.*]] = add nuw i8 [[NARROW]], [[TMP0]]
-; SSE2-NEXT:    [[SHR_1:%.*]] = lshr i8 [[A_SROA_2_0_EXTRACT_TRUNC]], 1
-; SSE2-NEXT:    [[SHR5_1:%.*]] = lshr i8 [[B_SROA_2_0_EXTRACT_TRUNC]], 1
-; SSE2-NEXT:    [[NARROW_1:%.*]] = add nuw i8 [[SHR5_1]], [[SHR_1]]
-; SSE2-NEXT:    [[OR21_1:%.*]] = or i8 [[B_SROA_2_0_EXTRACT_TRUNC]], [[A_SROA_2_0_EXTRACT_TRUNC]]
-; SSE2-NEXT:    [[TMP1:%.*]] = and i8 [[OR21_1]], 1
-; SSE2-NEXT:    [[ADD12_1:%.*]] = add nuw i8 [[NARROW_1]], [[TMP1]]
-; SSE2-NEXT:    [[SHR_2:%.*]] = lshr i8 [[A_SROA_3_0_EXTRACT_TRUNC]], 1
-; SSE2-NEXT:    [[SHR5_2:%.*]] = lshr i8 [[B_SROA_3_0_EXTRACT_TRUNC]], 1
-; SSE2-NEXT:    [[NARROW_2:%.*]] = add nuw i8 [[SHR5_2]], [[SHR_2]]
-; SSE2-NEXT:    [[OR21_2:%.*]] = or i8 [[B_SROA_3_0_EXTRACT_TRUNC]], [[A_SROA_3_0_EXTRACT_TRUNC]]
-; SSE2-NEXT:    [[TMP2:%.*]] = and i8 [[OR21_2]], 1
-; SSE2-NEXT:    [[ADD12_2:%.*]] = add nuw i8 [[NARROW_2]], [[TMP2]]
-; SSE2-NEXT:    [[SHR_3:%.*]] = lshr i8 [[A_SROA_4_0_EXTRACT_TRUNC]], 1
-; SSE2-NEXT:    [[SHR5_3:%.*]] = lshr i8 [[B_SROA_4_0_EXTRACT_TRUNC]], 1
-; SSE2-NEXT:    [[NARROW_3:%.*]] = add nuw i8 [[SHR5_3]], [[SHR_3]]
-; SSE2-NEXT:    [[OR21_3:%.*]] = or i8 [[B_SROA_4_0_EXTRACT_TRUNC]], [[A_SROA_4_0_EXTRACT_TRUNC]]
-; SSE2-NEXT:    [[TMP3:%.*]] = and i8 [[OR21_3]], 1
-; SSE2-NEXT:    [[ADD12_3:%.*]] = add nuw i8 [[NARROW_3]], [[TMP3]]
-; SSE2-NEXT:    [[SHR_4:%.*]] = lshr i8 [[A_SROA_5_0_EXTRACT_TRUNC]], 1
-; SSE2-NEXT:    [[SHR5_4:%.*]] = lshr i8 [[B_SROA_5_0_EXTRACT_TRUNC]], 1
-; SSE2-NEXT:    [[NARROW_4:%.*]] = add nuw i8 [[SHR5_4]], [[SHR_4]]
-; SSE2-NEXT:    [[OR21_4:%.*]] = or i8 [[B_SROA_5_0_EXTRACT_TRUNC]], [[A_SROA_5_0_EXTRACT_TRUNC]]
-; SSE2-NEXT:    [[TMP4:%.*]] = and i8 [[OR21_4]], 1
-; SSE2-NEXT:    [[ADD12_4:%.*]] = add nuw i8 [[NARROW_4]], [[TMP4]]
-; SSE2-NEXT:    [[SHR_5:%.*]] = lshr i8 [[A_SROA_6_0_EXTRACT_TRUNC]], 1
-; SSE2-NEXT:    [[SHR5_5:%.*]] = lshr i8 [[B_SROA_6_0_EXTRACT_TRUNC]], 1
-; SSE2-NEXT:    [[NARROW_5:%.*]] = add nuw i8 [[SHR5_5]], [[SHR_5]]
-; SSE2-NEXT:    [[OR21_5:%.*]] = or i8 [[B_SROA_6_0_EXTRACT_TRUNC]], [[A_SROA_6_0_EXTRACT_TRUNC]]
-; SSE2-NEXT:    [[TMP5:%.*]] = and i8 [[OR21_5]], 1
-; SSE2-NEXT:    [[ADD12_5:%.*]] = add nuw i8 [[NARROW_5]], [[TMP5]]
-; SSE2-NEXT:    [[SHR_6:%.*]] = lshr i8 [[A_SROA_7_0_EXTRACT_TRUNC]], 1
-; SSE2-NEXT:    [[SHR5_6:%.*]] = lshr i8 [[B_SROA_7_0_EXTRACT_TRUNC]], 1
-; SSE2-NEXT:    [[NARROW_6:%.*]] = add nuw i8 [[SHR5_6]], [[SHR_6]]
-; SSE2-NEXT:    [[OR21_6:%.*]] = or i8 [[B_SROA_7_0_EXTRACT_TRUNC]], [[A_SROA_7_0_EXTRACT_TRUNC]]
-; SSE2-NEXT:    [[TMP6:%.*]] = and i8 [[OR21_6]], 1
-; SSE2-NEXT:    [[ADD12_6:%.*]] = add nuw i8 [[NARROW_6]], [[TMP6]]
-; SSE2-NEXT:    [[SHR_7:%.*]] = lshr i8 [[A_SROA_8_0_EXTRACT_TRUNC]], 1
-; SSE2-NEXT:    [[SHR5_7:%.*]] = lshr i8 [[B_SROA_8_0_EXTRACT_TRUNC]], 1
-; SSE2-NEXT:    [[NARROW_7:%.*]] = add nuw i8 [[SHR5_7]], [[SHR_7]]
-; SSE2-NEXT:    [[OR21_7:%.*]] = or i8 [[B_SROA_8_0_EXTRACT_TRUNC]], [[A_SROA_8_0_EXTRACT_TRUNC]]
-; SSE2-NEXT:    [[TMP7:%.*]] = and i8 [[OR21_7]], 1
-; SSE2-NEXT:    [[ADD12_7:%.*]] = add nuw i8 [[NARROW_7]], [[TMP7]]
-; SSE2-NEXT:    [[SHR_8:%.*]] = lshr i8 [[A_SROA_9_8_EXTRACT_TRUNC]], 1
-; SSE2-NEXT:    [[SHR5_8:%.*]] = lshr i8 [[B_SROA_9_8_EXTRACT_TRUNC]], 1
-; SSE2-NEXT:    [[NARROW_8:%.*]] = add nuw i8 [[SHR5_8]], [[SHR_8]]
-; SSE2-NEXT:    [[OR21_8:%.*]] = or i8 [[B_SROA_9_8_EXTRACT_TRUNC]], [[A_SROA_9_8_EXTRACT_TRUNC]]
-; SSE2-NEXT:    [[TMP8:%.*]] = and i8 [[OR21_8]], 1
-; SSE2-NEXT:    [[ADD12_8:%.*]] = add nuw i8 [[NARROW_8]], [[TMP8]]
-; SSE2-NEXT:    [[SHR_9:%.*]] = lshr i8 [[A_SROA_11_8_EXTRACT_TRUNC]], 1
-; SSE2-NEXT:    [[SHR5_9:%.*]] = lshr i8 [[B_SROA_11_8_EXTRACT_TRUNC]], 1
-; SSE2-NEXT:    [[NARROW_9:%.*]] = add nuw i8 [[SHR5_9]], [[SHR_9]]
-; SSE2-NEXT:    [[OR21_9:%.*]] = or i8 [[B_SROA_11_8_EXTRACT_TRUNC]], [[A_SROA_11_8_EXTRACT_TRUNC]]
-; SSE2-NEXT:    [[TMP9:%.*]] = and i8 [[OR21_9]], 1
-; SSE2-NEXT:    [[ADD12_9:%.*]] = add nuw i8 [[NARROW_9]], [[TMP9]]
-; SSE2-NEXT:    [[SHR_10:%.*]] = lshr i8 [[A_SROA_12_8_EXTRACT_TRUNC]], 1
-; SSE2-NEXT:    [[SHR5_10:%.*]] = lshr i8 [[B_SROA_12_8_EXTRACT_TRUNC]], 1
-; SSE2-NEXT:    [[NARROW_10:%.*]] = add nuw i8 [[SHR5_10]], [[SHR_10]]
-; SSE2-NEXT:    [[OR21_10:%.*]] = or i8 [[B_SROA_12_8_EXTRACT_TRUNC]], [[A_SROA_12_8_EXTRACT_TRUNC]]
-; SSE2-NEXT:    [[TMP10:%.*]] = and i8 [[OR21_10]], 1
-; SSE2-NEXT:    [[ADD12_10:%.*]] = add nuw i8 [[NARROW_10]], [[TMP10]]
-; SSE2-NEXT:    [[SHR_11:%.*]] = lshr i8 [[A_SROA_13_8_EXTRACT_TRUNC]], 1
-; SSE2-NEXT:    [[SHR5_11:%.*]] = lshr i8 [[B_SROA_13_8_EXTRACT_TRUNC]], 1
-; SSE2-NEXT:    [[NARROW_11:%.*]] = add nuw i8 [[SHR5_11]], [[SHR_11]]
-; SSE2-NEXT:    [[OR21_11:%.*]] = or i8 [[B_SROA_13_8_EXTRACT_TRUNC]], [[A_SROA_13_8_EXTRACT_TRUNC]]
-; SSE2-NEXT:    [[TMP11:%.*]] = and i8 [[OR21_11]], 1
-; SSE2-NEXT:    [[ADD12_11:%.*]] = add nuw i8 [[NARROW_11]], [[TMP11]]
-; SSE2-NEXT:    [[SHR_12:%.*]] = lshr i8 [[A_SROA_14_8_EXTRACT_TRUNC]], 1
-; SSE2-NEXT:    [[SHR5_12:%.*]] = lshr i8 [[B_SROA_14_8_EXTRACT_TRUNC]], 1
-; SSE2-NEXT:    [[NARROW_12:%.*]] = add nuw i8 [[SHR5_12]], [[SHR_12]]
-; SSE2-NEXT:    [[OR21_12:%.*]] = or i8 [[B_SROA_14_8_EXTRACT_TRUNC]], [[A_SROA_14_8_EXTRACT_TRUNC]]
-; SSE2-NEXT:    [[TMP12:%.*]] = and i8 [[OR21_12]], 1
-; SSE2-NEXT:    [[ADD12_12:%.*]] = add nuw i8 [[NARROW_12]], [[TMP12]]
-; SSE2-NEXT:    [[SHR_13:%.*]] = lshr i8 [[A_SROA_15_8_EXTRACT_TRUNC]], 1
-; SSE2-NEXT:    [[SHR5_13:%.*]] = lshr i8 [[B_SROA_15_8_EXTRACT_TRUNC]], 1
-; SSE2-NEXT:    [[NARROW_13:%.*]] = add nuw i8 [[SHR5_13]], [[SHR_13]]
-; SSE2-NEXT:    [[OR21_13:%.*]] = or i8 [[B_SROA_15_8_EXTRACT_TRUNC]], [[A_SROA_15_8_EXTRACT_TRUNC]]
-; SSE2-NEXT:    [[TMP13:%.*]] = and i8 [[OR21_13]], 1
-; SSE2-NEXT:    [[ADD12_13:%.*]] = add nuw i8 [[NARROW_13]], [[TMP13]]
-; SSE2-NEXT:    [[SHR_14:%.*]] = lshr i8 [[A_SROA_16_8_EXTRACT_TRUNC]], 1
-; SSE2-NEXT:    [[SHR5_14:%.*]] = lshr i8 [[B_SROA_16_8_EXTRACT_TRUNC]], 1
-; SSE2-NEXT:    [[NARROW_14:%.*]] = add nuw i8 [[SHR5_14]], [[SHR_14]]
-; SSE2-NEXT:    [[OR21_14:%.*]] = or i8 [[B_SROA_16_8_EXTRACT_TRUNC]], [[A_SROA_16_8_EXTRACT_TRUNC]]
-; SSE2-NEXT:    [[TMP14:%.*]] = and i8 [[OR21_14]], 1
-; SSE2-NEXT:    [[ADD12_14:%.*]] = add nuw i8 [[NARROW_14]], [[TMP14]]
-; SSE2-NEXT:    [[SHR_15:%.*]] = lshr i8 [[A_SROA_17_8_EXTRACT_TRUNC]], 1
-; SSE2-NEXT:    [[SHR5_15:%.*]] = lshr i8 [[B_SROA_17_8_EXTRACT_TRUNC]], 1
-; SSE2-NEXT:    [[NARROW_15:%.*]] = add nuw i8 [[SHR5_15]], [[SHR_15]]
-; SSE2-NEXT:    [[OR21_15:%.*]] = or i8 [[B_SROA_17_8_EXTRACT_TRUNC]], [[A_SROA_17_8_EXTRACT_TRUNC]]
-; SSE2-NEXT:    [[TMP15:%.*]] = and i8 [[OR21_15]], 1
-; SSE2-NEXT:    [[ADD12_15:%.*]] = add nuw i8 [[NARROW_15]], [[TMP15]]
-; SSE2-NEXT:    [[RETVAL_SROA_8_0_INSERT_EXT:%.*]] = zext i8 [[ADD12_7]] to i64
-; SSE2-NEXT:    [[RETVAL_SROA_8_0_INSERT_SHIFT:%.*]] = shl nuw i64 [[RETVAL_SROA_8_0_INSERT_EXT]], 56
-; SSE2-NEXT:    [[RETVAL_SROA_7_0_INSERT_EXT:%.*]] = zext i8 [[ADD12_6]] to i64
-; SSE2-NEXT:    [[RETVAL_SROA_7_0_INSERT_SHIFT:%.*]] = shl nuw nsw i64 [[RETVAL_SROA_7_0_INSERT_EXT]], 48
-; SSE2-NEXT:    [[RETVAL_SROA_7_0_INSERT_INSERT:%.*]] = or disjoint i64 [[RETVAL_SROA_8_0_INSERT_SHIFT]], [[RETVAL_SROA_7_0_INSERT_SHIFT]]
-; SSE2-NEXT:    [[RETVAL_SROA_6_0_INSERT_EXT:%.*]] = zext i8 [[ADD12_5]] to i64
-; SSE2-NEXT:    [[RETVAL_SROA_6_0_INSERT_SHIFT:%.*]] = shl nuw nsw i64 [[RETVAL_SROA_6_0_INSERT_EXT]], 40
-; SSE2-NEXT:    [[RETVAL_SROA_6_0_INSERT_INSERT:%.*]] = or disjoint i64 [[RETVAL_SROA_7_0_INSERT_INSERT]], [[RETVAL_SROA_6_0_INSERT_SHIFT]]
-; SSE2-NEXT:    [[RETVAL_SROA_5_0_INSERT_EXT:%.*]] = zext i8 [[ADD12_4]] to i64
-; SSE2-NEXT:    [[RETVAL_SROA_5_0_INSERT_SHIFT:%.*]] = shl nuw nsw i64 [[RETVAL_SROA_5_0_INSERT_EXT]], 32
-; SSE2-NEXT:    [[RETVAL_SROA_5_0_INSERT_INSERT:%.*]] = or disjoint i64 [[RETVAL_SROA_6_0_INSERT_INSERT]], [[RETVAL_SROA_5_0_INSERT_SHIFT]]
-; SSE2-NEXT:    [[RETVAL_SROA_4_0_INSERT_EXT:%.*]] = zext i8 [[ADD12_3]] to i64
-; SSE2-NEXT:    [[RETVAL_SROA_4_0_INSERT_SHIFT:%.*]] = shl nuw nsw i64 [[RETVAL_SROA_4_0_INSERT_EXT]], 24
-; SSE2-NEXT:    [[RETVAL_SROA_4_0_INSERT_INSERT:%.*]] = or disjoint i64 [[RETVAL_SROA_5_0_INSERT_INSERT]], [[RETVAL_SROA_4_0_INSERT_SHIFT]]
-; SSE2-NEXT:    [[RETVAL_SROA_3_0_INSERT_EXT:%.*]] = zext i8 [[ADD12_2]] to i64
-; SSE2-NEXT:    [[RETVAL_SROA_3_0_INSERT_SHIFT:%.*]] = shl nuw nsw i64 [[RETVAL_SROA_3_0_INSERT_EXT]], 16
-; SSE2-NEXT:    [[RETVAL_SROA_2_0_INSERT_EXT:%.*]] = zext i8 [[ADD12_1]] to i64
-; SSE2-NEXT:    [[RETVAL_SROA_2_0_INSERT_SHIFT:%.*]] = shl nuw nsw i64 [[RETVAL_SROA_2_0_INSERT_EXT]], 8
-; SSE2-NEXT:    [[RETVAL_SROA_2_0_INSERT_MASK:%.*]] = or disjoint i64 [[RETVAL_SROA_4_0_INSERT_INSERT]], [[RETVAL_SROA_3_0_INSERT_SHIFT]]
-; SSE2-NEXT:    [[RETVAL_SROA_0_0_INSERT_EXT:%.*]] = zext i8 [[ADD12]] to i64
-; SSE2-NEXT:    [[RETVAL_SROA_0_0_INSERT_MASK:%.*]] = or i64 [[RETVAL_SROA_2_0_INSERT_MASK]], [[RETVAL_SROA_2_0_INSERT_SHIFT]]
-; SSE2-NEXT:    [[RETVAL_SROA_0_0_INSERT_INSERT:%.*]] = or i64 [[RETVAL_SROA_0_0_INSERT_MASK]], [[RETVAL_SROA_0_0_INSERT_EXT]]
-; SSE2-NEXT:    [[DOTFCA_0_INSERT:%.*]] = insertvalue { i64, i64 } poison, i64 [[RETVAL_SROA_0_0_INSERT_INSERT]], 0
-; SSE2-NEXT:    [[RETVAL_SROA_17_8_INSERT_EXT:%.*]] = zext i8 [[ADD12_15]] to i64
-; SSE2-NEXT:    [[RETVAL_SROA_17_8_INSERT_SHIFT:%.*]] = shl nuw i64 [[RETVAL_SROA_17_8_INSERT_EXT]], 56
-; SSE2-NEXT:    [[RETVAL_SROA_16_8_INSERT_EXT:%.*]] = zext i8 [[ADD12_14]] to i64
-; SSE2-NEXT:    [[RETVAL_SROA_16_8_INSERT_SHIFT:%.*]] = shl nuw nsw i64 [[RETVAL_SROA_16_8_INSERT_EXT]], 48
-; SSE2-NEXT:    [[RETVAL_SROA_16_8_INSERT_INSERT:%.*]] = or disjoint i64 [[RETVAL_SROA_17_8_INSERT_SHIFT]], [[RETVAL_SROA_16_8_INSERT_SHIFT]]
-; SSE2-NEXT:    [[RETVAL_SROA_15_8_INSERT_EXT:%.*]] = zext i8 [[ADD12_13]] to i64
-; SSE2-NEXT:    [[RETVAL_SROA_15_8_INSERT_SHIFT:%.*]] = shl nuw nsw i64 [[RETVAL_SROA_15_8_INSERT_EXT]], 40
-; SSE2-NEXT:    [[RETVAL_SROA_15_8_INSERT_INSERT:%.*]] = or disjoint i64 [[RETVAL_SROA_16_8_INSERT_INSERT]], [[RETVAL_SROA_15_8_INSERT_SHIFT]]
-; SSE2-NEXT:    [[RETVAL_SROA_14_8_INSERT_EXT:%.*]] = zext i8 [[ADD12_12]] to i64
-; SSE2-NEXT:    [[RETVAL_SROA_14_8_INSERT_SHIFT:%.*]] = shl nuw nsw i64 [[RETVAL_SROA_14_8_INSERT_EXT]], 32
-; SSE2-NEXT:    [[RETVAL_SROA_14_8_INSERT_INSERT:%.*]] = or disjoint i64 [[RETVAL_SROA_15_8_INSERT_INSERT]], [[RETVAL_SROA_14_8_INSERT_SHIFT]]
-; SSE2-NEXT:    [[RETVAL_SROA_13_8_INSERT_EXT:%.*]] = zext i8 [[ADD12_11]] to i64
-; SSE2-NEXT:    [[RETVAL_SROA_13_8_INSERT_SHIFT:%.*]] = shl nuw nsw i64 [[RETVAL_SROA_13_8_INSERT_EXT]], 24
-; SSE2-NEXT:    [[RETVAL_SROA_13_8_INSERT_INSERT:%.*]] = or disjoint i64 [[RETVAL_SROA_14_8_INSERT_INSERT]], [[RETVAL_SROA_13_8_INSERT_SHIFT]]
-; SSE2-NEXT:    [[RETVAL_SROA_12_8_INSERT_EXT:%.*]] = zext i8 [[ADD12_10]] to i64
-; SSE2-NEXT:    [[RETVAL_SROA_12_8_INSERT_SHIFT:%.*]] = shl nuw nsw i64 [[RETVAL_SROA_12_8_INSERT_EXT]], 16
-; SSE2-NEXT:    [[RETVAL_SROA_11_8_INSERT_EXT:%.*]] = zext i8 [[ADD12_9]] to i64
-; SSE2-NEXT:    [[RETVAL_SROA_11_8_INSERT_SHIFT:%.*]] = shl nuw nsw i64 [[RETVAL_SROA_11_8_INSERT_EXT]], 8
-; SSE2-NEXT:    [[RETVAL_SROA_11_8_INSERT_MASK:%.*]] = or disjoint i64 [[RETVAL_SROA_13_8_INSERT_INSERT]], [[RETVAL_SROA_12_8_INSERT_SHIFT]]
-; SSE2-NEXT:    [[RETVAL_SROA_9_8_INSERT_EXT:%.*]] = zext i8 [[ADD12_8]] to i64
-; SSE2-NEXT:    [[RETVAL_SROA_9_8_INSERT_MASK:%.*]] = or i64 [[RETVAL_SROA_11_8_INSERT_MASK]], [[RETVAL_SROA_11_8_INSERT_SHIFT]]
-; SSE2-NEXT:    [[RETVAL_SROA_9_8_INSERT_INSERT:%.*]] = or i64 [[RETVAL_SROA_9_8_INSERT_MASK]], [[RETVAL_SROA_9_8_INSERT_EXT]]
-; SSE2-NEXT:    [[DOTFCA_1_INSERT:%.*]] = insertvalue { i64, i64 } [[DOTFCA_0_INSERT]], i64 [[RETVAL_SROA_9_8_INSERT_INSERT]], 1
-; SSE2-NEXT:    ret { i64, i64 } [[DOTFCA_1_INSERT]]
-;
-; SSE4-LABEL: @avgr_16_u8_alt(
-; SSE4-NEXT:  entry:
-; SSE4-NEXT:    [[TMP0:%.*]] = insertelement <8 x i64> poison, i64 [[A_COERCE0:%.*]], i64 0
-; SSE4-NEXT:    [[TMP1:%.*]] = shufflevector <8 x i64> [[TMP0]], <8 x i64> poison, <8 x i32> zeroinitializer
-; SSE4-NEXT:    [[TMP2:%.*]] = lshr <8 x i64> [[TMP1]], <i64 0, i64 8, i64 16, i64 24, i64 32, i64 40, i64 48, i64 56>
-; SSE4-NEXT:    [[TMP3:%.*]] = trunc <8 x i64> [[TMP2]] to <8 x i8>
-; SSE4-NEXT:    [[TMP4:%.*]] = insertelement <8 x i64> poison, i64 [[B_COERCE0:%.*]], i64 0
-; SSE4-NEXT:    [[TMP5:%.*]] = shufflevector <8 x i64> [[TMP4]], <8 x i64> poison, <8 x i32> zeroinitializer
-; SSE4-NEXT:    [[TMP6:%.*]] = lshr <8 x i64> [[TMP5]], <i64 0, i64 8, i64 16, i64 24, i64 32, i64 40, i64 48, i64 56>
-; SSE4-NEXT:    [[TMP7:%.*]] = trunc <8 x i64> [[TMP6]] to <8 x i8>
-; SSE4-NEXT:    [[TMP8:%.*]] = lshr <8 x i8> [[TMP3]], splat (i8 1)
-; SSE4-NEXT:    [[TMP9:%.*]] = lshr <8 x i8> [[TMP7]], splat (i8 1)
-; SSE4-NEXT:    [[TMP10:%.*]] = add nuw <8 x i8> [[TMP9]], [[TMP8]]
-; SSE4-NEXT:    [[TMP11:%.*]] = or <8 x i8> [[TMP7]], [[TMP3]]
-; SSE4-NEXT:    [[TMP12:%.*]] = and <8 x i8> [[TMP11]], splat (i8 1)
-; SSE4-NEXT:    [[TMP13:%.*]] = add nuw <8 x i8> [[TMP10]], [[TMP12]]
-; SSE4-NEXT:    [[VEC2STRUCT_SLOT_SROA_0_0_VEC_EXTRACT:%.*]] = bitcast <8 x i8> [[TMP13]] to i64
-; SSE4-NEXT:    [[DOTFCA_0_INSERT:%.*]] = insertvalue { i64, i64 } poison, i64 [[VEC2STRUCT_SLOT_SROA_0_0_VEC_EXTRACT]], 0
-; SSE4-NEXT:    [[TMP17:%.*]] = insertelement <8 x i64> poison, i64 [[A_COERCE1:%.*]], i64 0
-; SSE4-NEXT:    [[TMP18:%.*]] = shufflevector <8 x i64> [[TMP17]], <8 x i64> poison, <8 x i32> zeroinitializer
-; SSE4-NEXT:    [[TMP19:%.*]] = lshr <8 x i64> [[TMP18]], <i64 0, i64 8, i64 16, i64 24, i64 32, i64 40, i64 48, i64 56>
-; SSE4-NEXT:    [[TMP20:%.*]] = trunc <8 x i64> [[TMP19]] to <8 x i8>
-; SSE4-NEXT:    [[TMP21:%.*]] = insertelement <8 x i64> poison, i64 [[B_COERCE1:%.*]], i64 0
-; SSE4-NEXT:    [[TMP22:%.*]] = shufflevector <8 x i64> [[TMP21]], <8 x i64> poison, <8 x i32> zeroinitializer
-; SSE4-NEXT:    [[TMP23:%.*]] = lshr <8 x i64> [[TMP22]], <i64 0, i64 8, i64 16, i64 24, i64 32, i64 40, i64 48, i64 56>
-; SSE4-NEXT:    [[TMP24:%.*]] = trunc <8 x i64> [[TMP23]] to <8 x i8>
-; SSE4-NEXT:    [[TMP25:%.*]] = lshr <8 x i8> [[TMP20]], splat (i8 1)
-; SSE4-NEXT:    [[TMP26:%.*]] = lshr <8 x i8> [[TMP24]], splat (i8 1)
-; SSE4-NEXT:    [[TMP27:%.*]] = add nuw <8 x i8> [[TMP26]], [[TMP25]]
-; SSE4-NEXT:    [[TMP28:%.*]] = or <8 x i8> [[TMP24]], [[TMP20]]
-; SSE4-NEXT:    [[TMP29:%.*]] = and <8 x i8> [[TMP28]], splat (i8 1)
-; SSE4-NEXT:    [[TMP30:%.*]] = add nuw <8 x i8> [[TMP27]], [[TMP29]]
-; SSE4-NEXT:    [[TMP49:%.*]] = bitcast <8 x i8> [[TMP30]] to i64
-; SSE4-NEXT:    [[DOTFCA_1_INSERT:%.*]] = insertvalue { i64, i64 } [[DOTFCA_0_INSERT]], i64 [[TMP49]], 1
-; SSE4-NEXT:    ret { i64, i64 } [[DOTFCA_1_INSERT]]
-;
-; AVX-LABEL: @avgr_16_u8_alt(
-; AVX-NEXT:  entry:
-; AVX-NEXT:    [[TMP0:%.*]] = insertelement <8 x i64> poison, i64 [[A_COERCE0:%.*]], i64 0
-; AVX-NEXT:    [[TMP1:%.*]] = shufflevector <8 x i64> [[TMP0]], <8 x i64> poison, <8 x i32> zeroinitializer
-; AVX-NEXT:    [[TMP2:%.*]] = lshr <8 x i64> [[TMP1]], <i64 0, i64 8, i64 16, i64 24, i64 32, i64 40, i64 48, i64 56>
-; AVX-NEXT:    [[TMP3:%.*]] = trunc <8 x i64> [[TMP2]] to <8 x i8>
-; AVX-NEXT:    [[TMP4:%.*]] = insertelement <8 x i64> poison, i64 [[B_COERCE0:%.*]], i64 0
-; AVX-NEXT:    [[TMP5:%.*]] = shufflevector <8 x i64> [[TMP4]], <8 x i64> poison, <8 x i32> zeroinitializer
-; AVX-NEXT:    [[TMP6:%.*]] = lshr <8 x i64> [[TMP5]], <i64 0, i64 8, i64 16, i64 24, i64 32, i64 40, i64 48, i64 56>
-; AVX-NEXT:    [[TMP7:%.*]] = trunc <8 x i64> [[TMP6]] to <8 x i8>
-; AVX-NEXT:    [[TMP8:%.*]] = lshr <8 x i8> [[TMP3]], splat (i8 1)
-; AVX-NEXT:    [[TMP9:%.*]] = lshr <8 x i8> [[TMP7]], splat (i8 1)
-; AVX-NEXT:    [[TMP10:%.*]] = add nuw <8 x i8> [[TMP9]], [[TMP8]]
-; AVX-NEXT:    [[TMP11:%.*]] = or <8 x i8> [[TMP7]], [[TMP3]]
-; AVX-NEXT:    [[TMP12:%.*]] = and <8 x i8> [[TMP11]], splat (i8 1)
-; AVX-NEXT:    [[TMP13:%.*]] = add nuw <8 x i8> [[TMP10]], [[TMP12]]
-; AVX-NEXT:    [[TMP16:%.*]] = bitcast <8 x i8> [[TMP13]] to i64
-; AVX-NEXT:    [[DOTFCA_0_INSERT:%.*]] = insertvalue { i64, i64 } poison, i64 [[TMP16]], 0
-; AVX-NEXT:    [[TMP17:%.*]] = insertelement <8 x i64> poison, i64 [[A_COERCE1:%.*]], i64 0
-; AVX-NEXT:    [[TMP18:%.*]] = shufflevector <8 x i64> [[TMP17]], <8 x i64> poison, <8 x i32> zeroinitializer
-; AVX-NEXT:    [[TMP19:%.*]] = lshr <8 x i64> [[TMP18]], <i64 0, i64 8, i64 16, i64 24, i64 32, i64 40, i64 48, i64 56>
-; AVX-NEXT:    [[TMP20:%.*]] = trunc <8 x i64> [[TMP19]] to <8 x i8>
-; AVX-NEXT:    [[TMP21:%.*]] = insertelement <8 x i64> poison, i64 [[B_COERCE1:%.*]], i64 0
-; AVX-NEXT:    [[TMP22:%.*]] = shufflevector <8 x i64> [[TMP21]], <8 x i64> poison, <8 x i32> zeroinitializer
-; AVX-NEXT:    [[TMP23:%.*]] = lshr <8 x i64> [[TMP22]], <i64 0, i64 8, i64 16, i64 24, i64 32, i64 40, i64 48, i64 56>
-; AVX-NEXT:    [[TMP24:%.*]] = trunc <8 x i64> [[TMP23]] to <8 x i8>
-; AVX-NEXT:    [[TMP25:%.*]] = lshr <8 x i8> [[TMP20]], splat (i8 1)
-; AVX-NEXT:    [[TMP26:%.*]] = lshr <8 x i8> [[TMP24]], splat (i8 1)
-; AVX-NEXT:    [[TMP27:%.*]] = add nuw <8 x i8> [[TMP26]], [[TMP25]]
-; AVX-NEXT:    [[TMP28:%.*]] = or <8 x i8> [[TMP24]], [[TMP20]]
-; AVX-NEXT:    [[TMP29:%.*]] = and <8 x i8> [[TMP28]], splat (i8 1)
-; AVX-NEXT:    [[TMP30:%.*]] = add nuw <8 x i8> [[TMP27]], [[TMP29]]
-; AVX-NEXT:    [[TMP33:%.*]] = bitcast <8 x i8> [[TMP30]] to i64
-; AVX-NEXT:    [[DOTFCA_1_INSERT:%.*]] = insertvalue { i64, i64 } [[DOTFCA_0_INSERT]], i64 [[TMP33]], 1
-; AVX-NEXT:    ret { i64, i64 } [[DOTFCA_1_INSERT]]
+; CHECK-LABEL: @avgr_16_u8_alt(
+; CHECK-NEXT:  entry:
+; CHECK-NEXT:    [[TMP0:%.*]] = bitcast i64 [[A_COERCE0:%.*]] to <8 x i8>
+; CHECK-NEXT:    [[TMP1:%.*]] = lshr <8 x i8> [[TMP0]], splat (i8 1)
+; CHECK-NEXT:    [[TMP2:%.*]] = bitcast i64 [[B_COERCE0:%.*]] to <8 x i8>
+; CHECK-NEXT:    [[TMP3:%.*]] = lshr <8 x i8> [[TMP2]], splat (i8 1)
+; CHECK-NEXT:    [[TMP4:%.*]] = add nuw <8 x i8> [[TMP3]], [[TMP1]]
+; CHECK-NEXT:    [[TMP5:%.*]] = or <8 x i8> [[TMP2]], [[TMP0]]
+; CHECK-NEXT:    [[TMP6:%.*]] = and <8 x i8> [[TMP5]], splat (i8 1)
+; CHECK-NEXT:    [[TMP7:%.*]] = add nuw <8 x i8> [[TMP4]], [[TMP6]]
+; CHECK-NEXT:    [[TMP8:%.*]] = bitcast <8 x i8> [[TMP7]] to i64
+; CHECK-NEXT:    [[DOTFCA_0_INSERT:%.*]] = insertvalue { i64, i64 } poison, i64 [[TMP8]], 0
+; CHECK-NEXT:    [[TMP9:%.*]] = bitcast i64 [[A_COERCE1:%.*]] to <8 x i8>
+; CHECK-NEXT:    [[TMP10:%.*]] = lshr <8 x i8> [[TMP9]], splat (i8 1)
+; CHECK-NEXT:    [[TMP11:%.*]] = bitcast i64 [[B_COERCE1:%.*]] to <8 x i8>
+; CHECK-NEXT:    [[TMP12:%.*]] = lshr <8 x i8> [[TMP11]], splat (i8 1)
+; CHECK-NEXT:    [[TMP13:%.*]] = add nuw <8 x i8> [[TMP12]], [[TMP10]]
+; CHECK-NEXT:    [[TMP14:%.*]] = or <8 x i8> [[TMP11]], [[TMP9]]
+; CHECK-NEXT:    [[TMP15:%.*]] = and <8 x i8> [[TMP14]], splat (i8 1)
+; CHECK-NEXT:    [[TMP16:%.*]] = add nuw <8 x i8> [[TMP13]], [[TMP15]]
+; CHECK-NEXT:    [[TMP17:%.*]] = bitcast <8 x i8> [[TMP16]] to i64
+; CHECK-NEXT:    [[DOTFCA_1_INSERT:%.*]] = insertvalue { i64, i64 } [[DOTFCA_0_INSERT]], i64 [[TMP17]], 1
+; CHECK-NEXT:    ret { i64, i64 } [[DOTFCA_1_INSERT]]
 ;
 entry:
   %retval = alloca %"struct.std::array16", align 1
@@ -787,210 +492,29 @@ for.body:                                         ; preds = %for.cond
 }
 
 define { i64, i64 } @avgr_8_u16_alt(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerce0, i64 %b.coerce1)  {
-; SSE2-LABEL: @avgr_8_u16_alt(
-; SSE2-NEXT:  entry:
-; SSE2-NEXT:    [[A_SROA_4_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE0:%.*]], 48
-; SSE2-NEXT:    [[A_SROA_3_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE0]], 32
-; SSE2-NEXT:    [[A_SROA_2_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE0]], 16
-; SSE2-NEXT:    [[B_SROA_4_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE0:%.*]], 48
-; SSE2-NEXT:    [[B_SROA_3_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE0]], 32
-; SSE2-NEXT:    [[B_SROA_2_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE0]], 16
-; SSE2-NEXT:    [[TMP0:%.*]] = trunc i64 [[A_COERCE0]] to i16
-; SSE2-NEXT:    [[TMP14:%.*]] = insertelement <2 x i16> poison, i16 [[TMP0]], i64 0
-; SSE2-NEXT:    [[TMP17:%.*]] = trunc i64 [[A_SROA_2_0_EXTRACT_SHIFT]] to i16
-; SSE2-NEXT:    [[TMP18:%.*]] = insertelement <2 x i16> [[TMP14]], i16 [[TMP17]], i64 1
-; SSE2-NEXT:    [[TMP4:%.*]] = trunc i64 [[B_COERCE0]] to i16
-; SSE2-NEXT:    [[TMP5:%.*]] = insertelement <2 x i16> poison, i16 [[TMP4]], i64 0
-; SSE2-NEXT:    [[TMP6:%.*]] = trunc i64 [[B_SROA_2_0_EXTRACT_SHIFT]] to i16
-; SSE2-NEXT:    [[TMP7:%.*]] = insertelement <2 x i16> [[TMP5]], i16 [[TMP6]], i64 1
-; SSE2-NEXT:    [[TMP19:%.*]] = lshr <2 x i16> [[TMP18]], splat (i16 1)
-; SSE2-NEXT:    [[TMP21:%.*]] = lshr <2 x i16> [[TMP7]], splat (i16 1)
-; SSE2-NEXT:    [[TMP10:%.*]] = add nuw <2 x i16> [[TMP21]], [[TMP19]]
-; SSE2-NEXT:    [[TMP37:%.*]] = shufflevector <2 x i16> [[TMP7]], <2 x i16> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
-; SSE2-NEXT:    [[TMP1:%.*]] = shufflevector <2 x i16> [[TMP18]], <2 x i16> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
-; SSE2-NEXT:    [[TMP44:%.*]] = shufflevector <2 x i16> [[TMP10]], <2 x i16> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
-; SSE2-NEXT:    [[A_SROA_9_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE1:%.*]], 48
-; SSE2-NEXT:    [[A_SROA_8_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE1]], 32
-; SSE2-NEXT:    [[A_SROA_7_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE1]], 16
-; SSE2-NEXT:    [[B_SROA_9_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE2:%.*]], 48
-; SSE2-NEXT:    [[B_SROA_8_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE2]], 32
-; SSE2-NEXT:    [[B_SROA_7_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE2]], 16
-; SSE2-NEXT:    [[B_SROA_9_8_EXTRACT_TRUNC:%.*]] = trunc i64 [[B_SROA_8_8_EXTRACT_SHIFT]] to i16
-; SSE2-NEXT:    [[B_SROA_8_8_EXTRACT_TRUNC:%.*]] = trunc nuw i64 [[B_SROA_9_8_EXTRACT_SHIFT]] to i16
-; SSE2-NEXT:    [[B_SROA_3_0_EXTRACT_TRUNC:%.*]] = trunc i64 [[B_SROA_3_0_EXTRACT_SHIFT]] to i16
-; SSE2-NEXT:    [[B_SROA_4_0_EXTRACT_TRUNC:%.*]] = trunc nuw i64 [[B_SROA_4_0_EXTRACT_SHIFT]] to i16
-; SSE2-NEXT:    [[TMP38:%.*]] = insertelement <4 x i16> [[TMP37]], i16 [[B_SROA_3_0_EXTRACT_TRUNC]], i64 2
-; SSE2-NEXT:    [[TMP39:%.*]] = insertelement <4 x i16> [[TMP38]], i16 [[B_SROA_4_0_EXTRACT_TRUNC]], i64 3
-; SSE2-NEXT:    [[NARROW:%.*]] = trunc i64 [[A_SROA_8_8_EXTRACT_SHIFT]] to i16
-; SSE2-NEXT:    [[A_SROA_5_8_EXTRACT_TRUNC:%.*]] = trunc nuw i64 [[A_SROA_9_8_EXTRACT_SHIFT]] to i16
-; SSE2-NEXT:    [[B_SROA_2_0_EXTRACT_TRUNC:%.*]] = trunc i64 [[A_SROA_3_0_EXTRACT_SHIFT]] to i16
-; SSE2-NEXT:    [[B_SROA_0_0_EXTRACT_TRUNC:%.*]] = trunc nuw i64 [[A_SROA_4_0_EXTRACT_SHIFT]] to i16
-; SSE2-NEXT:    [[TMP2:%.*]] = insertelement <4 x i16> [[TMP1]], i16 [[B_SROA_2_0_EXTRACT_TRUNC]], i64 2
-; SSE2-NEXT:    [[TMP3:%.*]] = insertelement <4 x i16> [[TMP2]], i16 [[B_SROA_0_0_EXTRACT_TRUNC]], i64 3
-; SSE2-NEXT:    [[TMP8:%.*]] = or <4 x i16> [[TMP39]], [[TMP3]]
-; SSE2-NEXT:    [[TMP9:%.*]] = and <4 x i16> [[TMP8]], splat (i16 1)
-; SSE2-NEXT:    [[TMP48:%.*]] = insertelement <4 x i16> poison, i16 [[B_SROA_0_0_EXTRACT_TRUNC]], i64 0
-; SSE2-NEXT:    [[TMP49:%.*]] = insertelement <4 x i16> [[TMP48]], i16 [[A_SROA_5_8_EXTRACT_TRUNC]], i64 1
-; SSE2-NEXT:    [[TMP12:%.*]] = insertelement <4 x i16> [[TMP49]], i16 [[B_SROA_2_0_EXTRACT_TRUNC]], i64 2
-; SSE2-NEXT:    [[TMP13:%.*]] = insertelement <4 x i16> [[TMP12]], i16 [[NARROW]], i64 3
-; SSE2-NEXT:    [[TMP50:%.*]] = lshr <4 x i16> [[TMP13]], splat (i16 1)
-; SSE2-NEXT:    [[TMP51:%.*]] = insertelement <4 x i16> poison, i16 [[B_SROA_4_0_EXTRACT_TRUNC]], i64 0
-; SSE2-NEXT:    [[TMP52:%.*]] = insertelement <4 x i16> [[TMP51]], i16 [[B_SROA_8_8_EXTRACT_TRUNC]], i64 1
-; SSE2-NEXT:    [[TMP53:%.*]] = insertelement <4 x i16> [[TMP52]], i16 [[B_SROA_3_0_EXTRACT_TRUNC]], i64 2
-; SSE2-NEXT:    [[TMP54:%.*]] = insertelement <4 x i16> [[TMP53]], i16 [[B_SROA_9_8_EXTRACT_TRUNC]], i64 3
-; SSE2-NEXT:    [[TMP55:%.*]] = lshr <4 x i16> [[TMP54]], splat (i16 1)
-; SSE2-NEXT:    [[TMP56:%.*]] = add nuw <4 x i16> [[TMP55]], [[TMP50]]
-; SSE2-NEXT:    [[TMP57:%.*]] = shufflevector <4 x i16> [[TMP44]], <4 x i16> [[TMP56]], <4 x i32> <i32 0, i32 1, i32 6, i32 4>
-; SSE2-NEXT:    [[TMP15:%.*]] = add nuw <4 x i16> [[TMP57]], [[TMP9]]
-; SSE2-NEXT:    [[TMP16:%.*]] = bitcast <4 x i16> [[TMP15]] to i64
-; SSE2-NEXT:    [[DOTFCA_0_INSERT:%.*]] = insertvalue { i64, i64 } poison, i64 [[TMP16]], 0
-; SSE2-NEXT:    [[TMP40:%.*]] = trunc i64 [[B_COERCE1]] to i16
-; SSE2-NEXT:    [[TMP41:%.*]] = insertelement <2 x i16> poison, i16 [[TMP40]], i64 0
-; SSE2-NEXT:    [[TMP42:%.*]] = trunc i64 [[A_SROA_7_8_EXTRACT_SHIFT]] to i16
-; SSE2-NEXT:    [[TMP27:%.*]] = insertelement <2 x i16> [[TMP41]], i16 [[TMP42]], i64 1
-; SSE2-NEXT:    [[TMP28:%.*]] = trunc i64 [[B_COERCE2]] to i16
-; SSE2-NEXT:    [[TMP29:%.*]] = insertelement <2 x i16> poison, i16 [[TMP28]], i64 0
-; SSE2-NEXT:    [[TMP45:%.*]] = trunc i64 [[B_SROA_7_8_EXTRACT_SHIFT]] to i16
-; SSE2-NEXT:    [[TMP46:%.*]] = insertelement <2 x i16> [[TMP29]], i16 [[TMP45]], i64 1
-; SSE2-NEXT:    [[TMP32:%.*]] = lshr <2 x i16> [[TMP27]], splat (i16 1)
-; SSE2-NEXT:    [[TMP47:%.*]] = lshr <2 x i16> [[TMP46]], splat (i16 1)
-; SSE2-NEXT:    [[TMP34:%.*]] = add nuw <2 x i16> [[TMP47]], [[TMP32]]
-; SSE2-NEXT:    [[TMP35:%.*]] = shufflevector <2 x i16> [[TMP46]], <2 x i16> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
-; SSE2-NEXT:    [[TMP36:%.*]] = insertelement <4 x i16> [[TMP35]], i16 [[B_SROA_9_8_EXTRACT_TRUNC]], i64 2
-; SSE2-NEXT:    [[TMP20:%.*]] = insertelement <4 x i16> [[TMP36]], i16 [[B_SROA_8_8_EXTRACT_TRUNC]], i64 3
-; SSE2-NEXT:    [[TMP22:%.*]] = shufflevector <2 x i16> [[TMP27]], <2 x i16> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
-; SSE2-NEXT:    [[TMP23:%.*]] = insertelement <4 x i16> [[TMP22]], i16 [[NARROW]], i64 2
-; SSE2-NEXT:    [[TMP24:%.*]] = insertelement <4 x i16> [[TMP23]], i16 [[A_SROA_5_8_EXTRACT_TRUNC]], i64 3
-; SSE2-NEXT:    [[TMP25:%.*]] = or <4 x i16> [[TMP20]], [[TMP24]]
-; SSE2-NEXT:    [[TMP26:%.*]] = and <4 x i16> [[TMP25]], splat (i16 1)
-; SSE2-NEXT:    [[TMP43:%.*]] = shufflevector <2 x i16> [[TMP34]], <2 x i16> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
-; SSE2-NEXT:    [[TMP30:%.*]] = shufflevector <4 x i16> [[TMP43]], <4 x i16> [[TMP56]], <4 x i32> <i32 0, i32 1, i32 7, i32 5>
-; SSE2-NEXT:    [[TMP31:%.*]] = add nuw <4 x i16> [[TMP30]], [[TMP26]]
-; SSE2-NEXT:    [[TMP33:%.*]] = bitcast <4 x i16> [[TMP31]] to i64
-; SSE2-NEXT:    [[DOTFCA_1_INSERT:%.*]] = insertvalue { i64, i64 } [[DOTFCA_0_INSERT]], i64 [[TMP33]], 1
-; SSE2-NEXT:    ret { i64, i64 } [[DOTFCA_1_INSERT]]
-;
-; SSE4-LABEL: @avgr_8_u16_alt(
-; SSE4-NEXT:  entry:
-; SSE4-NEXT:    [[A_SROA_4_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE0:%.*]], 48
-; SSE4-NEXT:    [[A_SROA_3_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE0]], 32
-; SSE4-NEXT:    [[A_SROA_2_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE0]], 16
-; SSE4-NEXT:    [[B_SROA_0_0_EXTRACT_TRUNC:%.*]] = trunc nuw i64 [[A_SROA_4_0_EXTRACT_SHIFT]] to i16
-; SSE4-NEXT:    [[B_SROA_2_0_EXTRACT_TRUNC:%.*]] = trunc i64 [[A_SROA_3_0_EXTRACT_SHIFT]] to i16
-; SSE4-NEXT:    [[B_SROA_4_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE0:%.*]], 48
-; SSE4-NEXT:    [[B_SROA_3_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE0]], 32
-; SSE4-NEXT:    [[B_SROA_2_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE0]], 16
-; SSE4-NEXT:    [[B_SROA_4_0_EXTRACT_TRUNC:%.*]] = trunc nuw i64 [[B_SROA_4_0_EXTRACT_SHIFT]] to i16
-; SSE4-NEXT:    [[B_SROA_3_0_EXTRACT_TRUNC:%.*]] = trunc i64 [[B_SROA_3_0_EXTRACT_SHIFT]] to i16
-; SSE4-NEXT:    [[SHR5:%.*]] = lshr i16 [[B_SROA_0_0_EXTRACT_TRUNC]], 1
-; SSE4-NEXT:    [[SHR5_1:%.*]] = lshr i16 [[B_SROA_2_0_EXTRACT_TRUNC]], 1
-; SSE4-NEXT:    [[SHR5_3:%.*]] = lshr i16 [[B_SROA_4_0_EXTRACT_TRUNC]], 1
-; SSE4-NEXT:    [[SHR5_2:%.*]] = lshr i16 [[B_SROA_3_0_EXTRACT_TRUNC]], 1
-; SSE4-NEXT:    [[NARROW:%.*]] = add nuw i16 [[SHR5_3]], [[SHR5]]
-; SSE4-NEXT:    [[NARROW_1:%.*]] = add nuw i16 [[SHR5_2]], [[SHR5_1]]
-; SSE4-NEXT:    [[TMP0:%.*]] = trunc i64 [[A_COERCE0]] to i16
-; SSE4-NEXT:    [[TMP14:%.*]] = insertelement <2 x i16> poison, i16 [[TMP0]], i64 0
-; SSE4-NEXT:    [[TMP17:%.*]] = trunc i64 [[A_SROA_2_0_EXTRACT_SHIFT]] to i16
-; SSE4-NEXT:    [[TMP18:%.*]] = insertelement <2 x i16> [[TMP14]], i16 [[TMP17]], i64 1
-; SSE4-NEXT:    [[TMP4:%.*]] = trunc i64 [[B_COERCE0]] to i16
-; SSE4-NEXT:    [[TMP5:%.*]] = insertelement <2 x i16> poison, i16 [[TMP4]], i64 0
-; SSE4-NEXT:    [[TMP6:%.*]] = trunc i64 [[B_SROA_2_0_EXTRACT_SHIFT]] to i16
-; SSE4-NEXT:    [[TMP7:%.*]] = insertelement <2 x i16> [[TMP5]], i16 [[TMP6]], i64 1
-; SSE4-NEXT:    [[TMP19:%.*]] = lshr <2 x i16> [[TMP18]], splat (i16 1)
-; SSE4-NEXT:    [[TMP21:%.*]] = lshr <2 x i16> [[TMP7]], splat (i16 1)
-; SSE4-NEXT:    [[TMP10:%.*]] = add nuw <2 x i16> [[TMP21]], [[TMP19]]
-; SSE4-NEXT:    [[TMP37:%.*]] = shufflevector <2 x i16> [[TMP7]], <2 x i16> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
-; SSE4-NEXT:    [[TMP38:%.*]] = insertelement <4 x i16> [[TMP37]], i16 [[B_SROA_3_0_EXTRACT_TRUNC]], i64 2
-; SSE4-NEXT:    [[TMP39:%.*]] = insertelement <4 x i16> [[TMP38]], i16 [[B_SROA_4_0_EXTRACT_TRUNC]], i64 3
-; SSE4-NEXT:    [[TMP1:%.*]] = shufflevector <2 x i16> [[TMP18]], <2 x i16> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
-; SSE4-NEXT:    [[TMP2:%.*]] = insertelement <4 x i16> [[TMP1]], i16 [[B_SROA_2_0_EXTRACT_TRUNC]], i64 2
-; SSE4-NEXT:    [[TMP3:%.*]] = insertelement <4 x i16> [[TMP2]], i16 [[B_SROA_0_0_EXTRACT_TRUNC]], i64 3
-; SSE4-NEXT:    [[TMP8:%.*]] = or <4 x i16> [[TMP39]], [[TMP3]]
-; SSE4-NEXT:    [[TMP9:%.*]] = and <4 x i16> [[TMP8]], splat (i16 1)
-; SSE4-NEXT:    [[TMP11:%.*]] = shufflevector <2 x i16> [[TMP10]], <2 x i16> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
-; SSE4-NEXT:    [[TMP12:%.*]] = insertelement <4 x i16> [[TMP11]], i16 [[NARROW_1]], i64 2
-; SSE4-NEXT:    [[TMP13:%.*]] = insertelement <4 x i16> [[TMP12]], i16 [[NARROW]], i64 3
-; SSE4-NEXT:    [[TMP15:%.*]] = add nuw <4 x i16> [[TMP13]], [[TMP9]]
-; SSE4-NEXT:    [[TMP16:%.*]] = bitcast <4 x i16> [[TMP15]] to i64
-; SSE4-NEXT:    [[DOTFCA_0_INSERT:%.*]] = insertvalue { i64, i64 } poison, i64 [[TMP16]], 0
-; SSE4-NEXT:    [[A_SROA_9_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE1:%.*]], 48
-; SSE4-NEXT:    [[B_SROA_8_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE1]], 32
-; SSE4-NEXT:    [[A_SROA_7_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE1]], 16
-; SSE4-NEXT:    [[A_SROA_5_8_EXTRACT_TRUNC:%.*]] = trunc nuw i64 [[A_SROA_9_8_EXTRACT_SHIFT]] to i16
-; SSE4-NEXT:    [[A_SROA_7_8_EXTRACT_TRUNC:%.*]] = trunc i64 [[B_SROA_8_8_EXTRACT_SHIFT]] to i16
-; SSE4-NEXT:    [[B_SROA_9_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE2:%.*]], 48
-; SSE4-NEXT:    [[B_SROA_8_8_EXTRACT_SHIFT1:%.*]] = lshr i64 [[B_COERCE2]], 32
-; SSE4-NEXT:    [[B_SROA_7_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE2]], 16
-; SSE4-NEXT:    [[B_SROA_8_8_EXTRACT_TRUNC:%.*]] = trunc nuw i64 [[B_SROA_9_8_EXTRACT_SHIFT]] to i16
-; SSE4-NEXT:    [[B_SROA_9_8_EXTRACT_TRUNC:%.*]] = trunc i64 [[B_SROA_8_8_EXTRACT_SHIFT1]] to i16
-; SSE4-NEXT:    [[SHR_6:%.*]] = lshr i16 [[A_SROA_5_8_EXTRACT_TRUNC]], 1
-; SSE4-NEXT:    [[SHR_7:%.*]] = lshr i16 [[A_SROA_7_8_EXTRACT_TRUNC]], 1
-; SSE4-NEXT:    [[SHR5_6:%.*]] = lshr i16 [[B_SROA_8_8_EXTRACT_TRUNC]], 1
-; SSE4-NEXT:    [[SHR5_7:%.*]] = lshr i16 [[B_SROA_9_8_EXTRACT_TRUNC]], 1
-; SSE4-NEXT:    [[NARROW_6:%.*]] = add nuw i16 [[SHR5_6]], [[SHR_6]]
-; SSE4-NEXT:    [[NARROW_7:%.*]] = add nuw i16 [[SHR5_7]], [[SHR_7]]
-; SSE4-NEXT:    [[TMP40:%.*]] = trunc i64 [[B_COERCE1]] to i16
-; SSE4-NEXT:    [[TMP41:%.*]] = insertelement <2 x i16> poison, i16 [[TMP40]], i64 0
-; SSE4-NEXT:    [[TMP42:%.*]] = trunc i64 [[A_SROA_7_8_EXTRACT_SHIFT]] to i16
-; SSE4-NEXT:    [[TMP27:%.*]] = insertelement <2 x i16> [[TMP41]], i16 [[TMP42]], i64 1
-; SSE4-NEXT:    [[TMP28:%.*]] = trunc i64 [[B_COERCE2]] to i16
-; SSE4-NEXT:    [[TMP29:%.*]] = insertelement <2 x i16> poison, i16 [[TMP28]], i64 0
-; SSE4-NEXT:    [[TMP45:%.*]] = trunc i64 [[B_SROA_7_8_EXTRACT_SHIFT]] to i16
-; SSE4-NEXT:    [[TMP46:%.*]] = insertelement <2 x i16> [[TMP29]], i16 [[TMP45]], i64 1
-; SSE4-NEXT:    [[TMP32:%.*]] = lshr <2 x i16> [[TMP27]], splat (i16 1)
-; SSE4-NEXT:    [[TMP47:%.*]] = lshr <2 x i16> [[TMP46]], splat (i16 1)
-; SSE4-NEXT:    [[TMP34:%.*]] = add nuw <2 x i16> [[TMP47]], [[TMP32]]
-; SSE4-NEXT:    [[TMP35:%.*]] = shufflevector <2 x i16> [[TMP46]], <2 x i16> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
-; SSE4-NEXT:    [[TMP36:%.*]] = insertelement <4 x i16> [[TMP35]], i16 [[B_SROA_9_8_EXTRACT_TRUNC]], i64 2
-; SSE4-NEXT:    [[TMP20:%.*]] = insertelement <4 x i16> [[TMP36]], i16 [[B_SROA_8_8_EXTRACT_TRUNC]], i64 3
-; SSE4-NEXT:    [[TMP22:%.*]] = shufflevector <2 x i16> [[TMP27]], <2 x i16> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
-; SSE4-NEXT:    [[TMP23:%.*]] = insertelement <4 x i16> [[TMP22]], i16 [[A_SROA_7_8_EXTRACT_TRUNC]], i64 2
-; SSE4-NEXT:    [[TMP24:%.*]] = insertelement <4 x i16> [[TMP23]], i16 [[A_SROA_5_8_EXTRACT_TRUNC]], i64 3
-; SSE4-NEXT:    [[TMP25:%.*]] = or <4 x i16> [[TMP20]], [[TMP24]]
-; SSE4-NEXT:    [[TMP26:%.*]] = and <4 x i16> [[TMP25]], splat (i16 1)
-; SSE4-NEXT:    [[TMP43:%.*]] = shufflevector <2 x i16> [[TMP34]], <2 x i16> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
-; SSE4-NEXT:    [[TMP44:%.*]] = insertelement <4 x i16> [[TMP43]], i16 [[NARROW_7]], i64 2
-; SSE4-NEXT:    [[TMP30:%.*]] = insertelement <4 x i16> [[TMP44]], i16 [[NARROW_6]], i64 3
-; SSE4-NEXT:    [[TMP31:%.*]] = add nuw <4 x i16> [[TMP30]], [[TMP26]]
-; SSE4-NEXT:    [[TMP33:%.*]] = bitcast <4 x i16> [[TMP31]] to i64
-; SSE4-NEXT:    [[DOTFCA_1_INSERT:%.*]] = insertvalue { i64, i64 } [[DOTFCA_0_INSERT]], i64 [[TMP33]], 1
-; SSE4-NEXT:    ret { i64, i64 } [[DOTFCA_1_INSERT]]
-;
-; AVX-LABEL: @avgr_8_u16_alt(
-; AVX-NEXT:  entry:
-; AVX-NEXT:    [[TMP0:%.*]] = insertelement <4 x i64> poison, i64 [[A_COERCE0:%.*]], i64 0
-; AVX-NEXT:    [[TMP1:%.*]] = shufflevector <4 x i64> [[TMP0]], <4 x i64> poison, <4 x i32> zeroinitializer
-; AVX-NEXT:    [[TMP2:%.*]] = lshr <4 x i64> [[TMP1]], <i64 0, i64 16, i64 32, i64 48>
-; AVX-NEXT:    [[TMP3:%.*]] = trunc <4 x i64> [[TMP2]] to <4 x i16>
-; AVX-NEXT:    [[TMP4:%.*]] = insertelement <4 x i64> poison, i64 [[B_COERCE0:%.*]], i64 0
-; AVX-NEXT:    [[TMP5:%.*]] = shufflevector <4 x i64> [[TMP4]], <4 x i64> poison, <4 x i32> zeroinitializer
-; AVX-NEXT:    [[TMP6:%.*]] = lshr <4 x i64> [[TMP5]], <i64 0, i64 16, i64 32, i64 48>
-; AVX-NEXT:    [[TMP7:%.*]] = trunc <4 x i64> [[TMP6]] to <4 x i16>
-; AVX-NEXT:    [[TMP8:%.*]] = lshr <4 x i16> [[TMP3]], splat (i16 1)
-; AVX-NEXT:    [[TMP9:%.*]] = lshr <4 x i16> [[TMP7]], splat (i16 1)
-; AVX-NEXT:    [[TMP10:%.*]] = add nuw <4 x i16> [[TMP9]], [[TMP8]]
-; AVX-NEXT:    [[TMP11:%.*]] = or <4 x i16> [[TMP7]], [[TMP3]]
-; AVX-NEXT:    [[TMP12:%.*]] = and <4 x i16> [[TMP11]], splat (i16 1)
-; AVX-NEXT:    [[TMP13:%.*]] = add nuw <4 x i16> [[TMP10]], [[TMP12]]
-; AVX-NEXT:    [[TMP14:%.*]] = bitcast <4 x i16> [[TMP13]] to i64
-; AVX-NEXT:    [[DOTFCA_0_INSERT:%.*]] = insertvalue { i64, i64 } poison, i64 [[TMP14]], 0
-; AVX-NEXT:    [[TMP15:%.*]] = insertelement <4 x i64> poison, i64 [[A_COERCE1:%.*]], i64 0
-; AVX-NEXT:    [[TMP16:%.*]] = shufflevector <4 x i64> [[TMP15]], <4 x i64> poison, <4 x i32> zeroinitializer
-; AVX-NEXT:    [[TMP17:%.*]] = lshr <4 x i64> [[TMP16]], <i64 0, i64 16, i64 32, i64 48>
-; AVX-NEXT:    [[TMP18:%.*]] = trunc <4 x i64> [[TMP17]] to <4 x i16>
-; AVX-NEXT:    [[TMP19:%.*]] = insertelement <4 x i64> poison, i64 [[B_COERCE1:%.*]], i64 0
-; AVX-NEXT:    [[TMP20:%.*]] = shufflevector <4 x i64> [[TMP19]], <4 x i64> poison, <4 x i32> zeroinitializer
-; AVX-NEXT:    [[TMP21:%.*]] = lshr <4 x i64> [[TMP20]], <i64 0, i64 16, i64 32, i64 48>
-; AVX-NEXT:    [[TMP22:%.*]] = trunc <4 x i64> [[TMP21]] to <4 x i16>
-; AVX-NEXT:    [[TMP23:%.*]] = lshr <4 x i16> [[TMP18]], splat (i16 1)
-; AVX-NEXT:    [[TMP24:%.*]] = lshr <4 x i16> [[TMP22]], splat (i16 1)
-; AVX-NEXT:    [[TMP25:%.*]] = add nuw <4 x i16> [[TMP24]], [[TMP23]]
-; AVX-NEXT:    [[TMP26:%.*]] = or <4 x i16> [[TMP22]], [[TMP18]]
-; AVX-NEXT:    [[TMP27:%.*]] = and <4 x i16> [[TMP26]], splat (i16 1)
-; AVX-NEXT:    [[TMP28:%.*]] = add nuw <4 x i16> [[TMP25]], [[TMP27]]
-; AVX-NEXT:    [[TMP29:%.*]] = bitcast <4 x i16> [[TMP28]] to i64
-; AVX-NEXT:    [[DOTFCA_1_INSERT:%.*]] = insertvalue { i64, i64 } [[DOTFCA_0_INSERT]], i64 [[TMP29]], 1
-; AVX-NEXT:    ret { i64, i64 } [[DOTFCA_1_INSERT]]
+; CHECK-LABEL: @avgr_8_u16_alt(
+; CHECK-NEXT:  entry:
+; CHECK-NEXT:    [[TMP0:%.*]] = bitcast i64 [[A_COERCE0:%.*]] to <4 x i16>
+; CHECK-NEXT:    [[TMP1:%.*]] = lshr <4 x i16> [[TMP0]], splat (i16 1)
+; CHECK-NEXT:    [[TMP2:%.*]] = bitcast i64 [[B_COERCE0:%.*]] to <4 x i16>
+; CHECK-NEXT:    [[TMP3:%.*]] = lshr <4 x i16> [[TMP2]], splat (i16 1)
+; CHECK-NEXT:    [[TMP4:%.*]] = add nuw <4 x i16> [[TMP3]], [[TMP1]]
+; CHECK-NEXT:    [[TMP5:%.*]] = or <4 x i16> [[TMP2]], [[TMP0]]
+; CHECK-NEXT:    [[TMP6:%.*]] = and <4 x i16> [[TMP5]], splat (i16 1)
+; CHECK-NEXT:    [[TMP7:%.*]] = add nuw <4 x i16> [[TMP4]], [[TMP6]]
+; CHECK-NEXT:    [[TMP8:%.*]] = bitcast <4 x i16> [[TMP7]] to i64
+; CHECK-NEXT:    [[DOTFCA_0_INSERT:%.*]] = insertvalue { i64, i64 } poison, i64 [[TMP8]], 0
+; CHECK-NEXT:    [[TMP9:%.*]] = bitcast i64 [[A_COERCE1:%.*]] to <4 x i16>
+; CHECK-NEXT:    [[TMP10:%.*]] = lshr <4 x i16> [[TMP9]], splat (i16 1)
+; CHECK-NEXT:    [[TMP11:%.*]] = bitcast i64 [[B_COERCE1:%.*]] to <4 x i16>
+; CHECK-NEXT:    [[TMP12:%.*]] = lshr <4 x i16> [[TMP11]], splat (i16 1)
+; CHECK-NEXT:    [[TMP13:%.*]] = add nuw <4 x i16> [[TMP12]], [[TMP10]]
+; CHECK-NEXT:    [[TMP14:%.*]] = or <4 x i16> [[TMP11]], [[TMP9]]
+; CHECK-NEXT:    [[TMP15:%.*]] = and <4 x i16> [[TMP14]], splat (i16 1)
+; CHECK-NEXT:    [[TMP16:%.*]] = add nuw <4 x i16> [[TMP13]], [[TMP15]]
+; CHECK-NEXT:    [[TMP17:%.*]] = bitcast <4 x i16> [[TMP16]] to i64
+; CHECK-NEXT:    [[DOTFCA_1_INSERT:%.*]] = insertvalue { i64, i64 } [[DOTFCA_0_INSERT]], i64 [[TMP17]], 1
+; CHECK-NEXT:    ret { i64, i64 } [[DOTFCA_1_INSERT]]
 ;
 entry:
   %retval = alloca %"struct.std::array8", align 2
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/extracted-subfields.ll b/llvm/test/Transforms/SLPVectorizer/X86/extracted-subfields.ll
index 7c58c1881aa515..f025d672aa2688 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/extracted-subfields.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/extracted-subfields.ll
@@ -10,34 +10,9 @@ define i16 @sum8_i64(ptr %x) {
 ; CHECK-SAME: ptr [[X:%.*]]) {
 ; CHECK-NEXT:  [[START:.*:]]
 ; CHECK-NEXT:    [[L:%.*]] = load i64, ptr [[X]], align 8
-; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[L]] to i16
-; CHECK-NEXT:    [[B0:%.*]] = and i16 [[T0]], 255
-; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[L]] to i16
-; CHECK-NEXT:    [[B1:%.*]] = lshr i16 [[T1]], 8
-; CHECK-NEXT:    [[A1:%.*]] = add nuw nsw i16 [[B0]], [[B1]]
-; CHECK-NEXT:    [[S2:%.*]] = lshr i64 [[L]], 16
-; CHECK-NEXT:    [[T2:%.*]] = trunc i64 [[S2]] to i16
-; CHECK-NEXT:    [[B2:%.*]] = and i16 [[T2]], 255
-; CHECK-NEXT:    [[A2:%.*]] = add nuw nsw i16 [[A1]], [[B2]]
-; CHECK-NEXT:    [[S3:%.*]] = lshr i64 [[L]], 24
-; CHECK-NEXT:    [[T3:%.*]] = trunc i64 [[S3]] to i16
-; CHECK-NEXT:    [[B3:%.*]] = and i16 [[T3]], 255
-; CHECK-NEXT:    [[A3:%.*]] = add nuw nsw i16 [[A2]], [[B3]]
-; CHECK-NEXT:    [[S4:%.*]] = lshr i64 [[L]], 32
-; CHECK-NEXT:    [[T4:%.*]] = trunc i64 [[S4]] to i16
-; CHECK-NEXT:    [[B4:%.*]] = and i16 [[T4]], 255
-; CHECK-NEXT:    [[A4:%.*]] = add nuw nsw i16 [[A3]], [[B4]]
-; CHECK-NEXT:    [[S5:%.*]] = lshr i64 [[L]], 40
-; CHECK-NEXT:    [[T5:%.*]] = trunc i64 [[S5]] to i16
-; CHECK-NEXT:    [[B5:%.*]] = and i16 [[T5]], 255
-; CHECK-NEXT:    [[A5:%.*]] = add nuw nsw i16 [[A4]], [[B5]]
-; CHECK-NEXT:    [[S6:%.*]] = lshr i64 [[L]], 48
-; CHECK-NEXT:    [[T6:%.*]] = trunc i64 [[S6]] to i16
-; CHECK-NEXT:    [[B6:%.*]] = and i16 [[T6]], 255
-; CHECK-NEXT:    [[A6:%.*]] = add nuw nsw i16 [[A5]], [[B6]]
-; CHECK-NEXT:    [[S7:%.*]] = lshr i64 [[L]], 56
-; CHECK-NEXT:    [[B7:%.*]] = trunc i64 [[S7]] to i16
-; CHECK-NEXT:    [[TMP2:%.*]] = add nuw nsw i16 [[A6]], [[B7]]
+; CHECK-NEXT:    [[TMP0:%.*]] = bitcast i64 [[L]] to <8 x i8>
+; CHECK-NEXT:    [[TMP1:%.*]] = zext <8 x i8> [[TMP0]] to <8 x i16>
+; CHECK-NEXT:    [[TMP2:%.*]] = call i16 @llvm.vector.reduce.add.v8i16(<8 x i16> [[TMP1]])
 ; CHECK-NEXT:    ret i16 [[TMP2]]
 ;
 start:
@@ -81,30 +56,8 @@ define i16 @sum8_i64_ashr(ptr %x) {
 ; CHECK-SAME: ptr [[X:%.*]]) {
 ; CHECK-NEXT:  [[START:.*:]]
 ; CHECK-NEXT:    [[L:%.*]] = load i64, ptr [[X]], align 8
-; CHECK-NEXT:    [[S7:%.*]] = ashr i64 [[L]], 56
-; CHECK-NEXT:    [[S6:%.*]] = ashr i64 [[L]], 48
-; CHECK-NEXT:    [[S5:%.*]] = ashr i64 [[L]], 40
-; CHECK-NEXT:    [[S4:%.*]] = ashr i64 [[L]], 32
-; CHECK-NEXT:    [[S3:%.*]] = ashr i64 [[L]], 24
-; CHECK-NEXT:    [[S2:%.*]] = ashr i64 [[L]], 16
-; CHECK-NEXT:    [[S1:%.*]] = ashr i64 [[L]], 8
-; CHECK-NEXT:    [[T7:%.*]] = trunc i64 [[S7]] to i16
-; CHECK-NEXT:    [[T6:%.*]] = trunc i64 [[S6]] to i16
-; CHECK-NEXT:    [[T5:%.*]] = trunc i64 [[S5]] to i16
-; CHECK-NEXT:    [[T4:%.*]] = trunc i64 [[S4]] to i16
-; CHECK-NEXT:    [[T3:%.*]] = trunc i64 [[S3]] to i16
-; CHECK-NEXT:    [[T2:%.*]] = trunc i64 [[S2]] to i16
-; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[S1]] to i16
-; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[L]] to i16
-; CHECK-NEXT:    [[TMP0:%.*]] = insertelement <8 x i16> poison, i16 [[T0]], i64 0
-; CHECK-NEXT:    [[TMP8:%.*]] = insertelement <8 x i16> [[TMP0]], i16 [[T1]], i64 1
-; CHECK-NEXT:    [[TMP9:%.*]] = insertelement <8 x i16> [[TMP8]], i16 [[T2]], i64 2
-; CHECK-NEXT:    [[TMP3:%.*]] = insertelement <8 x i16> [[TMP9]], i16 [[T3]], i64 3
-; CHECK-NEXT:    [[TMP4:%.*]] = insertelement <8 x i16> [[TMP3]], i16 [[T4]], i64 4
-; CHECK-NEXT:    [[TMP5:%.*]] = insertelement <8 x i16> [[TMP4]], i16 [[T5]], i64 5
-; CHECK-NEXT:    [[TMP6:%.*]] = insertelement <8 x i16> [[TMP5]], i16 [[T6]], i64 6
-; CHECK-NEXT:    [[TMP7:%.*]] = insertelement <8 x i16> [[TMP6]], i16 [[T7]], i64 7
-; CHECK-NEXT:    [[TMP1:%.*]] = and <8 x i16> [[TMP7]], splat (i16 255)
+; CHECK-NEXT:    [[TMP0:%.*]] = bitcast i64 [[L]] to <8 x i8>
+; CHECK-NEXT:    [[TMP1:%.*]] = zext <8 x i8> [[TMP0]] to <8 x i16>
 ; CHECK-NEXT:    [[TMP2:%.*]] = call i16 @llvm.vector.reduce.add.v8i16(<8 x i16> [[TMP1]])
 ; CHECK-NEXT:    ret i16 [[TMP2]]
 ;
@@ -151,19 +104,9 @@ define i32 @sum4_halves_i64(ptr %p) {
 ; CHECK-SAME: ptr [[P:%.*]]) {
 ; CHECK-NEXT:  [[START:.*:]]
 ; CHECK-NEXT:    [[L:%.*]] = load i64, ptr [[P]], align 8
-; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[L]] to i32
-; CHECK-NEXT:    [[B0:%.*]] = and i32 [[T0]], 65535
-; CHECK-NEXT:    [[S1:%.*]] = lshr i64 [[L]], 16
-; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[S1]] to i32
-; CHECK-NEXT:    [[B1:%.*]] = and i32 [[T1]], 65535
-; CHECK-NEXT:    [[S2:%.*]] = lshr i64 [[L]], 32
-; CHECK-NEXT:    [[T2:%.*]] = trunc i64 [[S2]] to i32
-; CHECK-NEXT:    [[B2:%.*]] = and i32 [[T2]], 65535
-; CHECK-NEXT:    [[S3:%.*]] = lshr i64 [[L]], 48
-; CHECK-NEXT:    [[B3:%.*]] = trunc i64 [[S3]] to i32
-; CHECK-NEXT:    [[A1:%.*]] = add nuw nsw i32 [[B0]], [[B1]]
-; CHECK-NEXT:    [[A2:%.*]] = add nuw nsw i32 [[A1]], [[B2]]
-; CHECK-NEXT:    [[A3:%.*]] = add nuw nsw i32 [[A2]], [[B3]]
+; CHECK-NEXT:    [[TMP0:%.*]] = bitcast i64 [[L]] to <4 x i16>
+; CHECK-NEXT:    [[TMP1:%.*]] = zext <4 x i16> [[TMP0]] to <4 x i32>
+; CHECK-NEXT:    [[A3:%.*]] = call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> [[TMP1]])
 ; CHECK-NEXT:    ret i32 [[A3]]
 ;
 start:
@@ -192,19 +135,9 @@ define i16 @sum4_i32(ptr %p) {
 ; CHECK-SAME: ptr [[P:%.*]]) {
 ; CHECK-NEXT:  [[START:.*:]]
 ; CHECK-NEXT:    [[L:%.*]] = load i32, ptr [[P]], align 4
-; CHECK-NEXT:    [[T0:%.*]] = trunc i32 [[L]] to i16
-; CHECK-NEXT:    [[B0:%.*]] = and i16 [[T0]], 255
-; CHECK-NEXT:    [[S1:%.*]] = lshr i32 [[L]], 8
-; CHECK-NEXT:    [[T1:%.*]] = trunc i32 [[S1]] to i16
-; CHECK-NEXT:    [[B1:%.*]] = and i16 [[T1]], 255
-; CHECK-NEXT:    [[S2:%.*]] = lshr i32 [[L]], 16
-; CHECK-NEXT:    [[T2:%.*]] = trunc i32 [[S2]] to i16
-; CHECK-NEXT:    [[B2:%.*]] = and i16 [[T2]], 255
-; CHECK-NEXT:    [[S3:%.*]] = lshr i32 [[L]], 24
-; CHECK-NEXT:    [[B3:%.*]] = trunc i32 [[S3]] to i16
-; CHECK-NEXT:    [[A1:%.*]] = add nuw nsw i16 [[B0]], [[B1]]
-; CHECK-NEXT:    [[A2:%.*]] = add nuw nsw i16 [[A1]], [[B2]]
-; CHECK-NEXT:    [[A3:%.*]] = add nuw nsw i16 [[A2]], [[B3]]
+; CHECK-NEXT:    [[TMP0:%.*]] = bitcast i32 [[L]] to <4 x i8>
+; CHECK-NEXT:    [[TMP1:%.*]] = zext <4 x i8> [[TMP0]] to <4 x i16>
+; CHECK-NEXT:    [[A3:%.*]] = call i16 @llvm.vector.reduce.add.v4i16(<4 x i16> [[TMP1]])
 ; CHECK-NEXT:    ret i16 [[A3]]
 ;
 start:
@@ -272,23 +205,10 @@ define void @store_bytes_i64(ptr %p, ptr %out) {
 ; CHECK-SAME: ptr [[P:%.*]], ptr [[OUT:%.*]]) {
 ; CHECK-NEXT:  [[START:.*:]]
 ; CHECK-NEXT:    [[L:%.*]] = load i64, ptr [[P]], align 8
-; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[L]] to i16
-; CHECK-NEXT:    [[B0:%.*]] = and i16 [[T0]], 255
-; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[L]] to i16
-; CHECK-NEXT:    [[B1:%.*]] = lshr i16 [[T1]], 8
-; CHECK-NEXT:    [[S2:%.*]] = lshr i64 [[L]], 16
-; CHECK-NEXT:    [[T2:%.*]] = trunc i64 [[S2]] to i16
-; CHECK-NEXT:    [[B2:%.*]] = and i16 [[T2]], 255
-; CHECK-NEXT:    [[S3:%.*]] = lshr i64 [[L]], 24
-; CHECK-NEXT:    [[T3:%.*]] = trunc i64 [[S3]] to i16
-; CHECK-NEXT:    [[B3:%.*]] = and i16 [[T3]], 255
-; CHECK-NEXT:    store i16 [[B0]], ptr [[OUT]], align 2
-; CHECK-NEXT:    [[O1:%.*]] = getelementptr i16, ptr [[OUT]], i64 1
-; CHECK-NEXT:    store i16 [[B1]], ptr [[O1]], align 2
-; CHECK-NEXT:    [[O2:%.*]] = getelementptr i16, ptr [[OUT]], i64 2
-; CHECK-NEXT:    store i16 [[B2]], ptr [[O2]], align 2
-; CHECK-NEXT:    [[O3:%.*]] = getelementptr i16, ptr [[OUT]], i64 3
-; CHECK-NEXT:    store i16 [[B3]], ptr [[O3]], align 2
+; CHECK-NEXT:    [[TMP0:%.*]] = bitcast i64 [[L]] to <8 x i8>
+; CHECK-NEXT:    [[TMP1:%.*]] = shufflevector <8 x i8> [[TMP0]], <8 x i8> poison, <4 x i32> <i32 0, i32 1, i32 2, i32 3>
+; CHECK-NEXT:    [[TMP2:%.*]] = zext <4 x i8> [[TMP1]] to <4 x i16>
+; CHECK-NEXT:    store <4 x i16> [[TMP2]], ptr [[OUT]], align 2
 ; CHECK-NEXT:    ret void
 ;
 start:
@@ -320,23 +240,9 @@ define void @store_halves_i64(ptr %p, ptr %out) {
 ; CHECK-SAME: ptr [[P:%.*]], ptr [[OUT:%.*]]) {
 ; CHECK-NEXT:  [[START:.*:]]
 ; CHECK-NEXT:    [[L:%.*]] = load i64, ptr [[P]], align 8
-; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[L]] to i32
-; CHECK-NEXT:    [[B0:%.*]] = and i32 [[T0]], 65535
-; CHECK-NEXT:    [[S1:%.*]] = lshr i64 [[L]], 16
-; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[S1]] to i32
-; CHECK-NEXT:    [[B1:%.*]] = and i32 [[T1]], 65535
-; CHECK-NEXT:    [[S2:%.*]] = lshr i64 [[L]], 32
-; CHECK-NEXT:    [[T2:%.*]] = trunc i64 [[S2]] to i32
-; CHECK-NEXT:    [[B2:%.*]] = and i32 [[T2]], 65535
-; CHECK-NEXT:    [[S3:%.*]] = lshr i64 [[L]], 48
-; CHECK-NEXT:    [[B3:%.*]] = trunc i64 [[S3]] to i32
-; CHECK-NEXT:    store i32 [[B0]], ptr [[OUT]], align 4
-; CHECK-NEXT:    [[O1:%.*]] = getelementptr i32, ptr [[OUT]], i64 1
-; CHECK-NEXT:    store i32 [[B1]], ptr [[O1]], align 4
-; CHECK-NEXT:    [[O2:%.*]] = getelementptr i32, ptr [[OUT]], i64 2
-; CHECK-NEXT:    store i32 [[B2]], ptr [[O2]], align 4
-; CHECK-NEXT:    [[O3:%.*]] = getelementptr i32, ptr [[OUT]], i64 3
-; CHECK-NEXT:    store i32 [[B3]], ptr [[O3]], align 4
+; CHECK-NEXT:    [[TMP0:%.*]] = bitcast i64 [[L]] to <4 x i16>
+; CHECK-NEXT:    [[TMP1:%.*]] = zext <4 x i16> [[TMP0]] to <4 x i32>
+; CHECK-NEXT:    store <4 x i32> [[TMP1]], ptr [[OUT]], align 4
 ; CHECK-NEXT:    ret void
 ;
 start:
@@ -392,30 +298,8 @@ define i16 @sum8_i64_scrambled(ptr %x) {
 ; CHECK-SAME: ptr [[X:%.*]]) {
 ; CHECK-NEXT:  [[START:.*:]]
 ; CHECK-NEXT:    [[L:%.*]] = load i64, ptr [[X]], align 8
-; CHECK-NEXT:    [[S7:%.*]] = lshr i64 [[L]], 56
-; CHECK-NEXT:    [[S4:%.*]] = lshr i64 [[L]], 32
-; CHECK-NEXT:    [[S6:%.*]] = lshr i64 [[L]], 48
-; CHECK-NEXT:    [[S2:%.*]] = lshr i64 [[L]], 16
-; CHECK-NEXT:    [[S1:%.*]] = lshr i64 [[L]], 8
-; CHECK-NEXT:    [[S5:%.*]] = lshr i64 [[L]], 40
-; CHECK-NEXT:    [[S3:%.*]] = lshr i64 [[L]], 24
-; CHECK-NEXT:    [[B7:%.*]] = trunc i64 [[S7]] to i16
-; CHECK-NEXT:    [[T4:%.*]] = trunc i64 [[S4]] to i16
-; CHECK-NEXT:    [[T6:%.*]] = trunc i64 [[S6]] to i16
-; CHECK-NEXT:    [[T2:%.*]] = trunc i64 [[S2]] to i16
-; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[S1]] to i16
-; CHECK-NEXT:    [[T5:%.*]] = trunc i64 [[S5]] to i16
-; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[L]] to i16
-; CHECK-NEXT:    [[T3:%.*]] = trunc i64 [[S3]] to i16
-; CHECK-NEXT:    [[TMP0:%.*]] = insertelement <8 x i16> poison, i16 [[T3]], i64 0
-; CHECK-NEXT:    [[TMP8:%.*]] = insertelement <8 x i16> [[TMP0]], i16 [[T0]], i64 1
-; CHECK-NEXT:    [[TMP9:%.*]] = insertelement <8 x i16> [[TMP8]], i16 [[T5]], i64 2
-; CHECK-NEXT:    [[TMP3:%.*]] = insertelement <8 x i16> [[TMP9]], i16 [[T1]], i64 3
-; CHECK-NEXT:    [[TMP4:%.*]] = insertelement <8 x i16> [[TMP3]], i16 [[T2]], i64 4
-; CHECK-NEXT:    [[TMP5:%.*]] = insertelement <8 x i16> [[TMP4]], i16 [[T6]], i64 5
-; CHECK-NEXT:    [[TMP6:%.*]] = insertelement <8 x i16> [[TMP5]], i16 [[T4]], i64 6
-; CHECK-NEXT:    [[TMP7:%.*]] = insertelement <8 x i16> [[TMP6]], i16 [[B7]], i64 7
-; CHECK-NEXT:    [[TMP1:%.*]] = and <8 x i16> [[TMP7]], <i16 255, i16 255, i16 255, i16 255, i16 255, i16 255, i16 255, i16 -1>
+; CHECK-NEXT:    [[TMP0:%.*]] = bitcast i64 [[L]] to <8 x i8>
+; CHECK-NEXT:    [[TMP1:%.*]] = zext <8 x i8> [[TMP0]] to <8 x i16>
 ; CHECK-NEXT:    [[TMP2:%.*]] = call i16 @llvm.vector.reduce.add.v8i16(<8 x i16> [[TMP1]])
 ; CHECK-NEXT:    ret i16 [[TMP2]]
 ;
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/reduce-or-bitpack.ll b/llvm/test/Transforms/SLPVectorizer/X86/reduce-or-bitpack.ll
index a5c51777871902..80c0c0c909798e 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/reduce-or-bitpack.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/reduce-or-bitpack.ll
@@ -8,11 +8,7 @@ define i32 @shl_and_pack(i32 %x) {
 ; CHECK-LABEL: define i32 @shl_and_pack(
 ; CHECK-SAME: i32 [[X:%.*]]) #[[ATTR0:[0-9]+]] {
 ; CHECK-NEXT:  [[ENTRY:.*:]]
-; CHECK-NEXT:    [[TMP0:%.*]] = insertelement <4 x i32> poison, i32 [[X]], i64 0
-; CHECK-NEXT:    [[TMP1:%.*]] = shufflevector <4 x i32> [[TMP0]], <4 x i32> poison, <4 x i32> zeroinitializer
-; CHECK-NEXT:    [[TMP2:%.*]] = lshr <4 x i32> [[TMP1]], <i32 0, i32 8, i32 16, i32 24>
-; CHECK-NEXT:    [[TMP3:%.*]] = and <4 x i32> [[TMP2]], <i32 255, i32 255, i32 255, i32 -1>
-; CHECK-NEXT:    [[TMP4:%.*]] = trunc <4 x i32> [[TMP3]] to <4 x i8>
+; CHECK-NEXT:    [[TMP4:%.*]] = bitcast i32 [[X]] to <4 x i8>
 ; CHECK-NEXT:    [[TMP6:%.*]] = bitcast <4 x i8> [[TMP4]] to i32
 ; CHECK-NEXT:    ret i32 [[TMP6]]
 ;



More information about the llvm-commits mailing list