[llvm] [SLP]Model gathers of extracted integer sub-fields as bitcast+permute+ext (PR #224919)
Alexey Bataev via llvm-commits
llvm-commits at lists.llvm.org
Thu Sep 24 06:30:18 PDT 2026
https://github.com/alexey-bataev updated https://github.com/llvm/llvm-project/pull/224919
>From 18397d8949537b75c58d58a25a94afedeaa4cfa0 Mon Sep 17 00:00:00 2001
From: Alexey Bataev <a.bataev at outlook.com>
Date: Sun, 20 Sep 2026 04:53:37 -0700
Subject: [PATCH] =?UTF-8?q?[=F0=9D=98=80=F0=9D=97=BD=F0=9D=97=BF]=20initia?=
=?UTF-8?q?l=20version?=
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
Created using spr 1.3.7
---
.../Transforms/Vectorize/SLPVectorizer.cpp | 130 +++-
.../Vectorize/SLPVectorizer/SLPUtils.cpp | 133 ++++
.../Vectorize/SLPVectorizer/SLPUtils.h | 9 +
.../AArch64/scalarize-load-ext-extract.ll | 7 +-
llvm/test/Transforms/PhaseOrdering/X86/avg.ll | 722 +++---------------
.../SLPVectorizer/X86/extracted-subfields.ll | 156 +---
.../SLPVectorizer/X86/reduce-or-bitpack.ll | 6 +-
7 files changed, 416 insertions(+), 747 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 205311d10e76c2..dfc2fe49b8cee6 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -2552,6 +2552,15 @@ class slpvectorizer::BoUpSLP {
template <typename BVTy, typename ResTy, typename... Args>
ResTy processBuildVector(const TreeEntry *E, Type *ScalarTy, Args &...Params);
+ /// Handles the gather of zero-extended sub-fields of the same wider integer
+ /// scalar, emitted as a bitcast to the field vector plus a permutation and
+ /// an extension to the lane type, if the gathered values match.
+ template <typename ResTy, typename BVTy>
+ std::optional<ResTy> processExtractedFieldsGather(
+ BVTy &ShuffleBuilder, const TreeEntry *E, Type *ScalarTy,
+ ArrayRef<Value *> VL,
+ ArrayRef<std::pair<const TreeEntry *, unsigned>> SubVectors);
+
/// Create a new vector from a list of scalar values. Produces a sequence
/// which exploits values reused across lanes, and arranges the inserts
/// for ease of later optimization.
@@ -12081,7 +12090,12 @@ void BoUpSLP::buildTreeRec(ArrayRef<Value *> VLRef, unsigned Depth,
TreeEntry::EntryState State = getScalarsVectorizationState(
S, VL, IsScatterVectorizeUserTE, CurrentOrder, PointerOps, SPtrInfo,
ExpandShuffleMask);
- if (State == TreeEntry::NeedToGather) {
+ // The zero-extended sub-fields of the same wider integer scalar are
+ // gathered and emitted as a bitcast to the field vector plus a permutation.
+ // For 2 lanes the emission is not cheaper than the insertelement chain.
+ if (State == TreeEntry::NeedToGather ||
+ (VL.size() > 2 && ReuseShuffleIndices.empty() &&
+ matchGatheredExtractedFields(VL, *DL))) {
newGatherTreeEntry(VL, S, UserTreeIdx, ReuseShuffleIndices);
return;
}
@@ -14145,7 +14159,9 @@ void BoUpSLP::transformNodes() {
// We use allSameOpcode instead of isAltShuffle because we don't
// want to use interchangeable instruction here.
!allSameOpcode(VL) || !allSameBlock(VL)) ||
- allConstant(VL) || isSplat(VL))
+ allConstant(VL) || isSplat(VL) ||
+ (E.ReuseShuffleIndices.empty() &&
+ matchGatheredExtractedFields(VL, *DL)))
continue;
if (ForceLoadGather && E.hasState() && E.getOpcode() == Instruction::Load)
continue;
@@ -15527,6 +15543,43 @@ class BoUpSLP::ShuffleCostEstimator : public BaseShuffleAnalysis {
cast<FixedVectorType>(Root->getType())->getNumElements()),
getAllOnesValue(*R.DL, ScalarTy->getScalarType()));
}
+ /// Estimates the cost of the gather of zero-extended sub-fields of width
+ /// \p FieldWidth of the same wider integer scalar \p Src, permuted by
+ /// \p Mask, as a bitcast to the field vector plus a permutation and an
+ /// extension to the lane type.
+ InstructionCost createExtractedFieldsVector(Value *Src, unsigned FieldWidth,
+ ArrayRef<int> Mask,
+ const TreeEntry &E) {
+ auto *FieldTy = IntegerType::get(Src->getContext(), FieldWidth);
+ unsigned SrcWidth = Src->getType()->getIntegerBitWidth();
+ assert(SrcWidth % FieldWidth == 0 &&
+ "Expected the field width to divide the source width.");
+ unsigned NumFields = SrcWidth / FieldWidth;
+ auto *FieldVecTy = cast<VectorType>(getWidenedType(FieldTy, NumFields));
+ TTI::CastContextHint CCH = R.getCastContextHint(E);
+ InstructionCost Cost = TTI.getCastInstrCost(
+ Instruction::BitCast, FieldVecTy, Src->getType(), CCH, CostKind);
+ // The permutation is elided only for the full-length identity mask.
+ if (Mask.size() != NumFields ||
+ !ShuffleVectorInst::isIdentityMask(Mask, NumFields))
+ Cost += getShuffleCost(TTI, TTI::SK_PermuteSingleSrc, FieldVecTy,
+ CostKind, Mask);
+ if (!ScalarTy->isIntegerTy(FieldWidth))
+ Cost += TTI.getCastInstrCost(
+ Instruction::ZExt, getWidenedType(ScalarTy, Mask.size()),
+ getWidenedType(FieldTy, Mask.size()), CCH, CostKind);
+ // The extraction instructions die once the fields are emitted from the
+ // source scalar; their scalar cost is credited back for the gathers
+ // directly feeding the reduction. Instructions vectorized elsewhere are
+ // skipped: their scalar cost is already a part of the scalar baseline.
+ if (R.UserIgnoreList &&
+ (!E.UserTreeIndex || E.UserTreeIndex.UserTE->Idx == 0))
+ for (Instruction *I : make_isa_range<Instruction>(E.Scalars))
+ if (CheckedExtracts.insert(I).second && !R.isVectorized(I) &&
+ R.areAllUsersVectorized(I, &VectorizedVals))
+ Cost -= TTI.getInstructionCost(I, CostKind);
+ return Cost;
+ }
InstructionCost createFreeze(InstructionCost Cost) { return Cost; }
/// Finalize emission of the shuffles.
InstructionCost finalize(
@@ -15664,6 +15717,17 @@ TTI::CastContextHint BoUpSLP::getCastContextHint(const TreeEntry &TE) const {
if (ShuffleVectorInst::isReverseMask(Mask, Mask.size()))
return TTI::CastContextHint::Reversed;
}
+ // A gather of extracted sub-fields inherits the context of the entry
+ // vectorizing the common source scalar, or of the source scalar load.
+ if (TE.isGather())
+ if (std::optional<std::tuple<Value *, unsigned, SmallVector<int>>> Fields =
+ matchGatheredExtractedFields(TE.Scalars, *DL)) {
+ Value *Src = std::get<0>(*Fields);
+ if (ArrayRef<TreeEntry *> TEs = getTreeEntries(Src); !TEs.empty())
+ return getCastContextHint(*TEs.front());
+ if (isa<LoadInst>(Src))
+ return TTI::CastContextHint::Normal;
+ }
return TTI::CastContextHint::None;
}
@@ -17458,6 +17522,7 @@ bool BoUpSLP::isFullyVectorizableTinyTree(bool ForReduction) const {
[this](Value *V) { return EphValues.contains(V); }) &&
(allConstant(TE->Scalars) || isSplat(TE->Scalars) ||
TE->Scalars.size() < Limit ||
+ matchGatheredExtractedFields(TE->Scalars, *DL) ||
// Nodes with copyable lanes may mix in non-extract lanes, which
// are not representable as a shuffle of the source vector.
(((TE->hasState() &&
@@ -22275,6 +22340,26 @@ class BoUpSLP::ShuffleInstructionBuilder final : public BaseShuffleAnalysis {
return createShuffle(V1, V2, Mask);
});
}
+ /// Emits the gather of zero-extended sub-fields of width \p FieldWidth of
+ /// the same wider integer scalar \p Src, permuted by \p Mask, as a bitcast
+ /// to the field vector plus a permutation and an extension to the lane type.
+ Value *createExtractedFieldsVector(Value *Src, unsigned FieldWidth,
+ ArrayRef<int> Mask, const TreeEntry &) {
+ auto *FieldTy = IntegerType::get(Src->getContext(), FieldWidth);
+ unsigned SrcWidth = Src->getType()->getIntegerBitWidth();
+ assert(SrcWidth % FieldWidth == 0 &&
+ "Expected the field width to divide the source width.");
+ Value *Vec = Builder.CreateBitCast(
+ Src, getWidenedType(FieldTy, SrcWidth / FieldWidth));
+ if (auto *I = dyn_cast<Instruction>(Vec)) {
+ R.GatherShuffleExtractSeq.insert(I);
+ R.CSEBlocks.insert(I->getParent());
+ }
+ Vec = createShuffle(Vec, /*V2=*/nullptr, Mask);
+ // The extracted fields are unsigned.
+ Vec = castToScalarTyElem(Vec, /*IsSigned=*/false);
+ return Vec;
+ }
Value *createFreeze(Value *V) { return Builder.CreateFreeze(V); }
/// Finalize emission of the shuffles.
/// \param Action the action (if any) to be performed before final applying of
@@ -22394,6 +22479,44 @@ Value *BoUpSLP::vectorizeOperand(TreeEntry *E, unsigned NodeIdx) {
return vectorizeTree(getOperandEntry(E, NodeIdx));
}
+template <typename ResTy, typename BVTy>
+std::optional<ResTy> BoUpSLP::processExtractedFieldsGather(
+ BVTy &ShuffleBuilder, const TreeEntry *E, Type *ScalarTy,
+ ArrayRef<Value *> VL,
+ ArrayRef<std::pair<const TreeEntry *, unsigned>> SubVectors) {
+ if (!SubVectors.empty() || !E->ReuseShuffleIndices.empty() ||
+ !ScalarTy->isIntegerTy())
+ return std::nullopt;
+ std::optional<std::tuple<Value *, unsigned, SmallVector<int>>>
+ ExtractedFields = matchGatheredExtractedFields(VL, *DL);
+ if (!ExtractedFields ||
+ ScalarTy->getIntegerBitWidth() < std::get<1>(*ExtractedFields))
+ return std::nullopt;
+ Value *Src = std::get<0>(*ExtractedFields);
+ unsigned FieldWidth = std::get<1>(*ExtractedFields);
+ SmallVector<int> &Mask = std::get<2>(*ExtractedFields);
+ unsigned NumFields = Src->getType()->getIntegerBitWidth() / FieldWidth;
+ // The lane order of the reduction root is unobservable when every gathered
+ // scalar is used only by the reduction operations and each lane holds a
+ // distinct field: emit the fields in the natural order and skip the
+ // permutation.
+ if (!E->UserTreeIndex && UserIgnoreList && Mask.size() == NumFields &&
+ all_of(make_isa_range<Instruction>(VL), [&](Instruction *I) {
+ return !I->hasNUsesOrMore(UsesLimit) &&
+ all_of(I->users(),
+ [&](User *U) { return UserIgnoreList->contains(U); });
+ })) {
+ SmallVector<int> SortedMask(Mask);
+ sort(SortedMask);
+ // Each field is held at most once; poison lanes are ignored.
+ if (adjacent_find(SortedMask, [](int A, int B) {
+ return A != PoisonMaskElem && A == B;
+ }) == SortedMask.end())
+ std::iota(Mask.begin(), Mask.end(), 0);
+ }
+ return ShuffleBuilder.createExtractedFieldsVector(Src, FieldWidth, Mask, *E);
+}
+
template <typename BVTy, typename ResTy, typename... Args>
ResTy BoUpSLP::processBuildVector(const TreeEntry *E, Type *ScalarTy,
Args &...Params) {
@@ -22505,6 +22628,9 @@ ResTy BoUpSLP::processBuildVector(const TreeEntry *E, Type *ScalarTy,
Type *OrigScalarTy = GatheredScalars.front()->getType();
auto *VecTy = getWidenedType(ScalarTy, GatheredScalars.size());
unsigned NumParts = getNumberOfParts(VecTy, ScalarTy, GatheredScalars.size());
+ if (std::optional<ResTy> Res = processExtractedFieldsGather<ResTy>(
+ ShuffleBuilder, E, ScalarTy, GatheredScalars, SubVectors))
+ return *Res;
if (!all_of(GatheredScalars, IsaPred<UndefValue>)) {
// Check for gathered extracts.
bool Resized = false;
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp
index cd6319fd6cce50..85ee4a6c0c4ab8 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp
@@ -1139,6 +1139,139 @@ TargetTransformInfo::TargetCostKind getSLPCostKind(const Function *F) {
return F->hasOptSize() ? TTI::TCK_CodeSize : TTI::TCK_RecipThroughput;
}
+/// Checks if \p V is a zero-extended sub-field of a wider integer scalar.
+/// Returns the source scalar, the field width and the field offset.
+static std::optional<std::tuple<Value *, unsigned, unsigned>>
+matchExtractedField(Value *V) {
+ if (!V->getType()->isIntegerTy())
+ return std::nullopt;
+ // Field offset for the field-aligned shift amount, if the shifted value of
+ // the given bit width keeps at least one full field.
+ auto GetFieldOffset = [](const APInt *Amt, unsigned BitWidth,
+ unsigned FieldWidth) -> std::optional<unsigned> {
+ uint64_t ShAmt = Amt->getLimitedValue(BitWidth);
+ if (ShAmt % FieldWidth != 0 || ShAmt + FieldWidth > BitWidth)
+ return std::nullopt;
+ return ShAmt / FieldWidth;
+ };
+ // Checks if the low bits of Val are a sub-field of the given width of a
+ // wider integer scalar. Val is a scalar integer, since V is one, and so is
+ // the matched source.
+ auto MatchLowField =
+ [&](Value *Val,
+ unsigned FieldWidth) -> std::optional<std::pair<Value *, unsigned>> {
+ Value *Src;
+ const APInt *Amt;
+ // Only the low bits of Val are observed, so lshr and ashr are equivalent.
+ if (match(Val, m_Trunc(m_Shr(m_Value(Src), m_APInt(Amt)))) ||
+ match(Val, m_Shr(m_Value(Src), m_APInt(Amt)))) {
+ if (std::optional<unsigned> Offset = GetFieldOffset(
+ Amt, Src->getType()->getIntegerBitWidth(), FieldWidth))
+ return std::make_pair(Src, *Offset);
+ return std::nullopt;
+ }
+ if (match(Val, m_Shr(m_Trunc(m_Value(Src)), m_APInt(Amt)))) {
+ if (std::optional<unsigned> Offset = GetFieldOffset(
+ Amt, Val->getType()->getIntegerBitWidth(), FieldWidth))
+ return std::make_pair(Src, *Offset);
+ return std::nullopt;
+ }
+ if (match(Val, m_Trunc(m_Value(Src))) &&
+ Src->getType()->getIntegerBitWidth() >= FieldWidth)
+ return std::make_pair(Src, 0u);
+ // Val itself is the source of its low field.
+ if (Val->getType()->getIntegerBitWidth() > FieldWidth)
+ return std::make_pair(Val, 0u);
+ return std::nullopt;
+ };
+ Value *Val;
+ const APInt *Mask;
+ // and Val, (1 << FieldWidth) - 1 or zext i<FieldWidth> Val - the low bits of
+ // Val.
+ unsigned FieldWidth = 0;
+ if (match(V, m_c_And(m_Value(Val), m_APInt(Mask))) && Mask->isMask())
+ FieldWidth = Mask->popcount();
+ else if (match(V, m_ZExt(m_Value(Val))))
+ FieldWidth = Val->getType()->getIntegerBitWidth();
+ if (FieldWidth != 0) {
+ if (std::optional<std::pair<Value *, unsigned>> Field =
+ MatchLowField(Val, FieldWidth))
+ return std::make_tuple(Field->first, FieldWidth, Field->second);
+ return std::nullopt;
+ }
+ unsigned LaneWidth = V->getType()->getIntegerBitWidth();
+ Value *Src;
+ const APInt *Amt;
+ if (match(V, m_Trunc(m_LShr(m_Value(Src), m_APInt(Amt))))) {
+ unsigned SrcWidth = Src->getType()->getIntegerBitWidth();
+ uint64_t ShAmt = Amt->getLimitedValue(SrcWidth);
+ // The field itself, if the lane width is the field width.
+ if (std::optional<unsigned> Offset =
+ GetFieldOffset(Amt, SrcWidth, LaneWidth))
+ return std::make_tuple(Src, LaneWidth, *Offset);
+ // The zero-extended top field of the source.
+ unsigned FieldWidth = SrcWidth - ShAmt;
+ if (FieldWidth > 0 && FieldWidth < LaneWidth && ShAmt % FieldWidth == 0)
+ return std::make_tuple(Src, FieldWidth, ShAmt / FieldWidth);
+ return std::nullopt;
+ }
+ if (match(V, m_LShr(m_Value(Src), m_APInt(Amt)))) {
+ // The zero-extended top field of the source, if the result keeps exactly
+ // one field. Look through a truncation of the shifted value.
+ unsigned ShfWidth = Src->getType()->getIntegerBitWidth();
+ uint64_t ShAmt = Amt->getLimitedValue(ShfWidth);
+ unsigned FieldWidth = ShfWidth - ShAmt;
+ if (FieldWidth > 0 && ShAmt % FieldWidth == 0) {
+ match(Src, m_Trunc(m_Value(Src)));
+ return std::make_tuple(Src, FieldWidth, ShAmt / FieldWidth);
+ }
+ return std::nullopt;
+ }
+ if (match(V, m_Trunc(m_Value(Src))))
+ return std::make_tuple(Src, LaneWidth, 0u);
+ return std::nullopt;
+}
+
+std::optional<std::tuple<Value *, unsigned, SmallVector<int>>>
+matchGatheredExtractedFields(ArrayRef<Value *> VL, const DataLayout &DL) {
+ // Splats are emitted as broadcasts, sub-fields of a constant are folded.
+ // The bitcast to the field vector maps lane 0 to the least significant
+ // field on little-endian targets only.
+ if (VL.size() < 2 || !VL.front()->getType()->isIntegerTy() || isSplat(VL) ||
+ DL.isBigEndian())
+ return std::nullopt;
+ Value *Src = nullptr;
+ unsigned FieldWidth = 0;
+ SmallVector<int> Mask(VL.size(), PoisonMaskElem);
+ for (auto [Idx, V] : enumerate(VL)) {
+ if (isa<UndefValue>(V))
+ continue;
+ if (V->getType() != VL.front()->getType())
+ return std::nullopt;
+ std::optional<std::tuple<Value *, unsigned, unsigned>> Field =
+ matchExtractedField(V);
+ if (!Field || (Src && (Src != std::get<0>(*Field) ||
+ FieldWidth != std::get<1>(*Field))))
+ return std::nullopt;
+ Src = std::get<0>(*Field);
+ FieldWidth = std::get<1>(*Field);
+ Mask[Idx] = std::get<2>(*Field);
+ }
+ // The field width is a whole number of bytes and divides the source
+ // exactly, same as for the packing layout, so the source bitcasts to the
+ // field vector.
+ if (!Src || isa<Constant>(Src) || FieldWidth % 8 != 0 ||
+ Src->getType()->getIntegerBitWidth() % FieldWidth != 0)
+ return std::nullopt;
+ // The same field in every lane is a splat, emitted as a broadcast.
+ if (all_of(Mask, [First = *find_if(Mask, not_equal_to(PoisonMaskElem))](
+ int MaskElt) {
+ return MaskElt == PoisonMaskElem || MaskElt == First;
+ }))
+ return std::nullopt;
+ return std::make_tuple(Src, FieldWidth, std::move(Mask));
+}
+
/// Deeper than the standard analysis recursion depth to keep the numeric
/// bound precise through arithmetic carry chains.
constexpr unsigned MaxBitPackAnalysisDepth = MaxAnalysisRecursionDepth + 2;
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.h b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.h
index bffe539822341c..4ac6880d0f17d0 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.h
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.h
@@ -30,6 +30,7 @@
#include <limits>
#include <optional>
#include <string>
+#include <tuple>
namespace llvm {
class AssumptionCache;
@@ -419,6 +420,14 @@ TargetTransformInfo::TargetCostKind getSLPCostKind(const Function *F);
/// loses it.
APInt getScalarMaxValue(const Value *V, unsigned Depth = 0);
+/// Checks if the values in \p VL are zero-extended sub-fields of the same
+/// wider integer scalar. Returns the source scalar, the field width and the
+/// field permutation mask. The extraction dual of the lane-packing layout.
+/// The field-to-lane mapping of the bitcast to the field vector is defined
+/// for little-endian targets only.
+std::optional<std::tuple<Value *, unsigned, SmallVector<int>>>
+matchGatheredExtractedFields(ArrayRef<Value *> VL, const DataLayout &DL);
+
/// Description of a bitfield packing of vector lanes into a scalar value:
/// every lane contributes a disjoint contiguous byte field of the result.
struct BitPackInfo {
diff --git a/llvm/test/Transforms/PhaseOrdering/AArch64/scalarize-load-ext-extract.ll b/llvm/test/Transforms/PhaseOrdering/AArch64/scalarize-load-ext-extract.ll
index 8af7ae1b0ac77c..ba0974ad8d80c6 100644
--- a/llvm/test/Transforms/PhaseOrdering/AArch64/scalarize-load-ext-extract.ll
+++ b/llvm/test/Transforms/PhaseOrdering/AArch64/scalarize-load-ext-extract.ll
@@ -5,11 +5,8 @@ define noundef i32 @load_ext_extract(ptr %src) {
; CHECK-LABEL: define noundef range(i32 0, 1021) i32 @load_ext_extract(
; CHECK-SAME: ptr nofree readonly captures(none) [[SRC:%.*]]) local_unnamed_addr #[[ATTR0:[0-9]+]] {
; CHECK-NEXT: [[ENTRY:.*:]]
-; CHECK-NEXT: [[TMP14:%.*]] = load i32, ptr [[SRC]], align 4
-; CHECK-NEXT: [[TMP0:%.*]] = insertelement <4 x i32> poison, i32 [[TMP14]], i64 0
-; CHECK-NEXT: [[TMP1:%.*]] = shufflevector <4 x i32> [[TMP0]], <4 x i32> poison, <4 x i32> zeroinitializer
-; CHECK-NEXT: [[TMP2:%.*]] = lshr <4 x i32> [[TMP1]], <i32 0, i32 8, i32 16, i32 24>
-; CHECK-NEXT: [[TMP5:%.*]] = and <4 x i32> [[TMP2]], <i32 255, i32 255, i32 255, i32 -1>
+; CHECK-NEXT: [[X12:%.*]] = load <4 x i8>, ptr [[SRC]], align 4
+; CHECK-NEXT: [[TMP5:%.*]] = zext <4 x i8> [[X12]] to <4 x i32>
; CHECK-NEXT: [[ADD3:%.*]] = tail call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> [[TMP5]])
; CHECK-NEXT: ret i32 [[ADD3]]
;
diff --git a/llvm/test/Transforms/PhaseOrdering/X86/avg.ll b/llvm/test/Transforms/PhaseOrdering/X86/avg.ll
index 8f8899d285ad37..be56447572ab37 100644
--- a/llvm/test/Transforms/PhaseOrdering/X86/avg.ll
+++ b/llvm/test/Transforms/PhaseOrdering/X86/avg.ll
@@ -1,12 +1,12 @@
; NOTE: Assertions have been autogenerated by utils/update_test_checks.py
; RUN: opt < %s -O3 -S -mtriple=x86_64-- -mcpu=x86-64 | FileCheck %s --check-prefixes=CHECK,SSE2
; RUN: opt < %s -O3 -S -mtriple=x86_64-- -mcpu=x86-64-v2 | FileCheck %s --check-prefixes=CHECK,SSE4
-; RUN: opt < %s -O3 -S -mtriple=x86_64-- -mcpu=x86-64-v3 | FileCheck %s --check-prefixes=CHECK,AVX,AVX2
-; RUN: opt < %s -O3 -S -mtriple=x86_64-- -mcpu=x86-64-v4 | FileCheck %s --check-prefixes=CHECK,AVX,AVX512
+; RUN: opt < %s -O3 -S -mtriple=x86_64-- -mcpu=x86-64-v3 | FileCheck %s --check-prefixes=CHECK,AVX2
+; RUN: opt < %s -O3 -S -mtriple=x86_64-- -mcpu=x86-64-v4 | FileCheck %s --check-prefixes=CHECK,AVX512
; RUN: opt < %s -passes="default<O3>" -S -mtriple=x86_64-- -mcpu=x86-64 | FileCheck %s --check-prefixes=CHECK,SSE2
; RUN: opt < %s -passes="default<O3>" -S -mtriple=x86_64-- -mcpu=x86-64-v2 | FileCheck %s --check-prefixes=CHECK,SSE4
-; RUN: opt < %s -passes="default<O3>" -S -mtriple=x86_64-- -mcpu=x86-64-v3 | FileCheck %s --check-prefixes=CHECK,AVX,AVX2
-; RUN: opt < %s -passes="default<O3>" -S -mtriple=x86_64-- -mcpu=x86-64-v4 | FileCheck %s --check-prefixes=CHECK,AVX,AVX512
+; RUN: opt < %s -passes="default<O3>" -S -mtriple=x86_64-- -mcpu=x86-64-v3 | FileCheck %s --check-prefixes=CHECK,AVX2
+; RUN: opt < %s -passes="default<O3>" -S -mtriple=x86_64-- -mcpu=x86-64-v4 | FileCheck %s --check-prefixes=CHECK,AVX512
; PR128424
@@ -95,121 +95,80 @@ define { i64, i64 } @avgr_16_u8(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerce0,
;
; SSE4-LABEL: @avgr_16_u8(
; SSE4-NEXT: entry:
-; SSE4-NEXT: [[TMP0:%.*]] = trunc i64 [[A_COERCE0:%.*]] to i16
-; SSE4-NEXT: [[TMP1:%.*]] = lshr i16 [[TMP0]], 8
-; SSE4-NEXT: [[A_SROA_3_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE0]], 16
-; SSE4-NEXT: [[A_SROA_4_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE0]], 24
-; SSE4-NEXT: [[A_SROA_5_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE0]], 32
-; SSE4-NEXT: [[A_SROA_6_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE0]], 40
-; SSE4-NEXT: [[A_SROA_7_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE0]], 48
-; SSE4-NEXT: [[A_SROA_8_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE0]], 56
-; SSE4-NEXT: [[TMP2:%.*]] = trunc i64 [[A_COERCE1:%.*]] to i16
-; SSE4-NEXT: [[TMP3:%.*]] = lshr i16 [[TMP2]], 8
-; SSE4-NEXT: [[A_SROA_12_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE1]], 16
-; SSE4-NEXT: [[A_SROA_13_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE1]], 24
-; SSE4-NEXT: [[A_SROA_14_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE1]], 32
-; SSE4-NEXT: [[A_SROA_15_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE1]], 40
-; SSE4-NEXT: [[A_SROA_16_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE1]], 48
-; SSE4-NEXT: [[A_SROA_17_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE1]], 56
-; SSE4-NEXT: [[TMP4:%.*]] = trunc i64 [[B_COERCE0:%.*]] to i16
-; SSE4-NEXT: [[TMP5:%.*]] = lshr i16 [[TMP4]], 8
-; SSE4-NEXT: [[B_SROA_3_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE0]], 16
-; SSE4-NEXT: [[B_SROA_4_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE0]], 24
-; SSE4-NEXT: [[B_SROA_5_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE0]], 32
-; SSE4-NEXT: [[B_SROA_6_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE0]], 40
-; SSE4-NEXT: [[B_SROA_7_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE0]], 48
-; SSE4-NEXT: [[B_SROA_8_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE0]], 56
-; SSE4-NEXT: [[TMP6:%.*]] = trunc i64 [[B_COERCE1:%.*]] to i16
-; SSE4-NEXT: [[TMP7:%.*]] = lshr i16 [[TMP6]], 8
-; SSE4-NEXT: [[B_SROA_12_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE1]], 16
-; SSE4-NEXT: [[B_SROA_13_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE1]], 24
-; SSE4-NEXT: [[B_SROA_14_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE1]], 32
-; SSE4-NEXT: [[B_SROA_15_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE1]], 40
-; SSE4-NEXT: [[B_SROA_16_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE1]], 48
-; SSE4-NEXT: [[B_SROA_17_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE1]], 56
-; SSE4-NEXT: [[CONV1:%.*]] = and i64 [[A_COERCE0]], 255
-; SSE4-NEXT: [[CONV4:%.*]] = and i64 [[B_COERCE0]], 255
-; SSE4-NEXT: [[ADD:%.*]] = add nuw nsw i64 [[CONV1]], 1
-; SSE4-NEXT: [[ADD5:%.*]] = add nuw nsw i64 [[ADD]], [[CONV4]]
-; SSE4-NEXT: [[SHR:%.*]] = lshr i64 [[ADD5]], 1
-; SSE4-NEXT: [[ADD_1:%.*]] = add nuw nsw i16 [[TMP1]], 1
-; SSE4-NEXT: [[ADD5_1:%.*]] = add nuw nsw i16 [[ADD_1]], [[TMP5]]
-; SSE4-NEXT: [[CONV1_6:%.*]] = and i64 [[A_SROA_7_0_EXTRACT_SHIFT]], 255
-; SSE4-NEXT: [[CONV4_6:%.*]] = and i64 [[B_SROA_7_0_EXTRACT_SHIFT]], 255
-; SSE4-NEXT: [[ADD_6:%.*]] = add nuw nsw i64 [[CONV1_6]], 1
-; SSE4-NEXT: [[ADD5_6:%.*]] = add nuw nsw i64 [[ADD_6]], [[CONV4_6]]
-; SSE4-NEXT: [[ADD_7:%.*]] = add nuw nsw i64 [[A_SROA_8_0_EXTRACT_SHIFT]], 1
-; SSE4-NEXT: [[ADD5_7:%.*]] = add nuw nsw i64 [[ADD_7]], [[B_SROA_8_0_EXTRACT_SHIFT]]
-; SSE4-NEXT: [[CONV1_8:%.*]] = and i64 [[A_COERCE1]], 255
-; SSE4-NEXT: [[CONV4_8:%.*]] = and i64 [[B_COERCE1]], 255
-; SSE4-NEXT: [[ADD_8:%.*]] = add nuw nsw i64 [[CONV1_8]], 1
-; SSE4-NEXT: [[ADD5_8:%.*]] = add nuw nsw i64 [[ADD_8]], [[CONV4_8]]
-; SSE4-NEXT: [[SHR_8:%.*]] = lshr i64 [[ADD5_8]], 1
-; SSE4-NEXT: [[ADD_9:%.*]] = add nuw nsw i16 [[TMP3]], 1
-; SSE4-NEXT: [[ADD5_9:%.*]] = add nuw nsw i16 [[ADD_9]], [[TMP7]]
-; SSE4-NEXT: [[CONV1_14:%.*]] = and i64 [[A_SROA_16_8_EXTRACT_SHIFT]], 255
-; SSE4-NEXT: [[CONV4_14:%.*]] = and i64 [[B_SROA_16_8_EXTRACT_SHIFT]], 255
-; SSE4-NEXT: [[ADD_14:%.*]] = add nuw nsw i64 [[CONV1_14]], 1
-; SSE4-NEXT: [[ADD5_14:%.*]] = add nuw nsw i64 [[ADD_14]], [[CONV4_14]]
-; SSE4-NEXT: [[ADD_15:%.*]] = add nuw nsw i64 [[A_SROA_17_8_EXTRACT_SHIFT]], 1
-; SSE4-NEXT: [[ADD5_15:%.*]] = add nuw nsw i64 [[ADD_15]], [[B_SROA_17_8_EXTRACT_SHIFT]]
-; SSE4-NEXT: [[TMP8:%.*]] = shl nuw i64 [[ADD5_7]], 55
-; SSE4-NEXT: [[RETVAL_SROA_8_0_INSERT_EXT:%.*]] = and i64 [[TMP8]], -72057594037927936
-; SSE4-NEXT: [[TMP9:%.*]] = shl nuw nsw i64 [[ADD5_6]], 47
-; SSE4-NEXT: [[RETVAL_SROA_7_0_INSERT_SHIFT:%.*]] = and i64 [[TMP9]], 71776119061217280
-; SSE4-NEXT: [[TMP10:%.*]] = insertelement <4 x i64> poison, i64 [[A_SROA_6_0_EXTRACT_SHIFT]], i64 0
-; SSE4-NEXT: [[TMP11:%.*]] = insertelement <4 x i64> [[TMP10]], i64 [[A_SROA_5_0_EXTRACT_SHIFT]], i64 1
-; SSE4-NEXT: [[TMP12:%.*]] = insertelement <4 x i64> [[TMP11]], i64 [[A_SROA_4_0_EXTRACT_SHIFT]], i64 2
-; SSE4-NEXT: [[TMP13:%.*]] = insertelement <4 x i64> [[TMP12]], i64 [[A_SROA_3_0_EXTRACT_SHIFT]], i64 3
-; SSE4-NEXT: [[TMP14:%.*]] = and <4 x i64> [[TMP13]], splat (i64 255)
-; SSE4-NEXT: [[TMP15:%.*]] = insertelement <4 x i64> poison, i64 [[B_SROA_6_0_EXTRACT_SHIFT]], i64 0
-; SSE4-NEXT: [[TMP16:%.*]] = insertelement <4 x i64> [[TMP15]], i64 [[B_SROA_5_0_EXTRACT_SHIFT]], i64 1
-; SSE4-NEXT: [[TMP17:%.*]] = insertelement <4 x i64> [[TMP16]], i64 [[B_SROA_4_0_EXTRACT_SHIFT]], i64 2
-; SSE4-NEXT: [[TMP18:%.*]] = insertelement <4 x i64> [[TMP17]], i64 [[B_SROA_3_0_EXTRACT_SHIFT]], i64 3
-; SSE4-NEXT: [[TMP19:%.*]] = and <4 x i64> [[TMP18]], splat (i64 255)
-; SSE4-NEXT: [[TMP20:%.*]] = add nuw nsw <4 x i64> [[TMP14]], splat (i64 1)
-; SSE4-NEXT: [[TMP21:%.*]] = add nuw nsw <4 x i64> [[TMP20]], [[TMP19]]
-; SSE4-NEXT: [[TMP22:%.*]] = trunc nuw nsw <4 x i64> [[TMP21]] to <4 x i32>
-; SSE4-NEXT: [[TMP23:%.*]] = lshr <4 x i32> [[TMP22]], splat (i32 1)
-; SSE4-NEXT: [[TMP24:%.*]] = bitcast <4 x i32> [[TMP23]] to <16 x i8>
-; SSE4-NEXT: [[TMP25:%.*]] = shufflevector <16 x i8> [[TMP24]], <16 x i8> <i8 0, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison>, <8 x i32> <i32 16, i32 16, i32 12, i32 8, i32 4, i32 0, i32 16, i32 16>
-; SSE4-NEXT: [[TMP26:%.*]] = bitcast <8 x i8> [[TMP25]] to i64
-; SSE4-NEXT: [[TMP27:%.*]] = shl nuw i16 [[ADD5_1]], 7
-; SSE4-NEXT: [[TMP28:%.*]] = and i16 [[TMP27]], -256
-; SSE4-NEXT: [[RETVAL_SROA_2_0_INSERT_SHIFT_MASKED:%.*]] = zext i16 [[TMP28]] to i64
-; SSE4-NEXT: [[OP_RDX5:%.*]] = or disjoint i64 [[RETVAL_SROA_8_0_INSERT_EXT]], [[TMP26]]
-; SSE4-NEXT: [[OP_RDX6:%.*]] = or disjoint i64 [[RETVAL_SROA_7_0_INSERT_SHIFT]], [[SHR]]
-; SSE4-NEXT: [[OP_RDX7:%.*]] = or disjoint i64 [[OP_RDX5]], [[OP_RDX6]]
-; SSE4-NEXT: [[TMP68:%.*]] = or i64 [[OP_RDX7]], [[RETVAL_SROA_2_0_INSERT_SHIFT_MASKED]]
+; SSE4-NEXT: [[TMP0:%.*]] = insertelement <2 x i64> poison, i64 [[A_COERCE0:%.*]], i64 0
+; SSE4-NEXT: [[TMP1:%.*]] = insertelement <2 x i64> [[TMP0]], i64 [[A_COERCE1:%.*]], i64 1
+; SSE4-NEXT: [[TMP2:%.*]] = trunc <2 x i64> [[TMP1]] to <2 x i16>
+; SSE4-NEXT: [[TMP3:%.*]] = lshr <2 x i64> [[TMP1]], splat (i64 16)
+; SSE4-NEXT: [[TMP4:%.*]] = lshr <2 x i64> [[TMP1]], splat (i64 24)
+; SSE4-NEXT: [[TMP5:%.*]] = lshr <2 x i64> [[TMP1]], splat (i64 32)
+; SSE4-NEXT: [[TMP6:%.*]] = lshr <2 x i64> [[TMP1]], splat (i64 40)
+; SSE4-NEXT: [[TMP7:%.*]] = lshr <2 x i64> [[TMP1]], splat (i64 48)
+; SSE4-NEXT: [[TMP8:%.*]] = lshr <2 x i64> [[TMP1]], splat (i64 56)
+; SSE4-NEXT: [[TMP9:%.*]] = insertelement <2 x i64> poison, i64 [[B_COERCE0:%.*]], i64 0
+; SSE4-NEXT: [[TMP10:%.*]] = insertelement <2 x i64> [[TMP9]], i64 [[B_COERCE1:%.*]], i64 1
+; SSE4-NEXT: [[TMP11:%.*]] = trunc <2 x i64> [[TMP10]] to <2 x i16>
+; SSE4-NEXT: [[TMP12:%.*]] = lshr <2 x i64> [[TMP10]], splat (i64 16)
+; SSE4-NEXT: [[TMP13:%.*]] = lshr <2 x i64> [[TMP10]], splat (i64 24)
+; SSE4-NEXT: [[TMP14:%.*]] = lshr <2 x i64> [[TMP10]], splat (i64 32)
+; SSE4-NEXT: [[TMP15:%.*]] = lshr <2 x i64> [[TMP10]], splat (i64 40)
+; SSE4-NEXT: [[TMP16:%.*]] = lshr <2 x i64> [[TMP10]], splat (i64 48)
+; SSE4-NEXT: [[TMP17:%.*]] = lshr <2 x i64> [[TMP10]], splat (i64 56)
+; SSE4-NEXT: [[TMP18:%.*]] = and <2 x i64> [[TMP1]], splat (i64 255)
+; SSE4-NEXT: [[TMP19:%.*]] = and <2 x i64> [[TMP10]], splat (i64 255)
+; SSE4-NEXT: [[TMP20:%.*]] = and <2 x i64> [[TMP7]], splat (i64 255)
+; SSE4-NEXT: [[TMP21:%.*]] = lshr <2 x i16> [[TMP2]], splat (i16 8)
+; SSE4-NEXT: [[TMP22:%.*]] = lshr <2 x i16> [[TMP11]], splat (i16 8)
+; SSE4-NEXT: [[TMP23:%.*]] = add nuw nsw <2 x i64> [[TMP18]], splat (i64 1)
+; SSE4-NEXT: [[TMP24:%.*]] = add nuw nsw <2 x i64> [[TMP23]], [[TMP19]]
+; SSE4-NEXT: [[TMP25:%.*]] = lshr <2 x i64> [[TMP24]], splat (i64 1)
+; SSE4-NEXT: [[TMP26:%.*]] = add nuw nsw <2 x i16> [[TMP21]], splat (i16 1)
+; SSE4-NEXT: [[TMP27:%.*]] = add nuw nsw <2 x i16> [[TMP26]], [[TMP22]]
+; SSE4-NEXT: [[TMP28:%.*]] = and <2 x i64> [[TMP3]], splat (i64 255)
+; SSE4-NEXT: [[TMP29:%.*]] = and <2 x i64> [[TMP12]], splat (i64 255)
+; SSE4-NEXT: [[TMP30:%.*]] = add nuw nsw <2 x i64> [[TMP28]], splat (i64 1)
+; SSE4-NEXT: [[TMP31:%.*]] = add nuw nsw <2 x i64> [[TMP30]], [[TMP29]]
+; SSE4-NEXT: [[TMP32:%.*]] = and <2 x i64> [[TMP4]], splat (i64 255)
+; SSE4-NEXT: [[TMP33:%.*]] = and <2 x i64> [[TMP13]], splat (i64 255)
+; SSE4-NEXT: [[TMP34:%.*]] = add nuw nsw <2 x i64> [[TMP32]], splat (i64 1)
+; SSE4-NEXT: [[TMP35:%.*]] = add nuw nsw <2 x i64> [[TMP34]], [[TMP33]]
+; SSE4-NEXT: [[TMP36:%.*]] = and <2 x i64> [[TMP5]], splat (i64 255)
+; SSE4-NEXT: [[TMP37:%.*]] = and <2 x i64> [[TMP14]], splat (i64 255)
+; SSE4-NEXT: [[TMP38:%.*]] = add nuw nsw <2 x i64> [[TMP36]], splat (i64 1)
+; SSE4-NEXT: [[TMP39:%.*]] = add nuw nsw <2 x i64> [[TMP38]], [[TMP37]]
+; SSE4-NEXT: [[TMP40:%.*]] = and <2 x i64> [[TMP6]], splat (i64 255)
+; SSE4-NEXT: [[TMP41:%.*]] = and <2 x i64> [[TMP15]], splat (i64 255)
+; SSE4-NEXT: [[TMP42:%.*]] = add nuw nsw <2 x i64> [[TMP40]], splat (i64 1)
+; SSE4-NEXT: [[TMP43:%.*]] = add nuw nsw <2 x i64> [[TMP42]], [[TMP41]]
+; SSE4-NEXT: [[TMP44:%.*]] = and <2 x i64> [[TMP16]], splat (i64 255)
+; SSE4-NEXT: [[TMP45:%.*]] = add nuw nsw <2 x i64> [[TMP20]], splat (i64 1)
+; SSE4-NEXT: [[TMP46:%.*]] = add nuw nsw <2 x i64> [[TMP45]], [[TMP44]]
+; SSE4-NEXT: [[TMP47:%.*]] = add nuw nsw <2 x i64> [[TMP8]], splat (i64 1)
+; SSE4-NEXT: [[TMP48:%.*]] = add nuw nsw <2 x i64> [[TMP47]], [[TMP17]]
+; SSE4-NEXT: [[TMP49:%.*]] = shl nuw <2 x i64> [[TMP48]], splat (i64 55)
+; SSE4-NEXT: [[TMP50:%.*]] = and <2 x i64> [[TMP49]], splat (i64 -72057594037927936)
+; SSE4-NEXT: [[TMP51:%.*]] = shl nuw nsw <2 x i64> [[TMP46]], splat (i64 47)
+; SSE4-NEXT: [[TMP52:%.*]] = and <2 x i64> [[TMP51]], splat (i64 71776119061217280)
+; SSE4-NEXT: [[TMP53:%.*]] = or disjoint <2 x i64> [[TMP50]], [[TMP52]]
+; SSE4-NEXT: [[TMP54:%.*]] = shl nuw nsw <2 x i64> [[TMP43]], splat (i64 39)
+; SSE4-NEXT: [[TMP55:%.*]] = and <2 x i64> [[TMP54]], splat (i64 280375465082880)
+; SSE4-NEXT: [[TMP56:%.*]] = or disjoint <2 x i64> [[TMP53]], [[TMP55]]
+; SSE4-NEXT: [[TMP57:%.*]] = shl nuw nsw <2 x i64> [[TMP39]], splat (i64 31)
+; SSE4-NEXT: [[TMP58:%.*]] = and <2 x i64> [[TMP57]], splat (i64 1095216660480)
+; SSE4-NEXT: [[TMP59:%.*]] = or disjoint <2 x i64> [[TMP56]], [[TMP58]]
+; SSE4-NEXT: [[TMP60:%.*]] = shl nuw nsw <2 x i64> [[TMP35]], splat (i64 23)
+; SSE4-NEXT: [[TMP61:%.*]] = and <2 x i64> [[TMP60]], splat (i64 4278190080)
+; SSE4-NEXT: [[TMP62:%.*]] = or disjoint <2 x i64> [[TMP59]], [[TMP61]]
+; SSE4-NEXT: [[TMP63:%.*]] = shl nuw nsw <2 x i64> [[TMP31]], splat (i64 15)
+; SSE4-NEXT: [[TMP64:%.*]] = and <2 x i64> [[TMP63]], splat (i64 16711680)
+; SSE4-NEXT: [[TMP65:%.*]] = shl nuw <2 x i16> [[TMP27]], splat (i16 7)
+; SSE4-NEXT: [[TMP66:%.*]] = or disjoint <2 x i64> [[TMP62]], [[TMP64]]
+; SSE4-NEXT: [[TMP67:%.*]] = and <2 x i16> [[TMP65]], splat (i16 -256)
+; SSE4-NEXT: [[TMP71:%.*]] = zext <2 x i16> [[TMP67]] to <2 x i64>
+; SSE4-NEXT: [[TMP72:%.*]] = or <2 x i64> [[TMP66]], [[TMP71]]
+; SSE4-NEXT: [[TMP70:%.*]] = or <2 x i64> [[TMP72]], [[TMP25]]
+; SSE4-NEXT: [[TMP68:%.*]] = extractelement <2 x i64> [[TMP70]], i64 0
; SSE4-NEXT: [[DOTFCA_0_INSERT:%.*]] = insertvalue { i64, i64 } poison, i64 [[TMP68]], 0
-; SSE4-NEXT: [[TMP29:%.*]] = shl nuw i64 [[ADD5_15]], 55
-; SSE4-NEXT: [[RETVAL_SROA_17_8_INSERT_EXT:%.*]] = and i64 [[TMP29]], -72057594037927936
-; SSE4-NEXT: [[TMP30:%.*]] = shl nuw nsw i64 [[ADD5_14]], 47
-; SSE4-NEXT: [[RETVAL_SROA_16_8_INSERT_SHIFT:%.*]] = and i64 [[TMP30]], 71776119061217280
-; SSE4-NEXT: [[TMP31:%.*]] = insertelement <4 x i64> poison, i64 [[A_SROA_15_8_EXTRACT_SHIFT]], i64 0
-; SSE4-NEXT: [[TMP32:%.*]] = insertelement <4 x i64> [[TMP31]], i64 [[A_SROA_14_8_EXTRACT_SHIFT]], i64 1
-; SSE4-NEXT: [[TMP33:%.*]] = insertelement <4 x i64> [[TMP32]], i64 [[A_SROA_13_8_EXTRACT_SHIFT]], i64 2
-; SSE4-NEXT: [[TMP34:%.*]] = insertelement <4 x i64> [[TMP33]], i64 [[A_SROA_12_8_EXTRACT_SHIFT]], i64 3
-; SSE4-NEXT: [[TMP35:%.*]] = and <4 x i64> [[TMP34]], splat (i64 255)
-; SSE4-NEXT: [[TMP36:%.*]] = insertelement <4 x i64> poison, i64 [[B_SROA_15_8_EXTRACT_SHIFT]], i64 0
-; SSE4-NEXT: [[TMP37:%.*]] = insertelement <4 x i64> [[TMP36]], i64 [[B_SROA_14_8_EXTRACT_SHIFT]], i64 1
-; SSE4-NEXT: [[TMP38:%.*]] = insertelement <4 x i64> [[TMP37]], i64 [[B_SROA_13_8_EXTRACT_SHIFT]], i64 2
-; SSE4-NEXT: [[TMP39:%.*]] = insertelement <4 x i64> [[TMP38]], i64 [[B_SROA_12_8_EXTRACT_SHIFT]], i64 3
-; SSE4-NEXT: [[TMP40:%.*]] = and <4 x i64> [[TMP39]], splat (i64 255)
-; SSE4-NEXT: [[TMP41:%.*]] = add nuw nsw <4 x i64> [[TMP35]], splat (i64 1)
-; SSE4-NEXT: [[TMP42:%.*]] = add nuw nsw <4 x i64> [[TMP41]], [[TMP40]]
-; SSE4-NEXT: [[TMP43:%.*]] = trunc nuw nsw <4 x i64> [[TMP42]] to <4 x i32>
-; SSE4-NEXT: [[TMP44:%.*]] = lshr <4 x i32> [[TMP43]], splat (i32 1)
-; SSE4-NEXT: [[TMP45:%.*]] = bitcast <4 x i32> [[TMP44]] to <16 x i8>
-; SSE4-NEXT: [[TMP46:%.*]] = shufflevector <16 x i8> [[TMP45]], <16 x i8> <i8 0, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison>, <8 x i32> <i32 16, i32 16, i32 12, i32 8, i32 4, i32 0, i32 16, i32 16>
-; SSE4-NEXT: [[TMP47:%.*]] = bitcast <8 x i8> [[TMP46]] to i64
-; SSE4-NEXT: [[TMP48:%.*]] = shl nuw i16 [[ADD5_9]], 7
-; SSE4-NEXT: [[TMP49:%.*]] = and i16 [[TMP48]], -256
-; SSE4-NEXT: [[RETVAL_SROA_11_8_INSERT_SHIFT_MASKED:%.*]] = zext i16 [[TMP49]] to i64
-; SSE4-NEXT: [[OP_RDX:%.*]] = or disjoint i64 [[RETVAL_SROA_17_8_INSERT_EXT]], [[TMP47]]
-; SSE4-NEXT: [[OP_RDX2:%.*]] = or disjoint i64 [[RETVAL_SROA_16_8_INSERT_SHIFT]], [[SHR_8]]
-; SSE4-NEXT: [[OP_RDX3:%.*]] = or disjoint i64 [[OP_RDX]], [[OP_RDX2]]
-; SSE4-NEXT: [[TMP69:%.*]] = or i64 [[OP_RDX3]], [[RETVAL_SROA_11_8_INSERT_SHIFT_MASKED]]
+; SSE4-NEXT: [[TMP69:%.*]] = extractelement <2 x i64> [[TMP70]], i64 1
; SSE4-NEXT: [[DOTFCA_1_INSERT:%.*]] = insertvalue { i64, i64 } [[DOTFCA_0_INSERT]], i64 [[TMP69]], 1
; SSE4-NEXT: ret { i64, i64 } [[DOTFCA_1_INSERT]]
;
@@ -380,283 +339,29 @@ for.body: ; preds = %for.cond
}
define { i64, i64 } @avgr_16_u8_alt(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerce0, i64 %b.coerce1) {
-; SSE2-LABEL: @avgr_16_u8_alt(
-; SSE2-NEXT: entry:
-; SSE2-NEXT: [[A_SROA_0_0_EXTRACT_TRUNC:%.*]] = trunc i64 [[A_COERCE0:%.*]] to i8
-; SSE2-NEXT: [[A_SROA_2_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE0]], 8
-; SSE2-NEXT: [[A_SROA_2_0_EXTRACT_TRUNC:%.*]] = trunc i64 [[A_SROA_2_0_EXTRACT_SHIFT]] to i8
-; SSE2-NEXT: [[A_SROA_3_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE0]], 16
-; SSE2-NEXT: [[A_SROA_3_0_EXTRACT_TRUNC:%.*]] = trunc i64 [[A_SROA_3_0_EXTRACT_SHIFT]] to i8
-; SSE2-NEXT: [[A_SROA_4_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE0]], 24
-; SSE2-NEXT: [[A_SROA_4_0_EXTRACT_TRUNC:%.*]] = trunc i64 [[A_SROA_4_0_EXTRACT_SHIFT]] to i8
-; SSE2-NEXT: [[A_SROA_5_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE0]], 32
-; SSE2-NEXT: [[A_SROA_5_0_EXTRACT_TRUNC:%.*]] = trunc i64 [[A_SROA_5_0_EXTRACT_SHIFT]] to i8
-; SSE2-NEXT: [[A_SROA_6_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE0]], 40
-; SSE2-NEXT: [[A_SROA_6_0_EXTRACT_TRUNC:%.*]] = trunc i64 [[A_SROA_6_0_EXTRACT_SHIFT]] to i8
-; SSE2-NEXT: [[A_SROA_7_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE0]], 48
-; SSE2-NEXT: [[A_SROA_7_0_EXTRACT_TRUNC:%.*]] = trunc i64 [[A_SROA_7_0_EXTRACT_SHIFT]] to i8
-; SSE2-NEXT: [[A_SROA_8_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE0]], 56
-; SSE2-NEXT: [[A_SROA_8_0_EXTRACT_TRUNC:%.*]] = trunc nuw i64 [[A_SROA_8_0_EXTRACT_SHIFT]] to i8
-; SSE2-NEXT: [[A_SROA_9_8_EXTRACT_TRUNC:%.*]] = trunc i64 [[A_COERCE1:%.*]] to i8
-; SSE2-NEXT: [[A_SROA_11_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE1]], 8
-; SSE2-NEXT: [[A_SROA_11_8_EXTRACT_TRUNC:%.*]] = trunc i64 [[A_SROA_11_8_EXTRACT_SHIFT]] to i8
-; SSE2-NEXT: [[A_SROA_12_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE1]], 16
-; SSE2-NEXT: [[A_SROA_12_8_EXTRACT_TRUNC:%.*]] = trunc i64 [[A_SROA_12_8_EXTRACT_SHIFT]] to i8
-; SSE2-NEXT: [[A_SROA_13_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE1]], 24
-; SSE2-NEXT: [[A_SROA_13_8_EXTRACT_TRUNC:%.*]] = trunc i64 [[A_SROA_13_8_EXTRACT_SHIFT]] to i8
-; SSE2-NEXT: [[A_SROA_14_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE1]], 32
-; SSE2-NEXT: [[A_SROA_14_8_EXTRACT_TRUNC:%.*]] = trunc i64 [[A_SROA_14_8_EXTRACT_SHIFT]] to i8
-; SSE2-NEXT: [[A_SROA_15_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE1]], 40
-; SSE2-NEXT: [[A_SROA_15_8_EXTRACT_TRUNC:%.*]] = trunc i64 [[A_SROA_15_8_EXTRACT_SHIFT]] to i8
-; SSE2-NEXT: [[A_SROA_16_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE1]], 48
-; SSE2-NEXT: [[A_SROA_16_8_EXTRACT_TRUNC:%.*]] = trunc i64 [[A_SROA_16_8_EXTRACT_SHIFT]] to i8
-; SSE2-NEXT: [[A_SROA_17_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE1]], 56
-; SSE2-NEXT: [[A_SROA_17_8_EXTRACT_TRUNC:%.*]] = trunc nuw i64 [[A_SROA_17_8_EXTRACT_SHIFT]] to i8
-; SSE2-NEXT: [[B_SROA_0_0_EXTRACT_TRUNC:%.*]] = trunc i64 [[B_COERCE0:%.*]] to i8
-; SSE2-NEXT: [[B_SROA_2_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE0]], 8
-; SSE2-NEXT: [[B_SROA_2_0_EXTRACT_TRUNC:%.*]] = trunc i64 [[B_SROA_2_0_EXTRACT_SHIFT]] to i8
-; SSE2-NEXT: [[B_SROA_3_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE0]], 16
-; SSE2-NEXT: [[B_SROA_3_0_EXTRACT_TRUNC:%.*]] = trunc i64 [[B_SROA_3_0_EXTRACT_SHIFT]] to i8
-; SSE2-NEXT: [[B_SROA_4_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE0]], 24
-; SSE2-NEXT: [[B_SROA_4_0_EXTRACT_TRUNC:%.*]] = trunc i64 [[B_SROA_4_0_EXTRACT_SHIFT]] to i8
-; SSE2-NEXT: [[B_SROA_5_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE0]], 32
-; SSE2-NEXT: [[B_SROA_5_0_EXTRACT_TRUNC:%.*]] = trunc i64 [[B_SROA_5_0_EXTRACT_SHIFT]] to i8
-; SSE2-NEXT: [[B_SROA_6_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE0]], 40
-; SSE2-NEXT: [[B_SROA_6_0_EXTRACT_TRUNC:%.*]] = trunc i64 [[B_SROA_6_0_EXTRACT_SHIFT]] to i8
-; SSE2-NEXT: [[B_SROA_7_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE0]], 48
-; SSE2-NEXT: [[B_SROA_7_0_EXTRACT_TRUNC:%.*]] = trunc i64 [[B_SROA_7_0_EXTRACT_SHIFT]] to i8
-; SSE2-NEXT: [[B_SROA_8_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE0]], 56
-; SSE2-NEXT: [[B_SROA_8_0_EXTRACT_TRUNC:%.*]] = trunc nuw i64 [[B_SROA_8_0_EXTRACT_SHIFT]] to i8
-; SSE2-NEXT: [[B_SROA_9_8_EXTRACT_TRUNC:%.*]] = trunc i64 [[B_COERCE1:%.*]] to i8
-; SSE2-NEXT: [[B_SROA_11_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE1]], 8
-; SSE2-NEXT: [[B_SROA_11_8_EXTRACT_TRUNC:%.*]] = trunc i64 [[B_SROA_11_8_EXTRACT_SHIFT]] to i8
-; SSE2-NEXT: [[B_SROA_12_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE1]], 16
-; SSE2-NEXT: [[B_SROA_12_8_EXTRACT_TRUNC:%.*]] = trunc i64 [[B_SROA_12_8_EXTRACT_SHIFT]] to i8
-; SSE2-NEXT: [[B_SROA_13_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE1]], 24
-; SSE2-NEXT: [[B_SROA_13_8_EXTRACT_TRUNC:%.*]] = trunc i64 [[B_SROA_13_8_EXTRACT_SHIFT]] to i8
-; SSE2-NEXT: [[B_SROA_14_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE1]], 32
-; SSE2-NEXT: [[B_SROA_14_8_EXTRACT_TRUNC:%.*]] = trunc i64 [[B_SROA_14_8_EXTRACT_SHIFT]] to i8
-; SSE2-NEXT: [[B_SROA_15_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE1]], 40
-; SSE2-NEXT: [[B_SROA_15_8_EXTRACT_TRUNC:%.*]] = trunc i64 [[B_SROA_15_8_EXTRACT_SHIFT]] to i8
-; SSE2-NEXT: [[B_SROA_16_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE1]], 48
-; SSE2-NEXT: [[B_SROA_16_8_EXTRACT_TRUNC:%.*]] = trunc i64 [[B_SROA_16_8_EXTRACT_SHIFT]] to i8
-; SSE2-NEXT: [[B_SROA_17_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE1]], 56
-; SSE2-NEXT: [[B_SROA_17_8_EXTRACT_TRUNC:%.*]] = trunc nuw i64 [[B_SROA_17_8_EXTRACT_SHIFT]] to i8
-; SSE2-NEXT: [[SHR:%.*]] = lshr i8 [[A_SROA_0_0_EXTRACT_TRUNC]], 1
-; SSE2-NEXT: [[SHR5:%.*]] = lshr i8 [[B_SROA_0_0_EXTRACT_TRUNC]], 1
-; SSE2-NEXT: [[NARROW:%.*]] = add nuw i8 [[SHR5]], [[SHR]]
-; SSE2-NEXT: [[OR21:%.*]] = or i8 [[B_SROA_0_0_EXTRACT_TRUNC]], [[A_SROA_0_0_EXTRACT_TRUNC]]
-; SSE2-NEXT: [[TMP0:%.*]] = and i8 [[OR21]], 1
-; SSE2-NEXT: [[ADD12:%.*]] = add nuw i8 [[NARROW]], [[TMP0]]
-; SSE2-NEXT: [[SHR_1:%.*]] = lshr i8 [[A_SROA_2_0_EXTRACT_TRUNC]], 1
-; SSE2-NEXT: [[SHR5_1:%.*]] = lshr i8 [[B_SROA_2_0_EXTRACT_TRUNC]], 1
-; SSE2-NEXT: [[NARROW_1:%.*]] = add nuw i8 [[SHR5_1]], [[SHR_1]]
-; SSE2-NEXT: [[OR21_1:%.*]] = or i8 [[B_SROA_2_0_EXTRACT_TRUNC]], [[A_SROA_2_0_EXTRACT_TRUNC]]
-; SSE2-NEXT: [[TMP1:%.*]] = and i8 [[OR21_1]], 1
-; SSE2-NEXT: [[ADD12_1:%.*]] = add nuw i8 [[NARROW_1]], [[TMP1]]
-; SSE2-NEXT: [[SHR_2:%.*]] = lshr i8 [[A_SROA_3_0_EXTRACT_TRUNC]], 1
-; SSE2-NEXT: [[SHR5_2:%.*]] = lshr i8 [[B_SROA_3_0_EXTRACT_TRUNC]], 1
-; SSE2-NEXT: [[NARROW_2:%.*]] = add nuw i8 [[SHR5_2]], [[SHR_2]]
-; SSE2-NEXT: [[OR21_2:%.*]] = or i8 [[B_SROA_3_0_EXTRACT_TRUNC]], [[A_SROA_3_0_EXTRACT_TRUNC]]
-; SSE2-NEXT: [[TMP2:%.*]] = and i8 [[OR21_2]], 1
-; SSE2-NEXT: [[ADD12_2:%.*]] = add nuw i8 [[NARROW_2]], [[TMP2]]
-; SSE2-NEXT: [[SHR_3:%.*]] = lshr i8 [[A_SROA_4_0_EXTRACT_TRUNC]], 1
-; SSE2-NEXT: [[SHR5_3:%.*]] = lshr i8 [[B_SROA_4_0_EXTRACT_TRUNC]], 1
-; SSE2-NEXT: [[NARROW_3:%.*]] = add nuw i8 [[SHR5_3]], [[SHR_3]]
-; SSE2-NEXT: [[OR21_3:%.*]] = or i8 [[B_SROA_4_0_EXTRACT_TRUNC]], [[A_SROA_4_0_EXTRACT_TRUNC]]
-; SSE2-NEXT: [[TMP3:%.*]] = and i8 [[OR21_3]], 1
-; SSE2-NEXT: [[ADD12_3:%.*]] = add nuw i8 [[NARROW_3]], [[TMP3]]
-; SSE2-NEXT: [[SHR_4:%.*]] = lshr i8 [[A_SROA_5_0_EXTRACT_TRUNC]], 1
-; SSE2-NEXT: [[SHR5_4:%.*]] = lshr i8 [[B_SROA_5_0_EXTRACT_TRUNC]], 1
-; SSE2-NEXT: [[NARROW_4:%.*]] = add nuw i8 [[SHR5_4]], [[SHR_4]]
-; SSE2-NEXT: [[OR21_4:%.*]] = or i8 [[B_SROA_5_0_EXTRACT_TRUNC]], [[A_SROA_5_0_EXTRACT_TRUNC]]
-; SSE2-NEXT: [[TMP4:%.*]] = and i8 [[OR21_4]], 1
-; SSE2-NEXT: [[ADD12_4:%.*]] = add nuw i8 [[NARROW_4]], [[TMP4]]
-; SSE2-NEXT: [[SHR_5:%.*]] = lshr i8 [[A_SROA_6_0_EXTRACT_TRUNC]], 1
-; SSE2-NEXT: [[SHR5_5:%.*]] = lshr i8 [[B_SROA_6_0_EXTRACT_TRUNC]], 1
-; SSE2-NEXT: [[NARROW_5:%.*]] = add nuw i8 [[SHR5_5]], [[SHR_5]]
-; SSE2-NEXT: [[OR21_5:%.*]] = or i8 [[B_SROA_6_0_EXTRACT_TRUNC]], [[A_SROA_6_0_EXTRACT_TRUNC]]
-; SSE2-NEXT: [[TMP5:%.*]] = and i8 [[OR21_5]], 1
-; SSE2-NEXT: [[ADD12_5:%.*]] = add nuw i8 [[NARROW_5]], [[TMP5]]
-; SSE2-NEXT: [[SHR_6:%.*]] = lshr i8 [[A_SROA_7_0_EXTRACT_TRUNC]], 1
-; SSE2-NEXT: [[SHR5_6:%.*]] = lshr i8 [[B_SROA_7_0_EXTRACT_TRUNC]], 1
-; SSE2-NEXT: [[NARROW_6:%.*]] = add nuw i8 [[SHR5_6]], [[SHR_6]]
-; SSE2-NEXT: [[OR21_6:%.*]] = or i8 [[B_SROA_7_0_EXTRACT_TRUNC]], [[A_SROA_7_0_EXTRACT_TRUNC]]
-; SSE2-NEXT: [[TMP6:%.*]] = and i8 [[OR21_6]], 1
-; SSE2-NEXT: [[ADD12_6:%.*]] = add nuw i8 [[NARROW_6]], [[TMP6]]
-; SSE2-NEXT: [[SHR_7:%.*]] = lshr i8 [[A_SROA_8_0_EXTRACT_TRUNC]], 1
-; SSE2-NEXT: [[SHR5_7:%.*]] = lshr i8 [[B_SROA_8_0_EXTRACT_TRUNC]], 1
-; SSE2-NEXT: [[NARROW_7:%.*]] = add nuw i8 [[SHR5_7]], [[SHR_7]]
-; SSE2-NEXT: [[OR21_7:%.*]] = or i8 [[B_SROA_8_0_EXTRACT_TRUNC]], [[A_SROA_8_0_EXTRACT_TRUNC]]
-; SSE2-NEXT: [[TMP7:%.*]] = and i8 [[OR21_7]], 1
-; SSE2-NEXT: [[ADD12_7:%.*]] = add nuw i8 [[NARROW_7]], [[TMP7]]
-; SSE2-NEXT: [[SHR_8:%.*]] = lshr i8 [[A_SROA_9_8_EXTRACT_TRUNC]], 1
-; SSE2-NEXT: [[SHR5_8:%.*]] = lshr i8 [[B_SROA_9_8_EXTRACT_TRUNC]], 1
-; SSE2-NEXT: [[NARROW_8:%.*]] = add nuw i8 [[SHR5_8]], [[SHR_8]]
-; SSE2-NEXT: [[OR21_8:%.*]] = or i8 [[B_SROA_9_8_EXTRACT_TRUNC]], [[A_SROA_9_8_EXTRACT_TRUNC]]
-; SSE2-NEXT: [[TMP8:%.*]] = and i8 [[OR21_8]], 1
-; SSE2-NEXT: [[ADD12_8:%.*]] = add nuw i8 [[NARROW_8]], [[TMP8]]
-; SSE2-NEXT: [[SHR_9:%.*]] = lshr i8 [[A_SROA_11_8_EXTRACT_TRUNC]], 1
-; SSE2-NEXT: [[SHR5_9:%.*]] = lshr i8 [[B_SROA_11_8_EXTRACT_TRUNC]], 1
-; SSE2-NEXT: [[NARROW_9:%.*]] = add nuw i8 [[SHR5_9]], [[SHR_9]]
-; SSE2-NEXT: [[OR21_9:%.*]] = or i8 [[B_SROA_11_8_EXTRACT_TRUNC]], [[A_SROA_11_8_EXTRACT_TRUNC]]
-; SSE2-NEXT: [[TMP9:%.*]] = and i8 [[OR21_9]], 1
-; SSE2-NEXT: [[ADD12_9:%.*]] = add nuw i8 [[NARROW_9]], [[TMP9]]
-; SSE2-NEXT: [[SHR_10:%.*]] = lshr i8 [[A_SROA_12_8_EXTRACT_TRUNC]], 1
-; SSE2-NEXT: [[SHR5_10:%.*]] = lshr i8 [[B_SROA_12_8_EXTRACT_TRUNC]], 1
-; SSE2-NEXT: [[NARROW_10:%.*]] = add nuw i8 [[SHR5_10]], [[SHR_10]]
-; SSE2-NEXT: [[OR21_10:%.*]] = or i8 [[B_SROA_12_8_EXTRACT_TRUNC]], [[A_SROA_12_8_EXTRACT_TRUNC]]
-; SSE2-NEXT: [[TMP10:%.*]] = and i8 [[OR21_10]], 1
-; SSE2-NEXT: [[ADD12_10:%.*]] = add nuw i8 [[NARROW_10]], [[TMP10]]
-; SSE2-NEXT: [[SHR_11:%.*]] = lshr i8 [[A_SROA_13_8_EXTRACT_TRUNC]], 1
-; SSE2-NEXT: [[SHR5_11:%.*]] = lshr i8 [[B_SROA_13_8_EXTRACT_TRUNC]], 1
-; SSE2-NEXT: [[NARROW_11:%.*]] = add nuw i8 [[SHR5_11]], [[SHR_11]]
-; SSE2-NEXT: [[OR21_11:%.*]] = or i8 [[B_SROA_13_8_EXTRACT_TRUNC]], [[A_SROA_13_8_EXTRACT_TRUNC]]
-; SSE2-NEXT: [[TMP11:%.*]] = and i8 [[OR21_11]], 1
-; SSE2-NEXT: [[ADD12_11:%.*]] = add nuw i8 [[NARROW_11]], [[TMP11]]
-; SSE2-NEXT: [[SHR_12:%.*]] = lshr i8 [[A_SROA_14_8_EXTRACT_TRUNC]], 1
-; SSE2-NEXT: [[SHR5_12:%.*]] = lshr i8 [[B_SROA_14_8_EXTRACT_TRUNC]], 1
-; SSE2-NEXT: [[NARROW_12:%.*]] = add nuw i8 [[SHR5_12]], [[SHR_12]]
-; SSE2-NEXT: [[OR21_12:%.*]] = or i8 [[B_SROA_14_8_EXTRACT_TRUNC]], [[A_SROA_14_8_EXTRACT_TRUNC]]
-; SSE2-NEXT: [[TMP12:%.*]] = and i8 [[OR21_12]], 1
-; SSE2-NEXT: [[ADD12_12:%.*]] = add nuw i8 [[NARROW_12]], [[TMP12]]
-; SSE2-NEXT: [[SHR_13:%.*]] = lshr i8 [[A_SROA_15_8_EXTRACT_TRUNC]], 1
-; SSE2-NEXT: [[SHR5_13:%.*]] = lshr i8 [[B_SROA_15_8_EXTRACT_TRUNC]], 1
-; SSE2-NEXT: [[NARROW_13:%.*]] = add nuw i8 [[SHR5_13]], [[SHR_13]]
-; SSE2-NEXT: [[OR21_13:%.*]] = or i8 [[B_SROA_15_8_EXTRACT_TRUNC]], [[A_SROA_15_8_EXTRACT_TRUNC]]
-; SSE2-NEXT: [[TMP13:%.*]] = and i8 [[OR21_13]], 1
-; SSE2-NEXT: [[ADD12_13:%.*]] = add nuw i8 [[NARROW_13]], [[TMP13]]
-; SSE2-NEXT: [[SHR_14:%.*]] = lshr i8 [[A_SROA_16_8_EXTRACT_TRUNC]], 1
-; SSE2-NEXT: [[SHR5_14:%.*]] = lshr i8 [[B_SROA_16_8_EXTRACT_TRUNC]], 1
-; SSE2-NEXT: [[NARROW_14:%.*]] = add nuw i8 [[SHR5_14]], [[SHR_14]]
-; SSE2-NEXT: [[OR21_14:%.*]] = or i8 [[B_SROA_16_8_EXTRACT_TRUNC]], [[A_SROA_16_8_EXTRACT_TRUNC]]
-; SSE2-NEXT: [[TMP14:%.*]] = and i8 [[OR21_14]], 1
-; SSE2-NEXT: [[ADD12_14:%.*]] = add nuw i8 [[NARROW_14]], [[TMP14]]
-; SSE2-NEXT: [[SHR_15:%.*]] = lshr i8 [[A_SROA_17_8_EXTRACT_TRUNC]], 1
-; SSE2-NEXT: [[SHR5_15:%.*]] = lshr i8 [[B_SROA_17_8_EXTRACT_TRUNC]], 1
-; SSE2-NEXT: [[NARROW_15:%.*]] = add nuw i8 [[SHR5_15]], [[SHR_15]]
-; SSE2-NEXT: [[OR21_15:%.*]] = or i8 [[B_SROA_17_8_EXTRACT_TRUNC]], [[A_SROA_17_8_EXTRACT_TRUNC]]
-; SSE2-NEXT: [[TMP15:%.*]] = and i8 [[OR21_15]], 1
-; SSE2-NEXT: [[ADD12_15:%.*]] = add nuw i8 [[NARROW_15]], [[TMP15]]
-; SSE2-NEXT: [[RETVAL_SROA_8_0_INSERT_EXT:%.*]] = zext i8 [[ADD12_7]] to i64
-; SSE2-NEXT: [[RETVAL_SROA_8_0_INSERT_SHIFT:%.*]] = shl nuw i64 [[RETVAL_SROA_8_0_INSERT_EXT]], 56
-; SSE2-NEXT: [[RETVAL_SROA_7_0_INSERT_EXT:%.*]] = zext i8 [[ADD12_6]] to i64
-; SSE2-NEXT: [[RETVAL_SROA_7_0_INSERT_SHIFT:%.*]] = shl nuw nsw i64 [[RETVAL_SROA_7_0_INSERT_EXT]], 48
-; SSE2-NEXT: [[RETVAL_SROA_7_0_INSERT_INSERT:%.*]] = or disjoint i64 [[RETVAL_SROA_8_0_INSERT_SHIFT]], [[RETVAL_SROA_7_0_INSERT_SHIFT]]
-; SSE2-NEXT: [[RETVAL_SROA_6_0_INSERT_EXT:%.*]] = zext i8 [[ADD12_5]] to i64
-; SSE2-NEXT: [[RETVAL_SROA_6_0_INSERT_SHIFT:%.*]] = shl nuw nsw i64 [[RETVAL_SROA_6_0_INSERT_EXT]], 40
-; SSE2-NEXT: [[RETVAL_SROA_6_0_INSERT_INSERT:%.*]] = or disjoint i64 [[RETVAL_SROA_7_0_INSERT_INSERT]], [[RETVAL_SROA_6_0_INSERT_SHIFT]]
-; SSE2-NEXT: [[RETVAL_SROA_5_0_INSERT_EXT:%.*]] = zext i8 [[ADD12_4]] to i64
-; SSE2-NEXT: [[RETVAL_SROA_5_0_INSERT_SHIFT:%.*]] = shl nuw nsw i64 [[RETVAL_SROA_5_0_INSERT_EXT]], 32
-; SSE2-NEXT: [[RETVAL_SROA_5_0_INSERT_INSERT:%.*]] = or disjoint i64 [[RETVAL_SROA_6_0_INSERT_INSERT]], [[RETVAL_SROA_5_0_INSERT_SHIFT]]
-; SSE2-NEXT: [[RETVAL_SROA_4_0_INSERT_EXT:%.*]] = zext i8 [[ADD12_3]] to i64
-; SSE2-NEXT: [[RETVAL_SROA_4_0_INSERT_SHIFT:%.*]] = shl nuw nsw i64 [[RETVAL_SROA_4_0_INSERT_EXT]], 24
-; SSE2-NEXT: [[RETVAL_SROA_4_0_INSERT_INSERT:%.*]] = or disjoint i64 [[RETVAL_SROA_5_0_INSERT_INSERT]], [[RETVAL_SROA_4_0_INSERT_SHIFT]]
-; SSE2-NEXT: [[RETVAL_SROA_3_0_INSERT_EXT:%.*]] = zext i8 [[ADD12_2]] to i64
-; SSE2-NEXT: [[RETVAL_SROA_3_0_INSERT_SHIFT:%.*]] = shl nuw nsw i64 [[RETVAL_SROA_3_0_INSERT_EXT]], 16
-; SSE2-NEXT: [[RETVAL_SROA_2_0_INSERT_EXT:%.*]] = zext i8 [[ADD12_1]] to i64
-; SSE2-NEXT: [[RETVAL_SROA_2_0_INSERT_SHIFT:%.*]] = shl nuw nsw i64 [[RETVAL_SROA_2_0_INSERT_EXT]], 8
-; SSE2-NEXT: [[RETVAL_SROA_2_0_INSERT_MASK:%.*]] = or disjoint i64 [[RETVAL_SROA_4_0_INSERT_INSERT]], [[RETVAL_SROA_3_0_INSERT_SHIFT]]
-; SSE2-NEXT: [[RETVAL_SROA_0_0_INSERT_EXT:%.*]] = zext i8 [[ADD12]] to i64
-; SSE2-NEXT: [[RETVAL_SROA_0_0_INSERT_MASK:%.*]] = or i64 [[RETVAL_SROA_2_0_INSERT_MASK]], [[RETVAL_SROA_2_0_INSERT_SHIFT]]
-; SSE2-NEXT: [[RETVAL_SROA_0_0_INSERT_INSERT:%.*]] = or i64 [[RETVAL_SROA_0_0_INSERT_MASK]], [[RETVAL_SROA_0_0_INSERT_EXT]]
-; SSE2-NEXT: [[DOTFCA_0_INSERT:%.*]] = insertvalue { i64, i64 } poison, i64 [[RETVAL_SROA_0_0_INSERT_INSERT]], 0
-; SSE2-NEXT: [[RETVAL_SROA_17_8_INSERT_EXT:%.*]] = zext i8 [[ADD12_15]] to i64
-; SSE2-NEXT: [[RETVAL_SROA_17_8_INSERT_SHIFT:%.*]] = shl nuw i64 [[RETVAL_SROA_17_8_INSERT_EXT]], 56
-; SSE2-NEXT: [[RETVAL_SROA_16_8_INSERT_EXT:%.*]] = zext i8 [[ADD12_14]] to i64
-; SSE2-NEXT: [[RETVAL_SROA_16_8_INSERT_SHIFT:%.*]] = shl nuw nsw i64 [[RETVAL_SROA_16_8_INSERT_EXT]], 48
-; SSE2-NEXT: [[RETVAL_SROA_16_8_INSERT_INSERT:%.*]] = or disjoint i64 [[RETVAL_SROA_17_8_INSERT_SHIFT]], [[RETVAL_SROA_16_8_INSERT_SHIFT]]
-; SSE2-NEXT: [[RETVAL_SROA_15_8_INSERT_EXT:%.*]] = zext i8 [[ADD12_13]] to i64
-; SSE2-NEXT: [[RETVAL_SROA_15_8_INSERT_SHIFT:%.*]] = shl nuw nsw i64 [[RETVAL_SROA_15_8_INSERT_EXT]], 40
-; SSE2-NEXT: [[RETVAL_SROA_15_8_INSERT_INSERT:%.*]] = or disjoint i64 [[RETVAL_SROA_16_8_INSERT_INSERT]], [[RETVAL_SROA_15_8_INSERT_SHIFT]]
-; SSE2-NEXT: [[RETVAL_SROA_14_8_INSERT_EXT:%.*]] = zext i8 [[ADD12_12]] to i64
-; SSE2-NEXT: [[RETVAL_SROA_14_8_INSERT_SHIFT:%.*]] = shl nuw nsw i64 [[RETVAL_SROA_14_8_INSERT_EXT]], 32
-; SSE2-NEXT: [[RETVAL_SROA_14_8_INSERT_INSERT:%.*]] = or disjoint i64 [[RETVAL_SROA_15_8_INSERT_INSERT]], [[RETVAL_SROA_14_8_INSERT_SHIFT]]
-; SSE2-NEXT: [[RETVAL_SROA_13_8_INSERT_EXT:%.*]] = zext i8 [[ADD12_11]] to i64
-; SSE2-NEXT: [[RETVAL_SROA_13_8_INSERT_SHIFT:%.*]] = shl nuw nsw i64 [[RETVAL_SROA_13_8_INSERT_EXT]], 24
-; SSE2-NEXT: [[RETVAL_SROA_13_8_INSERT_INSERT:%.*]] = or disjoint i64 [[RETVAL_SROA_14_8_INSERT_INSERT]], [[RETVAL_SROA_13_8_INSERT_SHIFT]]
-; SSE2-NEXT: [[RETVAL_SROA_12_8_INSERT_EXT:%.*]] = zext i8 [[ADD12_10]] to i64
-; SSE2-NEXT: [[RETVAL_SROA_12_8_INSERT_SHIFT:%.*]] = shl nuw nsw i64 [[RETVAL_SROA_12_8_INSERT_EXT]], 16
-; SSE2-NEXT: [[RETVAL_SROA_11_8_INSERT_EXT:%.*]] = zext i8 [[ADD12_9]] to i64
-; SSE2-NEXT: [[RETVAL_SROA_11_8_INSERT_SHIFT:%.*]] = shl nuw nsw i64 [[RETVAL_SROA_11_8_INSERT_EXT]], 8
-; SSE2-NEXT: [[RETVAL_SROA_11_8_INSERT_MASK:%.*]] = or disjoint i64 [[RETVAL_SROA_13_8_INSERT_INSERT]], [[RETVAL_SROA_12_8_INSERT_SHIFT]]
-; SSE2-NEXT: [[RETVAL_SROA_9_8_INSERT_EXT:%.*]] = zext i8 [[ADD12_8]] to i64
-; SSE2-NEXT: [[RETVAL_SROA_9_8_INSERT_MASK:%.*]] = or i64 [[RETVAL_SROA_11_8_INSERT_MASK]], [[RETVAL_SROA_11_8_INSERT_SHIFT]]
-; SSE2-NEXT: [[RETVAL_SROA_9_8_INSERT_INSERT:%.*]] = or i64 [[RETVAL_SROA_9_8_INSERT_MASK]], [[RETVAL_SROA_9_8_INSERT_EXT]]
-; SSE2-NEXT: [[DOTFCA_1_INSERT:%.*]] = insertvalue { i64, i64 } [[DOTFCA_0_INSERT]], i64 [[RETVAL_SROA_9_8_INSERT_INSERT]], 1
-; SSE2-NEXT: ret { i64, i64 } [[DOTFCA_1_INSERT]]
-;
-; SSE4-LABEL: @avgr_16_u8_alt(
-; SSE4-NEXT: entry:
-; SSE4-NEXT: [[TMP0:%.*]] = insertelement <8 x i64> poison, i64 [[A_COERCE0:%.*]], i64 0
-; SSE4-NEXT: [[TMP1:%.*]] = shufflevector <8 x i64> [[TMP0]], <8 x i64> poison, <8 x i32> zeroinitializer
-; SSE4-NEXT: [[TMP2:%.*]] = lshr <8 x i64> [[TMP1]], <i64 0, i64 8, i64 16, i64 24, i64 32, i64 40, i64 48, i64 56>
-; SSE4-NEXT: [[TMP3:%.*]] = trunc <8 x i64> [[TMP2]] to <8 x i8>
-; SSE4-NEXT: [[TMP4:%.*]] = insertelement <8 x i64> poison, i64 [[B_COERCE0:%.*]], i64 0
-; SSE4-NEXT: [[TMP5:%.*]] = shufflevector <8 x i64> [[TMP4]], <8 x i64> poison, <8 x i32> zeroinitializer
-; SSE4-NEXT: [[TMP6:%.*]] = lshr <8 x i64> [[TMP5]], <i64 0, i64 8, i64 16, i64 24, i64 32, i64 40, i64 48, i64 56>
-; SSE4-NEXT: [[TMP7:%.*]] = trunc <8 x i64> [[TMP6]] to <8 x i8>
-; SSE4-NEXT: [[TMP8:%.*]] = lshr <8 x i8> [[TMP3]], splat (i8 1)
-; SSE4-NEXT: [[TMP9:%.*]] = lshr <8 x i8> [[TMP7]], splat (i8 1)
-; SSE4-NEXT: [[TMP10:%.*]] = add nuw <8 x i8> [[TMP9]], [[TMP8]]
-; SSE4-NEXT: [[TMP11:%.*]] = or <8 x i8> [[TMP7]], [[TMP3]]
-; SSE4-NEXT: [[TMP12:%.*]] = and <8 x i8> [[TMP11]], splat (i8 1)
-; SSE4-NEXT: [[TMP13:%.*]] = add nuw <8 x i8> [[TMP10]], [[TMP12]]
-; SSE4-NEXT: [[VEC2STRUCT_SLOT_SROA_0_0_VEC_EXTRACT:%.*]] = bitcast <8 x i8> [[TMP13]] to i64
-; SSE4-NEXT: [[DOTFCA_0_INSERT:%.*]] = insertvalue { i64, i64 } poison, i64 [[VEC2STRUCT_SLOT_SROA_0_0_VEC_EXTRACT]], 0
-; SSE4-NEXT: [[TMP17:%.*]] = insertelement <8 x i64> poison, i64 [[A_COERCE1:%.*]], i64 0
-; SSE4-NEXT: [[TMP18:%.*]] = shufflevector <8 x i64> [[TMP17]], <8 x i64> poison, <8 x i32> zeroinitializer
-; SSE4-NEXT: [[TMP19:%.*]] = lshr <8 x i64> [[TMP18]], <i64 0, i64 8, i64 16, i64 24, i64 32, i64 40, i64 48, i64 56>
-; SSE4-NEXT: [[TMP20:%.*]] = trunc <8 x i64> [[TMP19]] to <8 x i8>
-; SSE4-NEXT: [[TMP21:%.*]] = insertelement <8 x i64> poison, i64 [[B_COERCE1:%.*]], i64 0
-; SSE4-NEXT: [[TMP22:%.*]] = shufflevector <8 x i64> [[TMP21]], <8 x i64> poison, <8 x i32> zeroinitializer
-; SSE4-NEXT: [[TMP23:%.*]] = lshr <8 x i64> [[TMP22]], <i64 0, i64 8, i64 16, i64 24, i64 32, i64 40, i64 48, i64 56>
-; SSE4-NEXT: [[TMP24:%.*]] = trunc <8 x i64> [[TMP23]] to <8 x i8>
-; SSE4-NEXT: [[TMP25:%.*]] = lshr <8 x i8> [[TMP20]], splat (i8 1)
-; SSE4-NEXT: [[TMP26:%.*]] = lshr <8 x i8> [[TMP24]], splat (i8 1)
-; SSE4-NEXT: [[TMP27:%.*]] = add nuw <8 x i8> [[TMP26]], [[TMP25]]
-; SSE4-NEXT: [[TMP28:%.*]] = or <8 x i8> [[TMP24]], [[TMP20]]
-; SSE4-NEXT: [[TMP29:%.*]] = and <8 x i8> [[TMP28]], splat (i8 1)
-; SSE4-NEXT: [[TMP30:%.*]] = add nuw <8 x i8> [[TMP27]], [[TMP29]]
-; SSE4-NEXT: [[TMP49:%.*]] = bitcast <8 x i8> [[TMP30]] to i64
-; SSE4-NEXT: [[DOTFCA_1_INSERT:%.*]] = insertvalue { i64, i64 } [[DOTFCA_0_INSERT]], i64 [[TMP49]], 1
-; SSE4-NEXT: ret { i64, i64 } [[DOTFCA_1_INSERT]]
-;
-; AVX-LABEL: @avgr_16_u8_alt(
-; AVX-NEXT: entry:
-; AVX-NEXT: [[TMP0:%.*]] = insertelement <8 x i64> poison, i64 [[A_COERCE0:%.*]], i64 0
-; AVX-NEXT: [[TMP1:%.*]] = shufflevector <8 x i64> [[TMP0]], <8 x i64> poison, <8 x i32> zeroinitializer
-; AVX-NEXT: [[TMP2:%.*]] = lshr <8 x i64> [[TMP1]], <i64 0, i64 8, i64 16, i64 24, i64 32, i64 40, i64 48, i64 56>
-; AVX-NEXT: [[TMP3:%.*]] = trunc <8 x i64> [[TMP2]] to <8 x i8>
-; AVX-NEXT: [[TMP4:%.*]] = insertelement <8 x i64> poison, i64 [[B_COERCE0:%.*]], i64 0
-; AVX-NEXT: [[TMP5:%.*]] = shufflevector <8 x i64> [[TMP4]], <8 x i64> poison, <8 x i32> zeroinitializer
-; AVX-NEXT: [[TMP6:%.*]] = lshr <8 x i64> [[TMP5]], <i64 0, i64 8, i64 16, i64 24, i64 32, i64 40, i64 48, i64 56>
-; AVX-NEXT: [[TMP7:%.*]] = trunc <8 x i64> [[TMP6]] to <8 x i8>
-; AVX-NEXT: [[TMP8:%.*]] = lshr <8 x i8> [[TMP3]], splat (i8 1)
-; AVX-NEXT: [[TMP9:%.*]] = lshr <8 x i8> [[TMP7]], splat (i8 1)
-; AVX-NEXT: [[TMP10:%.*]] = add nuw <8 x i8> [[TMP9]], [[TMP8]]
-; AVX-NEXT: [[TMP11:%.*]] = or <8 x i8> [[TMP7]], [[TMP3]]
-; AVX-NEXT: [[TMP12:%.*]] = and <8 x i8> [[TMP11]], splat (i8 1)
-; AVX-NEXT: [[TMP13:%.*]] = add nuw <8 x i8> [[TMP10]], [[TMP12]]
-; AVX-NEXT: [[TMP16:%.*]] = bitcast <8 x i8> [[TMP13]] to i64
-; AVX-NEXT: [[DOTFCA_0_INSERT:%.*]] = insertvalue { i64, i64 } poison, i64 [[TMP16]], 0
-; AVX-NEXT: [[TMP17:%.*]] = insertelement <8 x i64> poison, i64 [[A_COERCE1:%.*]], i64 0
-; AVX-NEXT: [[TMP18:%.*]] = shufflevector <8 x i64> [[TMP17]], <8 x i64> poison, <8 x i32> zeroinitializer
-; AVX-NEXT: [[TMP19:%.*]] = lshr <8 x i64> [[TMP18]], <i64 0, i64 8, i64 16, i64 24, i64 32, i64 40, i64 48, i64 56>
-; AVX-NEXT: [[TMP20:%.*]] = trunc <8 x i64> [[TMP19]] to <8 x i8>
-; AVX-NEXT: [[TMP21:%.*]] = insertelement <8 x i64> poison, i64 [[B_COERCE1:%.*]], i64 0
-; AVX-NEXT: [[TMP22:%.*]] = shufflevector <8 x i64> [[TMP21]], <8 x i64> poison, <8 x i32> zeroinitializer
-; AVX-NEXT: [[TMP23:%.*]] = lshr <8 x i64> [[TMP22]], <i64 0, i64 8, i64 16, i64 24, i64 32, i64 40, i64 48, i64 56>
-; AVX-NEXT: [[TMP24:%.*]] = trunc <8 x i64> [[TMP23]] to <8 x i8>
-; AVX-NEXT: [[TMP25:%.*]] = lshr <8 x i8> [[TMP20]], splat (i8 1)
-; AVX-NEXT: [[TMP26:%.*]] = lshr <8 x i8> [[TMP24]], splat (i8 1)
-; AVX-NEXT: [[TMP27:%.*]] = add nuw <8 x i8> [[TMP26]], [[TMP25]]
-; AVX-NEXT: [[TMP28:%.*]] = or <8 x i8> [[TMP24]], [[TMP20]]
-; AVX-NEXT: [[TMP29:%.*]] = and <8 x i8> [[TMP28]], splat (i8 1)
-; AVX-NEXT: [[TMP30:%.*]] = add nuw <8 x i8> [[TMP27]], [[TMP29]]
-; AVX-NEXT: [[TMP33:%.*]] = bitcast <8 x i8> [[TMP30]] to i64
-; AVX-NEXT: [[DOTFCA_1_INSERT:%.*]] = insertvalue { i64, i64 } [[DOTFCA_0_INSERT]], i64 [[TMP33]], 1
-; AVX-NEXT: ret { i64, i64 } [[DOTFCA_1_INSERT]]
+; CHECK-LABEL: @avgr_16_u8_alt(
+; CHECK-NEXT: entry:
+; CHECK-NEXT: [[TMP0:%.*]] = bitcast i64 [[A_COERCE0:%.*]] to <8 x i8>
+; CHECK-NEXT: [[TMP1:%.*]] = lshr <8 x i8> [[TMP0]], splat (i8 1)
+; CHECK-NEXT: [[TMP2:%.*]] = bitcast i64 [[B_COERCE0:%.*]] to <8 x i8>
+; CHECK-NEXT: [[TMP3:%.*]] = lshr <8 x i8> [[TMP2]], splat (i8 1)
+; CHECK-NEXT: [[TMP4:%.*]] = add nuw <8 x i8> [[TMP3]], [[TMP1]]
+; CHECK-NEXT: [[TMP5:%.*]] = or <8 x i8> [[TMP2]], [[TMP0]]
+; CHECK-NEXT: [[TMP6:%.*]] = and <8 x i8> [[TMP5]], splat (i8 1)
+; CHECK-NEXT: [[TMP7:%.*]] = add nuw <8 x i8> [[TMP4]], [[TMP6]]
+; CHECK-NEXT: [[TMP8:%.*]] = bitcast <8 x i8> [[TMP7]] to i64
+; CHECK-NEXT: [[DOTFCA_0_INSERT:%.*]] = insertvalue { i64, i64 } poison, i64 [[TMP8]], 0
+; CHECK-NEXT: [[TMP9:%.*]] = bitcast i64 [[A_COERCE1:%.*]] to <8 x i8>
+; CHECK-NEXT: [[TMP10:%.*]] = lshr <8 x i8> [[TMP9]], splat (i8 1)
+; CHECK-NEXT: [[TMP11:%.*]] = bitcast i64 [[B_COERCE1:%.*]] to <8 x i8>
+; CHECK-NEXT: [[TMP12:%.*]] = lshr <8 x i8> [[TMP11]], splat (i8 1)
+; CHECK-NEXT: [[TMP13:%.*]] = add nuw <8 x i8> [[TMP12]], [[TMP10]]
+; CHECK-NEXT: [[TMP14:%.*]] = or <8 x i8> [[TMP11]], [[TMP9]]
+; CHECK-NEXT: [[TMP15:%.*]] = and <8 x i8> [[TMP14]], splat (i8 1)
+; CHECK-NEXT: [[TMP16:%.*]] = add nuw <8 x i8> [[TMP13]], [[TMP15]]
+; CHECK-NEXT: [[TMP17:%.*]] = bitcast <8 x i8> [[TMP16]] to i64
+; CHECK-NEXT: [[DOTFCA_1_INSERT:%.*]] = insertvalue { i64, i64 } [[DOTFCA_0_INSERT]], i64 [[TMP17]], 1
+; CHECK-NEXT: ret { i64, i64 } [[DOTFCA_1_INSERT]]
;
entry:
%retval = alloca %"struct.std::array16", align 1
@@ -787,210 +492,29 @@ for.body: ; preds = %for.cond
}
define { i64, i64 } @avgr_8_u16_alt(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerce0, i64 %b.coerce1) {
-; SSE2-LABEL: @avgr_8_u16_alt(
-; SSE2-NEXT: entry:
-; SSE2-NEXT: [[A_SROA_4_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE0:%.*]], 48
-; SSE2-NEXT: [[A_SROA_3_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE0]], 32
-; SSE2-NEXT: [[A_SROA_2_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE0]], 16
-; SSE2-NEXT: [[B_SROA_4_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE0:%.*]], 48
-; SSE2-NEXT: [[B_SROA_3_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE0]], 32
-; SSE2-NEXT: [[B_SROA_2_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE0]], 16
-; SSE2-NEXT: [[TMP0:%.*]] = trunc i64 [[A_COERCE0]] to i16
-; SSE2-NEXT: [[TMP14:%.*]] = insertelement <2 x i16> poison, i16 [[TMP0]], i64 0
-; SSE2-NEXT: [[TMP17:%.*]] = trunc i64 [[A_SROA_2_0_EXTRACT_SHIFT]] to i16
-; SSE2-NEXT: [[TMP18:%.*]] = insertelement <2 x i16> [[TMP14]], i16 [[TMP17]], i64 1
-; SSE2-NEXT: [[TMP4:%.*]] = trunc i64 [[B_COERCE0]] to i16
-; SSE2-NEXT: [[TMP5:%.*]] = insertelement <2 x i16> poison, i16 [[TMP4]], i64 0
-; SSE2-NEXT: [[TMP6:%.*]] = trunc i64 [[B_SROA_2_0_EXTRACT_SHIFT]] to i16
-; SSE2-NEXT: [[TMP7:%.*]] = insertelement <2 x i16> [[TMP5]], i16 [[TMP6]], i64 1
-; SSE2-NEXT: [[TMP19:%.*]] = lshr <2 x i16> [[TMP18]], splat (i16 1)
-; SSE2-NEXT: [[TMP21:%.*]] = lshr <2 x i16> [[TMP7]], splat (i16 1)
-; SSE2-NEXT: [[TMP10:%.*]] = add nuw <2 x i16> [[TMP21]], [[TMP19]]
-; SSE2-NEXT: [[TMP37:%.*]] = shufflevector <2 x i16> [[TMP7]], <2 x i16> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
-; SSE2-NEXT: [[TMP1:%.*]] = shufflevector <2 x i16> [[TMP18]], <2 x i16> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
-; SSE2-NEXT: [[TMP44:%.*]] = shufflevector <2 x i16> [[TMP10]], <2 x i16> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
-; SSE2-NEXT: [[A_SROA_9_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE1:%.*]], 48
-; SSE2-NEXT: [[A_SROA_8_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE1]], 32
-; SSE2-NEXT: [[A_SROA_7_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE1]], 16
-; SSE2-NEXT: [[B_SROA_9_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE2:%.*]], 48
-; SSE2-NEXT: [[B_SROA_8_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE2]], 32
-; SSE2-NEXT: [[B_SROA_7_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE2]], 16
-; SSE2-NEXT: [[B_SROA_9_8_EXTRACT_TRUNC:%.*]] = trunc i64 [[B_SROA_8_8_EXTRACT_SHIFT]] to i16
-; SSE2-NEXT: [[B_SROA_8_8_EXTRACT_TRUNC:%.*]] = trunc nuw i64 [[B_SROA_9_8_EXTRACT_SHIFT]] to i16
-; SSE2-NEXT: [[B_SROA_3_0_EXTRACT_TRUNC:%.*]] = trunc i64 [[B_SROA_3_0_EXTRACT_SHIFT]] to i16
-; SSE2-NEXT: [[B_SROA_4_0_EXTRACT_TRUNC:%.*]] = trunc nuw i64 [[B_SROA_4_0_EXTRACT_SHIFT]] to i16
-; SSE2-NEXT: [[TMP38:%.*]] = insertelement <4 x i16> [[TMP37]], i16 [[B_SROA_3_0_EXTRACT_TRUNC]], i64 2
-; SSE2-NEXT: [[TMP39:%.*]] = insertelement <4 x i16> [[TMP38]], i16 [[B_SROA_4_0_EXTRACT_TRUNC]], i64 3
-; SSE2-NEXT: [[NARROW:%.*]] = trunc i64 [[A_SROA_8_8_EXTRACT_SHIFT]] to i16
-; SSE2-NEXT: [[A_SROA_5_8_EXTRACT_TRUNC:%.*]] = trunc nuw i64 [[A_SROA_9_8_EXTRACT_SHIFT]] to i16
-; SSE2-NEXT: [[B_SROA_2_0_EXTRACT_TRUNC:%.*]] = trunc i64 [[A_SROA_3_0_EXTRACT_SHIFT]] to i16
-; SSE2-NEXT: [[B_SROA_0_0_EXTRACT_TRUNC:%.*]] = trunc nuw i64 [[A_SROA_4_0_EXTRACT_SHIFT]] to i16
-; SSE2-NEXT: [[TMP2:%.*]] = insertelement <4 x i16> [[TMP1]], i16 [[B_SROA_2_0_EXTRACT_TRUNC]], i64 2
-; SSE2-NEXT: [[TMP3:%.*]] = insertelement <4 x i16> [[TMP2]], i16 [[B_SROA_0_0_EXTRACT_TRUNC]], i64 3
-; SSE2-NEXT: [[TMP8:%.*]] = or <4 x i16> [[TMP39]], [[TMP3]]
-; SSE2-NEXT: [[TMP9:%.*]] = and <4 x i16> [[TMP8]], splat (i16 1)
-; SSE2-NEXT: [[TMP48:%.*]] = insertelement <4 x i16> poison, i16 [[B_SROA_0_0_EXTRACT_TRUNC]], i64 0
-; SSE2-NEXT: [[TMP49:%.*]] = insertelement <4 x i16> [[TMP48]], i16 [[A_SROA_5_8_EXTRACT_TRUNC]], i64 1
-; SSE2-NEXT: [[TMP12:%.*]] = insertelement <4 x i16> [[TMP49]], i16 [[B_SROA_2_0_EXTRACT_TRUNC]], i64 2
-; SSE2-NEXT: [[TMP13:%.*]] = insertelement <4 x i16> [[TMP12]], i16 [[NARROW]], i64 3
-; SSE2-NEXT: [[TMP50:%.*]] = lshr <4 x i16> [[TMP13]], splat (i16 1)
-; SSE2-NEXT: [[TMP51:%.*]] = insertelement <4 x i16> poison, i16 [[B_SROA_4_0_EXTRACT_TRUNC]], i64 0
-; SSE2-NEXT: [[TMP52:%.*]] = insertelement <4 x i16> [[TMP51]], i16 [[B_SROA_8_8_EXTRACT_TRUNC]], i64 1
-; SSE2-NEXT: [[TMP53:%.*]] = insertelement <4 x i16> [[TMP52]], i16 [[B_SROA_3_0_EXTRACT_TRUNC]], i64 2
-; SSE2-NEXT: [[TMP54:%.*]] = insertelement <4 x i16> [[TMP53]], i16 [[B_SROA_9_8_EXTRACT_TRUNC]], i64 3
-; SSE2-NEXT: [[TMP55:%.*]] = lshr <4 x i16> [[TMP54]], splat (i16 1)
-; SSE2-NEXT: [[TMP56:%.*]] = add nuw <4 x i16> [[TMP55]], [[TMP50]]
-; SSE2-NEXT: [[TMP57:%.*]] = shufflevector <4 x i16> [[TMP44]], <4 x i16> [[TMP56]], <4 x i32> <i32 0, i32 1, i32 6, i32 4>
-; SSE2-NEXT: [[TMP15:%.*]] = add nuw <4 x i16> [[TMP57]], [[TMP9]]
-; SSE2-NEXT: [[TMP16:%.*]] = bitcast <4 x i16> [[TMP15]] to i64
-; SSE2-NEXT: [[DOTFCA_0_INSERT:%.*]] = insertvalue { i64, i64 } poison, i64 [[TMP16]], 0
-; SSE2-NEXT: [[TMP40:%.*]] = trunc i64 [[B_COERCE1]] to i16
-; SSE2-NEXT: [[TMP41:%.*]] = insertelement <2 x i16> poison, i16 [[TMP40]], i64 0
-; SSE2-NEXT: [[TMP42:%.*]] = trunc i64 [[A_SROA_7_8_EXTRACT_SHIFT]] to i16
-; SSE2-NEXT: [[TMP27:%.*]] = insertelement <2 x i16> [[TMP41]], i16 [[TMP42]], i64 1
-; SSE2-NEXT: [[TMP28:%.*]] = trunc i64 [[B_COERCE2]] to i16
-; SSE2-NEXT: [[TMP29:%.*]] = insertelement <2 x i16> poison, i16 [[TMP28]], i64 0
-; SSE2-NEXT: [[TMP45:%.*]] = trunc i64 [[B_SROA_7_8_EXTRACT_SHIFT]] to i16
-; SSE2-NEXT: [[TMP46:%.*]] = insertelement <2 x i16> [[TMP29]], i16 [[TMP45]], i64 1
-; SSE2-NEXT: [[TMP32:%.*]] = lshr <2 x i16> [[TMP27]], splat (i16 1)
-; SSE2-NEXT: [[TMP47:%.*]] = lshr <2 x i16> [[TMP46]], splat (i16 1)
-; SSE2-NEXT: [[TMP34:%.*]] = add nuw <2 x i16> [[TMP47]], [[TMP32]]
-; SSE2-NEXT: [[TMP35:%.*]] = shufflevector <2 x i16> [[TMP46]], <2 x i16> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
-; SSE2-NEXT: [[TMP36:%.*]] = insertelement <4 x i16> [[TMP35]], i16 [[B_SROA_9_8_EXTRACT_TRUNC]], i64 2
-; SSE2-NEXT: [[TMP20:%.*]] = insertelement <4 x i16> [[TMP36]], i16 [[B_SROA_8_8_EXTRACT_TRUNC]], i64 3
-; SSE2-NEXT: [[TMP22:%.*]] = shufflevector <2 x i16> [[TMP27]], <2 x i16> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
-; SSE2-NEXT: [[TMP23:%.*]] = insertelement <4 x i16> [[TMP22]], i16 [[NARROW]], i64 2
-; SSE2-NEXT: [[TMP24:%.*]] = insertelement <4 x i16> [[TMP23]], i16 [[A_SROA_5_8_EXTRACT_TRUNC]], i64 3
-; SSE2-NEXT: [[TMP25:%.*]] = or <4 x i16> [[TMP20]], [[TMP24]]
-; SSE2-NEXT: [[TMP26:%.*]] = and <4 x i16> [[TMP25]], splat (i16 1)
-; SSE2-NEXT: [[TMP43:%.*]] = shufflevector <2 x i16> [[TMP34]], <2 x i16> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
-; SSE2-NEXT: [[TMP30:%.*]] = shufflevector <4 x i16> [[TMP43]], <4 x i16> [[TMP56]], <4 x i32> <i32 0, i32 1, i32 7, i32 5>
-; SSE2-NEXT: [[TMP31:%.*]] = add nuw <4 x i16> [[TMP30]], [[TMP26]]
-; SSE2-NEXT: [[TMP33:%.*]] = bitcast <4 x i16> [[TMP31]] to i64
-; SSE2-NEXT: [[DOTFCA_1_INSERT:%.*]] = insertvalue { i64, i64 } [[DOTFCA_0_INSERT]], i64 [[TMP33]], 1
-; SSE2-NEXT: ret { i64, i64 } [[DOTFCA_1_INSERT]]
-;
-; SSE4-LABEL: @avgr_8_u16_alt(
-; SSE4-NEXT: entry:
-; SSE4-NEXT: [[A_SROA_4_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE0:%.*]], 48
-; SSE4-NEXT: [[A_SROA_3_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE0]], 32
-; SSE4-NEXT: [[A_SROA_2_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE0]], 16
-; SSE4-NEXT: [[B_SROA_0_0_EXTRACT_TRUNC:%.*]] = trunc nuw i64 [[A_SROA_4_0_EXTRACT_SHIFT]] to i16
-; SSE4-NEXT: [[B_SROA_2_0_EXTRACT_TRUNC:%.*]] = trunc i64 [[A_SROA_3_0_EXTRACT_SHIFT]] to i16
-; SSE4-NEXT: [[B_SROA_4_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE0:%.*]], 48
-; SSE4-NEXT: [[B_SROA_3_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE0]], 32
-; SSE4-NEXT: [[B_SROA_2_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE0]], 16
-; SSE4-NEXT: [[B_SROA_4_0_EXTRACT_TRUNC:%.*]] = trunc nuw i64 [[B_SROA_4_0_EXTRACT_SHIFT]] to i16
-; SSE4-NEXT: [[B_SROA_3_0_EXTRACT_TRUNC:%.*]] = trunc i64 [[B_SROA_3_0_EXTRACT_SHIFT]] to i16
-; SSE4-NEXT: [[SHR5:%.*]] = lshr i16 [[B_SROA_0_0_EXTRACT_TRUNC]], 1
-; SSE4-NEXT: [[SHR5_1:%.*]] = lshr i16 [[B_SROA_2_0_EXTRACT_TRUNC]], 1
-; SSE4-NEXT: [[SHR5_3:%.*]] = lshr i16 [[B_SROA_4_0_EXTRACT_TRUNC]], 1
-; SSE4-NEXT: [[SHR5_2:%.*]] = lshr i16 [[B_SROA_3_0_EXTRACT_TRUNC]], 1
-; SSE4-NEXT: [[NARROW:%.*]] = add nuw i16 [[SHR5_3]], [[SHR5]]
-; SSE4-NEXT: [[NARROW_1:%.*]] = add nuw i16 [[SHR5_2]], [[SHR5_1]]
-; SSE4-NEXT: [[TMP0:%.*]] = trunc i64 [[A_COERCE0]] to i16
-; SSE4-NEXT: [[TMP14:%.*]] = insertelement <2 x i16> poison, i16 [[TMP0]], i64 0
-; SSE4-NEXT: [[TMP17:%.*]] = trunc i64 [[A_SROA_2_0_EXTRACT_SHIFT]] to i16
-; SSE4-NEXT: [[TMP18:%.*]] = insertelement <2 x i16> [[TMP14]], i16 [[TMP17]], i64 1
-; SSE4-NEXT: [[TMP4:%.*]] = trunc i64 [[B_COERCE0]] to i16
-; SSE4-NEXT: [[TMP5:%.*]] = insertelement <2 x i16> poison, i16 [[TMP4]], i64 0
-; SSE4-NEXT: [[TMP6:%.*]] = trunc i64 [[B_SROA_2_0_EXTRACT_SHIFT]] to i16
-; SSE4-NEXT: [[TMP7:%.*]] = insertelement <2 x i16> [[TMP5]], i16 [[TMP6]], i64 1
-; SSE4-NEXT: [[TMP19:%.*]] = lshr <2 x i16> [[TMP18]], splat (i16 1)
-; SSE4-NEXT: [[TMP21:%.*]] = lshr <2 x i16> [[TMP7]], splat (i16 1)
-; SSE4-NEXT: [[TMP10:%.*]] = add nuw <2 x i16> [[TMP21]], [[TMP19]]
-; SSE4-NEXT: [[TMP37:%.*]] = shufflevector <2 x i16> [[TMP7]], <2 x i16> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
-; SSE4-NEXT: [[TMP38:%.*]] = insertelement <4 x i16> [[TMP37]], i16 [[B_SROA_3_0_EXTRACT_TRUNC]], i64 2
-; SSE4-NEXT: [[TMP39:%.*]] = insertelement <4 x i16> [[TMP38]], i16 [[B_SROA_4_0_EXTRACT_TRUNC]], i64 3
-; SSE4-NEXT: [[TMP1:%.*]] = shufflevector <2 x i16> [[TMP18]], <2 x i16> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
-; SSE4-NEXT: [[TMP2:%.*]] = insertelement <4 x i16> [[TMP1]], i16 [[B_SROA_2_0_EXTRACT_TRUNC]], i64 2
-; SSE4-NEXT: [[TMP3:%.*]] = insertelement <4 x i16> [[TMP2]], i16 [[B_SROA_0_0_EXTRACT_TRUNC]], i64 3
-; SSE4-NEXT: [[TMP8:%.*]] = or <4 x i16> [[TMP39]], [[TMP3]]
-; SSE4-NEXT: [[TMP9:%.*]] = and <4 x i16> [[TMP8]], splat (i16 1)
-; SSE4-NEXT: [[TMP11:%.*]] = shufflevector <2 x i16> [[TMP10]], <2 x i16> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
-; SSE4-NEXT: [[TMP12:%.*]] = insertelement <4 x i16> [[TMP11]], i16 [[NARROW_1]], i64 2
-; SSE4-NEXT: [[TMP13:%.*]] = insertelement <4 x i16> [[TMP12]], i16 [[NARROW]], i64 3
-; SSE4-NEXT: [[TMP15:%.*]] = add nuw <4 x i16> [[TMP13]], [[TMP9]]
-; SSE4-NEXT: [[TMP16:%.*]] = bitcast <4 x i16> [[TMP15]] to i64
-; SSE4-NEXT: [[DOTFCA_0_INSERT:%.*]] = insertvalue { i64, i64 } poison, i64 [[TMP16]], 0
-; SSE4-NEXT: [[A_SROA_9_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE1:%.*]], 48
-; SSE4-NEXT: [[B_SROA_8_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE1]], 32
-; SSE4-NEXT: [[A_SROA_7_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE1]], 16
-; SSE4-NEXT: [[A_SROA_5_8_EXTRACT_TRUNC:%.*]] = trunc nuw i64 [[A_SROA_9_8_EXTRACT_SHIFT]] to i16
-; SSE4-NEXT: [[A_SROA_7_8_EXTRACT_TRUNC:%.*]] = trunc i64 [[B_SROA_8_8_EXTRACT_SHIFT]] to i16
-; SSE4-NEXT: [[B_SROA_9_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE2:%.*]], 48
-; SSE4-NEXT: [[B_SROA_8_8_EXTRACT_SHIFT1:%.*]] = lshr i64 [[B_COERCE2]], 32
-; SSE4-NEXT: [[B_SROA_7_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE2]], 16
-; SSE4-NEXT: [[B_SROA_8_8_EXTRACT_TRUNC:%.*]] = trunc nuw i64 [[B_SROA_9_8_EXTRACT_SHIFT]] to i16
-; SSE4-NEXT: [[B_SROA_9_8_EXTRACT_TRUNC:%.*]] = trunc i64 [[B_SROA_8_8_EXTRACT_SHIFT1]] to i16
-; SSE4-NEXT: [[SHR_6:%.*]] = lshr i16 [[A_SROA_5_8_EXTRACT_TRUNC]], 1
-; SSE4-NEXT: [[SHR_7:%.*]] = lshr i16 [[A_SROA_7_8_EXTRACT_TRUNC]], 1
-; SSE4-NEXT: [[SHR5_6:%.*]] = lshr i16 [[B_SROA_8_8_EXTRACT_TRUNC]], 1
-; SSE4-NEXT: [[SHR5_7:%.*]] = lshr i16 [[B_SROA_9_8_EXTRACT_TRUNC]], 1
-; SSE4-NEXT: [[NARROW_6:%.*]] = add nuw i16 [[SHR5_6]], [[SHR_6]]
-; SSE4-NEXT: [[NARROW_7:%.*]] = add nuw i16 [[SHR5_7]], [[SHR_7]]
-; SSE4-NEXT: [[TMP40:%.*]] = trunc i64 [[B_COERCE1]] to i16
-; SSE4-NEXT: [[TMP41:%.*]] = insertelement <2 x i16> poison, i16 [[TMP40]], i64 0
-; SSE4-NEXT: [[TMP42:%.*]] = trunc i64 [[A_SROA_7_8_EXTRACT_SHIFT]] to i16
-; SSE4-NEXT: [[TMP27:%.*]] = insertelement <2 x i16> [[TMP41]], i16 [[TMP42]], i64 1
-; SSE4-NEXT: [[TMP28:%.*]] = trunc i64 [[B_COERCE2]] to i16
-; SSE4-NEXT: [[TMP29:%.*]] = insertelement <2 x i16> poison, i16 [[TMP28]], i64 0
-; SSE4-NEXT: [[TMP45:%.*]] = trunc i64 [[B_SROA_7_8_EXTRACT_SHIFT]] to i16
-; SSE4-NEXT: [[TMP46:%.*]] = insertelement <2 x i16> [[TMP29]], i16 [[TMP45]], i64 1
-; SSE4-NEXT: [[TMP32:%.*]] = lshr <2 x i16> [[TMP27]], splat (i16 1)
-; SSE4-NEXT: [[TMP47:%.*]] = lshr <2 x i16> [[TMP46]], splat (i16 1)
-; SSE4-NEXT: [[TMP34:%.*]] = add nuw <2 x i16> [[TMP47]], [[TMP32]]
-; SSE4-NEXT: [[TMP35:%.*]] = shufflevector <2 x i16> [[TMP46]], <2 x i16> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
-; SSE4-NEXT: [[TMP36:%.*]] = insertelement <4 x i16> [[TMP35]], i16 [[B_SROA_9_8_EXTRACT_TRUNC]], i64 2
-; SSE4-NEXT: [[TMP20:%.*]] = insertelement <4 x i16> [[TMP36]], i16 [[B_SROA_8_8_EXTRACT_TRUNC]], i64 3
-; SSE4-NEXT: [[TMP22:%.*]] = shufflevector <2 x i16> [[TMP27]], <2 x i16> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
-; SSE4-NEXT: [[TMP23:%.*]] = insertelement <4 x i16> [[TMP22]], i16 [[A_SROA_7_8_EXTRACT_TRUNC]], i64 2
-; SSE4-NEXT: [[TMP24:%.*]] = insertelement <4 x i16> [[TMP23]], i16 [[A_SROA_5_8_EXTRACT_TRUNC]], i64 3
-; SSE4-NEXT: [[TMP25:%.*]] = or <4 x i16> [[TMP20]], [[TMP24]]
-; SSE4-NEXT: [[TMP26:%.*]] = and <4 x i16> [[TMP25]], splat (i16 1)
-; SSE4-NEXT: [[TMP43:%.*]] = shufflevector <2 x i16> [[TMP34]], <2 x i16> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
-; SSE4-NEXT: [[TMP44:%.*]] = insertelement <4 x i16> [[TMP43]], i16 [[NARROW_7]], i64 2
-; SSE4-NEXT: [[TMP30:%.*]] = insertelement <4 x i16> [[TMP44]], i16 [[NARROW_6]], i64 3
-; SSE4-NEXT: [[TMP31:%.*]] = add nuw <4 x i16> [[TMP30]], [[TMP26]]
-; SSE4-NEXT: [[TMP33:%.*]] = bitcast <4 x i16> [[TMP31]] to i64
-; SSE4-NEXT: [[DOTFCA_1_INSERT:%.*]] = insertvalue { i64, i64 } [[DOTFCA_0_INSERT]], i64 [[TMP33]], 1
-; SSE4-NEXT: ret { i64, i64 } [[DOTFCA_1_INSERT]]
-;
-; AVX-LABEL: @avgr_8_u16_alt(
-; AVX-NEXT: entry:
-; AVX-NEXT: [[TMP0:%.*]] = insertelement <4 x i64> poison, i64 [[A_COERCE0:%.*]], i64 0
-; AVX-NEXT: [[TMP1:%.*]] = shufflevector <4 x i64> [[TMP0]], <4 x i64> poison, <4 x i32> zeroinitializer
-; AVX-NEXT: [[TMP2:%.*]] = lshr <4 x i64> [[TMP1]], <i64 0, i64 16, i64 32, i64 48>
-; AVX-NEXT: [[TMP3:%.*]] = trunc <4 x i64> [[TMP2]] to <4 x i16>
-; AVX-NEXT: [[TMP4:%.*]] = insertelement <4 x i64> poison, i64 [[B_COERCE0:%.*]], i64 0
-; AVX-NEXT: [[TMP5:%.*]] = shufflevector <4 x i64> [[TMP4]], <4 x i64> poison, <4 x i32> zeroinitializer
-; AVX-NEXT: [[TMP6:%.*]] = lshr <4 x i64> [[TMP5]], <i64 0, i64 16, i64 32, i64 48>
-; AVX-NEXT: [[TMP7:%.*]] = trunc <4 x i64> [[TMP6]] to <4 x i16>
-; AVX-NEXT: [[TMP8:%.*]] = lshr <4 x i16> [[TMP3]], splat (i16 1)
-; AVX-NEXT: [[TMP9:%.*]] = lshr <4 x i16> [[TMP7]], splat (i16 1)
-; AVX-NEXT: [[TMP10:%.*]] = add nuw <4 x i16> [[TMP9]], [[TMP8]]
-; AVX-NEXT: [[TMP11:%.*]] = or <4 x i16> [[TMP7]], [[TMP3]]
-; AVX-NEXT: [[TMP12:%.*]] = and <4 x i16> [[TMP11]], splat (i16 1)
-; AVX-NEXT: [[TMP13:%.*]] = add nuw <4 x i16> [[TMP10]], [[TMP12]]
-; AVX-NEXT: [[TMP14:%.*]] = bitcast <4 x i16> [[TMP13]] to i64
-; AVX-NEXT: [[DOTFCA_0_INSERT:%.*]] = insertvalue { i64, i64 } poison, i64 [[TMP14]], 0
-; AVX-NEXT: [[TMP15:%.*]] = insertelement <4 x i64> poison, i64 [[A_COERCE1:%.*]], i64 0
-; AVX-NEXT: [[TMP16:%.*]] = shufflevector <4 x i64> [[TMP15]], <4 x i64> poison, <4 x i32> zeroinitializer
-; AVX-NEXT: [[TMP17:%.*]] = lshr <4 x i64> [[TMP16]], <i64 0, i64 16, i64 32, i64 48>
-; AVX-NEXT: [[TMP18:%.*]] = trunc <4 x i64> [[TMP17]] to <4 x i16>
-; AVX-NEXT: [[TMP19:%.*]] = insertelement <4 x i64> poison, i64 [[B_COERCE1:%.*]], i64 0
-; AVX-NEXT: [[TMP20:%.*]] = shufflevector <4 x i64> [[TMP19]], <4 x i64> poison, <4 x i32> zeroinitializer
-; AVX-NEXT: [[TMP21:%.*]] = lshr <4 x i64> [[TMP20]], <i64 0, i64 16, i64 32, i64 48>
-; AVX-NEXT: [[TMP22:%.*]] = trunc <4 x i64> [[TMP21]] to <4 x i16>
-; AVX-NEXT: [[TMP23:%.*]] = lshr <4 x i16> [[TMP18]], splat (i16 1)
-; AVX-NEXT: [[TMP24:%.*]] = lshr <4 x i16> [[TMP22]], splat (i16 1)
-; AVX-NEXT: [[TMP25:%.*]] = add nuw <4 x i16> [[TMP24]], [[TMP23]]
-; AVX-NEXT: [[TMP26:%.*]] = or <4 x i16> [[TMP22]], [[TMP18]]
-; AVX-NEXT: [[TMP27:%.*]] = and <4 x i16> [[TMP26]], splat (i16 1)
-; AVX-NEXT: [[TMP28:%.*]] = add nuw <4 x i16> [[TMP25]], [[TMP27]]
-; AVX-NEXT: [[TMP29:%.*]] = bitcast <4 x i16> [[TMP28]] to i64
-; AVX-NEXT: [[DOTFCA_1_INSERT:%.*]] = insertvalue { i64, i64 } [[DOTFCA_0_INSERT]], i64 [[TMP29]], 1
-; AVX-NEXT: ret { i64, i64 } [[DOTFCA_1_INSERT]]
+; CHECK-LABEL: @avgr_8_u16_alt(
+; CHECK-NEXT: entry:
+; CHECK-NEXT: [[TMP0:%.*]] = bitcast i64 [[A_COERCE0:%.*]] to <4 x i16>
+; CHECK-NEXT: [[TMP1:%.*]] = lshr <4 x i16> [[TMP0]], splat (i16 1)
+; CHECK-NEXT: [[TMP2:%.*]] = bitcast i64 [[B_COERCE0:%.*]] to <4 x i16>
+; CHECK-NEXT: [[TMP3:%.*]] = lshr <4 x i16> [[TMP2]], splat (i16 1)
+; CHECK-NEXT: [[TMP4:%.*]] = add nuw <4 x i16> [[TMP3]], [[TMP1]]
+; CHECK-NEXT: [[TMP5:%.*]] = or <4 x i16> [[TMP2]], [[TMP0]]
+; CHECK-NEXT: [[TMP6:%.*]] = and <4 x i16> [[TMP5]], splat (i16 1)
+; CHECK-NEXT: [[TMP7:%.*]] = add nuw <4 x i16> [[TMP4]], [[TMP6]]
+; CHECK-NEXT: [[TMP8:%.*]] = bitcast <4 x i16> [[TMP7]] to i64
+; CHECK-NEXT: [[DOTFCA_0_INSERT:%.*]] = insertvalue { i64, i64 } poison, i64 [[TMP8]], 0
+; CHECK-NEXT: [[TMP9:%.*]] = bitcast i64 [[A_COERCE1:%.*]] to <4 x i16>
+; CHECK-NEXT: [[TMP10:%.*]] = lshr <4 x i16> [[TMP9]], splat (i16 1)
+; CHECK-NEXT: [[TMP11:%.*]] = bitcast i64 [[B_COERCE1:%.*]] to <4 x i16>
+; CHECK-NEXT: [[TMP12:%.*]] = lshr <4 x i16> [[TMP11]], splat (i16 1)
+; CHECK-NEXT: [[TMP13:%.*]] = add nuw <4 x i16> [[TMP12]], [[TMP10]]
+; CHECK-NEXT: [[TMP14:%.*]] = or <4 x i16> [[TMP11]], [[TMP9]]
+; CHECK-NEXT: [[TMP15:%.*]] = and <4 x i16> [[TMP14]], splat (i16 1)
+; CHECK-NEXT: [[TMP16:%.*]] = add nuw <4 x i16> [[TMP13]], [[TMP15]]
+; CHECK-NEXT: [[TMP17:%.*]] = bitcast <4 x i16> [[TMP16]] to i64
+; CHECK-NEXT: [[DOTFCA_1_INSERT:%.*]] = insertvalue { i64, i64 } [[DOTFCA_0_INSERT]], i64 [[TMP17]], 1
+; CHECK-NEXT: ret { i64, i64 } [[DOTFCA_1_INSERT]]
;
entry:
%retval = alloca %"struct.std::array8", align 2
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/extracted-subfields.ll b/llvm/test/Transforms/SLPVectorizer/X86/extracted-subfields.ll
index 7c58c1881aa515..f025d672aa2688 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/extracted-subfields.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/extracted-subfields.ll
@@ -10,34 +10,9 @@ define i16 @sum8_i64(ptr %x) {
; CHECK-SAME: ptr [[X:%.*]]) {
; CHECK-NEXT: [[START:.*:]]
; CHECK-NEXT: [[L:%.*]] = load i64, ptr [[X]], align 8
-; CHECK-NEXT: [[T0:%.*]] = trunc i64 [[L]] to i16
-; CHECK-NEXT: [[B0:%.*]] = and i16 [[T0]], 255
-; CHECK-NEXT: [[T1:%.*]] = trunc i64 [[L]] to i16
-; CHECK-NEXT: [[B1:%.*]] = lshr i16 [[T1]], 8
-; CHECK-NEXT: [[A1:%.*]] = add nuw nsw i16 [[B0]], [[B1]]
-; CHECK-NEXT: [[S2:%.*]] = lshr i64 [[L]], 16
-; CHECK-NEXT: [[T2:%.*]] = trunc i64 [[S2]] to i16
-; CHECK-NEXT: [[B2:%.*]] = and i16 [[T2]], 255
-; CHECK-NEXT: [[A2:%.*]] = add nuw nsw i16 [[A1]], [[B2]]
-; CHECK-NEXT: [[S3:%.*]] = lshr i64 [[L]], 24
-; CHECK-NEXT: [[T3:%.*]] = trunc i64 [[S3]] to i16
-; CHECK-NEXT: [[B3:%.*]] = and i16 [[T3]], 255
-; CHECK-NEXT: [[A3:%.*]] = add nuw nsw i16 [[A2]], [[B3]]
-; CHECK-NEXT: [[S4:%.*]] = lshr i64 [[L]], 32
-; CHECK-NEXT: [[T4:%.*]] = trunc i64 [[S4]] to i16
-; CHECK-NEXT: [[B4:%.*]] = and i16 [[T4]], 255
-; CHECK-NEXT: [[A4:%.*]] = add nuw nsw i16 [[A3]], [[B4]]
-; CHECK-NEXT: [[S5:%.*]] = lshr i64 [[L]], 40
-; CHECK-NEXT: [[T5:%.*]] = trunc i64 [[S5]] to i16
-; CHECK-NEXT: [[B5:%.*]] = and i16 [[T5]], 255
-; CHECK-NEXT: [[A5:%.*]] = add nuw nsw i16 [[A4]], [[B5]]
-; CHECK-NEXT: [[S6:%.*]] = lshr i64 [[L]], 48
-; CHECK-NEXT: [[T6:%.*]] = trunc i64 [[S6]] to i16
-; CHECK-NEXT: [[B6:%.*]] = and i16 [[T6]], 255
-; CHECK-NEXT: [[A6:%.*]] = add nuw nsw i16 [[A5]], [[B6]]
-; CHECK-NEXT: [[S7:%.*]] = lshr i64 [[L]], 56
-; CHECK-NEXT: [[B7:%.*]] = trunc i64 [[S7]] to i16
-; CHECK-NEXT: [[TMP2:%.*]] = add nuw nsw i16 [[A6]], [[B7]]
+; CHECK-NEXT: [[TMP0:%.*]] = bitcast i64 [[L]] to <8 x i8>
+; CHECK-NEXT: [[TMP1:%.*]] = zext <8 x i8> [[TMP0]] to <8 x i16>
+; CHECK-NEXT: [[TMP2:%.*]] = call i16 @llvm.vector.reduce.add.v8i16(<8 x i16> [[TMP1]])
; CHECK-NEXT: ret i16 [[TMP2]]
;
start:
@@ -81,30 +56,8 @@ define i16 @sum8_i64_ashr(ptr %x) {
; CHECK-SAME: ptr [[X:%.*]]) {
; CHECK-NEXT: [[START:.*:]]
; CHECK-NEXT: [[L:%.*]] = load i64, ptr [[X]], align 8
-; CHECK-NEXT: [[S7:%.*]] = ashr i64 [[L]], 56
-; CHECK-NEXT: [[S6:%.*]] = ashr i64 [[L]], 48
-; CHECK-NEXT: [[S5:%.*]] = ashr i64 [[L]], 40
-; CHECK-NEXT: [[S4:%.*]] = ashr i64 [[L]], 32
-; CHECK-NEXT: [[S3:%.*]] = ashr i64 [[L]], 24
-; CHECK-NEXT: [[S2:%.*]] = ashr i64 [[L]], 16
-; CHECK-NEXT: [[S1:%.*]] = ashr i64 [[L]], 8
-; CHECK-NEXT: [[T7:%.*]] = trunc i64 [[S7]] to i16
-; CHECK-NEXT: [[T6:%.*]] = trunc i64 [[S6]] to i16
-; CHECK-NEXT: [[T5:%.*]] = trunc i64 [[S5]] to i16
-; CHECK-NEXT: [[T4:%.*]] = trunc i64 [[S4]] to i16
-; CHECK-NEXT: [[T3:%.*]] = trunc i64 [[S3]] to i16
-; CHECK-NEXT: [[T2:%.*]] = trunc i64 [[S2]] to i16
-; CHECK-NEXT: [[T1:%.*]] = trunc i64 [[S1]] to i16
-; CHECK-NEXT: [[T0:%.*]] = trunc i64 [[L]] to i16
-; CHECK-NEXT: [[TMP0:%.*]] = insertelement <8 x i16> poison, i16 [[T0]], i64 0
-; CHECK-NEXT: [[TMP8:%.*]] = insertelement <8 x i16> [[TMP0]], i16 [[T1]], i64 1
-; CHECK-NEXT: [[TMP9:%.*]] = insertelement <8 x i16> [[TMP8]], i16 [[T2]], i64 2
-; CHECK-NEXT: [[TMP3:%.*]] = insertelement <8 x i16> [[TMP9]], i16 [[T3]], i64 3
-; CHECK-NEXT: [[TMP4:%.*]] = insertelement <8 x i16> [[TMP3]], i16 [[T4]], i64 4
-; CHECK-NEXT: [[TMP5:%.*]] = insertelement <8 x i16> [[TMP4]], i16 [[T5]], i64 5
-; CHECK-NEXT: [[TMP6:%.*]] = insertelement <8 x i16> [[TMP5]], i16 [[T6]], i64 6
-; CHECK-NEXT: [[TMP7:%.*]] = insertelement <8 x i16> [[TMP6]], i16 [[T7]], i64 7
-; CHECK-NEXT: [[TMP1:%.*]] = and <8 x i16> [[TMP7]], splat (i16 255)
+; CHECK-NEXT: [[TMP0:%.*]] = bitcast i64 [[L]] to <8 x i8>
+; CHECK-NEXT: [[TMP1:%.*]] = zext <8 x i8> [[TMP0]] to <8 x i16>
; CHECK-NEXT: [[TMP2:%.*]] = call i16 @llvm.vector.reduce.add.v8i16(<8 x i16> [[TMP1]])
; CHECK-NEXT: ret i16 [[TMP2]]
;
@@ -151,19 +104,9 @@ define i32 @sum4_halves_i64(ptr %p) {
; CHECK-SAME: ptr [[P:%.*]]) {
; CHECK-NEXT: [[START:.*:]]
; CHECK-NEXT: [[L:%.*]] = load i64, ptr [[P]], align 8
-; CHECK-NEXT: [[T0:%.*]] = trunc i64 [[L]] to i32
-; CHECK-NEXT: [[B0:%.*]] = and i32 [[T0]], 65535
-; CHECK-NEXT: [[S1:%.*]] = lshr i64 [[L]], 16
-; CHECK-NEXT: [[T1:%.*]] = trunc i64 [[S1]] to i32
-; CHECK-NEXT: [[B1:%.*]] = and i32 [[T1]], 65535
-; CHECK-NEXT: [[S2:%.*]] = lshr i64 [[L]], 32
-; CHECK-NEXT: [[T2:%.*]] = trunc i64 [[S2]] to i32
-; CHECK-NEXT: [[B2:%.*]] = and i32 [[T2]], 65535
-; CHECK-NEXT: [[S3:%.*]] = lshr i64 [[L]], 48
-; CHECK-NEXT: [[B3:%.*]] = trunc i64 [[S3]] to i32
-; CHECK-NEXT: [[A1:%.*]] = add nuw nsw i32 [[B0]], [[B1]]
-; CHECK-NEXT: [[A2:%.*]] = add nuw nsw i32 [[A1]], [[B2]]
-; CHECK-NEXT: [[A3:%.*]] = add nuw nsw i32 [[A2]], [[B3]]
+; CHECK-NEXT: [[TMP0:%.*]] = bitcast i64 [[L]] to <4 x i16>
+; CHECK-NEXT: [[TMP1:%.*]] = zext <4 x i16> [[TMP0]] to <4 x i32>
+; CHECK-NEXT: [[A3:%.*]] = call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> [[TMP1]])
; CHECK-NEXT: ret i32 [[A3]]
;
start:
@@ -192,19 +135,9 @@ define i16 @sum4_i32(ptr %p) {
; CHECK-SAME: ptr [[P:%.*]]) {
; CHECK-NEXT: [[START:.*:]]
; CHECK-NEXT: [[L:%.*]] = load i32, ptr [[P]], align 4
-; CHECK-NEXT: [[T0:%.*]] = trunc i32 [[L]] to i16
-; CHECK-NEXT: [[B0:%.*]] = and i16 [[T0]], 255
-; CHECK-NEXT: [[S1:%.*]] = lshr i32 [[L]], 8
-; CHECK-NEXT: [[T1:%.*]] = trunc i32 [[S1]] to i16
-; CHECK-NEXT: [[B1:%.*]] = and i16 [[T1]], 255
-; CHECK-NEXT: [[S2:%.*]] = lshr i32 [[L]], 16
-; CHECK-NEXT: [[T2:%.*]] = trunc i32 [[S2]] to i16
-; CHECK-NEXT: [[B2:%.*]] = and i16 [[T2]], 255
-; CHECK-NEXT: [[S3:%.*]] = lshr i32 [[L]], 24
-; CHECK-NEXT: [[B3:%.*]] = trunc i32 [[S3]] to i16
-; CHECK-NEXT: [[A1:%.*]] = add nuw nsw i16 [[B0]], [[B1]]
-; CHECK-NEXT: [[A2:%.*]] = add nuw nsw i16 [[A1]], [[B2]]
-; CHECK-NEXT: [[A3:%.*]] = add nuw nsw i16 [[A2]], [[B3]]
+; CHECK-NEXT: [[TMP0:%.*]] = bitcast i32 [[L]] to <4 x i8>
+; CHECK-NEXT: [[TMP1:%.*]] = zext <4 x i8> [[TMP0]] to <4 x i16>
+; CHECK-NEXT: [[A3:%.*]] = call i16 @llvm.vector.reduce.add.v4i16(<4 x i16> [[TMP1]])
; CHECK-NEXT: ret i16 [[A3]]
;
start:
@@ -272,23 +205,10 @@ define void @store_bytes_i64(ptr %p, ptr %out) {
; CHECK-SAME: ptr [[P:%.*]], ptr [[OUT:%.*]]) {
; CHECK-NEXT: [[START:.*:]]
; CHECK-NEXT: [[L:%.*]] = load i64, ptr [[P]], align 8
-; CHECK-NEXT: [[T0:%.*]] = trunc i64 [[L]] to i16
-; CHECK-NEXT: [[B0:%.*]] = and i16 [[T0]], 255
-; CHECK-NEXT: [[T1:%.*]] = trunc i64 [[L]] to i16
-; CHECK-NEXT: [[B1:%.*]] = lshr i16 [[T1]], 8
-; CHECK-NEXT: [[S2:%.*]] = lshr i64 [[L]], 16
-; CHECK-NEXT: [[T2:%.*]] = trunc i64 [[S2]] to i16
-; CHECK-NEXT: [[B2:%.*]] = and i16 [[T2]], 255
-; CHECK-NEXT: [[S3:%.*]] = lshr i64 [[L]], 24
-; CHECK-NEXT: [[T3:%.*]] = trunc i64 [[S3]] to i16
-; CHECK-NEXT: [[B3:%.*]] = and i16 [[T3]], 255
-; CHECK-NEXT: store i16 [[B0]], ptr [[OUT]], align 2
-; CHECK-NEXT: [[O1:%.*]] = getelementptr i16, ptr [[OUT]], i64 1
-; CHECK-NEXT: store i16 [[B1]], ptr [[O1]], align 2
-; CHECK-NEXT: [[O2:%.*]] = getelementptr i16, ptr [[OUT]], i64 2
-; CHECK-NEXT: store i16 [[B2]], ptr [[O2]], align 2
-; CHECK-NEXT: [[O3:%.*]] = getelementptr i16, ptr [[OUT]], i64 3
-; CHECK-NEXT: store i16 [[B3]], ptr [[O3]], align 2
+; CHECK-NEXT: [[TMP0:%.*]] = bitcast i64 [[L]] to <8 x i8>
+; CHECK-NEXT: [[TMP1:%.*]] = shufflevector <8 x i8> [[TMP0]], <8 x i8> poison, <4 x i32> <i32 0, i32 1, i32 2, i32 3>
+; CHECK-NEXT: [[TMP2:%.*]] = zext <4 x i8> [[TMP1]] to <4 x i16>
+; CHECK-NEXT: store <4 x i16> [[TMP2]], ptr [[OUT]], align 2
; CHECK-NEXT: ret void
;
start:
@@ -320,23 +240,9 @@ define void @store_halves_i64(ptr %p, ptr %out) {
; CHECK-SAME: ptr [[P:%.*]], ptr [[OUT:%.*]]) {
; CHECK-NEXT: [[START:.*:]]
; CHECK-NEXT: [[L:%.*]] = load i64, ptr [[P]], align 8
-; CHECK-NEXT: [[T0:%.*]] = trunc i64 [[L]] to i32
-; CHECK-NEXT: [[B0:%.*]] = and i32 [[T0]], 65535
-; CHECK-NEXT: [[S1:%.*]] = lshr i64 [[L]], 16
-; CHECK-NEXT: [[T1:%.*]] = trunc i64 [[S1]] to i32
-; CHECK-NEXT: [[B1:%.*]] = and i32 [[T1]], 65535
-; CHECK-NEXT: [[S2:%.*]] = lshr i64 [[L]], 32
-; CHECK-NEXT: [[T2:%.*]] = trunc i64 [[S2]] to i32
-; CHECK-NEXT: [[B2:%.*]] = and i32 [[T2]], 65535
-; CHECK-NEXT: [[S3:%.*]] = lshr i64 [[L]], 48
-; CHECK-NEXT: [[B3:%.*]] = trunc i64 [[S3]] to i32
-; CHECK-NEXT: store i32 [[B0]], ptr [[OUT]], align 4
-; CHECK-NEXT: [[O1:%.*]] = getelementptr i32, ptr [[OUT]], i64 1
-; CHECK-NEXT: store i32 [[B1]], ptr [[O1]], align 4
-; CHECK-NEXT: [[O2:%.*]] = getelementptr i32, ptr [[OUT]], i64 2
-; CHECK-NEXT: store i32 [[B2]], ptr [[O2]], align 4
-; CHECK-NEXT: [[O3:%.*]] = getelementptr i32, ptr [[OUT]], i64 3
-; CHECK-NEXT: store i32 [[B3]], ptr [[O3]], align 4
+; CHECK-NEXT: [[TMP0:%.*]] = bitcast i64 [[L]] to <4 x i16>
+; CHECK-NEXT: [[TMP1:%.*]] = zext <4 x i16> [[TMP0]] to <4 x i32>
+; CHECK-NEXT: store <4 x i32> [[TMP1]], ptr [[OUT]], align 4
; CHECK-NEXT: ret void
;
start:
@@ -392,30 +298,8 @@ define i16 @sum8_i64_scrambled(ptr %x) {
; CHECK-SAME: ptr [[X:%.*]]) {
; CHECK-NEXT: [[START:.*:]]
; CHECK-NEXT: [[L:%.*]] = load i64, ptr [[X]], align 8
-; CHECK-NEXT: [[S7:%.*]] = lshr i64 [[L]], 56
-; CHECK-NEXT: [[S4:%.*]] = lshr i64 [[L]], 32
-; CHECK-NEXT: [[S6:%.*]] = lshr i64 [[L]], 48
-; CHECK-NEXT: [[S2:%.*]] = lshr i64 [[L]], 16
-; CHECK-NEXT: [[S1:%.*]] = lshr i64 [[L]], 8
-; CHECK-NEXT: [[S5:%.*]] = lshr i64 [[L]], 40
-; CHECK-NEXT: [[S3:%.*]] = lshr i64 [[L]], 24
-; CHECK-NEXT: [[B7:%.*]] = trunc i64 [[S7]] to i16
-; CHECK-NEXT: [[T4:%.*]] = trunc i64 [[S4]] to i16
-; CHECK-NEXT: [[T6:%.*]] = trunc i64 [[S6]] to i16
-; CHECK-NEXT: [[T2:%.*]] = trunc i64 [[S2]] to i16
-; CHECK-NEXT: [[T1:%.*]] = trunc i64 [[S1]] to i16
-; CHECK-NEXT: [[T5:%.*]] = trunc i64 [[S5]] to i16
-; CHECK-NEXT: [[T0:%.*]] = trunc i64 [[L]] to i16
-; CHECK-NEXT: [[T3:%.*]] = trunc i64 [[S3]] to i16
-; CHECK-NEXT: [[TMP0:%.*]] = insertelement <8 x i16> poison, i16 [[T3]], i64 0
-; CHECK-NEXT: [[TMP8:%.*]] = insertelement <8 x i16> [[TMP0]], i16 [[T0]], i64 1
-; CHECK-NEXT: [[TMP9:%.*]] = insertelement <8 x i16> [[TMP8]], i16 [[T5]], i64 2
-; CHECK-NEXT: [[TMP3:%.*]] = insertelement <8 x i16> [[TMP9]], i16 [[T1]], i64 3
-; CHECK-NEXT: [[TMP4:%.*]] = insertelement <8 x i16> [[TMP3]], i16 [[T2]], i64 4
-; CHECK-NEXT: [[TMP5:%.*]] = insertelement <8 x i16> [[TMP4]], i16 [[T6]], i64 5
-; CHECK-NEXT: [[TMP6:%.*]] = insertelement <8 x i16> [[TMP5]], i16 [[T4]], i64 6
-; CHECK-NEXT: [[TMP7:%.*]] = insertelement <8 x i16> [[TMP6]], i16 [[B7]], i64 7
-; CHECK-NEXT: [[TMP1:%.*]] = and <8 x i16> [[TMP7]], <i16 255, i16 255, i16 255, i16 255, i16 255, i16 255, i16 255, i16 -1>
+; CHECK-NEXT: [[TMP0:%.*]] = bitcast i64 [[L]] to <8 x i8>
+; CHECK-NEXT: [[TMP1:%.*]] = zext <8 x i8> [[TMP0]] to <8 x i16>
; CHECK-NEXT: [[TMP2:%.*]] = call i16 @llvm.vector.reduce.add.v8i16(<8 x i16> [[TMP1]])
; CHECK-NEXT: ret i16 [[TMP2]]
;
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/reduce-or-bitpack.ll b/llvm/test/Transforms/SLPVectorizer/X86/reduce-or-bitpack.ll
index a5c51777871902..80c0c0c909798e 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/reduce-or-bitpack.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/reduce-or-bitpack.ll
@@ -8,11 +8,7 @@ define i32 @shl_and_pack(i32 %x) {
; CHECK-LABEL: define i32 @shl_and_pack(
; CHECK-SAME: i32 [[X:%.*]]) #[[ATTR0:[0-9]+]] {
; CHECK-NEXT: [[ENTRY:.*:]]
-; CHECK-NEXT: [[TMP0:%.*]] = insertelement <4 x i32> poison, i32 [[X]], i64 0
-; CHECK-NEXT: [[TMP1:%.*]] = shufflevector <4 x i32> [[TMP0]], <4 x i32> poison, <4 x i32> zeroinitializer
-; CHECK-NEXT: [[TMP2:%.*]] = lshr <4 x i32> [[TMP1]], <i32 0, i32 8, i32 16, i32 24>
-; CHECK-NEXT: [[TMP3:%.*]] = and <4 x i32> [[TMP2]], <i32 255, i32 255, i32 255, i32 -1>
-; CHECK-NEXT: [[TMP4:%.*]] = trunc <4 x i32> [[TMP3]] to <4 x i8>
+; CHECK-NEXT: [[TMP4:%.*]] = bitcast i32 [[X]] to <4 x i8>
; CHECK-NEXT: [[TMP6:%.*]] = bitcast <4 x i8> [[TMP4]] to i32
; CHECK-NEXT: ret i32 [[TMP6]]
;
More information about the llvm-commits
mailing list