[llvm] [SLP]Model or-reduction of masked shifted lanes as a bitfield pack (PR #219731)
Alexey Bataev via llvm-commits
llvm-commits at lists.llvm.org
Sun Sep 13 09:42:42 PDT 2026
https://github.com/alexey-bataev updated https://github.com/llvm/llvm-project/pull/219731
>From bdd550cda50721c234a809980c022944f01a1d1a Mon Sep 17 00:00:00 2001
From: Alexey Bataev <a.bataev at outlook.com>
Date: Sat, 29 Aug 2026 15:38:38 -0700
Subject: [PATCH] =?UTF-8?q?[=F0=9D=98=80=F0=9D=97=BD=F0=9D=97=BF]=20initia?=
=?UTF-8?q?l=20version?=
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
Created using spr 1.3.7
---
.../Transforms/Vectorize/SLPVectorizer.cpp | 316 +++++++++++++++---
.../SLPVectorizer/SLPCostAnalysis.cpp | 65 ++++
.../Vectorize/SLPVectorizer/SLPCostAnalysis.h | 15 +
.../Vectorize/SLPVectorizer/SLPUtils.cpp | 150 +++++++++
.../Vectorize/SLPVectorizer/SLPUtils.h | 41 +++
llvm/test/Transforms/PhaseOrdering/X86/avg.ll | 235 +++++++------
.../SLPVectorizer/X86/bad-reduction.ll | 18 +-
.../X86/load-merge-inseltpoison.ll | 4 +-
.../SLPVectorizer/X86/load-merge.ll | 4 +-
.../SLPVectorizer/X86/pr48879-sroa.ll | 66 ++--
.../SLPVectorizer/X86/reduce-or-bitpack.ll | 20 +-
.../SLPVectorizer/non-power-of-2-bswap.ll | 5 +-
12 files changed, 753 insertions(+), 186 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 5490fe24d5529..2b2e3c5758e53 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -904,7 +904,8 @@ class slpvectorizer::BoUpSLP {
(getRootNode().CombinedOp == TreeEntry::ReducedBitcast ||
getRootNode().CombinedOp == TreeEntry::ReducedBitcastBSwap ||
getRootNode().CombinedOp == TreeEntry::ReducedBitcastLoads ||
- getRootNode().CombinedOp == TreeEntry::ReducedBitcastBSwapLoads) &&
+ getRootNode().CombinedOp == TreeEntry::ReducedBitcastBSwapLoads ||
+ getRootNode().CombinedOp == TreeEntry::ReducedBitPack) &&
getRootNode().State == TreeEntry::Vectorize;
}
@@ -2933,6 +2934,20 @@ class slpvectorizer::BoUpSLP {
bool matchesShlZExt(const TreeEntry &TE, OrdersType &Order, bool &IsBSwap,
bool &ForLoads) const;
+ /// Computes the packing layout of an or reduction of masked shifted values
+ /// packing per-lane byte fields of the scalar result (and(shl(x, s), m) per
+ /// lane). The field analysis proves the disjointness of the fields, so the
+ /// reduction operations are not required to be marked disjoint.
+ std::optional<BitPackInfo> computeBitPackLayout(const TreeEntry &TE) const;
+
+ /// Checks if the tree entry is the root of such a reduction, the tree nodes
+ /// are in the expected state and it is profitable to model as a bitcast.
+ bool matchesBitPack(const TreeEntry &TE) const;
+
+ /// True if the node's lanes are a zext from i8, so compacting them to bytes
+ /// is free.
+ bool isByteZExtNode(const TreeEntry &TE) const;
+
/// Checks if the \p SelectTE matches zext+selects, which can be inversed for
/// better codegen in case like zext (icmp ne), select (icmp eq), ....
bool matchesInversedZExtSelect(
@@ -3086,6 +3101,7 @@ class slpvectorizer::BoUpSLP {
ReducedBitcastBSwap,
ReducedBitcastLoads,
ReducedBitcastBSwapLoads,
+ ReducedBitPack,
ReducedCmpBitcast,
};
CombinedOpcode CombinedOp = NotCombinedOp;
@@ -14827,6 +14843,126 @@ bool BoUpSLP::matchesShlZExt(const TreeEntry &TE, OrdersType &Order,
return BitcastCost < VecCost;
}
+std::optional<BitPackInfo>
+BoUpSLP::computeBitPackLayout(const TreeEntry &TE) const {
+ assert(TE.hasState() &&
+ (TE.getOpcode() == Instruction::And ||
+ TE.getOpcode() == Instruction::Shl) &&
+ "Expected And or Shl node.");
+ Type *ScalarTy = TE.getMainOp()->getType();
+ const unsigned BitWidth = DL->getTypeSizeInBits(ScalarTy);
+ const TreeEntry *ShlTE = &TE;
+ const TreeEntry *MaskTE = nullptr;
+ if (TE.getOpcode() == Instruction::And) {
+ MaskTE = getOperandEntry(&TE, 1);
+ ShlTE = getOperandEntry(&TE, 0);
+ }
+ const TreeEntry *ShiftTE = getOperandEntry(ShlTE, 1);
+ const TreeEntry *LhsTE = getOperandEntry(ShlTE, 0);
+
+ const unsigned VF = TE.getVectorFactor();
+ SmallVector<APInt> PossibleBits(VF), Masks(VF);
+ SmallVector<uint64_t> ShlAmts(VF);
+ auto IsCopyable = [](const TreeEntry *TE, unsigned Idx) {
+ return TE->hasCopyableElements() && TE->isCopyableElement(TE->Scalars[Idx]);
+ };
+ for (unsigned Idx : seq(VF)) {
+ uint64_t ShlAmt = 0;
+ const APInt *C;
+ if (!IsCopyable(ShlTE, Idx)) {
+ if (!match(ShiftTE->Scalars[Idx], m_APInt(C)) || C->uge(BitWidth))
+ return std::nullopt;
+ ShlAmt = C->getZExtValue();
+ }
+ APInt Mask = APInt::getAllOnes(BitWidth);
+ if (MaskTE && !IsCopyable(&TE, Idx)) {
+ if (!match(MaskTE->Scalars[Idx], m_APInt(C)))
+ return std::nullopt;
+ Mask = *C;
+ }
+ Value *X = LhsTE->Scalars[Idx];
+ if (isa<PoisonValue>(X)) {
+ PossibleBits[Idx] = APInt(BitWidth, 0);
+ } else {
+ PossibleBits[Idx] =
+ APInt::getLowBitsSet(BitWidth, getScalarMaxValue(X).getActiveBits()) &
+ ~computeKnownBits(X, SimplifyQuery(*DL)).Zero;
+ }
+ ShlAmts[Idx] = ShlAmt;
+ Masks[Idx] = Mask;
+ }
+ return computeBitPackInfo(BitWidth, PossibleBits, ShlAmts, Masks);
+}
+
+bool BoUpSLP::isByteZExtNode(const TreeEntry &TE) const {
+ return TE.getOpcode() == Instruction::ZExt &&
+ cast<ZExtInst>(TE.getMainOp())->getSrcTy()->isIntegerTy(8);
+}
+
+bool BoUpSLP::matchesBitPack(const TreeEntry &TE) const {
+ assert(TE.hasState() &&
+ (TE.getOpcode() == Instruction::And ||
+ TE.getOpcode() == Instruction::Shl) &&
+ "Expected And or Shl node.");
+ auto IsVectorizedOneUseNode = [&](const TreeEntry *N) {
+ return N->State == TreeEntry::Vectorize && !N->isAltShuffle() &&
+ N->ReorderIndices.empty() && N->ReuseShuffleIndices.empty() &&
+ !MinBWs.contains(N) && all_of(N->Scalars, [&](Value *V) {
+ return (N->hasCopyableElements() && N->isCopyableElement(V)) ||
+ V->hasOneUse();
+ });
+ };
+ if (!IsVectorizedOneUseNode(&TE))
+ return false;
+ Type *ScalarTy = TE.getMainOp()->getType();
+ if (!ScalarTy->isIntegerTy())
+ return false;
+ const unsigned BitWidth = DL->getTypeSizeInBits(ScalarTy);
+ if (!isPowerOf2_64(BitWidth))
+ return false;
+
+ const TreeEntry *ShlTE = &TE;
+ const TreeEntry *MaskTE = nullptr;
+ if (TE.getOpcode() == Instruction::And) {
+ MaskTE = getOperandEntry(&TE, 1);
+ ShlTE = getOperandEntry(&TE, 0);
+ if (!ShlTE->hasState() || ShlTE->getOpcode() != Instruction::Shl)
+ return false;
+ }
+ if (!IsVectorizedOneUseNode(ShlTE))
+ return false;
+ const TreeEntry *ShiftTE = getOperandEntry(ShlTE, 1);
+ const TreeEntry *LhsTE = getOperandEntry(ShlTE, 0);
+ if (!ShiftTE->isGather() || (MaskTE && !MaskTE->isGather()))
+ return false;
+ if (LhsTE->State != TreeEntry::Vectorize || LhsTE->isAltShuffle() ||
+ !LhsTE->ReorderIndices.empty() || !LhsTE->ReuseShuffleIndices.empty() ||
+ MinBWs.contains(LhsTE))
+ return false;
+ std::optional<BitPackInfo> Info = computeBitPackLayout(TE);
+ if (!Info)
+ return false;
+ auto *VecTy =
+ cast<VectorType>(getWidenedType(ScalarTy, TE.getVectorFactor()));
+ FastMathFlags FMF;
+ InstructionCost VecCost =
+ TTI->getArithmeticReductionCost(Instruction::Or, VecTy, FMF, CostKind) +
+ TTI->getArithmeticInstrCost(Instruction::Shl, VecTy, CostKind,
+ getOperandInfo(LhsTE->Scalars),
+ getOperandInfo(ShiftTE->Scalars), /*Args=*/{},
+ ShlTE->getMainOp(), TLI);
+ if (TE.getOpcode() == Instruction::And)
+ VecCost += TTI->getArithmeticInstrCost(
+ Instruction::And, VecTy, CostKind, getOperandInfo(ShlTE->Scalars),
+ getOperandInfo(MaskTE->Scalars), /*Args=*/{}, TE.getMainOp(), TLI);
+ unsigned ShiftWidth;
+ bool FreeByteTrunc = isByteZExtNode(*LhsTE);
+ InstructionCost PackCost =
+ getBitPackCost(*TTI, cast<FixedVectorType>(VecTy), ScalarTy, *Info,
+ FreeByteTrunc, CostKind, TLI, TE.getMainOp(), ShiftWidth);
+ return PackCost.isValid() && PackCost <= VecCost;
+}
+
bool BoUpSLP::matchesInversedZExtSelect(
const TreeEntry &SelectTE,
SmallVectorImpl<unsigned> &InversedCmpsIndices) const {
@@ -15464,8 +15600,10 @@ void BoUpSLP::transformNodes() {
}
break;
}
- case Instruction::Shl: {
- // Shl is not reassociated; guard since this case indexes operands 0/1.
+ case Instruction::Shl:
+ case Instruction::And: {
+ // Shl/And are not reassociated; guard since this case indexes operands
+ // 0/1.
if (E.getNumOperands() != 2)
break;
if (E.Idx != 0 || DL->isBigEndian())
@@ -15473,47 +15611,79 @@ void BoUpSLP::transformNodes() {
if (!UserIgnoreList)
break;
// Check that all reduction operands are disjoint or instructions.
+ if (E.getOpcode() == Instruction::Shl &&
+ all_of(*UserIgnoreList, [](Value *V) {
+ return match(V, m_DisjointOr(m_Value(), m_Value()));
+ })) {
+ OrdersType Order;
+ bool IsBSwap;
+ bool ForLoads;
+ if (matchesShlZExt(E, Order, IsBSwap, ForLoads)) {
+ // This node is a (reduced disjoint or) bitcast node.
+ TreeEntry::CombinedOpcode Code =
+ IsBSwap ? (ForLoads ? TreeEntry::ReducedBitcastBSwapLoads
+ : TreeEntry::ReducedBitcastBSwap)
+ : (ForLoads ? TreeEntry::ReducedBitcastLoads
+ : TreeEntry::ReducedBitcast);
+ E.CombinedOp = Code;
+ E.ReorderIndices = std::move(Order);
+ TreeEntry *ZExtEntry = getOperandEntry(&E, 0);
+ assert(ZExtEntry->UserTreeIndex &&
+ ZExtEntry->State == TreeEntry::Vectorize &&
+ ZExtEntry->getOpcode() == Instruction::ZExt &&
+ "Expected ZExt node.");
+ // The ZExt node is part of the combined node.
+ ZExtEntry->State = TreeEntry::CombinedVectorize;
+ ZExtEntry->CombinedOp = Code;
+ if (ForLoads) {
+ TreeEntry *LoadsEntry = getOperandEntry(ZExtEntry, 0);
+ assert(LoadsEntry->UserTreeIndex &&
+ LoadsEntry->State == TreeEntry::Vectorize &&
+ LoadsEntry->getOpcode() == Instruction::Load &&
+ "Expected Load node.");
+ // The Load node is part of the combined node.
+ LoadsEntry->State = TreeEntry::CombinedVectorize;
+ LoadsEntry->CombinedOp = Code;
+ }
+ TreeEntry *ConstEntry = getOperandEntry(&E, 1);
+ assert(ConstEntry->UserTreeIndex && ConstEntry->isGather() &&
+ "Expected ZExt node.");
+ // The ConstNode node is part of the combined node.
+ ConstEntry->State = TreeEntry::CombinedVectorize;
+ ConstEntry->CombinedOp = Code;
+ break;
+ }
+ }
+ // Check that all reduction operands are or instructions.
if (any_of(*UserIgnoreList, [](Value *V) {
- return !match(V, m_DisjointOr(m_Value(), m_Value()));
+ return !match(V, m_Or(m_Value(), m_Value()));
}))
break;
- OrdersType Order;
- bool IsBSwap;
- bool ForLoads;
- if (!matchesShlZExt(E, Order, IsBSwap, ForLoads))
+ if (!matchesBitPack(E))
break;
- // This node is a (reduced disjoint or) bitcast node.
- TreeEntry::CombinedOpcode Code =
- IsBSwap ? (ForLoads ? TreeEntry::ReducedBitcastBSwapLoads
- : TreeEntry::ReducedBitcastBSwap)
- : (ForLoads ? TreeEntry::ReducedBitcastLoads
- : TreeEntry::ReducedBitcast);
- E.CombinedOp = Code;
- E.ReorderIndices = std::move(Order);
- TreeEntry *ZExtEntry = getOperandEntry(&E, 0);
- assert(ZExtEntry->UserTreeIndex &&
- ZExtEntry->State == TreeEntry::Vectorize &&
- ZExtEntry->getOpcode() == Instruction::ZExt &&
- "Expected ZExt node.");
- // The ZExt node is part of the combined node.
- ZExtEntry->State = TreeEntry::CombinedVectorize;
- ZExtEntry->CombinedOp = Code;
- if (ForLoads) {
- TreeEntry *LoadsEntry = getOperandEntry(ZExtEntry, 0);
- assert(LoadsEntry->UserTreeIndex &&
- LoadsEntry->State == TreeEntry::Vectorize &&
- LoadsEntry->getOpcode() == Instruction::Load &&
- "Expected Load node.");
- // The Load node is part of the combined node.
- LoadsEntry->State = TreeEntry::CombinedVectorize;
- LoadsEntry->CombinedOp = Code;
- }
- TreeEntry *ConstEntry = getOperandEntry(&E, 1);
- assert(ConstEntry->UserTreeIndex && ConstEntry->isGather() &&
- "Expected ZExt node.");
- // The ConstNode node is part of the combined node.
- ConstEntry->State = TreeEntry::CombinedVectorize;
- ConstEntry->CombinedOp = Code;
+ // This node is a (reduced or) bit pack node.
+ E.CombinedOp = TreeEntry::ReducedBitPack;
+ TreeEntry *ShlEntry = &E;
+ if (E.getOpcode() == Instruction::And) {
+ ShlEntry = getOperandEntry(&E, 0);
+ assert(ShlEntry->UserTreeIndex &&
+ ShlEntry->State == TreeEntry::Vectorize &&
+ ShlEntry->getOpcode() == Instruction::Shl &&
+ "Expected Shl node.");
+ // The Shl node is part of the combined node.
+ ShlEntry->State = TreeEntry::CombinedVectorize;
+ ShlEntry->CombinedOp = TreeEntry::ReducedBitPack;
+ TreeEntry *MaskEntry = getOperandEntry(&E, 1);
+ assert(MaskEntry->UserTreeIndex && MaskEntry->isGather() &&
+ "Expected constants only.");
+ MaskEntry->State = TreeEntry::CombinedVectorize;
+ MaskEntry->CombinedOp = TreeEntry::ReducedBitPack;
+ }
+ TreeEntry *ShiftEntry = getOperandEntry(ShlEntry, 1);
+ assert(ShiftEntry->UserTreeIndex && ShiftEntry->isGather() &&
+ "Expected constants only.");
+ ShiftEntry->State = TreeEntry::CombinedVectorize;
+ ShiftEntry->CombinedOp = TreeEntry::ReducedBitPack;
break;
}
default:
@@ -16921,7 +17091,8 @@ BoUpSLP::getEntryCost(const TreeEntry *E, ArrayRef<Value *> VectorizedVals,
ScalarTy = IntegerType::get(F->getContext(), It->second.first);
if (VecTy)
ScalarTy = getWidenedType(ScalarTy, VecTy->getNumElements());
- } else if (E->Idx == 0 && isReducedBitcastRoot()) {
+ } else if (E->Idx == 0 && isReducedBitcastRoot() &&
+ getRootNode().CombinedOp != TreeEntry::ReducedBitPack) {
const TreeEntry *ZExt = getOperandEntry(E, /*Idx=*/0);
ScalarTy = cast<CastInst>(ZExt->getMainOp())->getSrcTy();
}
@@ -17743,6 +17914,37 @@ BoUpSLP::getEntryCost(const TreeEntry *E, ArrayRef<Value *> VectorizedVals,
};
return GetCostDiff(GetScalarCost, GetVectorCost);
}
+ case TreeEntry::ReducedBitPack: {
+ auto GetScalarCost = [&, &TTI = *TTI](unsigned Idx) {
+ Instruction *I;
+ if (!match(UniqueValues[Idx], m_Instruction(I)))
+ return InstructionCost(TTI::TCC_Free);
+ InstructionCost ScalarCost = TTI.getInstructionCost(I, CostKind);
+ if (match(I, m_And(m_Shl(m_Value(), m_Value()), m_Value())))
+ ScalarCost += TTI.getInstructionCost(
+ cast<Instruction>(I->getOperand(0)), CostKind);
+ return ScalarCost;
+ };
+ auto GetVectorCost = [&, &TTI = *TTI](InstructionCost CommonCost) {
+ const TreeEntry *ShlTE = E->getOpcode() == Instruction::And
+ ? getOperandEntry(E, /*Idx=*/0)
+ : E;
+ const TreeEntry *LhsTE = getOperandEntry(ShlTE, /*Idx=*/0);
+ std::optional<BitPackInfo> Info = computeBitPackLayout(*E);
+ if (!Info)
+ return InstructionCost::getInvalid();
+ unsigned ShiftWidth;
+ bool FreeByteTrunc = isByteZExtNode(*LhsTE);
+ return getBitPackCost(
+ TTI,
+ cast<FixedVectorType>(getWidenedType(
+ LhsTE->getMainOp()->getType(), LhsTE->getVectorFactor())),
+ ScalarTy, *Info, FreeByteTrunc, CostKind, TLI, E->getMainOp(),
+ ShiftWidth) +
+ CommonCost;
+ };
+ return GetCostDiff(GetScalarCost, GetVectorCost);
+ }
case TreeEntry::ReducedCmpBitcast: {
auto GetScalarCost = [&, &TTI = *TTI](unsigned Idx) {
if (isa<PoisonValue>(UniqueValues[Idx]))
@@ -18786,6 +18988,7 @@ InstructionCost BoUpSLP::getSpillCost() {
TEPtr->CombinedOp == TreeEntry::ReducedBitcastBSwap ||
TEPtr->CombinedOp == TreeEntry::ReducedBitcastLoads ||
TEPtr->CombinedOp == TreeEntry::ReducedBitcastBSwapLoads ||
+ TEPtr->CombinedOp == TreeEntry::ReducedBitPack ||
TEPtr->CombinedOp == TreeEntry::ReducedCmpBitcast) {
ScalarOrPseudoEntries.insert(TEPtr.get());
continue;
@@ -23478,6 +23681,7 @@ Value *BoUpSLP::vectorizeTree(TreeEntry *E) {
case TreeEntry::ReducedBitcastBSwap:
case TreeEntry::ReducedBitcastLoads:
case TreeEntry::ReducedBitcastBSwapLoads:
+ case TreeEntry::ReducedBitPack:
case TreeEntry::ReducedCmpBitcast:
ShuffleOrOp = E->CombinedOp;
break;
@@ -24993,6 +25197,34 @@ Value *BoUpSLP::vectorizeTree(TreeEntry *E) {
E->VectorizedValue = V;
return V;
}
+ case TreeEntry::ReducedBitPack: {
+ assert(UserIgnoreList && "Expected reduction operations only.");
+ setInsertPointAfterBundle(E);
+ TreeEntry *ShlTE = E->getOpcode() == Instruction::And
+ ? getOperandEntry(E, /*Idx=*/0)
+ : E;
+ SmallVector<TreeEntry *> CombinedTEs;
+ if (ShlTE != E)
+ CombinedTEs.append({ShlTE, getOperandEntry(E, /*Idx=*/1)});
+ CombinedTEs.push_back(getOperandEntry(ShlTE, /*Idx=*/1));
+ for (TreeEntry *TE : CombinedTEs)
+ TE->VectorizedValue = PoisonValue::get(getWidenedType(
+ TE->Scalars.front()->getType(), TE->getVectorFactor()));
+ Value *X = vectorizeOperand(ShlTE, /*NodeIdx=*/0);
+ std::optional<BitPackInfo> Info = computeBitPackLayout(*E);
+ assert(Info && "Expected bit pack info.");
+ unsigned ShiftWidth;
+ const TreeEntry *LhsTE = getOperandEntry(ShlTE, /*Idx=*/0);
+ bool FreeByteTrunc = isByteZExtNode(*LhsTE);
+ InstructionCost PackCost = getBitPackCost(
+ *TTI, cast<FixedVectorType>(X->getType()), ScalarTy, *Info,
+ FreeByteTrunc, CostKind, TLI, E->getMainOp(), ShiftWidth);
+ assert(PackCost.isValid() && "Expected valid cost.");
+ Value *V = buildBitPack(Builder, X, *Info, ShiftWidth);
+ ++NumVectorInstructions;
+ E->VectorizedValue = V;
+ return V;
+ }
case TreeEntry::ReducedCmpBitcast: {
assert(UserIgnoreList && "Expected reduction operations only.");
setInsertPointAfterBundle(E);
@@ -25511,6 +25743,7 @@ Value *BoUpSLP::vectorizeTree(
(TE->State == TreeEntry::CombinedVectorize &&
(TE->CombinedOp == TreeEntry::ReducedBitcast ||
TE->CombinedOp == TreeEntry::ReducedBitcastBSwap ||
+ TE->CombinedOp == TreeEntry::ReducedBitPack ||
((TE->CombinedOp == TreeEntry::ReducedBitcastLoads ||
TE->CombinedOp == TreeEntry::ReducedBitcastBSwapLoads ||
TE->CombinedOp == TreeEntry::ReducedCmpBitcast) &&
@@ -26191,6 +26424,7 @@ Value *BoUpSLP::vectorizeTree(
Entry->CombinedOp == TreeEntry::ReducedBitcastBSwap ||
Entry->CombinedOp == TreeEntry::ReducedBitcastLoads ||
Entry->CombinedOp == TreeEntry::ReducedBitcastBSwapLoads ||
+ Entry->CombinedOp == TreeEntry::ReducedBitPack ||
Entry->CombinedOp == TreeEntry::ReducedCmpBitcast) {
// Skip constant node
if (!Entry->hasState()) {
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.cpp
index de4c6ef1c7d1f..de10ca07c1c13 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.cpp
@@ -141,4 +141,69 @@ InstructionCost getBlendedLoadCost(const TargetTransformInfo &TTI, Type *VecTy,
CmpInst::BAD_ICMP_PREDICATE, CostKind);
}
+InstructionCost getBitPackCost(const TargetTransformInfo &TTI,
+ FixedVectorType *SrcTy, Type *ResultTy,
+ const BitPackInfo &Info, bool FreeByteTrunc,
+ TTI::TargetCostKind CostKind,
+ const TargetLibraryInfo *TLI,
+ const Instruction *CxtI, unsigned &ShiftWidth) {
+ unsigned BitWidth = SrcTy->getScalarSizeInBits();
+ unsigned NumElts = SrcTy->getNumElements();
+ uint64_t MaxAmt = *max_element(Info.LShrAmts);
+ // The shift amounts form a constant vector.
+ TTI::OperandValueInfo ShiftAmtInfo = {
+ all_of(Info.LShrAmts,
+ [&](uint64_t A) { return A == Info.LShrAmts.front(); })
+ ? TTI::OK_UniformConstantValue
+ : TTI::OK_NonUniformConstantValue,
+ all_of(Info.LShrAmts, isPowerOf2_64) ? TTI::OP_PowerOf2 : TTI::OP_None};
+ // After the shift the field content of each lane sits in the low bits of
+ // the lane, so the packing is a single byte shuffle of the shifted lanes.
+ // Pick the cheapest shift width: the narrowest type still holding the field
+ // content is not always the cheapest (e.g. missing narrow variable shifts).
+ Type *Int8Ty = IntegerType::get(SrcTy->getContext(), 8);
+ unsigned OutBytes = BitWidth / 8;
+ auto *PackTy = FixedVectorType::get(Int8Ty, OutBytes);
+ unsigned MinShiftWidth = 8;
+ while (MinShiftWidth < MaxAmt + Info.FieldWidth)
+ MinShiftWidth *= 2;
+ InstructionCost NewCost = InstructionCost::getInvalid();
+ ShiftWidth = 0;
+ for (unsigned W2 = MinShiftWidth; W2 <= BitWidth; W2 *= 2) {
+ auto *ShiftTy = FixedVectorType::get(
+ IntegerType::get(SrcTy->getContext(), W2), NumElts);
+ unsigned BytesPerLane = W2 / 8;
+ unsigned InBytes = NumElts * BytesPerLane;
+ SmallVector<int> Mask =
+ getBitPackMask(Info, OutBytes, NumElts, BytesPerLane);
+ InstructionCost C =
+ TTI.getCastInstrCost(Instruction::BitCast, ResultTy, PackTy,
+ TTI::CastContextHint::None, CostKind);
+ // A plain byte reversal of the shifted lanes is a bswap, no shuffle.
+ if (ShuffleVectorInst::isReverseMask(Mask, InBytes)) {
+ IntrinsicCostAttributes CostAttrs(Intrinsic::bswap, ResultTy, {ResultTy});
+ C += TTI.getIntrinsicInstrCost(CostAttrs, CostKind);
+ } else if (!ShuffleVectorInst::isIdentityMask(Mask, InBytes)) {
+ C += TTI.getShuffleCost(
+ is_contained(Info.LaneOfField, BitPackInfo::NoLane)
+ ? TargetTransformInfo::SK_PermuteTwoSrc
+ : TargetTransformInfo::SK_PermuteSingleSrc,
+ PackTy, FixedVectorType::get(Int8Ty, InBytes), CostKind, Mask,
+ /*Index=*/0, /*SubTp=*/nullptr, /*Args=*/{}, CxtI);
+ }
+ if (W2 != BitWidth && !(W2 == 8 && FreeByteTrunc))
+ C += TTI.getCastInstrCost(Instruction::Trunc, ShiftTy, SrcTy,
+ TTI::CastContextHint::None, CostKind);
+ if (Info.needsShift())
+ C += TTI.getArithmeticInstrCost(Instruction::LShr, ShiftTy, CostKind,
+ /*Opd1Info=*/{}, ShiftAmtInfo,
+ /*Args=*/{}, CxtI, TLI);
+ if (C.isValid() && (!NewCost.isValid() || C < NewCost)) {
+ NewCost = C;
+ ShiftWidth = W2;
+ }
+ }
+ return NewCost;
+}
+
} // namespace llvm::slpvectorizer
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.h b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.h
index 09f5ad640859a..218ccaaedbe5a 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.h
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.h
@@ -16,6 +16,7 @@
#ifndef LLVM_LIB_TRANSFORMS_VECTORIZE_SLPVECTORIZER_SLPCOSTANALYSIS_H
#define LLVM_LIB_TRANSFORMS_VECTORIZE_SLPVECTORIZER_SLPCOSTANALYSIS_H
+#include "SLPUtils.h"
#include "llvm/ADT/ArrayRef.h"
#include "llvm/Analysis/TargetTransformInfo.h"
#include "llvm/Support/InstructionCost.h"
@@ -23,6 +24,9 @@
#include <utility>
namespace llvm {
+class FixedVectorType;
+class Instruction;
+class TargetLibraryInfo;
class Type;
class Value;
class VectorType;
@@ -56,6 +60,17 @@ getBlendedLoadCost(const TargetTransformInfo &TTI, Type *VecTy, Align Alignment,
unsigned AddressSpace,
const TargetTransformInfo::TargetCostKind CostKind);
+/// Returns the cost of the bitfield packing of \p SrcTy into \p ResultTy,
+/// picking the cheapest shift width. The packing is a trunc, an lshr, a byte
+/// shuffle and a bitcast. \p FreeByteTrunc marks the lanes as a zext from i8,
+/// so compacting them to bytes is free.
+InstructionCost getBitPackCost(const TargetTransformInfo &TTI,
+ FixedVectorType *SrcTy, Type *ResultTy,
+ const BitPackInfo &Info, bool FreeByteTrunc,
+ TargetTransformInfo::TargetCostKind CostKind,
+ const TargetLibraryInfo *TLI,
+ const Instruction *CxtI, unsigned &ShiftWidth);
+
} // namespace llvm::slpvectorizer
#endif // LLVM_LIB_TRANSFORMS_VECTORIZE_SLPVECTORIZER_SLPCOSTANALYSIS_H
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp
index af6f8032c7b0b..5c0b4c2cf7ca5 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp
@@ -16,6 +16,7 @@
#include "llvm/IR/Constants.h"
#include "llvm/IR/DataLayout.h"
#include "llvm/IR/DerivedTypes.h"
+#include "llvm/IR/IRBuilder.h"
#include "llvm/IR/Instructions.h"
#include "llvm/IR/IntrinsicInst.h"
#include "llvm/IR/PatternMatch.h"
@@ -891,4 +892,153 @@ TargetTransformInfo::TargetCostKind getSLPCostKind(const Function *F) {
return F->hasOptSize() ? TTI::TCK_CodeSize : TTI::TCK_RecipThroughput;
}
+/// Deeper than the standard analysis recursion depth to keep the numeric
+/// bound precise through arithmetic carry chains.
+constexpr unsigned MaxBitPackAnalysisDepth = MaxAnalysisRecursionDepth + 2;
+
+APInt getScalarMaxValue(const Value *V, unsigned Depth) {
+ unsigned BitWidth = V->getType()->getScalarSizeInBits();
+ const APInt Unknown = APInt::getAllOnes(BitWidth);
+ if (Depth > MaxBitPackAnalysisDepth || !V->getType()->isIntegerTy())
+ return Unknown;
+ const APInt *C, *Amt;
+ if (match(V, m_APInt(C)))
+ return *C;
+ Value *L, *R;
+ if (match(V, m_Add(m_Value(L), m_Value(R))) ||
+ match(V, m_Or(m_Value(L), m_Value(R))) ||
+ match(V, m_Xor(m_Value(L), m_Value(R))))
+ return getScalarMaxValue(L, Depth + 1)
+ .uadd_sat(getScalarMaxValue(R, Depth + 1));
+ if (match(V, m_NUWSub(m_Value(L), m_Value(R))))
+ return getScalarMaxValue(L, Depth + 1);
+ if (match(V, m_Sub(m_Value(L), m_Value(R))))
+ return Unknown;
+ if (match(V, m_Mul(m_Value(L), m_Value(R))))
+ return getScalarMaxValue(L, Depth + 1)
+ .umul_sat(getScalarMaxValue(R, Depth + 1));
+ if (match(V, m_And(m_Value(L), m_Value(R))))
+ return APIntOps::umin(getScalarMaxValue(L, Depth + 1),
+ getScalarMaxValue(R, Depth + 1));
+ if (match(V, m_LShr(m_Value(L), m_APInt(Amt))) && Amt->ult(BitWidth))
+ return getScalarMaxValue(L, Depth + 1).lshr(*Amt);
+ if (match(V, m_Shl(m_Value(L), m_APInt(Amt))) && Amt->ult(BitWidth)) {
+ APInt LMax = getScalarMaxValue(L, Depth + 1);
+ return LMax.getActiveBits() + Amt->getZExtValue() <= BitWidth
+ ? LMax.shl(*Amt)
+ : Unknown;
+ }
+ if (match(V, m_ZExt(m_Value(L))))
+ return getScalarMaxValue(L, Depth + 1).zext(BitWidth);
+ if (match(V, m_Trunc(m_Value(L)))) {
+ APInt Max = getScalarMaxValue(L, Depth + 1);
+ return Max.getActiveBits() <= BitWidth ? Max.trunc(BitWidth) : Unknown;
+ }
+ if (match(V, m_SExt(m_Value(L)))) {
+ APInt Max = getScalarMaxValue(L, Depth + 1);
+ return Max.isNonNegative() ? Max.zext(BitWidth) : Unknown;
+ }
+ Value *F;
+ if (match(V, m_Select(m_Value(), m_Value(L), m_Value(F))))
+ return APIntOps::umax(getScalarMaxValue(L, Depth + 1),
+ getScalarMaxValue(F, Depth + 1));
+ return Unknown;
+}
+
+std::optional<BitPackInfo> computeBitPackInfo(unsigned BitWidth,
+ ArrayRef<APInt> PossibleBits,
+ ArrayRef<uint64_t> ShlAmts,
+ ArrayRef<APInt> Masks) {
+ unsigned NumElts = PossibleBits.size();
+ BitPackInfo Info;
+ Info.LShrAmts.assign(NumElts, 0);
+ for (unsigned Idx : seq(NumElts)) {
+ APInt Possible = PossibleBits[Idx];
+ Possible <<= ShlAmts[Idx];
+ Possible &= Masks[Idx];
+ if (Possible.isZero())
+ continue;
+ if (!Possible.isShiftedMask())
+ return std::nullopt;
+ unsigned Lo = Possible.countr_zero();
+ unsigned W = Possible.popcount();
+ if (Info.FieldWidth == 0) {
+ if (BitWidth % W != 0)
+ return std::nullopt;
+ Info.FieldWidth = W;
+ Info.LaneOfField.assign(BitWidth / W, BitPackInfo::NoLane);
+ }
+ if (W != Info.FieldWidth || Lo % W != 0 || Lo < ShlAmts[Idx])
+ return std::nullopt;
+ unsigned Field = Lo / W;
+ if (Info.LaneOfField[Field] != BitPackInfo::NoLane)
+ return std::nullopt;
+ Info.LaneOfField[Field] = Idx;
+ Info.LShrAmts[Idx] = Lo - ShlAmts[Idx];
+ }
+ if (Info.FieldWidth == 0 || Info.FieldWidth % 8 != 0)
+ return std::nullopt;
+ return Info;
+}
+
+SmallVector<int> getBitPackMask(const BitPackInfo &Info, unsigned NumBytes,
+ unsigned NumElts, unsigned BytesPerLane) {
+ unsigned BytesPerField = Info.FieldWidth / 8;
+ SmallVector<int> Mask;
+ for (unsigned J : seq(NumBytes)) {
+ unsigned Lane = Info.LaneOfField[J / BytesPerField];
+ Mask.push_back(Lane == BitPackInfo::NoLane
+ ? (int)(NumElts * BytesPerLane)
+ : (int)(Lane * BytesPerLane + J % BytesPerField));
+ }
+ return Mask;
+}
+
+Value *buildBitPack(IRBuilderBase &Builder, Value *X, const BitPackInfo &Info,
+ unsigned ShiftWidth) {
+ auto *VecTy = cast<FixedVectorType>(X->getType());
+ unsigned BitWidth = VecTy->getScalarSizeInBits();
+ unsigned NumElts = VecTy->getNumElements();
+ Value *Y = X;
+ if (ShiftWidth != BitWidth) {
+ // Compacting a byte zext is free, use its source directly.
+ if (auto *Z = dyn_cast<ZExtInst>(X);
+ Z && ShiftWidth == 8 &&
+ Z->getSrcTy() == FixedVectorType::get(Builder.getInt8Ty(), NumElts))
+ Y = Z->getOperand(0);
+ else
+ Y = Builder.CreateTrunc(
+ Y, FixedVectorType::get(IntegerType::get(X->getContext(), ShiftWidth),
+ NumElts));
+ }
+ if (Info.needsShift()) {
+ SmallVector<Constant *> Amts;
+ for (uint64_t A : Info.LShrAmts)
+ Amts.push_back(
+ ConstantInt::get(IntegerType::get(X->getContext(), ShiftWidth), A));
+ Y = Builder.CreateLShr(Y, ConstantVector::get(Amts));
+ }
+ unsigned InBytes = NumElts * (ShiftWidth / 8);
+ auto *ByteTy = FixedVectorType::get(Builder.getInt8Ty(), InBytes);
+ SmallVector<int> Mask =
+ getBitPackMask(Info, BitWidth / 8, NumElts, ShiftWidth / 8);
+ // A plain byte reversal of the shifted lanes is a bswap.
+ if (ShuffleVectorInst::isReverseMask(Mask, InBytes))
+ return Builder.CreateUnaryIntrinsic(
+ Intrinsic::bswap,
+ Builder.CreateBitCast(Y, IntegerType::get(X->getContext(), BitWidth)));
+ // An identity byte order needs no shuffle.
+ if (ShuffleVectorInst::isIdentityMask(Mask, InBytes))
+ return Builder.CreateBitCast(Y,
+ IntegerType::get(X->getContext(), BitWidth));
+ Value *Packed = Builder.CreateShuffleVector(
+ Builder.CreateBitCast(Y, ByteTy),
+ is_contained(Info.LaneOfField, BitPackInfo::NoLane)
+ ? Constant::getNullValue(ByteTy)
+ : PoisonValue::get(ByteTy),
+ Mask);
+ return Builder.CreateBitCast(Packed,
+ IntegerType::get(X->getContext(), BitWidth));
+}
+
} // namespace llvm::slpvectorizer
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.h b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.h
index e9aeff605eb91..d5a02a0e40c4b 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.h
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.h
@@ -18,18 +18,21 @@
#include "llvm/ADT/APInt.h"
#include "llvm/ADT/ArrayRef.h"
+#include "llvm/ADT/STLExtras.h"
#include "llvm/ADT/SmallBitVector.h"
#include "llvm/ADT/SmallVector.h"
#include "llvm/Analysis/MemoryLocation.h"
#include "llvm/Analysis/TargetTransformInfo.h"
#include "llvm/IR/Intrinsics.h"
+#include <cstdint>
#include <optional>
#include <string>
namespace llvm {
class Constant;
class DataLayout;
+class IRBuilderBase;
class Instruction;
class TargetLibraryInfo;
class Type;
@@ -365,6 +368,44 @@ void collectNarrowedLeaves(Value *V, unsigned RdxOpcode, unsigned WideBW,
TargetTransformInfo::TargetCostKind getSLPCostKind(const Function *F);
+/// Returns a saturating unsigned upper bound of the scalar V. The numeric
+/// bound keeps precision on arithmetic carries, where bit-wise analysis
+/// loses it.
+APInt getScalarMaxValue(const Value *V, unsigned Depth = 0);
+
+/// Description of a bitfield packing of vector lanes into a scalar value:
+/// every lane contributes a disjoint contiguous byte field of the result.
+struct BitPackInfo {
+ static constexpr unsigned NoLane = ~0u;
+ unsigned FieldWidth = 0;
+ /// Lane covering each field, NoLane if the field is always zero.
+ SmallVector<unsigned, 8> LaneOfField;
+ /// Per-lane right-shift amounts bringing the field content to the low bits.
+ SmallVector<uint64_t, 8> LShrAmts;
+
+ /// True if any lane needs a right shift to align its field content.
+ bool needsShift() const {
+ return any_of(LShrAmts, [](uint64_t A) { return A != 0; });
+ }
+};
+
+/// Computes the bitfield packing layout from the per-lane possibly set bits
+/// of the source values, the per-lane left-shift amounts and the per-lane
+/// masks (all-ones for unmasked lanes).
+std::optional<BitPackInfo> computeBitPackInfo(unsigned BitWidth,
+ ArrayRef<APInt> PossibleBits,
+ ArrayRef<uint64_t> ShlAmts,
+ ArrayRef<APInt> Masks);
+
+/// Returns the byte shuffle mask packing the per-lane fields of the shifted
+/// lanes (BytesPerLane bytes each) into the packed scalar of NumBytes bytes.
+SmallVector<int> getBitPackMask(const BitPackInfo &Info, unsigned NumBytes,
+ unsigned NumElts, unsigned BytesPerLane);
+
+/// Builds the bitfield packing of X per the layout and the shift width.
+Value *buildBitPack(IRBuilderBase &Builder, Value *X, const BitPackInfo &Info,
+ unsigned ShiftWidth);
+
} // namespace llvm::slpvectorizer
#endif // LLVM_LIB_TRANSFORMS_VECTORIZE_SLPVECTORIZER_SLPUTILS_H
diff --git a/llvm/test/Transforms/PhaseOrdering/X86/avg.ll b/llvm/test/Transforms/PhaseOrdering/X86/avg.ll
index e7d927156a9f1..8f8899d285ad3 100644
--- a/llvm/test/Transforms/PhaseOrdering/X86/avg.ll
+++ b/llvm/test/Transforms/PhaseOrdering/X86/avg.ll
@@ -95,80 +95,121 @@ define { i64, i64 } @avgr_16_u8(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerce0,
;
; SSE4-LABEL: @avgr_16_u8(
; SSE4-NEXT: entry:
-; SSE4-NEXT: [[TMP1:%.*]] = insertelement <2 x i64> poison, i64 [[A_COERCE0:%.*]], i64 0
-; SSE4-NEXT: [[TMP2:%.*]] = insertelement <2 x i64> [[TMP1]], i64 [[A_COERCE1:%.*]], i64 1
-; SSE4-NEXT: [[TMP19:%.*]] = trunc <2 x i64> [[TMP2]] to <2 x i16>
-; SSE4-NEXT: [[TMP3:%.*]] = lshr <2 x i64> [[TMP2]], splat (i64 16)
-; SSE4-NEXT: [[TMP4:%.*]] = lshr <2 x i64> [[TMP2]], splat (i64 24)
-; SSE4-NEXT: [[TMP5:%.*]] = lshr <2 x i64> [[TMP2]], splat (i64 32)
-; SSE4-NEXT: [[TMP6:%.*]] = lshr <2 x i64> [[TMP2]], splat (i64 40)
-; SSE4-NEXT: [[TMP45:%.*]] = lshr <2 x i64> [[TMP2]], splat (i64 48)
-; SSE4-NEXT: [[TMP46:%.*]] = lshr <2 x i64> [[TMP2]], splat (i64 56)
-; SSE4-NEXT: [[TMP9:%.*]] = insertelement <2 x i64> poison, i64 [[B_COERCE0:%.*]], i64 0
-; SSE4-NEXT: [[TMP10:%.*]] = insertelement <2 x i64> [[TMP9]], i64 [[B_COERCE1:%.*]], i64 1
-; SSE4-NEXT: [[TMP22:%.*]] = trunc <2 x i64> [[TMP10]] to <2 x i16>
-; SSE4-NEXT: [[TMP11:%.*]] = lshr <2 x i64> [[TMP10]], splat (i64 16)
-; SSE4-NEXT: [[TMP12:%.*]] = lshr <2 x i64> [[TMP10]], splat (i64 24)
-; SSE4-NEXT: [[TMP13:%.*]] = lshr <2 x i64> [[TMP10]], splat (i64 32)
-; SSE4-NEXT: [[TMP14:%.*]] = lshr <2 x i64> [[TMP10]], splat (i64 40)
-; SSE4-NEXT: [[TMP47:%.*]] = lshr <2 x i64> [[TMP10]], splat (i64 48)
-; SSE4-NEXT: [[TMP48:%.*]] = lshr <2 x i64> [[TMP10]], splat (i64 56)
-; SSE4-NEXT: [[TMP16:%.*]] = and <2 x i64> [[TMP2]], splat (i64 255)
-; SSE4-NEXT: [[TMP17:%.*]] = and <2 x i64> [[TMP10]], splat (i64 255)
-; SSE4-NEXT: [[TMP70:%.*]] = and <2 x i64> [[TMP45]], splat (i64 255)
-; SSE4-NEXT: [[TMP20:%.*]] = lshr <2 x i16> [[TMP19]], splat (i16 8)
-; SSE4-NEXT: [[TMP23:%.*]] = lshr <2 x i16> [[TMP22]], splat (i16 8)
-; SSE4-NEXT: [[TMP24:%.*]] = add nuw nsw <2 x i64> [[TMP16]], splat (i64 1)
-; SSE4-NEXT: [[TMP25:%.*]] = add nuw nsw <2 x i64> [[TMP24]], [[TMP17]]
-; SSE4-NEXT: [[TMP26:%.*]] = lshr <2 x i64> [[TMP25]], splat (i64 1)
-; SSE4-NEXT: [[TMP27:%.*]] = add nuw nsw <2 x i16> [[TMP20]], splat (i16 1)
-; SSE4-NEXT: [[TMP28:%.*]] = add nuw nsw <2 x i16> [[TMP27]], [[TMP23]]
-; SSE4-NEXT: [[TMP29:%.*]] = and <2 x i64> [[TMP3]], splat (i64 255)
-; SSE4-NEXT: [[TMP30:%.*]] = and <2 x i64> [[TMP11]], splat (i64 255)
-; SSE4-NEXT: [[TMP31:%.*]] = add nuw nsw <2 x i64> [[TMP29]], splat (i64 1)
-; SSE4-NEXT: [[TMP32:%.*]] = add nuw nsw <2 x i64> [[TMP31]], [[TMP30]]
-; SSE4-NEXT: [[TMP33:%.*]] = and <2 x i64> [[TMP4]], splat (i64 255)
-; SSE4-NEXT: [[TMP34:%.*]] = and <2 x i64> [[TMP12]], splat (i64 255)
-; SSE4-NEXT: [[TMP35:%.*]] = add nuw nsw <2 x i64> [[TMP33]], splat (i64 1)
-; SSE4-NEXT: [[TMP36:%.*]] = add nuw nsw <2 x i64> [[TMP35]], [[TMP34]]
-; SSE4-NEXT: [[TMP37:%.*]] = and <2 x i64> [[TMP5]], splat (i64 255)
-; SSE4-NEXT: [[TMP38:%.*]] = and <2 x i64> [[TMP13]], splat (i64 255)
-; SSE4-NEXT: [[TMP39:%.*]] = add nuw nsw <2 x i64> [[TMP37]], splat (i64 1)
-; SSE4-NEXT: [[TMP40:%.*]] = add nuw nsw <2 x i64> [[TMP39]], [[TMP38]]
-; SSE4-NEXT: [[TMP41:%.*]] = and <2 x i64> [[TMP6]], splat (i64 255)
-; SSE4-NEXT: [[TMP42:%.*]] = and <2 x i64> [[TMP14]], splat (i64 255)
-; SSE4-NEXT: [[TMP43:%.*]] = add nuw nsw <2 x i64> [[TMP41]], splat (i64 1)
-; SSE4-NEXT: [[TMP44:%.*]] = add nuw nsw <2 x i64> [[TMP43]], [[TMP42]]
-; SSE4-NEXT: [[TMP71:%.*]] = and <2 x i64> [[TMP47]], splat (i64 255)
-; SSE4-NEXT: [[TMP51:%.*]] = add nuw nsw <2 x i64> [[TMP70]], splat (i64 1)
-; SSE4-NEXT: [[TMP72:%.*]] = add nuw nsw <2 x i64> [[TMP51]], [[TMP71]]
-; SSE4-NEXT: [[TMP73:%.*]] = add nuw nsw <2 x i64> [[TMP46]], splat (i64 1)
-; SSE4-NEXT: [[TMP74:%.*]] = add nuw nsw <2 x i64> [[TMP73]], [[TMP48]]
-; SSE4-NEXT: [[TMP75:%.*]] = shl nuw <2 x i64> [[TMP74]], splat (i64 55)
-; SSE4-NEXT: [[TMP76:%.*]] = and <2 x i64> [[TMP75]], splat (i64 -72057594037927936)
-; SSE4-NEXT: [[TMP77:%.*]] = shl nuw nsw <2 x i64> [[TMP72]], splat (i64 47)
-; SSE4-NEXT: [[TMP78:%.*]] = and <2 x i64> [[TMP77]], splat (i64 71776119061217280)
-; SSE4-NEXT: [[TMP52:%.*]] = or disjoint <2 x i64> [[TMP76]], [[TMP78]]
-; SSE4-NEXT: [[TMP49:%.*]] = shl nuw nsw <2 x i64> [[TMP44]], splat (i64 39)
-; SSE4-NEXT: [[TMP50:%.*]] = and <2 x i64> [[TMP49]], splat (i64 280375465082880)
-; SSE4-NEXT: [[TMP53:%.*]] = or disjoint <2 x i64> [[TMP52]], [[TMP50]]
-; SSE4-NEXT: [[TMP54:%.*]] = shl nuw nsw <2 x i64> [[TMP40]], splat (i64 31)
-; SSE4-NEXT: [[TMP55:%.*]] = and <2 x i64> [[TMP54]], splat (i64 1095216660480)
-; SSE4-NEXT: [[TMP56:%.*]] = or disjoint <2 x i64> [[TMP53]], [[TMP55]]
-; SSE4-NEXT: [[TMP57:%.*]] = shl nuw nsw <2 x i64> [[TMP36]], splat (i64 23)
-; SSE4-NEXT: [[TMP58:%.*]] = and <2 x i64> [[TMP57]], splat (i64 4278190080)
-; SSE4-NEXT: [[TMP59:%.*]] = or disjoint <2 x i64> [[TMP56]], [[TMP58]]
-; SSE4-NEXT: [[TMP60:%.*]] = shl nuw nsw <2 x i64> [[TMP32]], splat (i64 15)
-; SSE4-NEXT: [[TMP61:%.*]] = and <2 x i64> [[TMP60]], splat (i64 16711680)
-; SSE4-NEXT: [[TMP62:%.*]] = shl nuw <2 x i16> [[TMP28]], splat (i16 7)
-; SSE4-NEXT: [[TMP63:%.*]] = or disjoint <2 x i64> [[TMP59]], [[TMP61]]
-; SSE4-NEXT: [[TMP64:%.*]] = and <2 x i16> [[TMP62]], splat (i16 -256)
-; SSE4-NEXT: [[TMP65:%.*]] = zext <2 x i16> [[TMP64]] to <2 x i64>
-; SSE4-NEXT: [[TMP66:%.*]] = or <2 x i64> [[TMP63]], [[TMP65]]
-; SSE4-NEXT: [[TMP67:%.*]] = or <2 x i64> [[TMP66]], [[TMP26]]
-; SSE4-NEXT: [[TMP68:%.*]] = extractelement <2 x i64> [[TMP67]], i64 0
+; SSE4-NEXT: [[TMP0:%.*]] = trunc i64 [[A_COERCE0:%.*]] to i16
+; SSE4-NEXT: [[TMP1:%.*]] = lshr i16 [[TMP0]], 8
+; SSE4-NEXT: [[A_SROA_3_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE0]], 16
+; SSE4-NEXT: [[A_SROA_4_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE0]], 24
+; SSE4-NEXT: [[A_SROA_5_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE0]], 32
+; SSE4-NEXT: [[A_SROA_6_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE0]], 40
+; SSE4-NEXT: [[A_SROA_7_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE0]], 48
+; SSE4-NEXT: [[A_SROA_8_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE0]], 56
+; SSE4-NEXT: [[TMP2:%.*]] = trunc i64 [[A_COERCE1:%.*]] to i16
+; SSE4-NEXT: [[TMP3:%.*]] = lshr i16 [[TMP2]], 8
+; SSE4-NEXT: [[A_SROA_12_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE1]], 16
+; SSE4-NEXT: [[A_SROA_13_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE1]], 24
+; SSE4-NEXT: [[A_SROA_14_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE1]], 32
+; SSE4-NEXT: [[A_SROA_15_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE1]], 40
+; SSE4-NEXT: [[A_SROA_16_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE1]], 48
+; SSE4-NEXT: [[A_SROA_17_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[A_COERCE1]], 56
+; SSE4-NEXT: [[TMP4:%.*]] = trunc i64 [[B_COERCE0:%.*]] to i16
+; SSE4-NEXT: [[TMP5:%.*]] = lshr i16 [[TMP4]], 8
+; SSE4-NEXT: [[B_SROA_3_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE0]], 16
+; SSE4-NEXT: [[B_SROA_4_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE0]], 24
+; SSE4-NEXT: [[B_SROA_5_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE0]], 32
+; SSE4-NEXT: [[B_SROA_6_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE0]], 40
+; SSE4-NEXT: [[B_SROA_7_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE0]], 48
+; SSE4-NEXT: [[B_SROA_8_0_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE0]], 56
+; SSE4-NEXT: [[TMP6:%.*]] = trunc i64 [[B_COERCE1:%.*]] to i16
+; SSE4-NEXT: [[TMP7:%.*]] = lshr i16 [[TMP6]], 8
+; SSE4-NEXT: [[B_SROA_12_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE1]], 16
+; SSE4-NEXT: [[B_SROA_13_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE1]], 24
+; SSE4-NEXT: [[B_SROA_14_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE1]], 32
+; SSE4-NEXT: [[B_SROA_15_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE1]], 40
+; SSE4-NEXT: [[B_SROA_16_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE1]], 48
+; SSE4-NEXT: [[B_SROA_17_8_EXTRACT_SHIFT:%.*]] = lshr i64 [[B_COERCE1]], 56
+; SSE4-NEXT: [[CONV1:%.*]] = and i64 [[A_COERCE0]], 255
+; SSE4-NEXT: [[CONV4:%.*]] = and i64 [[B_COERCE0]], 255
+; SSE4-NEXT: [[ADD:%.*]] = add nuw nsw i64 [[CONV1]], 1
+; SSE4-NEXT: [[ADD5:%.*]] = add nuw nsw i64 [[ADD]], [[CONV4]]
+; SSE4-NEXT: [[SHR:%.*]] = lshr i64 [[ADD5]], 1
+; SSE4-NEXT: [[ADD_1:%.*]] = add nuw nsw i16 [[TMP1]], 1
+; SSE4-NEXT: [[ADD5_1:%.*]] = add nuw nsw i16 [[ADD_1]], [[TMP5]]
+; SSE4-NEXT: [[CONV1_6:%.*]] = and i64 [[A_SROA_7_0_EXTRACT_SHIFT]], 255
+; SSE4-NEXT: [[CONV4_6:%.*]] = and i64 [[B_SROA_7_0_EXTRACT_SHIFT]], 255
+; SSE4-NEXT: [[ADD_6:%.*]] = add nuw nsw i64 [[CONV1_6]], 1
+; SSE4-NEXT: [[ADD5_6:%.*]] = add nuw nsw i64 [[ADD_6]], [[CONV4_6]]
+; SSE4-NEXT: [[ADD_7:%.*]] = add nuw nsw i64 [[A_SROA_8_0_EXTRACT_SHIFT]], 1
+; SSE4-NEXT: [[ADD5_7:%.*]] = add nuw nsw i64 [[ADD_7]], [[B_SROA_8_0_EXTRACT_SHIFT]]
+; SSE4-NEXT: [[CONV1_8:%.*]] = and i64 [[A_COERCE1]], 255
+; SSE4-NEXT: [[CONV4_8:%.*]] = and i64 [[B_COERCE1]], 255
+; SSE4-NEXT: [[ADD_8:%.*]] = add nuw nsw i64 [[CONV1_8]], 1
+; SSE4-NEXT: [[ADD5_8:%.*]] = add nuw nsw i64 [[ADD_8]], [[CONV4_8]]
+; SSE4-NEXT: [[SHR_8:%.*]] = lshr i64 [[ADD5_8]], 1
+; SSE4-NEXT: [[ADD_9:%.*]] = add nuw nsw i16 [[TMP3]], 1
+; SSE4-NEXT: [[ADD5_9:%.*]] = add nuw nsw i16 [[ADD_9]], [[TMP7]]
+; SSE4-NEXT: [[CONV1_14:%.*]] = and i64 [[A_SROA_16_8_EXTRACT_SHIFT]], 255
+; SSE4-NEXT: [[CONV4_14:%.*]] = and i64 [[B_SROA_16_8_EXTRACT_SHIFT]], 255
+; SSE4-NEXT: [[ADD_14:%.*]] = add nuw nsw i64 [[CONV1_14]], 1
+; SSE4-NEXT: [[ADD5_14:%.*]] = add nuw nsw i64 [[ADD_14]], [[CONV4_14]]
+; SSE4-NEXT: [[ADD_15:%.*]] = add nuw nsw i64 [[A_SROA_17_8_EXTRACT_SHIFT]], 1
+; SSE4-NEXT: [[ADD5_15:%.*]] = add nuw nsw i64 [[ADD_15]], [[B_SROA_17_8_EXTRACT_SHIFT]]
+; SSE4-NEXT: [[TMP8:%.*]] = shl nuw i64 [[ADD5_7]], 55
+; SSE4-NEXT: [[RETVAL_SROA_8_0_INSERT_EXT:%.*]] = and i64 [[TMP8]], -72057594037927936
+; SSE4-NEXT: [[TMP9:%.*]] = shl nuw nsw i64 [[ADD5_6]], 47
+; SSE4-NEXT: [[RETVAL_SROA_7_0_INSERT_SHIFT:%.*]] = and i64 [[TMP9]], 71776119061217280
+; SSE4-NEXT: [[TMP10:%.*]] = insertelement <4 x i64> poison, i64 [[A_SROA_6_0_EXTRACT_SHIFT]], i64 0
+; SSE4-NEXT: [[TMP11:%.*]] = insertelement <4 x i64> [[TMP10]], i64 [[A_SROA_5_0_EXTRACT_SHIFT]], i64 1
+; SSE4-NEXT: [[TMP12:%.*]] = insertelement <4 x i64> [[TMP11]], i64 [[A_SROA_4_0_EXTRACT_SHIFT]], i64 2
+; SSE4-NEXT: [[TMP13:%.*]] = insertelement <4 x i64> [[TMP12]], i64 [[A_SROA_3_0_EXTRACT_SHIFT]], i64 3
+; SSE4-NEXT: [[TMP14:%.*]] = and <4 x i64> [[TMP13]], splat (i64 255)
+; SSE4-NEXT: [[TMP15:%.*]] = insertelement <4 x i64> poison, i64 [[B_SROA_6_0_EXTRACT_SHIFT]], i64 0
+; SSE4-NEXT: [[TMP16:%.*]] = insertelement <4 x i64> [[TMP15]], i64 [[B_SROA_5_0_EXTRACT_SHIFT]], i64 1
+; SSE4-NEXT: [[TMP17:%.*]] = insertelement <4 x i64> [[TMP16]], i64 [[B_SROA_4_0_EXTRACT_SHIFT]], i64 2
+; SSE4-NEXT: [[TMP18:%.*]] = insertelement <4 x i64> [[TMP17]], i64 [[B_SROA_3_0_EXTRACT_SHIFT]], i64 3
+; SSE4-NEXT: [[TMP19:%.*]] = and <4 x i64> [[TMP18]], splat (i64 255)
+; SSE4-NEXT: [[TMP20:%.*]] = add nuw nsw <4 x i64> [[TMP14]], splat (i64 1)
+; SSE4-NEXT: [[TMP21:%.*]] = add nuw nsw <4 x i64> [[TMP20]], [[TMP19]]
+; SSE4-NEXT: [[TMP22:%.*]] = trunc nuw nsw <4 x i64> [[TMP21]] to <4 x i32>
+; SSE4-NEXT: [[TMP23:%.*]] = lshr <4 x i32> [[TMP22]], splat (i32 1)
+; SSE4-NEXT: [[TMP24:%.*]] = bitcast <4 x i32> [[TMP23]] to <16 x i8>
+; SSE4-NEXT: [[TMP25:%.*]] = shufflevector <16 x i8> [[TMP24]], <16 x i8> <i8 0, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison>, <8 x i32> <i32 16, i32 16, i32 12, i32 8, i32 4, i32 0, i32 16, i32 16>
+; SSE4-NEXT: [[TMP26:%.*]] = bitcast <8 x i8> [[TMP25]] to i64
+; SSE4-NEXT: [[TMP27:%.*]] = shl nuw i16 [[ADD5_1]], 7
+; SSE4-NEXT: [[TMP28:%.*]] = and i16 [[TMP27]], -256
+; SSE4-NEXT: [[RETVAL_SROA_2_0_INSERT_SHIFT_MASKED:%.*]] = zext i16 [[TMP28]] to i64
+; SSE4-NEXT: [[OP_RDX5:%.*]] = or disjoint i64 [[RETVAL_SROA_8_0_INSERT_EXT]], [[TMP26]]
+; SSE4-NEXT: [[OP_RDX6:%.*]] = or disjoint i64 [[RETVAL_SROA_7_0_INSERT_SHIFT]], [[SHR]]
+; SSE4-NEXT: [[OP_RDX7:%.*]] = or disjoint i64 [[OP_RDX5]], [[OP_RDX6]]
+; SSE4-NEXT: [[TMP68:%.*]] = or i64 [[OP_RDX7]], [[RETVAL_SROA_2_0_INSERT_SHIFT_MASKED]]
; SSE4-NEXT: [[DOTFCA_0_INSERT:%.*]] = insertvalue { i64, i64 } poison, i64 [[TMP68]], 0
-; SSE4-NEXT: [[TMP69:%.*]] = extractelement <2 x i64> [[TMP67]], i64 1
+; SSE4-NEXT: [[TMP29:%.*]] = shl nuw i64 [[ADD5_15]], 55
+; SSE4-NEXT: [[RETVAL_SROA_17_8_INSERT_EXT:%.*]] = and i64 [[TMP29]], -72057594037927936
+; SSE4-NEXT: [[TMP30:%.*]] = shl nuw nsw i64 [[ADD5_14]], 47
+; SSE4-NEXT: [[RETVAL_SROA_16_8_INSERT_SHIFT:%.*]] = and i64 [[TMP30]], 71776119061217280
+; SSE4-NEXT: [[TMP31:%.*]] = insertelement <4 x i64> poison, i64 [[A_SROA_15_8_EXTRACT_SHIFT]], i64 0
+; SSE4-NEXT: [[TMP32:%.*]] = insertelement <4 x i64> [[TMP31]], i64 [[A_SROA_14_8_EXTRACT_SHIFT]], i64 1
+; SSE4-NEXT: [[TMP33:%.*]] = insertelement <4 x i64> [[TMP32]], i64 [[A_SROA_13_8_EXTRACT_SHIFT]], i64 2
+; SSE4-NEXT: [[TMP34:%.*]] = insertelement <4 x i64> [[TMP33]], i64 [[A_SROA_12_8_EXTRACT_SHIFT]], i64 3
+; SSE4-NEXT: [[TMP35:%.*]] = and <4 x i64> [[TMP34]], splat (i64 255)
+; SSE4-NEXT: [[TMP36:%.*]] = insertelement <4 x i64> poison, i64 [[B_SROA_15_8_EXTRACT_SHIFT]], i64 0
+; SSE4-NEXT: [[TMP37:%.*]] = insertelement <4 x i64> [[TMP36]], i64 [[B_SROA_14_8_EXTRACT_SHIFT]], i64 1
+; SSE4-NEXT: [[TMP38:%.*]] = insertelement <4 x i64> [[TMP37]], i64 [[B_SROA_13_8_EXTRACT_SHIFT]], i64 2
+; SSE4-NEXT: [[TMP39:%.*]] = insertelement <4 x i64> [[TMP38]], i64 [[B_SROA_12_8_EXTRACT_SHIFT]], i64 3
+; SSE4-NEXT: [[TMP40:%.*]] = and <4 x i64> [[TMP39]], splat (i64 255)
+; SSE4-NEXT: [[TMP41:%.*]] = add nuw nsw <4 x i64> [[TMP35]], splat (i64 1)
+; SSE4-NEXT: [[TMP42:%.*]] = add nuw nsw <4 x i64> [[TMP41]], [[TMP40]]
+; SSE4-NEXT: [[TMP43:%.*]] = trunc nuw nsw <4 x i64> [[TMP42]] to <4 x i32>
+; SSE4-NEXT: [[TMP44:%.*]] = lshr <4 x i32> [[TMP43]], splat (i32 1)
+; SSE4-NEXT: [[TMP45:%.*]] = bitcast <4 x i32> [[TMP44]] to <16 x i8>
+; SSE4-NEXT: [[TMP46:%.*]] = shufflevector <16 x i8> [[TMP45]], <16 x i8> <i8 0, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison>, <8 x i32> <i32 16, i32 16, i32 12, i32 8, i32 4, i32 0, i32 16, i32 16>
+; SSE4-NEXT: [[TMP47:%.*]] = bitcast <8 x i8> [[TMP46]] to i64
+; SSE4-NEXT: [[TMP48:%.*]] = shl nuw i16 [[ADD5_9]], 7
+; SSE4-NEXT: [[TMP49:%.*]] = and i16 [[TMP48]], -256
+; SSE4-NEXT: [[RETVAL_SROA_11_8_INSERT_SHIFT_MASKED:%.*]] = zext i16 [[TMP49]] to i64
+; SSE4-NEXT: [[OP_RDX:%.*]] = or disjoint i64 [[RETVAL_SROA_17_8_INSERT_EXT]], [[TMP47]]
+; SSE4-NEXT: [[OP_RDX2:%.*]] = or disjoint i64 [[RETVAL_SROA_16_8_INSERT_SHIFT]], [[SHR_8]]
+; SSE4-NEXT: [[OP_RDX3:%.*]] = or disjoint i64 [[OP_RDX]], [[OP_RDX2]]
+; SSE4-NEXT: [[TMP69:%.*]] = or i64 [[OP_RDX3]], [[RETVAL_SROA_11_8_INSERT_SHIFT_MASKED]]
; SSE4-NEXT: [[DOTFCA_1_INSERT:%.*]] = insertvalue { i64, i64 } [[DOTFCA_0_INSERT]], i64 [[TMP69]], 1
; SSE4-NEXT: ret { i64, i64 } [[DOTFCA_1_INSERT]]
;
@@ -212,9 +253,11 @@ define { i64, i64 } @avgr_16_u8(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerce0,
; AVX2-NEXT: [[TMP27:%.*]] = and <8 x i64> [[TMP26]], <i64 -1, i64 255, i64 255, i64 255, i64 255, i64 255, i64 -1, i64 -1>
; AVX2-NEXT: [[TMP28:%.*]] = add nuw nsw <8 x i64> [[TMP27]], <i64 1, i64 1, i64 1, i64 1, i64 1, i64 1, i64 0, i64 0>
; AVX2-NEXT: [[TMP29:%.*]] = add nuw nsw <8 x i64> [[TMP28]], [[TMP13]]
-; AVX2-NEXT: [[TMP30:%.*]] = shl nuw <8 x i64> [[TMP29]], <i64 55, i64 47, i64 39, i64 31, i64 23, i64 15, i64 0, i64 0>
-; AVX2-NEXT: [[TMP31:%.*]] = and <8 x i64> [[TMP30]], <i64 -72057594037927936, i64 71776119061217280, i64 280375465082880, i64 1095216660480, i64 4278190080, i64 16711680, i64 -1, i64 -1>
-; AVX2-NEXT: [[TMP32:%.*]] = tail call i64 @llvm.vector.reduce.or.v8i64(<8 x i64> [[TMP31]])
+; AVX2-NEXT: [[TMP30:%.*]] = trunc <8 x i64> [[TMP29]] to <8 x i32>
+; AVX2-NEXT: [[TMP31:%.*]] = lshr <8 x i32> [[TMP30]], <i32 1, i32 1, i32 1, i32 1, i32 1, i32 1, i32 0, i32 8>
+; AVX2-NEXT: [[TMP41:%.*]] = bitcast <8 x i32> [[TMP31]] to <32 x i8>
+; AVX2-NEXT: [[TMP42:%.*]] = shufflevector <32 x i8> [[TMP41]], <32 x i8> poison, <8 x i32> <i32 24, i32 28, i32 20, i32 16, i32 12, i32 8, i32 4, i32 0>
+; AVX2-NEXT: [[TMP32:%.*]] = bitcast <8 x i8> [[TMP42]] to i64
; AVX2-NEXT: [[DOTFCA_0_INSERT:%.*]] = insertvalue { i64, i64 } poison, i64 [[TMP32]], 0
; AVX2-NEXT: [[TMP33:%.*]] = insertelement <8 x i64> poison, i64 [[A_COERCE1]], i64 0
; AVX2-NEXT: [[TMP34:%.*]] = insertelement <8 x i64> [[TMP33]], i64 [[ADD5_8]], i64 1
@@ -224,9 +267,11 @@ define { i64, i64 } @avgr_16_u8(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerce0,
; AVX2-NEXT: [[TMP38:%.*]] = and <8 x i64> [[TMP16]], <i64 -1, i64 255, i64 255, i64 255, i64 255, i64 255, i64 0, i64 0>
; AVX2-NEXT: [[TMP39:%.*]] = add nuw nsw <8 x i64> [[TMP37]], <i64 1, i64 1, i64 1, i64 1, i64 1, i64 1, i64 0, i64 0>
; AVX2-NEXT: [[TMP40:%.*]] = add nuw nsw <8 x i64> [[TMP39]], [[TMP38]]
-; AVX2-NEXT: [[TMP41:%.*]] = shl nuw <8 x i64> [[TMP40]], <i64 55, i64 47, i64 39, i64 31, i64 23, i64 15, i64 0, i64 0>
-; AVX2-NEXT: [[TMP42:%.*]] = and <8 x i64> [[TMP41]], <i64 -72057594037927936, i64 71776119061217280, i64 280375465082880, i64 1095216660480, i64 4278190080, i64 16711680, i64 -1, i64 -1>
-; AVX2-NEXT: [[TMP43:%.*]] = tail call i64 @llvm.vector.reduce.or.v8i64(<8 x i64> [[TMP42]])
+; AVX2-NEXT: [[TMP47:%.*]] = trunc <8 x i64> [[TMP40]] to <8 x i32>
+; AVX2-NEXT: [[TMP44:%.*]] = lshr <8 x i32> [[TMP47]], <i32 1, i32 1, i32 1, i32 1, i32 1, i32 1, i32 0, i32 8>
+; AVX2-NEXT: [[TMP45:%.*]] = bitcast <8 x i32> [[TMP44]] to <32 x i8>
+; AVX2-NEXT: [[TMP46:%.*]] = shufflevector <32 x i8> [[TMP45]], <32 x i8> poison, <8 x i32> <i32 24, i32 28, i32 20, i32 16, i32 12, i32 8, i32 4, i32 0>
+; AVX2-NEXT: [[TMP43:%.*]] = bitcast <8 x i8> [[TMP46]] to i64
; AVX2-NEXT: [[DOTFCA_1_INSERT:%.*]] = insertvalue { i64, i64 } [[DOTFCA_0_INSERT]], i64 [[TMP43]], 1
; AVX2-NEXT: ret { i64, i64 } [[DOTFCA_1_INSERT]]
;
@@ -270,9 +315,11 @@ define { i64, i64 } @avgr_16_u8(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerce0,
; AVX512-NEXT: [[TMP27:%.*]] = and <8 x i64> [[TMP26]], <i64 -1, i64 255, i64 255, i64 255, i64 255, i64 255, i64 -1, i64 -1>
; AVX512-NEXT: [[TMP28:%.*]] = add nuw nsw <8 x i64> [[TMP27]], <i64 1, i64 1, i64 1, i64 1, i64 1, i64 1, i64 0, i64 0>
; AVX512-NEXT: [[TMP29:%.*]] = add nuw nsw <8 x i64> [[TMP28]], [[TMP13]]
-; AVX512-NEXT: [[TMP30:%.*]] = shl nuw <8 x i64> [[TMP29]], <i64 55, i64 47, i64 39, i64 31, i64 23, i64 15, i64 0, i64 0>
-; AVX512-NEXT: [[TMP31:%.*]] = and <8 x i64> [[TMP30]], <i64 -72057594037927936, i64 71776119061217280, i64 280375465082880, i64 1095216660480, i64 4278190080, i64 16711680, i64 -1, i64 -1>
-; AVX512-NEXT: [[TMP32:%.*]] = tail call i64 @llvm.vector.reduce.or.v8i64(<8 x i64> [[TMP31]])
+; AVX512-NEXT: [[TMP30:%.*]] = trunc <8 x i64> [[TMP29]] to <8 x i16>
+; AVX512-NEXT: [[TMP31:%.*]] = lshr <8 x i16> [[TMP30]], <i16 1, i16 1, i16 1, i16 1, i16 1, i16 1, i16 0, i16 8>
+; AVX512-NEXT: [[TMP41:%.*]] = bitcast <8 x i16> [[TMP31]] to <16 x i8>
+; AVX512-NEXT: [[TMP42:%.*]] = shufflevector <16 x i8> [[TMP41]], <16 x i8> poison, <8 x i32> <i32 12, i32 14, i32 10, i32 8, i32 6, i32 4, i32 2, i32 0>
+; AVX512-NEXT: [[TMP32:%.*]] = bitcast <8 x i8> [[TMP42]] to i64
; AVX512-NEXT: [[DOTFCA_0_INSERT:%.*]] = insertvalue { i64, i64 } poison, i64 [[TMP32]], 0
; AVX512-NEXT: [[TMP33:%.*]] = insertelement <8 x i64> poison, i64 [[A_COERCE1]], i64 0
; AVX512-NEXT: [[TMP34:%.*]] = insertelement <8 x i64> [[TMP33]], i64 [[ADD5_8]], i64 1
@@ -282,9 +329,11 @@ define { i64, i64 } @avgr_16_u8(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerce0,
; AVX512-NEXT: [[TMP38:%.*]] = and <8 x i64> [[TMP16]], <i64 -1, i64 255, i64 255, i64 255, i64 255, i64 255, i64 0, i64 0>
; AVX512-NEXT: [[TMP39:%.*]] = add nuw nsw <8 x i64> [[TMP37]], <i64 1, i64 1, i64 1, i64 1, i64 1, i64 1, i64 0, i64 0>
; AVX512-NEXT: [[TMP40:%.*]] = add nuw nsw <8 x i64> [[TMP39]], [[TMP38]]
-; AVX512-NEXT: [[TMP41:%.*]] = shl nuw <8 x i64> [[TMP40]], <i64 55, i64 47, i64 39, i64 31, i64 23, i64 15, i64 0, i64 0>
-; AVX512-NEXT: [[TMP42:%.*]] = and <8 x i64> [[TMP41]], <i64 -72057594037927936, i64 71776119061217280, i64 280375465082880, i64 1095216660480, i64 4278190080, i64 16711680, i64 -1, i64 -1>
-; AVX512-NEXT: [[TMP43:%.*]] = tail call i64 @llvm.vector.reduce.or.v8i64(<8 x i64> [[TMP42]])
+; AVX512-NEXT: [[TMP47:%.*]] = trunc <8 x i64> [[TMP40]] to <8 x i16>
+; AVX512-NEXT: [[TMP44:%.*]] = lshr <8 x i16> [[TMP47]], <i16 1, i16 1, i16 1, i16 1, i16 1, i16 1, i16 0, i16 8>
+; AVX512-NEXT: [[TMP45:%.*]] = bitcast <8 x i16> [[TMP44]] to <16 x i8>
+; AVX512-NEXT: [[TMP46:%.*]] = shufflevector <16 x i8> [[TMP45]], <16 x i8> poison, <8 x i32> <i32 12, i32 14, i32 10, i32 8, i32 6, i32 4, i32 2, i32 0>
+; AVX512-NEXT: [[TMP43:%.*]] = bitcast <8 x i8> [[TMP46]] to i64
; AVX512-NEXT: [[DOTFCA_1_INSERT:%.*]] = insertvalue { i64, i64 } [[DOTFCA_0_INSERT]], i64 [[TMP43]], 1
; AVX512-NEXT: ret { i64, i64 } [[DOTFCA_1_INSERT]]
;
@@ -553,9 +602,7 @@ define { i64, i64 } @avgr_16_u8_alt(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerc
; SSE4-NEXT: [[TMP11:%.*]] = or <8 x i8> [[TMP7]], [[TMP3]]
; SSE4-NEXT: [[TMP12:%.*]] = and <8 x i8> [[TMP11]], splat (i8 1)
; SSE4-NEXT: [[TMP13:%.*]] = add nuw <8 x i8> [[TMP10]], [[TMP12]]
-; SSE4-NEXT: [[TMP14:%.*]] = zext <8 x i8> [[TMP13]] to <8 x i64>
-; SSE4-NEXT: [[TMP15:%.*]] = shl nuw <8 x i64> [[TMP14]], <i64 0, i64 8, i64 16, i64 24, i64 32, i64 40, i64 48, i64 56>
-; SSE4-NEXT: [[VEC2STRUCT_SLOT_SROA_0_0_VEC_EXTRACT:%.*]] = tail call i64 @llvm.vector.reduce.or.v8i64(<8 x i64> [[TMP15]])
+; SSE4-NEXT: [[VEC2STRUCT_SLOT_SROA_0_0_VEC_EXTRACT:%.*]] = bitcast <8 x i8> [[TMP13]] to i64
; SSE4-NEXT: [[DOTFCA_0_INSERT:%.*]] = insertvalue { i64, i64 } poison, i64 [[VEC2STRUCT_SLOT_SROA_0_0_VEC_EXTRACT]], 0
; SSE4-NEXT: [[TMP17:%.*]] = insertelement <8 x i64> poison, i64 [[A_COERCE1:%.*]], i64 0
; SSE4-NEXT: [[TMP18:%.*]] = shufflevector <8 x i64> [[TMP17]], <8 x i64> poison, <8 x i32> zeroinitializer
@@ -571,9 +618,7 @@ define { i64, i64 } @avgr_16_u8_alt(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerc
; SSE4-NEXT: [[TMP28:%.*]] = or <8 x i8> [[TMP24]], [[TMP20]]
; SSE4-NEXT: [[TMP29:%.*]] = and <8 x i8> [[TMP28]], splat (i8 1)
; SSE4-NEXT: [[TMP30:%.*]] = add nuw <8 x i8> [[TMP27]], [[TMP29]]
-; SSE4-NEXT: [[TMP31:%.*]] = zext <8 x i8> [[TMP30]] to <8 x i64>
-; SSE4-NEXT: [[TMP32:%.*]] = shl nuw <8 x i64> [[TMP31]], <i64 0, i64 8, i64 16, i64 24, i64 32, i64 40, i64 48, i64 56>
-; SSE4-NEXT: [[TMP49:%.*]] = tail call i64 @llvm.vector.reduce.or.v8i64(<8 x i64> [[TMP32]])
+; SSE4-NEXT: [[TMP49:%.*]] = bitcast <8 x i8> [[TMP30]] to i64
; SSE4-NEXT: [[DOTFCA_1_INSERT:%.*]] = insertvalue { i64, i64 } [[DOTFCA_0_INSERT]], i64 [[TMP49]], 1
; SSE4-NEXT: ret { i64, i64 } [[DOTFCA_1_INSERT]]
;
@@ -593,9 +638,7 @@ define { i64, i64 } @avgr_16_u8_alt(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerc
; AVX-NEXT: [[TMP11:%.*]] = or <8 x i8> [[TMP7]], [[TMP3]]
; AVX-NEXT: [[TMP12:%.*]] = and <8 x i8> [[TMP11]], splat (i8 1)
; AVX-NEXT: [[TMP13:%.*]] = add nuw <8 x i8> [[TMP10]], [[TMP12]]
-; AVX-NEXT: [[TMP14:%.*]] = zext <8 x i8> [[TMP13]] to <8 x i64>
-; AVX-NEXT: [[TMP15:%.*]] = shl nuw <8 x i64> [[TMP14]], <i64 0, i64 8, i64 16, i64 24, i64 32, i64 40, i64 48, i64 56>
-; AVX-NEXT: [[TMP16:%.*]] = tail call i64 @llvm.vector.reduce.or.v8i64(<8 x i64> [[TMP15]])
+; AVX-NEXT: [[TMP16:%.*]] = bitcast <8 x i8> [[TMP13]] to i64
; AVX-NEXT: [[DOTFCA_0_INSERT:%.*]] = insertvalue { i64, i64 } poison, i64 [[TMP16]], 0
; AVX-NEXT: [[TMP17:%.*]] = insertelement <8 x i64> poison, i64 [[A_COERCE1:%.*]], i64 0
; AVX-NEXT: [[TMP18:%.*]] = shufflevector <8 x i64> [[TMP17]], <8 x i64> poison, <8 x i32> zeroinitializer
@@ -611,9 +654,7 @@ define { i64, i64 } @avgr_16_u8_alt(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerc
; AVX-NEXT: [[TMP28:%.*]] = or <8 x i8> [[TMP24]], [[TMP20]]
; AVX-NEXT: [[TMP29:%.*]] = and <8 x i8> [[TMP28]], splat (i8 1)
; AVX-NEXT: [[TMP30:%.*]] = add nuw <8 x i8> [[TMP27]], [[TMP29]]
-; AVX-NEXT: [[TMP31:%.*]] = zext <8 x i8> [[TMP30]] to <8 x i64>
-; AVX-NEXT: [[TMP32:%.*]] = shl nuw <8 x i64> [[TMP31]], <i64 0, i64 8, i64 16, i64 24, i64 32, i64 40, i64 48, i64 56>
-; AVX-NEXT: [[TMP33:%.*]] = tail call i64 @llvm.vector.reduce.or.v8i64(<8 x i64> [[TMP32]])
+; AVX-NEXT: [[TMP33:%.*]] = bitcast <8 x i8> [[TMP30]] to i64
; AVX-NEXT: [[DOTFCA_1_INSERT:%.*]] = insertvalue { i64, i64 } [[DOTFCA_0_INSERT]], i64 [[TMP33]], 1
; AVX-NEXT: ret { i64, i64 } [[DOTFCA_1_INSERT]]
;
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/bad-reduction.ll b/llvm/test/Transforms/SLPVectorizer/X86/bad-reduction.ll
index 45d3405417b0e..77bb4a6b3ff27 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/bad-reduction.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/bad-reduction.ll
@@ -8,9 +8,8 @@
define i64 @load_bswap(ptr %p) {
; CHECK-LABEL: @load_bswap(
; CHECK-NEXT: [[TMP1:%.*]] = load <8 x i8>, ptr [[P:%.*]], align 1
-; CHECK-NEXT: [[TMP2:%.*]] = zext <8 x i8> [[TMP1]] to <8 x i64>
-; CHECK-NEXT: [[TMP3:%.*]] = shl nuw <8 x i64> [[TMP2]], <i64 56, i64 48, i64 40, i64 32, i64 24, i64 16, i64 8, i64 0>
-; CHECK-NEXT: [[OR01234567:%.*]] = call i64 @llvm.vector.reduce.or.v8i64(<8 x i64> [[TMP3]])
+; CHECK-NEXT: [[TMP2:%.*]] = bitcast <8 x i8> [[TMP1]] to i64
+; CHECK-NEXT: [[OR01234567:%.*]] = call i64 @llvm.bswap.i64(i64 [[TMP2]])
; CHECK-NEXT: ret i64 [[OR01234567]]
;
%g1 = getelementptr inbounds %v8i8, ptr %p, i64 0, i32 1
@@ -61,9 +60,8 @@ define i64 @load_bswap(ptr %p) {
define i64 @load_bswap_nop_shift(ptr %p) {
; CHECK-LABEL: @load_bswap_nop_shift(
; CHECK-NEXT: [[TMP1:%.*]] = load <8 x i8>, ptr [[P:%.*]], align 1
-; CHECK-NEXT: [[TMP2:%.*]] = zext <8 x i8> [[TMP1]] to <8 x i64>
-; CHECK-NEXT: [[TMP3:%.*]] = shl nuw <8 x i64> [[TMP2]], <i64 56, i64 48, i64 40, i64 32, i64 24, i64 16, i64 8, i64 0>
-; CHECK-NEXT: [[OR01234567:%.*]] = call i64 @llvm.vector.reduce.or.v8i64(<8 x i64> [[TMP3]])
+; CHECK-NEXT: [[TMP2:%.*]] = bitcast <8 x i8> [[TMP1]] to i64
+; CHECK-NEXT: [[OR01234567:%.*]] = call i64 @llvm.bswap.i64(i64 [[TMP2]])
; CHECK-NEXT: ret i64 [[OR01234567]]
;
%g1 = getelementptr inbounds %v8i8, ptr %p, i64 0, i32 1
@@ -116,9 +114,7 @@ define i64 @load_bswap_nop_shift(ptr %p) {
define i64 @load64le(ptr %arg) {
; CHECK-LABEL: @load64le(
; CHECK-NEXT: [[TMP1:%.*]] = load <8 x i8>, ptr [[ARG:%.*]], align 1
-; CHECK-NEXT: [[TMP2:%.*]] = zext <8 x i8> [[TMP1]] to <8 x i64>
-; CHECK-NEXT: [[TMP3:%.*]] = shl nuw <8 x i64> [[TMP2]], <i64 0, i64 8, i64 16, i64 24, i64 32, i64 40, i64 48, i64 56>
-; CHECK-NEXT: [[O7:%.*]] = call i64 @llvm.vector.reduce.or.v8i64(<8 x i64> [[TMP3]])
+; CHECK-NEXT: [[O7:%.*]] = bitcast <8 x i8> [[TMP1]] to i64
; CHECK-NEXT: ret i64 [[O7]]
;
%g1 = getelementptr inbounds i8, ptr %arg, i64 1
@@ -169,9 +165,7 @@ define i64 @load64le(ptr %arg) {
define i64 @load64le_nop_shift(ptr %arg) {
; CHECK-LABEL: @load64le_nop_shift(
; CHECK-NEXT: [[TMP1:%.*]] = load <8 x i8>, ptr [[ARG:%.*]], align 1
-; CHECK-NEXT: [[TMP2:%.*]] = zext <8 x i8> [[TMP1]] to <8 x i64>
-; CHECK-NEXT: [[TMP3:%.*]] = shl nuw <8 x i64> [[TMP2]], <i64 0, i64 8, i64 16, i64 24, i64 32, i64 40, i64 48, i64 56>
-; CHECK-NEXT: [[O7:%.*]] = call i64 @llvm.vector.reduce.or.v8i64(<8 x i64> [[TMP3]])
+; CHECK-NEXT: [[O7:%.*]] = bitcast <8 x i8> [[TMP1]] to i64
; CHECK-NEXT: ret i64 [[O7]]
;
%g1 = getelementptr inbounds i8, ptr %arg, i64 1
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/load-merge-inseltpoison.ll b/llvm/test/Transforms/SLPVectorizer/X86/load-merge-inseltpoison.ll
index daf93190cab6d..1c03d8ee6c582 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/load-merge-inseltpoison.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/load-merge-inseltpoison.ll
@@ -11,9 +11,7 @@ define i32 @_Z9load_le32Ph(ptr nocapture readonly %data) {
; CHECK-LABEL: @_Z9load_le32Ph(
; CHECK-NEXT: entry:
; CHECK-NEXT: [[TMP0:%.*]] = load <4 x i8>, ptr [[DATA:%.*]], align 1
-; CHECK-NEXT: [[TMP1:%.*]] = zext <4 x i8> [[TMP0]] to <4 x i32>
-; CHECK-NEXT: [[TMP2:%.*]] = shl nuw <4 x i32> [[TMP1]], <i32 0, i32 8, i32 16, i32 24>
-; CHECK-NEXT: [[OR11:%.*]] = call i32 @llvm.vector.reduce.or.v4i32(<4 x i32> [[TMP2]])
+; CHECK-NEXT: [[OR11:%.*]] = bitcast <4 x i8> [[TMP0]] to i32
; CHECK-NEXT: ret i32 [[OR11]]
;
entry:
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/load-merge.ll b/llvm/test/Transforms/SLPVectorizer/X86/load-merge.ll
index b407a83333502..74984d8703eb4 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/load-merge.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/load-merge.ll
@@ -11,9 +11,7 @@ define i32 @_Z9load_le32Ph(ptr nocapture readonly %data) {
; CHECK-LABEL: @_Z9load_le32Ph(
; CHECK-NEXT: entry:
; CHECK-NEXT: [[TMP0:%.*]] = load <4 x i8>, ptr [[DATA:%.*]], align 1
-; CHECK-NEXT: [[TMP1:%.*]] = zext <4 x i8> [[TMP0]] to <4 x i32>
-; CHECK-NEXT: [[TMP2:%.*]] = shl nuw <4 x i32> [[TMP1]], <i32 0, i32 8, i32 16, i32 24>
-; CHECK-NEXT: [[OR11:%.*]] = call i32 @llvm.vector.reduce.or.v4i32(<4 x i32> [[TMP2]])
+; CHECK-NEXT: [[OR11:%.*]] = bitcast <4 x i8> [[TMP0]] to i32
; CHECK-NEXT: ret i32 [[OR11]]
;
entry:
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/pr48879-sroa.ll b/llvm/test/Transforms/SLPVectorizer/X86/pr48879-sroa.ll
index 9d2ee4bafbbed..30877c242d71b 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/pr48879-sroa.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/pr48879-sroa.ll
@@ -2,7 +2,7 @@
; RUN: opt < %s -mtriple=x86_64-unknown -passes=slp-vectorizer -mcpu=x86-64 -S | FileCheck %s --check-prefixes=SSE
; RUN: opt < %s -mtriple=x86_64-unknown -passes=slp-vectorizer -mcpu=x86-64-v2 -S | FileCheck %s --check-prefixes=AVX
; RUN: opt < %s -mtriple=x86_64-unknown -passes=slp-vectorizer -mcpu=x86-64-v3 -S | FileCheck %s --check-prefixes=AVX2
-; RUN: opt < %s -mtriple=x86_64-unknown -passes=slp-vectorizer -mcpu=x86-64-v4 -S | FileCheck %s --check-prefixes=AVX2
+; RUN: opt < %s -mtriple=x86_64-unknown -passes=slp-vectorizer -mcpu=x86-64-v4 -S | FileCheck %s --check-prefixes=AVX512
define { i64, i64 } @compute_min(ptr nocapture noundef nonnull readonly align 2 dereferenceable(16) %x, ptr nocapture noundef nonnull readonly align 2 dereferenceable(16) %y) {
; SSE-LABEL: @compute_min(
@@ -13,15 +13,15 @@ define { i64, i64 } @compute_min(ptr nocapture noundef nonnull readonly align 2
; SSE-NEXT: [[TMP1:%.*]] = load <4 x i16>, ptr [[X]], align 2
; SSE-NEXT: [[TMP2:%.*]] = call <4 x i16> @llvm.smin.v4i16(<4 x i16> [[TMP0]], <4 x i16> [[TMP1]])
; SSE-NEXT: [[TMP3:%.*]] = zext <4 x i16> [[TMP2]] to <4 x i64>
-; SSE-NEXT: [[TMP4:%.*]] = shl nuw <4 x i64> [[TMP3]], <i64 0, i64 16, i64 32, i64 48>
-; SSE-NEXT: [[RETVAL_SROA_0_0_INSERT_INSERT:%.*]] = call i64 @llvm.vector.reduce.or.v4i64(<4 x i64> [[TMP4]])
+; SSE-NEXT: [[TMP4:%.*]] = trunc <4 x i64> [[TMP3]] to <4 x i16>
+; SSE-NEXT: [[RETVAL_SROA_0_0_INSERT_INSERT:%.*]] = bitcast <4 x i16> [[TMP4]] to i64
; SSE-NEXT: [[DOTFCA_0_INSERT:%.*]] = insertvalue { i64, i64 } poison, i64 [[RETVAL_SROA_0_0_INSERT_INSERT]], 0
; SSE-NEXT: [[TMP6:%.*]] = load <4 x i16>, ptr [[ARRAYIDX_I_I10_4]], align 2
; SSE-NEXT: [[TMP7:%.*]] = load <4 x i16>, ptr [[ARRAYIDX_I_I_4]], align 2
; SSE-NEXT: [[TMP8:%.*]] = call <4 x i16> @llvm.smin.v4i16(<4 x i16> [[TMP6]], <4 x i16> [[TMP7]])
; SSE-NEXT: [[TMP9:%.*]] = zext <4 x i16> [[TMP8]] to <4 x i64>
-; SSE-NEXT: [[TMP10:%.*]] = shl nuw <4 x i64> [[TMP9]], <i64 0, i64 16, i64 32, i64 48>
-; SSE-NEXT: [[RETVAL_SROA_5_8_INSERT_INSERT:%.*]] = call i64 @llvm.vector.reduce.or.v4i64(<4 x i64> [[TMP10]])
+; SSE-NEXT: [[TMP12:%.*]] = trunc <4 x i64> [[TMP9]] to <4 x i16>
+; SSE-NEXT: [[RETVAL_SROA_5_8_INSERT_INSERT:%.*]] = bitcast <4 x i16> [[TMP12]] to i64
; SSE-NEXT: [[DOTFCA_1_INSERT:%.*]] = insertvalue { i64, i64 } [[DOTFCA_0_INSERT]], i64 [[RETVAL_SROA_5_8_INSERT_INSERT]], 1
; SSE-NEXT: ret { i64, i64 } [[DOTFCA_1_INSERT]]
;
@@ -33,15 +33,19 @@ define { i64, i64 } @compute_min(ptr nocapture noundef nonnull readonly align 2
; AVX-NEXT: [[TMP1:%.*]] = load <4 x i16>, ptr [[X]], align 2
; AVX-NEXT: [[TMP2:%.*]] = call <4 x i16> @llvm.smin.v4i16(<4 x i16> [[TMP0]], <4 x i16> [[TMP1]])
; AVX-NEXT: [[TMP3:%.*]] = zext <4 x i16> [[TMP2]] to <4 x i64>
-; AVX-NEXT: [[TMP4:%.*]] = shl nuw <4 x i64> [[TMP3]], <i64 0, i64 16, i64 32, i64 48>
-; AVX-NEXT: [[TMP24:%.*]] = call i64 @llvm.vector.reduce.or.v4i64(<4 x i64> [[TMP4]])
+; AVX-NEXT: [[TMP4:%.*]] = trunc <4 x i64> [[TMP3]] to <4 x i32>
+; AVX-NEXT: [[TMP5:%.*]] = bitcast <4 x i32> [[TMP4]] to <16 x i8>
+; AVX-NEXT: [[TMP10:%.*]] = shufflevector <16 x i8> [[TMP5]], <16 x i8> poison, <8 x i32> <i32 0, i32 1, i32 4, i32 5, i32 8, i32 9, i32 12, i32 13>
+; AVX-NEXT: [[TMP24:%.*]] = bitcast <8 x i8> [[TMP10]] to i64
; AVX-NEXT: [[DOTFCA_0_INSERT:%.*]] = insertvalue { i64, i64 } poison, i64 [[TMP24]], 0
; AVX-NEXT: [[TMP6:%.*]] = load <4 x i16>, ptr [[ARRAYIDX_I_I10_4]], align 2
; AVX-NEXT: [[TMP7:%.*]] = load <4 x i16>, ptr [[ARRAYIDX_I_I_4]], align 2
; AVX-NEXT: [[TMP8:%.*]] = call <4 x i16> @llvm.smin.v4i16(<4 x i16> [[TMP6]], <4 x i16> [[TMP7]])
; AVX-NEXT: [[TMP9:%.*]] = zext <4 x i16> [[TMP8]] to <4 x i64>
-; AVX-NEXT: [[TMP10:%.*]] = shl nuw <4 x i64> [[TMP9]], <i64 0, i64 16, i64 32, i64 48>
-; AVX-NEXT: [[TMP25:%.*]] = call i64 @llvm.vector.reduce.or.v4i64(<4 x i64> [[TMP10]])
+; AVX-NEXT: [[TMP12:%.*]] = trunc <4 x i64> [[TMP9]] to <4 x i32>
+; AVX-NEXT: [[TMP13:%.*]] = bitcast <4 x i32> [[TMP12]] to <16 x i8>
+; AVX-NEXT: [[TMP14:%.*]] = shufflevector <16 x i8> [[TMP13]], <16 x i8> poison, <8 x i32> <i32 0, i32 1, i32 4, i32 5, i32 8, i32 9, i32 12, i32 13>
+; AVX-NEXT: [[TMP25:%.*]] = bitcast <8 x i8> [[TMP14]] to i64
; AVX-NEXT: [[DOTFCA_1_INSERT:%.*]] = insertvalue { i64, i64 } [[DOTFCA_0_INSERT]], i64 [[TMP25]], 1
; AVX-NEXT: ret { i64, i64 } [[DOTFCA_1_INSERT]]
;
@@ -53,18 +57,42 @@ define { i64, i64 } @compute_min(ptr nocapture noundef nonnull readonly align 2
; AVX2-NEXT: [[TMP1:%.*]] = load <4 x i16>, ptr [[X]], align 2
; AVX2-NEXT: [[TMP2:%.*]] = call <4 x i16> @llvm.smin.v4i16(<4 x i16> [[TMP0]], <4 x i16> [[TMP1]])
; AVX2-NEXT: [[TMP3:%.*]] = zext <4 x i16> [[TMP2]] to <4 x i64>
-; AVX2-NEXT: [[TMP4:%.*]] = shl nuw <4 x i64> [[TMP3]], <i64 0, i64 16, i64 32, i64 48>
-; AVX2-NEXT: [[TMP24:%.*]] = call i64 @llvm.vector.reduce.or.v4i64(<4 x i64> [[TMP4]])
-; AVX2-NEXT: [[DOTFCA_0_INSERT:%.*]] = insertvalue { i64, i64 } poison, i64 [[TMP24]], 0
-; AVX2-NEXT: [[TMP6:%.*]] = load <4 x i16>, ptr [[ARRAYIDX_I_I10_4]], align 2
-; AVX2-NEXT: [[TMP7:%.*]] = load <4 x i16>, ptr [[ARRAYIDX_I_I_4]], align 2
-; AVX2-NEXT: [[TMP8:%.*]] = call <4 x i16> @llvm.smin.v4i16(<4 x i16> [[TMP6]], <4 x i16> [[TMP7]])
-; AVX2-NEXT: [[TMP9:%.*]] = zext <4 x i16> [[TMP8]] to <4 x i64>
-; AVX2-NEXT: [[TMP10:%.*]] = shl nuw <4 x i64> [[TMP9]], <i64 0, i64 16, i64 32, i64 48>
-; AVX2-NEXT: [[TMP25:%.*]] = call i64 @llvm.vector.reduce.or.v4i64(<4 x i64> [[TMP10]])
-; AVX2-NEXT: [[DOTFCA_1_INSERT:%.*]] = insertvalue { i64, i64 } [[DOTFCA_0_INSERT]], i64 [[TMP25]], 1
+; AVX2-NEXT: [[TMP4:%.*]] = trunc <4 x i64> [[TMP3]] to <4 x i32>
+; AVX2-NEXT: [[TMP5:%.*]] = bitcast <4 x i32> [[TMP4]] to <16 x i8>
+; AVX2-NEXT: [[TMP6:%.*]] = shufflevector <16 x i8> [[TMP5]], <16 x i8> poison, <8 x i32> <i32 0, i32 1, i32 4, i32 5, i32 8, i32 9, i32 12, i32 13>
+; AVX2-NEXT: [[TMP7:%.*]] = bitcast <8 x i8> [[TMP6]] to i64
+; AVX2-NEXT: [[DOTFCA_0_INSERT:%.*]] = insertvalue { i64, i64 } poison, i64 [[TMP7]], 0
+; AVX2-NEXT: [[TMP8:%.*]] = load <4 x i16>, ptr [[ARRAYIDX_I_I10_4]], align 2
+; AVX2-NEXT: [[TMP9:%.*]] = load <4 x i16>, ptr [[ARRAYIDX_I_I_4]], align 2
+; AVX2-NEXT: [[TMP10:%.*]] = call <4 x i16> @llvm.smin.v4i16(<4 x i16> [[TMP8]], <4 x i16> [[TMP9]])
+; AVX2-NEXT: [[TMP11:%.*]] = zext <4 x i16> [[TMP10]] to <4 x i64>
+; AVX2-NEXT: [[TMP12:%.*]] = trunc <4 x i64> [[TMP11]] to <4 x i32>
+; AVX2-NEXT: [[TMP13:%.*]] = bitcast <4 x i32> [[TMP12]] to <16 x i8>
+; AVX2-NEXT: [[TMP14:%.*]] = shufflevector <16 x i8> [[TMP13]], <16 x i8> poison, <8 x i32> <i32 0, i32 1, i32 4, i32 5, i32 8, i32 9, i32 12, i32 13>
+; AVX2-NEXT: [[TMP15:%.*]] = bitcast <8 x i8> [[TMP14]] to i64
+; AVX2-NEXT: [[DOTFCA_1_INSERT:%.*]] = insertvalue { i64, i64 } [[DOTFCA_0_INSERT]], i64 [[TMP15]], 1
; AVX2-NEXT: ret { i64, i64 } [[DOTFCA_1_INSERT]]
;
+; AVX512-LABEL: @compute_min(
+; AVX512-NEXT: entry:
+; AVX512-NEXT: [[ARRAYIDX_I_I_4:%.*]] = getelementptr inbounds [8 x i16], ptr [[X:%.*]], i64 0, i64 4
+; AVX512-NEXT: [[ARRAYIDX_I_I10_4:%.*]] = getelementptr inbounds [8 x i16], ptr [[Y:%.*]], i64 0, i64 4
+; AVX512-NEXT: [[TMP0:%.*]] = load <4 x i16>, ptr [[Y]], align 2
+; AVX512-NEXT: [[TMP1:%.*]] = load <4 x i16>, ptr [[X]], align 2
+; AVX512-NEXT: [[TMP2:%.*]] = call <4 x i16> @llvm.smin.v4i16(<4 x i16> [[TMP0]], <4 x i16> [[TMP1]])
+; AVX512-NEXT: [[TMP3:%.*]] = zext <4 x i16> [[TMP2]] to <4 x i64>
+; AVX512-NEXT: [[TMP4:%.*]] = trunc <4 x i64> [[TMP3]] to <4 x i16>
+; AVX512-NEXT: [[TMP5:%.*]] = bitcast <4 x i16> [[TMP4]] to i64
+; AVX512-NEXT: [[DOTFCA_0_INSERT:%.*]] = insertvalue { i64, i64 } poison, i64 [[TMP5]], 0
+; AVX512-NEXT: [[TMP6:%.*]] = load <4 x i16>, ptr [[ARRAYIDX_I_I10_4]], align 2
+; AVX512-NEXT: [[TMP7:%.*]] = load <4 x i16>, ptr [[ARRAYIDX_I_I_4]], align 2
+; AVX512-NEXT: [[TMP8:%.*]] = call <4 x i16> @llvm.smin.v4i16(<4 x i16> [[TMP6]], <4 x i16> [[TMP7]])
+; AVX512-NEXT: [[TMP9:%.*]] = zext <4 x i16> [[TMP8]] to <4 x i64>
+; AVX512-NEXT: [[TMP10:%.*]] = trunc <4 x i64> [[TMP9]] to <4 x i16>
+; AVX512-NEXT: [[TMP11:%.*]] = bitcast <4 x i16> [[TMP10]] to i64
+; AVX512-NEXT: [[DOTFCA_1_INSERT:%.*]] = insertvalue { i64, i64 } [[DOTFCA_0_INSERT]], i64 [[TMP11]], 1
+; AVX512-NEXT: ret { i64, i64 } [[DOTFCA_1_INSERT]]
+;
entry:
%0 = load i16, ptr %y, align 2
%1 = load i16, ptr %x, align 2
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/reduce-or-bitpack.ll b/llvm/test/Transforms/SLPVectorizer/X86/reduce-or-bitpack.ll
index 6c4ec216e1b2f..a5c5177787190 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/reduce-or-bitpack.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/reduce-or-bitpack.ll
@@ -12,8 +12,8 @@ define i32 @shl_and_pack(i32 %x) {
; CHECK-NEXT: [[TMP1:%.*]] = shufflevector <4 x i32> [[TMP0]], <4 x i32> poison, <4 x i32> zeroinitializer
; CHECK-NEXT: [[TMP2:%.*]] = lshr <4 x i32> [[TMP1]], <i32 0, i32 8, i32 16, i32 24>
; CHECK-NEXT: [[TMP3:%.*]] = and <4 x i32> [[TMP2]], <i32 255, i32 255, i32 255, i32 -1>
-; CHECK-NEXT: [[TMP4:%.*]] = shl nuw <4 x i32> [[TMP3]], <i32 0, i32 8, i32 16, i32 24>
-; CHECK-NEXT: [[TMP6:%.*]] = call i32 @llvm.vector.reduce.or.v4i32(<4 x i32> [[TMP4]])
+; CHECK-NEXT: [[TMP4:%.*]] = trunc <4 x i32> [[TMP3]] to <4 x i8>
+; CHECK-NEXT: [[TMP6:%.*]] = bitcast <4 x i8> [[TMP4]] to i32
; CHECK-NEXT: ret i32 [[TMP6]]
;
entry:
@@ -109,9 +109,11 @@ define dso_local { i64, i64 } @and_shl_add_pack(i64 %0, i64 %1, i64 %2, i64 %3)
; CHECK-NEXT: [[TMP42:%.*]] = and <8 x i64> [[TMP41]], <i64 -1, i64 255, i64 255, i64 255, i64 255, i64 255, i64 -1, i64 -1>
; CHECK-NEXT: [[TMP43:%.*]] = add nuw nsw <8 x i64> [[TMP42]], <i64 1, i64 1, i64 1, i64 1, i64 1, i64 1, i64 0, i64 0>
; CHECK-NEXT: [[TMP44:%.*]] = add nuw nsw <8 x i64> [[TMP43]], [[TMP26]]
-; CHECK-NEXT: [[TMP45:%.*]] = shl nuw <8 x i64> [[TMP44]], <i64 55, i64 47, i64 39, i64 31, i64 23, i64 15, i64 0, i64 0>
-; CHECK-NEXT: [[TMP46:%.*]] = and <8 x i64> [[TMP45]], <i64 -72057594037927936, i64 71776119061217280, i64 280375465082880, i64 1095216660480, i64 4278190080, i64 16711680, i64 -1, i64 -1>
-; CHECK-NEXT: [[TMP49:%.*]] = call i64 @llvm.vector.reduce.or.v8i64(<8 x i64> [[TMP46]])
+; CHECK-NEXT: [[TMP45:%.*]] = trunc <8 x i64> [[TMP44]] to <8 x i32>
+; CHECK-NEXT: [[TMP46:%.*]] = lshr <8 x i32> [[TMP45]], <i32 1, i32 1, i32 1, i32 1, i32 1, i32 1, i32 0, i32 8>
+; CHECK-NEXT: [[TMP47:%.*]] = bitcast <8 x i32> [[TMP46]] to <32 x i8>
+; CHECK-NEXT: [[TMP48:%.*]] = shufflevector <32 x i8> [[TMP47]], <32 x i8> poison, <8 x i32> <i32 24, i32 28, i32 20, i32 16, i32 12, i32 8, i32 4, i32 0>
+; CHECK-NEXT: [[TMP49:%.*]] = bitcast <8 x i8> [[TMP48]] to i64
; CHECK-NEXT: [[TMP50:%.*]] = insertvalue { i64, i64 } poison, i64 [[TMP49]], 0
; CHECK-NEXT: [[TMP51:%.*]] = insertelement <8 x i64> poison, i64 [[TMP1]], i64 0
; CHECK-NEXT: [[TMP52:%.*]] = insertelement <8 x i64> [[TMP51]], i64 [[TMP20]], i64 1
@@ -123,9 +125,11 @@ define dso_local { i64, i64 } @and_shl_add_pack(i64 %0, i64 %1, i64 %2, i64 %3)
; CHECK-NEXT: [[TMP58:%.*]] = and <8 x i64> <i64 -1, i64 255, i64 255, i64 255, i64 255, i64 255, i64 0, i64 0>, [[TMP29]]
; CHECK-NEXT: [[TMP59:%.*]] = add nuw nsw <8 x i64> [[TMP57]], <i64 1, i64 1, i64 1, i64 1, i64 1, i64 1, i64 0, i64 0>
; CHECK-NEXT: [[TMP60:%.*]] = add nuw nsw <8 x i64> [[TMP59]], [[TMP58]]
-; CHECK-NEXT: [[TMP61:%.*]] = shl nuw <8 x i64> [[TMP60]], <i64 55, i64 47, i64 39, i64 31, i64 23, i64 15, i64 0, i64 0>
-; CHECK-NEXT: [[TMP62:%.*]] = and <8 x i64> [[TMP61]], <i64 -72057594037927936, i64 71776119061217280, i64 280375465082880, i64 1095216660480, i64 4278190080, i64 16711680, i64 -1, i64 -1>
-; CHECK-NEXT: [[TMP65:%.*]] = call i64 @llvm.vector.reduce.or.v8i64(<8 x i64> [[TMP62]])
+; CHECK-NEXT: [[TMP61:%.*]] = trunc <8 x i64> [[TMP60]] to <8 x i32>
+; CHECK-NEXT: [[TMP62:%.*]] = lshr <8 x i32> [[TMP61]], <i32 1, i32 1, i32 1, i32 1, i32 1, i32 1, i32 0, i32 8>
+; CHECK-NEXT: [[TMP63:%.*]] = bitcast <8 x i32> [[TMP62]] to <32 x i8>
+; CHECK-NEXT: [[TMP64:%.*]] = shufflevector <32 x i8> [[TMP63]], <32 x i8> poison, <8 x i32> <i32 24, i32 28, i32 20, i32 16, i32 12, i32 8, i32 4, i32 0>
+; CHECK-NEXT: [[TMP65:%.*]] = bitcast <8 x i8> [[TMP64]] to i64
; CHECK-NEXT: [[TMP66:%.*]] = insertvalue { i64, i64 } [[TMP50]], i64 [[TMP65]], 1
; CHECK-NEXT: ret { i64, i64 } [[TMP66]]
;
diff --git a/llvm/test/Transforms/SLPVectorizer/non-power-of-2-bswap.ll b/llvm/test/Transforms/SLPVectorizer/non-power-of-2-bswap.ll
index 923663c2adb22..7732167767b8a 100644
--- a/llvm/test/Transforms/SLPVectorizer/non-power-of-2-bswap.ll
+++ b/llvm/test/Transforms/SLPVectorizer/non-power-of-2-bswap.ll
@@ -7,9 +7,8 @@ define i64 @bswap_i24(ptr noalias %p, ptr noalias %p1) {
; CHECK-NEXT: [[TMP1:%.*]] = load <3 x i8>, ptr [[P]], align 1
; CHECK-NEXT: [[TMP2:%.*]] = load <3 x i8>, ptr [[P1]], align 1
; CHECK-NEXT: [[TMP3:%.*]] = add <3 x i8> [[TMP1]], [[TMP2]]
-; CHECK-NEXT: [[TMP4:%.*]] = zext <3 x i8> [[TMP3]] to <3 x i32>
-; CHECK-NEXT: [[TMP5:%.*]] = shl <3 x i32> [[TMP4]], <i32 16, i32 8, i32 0>
-; CHECK-NEXT: [[TMP8:%.*]] = call i32 @llvm.vector.reduce.or.v3i32(<3 x i32> [[TMP5]])
+; CHECK-NEXT: [[TMP6:%.*]] = shufflevector <3 x i8> [[TMP3]], <3 x i8> zeroinitializer, <4 x i32> <i32 2, i32 1, i32 0, i32 3>
+; CHECK-NEXT: [[TMP8:%.*]] = bitcast <4 x i8> [[TMP6]] to i32
; CHECK-NEXT: [[TMP9:%.*]] = zext i32 [[TMP8]] to i64
; CHECK-NEXT: ret i64 [[TMP9]]
;
More information about the llvm-commits
mailing list