[llvm] [SLP] Fix CmpInst type handling in cost model (PR #190618)
via llvm-commits
llvm-commits at lists.llvm.org
Mon Apr 6 08:24:36 PDT 2026
llvmbot wrote:
<!--LLVM PR SUMMARY COMMENT-->
@llvm/pr-subscribers-vectorizers
Author: Alexey Bataev (alexey-bataev)
<details>
<summary>Changes</summary>
Previously, getValueType() always returned the compared operand type
(e.g. i32) for CmpInst, which was incorrect for gather cost estimation
and codegen where the result type (i1) is needed. This caused ad-hoc
fixups scattered across getEntryCost, calculateTreeCostAndTrimNonProfitable,
and vectorizeTree that overrode ScalarTy back to i1 for CmpInsts.
Add a LookThroughCmp parameter to getValueType() (default: false) so
callers that need the operand type for vector width calculations can
explicitly opt in. This removes the need for the scattered CmpInst
special cases:
- getEntryCost gather path: remove `if (isa<CmpInst>) ScalarTy = i1`
- calculateTreeCostAndTrimNonProfitable: remove same override
- vectorizeTree: simplify `if (!isa<CmpInst>) ScalarTy = getValueType(V)`
to just `getValueType(V)`
For the ICmp/FCmp cost case in getEntryCost, add a fallthrough from
ICmp/FCmp to Select that overrides ScalarTy with the compared operand
type via getValueType(VL0, true), since getCmpSelInstrCost expects the
compared type as its first argument. Fix the condition type argument
passed to getCmpSelInstrCost for both scalar and vector paths: use the
actual condition/result type instead of always Builder.getInt1Ty().
---
Patch is 43.17 KiB, truncated to 20.00 KiB below, full version: https://github.com/llvm/llvm-project/pull/190618.diff
10 Files Affected:
- (modified) llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp (+45-29)
- (modified) llvm/test/Transforms/SLPVectorizer/AArch64/extracts-from-scalarizable-vector.ll (+8-19)
- (modified) llvm/test/Transforms/SLPVectorizer/AArch64/vectorizable-selects-min-max.ll (+13-6)
- (modified) llvm/test/Transforms/SLPVectorizer/RISCV/remarks-insert-into-small-vector.ll (+1-1)
- (modified) llvm/test/Transforms/SLPVectorizer/X86/bool-mask.ll (+13-96)
- (modified) llvm/test/Transforms/SLPVectorizer/X86/cmp-as-alternate-ops.ll (+7-5)
- (modified) llvm/test/Transforms/SLPVectorizer/X86/identity-match-splat-less-defined.ll (+42-1)
- (modified) llvm/test/Transforms/SLPVectorizer/X86/inversed-icmp-to-gather.ll (+5-16)
- (modified) llvm/test/Transforms/SLPVectorizer/X86/non-load-reduced-as-part-of-bv.ll (+10-10)
- (modified) llvm/test/Transforms/SLPVectorizer/X86/reduced-value-stored.ll (+17-18)
``````````diff
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index f2ccf198c4c81..1ca868a111bc0 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -258,15 +258,18 @@ static bool isValidElementType(Type *Ty) {
!Ty->isPPC_FP128Ty();
}
-/// Returns the type of the given value/instruction \p V. If it is store,
-/// returns the type of its value operand, for Cmp - the types of the compare
-/// operands and for insertelement - the type os the inserted operand.
-/// Otherwise, just the type of the value is returned.
-static Type *getValueType(Value *V) {
+/// Returns the "element type" of the given value/instruction \p V.
+/// For stores, returns the stored value type; for insertelement (when ReVec is
+/// off), the inserted operand type. For compares, the default is to return the
+/// result type (i1); when \p LookThroughCmp is true, returns the type of the
+/// compared operands instead, which is needed for vector width calculations
+/// (the width is determined by the operand type, not the i1 result).
+static Type *getValueType(Value *V, bool LookThroughCmp = false) {
if (auto *SI = dyn_cast<StoreInst>(V))
return SI->getValueOperand()->getType();
- if (auto *CI = dyn_cast<CmpInst>(V))
- return CI->getOperand(0)->getType();
+ if (LookThroughCmp)
+ if (auto *CI = dyn_cast<CmpInst>(V))
+ return CI->getOperand(0)->getType();
if (!SLPReVec)
if (auto *IE = dyn_cast<InsertElementInst>(V))
return IE->getOperand(1)->getType();
@@ -4329,7 +4332,8 @@ class slpvectorizer::BoUpSLP {
bool
hasNonWholeRegisterOrNonPowerOf2Vec(const TargetTransformInfo &TTI) const {
bool IsNonPowerOf2 = !hasFullVectorsOrPowerOf2(
- TTI, getValueType(Scalars.front()), Scalars.size());
+ TTI, getValueType(Scalars.front(), /*LookThroughCmp=*/true),
+ Scalars.size());
assert((!IsNonPowerOf2 || ReuseShuffleIndices.empty()) &&
"Reshuffling not supported with non-power-of-2 vectors yet.");
return IsNonPowerOf2;
@@ -4492,10 +4496,11 @@ class slpvectorizer::BoUpSLP {
std::make_pair(UserTreeIdx.UserTE, UserTreeIdx.EdgeIdx), Last);
// FIXME: Remove once support for ReuseShuffleIndices has been implemented
// for non-power-of-two vectors.
- assert(
- (hasFullVectorsOrPowerOf2(*TTI, getValueType(VL.front()), VL.size()) ||
- ReuseShuffleIndices.empty()) &&
- "Reshuffling scalars not yet supported for nodes with padding");
+ assert((hasFullVectorsOrPowerOf2(
+ *TTI, getValueType(VL.front(), /*LookThroughCmp=*/true),
+ VL.size()) ||
+ ReuseShuffleIndices.empty()) &&
+ "Reshuffling scalars not yet supported for nodes with padding");
Last->ReuseShuffleIndices.append(ReuseShuffleIndices.begin(),
ReuseShuffleIndices.end());
if (ReorderIndices.empty()) {
@@ -10964,7 +10969,8 @@ static bool tryToFindDuplicates(SmallVectorImpl<Value *> &VL,
// Easy case: VL has unique values and a "natural" size
size_t NumUniqueScalarValues = UniqueValues.size();
bool IsFullVectors = hasFullVectorsOrPowerOf2(
- TTI, getValueType(UniqueValues.front()), NumUniqueScalarValues);
+ TTI, getValueType(UniqueValues.front(), /*LookThroughCmp=*/true),
+ NumUniqueScalarValues);
if (NumUniqueScalarValues == VL.size() &&
(VectorizeNonPowerOf2 || IsFullVectors)) {
ReuseShuffleIndices.clear();
@@ -10974,7 +10980,8 @@ static bool tryToFindDuplicates(SmallVectorImpl<Value *> &VL,
// FIXME: Reshuffing scalars is not supported yet for non-power-of-2 ops.
if ((UserTreeIdx.UserTE &&
UserTreeIdx.UserTE->hasNonWholeRegisterOrNonPowerOf2Vec(TTI)) ||
- !hasFullVectorsOrPowerOf2(TTI, getValueType(VL.front()), VL.size())) {
+ !hasFullVectorsOrPowerOf2(
+ TTI, getValueType(VL.front(), /*LookThroughCmp=*/true), VL.size())) {
LLVM_DEBUG(dbgs() << "SLP: Reshuffling scalars not yet supported "
"for nodes with padding.\n");
ReuseShuffleIndices.clear();
@@ -15687,8 +15694,6 @@ BoUpSLP::getEntryCost(const TreeEntry *E, ArrayRef<Value *> VectorizedVals,
return 0;
if (isa<InsertElementInst>(VL[0]))
return InstructionCost::getInvalid();
- if (isa<CmpInst>(VL.front()))
- ScalarTy = VL.front()->getType();
return SpillsReloads +
processBuildVector<ShuffleCostEstimator, InstructionCost>(
E, ScalarTy, *TTI, VectorizedVals, *this, CheckedExtracts);
@@ -16152,6 +16157,12 @@ BoUpSLP::getEntryCost(const TreeEntry *E, ArrayRef<Value *> VectorizedVals,
}
case Instruction::FCmp:
case Instruction::ICmp:
+ // Override ScalarTy/VecTy with the compared operand type (not i1). The
+ // cost of a compare instruction is determined by the operand width, and
+ // getCmpSelInstrCost expects the compared type as its first type arg.
+ OrigScalarTy = ScalarTy = getValueType(VL0, /*LookThroughCmp=*/true);
+ VecTy = getWidenedType(ScalarTy, VL.size());
+ [[fallthrough]];
case Instruction::Select: {
CmpPredicate VecPred, SwappedVecPred;
auto MatchCmp = m_Cmp(VecPred, m_Value(), m_Value());
@@ -16183,9 +16194,13 @@ BoUpSLP::getEntryCost(const TreeEntry *E, ArrayRef<Value *> VectorizedVals,
? CmpInst::BAD_FCMP_PREDICATE
: CmpInst::BAD_ICMP_PREDICATE;
+ // For selects, the "condition type" arg is the condition operand's
+ // type; for standalone compares, it is the result type (i1).
InstructionCost ScalarCost = TTI->getCmpSelInstrCost(
- E->getOpcode(), OrigScalarTy, Builder.getInt1Ty(), CurrentPred,
- CostKind,
+ E->getOpcode(), OrigScalarTy,
+ ShuffleOrOp == Instruction::Select ? VL0->getOperand(0)->getType()
+ : VL0->getType(),
+ CurrentPred, CostKind,
getOperandInfo(
VI->getOperand(ShuffleOrOp == Instruction::Select ? 1 : 0)),
getOperandInfo(
@@ -16198,7 +16213,14 @@ BoUpSLP::getEntryCost(const TreeEntry *E, ArrayRef<Value *> VectorizedVals,
return ScalarCost;
};
auto GetVectorCost = [&](InstructionCost CommonCost) {
- auto *MaskTy = getWidenedType(Builder.getInt1Ty(), VL.size());
+ // For selects, the condition type may differ from the result type
+ // (e.g. condition is <N x i1> while result is <N x i32>). For
+ // compares, the result type IS the mask (i1/vNi1). Construct the
+ // right type so getCmpSelInstrCost sees the actual mask/result width.
+ auto *MaskTy = getWidenedType(ShuffleOrOp == Instruction::Select
+ ? VL0->getOperand(0)->getType()
+ : VL0->getType(),
+ VL.size());
InstructionCost VecCost = TTI->getCmpSelInstrCost(
E->getOpcode(), VecTy, MaskTy, VecPred, CostKind,
@@ -16207,10 +16229,8 @@ BoUpSLP::getEntryCost(const TreeEntry *E, ArrayRef<Value *> VectorizedVals,
getOperandInfo(
E->getOperand(ShuffleOrOp == Instruction::Select ? 2 : 1)),
VL0);
- if (auto *SI = dyn_cast<SelectInst>(VL0)) {
- auto *CondType =
- getWidenedType(SI->getCondition()->getType(), VL.size());
- unsigned CondNumElements = CondType->getNumElements();
+ if (isa<SelectInst>(VL0)) {
+ unsigned CondNumElements = getNumElements(MaskTy);
unsigned VecTyNumElements = getNumElements(VecTy);
assert(VecTyNumElements >= CondNumElements &&
VecTyNumElements % CondNumElements == 0 &&
@@ -16219,7 +16239,7 @@ BoUpSLP::getEntryCost(const TreeEntry *E, ArrayRef<Value *> VectorizedVals,
// When the return type is i1 but the source is fixed vector type, we
// need to duplicate the condition value.
VecCost += ::getShuffleCost(
- *TTI, TTI::SK_PermuteSingleSrc, CondType,
+ *TTI, TTI::SK_PermuteSingleSrc, MaskTy,
createReplicatedMask(VecTyNumElements / CondNumElements,
CondNumElements));
}
@@ -17855,8 +17875,6 @@ InstructionCost BoUpSLP::calculateTreeCostAndTrimNonProfitable(
auto It = MinBWs.find(TE);
if (It != MinBWs.end())
ScalarTy = IntegerType::get(ScalarTy->getContext(), It->second.first);
- if (isa<CmpInst>(TE->Scalars.front()))
- ScalarTy = TE->Scalars.front()->getType();
auto *VecTy = getWidenedType(ScalarTy, Sz);
const unsigned EntryVF = TE->getVectorFactor();
auto *FinalVecTy = getWidenedType(ScalarTy, EntryVF);
@@ -21215,9 +21233,7 @@ Value *BoUpSLP::vectorizeTree(TreeEntry *E) {
IRBuilderBase::InsertPointGuard Guard(Builder);
Value *V = E->Scalars.front();
- Type *ScalarTy = V->getType();
- if (!isa<CmpInst>(V))
- ScalarTy = getValueType(V);
+ Type *ScalarTy = getValueType(V);
auto It = MinBWs.find(E);
if (It != MinBWs.end()) {
auto *VecTy = dyn_cast<FixedVectorType>(ScalarTy);
diff --git a/llvm/test/Transforms/SLPVectorizer/AArch64/extracts-from-scalarizable-vector.ll b/llvm/test/Transforms/SLPVectorizer/AArch64/extracts-from-scalarizable-vector.ll
index c99dd53117e5f..54c31285c194e 100644
--- a/llvm/test/Transforms/SLPVectorizer/AArch64/extracts-from-scalarizable-vector.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AArch64/extracts-from-scalarizable-vector.ll
@@ -4,15 +4,7 @@
define i1 @degenerate() {
; CHECK-LABEL: define i1 @degenerate() {
; CHECK-NEXT: entry:
-; CHECK-NEXT: [[TMP0:%.*]] = extractelement <4 x fp128> zeroinitializer, i32 0
-; CHECK-NEXT: [[CMP:%.*]] = fcmp ogt fp128 [[TMP0]], 0xL00000000000000000000000000000000
-; CHECK-NEXT: [[CMP3:%.*]] = fcmp olt fp128 [[TMP0]], 0xL00000000000000000000000000000000
-; CHECK-NEXT: [[OR_COND:%.*]] = and i1 [[CMP]], [[CMP3]]
-; CHECK-NEXT: [[TMP1:%.*]] = extractelement <4 x fp128> zeroinitializer, i32 0
-; CHECK-NEXT: [[CMP6:%.*]] = fcmp ogt fp128 [[TMP1]], 0xL00000000000000000000000000000000
-; CHECK-NEXT: [[OR_COND29:%.*]] = select i1 [[OR_COND]], i1 [[CMP6]], i1 false
-; CHECK-NEXT: [[CMP10:%.*]] = fcmp olt fp128 [[TMP1]], 0xL00000000000000000000000000000000
-; CHECK-NEXT: [[OR_COND30:%.*]] = select i1 [[OR_COND29]], i1 [[CMP10]], i1 false
+; CHECK-NEXT: [[OR_COND30:%.*]] = call i1 @llvm.vector.reduce.and.v4i1(<4 x i1> zeroinitializer)
; CHECK-NEXT: ret i1 [[OR_COND30]]
;
entry:
@@ -32,16 +24,13 @@ define i1 @with_inputs(<4 x fp128> %a) {
; CHECK-LABEL: define i1 @with_inputs
; CHECK-SAME: (<4 x fp128> [[A:%.*]]) {
; CHECK-NEXT: entry:
-; CHECK-NEXT: [[TMP0:%.*]] = extractelement <4 x fp128> [[A]], i32 0
-; CHECK-NEXT: [[CMP:%.*]] = fcmp ogt fp128 [[TMP0]], 0xL00000000000000000000000000000000
-; CHECK-NEXT: [[CMP3:%.*]] = fcmp olt fp128 [[TMP0]], 0xL00000000000000000000000000000000
-; CHECK-NEXT: [[OR_COND:%.*]] = and i1 [[CMP]], [[CMP3]]
-; CHECK-NEXT: [[TMP1:%.*]] = extractelement <4 x fp128> [[A]], i32 1
-; CHECK-NEXT: [[CMP6:%.*]] = fcmp ogt fp128 [[TMP1]], 0xL00000000000000000000000000000000
-; CHECK-NEXT: [[OR_COND29:%.*]] = select i1 [[OR_COND]], i1 [[CMP6]], i1 false
-; CHECK-NEXT: [[CMP10:%.*]] = fcmp olt fp128 [[TMP1]], 0xL00000000000000000000000000000000
-; CHECK-NEXT: [[OR_COND30:%.*]] = select i1 [[OR_COND29]], i1 [[CMP10]], i1 false
-; CHECK-NEXT: ret i1 [[OR_COND30]]
+; CHECK-NEXT: [[TMP0:%.*]] = shufflevector <4 x fp128> [[A]], <4 x fp128> poison, <4 x i32> <i32 0, i32 1, i32 0, i32 1>
+; CHECK-NEXT: [[TMP1:%.*]] = fcmp olt <4 x fp128> [[TMP0]], zeroinitializer
+; CHECK-NEXT: [[TMP2:%.*]] = fcmp ogt <4 x fp128> [[TMP0]], zeroinitializer
+; CHECK-NEXT: [[TMP3:%.*]] = shufflevector <4 x i1> [[TMP1]], <4 x i1> [[TMP2]], <4 x i32> <i32 0, i32 1, i32 6, i32 7>
+; CHECK-NEXT: [[TMP4:%.*]] = freeze <4 x i1> [[TMP3]]
+; CHECK-NEXT: [[TMP5:%.*]] = call i1 @llvm.vector.reduce.and.v4i1(<4 x i1> [[TMP4]])
+; CHECK-NEXT: ret i1 [[TMP5]]
;
entry:
%0 = extractelement <4 x fp128> %a, i32 0
diff --git a/llvm/test/Transforms/SLPVectorizer/AArch64/vectorizable-selects-min-max.ll b/llvm/test/Transforms/SLPVectorizer/AArch64/vectorizable-selects-min-max.ll
index c0b882b98d15b..413a09bbdce49 100644
--- a/llvm/test/Transforms/SLPVectorizer/AArch64/vectorizable-selects-min-max.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AArch64/vectorizable-selects-min-max.ll
@@ -103,12 +103,19 @@ entry:
define void @select_ule_ugt_mix_4xi32(ptr %ptr, i32 %x) {
; CHECK-LABEL: @select_ule_ugt_mix_4xi32(
; CHECK-NEXT: entry:
-; CHECK-NEXT: [[TMP1:%.*]] = load <4 x i32>, ptr [[PTR:%.*]], align 4
-; CHECK-NEXT: [[TMP2:%.*]] = icmp ult <4 x i32> [[TMP1]], splat (i32 16383)
-; CHECK-NEXT: [[TMP3:%.*]] = icmp ugt <4 x i32> [[TMP1]], splat (i32 16383)
-; CHECK-NEXT: [[TMP4:%.*]] = shufflevector <4 x i1> [[TMP2]], <4 x i1> [[TMP3]], <4 x i32> <i32 0, i32 5, i32 2, i32 7>
-; CHECK-NEXT: [[TMP5:%.*]] = select <4 x i1> [[TMP4]], <4 x i32> [[TMP1]], <4 x i32> splat (i32 16383)
-; CHECK-NEXT: store <4 x i32> [[TMP5]], ptr [[PTR]], align 4
+; CHECK-NEXT: [[TMP0:%.*]] = load <2 x i32>, ptr [[PTR:%.*]], align 4
+; CHECK-NEXT: [[TMP1:%.*]] = shufflevector <2 x i32> [[TMP0]], <2 x i32> <i32 poison, i32 16383>, <2 x i32> <i32 0, i32 3>
+; CHECK-NEXT: [[TMP2:%.*]] = shufflevector <2 x i32> [[TMP0]], <2 x i32> <i32 16383, i32 poison>, <2 x i32> <i32 2, i32 1>
+; CHECK-NEXT: [[TMP3:%.*]] = icmp ult <2 x i32> [[TMP1]], [[TMP2]]
+; CHECK-NEXT: [[TMP4:%.*]] = select <2 x i1> [[TMP3]], <2 x i32> [[TMP0]], <2 x i32> splat (i32 16383)
+; CHECK-NEXT: store <2 x i32> [[TMP4]], ptr [[PTR]], align 4
+; CHECK-NEXT: [[GEP_2:%.*]] = getelementptr inbounds i32, ptr [[PTR]], i32 2
+; CHECK-NEXT: [[TMP5:%.*]] = load <2 x i32>, ptr [[GEP_2]], align 4
+; CHECK-NEXT: [[TMP6:%.*]] = shufflevector <2 x i32> [[TMP5]], <2 x i32> <i32 poison, i32 16383>, <2 x i32> <i32 0, i32 3>
+; CHECK-NEXT: [[TMP7:%.*]] = shufflevector <2 x i32> [[TMP5]], <2 x i32> <i32 16383, i32 poison>, <2 x i32> <i32 2, i32 1>
+; CHECK-NEXT: [[TMP8:%.*]] = icmp ult <2 x i32> [[TMP6]], [[TMP7]]
+; CHECK-NEXT: [[TMP9:%.*]] = select <2 x i1> [[TMP8]], <2 x i32> [[TMP5]], <2 x i32> splat (i32 16383)
+; CHECK-NEXT: store <2 x i32> [[TMP9]], ptr [[GEP_2]], align 4
; CHECK-NEXT: ret void
;
entry:
diff --git a/llvm/test/Transforms/SLPVectorizer/RISCV/remarks-insert-into-small-vector.ll b/llvm/test/Transforms/SLPVectorizer/RISCV/remarks-insert-into-small-vector.ll
index 7fd88e54d4440..e39d887ec75d4 100644
--- a/llvm/test/Transforms/SLPVectorizer/RISCV/remarks-insert-into-small-vector.ll
+++ b/llvm/test/Transforms/SLPVectorizer/RISCV/remarks-insert-into-small-vector.ll
@@ -8,7 +8,7 @@
; YAML-NEXT: Function: test
; YAML-NEXT: Args:
; YAML-NEXT: - String: 'Stores SLP vectorized with cost '
-; YAML-NEXT: - Cost: '-2'
+; YAML-NEXT: - Cost: '0'
; YAML-NEXT: - String: ' and with tree size '
; YAML-NEXT: - TreeSize: '9'
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/bool-mask.ll b/llvm/test/Transforms/SLPVectorizer/X86/bool-mask.ll
index 2780dd363035f..426043033da90 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/bool-mask.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/bool-mask.ll
@@ -258,44 +258,12 @@ entry:
define i64 @bitmask_8xi64(ptr nocapture noundef readonly %src) {
; SSE2-LABEL: @bitmask_8xi64(
; SSE2-NEXT: entry:
-; SSE2-NEXT: [[TMP0:%.*]] = load i64, ptr [[SRC:%.*]], align 8
-; SSE2-NEXT: [[TOBOOL_NOT:%.*]] = icmp ne i64 [[TMP0]], 0
-; SSE2-NEXT: [[OR:%.*]] = zext i1 [[TOBOOL_NOT]] to i64
-; SSE2-NEXT: [[ARRAYIDX_1:%.*]] = getelementptr inbounds i64, ptr [[SRC]], i64 1
-; SSE2-NEXT: [[TMP1:%.*]] = load i64, ptr [[ARRAYIDX_1]], align 8
-; SSE2-NEXT: [[TOBOOL_NOT_1:%.*]] = icmp eq i64 [[TMP1]], 0
-; SSE2-NEXT: [[OR_1:%.*]] = select i1 [[TOBOOL_NOT_1]], i64 0, i64 2
-; SSE2-NEXT: [[MASK_1_1:%.*]] = or i64 [[OR_1]], [[OR]]
-; SSE2-NEXT: [[ARRAYIDX_2:%.*]] = getelementptr inbounds i64, ptr [[SRC]], i64 2
-; SSE2-NEXT: [[TMP2:%.*]] = load i64, ptr [[ARRAYIDX_2]], align 8
-; SSE2-NEXT: [[TOBOOL_NOT_2:%.*]] = icmp eq i64 [[TMP2]], 0
-; SSE2-NEXT: [[OR_2:%.*]] = select i1 [[TOBOOL_NOT_2]], i64 0, i64 4
-; SSE2-NEXT: [[MASK_1_2:%.*]] = or i64 [[OR_2]], [[MASK_1_1]]
-; SSE2-NEXT: [[ARRAYIDX_3:%.*]] = getelementptr inbounds i64, ptr [[SRC]], i64 3
-; SSE2-NEXT: [[TMP3:%.*]] = load i64, ptr [[ARRAYIDX_3]], align 8
-; SSE2-NEXT: [[TOBOOL_NOT_3:%.*]] = icmp eq i64 [[TMP3]], 0
-; SSE2-NEXT: [[OR_3:%.*]] = select i1 [[TOBOOL_NOT_3]], i64 0, i64 8
-; SSE2-NEXT: [[MASK_1_3:%.*]] = or i64 [[OR_3]], [[MASK_1_2]]
-; SSE2-NEXT: [[ARRAYIDX_4:%.*]] = getelementptr inbounds i64, ptr [[SRC]], i64 4
-; SSE2-NEXT: [[TMP4:%.*]] = load i64, ptr [[ARRAYIDX_4]], align 8
-; SSE2-NEXT: [[TOBOOL_NOT_4:%.*]] = icmp eq i64 [[TMP4]], 0
-; SSE2-NEXT: [[OR_4:%.*]] = select i1 [[TOBOOL_NOT_4]], i64 0, i64 16
-; SSE2-NEXT: [[MASK_1_4:%.*]] = or i64 [[OR_4]], [[MASK_1_3]]
-; SSE2-NEXT: [[ARRAYIDX_5:%.*]] = getelementptr inbounds i64, ptr [[SRC]], i64 5
-; SSE2-NEXT: [[TMP5:%.*]] = load i64, ptr [[ARRAYIDX_5]], align 8
-; SSE2-NEXT: [[TOBOOL_NOT_5:%.*]] = icmp eq i64 [[TMP5]], 0
-; SSE2-NEXT: [[OR_5:%.*]] = select i1 [[TOBOOL_NOT_5]], i64 0, i64 32
-; SSE2-NEXT: [[MASK_1_5:%.*]] = or i64 [[OR_5]], [[MASK_1_4]]
-; SSE2-NEXT: [[ARRAYIDX_6:%.*]] = getelementptr inbounds i64, ptr [[SRC]], i64 6
-; SSE2-NEXT: [[TMP6:%.*]] = load i64, ptr [[ARRAYIDX_6]], align 8
-; SSE2-NEXT: [[TOBOOL_NOT_6:%.*]] = icmp eq i64 [[TMP6]], 0
-; SSE2-NEXT: [[OR_6:%.*]] = select i1 [[TOBOOL_NOT_6]], i64 0, i64 64
-; SSE2-NEXT: [[MASK_1_6:%.*]] = or i64 [[OR_6]], [[MASK_1_5]]
-; SSE2-NEXT: [[ARRAYIDX_7:%.*]] = getelementptr inbounds i64, ptr [[SRC]], i64 7
-; SSE2-NEXT: [[TMP7:%.*]] = load i64, ptr [[ARRAYIDX_7]], align 8
-; SSE2-NEXT: [[TOBOOL_NOT_7:%.*]] = icmp eq i64 [[TMP7]], 0
-; SSE2-NEXT: [[OR_7:%.*]] = select i1 [[TOBOOL_NOT_7]], i64 0, i64 128
-; SSE2-NEXT: [[MASK_1_7:%.*]] = or i64 [[OR_7]], [[MASK_1_6]]
+; SSE2-NEXT: [[TMP0:%.*]] = load <8 x i64>, ptr [[SRC:%.*]], align 8
+; SSE2-NEXT: [[TMP1:%.*]] = icmp ne <8 x i64> [[TMP0]], zeroinitializer
+; SSE2-NEXT: [[TMP2:%.*]] = icmp eq <8 x i64> [[TMP0]], zeroinitializer
+; SSE2-NEXT: [[TMP3:%.*]] = shufflevector <8 x i1> [[TMP1]], <8 x i1> [[TMP2]], <8 x i32> <i32 0, i32 9, i32 10, i32 11, i32 12, i32 13, i32 14, i32 15>
+; SSE2-NEXT: [[TMP4:%.*]] = select <8 x i1> [[TMP3]], <8 x i64> <i64 1, i64 0, i64 0, i64 0, i64 0, i64 0, i64 0, i64 0>, <8 x i64> <i64 0, i64 2, i64 4, i64 8, i64 16, i64 32, i64 64, i64 128>
+; SSE2-NEXT: [[MASK_1_7:%.*]] = call i64 @llvm.vector.reduce.or.v8i64(<8 x i64> [[TMP4]])
; SSE2-NEXT: ret i64 [[MASK_1_7]]
;
; SSE4-LABEL: @bitmask_8xi64(
@@ -367,64 +335,13 @@ entry:
define i64 @combined(ptr nocapture noundef readonly %src) {
; SSE2-LABEL: @combined(
; SSE2-NEXT: entry:
-; SSE2-NEXT: [[TMP0:%.*]] = load i64, ptr [[SRC:%.*]], align 2
-; SSE2-NEXT: [[TOBOOL_NOT:%.*]] = icmp ne i64 [[TMP0]], 0
-; SSE2-NEXT: [[OR:%.*]] = zext i1 [[TOBOOL_NOT]] to i64
-; SSE2-NEXT: [[ARRAYIDX_1:%.*]] = getelementptr inbounds i64, ptr [[SRC]], i64 1
-; SSE2-NEXT: [[TMP1:%.*]] = load i64, ptr [[ARRAYIDX_1]], align 2
-; SSE2-NEXT: [[TOBOOL_NOT_1:%.*]] = icmp eq i64 [[TMP1]], 0
-; SSE2-NEXT: [[OR_1:%.*]] = select i1 [[TOBOOL_NOT_1]], i64 0, i64 2
-; SSE2-NEXT: [[MASK_1_1:%.*]] = or i64 [[OR_1]], [[OR]]
-; SSE2-NEXT: [[ARRAYIDX_2:%.*]] = getelementptr inbounds i64, ptr [[SRC]], i64 2
-; SSE2-NEXT: [[TMP2:%.*]] = load i64, ptr [[ARRAYIDX_2]], align 2
-; SSE2-NEXT: [[TOBOOL_NOT_2:%.*]] = icmp eq i64 [[TMP2]], 0
-; SSE2-NEXT: [[OR_2:%.*]] = select i1 [[TOBOOL_NOT_2]], i64 0, i64 4
-; SSE2-NEXT: [[MASK_1_2:%.*]] = or i64 [[OR_2]], [[MASK_1_1]]
-; SSE2-NEXT: [[ARRAYIDX_3:%.*]] = getelementptr inbounds i64, ptr [[SRC]], i64 3
-; SSE2-NEXT: [[TMP3:%.*]] = load i64, ptr [[ARRAYIDX_3]], align 2
-; SSE2-NEXT: [[TOBOOL_NOT_3:%.*]] = icmp eq i64 [[TMP3]], 0
-; SSE2-NEXT: [[OR_3:%.*]] = select i1 [[TOBOOL_NOT_3]], i64 0,...
[truncated]
``````````
</details>
https://github.com/llvm/llvm-project/pull/190618
More information about the llvm-commits
mailing list