[llvm] [SLP] Vectorize zero-tested OR/UMax reductions (PR #205473)
via llvm-commits
llvm-commits at lists.llvm.org
Mon Aug 10 20:47:27 PDT 2026
https://github.com/ParkHanbum updated https://github.com/llvm/llvm-project/pull/205473
>From c54de8da5878b93eb8be0b2eaed21ec1ffab929a Mon Sep 17 00:00:00 2001
From: hanbeom <kese111 at gmail.com>
Date: Sat, 25 Jul 2026 03:32:30 +0900
Subject: [PATCH 01/11] [SLP] Vectorize zero-tested OR/UMax reductions
When a scalar OR or UMax reduction has a single eq/ne-zero use, form
an equivalent lane-wise vector comparison followed by an i1 AND/OR
reduction.
Model the vectorized form as a vector comparison plus the corresponding
boolean reduction, avoiding an integer reduction whose scalar result is
used only by the zero test.
Fixes #195118
---
.../Transforms/Vectorize/SLPVectorizer.cpp | 130 ++++++++++++++++--
.../WebAssembly/or-reduction-zero-test.ll | 42 +-----
.../X86/reduction-zero-test-ctpop.ll | 20 +--
3 files changed, 132 insertions(+), 60 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index a04877d183516..5243ed4e70dd2 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -29649,6 +29649,11 @@ namespace {
class HorizontalReduction {
using ReductionOpsType = SmallVector<Value *, 16>;
using ReductionOpsListType = SmallVector<ReductionOpsType, 2>;
+ enum class ReductionContext {
+ None,
+ CmpZero,
+ };
+
ReductionOpsListType ReductionOps;
/// List of possibly reduced values.
SmallVector<SmallVector<Value *>> ReducedVals;
@@ -29680,6 +29685,26 @@ class HorizontalReduction {
(match(I, m_LogicalAnd()) || match(I, m_LogicalOr()));
}
+ /// Return CmpZero for a scalar OR/UMax reduction whose only use is an eq/ne
+ /// comparison against zero.
+ ReductionContext getReductionContext() const {
+ auto *Root = dyn_cast<Instruction>(ReductionRoot);
+ if (!Root || !Root->getType()->isIntegerTy() || !Root->hasOneUse() ||
+ (RdxKind != RecurKind::Or && RdxKind != RecurKind::UMax))
+ return ReductionContext::None;
+
+ CmpPredicate Pred;
+ if (!match(*Root->user_begin(),
+ m_c_ICmp(Pred, m_Specific(Root), m_ZeroInt())) ||
+ !ICmpInst::isEquality(Pred))
+ return ReductionContext::None;
+ return ReductionContext::CmpZero;
+ }
+
+ static bool isZeroCmpContext(ReductionContext Context) {
+ return Context == ReductionContext::CmpZero;
+ }
+
/// Checks if instruction is associative and can be vectorized.
enum class ReductionOrdering { Unordered, Ordered, None };
ReductionOrdering RK = ReductionOrdering::None;
@@ -30484,6 +30509,7 @@ class HorizontalReduction {
if (RK == ReductionOrdering::Ordered)
IgnoreList.clear();
bool IsCmpSelMinMax = isCmpSelMinMax(cast<Instruction>(ReductionRoot));
+ ReductionContext RdxContext = getReductionContext();
// Need to track reduced vals, they may be changed during vectorization of
// subvectors.
@@ -30503,6 +30529,7 @@ class HorizontalReduction {
// nodes and thus requiring extract if fully vectorized in other trees.
SmallPtrSet<Value *, 4> RequiredExtract;
WeakTrackingVH VectorizedTree = nullptr;
+ ReductionContext VectorizedReductionContext = ReductionContext::None;
bool CheckForReusedReductionOps = false;
// Try to vectorize elements based on their type.
SmallVector<InstructionsState> States;
@@ -30881,6 +30908,17 @@ class HorizontalReduction {
LocalExternallyUsedValues.insert(RdxVal);
V.buildExternalUses(LocalExternallyUsedValues);
+ // Use the zero-test cost only for a complete, standalone scalar
+ // reduction that can become one vector comparison and i1 reduction.
+ ReductionContext CostContext = ReductionContext::None;
+ if (isZeroCmpContext(RdxContext) &&
+ this->ReducedVals.size() == 1 &&
+ VL.size() == this->ReducedVals.front().size() &&
+ VectorValuesAndScales.empty() && !VectorizedTree &&
+ !isa<VectorType>(VL.front()->getType()) && !allConstant(VL) &&
+ !V.isReducedBitcastRoot() && !V.isReducedCmpBitcastRoot())
+ CostContext = RdxContext;
+
// Estimate cost.
InstructionCost ReductionCost;
if (RK == ReductionOrdering::Ordered || V.isReducedBitcastRoot() ||
@@ -30889,7 +30927,7 @@ class HorizontalReduction {
else
ReductionCost =
getReductionCost(TTI, VL, SameValuesCounter, IsCmpSelMinMax,
- RdxFMF, V, DT, DL, TLI);
+ CostContext, RdxFMF, V, DT, DL, TLI);
// If the root is a select (min/max idiom), the insert point is the
// compare condition of that select.
Instruction *RdxRootInst = cast<Instruction>(ReductionRoot);
@@ -30993,6 +31031,8 @@ class HorizontalReduction {
Type *ScalarTy = VL.front()->getType();
Type *VecTy = VectorizedRoot->getType();
Type *RedScalarTy = VecTy->getScalarType();
+ if (isZeroCmpContext(CostContext))
+ VectorizedReductionContext = CostContext;
VectorValuesAndScales.emplace_back(
VectorizedRoot,
OptReusedScalars && SameScaleFactor
@@ -31038,8 +31078,8 @@ class HorizontalReduction {
if (!VectorValuesAndScales.empty())
VectorizedTree = GetNewVectorizedTree(
- VectorizedTree,
- emitReduction(Builder, *TTI, ReductionRoot->getType()));
+ VectorizedTree, emitReduction(Builder, *TTI, ReductionRoot->getType(),
+ VectorizedReductionContext));
if (!VectorizedTree) {
if (!CheckForReusedReductionOps) {
@@ -31157,7 +31197,18 @@ class HorizontalReduction {
}
VectorizedTree = ExtraReductions.front().second;
- ReductionRoot->replaceAllUsesWith(VectorizedTree);
+ if (isZeroCmpContext(VectorizedReductionContext)) {
+ auto *Cmp =
+ cast<ICmpInst>(*cast<Instruction>(ReductionRoot)->user_begin());
+ VectorizedTree->takeName(Cmp);
+ Cmp->replaceAllUsesWith(VectorizedTree);
+ salvageDebugInfo(*Cmp);
+ Cmp->dropAllReferences();
+ Cmp->removeFromParent();
+ V.eraseInstruction(Cmp);
+ } else {
+ ReductionRoot->replaceAllUsesWith(VectorizedTree);
+ }
// The original scalar reduction is expected to have no remaining
// uses outside the reduction tree itself. Assert that we got this
@@ -31312,9 +31363,10 @@ class HorizontalReduction {
V.calculateTreeCostAndTrimNonProfitable(VL, RdxRootInst);
V.buildExternalUses(LocalExternallyUsedValues);
- InstructionCost ReductionCost =
- getReductionCost(TTI, VL, EmptySameValuesCounter,
- /*IsCmpSelMinMax=*/false, RdxFMF, V, DT, DL, TLI);
+ InstructionCost ReductionCost = getReductionCost(
+ TTI, VL, EmptySameValuesCounter,
+ /*IsCmpSelMinMax=*/false, ReductionContext::None, RdxFMF, V, DT, DL,
+ TLI);
InstructionCost Cost =
V.getTreeCost(TreeCost, VL, ReductionCost, RdxRootInst);
LLVM_DEBUG(dbgs() << "SLP: Found cost = " << Cost
@@ -31497,8 +31549,9 @@ class HorizontalReduction {
InstructionCost getReductionCost(
TargetTransformInfo *TTI, ArrayRef<Value *> ReducedVals,
const SmallMapVector<Value *, unsigned, 16> SameValuesCounter,
- bool IsCmpSelMinMax, FastMathFlags FMF, const BoUpSLP &R,
- DominatorTree &DT, const DataLayout &DL, const TargetLibraryInfo &TLI) {
+ bool IsCmpSelMinMax, ReductionContext Context, FastMathFlags FMF,
+ const BoUpSLP &R, DominatorTree &DT, const DataLayout &DL,
+ const TargetLibraryInfo &TLI) {
TTI::TargetCostKind CostKind = TTI::TCK_RecipThroughput;
Type *ScalarTy = ReducedVals.front()->getType();
unsigned ReduxWidth = ReducedVals.size();
@@ -31581,6 +31634,36 @@ class HorizontalReduction {
// 2. The storage does not have any vector with full vector use (first
// vector with full register use).
bool DoesRequireReductionOp = !AllConsts && VectorValuesAndScales.empty();
+ InstructionCost CmpReductionCost = InstructionCost::getInvalid();
+ if (isZeroCmpContext(Context)) {
+ // For a complete OR/UMax reduction used only by an eq/ne zero test,
+ // account for moving the comparison before the reduction:
+ //
+ // %rdx = call iN @llvm.vector.reduce.or/umax(<VF x iN> %vec)
+ // %cmp = icmp eq/ne iN %rdx, 0
+ //
+ // becomes:
+ //
+ // %lane.cmp = icmp eq/ne <VF x iN> %vec, zeroinitializer
+ // %cmp = call i1 @llvm.vector.reduce.and/or(<VF x i1> %lane.cmp)
+ //
+ // Equality requires every lane to be zero, so it uses an AND reduction;
+ // inequality requires any lane to be nonzero, so it uses OR. Add the
+ // vector comparison and boolean reduction costs.
+ assert(DoesRequireReductionOp && !isa<VectorType>(ScalarTy) &&
+ (RdxKind == RecurKind::Or || RdxKind == RecurKind::UMax) &&
+ "Unexpected zero comparison reduction");
+ auto *ScalarCmp =
+ cast<ICmpInst>(*cast<Instruction>(ReductionRoot)->user_begin());
+ CmpPredicate Pred = ScalarCmp->getPredicate();
+ auto *CmpTy = cast<VectorType>(CmpInst::makeCmpResultType(VectorTy));
+ unsigned ReductionOpcode =
+ Pred == ICmpInst::ICMP_EQ ? Instruction::And : Instruction::Or;
+ CmpReductionCost = TTI->getCmpSelInstrCost(Instruction::ICmp, VectorTy,
+ CmpTy, Pred, CostKind) +
+ TTI->getArithmeticReductionCost(ReductionOpcode, CmpTy,
+ {}, CostKind);
+ }
switch (RdxKind) {
case RecurKind::Add:
case RecurKind::Mul:
@@ -31592,7 +31675,9 @@ class HorizontalReduction {
unsigned RdxOpcode = RecurrenceDescriptor::getOpcode(RdxKind);
if (!AllConsts) {
if (DoesRequireReductionOp) {
- if (auto *VecTy = dyn_cast<FixedVectorType>(ScalarTy)) {
+ if (isZeroCmpContext(Context)) {
+ VectorCost = CmpReductionCost;
+ } else if (auto *VecTy = dyn_cast<FixedVectorType>(ScalarTy)) {
assert(SLPReVec && "FixedVectorType is not expected.");
unsigned ScalarTyNumElements = VecTy->getNumElements();
for (unsigned I : seq<unsigned>(ReducedVals.size())) {
@@ -31697,7 +31782,11 @@ class HorizontalReduction {
Intrinsic::ID Id = getMinMaxReductionIntrinsicOp(RdxKind);
if (!AllConsts) {
if (DoesRequireReductionOp) {
- VectorCost = TTI->getMinMaxReductionCost(Id, VectorTy, FMF, CostKind);
+ if (isZeroCmpContext(Context))
+ VectorCost = CmpReductionCost;
+ else
+ VectorCost =
+ TTI->getMinMaxReductionCost(Id, VectorTy, FMF, CostKind);
} else {
// Check if the previous reduction already exists and account it as
// series of operations + single reduction.
@@ -31737,7 +31826,24 @@ class HorizontalReduction {
/// sub-registers, combines them with the given reduction operation as a
/// vector operation and then performs single (small enough) reduction.
Value *emitReduction(IRBuilderBase &Builder, const TargetTransformInfo &TTI,
- Type *DestTy) {
+ Type *DestTy, ReductionContext Context) {
+ if (isZeroCmpContext(Context)) {
+ assert(VectorValuesAndScales.size() == 1 &&
+ !std::get<3>(VectorValuesAndScales.front()) &&
+ "Expected one complete vector reduction");
+ Value *Vec = std::get<0>(VectorValuesAndScales.front());
+ auto *VecTy = cast<VectorType>(Vec->getType());
+ auto *ScalarCmp =
+ cast<ICmpInst>(*cast<Instruction>(ReductionRoot)->user_begin());
+ CmpPredicate Pred = ScalarCmp->getPredicate();
+ Builder.SetCurrentDebugLocation(ScalarCmp->getDebugLoc());
+ Value *Cmp = Builder.CreateICmp(Pred, Vec, Constant::getNullValue(VecTy));
+ RecurKind BoolRdxKind =
+ Pred == ICmpInst::ICMP_EQ ? RecurKind::And : RecurKind::Or;
+ NumVectorInstructions += 2;
+ return createSimpleReduction(Builder, Cmp, BoolRdxKind);
+ }
+
Value *ReducedSubTree = nullptr;
// Creates reduction and combines with the previous reduction.
auto CreateSingleOp = [&](Value *Vec, unsigned Scale, bool IsSigned,
diff --git a/llvm/test/Transforms/SLPVectorizer/WebAssembly/or-reduction-zero-test.ll b/llvm/test/Transforms/SLPVectorizer/WebAssembly/or-reduction-zero-test.ll
index 34ab34d8a2538..94d4e090f2b8a 100644
--- a/llvm/test/Transforms/SLPVectorizer/WebAssembly/or-reduction-zero-test.ll
+++ b/llvm/test/Transforms/SLPVectorizer/WebAssembly/or-reduction-zero-test.ll
@@ -6,25 +6,8 @@ define i1 @or_reduction_nonzero(ptr %p) {
; CHECK-SAME: ptr [[P:%.*]]) #[[ATTR0:[0-9]+]] {
; CHECK-NEXT: [[ENTRY:.*:]]
; CHECK-NEXT: [[TMP0:%.*]] = load <16 x i8>, ptr [[P]], align 1
-; CHECK-NEXT: [[TMP18:%.*]] = shufflevector <16 x i8> [[TMP0]], <16 x i8> poison, <2 x i32> <i32 0, i32 8>
-; CHECK-NEXT: [[TMP2:%.*]] = shufflevector <16 x i8> [[TMP0]], <16 x i8> poison, <2 x i32> <i32 1, i32 9>
-; CHECK-NEXT: [[TMP3:%.*]] = or <2 x i8> [[TMP18]], [[TMP2]]
-; CHECK-NEXT: [[TMP4:%.*]] = shufflevector <16 x i8> [[TMP0]], <16 x i8> poison, <2 x i32> <i32 2, i32 10>
-; CHECK-NEXT: [[TMP5:%.*]] = shufflevector <16 x i8> [[TMP0]], <16 x i8> poison, <2 x i32> <i32 3, i32 11>
-; CHECK-NEXT: [[TMP6:%.*]] = or <2 x i8> [[TMP4]], [[TMP5]]
-; CHECK-NEXT: [[TMP7:%.*]] = shufflevector <16 x i8> [[TMP0]], <16 x i8> poison, <2 x i32> <i32 4, i32 12>
-; CHECK-NEXT: [[TMP8:%.*]] = shufflevector <16 x i8> [[TMP0]], <16 x i8> poison, <2 x i32> <i32 5, i32 13>
-; CHECK-NEXT: [[TMP9:%.*]] = or <2 x i8> [[TMP7]], [[TMP8]]
-; CHECK-NEXT: [[TMP10:%.*]] = shufflevector <16 x i8> [[TMP0]], <16 x i8> poison, <2 x i32> <i32 6, i32 14>
-; CHECK-NEXT: [[TMP11:%.*]] = shufflevector <16 x i8> [[TMP0]], <16 x i8> poison, <2 x i32> <i32 7, i32 15>
-; CHECK-NEXT: [[TMP12:%.*]] = or <2 x i8> [[TMP10]], [[TMP11]]
-; CHECK-NEXT: [[TMP13:%.*]] = or <2 x i8> [[TMP3]], [[TMP6]]
-; CHECK-NEXT: [[TMP14:%.*]] = or <2 x i8> [[TMP9]], [[TMP12]]
-; CHECK-NEXT: [[TMP15:%.*]] = or <2 x i8> [[TMP13]], [[TMP14]]
-; CHECK-NEXT: [[TMP16:%.*]] = extractelement <2 x i8> [[TMP15]], i64 0
-; CHECK-NEXT: [[TMP17:%.*]] = extractelement <2 x i8> [[TMP15]], i64 1
-; CHECK-NEXT: [[TMP1:%.*]] = or i8 [[TMP16]], [[TMP17]]
-; CHECK-NEXT: [[CMP:%.*]] = icmp ne i8 [[TMP1]], 0
+; CHECK-NEXT: [[TMP1:%.*]] = icmp ne <16 x i8> [[TMP0]], zeroinitializer
+; CHECK-NEXT: [[CMP:%.*]] = call i1 @llvm.vector.reduce.or.v16i1(<16 x i1> [[TMP1]])
; CHECK-NEXT: ret i1 [[CMP]]
;
entry:
@@ -83,25 +66,8 @@ define i1 @or_reduction_zero(ptr %p) {
; CHECK-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
; CHECK-NEXT: [[ENTRY:.*:]]
; CHECK-NEXT: [[TMP0:%.*]] = load <16 x i8>, ptr [[P]], align 1
-; CHECK-NEXT: [[TMP18:%.*]] = shufflevector <16 x i8> [[TMP0]], <16 x i8> poison, <2 x i32> <i32 0, i32 8>
-; CHECK-NEXT: [[TMP2:%.*]] = shufflevector <16 x i8> [[TMP0]], <16 x i8> poison, <2 x i32> <i32 1, i32 9>
-; CHECK-NEXT: [[TMP3:%.*]] = or <2 x i8> [[TMP18]], [[TMP2]]
-; CHECK-NEXT: [[TMP4:%.*]] = shufflevector <16 x i8> [[TMP0]], <16 x i8> poison, <2 x i32> <i32 2, i32 10>
-; CHECK-NEXT: [[TMP5:%.*]] = shufflevector <16 x i8> [[TMP0]], <16 x i8> poison, <2 x i32> <i32 3, i32 11>
-; CHECK-NEXT: [[TMP6:%.*]] = or <2 x i8> [[TMP4]], [[TMP5]]
-; CHECK-NEXT: [[TMP7:%.*]] = shufflevector <16 x i8> [[TMP0]], <16 x i8> poison, <2 x i32> <i32 4, i32 12>
-; CHECK-NEXT: [[TMP8:%.*]] = shufflevector <16 x i8> [[TMP0]], <16 x i8> poison, <2 x i32> <i32 5, i32 13>
-; CHECK-NEXT: [[TMP9:%.*]] = or <2 x i8> [[TMP7]], [[TMP8]]
-; CHECK-NEXT: [[TMP10:%.*]] = shufflevector <16 x i8> [[TMP0]], <16 x i8> poison, <2 x i32> <i32 6, i32 14>
-; CHECK-NEXT: [[TMP11:%.*]] = shufflevector <16 x i8> [[TMP0]], <16 x i8> poison, <2 x i32> <i32 7, i32 15>
-; CHECK-NEXT: [[TMP12:%.*]] = or <2 x i8> [[TMP10]], [[TMP11]]
-; CHECK-NEXT: [[TMP13:%.*]] = or <2 x i8> [[TMP3]], [[TMP6]]
-; CHECK-NEXT: [[TMP14:%.*]] = or <2 x i8> [[TMP9]], [[TMP12]]
-; CHECK-NEXT: [[TMP15:%.*]] = or <2 x i8> [[TMP13]], [[TMP14]]
-; CHECK-NEXT: [[TMP16:%.*]] = extractelement <2 x i8> [[TMP15]], i64 0
-; CHECK-NEXT: [[TMP17:%.*]] = extractelement <2 x i8> [[TMP15]], i64 1
-; CHECK-NEXT: [[TMP1:%.*]] = or i8 [[TMP16]], [[TMP17]]
-; CHECK-NEXT: [[CMP:%.*]] = icmp eq i8 [[TMP1]], 0
+; CHECK-NEXT: [[TMP1:%.*]] = icmp eq <16 x i8> [[TMP0]], zeroinitializer
+; CHECK-NEXT: [[CMP:%.*]] = call i1 @llvm.vector.reduce.and.v16i1(<16 x i1> [[TMP1]])
; CHECK-NEXT: ret i1 [[CMP]]
;
entry:
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/reduction-zero-test-ctpop.ll b/llvm/test/Transforms/SLPVectorizer/X86/reduction-zero-test-ctpop.ll
index 67fd3f4acacaa..4f74d64460e90 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/reduction-zero-test-ctpop.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/reduction-zero-test-ctpop.ll
@@ -5,8 +5,8 @@ define i32 @or_nonzero(ptr %p) {
; CHECK-LABEL: define i32 @or_nonzero(
; CHECK-SAME: ptr [[P:%.*]]) #[[ATTR0:[0-9]+]] {
; CHECK-NEXT: [[INPUT:%.*]] = load <8 x i8>, ptr [[P]], align 1
-; CHECK-NEXT: [[TMP3:%.*]] = call i8 @llvm.vector.reduce.or.v8i8(<8 x i8> [[INPUT]])
-; CHECK-NEXT: [[CMP:%.*]] = icmp ne i8 [[TMP3]], 0
+; CHECK-NEXT: [[TMP1:%.*]] = icmp ne <8 x i8> [[INPUT]], zeroinitializer
+; CHECK-NEXT: [[CMP:%.*]] = call i1 @llvm.vector.reduce.or.v8i1(<8 x i1> [[TMP1]])
; CHECK-NEXT: [[RESULT:%.*]] = zext i1 [[CMP]] to i32
; CHECK-NEXT: ret i32 [[RESULT]]
;
@@ -35,8 +35,8 @@ define i32 @or_nonzero_commuted(ptr %p) {
; CHECK-LABEL: define i32 @or_nonzero_commuted(
; CHECK-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
; CHECK-NEXT: [[INPUT:%.*]] = load <8 x i8>, ptr [[P]], align 1
-; CHECK-NEXT: [[TMP1:%.*]] = call i8 @llvm.vector.reduce.or.v8i8(<8 x i8> [[INPUT]])
-; CHECK-NEXT: [[CMP:%.*]] = icmp ne i8 0, [[TMP1]]
+; CHECK-NEXT: [[TMP1:%.*]] = icmp ne <8 x i8> [[INPUT]], zeroinitializer
+; CHECK-NEXT: [[CMP:%.*]] = call i1 @llvm.vector.reduce.or.v8i1(<8 x i1> [[TMP1]])
; CHECK-NEXT: [[RESULT:%.*]] = zext i1 [[CMP]] to i32
; CHECK-NEXT: ret i32 [[RESULT]]
;
@@ -65,8 +65,8 @@ define i32 @or_zero(ptr %p) {
; CHECK-LABEL: define i32 @or_zero(
; CHECK-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
; CHECK-NEXT: [[INPUT:%.*]] = load <8 x i8>, ptr [[P]], align 1
-; CHECK-NEXT: [[TMP3:%.*]] = call i8 @llvm.vector.reduce.or.v8i8(<8 x i8> [[INPUT]])
-; CHECK-NEXT: [[CMP:%.*]] = icmp eq i8 [[TMP3]], 0
+; CHECK-NEXT: [[TMP1:%.*]] = icmp eq <8 x i8> [[INPUT]], zeroinitializer
+; CHECK-NEXT: [[CMP:%.*]] = call i1 @llvm.vector.reduce.and.v8i1(<8 x i1> [[TMP1]])
; CHECK-NEXT: [[RESULT:%.*]] = zext i1 [[CMP]] to i32
; CHECK-NEXT: ret i32 [[RESULT]]
;
@@ -95,8 +95,8 @@ define i32 @umax_nonzero(ptr %p) {
; CHECK-LABEL: define i32 @umax_nonzero(
; CHECK-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
; CHECK-NEXT: [[INPUT:%.*]] = load <8 x i8>, ptr [[P]], align 1
-; CHECK-NEXT: [[TMP3:%.*]] = call i8 @llvm.vector.reduce.umax.v8i8(<8 x i8> [[INPUT]])
-; CHECK-NEXT: [[CMP:%.*]] = icmp ne i8 [[TMP3]], 0
+; CHECK-NEXT: [[TMP1:%.*]] = icmp ne <8 x i8> [[INPUT]], zeroinitializer
+; CHECK-NEXT: [[CMP:%.*]] = call i1 @llvm.vector.reduce.or.v8i1(<8 x i1> [[TMP1]])
; CHECK-NEXT: [[RESULT:%.*]] = zext i1 [[CMP]] to i32
; CHECK-NEXT: ret i32 [[RESULT]]
;
@@ -132,8 +132,8 @@ define i32 @umax_zero_commuted(ptr %p) {
; CHECK-LABEL: define i32 @umax_zero_commuted(
; CHECK-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
; CHECK-NEXT: [[INPUT:%.*]] = load <8 x i8>, ptr [[P]], align 1
-; CHECK-NEXT: [[TMP1:%.*]] = call i8 @llvm.vector.reduce.umax.v8i8(<8 x i8> [[INPUT]])
-; CHECK-NEXT: [[CMP:%.*]] = icmp eq i8 0, [[TMP1]]
+; CHECK-NEXT: [[TMP1:%.*]] = icmp eq <8 x i8> [[INPUT]], zeroinitializer
+; CHECK-NEXT: [[CMP:%.*]] = call i1 @llvm.vector.reduce.and.v8i1(<8 x i1> [[TMP1]])
; CHECK-NEXT: [[RESULT:%.*]] = zext i1 [[CMP]] to i32
; CHECK-NEXT: ret i32 [[RESULT]]
;
>From 9cd14a90d98eba79f4acbdfd1c67d7e1f0ff1919 Mon Sep 17 00:00:00 2001
From: hanbeom <kese111 at gmail.com>
Date: Sun, 26 Jul 2026 00:45:56 +0900
Subject: [PATCH 02/11] Add cost coverage for complete and partial zero-tested
reductions
---
.../X86/reduction-zero-test-partial-cost.ll | 143 ++++++++++++++++++
1 file changed, 143 insertions(+)
create mode 100644 llvm/test/Transforms/SLPVectorizer/X86/reduction-zero-test-partial-cost.ll
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/reduction-zero-test-partial-cost.ll b/llvm/test/Transforms/SLPVectorizer/X86/reduction-zero-test-partial-cost.ll
new file mode 100644
index 0000000000000..73fe770795444
--- /dev/null
+++ b/llvm/test/Transforms/SLPVectorizer/X86/reduction-zero-test-partial-cost.ll
@@ -0,0 +1,143 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -mtriple=x86_64-unknown-linux-gnu -mcpu=x86-64-v4 -passes=slp-vectorizer -slp-threshold=19 -pass-remarks-output=%t -S < %s | FileCheck %s
+; RUN: cat %t | FileCheck -check-prefix=COST %s
+
+; COST-LABEL: Function: or_reduction_nonzero_scalar
+; COST: Cost: '-20'
+define i1 @or_reduction_nonzero_scalar(ptr %p) {
+; CHECK-LABEL: define i1 @or_reduction_nonzero_scalar(
+; CHECK-SAME: ptr [[P:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT: [[TMP1:%.*]] = load <16 x i8>, ptr [[P]], align 1
+; CHECK-NEXT: [[TMP2:%.*]] = icmp ne <16 x i8> [[TMP1]], zeroinitializer
+; CHECK-NEXT: [[CMP:%.*]] = call i1 @llvm.vector.reduce.or.v16i1(<16 x i1> [[TMP2]])
+; CHECK-NEXT: ret i1 [[CMP]]
+;
+ %v0 = load i8, ptr %p, align 1
+ %p1 = getelementptr inbounds i8, ptr %p, i32 1
+ %v1 = load i8, ptr %p1, align 1
+ %p2 = getelementptr inbounds i8, ptr %p, i32 2
+ %v2 = load i8, ptr %p2, align 1
+ %p3 = getelementptr inbounds i8, ptr %p, i32 3
+ %v3 = load i8, ptr %p3, align 1
+ %p4 = getelementptr inbounds i8, ptr %p, i32 4
+ %v4 = load i8, ptr %p4, align 1
+ %p5 = getelementptr inbounds i8, ptr %p, i32 5
+ %v5 = load i8, ptr %p5, align 1
+ %p6 = getelementptr inbounds i8, ptr %p, i32 6
+ %v6 = load i8, ptr %p6, align 1
+ %p7 = getelementptr inbounds i8, ptr %p, i32 7
+ %v7 = load i8, ptr %p7, align 1
+ %p8 = getelementptr inbounds i8, ptr %p, i32 8
+ %v8 = load i8, ptr %p8, align 1
+ %p9 = getelementptr inbounds i8, ptr %p, i32 9
+ %v9 = load i8, ptr %p9, align 1
+ %p10 = getelementptr inbounds i8, ptr %p, i32 10
+ %v10 = load i8, ptr %p10, align 1
+ %p11 = getelementptr inbounds i8, ptr %p, i32 11
+ %v11 = load i8, ptr %p11, align 1
+ %p12 = getelementptr inbounds i8, ptr %p, i32 12
+ %v12 = load i8, ptr %p12, align 1
+ %p13 = getelementptr inbounds i8, ptr %p, i32 13
+ %v13 = load i8, ptr %p13, align 1
+ %p14 = getelementptr inbounds i8, ptr %p, i32 14
+ %v14 = load i8, ptr %p14, align 1
+ %p15 = getelementptr inbounds i8, ptr %p, i32 15
+ %v15 = load i8, ptr %p15, align 1
+
+ %o01 = or i8 %v0, %v1
+ %o23 = or i8 %v2, %v3
+ %o45 = or i8 %v4, %v5
+ %o67 = or i8 %v6, %v7
+ %o89 = or i8 %v8, %v9
+ %o1011 = or i8 %v10, %v11
+ %o1213 = or i8 %v12, %v13
+ %o1415 = or i8 %v14, %v15
+
+ %o0123 = or i8 %o01, %o23
+ %o4567 = or i8 %o45, %o67
+ %o891011 = or i8 %o89, %o1011
+ %o12131415 = or i8 %o1213, %o1415
+
+ %o07 = or i8 %o0123, %o4567
+ %o815 = or i8 %o891011, %o12131415
+ %o015 = or i8 %o07, %o815
+
+ %cmp = icmp ne i8 %o015, 0
+ ret i1 %cmp
+}
+
+; COST-LABEL: Function: or_reduction_nonzero_scalar_remainder
+; COST: Cost: '-21'
+define i1 @or_reduction_nonzero_scalar_remainder(ptr %p) {
+; CHECK-LABEL: define i1 @or_reduction_nonzero_scalar_remainder(
+; CHECK-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[TMP1:%.*]] = load <16 x i8>, ptr [[P]], align 1
+; CHECK-NEXT: [[P16:%.*]] = getelementptr inbounds i8, ptr [[P]], i32 16
+; CHECK-NEXT: [[V16:%.*]] = load i8, ptr [[P16]], align 1
+; CHECK-NEXT: [[P17:%.*]] = getelementptr inbounds i8, ptr [[P]], i32 17
+; CHECK-NEXT: [[V17:%.*]] = load i8, ptr [[P17]], align 1
+; CHECK-NEXT: [[TMP2:%.*]] = call i8 @llvm.vector.reduce.or.v16i8(<16 x i8> [[TMP1]])
+; CHECK-NEXT: [[OP_RDX:%.*]] = or i8 [[TMP2]], [[V16]]
+; CHECK-NEXT: [[OP_RDX1:%.*]] = or i8 [[OP_RDX]], [[V17]]
+; CHECK-NEXT: [[CMP:%.*]] = icmp ne i8 [[OP_RDX1]], 0
+; CHECK-NEXT: ret i1 [[CMP]]
+;
+ %v0 = load i8, ptr %p, align 1
+ %p1 = getelementptr inbounds i8, ptr %p, i32 1
+ %v1 = load i8, ptr %p1, align 1
+ %p2 = getelementptr inbounds i8, ptr %p, i32 2
+ %v2 = load i8, ptr %p2, align 1
+ %p3 = getelementptr inbounds i8, ptr %p, i32 3
+ %v3 = load i8, ptr %p3, align 1
+ %p4 = getelementptr inbounds i8, ptr %p, i32 4
+ %v4 = load i8, ptr %p4, align 1
+ %p5 = getelementptr inbounds i8, ptr %p, i32 5
+ %v5 = load i8, ptr %p5, align 1
+ %p6 = getelementptr inbounds i8, ptr %p, i32 6
+ %v6 = load i8, ptr %p6, align 1
+ %p7 = getelementptr inbounds i8, ptr %p, i32 7
+ %v7 = load i8, ptr %p7, align 1
+ %p8 = getelementptr inbounds i8, ptr %p, i32 8
+ %v8 = load i8, ptr %p8, align 1
+ %p9 = getelementptr inbounds i8, ptr %p, i32 9
+ %v9 = load i8, ptr %p9, align 1
+ %p10 = getelementptr inbounds i8, ptr %p, i32 10
+ %v10 = load i8, ptr %p10, align 1
+ %p11 = getelementptr inbounds i8, ptr %p, i32 11
+ %v11 = load i8, ptr %p11, align 1
+ %p12 = getelementptr inbounds i8, ptr %p, i32 12
+ %v12 = load i8, ptr %p12, align 1
+ %p13 = getelementptr inbounds i8, ptr %p, i32 13
+ %v13 = load i8, ptr %p13, align 1
+ %p14 = getelementptr inbounds i8, ptr %p, i32 14
+ %v14 = load i8, ptr %p14, align 1
+ %p15 = getelementptr inbounds i8, ptr %p, i32 15
+ %v15 = load i8, ptr %p15, align 1
+ %p16 = getelementptr inbounds i8, ptr %p, i32 16
+ %v16 = load i8, ptr %p16, align 1
+ %p17 = getelementptr inbounds i8, ptr %p, i32 17
+ %v17 = load i8, ptr %p17, align 1
+
+ %o01 = or i8 %v0, %v1
+ %o23 = or i8 %v2, %v3
+ %o45 = or i8 %v4, %v5
+ %o67 = or i8 %v6, %v7
+ %o89 = or i8 %v8, %v9
+ %o1011 = or i8 %v10, %v11
+ %o1213 = or i8 %v12, %v13
+ %o1415 = or i8 %v14, %v15
+ %o1617 = or i8 %v16, %v17
+
+ %o0123 = or i8 %o01, %o23
+ %o4567 = or i8 %o45, %o67
+ %o891011 = or i8 %o89, %o1011
+ %o12131415 = or i8 %o1213, %o1415
+
+ %o07 = or i8 %o0123, %o4567
+ %o815 = or i8 %o891011, %o12131415
+ %o015 = or i8 %o07, %o815
+ %o017 = or i8 %o015, %o1617
+
+ %cmp = icmp ne i8 %o017, 0
+ ret i1 %cmp
+}
>From 2b4b4ceabb6947a64154093cdc1d68f45786fd11 Mon Sep 17 00:00:00 2001
From: hanbeom <kese111 at gmail.com>
Date: Fri, 31 Jul 2026 19:14:41 +0900
Subject: [PATCH 03/11] Account for removed scalar compare in zero-test costs
Subtract the scalar comparison eliminated when a complete
zero-tested reduction is lowered to a vector comparison
and i1 reduction.
---
llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp | 6 ++++--
.../SLPVectorizer/X86/reduction-zero-test-partial-cost.ll | 2 +-
2 files changed, 5 insertions(+), 3 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 5243ed4e70dd2..7dbd3a2ec95a2 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -31649,7 +31649,8 @@ class HorizontalReduction {
//
// Equality requires every lane to be zero, so it uses an AND reduction;
// inequality requires any lane to be nonzero, so it uses OR. Add the
- // vector comparison and boolean reduction costs.
+ // vector comparison and boolean reduction costs, then remove the scalar
+ // comparison cost eliminated by this replacement.
assert(DoesRequireReductionOp && !isa<VectorType>(ScalarTy) &&
(RdxKind == RecurKind::Or || RdxKind == RecurKind::UMax) &&
"Unexpected zero comparison reduction");
@@ -31662,7 +31663,8 @@ class HorizontalReduction {
CmpReductionCost = TTI->getCmpSelInstrCost(Instruction::ICmp, VectorTy,
CmpTy, Pred, CostKind) +
TTI->getArithmeticReductionCost(ReductionOpcode, CmpTy,
- {}, CostKind);
+ {}, CostKind) -
+ TTI->getInstructionCost(ScalarCmp, CostKind);
}
switch (RdxKind) {
case RecurKind::Add:
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/reduction-zero-test-partial-cost.ll b/llvm/test/Transforms/SLPVectorizer/X86/reduction-zero-test-partial-cost.ll
index 73fe770795444..139cc786bde67 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/reduction-zero-test-partial-cost.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/reduction-zero-test-partial-cost.ll
@@ -3,7 +3,7 @@
; RUN: cat %t | FileCheck -check-prefix=COST %s
; COST-LABEL: Function: or_reduction_nonzero_scalar
-; COST: Cost: '-20'
+; COST: Cost: '-21'
define i1 @or_reduction_nonzero_scalar(ptr %p) {
; CHECK-LABEL: define i1 @or_reduction_nonzero_scalar(
; CHECK-SAME: ptr [[P:%.*]]) #[[ATTR0:[0-9]+]] {
>From 0a81d7589bb2ef4e824c89d25ecbb45345412719 Mon Sep 17 00:00:00 2001
From: hanbeom <kese111 at gmail.com>
Date: Sat, 1 Aug 2026 10:46:52 +0900
Subject: [PATCH 04/11] [TTI][SLP] Use VectorInstrContext for zero-tested
reductions
Encode the zero-comparison reduction context in TTI::VectorInstrContext and use it for SLP reduction costing and lowering instead of an SLP-local enum.
---
.../llvm/Analysis/TargetTransformInfo.h | 1 +
.../Transforms/Vectorize/SLPVectorizer.cpp | 37 +++++++++----------
2 files changed, 19 insertions(+), 19 deletions(-)
diff --git a/llvm/include/llvm/Analysis/TargetTransformInfo.h b/llvm/include/llvm/Analysis/TargetTransformInfo.h
index 107ae4dba5075..2cf60d3bedb59 100644
--- a/llvm/include/llvm/Analysis/TargetTransformInfo.h
+++ b/llvm/include/llvm/Analysis/TargetTransformInfo.h
@@ -191,6 +191,7 @@ enum class VectorInstrContext : uint8_t {
Load, ///< The value being inserted comes from a load (InsertElement only).
Store, ///< The extracted value is stored (ExtractElement only).
BinaryOp, ///< One of the operands is a binary op.
+ CmpZero, ///< The reduction result is compared against zero.
};
class IntrinsicCostAttributes {
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 7dbd3a2ec95a2..de62f597a8c2f 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -29649,10 +29649,6 @@ namespace {
class HorizontalReduction {
using ReductionOpsType = SmallVector<Value *, 16>;
using ReductionOpsListType = SmallVector<ReductionOpsType, 2>;
- enum class ReductionContext {
- None,
- CmpZero,
- };
ReductionOpsListType ReductionOps;
/// List of possibly reduced values.
@@ -29687,22 +29683,22 @@ class HorizontalReduction {
/// Return CmpZero for a scalar OR/UMax reduction whose only use is an eq/ne
/// comparison against zero.
- ReductionContext getReductionContext() const {
+ TTI::VectorInstrContext getReductionContext() const {
auto *Root = dyn_cast<Instruction>(ReductionRoot);
if (!Root || !Root->getType()->isIntegerTy() || !Root->hasOneUse() ||
(RdxKind != RecurKind::Or && RdxKind != RecurKind::UMax))
- return ReductionContext::None;
+ return TTI::VectorInstrContext::None;
CmpPredicate Pred;
if (!match(*Root->user_begin(),
m_c_ICmp(Pred, m_Specific(Root), m_ZeroInt())) ||
!ICmpInst::isEquality(Pred))
- return ReductionContext::None;
- return ReductionContext::CmpZero;
+ return TTI::VectorInstrContext::None;
+ return TTI::VectorInstrContext::CmpZero;
}
- static bool isZeroCmpContext(ReductionContext Context) {
- return Context == ReductionContext::CmpZero;
+ static bool isZeroCmpContext(TTI::VectorInstrContext Context) {
+ return Context == TTI::VectorInstrContext::CmpZero;
}
/// Checks if instruction is associative and can be vectorized.
@@ -30509,7 +30505,7 @@ class HorizontalReduction {
if (RK == ReductionOrdering::Ordered)
IgnoreList.clear();
bool IsCmpSelMinMax = isCmpSelMinMax(cast<Instruction>(ReductionRoot));
- ReductionContext RdxContext = getReductionContext();
+ TTI::VectorInstrContext RdxContext = getReductionContext();
// Need to track reduced vals, they may be changed during vectorization of
// subvectors.
@@ -30529,7 +30525,8 @@ class HorizontalReduction {
// nodes and thus requiring extract if fully vectorized in other trees.
SmallPtrSet<Value *, 4> RequiredExtract;
WeakTrackingVH VectorizedTree = nullptr;
- ReductionContext VectorizedReductionContext = ReductionContext::None;
+ TTI::VectorInstrContext VectorizedReductionContext =
+ TTI::VectorInstrContext::None;
bool CheckForReusedReductionOps = false;
// Try to vectorize elements based on their type.
SmallVector<InstructionsState> States;
@@ -30910,7 +30907,8 @@ class HorizontalReduction {
// Use the zero-test cost only for a complete, standalone scalar
// reduction that can become one vector comparison and i1 reduction.
- ReductionContext CostContext = ReductionContext::None;
+ TTI::VectorInstrContext CostContext =
+ TTI::VectorInstrContext::None;
if (isZeroCmpContext(RdxContext) &&
this->ReducedVals.size() == 1 &&
VL.size() == this->ReducedVals.front().size() &&
@@ -31365,8 +31363,8 @@ class HorizontalReduction {
InstructionCost ReductionCost = getReductionCost(
TTI, VL, EmptySameValuesCounter,
- /*IsCmpSelMinMax=*/false, ReductionContext::None, RdxFMF, V, DT, DL,
- TLI);
+ /*IsCmpSelMinMax=*/false, TTI::VectorInstrContext::None, RdxFMF, V,
+ DT, DL, TLI);
InstructionCost Cost =
V.getTreeCost(TreeCost, VL, ReductionCost, RdxRootInst);
LLVM_DEBUG(dbgs() << "SLP: Found cost = " << Cost
@@ -31549,9 +31547,9 @@ class HorizontalReduction {
InstructionCost getReductionCost(
TargetTransformInfo *TTI, ArrayRef<Value *> ReducedVals,
const SmallMapVector<Value *, unsigned, 16> SameValuesCounter,
- bool IsCmpSelMinMax, ReductionContext Context, FastMathFlags FMF,
- const BoUpSLP &R, DominatorTree &DT, const DataLayout &DL,
- const TargetLibraryInfo &TLI) {
+ bool IsCmpSelMinMax, TargetTransformInfo::VectorInstrContext Context,
+ FastMathFlags FMF, const BoUpSLP &R, DominatorTree &DT,
+ const DataLayout &DL, const TargetLibraryInfo &TLI) {
TTI::TargetCostKind CostKind = TTI::TCK_RecipThroughput;
Type *ScalarTy = ReducedVals.front()->getType();
unsigned ReduxWidth = ReducedVals.size();
@@ -31828,7 +31826,8 @@ class HorizontalReduction {
/// sub-registers, combines them with the given reduction operation as a
/// vector operation and then performs single (small enough) reduction.
Value *emitReduction(IRBuilderBase &Builder, const TargetTransformInfo &TTI,
- Type *DestTy, ReductionContext Context) {
+ Type *DestTy,
+ TargetTransformInfo::VectorInstrContext Context) {
if (isZeroCmpContext(Context)) {
assert(VectorValuesAndScales.size() == 1 &&
!std::get<3>(VectorValuesAndScales.front()) &&
>From 610834df0bb504926a894a2aa4c9ea54dca65ccb Mon Sep 17 00:00:00 2001
From: hanbeom <kese111 at gmail.com>
Date: Sat, 1 Aug 2026 11:33:17 +0900
Subject: [PATCH 05/11] formatting
---
llvm/include/llvm/Analysis/TargetTransformInfo.h | 2 +-
llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp | 6 ++----
2 files changed, 3 insertions(+), 5 deletions(-)
diff --git a/llvm/include/llvm/Analysis/TargetTransformInfo.h b/llvm/include/llvm/Analysis/TargetTransformInfo.h
index 2cf60d3bedb59..10cc047771967 100644
--- a/llvm/include/llvm/Analysis/TargetTransformInfo.h
+++ b/llvm/include/llvm/Analysis/TargetTransformInfo.h
@@ -191,7 +191,7 @@ enum class VectorInstrContext : uint8_t {
Load, ///< The value being inserted comes from a load (InsertElement only).
Store, ///< The extracted value is stored (ExtractElement only).
BinaryOp, ///< One of the operands is a binary op.
- CmpZero, ///< The reduction result is compared against zero.
+ CmpZero, ///< The reduction result is compared against zero.
};
class IntrinsicCostAttributes {
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index de62f597a8c2f..1ff2d817cacd3 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -30907,10 +30907,8 @@ class HorizontalReduction {
// Use the zero-test cost only for a complete, standalone scalar
// reduction that can become one vector comparison and i1 reduction.
- TTI::VectorInstrContext CostContext =
- TTI::VectorInstrContext::None;
- if (isZeroCmpContext(RdxContext) &&
- this->ReducedVals.size() == 1 &&
+ TTI::VectorInstrContext CostContext = TTI::VectorInstrContext::None;
+ if (isZeroCmpContext(RdxContext) && this->ReducedVals.size() == 1 &&
VL.size() == this->ReducedVals.front().size() &&
VectorValuesAndScales.empty() && !VectorizedTree &&
!isa<VectorType>(VL.front()->getType()) && !allConstant(VL) &&
>From 7118f719ddcd3e808f96af1f8e4eac2e790d132c Mon Sep 17 00:00:00 2001
From: hanbeom <kese111 at gmail.com>
Date: Mon, 10 Aug 2026 13:47:50 +0900
Subject: [PATCH 06/11] restore ReductionContext
---
llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp | 4 ++++
1 file changed, 4 insertions(+)
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 1ff2d817cacd3..ab2fafeb18bb6 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -29649,6 +29649,10 @@ namespace {
class HorizontalReduction {
using ReductionOpsType = SmallVector<Value *, 16>;
using ReductionOpsListType = SmallVector<ReductionOpsType, 2>;
+ enum class ReductionContext {
+ None,
+ CmpZero,
+ };
ReductionOpsListType ReductionOps;
/// List of possibly reduced values.
>From 1bcf72d9244128151e1226cef51ca4246932911e Mon Sep 17 00:00:00 2001
From: hanbeom <kese111 at gmail.com>
Date: Mon, 10 Aug 2026 14:49:36 +0900
Subject: [PATCH 07/11] change order of CostContext for getReductionCost
---
llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp | 16 ++++++++--------
1 file changed, 8 insertions(+), 8 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index ab2fafeb18bb6..2a187a7e48532 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -30927,7 +30927,7 @@ class HorizontalReduction {
else
ReductionCost =
getReductionCost(TTI, VL, SameValuesCounter, IsCmpSelMinMax,
- CostContext, RdxFMF, V, DT, DL, TLI);
+ RdxFMF, V, DT, DL, TLI, CostContext);
// If the root is a select (min/max idiom), the insert point is the
// compare condition of that select.
Instruction *RdxRootInst = cast<Instruction>(ReductionRoot);
@@ -31363,10 +31363,10 @@ class HorizontalReduction {
V.calculateTreeCostAndTrimNonProfitable(VL, RdxRootInst);
V.buildExternalUses(LocalExternallyUsedValues);
- InstructionCost ReductionCost = getReductionCost(
- TTI, VL, EmptySameValuesCounter,
- /*IsCmpSelMinMax=*/false, TTI::VectorInstrContext::None, RdxFMF, V,
- DT, DL, TLI);
+ InstructionCost ReductionCost =
+ getReductionCost(TTI, VL, EmptySameValuesCounter,
+ /*IsCmpSelMinMax=*/false, RdxFMF, V, DT, DL, TLI,
+ TTI::VectorInstrContext::None);
InstructionCost Cost =
V.getTreeCost(TreeCost, VL, ReductionCost, RdxRootInst);
LLVM_DEBUG(dbgs() << "SLP: Found cost = " << Cost
@@ -31549,9 +31549,9 @@ class HorizontalReduction {
InstructionCost getReductionCost(
TargetTransformInfo *TTI, ArrayRef<Value *> ReducedVals,
const SmallMapVector<Value *, unsigned, 16> SameValuesCounter,
- bool IsCmpSelMinMax, TargetTransformInfo::VectorInstrContext Context,
- FastMathFlags FMF, const BoUpSLP &R, DominatorTree &DT,
- const DataLayout &DL, const TargetLibraryInfo &TLI) {
+ bool IsCmpSelMinMax, FastMathFlags FMF, const BoUpSLP &R,
+ DominatorTree &DT, const DataLayout &DL, const TargetLibraryInfo &TLI,
+ TargetTransformInfo::VectorInstrContext Context) {
TTI::TargetCostKind CostKind = TTI::TCK_RecipThroughput;
Type *ScalarTy = ReducedVals.front()->getType();
unsigned ReduxWidth = ReducedVals.size();
>From af2e4da578ee7dc367d4523b5d33ac4f1d9e1431 Mon Sep 17 00:00:00 2001
From: hanbeom <kese111 at gmail.com>
Date: Mon, 10 Aug 2026 14:34:54 +0900
Subject: [PATCH 08/11] modify condition check for reduction context
---
.../lib/Transforms/Vectorize/SLPVectorizer.cpp | 18 +++++++++++++-----
1 file changed, 13 insertions(+), 5 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 2a187a7e48532..787553e3efb0a 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -29688,17 +29688,25 @@ class HorizontalReduction {
/// Return CmpZero for a scalar OR/UMax reduction whose only use is an eq/ne
/// comparison against zero.
TTI::VectorInstrContext getReductionContext() const {
+ TTI::VectorInstrContext Context = TTI::VectorInstrContext::None;
+
auto *Root = dyn_cast<Instruction>(ReductionRoot);
- if (!Root || !Root->getType()->isIntegerTy() || !Root->hasOneUse() ||
- (RdxKind != RecurKind::Or && RdxKind != RecurKind::UMax))
- return TTI::VectorInstrContext::None;
+ if (!Root || !Root->getType()->isIntegerTy() || !Root->hasOneUse())
+ return Context;
CmpPredicate Pred;
if (!match(*Root->user_begin(),
m_c_ICmp(Pred, m_Specific(Root), m_ZeroInt())) ||
!ICmpInst::isEquality(Pred))
- return TTI::VectorInstrContext::None;
- return TTI::VectorInstrContext::CmpZero;
+ return Context;
+
+ switch (RdxKind) {
+ case RecurKind::Or:
+ case RecurKind::UMax:
+ return TTI::VectorInstrContext::CmpZero;
+ default:
+ return Context;
+ }
}
static bool isZeroCmpContext(TTI::VectorInstrContext Context) {
>From 10ce3e7af3251452be1a29ab2bf1efb59299224c Mon Sep 17 00:00:00 2001
From: hanbeom <kese111 at gmail.com>
Date: Mon, 10 Aug 2026 16:43:02 +0900
Subject: [PATCH 09/11] Keep zero-test context out of reduction emission
---
.../Transforms/Vectorize/SLPVectorizer.cpp | 41 ++-----------------
1 file changed, 4 insertions(+), 37 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 787553e3efb0a..6f4cd655845ac 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -30537,8 +30537,6 @@ class HorizontalReduction {
// nodes and thus requiring extract if fully vectorized in other trees.
SmallPtrSet<Value *, 4> RequiredExtract;
WeakTrackingVH VectorizedTree = nullptr;
- TTI::VectorInstrContext VectorizedReductionContext =
- TTI::VectorInstrContext::None;
bool CheckForReusedReductionOps = false;
// Try to vectorize elements based on their type.
SmallVector<InstructionsState> States;
@@ -31039,8 +31037,6 @@ class HorizontalReduction {
Type *ScalarTy = VL.front()->getType();
Type *VecTy = VectorizedRoot->getType();
Type *RedScalarTy = VecTy->getScalarType();
- if (isZeroCmpContext(CostContext))
- VectorizedReductionContext = CostContext;
VectorValuesAndScales.emplace_back(
VectorizedRoot,
OptReusedScalars && SameScaleFactor
@@ -31086,8 +31082,8 @@ class HorizontalReduction {
if (!VectorValuesAndScales.empty())
VectorizedTree = GetNewVectorizedTree(
- VectorizedTree, emitReduction(Builder, *TTI, ReductionRoot->getType(),
- VectorizedReductionContext));
+ VectorizedTree,
+ emitReduction(Builder, *TTI, ReductionRoot->getType()));
if (!VectorizedTree) {
if (!CheckForReusedReductionOps) {
@@ -31205,18 +31201,7 @@ class HorizontalReduction {
}
VectorizedTree = ExtraReductions.front().second;
- if (isZeroCmpContext(VectorizedReductionContext)) {
- auto *Cmp =
- cast<ICmpInst>(*cast<Instruction>(ReductionRoot)->user_begin());
- VectorizedTree->takeName(Cmp);
- Cmp->replaceAllUsesWith(VectorizedTree);
- salvageDebugInfo(*Cmp);
- Cmp->dropAllReferences();
- Cmp->removeFromParent();
- V.eraseInstruction(Cmp);
- } else {
- ReductionRoot->replaceAllUsesWith(VectorizedTree);
- }
+ ReductionRoot->replaceAllUsesWith(VectorizedTree);
// The original scalar reduction is expected to have no remaining
// uses outside the reduction tree itself. Assert that we got this
@@ -31836,25 +31821,7 @@ class HorizontalReduction {
/// sub-registers, combines them with the given reduction operation as a
/// vector operation and then performs single (small enough) reduction.
Value *emitReduction(IRBuilderBase &Builder, const TargetTransformInfo &TTI,
- Type *DestTy,
- TargetTransformInfo::VectorInstrContext Context) {
- if (isZeroCmpContext(Context)) {
- assert(VectorValuesAndScales.size() == 1 &&
- !std::get<3>(VectorValuesAndScales.front()) &&
- "Expected one complete vector reduction");
- Value *Vec = std::get<0>(VectorValuesAndScales.front());
- auto *VecTy = cast<VectorType>(Vec->getType());
- auto *ScalarCmp =
- cast<ICmpInst>(*cast<Instruction>(ReductionRoot)->user_begin());
- CmpPredicate Pred = ScalarCmp->getPredicate();
- Builder.SetCurrentDebugLocation(ScalarCmp->getDebugLoc());
- Value *Cmp = Builder.CreateICmp(Pred, Vec, Constant::getNullValue(VecTy));
- RecurKind BoolRdxKind =
- Pred == ICmpInst::ICMP_EQ ? RecurKind::And : RecurKind::Or;
- NumVectorInstructions += 2;
- return createSimpleReduction(Builder, Cmp, BoolRdxKind);
- }
-
+ Type *DestTy) {
Value *ReducedSubTree = nullptr;
// Creates reduction and combines with the previous reduction.
auto CreateSingleOp = [&](Value *Vec, unsigned Scale, bool IsSigned,
>From 4d29bfe8a82fab95e75761129e94c42b5d356107 Mon Sep 17 00:00:00 2001
From: hanbeom <kese111 at gmail.com>
Date: Mon, 10 Aug 2026 16:44:59 +0900
Subject: [PATCH 10/11] Restrict zero-test cost to complete reductions
---
.../Transforms/Vectorize/SLPVectorizer.cpp | 31 ++++++-------------
1 file changed, 10 insertions(+), 21 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 6f4cd655845ac..c01a7e08187df 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -29649,11 +29649,6 @@ namespace {
class HorizontalReduction {
using ReductionOpsType = SmallVector<Value *, 16>;
using ReductionOpsListType = SmallVector<ReductionOpsType, 2>;
- enum class ReductionContext {
- None,
- CmpZero,
- };
-
ReductionOpsListType ReductionOps;
/// List of possibly reduced values.
SmallVector<SmallVector<Value *>> ReducedVals;
@@ -30915,16 +30910,6 @@ class HorizontalReduction {
LocalExternallyUsedValues.insert(RdxVal);
V.buildExternalUses(LocalExternallyUsedValues);
- // Use the zero-test cost only for a complete, standalone scalar
- // reduction that can become one vector comparison and i1 reduction.
- TTI::VectorInstrContext CostContext = TTI::VectorInstrContext::None;
- if (isZeroCmpContext(RdxContext) && this->ReducedVals.size() == 1 &&
- VL.size() == this->ReducedVals.front().size() &&
- VectorValuesAndScales.empty() && !VectorizedTree &&
- !isa<VectorType>(VL.front()->getType()) && !allConstant(VL) &&
- !V.isReducedBitcastRoot() && !V.isReducedCmpBitcastRoot())
- CostContext = RdxContext;
-
// Estimate cost.
InstructionCost ReductionCost;
if (RK == ReductionOrdering::Ordered || V.isReducedBitcastRoot() ||
@@ -30933,7 +30918,7 @@ class HorizontalReduction {
else
ReductionCost =
getReductionCost(TTI, VL, SameValuesCounter, IsCmpSelMinMax,
- RdxFMF, V, DT, DL, TLI, CostContext);
+ RdxFMF, V, DT, DL, TLI, RdxContext);
// If the root is a select (min/max idiom), the insert point is the
// compare condition of that select.
Instruction *RdxRootInst = cast<Instruction>(ReductionRoot);
@@ -31627,8 +31612,13 @@ class HorizontalReduction {
// 2. The storage does not have any vector with full vector use (first
// vector with full register use).
bool DoesRequireReductionOp = !AllConsts && VectorValuesAndScales.empty();
+ bool IsCompleteReduction =
+ this->ReducedVals.size() == 1 &&
+ ReducedVals.size() == this->ReducedVals.front().size();
+ bool UseCmpZeroCost = isZeroCmpContext(Context) && IsCompleteReduction &&
+ DoesRequireReductionOp && !isa<VectorType>(ScalarTy);
InstructionCost CmpReductionCost = InstructionCost::getInvalid();
- if (isZeroCmpContext(Context)) {
+ if (UseCmpZeroCost) {
// For a complete OR/UMax reduction used only by an eq/ne zero test,
// account for moving the comparison before the reduction:
//
@@ -31644,8 +31634,7 @@ class HorizontalReduction {
// inequality requires any lane to be nonzero, so it uses OR. Add the
// vector comparison and boolean reduction costs, then remove the scalar
// comparison cost eliminated by this replacement.
- assert(DoesRequireReductionOp && !isa<VectorType>(ScalarTy) &&
- (RdxKind == RecurKind::Or || RdxKind == RecurKind::UMax) &&
+ assert((RdxKind == RecurKind::Or || RdxKind == RecurKind::UMax) &&
"Unexpected zero comparison reduction");
auto *ScalarCmp =
cast<ICmpInst>(*cast<Instruction>(ReductionRoot)->user_begin());
@@ -31670,7 +31659,7 @@ class HorizontalReduction {
unsigned RdxOpcode = RecurrenceDescriptor::getOpcode(RdxKind);
if (!AllConsts) {
if (DoesRequireReductionOp) {
- if (isZeroCmpContext(Context)) {
+ if (UseCmpZeroCost) {
VectorCost = CmpReductionCost;
} else if (auto *VecTy = dyn_cast<FixedVectorType>(ScalarTy)) {
assert(SLPReVec && "FixedVectorType is not expected.");
@@ -31777,7 +31766,7 @@ class HorizontalReduction {
Intrinsic::ID Id = getMinMaxReductionIntrinsicOp(RdxKind);
if (!AllConsts) {
if (DoesRequireReductionOp) {
- if (isZeroCmpContext(Context))
+ if (UseCmpZeroCost)
VectorCost = CmpReductionCost;
else
VectorCost =
>From e37ffb6e869bf6d2213ab9272e8feaed6e74d79d Mon Sep 17 00:00:00 2001
From: hanbeom <kese111 at gmail.com>
Date: Tue, 11 Aug 2026 01:14:33 +0900
Subject: [PATCH 11/11] update tests
---
.../WebAssembly/or-reduction-zero-test.ll | 8 ++++----
.../X86/reduction-zero-test-ctpop.ll | 20 +++++++++----------
.../X86/reduction-zero-test-partial-cost.ll | 6 +++---
3 files changed, 17 insertions(+), 17 deletions(-)
diff --git a/llvm/test/Transforms/SLPVectorizer/WebAssembly/or-reduction-zero-test.ll b/llvm/test/Transforms/SLPVectorizer/WebAssembly/or-reduction-zero-test.ll
index 94d4e090f2b8a..d5667fb9f6a7a 100644
--- a/llvm/test/Transforms/SLPVectorizer/WebAssembly/or-reduction-zero-test.ll
+++ b/llvm/test/Transforms/SLPVectorizer/WebAssembly/or-reduction-zero-test.ll
@@ -6,8 +6,8 @@ define i1 @or_reduction_nonzero(ptr %p) {
; CHECK-SAME: ptr [[P:%.*]]) #[[ATTR0:[0-9]+]] {
; CHECK-NEXT: [[ENTRY:.*:]]
; CHECK-NEXT: [[TMP0:%.*]] = load <16 x i8>, ptr [[P]], align 1
-; CHECK-NEXT: [[TMP1:%.*]] = icmp ne <16 x i8> [[TMP0]], zeroinitializer
-; CHECK-NEXT: [[CMP:%.*]] = call i1 @llvm.vector.reduce.or.v16i1(<16 x i1> [[TMP1]])
+; CHECK-NEXT: [[TMP1:%.*]] = call i8 @llvm.vector.reduce.or.v16i8(<16 x i8> [[TMP0]])
+; CHECK-NEXT: [[CMP:%.*]] = icmp ne i8 [[TMP1]], 0
; CHECK-NEXT: ret i1 [[CMP]]
;
entry:
@@ -66,8 +66,8 @@ define i1 @or_reduction_zero(ptr %p) {
; CHECK-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
; CHECK-NEXT: [[ENTRY:.*:]]
; CHECK-NEXT: [[TMP0:%.*]] = load <16 x i8>, ptr [[P]], align 1
-; CHECK-NEXT: [[TMP1:%.*]] = icmp eq <16 x i8> [[TMP0]], zeroinitializer
-; CHECK-NEXT: [[CMP:%.*]] = call i1 @llvm.vector.reduce.and.v16i1(<16 x i1> [[TMP1]])
+; CHECK-NEXT: [[TMP1:%.*]] = call i8 @llvm.vector.reduce.or.v16i8(<16 x i8> [[TMP0]])
+; CHECK-NEXT: [[CMP:%.*]] = icmp eq i8 [[TMP1]], 0
; CHECK-NEXT: ret i1 [[CMP]]
;
entry:
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/reduction-zero-test-ctpop.ll b/llvm/test/Transforms/SLPVectorizer/X86/reduction-zero-test-ctpop.ll
index 4f74d64460e90..9c7e19c25b3e9 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/reduction-zero-test-ctpop.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/reduction-zero-test-ctpop.ll
@@ -5,8 +5,8 @@ define i32 @or_nonzero(ptr %p) {
; CHECK-LABEL: define i32 @or_nonzero(
; CHECK-SAME: ptr [[P:%.*]]) #[[ATTR0:[0-9]+]] {
; CHECK-NEXT: [[INPUT:%.*]] = load <8 x i8>, ptr [[P]], align 1
-; CHECK-NEXT: [[TMP1:%.*]] = icmp ne <8 x i8> [[INPUT]], zeroinitializer
-; CHECK-NEXT: [[CMP:%.*]] = call i1 @llvm.vector.reduce.or.v8i1(<8 x i1> [[TMP1]])
+; CHECK-NEXT: [[TMP1:%.*]] = call i8 @llvm.vector.reduce.or.v8i8(<8 x i8> [[INPUT]])
+; CHECK-NEXT: [[CMP:%.*]] = icmp ne i8 [[TMP1]], 0
; CHECK-NEXT: [[RESULT:%.*]] = zext i1 [[CMP]] to i32
; CHECK-NEXT: ret i32 [[RESULT]]
;
@@ -35,8 +35,8 @@ define i32 @or_nonzero_commuted(ptr %p) {
; CHECK-LABEL: define i32 @or_nonzero_commuted(
; CHECK-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
; CHECK-NEXT: [[INPUT:%.*]] = load <8 x i8>, ptr [[P]], align 1
-; CHECK-NEXT: [[TMP1:%.*]] = icmp ne <8 x i8> [[INPUT]], zeroinitializer
-; CHECK-NEXT: [[CMP:%.*]] = call i1 @llvm.vector.reduce.or.v8i1(<8 x i1> [[TMP1]])
+; CHECK-NEXT: [[TMP1:%.*]] = call i8 @llvm.vector.reduce.or.v8i8(<8 x i8> [[INPUT]])
+; CHECK-NEXT: [[CMP:%.*]] = icmp ne i8 0, [[TMP1]]
; CHECK-NEXT: [[RESULT:%.*]] = zext i1 [[CMP]] to i32
; CHECK-NEXT: ret i32 [[RESULT]]
;
@@ -65,8 +65,8 @@ define i32 @or_zero(ptr %p) {
; CHECK-LABEL: define i32 @or_zero(
; CHECK-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
; CHECK-NEXT: [[INPUT:%.*]] = load <8 x i8>, ptr [[P]], align 1
-; CHECK-NEXT: [[TMP1:%.*]] = icmp eq <8 x i8> [[INPUT]], zeroinitializer
-; CHECK-NEXT: [[CMP:%.*]] = call i1 @llvm.vector.reduce.and.v8i1(<8 x i1> [[TMP1]])
+; CHECK-NEXT: [[TMP1:%.*]] = call i8 @llvm.vector.reduce.or.v8i8(<8 x i8> [[INPUT]])
+; CHECK-NEXT: [[CMP:%.*]] = icmp eq i8 [[TMP1]], 0
; CHECK-NEXT: [[RESULT:%.*]] = zext i1 [[CMP]] to i32
; CHECK-NEXT: ret i32 [[RESULT]]
;
@@ -95,8 +95,8 @@ define i32 @umax_nonzero(ptr %p) {
; CHECK-LABEL: define i32 @umax_nonzero(
; CHECK-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
; CHECK-NEXT: [[INPUT:%.*]] = load <8 x i8>, ptr [[P]], align 1
-; CHECK-NEXT: [[TMP1:%.*]] = icmp ne <8 x i8> [[INPUT]], zeroinitializer
-; CHECK-NEXT: [[CMP:%.*]] = call i1 @llvm.vector.reduce.or.v8i1(<8 x i1> [[TMP1]])
+; CHECK-NEXT: [[TMP1:%.*]] = call i8 @llvm.vector.reduce.umax.v8i8(<8 x i8> [[INPUT]])
+; CHECK-NEXT: [[CMP:%.*]] = icmp ne i8 [[TMP1]], 0
; CHECK-NEXT: [[RESULT:%.*]] = zext i1 [[CMP]] to i32
; CHECK-NEXT: ret i32 [[RESULT]]
;
@@ -132,8 +132,8 @@ define i32 @umax_zero_commuted(ptr %p) {
; CHECK-LABEL: define i32 @umax_zero_commuted(
; CHECK-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
; CHECK-NEXT: [[INPUT:%.*]] = load <8 x i8>, ptr [[P]], align 1
-; CHECK-NEXT: [[TMP1:%.*]] = icmp eq <8 x i8> [[INPUT]], zeroinitializer
-; CHECK-NEXT: [[CMP:%.*]] = call i1 @llvm.vector.reduce.and.v8i1(<8 x i1> [[TMP1]])
+; CHECK-NEXT: [[TMP1:%.*]] = call i8 @llvm.vector.reduce.umax.v8i8(<8 x i8> [[INPUT]])
+; CHECK-NEXT: [[CMP:%.*]] = icmp eq i8 0, [[TMP1]]
; CHECK-NEXT: [[RESULT:%.*]] = zext i1 [[CMP]] to i32
; CHECK-NEXT: ret i32 [[RESULT]]
;
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/reduction-zero-test-partial-cost.ll b/llvm/test/Transforms/SLPVectorizer/X86/reduction-zero-test-partial-cost.ll
index 139cc786bde67..69061b520c7e4 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/reduction-zero-test-partial-cost.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/reduction-zero-test-partial-cost.ll
@@ -8,8 +8,8 @@ define i1 @or_reduction_nonzero_scalar(ptr %p) {
; CHECK-LABEL: define i1 @or_reduction_nonzero_scalar(
; CHECK-SAME: ptr [[P:%.*]]) #[[ATTR0:[0-9]+]] {
; CHECK-NEXT: [[TMP1:%.*]] = load <16 x i8>, ptr [[P]], align 1
-; CHECK-NEXT: [[TMP2:%.*]] = icmp ne <16 x i8> [[TMP1]], zeroinitializer
-; CHECK-NEXT: [[CMP:%.*]] = call i1 @llvm.vector.reduce.or.v16i1(<16 x i1> [[TMP2]])
+; CHECK-NEXT: [[TMP2:%.*]] = call i8 @llvm.vector.reduce.or.v16i8(<16 x i8> [[TMP1]])
+; CHECK-NEXT: [[CMP:%.*]] = icmp ne i8 [[TMP2]], 0
; CHECK-NEXT: ret i1 [[CMP]]
;
%v0 = load i8, ptr %p, align 1
@@ -67,7 +67,7 @@ define i1 @or_reduction_nonzero_scalar(ptr %p) {
}
; COST-LABEL: Function: or_reduction_nonzero_scalar_remainder
-; COST: Cost: '-21'
+; COST: Cost: '-28'
define i1 @or_reduction_nonzero_scalar_remainder(ptr %p) {
; CHECK-LABEL: define i1 @or_reduction_nonzero_scalar_remainder(
; CHECK-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
More information about the llvm-commits
mailing list