[llvm] cb02999 - [SLP]Allow min-VF vectorization of seed-level reduction groups
via llvm-commits
llvm-commits at lists.llvm.org
Fri Sep 11 10:56:50 PDT 2026
Author: Alexey Bataev
Date: 2026-09-11T13:56:45-04:00
New Revision: cb029994abbda52adf1986df5c6198c3e1fff4da
URL: https://github.com/llvm/llvm-project/commit/cb029994abbda52adf1986df5c6198c3e1fff4da
DIFF: https://github.com/llvm/llvm-project/commit/cb029994abbda52adf1986df5c6198c3e1fff4da.diff
LOG: [SLP]Allow min-VF vectorization of seed-level reduction groups
Small reduced-value groups of single-use seed-level reductions may
vectorize at VF=2, if the values are used by the reduction operations
only and no scalar leftovers remain. For fadd this replaces the
ordered reduction fallback; for the other kinds trunk leaves such
groups scalar.
Reviewers: RKSimon, bababuck, MrSidims
Pull Request: https://github.com/llvm/llvm-project/pull/222757
Added:
Modified:
llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
llvm/test/Transforms/PhaseOrdering/X86/fma-reassociate-pairs.ll
llvm/test/Transforms/SLPVectorizer/X86/fma-operand-index.ll
llvm/test/Transforms/SLPVectorizer/X86/reassociated-fma-reduction.ll
llvm/test/Transforms/SLPVectorizer/X86/revectorized_rdx_crash.ll
Removed:
################################################################################
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index bcf79dea600f0..488bfb36b6ccc 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -29844,6 +29844,12 @@ class HorizontalReduction {
I->hasAllowReassoc() && I->hasNoSignedZeros();
}
+ /// A group of loads or non-instructions rarely amortizes on its own the
+ /// cost of the reduction operation, charged to the first vectorized group.
+ static bool isCheapReductionGroup(ArrayRef<Value *> P) {
+ return !isa<Instruction>(P.front()) || isa<LoadInst>(P.front());
+ }
+
/// Try to find a reduction tree. If \p FlattenNegations is false, the
/// reassociable fsub/fneg links of an fadd chain are not flattened (they are
/// leaves of the reduction).
@@ -30250,11 +30256,8 @@ class HorizontalReduction {
return P1.size() > P2.size();
if (NegatedReducedVals.empty())
return false;
- auto IsCheapGroup = [](ArrayRef<Value *> P) {
- return !isa<Instruction>(P.front()) || isa<LoadInst>(P.front());
- };
- bool Cheap1 = IsCheapGroup(P1);
- bool Cheap2 = IsCheapGroup(P2);
+ bool Cheap1 = isCheapReductionGroup(P1);
+ bool Cheap2 = isCheapReductionGroup(P2);
if (Cheap1 != Cheap2)
return Cheap2;
return NegatedReducedVals.contains(P1.front()) <
@@ -30303,9 +30306,11 @@ class HorizontalReduction {
}
/// Attempt to vectorize the tree found by matchAssociativeReduction.
+ /// \p IsSeedRoot allows vectorizing the groups at the minimum vector
+ /// factor. Set for single-use seed-level roots only.
Value *tryToReduce(BoUpSLP &V, const DataLayout &DL, TargetTransformInfo *TTI,
const TargetLibraryInfo &TLI, AssumptionCache *AC,
- DominatorTree &DT) {
+ DominatorTree &DT, bool IsSeedRoot = false) {
constexpr unsigned RegMaxNumber = 4;
const unsigned RedValsMaxNumber =
(RK == ReductionOrdering::Ordered &&
@@ -30553,6 +30558,39 @@ class HorizontalReduction {
}
ReducedVals.swap(LocalReducedVals);
FastMathFlags GroupRdxFMF = RdxFMF;
+ // The minimum vector factor pays off only when it covers the whole
+ // reduction: every group must be either a pair of distinct values or
+ // large enough for the regular vector factor.
+ const bool NoScalarLeftovers =
+ all_of(ReducedVals, [this](ArrayRef<Value *> Vals) {
+ return Vals.size() >= ReductionLimit ||
+ (Vals.size() == 2 && Vals.front() != Vals.back());
+ });
+ // Same-size groups of the seed-level reduction get the profitability
+ // ordering (cheap groups go last) when the minimum vector factor covers
+ // the whole reduction.
+ if (IsSeedRoot && NegatedReducedVals.empty() && NoScalarLeftovers &&
+ any_of(ReducedVals,
+ [](ArrayRef<Value *> Vals) { return Vals.size() == 2; })) {
+ SmallVector<unsigned> GroupOrder(ReducedVals.size());
+ std::iota(GroupOrder.begin(), GroupOrder.end(), 0);
+ stable_sort(GroupOrder, [&](unsigned L, unsigned R) {
+ ArrayRef<Value *> P1 = ReducedVals[L], P2 = ReducedVals[R];
+ if (P1.size() != P2.size())
+ return P1.size() > P2.size();
+ return !isCheapReductionGroup(P1) && isCheapReductionGroup(P2);
+ });
+ SmallVector<SmallVector<Value *>> SortedVals;
+ SmallVector<InstructionsState> SortedStates;
+ SortedVals.reserve(ReducedVals.size());
+ SortedStates.reserve(States.size());
+ for (unsigned Idx : GroupOrder) {
+ SortedVals.emplace_back(std::move(ReducedVals[Idx]));
+ SortedStates.push_back(States[Idx]);
+ }
+ ReducedVals = std::move(SortedVals);
+ States = std::move(SortedStates);
+ }
for (unsigned I = 0, E = ReducedVals.size(); I < E; ++I) {
ArrayRef<Value *> OrigReducedVals = ReducedVals[I];
InstructionsState S = States[I];
@@ -30645,11 +30683,24 @@ class HorizontalReduction {
}
unsigned NumReducedVals = Candidates.size();
+ auto UsedByReductionOnly = [&](Value *V) {
+ if (!V->hasUseList())
+ return true;
+ // Bail out if we have too many uses to save compilation time.
+ if (V->hasNUsesOrMore(UsesLimit))
+ return false;
+ return all_of(V->users(),
+ [&](User *U) { return IgnoreList.contains(U); });
+ };
// Sign-aware reductions pair small positive/negative groups: allow
- // non-splat groups down to 2 elements.
+ // non-splat groups down to 2 elements. Seed-level reductions get the
+ // same for groups whose values are used by the reduction operations
+ // only, if the minimum vector factor covers the whole reduction.
+ const bool MinVFAllowed = IsSeedRoot && NoScalarLeftovers &&
+ all_of(Candidates, UsedByReductionOnly);
if (NumReducedVals < ReductionLimit &&
- (NumReducedVals < 2 ||
- (!isSplat(Candidates) && NegatedReducedVals.empty())))
+ (NumReducedVals < 2 || (!isSplat(Candidates) &&
+ NegatedReducedVals.empty() && !MinVFAllowed)))
continue;
// Check if we support repeated scalar values processing (optimization of
@@ -30789,9 +30840,12 @@ class HorizontalReduction {
};
bool AnyVectorized = false;
SmallDenseSet<std::pair<unsigned, unsigned>, 8> IgnoredCandidates;
- // Same small-group allowance for the vector width.
+ // Same small-group allowance for the vector width, if it covers the
+ // whole group.
const unsigned MinReduxWidth =
- NegatedReducedVals.empty() ? ReductionLimit : 2;
+ !NegatedReducedVals.empty() || (MinVFAllowed && NumReducedVals == 2)
+ ? 2
+ : ReductionLimit;
while (Pos < NumReducedVals - ReduxWidth + 1 &&
ReduxWidth >= MinReduxWidth) {
// Dependency in tree of the reduction ops - drop this attempt, try
@@ -32577,7 +32631,8 @@ bool SLPVectorizerPass::vectorizeHorReduction(
Stack.emplace(SelectRoot(), 0);
SmallPtrSet<Value *, 8> VisitedInstrs;
bool Res = false;
- auto TryToReduce = [this, &R, TTI = TTI](Instruction *Inst) -> Value * {
+ auto TryToReduce = [this, &R, TTI = TTI](Instruction *Inst,
+ bool IsSeedRoot) -> Value * {
if (R.isAnalyzedReductionRoot(Inst))
return nullptr;
if (!isReductionCandidate(Inst))
@@ -32585,7 +32640,7 @@ bool SLPVectorizerPass::vectorizeHorReduction(
HorizontalReduction HorRdx;
Value *Res = nullptr;
if (HorRdx.matchAssociativeReduction(R, Inst, *SE, *DT, *DL, *TTI, *TLI)) {
- Value *Red = HorRdx.tryToReduce(R, *DL, TTI, *TLI, AC, *DT);
+ Value *Red = HorRdx.tryToReduce(R, *DL, TTI, *TLI, AC, *DT, IsSeedRoot);
// The chain with the flattened fsub/fneg links produced no vector part:
// retry with the fsub/fneg links as leaves, they may be vectorizable on
// their own.
@@ -32593,7 +32648,8 @@ bool SLPVectorizerPass::vectorizeHorReduction(
HorizontalReduction UnflattenedHorRdx;
if (UnflattenedHorRdx.matchAssociativeReduction(
R, Inst, *SE, *DT, *DL, *TTI, *TLI, /*FlattenNegations=*/false))
- Red = UnflattenedHorRdx.tryToReduce(R, *DL, TTI, *TLI, AC, *DT);
+ Red = UnflattenedHorRdx.tryToReduce(R, *DL, TTI, *TLI, AC, *DT,
+ IsSeedRoot);
}
if (Red) {
if (Red != Inst)
@@ -32638,7 +32694,10 @@ bool SLPVectorizerPass::vectorizeHorReduction(
// iteration while stack was populated before that happened.
if (R.isDeleted(Inst))
continue;
- if (Value *VectorizedV = TryToReduce(Inst)) {
+ // The minimum-vector-factor allowance applies to single-use seed-level
+ // roots only.
+ if (Value *VectorizedV =
+ TryToReduce(Inst, /*IsSeedRoot=*/Level == 0 && Inst->hasOneUse())) {
Res = true;
if (auto *I = dyn_cast<Instruction>(VectorizedV); I && I != Inst) {
// Try to find another reduction.
diff --git a/llvm/test/Transforms/PhaseOrdering/X86/fma-reassociate-pairs.ll b/llvm/test/Transforms/PhaseOrdering/X86/fma-reassociate-pairs.ll
index 1d2985733a698..f0d5eee0c7283 100644
--- a/llvm/test/Transforms/PhaseOrdering/X86/fma-reassociate-pairs.ll
+++ b/llvm/test/Transforms/PhaseOrdering/X86/fma-reassociate-pairs.ll
@@ -64,7 +64,7 @@ define double @fadd_fmul_2_right(ptr %x, ptr %y, ptr %z) {
; CHECK-NEXT: [[TMP2:%.*]] = load <2 x double>, ptr [[Y]], align 8
; CHECK-NEXT: [[TMP3:%.*]] = fmul reassoc nsz contract <2 x double> [[TMP2]], [[TMP1]]
; CHECK-NEXT: [[TMP4:%.*]] = load <2 x double>, ptr [[Z]], align 8
-; CHECK-NEXT: [[TMP5:%.*]] = fadd reassoc nsz contract <2 x double> [[TMP4]], [[TMP3]]
+; CHECK-NEXT: [[TMP5:%.*]] = fadd reassoc nsz contract <2 x double> [[TMP3]], [[TMP4]]
; CHECK-NEXT: [[R:%.*]] = tail call reassoc nsz contract double @llvm.vector.reduce.fadd.v2f64(double 0.000000e+00, <2 x double> [[TMP5]])
; CHECK-NEXT: ret double [[R]]
;
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/fma-operand-index.ll b/llvm/test/Transforms/SLPVectorizer/X86/fma-operand-index.ll
index 0fa4058f874d2..154d15f4d2dad 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/fma-operand-index.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/fma-operand-index.ll
@@ -1,26 +1,22 @@
; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
; RUN: opt -S --passes=slp-vectorizer -mtriple=x86_64-unknown-linux-gnu -mcpu=znver4 < %s | FileCheck %s
-; A one-use fmul is left scalar so the backend can fuse it, but only operand 0
-; of the fadd/fsub is checked. The same fmul is kept or gathered depending on
-; which side it sits on. To be addressed in a follow-up.
+; Seed-level reassociable fadd/fsub chains vectorize as unordered reductions
+; when the reduced values split into 2-element groups: the groups share the
+; reduction operation cost. The vectorized fmul still fuses into a vector fma
+; in the backend, whichever operand of the fadd/fsub it feeds.
-; Both fmuls are operand 1, so they gather and cannot fuse.
+; Both fmuls are operand 1 of the fadds.
define double @fmul_rhs_fadd(ptr %x, ptr %y, ptr %z) {
; CHECK-LABEL: define double @fmul_rhs_fadd(
; CHECK-SAME: ptr [[X:%.*]], ptr [[Y:%.*]], ptr [[Z:%.*]]) #[[ATTR0:[0-9]+]] {
; CHECK-NEXT: [[ENTRY:.*:]]
-; CHECK-NEXT: [[Z8:%.*]] = getelementptr inbounds nuw i8, ptr [[Z]], i64 8
-; CHECK-NEXT: [[Z0:%.*]] = load double, ptr [[Z]], align 8
; CHECK-NEXT: [[TMP0:%.*]] = load <2 x double>, ptr [[X]], align 8
; CHECK-NEXT: [[TMP1:%.*]] = load <2 x double>, ptr [[Y]], align 8
; CHECK-NEXT: [[TMP2:%.*]] = fmul reassoc nsz contract <2 x double> [[TMP1]], [[TMP0]]
-; CHECK-NEXT: [[Z1:%.*]] = load double, ptr [[Z8]], align 8
-; CHECK-NEXT: [[ZSUM:%.*]] = fadd reassoc nsz contract double [[Z0]], [[Z1]]
-; CHECK-NEXT: [[MUL0:%.*]] = extractelement <2 x double> [[TMP2]], i64 0
-; CHECK-NEXT: [[ADD0:%.*]] = fadd reassoc nsz contract double [[ZSUM]], [[MUL0]]
-; CHECK-NEXT: [[MUL1:%.*]] = extractelement <2 x double> [[TMP2]], i64 1
-; CHECK-NEXT: [[ADD1:%.*]] = fadd reassoc nsz contract double [[ADD0]], [[MUL1]]
+; CHECK-NEXT: [[TMP3:%.*]] = load <2 x double>, ptr [[Z]], align 8
+; CHECK-NEXT: [[RDX_OP:%.*]] = fadd reassoc nsz contract <2 x double> [[TMP2]], [[TMP3]]
+; CHECK-NEXT: [[ADD1:%.*]] = call reassoc nsz contract double @llvm.vector.reduce.fadd.v2f64(double 0.000000e+00, <2 x double> [[RDX_OP]])
; CHECK-NEXT: ret double [[ADD1]]
;
entry:
@@ -72,25 +68,17 @@ entry:
ret double %sub1
}
-; Operand 0 is the one checked, so these stay scalar and fuse.
+; Both fmuls are operand 0 of the fadds.
define double @fmul_lhs_fadd(ptr %x, ptr %y, ptr %z) {
; CHECK-LABEL: define double @fmul_lhs_fadd(
; CHECK-SAME: ptr [[X:%.*]], ptr [[Y:%.*]], ptr [[Z:%.*]]) #[[ATTR0]] {
; CHECK-NEXT: [[ENTRY:.*:]]
-; CHECK-NEXT: [[X8:%.*]] = getelementptr inbounds nuw i8, ptr [[X]], i64 8
-; CHECK-NEXT: [[Y8:%.*]] = getelementptr inbounds nuw i8, ptr [[Y]], i64 8
-; CHECK-NEXT: [[Z8:%.*]] = getelementptr inbounds nuw i8, ptr [[Z]], i64 8
-; CHECK-NEXT: [[X0:%.*]] = load double, ptr [[X]], align 8
-; CHECK-NEXT: [[Y0:%.*]] = load double, ptr [[Y]], align 8
-; CHECK-NEXT: [[MUL0:%.*]] = fmul reassoc nsz contract double [[Y0]], [[X0]]
-; CHECK-NEXT: [[Z0:%.*]] = load double, ptr [[Z]], align 8
-; CHECK-NEXT: [[X1:%.*]] = load double, ptr [[X8]], align 8
-; CHECK-NEXT: [[Y1:%.*]] = load double, ptr [[Y8]], align 8
-; CHECK-NEXT: [[MUL1:%.*]] = fmul reassoc nsz contract double [[Y1]], [[X1]]
-; CHECK-NEXT: [[Z1:%.*]] = load double, ptr [[Z8]], align 8
-; CHECK-NEXT: [[ZSUM:%.*]] = fadd reassoc nsz contract double [[Z0]], [[Z1]]
-; CHECK-NEXT: [[ADD0:%.*]] = fadd reassoc nsz contract double [[MUL0]], [[ZSUM]]
-; CHECK-NEXT: [[ADD1:%.*]] = fadd reassoc nsz contract double [[MUL1]], [[ADD0]]
+; CHECK-NEXT: [[TMP0:%.*]] = load <2 x double>, ptr [[X]], align 8
+; CHECK-NEXT: [[TMP1:%.*]] = load <2 x double>, ptr [[Y]], align 8
+; CHECK-NEXT: [[TMP2:%.*]] = fmul reassoc nsz contract <2 x double> [[TMP1]], [[TMP0]]
+; CHECK-NEXT: [[TMP3:%.*]] = load <2 x double>, ptr [[Z]], align 8
+; CHECK-NEXT: [[RDX_OP:%.*]] = fadd reassoc nsz contract <2 x double> [[TMP2]], [[TMP3]]
+; CHECK-NEXT: [[ADD1:%.*]] = call reassoc nsz contract double @llvm.vector.reduce.fadd.v2f64(double 0.000000e+00, <2 x double> [[RDX_OP]])
; CHECK-NEXT: ret double [[ADD1]]
;
entry:
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/reassociated-fma-reduction.ll b/llvm/test/Transforms/SLPVectorizer/X86/reassociated-fma-reduction.ll
index 179d94b02307c..1fd49b8e58101 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/reassociated-fma-reduction.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/reassociated-fma-reduction.ll
@@ -5,20 +5,12 @@ define double @test(ptr %x, ptr %y, ptr %z) {
; CHECK-LABEL: define double @test(
; CHECK-SAME: ptr [[X:%.*]], ptr [[Y:%.*]], ptr [[Z:%.*]]) #[[ATTR0:[0-9]+]] {
; CHECK-NEXT: [[ENTRY:.*:]]
-; CHECK-NEXT: [[X0:%.*]] = load double, ptr [[X]], align 8
-; CHECK-NEXT: [[Y0:%.*]] = load double, ptr [[Y]], align 8
-; CHECK-NEXT: [[MUL0:%.*]] = fmul reassoc nsz contract double [[X0]], [[Y0]]
-; CHECK-NEXT: [[Z0:%.*]] = load double, ptr [[Z]], align 8
-; CHECK-NEXT: [[ADD0:%.*]] = fadd reassoc nsz contract double [[MUL0]], [[Z0]]
-; CHECK-NEXT: [[X1P:%.*]] = getelementptr inbounds i8, ptr [[X]], i64 8
-; CHECK-NEXT: [[X1:%.*]] = load double, ptr [[X1P]], align 8
-; CHECK-NEXT: [[Y1P:%.*]] = getelementptr inbounds i8, ptr [[Y]], i64 8
-; CHECK-NEXT: [[Y1:%.*]] = load double, ptr [[Y1P]], align 8
-; CHECK-NEXT: [[MUL1:%.*]] = fmul reassoc nsz contract double [[X1]], [[Y1]]
-; CHECK-NEXT: [[Z1P:%.*]] = getelementptr inbounds i8, ptr [[Z]], i64 8
-; CHECK-NEXT: [[Z1:%.*]] = load double, ptr [[Z1P]], align 8
-; CHECK-NEXT: [[ADD1:%.*]] = fadd reassoc nsz contract double [[ADD0]], [[Z1]]
-; CHECK-NEXT: [[ADD2:%.*]] = fadd reassoc nsz contract double [[ADD1]], [[MUL1]]
+; CHECK-NEXT: [[TMP0:%.*]] = load <2 x double>, ptr [[X]], align 8
+; CHECK-NEXT: [[TMP1:%.*]] = load <2 x double>, ptr [[Y]], align 8
+; CHECK-NEXT: [[TMP2:%.*]] = fmul reassoc nsz contract <2 x double> [[TMP0]], [[TMP1]]
+; CHECK-NEXT: [[TMP3:%.*]] = load <2 x double>, ptr [[Z]], align 8
+; CHECK-NEXT: [[RDX_OP:%.*]] = fadd reassoc nsz contract <2 x double> [[TMP2]], [[TMP3]]
+; CHECK-NEXT: [[ADD2:%.*]] = call reassoc nsz contract double @llvm.vector.reduce.fadd.v2f64(double 0.000000e+00, <2 x double> [[RDX_OP]])
; CHECK-NEXT: ret double [[ADD2]]
;
; OFF-LABEL: @reassoc_fma_reduction(
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/revectorized_rdx_crash.ll b/llvm/test/Transforms/SLPVectorizer/X86/revectorized_rdx_crash.ll
index 48b2174cac688..c5bdff3bf59d1 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/revectorized_rdx_crash.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/revectorized_rdx_crash.ll
@@ -19,19 +19,21 @@ define void @test(i1 %arg, ptr %p) {
; CHECK: for.cond.preheader:
; CHECK-NEXT: [[I:%.*]] = getelementptr inbounds [100 x i32], ptr [[P:%.*]], i64 0, i64 2
; CHECK-NEXT: [[I1:%.*]] = getelementptr inbounds [100 x i32], ptr [[P]], i64 0, i64 3
-; CHECK-NEXT: [[TMP0:%.*]] = load <4 x i32>, ptr [[I]], align 8
-; CHECK-NEXT: [[TMP1:%.*]] = call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> [[TMP0]])
-; CHECK-NEXT: [[OP_RDX3:%.*]] = add i32 0, [[TMP1]]
-; CHECK-NEXT: [[TMP2:%.*]] = load <4 x i32>, ptr [[I1]], align 4
-; CHECK-NEXT: [[TMP3:%.*]] = call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> [[TMP2]])
-; CHECK-NEXT: [[OP_RDX2:%.*]] = add i32 0, [[TMP3]]
+; CHECK-NEXT: [[I2:%.*]] = getelementptr inbounds [100 x i32], ptr [[P]], i64 0, i64 4
+; CHECK-NEXT: [[I3:%.*]] = getelementptr inbounds [100 x i32], ptr [[P]], i64 0, i64 5
+; CHECK-NEXT: [[TMP0:%.*]] = load <2 x i32>, ptr [[I3]], align 4
+; CHECK-NEXT: [[TMP1:%.*]] = load <2 x i32>, ptr [[I2]], align 16
+; CHECK-NEXT: [[TMP2:%.*]] = load <2 x i32>, ptr [[I1]], align 4
+; CHECK-NEXT: [[TMP3:%.*]] = load <2 x i32>, ptr [[I]], align 8
+; CHECK-NEXT: [[TMP7:%.*]] = add <2 x i32> [[TMP0]], [[TMP1]]
+; CHECK-NEXT: [[TMP5:%.*]] = add <2 x i32> [[TMP2]], [[TMP3]]
+; CHECK-NEXT: [[TMP6:%.*]] = add <2 x i32> [[TMP7]], [[TMP5]]
+; CHECK-NEXT: [[OP_RDX3:%.*]] = call i32 @llvm.vector.reduce.add.v2i32(<2 x i32> [[TMP6]])
; CHECK-NEXT: [[TMP4:%.*]] = mul i32 [[OP_RDX3]], 2
; CHECK-NEXT: [[OP_RDX:%.*]] = add i32 0, [[TMP4]]
-; CHECK-NEXT: [[TMP5:%.*]] = mul i32 [[OP_RDX2]], 2
-; CHECK-NEXT: [[OP_RDX1:%.*]] = add i32 [[OP_RDX]], [[TMP5]]
; CHECK-NEXT: br label [[IF_END]]
; CHECK: if.end:
-; CHECK-NEXT: [[R:%.*]] = phi i32 [ [[OP_RDX1]], [[FOR_COND_PREHEADER]] ], [ 0, [[ENTRY:%.*]] ]
+; CHECK-NEXT: [[R:%.*]] = phi i32 [ [[OP_RDX]], [[FOR_COND_PREHEADER]] ], [ 0, [[ENTRY:%.*]] ]
; CHECK-NEXT: ret void
;
entry:
More information about the llvm-commits
mailing list