[llvm] [LV] Always fold constant branches via removeBranchOnConst. (PR #206673)
via llvm-commits
llvm-commits at lists.llvm.org
Tue Jun 30 01:09:11 PDT 2026
llvmorg-github-actions[bot] wrote:
<!--LLVM PR SUMMARY COMMENT-->
@llvm/pr-subscribers-vectorizers
@llvm/pr-subscribers-llvm-transforms
Author: Florian Hahn (fhahn)
<details>
<summary>Changes</summary>
Migrate the remaining final removeBranchOnConst call in executePlan from
OnlyLatches=(EpilogueVecKind != None) to false, so it also folds constant
BranchOnCond recipes during epilogue vectorization. With both call sites
migrated, the OnlyLatches mode is no longer used, so drop the parameter.
The epilogue related code was already prepared to tolerate such folds in
the preceding change.
This is another step to unify the regular VPlan and epilogue codegen
paths.
Depends on https://github.com/llvm/llvm-project/pull/204243
---
Patch is 362.96 KiB, truncated to 20.00 KiB below, full version: https://github.com/llvm/llvm-project/pull/206673.diff
38 Files Affected:
- (modified) llvm/lib/Transforms/Vectorize/LoopVectorize.cpp (+96-58)
- (modified) llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp (+2-9)
- (modified) llvm/lib/Transforms/Vectorize/VPlanTransforms.h (+3-4)
- (modified) llvm/test/Transforms/LoopVectorize/AArch64/conditional-branches-cost.ll (+6-7)
- (modified) llvm/test/Transforms/LoopVectorize/AArch64/deterministic-type-shrinkage.ll (+18-21)
- (modified) llvm/test/Transforms/LoopVectorize/AArch64/epilog-vectorization-factors.ll (+10-10)
- (modified) llvm/test/Transforms/LoopVectorize/AArch64/epilog-vectorization-widen-inductions.ll (+13-46)
- (modified) llvm/test/Transforms/LoopVectorize/AArch64/epilogue-vectorization-fix-scalar-resume-values.ll (+24-85)
- (modified) llvm/test/Transforms/LoopVectorize/AArch64/force-target-instruction-cost.ll (+14-22)
- (modified) llvm/test/Transforms/LoopVectorize/AArch64/induction-costs.ll (+10-15)
- (modified) llvm/test/Transforms/LoopVectorize/AArch64/interleave-with-runtime-checks.ll (+5-5)
- (modified) llvm/test/Transforms/LoopVectorize/AArch64/neon-inloop-reductions.ll (+19-22)
- (modified) llvm/test/Transforms/LoopVectorize/AArch64/partial-reduce-sub-epilogue-vec.ll (+44-101)
- (modified) llvm/test/Transforms/LoopVectorize/AArch64/pr60831-sve-inv-store-crash.ll (+16-18)
- (modified) llvm/test/Transforms/LoopVectorize/AArch64/reduction-cost.ll (+16-26)
- (modified) llvm/test/Transforms/LoopVectorize/AArch64/store-costs-sve.ll (+7-7)
- (modified) llvm/test/Transforms/LoopVectorize/AArch64/sve-epilog-vect.ll (+38-57)
- (modified) llvm/test/Transforms/LoopVectorize/AArch64/transform-narrow-interleave-to-widen-memory-cost.ll (+7-11)
- (modified) llvm/test/Transforms/LoopVectorize/X86/cost-conditional-branches.ll (+28-27)
- (modified) llvm/test/Transforms/LoopVectorize/X86/epilog-vectorization-inductions.ll (+10-16)
- (modified) llvm/test/Transforms/LoopVectorize/X86/epilog-vectorization-ordered-reduction.ll (+39-43)
- (modified) llvm/test/Transforms/LoopVectorize/X86/induction-costs.ll (+10-14)
- (modified) llvm/test/Transforms/LoopVectorize/X86/invariant-store-vectorization.ll (+12-14)
- (modified) llvm/test/Transforms/LoopVectorize/X86/limit-vf-by-tripcount.ll (+29-32)
- (modified) llvm/test/Transforms/LoopVectorize/X86/masked-store-cost.ll (+10-14)
- (modified) llvm/test/Transforms/LoopVectorize/X86/masked_load_store.ll (+63-67)
- (modified) llvm/test/Transforms/LoopVectorize/X86/multi-exit-cost.ll (+11-26)
- (modified) llvm/test/Transforms/LoopVectorize/X86/pr56319-vector-exit-cond-optimization-epilogue-vectorization.ll (+7-8)
- (modified) llvm/test/Transforms/LoopVectorize/X86/strided_load_cost.ll (+28-32)
- (modified) llvm/test/Transforms/LoopVectorize/X86/transform-narrow-interleave-to-widen-memory-epilogue-vec.ll (+28-29)
- (modified) llvm/test/Transforms/LoopVectorize/X86/transform-narrow-interleave-to-widen-memory-live-outs.ll (+3-3)
- (modified) llvm/test/Transforms/LoopVectorize/X86/transform-narrow-interleave-to-widen-memory.ll (+6-6)
- (modified) llvm/test/Transforms/LoopVectorize/X86/vect.omp.force.small-tc.ll (+7-8)
- (modified) llvm/test/Transforms/LoopVectorize/epilog-iv-select-cmp.ll (+61-78)
- (modified) llvm/test/Transforms/LoopVectorize/epilog-vectorization-any-of-reductions.ll (+87)
- (modified) llvm/test/Transforms/LoopVectorize/epilog-vectorization-reductions.ll (+63-169)
- (modified) llvm/test/Transforms/LoopVectorize/epilog-vectorization-scev-expansion.ll (+6-7)
- (modified) llvm/test/Transforms/LoopVectorize/optimal-epilog-vectorization.ll (+28-125)
``````````diff
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 60fe8fda120b4..9dbf3fdde0214 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -5945,8 +5945,7 @@ DenseMap<const SCEV *, Value *> LoopVectorizationPlanner::executePlan(
PSE);
RUN_VPLAN_PASS(VPlanTransforms::simplifyRecipes, BestVPlan);
if (EpilogueVecKind == EpilogueVectorizationKind::None)
- RUN_VPLAN_PASS(VPlanTransforms::removeBranchOnConst, BestVPlan,
- /*OnlyLatches=*/false);
+ RUN_VPLAN_PASS(VPlanTransforms::removeBranchOnConst, BestVPlan);
if (BestVPlan.getEntry()->getSingleSuccessor() ==
BestVPlan.getScalarPreheader()) {
// TODO: The vector loop would be dead, should not even try to vectorize.
@@ -5973,19 +5972,20 @@ DenseMap<const SCEV *, Value *> LoopVectorizationPlanner::executePlan(
RUN_VPLAN_PASS(VPlanTransforms::expandBranchOnTwoConds, BestVPlan);
// Convert loops with variable-length stepping after regions are dissolved.
RUN_VPLAN_PASS(VPlanTransforms::convertToVariableLengthStep, BestVPlan);
- // Remove dead back-edges for single-iteration loops with BranchOnCond(true).
- // Only process loop latches to avoid removing edges from the middle block,
- // which may be needed for epilogue vectorization.
- VPlanTransforms::removeBranchOnConst(BestVPlan, /*OnlyLatches=*/true);
- VPlanTransforms::materializeBackedgeTakenCount(BestVPlan, VectorPH);
- std::optional<uint64_t> MaxRuntimeStep;
- if (auto MaxVScale = getMaxVScale(*CM.TheFunction, CM.TTI))
- MaxRuntimeStep = uint64_t(*MaxVScale) * BestVF.getKnownMinValue() * BestUF;
- VPlanTransforms::materializeVectorTripCount(
- BestVPlan, VectorPH, CM.foldTailByMasking(),
- CM.requiresScalarEpilogue(BestVF.isVector()), &BestVPlan.getVFxUF(),
- MaxRuntimeStep);
- VPlanTransforms::materializeFactors(BestVPlan, VectorPH, BestVF);
+ // Fold any remaining BranchOnCond with constant condition.
+ VPlanTransforms::removeBranchOnConst(BestVPlan);
+ if (VectorPH->hasPredecessors()) {
+ VPlanTransforms::materializeBackedgeTakenCount(BestVPlan, VectorPH);
+ std::optional<uint64_t> MaxRuntimeStep;
+ if (auto MaxVScale = getMaxVScale(*CM.TheFunction, CM.TTI))
+ MaxRuntimeStep =
+ uint64_t(*MaxVScale) * BestVF.getKnownMinValue() * BestUF;
+ VPlanTransforms::materializeVectorTripCount(
+ BestVPlan, VectorPH, CM.foldTailByMasking(),
+ CM.requiresScalarEpilogue(BestVF.isVector()), &BestVPlan.getVFxUF(),
+ MaxRuntimeStep);
+ VPlanTransforms::materializeFactors(BestVPlan, VectorPH, BestVF);
+ }
// Limit expansions to VPInstruction to when not vectorizing the epilogue.
// Currently this code path still relies on code re-using SCEVs expanded
// directly to IR instructions.
@@ -5995,9 +5995,7 @@ DenseMap<const SCEV *, Value *> LoopVectorizationPlanner::executePlan(
VPlanTransforms::simplifyRecipes(BestVPlan);
// Removing branches and incoming values may expose additional simplification
// opportunities.
- if (VPlanTransforms::removeBranchOnConst(BestVPlan,
- /*OnlyLatches=*/EpilogueVecKind !=
- EpilogueVectorizationKind::None))
+ if (VPlanTransforms::removeBranchOnConst(BestVPlan))
VPlanTransforms::simplifyRecipes(BestVPlan);
VPlanTransforms::simplifyKnownEVL(BestVPlan, BestVF, PSE);
@@ -6836,7 +6834,7 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(VPlanPtr Plan,
if (!RUN_VPLAN_PASS(VPlanTransforms::handleFindLastReductions, *Plan))
return nullptr;
- RUN_VPLAN_PASS(VPlanTransforms::removeBranchOnConst, *Plan, false);
+ RUN_VPLAN_PASS(VPlanTransforms::removeBranchOnConst, *Plan);
// Create partial reduction recipes for scaled reductions and transform
// recipes to abstract recipes if it is legal and beneficial and clamp the
@@ -7621,7 +7619,6 @@ static SmallVector<Instruction *> preparePlanForEpilogueVectorLoop(
RecurKind RK = ReductionPhi->getRecurrenceKind();
if (RecurrenceDescriptor::isAnyOfRecurrenceKind(RK) || IsFindIV) {
- auto *ResumePhi = cast<PHINode>(ResumeV);
VPValue *BypassOp = ResumeForEpi->getOperand(1);
assert((isa<VPIRValue>(BypassOp) ||
VPlanPatternMatch::match(
@@ -7629,8 +7626,11 @@ static SmallVector<Instruction *> preparePlanForEpilogueVectorLoop(
m_VPInstruction<Instruction::Freeze>(m_VPValue()))) &&
"expected live-in or Freeze");
Value *StartV = BypassOp->getUnderlyingValue();
- IRBuilder<> Builder(ResumePhi->getParent(),
- ResumePhi->getParent()->getFirstNonPHIIt());
+ BasicBlock::iterator InsertAt =
+ L->getLoopPreheader()->getFirstNonPHIIt();
+ if (auto *ResumeI = dyn_cast<Instruction>(ResumeV))
+ InsertAt = *ResumeI->getInsertionPointAfterDef();
+ IRBuilder<> Builder(InsertAt->getParent(), InsertAt);
if (RecurrenceDescriptor::isAnyOfRecurrenceKind(RK)) {
// VPReductionPHIRecipes for AnyOf reductions expect a boolean as
@@ -7743,34 +7743,60 @@ static SmallVector<Instruction *> preparePlanForEpilogueVectorLoop(
return InstsToMove;
}
-static void
-fixScalarResumeValuesFromBypass(BasicBlock *BypassBlock, Loop *L,
- VPlan &BestEpiPlan,
- ArrayRef<VPInstruction *> ResumeValues) {
+static void fixScalarResumeValuesFromBypass(
+ BasicBlock *BypassBlock, Loop *L, VPlan &BestEpiPlan,
+ ArrayRef<VPInstruction *> ResumeValues, BasicBlock *VecEpiloguePreHeader,
+ const DominatorTree &DT) {
// Fix resume values from the additional bypass block.
BasicBlock *PH = L->getLoopPreheader();
for (auto *Pred : predecessors(PH)) {
for (PHINode &Phi : PH->phis()) {
if (Phi.getBasicBlockIndex(Pred) != -1)
continue;
+ // BypassBlock may have been folded away by removeBranchOnConst; skip
+ // phis that no longer carry an incoming from it.
+ if (Phi.getBasicBlockIndex(BypassBlock) == -1)
+ continue;
Phi.addIncoming(Phi.getIncomingValueForBlock(BypassBlock), Pred);
}
}
auto *ScalarPH = cast<VPIRBasicBlock>(BestEpiPlan.getScalarPreheader());
- if (ScalarPH->hasPredecessors()) {
+ BasicBlock *ScalarPHBB = ScalarPH->getIRBasicBlock();
+ if (ScalarPH->hasPredecessors() &&
+ is_contained(predecessors(ScalarPHBB), BypassBlock)) {
// Fix resume values for inductions and reductions from the additional
// bypass block using the incoming values from the main loop's resume phis.
// ResumeValues correspond 1:1 with the scalar loop header phis.
for (auto [ResumeV, HeaderPhi] :
- zip(ResumeValues, BestEpiPlan.getScalarHeader()->phis())) {
- auto *HeaderPhiR = cast<VPIRPhi>(&HeaderPhi);
- auto *EpiResumePhi =
- cast<PHINode>(HeaderPhiR->getIRPhi().getIncomingValueForBlock(PH));
+ zip_equal(ResumeValues, BestEpiPlan.getScalarHeader()->phis())) {
+ PHINode &IRHeaderPhi = cast<VPIRPhi>(&HeaderPhi)->getIRPhi();
+ Value *IncomingVal = IRHeaderPhi.getIncomingValueForBlock(PH);
+ auto *EpiResumePhi = dyn_cast<PHINode>(IncomingVal);
+ if (!EpiResumePhi) {
+ // Simplifications folded the resume phi to IncomingVal because all
+ // incomings shared a single VPValue. Recreate it so BypassBlock can
+ // carry a distinct value below.
+ assert(none_of(predecessors(ScalarPHBB),
+ [&](BasicBlock *Pred) {
+ return DT.dominates(VecEpiloguePreHeader, Pred);
+ }) &&
+ "epilogue middle block already reaches scalar.ph");
+ EpiResumePhi =
+ PHINode::Create(IncomingVal->getType(), pred_size(ScalarPHBB),
+ "resume", ScalarPHBB->getFirstNonPHIIt());
+ for (BasicBlock *Pred : predecessors(ScalarPHBB))
+ EpiResumePhi->addIncoming(IncomingVal, Pred);
+ IRHeaderPhi.setIncomingValueForBlock(PH, EpiResumePhi);
+ }
if (EpiResumePhi->getBasicBlockIndex(BypassBlock) == -1)
continue;
- auto *MainResumePhi = cast<PHINode>(ResumeV->getUnderlyingValue());
- EpiResumePhi->setIncomingValueForBlock(
- BypassBlock, MainResumePhi->getIncomingValueForBlock(BypassBlock));
+ // If the resume phi was folded, ResumeV is the common incoming value.
+ Value *BypassIncoming = ResumeV->getUnderlyingValue();
+ assert(BypassIncoming && "no IR value");
+ if (auto *MainResumePhi = dyn_cast<PHINode>(BypassIncoming))
+ if (MainResumePhi->getBasicBlockIndex(BypassBlock) != -1)
+ BypassIncoming = MainResumePhi->getIncomingValueForBlock(BypassBlock);
+ EpiResumePhi->setIncomingValueForBlock(BypassBlock, BypassIncoming);
}
}
}
@@ -7799,8 +7825,11 @@ static void connectEpilogueVectorLoop(VPlan &EpiPlan, Loop *L,
DomTreeUpdater DTU(DT, DomTreeUpdater::UpdateStrategy::Eager);
// Helper to redirect an edge from \p BB to \p VecEpilogueIterationCountCheck
- // to \p NewSucc instead, updating the DomTree.
+ // to \p NewSucc instead, updating the DomTree. No-op if \p BB doesn't branch
+ // there (e.g. its check folded by removeBranchOnConst).
auto RedirectEdge = [&](BasicBlock *BB, BasicBlock *NewSucc) {
+ if (!is_contained(successors(BB), VecEpilogueIterationCountCheck))
+ return;
BB->getTerminator()->replaceUsesOfWith(VecEpilogueIterationCountCheck,
NewSucc);
DTU.applyUpdates(
@@ -7808,7 +7837,10 @@ static void connectEpilogueVectorLoop(VPlan &EpiPlan, Loop *L,
{DominatorTree::Insert, BB, NewSucc}});
};
- RedirectEdge(EPI.MainLoopIterationCountCheck, VecEpiloguePreHeader);
+ // If the main-loop bypass folded to unconditional, its only successor is
+ // the epilogue check and must be preserved.
+ if (isa<CondBrInst>(EPI.MainLoopIterationCountCheck->getTerminator()))
+ RedirectEdge(EPI.MainLoopIterationCountCheck, VecEpiloguePreHeader);
BasicBlock *ScalarPH =
cast<VPIRBasicBlock>(EpiPlan.getScalarPreheader())->getIRBasicBlock();
@@ -7835,18 +7867,14 @@ static void connectEpilogueVectorLoop(VPlan &EpiPlan, Loop *L,
VecEpilogueIterationCountCheck->getSinglePredecessor(),
VecEpilogueIterationCountCheck);
- // If the phi doesn't have an incoming value from the
- // EpilogueIterationCountCheck, we are done. Otherwise remove the
- // incoming value and also those from other check blocks. This is needed
- // for reduction phis only.
- if (none_of(Phi->blocks(), [&](BasicBlock *IncB) {
- return EPI.EpilogueIterationCountCheck == IncB;
- }))
- continue;
+ // The check blocks no longer reach the phi's new parent (their edges
+ // were redirected above or folded by removeBranchOnConst), so drop any
+ // stale incoming values still recorded for them.
for (BasicBlock *BB :
{EPI.EpilogueIterationCountCheck, SCEVCheckBlock, MemCheckBlock}) {
- if (BB)
- Phi->removeIncomingValue(BB);
+ if (!BB || Phi->getBasicBlockIndex(BB) == -1)
+ continue;
+ Phi->removeIncomingValue(BB);
}
}
@@ -7858,7 +7886,7 @@ static void connectEpilogueVectorLoop(VPlan &EpiPlan, Loop *L,
// after executing the main loop. We need to update the resume values of
// inductions and reductions during epilogue vectorization.
fixScalarResumeValuesFromBypass(VecEpilogueIterationCountCheck, L, EpiPlan,
- ResumeValues);
+ ResumeValues, VecEpiloguePreHeader, *DT);
// Remove dead phis that were moved to the epilogue preheader but are unused
// (e.g., resume phis for inductions not widened in the epilogue vector loop).
@@ -8318,23 +8346,33 @@ bool LoopVectorizePass::processLoop(Loop *L) {
LoopVectorizationPlanner::EpilogueVectorizationKind::MainLoop);
++LoopsVectorized;
- // Derive EPI fields from VPlan-generated IR.
+ // Skip epilogue if the main loop covered everything or the chain folded.
+ if (!BestMainPlan.getScalarPreheader()->hasPredecessors())
+ return true;
+
+ // Check chain: Entry -> [SCEV] -> [Mem] -> MainCheck -> VecPH. A check
+ // block is detached if removeBranchOnConst dropped its BranchOnCond before
+ // execution, leaving its UnreachableInst terminator.
BasicBlock *EntryBB =
cast<VPIRBasicBlock>(BestMainPlan.getEntry())->getIRBasicBlock();
EntryBB->setName("iter.check");
EPI.EpilogueIterationCountCheck = EntryBB;
- // The check chain is: Entry -> [SCEV] -> [Mem] -> MainCheck -> VecPH.
- // MainCheck is the non-bypass successor of the last runtime check block
- // (or Entry if there are no runtime checks).
+
BasicBlock *LastCheck = EntryBB;
- if (BasicBlock *MemBB = Checks.getMemRuntimeChecks().second)
- LastCheck = MemBB;
- else if (BasicBlock *SCEVBB = Checks.getSCEVChecks().second)
- LastCheck = SCEVBB;
+ auto IsLive = [](BasicBlock *BB) {
+ return BB && !isa<UnreachableInst>(BB->getTerminator());
+ };
+ if (BasicBlock *BB = Checks.getMemRuntimeChecks().second; IsLive(BB))
+ LastCheck = BB;
+ else if (BasicBlock *BB = Checks.getSCEVChecks().second; IsLive(BB))
+ LastCheck = BB;
+
BasicBlock *ScalarPH = L->getLoopPreheader();
- auto *BI = cast<CondBrInst>(LastCheck->getTerminator());
- EPI.MainLoopIterationCountCheck =
- BI->getSuccessor(BI->getSuccessor(0) == ScalarPH);
+ Instruction *BI = LastCheck->getTerminator();
+ EPI.MainLoopIterationCountCheck = BI->getSuccessor(
+ BI->getNumSuccessors() == 2 && BI->getSuccessor(0) == ScalarPH);
+ if (EPI.MainLoopIterationCountCheck == ScalarPH)
+ return true;
// Second pass vectorizes the epilogue and adjusts the control flow
// edges from the first pass.
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index 46c63bb47b165..ac7a2be131498 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -2773,11 +2773,7 @@ void VPlanTransforms::truncateToMinimalBitwidths(
}
}
-bool VPlanTransforms::removeBranchOnConst(VPlan &Plan, bool OnlyLatches) {
- std::optional<VPDominatorTree> VPDT;
- if (OnlyLatches)
- VPDT.emplace(Plan);
-
+bool VPlanTransforms::removeBranchOnConst(VPlan &Plan) {
// Collect all blocks before modifying the CFG so we can identify unreachable
// ones after constant branch removal.
SmallVector<VPBlockBase *> AllBlocks(vp_depth_first_shallow(Plan.getEntry()));
@@ -2789,9 +2785,6 @@ bool VPlanTransforms::removeBranchOnConst(VPlan &Plan, bool OnlyLatches) {
if (VPBB->empty() || !match(&VPBB->back(), m_BranchOnCond(m_VPValue(Cond))))
continue;
- if (OnlyLatches && !VPBlockUtils::isLatch(VPBB, *VPDT))
- continue;
-
assert(VPBB->getNumSuccessors() == 2 &&
"Two successors expected for BranchOnCond");
unsigned RemovedIdx;
@@ -2859,7 +2852,7 @@ void VPlanTransforms::optimize(VPlan &Plan) {
RUN_VPLAN_PASS(removeRedundantExpandSCEVRecipes, Plan);
RUN_VPLAN_PASS(reassociateHeaderMask, Plan);
RUN_VPLAN_PASS(simplifyRecipes, Plan);
- RUN_VPLAN_PASS(removeBranchOnConst, Plan, /*OnlyLatches=*/false);
+ RUN_VPLAN_PASS(removeBranchOnConst, Plan);
RUN_VPLAN_PASS(simplifyReverses, Plan);
RUN_VPLAN_PASS(removeDeadRecipes, Plan);
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.h b/llvm/lib/Transforms/Vectorize/VPlanTransforms.h
index 8fbf6793e90c5..d3f2b6802d205 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.h
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.h
@@ -405,10 +405,9 @@ struct VPlanTransforms {
static void simplifyReverses(VPlan &Plan);
/// Remove BranchOnCond recipes with true or false conditions together with
- /// removing dead edges to their successors. If \p OnlyLatches is true, only
- /// process loop latches. Returns true if incoming values from any phi-like
- /// recipe have been removed.
- static bool removeBranchOnConst(VPlan &Plan, bool OnlyLatches = false);
+ /// removing dead edges to their successors. Returns true if incoming values
+ /// from any phi-like recipe have been removed.
+ static bool removeBranchOnConst(VPlan &Plan);
/// Perform common-subexpression-elimination on \p Plan.
static void cse(VPlan &Plan);
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/conditional-branches-cost.ll b/llvm/test/Transforms/LoopVectorize/AArch64/conditional-branches-cost.ll
index ec5c95fb0b17d..1a59223c63916 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/conditional-branches-cost.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/conditional-branches-cost.ll
@@ -250,9 +250,9 @@ define void @latch_branch_cost(ptr %dst) {
; DEFAULT-LABEL: define void @latch_branch_cost(
; DEFAULT-SAME: ptr [[DST:%.*]]) {
; DEFAULT-NEXT: [[ITER_CHECK:.*:]]
-; DEFAULT-NEXT: br i1 false, label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; DEFAULT-NEXT: br label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
; DEFAULT: [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
-; DEFAULT-NEXT: br i1 false, label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_PH:.*]]
+; DEFAULT-NEXT: br label %[[VECTOR_PH:.*]]
; DEFAULT: [[VECTOR_PH]]:
; DEFAULT-NEXT: br label %[[VECTOR_BODY:.*]]
; DEFAULT: [[VECTOR_BODY]]:
@@ -265,21 +265,20 @@ define void @latch_branch_cost(ptr %dst) {
; DEFAULT-NEXT: [[TMP4:%.*]] = icmp eq i64 [[INDEX_NEXT]], 96
; DEFAULT-NEXT: br i1 [[TMP4]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
; DEFAULT: [[MIDDLE_BLOCK]]:
-; DEFAULT-NEXT: br i1 false, [[EXIT:label %.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
+; DEFAULT-NEXT: br label %[[VEC_EPILOG_ITER_CHECK:.*]]
; DEFAULT: [[VEC_EPILOG_ITER_CHECK]]:
-; DEFAULT-NEXT: br i1 false, label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF5:![0-9]+]]
+; DEFAULT-NEXT: br i1 false, label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VEC_EPILOG_PH:.*]], !prof [[PROF5:![0-9]+]]
; DEFAULT: [[VEC_EPILOG_PH]]:
-; DEFAULT-NEXT: [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ 96, %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
; DEFAULT-NEXT: br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
; DEFAULT: [[VEC_EPILOG_VECTOR_BODY]]:
-; DEFAULT-NEXT: [[INDEX1:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT2:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; DEFAULT-NEXT: [[INDEX1:%.*]] = phi i64 [ 96, %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT2:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
; DEFAULT-NEXT: [[TMP8:%.*]] = getelementptr i8, ptr [[DST]], i64 [[INDEX1]]
; DEFAULT-NEXT: store <4 x i8> zeroinitializer, ptr [[TMP8]], align 1
; DEFAULT-NEXT: [[INDEX_NEXT2]] = add nuw i64 [[INDEX1]], 4
; DEFAULT-NEXT: [[TMP10:%.*]] = icmp eq i64 [[INDEX_NEXT2]], 100
; DEFAULT-NEXT: br i1 [[TMP10]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP6:![0-9]+]]
; DEFAULT: [[VEC_EPILOG_MIDDLE_BLOCK]]:
-; DEFAULT-NEXT: br i1 true, [[EXIT]], label %[[VEC_EPILOG_SCALAR_PH]]
+; DEFAULT-NEXT: br [[EXIT:label %.*]]
; DEFAULT: [[VEC_EPILOG_SCALAR_PH]]:
;
; PRED-LABEL: define void @latch_branch_cost(
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/deterministic-type-shrinkage.ll b/llvm/test/Transforms/LoopVectorize/AArch64/deterministic-type-shrinkage.ll
index f0664197dcb94..fef79e4da9e9f 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/deterministic-type-shrinkage.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/deterministic-type-shrinkage.ll
@@ -121,9 +121,9 @@ define void @test_shrink_zext_in_preheader(ptr noalias %src, ptr noalias %dst, i
; CHECK-SAME: ptr noalias [[SRC:%.*]], ptr noalias [[DST:%.*]], i32 [[A:%.*]], i16 [[B:%.*]]) {
; CHECK-NEXT: [[ITER_CHECK:.*:]]
; CHECK-NEXT: [[CONV10:%.*]] = zext i16 [[B]] to i32
-; CHECK-NEXT: br i1 false, label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; CHECK-NEXT: br label %[[VECTOR_MAIN_LOOP_ITE...
[truncated]
``````````
</details>
https://github.com/llvm/llvm-project/pull/206673
More information about the llvm-commits
mailing list