[llvm] c4ff7a7 - [LV] Use SCEV loop-uniformity for outer-loop branch legality (#199632)
via llvm-commits
llvm-commits at lists.llvm.org
Wed Sep 30 20:48:46 PDT 2026
Author: Ming Yan
Date: 2026-10-01T03:48:36Z
New Revision: c4ff7a767a271c02dd37312999c80f404219dd30
URL: https://github.com/llvm/llvm-project/commit/c4ff7a767a271c02dd37312999c80f404219dd30
DIFF: https://github.com/llvm/llvm-project/commit/c4ff7a767a271c02dd37312999c80f404219dd30.diff
LOG: [LV] Use SCEV loop-uniformity for outer-loop branch legality (#199632)
This patch refactors the outer-loop vectorization branch legality checks
to reason about conditional branches directly instead of using the old
recursive inner-loop shape check.
The new check allows conditional branches when their condition is
either:
- loop-invariant with respect to the vectorized outer loop, or
- a compare whose operands are both SCEV loop-uniform with respect to
the vectorized outer loop.
Divergent conditional branches are still rejected, now with a more
specific diagnostic.
Added:
Modified:
llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp
llvm/test/Transforms/LoopVectorize/explicit_outer_nonuniform_inner.ll
llvm/test/Transforms/LoopVectorize/explicit_outer_uniform_diverg_branch.ll
llvm/test/Transforms/LoopVectorize/outer_loop_early_exit.ll
llvm/test/Transforms/LoopVectorize/outer_loop_inner_loop_exits.ll
Removed:
################################################################################
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp
index 7c9baf763b8cf..8d58ef9ed9c4f 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp
@@ -324,86 +324,6 @@ void LoopVectorizeHints::setHint(StringRef Name, Metadata *Arg) {
}
}
-// Return true if the inner loop \p Lp is uniform with regard to the outer loop
-// \p OuterLp (i.e., if the outer loop is vectorized, all the vector lanes
-// executing the inner loop will execute the same iterations). This check is
-// very constrained for now but it will be relaxed in the future. \p Lp is
-// considered uniform if it meets all the following conditions:
-// 1) it has a canonical IV (starting from 0 and with stride 1),
-// 2) its latch terminator is a conditional branch and,
-// 3) its latch condition is a compare instruction whose operands are the
-// canonical IV and an OuterLp invariant.
-// This check doesn't take into account the uniformity of other conditions not
-// related to the loop latch because they don't affect the loop uniformity.
-//
-// NOTE: We decided to keep all these checks and its associated documentation
-// together so that we can easily have a picture of the current supported loop
-// nests. However, some of the current checks don't depend on \p OuterLp and
-// would be redundantly executed for each \p Lp if we invoked this function for
-//
diff erent candidate outer loops. This is not the case for now because we
-// don't currently have the infrastructure to evaluate multiple candidate outer
-// loops and \p OuterLp will be a fixed parameter while we only support explicit
-// outer loop vectorization. It's also very likely that these checks go away
-// before introducing the aforementioned infrastructure. However, if this is not
-// the case, we should move the \p OuterLp independent checks to a separate
-// function that is only executed once for each \p Lp.
-static bool isUniformLoop(Loop *Lp, Loop *OuterLp) {
- assert(Lp->getLoopLatch() && "Expected loop with a single latch.");
-
- // If Lp is the outer loop, it's uniform by definition.
- if (Lp == OuterLp)
- return true;
- assert(OuterLp->contains(Lp) && "OuterLp must contain Lp.");
-
- // 1.
- PHINode *IV = Lp->getCanonicalInductionVariable();
- if (!IV) {
- LLVM_DEBUG(dbgs() << "LV: Canonical IV not found.\n");
- return false;
- }
-
- // 2.
- BasicBlock *Latch = Lp->getLoopLatch();
- auto *LatchBr = dyn_cast<CondBrInst>(Latch->getTerminator());
- if (!LatchBr) {
- LLVM_DEBUG(dbgs() << "LV: Unsupported loop latch branch.\n");
- return false;
- }
-
- // 3.
- auto *LatchCmp = dyn_cast<CmpInst>(LatchBr->getCondition());
- if (!LatchCmp) {
- LLVM_DEBUG(
- dbgs() << "LV: Loop latch condition is not a compare instruction.\n");
- return false;
- }
-
- Value *CondOp0 = LatchCmp->getOperand(0);
- Value *CondOp1 = LatchCmp->getOperand(1);
- Value *IVUpdate = IV->getIncomingValueForBlock(Latch);
- if (!(CondOp0 == IVUpdate && OuterLp->isLoopInvariant(CondOp1)) &&
- !(CondOp1 == IVUpdate && OuterLp->isLoopInvariant(CondOp0))) {
- LLVM_DEBUG(dbgs() << "LV: Loop latch condition is not uniform.\n");
- return false;
- }
-
- return true;
-}
-
-// Return true if \p Lp and all its nested loops are uniform with regard to \p
-// OuterLp.
-static bool isUniformLoopNest(Loop *Lp, Loop *OuterLp) {
- if (!isUniformLoop(Lp, OuterLp))
- return false;
-
- // Check if nested loops are uniform.
- for (Loop *SubLp : *Lp)
- if (!isUniformLoopNest(SubLp, OuterLp))
- return false;
-
- return true;
-}
-
static IntegerType *getInductionIntegerTy(const DataLayout &DL, Type *Ty) {
assert(Ty->isIntOrPtrTy() && "Expected integer or pointer type");
@@ -652,30 +572,46 @@ bool LoopVectorizationLegality::canVectorizeOuterLoop() {
}
// Check whether the branch is a supported one. Only unconditional
- // branches, conditional branches with an outer loop invariant condition or
+ // branches, conditional branches with an outer loop uniform condition or
// backedges are supported.
// FIXME: We skip these checks when VPlan predication is enabled as we
// want to allow divergent branches. This whole check will be removed
// once VPlan predication is on by default.
auto *Br = dyn_cast<CondBrInst>(Term);
- if (Br && !TheLoop->isLoopInvariant(Br->getCondition()) &&
- !LI->isLoopHeader(Br->getSuccessor(0)) &&
- !LI->isLoopHeader(Br->getSuccessor(1))) {
- reportVectorizationFailure(
- "Unsupported conditional branch",
- "loop control flow is not understood by vectorizer",
- "CFGNotUnderstood", ORE, TheLoop);
- if (DoExtraAnalysis)
- Result = false;
- else
- return false;
+ if (Br && !TheLoop->isLoopLatch(BB)) {
+ bool IsUniformCondBr = TheLoop->isLoopInvariant(Br->getCondition());
+
+ Value *Lhs = nullptr;
+ Value *Rhs = nullptr;
+ auto *SE = PSE.getSE();
+ if (match(Br->getCondition(), m_c_ICmp(m_Value(Lhs), m_Value(Rhs))) &&
+ !IsUniformCondBr && SE->isSCEVable(Lhs->getType())) {
+ const SCEV *LhsExpr = PSE.getSCEV(Lhs);
+ const SCEV *RhsExpr = PSE.getSCEV(Rhs);
+ IsUniformCondBr |= (SE->isLoopUniform(LhsExpr, TheLoop) &&
+ SE->isLoopUniform(RhsExpr, TheLoop));
+ }
+
+ // If the condition is not uniform, report a failure. We currently require
+ // uniform conditions to avoid the complexity of vectorizing divergent
+ // control flow in the outer loop.
+ if (!IsUniformCondBr) {
+ reportVectorizationFailure(
+ "Outer loop contains divergent conditional branch",
+ "loop control flow is not understood by vectorizer",
+ "CFGNotUnderstood", ORE, TheLoop);
+ if (DoExtraAnalysis)
+ Result = false;
+ else
+ return false;
+ }
}
}
// Each nested loop must exit via its latch only, as a region with the latch
// as its only exiting block is created for it. Note that the branch check
- // above rejects divergent exits, but exits with an outer-loop invariant
- // condition are allowed through.
+ // rejects divergent exits, but exits with an outer-loop uniform condition
+ // are allowed through.
SmallVector<Loop *, 4> LoopNest = TheLoop->getLoopsInPreorder();
for (Loop *Lp : drop_begin(LoopNest)) {
if (Lp->getExitingBlock() != Lp->getLoopLatch()) {
@@ -690,20 +626,6 @@ bool LoopVectorizationLegality::canVectorizeOuterLoop() {
}
}
- // Check whether inner loops are uniform. At this point, we only support
- // simple outer loops scenarios with uniform nested loops.
- if (!isUniformLoopNest(TheLoop /*loop nest*/,
- TheLoop /*context outer loop*/)) {
- reportVectorizationFailure(
- "Outer loop contains divergent loops",
- "loop control flow is not understood by vectorizer", "CFGNotUnderstood",
- ORE, TheLoop);
- if (DoExtraAnalysis)
- Result = false;
- else
- return false;
- }
-
// Check whether we are able to set up outer loop induction.
if (!setupOuterLoopInductions()) {
reportVectorizationFailure("Unsupported outer loop Phi(s)",
diff --git a/llvm/test/Transforms/LoopVectorize/explicit_outer_nonuniform_inner.ll b/llvm/test/Transforms/LoopVectorize/explicit_outer_nonuniform_inner.ll
index 9c87b9cb1e818..61acce96ad283 100644
--- a/llvm/test/Transforms/LoopVectorize/explicit_outer_nonuniform_inner.ll
+++ b/llvm/test/Transforms/LoopVectorize/explicit_outer_nonuniform_inner.ll
@@ -20,7 +20,7 @@
; Case 1 (for (j = i; j < M; j++)): Inner loop with divergent IV start.
; CHECK-LABEL: iv_start
-; CHECK: LV: Not vectorizing: Outer loop contains divergent loops.
+; CHECK: LV: Not vectorizing: Outer loop contains divergent conditional branch.
; CHECK: LV: Not vectorizing: Unsupported outer loop.
target datalayout = "e-m:e-i64:64-f80:128-n8:16:32:64-S128"
@@ -70,7 +70,7 @@ for.end15:
; Case 2 (for (j = 0; j < i; j++)): Inner loop with divergent upper-bound.
; CHECK-LABEL: loop_ub
-; CHECK: LV: Not vectorizing: Outer loop contains divergent loops.
+; CHECK: LV: Not vectorizing: Outer loop contains divergent conditional branch.
; CHECK: LV: Not vectorizing: Unsupported outer loop.
define void @loop_ub(ptr nocapture %a, ptr nocapture readonly %b, i32 %N, i32 %M) {
@@ -116,7 +116,7 @@ for.end15:
; Case 3 (for (j = 0; j < M; j+=i)): Inner loop with divergent step.
; CHECK-LABEL: iv_step
-; CHECK: LV: Not vectorizing: Outer loop contains divergent loops.
+; CHECK: LV: Not vectorizing: Outer loop contains divergent conditional branch.
; CHECK: LV: Not vectorizing: Unsupported outer loop.
define void @iv_step(ptr nocapture %a, ptr nocapture readonly %b, i32 %N, i32 %M) {
diff --git a/llvm/test/Transforms/LoopVectorize/explicit_outer_uniform_diverg_branch.ll b/llvm/test/Transforms/LoopVectorize/explicit_outer_uniform_diverg_branch.ll
index 75858e626b728..56d5ea6b66f6f 100644
--- a/llvm/test/Transforms/LoopVectorize/explicit_outer_uniform_diverg_branch.ll
+++ b/llvm/test/Transforms/LoopVectorize/explicit_outer_uniform_diverg_branch.ll
@@ -4,7 +4,7 @@
; Verify that LV can handle explicit vectorization outer loops with uniform branches
; but bails out on outer loops with divergent branches.
-; Root C/C++ source code for the test cases
+; Root C/C++ source code for the first two test cases.
; void foo(int *a, int *b, int N, int M)
; {
; int i, j;
@@ -74,7 +74,7 @@ for.end19:
; Case 2 (COND => B[i * M] == 0): Outer loop with divergent conditional branch.
; CHECK-LABEL: divergent_branch
-; CHECK: Unsupported conditional branch.
+; CHECK: LV: Not vectorizing: Outer loop contains divergent conditional branch.
; CHECK: LV: Not vectorizing: Unsupported outer loop.
define void @divergent_branch(ptr nocapture %a, ptr nocapture readonly %b, i32 %N, i32 %M) {
@@ -122,6 +122,62 @@ for.end19:
ret void
}
+; Case 3: Three-level loop nest with a triangular innermost loop.
+;
+; #pragma clang loop vectorize(enable) vectorize_width(8)
+; for (size_t i = 0; i < M; ++i)
+; for (size_t j = 0; j < N; ++j)
+; for (size_t k = 0; k < j; ++k)
+; a[i * N * N + j * N + k] = b[i * N * N + j * N + k];
+;
+; The innermost loop latch condition depends on the middle loop IV, but is
+; uniform with respect to the vectorized outer loop.
+
+; CHECK-LABEL: uniform_triangular_inner_loop
+; CHECK: LV: We can vectorize this outer loop!
+
+define void @uniform_triangular_inner_loop(ptr nocapture %a, ptr nocapture readonly %b, i64 %M, i64 %N) {
+entry:
+ br label %outer.header
+
+outer.header:
+ %i = phi i64 [ 0, %entry ], [ %i.next, %outer.latch ]
+ br label %middle.header
+
+middle.header:
+ %j = phi i64 [ 0, %outer.header ], [ %j.next, %middle.latch ]
+ %cmp.k = icmp eq i64 %j, 0
+ br i1 %cmp.k, label %middle.latch, label %inner.body
+
+inner.body:
+ %k = phi i64 [ 0, %middle.header ], [ %k.next, %inner.body ]
+ %mul.n.n = mul nuw i64 %N, %N
+ %mul.i = mul nuw i64 %i, %mul.n.n
+ %mul.j = mul nuw i64 %j, %N
+ %add.j = add nuw i64 %mul.i, %mul.j
+ %idx = add nuw i64 %add.j, %k
+ %arrayidx.b = getelementptr inbounds i32, ptr %b, i64 %idx
+ %0 = load i32, ptr %arrayidx.b, align 4
+ %arrayidx.a = getelementptr inbounds i32, ptr %a, i64 %idx
+ store i32 %0, ptr %arrayidx.a, align 4
+ %k.next = add nuw i64 %k, 1
+ %exitcond.k = icmp eq i64 %k.next, %j
+ br i1 %exitcond.k, label %middle.latch, label %inner.body
+
+middle.latch:
+ %j.next = add nuw i64 %j, 1
+ %exitcond.j = icmp eq i64 %j.next, %N
+ br i1 %exitcond.j, label %outer.latch, label %middle.header
+
+outer.latch:
+ %i.next = add nuw i64 %i, 1
+ %exitcond.i = icmp eq i64 %i.next, %M
+ br i1 %exitcond.i, label %exit, label %outer.header, !llvm.loop !6
+
+exit:
+ ret void
+}
+
!6 = distinct !{!6, !7, !8}
!7 = !{!"llvm.loop.vectorize.width", i32 8}
!8 = !{!"llvm.loop.vectorize.enable"}
diff --git a/llvm/test/Transforms/LoopVectorize/outer_loop_early_exit.ll b/llvm/test/Transforms/LoopVectorize/outer_loop_early_exit.ll
index a5442f4cdb7f0..91903ad7587d0 100644
--- a/llvm/test/Transforms/LoopVectorize/outer_loop_early_exit.ll
+++ b/llvm/test/Transforms/LoopVectorize/outer_loop_early_exit.ll
@@ -2,7 +2,7 @@
; RUN: opt -S -passes=loop-vectorize -pass-remarks-analysis=loop-vectorize -enable-vplan-native-path -disable-output -debug 2>&1 < %s | FileCheck %s
; CHECK-LABEL: LV: Found a loop: for.body
-; CHECK: LV: Not vectorizing: Unsupported conditional branch.
+; CHECK: LV: Not vectorizing: Outer loop contains divergent conditional branch.
; CHECK: loop not vectorized: loop control flow is not understood by vectorizer
; CHECK: LV: Not vectorizing: Unsupported outer loop.
diff --git a/llvm/test/Transforms/LoopVectorize/outer_loop_inner_loop_exits.ll b/llvm/test/Transforms/LoopVectorize/outer_loop_inner_loop_exits.ll
index 45e5314f313fa..2308042a4b6b9 100644
--- a/llvm/test/Transforms/LoopVectorize/outer_loop_inner_loop_exits.ll
+++ b/llvm/test/Transforms/LoopVectorize/outer_loop_inner_loop_exits.ll
@@ -91,6 +91,40 @@ exit:
ret void
}
+; The inner loop's exit condition depends on the outer loop's induction
+; variable, so its backedge branch is divergent across outer-loop iterations
+; and the outer loop must not be vectorized.
+define void @inner_loop_divergence_exit(ptr %a, i64 %N) {
+; CHECK-LABEL: LV: Checking a loop in 'inner_loop_divergence_exit'
+; CHECK: LV: Not vectorizing: Outer loop contains divergent conditional branch.
+; CHECK: LV: Not vectorizing: Unsupported outer loop.
+; CHECK: LV: Not vectorizing: Cannot prove legality.
+entry:
+ br label %outer.header
+
+outer.header:
+ %outer.iv = phi i64 [ 0, %entry ], [ %inc6, %outer.latch ]
+ %invariant.gep = getelementptr [4 x i8], ptr %a, i64 %outer.iv
+ br label %inner.body
+
+inner.body:
+ %inner.iv = phi i64 [ %outer.iv, %outer.header ], [ %inc, %inner.body ]
+ %mul = mul i64 %inner.iv, %N
+ %gep = getelementptr [4 x i8], ptr %invariant.gep, i64 %mul
+ store i32 0, ptr %gep
+ %inc = add nuw i64 %inner.iv, 1
+ %exitcond.not = icmp eq i64 %inc, %N
+ br i1 %exitcond.not, label %outer.latch, label %inner.body
+
+outer.latch:
+ %inc6 = add nuw i64 %outer.iv, 1
+ %exitcond18.not = icmp eq i64 %inc6, %N
+ br i1 %exitcond18.not, label %exit, label %outer.header, !llvm.loop !0
+
+exit:
+ ret void
+}
+
!0 = distinct !{!0, !1, !2}
!1 = !{!"llvm.loop.vectorize.width", i32 4}
!2 = !{!"llvm.loop.vectorize.enable"}
More information about the llvm-commits
mailing list