[llvm] c4ff7a7 - [LV] Use SCEV loop-uniformity for outer-loop branch legality (#199632)

via llvm-commits llvm-commits at lists.llvm.org
Wed Sep 30 20:48:46 PDT 2026


Author: Ming Yan
Date: 2026-10-01T03:48:36Z
New Revision: c4ff7a767a271c02dd37312999c80f404219dd30

URL: https://github.com/llvm/llvm-project/commit/c4ff7a767a271c02dd37312999c80f404219dd30
DIFF: https://github.com/llvm/llvm-project/commit/c4ff7a767a271c02dd37312999c80f404219dd30.diff

LOG: [LV] Use SCEV loop-uniformity for outer-loop branch legality (#199632)

This patch refactors the outer-loop vectorization branch legality checks
to reason about conditional branches directly instead of using the old
recursive inner-loop shape check.

The new check allows conditional branches when their condition is
either:

- loop-invariant with respect to the vectorized outer loop, or
- a compare whose operands are both SCEV loop-uniform with respect to
the vectorized outer loop.

Divergent conditional branches are still rejected, now with a more
specific diagnostic.

Added: 
    

Modified: 
    llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp
    llvm/test/Transforms/LoopVectorize/explicit_outer_nonuniform_inner.ll
    llvm/test/Transforms/LoopVectorize/explicit_outer_uniform_diverg_branch.ll
    llvm/test/Transforms/LoopVectorize/outer_loop_early_exit.ll
    llvm/test/Transforms/LoopVectorize/outer_loop_inner_loop_exits.ll

Removed: 
    


################################################################################
diff  --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp
index 7c9baf763b8cf..8d58ef9ed9c4f 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp
@@ -324,86 +324,6 @@ void LoopVectorizeHints::setHint(StringRef Name, Metadata *Arg) {
   }
 }
 
-// Return true if the inner loop \p Lp is uniform with regard to the outer loop
-// \p OuterLp (i.e., if the outer loop is vectorized, all the vector lanes
-// executing the inner loop will execute the same iterations). This check is
-// very constrained for now but it will be relaxed in the future. \p Lp is
-// considered uniform if it meets all the following conditions:
-//   1) it has a canonical IV (starting from 0 and with stride 1),
-//   2) its latch terminator is a conditional branch and,
-//   3) its latch condition is a compare instruction whose operands are the
-//      canonical IV and an OuterLp invariant.
-// This check doesn't take into account the uniformity of other conditions not
-// related to the loop latch because they don't affect the loop uniformity.
-//
-// NOTE: We decided to keep all these checks and its associated documentation
-// together so that we can easily have a picture of the current supported loop
-// nests. However, some of the current checks don't depend on \p OuterLp and
-// would be redundantly executed for each \p Lp if we invoked this function for
-// 
diff erent candidate outer loops. This is not the case for now because we
-// don't currently have the infrastructure to evaluate multiple candidate outer
-// loops and \p OuterLp will be a fixed parameter while we only support explicit
-// outer loop vectorization. It's also very likely that these checks go away
-// before introducing the aforementioned infrastructure. However, if this is not
-// the case, we should move the \p OuterLp independent checks to a separate
-// function that is only executed once for each \p Lp.
-static bool isUniformLoop(Loop *Lp, Loop *OuterLp) {
-  assert(Lp->getLoopLatch() && "Expected loop with a single latch.");
-
-  // If Lp is the outer loop, it's uniform by definition.
-  if (Lp == OuterLp)
-    return true;
-  assert(OuterLp->contains(Lp) && "OuterLp must contain Lp.");
-
-  // 1.
-  PHINode *IV = Lp->getCanonicalInductionVariable();
-  if (!IV) {
-    LLVM_DEBUG(dbgs() << "LV: Canonical IV not found.\n");
-    return false;
-  }
-
-  // 2.
-  BasicBlock *Latch = Lp->getLoopLatch();
-  auto *LatchBr = dyn_cast<CondBrInst>(Latch->getTerminator());
-  if (!LatchBr) {
-    LLVM_DEBUG(dbgs() << "LV: Unsupported loop latch branch.\n");
-    return false;
-  }
-
-  // 3.
-  auto *LatchCmp = dyn_cast<CmpInst>(LatchBr->getCondition());
-  if (!LatchCmp) {
-    LLVM_DEBUG(
-        dbgs() << "LV: Loop latch condition is not a compare instruction.\n");
-    return false;
-  }
-
-  Value *CondOp0 = LatchCmp->getOperand(0);
-  Value *CondOp1 = LatchCmp->getOperand(1);
-  Value *IVUpdate = IV->getIncomingValueForBlock(Latch);
-  if (!(CondOp0 == IVUpdate && OuterLp->isLoopInvariant(CondOp1)) &&
-      !(CondOp1 == IVUpdate && OuterLp->isLoopInvariant(CondOp0))) {
-    LLVM_DEBUG(dbgs() << "LV: Loop latch condition is not uniform.\n");
-    return false;
-  }
-
-  return true;
-}
-
-// Return true if \p Lp and all its nested loops are uniform with regard to \p
-// OuterLp.
-static bool isUniformLoopNest(Loop *Lp, Loop *OuterLp) {
-  if (!isUniformLoop(Lp, OuterLp))
-    return false;
-
-  // Check if nested loops are uniform.
-  for (Loop *SubLp : *Lp)
-    if (!isUniformLoopNest(SubLp, OuterLp))
-      return false;
-
-  return true;
-}
-
 static IntegerType *getInductionIntegerTy(const DataLayout &DL, Type *Ty) {
   assert(Ty->isIntOrPtrTy() && "Expected integer or pointer type");
 
@@ -652,30 +572,46 @@ bool LoopVectorizationLegality::canVectorizeOuterLoop() {
     }
 
     // Check whether the branch is a supported one. Only unconditional
-    // branches, conditional branches with an outer loop invariant condition or
+    // branches, conditional branches with an outer loop uniform condition or
     // backedges are supported.
     // FIXME: We skip these checks when VPlan predication is enabled as we
     // want to allow divergent branches. This whole check will be removed
     // once VPlan predication is on by default.
     auto *Br = dyn_cast<CondBrInst>(Term);
-    if (Br && !TheLoop->isLoopInvariant(Br->getCondition()) &&
-        !LI->isLoopHeader(Br->getSuccessor(0)) &&
-        !LI->isLoopHeader(Br->getSuccessor(1))) {
-      reportVectorizationFailure(
-          "Unsupported conditional branch",
-          "loop control flow is not understood by vectorizer",
-          "CFGNotUnderstood", ORE, TheLoop);
-      if (DoExtraAnalysis)
-        Result = false;
-      else
-        return false;
+    if (Br && !TheLoop->isLoopLatch(BB)) {
+      bool IsUniformCondBr = TheLoop->isLoopInvariant(Br->getCondition());
+
+      Value *Lhs = nullptr;
+      Value *Rhs = nullptr;
+      auto *SE = PSE.getSE();
+      if (match(Br->getCondition(), m_c_ICmp(m_Value(Lhs), m_Value(Rhs))) &&
+          !IsUniformCondBr && SE->isSCEVable(Lhs->getType())) {
+        const SCEV *LhsExpr = PSE.getSCEV(Lhs);
+        const SCEV *RhsExpr = PSE.getSCEV(Rhs);
+        IsUniformCondBr |= (SE->isLoopUniform(LhsExpr, TheLoop) &&
+                            SE->isLoopUniform(RhsExpr, TheLoop));
+      }
+
+      // If the condition is not uniform, report a failure. We currently require
+      // uniform conditions to avoid the complexity of vectorizing divergent
+      // control flow in the outer loop.
+      if (!IsUniformCondBr) {
+        reportVectorizationFailure(
+            "Outer loop contains divergent conditional branch",
+            "loop control flow is not understood by vectorizer",
+            "CFGNotUnderstood", ORE, TheLoop);
+        if (DoExtraAnalysis)
+          Result = false;
+        else
+          return false;
+      }
     }
   }
 
   // Each nested loop must exit via its latch only, as a region with the latch
   // as its only exiting block is created for it. Note that the branch check
-  // above rejects divergent exits, but exits with an outer-loop invariant
-  // condition are allowed through.
+  // rejects divergent exits, but exits with an outer-loop uniform condition
+  // are allowed through.
   SmallVector<Loop *, 4> LoopNest = TheLoop->getLoopsInPreorder();
   for (Loop *Lp : drop_begin(LoopNest)) {
     if (Lp->getExitingBlock() != Lp->getLoopLatch()) {
@@ -690,20 +626,6 @@ bool LoopVectorizationLegality::canVectorizeOuterLoop() {
     }
   }
 
-  // Check whether inner loops are uniform. At this point, we only support
-  // simple outer loops scenarios with uniform nested loops.
-  if (!isUniformLoopNest(TheLoop /*loop nest*/,
-                         TheLoop /*context outer loop*/)) {
-    reportVectorizationFailure(
-        "Outer loop contains divergent loops",
-        "loop control flow is not understood by vectorizer", "CFGNotUnderstood",
-        ORE, TheLoop);
-    if (DoExtraAnalysis)
-      Result = false;
-    else
-      return false;
-  }
-
   // Check whether we are able to set up outer loop induction.
   if (!setupOuterLoopInductions()) {
     reportVectorizationFailure("Unsupported outer loop Phi(s)",

diff  --git a/llvm/test/Transforms/LoopVectorize/explicit_outer_nonuniform_inner.ll b/llvm/test/Transforms/LoopVectorize/explicit_outer_nonuniform_inner.ll
index 9c87b9cb1e818..61acce96ad283 100644
--- a/llvm/test/Transforms/LoopVectorize/explicit_outer_nonuniform_inner.ll
+++ b/llvm/test/Transforms/LoopVectorize/explicit_outer_nonuniform_inner.ll
@@ -20,7 +20,7 @@
 ; Case 1 (for (j = i; j < M; j++)): Inner loop with divergent IV start.
 
 ; CHECK-LABEL: iv_start
-; CHECK: LV: Not vectorizing: Outer loop contains divergent loops.
+; CHECK: LV: Not vectorizing: Outer loop contains divergent conditional branch.
 ; CHECK: LV: Not vectorizing: Unsupported outer loop.
 
 target datalayout = "e-m:e-i64:64-f80:128-n8:16:32:64-S128"
@@ -70,7 +70,7 @@ for.end15:
 ; Case 2 (for (j = 0; j < i; j++)): Inner loop with divergent upper-bound.
 
 ; CHECK-LABEL: loop_ub
-; CHECK: LV: Not vectorizing: Outer loop contains divergent loops.
+; CHECK: LV: Not vectorizing: Outer loop contains divergent conditional branch.
 ; CHECK: LV: Not vectorizing: Unsupported outer loop.
 
 define void @loop_ub(ptr nocapture %a, ptr nocapture readonly %b, i32 %N, i32 %M) {
@@ -116,7 +116,7 @@ for.end15:
 ; Case 3 (for (j = 0; j < M; j+=i)): Inner loop with divergent step.
 
 ; CHECK-LABEL: iv_step
-; CHECK: LV: Not vectorizing: Outer loop contains divergent loops.
+; CHECK: LV: Not vectorizing: Outer loop contains divergent conditional branch.
 ; CHECK: LV: Not vectorizing: Unsupported outer loop.
 
 define void @iv_step(ptr nocapture %a, ptr nocapture readonly %b, i32 %N, i32 %M) {

diff  --git a/llvm/test/Transforms/LoopVectorize/explicit_outer_uniform_diverg_branch.ll b/llvm/test/Transforms/LoopVectorize/explicit_outer_uniform_diverg_branch.ll
index 75858e626b728..56d5ea6b66f6f 100644
--- a/llvm/test/Transforms/LoopVectorize/explicit_outer_uniform_diverg_branch.ll
+++ b/llvm/test/Transforms/LoopVectorize/explicit_outer_uniform_diverg_branch.ll
@@ -4,7 +4,7 @@
 ; Verify that LV can handle explicit vectorization outer loops with uniform branches
 ; but bails out on outer loops with divergent branches.
 
-; Root C/C++ source code for the test cases
+; Root C/C++ source code for the first two test cases.
 ; void foo(int *a, int *b, int N, int M)
 ; {
 ;   int i, j;
@@ -74,7 +74,7 @@ for.end19:
 ; Case 2 (COND => B[i * M] == 0): Outer loop with divergent conditional branch.
 
 ; CHECK-LABEL: divergent_branch
-; CHECK: Unsupported conditional branch.
+; CHECK: LV: Not vectorizing: Outer loop contains divergent conditional branch.
 ; CHECK: LV: Not vectorizing: Unsupported outer loop.
 
 define void @divergent_branch(ptr nocapture %a, ptr nocapture readonly %b, i32 %N, i32 %M) {
@@ -122,6 +122,62 @@ for.end19:
   ret void
 }
 
+; Case 3: Three-level loop nest with a triangular innermost loop.
+;
+;   #pragma clang loop vectorize(enable) vectorize_width(8)
+;   for (size_t i = 0; i < M; ++i)
+;     for (size_t j = 0; j < N; ++j)
+;       for (size_t k = 0; k < j; ++k)
+;         a[i * N * N + j * N + k] = b[i * N * N + j * N + k];
+;
+; The innermost loop latch condition depends on the middle loop IV, but is
+; uniform with respect to the vectorized outer loop.
+
+; CHECK-LABEL: uniform_triangular_inner_loop
+; CHECK: LV: We can vectorize this outer loop!
+
+define void @uniform_triangular_inner_loop(ptr nocapture %a, ptr nocapture readonly %b, i64 %M, i64 %N) {
+entry:
+  br label %outer.header
+
+outer.header:
+  %i = phi i64 [ 0, %entry ], [ %i.next, %outer.latch ]
+  br label %middle.header
+
+middle.header:
+  %j = phi i64 [ 0, %outer.header ], [ %j.next, %middle.latch ]
+  %cmp.k = icmp eq i64 %j, 0
+  br i1 %cmp.k, label %middle.latch, label %inner.body
+
+inner.body:
+  %k = phi i64 [ 0, %middle.header ], [ %k.next, %inner.body ]
+  %mul.n.n = mul nuw i64 %N, %N
+  %mul.i = mul nuw i64 %i, %mul.n.n
+  %mul.j = mul nuw i64 %j, %N
+  %add.j = add nuw i64 %mul.i, %mul.j
+  %idx = add nuw i64 %add.j, %k
+  %arrayidx.b = getelementptr inbounds i32, ptr %b, i64 %idx
+  %0 = load i32, ptr %arrayidx.b, align 4
+  %arrayidx.a = getelementptr inbounds i32, ptr %a, i64 %idx
+  store i32 %0, ptr %arrayidx.a, align 4
+  %k.next = add nuw i64 %k, 1
+  %exitcond.k = icmp eq i64 %k.next, %j
+  br i1 %exitcond.k, label %middle.latch, label %inner.body
+
+middle.latch:
+  %j.next = add nuw i64 %j, 1
+  %exitcond.j = icmp eq i64 %j.next, %N
+  br i1 %exitcond.j, label %outer.latch, label %middle.header
+
+outer.latch:
+  %i.next = add nuw i64 %i, 1
+  %exitcond.i = icmp eq i64 %i.next, %M
+  br i1 %exitcond.i, label %exit, label %outer.header, !llvm.loop !6
+
+exit:
+  ret void
+}
+
 !6 = distinct !{!6, !7, !8}
 !7 = !{!"llvm.loop.vectorize.width", i32 8}
 !8 = !{!"llvm.loop.vectorize.enable"}

diff  --git a/llvm/test/Transforms/LoopVectorize/outer_loop_early_exit.ll b/llvm/test/Transforms/LoopVectorize/outer_loop_early_exit.ll
index a5442f4cdb7f0..91903ad7587d0 100644
--- a/llvm/test/Transforms/LoopVectorize/outer_loop_early_exit.ll
+++ b/llvm/test/Transforms/LoopVectorize/outer_loop_early_exit.ll
@@ -2,7 +2,7 @@
 ; RUN: opt -S -passes=loop-vectorize -pass-remarks-analysis=loop-vectorize -enable-vplan-native-path -disable-output -debug 2>&1 < %s | FileCheck %s
 
 ; CHECK-LABEL: LV: Found a loop: for.body
-; CHECK: LV: Not vectorizing: Unsupported conditional branch.
+; CHECK: LV: Not vectorizing: Outer loop contains divergent conditional branch.
 ; CHECK: loop not vectorized: loop control flow is not understood by vectorizer
 ; CHECK: LV: Not vectorizing: Unsupported outer loop.
 

diff  --git a/llvm/test/Transforms/LoopVectorize/outer_loop_inner_loop_exits.ll b/llvm/test/Transforms/LoopVectorize/outer_loop_inner_loop_exits.ll
index 45e5314f313fa..2308042a4b6b9 100644
--- a/llvm/test/Transforms/LoopVectorize/outer_loop_inner_loop_exits.ll
+++ b/llvm/test/Transforms/LoopVectorize/outer_loop_inner_loop_exits.ll
@@ -91,6 +91,40 @@ exit:
   ret void
 }
 
+; The inner loop's exit condition depends on the outer loop's induction
+; variable, so its backedge branch is divergent across outer-loop iterations
+; and the outer loop must not be vectorized.
+define void @inner_loop_divergence_exit(ptr %a, i64 %N) {
+; CHECK-LABEL: LV: Checking a loop in 'inner_loop_divergence_exit'
+; CHECK: LV: Not vectorizing: Outer loop contains divergent conditional branch.
+; CHECK: LV: Not vectorizing: Unsupported outer loop.
+; CHECK: LV: Not vectorizing: Cannot prove legality.
+entry:
+  br label %outer.header
+
+outer.header:
+  %outer.iv = phi i64 [ 0, %entry ], [ %inc6, %outer.latch ]
+  %invariant.gep = getelementptr [4 x i8], ptr %a, i64 %outer.iv
+  br label %inner.body
+
+inner.body:
+  %inner.iv = phi i64 [ %outer.iv, %outer.header ], [ %inc, %inner.body ]
+  %mul = mul i64 %inner.iv, %N
+  %gep = getelementptr [4 x i8], ptr %invariant.gep, i64 %mul
+  store i32 0, ptr %gep
+  %inc = add nuw i64 %inner.iv, 1
+  %exitcond.not = icmp eq i64 %inc, %N
+  br i1 %exitcond.not, label %outer.latch, label %inner.body
+
+outer.latch:
+  %inc6 = add nuw i64 %outer.iv, 1
+  %exitcond18.not = icmp eq i64 %inc6, %N
+  br i1 %exitcond18.not, label %exit, label %outer.header, !llvm.loop !0
+
+exit:
+  ret void
+}
+
 !0 = distinct !{!0, !1, !2}
 !1 = !{!"llvm.loop.vectorize.width", i32 4}
 !2 = !{!"llvm.loop.vectorize.enable"}


        


More information about the llvm-commits mailing list