[llvm] [LoopFusion] Fix false-positive dependency blocking fusion of idempot… (PR #206401)
via llvm-commits
llvm-commits at lists.llvm.org
Sun Jun 28 22:39:42 PDT 2026
llvmorg-github-actions[bot] wrote:
<!--LLVM PR SUMMARY COMMENT-->
@llvm/pr-subscribers-llvm-transforms
Author: AntonyCJ30
<details>
<summary>Changes</summary>
Loop Fusion incorrectly blocks fusion of loops that write the same value to different arrays. Dependence Analysis conservatively reports MayAlias for function argument pointers, generating a false-positive store-store dependency that blocks fusion even when it is clearly safe.
Fix
For store-store pairs where both stores write the same value to provably different underlying objects, the dependency is a false positive. Writing the same value twice produces identical observable results regardless of execution order, so fusion is safe.
Two cases are handled:
Same value operand (constants, function arguments, hoisted expressions)
Different value operands but identical loop-invariant SCEV (e.g. val+unknown computed separately in each preheader)
Limitations
IV-dependent expressions like a[i]=i*2+1; b[i]=i*2+1 are not handled — their SCEVs have different loop anchors and require SCEVParameterRewriter for comparison. Left as a TODO for a follow-up.
Testing
Added test cases to loop_invariant.ll covering safe fusion (same argument, same expression) and correct blocking (different values).
Fixes 94676
---
Full diff: https://github.com/llvm/llvm-project/pull/206401.diff
2 Files Affected:
- (modified) llvm/lib/Transforms/Scalar/LoopFuse.cpp (+32-2)
- (modified) llvm/test/Transforms/LoopFusion/loop_invariant.ll (+122)
``````````diff
diff --git a/llvm/lib/Transforms/Scalar/LoopFuse.cpp b/llvm/lib/Transforms/Scalar/LoopFuse.cpp
index 1b83c971c01bf..6502e3a60a814 100644
--- a/llvm/lib/Transforms/Scalar/LoopFuse.cpp
+++ b/llvm/lib/Transforms/Scalar/LoopFuse.cpp
@@ -54,6 +54,7 @@
#include "llvm/Analysis/PostDominators.h"
#include "llvm/Analysis/ScalarEvolution.h"
#include "llvm/Analysis/TargetTransformInfo.h"
+#include "llvm/Analysis/ValueTracking.h"
#include "llvm/IR/Function.h"
#include "llvm/IR/Verifier.h"
#include "llvm/Support/CommandLine.h"
@@ -64,7 +65,6 @@
#include "llvm/Transforms/Utils/LoopPeel.h"
#include "llvm/Transforms/Utils/LoopSimplify.h"
#include <list>
-
using namespace llvm;
#define DEBUG_TYPE "loop-fusion"
@@ -1113,6 +1113,36 @@ struct LoopFuser {
auto DepResult = DI.depends(&I0, &I1);
if (!DepResult)
return true;
+ // Dependence Analysis is conservative for function arguments and may
+ // report MayAlias for pointers that provably refer to different objects.
+ // For store-store pairs writing the same value to different underlying
+ // objects, the dependency is a false positive: fusing is safe because
+ // the observable result is identical regardless of execution order.
+ if (isa<StoreInst>(I0) && isa<StoreInst>(I1)) {
+ auto *S0 = cast<StoreInst>(&I0);
+ auto *S1 = cast<StoreInst>(&I1);
+
+ Value *P0 = getUnderlyingObject(S0->getPointerOperand());
+ Value *P1 = getUnderlyingObject(S1->getPointerOperand());
+
+ if (P0 != P1) {
+ // Case 1: Same Value* (constants, same variables, hoisted exprs)
+ if (S0->getValueOperand() == S1->getValueOperand())
+ return true;
+
+ // Case 2: Loop-invariant expressions with identical SCEV
+ // (e.g. val+unknown computed separately but symbolically equal)
+ // Note: this only works for loop-invariant SCEVs; IV-dependent
+ // expressions like i*2+1 have different loop anchors and won't
+ // match. Those require SCEVParameterRewriter (future work).
+ const SCEV *SCEV0 = SE.getSCEV(S0->getValueOperand());
+ const SCEV *SCEV1 = SE.getSCEV(S1->getValueOperand());
+ if (SCEV0 && SCEV1 && SCEV0 == SCEV1)
+ return true;
+ }
+ }
+ // TODO: Handle structurally identical IV-dependent expressions
+ // (e.g. a[i]=i*2+1; b[i]=i*2+1) using SCEVParameterRewriter.
#ifndef NDEBUG
if (VerboseFusionDebugging) {
LLVM_DEBUG(dbgs() << "DA res: "; DepResult->dump(dbgs());
@@ -1148,7 +1178,6 @@ struct LoopFuser {
}
assert(CurLoopLevel > Levels && "Fusion candidates are not separated");
-
if (DepResult->isScalar(CurLoopLevel, true)) {
if (DepResult->isInput() || DepResult->isOutput()) {
LLVM_DEBUG(dbgs() << "Safe to fuse due to a loop-invariant "
@@ -1157,6 +1186,7 @@ struct LoopFuser {
NumDA++;
return true;
}
+
LLVM_DEBUG(
dbgs() << "Not safe to fuse due to a scalar flow dependency\n");
return false;
diff --git a/llvm/test/Transforms/LoopFusion/loop_invariant.ll b/llvm/test/Transforms/LoopFusion/loop_invariant.ll
index ea682ac363aad..26b63531a9703 100644
--- a/llvm/test/Transforms/LoopFusion/loop_invariant.ll
+++ b/llvm/test/Transforms/LoopFusion/loop_invariant.ll
@@ -60,3 +60,125 @@ body2: ; preds = %pre2, %body2
exit:
ret void
}
+
+; Test idempotent store-store pairs: both loops write the same value to
+; provably different underlying objects. Safe to fuse despite MayAlias.
+
+define void @idempotent_same_arg(ptr %a, ptr %b, i32 %n, i32 %val) {
+; CHECK-DA: Performing Loop Fusion on function idempotent_same_arg
+; CHECK-DA: Fusion is performed
+entry:
+ %cmp = icmp sgt i32 %n, 0
+ br i1 %cmp, label %for.body.preheader, label %for.cond2.preheader
+
+for.body.preheader:
+ %wide.trip.count = zext nneg i32 %n to i64
+ br label %for.body
+
+for.body:
+ %iv = phi i64 [ 0, %for.body.preheader ], [ %iv.next, %for.body ]
+ %gep.a = getelementptr inbounds i32, ptr %a, i64 %iv
+ store i32 %val, ptr %gep.a, align 4
+ %iv.next = add nuw nsw i64 %iv, 1
+ %exit = icmp eq i64 %iv.next, %wide.trip.count
+ br i1 %exit, label %for.cond2.preheader, label %for.body
+
+for.cond2.preheader:
+ %cmp2 = icmp sgt i32 %n, 0
+ br i1 %cmp2, label %for.body5.preheader, label %for.cond.cleanup4
+
+for.body5.preheader:
+ %wide.trip.count2 = zext nneg i32 %n to i64
+ br label %for.body5
+
+for.body5:
+ %iv2 = phi i64 [ 0, %for.body5.preheader ], [ %iv2.next, %for.body5 ]
+ %gep.b = getelementptr inbounds i32, ptr %b, i64 %iv2
+ store i32 %val, ptr %gep.b, align 4
+ %iv2.next = add nuw nsw i64 %iv2, 1
+ %exit2 = icmp eq i64 %iv2.next, %wide.trip.count2
+ br i1 %exit2, label %for.cond.cleanup4, label %for.body5
+
+for.cond.cleanup4:
+ ret void
+}
+
+define void @idempotent_same_expr(ptr %a, ptr %b, i32 %n, i32 %val, i32 %unknown) {
+; CHECK-DA: Performing Loop Fusion on function idempotent_same_expr
+; CHECK-DA: Fusion is performed
+entry:
+ %cmp = icmp sgt i32 %n, 0
+ br i1 %cmp, label %for.body.lr.ph, label %for.cond2.preheader
+
+for.body.lr.ph:
+ %add = add nsw i32 %unknown, %val
+ %wide.trip.count = zext nneg i32 %n to i64
+ br label %for.body
+
+for.body:
+ %iv = phi i64 [ 0, %for.body.lr.ph ], [ %iv.next, %for.body ]
+ %gep.a = getelementptr inbounds i32, ptr %a, i64 %iv
+ store i32 %add, ptr %gep.a, align 4
+ %iv.next = add nuw nsw i64 %iv, 1
+ %exit = icmp eq i64 %iv.next, %wide.trip.count
+ br i1 %exit, label %for.cond2.preheader, label %for.body
+
+for.cond2.preheader:
+ %cmp2 = icmp sgt i32 %n, 0
+ br i1 %cmp2, label %for.body5.lr.ph, label %for.cond.cleanup4
+
+for.body5.lr.ph:
+ %add6 = add nsw i32 %unknown, %val
+ %wide.trip.count2 = zext nneg i32 %n to i64
+ br label %for.body5
+
+for.body5:
+ %iv2 = phi i64 [ 0, %for.body5.lr.ph ], [ %iv2.next, %for.body5 ]
+ %gep.b = getelementptr inbounds i32, ptr %b, i64 %iv2
+ store i32 %add6, ptr %gep.b, align 4
+ %iv2.next = add nuw nsw i64 %iv2, 1
+ %exit2 = icmp eq i64 %iv2.next, %wide.trip.count2
+ br i1 %exit2, label %for.cond.cleanup4, label %for.body5
+
+for.cond.cleanup4:
+ ret void
+}
+
+define void @idempotent_different_values(ptr %a, ptr %b, i32 %n, i32 %t) {
+; CHECK-DA: Performing Loop Fusion on function idempotent_different_values
+; CHECK-DA: Memory dependencies do not allow fusion!
+entry:
+ %cmp = icmp sgt i32 %n, 0
+ br i1 %cmp, label %for.body.preheader, label %for.cond2.preheader
+
+for.body.preheader:
+ %wide.trip.count = zext nneg i32 %n to i64
+ br label %for.body
+
+for.body:
+ %iv = phi i64 [ 0, %for.body.preheader ], [ %iv.next, %for.body ]
+ %gep.a = getelementptr inbounds i32, ptr %a, i64 %iv
+ store i32 %t, ptr %gep.a, align 4
+ %iv.next = add nuw nsw i64 %iv, 1
+ %exit = icmp eq i64 %iv.next, %wide.trip.count
+ br i1 %exit, label %for.cond2.preheader, label %for.body
+
+for.cond2.preheader:
+ %cmp2 = icmp sgt i32 %n, 0
+ br i1 %cmp2, label %for.body5.preheader, label %for.cond.cleanup4
+
+for.body5.preheader:
+ %wide.trip.count2 = zext nneg i32 %n to i64
+ br label %for.body5
+
+for.body5:
+ %iv2 = phi i64 [ 0, %for.body5.preheader ], [ %iv2.next, %for.body5 ]
+ %gep.b = getelementptr inbounds i32, ptr %b, i64 %iv2
+ store i32 42, ptr %gep.b, align 4
+ %iv2.next = add nuw nsw i64 %iv2, 1
+ %exit2 = icmp eq i64 %iv2.next, %wide.trip.count2
+ br i1 %exit2, label %for.cond.cleanup4, label %for.body5
+
+for.cond.cleanup4:
+ ret void
+}
``````````
</details>
https://github.com/llvm/llvm-project/pull/206401
More information about the llvm-commits
mailing list