[llvm] [DA] Refine identical SameSD addrec dependences (PR #206425)

via llvm-commits llvm-commits at lists.llvm.org
Mon Jun 29 01:47:03 PDT 2026


https://github.com/XiaShark updated https://github.com/llvm/llvm-project/pull/206425

>From b92433a652f295793963d29ac498ef7ec7495fdd Mon Sep 17 00:00:00 2001
From: XiaShark <xiajingze1 at huawei.com>
Date: Mon, 29 Jun 2026 14:08:37 +0800
Subject: [PATCH] [DA] Refine identical SameSD addrec dependences

LoopFusion now uses DependenceAnalysis by default. In a case with two
adjacent loops over the same iteration space, DA can fail to classify a
store/load pair as SIV even though both subscripts are identical addrecs on
SameSD loops.

The missed case has equivalent address recurrences in sibling loops:

  SrcSCEV = {A,+,4}<L0>
  DstSCEV = {A,+,4}<L1>

where L0 and L1 are sibling loops with the same trip count and depth. The
generic subscript classifier requires addrecs accepted by checkSubscript().
Depending on how SCEV derives nowrap flags, one side may not carry a signed
no-wrap flag even when the corresponding induction increment has nsw. As a
result, the pair is classified as NonLinear before DA can recognize that the
two SameSD recurrences are identical, and LoopFusion rejects a safe fusion.

Teach DA to recognize identical addrecs in SameSD loops as SIV directly when
they have the same start SCEV, their steps match, and their start/step
expressions are loop-invariant for the corresponding loop nests.

Once classified as SIV, refine identical recurrences in Strong SIV to an equal
direction. They only have same-iteration dependences, so after fusion the
store still precedes the load in the fused iteration and there is no backward
loop-carried dependence.

Add a LoopFusion regression test mirroring the motivating shape, with an
additional pointer recurrence and readonly call in the first loop. The checks
are generated with update_test_checks.py.
---
 llvm/lib/Analysis/DependenceAnalysis.cpp      | 29 +++++++
 .../LoopFusion/da-conservative-direction.ll   | 76 +++++++++++++++++++
 2 files changed, 105 insertions(+)
 create mode 100644 llvm/test/Transforms/LoopFusion/da-conservative-direction.ll

diff --git a/llvm/lib/Analysis/DependenceAnalysis.cpp b/llvm/lib/Analysis/DependenceAnalysis.cpp
index 9d5a555fb8998..c0a4625d8176c 100644
--- a/llvm/lib/Analysis/DependenceAnalysis.cpp
+++ b/llvm/lib/Analysis/DependenceAnalysis.cpp
@@ -842,6 +842,25 @@ DependenceInfo::classifyPair(const SCEV *Src, const Loop *SrcLoopNest,
                              SmallBitVector &Loops) {
   SmallBitVector SrcLoops(MaxLevels + 1);
   SmallBitVector DstLoops(MaxLevels + 1);
+  const SCEVAddRecExpr *SrcAddRec = dyn_cast<SCEVAddRecExpr>(Src);
+  const SCEVAddRecExpr *DstAddRec = dyn_cast<SCEVAddRecExpr>(Dst);
+  if (SrcAddRec && DstAddRec &&
+      SrcAddRec->getStart() == DstAddRec->getStart() &&
+      SrcAddRec->getStepRecurrence(*SE) == DstAddRec->getStepRecurrence(*SE) &&
+      SrcAddRec->getLoop() != DstAddRec->getLoop() &&
+      haveSameSD(SrcAddRec->getLoop(), DstAddRec->getLoop()) &&
+      isLoopInvariant(SrcAddRec->getStart(), SrcLoopNest) &&
+      isLoopInvariant(DstAddRec->getStart(), DstLoopNest) &&
+      isLoopInvariant(SrcAddRec->getStepRecurrence(*SE), SrcLoopNest) &&
+      isLoopInvariant(DstAddRec->getStepRecurrence(*SE), DstLoopNest)) {
+    SrcLoops.set(mapSrcLoop(SrcAddRec->getLoop()));
+    DstLoops.set(mapDstLoop(DstAddRec->getLoop()));
+    Loops = SrcLoops;
+    Loops |= DstLoops;
+    assert(Loops.count() == 1 && "Expected one SameSD level");
+    return Subscript::SIV;
+  }
+
   if (!checkSrcSubscript(Src, SrcLoopNest, SrcLoops))
     return Subscript::NonLinear;
   if (!checkDstSubscript(Dst, DstLoopNest, DstLoops))
@@ -984,8 +1003,18 @@ bool DependenceInfo::strongSIVtest(const SCEVAddRecExpr *Src,
   LLVM_DEBUG(dbgs() << ", " << *DstConst->getType() << "\n");
   ++StrongSIVapplications;
   assert(0 < Level && Level <= CommonLevels && "level out of range");
+  bool IsSameSDLevel = SameSDLevels > 0 && Level > CommonLevels - SameSDLevels;
   Level--;
 
+  // Identical recurrences only have same-iteration dependences. This is useful
+  // for SameSD loops, where the addrecs are in different loops but share the
+  // same start and step.
+  if (IsSameSDLevel && SrcConst == DstConst) {
+    Result.DV[Level].Direction &= Dependence::DVEntry::EQ;
+    ++StrongSIVsuccesses;
+    return false;
+  }
+
   const SCEV *Delta = minusSCEVNoSignedOverflow(SrcConst, DstConst, *SE);
   if (!Delta)
     return false;
diff --git a/llvm/test/Transforms/LoopFusion/da-conservative-direction.ll b/llvm/test/Transforms/LoopFusion/da-conservative-direction.ll
new file mode 100644
index 0000000000000..83c49186d9d68
--- /dev/null
+++ b/llvm/test/Transforms/LoopFusion/da-conservative-direction.ll
@@ -0,0 +1,76 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -S -passes=loop-fusion < %s | FileCheck %s
+
+; DA used to classify the store/load dependence as non-linear even though the
+; subscripts are identical addrecs in SameSD loops. This shape intentionally
+; mirrors a loop nest where the first loop has another pointer recurrence and a
+; readonly call; simply adding nsw to the induction increments is not enough for
+; trunk to fuse it.
+define i64 @fuse_identical_samesd_addrecs(ptr %a, ptr %x, ptr %y, i64 %stride, i64 %n) {
+; CHECK-LABEL: define i64 @fuse_identical_samesd_addrecs(
+; CHECK-SAME: ptr [[A:%.*]], ptr [[X:%.*]], ptr [[Y:%.*]], i64 [[STRIDE:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[CMP0:%.*]] = icmp eq i64 [[N]], 0
+; CHECK-NEXT:    br i1 [[CMP0]], label %[[EXIT:.*]], label %[[FOR_1_PREHEADER:.*]]
+; CHECK:       [[FOR_1_PREHEADER]]:
+; CHECK-NEXT:    br label %[[FOR_1:.*]]
+; CHECK:       [[FOR_1]]:
+; CHECK-NEXT:    [[I:%.*]] = phi i64 [ [[I_NEXT:%.*]], %[[FOR_1]] ], [ 0, %[[FOR_1_PREHEADER]] ]
+; CHECK-NEXT:    [[P:%.*]] = phi ptr [ [[P_NEXT:%.*]], %[[FOR_1]] ], [ [[Y]], %[[FOR_1_PREHEADER]] ]
+; CHECK-NEXT:    [[J:%.*]] = phi i64 [ [[J_NEXT:%.*]], %[[FOR_1]] ], [ 0, %[[FOR_1_PREHEADER]] ]
+; CHECK-NEXT:    [[BEST:%.*]] = phi float [ [[BEST_NEXT:%.*]], %[[FOR_1]] ], [ +inf, %[[FOR_1_PREHEADER]] ]
+; CHECK-NEXT:    [[IDX:%.*]] = phi i64 [ [[IDX_NEXT:%.*]], %[[FOR_1]] ], [ 0, %[[FOR_1_PREHEADER]] ]
+; CHECK-NEXT:    [[CALL:%.*]] = tail call float @readonly_call(ptr [[X]], ptr [[P]], i64 [[STRIDE]])
+; CHECK-NEXT:    [[GEP_A_1:%.*]] = getelementptr inbounds nuw float, ptr [[A]], i64 [[I]]
+; CHECK-NEXT:    store float [[CALL]], ptr [[GEP_A_1]], align 4
+; CHECK-NEXT:    [[P_NEXT]] = getelementptr inbounds nuw float, ptr [[P]], i64 [[STRIDE]]
+; CHECK-NEXT:    [[I_NEXT]] = add nsw i64 [[I]], 1
+; CHECK-NEXT:    [[DONE_1:%.*]] = icmp eq i64 [[I_NEXT]], [[N]]
+; CHECK-NEXT:    [[GEP_A_2:%.*]] = getelementptr inbounds nuw float, ptr [[A]], i64 [[J]]
+; CHECK-NEXT:    [[V:%.*]] = load float, ptr [[GEP_A_2]], align 4
+; CHECK-NEXT:    [[CMP:%.*]] = fcmp olt float [[V]], [[BEST]]
+; CHECK-NEXT:    [[IDX_NEXT]] = select i1 [[CMP]], i64 [[J]], i64 [[IDX]]
+; CHECK-NEXT:    [[BEST_NEXT]] = select i1 [[CMP]], float [[V]], float [[BEST]]
+; CHECK-NEXT:    [[J_NEXT]] = add nsw i64 [[J]], 1
+; CHECK-NEXT:    [[DONE_2:%.*]] = icmp eq i64 [[J_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[DONE_2]], label %[[EXIT_LOOPEXIT:.*]], label %[[FOR_1]]
+; CHECK:       [[EXIT_LOOPEXIT]]:
+; CHECK-NEXT:    br label %[[EXIT]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    [[RES:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IDX_NEXT]], %[[EXIT_LOOPEXIT]] ]
+; CHECK-NEXT:    ret i64 [[RES]]
+;
+entry:
+  %cmp0 = icmp eq i64 %n, 0
+  br i1 %cmp0, label %exit, label %for.1
+
+for.1:
+  %i = phi i64 [ 0, %entry ], [ %i.next, %for.1 ]
+  %p = phi ptr [ %y, %entry ], [ %p.next, %for.1 ]
+  %call = tail call float @readonly_call(ptr %x, ptr %p, i64 %stride)
+  %gep.a.1 = getelementptr inbounds nuw float, ptr %a, i64 %i
+  store float %call, ptr %gep.a.1, align 4
+  %p.next = getelementptr inbounds nuw float, ptr %p, i64 %stride
+  %i.next = add nsw i64 %i, 1
+  %done.1 = icmp eq i64 %i.next, %n
+  br i1 %done.1, label %for.2, label %for.1
+
+for.2:
+  %j = phi i64 [ 0, %for.1 ], [ %j.next, %for.2 ]
+  %best = phi float [ 0x7FF0000000000000, %for.1 ], [ %best.next, %for.2 ]
+  %idx = phi i64 [ 0, %for.1 ], [ %idx.next, %for.2 ]
+  %gep.a.2 = getelementptr inbounds nuw float, ptr %a, i64 %j
+  %v = load float, ptr %gep.a.2, align 4
+  %cmp = fcmp olt float %v, %best
+  %idx.next = select i1 %cmp, i64 %j, i64 %idx
+  %best.next = select i1 %cmp, float %v, float %best
+  %j.next = add nsw i64 %j, 1
+  %done.2 = icmp eq i64 %j.next, %n
+  br i1 %done.2, label %exit, label %for.2
+
+exit:
+  %res = phi i64 [ 0, %entry ], [ %idx.next, %for.2 ]
+  ret i64 %res
+}
+
+declare float @readonly_call(ptr, ptr, i64) nounwind readonly



More information about the llvm-commits mailing list