[llvm] [LICM] Rewrite noConflictingReadWrites to walk MemorySSA graph (PR #191468)

Sebastian Pop via llvm-commits llvm-commits at lists.llvm.org
Fri Apr 10 10:52:45 PDT 2026


https://github.com/sebpop updated https://github.com/llvm/llvm-project/pull/191468

>From d121e4320f45dd36846281eadb8830969a9ca0fa Mon Sep 17 00:00:00 2001
From: Sebastian Pop <spop at nvidia.com>
Date: Fri, 10 Apr 2026 10:53:24 -0500
Subject: [PATCH] [LICM] Rewrite noConflictingReadWrites to walk MemorySSA
 graph

Instead of scanning all memory accesses in the loop and checking each
one with a clobber walk, traverse the MemorySSA use-def graph starting
from the loop header's MemoryPhi. In optimized MemorySSA, non-aliasing
MemoryUses have their defining access redirected outside the loop, so
they naturally won't appear as users of loop-internal accesses.

This removes: the O(all accesses) block scan, the tooManyMemoryAccesses
bailout, the incorrect dominance check (PR #187529), and the FIXME
about imprecise alias checking for MemoryUses.
---
 llvm/lib/Transforms/Scalar/LICM.cpp        | 107 ++++++++++++---------
 llvm/test/Analysis/MemorySSA/pr43427.ll    |  16 +--
 llvm/test/Transforms/LICM/call-hoisting.ll |  29 ++++++
 llvm/test/Transforms/LICM/pr54495.ll       |   2 +-
 4 files changed, 100 insertions(+), 54 deletions(-)

diff --git a/llvm/lib/Transforms/Scalar/LICM.cpp b/llvm/lib/Transforms/Scalar/LICM.cpp
index 733c35ec797cd..88a83301171d2 100644
--- a/llvm/lib/Transforms/Scalar/LICM.cpp
+++ b/llvm/lib/Transforms/Scalar/LICM.cpp
@@ -2302,15 +2302,15 @@ collectPromotionCandidates(MemorySSA *MSSA, AliasAnalysis *AA, Loop *L) {
 
 // For a given store instruction or writeonly call instruction, this function
 // checks that there are no read or writes that conflict with the memory
-// access in the instruction
+// access in the instruction.  Instead of scanning all memory accesses in the
+// loop, walk the MemorySSA use-def graph starting from the loop header's
+// MemoryPhi.  This visits only MemoryDefs and MemoryUses that are reachable
+// through loop-internal accesses, skipping uses that have already been
+// optimized to point outside the loop.
 static bool noConflictingReadWrites(Instruction *I, MemorySSA *MSSA,
                                     AAResults *AA, Loop *CurLoop,
                                     SinkAndHoistLICMFlags &Flags) {
   assert(isa<CallInst>(*I) || isa<StoreInst>(*I));
-  // If there are more accesses than the Promotion cap, then give up as we're
-  // not walking a list that long.
-  if (Flags.tooManyMemoryAccesses())
-    return false;
 
   auto *IMD = MSSA->getMemoryAccess(I);
   BatchAAResults BAA(*AA);
@@ -2319,54 +2319,71 @@ static bool noConflictingReadWrites(Instruction *I, MemorySSA *MSSA,
   if (!MSSA->isLiveOnEntryDef(Source) && CurLoop->contains(Source->getBlock()))
     return false;
 
-  // If there are interfering Uses (i.e. their defining access is in the
-  // loop), or ordered loads (stored as Defs!), don't move this store.
-  // Could do better here, but this is conservatively correct.
-  // TODO: Cache set of Uses on the first walk in runOnLoop, update when
-  // moving accesses. Can also extend to dominating uses.
-  for (auto *BB : CurLoop->getBlocks()) {
-    auto *Accesses = MSSA->getBlockAccesses(BB);
-    if (!Accesses)
-      continue;
-    for (const auto &MA : *Accesses)
-      if (const auto *MU = dyn_cast<MemoryUse>(&MA)) {
-        auto *MD = getClobberingMemoryAccess(*MSSA, BAA, Flags,
-                                             const_cast<MemoryUse *>(MU));
-        if (!MSSA->isLiveOnEntryDef(MD) && CurLoop->contains(MD->getBlock()))
-          return false;
-        // Disable hoisting past potentially interfering loads. Optimized
-        // Uses may point to an access outside the loop, as getClobbering
-        // checks the previous iteration when walking the backedge.
-        // FIXME: More precise: no Uses that alias I.
-        if (!Flags.getIsSink() && !MSSA->dominates(IMD, MU))
-          return false;
-      } else if (const auto *MD = dyn_cast<MemoryDef>(&MA)) {
+  // Walk the MemorySSA graph from the loop header's MemoryPhi.  Every
+  // MemoryDef and (in optimized MSSA) every aliasing MemoryUse in the loop
+  // is reachable through the users of the MemoryPhi and loop-internal defs.
+  auto *HeaderPhi = MSSA->getMemoryAccess(CurLoop->getHeader());
+  if (!HeaderPhi)
+    return true;
+
+  SmallVector<const MemoryAccess *, 8> Worklist;
+  SmallPtrSet<const MemoryAccess *, 8> Visited;
+  Worklist.push_back(HeaderPhi);
+  Visited.insert(HeaderPhi);
+
+  while (!Worklist.empty()) {
+    const MemoryAccess *MA = Worklist.pop_back_val();
+    for (const User *U : MA->users()) {
+      const auto *UserMA = cast<MemoryAccess>(U);
+      if (!Visited.insert(UserMA).second)
+        continue;
+      // Skip accesses outside the loop (e.g., LCSSA phi users).
+      if (!CurLoop->contains(UserMA->getBlock()))
+        continue;
+
+      if (const auto *MU = dyn_cast<MemoryUse>(UserMA)) {
+        // A MemoryUse whose defining access is inside the loop may or may
+        // not alias with I (MemorySSA uses may not be optimized yet).
+        // Check directly whether this use reads from I's write location.
+        Instruction *UseInst = MU->getMemoryInst();
+        if (auto *SI = dyn_cast<StoreInst>(I)) {
+          if (isRefSet(BAA.getModRefInfo(UseInst, MemoryLocation::get(SI))))
+            return false;
+        } else {
+          if (UseInst != I &&
+              isRefSet(BAA.getModRefInfo(UseInst, cast<CallInst>(I))))
+            return false;
+        }
+        continue;
+      }
+
+      if (const auto *MD = dyn_cast<MemoryDef>(UserMA)) {
+        // Ordered loads are stored as MemoryDefs; always reject.
         if (auto *LI = dyn_cast<LoadInst>(MD->getMemoryInst())) {
-          (void)LI; // Silence warning.
+          (void)LI;
           assert(!LI->isUnordered() && "Expected unordered load");
           return false;
         }
-        // Any call, while it may not be clobbering I, it may be a use.
+        // A call may read from the location written by I.
         if (auto *CI = dyn_cast<CallInst>(MD->getMemoryInst())) {
-          // Check if the call may read from the memory location written
-          // to by I. Check CI's attributes and arguments; the number of
-          // such checks performed is limited above by NoOfMemAccTooLarge.
-          if (auto *SI = dyn_cast<StoreInst>(I)) {
-            ModRefInfo MRI = BAA.getModRefInfo(CI, MemoryLocation::get(SI));
-            if (isModOrRefSet(MRI))
-              return false;
-          } else {
-            auto *SCI = cast<CallInst>(I);
-            // If the instruction we are wanting to hoist is also a call
-            // instruction then we need not check mod/ref info with itself
-            if (SCI == CI)
-              continue;
-            ModRefInfo MRI = BAA.getModRefInfo(CI, SCI);
-            if (isModOrRefSet(MRI))
-              return false;
+          if (CI != I) {
+            if (auto *SI = dyn_cast<StoreInst>(I)) {
+              if (isModOrRefSet(BAA.getModRefInfo(CI, MemoryLocation::get(SI))))
+                return false;
+            } else {
+              if (isModOrRefSet(BAA.getModRefInfo(CI, cast<CallInst>(I))))
+                return false;
+            }
           }
         }
+        // Follow defs inside the loop to find their users.
+        Worklist.push_back(MD);
       }
+
+      // MemoryPhis in sub-loops: follow them.
+      if (isa<MemoryPhi>(UserMA))
+        Worklist.push_back(UserMA);
+    }
   }
   return true;
 }
diff --git a/llvm/test/Analysis/MemorySSA/pr43427.ll b/llvm/test/Analysis/MemorySSA/pr43427.ll
index 254fb1104c590..f9338de8c6ee0 100644
--- a/llvm/test/Analysis/MemorySSA/pr43427.ll
+++ b/llvm/test/Analysis/MemorySSA/pr43427.ll
@@ -2,9 +2,13 @@
 
 ; CHECK-LABEL: @f(i1 %arg)
 
+; CHECK: entry:
+; CHECK:      ; [[NO1:.*]] = MemoryDef(liveOnEntry)
+; CHECK-NEXT:  store i16 undef, ptr %e, align 1
+
 ; CHECK: lbl1:
-; CHECK-NEXT: ; [[NO4:.*]] = MemoryPhi({entry,liveOnEntry},{lbl1.backedge,[[NO9:.*]]})
-; CHECK-NEXT: ; [[NO2:.*]] = MemoryDef([[NO4]])
+; CHECK-NEXT: ; [[NO4:.*]] = MemoryPhi({entry,[[NO1]]},{lbl1.backedge,[[NO2:.*]]})
+; CHECK-NEXT: ; [[NO2]] = MemoryDef([[NO4]])
 ; CHECK-NEXT:  call void @g()
 ; CHECK-NEXT:  br i1 %arg, label %for.end, label %if.else
 
@@ -12,24 +16,20 @@
 ; CHECK-NEXT:  br i1 %arg, label %lbl3, label %lbl2
 
 ; CHECK: lbl2:
-; CHECK-NEXT: ; [[NO8:.*]] = MemoryPhi({lbl3,[[NO7:.*]]},{for.end,[[NO2]]})
 ; CHECK-NEXT:  br label %lbl3
 
 ; CHECK: lbl3:
-; CHECK-NEXT: [[NO7]] = MemoryPhi({lbl2,[[NO8]]},{for.end,2})
+; CHECK-NEXT:  br i1 %arg, label %lbl2, label %cleanup
 
 ; CHECK: cleanup:
 ; CHECK-NEXT: MemoryUse([[NO2]])
 ; CHECK-NEXT:  %cleanup.dest = load i32, ptr undef, align 1
 
 ; CHECK: lbl1.backedge:
-; CHECK-NEXT:  [[NO9]] = MemoryPhi({cleanup,[[NO7]]},{if.else,2})
 ; CHECK-NEXT:   br label %lbl1
 
 ; CHECK: cleanup.cont:
-; CHECK-NEXT: ; [[NO6:.*]] = MemoryDef([[NO7]])
-; CHECK-NEXT:   store i16 undef, ptr %e, align 1
-; CHECK-NEXT:  3 = MemoryDef([[NO6]])
+; CHECK-NEXT: ; [[NO3:.*]] = MemoryDef([[NO2]])
 ; CHECK-NEXT:   call void @g()
 
 define void @f(i1 %arg) {
diff --git a/llvm/test/Transforms/LICM/call-hoisting.ll b/llvm/test/Transforms/LICM/call-hoisting.ll
index bb28d1ca93233..d135b7440b4fe 100644
--- a/llvm/test/Transforms/LICM/call-hoisting.ll
+++ b/llvm/test/Transforms/LICM/call-hoisting.ll
@@ -277,6 +277,35 @@ exit:
   ret i32 %val
 }
 
+define i32 @unrelated_read(ptr noalias %loc, ptr noalias %otherloc) {
+; CHECK-LABEL: define i32 @unrelated_read(
+; CHECK-SAME: ptr noalias [[LOC:%.*]], ptr noalias [[OTHERLOC:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    call void @store(i32 0, ptr [[LOC]])
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i32 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[VAL:%.*]] = call i32 @load(i32 [[IV]], ptr [[OTHERLOC]])
+; CHECK-NEXT:    [[IV_NEXT]] = add i32 [[IV]], 1
+; CHECK-NEXT:    [[CMP:%.*]] = icmp slt i32 [[IV]], 200
+; CHECK-NEXT:    br i1 [[CMP]], label %[[LOOP]], label %[[EXIT:.*]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    [[VAL_LCSSA:%.*]] = phi i32 [ [[VAL]], %[[LOOP]] ]
+; CHECK-NEXT:    ret i32 [[VAL_LCSSA]]
+;
+entry:
+  br label %loop
+loop:
+  %iv = phi i32 [0, %entry], [%iv.next, %loop]
+  %val = call i32 @load(i32 %iv, ptr %otherloc)
+  call void @store(i32 0, ptr %loc)
+  %iv.next = add i32 %iv, 1
+  %cmp = icmp slt i32 %iv, 200
+  br i1 %cmp, label %loop, label %exit
+exit:
+  ret i32 %val
+}
+
 define void @neg_lv_value(ptr %loc) {
 ; CHECK-LABEL: define void @neg_lv_value(
 ; CHECK-SAME: ptr [[LOC:%.*]]) {
diff --git a/llvm/test/Transforms/LICM/pr54495.ll b/llvm/test/Transforms/LICM/pr54495.ll
index d01ca69d55242..5e66758257ef0 100644
--- a/llvm/test/Transforms/LICM/pr54495.ll
+++ b/llvm/test/Transforms/LICM/pr54495.ll
@@ -6,6 +6,7 @@
 define void @test(ptr %p1, ptr %p2, ptr noalias %p3) {
 ; CHECK-LABEL: @test(
 ; CHECK-NEXT:  entry:
+; CHECK-NEXT:    store ptr [[P3:%.*]], ptr [[P3]], align 8
 ; CHECK-NEXT:    br label [[LOOP:%.*]]
 ; CHECK:       loop:
 ; CHECK-NEXT:    [[P:%.*]] = phi ptr [ [[P1:%.*]], [[ENTRY:%.*]] ], [ [[P2:%.*]], [[LOOP]] ]
@@ -13,7 +14,6 @@ define void @test(ptr %p1, ptr %p2, ptr noalias %p3) {
 ; CHECK-NEXT:    [[CMP:%.*]] = icmp eq i64 [[V]], 0
 ; CHECK-NEXT:    br i1 [[CMP]], label [[LOOP]], label [[LOOP_EXIT:%.*]]
 ; CHECK:       loop.exit:
-; CHECK-NEXT:    store ptr [[P3:%.*]], ptr [[P3]], align 8
 ; CHECK-NEXT:    ret void
 ;
 entry:



More information about the llvm-commits mailing list