[llvm] 9073b4d - [PromoteMemToReg] Hoist removeIntrinsicUsers into a pre-pass before promotion (#224617)

via llvm-commits llvm-commits at lists.llvm.org
Tue Sep 22 01:48:41 PDT 2026


Author: Marcin Radomski
Date: 2026-09-22T10:48:35+02:00
New Revision: 9073b4d71e9d66c495df5f6ec42ab0279326f762

URL: https://github.com/llvm/llvm-project/commit/9073b4d71e9d66c495df5f6ec42ab0279326f762
DIFF: https://github.com/llvm/llvm-project/commit/9073b4d71e9d66c495df5f6ec42ab0279326f762.diff

LOG: [PromoteMemToReg] Hoist removeIntrinsicUsers into a pre-pass before promotion (#224617)

Calling removeIntrinsicUsers(AI) inside the per-alloca promotion loop in
PromoteMem2Reg::run() inserts StoreInst(undef) instructions into basic
blocks after LargeBlockInfo (LBI) has already cached instruction indices
for those blocks during earlier iterations. Every subsequent
LBI.getInstructionIndex query on newly inserted StoreInst(undef)
instructions misses in LBI.InstNumbers and triggers a full O(|BB|) basic
block rescan, resulting in O(K * |BB|) compilation time blowup in
promoteSingleBlockAlloca when a basic block contains K scoped allocas.

Hoist removeIntrinsicUsers(AI) into a pre-pass loop over all Allocas
upfront in PromoteMem2Reg::run() before the main AllocaNum promotion
loop begins so all synthetic StoreInst(undef) instructions are inserted
prior to LBI caching.

Attached test verifies this PR doesn't change the behavior compared to
HEAD.

This improves the compile time of some CUDA kernels
(https://github.com/llvm/llvm-project/pull/191909#issuecomment-5569614846)
by ~9x.

Assisted-by: Gemini

Added: 
    

Modified: 
    llvm/lib/Transforms/Utils/PromoteMemoryToRegister.cpp
    llvm/test/Transforms/Mem2Reg/ignore-droppable.ll
    llvm/test/Transforms/Mem2Reg/ignore-lifetime.ll

Removed: 
    


################################################################################
diff  --git a/llvm/lib/Transforms/Utils/PromoteMemoryToRegister.cpp b/llvm/lib/Transforms/Utils/PromoteMemoryToRegister.cpp
index 21ddc78f1469b..157c3de56be02 100644
--- a/llvm/lib/Transforms/Utils/PromoteMemoryToRegister.cpp
+++ b/llvm/lib/Transforms/Utils/PromoteMemoryToRegister.cpp
@@ -810,14 +810,19 @@ void PromoteMem2Reg::run() {
 
   NoSignedZeros = F.getFnAttribute("no-signed-zeros-fp-math").getValueAsBool();
 
-  for (unsigned AllocaNum = 0; AllocaNum != Allocas.size(); ++AllocaNum) {
-    AllocaInst *AI = Allocas[AllocaNum];
-
+  // removeIntrinsicUsers inserts StoreInst(undef). A cache miss on
+  // LBI.getInstructionIndex lookup causes a full O(|BB|) basic block rescan.
+  // Doing all inserts in advance lets us capture all the new instructions in
+  // just a single rescan.
+  for (AllocaInst *AI : Allocas) {
     assert(isAllocaPromotable(AI) && "Cannot promote non-promotable alloca!");
     assert(AI->getParent()->getParent() == &F &&
            "All allocas should be in the same function, which is same as DF!");
-
     removeIntrinsicUsers(AI);
+  }
+
+  for (unsigned AllocaNum = 0; AllocaNum != Allocas.size(); ++AllocaNum) {
+    AllocaInst *AI = Allocas[AllocaNum];
 
     if (AI->use_empty()) {
       // If there are no uses of the alloca, just delete it now.

diff  --git a/llvm/test/Transforms/Mem2Reg/ignore-droppable.ll b/llvm/test/Transforms/Mem2Reg/ignore-droppable.ll
index a876319281b17..e5236c7b3d96d 100644
--- a/llvm/test/Transforms/Mem2Reg/ignore-droppable.ll
+++ b/llvm/test/Transforms/Mem2Reg/ignore-droppable.ll
@@ -78,3 +78,58 @@ define void @positive_mixed_assume_uses() {
   call void @llvm.assume(i1 true) ["nonnull"(ptr %A), "align"(ptr %A, i64 2), "nonnull"(ptr %A)]
   ret void
 }
+
+declare void @use(i32)
+
+; Verify that lifetime.start and lifetime.end on an addrspace(5) alloca
+; alongside zero-index GEP droppable users insert store undef in the pre-pass,
+; breaking loop-carried back-edge dependencies.
+define void @lifetime_addrspace5_and_gep_in_loop(i32 %val, i1 %c1, i1 %c2) {
+; CHECK-LABEL: @lifetime_addrspace5_and_gep_in_loop(
+; CHECK-NEXT:  entry:
+; CHECK-NEXT:    br label [[LOOP:%.*]]
+; CHECK:       loop:
+; CHECK-NEXT:    call void @llvm.assume(i1 true) [ "ignore"(ptr addrspace(5) poison, i64 4) ]
+; CHECK-NEXT:    br i1 [[C1:%.*]], label [[INIT:%.*]], label [[SKIP:%.*]]
+; CHECK:       init:
+; CHECK-NEXT:    br label [[READ:%.*]]
+; CHECK:       skip:
+; CHECK-NEXT:    br label [[READ]]
+; CHECK:       read:
+; CHECK-NEXT:    [[A_0:%.*]] = phi i32 [ [[VAL:%.*]], [[INIT]] ], [ undef, [[SKIP]] ]
+; CHECK-NEXT:    call void @use(i32 [[A_0]])
+; CHECK-NEXT:    br label [[CLEANUP:%.*]]
+; CHECK:       cleanup:
+; CHECK-NEXT:    br i1 [[C2:%.*]], label [[LOOP]], label [[EXIT:%.*]]
+; CHECK:       exit:
+; CHECK-NEXT:    ret void
+;
+entry:
+  %a = alloca i32, align 4, addrspace(5)
+  br label %loop
+
+loop:
+  call void @llvm.lifetime.start.p5(ptr addrspace(5) %a)
+  %gep = getelementptr inbounds i32, ptr addrspace(5) %a, i64 0
+  call void @llvm.assume(i1 true) ["align"(ptr addrspace(5) %gep, i64 4)]
+  br i1 %c1, label %init, label %skip
+
+init:
+  store i32 %val, ptr addrspace(5) %a, align 4
+  br label %read
+
+skip:
+  br label %read
+
+read:
+  %v = load i32, ptr addrspace(5) %a, align 4
+  call void @use(i32 %v)
+  br label %cleanup
+
+cleanup:
+  call void @llvm.lifetime.end.p5(ptr addrspace(5) %a)
+  br i1 %c2, label %loop, label %exit
+
+exit:
+  ret void
+}

diff  --git a/llvm/test/Transforms/Mem2Reg/ignore-lifetime.ll b/llvm/test/Transforms/Mem2Reg/ignore-lifetime.ll
index 510fb2b8638e0..c0cd14fc7c5df 100644
--- a/llvm/test/Transforms/Mem2Reg/ignore-lifetime.ll
+++ b/llvm/test/Transforms/Mem2Reg/ignore-lifetime.ll
@@ -22,3 +22,26 @@ define void @test2() {
   call void @llvm.lifetime.end.p0(ptr %A)
   ret void
 }
+
+; Verify that multiple single-block allocas with lifetime markers in the same
+; block are promoted accurately when removeIntrinsicUsers runs as a pre-pass
+; before LargeBlockInfo caches instruction indices.
+define i32 @multiple_single_block_allocas_with_lifetime(i32 %x, i32 %y) {
+; CHECK-LABEL: define i32 @multiple_single_block_allocas_with_lifetime(
+; CHECK-SAME: i32 [[X:%.*]], i32 [[Y:%.*]]) {
+; CHECK-NEXT:    [[SUM:%.*]] = add i32 [[X]], [[Y]]
+; CHECK-NEXT:    ret i32 [[SUM]]
+;
+  %a = alloca i32, align 4
+  %b = alloca i32, align 4
+  call void @llvm.lifetime.start.p0(ptr %a)
+  store i32 %x, ptr %a, align 4
+  %va = load i32, ptr %a, align 4
+  call void @llvm.lifetime.end.p0(ptr %a)
+  call void @llvm.lifetime.start.p0(ptr %b)
+  store i32 %y, ptr %b, align 4
+  %vb = load i32, ptr %b, align 4
+  call void @llvm.lifetime.end.p0(ptr %b)
+  %sum = add i32 %va, %vb
+  ret i32 %sum
+}


        


More information about the llvm-commits mailing list