[llvm] [AMDGPU] Do not coerce values defined by terminators in LiveRegOptimizer (PR #218868)

Arseniy Obolenskiy via llvm-commits llvm-commits at lists.llvm.org
Wed Aug 26 03:00:30 PDT 2026


https://github.com/aobolensk updated https://github.com/llvm/llvm-project/pull/218868

>From 0ee3c910290cafc6cb03090ca01ad7fb096b3c4e Mon Sep 17 00:00:00 2001
From: Arseniy Obolenskiy <arseniy.obolenskiy at amd.com>
Date: Wed, 26 Aug 2026 11:24:26 +0200
Subject: [PATCH 1/2] [AMDGPU] Do not coerce values defined by terminators in
 LiveRegOptimizer

Inserting the coercion after a value-producing terminator (invoke, callbr) lands past the end of its block, leaving the IR without a terminator
---
 .../AMDGPU/AMDGPULateCodeGenPrepare.cpp       |  12 +-
 .../AMDGPU/amdgpu-late-codegenprepare.ll      | 229 ++++++++++++++++++
 2 files changed, 238 insertions(+), 3 deletions(-)

diff --git a/llvm/lib/Target/AMDGPU/AMDGPULateCodeGenPrepare.cpp b/llvm/lib/Target/AMDGPU/AMDGPULateCodeGenPrepare.cpp
index be0b221456077..0b4aebb09f7e5 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPULateCodeGenPrepare.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPULateCodeGenPrepare.cpp
@@ -385,6 +385,9 @@ bool LiveRegOptimizer::optimizeLiveType(
 
   // Coerce and track the defs.
   for (Instruction *D : Defs) {
+    // Terminators can't be coerced in-block, so skip them.
+    if (D->isTerminator())
+      continue;
     if (!ValMap.contains(D)) {
       BasicBlock::iterator InsertPt = std::next(D->getIterator());
       Value *ConvertVal = convertToOptType(D, InsertPt);
@@ -434,12 +437,14 @@ bool LiveRegOptimizer::optimizeLiveType(
         if (OriginalPhi != PhiNodes.end())
           ValMap.erase(*OriginalPhi);
 
-        DeadInsts.emplace_back(cast<Instruction>(NextDeadValue));
-
         for (User *U : NextDeadValue->users()) {
           if (!VisitedPhis.contains(cast<PHINode>(U)))
             PHIWorklist.push_back(U);
         }
+        NextDeadValue->replaceAllUsesWith(
+            PoisonValue::get(NextDeadValue->getType()));
+
+        DeadInsts.emplace_back(cast<Instruction>(NextDeadValue));
       }
     } else {
       DeadInsts.emplace_back(cast<Instruction>(Phi));
@@ -455,7 +460,8 @@ bool LiveRegOptimizer::optimizeLiveType(
             BBUseValMap[U->getParent()].contains(Val))
           NewVal = BBUseValMap[U->getParent()][Val];
         else {
-          BasicBlock::iterator InsertPt = U->getParent()->getFirstNonPHIIt();
+          // Not getFirstNonPHIIt, which would insert in front of a landingpad.
+          BasicBlock::iterator InsertPt = U->getParent()->getFirstInsertionPt();
           // We may pick up ops that were previously converted for users in
           // other blocks. If there is an originally typed definition of the Op
           // already in this block, simply reuse it.
diff --git a/llvm/test/CodeGen/AMDGPU/amdgpu-late-codegenprepare.ll b/llvm/test/CodeGen/AMDGPU/amdgpu-late-codegenprepare.ll
index a5fecc4913432..d12267909fd08 100644
--- a/llvm/test/CodeGen/AMDGPU/amdgpu-late-codegenprepare.ll
+++ b/llvm/test/CodeGen/AMDGPU/amdgpu-late-codegenprepare.ll
@@ -271,3 +271,232 @@ define amdgpu_kernel void @no_widen_insufficient_dereferenceable(ptr addrspace(4
 
 !0 = !{}
 !1 = !{i32 3}
+
+; Coercing a value-producing terminator's def used to insert past the block end.
+define amdgpu_kernel void @callbr_def_cross_block_use(ptr addrspace(1) %out) {
+;
+; GFX9-LABEL: @callbr_def_cross_block_use(
+; GFX9-NEXT:  entry:
+; GFX9-NEXT:    [[V:%.*]] = callbr <4 x i8> asm "", "=v"()
+; GFX9-NEXT:            to label [[USE:%.*]] []
+; GFX9:       use:
+; GFX9-NEXT:    store <4 x i8> [[V]], ptr addrspace(1) [[OUT:%.*]], align 4
+; GFX9-NEXT:    ret void
+;
+; GFX12-LABEL: @callbr_def_cross_block_use(
+; GFX12-NEXT:  entry:
+; GFX12-NEXT:    [[V:%.*]] = callbr <4 x i8> asm "", "=v"()
+; GFX12-NEXT:            to label [[USE:%.*]] []
+; GFX12:       use:
+; GFX12-NEXT:    store <4 x i8> [[V]], ptr addrspace(1) [[OUT:%.*]], align 4
+; GFX12-NEXT:    ret void
+;
+entry:
+  %v = callbr <4 x i8> asm "", "=v"() to label %use []
+use:
+  store <4 x i8> %v, ptr addrspace(1) %out
+  ret void
+}
+
+; Coercible incoming value shares a phi with the callbr def; the new phi must unwind.
+define amdgpu_kernel void @callbr_def_phi_incoming(ptr addrspace(1) %in, ptr addrspace(1) %out, i1 %c) {
+;
+; GFX9-LABEL: @callbr_def_phi_incoming(
+; GFX9-NEXT:  entry:
+; GFX9-NEXT:    br i1 [[C:%.*]], label [[A:%.*]], label [[B:%.*]]
+; GFX9:       a:
+; GFX9-NEXT:    [[V:%.*]] = callbr <4 x i8> asm "", "=v"()
+; GFX9-NEXT:            to label [[JOIN:%.*]] []
+; GFX9:       b:
+; GFX9-NEXT:    [[W:%.*]] = load <4 x i8>, ptr addrspace(1) [[IN:%.*]], align 4
+; GFX9-NEXT:    br label [[JOIN]]
+; GFX9:       join:
+; GFX9-NEXT:    [[P:%.*]] = phi <4 x i8> [ [[V]], [[A]] ], [ [[W]], [[B]] ]
+; GFX9-NEXT:    store <4 x i8> [[P]], ptr addrspace(1) [[OUT:%.*]], align 4
+; GFX9-NEXT:    ret void
+;
+; GFX12-LABEL: @callbr_def_phi_incoming(
+; GFX12-NEXT:  entry:
+; GFX12-NEXT:    br i1 [[C:%.*]], label [[A:%.*]], label [[B:%.*]]
+; GFX12:       a:
+; GFX12-NEXT:    [[V:%.*]] = callbr <4 x i8> asm "", "=v"()
+; GFX12-NEXT:            to label [[JOIN:%.*]] []
+; GFX12:       b:
+; GFX12-NEXT:    [[W:%.*]] = load <4 x i8>, ptr addrspace(1) [[IN:%.*]], align 4
+; GFX12-NEXT:    br label [[JOIN]]
+; GFX12:       join:
+; GFX12-NEXT:    [[P:%.*]] = phi <4 x i8> [ [[V]], [[A]] ], [ [[W]], [[B]] ]
+; GFX12-NEXT:    store <4 x i8> [[P]], ptr addrspace(1) [[OUT:%.*]], align 4
+; GFX12-NEXT:    ret void
+;
+entry:
+  br i1 %c, label %a, label %b
+a:
+  %v = callbr <4 x i8> asm "", "=v"() to label %join []
+b:
+  %w = load <4 x i8>, ptr addrspace(1) %in
+  br label %join
+join:
+  %p = phi <4 x i8> [ %v, %a ], [ %w, %b ]
+  store <4 x i8> %p, ptr addrspace(1) %out
+  ret void
+}
+
+; Loop-carried phi feeding itself: unwound replacement phis form a use cycle.
+define amdgpu_kernel void @callbr_def_self_loop_phi(ptr addrspace(1) %out, i1 %c) {
+;
+; GFX9-LABEL: @callbr_def_self_loop_phi(
+; GFX9-NEXT:  entry:
+; GFX9-NEXT:    [[V:%.*]] = callbr <4 x i8> asm "", "=v"()
+; GFX9-NEXT:            to label [[LOOP:%.*]] []
+; GFX9:       loop:
+; GFX9-NEXT:    [[P:%.*]] = phi <4 x i8> [ [[V]], [[ENTRY:%.*]] ], [ [[P]], [[LOOP]] ]
+; GFX9-NEXT:    br i1 [[C:%.*]], label [[LOOP]], label [[EXIT:%.*]]
+; GFX9:       exit:
+; GFX9-NEXT:    store <4 x i8> [[P]], ptr addrspace(1) [[OUT:%.*]], align 4
+; GFX9-NEXT:    ret void
+;
+; GFX12-LABEL: @callbr_def_self_loop_phi(
+; GFX12-NEXT:  entry:
+; GFX12-NEXT:    [[V:%.*]] = callbr <4 x i8> asm "", "=v"()
+; GFX12-NEXT:            to label [[LOOP:%.*]] []
+; GFX12:       loop:
+; GFX12-NEXT:    [[P:%.*]] = phi <4 x i8> [ [[V]], [[ENTRY:%.*]] ], [ [[P]], [[LOOP]] ]
+; GFX12-NEXT:    br i1 [[C:%.*]], label [[LOOP]], label [[EXIT:%.*]]
+; GFX12:       exit:
+; GFX12-NEXT:    store <4 x i8> [[P]], ptr addrspace(1) [[OUT:%.*]], align 4
+; GFX12-NEXT:    ret void
+;
+entry:
+  %v = callbr <4 x i8> asm "", "=v"() to label %loop []
+loop:
+  %p = phi <4 x i8> [ %v, %entry ], [ %p, %loop ]
+  br i1 %c, label %loop, label %exit
+exit:
+  store <4 x i8> %p, ptr addrspace(1) %out
+  ret void
+}
+
+; Used on an indirect edge, so there is no single block to insert the coercion at.
+define amdgpu_kernel void @callbr_def_indirect_use(ptr addrspace(1) %out) {
+;
+; GFX9-LABEL: @callbr_def_indirect_use(
+; GFX9-NEXT:  entry:
+; GFX9-NEXT:    [[V:%.*]] = callbr <4 x i8> asm "", "=v,!i"()
+; GFX9-NEXT:            to label [[FALLTHROUGH:%.*]] [label [[INDIRECT:%.*]]]
+; GFX9:       fallthrough:
+; GFX9-NEXT:    store <4 x i8> zeroinitializer, ptr addrspace(1) [[OUT:%.*]], align 4
+; GFX9-NEXT:    ret void
+; GFX9:       indirect:
+; GFX9-NEXT:    store <4 x i8> [[V]], ptr addrspace(1) [[OUT]], align 4
+; GFX9-NEXT:    ret void
+;
+; GFX12-LABEL: @callbr_def_indirect_use(
+; GFX12-NEXT:  entry:
+; GFX12-NEXT:    [[V:%.*]] = callbr <4 x i8> asm "", "=v,!i"()
+; GFX12-NEXT:            to label [[FALLTHROUGH:%.*]] [label [[INDIRECT:%.*]]]
+; GFX12:       fallthrough:
+; GFX12-NEXT:    store <4 x i8> zeroinitializer, ptr addrspace(1) [[OUT:%.*]], align 4
+; GFX12-NEXT:    ret void
+; GFX12:       indirect:
+; GFX12-NEXT:    store <4 x i8> [[V]], ptr addrspace(1) [[OUT]], align 4
+; GFX12-NEXT:    ret void
+;
+entry:
+  %v = callbr <4 x i8> asm "", "=v,!i"() to label %fallthrough [label %indirect]
+fallthrough:
+  store <4 x i8> zeroinitializer, ptr addrspace(1) %out
+  ret void
+indirect:
+  store <4 x i8> %v, ptr addrspace(1) %out
+  ret void
+}
+
+define amdgpu_kernel void @invoke_def_cross_block_use(ptr addrspace(1) %out) personality ptr @__gxx_personality_v0 {
+;
+; GFX9-LABEL: @invoke_def_cross_block_use(
+; GFX9-NEXT:  entry:
+; GFX9-NEXT:    [[V:%.*]] = invoke <4 x i8> @ret_v4i8()
+; GFX9-NEXT:            to label [[USE:%.*]] unwind label [[LPAD:%.*]]
+; GFX9:       use:
+; GFX9-NEXT:    store <4 x i8> [[V]], ptr addrspace(1) [[OUT:%.*]], align 4
+; GFX9-NEXT:    ret void
+; GFX9:       lpad:
+; GFX9-NEXT:    [[LP:%.*]] = landingpad { ptr, i32 }
+; GFX9-NEXT:            cleanup
+; GFX9-NEXT:    ret void
+;
+; GFX12-LABEL: @invoke_def_cross_block_use(
+; GFX12-NEXT:  entry:
+; GFX12-NEXT:    [[V:%.*]] = invoke <4 x i8> @ret_v4i8()
+; GFX12-NEXT:            to label [[USE:%.*]] unwind label [[LPAD:%.*]]
+; GFX12:       use:
+; GFX12-NEXT:    store <4 x i8> [[V]], ptr addrspace(1) [[OUT:%.*]], align 4
+; GFX12-NEXT:    ret void
+; GFX12:       lpad:
+; GFX12-NEXT:    [[LP:%.*]] = landingpad { ptr, i32 }
+; GFX12-NEXT:            cleanup
+; GFX12-NEXT:    ret void
+;
+entry:
+  %v = invoke <4 x i8> @ret_v4i8() to label %use unwind label %lpad
+use:
+  store <4 x i8> %v, ptr addrspace(1) %out
+  ret void
+lpad:
+  %lp = landingpad { ptr, i32 } cleanup
+  ret void
+}
+
+; Conversion must insert after the landingpad, not before it.
+define amdgpu_kernel void @use_in_landingpad_block(ptr addrspace(1) %in, ptr addrspace(1) %out) personality ptr @__gxx_personality_v0 {
+;
+; GFX9-LABEL: @use_in_landingpad_block(
+; GFX9-NEXT:  entry:
+; GFX9-NEXT:    [[V:%.*]] = load <4 x i8>, ptr addrspace(1) [[IN:%.*]], align 4
+; GFX9-NEXT:    [[V_BC:%.*]] = bitcast <4 x i8> [[V]] to i32
+; GFX9-NEXT:    invoke void @maybe_throw()
+; GFX9-NEXT:            to label [[CONT:%.*]] unwind label [[LPAD:%.*]]
+; GFX9:       cont:
+; GFX9-NEXT:    [[V_BC_BC1:%.*]] = bitcast i32 [[V_BC]] to <4 x i8>
+; GFX9-NEXT:    store <4 x i8> [[V_BC_BC1]], ptr addrspace(1) [[OUT:%.*]], align 4
+; GFX9-NEXT:    ret void
+; GFX9:       lpad:
+; GFX9-NEXT:    [[LP:%.*]] = landingpad { ptr, i32 }
+; GFX9-NEXT:            cleanup
+; GFX9-NEXT:    [[V_BC_BC:%.*]] = bitcast i32 [[V_BC]] to <4 x i8>
+; GFX9-NEXT:    store <4 x i8> [[V_BC_BC]], ptr addrspace(1) [[OUT]], align 4
+; GFX9-NEXT:    ret void
+;
+; GFX12-LABEL: @use_in_landingpad_block(
+; GFX12-NEXT:  entry:
+; GFX12-NEXT:    [[V:%.*]] = load <4 x i8>, ptr addrspace(1) [[IN:%.*]], align 4
+; GFX12-NEXT:    [[V_BC:%.*]] = bitcast <4 x i8> [[V]] to i32
+; GFX12-NEXT:    invoke void @maybe_throw()
+; GFX12-NEXT:            to label [[CONT:%.*]] unwind label [[LPAD:%.*]]
+; GFX12:       cont:
+; GFX12-NEXT:    [[V_BC_BC1:%.*]] = bitcast i32 [[V_BC]] to <4 x i8>
+; GFX12-NEXT:    store <4 x i8> [[V_BC_BC1]], ptr addrspace(1) [[OUT:%.*]], align 4
+; GFX12-NEXT:    ret void
+; GFX12:       lpad:
+; GFX12-NEXT:    [[LP:%.*]] = landingpad { ptr, i32 }
+; GFX12-NEXT:            cleanup
+; GFX12-NEXT:    [[V_BC_BC:%.*]] = bitcast i32 [[V_BC]] to <4 x i8>
+; GFX12-NEXT:    store <4 x i8> [[V_BC_BC]], ptr addrspace(1) [[OUT]], align 4
+; GFX12-NEXT:    ret void
+;
+entry:
+  %v = load <4 x i8>, ptr addrspace(1) %in
+  invoke void @maybe_throw() to label %cont unwind label %lpad
+cont:
+  store <4 x i8> %v, ptr addrspace(1) %out
+  ret void
+lpad:
+  %lp = landingpad { ptr, i32 } cleanup
+  store <4 x i8> %v, ptr addrspace(1) %out
+  ret void
+}
+
+declare <4 x i8> @ret_v4i8()
+declare void @maybe_throw()
+declare i32 @__gxx_personality_v0(...)

>From e064a8768d497afe715a0c3e9894befa2de50a99 Mon Sep 17 00:00:00 2001
From: Arseniy Obolenskiy <arseniy.obolenskiy at amd.com>
Date: Wed, 26 Aug 2026 12:00:17 +0200
Subject: [PATCH 2/2] comment

---
 llvm/lib/Target/AMDGPU/AMDGPULateCodeGenPrepare.cpp | 7 ++-----
 1 file changed, 2 insertions(+), 5 deletions(-)

diff --git a/llvm/lib/Target/AMDGPU/AMDGPULateCodeGenPrepare.cpp b/llvm/lib/Target/AMDGPU/AMDGPULateCodeGenPrepare.cpp
index 0b4aebb09f7e5..0f9fdfb46ce23 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPULateCodeGenPrepare.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPULateCodeGenPrepare.cpp
@@ -359,7 +359,7 @@ bool LiveRegOptimizer::optimizeLiveType(
           return false;
 
         // Collect all other incoming values for coercion.
-        if (IncInst)
+        if (IncInst && !IncInst->isTerminator())
           Defs.insert(IncInst);
       }
     }
@@ -377,7 +377,7 @@ bool LiveRegOptimizer::optimizeLiveType(
       // Collect all uses of PHINodes and any use the crosses BB boundaries.
       if (UseInst->getParent() != II->getParent() || isa<PHINode>(II)) {
         Uses.insert(UseInst);
-        if (!isa<PHINode>(II))
+        if (!isa<PHINode>(II) && !II->isTerminator())
           Defs.insert(II);
       }
     }
@@ -385,9 +385,6 @@ bool LiveRegOptimizer::optimizeLiveType(
 
   // Coerce and track the defs.
   for (Instruction *D : Defs) {
-    // Terminators can't be coerced in-block, so skip them.
-    if (D->isTerminator())
-      continue;
     if (!ValMap.contains(D)) {
       BasicBlock::iterator InsertPt = std::next(D->getIterator());
       Value *ConvertVal = convertToOptType(D, InsertPt);



More information about the llvm-commits mailing list