[llvm] [GlobalISel] Don't treat intrinsic callbr indirect dest as machine successor. (PR #221685)

Vikash Gupta via llvm-commits llvm-commits at lists.llvm.org
Mon Sep 7 02:00:26 PDT 2026


https://github.com/vg0204 created https://github.com/llvm/llvm-project/pull/221685

Intrinsic `callbr` is implicit control flow: the selected MI does not name the indirect dest, so that MBB is not a legal successor. GISel still added the edge and set `IsInlineAsmBrIndirectTarget`, emitting a fake asm-goto block for `llvm.amdgcn.kill`. SelectionDAG already omits that edge.

RTranslator now matches `visitCallBr` (default dest only, then G_BR). InstructionSelect never selects the now-unreachable dest; after MBB is cleared-up which are unreachable, it also strips leftover successors/PHI uses, as 'UnreachableMachineBlockElim' does for the MBB. As a result, tests drop the fake asm-goto check lines.

>From dab81608ee81e65aa021ae1e2846146f403ceeed Mon Sep 17 00:00:00 2001
From: vg0204 <Vikash.Gupta at amd.com>
Date: Mon, 7 Sep 2026 14:26:36 +0530
Subject: [PATCH] [GlobalISel] Don't treat intrinsic callbr indirect dest as
 machine succs

Intrinsic callbr is implicit control flow: the selected MI does not name
the indirect dest, so that MBB is not a legal successor. GISel still
added the edge and set IsInlineAsmBrIndirectTarget, emitting a fake
asm-goto block for llvm.amdgcn.kill. SelectionDAG already omits that
edge.

RTranslator now matches visitCallBr (default dest only, then G_BR).
InstructionSelect never selects the now-unreachable dest; after
MBB.clear() it also strips leftover successors/PHI uses, as
UnreachableMachineBlockElim does. As a result, tests drop the fake
asm-goto check lines.
---
 llvm/lib/CodeGen/GlobalISel/IRTranslator.cpp  | 25 ++++++++-----------
 .../CodeGen/GlobalISel/InstructionSelect.cpp  | 15 ++++++++---
 llvm/test/CodeGen/AMDGPU/callbr-intrinsics.ll | 24 ++++--------------
 3 files changed, 27 insertions(+), 37 deletions(-)

diff --git a/llvm/lib/CodeGen/GlobalISel/IRTranslator.cpp b/llvm/lib/CodeGen/GlobalISel/IRTranslator.cpp
index ee4bcf773433d..d32428be0eb1c 100644
--- a/llvm/lib/CodeGen/GlobalISel/IRTranslator.cpp
+++ b/llvm/lib/CodeGen/GlobalISel/IRTranslator.cpp
@@ -3913,25 +3913,20 @@ bool IRTranslatorImpl::translateCallBr(const User &U,
   if (!translateIntrinsic(I, IID, MIRBuilder))
     return false;
 
-  // Retrieve successors.
-  SmallPtrSet<BasicBlock *, 8> Dests = {I.getDefaultDest()};
+  // Retrieve successors. Intrinsic callbr models implicit control flow (e.g.
+  // amdgcn.kill); the selected instruction does not name the indirect dest as a
+  // machine operand, so it is not a machine successor. Match
+  // SelectionDAGBuilder::visitCallBr, which only adds indirect dests for
+  // inline asm (still unsupported here).
   MachineBasicBlock *Return = &getMBB(*I.getDefaultDest());
 
   // Update successor info.
   addSuccessorWithProb(CallBrMBB, Return, BranchProbability::getOne());
-
-  // Add indirect targets as successors. For intrinsic callbr, these represent
-  // implicit control flow (e.g., the "kill" path for amdgcn.kill). We mark them
-  // with setIsInlineAsmBrIndirectTarget so the machine verifier accepts them as
-  // valid successors, even though they're not from inline asm.
-  for (BasicBlock *Dest : I.getIndirectDests()) {
-    MachineBasicBlock &Target = getMBB(*Dest);
-    Target.setIsInlineAsmBrIndirectTarget();
-    Target.setLabelMustBeEmitted();
-    // Don't add duplicate machine successors.
-    if (Dests.insert(Dest).second)
-      addSuccessorWithProb(CallBrMBB, &Target, BranchProbability::getZero());
-  }
+  // TODO: For most of the cases where there is an intrinsic callbr, we're
+  // having exactly one indirect target, which will be unreachable. As soon as
+  // this changes, we might need to enhance
+  // Target->setIsInlineAsmBrIndirectTarget or add something similar for
+  // intrinsic indirect branches.
 
   CallBrMBB->normalizeSuccProbs();
 
diff --git a/llvm/lib/CodeGen/GlobalISel/InstructionSelect.cpp b/llvm/lib/CodeGen/GlobalISel/InstructionSelect.cpp
index ccb1828ba3f96..7352388bfa847 100644
--- a/llvm/lib/CodeGen/GlobalISel/InstructionSelect.cpp
+++ b/llvm/lib/CodeGen/GlobalISel/InstructionSelect.cpp
@@ -240,18 +240,27 @@ bool InstructionSelectImpl::selectMachineFunction(MachineFunction &MF) {
   }
 
   for (MachineBasicBlock &MBB : MF) {
-    if (MBB.empty())
-      continue;
-
     if (!SelectedBlocks.contains(&MBB)) {
       // This is an unreachable block and therefore hasn't been selected, since
       // the main selection loop above uses a postorder block traversal.
       // We delete all the instructions in this block since it's unreachable.
       MBB.clear();
+
+      // Clearing dropped the terminators that encoded this block's successors,
+      // which can leave an empty MBB with a stale successor list.
+      // Therefore, trim leftover successors and PHI uses the same way
+      // UnreachableMachineBlockElim does.
       // Don't delete the block in case the block has it's address taken or is
       // still being referenced by a phi somewhere.
+      while (!MBB.succ_empty()) {
+        (*MBB.succ_begin())->removePHIsIncomingValuesForPredecessor(MBB);
+        MBB.removeSuccessor(MBB.succ_begin());
+      }
       continue;
     }
+
+    if (MBB.empty())
+      continue;
     // Try to find redundant copies b/w vregs of the same register class.
     for (auto MII = MBB.rbegin(), End = MBB.rend(); MII != End;) {
       MachineInstr &MI = *MII;
diff --git a/llvm/test/CodeGen/AMDGPU/callbr-intrinsics.ll b/llvm/test/CodeGen/AMDGPU/callbr-intrinsics.ll
index 39bf3ba0306e6..5695916f2b165 100644
--- a/llvm/test/CodeGen/AMDGPU/callbr-intrinsics.ll
+++ b/llvm/test/CodeGen/AMDGPU/callbr-intrinsics.ll
@@ -1,6 +1,6 @@
 ; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 5
 ; RUN: llc -global-isel=0 -mtriple=amdgpu9.0a < %s | FileCheck %s
-; RUN: llc -global-isel=1 -mtriple=amdgpu9.0a < %s | FileCheck --check-prefix=GISEL %s
+; RUN: llc -global-isel=1 -verify-machineinstrs -mtriple=amdgpu9.0a < %s | FileCheck --check-prefix=GISEL %s
 
 define void @test_kill(ptr %src, ptr %dst, i1 %c) {
 ; CHECK-LABEL: test_kill:
@@ -32,21 +32,14 @@ define void @test_kill(ptr %src, ptr %dst, i1 %c) {
 ; GISEL-NEXT:    s_mov_b64 s[4:5], exec
 ; GISEL-NEXT:    s_andn2_b64 s[6:7], exec, vcc
 ; GISEL-NEXT:    s_andn2_b64 s[4:5], s[4:5], s[6:7]
-; GISEL-NEXT:    s_cbranch_scc0 .LBB0_4
+; GISEL-NEXT:    s_cbranch_scc0 .LBB0_2
 ; GISEL-NEXT:  ; %bb.1:
 ; GISEL-NEXT:    s_and_b64 exec, exec, s[4:5]
-; GISEL-NEXT:  ; %bb.2: ; %cont
 ; GISEL-NEXT:    s_waitcnt vmcnt(0) lgkmcnt(0)
 ; GISEL-NEXT:    flat_store_dword v[2:3], v0
 ; GISEL-NEXT:    s_waitcnt vmcnt(0) lgkmcnt(0)
 ; GISEL-NEXT:    s_setpc_b64 s[30:31]
-; GISEL-NEXT:  .LBB0_3: ; Inline asm indirect target
-; GISEL-NEXT:    ; %kill
-; GISEL-NEXT:    ; Label of block must be emitted
-; GISEL-NEXT:    ; divergent unreachable
-; GISEL-NEXT:    s_waitcnt vmcnt(0) lgkmcnt(0)
-; GISEL-NEXT:    s_setpc_b64 s[30:31]
-; GISEL-NEXT:  .LBB0_4:
+; GISEL-NEXT:  .LBB0_2:
 ; GISEL-NEXT:    s_mov_b64 exec, 0
 ; GISEL-NEXT:    s_endpgm
   %a = load i32, ptr %src, align 4
@@ -88,21 +81,14 @@ define void @test_kill_block_order(ptr %src, ptr %dst, i1 %c) {
 ; GISEL-NEXT:    s_mov_b64 s[4:5], exec
 ; GISEL-NEXT:    s_andn2_b64 s[6:7], exec, vcc
 ; GISEL-NEXT:    s_andn2_b64 s[4:5], s[4:5], s[6:7]
-; GISEL-NEXT:    s_cbranch_scc0 .LBB1_4
+; GISEL-NEXT:    s_cbranch_scc0 .LBB1_2
 ; GISEL-NEXT:  ; %bb.1:
 ; GISEL-NEXT:    s_and_b64 exec, exec, s[4:5]
-; GISEL-NEXT:  ; %bb.2: ; %cont
 ; GISEL-NEXT:    s_waitcnt vmcnt(0) lgkmcnt(0)
 ; GISEL-NEXT:    flat_store_dword v[2:3], v0
 ; GISEL-NEXT:    s_waitcnt vmcnt(0) lgkmcnt(0)
 ; GISEL-NEXT:    s_setpc_b64 s[30:31]
-; GISEL-NEXT:  .LBB1_3: ; Inline asm indirect target
-; GISEL-NEXT:    ; %kill
-; GISEL-NEXT:    ; Label of block must be emitted
-; GISEL-NEXT:    ; divergent unreachable
-; GISEL-NEXT:    s_waitcnt vmcnt(0) lgkmcnt(0)
-; GISEL-NEXT:    s_setpc_b64 s[30:31]
-; GISEL-NEXT:  .LBB1_4:
+; GISEL-NEXT:  .LBB1_2:
 ; GISEL-NEXT:    s_mov_b64 exec, 0
 ; GISEL-NEXT:    s_endpgm
   %a = load i32, ptr %src, align 4



More information about the llvm-commits mailing list