[llvm] [AMDGPU] Fix mode register intersection across predecessors (PR #222244)

Arseniy Obolenskiy via llvm-commits llvm-commits at lists.llvm.org
Thu Sep 17 05:51:39 PDT 2026


https://github.com/aobolensk updated https://github.com/llvm/llvm-project/pull/222244

>From 9cdb289bda5bef4e4c365c7facb10d8e577f89a7 Mon Sep 17 00:00:00 2001
From: Arseniy Obolenskiy <arseniy.obolenskiy at amd.com>
Date: Wed, 9 Sep 2026 07:47:55 +0200
Subject: [PATCH 1/4] [AMDGPU] Fix mode register intersection across
 predecessors

The merge loop tested a stale ExitSet flag from an earlier visit, so each known predecessor overwrote the entry mode instead of intersecting into it
---
 llvm/lib/Target/AMDGPU/SIModeRegister.cpp     | 49 ++++++++--------
 .../AMDGPU/mode-register-fpconstrain.ll       | 51 +++++++++++++++++
 llvm/test/CodeGen/AMDGPU/mode-register.mir    | 56 +++++++++++++++++++
 3 files changed, 129 insertions(+), 27 deletions(-)

diff --git a/llvm/lib/Target/AMDGPU/SIModeRegister.cpp b/llvm/lib/Target/AMDGPU/SIModeRegister.cpp
index dbe27a8030d4e..6bb99944facfc 100644
--- a/llvm/lib/Target/AMDGPU/SIModeRegister.cpp
+++ b/llvm/lib/Target/AMDGPU/SIModeRegister.cpp
@@ -372,37 +372,32 @@ void SIModeRegister::processBlockPhase2(MachineBasicBlock &MBB,
     // Mask bits (which represent the Mode bits with a known value) can only be
     // added by explicit SETREG instructions or the initial default value -
     // the intersection process may remove Mask bits.
-    // If we find a predecessor that has not yet had an exit value determined
-    // (this can happen for example if a block is its own predecessor) we defer
-    // use of that value as the Mask will be all zero, and we will revisit this
-    // block again later (unless the only predecessor without an exit value is
-    // this block).
-    MachineBasicBlock::pred_iterator P = MBB.pred_begin(), E = MBB.pred_end();
-    MachineBasicBlock &PB = *(*P);
-    unsigned PredBlock = PB.getNumber();
-    if ((ThisBlock == PredBlock) && (std::next(P) == E)) {
-      BlockInfo[ThisBlock]->Pred = DefaultStatus;
+    BlockData &Info = *BlockInfo[ThisBlock];
+    bool SelfPredPending = false;
+    // An entry block is also entered from the function entry, so the default
+    // applies even when it has predecessors.
+    if (MBB.isEntryBlock()) {
+      Info.Pred = DefaultStatus;
       ExitSet = true;
-    } else if (BlockInfo[PredBlock]->ExitSet) {
-      BlockInfo[ThisBlock]->Pred = BlockInfo[PredBlock]->Exit;
-      ExitSet = true;
-    } else if (PredBlock != ThisBlock)
-      RevisitRequired = true;
-
-    for (P = std::next(P); P != E; P = std::next(P)) {
-      MachineBasicBlock *Pred = *P;
+    }
+    for (MachineBasicBlock *Pred : MBB.predecessors()) {
       unsigned PredBlock = Pred->getNumber();
-      if (BlockInfo[PredBlock]->ExitSet) {
-        if (BlockInfo[ThisBlock]->ExitSet) {
-          BlockInfo[ThisBlock]->Pred =
-              BlockInfo[ThisBlock]->Pred.intersect(BlockInfo[PredBlock]->Exit);
-        } else {
-          BlockInfo[ThisBlock]->Pred = BlockInfo[PredBlock]->Exit;
-        }
+      const BlockData &PredInfo = *BlockInfo[PredBlock];
+      if (!PredInfo.ExitSet) {
+        if (PredBlock == ThisBlock)
+          SelfPredPending = true;
+        else
+          RevisitRequired = true;
+      } else if (ExitSet) {
+        Info.Pred = Info.Pred.intersect(PredInfo.Exit);
+      } else {
+        Info.Pred = PredInfo.Exit;
         ExitSet = true;
-      } else if (PredBlock != ThisBlock)
-        RevisitRequired = true;
+      }
     }
+    // ExitSet gating stops an unreachable self-only block from requeuing
+    // forever, as its exit never becomes known.
+    RevisitRequired |= SelfPredPending && ExitSet;
   }
   Status TmpStatus =
       BlockInfo[ThisBlock]->Pred.merge(BlockInfo[ThisBlock]->Change);
diff --git a/llvm/test/CodeGen/AMDGPU/mode-register-fpconstrain.ll b/llvm/test/CodeGen/AMDGPU/mode-register-fpconstrain.ll
index 174435795c5ef..a7932eca0e4ba 100644
--- a/llvm/test/CodeGen/AMDGPU/mode-register-fpconstrain.ll
+++ b/llvm/test/CodeGen/AMDGPU/mode-register-fpconstrain.ll
@@ -30,6 +30,57 @@ entry:
   ret double %val
 }
 
+; The entry fadd is load-bearing: it makes every loop predecessor exit stable
+; from the start, which is what hid the loop exit from phase-2 intersection.
+
+define amdgpu_kernel void @loop_carried_round_mode(ptr addrspace(1) %out, double %a, double %b, i32 %n) {
+; GCN-LABEL: loop_carried_round_mode:
+; GCN:       ; %bb.0: ; %entry
+; GCN-NEXT:    s_load_dwordx2 s[4:5], s[8:9], 0x10
+; GCN-NEXT:    s_load_dwordx4 s[0:3], s[8:9], 0x0
+; GCN-NEXT:    s_load_dword s6, s[8:9], 0x18
+; GCN-NEXT:    v_mov_b32_e32 v2, 0
+; GCN-NEXT:    s_mov_b32 s7, 0
+; GCN-NEXT:    s_waitcnt lgkmcnt(0)
+; GCN-NEXT:    v_mov_b32_e32 v0, s4
+; GCN-NEXT:    v_mov_b32_e32 v1, s5
+; GCN-NEXT:    v_add_f64 v[0:1], s[2:3], v[0:1]
+; GCN-NEXT:    global_store_dwordx2 v2, v[0:1], s[0:1]
+; GCN-NEXT:    s_waitcnt vmcnt(0)
+; GCN-NEXT:    v_mov_b32_e32 v0, s2
+; GCN-NEXT:    v_mov_b32_e32 v1, s3
+; GCN-NEXT:  .LBB2_1: ; %loop
+; GCN-NEXT:    ; =>This Inner Loop Header: Depth=1
+; GCN-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_MODE, 2, 1), 0
+; GCN-NEXT:    v_add_f64 v[0:1], v[0:1], s[4:5]
+; GCN-NEXT:    s_add_i32 s7, s7, 1
+; GCN-NEXT:    s_cmp_lt_i32 s7, s6
+; GCN-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_MODE, 2, 2), 1
+; GCN-NEXT:    v_cvt_f32_f64_e32 v3, v[0:1]
+; GCN-NEXT:    global_store_dword v2, v3, s[0:1]
+; GCN-NEXT:    s_waitcnt vmcnt(0)
+; GCN-NEXT:    s_cbranch_scc1 .LBB2_1
+; GCN-NEXT:  ; %bb.2: ; %exit
+; GCN-NEXT:    s_endpgm
+entry:
+  %e = fadd double %a, %b
+  store volatile double %e, ptr addrspace(1) %out
+  br label %loop
+
+loop:
+  %i = phi i32 [ 0, %entry ], [ %i.next, %loop ]
+  %acc = phi double [ %a, %entry ], [ %sum, %loop ]
+  %sum = fadd double %acc, %b
+  %t = call float @llvm.fptrunc.round.f32.f64(double %sum, metadata !"round.upward")
+  store volatile float %t, ptr addrspace(1) %out
+  %i.next = add i32 %i, 1
+  %cc = icmp slt i32 %i.next, %n
+  br i1 %cc, label %loop, label %exit
+
+exit:
+  ret void
+}
+
 declare void @llvm.amdgcn.s.setreg(i32 immarg, i32)
 
 declare double @llvm.experimental.constrained.fadd.f64(double, double, metadata, metadata)
diff --git a/llvm/test/CodeGen/AMDGPU/mode-register.mir b/llvm/test/CodeGen/AMDGPU/mode-register.mir
index 1dd6499ca2733..24611e4238257 100644
--- a/llvm/test/CodeGen/AMDGPU/mode-register.mir
+++ b/llvm/test/CodeGen/AMDGPU/mode-register.mir
@@ -510,3 +510,59 @@ body: |
   bb.2:
     S_ENDPGM 0
 ...
+---
+# Predecessor exits used to overwrite rather than intersect, so the last
+# predecessor won and the RTN restore in bb.3 was dropped.
+# CHECK-LABEL: name: join_pred_intersect
+# CHECK-LABEL: bb.3:
+# CHECK: S_SETREG_IMM32_B32 0, 2177
+# CHECK-NEXT: V_ADD_F64_e64
+
+name: join_pred_intersect
+
+body: |
+  bb.0:
+    successors: %bb.1, %bb.2
+    S_CBRANCH_VCCZ %bb.2, implicit $vcc
+    S_BRANCH %bb.1
+
+  bb.1:
+    successors: %bb.3
+    S_SETREG_IMM32_B32 3, 2177, implicit-def $mode, implicit $mode
+    S_BRANCH %bb.3
+
+  bb.2:
+    successors: %bb.3
+    liveins: $vgpr0_vgpr1, $vgpr2_vgpr3
+    $vgpr4_vgpr5 = V_ADD_F64_e64 0, $vgpr0_vgpr1, 0, $vgpr2_vgpr3, 0, 0, implicit $mode, implicit $exec
+    S_BRANCH %bb.3
+
+  bb.3:
+    liveins: $vgpr0_vgpr1, $vgpr2_vgpr3
+    $vgpr6_vgpr7 = V_ADD_F64_e64 0, $vgpr0_vgpr1, 0, $vgpr2_vgpr3, 0, 0, implicit $mode, implicit $exec
+    S_ENDPGM 0
+...
+---
+# A self-looping entry block used to be pinned to the default status without
+# folding in its own exit, dropping the RTN restore for the V_ADD_F64.
+# CHECK-LABEL: name: entry_self_loop_carried_mode
+# CHECK-LABEL: bb.0:
+# CHECK: S_SETREG_IMM32_B32 0, 2177
+# CHECK-NEXT: V_ADD_F64_e64
+# CHECK: S_SETREG_IMM32_B32 3, 2177
+# CHECK-NEXT: V_CVT_F32_F64_e32
+
+name: entry_self_loop_carried_mode
+
+body: |
+  bb.0:
+    successors: %bb.0, %bb.1
+    liveins: $vgpr0_vgpr1, $vgpr2_vgpr3
+    $vgpr4_vgpr5 = V_ADD_F64_e64 0, $vgpr0_vgpr1, 0, $vgpr2_vgpr3, 0, 0, implicit $mode, implicit $exec
+    $vgpr6 = FPTRUNC_ROUND_F32_F64_PSEUDO $vgpr4_vgpr5, 3, implicit $mode, implicit $exec
+    S_CBRANCH_VCCZ %bb.0, implicit $vcc
+    S_BRANCH %bb.1
+
+  bb.1:
+    S_ENDPGM 0
+...

>From f7299501cc96cb2d00573a20628152d49780e32c Mon Sep 17 00:00:00 2001
From: Arseniy Obolenskiy <arseniy.obolenskiy at amd.com>
Date: Sun, 13 Sep 2026 21:00:07 +0200
Subject: [PATCH 2/4] address comments

---
 llvm/lib/Target/AMDGPU/SIModeRegister.cpp  | 11 +++--------
 llvm/test/CodeGen/AMDGPU/mode-register.mir | 22 +++++++++++++---------
 2 files changed, 16 insertions(+), 17 deletions(-)

diff --git a/llvm/lib/Target/AMDGPU/SIModeRegister.cpp b/llvm/lib/Target/AMDGPU/SIModeRegister.cpp
index 6bb99944facfc..93371478b9a89 100644
--- a/llvm/lib/Target/AMDGPU/SIModeRegister.cpp
+++ b/llvm/lib/Target/AMDGPU/SIModeRegister.cpp
@@ -362,8 +362,9 @@ void SIModeRegister::processBlockPhase2(MachineBasicBlock &MBB,
   bool RevisitRequired = false;
   bool ExitSet = false;
   unsigned ThisBlock = MBB.getNumber();
-  if (MBB.pred_empty()) {
-    // There are no predecessors, so use the default starting status.
+  if (MBB.pred_empty() || MBB.isEntryBlock()) {
+    // There are no predecessors, or the block is only entered from the function
+    // entry, so use the default starting status.
     BlockInfo[ThisBlock]->Pred = DefaultStatus;
     ExitSet = true;
   } else {
@@ -374,12 +375,6 @@ void SIModeRegister::processBlockPhase2(MachineBasicBlock &MBB,
     // the intersection process may remove Mask bits.
     BlockData &Info = *BlockInfo[ThisBlock];
     bool SelfPredPending = false;
-    // An entry block is also entered from the function entry, so the default
-    // applies even when it has predecessors.
-    if (MBB.isEntryBlock()) {
-      Info.Pred = DefaultStatus;
-      ExitSet = true;
-    }
     for (MachineBasicBlock *Pred : MBB.predecessors()) {
       unsigned PredBlock = Pred->getNumber();
       const BlockData &PredInfo = *BlockInfo[PredBlock];
diff --git a/llvm/test/CodeGen/AMDGPU/mode-register.mir b/llvm/test/CodeGen/AMDGPU/mode-register.mir
index 24611e4238257..f48aebbf26951 100644
--- a/llvm/test/CodeGen/AMDGPU/mode-register.mir
+++ b/llvm/test/CodeGen/AMDGPU/mode-register.mir
@@ -543,26 +543,30 @@ body: |
     S_ENDPGM 0
 ...
 ---
-# A self-looping entry block used to be pinned to the default status without
-# folding in its own exit, dropping the RTN restore for the V_ADD_F64.
-# CHECK-LABEL: name: entry_self_loop_carried_mode
-# CHECK-LABEL: bb.0:
+# A self-looping block only folded in the exit of its other predecessor, so the
+# RTN restore for the V_ADD_F64 on the loop-carried edge was dropped.
+# CHECK-LABEL: name: self_loop_carried_mode
+# CHECK-LABEL: bb.1:
 # CHECK: S_SETREG_IMM32_B32 0, 2177
 # CHECK-NEXT: V_ADD_F64_e64
 # CHECK: S_SETREG_IMM32_B32 3, 2177
 # CHECK-NEXT: V_CVT_F32_F64_e32
 
-name: entry_self_loop_carried_mode
+name: self_loop_carried_mode
 
 body: |
   bb.0:
-    successors: %bb.0, %bb.1
+    successors: %bb.1
+    S_BRANCH %bb.1
+
+  bb.1:
+    successors: %bb.1, %bb.2
     liveins: $vgpr0_vgpr1, $vgpr2_vgpr3
     $vgpr4_vgpr5 = V_ADD_F64_e64 0, $vgpr0_vgpr1, 0, $vgpr2_vgpr3, 0, 0, implicit $mode, implicit $exec
     $vgpr6 = FPTRUNC_ROUND_F32_F64_PSEUDO $vgpr4_vgpr5, 3, implicit $mode, implicit $exec
-    S_CBRANCH_VCCZ %bb.0, implicit $vcc
-    S_BRANCH %bb.1
+    S_CBRANCH_VCCZ %bb.1, implicit $vcc
+    S_BRANCH %bb.2
 
-  bb.1:
+  bb.2:
     S_ENDPGM 0
 ...

>From 0e879c03455f641b09b97cd21bbf257501108c34 Mon Sep 17 00:00:00 2001
From: Arseniy Obolenskiy <arseniy.obolenskiy at amd.com>
Date: Thu, 17 Sep 2026 11:56:11 +0200
Subject: [PATCH 3/4] rm

---
 llvm/lib/Target/AMDGPU/SIModeRegister.cpp | 5 ++---
 1 file changed, 2 insertions(+), 3 deletions(-)

diff --git a/llvm/lib/Target/AMDGPU/SIModeRegister.cpp b/llvm/lib/Target/AMDGPU/SIModeRegister.cpp
index 93371478b9a89..cc95f2d73d3f0 100644
--- a/llvm/lib/Target/AMDGPU/SIModeRegister.cpp
+++ b/llvm/lib/Target/AMDGPU/SIModeRegister.cpp
@@ -362,9 +362,8 @@ void SIModeRegister::processBlockPhase2(MachineBasicBlock &MBB,
   bool RevisitRequired = false;
   bool ExitSet = false;
   unsigned ThisBlock = MBB.getNumber();
-  if (MBB.pred_empty() || MBB.isEntryBlock()) {
-    // There are no predecessors, or the block is only entered from the function
-    // entry, so use the default starting status.
+  if (MBB.pred_empty()) {
+    // There are no predecessors, so use the default starting status.
     BlockInfo[ThisBlock]->Pred = DefaultStatus;
     ExitSet = true;
   } else {

>From d88ba4fee2d7681f107f04eb897df9e31b7103ed Mon Sep 17 00:00:00 2001
From: Arseniy Obolenskiy <arseniy.obolenskiy at amd.com>
Date: Thu, 17 Sep 2026 14:51:23 +0200
Subject: [PATCH 4/4] fix: Seed mode register dataflow from entry and
 unreachable blocks

---
 llvm/lib/Target/AMDGPU/SIModeRegister.cpp     | 82 +++++++++---------
 .../AMDGPU/mode-register-fpconstrain.ll       |  4 +-
 llvm/test/CodeGen/AMDGPU/mode-register.mir    | 86 ++++++++++++++++++-
 3 files changed, 124 insertions(+), 48 deletions(-)

diff --git a/llvm/lib/Target/AMDGPU/SIModeRegister.cpp b/llvm/lib/Target/AMDGPU/SIModeRegister.cpp
index cc95f2d73d3f0..1cc05f592b220 100644
--- a/llvm/lib/Target/AMDGPU/SIModeRegister.cpp
+++ b/llvm/lib/Target/AMDGPU/SIModeRegister.cpp
@@ -15,6 +15,7 @@
 //
 #include "AMDGPU.h"
 #include "GCNSubtarget.h"
+#include "llvm/ADT/DepthFirstIterator.h"
 #include "llvm/ADT/Statistic.h"
 #include "llvm/CodeGen/MachineFunctionPass.h"
 #include <queue>
@@ -360,49 +361,37 @@ void SIModeRegister::processBlockPhase1(MachineBasicBlock &MBB,
 void SIModeRegister::processBlockPhase2(MachineBasicBlock &MBB,
                                         const SIInstrInfo *TII) {
   bool RevisitRequired = false;
-  bool ExitSet = false;
-  unsigned ThisBlock = MBB.getNumber();
-  if (MBB.pred_empty()) {
-    // There are no predecessors, so use the default starting status.
-    BlockInfo[ThisBlock]->Pred = DefaultStatus;
-    ExitSet = true;
-  } else {
-    // Build a status that is common to all the predecessors by intersecting
-    // all the predecessor exit status values.
-    // Mask bits (which represent the Mode bits with a known value) can only be
-    // added by explicit SETREG instructions or the initial default value -
-    // the intersection process may remove Mask bits.
-    BlockData &Info = *BlockInfo[ThisBlock];
-    bool SelfPredPending = false;
-    for (MachineBasicBlock *Pred : MBB.predecessors()) {
-      unsigned PredBlock = Pred->getNumber();
-      const BlockData &PredInfo = *BlockInfo[PredBlock];
-      if (!PredInfo.ExitSet) {
-        if (PredBlock == ThisBlock)
-          SelfPredPending = true;
-        else
-          RevisitRequired = true;
-      } else if (ExitSet) {
-        Info.Pred = Info.Pred.intersect(PredInfo.Exit);
-      } else {
-        Info.Pred = PredInfo.Exit;
-        ExitSet = true;
-      }
+  BlockData &Info = *BlockInfo[MBB.getNumber()];
+  // The entry block is entered with the default mode even if it has preds.
+  bool ExitSet = MBB.isEntryBlock();
+  if (ExitSet)
+    Info.Pred = DefaultStatus;
+  // Build a status that is common to all the predecessors by intersecting
+  // all the predecessor exit status values.
+  // Mask bits (which represent the Mode bits with a known value) can only be
+  // added by explicit SETREG instructions or the initial default value -
+  // the intersection process may remove Mask bits.
+  // A predecessor with no exit value yet defers the block rather than guessing.
+  for (MachineBasicBlock *Pred : MBB.predecessors()) {
+    const BlockData &PredInfo = *BlockInfo[Pred->getNumber()];
+    if (!PredInfo.ExitSet) {
+      RevisitRequired = true;
+    } else if (ExitSet) {
+      Info.Pred = Info.Pred.intersect(PredInfo.Exit);
+    } else {
+      Info.Pred = PredInfo.Exit;
+      ExitSet = true;
     }
-    // ExitSet gating stops an unreachable self-only block from requeuing
-    // forever, as its exit never becomes known.
-    RevisitRequired |= SelfPredPending && ExitSet;
   }
-  Status TmpStatus =
-      BlockInfo[ThisBlock]->Pred.merge(BlockInfo[ThisBlock]->Change);
-  if (BlockInfo[ThisBlock]->Exit != TmpStatus) {
-    BlockInfo[ThisBlock]->Exit = TmpStatus;
+  Status TmpStatus = Info.Pred.merge(Info.Change);
+  if (Info.Exit != TmpStatus) {
+    Info.Exit = TmpStatus;
     // Add the successors to the work list so we can propagate the changed exit
     // status.
     for (MachineBasicBlock *Succ : MBB.successors())
       Phase2List.push(Succ);
   }
-  BlockInfo[ThisBlock]->ExitSet = ExitSet;
+  Info.ExitSet = ExitSet;
   if (RevisitRequired)
     Phase2List.push(&MBB);
 }
@@ -445,6 +434,8 @@ bool SIModeRegister::run(MachineFunction &MF) {
   const Function &F = MF.getFunction();
   if (F.hasFnAttribute(llvm::Attribute::StrictFP))
     return Changed;
+  if (MF.empty())
+    return Changed;
   BlockInfo.resize(MF.getNumBlockIDs());
   const GCNSubtarget &ST = MF.getSubtarget<GCNSubtarget>();
   const SIInstrInfo *TII = ST.getInstrInfo();
@@ -456,11 +447,20 @@ bool SIModeRegister::run(MachineFunction &MF) {
   for (MachineBasicBlock &BB : MF)
     processBlockPhase1(BB, TII);
 
-  // Phase 2 - determine the exit mode from each block. We add all blocks to the
-  // list here, but will also add any that need to be revisited during Phase 2
-  // processing.
-  for (MachineBasicBlock &BB : MF)
-    Phase2List.push(&BB);
+  // Phase 2 - determine the exit mode from each block. We add all reachable
+  // blocks to the list here, and any that need revisiting during Phase 2.
+  // Unreachable blocks never derive an exit value, so they are seeded instead.
+  df_iterator_default_set<MachineBasicBlock *> Reachable;
+  for (MachineBasicBlock *BB : depth_first_ext(&MF.front(), Reachable))
+    Phase2List.push(BB);
+  for (MachineBasicBlock &BB : MF) {
+    if (Reachable.contains(&BB))
+      continue;
+    BlockData &Info = *BlockInfo[BB.getNumber()];
+    Info.Pred = DefaultStatus;
+    Info.Exit = Info.Pred.merge(Info.Change);
+    Info.ExitSet = true;
+  }
   while (!Phase2List.empty()) {
     processBlockPhase2(*Phase2List.front(), TII);
     Phase2List.pop();
diff --git a/llvm/test/CodeGen/AMDGPU/mode-register-fpconstrain.ll b/llvm/test/CodeGen/AMDGPU/mode-register-fpconstrain.ll
index a7932eca0e4ba..0a9a5f4ec31fa 100644
--- a/llvm/test/CodeGen/AMDGPU/mode-register-fpconstrain.ll
+++ b/llvm/test/CodeGen/AMDGPU/mode-register-fpconstrain.ll
@@ -30,9 +30,7 @@ entry:
   ret double %val
 }
 
-; The entry fadd is load-bearing: it makes every loop predecessor exit stable
-; from the start, which is what hid the loop exit from phase-2 intersection.
-
+; The entry fadd is load-bearing: it gives every loop predecessor a stable exit.
 define amdgpu_kernel void @loop_carried_round_mode(ptr addrspace(1) %out, double %a, double %b, i32 %n) {
 ; GCN-LABEL: loop_carried_round_mode:
 ; GCN:       ; %bb.0: ; %entry
diff --git a/llvm/test/CodeGen/AMDGPU/mode-register.mir b/llvm/test/CodeGen/AMDGPU/mode-register.mir
index f48aebbf26951..8a20596d9a620 100644
--- a/llvm/test/CodeGen/AMDGPU/mode-register.mir
+++ b/llvm/test/CodeGen/AMDGPU/mode-register.mir
@@ -511,8 +511,7 @@ body: |
     S_ENDPGM 0
 ...
 ---
-# Predecessor exits used to overwrite rather than intersect, so the last
-# predecessor won and the RTN restore in bb.3 was dropped.
+# check that predecessor exits are intersected, not overwritten, at a join
 # CHECK-LABEL: name: join_pred_intersect
 # CHECK-LABEL: bb.3:
 # CHECK: S_SETREG_IMM32_B32 0, 2177
@@ -543,8 +542,8 @@ body: |
     S_ENDPGM 0
 ...
 ---
-# A self-looping block only folded in the exit of its other predecessor, so the
-# RTN restore for the V_ADD_F64 on the loop-carried edge was dropped.
+# check that a self-looping block intersects its own exit with its other
+# predecessor
 # CHECK-LABEL: name: self_loop_carried_mode
 # CHECK-LABEL: bb.1:
 # CHECK: S_SETREG_IMM32_B32 0, 2177
@@ -570,3 +569,82 @@ body: |
   bb.2:
     S_ENDPGM 0
 ...
+---
+# check that a block whose only predecessor is itself intersects the
+# loop-carried exit with the default, rather than pinning the default
+# CHECK-LABEL: name: self_only_pred_mode
+# CHECK-LABEL: bb.0:
+# CHECK: S_SETREG_IMM32_B32 0, 2177
+# CHECK-NEXT: V_ADD_F64_e64
+# CHECK-NEXT: S_SETREG_IMM32_B32 3, 2177
+# CHECK-NEXT: V_CVT_F32_F64_e32
+# CHECK-NOT: S_SETREG_IMM32_B32
+
+name: self_only_pred_mode
+
+body: |
+  bb.0:
+    successors: %bb.0, %bb.1
+    liveins: $vgpr0_vgpr1, $vgpr2_vgpr3
+    $vgpr4_vgpr5 = V_ADD_F64_e64 0, $vgpr0_vgpr1, 0, $vgpr2_vgpr3, 0, 0, implicit $mode, implicit $exec
+    $vgpr6 = FPTRUNC_ROUND_F32_F64_PSEUDO $vgpr4_vgpr5, 3, implicit $mode, implicit $exec
+    S_CBRANCH_VCCZ %bb.0, implicit $vcc
+    S_BRANCH %bb.1
+
+  bb.1:
+    S_ENDPGM 0
+...
+---
+# check that an unreachable cycle is not requeued forever
+# CHECK-LABEL: name: unreachable_cycle
+# CHECK-NOT: S_SETREG_IMM32_B32
+
+name: unreachable_cycle
+
+body: |
+  bb.0:
+    S_ENDPGM 0
+
+  bb.1:
+    successors: %bb.2
+    S_BRANCH %bb.2
+
+  bb.2:
+    successors: %bb.1
+    S_BRANCH %bb.1
+...
+---
+# check that a block whose predecessors are all later in layout order is not
+# given a guessed default status, which the intersection could never recover from
+# CHECK-LABEL: name: pred_later_in_layout
+# CHECK: S_SETREG_IMM32_B32 3, 2177
+# CHECK-NOT: S_SETREG_IMM32_B32
+
+name: pred_later_in_layout
+
+body: |
+  bb.0:
+    successors: %bb.2
+    S_SETREG_IMM32_B32 3, 2177, implicit-def $mode, implicit $mode
+    S_BRANCH %bb.2
+
+  bb.1:
+    successors: %bb.2, %bb.3
+    S_CBRANCH_VCCZ %bb.2, implicit $vcc
+    S_BRANCH %bb.3
+
+  bb.2:
+    successors: %bb.1
+    S_BRANCH %bb.1
+
+  bb.3:
+    liveins: $vgpr4_vgpr5
+    $vgpr6 = FPTRUNC_ROUND_F32_F64_PSEUDO $vgpr4_vgpr5, 3, implicit $mode, implicit $exec
+    S_ENDPGM 0
+...
+---
+# check that a function with no blocks does not crash phase 2
+# CHECK-LABEL: name: empty_body
+
+name: empty_body
+...



More information about the llvm-commits mailing list