[llvm] AMDGPU: Mark the SCC def dead when expanding CF pseudos (PR #226249)
via llvm-commits
llvm-commits at lists.llvm.org
Thu Sep 24 10:49:43 PDT 2026
llvmorg-github-actions[bot] wrote:
<!--LLVM PR SUMMARY COMMENT-->
@llvm/pr-subscribers-backend-amdgpu
Author: Matt Arsenault (arsenm)
<details>
<summary>Changes</summary>
The control flow pseudos start with dead flags on the implicit scc def,
but did not transfer to the replacement instruction
Co-Authored-By: Claude Opus 5 <noreply@<!-- -->anthropic.com>
---
Patch is 78.70 KiB, truncated to 20.00 KiB below, full version: https://github.com/llvm/llvm-project/pull/226249.diff
19 Files Affected:
- (modified) llvm/lib/Target/AMDGPU/SILowerControlFlow.cpp (+19-6)
- (modified) llvm/test/CodeGen/AMDGPU/GlobalISel/divergence-divergent-i1-phis-no-lane-mask-merging.ll (+3-3)
- (modified) llvm/test/CodeGen/AMDGPU/GlobalISel/divergence-divergent-i1-used-outside-loop.ll (+24-27)
- (modified) llvm/test/CodeGen/AMDGPU/GlobalISel/divergence-structurizer.ll (+3-3)
- (modified) llvm/test/CodeGen/AMDGPU/GlobalISel/divergence-temporal-divergent-i1.ll (+16-16)
- (modified) llvm/test/CodeGen/AMDGPU/block-should-not-be-in-alive-blocks.mir (+4-4)
- (modified) llvm/test/CodeGen/AMDGPU/branch-folding-implicit-def-subreg.ll (+20-20)
- (modified) llvm/test/CodeGen/AMDGPU/collapse-endcf.mir (+25-25)
- (modified) llvm/test/CodeGen/AMDGPU/divergent-branch-uniform-condition.ll (+2-2)
- (modified) llvm/test/CodeGen/AMDGPU/i1-copy-from-loop.ll (+1-1)
- (modified) llvm/test/CodeGen/AMDGPU/insert-waitcnts-merge.ll (+5-5)
- (modified) llvm/test/CodeGen/AMDGPU/loop_break.ll (+5-5)
- (modified) llvm/test/CodeGen/AMDGPU/lower-control-flow-live-intervals.mir (+10-10)
- (modified) llvm/test/CodeGen/AMDGPU/lower-control-flow-live-variables-update.mir (+3-3)
- (modified) llvm/test/CodeGen/AMDGPU/lower-control-flow-other-terminators.mir (+5-5)
- (modified) llvm/test/CodeGen/AMDGPU/sgpr-scavenge-fi-stack-id.ll (+4-5)
- (modified) llvm/test/CodeGen/AMDGPU/si-lower-control-flow-remove-redundant-block-liveintervals.mir (+3-3)
- (modified) llvm/test/CodeGen/AMDGPU/si-lower-control-flow.mir (+6-6)
- (modified) llvm/test/CodeGen/AMDGPU/transform-block-with-return-to-epilog.ll (+6-6)
``````````diff
diff --git a/llvm/lib/Target/AMDGPU/SILowerControlFlow.cpp b/llvm/lib/Target/AMDGPU/SILowerControlFlow.cpp
index 6c87ba5438f3d..9dccac3b9d8bf 100644
--- a/llvm/lib/Target/AMDGPU/SILowerControlFlow.cpp
+++ b/llvm/lib/Target/AMDGPU/SILowerControlFlow.cpp
@@ -180,6 +180,11 @@ static void setImpSCCDefDead(MachineInstr &MI, bool IsDead) {
ImpDefSCC.setIsDead(IsDead);
}
+static void copySCCDefDead(MachineInstr &MI, const MachineOperand &OrigSCCDef) {
+ assert(OrigSCCDef.getReg() == AMDGPU::SCC && OrigSCCDef.isDef());
+ setImpSCCDefDead(MI, OrigSCCDef.isDead());
+}
+
char &llvm::SILowerControlFlowLegacyID = SILowerControlFlowLegacy::ID;
bool SILowerControlFlow::hasKill(const MachineBasicBlock *Begin,
@@ -221,9 +226,6 @@ void SILowerControlFlow::emitIf(MachineInstr &MI) {
MachineOperand& Cond = MI.getOperand(1);
assert(Cond.getSubReg() == AMDGPU::NoSubRegister);
- MachineOperand &ImpDefSCC = MI.getOperand(4);
- assert(ImpDefSCC.getReg() == AMDGPU::SCC && ImpDefSCC.isDef());
-
// If there is only one use of save exec register and that use is SI_END_CF,
// we can optimize SI_IF by returning the full saved exec mask instead of
// just cleared bits.
@@ -249,17 +251,17 @@ void SILowerControlFlow::emitIf(MachineInstr &MI) {
MachineInstr *And =
BuildMI(MBB, I, DL, TII->get(LMC.AndOpc), Tmp).addReg(CopyReg).add(Cond);
+ setImpSCCDefDead(*And, /*IsDead=*/true);
+
if (LV)
LV->replaceKillInstruction(Cond.getReg(), MI, *And);
- setImpSCCDefDead(*And, true);
-
MachineInstr *Xor = nullptr;
if (!SimpleIf) {
Xor = BuildMI(MBB, I, DL, TII->get(LMC.XorOpc), SaveExecReg)
.addReg(Tmp)
.addReg(CopyReg);
- setImpSCCDefDead(*Xor, ImpDefSCC.isDead());
+ copySCCDefDead(*Xor, MI.getOperand(4));
}
// Use a copy that is a terminator to get correct spill code placement it with
@@ -321,6 +323,7 @@ void SILowerControlFlow::emitElse(MachineInstr &MI) {
MachineInstr *OrSaveExec =
BuildMI(MBB, Start, DL, TII->get(LMC.OrSaveExecOpc), SaveReg)
.add(MI.getOperand(1)); // Saved EXEC
+ setImpSCCDefDead(*OrSaveExec, /*IsDead=*/true);
if (LV)
LV->replaceKillInstruction(SrcReg, MI, *OrSaveExec);
@@ -333,11 +336,13 @@ void SILowerControlFlow::emitElse(MachineInstr &MI) {
MachineInstr *And = BuildMI(MBB, ElsePt, DL, TII->get(LMC.AndOpc), DstReg)
.addReg(LMC.ExecReg)
.addReg(SaveReg);
+ setImpSCCDefDead(*And, /*IsDead=*/true);
MachineInstr *Xor =
BuildMI(MBB, ElsePt, DL, TII->get(LMC.XorTermOpc), LMC.ExecReg)
.addReg(LMC.ExecReg)
.addReg(DstReg);
+ copySCCDefDead(*Xor, MI.getOperand(4));
// Skip ahead to the unconditional branch in case there are other terminators
// present.
@@ -392,6 +397,7 @@ void SILowerControlFlow::emitIfBreak(MachineInstr &MI) {
And = BuildMI(MBB, &MI, DL, TII->get(LMC.AndOpc), AndReg)
.addReg(LMC.ExecReg)
.add(MI.getOperand(1));
+ setImpSCCDefDead(*And, /*IsDead=*/true);
if (LV)
LV->replaceKillInstruction(MI.getOperand(1).getReg(), MI, *And);
Or = BuildMI(MBB, &MI, DL, TII->get(LMC.OrOpc), Dst)
@@ -404,6 +410,9 @@ void SILowerControlFlow::emitIfBreak(MachineInstr &MI) {
if (LV)
LV->replaceKillInstruction(MI.getOperand(1).getReg(), MI, *Or);
}
+
+ copySCCDefDead(*Or, MI.getOperand(3));
+
if (LV)
LV->replaceKillInstruction(MI.getOperand(2).getReg(), MI, *Or);
@@ -428,6 +437,8 @@ void SILowerControlFlow::emitLoop(MachineInstr &MI) {
BuildMI(MBB, &MI, DL, TII->get(LMC.AndN2TermOpc), LMC.ExecReg)
.addReg(LMC.ExecReg)
.add(MI.getOperand(0));
+ copySCCDefDead(*AndN2, MI.getOperand(3));
+
if (LV)
LV->replaceKillInstruction(MI.getOperand(0).getReg(), MI, *AndN2);
@@ -518,6 +529,8 @@ MachineBasicBlock *SILowerControlFlow::emitEndCf(MachineInstr &MI) {
MachineInstr *NewMI = BuildMI(MBB, InsPt, DL, TII->get(Opcode), LMC.ExecReg)
.addReg(LMC.ExecReg)
.add(MI.getOperand(0));
+ copySCCDefDead(*NewMI, MI.getOperand(2));
+
if (LV) {
LV->replaceKillInstruction(DataReg, MI, *NewMI);
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/divergence-divergent-i1-phis-no-lane-mask-merging.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/divergence-divergent-i1-phis-no-lane-mask-merging.ll
index 4371804ed1bdd..ef77dce2b0c9c 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/divergence-divergent-i1-phis-no-lane-mask-merging.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/divergence-divergent-i1-phis-no-lane-mask-merging.ll
@@ -112,11 +112,11 @@ define void @divergent_i1_phi_used_inside_loop(float %val, ptr %addr) {
; GFX10-NEXT: s_cmp_lg_u32 s8, 0
; GFX10-NEXT: s_cselect_b32 s8, exec_lo, 0
; GFX10-NEXT: v_cmp_gt_f32_e32 vcc_lo, v3, v0
+; GFX10-NEXT: s_andn2_b32 s7, s7, exec_lo
+; GFX10-NEXT: s_and_b32 s8, exec_lo, s8
; GFX10-NEXT: s_xor_b32 s5, s5, 1
; GFX10-NEXT: s_add_i32 s6, s6, 1
; GFX10-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX10-NEXT: s_andn2_b32 s7, s7, exec_lo
-; GFX10-NEXT: s_and_b32 s8, exec_lo, s8
; GFX10-NEXT: s_or_b32 s7, s7, s8
; GFX10-NEXT: s_andn2_b32 exec_lo, exec_lo, s4
; GFX10-NEXT: s_cbranch_execnz .LBB2_1
@@ -255,10 +255,10 @@ define amdgpu_cs void @single_lane_execution_attribute(i32 inreg %.userdata0, <3
; GFX10-NEXT: s_add_i32 s3, s3, 4
; GFX10-NEXT: buffer_load_dword v3, v3, s[4:7], 0 offen
; GFX10-NEXT: v_cmp_eq_u32_e64 s0, 0, v2
+; GFX10-NEXT: s_or_b32 s12, s0, s12
; GFX10-NEXT: s_waitcnt vmcnt(0)
; GFX10-NEXT: v_readfirstlane_b32 s14, v3
; GFX10-NEXT: s_add_i32 s13, s14, s13
-; GFX10-NEXT: s_or_b32 s12, s0, s12
; GFX10-NEXT: v_mov_b32_e32 v3, s13
; GFX10-NEXT: s_andn2_b32 exec_lo, exec_lo, s12
; GFX10-NEXT: s_cbranch_execnz .LBB4_2
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/divergence-divergent-i1-used-outside-loop.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/divergence-divergent-i1-used-outside-loop.ll
index 34f0a85b6493f..b5d96f2f5dae7 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/divergence-divergent-i1-used-outside-loop.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/divergence-divergent-i1-used-outside-loop.ll
@@ -21,14 +21,13 @@ define void @divergent_i1_phi_used_outside_loop(float %val, float %pre.cond.val,
; GFX10-NEXT: ; =>This Inner Loop Header: Depth=1
; GFX10-NEXT: v_cvt_f32_u32_e32 v1, s6
; GFX10-NEXT: s_mov_b32 s8, exec_lo
-; GFX10-NEXT: s_add_i32 s6, s6, 1
-; GFX10-NEXT: s_xor_b32 s8, s5, s8
-; GFX10-NEXT: v_cmp_gt_f32_e32 vcc_lo, v1, v0
-; GFX10-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX10-NEXT: s_andn2_b32 s7, s7, exec_lo
; GFX10-NEXT: s_and_b32 s9, exec_lo, s5
-; GFX10-NEXT: s_mov_b32 s5, s8
+; GFX10-NEXT: s_add_i32 s6, s6, 1
+; GFX10-NEXT: v_cmp_gt_f32_e32 vcc_lo, v1, v0
+; GFX10-NEXT: s_xor_b32 s5, s5, s8
; GFX10-NEXT: s_or_b32 s7, s7, s9
+; GFX10-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX10-NEXT: s_andn2_b32 exec_lo, exec_lo, s4
; GFX10-NEXT: s_cbranch_execnz .LBB0_1
; GFX10-NEXT: ; %bb.2: ; %exit
@@ -141,14 +140,13 @@ define void @divergent_i1_xor_used_outside_loop(float %val, float %pre.cond.val,
; GFX10-NEXT: ; =>This Inner Loop Header: Depth=1
; GFX10-NEXT: v_cvt_f32_u32_e32 v1, s6
; GFX10-NEXT: s_mov_b32 s8, exec_lo
-; GFX10-NEXT: s_add_i32 s6, s6, 1
-; GFX10-NEXT: s_xor_b32 s8, s5, s8
-; GFX10-NEXT: v_cmp_gt_f32_e32 vcc_lo, v1, v0
-; GFX10-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX10-NEXT: s_andn2_b32 s7, s7, exec_lo
; GFX10-NEXT: s_and_b32 s9, exec_lo, s5
-; GFX10-NEXT: s_mov_b32 s5, s8
+; GFX10-NEXT: s_add_i32 s6, s6, 1
+; GFX10-NEXT: v_cmp_gt_f32_e32 vcc_lo, v1, v0
+; GFX10-NEXT: s_xor_b32 s5, s5, s8
; GFX10-NEXT: s_or_b32 s7, s7, s9
+; GFX10-NEXT: s_or_b32 s4, vcc_lo, s4
; GFX10-NEXT: s_andn2_b32 exec_lo, exec_lo, s4
; GFX10-NEXT: s_cbranch_execnz .LBB2_1
; GFX10-NEXT: ; %bb.2: ; %exit
@@ -182,26 +180,25 @@ define void @divergent_i1_xor_used_outside_loop_twice(float %val, float %pre.con
; GFX10-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX10-NEXT: v_cmp_lt_f32_e64 s5, 1.0, v1
; GFX10-NEXT: s_mov_b32 s4, 0
-; GFX10-NEXT: s_mov_b32 s6, 0
-; GFX10-NEXT: ; implicit-def: $sgpr7
+; GFX10-NEXT: s_mov_b32 s7, 0
+; GFX10-NEXT: ; implicit-def: $sgpr6
; GFX10-NEXT: .LBB3_1: ; %loop
; GFX10-NEXT: ; =>This Inner Loop Header: Depth=1
-; GFX10-NEXT: v_cvt_f32_u32_e32 v1, s6
+; GFX10-NEXT: v_cvt_f32_u32_e32 v1, s7
; GFX10-NEXT: s_mov_b32 s8, exec_lo
-; GFX10-NEXT: s_add_i32 s6, s6, 1
-; GFX10-NEXT: s_xor_b32 s8, s5, s8
+; GFX10-NEXT: s_andn2_b32 s6, s6, exec_lo
+; GFX10-NEXT: s_and_b32 s9, exec_lo, s5
+; GFX10-NEXT: s_add_i32 s7, s7, 1
; GFX10-NEXT: v_cmp_gt_f32_e32 vcc_lo, v1, v0
+; GFX10-NEXT: s_xor_b32 s5, s5, s8
+; GFX10-NEXT: s_or_b32 s6, s6, s9
; GFX10-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX10-NEXT: s_andn2_b32 s7, s7, exec_lo
-; GFX10-NEXT: s_and_b32 s9, exec_lo, s5
-; GFX10-NEXT: s_mov_b32 s5, s8
-; GFX10-NEXT: s_or_b32 s7, s7, s9
; GFX10-NEXT: s_andn2_b32 exec_lo, exec_lo, s4
; GFX10-NEXT: s_cbranch_execnz .LBB3_1
; GFX10-NEXT: ; %bb.2: ; %exit
; GFX10-NEXT: s_or_b32 exec_lo, exec_lo, s4
-; GFX10-NEXT: v_cndmask_b32_e64 v0, 1.0, 0, s7
-; GFX10-NEXT: v_cndmask_b32_e64 v1, 2.0, -1.0, s7
+; GFX10-NEXT: v_cndmask_b32_e64 v0, 1.0, 0, s6
+; GFX10-NEXT: v_cndmask_b32_e64 v1, 2.0, -1.0, s6
; GFX10-NEXT: flat_store_dword v[2:3], v0
; GFX10-NEXT: flat_store_dword v[4:5], v1
; GFX10-NEXT: s_waitcnt lgkmcnt(0)
@@ -256,9 +253,9 @@ define void @divergent_i1_xor_used_outside_loop_larger_loop_body(i32 %num.elts,
; GFX10-NEXT: s_or_b32 exec_lo, exec_lo, s5
; GFX10-NEXT: s_xor_b32 s5, s11, exec_lo
; GFX10-NEXT: s_and_b32 s12, exec_lo, s10
-; GFX10-NEXT: s_or_b32 s8, s12, s8
; GFX10-NEXT: s_andn2_b32 s9, s9, exec_lo
; GFX10-NEXT: s_and_b32 s5, exec_lo, s5
+; GFX10-NEXT: s_or_b32 s8, s12, s8
; GFX10-NEXT: s_or_b32 s9, s9, s5
; GFX10-NEXT: s_andn2_b32 exec_lo, exec_lo, s8
; GFX10-NEXT: s_cbranch_execz .LBB4_5
@@ -468,11 +465,11 @@ define amdgpu_ps void @divergent_i1_freeze_used_outside_loop(i32 %n, ptr addrspa
; GFX10-NEXT: ; in Loop: Header=BB6_2 Depth=1
; GFX10-NEXT: s_or_b32 exec_lo, exec_lo, s5
; GFX10-NEXT: v_cmp_lt_i32_e32 vcc_lo, s0, v0
-; GFX10-NEXT: s_add_i32 s0, s0, 1
-; GFX10-NEXT: s_or_b32 s2, vcc_lo, s2
; GFX10-NEXT: s_andn2_b32 s3, s3, exec_lo
; GFX10-NEXT: s_and_b32 s5, exec_lo, s4
; GFX10-NEXT: s_andn2_b32 s1, s1, exec_lo
+; GFX10-NEXT: s_add_i32 s0, s0, 1
+; GFX10-NEXT: s_or_b32 s2, vcc_lo, s2
; GFX10-NEXT: s_or_b32 s3, s3, s5
; GFX10-NEXT: s_or_b32 s1, s1, s5
; GFX10-NEXT: s_andn2_b32 exec_lo, exec_lo, s2
@@ -546,10 +543,10 @@ define amdgpu_cs void @loop_with_1break(ptr addrspace(1) %x, ptr addrspace(1) %a
; GFX10-NEXT: ; in Loop: Header=BB7_2 Depth=1
; GFX10-NEXT: s_or_b32 exec_lo, exec_lo, s5
; GFX10-NEXT: s_and_b32 s5, exec_lo, s3
-; GFX10-NEXT: s_or_b32 s0, s5, s0
; GFX10-NEXT: s_andn2_b32 s2, s2, exec_lo
-; GFX10-NEXT: s_and_b32 s5, exec_lo, s4
-; GFX10-NEXT: s_or_b32 s2, s2, s5
+; GFX10-NEXT: s_and_b32 s6, exec_lo, s4
+; GFX10-NEXT: s_or_b32 s0, s5, s0
+; GFX10-NEXT: s_or_b32 s2, s2, s6
; GFX10-NEXT: s_andn2_b32 exec_lo, exec_lo, s0
; GFX10-NEXT: s_cbranch_execz .LBB7_4
; GFX10-NEXT: .LBB7_2: ; %A
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/divergence-structurizer.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/divergence-structurizer.ll
index 1603d5211c686..8ee7b491eef98 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/divergence-structurizer.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/divergence-structurizer.ll
@@ -384,10 +384,10 @@ define amdgpu_cs void @loop_with_div_break_with_body(ptr addrspace(1) %x, ptr ad
; GFX10-NEXT: ; in Loop: Header=BB5_2 Depth=1
; GFX10-NEXT: s_or_b32 exec_lo, exec_lo, s5
; GFX10-NEXT: s_and_b32 s5, exec_lo, s3
-; GFX10-NEXT: s_or_b32 s0, s5, s0
; GFX10-NEXT: s_andn2_b32 s2, s2, exec_lo
-; GFX10-NEXT: s_and_b32 s5, exec_lo, s4
-; GFX10-NEXT: s_or_b32 s2, s2, s5
+; GFX10-NEXT: s_and_b32 s6, exec_lo, s4
+; GFX10-NEXT: s_or_b32 s0, s5, s0
+; GFX10-NEXT: s_or_b32 s2, s2, s6
; GFX10-NEXT: s_andn2_b32 exec_lo, exec_lo, s0
; GFX10-NEXT: s_cbranch_execz .LBB5_4
; GFX10-NEXT: .LBB5_2: ; %A
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/divergence-temporal-divergent-i1.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/divergence-temporal-divergent-i1.ll
index 8c344daa5c38e..31745ec365e50 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/divergence-temporal-divergent-i1.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/divergence-temporal-divergent-i1.ll
@@ -16,11 +16,11 @@ define void @temporal_divergent_i1_phi(float %val, ptr %addr) {
; GFX10-NEXT: s_cmp_lg_u32 s8, 0
; GFX10-NEXT: s_cselect_b32 s8, exec_lo, 0
; GFX10-NEXT: v_cmp_gt_f32_e32 vcc_lo, v3, v0
+; GFX10-NEXT: s_andn2_b32 s7, s7, exec_lo
+; GFX10-NEXT: s_and_b32 s8, exec_lo, s8
; GFX10-NEXT: s_xor_b32 s5, s5, 1
; GFX10-NEXT: s_add_i32 s6, s6, 1
; GFX10-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX10-NEXT: s_andn2_b32 s7, s7, exec_lo
-; GFX10-NEXT: s_and_b32 s8, exec_lo, s8
; GFX10-NEXT: s_or_b32 s7, s7, s8
; GFX10-NEXT: s_andn2_b32 exec_lo, exec_lo, s4
; GFX10-NEXT: s_cbranch_execnz .LBB0_1
@@ -63,11 +63,11 @@ define void @temporal_divergent_i1_non_phi(float %val, ptr %addr) {
; GFX10-NEXT: s_cmp_lg_u32 s8, 0
; GFX10-NEXT: s_cselect_b32 s8, exec_lo, 0
; GFX10-NEXT: v_cmp_gt_f32_e32 vcc_lo, v3, v0
+; GFX10-NEXT: s_andn2_b32 s7, s7, exec_lo
+; GFX10-NEXT: s_and_b32 s8, exec_lo, s8
; GFX10-NEXT: s_xor_b32 s5, s5, 1
; GFX10-NEXT: s_add_i32 s6, s6, 1
; GFX10-NEXT: s_or_b32 s4, vcc_lo, s4
-; GFX10-NEXT: s_andn2_b32 s7, s7, exec_lo
-; GFX10-NEXT: s_and_b32 s8, exec_lo, s8
; GFX10-NEXT: s_or_b32 s7, s7, s8
; GFX10-NEXT: s_andn2_b32 exec_lo, exec_lo, s4
; GFX10-NEXT: s_cbranch_execnz .LBB1_1
@@ -127,10 +127,10 @@ define amdgpu_cs void @loop_with_1break(ptr addrspace(1) %x, i32 %x.size, ptr ad
; GFX10-NEXT: s_cmp_lg_u32 s5, 0
; GFX10-NEXT: s_cselect_b32 s5, exec_lo, 0
; GFX10-NEXT: s_and_b32 s6, exec_lo, s10
-; GFX10-NEXT: s_or_b32 s8, s6, s8
-; GFX10-NEXT: s_andn2_b32 s6, s9, exec_lo
+; GFX10-NEXT: s_andn2_b32 s7, s9, exec_lo
; GFX10-NEXT: s_and_b32 s5, exec_lo, s5
-; GFX10-NEXT: s_or_b32 s9, s6, s5
+; GFX10-NEXT: s_or_b32 s8, s6, s8
+; GFX10-NEXT: s_or_b32 s9, s7, s5
; GFX10-NEXT: s_waitcnt_depctr depctr_vm_vsrc(0)
; GFX10-NEXT: s_andn2_b32 exec_lo, exec_lo, s8
; GFX10-NEXT: s_cbranch_execz .LBB2_5
@@ -219,14 +219,14 @@ define void @nested_loops_temporal_divergence_inner(float %pre.cond.val, i32 %n.
; GFX10-NEXT: ; => This Inner Loop Header: Depth=2
; GFX10-NEXT: v_cvt_f32_u32_e32 v6, s11
; GFX10-NEXT: s_mov_b32 s12, exec_lo
-; GFX10-NEXT: s_add_i32 s11, s11, 1
+; GFX10-NEXT: s_andn2_b32 s9, s9, exec_lo
; GFX10-NEXT: s_xor_b32 s4, s4, s12
+; GFX10-NEXT: s_add_i32 s11, s11, 1
; GFX10-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
; GFX10-NEXT: v_cmp_gt_f32_e32 vcc_lo, v6, v0
-; GFX10-NEXT: s_or_b32 s10, vcc_lo, s10
-; GFX10-NEXT: s_andn2_b32 s9, s9, exec_lo
; GFX10-NEXT: s_and_b32 s12, exec_lo, s4
; GFX10-NEXT: s_or_b32 s9, s9, s12
+; GFX10-NEXT: s_or_b32 s10, vcc_lo, s10
; GFX10-NEXT: s_andn2_b32 exec_lo, exec_lo, s10
; GFX10-NEXT: s_cbranch_execnz .LBB3_2
; GFX10-NEXT: ; %bb.3: ; %UseInst
@@ -310,14 +310,14 @@ define void @nested_loops_temporal_divergence_outer(float %pre.cond.val, i32 %n.
; GFX10-NEXT: ; => This Inner Loop Header: Depth=2
; GFX10-NEXT: v_cvt_f32_u32_e32 v6, s11
; GFX10-NEXT: s_mov_b32 s12, exec_lo
-; GFX10-NEXT: s_add_i32 s11, s11, 1
+; GFX10-NEXT: s_andn2_b32 s9, s9, exec_lo
; GFX10-NEXT: s_xor_b32 s4, s4, s12
+; GFX10-NEXT: s_add_i32 s11, s11, 1
; GFX10-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
; GFX10-NEXT: v_cmp_gt_f32_e32 vcc_lo, v6, v0
-; GFX10-NEXT: s_or_b32 s10, vcc_lo, s10
-; GFX10-NEXT: s_andn2_b32 s9, s9, exec_lo
; GFX10-NEXT: s_and_b32 s12, exec_lo, s4
; GFX10-NEXT: s_or_b32 s9, s9, s12
+; GFX10-NEXT: s_or_b32 s10, vcc_lo, s10
; GFX10-NEXT: s_andn2_b32 exec_lo, exec_lo, s10
; GFX10-NEXT: s_cbranch_execnz .LBB4_2
; GFX10-NEXT: ; %bb.3: ; %UseInst
@@ -401,14 +401,14 @@ define void @nested_loops_temporal_divergence_both(float %pre.cond.val, i32 %n.i
; GFX10-NEXT: ; => This Inner Loop Header: Depth=2
; GFX10-NEXT: v_cvt_f32_u32_e32 v8, s11
; GFX10-NEXT: s_mov_b32 s12, exec_lo
-; GFX10-NEXT: s_add_i32 s11, s11, 1
+; GFX10-NEXT: s_andn2_b32 s9, s9, exec_lo
; GFX10-NEXT: s_xor_b32 s4, s4, s12
+; GFX10-NEXT: s_add_i32 s11, s11, 1
; GFX10-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
; GFX10-NEXT: v_cmp_gt_f32_e32 vcc_lo, v8, v0
-; GFX10-NEXT: s_or_b32 s10, vcc_lo, s10
-; GFX10-NEXT: s_andn2_b32 s9, s9, exec_lo
; GFX10-NEXT: s_and_b32 s12, exec_lo, s4
; GFX10-NEXT: s_or_b32 s9, s9, s12
+; GFX10-NEXT: s_or_b32 s10, vcc_lo, s10
; GFX10-NEXT: s_andn2_b32 exec_lo, exec_lo, s10
; GFX10-NEXT: s_cbranch_execnz .LBB5_2
; GFX10-NEXT: ; %bb.3: ; %UseInst
diff --git a/llvm/test/CodeGen/AMDGPU/block-should-not-be-in-alive-blocks.mir b/llvm/test/CodeGen/AMDGPU/block-should-not-be-in-alive-blocks.mir
index 7ecb1cd7a5343..875e2fcc671e6 100644
--- a/llvm/test/CodeGen/AMDGPU/block-should-not-be-in-alive-blocks.mir
+++ b/llvm/test/CodeGen/AMDGPU/block-should-not-be-in-alive-blocks.mir
@@ -61,10 +61,10 @@ body: |
; CHECK-NEXT: bb.5:
; CHECK-NEXT: successors: %bb.1(0x40000000), %bb.7(0x40000000)
; CHECK-NEXT: {{ $}}
- ; CHECK-NEXT: [[S_OR_SAVEEXEC_B32_:%[0-9]+]]:sreg_32 = S_OR_SAVEEXEC_B32 killed [[S_XOR_B32_]], implicit-def $exec, implicit-def $scc, implicit $exec
+ ; CHECK-NEXT: [[S_OR_SAVEEXEC_B32_:%[0-9]+]]:sreg_32 = S_OR_SAVEEXEC_B32 killed [[S_XOR_B32_]], implicit-def $exec, implicit-def dead $scc, implicit $exec
; CHECK-NEXT: [[COPY4:%[0-9]+]]:vgpr_32 = COPY killed [[COPY2]]
- ; CHECK-NEXT: [[S_AND_B32_1:%[0-9]+]]:sreg_32 = S_AND_B32 $exec_lo, [[S_OR_SAVEEXEC_B32_]], implicit-def $scc
- ; CHECK-NEXT: $exec_lo = S_XOR_B32_term $exec_lo, [[S_AND_B32_1]], implicit-def $scc
+ ; CHECK-NEXT: [[S_AND_B32_1:%[0-9]+]]:sreg_32 = S_AND_B32 $exec_lo, [[S_OR_SAVEEXEC_B32_]], implicit-def dead $scc
+ ; CHECK-NEXT: $exec_lo = S_XOR_B32_term $exec_lo, [[S_AND_B32_1]], implicit-def dead $scc
; CHECK-NEXT: S_CBRANCH_EXECZ %bb.7, implicit $exec
; CHECK-NEXT: S_BRANCH %bb.1
; CHECK-NEXT: {{ $}}
@@ -75,7 +75,7 @@ body: |
; CHECK-NEXT: S_BRANCH %bb.5
; CHECK-NEXT: {{ $}}
; CHECK-NEXT: bb.7:
- ; CHECK-NEXT: $exec_lo = S_OR_B32 $exec_lo, killed [[S_AND_B32_1]], implicit-def $scc
+ ; CHECK-NEXT: $exec_lo = S_OR_B32 $exec_lo, killed [[S_AND_B32_1]], implicit-def dead $scc
; CHECK-NEXT: S_ENDPGM 0
bb.0:
successors: %bb.2(0x40000000), %bb.5(0x40000000)
diff --git a/llvm/test/CodeGen/AMDGPU/branch-folding-implicit-def-subreg.ll b/llvm/test/CodeGen/AMDGPU/branch-folding-implicit-def-subreg.ll
index 210eeab4b99aa..41b3bfdf523e8 100644
--- a/llvm/test/CodeGen/AMDGPU/branch-folding-implicit-def-subreg.ll
+++ b/llvm/test/CodeGen/AMDGPU/branch-folding-implicit-def-subreg.ll
@@ -133,7 +133,7 @@ define amdgpu_kernel void @f1(ptr addrspace(1) %arg, ptr addrspace(1) %arg1, i64
; GFX90A-NEXT: successors: %bb.9(0x40000000), %bb.10(0x40000000)
; GFX90A-NEXT: liveins: $sgpr14, $sgpr16, $sgpr17, $vgpr31, $sgpr4_sgpr5, $sgpr6_sgpr7, $sgpr8_sgpr9:0x000000000000000F, $sgpr10_sgpr11, $sgpr18_sgpr19, $sgpr34_sgpr35, $sgpr38_sgpr39, $sgpr40_sgpr41, $sgpr42_sgpr43, $sgpr44_sgpr45, $sgpr46_sgpr47, $sgpr48_sgpr49, $sgpr50_sgpr51, $sgpr52_sgpr53, $sgpr54_sgpr55, $sgpr64_sgpr65, $sgpr66_sgpr67, $sgpr68_sgpr69, $vgpr0_vgpr1:0x000000000000000F, $vgpr6_vgpr7:0x000000000000000F, $vgpr8_vgpr9:0x000000000000000F, $vgpr40_vgpr41:0x000000000000000F, $vgpr42_vgpr43:0x000000000000000F, $vgpr44_vgpr45:0x000000000000000F, $vgpr46_vgpr47:0x000000000000000F, $vgpr54_vgpr55:0x000000000000000F, $vgpr56_vgpr57:0x000000000000000F, $vgpr58_vgpr59:0x000000000000000F, $vgpr60_vgpr61:0x000000000000000F, $vgpr62_vg...
[truncated]
``````````
</details>
https://github.com/llvm/llvm-project/pull/226249
More information about the llvm-commits
mailing list