[llvm] 67f26e1 - AMDGPU: Restore dropped EXEC/SCC defs on SI_KILL_F32_COND_IMM (#227604)
via llvm-commits
llvm-commits at lists.llvm.org
Wed Sep 30 03:27:56 PDT 2026
Author: Matt Arsenault
Date: 2026-09-30T12:27:49+02:00
New Revision: 67f26e15725a56f302737c576e3f6f52a97cc4ac
URL: https://github.com/llvm/llvm-project/commit/67f26e15725a56f302737c576e3f6f52a97cc4ac
DIFF: https://github.com/llvm/llvm-project/commit/67f26e15725a56f302737c576e3f6f52a97cc4ac.diff
LOG: AMDGPU: Restore dropped EXEC/SCC defs on SI_KILL_F32_COND_IMM (#227604)
These pseudos may clobber scc, so they need the def on the instruction
definition. c16f776028dd added a let block over the defm, overriding the
[EXEC, SCC] defs in the base PseudoInstKill class.
With the defs gone, selection was free to leave a compare result in SCC
across the kill. #134718 then marked SCC live-in to the split block to
satisfy the verifier, which can probably cleaned up. The result is
visible in skip-if-dead.ll: in scc_use_after_kill_inst the s_cbranch_scc0 after
the kill consumed the exec update's SCC instead of the s_cmp_lg_u32 it was
selected for.
Pass the extra defs as a multiclass parameter so an override cannot drop
the mandatory ones again. Selection now sinks the compare past the kill,
so SCC is never live across it and the dead flags from a3cf6f45fb68 are
correct as written.
Co-authored-by: Claude Opus 5 <noreply at anthropic.com>
Added:
Modified:
llvm/lib/Target/AMDGPU/SIInstructions.td
llvm/test/CodeGen/AMDGPU/finalize-isel-kill-scc-vcc.mir
llvm/test/CodeGen/AMDGPU/skip-if-dead.ll
llvm/test/CodeGen/AMDGPU/wqm.mir
Removed:
################################################################################
diff --git a/llvm/lib/Target/AMDGPU/SIInstructions.td b/llvm/lib/Target/AMDGPU/SIInstructions.td
index 823a8b3a0c9f9..c7976983b727f 100644
--- a/llvm/lib/Target/AMDGPU/SIInstructions.td
+++ b/llvm/lib/Target/AMDGPU/SIInstructions.td
@@ -656,13 +656,13 @@ def SI_EARLY_TERMINATE_SCC0 : SPseudoInstSI <(outs), (ins)> {
let Uses = [EXEC] in {
-multiclass PseudoInstKill <dag ins> {
+multiclass PseudoInstKill <dag ins, list<Register> extraDefs = []> {
// Even though this pseudo can usually be expanded without an SCC def, we
// conservatively assume that it has an SCC def, both because it is sometimes
// required in degenerate cases (when V_CMPX cannot be used due to constant
// bus limitations) and because it allows us to avoid having to track SCC
// liveness across basic blocks.
- let isConvergent = 1, Defs = [EXEC,SCC] in {
+ let isConvergent = 1, Defs = !listconcat([EXEC, SCC], extraDefs) in {
def _PSEUDO : PseudoInstSI <(outs), ins> {
let usesCustomInserter = 1;
}
@@ -673,15 +673,14 @@ multiclass PseudoInstKill <dag ins> {
}
defm SI_KILL_I1 : PseudoInstKill <(ins SCSrc_i1:$src, i1imm:$killvalue)>;
-let Defs = [VCC] in
-defm SI_KILL_F32_COND_IMM : PseudoInstKill <(ins VSrc_b32:$src0, i32imm:$src1, i32imm:$cond)>;
+defm SI_KILL_F32_COND_IMM : PseudoInstKill <(ins VSrc_b32:$src0, i32imm:$src1, i32imm:$cond), [VCC]>;
let Defs = [EXEC,VCC] in
def SI_ILLEGAL_COPY : SPseudoInstSI <
(outs unknown:$dst), (ins unknown:$src),
[], " ; illegal copy $src to $dst">;
-} // End Uses = [EXEC], Defs = [EXEC,VCC]
+} // End Uses = [EXEC]
// Branch on undef scc. Used to avoid intermediate copy from
// IMPLICIT_DEF to SCC.
diff --git a/llvm/test/CodeGen/AMDGPU/finalize-isel-kill-scc-vcc.mir b/llvm/test/CodeGen/AMDGPU/finalize-isel-kill-scc-vcc.mir
index ce65c4d4d2102..6344ac034d87f 100644
--- a/llvm/test/CodeGen/AMDGPU/finalize-isel-kill-scc-vcc.mir
+++ b/llvm/test/CodeGen/AMDGPU/finalize-isel-kill-scc-vcc.mir
@@ -21,7 +21,7 @@ body: |
; CHECK-NEXT: [[COPY3:%[0-9]+]]:sgpr_32 = COPY [[V_CNDMASK_B32_e64_]]
; CHECK-NEXT: [[S_MOV_B32_3:%[0-9]+]]:sreg_32 = S_MOV_B32 0
; CHECK-NEXT: S_CMP_LG_U32 [[COPY]], killed [[S_MOV_B32_3]], implicit-def $scc
- ; CHECK-NEXT: SI_KILL_F32_COND_IMM_TERMINATOR [[V_ADD_F32_e64_]], 0, 2, implicit-def $vcc_lo, implicit $exec
+ ; CHECK-NEXT: SI_KILL_F32_COND_IMM_TERMINATOR [[V_ADD_F32_e64_]], 0, 2, implicit-def $exec, implicit-def $scc, implicit-def $vcc_lo, implicit $exec
; CHECK-NEXT: {{ $}}
; CHECK-NEXT: bb.3:
; CHECK-NEXT: successors: %bb.1(0x40000000), %bb.2(0x40000000)
@@ -55,7 +55,7 @@ body: |
%0:sgpr_32 = COPY %10:vgpr_32
%12:sreg_32 = S_MOV_B32 0
S_CMP_LG_U32 %3:sgpr_32, killed %12:sreg_32, implicit-def $scc
- SI_KILL_F32_COND_IMM_PSEUDO %6:vgpr_32, 0, 2, implicit-def $vcc, implicit $exec
+ SI_KILL_F32_COND_IMM_PSEUDO %6:vgpr_32, 0, 2, implicit-def $exec, implicit-def $scc, implicit-def $vcc, implicit $exec
S_CBRANCH_SCC1 %bb.1, implicit $scc
S_CBRANCH_VCCNZ %bb.2, implicit $vcc
S_BRANCH %bb.2
diff --git a/llvm/test/CodeGen/AMDGPU/skip-if-dead.ll b/llvm/test/CodeGen/AMDGPU/skip-if-dead.ll
index cb8974c0ff87c..c3dd793907beb 100644
--- a/llvm/test/CodeGen/AMDGPU/skip-if-dead.ll
+++ b/llvm/test/CodeGen/AMDGPU/skip-if-dead.ll
@@ -1962,14 +1962,13 @@ define amdgpu_ps void @scc_use_after_kill_inst(float inreg %x, i32 inreg %y) #0
; SI: ; %bb.0: ; %bb
; SI-NEXT: v_add_f32_e64 v1, s0, 1.0
; SI-NEXT: v_cmp_lt_f32_e32 vcc, 0, v1
-; SI-NEXT: s_mov_b64 s[2:3], exec
-; SI-NEXT: s_cmp_lg_u32 s1, 0
; SI-NEXT: v_cndmask_b32_e64 v0, 0, -1.0, vcc
; SI-NEXT: v_cmp_nlt_f32_e32 vcc, 0, v1
-; SI-NEXT: s_andn2_b64 s[2:3], s[2:3], vcc
+; SI-NEXT: s_andn2_b64 exec, exec, vcc
; SI-NEXT: s_cbranch_scc0 .LBB17_6
; SI-NEXT: ; %bb.1: ; %bb
; SI-NEXT: s_andn2_b64 exec, exec, vcc
+; SI-NEXT: s_cmp_lg_u32 s1, 0
; SI-NEXT: s_cbranch_scc0 .LBB17_3
; SI-NEXT: ; %bb.2: ; %bb8
; SI-NEXT: s_mov_b32 s3, 0xf000
@@ -1997,15 +1996,14 @@ define amdgpu_ps void @scc_use_after_kill_inst(float inreg %x, i32 inreg %y) #0
; GFX10-WAVE64-LABEL: scc_use_after_kill_inst:
; GFX10-WAVE64: ; %bb.0: ; %bb
; GFX10-WAVE64-NEXT: v_add_f32_e64 v1, s0, 1.0
-; GFX10-WAVE64-NEXT: s_mov_b64 s[2:3], exec
-; GFX10-WAVE64-NEXT: s_cmp_lg_u32 s1, 0
; GFX10-WAVE64-NEXT: v_cmp_lt_f32_e32 vcc, 0, v1
; GFX10-WAVE64-NEXT: v_cndmask_b32_e64 v0, 0, -1.0, vcc
; GFX10-WAVE64-NEXT: v_cmp_nlt_f32_e32 vcc, 0, v1
-; GFX10-WAVE64-NEXT: s_andn2_b64 s[2:3], s[2:3], vcc
+; GFX10-WAVE64-NEXT: s_andn2_b64 exec, exec, vcc
; GFX10-WAVE64-NEXT: s_cbranch_scc0 .LBB17_6
; GFX10-WAVE64-NEXT: ; %bb.1: ; %bb
; GFX10-WAVE64-NEXT: s_andn2_b64 exec, exec, vcc
+; GFX10-WAVE64-NEXT: s_cmp_lg_u32 s1, 0
; GFX10-WAVE64-NEXT: s_cbranch_scc0 .LBB17_3
; GFX10-WAVE64-NEXT: ; %bb.2: ; %bb8
; GFX10-WAVE64-NEXT: v_mov_b32_e32 v1, 8
@@ -2029,15 +2027,14 @@ define amdgpu_ps void @scc_use_after_kill_inst(float inreg %x, i32 inreg %y) #0
; GFX10-WAVE32-LABEL: scc_use_after_kill_inst:
; GFX10-WAVE32: ; %bb.0: ; %bb
; GFX10-WAVE32-NEXT: v_add_f32_e64 v1, s0, 1.0
-; GFX10-WAVE32-NEXT: s_mov_b32 s2, exec_lo
-; GFX10-WAVE32-NEXT: s_cmp_lg_u32 s1, 0
; GFX10-WAVE32-NEXT: v_cmp_lt_f32_e32 vcc_lo, 0, v1
; GFX10-WAVE32-NEXT: v_cndmask_b32_e64 v0, 0, -1.0, vcc_lo
; GFX10-WAVE32-NEXT: v_cmp_nlt_f32_e32 vcc_lo, 0, v1
-; GFX10-WAVE32-NEXT: s_andn2_b32 s2, s2, vcc_lo
+; GFX10-WAVE32-NEXT: s_andn2_b32 exec_lo, exec_lo, vcc_lo
; GFX10-WAVE32-NEXT: s_cbranch_scc0 .LBB17_6
; GFX10-WAVE32-NEXT: ; %bb.1: ; %bb
; GFX10-WAVE32-NEXT: s_andn2_b32 exec_lo, exec_lo, vcc_lo
+; GFX10-WAVE32-NEXT: s_cmp_lg_u32 s1, 0
; GFX10-WAVE32-NEXT: s_cbranch_scc0 .LBB17_3
; GFX10-WAVE32-NEXT: ; %bb.2: ; %bb8
; GFX10-WAVE32-NEXT: v_mov_b32_e32 v1, 8
@@ -2061,17 +2058,16 @@ define amdgpu_ps void @scc_use_after_kill_inst(float inreg %x, i32 inreg %y) #0
; GFX11-LABEL: scc_use_after_kill_inst:
; GFX11: ; %bb.0: ; %bb
; GFX11-NEXT: v_add_f32_e64 v1, s0, 1.0
-; GFX11-NEXT: s_mov_b64 s[2:3], exec
-; GFX11-NEXT: s_cmp_lg_u32 s1, 0
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_cmp_lt_f32_e32 vcc, 0, v1
; GFX11-NEXT: v_cndmask_b32_e64 v0, 0, -1.0, vcc
; GFX11-NEXT: v_cmp_nlt_f32_e32 vcc, 0, v1
; GFX11-NEXT: s_waitcnt_depctr depctr_va_vcc(0)
-; GFX11-NEXT: s_and_not1_b64 s[2:3], s[2:3], vcc
+; GFX11-NEXT: s_and_not1_b64 exec, exec, vcc
; GFX11-NEXT: s_cbranch_scc0 .LBB17_6
; GFX11-NEXT: ; %bb.1: ; %bb
; GFX11-NEXT: s_and_not1_b64 exec, exec, vcc
+; GFX11-NEXT: s_cmp_lg_u32 s1, 0
; GFX11-NEXT: s_cbranch_scc0 .LBB17_3
; GFX11-NEXT: ; %bb.2: ; %bb8
; GFX11-NEXT: v_mov_b32_e32 v1, 8
diff --git a/llvm/test/CodeGen/AMDGPU/wqm.mir b/llvm/test/CodeGen/AMDGPU/wqm.mir
index c8530db067fa5..0dc193e087663 100644
--- a/llvm/test/CodeGen/AMDGPU/wqm.mir
+++ b/llvm/test/CodeGen/AMDGPU/wqm.mir
@@ -646,7 +646,7 @@ body: |
%0:vgpr_32 = COPY $vgpr0
%1:sreg_64 = COPY $vcc
- SI_KILL_F32_COND_IMM_TERMINATOR %0, 0, 4, implicit-def $vcc, implicit $exec
+ SI_KILL_F32_COND_IMM_TERMINATOR %0, 0, 4, implicit-def $exec, implicit-def $scc, implicit-def $vcc, implicit $exec
bb.1:
$exec = S_AND_B64 $exec, %1, implicit-def $scc
More information about the llvm-commits
mailing list