[llvm] [AMDGPU] Preserve live SCC when shrinking disjoint S_OR_B32 to S_ADDK_I32 (PR #228025)
Teja Alaghari via llvm-commits
llvm-commits at lists.llvm.org
Sun Oct 4 22:52:15 PDT 2026
https://github.com/TejaX-Alaghari updated https://github.com/llvm/llvm-project/pull/228025
>From 4974a045a565b618d3c8a811463a81fc4982c9e7 Mon Sep 17 00:00:00 2001
From: tmahatej_amdeng <teja.alaghari at amd.com>
Date: Thu, 1 Oct 2026 15:48:47 +0530
Subject: [PATCH 1/3] Precommit test case for s_or incorrectly transformed to
s_addk when SCC is live
---
.../CodeGen/AMDGPU/s_or_b32_transformation.ll | 28 +++++++++++++++++++
1 file changed, 28 insertions(+)
diff --git a/llvm/test/CodeGen/AMDGPU/s_or_b32_transformation.ll b/llvm/test/CodeGen/AMDGPU/s_or_b32_transformation.ll
index 68f90e7a73e2795..2e8cd9052860703 100644
--- a/llvm/test/CodeGen/AMDGPU/s_or_b32_transformation.ll
+++ b/llvm/test/CodeGen/AMDGPU/s_or_b32_transformation.ll
@@ -29,3 +29,31 @@ define amdgpu_ps i32 @s_or_b32_to_s_bitset1_b32(i32 inreg %x) {
ret i32 %or
}
+define amdgpu_ps i32 @s_or_b32_disjoint_live_scc(i32 inreg %n) {
+; CHECK-LABEL: s_or_b32_disjoint_live_scc:
+; CHECK: ; %bb.0: ; %entry
+; CHECK-NEXT: s_lshl_b32 s0, s0, 31
+; CHECK-NEXT: s_addk_i32 s0, 0x800
+; CHECK-NEXT: s_cselect_b32 s0, 0, 1
+; CHECK-NEXT: ; return to shader part epilog
+entry:
+ switch i32 0, label %tail [
+ i32 0, label %case
+ ]
+
+case:
+ br label %tail
+
+tail:
+ %phi = phi i32 [ 32, %case ], [ 0, %entry ]
+ %ins0 = insertelement <2 x i32> poison, i32 %n, i32 0
+ %ins1 = insertelement <2 x i32> %ins0, i32 %phi, i32 1
+ %mul = mul <2 x i32> %ins1, <i32 -2147483648, i32 64>
+ %e0 = extractelement <2 x i32> %mul, i32 0
+ %e1 = extractelement <2 x i32> %mul, i32 1
+ %s1 = add i32 %e0, %e1
+ %cmp = icmp eq i32 %s1, 0
+ %sel = select i1 %cmp, i32 1, i32 0
+ ret i32 %sel
+}
+
>From eef9b4d9ccf17356cb309b2b7869dd76bfe6b2bb Mon Sep 17 00:00:00 2001
From: tmahatej_amdeng <teja.alaghari at amd.com>
Date: Thu, 1 Oct 2026 15:45:24 +0530
Subject: [PATCH 2/3] [AMDGPU] Preserve live SCC when shrinking disjoint
S_OR_B32
---
llvm/lib/Target/AMDGPU/SIShrinkInstructions.cpp | 6 ++++++
llvm/test/CodeGen/AMDGPU/s_or_b32_transformation.ll | 3 ++-
2 files changed, 8 insertions(+), 1 deletion(-)
diff --git a/llvm/lib/Target/AMDGPU/SIShrinkInstructions.cpp b/llvm/lib/Target/AMDGPU/SIShrinkInstructions.cpp
index 1483fd43ad8b1da..41afb13ea43077c 100644
--- a/llvm/lib/Target/AMDGPU/SIShrinkInstructions.cpp
+++ b/llvm/lib/Target/AMDGPU/SIShrinkInstructions.cpp
@@ -953,6 +953,12 @@ bool SIShrinkInstructions::run(MachineFunction &MF) {
}
if (Src0->isReg() && Src0->getReg() == Dest->getReg()) {
if (Src1->isImm() && isKImmOperand(MI, *Src1)) {
+ // S_OR_B32 sets SCC to (result != 0), but S_ADDK_I32 sets it to
+ // signed overflow, so only shrink the OR when SCC is dead.
+ if (MI.getOpcode() == AMDGPU::S_OR_B32 &&
+ !MI.findRegisterDefOperand(AMDGPU::SCC, /*TRI=*/nullptr)
+ ->isDead())
+ continue;
unsigned Opc = (MI.getOpcode() == AMDGPU::S_MUL_I32)
? AMDGPU::S_MULK_I32
: AMDGPU::S_ADDK_I32;
diff --git a/llvm/test/CodeGen/AMDGPU/s_or_b32_transformation.ll b/llvm/test/CodeGen/AMDGPU/s_or_b32_transformation.ll
index 2e8cd9052860703..53a1fe89af5207c 100644
--- a/llvm/test/CodeGen/AMDGPU/s_or_b32_transformation.ll
+++ b/llvm/test/CodeGen/AMDGPU/s_or_b32_transformation.ll
@@ -1,6 +1,7 @@
; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
; RUN: llc -mtriple=amdgpu9.00 < %s | FileCheck %s
; This tests if disjoint s_or_b32 gets transformed to s_addk_i32 when we can't use s_bitset1_b32
+; and SCC is dead.
define amdgpu_ps i32 @s_or_b32_i32(i32 inreg %x) {
; CHECK-LABEL: s_or_b32_i32:
@@ -33,7 +34,7 @@ define amdgpu_ps i32 @s_or_b32_disjoint_live_scc(i32 inreg %n) {
; CHECK-LABEL: s_or_b32_disjoint_live_scc:
; CHECK: ; %bb.0: ; %entry
; CHECK-NEXT: s_lshl_b32 s0, s0, 31
-; CHECK-NEXT: s_addk_i32 s0, 0x800
+; CHECK-NEXT: s_or_b32 s0, s0, 0x800
; CHECK-NEXT: s_cselect_b32 s0, 0, 1
; CHECK-NEXT: ; return to shader part epilog
entry:
>From c434cf438f046f6474cd9b91f53a2c6f184c4eba Mon Sep 17 00:00:00 2001
From: tmahatej_amdeng <teja.alaghari at amd.com>
Date: Mon, 5 Oct 2026 11:21:12 +0530
Subject: [PATCH 3/3] Address review feedback: 1. Use allImplicitDefsAreDead()
helper to check whether SCC is dead 2. Simplify .ll test case 3. Add a new
.mir test case
---
.../Target/AMDGPU/SIShrinkInstructions.cpp | 3 +-
.../CodeGen/AMDGPU/s_or_b32_transformation.ll | 14 ++++----
.../CodeGen/AMDGPU/shrink-disjoint-or-scc.mir | 35 +++++++++++++++++++
3 files changed, 42 insertions(+), 10 deletions(-)
create mode 100644 llvm/test/CodeGen/AMDGPU/shrink-disjoint-or-scc.mir
diff --git a/llvm/lib/Target/AMDGPU/SIShrinkInstructions.cpp b/llvm/lib/Target/AMDGPU/SIShrinkInstructions.cpp
index 41afb13ea43077c..098f39cb49f462e 100644
--- a/llvm/lib/Target/AMDGPU/SIShrinkInstructions.cpp
+++ b/llvm/lib/Target/AMDGPU/SIShrinkInstructions.cpp
@@ -956,8 +956,7 @@ bool SIShrinkInstructions::run(MachineFunction &MF) {
// S_OR_B32 sets SCC to (result != 0), but S_ADDK_I32 sets it to
// signed overflow, so only shrink the OR when SCC is dead.
if (MI.getOpcode() == AMDGPU::S_OR_B32 &&
- !MI.findRegisterDefOperand(AMDGPU::SCC, /*TRI=*/nullptr)
- ->isDead())
+ !MI.allImplicitDefsAreDead())
continue;
unsigned Opc = (MI.getOpcode() == AMDGPU::S_MUL_I32)
? AMDGPU::S_MULK_I32
diff --git a/llvm/test/CodeGen/AMDGPU/s_or_b32_transformation.ll b/llvm/test/CodeGen/AMDGPU/s_or_b32_transformation.ll
index 53a1fe89af5207c..8e46a6381186f66 100644
--- a/llvm/test/CodeGen/AMDGPU/s_or_b32_transformation.ll
+++ b/llvm/test/CodeGen/AMDGPU/s_or_b32_transformation.ll
@@ -38,22 +38,20 @@ define amdgpu_ps i32 @s_or_b32_disjoint_live_scc(i32 inreg %n) {
; CHECK-NEXT: s_cselect_b32 s0, 0, 1
; CHECK-NEXT: ; return to shader part epilog
entry:
- switch i32 0, label %tail [
- i32 0, label %case
- ]
+ br i1 true, label %tail, label %dead
-case:
+dead:
br label %tail
tail:
- %phi = phi i32 [ 32, %case ], [ 0, %entry ]
+ %addend = phi i32 [ 32, %entry ], [ 0, %dead ]
%ins0 = insertelement <2 x i32> poison, i32 %n, i32 0
- %ins1 = insertelement <2 x i32> %ins0, i32 %phi, i32 1
+ %ins1 = insertelement <2 x i32> %ins0, i32 %addend, i32 1
%mul = mul <2 x i32> %ins1, <i32 -2147483648, i32 64>
%e0 = extractelement <2 x i32> %mul, i32 0
%e1 = extractelement <2 x i32> %mul, i32 1
- %s1 = add i32 %e0, %e1
- %cmp = icmp eq i32 %s1, 0
+ %sum = add i32 %e0, %e1
+ %cmp = icmp eq i32 %sum, 0
%sel = select i1 %cmp, i32 1, i32 0
ret i32 %sel
}
diff --git a/llvm/test/CodeGen/AMDGPU/shrink-disjoint-or-scc.mir b/llvm/test/CodeGen/AMDGPU/shrink-disjoint-or-scc.mir
new file mode 100644
index 000000000000000..20e0b6c0205cdc1
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/shrink-disjoint-or-scc.mir
@@ -0,0 +1,35 @@
+# NOTE: Assertions have been autogenerated by utils/update_mir_test_checks.py UTC_ARGS: --version 3
+# RUN: llc -mtriple=amdgpu9.00 -verify-machineinstrs -run-pass=si-shrink-instructions -o - %s | FileCheck -check-prefix=GCN %s
+# RUN: llc -mtriple=amdgpu9.00 -verify-machineinstrs -passes=si-shrink-instructions -o - %s | FileCheck -check-prefix=GCN %s
+
+# Disjoint S_OR_B32 sets SCC to (result != 0). S_ADDK_I32 sets SCC to signed
+# overflow, so the shrink is refused while that implicit def is live.
+
+---
+name: shrink_disjoint_or_live_scc
+tracksRegLiveness: true
+body: |
+ bb.0:
+ liveins: $sgpr0
+ ; GCN-LABEL: name: shrink_disjoint_or_live_scc
+ ; GCN: liveins: $sgpr0
+ ; GCN-NEXT: {{ $}}
+ ; GCN-NEXT: $sgpr0 = disjoint S_OR_B32 $sgpr0, 257, implicit-def $scc
+ ; GCN-NEXT: $sgpr1 = S_CSELECT_B32 $sgpr0, 0, implicit $scc
+ $sgpr0 = disjoint S_OR_B32 $sgpr0, 257, implicit-def $scc
+ $sgpr1 = S_CSELECT_B32 $sgpr0, 0, implicit $scc
+
+...
+---
+name: shrink_disjoint_or_dead_scc
+tracksRegLiveness: true
+body: |
+ bb.0:
+ liveins: $sgpr0
+ ; GCN-LABEL: name: shrink_disjoint_or_dead_scc
+ ; GCN: liveins: $sgpr0
+ ; GCN-NEXT: {{ $}}
+ ; GCN-NEXT: $sgpr0 = disjoint S_ADDK_I32 $sgpr0, 257, implicit-def dead $scc
+ $sgpr0 = disjoint S_OR_B32 $sgpr0, 257, implicit-def dead $scc
+
+...
More information about the llvm-commits
mailing list