[llvm] [AMDGPU] Generalize deletion of redundant s_or_b32 of a 64-bit select (PR #228034)
Jay Foad via llvm-commits
llvm-commits at lists.llvm.org
Fri Oct 2 00:15:55 PDT 2026
https://github.com/jayfoad updated https://github.com/llvm/llvm-project/pull/228034
>From f77792f9decf784a3937afe298ce5b85b99b8510 Mon Sep 17 00:00:00 2001
From: Jay Foad <jay.foad at amd.com>
Date: Thu, 1 Oct 2026 11:49:24 +0100
Subject: [PATCH 1/2] [AMDGPU] Generalize deletion of redundant s_or_b32 of a
64-bit select
When deleting s_or_b32 of the two halves of a 64-bit S_CSELECT, use
getSelectConstants instead of foldableSelect. This handles materialized
constants, and a select whose true value is 0 by inverting the SCC uses.
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply at anthropic.com>
---
llvm/lib/Target/AMDGPU/SIInstrInfo.cpp | 26 ++--
llvm/test/CodeGen/AMDGPU/optimize-compare.mir | 132 ++++++++++++++++++
2 files changed, 142 insertions(+), 16 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/SIInstrInfo.cpp b/llvm/lib/Target/AMDGPU/SIInstrInfo.cpp
index ad7073fd66ef9..8d1013735392a 100644
--- a/llvm/lib/Target/AMDGPU/SIInstrInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/SIInstrInfo.cpp
@@ -11660,17 +11660,6 @@ bool SIInstrInfo::optimizeSCC(MachineInstr *SCCValid, MachineInstr *SCCRedefine,
return true;
}
-static bool foldableSelect(const MachineInstr &Def) {
- if (Def.getOpcode() != AMDGPU::S_CSELECT_B32 &&
- Def.getOpcode() != AMDGPU::S_CSELECT_B64)
- return false;
- bool Op1IsNonZeroImm =
- Def.getOperand(1).isImm() && Def.getOperand(1).getImm() != 0;
- bool Op2IsZeroImm =
- Def.getOperand(2).isImm() && Def.getOperand(2).getImm() == 0;
- return Op1IsNonZeroImm && Op2IsZeroImm;
-}
-
/// If \p Sel is an S_CSELECT* of two different constants A and B, return them,
/// truncated to the width of the select.
static std::optional<std::pair<int64_t, int64_t>>
@@ -11780,8 +11769,8 @@ bool SIInstrInfo::optimizeCompareInstr(MachineInstr &CmpInstr, Register SrcReg,
// If s_or_b32 result, sY, is unused (i.e. it is effectively a 64-bit
// s_cmp_lg of a register pair) and the inputs are the hi and lo-halves of a
- // 64-bit foldableSelect then delete s_or_b32 in the sequence:
- // sX = s_cselect_b64 (non-zero imm), 0
+ // 64-bit select then delete s_or_b32 in the sequence:
+ // sX = s_cselect_b64 A, B (A != B, one of them 0)
// sLo = copy sX.sub0
// sHi = copy sX.sub1
// sY = s_or_b32 sLo, sHi
@@ -11798,9 +11787,14 @@ bool SIInstrInfo::optimizeCompareInstr(MachineInstr &CmpInstr, Register SrcReg,
Def1->getOperand(1).getSubReg() == AMDGPU::sub0 &&
Def2->getOperand(1).getSubReg() == AMDGPU::sub1 &&
Def1->getOperand(1).getReg() == Def2->getOperand(1).getReg()) {
- MachineInstr *Select = MRI->getVRegDef(Def1->getOperand(1).getReg());
- if (Select && foldableSelect(*Select))
- optimizeSCC(Select, Def, /*NeedInversion=*/false);
+ if (MachineInstr *Select =
+ MRI->getVRegDef(Def1->getOperand(1).getReg())) {
+ if (auto Consts = getSelectConstants(*this, *MRI, *Select)) {
+ auto [A, B] = *Consts;
+ if (A == 0 || B == 0)
+ optimizeSCC(Select, Def, /*NeedInversion=*/A == 0);
+ }
+ }
}
}
}
diff --git a/llvm/test/CodeGen/AMDGPU/optimize-compare.mir b/llvm/test/CodeGen/AMDGPU/optimize-compare.mir
index ac87460abcaef..c90586db5809a 100644
--- a/llvm/test/CodeGen/AMDGPU/optimize-compare.mir
+++ b/llvm/test/CodeGen/AMDGPU/optimize-compare.mir
@@ -2447,6 +2447,138 @@ body: |
...
+---
+# Delete s_or_b32 and invert the SCC use since the select's true value is 0.
+name: s_cselect_b64_0_x_s_or_b32_s_cmp_lg_u32_0x00000000
+body: |
+ ; GCN-LABEL: name: s_cselect_b64_0_x_s_or_b32_s_cmp_lg_u32_0x00000000
+ ; GCN: bb.0:
+ ; GCN-NEXT: successors: %bb.1(0x40000000), %bb.2(0x40000000)
+ ; GCN-NEXT: liveins: $sgpr0
+ ; GCN-NEXT: {{ $}}
+ ; GCN-NEXT: [[COPY:%[0-9]+]]:sreg_32 = COPY $sgpr0
+ ; GCN-NEXT: S_CMP_LG_U32 [[COPY]], 0, implicit-def $scc
+ ; GCN-NEXT: [[S_CSELECT_B64_:%[0-9]+]]:sreg_64_xexec = S_CSELECT_B64 0, -1, implicit $scc
+ ; GCN-NEXT: [[COPY1:%[0-9]+]]:sreg_32_xm0_xexec = COPY [[S_CSELECT_B64_]].sub0
+ ; GCN-NEXT: [[COPY2:%[0-9]+]]:sreg_32_xm0_xexec = COPY [[S_CSELECT_B64_]].sub1
+ ; GCN-NEXT: S_CBRANCH_SCC1 %bb.2, implicit $scc
+ ; GCN-NEXT: S_BRANCH %bb.1
+ ; GCN-NEXT: {{ $}}
+ ; GCN-NEXT: bb.1:
+ ; GCN-NEXT: successors: %bb.2(0x80000000)
+ ; GCN-NEXT: {{ $}}
+ ; GCN-NEXT: bb.2:
+ ; GCN-NEXT: S_ENDPGM 0
+ bb.0:
+ successors: %bb.1, %bb.2
+ liveins: $sgpr0
+ %2:sreg_32 = COPY $sgpr0
+ S_CMP_LG_U32 %2, 0, implicit-def $scc
+ %31:sreg_64_xexec = S_CSELECT_B64 0, -1, implicit $scc
+ %40:sreg_32_xm0_xexec = COPY %31.sub0:sreg_64_xexec
+ %41:sreg_32_xm0_xexec = COPY %31.sub1:sreg_64_xexec
+ %sgpr4:sreg_32 = S_OR_B32 %40:sreg_32_xm0_xexec, %41:sreg_32_xm0_xexec, implicit-def $scc
+ S_CMP_LG_U32 %sgpr4, 0, implicit-def $scc
+ S_CBRANCH_SCC0 %bb.2, implicit $scc
+ S_BRANCH %bb.1
+
+ bb.1:
+ successors: %bb.2
+
+ bb.2:
+ S_ENDPGM 0
+
+...
+
+---
+# Delete s_or_b32. The inversions for s_cmp_eq and for the select's true value
+# being 0 cancel out.
+name: s_cselect_b64_0_x_s_or_b32_s_cmp_eq_u32_0x00000000
+body: |
+ ; GCN-LABEL: name: s_cselect_b64_0_x_s_or_b32_s_cmp_eq_u32_0x00000000
+ ; GCN: bb.0:
+ ; GCN-NEXT: successors: %bb.1(0x40000000), %bb.2(0x40000000)
+ ; GCN-NEXT: liveins: $sgpr0
+ ; GCN-NEXT: {{ $}}
+ ; GCN-NEXT: [[COPY:%[0-9]+]]:sreg_32 = COPY $sgpr0
+ ; GCN-NEXT: S_CMP_LG_U32 [[COPY]], 0, implicit-def $scc
+ ; GCN-NEXT: [[S_CSELECT_B64_:%[0-9]+]]:sreg_64_xexec = S_CSELECT_B64 0, -1, implicit $scc
+ ; GCN-NEXT: [[COPY1:%[0-9]+]]:sreg_32_xm0_xexec = COPY [[S_CSELECT_B64_]].sub0
+ ; GCN-NEXT: [[COPY2:%[0-9]+]]:sreg_32_xm0_xexec = COPY [[S_CSELECT_B64_]].sub1
+ ; GCN-NEXT: S_CBRANCH_SCC0 %bb.2, implicit $scc
+ ; GCN-NEXT: S_BRANCH %bb.1
+ ; GCN-NEXT: {{ $}}
+ ; GCN-NEXT: bb.1:
+ ; GCN-NEXT: successors: %bb.2(0x80000000)
+ ; GCN-NEXT: {{ $}}
+ ; GCN-NEXT: bb.2:
+ ; GCN-NEXT: S_ENDPGM 0
+ bb.0:
+ successors: %bb.1, %bb.2
+ liveins: $sgpr0
+ %2:sreg_32 = COPY $sgpr0
+ S_CMP_LG_U32 %2, 0, implicit-def $scc
+ %31:sreg_64_xexec = S_CSELECT_B64 0, -1, implicit $scc
+ %40:sreg_32_xm0_xexec = COPY %31.sub0:sreg_64_xexec
+ %41:sreg_32_xm0_xexec = COPY %31.sub1:sreg_64_xexec
+ %sgpr4:sreg_32 = S_OR_B32 %40:sreg_32_xm0_xexec, %41:sreg_32_xm0_xexec, implicit-def $scc
+ S_CMP_EQ_U32 %sgpr4, 0, implicit-def $scc
+ S_CBRANCH_SCC0 %bb.2, implicit $scc
+ S_BRANCH %bb.1
+
+ bb.1:
+ successors: %bb.2
+
+ bb.2:
+ S_ENDPGM 0
+
+...
+
+---
+# Delete s_or_b32 when the select's non-zero value is materialized.
+name: s_cselect_b64_mov_0_s_or_b32_s_cmp_lg_u32_0x00000000
+body: |
+ ; GCN-LABEL: name: s_cselect_b64_mov_0_s_or_b32_s_cmp_lg_u32_0x00000000
+ ; GCN: bb.0:
+ ; GCN-NEXT: successors: %bb.1(0x40000000), %bb.2(0x40000000)
+ ; GCN-NEXT: liveins: $sgpr0
+ ; GCN-NEXT: {{ $}}
+ ; GCN-NEXT: [[COPY:%[0-9]+]]:sreg_32 = COPY $sgpr0
+ ; GCN-NEXT: [[S_MOV_B64_:%[0-9]+]]:sreg_64 = S_MOV_B64 -1
+ ; GCN-NEXT: S_CMP_LG_U32 [[COPY]], 0, implicit-def $scc
+ ; GCN-NEXT: [[S_CSELECT_B64_:%[0-9]+]]:sreg_64_xexec = S_CSELECT_B64 [[S_MOV_B64_]], 0, implicit $scc
+ ; GCN-NEXT: [[COPY1:%[0-9]+]]:sreg_32_xm0_xexec = COPY [[S_CSELECT_B64_]].sub0
+ ; GCN-NEXT: [[COPY2:%[0-9]+]]:sreg_32_xm0_xexec = COPY [[S_CSELECT_B64_]].sub1
+ ; GCN-NEXT: S_CBRANCH_SCC0 %bb.2, implicit $scc
+ ; GCN-NEXT: S_BRANCH %bb.1
+ ; GCN-NEXT: {{ $}}
+ ; GCN-NEXT: bb.1:
+ ; GCN-NEXT: successors: %bb.2(0x80000000)
+ ; GCN-NEXT: {{ $}}
+ ; GCN-NEXT: bb.2:
+ ; GCN-NEXT: S_ENDPGM 0
+ bb.0:
+ successors: %bb.1, %bb.2
+ liveins: $sgpr0
+ %2:sreg_32 = COPY $sgpr0
+ %5:sreg_64 = S_MOV_B64 -1
+ S_CMP_LG_U32 %2, 0, implicit-def $scc
+ %31:sreg_64_xexec = S_CSELECT_B64 %5, 0, implicit $scc
+ %40:sreg_32_xm0_xexec = COPY %31.sub0:sreg_64_xexec
+ %41:sreg_32_xm0_xexec = COPY %31.sub1:sreg_64_xexec
+ %sgpr4:sreg_32 = S_OR_B32 %40:sreg_32_xm0_xexec, %41:sreg_32_xm0_xexec, implicit-def $scc
+ S_CMP_LG_U32 %sgpr4, 0, implicit-def $scc
+ S_CBRANCH_SCC0 %bb.2, implicit $scc
+ S_BRANCH %bb.1
+
+ bb.1:
+ successors: %bb.2
+
+ bb.2:
+ S_ENDPGM 0
+
+...
+
# STARTT
---
# Delete s_cmp after s_add_u32 X, 1
>From b23db15bb31a7220b7669c1aa494fc99936a0b96 Mon Sep 17 00:00:00 2001
From: Jay Foad <jay.foad at amd.com>
Date: Thu, 1 Oct 2026 15:18:12 +0100
Subject: [PATCH 2/2] Compact MIR vreg numbers
---
llvm/test/CodeGen/AMDGPU/optimize-compare.mir | 41 +++++++++----------
1 file changed, 19 insertions(+), 22 deletions(-)
diff --git a/llvm/test/CodeGen/AMDGPU/optimize-compare.mir b/llvm/test/CodeGen/AMDGPU/optimize-compare.mir
index c90586db5809a..9c2d0b14dcb3d 100644
--- a/llvm/test/CodeGen/AMDGPU/optimize-compare.mir
+++ b/llvm/test/CodeGen/AMDGPU/optimize-compare.mir
@@ -2472,12 +2472,12 @@ body: |
bb.0:
successors: %bb.1, %bb.2
liveins: $sgpr0
- %2:sreg_32 = COPY $sgpr0
- S_CMP_LG_U32 %2, 0, implicit-def $scc
- %31:sreg_64_xexec = S_CSELECT_B64 0, -1, implicit $scc
- %40:sreg_32_xm0_xexec = COPY %31.sub0:sreg_64_xexec
- %41:sreg_32_xm0_xexec = COPY %31.sub1:sreg_64_xexec
- %sgpr4:sreg_32 = S_OR_B32 %40:sreg_32_xm0_xexec, %41:sreg_32_xm0_xexec, implicit-def $scc
+ %0:sreg_32 = COPY $sgpr0
+ S_CMP_LG_U32 %0, 0, implicit-def $scc
+ %1:sreg_64_xexec = S_CSELECT_B64 0, -1, implicit $scc
+ %2:sreg_32_xm0_xexec = COPY %1.sub0
+ %3:sreg_32_xm0_xexec = COPY %1.sub1
+ %sgpr4:sreg_32 = S_OR_B32 %2, %3, implicit-def $scc
S_CMP_LG_U32 %sgpr4, 0, implicit-def $scc
S_CBRANCH_SCC0 %bb.2, implicit $scc
S_BRANCH %bb.1
@@ -2487,7 +2487,6 @@ body: |
bb.2:
S_ENDPGM 0
-
...
---
@@ -2516,12 +2515,12 @@ body: |
bb.0:
successors: %bb.1, %bb.2
liveins: $sgpr0
- %2:sreg_32 = COPY $sgpr0
- S_CMP_LG_U32 %2, 0, implicit-def $scc
- %31:sreg_64_xexec = S_CSELECT_B64 0, -1, implicit $scc
- %40:sreg_32_xm0_xexec = COPY %31.sub0:sreg_64_xexec
- %41:sreg_32_xm0_xexec = COPY %31.sub1:sreg_64_xexec
- %sgpr4:sreg_32 = S_OR_B32 %40:sreg_32_xm0_xexec, %41:sreg_32_xm0_xexec, implicit-def $scc
+ %0:sreg_32 = COPY $sgpr0
+ S_CMP_LG_U32 %0, 0, implicit-def $scc
+ %1:sreg_64_xexec = S_CSELECT_B64 0, -1, implicit $scc
+ %2:sreg_32_xm0_xexec = COPY %1.sub0
+ %3:sreg_32_xm0_xexec = COPY %1.sub1
+ %sgpr4:sreg_32 = S_OR_B32 %2, %3, implicit-def $scc
S_CMP_EQ_U32 %sgpr4, 0, implicit-def $scc
S_CBRANCH_SCC0 %bb.2, implicit $scc
S_BRANCH %bb.1
@@ -2531,7 +2530,6 @@ body: |
bb.2:
S_ENDPGM 0
-
...
---
@@ -2560,13 +2558,13 @@ body: |
bb.0:
successors: %bb.1, %bb.2
liveins: $sgpr0
- %2:sreg_32 = COPY $sgpr0
- %5:sreg_64 = S_MOV_B64 -1
- S_CMP_LG_U32 %2, 0, implicit-def $scc
- %31:sreg_64_xexec = S_CSELECT_B64 %5, 0, implicit $scc
- %40:sreg_32_xm0_xexec = COPY %31.sub0:sreg_64_xexec
- %41:sreg_32_xm0_xexec = COPY %31.sub1:sreg_64_xexec
- %sgpr4:sreg_32 = S_OR_B32 %40:sreg_32_xm0_xexec, %41:sreg_32_xm0_xexec, implicit-def $scc
+ %0:sreg_32 = COPY $sgpr0
+ %1:sreg_64 = S_MOV_B64 -1
+ S_CMP_LG_U32 %0, 0, implicit-def $scc
+ %2:sreg_64_xexec = S_CSELECT_B64 %1, 0, implicit $scc
+ %3:sreg_32_xm0_xexec = COPY %2.sub0
+ %4:sreg_32_xm0_xexec = COPY %2.sub1
+ %sgpr4:sreg_32 = S_OR_B32 %3, %4, implicit-def $scc
S_CMP_LG_U32 %sgpr4, 0, implicit-def $scc
S_CBRANCH_SCC0 %bb.2, implicit $scc
S_BRANCH %bb.1
@@ -2576,7 +2574,6 @@ body: |
bb.2:
S_ENDPGM 0
-
...
# STARTT
More information about the llvm-commits
mailing list