[llvm] 4c53405 - [AMDGPU] Generalize deletion of redundant s_or_b32 of a 64-bit select (#228034)
via llvm-commits
llvm-commits at lists.llvm.org
Fri Oct 2 01:05:12 PDT 2026
Author: Jay Foad
Date: 2026-10-02T08:04:56Z
New Revision: 4c53405cace11d69e1f9870ca51626f366aa3110
URL: https://github.com/llvm/llvm-project/commit/4c53405cace11d69e1f9870ca51626f366aa3110
DIFF: https://github.com/llvm/llvm-project/commit/4c53405cace11d69e1f9870ca51626f366aa3110.diff
LOG: [AMDGPU] Generalize deletion of redundant s_or_b32 of a 64-bit select (#228034)
When deleting s_or_b32 of the two halves of a 64-bit S_CSELECT, use
getSelectConstants instead of foldableSelect. This handles materialized
constants, and a select whose true value is 0 by inverting the SCC uses.
Co-authored-by: Claude Opus 5.5 (1M context) <noreply at anthropic.com>
Added:
Modified:
llvm/lib/Target/AMDGPU/SIInstrInfo.cpp
llvm/test/CodeGen/AMDGPU/optimize-compare.mir
Removed:
################################################################################
diff --git a/llvm/lib/Target/AMDGPU/SIInstrInfo.cpp b/llvm/lib/Target/AMDGPU/SIInstrInfo.cpp
index a69898061c1f9..78ba4d6e01e2e 100644
--- a/llvm/lib/Target/AMDGPU/SIInstrInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/SIInstrInfo.cpp
@@ -11643,17 +11643,6 @@ bool SIInstrInfo::optimizeSCC(MachineInstr *SCCValid, MachineInstr *SCCRedefine,
return true;
}
-static bool foldableSelect(const MachineInstr &Def) {
- if (Def.getOpcode() != AMDGPU::S_CSELECT_B32 &&
- Def.getOpcode() != AMDGPU::S_CSELECT_B64)
- return false;
- bool Op1IsNonZeroImm =
- Def.getOperand(1).isImm() && Def.getOperand(1).getImm() != 0;
- bool Op2IsZeroImm =
- Def.getOperand(2).isImm() && Def.getOperand(2).getImm() == 0;
- return Op1IsNonZeroImm && Op2IsZeroImm;
-}
-
/// If \p Sel is an S_CSELECT* of two
diff erent constants A and B, return them,
/// truncated to the width of the select.
static std::optional<std::pair<int64_t, int64_t>>
@@ -11765,8 +11754,8 @@ bool SIInstrInfo::optimizeCompareInstr(MachineInstr &CmpInstr, Register SrcReg,
// If s_or_b32 result, sY, is unused (i.e. it is effectively a 64-bit
// s_cmp_lg of a register pair) and the inputs are the hi and lo-halves of a
- // 64-bit foldableSelect then delete s_or_b32 in the sequence:
- // sX = s_cselect_b64 (non-zero imm), 0
+ // 64-bit select then delete s_or_b32 in the sequence:
+ // sX = s_cselect_b64 A, B (A != B, one of them 0)
// sLo = copy sX.sub0
// sHi = copy sX.sub1
// sY = s_or_b32 sLo, sHi
@@ -11783,9 +11772,14 @@ bool SIInstrInfo::optimizeCompareInstr(MachineInstr &CmpInstr, Register SrcReg,
Def1->getOperand(1).getSubReg() == AMDGPU::sub0 &&
Def2->getOperand(1).getSubReg() == AMDGPU::sub1 &&
Def1->getOperand(1).getReg() == Def2->getOperand(1).getReg()) {
- MachineInstr *Select = MRI->getVRegDef(Def1->getOperand(1).getReg());
- if (Select && foldableSelect(*Select))
- optimizeSCC(Select, Def, /*NeedInversion=*/false);
+ if (MachineInstr *Select =
+ MRI->getVRegDef(Def1->getOperand(1).getReg())) {
+ if (auto Consts = getSelectConstants(*this, *MRI, *Select)) {
+ auto [A, B] = *Consts;
+ if (A == 0 || B == 0)
+ optimizeSCC(Select, Def, /*NeedInversion=*/A == 0);
+ }
+ }
}
}
}
diff --git a/llvm/test/CodeGen/AMDGPU/optimize-compare.mir b/llvm/test/CodeGen/AMDGPU/optimize-compare.mir
index ac87460abcaef..9c2d0b14dcb3d 100644
--- a/llvm/test/CodeGen/AMDGPU/optimize-compare.mir
+++ b/llvm/test/CodeGen/AMDGPU/optimize-compare.mir
@@ -2447,6 +2447,135 @@ body: |
...
+---
+# Delete s_or_b32 and invert the SCC use since the select's true value is 0.
+name: s_cselect_b64_0_x_s_or_b32_s_cmp_lg_u32_0x00000000
+body: |
+ ; GCN-LABEL: name: s_cselect_b64_0_x_s_or_b32_s_cmp_lg_u32_0x00000000
+ ; GCN: bb.0:
+ ; GCN-NEXT: successors: %bb.1(0x40000000), %bb.2(0x40000000)
+ ; GCN-NEXT: liveins: $sgpr0
+ ; GCN-NEXT: {{ $}}
+ ; GCN-NEXT: [[COPY:%[0-9]+]]:sreg_32 = COPY $sgpr0
+ ; GCN-NEXT: S_CMP_LG_U32 [[COPY]], 0, implicit-def $scc
+ ; GCN-NEXT: [[S_CSELECT_B64_:%[0-9]+]]:sreg_64_xexec = S_CSELECT_B64 0, -1, implicit $scc
+ ; GCN-NEXT: [[COPY1:%[0-9]+]]:sreg_32_xm0_xexec = COPY [[S_CSELECT_B64_]].sub0
+ ; GCN-NEXT: [[COPY2:%[0-9]+]]:sreg_32_xm0_xexec = COPY [[S_CSELECT_B64_]].sub1
+ ; GCN-NEXT: S_CBRANCH_SCC1 %bb.2, implicit $scc
+ ; GCN-NEXT: S_BRANCH %bb.1
+ ; GCN-NEXT: {{ $}}
+ ; GCN-NEXT: bb.1:
+ ; GCN-NEXT: successors: %bb.2(0x80000000)
+ ; GCN-NEXT: {{ $}}
+ ; GCN-NEXT: bb.2:
+ ; GCN-NEXT: S_ENDPGM 0
+ bb.0:
+ successors: %bb.1, %bb.2
+ liveins: $sgpr0
+ %0:sreg_32 = COPY $sgpr0
+ S_CMP_LG_U32 %0, 0, implicit-def $scc
+ %1:sreg_64_xexec = S_CSELECT_B64 0, -1, implicit $scc
+ %2:sreg_32_xm0_xexec = COPY %1.sub0
+ %3:sreg_32_xm0_xexec = COPY %1.sub1
+ %sgpr4:sreg_32 = S_OR_B32 %2, %3, implicit-def $scc
+ S_CMP_LG_U32 %sgpr4, 0, implicit-def $scc
+ S_CBRANCH_SCC0 %bb.2, implicit $scc
+ S_BRANCH %bb.1
+
+ bb.1:
+ successors: %bb.2
+
+ bb.2:
+ S_ENDPGM 0
+...
+
+---
+# Delete s_or_b32. The inversions for s_cmp_eq and for the select's true value
+# being 0 cancel out.
+name: s_cselect_b64_0_x_s_or_b32_s_cmp_eq_u32_0x00000000
+body: |
+ ; GCN-LABEL: name: s_cselect_b64_0_x_s_or_b32_s_cmp_eq_u32_0x00000000
+ ; GCN: bb.0:
+ ; GCN-NEXT: successors: %bb.1(0x40000000), %bb.2(0x40000000)
+ ; GCN-NEXT: liveins: $sgpr0
+ ; GCN-NEXT: {{ $}}
+ ; GCN-NEXT: [[COPY:%[0-9]+]]:sreg_32 = COPY $sgpr0
+ ; GCN-NEXT: S_CMP_LG_U32 [[COPY]], 0, implicit-def $scc
+ ; GCN-NEXT: [[S_CSELECT_B64_:%[0-9]+]]:sreg_64_xexec = S_CSELECT_B64 0, -1, implicit $scc
+ ; GCN-NEXT: [[COPY1:%[0-9]+]]:sreg_32_xm0_xexec = COPY [[S_CSELECT_B64_]].sub0
+ ; GCN-NEXT: [[COPY2:%[0-9]+]]:sreg_32_xm0_xexec = COPY [[S_CSELECT_B64_]].sub1
+ ; GCN-NEXT: S_CBRANCH_SCC0 %bb.2, implicit $scc
+ ; GCN-NEXT: S_BRANCH %bb.1
+ ; GCN-NEXT: {{ $}}
+ ; GCN-NEXT: bb.1:
+ ; GCN-NEXT: successors: %bb.2(0x80000000)
+ ; GCN-NEXT: {{ $}}
+ ; GCN-NEXT: bb.2:
+ ; GCN-NEXT: S_ENDPGM 0
+ bb.0:
+ successors: %bb.1, %bb.2
+ liveins: $sgpr0
+ %0:sreg_32 = COPY $sgpr0
+ S_CMP_LG_U32 %0, 0, implicit-def $scc
+ %1:sreg_64_xexec = S_CSELECT_B64 0, -1, implicit $scc
+ %2:sreg_32_xm0_xexec = COPY %1.sub0
+ %3:sreg_32_xm0_xexec = COPY %1.sub1
+ %sgpr4:sreg_32 = S_OR_B32 %2, %3, implicit-def $scc
+ S_CMP_EQ_U32 %sgpr4, 0, implicit-def $scc
+ S_CBRANCH_SCC0 %bb.2, implicit $scc
+ S_BRANCH %bb.1
+
+ bb.1:
+ successors: %bb.2
+
+ bb.2:
+ S_ENDPGM 0
+...
+
+---
+# Delete s_or_b32 when the select's non-zero value is materialized.
+name: s_cselect_b64_mov_0_s_or_b32_s_cmp_lg_u32_0x00000000
+body: |
+ ; GCN-LABEL: name: s_cselect_b64_mov_0_s_or_b32_s_cmp_lg_u32_0x00000000
+ ; GCN: bb.0:
+ ; GCN-NEXT: successors: %bb.1(0x40000000), %bb.2(0x40000000)
+ ; GCN-NEXT: liveins: $sgpr0
+ ; GCN-NEXT: {{ $}}
+ ; GCN-NEXT: [[COPY:%[0-9]+]]:sreg_32 = COPY $sgpr0
+ ; GCN-NEXT: [[S_MOV_B64_:%[0-9]+]]:sreg_64 = S_MOV_B64 -1
+ ; GCN-NEXT: S_CMP_LG_U32 [[COPY]], 0, implicit-def $scc
+ ; GCN-NEXT: [[S_CSELECT_B64_:%[0-9]+]]:sreg_64_xexec = S_CSELECT_B64 [[S_MOV_B64_]], 0, implicit $scc
+ ; GCN-NEXT: [[COPY1:%[0-9]+]]:sreg_32_xm0_xexec = COPY [[S_CSELECT_B64_]].sub0
+ ; GCN-NEXT: [[COPY2:%[0-9]+]]:sreg_32_xm0_xexec = COPY [[S_CSELECT_B64_]].sub1
+ ; GCN-NEXT: S_CBRANCH_SCC0 %bb.2, implicit $scc
+ ; GCN-NEXT: S_BRANCH %bb.1
+ ; GCN-NEXT: {{ $}}
+ ; GCN-NEXT: bb.1:
+ ; GCN-NEXT: successors: %bb.2(0x80000000)
+ ; GCN-NEXT: {{ $}}
+ ; GCN-NEXT: bb.2:
+ ; GCN-NEXT: S_ENDPGM 0
+ bb.0:
+ successors: %bb.1, %bb.2
+ liveins: $sgpr0
+ %0:sreg_32 = COPY $sgpr0
+ %1:sreg_64 = S_MOV_B64 -1
+ S_CMP_LG_U32 %0, 0, implicit-def $scc
+ %2:sreg_64_xexec = S_CSELECT_B64 %1, 0, implicit $scc
+ %3:sreg_32_xm0_xexec = COPY %2.sub0
+ %4:sreg_32_xm0_xexec = COPY %2.sub1
+ %sgpr4:sreg_32 = S_OR_B32 %3, %4, implicit-def $scc
+ S_CMP_LG_U32 %sgpr4, 0, implicit-def $scc
+ S_CBRANCH_SCC0 %bb.2, implicit $scc
+ S_BRANCH %bb.1
+
+ bb.1:
+ successors: %bb.2
+
+ bb.2:
+ S_ENDPGM 0
+...
+
# STARTT
---
# Delete s_cmp after s_add_u32 X, 1
More information about the llvm-commits
mailing list