[llvm] [AMDGPU] Generalize deletion of redundant s_or_b32 of a 64-bit select (PR #228034)

Jay Foad via llvm-commits llvm-commits at lists.llvm.org
Fri Oct 2 00:15:55 PDT 2026


https://github.com/jayfoad updated https://github.com/llvm/llvm-project/pull/228034

>From f77792f9decf784a3937afe298ce5b85b99b8510 Mon Sep 17 00:00:00 2001
From: Jay Foad <jay.foad at amd.com>
Date: Thu, 1 Oct 2026 11:49:24 +0100
Subject: [PATCH 1/2] [AMDGPU] Generalize deletion of redundant s_or_b32 of a
 64-bit select

When deleting s_or_b32 of the two halves of a 64-bit S_CSELECT, use
getSelectConstants instead of foldableSelect. This handles materialized
constants, and a select whose true value is 0 by inverting the SCC uses.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply at anthropic.com>
---
 llvm/lib/Target/AMDGPU/SIInstrInfo.cpp        |  26 ++--
 llvm/test/CodeGen/AMDGPU/optimize-compare.mir | 132 ++++++++++++++++++
 2 files changed, 142 insertions(+), 16 deletions(-)

diff --git a/llvm/lib/Target/AMDGPU/SIInstrInfo.cpp b/llvm/lib/Target/AMDGPU/SIInstrInfo.cpp
index ad7073fd66ef9..8d1013735392a 100644
--- a/llvm/lib/Target/AMDGPU/SIInstrInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/SIInstrInfo.cpp
@@ -11660,17 +11660,6 @@ bool SIInstrInfo::optimizeSCC(MachineInstr *SCCValid, MachineInstr *SCCRedefine,
   return true;
 }
 
-static bool foldableSelect(const MachineInstr &Def) {
-  if (Def.getOpcode() != AMDGPU::S_CSELECT_B32 &&
-      Def.getOpcode() != AMDGPU::S_CSELECT_B64)
-    return false;
-  bool Op1IsNonZeroImm =
-      Def.getOperand(1).isImm() && Def.getOperand(1).getImm() != 0;
-  bool Op2IsZeroImm =
-      Def.getOperand(2).isImm() && Def.getOperand(2).getImm() == 0;
-  return Op1IsNonZeroImm && Op2IsZeroImm;
-}
-
 /// If \p Sel is an S_CSELECT* of two different constants A and B, return them,
 /// truncated to the width of the select.
 static std::optional<std::pair<int64_t, int64_t>>
@@ -11780,8 +11769,8 @@ bool SIInstrInfo::optimizeCompareInstr(MachineInstr &CmpInstr, Register SrcReg,
 
     // If s_or_b32 result, sY, is unused (i.e. it is effectively a 64-bit
     // s_cmp_lg of a register pair) and the inputs are the hi and lo-halves of a
-    // 64-bit foldableSelect then delete s_or_b32 in the sequence:
-    //    sX = s_cselect_b64 (non-zero imm), 0
+    // 64-bit select then delete s_or_b32 in the sequence:
+    //    sX = s_cselect_b64 A, B  (A != B, one of them 0)
     //    sLo = copy sX.sub0
     //    sHi = copy sX.sub1
     //    sY = s_or_b32 sLo, sHi
@@ -11798,9 +11787,14 @@ bool SIInstrInfo::optimizeCompareInstr(MachineInstr &CmpInstr, Register SrcReg,
             Def1->getOperand(1).getSubReg() == AMDGPU::sub0 &&
             Def2->getOperand(1).getSubReg() == AMDGPU::sub1 &&
             Def1->getOperand(1).getReg() == Def2->getOperand(1).getReg()) {
-          MachineInstr *Select = MRI->getVRegDef(Def1->getOperand(1).getReg());
-          if (Select && foldableSelect(*Select))
-            optimizeSCC(Select, Def, /*NeedInversion=*/false);
+          if (MachineInstr *Select =
+                  MRI->getVRegDef(Def1->getOperand(1).getReg())) {
+            if (auto Consts = getSelectConstants(*this, *MRI, *Select)) {
+              auto [A, B] = *Consts;
+              if (A == 0 || B == 0)
+                optimizeSCC(Select, Def, /*NeedInversion=*/A == 0);
+            }
+          }
         }
       }
     }
diff --git a/llvm/test/CodeGen/AMDGPU/optimize-compare.mir b/llvm/test/CodeGen/AMDGPU/optimize-compare.mir
index ac87460abcaef..c90586db5809a 100644
--- a/llvm/test/CodeGen/AMDGPU/optimize-compare.mir
+++ b/llvm/test/CodeGen/AMDGPU/optimize-compare.mir
@@ -2447,6 +2447,138 @@ body:             |
 
 ...
 
+---
+# Delete s_or_b32 and invert the SCC use since the select's true value is 0.
+name:            s_cselect_b64_0_x_s_or_b32_s_cmp_lg_u32_0x00000000
+body:             |
+  ; GCN-LABEL: name: s_cselect_b64_0_x_s_or_b32_s_cmp_lg_u32_0x00000000
+  ; GCN: bb.0:
+  ; GCN-NEXT:   successors: %bb.1(0x40000000), %bb.2(0x40000000)
+  ; GCN-NEXT:   liveins: $sgpr0
+  ; GCN-NEXT: {{  $}}
+  ; GCN-NEXT:   [[COPY:%[0-9]+]]:sreg_32 = COPY $sgpr0
+  ; GCN-NEXT:   S_CMP_LG_U32 [[COPY]], 0, implicit-def $scc
+  ; GCN-NEXT:   [[S_CSELECT_B64_:%[0-9]+]]:sreg_64_xexec = S_CSELECT_B64 0, -1, implicit $scc
+  ; GCN-NEXT:   [[COPY1:%[0-9]+]]:sreg_32_xm0_xexec = COPY [[S_CSELECT_B64_]].sub0
+  ; GCN-NEXT:   [[COPY2:%[0-9]+]]:sreg_32_xm0_xexec = COPY [[S_CSELECT_B64_]].sub1
+  ; GCN-NEXT:   S_CBRANCH_SCC1 %bb.2, implicit $scc
+  ; GCN-NEXT:   S_BRANCH %bb.1
+  ; GCN-NEXT: {{  $}}
+  ; GCN-NEXT: bb.1:
+  ; GCN-NEXT:   successors: %bb.2(0x80000000)
+  ; GCN-NEXT: {{  $}}
+  ; GCN-NEXT: bb.2:
+  ; GCN-NEXT:   S_ENDPGM 0
+  bb.0:
+    successors: %bb.1, %bb.2
+    liveins: $sgpr0
+    %2:sreg_32 = COPY $sgpr0
+    S_CMP_LG_U32 %2, 0, implicit-def $scc
+    %31:sreg_64_xexec = S_CSELECT_B64 0, -1, implicit $scc
+    %40:sreg_32_xm0_xexec = COPY %31.sub0:sreg_64_xexec
+    %41:sreg_32_xm0_xexec = COPY %31.sub1:sreg_64_xexec
+    %sgpr4:sreg_32 = S_OR_B32 %40:sreg_32_xm0_xexec, %41:sreg_32_xm0_xexec, implicit-def $scc
+    S_CMP_LG_U32 %sgpr4, 0, implicit-def $scc
+    S_CBRANCH_SCC0 %bb.2, implicit $scc
+    S_BRANCH %bb.1
+
+  bb.1:
+    successors: %bb.2
+
+  bb.2:
+    S_ENDPGM 0
+
+...
+
+---
+# Delete s_or_b32. The inversions for s_cmp_eq and for the select's true value
+# being 0 cancel out.
+name:            s_cselect_b64_0_x_s_or_b32_s_cmp_eq_u32_0x00000000
+body:             |
+  ; GCN-LABEL: name: s_cselect_b64_0_x_s_or_b32_s_cmp_eq_u32_0x00000000
+  ; GCN: bb.0:
+  ; GCN-NEXT:   successors: %bb.1(0x40000000), %bb.2(0x40000000)
+  ; GCN-NEXT:   liveins: $sgpr0
+  ; GCN-NEXT: {{  $}}
+  ; GCN-NEXT:   [[COPY:%[0-9]+]]:sreg_32 = COPY $sgpr0
+  ; GCN-NEXT:   S_CMP_LG_U32 [[COPY]], 0, implicit-def $scc
+  ; GCN-NEXT:   [[S_CSELECT_B64_:%[0-9]+]]:sreg_64_xexec = S_CSELECT_B64 0, -1, implicit $scc
+  ; GCN-NEXT:   [[COPY1:%[0-9]+]]:sreg_32_xm0_xexec = COPY [[S_CSELECT_B64_]].sub0
+  ; GCN-NEXT:   [[COPY2:%[0-9]+]]:sreg_32_xm0_xexec = COPY [[S_CSELECT_B64_]].sub1
+  ; GCN-NEXT:   S_CBRANCH_SCC0 %bb.2, implicit $scc
+  ; GCN-NEXT:   S_BRANCH %bb.1
+  ; GCN-NEXT: {{  $}}
+  ; GCN-NEXT: bb.1:
+  ; GCN-NEXT:   successors: %bb.2(0x80000000)
+  ; GCN-NEXT: {{  $}}
+  ; GCN-NEXT: bb.2:
+  ; GCN-NEXT:   S_ENDPGM 0
+  bb.0:
+    successors: %bb.1, %bb.2
+    liveins: $sgpr0
+    %2:sreg_32 = COPY $sgpr0
+    S_CMP_LG_U32 %2, 0, implicit-def $scc
+    %31:sreg_64_xexec = S_CSELECT_B64 0, -1, implicit $scc
+    %40:sreg_32_xm0_xexec = COPY %31.sub0:sreg_64_xexec
+    %41:sreg_32_xm0_xexec = COPY %31.sub1:sreg_64_xexec
+    %sgpr4:sreg_32 = S_OR_B32 %40:sreg_32_xm0_xexec, %41:sreg_32_xm0_xexec, implicit-def $scc
+    S_CMP_EQ_U32 %sgpr4, 0, implicit-def $scc
+    S_CBRANCH_SCC0 %bb.2, implicit $scc
+    S_BRANCH %bb.1
+
+  bb.1:
+    successors: %bb.2
+
+  bb.2:
+    S_ENDPGM 0
+
+...
+
+---
+# Delete s_or_b32 when the select's non-zero value is materialized.
+name:            s_cselect_b64_mov_0_s_or_b32_s_cmp_lg_u32_0x00000000
+body:             |
+  ; GCN-LABEL: name: s_cselect_b64_mov_0_s_or_b32_s_cmp_lg_u32_0x00000000
+  ; GCN: bb.0:
+  ; GCN-NEXT:   successors: %bb.1(0x40000000), %bb.2(0x40000000)
+  ; GCN-NEXT:   liveins: $sgpr0
+  ; GCN-NEXT: {{  $}}
+  ; GCN-NEXT:   [[COPY:%[0-9]+]]:sreg_32 = COPY $sgpr0
+  ; GCN-NEXT:   [[S_MOV_B64_:%[0-9]+]]:sreg_64 = S_MOV_B64 -1
+  ; GCN-NEXT:   S_CMP_LG_U32 [[COPY]], 0, implicit-def $scc
+  ; GCN-NEXT:   [[S_CSELECT_B64_:%[0-9]+]]:sreg_64_xexec = S_CSELECT_B64 [[S_MOV_B64_]], 0, implicit $scc
+  ; GCN-NEXT:   [[COPY1:%[0-9]+]]:sreg_32_xm0_xexec = COPY [[S_CSELECT_B64_]].sub0
+  ; GCN-NEXT:   [[COPY2:%[0-9]+]]:sreg_32_xm0_xexec = COPY [[S_CSELECT_B64_]].sub1
+  ; GCN-NEXT:   S_CBRANCH_SCC0 %bb.2, implicit $scc
+  ; GCN-NEXT:   S_BRANCH %bb.1
+  ; GCN-NEXT: {{  $}}
+  ; GCN-NEXT: bb.1:
+  ; GCN-NEXT:   successors: %bb.2(0x80000000)
+  ; GCN-NEXT: {{  $}}
+  ; GCN-NEXT: bb.2:
+  ; GCN-NEXT:   S_ENDPGM 0
+  bb.0:
+    successors: %bb.1, %bb.2
+    liveins: $sgpr0
+    %2:sreg_32 = COPY $sgpr0
+    %5:sreg_64 = S_MOV_B64 -1
+    S_CMP_LG_U32 %2, 0, implicit-def $scc
+    %31:sreg_64_xexec = S_CSELECT_B64 %5, 0, implicit $scc
+    %40:sreg_32_xm0_xexec = COPY %31.sub0:sreg_64_xexec
+    %41:sreg_32_xm0_xexec = COPY %31.sub1:sreg_64_xexec
+    %sgpr4:sreg_32 = S_OR_B32 %40:sreg_32_xm0_xexec, %41:sreg_32_xm0_xexec, implicit-def $scc
+    S_CMP_LG_U32 %sgpr4, 0, implicit-def $scc
+    S_CBRANCH_SCC0 %bb.2, implicit $scc
+    S_BRANCH %bb.1
+
+  bb.1:
+    successors: %bb.2
+
+  bb.2:
+    S_ENDPGM 0
+
+...
+
 # STARTT
 ---
 # Delete s_cmp after s_add_u32 X, 1

>From b23db15bb31a7220b7669c1aa494fc99936a0b96 Mon Sep 17 00:00:00 2001
From: Jay Foad <jay.foad at amd.com>
Date: Thu, 1 Oct 2026 15:18:12 +0100
Subject: [PATCH 2/2] Compact MIR vreg numbers

---
 llvm/test/CodeGen/AMDGPU/optimize-compare.mir | 41 +++++++++----------
 1 file changed, 19 insertions(+), 22 deletions(-)

diff --git a/llvm/test/CodeGen/AMDGPU/optimize-compare.mir b/llvm/test/CodeGen/AMDGPU/optimize-compare.mir
index c90586db5809a..9c2d0b14dcb3d 100644
--- a/llvm/test/CodeGen/AMDGPU/optimize-compare.mir
+++ b/llvm/test/CodeGen/AMDGPU/optimize-compare.mir
@@ -2472,12 +2472,12 @@ body:             |
   bb.0:
     successors: %bb.1, %bb.2
     liveins: $sgpr0
-    %2:sreg_32 = COPY $sgpr0
-    S_CMP_LG_U32 %2, 0, implicit-def $scc
-    %31:sreg_64_xexec = S_CSELECT_B64 0, -1, implicit $scc
-    %40:sreg_32_xm0_xexec = COPY %31.sub0:sreg_64_xexec
-    %41:sreg_32_xm0_xexec = COPY %31.sub1:sreg_64_xexec
-    %sgpr4:sreg_32 = S_OR_B32 %40:sreg_32_xm0_xexec, %41:sreg_32_xm0_xexec, implicit-def $scc
+    %0:sreg_32 = COPY $sgpr0
+    S_CMP_LG_U32 %0, 0, implicit-def $scc
+    %1:sreg_64_xexec = S_CSELECT_B64 0, -1, implicit $scc
+    %2:sreg_32_xm0_xexec = COPY %1.sub0
+    %3:sreg_32_xm0_xexec = COPY %1.sub1
+    %sgpr4:sreg_32 = S_OR_B32 %2, %3, implicit-def $scc
     S_CMP_LG_U32 %sgpr4, 0, implicit-def $scc
     S_CBRANCH_SCC0 %bb.2, implicit $scc
     S_BRANCH %bb.1
@@ -2487,7 +2487,6 @@ body:             |
 
   bb.2:
     S_ENDPGM 0
-
 ...
 
 ---
@@ -2516,12 +2515,12 @@ body:             |
   bb.0:
     successors: %bb.1, %bb.2
     liveins: $sgpr0
-    %2:sreg_32 = COPY $sgpr0
-    S_CMP_LG_U32 %2, 0, implicit-def $scc
-    %31:sreg_64_xexec = S_CSELECT_B64 0, -1, implicit $scc
-    %40:sreg_32_xm0_xexec = COPY %31.sub0:sreg_64_xexec
-    %41:sreg_32_xm0_xexec = COPY %31.sub1:sreg_64_xexec
-    %sgpr4:sreg_32 = S_OR_B32 %40:sreg_32_xm0_xexec, %41:sreg_32_xm0_xexec, implicit-def $scc
+    %0:sreg_32 = COPY $sgpr0
+    S_CMP_LG_U32 %0, 0, implicit-def $scc
+    %1:sreg_64_xexec = S_CSELECT_B64 0, -1, implicit $scc
+    %2:sreg_32_xm0_xexec = COPY %1.sub0
+    %3:sreg_32_xm0_xexec = COPY %1.sub1
+    %sgpr4:sreg_32 = S_OR_B32 %2, %3, implicit-def $scc
     S_CMP_EQ_U32 %sgpr4, 0, implicit-def $scc
     S_CBRANCH_SCC0 %bb.2, implicit $scc
     S_BRANCH %bb.1
@@ -2531,7 +2530,6 @@ body:             |
 
   bb.2:
     S_ENDPGM 0
-
 ...
 
 ---
@@ -2560,13 +2558,13 @@ body:             |
   bb.0:
     successors: %bb.1, %bb.2
     liveins: $sgpr0
-    %2:sreg_32 = COPY $sgpr0
-    %5:sreg_64 = S_MOV_B64 -1
-    S_CMP_LG_U32 %2, 0, implicit-def $scc
-    %31:sreg_64_xexec = S_CSELECT_B64 %5, 0, implicit $scc
-    %40:sreg_32_xm0_xexec = COPY %31.sub0:sreg_64_xexec
-    %41:sreg_32_xm0_xexec = COPY %31.sub1:sreg_64_xexec
-    %sgpr4:sreg_32 = S_OR_B32 %40:sreg_32_xm0_xexec, %41:sreg_32_xm0_xexec, implicit-def $scc
+    %0:sreg_32 = COPY $sgpr0
+    %1:sreg_64 = S_MOV_B64 -1
+    S_CMP_LG_U32 %0, 0, implicit-def $scc
+    %2:sreg_64_xexec = S_CSELECT_B64 %1, 0, implicit $scc
+    %3:sreg_32_xm0_xexec = COPY %2.sub0
+    %4:sreg_32_xm0_xexec = COPY %2.sub1
+    %sgpr4:sreg_32 = S_OR_B32 %3, %4, implicit-def $scc
     S_CMP_LG_U32 %sgpr4, 0, implicit-def $scc
     S_CBRANCH_SCC0 %bb.2, implicit $scc
     S_BRANCH %bb.1
@@ -2576,7 +2574,6 @@ body:             |
 
   bb.2:
     S_ENDPGM 0
-
 ...
 
 # STARTT



More information about the llvm-commits mailing list