[llvm] [AMDGPU] Remove redundant S_AND handling from optimizeVccBranch (PR #228417)

Jay Foad via llvm-commits llvm-commits at lists.llvm.org
Fri Oct 2 07:23:14 PDT 2026


https://github.com/jayfoad updated https://github.com/llvm/llvm-project/pull/228417

>From 62d82c29da7f50847c7d57d044ff38b6afc4d702 Mon Sep 17 00:00:00 2001
From: Jay Foad <jay.foad at amd.com>
Date: Fri, 2 Oct 2026 13:26:11 +0100
Subject: [PATCH] [AMDGPU] Remove redundant S_AND handling from
 optimizeVccBranch

PR #227729 moved this optimization to SIFoldOperands::tryFoldAndExec
where it is more effective.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply at anthropic.com>
---
 llvm/lib/Target/AMDGPU/SIPreEmitPeephole.cpp  |  20 ---
 .../CodeGen/AMDGPU/insert-skip-from-vcc.mir   | 136 ------------------
 ...i-pre-emit-peephole-preserve-loop-info.mir |  26 ----
 3 files changed, 182 deletions(-)

diff --git a/llvm/lib/Target/AMDGPU/SIPreEmitPeephole.cpp b/llvm/lib/Target/AMDGPU/SIPreEmitPeephole.cpp
index 9b67cdd6be6f458..89461de967b85d5 100644
--- a/llvm/lib/Target/AMDGPU/SIPreEmitPeephole.cpp
+++ b/llvm/lib/Target/AMDGPU/SIPreEmitPeephole.cpp
@@ -180,14 +180,6 @@ bool SIPreEmitPeephole::optimizeVccBranch(MachineInstr &MI) const {
   // We end up with this pattern sometimes after basic block placement.
   // It happens while combining a block which assigns -1 or 0 to a saved mask
   // and another block which consumes that saved mask and then a branch.
-  //
-  // While searching this also performs the following substitution:
-  // vcc = V_CMP
-  // vcc = S_AND exec, vcc
-  // S_CBRANCH_VCC[N]Z
-  // =>
-  // vcc = V_CMP
-  // S_CBRANCH_VCC[N]Z
 
   bool Changed = false;
   MachineBasicBlock &MBB = *MI.getParent();
@@ -237,27 +229,15 @@ bool SIPreEmitPeephole::optimizeVccBranch(MachineInstr &MI) const {
     SReg = Op2.getReg();
     auto M = std::next(A);
     bool ReadsSreg = false;
-    bool ModifiesExec = false;
     for (; M != E; ++M) {
       if (M->definesRegister(SReg, TRI))
         break;
       if (M->modifiesRegister(SReg, TRI))
         return Changed;
       ReadsSreg |= M->readsRegister(SReg, TRI);
-      ModifiesExec |= M->modifiesRegister(ExecReg, TRI);
     }
     if (M == E)
       return Changed;
-    // If SReg is VCC and SReg definition is a VALU comparison.
-    // This means S_AND with EXEC is not required, unless
-    // the implicit def of SCC is alive.
-    // Erase the S_AND and return.
-    // Note: isVOPC is used instead of isCompare to catch V_CMP_CLASS
-    if (A->getOpcode() == And && SReg == CondReg && !ModifiesExec &&
-        TII->isVOPC(*M) && A->allImplicitDefsAreDead()) {
-      A->eraseFromParent();
-      return true;
-    }
 
     if (!M->isMoveImmediate() || !M->getOperand(1).isImm() ||
         (M->getOperand(1).getImm() != -1 && M->getOperand(1).getImm() != 0))
diff --git a/llvm/test/CodeGen/AMDGPU/insert-skip-from-vcc.mir b/llvm/test/CodeGen/AMDGPU/insert-skip-from-vcc.mir
index 3081a2c1d1fe1a0..92bd6324c31ca84 100644
--- a/llvm/test/CodeGen/AMDGPU/insert-skip-from-vcc.mir
+++ b/llvm/test/CodeGen/AMDGPU/insert-skip-from-vcc.mir
@@ -536,123 +536,6 @@ body:             |
     S_CBRANCH_VCCZ %bb.1, implicit $vcc
     S_ENDPGM 0
 ...
----
-# GCN-LABEL: name: and_cmp_vccz
-# GCN: V_CMP_EQ_U32_e32 0, killed $vgpr0, implicit-def $vcc, implicit $exec
-# GCN-NOT: S_AND_
-# GCN: S_CBRANCH_VCCZ %bb.1, implicit killed $vcc
-name:            and_cmp_vccz
-body:             |
-  bb.0:
-    S_NOP 0
-
-  bb.1:
-    S_NOP 0
-
-  bb.2:
-    V_CMP_EQ_U32_e32 0, killed $vgpr0, implicit-def $vcc, implicit $exec
-    $vcc = S_AND_B64 $exec, $vcc, implicit-def dead $scc
-    S_CBRANCH_VCCZ %bb.1, implicit killed $vcc
-    S_ENDPGM 0
-...
----
-# GCN-LABEL: name: and_cmp_vccnz
-# GCN: V_CMP_EQ_U32_e32 0, killed $vgpr0, implicit-def $vcc, implicit $exec
-# GCN-NOT: S_AND_
-# GCN: S_CBRANCH_VCCNZ %bb.1, implicit killed $vcc
-name:            and_cmp_vccnz
-body:             |
-  bb.0:
-    S_NOP 0
-
-  bb.1:
-    S_NOP 0
-
-  bb.2:
-    V_CMP_EQ_U32_e32 0, killed $vgpr0, implicit-def $vcc, implicit $exec
-    $vcc = S_AND_B64 $exec, $vcc, implicit-def dead $scc
-    S_CBRANCH_VCCNZ %bb.1, implicit killed $vcc
-    S_ENDPGM 0
-...
----
-# GCN-LABEL: name: andn2_cmp_vccz
-# GCN: V_CMP_EQ_U32_e32 0, killed $vgpr0, implicit-def $vcc, implicit $exec
-# GCN: $vcc = S_ANDN2_B64 $exec, $vcc, implicit-def dead $scc
-# GCN: S_CBRANCH_VCCZ %bb.1, implicit killed $vcc
-name:            andn2_cmp_vccz
-body:             |
-  bb.0:
-    S_NOP 0
-
-  bb.1:
-    S_NOP 0
-
-  bb.2:
-    V_CMP_EQ_U32_e32 0, killed $vgpr0, implicit-def $vcc, implicit $exec
-    $vcc = S_ANDN2_B64 $exec, $vcc, implicit-def dead $scc
-    S_CBRANCH_VCCZ %bb.1, implicit killed $vcc
-    S_ENDPGM 0
-...
----
-# GCN-LABEL: name: and_cmpclass_vccz
-# GCN: V_CMP_CLASS_F32_e32 killed $sgpr0, killed $vgpr0, implicit-def $vcc, implicit $exec
-# GCN-NOT: S_AND_
-# GCN: S_CBRANCH_VCCZ %bb.1, implicit killed $vcc
-name:            and_cmpclass_vccz
-body:             |
-  bb.0:
-    S_NOP 0
-
-  bb.1:
-    S_NOP 0
-
-  bb.2:
-    V_CMP_CLASS_F32_e32 killed $sgpr0, killed $vgpr0, implicit-def $vcc, implicit $exec
-    $vcc = S_AND_B64 $exec, $vcc, implicit-def dead $scc
-    S_CBRANCH_VCCZ %bb.1, implicit killed $vcc
-    S_ENDPGM 0
-...
----
-# GCN-LABEL: name: and_cmpx_vccz
-# GCN: V_CMPX_EQ_U32_e32 0, killed $vgpr0, implicit-def $vcc, implicit-def $exec, implicit $exec
-# GCN-NOT: S_AND_
-# GCN: S_CBRANCH_VCCZ %bb.1, implicit killed $vcc
-name:            and_cmpx_vccz
-body:             |
-  bb.0:
-    S_NOP 0
-
-  bb.1:
-    S_NOP 0
-
-  bb.2:
-    V_CMPX_EQ_U32_e32 0, killed $vgpr0, implicit-def $vcc, implicit-def $exec, implicit $exec
-    $vcc = S_AND_B64 $exec, $vcc, implicit-def dead $scc
-    S_CBRANCH_VCCZ %bb.1, implicit killed $vcc
-    S_ENDPGM 0
-...
----
-# GCN-LABEL: name: and_or_cmp_vccz
-# GCN: V_CMP_EQ_U32_e32 0, killed $vgpr0, implicit-def $vcc, implicit $exec
-# GCN: $exec = S_OR_B64 $exec, $sgpr0_sgpr1, implicit-def dead $scc
-# GCN: $vcc = S_AND_B64 $exec, $vcc, implicit-def dead $scc
-# GCN: S_CBRANCH_VCCZ %bb.1, implicit killed $vcc
-name:            and_or_cmp_vccz
-body:             |
-  bb.0:
-    S_NOP 0
-
-  bb.1:
-    S_NOP 0
-
-  bb.2:
-    V_CMP_EQ_U32_e32 0, killed $vgpr0, implicit-def $vcc, implicit $exec
-    $exec = S_OR_B64 $exec, $sgpr0_sgpr1, implicit-def dead $scc
-    $vcc = S_AND_B64 $exec, $vcc, implicit-def dead $scc
-    S_CBRANCH_VCCZ %bb.1, implicit killed $vcc
-    S_ENDPGM 0
-...
-
 ---
 # GCN-LABEL: name: issue176578_src0_nonreg
 # GCN: bb.2:
@@ -708,22 +591,3 @@ body:             |
     S_CBRANCH_VCCZ %bb.1, implicit killed $vcc
     S_ENDPGM 0
 ...
-
----
-# GCN-LABEL: name: and_cmp_vccnz_live_scc
-# GCN: V_CMP_GT_I64_e32 $sgpr0_sgpr1, $vgpr0_vgpr1, implicit-def $vcc, implicit $exec
-# GCN-NEXT: $vcc = S_AND_B64 $exec, $vcc, implicit-def $scc
-# GCN-NEXT: S_CBRANCH_VCCNZ %bb.1, implicit $vcc
-# GCN: $sgpr0 = S_CSELECT_B32 $sgpr2, $sgpr3, implicit $scc
-name:            and_cmp_vccnz_live_scc
-body:             |
-  bb.0:
-    liveins: $sgpr0_sgpr1, $sgpr2, $sgpr3, $vgpr0_vgpr1
-    V_CMP_GT_I64_e32 $sgpr0_sgpr1, $vgpr0_vgpr1, implicit-def $vcc, implicit $exec
-    $vcc = S_AND_B64 $exec, $vcc, implicit-def $scc
-    S_CBRANCH_VCCNZ %bb.1, implicit $vcc
-
-  bb.1:
-    $sgpr0 = S_CSELECT_B32 $sgpr2, $sgpr3, implicit $scc
-    S_ENDPGM 0
-...
diff --git a/llvm/test/CodeGen/AMDGPU/si-pre-emit-peephole-preserve-loop-info.mir b/llvm/test/CodeGen/AMDGPU/si-pre-emit-peephole-preserve-loop-info.mir
index 89424af72127745..a6a535bcb4a54e0 100644
--- a/llvm/test/CodeGen/AMDGPU/si-pre-emit-peephole-preserve-loop-info.mir
+++ b/llvm/test/CodeGen/AMDGPU/si-pre-emit-peephole-preserve-loop-info.mir
@@ -1,31 +1,5 @@
 # RUN: llc -mtriple=amdgpu9.00-amd-amdhsa -passes="require<machine-dom-tree>,require<machine-loops>,si-pre-emit-peephole,print<machine-loops>" -debug-pass-manager -filetype=null %s 2>&1 | FileCheck %s
 
-# CHECK: Running analysis: MachineDominatorTreeAnalysis on vcc_and_removal_preserves_mli
-# CHECK: Running analysis: MachineLoopAnalysis on vcc_and_removal_preserves_mli
-# CHECK-NEXT: Running pass: SIPreEmitPeepholePass on vcc_and_removal_preserves_mli
-# CHECK-NEXT: Invalidating analysis: MachineDominatorTreeAnalysis on vcc_and_removal_preserves_mli
-# CHECK-NEXT: Running pass: MachineLoopPrinterPass on vcc_and_removal_preserves_mli
-# CHECK-NEXT: Machine loop info for machine function 'vcc_and_removal_preserves_mli':
-# CHECK-NOT: Running analysis: MachineLoopAnalysis on vcc_and_removal_preserves_mli
-# CHECK-NEXT: Loop at depth 1 containing: %bb.1<header><latch><exiting>
-
----
-name: vcc_and_removal_preserves_mli
-body: |
-  bb.0:
-    S_BRANCH %bb.1
-
-  ; S_AND gets removed
-  bb.1:
-    V_CMP_EQ_U32_e32 0, $vgpr0, implicit-def $vcc, implicit $exec
-    $vcc = S_AND_B64 $exec, $vcc, implicit-def dead $scc
-    S_CBRANCH_VCCNZ %bb.1, implicit $vcc
-    S_BRANCH %bb.2
-
-  bb.2:
-    S_ENDPGM 0
-...
-
 # CHECK-LABEL: Running pass: SIPreEmitPeepholePass on vcc_branch_destroys_loop
 # CHECK-NOT: Running analysis: MachineLoopAnalysis on vcc_branch_destroys_loop
 # CHECK: Machine loop info for machine function 'vcc_branch_destroys_loop':



More information about the llvm-commits mailing list