[llvm] [CodeGenPrepare] Don't sink icmp eq (and X, mask), 0 into return blocks (PR #200460)

Aayush Shrivastava via llvm-commits llvm-commits at lists.llvm.org
Fri Jun 26 15:49:06 PDT 2026


https://github.com/iamaayushrivastava updated https://github.com/llvm/llvm-project/pull/200460

>From 10082bd583473730a5f96befebf8a16361c4f9ca Mon Sep 17 00:00:00 2001
From: Aayush Shrivastava <iamaayushrivastava at gmail.com>
Date: Mon, 8 Jun 2026 20:44:57 +0530
Subject: [PATCH 1/3] [CodeGenPrepare][X86] Tests for icmp eq (and X, mask), 0
 return-block sinking

---
 .../CodeGen/X86/bool-ret-from-branch-cond.ll  | 31 +++++++++
 .../X86/sink-cmp-no-sink-into-ret.ll          | 68 +++++++++++++++++++
 2 files changed, 99 insertions(+)
 create mode 100644 llvm/test/CodeGen/X86/bool-ret-from-branch-cond.ll
 create mode 100644 llvm/test/Transforms/CodeGenPrepare/X86/sink-cmp-no-sink-into-ret.ll

diff --git a/llvm/test/CodeGen/X86/bool-ret-from-branch-cond.ll b/llvm/test/CodeGen/X86/bool-ret-from-branch-cond.ll
new file mode 100644
index 0000000000000..36ecec6da2117
--- /dev/null
+++ b/llvm/test/CodeGen/X86/bool-ret-from-branch-cond.ll
@@ -0,0 +1,31 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 2
+; RUN: llc -mtriple=x86_64 -O2 < %s | FileCheck %s
+
+; Returning a boolean that is also the branch condition currently produces a
+; redundant test/cmp instruction: the generated assembly contains a duplicate
+; "testb $3, %dil; sete %al" sequence in the shared return block, in addition
+; to the "testb $3, %dil; jne" used for the branch.
+
+define i1 @bool_ret(i64 %x, ptr %y) {
+; CHECK-LABEL: bool_ret:
+; CHECK:       # %bb.0: # %entry
+; CHECK-NEXT:    testb $3, %dil
+; CHECK-NEXT:    jne .LBB0_2
+; CHECK-NEXT:  # %bb.1: # %if.then
+; CHECK-NEXT:    movl $0, (%rsi)
+; CHECK-NEXT:  .LBB0_2: # %return
+; CHECK-NEXT:    testb $3, %dil
+; CHECK-NEXT:    sete %al
+; CHECK-NEXT:    retq
+entry:
+  %rem = and i64 %x, 3
+  %cmp = icmp eq i64 %rem, 0
+  br i1 %cmp, label %if.then, label %return
+
+if.then:
+  store i32 0, ptr %y, align 4
+  br label %return
+
+return:
+  ret i1 %cmp
+}
diff --git a/llvm/test/Transforms/CodeGenPrepare/X86/sink-cmp-no-sink-into-ret.ll b/llvm/test/Transforms/CodeGenPrepare/X86/sink-cmp-no-sink-into-ret.ll
new file mode 100644
index 0000000000000..ff51704524209
--- /dev/null
+++ b/llvm/test/Transforms/CodeGenPrepare/X86/sink-cmp-no-sink-into-ret.ll
@@ -0,0 +1,68 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 2
+; RUN: opt -S -mtriple=x86_64-- -codegenprepare < %s | FileCheck %s
+
+; sinkCmpExpression currently sinks an "icmp eq (and X, mask), 0" comparison
+; into a block that only returns the result, giving the 'and' operand a second
+; cross-block use.  This causes sinkAndCmp0Expression to also duplicate the
+; 'and' in the return block, producing a redundant TEST+SETCC sequence in the
+; generated code (see bool-ret-from-branch-cond.ll).
+
+define i1 @test_no_sink_into_ret(i64 %x, ptr %y) {
+; CHECK-LABEL: define i1 @test_no_sink_into_ret
+; CHECK-SAME: (i64 [[X:%.*]], ptr [[Y:%.*]]) {
+; CHECK-NEXT:  entry:
+; CHECK-NEXT:    [[TMP0:%.*]] = and i64 [[X]], 3
+; CHECK-NEXT:    [[CMP:%.*]] = icmp eq i64 [[TMP0]], 0
+; CHECK-NEXT:    br i1 [[CMP]], label [[IF_THEN:%.*]], label [[RETURN:%.*]]
+; CHECK:       if.then:
+; CHECK-NEXT:    store i32 0, ptr [[Y]], align 4
+; CHECK-NEXT:    br label [[RETURN]]
+; CHECK:       return:
+; CHECK-NEXT:    [[TMP1:%.*]] = and i64 [[X]], 3
+; CHECK-NEXT:    [[TMP2:%.*]] = icmp eq i64 [[TMP1]], 0
+; CHECK-NEXT:    ret i1 [[TMP2]]
+;
+entry:
+  %rem = and i64 %x, 3
+  %cmp = icmp eq i64 %rem, 0
+  br i1 %cmp, label %if.then, label %return
+
+if.then:
+  store i32 0, ptr %y, align 4
+  br label %return
+
+return:
+  ret i1 %cmp
+}
+
+; Multiple predecessors: condition is known at each predecessor but must
+; not be sunk when it fits the and+icmp-eq-0 / TEST pattern.
+define i1 @test_no_sink_into_ret_multi_pred(i64 %x, ptr %y, ptr %z) {
+; CHECK-LABEL: define i1 @test_no_sink_into_ret_multi_pred
+; CHECK-SAME: (i64 [[X:%.*]], ptr [[Y:%.*]], ptr [[Z:%.*]]) {
+; CHECK-NEXT:  entry:
+; CHECK-NEXT:    [[TMP0:%.*]] = and i64 [[X]], 3
+; CHECK-NEXT:    [[CMP:%.*]] = icmp eq i64 [[TMP0]], 0
+; CHECK-NEXT:    br i1 [[CMP]], label [[IF_THEN:%.*]], label [[RETURN:%.*]]
+; CHECK:       if.then:
+; CHECK-NEXT:    store i32 0, ptr [[Y]], align 4
+; CHECK-NEXT:    store i32 1, ptr [[Z]], align 4
+; CHECK-NEXT:    br label [[RETURN]]
+; CHECK:       return:
+; CHECK-NEXT:    [[TMP1:%.*]] = and i64 [[X]], 3
+; CHECK-NEXT:    [[TMP2:%.*]] = icmp eq i64 [[TMP1]], 0
+; CHECK-NEXT:    ret i1 [[TMP2]]
+;
+entry:
+  %rem = and i64 %x, 3
+  %cmp = icmp eq i64 %rem, 0
+  br i1 %cmp, label %if.then, label %return
+
+if.then:
+  store i32 0, ptr %y, align 4
+  store i32 1, ptr %z, align 4
+  br label %return
+
+return:
+  ret i1 %cmp
+}

>From e56279af3b7a0f21e8d03e0bd866a2e00c81e7d9 Mon Sep 17 00:00:00 2001
From: Aayush Shrivastava <iamaayushrivastava at gmail.com>
Date: Mon, 8 Jun 2026 20:47:10 +0530
Subject: [PATCH 2/3] [CodeGenPrepare] Avoid sinking icmp eq (and X, mask), 0
 into return blocks

---
 llvm/lib/CodeGen/CodeGenPrepare.cpp           | 21 +++++++++++++++
 .../CodeGen/X86/bool-ret-from-branch-cond.ll  | 26 +++++++++---------
 .../X86/sink-cmp-no-sink-into-ret.ll          | 27 +++++++++----------
 3 files changed, 47 insertions(+), 27 deletions(-)

diff --git a/llvm/lib/CodeGen/CodeGenPrepare.cpp b/llvm/lib/CodeGen/CodeGenPrepare.cpp
index 047ca6cebb1c6..8ef24fefb7091 100644
--- a/llvm/lib/CodeGen/CodeGenPrepare.cpp
+++ b/llvm/lib/CodeGen/CodeGenPrepare.cpp
@@ -1911,6 +1911,27 @@ static bool sinkCmpExpression(CmpInst *Cmp, const TargetLowering &TLI,
     if (isa<PHINode>(User))
       continue;
 
+    // Don't sink an "icmp eq (and X, mask), 0" into a return block when the
+    // 'and' currently has a single use in the same block as the cmp. Sinking
+    // would give the 'and' a second use (in the return block), causing
+    // sinkAndCmp0Expression to duplicate the 'and' there too, creating a
+    // redundant TEST+SETCC sequence. The backend's CopyToExportRegsIfNeeded
+    // mechanism handles this cross-block use of the condition without any
+    // extra work.
+    if (isa<ReturnInst>(User)) {
+      if (auto *ICmpI = dyn_cast<ICmpInst>(Cmp)) {
+        if (auto *AndI = dyn_cast<BinaryOperator>(ICmpI->getOperand(0))) {
+          if (AndI->getOpcode() == Instruction::And && AndI->hasOneUse() &&
+              AndI->getParent() == Cmp->getParent()) {
+            auto *CmpC = dyn_cast<ConstantInt>(ICmpI->getOperand(1));
+            if (CmpC && CmpC->isZero() &&
+                TLI.isMaskAndCmp0FoldingBeneficial(*AndI))
+              continue;
+          }
+        }
+      }
+    }
+
     // Figure out which BB this cmp is used in.
     BasicBlock *UserBB = User->getParent();
     BasicBlock *DefBB = Cmp->getParent();
diff --git a/llvm/test/CodeGen/X86/bool-ret-from-branch-cond.ll b/llvm/test/CodeGen/X86/bool-ret-from-branch-cond.ll
index 36ecec6da2117..f675aa6af8ad7 100644
--- a/llvm/test/CodeGen/X86/bool-ret-from-branch-cond.ll
+++ b/llvm/test/CodeGen/X86/bool-ret-from-branch-cond.ll
@@ -1,21 +1,23 @@
 ; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 2
 ; RUN: llc -mtriple=x86_64 -O2 < %s | FileCheck %s
 
-; Returning a boolean that is also the branch condition currently produces a
-; redundant test/cmp instruction: the generated assembly contains a duplicate
-; "testb $3, %dil; sete %al" sequence in the shared return block, in addition
-; to the "testb $3, %dil; jne" used for the branch.
+; Verify that returning a boolean that is also the branch condition does not
+; generate a redundant test/cmp instruction.  Before the fix, the assembly had
+; a duplicate "testb $3, %dil; sete %al" in the shared return block.
+;
+; With the fix the condition is materialised once right after the first testb,
+; and both return paths share that register - no second testb.
 
 define i1 @bool_ret(i64 %x, ptr %y) {
 ; CHECK-LABEL: bool_ret:
-; CHECK:       # %bb.0: # %entry
-; CHECK-NEXT:    testb $3, %dil
-; CHECK-NEXT:    jne .LBB0_2
-; CHECK-NEXT:  # %bb.1: # %if.then
-; CHECK-NEXT:    movl $0, (%rsi)
-; CHECK-NEXT:  .LBB0_2: # %return
-; CHECK-NEXT:    testb $3, %dil
-; CHECK-NEXT:    sete %al
+; CHECK:       # %bb.0:                                # %entry
+; CHECK-NEXT:    testb	$3, %dil
+; CHECK-NEXT:    sete	%al
+; CHECK-NEXT:    je	.LBB0_1
+; CHECK-NEXT:  # %bb.2:                                # %return
+; CHECK-NEXT:    retq
+; CHECK-NEXT:  .LBB0_1:                                # %if.then
+; CHECK-NEXT:    movl	$0, (%rsi)
 ; CHECK-NEXT:    retq
 entry:
   %rem = and i64 %x, 3
diff --git a/llvm/test/Transforms/CodeGenPrepare/X86/sink-cmp-no-sink-into-ret.ll b/llvm/test/Transforms/CodeGenPrepare/X86/sink-cmp-no-sink-into-ret.ll
index ff51704524209..145cea731ebe1 100644
--- a/llvm/test/Transforms/CodeGenPrepare/X86/sink-cmp-no-sink-into-ret.ll
+++ b/llvm/test/Transforms/CodeGenPrepare/X86/sink-cmp-no-sink-into-ret.ll
@@ -1,26 +1,25 @@
 ; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 2
 ; RUN: opt -S -mtriple=x86_64-- -codegenprepare < %s | FileCheck %s
 
-; sinkCmpExpression currently sinks an "icmp eq (and X, mask), 0" comparison
-; into a block that only returns the result, giving the 'and' operand a second
-; cross-block use.  This causes sinkAndCmp0Expression to also duplicate the
-; 'and' in the return block, producing a redundant TEST+SETCC sequence in the
-; generated code (see bool-ret-from-branch-cond.ll).
+; sinkCmpExpression must not sink an "icmp eq (and X, mask), 0" comparison into
+; a block that only returns the result.  Sinking would give the 'and' operand a
+; second cross-block use, causing sinkAndCmp0Expression to also duplicate the
+; 'and' in the return block and produce a redundant TEST+SETCC sequence.
+; The backend's CopyToExportRegsIfNeeded mechanism handles this case without
+; any extra work.
 
 define i1 @test_no_sink_into_ret(i64 %x, ptr %y) {
 ; CHECK-LABEL: define i1 @test_no_sink_into_ret
 ; CHECK-SAME: (i64 [[X:%.*]], ptr [[Y:%.*]]) {
 ; CHECK-NEXT:  entry:
-; CHECK-NEXT:    [[TMP0:%.*]] = and i64 [[X]], 3
-; CHECK-NEXT:    [[CMP:%.*]] = icmp eq i64 [[TMP0]], 0
+; CHECK-NEXT:    [[REM:%.*]] = and i64 [[X]], 3
+; CHECK-NEXT:    [[CMP:%.*]] = icmp eq i64 [[REM]], 0
 ; CHECK-NEXT:    br i1 [[CMP]], label [[IF_THEN:%.*]], label [[RETURN:%.*]]
 ; CHECK:       if.then:
 ; CHECK-NEXT:    store i32 0, ptr [[Y]], align 4
 ; CHECK-NEXT:    br label [[RETURN]]
 ; CHECK:       return:
-; CHECK-NEXT:    [[TMP1:%.*]] = and i64 [[X]], 3
-; CHECK-NEXT:    [[TMP2:%.*]] = icmp eq i64 [[TMP1]], 0
-; CHECK-NEXT:    ret i1 [[TMP2]]
+; CHECK-NEXT:    ret i1 [[CMP]]
 ;
 entry:
   %rem = and i64 %x, 3
@@ -41,17 +40,15 @@ define i1 @test_no_sink_into_ret_multi_pred(i64 %x, ptr %y, ptr %z) {
 ; CHECK-LABEL: define i1 @test_no_sink_into_ret_multi_pred
 ; CHECK-SAME: (i64 [[X:%.*]], ptr [[Y:%.*]], ptr [[Z:%.*]]) {
 ; CHECK-NEXT:  entry:
-; CHECK-NEXT:    [[TMP0:%.*]] = and i64 [[X]], 3
-; CHECK-NEXT:    [[CMP:%.*]] = icmp eq i64 [[TMP0]], 0
+; CHECK-NEXT:    [[REM:%.*]] = and i64 [[X]], 3
+; CHECK-NEXT:    [[CMP:%.*]] = icmp eq i64 [[REM]], 0
 ; CHECK-NEXT:    br i1 [[CMP]], label [[IF_THEN:%.*]], label [[RETURN:%.*]]
 ; CHECK:       if.then:
 ; CHECK-NEXT:    store i32 0, ptr [[Y]], align 4
 ; CHECK-NEXT:    store i32 1, ptr [[Z]], align 4
 ; CHECK-NEXT:    br label [[RETURN]]
 ; CHECK:       return:
-; CHECK-NEXT:    [[TMP1:%.*]] = and i64 [[X]], 3
-; CHECK-NEXT:    [[TMP2:%.*]] = icmp eq i64 [[TMP1]], 0
-; CHECK-NEXT:    ret i1 [[TMP2]]
+; CHECK-NEXT:    ret i1 [[CMP]]
 ;
 entry:
   %rem = and i64 %x, 3

>From 0a7df73f2dfb9c17f7127bb8cce80aec36eb21e7 Mon Sep 17 00:00:00 2001
From: Aayush Shrivastava <iamaayushrivastava at gmail.com>
Date: Mon, 8 Jun 2026 20:47:39 +0530
Subject: [PATCH 3/3] [CodeGenPrepare] Use match() in sinkCmpExpression
 return-block guard

---
 llvm/lib/CodeGen/CodeGenPrepare.cpp | 21 ++++++++-------------
 1 file changed, 8 insertions(+), 13 deletions(-)

diff --git a/llvm/lib/CodeGen/CodeGenPrepare.cpp b/llvm/lib/CodeGen/CodeGenPrepare.cpp
index 8ef24fefb7091..2a944d71debb2 100644
--- a/llvm/lib/CodeGen/CodeGenPrepare.cpp
+++ b/llvm/lib/CodeGen/CodeGenPrepare.cpp
@@ -1911,25 +1911,20 @@ static bool sinkCmpExpression(CmpInst *Cmp, const TargetLowering &TLI,
     if (isa<PHINode>(User))
       continue;
 
-    // Don't sink an "icmp eq (and X, mask), 0" into a return block when the
+    // Don't sink an "icmp (and X, mask), 0" into a return block when the
     // 'and' currently has a single use in the same block as the cmp. Sinking
     // would give the 'and' a second use (in the return block), causing
     // sinkAndCmp0Expression to duplicate the 'and' there too, creating a
     // redundant TEST+SETCC sequence. The backend's CopyToExportRegsIfNeeded
     // mechanism handles this cross-block use of the condition without any
     // extra work.
-    if (isa<ReturnInst>(User)) {
-      if (auto *ICmpI = dyn_cast<ICmpInst>(Cmp)) {
-        if (auto *AndI = dyn_cast<BinaryOperator>(ICmpI->getOperand(0))) {
-          if (AndI->getOpcode() == Instruction::And && AndI->hasOneUse() &&
-              AndI->getParent() == Cmp->getParent()) {
-            auto *CmpC = dyn_cast<ConstantInt>(ICmpI->getOperand(1));
-            if (CmpC && CmpC->isZero() &&
-                TLI.isMaskAndCmp0FoldingBeneficial(*AndI))
-              continue;
-          }
-        }
-      }
+    if (isa<ReturnInst>(User) &&
+        match(Cmp,
+              m_ICmp(m_OneUse(m_And(m_Value(), m_Value())), m_ZeroInt()))) {
+      auto *AndI = cast<BinaryOperator>(Cmp->getOperand(0));
+      if (AndI->getParent() == Cmp->getParent() &&
+          TLI.isMaskAndCmp0FoldingBeneficial(*AndI))
+        continue;
     }
 
     // Figure out which BB this cmp is used in.



More information about the llvm-commits mailing list