[llvm] [CodeGenPrepare] Don't sink icmp eq (and X, mask), 0 into return blocks (PR #200460)
Aayush Shrivastava via llvm-commits
llvm-commits at lists.llvm.org
Fri Jun 26 15:49:06 PDT 2026
https://github.com/iamaayushrivastava updated https://github.com/llvm/llvm-project/pull/200460
>From 10082bd583473730a5f96befebf8a16361c4f9ca Mon Sep 17 00:00:00 2001
From: Aayush Shrivastava <iamaayushrivastava at gmail.com>
Date: Mon, 8 Jun 2026 20:44:57 +0530
Subject: [PATCH 1/3] [CodeGenPrepare][X86] Tests for icmp eq (and X, mask), 0
return-block sinking
---
.../CodeGen/X86/bool-ret-from-branch-cond.ll | 31 +++++++++
.../X86/sink-cmp-no-sink-into-ret.ll | 68 +++++++++++++++++++
2 files changed, 99 insertions(+)
create mode 100644 llvm/test/CodeGen/X86/bool-ret-from-branch-cond.ll
create mode 100644 llvm/test/Transforms/CodeGenPrepare/X86/sink-cmp-no-sink-into-ret.ll
diff --git a/llvm/test/CodeGen/X86/bool-ret-from-branch-cond.ll b/llvm/test/CodeGen/X86/bool-ret-from-branch-cond.ll
new file mode 100644
index 0000000000000..36ecec6da2117
--- /dev/null
+++ b/llvm/test/CodeGen/X86/bool-ret-from-branch-cond.ll
@@ -0,0 +1,31 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 2
+; RUN: llc -mtriple=x86_64 -O2 < %s | FileCheck %s
+
+; Returning a boolean that is also the branch condition currently produces a
+; redundant test/cmp instruction: the generated assembly contains a duplicate
+; "testb $3, %dil; sete %al" sequence in the shared return block, in addition
+; to the "testb $3, %dil; jne" used for the branch.
+
+define i1 @bool_ret(i64 %x, ptr %y) {
+; CHECK-LABEL: bool_ret:
+; CHECK: # %bb.0: # %entry
+; CHECK-NEXT: testb $3, %dil
+; CHECK-NEXT: jne .LBB0_2
+; CHECK-NEXT: # %bb.1: # %if.then
+; CHECK-NEXT: movl $0, (%rsi)
+; CHECK-NEXT: .LBB0_2: # %return
+; CHECK-NEXT: testb $3, %dil
+; CHECK-NEXT: sete %al
+; CHECK-NEXT: retq
+entry:
+ %rem = and i64 %x, 3
+ %cmp = icmp eq i64 %rem, 0
+ br i1 %cmp, label %if.then, label %return
+
+if.then:
+ store i32 0, ptr %y, align 4
+ br label %return
+
+return:
+ ret i1 %cmp
+}
diff --git a/llvm/test/Transforms/CodeGenPrepare/X86/sink-cmp-no-sink-into-ret.ll b/llvm/test/Transforms/CodeGenPrepare/X86/sink-cmp-no-sink-into-ret.ll
new file mode 100644
index 0000000000000..ff51704524209
--- /dev/null
+++ b/llvm/test/Transforms/CodeGenPrepare/X86/sink-cmp-no-sink-into-ret.ll
@@ -0,0 +1,68 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 2
+; RUN: opt -S -mtriple=x86_64-- -codegenprepare < %s | FileCheck %s
+
+; sinkCmpExpression currently sinks an "icmp eq (and X, mask), 0" comparison
+; into a block that only returns the result, giving the 'and' operand a second
+; cross-block use. This causes sinkAndCmp0Expression to also duplicate the
+; 'and' in the return block, producing a redundant TEST+SETCC sequence in the
+; generated code (see bool-ret-from-branch-cond.ll).
+
+define i1 @test_no_sink_into_ret(i64 %x, ptr %y) {
+; CHECK-LABEL: define i1 @test_no_sink_into_ret
+; CHECK-SAME: (i64 [[X:%.*]], ptr [[Y:%.*]]) {
+; CHECK-NEXT: entry:
+; CHECK-NEXT: [[TMP0:%.*]] = and i64 [[X]], 3
+; CHECK-NEXT: [[CMP:%.*]] = icmp eq i64 [[TMP0]], 0
+; CHECK-NEXT: br i1 [[CMP]], label [[IF_THEN:%.*]], label [[RETURN:%.*]]
+; CHECK: if.then:
+; CHECK-NEXT: store i32 0, ptr [[Y]], align 4
+; CHECK-NEXT: br label [[RETURN]]
+; CHECK: return:
+; CHECK-NEXT: [[TMP1:%.*]] = and i64 [[X]], 3
+; CHECK-NEXT: [[TMP2:%.*]] = icmp eq i64 [[TMP1]], 0
+; CHECK-NEXT: ret i1 [[TMP2]]
+;
+entry:
+ %rem = and i64 %x, 3
+ %cmp = icmp eq i64 %rem, 0
+ br i1 %cmp, label %if.then, label %return
+
+if.then:
+ store i32 0, ptr %y, align 4
+ br label %return
+
+return:
+ ret i1 %cmp
+}
+
+; Multiple predecessors: condition is known at each predecessor but must
+; not be sunk when it fits the and+icmp-eq-0 / TEST pattern.
+define i1 @test_no_sink_into_ret_multi_pred(i64 %x, ptr %y, ptr %z) {
+; CHECK-LABEL: define i1 @test_no_sink_into_ret_multi_pred
+; CHECK-SAME: (i64 [[X:%.*]], ptr [[Y:%.*]], ptr [[Z:%.*]]) {
+; CHECK-NEXT: entry:
+; CHECK-NEXT: [[TMP0:%.*]] = and i64 [[X]], 3
+; CHECK-NEXT: [[CMP:%.*]] = icmp eq i64 [[TMP0]], 0
+; CHECK-NEXT: br i1 [[CMP]], label [[IF_THEN:%.*]], label [[RETURN:%.*]]
+; CHECK: if.then:
+; CHECK-NEXT: store i32 0, ptr [[Y]], align 4
+; CHECK-NEXT: store i32 1, ptr [[Z]], align 4
+; CHECK-NEXT: br label [[RETURN]]
+; CHECK: return:
+; CHECK-NEXT: [[TMP1:%.*]] = and i64 [[X]], 3
+; CHECK-NEXT: [[TMP2:%.*]] = icmp eq i64 [[TMP1]], 0
+; CHECK-NEXT: ret i1 [[TMP2]]
+;
+entry:
+ %rem = and i64 %x, 3
+ %cmp = icmp eq i64 %rem, 0
+ br i1 %cmp, label %if.then, label %return
+
+if.then:
+ store i32 0, ptr %y, align 4
+ store i32 1, ptr %z, align 4
+ br label %return
+
+return:
+ ret i1 %cmp
+}
>From e56279af3b7a0f21e8d03e0bd866a2e00c81e7d9 Mon Sep 17 00:00:00 2001
From: Aayush Shrivastava <iamaayushrivastava at gmail.com>
Date: Mon, 8 Jun 2026 20:47:10 +0530
Subject: [PATCH 2/3] [CodeGenPrepare] Avoid sinking icmp eq (and X, mask), 0
into return blocks
---
llvm/lib/CodeGen/CodeGenPrepare.cpp | 21 +++++++++++++++
.../CodeGen/X86/bool-ret-from-branch-cond.ll | 26 +++++++++---------
.../X86/sink-cmp-no-sink-into-ret.ll | 27 +++++++++----------
3 files changed, 47 insertions(+), 27 deletions(-)
diff --git a/llvm/lib/CodeGen/CodeGenPrepare.cpp b/llvm/lib/CodeGen/CodeGenPrepare.cpp
index 047ca6cebb1c6..8ef24fefb7091 100644
--- a/llvm/lib/CodeGen/CodeGenPrepare.cpp
+++ b/llvm/lib/CodeGen/CodeGenPrepare.cpp
@@ -1911,6 +1911,27 @@ static bool sinkCmpExpression(CmpInst *Cmp, const TargetLowering &TLI,
if (isa<PHINode>(User))
continue;
+ // Don't sink an "icmp eq (and X, mask), 0" into a return block when the
+ // 'and' currently has a single use in the same block as the cmp. Sinking
+ // would give the 'and' a second use (in the return block), causing
+ // sinkAndCmp0Expression to duplicate the 'and' there too, creating a
+ // redundant TEST+SETCC sequence. The backend's CopyToExportRegsIfNeeded
+ // mechanism handles this cross-block use of the condition without any
+ // extra work.
+ if (isa<ReturnInst>(User)) {
+ if (auto *ICmpI = dyn_cast<ICmpInst>(Cmp)) {
+ if (auto *AndI = dyn_cast<BinaryOperator>(ICmpI->getOperand(0))) {
+ if (AndI->getOpcode() == Instruction::And && AndI->hasOneUse() &&
+ AndI->getParent() == Cmp->getParent()) {
+ auto *CmpC = dyn_cast<ConstantInt>(ICmpI->getOperand(1));
+ if (CmpC && CmpC->isZero() &&
+ TLI.isMaskAndCmp0FoldingBeneficial(*AndI))
+ continue;
+ }
+ }
+ }
+ }
+
// Figure out which BB this cmp is used in.
BasicBlock *UserBB = User->getParent();
BasicBlock *DefBB = Cmp->getParent();
diff --git a/llvm/test/CodeGen/X86/bool-ret-from-branch-cond.ll b/llvm/test/CodeGen/X86/bool-ret-from-branch-cond.ll
index 36ecec6da2117..f675aa6af8ad7 100644
--- a/llvm/test/CodeGen/X86/bool-ret-from-branch-cond.ll
+++ b/llvm/test/CodeGen/X86/bool-ret-from-branch-cond.ll
@@ -1,21 +1,23 @@
; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 2
; RUN: llc -mtriple=x86_64 -O2 < %s | FileCheck %s
-; Returning a boolean that is also the branch condition currently produces a
-; redundant test/cmp instruction: the generated assembly contains a duplicate
-; "testb $3, %dil; sete %al" sequence in the shared return block, in addition
-; to the "testb $3, %dil; jne" used for the branch.
+; Verify that returning a boolean that is also the branch condition does not
+; generate a redundant test/cmp instruction. Before the fix, the assembly had
+; a duplicate "testb $3, %dil; sete %al" in the shared return block.
+;
+; With the fix the condition is materialised once right after the first testb,
+; and both return paths share that register - no second testb.
define i1 @bool_ret(i64 %x, ptr %y) {
; CHECK-LABEL: bool_ret:
-; CHECK: # %bb.0: # %entry
-; CHECK-NEXT: testb $3, %dil
-; CHECK-NEXT: jne .LBB0_2
-; CHECK-NEXT: # %bb.1: # %if.then
-; CHECK-NEXT: movl $0, (%rsi)
-; CHECK-NEXT: .LBB0_2: # %return
-; CHECK-NEXT: testb $3, %dil
-; CHECK-NEXT: sete %al
+; CHECK: # %bb.0: # %entry
+; CHECK-NEXT: testb $3, %dil
+; CHECK-NEXT: sete %al
+; CHECK-NEXT: je .LBB0_1
+; CHECK-NEXT: # %bb.2: # %return
+; CHECK-NEXT: retq
+; CHECK-NEXT: .LBB0_1: # %if.then
+; CHECK-NEXT: movl $0, (%rsi)
; CHECK-NEXT: retq
entry:
%rem = and i64 %x, 3
diff --git a/llvm/test/Transforms/CodeGenPrepare/X86/sink-cmp-no-sink-into-ret.ll b/llvm/test/Transforms/CodeGenPrepare/X86/sink-cmp-no-sink-into-ret.ll
index ff51704524209..145cea731ebe1 100644
--- a/llvm/test/Transforms/CodeGenPrepare/X86/sink-cmp-no-sink-into-ret.ll
+++ b/llvm/test/Transforms/CodeGenPrepare/X86/sink-cmp-no-sink-into-ret.ll
@@ -1,26 +1,25 @@
; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 2
; RUN: opt -S -mtriple=x86_64-- -codegenprepare < %s | FileCheck %s
-; sinkCmpExpression currently sinks an "icmp eq (and X, mask), 0" comparison
-; into a block that only returns the result, giving the 'and' operand a second
-; cross-block use. This causes sinkAndCmp0Expression to also duplicate the
-; 'and' in the return block, producing a redundant TEST+SETCC sequence in the
-; generated code (see bool-ret-from-branch-cond.ll).
+; sinkCmpExpression must not sink an "icmp eq (and X, mask), 0" comparison into
+; a block that only returns the result. Sinking would give the 'and' operand a
+; second cross-block use, causing sinkAndCmp0Expression to also duplicate the
+; 'and' in the return block and produce a redundant TEST+SETCC sequence.
+; The backend's CopyToExportRegsIfNeeded mechanism handles this case without
+; any extra work.
define i1 @test_no_sink_into_ret(i64 %x, ptr %y) {
; CHECK-LABEL: define i1 @test_no_sink_into_ret
; CHECK-SAME: (i64 [[X:%.*]], ptr [[Y:%.*]]) {
; CHECK-NEXT: entry:
-; CHECK-NEXT: [[TMP0:%.*]] = and i64 [[X]], 3
-; CHECK-NEXT: [[CMP:%.*]] = icmp eq i64 [[TMP0]], 0
+; CHECK-NEXT: [[REM:%.*]] = and i64 [[X]], 3
+; CHECK-NEXT: [[CMP:%.*]] = icmp eq i64 [[REM]], 0
; CHECK-NEXT: br i1 [[CMP]], label [[IF_THEN:%.*]], label [[RETURN:%.*]]
; CHECK: if.then:
; CHECK-NEXT: store i32 0, ptr [[Y]], align 4
; CHECK-NEXT: br label [[RETURN]]
; CHECK: return:
-; CHECK-NEXT: [[TMP1:%.*]] = and i64 [[X]], 3
-; CHECK-NEXT: [[TMP2:%.*]] = icmp eq i64 [[TMP1]], 0
-; CHECK-NEXT: ret i1 [[TMP2]]
+; CHECK-NEXT: ret i1 [[CMP]]
;
entry:
%rem = and i64 %x, 3
@@ -41,17 +40,15 @@ define i1 @test_no_sink_into_ret_multi_pred(i64 %x, ptr %y, ptr %z) {
; CHECK-LABEL: define i1 @test_no_sink_into_ret_multi_pred
; CHECK-SAME: (i64 [[X:%.*]], ptr [[Y:%.*]], ptr [[Z:%.*]]) {
; CHECK-NEXT: entry:
-; CHECK-NEXT: [[TMP0:%.*]] = and i64 [[X]], 3
-; CHECK-NEXT: [[CMP:%.*]] = icmp eq i64 [[TMP0]], 0
+; CHECK-NEXT: [[REM:%.*]] = and i64 [[X]], 3
+; CHECK-NEXT: [[CMP:%.*]] = icmp eq i64 [[REM]], 0
; CHECK-NEXT: br i1 [[CMP]], label [[IF_THEN:%.*]], label [[RETURN:%.*]]
; CHECK: if.then:
; CHECK-NEXT: store i32 0, ptr [[Y]], align 4
; CHECK-NEXT: store i32 1, ptr [[Z]], align 4
; CHECK-NEXT: br label [[RETURN]]
; CHECK: return:
-; CHECK-NEXT: [[TMP1:%.*]] = and i64 [[X]], 3
-; CHECK-NEXT: [[TMP2:%.*]] = icmp eq i64 [[TMP1]], 0
-; CHECK-NEXT: ret i1 [[TMP2]]
+; CHECK-NEXT: ret i1 [[CMP]]
;
entry:
%rem = and i64 %x, 3
>From 0a7df73f2dfb9c17f7127bb8cce80aec36eb21e7 Mon Sep 17 00:00:00 2001
From: Aayush Shrivastava <iamaayushrivastava at gmail.com>
Date: Mon, 8 Jun 2026 20:47:39 +0530
Subject: [PATCH 3/3] [CodeGenPrepare] Use match() in sinkCmpExpression
return-block guard
---
llvm/lib/CodeGen/CodeGenPrepare.cpp | 21 ++++++++-------------
1 file changed, 8 insertions(+), 13 deletions(-)
diff --git a/llvm/lib/CodeGen/CodeGenPrepare.cpp b/llvm/lib/CodeGen/CodeGenPrepare.cpp
index 8ef24fefb7091..2a944d71debb2 100644
--- a/llvm/lib/CodeGen/CodeGenPrepare.cpp
+++ b/llvm/lib/CodeGen/CodeGenPrepare.cpp
@@ -1911,25 +1911,20 @@ static bool sinkCmpExpression(CmpInst *Cmp, const TargetLowering &TLI,
if (isa<PHINode>(User))
continue;
- // Don't sink an "icmp eq (and X, mask), 0" into a return block when the
+ // Don't sink an "icmp (and X, mask), 0" into a return block when the
// 'and' currently has a single use in the same block as the cmp. Sinking
// would give the 'and' a second use (in the return block), causing
// sinkAndCmp0Expression to duplicate the 'and' there too, creating a
// redundant TEST+SETCC sequence. The backend's CopyToExportRegsIfNeeded
// mechanism handles this cross-block use of the condition without any
// extra work.
- if (isa<ReturnInst>(User)) {
- if (auto *ICmpI = dyn_cast<ICmpInst>(Cmp)) {
- if (auto *AndI = dyn_cast<BinaryOperator>(ICmpI->getOperand(0))) {
- if (AndI->getOpcode() == Instruction::And && AndI->hasOneUse() &&
- AndI->getParent() == Cmp->getParent()) {
- auto *CmpC = dyn_cast<ConstantInt>(ICmpI->getOperand(1));
- if (CmpC && CmpC->isZero() &&
- TLI.isMaskAndCmp0FoldingBeneficial(*AndI))
- continue;
- }
- }
- }
+ if (isa<ReturnInst>(User) &&
+ match(Cmp,
+ m_ICmp(m_OneUse(m_And(m_Value(), m_Value())), m_ZeroInt()))) {
+ auto *AndI = cast<BinaryOperator>(Cmp->getOperand(0));
+ if (AndI->getParent() == Cmp->getParent() &&
+ TLI.isMaskAndCmp0FoldingBeneficial(*AndI))
+ continue;
}
// Figure out which BB this cmp is used in.
More information about the llvm-commits
mailing list