[llvm] [CGP] Allow replaceMathCmpWithIntrinsic() in some multi-BB situations (PR #227681)
Hans Wennborg via llvm-commits
llvm-commits at lists.llvm.org
Wed Sep 30 05:06:17 PDT 2026
https://github.com/zmodem created https://github.com/llvm/llvm-project/pull/227681
CGP can replace some combinations of math and comparisons, such as:
```
%c = icmp ult i32 %a, 10
br i1 %c, label %then, label %else
..
%l = sub i32 %a, 10
```
with an intrinsic that sets the overflow flag:
```
%0 = call { i32, i1 } @llvm.usub.with.overflow.i32(i32 %a, i32 10)
%l = extractvalue { i32, i1 } %0, 0
%ov = extractvalue { i32, i1 } %1, 1
br i1 %ov, label %then, label %else
```
which the backend will lover efficiently.
However, 5ab41a7a0552690e9f7ca657bee1d0507baaddfb disabled the transform across multiple blocks because the dominance check can be expensive and the transform could increase register pressure (in the example, `%l` is computed earlier, so is live longer).
This PR re-enables the transform across multiple blocks in simple situations: dominance is trivial if the binary op block is uniquely preceded by the cmp block, and if there are no other uses of the binary op's operand we're replacing one live value with another (in the example, `%a` is no longer live after `%l` is computed) so there is no increased register pressure.
Fixes #156015 which is a subset of #58636.
>From c427d46f891b6e06fd008efa0935d9cdd1879124 Mon Sep 17 00:00:00 2001
From: Hans Wennborg <hans at chromium.org>
Date: Tue, 29 Sep 2026 16:03:41 +0200
Subject: [PATCH 1/5] allow replaceMathCmpWithIntrinsic when CMP is a unique
predecessor to BO
---
llvm/lib/CodeGen/CodeGenPrepare.cpp | 4 +++-
1 file changed, 3 insertions(+), 1 deletion(-)
diff --git a/llvm/lib/CodeGen/CodeGenPrepare.cpp b/llvm/lib/CodeGen/CodeGenPrepare.cpp
index ff40f210c57902..45d622e9de7678 100644
--- a/llvm/lib/CodeGen/CodeGenPrepare.cpp
+++ b/llvm/lib/CodeGen/CodeGenPrepare.cpp
@@ -1656,7 +1656,9 @@ bool CodeGenPrepare::replaceMathCmpWithIntrinsic(BinaryOperator *BO,
// Otherwise, special case the single use in the phi recurrence.
return BO->hasOneUse() && DT.dominates(Cmp->getParent(), L->getLoopLatch());
};
- if (BO->getParent() != Cmp->getParent() && !IsReplacableIVIncrement(BO)) {
+ if (BO->getParent() != Cmp->getParent() &&
+ BO->getParent()->getUniquePredecessor() != Cmp->getParent() &&
+ !IsReplacableIVIncrement(BO)) {
// We used to use a dominator tree here to allow multi-block optimization.
// But that was problematic because:
// 1. It could cause a perf regression by hoisting the math op into the
>From 0730a27488b5cd0908167b45e23fc4713415e453 Mon Sep 17 00:00:00 2001
From: Hans Wennborg <hans at chromium.org>
Date: Wed, 30 Sep 2026 11:58:55 +0200
Subject: [PATCH 2/5] do the quick dominance and use check
---
llvm/lib/CodeGen/CodeGenPrepare.cpp | 24 ++++++++++++++++++++++--
1 file changed, 22 insertions(+), 2 deletions(-)
diff --git a/llvm/lib/CodeGen/CodeGenPrepare.cpp b/llvm/lib/CodeGen/CodeGenPrepare.cpp
index 45d622e9de7678..6110f32e2b7e4f 100644
--- a/llvm/lib/CodeGen/CodeGenPrepare.cpp
+++ b/llvm/lib/CodeGen/CodeGenPrepare.cpp
@@ -1656,8 +1656,25 @@ bool CodeGenPrepare::replaceMathCmpWithIntrinsic(BinaryOperator *BO,
// Otherwise, special case the single use in the phi recurrence.
return BO->hasOneUse() && DT.dominates(Cmp->getParent(), L->getLoopLatch());
};
- if (BO->getParent() != Cmp->getParent() &&
- BO->getParent()->getUniquePredecessor() != Cmp->getParent() &&
+
+ auto QuickDomAndPressureCheck = [=]{
+ // A cheap dominance check.
+ if (BO->getParent()->getUniquePredecessor() != Cmp->getParent())
+ return false;
+
+ // If the only uses of Arg0/Arg1 are the BO and Cmp, replacing them
+ // with an intrinsic means we're replacing one live value (the non-constnat
+ // argument) with another (the intrinsic), thus keeping register preassure
+ // unchanged.
+ if (!isa<ConstantInt>(Arg0) && Arg0->hasNUsesOrMore(3))
+ return false;
+ if (!isa<ConstantInt>(Arg1) && Arg1->hasNUsesOrMore(3))
+ return false;
+
+ return true;
+ };
+
+ if (BO->getParent() != Cmp->getParent() && !QuickDomAndPressureCheck() &&
!IsReplacableIVIncrement(BO)) {
// We used to use a dominator tree here to allow multi-block optimization.
// But that was problematic because:
@@ -1677,6 +1694,9 @@ bool CodeGenPrepare::replaceMathCmpWithIntrinsic(BinaryOperator *BO,
// - Upon computing Cmp, we effectively compute something equivalent to the
// IV increment (despite it loops differently in the IR). So moving it up
// to the cmp point does not really increase register pressure.
+ //
+ // Also, if Cmp's block trivially dominates BO's and the non-const
+ // argument doesn't have any other uses, we handle that too.
return false;
}
>From 1d49d91691f14e8e5355e72af690480deca6096d Mon Sep 17 00:00:00 2001
From: Hans Wennborg <hans at chromium.org>
Date: Wed, 30 Sep 2026 13:18:12 +0200
Subject: [PATCH 3/5] update llvm/test/CodeGen/X86/pr44412.ll
---
llvm/test/CodeGen/X86/pr44412.ll | 26 ++++++++++----------------
1 file changed, 10 insertions(+), 16 deletions(-)
diff --git a/llvm/test/CodeGen/X86/pr44412.ll b/llvm/test/CodeGen/X86/pr44412.ll
index 546dbcc1561299..cc46e290660a7c 100644
--- a/llvm/test/CodeGen/X86/pr44412.ll
+++ b/llvm/test/CodeGen/X86/pr44412.ll
@@ -4,21 +4,18 @@
define void @bar(i32 %0, i32 %1) nounwind {
; CHECK-LABEL: bar:
; CHECK: # %bb.0:
-; CHECK-NEXT: testl %edi, %edi
-; CHECK-NEXT: je .LBB0_4
-; CHECK-NEXT: # %bb.1: # %.preheader
; CHECK-NEXT: pushq %rbx
; CHECK-NEXT: movl %edi, %ebx
-; CHECK-NEXT: decl %ebx
+; CHECK-NEXT: subl $1, %ebx
+; CHECK-NEXT: jb .LBB0_2
; CHECK-NEXT: .p2align 4
-; CHECK-NEXT: .LBB0_2: # =>This Inner Loop Header: Depth=1
+; CHECK-NEXT: .LBB0_1: # =>This Inner Loop Header: Depth=1
; CHECK-NEXT: movl %ebx, %edi
; CHECK-NEXT: callq foo at PLT
; CHECK-NEXT: addl $-1, %ebx
-; CHECK-NEXT: jb .LBB0_2
-; CHECK-NEXT: # %bb.3:
+; CHECK-NEXT: jb .LBB0_1
+; CHECK-NEXT: .LBB0_2:
; CHECK-NEXT: popq %rbx
-; CHECK-NEXT: .LBB0_4:
; CHECK-NEXT: retq
%3 = icmp eq i32 %0, 0
br i1 %3, label %8, label %4
@@ -37,21 +34,18 @@ define void @bar(i32 %0, i32 %1) nounwind {
define void @baz(i32 %0, i32 %1) nounwind {
; CHECK-LABEL: baz:
; CHECK: # %bb.0:
-; CHECK-NEXT: testl %edi, %edi
-; CHECK-NEXT: je .LBB1_4
-; CHECK-NEXT: # %bb.1: # %.preheader
; CHECK-NEXT: pushq %rbx
; CHECK-NEXT: movl %edi, %ebx
-; CHECK-NEXT: decl %ebx
+; CHECK-NEXT: subl $1, %ebx
+; CHECK-NEXT: jb .LBB1_2
; CHECK-NEXT: .p2align 4
-; CHECK-NEXT: .LBB1_2: # =>This Inner Loop Header: Depth=1
+; CHECK-NEXT: .LBB1_1: # =>This Inner Loop Header: Depth=1
; CHECK-NEXT: movl %ebx, %edi
; CHECK-NEXT: callq foo at PLT
; CHECK-NEXT: addl $-1, %ebx
-; CHECK-NEXT: jae .LBB1_2
-; CHECK-NEXT: # %bb.3:
+; CHECK-NEXT: jae .LBB1_1
+; CHECK-NEXT: .LBB1_2:
; CHECK-NEXT: popq %rbx
-; CHECK-NEXT: .LBB1_4:
; CHECK-NEXT: retq
%3 = icmp eq i32 %0, 0
br i1 %3, label %8, label %4
>From 9b4a35f4e50afa9f7c970293172ef46e8d6b434e Mon Sep 17 00:00:00 2001
From: Hans Wennborg <hans at chromium.org>
Date: Wed, 30 Sep 2026 13:20:18 +0200
Subject: [PATCH 4/5] update llvm/test/CodeGen/X86/fold-loop-of-urem.ll
---
llvm/test/CodeGen/X86/fold-loop-of-urem.ll | 12 +++++-------
1 file changed, 5 insertions(+), 7 deletions(-)
diff --git a/llvm/test/CodeGen/X86/fold-loop-of-urem.ll b/llvm/test/CodeGen/X86/fold-loop-of-urem.ll
index f3b9af4eb08e87..02bab1548e1371 100644
--- a/llvm/test/CodeGen/X86/fold-loop-of-urem.ll
+++ b/llvm/test/CodeGen/X86/fold-loop-of-urem.ll
@@ -890,9 +890,6 @@ for.body:
define void @simple_urem_multi_latch_non_canonical(i32 %N, i32 %rem_amt) nounwind {
; CHECK-LABEL: simple_urem_multi_latch_non_canonical:
; CHECK: # %bb.0: # %entry
-; CHECK-NEXT: testl %edi, %edi
-; CHECK-NEXT: je .LBB15_6
-; CHECK-NEXT: # %bb.1: # %for.body.preheader
; CHECK-NEXT: pushq %rbp
; CHECK-NEXT: pushq %r15
; CHECK-NEXT: pushq %r14
@@ -900,9 +897,11 @@ define void @simple_urem_multi_latch_non_canonical(i32 %N, i32 %rem_amt) nounwin
; CHECK-NEXT: pushq %r12
; CHECK-NEXT: pushq %rbx
; CHECK-NEXT: pushq %rax
-; CHECK-NEXT: movl %esi, %ebx
; CHECK-NEXT: movl %edi, %ebp
-; CHECK-NEXT: decl %ebp
+; CHECK-NEXT: subl $1, %ebp
+; CHECK-NEXT: jb .LBB15_5
+; CHECK-NEXT: # %bb.1: # %for.body.preheader
+; CHECK-NEXT: movl %esi, %ebx
; CHECK-NEXT: xorl %r12d, %r12d
; CHECK-NEXT: xorl %r14d, %r14d
; CHECK-NEXT: xorl %r13d, %r13d
@@ -928,7 +927,7 @@ define void @simple_urem_multi_latch_non_canonical(i32 %N, i32 %rem_amt) nounwin
; CHECK-NEXT: callq do_stuff1 at PLT
; CHECK-NEXT: cmpl %r13d, %ebp
; CHECK-NEXT: jne .LBB15_3
-; CHECK-NEXT: # %bb.5:
+; CHECK-NEXT: .LBB15_5: # %for.cond.cleanup
; CHECK-NEXT: addq $8, %rsp
; CHECK-NEXT: popq %rbx
; CHECK-NEXT: popq %r12
@@ -936,7 +935,6 @@ define void @simple_urem_multi_latch_non_canonical(i32 %N, i32 %rem_amt) nounwin
; CHECK-NEXT: popq %r14
; CHECK-NEXT: popq %r15
; CHECK-NEXT: popq %rbp
-; CHECK-NEXT: .LBB15_6: # %for.cond.cleanup
; CHECK-NEXT: retq
entry:
%cmp3.not = icmp eq i32 %N, 0
>From 7cd4f9939858c570e4c09f8f987a85e6f60b5528 Mon Sep 17 00:00:00 2001
From: Hans Wennborg <hans at chromium.org>
Date: Wed, 30 Sep 2026 13:37:15 +0200
Subject: [PATCH 5/5] new tests
---
llvm/test/CodeGen/X86/cgp-usubo.ll | 20 +++++++++++++++
.../CodeGenPrepare/X86/overflow-intrinsics.ll | 25 +++++++++++++++++++
2 files changed, 45 insertions(+)
diff --git a/llvm/test/CodeGen/X86/cgp-usubo.ll b/llvm/test/CodeGen/X86/cgp-usubo.ll
index 57e2a2b22bc9bc..82d019ce63b410 100644
--- a/llvm/test/CodeGen/X86/cgp-usubo.ll
+++ b/llvm/test/CodeGen/X86/cgp-usubo.ll
@@ -258,3 +258,23 @@ define i32 @PR42571(i32 %x, i32 %y) {
%cond = select i1 %tobool, i32 %y, i32 %and
ret i32 %cond
}
+
+; Some simple multi-BB situations can still be handled.
+
+declare dso_local fastcc void @use(i32)
+define void @Issue156015(i32 %a) {
+; CHECK-LABEL: Issue156015:
+; CHECK: # %bb.0:
+; CHECK-NEXT: subl $10, %edi
+; CHECK-NEXT: jae use # TAILCALL
+; CHECK-NEXT: # %bb.1: # %then
+; CHECK-NEXT: retq
+ %c = icmp ult i32 %a, 10
+ br i1 %c, label %then, label %else
+then:
+ ret void
+else:
+ %l = sub i32 %a, 10
+ tail call fastcc void @use(i32 %l)
+ ret void
+}
diff --git a/llvm/test/Transforms/CodeGenPrepare/X86/overflow-intrinsics.ll b/llvm/test/Transforms/CodeGenPrepare/X86/overflow-intrinsics.ll
index 653f3463564888..5679a2e1450274 100644
--- a/llvm/test/Transforms/CodeGenPrepare/X86/overflow-intrinsics.ll
+++ b/llvm/test/Transforms/CodeGenPrepare/X86/overflow-intrinsics.ll
@@ -636,6 +636,31 @@ exit:
ret void
}
+; Some simple multi-BB situations can still be handled.
+
+declare dso_local fastcc void @use32(i32)
+define void @Issue156015(i32 %a) {
+; CHECK-LABEL: @Issue156015(
+; CHECK-NEXT: [[TMP1:%.*]] = call { i32, i1 } @llvm.usub.with.overflow.i32(i32 [[A:%.*]], i32 10)
+; CHECK-NEXT: [[MATH:%.*]] = extractvalue { i32, i1 } [[TMP1]], 0
+; CHECK-NEXT: [[OV:%.*]] = extractvalue { i32, i1 } [[TMP1]], 1
+; CHECK-NEXT: br i1 [[OV]], label [[THEN:%.*]], label [[ELSE:%.*]]
+; CHECK: then:
+; CHECK-NEXT: ret void
+; CHECK: else:
+; CHECK-NEXT: tail call fastcc void @use32(i32 [[MATH]])
+; CHECK-NEXT: ret void
+;
+ %c = icmp ult i32 %a, 10
+ br i1 %c, label %then, label %else
+then:
+ ret void
+else:
+ %l = sub i32 %a, 10
+ tail call fastcc void @use32(i32 %l)
+ ret void
+}
+
; Check that every instruction inserted by -passes='require<profile-summary>,function(codegenprepare)' has a debug location.
; DEBUG: CheckModuleDebugify: PASS
More information about the llvm-commits
mailing list