[llvm] [CGP] Allow replaceMathCmpWithIntrinsic() in some multi-BB situations (PR #227681)
Hans Wennborg via llvm-commits
llvm-commits at lists.llvm.org
Wed Sep 30 05:12:42 PDT 2026
https://github.com/zmodem updated https://github.com/llvm/llvm-project/pull/227681
>From c427d46f891b6e06fd008efa0935d9cdd1879124 Mon Sep 17 00:00:00 2001
From: Hans Wennborg <hans at chromium.org>
Date: Tue, 29 Sep 2026 16:03:41 +0200
Subject: [PATCH 1/6] allow replaceMathCmpWithIntrinsic when CMP is a unique
predecessor to BO
---
llvm/lib/CodeGen/CodeGenPrepare.cpp | 4 +++-
1 file changed, 3 insertions(+), 1 deletion(-)
diff --git a/llvm/lib/CodeGen/CodeGenPrepare.cpp b/llvm/lib/CodeGen/CodeGenPrepare.cpp
index ff40f210c5790..45d622e9de767 100644
--- a/llvm/lib/CodeGen/CodeGenPrepare.cpp
+++ b/llvm/lib/CodeGen/CodeGenPrepare.cpp
@@ -1656,7 +1656,9 @@ bool CodeGenPrepare::replaceMathCmpWithIntrinsic(BinaryOperator *BO,
// Otherwise, special case the single use in the phi recurrence.
return BO->hasOneUse() && DT.dominates(Cmp->getParent(), L->getLoopLatch());
};
- if (BO->getParent() != Cmp->getParent() && !IsReplacableIVIncrement(BO)) {
+ if (BO->getParent() != Cmp->getParent() &&
+ BO->getParent()->getUniquePredecessor() != Cmp->getParent() &&
+ !IsReplacableIVIncrement(BO)) {
// We used to use a dominator tree here to allow multi-block optimization.
// But that was problematic because:
// 1. It could cause a perf regression by hoisting the math op into the
>From 0730a27488b5cd0908167b45e23fc4713415e453 Mon Sep 17 00:00:00 2001
From: Hans Wennborg <hans at chromium.org>
Date: Wed, 30 Sep 2026 11:58:55 +0200
Subject: [PATCH 2/6] do the quick dominance and use check
---
llvm/lib/CodeGen/CodeGenPrepare.cpp | 24 ++++++++++++++++++++++--
1 file changed, 22 insertions(+), 2 deletions(-)
diff --git a/llvm/lib/CodeGen/CodeGenPrepare.cpp b/llvm/lib/CodeGen/CodeGenPrepare.cpp
index 45d622e9de767..6110f32e2b7e4 100644
--- a/llvm/lib/CodeGen/CodeGenPrepare.cpp
+++ b/llvm/lib/CodeGen/CodeGenPrepare.cpp
@@ -1656,8 +1656,25 @@ bool CodeGenPrepare::replaceMathCmpWithIntrinsic(BinaryOperator *BO,
// Otherwise, special case the single use in the phi recurrence.
return BO->hasOneUse() && DT.dominates(Cmp->getParent(), L->getLoopLatch());
};
- if (BO->getParent() != Cmp->getParent() &&
- BO->getParent()->getUniquePredecessor() != Cmp->getParent() &&
+
+ auto QuickDomAndPressureCheck = [=]{
+ // A cheap dominance check.
+ if (BO->getParent()->getUniquePredecessor() != Cmp->getParent())
+ return false;
+
+ // If the only uses of Arg0/Arg1 are the BO and Cmp, replacing them
+ // with an intrinsic means we're replacing one live value (the non-constnat
+ // argument) with another (the intrinsic), thus keeping register preassure
+ // unchanged.
+ if (!isa<ConstantInt>(Arg0) && Arg0->hasNUsesOrMore(3))
+ return false;
+ if (!isa<ConstantInt>(Arg1) && Arg1->hasNUsesOrMore(3))
+ return false;
+
+ return true;
+ };
+
+ if (BO->getParent() != Cmp->getParent() && !QuickDomAndPressureCheck() &&
!IsReplacableIVIncrement(BO)) {
// We used to use a dominator tree here to allow multi-block optimization.
// But that was problematic because:
@@ -1677,6 +1694,9 @@ bool CodeGenPrepare::replaceMathCmpWithIntrinsic(BinaryOperator *BO,
// - Upon computing Cmp, we effectively compute something equivalent to the
// IV increment (despite it loops differently in the IR). So moving it up
// to the cmp point does not really increase register pressure.
+ //
+ // Also, if Cmp's block trivially dominates BO's and the non-const
+ // argument doesn't have any other uses, we handle that too.
return false;
}
>From 1d49d91691f14e8e5355e72af690480deca6096d Mon Sep 17 00:00:00 2001
From: Hans Wennborg <hans at chromium.org>
Date: Wed, 30 Sep 2026 13:18:12 +0200
Subject: [PATCH 3/6] update llvm/test/CodeGen/X86/pr44412.ll
---
llvm/test/CodeGen/X86/pr44412.ll | 26 ++++++++++----------------
1 file changed, 10 insertions(+), 16 deletions(-)
diff --git a/llvm/test/CodeGen/X86/pr44412.ll b/llvm/test/CodeGen/X86/pr44412.ll
index 546dbcc156129..cc46e290660a7 100644
--- a/llvm/test/CodeGen/X86/pr44412.ll
+++ b/llvm/test/CodeGen/X86/pr44412.ll
@@ -4,21 +4,18 @@
define void @bar(i32 %0, i32 %1) nounwind {
; CHECK-LABEL: bar:
; CHECK: # %bb.0:
-; CHECK-NEXT: testl %edi, %edi
-; CHECK-NEXT: je .LBB0_4
-; CHECK-NEXT: # %bb.1: # %.preheader
; CHECK-NEXT: pushq %rbx
; CHECK-NEXT: movl %edi, %ebx
-; CHECK-NEXT: decl %ebx
+; CHECK-NEXT: subl $1, %ebx
+; CHECK-NEXT: jb .LBB0_2
; CHECK-NEXT: .p2align 4
-; CHECK-NEXT: .LBB0_2: # =>This Inner Loop Header: Depth=1
+; CHECK-NEXT: .LBB0_1: # =>This Inner Loop Header: Depth=1
; CHECK-NEXT: movl %ebx, %edi
; CHECK-NEXT: callq foo at PLT
; CHECK-NEXT: addl $-1, %ebx
-; CHECK-NEXT: jb .LBB0_2
-; CHECK-NEXT: # %bb.3:
+; CHECK-NEXT: jb .LBB0_1
+; CHECK-NEXT: .LBB0_2:
; CHECK-NEXT: popq %rbx
-; CHECK-NEXT: .LBB0_4:
; CHECK-NEXT: retq
%3 = icmp eq i32 %0, 0
br i1 %3, label %8, label %4
@@ -37,21 +34,18 @@ define void @bar(i32 %0, i32 %1) nounwind {
define void @baz(i32 %0, i32 %1) nounwind {
; CHECK-LABEL: baz:
; CHECK: # %bb.0:
-; CHECK-NEXT: testl %edi, %edi
-; CHECK-NEXT: je .LBB1_4
-; CHECK-NEXT: # %bb.1: # %.preheader
; CHECK-NEXT: pushq %rbx
; CHECK-NEXT: movl %edi, %ebx
-; CHECK-NEXT: decl %ebx
+; CHECK-NEXT: subl $1, %ebx
+; CHECK-NEXT: jb .LBB1_2
; CHECK-NEXT: .p2align 4
-; CHECK-NEXT: .LBB1_2: # =>This Inner Loop Header: Depth=1
+; CHECK-NEXT: .LBB1_1: # =>This Inner Loop Header: Depth=1
; CHECK-NEXT: movl %ebx, %edi
; CHECK-NEXT: callq foo at PLT
; CHECK-NEXT: addl $-1, %ebx
-; CHECK-NEXT: jae .LBB1_2
-; CHECK-NEXT: # %bb.3:
+; CHECK-NEXT: jae .LBB1_1
+; CHECK-NEXT: .LBB1_2:
; CHECK-NEXT: popq %rbx
-; CHECK-NEXT: .LBB1_4:
; CHECK-NEXT: retq
%3 = icmp eq i32 %0, 0
br i1 %3, label %8, label %4
>From 9b4a35f4e50afa9f7c970293172ef46e8d6b434e Mon Sep 17 00:00:00 2001
From: Hans Wennborg <hans at chromium.org>
Date: Wed, 30 Sep 2026 13:20:18 +0200
Subject: [PATCH 4/6] update llvm/test/CodeGen/X86/fold-loop-of-urem.ll
---
llvm/test/CodeGen/X86/fold-loop-of-urem.ll | 12 +++++-------
1 file changed, 5 insertions(+), 7 deletions(-)
diff --git a/llvm/test/CodeGen/X86/fold-loop-of-urem.ll b/llvm/test/CodeGen/X86/fold-loop-of-urem.ll
index f3b9af4eb08e8..02bab1548e137 100644
--- a/llvm/test/CodeGen/X86/fold-loop-of-urem.ll
+++ b/llvm/test/CodeGen/X86/fold-loop-of-urem.ll
@@ -890,9 +890,6 @@ for.body:
define void @simple_urem_multi_latch_non_canonical(i32 %N, i32 %rem_amt) nounwind {
; CHECK-LABEL: simple_urem_multi_latch_non_canonical:
; CHECK: # %bb.0: # %entry
-; CHECK-NEXT: testl %edi, %edi
-; CHECK-NEXT: je .LBB15_6
-; CHECK-NEXT: # %bb.1: # %for.body.preheader
; CHECK-NEXT: pushq %rbp
; CHECK-NEXT: pushq %r15
; CHECK-NEXT: pushq %r14
@@ -900,9 +897,11 @@ define void @simple_urem_multi_latch_non_canonical(i32 %N, i32 %rem_amt) nounwin
; CHECK-NEXT: pushq %r12
; CHECK-NEXT: pushq %rbx
; CHECK-NEXT: pushq %rax
-; CHECK-NEXT: movl %esi, %ebx
; CHECK-NEXT: movl %edi, %ebp
-; CHECK-NEXT: decl %ebp
+; CHECK-NEXT: subl $1, %ebp
+; CHECK-NEXT: jb .LBB15_5
+; CHECK-NEXT: # %bb.1: # %for.body.preheader
+; CHECK-NEXT: movl %esi, %ebx
; CHECK-NEXT: xorl %r12d, %r12d
; CHECK-NEXT: xorl %r14d, %r14d
; CHECK-NEXT: xorl %r13d, %r13d
@@ -928,7 +927,7 @@ define void @simple_urem_multi_latch_non_canonical(i32 %N, i32 %rem_amt) nounwin
; CHECK-NEXT: callq do_stuff1 at PLT
; CHECK-NEXT: cmpl %r13d, %ebp
; CHECK-NEXT: jne .LBB15_3
-; CHECK-NEXT: # %bb.5:
+; CHECK-NEXT: .LBB15_5: # %for.cond.cleanup
; CHECK-NEXT: addq $8, %rsp
; CHECK-NEXT: popq %rbx
; CHECK-NEXT: popq %r12
@@ -936,7 +935,6 @@ define void @simple_urem_multi_latch_non_canonical(i32 %N, i32 %rem_amt) nounwin
; CHECK-NEXT: popq %r14
; CHECK-NEXT: popq %r15
; CHECK-NEXT: popq %rbp
-; CHECK-NEXT: .LBB15_6: # %for.cond.cleanup
; CHECK-NEXT: retq
entry:
%cmp3.not = icmp eq i32 %N, 0
>From 7cd4f9939858c570e4c09f8f987a85e6f60b5528 Mon Sep 17 00:00:00 2001
From: Hans Wennborg <hans at chromium.org>
Date: Wed, 30 Sep 2026 13:37:15 +0200
Subject: [PATCH 5/6] new tests
---
llvm/test/CodeGen/X86/cgp-usubo.ll | 20 +++++++++++++++
.../CodeGenPrepare/X86/overflow-intrinsics.ll | 25 +++++++++++++++++++
2 files changed, 45 insertions(+)
diff --git a/llvm/test/CodeGen/X86/cgp-usubo.ll b/llvm/test/CodeGen/X86/cgp-usubo.ll
index 57e2a2b22bc9b..82d019ce63b41 100644
--- a/llvm/test/CodeGen/X86/cgp-usubo.ll
+++ b/llvm/test/CodeGen/X86/cgp-usubo.ll
@@ -258,3 +258,23 @@ define i32 @PR42571(i32 %x, i32 %y) {
%cond = select i1 %tobool, i32 %y, i32 %and
ret i32 %cond
}
+
+; Some simple multi-BB situations can still be handled.
+
+declare dso_local fastcc void @use(i32)
+define void @Issue156015(i32 %a) {
+; CHECK-LABEL: Issue156015:
+; CHECK: # %bb.0:
+; CHECK-NEXT: subl $10, %edi
+; CHECK-NEXT: jae use # TAILCALL
+; CHECK-NEXT: # %bb.1: # %then
+; CHECK-NEXT: retq
+ %c = icmp ult i32 %a, 10
+ br i1 %c, label %then, label %else
+then:
+ ret void
+else:
+ %l = sub i32 %a, 10
+ tail call fastcc void @use(i32 %l)
+ ret void
+}
diff --git a/llvm/test/Transforms/CodeGenPrepare/X86/overflow-intrinsics.ll b/llvm/test/Transforms/CodeGenPrepare/X86/overflow-intrinsics.ll
index 653f346356488..5679a2e145027 100644
--- a/llvm/test/Transforms/CodeGenPrepare/X86/overflow-intrinsics.ll
+++ b/llvm/test/Transforms/CodeGenPrepare/X86/overflow-intrinsics.ll
@@ -636,6 +636,31 @@ exit:
ret void
}
+; Some simple multi-BB situations can still be handled.
+
+declare dso_local fastcc void @use32(i32)
+define void @Issue156015(i32 %a) {
+; CHECK-LABEL: @Issue156015(
+; CHECK-NEXT: [[TMP1:%.*]] = call { i32, i1 } @llvm.usub.with.overflow.i32(i32 [[A:%.*]], i32 10)
+; CHECK-NEXT: [[MATH:%.*]] = extractvalue { i32, i1 } [[TMP1]], 0
+; CHECK-NEXT: [[OV:%.*]] = extractvalue { i32, i1 } [[TMP1]], 1
+; CHECK-NEXT: br i1 [[OV]], label [[THEN:%.*]], label [[ELSE:%.*]]
+; CHECK: then:
+; CHECK-NEXT: ret void
+; CHECK: else:
+; CHECK-NEXT: tail call fastcc void @use32(i32 [[MATH]])
+; CHECK-NEXT: ret void
+;
+ %c = icmp ult i32 %a, 10
+ br i1 %c, label %then, label %else
+then:
+ ret void
+else:
+ %l = sub i32 %a, 10
+ tail call fastcc void @use32(i32 %l)
+ ret void
+}
+
; Check that every instruction inserted by -passes='require<profile-summary>,function(codegenprepare)' has a debug location.
; DEBUG: CheckModuleDebugify: PASS
>From 3d28239e0114f80187f6ff202f2e0d7e3c60e6dd Mon Sep 17 00:00:00 2001
From: Hans Wennborg <hans at chromium.org>
Date: Wed, 30 Sep 2026 14:12:25 +0200
Subject: [PATCH 6/6] format
---
llvm/lib/CodeGen/CodeGenPrepare.cpp | 2 +-
1 file changed, 1 insertion(+), 1 deletion(-)
diff --git a/llvm/lib/CodeGen/CodeGenPrepare.cpp b/llvm/lib/CodeGen/CodeGenPrepare.cpp
index 6110f32e2b7e4..96380d61a3276 100644
--- a/llvm/lib/CodeGen/CodeGenPrepare.cpp
+++ b/llvm/lib/CodeGen/CodeGenPrepare.cpp
@@ -1657,7 +1657,7 @@ bool CodeGenPrepare::replaceMathCmpWithIntrinsic(BinaryOperator *BO,
return BO->hasOneUse() && DT.dominates(Cmp->getParent(), L->getLoopLatch());
};
- auto QuickDomAndPressureCheck = [=]{
+ auto QuickDomAndPressureCheck = [=] {
// A cheap dominance check.
if (BO->getParent()->getUniquePredecessor() != Cmp->getParent())
return false;
More information about the llvm-commits
mailing list