[llvm] [CGP] Fold cmp+br+sub to usub+br on the overflow flag (PR #226496)
Hans Wennborg via llvm-commits
llvm-commits at lists.llvm.org
Fri Sep 25 06:47:04 PDT 2026
https://github.com/zmodem created https://github.com/llvm/llvm-project/pull/226496
None
>From 501686eaeb0dffe34303743d6fec7093078ec715 Mon Sep 17 00:00:00 2001
From: Hans Wennborg <hans at chromium.org>
Date: Thu, 24 Sep 2026 16:57:14 +0200
Subject: [PATCH] [CGP] Fold cmp+br+sub to usub+br on the overflow flag
---
llvm/include/llvm/CodeGen/TargetLowering.h | 5 ++
llvm/lib/CodeGen/CodeGenPrepare.cpp | 39 ++++++++++++
llvm/lib/Target/X86/X86ISelLowering.cpp | 2 +
llvm/lib/Target/X86/X86ISelLowering.h | 2 +
.../CodeGen/X86/branch-on-usub-overflow.ll | 63 +++++++++++++++++++
.../X86/branch-on-usub-overflow.ll | 54 ++++++++++++++++
6 files changed, 165 insertions(+)
create mode 100644 llvm/test/CodeGen/X86/branch-on-usub-overflow.ll
create mode 100644 llvm/test/Transforms/CodeGenPrepare/X86/branch-on-usub-overflow.ll
diff --git a/llvm/include/llvm/CodeGen/TargetLowering.h b/llvm/include/llvm/CodeGen/TargetLowering.h
index 283b0f811dc2f..8a16c01507f11 100644
--- a/llvm/include/llvm/CodeGen/TargetLowering.h
+++ b/llvm/include/llvm/CodeGen/TargetLowering.h
@@ -770,6 +770,11 @@ class LLVM_ABI TargetLoweringBase {
/// gen prepare.
virtual bool preferZeroCompareBranch() const { return false; }
+ /// Return true if the heuristic to prefer branching on
+ /// @llvm.usub.with.overflow's overflow flag should be used in
+ /// codegenprepare. Requires preferZeroCompareBranch() to be enabled.
+ virtual bool preferUSubOverflowBranch() const { return false; }
+
/// Return true if it is cheaper to split the store of a merged int val
/// from a pair of smaller values into multiple stores.
virtual bool isMultiStoresCheaperThanBitsMerge(EVT LTy, EVT HTy) const {
diff --git a/llvm/lib/CodeGen/CodeGenPrepare.cpp b/llvm/lib/CodeGen/CodeGenPrepare.cpp
index ff40f210c5790..e246ad3db1b62 100644
--- a/llvm/lib/CodeGen/CodeGenPrepare.cpp
+++ b/llvm/lib/CodeGen/CodeGenPrepare.cpp
@@ -8878,6 +8878,10 @@ static bool tryUnmergingGEPsAcrossIndirectBr(GetElementPtrInst *GEPI,
static bool optimizeBranch(CondBrInst *Branch, const TargetLowering &TLI,
SmallPtrSet<BasicBlock *, 32> &FreshBBs,
bool IsHugeFunc) {
+ if (TLI.preferUSubOverflowBranch())
+ assert(TLI.preferZeroCompareBranch() &&
+ "preferUSubOverflowBranch requires preferZeroCompareBranch");
+
// Try and convert
// %c = icmp ult %x, 8
// br %c, bla, blb
@@ -8925,6 +8929,7 @@ static bool optimizeBranch(CondBrInst *Branch, const TargetLowering &TLI,
replaceAllUsesWith(Cmp, NewCmp, FreshBBs, IsHugeFunc);
return true;
}
+
if (Cmp->isEquality() &&
(match(UI, m_Add(m_Specific(X), m_SpecificInt(-CmpC))) ||
match(UI, m_Sub(m_Specific(X), m_SpecificInt(CmpC))) ||
@@ -8940,6 +8945,40 @@ static bool optimizeBranch(CondBrInst *Branch, const TargetLowering &TLI,
replaceAllUsesWith(Cmp, NewCmp, FreshBBs, IsHugeFunc);
return true;
}
+
+ if (TLI.preferUSubOverflowBranch() &&
+ Cmp->getPredicate() == ICmpInst::ICMP_ULT &&
+ (match(UI, m_Add(m_Specific(X), m_SpecificInt(-CmpC))) ||
+ match(UI, m_Sub(m_Specific(X), m_SpecificInt(CmpC))))) {
+ // Convert
+ //
+ // %cmp = icmp ult i32 %x, 3
+ // br i1 %cmp, label %bb1, label %bb2
+ // bb2:
+ // %sub = add i32 %x, -3
+ //
+ // into
+ //
+ // %usub = call { i32, i1 } @llvm.usub.with.overflow.i32(i32 %x, i32 -3)
+ // %sub = extractvalue { i32, i1 } %usub, 0
+ // %oflow = extractvalue { i32, i1 } %usub, 1
+ // br i1 %oflow, label %bb1, label %bb2
+ //
+ // for targets where lowering the intrinsic and branching on its overflow
+ // flag is efficient.
+ Function *USubWithOverflow = Intrinsic::getOrInsertDeclaration(
+ Branch->getModule(), Intrinsic::usub_with_overflow, {UI->getType()});
+ IRBuilder<> Builder(Branch);
+ Value *Call = Builder.CreateCall(
+ USubWithOverflow, {UI->getOperand(0), UI->getOperand(1)}, "usub");
+ Value *Sub = Builder.CreateExtractValue(Call, 0, "sub");
+ Value *Overflow = Builder.CreateExtractValue(Call, 1, "oflow");
+ LLVM_DEBUG(dbgs() << "Converting " << *Cmp << "\n");
+ LLVM_DEBUG(dbgs() << " to llvm.usub.with.overflow: " << *Call << "\n");
+ replaceAllUsesWith(Cmp, Overflow, FreshBBs, IsHugeFunc);
+ replaceAllUsesWith(UI, Sub, FreshBBs, IsHugeFunc);
+ return true;
+ }
}
return false;
}
diff --git a/llvm/lib/Target/X86/X86ISelLowering.cpp b/llvm/lib/Target/X86/X86ISelLowering.cpp
index e2f3b5f3cd3d5..f77bac51ba756 100644
--- a/llvm/lib/Target/X86/X86ISelLowering.cpp
+++ b/llvm/lib/Target/X86/X86ISelLowering.cpp
@@ -3804,6 +3804,8 @@ bool X86TargetLowering::isCtlzFast() const {
bool X86TargetLowering::preferZeroCompareBranch() const { return true; }
+bool X86TargetLowering::preferUSubOverflowBranch() const { return true; }
+
bool X86TargetLowering::isMaskAndCmp0FoldingBeneficial(
const Instruction &AndI) const {
return true;
diff --git a/llvm/lib/Target/X86/X86ISelLowering.h b/llvm/lib/Target/X86/X86ISelLowering.h
index 34bb76982eecb..119bbecb0293d 100644
--- a/llvm/lib/Target/X86/X86ISelLowering.h
+++ b/llvm/lib/Target/X86/X86ISelLowering.h
@@ -240,6 +240,8 @@ namespace llvm {
bool preferZeroCompareBranch() const override;
+ bool preferUSubOverflowBranch() const override;
+
bool isMultiStoresCheaperThanBitsMerge(EVT LTy, EVT HTy) const override {
// If the pair to store is a mixture of float and int values, we will
// save two bitwise instructions and one float-to-int instruction and
diff --git a/llvm/test/CodeGen/X86/branch-on-usub-overflow.ll b/llvm/test/CodeGen/X86/branch-on-usub-overflow.ll
new file mode 100644
index 0000000000000..8ed6d2a0ca81a
--- /dev/null
+++ b/llvm/test/CodeGen/X86/branch-on-usub-overflow.ll
@@ -0,0 +1,63 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
+; RUN: llc -mtriple x86_64-unknown-linux-gnu -o - %s | FileCheck %s --check-prefix=CHECK64
+; RUN: llc -mtriple i686-unknown-linux-gnu -o - %s | FileCheck %s --check-prefix=CHECK32
+
+declare dso_local fastcc void @use(i32)
+
+define void @ult_sub10(i32 %a) nounwind {
+; CHECK64-LABEL: ult_sub10:
+; CHECK64: # %bb.0: # %entry
+; CHECK64-NEXT: subl $10, %edi
+; CHECK64-NEXT: jae use # TAILCALL
+; CHECK64-NEXT: # %bb.1: # %then
+; CHECK64-NEXT: retq
+;
+; CHECK32-LABEL: ult_sub10:
+; CHECK32: # %bb.0: # %entry
+; CHECK32-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; CHECK32-NEXT: subl $10, %ecx
+; CHECK32-NEXT: jae use # TAILCALL
+; CHECK32-NEXT: # %bb.1: # %then
+; CHECK32-NEXT: retl
+entry:
+ %c = icmp ult i32 %a, 10
+ br i1 %c, label %then, label %else
+then:
+ ret void
+else:
+ %l = sub i32 %a, 10
+ tail call fastcc void @use(i32 %l)
+ ret void
+}
+
+define void @ule_sub10(i32 %a) nounwind {
+; CHECK64-LABEL: ule_sub10:
+; CHECK64: # %bb.0: # %entry
+; CHECK64-NEXT: cmpl $10, %edi
+; CHECK64-NEXT: ja .LBB1_2
+; CHECK64-NEXT: # %bb.1: # %then
+; CHECK64-NEXT: retq
+; CHECK64-NEXT: .LBB1_2: # %else
+; CHECK64-NEXT: addl $-10, %edi
+; CHECK64-NEXT: jmp use # TAILCALL
+;
+; CHECK32-LABEL: ule_sub10:
+; CHECK32: # %bb.0: # %entry
+; CHECK32-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; CHECK32-NEXT: cmpl $10, %ecx
+; CHECK32-NEXT: ja .LBB1_2
+; CHECK32-NEXT: # %bb.1: # %then
+; CHECK32-NEXT: retl
+; CHECK32-NEXT: .LBB1_2: # %else
+; CHECK32-NEXT: addl $-10, %ecx
+; CHECK32-NEXT: jmp use # TAILCALL
+entry:
+ %c = icmp ule i32 %a, 10
+ br i1 %c, label %then, label %else
+then:
+ ret void
+else:
+ %l = sub i32 %a, 10
+ tail call fastcc void @use(i32 %l)
+ ret void
+}
diff --git a/llvm/test/Transforms/CodeGenPrepare/X86/branch-on-usub-overflow.ll b/llvm/test/Transforms/CodeGenPrepare/X86/branch-on-usub-overflow.ll
new file mode 100644
index 0000000000000..b77b6eb231ed3
--- /dev/null
+++ b/llvm/test/Transforms/CodeGenPrepare/X86/branch-on-usub-overflow.ll
@@ -0,0 +1,54 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py
+; RUN: opt -S -passes='require<profile-summary>,function(codegenprepare)' < %s | FileCheck %s
+
+target triple = "x86_64-unknown-linux-gnu"
+
+declare dso_local fastcc void @use(i32)
+
+define void @ult_sub10(i32 %a) nounwind {
+; CHECK-LABEL: @ult_sub10(
+; CHECK-NEXT: entry:
+; CHECK-NEXT: [[USUB:%.*]] = call { i32, i1 } @llvm.usub.with.overflow.i32(i32 [[A:%.*]], i32 10)
+; CHECK-NEXT: [[SUB:%.*]] = extractvalue { i32, i1 } [[USUB]], 0
+; CHECK-NEXT: [[OFLOW:%.*]] = extractvalue { i32, i1 } [[USUB]], 1
+; CHECK-NEXT: br i1 [[OFLOW]], label [[THEN:%.*]], label [[ELSE:%.*]]
+; CHECK: then:
+; CHECK-NEXT: ret void
+; CHECK: else:
+; CHECK-NEXT: [[L:%.*]] = sub i32 [[A]], 10
+; CHECK-NEXT: tail call fastcc void @use(i32 [[SUB]])
+; CHECK-NEXT: ret void
+;
+entry:
+ %c = icmp ult i32 %a, 10
+ br i1 %c, label %then, label %else
+then:
+ ret void
+else:
+ %l = sub i32 %a, 10
+ tail call fastcc void @use(i32 %l)
+ ret void
+}
+
+define void @ule_sub10(i32 %a) nounwind {
+; CHECK-LABEL: @ule_sub10(
+; CHECK-NEXT: entry:
+; CHECK-NEXT: [[C:%.*]] = icmp ule i32 [[A:%.*]], 10
+; CHECK-NEXT: br i1 [[C]], label [[THEN:%.*]], label [[ELSE:%.*]]
+; CHECK: then:
+; CHECK-NEXT: ret void
+; CHECK: else:
+; CHECK-NEXT: [[L:%.*]] = sub i32 [[A]], 10
+; CHECK-NEXT: tail call fastcc void @use(i32 [[L]])
+; CHECK-NEXT: ret void
+;
+entry:
+ %c = icmp ule i32 %a, 10
+ br i1 %c, label %then, label %else
+then:
+ ret void
+else:
+ %l = sub i32 %a, 10
+ tail call fastcc void @use(i32 %l)
+ ret void
+}
More information about the llvm-commits
mailing list