[llvm] [CGP] Fold cmp+br+sub to usub+br on the overflow flag (PR #226496)

Hans Wennborg via llvm-commits llvm-commits at lists.llvm.org
Fri Sep 25 06:47:04 PDT 2026


https://github.com/zmodem created https://github.com/llvm/llvm-project/pull/226496

None

>From 501686eaeb0dffe34303743d6fec7093078ec715 Mon Sep 17 00:00:00 2001
From: Hans Wennborg <hans at chromium.org>
Date: Thu, 24 Sep 2026 16:57:14 +0200
Subject: [PATCH] [CGP] Fold cmp+br+sub to usub+br on the overflow flag

---
 llvm/include/llvm/CodeGen/TargetLowering.h    |  5 ++
 llvm/lib/CodeGen/CodeGenPrepare.cpp           | 39 ++++++++++++
 llvm/lib/Target/X86/X86ISelLowering.cpp       |  2 +
 llvm/lib/Target/X86/X86ISelLowering.h         |  2 +
 .../CodeGen/X86/branch-on-usub-overflow.ll    | 63 +++++++++++++++++++
 .../X86/branch-on-usub-overflow.ll            | 54 ++++++++++++++++
 6 files changed, 165 insertions(+)
 create mode 100644 llvm/test/CodeGen/X86/branch-on-usub-overflow.ll
 create mode 100644 llvm/test/Transforms/CodeGenPrepare/X86/branch-on-usub-overflow.ll

diff --git a/llvm/include/llvm/CodeGen/TargetLowering.h b/llvm/include/llvm/CodeGen/TargetLowering.h
index 283b0f811dc2f..8a16c01507f11 100644
--- a/llvm/include/llvm/CodeGen/TargetLowering.h
+++ b/llvm/include/llvm/CodeGen/TargetLowering.h
@@ -770,6 +770,11 @@ class LLVM_ABI TargetLoweringBase {
   /// gen prepare.
   virtual bool preferZeroCompareBranch() const { return false; }
 
+  /// Return true if the heuristic to prefer branching on
+  /// @llvm.usub.with.overflow's overflow flag should be used in
+  /// codegenprepare. Requires preferZeroCompareBranch() to be enabled.
+  virtual bool preferUSubOverflowBranch() const { return false; }
+
   /// Return true if it is cheaper to split the store of a merged int val
   /// from a pair of smaller values into multiple stores.
   virtual bool isMultiStoresCheaperThanBitsMerge(EVT LTy, EVT HTy) const {
diff --git a/llvm/lib/CodeGen/CodeGenPrepare.cpp b/llvm/lib/CodeGen/CodeGenPrepare.cpp
index ff40f210c5790..e246ad3db1b62 100644
--- a/llvm/lib/CodeGen/CodeGenPrepare.cpp
+++ b/llvm/lib/CodeGen/CodeGenPrepare.cpp
@@ -8878,6 +8878,10 @@ static bool tryUnmergingGEPsAcrossIndirectBr(GetElementPtrInst *GEPI,
 static bool optimizeBranch(CondBrInst *Branch, const TargetLowering &TLI,
                            SmallPtrSet<BasicBlock *, 32> &FreshBBs,
                            bool IsHugeFunc) {
+  if (TLI.preferUSubOverflowBranch())
+    assert(TLI.preferZeroCompareBranch() &&
+           "preferUSubOverflowBranch requires preferZeroCompareBranch");
+
   // Try and convert
   //  %c = icmp ult %x, 8
   //  br %c, bla, blb
@@ -8925,6 +8929,7 @@ static bool optimizeBranch(CondBrInst *Branch, const TargetLowering &TLI,
       replaceAllUsesWith(Cmp, NewCmp, FreshBBs, IsHugeFunc);
       return true;
     }
+
     if (Cmp->isEquality() &&
         (match(UI, m_Add(m_Specific(X), m_SpecificInt(-CmpC))) ||
          match(UI, m_Sub(m_Specific(X), m_SpecificInt(CmpC))) ||
@@ -8940,6 +8945,40 @@ static bool optimizeBranch(CondBrInst *Branch, const TargetLowering &TLI,
       replaceAllUsesWith(Cmp, NewCmp, FreshBBs, IsHugeFunc);
       return true;
     }
+
+    if (TLI.preferUSubOverflowBranch() &&
+        Cmp->getPredicate() == ICmpInst::ICMP_ULT &&
+        (match(UI, m_Add(m_Specific(X), m_SpecificInt(-CmpC))) ||
+         match(UI, m_Sub(m_Specific(X), m_SpecificInt(CmpC))))) {
+      // Convert
+      //
+      //   %cmp = icmp ult i32 %x, 3
+      //   br i1 %cmp, label %bb1, label %bb2
+      //   bb2:
+      //   %sub = add i32 %x, -3
+      //
+      // into
+      //
+      //   %usub = call { i32, i1 } @llvm.usub.with.overflow.i32(i32 %x, i32 -3)
+      //   %sub = extractvalue { i32, i1 } %usub, 0
+      //   %oflow = extractvalue { i32, i1 } %usub, 1
+      //   br i1 %oflow, label %bb1, label %bb2
+      //
+      // for targets where lowering the intrinsic and branching on its overflow
+      // flag is efficient.
+      Function *USubWithOverflow = Intrinsic::getOrInsertDeclaration(
+          Branch->getModule(), Intrinsic::usub_with_overflow, {UI->getType()});
+      IRBuilder<> Builder(Branch);
+      Value *Call = Builder.CreateCall(
+          USubWithOverflow, {UI->getOperand(0), UI->getOperand(1)}, "usub");
+      Value *Sub = Builder.CreateExtractValue(Call, 0, "sub");
+      Value *Overflow = Builder.CreateExtractValue(Call, 1, "oflow");
+      LLVM_DEBUG(dbgs() << "Converting " << *Cmp << "\n");
+      LLVM_DEBUG(dbgs() << " to llvm.usub.with.overflow: " << *Call << "\n");
+      replaceAllUsesWith(Cmp, Overflow, FreshBBs, IsHugeFunc);
+      replaceAllUsesWith(UI, Sub, FreshBBs, IsHugeFunc);
+      return true;
+    }
   }
   return false;
 }
diff --git a/llvm/lib/Target/X86/X86ISelLowering.cpp b/llvm/lib/Target/X86/X86ISelLowering.cpp
index e2f3b5f3cd3d5..f77bac51ba756 100644
--- a/llvm/lib/Target/X86/X86ISelLowering.cpp
+++ b/llvm/lib/Target/X86/X86ISelLowering.cpp
@@ -3804,6 +3804,8 @@ bool X86TargetLowering::isCtlzFast() const {
 
 bool X86TargetLowering::preferZeroCompareBranch() const { return true; }
 
+bool X86TargetLowering::preferUSubOverflowBranch() const { return true; }
+
 bool X86TargetLowering::isMaskAndCmp0FoldingBeneficial(
     const Instruction &AndI) const {
   return true;
diff --git a/llvm/lib/Target/X86/X86ISelLowering.h b/llvm/lib/Target/X86/X86ISelLowering.h
index 34bb76982eecb..119bbecb0293d 100644
--- a/llvm/lib/Target/X86/X86ISelLowering.h
+++ b/llvm/lib/Target/X86/X86ISelLowering.h
@@ -240,6 +240,8 @@ namespace llvm {
 
     bool preferZeroCompareBranch() const override;
 
+    bool preferUSubOverflowBranch() const override;
+
     bool isMultiStoresCheaperThanBitsMerge(EVT LTy, EVT HTy) const override {
       // If the pair to store is a mixture of float and int values, we will
       // save two bitwise instructions and one float-to-int instruction and
diff --git a/llvm/test/CodeGen/X86/branch-on-usub-overflow.ll b/llvm/test/CodeGen/X86/branch-on-usub-overflow.ll
new file mode 100644
index 0000000000000..8ed6d2a0ca81a
--- /dev/null
+++ b/llvm/test/CodeGen/X86/branch-on-usub-overflow.ll
@@ -0,0 +1,63 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
+; RUN: llc -mtriple x86_64-unknown-linux-gnu -o - %s | FileCheck %s --check-prefix=CHECK64
+; RUN: llc -mtriple i686-unknown-linux-gnu -o - %s   | FileCheck %s --check-prefix=CHECK32
+
+declare dso_local fastcc void @use(i32)
+
+define void @ult_sub10(i32 %a) nounwind {
+; CHECK64-LABEL: ult_sub10:
+; CHECK64:       # %bb.0: # %entry
+; CHECK64-NEXT:    subl $10, %edi
+; CHECK64-NEXT:    jae use # TAILCALL
+; CHECK64-NEXT:  # %bb.1: # %then
+; CHECK64-NEXT:    retq
+;
+; CHECK32-LABEL: ult_sub10:
+; CHECK32:       # %bb.0: # %entry
+; CHECK32-NEXT:    movl {{[0-9]+}}(%esp), %ecx
+; CHECK32-NEXT:    subl $10, %ecx
+; CHECK32-NEXT:    jae use # TAILCALL
+; CHECK32-NEXT:  # %bb.1: # %then
+; CHECK32-NEXT:    retl
+entry:
+  %c = icmp ult i32 %a, 10
+  br i1 %c, label %then, label %else
+then:
+  ret void
+else:
+  %l = sub i32 %a, 10
+  tail call fastcc void @use(i32 %l)
+  ret void
+}
+
+define void @ule_sub10(i32 %a) nounwind {
+; CHECK64-LABEL: ule_sub10:
+; CHECK64:       # %bb.0: # %entry
+; CHECK64-NEXT:    cmpl $10, %edi
+; CHECK64-NEXT:    ja .LBB1_2
+; CHECK64-NEXT:  # %bb.1: # %then
+; CHECK64-NEXT:    retq
+; CHECK64-NEXT:  .LBB1_2: # %else
+; CHECK64-NEXT:    addl $-10, %edi
+; CHECK64-NEXT:    jmp use # TAILCALL
+;
+; CHECK32-LABEL: ule_sub10:
+; CHECK32:       # %bb.0: # %entry
+; CHECK32-NEXT:    movl {{[0-9]+}}(%esp), %ecx
+; CHECK32-NEXT:    cmpl $10, %ecx
+; CHECK32-NEXT:    ja .LBB1_2
+; CHECK32-NEXT:  # %bb.1: # %then
+; CHECK32-NEXT:    retl
+; CHECK32-NEXT:  .LBB1_2: # %else
+; CHECK32-NEXT:    addl $-10, %ecx
+; CHECK32-NEXT:    jmp use # TAILCALL
+entry:
+  %c = icmp ule i32 %a, 10
+  br i1 %c, label %then, label %else
+then:
+  ret void
+else:
+  %l = sub i32 %a, 10
+  tail call fastcc void @use(i32 %l)
+  ret void
+}
diff --git a/llvm/test/Transforms/CodeGenPrepare/X86/branch-on-usub-overflow.ll b/llvm/test/Transforms/CodeGenPrepare/X86/branch-on-usub-overflow.ll
new file mode 100644
index 0000000000000..b77b6eb231ed3
--- /dev/null
+++ b/llvm/test/Transforms/CodeGenPrepare/X86/branch-on-usub-overflow.ll
@@ -0,0 +1,54 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py
+; RUN: opt -S -passes='require<profile-summary>,function(codegenprepare)' < %s | FileCheck %s
+
+target triple = "x86_64-unknown-linux-gnu"
+
+declare dso_local fastcc void @use(i32)
+
+define void @ult_sub10(i32 %a) nounwind {
+; CHECK-LABEL: @ult_sub10(
+; CHECK-NEXT:  entry:
+; CHECK-NEXT:    [[USUB:%.*]] = call { i32, i1 } @llvm.usub.with.overflow.i32(i32 [[A:%.*]], i32 10)
+; CHECK-NEXT:    [[SUB:%.*]] = extractvalue { i32, i1 } [[USUB]], 0
+; CHECK-NEXT:    [[OFLOW:%.*]] = extractvalue { i32, i1 } [[USUB]], 1
+; CHECK-NEXT:    br i1 [[OFLOW]], label [[THEN:%.*]], label [[ELSE:%.*]]
+; CHECK:       then:
+; CHECK-NEXT:    ret void
+; CHECK:       else:
+; CHECK-NEXT:    [[L:%.*]] = sub i32 [[A]], 10
+; CHECK-NEXT:    tail call fastcc void @use(i32 [[SUB]])
+; CHECK-NEXT:    ret void
+;
+entry:
+  %c = icmp ult i32 %a, 10
+  br i1 %c, label %then, label %else
+then:
+  ret void
+else:
+  %l = sub i32 %a, 10
+  tail call fastcc void @use(i32 %l)
+  ret void
+}
+
+define void @ule_sub10(i32 %a) nounwind {
+; CHECK-LABEL: @ule_sub10(
+; CHECK-NEXT:  entry:
+; CHECK-NEXT:    [[C:%.*]] = icmp ule i32 [[A:%.*]], 10
+; CHECK-NEXT:    br i1 [[C]], label [[THEN:%.*]], label [[ELSE:%.*]]
+; CHECK:       then:
+; CHECK-NEXT:    ret void
+; CHECK:       else:
+; CHECK-NEXT:    [[L:%.*]] = sub i32 [[A]], 10
+; CHECK-NEXT:    tail call fastcc void @use(i32 [[L]])
+; CHECK-NEXT:    ret void
+;
+entry:
+  %c = icmp ule i32 %a, 10
+  br i1 %c, label %then, label %else
+then:
+  ret void
+else:
+  %l = sub i32 %a, 10
+  tail call fastcc void @use(i32 %l)
+  ret void
+}



More information about the llvm-commits mailing list