[llvm] [X86] Fold dead-result SUB into successor recomputations (PR #208633)

via llvm-commits llvm-commits at lists.llvm.org
Fri Jul 10 00:04:16 PDT 2026


https://github.com/FathimaHaris created https://github.com/llvm/llvm-project/pull/208633

optimizeCompareInstr converted dead-result SUB instructions into CMPs even if their destination 
could be preserved to eliminate identical arithmetic recomputations in predictable successor blocks. 
 
Add a local helper foldRedundantRecompute to check single-predecessor successor blocks
for a matching offset recomputation. If found, the helper reused the SUB instruction's destination,
replaces the redundant instruction's destination uses, and erases the redundant operation, 
eliminating an unnecessary instruction.

Fixes #195589 

>From adeb1821803eaf1b46f80a7c87cc8e321aeef847 Mon Sep 17 00:00:00 2001
From: FathimaHaris <fathimarazack4 at gmail.com>
Date: Fri, 10 Jul 2026 06:36:50 +0000
Subject: [PATCH 1/2] [X86] Add test for missed reuse of dead-result SUB in
 redundant recompute

Test for a case where a dead-result SUB instruction used for comparison
fails to reuse its destination to eliminate a redundant recomputation
in a successor block, and is instead converted into a CMP.
---
 llvm/test/CodeGen/X86/cmp-merge.ll | 271 +++++++++++++++++++++++++++++
 1 file changed, 271 insertions(+)

diff --git a/llvm/test/CodeGen/X86/cmp-merge.ll b/llvm/test/CodeGen/X86/cmp-merge.ll
index 0c355af64b027..1b67747c2472f 100644
--- a/llvm/test/CodeGen/X86/cmp-merge.ll
+++ b/llvm/test/CodeGen/X86/cmp-merge.ll
@@ -151,4 +151,275 @@ cond.end:
   ret void
 }
 
+
+;Check that a dead-result comparison (SUB) can be preserved
+; and reused to eliminate a redundant recomputation in a successor block.
+
+define i32 @fold_add_i32(i32 %x) {
+; X86-LABEL: fold_add_i32:
+; X86:       # %bb.0: # %entry
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %ecx
+; X86-NEXT:    xorl %eax, %eax
+; X86-NEXT:    cmpl $5, %ecx
+; X86-NEXT:    jl .LBB4_2
+; X86-NEXT:  # %bb.1: # %bb.nph
+; X86-NEXT:    addl $-5, %ecx
+; X86-NEXT:    movl %ecx, %eax
+; X86-NEXT:  .LBB4_2: # %ret
+; X86-NEXT:    retl
+;
+; X64-LABEL: fold_add_i32:
+; X64:       # %bb.0: # %entry
+; X64-NEXT:    xorl %eax, %eax
+; X64-NEXT:    cmpl $5, %edi
+; X64-NEXT:    jl .LBB4_2
+; X64-NEXT:  # %bb.1: # %bb.nph
+; X64-NEXT:    addl $-5, %edi
+; X64-NEXT:    movl %edi, %eax
+; X64-NEXT:  .LBB4_2: # %ret
+; X64-NEXT:    retq
+entry:
+  %cmp = icmp sgt i32 %x, 4
+  br i1 %cmp, label %bb.nph, label %ret
+
+bb.nph:
+  %t = add i32 %x, -5
+  br label %ret
+
+ret:
+  %r = phi i32 [ %t, %bb.nph ], [ 0, %entry ]
+  ret i32 %r
+}
+
+
+define i8 @fold_sub_i8(i8 %x)  {
+; X86-LABEL: fold_sub_i8:
+; X86:       # %bb.0: # %entry
+; X86-NEXT:    movzbl {{[0-9]+}}(%esp), %eax
+; X86-NEXT:    cmpb $5, %al
+; X86-NEXT:    jl .LBB5_1
+; X86-NEXT:  # %bb.2: # %bb.nph
+; X86-NEXT:    addb $-5, %al
+; X86-NEXT:    # kill: def $al killed $al killed $eax
+; X86-NEXT:    retl
+; X86-NEXT:  .LBB5_1:
+; X86-NEXT:    xorl %eax, %eax
+; X86-NEXT:    # kill: def $al killed $al killed $eax
+; X86-NEXT:    retl
+;
+; X64-LABEL: fold_sub_i8:
+; X64:       # %bb.0: # %entry
+; X64-NEXT:    movl %edi, %eax
+; X64-NEXT:    cmpb $5, %al
+; X64-NEXT:    jl .LBB5_1
+; X64-NEXT:  # %bb.2: # %bb.nph
+; X64-NEXT:    addb $-5, %al
+; X64-NEXT:    # kill: def $al killed $al killed $eax
+; X64-NEXT:    retq
+; X64-NEXT:  .LBB5_1:
+; X64-NEXT:    xorl %eax, %eax
+; X64-NEXT:    # kill: def $al killed $al killed $eax
+; X64-NEXT:    retq
+entry:
+  %cmp = icmp sgt i8 %x, 4
+  br i1 %cmp, label %bb.nph, label %ret
+
+bb.nph:
+  %t = sub i8 %x, 5
+  br label %ret
+
+ret:
+  %r = phi i8 [ %t, %bb.nph ], [ 0, %entry ]
+  ret i8 %r
+}
+
+
+
+;from issue report 195589 .
+define internal void @example.emit(ptr %0, i64 %1, ptr  %2, i64 %3)  {
+; X86-LABEL: example.emit:
+; X86:       # %bb.0: # %Entry
+; X86-NEXT:    pushl %ebp
+; X86-NEXT:    .cfi_def_cfa_offset 8
+; X86-NEXT:    pushl %ebx
+; X86-NEXT:    .cfi_def_cfa_offset 12
+; X86-NEXT:    pushl %edi
+; X86-NEXT:    .cfi_def_cfa_offset 16
+; X86-NEXT:    pushl %esi
+; X86-NEXT:    .cfi_def_cfa_offset 20
+; X86-NEXT:    subl $36, %esp
+; X86-NEXT:    .cfi_def_cfa_offset 56
+; X86-NEXT:    .cfi_offset %esi, -20
+; X86-NEXT:    .cfi_offset %edi, -16
+; X86-NEXT:    .cfi_offset %ebx, -12
+; X86-NEXT:    .cfi_offset %ebp, -8
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax
+; X86-NEXT:    movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax
+; X86-NEXT:    movl %eax, (%esp) # 4-byte Spill
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %edx
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %esi
+; X86-NEXT:    .p2align 4
+; X86-NEXT:  .LBB6_1: # %Loop
+; X86-NEXT:    # =>This Inner Loop Header: Depth=1
+; X86-NEXT:    movl 12(%esi), %eax
+; X86-NEXT:    movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
+; X86-NEXT:    movl (%esi), %eax
+; X86-NEXT:    movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
+; X86-NEXT:    movl 4(%esi), %eax
+; X86-NEXT:    movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
+; X86-NEXT:    movl 8(%esi), %eax
+; X86-NEXT:    movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
+; X86-NEXT:    movl 16(%esi), %eax
+; X86-NEXT:    movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
+; X86-NEXT:    movl 20(%esi), %edi
+; X86-NEXT:    movl 24(%esi), %ecx
+; X86-NEXT:    movl 28(%esi), %eax
+; X86-NEXT:    movl 32(%esi), %ebx
+; X86-NEXT:    movl %ebx, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
+; X86-NEXT:    movl 36(%esi), %ebx
+; X86-NEXT:    movl %ebx, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
+; X86-NEXT:    movl 40(%esi), %ebp
+; X86-NEXT:    movl 44(%esi), %ebx
+; X86-NEXT:    movl %eax, 28(%edx)
+; X86-NEXT:    movl %ecx, 24(%edx)
+; X86-NEXT:    movl %edi, 20(%edx)
+; X86-NEXT:    movl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 4-byte Reload
+; X86-NEXT:    movl %eax, 16(%edx)
+; X86-NEXT:    movl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 4-byte Reload
+; X86-NEXT:    movl %eax, 8(%edx)
+; X86-NEXT:    movl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 4-byte Reload
+; X86-NEXT:    movl %eax, 4(%edx)
+; X86-NEXT:    movl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 4-byte Reload
+; X86-NEXT:    movl %eax, (%edx)
+; X86-NEXT:    movl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 4-byte Reload
+; X86-NEXT:    movl %eax, 12(%edx)
+; X86-NEXT:    movl %ebx, 44(%edx)
+; X86-NEXT:    movl %ebp, 40(%edx)
+; X86-NEXT:    movl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 4-byte Reload
+; X86-NEXT:    movl %eax, 36(%edx)
+; X86-NEXT:    movl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 4-byte Reload
+; X86-NEXT:    movl %eax, 32(%edx)
+; X86-NEXT:    movl (%esp), %edi # 4-byte Reload
+; X86-NEXT:    movl $64, %eax
+; X86-NEXT:    cmpl %edi, %eax
+; X86-NEXT:    movl $0, %eax
+; X86-NEXT:    movl {{[-0-9]+}}(%e{{[sb]}}p), %ecx # 4-byte Reload
+; X86-NEXT:    sbbl %ecx, %eax
+; X86-NEXT:    jae .LBB6_2
+; X86-NEXT:  # %bb.3: # %Else
+; X86-NEXT:    # in Loop: Header=BB6_1 Depth=1
+; X86-NEXT:    addl $-64, %edi
+; X86-NEXT:    movl %edi, (%esp) # 4-byte Spill
+; X86-NEXT:    adcl $-1, %ecx
+; X86-NEXT:    movl %ecx, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
+; X86-NEXT:    addl $64, %esi
+; X86-NEXT:    addl $48, %edx
+; X86-NEXT:    jmp .LBB6_1
+; X86-NEXT:  .LBB6_2: # %Then
+; X86-NEXT:    addl $36, %esp
+; X86-NEXT:    .cfi_def_cfa_offset 20
+; X86-NEXT:    popl %esi
+; X86-NEXT:    .cfi_def_cfa_offset 16
+; X86-NEXT:    popl %edi
+; X86-NEXT:    .cfi_def_cfa_offset 12
+; X86-NEXT:    popl %ebx
+; X86-NEXT:    .cfi_def_cfa_offset 8
+; X86-NEXT:    popl %ebp
+; X86-NEXT:    .cfi_def_cfa_offset 4
+; X86-NEXT:    retl
+;
+; X64-LABEL: example.emit:
+; X64:       # %bb.0: # %Entry
+; X64-NEXT:    .p2align 4
+; X64-NEXT:  .LBB6_1: # %Loop
+; X64-NEXT:    # =>This Inner Loop Header: Depth=1
+; X64-NEXT:    movups (%rdi), %xmm0
+; X64-NEXT:    movups 16(%rdi), %xmm1
+; X64-NEXT:    movups 32(%rdi), %xmm2
+; X64-NEXT:    movups %xmm0, (%rdx)
+; X64-NEXT:    movups %xmm1, 16(%rdx)
+; X64-NEXT:    movups %xmm2, 32(%rdx)
+; X64-NEXT:    cmpq $64, %rsi
+; X64-NEXT:    jbe .LBB6_2
+; X64-NEXT:  # %bb.3: # %Else
+; X64-NEXT:    # in Loop: Header=BB6_1 Depth=1
+; X64-NEXT:    addq $64, %rdi
+; X64-NEXT:    addq $-64, %rsi
+; X64-NEXT:    addq $48, %rdx
+; X64-NEXT:    jmp .LBB6_1
+; X64-NEXT:  .LBB6_2: # %Then
+; X64-NEXT:    retq
+Entry:
+  br label %Loop
+
+Loop:
+  %.sroa.4.0 = phi i64 [ %1, %Entry ], [ %8, %Else ]
+  %.sroa.0.0 = phi ptr [ %0, %Entry ], [ %7, %Else ]
+  %.sroa.032.0 = phi ptr [ %2, %Entry ], [ %9, %Else ]
+  %.sroa.33.0..sroa_idx = getelementptr i8, ptr %.sroa.0.0, i64 32
+  %4 = load <32 x i8>, ptr %.sroa.0.0, align 1
+  %.sroa.3369.0..sroa_idx = getelementptr i8, ptr %.sroa.032.0, i64 32
+  %5 = load <16 x i8>, ptr %.sroa.33.0..sroa_idx, align 1
+  store <32 x i8> %4, ptr %.sroa.032.0, align 1
+  store <16 x i8> %5, ptr %.sroa.3369.0..sroa_idx, align 1
+  %6 = icmp ult i64 %.sroa.4.0, 65
+  br i1 %6, label %Then, label %Else
+
+Then:
+  ret void
+
+Else:
+  %7 = getelementptr i8, ptr %.sroa.0.0, i64 64
+  %8 = add i64 %.sroa.4.0, -64
+  %9 = getelementptr i8, ptr %.sroa.032.0, i64 48
+  br label %Loop
+}
+
+
+;negative case
+define i32 @no_fold_when_srcreg_reused(i32 %x_offs) nounwind readnone {
+; X86-LABEL: no_fold_when_srcreg_reused:
+; X86:       # %bb.0: # %entry
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax
+; X86-NEXT:    cmpl $7, %eax
+; X86-NEXT:    jl .LBB7_2
+; X86-NEXT:  # %bb.1: # %bb.nph
+; X86-NEXT:    leal -5(%eax), %ecx
+; X86-NEXT:    andl $-4, %ecx
+; X86-NEXT:    negl %ecx
+; X86-NEXT:    leal -4(%eax,%ecx), %eax
+; X86-NEXT:  .LBB7_2: # %bb2
+; X86-NEXT:    retl
+;
+; X64-LABEL: no_fold_when_srcreg_reused:
+; X64:       # %bb.0: # %entry
+; X64-NEXT:    # kill: def $edi killed $edi def $rdi
+; X64-NEXT:    cmpl $7, %edi
+; X64-NEXT:    jl .LBB7_2
+; X64-NEXT:  # %bb.1: # %bb.nph
+; X64-NEXT:    leal -5(%rdi), %eax
+; X64-NEXT:    andl $-4, %eax
+; X64-NEXT:    negl %eax
+; X64-NEXT:    leal -4(%rdi,%rax), %eax
+; X64-NEXT:    retq
+; X64-NEXT:  .LBB7_2: # %bb2
+; X64-NEXT:    movl %edi, %eax
+; X64-NEXT:    retq
+entry:
+  %t0 = icmp sgt i32 %x_offs, 6
+  br i1 %t0, label %bb.nph, label %bb2
+
+bb.nph:
+  %tmp = add i32 %x_offs, -5
+  %tmp6 = lshr i32 %tmp, 2
+  %tmp7 = mul i32 %tmp6, -4
+  %tmp8 = add i32 %tmp7, %x_offs
+  %tmp9 = add i32 %tmp8, -4
+  ret i32 %tmp9
+
+bb2:
+  ret i32 %x_offs
+}
+
 declare void @use(i32)

>From b30d9d2625a1475348eb82d0c4d670ba2e9149ff Mon Sep 17 00:00:00 2001
From: FathimaHaris <fathimarazack4 at gmail.com>
Date: Fri, 10 Jul 2026 06:53:38 +0000
Subject: [PATCH 2/2] [X86] Preserve dead-result SUB to eliminate successor
 recomputations

Update optimizeCompareInstr to check if a dead-result SUB can be preserved
and reused rather than converted into a CMP. When an identical offset
recomputation exists in a single-predecessor successor block, we reuse the
SUB destination register and erase the redundant operation.
---
 llvm/lib/Target/X86/X86InstrInfo.cpp          | 97 +++++++++++++++++++
 .../test/CodeGen/X86/2006-05-11-InstrSched.ll |  2 +-
 llvm/test/CodeGen/X86/cmp-merge.ll            | 31 +++---
 3 files changed, 109 insertions(+), 21 deletions(-)

diff --git a/llvm/lib/Target/X86/X86InstrInfo.cpp b/llvm/lib/Target/X86/X86InstrInfo.cpp
index 7ff2400d06d1d..7cba1ada7987b 100644
--- a/llvm/lib/Target/X86/X86InstrInfo.cpp
+++ b/llvm/lib/Target/X86/X86InstrInfo.cpp
@@ -5285,6 +5285,96 @@ static std::pair<X86::CondCode, unsigned> isUseDefConvertible(const MachineInstr
   }
 }
 
+// If CmpInstr is a dead-result SUB reg, imm, and a single-pred
+// successor redundantly recomputes reg - imm, reuse CmpInstr's
+// destination for that instead of erasing it, and delete the redundant
+// recompute. Returns true if something was folded.
+
+static bool foldRedundantRecompute(MachineInstr &CmpInstr, Register SrcReg,
+                                   int64_t CmpValue) {
+
+  Register DstReg = CmpInstr.getOperand(0).getReg();
+
+  // Transformation currently requires SSA values.
+  if (!SrcReg.isVirtual() || !DstReg.isVirtual())
+    return false;
+
+  MachineBasicBlock &MBB = *CmpInstr.getParent();
+  MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo();
+  MachineBasicBlock *TBB = nullptr, *FBB = nullptr;
+  SmallVector<MachineOperand, 4> Cond;
+  const TargetInstrInfo *TII = MBB.getParent()->getSubtarget().getInstrInfo();
+  if (TII->analyzeBranch(MBB, TBB, FBB, Cond, /*AllowModify=*/false))
+    return false;
+
+  if (Cond.empty())
+    return false;
+  MachineBasicBlock *Fallthrough = MBB.getFallThrough();
+  for (MachineBasicBlock *Succ : {TBB, FBB ? FBB : Fallthrough}) {
+    if (!Succ || Succ->pred_size() != 1)
+      continue;
+
+    for (MachineInstr &MI : *Succ) {
+      if (MI.isDebugInstr())
+        continue;
+
+      bool IsMatch = false;
+
+      switch (MI.getOpcode()) {
+        CASE_ND(ADD64ri32)
+        CASE_ND(ADD32ri)
+        CASE_ND(ADD16ri)
+        CASE_ND(ADD8ri)
+        IsMatch = MI.getOperand(1).getReg() == SrcReg &&
+                  MI.getOperand(2).isImm() &&
+                  MI.getOperand(2).getImm() == -CmpValue;
+        break;
+        CASE_ND(SUB64ri32)
+        CASE_ND(SUB32ri)
+        CASE_ND(SUB16ri)
+        CASE_ND(SUB8ri)
+        IsMatch = MI.getOperand(1).getReg() == SrcReg &&
+                  MI.getOperand(2).isImm() &&
+                  MI.getOperand(2).getImm() == CmpValue;
+        break;
+      default:
+        break;
+      }
+
+      if (IsMatch) {
+
+        // Ensure the matching instruction does not define a live EFLAGS value
+        // that is required downstream.
+        MachineOperand *FlagDef =
+            MI.findRegisterDefOperand(X86::EFLAGS, /*TRI=*/nullptr);
+        if (FlagDef && !FlagDef->isDead())
+          return false;
+
+        // Register classes must match exactly to safely reuse the destination.
+        Register OldDst = MI.getOperand(0).getReg();
+        if (!OldDst.isVirtual() ||
+            MRI.getRegClass(OldDst) != MRI.getRegClass(DstReg))
+          return false;
+
+        // Restrict the fold to cases where SrcReg has no uses other than
+        // CmpInstr and MI. Otherwise, preserving SrcReg for its remaining
+        // uses can result in additional copies during register allocation.
+        for (MachineInstr &UseMI : MRI.use_nodbg_instructions(SrcReg)) {
+          if (&UseMI != &CmpInstr && &UseMI != &MI)
+            return false;
+        }
+
+        CmpInstr.getOperand(0).setIsDead(false);
+        MRI.replaceRegWith(OldDst, DstReg);
+        MI.eraseFromParent();
+
+        return true;
+      }
+    }
+  }
+  return false;
+}
+
 /// Check if there exists an earlier instruction that
 /// operates on the same source operands and sets flags in the same way as
 /// Compare; remove Compare if possible.
@@ -5310,6 +5400,13 @@ bool X86InstrInfo::optimizeCompareInstr(MachineInstr &CmpInstr, Register SrcReg,
   CASE_ND(SUB8rr) {
     if (!MRI->use_nodbg_empty(CmpInstr.getOperand(0).getReg()))
       return false;
+
+    // Before discarding SUB's dead destination to make a plain CMP, see
+    // if a successor redundantly recomputes the same value and can reuse
+    // it instead.
+    if (CmpMask != 0 && foldRedundantRecompute(CmpInstr, SrcReg, CmpValue))
+      return true;
+
     // There is no use of the destination register, we can replace SUB with CMP.
     unsigned NewOpcode = 0;
 #define FROM_TO(A, B)                                                          \
diff --git a/llvm/test/CodeGen/X86/2006-05-11-InstrSched.ll b/llvm/test/CodeGen/X86/2006-05-11-InstrSched.ll
index a8fecba27bf3c..33eb2b04f28c0 100644
--- a/llvm/test/CodeGen/X86/2006-05-11-InstrSched.ll
+++ b/llvm/test/CodeGen/X86/2006-05-11-InstrSched.ll
@@ -1,6 +1,6 @@
 ; REQUIRES: asserts
 ; RUN: llc < %s -mtriple=i386-linux-gnu -mcpu=penryn -mattr=+sse2 -stats 2>&1 | \
-; RUN:     grep "asm-printer" | grep 33
+; RUN:     grep "asm-printer" | grep 32
 
 target datalayout = "e-p:32:32"
 define void @foo(ptr %mc, ptr %bp, ptr %ms, ptr %xmb, ptr %mpp, ptr %tpmm, ptr %ip, ptr %tpim, ptr %dpp, ptr %tpdm, ptr %bpi, i32 %M) nounwind {
diff --git a/llvm/test/CodeGen/X86/cmp-merge.ll b/llvm/test/CodeGen/X86/cmp-merge.ll
index 1b67747c2472f..f876aabc81696 100644
--- a/llvm/test/CodeGen/X86/cmp-merge.ll
+++ b/llvm/test/CodeGen/X86/cmp-merge.ll
@@ -160,10 +160,9 @@ define i32 @fold_add_i32(i32 %x) {
 ; X86:       # %bb.0: # %entry
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %ecx
 ; X86-NEXT:    xorl %eax, %eax
-; X86-NEXT:    cmpl $5, %ecx
+; X86-NEXT:    subl $5, %ecx
 ; X86-NEXT:    jl .LBB4_2
 ; X86-NEXT:  # %bb.1: # %bb.nph
-; X86-NEXT:    addl $-5, %ecx
 ; X86-NEXT:    movl %ecx, %eax
 ; X86-NEXT:  .LBB4_2: # %ret
 ; X86-NEXT:    retl
@@ -171,10 +170,9 @@ define i32 @fold_add_i32(i32 %x) {
 ; X64-LABEL: fold_add_i32:
 ; X64:       # %bb.0: # %entry
 ; X64-NEXT:    xorl %eax, %eax
-; X64-NEXT:    cmpl $5, %edi
+; X64-NEXT:    subl $5, %edi
 ; X64-NEXT:    jl .LBB4_2
 ; X64-NEXT:  # %bb.1: # %bb.nph
-; X64-NEXT:    addl $-5, %edi
 ; X64-NEXT:    movl %edi, %eax
 ; X64-NEXT:  .LBB4_2: # %ret
 ; X64-NEXT:    retq
@@ -196,28 +194,22 @@ define i8 @fold_sub_i8(i8 %x)  {
 ; X86-LABEL: fold_sub_i8:
 ; X86:       # %bb.0: # %entry
 ; X86-NEXT:    movzbl {{[0-9]+}}(%esp), %eax
-; X86-NEXT:    cmpb $5, %al
-; X86-NEXT:    jl .LBB5_1
-; X86-NEXT:  # %bb.2: # %bb.nph
-; X86-NEXT:    addb $-5, %al
-; X86-NEXT:    # kill: def $al killed $al killed $eax
-; X86-NEXT:    retl
-; X86-NEXT:  .LBB5_1:
+; X86-NEXT:    subb $5, %al
+; X86-NEXT:    jge .LBB5_2
+; X86-NEXT:  # %bb.1:
 ; X86-NEXT:    xorl %eax, %eax
+; X86-NEXT:  .LBB5_2: # %ret
 ; X86-NEXT:    # kill: def $al killed $al killed $eax
 ; X86-NEXT:    retl
 ;
 ; X64-LABEL: fold_sub_i8:
 ; X64:       # %bb.0: # %entry
 ; X64-NEXT:    movl %edi, %eax
-; X64-NEXT:    cmpb $5, %al
-; X64-NEXT:    jl .LBB5_1
-; X64-NEXT:  # %bb.2: # %bb.nph
-; X64-NEXT:    addb $-5, %al
-; X64-NEXT:    # kill: def $al killed $al killed $eax
-; X64-NEXT:    retq
-; X64-NEXT:  .LBB5_1:
+; X64-NEXT:    subb $5, %al
+; X64-NEXT:    jge .LBB5_2
+; X64-NEXT:  # %bb.1:
 ; X64-NEXT:    xorl %eax, %eax
+; X64-NEXT:  .LBB5_2: # %ret
 ; X64-NEXT:    # kill: def $al killed $al killed $eax
 ; X64-NEXT:    retq
 entry:
@@ -340,12 +332,11 @@ define internal void @example.emit(ptr %0, i64 %1, ptr  %2, i64 %3)  {
 ; X64-NEXT:    movups %xmm0, (%rdx)
 ; X64-NEXT:    movups %xmm1, 16(%rdx)
 ; X64-NEXT:    movups %xmm2, 32(%rdx)
-; X64-NEXT:    cmpq $64, %rsi
+; X64-NEXT:    subq $64, %rsi
 ; X64-NEXT:    jbe .LBB6_2
 ; X64-NEXT:  # %bb.3: # %Else
 ; X64-NEXT:    # in Loop: Header=BB6_1 Depth=1
 ; X64-NEXT:    addq $64, %rdi
-; X64-NEXT:    addq $-64, %rsi
 ; X64-NEXT:    addq $48, %rdx
 ; X64-NEXT:    jmp .LBB6_1
 ; X64-NEXT:  .LBB6_2: # %Then



More information about the llvm-commits mailing list