[llvm] [X86] Use CF from SHR for d & 1 when d >> 1 is also computed (PR #228181)
via llvm-commits
llvm-commits at lists.llvm.org
Fri Oct 2 05:10:06 PDT 2026
https://github.com/addmisol updated https://github.com/llvm/llvm-project/pull/228181
>From 6de647b55edc0b9e82a103a904a72341a52d7c8d Mon Sep 17 00:00:00 2001
From: addmisol <addmisol9 at gmail.com>
Date: Thu, 1 Oct 2026 23:23:44 +0530
Subject: [PATCH 1/3] [X86] Use CF from SHR for d & 1 when d >> 1 is also
computed
Signed-off-by: addmisol <addmisol9 at gmail.com>
---
llvm/lib/Target/X86/X86ISelLowering.cpp | 38 ++++-
llvm/lib/Target/X86/X86InstrFragments.td | 12 ++
llvm/lib/Target/X86/X86InstrShiftRotate.td | 52 +++++++
llvm/test/CodeGen/X86/shr-cf-combine.ll | 170 +++++++++++++++++++++
4 files changed, 271 insertions(+), 1 deletion(-)
create mode 100644 llvm/test/CodeGen/X86/shr-cf-combine.ll
diff --git a/llvm/lib/Target/X86/X86ISelLowering.cpp b/llvm/lib/Target/X86/X86ISelLowering.cpp
index 7817c86155e5b05..ddd0199ea3f4f12 100644
--- a/llvm/lib/Target/X86/X86ISelLowering.cpp
+++ b/llvm/lib/Target/X86/X86ISelLowering.cpp
@@ -53836,6 +53836,21 @@ static SDValue combineOrCmpEqZeroToCtlzSrl(SDNode *N, SelectionDAG &DAG,
return DAG.getNode(ISD::ZERO_EXTEND, SDLoc(N), N->getValueType(0), Ret);
}
+/// Find a SRL by 1 node that uses the same operand as D, if it exists.
+/// This is used to detect patterns where both (and D, 1) and (srl D, 1)
+/// are computed, allowing us to use the carry flag from SHR for both.
+static SDNode *findSrlBy1User(SDValue D) {
+ for (SDNode *User : D.getNode()->users()) {
+ if (User->getOpcode() == ISD::SRL && User->getOperand(0) == D) {
+ if (auto *ShAmtC = dyn_cast<ConstantSDNode>(User->getOperand(1))) {
+ if (ShAmtC->getZExtValue() == 1)
+ return User;
+ }
+ }
+ }
+ return nullptr;
+}
+
/// If this is an add or subtract where one operand is produced by a cmp+setcc,
/// then try to convert it to an ADC or SBB. This replaces TEST+SET+{ADD/SUB}
/// with CMP+{ADC, SBB}.
@@ -53854,12 +53869,33 @@ static SDValue combineAddOrSubToADCOrSBB(bool IsSub, const SDLoc &DL, EVT VT,
X86::CondCode CC;
SDValue EFLAGS;
+ SDNode *SrlBy1Node = nullptr;
if (Y.getOpcode() == X86ISD::SETCC && Y.hasOneUse()) {
CC = (X86::CondCode)Y.getConstantOperandVal(0);
EFLAGS = Y.getOperand(1);
} else if (Y.getOpcode() == ISD::AND && isOneConstant(Y.getOperand(1)) &&
Y.hasOneUse()) {
- EFLAGS = LowerAndToBT(Y, ISD::SETNE, DL, DAG, CC);
+ // Check if we have both (and D, 1) and (srl D, 1) for the same D.
+ // If so, we can use X86ISD::SHR_FLAG to get both the shifted result
+ // and the carry flag (which contains the LSB), avoiding a separate BT.
+ SDValue D = Y.getOperand(0);
+ SrlBy1Node = findSrlBy1User(D);
+ if (SrlBy1Node) {
+ // Create X86ISD::SHR_FLAG which produces (D >> 1, EFLAGS with CF = D & 1)
+ EVT SrlVT = SrlBy1Node->getValueType(0);
+ SDVTList VTs = DAG.getVTList(SrlVT, MVT::i32);
+ SDValue ShAmt = DAG.getConstant(1, DL, MVT::i8);
+ SDValue ShrFlag = DAG.getNode(X86ISD::SHR_FLAG, DL, VTs, D, ShAmt);
+
+ // Replace uses of the original SRL with the shifted result from SHR_FLAG
+ DAG.ReplaceAllUsesOfValueWith(SDValue(SrlBy1Node, 0), ShrFlag.getValue(0));
+
+ // Use the EFLAGS output from SHR_FLAG
+ EFLAGS = ShrFlag.getValue(1);
+ CC = X86::COND_B; // CF is set when (D & 1) is 1
+ } else {
+ EFLAGS = LowerAndToBT(Y, ISD::SETNE, DL, DAG, CC);
+ }
}
if (!EFLAGS)
diff --git a/llvm/lib/Target/X86/X86InstrFragments.td b/llvm/lib/Target/X86/X86InstrFragments.td
index 383e713f93810d3..e3d48e2e18ed7a6 100644
--- a/llvm/lib/Target/X86/X86InstrFragments.td
+++ b/llvm/lib/Target/X86/X86InstrFragments.td
@@ -38,6 +38,13 @@ def SDTBinaryArithWithFlags : SDTypeProfile<2, 2,
SDTCisSameAs<0, 3>,
SDTCisInt<0>, SDTCisVT<1, i32>]>;
+// SDTShiftWithFlags - RES, EFLAGS = op SRC, SHAMT
+// Shift amount is i8, result matches source type.
+def SDTShiftWithFlags : SDTypeProfile<2, 2,
+ [SDTCisSameAs<0, 2>,
+ SDTCisInt<0>, SDTCisVT<1, i32>,
+ SDTCisVT<3, i8>]>;
+
// SDTBinaryArithWithFlagsInOut - RES1, EFLAGS = op LHS, RHS, EFLAGS
def SDTBinaryArithWithFlagsInOut : SDTypeProfile<2, 3,
[SDTCisSameAs<0, 2>,
@@ -400,6 +407,11 @@ def X86xor_flag : SDNode<"X86ISD::XOR", SDTBinaryArithWithFlags,
def X86and_flag : SDNode<"X86ISD::AND", SDTBinaryArithWithFlags,
[SDNPCommutative]>;
+// Shift right logical with flags. Used when we need both the shifted result
+// and the carry flag (which contains the shifted-out bit).
+// RES, EFLAGS = SHR_FLAG SRC, SHAMT
+def X86shr_flag : SDNode<"X86ISD::SHR_FLAG", SDTShiftWithFlags>;
+
// LOCK-prefixed arithmetic read-modify-write instructions.
// EFLAGS, OUTCHAIN = LADD(INCHAIN, PTR, RHS)
def X86lock_add : SDNode<"X86ISD::LADD", SDTLockBinaryArithWithFlags,
diff --git a/llvm/lib/Target/X86/X86InstrShiftRotate.td b/llvm/lib/Target/X86/X86InstrShiftRotate.td
index 7e7c2f97c57937f..4ba919c2ec27bb3 100644
--- a/llvm/lib/Target/X86/X86InstrShiftRotate.td
+++ b/llvm/lib/Target/X86/X86InstrShiftRotate.td
@@ -689,3 +689,55 @@ let Predicates = [HasBMI2, HasEGPR] in {
defm SHRX : ShiftX_Pats<srl, "_EVEX">;
defm SHLX : ShiftX_Pats<shl, "_EVEX">;
}
+
+// Patterns for X86ISD::SHR_FLAG - shift right with flags output.
+// Used when we need both the shifted result and the carry flag
+// (which contains the shifted-out bit). This is primarily used for
+// optimizing patterns like: (sub X, (and D, 1)) combined with (srl D, 1)
+// where the shift by 1 already puts the LSB into CF.
+
+// SHR_FLAG with immediate 1 - use the shift-by-1 instructions.
+let Predicates = [NoNDD] in {
+ def : Pat<(X86shr_flag GR8:$src, (i8 1)),
+ (SHR8r1 GR8:$src)>;
+ def : Pat<(X86shr_flag GR16:$src, (i8 1)),
+ (SHR16r1 GR16:$src)>;
+ def : Pat<(X86shr_flag GR32:$src, (i8 1)),
+ (SHR32r1 GR32:$src)>;
+ def : Pat<(X86shr_flag GR64:$src, (i8 1)),
+ (SHR64r1 GR64:$src)>;
+}
+
+let Predicates = [HasNDD, In64BitMode] in {
+ def : Pat<(X86shr_flag GR8:$src, (i8 1)),
+ (SHR8r1_ND GR8:$src)>;
+ def : Pat<(X86shr_flag GR16:$src, (i8 1)),
+ (SHR16r1_ND GR16:$src)>;
+ def : Pat<(X86shr_flag GR32:$src, (i8 1)),
+ (SHR32r1_ND GR32:$src)>;
+ def : Pat<(X86shr_flag GR64:$src, (i8 1)),
+ (SHR64r1_ND GR64:$src)>;
+}
+
+// SHR_FLAG with immediate > 1 - use the shift-by-immediate instructions.
+let Predicates = [NoNDD] in {
+ def : Pat<(X86shr_flag GR8:$src, (i8 timm:$amt)),
+ (SHR8ri GR8:$src, timm:$amt)>;
+ def : Pat<(X86shr_flag GR16:$src, (i8 timm:$amt)),
+ (SHR16ri GR16:$src, timm:$amt)>;
+ def : Pat<(X86shr_flag GR32:$src, (i8 timm:$amt)),
+ (SHR32ri GR32:$src, timm:$amt)>;
+ def : Pat<(X86shr_flag GR64:$src, (i8 timm:$amt)),
+ (SHR64ri GR64:$src, timm:$amt)>;
+}
+
+let Predicates = [HasNDD, In64BitMode] in {
+ def : Pat<(X86shr_flag GR8:$src, (i8 timm:$amt)),
+ (SHR8ri_ND GR8:$src, timm:$amt)>;
+ def : Pat<(X86shr_flag GR16:$src, (i8 timm:$amt)),
+ (SHR16ri_ND GR16:$src, timm:$amt)>;
+ def : Pat<(X86shr_flag GR32:$src, (i8 timm:$amt)),
+ (SHR32ri_ND GR32:$src, timm:$amt)>;
+ def : Pat<(X86shr_flag GR64:$src, (i8 timm:$amt)),
+ (SHR64ri_ND GR64:$src, timm:$amt)>;
+}
diff --git a/llvm/test/CodeGen/X86/shr-cf-combine.ll b/llvm/test/CodeGen/X86/shr-cf-combine.ll
new file mode 100644
index 000000000000000..e6424682e8d1f1d
--- /dev/null
+++ b/llvm/test/CodeGen/X86/shr-cf-combine.ll
@@ -0,0 +1,170 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 5
+; RUN: llc < %s -mtriple=x86_64-unknown-unknown | FileCheck %s
+
+; Test that when both (and D, 1) and (lshr D, 1) are needed, we use the
+; carry flag from SHR for the (and D, 1) via SBB, instead of computing it
+; separately with AND + NEG/SUB.
+;
+; This is a common pattern in GMP's reciprocal computation (mpn/x86_64/invert_limb.asm).
+; See: https://github.com/llvm/llvm-project/issues/228046
+
+define { i64, i64 } @neg_lsb(i64 %d) {
+; CHECK-LABEL: neg_lsb:
+; CHECK: # %bb.0:
+; CHECK-NEXT: xorl %eax, %eax
+; CHECK-NEXT: shrq %rdi
+; CHECK-NEXT: sbbq %rax, %rax
+; CHECK-NEXT: movq %rdi, %rdx
+; CHECK-NEXT: retq
+ %lsb = and i64 %d, 1
+ %neg = sub i64 0, %lsb
+ %half = lshr i64 %d, 1
+ %r0 = insertvalue { i64, i64 } poison, i64 %neg, 0
+ %r1 = insertvalue { i64, i64 } %r0, i64 %half, 1
+ ret { i64, i64 } %r1
+}
+
+define { i64, i64 } @sub_lsb(i64 %d, i64 %x) {
+; CHECK-LABEL: sub_lsb:
+; CHECK: # %bb.0:
+; CHECK-NEXT: movq %rsi, %rax
+; CHECK-NEXT: shrq %rdi
+; CHECK-NEXT: sbbq $0, %rax
+; CHECK-NEXT: movq %rdi, %rdx
+; CHECK-NEXT: retq
+ %lsb = and i64 %d, 1
+ %sub = sub i64 %x, %lsb
+ %half = lshr i64 %d, 1
+ %r0 = insertvalue { i64, i64 } poison, i64 %sub, 0
+ %r1 = insertvalue { i64, i64 } %r0, i64 %half, 1
+ ret { i64, i64 } %r1
+}
+
+; Test 32-bit version
+define { i32, i32 } @neg_lsb_32(i32 %d) {
+; CHECK-LABEL: neg_lsb_32:
+; CHECK: # %bb.0:
+; CHECK-NEXT: xorl %eax, %eax
+; CHECK-NEXT: shrl %edi
+; CHECK-NEXT: sbbl %eax, %eax
+; CHECK-NEXT: movl %edi, %edx
+; CHECK-NEXT: retq
+ %lsb = and i32 %d, 1
+ %neg = sub i32 0, %lsb
+ %half = lshr i32 %d, 1
+ %r0 = insertvalue { i32, i32 } poison, i32 %neg, 0
+ %r1 = insertvalue { i32, i32 } %r0, i32 %half, 1
+ ret { i32, i32 } %r1
+}
+
+define { i32, i32 } @sub_lsb_32(i32 %d, i32 %x) {
+; CHECK-LABEL: sub_lsb_32:
+; CHECK: # %bb.0:
+; CHECK-NEXT: movl %esi, %eax
+; CHECK-NEXT: shrl %edi
+; CHECK-NEXT: sbbl $0, %eax
+; CHECK-NEXT: movl %edi, %edx
+; CHECK-NEXT: retq
+ %lsb = and i32 %d, 1
+ %sub = sub i32 %x, %lsb
+ %half = lshr i32 %d, 1
+ %r0 = insertvalue { i32, i32 } poison, i32 %sub, 0
+ %r1 = insertvalue { i32, i32 } %r0, i32 %half, 1
+ ret { i32, i32 } %r1
+}
+
+; Test 16-bit version
+define { i16, i16 } @neg_lsb_16(i16 %d) {
+; CHECK-LABEL: neg_lsb_16:
+; CHECK: # %bb.0:
+; CHECK-NEXT: xorl %eax, %eax
+; CHECK-NEXT: shrw %di
+; CHECK-NEXT: sbbl %eax, %eax
+; CHECK-NEXT: # kill: def $ax killed $ax killed $eax
+; CHECK-NEXT: movl %edi, %edx
+; CHECK-NEXT: retq
+ %lsb = and i16 %d, 1
+ %neg = sub i16 0, %lsb
+ %half = lshr i16 %d, 1
+ %r0 = insertvalue { i16, i16 } poison, i16 %neg, 0
+ %r1 = insertvalue { i16, i16 } %r0, i16 %half, 1
+ ret { i16, i16 } %r1
+}
+
+define { i16, i16 } @sub_lsb_16(i16 %d, i16 %x) {
+; CHECK-LABEL: sub_lsb_16:
+; CHECK: # %bb.0:
+; CHECK-NEXT: movl %esi, %eax
+; CHECK-NEXT: shrw %di
+; CHECK-NEXT: sbbw $0, %ax
+; CHECK-NEXT: # kill: def $ax killed $ax killed $eax
+; CHECK-NEXT: movl %edi, %edx
+; CHECK-NEXT: retq
+ %lsb = and i16 %d, 1
+ %sub = sub i16 %x, %lsb
+ %half = lshr i16 %d, 1
+ %r0 = insertvalue { i16, i16 } poison, i16 %sub, 0
+ %r1 = insertvalue { i16, i16 } %r0, i16 %half, 1
+ ret { i16, i16 } %r1
+}
+
+; Test 8-bit version
+define { i8, i8 } @neg_lsb_8(i8 %d) {
+; CHECK-LABEL: neg_lsb_8:
+; CHECK: # %bb.0:
+; CHECK-NEXT: xorl %eax, %eax
+; CHECK-NEXT: shrb %dil
+; CHECK-NEXT: sbbl %eax, %eax
+; CHECK-NEXT: # kill: def $al killed $al killed $eax
+; CHECK-NEXT: movl %edi, %edx
+; CHECK-NEXT: retq
+ %lsb = and i8 %d, 1
+ %neg = sub i8 0, %lsb
+ %half = lshr i8 %d, 1
+ %r0 = insertvalue { i8, i8 } poison, i8 %neg, 0
+ %r1 = insertvalue { i8, i8 } %r0, i8 %half, 1
+ ret { i8, i8 } %r1
+}
+
+define { i8, i8 } @sub_lsb_8(i8 %d, i8 %x) {
+; CHECK-LABEL: sub_lsb_8:
+; CHECK: # %bb.0:
+; CHECK-NEXT: movl %esi, %eax
+; CHECK-NEXT: shrb %dil
+; CHECK-NEXT: sbbb $0, %al
+; CHECK-NEXT: # kill: def $al killed $al killed $eax
+; CHECK-NEXT: movl %edi, %edx
+; CHECK-NEXT: retq
+ %lsb = and i8 %d, 1
+ %sub = sub i8 %x, %lsb
+ %half = lshr i8 %d, 1
+ %r0 = insertvalue { i8, i8 } poison, i8 %sub, 0
+ %r1 = insertvalue { i8, i8 } %r0, i8 %half, 1
+ ret { i8, i8 } %r1
+}
+
+; Negative test: when only (and D, 1) is used (no corresponding srl D, 1),
+; we should NOT use SHR_FLAG (falls back to the normal AND+NEG pattern).
+define i64 @only_lsb(i64 %d) {
+; CHECK-LABEL: only_lsb:
+; CHECK: # %bb.0:
+; CHECK-NEXT: movq %rdi, %rax
+; CHECK-NEXT: andl $1, %eax
+; CHECK-NEXT: negq %rax
+; CHECK-NEXT: retq
+ %lsb = and i64 %d, 1
+ %neg = sub i64 0, %lsb
+ ret i64 %neg
+}
+
+; Negative test: when only (srl D, 1) is used (no corresponding and D, 1),
+; we should just use normal SHR.
+define i64 @only_srl(i64 %d) {
+; CHECK-LABEL: only_srl:
+; CHECK: # %bb.0:
+; CHECK-NEXT: movq %rdi, %rax
+; CHECK-NEXT: shrq %rax
+; CHECK-NEXT: retq
+ %half = lshr i64 %d, 1
+ ret i64 %half
+}
>From b9fff5ceeaa985f63348cfeb10b409c47c83780d Mon Sep 17 00:00:00 2001
From: addmisol <addmisol9 at gmail.com>
Date: Fri, 2 Oct 2026 14:11:36 +0530
Subject: [PATCH 2/3] Fix : clang code formatter
Signed-off-by: addmisol <addmisol9 at gmail.com>
---
llvm/lib/Target/X86/X86ISelLowering.cpp | 3 ++-
1 file changed, 2 insertions(+), 1 deletion(-)
diff --git a/llvm/lib/Target/X86/X86ISelLowering.cpp b/llvm/lib/Target/X86/X86ISelLowering.cpp
index ddd0199ea3f4f12..142f3d522d93234 100644
--- a/llvm/lib/Target/X86/X86ISelLowering.cpp
+++ b/llvm/lib/Target/X86/X86ISelLowering.cpp
@@ -53888,7 +53888,8 @@ static SDValue combineAddOrSubToADCOrSBB(bool IsSub, const SDLoc &DL, EVT VT,
SDValue ShrFlag = DAG.getNode(X86ISD::SHR_FLAG, DL, VTs, D, ShAmt);
// Replace uses of the original SRL with the shifted result from SHR_FLAG
- DAG.ReplaceAllUsesOfValueWith(SDValue(SrlBy1Node, 0), ShrFlag.getValue(0));
+ DAG.ReplaceAllUsesOfValueWith(SDValue(SrlBy1Node, 0),
+ ShrFlag.getValue(0));
// Use the EFLAGS output from SHR_FLAG
EFLAGS = ShrFlag.getValue(1);
>From 246401073c9868ef977045d96f068077f91bd5fa Mon Sep 17 00:00:00 2001
From: Addmisol <addmisol9 at gmail.com>
Date: Fri, 2 Oct 2026 17:39:48 +0530
Subject: [PATCH 3/3] fix: change to version 6
---
llvm/test/CodeGen/X86/shr-cf-combine.ll | 2 +-
1 file changed, 1 insertion(+), 1 deletion(-)
diff --git a/llvm/test/CodeGen/X86/shr-cf-combine.ll b/llvm/test/CodeGen/X86/shr-cf-combine.ll
index e6424682e8d1f1d..48e420cadf84890 100644
--- a/llvm/test/CodeGen/X86/shr-cf-combine.ll
+++ b/llvm/test/CodeGen/X86/shr-cf-combine.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 5
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
; RUN: llc < %s -mtriple=x86_64-unknown-unknown | FileCheck %s
; Test that when both (and D, 1) and (lshr D, 1) are needed, we use the
More information about the llvm-commits
mailing list