[llvm] [X86][CCMP] Lower select(and/or(setcc,...), T, F) as a CCMP chain (PR #207929)
Phoebe Wang via llvm-commits
llvm-commits at lists.llvm.org
Tue Jul 7 06:13:06 PDT 2026
https://github.com/phoebewang updated https://github.com/llvm/llvm-project/pull/207929
>From aceb39d0b25f1af4132ea1f07264f2b476b4cc3a Mon Sep 17 00:00:00 2001
From: Phoebe Wang <phoebe.wang at intel.com>
Date: Tue, 7 Jul 2026 05:36:53 -0700
Subject: [PATCH] [X86][CCMP] Lower select(and/or(setcc,...), T, F) as a CCMP
chain
---
llvm/lib/Target/X86/X86ISelLowering.cpp | 108 +++++++++++++++++++++++-
llvm/lib/Target/X86/X86ISelLowering.h | 3 +
llvm/test/CodeGen/X86/apx/ccmp.ll | 80 ++++++++++++++++++
3 files changed, 190 insertions(+), 1 deletion(-)
diff --git a/llvm/lib/Target/X86/X86ISelLowering.cpp b/llvm/lib/Target/X86/X86ISelLowering.cpp
index e97b4e6d84a9f..fa0c86d4cfc6b 100644
--- a/llvm/lib/Target/X86/X86ISelLowering.cpp
+++ b/llvm/lib/Target/X86/X86ISelLowering.cpp
@@ -3533,6 +3533,13 @@ bool X86TargetLowering::convertSelectOfConstantsToMath(EVT VT) const {
return true;
}
+bool X86TargetLowering::shouldNormalizeToSelectSequence(LLVMContext &, EVT VT,
+ EVT) const {
+ // With CCMP, keep and/or(setcc, setcc) trees intact so LowerSELECT can
+ // emit them as CCMP chains rather than splitting into chained selects.
+ return !(Subtarget.hasCCMP() && VT.isScalarInteger());
+}
+
bool X86TargetLowering::decomposeMulByConstant(LLVMContext &Context, EVT VT,
SDValue C) const {
// TODO: We handle scalars using custom code, but generic combining could make
@@ -25308,7 +25315,7 @@ static SDValue LowerXALUO(SDValue Op, SelectionDAG &DAG) {
static bool isX86LogicalCmp(SDValue Op) {
unsigned Opc = Op.getOpcode();
if (Opc == X86ISD::CMP || Opc == X86ISD::COMI || Opc == X86ISD::UCOMI ||
- Opc == X86ISD::FCMP)
+ Opc == X86ISD::FCMP || Opc == X86ISD::CCMP || Opc == X86ISD::CTEST)
return true;
if (Op.getResNo() == 1 &&
(Opc == X86ISD::ADD || Opc == X86ISD::SUB || Opc == X86ISD::ADC ||
@@ -25462,6 +25469,93 @@ static SDValue LowerSELECTWithCmpZero(SDValue CmpVal, SDValue LHS, SDValue RHS,
return SDValue();
}
+// Return true if Val is an integer ISD::SETCC or an AND/OR tree thereof,
+// suitable for lowering to a CCMP chain.
+static bool canEmitConjunctionForCCMP(SDValue Val) {
+ unsigned Opc = Val.getOpcode();
+ if (Opc == ISD::SETCC)
+ return Val.getOperand(0).getSimpleValueType().isInteger();
+ if ((Opc == ISD::AND || Opc == ISD::OR) && Val.hasOneUse())
+ return canEmitConjunctionForCCMP(Val.getOperand(0)) &&
+ canEmitConjunctionForCCMP(Val.getOperand(1));
+ return false;
+}
+
+// Recursively emit a CCMP chain for an AND/OR tree of integer SETCCs.
+// CCOp: incoming flags value (null for the first/root comparison)
+// Predicate: condition under which CCOp was produced (COND_INVALID at root)
+// OutCC: set to the condition code to test after the whole chain
+// Returns the flags-producing node (SUB or CCMP).
+//
+// AND(cc1, cc2): emit cc1 first; CCMP(cc2) fires when cc1 is true.
+// SrcCC = cc1, DCF forces cc2 false when cc1 is false.
+// OR(cc1, cc2): emit cc1 first; CCMP(cc2) fires when cc1 is false.
+// SrcCC = ~cc1, DCF forces cc2 true when cc1 is true.
+static SDValue emitConjunctionForCCMPRec(SDValue Val, X86::CondCode &OutCC,
+ SDValue CCOp, X86::CondCode Predicate,
+ SelectionDAG &DAG,
+ const X86Subtarget &Subtarget) {
+ SDLoc DL(Val);
+
+ if (Val.getOpcode() == ISD::SETCC) {
+ SDValue LHS = Val.getOperand(0), RHS = Val.getOperand(1);
+ ISD::CondCode CC = cast<CondCodeSDNode>(Val.getOperand(2))->get();
+ X86::CondCode X86CC = TranslateX86CC(CC, DL, /*IsFP=*/false, LHS, RHS, DAG);
+ assert(X86CC != X86::COND_INVALID);
+ OutCC = X86CC;
+
+ SDValue Flags = EmitCmp(LHS, RHS, X86CC, DL, DAG, Subtarget);
+ if (!CCOp)
+ return Flags;
+
+ SDNode *FlagsNode = Flags.getNode();
+ X86::CondCode DCFCode = X86::GetOppositeBranchCondition(X86CC);
+ SDValue CFlags = DAG.getTargetConstant(
+ X86::getCCMPCondFlagsFromCondCode(DCFCode), DL, MVT::i8);
+ SDValue SrcCC = DAG.getTargetConstant(Predicate, DL, MVT::i8);
+ return DAG.getNode(X86ISD::CCMP, DL, MVT::i32,
+ {FlagsNode->getOperand(0), FlagsNode->getOperand(1),
+ CFlags, SrcCC, CCOp});
+ }
+
+ bool IsOR = Val.getOpcode() == ISD::OR;
+ SDValue LHS = Val.getOperand(0), RHS = Val.getOperand(1);
+
+ // Emit the left subtree first (provides CCOp for the right subtree's CCMP).
+ X86::CondCode LHSCC;
+ SDValue CmpL =
+ emitConjunctionForCCMPRec(LHS, LHSCC, CCOp, Predicate, DAG, Subtarget);
+
+ // For AND: right CCMP fires when left is true, SrcCC = LHSCC.
+ // For OR: right CCMP fires when left is false, SrcCC = !LHSCC.
+ X86::CondCode NextPred =
+ IsOR ? X86::GetOppositeBranchCondition(LHSCC) : LHSCC;
+
+ SDValue CmpR =
+ emitConjunctionForCCMPRec(RHS, OutCC, CmpL, NextPred, DAG, Subtarget);
+
+ // For OR, the DCF inside the right leaf was computed as ~OutCC (forces
+ // OutCC false on skip). We need it to force OutCC TRUE on skip instead.
+ // Patch the DCF of the last-emitted CCMP node.
+ if (IsOR && CmpR.getOpcode() == X86ISD::CCMP) {
+ SDValue CFlags = DAG.getTargetConstant(
+ X86::getCCMPCondFlagsFromCondCode(OutCC), DL, MVT::i8);
+ CmpR = DAG.getNode(X86ISD::CCMP, DL, MVT::i32,
+ {CmpR.getOperand(0), CmpR.getOperand(1), CFlags,
+ CmpR.getOperand(3), CmpR.getOperand(4)});
+ }
+ return CmpR;
+}
+
+static SDValue emitConjunctionForCCMP(SDValue Val, X86::CondCode &OutCC,
+ SelectionDAG &DAG,
+ const X86Subtarget &Subtarget) {
+ if (!canEmitConjunctionForCCMP(Val))
+ return SDValue();
+ return emitConjunctionForCCMPRec(Val, OutCC, SDValue(), X86::COND_INVALID,
+ DAG, Subtarget);
+}
+
SDValue X86TargetLowering::LowerSELECT(SDValue Op, SelectionDAG &DAG) const {
bool AddTest = true;
SDValue Cond = Op.getOperand(0);
@@ -25537,6 +25631,18 @@ SDValue X86TargetLowering::LowerSELECT(SDValue Op, SelectionDAG &DAG) const {
return DAG.getNode(X86ISD::SELECTS, DL, VT, Cmp, Op1, Op2);
}
+ // Lower select(and/or(setcc,...), T, F) as a CCMP chain.
+ if (Subtarget.hasCCMP() && !VT.isVector() &&
+ (Cond.getOpcode() == ISD::AND || Cond.getOpcode() == ISD::OR)) {
+ X86::CondCode CCMPOutCC;
+ if (SDValue Flags =
+ emitConjunctionForCCMP(Cond, CCMPOutCC, DAG, Subtarget)) {
+ SDValue X86CC = DAG.getTargetConstant(CCMPOutCC, DL, MVT::i8);
+ Cond = DAG.getNode(X86ISD::SETCC, DL, MVT::i8, X86CC, Flags);
+ AddTest = false;
+ }
+ }
+
if (Cond.getOpcode() == ISD::SETCC &&
!isSoftF16(Cond.getOperand(0).getSimpleValueType(), Subtarget)) {
if (SDValue NewCond = LowerSETCC(Cond, DAG)) {
diff --git a/llvm/lib/Target/X86/X86ISelLowering.h b/llvm/lib/Target/X86/X86ISelLowering.h
index 5283c377b97c0..02cd89c2c9881 100644
--- a/llvm/lib/Target/X86/X86ISelLowering.h
+++ b/llvm/lib/Target/X86/X86ISelLowering.h
@@ -566,6 +566,9 @@ namespace llvm {
bool convertSelectOfConstantsToMath(EVT VT) const override;
+ bool shouldNormalizeToSelectSequence(LLVMContext &Context, EVT VT,
+ EVT CCVT) const override;
+
bool decomposeMulByConstant(LLVMContext &Context, EVT VT,
SDValue C) const override;
diff --git a/llvm/test/CodeGen/X86/apx/ccmp.ll b/llvm/test/CodeGen/X86/apx/ccmp.ll
index 94cbb7786721c..cc6cb6b78fcf1 100644
--- a/llvm/test/CodeGen/X86/apx/ccmp.ll
+++ b/llvm/test/CodeGen/X86/apx/ccmp.ll
@@ -2098,5 +2098,85 @@ define i32 @test_or_fp_int(double %a, double %b, i32 %c, i32 %d) {
ret i32 %ext
}
+; (b != d) && (a < c): AND of two comparisons - fold chained CMOV into CCMP+CMOV
+define i32 @ccmp_cmov_and(i32 %a, i32 %b, i32 %c, i32 %d) {
+; CHECK-LABEL: ccmp_cmov_and:
+; CHECK: # %bb.0: # %entry
+; CHECK-NEXT: movl %edi, %eax # encoding: [0x89,0xf8]
+; CHECK-NEXT: cmpl %ecx, %esi # encoding: [0x39,0xce]
+; CHECK-NEXT: ccmpnel {dfv=} %edx, %edi # encoding: [0x62,0xf4,0x04,0x05,0x39,0xd7]
+; CHECK-NEXT: cmovll %esi, %eax # encoding: [0x0f,0x4c,0xc6]
+; CHECK-NEXT: retq # encoding: [0xc3]
+;
+; NDD-LABEL: ccmp_cmov_and:
+; NDD: # %bb.0: # %entry
+; NDD-NEXT: cmpl %ecx, %esi # encoding: [0x39,0xce]
+; NDD-NEXT: ccmpnel {dfv=} %edx, %edi # encoding: [0x62,0xf4,0x04,0x05,0x39,0xd7]
+; NDD-NEXT: cmovll %esi, %edi, %eax # encoding: [0x62,0xf4,0x7c,0x18,0x4c,0xfe]
+; NDD-NEXT: retq # encoding: [0xc3]
+entry:
+ %cmp1 = icmp ne i32 %b, %d
+ %cmp2 = icmp slt i32 %a, %c
+ %and = and i1 %cmp1, %cmp2
+ %sel = select i1 %and, i32 %b, i32 %a
+ ret i32 %sel
+}
+
+; (b != d && a < c) || (a > d): AND+OR - fold chained CMOVs into CCMP+CCMP+CMOV
+define i32 @ccmp_cmov_and_or(i32 %a, i32 %b, i32 %c, i32 %d) {
+; CHECK-LABEL: ccmp_cmov_and_or:
+; CHECK: # %bb.0: # %entry
+; CHECK-NEXT: movl %edi, %eax # encoding: [0x89,0xf8]
+; CHECK-NEXT: cmpl %ecx, %esi # encoding: [0x39,0xce]
+; CHECK-NEXT: ccmpnel {dfv=} %edx, %edi # encoding: [0x62,0xf4,0x04,0x05,0x39,0xd7]
+; CHECK-NEXT: ccmpgel {dfv=} %ecx, %edi # encoding: [0x62,0xf4,0x04,0x0d,0x39,0xcf]
+; CHECK-NEXT: cmovgl %esi, %eax # encoding: [0x0f,0x4f,0xc6]
+; CHECK-NEXT: retq # encoding: [0xc3]
+;
+; NDD-LABEL: ccmp_cmov_and_or:
+; NDD: # %bb.0: # %entry
+; NDD-NEXT: cmpl %ecx, %esi # encoding: [0x39,0xce]
+; NDD-NEXT: ccmpnel {dfv=} %edx, %edi # encoding: [0x62,0xf4,0x04,0x05,0x39,0xd7]
+; NDD-NEXT: ccmpgel {dfv=} %ecx, %edi # encoding: [0x62,0xf4,0x04,0x0d,0x39,0xcf]
+; NDD-NEXT: cmovgl %esi, %edi, %eax # encoding: [0x62,0xf4,0x7c,0x18,0x4f,0xfe]
+; NDD-NEXT: retq # encoding: [0xc3]
+entry:
+ %cmp1 = icmp ne i32 %b, %d
+ %cmp2 = icmp slt i32 %a, %c
+ %and = and i1 %cmp1, %cmp2
+ %cmp3 = icmp sgt i32 %a, %d
+ %or = or i1 %and, %cmp3
+ %sel = select i1 %or, i32 %b, i32 %a
+ ret i32 %sel
+}
+
+; (a > d) || (a < c && b != d): AND+OR with outer OR first
+define i32 @ccmp_cmov_and_or_c(i32 %a, i32 %b, i32 %c, i32 %d) {
+; CHECK-LABEL: ccmp_cmov_and_or_c:
+; CHECK: # %bb.0: # %entry
+; CHECK-NEXT: movl %edi, %eax # encoding: [0x89,0xf8]
+; CHECK-NEXT: cmpl %ecx, %edi # encoding: [0x39,0xcf]
+; CHECK-NEXT: ccmplel {dfv=} %edx, %edi # encoding: [0x62,0xf4,0x04,0x0e,0x39,0xd7]
+; CHECK-NEXT: ccmpll {dfv=} %ecx, %esi # encoding: [0x62,0xf4,0x04,0x0c,0x39,0xce]
+; CHECK-NEXT: cmovnel %esi, %eax # encoding: [0x0f,0x45,0xc6]
+; CHECK-NEXT: retq # encoding: [0xc3]
+;
+; NDD-LABEL: ccmp_cmov_and_or_c:
+; NDD: # %bb.0: # %entry
+; NDD-NEXT: cmpl %ecx, %edi # encoding: [0x39,0xcf]
+; NDD-NEXT: ccmplel {dfv=} %edx, %edi # encoding: [0x62,0xf4,0x04,0x0e,0x39,0xd7]
+; NDD-NEXT: ccmpll {dfv=} %ecx, %esi # encoding: [0x62,0xf4,0x04,0x0c,0x39,0xce]
+; NDD-NEXT: cmovnel %esi, %edi, %eax # encoding: [0x62,0xf4,0x7c,0x18,0x45,0xfe]
+; NDD-NEXT: retq # encoding: [0xc3]
+entry:
+ %cmp1 = icmp ne i32 %b, %d
+ %cmp2 = icmp slt i32 %a, %c
+ %and = and i1 %cmp2, %cmp1
+ %cmp3 = icmp sgt i32 %a, %d
+ %or = or i1 %cmp3, %and
+ %sel = select i1 %or, i32 %b, i32 %a
+ ret i32 %sel
+}
+
declare dso_local void @foo(...)
declare {i64, i1} @llvm.ssub.with.overflow.i64(i64, i64) nounwind readnone
More information about the llvm-commits
mailing list