[llvm] [ConstantTime][LLVM] Add llvm.ct.select intrinsic with generic SelectionDAG lowering (PR #166702)
Akshay K via llvm-commits
llvm-commits at lists.llvm.org
Wed Sep 30 07:46:02 PDT 2026
https://github.com/kumarak updated https://github.com/llvm/llvm-project/pull/166702
>From 4abdc461cf80cd769d3980334fd73cb33a83846d Mon Sep 17 00:00:00 2001
From: wizardengineer <juliuswoosebert at gmail.com>
Date: Wed, 5 Nov 2025 10:51:08 -0500
Subject: [PATCH 01/12] [ConstantTime][LLVM] Add llvm.ct.select intrinsic with
generic SelectionDAG lowering
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
[LLVM][CodeGen] Improve CTSELECT fallback lowering and target support modeling (#179395)
This pull request refactors and improves the **fallback handling** of
constant-time select (CTSELECT) in LLVM’s code generation
infrastructure. The changes clarify semantics, simplify
target-capability checks, and improve the correctness and
maintainability of fallback lowering, without changing the intended
constant-time guarantees.
- **CTSELECT semantics**
- Clarified documentation for the `CTSELECT` node to explicitly describe
its operands and its role as the lowering target for the constant-time
select intrinsic.
- **TargetLowering cleanup**
- Removed CTSELECT-specific entries from `SelectSupportKind`.
- Introduced a dedicated `isCtSelectSupported(EVT)` hook to cleanly
separate ISA select support from constant-time guarantees.
- Updated intrinsic lowering to use this new capability check.
- **Fallback implementation improvements**
- Updated the generic SelectionDAG fallback lowering to use the
canonical bitwise formulation:
```
F ^ ((T ^ F) & Mask)
```
- Simplified handling of vector splats and bitcasting in the fallback
path.
- **DAGCombiner refactor**
- Renamed and clarified CTSELECT-related DAGCombiner code paths for
improved readability.
- **Miscellaneous cleanups**
- Improved documentation headers for constant-time intrinsics.
- Removed unused flags and minor nits uncovered during review.
These changes primarily affect the **generic fallback path** used when
targets do not provide specialized CTSELECT lowering. Target-specific
implementations are handled in follow-up PRs
---
llvm/include/llvm/CodeGen/ISDOpcodes.h | 4 +
llvm/include/llvm/CodeGen/SelectionDAG.h | 7 +
llvm/include/llvm/CodeGen/TargetLowering.h | 7 +-
llvm/include/llvm/IR/Intrinsics.td | 10 +-
.../include/llvm/Target/TargetSelectionDAG.td | 6 +
llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp | 74 +
llvm/lib/CodeGen/SelectionDAG/LegalizeDAG.cpp | 78 +-
.../SelectionDAG/LegalizeFloatTypes.cpp | 19 +
.../SelectionDAG/LegalizeIntegerTypes.cpp | 20 +
llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h | 7 +-
.../SelectionDAG/LegalizeTypesGeneric.cpp | 6 +
.../SelectionDAG/LegalizeVectorTypes.cpp | 13 +
.../SelectionDAG/SelectionDAGBuilder.cpp | 106 ++
.../SelectionDAG/SelectionDAGBuilder.h | 3 +
.../SelectionDAG/SelectionDAGDumper.cpp | 1 +
llvm/lib/Target/AArch64/AArch64ISelLowering.h | 2 +
llvm/lib/Target/ARM/ARMISelLowering.h | 2 +
llvm/lib/Target/X86/X86ISelLowering.h | 2 +
llvm/test/CodeGen/RISCV/ctselect-fallback.ll | 754 ++++++++
llvm/test/CodeGen/X86/ctselect.ll | 1509 +++++++++++++++++
20 files changed, 2623 insertions(+), 7 deletions(-)
create mode 100644 llvm/test/CodeGen/RISCV/ctselect-fallback.ll
create mode 100644 llvm/test/CodeGen/X86/ctselect.ll
diff --git a/llvm/include/llvm/CodeGen/ISDOpcodes.h b/llvm/include/llvm/CodeGen/ISDOpcodes.h
index 9c3757203299e..89fe9e07e3583 100644
--- a/llvm/include/llvm/CodeGen/ISDOpcodes.h
+++ b/llvm/include/llvm/CodeGen/ISDOpcodes.h
@@ -805,6 +805,10 @@ enum NodeType {
/// i1 then the high bits must conform to getBooleanContents.
SELECT,
+ /// CTSELECT(Cond, TrueVal, FalseVal). Cond is i1 and the value operands must
+ /// have the same type. Used to lower the constant-time select intrinsic.
+ CTSELECT,
+
/// Select with a vector condition (op #0) and two vector operands (ops #1
/// and #2), returning a vector result. All vectors have the same length.
/// Much like the scalar select and setcc, each bit in the condition selects
diff --git a/llvm/include/llvm/CodeGen/SelectionDAG.h b/llvm/include/llvm/CodeGen/SelectionDAG.h
index 5dd81fc7c5a94..47b27aaf75ed9 100644
--- a/llvm/include/llvm/CodeGen/SelectionDAG.h
+++ b/llvm/include/llvm/CodeGen/SelectionDAG.h
@@ -1360,6 +1360,13 @@ class SelectionDAG {
return getNode(Opcode, DL, VT, Cond, LHS, RHS, Flags);
}
+ SDValue getCTSelect(const SDLoc &DL, EVT VT, SDValue Cond, SDValue LHS,
+ SDValue RHS, SDNodeFlags Flags = SDNodeFlags()) {
+ assert(LHS.getValueType() == VT && RHS.getValueType() == VT &&
+ "Cannot use select on differing types");
+ return getNode(ISD::CTSELECT, DL, VT, Cond, LHS, RHS, Flags);
+ }
+
/// Helper function to make it easier to build SelectCC's if you just have an
/// ISD::CondCode instead of an SDValue.
SDValue getSelectCC(const SDLoc &DL, SDValue LHS, SDValue RHS, SDValue True,
diff --git a/llvm/include/llvm/CodeGen/TargetLowering.h b/llvm/include/llvm/CodeGen/TargetLowering.h
index 4006ad3bda5d6..5dcb5ffbf08b1 100644
--- a/llvm/include/llvm/CodeGen/TargetLowering.h
+++ b/llvm/include/llvm/CodeGen/TargetLowering.h
@@ -506,9 +506,10 @@ class LLVM_ABI TargetLoweringBase {
MachineMemOperand::Flags
getVPIntrinsicMemOperandFlags(const VPIntrinsic &VPIntrin) const;
- virtual bool isSelectSupported(SelectSupportKind /*kind*/) const {
- return true;
- }
+ virtual bool isSelectSupported(SelectSupportKind kind) const { return true; }
+
+ /// Return true if the target has custom lowering for constant-time select.
+ virtual bool isCtSelectSupported(EVT VT) const { return false; }
/// Return true if the @llvm.get.active.lane.mask intrinsic should be expanded
/// using generic code in SelectionDAGBuilder.
diff --git a/llvm/include/llvm/IR/Intrinsics.td b/llvm/include/llvm/IR/Intrinsics.td
index a376d4aa12af1..cbfc08de904cd 100644
--- a/llvm/include/llvm/IR/Intrinsics.td
+++ b/llvm/include/llvm/IR/Intrinsics.td
@@ -2066,8 +2066,16 @@ def int_coro_subfn_addr : DefaultAttrsIntrinsic<
[IntrReadMem, IntrArgMemOnly, ReadOnly<ArgIndex<0>>,
NoCapture<ArgIndex<0>>]>;
-///===------------------------- Other Intrinsics --------------------------===//
+///===---------------------- Constant Time Intrinsics ----------------------===//
//
+// Intrinsic to support constant time select
+def int_ct_select
+ : DefaultAttrsIntrinsic<[llvm_any_ty],
+ [llvm_i1_ty, LLVMMatchType<0>, LLVMMatchType<0>],
+ [IntrWriteMem, IntrWillReturn, NoUndef<RetIndex>]>;
+
+
+////===-------------------------- Other Intrinsics --------------------------===//
// TODO: We should introduce a new memory kind fo traps (and other side effects
// we only model to keep things alive).
def int_trap : Intrinsic<[], [],
diff --git a/llvm/include/llvm/Target/TargetSelectionDAG.td b/llvm/include/llvm/Target/TargetSelectionDAG.td
index b0fd9c733e5b6..94c8fddd9aed5 100644
--- a/llvm/include/llvm/Target/TargetSelectionDAG.td
+++ b/llvm/include/llvm/Target/TargetSelectionDAG.td
@@ -221,6 +221,11 @@ def SDTSelect : SDTypeProfile<1, 3, [ // select
SDTCisInt<1>, SDTCisSameAs<0, 2>, SDTCisSameAs<2, 3>
]>;
+def SDTCtSelect
+ : SDTypeProfile<1, 3,
+ [ // ctselect
+ SDTCisInt<1>, SDTCisSameAs<0, 2>, SDTCisSameAs<2, 3>]>;
+
def SDTVSelect : SDTypeProfile<1, 3, [ // vselect
SDTCisVec<0>, SDTCisInt<1>, SDTCisSameAs<0, 2>, SDTCisSameAs<2, 3>, SDTCisSameNumEltsAs<0, 1>
]>;
@@ -785,6 +790,7 @@ def reset_fpmode : SDNode<"ISD::RESET_FPMODE", SDTNone, [SDNPHasChain]>;
def setcc : SDNode<"ISD::SETCC" , SDTSetCC>;
def select : SDNode<"ISD::SELECT" , SDTSelect>;
+def ctselect : SDNode<"ISD::CTSELECT", SDTCtSelect>;
def vselect : SDNode<"ISD::VSELECT" , SDTVSelect>;
def selectcc : SDNode<"ISD::SELECT_CC" , SDTSelectCC>;
diff --git a/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp b/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp
index a829d7a34d5a6..075555e009097 100644
--- a/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp
@@ -482,6 +482,9 @@ namespace {
SDValue visitCTTZ_ZERO_POISON(SDNode *N);
SDValue visitCTPOP(SDNode *N);
SDValue visitSELECT(SDNode *N);
+ // ISD::CTSELECT - Constant-Time SELECT (not related to CT in
+ // CTPOP/CTLZ/CTTZ where CT means "count").
+ SDValue visitCT_SELECT(SDNode *N);
SDValue visitVSELECT(SDNode *N);
SDValue visitSELECT_CC(SDNode *N);
SDValue visitSETCC(SDNode *N);
@@ -2037,6 +2040,7 @@ SDValue DAGCombiner::visit(SDNode *N) {
case ISD::CTTZ_ZERO_POISON: return visitCTTZ_ZERO_POISON(N);
case ISD::CTPOP: return visitCTPOP(N);
case ISD::SELECT: return visitSELECT(N);
+ case ISD::CTSELECT: return visitCT_SELECT(N);
case ISD::VSELECT: return visitVSELECT(N);
case ISD::SELECT_CC: return visitSELECT_CC(N);
case ISD::SETCC: return visitSETCC(N);
@@ -13682,6 +13686,76 @@ SDValue DAGCombiner::visitSELECT(SDNode *N) {
return SDValue();
}
+// Keep CTSELECT combines deliberately conservative to preserve constant-time
+// intent across generic DAG combines. We only accept:
+// - canonicalization of negated conditions (flip true/false operands), and
+// - i1 CTSELECT nesting merges via AND/OR that keep the result as CTSELECT.
+// Broader rewrites should be done in target-specific lowering when stronger
+// guarantees about legality and constant-time preservation are available.
+SDValue DAGCombiner::visitCT_SELECT(SDNode *N) {
+ SDValue N0 = N->getOperand(0);
+ SDValue N1 = N->getOperand(1);
+ SDValue N2 = N->getOperand(2);
+ EVT VT = N->getValueType(0);
+ EVT VT0 = N0.getValueType();
+ SDLoc DL(N);
+ SDNodeFlags Flags = N->getFlags();
+
+ // ctselect (not Cond), N1, N2 -> ctselect Cond, N2, N1
+ // This is a CT-safe canonicalization: flip negated condition by swapping
+ // arms. extractBooleanFlip only matches boolean xor-with-1, so this preserves
+ // dataflow semantics and does not introduce data-dependent control flow.
+ if (SDValue F = extractBooleanFlip(N0, DAG, TLI, false)) {
+ SDValue SelectOp = DAG.getNode(ISD::CTSELECT, DL, VT, F, N2, N1);
+ SelectOp->setFlags(Flags);
+ return SelectOp;
+ }
+
+ if (VT0 == MVT::i1) {
+ // Nested CTSELECT merging optimizations for i1 conditions.
+ // These are CT-safe because:
+ // 1. AND/OR are bitwise operations that execute in constant time
+ // 2. The optimization combines two sequential CTSELECTs into one,
+ // reducing the total number of constant-time operations without
+ // changing semantics
+ // 3. No data-dependent branches or memory accesses are introduced
+ //
+ // ctselect C0, (ctselect C1, X, Y), Y -> ctselect (C0 & C1), X, Y
+ // Semantic equivalence: If C0 is true, evaluate inner select (C1 ? X :
+ // Y). If C0 is false, choose Y. This is equivalent to (C0 && C1) ? X : Y.
+ if (N1->getOpcode() == ISD::CTSELECT && N1->hasOneUse()) {
+ SDValue N1_0 = N1->getOperand(0);
+ SDValue N1_1 = N1->getOperand(1);
+ SDValue N1_2 = N1->getOperand(2);
+ if (N1_2 == N2 && N0.getValueType() == N1_0.getValueType()) {
+ SDValue And = DAG.getNode(ISD::AND, DL, N0.getValueType(), N0, N1_0);
+ SDValue SelectOp =
+ DAG.getNode(ISD::CTSELECT, DL, N1.getValueType(), And, N1_1, N2);
+ SelectOp->setFlags(Flags);
+ return SelectOp;
+ }
+ }
+
+ // ctselect C0, X, (ctselect C1, X, Y) -> ctselect (C0 | C1), X, Y
+ // Semantic equivalence: If C0 is true, choose X. If C0 is false, evaluate
+ // inner select (C1 ? X : Y). This is equivalent to (C0 || C1) ? X : Y.
+ if (N2->getOpcode() == ISD::CTSELECT && N2->hasOneUse()) {
+ SDValue N2_0 = N2->getOperand(0);
+ SDValue N2_1 = N2->getOperand(1);
+ SDValue N2_2 = N2->getOperand(2);
+ if (N2_1 == N1 && N0.getValueType() == N2_0.getValueType()) {
+ SDValue Or = DAG.getNode(ISD::OR, DL, N0.getValueType(), N0, N2_0);
+ SDValue SelectOp =
+ DAG.getNode(ISD::CTSELECT, DL, N1.getValueType(), Or, N1, N2_2);
+ SelectOp->setFlags(Flags);
+ return SelectOp;
+ }
+ }
+ }
+
+ return SDValue();
+}
+
// This function assumes all the vselect's arguments are CONCAT_VECTOR
// nodes and that the condition is a BV of ConstantSDNodes (or undefs).
static SDValue ConvertSelectToConcatVector(SDNode *N, SelectionDAG &DAG) {
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeDAG.cpp b/llvm/lib/CodeGen/SelectionDAG/LegalizeDAG.cpp
index 29f1335563166..4025f82e06478 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeDAG.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeDAG.cpp
@@ -4336,6 +4336,78 @@ bool SelectionDAGLegalize::ExpandNode(SDNode *Node) {
}
Results.push_back(Tmp1);
break;
+ case ISD::CTSELECT: {
+ Tmp1 = Node->getOperand(0);
+ Tmp2 = Node->getOperand(1);
+ Tmp3 = Node->getOperand(2);
+ EVT VT = Tmp2.getValueType();
+ if (VT.isVector()) {
+ // Constant-time vector blending using pattern F ^ ((T ^ F) & Mask)
+ // where Mask = broadcast(i1 ? -1 : 0) to match vector element width.
+ //
+ // This formulation uses only XOR and AND operations, avoiding branches
+ // that would leak timing information. It's equivalent to:
+ // Mask==0xFF: F ^ ((T ^ F) & 0xFF) = F ^ (T ^ F) = T
+ // Mask==0x00: F ^ ((T ^ F) & 0x00) = F ^ 0 = F
+
+ EVT IntVT = VT;
+ SDValue T = Tmp2; // True value
+ SDValue F = Tmp3; // False value
+
+ // Step 1: Handle floating-point vectors by bitcasting to integer
+ if (VT.isFloatingPoint()) {
+ IntVT = EVT::getVectorVT(
+ *DAG.getContext(),
+ EVT::getIntegerVT(*DAG.getContext(), VT.getScalarSizeInBits()),
+ VT.getVectorElementCount());
+ T = DAG.getNode(ISD::BITCAST, dl, IntVT, T);
+ F = DAG.getNode(ISD::BITCAST, dl, IntVT, F);
+ }
+
+ // Step 2: Broadcast the i1 condition to a vector of i1s
+ // Creates [cond, cond, cond, ...] with i1 elements
+ EVT VecI1Ty = EVT::getVectorVT(*DAG.getContext(), MVT::i1,
+ VT.getVectorNumElements());
+ SDValue VecCond = DAG.getSplatBuildVector(VecI1Ty, dl, Tmp1);
+
+ // Step 3: Sign-extend i1 vector to get all-bits mask
+ // true (i1=1) -> 0xFFFFFFFF..., false (i1=0) -> 0x00000000
+ // Sign extension is constant-time: pure arithmetic, no branches
+ SDValue Mask = DAG.getNode(ISD::SIGN_EXTEND, dl, IntVT, VecCond);
+
+ // Step 4: Compute constant-time blend: F ^ ((T ^ F) & Mask)
+ // All operations (XOR, AND) execute in constant time
+ SDValue TXorF = DAG.getNode(ISD::XOR, dl, IntVT, T, F);
+ SDValue MaskedDiff = DAG.getNode(ISD::AND, dl, IntVT, TXorF, Mask);
+ Tmp1 = DAG.getNode(ISD::XOR, dl, IntVT, F, MaskedDiff);
+
+ // Step 5: Bitcast back to original floating-point type if needed
+ if (VT.isFloatingPoint()) {
+ Tmp1 = DAG.getNode(ISD::BITCAST, dl, VT, Tmp1);
+ }
+
+ Tmp1->setFlags(Node->getFlags());
+ } else if (VT.isFloatingPoint()) {
+ EVT IntegerVT = EVT::getIntegerVT(*DAG.getContext(), VT.getSizeInBits());
+ Tmp2 = DAG.getBitcast(IntegerVT, Tmp2);
+ Tmp3 = DAG.getBitcast(IntegerVT, Tmp3);
+ Tmp1 = DAG.getBitcast(VT, DAG.getCTSelect(dl, IntegerVT, Tmp1, Tmp2, Tmp3,
+ Node->getFlags()));
+ } else {
+ assert(VT.isInteger());
+ EVT HalfVT = VT.getHalfSizedIntegerVT(*DAG.getContext());
+ auto [Tmp2Lo, Tmp2Hi] = DAG.SplitScalar(Tmp2, dl, HalfVT, HalfVT);
+ auto [Tmp3Lo, Tmp3Hi] = DAG.SplitScalar(Tmp3, dl, HalfVT, HalfVT);
+ SDValue ResLo =
+ DAG.getCTSelect(dl, HalfVT, Tmp1, Tmp2Lo, Tmp3Lo, Node->getFlags());
+ SDValue ResHi =
+ DAG.getCTSelect(dl, HalfVT, Tmp1, Tmp2Hi, Tmp3Hi, Node->getFlags());
+ Tmp1 = DAG.getNode(ISD::BUILD_PAIR, dl, VT, ResLo, ResHi);
+ Tmp1->setFlags(Node->getFlags());
+ }
+ Results.push_back(Tmp1);
+ break;
+ }
case ISD::BR_JT: {
SDValue Chain = Node->getOperand(0);
SDValue Table = Node->getOperand(1);
@@ -5661,7 +5733,8 @@ void SelectionDAGLegalize::PromoteNode(SDNode *Node) {
Results.push_back(DAG.getNode(ISD::TRUNCATE, dl, OVT, Tmp2));
break;
}
- case ISD::SELECT: {
+ case ISD::SELECT:
+ case ISD::CTSELECT: {
unsigned ExtOp, TruncOp;
if (Node->getValueType(0).isVector() ||
Node->getValueType(0).getSizeInBits() == NVT.getSizeInBits()) {
@@ -5679,7 +5752,8 @@ void SelectionDAGLegalize::PromoteNode(SDNode *Node) {
Tmp2 = DAG.getNode(ExtOp, dl, NVT, Node->getOperand(1));
Tmp3 = DAG.getNode(ExtOp, dl, NVT, Node->getOperand(2));
// Perform the larger operation, then round down.
- Tmp1 = DAG.getSelect(dl, NVT, Tmp1, Tmp2, Tmp3);
+ Tmp1 = DAG.getNode(Node->getOpcode(), dl, NVT, Tmp1, Tmp2, Tmp3);
+ Tmp1->setFlags(Node->getFlags());
if (TruncOp != ISD::FP_ROUND)
Tmp1 = DAG.getNode(TruncOp, dl, Node->getValueType(0), Tmp1);
else
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeFloatTypes.cpp b/llvm/lib/CodeGen/SelectionDAG/LegalizeFloatTypes.cpp
index b18da75e46bae..4791130d75621 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeFloatTypes.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeFloatTypes.cpp
@@ -161,6 +161,7 @@ void DAGTypeLegalizer::SoftenFloatResult(SDNode *N, unsigned ResNo) {
case ISD::ATOMIC_LOAD: R = SoftenFloatRes_ATOMIC_LOAD(N); break;
case ISD::ATOMIC_SWAP: R = BitcastToInt_ATOMIC_SWAP(N); break;
case ISD::SELECT: R = SoftenFloatRes_SELECT(N); break;
+ case ISD::CTSELECT: R = SoftenFloatRes_CTSELECT(N); break;
case ISD::SELECT_CC: R = SoftenFloatRes_SELECT_CC(N); break;
case ISD::FREEZE: R = SoftenFloatRes_FREEZE(N); break;
case ISD::STRICT_SINT_TO_FP:
@@ -966,6 +967,13 @@ SDValue DAGTypeLegalizer::SoftenFloatRes_SELECT(SDNode *N) {
LHS.getValueType(), N->getOperand(0), LHS, RHS);
}
+SDValue DAGTypeLegalizer::SoftenFloatRes_CTSELECT(SDNode *N) {
+ SDValue LHS = GetSoftenedFloat(N->getOperand(1));
+ SDValue RHS = GetSoftenedFloat(N->getOperand(2));
+ return DAG.getCTSelect(SDLoc(N), LHS.getValueType(), N->getOperand(0), LHS,
+ RHS);
+}
+
SDValue DAGTypeLegalizer::SoftenFloatRes_SELECT_CC(SDNode *N) {
SDValue LHS = GetSoftenedFloat(N->getOperand(2));
SDValue RHS = GetSoftenedFloat(N->getOperand(3));
@@ -1529,6 +1537,7 @@ void DAGTypeLegalizer::ExpandFloatResult(SDNode *N, unsigned ResNo) {
case ISD::POISON:
case ISD::UNDEF: SplitRes_UNDEF(N, Lo, Hi); break;
case ISD::SELECT: SplitRes_Select(N, Lo, Hi); break;
+ case ISD::CTSELECT: SplitRes_CTSELECT(N, Lo, Hi); break;
case ISD::SELECT_CC: SplitRes_SELECT_CC(N, Lo, Hi); break;
case ISD::MERGE_VALUES: ExpandRes_MERGE_VALUES(N, ResNo, Lo, Hi); break;
@@ -2581,6 +2590,9 @@ void DAGTypeLegalizer::SoftPromoteHalfResult(SDNode *N, unsigned ResNo) {
R = SoftPromoteHalfRes_ATOMIC_LOAD(N);
break;
case ISD::SELECT: R = SoftPromoteHalfRes_SELECT(N); break;
+ case ISD::CTSELECT:
+ R = SoftPromoteHalfRes_CTSELECT(N);
+ break;
case ISD::SELECT_CC: R = SoftPromoteHalfRes_SELECT_CC(N); break;
case ISD::STRICT_SINT_TO_FP:
case ISD::STRICT_UINT_TO_FP:
@@ -2850,6 +2862,13 @@ SDValue DAGTypeLegalizer::SoftPromoteHalfRes_SELECT(SDNode *N) {
N->getFlags());
}
+SDValue DAGTypeLegalizer::SoftPromoteHalfRes_CTSELECT(SDNode *N) {
+ SDValue Op1 = GetSoftPromotedHalf(N->getOperand(1));
+ SDValue Op2 = GetSoftPromotedHalf(N->getOperand(2));
+ return DAG.getCTSelect(SDLoc(N), Op1.getValueType(), N->getOperand(0), Op1,
+ Op2);
+}
+
SDValue DAGTypeLegalizer::SoftPromoteHalfRes_SELECT_CC(SDNode *N) {
SDValue Op2 = GetSoftPromotedHalf(N->getOperand(2));
SDValue Op3 = GetSoftPromotedHalf(N->getOperand(3));
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeIntegerTypes.cpp b/llvm/lib/CodeGen/SelectionDAG/LegalizeIntegerTypes.cpp
index 43e41ed39ca1d..d3b419f32d8a9 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeIntegerTypes.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeIntegerTypes.cpp
@@ -91,6 +91,7 @@ void DAGTypeLegalizer::PromoteIntegerResult(SDNode *N, unsigned ResNo) {
Res = PromoteIntRes_VECTOR_COMPRESS(N);
break;
case ISD::SELECT:
+ case ISD::CTSELECT:
case ISD::VSELECT:
case ISD::VP_MERGE:
Res = PromoteIntRes_Select(N);
@@ -1995,6 +1996,9 @@ bool DAGTypeLegalizer::PromoteIntegerOperand(SDNode *N, unsigned OpNo) {
break;
case ISD::VSELECT:
case ISD::SELECT: Res = PromoteIntOp_SELECT(N, OpNo); break;
+ case ISD::CTSELECT:
+ Res = PromoteIntOp_CTSELECT(N, OpNo);
+ break;
case ISD::SELECT_CC: Res = PromoteIntOp_SELECT_CC(N, OpNo); break;
case ISD::SETCC: Res = PromoteIntOp_SETCC(N, OpNo); break;
case ISD::SIGN_EXTEND: Res = PromoteIntOp_SIGN_EXTEND(N); break;
@@ -2405,6 +2409,19 @@ SDValue DAGTypeLegalizer::PromoteIntOp_SELECT(SDNode *N, unsigned OpNo) {
N->getOperand(2)), 0);
}
+SDValue DAGTypeLegalizer::PromoteIntOp_CTSELECT(SDNode *N, unsigned OpNo) {
+ assert(OpNo == 0 && "Only know how to promote the condition!");
+ SDValue Cond = N->getOperand(0);
+ EVT OpTy = N->getOperand(1).getValueType();
+
+ // Promote all the way up to the canonical SetCC type.
+ EVT OpVT = N->getOpcode() == ISD::CTSELECT ? OpTy.getScalarType() : OpTy;
+ Cond = PromoteTargetBoolean(Cond, OpVT);
+
+ return SDValue(
+ DAG.UpdateNodeOperands(N, Cond, N->getOperand(1), N->getOperand(2)), 0);
+}
+
SDValue DAGTypeLegalizer::PromoteIntOp_SELECT_CC(SDNode *N, unsigned OpNo) {
assert(OpNo == 0 && "Don't know how to promote this operand!");
@@ -3013,6 +3030,9 @@ void DAGTypeLegalizer::ExpandIntegerResult(SDNode *N, unsigned ResNo) {
case ISD::ARITH_FENCE: SplitRes_ARITH_FENCE(N, Lo, Hi); break;
case ISD::MERGE_VALUES: SplitRes_MERGE_VALUES(N, ResNo, Lo, Hi); break;
case ISD::SELECT: SplitRes_Select(N, Lo, Hi); break;
+ case ISD::CTSELECT:
+ SplitRes_CTSELECT(N, Lo, Hi);
+ break;
case ISD::SELECT_CC: SplitRes_SELECT_CC(N, Lo, Hi); break;
case ISD::POISON:
case ISD::UNDEF: SplitRes_UNDEF(N, Lo, Hi); break;
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h b/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h
index f1458b22532a5..8036ac9b040ee 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h
@@ -382,6 +382,7 @@ class LLVM_LIBRARY_VISIBILITY DAGTypeLegalizer {
SDValue PromoteIntOp_CONCAT_VECTORS(SDNode *N);
SDValue PromoteIntOp_ScalarOp(SDNode *N);
SDValue PromoteIntOp_SELECT(SDNode *N, unsigned OpNo);
+ SDValue PromoteIntOp_CTSELECT(SDNode *N, unsigned OpNo);
SDValue PromoteIntOp_SELECT_CC(SDNode *N, unsigned OpNo);
SDValue PromoteIntOp_SETCC(SDNode *N, unsigned OpNo);
SDValue PromoteIntOp_Shift(SDNode *N);
@@ -622,6 +623,7 @@ class LLVM_LIBRARY_VISIBILITY DAGTypeLegalizer {
SDValue SoftenFloatRes_LOAD(SDNode *N);
SDValue SoftenFloatRes_ATOMIC_LOAD(SDNode *N);
SDValue SoftenFloatRes_SELECT(SDNode *N);
+ SDValue SoftenFloatRes_CTSELECT(SDNode *N);
SDValue SoftenFloatRes_SELECT_CC(SDNode *N);
SDValue SoftenFloatRes_UNDEF(SDNode *N);
SDValue SoftenFloatRes_VAARG(SDNode *N);
@@ -773,6 +775,7 @@ class LLVM_LIBRARY_VISIBILITY DAGTypeLegalizer {
SDValue SoftPromoteHalfRes_LOAD(SDNode *N);
SDValue SoftPromoteHalfRes_ATOMIC_LOAD(SDNode *N);
SDValue SoftPromoteHalfRes_SELECT(SDNode *N);
+ SDValue SoftPromoteHalfRes_CTSELECT(SDNode *N);
SDValue SoftPromoteHalfRes_SELECT_CC(SDNode *N);
SDValue SoftPromoteHalfRes_UnaryOp(SDNode *N);
SDValue SoftPromoteHalfRes_FABS(SDNode *N);
@@ -845,6 +848,7 @@ class LLVM_LIBRARY_VISIBILITY DAGTypeLegalizer {
SDValue ScalarizeVecRes_VECTOR_INTERLEAVE_DEINTERLEAVE(SDNode *N);
SDValue ScalarizeVecRes_VSELECT(SDNode *N);
SDValue ScalarizeVecRes_SELECT(SDNode *N);
+ SDValue ScalarizeVecRes_CTSELECT(SDNode *N);
SDValue ScalarizeVecRes_SELECT_CC(SDNode *N);
SDValue ScalarizeVecRes_SETCC(SDNode *N);
SDValue ScalarizeVecRes_UNDEF(SDNode *N);
@@ -1195,7 +1199,8 @@ class LLVM_LIBRARY_VISIBILITY DAGTypeLegalizer {
void SplitVecRes_AssertZext(SDNode *N, SDValue &Lo, SDValue &Hi);
void SplitVecRes_AssertSext(SDNode *N, SDValue &Lo, SDValue &Hi);
void SplitRes_ARITH_FENCE (SDNode *N, SDValue &Lo, SDValue &Hi);
- void SplitRes_Select (SDNode *N, SDValue &Lo, SDValue &Hi);
+ void SplitRes_Select(SDNode *N, SDValue &Lo, SDValue &Hi);
+ void SplitRes_CTSELECT(SDNode *N, SDValue &Lo, SDValue &Hi);
void SplitRes_SELECT_CC (SDNode *N, SDValue &Lo, SDValue &Hi);
void SplitRes_UNDEF (SDNode *N, SDValue &Lo, SDValue &Hi);
void SplitRes_FREEZE (SDNode *N, SDValue &Lo, SDValue &Hi);
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeTypesGeneric.cpp b/llvm/lib/CodeGen/SelectionDAG/LegalizeTypesGeneric.cpp
index 8c252c3491540..060f7a4ae6873 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeTypesGeneric.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeTypesGeneric.cpp
@@ -569,6 +569,12 @@ void DAGTypeLegalizer::SplitRes_Select(SDNode *N, SDValue &Lo, SDValue &Hi) {
Hi = DAG.getNode(Opcode, dl, LH.getValueType(), CH, LH, RH, EVLHi);
}
+void DAGTypeLegalizer::SplitRes_CTSELECT(SDNode *N, SDValue &Lo, SDValue &Hi) {
+ // Reuse generic select splitting to support scalar and vector conditions.
+ // SplitRes_Select rebuilds with N->getOpcode(), so CTSELECT is preserved.
+ SplitRes_Select(N, Lo, Hi);
+}
+
void DAGTypeLegalizer::SplitRes_SELECT_CC(SDNode *N, SDValue &Lo,
SDValue &Hi) {
SDValue LL, LH, RL, RH;
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp b/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp
index 35ea443edbab6..8ddb9a1e78208 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp
@@ -91,6 +91,9 @@ void DAGTypeLegalizer::ScalarizeVectorResult(SDNode *N, unsigned ResNo) {
case ISD::SIGN_EXTEND_INREG: R = ScalarizeVecRes_InregOp(N); break;
case ISD::VSELECT: R = ScalarizeVecRes_VSELECT(N); break;
case ISD::SELECT: R = ScalarizeVecRes_SELECT(N); break;
+ case ISD::CTSELECT:
+ R = ScalarizeVecRes_CTSELECT(N);
+ break;
case ISD::SELECT_CC: R = ScalarizeVecRes_SELECT_CC(N); break;
case ISD::SETCC: R = ScalarizeVecRes_SETCC(N); break;
case ISD::VECTOR_MATCH:
@@ -762,6 +765,12 @@ SDValue DAGTypeLegalizer::ScalarizeVecRes_SELECT(SDNode *N) {
GetScalarizedVector(N->getOperand(2)));
}
+SDValue DAGTypeLegalizer::ScalarizeVecRes_CTSELECT(SDNode *N) {
+ SDValue LHS = GetScalarizedVector(N->getOperand(1));
+ return DAG.getCTSelect(SDLoc(N), LHS.getValueType(), N->getOperand(0), LHS,
+ GetScalarizedVector(N->getOperand(2)));
+}
+
SDValue DAGTypeLegalizer::ScalarizeVecRes_SELECT_CC(SDNode *N) {
SDValue LHS = GetScalarizedVector(N->getOperand(2));
return DAG.getNode(ISD::SELECT_CC, SDLoc(N), LHS.getValueType(),
@@ -1400,6 +1409,9 @@ void DAGTypeLegalizer::SplitVectorResult(SDNode *N, unsigned ResNo) {
case ISD::VSELECT:
case ISD::SELECT:
case ISD::VP_MERGE: SplitRes_Select(N, Lo, Hi); break;
+ case ISD::CTSELECT:
+ SplitRes_CTSELECT(N, Lo, Hi);
+ break;
case ISD::SELECT_CC: SplitRes_SELECT_CC(N, Lo, Hi); break;
case ISD::POISON:
case ISD::UNDEF: SplitRes_UNDEF(N, Lo, Hi); break;
@@ -5284,6 +5296,7 @@ void DAGTypeLegalizer::WidenVectorResult(SDNode *N, unsigned ResNo) {
case ISD::SIGN_EXTEND_INREG: Res = WidenVecRes_InregOp(N); break;
case ISD::VSELECT:
case ISD::SELECT:
+ case ISD::CTSELECT:
case ISD::VP_MERGE:
Res = WidenVecRes_Select(N);
break;
diff --git a/llvm/lib/CodeGen/SelectionDAG/SelectionDAGBuilder.cpp b/llvm/lib/CodeGen/SelectionDAG/SelectionDAGBuilder.cpp
index 794a4e6e1b537..aac1442347fbd 100644
--- a/llvm/lib/CodeGen/SelectionDAG/SelectionDAGBuilder.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/SelectionDAGBuilder.cpp
@@ -6704,6 +6704,82 @@ void SelectionDAGBuilder::visitVectorExtractLastActive(const CallInst &I,
setValue(&I, Result);
}
+/// Fallback implementation for constant-time select using DAG chaining.
+/// This implementation uses data dependencies through virtual registers to
+/// prevent optimizations from breaking the constant-time property. It is a
+/// best-effort safeguard; for stronger guarantees we prefer target-specific
+/// lowering pipelines that preserve the select pattern by construction.
+///
+/// It handles scalars, vectors (fixed and scalable), and floating-point types.
+SDValue SelectionDAGBuilder::createProtectedCtSelectFallback(
+ SelectionDAG &DAG, const SDLoc &DL, SDValue Cond, SDValue T, SDValue F,
+ EVT VT) {
+
+ SDValue WorkingT = T;
+ SDValue WorkingF = F;
+ EVT WorkingVT = VT;
+
+ SDValue Chain = DAG.getEntryNode();
+ MachineRegisterInfo &MRI = DAG.getMachineFunction().getRegInfo();
+
+ // Handle vector condition: splat scalar condition to vector
+ if (VT.isVector() && !Cond.getValueType().isVector()) {
+ ElementCount NumElems = VT.getVectorElementCount();
+ EVT CondVT = EVT::getVectorVT(*DAG.getContext(), MVT::i1, NumElems);
+ Cond = DAG.getSplat(CondVT, DL, Cond);
+ }
+
+ // Handle floating-point types: bitcast to integer for bitwise operations
+ if (VT.isFloatingPoint()) {
+ WorkingVT = VT.changeTypeToInteger();
+ WorkingT = DAG.getBitcast(WorkingVT, T);
+ WorkingF = DAG.getBitcast(WorkingVT, F);
+ }
+
+ // Create mask: sign-extend condition to all bits
+ SDValue Mask = DAG.getSExtOrTrunc(Cond, DL, WorkingVT);
+
+ // Compute: F ^ ((T ^ F) & Mask)
+ // This is constant-time because both branches are always computed
+ SDValue XorTF = DAG.getNode(ISD::XOR, DL, WorkingVT, WorkingT, WorkingF);
+ SDValue TM = DAG.getNode(ISD::AND, DL, WorkingVT, XorTF, Mask);
+
+ // DAG chaining: create data dependency through virtual register
+ // This prevents optimizations from reordering or eliminating operations
+ const TargetLowering &TLI = DAG.getTargetLoweringInfo();
+ bool CanUseChaining = false;
+
+ if (!WorkingVT.isScalableVector()) {
+ // For fixed-size vectors and scalars, chaining is a best-effort hardening
+ // step. The CT guarantee comes from the dataflow-only select
+ // pattern (both sides computed, no control-flow). Chaining only adds an
+ // extra dependency to discourage later combines.
+ CanUseChaining = TLI.isTypeLegal(WorkingVT.getSimpleVT());
+ } else {
+ // For scalable vectors, skip chaining because there is no stable register
+ // class to copy through. CT behavior still relies on the masking/select
+ // pattern above.
+ CanUseChaining = false;
+ }
+
+ if (CanUseChaining) {
+ // Apply chaining through registers for additional protection
+ const TargetRegisterClass *RC = TLI.getRegClassFor(WorkingVT.getSimpleVT());
+ Register TMReg = MRI.createVirtualRegister(RC);
+ Chain = DAG.getCopyToReg(Chain, DL, TMReg, TM);
+ TM = DAG.getCopyFromReg(Chain, DL, TMReg, WorkingVT);
+ }
+
+ SDValue Result = DAG.getNode(ISD::XOR, DL, WorkingVT, WorkingF, TM);
+
+ // Convert back to original type if needed
+ if (WorkingVT != VT) {
+ Result = DAG.getBitcast(VT, Result);
+ }
+
+ return Result;
+}
+
/// Lower the call to the specified intrinsic function.
void SelectionDAGBuilder::visitIntrinsicCall(const CallInst &I,
unsigned Intrinsic) {
@@ -6905,6 +6981,36 @@ void SelectionDAGBuilder::visitIntrinsicCall(const CallInst &I,
updateDAGForMaybeTailCall(MC);
return;
}
+ case Intrinsic::ct_select: {
+ // Set function attribute to indicate ct.select usage
+ Function &F = DAG.getMachineFunction().getFunction();
+ F.addFnAttr("ct-select");
+
+ SDLoc DL = getCurSDLoc();
+
+ SDValue Cond = getValue(I.getArgOperand(0)); // i1
+ SDValue A = getValue(I.getArgOperand(1)); // T
+ SDValue B = getValue(I.getArgOperand(2)); // T
+
+ assert((A.getValueType() == B.getValueType()) &&
+ "Operands are of different types");
+
+ EVT VT = A.getValueType();
+ EVT CondVT = Cond.getValueType();
+
+ // assert if Cond type is Vector
+ assert(!CondVT.isVector() && "Vector type cond not supported yet");
+
+ // Handle scalar types
+ if (TLI.isCtSelectSupported(VT) && !CondVT.isVector()) {
+ SDValue Result = DAG.getNode(ISD::CTSELECT, DL, VT, Cond, A, B);
+ setValue(&I, Result);
+ return;
+ }
+
+ setValue(&I, createProtectedCtSelectFallback(DAG, DL, Cond, A, B, VT));
+ return;
+ }
case Intrinsic::call_preallocated_setup: {
const CallBase *PreallocatedCall = FindPreallocatedCall(&I);
SDValue SrcValue = DAG.getSrcValue(PreallocatedCall);
diff --git a/llvm/lib/CodeGen/SelectionDAG/SelectionDAGBuilder.h b/llvm/lib/CodeGen/SelectionDAG/SelectionDAGBuilder.h
index 7a925c45c3a13..45efe4c81a9e0 100644
--- a/llvm/lib/CodeGen/SelectionDAG/SelectionDAGBuilder.h
+++ b/llvm/lib/CodeGen/SelectionDAG/SelectionDAGBuilder.h
@@ -219,6 +219,9 @@ class SelectionDAGBuilder {
peelDominantCaseCluster(const SwitchInst &SI,
SwitchCG::CaseClusterVector &Clusters,
BranchProbability &PeeledCaseProb);
+ SDValue createProtectedCtSelectFallback(SelectionDAG &DAG, const SDLoc &DL,
+ SDValue Cond, SDValue T, SDValue F,
+ EVT VT);
private:
const TargetMachine &TM;
diff --git a/llvm/lib/CodeGen/SelectionDAG/SelectionDAGDumper.cpp b/llvm/lib/CodeGen/SelectionDAG/SelectionDAGDumper.cpp
index 32e0cafd9f71c..f99394da31eeb 100644
--- a/llvm/lib/CodeGen/SelectionDAG/SelectionDAGDumper.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/SelectionDAGDumper.cpp
@@ -346,6 +346,7 @@ std::string SDNode::getOperationName(const SelectionDAG *G) const {
case ISD::FPOWI: return "fpowi";
case ISD::STRICT_FPOWI: return "strict_fpowi";
case ISD::SETCC: return "setcc";
+ case ISD::CTSELECT: return "ctselect";
case ISD::SETCCCARRY: return "setcccarry";
case ISD::STRICT_FSETCC: return "strict_fsetcc";
case ISD::STRICT_FSETCCS: return "strict_fsetccs";
diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.h b/llvm/lib/Target/AArch64/AArch64ISelLowering.h
index 1b622e93480ec..f588d46ea9d2c 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.h
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.h
@@ -167,6 +167,8 @@ class AArch64TargetLowering : public TargetLowering {
EVT getSetCCResultType(const DataLayout &DL, LLVMContext &Context,
EVT VT) const override;
+ bool isCtSelectSupported(EVT VT) const override { return false; }
+
SDValue ReconstructShuffle(SDValue Op, SelectionDAG &DAG) const;
MachineBasicBlock *EmitF128CSEL(MachineInstr &MI,
diff --git a/llvm/lib/Target/ARM/ARMISelLowering.h b/llvm/lib/Target/ARM/ARMISelLowering.h
index 4f95a1cdeb2ef..0d0cfd8c0cd68 100644
--- a/llvm/lib/Target/ARM/ARMISelLowering.h
+++ b/llvm/lib/Target/ARM/ARMISelLowering.h
@@ -119,6 +119,8 @@ class VectorType;
return (Kind != ScalarCondVectorVal);
}
+ bool isCtSelectSupported(EVT VT) const override { return false; }
+
bool isReadOnly(const GlobalValue *GV) const;
/// getSetCCResultType - Return the value type to use for ISD::SETCC.
diff --git a/llvm/lib/Target/X86/X86ISelLowering.h b/llvm/lib/Target/X86/X86ISelLowering.h
index 5215bb0704dca..455125d94518d 100644
--- a/llvm/lib/Target/X86/X86ISelLowering.h
+++ b/llvm/lib/Target/X86/X86ISelLowering.h
@@ -111,6 +111,8 @@ namespace llvm {
unsigned getJumpTableEncoding() const override;
bool useSoftFloat() const override;
+ bool isCtSelectSupported(EVT VT) const override { return false; }
+
void markLibCallAttributes(MachineFunction *MF, unsigned CC,
ArgListTy &Args) const override;
diff --git a/llvm/test/CodeGen/RISCV/ctselect-fallback.ll b/llvm/test/CodeGen/RISCV/ctselect-fallback.ll
new file mode 100644
index 0000000000000..c624c17d7e33e
--- /dev/null
+++ b/llvm/test/CodeGen/RISCV/ctselect-fallback.ll
@@ -0,0 +1,754 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 5
+; RUN: llc < %s -mtriple=riscv64 -O3 | FileCheck %s --check-prefix=RV64
+; RUN: llc < %s -mtriple=riscv32 -O3 | FileCheck %s --check-prefix=RV32
+
+; Test basic ct.select functionality for scalar types
+define i8 @test_ctselect_i8(i1 %cond, i8 %a, i8 %b) {
+; RV64-LABEL: test_ctselect_i8:
+; RV64: # %bb.0:
+; RV64-NEXT: xor a1, a1, a2
+; RV64-NEXT: slli a0, a0, 63
+; RV64-NEXT: srai a0, a0, 63
+; RV64-NEXT: and a0, a1, a0
+; RV64-NEXT: xor a0, a2, a0
+; RV64-NEXT: ret
+;
+; RV32-LABEL: test_ctselect_i8:
+; RV32: # %bb.0:
+; RV32-NEXT: xor a1, a1, a2
+; RV32-NEXT: slli a0, a0, 31
+; RV32-NEXT: srai a0, a0, 31
+; RV32-NEXT: and a0, a1, a0
+; RV32-NEXT: xor a0, a2, a0
+; RV32-NEXT: ret
+ %result = call i8 @llvm.ct.select.i8(i1 %cond, i8 %a, i8 %b)
+ ret i8 %result
+}
+define i32 @test_ctselect_i32(i1 %cond, i32 %a, i32 %b) {
+; RV64-LABEL: test_ctselect_i32:
+; RV64: # %bb.0:
+; RV64-NEXT: xor a1, a1, a2
+; RV64-NEXT: slli a0, a0, 63
+; RV64-NEXT: srai a0, a0, 63
+; RV64-NEXT: and a0, a1, a0
+; RV64-NEXT: xor a0, a2, a0
+; RV64-NEXT: ret
+;
+; RV32-LABEL: test_ctselect_i32:
+; RV32: # %bb.0:
+; RV32-NEXT: xor a1, a1, a2
+; RV32-NEXT: slli a0, a0, 31
+; RV32-NEXT: srai a0, a0, 31
+; RV32-NEXT: and a0, a1, a0
+; RV32-NEXT: xor a0, a2, a0
+; RV32-NEXT: ret
+ %result = call i32 @llvm.ct.select.i32(i1 %cond, i32 %a, i32 %b)
+ ret i32 %result
+}
+
+define i64 @test_ctselect_i64(i1 %cond, i64 %a, i64 %b) {
+; RV64-LABEL: test_ctselect_i64:
+; RV64: # %bb.0:
+; RV64-NEXT: xor a1, a1, a2
+; RV64-NEXT: slli a0, a0, 63
+; RV64-NEXT: srai a0, a0, 63
+; RV64-NEXT: and a0, a1, a0
+; RV64-NEXT: xor a0, a2, a0
+; RV64-NEXT: ret
+;
+; RV32-LABEL: test_ctselect_i64:
+; RV32: # %bb.0:
+; RV32-NEXT: xor a1, a1, a3
+; RV32-NEXT: slli a0, a0, 31
+; RV32-NEXT: xor a2, a2, a4
+; RV32-NEXT: srai a0, a0, 31
+; RV32-NEXT: and a1, a1, a0
+; RV32-NEXT: and a2, a2, a0
+; RV32-NEXT: xor a0, a3, a1
+; RV32-NEXT: xor a1, a4, a2
+; RV32-NEXT: ret
+ %result = call i64 @llvm.ct.select.i64(i1 %cond, i64 %a, i64 %b)
+ ret i64 %result
+}
+
+define ptr @test_ctselect_ptr(i1 %cond, ptr %a, ptr %b) {
+; RV64-LABEL: test_ctselect_ptr:
+; RV64: # %bb.0:
+; RV64-NEXT: xor a1, a1, a2
+; RV64-NEXT: slli a0, a0, 63
+; RV64-NEXT: srai a0, a0, 63
+; RV64-NEXT: and a0, a1, a0
+; RV64-NEXT: xor a0, a2, a0
+; RV64-NEXT: ret
+;
+; RV32-LABEL: test_ctselect_ptr:
+; RV32: # %bb.0:
+; RV32-NEXT: xor a1, a1, a2
+; RV32-NEXT: slli a0, a0, 31
+; RV32-NEXT: srai a0, a0, 31
+; RV32-NEXT: and a0, a1, a0
+; RV32-NEXT: xor a0, a2, a0
+; RV32-NEXT: ret
+ %result = call ptr @llvm.ct.select.p0(i1 %cond, ptr %a, ptr %b)
+ ret ptr %result
+}
+
+; Test with constant conditions
+define i32 @test_ctselect_const_true(i32 %a, i32 %b) {
+; RV64-LABEL: test_ctselect_const_true:
+; RV64: # %bb.0:
+; RV64-NEXT: ret
+;
+; RV32-LABEL: test_ctselect_const_true:
+; RV32: # %bb.0:
+; RV32-NEXT: xor a0, a0, a1
+; RV32-NEXT: xor a0, a1, a0
+; RV32-NEXT: ret
+ %result = call i32 @llvm.ct.select.i32(i1 true, i32 %a, i32 %b)
+ ret i32 %result
+}
+
+define i32 @test_ctselect_const_false(i32 %a, i32 %b) {
+; RV64-LABEL: test_ctselect_const_false:
+; RV64: # %bb.0:
+; RV64-NEXT: mv a0, a1
+; RV64-NEXT: ret
+;
+; RV32-LABEL: test_ctselect_const_false:
+; RV32: # %bb.0:
+; RV32-NEXT: mv a0, a1
+; RV32-NEXT: ret
+ %result = call i32 @llvm.ct.select.i32(i1 false, i32 %a, i32 %b)
+ ret i32 %result
+}
+
+; Test with comparison conditions
+define i32 @test_ctselect_icmp_eq(i32 %x, i32 %y, i32 %a, i32 %b) {
+; RV64-LABEL: test_ctselect_icmp_eq:
+; RV64: # %bb.0:
+; RV64-NEXT: sext.w a1, a1
+; RV64-NEXT: sext.w a0, a0
+; RV64-NEXT: xor a0, a0, a1
+; RV64-NEXT: snez a0, a0
+; RV64-NEXT: xor a2, a2, a3
+; RV64-NEXT: addi a0, a0, -1
+; RV64-NEXT: and a0, a2, a0
+; RV64-NEXT: xor a0, a3, a0
+; RV64-NEXT: ret
+;
+; RV32-LABEL: test_ctselect_icmp_eq:
+; RV32: # %bb.0:
+; RV32-NEXT: xor a0, a0, a1
+; RV32-NEXT: snez a0, a0
+; RV32-NEXT: xor a2, a2, a3
+; RV32-NEXT: addi a0, a0, -1
+; RV32-NEXT: and a0, a2, a0
+; RV32-NEXT: xor a0, a3, a0
+; RV32-NEXT: ret
+ %cond = icmp eq i32 %x, %y
+ %result = call i32 @llvm.ct.select.i32(i1 %cond, i32 %a, i32 %b)
+ ret i32 %result
+}
+define i32 @test_ctselect_icmp_ult(i32 %x, i32 %y, i32 %a, i32 %b) {
+; RV64-LABEL: test_ctselect_icmp_ult:
+; RV64: # %bb.0:
+; RV64-NEXT: sext.w a1, a1
+; RV64-NEXT: sext.w a0, a0
+; RV64-NEXT: sltu a0, a0, a1
+; RV64-NEXT: xor a2, a2, a3
+; RV64-NEXT: neg a0, a0
+; RV64-NEXT: and a0, a2, a0
+; RV64-NEXT: xor a0, a3, a0
+; RV64-NEXT: ret
+;
+; RV32-LABEL: test_ctselect_icmp_ult:
+; RV32: # %bb.0:
+; RV32-NEXT: sltu a0, a0, a1
+; RV32-NEXT: xor a2, a2, a3
+; RV32-NEXT: neg a0, a0
+; RV32-NEXT: and a0, a2, a0
+; RV32-NEXT: xor a0, a3, a0
+; RV32-NEXT: ret
+ %cond = icmp ult i32 %x, %y
+ %result = call i32 @llvm.ct.select.i32(i1 %cond, i32 %a, i32 %b)
+ ret i32 %result
+}
+
+; Test with memory operands
+define i32 @test_ctselect_load(i1 %cond, ptr %p1, ptr %p2) {
+; RV64-LABEL: test_ctselect_load:
+; RV64: # %bb.0:
+; RV64-NEXT: lw a1, 0(a1)
+; RV64-NEXT: lw a2, 0(a2)
+; RV64-NEXT: slli a0, a0, 63
+; RV64-NEXT: xor a1, a1, a2
+; RV64-NEXT: srai a0, a0, 63
+; RV64-NEXT: and a0, a1, a0
+; RV64-NEXT: xor a0, a2, a0
+; RV64-NEXT: ret
+;
+; RV32-LABEL: test_ctselect_load:
+; RV32: # %bb.0:
+; RV32-NEXT: lw a1, 0(a1)
+; RV32-NEXT: lw a2, 0(a2)
+; RV32-NEXT: slli a0, a0, 31
+; RV32-NEXT: xor a1, a1, a2
+; RV32-NEXT: srai a0, a0, 31
+; RV32-NEXT: and a0, a1, a0
+; RV32-NEXT: xor a0, a2, a0
+; RV32-NEXT: ret
+ %a = load i32, ptr %p1
+ %b = load i32, ptr %p2
+ %result = call i32 @llvm.ct.select.i32(i1 %cond, i32 %a, i32 %b)
+ ret i32 %result
+}
+
+; Test nested CTSELECT pattern with AND merging on i1 values
+; Pattern: ctselect C0, (ctselect C1, X, Y), Y -> ctselect (C0 & C1), X, Y
+define i32 @test_ctselect_nested_and_i1_to_i32(i1 %c0, i1 %c1, i32 %x, i32 %y) {
+; RV64-LABEL: test_ctselect_nested_and_i1_to_i32:
+; RV64: # %bb.0:
+; RV64-NEXT: and a0, a1, a0
+; RV64-NEXT: xor a2, a2, a3
+; RV64-NEXT: slli a0, a0, 63
+; RV64-NEXT: srai a0, a0, 63
+; RV64-NEXT: and a0, a2, a0
+; RV64-NEXT: xor a0, a3, a0
+; RV64-NEXT: ret
+;
+; RV32-LABEL: test_ctselect_nested_and_i1_to_i32:
+; RV32: # %bb.0:
+; RV32-NEXT: and a0, a1, a0
+; RV32-NEXT: xor a2, a2, a3
+; RV32-NEXT: slli a0, a0, 31
+; RV32-NEXT: srai a0, a0, 31
+; RV32-NEXT: and a0, a2, a0
+; RV32-NEXT: xor a0, a3, a0
+; RV32-NEXT: ret
+ %inner = call i1 @llvm.ct.select.i1(i1 %c1, i1 true, i1 false)
+ %cond = call i1 @llvm.ct.select.i1(i1 %c0, i1 %inner, i1 false)
+ %result = call i32 @llvm.ct.select.i32(i1 %cond, i32 %x, i32 %y)
+ ret i32 %result
+}
+
+; Test nested CTSELECT pattern with OR merging on i1 values
+; Pattern: ctselect C0, X, (ctselect C1, X, Y) -> ctselect (C0 | C1), X, Y
+define i32 @test_ctselect_nested_or_i1_to_i32(i1 %c0, i1 %c1, i32 %x, i32 %y) {
+; RV64-LABEL: test_ctselect_nested_or_i1_to_i32:
+; RV64: # %bb.0:
+; RV64-NEXT: or a0, a0, a1
+; RV64-NEXT: xor a2, a2, a3
+; RV64-NEXT: slli a0, a0, 63
+; RV64-NEXT: srai a0, a0, 63
+; RV64-NEXT: and a0, a2, a0
+; RV64-NEXT: xor a0, a3, a0
+; RV64-NEXT: ret
+;
+; RV32-LABEL: test_ctselect_nested_or_i1_to_i32:
+; RV32: # %bb.0:
+; RV32-NEXT: or a0, a0, a1
+; RV32-NEXT: xor a2, a2, a3
+; RV32-NEXT: slli a0, a0, 31
+; RV32-NEXT: srai a0, a0, 31
+; RV32-NEXT: and a0, a2, a0
+; RV32-NEXT: xor a0, a3, a0
+; RV32-NEXT: ret
+ %inner = call i1 @llvm.ct.select.i1(i1 %c1, i1 true, i1 false)
+ %cond = call i1 @llvm.ct.select.i1(i1 %c0, i1 true, i1 %inner)
+ %result = call i32 @llvm.ct.select.i32(i1 %cond, i32 %x, i32 %y)
+ ret i32 %result
+}
+
+; Test double nested CTSELECT with recursive AND merging
+; Pattern: ctselect C0, (ctselect C1, (ctselect C2, X, Y), Y), Y
+; -> ctselect (C0 & C1 & C2), X, Y
+define i32 @test_ctselect_double_nested_and_i1(i1 %c0, i1 %c1, i1 %c2, i32 %x, i32 %y) {
+; RV64-LABEL: test_ctselect_double_nested_and_i1:
+; RV64: # %bb.0:
+; RV64-NEXT: and a1, a2, a1
+; RV64-NEXT: and a0, a1, a0
+; RV64-NEXT: xor a3, a3, a4
+; RV64-NEXT: slli a0, a0, 63
+; RV64-NEXT: srai a0, a0, 63
+; RV64-NEXT: and a0, a3, a0
+; RV64-NEXT: xor a0, a4, a0
+; RV64-NEXT: ret
+;
+; RV32-LABEL: test_ctselect_double_nested_and_i1:
+; RV32: # %bb.0:
+; RV32-NEXT: and a1, a2, a1
+; RV32-NEXT: and a0, a1, a0
+; RV32-NEXT: xor a3, a3, a4
+; RV32-NEXT: slli a0, a0, 31
+; RV32-NEXT: srai a0, a0, 31
+; RV32-NEXT: and a0, a3, a0
+; RV32-NEXT: xor a0, a4, a0
+; RV32-NEXT: ret
+ %inner2 = call i1 @llvm.ct.select.i1(i1 %c2, i1 true, i1 false)
+ %inner1 = call i1 @llvm.ct.select.i1(i1 %c1, i1 %inner2, i1 false)
+ %cond = call i1 @llvm.ct.select.i1(i1 %c0, i1 %inner1, i1 false)
+ %result = call i32 @llvm.ct.select.i32(i1 %cond, i32 %x, i32 %y)
+ ret i32 %result
+}
+
+; Test double nested CTSELECT with mixed AND/OR patterns
+define i32 @test_ctselect_double_nested_mixed_i1(i1 %c0, i1 %c1, i1 %c2, i32 %x, i32 %y, i32 %z) {
+; RV64-LABEL: test_ctselect_double_nested_mixed_i1:
+; RV64: # %bb.0:
+; RV64-NEXT: and a0, a1, a0
+; RV64-NEXT: xor a3, a3, a4
+; RV64-NEXT: or a0, a0, a2
+; RV64-NEXT: slli a0, a0, 63
+; RV64-NEXT: srai a0, a0, 63
+; RV64-NEXT: and a3, a3, a0
+; RV64-NEXT: xor a4, a4, a5
+; RV64-NEXT: xor a3, a4, a3
+; RV64-NEXT: and a0, a3, a0
+; RV64-NEXT: xor a0, a5, a0
+; RV64-NEXT: ret
+;
+; RV32-LABEL: test_ctselect_double_nested_mixed_i1:
+; RV32: # %bb.0:
+; RV32-NEXT: and a0, a1, a0
+; RV32-NEXT: xor a3, a3, a4
+; RV32-NEXT: or a0, a0, a2
+; RV32-NEXT: slli a0, a0, 31
+; RV32-NEXT: srai a0, a0, 31
+; RV32-NEXT: and a3, a3, a0
+; RV32-NEXT: xor a4, a4, a5
+; RV32-NEXT: xor a3, a4, a3
+; RV32-NEXT: and a0, a3, a0
+; RV32-NEXT: xor a0, a5, a0
+; RV32-NEXT: ret
+ %inner1 = call i1 @llvm.ct.select.i1(i1 %c1, i1 true, i1 false)
+ %and_cond = call i1 @llvm.ct.select.i1(i1 %c0, i1 %inner1, i1 false)
+ %inner2 = call i1 @llvm.ct.select.i1(i1 %c2, i1 true, i1 false)
+ %or_cond = call i1 @llvm.ct.select.i1(i1 %and_cond, i1 true, i1 %inner2)
+ %inner_result = call i32 @llvm.ct.select.i32(i1 %or_cond, i32 %x, i32 %y)
+ %result = call i32 @llvm.ct.select.i32(i1 %or_cond, i32 %inner_result, i32 %z)
+ ret i32 %result
+}
+
+; Test nested ctselect calls
+define i32 @test_ctselect_nested(i1 %cond1, i1 %cond2, i32 %a, i32 %b, i32 %c) {
+; RV64-LABEL: test_ctselect_nested:
+; RV64: # %bb.0:
+; RV64-NEXT: xor a2, a2, a3
+; RV64-NEXT: slli a1, a1, 63
+; RV64-NEXT: xor a3, a3, a4
+; RV64-NEXT: slli a0, a0, 63
+; RV64-NEXT: srai a1, a1, 63
+; RV64-NEXT: and a1, a2, a1
+; RV64-NEXT: xor a1, a3, a1
+; RV64-NEXT: srai a0, a0, 63
+; RV64-NEXT: and a0, a1, a0
+; RV64-NEXT: xor a0, a4, a0
+; RV64-NEXT: ret
+;
+; RV32-LABEL: test_ctselect_nested:
+; RV32: # %bb.0:
+; RV32-NEXT: xor a2, a2, a3
+; RV32-NEXT: slli a1, a1, 31
+; RV32-NEXT: xor a3, a3, a4
+; RV32-NEXT: slli a0, a0, 31
+; RV32-NEXT: srai a1, a1, 31
+; RV32-NEXT: and a1, a2, a1
+; RV32-NEXT: xor a1, a3, a1
+; RV32-NEXT: srai a0, a0, 31
+; RV32-NEXT: and a0, a1, a0
+; RV32-NEXT: xor a0, a4, a0
+; RV32-NEXT: ret
+ %inner = call i32 @llvm.ct.select.i32(i1 %cond2, i32 %a, i32 %b)
+ %result = call i32 @llvm.ct.select.i32(i1 %cond1, i32 %inner, i32 %c)
+ ret i32 %result
+}
+
+; Test floating-point ct.select selecting between NaN and Inf
+define float @test_ctselect_f32_nan_inf(i1 %cond) {
+; RV64-LABEL: test_ctselect_f32_nan_inf:
+; RV64: # %bb.0:
+; RV64-NEXT: slli a0, a0, 63
+; RV64-NEXT: lui a1, 1024
+; RV64-NEXT: srai a0, a0, 63
+; RV64-NEXT: and a0, a0, a1
+; RV64-NEXT: lui a1, 522240
+; RV64-NEXT: or a0, a0, a1
+; RV64-NEXT: ret
+;
+; RV32-LABEL: test_ctselect_f32_nan_inf:
+; RV32: # %bb.0:
+; RV32-NEXT: slli a0, a0, 31
+; RV32-NEXT: lui a1, 1024
+; RV32-NEXT: srai a0, a0, 31
+; RV32-NEXT: and a0, a0, a1
+; RV32-NEXT: lui a1, 522240
+; RV32-NEXT: xor a0, a0, a1
+; RV32-NEXT: ret
+ %result = call float @llvm.ct.select.f32(i1 %cond, float 0x7FF8000000000000, float 0x7FF0000000000000)
+ ret float %result
+}
+
+define double @test_ctselect_f64_nan_inf(i1 %cond) {
+; RV64-LABEL: test_ctselect_f64_nan_inf:
+; RV64: # %bb.0:
+; RV64-NEXT: slli a0, a0, 63
+; RV64-NEXT: li a1, 1
+; RV64-NEXT: srai a0, a0, 63
+; RV64-NEXT: slli a1, a1, 51
+; RV64-NEXT: and a0, a0, a1
+; RV64-NEXT: li a1, 2047
+; RV64-NEXT: slli a1, a1, 52
+; RV64-NEXT: xor a0, a0, a1
+; RV64-NEXT: ret
+;
+; RV32-LABEL: test_ctselect_f64_nan_inf:
+; RV32: # %bb.0:
+; RV32-NEXT: slli a0, a0, 31
+; RV32-NEXT: lui a1, 128
+; RV32-NEXT: srai a0, a0, 31
+; RV32-NEXT: and a0, a0, a1
+; RV32-NEXT: lui a1, 524032
+; RV32-NEXT: or a1, a0, a1
+; RV32-NEXT: li a0, 0
+; RV32-NEXT: ret
+ %result = call double @llvm.ct.select.f64(i1 %cond, double 0x7FF8000000000000, double 0x7FF0000000000000)
+ ret double %result
+}
+
+; Test basic floating-point ct.select
+define float @test_ctselect_f32(i1 %cond, float %a, float %b) {
+; RV64-LABEL: test_ctselect_f32:
+; RV64: # %bb.0:
+; RV64-NEXT: xor a1, a1, a2
+; RV64-NEXT: slli a0, a0, 63
+; RV64-NEXT: srai a0, a0, 63
+; RV64-NEXT: and a0, a1, a0
+; RV64-NEXT: xor a0, a2, a0
+; RV64-NEXT: ret
+;
+; RV32-LABEL: test_ctselect_f32:
+; RV32: # %bb.0:
+; RV32-NEXT: xor a1, a1, a2
+; RV32-NEXT: slli a0, a0, 31
+; RV32-NEXT: srai a0, a0, 31
+; RV32-NEXT: and a0, a1, a0
+; RV32-NEXT: xor a0, a2, a0
+; RV32-NEXT: ret
+ %result = call float @llvm.ct.select.f32(i1 %cond, float %a, float %b)
+ ret float %result
+}
+
+define double @test_ctselect_f64(i1 %cond, double %a, double %b) {
+; RV64-LABEL: test_ctselect_f64:
+; RV64: # %bb.0:
+; RV64-NEXT: xor a1, a1, a2
+; RV64-NEXT: slli a0, a0, 63
+; RV64-NEXT: srai a0, a0, 63
+; RV64-NEXT: and a0, a1, a0
+; RV64-NEXT: xor a0, a2, a0
+; RV64-NEXT: ret
+;
+; RV32-LABEL: test_ctselect_f64:
+; RV32: # %bb.0:
+; RV32-NEXT: xor a1, a1, a3
+; RV32-NEXT: slli a0, a0, 31
+; RV32-NEXT: xor a2, a2, a4
+; RV32-NEXT: srai a0, a0, 31
+; RV32-NEXT: and a1, a1, a0
+; RV32-NEXT: and a2, a2, a0
+; RV32-NEXT: xor a0, a3, a1
+; RV32-NEXT: xor a1, a4, a2
+; RV32-NEXT: ret
+ %result = call double @llvm.ct.select.f64(i1 %cond, double %a, double %b)
+ ret double %result
+}
+
+; Test vector ct.select with integer vectors
+define <4 x i32> @test_ctselect_v4i32(i1 %cond, <4 x i32> %a, <4 x i32> %b) {
+; RV64-LABEL: test_ctselect_v4i32:
+; RV64: # %bb.0:
+; RV64-NEXT: lw a4, 0(a3)
+; RV64-NEXT: lw a5, 8(a3)
+; RV64-NEXT: lw a6, 16(a3)
+; RV64-NEXT: lw a3, 24(a3)
+; RV64-NEXT: lw a7, 0(a2)
+; RV64-NEXT: lw t0, 8(a2)
+; RV64-NEXT: lw t1, 16(a2)
+; RV64-NEXT: lw a2, 24(a2)
+; RV64-NEXT: slli a1, a1, 63
+; RV64-NEXT: srai a1, a1, 63
+; RV64-NEXT: xor a7, a7, a4
+; RV64-NEXT: xor t0, t0, a5
+; RV64-NEXT: xor t1, t1, a6
+; RV64-NEXT: xor a2, a2, a3
+; RV64-NEXT: and a7, a7, a1
+; RV64-NEXT: and t0, t0, a1
+; RV64-NEXT: and t1, t1, a1
+; RV64-NEXT: and a1, a2, a1
+; RV64-NEXT: xor a2, a4, a7
+; RV64-NEXT: xor a4, a5, t0
+; RV64-NEXT: xor a5, a6, t1
+; RV64-NEXT: xor a1, a3, a1
+; RV64-NEXT: sw a2, 0(a0)
+; RV64-NEXT: sw a4, 4(a0)
+; RV64-NEXT: sw a5, 8(a0)
+; RV64-NEXT: sw a1, 12(a0)
+; RV64-NEXT: ret
+;
+; RV32-LABEL: test_ctselect_v4i32:
+; RV32: # %bb.0:
+; RV32-NEXT: lw a4, 0(a3)
+; RV32-NEXT: lw a5, 4(a3)
+; RV32-NEXT: lw a6, 8(a3)
+; RV32-NEXT: lw a3, 12(a3)
+; RV32-NEXT: lw a7, 0(a2)
+; RV32-NEXT: lw t0, 4(a2)
+; RV32-NEXT: lw t1, 8(a2)
+; RV32-NEXT: lw a2, 12(a2)
+; RV32-NEXT: slli a1, a1, 31
+; RV32-NEXT: srai a1, a1, 31
+; RV32-NEXT: xor a7, a7, a4
+; RV32-NEXT: xor t0, t0, a5
+; RV32-NEXT: xor t1, t1, a6
+; RV32-NEXT: xor a2, a2, a3
+; RV32-NEXT: and a7, a7, a1
+; RV32-NEXT: and t0, t0, a1
+; RV32-NEXT: and t1, t1, a1
+; RV32-NEXT: and a1, a2, a1
+; RV32-NEXT: xor a2, a4, a7
+; RV32-NEXT: xor a4, a5, t0
+; RV32-NEXT: xor a5, a6, t1
+; RV32-NEXT: xor a1, a3, a1
+; RV32-NEXT: sw a2, 0(a0)
+; RV32-NEXT: sw a4, 4(a0)
+; RV32-NEXT: sw a5, 8(a0)
+; RV32-NEXT: sw a1, 12(a0)
+; RV32-NEXT: ret
+ %result = call <4 x i32> @llvm.ct.select.v4i32(i1 %cond, <4 x i32> %a, <4 x i32> %b)
+ ret <4 x i32> %result
+}
+define <4 x float> @test_ctselect_v4f32(i1 %cond, <4 x float> %a, <4 x float> %b) {
+; RV64-LABEL: test_ctselect_v4f32:
+; RV64: # %bb.0:
+; RV64-NEXT: lw a4, 0(a3)
+; RV64-NEXT: lw a5, 8(a3)
+; RV64-NEXT: lw a6, 16(a3)
+; RV64-NEXT: lw a3, 24(a3)
+; RV64-NEXT: lw a7, 0(a2)
+; RV64-NEXT: lw t0, 8(a2)
+; RV64-NEXT: lw t1, 16(a2)
+; RV64-NEXT: lw a2, 24(a2)
+; RV64-NEXT: slli a1, a1, 63
+; RV64-NEXT: srai a1, a1, 63
+; RV64-NEXT: xor a7, a7, a4
+; RV64-NEXT: xor t0, t0, a5
+; RV64-NEXT: xor t1, t1, a6
+; RV64-NEXT: xor a2, a2, a3
+; RV64-NEXT: and a7, a7, a1
+; RV64-NEXT: and t0, t0, a1
+; RV64-NEXT: and t1, t1, a1
+; RV64-NEXT: and a1, a2, a1
+; RV64-NEXT: xor a2, a4, a7
+; RV64-NEXT: xor a4, a5, t0
+; RV64-NEXT: xor a5, a6, t1
+; RV64-NEXT: xor a1, a3, a1
+; RV64-NEXT: sw a2, 0(a0)
+; RV64-NEXT: sw a4, 4(a0)
+; RV64-NEXT: sw a5, 8(a0)
+; RV64-NEXT: sw a1, 12(a0)
+; RV64-NEXT: ret
+;
+; RV32-LABEL: test_ctselect_v4f32:
+; RV32: # %bb.0:
+; RV32-NEXT: lw a4, 0(a3)
+; RV32-NEXT: lw a5, 4(a3)
+; RV32-NEXT: lw a6, 8(a3)
+; RV32-NEXT: lw a3, 12(a3)
+; RV32-NEXT: lw a7, 0(a2)
+; RV32-NEXT: lw t0, 4(a2)
+; RV32-NEXT: lw t1, 8(a2)
+; RV32-NEXT: lw a2, 12(a2)
+; RV32-NEXT: slli a1, a1, 31
+; RV32-NEXT: srai a1, a1, 31
+; RV32-NEXT: xor a7, a7, a4
+; RV32-NEXT: xor t0, t0, a5
+; RV32-NEXT: xor t1, t1, a6
+; RV32-NEXT: xor a2, a2, a3
+; RV32-NEXT: and a7, a7, a1
+; RV32-NEXT: and t0, t0, a1
+; RV32-NEXT: and t1, t1, a1
+; RV32-NEXT: and a1, a2, a1
+; RV32-NEXT: xor a2, a4, a7
+; RV32-NEXT: xor a4, a5, t0
+; RV32-NEXT: xor a5, a6, t1
+; RV32-NEXT: xor a1, a3, a1
+; RV32-NEXT: sw a2, 0(a0)
+; RV32-NEXT: sw a4, 4(a0)
+; RV32-NEXT: sw a5, 8(a0)
+; RV32-NEXT: sw a1, 12(a0)
+; RV32-NEXT: ret
+ %result = call <4 x float> @llvm.ct.select.v4f32(i1 %cond, <4 x float> %a, <4 x float> %b)
+ ret <4 x float> %result
+}
+define <8 x i32> @test_ctselect_v8i32(i1 %cond, <8 x i32> %a, <8 x i32> %b) {
+; RV64-LABEL: test_ctselect_v8i32:
+; RV64: # %bb.0:
+; RV64-NEXT: addi sp, sp, -32
+; RV64-NEXT: .cfi_def_cfa_offset 32
+; RV64-NEXT: sd s0, 24(sp) # 8-byte Folded Spill
+; RV64-NEXT: sd s1, 16(sp) # 8-byte Folded Spill
+; RV64-NEXT: sd s2, 8(sp) # 8-byte Folded Spill
+; RV64-NEXT: .cfi_offset s0, -8
+; RV64-NEXT: .cfi_offset s1, -16
+; RV64-NEXT: .cfi_offset s2, -24
+; RV64-NEXT: lw a7, 32(a3)
+; RV64-NEXT: lw a6, 40(a3)
+; RV64-NEXT: lw a5, 48(a3)
+; RV64-NEXT: lw a4, 56(a3)
+; RV64-NEXT: lw t0, 32(a2)
+; RV64-NEXT: lw t1, 40(a2)
+; RV64-NEXT: lw t2, 48(a2)
+; RV64-NEXT: lw t3, 56(a2)
+; RV64-NEXT: lw t4, 0(a3)
+; RV64-NEXT: lw t5, 8(a3)
+; RV64-NEXT: lw t6, 16(a3)
+; RV64-NEXT: lw a3, 24(a3)
+; RV64-NEXT: lw s0, 0(a2)
+; RV64-NEXT: lw s1, 8(a2)
+; RV64-NEXT: lw s2, 16(a2)
+; RV64-NEXT: lw a2, 24(a2)
+; RV64-NEXT: slli a1, a1, 63
+; RV64-NEXT: srai a1, a1, 63
+; RV64-NEXT: xor s0, s0, t4
+; RV64-NEXT: xor s1, s1, t5
+; RV64-NEXT: xor s2, s2, t6
+; RV64-NEXT: xor a2, a2, a3
+; RV64-NEXT: xor t0, t0, a7
+; RV64-NEXT: xor t1, t1, a6
+; RV64-NEXT: xor t2, t2, a5
+; RV64-NEXT: xor t3, t3, a4
+; RV64-NEXT: and s0, s0, a1
+; RV64-NEXT: and s1, s1, a1
+; RV64-NEXT: and s2, s2, a1
+; RV64-NEXT: and a2, a2, a1
+; RV64-NEXT: and t0, t0, a1
+; RV64-NEXT: and t1, t1, a1
+; RV64-NEXT: and t2, t2, a1
+; RV64-NEXT: and a1, t3, a1
+; RV64-NEXT: xor t3, t4, s0
+; RV64-NEXT: xor t4, t5, s1
+; RV64-NEXT: xor t5, t6, s2
+; RV64-NEXT: xor a2, a3, a2
+; RV64-NEXT: xor a3, a7, t0
+; RV64-NEXT: xor a6, a6, t1
+; RV64-NEXT: xor a5, a5, t2
+; RV64-NEXT: xor a1, a4, a1
+; RV64-NEXT: sw a3, 16(a0)
+; RV64-NEXT: sw a6, 20(a0)
+; RV64-NEXT: sw a5, 24(a0)
+; RV64-NEXT: sw a1, 28(a0)
+; RV64-NEXT: sw t3, 0(a0)
+; RV64-NEXT: sw t4, 4(a0)
+; RV64-NEXT: sw t5, 8(a0)
+; RV64-NEXT: sw a2, 12(a0)
+; RV64-NEXT: ld s0, 24(sp) # 8-byte Folded Reload
+; RV64-NEXT: ld s1, 16(sp) # 8-byte Folded Reload
+; RV64-NEXT: ld s2, 8(sp) # 8-byte Folded Reload
+; RV64-NEXT: .cfi_restore s0
+; RV64-NEXT: .cfi_restore s1
+; RV64-NEXT: .cfi_restore s2
+; RV64-NEXT: addi sp, sp, 32
+; RV64-NEXT: .cfi_def_cfa_offset 0
+; RV64-NEXT: ret
+;
+; RV32-LABEL: test_ctselect_v8i32:
+; RV32: # %bb.0:
+; RV32-NEXT: addi sp, sp, -16
+; RV32-NEXT: .cfi_def_cfa_offset 16
+; RV32-NEXT: sw s0, 12(sp) # 4-byte Folded Spill
+; RV32-NEXT: sw s1, 8(sp) # 4-byte Folded Spill
+; RV32-NEXT: sw s2, 4(sp) # 4-byte Folded Spill
+; RV32-NEXT: .cfi_offset s0, -4
+; RV32-NEXT: .cfi_offset s1, -8
+; RV32-NEXT: .cfi_offset s2, -12
+; RV32-NEXT: lw a7, 16(a3)
+; RV32-NEXT: lw a6, 20(a3)
+; RV32-NEXT: lw a5, 24(a3)
+; RV32-NEXT: lw a4, 28(a3)
+; RV32-NEXT: lw t0, 16(a2)
+; RV32-NEXT: lw t1, 20(a2)
+; RV32-NEXT: lw t2, 24(a2)
+; RV32-NEXT: lw t3, 28(a2)
+; RV32-NEXT: lw t4, 0(a3)
+; RV32-NEXT: lw t5, 4(a3)
+; RV32-NEXT: lw t6, 8(a3)
+; RV32-NEXT: lw a3, 12(a3)
+; RV32-NEXT: lw s0, 0(a2)
+; RV32-NEXT: lw s1, 4(a2)
+; RV32-NEXT: lw s2, 8(a2)
+; RV32-NEXT: lw a2, 12(a2)
+; RV32-NEXT: slli a1, a1, 31
+; RV32-NEXT: srai a1, a1, 31
+; RV32-NEXT: xor s0, s0, t4
+; RV32-NEXT: xor s1, s1, t5
+; RV32-NEXT: xor s2, s2, t6
+; RV32-NEXT: xor a2, a2, a3
+; RV32-NEXT: xor t0, t0, a7
+; RV32-NEXT: xor t1, t1, a6
+; RV32-NEXT: xor t2, t2, a5
+; RV32-NEXT: xor t3, t3, a4
+; RV32-NEXT: and s0, s0, a1
+; RV32-NEXT: and s1, s1, a1
+; RV32-NEXT: and s2, s2, a1
+; RV32-NEXT: and a2, a2, a1
+; RV32-NEXT: and t0, t0, a1
+; RV32-NEXT: and t1, t1, a1
+; RV32-NEXT: and t2, t2, a1
+; RV32-NEXT: and a1, t3, a1
+; RV32-NEXT: xor t3, t4, s0
+; RV32-NEXT: xor t4, t5, s1
+; RV32-NEXT: xor t5, t6, s2
+; RV32-NEXT: xor a2, a3, a2
+; RV32-NEXT: xor a3, a7, t0
+; RV32-NEXT: xor a6, a6, t1
+; RV32-NEXT: xor a5, a5, t2
+; RV32-NEXT: xor a1, a4, a1
+; RV32-NEXT: sw a3, 16(a0)
+; RV32-NEXT: sw a6, 20(a0)
+; RV32-NEXT: sw a5, 24(a0)
+; RV32-NEXT: sw a1, 28(a0)
+; RV32-NEXT: sw t3, 0(a0)
+; RV32-NEXT: sw t4, 4(a0)
+; RV32-NEXT: sw t5, 8(a0)
+; RV32-NEXT: sw a2, 12(a0)
+; RV32-NEXT: lw s0, 12(sp) # 4-byte Folded Reload
+; RV32-NEXT: lw s1, 8(sp) # 4-byte Folded Reload
+; RV32-NEXT: lw s2, 4(sp) # 4-byte Folded Reload
+; RV32-NEXT: .cfi_restore s0
+; RV32-NEXT: .cfi_restore s1
+; RV32-NEXT: .cfi_restore s2
+; RV32-NEXT: addi sp, sp, 16
+; RV32-NEXT: .cfi_def_cfa_offset 0
+; RV32-NEXT: ret
+ %result = call <8 x i32> @llvm.ct.select.v8i32(i1 %cond, <8 x i32> %a, <8 x i32> %b)
+ ret <8 x i32> %result
+}
+
+; Declare the intrinsics
+declare i1 @llvm.ct.select.i1(i1, i1, i1)
+declare i8 @llvm.ct.select.i8(i1, i8, i8)
+declare i16 @llvm.ct.select.i16(i1, i16, i16)
+declare i32 @llvm.ct.select.i32(i1, i32, i32)
+declare i64 @llvm.ct.select.i64(i1, i64, i64)
+declare ptr @llvm.ct.select.p0(i1, ptr, ptr)
+declare float @llvm.ct.select.f32(i1, float, float)
+declare double @llvm.ct.select.f64(i1, double, double)
+
+; Vector intrinsics
+declare <4 x i32> @llvm.ct.select.v4i32(i1, <4 x i32>, <4 x i32>)
+declare <2 x i64> @llvm.ct.select.v2i64(i1, <2 x i64>, <2 x i64>)
+declare <8 x i16> @llvm.ct.select.v8i16(i1, <8 x i16>, <8 x i16>)
+declare <16 x i8> @llvm.ct.select.v16i8(i1, <16 x i8>, <16 x i8>)
+declare <4 x float> @llvm.ct.select.v4f32(i1, <4 x float>, <4 x float>)
+declare <2 x double> @llvm.ct.select.v2f64(i1, <2 x double>, <2 x double>)
+declare <8 x i32> @llvm.ct.select.v8i32(i1, <8 x i32>, <8 x i32>)
diff --git a/llvm/test/CodeGen/X86/ctselect.ll b/llvm/test/CodeGen/X86/ctselect.ll
new file mode 100644
index 0000000000000..a970fd8933b9b
--- /dev/null
+++ b/llvm/test/CodeGen/X86/ctselect.ll
@@ -0,0 +1,1509 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
+; RUN: llc < %s -mtriple=x86_64-unknown-linux-gnu -mattr=+cmov | FileCheck %s --check-prefix=X64
+; RUN: llc < %s -mtriple=i386-unknown-linux-gnu -mattr=+cmov | FileCheck %s --check-prefix=X32
+; RUN: llc < %s -mtriple=i386-unknown-linux-gnu -mattr=-cmov | FileCheck %s --check-prefix=X32-NOCMOV
+
+; Test basic ct.select functionality for scalar types
+
+define i8 @test_ctselect_i8(i1 %cond, i8 %a, i8 %b) {
+; X64-LABEL: test_ctselect_i8:
+; X64: # %bb.0:
+; X64-NEXT: movl %edi, %eax
+; X64-NEXT: xorl %edx, %esi
+; X64-NEXT: andb $1, %al
+; X64-NEXT: negb %al
+; X64-NEXT: andb %sil, %al
+; X64-NEXT: xorb %dl, %al
+; X64-NEXT: # kill: def $al killed $al killed $eax
+; X64-NEXT: retq
+;
+; X32-LABEL: test_ctselect_i8:
+; X32: # %bb.0:
+; X32-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: movzbl {{[0-9]+}}(%esp), %edx
+; X32-NEXT: xorb %cl, %dl
+; X32-NEXT: andb $1, %al
+; X32-NEXT: negb %al
+; X32-NEXT: andb %dl, %al
+; X32-NEXT: xorb %cl, %al
+; X32-NEXT: retl
+;
+; X32-NOCMOV-LABEL: test_ctselect_i8:
+; X32-NOCMOV: # %bb.0:
+; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %edx
+; X32-NOCMOV-NEXT: xorb %cl, %dl
+; X32-NOCMOV-NEXT: andb $1, %al
+; X32-NOCMOV-NEXT: negb %al
+; X32-NOCMOV-NEXT: andb %dl, %al
+; X32-NOCMOV-NEXT: xorb %cl, %al
+; X32-NOCMOV-NEXT: retl
+ %result = call i8 @llvm.ct.select.i8(i1 %cond, i8 %a, i8 %b)
+ ret i8 %result
+}
+
+define i32 @test_ctselect_i32(i1 %cond, i32 %a, i32 %b) {
+; X64-LABEL: test_ctselect_i32:
+; X64: # %bb.0:
+; X64-NEXT: movl %edi, %eax
+; X64-NEXT: xorl %edx, %esi
+; X64-NEXT: andl $1, %eax
+; X64-NEXT: negl %eax
+; X64-NEXT: andl %esi, %eax
+; X64-NEXT: xorl %edx, %eax
+; X64-NEXT: retq
+;
+; X32-LABEL: test_ctselect_i32:
+; X32: # %bb.0:
+; X32-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NEXT: xorl %ecx, %edx
+; X32-NEXT: andl $1, %eax
+; X32-NEXT: negl %eax
+; X32-NEXT: andl %edx, %eax
+; X32-NEXT: xorl %ecx, %eax
+; X32-NEXT: retl
+;
+; X32-NOCMOV-LABEL: test_ctselect_i32:
+; X32-NOCMOV: # %bb.0:
+; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NOCMOV-NEXT: xorl %ecx, %edx
+; X32-NOCMOV-NEXT: andl $1, %eax
+; X32-NOCMOV-NEXT: negl %eax
+; X32-NOCMOV-NEXT: andl %edx, %eax
+; X32-NOCMOV-NEXT: xorl %ecx, %eax
+; X32-NOCMOV-NEXT: retl
+ %result = call i32 @llvm.ct.select.i32(i1 %cond, i32 %a, i32 %b)
+ ret i32 %result
+}
+
+define i64 @test_ctselect_i64(i1 %cond, i64 %a, i64 %b) {
+; X64-LABEL: test_ctselect_i64:
+; X64: # %bb.0:
+; X64-NEXT: movl %edi, %eax
+; X64-NEXT: xorq %rdx, %rsi
+; X64-NEXT: andl $1, %eax
+; X64-NEXT: negq %rax
+; X64-NEXT: andq %rsi, %rax
+; X64-NEXT: xorq %rdx, %rax
+; X64-NEXT: retq
+;
+; X32-LABEL: test_ctselect_i64:
+; X32: # %bb.0:
+; X32-NEXT: pushl %esi
+; X32-NEXT: .cfi_def_cfa_offset 8
+; X32-NEXT: .cfi_offset %esi, -8
+; X32-NEXT: movzbl {{[0-9]+}}(%esp), %esi
+; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: xorl %edx, %eax
+; X32-NEXT: andl $1, %esi
+; X32-NEXT: negl %esi
+; X32-NEXT: andl %esi, %eax
+; X32-NEXT: xorl %edx, %eax
+; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NEXT: xorl %ecx, %edx
+; X32-NEXT: andl %esi, %edx
+; X32-NEXT: xorl %ecx, %edx
+; X32-NEXT: popl %esi
+; X32-NEXT: .cfi_def_cfa_offset 4
+; X32-NEXT: retl
+;
+; X32-NOCMOV-LABEL: test_ctselect_i64:
+; X32-NOCMOV: # %bb.0:
+; X32-NOCMOV-NEXT: pushl %esi
+; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 8
+; X32-NOCMOV-NEXT: .cfi_offset %esi, -8
+; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %esi
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: xorl %edx, %eax
+; X32-NOCMOV-NEXT: andl $1, %esi
+; X32-NOCMOV-NEXT: negl %esi
+; X32-NOCMOV-NEXT: andl %esi, %eax
+; X32-NOCMOV-NEXT: xorl %edx, %eax
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NOCMOV-NEXT: xorl %ecx, %edx
+; X32-NOCMOV-NEXT: andl %esi, %edx
+; X32-NOCMOV-NEXT: xorl %ecx, %edx
+; X32-NOCMOV-NEXT: popl %esi
+; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 4
+; X32-NOCMOV-NEXT: retl
+ %result = call i64 @llvm.ct.select.i64(i1 %cond, i64 %a, i64 %b)
+ ret i64 %result
+}
+
+define float @test_ctselect_f32(i1 %cond, float %a, float %b) {
+; X64-LABEL: test_ctselect_f32:
+; X64: # %bb.0:
+; X64-NEXT: movd %xmm1, %eax
+; X64-NEXT: pxor %xmm1, %xmm0
+; X64-NEXT: movd %xmm0, %ecx
+; X64-NEXT: andl $1, %edi
+; X64-NEXT: negl %edi
+; X64-NEXT: andl %ecx, %edi
+; X64-NEXT: xorl %eax, %edi
+; X64-NEXT: movd %edi, %xmm0
+; X64-NEXT: retq
+;
+; X32-LABEL: test_ctselect_f32:
+; X32: # %bb.0:
+; X32-NEXT: pushl %eax
+; X32-NEXT: .cfi_def_cfa_offset 8
+; X32-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NEXT: xorl %ecx, %edx
+; X32-NEXT: andl $1, %eax
+; X32-NEXT: negl %eax
+; X32-NEXT: andl %edx, %eax
+; X32-NEXT: xorl %ecx, %eax
+; X32-NEXT: movl %eax, (%esp)
+; X32-NEXT: flds (%esp)
+; X32-NEXT: popl %eax
+; X32-NEXT: .cfi_def_cfa_offset 4
+; X32-NEXT: retl
+;
+; X32-NOCMOV-LABEL: test_ctselect_f32:
+; X32-NOCMOV: # %bb.0:
+; X32-NOCMOV-NEXT: pushl %eax
+; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 8
+; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NOCMOV-NEXT: xorl %ecx, %edx
+; X32-NOCMOV-NEXT: andl $1, %eax
+; X32-NOCMOV-NEXT: negl %eax
+; X32-NOCMOV-NEXT: andl %edx, %eax
+; X32-NOCMOV-NEXT: xorl %ecx, %eax
+; X32-NOCMOV-NEXT: movl %eax, (%esp)
+; X32-NOCMOV-NEXT: flds (%esp)
+; X32-NOCMOV-NEXT: popl %eax
+; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 4
+; X32-NOCMOV-NEXT: retl
+ %result = call float @llvm.ct.select.f32(i1 %cond, float %a, float %b)
+ ret float %result
+}
+
+define double @test_ctselect_f64(i1 %cond, double %a, double %b) {
+; X64-LABEL: test_ctselect_f64:
+; X64: # %bb.0:
+; X64-NEXT: # kill: def $edi killed $edi def $rdi
+; X64-NEXT: movq %xmm1, %rax
+; X64-NEXT: pxor %xmm1, %xmm0
+; X64-NEXT: movq %xmm0, %rcx
+; X64-NEXT: andl $1, %edi
+; X64-NEXT: negq %rdi
+; X64-NEXT: andq %rcx, %rdi
+; X64-NEXT: xorq %rax, %rdi
+; X64-NEXT: movq %rdi, %xmm0
+; X64-NEXT: retq
+;
+; X32-LABEL: test_ctselect_f64:
+; X32: # %bb.0:
+; X32-NEXT: pushl %esi
+; X32-NEXT: .cfi_def_cfa_offset 8
+; X32-NEXT: subl $8, %esp
+; X32-NEXT: .cfi_def_cfa_offset 16
+; X32-NEXT: .cfi_offset %esi, -8
+; X32-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NEXT: movl {{[0-9]+}}(%esp), %esi
+; X32-NEXT: xorl %edx, %esi
+; X32-NEXT: andl $1, %ecx
+; X32-NEXT: negl %ecx
+; X32-NEXT: andl %ecx, %esi
+; X32-NEXT: xorl %edx, %esi
+; X32-NEXT: movl %esi, {{[0-9]+}}(%esp)
+; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NEXT: xorl %eax, %edx
+; X32-NEXT: andl %ecx, %edx
+; X32-NEXT: xorl %eax, %edx
+; X32-NEXT: movl %edx, (%esp)
+; X32-NEXT: fldl (%esp)
+; X32-NEXT: addl $8, %esp
+; X32-NEXT: .cfi_def_cfa_offset 8
+; X32-NEXT: popl %esi
+; X32-NEXT: .cfi_def_cfa_offset 4
+; X32-NEXT: retl
+;
+; X32-NOCMOV-LABEL: test_ctselect_f64:
+; X32-NOCMOV: # %bb.0:
+; X32-NOCMOV-NEXT: pushl %esi
+; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 8
+; X32-NOCMOV-NEXT: subl $8, %esp
+; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 16
+; X32-NOCMOV-NEXT: .cfi_offset %esi, -8
+; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %esi
+; X32-NOCMOV-NEXT: xorl %edx, %esi
+; X32-NOCMOV-NEXT: andl $1, %ecx
+; X32-NOCMOV-NEXT: negl %ecx
+; X32-NOCMOV-NEXT: andl %ecx, %esi
+; X32-NOCMOV-NEXT: xorl %edx, %esi
+; X32-NOCMOV-NEXT: movl %esi, {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NOCMOV-NEXT: xorl %eax, %edx
+; X32-NOCMOV-NEXT: andl %ecx, %edx
+; X32-NOCMOV-NEXT: xorl %eax, %edx
+; X32-NOCMOV-NEXT: movl %edx, (%esp)
+; X32-NOCMOV-NEXT: fldl (%esp)
+; X32-NOCMOV-NEXT: addl $8, %esp
+; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 8
+; X32-NOCMOV-NEXT: popl %esi
+; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 4
+; X32-NOCMOV-NEXT: retl
+ %result = call double @llvm.ct.select.f64(i1 %cond, double %a, double %b)
+ ret double %result
+}
+
+define ptr @test_ctselect_ptr(i1 %cond, ptr %a, ptr %b) {
+; X64-LABEL: test_ctselect_ptr:
+; X64: # %bb.0:
+; X64-NEXT: movl %edi, %eax
+; X64-NEXT: xorq %rdx, %rsi
+; X64-NEXT: andl $1, %eax
+; X64-NEXT: negq %rax
+; X64-NEXT: andq %rsi, %rax
+; X64-NEXT: xorq %rdx, %rax
+; X64-NEXT: retq
+;
+; X32-LABEL: test_ctselect_ptr:
+; X32: # %bb.0:
+; X32-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NEXT: xorl %ecx, %edx
+; X32-NEXT: andl $1, %eax
+; X32-NEXT: negl %eax
+; X32-NEXT: andl %edx, %eax
+; X32-NEXT: xorl %ecx, %eax
+; X32-NEXT: retl
+;
+; X32-NOCMOV-LABEL: test_ctselect_ptr:
+; X32-NOCMOV: # %bb.0:
+; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NOCMOV-NEXT: xorl %ecx, %edx
+; X32-NOCMOV-NEXT: andl $1, %eax
+; X32-NOCMOV-NEXT: negl %eax
+; X32-NOCMOV-NEXT: andl %edx, %eax
+; X32-NOCMOV-NEXT: xorl %ecx, %eax
+; X32-NOCMOV-NEXT: retl
+ %result = call ptr @llvm.ct.select.p0(i1 %cond, ptr %a, ptr %b)
+ ret ptr %result
+}
+
+; Test with constant conditions
+define i32 @test_ctselect_const_true(i32 %a, i32 %b) {
+; X64-LABEL: test_ctselect_const_true:
+; X64: # %bb.0:
+; X64-NEXT: movl %edi, %eax
+; X64-NEXT: xorl %esi, %eax
+; X64-NEXT: xorl %esi, %eax
+; X64-NEXT: retq
+;
+; X32-LABEL: test_ctselect_const_true:
+; X32: # %bb.0:
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: xorl %ecx, %eax
+; X32-NEXT: xorl %ecx, %eax
+; X32-NEXT: retl
+;
+; X32-NOCMOV-LABEL: test_ctselect_const_true:
+; X32-NOCMOV: # %bb.0:
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: xorl %ecx, %eax
+; X32-NOCMOV-NEXT: xorl %ecx, %eax
+; X32-NOCMOV-NEXT: retl
+ %result = call i32 @llvm.ct.select.i32(i1 true, i32 %a, i32 %b)
+ ret i32 %result
+}
+
+define i32 @test_ctselect_const_false(i32 %a, i32 %b) {
+; X64-LABEL: test_ctselect_const_false:
+; X64: # %bb.0:
+; X64-NEXT: movl %esi, %eax
+; X64-NEXT: retq
+;
+; X32-LABEL: test_ctselect_const_false:
+; X32: # %bb.0:
+; X32-NEXT: xorl %eax, %eax
+; X32-NEXT: xorl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: retl
+;
+; X32-NOCMOV-LABEL: test_ctselect_const_false:
+; X32-NOCMOV: # %bb.0:
+; X32-NOCMOV-NEXT: xorl %eax, %eax
+; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: retl
+ %result = call i32 @llvm.ct.select.i32(i1 false, i32 %a, i32 %b)
+ ret i32 %result
+}
+
+; Test with comparison conditions
+define i32 @test_ctselect_icmp_eq(i32 %x, i32 %y, i32 %a, i32 %b) {
+; X64-LABEL: test_ctselect_icmp_eq:
+; X64: # %bb.0:
+; X64-NEXT: xorl %eax, %eax
+; X64-NEXT: cmpl %esi, %edi
+; X64-NEXT: sete %al
+; X64-NEXT: xorl %ecx, %edx
+; X64-NEXT: negl %eax
+; X64-NEXT: andl %edx, %eax
+; X64-NEXT: xorl %ecx, %eax
+; X64-NEXT: retq
+;
+; X32-LABEL: test_ctselect_icmp_eq:
+; X32: # %bb.0:
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NEXT: xorl %eax, %eax
+; X32-NEXT: cmpl {{[0-9]+}}(%esp), %edx
+; X32-NEXT: sete %al
+; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NEXT: xorl %ecx, %edx
+; X32-NEXT: negl %eax
+; X32-NEXT: andl %edx, %eax
+; X32-NEXT: xorl %ecx, %eax
+; X32-NEXT: retl
+;
+; X32-NOCMOV-LABEL: test_ctselect_icmp_eq:
+; X32-NOCMOV: # %bb.0:
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NOCMOV-NEXT: xorl %eax, %eax
+; X32-NOCMOV-NEXT: cmpl {{[0-9]+}}(%esp), %edx
+; X32-NOCMOV-NEXT: sete %al
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NOCMOV-NEXT: xorl %ecx, %edx
+; X32-NOCMOV-NEXT: negl %eax
+; X32-NOCMOV-NEXT: andl %edx, %eax
+; X32-NOCMOV-NEXT: xorl %ecx, %eax
+; X32-NOCMOV-NEXT: retl
+ %cond = icmp eq i32 %x, %y
+ %result = call i32 @llvm.ct.select.i32(i1 %cond, i32 %a, i32 %b)
+ ret i32 %result
+}
+
+define i32 @test_ctselect_icmp_ult(i32 %x, i32 %y, i32 %a, i32 %b) {
+; X64-LABEL: test_ctselect_icmp_ult:
+; X64: # %bb.0:
+; X64-NEXT: xorl %ecx, %edx
+; X64-NEXT: xorl %eax, %eax
+; X64-NEXT: cmpl %esi, %edi
+; X64-NEXT: sbbl %eax, %eax
+; X64-NEXT: andl %edx, %eax
+; X64-NEXT: xorl %ecx, %eax
+; X64-NEXT: retq
+;
+; X32-LABEL: test_ctselect_icmp_ult:
+; X32: # %bb.0:
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: xorl %edx, %edx
+; X32-NEXT: cmpl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: sbbl %edx, %edx
+; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: xorl %ecx, %eax
+; X32-NEXT: andl %edx, %eax
+; X32-NEXT: xorl %ecx, %eax
+; X32-NEXT: retl
+;
+; X32-NOCMOV-LABEL: test_ctselect_icmp_ult:
+; X32-NOCMOV: # %bb.0:
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: xorl %edx, %edx
+; X32-NOCMOV-NEXT: cmpl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: sbbl %edx, %edx
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: xorl %ecx, %eax
+; X32-NOCMOV-NEXT: andl %edx, %eax
+; X32-NOCMOV-NEXT: xorl %ecx, %eax
+; X32-NOCMOV-NEXT: retl
+ %cond = icmp ult i32 %x, %y
+ %result = call i32 @llvm.ct.select.i32(i1 %cond, i32 %a, i32 %b)
+ ret i32 %result
+}
+
+define float @test_ctselect_fcmp_oeq(float %x, float %y, float %a, float %b) {
+; X64-LABEL: test_ctselect_fcmp_oeq:
+; X64: # %bb.0:
+; X64-NEXT: movd %xmm3, %eax
+; X64-NEXT: cmpeqss %xmm1, %xmm0
+; X64-NEXT: pxor %xmm3, %xmm2
+; X64-NEXT: pand %xmm0, %xmm2
+; X64-NEXT: movd %xmm2, %ecx
+; X64-NEXT: xorl %eax, %ecx
+; X64-NEXT: movd %ecx, %xmm0
+; X64-NEXT: retq
+;
+; X32-LABEL: test_ctselect_fcmp_oeq:
+; X32: # %bb.0:
+; X32-NEXT: pushl %eax
+; X32-NEXT: .cfi_def_cfa_offset 8
+; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NEXT: fucompi %st(1), %st
+; X32-NEXT: fstp %st(0)
+; X32-NEXT: setnp %cl
+; X32-NEXT: sete %dl
+; X32-NEXT: andb %cl, %dl
+; X32-NEXT: movzbl %dl, %ecx
+; X32-NEXT: negl %ecx
+; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NEXT: xorl %eax, %edx
+; X32-NEXT: andl %ecx, %edx
+; X32-NEXT: xorl %eax, %edx
+; X32-NEXT: movl %edx, (%esp)
+; X32-NEXT: flds (%esp)
+; X32-NEXT: popl %eax
+; X32-NEXT: .cfi_def_cfa_offset 4
+; X32-NEXT: retl
+;
+; X32-NOCMOV-LABEL: test_ctselect_fcmp_oeq:
+; X32-NOCMOV: # %bb.0:
+; X32-NOCMOV-NEXT: pushl %eax
+; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 8
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: fucompp
+; X32-NOCMOV-NEXT: fnstsw %ax
+; X32-NOCMOV-NEXT: # kill: def $ah killed $ah killed $ax
+; X32-NOCMOV-NEXT: sahf
+; X32-NOCMOV-NEXT: setnp %al
+; X32-NOCMOV-NEXT: sete %dl
+; X32-NOCMOV-NEXT: andb %al, %dl
+; X32-NOCMOV-NEXT: movzbl %dl, %eax
+; X32-NOCMOV-NEXT: negl %eax
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NOCMOV-NEXT: xorl %ecx, %edx
+; X32-NOCMOV-NEXT: andl %eax, %edx
+; X32-NOCMOV-NEXT: xorl %ecx, %edx
+; X32-NOCMOV-NEXT: movl %edx, (%esp)
+; X32-NOCMOV-NEXT: flds (%esp)
+; X32-NOCMOV-NEXT: popl %eax
+; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 4
+; X32-NOCMOV-NEXT: retl
+ %cond = fcmp oeq float %x, %y
+ %result = call float @llvm.ct.select.f32(i1 %cond, float %a, float %b)
+ ret float %result
+}
+
+; Test with memory operands
+define i32 @test_ctselect_load(i1 %cond, ptr %p1, ptr %p2) {
+; X64-LABEL: test_ctselect_load:
+; X64: # %bb.0:
+; X64-NEXT: movl (%rdx), %ecx
+; X64-NEXT: movl (%rsi), %eax
+; X64-NEXT: xorl %ecx, %eax
+; X64-NEXT: andl $1, %edi
+; X64-NEXT: negl %edi
+; X64-NEXT: andl %edi, %eax
+; X64-NEXT: xorl %ecx, %eax
+; X64-NEXT: retq
+;
+; X32-LABEL: test_ctselect_load:
+; X32: # %bb.0:
+; X32-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NEXT: movl (%edx), %edx
+; X32-NEXT: movl (%ecx), %ecx
+; X32-NEXT: xorl %edx, %ecx
+; X32-NEXT: andl $1, %eax
+; X32-NEXT: negl %eax
+; X32-NEXT: andl %ecx, %eax
+; X32-NEXT: xorl %edx, %eax
+; X32-NEXT: retl
+;
+; X32-NOCMOV-LABEL: test_ctselect_load:
+; X32-NOCMOV: # %bb.0:
+; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NOCMOV-NEXT: movl (%edx), %edx
+; X32-NOCMOV-NEXT: movl (%ecx), %ecx
+; X32-NOCMOV-NEXT: xorl %edx, %ecx
+; X32-NOCMOV-NEXT: andl $1, %eax
+; X32-NOCMOV-NEXT: negl %eax
+; X32-NOCMOV-NEXT: andl %ecx, %eax
+; X32-NOCMOV-NEXT: xorl %edx, %eax
+; X32-NOCMOV-NEXT: retl
+ %a = load i32, ptr %p1
+ %b = load i32, ptr %p2
+ %result = call i32 @llvm.ct.select.i32(i1 %cond, i32 %a, i32 %b)
+ ret i32 %result
+}
+
+; Test nested ctselect calls
+define i32 @test_ctselect_nested(i1 %cond1, i1 %cond2, i32 %a, i32 %b, i32 %c) {
+; X64-LABEL: test_ctselect_nested:
+; X64: # %bb.0:
+; X64-NEXT: movl %edi, %eax
+; X64-NEXT: xorl %ecx, %edx
+; X64-NEXT: andl $1, %esi
+; X64-NEXT: negl %esi
+; X64-NEXT: andl %edx, %esi
+; X64-NEXT: xorl %r8d, %ecx
+; X64-NEXT: xorl %esi, %ecx
+; X64-NEXT: andl $1, %eax
+; X64-NEXT: negl %eax
+; X64-NEXT: andl %ecx, %eax
+; X64-NEXT: xorl %r8d, %eax
+; X64-NEXT: retq
+;
+; X32-LABEL: test_ctselect_nested:
+; X32: # %bb.0:
+; X32-NEXT: pushl %edi
+; X32-NEXT: .cfi_def_cfa_offset 8
+; X32-NEXT: pushl %esi
+; X32-NEXT: .cfi_def_cfa_offset 12
+; X32-NEXT: .cfi_offset %esi, -12
+; X32-NEXT: .cfi_offset %edi, -8
+; X32-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: movzbl {{[0-9]+}}(%esp), %esi
+; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NEXT: movl {{[0-9]+}}(%esp), %edi
+; X32-NEXT: xorl %edx, %edi
+; X32-NEXT: andl $1, %esi
+; X32-NEXT: negl %esi
+; X32-NEXT: andl %edi, %esi
+; X32-NEXT: xorl %ecx, %edx
+; X32-NEXT: xorl %esi, %edx
+; X32-NEXT: andl $1, %eax
+; X32-NEXT: negl %eax
+; X32-NEXT: andl %edx, %eax
+; X32-NEXT: xorl %ecx, %eax
+; X32-NEXT: popl %esi
+; X32-NEXT: .cfi_def_cfa_offset 8
+; X32-NEXT: popl %edi
+; X32-NEXT: .cfi_def_cfa_offset 4
+; X32-NEXT: retl
+;
+; X32-NOCMOV-LABEL: test_ctselect_nested:
+; X32-NOCMOV: # %bb.0:
+; X32-NOCMOV-NEXT: pushl %edi
+; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 8
+; X32-NOCMOV-NEXT: pushl %esi
+; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 12
+; X32-NOCMOV-NEXT: .cfi_offset %esi, -12
+; X32-NOCMOV-NEXT: .cfi_offset %edi, -8
+; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %esi
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edi
+; X32-NOCMOV-NEXT: xorl %edx, %edi
+; X32-NOCMOV-NEXT: andl $1, %esi
+; X32-NOCMOV-NEXT: negl %esi
+; X32-NOCMOV-NEXT: andl %edi, %esi
+; X32-NOCMOV-NEXT: xorl %ecx, %edx
+; X32-NOCMOV-NEXT: xorl %esi, %edx
+; X32-NOCMOV-NEXT: andl $1, %eax
+; X32-NOCMOV-NEXT: negl %eax
+; X32-NOCMOV-NEXT: andl %edx, %eax
+; X32-NOCMOV-NEXT: xorl %ecx, %eax
+; X32-NOCMOV-NEXT: popl %esi
+; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 8
+; X32-NOCMOV-NEXT: popl %edi
+; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 4
+; X32-NOCMOV-NEXT: retl
+ %inner = call i32 @llvm.ct.select.i32(i1 %cond2, i32 %a, i32 %b)
+ %result = call i32 @llvm.ct.select.i32(i1 %cond1, i32 %inner, i32 %c)
+ ret i32 %result
+}
+
+; Test nested CTSELECT pattern with AND merging on i1 values
+; Pattern: ctselect C0, (ctselect C1, X, Y), Y -> ctselect (C0 & C1), X, Y
+; This optimization only applies when selecting between i1 values (boolean logic)
+define i32 @test_ctselect_nested_and_i1_to_i32(i1 %c0, i1 %c1, i32 %x, i32 %y) {
+; X64-LABEL: test_ctselect_nested_and_i1_to_i32:
+; X64: # %bb.0:
+; X64-NEXT: movl %edi, %eax
+; X64-NEXT: andl %esi, %eax
+; X64-NEXT: xorl %ecx, %edx
+; X64-NEXT: andl $1, %eax
+; X64-NEXT: negl %eax
+; X64-NEXT: andl %edx, %eax
+; X64-NEXT: xorl %ecx, %eax
+; X64-NEXT: retq
+;
+; X32-LABEL: test_ctselect_nested_and_i1_to_i32:
+; X32: # %bb.0:
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: andb {{[0-9]+}}(%esp), %al
+; X32-NEXT: movzbl %al, %eax
+; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NEXT: xorl %ecx, %edx
+; X32-NEXT: andl $1, %eax
+; X32-NEXT: negl %eax
+; X32-NEXT: andl %edx, %eax
+; X32-NEXT: xorl %ecx, %eax
+; X32-NEXT: retl
+;
+; X32-NOCMOV-LABEL: test_ctselect_nested_and_i1_to_i32:
+; X32-NOCMOV: # %bb.0:
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: andb {{[0-9]+}}(%esp), %al
+; X32-NOCMOV-NEXT: movzbl %al, %eax
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NOCMOV-NEXT: xorl %ecx, %edx
+; X32-NOCMOV-NEXT: andl $1, %eax
+; X32-NOCMOV-NEXT: negl %eax
+; X32-NOCMOV-NEXT: andl %edx, %eax
+; X32-NOCMOV-NEXT: xorl %ecx, %eax
+; X32-NOCMOV-NEXT: retl
+ %inner = call i1 @llvm.ct.select.i1(i1 %c1, i1 true, i1 false)
+ %cond = call i1 @llvm.ct.select.i1(i1 %c0, i1 %inner, i1 false)
+ %result = call i32 @llvm.ct.select.i32(i1 %cond, i32 %x, i32 %y)
+ ret i32 %result
+}
+
+; Test nested CTSELECT pattern with OR merging on i1 values
+; Pattern: ctselect C0, X, (ctselect C1, X, Y) -> ctselect (C0 | C1), X, Y
+; This optimization only applies when selecting between i1 values (boolean logic)
+define i32 @test_ctselect_nested_or_i1_to_i32(i1 %c0, i1 %c1, i32 %x, i32 %y) {
+; X64-LABEL: test_ctselect_nested_or_i1_to_i32:
+; X64: # %bb.0:
+; X64-NEXT: movl %edi, %eax
+; X64-NEXT: orl %esi, %eax
+; X64-NEXT: xorl %ecx, %edx
+; X64-NEXT: andl $1, %eax
+; X64-NEXT: negl %eax
+; X64-NEXT: andl %edx, %eax
+; X64-NEXT: xorl %ecx, %eax
+; X64-NEXT: retq
+;
+; X32-LABEL: test_ctselect_nested_or_i1_to_i32:
+; X32: # %bb.0:
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: orb {{[0-9]+}}(%esp), %al
+; X32-NEXT: movzbl %al, %eax
+; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NEXT: xorl %ecx, %edx
+; X32-NEXT: andl $1, %eax
+; X32-NEXT: negl %eax
+; X32-NEXT: andl %edx, %eax
+; X32-NEXT: xorl %ecx, %eax
+; X32-NEXT: retl
+;
+; X32-NOCMOV-LABEL: test_ctselect_nested_or_i1_to_i32:
+; X32-NOCMOV: # %bb.0:
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: orb {{[0-9]+}}(%esp), %al
+; X32-NOCMOV-NEXT: movzbl %al, %eax
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NOCMOV-NEXT: xorl %ecx, %edx
+; X32-NOCMOV-NEXT: andl $1, %eax
+; X32-NOCMOV-NEXT: negl %eax
+; X32-NOCMOV-NEXT: andl %edx, %eax
+; X32-NOCMOV-NEXT: xorl %ecx, %eax
+; X32-NOCMOV-NEXT: retl
+ %inner = call i1 @llvm.ct.select.i1(i1 %c1, i1 true, i1 false)
+ %cond = call i1 @llvm.ct.select.i1(i1 %c0, i1 true, i1 %inner)
+ %result = call i32 @llvm.ct.select.i32(i1 %cond, i32 %x, i32 %y)
+ ret i32 %result
+}
+
+; Test double nested CTSELECT with recursive AND merging
+; Pattern: ctselect C0, (ctselect C1, (ctselect C2, X, Y), Y), Y
+; -> ctselect C0, (ctselect (C1 & C2), X, Y), Y
+; -> ctselect (C0 & (C1 & C2)), X, Y
+; This tests that the optimization can be applied recursively
+define i32 @test_ctselect_double_nested_and_i1(i1 %c0, i1 %c1, i1 %c2, i32 %x, i32 %y) {
+; X64-LABEL: test_ctselect_double_nested_and_i1:
+; X64: # %bb.0:
+; X64-NEXT: movl %esi, %eax
+; X64-NEXT: andl %edx, %eax
+; X64-NEXT: andl %edi, %eax
+; X64-NEXT: xorl %r8d, %ecx
+; X64-NEXT: andl $1, %eax
+; X64-NEXT: negl %eax
+; X64-NEXT: andl %ecx, %eax
+; X64-NEXT: xorl %r8d, %eax
+; X64-NEXT: retq
+;
+; X32-LABEL: test_ctselect_double_nested_and_i1:
+; X32: # %bb.0:
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: andb {{[0-9]+}}(%esp), %al
+; X32-NEXT: andb {{[0-9]+}}(%esp), %al
+; X32-NEXT: movzbl %al, %eax
+; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NEXT: xorl %ecx, %edx
+; X32-NEXT: andl $1, %eax
+; X32-NEXT: negl %eax
+; X32-NEXT: andl %edx, %eax
+; X32-NEXT: xorl %ecx, %eax
+; X32-NEXT: retl
+;
+; X32-NOCMOV-LABEL: test_ctselect_double_nested_and_i1:
+; X32-NOCMOV: # %bb.0:
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: andb {{[0-9]+}}(%esp), %al
+; X32-NOCMOV-NEXT: andb {{[0-9]+}}(%esp), %al
+; X32-NOCMOV-NEXT: movzbl %al, %eax
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NOCMOV-NEXT: xorl %ecx, %edx
+; X32-NOCMOV-NEXT: andl $1, %eax
+; X32-NOCMOV-NEXT: negl %eax
+; X32-NOCMOV-NEXT: andl %edx, %eax
+; X32-NOCMOV-NEXT: xorl %ecx, %eax
+; X32-NOCMOV-NEXT: retl
+ %inner2 = call i1 @llvm.ct.select.i1(i1 %c2, i1 true, i1 false)
+ %inner1 = call i1 @llvm.ct.select.i1(i1 %c1, i1 %inner2, i1 false)
+ %cond = call i1 @llvm.ct.select.i1(i1 %c0, i1 %inner1, i1 false)
+ %result = call i32 @llvm.ct.select.i32(i1 %cond, i32 %x, i32 %y)
+ ret i32 %result
+}
+
+; Vector CTSELECT Tests
+; ============================================================================
+
+; Test vector CTSELECT with v4i32 (128-bit vector with single i1 mask)
+; NOW CONSTANT-TIME: Uses bitwise XOR/AND operations instead of branches!
+define <4 x i32> @test_ctselect_v4i32(i1 %cond, <4 x i32> %a, <4 x i32> %b) {
+; X64-LABEL: test_ctselect_v4i32:
+; X64: # %bb.0:
+; X64-NEXT: pxor %xmm1, %xmm0
+; X64-NEXT: movd %edi, %xmm2
+; X64-NEXT: pshufd {{.*#+}} xmm2 = xmm2[0,0,0,0]
+; X64-NEXT: pslld $31, %xmm2
+; X64-NEXT: psrad $31, %xmm2
+; X64-NEXT: pand %xmm2, %xmm0
+; X64-NEXT: pxor %xmm1, %xmm0
+; X64-NEXT: retq
+;
+; X32-LABEL: test_ctselect_v4i32:
+; X32: # %bb.0:
+; X32-NEXT: pushl %ebp
+; X32-NEXT: .cfi_def_cfa_offset 8
+; X32-NEXT: pushl %ebx
+; X32-NEXT: .cfi_def_cfa_offset 12
+; X32-NEXT: pushl %edi
+; X32-NEXT: .cfi_def_cfa_offset 16
+; X32-NEXT: pushl %esi
+; X32-NEXT: .cfi_def_cfa_offset 20
+; X32-NEXT: .cfi_offset %esi, -20
+; X32-NEXT: .cfi_offset %edi, -16
+; X32-NEXT: .cfi_offset %ebx, -12
+; X32-NEXT: .cfi_offset %ebp, -8
+; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: movl {{[0-9]+}}(%esp), %esi
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ebp
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ebx
+; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NEXT: xorl %ebx, %edx
+; X32-NEXT: movzbl {{[0-9]+}}(%esp), %edi
+; X32-NEXT: andl $1, %edi
+; X32-NEXT: negl %edi
+; X32-NEXT: andl %edi, %edx
+; X32-NEXT: xorl %ebx, %edx
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ebx
+; X32-NEXT: xorl %ebp, %ebx
+; X32-NEXT: andl %edi, %ebx
+; X32-NEXT: xorl %ebp, %ebx
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ebp
+; X32-NEXT: xorl %esi, %ebp
+; X32-NEXT: andl %edi, %ebp
+; X32-NEXT: xorl %esi, %ebp
+; X32-NEXT: movl {{[0-9]+}}(%esp), %esi
+; X32-NEXT: xorl %ecx, %esi
+; X32-NEXT: andl %edi, %esi
+; X32-NEXT: xorl %ecx, %esi
+; X32-NEXT: movl %esi, 12(%eax)
+; X32-NEXT: movl %ebp, 8(%eax)
+; X32-NEXT: movl %ebx, 4(%eax)
+; X32-NEXT: movl %edx, (%eax)
+; X32-NEXT: popl %esi
+; X32-NEXT: .cfi_def_cfa_offset 16
+; X32-NEXT: popl %edi
+; X32-NEXT: .cfi_def_cfa_offset 12
+; X32-NEXT: popl %ebx
+; X32-NEXT: .cfi_def_cfa_offset 8
+; X32-NEXT: popl %ebp
+; X32-NEXT: .cfi_def_cfa_offset 4
+; X32-NEXT: retl $4
+;
+; X32-NOCMOV-LABEL: test_ctselect_v4i32:
+; X32-NOCMOV: # %bb.0:
+; X32-NOCMOV-NEXT: pushl %ebp
+; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 8
+; X32-NOCMOV-NEXT: pushl %ebx
+; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 12
+; X32-NOCMOV-NEXT: pushl %edi
+; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 16
+; X32-NOCMOV-NEXT: pushl %esi
+; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 20
+; X32-NOCMOV-NEXT: .cfi_offset %esi, -20
+; X32-NOCMOV-NEXT: .cfi_offset %edi, -16
+; X32-NOCMOV-NEXT: .cfi_offset %ebx, -12
+; X32-NOCMOV-NEXT: .cfi_offset %ebp, -8
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %esi
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebp
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebx
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NOCMOV-NEXT: xorl %ebx, %edx
+; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %edi
+; X32-NOCMOV-NEXT: andl $1, %edi
+; X32-NOCMOV-NEXT: negl %edi
+; X32-NOCMOV-NEXT: andl %edi, %edx
+; X32-NOCMOV-NEXT: xorl %ebx, %edx
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebx
+; X32-NOCMOV-NEXT: xorl %ebp, %ebx
+; X32-NOCMOV-NEXT: andl %edi, %ebx
+; X32-NOCMOV-NEXT: xorl %ebp, %ebx
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebp
+; X32-NOCMOV-NEXT: xorl %esi, %ebp
+; X32-NOCMOV-NEXT: andl %edi, %ebp
+; X32-NOCMOV-NEXT: xorl %esi, %ebp
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %esi
+; X32-NOCMOV-NEXT: xorl %ecx, %esi
+; X32-NOCMOV-NEXT: andl %edi, %esi
+; X32-NOCMOV-NEXT: xorl %ecx, %esi
+; X32-NOCMOV-NEXT: movl %esi, 12(%eax)
+; X32-NOCMOV-NEXT: movl %ebp, 8(%eax)
+; X32-NOCMOV-NEXT: movl %ebx, 4(%eax)
+; X32-NOCMOV-NEXT: movl %edx, (%eax)
+; X32-NOCMOV-NEXT: popl %esi
+; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 16
+; X32-NOCMOV-NEXT: popl %edi
+; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 12
+; X32-NOCMOV-NEXT: popl %ebx
+; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 8
+; X32-NOCMOV-NEXT: popl %ebp
+; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 4
+; X32-NOCMOV-NEXT: retl $4
+ %result = call <4 x i32> @llvm.ct.select.v4i32(i1 %cond, <4 x i32> %a, <4 x i32> %b)
+ ret <4 x i32> %result
+}
+define <4 x float> @test_ctselect_v4f32(i1 %cond, <4 x float> %a, <4 x float> %b) {
+; X64-LABEL: test_ctselect_v4f32:
+; X64: # %bb.0:
+; X64-NEXT: pxor %xmm1, %xmm0
+; X64-NEXT: movd %edi, %xmm2
+; X64-NEXT: pshufd {{.*#+}} xmm2 = xmm2[0,0,0,0]
+; X64-NEXT: pslld $31, %xmm2
+; X64-NEXT: psrad $31, %xmm2
+; X64-NEXT: pand %xmm2, %xmm0
+; X64-NEXT: pxor %xmm1, %xmm0
+; X64-NEXT: retq
+;
+; X32-LABEL: test_ctselect_v4f32:
+; X32: # %bb.0:
+; X32-NEXT: pushl %ebp
+; X32-NEXT: .cfi_def_cfa_offset 8
+; X32-NEXT: pushl %ebx
+; X32-NEXT: .cfi_def_cfa_offset 12
+; X32-NEXT: pushl %edi
+; X32-NEXT: .cfi_def_cfa_offset 16
+; X32-NEXT: pushl %esi
+; X32-NEXT: .cfi_def_cfa_offset 20
+; X32-NEXT: .cfi_offset %esi, -20
+; X32-NEXT: .cfi_offset %edi, -16
+; X32-NEXT: .cfi_offset %ebx, -12
+; X32-NEXT: .cfi_offset %ebp, -8
+; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: movl {{[0-9]+}}(%esp), %esi
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ebp
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ebx
+; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NEXT: xorl %ebx, %edx
+; X32-NEXT: movzbl {{[0-9]+}}(%esp), %edi
+; X32-NEXT: andl $1, %edi
+; X32-NEXT: negl %edi
+; X32-NEXT: andl %edi, %edx
+; X32-NEXT: xorl %ebx, %edx
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ebx
+; X32-NEXT: xorl %ebp, %ebx
+; X32-NEXT: andl %edi, %ebx
+; X32-NEXT: xorl %ebp, %ebx
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ebp
+; X32-NEXT: xorl %esi, %ebp
+; X32-NEXT: andl %edi, %ebp
+; X32-NEXT: xorl %esi, %ebp
+; X32-NEXT: movl {{[0-9]+}}(%esp), %esi
+; X32-NEXT: xorl %ecx, %esi
+; X32-NEXT: andl %edi, %esi
+; X32-NEXT: xorl %ecx, %esi
+; X32-NEXT: movl %esi, 12(%eax)
+; X32-NEXT: movl %ebp, 8(%eax)
+; X32-NEXT: movl %ebx, 4(%eax)
+; X32-NEXT: movl %edx, (%eax)
+; X32-NEXT: popl %esi
+; X32-NEXT: .cfi_def_cfa_offset 16
+; X32-NEXT: popl %edi
+; X32-NEXT: .cfi_def_cfa_offset 12
+; X32-NEXT: popl %ebx
+; X32-NEXT: .cfi_def_cfa_offset 8
+; X32-NEXT: popl %ebp
+; X32-NEXT: .cfi_def_cfa_offset 4
+; X32-NEXT: retl $4
+;
+; X32-NOCMOV-LABEL: test_ctselect_v4f32:
+; X32-NOCMOV: # %bb.0:
+; X32-NOCMOV-NEXT: pushl %ebp
+; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 8
+; X32-NOCMOV-NEXT: pushl %ebx
+; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 12
+; X32-NOCMOV-NEXT: pushl %edi
+; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 16
+; X32-NOCMOV-NEXT: pushl %esi
+; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 20
+; X32-NOCMOV-NEXT: .cfi_offset %esi, -20
+; X32-NOCMOV-NEXT: .cfi_offset %edi, -16
+; X32-NOCMOV-NEXT: .cfi_offset %ebx, -12
+; X32-NOCMOV-NEXT: .cfi_offset %ebp, -8
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %esi
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebp
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebx
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NOCMOV-NEXT: xorl %ebx, %edx
+; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %edi
+; X32-NOCMOV-NEXT: andl $1, %edi
+; X32-NOCMOV-NEXT: negl %edi
+; X32-NOCMOV-NEXT: andl %edi, %edx
+; X32-NOCMOV-NEXT: xorl %ebx, %edx
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebx
+; X32-NOCMOV-NEXT: xorl %ebp, %ebx
+; X32-NOCMOV-NEXT: andl %edi, %ebx
+; X32-NOCMOV-NEXT: xorl %ebp, %ebx
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebp
+; X32-NOCMOV-NEXT: xorl %esi, %ebp
+; X32-NOCMOV-NEXT: andl %edi, %ebp
+; X32-NOCMOV-NEXT: xorl %esi, %ebp
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %esi
+; X32-NOCMOV-NEXT: xorl %ecx, %esi
+; X32-NOCMOV-NEXT: andl %edi, %esi
+; X32-NOCMOV-NEXT: xorl %ecx, %esi
+; X32-NOCMOV-NEXT: movl %esi, 12(%eax)
+; X32-NOCMOV-NEXT: movl %ebp, 8(%eax)
+; X32-NOCMOV-NEXT: movl %ebx, 4(%eax)
+; X32-NOCMOV-NEXT: movl %edx, (%eax)
+; X32-NOCMOV-NEXT: popl %esi
+; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 16
+; X32-NOCMOV-NEXT: popl %edi
+; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 12
+; X32-NOCMOV-NEXT: popl %ebx
+; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 8
+; X32-NOCMOV-NEXT: popl %ebp
+; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 4
+; X32-NOCMOV-NEXT: retl $4
+ %result = call <4 x float> @llvm.ct.select.v4f32(i1 %cond, <4 x float> %a, <4 x float> %b)
+ ret <4 x float> %result
+}
+
+define <8 x i32> @test_ctselect_v8i32_avx(i1 %cond, <8 x i32> %a, <8 x i32> %b) {
+; X64-LABEL: test_ctselect_v8i32_avx:
+; X64: # %bb.0:
+; X64-NEXT: movd %edi, %xmm4
+; X64-NEXT: pshufd {{.*#+}} xmm4 = xmm4[0,0,0,0]
+; X64-NEXT: pslld $31, %xmm4
+; X64-NEXT: psrad $31, %xmm4
+; X64-NEXT: movdqa %xmm4, %xmm5
+; X64-NEXT: pandn %xmm2, %xmm5
+; X64-NEXT: pand %xmm4, %xmm0
+; X64-NEXT: por %xmm5, %xmm0
+; X64-NEXT: pand %xmm4, %xmm1
+; X64-NEXT: pandn %xmm3, %xmm4
+; X64-NEXT: por %xmm4, %xmm1
+; X64-NEXT: retq
+;
+; X32-LABEL: test_ctselect_v8i32_avx:
+; X32: # %bb.0:
+; X32-NEXT: pushl %ebp
+; X32-NEXT: .cfi_def_cfa_offset 8
+; X32-NEXT: pushl %ebx
+; X32-NEXT: .cfi_def_cfa_offset 12
+; X32-NEXT: pushl %edi
+; X32-NEXT: .cfi_def_cfa_offset 16
+; X32-NEXT: pushl %esi
+; X32-NEXT: .cfi_def_cfa_offset 20
+; X32-NEXT: subl $8, %esp
+; X32-NEXT: .cfi_def_cfa_offset 28
+; X32-NEXT: .cfi_offset %esi, -20
+; X32-NEXT: .cfi_offset %edi, -16
+; X32-NEXT: .cfi_offset %ebx, -12
+; X32-NEXT: .cfi_offset %ebp, -8
+; X32-NEXT: movl {{[0-9]+}}(%esp), %edi
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ebp
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ebx
+; X32-NEXT: movl {{[0-9]+}}(%esp), %esi
+; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: xorl %eax, %ecx
+; X32-NEXT: movzbl {{[0-9]+}}(%esp), %edx
+; X32-NEXT: andl $1, %edx
+; X32-NEXT: negl %edx
+; X32-NEXT: andl %edx, %ecx
+; X32-NEXT: xorl %eax, %ecx
+; X32-NEXT: movl %ecx, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
+; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: xorl %esi, %eax
+; X32-NEXT: andl %edx, %eax
+; X32-NEXT: xorl %esi, %eax
+; X32-NEXT: movl %eax, (%esp) # 4-byte Spill
+; X32-NEXT: movl {{[0-9]+}}(%esp), %esi
+; X32-NEXT: xorl %ebx, %esi
+; X32-NEXT: andl %edx, %esi
+; X32-NEXT: xorl %ebx, %esi
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ebx
+; X32-NEXT: xorl %ebp, %ebx
+; X32-NEXT: andl %edx, %ebx
+; X32-NEXT: xorl %ebp, %ebx
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ebp
+; X32-NEXT: xorl %edi, %ebp
+; X32-NEXT: andl %edx, %ebp
+; X32-NEXT: xorl %edi, %ebp
+; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: movl {{[0-9]+}}(%esp), %edi
+; X32-NEXT: xorl %eax, %edi
+; X32-NEXT: andl %edx, %edi
+; X32-NEXT: xorl %eax, %edi
+; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: xorl %eax, %ecx
+; X32-NEXT: andl %edx, %ecx
+; X32-NEXT: xorl %eax, %ecx
+; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: xorl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: andl %edx, %eax
+; X32-NEXT: xorl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NEXT: movl %eax, 28(%edx)
+; X32-NEXT: movl %ecx, 24(%edx)
+; X32-NEXT: movl %edi, 20(%edx)
+; X32-NEXT: movl %ebp, 16(%edx)
+; X32-NEXT: movl %ebx, 12(%edx)
+; X32-NEXT: movl %esi, 8(%edx)
+; X32-NEXT: movl (%esp), %eax # 4-byte Reload
+; X32-NEXT: movl %eax, 4(%edx)
+; X32-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 4-byte Reload
+; X32-NEXT: movl %eax, (%edx)
+; X32-NEXT: movl %edx, %eax
+; X32-NEXT: addl $8, %esp
+; X32-NEXT: .cfi_def_cfa_offset 20
+; X32-NEXT: popl %esi
+; X32-NEXT: .cfi_def_cfa_offset 16
+; X32-NEXT: popl %edi
+; X32-NEXT: .cfi_def_cfa_offset 12
+; X32-NEXT: popl %ebx
+; X32-NEXT: .cfi_def_cfa_offset 8
+; X32-NEXT: popl %ebp
+; X32-NEXT: .cfi_def_cfa_offset 4
+; X32-NEXT: retl $4
+;
+; X32-NOCMOV-LABEL: test_ctselect_v8i32_avx:
+; X32-NOCMOV: # %bb.0:
+; X32-NOCMOV-NEXT: pushl %ebp
+; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 8
+; X32-NOCMOV-NEXT: pushl %ebx
+; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 12
+; X32-NOCMOV-NEXT: pushl %edi
+; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 16
+; X32-NOCMOV-NEXT: pushl %esi
+; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 20
+; X32-NOCMOV-NEXT: subl $8, %esp
+; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 28
+; X32-NOCMOV-NEXT: .cfi_offset %esi, -20
+; X32-NOCMOV-NEXT: .cfi_offset %edi, -16
+; X32-NOCMOV-NEXT: .cfi_offset %ebx, -12
+; X32-NOCMOV-NEXT: .cfi_offset %ebp, -8
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edi
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebp
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebx
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %esi
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: xorl %eax, %ecx
+; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %edx
+; X32-NOCMOV-NEXT: andl $1, %edx
+; X32-NOCMOV-NEXT: negl %edx
+; X32-NOCMOV-NEXT: andl %edx, %ecx
+; X32-NOCMOV-NEXT: xorl %eax, %ecx
+; X32-NOCMOV-NEXT: movl %ecx, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: xorl %esi, %eax
+; X32-NOCMOV-NEXT: andl %edx, %eax
+; X32-NOCMOV-NEXT: xorl %esi, %eax
+; X32-NOCMOV-NEXT: movl %eax, (%esp) # 4-byte Spill
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %esi
+; X32-NOCMOV-NEXT: xorl %ebx, %esi
+; X32-NOCMOV-NEXT: andl %edx, %esi
+; X32-NOCMOV-NEXT: xorl %ebx, %esi
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebx
+; X32-NOCMOV-NEXT: xorl %ebp, %ebx
+; X32-NOCMOV-NEXT: andl %edx, %ebx
+; X32-NOCMOV-NEXT: xorl %ebp, %ebx
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebp
+; X32-NOCMOV-NEXT: xorl %edi, %ebp
+; X32-NOCMOV-NEXT: andl %edx, %ebp
+; X32-NOCMOV-NEXT: xorl %edi, %ebp
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edi
+; X32-NOCMOV-NEXT: xorl %eax, %edi
+; X32-NOCMOV-NEXT: andl %edx, %edi
+; X32-NOCMOV-NEXT: xorl %eax, %edi
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: xorl %eax, %ecx
+; X32-NOCMOV-NEXT: andl %edx, %ecx
+; X32-NOCMOV-NEXT: xorl %eax, %ecx
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: andl %edx, %eax
+; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NOCMOV-NEXT: movl %eax, 28(%edx)
+; X32-NOCMOV-NEXT: movl %ecx, 24(%edx)
+; X32-NOCMOV-NEXT: movl %edi, 20(%edx)
+; X32-NOCMOV-NEXT: movl %ebp, 16(%edx)
+; X32-NOCMOV-NEXT: movl %ebx, 12(%edx)
+; X32-NOCMOV-NEXT: movl %esi, 8(%edx)
+; X32-NOCMOV-NEXT: movl (%esp), %eax # 4-byte Reload
+; X32-NOCMOV-NEXT: movl %eax, 4(%edx)
+; X32-NOCMOV-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 4-byte Reload
+; X32-NOCMOV-NEXT: movl %eax, (%edx)
+; X32-NOCMOV-NEXT: movl %edx, %eax
+; X32-NOCMOV-NEXT: addl $8, %esp
+; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 20
+; X32-NOCMOV-NEXT: popl %esi
+; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 16
+; X32-NOCMOV-NEXT: popl %edi
+; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 12
+; X32-NOCMOV-NEXT: popl %ebx
+; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 8
+; X32-NOCMOV-NEXT: popl %ebp
+; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 4
+; X32-NOCMOV-NEXT: retl $4
+ %result = call <8 x i32> @llvm.ct.select.v8i32(i1 %cond, <8 x i32> %a, <8 x i32> %b)
+ ret <8 x i32> %result
+}
+
+define <8 x float> @test_ctselect_v8f32(i1 %cond, <8 x float> %a, <8 x float> %b) {
+; X64-LABEL: test_ctselect_v8f32:
+; X64: # %bb.0:
+; X64-NEXT: movd %edi, %xmm4
+; X64-NEXT: pshufd {{.*#+}} xmm4 = xmm4[0,0,0,0]
+; X64-NEXT: pslld $31, %xmm4
+; X64-NEXT: psrad $31, %xmm4
+; X64-NEXT: movdqa %xmm4, %xmm5
+; X64-NEXT: pandn %xmm2, %xmm5
+; X64-NEXT: pand %xmm4, %xmm0
+; X64-NEXT: por %xmm5, %xmm0
+; X64-NEXT: pand %xmm4, %xmm1
+; X64-NEXT: pandn %xmm3, %xmm4
+; X64-NEXT: por %xmm4, %xmm1
+; X64-NEXT: retq
+;
+; X32-LABEL: test_ctselect_v8f32:
+; X32: # %bb.0:
+; X32-NEXT: pushl %ebp
+; X32-NEXT: .cfi_def_cfa_offset 8
+; X32-NEXT: pushl %ebx
+; X32-NEXT: .cfi_def_cfa_offset 12
+; X32-NEXT: pushl %edi
+; X32-NEXT: .cfi_def_cfa_offset 16
+; X32-NEXT: pushl %esi
+; X32-NEXT: .cfi_def_cfa_offset 20
+; X32-NEXT: subl $8, %esp
+; X32-NEXT: .cfi_def_cfa_offset 28
+; X32-NEXT: .cfi_offset %esi, -20
+; X32-NEXT: .cfi_offset %edi, -16
+; X32-NEXT: .cfi_offset %ebx, -12
+; X32-NEXT: .cfi_offset %ebp, -8
+; X32-NEXT: movl {{[0-9]+}}(%esp), %edi
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ebp
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ebx
+; X32-NEXT: movl {{[0-9]+}}(%esp), %esi
+; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: xorl %eax, %ecx
+; X32-NEXT: movzbl {{[0-9]+}}(%esp), %edx
+; X32-NEXT: andl $1, %edx
+; X32-NEXT: negl %edx
+; X32-NEXT: andl %edx, %ecx
+; X32-NEXT: xorl %eax, %ecx
+; X32-NEXT: movl %ecx, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
+; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: xorl %esi, %eax
+; X32-NEXT: andl %edx, %eax
+; X32-NEXT: xorl %esi, %eax
+; X32-NEXT: movl %eax, (%esp) # 4-byte Spill
+; X32-NEXT: movl {{[0-9]+}}(%esp), %esi
+; X32-NEXT: xorl %ebx, %esi
+; X32-NEXT: andl %edx, %esi
+; X32-NEXT: xorl %ebx, %esi
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ebx
+; X32-NEXT: xorl %ebp, %ebx
+; X32-NEXT: andl %edx, %ebx
+; X32-NEXT: xorl %ebp, %ebx
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ebp
+; X32-NEXT: xorl %edi, %ebp
+; X32-NEXT: andl %edx, %ebp
+; X32-NEXT: xorl %edi, %ebp
+; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: movl {{[0-9]+}}(%esp), %edi
+; X32-NEXT: xorl %eax, %edi
+; X32-NEXT: andl %edx, %edi
+; X32-NEXT: xorl %eax, %edi
+; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: xorl %eax, %ecx
+; X32-NEXT: andl %edx, %ecx
+; X32-NEXT: xorl %eax, %ecx
+; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: xorl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: andl %edx, %eax
+; X32-NEXT: xorl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NEXT: movl %eax, 28(%edx)
+; X32-NEXT: movl %ecx, 24(%edx)
+; X32-NEXT: movl %edi, 20(%edx)
+; X32-NEXT: movl %ebp, 16(%edx)
+; X32-NEXT: movl %ebx, 12(%edx)
+; X32-NEXT: movl %esi, 8(%edx)
+; X32-NEXT: movl (%esp), %eax # 4-byte Reload
+; X32-NEXT: movl %eax, 4(%edx)
+; X32-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 4-byte Reload
+; X32-NEXT: movl %eax, (%edx)
+; X32-NEXT: movl %edx, %eax
+; X32-NEXT: addl $8, %esp
+; X32-NEXT: .cfi_def_cfa_offset 20
+; X32-NEXT: popl %esi
+; X32-NEXT: .cfi_def_cfa_offset 16
+; X32-NEXT: popl %edi
+; X32-NEXT: .cfi_def_cfa_offset 12
+; X32-NEXT: popl %ebx
+; X32-NEXT: .cfi_def_cfa_offset 8
+; X32-NEXT: popl %ebp
+; X32-NEXT: .cfi_def_cfa_offset 4
+; X32-NEXT: retl $4
+;
+; X32-NOCMOV-LABEL: test_ctselect_v8f32:
+; X32-NOCMOV: # %bb.0:
+; X32-NOCMOV-NEXT: pushl %ebp
+; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 8
+; X32-NOCMOV-NEXT: pushl %ebx
+; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 12
+; X32-NOCMOV-NEXT: pushl %edi
+; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 16
+; X32-NOCMOV-NEXT: pushl %esi
+; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 20
+; X32-NOCMOV-NEXT: subl $8, %esp
+; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 28
+; X32-NOCMOV-NEXT: .cfi_offset %esi, -20
+; X32-NOCMOV-NEXT: .cfi_offset %edi, -16
+; X32-NOCMOV-NEXT: .cfi_offset %ebx, -12
+; X32-NOCMOV-NEXT: .cfi_offset %ebp, -8
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edi
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebp
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebx
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %esi
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: xorl %eax, %ecx
+; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %edx
+; X32-NOCMOV-NEXT: andl $1, %edx
+; X32-NOCMOV-NEXT: negl %edx
+; X32-NOCMOV-NEXT: andl %edx, %ecx
+; X32-NOCMOV-NEXT: xorl %eax, %ecx
+; X32-NOCMOV-NEXT: movl %ecx, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: xorl %esi, %eax
+; X32-NOCMOV-NEXT: andl %edx, %eax
+; X32-NOCMOV-NEXT: xorl %esi, %eax
+; X32-NOCMOV-NEXT: movl %eax, (%esp) # 4-byte Spill
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %esi
+; X32-NOCMOV-NEXT: xorl %ebx, %esi
+; X32-NOCMOV-NEXT: andl %edx, %esi
+; X32-NOCMOV-NEXT: xorl %ebx, %esi
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebx
+; X32-NOCMOV-NEXT: xorl %ebp, %ebx
+; X32-NOCMOV-NEXT: andl %edx, %ebx
+; X32-NOCMOV-NEXT: xorl %ebp, %ebx
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebp
+; X32-NOCMOV-NEXT: xorl %edi, %ebp
+; X32-NOCMOV-NEXT: andl %edx, %ebp
+; X32-NOCMOV-NEXT: xorl %edi, %ebp
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edi
+; X32-NOCMOV-NEXT: xorl %eax, %edi
+; X32-NOCMOV-NEXT: andl %edx, %edi
+; X32-NOCMOV-NEXT: xorl %eax, %edi
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: xorl %eax, %ecx
+; X32-NOCMOV-NEXT: andl %edx, %ecx
+; X32-NOCMOV-NEXT: xorl %eax, %ecx
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: andl %edx, %eax
+; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NOCMOV-NEXT: movl %eax, 28(%edx)
+; X32-NOCMOV-NEXT: movl %ecx, 24(%edx)
+; X32-NOCMOV-NEXT: movl %edi, 20(%edx)
+; X32-NOCMOV-NEXT: movl %ebp, 16(%edx)
+; X32-NOCMOV-NEXT: movl %ebx, 12(%edx)
+; X32-NOCMOV-NEXT: movl %esi, 8(%edx)
+; X32-NOCMOV-NEXT: movl (%esp), %eax # 4-byte Reload
+; X32-NOCMOV-NEXT: movl %eax, 4(%edx)
+; X32-NOCMOV-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 4-byte Reload
+; X32-NOCMOV-NEXT: movl %eax, (%edx)
+; X32-NOCMOV-NEXT: movl %edx, %eax
+; X32-NOCMOV-NEXT: addl $8, %esp
+; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 20
+; X32-NOCMOV-NEXT: popl %esi
+; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 16
+; X32-NOCMOV-NEXT: popl %edi
+; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 12
+; X32-NOCMOV-NEXT: popl %ebx
+; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 8
+; X32-NOCMOV-NEXT: popl %ebp
+; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 4
+; X32-NOCMOV-NEXT: retl $4
+ %result = call <8 x float> @llvm.ct.select.v8f32(i1 %cond, <8 x float> %a, <8 x float> %b)
+ ret <8 x float> %result
+}
+
+define float @test_ctselect_f32_nan_inf(i1 %cond) {
+; X64-LABEL: test_ctselect_f32_nan_inf:
+; X64: # %bb.0:
+; X64-NEXT: andl $1, %edi
+; X64-NEXT: negl %edi
+; X64-NEXT: andl $4194304, %edi # imm = 0x400000
+; X64-NEXT: xorl $2139095040, %edi # imm = 0x7F800000
+; X64-NEXT: movd %edi, %xmm0
+; X64-NEXT: retq
+;
+; X32-LABEL: test_ctselect_f32_nan_inf:
+; X32: # %bb.0:
+; X32-NEXT: pushl %eax
+; X32-NEXT: .cfi_def_cfa_offset 8
+; X32-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: andl $1, %eax
+; X32-NEXT: negl %eax
+; X32-NEXT: andl $4194304, %eax # imm = 0x400000
+; X32-NEXT: xorl $2139095040, %eax # imm = 0x7F800000
+; X32-NEXT: movl %eax, (%esp)
+; X32-NEXT: flds (%esp)
+; X32-NEXT: popl %eax
+; X32-NEXT: .cfi_def_cfa_offset 4
+; X32-NEXT: retl
+;
+; X32-NOCMOV-LABEL: test_ctselect_f32_nan_inf:
+; X32-NOCMOV: # %bb.0:
+; X32-NOCMOV-NEXT: pushl %eax
+; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 8
+; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: andl $1, %eax
+; X32-NOCMOV-NEXT: negl %eax
+; X32-NOCMOV-NEXT: andl $4194304, %eax # imm = 0x400000
+; X32-NOCMOV-NEXT: xorl $2139095040, %eax # imm = 0x7F800000
+; X32-NOCMOV-NEXT: movl %eax, (%esp)
+; X32-NOCMOV-NEXT: flds (%esp)
+; X32-NOCMOV-NEXT: popl %eax
+; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 4
+; X32-NOCMOV-NEXT: retl
+ %result = call float @llvm.ct.select.f32(i1 %cond, float 0x7FF8000000000000, float 0x7FF0000000000000)
+ ret float %result
+}
+
+define double @test_ctselect_f64_nan_inf(i1 %cond) {
+; X64-LABEL: test_ctselect_f64_nan_inf:
+; X64: # %bb.0:
+; X64-NEXT: # kill: def $edi killed $edi def $rdi
+; X64-NEXT: andl $1, %edi
+; X64-NEXT: negq %rdi
+; X64-NEXT: movabsq $2251799813685248, %rax # imm = 0x8000000000000
+; X64-NEXT: andq %rdi, %rax
+; X64-NEXT: movabsq $9218868437227405312, %rcx # imm = 0x7FF0000000000000
+; X64-NEXT: xorq %rax, %rcx
+; X64-NEXT: movq %rcx, %xmm0
+; X64-NEXT: retq
+;
+; X32-LABEL: test_ctselect_f64_nan_inf:
+; X32: # %bb.0:
+; X32-NEXT: subl $12, %esp
+; X32-NEXT: .cfi_def_cfa_offset 16
+; X32-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: andl $1, %eax
+; X32-NEXT: negl %eax
+; X32-NEXT: andl $524288, %eax # imm = 0x80000
+; X32-NEXT: orl $2146435072, %eax # imm = 0x7FF00000
+; X32-NEXT: movl %eax, {{[0-9]+}}(%esp)
+; X32-NEXT: movl $0, (%esp)
+; X32-NEXT: fldl (%esp)
+; X32-NEXT: addl $12, %esp
+; X32-NEXT: .cfi_def_cfa_offset 4
+; X32-NEXT: retl
+;
+; X32-NOCMOV-LABEL: test_ctselect_f64_nan_inf:
+; X32-NOCMOV: # %bb.0:
+; X32-NOCMOV-NEXT: subl $12, %esp
+; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 16
+; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: andl $1, %eax
+; X32-NOCMOV-NEXT: negl %eax
+; X32-NOCMOV-NEXT: andl $524288, %eax # imm = 0x80000
+; X32-NOCMOV-NEXT: orl $2146435072, %eax # imm = 0x7FF00000
+; X32-NOCMOV-NEXT: movl %eax, {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: movl $0, (%esp)
+; X32-NOCMOV-NEXT: fldl (%esp)
+; X32-NOCMOV-NEXT: addl $12, %esp
+; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 4
+; X32-NOCMOV-NEXT: retl
+ %result = call double @llvm.ct.select.f64(i1 %cond, double 0x7FF8000000000000, double 0x7FF0000000000000)
+ ret double %result
+}
+
+; Declare the intrinsics
+declare i1 @llvm.ct.select.i1(i1, i1, i1)
+declare i8 @llvm.ct.select.i8(i1, i8, i8)
+declare i16 @llvm.ct.select.i16(i1, i16, i16)
+declare i32 @llvm.ct.select.i32(i1, i32, i32)
+declare i64 @llvm.ct.select.i64(i1, i64, i64)
+declare float @llvm.ct.select.f32(i1, float, float)
+declare double @llvm.ct.select.f64(i1, double, double)
+declare ptr @llvm.ct.select.p0(i1, ptr, ptr)
+
+; Vector intrinsics
+declare <4 x i32> @llvm.ct.select.v4i32(i1, <4 x i32>, <4 x i32>)
+declare <2 x i64> @llvm.ct.select.v2i64(i1, <2 x i64>, <2 x i64>)
+declare <8 x i16> @llvm.ct.select.v8i16(i1, <8 x i16>, <8 x i16>)
+declare <16 x i8> @llvm.ct.select.v16i8(i1, <16 x i8>, <16 x i8>)
+declare <4 x float> @llvm.ct.select.v4f32(i1, <4 x float>, <4 x float>)
+declare <2 x double> @llvm.ct.select.v2f64(i1, <2 x double>, <2 x double>)
+declare <8 x i32> @llvm.ct.select.v8i32(i1, <8 x i32>, <8 x i32>)
+declare <8 x float> @llvm.ct.select.v8f32(i1, <8 x float>, <8 x float>)
>From 2703c5009d0419196c62149b785d71d14c1a40e2 Mon Sep 17 00:00:00 2001
From: wizardengineer <juliuswoosebert at gmail.com>
Date: Thu, 19 Feb 2026 23:13:21 -0500
Subject: [PATCH 02/12] [ConstantTime] Changed CTSELECT instances to CT_SELECT
---
llvm/include/llvm/CodeGen/ISDOpcodes.h | 4 +--
llvm/include/llvm/CodeGen/SelectionDAG.h | 2 +-
llvm/include/llvm/CodeGen/TargetLowering.h | 7 ++---
.../include/llvm/Target/TargetSelectionDAG.td | 2 +-
llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp | 28 +++++++++----------
llvm/lib/CodeGen/SelectionDAG/LegalizeDAG.cpp | 4 +--
.../SelectionDAG/LegalizeFloatTypes.cpp | 12 ++++----
.../SelectionDAG/LegalizeIntegerTypes.cpp | 14 +++++-----
llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h | 10 +++----
.../SelectionDAG/LegalizeTypesGeneric.cpp | 4 +--
.../SelectionDAG/LegalizeVectorTypes.cpp | 14 +++++-----
.../SelectionDAG/SelectionDAGBuilder.cpp | 5 ++--
.../SelectionDAG/SelectionDAGDumper.cpp | 2 +-
llvm/lib/CodeGen/TargetLoweringBase.cpp | 3 +-
llvm/lib/Target/AArch64/AArch64ISelLowering.h | 2 --
llvm/lib/Target/ARM/ARMISelLowering.h | 2 --
llvm/lib/Target/X86/X86ISelLowering.h | 2 --
llvm/test/CodeGen/RISCV/ctselect-fallback.ll | 18 ++++++------
llvm/test/CodeGen/X86/ctselect.ll | 22 +++++++--------
19 files changed, 76 insertions(+), 81 deletions(-)
diff --git a/llvm/include/llvm/CodeGen/ISDOpcodes.h b/llvm/include/llvm/CodeGen/ISDOpcodes.h
index 89fe9e07e3583..1e3b52494c0a9 100644
--- a/llvm/include/llvm/CodeGen/ISDOpcodes.h
+++ b/llvm/include/llvm/CodeGen/ISDOpcodes.h
@@ -805,9 +805,9 @@ enum NodeType {
/// i1 then the high bits must conform to getBooleanContents.
SELECT,
- /// CTSELECT(Cond, TrueVal, FalseVal). Cond is i1 and the value operands must
+ /// CT_SELECT(Cond, TrueVal, FalseVal). Cond is i1 and the value operands must
/// have the same type. Used to lower the constant-time select intrinsic.
- CTSELECT,
+ CT_SELECT,
/// Select with a vector condition (op #0) and two vector operands (ops #1
/// and #2), returning a vector result. All vectors have the same length.
diff --git a/llvm/include/llvm/CodeGen/SelectionDAG.h b/llvm/include/llvm/CodeGen/SelectionDAG.h
index 47b27aaf75ed9..60f8af27101aa 100644
--- a/llvm/include/llvm/CodeGen/SelectionDAG.h
+++ b/llvm/include/llvm/CodeGen/SelectionDAG.h
@@ -1364,7 +1364,7 @@ class SelectionDAG {
SDValue RHS, SDNodeFlags Flags = SDNodeFlags()) {
assert(LHS.getValueType() == VT && RHS.getValueType() == VT &&
"Cannot use select on differing types");
- return getNode(ISD::CTSELECT, DL, VT, Cond, LHS, RHS, Flags);
+ return getNode(ISD::CT_SELECT, DL, VT, Cond, LHS, RHS, Flags);
}
/// Helper function to make it easier to build SelectCC's if you just have an
diff --git a/llvm/include/llvm/CodeGen/TargetLowering.h b/llvm/include/llvm/CodeGen/TargetLowering.h
index 5dcb5ffbf08b1..4006ad3bda5d6 100644
--- a/llvm/include/llvm/CodeGen/TargetLowering.h
+++ b/llvm/include/llvm/CodeGen/TargetLowering.h
@@ -506,10 +506,9 @@ class LLVM_ABI TargetLoweringBase {
MachineMemOperand::Flags
getVPIntrinsicMemOperandFlags(const VPIntrinsic &VPIntrin) const;
- virtual bool isSelectSupported(SelectSupportKind kind) const { return true; }
-
- /// Return true if the target has custom lowering for constant-time select.
- virtual bool isCtSelectSupported(EVT VT) const { return false; }
+ virtual bool isSelectSupported(SelectSupportKind /*kind*/) const {
+ return true;
+ }
/// Return true if the @llvm.get.active.lane.mask intrinsic should be expanded
/// using generic code in SelectionDAGBuilder.
diff --git a/llvm/include/llvm/Target/TargetSelectionDAG.td b/llvm/include/llvm/Target/TargetSelectionDAG.td
index 94c8fddd9aed5..4c5a4c884e745 100644
--- a/llvm/include/llvm/Target/TargetSelectionDAG.td
+++ b/llvm/include/llvm/Target/TargetSelectionDAG.td
@@ -790,7 +790,7 @@ def reset_fpmode : SDNode<"ISD::RESET_FPMODE", SDTNone, [SDNPHasChain]>;
def setcc : SDNode<"ISD::SETCC" , SDTSetCC>;
def select : SDNode<"ISD::SELECT" , SDTSelect>;
-def ctselect : SDNode<"ISD::CTSELECT", SDTCtSelect>;
+def ct_select : SDNode<"ISD::CT_SELECT", SDTCtSelect>;
def vselect : SDNode<"ISD::VSELECT" , SDTVSelect>;
def selectcc : SDNode<"ISD::SELECT_CC" , SDTSelectCC>;
diff --git a/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp b/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp
index 075555e009097..0e46efe269ed3 100644
--- a/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp
@@ -482,7 +482,7 @@ namespace {
SDValue visitCTTZ_ZERO_POISON(SDNode *N);
SDValue visitCTPOP(SDNode *N);
SDValue visitSELECT(SDNode *N);
- // ISD::CTSELECT - Constant-Time SELECT (not related to CT in
+ // ISD::CT_SELECT - Constant-Time SELECT (not related to CT in
// CTPOP/CTLZ/CTTZ where CT means "count").
SDValue visitCT_SELECT(SDNode *N);
SDValue visitVSELECT(SDNode *N);
@@ -2040,7 +2040,7 @@ SDValue DAGCombiner::visit(SDNode *N) {
case ISD::CTTZ_ZERO_POISON: return visitCTTZ_ZERO_POISON(N);
case ISD::CTPOP: return visitCTPOP(N);
case ISD::SELECT: return visitSELECT(N);
- case ISD::CTSELECT: return visitCT_SELECT(N);
+ case ISD::CT_SELECT: return visitCT_SELECT(N);
case ISD::VSELECT: return visitVSELECT(N);
case ISD::SELECT_CC: return visitSELECT_CC(N);
case ISD::SETCC: return visitSETCC(N);
@@ -13686,10 +13686,10 @@ SDValue DAGCombiner::visitSELECT(SDNode *N) {
return SDValue();
}
-// Keep CTSELECT combines deliberately conservative to preserve constant-time
+// Keep CT_SELECT combines deliberately conservative to preserve constant-time
// intent across generic DAG combines. We only accept:
// - canonicalization of negated conditions (flip true/false operands), and
-// - i1 CTSELECT nesting merges via AND/OR that keep the result as CTSELECT.
+// - i1 CT_SELECT nesting merges via AND/OR that keep the result as CT_SELECT.
// Broader rewrites should be done in target-specific lowering when stronger
// guarantees about legality and constant-time preservation are available.
SDValue DAGCombiner::visitCT_SELECT(SDNode *N) {
@@ -13701,52 +13701,52 @@ SDValue DAGCombiner::visitCT_SELECT(SDNode *N) {
SDLoc DL(N);
SDNodeFlags Flags = N->getFlags();
- // ctselect (not Cond), N1, N2 -> ctselect Cond, N2, N1
+ // ct_select (not Cond), N1, N2 -> ct_select Cond, N2, N1
// This is a CT-safe canonicalization: flip negated condition by swapping
// arms. extractBooleanFlip only matches boolean xor-with-1, so this preserves
// dataflow semantics and does not introduce data-dependent control flow.
if (SDValue F = extractBooleanFlip(N0, DAG, TLI, false)) {
- SDValue SelectOp = DAG.getNode(ISD::CTSELECT, DL, VT, F, N2, N1);
+ SDValue SelectOp = DAG.getNode(ISD::CT_SELECT, DL, VT, F, N2, N1);
SelectOp->setFlags(Flags);
return SelectOp;
}
if (VT0 == MVT::i1) {
- // Nested CTSELECT merging optimizations for i1 conditions.
+ // Nested CT_SELECT merging optimizations for i1 conditions.
// These are CT-safe because:
// 1. AND/OR are bitwise operations that execute in constant time
- // 2. The optimization combines two sequential CTSELECTs into one,
+ // 2. The optimization combines two sequential CT_SELECTs into one,
// reducing the total number of constant-time operations without
// changing semantics
// 3. No data-dependent branches or memory accesses are introduced
//
- // ctselect C0, (ctselect C1, X, Y), Y -> ctselect (C0 & C1), X, Y
+ // ct_select C0, (ct_select C1, X, Y), Y -> ct_select (C0 & C1), X, Y
// Semantic equivalence: If C0 is true, evaluate inner select (C1 ? X :
// Y). If C0 is false, choose Y. This is equivalent to (C0 && C1) ? X : Y.
- if (N1->getOpcode() == ISD::CTSELECT && N1->hasOneUse()) {
+ if (N1->getOpcode() == ISD::CT_SELECT && N1->hasOneUse()) {
SDValue N1_0 = N1->getOperand(0);
SDValue N1_1 = N1->getOperand(1);
SDValue N1_2 = N1->getOperand(2);
if (N1_2 == N2 && N0.getValueType() == N1_0.getValueType()) {
SDValue And = DAG.getNode(ISD::AND, DL, N0.getValueType(), N0, N1_0);
SDValue SelectOp =
- DAG.getNode(ISD::CTSELECT, DL, N1.getValueType(), And, N1_1, N2);
+ DAG.getNode(ISD::CT_SELECT, DL, N1.getValueType(), And, N1_1, N2);
SelectOp->setFlags(Flags);
return SelectOp;
}
}
- // ctselect C0, X, (ctselect C1, X, Y) -> ctselect (C0 | C1), X, Y
+ // ct_select C0, X, (ct_select C1, X, Y) -> ct_select (C0 | C1), X, Y
// Semantic equivalence: If C0 is true, choose X. If C0 is false, evaluate
// inner select (C1 ? X : Y). This is equivalent to (C0 || C1) ? X : Y.
- if (N2->getOpcode() == ISD::CTSELECT && N2->hasOneUse()) {
+ if (N2->getOpcode() == ISD::CT_SELECT && N2->hasOneUse()) {
SDValue N2_0 = N2->getOperand(0);
SDValue N2_1 = N2->getOperand(1);
SDValue N2_2 = N2->getOperand(2);
if (N2_1 == N1 && N0.getValueType() == N2_0.getValueType()) {
SDValue Or = DAG.getNode(ISD::OR, DL, N0.getValueType(), N0, N2_0);
SDValue SelectOp =
- DAG.getNode(ISD::CTSELECT, DL, N1.getValueType(), Or, N1, N2_2);
+ DAG.getNode(ISD::CT_SELECT, DL, N1.getValueType(), Or, N1, N2_2);
SelectOp->setFlags(Flags);
return SelectOp;
}
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeDAG.cpp b/llvm/lib/CodeGen/SelectionDAG/LegalizeDAG.cpp
index 4025f82e06478..43a94d17b8aee 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeDAG.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeDAG.cpp
@@ -4336,7 +4336,7 @@ bool SelectionDAGLegalize::ExpandNode(SDNode *Node) {
}
Results.push_back(Tmp1);
break;
- case ISD::CTSELECT: {
+ case ISD::CT_SELECT: {
Tmp1 = Node->getOperand(0);
Tmp2 = Node->getOperand(1);
Tmp3 = Node->getOperand(2);
@@ -5734,7 +5734,7 @@ void SelectionDAGLegalize::PromoteNode(SDNode *Node) {
break;
}
case ISD::SELECT:
- case ISD::CTSELECT: {
+ case ISD::CT_SELECT: {
unsigned ExtOp, TruncOp;
if (Node->getValueType(0).isVector() ||
Node->getValueType(0).getSizeInBits() == NVT.getSizeInBits()) {
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeFloatTypes.cpp b/llvm/lib/CodeGen/SelectionDAG/LegalizeFloatTypes.cpp
index 4791130d75621..17a837bb42770 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeFloatTypes.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeFloatTypes.cpp
@@ -161,7 +161,7 @@ void DAGTypeLegalizer::SoftenFloatResult(SDNode *N, unsigned ResNo) {
case ISD::ATOMIC_LOAD: R = SoftenFloatRes_ATOMIC_LOAD(N); break;
case ISD::ATOMIC_SWAP: R = BitcastToInt_ATOMIC_SWAP(N); break;
case ISD::SELECT: R = SoftenFloatRes_SELECT(N); break;
- case ISD::CTSELECT: R = SoftenFloatRes_CTSELECT(N); break;
+ case ISD::CT_SELECT: R = SoftenFloatRes_CT_SELECT(N); break;
case ISD::SELECT_CC: R = SoftenFloatRes_SELECT_CC(N); break;
case ISD::FREEZE: R = SoftenFloatRes_FREEZE(N); break;
case ISD::STRICT_SINT_TO_FP:
@@ -967,7 +967,7 @@ SDValue DAGTypeLegalizer::SoftenFloatRes_SELECT(SDNode *N) {
LHS.getValueType(), N->getOperand(0), LHS, RHS);
}
-SDValue DAGTypeLegalizer::SoftenFloatRes_CTSELECT(SDNode *N) {
+SDValue DAGTypeLegalizer::SoftenFloatRes_CT_SELECT(SDNode *N) {
SDValue LHS = GetSoftenedFloat(N->getOperand(1));
SDValue RHS = GetSoftenedFloat(N->getOperand(2));
return DAG.getCTSelect(SDLoc(N), LHS.getValueType(), N->getOperand(0), LHS,
@@ -1537,7 +1537,7 @@ void DAGTypeLegalizer::ExpandFloatResult(SDNode *N, unsigned ResNo) {
case ISD::POISON:
case ISD::UNDEF: SplitRes_UNDEF(N, Lo, Hi); break;
case ISD::SELECT: SplitRes_Select(N, Lo, Hi); break;
- case ISD::CTSELECT: SplitRes_CTSELECT(N, Lo, Hi); break;
+ case ISD::CT_SELECT: SplitRes_CT_SELECT(N, Lo, Hi); break;
case ISD::SELECT_CC: SplitRes_SELECT_CC(N, Lo, Hi); break;
case ISD::MERGE_VALUES: ExpandRes_MERGE_VALUES(N, ResNo, Lo, Hi); break;
@@ -2590,8 +2590,8 @@ void DAGTypeLegalizer::SoftPromoteHalfResult(SDNode *N, unsigned ResNo) {
R = SoftPromoteHalfRes_ATOMIC_LOAD(N);
break;
case ISD::SELECT: R = SoftPromoteHalfRes_SELECT(N); break;
- case ISD::CTSELECT:
- R = SoftPromoteHalfRes_CTSELECT(N);
+ case ISD::CT_SELECT:
+ R = SoftPromoteHalfRes_CT_SELECT(N);
break;
case ISD::SELECT_CC: R = SoftPromoteHalfRes_SELECT_CC(N); break;
case ISD::STRICT_SINT_TO_FP:
@@ -2862,7 +2862,7 @@ SDValue DAGTypeLegalizer::SoftPromoteHalfRes_SELECT(SDNode *N) {
N->getFlags());
}
-SDValue DAGTypeLegalizer::SoftPromoteHalfRes_CTSELECT(SDNode *N) {
+SDValue DAGTypeLegalizer::SoftPromoteHalfRes_CT_SELECT(SDNode *N) {
SDValue Op1 = GetSoftPromotedHalf(N->getOperand(1));
SDValue Op2 = GetSoftPromotedHalf(N->getOperand(2));
return DAG.getCTSelect(SDLoc(N), Op1.getValueType(), N->getOperand(0), Op1,
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeIntegerTypes.cpp b/llvm/lib/CodeGen/SelectionDAG/LegalizeIntegerTypes.cpp
index d3b419f32d8a9..0cc5e14834442 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeIntegerTypes.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeIntegerTypes.cpp
@@ -91,7 +91,7 @@ void DAGTypeLegalizer::PromoteIntegerResult(SDNode *N, unsigned ResNo) {
Res = PromoteIntRes_VECTOR_COMPRESS(N);
break;
case ISD::SELECT:
- case ISD::CTSELECT:
+ case ISD::CT_SELECT:
case ISD::VSELECT:
case ISD::VP_MERGE:
Res = PromoteIntRes_Select(N);
@@ -1996,8 +1996,8 @@ bool DAGTypeLegalizer::PromoteIntegerOperand(SDNode *N, unsigned OpNo) {
break;
case ISD::VSELECT:
case ISD::SELECT: Res = PromoteIntOp_SELECT(N, OpNo); break;
- case ISD::CTSELECT:
- Res = PromoteIntOp_CTSELECT(N, OpNo);
+ case ISD::CT_SELECT:
+ Res = PromoteIntOp_CT_SELECT(N, OpNo);
break;
case ISD::SELECT_CC: Res = PromoteIntOp_SELECT_CC(N, OpNo); break;
case ISD::SETCC: Res = PromoteIntOp_SETCC(N, OpNo); break;
@@ -2409,13 +2409,13 @@ SDValue DAGTypeLegalizer::PromoteIntOp_SELECT(SDNode *N, unsigned OpNo) {
N->getOperand(2)), 0);
}
-SDValue DAGTypeLegalizer::PromoteIntOp_CTSELECT(SDNode *N, unsigned OpNo) {
+SDValue DAGTypeLegalizer::PromoteIntOp_CT_SELECT(SDNode *N, unsigned OpNo) {
assert(OpNo == 0 && "Only know how to promote the condition!");
SDValue Cond = N->getOperand(0);
EVT OpTy = N->getOperand(1).getValueType();
// Promote all the way up to the canonical SetCC type.
- EVT OpVT = N->getOpcode() == ISD::CTSELECT ? OpTy.getScalarType() : OpTy;
+ EVT OpVT = N->getOpcode() == ISD::CT_SELECT ? OpTy.getScalarType() : OpTy;
Cond = PromoteTargetBoolean(Cond, OpVT);
return SDValue(
@@ -3030,8 +3030,8 @@ void DAGTypeLegalizer::ExpandIntegerResult(SDNode *N, unsigned ResNo) {
case ISD::ARITH_FENCE: SplitRes_ARITH_FENCE(N, Lo, Hi); break;
case ISD::MERGE_VALUES: SplitRes_MERGE_VALUES(N, ResNo, Lo, Hi); break;
case ISD::SELECT: SplitRes_Select(N, Lo, Hi); break;
- case ISD::CTSELECT:
- SplitRes_CTSELECT(N, Lo, Hi);
+ case ISD::CT_SELECT:
+ SplitRes_CT_SELECT(N, Lo, Hi);
break;
case ISD::SELECT_CC: SplitRes_SELECT_CC(N, Lo, Hi); break;
case ISD::POISON:
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h b/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h
index 8036ac9b040ee..10cb2f619ce7e 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h
@@ -382,7 +382,7 @@ class LLVM_LIBRARY_VISIBILITY DAGTypeLegalizer {
SDValue PromoteIntOp_CONCAT_VECTORS(SDNode *N);
SDValue PromoteIntOp_ScalarOp(SDNode *N);
SDValue PromoteIntOp_SELECT(SDNode *N, unsigned OpNo);
- SDValue PromoteIntOp_CTSELECT(SDNode *N, unsigned OpNo);
+ SDValue PromoteIntOp_CT_SELECT(SDNode *N, unsigned OpNo);
SDValue PromoteIntOp_SELECT_CC(SDNode *N, unsigned OpNo);
SDValue PromoteIntOp_SETCC(SDNode *N, unsigned OpNo);
SDValue PromoteIntOp_Shift(SDNode *N);
@@ -623,7 +623,7 @@ class LLVM_LIBRARY_VISIBILITY DAGTypeLegalizer {
SDValue SoftenFloatRes_LOAD(SDNode *N);
SDValue SoftenFloatRes_ATOMIC_LOAD(SDNode *N);
SDValue SoftenFloatRes_SELECT(SDNode *N);
- SDValue SoftenFloatRes_CTSELECT(SDNode *N);
+ SDValue SoftenFloatRes_CT_SELECT(SDNode *N);
SDValue SoftenFloatRes_SELECT_CC(SDNode *N);
SDValue SoftenFloatRes_UNDEF(SDNode *N);
SDValue SoftenFloatRes_VAARG(SDNode *N);
@@ -775,7 +775,7 @@ class LLVM_LIBRARY_VISIBILITY DAGTypeLegalizer {
SDValue SoftPromoteHalfRes_LOAD(SDNode *N);
SDValue SoftPromoteHalfRes_ATOMIC_LOAD(SDNode *N);
SDValue SoftPromoteHalfRes_SELECT(SDNode *N);
- SDValue SoftPromoteHalfRes_CTSELECT(SDNode *N);
+ SDValue SoftPromoteHalfRes_CT_SELECT(SDNode *N);
SDValue SoftPromoteHalfRes_SELECT_CC(SDNode *N);
SDValue SoftPromoteHalfRes_UnaryOp(SDNode *N);
SDValue SoftPromoteHalfRes_FABS(SDNode *N);
@@ -848,7 +848,7 @@ class LLVM_LIBRARY_VISIBILITY DAGTypeLegalizer {
SDValue ScalarizeVecRes_VECTOR_INTERLEAVE_DEINTERLEAVE(SDNode *N);
SDValue ScalarizeVecRes_VSELECT(SDNode *N);
SDValue ScalarizeVecRes_SELECT(SDNode *N);
- SDValue ScalarizeVecRes_CTSELECT(SDNode *N);
+ SDValue ScalarizeVecRes_CT_SELECT(SDNode *N);
SDValue ScalarizeVecRes_SELECT_CC(SDNode *N);
SDValue ScalarizeVecRes_SETCC(SDNode *N);
SDValue ScalarizeVecRes_UNDEF(SDNode *N);
@@ -1200,7 +1200,7 @@ class LLVM_LIBRARY_VISIBILITY DAGTypeLegalizer {
void SplitVecRes_AssertSext(SDNode *N, SDValue &Lo, SDValue &Hi);
void SplitRes_ARITH_FENCE (SDNode *N, SDValue &Lo, SDValue &Hi);
void SplitRes_Select(SDNode *N, SDValue &Lo, SDValue &Hi);
- void SplitRes_CTSELECT(SDNode *N, SDValue &Lo, SDValue &Hi);
+ void SplitRes_CT_SELECT(SDNode *N, SDValue &Lo, SDValue &Hi);
void SplitRes_SELECT_CC (SDNode *N, SDValue &Lo, SDValue &Hi);
void SplitRes_UNDEF (SDNode *N, SDValue &Lo, SDValue &Hi);
void SplitRes_FREEZE (SDNode *N, SDValue &Lo, SDValue &Hi);
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeTypesGeneric.cpp b/llvm/lib/CodeGen/SelectionDAG/LegalizeTypesGeneric.cpp
index 060f7a4ae6873..5b452eeac4866 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeTypesGeneric.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeTypesGeneric.cpp
@@ -569,9 +569,9 @@ void DAGTypeLegalizer::SplitRes_Select(SDNode *N, SDValue &Lo, SDValue &Hi) {
Hi = DAG.getNode(Opcode, dl, LH.getValueType(), CH, LH, RH, EVLHi);
}
-void DAGTypeLegalizer::SplitRes_CTSELECT(SDNode *N, SDValue &Lo, SDValue &Hi) {
+void DAGTypeLegalizer::SplitRes_CT_SELECT(SDNode *N, SDValue &Lo, SDValue &Hi) {
// Reuse generic select splitting to support scalar and vector conditions.
- // SplitRes_Select rebuilds with N->getOpcode(), so CTSELECT is preserved.
+ // SplitRes_Select rebuilds with N->getOpcode(), so CT_SELECT is preserved.
SplitRes_Select(N, Lo, Hi);
}
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp b/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp
index 8ddb9a1e78208..6d25c26cbe914 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp
@@ -91,8 +91,8 @@ void DAGTypeLegalizer::ScalarizeVectorResult(SDNode *N, unsigned ResNo) {
case ISD::SIGN_EXTEND_INREG: R = ScalarizeVecRes_InregOp(N); break;
case ISD::VSELECT: R = ScalarizeVecRes_VSELECT(N); break;
case ISD::SELECT: R = ScalarizeVecRes_SELECT(N); break;
- case ISD::CTSELECT:
- R = ScalarizeVecRes_CTSELECT(N);
+ case ISD::CT_SELECT:
+ R = ScalarizeVecRes_CT_SELECT(N);
break;
case ISD::SELECT_CC: R = ScalarizeVecRes_SELECT_CC(N); break;
case ISD::SETCC: R = ScalarizeVecRes_SETCC(N); break;
@@ -765,7 +765,7 @@ SDValue DAGTypeLegalizer::ScalarizeVecRes_SELECT(SDNode *N) {
GetScalarizedVector(N->getOperand(2)));
}
-SDValue DAGTypeLegalizer::ScalarizeVecRes_CTSELECT(SDNode *N) {
+SDValue DAGTypeLegalizer::ScalarizeVecRes_CT_SELECT(SDNode *N) {
SDValue LHS = GetScalarizedVector(N->getOperand(1));
return DAG.getCTSelect(SDLoc(N), LHS.getValueType(), N->getOperand(0), LHS,
GetScalarizedVector(N->getOperand(2)));
@@ -1408,9 +1408,9 @@ void DAGTypeLegalizer::SplitVectorResult(SDNode *N, unsigned ResNo) {
case ISD::AssertSext: SplitVecRes_AssertSext(N, Lo, Hi); break;
case ISD::VSELECT:
case ISD::SELECT:
- case ISD::VP_MERGE: SplitRes_Select(N, Lo, Hi); break;
- case ISD::CTSELECT:
- SplitRes_CTSELECT(N, Lo, Hi);
+ case ISD::VP_MERGE:
+ case ISD::CT_SELECT:
+ SplitRes_CT_SELECT(N, Lo, Hi);
break;
case ISD::SELECT_CC: SplitRes_SELECT_CC(N, Lo, Hi); break;
case ISD::POISON:
@@ -5296,7 +5296,7 @@ void DAGTypeLegalizer::WidenVectorResult(SDNode *N, unsigned ResNo) {
case ISD::SIGN_EXTEND_INREG: Res = WidenVecRes_InregOp(N); break;
case ISD::VSELECT:
case ISD::SELECT:
- case ISD::CTSELECT:
+ case ISD::CT_SELECT:
case ISD::VP_MERGE:
Res = WidenVecRes_Select(N);
break;
diff --git a/llvm/lib/CodeGen/SelectionDAG/SelectionDAGBuilder.cpp b/llvm/lib/CodeGen/SelectionDAG/SelectionDAGBuilder.cpp
index aac1442347fbd..de971b8c8a562 100644
--- a/llvm/lib/CodeGen/SelectionDAG/SelectionDAGBuilder.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/SelectionDAGBuilder.cpp
@@ -7002,8 +7002,9 @@ void SelectionDAGBuilder::visitIntrinsicCall(const CallInst &I,
assert(!CondVT.isVector() && "Vector type cond not supported yet");
// Handle scalar types
- if (TLI.isCtSelectSupported(VT) && !CondVT.isVector()) {
- SDValue Result = DAG.getNode(ISD::CTSELECT, DL, VT, Cond, A, B);
+ if (TLI.isOperationLegalOrCustom(ISD::CT_SELECT, VT) &&
+ !CondVT.isVector()) {
+ SDValue Result = DAG.getNode(ISD::CT_SELECT, DL, VT, Cond, A, B);
setValue(&I, Result);
return;
}
diff --git a/llvm/lib/CodeGen/SelectionDAG/SelectionDAGDumper.cpp b/llvm/lib/CodeGen/SelectionDAG/SelectionDAGDumper.cpp
index f99394da31eeb..0263bd9622b41 100644
--- a/llvm/lib/CodeGen/SelectionDAG/SelectionDAGDumper.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/SelectionDAGDumper.cpp
@@ -346,7 +346,7 @@ std::string SDNode::getOperationName(const SelectionDAG *G) const {
case ISD::FPOWI: return "fpowi";
case ISD::STRICT_FPOWI: return "strict_fpowi";
case ISD::SETCC: return "setcc";
- case ISD::CTSELECT: return "ctselect";
+ case ISD::CT_SELECT: return "ct_select";
case ISD::SETCCCARRY: return "setcccarry";
case ISD::STRICT_FSETCC: return "strict_fsetcc";
case ISD::STRICT_FSETCCS: return "strict_fsetccs";
diff --git a/llvm/lib/CodeGen/TargetLoweringBase.cpp b/llvm/lib/CodeGen/TargetLoweringBase.cpp
index bd9b9b47c8b82..2af0cf72700d2 100644
--- a/llvm/lib/CodeGen/TargetLoweringBase.cpp
+++ b/llvm/lib/CodeGen/TargetLoweringBase.cpp
@@ -862,7 +862,8 @@ void TargetLoweringBase::initActions() {
ISD::FTANH, ISD::FATAN2,
ISD::FMULADD, ISD::CONVERT_FROM_ARBITRARY_FP,
ISD::CONVERT_TO_ARBITRARY_FP,
- ISD::PSEUDO_FMIN, ISD::PSEUDO_FMAX},
+ ISD::PSEUDO_FMIN, ISD::PSEUDO_FMAX,
+ ISD::CT_SELECT},
VT, Expand);
// clang-format on
diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.h b/llvm/lib/Target/AArch64/AArch64ISelLowering.h
index f588d46ea9d2c..1b622e93480ec 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.h
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.h
@@ -167,8 +167,6 @@ class AArch64TargetLowering : public TargetLowering {
EVT getSetCCResultType(const DataLayout &DL, LLVMContext &Context,
EVT VT) const override;
- bool isCtSelectSupported(EVT VT) const override { return false; }
-
SDValue ReconstructShuffle(SDValue Op, SelectionDAG &DAG) const;
MachineBasicBlock *EmitF128CSEL(MachineInstr &MI,
diff --git a/llvm/lib/Target/ARM/ARMISelLowering.h b/llvm/lib/Target/ARM/ARMISelLowering.h
index 0d0cfd8c0cd68..4f95a1cdeb2ef 100644
--- a/llvm/lib/Target/ARM/ARMISelLowering.h
+++ b/llvm/lib/Target/ARM/ARMISelLowering.h
@@ -119,8 +119,6 @@ class VectorType;
return (Kind != ScalarCondVectorVal);
}
- bool isCtSelectSupported(EVT VT) const override { return false; }
-
bool isReadOnly(const GlobalValue *GV) const;
/// getSetCCResultType - Return the value type to use for ISD::SETCC.
diff --git a/llvm/lib/Target/X86/X86ISelLowering.h b/llvm/lib/Target/X86/X86ISelLowering.h
index 455125d94518d..5215bb0704dca 100644
--- a/llvm/lib/Target/X86/X86ISelLowering.h
+++ b/llvm/lib/Target/X86/X86ISelLowering.h
@@ -111,8 +111,6 @@ namespace llvm {
unsigned getJumpTableEncoding() const override;
bool useSoftFloat() const override;
- bool isCtSelectSupported(EVT VT) const override { return false; }
-
void markLibCallAttributes(MachineFunction *MF, unsigned CC,
ArgListTy &Args) const override;
diff --git a/llvm/test/CodeGen/RISCV/ctselect-fallback.ll b/llvm/test/CodeGen/RISCV/ctselect-fallback.ll
index c624c17d7e33e..d4617c7e75da7 100644
--- a/llvm/test/CodeGen/RISCV/ctselect-fallback.ll
+++ b/llvm/test/CodeGen/RISCV/ctselect-fallback.ll
@@ -203,8 +203,8 @@ define i32 @test_ctselect_load(i1 %cond, ptr %p1, ptr %p2) {
ret i32 %result
}
-; Test nested CTSELECT pattern with AND merging on i1 values
-; Pattern: ctselect C0, (ctselect C1, X, Y), Y -> ctselect (C0 & C1), X, Y
+; Test nested CT_SELECT pattern with AND merging on i1 values
+; Pattern: ct_select C0, (ct_select C1, X, Y), Y -> ct_select (C0 & C1), X, Y
define i32 @test_ctselect_nested_and_i1_to_i32(i1 %c0, i1 %c1, i32 %x, i32 %y) {
; RV64-LABEL: test_ctselect_nested_and_i1_to_i32:
; RV64: # %bb.0:
@@ -231,8 +231,8 @@ define i32 @test_ctselect_nested_and_i1_to_i32(i1 %c0, i1 %c1, i32 %x, i32 %y) {
ret i32 %result
}
-; Test nested CTSELECT pattern with OR merging on i1 values
-; Pattern: ctselect C0, X, (ctselect C1, X, Y) -> ctselect (C0 | C1), X, Y
+; Test nested CT_SELECT pattern with OR merging on i1 values
+; Pattern: ct_select C0, X, (ct_select C1, X, Y) -> ct_select (C0 | C1), X, Y
define i32 @test_ctselect_nested_or_i1_to_i32(i1 %c0, i1 %c1, i32 %x, i32 %y) {
; RV64-LABEL: test_ctselect_nested_or_i1_to_i32:
; RV64: # %bb.0:
@@ -259,9 +259,9 @@ define i32 @test_ctselect_nested_or_i1_to_i32(i1 %c0, i1 %c1, i32 %x, i32 %y) {
ret i32 %result
}
-; Test double nested CTSELECT with recursive AND merging
-; Pattern: ctselect C0, (ctselect C1, (ctselect C2, X, Y), Y), Y
-; -> ctselect (C0 & C1 & C2), X, Y
+; Test double nested CT_SELECT with recursive AND merging
+; Pattern: ct_select C0, (ct_select C1, (ct_select C2, X, Y), Y), Y
+; -> ct_select (C0 & C1 & C2), X, Y
define i32 @test_ctselect_double_nested_and_i1(i1 %c0, i1 %c1, i1 %c2, i32 %x, i32 %y) {
; RV64-LABEL: test_ctselect_double_nested_and_i1:
; RV64: # %bb.0:
@@ -291,7 +291,7 @@ define i32 @test_ctselect_double_nested_and_i1(i1 %c0, i1 %c1, i1 %c2, i32 %x, i
ret i32 %result
}
-; Test double nested CTSELECT with mixed AND/OR patterns
+; Test double nested CT_SELECT with mixed AND/OR patterns
define i32 @test_ctselect_double_nested_mixed_i1(i1 %c0, i1 %c1, i1 %c2, i32 %x, i32 %y, i32 %z) {
; RV64-LABEL: test_ctselect_double_nested_mixed_i1:
; RV64: # %bb.0:
@@ -329,7 +329,7 @@ define i32 @test_ctselect_double_nested_mixed_i1(i1 %c0, i1 %c1, i1 %c2, i32 %x,
ret i32 %result
}
-; Test nested ctselect calls
+; Test nested ct_select calls
define i32 @test_ctselect_nested(i1 %cond1, i1 %cond2, i32 %a, i32 %b, i32 %c) {
; RV64-LABEL: test_ctselect_nested:
; RV64: # %bb.0:
diff --git a/llvm/test/CodeGen/X86/ctselect.ll b/llvm/test/CodeGen/X86/ctselect.ll
index a970fd8933b9b..bf65e04721df1 100644
--- a/llvm/test/CodeGen/X86/ctselect.ll
+++ b/llvm/test/CodeGen/X86/ctselect.ll
@@ -552,7 +552,7 @@ define i32 @test_ctselect_load(i1 %cond, ptr %p1, ptr %p2) {
ret i32 %result
}
-; Test nested ctselect calls
+; Test nested ct_select calls
define i32 @test_ctselect_nested(i1 %cond1, i1 %cond2, i32 %a, i32 %b, i32 %c) {
; X64-LABEL: test_ctselect_nested:
; X64: # %bb.0:
@@ -631,8 +631,8 @@ define i32 @test_ctselect_nested(i1 %cond1, i1 %cond2, i32 %a, i32 %b, i32 %c) {
ret i32 %result
}
-; Test nested CTSELECT pattern with AND merging on i1 values
-; Pattern: ctselect C0, (ctselect C1, X, Y), Y -> ctselect (C0 & C1), X, Y
+; Test nested CT_SELECT pattern with AND merging on i1 values
+; Pattern: ct_select C0, (ct_select C1, X, Y), Y -> ct_select (C0 & C1), X, Y
; This optimization only applies when selecting between i1 values (boolean logic)
define i32 @test_ctselect_nested_and_i1_to_i32(i1 %c0, i1 %c1, i32 %x, i32 %y) {
; X64-LABEL: test_ctselect_nested_and_i1_to_i32:
@@ -679,8 +679,8 @@ define i32 @test_ctselect_nested_and_i1_to_i32(i1 %c0, i1 %c1, i32 %x, i32 %y) {
ret i32 %result
}
-; Test nested CTSELECT pattern with OR merging on i1 values
-; Pattern: ctselect C0, X, (ctselect C1, X, Y) -> ctselect (C0 | C1), X, Y
+; Test nested CT_SELECT pattern with OR merging on i1 values
+; Pattern: ct_select C0, X, (ct_select C1, X, Y) -> ct_select (C0 | C1), X, Y
; This optimization only applies when selecting between i1 values (boolean logic)
define i32 @test_ctselect_nested_or_i1_to_i32(i1 %c0, i1 %c1, i32 %x, i32 %y) {
; X64-LABEL: test_ctselect_nested_or_i1_to_i32:
@@ -727,10 +727,10 @@ define i32 @test_ctselect_nested_or_i1_to_i32(i1 %c0, i1 %c1, i32 %x, i32 %y) {
ret i32 %result
}
-; Test double nested CTSELECT with recursive AND merging
-; Pattern: ctselect C0, (ctselect C1, (ctselect C2, X, Y), Y), Y
-; -> ctselect C0, (ctselect (C1 & C2), X, Y), Y
-; -> ctselect (C0 & (C1 & C2)), X, Y
+; Test double nested CT_SELECT with recursive AND merging
+; Pattern: ct_select C0, (ct_select C1, (ct_select C2, X, Y), Y), Y
+; -> ct_select C0, (ct_select (C1 & C2), X, Y), Y
+; -> ct_select (C0 & (C1 & C2)), X, Y
; This tests that the optimization can be applied recursively
define i32 @test_ctselect_double_nested_and_i1(i1 %c0, i1 %c1, i1 %c2, i32 %x, i32 %y) {
; X64-LABEL: test_ctselect_double_nested_and_i1:
@@ -781,10 +781,10 @@ define i32 @test_ctselect_double_nested_and_i1(i1 %c0, i1 %c1, i1 %c2, i32 %x, i
ret i32 %result
}
-; Vector CTSELECT Tests
+; Vector CT_SELECT Tests
; ============================================================================
-; Test vector CTSELECT with v4i32 (128-bit vector with single i1 mask)
+; Test vector CT_SELECT with v4i32 (128-bit vector with single i1 mask)
; NOW CONSTANT-TIME: Uses bitwise XOR/AND operations instead of branches!
define <4 x i32> @test_ctselect_v4i32(i1 %cond, <4 x i32> %a, <4 x i32> %b) {
; X64-LABEL: test_ctselect_v4i32:
>From 57530a15d66424f78c2d3d3e48386cf3425a15b9 Mon Sep 17 00:00:00 2001
From: wizardengineer <juliuswoosebert at gmail.com>
Date: Sat, 7 Mar 2026 15:36:09 -0500
Subject: [PATCH 03/12] [ConstantTime] Fix CT_SELECT expansion to preserve
constant-time guarantees
Create CT_SELECT nodes for scalar types regardless of target support, so
they survive DAGCombiner (visitCT_SELECT is conservative). Expand to
AND/OR/XOR during operation legalization after SETCC is lowered, preventing
the sext(setcc)->select fold chain that converts constant-time patterns
into data-dependent conditional moves (e.g. movn/movz on MIPS).
The mask uses SUB(0, AND(Cond, 1)) instead of SIGN_EXTEND because type
legalization already promoted i1 to the SetCC result type, making
SIGN_EXTEND a no-op for same-width types.
---
llvm/lib/CodeGen/SelectionDAG/LegalizeDAG.cpp | 20 +-
.../SelectionDAG/SelectionDAGBuilder.cpp | 16 +-
llvm/test/CodeGen/RISCV/ctselect-fallback.ll | 22 +-
llvm/test/CodeGen/X86/ctselect.ll | 259 ++++++++++--------
4 files changed, 179 insertions(+), 138 deletions(-)
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeDAG.cpp b/llvm/lib/CodeGen/SelectionDAG/LegalizeDAG.cpp
index 43a94d17b8aee..a494ff2624aa8 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeDAG.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeDAG.cpp
@@ -4395,14 +4395,18 @@ bool SelectionDAGLegalize::ExpandNode(SDNode *Node) {
Node->getFlags()));
} else {
assert(VT.isInteger());
- EVT HalfVT = VT.getHalfSizedIntegerVT(*DAG.getContext());
- auto [Tmp2Lo, Tmp2Hi] = DAG.SplitScalar(Tmp2, dl, HalfVT, HalfVT);
- auto [Tmp3Lo, Tmp3Hi] = DAG.SplitScalar(Tmp3, dl, HalfVT, HalfVT);
- SDValue ResLo =
- DAG.getCTSelect(dl, HalfVT, Tmp1, Tmp2Lo, Tmp3Lo, Node->getFlags());
- SDValue ResHi =
- DAG.getCTSelect(dl, HalfVT, Tmp1, Tmp2Hi, Tmp3Hi, Node->getFlags());
- Tmp1 = DAG.getNode(ISD::BUILD_PAIR, dl, VT, ResLo, ResHi);
+ // Expand: Result = F ^ ((T ^ F) & Mask), Mask = 0 - (Cond & 1).
+ // SUB+AND creates the mask because i1 is already type-promoted;
+ // SIGN_EXTEND(i32, i32) would be a no-op leaving mask as 0/1.
+ SDValue Cond = Tmp1;
+ if (Cond.getValueType() != VT)
+ Cond = DAG.getNode(ISD::ANY_EXTEND, dl, VT, Cond);
+ SDValue Mask = DAG.getNode(
+ ISD::SUB, dl, VT, DAG.getConstant(0, dl, VT),
+ DAG.getNode(ISD::AND, dl, VT, Cond, DAG.getConstant(1, dl, VT)));
+ SDValue Diff = DAG.getNode(ISD::XOR, dl, VT, Tmp2, Tmp3);
+ Tmp1 = DAG.getNode(ISD::XOR, dl, VT, Tmp3,
+ DAG.getNode(ISD::AND, dl, VT, Diff, Mask));
Tmp1->setFlags(Node->getFlags());
}
Results.push_back(Tmp1);
diff --git a/llvm/lib/CodeGen/SelectionDAG/SelectionDAGBuilder.cpp b/llvm/lib/CodeGen/SelectionDAG/SelectionDAGBuilder.cpp
index de971b8c8a562..aa28b25dbeb51 100644
--- a/llvm/lib/CodeGen/SelectionDAG/SelectionDAGBuilder.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/SelectionDAGBuilder.cpp
@@ -7001,9 +7001,19 @@ void SelectionDAGBuilder::visitIntrinsicCall(const CallInst &I,
// assert if Cond type is Vector
assert(!CondVT.isVector() && "Vector type cond not supported yet");
- // Handle scalar types
- if (TLI.isOperationLegalOrCustom(ISD::CT_SELECT, VT) &&
- !CondVT.isVector()) {
+ // Create a CT_SELECT node for scalar types so it survives DAGCombiner
+ // (visitCT_SELECT is conservative) and expands to AND/OR/XOR during
+ // operation legalization, after SETCC is lowered. Unsupported vectors
+ // and floats with illegal integer equivalents (e.g. f64 on i386) use
+ // the inline fallback which runs before type legalization.
+ bool CreateNode =
+ TLI.isOperationLegalOrCustom(ISD::CT_SELECT, VT) ||
+ (!VT.isVector() &&
+ (!VT.isFloatingPoint() ||
+ TLI.isTypeLegal(
+ EVT::getIntegerVT(*DAG.getContext(), VT.getSizeInBits()))));
+
+ if (CreateNode) {
SDValue Result = DAG.getNode(ISD::CT_SELECT, DL, VT, Cond, A, B);
setValue(&I, Result);
return;
diff --git a/llvm/test/CodeGen/RISCV/ctselect-fallback.ll b/llvm/test/CodeGen/RISCV/ctselect-fallback.ll
index d4617c7e75da7..ee8072703ee31 100644
--- a/llvm/test/CodeGen/RISCV/ctselect-fallback.ll
+++ b/llvm/test/CodeGen/RISCV/ctselect-fallback.ll
@@ -101,8 +101,6 @@ define i32 @test_ctselect_const_true(i32 %a, i32 %b) {
;
; RV32-LABEL: test_ctselect_const_true:
; RV32: # %bb.0:
-; RV32-NEXT: xor a0, a0, a1
-; RV32-NEXT: xor a0, a1, a0
; RV32-NEXT: ret
%result = call i32 @llvm.ct.select.i32(i1 true, i32 %a, i32 %b)
ret i32 %result
@@ -208,7 +206,7 @@ define i32 @test_ctselect_load(i1 %cond, ptr %p1, ptr %p2) {
define i32 @test_ctselect_nested_and_i1_to_i32(i1 %c0, i1 %c1, i32 %x, i32 %y) {
; RV64-LABEL: test_ctselect_nested_and_i1_to_i32:
; RV64: # %bb.0:
-; RV64-NEXT: and a0, a1, a0
+; RV64-NEXT: and a0, a0, a1
; RV64-NEXT: xor a2, a2, a3
; RV64-NEXT: slli a0, a0, 63
; RV64-NEXT: srai a0, a0, 63
@@ -218,7 +216,7 @@ define i32 @test_ctselect_nested_and_i1_to_i32(i1 %c0, i1 %c1, i32 %x, i32 %y) {
;
; RV32-LABEL: test_ctselect_nested_and_i1_to_i32:
; RV32: # %bb.0:
-; RV32-NEXT: and a0, a1, a0
+; RV32-NEXT: and a0, a0, a1
; RV32-NEXT: xor a2, a2, a3
; RV32-NEXT: slli a0, a0, 31
; RV32-NEXT: srai a0, a0, 31
@@ -265,8 +263,8 @@ define i32 @test_ctselect_nested_or_i1_to_i32(i1 %c0, i1 %c1, i32 %x, i32 %y) {
define i32 @test_ctselect_double_nested_and_i1(i1 %c0, i1 %c1, i1 %c2, i32 %x, i32 %y) {
; RV64-LABEL: test_ctselect_double_nested_and_i1:
; RV64: # %bb.0:
-; RV64-NEXT: and a1, a2, a1
-; RV64-NEXT: and a0, a1, a0
+; RV64-NEXT: and a0, a0, a1
+; RV64-NEXT: and a0, a0, a2
; RV64-NEXT: xor a3, a3, a4
; RV64-NEXT: slli a0, a0, 63
; RV64-NEXT: srai a0, a0, 63
@@ -276,8 +274,8 @@ define i32 @test_ctselect_double_nested_and_i1(i1 %c0, i1 %c1, i1 %c2, i32 %x, i
;
; RV32-LABEL: test_ctselect_double_nested_and_i1:
; RV32: # %bb.0:
-; RV32-NEXT: and a1, a2, a1
-; RV32-NEXT: and a0, a1, a0
+; RV32-NEXT: and a0, a0, a1
+; RV32-NEXT: and a0, a0, a2
; RV32-NEXT: xor a3, a3, a4
; RV32-NEXT: slli a0, a0, 31
; RV32-NEXT: srai a0, a0, 31
@@ -295,7 +293,7 @@ define i32 @test_ctselect_double_nested_and_i1(i1 %c0, i1 %c1, i1 %c2, i32 %x, i
define i32 @test_ctselect_double_nested_mixed_i1(i1 %c0, i1 %c1, i1 %c2, i32 %x, i32 %y, i32 %z) {
; RV64-LABEL: test_ctselect_double_nested_mixed_i1:
; RV64: # %bb.0:
-; RV64-NEXT: and a0, a1, a0
+; RV64-NEXT: and a0, a0, a1
; RV64-NEXT: xor a3, a3, a4
; RV64-NEXT: or a0, a0, a2
; RV64-NEXT: slli a0, a0, 63
@@ -309,7 +307,7 @@ define i32 @test_ctselect_double_nested_mixed_i1(i1 %c0, i1 %c1, i1 %c2, i32 %x,
;
; RV32-LABEL: test_ctselect_double_nested_mixed_i1:
; RV32: # %bb.0:
-; RV32-NEXT: and a0, a1, a0
+; RV32-NEXT: and a0, a0, a1
; RV32-NEXT: xor a3, a3, a4
; RV32-NEXT: or a0, a0, a2
; RV32-NEXT: slli a0, a0, 31
@@ -382,7 +380,7 @@ define float @test_ctselect_f32_nan_inf(i1 %cond) {
; RV32-NEXT: srai a0, a0, 31
; RV32-NEXT: and a0, a0, a1
; RV32-NEXT: lui a1, 522240
-; RV32-NEXT: xor a0, a0, a1
+; RV32-NEXT: or a0, a0, a1
; RV32-NEXT: ret
%result = call float @llvm.ct.select.f32(i1 %cond, float 0x7FF8000000000000, float 0x7FF0000000000000)
ret float %result
@@ -398,7 +396,7 @@ define double @test_ctselect_f64_nan_inf(i1 %cond) {
; RV64-NEXT: and a0, a0, a1
; RV64-NEXT: li a1, 2047
; RV64-NEXT: slli a1, a1, 52
-; RV64-NEXT: xor a0, a0, a1
+; RV64-NEXT: or a0, a0, a1
; RV64-NEXT: ret
;
; RV32-LABEL: test_ctselect_f64_nan_inf:
diff --git a/llvm/test/CodeGen/X86/ctselect.ll b/llvm/test/CodeGen/X86/ctselect.ll
index bf65e04721df1..e1abae80cef4f 100644
--- a/llvm/test/CodeGen/X86/ctselect.ll
+++ b/llvm/test/CodeGen/X86/ctselect.ll
@@ -9,8 +9,8 @@ define i8 @test_ctselect_i8(i1 %cond, i8 %a, i8 %b) {
; X64-LABEL: test_ctselect_i8:
; X64: # %bb.0:
; X64-NEXT: movl %edi, %eax
-; X64-NEXT: xorl %edx, %esi
; X64-NEXT: andb $1, %al
+; X64-NEXT: xorl %edx, %esi
; X64-NEXT: negb %al
; X64-NEXT: andb %sil, %al
; X64-NEXT: xorb %dl, %al
@@ -20,10 +20,10 @@ define i8 @test_ctselect_i8(i1 %cond, i8 %a, i8 %b) {
; X32-LABEL: test_ctselect_i8:
; X32: # %bb.0:
; X32-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: andb $1, %al
; X32-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
; X32-NEXT: movzbl {{[0-9]+}}(%esp), %edx
; X32-NEXT: xorb %cl, %dl
-; X32-NEXT: andb $1, %al
; X32-NEXT: negb %al
; X32-NEXT: andb %dl, %al
; X32-NEXT: xorb %cl, %al
@@ -32,10 +32,10 @@ define i8 @test_ctselect_i8(i1 %cond, i8 %a, i8 %b) {
; X32-NOCMOV-LABEL: test_ctselect_i8:
; X32-NOCMOV: # %bb.0:
; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: andb $1, %al
; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %edx
; X32-NOCMOV-NEXT: xorb %cl, %dl
-; X32-NOCMOV-NEXT: andb $1, %al
; X32-NOCMOV-NEXT: negb %al
; X32-NOCMOV-NEXT: andb %dl, %al
; X32-NOCMOV-NEXT: xorb %cl, %al
@@ -58,10 +58,11 @@ define i32 @test_ctselect_i32(i1 %cond, i32 %a, i32 %b) {
; X32-LABEL: test_ctselect_i32:
; X32: # %bb.0:
; X32-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: andb $1, %al
; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
; X32-NEXT: xorl %ecx, %edx
-; X32-NEXT: andl $1, %eax
+; X32-NEXT: movzbl %al, %eax
; X32-NEXT: negl %eax
; X32-NEXT: andl %edx, %eax
; X32-NEXT: xorl %ecx, %eax
@@ -70,10 +71,11 @@ define i32 @test_ctselect_i32(i1 %cond, i32 %a, i32 %b) {
; X32-NOCMOV-LABEL: test_ctselect_i32:
; X32-NOCMOV: # %bb.0:
; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: andb $1, %al
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
; X32-NOCMOV-NEXT: xorl %ecx, %edx
-; X32-NOCMOV-NEXT: andl $1, %eax
+; X32-NOCMOV-NEXT: movzbl %al, %eax
; X32-NOCMOV-NEXT: negl %eax
; X32-NOCMOV-NEXT: andl %edx, %eax
; X32-NOCMOV-NEXT: xorl %ecx, %eax
@@ -95,45 +97,57 @@ define i64 @test_ctselect_i64(i1 %cond, i64 %a, i64 %b) {
;
; X32-LABEL: test_ctselect_i64:
; X32: # %bb.0:
-; X32-NEXT: pushl %esi
+; X32-NEXT: pushl %edi
; X32-NEXT: .cfi_def_cfa_offset 8
-; X32-NEXT: .cfi_offset %esi, -8
-; X32-NEXT: movzbl {{[0-9]+}}(%esp), %esi
-; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NEXT: pushl %esi
+; X32-NEXT: .cfi_def_cfa_offset 12
+; X32-NEXT: .cfi_offset %esi, -12
+; X32-NEXT: .cfi_offset %edi, -8
+; X32-NEXT: movzbl {{[0-9]+}}(%esp), %edx
+; X32-NEXT: andb $1, %dl
+; X32-NEXT: movl {{[0-9]+}}(%esp), %esi
; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
-; X32-NEXT: xorl %edx, %eax
-; X32-NEXT: andl $1, %esi
-; X32-NEXT: negl %esi
-; X32-NEXT: andl %esi, %eax
-; X32-NEXT: xorl %edx, %eax
+; X32-NEXT: xorl %esi, %eax
+; X32-NEXT: movzbl %dl, %edi
+; X32-NEXT: negl %edi
+; X32-NEXT: andl %edi, %eax
+; X32-NEXT: xorl %esi, %eax
; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
; X32-NEXT: xorl %ecx, %edx
-; X32-NEXT: andl %esi, %edx
+; X32-NEXT: andl %edi, %edx
; X32-NEXT: xorl %ecx, %edx
; X32-NEXT: popl %esi
+; X32-NEXT: .cfi_def_cfa_offset 8
+; X32-NEXT: popl %edi
; X32-NEXT: .cfi_def_cfa_offset 4
; X32-NEXT: retl
;
; X32-NOCMOV-LABEL: test_ctselect_i64:
; X32-NOCMOV: # %bb.0:
-; X32-NOCMOV-NEXT: pushl %esi
+; X32-NOCMOV-NEXT: pushl %edi
; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 8
-; X32-NOCMOV-NEXT: .cfi_offset %esi, -8
-; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %esi
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NOCMOV-NEXT: pushl %esi
+; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 12
+; X32-NOCMOV-NEXT: .cfi_offset %esi, -12
+; X32-NOCMOV-NEXT: .cfi_offset %edi, -8
+; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %edx
+; X32-NOCMOV-NEXT: andb $1, %dl
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %esi
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
-; X32-NOCMOV-NEXT: xorl %edx, %eax
-; X32-NOCMOV-NEXT: andl $1, %esi
-; X32-NOCMOV-NEXT: negl %esi
-; X32-NOCMOV-NEXT: andl %esi, %eax
-; X32-NOCMOV-NEXT: xorl %edx, %eax
+; X32-NOCMOV-NEXT: xorl %esi, %eax
+; X32-NOCMOV-NEXT: movzbl %dl, %edi
+; X32-NOCMOV-NEXT: negl %edi
+; X32-NOCMOV-NEXT: andl %edi, %eax
+; X32-NOCMOV-NEXT: xorl %esi, %eax
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
; X32-NOCMOV-NEXT: xorl %ecx, %edx
-; X32-NOCMOV-NEXT: andl %esi, %edx
+; X32-NOCMOV-NEXT: andl %edi, %edx
; X32-NOCMOV-NEXT: xorl %ecx, %edx
; X32-NOCMOV-NEXT: popl %esi
+; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 8
+; X32-NOCMOV-NEXT: popl %edi
; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 4
; X32-NOCMOV-NEXT: retl
%result = call i64 @llvm.ct.select.i64(i1 %cond, i64 %a, i64 %b)
@@ -155,37 +169,47 @@ define float @test_ctselect_f32(i1 %cond, float %a, float %b) {
;
; X32-LABEL: test_ctselect_f32:
; X32: # %bb.0:
-; X32-NEXT: pushl %eax
-; X32-NEXT: .cfi_def_cfa_offset 8
+; X32-NEXT: subl $12, %esp
+; X32-NEXT: .cfi_def_cfa_offset 16
+; X32-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NEXT: fstps {{[0-9]+}}(%esp)
+; X32-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NEXT: fstps (%esp)
; X32-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: andb $1, %al
; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
-; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
-; X32-NEXT: xorl %ecx, %edx
-; X32-NEXT: andl $1, %eax
+; X32-NEXT: movzbl %al, %eax
; X32-NEXT: negl %eax
-; X32-NEXT: andl %edx, %eax
-; X32-NEXT: xorl %ecx, %eax
-; X32-NEXT: movl %eax, (%esp)
-; X32-NEXT: flds (%esp)
-; X32-NEXT: popl %eax
+; X32-NEXT: movl (%esp), %edx
+; X32-NEXT: xorl %ecx, %edx
+; X32-NEXT: andl %eax, %edx
+; X32-NEXT: xorl %ecx, %edx
+; X32-NEXT: movl %edx, {{[0-9]+}}(%esp)
+; X32-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NEXT: addl $12, %esp
; X32-NEXT: .cfi_def_cfa_offset 4
; X32-NEXT: retl
;
; X32-NOCMOV-LABEL: test_ctselect_f32:
; X32-NOCMOV: # %bb.0:
-; X32-NOCMOV-NEXT: pushl %eax
-; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 8
+; X32-NOCMOV-NEXT: subl $12, %esp
+; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 16
+; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: fstps {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: fstps (%esp)
; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: andb $1, %al
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
-; X32-NOCMOV-NEXT: xorl %ecx, %edx
-; X32-NOCMOV-NEXT: andl $1, %eax
+; X32-NOCMOV-NEXT: movzbl %al, %eax
; X32-NOCMOV-NEXT: negl %eax
-; X32-NOCMOV-NEXT: andl %edx, %eax
-; X32-NOCMOV-NEXT: xorl %ecx, %eax
-; X32-NOCMOV-NEXT: movl %eax, (%esp)
-; X32-NOCMOV-NEXT: flds (%esp)
-; X32-NOCMOV-NEXT: popl %eax
+; X32-NOCMOV-NEXT: movl (%esp), %edx
+; X32-NOCMOV-NEXT: xorl %ecx, %edx
+; X32-NOCMOV-NEXT: andl %eax, %edx
+; X32-NOCMOV-NEXT: xorl %ecx, %edx
+; X32-NOCMOV-NEXT: movl %edx, {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: addl $12, %esp
; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 4
; X32-NOCMOV-NEXT: retl
%result = call float @llvm.ct.select.f32(i1 %cond, float %a, float %b)
@@ -281,10 +305,11 @@ define ptr @test_ctselect_ptr(i1 %cond, ptr %a, ptr %b) {
; X32-LABEL: test_ctselect_ptr:
; X32: # %bb.0:
; X32-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: andb $1, %al
; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
; X32-NEXT: xorl %ecx, %edx
-; X32-NEXT: andl $1, %eax
+; X32-NEXT: movzbl %al, %eax
; X32-NEXT: negl %eax
; X32-NEXT: andl %edx, %eax
; X32-NEXT: xorl %ecx, %eax
@@ -293,10 +318,11 @@ define ptr @test_ctselect_ptr(i1 %cond, ptr %a, ptr %b) {
; X32-NOCMOV-LABEL: test_ctselect_ptr:
; X32-NOCMOV: # %bb.0:
; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: andb $1, %al
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
; X32-NOCMOV-NEXT: xorl %ecx, %edx
-; X32-NOCMOV-NEXT: andl $1, %eax
+; X32-NOCMOV-NEXT: movzbl %al, %eax
; X32-NOCMOV-NEXT: negl %eax
; X32-NOCMOV-NEXT: andl %edx, %eax
; X32-NOCMOV-NEXT: xorl %ecx, %eax
@@ -310,24 +336,16 @@ define i32 @test_ctselect_const_true(i32 %a, i32 %b) {
; X64-LABEL: test_ctselect_const_true:
; X64: # %bb.0:
; X64-NEXT: movl %edi, %eax
-; X64-NEXT: xorl %esi, %eax
-; X64-NEXT: xorl %esi, %eax
; X64-NEXT: retq
;
; X32-LABEL: test_ctselect_const_true:
; X32: # %bb.0:
-; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
-; X32-NEXT: xorl %ecx, %eax
-; X32-NEXT: xorl %ecx, %eax
; X32-NEXT: retl
;
; X32-NOCMOV-LABEL: test_ctselect_const_true:
; X32-NOCMOV: # %bb.0:
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
-; X32-NOCMOV-NEXT: xorl %ecx, %eax
-; X32-NOCMOV-NEXT: xorl %ecx, %eax
; X32-NOCMOV-NEXT: retl
%result = call i32 @llvm.ct.select.i32(i1 true, i32 %a, i32 %b)
ret i32 %result
@@ -341,14 +359,12 @@ define i32 @test_ctselect_const_false(i32 %a, i32 %b) {
;
; X32-LABEL: test_ctselect_const_false:
; X32: # %bb.0:
-; X32-NEXT: xorl %eax, %eax
-; X32-NEXT: xorl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
; X32-NEXT: retl
;
; X32-NOCMOV-LABEL: test_ctselect_const_false:
; X32-NOCMOV: # %bb.0:
-; X32-NOCMOV-NEXT: xorl %eax, %eax
-; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
; X32-NOCMOV-NEXT: retl
%result = call i32 @llvm.ct.select.i32(i1 false, i32 %a, i32 %b)
ret i32 %result
@@ -443,19 +459,20 @@ define i32 @test_ctselect_icmp_ult(i32 %x, i32 %y, i32 %a, i32 %b) {
define float @test_ctselect_fcmp_oeq(float %x, float %y, float %a, float %b) {
; X64-LABEL: test_ctselect_fcmp_oeq:
; X64: # %bb.0:
-; X64-NEXT: movd %xmm3, %eax
; X64-NEXT: cmpeqss %xmm1, %xmm0
-; X64-NEXT: pxor %xmm3, %xmm2
-; X64-NEXT: pand %xmm0, %xmm2
-; X64-NEXT: movd %xmm2, %ecx
-; X64-NEXT: xorl %eax, %ecx
-; X64-NEXT: movd %ecx, %xmm0
+; X64-NEXT: xorps %xmm3, %xmm2
+; X64-NEXT: andps %xmm2, %xmm0
+; X64-NEXT: xorps %xmm3, %xmm0
; X64-NEXT: retq
;
; X32-LABEL: test_ctselect_fcmp_oeq:
; X32: # %bb.0:
-; X32-NEXT: pushl %eax
-; X32-NEXT: .cfi_def_cfa_offset 8
+; X32-NEXT: subl $12, %esp
+; X32-NEXT: .cfi_def_cfa_offset 16
+; X32-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NEXT: fstps {{[0-9]+}}(%esp)
+; X32-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NEXT: fstps (%esp)
; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
; X32-NEXT: flds {{[0-9]+}}(%esp)
; X32-NEXT: flds {{[0-9]+}}(%esp)
@@ -466,20 +483,24 @@ define float @test_ctselect_fcmp_oeq(float %x, float %y, float %a, float %b) {
; X32-NEXT: andb %cl, %dl
; X32-NEXT: movzbl %dl, %ecx
; X32-NEXT: negl %ecx
-; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NEXT: movl (%esp), %edx
; X32-NEXT: xorl %eax, %edx
; X32-NEXT: andl %ecx, %edx
; X32-NEXT: xorl %eax, %edx
-; X32-NEXT: movl %edx, (%esp)
-; X32-NEXT: flds (%esp)
-; X32-NEXT: popl %eax
+; X32-NEXT: movl %edx, {{[0-9]+}}(%esp)
+; X32-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NEXT: addl $12, %esp
; X32-NEXT: .cfi_def_cfa_offset 4
; X32-NEXT: retl
;
; X32-NOCMOV-LABEL: test_ctselect_fcmp_oeq:
; X32-NOCMOV: # %bb.0:
-; X32-NOCMOV-NEXT: pushl %eax
-; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 8
+; X32-NOCMOV-NEXT: subl $12, %esp
+; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 16
+; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: fstps {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: fstps (%esp)
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
@@ -492,13 +513,13 @@ define float @test_ctselect_fcmp_oeq(float %x, float %y, float %a, float %b) {
; X32-NOCMOV-NEXT: andb %al, %dl
; X32-NOCMOV-NEXT: movzbl %dl, %eax
; X32-NOCMOV-NEXT: negl %eax
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NOCMOV-NEXT: movl (%esp), %edx
; X32-NOCMOV-NEXT: xorl %ecx, %edx
; X32-NOCMOV-NEXT: andl %eax, %edx
; X32-NOCMOV-NEXT: xorl %ecx, %edx
-; X32-NOCMOV-NEXT: movl %edx, (%esp)
-; X32-NOCMOV-NEXT: flds (%esp)
-; X32-NOCMOV-NEXT: popl %eax
+; X32-NOCMOV-NEXT: movl %edx, {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: addl $12, %esp
; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 4
; X32-NOCMOV-NEXT: retl
%cond = fcmp oeq float %x, %y
@@ -522,12 +543,13 @@ define i32 @test_ctselect_load(i1 %cond, ptr %p1, ptr %p2) {
; X32-LABEL: test_ctselect_load:
; X32: # %bb.0:
; X32-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: andb $1, %al
; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
; X32-NEXT: movl (%edx), %edx
; X32-NEXT: movl (%ecx), %ecx
; X32-NEXT: xorl %edx, %ecx
-; X32-NEXT: andl $1, %eax
+; X32-NEXT: movzbl %al, %eax
; X32-NEXT: negl %eax
; X32-NEXT: andl %ecx, %eax
; X32-NEXT: xorl %edx, %eax
@@ -536,12 +558,13 @@ define i32 @test_ctselect_load(i1 %cond, ptr %p1, ptr %p2) {
; X32-NOCMOV-LABEL: test_ctselect_load:
; X32-NOCMOV: # %bb.0:
; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: andb $1, %al
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
; X32-NOCMOV-NEXT: movl (%edx), %edx
; X32-NOCMOV-NEXT: movl (%ecx), %ecx
; X32-NOCMOV-NEXT: xorl %edx, %ecx
-; X32-NOCMOV-NEXT: andl $1, %eax
+; X32-NOCMOV-NEXT: movzbl %al, %eax
; X32-NOCMOV-NEXT: negl %eax
; X32-NOCMOV-NEXT: andl %ecx, %eax
; X32-NOCMOV-NEXT: xorl %edx, %eax
@@ -578,17 +601,19 @@ define i32 @test_ctselect_nested(i1 %cond1, i1 %cond2, i32 %a, i32 %b, i32 %c) {
; X32-NEXT: .cfi_offset %esi, -12
; X32-NEXT: .cfi_offset %edi, -8
; X32-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: andb $1, %al
+; X32-NEXT: movb {{[0-9]+}}(%esp), %ah
+; X32-NEXT: andb $1, %ah
; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
-; X32-NEXT: movzbl {{[0-9]+}}(%esp), %esi
; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
-; X32-NEXT: movl {{[0-9]+}}(%esp), %edi
-; X32-NEXT: xorl %edx, %edi
-; X32-NEXT: andl $1, %esi
-; X32-NEXT: negl %esi
-; X32-NEXT: andl %edi, %esi
+; X32-NEXT: movl {{[0-9]+}}(%esp), %esi
+; X32-NEXT: xorl %edx, %esi
+; X32-NEXT: movzbl %ah, %edi
+; X32-NEXT: negl %edi
+; X32-NEXT: andl %esi, %edi
; X32-NEXT: xorl %ecx, %edx
-; X32-NEXT: xorl %esi, %edx
-; X32-NEXT: andl $1, %eax
+; X32-NEXT: xorl %edi, %edx
+; X32-NEXT: movzbl %al, %eax
; X32-NEXT: negl %eax
; X32-NEXT: andl %edx, %eax
; X32-NEXT: xorl %ecx, %eax
@@ -607,17 +632,19 @@ define i32 @test_ctselect_nested(i1 %cond1, i1 %cond2, i32 %a, i32 %b, i32 %c) {
; X32-NOCMOV-NEXT: .cfi_offset %esi, -12
; X32-NOCMOV-NEXT: .cfi_offset %edi, -8
; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: andb $1, %al
+; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %ah
+; X32-NOCMOV-NEXT: andb $1, %ah
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
-; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %esi
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edi
-; X32-NOCMOV-NEXT: xorl %edx, %edi
-; X32-NOCMOV-NEXT: andl $1, %esi
-; X32-NOCMOV-NEXT: negl %esi
-; X32-NOCMOV-NEXT: andl %edi, %esi
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %esi
+; X32-NOCMOV-NEXT: xorl %edx, %esi
+; X32-NOCMOV-NEXT: movzbl %ah, %edi
+; X32-NOCMOV-NEXT: negl %edi
+; X32-NOCMOV-NEXT: andl %esi, %edi
; X32-NOCMOV-NEXT: xorl %ecx, %edx
-; X32-NOCMOV-NEXT: xorl %esi, %edx
-; X32-NOCMOV-NEXT: andl $1, %eax
+; X32-NOCMOV-NEXT: xorl %edi, %edx
+; X32-NOCMOV-NEXT: movzbl %al, %eax
; X32-NOCMOV-NEXT: negl %eax
; X32-NOCMOV-NEXT: andl %edx, %eax
; X32-NOCMOV-NEXT: xorl %ecx, %eax
@@ -651,10 +678,10 @@ define i32 @test_ctselect_nested_and_i1_to_i32(i1 %c0, i1 %c1, i32 %x, i32 %y) {
; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
; X32-NEXT: movzbl {{[0-9]+}}(%esp), %eax
; X32-NEXT: andb {{[0-9]+}}(%esp), %al
-; X32-NEXT: movzbl %al, %eax
+; X32-NEXT: andb $1, %al
; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
; X32-NEXT: xorl %ecx, %edx
-; X32-NEXT: andl $1, %eax
+; X32-NEXT: movzbl %al, %eax
; X32-NEXT: negl %eax
; X32-NEXT: andl %edx, %eax
; X32-NEXT: xorl %ecx, %eax
@@ -665,10 +692,10 @@ define i32 @test_ctselect_nested_and_i1_to_i32(i1 %c0, i1 %c1, i32 %x, i32 %y) {
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %eax
; X32-NOCMOV-NEXT: andb {{[0-9]+}}(%esp), %al
-; X32-NOCMOV-NEXT: movzbl %al, %eax
+; X32-NOCMOV-NEXT: andb $1, %al
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
; X32-NOCMOV-NEXT: xorl %ecx, %edx
-; X32-NOCMOV-NEXT: andl $1, %eax
+; X32-NOCMOV-NEXT: movzbl %al, %eax
; X32-NOCMOV-NEXT: negl %eax
; X32-NOCMOV-NEXT: andl %edx, %eax
; X32-NOCMOV-NEXT: xorl %ecx, %eax
@@ -699,10 +726,10 @@ define i32 @test_ctselect_nested_or_i1_to_i32(i1 %c0, i1 %c1, i32 %x, i32 %y) {
; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
; X32-NEXT: movzbl {{[0-9]+}}(%esp), %eax
; X32-NEXT: orb {{[0-9]+}}(%esp), %al
-; X32-NEXT: movzbl %al, %eax
+; X32-NEXT: andb $1, %al
; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
; X32-NEXT: xorl %ecx, %edx
-; X32-NEXT: andl $1, %eax
+; X32-NEXT: movzbl %al, %eax
; X32-NEXT: negl %eax
; X32-NEXT: andl %edx, %eax
; X32-NEXT: xorl %ecx, %eax
@@ -713,10 +740,10 @@ define i32 @test_ctselect_nested_or_i1_to_i32(i1 %c0, i1 %c1, i32 %x, i32 %y) {
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %eax
; X32-NOCMOV-NEXT: orb {{[0-9]+}}(%esp), %al
-; X32-NOCMOV-NEXT: movzbl %al, %eax
+; X32-NOCMOV-NEXT: andb $1, %al
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
; X32-NOCMOV-NEXT: xorl %ecx, %edx
-; X32-NOCMOV-NEXT: andl $1, %eax
+; X32-NOCMOV-NEXT: movzbl %al, %eax
; X32-NOCMOV-NEXT: negl %eax
; X32-NOCMOV-NEXT: andl %edx, %eax
; X32-NOCMOV-NEXT: xorl %ecx, %eax
@@ -735,9 +762,9 @@ define i32 @test_ctselect_nested_or_i1_to_i32(i1 %c0, i1 %c1, i32 %x, i32 %y) {
define i32 @test_ctselect_double_nested_and_i1(i1 %c0, i1 %c1, i1 %c2, i32 %x, i32 %y) {
; X64-LABEL: test_ctselect_double_nested_and_i1:
; X64: # %bb.0:
-; X64-NEXT: movl %esi, %eax
+; X64-NEXT: movl %edi, %eax
+; X64-NEXT: andl %esi, %eax
; X64-NEXT: andl %edx, %eax
-; X64-NEXT: andl %edi, %eax
; X64-NEXT: xorl %r8d, %ecx
; X64-NEXT: andl $1, %eax
; X64-NEXT: negl %eax
@@ -751,10 +778,10 @@ define i32 @test_ctselect_double_nested_and_i1(i1 %c0, i1 %c1, i1 %c2, i32 %x, i
; X32-NEXT: movzbl {{[0-9]+}}(%esp), %eax
; X32-NEXT: andb {{[0-9]+}}(%esp), %al
; X32-NEXT: andb {{[0-9]+}}(%esp), %al
-; X32-NEXT: movzbl %al, %eax
+; X32-NEXT: andb $1, %al
; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
; X32-NEXT: xorl %ecx, %edx
-; X32-NEXT: andl $1, %eax
+; X32-NEXT: movzbl %al, %eax
; X32-NEXT: negl %eax
; X32-NEXT: andl %edx, %eax
; X32-NEXT: xorl %ecx, %eax
@@ -766,10 +793,10 @@ define i32 @test_ctselect_double_nested_and_i1(i1 %c0, i1 %c1, i1 %c2, i32 %x, i
; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %eax
; X32-NOCMOV-NEXT: andb {{[0-9]+}}(%esp), %al
; X32-NOCMOV-NEXT: andb {{[0-9]+}}(%esp), %al
-; X32-NOCMOV-NEXT: movzbl %al, %eax
+; X32-NOCMOV-NEXT: andb $1, %al
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
; X32-NOCMOV-NEXT: xorl %ecx, %edx
-; X32-NOCMOV-NEXT: andl $1, %eax
+; X32-NOCMOV-NEXT: movzbl %al, %eax
; X32-NOCMOV-NEXT: negl %eax
; X32-NOCMOV-NEXT: andl %edx, %eax
; X32-NOCMOV-NEXT: xorl %ecx, %eax
@@ -1403,7 +1430,7 @@ define float @test_ctselect_f32_nan_inf(i1 %cond) {
; X64-NEXT: andl $1, %edi
; X64-NEXT: negl %edi
; X64-NEXT: andl $4194304, %edi # imm = 0x400000
-; X64-NEXT: xorl $2139095040, %edi # imm = 0x7F800000
+; X64-NEXT: orl $2139095040, %edi # imm = 0x7F800000
; X64-NEXT: movd %edi, %xmm0
; X64-NEXT: retq
;
@@ -1412,10 +1439,11 @@ define float @test_ctselect_f32_nan_inf(i1 %cond) {
; X32-NEXT: pushl %eax
; X32-NEXT: .cfi_def_cfa_offset 8
; X32-NEXT: movzbl {{[0-9]+}}(%esp), %eax
-; X32-NEXT: andl $1, %eax
+; X32-NEXT: andb $1, %al
+; X32-NEXT: movzbl %al, %eax
; X32-NEXT: negl %eax
; X32-NEXT: andl $4194304, %eax # imm = 0x400000
-; X32-NEXT: xorl $2139095040, %eax # imm = 0x7F800000
+; X32-NEXT: orl $2139095040, %eax # imm = 0x7F800000
; X32-NEXT: movl %eax, (%esp)
; X32-NEXT: flds (%esp)
; X32-NEXT: popl %eax
@@ -1427,10 +1455,11 @@ define float @test_ctselect_f32_nan_inf(i1 %cond) {
; X32-NOCMOV-NEXT: pushl %eax
; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 8
; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %eax
-; X32-NOCMOV-NEXT: andl $1, %eax
+; X32-NOCMOV-NEXT: andb $1, %al
+; X32-NOCMOV-NEXT: movzbl %al, %eax
; X32-NOCMOV-NEXT: negl %eax
; X32-NOCMOV-NEXT: andl $4194304, %eax # imm = 0x400000
-; X32-NOCMOV-NEXT: xorl $2139095040, %eax # imm = 0x7F800000
+; X32-NOCMOV-NEXT: orl $2139095040, %eax # imm = 0x7F800000
; X32-NOCMOV-NEXT: movl %eax, (%esp)
; X32-NOCMOV-NEXT: flds (%esp)
; X32-NOCMOV-NEXT: popl %eax
@@ -1449,7 +1478,7 @@ define double @test_ctselect_f64_nan_inf(i1 %cond) {
; X64-NEXT: movabsq $2251799813685248, %rax # imm = 0x8000000000000
; X64-NEXT: andq %rdi, %rax
; X64-NEXT: movabsq $9218868437227405312, %rcx # imm = 0x7FF0000000000000
-; X64-NEXT: xorq %rax, %rcx
+; X64-NEXT: orq %rax, %rcx
; X64-NEXT: movq %rcx, %xmm0
; X64-NEXT: retq
;
>From 28a2da723e47c0be2bbdfe5bb747169c3939cf0c Mon Sep 17 00:00:00 2001
From: AkshayK <iit.akshay at gmail.com>
Date: Thu, 21 May 2026 17:19:01 -0400
Subject: [PATCH 04/12] [ConstantTime] Move ct.select lowering to legalizer;
misc fixes
- LegalizeDAG owns CT_SELECT lowering: scalar-mask+splat for vectors
(avoids illegal vNi1), memory-blend for FP types lacking a legal
same-size integer (f64 on i386 no-SSE, fp128, x86_fp80)
- Add InstSimplify folds for constant cond and identical arms
- Extend X86 tests to half/bfloat/fp128/x86_fp80; nounwind cleanup
- Preserve flags through legalizer; reuse SDTSelect
---
.../include/llvm/Target/TargetSelectionDAG.td | 7 +-
llvm/lib/Analysis/InstructionSimplify.cpp | 16 +
llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp | 17 +-
llvm/lib/CodeGen/SelectionDAG/LegalizeDAG.cpp | 208 +-
.../SelectionDAG/LegalizeFloatTypes.cpp | 4 +-
.../SelectionDAG/LegalizeVectorTypes.cpp | 2 +-
.../SelectionDAG/SelectionDAGBuilder.cpp | 123 +-
.../SelectionDAG/SelectionDAGBuilder.h | 3 -
llvm/test/CodeGen/RISCV/ctselect-fallback.ll | 74 +-
llvm/test/CodeGen/X86/ctselect.ll | 2080 ++++++++++++-----
.../test/Transforms/InstSimplify/ct-select.ll | 102 +
11 files changed, 1847 insertions(+), 789 deletions(-)
create mode 100644 llvm/test/Transforms/InstSimplify/ct-select.ll
diff --git a/llvm/include/llvm/Target/TargetSelectionDAG.td b/llvm/include/llvm/Target/TargetSelectionDAG.td
index 4c5a4c884e745..a855f360dcd47 100644
--- a/llvm/include/llvm/Target/TargetSelectionDAG.td
+++ b/llvm/include/llvm/Target/TargetSelectionDAG.td
@@ -221,11 +221,6 @@ def SDTSelect : SDTypeProfile<1, 3, [ // select
SDTCisInt<1>, SDTCisSameAs<0, 2>, SDTCisSameAs<2, 3>
]>;
-def SDTCtSelect
- : SDTypeProfile<1, 3,
- [ // ctselect
- SDTCisInt<1>, SDTCisSameAs<0, 2>, SDTCisSameAs<2, 3>]>;
-
def SDTVSelect : SDTypeProfile<1, 3, [ // vselect
SDTCisVec<0>, SDTCisInt<1>, SDTCisSameAs<0, 2>, SDTCisSameAs<2, 3>, SDTCisSameNumEltsAs<0, 1>
]>;
@@ -790,7 +785,7 @@ def reset_fpmode : SDNode<"ISD::RESET_FPMODE", SDTNone, [SDNPHasChain]>;
def setcc : SDNode<"ISD::SETCC" , SDTSetCC>;
def select : SDNode<"ISD::SELECT" , SDTSelect>;
-def ct_select : SDNode<"ISD::CT_SELECT", SDTCtSelect>;
+def ct_select : SDNode<"ISD::CT_SELECT", SDTSelect>;
def vselect : SDNode<"ISD::VSELECT" , SDTVSelect>;
def selectcc : SDNode<"ISD::SELECT_CC" , SDTSelectCC>;
diff --git a/llvm/lib/Analysis/InstructionSimplify.cpp b/llvm/lib/Analysis/InstructionSimplify.cpp
index d9aebc757d636..602eb6b63e133 100644
--- a/llvm/lib/Analysis/InstructionSimplify.cpp
+++ b/llvm/lib/Analysis/InstructionSimplify.cpp
@@ -7615,6 +7615,22 @@ static Value *simplifyIntrinsic(CallBase *Call, ArrayRef<Value *> Args,
}
return nullptr;
}
+ case Intrinsic::ct_select: {
+ // Only fold on a literal IR-constant condition or identical arms. Folding
+ // through ValueTracking-derived known bits would defeat the constant-time
+ // contract on conditions the user wants kept opaque.
+ Value *Cond = Args[0], *TrueVal = Args[1], *FalseVal = Args[2];
+ if (auto *CI = dyn_cast<ConstantInt>(Cond)) {
+ if (CI->isOne())
+ return TrueVal;
+ if (CI->isZero())
+ return FalseVal;
+ }
+ // ct.select C, X, X -> X: the condition is unused, so no secret is exposed.
+ if (TrueVal == FalseVal)
+ return TrueVal;
+ return nullptr;
+ }
default: {
// Use the default FP environment if none is found.
fp::ExceptionBehavior ExBehavior = fp::ebIgnore;
diff --git a/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp b/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp
index 0e46efe269ed3..6016f37e855c5 100644
--- a/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp
@@ -13705,11 +13705,8 @@ SDValue DAGCombiner::visitCT_SELECT(SDNode *N) {
// This is a CT-safe canonicalization: flip negated condition by swapping
// arms. extractBooleanFlip only matches boolean xor-with-1, so this preserves
// dataflow semantics and does not introduce data-dependent control flow.
- if (SDValue F = extractBooleanFlip(N0, DAG, TLI, false)) {
- SDValue SelectOp = DAG.getNode(ISD::CT_SELECT, DL, VT, F, N2, N1);
- SelectOp->setFlags(Flags);
- return SelectOp;
- }
+ if (SDValue F = extractBooleanFlip(N0, DAG, TLI, false))
+ return DAG.getCTSelect(DL, VT, F, N2, N1, Flags);
if (VT0 == MVT::i1) {
// Nested CT_SELECT merging optimizations for i1 conditions.
@@ -13729,10 +13726,7 @@ SDValue DAGCombiner::visitCT_SELECT(SDNode *N) {
SDValue N1_2 = N1->getOperand(2);
if (N1_2 == N2 && N0.getValueType() == N1_0.getValueType()) {
SDValue And = DAG.getNode(ISD::AND, DL, N0.getValueType(), N0, N1_0);
- SDValue SelectOp =
- DAG.getNode(ISD::CT_SELECT, DL, N1.getValueType(), And, N1_1, N2);
- SelectOp->setFlags(Flags);
- return SelectOp;
+ return DAG.getCTSelect(DL, N1.getValueType(), And, N1_1, N2, Flags);
}
}
@@ -13745,10 +13739,7 @@ SDValue DAGCombiner::visitCT_SELECT(SDNode *N) {
SDValue N2_2 = N2->getOperand(2);
if (N2_1 == N1 && N0.getValueType() == N2_0.getValueType()) {
SDValue Or = DAG.getNode(ISD::OR, DL, N0.getValueType(), N0, N2_0);
- SDValue SelectOp =
- DAG.getNode(ISD::CT_SELECT, DL, N1.getValueType(), Or, N1, N2_2);
- SelectOp->setFlags(Flags);
- return SelectOp;
+ return DAG.getCTSelect(DL, N1.getValueType(), Or, N1, N2_2, Flags);
}
}
}
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeDAG.cpp b/llvm/lib/CodeGen/SelectionDAG/LegalizeDAG.cpp
index a494ff2624aa8..92957d2b22bf9 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeDAG.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeDAG.cpp
@@ -26,6 +26,7 @@
#include "llvm/CodeGen/MachineFunction.h"
#include "llvm/CodeGen/MachineJumpTableInfo.h"
#include "llvm/CodeGen/MachineMemOperand.h"
+#include "llvm/CodeGen/MachineRegisterInfo.h"
#include "llvm/CodeGen/RuntimeLibcallUtil.h"
#include "llvm/CodeGen/SelectionDAG.h"
#include "llvm/CodeGen/SelectionDAGNodes.h"
@@ -4337,78 +4338,155 @@ bool SelectionDAGLegalize::ExpandNode(SDNode *Node) {
Results.push_back(Tmp1);
break;
case ISD::CT_SELECT: {
- Tmp1 = Node->getOperand(0);
- Tmp2 = Node->getOperand(1);
- Tmp3 = Node->getOperand(2);
+ // Constant-time select: F ^ ((T ^ F) & Mask), Mask = 0 - (cond & 1).
+ // Bitwise-only — no select/cmov SDNode is constructed, so the CT property
+ // holds against any combiner that targets those opcodes. FP types operate
+ // on the same-size integer; vectors build the mask as a scalar then splat
+ // (avoids illegal vNi1).
+ //
+ // The masked-diff is routed through a virtual register (CopyToReg /
+ // CopyFromReg) below as a forward-looking DAGCombine barrier. This is
+ // *not* required for correctness against any combiner in tree today —
+ // DAGCombiner has no rewrite that recognizes XOR/AND/XOR-with-sext-mask
+ // and reconstructs a SELECT. The chain edge is defense-in-depth against
+ // a hypothetical future fold of that form: the dependency partitions the
+ // bitwise sequence into a region the combiner can't see through. Cost is
+ // at most a coalesce-able MOV per call. Swap for ARITH_FENCE when its
+ // int/vector extension lands.
+ Tmp1 = Node->getOperand(0); // cond
+ Tmp2 = Node->getOperand(1); // T
+ Tmp3 = Node->getOperand(2); // F
EVT VT = Tmp2.getValueType();
- if (VT.isVector()) {
- // Constant-time vector blending using pattern F ^ ((T ^ F) & Mask)
- // where Mask = broadcast(i1 ? -1 : 0) to match vector element width.
- //
- // This formulation uses only XOR and AND operations, avoiding branches
- // that would leak timing information. It's equivalent to:
- // Mask==0xFF: F ^ ((T ^ F) & 0xFF) = F ^ (T ^ F) = T
- // Mask==0x00: F ^ ((T ^ F) & 0x00) = F ^ 0 = F
-
- EVT IntVT = VT;
- SDValue T = Tmp2; // True value
- SDValue F = Tmp3; // False value
-
- // Step 1: Handle floating-point vectors by bitcasting to integer
- if (VT.isFloatingPoint()) {
- IntVT = EVT::getVectorVT(
+
+ // Memory-blend for FP scalars whose same-size integer isn't legal (f64 on
+ // i386 no-SSE, x86_fp80, fp128). The bitcast-to-int expansion below can't
+ // run since LegalizeDAG is the last legalization stage. Spill T/F, blend
+ // chunk-by-chunk at a legal int width, reload as the FP type. Fixed and
+ // scalable FP vectors fall through to the unified path below.
+ if (VT.isFloatingPoint() && !VT.isVector() &&
+ !TLI.isTypeLegal(VT.changeTypeToInteger())) {
+ const DataLayout &DL = DAG.getDataLayout();
+ Type *VTTy = VT.getTypeForEVT(*DAG.getContext());
+ unsigned StorageBytes = DL.getTypeStoreSize(VTTy);
+ assert(StorageBytes > 0 && "FP type with zero storage size");
+
+ // Pick the largest legal scalar integer chunk that divides StorageBytes
+ // evenly. i8 is the universal fallback (legal on every target).
+ MVT ChunkVT = MVT::i8;
+ for (MVT MV : {MVT::i64, MVT::i32, MVT::i16}) {
+ unsigned MVBytes = MV.getSizeInBits() / 8;
+ if (TLI.isTypeLegal(MV) && MVBytes <= StorageBytes &&
+ StorageBytes % MVBytes == 0) {
+ ChunkVT = MV;
+ break;
+ }
+ }
+ unsigned ChunkBytes = ChunkVT.getSizeInBits() / 8;
+ unsigned NumChunks = StorageBytes / ChunkBytes;
+
+ MachineFunction &MF = DAG.getMachineFunction();
+ SDValue StackT = DAG.CreateStackTemporary(VT);
+ SDValue StackF = DAG.CreateStackTemporary(VT);
+ SDValue StackR = DAG.CreateStackTemporary(VT);
+ int FIT = cast<FrameIndexSDNode>(StackT.getNode())->getIndex();
+ int FIF = cast<FrameIndexSDNode>(StackF.getNode())->getIndex();
+ int FIR = cast<FrameIndexSDNode>(StackR.getNode())->getIndex();
+ MachinePointerInfo PIT = MachinePointerInfo::getFixedStack(MF, FIT);
+ MachinePointerInfo PIF = MachinePointerInfo::getFixedStack(MF, FIF);
+ MachinePointerInfo PIR = MachinePointerInfo::getFixedStack(MF, FIR);
+
+ SDValue Chain = DAG.getEntryNode();
+ Chain = DAG.getStore(Chain, dl, Tmp2, StackT, PIT);
+ Chain = DAG.getStore(Chain, dl, Tmp3, StackF, PIF);
+
+ for (unsigned i = 0; i < NumChunks; ++i) {
+ TypeSize Off = TypeSize::getFixed(i * ChunkBytes);
+ SDValue TPtr = DAG.getMemBasePlusOffset(StackT, Off, dl);
+ SDValue FPtr = DAG.getMemBasePlusOffset(StackF, Off, dl);
+ SDValue RPtr = DAG.getMemBasePlusOffset(StackR, Off, dl);
+
+ SDValue Ti = DAG.getLoad(ChunkVT, dl, Chain, TPtr,
+ PIT.getWithOffset(i * ChunkBytes));
+ Chain = Ti.getValue(1);
+ SDValue Fi = DAG.getLoad(ChunkVT, dl, Chain, FPtr,
+ PIF.getWithOffset(i * ChunkBytes));
+ Chain = Fi.getValue(1);
+
+ // Blend this chunk via CT_SELECT on a legal integer type. The
+ // recursive node will be Expand'd by the scalar-int branch below.
+ SDValue Ri =
+ DAG.getCTSelect(dl, ChunkVT, Tmp1, Ti, Fi, Node->getFlags());
+
+ Chain = DAG.getStore(Chain, dl, Ri, RPtr,
+ PIR.getWithOffset(i * ChunkBytes));
+ }
+
+ Tmp1 = DAG.getLoad(VT, dl, Chain, StackR, PIR);
+ Tmp1->setFlags(Node->getFlags());
+ Results.push_back(Tmp1);
+ break;
+ }
+
+ SDValue WorkingT = Tmp2;
+ SDValue WorkingF = Tmp3;
+ EVT WorkingVT = VT;
+
+ bool IsFP = VT.isVector() ? VT.getVectorElementType().isFloatingPoint()
+ : VT.isFloatingPoint();
+ if (IsFP) {
+ if (VT.isVector())
+ WorkingVT = EVT::getVectorVT(
*DAG.getContext(),
EVT::getIntegerVT(*DAG.getContext(), VT.getScalarSizeInBits()),
VT.getVectorElementCount());
- T = DAG.getNode(ISD::BITCAST, dl, IntVT, T);
- F = DAG.getNode(ISD::BITCAST, dl, IntVT, F);
- }
+ else
+ WorkingVT = VT.changeTypeToInteger();
+ WorkingT = DAG.getBitcast(WorkingVT, Tmp2);
+ WorkingF = DAG.getBitcast(WorkingVT, Tmp3);
+ }
- // Step 2: Broadcast the i1 condition to a vector of i1s
- // Creates [cond, cond, cond, ...] with i1 elements
- EVT VecI1Ty = EVT::getVectorVT(*DAG.getContext(), MVT::i1,
- VT.getVectorNumElements());
- SDValue VecCond = DAG.getSplatBuildVector(VecI1Ty, dl, Tmp1);
-
- // Step 3: Sign-extend i1 vector to get all-bits mask
- // true (i1=1) -> 0xFFFFFFFF..., false (i1=0) -> 0x00000000
- // Sign extension is constant-time: pure arithmetic, no branches
- SDValue Mask = DAG.getNode(ISD::SIGN_EXTEND, dl, IntVT, VecCond);
-
- // Step 4: Compute constant-time blend: F ^ ((T ^ F) & Mask)
- // All operations (XOR, AND) execute in constant time
- SDValue TXorF = DAG.getNode(ISD::XOR, dl, IntVT, T, F);
- SDValue MaskedDiff = DAG.getNode(ISD::AND, dl, IntVT, TXorF, Mask);
- Tmp1 = DAG.getNode(ISD::XOR, dl, IntVT, F, MaskedDiff);
-
- // Step 5: Bitcast back to original floating-point type if needed
- if (VT.isFloatingPoint()) {
- Tmp1 = DAG.getNode(ISD::BITCAST, dl, VT, Tmp1);
+ // Compute the all-ones/all-zeros mask as a scalar, then splat for vectors.
+ EVT MaskEltVT =
+ WorkingVT.isVector() ? WorkingVT.getVectorElementType() : WorkingVT;
+ SDValue ScalarCond = Tmp1;
+ if (ScalarCond.getValueType() != MaskEltVT)
+ ScalarCond = DAG.getAnyExtOrTrunc(ScalarCond, dl, MaskEltVT);
+ SDValue ScalarMask =
+ DAG.getNode(ISD::SUB, dl, MaskEltVT, DAG.getConstant(0, dl, MaskEltVT),
+ DAG.getNode(ISD::AND, dl, MaskEltVT, ScalarCond,
+ DAG.getConstant(1, dl, MaskEltVT)));
+ SDValue Mask = WorkingVT.isVector()
+ ? DAG.getSplat(WorkingVT, dl, ScalarMask)
+ : ScalarMask;
+
+ // F ^ ((T ^ F) & Mask)
+ SDValue XorTF = DAG.getNode(ISD::XOR, dl, WorkingVT, WorkingT, WorkingF);
+ SDValue TM = DAG.getNode(ISD::AND, dl, WorkingVT, XorTF, Mask);
+
+ // Forward-looking DAGCombine barrier (see header): route the masked-diff
+ // through a vreg so the chain edge partitions it from the surrounding
+ // bitwise ops, guarding against a future combiner that might fold
+ // XOR/AND/XOR-with-sext-mask back to SELECT. Skipped where no register
+ // class is available (scalable vectors, non-simple or illegal types, or
+ // a legal type the target has no register class for).
+ if (WorkingVT.isSimple() && !WorkingVT.isScalableVector() &&
+ TLI.isTypeLegal(WorkingVT.getSimpleVT())) {
+ if (const TargetRegisterClass *RC =
+ TLI.getRegClassFor(WorkingVT.getSimpleVT())) {
+ MachineRegisterInfo &MRI = DAG.getMachineFunction().getRegInfo();
+ Register TMReg = MRI.createVirtualRegister(RC);
+ SDValue Chain = DAG.getEntryNode();
+ Chain = DAG.getCopyToReg(Chain, dl, TMReg, TM);
+ TM = DAG.getCopyFromReg(Chain, dl, TMReg, WorkingVT);
}
-
- Tmp1->setFlags(Node->getFlags());
- } else if (VT.isFloatingPoint()) {
- EVT IntegerVT = EVT::getIntegerVT(*DAG.getContext(), VT.getSizeInBits());
- Tmp2 = DAG.getBitcast(IntegerVT, Tmp2);
- Tmp3 = DAG.getBitcast(IntegerVT, Tmp3);
- Tmp1 = DAG.getBitcast(VT, DAG.getCTSelect(dl, IntegerVT, Tmp1, Tmp2, Tmp3,
- Node->getFlags()));
- } else {
- assert(VT.isInteger());
- // Expand: Result = F ^ ((T ^ F) & Mask), Mask = 0 - (Cond & 1).
- // SUB+AND creates the mask because i1 is already type-promoted;
- // SIGN_EXTEND(i32, i32) would be a no-op leaving mask as 0/1.
- SDValue Cond = Tmp1;
- if (Cond.getValueType() != VT)
- Cond = DAG.getNode(ISD::ANY_EXTEND, dl, VT, Cond);
- SDValue Mask = DAG.getNode(
- ISD::SUB, dl, VT, DAG.getConstant(0, dl, VT),
- DAG.getNode(ISD::AND, dl, VT, Cond, DAG.getConstant(1, dl, VT)));
- SDValue Diff = DAG.getNode(ISD::XOR, dl, VT, Tmp2, Tmp3);
- Tmp1 = DAG.getNode(ISD::XOR, dl, VT, Tmp3,
- DAG.getNode(ISD::AND, dl, VT, Diff, Mask));
- Tmp1->setFlags(Node->getFlags());
}
+
+ Tmp1 = DAG.getNode(ISD::XOR, dl, WorkingVT, WorkingF, TM);
+
+ if (WorkingVT != VT)
+ Tmp1 = DAG.getBitcast(VT, Tmp1);
+
+ Tmp1->setFlags(Node->getFlags());
Results.push_back(Tmp1);
break;
}
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeFloatTypes.cpp b/llvm/lib/CodeGen/SelectionDAG/LegalizeFloatTypes.cpp
index 17a837bb42770..04257589c3e4d 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeFloatTypes.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeFloatTypes.cpp
@@ -971,7 +971,7 @@ SDValue DAGTypeLegalizer::SoftenFloatRes_CT_SELECT(SDNode *N) {
SDValue LHS = GetSoftenedFloat(N->getOperand(1));
SDValue RHS = GetSoftenedFloat(N->getOperand(2));
return DAG.getCTSelect(SDLoc(N), LHS.getValueType(), N->getOperand(0), LHS,
- RHS);
+ RHS, N->getFlags());
}
SDValue DAGTypeLegalizer::SoftenFloatRes_SELECT_CC(SDNode *N) {
@@ -2866,7 +2866,7 @@ SDValue DAGTypeLegalizer::SoftPromoteHalfRes_CT_SELECT(SDNode *N) {
SDValue Op1 = GetSoftPromotedHalf(N->getOperand(1));
SDValue Op2 = GetSoftPromotedHalf(N->getOperand(2));
return DAG.getCTSelect(SDLoc(N), Op1.getValueType(), N->getOperand(0), Op1,
- Op2);
+ Op2, N->getFlags());
}
SDValue DAGTypeLegalizer::SoftPromoteHalfRes_SELECT_CC(SDNode *N) {
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp b/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp
index 6d25c26cbe914..872b1c558aa6c 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp
@@ -768,7 +768,7 @@ SDValue DAGTypeLegalizer::ScalarizeVecRes_SELECT(SDNode *N) {
SDValue DAGTypeLegalizer::ScalarizeVecRes_CT_SELECT(SDNode *N) {
SDValue LHS = GetScalarizedVector(N->getOperand(1));
return DAG.getCTSelect(SDLoc(N), LHS.getValueType(), N->getOperand(0), LHS,
- GetScalarizedVector(N->getOperand(2)));
+ GetScalarizedVector(N->getOperand(2)), N->getFlags());
}
SDValue DAGTypeLegalizer::ScalarizeVecRes_SELECT_CC(SDNode *N) {
diff --git a/llvm/lib/CodeGen/SelectionDAG/SelectionDAGBuilder.cpp b/llvm/lib/CodeGen/SelectionDAG/SelectionDAGBuilder.cpp
index aa28b25dbeb51..74f514d789125 100644
--- a/llvm/lib/CodeGen/SelectionDAG/SelectionDAGBuilder.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/SelectionDAGBuilder.cpp
@@ -6704,82 +6704,6 @@ void SelectionDAGBuilder::visitVectorExtractLastActive(const CallInst &I,
setValue(&I, Result);
}
-/// Fallback implementation for constant-time select using DAG chaining.
-/// This implementation uses data dependencies through virtual registers to
-/// prevent optimizations from breaking the constant-time property. It is a
-/// best-effort safeguard; for stronger guarantees we prefer target-specific
-/// lowering pipelines that preserve the select pattern by construction.
-///
-/// It handles scalars, vectors (fixed and scalable), and floating-point types.
-SDValue SelectionDAGBuilder::createProtectedCtSelectFallback(
- SelectionDAG &DAG, const SDLoc &DL, SDValue Cond, SDValue T, SDValue F,
- EVT VT) {
-
- SDValue WorkingT = T;
- SDValue WorkingF = F;
- EVT WorkingVT = VT;
-
- SDValue Chain = DAG.getEntryNode();
- MachineRegisterInfo &MRI = DAG.getMachineFunction().getRegInfo();
-
- // Handle vector condition: splat scalar condition to vector
- if (VT.isVector() && !Cond.getValueType().isVector()) {
- ElementCount NumElems = VT.getVectorElementCount();
- EVT CondVT = EVT::getVectorVT(*DAG.getContext(), MVT::i1, NumElems);
- Cond = DAG.getSplat(CondVT, DL, Cond);
- }
-
- // Handle floating-point types: bitcast to integer for bitwise operations
- if (VT.isFloatingPoint()) {
- WorkingVT = VT.changeTypeToInteger();
- WorkingT = DAG.getBitcast(WorkingVT, T);
- WorkingF = DAG.getBitcast(WorkingVT, F);
- }
-
- // Create mask: sign-extend condition to all bits
- SDValue Mask = DAG.getSExtOrTrunc(Cond, DL, WorkingVT);
-
- // Compute: F ^ ((T ^ F) & Mask)
- // This is constant-time because both branches are always computed
- SDValue XorTF = DAG.getNode(ISD::XOR, DL, WorkingVT, WorkingT, WorkingF);
- SDValue TM = DAG.getNode(ISD::AND, DL, WorkingVT, XorTF, Mask);
-
- // DAG chaining: create data dependency through virtual register
- // This prevents optimizations from reordering or eliminating operations
- const TargetLowering &TLI = DAG.getTargetLoweringInfo();
- bool CanUseChaining = false;
-
- if (!WorkingVT.isScalableVector()) {
- // For fixed-size vectors and scalars, chaining is a best-effort hardening
- // step. The CT guarantee comes from the dataflow-only select
- // pattern (both sides computed, no control-flow). Chaining only adds an
- // extra dependency to discourage later combines.
- CanUseChaining = TLI.isTypeLegal(WorkingVT.getSimpleVT());
- } else {
- // For scalable vectors, skip chaining because there is no stable register
- // class to copy through. CT behavior still relies on the masking/select
- // pattern above.
- CanUseChaining = false;
- }
-
- if (CanUseChaining) {
- // Apply chaining through registers for additional protection
- const TargetRegisterClass *RC = TLI.getRegClassFor(WorkingVT.getSimpleVT());
- Register TMReg = MRI.createVirtualRegister(RC);
- Chain = DAG.getCopyToReg(Chain, DL, TMReg, TM);
- TM = DAG.getCopyFromReg(Chain, DL, TMReg, WorkingVT);
- }
-
- SDValue Result = DAG.getNode(ISD::XOR, DL, WorkingVT, WorkingF, TM);
-
- // Convert back to original type if needed
- if (WorkingVT != VT) {
- Result = DAG.getBitcast(VT, Result);
- }
-
- return Result;
-}
-
/// Lower the call to the specified intrinsic function.
void SelectionDAGBuilder::visitIntrinsicCall(const CallInst &I,
unsigned Intrinsic) {
@@ -6982,44 +6906,21 @@ void SelectionDAGBuilder::visitIntrinsicCall(const CallInst &I,
return;
}
case Intrinsic::ct_select: {
- // Set function attribute to indicate ct.select usage
- Function &F = DAG.getMachineFunction().getFunction();
- F.addFnAttr("ct-select");
-
- SDLoc DL = getCurSDLoc();
-
- SDValue Cond = getValue(I.getArgOperand(0)); // i1
- SDValue A = getValue(I.getArgOperand(1)); // T
- SDValue B = getValue(I.getArgOperand(2)); // T
-
- assert((A.getValueType() == B.getValueType()) &&
+ SDValue Cond = getValue(I.getArgOperand(0));
+ SDValue A = getValue(I.getArgOperand(1));
+ SDValue B = getValue(I.getArgOperand(2));
+ assert(A.getValueType() == B.getValueType() &&
"Operands are of different types");
+ assert(!Cond.getValueType().isVector() && "Vector condition not supported");
- EVT VT = A.getValueType();
- EVT CondVT = Cond.getValueType();
-
- // assert if Cond type is Vector
- assert(!CondVT.isVector() && "Vector type cond not supported yet");
-
- // Create a CT_SELECT node for scalar types so it survives DAGCombiner
- // (visitCT_SELECT is conservative) and expands to AND/OR/XOR during
- // operation legalization, after SETCC is lowered. Unsupported vectors
- // and floats with illegal integer equivalents (e.g. f64 on i386) use
- // the inline fallback which runs before type legalization.
- bool CreateNode =
- TLI.isOperationLegalOrCustom(ISD::CT_SELECT, VT) ||
- (!VT.isVector() &&
- (!VT.isFloatingPoint() ||
- TLI.isTypeLegal(
- EVT::getIntegerVT(*DAG.getContext(), VT.getSizeInBits()))));
-
- if (CreateNode) {
- SDValue Result = DAG.getNode(ISD::CT_SELECT, DL, VT, Cond, A, B);
- setValue(&I, Result);
- return;
- }
+ // TODO: a follow-up target-specific bundling pass needs to know the
+ // function uses ct.select. When that consumer lands, record it in the
+ // target's MachineFunctionInfo instead of mutating the IR from codegen.
+ // Kept commented for reference until then:
+ // DAG.getMachineFunction().getFunction().addFnAttr("ct-select");
- setValue(&I, createProtectedCtSelectFallback(DAG, DL, Cond, A, B, VT));
+ setValue(&I, DAG.getNode(ISD::CT_SELECT, getCurSDLoc(), A.getValueType(),
+ Cond, A, B));
return;
}
case Intrinsic::call_preallocated_setup: {
diff --git a/llvm/lib/CodeGen/SelectionDAG/SelectionDAGBuilder.h b/llvm/lib/CodeGen/SelectionDAG/SelectionDAGBuilder.h
index 45efe4c81a9e0..7a925c45c3a13 100644
--- a/llvm/lib/CodeGen/SelectionDAG/SelectionDAGBuilder.h
+++ b/llvm/lib/CodeGen/SelectionDAG/SelectionDAGBuilder.h
@@ -219,9 +219,6 @@ class SelectionDAGBuilder {
peelDominantCaseCluster(const SwitchInst &SI,
SwitchCG::CaseClusterVector &Clusters,
BranchProbability &PeeledCaseProb);
- SDValue createProtectedCtSelectFallback(SelectionDAG &DAG, const SDLoc &DL,
- SDValue Cond, SDValue T, SDValue F,
- EVT VT);
private:
const TargetMachine &TM;
diff --git a/llvm/test/CodeGen/RISCV/ctselect-fallback.ll b/llvm/test/CodeGen/RISCV/ctselect-fallback.ll
index ee8072703ee31..0df231b7a98d9 100644
--- a/llvm/test/CodeGen/RISCV/ctselect-fallback.ll
+++ b/llvm/test/CodeGen/RISCV/ctselect-fallback.ll
@@ -97,10 +97,14 @@ define ptr @test_ctselect_ptr(i1 %cond, ptr %a, ptr %b) {
define i32 @test_ctselect_const_true(i32 %a, i32 %b) {
; RV64-LABEL: test_ctselect_const_true:
; RV64: # %bb.0:
+; RV64-NEXT: xor a0, a0, a1
+; RV64-NEXT: xor a0, a1, a0
; RV64-NEXT: ret
;
; RV32-LABEL: test_ctselect_const_true:
; RV32: # %bb.0:
+; RV32-NEXT: xor a0, a0, a1
+; RV32-NEXT: xor a0, a1, a0
; RV32-NEXT: ret
%result = call i32 @llvm.ct.select.i32(i1 true, i32 %a, i32 %b)
ret i32 %result
@@ -207,6 +211,7 @@ define i32 @test_ctselect_nested_and_i1_to_i32(i1 %c0, i1 %c1, i32 %x, i32 %y) {
; RV64-LABEL: test_ctselect_nested_and_i1_to_i32:
; RV64: # %bb.0:
; RV64-NEXT: and a0, a0, a1
+; RV64-NEXT: andi a0, a0, 1
; RV64-NEXT: xor a2, a2, a3
; RV64-NEXT: slli a0, a0, 63
; RV64-NEXT: srai a0, a0, 63
@@ -217,6 +222,7 @@ define i32 @test_ctselect_nested_and_i1_to_i32(i1 %c0, i1 %c1, i32 %x, i32 %y) {
; RV32-LABEL: test_ctselect_nested_and_i1_to_i32:
; RV32: # %bb.0:
; RV32-NEXT: and a0, a0, a1
+; RV32-NEXT: andi a0, a0, 1
; RV32-NEXT: xor a2, a2, a3
; RV32-NEXT: slli a0, a0, 31
; RV32-NEXT: srai a0, a0, 31
@@ -235,6 +241,7 @@ define i32 @test_ctselect_nested_or_i1_to_i32(i1 %c0, i1 %c1, i32 %x, i32 %y) {
; RV64-LABEL: test_ctselect_nested_or_i1_to_i32:
; RV64: # %bb.0:
; RV64-NEXT: or a0, a0, a1
+; RV64-NEXT: andi a0, a0, 1
; RV64-NEXT: xor a2, a2, a3
; RV64-NEXT: slli a0, a0, 63
; RV64-NEXT: srai a0, a0, 63
@@ -245,6 +252,7 @@ define i32 @test_ctselect_nested_or_i1_to_i32(i1 %c0, i1 %c1, i32 %x, i32 %y) {
; RV32-LABEL: test_ctselect_nested_or_i1_to_i32:
; RV32: # %bb.0:
; RV32-NEXT: or a0, a0, a1
+; RV32-NEXT: andi a0, a0, 1
; RV32-NEXT: xor a2, a2, a3
; RV32-NEXT: slli a0, a0, 31
; RV32-NEXT: srai a0, a0, 31
@@ -265,6 +273,7 @@ define i32 @test_ctselect_double_nested_and_i1(i1 %c0, i1 %c1, i1 %c2, i32 %x, i
; RV64: # %bb.0:
; RV64-NEXT: and a0, a0, a1
; RV64-NEXT: and a0, a0, a2
+; RV64-NEXT: andi a0, a0, 1
; RV64-NEXT: xor a3, a3, a4
; RV64-NEXT: slli a0, a0, 63
; RV64-NEXT: srai a0, a0, 63
@@ -276,6 +285,7 @@ define i32 @test_ctselect_double_nested_and_i1(i1 %c0, i1 %c1, i1 %c2, i32 %x, i
; RV32: # %bb.0:
; RV32-NEXT: and a0, a0, a1
; RV32-NEXT: and a0, a0, a2
+; RV32-NEXT: andi a0, a0, 1
; RV32-NEXT: xor a3, a3, a4
; RV32-NEXT: slli a0, a0, 31
; RV32-NEXT: srai a0, a0, 31
@@ -295,7 +305,9 @@ define i32 @test_ctselect_double_nested_mixed_i1(i1 %c0, i1 %c1, i1 %c2, i32 %x,
; RV64: # %bb.0:
; RV64-NEXT: and a0, a0, a1
; RV64-NEXT: xor a3, a3, a4
+; RV64-NEXT: andi a0, a0, 1
; RV64-NEXT: or a0, a0, a2
+; RV64-NEXT: andi a0, a0, 1
; RV64-NEXT: slli a0, a0, 63
; RV64-NEXT: srai a0, a0, 63
; RV64-NEXT: and a3, a3, a0
@@ -309,7 +321,9 @@ define i32 @test_ctselect_double_nested_mixed_i1(i1 %c0, i1 %c1, i1 %c2, i32 %x,
; RV32: # %bb.0:
; RV32-NEXT: and a0, a0, a1
; RV32-NEXT: xor a3, a3, a4
+; RV32-NEXT: andi a0, a0, 1
; RV32-NEXT: or a0, a0, a2
+; RV32-NEXT: andi a0, a0, 1
; RV32-NEXT: slli a0, a0, 31
; RV32-NEXT: srai a0, a0, 31
; RV32-NEXT: and a3, a3, a0
@@ -370,7 +384,7 @@ define float @test_ctselect_f32_nan_inf(i1 %cond) {
; RV64-NEXT: srai a0, a0, 63
; RV64-NEXT: and a0, a0, a1
; RV64-NEXT: lui a1, 522240
-; RV64-NEXT: or a0, a0, a1
+; RV64-NEXT: xor a0, a0, a1
; RV64-NEXT: ret
;
; RV32-LABEL: test_ctselect_f32_nan_inf:
@@ -380,7 +394,7 @@ define float @test_ctselect_f32_nan_inf(i1 %cond) {
; RV32-NEXT: srai a0, a0, 31
; RV32-NEXT: and a0, a0, a1
; RV32-NEXT: lui a1, 522240
-; RV32-NEXT: or a0, a0, a1
+; RV32-NEXT: xor a0, a0, a1
; RV32-NEXT: ret
%result = call float @llvm.ct.select.f32(i1 %cond, float 0x7FF8000000000000, float 0x7FF0000000000000)
ret float %result
@@ -396,7 +410,7 @@ define double @test_ctselect_f64_nan_inf(i1 %cond) {
; RV64-NEXT: and a0, a0, a1
; RV64-NEXT: li a1, 2047
; RV64-NEXT: slli a1, a1, 52
-; RV64-NEXT: or a0, a0, a1
+; RV64-NEXT: xor a0, a0, a1
; RV64-NEXT: ret
;
; RV32-LABEL: test_ctselect_f64_nan_inf:
@@ -406,7 +420,7 @@ define double @test_ctselect_f64_nan_inf(i1 %cond) {
; RV32-NEXT: srai a0, a0, 31
; RV32-NEXT: and a0, a0, a1
; RV32-NEXT: lui a1, 524032
-; RV32-NEXT: or a1, a0, a1
+; RV32-NEXT: xor a1, a0, a1
; RV32-NEXT: li a0, 0
; RV32-NEXT: ret
%result = call double @llvm.ct.select.f64(i1 %cond, double 0x7FF8000000000000, double 0x7FF0000000000000)
@@ -599,10 +613,10 @@ define <8 x i32> @test_ctselect_v8i32(i1 %cond, <8 x i32> %a, <8 x i32> %b) {
; RV64-NEXT: .cfi_offset s0, -8
; RV64-NEXT: .cfi_offset s1, -16
; RV64-NEXT: .cfi_offset s2, -24
-; RV64-NEXT: lw a7, 32(a3)
-; RV64-NEXT: lw a6, 40(a3)
-; RV64-NEXT: lw a5, 48(a3)
-; RV64-NEXT: lw a4, 56(a3)
+; RV64-NEXT: lw a4, 32(a3)
+; RV64-NEXT: lw a5, 40(a3)
+; RV64-NEXT: lw a6, 48(a3)
+; RV64-NEXT: lw a7, 56(a3)
; RV64-NEXT: lw t0, 32(a2)
; RV64-NEXT: lw t1, 40(a2)
; RV64-NEXT: lw t2, 48(a2)
@@ -621,10 +635,10 @@ define <8 x i32> @test_ctselect_v8i32(i1 %cond, <8 x i32> %a, <8 x i32> %b) {
; RV64-NEXT: xor s1, s1, t5
; RV64-NEXT: xor s2, s2, t6
; RV64-NEXT: xor a2, a2, a3
-; RV64-NEXT: xor t0, t0, a7
-; RV64-NEXT: xor t1, t1, a6
-; RV64-NEXT: xor t2, t2, a5
-; RV64-NEXT: xor t3, t3, a4
+; RV64-NEXT: xor t0, t0, a4
+; RV64-NEXT: xor t1, t1, a5
+; RV64-NEXT: xor t2, t2, a6
+; RV64-NEXT: xor t3, t3, a7
; RV64-NEXT: and s0, s0, a1
; RV64-NEXT: and s1, s1, a1
; RV64-NEXT: and s2, s2, a1
@@ -637,12 +651,12 @@ define <8 x i32> @test_ctselect_v8i32(i1 %cond, <8 x i32> %a, <8 x i32> %b) {
; RV64-NEXT: xor t4, t5, s1
; RV64-NEXT: xor t5, t6, s2
; RV64-NEXT: xor a2, a3, a2
-; RV64-NEXT: xor a3, a7, t0
-; RV64-NEXT: xor a6, a6, t1
-; RV64-NEXT: xor a5, a5, t2
-; RV64-NEXT: xor a1, a4, a1
+; RV64-NEXT: xor a3, a4, t0
+; RV64-NEXT: xor a4, a5, t1
+; RV64-NEXT: xor a5, a6, t2
+; RV64-NEXT: xor a1, a7, a1
; RV64-NEXT: sw a3, 16(a0)
-; RV64-NEXT: sw a6, 20(a0)
+; RV64-NEXT: sw a4, 20(a0)
; RV64-NEXT: sw a5, 24(a0)
; RV64-NEXT: sw a1, 28(a0)
; RV64-NEXT: sw t3, 0(a0)
@@ -669,10 +683,10 @@ define <8 x i32> @test_ctselect_v8i32(i1 %cond, <8 x i32> %a, <8 x i32> %b) {
; RV32-NEXT: .cfi_offset s0, -4
; RV32-NEXT: .cfi_offset s1, -8
; RV32-NEXT: .cfi_offset s2, -12
-; RV32-NEXT: lw a7, 16(a3)
-; RV32-NEXT: lw a6, 20(a3)
-; RV32-NEXT: lw a5, 24(a3)
-; RV32-NEXT: lw a4, 28(a3)
+; RV32-NEXT: lw a4, 16(a3)
+; RV32-NEXT: lw a5, 20(a3)
+; RV32-NEXT: lw a6, 24(a3)
+; RV32-NEXT: lw a7, 28(a3)
; RV32-NEXT: lw t0, 16(a2)
; RV32-NEXT: lw t1, 20(a2)
; RV32-NEXT: lw t2, 24(a2)
@@ -691,10 +705,10 @@ define <8 x i32> @test_ctselect_v8i32(i1 %cond, <8 x i32> %a, <8 x i32> %b) {
; RV32-NEXT: xor s1, s1, t5
; RV32-NEXT: xor s2, s2, t6
; RV32-NEXT: xor a2, a2, a3
-; RV32-NEXT: xor t0, t0, a7
-; RV32-NEXT: xor t1, t1, a6
-; RV32-NEXT: xor t2, t2, a5
-; RV32-NEXT: xor t3, t3, a4
+; RV32-NEXT: xor t0, t0, a4
+; RV32-NEXT: xor t1, t1, a5
+; RV32-NEXT: xor t2, t2, a6
+; RV32-NEXT: xor t3, t3, a7
; RV32-NEXT: and s0, s0, a1
; RV32-NEXT: and s1, s1, a1
; RV32-NEXT: and s2, s2, a1
@@ -707,12 +721,12 @@ define <8 x i32> @test_ctselect_v8i32(i1 %cond, <8 x i32> %a, <8 x i32> %b) {
; RV32-NEXT: xor t4, t5, s1
; RV32-NEXT: xor t5, t6, s2
; RV32-NEXT: xor a2, a3, a2
-; RV32-NEXT: xor a3, a7, t0
-; RV32-NEXT: xor a6, a6, t1
-; RV32-NEXT: xor a5, a5, t2
-; RV32-NEXT: xor a1, a4, a1
+; RV32-NEXT: xor a3, a4, t0
+; RV32-NEXT: xor a4, a5, t1
+; RV32-NEXT: xor a5, a6, t2
+; RV32-NEXT: xor a1, a7, a1
; RV32-NEXT: sw a3, 16(a0)
-; RV32-NEXT: sw a6, 20(a0)
+; RV32-NEXT: sw a4, 20(a0)
; RV32-NEXT: sw a5, 24(a0)
; RV32-NEXT: sw a1, 28(a0)
; RV32-NEXT: sw t3, 0(a0)
diff --git a/llvm/test/CodeGen/X86/ctselect.ll b/llvm/test/CodeGen/X86/ctselect.ll
index e1abae80cef4f..9d54c29db9683 100644
--- a/llvm/test/CodeGen/X86/ctselect.ll
+++ b/llvm/test/CodeGen/X86/ctselect.ll
@@ -5,7 +5,7 @@
; Test basic ct.select functionality for scalar types
-define i8 @test_ctselect_i8(i1 %cond, i8 %a, i8 %b) {
+define i8 @test_ctselect_i8(i1 %cond, i8 %a, i8 %b) #0 {
; X64-LABEL: test_ctselect_i8:
; X64: # %bb.0:
; X64-NEXT: movl %edi, %eax
@@ -44,7 +44,7 @@ define i8 @test_ctselect_i8(i1 %cond, i8 %a, i8 %b) {
ret i8 %result
}
-define i32 @test_ctselect_i32(i1 %cond, i32 %a, i32 %b) {
+define i32 @test_ctselect_i32(i1 %cond, i32 %a, i32 %b) #0 {
; X64-LABEL: test_ctselect_i32:
; X64: # %bb.0:
; X64-NEXT: movl %edi, %eax
@@ -84,7 +84,7 @@ define i32 @test_ctselect_i32(i1 %cond, i32 %a, i32 %b) {
ret i32 %result
}
-define i64 @test_ctselect_i64(i1 %cond, i64 %a, i64 %b) {
+define i64 @test_ctselect_i64(i1 %cond, i64 %a, i64 %b) #0 {
; X64-LABEL: test_ctselect_i64:
; X64: # %bb.0:
; X64-NEXT: movl %edi, %eax
@@ -98,11 +98,7 @@ define i64 @test_ctselect_i64(i1 %cond, i64 %a, i64 %b) {
; X32-LABEL: test_ctselect_i64:
; X32: # %bb.0:
; X32-NEXT: pushl %edi
-; X32-NEXT: .cfi_def_cfa_offset 8
; X32-NEXT: pushl %esi
-; X32-NEXT: .cfi_def_cfa_offset 12
-; X32-NEXT: .cfi_offset %esi, -12
-; X32-NEXT: .cfi_offset %edi, -8
; X32-NEXT: movzbl {{[0-9]+}}(%esp), %edx
; X32-NEXT: andb $1, %dl
; X32-NEXT: movl {{[0-9]+}}(%esp), %esi
@@ -118,19 +114,13 @@ define i64 @test_ctselect_i64(i1 %cond, i64 %a, i64 %b) {
; X32-NEXT: andl %edi, %edx
; X32-NEXT: xorl %ecx, %edx
; X32-NEXT: popl %esi
-; X32-NEXT: .cfi_def_cfa_offset 8
; X32-NEXT: popl %edi
-; X32-NEXT: .cfi_def_cfa_offset 4
; X32-NEXT: retl
;
; X32-NOCMOV-LABEL: test_ctselect_i64:
; X32-NOCMOV: # %bb.0:
; X32-NOCMOV-NEXT: pushl %edi
-; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 8
; X32-NOCMOV-NEXT: pushl %esi
-; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 12
-; X32-NOCMOV-NEXT: .cfi_offset %esi, -12
-; X32-NOCMOV-NEXT: .cfi_offset %edi, -8
; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %edx
; X32-NOCMOV-NEXT: andb $1, %dl
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %esi
@@ -146,15 +136,13 @@ define i64 @test_ctselect_i64(i1 %cond, i64 %a, i64 %b) {
; X32-NOCMOV-NEXT: andl %edi, %edx
; X32-NOCMOV-NEXT: xorl %ecx, %edx
; X32-NOCMOV-NEXT: popl %esi
-; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 8
; X32-NOCMOV-NEXT: popl %edi
-; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 4
; X32-NOCMOV-NEXT: retl
%result = call i64 @llvm.ct.select.i64(i1 %cond, i64 %a, i64 %b)
ret i64 %result
}
-define float @test_ctselect_f32(i1 %cond, float %a, float %b) {
+define float @test_ctselect_f32(i1 %cond, float %a, float %b) #0 {
; X64-LABEL: test_ctselect_f32:
; X64: # %bb.0:
; X64-NEXT: movd %xmm1, %eax
@@ -170,7 +158,6 @@ define float @test_ctselect_f32(i1 %cond, float %a, float %b) {
; X32-LABEL: test_ctselect_f32:
; X32: # %bb.0:
; X32-NEXT: subl $12, %esp
-; X32-NEXT: .cfi_def_cfa_offset 16
; X32-NEXT: flds {{[0-9]+}}(%esp)
; X32-NEXT: fstps {{[0-9]+}}(%esp)
; X32-NEXT: flds {{[0-9]+}}(%esp)
@@ -187,13 +174,11 @@ define float @test_ctselect_f32(i1 %cond, float %a, float %b) {
; X32-NEXT: movl %edx, {{[0-9]+}}(%esp)
; X32-NEXT: flds {{[0-9]+}}(%esp)
; X32-NEXT: addl $12, %esp
-; X32-NEXT: .cfi_def_cfa_offset 4
; X32-NEXT: retl
;
; X32-NOCMOV-LABEL: test_ctselect_f32:
; X32-NOCMOV: # %bb.0:
; X32-NOCMOV-NEXT: subl $12, %esp
-; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 16
; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
; X32-NOCMOV-NEXT: fstps {{[0-9]+}}(%esp)
; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
@@ -210,13 +195,12 @@ define float @test_ctselect_f32(i1 %cond, float %a, float %b) {
; X32-NOCMOV-NEXT: movl %edx, {{[0-9]+}}(%esp)
; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
; X32-NOCMOV-NEXT: addl $12, %esp
-; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 4
; X32-NOCMOV-NEXT: retl
%result = call float @llvm.ct.select.f32(i1 %cond, float %a, float %b)
ret float %result
}
-define double @test_ctselect_f64(i1 %cond, double %a, double %b) {
+define double @test_ctselect_f64(i1 %cond, double %a, double %b) #0 {
; X64-LABEL: test_ctselect_f64:
; X64: # %bb.0:
; X64-NEXT: # kill: def $edi killed $edi def $rdi
@@ -233,65 +217,426 @@ define double @test_ctselect_f64(i1 %cond, double %a, double %b) {
; X32-LABEL: test_ctselect_f64:
; X32: # %bb.0:
; X32-NEXT: pushl %esi
-; X32-NEXT: .cfi_def_cfa_offset 8
-; X32-NEXT: subl $8, %esp
-; X32-NEXT: .cfi_def_cfa_offset 16
-; X32-NEXT: .cfi_offset %esi, -8
-; X32-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
-; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: subl $24, %esp
+; X32-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: andb $1, %al
+; X32-NEXT: fldl {{[0-9]+}}(%esp)
+; X32-NEXT: fldl {{[0-9]+}}(%esp)
+; X32-NEXT: fstpl {{[0-9]+}}(%esp)
+; X32-NEXT: fstpl {{[0-9]+}}(%esp)
+; X32-NEXT: movzbl %al, %eax
+; X32-NEXT: negl %eax
; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
; X32-NEXT: movl {{[0-9]+}}(%esp), %esi
; X32-NEXT: xorl %edx, %esi
-; X32-NEXT: andl $1, %ecx
-; X32-NEXT: negl %ecx
-; X32-NEXT: andl %ecx, %esi
+; X32-NEXT: andl %eax, %esi
; X32-NEXT: xorl %edx, %esi
-; X32-NEXT: movl %esi, {{[0-9]+}}(%esp)
+; X32-NEXT: movl %esi, (%esp)
; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
-; X32-NEXT: xorl %eax, %edx
-; X32-NEXT: andl %ecx, %edx
-; X32-NEXT: xorl %eax, %edx
-; X32-NEXT: movl %edx, (%esp)
+; X32-NEXT: xorl %ecx, %edx
+; X32-NEXT: andl %eax, %edx
+; X32-NEXT: xorl %ecx, %edx
+; X32-NEXT: movl %edx, {{[0-9]+}}(%esp)
; X32-NEXT: fldl (%esp)
-; X32-NEXT: addl $8, %esp
-; X32-NEXT: .cfi_def_cfa_offset 8
+; X32-NEXT: addl $24, %esp
; X32-NEXT: popl %esi
-; X32-NEXT: .cfi_def_cfa_offset 4
; X32-NEXT: retl
;
; X32-NOCMOV-LABEL: test_ctselect_f64:
; X32-NOCMOV: # %bb.0:
; X32-NOCMOV-NEXT: pushl %esi
-; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 8
-; X32-NOCMOV-NEXT: subl $8, %esp
-; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 16
-; X32-NOCMOV-NEXT: .cfi_offset %esi, -8
-; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: subl $24, %esp
+; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: andb $1, %al
+; X32-NOCMOV-NEXT: fldl {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: fldl {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: fstpl {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: fstpl {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: movzbl %al, %eax
+; X32-NOCMOV-NEXT: negl %eax
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %esi
; X32-NOCMOV-NEXT: xorl %edx, %esi
-; X32-NOCMOV-NEXT: andl $1, %ecx
-; X32-NOCMOV-NEXT: negl %ecx
-; X32-NOCMOV-NEXT: andl %ecx, %esi
+; X32-NOCMOV-NEXT: andl %eax, %esi
; X32-NOCMOV-NEXT: xorl %edx, %esi
-; X32-NOCMOV-NEXT: movl %esi, {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: movl %esi, (%esp)
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
-; X32-NOCMOV-NEXT: xorl %eax, %edx
-; X32-NOCMOV-NEXT: andl %ecx, %edx
-; X32-NOCMOV-NEXT: xorl %eax, %edx
-; X32-NOCMOV-NEXT: movl %edx, (%esp)
+; X32-NOCMOV-NEXT: xorl %ecx, %edx
+; X32-NOCMOV-NEXT: andl %eax, %edx
+; X32-NOCMOV-NEXT: xorl %ecx, %edx
+; X32-NOCMOV-NEXT: movl %edx, {{[0-9]+}}(%esp)
; X32-NOCMOV-NEXT: fldl (%esp)
-; X32-NOCMOV-NEXT: addl $8, %esp
-; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 8
+; X32-NOCMOV-NEXT: addl $24, %esp
; X32-NOCMOV-NEXT: popl %esi
-; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 4
; X32-NOCMOV-NEXT: retl
%result = call double @llvm.ct.select.f64(i1 %cond, double %a, double %b)
ret double %result
}
-define ptr @test_ctselect_ptr(i1 %cond, ptr %a, ptr %b) {
+define half @test_ctselect_f16(i1 %cond, half %a, half %b) #0 {
+; X64-LABEL: test_ctselect_f16:
+; X64: # %bb.0:
+; X64-NEXT: pextrw $0, %xmm1, %eax
+; X64-NEXT: pextrw $0, %xmm0, %ecx
+; X64-NEXT: xorl %eax, %ecx
+; X64-NEXT: andl $1, %edi
+; X64-NEXT: negl %edi
+; X64-NEXT: andl %ecx, %edi
+; X64-NEXT: xorl %eax, %edi
+; X64-NEXT: pinsrw $0, %edi, %xmm0
+; X64-NEXT: retq
+;
+; X32-LABEL: test_ctselect_f16:
+; X32: # %bb.0:
+; X32-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: andb $1, %al
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: movzwl {{[0-9]+}}(%esp), %edx
+; X32-NEXT: xorw %cx, %dx
+; X32-NEXT: movzbl %al, %eax
+; X32-NEXT: negl %eax
+; X32-NEXT: andl %edx, %eax
+; X32-NEXT: xorl %ecx, %eax
+; X32-NEXT: # kill: def $ax killed $ax killed $eax
+; X32-NEXT: retl
+;
+; X32-NOCMOV-LABEL: test_ctselect_f16:
+; X32-NOCMOV: # %bb.0:
+; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: andb $1, %al
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: movzwl {{[0-9]+}}(%esp), %edx
+; X32-NOCMOV-NEXT: xorw %cx, %dx
+; X32-NOCMOV-NEXT: movzbl %al, %eax
+; X32-NOCMOV-NEXT: negl %eax
+; X32-NOCMOV-NEXT: andl %edx, %eax
+; X32-NOCMOV-NEXT: xorl %ecx, %eax
+; X32-NOCMOV-NEXT: # kill: def $ax killed $ax killed $eax
+; X32-NOCMOV-NEXT: retl
+ %result = call half @llvm.ct.select.f16(i1 %cond, half %a, half %b)
+ ret half %result
+}
+
+define bfloat @test_ctselect_bf16(i1 %cond, bfloat %a, bfloat %b) #0 {
+; X64-LABEL: test_ctselect_bf16:
+; X64: # %bb.0:
+; X64-NEXT: pextrw $0, %xmm1, %eax
+; X64-NEXT: pextrw $0, %xmm0, %ecx
+; X64-NEXT: xorl %eax, %ecx
+; X64-NEXT: andl $1, %edi
+; X64-NEXT: negl %edi
+; X64-NEXT: andl %ecx, %edi
+; X64-NEXT: xorl %eax, %edi
+; X64-NEXT: pinsrw $0, %edi, %xmm0
+; X64-NEXT: retq
+;
+; X32-LABEL: test_ctselect_bf16:
+; X32: # %bb.0:
+; X32-NEXT: pushl %esi
+; X32-NEXT: subl $8, %esp
+; X32-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NEXT: fstps (%esp)
+; X32-NEXT: calll __truncsfbf2
+; X32-NEXT: movl %eax, %esi
+; X32-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NEXT: fstps (%esp)
+; X32-NEXT: calll __truncsfbf2
+; X32-NEXT: # kill: def $ax killed $ax def $eax
+; X32-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: andb $1, %cl
+; X32-NEXT: xorl %esi, %eax
+; X32-NEXT: movzbl %cl, %ecx
+; X32-NEXT: negl %ecx
+; X32-NEXT: andl %eax, %ecx
+; X32-NEXT: xorl %esi, %ecx
+; X32-NEXT: shll $16, %ecx
+; X32-NEXT: movl %ecx, {{[0-9]+}}(%esp)
+; X32-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NEXT: addl $8, %esp
+; X32-NEXT: popl %esi
+; X32-NEXT: retl
+;
+; X32-NOCMOV-LABEL: test_ctselect_bf16:
+; X32-NOCMOV: # %bb.0:
+; X32-NOCMOV-NEXT: pushl %esi
+; X32-NOCMOV-NEXT: subl $8, %esp
+; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: fstps (%esp)
+; X32-NOCMOV-NEXT: calll __truncsfbf2
+; X32-NOCMOV-NEXT: movl %eax, %esi
+; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: fstps (%esp)
+; X32-NOCMOV-NEXT: calll __truncsfbf2
+; X32-NOCMOV-NEXT: # kill: def $ax killed $ax def $eax
+; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: andb $1, %cl
+; X32-NOCMOV-NEXT: xorl %esi, %eax
+; X32-NOCMOV-NEXT: movzbl %cl, %ecx
+; X32-NOCMOV-NEXT: negl %ecx
+; X32-NOCMOV-NEXT: andl %eax, %ecx
+; X32-NOCMOV-NEXT: xorl %esi, %ecx
+; X32-NOCMOV-NEXT: shll $16, %ecx
+; X32-NOCMOV-NEXT: movl %ecx, {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: addl $8, %esp
+; X32-NOCMOV-NEXT: popl %esi
+; X32-NOCMOV-NEXT: retl
+ %result = call bfloat @llvm.ct.select.bf16(i1 %cond, bfloat %a, bfloat %b)
+ ret bfloat %result
+}
+
+define fp128 @test_ctselect_f128(i1 %cond, fp128 %a, fp128 %b) #0 {
+; X64-LABEL: test_ctselect_f128:
+; X64: # %bb.0:
+; X64-NEXT: # kill: def $edi killed $edi def $rdi
+; X64-NEXT: movaps %xmm1, -{{[0-9]+}}(%rsp)
+; X64-NEXT: movaps %xmm0, -{{[0-9]+}}(%rsp)
+; X64-NEXT: andl $1, %edi
+; X64-NEXT: negq %rdi
+; X64-NEXT: movq -{{[0-9]+}}(%rsp), %rax
+; X64-NEXT: movq -{{[0-9]+}}(%rsp), %rcx
+; X64-NEXT: movq -{{[0-9]+}}(%rsp), %rdx
+; X64-NEXT: xorq %rax, %rdx
+; X64-NEXT: andq %rdi, %rdx
+; X64-NEXT: xorq %rax, %rdx
+; X64-NEXT: movq %rdx, -{{[0-9]+}}(%rsp)
+; X64-NEXT: movq -{{[0-9]+}}(%rsp), %rax
+; X64-NEXT: xorq %rcx, %rax
+; X64-NEXT: andq %rdi, %rax
+; X64-NEXT: xorq %rcx, %rax
+; X64-NEXT: movq %rax, -{{[0-9]+}}(%rsp)
+; X64-NEXT: movaps -{{[0-9]+}}(%rsp), %xmm0
+; X64-NEXT: retq
+;
+; X32-LABEL: test_ctselect_f128:
+; X32: # %bb.0:
+; X32-NEXT: pushl %ebp
+; X32-NEXT: pushl %ebx
+; X32-NEXT: pushl %edi
+; X32-NEXT: pushl %esi
+; X32-NEXT: subl $12, %esp
+; X32-NEXT: movzbl {{[0-9]+}}(%esp), %ebx
+; X32-NEXT: andb $1, %bl
+; X32-NEXT: movl {{[0-9]+}}(%esp), %edi
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: xorl %edi, %ecx
+; X32-NEXT: movzbl %bl, %ebp
+; X32-NEXT: negl %ebp
+; X32-NEXT: andl %ebp, %ecx
+; X32-NEXT: movl {{[0-9]+}}(%esp), %esi
+; X32-NEXT: xorl {{[0-9]+}}(%esp), %esi
+; X32-NEXT: andl %ebp, %esi
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ebx
+; X32-NEXT: xorl {{[0-9]+}}(%esp), %ebx
+; X32-NEXT: andl %ebp, %ebx
+; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: xorl %edx, %eax
+; X32-NEXT: andl %ebp, %eax
+; X32-NEXT: xorl %edi, %ecx
+; X32-NEXT: xorl {{[0-9]+}}(%esp), %esi
+; X32-NEXT: xorl {{[0-9]+}}(%esp), %ebx
+; X32-NEXT: xorl %edx, %eax
+; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NEXT: movl %eax, 12(%edx)
+; X32-NEXT: movl %ebx, 8(%edx)
+; X32-NEXT: movl %esi, 4(%edx)
+; X32-NEXT: movl %ecx, (%edx)
+; X32-NEXT: movl %edx, %eax
+; X32-NEXT: addl $12, %esp
+; X32-NEXT: popl %esi
+; X32-NEXT: popl %edi
+; X32-NEXT: popl %ebx
+; X32-NEXT: popl %ebp
+; X32-NEXT: retl $4
+;
+; X32-NOCMOV-LABEL: test_ctselect_f128:
+; X32-NOCMOV: # %bb.0:
+; X32-NOCMOV-NEXT: pushl %ebp
+; X32-NOCMOV-NEXT: pushl %ebx
+; X32-NOCMOV-NEXT: pushl %edi
+; X32-NOCMOV-NEXT: pushl %esi
+; X32-NOCMOV-NEXT: subl $12, %esp
+; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %ebx
+; X32-NOCMOV-NEXT: andb $1, %bl
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edi
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: xorl %edi, %ecx
+; X32-NOCMOV-NEXT: movzbl %bl, %ebp
+; X32-NOCMOV-NEXT: negl %ebp
+; X32-NOCMOV-NEXT: andl %ebp, %ecx
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %esi
+; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %esi
+; X32-NOCMOV-NEXT: andl %ebp, %esi
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebx
+; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %ebx
+; X32-NOCMOV-NEXT: andl %ebp, %ebx
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: xorl %edx, %eax
+; X32-NOCMOV-NEXT: andl %ebp, %eax
+; X32-NOCMOV-NEXT: xorl %edi, %ecx
+; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %esi
+; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %ebx
+; X32-NOCMOV-NEXT: xorl %edx, %eax
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NOCMOV-NEXT: movl %eax, 12(%edx)
+; X32-NOCMOV-NEXT: movl %ebx, 8(%edx)
+; X32-NOCMOV-NEXT: movl %esi, 4(%edx)
+; X32-NOCMOV-NEXT: movl %ecx, (%edx)
+; X32-NOCMOV-NEXT: movl %edx, %eax
+; X32-NOCMOV-NEXT: addl $12, %esp
+; X32-NOCMOV-NEXT: popl %esi
+; X32-NOCMOV-NEXT: popl %edi
+; X32-NOCMOV-NEXT: popl %ebx
+; X32-NOCMOV-NEXT: popl %ebp
+; X32-NOCMOV-NEXT: retl $4
+ %result = call fp128 @llvm.ct.select.f128(i1 %cond, fp128 %a, fp128 %b)
+ ret fp128 %result
+}
+
+define x86_fp80 @test_ctselect_f80(i1 %cond, x86_fp80 %a, x86_fp80 %b) #0 {
+; X64-LABEL: test_ctselect_f80:
+; X64: # %bb.0:
+; X64-NEXT: fldt {{[0-9]+}}(%rsp)
+; X64-NEXT: fldt {{[0-9]+}}(%rsp)
+; X64-NEXT: fstpt -{{[0-9]+}}(%rsp)
+; X64-NEXT: fstpt -{{[0-9]+}}(%rsp)
+; X64-NEXT: movl -{{[0-9]+}}(%rsp), %ecx
+; X64-NEXT: movl -{{[0-9]+}}(%rsp), %eax
+; X64-NEXT: movzwl -{{[0-9]+}}(%rsp), %edx
+; X64-NEXT: xorw %cx, %dx
+; X64-NEXT: andl $1, %edi
+; X64-NEXT: negl %edi
+; X64-NEXT: andl %edi, %edx
+; X64-NEXT: xorl %ecx, %edx
+; X64-NEXT: movw %dx, -{{[0-9]+}}(%rsp)
+; X64-NEXT: movzwl -{{[0-9]+}}(%rsp), %ecx
+; X64-NEXT: movzwl -{{[0-9]+}}(%rsp), %edx
+; X64-NEXT: xorw %cx, %dx
+; X64-NEXT: andl %edi, %edx
+; X64-NEXT: xorl %ecx, %edx
+; X64-NEXT: movw %dx, -{{[0-9]+}}(%rsp)
+; X64-NEXT: movzwl -{{[0-9]+}}(%rsp), %ecx
+; X64-NEXT: xorw %ax, %cx
+; X64-NEXT: andl %edi, %ecx
+; X64-NEXT: xorl %eax, %ecx
+; X64-NEXT: movw %cx, -{{[0-9]+}}(%rsp)
+; X64-NEXT: movzwl -{{[0-9]+}}(%rsp), %eax
+; X64-NEXT: movzwl -{{[0-9]+}}(%rsp), %ecx
+; X64-NEXT: xorw %ax, %cx
+; X64-NEXT: andl %edi, %ecx
+; X64-NEXT: xorl %eax, %ecx
+; X64-NEXT: movw %cx, -{{[0-9]+}}(%rsp)
+; X64-NEXT: movl -{{[0-9]+}}(%rsp), %eax
+; X64-NEXT: movzwl -{{[0-9]+}}(%rsp), %ecx
+; X64-NEXT: xorw %ax, %cx
+; X64-NEXT: andl %edi, %ecx
+; X64-NEXT: xorl %eax, %ecx
+; X64-NEXT: movw %cx, -{{[0-9]+}}(%rsp)
+; X64-NEXT: fldt -{{[0-9]+}}(%rsp)
+; X64-NEXT: retq
+;
+; X32-LABEL: test_ctselect_f80:
+; X32: # %bb.0:
+; X32-NEXT: pushl %esi
+; X32-NEXT: subl $36, %esp
+; X32-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: andb $1, %al
+; X32-NEXT: fldt {{[0-9]+}}(%esp)
+; X32-NEXT: fldt {{[0-9]+}}(%esp)
+; X32-NEXT: fstpt {{[0-9]+}}(%esp)
+; X32-NEXT: fstpt {{[0-9]+}}(%esp)
+; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: movzwl {{[0-9]+}}(%esp), %esi
+; X32-NEXT: xorw %dx, %si
+; X32-NEXT: movzbl %al, %eax
+; X32-NEXT: negl %eax
+; X32-NEXT: andl %eax, %esi
+; X32-NEXT: xorl %edx, %esi
+; X32-NEXT: movw %si, (%esp)
+; X32-NEXT: movzwl {{[0-9]+}}(%esp), %edx
+; X32-NEXT: movzwl {{[0-9]+}}(%esp), %esi
+; X32-NEXT: xorw %dx, %si
+; X32-NEXT: andl %eax, %esi
+; X32-NEXT: xorl %edx, %esi
+; X32-NEXT: movw %si, {{[0-9]+}}(%esp)
+; X32-NEXT: movzwl {{[0-9]+}}(%esp), %edx
+; X32-NEXT: xorw %cx, %dx
+; X32-NEXT: andl %eax, %edx
+; X32-NEXT: xorl %ecx, %edx
+; X32-NEXT: movw %dx, {{[0-9]+}}(%esp)
+; X32-NEXT: movzwl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: movzwl {{[0-9]+}}(%esp), %edx
+; X32-NEXT: xorw %cx, %dx
+; X32-NEXT: andl %eax, %edx
+; X32-NEXT: xorl %ecx, %edx
+; X32-NEXT: movw %dx, {{[0-9]+}}(%esp)
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: movzwl {{[0-9]+}}(%esp), %edx
+; X32-NEXT: xorw %cx, %dx
+; X32-NEXT: andl %eax, %edx
+; X32-NEXT: xorl %ecx, %edx
+; X32-NEXT: movw %dx, {{[0-9]+}}(%esp)
+; X32-NEXT: fldt (%esp)
+; X32-NEXT: addl $36, %esp
+; X32-NEXT: popl %esi
+; X32-NEXT: retl
+;
+; X32-NOCMOV-LABEL: test_ctselect_f80:
+; X32-NOCMOV: # %bb.0:
+; X32-NOCMOV-NEXT: pushl %esi
+; X32-NOCMOV-NEXT: subl $36, %esp
+; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: andb $1, %al
+; X32-NOCMOV-NEXT: fldt {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: fldt {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: fstpt {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: fstpt {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: movzwl {{[0-9]+}}(%esp), %esi
+; X32-NOCMOV-NEXT: xorw %dx, %si
+; X32-NOCMOV-NEXT: movzbl %al, %eax
+; X32-NOCMOV-NEXT: negl %eax
+; X32-NOCMOV-NEXT: andl %eax, %esi
+; X32-NOCMOV-NEXT: xorl %edx, %esi
+; X32-NOCMOV-NEXT: movw %si, (%esp)
+; X32-NOCMOV-NEXT: movzwl {{[0-9]+}}(%esp), %edx
+; X32-NOCMOV-NEXT: movzwl {{[0-9]+}}(%esp), %esi
+; X32-NOCMOV-NEXT: xorw %dx, %si
+; X32-NOCMOV-NEXT: andl %eax, %esi
+; X32-NOCMOV-NEXT: xorl %edx, %esi
+; X32-NOCMOV-NEXT: movw %si, {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: movzwl {{[0-9]+}}(%esp), %edx
+; X32-NOCMOV-NEXT: xorw %cx, %dx
+; X32-NOCMOV-NEXT: andl %eax, %edx
+; X32-NOCMOV-NEXT: xorl %ecx, %edx
+; X32-NOCMOV-NEXT: movw %dx, {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: movzwl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: movzwl {{[0-9]+}}(%esp), %edx
+; X32-NOCMOV-NEXT: xorw %cx, %dx
+; X32-NOCMOV-NEXT: andl %eax, %edx
+; X32-NOCMOV-NEXT: xorl %ecx, %edx
+; X32-NOCMOV-NEXT: movw %dx, {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: movzwl {{[0-9]+}}(%esp), %edx
+; X32-NOCMOV-NEXT: xorw %cx, %dx
+; X32-NOCMOV-NEXT: andl %eax, %edx
+; X32-NOCMOV-NEXT: xorl %ecx, %edx
+; X32-NOCMOV-NEXT: movw %dx, {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: fldt (%esp)
+; X32-NOCMOV-NEXT: addl $36, %esp
+; X32-NOCMOV-NEXT: popl %esi
+; X32-NOCMOV-NEXT: retl
+ %result = call x86_fp80 @llvm.ct.select.f80(i1 %cond, x86_fp80 %a, x86_fp80 %b)
+ ret x86_fp80 %result
+}
+
+define ptr @test_ctselect_ptr(i1 %cond, ptr %a, ptr %b) #0 {
; X64-LABEL: test_ctselect_ptr:
; X64: # %bb.0:
; X64-NEXT: movl %edi, %eax
@@ -332,26 +677,34 @@ define ptr @test_ctselect_ptr(i1 %cond, ptr %a, ptr %b) {
}
; Test with constant conditions
-define i32 @test_ctselect_const_true(i32 %a, i32 %b) {
+define i32 @test_ctselect_const_true(i32 %a, i32 %b) #0 {
; X64-LABEL: test_ctselect_const_true:
; X64: # %bb.0:
; X64-NEXT: movl %edi, %eax
+; X64-NEXT: xorl %esi, %eax
+; X64-NEXT: xorl %esi, %eax
; X64-NEXT: retq
;
; X32-LABEL: test_ctselect_const_true:
; X32: # %bb.0:
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: xorl %ecx, %eax
+; X32-NEXT: xorl %ecx, %eax
; X32-NEXT: retl
;
; X32-NOCMOV-LABEL: test_ctselect_const_true:
; X32-NOCMOV: # %bb.0:
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: xorl %ecx, %eax
+; X32-NOCMOV-NEXT: xorl %ecx, %eax
; X32-NOCMOV-NEXT: retl
%result = call i32 @llvm.ct.select.i32(i1 true, i32 %a, i32 %b)
ret i32 %result
}
-define i32 @test_ctselect_const_false(i32 %a, i32 %b) {
+define i32 @test_ctselect_const_false(i32 %a, i32 %b) #0 {
; X64-LABEL: test_ctselect_const_false:
; X64: # %bb.0:
; X64-NEXT: movl %esi, %eax
@@ -359,19 +712,21 @@ define i32 @test_ctselect_const_false(i32 %a, i32 %b) {
;
; X32-LABEL: test_ctselect_const_false:
; X32: # %bb.0:
-; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: xorl %eax, %eax
+; X32-NEXT: xorl {{[0-9]+}}(%esp), %eax
; X32-NEXT: retl
;
; X32-NOCMOV-LABEL: test_ctselect_const_false:
; X32-NOCMOV: # %bb.0:
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: xorl %eax, %eax
+; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %eax
; X32-NOCMOV-NEXT: retl
%result = call i32 @llvm.ct.select.i32(i1 false, i32 %a, i32 %b)
ret i32 %result
}
; Test with comparison conditions
-define i32 @test_ctselect_icmp_eq(i32 %x, i32 %y, i32 %a, i32 %b) {
+define i32 @test_ctselect_icmp_eq(i32 %x, i32 %y, i32 %a, i32 %b) #0 {
; X64-LABEL: test_ctselect_icmp_eq:
; X64: # %bb.0:
; X64-NEXT: xorl %eax, %eax
@@ -415,7 +770,7 @@ define i32 @test_ctselect_icmp_eq(i32 %x, i32 %y, i32 %a, i32 %b) {
ret i32 %result
}
-define i32 @test_ctselect_icmp_ult(i32 %x, i32 %y, i32 %a, i32 %b) {
+define i32 @test_ctselect_icmp_ult(i32 %x, i32 %y, i32 %a, i32 %b) #0 {
; X64-LABEL: test_ctselect_icmp_ult:
; X64: # %bb.0:
; X64-NEXT: xorl %ecx, %edx
@@ -456,19 +811,21 @@ define i32 @test_ctselect_icmp_ult(i32 %x, i32 %y, i32 %a, i32 %b) {
ret i32 %result
}
-define float @test_ctselect_fcmp_oeq(float %x, float %y, float %a, float %b) {
+define float @test_ctselect_fcmp_oeq(float %x, float %y, float %a, float %b) #0 {
; X64-LABEL: test_ctselect_fcmp_oeq:
; X64: # %bb.0:
+; X64-NEXT: movd %xmm3, %eax
; X64-NEXT: cmpeqss %xmm1, %xmm0
-; X64-NEXT: xorps %xmm3, %xmm2
-; X64-NEXT: andps %xmm2, %xmm0
-; X64-NEXT: xorps %xmm3, %xmm0
+; X64-NEXT: pxor %xmm3, %xmm2
+; X64-NEXT: pand %xmm0, %xmm2
+; X64-NEXT: movd %xmm2, %ecx
+; X64-NEXT: xorl %eax, %ecx
+; X64-NEXT: movd %ecx, %xmm0
; X64-NEXT: retq
;
; X32-LABEL: test_ctselect_fcmp_oeq:
; X32: # %bb.0:
; X32-NEXT: subl $12, %esp
-; X32-NEXT: .cfi_def_cfa_offset 16
; X32-NEXT: flds {{[0-9]+}}(%esp)
; X32-NEXT: fstps {{[0-9]+}}(%esp)
; X32-NEXT: flds {{[0-9]+}}(%esp)
@@ -490,13 +847,11 @@ define float @test_ctselect_fcmp_oeq(float %x, float %y, float %a, float %b) {
; X32-NEXT: movl %edx, {{[0-9]+}}(%esp)
; X32-NEXT: flds {{[0-9]+}}(%esp)
; X32-NEXT: addl $12, %esp
-; X32-NEXT: .cfi_def_cfa_offset 4
; X32-NEXT: retl
;
; X32-NOCMOV-LABEL: test_ctselect_fcmp_oeq:
; X32-NOCMOV: # %bb.0:
; X32-NOCMOV-NEXT: subl $12, %esp
-; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 16
; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
; X32-NOCMOV-NEXT: fstps {{[0-9]+}}(%esp)
; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
@@ -520,7 +875,6 @@ define float @test_ctselect_fcmp_oeq(float %x, float %y, float %a, float %b) {
; X32-NOCMOV-NEXT: movl %edx, {{[0-9]+}}(%esp)
; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
; X32-NOCMOV-NEXT: addl $12, %esp
-; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 4
; X32-NOCMOV-NEXT: retl
%cond = fcmp oeq float %x, %y
%result = call float @llvm.ct.select.f32(i1 %cond, float %a, float %b)
@@ -528,7 +882,7 @@ define float @test_ctselect_fcmp_oeq(float %x, float %y, float %a, float %b) {
}
; Test with memory operands
-define i32 @test_ctselect_load(i1 %cond, ptr %p1, ptr %p2) {
+define i32 @test_ctselect_load(i1 %cond, ptr %p1, ptr %p2) #0 {
; X64-LABEL: test_ctselect_load:
; X64: # %bb.0:
; X64-NEXT: movl (%rdx), %ecx
@@ -576,7 +930,7 @@ define i32 @test_ctselect_load(i1 %cond, ptr %p1, ptr %p2) {
}
; Test nested ct_select calls
-define i32 @test_ctselect_nested(i1 %cond1, i1 %cond2, i32 %a, i32 %b, i32 %c) {
+define i32 @test_ctselect_nested(i1 %cond1, i1 %cond2, i32 %a, i32 %b, i32 %c) #0 {
; X64-LABEL: test_ctselect_nested:
; X64: # %bb.0:
; X64-NEXT: movl %edi, %eax
@@ -595,11 +949,7 @@ define i32 @test_ctselect_nested(i1 %cond1, i1 %cond2, i32 %a, i32 %b, i32 %c) {
; X32-LABEL: test_ctselect_nested:
; X32: # %bb.0:
; X32-NEXT: pushl %edi
-; X32-NEXT: .cfi_def_cfa_offset 8
; X32-NEXT: pushl %esi
-; X32-NEXT: .cfi_def_cfa_offset 12
-; X32-NEXT: .cfi_offset %esi, -12
-; X32-NEXT: .cfi_offset %edi, -8
; X32-NEXT: movzbl {{[0-9]+}}(%esp), %eax
; X32-NEXT: andb $1, %al
; X32-NEXT: movb {{[0-9]+}}(%esp), %ah
@@ -618,19 +968,13 @@ define i32 @test_ctselect_nested(i1 %cond1, i1 %cond2, i32 %a, i32 %b, i32 %c) {
; X32-NEXT: andl %edx, %eax
; X32-NEXT: xorl %ecx, %eax
; X32-NEXT: popl %esi
-; X32-NEXT: .cfi_def_cfa_offset 8
; X32-NEXT: popl %edi
-; X32-NEXT: .cfi_def_cfa_offset 4
; X32-NEXT: retl
;
; X32-NOCMOV-LABEL: test_ctselect_nested:
; X32-NOCMOV: # %bb.0:
; X32-NOCMOV-NEXT: pushl %edi
-; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 8
; X32-NOCMOV-NEXT: pushl %esi
-; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 12
-; X32-NOCMOV-NEXT: .cfi_offset %esi, -12
-; X32-NOCMOV-NEXT: .cfi_offset %edi, -8
; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %eax
; X32-NOCMOV-NEXT: andb $1, %al
; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %ah
@@ -649,9 +993,7 @@ define i32 @test_ctselect_nested(i1 %cond1, i1 %cond2, i32 %a, i32 %b, i32 %c) {
; X32-NOCMOV-NEXT: andl %edx, %eax
; X32-NOCMOV-NEXT: xorl %ecx, %eax
; X32-NOCMOV-NEXT: popl %esi
-; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 8
; X32-NOCMOV-NEXT: popl %edi
-; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 4
; X32-NOCMOV-NEXT: retl
%inner = call i32 @llvm.ct.select.i32(i1 %cond2, i32 %a, i32 %b)
%result = call i32 @llvm.ct.select.i32(i1 %cond1, i32 %inner, i32 %c)
@@ -661,13 +1003,14 @@ define i32 @test_ctselect_nested(i1 %cond1, i1 %cond2, i32 %a, i32 %b, i32 %c) {
; Test nested CT_SELECT pattern with AND merging on i1 values
; Pattern: ct_select C0, (ct_select C1, X, Y), Y -> ct_select (C0 & C1), X, Y
; This optimization only applies when selecting between i1 values (boolean logic)
-define i32 @test_ctselect_nested_and_i1_to_i32(i1 %c0, i1 %c1, i32 %x, i32 %y) {
+define i32 @test_ctselect_nested_and_i1_to_i32(i1 %c0, i1 %c1, i32 %x, i32 %y) #0 {
; X64-LABEL: test_ctselect_nested_and_i1_to_i32:
; X64: # %bb.0:
-; X64-NEXT: movl %edi, %eax
-; X64-NEXT: andl %esi, %eax
+; X64-NEXT: andl %esi, %edi
+; X64-NEXT: andb $1, %dil
+; X64-NEXT: andb $1, %dil
; X64-NEXT: xorl %ecx, %edx
-; X64-NEXT: andl $1, %eax
+; X64-NEXT: movzbl %dil, %eax
; X64-NEXT: negl %eax
; X64-NEXT: andl %edx, %eax
; X64-NEXT: xorl %ecx, %eax
@@ -679,6 +1022,7 @@ define i32 @test_ctselect_nested_and_i1_to_i32(i1 %c0, i1 %c1, i32 %x, i32 %y) {
; X32-NEXT: movzbl {{[0-9]+}}(%esp), %eax
; X32-NEXT: andb {{[0-9]+}}(%esp), %al
; X32-NEXT: andb $1, %al
+; X32-NEXT: andb $1, %al
; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
; X32-NEXT: xorl %ecx, %edx
; X32-NEXT: movzbl %al, %eax
@@ -693,6 +1037,7 @@ define i32 @test_ctselect_nested_and_i1_to_i32(i1 %c0, i1 %c1, i32 %x, i32 %y) {
; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %eax
; X32-NOCMOV-NEXT: andb {{[0-9]+}}(%esp), %al
; X32-NOCMOV-NEXT: andb $1, %al
+; X32-NOCMOV-NEXT: andb $1, %al
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
; X32-NOCMOV-NEXT: xorl %ecx, %edx
; X32-NOCMOV-NEXT: movzbl %al, %eax
@@ -709,13 +1054,14 @@ define i32 @test_ctselect_nested_and_i1_to_i32(i1 %c0, i1 %c1, i32 %x, i32 %y) {
; Test nested CT_SELECT pattern with OR merging on i1 values
; Pattern: ct_select C0, X, (ct_select C1, X, Y) -> ct_select (C0 | C1), X, Y
; This optimization only applies when selecting between i1 values (boolean logic)
-define i32 @test_ctselect_nested_or_i1_to_i32(i1 %c0, i1 %c1, i32 %x, i32 %y) {
+define i32 @test_ctselect_nested_or_i1_to_i32(i1 %c0, i1 %c1, i32 %x, i32 %y) #0 {
; X64-LABEL: test_ctselect_nested_or_i1_to_i32:
; X64: # %bb.0:
-; X64-NEXT: movl %edi, %eax
-; X64-NEXT: orl %esi, %eax
+; X64-NEXT: orl %esi, %edi
+; X64-NEXT: andb $1, %dil
+; X64-NEXT: andb $1, %dil
; X64-NEXT: xorl %ecx, %edx
-; X64-NEXT: andl $1, %eax
+; X64-NEXT: movzbl %dil, %eax
; X64-NEXT: negl %eax
; X64-NEXT: andl %edx, %eax
; X64-NEXT: xorl %ecx, %eax
@@ -727,6 +1073,7 @@ define i32 @test_ctselect_nested_or_i1_to_i32(i1 %c0, i1 %c1, i32 %x, i32 %y) {
; X32-NEXT: movzbl {{[0-9]+}}(%esp), %eax
; X32-NEXT: orb {{[0-9]+}}(%esp), %al
; X32-NEXT: andb $1, %al
+; X32-NEXT: andb $1, %al
; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
; X32-NEXT: xorl %ecx, %edx
; X32-NEXT: movzbl %al, %eax
@@ -741,6 +1088,7 @@ define i32 @test_ctselect_nested_or_i1_to_i32(i1 %c0, i1 %c1, i32 %x, i32 %y) {
; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %eax
; X32-NOCMOV-NEXT: orb {{[0-9]+}}(%esp), %al
; X32-NOCMOV-NEXT: andb $1, %al
+; X32-NOCMOV-NEXT: andb $1, %al
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
; X32-NOCMOV-NEXT: xorl %ecx, %edx
; X32-NOCMOV-NEXT: movzbl %al, %eax
@@ -759,14 +1107,15 @@ define i32 @test_ctselect_nested_or_i1_to_i32(i1 %c0, i1 %c1, i32 %x, i32 %y) {
; -> ct_select C0, (ct_select (C1 & C2), X, Y), Y
; -> ct_select (C0 & (C1 & C2)), X, Y
; This tests that the optimization can be applied recursively
-define i32 @test_ctselect_double_nested_and_i1(i1 %c0, i1 %c1, i1 %c2, i32 %x, i32 %y) {
+define i32 @test_ctselect_double_nested_and_i1(i1 %c0, i1 %c1, i1 %c2, i32 %x, i32 %y) #0 {
; X64-LABEL: test_ctselect_double_nested_and_i1:
; X64: # %bb.0:
-; X64-NEXT: movl %edi, %eax
-; X64-NEXT: andl %esi, %eax
-; X64-NEXT: andl %edx, %eax
+; X64-NEXT: andl %esi, %edi
+; X64-NEXT: andl %edx, %edi
+; X64-NEXT: andb $1, %dil
+; X64-NEXT: andb $1, %dil
; X64-NEXT: xorl %r8d, %ecx
-; X64-NEXT: andl $1, %eax
+; X64-NEXT: movzbl %dil, %eax
; X64-NEXT: negl %eax
; X64-NEXT: andl %ecx, %eax
; X64-NEXT: xorl %r8d, %eax
@@ -779,6 +1128,7 @@ define i32 @test_ctselect_double_nested_and_i1(i1 %c0, i1 %c1, i1 %c2, i32 %x, i
; X32-NEXT: andb {{[0-9]+}}(%esp), %al
; X32-NEXT: andb {{[0-9]+}}(%esp), %al
; X32-NEXT: andb $1, %al
+; X32-NEXT: andb $1, %al
; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
; X32-NEXT: xorl %ecx, %edx
; X32-NEXT: movzbl %al, %eax
@@ -794,6 +1144,7 @@ define i32 @test_ctselect_double_nested_and_i1(i1 %c0, i1 %c1, i1 %c2, i32 %x, i
; X32-NOCMOV-NEXT: andb {{[0-9]+}}(%esp), %al
; X32-NOCMOV-NEXT: andb {{[0-9]+}}(%esp), %al
; X32-NOCMOV-NEXT: andb $1, %al
+; X32-NOCMOV-NEXT: andb $1, %al
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
; X32-NOCMOV-NEXT: xorl %ecx, %edx
; X32-NOCMOV-NEXT: movzbl %al, %eax
@@ -813,14 +1164,14 @@ define i32 @test_ctselect_double_nested_and_i1(i1 %c0, i1 %c1, i1 %c2, i32 %x, i
; Test vector CT_SELECT with v4i32 (128-bit vector with single i1 mask)
; NOW CONSTANT-TIME: Uses bitwise XOR/AND operations instead of branches!
-define <4 x i32> @test_ctselect_v4i32(i1 %cond, <4 x i32> %a, <4 x i32> %b) {
+define <4 x i32> @test_ctselect_v4i32(i1 %cond, <4 x i32> %a, <4 x i32> %b) #0 {
; X64-LABEL: test_ctselect_v4i32:
; X64: # %bb.0:
; X64-NEXT: pxor %xmm1, %xmm0
+; X64-NEXT: andl $1, %edi
+; X64-NEXT: negl %edi
; X64-NEXT: movd %edi, %xmm2
; X64-NEXT: pshufd {{.*#+}} xmm2 = xmm2[0,0,0,0]
-; X64-NEXT: pslld $31, %xmm2
-; X64-NEXT: psrad $31, %xmm2
; X64-NEXT: pand %xmm2, %xmm0
; X64-NEXT: pxor %xmm1, %xmm0
; X64-NEXT: retq
@@ -828,117 +1179,93 @@ define <4 x i32> @test_ctselect_v4i32(i1 %cond, <4 x i32> %a, <4 x i32> %b) {
; X32-LABEL: test_ctselect_v4i32:
; X32: # %bb.0:
; X32-NEXT: pushl %ebp
-; X32-NEXT: .cfi_def_cfa_offset 8
; X32-NEXT: pushl %ebx
-; X32-NEXT: .cfi_def_cfa_offset 12
; X32-NEXT: pushl %edi
-; X32-NEXT: .cfi_def_cfa_offset 16
; X32-NEXT: pushl %esi
-; X32-NEXT: .cfi_def_cfa_offset 20
-; X32-NEXT: .cfi_offset %esi, -20
-; X32-NEXT: .cfi_offset %edi, -16
-; X32-NEXT: .cfi_offset %ebx, -12
-; X32-NEXT: .cfi_offset %ebp, -8
-; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: movzbl {{[0-9]+}}(%esp), %ebx
+; X32-NEXT: andb $1, %bl
+; X32-NEXT: movl {{[0-9]+}}(%esp), %edi
; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: xorl %edi, %ecx
+; X32-NEXT: movzbl %bl, %ebp
+; X32-NEXT: negl %ebp
+; X32-NEXT: andl %ebp, %ecx
; X32-NEXT: movl {{[0-9]+}}(%esp), %esi
-; X32-NEXT: movl {{[0-9]+}}(%esp), %ebp
+; X32-NEXT: xorl {{[0-9]+}}(%esp), %esi
+; X32-NEXT: andl %ebp, %esi
; X32-NEXT: movl {{[0-9]+}}(%esp), %ebx
+; X32-NEXT: xorl {{[0-9]+}}(%esp), %ebx
+; X32-NEXT: andl %ebp, %ebx
; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
-; X32-NEXT: xorl %ebx, %edx
-; X32-NEXT: movzbl {{[0-9]+}}(%esp), %edi
-; X32-NEXT: andl $1, %edi
-; X32-NEXT: negl %edi
-; X32-NEXT: andl %edi, %edx
-; X32-NEXT: xorl %ebx, %edx
-; X32-NEXT: movl {{[0-9]+}}(%esp), %ebx
-; X32-NEXT: xorl %ebp, %ebx
-; X32-NEXT: andl %edi, %ebx
-; X32-NEXT: xorl %ebp, %ebx
-; X32-NEXT: movl {{[0-9]+}}(%esp), %ebp
-; X32-NEXT: xorl %esi, %ebp
-; X32-NEXT: andl %edi, %ebp
-; X32-NEXT: xorl %esi, %ebp
-; X32-NEXT: movl {{[0-9]+}}(%esp), %esi
-; X32-NEXT: xorl %ecx, %esi
-; X32-NEXT: andl %edi, %esi
-; X32-NEXT: xorl %ecx, %esi
-; X32-NEXT: movl %esi, 12(%eax)
-; X32-NEXT: movl %ebp, 8(%eax)
-; X32-NEXT: movl %ebx, 4(%eax)
-; X32-NEXT: movl %edx, (%eax)
+; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: xorl %edx, %eax
+; X32-NEXT: andl %ebp, %eax
+; X32-NEXT: xorl %edi, %ecx
+; X32-NEXT: xorl {{[0-9]+}}(%esp), %esi
+; X32-NEXT: xorl {{[0-9]+}}(%esp), %ebx
+; X32-NEXT: xorl %edx, %eax
+; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NEXT: movl %eax, 12(%edx)
+; X32-NEXT: movl %ebx, 8(%edx)
+; X32-NEXT: movl %esi, 4(%edx)
+; X32-NEXT: movl %ecx, (%edx)
+; X32-NEXT: movl %edx, %eax
; X32-NEXT: popl %esi
-; X32-NEXT: .cfi_def_cfa_offset 16
; X32-NEXT: popl %edi
-; X32-NEXT: .cfi_def_cfa_offset 12
; X32-NEXT: popl %ebx
-; X32-NEXT: .cfi_def_cfa_offset 8
; X32-NEXT: popl %ebp
-; X32-NEXT: .cfi_def_cfa_offset 4
; X32-NEXT: retl $4
;
; X32-NOCMOV-LABEL: test_ctselect_v4i32:
; X32-NOCMOV: # %bb.0:
; X32-NOCMOV-NEXT: pushl %ebp
-; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 8
; X32-NOCMOV-NEXT: pushl %ebx
-; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 12
; X32-NOCMOV-NEXT: pushl %edi
-; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 16
; X32-NOCMOV-NEXT: pushl %esi
-; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 20
-; X32-NOCMOV-NEXT: .cfi_offset %esi, -20
-; X32-NOCMOV-NEXT: .cfi_offset %edi, -16
-; X32-NOCMOV-NEXT: .cfi_offset %ebx, -12
-; X32-NOCMOV-NEXT: .cfi_offset %ebp, -8
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %ebx
+; X32-NOCMOV-NEXT: andb $1, %bl
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edi
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: xorl %edi, %ecx
+; X32-NOCMOV-NEXT: movzbl %bl, %ebp
+; X32-NOCMOV-NEXT: negl %ebp
+; X32-NOCMOV-NEXT: andl %ebp, %ecx
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %esi
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebp
+; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %esi
+; X32-NOCMOV-NEXT: andl %ebp, %esi
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebx
+; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %ebx
+; X32-NOCMOV-NEXT: andl %ebp, %ebx
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
-; X32-NOCMOV-NEXT: xorl %ebx, %edx
-; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %edi
-; X32-NOCMOV-NEXT: andl $1, %edi
-; X32-NOCMOV-NEXT: negl %edi
-; X32-NOCMOV-NEXT: andl %edi, %edx
-; X32-NOCMOV-NEXT: xorl %ebx, %edx
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebx
-; X32-NOCMOV-NEXT: xorl %ebp, %ebx
-; X32-NOCMOV-NEXT: andl %edi, %ebx
-; X32-NOCMOV-NEXT: xorl %ebp, %ebx
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebp
-; X32-NOCMOV-NEXT: xorl %esi, %ebp
-; X32-NOCMOV-NEXT: andl %edi, %ebp
-; X32-NOCMOV-NEXT: xorl %esi, %ebp
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %esi
-; X32-NOCMOV-NEXT: xorl %ecx, %esi
-; X32-NOCMOV-NEXT: andl %edi, %esi
-; X32-NOCMOV-NEXT: xorl %ecx, %esi
-; X32-NOCMOV-NEXT: movl %esi, 12(%eax)
-; X32-NOCMOV-NEXT: movl %ebp, 8(%eax)
-; X32-NOCMOV-NEXT: movl %ebx, 4(%eax)
-; X32-NOCMOV-NEXT: movl %edx, (%eax)
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: xorl %edx, %eax
+; X32-NOCMOV-NEXT: andl %ebp, %eax
+; X32-NOCMOV-NEXT: xorl %edi, %ecx
+; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %esi
+; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %ebx
+; X32-NOCMOV-NEXT: xorl %edx, %eax
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NOCMOV-NEXT: movl %eax, 12(%edx)
+; X32-NOCMOV-NEXT: movl %ebx, 8(%edx)
+; X32-NOCMOV-NEXT: movl %esi, 4(%edx)
+; X32-NOCMOV-NEXT: movl %ecx, (%edx)
+; X32-NOCMOV-NEXT: movl %edx, %eax
; X32-NOCMOV-NEXT: popl %esi
-; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 16
; X32-NOCMOV-NEXT: popl %edi
-; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 12
; X32-NOCMOV-NEXT: popl %ebx
-; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 8
; X32-NOCMOV-NEXT: popl %ebp
-; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 4
; X32-NOCMOV-NEXT: retl $4
%result = call <4 x i32> @llvm.ct.select.v4i32(i1 %cond, <4 x i32> %a, <4 x i32> %b)
ret <4 x i32> %result
}
-define <4 x float> @test_ctselect_v4f32(i1 %cond, <4 x float> %a, <4 x float> %b) {
+define <4 x float> @test_ctselect_v4f32(i1 %cond, <4 x float> %a, <4 x float> %b) #0 {
; X64-LABEL: test_ctselect_v4f32:
; X64: # %bb.0:
; X64-NEXT: pxor %xmm1, %xmm0
+; X64-NEXT: andl $1, %edi
+; X64-NEXT: negl %edi
; X64-NEXT: movd %edi, %xmm2
; X64-NEXT: pshufd {{.*#+}} xmm2 = xmm2[0,0,0,0]
-; X64-NEXT: pslld $31, %xmm2
-; X64-NEXT: psrad $31, %xmm2
; X64-NEXT: pand %xmm2, %xmm0
; X64-NEXT: pxor %xmm1, %xmm0
; X64-NEXT: retq
@@ -946,530 +1273,1133 @@ define <4 x float> @test_ctselect_v4f32(i1 %cond, <4 x float> %a, <4 x float> %b
; X32-LABEL: test_ctselect_v4f32:
; X32: # %bb.0:
; X32-NEXT: pushl %ebp
-; X32-NEXT: .cfi_def_cfa_offset 8
; X32-NEXT: pushl %ebx
-; X32-NEXT: .cfi_def_cfa_offset 12
; X32-NEXT: pushl %edi
-; X32-NEXT: .cfi_def_cfa_offset 16
; X32-NEXT: pushl %esi
-; X32-NEXT: .cfi_def_cfa_offset 20
-; X32-NEXT: .cfi_offset %esi, -20
-; X32-NEXT: .cfi_offset %edi, -16
-; X32-NEXT: .cfi_offset %ebx, -12
-; X32-NEXT: .cfi_offset %ebp, -8
+; X32-NEXT: subl $48, %esp
+; X32-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NEXT: fstps {{[0-9]+}}(%esp)
+; X32-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NEXT: fstps {{[0-9]+}}(%esp)
+; X32-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NEXT: fstps {{[0-9]+}}(%esp)
+; X32-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NEXT: fstps {{[0-9]+}}(%esp)
+; X32-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NEXT: fstps {{[0-9]+}}(%esp)
+; X32-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NEXT: fstps {{[0-9]+}}(%esp)
+; X32-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NEXT: fstps {{[0-9]+}}(%esp)
+; X32-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NEXT: fstps (%esp)
+; X32-NEXT: movzbl {{[0-9]+}}(%esp), %edx
+; X32-NEXT: andb $1, %dl
; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
; X32-NEXT: movl {{[0-9]+}}(%esp), %esi
-; X32-NEXT: movl {{[0-9]+}}(%esp), %ebp
-; X32-NEXT: movl {{[0-9]+}}(%esp), %ebx
-; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
-; X32-NEXT: xorl %ebx, %edx
-; X32-NEXT: movzbl {{[0-9]+}}(%esp), %edi
-; X32-NEXT: andl $1, %edi
-; X32-NEXT: negl %edi
-; X32-NEXT: andl %edi, %edx
-; X32-NEXT: xorl %ebx, %edx
+; X32-NEXT: movl {{[0-9]+}}(%esp), %edi
; X32-NEXT: movl {{[0-9]+}}(%esp), %ebx
-; X32-NEXT: xorl %ebp, %ebx
-; X32-NEXT: andl %edi, %ebx
-; X32-NEXT: xorl %ebp, %ebx
+; X32-NEXT: movzbl %dl, %edx
+; X32-NEXT: negl %edx
; X32-NEXT: movl {{[0-9]+}}(%esp), %ebp
-; X32-NEXT: xorl %esi, %ebp
-; X32-NEXT: andl %edi, %ebp
-; X32-NEXT: xorl %esi, %ebp
-; X32-NEXT: movl {{[0-9]+}}(%esp), %esi
+; X32-NEXT: xorl %ebx, %ebp
+; X32-NEXT: andl %edx, %ebp
+; X32-NEXT: xorl %ebx, %ebp
+; X32-NEXT: movl %ebp, {{[0-9]+}}(%esp)
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ebx
+; X32-NEXT: xorl %edi, %ebx
+; X32-NEXT: andl %edx, %ebx
+; X32-NEXT: xorl %edi, %ebx
+; X32-NEXT: movl %ebx, {{[0-9]+}}(%esp)
+; X32-NEXT: movl {{[0-9]+}}(%esp), %edi
+; X32-NEXT: xorl %esi, %edi
+; X32-NEXT: andl %edx, %edi
+; X32-NEXT: xorl %esi, %edi
+; X32-NEXT: movl %edi, {{[0-9]+}}(%esp)
+; X32-NEXT: movl (%esp), %esi
; X32-NEXT: xorl %ecx, %esi
-; X32-NEXT: andl %edi, %esi
+; X32-NEXT: andl %edx, %esi
; X32-NEXT: xorl %ecx, %esi
-; X32-NEXT: movl %esi, 12(%eax)
-; X32-NEXT: movl %ebp, 8(%eax)
-; X32-NEXT: movl %ebx, 4(%eax)
-; X32-NEXT: movl %edx, (%eax)
+; X32-NEXT: movl %esi, {{[0-9]+}}(%esp)
+; X32-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NEXT: fstps 12(%eax)
+; X32-NEXT: fstps 8(%eax)
+; X32-NEXT: fstps 4(%eax)
+; X32-NEXT: fstps (%eax)
+; X32-NEXT: addl $48, %esp
; X32-NEXT: popl %esi
-; X32-NEXT: .cfi_def_cfa_offset 16
; X32-NEXT: popl %edi
-; X32-NEXT: .cfi_def_cfa_offset 12
; X32-NEXT: popl %ebx
-; X32-NEXT: .cfi_def_cfa_offset 8
; X32-NEXT: popl %ebp
-; X32-NEXT: .cfi_def_cfa_offset 4
; X32-NEXT: retl $4
;
; X32-NOCMOV-LABEL: test_ctselect_v4f32:
; X32-NOCMOV: # %bb.0:
; X32-NOCMOV-NEXT: pushl %ebp
-; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 8
; X32-NOCMOV-NEXT: pushl %ebx
-; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 12
; X32-NOCMOV-NEXT: pushl %edi
-; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 16
; X32-NOCMOV-NEXT: pushl %esi
-; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 20
-; X32-NOCMOV-NEXT: .cfi_offset %esi, -20
-; X32-NOCMOV-NEXT: .cfi_offset %edi, -16
-; X32-NOCMOV-NEXT: .cfi_offset %ebx, -12
-; X32-NOCMOV-NEXT: .cfi_offset %ebp, -8
+; X32-NOCMOV-NEXT: subl $48, %esp
+; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: fstps {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: fstps {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: fstps {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: fstps {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: fstps {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: fstps {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: fstps {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: fstps (%esp)
+; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %edx
+; X32-NOCMOV-NEXT: andb $1, %dl
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %esi
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebp
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebx
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
-; X32-NOCMOV-NEXT: xorl %ebx, %edx
-; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %edi
-; X32-NOCMOV-NEXT: andl $1, %edi
-; X32-NOCMOV-NEXT: negl %edi
-; X32-NOCMOV-NEXT: andl %edi, %edx
-; X32-NOCMOV-NEXT: xorl %ebx, %edx
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edi
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebx
-; X32-NOCMOV-NEXT: xorl %ebp, %ebx
-; X32-NOCMOV-NEXT: andl %edi, %ebx
-; X32-NOCMOV-NEXT: xorl %ebp, %ebx
+; X32-NOCMOV-NEXT: movzbl %dl, %edx
+; X32-NOCMOV-NEXT: negl %edx
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebp
-; X32-NOCMOV-NEXT: xorl %esi, %ebp
-; X32-NOCMOV-NEXT: andl %edi, %ebp
-; X32-NOCMOV-NEXT: xorl %esi, %ebp
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %esi
+; X32-NOCMOV-NEXT: xorl %ebx, %ebp
+; X32-NOCMOV-NEXT: andl %edx, %ebp
+; X32-NOCMOV-NEXT: xorl %ebx, %ebp
+; X32-NOCMOV-NEXT: movl %ebp, {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebx
+; X32-NOCMOV-NEXT: xorl %edi, %ebx
+; X32-NOCMOV-NEXT: andl %edx, %ebx
+; X32-NOCMOV-NEXT: xorl %edi, %ebx
+; X32-NOCMOV-NEXT: movl %ebx, {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edi
+; X32-NOCMOV-NEXT: xorl %esi, %edi
+; X32-NOCMOV-NEXT: andl %edx, %edi
+; X32-NOCMOV-NEXT: xorl %esi, %edi
+; X32-NOCMOV-NEXT: movl %edi, {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: movl (%esp), %esi
; X32-NOCMOV-NEXT: xorl %ecx, %esi
-; X32-NOCMOV-NEXT: andl %edi, %esi
+; X32-NOCMOV-NEXT: andl %edx, %esi
; X32-NOCMOV-NEXT: xorl %ecx, %esi
-; X32-NOCMOV-NEXT: movl %esi, 12(%eax)
-; X32-NOCMOV-NEXT: movl %ebp, 8(%eax)
-; X32-NOCMOV-NEXT: movl %ebx, 4(%eax)
-; X32-NOCMOV-NEXT: movl %edx, (%eax)
+; X32-NOCMOV-NEXT: movl %esi, {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: fstps 12(%eax)
+; X32-NOCMOV-NEXT: fstps 8(%eax)
+; X32-NOCMOV-NEXT: fstps 4(%eax)
+; X32-NOCMOV-NEXT: fstps (%eax)
+; X32-NOCMOV-NEXT: addl $48, %esp
; X32-NOCMOV-NEXT: popl %esi
-; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 16
; X32-NOCMOV-NEXT: popl %edi
-; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 12
; X32-NOCMOV-NEXT: popl %ebx
-; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 8
; X32-NOCMOV-NEXT: popl %ebp
-; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 4
; X32-NOCMOV-NEXT: retl $4
%result = call <4 x float> @llvm.ct.select.v4f32(i1 %cond, <4 x float> %a, <4 x float> %b)
ret <4 x float> %result
}
-define <8 x i32> @test_ctselect_v8i32_avx(i1 %cond, <8 x i32> %a, <8 x i32> %b) {
+define <8 x i32> @test_ctselect_v8i32_avx(i1 %cond, <8 x i32> %a, <8 x i32> %b) #0 {
; X64-LABEL: test_ctselect_v8i32_avx:
; X64: # %bb.0:
+; X64-NEXT: pxor %xmm2, %xmm0
+; X64-NEXT: andl $1, %edi
+; X64-NEXT: negl %edi
; X64-NEXT: movd %edi, %xmm4
; X64-NEXT: pshufd {{.*#+}} xmm4 = xmm4[0,0,0,0]
-; X64-NEXT: pslld $31, %xmm4
-; X64-NEXT: psrad $31, %xmm4
-; X64-NEXT: movdqa %xmm4, %xmm5
-; X64-NEXT: pandn %xmm2, %xmm5
; X64-NEXT: pand %xmm4, %xmm0
-; X64-NEXT: por %xmm5, %xmm0
+; X64-NEXT: pxor %xmm2, %xmm0
+; X64-NEXT: pxor %xmm3, %xmm1
; X64-NEXT: pand %xmm4, %xmm1
-; X64-NEXT: pandn %xmm3, %xmm4
-; X64-NEXT: por %xmm4, %xmm1
+; X64-NEXT: pxor %xmm3, %xmm1
; X64-NEXT: retq
;
; X32-LABEL: test_ctselect_v8i32_avx:
; X32: # %bb.0:
; X32-NEXT: pushl %ebp
-; X32-NEXT: .cfi_def_cfa_offset 8
; X32-NEXT: pushl %ebx
-; X32-NEXT: .cfi_def_cfa_offset 12
; X32-NEXT: pushl %edi
-; X32-NEXT: .cfi_def_cfa_offset 16
; X32-NEXT: pushl %esi
-; X32-NEXT: .cfi_def_cfa_offset 20
; X32-NEXT: subl $8, %esp
-; X32-NEXT: .cfi_def_cfa_offset 28
-; X32-NEXT: .cfi_offset %esi, -20
-; X32-NEXT: .cfi_offset %edi, -16
-; X32-NEXT: .cfi_offset %ebx, -12
-; X32-NEXT: .cfi_offset %ebp, -8
-; X32-NEXT: movl {{[0-9]+}}(%esp), %edi
+; X32-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: andb $1, %al
+; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NEXT: xorl {{[0-9]+}}(%esp), %edx
+; X32-NEXT: movzbl %al, %ecx
+; X32-NEXT: negl %ecx
+; X32-NEXT: andl %ecx, %edx
+; X32-NEXT: movl %edx, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
+; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: xorl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: andl %ecx, %eax
+; X32-NEXT: movl %eax, (%esp) # 4-byte Spill
; X32-NEXT: movl {{[0-9]+}}(%esp), %ebp
+; X32-NEXT: xorl {{[0-9]+}}(%esp), %ebp
+; X32-NEXT: andl %ecx, %ebp
; X32-NEXT: movl {{[0-9]+}}(%esp), %ebx
+; X32-NEXT: xorl {{[0-9]+}}(%esp), %ebx
+; X32-NEXT: andl %ecx, %ebx
+; X32-NEXT: movl {{[0-9]+}}(%esp), %edi
+; X32-NEXT: xorl {{[0-9]+}}(%esp), %edi
+; X32-NEXT: andl %ecx, %edi
; X32-NEXT: movl {{[0-9]+}}(%esp), %esi
+; X32-NEXT: xorl {{[0-9]+}}(%esp), %esi
+; X32-NEXT: andl %ecx, %esi
+; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NEXT: xorl {{[0-9]+}}(%esp), %edx
+; X32-NEXT: andl %ecx, %edx
; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: xorl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: andl %ecx, %eax
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: xorl %ecx, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Folded Spill
; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
-; X32-NEXT: xorl %eax, %ecx
+; X32-NEXT: xorl %ecx, (%esp) # 4-byte Folded Spill
+; X32-NEXT: xorl {{[0-9]+}}(%esp), %ebp
+; X32-NEXT: xorl {{[0-9]+}}(%esp), %ebx
+; X32-NEXT: xorl {{[0-9]+}}(%esp), %edi
+; X32-NEXT: xorl {{[0-9]+}}(%esp), %esi
+; X32-NEXT: xorl {{[0-9]+}}(%esp), %edx
+; X32-NEXT: xorl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: movl %eax, 28(%ecx)
+; X32-NEXT: movl %edx, 24(%ecx)
+; X32-NEXT: movl %esi, 20(%ecx)
+; X32-NEXT: movl %edi, 16(%ecx)
+; X32-NEXT: movl %ebx, 12(%ecx)
+; X32-NEXT: movl %ebp, 8(%ecx)
+; X32-NEXT: movl (%esp), %eax # 4-byte Reload
+; X32-NEXT: movl %eax, 4(%ecx)
+; X32-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 4-byte Reload
+; X32-NEXT: movl %eax, (%ecx)
+; X32-NEXT: movl %ecx, %eax
+; X32-NEXT: addl $8, %esp
+; X32-NEXT: popl %esi
+; X32-NEXT: popl %edi
+; X32-NEXT: popl %ebx
+; X32-NEXT: popl %ebp
+; X32-NEXT: retl $4
+;
+; X32-NOCMOV-LABEL: test_ctselect_v8i32_avx:
+; X32-NOCMOV: # %bb.0:
+; X32-NOCMOV-NEXT: pushl %ebp
+; X32-NOCMOV-NEXT: pushl %ebx
+; X32-NOCMOV-NEXT: pushl %edi
+; X32-NOCMOV-NEXT: pushl %esi
+; X32-NOCMOV-NEXT: subl $8, %esp
+; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: andb $1, %al
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %edx
+; X32-NOCMOV-NEXT: movzbl %al, %ecx
+; X32-NOCMOV-NEXT: negl %ecx
+; X32-NOCMOV-NEXT: andl %ecx, %edx
+; X32-NOCMOV-NEXT: movl %edx, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: andl %ecx, %eax
+; X32-NOCMOV-NEXT: movl %eax, (%esp) # 4-byte Spill
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebp
+; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %ebp
+; X32-NOCMOV-NEXT: andl %ecx, %ebp
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebx
+; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %ebx
+; X32-NOCMOV-NEXT: andl %ecx, %ebx
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edi
+; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %edi
+; X32-NOCMOV-NEXT: andl %ecx, %edi
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %esi
+; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %esi
+; X32-NOCMOV-NEXT: andl %ecx, %esi
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %edx
+; X32-NOCMOV-NEXT: andl %ecx, %edx
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: andl %ecx, %eax
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: xorl %ecx, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Folded Spill
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: xorl %ecx, (%esp) # 4-byte Folded Spill
+; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %ebp
+; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %ebx
+; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %edi
+; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %esi
+; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %edx
+; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: movl %eax, 28(%ecx)
+; X32-NOCMOV-NEXT: movl %edx, 24(%ecx)
+; X32-NOCMOV-NEXT: movl %esi, 20(%ecx)
+; X32-NOCMOV-NEXT: movl %edi, 16(%ecx)
+; X32-NOCMOV-NEXT: movl %ebx, 12(%ecx)
+; X32-NOCMOV-NEXT: movl %ebp, 8(%ecx)
+; X32-NOCMOV-NEXT: movl (%esp), %eax # 4-byte Reload
+; X32-NOCMOV-NEXT: movl %eax, 4(%ecx)
+; X32-NOCMOV-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 4-byte Reload
+; X32-NOCMOV-NEXT: movl %eax, (%ecx)
+; X32-NOCMOV-NEXT: movl %ecx, %eax
+; X32-NOCMOV-NEXT: addl $8, %esp
+; X32-NOCMOV-NEXT: popl %esi
+; X32-NOCMOV-NEXT: popl %edi
+; X32-NOCMOV-NEXT: popl %ebx
+; X32-NOCMOV-NEXT: popl %ebp
+; X32-NOCMOV-NEXT: retl $4
+ %result = call <8 x i32> @llvm.ct.select.v8i32(i1 %cond, <8 x i32> %a, <8 x i32> %b)
+ ret <8 x i32> %result
+}
+
+define <8 x float> @test_ctselect_v8f32(i1 %cond, <8 x float> %a, <8 x float> %b) #0 {
+; X64-LABEL: test_ctselect_v8f32:
+; X64: # %bb.0:
+; X64-NEXT: pxor %xmm2, %xmm0
+; X64-NEXT: andl $1, %edi
+; X64-NEXT: negl %edi
+; X64-NEXT: movd %edi, %xmm4
+; X64-NEXT: pshufd {{.*#+}} xmm4 = xmm4[0,0,0,0]
+; X64-NEXT: pand %xmm4, %xmm0
+; X64-NEXT: pxor %xmm2, %xmm0
+; X64-NEXT: pxor %xmm3, %xmm1
+; X64-NEXT: pand %xmm4, %xmm1
+; X64-NEXT: pxor %xmm3, %xmm1
+; X64-NEXT: retq
+;
+; X32-LABEL: test_ctselect_v8f32:
+; X32: # %bb.0:
+; X32-NEXT: pushl %ebp
+; X32-NEXT: pushl %ebx
+; X32-NEXT: pushl %edi
+; X32-NEXT: pushl %esi
+; X32-NEXT: subl $104, %esp
+; X32-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NEXT: fstps {{[0-9]+}}(%esp)
+; X32-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NEXT: fstps {{[0-9]+}}(%esp)
+; X32-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NEXT: fstps {{[0-9]+}}(%esp)
+; X32-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NEXT: fstps {{[0-9]+}}(%esp)
+; X32-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NEXT: fstps {{[0-9]+}}(%esp)
+; X32-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NEXT: fstps {{[0-9]+}}(%esp)
+; X32-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NEXT: fstps {{[0-9]+}}(%esp)
+; X32-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NEXT: fstps {{[0-9]+}}(%esp)
+; X32-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NEXT: fstps {{[0-9]+}}(%esp)
+; X32-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NEXT: fstps {{[0-9]+}}(%esp)
+; X32-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NEXT: fstps {{[0-9]+}}(%esp)
+; X32-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NEXT: fstps {{[0-9]+}}(%esp)
+; X32-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NEXT: fstps {{[0-9]+}}(%esp)
+; X32-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NEXT: fstps {{[0-9]+}}(%esp)
+; X32-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NEXT: fstps {{[0-9]+}}(%esp)
+; X32-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NEXT: fstps {{[0-9]+}}(%esp)
; X32-NEXT: movzbl {{[0-9]+}}(%esp), %edx
-; X32-NEXT: andl $1, %edx
-; X32-NEXT: negl %edx
-; X32-NEXT: andl %edx, %ecx
-; X32-NEXT: xorl %eax, %ecx
-; X32-NEXT: movl %ecx, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
+; X32-NEXT: andb $1, %dl
; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
-; X32-NEXT: xorl %esi, %eax
-; X32-NEXT: andl %edx, %eax
-; X32-NEXT: xorl %esi, %eax
; X32-NEXT: movl %eax, (%esp) # 4-byte Spill
+; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
; X32-NEXT: movl {{[0-9]+}}(%esp), %esi
-; X32-NEXT: xorl %ebx, %esi
-; X32-NEXT: andl %edx, %esi
-; X32-NEXT: xorl %ebx, %esi
+; X32-NEXT: movl {{[0-9]+}}(%esp), %edi
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ebx
+; X32-NEXT: movzbl %dl, %edx
+; X32-NEXT: negl %edx
+; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: xorl %ebx, %eax
+; X32-NEXT: andl %edx, %eax
+; X32-NEXT: xorl %ebx, %eax
; X32-NEXT: movl {{[0-9]+}}(%esp), %ebx
-; X32-NEXT: xorl %ebp, %ebx
-; X32-NEXT: andl %edx, %ebx
-; X32-NEXT: xorl %ebp, %ebx
; X32-NEXT: movl {{[0-9]+}}(%esp), %ebp
-; X32-NEXT: xorl %edi, %ebp
-; X32-NEXT: andl %edx, %ebp
-; X32-NEXT: xorl %edi, %ebp
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: movl %eax, {{[0-9]+}}(%esp)
+; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: xorl %ecx, %eax
+; X32-NEXT: andl %edx, %eax
+; X32-NEXT: xorl %ecx, %eax
+; X32-NEXT: movl %eax, {{[0-9]+}}(%esp)
+; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: xorl %ebp, %eax
+; X32-NEXT: andl %edx, %eax
+; X32-NEXT: xorl %ebp, %eax
+; X32-NEXT: movl %eax, {{[0-9]+}}(%esp)
+; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: xorl %ebx, %eax
+; X32-NEXT: andl %edx, %eax
+; X32-NEXT: xorl %ebx, %eax
+; X32-NEXT: movl %eax, {{[0-9]+}}(%esp)
+; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: xorl %edi, %eax
+; X32-NEXT: andl %edx, %eax
+; X32-NEXT: xorl %edi, %eax
+; X32-NEXT: movl %eax, {{[0-9]+}}(%esp)
+; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: xorl %esi, %eax
+; X32-NEXT: andl %edx, %eax
+; X32-NEXT: xorl %esi, %eax
+; X32-NEXT: movl %eax, {{[0-9]+}}(%esp)
+; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %ecx # 4-byte Reload
+; X32-NEXT: xorl %ecx, %eax
+; X32-NEXT: andl %edx, %eax
+; X32-NEXT: xorl %ecx, %eax
+; X32-NEXT: movl %eax, {{[0-9]+}}(%esp)
+; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: movl (%esp), %ecx # 4-byte Reload
+; X32-NEXT: xorl %ecx, %eax
+; X32-NEXT: andl %edx, %eax
+; X32-NEXT: xorl %ecx, %eax
+; X32-NEXT: movl %eax, {{[0-9]+}}(%esp)
; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NEXT: fstps (%esp) # 4-byte Folded Spill
+; X32-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NEXT: fstps 28(%eax)
+; X32-NEXT: fstps 24(%eax)
+; X32-NEXT: fstps 20(%eax)
+; X32-NEXT: fstps 16(%eax)
+; X32-NEXT: fstps 12(%eax)
+; X32-NEXT: fstps 8(%eax)
+; X32-NEXT: fstps 4(%eax)
+; X32-NEXT: flds (%esp) # 4-byte Folded Reload
+; X32-NEXT: fstps (%eax)
+; X32-NEXT: addl $104, %esp
+; X32-NEXT: popl %esi
+; X32-NEXT: popl %edi
+; X32-NEXT: popl %ebx
+; X32-NEXT: popl %ebp
+; X32-NEXT: retl $4
+;
+; X32-NOCMOV-LABEL: test_ctselect_v8f32:
+; X32-NOCMOV: # %bb.0:
+; X32-NOCMOV-NEXT: pushl %ebp
+; X32-NOCMOV-NEXT: pushl %ebx
+; X32-NOCMOV-NEXT: pushl %edi
+; X32-NOCMOV-NEXT: pushl %esi
+; X32-NOCMOV-NEXT: subl $104, %esp
+; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: fstps {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: fstps {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: fstps {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: fstps {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: fstps {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: fstps {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: fstps {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: fstps {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: fstps {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: fstps {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: fstps {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: fstps {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: fstps {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: fstps {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: fstps {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: fstps {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %edx
+; X32-NOCMOV-NEXT: andb $1, %dl
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: movl %eax, (%esp) # 4-byte Spill
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %esi
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edi
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebx
+; X32-NOCMOV-NEXT: movzbl %dl, %edx
+; X32-NOCMOV-NEXT: negl %edx
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: xorl %ebx, %eax
+; X32-NOCMOV-NEXT: andl %edx, %eax
+; X32-NOCMOV-NEXT: xorl %ebx, %eax
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebx
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebp
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: movl %eax, {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: xorl %ecx, %eax
+; X32-NOCMOV-NEXT: andl %edx, %eax
+; X32-NOCMOV-NEXT: xorl %ecx, %eax
+; X32-NOCMOV-NEXT: movl %eax, {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: xorl %ebp, %eax
+; X32-NOCMOV-NEXT: andl %edx, %eax
+; X32-NOCMOV-NEXT: xorl %ebp, %eax
+; X32-NOCMOV-NEXT: movl %eax, {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: xorl %ebx, %eax
+; X32-NOCMOV-NEXT: andl %edx, %eax
+; X32-NOCMOV-NEXT: xorl %ebx, %eax
+; X32-NOCMOV-NEXT: movl %eax, {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: xorl %edi, %eax
+; X32-NOCMOV-NEXT: andl %edx, %eax
+; X32-NOCMOV-NEXT: xorl %edi, %eax
+; X32-NOCMOV-NEXT: movl %eax, {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: xorl %esi, %eax
+; X32-NOCMOV-NEXT: andl %edx, %eax
+; X32-NOCMOV-NEXT: xorl %esi, %eax
+; X32-NOCMOV-NEXT: movl %eax, {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %ecx # 4-byte Reload
+; X32-NOCMOV-NEXT: xorl %ecx, %eax
+; X32-NOCMOV-NEXT: andl %edx, %eax
+; X32-NOCMOV-NEXT: xorl %ecx, %eax
+; X32-NOCMOV-NEXT: movl %eax, {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: movl (%esp), %ecx # 4-byte Reload
+; X32-NOCMOV-NEXT: xorl %ecx, %eax
+; X32-NOCMOV-NEXT: andl %edx, %eax
+; X32-NOCMOV-NEXT: xorl %ecx, %eax
+; X32-NOCMOV-NEXT: movl %eax, {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: fstps (%esp) # 4-byte Folded Spill
+; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: fstps 28(%eax)
+; X32-NOCMOV-NEXT: fstps 24(%eax)
+; X32-NOCMOV-NEXT: fstps 20(%eax)
+; X32-NOCMOV-NEXT: fstps 16(%eax)
+; X32-NOCMOV-NEXT: fstps 12(%eax)
+; X32-NOCMOV-NEXT: fstps 8(%eax)
+; X32-NOCMOV-NEXT: fstps 4(%eax)
+; X32-NOCMOV-NEXT: flds (%esp) # 4-byte Folded Reload
+; X32-NOCMOV-NEXT: fstps (%eax)
+; X32-NOCMOV-NEXT: addl $104, %esp
+; X32-NOCMOV-NEXT: popl %esi
+; X32-NOCMOV-NEXT: popl %edi
+; X32-NOCMOV-NEXT: popl %ebx
+; X32-NOCMOV-NEXT: popl %ebp
+; X32-NOCMOV-NEXT: retl $4
+ %result = call <8 x float> @llvm.ct.select.v8f32(i1 %cond, <8 x float> %a, <8 x float> %b)
+ ret <8 x float> %result
+}
+
+define <8 x half> @test_ctselect_v8f16(i1 %cond, <8 x half> %a, <8 x half> %b) #0 {
+; X64-LABEL: test_ctselect_v8f16:
+; X64: # %bb.0:
+; X64-NEXT: pxor %xmm1, %xmm0
+; X64-NEXT: andl $1, %edi
+; X64-NEXT: negl %edi
+; X64-NEXT: movd %edi, %xmm2
+; X64-NEXT: pshuflw {{.*#+}} xmm2 = xmm2[0,0,0,0,4,5,6,7]
+; X64-NEXT: pshufd {{.*#+}} xmm2 = xmm2[0,1,0,1]
+; X64-NEXT: pand %xmm2, %xmm0
+; X64-NEXT: pxor %xmm1, %xmm0
+; X64-NEXT: retq
+;
+; X32-LABEL: test_ctselect_v8f16:
+; X32: # %bb.0:
+; X32-NEXT: pushl %ebp
+; X32-NEXT: pushl %ebx
+; X32-NEXT: pushl %edi
+; X32-NEXT: pushl %esi
+; X32-NEXT: subl $12, %esp
+; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NEXT: movl {{[0-9]+}}(%esp), %esi
+; X32-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: andb $1, %al
+; X32-NEXT: movl {{[0-9]+}}(%esp), %edi
+; X32-NEXT: movzwl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: xorw %di, %cx
+; X32-NEXT: movzbl %al, %eax
+; X32-NEXT: negl %eax
+; X32-NEXT: andl %eax, %ecx
+; X32-NEXT: movl %ecx, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
+; X32-NEXT: movzwl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: xorw %si, %cx
+; X32-NEXT: andl %eax, %ecx
+; X32-NEXT: movl %ecx, (%esp) # 4-byte Spill
+; X32-NEXT: movzwl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: xorw %dx, %cx
+; X32-NEXT: andl %eax, %ecx
+; X32-NEXT: movl %ecx, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: movzwl {{[0-9]+}}(%esp), %ebx
+; X32-NEXT: xorw %cx, %bx
+; X32-NEXT: andl %eax, %ebx
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: movzwl {{[0-9]+}}(%esp), %ebp
+; X32-NEXT: xorw %cx, %bp
+; X32-NEXT: andl %eax, %ebp
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: movzwl {{[0-9]+}}(%esp), %esi
+; X32-NEXT: xorw %cx, %si
+; X32-NEXT: andl %eax, %esi
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: movzwl {{[0-9]+}}(%esp), %edx
+; X32-NEXT: xorw %cx, %dx
+; X32-NEXT: andl %eax, %edx
+; X32-NEXT: movzwl {{[0-9]+}}(%esp), %ecx
; X32-NEXT: movl {{[0-9]+}}(%esp), %edi
-; X32-NEXT: xorl %eax, %edi
-; X32-NEXT: andl %edx, %edi
-; X32-NEXT: xorl %eax, %edi
+; X32-NEXT: xorw %di, %cx
+; X32-NEXT: andl %eax, %ecx
; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
-; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
-; X32-NEXT: xorl %eax, %ecx
-; X32-NEXT: andl %edx, %ecx
-; X32-NEXT: xorl %eax, %ecx
+; X32-NEXT: xorl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Folded Spill
; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
-; X32-NEXT: xorl {{[0-9]+}}(%esp), %eax
-; X32-NEXT: andl %edx, %eax
-; X32-NEXT: xorl {{[0-9]+}}(%esp), %eax
-; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
-; X32-NEXT: movl %eax, 28(%edx)
-; X32-NEXT: movl %ecx, 24(%edx)
-; X32-NEXT: movl %edi, 20(%edx)
-; X32-NEXT: movl %ebp, 16(%edx)
-; X32-NEXT: movl %ebx, 12(%edx)
-; X32-NEXT: movl %esi, 8(%edx)
-; X32-NEXT: movl (%esp), %eax # 4-byte Reload
-; X32-NEXT: movl %eax, 4(%edx)
-; X32-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 4-byte Reload
-; X32-NEXT: movl %eax, (%edx)
-; X32-NEXT: movl %edx, %eax
-; X32-NEXT: addl $8, %esp
-; X32-NEXT: .cfi_def_cfa_offset 20
+; X32-NEXT: xorl %eax, (%esp) # 4-byte Folded Spill
+; X32-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %edi # 4-byte Reload
+; X32-NEXT: xorl {{[0-9]+}}(%esp), %edi
+; X32-NEXT: xorl {{[0-9]+}}(%esp), %ebx
+; X32-NEXT: xorl {{[0-9]+}}(%esp), %ebp
+; X32-NEXT: xorl {{[0-9]+}}(%esp), %esi
+; X32-NEXT: xorl {{[0-9]+}}(%esp), %edx
+; X32-NEXT: xorl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: movw %cx, 14(%eax)
+; X32-NEXT: movw %dx, 12(%eax)
+; X32-NEXT: movw %si, 10(%eax)
+; X32-NEXT: movw %bp, 8(%eax)
+; X32-NEXT: movw %bx, 6(%eax)
+; X32-NEXT: movw %di, 4(%eax)
+; X32-NEXT: movl (%esp), %ecx # 4-byte Reload
+; X32-NEXT: movw %cx, 2(%eax)
+; X32-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %ecx # 4-byte Reload
+; X32-NEXT: movw %cx, (%eax)
+; X32-NEXT: addl $12, %esp
; X32-NEXT: popl %esi
-; X32-NEXT: .cfi_def_cfa_offset 16
; X32-NEXT: popl %edi
-; X32-NEXT: .cfi_def_cfa_offset 12
; X32-NEXT: popl %ebx
-; X32-NEXT: .cfi_def_cfa_offset 8
; X32-NEXT: popl %ebp
-; X32-NEXT: .cfi_def_cfa_offset 4
; X32-NEXT: retl $4
;
-; X32-NOCMOV-LABEL: test_ctselect_v8i32_avx:
+; X32-NOCMOV-LABEL: test_ctselect_v8f16:
; X32-NOCMOV: # %bb.0:
; X32-NOCMOV-NEXT: pushl %ebp
-; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 8
; X32-NOCMOV-NEXT: pushl %ebx
-; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 12
; X32-NOCMOV-NEXT: pushl %edi
-; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 16
; X32-NOCMOV-NEXT: pushl %esi
-; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 20
-; X32-NOCMOV-NEXT: subl $8, %esp
-; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 28
-; X32-NOCMOV-NEXT: .cfi_offset %esi, -20
-; X32-NOCMOV-NEXT: .cfi_offset %edi, -16
-; X32-NOCMOV-NEXT: .cfi_offset %ebx, -12
-; X32-NOCMOV-NEXT: .cfi_offset %ebp, -8
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edi
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebp
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebx
+; X32-NOCMOV-NEXT: subl $12, %esp
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %esi
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
-; X32-NOCMOV-NEXT: xorl %eax, %ecx
-; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %edx
-; X32-NOCMOV-NEXT: andl $1, %edx
-; X32-NOCMOV-NEXT: negl %edx
-; X32-NOCMOV-NEXT: andl %edx, %ecx
-; X32-NOCMOV-NEXT: xorl %eax, %ecx
+; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: andb $1, %al
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edi
+; X32-NOCMOV-NEXT: movzwl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: xorw %di, %cx
+; X32-NOCMOV-NEXT: movzbl %al, %eax
+; X32-NOCMOV-NEXT: negl %eax
+; X32-NOCMOV-NEXT: andl %eax, %ecx
; X32-NOCMOV-NEXT: movl %ecx, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
-; X32-NOCMOV-NEXT: xorl %esi, %eax
-; X32-NOCMOV-NEXT: andl %edx, %eax
-; X32-NOCMOV-NEXT: xorl %esi, %eax
-; X32-NOCMOV-NEXT: movl %eax, (%esp) # 4-byte Spill
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %esi
-; X32-NOCMOV-NEXT: xorl %ebx, %esi
-; X32-NOCMOV-NEXT: andl %edx, %esi
-; X32-NOCMOV-NEXT: xorl %ebx, %esi
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebx
-; X32-NOCMOV-NEXT: xorl %ebp, %ebx
-; X32-NOCMOV-NEXT: andl %edx, %ebx
-; X32-NOCMOV-NEXT: xorl %ebp, %ebx
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebp
-; X32-NOCMOV-NEXT: xorl %edi, %ebp
-; X32-NOCMOV-NEXT: andl %edx, %ebp
-; X32-NOCMOV-NEXT: xorl %edi, %ebp
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: movzwl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: xorw %si, %cx
+; X32-NOCMOV-NEXT: andl %eax, %ecx
+; X32-NOCMOV-NEXT: movl %ecx, (%esp) # 4-byte Spill
+; X32-NOCMOV-NEXT: movzwl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: xorw %dx, %cx
+; X32-NOCMOV-NEXT: andl %eax, %ecx
+; X32-NOCMOV-NEXT: movl %ecx, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: movzwl {{[0-9]+}}(%esp), %ebx
+; X32-NOCMOV-NEXT: xorw %cx, %bx
+; X32-NOCMOV-NEXT: andl %eax, %ebx
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: movzwl {{[0-9]+}}(%esp), %ebp
+; X32-NOCMOV-NEXT: xorw %cx, %bp
+; X32-NOCMOV-NEXT: andl %eax, %ebp
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: movzwl {{[0-9]+}}(%esp), %esi
+; X32-NOCMOV-NEXT: xorw %cx, %si
+; X32-NOCMOV-NEXT: andl %eax, %esi
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: movzwl {{[0-9]+}}(%esp), %edx
+; X32-NOCMOV-NEXT: xorw %cx, %dx
+; X32-NOCMOV-NEXT: andl %eax, %edx
+; X32-NOCMOV-NEXT: movzwl {{[0-9]+}}(%esp), %ecx
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edi
-; X32-NOCMOV-NEXT: xorl %eax, %edi
-; X32-NOCMOV-NEXT: andl %edx, %edi
-; X32-NOCMOV-NEXT: xorl %eax, %edi
+; X32-NOCMOV-NEXT: xorw %di, %cx
+; X32-NOCMOV-NEXT: andl %eax, %ecx
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
-; X32-NOCMOV-NEXT: xorl %eax, %ecx
-; X32-NOCMOV-NEXT: andl %edx, %ecx
-; X32-NOCMOV-NEXT: xorl %eax, %ecx
+; X32-NOCMOV-NEXT: xorl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Folded Spill
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
-; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %eax
-; X32-NOCMOV-NEXT: andl %edx, %eax
-; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %eax
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
-; X32-NOCMOV-NEXT: movl %eax, 28(%edx)
-; X32-NOCMOV-NEXT: movl %ecx, 24(%edx)
-; X32-NOCMOV-NEXT: movl %edi, 20(%edx)
-; X32-NOCMOV-NEXT: movl %ebp, 16(%edx)
-; X32-NOCMOV-NEXT: movl %ebx, 12(%edx)
-; X32-NOCMOV-NEXT: movl %esi, 8(%edx)
-; X32-NOCMOV-NEXT: movl (%esp), %eax # 4-byte Reload
-; X32-NOCMOV-NEXT: movl %eax, 4(%edx)
-; X32-NOCMOV-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 4-byte Reload
-; X32-NOCMOV-NEXT: movl %eax, (%edx)
-; X32-NOCMOV-NEXT: movl %edx, %eax
-; X32-NOCMOV-NEXT: addl $8, %esp
-; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 20
+; X32-NOCMOV-NEXT: xorl %eax, (%esp) # 4-byte Folded Spill
+; X32-NOCMOV-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %edi # 4-byte Reload
+; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %edi
+; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %ebx
+; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %ebp
+; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %esi
+; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %edx
+; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: movw %cx, 14(%eax)
+; X32-NOCMOV-NEXT: movw %dx, 12(%eax)
+; X32-NOCMOV-NEXT: movw %si, 10(%eax)
+; X32-NOCMOV-NEXT: movw %bp, 8(%eax)
+; X32-NOCMOV-NEXT: movw %bx, 6(%eax)
+; X32-NOCMOV-NEXT: movw %di, 4(%eax)
+; X32-NOCMOV-NEXT: movl (%esp), %ecx # 4-byte Reload
+; X32-NOCMOV-NEXT: movw %cx, 2(%eax)
+; X32-NOCMOV-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %ecx # 4-byte Reload
+; X32-NOCMOV-NEXT: movw %cx, (%eax)
+; X32-NOCMOV-NEXT: addl $12, %esp
; X32-NOCMOV-NEXT: popl %esi
-; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 16
; X32-NOCMOV-NEXT: popl %edi
-; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 12
; X32-NOCMOV-NEXT: popl %ebx
-; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 8
; X32-NOCMOV-NEXT: popl %ebp
-; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 4
; X32-NOCMOV-NEXT: retl $4
- %result = call <8 x i32> @llvm.ct.select.v8i32(i1 %cond, <8 x i32> %a, <8 x i32> %b)
- ret <8 x i32> %result
+ %result = call <8 x half> @llvm.ct.select.v8f16(i1 %cond, <8 x half> %a, <8 x half> %b)
+ ret <8 x half> %result
}
-define <8 x float> @test_ctselect_v8f32(i1 %cond, <8 x float> %a, <8 x float> %b) {
-; X64-LABEL: test_ctselect_v8f32:
+define <8 x bfloat> @test_ctselect_v8bf16(i1 %cond, <8 x bfloat> %a, <8 x bfloat> %b) #0 {
+; X64-LABEL: test_ctselect_v8bf16:
; X64: # %bb.0:
-; X64-NEXT: movd %edi, %xmm4
-; X64-NEXT: pshufd {{.*#+}} xmm4 = xmm4[0,0,0,0]
-; X64-NEXT: pslld $31, %xmm4
-; X64-NEXT: psrad $31, %xmm4
-; X64-NEXT: movdqa %xmm4, %xmm5
-; X64-NEXT: pandn %xmm2, %xmm5
-; X64-NEXT: pand %xmm4, %xmm0
-; X64-NEXT: por %xmm5, %xmm0
-; X64-NEXT: pand %xmm4, %xmm1
-; X64-NEXT: pandn %xmm3, %xmm4
-; X64-NEXT: por %xmm4, %xmm1
+; X64-NEXT: movq %xmm1, %rax
+; X64-NEXT: movq %xmm0, %rcx
+; X64-NEXT: punpckhqdq {{.*#+}} xmm1 = xmm1[1,1]
+; X64-NEXT: movq %xmm1, %rsi
+; X64-NEXT: punpckhqdq {{.*#+}} xmm0 = xmm0[1,1]
+; X64-NEXT: movq %xmm0, %r9
+; X64-NEXT: movq %r9, %rdx
+; X64-NEXT: movl %esi, %r8d
+; X64-NEXT: shrl $16, %r8d
+; X64-NEXT: movl %r9d, %r10d
+; X64-NEXT: shrl $16, %r10d
+; X64-NEXT: xorl %r8d, %r10d
+; X64-NEXT: andl $1, %edi
+; X64-NEXT: negl %edi
+; X64-NEXT: andl %edi, %r10d
+; X64-NEXT: xorl %r8d, %r10d
+; X64-NEXT: movq %r9, %r8
+; X64-NEXT: xorl %esi, %r9d
+; X64-NEXT: andl %edi, %r9d
+; X64-NEXT: shll $16, %r10d
+; X64-NEXT: xorl %esi, %r9d
+; X64-NEXT: movzwl %r9w, %r9d
+; X64-NEXT: orl %r10d, %r9d
+; X64-NEXT: movq %rsi, %r10
+; X64-NEXT: shrq $48, %r10
+; X64-NEXT: shrq $48, %r8
+; X64-NEXT: xorl %r10d, %r8d
+; X64-NEXT: andl %edi, %r8d
+; X64-NEXT: xorl %r10d, %r8d
+; X64-NEXT: shrq $32, %rsi
+; X64-NEXT: shrq $32, %rdx
+; X64-NEXT: xorl %esi, %edx
+; X64-NEXT: andl %edi, %edx
+; X64-NEXT: xorl %esi, %edx
+; X64-NEXT: movq %rcx, %rsi
+; X64-NEXT: shll $16, %r8d
+; X64-NEXT: movzwl %dx, %edx
+; X64-NEXT: orl %r8d, %edx
+; X64-NEXT: movl %eax, %r8d
+; X64-NEXT: shrl $16, %r8d
+; X64-NEXT: shlq $32, %rdx
+; X64-NEXT: orq %r9, %rdx
+; X64-NEXT: movl %ecx, %r9d
+; X64-NEXT: shrl $16, %r9d
+; X64-NEXT: xorl %r8d, %r9d
+; X64-NEXT: andl %edi, %r9d
+; X64-NEXT: xorl %r8d, %r9d
+; X64-NEXT: movq %rcx, %r8
+; X64-NEXT: xorl %eax, %ecx
+; X64-NEXT: andl %edi, %ecx
+; X64-NEXT: shll $16, %r9d
+; X64-NEXT: xorl %eax, %ecx
+; X64-NEXT: movzwl %cx, %ecx
+; X64-NEXT: orl %r9d, %ecx
+; X64-NEXT: movq %rax, %r9
+; X64-NEXT: shrq $32, %rax
+; X64-NEXT: shrq $32, %rsi
+; X64-NEXT: shrq $48, %r9
+; X64-NEXT: shrq $48, %r8
+; X64-NEXT: xorl %r9d, %r8d
+; X64-NEXT: andl %edi, %r8d
+; X64-NEXT: xorl %eax, %esi
+; X64-NEXT: andl %edi, %esi
+; X64-NEXT: xorl %r9d, %r8d
+; X64-NEXT: xorl %eax, %esi
+; X64-NEXT: shll $16, %r8d
+; X64-NEXT: movzwl %si, %eax
+; X64-NEXT: orl %r8d, %eax
+; X64-NEXT: shlq $32, %rax
+; X64-NEXT: orq %rcx, %rax
+; X64-NEXT: movq %rax, %xmm0
+; X64-NEXT: movq %rdx, %xmm1
+; X64-NEXT: punpcklqdq {{.*#+}} xmm0 = xmm0[0],xmm1[0]
; X64-NEXT: retq
;
-; X32-LABEL: test_ctselect_v8f32:
+; X32-LABEL: test_ctselect_v8bf16:
; X32: # %bb.0:
; X32-NEXT: pushl %ebp
-; X32-NEXT: .cfi_def_cfa_offset 8
; X32-NEXT: pushl %ebx
-; X32-NEXT: .cfi_def_cfa_offset 12
; X32-NEXT: pushl %edi
-; X32-NEXT: .cfi_def_cfa_offset 16
; X32-NEXT: pushl %esi
-; X32-NEXT: .cfi_def_cfa_offset 20
-; X32-NEXT: subl $8, %esp
-; X32-NEXT: .cfi_def_cfa_offset 28
-; X32-NEXT: .cfi_offset %esi, -20
-; X32-NEXT: .cfi_offset %edi, -16
-; X32-NEXT: .cfi_offset %ebx, -12
-; X32-NEXT: .cfi_offset %ebp, -8
-; X32-NEXT: movl {{[0-9]+}}(%esp), %edi
-; X32-NEXT: movl {{[0-9]+}}(%esp), %ebp
-; X32-NEXT: movl {{[0-9]+}}(%esp), %ebx
-; X32-NEXT: movl {{[0-9]+}}(%esp), %esi
-; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
-; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
-; X32-NEXT: xorl %eax, %ecx
-; X32-NEXT: movzbl {{[0-9]+}}(%esp), %edx
-; X32-NEXT: andl $1, %edx
-; X32-NEXT: negl %edx
-; X32-NEXT: andl %edx, %ecx
-; X32-NEXT: xorl %eax, %ecx
+; X32-NEXT: subl $60, %esp
+; X32-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NEXT: fstps (%esp)
+; X32-NEXT: calll __truncsfbf2
+; X32-NEXT: # kill: def $ax killed $ax def $eax
+; X32-NEXT: movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
+; X32-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NEXT: fstps (%esp)
+; X32-NEXT: calll __truncsfbf2
+; X32-NEXT: movl %eax, %ebp
+; X32-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NEXT: fstps (%esp)
+; X32-NEXT: calll __truncsfbf2
+; X32-NEXT: # kill: def $ax killed $ax def $eax
+; X32-NEXT: movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
+; X32-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NEXT: fstps (%esp)
+; X32-NEXT: calll __truncsfbf2
+; X32-NEXT: # kill: def $ax killed $ax def $eax
+; X32-NEXT: movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
+; X32-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NEXT: fstps (%esp)
+; X32-NEXT: calll __truncsfbf2
+; X32-NEXT: # kill: def $ax killed $ax def $eax
+; X32-NEXT: movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
+; X32-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NEXT: fstps (%esp)
+; X32-NEXT: calll __truncsfbf2
+; X32-NEXT: # kill: def $ax killed $ax def $eax
+; X32-NEXT: movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
+; X32-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NEXT: fstps (%esp)
+; X32-NEXT: calll __truncsfbf2
+; X32-NEXT: # kill: def $ax killed $ax def $eax
+; X32-NEXT: movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
+; X32-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NEXT: fstps (%esp)
+; X32-NEXT: calll __truncsfbf2
+; X32-NEXT: # kill: def $ax killed $ax def $eax
+; X32-NEXT: movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
+; X32-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NEXT: fstps (%esp)
+; X32-NEXT: calll __truncsfbf2
+; X32-NEXT: # kill: def $ax killed $ax def $eax
+; X32-NEXT: movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
+; X32-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NEXT: fstps (%esp)
+; X32-NEXT: calll __truncsfbf2
+; X32-NEXT: # kill: def $ax killed $ax def $eax
+; X32-NEXT: movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
+; X32-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NEXT: fstps (%esp)
+; X32-NEXT: calll __truncsfbf2
+; X32-NEXT: # kill: def $ax killed $ax def $eax
+; X32-NEXT: movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
+; X32-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NEXT: fstps (%esp)
+; X32-NEXT: calll __truncsfbf2
+; X32-NEXT: movl %eax, %edi
+; X32-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NEXT: fstps (%esp)
+; X32-NEXT: calll __truncsfbf2
+; X32-NEXT: # kill: def $ax killed $ax def $eax
+; X32-NEXT: movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
+; X32-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NEXT: fstps (%esp)
+; X32-NEXT: calll __truncsfbf2
+; X32-NEXT: movl %eax, %esi
+; X32-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NEXT: fstps (%esp)
+; X32-NEXT: calll __truncsfbf2
+; X32-NEXT: # kill: def $ax killed $ax def $eax
+; X32-NEXT: movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
+; X32-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NEXT: fstps (%esp)
+; X32-NEXT: calll __truncsfbf2
+; X32-NEXT: # kill: def $ax killed $ax def $eax
+; X32-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: andb $1, %cl
+; X32-NEXT: xorl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 4-byte Folded Reload
+; X32-NEXT: movzbl %cl, %ecx
+; X32-NEXT: negl %ecx
+; X32-NEXT: andl %ecx, %eax
+; X32-NEXT: movl %eax, %ebx
+; X32-NEXT: movl %ebp, %eax
+; X32-NEXT: xorl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 4-byte Folded Reload
+; X32-NEXT: andl %ecx, %eax
+; X32-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %ebp # 4-byte Reload
+; X32-NEXT: xorl {{[-0-9]+}}(%e{{[sb]}}p), %ebp # 4-byte Folded Reload
+; X32-NEXT: andl %ecx, %ebp
+; X32-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %edx # 4-byte Reload
+; X32-NEXT: xorl {{[-0-9]+}}(%e{{[sb]}}p), %edx # 4-byte Folded Reload
+; X32-NEXT: andl %ecx, %edx
+; X32-NEXT: movl %edx, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
+; X32-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %edx # 4-byte Reload
+; X32-NEXT: xorl {{[-0-9]+}}(%e{{[sb]}}p), %edx # 4-byte Folded Reload
+; X32-NEXT: andl %ecx, %edx
+; X32-NEXT: movl %edx, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
+; X32-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %edx # 4-byte Reload
+; X32-NEXT: xorl {{[-0-9]+}}(%e{{[sb]}}p), %edx # 4-byte Folded Reload
+; X32-NEXT: andl %ecx, %edx
+; X32-NEXT: xorl {{[-0-9]+}}(%e{{[sb]}}p), %edi # 4-byte Folded Reload
+; X32-NEXT: andl %ecx, %edi
+; X32-NEXT: movl %edi, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
+; X32-NEXT: xorl {{[-0-9]+}}(%e{{[sb]}}p), %esi # 4-byte Folded Reload
+; X32-NEXT: andl %ecx, %esi
+; X32-NEXT: movl %esi, %ecx
+; X32-NEXT: xorl {{[-0-9]+}}(%e{{[sb]}}p), %ebx # 4-byte Folded Reload
+; X32-NEXT: movl %ebx, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
+; X32-NEXT: xorl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 4-byte Folded Reload
+; X32-NEXT: movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
+; X32-NEXT: xorl {{[-0-9]+}}(%e{{[sb]}}p), %ebp # 4-byte Folded Reload
+; X32-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %edi # 4-byte Reload
+; X32-NEXT: xorl {{[-0-9]+}}(%e{{[sb]}}p), %edi # 4-byte Folded Reload
+; X32-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %esi # 4-byte Reload
+; X32-NEXT: xorl {{[-0-9]+}}(%e{{[sb]}}p), %esi # 4-byte Folded Reload
+; X32-NEXT: xorl {{[-0-9]+}}(%e{{[sb]}}p), %edx # 4-byte Folded Reload
+; X32-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 4-byte Reload
+; X32-NEXT: xorl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 4-byte Folded Reload
+; X32-NEXT: xorl {{[-0-9]+}}(%e{{[sb]}}p), %ecx # 4-byte Folded Reload
; X32-NEXT: movl %ecx, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
-; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
-; X32-NEXT: xorl %esi, %eax
-; X32-NEXT: andl %edx, %eax
-; X32-NEXT: xorl %esi, %eax
-; X32-NEXT: movl %eax, (%esp) # 4-byte Spill
-; X32-NEXT: movl {{[0-9]+}}(%esp), %esi
-; X32-NEXT: xorl %ebx, %esi
-; X32-NEXT: andl %edx, %esi
-; X32-NEXT: xorl %ebx, %esi
-; X32-NEXT: movl {{[0-9]+}}(%esp), %ebx
-; X32-NEXT: xorl %ebp, %ebx
-; X32-NEXT: andl %edx, %ebx
-; X32-NEXT: xorl %ebp, %ebx
-; X32-NEXT: movl {{[0-9]+}}(%esp), %ebp
-; X32-NEXT: xorl %edi, %ebp
-; X32-NEXT: andl %edx, %ebp
-; X32-NEXT: xorl %edi, %ebp
-; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
-; X32-NEXT: movl {{[0-9]+}}(%esp), %edi
-; X32-NEXT: xorl %eax, %edi
-; X32-NEXT: andl %edx, %edi
-; X32-NEXT: xorl %eax, %edi
-; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
-; X32-NEXT: xorl %eax, %ecx
-; X32-NEXT: andl %edx, %ecx
-; X32-NEXT: xorl %eax, %ecx
-; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
-; X32-NEXT: xorl {{[0-9]+}}(%esp), %eax
-; X32-NEXT: andl %edx, %eax
-; X32-NEXT: xorl {{[0-9]+}}(%esp), %eax
-; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
-; X32-NEXT: movl %eax, 28(%edx)
-; X32-NEXT: movl %ecx, 24(%edx)
-; X32-NEXT: movl %edi, 20(%edx)
-; X32-NEXT: movl %ebp, 16(%edx)
-; X32-NEXT: movl %ebx, 12(%edx)
-; X32-NEXT: movl %esi, 8(%edx)
-; X32-NEXT: movl (%esp), %eax # 4-byte Reload
-; X32-NEXT: movl %eax, 4(%edx)
+; X32-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %ebx # 4-byte Reload
+; X32-NEXT: movw %bx, 14(%ecx)
+; X32-NEXT: movw %ax, 12(%ecx)
+; X32-NEXT: movw %dx, 10(%ecx)
+; X32-NEXT: movw %si, 8(%ecx)
+; X32-NEXT: movw %di, 6(%ecx)
+; X32-NEXT: movw %bp, 4(%ecx)
; X32-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 4-byte Reload
-; X32-NEXT: movl %eax, (%edx)
-; X32-NEXT: movl %edx, %eax
-; X32-NEXT: addl $8, %esp
-; X32-NEXT: .cfi_def_cfa_offset 20
+; X32-NEXT: movw %ax, 2(%ecx)
+; X32-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 4-byte Reload
+; X32-NEXT: movw %ax, (%ecx)
+; X32-NEXT: movl %ecx, %eax
+; X32-NEXT: addl $60, %esp
; X32-NEXT: popl %esi
-; X32-NEXT: .cfi_def_cfa_offset 16
; X32-NEXT: popl %edi
-; X32-NEXT: .cfi_def_cfa_offset 12
; X32-NEXT: popl %ebx
-; X32-NEXT: .cfi_def_cfa_offset 8
; X32-NEXT: popl %ebp
-; X32-NEXT: .cfi_def_cfa_offset 4
; X32-NEXT: retl $4
;
-; X32-NOCMOV-LABEL: test_ctselect_v8f32:
+; X32-NOCMOV-LABEL: test_ctselect_v8bf16:
; X32-NOCMOV: # %bb.0:
; X32-NOCMOV-NEXT: pushl %ebp
-; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 8
; X32-NOCMOV-NEXT: pushl %ebx
-; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 12
; X32-NOCMOV-NEXT: pushl %edi
-; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 16
; X32-NOCMOV-NEXT: pushl %esi
-; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 20
-; X32-NOCMOV-NEXT: subl $8, %esp
-; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 28
-; X32-NOCMOV-NEXT: .cfi_offset %esi, -20
-; X32-NOCMOV-NEXT: .cfi_offset %edi, -16
-; X32-NOCMOV-NEXT: .cfi_offset %ebx, -12
-; X32-NOCMOV-NEXT: .cfi_offset %ebp, -8
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edi
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebp
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebx
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %esi
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
-; X32-NOCMOV-NEXT: xorl %eax, %ecx
-; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %edx
-; X32-NOCMOV-NEXT: andl $1, %edx
-; X32-NOCMOV-NEXT: negl %edx
-; X32-NOCMOV-NEXT: andl %edx, %ecx
-; X32-NOCMOV-NEXT: xorl %eax, %ecx
+; X32-NOCMOV-NEXT: subl $60, %esp
+; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: fstps (%esp)
+; X32-NOCMOV-NEXT: calll __truncsfbf2
+; X32-NOCMOV-NEXT: # kill: def $ax killed $ax def $eax
+; X32-NOCMOV-NEXT: movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
+; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: fstps (%esp)
+; X32-NOCMOV-NEXT: calll __truncsfbf2
+; X32-NOCMOV-NEXT: movl %eax, %ebp
+; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: fstps (%esp)
+; X32-NOCMOV-NEXT: calll __truncsfbf2
+; X32-NOCMOV-NEXT: # kill: def $ax killed $ax def $eax
+; X32-NOCMOV-NEXT: movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
+; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: fstps (%esp)
+; X32-NOCMOV-NEXT: calll __truncsfbf2
+; X32-NOCMOV-NEXT: # kill: def $ax killed $ax def $eax
+; X32-NOCMOV-NEXT: movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
+; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: fstps (%esp)
+; X32-NOCMOV-NEXT: calll __truncsfbf2
+; X32-NOCMOV-NEXT: # kill: def $ax killed $ax def $eax
+; X32-NOCMOV-NEXT: movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
+; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: fstps (%esp)
+; X32-NOCMOV-NEXT: calll __truncsfbf2
+; X32-NOCMOV-NEXT: # kill: def $ax killed $ax def $eax
+; X32-NOCMOV-NEXT: movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
+; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: fstps (%esp)
+; X32-NOCMOV-NEXT: calll __truncsfbf2
+; X32-NOCMOV-NEXT: # kill: def $ax killed $ax def $eax
+; X32-NOCMOV-NEXT: movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
+; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: fstps (%esp)
+; X32-NOCMOV-NEXT: calll __truncsfbf2
+; X32-NOCMOV-NEXT: # kill: def $ax killed $ax def $eax
+; X32-NOCMOV-NEXT: movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
+; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: fstps (%esp)
+; X32-NOCMOV-NEXT: calll __truncsfbf2
+; X32-NOCMOV-NEXT: # kill: def $ax killed $ax def $eax
+; X32-NOCMOV-NEXT: movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
+; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: fstps (%esp)
+; X32-NOCMOV-NEXT: calll __truncsfbf2
+; X32-NOCMOV-NEXT: # kill: def $ax killed $ax def $eax
+; X32-NOCMOV-NEXT: movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
+; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: fstps (%esp)
+; X32-NOCMOV-NEXT: calll __truncsfbf2
+; X32-NOCMOV-NEXT: # kill: def $ax killed $ax def $eax
+; X32-NOCMOV-NEXT: movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
+; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: fstps (%esp)
+; X32-NOCMOV-NEXT: calll __truncsfbf2
+; X32-NOCMOV-NEXT: movl %eax, %edi
+; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: fstps (%esp)
+; X32-NOCMOV-NEXT: calll __truncsfbf2
+; X32-NOCMOV-NEXT: # kill: def $ax killed $ax def $eax
+; X32-NOCMOV-NEXT: movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
+; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: fstps (%esp)
+; X32-NOCMOV-NEXT: calll __truncsfbf2
+; X32-NOCMOV-NEXT: movl %eax, %esi
+; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: fstps (%esp)
+; X32-NOCMOV-NEXT: calll __truncsfbf2
+; X32-NOCMOV-NEXT: # kill: def $ax killed $ax def $eax
+; X32-NOCMOV-NEXT: movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
+; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: fstps (%esp)
+; X32-NOCMOV-NEXT: calll __truncsfbf2
+; X32-NOCMOV-NEXT: # kill: def $ax killed $ax def $eax
+; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: andb $1, %cl
+; X32-NOCMOV-NEXT: xorl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 4-byte Folded Reload
+; X32-NOCMOV-NEXT: movzbl %cl, %ecx
+; X32-NOCMOV-NEXT: negl %ecx
+; X32-NOCMOV-NEXT: andl %ecx, %eax
+; X32-NOCMOV-NEXT: movl %eax, %ebx
+; X32-NOCMOV-NEXT: movl %ebp, %eax
+; X32-NOCMOV-NEXT: xorl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 4-byte Folded Reload
+; X32-NOCMOV-NEXT: andl %ecx, %eax
+; X32-NOCMOV-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %ebp # 4-byte Reload
+; X32-NOCMOV-NEXT: xorl {{[-0-9]+}}(%e{{[sb]}}p), %ebp # 4-byte Folded Reload
+; X32-NOCMOV-NEXT: andl %ecx, %ebp
+; X32-NOCMOV-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %edx # 4-byte Reload
+; X32-NOCMOV-NEXT: xorl {{[-0-9]+}}(%e{{[sb]}}p), %edx # 4-byte Folded Reload
+; X32-NOCMOV-NEXT: andl %ecx, %edx
+; X32-NOCMOV-NEXT: movl %edx, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
+; X32-NOCMOV-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %edx # 4-byte Reload
+; X32-NOCMOV-NEXT: xorl {{[-0-9]+}}(%e{{[sb]}}p), %edx # 4-byte Folded Reload
+; X32-NOCMOV-NEXT: andl %ecx, %edx
+; X32-NOCMOV-NEXT: movl %edx, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
+; X32-NOCMOV-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %edx # 4-byte Reload
+; X32-NOCMOV-NEXT: xorl {{[-0-9]+}}(%e{{[sb]}}p), %edx # 4-byte Folded Reload
+; X32-NOCMOV-NEXT: andl %ecx, %edx
+; X32-NOCMOV-NEXT: xorl {{[-0-9]+}}(%e{{[sb]}}p), %edi # 4-byte Folded Reload
+; X32-NOCMOV-NEXT: andl %ecx, %edi
+; X32-NOCMOV-NEXT: movl %edi, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
+; X32-NOCMOV-NEXT: xorl {{[-0-9]+}}(%e{{[sb]}}p), %esi # 4-byte Folded Reload
+; X32-NOCMOV-NEXT: andl %ecx, %esi
+; X32-NOCMOV-NEXT: movl %esi, %ecx
+; X32-NOCMOV-NEXT: xorl {{[-0-9]+}}(%e{{[sb]}}p), %ebx # 4-byte Folded Reload
+; X32-NOCMOV-NEXT: movl %ebx, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
+; X32-NOCMOV-NEXT: xorl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 4-byte Folded Reload
+; X32-NOCMOV-NEXT: movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
+; X32-NOCMOV-NEXT: xorl {{[-0-9]+}}(%e{{[sb]}}p), %ebp # 4-byte Folded Reload
+; X32-NOCMOV-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %edi # 4-byte Reload
+; X32-NOCMOV-NEXT: xorl {{[-0-9]+}}(%e{{[sb]}}p), %edi # 4-byte Folded Reload
+; X32-NOCMOV-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %esi # 4-byte Reload
+; X32-NOCMOV-NEXT: xorl {{[-0-9]+}}(%e{{[sb]}}p), %esi # 4-byte Folded Reload
+; X32-NOCMOV-NEXT: xorl {{[-0-9]+}}(%e{{[sb]}}p), %edx # 4-byte Folded Reload
+; X32-NOCMOV-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 4-byte Reload
+; X32-NOCMOV-NEXT: xorl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 4-byte Folded Reload
+; X32-NOCMOV-NEXT: xorl {{[-0-9]+}}(%e{{[sb]}}p), %ecx # 4-byte Folded Reload
; X32-NOCMOV-NEXT: movl %ecx, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
-; X32-NOCMOV-NEXT: xorl %esi, %eax
-; X32-NOCMOV-NEXT: andl %edx, %eax
-; X32-NOCMOV-NEXT: xorl %esi, %eax
-; X32-NOCMOV-NEXT: movl %eax, (%esp) # 4-byte Spill
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %esi
-; X32-NOCMOV-NEXT: xorl %ebx, %esi
-; X32-NOCMOV-NEXT: andl %edx, %esi
-; X32-NOCMOV-NEXT: xorl %ebx, %esi
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebx
-; X32-NOCMOV-NEXT: xorl %ebp, %ebx
-; X32-NOCMOV-NEXT: andl %edx, %ebx
-; X32-NOCMOV-NEXT: xorl %ebp, %ebx
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebp
-; X32-NOCMOV-NEXT: xorl %edi, %ebp
-; X32-NOCMOV-NEXT: andl %edx, %ebp
-; X32-NOCMOV-NEXT: xorl %edi, %ebp
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edi
-; X32-NOCMOV-NEXT: xorl %eax, %edi
-; X32-NOCMOV-NEXT: andl %edx, %edi
-; X32-NOCMOV-NEXT: xorl %eax, %edi
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
-; X32-NOCMOV-NEXT: xorl %eax, %ecx
-; X32-NOCMOV-NEXT: andl %edx, %ecx
-; X32-NOCMOV-NEXT: xorl %eax, %ecx
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
-; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %eax
-; X32-NOCMOV-NEXT: andl %edx, %eax
-; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %eax
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
-; X32-NOCMOV-NEXT: movl %eax, 28(%edx)
-; X32-NOCMOV-NEXT: movl %ecx, 24(%edx)
-; X32-NOCMOV-NEXT: movl %edi, 20(%edx)
-; X32-NOCMOV-NEXT: movl %ebp, 16(%edx)
-; X32-NOCMOV-NEXT: movl %ebx, 12(%edx)
-; X32-NOCMOV-NEXT: movl %esi, 8(%edx)
-; X32-NOCMOV-NEXT: movl (%esp), %eax # 4-byte Reload
-; X32-NOCMOV-NEXT: movl %eax, 4(%edx)
+; X32-NOCMOV-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %ebx # 4-byte Reload
+; X32-NOCMOV-NEXT: movw %bx, 14(%ecx)
+; X32-NOCMOV-NEXT: movw %ax, 12(%ecx)
+; X32-NOCMOV-NEXT: movw %dx, 10(%ecx)
+; X32-NOCMOV-NEXT: movw %si, 8(%ecx)
+; X32-NOCMOV-NEXT: movw %di, 6(%ecx)
+; X32-NOCMOV-NEXT: movw %bp, 4(%ecx)
; X32-NOCMOV-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 4-byte Reload
-; X32-NOCMOV-NEXT: movl %eax, (%edx)
-; X32-NOCMOV-NEXT: movl %edx, %eax
-; X32-NOCMOV-NEXT: addl $8, %esp
-; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 20
+; X32-NOCMOV-NEXT: movw %ax, 2(%ecx)
+; X32-NOCMOV-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 4-byte Reload
+; X32-NOCMOV-NEXT: movw %ax, (%ecx)
+; X32-NOCMOV-NEXT: movl %ecx, %eax
+; X32-NOCMOV-NEXT: addl $60, %esp
; X32-NOCMOV-NEXT: popl %esi
-; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 16
; X32-NOCMOV-NEXT: popl %edi
-; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 12
; X32-NOCMOV-NEXT: popl %ebx
-; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 8
; X32-NOCMOV-NEXT: popl %ebp
-; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 4
; X32-NOCMOV-NEXT: retl $4
- %result = call <8 x float> @llvm.ct.select.v8f32(i1 %cond, <8 x float> %a, <8 x float> %b)
- ret <8 x float> %result
+ %result = call <8 x bfloat> @llvm.ct.select.v8bf16(i1 %cond, <8 x bfloat> %a, <8 x bfloat> %b)
+ ret <8 x bfloat> %result
}
-define float @test_ctselect_f32_nan_inf(i1 %cond) {
+define float @test_ctselect_f32_nan_inf(i1 %cond) #0 {
; X64-LABEL: test_ctselect_f32_nan_inf:
; X64: # %bb.0:
; X64-NEXT: andl $1, %edi
; X64-NEXT: negl %edi
; X64-NEXT: andl $4194304, %edi # imm = 0x400000
-; X64-NEXT: orl $2139095040, %edi # imm = 0x7F800000
+; X64-NEXT: xorl $2139095040, %edi # imm = 0x7F800000
; X64-NEXT: movd %edi, %xmm0
; X64-NEXT: retq
;
; X32-LABEL: test_ctselect_f32_nan_inf:
; X32: # %bb.0:
; X32-NEXT: pushl %eax
-; X32-NEXT: .cfi_def_cfa_offset 8
; X32-NEXT: movzbl {{[0-9]+}}(%esp), %eax
; X32-NEXT: andb $1, %al
; X32-NEXT: movzbl %al, %eax
; X32-NEXT: negl %eax
; X32-NEXT: andl $4194304, %eax # imm = 0x400000
-; X32-NEXT: orl $2139095040, %eax # imm = 0x7F800000
+; X32-NEXT: xorl $2139095040, %eax # imm = 0x7F800000
; X32-NEXT: movl %eax, (%esp)
; X32-NEXT: flds (%esp)
; X32-NEXT: popl %eax
-; X32-NEXT: .cfi_def_cfa_offset 4
; X32-NEXT: retl
;
; X32-NOCMOV-LABEL: test_ctselect_f32_nan_inf:
; X32-NOCMOV: # %bb.0:
; X32-NOCMOV-NEXT: pushl %eax
-; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 8
; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %eax
; X32-NOCMOV-NEXT: andb $1, %al
; X32-NOCMOV-NEXT: movzbl %al, %eax
; X32-NOCMOV-NEXT: negl %eax
; X32-NOCMOV-NEXT: andl $4194304, %eax # imm = 0x400000
-; X32-NOCMOV-NEXT: orl $2139095040, %eax # imm = 0x7F800000
+; X32-NOCMOV-NEXT: xorl $2139095040, %eax # imm = 0x7F800000
; X32-NOCMOV-NEXT: movl %eax, (%esp)
; X32-NOCMOV-NEXT: flds (%esp)
; X32-NOCMOV-NEXT: popl %eax
-; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 4
; X32-NOCMOV-NEXT: retl
%result = call float @llvm.ct.select.f32(i1 %cond, float 0x7FF8000000000000, float 0x7FF0000000000000)
ret float %result
}
-define double @test_ctselect_f64_nan_inf(i1 %cond) {
+define double @test_ctselect_f64_nan_inf(i1 %cond) #0 {
; X64-LABEL: test_ctselect_f64_nan_inf:
; X64: # %bb.0:
; X64-NEXT: # kill: def $edi killed $edi def $rdi
@@ -1478,40 +2408,66 @@ define double @test_ctselect_f64_nan_inf(i1 %cond) {
; X64-NEXT: movabsq $2251799813685248, %rax # imm = 0x8000000000000
; X64-NEXT: andq %rdi, %rax
; X64-NEXT: movabsq $9218868437227405312, %rcx # imm = 0x7FF0000000000000
-; X64-NEXT: orq %rax, %rcx
+; X64-NEXT: xorq %rax, %rcx
; X64-NEXT: movq %rcx, %xmm0
; X64-NEXT: retq
;
; X32-LABEL: test_ctselect_f64_nan_inf:
; X32: # %bb.0:
-; X32-NEXT: subl $12, %esp
-; X32-NEXT: .cfi_def_cfa_offset 16
+; X32-NEXT: pushl %esi
+; X32-NEXT: subl $24, %esp
; X32-NEXT: movzbl {{[0-9]+}}(%esp), %eax
-; X32-NEXT: andl $1, %eax
+; X32-NEXT: andb $1, %al
+; X32-NEXT: flds {{\.?LCPI[0-9]+_[0-9]+}}
+; X32-NEXT: fstpl {{[0-9]+}}(%esp)
+; X32-NEXT: flds {{\.?LCPI[0-9]+_[0-9]+}}
+; X32-NEXT: fstpl {{[0-9]+}}(%esp)
+; X32-NEXT: movzbl %al, %eax
; X32-NEXT: negl %eax
-; X32-NEXT: andl $524288, %eax # imm = 0x80000
-; X32-NEXT: orl $2146435072, %eax # imm = 0x7FF00000
-; X32-NEXT: movl %eax, {{[0-9]+}}(%esp)
-; X32-NEXT: movl $0, (%esp)
+; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: movl {{[0-9]+}}(%esp), %esi
+; X32-NEXT: xorl %edx, %esi
+; X32-NEXT: andl %eax, %esi
+; X32-NEXT: xorl %edx, %esi
+; X32-NEXT: movl %esi, (%esp)
+; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NEXT: xorl %ecx, %edx
+; X32-NEXT: andl %eax, %edx
+; X32-NEXT: xorl %ecx, %edx
+; X32-NEXT: movl %edx, {{[0-9]+}}(%esp)
; X32-NEXT: fldl (%esp)
-; X32-NEXT: addl $12, %esp
-; X32-NEXT: .cfi_def_cfa_offset 4
+; X32-NEXT: addl $24, %esp
+; X32-NEXT: popl %esi
; X32-NEXT: retl
;
; X32-NOCMOV-LABEL: test_ctselect_f64_nan_inf:
; X32-NOCMOV: # %bb.0:
-; X32-NOCMOV-NEXT: subl $12, %esp
-; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 16
+; X32-NOCMOV-NEXT: pushl %esi
+; X32-NOCMOV-NEXT: subl $24, %esp
; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %eax
-; X32-NOCMOV-NEXT: andl $1, %eax
+; X32-NOCMOV-NEXT: andb $1, %al
+; X32-NOCMOV-NEXT: flds {{\.?LCPI[0-9]+_[0-9]+}}
+; X32-NOCMOV-NEXT: fstpl {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: flds {{\.?LCPI[0-9]+_[0-9]+}}
+; X32-NOCMOV-NEXT: fstpl {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: movzbl %al, %eax
; X32-NOCMOV-NEXT: negl %eax
-; X32-NOCMOV-NEXT: andl $524288, %eax # imm = 0x80000
-; X32-NOCMOV-NEXT: orl $2146435072, %eax # imm = 0x7FF00000
-; X32-NOCMOV-NEXT: movl %eax, {{[0-9]+}}(%esp)
-; X32-NOCMOV-NEXT: movl $0, (%esp)
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %esi
+; X32-NOCMOV-NEXT: xorl %edx, %esi
+; X32-NOCMOV-NEXT: andl %eax, %esi
+; X32-NOCMOV-NEXT: xorl %edx, %esi
+; X32-NOCMOV-NEXT: movl %esi, (%esp)
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NOCMOV-NEXT: xorl %ecx, %edx
+; X32-NOCMOV-NEXT: andl %eax, %edx
+; X32-NOCMOV-NEXT: xorl %ecx, %edx
+; X32-NOCMOV-NEXT: movl %edx, {{[0-9]+}}(%esp)
; X32-NOCMOV-NEXT: fldl (%esp)
-; X32-NOCMOV-NEXT: addl $12, %esp
-; X32-NOCMOV-NEXT: .cfi_def_cfa_offset 4
+; X32-NOCMOV-NEXT: addl $24, %esp
+; X32-NOCMOV-NEXT: popl %esi
; X32-NOCMOV-NEXT: retl
%result = call double @llvm.ct.select.f64(i1 %cond, double 0x7FF8000000000000, double 0x7FF0000000000000)
ret double %result
@@ -1525,6 +2481,10 @@ declare i32 @llvm.ct.select.i32(i1, i32, i32)
declare i64 @llvm.ct.select.i64(i1, i64, i64)
declare float @llvm.ct.select.f32(i1, float, float)
declare double @llvm.ct.select.f64(i1, double, double)
+declare half @llvm.ct.select.f16(i1, half, half)
+declare bfloat @llvm.ct.select.bf16(i1, bfloat, bfloat)
+declare fp128 @llvm.ct.select.f128(i1, fp128, fp128)
+declare x86_fp80 @llvm.ct.select.f80(i1, x86_fp80, x86_fp80)
declare ptr @llvm.ct.select.p0(i1, ptr, ptr)
; Vector intrinsics
@@ -1536,3 +2496,7 @@ declare <4 x float> @llvm.ct.select.v4f32(i1, <4 x float>, <4 x float>)
declare <2 x double> @llvm.ct.select.v2f64(i1, <2 x double>, <2 x double>)
declare <8 x i32> @llvm.ct.select.v8i32(i1, <8 x i32>, <8 x i32>)
declare <8 x float> @llvm.ct.select.v8f32(i1, <8 x float>, <8 x float>)
+declare <8 x half> @llvm.ct.select.v8f16(i1, <8 x half>, <8 x half>)
+declare <8 x bfloat> @llvm.ct.select.v8bf16(i1, <8 x bfloat>, <8 x bfloat>)
+
+attributes #0 = { nounwind }
diff --git a/llvm/test/Transforms/InstSimplify/ct-select.ll b/llvm/test/Transforms/InstSimplify/ct-select.ll
new file mode 100644
index 0000000000000..3a89009be8480
--- /dev/null
+++ b/llvm/test/Transforms/InstSimplify/ct-select.ll
@@ -0,0 +1,102 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py
+; RUN: opt < %s -passes=instsimplify -S | FileCheck %s
+
+declare i32 @llvm.ct.select.i32(i1, i32, i32)
+declare i64 @llvm.ct.select.i64(i1, i64, i64)
+declare float @llvm.ct.select.f32(i1, float, float)
+declare ptr @llvm.ct.select.p0(i1, ptr, ptr)
+declare <4 x i32> @llvm.ct.select.v4i32(i1, <4 x i32>, <4 x i32>)
+
+define i32 @ct_select_true(i32 %x, i32 %y) {
+; CHECK-LABEL: @ct_select_true(
+; CHECK-NEXT: [[R:%.*]] = call i32 @llvm.ct.select.i32(i1 true, i32 [[X:%.*]], i32 [[Y:%.*]])
+; CHECK-NEXT: ret i32 [[X]]
+;
+ %r = call i32 @llvm.ct.select.i32(i1 true, i32 %x, i32 %y)
+ ret i32 %r
+}
+
+define i32 @ct_select_false(i32 %x, i32 %y) {
+; CHECK-LABEL: @ct_select_false(
+; CHECK-NEXT: [[R:%.*]] = call i32 @llvm.ct.select.i32(i1 false, i32 [[X:%.*]], i32 [[Y:%.*]])
+; CHECK-NEXT: ret i32 [[Y]]
+;
+ %r = call i32 @llvm.ct.select.i32(i1 false, i32 %x, i32 %y)
+ ret i32 %r
+}
+
+define i64 @ct_select_true_i64(i64 %x, i64 %y) {
+; CHECK-LABEL: @ct_select_true_i64(
+; CHECK-NEXT: [[R:%.*]] = call i64 @llvm.ct.select.i64(i1 true, i64 [[X:%.*]], i64 [[Y:%.*]])
+; CHECK-NEXT: ret i64 [[X]]
+;
+ %r = call i64 @llvm.ct.select.i64(i1 true, i64 %x, i64 %y)
+ ret i64 %r
+}
+
+define float @ct_select_false_f32(float %x, float %y) {
+; CHECK-LABEL: @ct_select_false_f32(
+; CHECK-NEXT: [[R:%.*]] = call float @llvm.ct.select.f32(i1 false, float [[X:%.*]], float [[Y:%.*]])
+; CHECK-NEXT: ret float [[Y]]
+;
+ %r = call float @llvm.ct.select.f32(i1 false, float %x, float %y)
+ ret float %r
+}
+
+define ptr @ct_select_true_ptr(ptr %x, ptr %y) {
+; CHECK-LABEL: @ct_select_true_ptr(
+; CHECK-NEXT: [[R:%.*]] = call ptr @llvm.ct.select.p0(i1 true, ptr [[X:%.*]], ptr [[Y:%.*]])
+; CHECK-NEXT: ret ptr [[X]]
+;
+ %r = call ptr @llvm.ct.select.p0(i1 true, ptr %x, ptr %y)
+ ret ptr %r
+}
+
+define <4 x i32> @ct_select_true_v4i32(<4 x i32> %x, <4 x i32> %y) {
+; CHECK-LABEL: @ct_select_true_v4i32(
+; CHECK-NEXT: [[R:%.*]] = call <4 x i32> @llvm.ct.select.v4i32(i1 true, <4 x i32> [[X:%.*]], <4 x i32> [[Y:%.*]])
+; CHECK-NEXT: ret <4 x i32> [[X]]
+;
+ %r = call <4 x i32> @llvm.ct.select.v4i32(i1 true, <4 x i32> %x, <4 x i32> %y)
+ ret <4 x i32> %r
+}
+
+define i32 @ct_select_same_arms(i1 %c, i32 %x) {
+; CHECK-LABEL: @ct_select_same_arms(
+; CHECK-NEXT: [[R:%.*]] = call i32 @llvm.ct.select.i32(i1 [[C:%.*]], i32 [[X:%.*]], i32 [[X]])
+; CHECK-NEXT: ret i32 [[X]]
+;
+ %r = call i32 @llvm.ct.select.i32(i1 %c, i32 %x, i32 %x)
+ ret i32 %r
+}
+
+define <4 x i32> @ct_select_same_arms_vec(i1 %c, <4 x i32> %x) {
+; CHECK-LABEL: @ct_select_same_arms_vec(
+; CHECK-NEXT: [[R:%.*]] = call <4 x i32> @llvm.ct.select.v4i32(i1 [[C:%.*]], <4 x i32> [[X:%.*]], <4 x i32> [[X]])
+; CHECK-NEXT: ret <4 x i32> [[X]]
+;
+ %r = call <4 x i32> @llvm.ct.select.v4i32(i1 %c, <4 x i32> %x, <4 x i32> %x)
+ ret <4 x i32> %r
+}
+
+; Negative test: must NOT fold when condition is a non-literal value, even if
+; analyses could prove it constant. The whole point of ct.select is to keep
+; the lowering even when the cond looks statically derivable.
+define i32 @ct_select_runtime_cond_no_fold(i1 %c, i32 %x, i32 %y) {
+; CHECK-LABEL: @ct_select_runtime_cond_no_fold(
+; CHECK-NEXT: [[R:%.*]] = call i32 @llvm.ct.select.i32(i1 [[C:%.*]], i32 [[X:%.*]], i32 [[Y:%.*]])
+; CHECK-NEXT: ret i32 [[R]]
+;
+ %r = call i32 @llvm.ct.select.i32(i1 %c, i32 %x, i32 %y)
+ ret i32 %r
+}
+
+; Negative test: distinct arms with a runtime condition must not fold.
+define i32 @ct_select_distinct_arms_no_fold(i1 %c, i32 %x, i32 %y) {
+; CHECK-LABEL: @ct_select_distinct_arms_no_fold(
+; CHECK-NEXT: [[R:%.*]] = call i32 @llvm.ct.select.i32(i1 [[C:%.*]], i32 [[X:%.*]], i32 [[Y:%.*]])
+; CHECK-NEXT: ret i32 [[R]]
+;
+ %r = call i32 @llvm.ct.select.i32(i1 %c, i32 %x, i32 %y)
+ ret i32 %r
+}
>From b0cd65ef822d60cd6b9776389c644ba8d1685b75 Mon Sep 17 00:00:00 2001
From: AkshayK <iit.akshay at gmail.com>
Date: Fri, 22 May 2026 15:12:51 -0400
Subject: [PATCH 05/12] [ConstantTime] Reword comments and LangRef doc
---
llvm/include/llvm/CodeGen/ISDOpcodes.h | 7 +++++--
llvm/lib/CodeGen/SelectionDAG/SelectionDAGBuilder.cpp | 7 -------
2 files changed, 5 insertions(+), 9 deletions(-)
diff --git a/llvm/include/llvm/CodeGen/ISDOpcodes.h b/llvm/include/llvm/CodeGen/ISDOpcodes.h
index 1e3b52494c0a9..f5534292869ab 100644
--- a/llvm/include/llvm/CodeGen/ISDOpcodes.h
+++ b/llvm/include/llvm/CodeGen/ISDOpcodes.h
@@ -805,8 +805,11 @@ enum NodeType {
/// i1 then the high bits must conform to getBooleanContents.
SELECT,
- /// CT_SELECT(Cond, TrueVal, FalseVal). Cond is i1 and the value operands must
- /// have the same type. Used to lower the constant-time select intrinsic.
+ /// CT_SELECT(COND, TRUEVAL, FALSEVAL) - Constant-time select: returns TRUEVAL
+ /// if COND (an i1) is true, else FALSEVAL; the value operands and result
+ /// share one type. Unlike SELECT, it must lower to code whose timing is
+ /// independent of COND -- no data-dependent branches, both arms always
+ /// evaluated -- and combines must preserve that. Node for llvm.ct.select.*.
CT_SELECT,
/// Select with a vector condition (op #0) and two vector operands (ops #1
diff --git a/llvm/lib/CodeGen/SelectionDAG/SelectionDAGBuilder.cpp b/llvm/lib/CodeGen/SelectionDAG/SelectionDAGBuilder.cpp
index 74f514d789125..8e96d9bb8c76f 100644
--- a/llvm/lib/CodeGen/SelectionDAG/SelectionDAGBuilder.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/SelectionDAGBuilder.cpp
@@ -6912,13 +6912,6 @@ void SelectionDAGBuilder::visitIntrinsicCall(const CallInst &I,
assert(A.getValueType() == B.getValueType() &&
"Operands are of different types");
assert(!Cond.getValueType().isVector() && "Vector condition not supported");
-
- // TODO: a follow-up target-specific bundling pass needs to know the
- // function uses ct.select. When that consumer lands, record it in the
- // target's MachineFunctionInfo instead of mutating the IR from codegen.
- // Kept commented for reference until then:
- // DAG.getMachineFunction().getFunction().addFnAttr("ct-select");
-
setValue(&I, DAG.getNode(ISD::CT_SELECT, getCurSDLoc(), A.getValueType(),
Cond, A, B));
return;
>From cd036cff2af4101001618186dac7ef890228ee94 Mon Sep 17 00:00:00 2001
From: AkshayK <iit.akshay at gmail.com>
Date: Tue, 14 Jul 2026 19:39:41 -0400
Subject: [PATCH 06/12] [ConstantTime] Address reviewer feedback for
llvm.ct.select core
Model llvm.ct.select as IntrInaccessibleMemOnly so the call stays pinned
without pessimizing alias analysis. Make CT_SELECT flagless (drop the
getCTSelect flags parameter, dead flag propagation, and
setFlags-after-getNode) and assert its condition is scalar. Blend the FP
memory fallback at the widest legal integer width with ext-load and
trunc-store tails instead of illegal i8 chunks. Restore the LangRef
section dropped in the rebase onto the Markdown docs migration and note
RISC-V Zkt/Zvkt. Apply review style fixes and regenerate affected tests.
---
llvm/docs/LangRef.md | 125 ++++++++++
llvm/include/llvm/CodeGen/ISDOpcodes.h | 4 +-
llvm/include/llvm/CodeGen/SelectionDAG.h | 9 +-
llvm/include/llvm/IR/Intrinsics.td | 10 +-
.../include/llvm/Target/TargetSelectionDAG.td | 2 +-
llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp | 35 ++-
llvm/lib/CodeGen/SelectionDAG/LegalizeDAG.cpp | 66 ++---
.../SelectionDAG/LegalizeFloatTypes.cpp | 8 +-
.../SelectionDAG/LegalizeIntegerTypes.cpp | 6 +-
.../SelectionDAG/LegalizeVectorTypes.cpp | 2 +-
.../SelectionDAG/SelectionDAGBuilder.cpp | 6 +-
.../SelectionDAG/SelectionDAGDumper.cpp | 2 +-
llvm/test/CodeGen/RISCV/ctselect-fallback.ll | 226 +++++++++---------
llvm/test/CodeGen/X86/ctselect.ll | 91 ++-----
14 files changed, 347 insertions(+), 245 deletions(-)
diff --git a/llvm/docs/LangRef.md b/llvm/docs/LangRef.md
index a6a7fa8926f2f..6de11a8d0ae9d 100644
--- a/llvm/docs/LangRef.md
+++ b/llvm/docs/LangRef.md
@@ -16038,6 +16038,131 @@ catch:
ret void
```
+### Constant-Time Intrinsics
+
+These intrinsics are provided to support constant-time operations for
+security-sensitive code. Constant-time operations execute in time independent
+of secret data values, preventing timing side-channel leaks.
+
+(int_ct_select)=
+
+#### '`llvm.ct.select.*`' Intrinsic
+
+##### Syntax:
+
+This is an overloaded intrinsic. You can use `llvm.ct.select` on any
+integer or floating-point type, pointer types, or vectors types.
+
+```
+declare i32 @llvm.ct.select.i32(i1 <cond>, i32 <val1>, i32 <val2>)
+declare i64 @llvm.ct.select.i64(i1 <cond>, i64 <val1>, i64 <val2>)
+declare float @llvm.ct.select.f32(i1 <cond>, float <val1>, float <val2>)
+declare double @llvm.ct.select.f64(i1 <cond>, double <val1>, double <val2>)
+declare ptr @llvm.ct.select.p0(i1 <cond>, ptr <val1>, ptr <val2>)
+
+; 128-bit vectors
+declare <4 x i32> @llvm.ct.select.v4i32(i1 <cond>, <4 x i32> <val1>, <4 x i32> <val2>)
+declare <2 x i64> @llvm.ct.select.v2i64(i1 <cond>, <2 x i64> <val1>, <2 x i64> <val2>)
+declare <4 x float> @llvm.ct.select.v4f32(i1 <cond>, <4 x float> <val1>, <4 x float> <val2>)
+declare <2 x double> @llvm.ct.select.v2f64(i1 <cond>, <2 x double> <val1>, <2 x double> <val2>)
+
+; 256-bit vectors
+declare <8 x i32> @llvm.ct.select.v8i32(i1 <cond>, <8 x i32> <val1>, <8 x i32> <val2>)
+declare <8 x float> @llvm.ct.select.v8f32(i1 <cond>, <8 x float> <val1>, <8 x float> <val2>)
+declare <4 x double> @llvm.ct.select.v4f64(i1 <cond>, <4 x double> <val1>, <4 x double> <val2>)
+```
+
+##### Overview:
+
+The '`llvm.ct.select`' family of intrinsic functions selects one of two
+values based on a condition, with the guarantee that the operation executes
+in constant time. Unlike the standard {ref}`select <i_select>` instruction,
+`llvm.ct.select` ensures that the execution time and observable behavior
+do not depend on the condition value, preventing timing-based side-channel
+leaks.
+
+##### Arguments:
+
+The '`llvm.ct.select`' intrinsic requires three arguments:
+
+1. The condition, which must be a scalar value of type 'i1'. Unlike
+ {ref}`select <i_select>` which accepts both scalar 'i1' and vector
+ '`<N x i1>`' conditions, `llvm.ct.select` only accepts a scalar 'i1'
+ condition. Vector conditions are not supported.
+2. The first value argument of any {ref}`first class <t_firstclass>` type.
+ This can be a scalar or vector type.
+3. The second value argument, which must have the same type as the first
+ value argument.
+
+##### Semantics:
+
+If the condition evaluates to true, the intrinsic returns the first value
+argument; otherwise, it returns the second value argument.
+
+The key semantic difference from {ref}`select <i_select>` is the constant-time
+code generation guarantee: the intrinsic must be lowered to machine code that:
+
+- Does not introduce data-dependent control flow based on the condition value
+- Executes the same sequence of instructions regardless of the condition value
+- Computes both value arguments before performing the selection
+
+**Platform Requirements:** The constant-time guarantee is conditional on
+hardware support for data-independent execution timing. This may be a
+processor mode that the platform enables, such as Arm DIT (Data Independent
+Timing) or x86 DOIT (Data Operand Independent Timing), or a static
+architectural guarantee, such as the RISC-V `Zkt` and `Zvkt` extensions,
+which certify data-independent latency for the scalar and vector instructions
+emitted by the lowering. Without such hardware support, the generated code
+will still be free from data-dependent control flow, but microarchitectural
+timing variations may still occur.
+
+The typical implementation uses bitwise operations to blend the two values
+based on a mask derived from the condition:
+
+```
+mask = sext(cond) ; sign-extend condition to all 1s or all 0s
+result = val2 ^ ((val1 ^ val2) & mask)
+```
+
+Targets with native constant-time select support use target-specific
+instructions to generate optimized bitwise operations with stronger guarantees.
+Targets without native support lower the intrinsic to a sequence of generic
+bitwise operations as shown above, structured to resist pattern recognition
+and preserve the constant-time property through optimization passes.
+
+Optimizations must preserve the constant-time code generation semantics.
+Transforms that would introduce data-dependent control flow are not permitted.
+This includes converting to conditional branches, using predicated instructions
+with data-dependent timing, or optimizing away either value argument before the
+selection completes (both paths must be computed).
+
+##### Examples:
+
+```llvm
+; Constant-time integer selection
+%x = call i32 @llvm.ct.select.i32(i1 %cond, i32 42, i32 17)
+%key = call i64 @llvm.ct.select.i64(i1 %cond, i64 %k_a, i64 %k_b)
+
+; Constant-time 128-bit integer vector selection (scalar condition broadcast to all lanes)
+%v4 = call <4 x i32> @llvm.ct.select.v4i32(i1 %cond,
+ <4 x i32> <i32 1, i32 2, i32 3, i32 4>,
+ <4 x i32> <i32 5, i32 6, i32 7, i32 8>)
+
+; Constant-time 256-bit integer vector selection
+%v8 = call <8 x i32> @llvm.ct.select.v8i32(i1 %cond,
+ <8 x i32> %vec_a, <8 x i32> %vec_b)
+
+; Constant-time 256-bit float vector selection
+%v8f = call <8 x float> @llvm.ct.select.v8f32(i1 %cond,
+ <8 x float> %fvec_a, <8 x float> %fvec_b)
+
+; Constant-time float selection
+%f = call float @llvm.ct.select.f32(i1 %cond, float 1.0, float 0.0)
+
+; Constant-time pointer selection
+%ptr = call ptr @llvm.ct.select.p0(i1 %cond, ptr %ptr_a, ptr %ptr_b)
+```
+
### Standard C/C++ Library Intrinsics
LLVM provides intrinsics for a few important standard C/C++ library
diff --git a/llvm/include/llvm/CodeGen/ISDOpcodes.h b/llvm/include/llvm/CodeGen/ISDOpcodes.h
index f5534292869ab..c5cbd95d6e4ae 100644
--- a/llvm/include/llvm/CodeGen/ISDOpcodes.h
+++ b/llvm/include/llvm/CodeGen/ISDOpcodes.h
@@ -809,7 +809,9 @@ enum NodeType {
/// if COND (an i1) is true, else FALSEVAL; the value operands and result
/// share one type. Unlike SELECT, it must lower to code whose timing is
/// independent of COND -- no data-dependent branches, both arms always
- /// evaluated -- and combines must preserve that. Node for llvm.ct.select.*.
+ /// evaluated -- and combines must preserve that. Carries no SDNodeFlags;
+ /// FMF-driven select combines (e.g. select -> fminnum) must not be applied.
+ /// Node for llvm.ct.select.*.
CT_SELECT,
/// Select with a vector condition (op #0) and two vector operands (ops #1
diff --git a/llvm/include/llvm/CodeGen/SelectionDAG.h b/llvm/include/llvm/CodeGen/SelectionDAG.h
index 60f8af27101aa..2fe58e25d4475 100644
--- a/llvm/include/llvm/CodeGen/SelectionDAG.h
+++ b/llvm/include/llvm/CodeGen/SelectionDAG.h
@@ -1360,11 +1360,16 @@ class SelectionDAG {
return getNode(Opcode, DL, VT, Cond, LHS, RHS, Flags);
}
+ /// Helper function to build CT_SELECT nodes. Unlike select, ct.select only
+ /// accepts a scalar condition, shared by all lanes of vector operands, and
+ /// carries no SDNodeFlags (see the ISD::CT_SELECT documentation).
SDValue getCTSelect(const SDLoc &DL, EVT VT, SDValue Cond, SDValue LHS,
- SDValue RHS, SDNodeFlags Flags = SDNodeFlags()) {
+ SDValue RHS) {
assert(LHS.getValueType() == VT && RHS.getValueType() == VT &&
"Cannot use select on differing types");
- return getNode(ISD::CT_SELECT, DL, VT, Cond, LHS, RHS, Flags);
+ assert(!Cond.getValueType().isVector() &&
+ "ct.select condition must be a scalar");
+ return getNode(ISD::CT_SELECT, DL, VT, Cond, LHS, RHS);
}
/// Helper function to make it easier to build SelectCC's if you just have an
diff --git a/llvm/include/llvm/IR/Intrinsics.td b/llvm/include/llvm/IR/Intrinsics.td
index cbfc08de904cd..a88d682ad68e0 100644
--- a/llvm/include/llvm/IR/Intrinsics.td
+++ b/llvm/include/llvm/IR/Intrinsics.td
@@ -2068,11 +2068,17 @@ def int_coro_subfn_addr : DefaultAttrsIntrinsic<
///===---------------------- Constant Time Intrinsics ----------------------===//
//
-// Intrinsic to support constant time select
+// Intrinsic to support constant-time select. The selected value is pure, but
+// the intrinsic is intentionally modeled like llvm.sideeffect: it only touches
+// inaccessible memory so it remains an opaque barrier to generic transforms.
+// This prevents passes from deleting it when unused, speculating it, or
+// rewriting it into an ordinary select, while avoiding claims about writes to
+// accessible program memory. The constant-time lowering contract is described
+// in LangRef.
def int_ct_select
: DefaultAttrsIntrinsic<[llvm_any_ty],
[llvm_i1_ty, LLVMMatchType<0>, LLVMMatchType<0>],
- [IntrWriteMem, IntrWillReturn, NoUndef<RetIndex>]>;
+ [IntrInaccessibleMemOnly, NoUndef<RetIndex>]>;
////===-------------------------- Other Intrinsics --------------------------===//
diff --git a/llvm/include/llvm/Target/TargetSelectionDAG.td b/llvm/include/llvm/Target/TargetSelectionDAG.td
index a855f360dcd47..f19d2d4e99c0c 100644
--- a/llvm/include/llvm/Target/TargetSelectionDAG.td
+++ b/llvm/include/llvm/Target/TargetSelectionDAG.td
@@ -785,7 +785,7 @@ def reset_fpmode : SDNode<"ISD::RESET_FPMODE", SDTNone, [SDNPHasChain]>;
def setcc : SDNode<"ISD::SETCC" , SDTSetCC>;
def select : SDNode<"ISD::SELECT" , SDTSelect>;
-def ct_select : SDNode<"ISD::CT_SELECT", SDTSelect>;
+def ct_select : SDNode<"ISD::CT_SELECT" , SDTSelect>;
def vselect : SDNode<"ISD::VSELECT" , SDTVSelect>;
def selectcc : SDNode<"ISD::SELECT_CC" , SDTSelectCC>;
diff --git a/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp b/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp
index 6016f37e855c5..bda6534b34808 100644
--- a/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp
@@ -482,8 +482,6 @@ namespace {
SDValue visitCTTZ_ZERO_POISON(SDNode *N);
SDValue visitCTPOP(SDNode *N);
SDValue visitSELECT(SDNode *N);
- // ISD::CT_SELECT - Constant-Time SELECT (not related to CT in
- // CTPOP/CTLZ/CTTZ where CT means "count").
SDValue visitCT_SELECT(SDNode *N);
SDValue visitVSELECT(SDNode *N);
SDValue visitSELECT_CC(SDNode *N);
@@ -2040,7 +2038,7 @@ SDValue DAGCombiner::visit(SDNode *N) {
case ISD::CTTZ_ZERO_POISON: return visitCTTZ_ZERO_POISON(N);
case ISD::CTPOP: return visitCTPOP(N);
case ISD::SELECT: return visitSELECT(N);
- case ISD::CT_SELECT: return visitCT_SELECT(N);
+ case ISD::CT_SELECT: return visitCT_SELECT(N);
case ISD::VSELECT: return visitVSELECT(N);
case ISD::SELECT_CC: return visitSELECT_CC(N);
case ISD::SETCC: return visitSETCC(N);
@@ -13699,14 +13697,13 @@ SDValue DAGCombiner::visitCT_SELECT(SDNode *N) {
EVT VT = N->getValueType(0);
EVT VT0 = N0.getValueType();
SDLoc DL(N);
- SDNodeFlags Flags = N->getFlags();
// ct_select (not Cond), N1, N2 -> ct_select Cond, N2, N1
// This is a CT-safe canonicalization: flip negated condition by swapping
// arms. extractBooleanFlip only matches boolean xor-with-1, so this preserves
// dataflow semantics and does not introduce data-dependent control flow.
if (SDValue F = extractBooleanFlip(N0, DAG, TLI, false))
- return DAG.getCTSelect(DL, VT, F, N2, N1, Flags);
+ return DAG.getCTSelect(DL, VT, F, N2, N1);
if (VT0 == MVT::i1) {
// Nested CT_SELECT merging optimizations for i1 conditions.
@@ -13720,26 +13717,26 @@ SDValue DAGCombiner::visitCT_SELECT(SDNode *N) {
// ct_select C0, (ct_select C1, X, Y), Y -> ct_select (C0 & C1), X, Y
// Semantic equivalence: If C0 is true, evaluate inner select (C1 ? X :
// Y). If C0 is false, choose Y. This is equivalent to (C0 && C1) ? X : Y.
- if (N1->getOpcode() == ISD::CT_SELECT && N1->hasOneUse()) {
- SDValue N1_0 = N1->getOperand(0);
- SDValue N1_1 = N1->getOperand(1);
- SDValue N1_2 = N1->getOperand(2);
- if (N1_2 == N2 && N0.getValueType() == N1_0.getValueType()) {
- SDValue And = DAG.getNode(ISD::AND, DL, N0.getValueType(), N0, N1_0);
- return DAG.getCTSelect(DL, N1.getValueType(), And, N1_1, N2, Flags);
+ if (N1.getOpcode() == ISD::CT_SELECT && N1.hasOneUse()) {
+ SDValue N10 = N1.getOperand(0);
+ SDValue N11 = N1.getOperand(1);
+ SDValue N12 = N1.getOperand(2);
+ if (N12 == N2 && N0.getValueType() == N10.getValueType()) {
+ SDValue And = DAG.getNode(ISD::AND, DL, N0.getValueType(), N0, N10);
+ return DAG.getCTSelect(DL, N1.getValueType(), And, N11, N2);
}
}
// ct_select C0, X, (ct_select C1, X, Y) -> ct_select (C0 | C1), X, Y
// Semantic equivalence: If C0 is true, choose X. If C0 is false, evaluate
// inner select (C1 ? X : Y). This is equivalent to (C0 || C1) ? X : Y.
- if (N2->getOpcode() == ISD::CT_SELECT && N2->hasOneUse()) {
- SDValue N2_0 = N2->getOperand(0);
- SDValue N2_1 = N2->getOperand(1);
- SDValue N2_2 = N2->getOperand(2);
- if (N2_1 == N1 && N0.getValueType() == N2_0.getValueType()) {
- SDValue Or = DAG.getNode(ISD::OR, DL, N0.getValueType(), N0, N2_0);
- return DAG.getCTSelect(DL, N1.getValueType(), Or, N1, N2_2, Flags);
+ if (N2.getOpcode() == ISD::CT_SELECT && N2.hasOneUse()) {
+ SDValue N20 = N2.getOperand(0);
+ SDValue N21 = N2.getOperand(1);
+ SDValue N22 = N2.getOperand(2);
+ if (N21 == N1 && N0.getValueType() == N20.getValueType()) {
+ SDValue Or = DAG.getNode(ISD::OR, DL, N0.getValueType(), N0, N20);
+ return DAG.getCTSelect(DL, N1.getValueType(), Or, N1, N22);
}
}
}
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeDAG.cpp b/llvm/lib/CodeGen/SelectionDAG/LegalizeDAG.cpp
index 92957d2b22bf9..54715a5a31d03 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeDAG.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeDAG.cpp
@@ -4370,19 +4370,17 @@ bool SelectionDAGLegalize::ExpandNode(SDNode *Node) {
unsigned StorageBytes = DL.getTypeStoreSize(VTTy);
assert(StorageBytes > 0 && "FP type with zero storage size");
- // Pick the largest legal scalar integer chunk that divides StorageBytes
- // evenly. i8 is the universal fallback (legal on every target).
- MVT ChunkVT = MVT::i8;
- for (MVT MV : {MVT::i64, MVT::i32, MVT::i16}) {
- unsigned MVBytes = MV.getSizeInBits() / 8;
- if (TLI.isTypeLegal(MV) && MVBytes <= StorageBytes &&
- StorageBytes % MVBytes == 0) {
- ChunkVT = MV;
+ // Blend at the widest legal scalar integer width. Chunks narrower than
+ // that (e.g. the 2-byte tail of x86_fp80) are zero-extended on load and
+ // truncated on store, so every value in the DAG has a legal type.
+ MVT BlendVT;
+ for (MVT MV : {MVT::i64, MVT::i32, MVT::i16, MVT::i8})
+ if (TLI.isTypeLegal(MV)) {
+ BlendVT = MV;
break;
}
- }
- unsigned ChunkBytes = ChunkVT.getSizeInBits() / 8;
- unsigned NumChunks = StorageBytes / ChunkBytes;
+ assert(BlendVT.isValid() && "no legal scalar integer type");
+ unsigned BlendBytes = BlendVT.getSizeInBits() / 8;
MachineFunction &MF = DAG.getMachineFunction();
SDValue StackT = DAG.CreateStackTemporary(VT);
@@ -4399,30 +4397,45 @@ bool SelectionDAGLegalize::ExpandNode(SDNode *Node) {
Chain = DAG.getStore(Chain, dl, Tmp2, StackT, PIT);
Chain = DAG.getStore(Chain, dl, Tmp3, StackF, PIF);
- for (unsigned i = 0; i < NumChunks; ++i) {
- TypeSize Off = TypeSize::getFixed(i * ChunkBytes);
+ // Walk the storage in power-of-2 chunks, widest first, so every access
+ // stays naturally aligned within the max-aligned stack temporaries.
+ unsigned Offset = 0;
+ while (Offset < StorageBytes) {
+ unsigned ChunkBytes =
+ std::min(BlendBytes, llvm::bit_floor(StorageBytes - Offset));
+ MVT MemVT = MVT::getIntegerVT(ChunkBytes * 8);
+ TypeSize Off = TypeSize::getFixed(Offset);
SDValue TPtr = DAG.getMemBasePlusOffset(StackT, Off, dl);
SDValue FPtr = DAG.getMemBasePlusOffset(StackF, Off, dl);
SDValue RPtr = DAG.getMemBasePlusOffset(StackR, Off, dl);
- SDValue Ti = DAG.getLoad(ChunkVT, dl, Chain, TPtr,
- PIT.getWithOffset(i * ChunkBytes));
- Chain = Ti.getValue(1);
- SDValue Fi = DAG.getLoad(ChunkVT, dl, Chain, FPtr,
- PIF.getWithOffset(i * ChunkBytes));
+ SDValue Ti, Fi;
+ if (MemVT == BlendVT) {
+ Ti = DAG.getLoad(BlendVT, dl, Chain, TPtr, PIT.getWithOffset(Offset));
+ Chain = Ti.getValue(1);
+ Fi = DAG.getLoad(BlendVT, dl, Chain, FPtr, PIF.getWithOffset(Offset));
+ } else {
+ Ti = DAG.getExtLoad(ISD::ZEXTLOAD, dl, BlendVT, Chain, TPtr,
+ PIT.getWithOffset(Offset), MemVT);
+ Chain = Ti.getValue(1);
+ Fi = DAG.getExtLoad(ISD::ZEXTLOAD, dl, BlendVT, Chain, FPtr,
+ PIF.getWithOffset(Offset), MemVT);
+ }
Chain = Fi.getValue(1);
- // Blend this chunk via CT_SELECT on a legal integer type. The
+ // Blend this chunk via CT_SELECT on the legal integer type. The
// recursive node will be Expand'd by the scalar-int branch below.
- SDValue Ri =
- DAG.getCTSelect(dl, ChunkVT, Tmp1, Ti, Fi, Node->getFlags());
+ SDValue Ri = DAG.getCTSelect(dl, BlendVT, Tmp1, Ti, Fi);
- Chain = DAG.getStore(Chain, dl, Ri, RPtr,
- PIR.getWithOffset(i * ChunkBytes));
+ if (MemVT == BlendVT)
+ Chain = DAG.getStore(Chain, dl, Ri, RPtr, PIR.getWithOffset(Offset));
+ else
+ Chain = DAG.getTruncStore(Chain, dl, Ri, RPtr,
+ PIR.getWithOffset(Offset), MemVT);
+ Offset += ChunkBytes;
}
Tmp1 = DAG.getLoad(VT, dl, Chain, StackR, PIR);
- Tmp1->setFlags(Node->getFlags());
Results.push_back(Tmp1);
break;
}
@@ -4486,7 +4499,6 @@ bool SelectionDAGLegalize::ExpandNode(SDNode *Node) {
if (WorkingVT != VT)
Tmp1 = DAG.getBitcast(VT, Tmp1);
- Tmp1->setFlags(Node->getFlags());
Results.push_back(Tmp1);
break;
}
@@ -5834,8 +5846,8 @@ void SelectionDAGLegalize::PromoteNode(SDNode *Node) {
Tmp2 = DAG.getNode(ExtOp, dl, NVT, Node->getOperand(1));
Tmp3 = DAG.getNode(ExtOp, dl, NVT, Node->getOperand(2));
// Perform the larger operation, then round down.
- Tmp1 = DAG.getNode(Node->getOpcode(), dl, NVT, Tmp1, Tmp2, Tmp3);
- Tmp1->setFlags(Node->getFlags());
+ Tmp1 = DAG.getNode(Node->getOpcode(), dl, NVT, Tmp1, Tmp2, Tmp3,
+ Node->getFlags());
if (TruncOp != ISD::FP_ROUND)
Tmp1 = DAG.getNode(TruncOp, dl, Node->getValueType(0), Tmp1);
else
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeFloatTypes.cpp b/llvm/lib/CodeGen/SelectionDAG/LegalizeFloatTypes.cpp
index 04257589c3e4d..28d5279d98453 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeFloatTypes.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeFloatTypes.cpp
@@ -161,7 +161,7 @@ void DAGTypeLegalizer::SoftenFloatResult(SDNode *N, unsigned ResNo) {
case ISD::ATOMIC_LOAD: R = SoftenFloatRes_ATOMIC_LOAD(N); break;
case ISD::ATOMIC_SWAP: R = BitcastToInt_ATOMIC_SWAP(N); break;
case ISD::SELECT: R = SoftenFloatRes_SELECT(N); break;
- case ISD::CT_SELECT: R = SoftenFloatRes_CT_SELECT(N); break;
+ case ISD::CT_SELECT: R = SoftenFloatRes_CT_SELECT(N); break;
case ISD::SELECT_CC: R = SoftenFloatRes_SELECT_CC(N); break;
case ISD::FREEZE: R = SoftenFloatRes_FREEZE(N); break;
case ISD::STRICT_SINT_TO_FP:
@@ -971,7 +971,7 @@ SDValue DAGTypeLegalizer::SoftenFloatRes_CT_SELECT(SDNode *N) {
SDValue LHS = GetSoftenedFloat(N->getOperand(1));
SDValue RHS = GetSoftenedFloat(N->getOperand(2));
return DAG.getCTSelect(SDLoc(N), LHS.getValueType(), N->getOperand(0), LHS,
- RHS, N->getFlags());
+ RHS);
}
SDValue DAGTypeLegalizer::SoftenFloatRes_SELECT_CC(SDNode *N) {
@@ -1537,7 +1537,7 @@ void DAGTypeLegalizer::ExpandFloatResult(SDNode *N, unsigned ResNo) {
case ISD::POISON:
case ISD::UNDEF: SplitRes_UNDEF(N, Lo, Hi); break;
case ISD::SELECT: SplitRes_Select(N, Lo, Hi); break;
- case ISD::CT_SELECT: SplitRes_CT_SELECT(N, Lo, Hi); break;
+ case ISD::CT_SELECT: SplitRes_CT_SELECT(N, Lo, Hi); break;
case ISD::SELECT_CC: SplitRes_SELECT_CC(N, Lo, Hi); break;
case ISD::MERGE_VALUES: ExpandRes_MERGE_VALUES(N, ResNo, Lo, Hi); break;
@@ -2866,7 +2866,7 @@ SDValue DAGTypeLegalizer::SoftPromoteHalfRes_CT_SELECT(SDNode *N) {
SDValue Op1 = GetSoftPromotedHalf(N->getOperand(1));
SDValue Op2 = GetSoftPromotedHalf(N->getOperand(2));
return DAG.getCTSelect(SDLoc(N), Op1.getValueType(), N->getOperand(0), Op1,
- Op2, N->getFlags());
+ Op2);
}
SDValue DAGTypeLegalizer::SoftPromoteHalfRes_SELECT_CC(SDNode *N) {
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeIntegerTypes.cpp b/llvm/lib/CodeGen/SelectionDAG/LegalizeIntegerTypes.cpp
index 0cc5e14834442..6f8a2db992d42 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeIntegerTypes.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeIntegerTypes.cpp
@@ -2414,9 +2414,9 @@ SDValue DAGTypeLegalizer::PromoteIntOp_CT_SELECT(SDNode *N, unsigned OpNo) {
SDValue Cond = N->getOperand(0);
EVT OpTy = N->getOperand(1).getValueType();
- // Promote all the way up to the canonical SetCC type.
- EVT OpVT = N->getOpcode() == ISD::CT_SELECT ? OpTy.getScalarType() : OpTy;
- Cond = PromoteTargetBoolean(Cond, OpVT);
+ // Promote all the way up to the canonical SetCC type. The condition is
+ // always scalar, so derive the boolean type from the operands' scalar type.
+ Cond = PromoteTargetBoolean(Cond, OpTy.getScalarType());
return SDValue(
DAG.UpdateNodeOperands(N, Cond, N->getOperand(1), N->getOperand(2)), 0);
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp b/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp
index 872b1c558aa6c..6d25c26cbe914 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp
@@ -768,7 +768,7 @@ SDValue DAGTypeLegalizer::ScalarizeVecRes_SELECT(SDNode *N) {
SDValue DAGTypeLegalizer::ScalarizeVecRes_CT_SELECT(SDNode *N) {
SDValue LHS = GetScalarizedVector(N->getOperand(1));
return DAG.getCTSelect(SDLoc(N), LHS.getValueType(), N->getOperand(0), LHS,
- GetScalarizedVector(N->getOperand(2)), N->getFlags());
+ GetScalarizedVector(N->getOperand(2)));
}
SDValue DAGTypeLegalizer::ScalarizeVecRes_SELECT_CC(SDNode *N) {
diff --git a/llvm/lib/CodeGen/SelectionDAG/SelectionDAGBuilder.cpp b/llvm/lib/CodeGen/SelectionDAG/SelectionDAGBuilder.cpp
index 8e96d9bb8c76f..3073a1a19b263 100644
--- a/llvm/lib/CodeGen/SelectionDAG/SelectionDAGBuilder.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/SelectionDAGBuilder.cpp
@@ -6909,11 +6909,7 @@ void SelectionDAGBuilder::visitIntrinsicCall(const CallInst &I,
SDValue Cond = getValue(I.getArgOperand(0));
SDValue A = getValue(I.getArgOperand(1));
SDValue B = getValue(I.getArgOperand(2));
- assert(A.getValueType() == B.getValueType() &&
- "Operands are of different types");
- assert(!Cond.getValueType().isVector() && "Vector condition not supported");
- setValue(&I, DAG.getNode(ISD::CT_SELECT, getCurSDLoc(), A.getValueType(),
- Cond, A, B));
+ setValue(&I, DAG.getCTSelect(getCurSDLoc(), A.getValueType(), Cond, A, B));
return;
}
case Intrinsic::call_preallocated_setup: {
diff --git a/llvm/lib/CodeGen/SelectionDAG/SelectionDAGDumper.cpp b/llvm/lib/CodeGen/SelectionDAG/SelectionDAGDumper.cpp
index 0263bd9622b41..81df3934ce8f4 100644
--- a/llvm/lib/CodeGen/SelectionDAG/SelectionDAGDumper.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/SelectionDAGDumper.cpp
@@ -346,7 +346,7 @@ std::string SDNode::getOperationName(const SelectionDAG *G) const {
case ISD::FPOWI: return "fpowi";
case ISD::STRICT_FPOWI: return "strict_fpowi";
case ISD::SETCC: return "setcc";
- case ISD::CT_SELECT: return "ct_select";
+ case ISD::CT_SELECT: return "ct_select";
case ISD::SETCCCARRY: return "setcccarry";
case ISD::STRICT_FSETCC: return "strict_fsetcc";
case ISD::STRICT_FSETCCS: return "strict_fsetccs";
diff --git a/llvm/test/CodeGen/RISCV/ctselect-fallback.ll b/llvm/test/CodeGen/RISCV/ctselect-fallback.ll
index 0df231b7a98d9..68d5489895a6a 100644
--- a/llvm/test/CodeGen/RISCV/ctselect-fallback.ll
+++ b/llvm/test/CodeGen/RISCV/ctselect-fallback.ll
@@ -58,10 +58,10 @@ define i64 @test_ctselect_i64(i1 %cond, i64 %a, i64 %b) {
;
; RV32-LABEL: test_ctselect_i64:
; RV32: # %bb.0:
-; RV32-NEXT: xor a1, a1, a3
; RV32-NEXT: slli a0, a0, 31
-; RV32-NEXT: xor a2, a2, a4
+; RV32-NEXT: xor a1, a1, a3
; RV32-NEXT: srai a0, a0, 31
+; RV32-NEXT: xor a2, a2, a4
; RV32-NEXT: and a1, a1, a0
; RV32-NEXT: and a2, a2, a0
; RV32-NEXT: xor a0, a3, a1
@@ -212,8 +212,8 @@ define i32 @test_ctselect_nested_and_i1_to_i32(i1 %c0, i1 %c1, i32 %x, i32 %y) {
; RV64: # %bb.0:
; RV64-NEXT: and a0, a0, a1
; RV64-NEXT: andi a0, a0, 1
-; RV64-NEXT: xor a2, a2, a3
; RV64-NEXT: slli a0, a0, 63
+; RV64-NEXT: xor a2, a2, a3
; RV64-NEXT: srai a0, a0, 63
; RV64-NEXT: and a0, a2, a0
; RV64-NEXT: xor a0, a3, a0
@@ -223,8 +223,8 @@ define i32 @test_ctselect_nested_and_i1_to_i32(i1 %c0, i1 %c1, i32 %x, i32 %y) {
; RV32: # %bb.0:
; RV32-NEXT: and a0, a0, a1
; RV32-NEXT: andi a0, a0, 1
-; RV32-NEXT: xor a2, a2, a3
; RV32-NEXT: slli a0, a0, 31
+; RV32-NEXT: xor a2, a2, a3
; RV32-NEXT: srai a0, a0, 31
; RV32-NEXT: and a0, a2, a0
; RV32-NEXT: xor a0, a3, a0
@@ -242,8 +242,8 @@ define i32 @test_ctselect_nested_or_i1_to_i32(i1 %c0, i1 %c1, i32 %x, i32 %y) {
; RV64: # %bb.0:
; RV64-NEXT: or a0, a0, a1
; RV64-NEXT: andi a0, a0, 1
-; RV64-NEXT: xor a2, a2, a3
; RV64-NEXT: slli a0, a0, 63
+; RV64-NEXT: xor a2, a2, a3
; RV64-NEXT: srai a0, a0, 63
; RV64-NEXT: and a0, a2, a0
; RV64-NEXT: xor a0, a3, a0
@@ -253,8 +253,8 @@ define i32 @test_ctselect_nested_or_i1_to_i32(i1 %c0, i1 %c1, i32 %x, i32 %y) {
; RV32: # %bb.0:
; RV32-NEXT: or a0, a0, a1
; RV32-NEXT: andi a0, a0, 1
-; RV32-NEXT: xor a2, a2, a3
; RV32-NEXT: slli a0, a0, 31
+; RV32-NEXT: xor a2, a2, a3
; RV32-NEXT: srai a0, a0, 31
; RV32-NEXT: and a0, a2, a0
; RV32-NEXT: xor a0, a3, a0
@@ -274,8 +274,8 @@ define i32 @test_ctselect_double_nested_and_i1(i1 %c0, i1 %c1, i1 %c2, i32 %x, i
; RV64-NEXT: and a0, a0, a1
; RV64-NEXT: and a0, a0, a2
; RV64-NEXT: andi a0, a0, 1
-; RV64-NEXT: xor a3, a3, a4
; RV64-NEXT: slli a0, a0, 63
+; RV64-NEXT: xor a3, a3, a4
; RV64-NEXT: srai a0, a0, 63
; RV64-NEXT: and a0, a3, a0
; RV64-NEXT: xor a0, a4, a0
@@ -286,8 +286,8 @@ define i32 @test_ctselect_double_nested_and_i1(i1 %c0, i1 %c1, i1 %c2, i32 %x, i
; RV32-NEXT: and a0, a0, a1
; RV32-NEXT: and a0, a0, a2
; RV32-NEXT: andi a0, a0, 1
-; RV32-NEXT: xor a3, a3, a4
; RV32-NEXT: slli a0, a0, 31
+; RV32-NEXT: xor a3, a3, a4
; RV32-NEXT: srai a0, a0, 31
; RV32-NEXT: and a0, a3, a0
; RV32-NEXT: xor a0, a4, a0
@@ -304,11 +304,11 @@ define i32 @test_ctselect_double_nested_mixed_i1(i1 %c0, i1 %c1, i1 %c2, i32 %x,
; RV64-LABEL: test_ctselect_double_nested_mixed_i1:
; RV64: # %bb.0:
; RV64-NEXT: and a0, a0, a1
-; RV64-NEXT: xor a3, a3, a4
; RV64-NEXT: andi a0, a0, 1
; RV64-NEXT: or a0, a0, a2
; RV64-NEXT: andi a0, a0, 1
; RV64-NEXT: slli a0, a0, 63
+; RV64-NEXT: xor a3, a3, a4
; RV64-NEXT: srai a0, a0, 63
; RV64-NEXT: and a3, a3, a0
; RV64-NEXT: xor a4, a4, a5
@@ -320,11 +320,11 @@ define i32 @test_ctselect_double_nested_mixed_i1(i1 %c0, i1 %c1, i1 %c2, i32 %x,
; RV32-LABEL: test_ctselect_double_nested_mixed_i1:
; RV32: # %bb.0:
; RV32-NEXT: and a0, a0, a1
-; RV32-NEXT: xor a3, a3, a4
; RV32-NEXT: andi a0, a0, 1
; RV32-NEXT: or a0, a0, a2
; RV32-NEXT: andi a0, a0, 1
; RV32-NEXT: slli a0, a0, 31
+; RV32-NEXT: xor a3, a3, a4
; RV32-NEXT: srai a0, a0, 31
; RV32-NEXT: and a3, a3, a0
; RV32-NEXT: xor a4, a4, a5
@@ -345,12 +345,12 @@ define i32 @test_ctselect_double_nested_mixed_i1(i1 %c0, i1 %c1, i1 %c2, i32 %x,
define i32 @test_ctselect_nested(i1 %cond1, i1 %cond2, i32 %a, i32 %b, i32 %c) {
; RV64-LABEL: test_ctselect_nested:
; RV64: # %bb.0:
-; RV64-NEXT: xor a2, a2, a3
; RV64-NEXT: slli a1, a1, 63
-; RV64-NEXT: xor a3, a3, a4
-; RV64-NEXT: slli a0, a0, 63
+; RV64-NEXT: xor a2, a2, a3
; RV64-NEXT: srai a1, a1, 63
; RV64-NEXT: and a1, a2, a1
+; RV64-NEXT: xor a3, a3, a4
+; RV64-NEXT: slli a0, a0, 63
; RV64-NEXT: xor a1, a3, a1
; RV64-NEXT: srai a0, a0, 63
; RV64-NEXT: and a0, a1, a0
@@ -359,12 +359,12 @@ define i32 @test_ctselect_nested(i1 %cond1, i1 %cond2, i32 %a, i32 %b, i32 %c) {
;
; RV32-LABEL: test_ctselect_nested:
; RV32: # %bb.0:
-; RV32-NEXT: xor a2, a2, a3
; RV32-NEXT: slli a1, a1, 31
-; RV32-NEXT: xor a3, a3, a4
-; RV32-NEXT: slli a0, a0, 31
+; RV32-NEXT: xor a2, a2, a3
; RV32-NEXT: srai a1, a1, 31
; RV32-NEXT: and a1, a2, a1
+; RV32-NEXT: xor a3, a3, a4
+; RV32-NEXT: slli a0, a0, 31
; RV32-NEXT: xor a1, a3, a1
; RV32-NEXT: srai a0, a0, 31
; RV32-NEXT: and a0, a1, a0
@@ -379,8 +379,8 @@ define i32 @test_ctselect_nested(i1 %cond1, i1 %cond2, i32 %a, i32 %b, i32 %c) {
define float @test_ctselect_f32_nan_inf(i1 %cond) {
; RV64-LABEL: test_ctselect_f32_nan_inf:
; RV64: # %bb.0:
-; RV64-NEXT: slli a0, a0, 63
; RV64-NEXT: lui a1, 1024
+; RV64-NEXT: slli a0, a0, 63
; RV64-NEXT: srai a0, a0, 63
; RV64-NEXT: and a0, a0, a1
; RV64-NEXT: lui a1, 522240
@@ -389,8 +389,8 @@ define float @test_ctselect_f32_nan_inf(i1 %cond) {
;
; RV32-LABEL: test_ctselect_f32_nan_inf:
; RV32: # %bb.0:
-; RV32-NEXT: slli a0, a0, 31
; RV32-NEXT: lui a1, 1024
+; RV32-NEXT: slli a0, a0, 31
; RV32-NEXT: srai a0, a0, 31
; RV32-NEXT: and a0, a0, a1
; RV32-NEXT: lui a1, 522240
@@ -403,20 +403,20 @@ define float @test_ctselect_f32_nan_inf(i1 %cond) {
define double @test_ctselect_f64_nan_inf(i1 %cond) {
; RV64-LABEL: test_ctselect_f64_nan_inf:
; RV64: # %bb.0:
-; RV64-NEXT: slli a0, a0, 63
; RV64-NEXT: li a1, 1
+; RV64-NEXT: slli a0, a0, 63
; RV64-NEXT: srai a0, a0, 63
; RV64-NEXT: slli a1, a1, 51
+; RV64-NEXT: li a2, 2047
; RV64-NEXT: and a0, a0, a1
-; RV64-NEXT: li a1, 2047
-; RV64-NEXT: slli a1, a1, 52
-; RV64-NEXT: xor a0, a0, a1
+; RV64-NEXT: slli a2, a2, 52
+; RV64-NEXT: xor a0, a0, a2
; RV64-NEXT: ret
;
; RV32-LABEL: test_ctselect_f64_nan_inf:
; RV32: # %bb.0:
-; RV32-NEXT: slli a0, a0, 31
; RV32-NEXT: lui a1, 128
+; RV32-NEXT: slli a0, a0, 31
; RV32-NEXT: srai a0, a0, 31
; RV32-NEXT: and a0, a0, a1
; RV32-NEXT: lui a1, 524032
@@ -462,10 +462,10 @@ define double @test_ctselect_f64(i1 %cond, double %a, double %b) {
;
; RV32-LABEL: test_ctselect_f64:
; RV32: # %bb.0:
-; RV32-NEXT: xor a1, a1, a3
; RV32-NEXT: slli a0, a0, 31
-; RV32-NEXT: xor a2, a2, a4
+; RV32-NEXT: xor a1, a1, a3
; RV32-NEXT: srai a0, a0, 31
+; RV32-NEXT: xor a2, a2, a4
; RV32-NEXT: and a1, a1, a0
; RV32-NEXT: and a2, a2, a0
; RV32-NEXT: xor a0, a3, a1
@@ -480,29 +480,29 @@ define <4 x i32> @test_ctselect_v4i32(i1 %cond, <4 x i32> %a, <4 x i32> %b) {
; RV64-LABEL: test_ctselect_v4i32:
; RV64: # %bb.0:
; RV64-NEXT: lw a4, 0(a3)
-; RV64-NEXT: lw a5, 8(a3)
-; RV64-NEXT: lw a6, 16(a3)
+; RV64-NEXT: lw a5, 0(a2)
+; RV64-NEXT: lw a6, 8(a2)
+; RV64-NEXT: lw a7, 8(a3)
+; RV64-NEXT: lw t0, 16(a3)
; RV64-NEXT: lw a3, 24(a3)
-; RV64-NEXT: lw a7, 0(a2)
-; RV64-NEXT: lw t0, 8(a2)
; RV64-NEXT: lw t1, 16(a2)
; RV64-NEXT: lw a2, 24(a2)
+; RV64-NEXT: xor a5, a5, a4
; RV64-NEXT: slli a1, a1, 63
; RV64-NEXT: srai a1, a1, 63
-; RV64-NEXT: xor a7, a7, a4
-; RV64-NEXT: xor t0, t0, a5
-; RV64-NEXT: xor t1, t1, a6
+; RV64-NEXT: xor a6, a6, a7
+; RV64-NEXT: and a5, a5, a1
+; RV64-NEXT: and a6, a6, a1
+; RV64-NEXT: xor t1, t1, t0
; RV64-NEXT: xor a2, a2, a3
-; RV64-NEXT: and a7, a7, a1
-; RV64-NEXT: and t0, t0, a1
; RV64-NEXT: and t1, t1, a1
; RV64-NEXT: and a1, a2, a1
-; RV64-NEXT: xor a2, a4, a7
-; RV64-NEXT: xor a4, a5, t0
-; RV64-NEXT: xor a5, a6, t1
+; RV64-NEXT: xor a4, a4, a5
+; RV64-NEXT: xor a2, a7, a6
+; RV64-NEXT: xor a5, t0, t1
; RV64-NEXT: xor a1, a3, a1
-; RV64-NEXT: sw a2, 0(a0)
-; RV64-NEXT: sw a4, 4(a0)
+; RV64-NEXT: sw a4, 0(a0)
+; RV64-NEXT: sw a2, 4(a0)
; RV64-NEXT: sw a5, 8(a0)
; RV64-NEXT: sw a1, 12(a0)
; RV64-NEXT: ret
@@ -510,29 +510,29 @@ define <4 x i32> @test_ctselect_v4i32(i1 %cond, <4 x i32> %a, <4 x i32> %b) {
; RV32-LABEL: test_ctselect_v4i32:
; RV32: # %bb.0:
; RV32-NEXT: lw a4, 0(a3)
-; RV32-NEXT: lw a5, 4(a3)
-; RV32-NEXT: lw a6, 8(a3)
+; RV32-NEXT: lw a5, 0(a2)
+; RV32-NEXT: lw a6, 4(a2)
+; RV32-NEXT: lw a7, 4(a3)
+; RV32-NEXT: lw t0, 8(a3)
; RV32-NEXT: lw a3, 12(a3)
-; RV32-NEXT: lw a7, 0(a2)
-; RV32-NEXT: lw t0, 4(a2)
; RV32-NEXT: lw t1, 8(a2)
; RV32-NEXT: lw a2, 12(a2)
+; RV32-NEXT: xor a5, a5, a4
; RV32-NEXT: slli a1, a1, 31
; RV32-NEXT: srai a1, a1, 31
-; RV32-NEXT: xor a7, a7, a4
-; RV32-NEXT: xor t0, t0, a5
-; RV32-NEXT: xor t1, t1, a6
+; RV32-NEXT: xor a6, a6, a7
+; RV32-NEXT: and a5, a5, a1
+; RV32-NEXT: and a6, a6, a1
+; RV32-NEXT: xor t1, t1, t0
; RV32-NEXT: xor a2, a2, a3
-; RV32-NEXT: and a7, a7, a1
-; RV32-NEXT: and t0, t0, a1
; RV32-NEXT: and t1, t1, a1
; RV32-NEXT: and a1, a2, a1
-; RV32-NEXT: xor a2, a4, a7
-; RV32-NEXT: xor a4, a5, t0
-; RV32-NEXT: xor a5, a6, t1
+; RV32-NEXT: xor a4, a4, a5
+; RV32-NEXT: xor a2, a7, a6
+; RV32-NEXT: xor a5, t0, t1
; RV32-NEXT: xor a1, a3, a1
-; RV32-NEXT: sw a2, 0(a0)
-; RV32-NEXT: sw a4, 4(a0)
+; RV32-NEXT: sw a4, 0(a0)
+; RV32-NEXT: sw a2, 4(a0)
; RV32-NEXT: sw a5, 8(a0)
; RV32-NEXT: sw a1, 12(a0)
; RV32-NEXT: ret
@@ -543,29 +543,29 @@ define <4 x float> @test_ctselect_v4f32(i1 %cond, <4 x float> %a, <4 x float> %b
; RV64-LABEL: test_ctselect_v4f32:
; RV64: # %bb.0:
; RV64-NEXT: lw a4, 0(a3)
-; RV64-NEXT: lw a5, 8(a3)
-; RV64-NEXT: lw a6, 16(a3)
+; RV64-NEXT: lw a5, 0(a2)
+; RV64-NEXT: lw a6, 8(a2)
+; RV64-NEXT: lw a7, 8(a3)
+; RV64-NEXT: lw t0, 16(a3)
; RV64-NEXT: lw a3, 24(a3)
-; RV64-NEXT: lw a7, 0(a2)
-; RV64-NEXT: lw t0, 8(a2)
; RV64-NEXT: lw t1, 16(a2)
; RV64-NEXT: lw a2, 24(a2)
+; RV64-NEXT: xor a5, a5, a4
; RV64-NEXT: slli a1, a1, 63
; RV64-NEXT: srai a1, a1, 63
-; RV64-NEXT: xor a7, a7, a4
-; RV64-NEXT: xor t0, t0, a5
-; RV64-NEXT: xor t1, t1, a6
+; RV64-NEXT: xor a6, a6, a7
+; RV64-NEXT: and a5, a5, a1
+; RV64-NEXT: and a6, a6, a1
+; RV64-NEXT: xor t1, t1, t0
; RV64-NEXT: xor a2, a2, a3
-; RV64-NEXT: and a7, a7, a1
-; RV64-NEXT: and t0, t0, a1
; RV64-NEXT: and t1, t1, a1
; RV64-NEXT: and a1, a2, a1
-; RV64-NEXT: xor a2, a4, a7
-; RV64-NEXT: xor a4, a5, t0
-; RV64-NEXT: xor a5, a6, t1
+; RV64-NEXT: xor a4, a4, a5
+; RV64-NEXT: xor a2, a7, a6
+; RV64-NEXT: xor a5, t0, t1
; RV64-NEXT: xor a1, a3, a1
-; RV64-NEXT: sw a2, 0(a0)
-; RV64-NEXT: sw a4, 4(a0)
+; RV64-NEXT: sw a4, 0(a0)
+; RV64-NEXT: sw a2, 4(a0)
; RV64-NEXT: sw a5, 8(a0)
; RV64-NEXT: sw a1, 12(a0)
; RV64-NEXT: ret
@@ -573,29 +573,29 @@ define <4 x float> @test_ctselect_v4f32(i1 %cond, <4 x float> %a, <4 x float> %b
; RV32-LABEL: test_ctselect_v4f32:
; RV32: # %bb.0:
; RV32-NEXT: lw a4, 0(a3)
-; RV32-NEXT: lw a5, 4(a3)
-; RV32-NEXT: lw a6, 8(a3)
+; RV32-NEXT: lw a5, 0(a2)
+; RV32-NEXT: lw a6, 4(a2)
+; RV32-NEXT: lw a7, 4(a3)
+; RV32-NEXT: lw t0, 8(a3)
; RV32-NEXT: lw a3, 12(a3)
-; RV32-NEXT: lw a7, 0(a2)
-; RV32-NEXT: lw t0, 4(a2)
; RV32-NEXT: lw t1, 8(a2)
; RV32-NEXT: lw a2, 12(a2)
+; RV32-NEXT: xor a5, a5, a4
; RV32-NEXT: slli a1, a1, 31
; RV32-NEXT: srai a1, a1, 31
-; RV32-NEXT: xor a7, a7, a4
-; RV32-NEXT: xor t0, t0, a5
-; RV32-NEXT: xor t1, t1, a6
+; RV32-NEXT: xor a6, a6, a7
+; RV32-NEXT: and a5, a5, a1
+; RV32-NEXT: and a6, a6, a1
+; RV32-NEXT: xor t1, t1, t0
; RV32-NEXT: xor a2, a2, a3
-; RV32-NEXT: and a7, a7, a1
-; RV32-NEXT: and t0, t0, a1
; RV32-NEXT: and t1, t1, a1
; RV32-NEXT: and a1, a2, a1
-; RV32-NEXT: xor a2, a4, a7
-; RV32-NEXT: xor a4, a5, t0
-; RV32-NEXT: xor a5, a6, t1
+; RV32-NEXT: xor a4, a4, a5
+; RV32-NEXT: xor a2, a7, a6
+; RV32-NEXT: xor a5, t0, t1
; RV32-NEXT: xor a1, a3, a1
-; RV32-NEXT: sw a2, 0(a0)
-; RV32-NEXT: sw a4, 4(a0)
+; RV32-NEXT: sw a4, 0(a0)
+; RV32-NEXT: sw a2, 4(a0)
; RV32-NEXT: sw a5, 8(a0)
; RV32-NEXT: sw a1, 12(a0)
; RV32-NEXT: ret
@@ -622,34 +622,34 @@ define <8 x i32> @test_ctselect_v8i32(i1 %cond, <8 x i32> %a, <8 x i32> %b) {
; RV64-NEXT: lw t2, 48(a2)
; RV64-NEXT: lw t3, 56(a2)
; RV64-NEXT: lw t4, 0(a3)
-; RV64-NEXT: lw t5, 8(a3)
-; RV64-NEXT: lw t6, 16(a3)
+; RV64-NEXT: lw t5, 0(a2)
+; RV64-NEXT: lw t6, 8(a2)
+; RV64-NEXT: lw s0, 8(a3)
+; RV64-NEXT: lw s1, 16(a3)
; RV64-NEXT: lw a3, 24(a3)
-; RV64-NEXT: lw s0, 0(a2)
-; RV64-NEXT: lw s1, 8(a2)
; RV64-NEXT: lw s2, 16(a2)
; RV64-NEXT: lw a2, 24(a2)
+; RV64-NEXT: xor t5, t5, t4
; RV64-NEXT: slli a1, a1, 63
; RV64-NEXT: srai a1, a1, 63
-; RV64-NEXT: xor s0, s0, t4
-; RV64-NEXT: xor s1, s1, t5
-; RV64-NEXT: xor s2, s2, t6
+; RV64-NEXT: xor t6, t6, s0
+; RV64-NEXT: and t5, t5, a1
+; RV64-NEXT: and t6, t6, a1
+; RV64-NEXT: xor s2, s2, s1
; RV64-NEXT: xor a2, a2, a3
-; RV64-NEXT: xor t0, t0, a4
-; RV64-NEXT: xor t1, t1, a5
-; RV64-NEXT: xor t2, t2, a6
-; RV64-NEXT: xor t3, t3, a7
-; RV64-NEXT: and s0, s0, a1
-; RV64-NEXT: and s1, s1, a1
; RV64-NEXT: and s2, s2, a1
; RV64-NEXT: and a2, a2, a1
+; RV64-NEXT: xor t0, t0, a4
+; RV64-NEXT: xor t1, t1, a5
; RV64-NEXT: and t0, t0, a1
; RV64-NEXT: and t1, t1, a1
+; RV64-NEXT: xor t2, t2, a6
+; RV64-NEXT: xor t3, t3, a7
; RV64-NEXT: and t2, t2, a1
; RV64-NEXT: and a1, t3, a1
-; RV64-NEXT: xor t3, t4, s0
-; RV64-NEXT: xor t4, t5, s1
-; RV64-NEXT: xor t5, t6, s2
+; RV64-NEXT: xor t3, t4, t5
+; RV64-NEXT: xor t4, s0, t6
+; RV64-NEXT: xor t5, s1, s2
; RV64-NEXT: xor a2, a3, a2
; RV64-NEXT: xor a3, a4, t0
; RV64-NEXT: xor a4, a5, t1
@@ -692,34 +692,34 @@ define <8 x i32> @test_ctselect_v8i32(i1 %cond, <8 x i32> %a, <8 x i32> %b) {
; RV32-NEXT: lw t2, 24(a2)
; RV32-NEXT: lw t3, 28(a2)
; RV32-NEXT: lw t4, 0(a3)
-; RV32-NEXT: lw t5, 4(a3)
-; RV32-NEXT: lw t6, 8(a3)
+; RV32-NEXT: lw t5, 0(a2)
+; RV32-NEXT: lw t6, 4(a2)
+; RV32-NEXT: lw s0, 4(a3)
+; RV32-NEXT: lw s1, 8(a3)
; RV32-NEXT: lw a3, 12(a3)
-; RV32-NEXT: lw s0, 0(a2)
-; RV32-NEXT: lw s1, 4(a2)
; RV32-NEXT: lw s2, 8(a2)
; RV32-NEXT: lw a2, 12(a2)
+; RV32-NEXT: xor t5, t5, t4
; RV32-NEXT: slli a1, a1, 31
; RV32-NEXT: srai a1, a1, 31
-; RV32-NEXT: xor s0, s0, t4
-; RV32-NEXT: xor s1, s1, t5
-; RV32-NEXT: xor s2, s2, t6
+; RV32-NEXT: xor t6, t6, s0
+; RV32-NEXT: and t5, t5, a1
+; RV32-NEXT: and t6, t6, a1
+; RV32-NEXT: xor s2, s2, s1
; RV32-NEXT: xor a2, a2, a3
-; RV32-NEXT: xor t0, t0, a4
-; RV32-NEXT: xor t1, t1, a5
-; RV32-NEXT: xor t2, t2, a6
-; RV32-NEXT: xor t3, t3, a7
-; RV32-NEXT: and s0, s0, a1
-; RV32-NEXT: and s1, s1, a1
; RV32-NEXT: and s2, s2, a1
; RV32-NEXT: and a2, a2, a1
+; RV32-NEXT: xor t0, t0, a4
+; RV32-NEXT: xor t1, t1, a5
; RV32-NEXT: and t0, t0, a1
; RV32-NEXT: and t1, t1, a1
+; RV32-NEXT: xor t2, t2, a6
+; RV32-NEXT: xor t3, t3, a7
; RV32-NEXT: and t2, t2, a1
; RV32-NEXT: and a1, t3, a1
-; RV32-NEXT: xor t3, t4, s0
-; RV32-NEXT: xor t4, t5, s1
-; RV32-NEXT: xor t5, t6, s2
+; RV32-NEXT: xor t3, t4, t5
+; RV32-NEXT: xor t4, s0, t6
+; RV32-NEXT: xor t5, s1, s2
; RV32-NEXT: xor a2, a3, a2
; RV32-NEXT: xor a3, a4, t0
; RV32-NEXT: xor a4, a5, t1
diff --git a/llvm/test/CodeGen/X86/ctselect.ll b/llvm/test/CodeGen/X86/ctselect.ll
index 9d54c29db9683..5b23642b1c5d0 100644
--- a/llvm/test/CodeGen/X86/ctselect.ll
+++ b/llvm/test/CodeGen/X86/ctselect.ll
@@ -500,42 +500,25 @@ define fp128 @test_ctselect_f128(i1 %cond, fp128 %a, fp128 %b) #0 {
define x86_fp80 @test_ctselect_f80(i1 %cond, x86_fp80 %a, x86_fp80 %b) #0 {
; X64-LABEL: test_ctselect_f80:
; X64: # %bb.0:
+; X64-NEXT: # kill: def $edi killed $edi def $rdi
; X64-NEXT: fldt {{[0-9]+}}(%rsp)
; X64-NEXT: fldt {{[0-9]+}}(%rsp)
; X64-NEXT: fstpt -{{[0-9]+}}(%rsp)
; X64-NEXT: fstpt -{{[0-9]+}}(%rsp)
-; X64-NEXT: movl -{{[0-9]+}}(%rsp), %ecx
-; X64-NEXT: movl -{{[0-9]+}}(%rsp), %eax
-; X64-NEXT: movzwl -{{[0-9]+}}(%rsp), %edx
-; X64-NEXT: xorw %cx, %dx
; X64-NEXT: andl $1, %edi
-; X64-NEXT: negl %edi
-; X64-NEXT: andl %edi, %edx
-; X64-NEXT: xorl %ecx, %edx
-; X64-NEXT: movw %dx, -{{[0-9]+}}(%rsp)
-; X64-NEXT: movzwl -{{[0-9]+}}(%rsp), %ecx
-; X64-NEXT: movzwl -{{[0-9]+}}(%rsp), %edx
-; X64-NEXT: xorw %cx, %dx
-; X64-NEXT: andl %edi, %edx
-; X64-NEXT: xorl %ecx, %edx
-; X64-NEXT: movw %dx, -{{[0-9]+}}(%rsp)
-; X64-NEXT: movzwl -{{[0-9]+}}(%rsp), %ecx
-; X64-NEXT: xorw %ax, %cx
-; X64-NEXT: andl %edi, %ecx
-; X64-NEXT: xorl %eax, %ecx
-; X64-NEXT: movw %cx, -{{[0-9]+}}(%rsp)
+; X64-NEXT: negq %rdi
+; X64-NEXT: movq -{{[0-9]+}}(%rsp), %rax
+; X64-NEXT: movq -{{[0-9]+}}(%rsp), %rcx
+; X64-NEXT: xorq %rax, %rcx
+; X64-NEXT: andq %rdi, %rcx
+; X64-NEXT: xorq %rax, %rcx
+; X64-NEXT: movq %rcx, -{{[0-9]+}}(%rsp)
; X64-NEXT: movzwl -{{[0-9]+}}(%rsp), %eax
; X64-NEXT: movzwl -{{[0-9]+}}(%rsp), %ecx
-; X64-NEXT: xorw %ax, %cx
-; X64-NEXT: andl %edi, %ecx
; X64-NEXT: xorl %eax, %ecx
-; X64-NEXT: movw %cx, -{{[0-9]+}}(%rsp)
-; X64-NEXT: movl -{{[0-9]+}}(%rsp), %eax
-; X64-NEXT: movzwl -{{[0-9]+}}(%rsp), %ecx
-; X64-NEXT: xorw %ax, %cx
-; X64-NEXT: andl %edi, %ecx
-; X64-NEXT: xorl %eax, %ecx
-; X64-NEXT: movw %cx, -{{[0-9]+}}(%rsp)
+; X64-NEXT: andl %ecx, %edi
+; X64-NEXT: xorl %eax, %edi
+; X64-NEXT: movw %di, -{{[0-9]+}}(%rsp)
; X64-NEXT: fldt -{{[0-9]+}}(%rsp)
; X64-NEXT: retq
;
@@ -549,35 +532,23 @@ define x86_fp80 @test_ctselect_f80(i1 %cond, x86_fp80 %a, x86_fp80 %b) #0 {
; X32-NEXT: fldt {{[0-9]+}}(%esp)
; X32-NEXT: fstpt {{[0-9]+}}(%esp)
; X32-NEXT: fstpt {{[0-9]+}}(%esp)
-; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
-; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
-; X32-NEXT: movzwl {{[0-9]+}}(%esp), %esi
-; X32-NEXT: xorw %dx, %si
; X32-NEXT: movzbl %al, %eax
; X32-NEXT: negl %eax
-; X32-NEXT: andl %eax, %esi
+; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: movl {{[0-9]+}}(%esp), %esi
; X32-NEXT: xorl %edx, %esi
-; X32-NEXT: movw %si, (%esp)
-; X32-NEXT: movzwl {{[0-9]+}}(%esp), %edx
-; X32-NEXT: movzwl {{[0-9]+}}(%esp), %esi
-; X32-NEXT: xorw %dx, %si
; X32-NEXT: andl %eax, %esi
; X32-NEXT: xorl %edx, %esi
-; X32-NEXT: movw %si, {{[0-9]+}}(%esp)
-; X32-NEXT: movzwl {{[0-9]+}}(%esp), %edx
-; X32-NEXT: xorw %cx, %dx
+; X32-NEXT: movl %esi, (%esp)
+; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NEXT: xorl %ecx, %edx
; X32-NEXT: andl %eax, %edx
; X32-NEXT: xorl %ecx, %edx
-; X32-NEXT: movw %dx, {{[0-9]+}}(%esp)
+; X32-NEXT: movl %edx, {{[0-9]+}}(%esp)
; X32-NEXT: movzwl {{[0-9]+}}(%esp), %ecx
; X32-NEXT: movzwl {{[0-9]+}}(%esp), %edx
-; X32-NEXT: xorw %cx, %dx
-; X32-NEXT: andl %eax, %edx
; X32-NEXT: xorl %ecx, %edx
-; X32-NEXT: movw %dx, {{[0-9]+}}(%esp)
-; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
-; X32-NEXT: movzwl {{[0-9]+}}(%esp), %edx
-; X32-NEXT: xorw %cx, %dx
; X32-NEXT: andl %eax, %edx
; X32-NEXT: xorl %ecx, %edx
; X32-NEXT: movw %dx, {{[0-9]+}}(%esp)
@@ -596,35 +567,23 @@ define x86_fp80 @test_ctselect_f80(i1 %cond, x86_fp80 %a, x86_fp80 %b) #0 {
; X32-NOCMOV-NEXT: fldt {{[0-9]+}}(%esp)
; X32-NOCMOV-NEXT: fstpt {{[0-9]+}}(%esp)
; X32-NOCMOV-NEXT: fstpt {{[0-9]+}}(%esp)
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
-; X32-NOCMOV-NEXT: movzwl {{[0-9]+}}(%esp), %esi
-; X32-NOCMOV-NEXT: xorw %dx, %si
; X32-NOCMOV-NEXT: movzbl %al, %eax
; X32-NOCMOV-NEXT: negl %eax
-; X32-NOCMOV-NEXT: andl %eax, %esi
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %esi
; X32-NOCMOV-NEXT: xorl %edx, %esi
-; X32-NOCMOV-NEXT: movw %si, (%esp)
-; X32-NOCMOV-NEXT: movzwl {{[0-9]+}}(%esp), %edx
-; X32-NOCMOV-NEXT: movzwl {{[0-9]+}}(%esp), %esi
-; X32-NOCMOV-NEXT: xorw %dx, %si
; X32-NOCMOV-NEXT: andl %eax, %esi
; X32-NOCMOV-NEXT: xorl %edx, %esi
-; X32-NOCMOV-NEXT: movw %si, {{[0-9]+}}(%esp)
-; X32-NOCMOV-NEXT: movzwl {{[0-9]+}}(%esp), %edx
-; X32-NOCMOV-NEXT: xorw %cx, %dx
+; X32-NOCMOV-NEXT: movl %esi, (%esp)
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NOCMOV-NEXT: xorl %ecx, %edx
; X32-NOCMOV-NEXT: andl %eax, %edx
; X32-NOCMOV-NEXT: xorl %ecx, %edx
-; X32-NOCMOV-NEXT: movw %dx, {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: movl %edx, {{[0-9]+}}(%esp)
; X32-NOCMOV-NEXT: movzwl {{[0-9]+}}(%esp), %ecx
; X32-NOCMOV-NEXT: movzwl {{[0-9]+}}(%esp), %edx
-; X32-NOCMOV-NEXT: xorw %cx, %dx
-; X32-NOCMOV-NEXT: andl %eax, %edx
; X32-NOCMOV-NEXT: xorl %ecx, %edx
-; X32-NOCMOV-NEXT: movw %dx, {{[0-9]+}}(%esp)
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
-; X32-NOCMOV-NEXT: movzwl {{[0-9]+}}(%esp), %edx
-; X32-NOCMOV-NEXT: xorw %cx, %dx
; X32-NOCMOV-NEXT: andl %eax, %edx
; X32-NOCMOV-NEXT: xorl %ecx, %edx
; X32-NOCMOV-NEXT: movw %dx, {{[0-9]+}}(%esp)
>From ba125cb7e29b2871e0d080d439b93215ae4a5c63 Mon Sep 17 00:00:00 2001
From: AkshayK <iit.akshay at gmail.com>
Date: Thu, 16 Jul 2026 21:38:38 -0400
Subject: [PATCH 07/12] [ConstantTime] Apply review fixes for llvm.ct.select
core
- Verifier: reject aggregate types instead of crashing ISel; add test.
- Fix scalable-vector expansion crash when the element type is not a
legal scalar; add RVV test.
- LangRef: document poison/undef, noundef, permitted folds, and FMF
behavior; rewrite Syntax/Overview.
- Deduplicate CT_SELECT type-legalizer handlers into the SELECT paths.
- Misc: FP int-twin assert, dead null check, in-place memory blend,
flag-free PromoteNode, GlobalISel fallback test, comment cleanups.
---
llvm/docs/LangRef.md | 90 +++++++++-----
llvm/include/llvm/CodeGen/SelectionDAG.h | 2 +-
llvm/include/llvm/IR/Intrinsics.td | 7 +-
llvm/lib/Analysis/InstructionSimplify.cpp | 5 +-
llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp | 15 +--
llvm/lib/CodeGen/SelectionDAG/LegalizeDAG.cpp | 115 ++++++++++-------
.../SelectionDAG/LegalizeFloatTypes.cpp | 34 ++---
.../SelectionDAG/LegalizeIntegerTypes.cpp | 26 ++--
llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h | 5 -
.../SelectionDAG/LegalizeTypesGeneric.cpp | 6 -
.../SelectionDAG/LegalizeVectorTypes.cpp | 18 +--
.../SelectionDAG/SelectionDAGBuilder.cpp | 2 +
llvm/lib/IR/Verifier.cpp | 11 ++
.../CodeGen/RISCV/ctselect-fallback-gisel.ll | 35 ++++++
.../RISCV/ctselect-fallback-scalable.ll | 117 ++++++++++++++++++
llvm/test/CodeGen/X86/ctselect.ll | 48 +++----
llvm/test/Verifier/ct-select.ll | 20 +++
17 files changed, 369 insertions(+), 187 deletions(-)
create mode 100644 llvm/test/CodeGen/RISCV/ctselect-fallback-gisel.ll
create mode 100644 llvm/test/CodeGen/RISCV/ctselect-fallback-scalable.ll
create mode 100644 llvm/test/Verifier/ct-select.ll
diff --git a/llvm/docs/LangRef.md b/llvm/docs/LangRef.md
index 6de11a8d0ae9d..aaf21cdca3097 100644
--- a/llvm/docs/LangRef.md
+++ b/llvm/docs/LangRef.md
@@ -16041,8 +16041,8 @@ catch:
### Constant-Time Intrinsics
These intrinsics are provided to support constant-time operations for
-security-sensitive code. Constant-time operations execute in time independent
-of secret data values, preventing timing side-channel leaks.
+security-sensitive code. Constant-time operations are designed to execute in
+time independent of secret data values, preventing timing side-channel leaks.
(int_ct_select)=
@@ -16050,36 +16050,33 @@ of secret data values, preventing timing side-channel leaks.
##### Syntax:
-This is an overloaded intrinsic. You can use `llvm.ct.select` on any
-integer or floating-point type, pointer types, or vectors types.
+This is an overloaded intrinsic. You can use `llvm.ct.select` on any integer,
+floating-point, or pointer type, or on any vector of those types, including
+scalable vectors. The declarations below are a representative sample:
```
+declare i8 @llvm.ct.select.i8(i1 <cond>, i8 <val1>, i8 <val2>)
declare i32 @llvm.ct.select.i32(i1 <cond>, i32 <val1>, i32 <val2>)
declare i64 @llvm.ct.select.i64(i1 <cond>, i64 <val1>, i64 <val2>)
+declare half @llvm.ct.select.f16(i1 <cond>, half <val1>, half <val2>)
declare float @llvm.ct.select.f32(i1 <cond>, float <val1>, float <val2>)
declare double @llvm.ct.select.f64(i1 <cond>, double <val1>, double <val2>)
+declare fp128 @llvm.ct.select.f128(i1 <cond>, fp128 <val1>, fp128 <val2>)
declare ptr @llvm.ct.select.p0(i1 <cond>, ptr <val1>, ptr <val2>)
-
-; 128-bit vectors
declare <4 x i32> @llvm.ct.select.v4i32(i1 <cond>, <4 x i32> <val1>, <4 x i32> <val2>)
-declare <2 x i64> @llvm.ct.select.v2i64(i1 <cond>, <2 x i64> <val1>, <2 x i64> <val2>)
-declare <4 x float> @llvm.ct.select.v4f32(i1 <cond>, <4 x float> <val1>, <4 x float> <val2>)
declare <2 x double> @llvm.ct.select.v2f64(i1 <cond>, <2 x double> <val1>, <2 x double> <val2>)
-
-; 256-bit vectors
-declare <8 x i32> @llvm.ct.select.v8i32(i1 <cond>, <8 x i32> <val1>, <8 x i32> <val2>)
-declare <8 x float> @llvm.ct.select.v8f32(i1 <cond>, <8 x float> <val1>, <8 x float> <val2>)
-declare <4 x double> @llvm.ct.select.v4f64(i1 <cond>, <4 x double> <val1>, <4 x double> <val2>)
+declare <2 x ptr> @llvm.ct.select.v2p0(i1 <cond>, <2 x ptr> <val1>, <2 x ptr> <val2>)
+declare <vscale x 4 x i32> @llvm.ct.select.nxv4i32(i1 <cond>, <vscale x 4 x i32> <val1>, <vscale x 4 x i32> <val2>)
```
##### Overview:
The '`llvm.ct.select`' family of intrinsic functions selects one of two
-values based on a condition, with the guarantee that the operation executes
-in constant time. Unlike the standard {ref}`select <i_select>` instruction,
-`llvm.ct.select` ensures that the execution time and observable behavior
-do not depend on the condition value, preventing timing-based side-channel
-leaks.
+values based on a condition, like the standard {ref}`select <i_select>`
+instruction, but is lowered to branchless code whose execution time does not
+depend on the condition value. This keeps the condition from leaking through
+timing side channels; see the Semantics section for the exact guarantee and
+its platform requirements.
##### Arguments:
@@ -16089,8 +16086,10 @@ The '`llvm.ct.select`' intrinsic requires three arguments:
{ref}`select <i_select>` which accepts both scalar 'i1' and vector
'`<N x i1>`' conditions, `llvm.ct.select` only accepts a scalar 'i1'
condition. Vector conditions are not supported.
-2. The first value argument of any {ref}`first class <t_firstclass>` type.
- This can be a scalar or vector type.
+2. The first value argument, which must be an integer, floating-point, or
+ pointer type, or a vector of those types. Other
+ {ref}`first class <t_firstclass>` types, such as aggregates, are not
+ supported.
3. The second value argument, which must have the same type as the first
value argument.
@@ -16099,12 +16098,25 @@ The '`llvm.ct.select`' intrinsic requires three arguments:
If the condition evaluates to true, the intrinsic returns the first value
argument; otherwise, it returns the second value argument.
+If any argument is poison, the result is poison. Unlike `select`, this
+includes the unselected value argument, since the lowering blends both
+arguments with bitwise operations. If any argument is undef, the result is
+undef; an undef condition can blend bits from both arguments, so the result
+is not necessarily one of them. `llvm.ct.select` therefore cannot be used as
+a poison barrier.
+
+The return value carries the `noundef` attribute. A call where the condition
+or either value argument is undef or poison would return undef or poison, and
+is therefore undefined behavior. Frontends that cannot rule this out must
+{ref}`freeze <i_freeze>` the operands first.
+
The key semantic difference from {ref}`select <i_select>` is the constant-time
code generation guarantee: the intrinsic must be lowered to machine code that:
- Does not introduce data-dependent control flow based on the condition value
- Executes the same sequence of instructions regardless of the condition value
-- Computes both value arguments before performing the selection
+- Uses both value arguments unconditionally; the not-selected argument is
+ never skipped or branched around
**Platform Requirements:** The constant-time guarantee is conditional on
hardware support for data-independent execution timing. This may be a
@@ -16124,17 +16136,26 @@ mask = sext(cond) ; sign-extend condition to all 1s or all 0s
result = val2 ^ ((val1 ^ val2) & mask)
```
-Targets with native constant-time select support use target-specific
-instructions to generate optimized bitwise operations with stronger guarantees.
-Targets without native support lower the intrinsic to a sequence of generic
-bitwise operations as shown above, structured to resist pattern recognition
-and preserve the constant-time property through optimization passes.
+Targets with native constant-time select support lower the intrinsic to
+suitable target instructions instead. Targets without native support use the
+generic bitwise expansion shown above. In either case the selection is kept
+as a single opaque operation until late in code generation, so optimization
+passes never see a pattern they could rewrite into a select or a conditional
+branch.
Optimizations must preserve the constant-time code generation semantics.
Transforms that would introduce data-dependent control flow are not permitted.
This includes converting to conditional branches, using predicated instructions
-with data-dependent timing, or optimizing away either value argument before the
-selection completes (both paths must be computed).
+with data-dependent timing, or, except as described below, optimizing away
+either value argument before the selection completes.
+
+The call may be folded to one of its value arguments only when the condition
+is a literal `i1` constant or both value arguments are the same SSA value;
+the unused argument then becomes ordinary dead code. Proving the condition
+constant by other means (known-bits, range, or dominating conditions) does
+not permit the fold. Rewrites that keep the `llvm.ct.select`, such as
+swapping the arguments to remove a negated condition, are allowed. Fast-math
+flags on a call to `llvm.ct.select` are ignored.
##### Examples:
@@ -16143,19 +16164,20 @@ selection completes (both paths must be computed).
%x = call i32 @llvm.ct.select.i32(i1 %cond, i32 42, i32 17)
%key = call i64 @llvm.ct.select.i64(i1 %cond, i64 %k_a, i64 %k_b)
-; Constant-time 128-bit integer vector selection (scalar condition broadcast to all lanes)
+; Constant-time integer vector selection (scalar condition broadcast to all lanes)
%v4 = call <4 x i32> @llvm.ct.select.v4i32(i1 %cond,
<4 x i32> <i32 1, i32 2, i32 3, i32 4>,
<4 x i32> <i32 5, i32 6, i32 7, i32 8>)
-; Constant-time 256-bit integer vector selection
-%v8 = call <8 x i32> @llvm.ct.select.v8i32(i1 %cond,
- <8 x i32> %vec_a, <8 x i32> %vec_b)
-
-; Constant-time 256-bit float vector selection
+; Constant-time float vector selection
%v8f = call <8 x float> @llvm.ct.select.v8f32(i1 %cond,
<8 x float> %fvec_a, <8 x float> %fvec_b)
+; Constant-time scalable vector selection
+%sv = call <vscale x 4 x i32> @llvm.ct.select.nxv4i32(i1 %cond,
+ <vscale x 4 x i32> %sv_a,
+ <vscale x 4 x i32> %sv_b)
+
; Constant-time float selection
%f = call float @llvm.ct.select.f32(i1 %cond, float 1.0, float 0.0)
diff --git a/llvm/include/llvm/CodeGen/SelectionDAG.h b/llvm/include/llvm/CodeGen/SelectionDAG.h
index 2fe58e25d4475..12913b9edbc9f 100644
--- a/llvm/include/llvm/CodeGen/SelectionDAG.h
+++ b/llvm/include/llvm/CodeGen/SelectionDAG.h
@@ -1366,7 +1366,7 @@ class SelectionDAG {
SDValue getCTSelect(const SDLoc &DL, EVT VT, SDValue Cond, SDValue LHS,
SDValue RHS) {
assert(LHS.getValueType() == VT && RHS.getValueType() == VT &&
- "Cannot use select on differing types");
+ "Cannot use ct.select on differing types");
assert(!Cond.getValueType().isVector() &&
"ct.select condition must be a scalar");
return getNode(ISD::CT_SELECT, DL, VT, Cond, LHS, RHS);
diff --git a/llvm/include/llvm/IR/Intrinsics.td b/llvm/include/llvm/IR/Intrinsics.td
index a88d682ad68e0..bb0fec8e49929 100644
--- a/llvm/include/llvm/IR/Intrinsics.td
+++ b/llvm/include/llvm/IR/Intrinsics.td
@@ -2074,14 +2074,15 @@ def int_coro_subfn_addr : DefaultAttrsIntrinsic<
// This prevents passes from deleting it when unused, speculating it, or
// rewriting it into an ordinary select, while avoiding claims about writes to
// accessible program memory. The constant-time lowering contract is described
-// in LangRef.
+// in LangRef. llvm_any_ty alone would also admit aggregates, which codegen
+// does not support; the Verifier restricts the value type to integer,
+// floating-point, or pointer types, or vectors of them.
def int_ct_select
: DefaultAttrsIntrinsic<[llvm_any_ty],
[llvm_i1_ty, LLVMMatchType<0>, LLVMMatchType<0>],
[IntrInaccessibleMemOnly, NoUndef<RetIndex>]>;
-
-////===-------------------------- Other Intrinsics --------------------------===//
+///===------------------------- Other Intrinsics --------------------------===//
// TODO: We should introduce a new memory kind fo traps (and other side effects
// we only model to keep things alive).
def int_trap : Intrinsic<[], [],
diff --git a/llvm/lib/Analysis/InstructionSimplify.cpp b/llvm/lib/Analysis/InstructionSimplify.cpp
index 602eb6b63e133..dd83223f351df 100644
--- a/llvm/lib/Analysis/InstructionSimplify.cpp
+++ b/llvm/lib/Analysis/InstructionSimplify.cpp
@@ -7616,8 +7616,9 @@ static Value *simplifyIntrinsic(CallBase *Call, ArrayRef<Value *> Args,
return nullptr;
}
case Intrinsic::ct_select: {
- // Only fold on a literal IR-constant condition or identical arms. Folding
- // through ValueTracking-derived known bits would defeat the constant-time
+ // Only fold on a literal IR-constant condition or identical arms; these
+ // are the only two folds LangRef permits for ct.select. Folding through
+ // ValueTracking-derived known bits would defeat the constant-time
// contract on conditions the user wants kept opaque.
Value *Cond = Args[0], *TrueVal = Args[1], *FalseVal = Args[2];
if (auto *CI = dyn_cast<ConstantInt>(Cond)) {
diff --git a/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp b/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp
index bda6534b34808..cd2abe15ca517 100644
--- a/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp
@@ -13706,17 +13706,10 @@ SDValue DAGCombiner::visitCT_SELECT(SDNode *N) {
return DAG.getCTSelect(DL, VT, F, N2, N1);
if (VT0 == MVT::i1) {
- // Nested CT_SELECT merging optimizations for i1 conditions.
- // These are CT-safe because:
- // 1. AND/OR are bitwise operations that execute in constant time
- // 2. The optimization combines two sequential CT_SELECTs into one,
- // reducing the total number of constant-time operations without
- // changing semantics
- // 3. No data-dependent branches or memory accesses are introduced
- //
+ // Merge nested i1 ct_selects. The AND/OR of the two conditions is itself
+ // constant-time and the result remains a CT_SELECT.
+
// ct_select C0, (ct_select C1, X, Y), Y -> ct_select (C0 & C1), X, Y
- // Semantic equivalence: If C0 is true, evaluate inner select (C1 ? X :
- // Y). If C0 is false, choose Y. This is equivalent to (C0 && C1) ? X : Y.
if (N1.getOpcode() == ISD::CT_SELECT && N1.hasOneUse()) {
SDValue N10 = N1.getOperand(0);
SDValue N11 = N1.getOperand(1);
@@ -13728,8 +13721,6 @@ SDValue DAGCombiner::visitCT_SELECT(SDNode *N) {
}
// ct_select C0, X, (ct_select C1, X, Y) -> ct_select (C0 | C1), X, Y
- // Semantic equivalence: If C0 is true, choose X. If C0 is false, evaluate
- // inner select (C1 ? X : Y). This is equivalent to (C0 || C1) ? X : Y.
if (N2.getOpcode() == ISD::CT_SELECT && N2.hasOneUse()) {
SDValue N20 = N2.getOperand(0);
SDValue N21 = N2.getOperand(1);
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeDAG.cpp b/llvm/lib/CodeGen/SelectionDAG/LegalizeDAG.cpp
index 54715a5a31d03..afa49d0ec372d 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeDAG.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeDAG.cpp
@@ -4351,8 +4351,9 @@ bool SelectionDAGLegalize::ExpandNode(SDNode *Node) {
// and reconstructs a SELECT. The chain edge is defense-in-depth against
// a hypothetical future fold of that form: the dependency partitions the
// bitwise sequence into a region the combiner can't see through. Cost is
- // at most a coalesce-able MOV per call. Swap for ARITH_FENCE when its
- // int/vector extension lands.
+ // at most a coalesce-able MOV per call. ISD::ARITH_FENCE would serve the
+ // same purpose (ISel already selects it for any legal type); switching to
+ // it is left to a follow-up.
Tmp1 = Node->getOperand(0); // cond
Tmp2 = Node->getOperand(1); // T
Tmp3 = Node->getOperand(2); // F
@@ -4361,8 +4362,9 @@ bool SelectionDAGLegalize::ExpandNode(SDNode *Node) {
// Memory-blend for FP scalars whose same-size integer isn't legal (f64 on
// i386 no-SSE, x86_fp80, fp128). The bitcast-to-int expansion below can't
// run since LegalizeDAG is the last legalization stage. Spill T/F, blend
- // chunk-by-chunk at a legal int width, reload as the FP type. Fixed and
- // scalable FP vectors fall through to the unified path below.
+ // chunk-by-chunk at a legal int width in place over the T slot, reload it
+ // as the FP type. Fixed and scalable FP vectors fall through to the
+ // unified path below.
if (VT.isFloatingPoint() && !VT.isVector() &&
!TLI.isTypeLegal(VT.changeTypeToInteger())) {
const DataLayout &DL = DAG.getDataLayout();
@@ -4385,13 +4387,10 @@ bool SelectionDAGLegalize::ExpandNode(SDNode *Node) {
MachineFunction &MF = DAG.getMachineFunction();
SDValue StackT = DAG.CreateStackTemporary(VT);
SDValue StackF = DAG.CreateStackTemporary(VT);
- SDValue StackR = DAG.CreateStackTemporary(VT);
int FIT = cast<FrameIndexSDNode>(StackT.getNode())->getIndex();
int FIF = cast<FrameIndexSDNode>(StackF.getNode())->getIndex();
- int FIR = cast<FrameIndexSDNode>(StackR.getNode())->getIndex();
MachinePointerInfo PIT = MachinePointerInfo::getFixedStack(MF, FIT);
MachinePointerInfo PIF = MachinePointerInfo::getFixedStack(MF, FIF);
- MachinePointerInfo PIR = MachinePointerInfo::getFixedStack(MF, FIR);
SDValue Chain = DAG.getEntryNode();
Chain = DAG.getStore(Chain, dl, Tmp2, StackT, PIT);
@@ -4407,7 +4406,6 @@ bool SelectionDAGLegalize::ExpandNode(SDNode *Node) {
TypeSize Off = TypeSize::getFixed(Offset);
SDValue TPtr = DAG.getMemBasePlusOffset(StackT, Off, dl);
SDValue FPtr = DAG.getMemBasePlusOffset(StackF, Off, dl);
- SDValue RPtr = DAG.getMemBasePlusOffset(StackR, Off, dl);
SDValue Ti, Fi;
if (MemVT == BlendVT) {
@@ -4423,19 +4421,21 @@ bool SelectionDAGLegalize::ExpandNode(SDNode *Node) {
}
Chain = Fi.getValue(1);
- // Blend this chunk via CT_SELECT on the legal integer type. The
- // recursive node will be Expand'd by the scalar-int branch below.
+ // Blend this chunk via CT_SELECT on the legal integer type (the
+ // recursive node is Expand'd by the scalar-int branch below), then
+ // store it back over the now-dead chunk of the T slot. The serial
+ // chain keeps each chunk's loads ahead of the store.
SDValue Ri = DAG.getCTSelect(dl, BlendVT, Tmp1, Ti, Fi);
if (MemVT == BlendVT)
- Chain = DAG.getStore(Chain, dl, Ri, RPtr, PIR.getWithOffset(Offset));
+ Chain = DAG.getStore(Chain, dl, Ri, TPtr, PIT.getWithOffset(Offset));
else
- Chain = DAG.getTruncStore(Chain, dl, Ri, RPtr,
- PIR.getWithOffset(Offset), MemVT);
+ Chain = DAG.getTruncStore(Chain, dl, Ri, TPtr,
+ PIT.getWithOffset(Offset), MemVT);
Offset += ChunkBytes;
}
- Tmp1 = DAG.getLoad(VT, dl, Chain, StackR, PIR);
+ Tmp1 = DAG.getLoad(VT, dl, Chain, StackT, PIT);
Results.push_back(Tmp1);
break;
}
@@ -4447,30 +4447,53 @@ bool SelectionDAGLegalize::ExpandNode(SDNode *Node) {
bool IsFP = VT.isVector() ? VT.getVectorElementType().isFloatingPoint()
: VT.isFloatingPoint();
if (IsFP) {
- if (VT.isVector())
- WorkingVT = EVT::getVectorVT(
- *DAG.getContext(),
- EVT::getIntegerVT(*DAG.getContext(), VT.getScalarSizeInBits()),
- VT.getVectorElementCount());
- else
- WorkingVT = VT.changeTypeToInteger();
+ WorkingVT = VT.changeTypeToInteger();
+ // Scalars with an illegal integer twin took the memory path above; FP
+ // vector types are expected to have a legal integer counterpart.
+ assert(TLI.isTypeLegal(WorkingVT) &&
+ "no legal same-width integer type for CT_SELECT expansion");
WorkingT = DAG.getBitcast(WorkingVT, Tmp2);
WorkingF = DAG.getBitcast(WorkingVT, Tmp3);
}
// Compute the all-ones/all-zeros mask as a scalar, then splat for vectors.
+ // The element type of a legal vector is not always a legal scalar type
+ // (i32 on riscv64, i64 on riscv32), so build the scalar mask at a legal
+ // width: BUILD_VECTOR and SPLAT_VECTOR implicitly truncate a wider
+ // scalar, and if every legal scalar is narrower than the element, splat
+ // at that width and sign-extend the mask vector (0 and -1 survive both).
EVT MaskEltVT =
WorkingVT.isVector() ? WorkingVT.getVectorElementType() : WorkingVT;
+ EVT MaskSclVT = MaskEltVT;
+ if (!TLI.isTypeLegal(MaskSclVT)) {
+ assert(WorkingVT.isVector() && "scalar CT_SELECT type must be legal");
+ for (MVT MV : {MVT::i64, MVT::i32, MVT::i16, MVT::i8})
+ if (TLI.isTypeLegal(MV)) {
+ MaskSclVT = MV;
+ break;
+ }
+ assert(TLI.isTypeLegal(MaskSclVT) && "no legal scalar integer type");
+ }
SDValue ScalarCond = Tmp1;
- if (ScalarCond.getValueType() != MaskEltVT)
- ScalarCond = DAG.getAnyExtOrTrunc(ScalarCond, dl, MaskEltVT);
+ if (ScalarCond.getValueType() != MaskSclVT)
+ ScalarCond = DAG.getAnyExtOrTrunc(ScalarCond, dl, MaskSclVT);
SDValue ScalarMask =
- DAG.getNode(ISD::SUB, dl, MaskEltVT, DAG.getConstant(0, dl, MaskEltVT),
- DAG.getNode(ISD::AND, dl, MaskEltVT, ScalarCond,
- DAG.getConstant(1, dl, MaskEltVT)));
- SDValue Mask = WorkingVT.isVector()
- ? DAG.getSplat(WorkingVT, dl, ScalarMask)
- : ScalarMask;
+ DAG.getNode(ISD::SUB, dl, MaskSclVT, DAG.getConstant(0, dl, MaskSclVT),
+ DAG.getNode(ISD::AND, dl, MaskSclVT, ScalarCond,
+ DAG.getConstant(1, dl, MaskSclVT)));
+ SDValue Mask;
+ if (!WorkingVT.isVector()) {
+ Mask = ScalarMask;
+ } else if (MaskSclVT.bitsGE(MaskEltVT)) {
+ Mask = DAG.getSplat(WorkingVT, dl, ScalarMask);
+ } else {
+ EVT NarrowVT =
+ WorkingVT.changeVectorElementType(*DAG.getContext(), MaskSclVT);
+ assert(TLI.isTypeLegal(NarrowVT) &&
+ "no legal vector type for the CT_SELECT mask splat");
+ Mask = DAG.getNode(ISD::SIGN_EXTEND, dl, WorkingVT,
+ DAG.getSplat(NarrowVT, dl, ScalarMask));
+ }
// F ^ ((T ^ F) & Mask)
SDValue XorTF = DAG.getNode(ISD::XOR, dl, WorkingVT, WorkingT, WorkingF);
@@ -4479,19 +4502,17 @@ bool SelectionDAGLegalize::ExpandNode(SDNode *Node) {
// Forward-looking DAGCombine barrier (see header): route the masked-diff
// through a vreg so the chain edge partitions it from the surrounding
// bitwise ops, guarding against a future combiner that might fold
- // XOR/AND/XOR-with-sext-mask back to SELECT. Skipped where no register
- // class is available (scalable vectors, non-simple or illegal types, or
- // a legal type the target has no register class for).
- if (WorkingVT.isSimple() && !WorkingVT.isScalableVector() &&
- TLI.isTypeLegal(WorkingVT.getSimpleVT())) {
- if (const TargetRegisterClass *RC =
- TLI.getRegClassFor(WorkingVT.getSimpleVT())) {
- MachineRegisterInfo &MRI = DAG.getMachineFunction().getRegInfo();
- Register TMReg = MRI.createVirtualRegister(RC);
- SDValue Chain = DAG.getEntryNode();
- Chain = DAG.getCopyToReg(Chain, dl, TMReg, TM);
- TM = DAG.getCopyFromReg(Chain, dl, TMReg, WorkingVT);
- }
+ // XOR/AND/XOR-with-sext-mask back to SELECT. WorkingVT is legal here, so
+ // it always has a register class; only scalable vectors are skipped,
+ // conservatively.
+ if (!WorkingVT.isScalableVector()) {
+ const TargetRegisterClass *RC =
+ TLI.getRegClassFor(WorkingVT.getSimpleVT());
+ MachineRegisterInfo &MRI = DAG.getMachineFunction().getRegInfo();
+ Register TMReg = MRI.createVirtualRegister(RC);
+ SDValue Chain = DAG.getEntryNode();
+ Chain = DAG.getCopyToReg(Chain, dl, TMReg, TM);
+ TM = DAG.getCopyFromReg(Chain, dl, TMReg, WorkingVT);
}
Tmp1 = DAG.getNode(ISD::XOR, dl, WorkingVT, WorkingF, TM);
@@ -5845,9 +5866,13 @@ void SelectionDAGLegalize::PromoteNode(SDNode *Node) {
// Promote each of the values to the new type.
Tmp2 = DAG.getNode(ExtOp, dl, NVT, Node->getOperand(1));
Tmp3 = DAG.getNode(ExtOp, dl, NVT, Node->getOperand(2));
- // Perform the larger operation, then round down.
- Tmp1 = DAG.getNode(Node->getOpcode(), dl, NVT, Tmp1, Tmp2, Tmp3,
- Node->getFlags());
+ // Perform the larger operation, then round down. CT_SELECT carries no
+ // SDNodeFlags (see ISDOpcodes.h); rebuild it through the helper.
+ if (Node->getOpcode() == ISD::CT_SELECT)
+ Tmp1 = DAG.getCTSelect(dl, NVT, Tmp1, Tmp2, Tmp3);
+ else
+ Tmp1 =
+ DAG.getNode(ISD::SELECT, dl, NVT, Tmp1, Tmp2, Tmp3, Node->getFlags());
if (TruncOp != ISD::FP_ROUND)
Tmp1 = DAG.getNode(TruncOp, dl, Node->getValueType(0), Tmp1);
else
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeFloatTypes.cpp b/llvm/lib/CodeGen/SelectionDAG/LegalizeFloatTypes.cpp
index 28d5279d98453..1c3fdeb74d6b3 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeFloatTypes.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeFloatTypes.cpp
@@ -160,8 +160,8 @@ void DAGTypeLegalizer::SoftenFloatResult(SDNode *N, unsigned ResNo) {
case ISD::LOAD: R = SoftenFloatRes_LOAD(N); break;
case ISD::ATOMIC_LOAD: R = SoftenFloatRes_ATOMIC_LOAD(N); break;
case ISD::ATOMIC_SWAP: R = BitcastToInt_ATOMIC_SWAP(N); break;
- case ISD::SELECT: R = SoftenFloatRes_SELECT(N); break;
- case ISD::CT_SELECT: R = SoftenFloatRes_CT_SELECT(N); break;
+ case ISD::SELECT:
+ case ISD::CT_SELECT: R = SoftenFloatRes_SELECT(N); break;
case ISD::SELECT_CC: R = SoftenFloatRes_SELECT_CC(N); break;
case ISD::FREEZE: R = SoftenFloatRes_FREEZE(N); break;
case ISD::STRICT_SINT_TO_FP:
@@ -963,15 +963,8 @@ SDValue DAGTypeLegalizer::SoftenFloatRes_ATOMIC_LOAD(SDNode *N) {
SDValue DAGTypeLegalizer::SoftenFloatRes_SELECT(SDNode *N) {
SDValue LHS = GetSoftenedFloat(N->getOperand(1));
SDValue RHS = GetSoftenedFloat(N->getOperand(2));
- return DAG.getSelect(SDLoc(N),
- LHS.getValueType(), N->getOperand(0), LHS, RHS);
-}
-
-SDValue DAGTypeLegalizer::SoftenFloatRes_CT_SELECT(SDNode *N) {
- SDValue LHS = GetSoftenedFloat(N->getOperand(1));
- SDValue RHS = GetSoftenedFloat(N->getOperand(2));
- return DAG.getCTSelect(SDLoc(N), LHS.getValueType(), N->getOperand(0), LHS,
- RHS);
+ return DAG.getNode(N->getOpcode(), SDLoc(N), LHS.getValueType(),
+ N->getOperand(0), LHS, RHS);
}
SDValue DAGTypeLegalizer::SoftenFloatRes_SELECT_CC(SDNode *N) {
@@ -1536,8 +1529,8 @@ void DAGTypeLegalizer::ExpandFloatResult(SDNode *N, unsigned ResNo) {
// clang-format off
case ISD::POISON:
case ISD::UNDEF: SplitRes_UNDEF(N, Lo, Hi); break;
- case ISD::SELECT: SplitRes_Select(N, Lo, Hi); break;
- case ISD::CT_SELECT: SplitRes_CT_SELECT(N, Lo, Hi); break;
+ case ISD::SELECT:
+ case ISD::CT_SELECT: SplitRes_Select(N, Lo, Hi); break;
case ISD::SELECT_CC: SplitRes_SELECT_CC(N, Lo, Hi); break;
case ISD::MERGE_VALUES: ExpandRes_MERGE_VALUES(N, ResNo, Lo, Hi); break;
@@ -2589,9 +2582,9 @@ void DAGTypeLegalizer::SoftPromoteHalfResult(SDNode *N, unsigned ResNo) {
case ISD::ATOMIC_LOAD:
R = SoftPromoteHalfRes_ATOMIC_LOAD(N);
break;
- case ISD::SELECT: R = SoftPromoteHalfRes_SELECT(N); break;
+ case ISD::SELECT:
case ISD::CT_SELECT:
- R = SoftPromoteHalfRes_CT_SELECT(N);
+ R = SoftPromoteHalfRes_SELECT(N);
break;
case ISD::SELECT_CC: R = SoftPromoteHalfRes_SELECT_CC(N); break;
case ISD::STRICT_SINT_TO_FP:
@@ -2858,15 +2851,8 @@ SDValue DAGTypeLegalizer::SoftPromoteHalfRes_ATOMIC_LOAD(SDNode *N) {
SDValue DAGTypeLegalizer::SoftPromoteHalfRes_SELECT(SDNode *N) {
SDValue Op1 = GetSoftPromotedHalf(N->getOperand(1));
SDValue Op2 = GetSoftPromotedHalf(N->getOperand(2));
- return DAG.getSelect(SDLoc(N), Op1.getValueType(), N->getOperand(0), Op1, Op2,
- N->getFlags());
-}
-
-SDValue DAGTypeLegalizer::SoftPromoteHalfRes_CT_SELECT(SDNode *N) {
- SDValue Op1 = GetSoftPromotedHalf(N->getOperand(1));
- SDValue Op2 = GetSoftPromotedHalf(N->getOperand(2));
- return DAG.getCTSelect(SDLoc(N), Op1.getValueType(), N->getOperand(0), Op1,
- Op2);
+ return DAG.getNode(N->getOpcode(), SDLoc(N), Op1.getValueType(),
+ N->getOperand(0), Op1, Op2, N->getFlags());
}
SDValue DAGTypeLegalizer::SoftPromoteHalfRes_SELECT_CC(SDNode *N) {
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeIntegerTypes.cpp b/llvm/lib/CodeGen/SelectionDAG/LegalizeIntegerTypes.cpp
index 6f8a2db992d42..61a1c87735b0b 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeIntegerTypes.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeIntegerTypes.cpp
@@ -1995,9 +1995,9 @@ bool DAGTypeLegalizer::PromoteIntegerOperand(SDNode *N, unsigned OpNo) {
Res = PromoteIntOp_ScalarOp(N);
break;
case ISD::VSELECT:
- case ISD::SELECT: Res = PromoteIntOp_SELECT(N, OpNo); break;
+ case ISD::SELECT:
case ISD::CT_SELECT:
- Res = PromoteIntOp_CT_SELECT(N, OpNo);
+ Res = PromoteIntOp_SELECT(N, OpNo);
break;
case ISD::SELECT_CC: Res = PromoteIntOp_SELECT_CC(N, OpNo); break;
case ISD::SETCC: Res = PromoteIntOp_SETCC(N, OpNo); break;
@@ -2401,27 +2401,15 @@ SDValue DAGTypeLegalizer::PromoteIntOp_SELECT(SDNode *N, unsigned OpNo) {
return DAG.getNode(N->getOpcode(), SDLoc(N), N->getValueType(0),
Res, N->getOperand(1), N->getOperand(2));
- // Promote all the way up to the canonical SetCC type.
- EVT OpVT = N->getOpcode() == ISD::SELECT ? OpTy.getScalarType() : OpTy;
+ // Promote all the way up to the canonical SetCC type. SELECT and CT_SELECT
+ // have a scalar condition; VSELECT's mask type matches the vector operands.
+ EVT OpVT = N->getOpcode() == ISD::VSELECT ? OpTy : OpTy.getScalarType();
Cond = PromoteTargetBoolean(Cond, OpVT);
return SDValue(DAG.UpdateNodeOperands(N, Cond, N->getOperand(1),
N->getOperand(2)), 0);
}
-SDValue DAGTypeLegalizer::PromoteIntOp_CT_SELECT(SDNode *N, unsigned OpNo) {
- assert(OpNo == 0 && "Only know how to promote the condition!");
- SDValue Cond = N->getOperand(0);
- EVT OpTy = N->getOperand(1).getValueType();
-
- // Promote all the way up to the canonical SetCC type. The condition is
- // always scalar, so derive the boolean type from the operands' scalar type.
- Cond = PromoteTargetBoolean(Cond, OpTy.getScalarType());
-
- return SDValue(
- DAG.UpdateNodeOperands(N, Cond, N->getOperand(1), N->getOperand(2)), 0);
-}
-
SDValue DAGTypeLegalizer::PromoteIntOp_SELECT_CC(SDNode *N, unsigned OpNo) {
assert(OpNo == 0 && "Don't know how to promote this operand!");
@@ -3029,9 +3017,9 @@ void DAGTypeLegalizer::ExpandIntegerResult(SDNode *N, unsigned ResNo) {
case ISD::ARITH_FENCE: SplitRes_ARITH_FENCE(N, Lo, Hi); break;
case ISD::MERGE_VALUES: SplitRes_MERGE_VALUES(N, ResNo, Lo, Hi); break;
- case ISD::SELECT: SplitRes_Select(N, Lo, Hi); break;
+ case ISD::SELECT:
case ISD::CT_SELECT:
- SplitRes_CT_SELECT(N, Lo, Hi);
+ SplitRes_Select(N, Lo, Hi);
break;
case ISD::SELECT_CC: SplitRes_SELECT_CC(N, Lo, Hi); break;
case ISD::POISON:
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h b/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h
index 10cb2f619ce7e..5396ee06eccda 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h
@@ -382,7 +382,6 @@ class LLVM_LIBRARY_VISIBILITY DAGTypeLegalizer {
SDValue PromoteIntOp_CONCAT_VECTORS(SDNode *N);
SDValue PromoteIntOp_ScalarOp(SDNode *N);
SDValue PromoteIntOp_SELECT(SDNode *N, unsigned OpNo);
- SDValue PromoteIntOp_CT_SELECT(SDNode *N, unsigned OpNo);
SDValue PromoteIntOp_SELECT_CC(SDNode *N, unsigned OpNo);
SDValue PromoteIntOp_SETCC(SDNode *N, unsigned OpNo);
SDValue PromoteIntOp_Shift(SDNode *N);
@@ -623,7 +622,6 @@ class LLVM_LIBRARY_VISIBILITY DAGTypeLegalizer {
SDValue SoftenFloatRes_LOAD(SDNode *N);
SDValue SoftenFloatRes_ATOMIC_LOAD(SDNode *N);
SDValue SoftenFloatRes_SELECT(SDNode *N);
- SDValue SoftenFloatRes_CT_SELECT(SDNode *N);
SDValue SoftenFloatRes_SELECT_CC(SDNode *N);
SDValue SoftenFloatRes_UNDEF(SDNode *N);
SDValue SoftenFloatRes_VAARG(SDNode *N);
@@ -775,7 +773,6 @@ class LLVM_LIBRARY_VISIBILITY DAGTypeLegalizer {
SDValue SoftPromoteHalfRes_LOAD(SDNode *N);
SDValue SoftPromoteHalfRes_ATOMIC_LOAD(SDNode *N);
SDValue SoftPromoteHalfRes_SELECT(SDNode *N);
- SDValue SoftPromoteHalfRes_CT_SELECT(SDNode *N);
SDValue SoftPromoteHalfRes_SELECT_CC(SDNode *N);
SDValue SoftPromoteHalfRes_UnaryOp(SDNode *N);
SDValue SoftPromoteHalfRes_FABS(SDNode *N);
@@ -848,7 +845,6 @@ class LLVM_LIBRARY_VISIBILITY DAGTypeLegalizer {
SDValue ScalarizeVecRes_VECTOR_INTERLEAVE_DEINTERLEAVE(SDNode *N);
SDValue ScalarizeVecRes_VSELECT(SDNode *N);
SDValue ScalarizeVecRes_SELECT(SDNode *N);
- SDValue ScalarizeVecRes_CT_SELECT(SDNode *N);
SDValue ScalarizeVecRes_SELECT_CC(SDNode *N);
SDValue ScalarizeVecRes_SETCC(SDNode *N);
SDValue ScalarizeVecRes_UNDEF(SDNode *N);
@@ -1200,7 +1196,6 @@ class LLVM_LIBRARY_VISIBILITY DAGTypeLegalizer {
void SplitVecRes_AssertSext(SDNode *N, SDValue &Lo, SDValue &Hi);
void SplitRes_ARITH_FENCE (SDNode *N, SDValue &Lo, SDValue &Hi);
void SplitRes_Select(SDNode *N, SDValue &Lo, SDValue &Hi);
- void SplitRes_CT_SELECT(SDNode *N, SDValue &Lo, SDValue &Hi);
void SplitRes_SELECT_CC (SDNode *N, SDValue &Lo, SDValue &Hi);
void SplitRes_UNDEF (SDNode *N, SDValue &Lo, SDValue &Hi);
void SplitRes_FREEZE (SDNode *N, SDValue &Lo, SDValue &Hi);
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeTypesGeneric.cpp b/llvm/lib/CodeGen/SelectionDAG/LegalizeTypesGeneric.cpp
index 5b452eeac4866..8c252c3491540 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeTypesGeneric.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeTypesGeneric.cpp
@@ -569,12 +569,6 @@ void DAGTypeLegalizer::SplitRes_Select(SDNode *N, SDValue &Lo, SDValue &Hi) {
Hi = DAG.getNode(Opcode, dl, LH.getValueType(), CH, LH, RH, EVLHi);
}
-void DAGTypeLegalizer::SplitRes_CT_SELECT(SDNode *N, SDValue &Lo, SDValue &Hi) {
- // Reuse generic select splitting to support scalar and vector conditions.
- // SplitRes_Select rebuilds with N->getOpcode(), so CT_SELECT is preserved.
- SplitRes_Select(N, Lo, Hi);
-}
-
void DAGTypeLegalizer::SplitRes_SELECT_CC(SDNode *N, SDValue &Lo,
SDValue &Hi) {
SDValue LL, LH, RL, RH;
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp b/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp
index 6d25c26cbe914..7deb85ce9c44e 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp
@@ -90,9 +90,9 @@ void DAGTypeLegalizer::ScalarizeVectorResult(SDNode *N, unsigned ResNo) {
break;
case ISD::SIGN_EXTEND_INREG: R = ScalarizeVecRes_InregOp(N); break;
case ISD::VSELECT: R = ScalarizeVecRes_VSELECT(N); break;
- case ISD::SELECT: R = ScalarizeVecRes_SELECT(N); break;
+ case ISD::SELECT:
case ISD::CT_SELECT:
- R = ScalarizeVecRes_CT_SELECT(N);
+ R = ScalarizeVecRes_SELECT(N);
break;
case ISD::SELECT_CC: R = ScalarizeVecRes_SELECT_CC(N); break;
case ISD::SETCC: R = ScalarizeVecRes_SETCC(N); break;
@@ -760,15 +760,9 @@ SDValue DAGTypeLegalizer::ScalarizeVecRes_VSELECT(SDNode *N) {
SDValue DAGTypeLegalizer::ScalarizeVecRes_SELECT(SDNode *N) {
SDValue LHS = GetScalarizedVector(N->getOperand(1));
- return DAG.getSelect(SDLoc(N),
- LHS.getValueType(), N->getOperand(0), LHS,
- GetScalarizedVector(N->getOperand(2)));
-}
-
-SDValue DAGTypeLegalizer::ScalarizeVecRes_CT_SELECT(SDNode *N) {
- SDValue LHS = GetScalarizedVector(N->getOperand(1));
- return DAG.getCTSelect(SDLoc(N), LHS.getValueType(), N->getOperand(0), LHS,
- GetScalarizedVector(N->getOperand(2)));
+ return DAG.getNode(N->getOpcode(), SDLoc(N), LHS.getValueType(),
+ N->getOperand(0), LHS,
+ GetScalarizedVector(N->getOperand(2)));
}
SDValue DAGTypeLegalizer::ScalarizeVecRes_SELECT_CC(SDNode *N) {
@@ -1408,10 +1402,10 @@ void DAGTypeLegalizer::SplitVectorResult(SDNode *N, unsigned ResNo) {
case ISD::AssertSext: SplitVecRes_AssertSext(N, Lo, Hi); break;
case ISD::VSELECT:
case ISD::SELECT:
- case ISD::VP_MERGE:
case ISD::CT_SELECT:
SplitRes_CT_SELECT(N, Lo, Hi);
break;
+ case ISD::VP_MERGE:
case ISD::SELECT_CC: SplitRes_SELECT_CC(N, Lo, Hi); break;
case ISD::POISON:
case ISD::UNDEF: SplitRes_UNDEF(N, Lo, Hi); break;
diff --git a/llvm/lib/CodeGen/SelectionDAG/SelectionDAGBuilder.cpp b/llvm/lib/CodeGen/SelectionDAG/SelectionDAGBuilder.cpp
index 3073a1a19b263..1d42f040e2fd1 100644
--- a/llvm/lib/CodeGen/SelectionDAG/SelectionDAGBuilder.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/SelectionDAGBuilder.cpp
@@ -6906,6 +6906,8 @@ void SelectionDAGBuilder::visitIntrinsicCall(const CallInst &I,
return;
}
case Intrinsic::ct_select: {
+ // Fast-math flags on the call are intentionally dropped: CT_SELECT
+ // carries no SDNodeFlags, so no FMF-driven combine can apply to it.
SDValue Cond = getValue(I.getArgOperand(0));
SDValue A = getValue(I.getArgOperand(1));
SDValue B = getValue(I.getArgOperand(2));
diff --git a/llvm/lib/IR/Verifier.cpp b/llvm/lib/IR/Verifier.cpp
index c60c1b6ac11b6..37eba3c0c6791 100644
--- a/llvm/lib/IR/Verifier.cpp
+++ b/llvm/lib/IR/Verifier.cpp
@@ -6268,6 +6268,17 @@ void Verifier::visitIntrinsicCall(Intrinsic::ID ID, CallBase &Call) {
}
break;
}
+ case Intrinsic::ct_select: {
+ // The value type is only constrained to first-class by tablegen's
+ // llvm_any_ty; SelectionDAG lowering only supports single-value types.
+ Type *ValTy = Call.getType();
+ Check(ValTy->isIntOrIntVectorTy() || ValTy->isFPOrFPVectorTy() ||
+ ValTy->isPtrOrPtrVectorTy(),
+ "llvm.ct.select only supports integer, floating-point, or pointer "
+ "types, or vectors of them",
+ Call);
+ break;
+ }
case Intrinsic::coro_begin:
case Intrinsic::coro_begin_custom_abi:
Check(isa<AnyCoroIdInst>(Call.getArgOperand(0)),
diff --git a/llvm/test/CodeGen/RISCV/ctselect-fallback-gisel.ll b/llvm/test/CodeGen/RISCV/ctselect-fallback-gisel.ll
new file mode 100644
index 0000000000000..756d1154b5ef8
--- /dev/null
+++ b/llvm/test/CodeGen/RISCV/ctselect-fallback-gisel.ll
@@ -0,0 +1,35 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc -mtriple=riscv64 -global-isel -global-isel-abort=0 < %s | FileCheck %s
+
+; llvm.ct.select has no GlobalISel lowering yet. This pins the per-function
+; SelectionDAG fallback: the function must still compile to the branchless
+; expansion rather than failing in the GlobalISel legalizer.
+
+declare i32 @llvm.ct.select.i32(i1, i32, i32)
+declare i64 @llvm.ct.select.i64(i1, i64, i64)
+
+define i32 @ctsel_fallback_i32(i1 %c, i32 %x, i32 %y) {
+; CHECK-LABEL: ctsel_fallback_i32:
+; CHECK: # %bb.0:
+; CHECK-NEXT: xor a1, a1, a2
+; CHECK-NEXT: slli a0, a0, 63
+; CHECK-NEXT: srai a0, a0, 63
+; CHECK-NEXT: and a0, a1, a0
+; CHECK-NEXT: xor a0, a2, a0
+; CHECK-NEXT: ret
+ %r = call i32 @llvm.ct.select.i32(i1 %c, i32 %x, i32 %y)
+ ret i32 %r
+}
+
+define i64 @ctsel_fallback_i64(i1 %c, i64 %x, i64 %y) {
+; CHECK-LABEL: ctsel_fallback_i64:
+; CHECK: # %bb.0:
+; CHECK-NEXT: xor a1, a1, a2
+; CHECK-NEXT: slli a0, a0, 63
+; CHECK-NEXT: srai a0, a0, 63
+; CHECK-NEXT: and a0, a1, a0
+; CHECK-NEXT: xor a0, a2, a0
+; CHECK-NEXT: ret
+ %r = call i64 @llvm.ct.select.i64(i1 %c, i64 %x, i64 %y)
+ ret i64 %r
+}
diff --git a/llvm/test/CodeGen/RISCV/ctselect-fallback-scalable.ll b/llvm/test/CodeGen/RISCV/ctselect-fallback-scalable.ll
new file mode 100644
index 0000000000000..d721ab1d76a40
--- /dev/null
+++ b/llvm/test/CodeGen/RISCV/ctselect-fallback-scalable.ll
@@ -0,0 +1,117 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc -mtriple=riscv64 -mattr=+v < %s | FileCheck %s --check-prefix=RV64V
+; RUN: llc -mtriple=riscv32 -mattr=+v < %s | FileCheck %s --check-prefix=RV32V
+
+; Generic CT_SELECT expansion for scalable vectors. The mask is built at a
+; legal scalar width: elements narrower than it splat with implicit
+; truncation (nxv4i32 on rv64), and elements wider than every legal scalar
+; splat narrow and sign-extend (nxv2i64 on rv32).
+
+declare <vscale x 4 x i32> @llvm.ct.select.nxv4i32(i1, <vscale x 4 x i32>, <vscale x 4 x i32>)
+declare <vscale x 2 x i64> @llvm.ct.select.nxv2i64(i1, <vscale x 2 x i64>, <vscale x 2 x i64>)
+declare <vscale x 4 x float> @llvm.ct.select.nxv4f32(i1, <vscale x 4 x float>, <vscale x 4 x float>)
+declare <vscale x 2 x double> @llvm.ct.select.nxv2f64(i1, <vscale x 2 x double>, <vscale x 2 x double>)
+
+define <vscale x 4 x i32> @ctsel_nxv4i32(i1 %c, <vscale x 4 x i32> %x, <vscale x 4 x i32> %y) {
+; RV64V-LABEL: ctsel_nxv4i32:
+; RV64V: # %bb.0:
+; RV64V-NEXT: vsetvli a1, zero, e32, m2, ta, ma
+; RV64V-NEXT: vxor.vv v8, v8, v10
+; RV64V-NEXT: slli a0, a0, 63
+; RV64V-NEXT: srai a0, a0, 63
+; RV64V-NEXT: vand.vx v8, v8, a0
+; RV64V-NEXT: vxor.vv v8, v10, v8
+; RV64V-NEXT: ret
+;
+; RV32V-LABEL: ctsel_nxv4i32:
+; RV32V: # %bb.0:
+; RV32V-NEXT: vsetvli a1, zero, e32, m2, ta, ma
+; RV32V-NEXT: vxor.vv v8, v8, v10
+; RV32V-NEXT: slli a0, a0, 31
+; RV32V-NEXT: srai a0, a0, 31
+; RV32V-NEXT: vand.vx v8, v8, a0
+; RV32V-NEXT: vxor.vv v8, v10, v8
+; RV32V-NEXT: ret
+ %r = call <vscale x 4 x i32> @llvm.ct.select.nxv4i32(i1 %c, <vscale x 4 x i32> %x, <vscale x 4 x i32> %y)
+ ret <vscale x 4 x i32> %r
+}
+
+define <vscale x 2 x i64> @ctsel_nxv2i64(i1 %c, <vscale x 2 x i64> %x, <vscale x 2 x i64> %y) {
+; RV64V-LABEL: ctsel_nxv2i64:
+; RV64V: # %bb.0:
+; RV64V-NEXT: vsetvli a1, zero, e64, m2, ta, ma
+; RV64V-NEXT: vxor.vv v8, v8, v10
+; RV64V-NEXT: slli a0, a0, 63
+; RV64V-NEXT: srai a0, a0, 63
+; RV64V-NEXT: vand.vx v8, v8, a0
+; RV64V-NEXT: vxor.vv v8, v10, v8
+; RV64V-NEXT: ret
+;
+; RV32V-LABEL: ctsel_nxv2i64:
+; RV32V: # %bb.0:
+; RV32V-NEXT: slli a0, a0, 31
+; RV32V-NEXT: srai a0, a0, 31
+; RV32V-NEXT: vsetvli a1, zero, e64, m2, ta, ma
+; RV32V-NEXT: vxor.vv v8, v8, v10
+; RV32V-NEXT: vsetvli zero, zero, e32, m1, ta, ma
+; RV32V-NEXT: vmv.v.x v14, a0
+; RV32V-NEXT: vsetvli zero, zero, e64, m2, ta, ma
+; RV32V-NEXT: vsext.vf2 v12, v14
+; RV32V-NEXT: vand.vv v8, v8, v12
+; RV32V-NEXT: vxor.vv v8, v10, v8
+; RV32V-NEXT: ret
+ %r = call <vscale x 2 x i64> @llvm.ct.select.nxv2i64(i1 %c, <vscale x 2 x i64> %x, <vscale x 2 x i64> %y)
+ ret <vscale x 2 x i64> %r
+}
+
+define <vscale x 4 x float> @ctsel_nxv4f32(i1 %c, <vscale x 4 x float> %x, <vscale x 4 x float> %y) {
+; RV64V-LABEL: ctsel_nxv4f32:
+; RV64V: # %bb.0:
+; RV64V-NEXT: vsetvli a1, zero, e32, m2, ta, ma
+; RV64V-NEXT: vxor.vv v8, v8, v10
+; RV64V-NEXT: slli a0, a0, 63
+; RV64V-NEXT: srai a0, a0, 63
+; RV64V-NEXT: vand.vx v8, v8, a0
+; RV64V-NEXT: vxor.vv v8, v10, v8
+; RV64V-NEXT: ret
+;
+; RV32V-LABEL: ctsel_nxv4f32:
+; RV32V: # %bb.0:
+; RV32V-NEXT: vsetvli a1, zero, e32, m2, ta, ma
+; RV32V-NEXT: vxor.vv v8, v8, v10
+; RV32V-NEXT: slli a0, a0, 31
+; RV32V-NEXT: srai a0, a0, 31
+; RV32V-NEXT: vand.vx v8, v8, a0
+; RV32V-NEXT: vxor.vv v8, v10, v8
+; RV32V-NEXT: ret
+ %r = call <vscale x 4 x float> @llvm.ct.select.nxv4f32(i1 %c, <vscale x 4 x float> %x, <vscale x 4 x float> %y)
+ ret <vscale x 4 x float> %r
+}
+
+define <vscale x 2 x double> @ctsel_nxv2f64(i1 %c, <vscale x 2 x double> %x, <vscale x 2 x double> %y) {
+; RV64V-LABEL: ctsel_nxv2f64:
+; RV64V: # %bb.0:
+; RV64V-NEXT: vsetvli a1, zero, e64, m2, ta, ma
+; RV64V-NEXT: vxor.vv v8, v8, v10
+; RV64V-NEXT: slli a0, a0, 63
+; RV64V-NEXT: srai a0, a0, 63
+; RV64V-NEXT: vand.vx v8, v8, a0
+; RV64V-NEXT: vxor.vv v8, v10, v8
+; RV64V-NEXT: ret
+;
+; RV32V-LABEL: ctsel_nxv2f64:
+; RV32V: # %bb.0:
+; RV32V-NEXT: slli a0, a0, 31
+; RV32V-NEXT: srai a0, a0, 31
+; RV32V-NEXT: vsetvli a1, zero, e64, m2, ta, ma
+; RV32V-NEXT: vxor.vv v8, v8, v10
+; RV32V-NEXT: vsetvli zero, zero, e32, m1, ta, ma
+; RV32V-NEXT: vmv.v.x v14, a0
+; RV32V-NEXT: vsetvli zero, zero, e64, m2, ta, ma
+; RV32V-NEXT: vsext.vf2 v12, v14
+; RV32V-NEXT: vand.vv v8, v8, v12
+; RV32V-NEXT: vxor.vv v8, v10, v8
+; RV32V-NEXT: ret
+ %r = call <vscale x 2 x double> @llvm.ct.select.nxv2f64(i1 %c, <vscale x 2 x double> %x, <vscale x 2 x double> %y)
+ ret <vscale x 2 x double> %r
+}
diff --git a/llvm/test/CodeGen/X86/ctselect.ll b/llvm/test/CodeGen/X86/ctselect.ll
index 5b23642b1c5d0..2246530619150 100644
--- a/llvm/test/CodeGen/X86/ctselect.ll
+++ b/llvm/test/CodeGen/X86/ctselect.ll
@@ -217,18 +217,18 @@ define double @test_ctselect_f64(i1 %cond, double %a, double %b) #0 {
; X32-LABEL: test_ctselect_f64:
; X32: # %bb.0:
; X32-NEXT: pushl %esi
-; X32-NEXT: subl $24, %esp
+; X32-NEXT: subl $16, %esp
; X32-NEXT: movzbl {{[0-9]+}}(%esp), %eax
; X32-NEXT: andb $1, %al
; X32-NEXT: fldl {{[0-9]+}}(%esp)
; X32-NEXT: fldl {{[0-9]+}}(%esp)
; X32-NEXT: fstpl {{[0-9]+}}(%esp)
-; X32-NEXT: fstpl {{[0-9]+}}(%esp)
+; X32-NEXT: fstpl (%esp)
; X32-NEXT: movzbl %al, %eax
; X32-NEXT: negl %eax
; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
-; X32-NEXT: movl {{[0-9]+}}(%esp), %esi
+; X32-NEXT: movl (%esp), %esi
; X32-NEXT: xorl %edx, %esi
; X32-NEXT: andl %eax, %esi
; X32-NEXT: xorl %edx, %esi
@@ -239,25 +239,25 @@ define double @test_ctselect_f64(i1 %cond, double %a, double %b) #0 {
; X32-NEXT: xorl %ecx, %edx
; X32-NEXT: movl %edx, {{[0-9]+}}(%esp)
; X32-NEXT: fldl (%esp)
-; X32-NEXT: addl $24, %esp
+; X32-NEXT: addl $16, %esp
; X32-NEXT: popl %esi
; X32-NEXT: retl
;
; X32-NOCMOV-LABEL: test_ctselect_f64:
; X32-NOCMOV: # %bb.0:
; X32-NOCMOV-NEXT: pushl %esi
-; X32-NOCMOV-NEXT: subl $24, %esp
+; X32-NOCMOV-NEXT: subl $16, %esp
; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %eax
; X32-NOCMOV-NEXT: andb $1, %al
; X32-NOCMOV-NEXT: fldl {{[0-9]+}}(%esp)
; X32-NOCMOV-NEXT: fldl {{[0-9]+}}(%esp)
; X32-NOCMOV-NEXT: fstpl {{[0-9]+}}(%esp)
-; X32-NOCMOV-NEXT: fstpl {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: fstpl (%esp)
; X32-NOCMOV-NEXT: movzbl %al, %eax
; X32-NOCMOV-NEXT: negl %eax
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %esi
+; X32-NOCMOV-NEXT: movl (%esp), %esi
; X32-NOCMOV-NEXT: xorl %edx, %esi
; X32-NOCMOV-NEXT: andl %eax, %esi
; X32-NOCMOV-NEXT: xorl %edx, %esi
@@ -268,7 +268,7 @@ define double @test_ctselect_f64(i1 %cond, double %a, double %b) #0 {
; X32-NOCMOV-NEXT: xorl %ecx, %edx
; X32-NOCMOV-NEXT: movl %edx, {{[0-9]+}}(%esp)
; X32-NOCMOV-NEXT: fldl (%esp)
-; X32-NOCMOV-NEXT: addl $24, %esp
+; X32-NOCMOV-NEXT: addl $16, %esp
; X32-NOCMOV-NEXT: popl %esi
; X32-NOCMOV-NEXT: retl
%result = call double @llvm.ct.select.f64(i1 %cond, double %a, double %b)
@@ -525,18 +525,18 @@ define x86_fp80 @test_ctselect_f80(i1 %cond, x86_fp80 %a, x86_fp80 %b) #0 {
; X32-LABEL: test_ctselect_f80:
; X32: # %bb.0:
; X32-NEXT: pushl %esi
-; X32-NEXT: subl $36, %esp
+; X32-NEXT: subl $24, %esp
; X32-NEXT: movzbl {{[0-9]+}}(%esp), %eax
; X32-NEXT: andb $1, %al
; X32-NEXT: fldt {{[0-9]+}}(%esp)
; X32-NEXT: fldt {{[0-9]+}}(%esp)
; X32-NEXT: fstpt {{[0-9]+}}(%esp)
-; X32-NEXT: fstpt {{[0-9]+}}(%esp)
+; X32-NEXT: fstpt (%esp)
; X32-NEXT: movzbl %al, %eax
; X32-NEXT: negl %eax
; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
-; X32-NEXT: movl {{[0-9]+}}(%esp), %esi
+; X32-NEXT: movl (%esp), %esi
; X32-NEXT: xorl %edx, %esi
; X32-NEXT: andl %eax, %esi
; X32-NEXT: xorl %edx, %esi
@@ -553,25 +553,25 @@ define x86_fp80 @test_ctselect_f80(i1 %cond, x86_fp80 %a, x86_fp80 %b) #0 {
; X32-NEXT: xorl %ecx, %edx
; X32-NEXT: movw %dx, {{[0-9]+}}(%esp)
; X32-NEXT: fldt (%esp)
-; X32-NEXT: addl $36, %esp
+; X32-NEXT: addl $24, %esp
; X32-NEXT: popl %esi
; X32-NEXT: retl
;
; X32-NOCMOV-LABEL: test_ctselect_f80:
; X32-NOCMOV: # %bb.0:
; X32-NOCMOV-NEXT: pushl %esi
-; X32-NOCMOV-NEXT: subl $36, %esp
+; X32-NOCMOV-NEXT: subl $24, %esp
; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %eax
; X32-NOCMOV-NEXT: andb $1, %al
; X32-NOCMOV-NEXT: fldt {{[0-9]+}}(%esp)
; X32-NOCMOV-NEXT: fldt {{[0-9]+}}(%esp)
; X32-NOCMOV-NEXT: fstpt {{[0-9]+}}(%esp)
-; X32-NOCMOV-NEXT: fstpt {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: fstpt (%esp)
; X32-NOCMOV-NEXT: movzbl %al, %eax
; X32-NOCMOV-NEXT: negl %eax
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %esi
+; X32-NOCMOV-NEXT: movl (%esp), %esi
; X32-NOCMOV-NEXT: xorl %edx, %esi
; X32-NOCMOV-NEXT: andl %eax, %esi
; X32-NOCMOV-NEXT: xorl %edx, %esi
@@ -588,7 +588,7 @@ define x86_fp80 @test_ctselect_f80(i1 %cond, x86_fp80 %a, x86_fp80 %b) #0 {
; X32-NOCMOV-NEXT: xorl %ecx, %edx
; X32-NOCMOV-NEXT: movw %dx, {{[0-9]+}}(%esp)
; X32-NOCMOV-NEXT: fldt (%esp)
-; X32-NOCMOV-NEXT: addl $36, %esp
+; X32-NOCMOV-NEXT: addl $24, %esp
; X32-NOCMOV-NEXT: popl %esi
; X32-NOCMOV-NEXT: retl
%result = call x86_fp80 @llvm.ct.select.f80(i1 %cond, x86_fp80 %a, x86_fp80 %b)
@@ -2374,18 +2374,18 @@ define double @test_ctselect_f64_nan_inf(i1 %cond) #0 {
; X32-LABEL: test_ctselect_f64_nan_inf:
; X32: # %bb.0:
; X32-NEXT: pushl %esi
-; X32-NEXT: subl $24, %esp
+; X32-NEXT: subl $16, %esp
; X32-NEXT: movzbl {{[0-9]+}}(%esp), %eax
; X32-NEXT: andb $1, %al
; X32-NEXT: flds {{\.?LCPI[0-9]+_[0-9]+}}
; X32-NEXT: fstpl {{[0-9]+}}(%esp)
; X32-NEXT: flds {{\.?LCPI[0-9]+_[0-9]+}}
-; X32-NEXT: fstpl {{[0-9]+}}(%esp)
+; X32-NEXT: fstpl (%esp)
; X32-NEXT: movzbl %al, %eax
; X32-NEXT: negl %eax
; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
-; X32-NEXT: movl {{[0-9]+}}(%esp), %esi
+; X32-NEXT: movl (%esp), %esi
; X32-NEXT: xorl %edx, %esi
; X32-NEXT: andl %eax, %esi
; X32-NEXT: xorl %edx, %esi
@@ -2396,25 +2396,25 @@ define double @test_ctselect_f64_nan_inf(i1 %cond) #0 {
; X32-NEXT: xorl %ecx, %edx
; X32-NEXT: movl %edx, {{[0-9]+}}(%esp)
; X32-NEXT: fldl (%esp)
-; X32-NEXT: addl $24, %esp
+; X32-NEXT: addl $16, %esp
; X32-NEXT: popl %esi
; X32-NEXT: retl
;
; X32-NOCMOV-LABEL: test_ctselect_f64_nan_inf:
; X32-NOCMOV: # %bb.0:
; X32-NOCMOV-NEXT: pushl %esi
-; X32-NOCMOV-NEXT: subl $24, %esp
+; X32-NOCMOV-NEXT: subl $16, %esp
; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %eax
; X32-NOCMOV-NEXT: andb $1, %al
; X32-NOCMOV-NEXT: flds {{\.?LCPI[0-9]+_[0-9]+}}
; X32-NOCMOV-NEXT: fstpl {{[0-9]+}}(%esp)
; X32-NOCMOV-NEXT: flds {{\.?LCPI[0-9]+_[0-9]+}}
-; X32-NOCMOV-NEXT: fstpl {{[0-9]+}}(%esp)
+; X32-NOCMOV-NEXT: fstpl (%esp)
; X32-NOCMOV-NEXT: movzbl %al, %eax
; X32-NOCMOV-NEXT: negl %eax
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %esi
+; X32-NOCMOV-NEXT: movl (%esp), %esi
; X32-NOCMOV-NEXT: xorl %edx, %esi
; X32-NOCMOV-NEXT: andl %eax, %esi
; X32-NOCMOV-NEXT: xorl %edx, %esi
@@ -2425,7 +2425,7 @@ define double @test_ctselect_f64_nan_inf(i1 %cond) #0 {
; X32-NOCMOV-NEXT: xorl %ecx, %edx
; X32-NOCMOV-NEXT: movl %edx, {{[0-9]+}}(%esp)
; X32-NOCMOV-NEXT: fldl (%esp)
-; X32-NOCMOV-NEXT: addl $24, %esp
+; X32-NOCMOV-NEXT: addl $16, %esp
; X32-NOCMOV-NEXT: popl %esi
; X32-NOCMOV-NEXT: retl
%result = call double @llvm.ct.select.f64(i1 %cond, double 0x7FF8000000000000, double 0x7FF0000000000000)
diff --git a/llvm/test/Verifier/ct-select.ll b/llvm/test/Verifier/ct-select.ll
new file mode 100644
index 0000000000000..56f897a8a53c3
--- /dev/null
+++ b/llvm/test/Verifier/ct-select.ll
@@ -0,0 +1,20 @@
+; RUN: not llvm-as -disable-output < %s 2>&1 | FileCheck %s
+
+; ct.select's value type is restricted to single-value types: aggregates are
+; accepted by the intrinsic signature (llvm_any_ty) but rejected here because
+; codegen does not support them.
+
+declare {i32, i32} @llvm.ct.select.sl_i32i32s(i1, {i32, i32}, {i32, i32})
+declare [2 x i64] @llvm.ct.select.a2i64(i1, [2 x i64], [2 x i64])
+
+; CHECK: llvm.ct.select only supports integer, floating-point, or pointer types, or vectors of them
+define {i32, i32} @ct_select_struct(i1 %c, {i32, i32} %x, {i32, i32} %y) {
+ %r = call {i32, i32} @llvm.ct.select.sl_i32i32s(i1 %c, {i32, i32} %x, {i32, i32} %y)
+ ret {i32, i32} %r
+}
+
+; CHECK: llvm.ct.select only supports integer, floating-point, or pointer types, or vectors of them
+define [2 x i64] @ct_select_array(i1 %c, [2 x i64] %x, [2 x i64] %y) {
+ %r = call [2 x i64] @llvm.ct.select.a2i64(i1 %c, [2 x i64] %x, [2 x i64] %y)
+ ret [2 x i64] %r
+}
>From a3f2ffa47efc62accafd3e4f18773b1d16bd4ceb Mon Sep 17 00:00:00 2001
From: AkshayK <iit.akshay at gmail.com>
Date: Wed, 26 Aug 2026 16:34:01 -0400
Subject: [PATCH 08/12] [ConstantTime] Address reviewer feedback on
llvm.ct.select core
- Fix vector-split build error: call the existing SplitRes_Select for
CT_SELECT instead of the nonexistent SplitRes_CT_SELECT.
- Fix a VP_MERGE result-split regression: the CT_SELECT switch reformat had
moved ISD::VP_MERGE onto the SplitRes_SELECT_CC line, which reads operand 4
(out of bounds for VP_MERGE's 4 operands) and builds a SELECT_CC without
splitting the EVL, crashing or miscompiling any vp.merge whose result
vector must be split. Restore it to the SplitRes_Select group.
- Accept byte types (bN and vectors of them) in the Verifier and document
them in LangRef; add byte-typed X86 codegen and Verifier test coverage.
- Drop the redundant `VT0 == MVT::i1` guard in visitCT_SELECT; the inner
condition-type checks already cover correctness after promotion.
- LangRef: reword the constant-fold rule to the enforceable "constant
operand" form, restate fast-math flags via the general FP-call rule (only
nnan/ninf are poison-generating), and switch undef/poison to match select,
dropping the noundef return attribute so poison propagates (both arms always
evaluate, so poison in either yields poison).
- Revert an unrelated whitespace change in LegalizeTypes.h.
- Drop redundant intrinsic declarations from the InstSimplify test and soften
the SelectionDAGBuilder fast-math comment.
---
llvm/docs/LangRef.md | 46 +--
llvm/include/llvm/IR/Intrinsics.td | 4 +-
llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp | 43 ++-
llvm/lib/CodeGen/SelectionDAG/LegalizeDAG.cpp | 6 +-
llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h | 2 +-
.../SelectionDAG/LegalizeVectorTypes.cpp | 4 +-
.../SelectionDAG/SelectionDAGBuilder.cpp | 6 +-
llvm/lib/IR/Verifier.cpp | 6 +-
llvm/test/CodeGen/X86/ctselect.ll | 354 ++++++++++++++++++
.../test/Transforms/InstSimplify/ct-select.ll | 12 +-
llvm/test/Verifier/ct-select.ll | 4 +-
11 files changed, 417 insertions(+), 70 deletions(-)
diff --git a/llvm/docs/LangRef.md b/llvm/docs/LangRef.md
index aaf21cdca3097..0457672ef496c 100644
--- a/llvm/docs/LangRef.md
+++ b/llvm/docs/LangRef.md
@@ -16051,11 +16051,12 @@ time independent of secret data values, preventing timing side-channel leaks.
##### Syntax:
This is an overloaded intrinsic. You can use `llvm.ct.select` on any integer,
-floating-point, or pointer type, or on any vector of those types, including
-scalable vectors. The declarations below are a representative sample:
+byte, floating-point, or pointer type, or on any vector of those types,
+including scalable vectors. The declarations below are a representative sample:
```
declare i8 @llvm.ct.select.i8(i1 <cond>, i8 <val1>, i8 <val2>)
+declare b8 @llvm.ct.select.b8(i1 <cond>, b8 <val1>, b8 <val2>)
declare i32 @llvm.ct.select.i32(i1 <cond>, i32 <val1>, i32 <val2>)
declare i64 @llvm.ct.select.i64(i1 <cond>, i64 <val1>, i64 <val2>)
declare half @llvm.ct.select.f16(i1 <cond>, half <val1>, half <val2>)
@@ -16086,8 +16087,8 @@ The '`llvm.ct.select`' intrinsic requires three arguments:
{ref}`select <i_select>` which accepts both scalar 'i1' and vector
'`<N x i1>`' conditions, `llvm.ct.select` only accepts a scalar 'i1'
condition. Vector conditions are not supported.
-2. The first value argument, which must be an integer, floating-point, or
- pointer type, or a vector of those types. Other
+2. The first value argument, which must be an integer, byte, floating-point,
+ or pointer type, or a vector of those types. Other
{ref}`first class <t_firstclass>` types, such as aggregates, are not
supported.
3. The second value argument, which must have the same type as the first
@@ -16098,17 +16099,10 @@ The '`llvm.ct.select`' intrinsic requires three arguments:
If the condition evaluates to true, the intrinsic returns the first value
argument; otherwise, it returns the second value argument.
-If any argument is poison, the result is poison. Unlike `select`, this
-includes the unselected value argument, since the lowering blends both
-arguments with bitwise operations. If any argument is undef, the result is
-undef; an undef condition can blend bits from both arguments, so the result
-is not necessarily one of them. `llvm.ct.select` therefore cannot be used as
-a poison barrier.
-
-The return value carries the `noundef` attribute. A call where the condition
-or either value argument is undef or poison would return undef or poison, and
-is therefore undefined behavior. Frontends that cannot rule this out must
-{ref}`freeze <i_freeze>` the operands first.
+Poison and undef propagate as for {ref}`select <i_select>`: a `poison`
+condition yields `poison`, an `undef` condition yields `undef`. Unlike
+`select`, both value arguments are always evaluated, so a `poison` value in
+either one yields `poison`.
The key semantic difference from {ref}`select <i_select>` is the constant-time
code generation guarantee: the intrinsic must be lowered to machine code that:
@@ -16149,13 +16143,21 @@ This includes converting to conditional branches, using predicated instructions
with data-dependent timing, or, except as described below, optimizing away
either value argument before the selection completes.
-The call may be folded to one of its value arguments only when the condition
-is a literal `i1` constant or both value arguments are the same SSA value;
-the unused argument then becomes ordinary dead code. Proving the condition
-constant by other means (known-bits, range, or dominating conditions) does
-not permit the fold. Rewrites that keep the `llvm.ct.select`, such as
-swapping the arguments to remove a negated condition, are allowed. Fast-math
-flags on a call to `llvm.ct.select` are ignored.
+The call folds to one of its value arguments when the condition operand is a
+constant `i1`, or when both value arguments are the same value; the unused
+argument then becomes ordinary dead code. Optimizers must not derive the fold
+by running value analyses (known-bits, range, or dominating conditions) on a
+non-constant condition. A condition that another pass has already proved to be
+a compile-time constant is no longer secret-dependent, so folding it is
+permitted. Rewrites that keep the `llvm.ct.select`, such as swapping the
+arguments to remove a negated condition, are allowed.
+
+Like other floating-point calls, `llvm.ct.select` may carry fast-math flags
+when it returns a supported floating-point type; the poison-generating flags
+(`nnan`, `ninf`) apply to both value arguments (the lowering blends both),
+and the optimizer may use them. No fast-math flag causes
+the intrinsic itself to be algebraically rewritten, so its constant-time
+lowering is unaffected.
##### Examples:
diff --git a/llvm/include/llvm/IR/Intrinsics.td b/llvm/include/llvm/IR/Intrinsics.td
index bb0fec8e49929..0533074a3f863 100644
--- a/llvm/include/llvm/IR/Intrinsics.td
+++ b/llvm/include/llvm/IR/Intrinsics.td
@@ -2075,12 +2075,12 @@ def int_coro_subfn_addr : DefaultAttrsIntrinsic<
// rewriting it into an ordinary select, while avoiding claims about writes to
// accessible program memory. The constant-time lowering contract is described
// in LangRef. llvm_any_ty alone would also admit aggregates, which codegen
-// does not support; the Verifier restricts the value type to integer,
+// does not support; the Verifier restricts the value type to integer, byte,
// floating-point, or pointer types, or vectors of them.
def int_ct_select
: DefaultAttrsIntrinsic<[llvm_any_ty],
[llvm_i1_ty, LLVMMatchType<0>, LLVMMatchType<0>],
- [IntrInaccessibleMemOnly, NoUndef<RetIndex>]>;
+ [IntrInaccessibleMemOnly]>;
///===------------------------- Other Intrinsics --------------------------===//
// TODO: We should introduce a new memory kind fo traps (and other side effects
diff --git a/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp b/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp
index cd2abe15ca517..374ba6cd09439 100644
--- a/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp
@@ -13695,7 +13695,6 @@ SDValue DAGCombiner::visitCT_SELECT(SDNode *N) {
SDValue N1 = N->getOperand(1);
SDValue N2 = N->getOperand(2);
EVT VT = N->getValueType(0);
- EVT VT0 = N0.getValueType();
SDLoc DL(N);
// ct_select (not Cond), N1, N2 -> ct_select Cond, N2, N1
@@ -13705,30 +13704,30 @@ SDValue DAGCombiner::visitCT_SELECT(SDNode *N) {
if (SDValue F = extractBooleanFlip(N0, DAG, TLI, false))
return DAG.getCTSelect(DL, VT, F, N2, N1);
- if (VT0 == MVT::i1) {
- // Merge nested i1 ct_selects. The AND/OR of the two conditions is itself
- // constant-time and the result remains a CT_SELECT.
+ // Merge nested ct_selects. The AND/OR of the two conditions is itself
+ // constant-time and the result remains a CT_SELECT. The condition-type
+ // equality checks below keep this correct regardless of whether the scalar
+ // bool condition has been promoted past i1 by type legalization.
- // ct_select C0, (ct_select C1, X, Y), Y -> ct_select (C0 & C1), X, Y
- if (N1.getOpcode() == ISD::CT_SELECT && N1.hasOneUse()) {
- SDValue N10 = N1.getOperand(0);
- SDValue N11 = N1.getOperand(1);
- SDValue N12 = N1.getOperand(2);
- if (N12 == N2 && N0.getValueType() == N10.getValueType()) {
- SDValue And = DAG.getNode(ISD::AND, DL, N0.getValueType(), N0, N10);
- return DAG.getCTSelect(DL, N1.getValueType(), And, N11, N2);
- }
+ // ct_select C0, (ct_select C1, X, Y), Y -> ct_select (C0 & C1), X, Y
+ if (N1.getOpcode() == ISD::CT_SELECT && N1.hasOneUse()) {
+ SDValue N10 = N1.getOperand(0);
+ SDValue N11 = N1.getOperand(1);
+ SDValue N12 = N1.getOperand(2);
+ if (N12 == N2 && N0.getValueType() == N10.getValueType()) {
+ SDValue And = DAG.getNode(ISD::AND, DL, N0.getValueType(), N0, N10);
+ return DAG.getCTSelect(DL, N1.getValueType(), And, N11, N2);
}
+ }
- // ct_select C0, X, (ct_select C1, X, Y) -> ct_select (C0 | C1), X, Y
- if (N2.getOpcode() == ISD::CT_SELECT && N2.hasOneUse()) {
- SDValue N20 = N2.getOperand(0);
- SDValue N21 = N2.getOperand(1);
- SDValue N22 = N2.getOperand(2);
- if (N21 == N1 && N0.getValueType() == N20.getValueType()) {
- SDValue Or = DAG.getNode(ISD::OR, DL, N0.getValueType(), N0, N20);
- return DAG.getCTSelect(DL, N1.getValueType(), Or, N1, N22);
- }
+ // ct_select C0, X, (ct_select C1, X, Y) -> ct_select (C0 | C1), X, Y
+ if (N2.getOpcode() == ISD::CT_SELECT && N2.hasOneUse()) {
+ SDValue N20 = N2.getOperand(0);
+ SDValue N21 = N2.getOperand(1);
+ SDValue N22 = N2.getOperand(2);
+ if (N21 == N1 && N0.getValueType() == N20.getValueType()) {
+ SDValue Or = DAG.getNode(ISD::OR, DL, N0.getValueType(), N0, N20);
+ return DAG.getCTSelect(DL, N1.getValueType(), Or, N1, N22);
}
}
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeDAG.cpp b/llvm/lib/CodeGen/SelectionDAG/LegalizeDAG.cpp
index afa49d0ec372d..708d7c7df12a9 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeDAG.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeDAG.cpp
@@ -4467,11 +4467,7 @@ bool SelectionDAGLegalize::ExpandNode(SDNode *Node) {
EVT MaskSclVT = MaskEltVT;
if (!TLI.isTypeLegal(MaskSclVT)) {
assert(WorkingVT.isVector() && "scalar CT_SELECT type must be legal");
- for (MVT MV : {MVT::i64, MVT::i32, MVT::i16, MVT::i8})
- if (TLI.isTypeLegal(MV)) {
- MaskSclVT = MV;
- break;
- }
+ MaskSclVT = TLI.getLegalTypeToTransformTo(*DAG.getContext(), MaskSclVT);
assert(TLI.isTypeLegal(MaskSclVT) && "no legal scalar integer type");
}
SDValue ScalarCond = Tmp1;
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h b/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h
index 5396ee06eccda..f1458b22532a5 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h
@@ -1195,7 +1195,7 @@ class LLVM_LIBRARY_VISIBILITY DAGTypeLegalizer {
void SplitVecRes_AssertZext(SDNode *N, SDValue &Lo, SDValue &Hi);
void SplitVecRes_AssertSext(SDNode *N, SDValue &Lo, SDValue &Hi);
void SplitRes_ARITH_FENCE (SDNode *N, SDValue &Lo, SDValue &Hi);
- void SplitRes_Select(SDNode *N, SDValue &Lo, SDValue &Hi);
+ void SplitRes_Select (SDNode *N, SDValue &Lo, SDValue &Hi);
void SplitRes_SELECT_CC (SDNode *N, SDValue &Lo, SDValue &Hi);
void SplitRes_UNDEF (SDNode *N, SDValue &Lo, SDValue &Hi);
void SplitRes_FREEZE (SDNode *N, SDValue &Lo, SDValue &Hi);
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp b/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp
index 7deb85ce9c44e..220757dfa5258 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp
@@ -1403,9 +1403,9 @@ void DAGTypeLegalizer::SplitVectorResult(SDNode *N, unsigned ResNo) {
case ISD::VSELECT:
case ISD::SELECT:
case ISD::CT_SELECT:
- SplitRes_CT_SELECT(N, Lo, Hi);
- break;
case ISD::VP_MERGE:
+ SplitRes_Select(N, Lo, Hi);
+ break;
case ISD::SELECT_CC: SplitRes_SELECT_CC(N, Lo, Hi); break;
case ISD::POISON:
case ISD::UNDEF: SplitRes_UNDEF(N, Lo, Hi); break;
diff --git a/llvm/lib/CodeGen/SelectionDAG/SelectionDAGBuilder.cpp b/llvm/lib/CodeGen/SelectionDAG/SelectionDAGBuilder.cpp
index 1d42f040e2fd1..410b670ca3f18 100644
--- a/llvm/lib/CodeGen/SelectionDAG/SelectionDAGBuilder.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/SelectionDAGBuilder.cpp
@@ -6906,8 +6906,10 @@ void SelectionDAGBuilder::visitIntrinsicCall(const CallInst &I,
return;
}
case Intrinsic::ct_select: {
- // Fast-math flags on the call are intentionally dropped: CT_SELECT
- // carries no SDNodeFlags, so no FMF-driven combine can apply to it.
+ // CT_SELECT carries no SDNodeFlags, so any fast-math flags on the call are
+ // dropped here. This is conservative: the poison-generating flags were
+ // already available to the middle end, and dropping them only removes an
+ // assumption, keeping the node opaque to FMF-driven combines.
SDValue Cond = getValue(I.getArgOperand(0));
SDValue A = getValue(I.getArgOperand(1));
SDValue B = getValue(I.getArgOperand(2));
diff --git a/llvm/lib/IR/Verifier.cpp b/llvm/lib/IR/Verifier.cpp
index 37eba3c0c6791..2e74643984158 100644
--- a/llvm/lib/IR/Verifier.cpp
+++ b/llvm/lib/IR/Verifier.cpp
@@ -6273,9 +6273,9 @@ void Verifier::visitIntrinsicCall(Intrinsic::ID ID, CallBase &Call) {
// llvm_any_ty; SelectionDAG lowering only supports single-value types.
Type *ValTy = Call.getType();
Check(ValTy->isIntOrIntVectorTy() || ValTy->isFPOrFPVectorTy() ||
- ValTy->isPtrOrPtrVectorTy(),
- "llvm.ct.select only supports integer, floating-point, or pointer "
- "types, or vectors of them",
+ ValTy->isPtrOrPtrVectorTy() || ValTy->isByteOrByteVectorTy(),
+ "llvm.ct.select only supports integer, byte, floating-point, or "
+ "pointer types, or vectors of them",
Call);
break;
}
diff --git a/llvm/test/CodeGen/X86/ctselect.ll b/llvm/test/CodeGen/X86/ctselect.ll
index 2246530619150..096b5213ec69e 100644
--- a/llvm/test/CodeGen/X86/ctselect.ll
+++ b/llvm/test/CodeGen/X86/ctselect.ll
@@ -44,6 +44,45 @@ define i8 @test_ctselect_i8(i1 %cond, i8 %a, i8 %b) #0 {
ret i8 %result
}
+define b8 @test_ctselect_b8(i1 %cond, b8 %a, b8 %b) #0 {
+; X64-LABEL: test_ctselect_b8:
+; X64: # %bb.0:
+; X64-NEXT: movl %edi, %eax
+; X64-NEXT: andb $1, %al
+; X64-NEXT: xorl %edx, %esi
+; X64-NEXT: negb %al
+; X64-NEXT: andb %sil, %al
+; X64-NEXT: xorb %dl, %al
+; X64-NEXT: # kill: def $al killed $al killed $eax
+; X64-NEXT: retq
+;
+; X32-LABEL: test_ctselect_b8:
+; X32: # %bb.0:
+; X32-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: andb $1, %al
+; X32-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: movzbl {{[0-9]+}}(%esp), %edx
+; X32-NEXT: xorb %cl, %dl
+; X32-NEXT: negb %al
+; X32-NEXT: andb %dl, %al
+; X32-NEXT: xorb %cl, %al
+; X32-NEXT: retl
+;
+; X32-NOCMOV-LABEL: test_ctselect_b8:
+; X32-NOCMOV: # %bb.0:
+; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: andb $1, %al
+; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %edx
+; X32-NOCMOV-NEXT: xorb %cl, %dl
+; X32-NOCMOV-NEXT: negb %al
+; X32-NOCMOV-NEXT: andb %dl, %al
+; X32-NOCMOV-NEXT: xorb %cl, %al
+; X32-NOCMOV-NEXT: retl
+ %result = call b8 @llvm.ct.select.b8(i1 %cond, b8 %a, b8 %b)
+ ret b8 %result
+}
+
define i32 @test_ctselect_i32(i1 %cond, i32 %a, i32 %b) #0 {
; X64-LABEL: test_ctselect_i32:
; X64: # %bb.0:
@@ -1118,9 +1157,322 @@ define i32 @test_ctselect_double_nested_and_i1(i1 %c0, i1 %c1, i1 %c2, i32 %x, i
ret i32 %result
}
+; ct_select (not C), A, B -> ct_select C, B, A. The xor is folded into swapped
+; arms, so the condition is consumed directly and no not/xor materializes the
+; inverted predicate.
+define i32 @test_ctselect_negated_cond(i1 %c, i32 %a, i32 %b) #0 {
+; X64-LABEL: test_ctselect_negated_cond:
+; X64: # %bb.0:
+; X64-NEXT: movl %edi, %eax
+; X64-NEXT: xorl %esi, %edx
+; X64-NEXT: andl $1, %eax
+; X64-NEXT: negl %eax
+; X64-NEXT: andl %edx, %eax
+; X64-NEXT: xorl %esi, %eax
+; X64-NEXT: retq
+;
+; X32-LABEL: test_ctselect_negated_cond:
+; X32: # %bb.0:
+; X32-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: andb $1, %al
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NEXT: xorl %ecx, %edx
+; X32-NEXT: movzbl %al, %eax
+; X32-NEXT: negl %eax
+; X32-NEXT: andl %edx, %eax
+; X32-NEXT: xorl %ecx, %eax
+; X32-NEXT: retl
+;
+; X32-NOCMOV-LABEL: test_ctselect_negated_cond:
+; X32-NOCMOV: # %bb.0:
+; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: andb $1, %al
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NOCMOV-NEXT: xorl %ecx, %edx
+; X32-NOCMOV-NEXT: movzbl %al, %eax
+; X32-NOCMOV-NEXT: negl %eax
+; X32-NOCMOV-NEXT: andl %edx, %eax
+; X32-NOCMOV-NEXT: xorl %ecx, %eax
+; X32-NOCMOV-NEXT: retl
+ %not = xor i1 %c, true
+ %result = call i32 @llvm.ct.select.i32(i1 %not, i32 %a, i32 %b)
+ ret i32 %result
+}
+
; Vector CT_SELECT Tests
; ============================================================================
+; Byte-typed vector: bytes lower like the same-width integer vector.
+define <16 x b8> @test_ctselect_v16b8(i1 %cond, <16 x b8> %a, <16 x b8> %b) #0 {
+; X64-LABEL: test_ctselect_v16b8:
+; X64: # %bb.0:
+; X64-NEXT: andb $1, %dil
+; X64-NEXT: pxor %xmm1, %xmm0
+; X64-NEXT: negb %dil
+; X64-NEXT: movzbl %dil, %eax
+; X64-NEXT: movd %eax, %xmm2
+; X64-NEXT: punpcklbw {{.*#+}} xmm2 = xmm2[0,0,1,1,2,2,3,3,4,4,5,5,6,6,7,7]
+; X64-NEXT: pshuflw {{.*#+}} xmm2 = xmm2[0,0,0,0,4,5,6,7]
+; X64-NEXT: pshufd {{.*#+}} xmm2 = xmm2[0,1,0,1]
+; X64-NEXT: pand %xmm2, %xmm0
+; X64-NEXT: pxor %xmm1, %xmm0
+; X64-NEXT: retq
+;
+; X32-LABEL: test_ctselect_v16b8:
+; X32: # %bb.0:
+; X32-NEXT: pushl %ebx
+; X32-NEXT: subl $12, %esp
+; X32-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: andb $1, %al
+; X32-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: xorb {{[0-9]+}}(%esp), %cl
+; X32-NEXT: negb %al
+; X32-NEXT: andb %al, %cl
+; X32-NEXT: movb %cl, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
+; X32-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: xorb {{[0-9]+}}(%esp), %cl
+; X32-NEXT: andb %al, %cl
+; X32-NEXT: movb %cl, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
+; X32-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: xorb {{[0-9]+}}(%esp), %cl
+; X32-NEXT: andb %al, %cl
+; X32-NEXT: movb %cl, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
+; X32-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: xorb {{[0-9]+}}(%esp), %cl
+; X32-NEXT: andb %al, %cl
+; X32-NEXT: movb %cl, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
+; X32-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: xorb {{[0-9]+}}(%esp), %cl
+; X32-NEXT: andb %al, %cl
+; X32-NEXT: movb %cl, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
+; X32-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: xorb {{[0-9]+}}(%esp), %cl
+; X32-NEXT: andb %al, %cl
+; X32-NEXT: movb %cl, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
+; X32-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: xorb {{[0-9]+}}(%esp), %cl
+; X32-NEXT: andb %al, %cl
+; X32-NEXT: movb %cl, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
+; X32-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: xorb {{[0-9]+}}(%esp), %cl
+; X32-NEXT: andb %al, %cl
+; X32-NEXT: movb %cl, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
+; X32-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: xorb {{[0-9]+}}(%esp), %cl
+; X32-NEXT: andb %al, %cl
+; X32-NEXT: movb %cl, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
+; X32-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: xorb {{[0-9]+}}(%esp), %cl
+; X32-NEXT: andb %al, %cl
+; X32-NEXT: movb %cl, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
+; X32-NEXT: movb {{[0-9]+}}(%esp), %bh
+; X32-NEXT: xorb {{[0-9]+}}(%esp), %bh
+; X32-NEXT: andb %al, %bh
+; X32-NEXT: movb {{[0-9]+}}(%esp), %bl
+; X32-NEXT: xorb {{[0-9]+}}(%esp), %bl
+; X32-NEXT: andb %al, %bl
+; X32-NEXT: movb {{[0-9]+}}(%esp), %dh
+; X32-NEXT: xorb {{[0-9]+}}(%esp), %dh
+; X32-NEXT: andb %al, %dh
+; X32-NEXT: movb {{[0-9]+}}(%esp), %ch
+; X32-NEXT: xorb {{[0-9]+}}(%esp), %ch
+; X32-NEXT: andb %al, %ch
+; X32-NEXT: movb {{[0-9]+}}(%esp), %dl
+; X32-NEXT: xorb {{[0-9]+}}(%esp), %dl
+; X32-NEXT: andb %al, %dl
+; X32-NEXT: movb {{[0-9]+}}(%esp), %ah
+; X32-NEXT: movb {{[0-9]+}}(%esp), %cl
+; X32-NEXT: xorb %ah, %cl
+; X32-NEXT: andb %al, %cl
+; X32-NEXT: movb {{[-0-9]+}}(%e{{[sb]}}p), %al # 1-byte Reload
+; X32-NEXT: xorb {{[0-9]+}}(%esp), %al
+; X32-NEXT: movb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
+; X32-NEXT: movb {{[0-9]+}}(%esp), %al
+; X32-NEXT: xorb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Folded Spill
+; X32-NEXT: movb {{[0-9]+}}(%esp), %al
+; X32-NEXT: xorb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Folded Spill
+; X32-NEXT: movb {{[0-9]+}}(%esp), %al
+; X32-NEXT: xorb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Folded Spill
+; X32-NEXT: movb {{[0-9]+}}(%esp), %al
+; X32-NEXT: xorb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Folded Spill
+; X32-NEXT: movb {{[0-9]+}}(%esp), %al
+; X32-NEXT: xorb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Folded Spill
+; X32-NEXT: movb {{[0-9]+}}(%esp), %al
+; X32-NEXT: xorb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Folded Spill
+; X32-NEXT: movb {{[0-9]+}}(%esp), %al
+; X32-NEXT: xorb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Folded Spill
+; X32-NEXT: movb {{[0-9]+}}(%esp), %al
+; X32-NEXT: xorb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Folded Spill
+; X32-NEXT: movb {{[-0-9]+}}(%e{{[sb]}}p), %al # 1-byte Reload
+; X32-NEXT: xorb {{[0-9]+}}(%esp), %al
+; X32-NEXT: movb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
+; X32-NEXT: xorb {{[0-9]+}}(%esp), %bh
+; X32-NEXT: xorb {{[0-9]+}}(%esp), %bl
+; X32-NEXT: xorb {{[0-9]+}}(%esp), %dh
+; X32-NEXT: xorb {{[0-9]+}}(%esp), %ch
+; X32-NEXT: xorb {{[0-9]+}}(%esp), %dl
+; X32-NEXT: xorb %ah, %cl
+; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: movb %cl, 15(%eax)
+; X32-NEXT: movb %dl, 14(%eax)
+; X32-NEXT: movb %ch, 13(%eax)
+; X32-NEXT: movb %dh, 12(%eax)
+; X32-NEXT: movb %bl, 11(%eax)
+; X32-NEXT: movb %bh, 10(%eax)
+; X32-NEXT: movzbl {{[-0-9]+}}(%e{{[sb]}}p), %ecx # 1-byte Folded Reload
+; X32-NEXT: movb %cl, 9(%eax)
+; X32-NEXT: movzbl {{[-0-9]+}}(%e{{[sb]}}p), %ecx # 1-byte Folded Reload
+; X32-NEXT: movb %cl, 8(%eax)
+; X32-NEXT: movzbl {{[-0-9]+}}(%e{{[sb]}}p), %ecx # 1-byte Folded Reload
+; X32-NEXT: movb %cl, 7(%eax)
+; X32-NEXT: movzbl {{[-0-9]+}}(%e{{[sb]}}p), %ecx # 1-byte Folded Reload
+; X32-NEXT: movb %cl, 6(%eax)
+; X32-NEXT: movzbl {{[-0-9]+}}(%e{{[sb]}}p), %ecx # 1-byte Folded Reload
+; X32-NEXT: movb %cl, 5(%eax)
+; X32-NEXT: movzbl {{[-0-9]+}}(%e{{[sb]}}p), %ecx # 1-byte Folded Reload
+; X32-NEXT: movb %cl, 4(%eax)
+; X32-NEXT: movzbl {{[-0-9]+}}(%e{{[sb]}}p), %ecx # 1-byte Folded Reload
+; X32-NEXT: movb %cl, 3(%eax)
+; X32-NEXT: movzbl {{[-0-9]+}}(%e{{[sb]}}p), %ecx # 1-byte Folded Reload
+; X32-NEXT: movb %cl, 2(%eax)
+; X32-NEXT: movzbl {{[-0-9]+}}(%e{{[sb]}}p), %ecx # 1-byte Folded Reload
+; X32-NEXT: movb %cl, 1(%eax)
+; X32-NEXT: movzbl {{[-0-9]+}}(%e{{[sb]}}p), %ecx # 1-byte Folded Reload
+; X32-NEXT: movb %cl, (%eax)
+; X32-NEXT: addl $12, %esp
+; X32-NEXT: popl %ebx
+; X32-NEXT: retl $4
+;
+; X32-NOCMOV-LABEL: test_ctselect_v16b8:
+; X32-NOCMOV: # %bb.0:
+; X32-NOCMOV-NEXT: pushl %ebx
+; X32-NOCMOV-NEXT: subl $12, %esp
+; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: andb $1, %al
+; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: xorb {{[0-9]+}}(%esp), %cl
+; X32-NOCMOV-NEXT: negb %al
+; X32-NOCMOV-NEXT: andb %al, %cl
+; X32-NOCMOV-NEXT: movb %cl, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
+; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: xorb {{[0-9]+}}(%esp), %cl
+; X32-NOCMOV-NEXT: andb %al, %cl
+; X32-NOCMOV-NEXT: movb %cl, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
+; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: xorb {{[0-9]+}}(%esp), %cl
+; X32-NOCMOV-NEXT: andb %al, %cl
+; X32-NOCMOV-NEXT: movb %cl, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
+; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: xorb {{[0-9]+}}(%esp), %cl
+; X32-NOCMOV-NEXT: andb %al, %cl
+; X32-NOCMOV-NEXT: movb %cl, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
+; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: xorb {{[0-9]+}}(%esp), %cl
+; X32-NOCMOV-NEXT: andb %al, %cl
+; X32-NOCMOV-NEXT: movb %cl, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
+; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: xorb {{[0-9]+}}(%esp), %cl
+; X32-NOCMOV-NEXT: andb %al, %cl
+; X32-NOCMOV-NEXT: movb %cl, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
+; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: xorb {{[0-9]+}}(%esp), %cl
+; X32-NOCMOV-NEXT: andb %al, %cl
+; X32-NOCMOV-NEXT: movb %cl, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
+; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: xorb {{[0-9]+}}(%esp), %cl
+; X32-NOCMOV-NEXT: andb %al, %cl
+; X32-NOCMOV-NEXT: movb %cl, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
+; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: xorb {{[0-9]+}}(%esp), %cl
+; X32-NOCMOV-NEXT: andb %al, %cl
+; X32-NOCMOV-NEXT: movb %cl, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
+; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: xorb {{[0-9]+}}(%esp), %cl
+; X32-NOCMOV-NEXT: andb %al, %cl
+; X32-NOCMOV-NEXT: movb %cl, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
+; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %bh
+; X32-NOCMOV-NEXT: xorb {{[0-9]+}}(%esp), %bh
+; X32-NOCMOV-NEXT: andb %al, %bh
+; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %bl
+; X32-NOCMOV-NEXT: xorb {{[0-9]+}}(%esp), %bl
+; X32-NOCMOV-NEXT: andb %al, %bl
+; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %dh
+; X32-NOCMOV-NEXT: xorb {{[0-9]+}}(%esp), %dh
+; X32-NOCMOV-NEXT: andb %al, %dh
+; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %ch
+; X32-NOCMOV-NEXT: xorb {{[0-9]+}}(%esp), %ch
+; X32-NOCMOV-NEXT: andb %al, %ch
+; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %dl
+; X32-NOCMOV-NEXT: xorb {{[0-9]+}}(%esp), %dl
+; X32-NOCMOV-NEXT: andb %al, %dl
+; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %ah
+; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %cl
+; X32-NOCMOV-NEXT: xorb %ah, %cl
+; X32-NOCMOV-NEXT: andb %al, %cl
+; X32-NOCMOV-NEXT: movb {{[-0-9]+}}(%e{{[sb]}}p), %al # 1-byte Reload
+; X32-NOCMOV-NEXT: xorb {{[0-9]+}}(%esp), %al
+; X32-NOCMOV-NEXT: movb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
+; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %al
+; X32-NOCMOV-NEXT: xorb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Folded Spill
+; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %al
+; X32-NOCMOV-NEXT: xorb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Folded Spill
+; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %al
+; X32-NOCMOV-NEXT: xorb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Folded Spill
+; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %al
+; X32-NOCMOV-NEXT: xorb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Folded Spill
+; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %al
+; X32-NOCMOV-NEXT: xorb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Folded Spill
+; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %al
+; X32-NOCMOV-NEXT: xorb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Folded Spill
+; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %al
+; X32-NOCMOV-NEXT: xorb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Folded Spill
+; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %al
+; X32-NOCMOV-NEXT: xorb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Folded Spill
+; X32-NOCMOV-NEXT: movb {{[-0-9]+}}(%e{{[sb]}}p), %al # 1-byte Reload
+; X32-NOCMOV-NEXT: xorb {{[0-9]+}}(%esp), %al
+; X32-NOCMOV-NEXT: movb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
+; X32-NOCMOV-NEXT: xorb {{[0-9]+}}(%esp), %bh
+; X32-NOCMOV-NEXT: xorb {{[0-9]+}}(%esp), %bl
+; X32-NOCMOV-NEXT: xorb {{[0-9]+}}(%esp), %dh
+; X32-NOCMOV-NEXT: xorb {{[0-9]+}}(%esp), %ch
+; X32-NOCMOV-NEXT: xorb {{[0-9]+}}(%esp), %dl
+; X32-NOCMOV-NEXT: xorb %ah, %cl
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: movb %cl, 15(%eax)
+; X32-NOCMOV-NEXT: movb %dl, 14(%eax)
+; X32-NOCMOV-NEXT: movb %ch, 13(%eax)
+; X32-NOCMOV-NEXT: movb %dh, 12(%eax)
+; X32-NOCMOV-NEXT: movb %bl, 11(%eax)
+; X32-NOCMOV-NEXT: movb %bh, 10(%eax)
+; X32-NOCMOV-NEXT: movzbl {{[-0-9]+}}(%e{{[sb]}}p), %ecx # 1-byte Folded Reload
+; X32-NOCMOV-NEXT: movb %cl, 9(%eax)
+; X32-NOCMOV-NEXT: movzbl {{[-0-9]+}}(%e{{[sb]}}p), %ecx # 1-byte Folded Reload
+; X32-NOCMOV-NEXT: movb %cl, 8(%eax)
+; X32-NOCMOV-NEXT: movzbl {{[-0-9]+}}(%e{{[sb]}}p), %ecx # 1-byte Folded Reload
+; X32-NOCMOV-NEXT: movb %cl, 7(%eax)
+; X32-NOCMOV-NEXT: movzbl {{[-0-9]+}}(%e{{[sb]}}p), %ecx # 1-byte Folded Reload
+; X32-NOCMOV-NEXT: movb %cl, 6(%eax)
+; X32-NOCMOV-NEXT: movzbl {{[-0-9]+}}(%e{{[sb]}}p), %ecx # 1-byte Folded Reload
+; X32-NOCMOV-NEXT: movb %cl, 5(%eax)
+; X32-NOCMOV-NEXT: movzbl {{[-0-9]+}}(%e{{[sb]}}p), %ecx # 1-byte Folded Reload
+; X32-NOCMOV-NEXT: movb %cl, 4(%eax)
+; X32-NOCMOV-NEXT: movzbl {{[-0-9]+}}(%e{{[sb]}}p), %ecx # 1-byte Folded Reload
+; X32-NOCMOV-NEXT: movb %cl, 3(%eax)
+; X32-NOCMOV-NEXT: movzbl {{[-0-9]+}}(%e{{[sb]}}p), %ecx # 1-byte Folded Reload
+; X32-NOCMOV-NEXT: movb %cl, 2(%eax)
+; X32-NOCMOV-NEXT: movzbl {{[-0-9]+}}(%e{{[sb]}}p), %ecx # 1-byte Folded Reload
+; X32-NOCMOV-NEXT: movb %cl, 1(%eax)
+; X32-NOCMOV-NEXT: movzbl {{[-0-9]+}}(%e{{[sb]}}p), %ecx # 1-byte Folded Reload
+; X32-NOCMOV-NEXT: movb %cl, (%eax)
+; X32-NOCMOV-NEXT: addl $12, %esp
+; X32-NOCMOV-NEXT: popl %ebx
+; X32-NOCMOV-NEXT: retl $4
+ %result = call <16 x b8> @llvm.ct.select.v16b8(i1 %cond, <16 x b8> %a, <16 x b8> %b)
+ ret <16 x b8> %result
+}
+
; Test vector CT_SELECT with v4i32 (128-bit vector with single i1 mask)
; NOW CONSTANT-TIME: Uses bitwise XOR/AND operations instead of branches!
define <4 x i32> @test_ctselect_v4i32(i1 %cond, <4 x i32> %a, <4 x i32> %b) #0 {
@@ -2435,6 +2787,7 @@ define double @test_ctselect_f64_nan_inf(i1 %cond) #0 {
; Declare the intrinsics
declare i1 @llvm.ct.select.i1(i1, i1, i1)
declare i8 @llvm.ct.select.i8(i1, i8, i8)
+declare b8 @llvm.ct.select.b8(i1, b8, b8)
declare i16 @llvm.ct.select.i16(i1, i16, i16)
declare i32 @llvm.ct.select.i32(i1, i32, i32)
declare i64 @llvm.ct.select.i64(i1, i64, i64)
@@ -2451,6 +2804,7 @@ declare <4 x i32> @llvm.ct.select.v4i32(i1, <4 x i32>, <4 x i32>)
declare <2 x i64> @llvm.ct.select.v2i64(i1, <2 x i64>, <2 x i64>)
declare <8 x i16> @llvm.ct.select.v8i16(i1, <8 x i16>, <8 x i16>)
declare <16 x i8> @llvm.ct.select.v16i8(i1, <16 x i8>, <16 x i8>)
+declare <16 x b8> @llvm.ct.select.v16b8(i1, <16 x b8>, <16 x b8>)
declare <4 x float> @llvm.ct.select.v4f32(i1, <4 x float>, <4 x float>)
declare <2 x double> @llvm.ct.select.v2f64(i1, <2 x double>, <2 x double>)
declare <8 x i32> @llvm.ct.select.v8i32(i1, <8 x i32>, <8 x i32>)
diff --git a/llvm/test/Transforms/InstSimplify/ct-select.ll b/llvm/test/Transforms/InstSimplify/ct-select.ll
index 3a89009be8480..b5802497cbed0 100644
--- a/llvm/test/Transforms/InstSimplify/ct-select.ll
+++ b/llvm/test/Transforms/InstSimplify/ct-select.ll
@@ -1,12 +1,6 @@
; NOTE: Assertions have been autogenerated by utils/update_test_checks.py
; RUN: opt < %s -passes=instsimplify -S | FileCheck %s
-declare i32 @llvm.ct.select.i32(i1, i32, i32)
-declare i64 @llvm.ct.select.i64(i1, i64, i64)
-declare float @llvm.ct.select.f32(i1, float, float)
-declare ptr @llvm.ct.select.p0(i1, ptr, ptr)
-declare <4 x i32> @llvm.ct.select.v4i32(i1, <4 x i32>, <4 x i32>)
-
define i32 @ct_select_true(i32 %x, i32 %y) {
; CHECK-LABEL: @ct_select_true(
; CHECK-NEXT: [[R:%.*]] = call i32 @llvm.ct.select.i32(i1 true, i32 [[X:%.*]], i32 [[Y:%.*]])
@@ -79,9 +73,9 @@ define <4 x i32> @ct_select_same_arms_vec(i1 %c, <4 x i32> %x) {
ret <4 x i32> %r
}
-; Negative test: must NOT fold when condition is a non-literal value, even if
-; analyses could prove it constant. The whole point of ct.select is to keep
-; the lowering even when the cond looks statically derivable.
+; Negative test: must NOT fold when the condition operand is not a constant.
+; InstSimplify deliberately does not run value analyses (known-bits, range) on
+; the condition to derive a fold, so a plain runtime %c keeps the ct.select.
define i32 @ct_select_runtime_cond_no_fold(i1 %c, i32 %x, i32 %y) {
; CHECK-LABEL: @ct_select_runtime_cond_no_fold(
; CHECK-NEXT: [[R:%.*]] = call i32 @llvm.ct.select.i32(i1 [[C:%.*]], i32 [[X:%.*]], i32 [[Y:%.*]])
diff --git a/llvm/test/Verifier/ct-select.ll b/llvm/test/Verifier/ct-select.ll
index 56f897a8a53c3..5c2f441717f5f 100644
--- a/llvm/test/Verifier/ct-select.ll
+++ b/llvm/test/Verifier/ct-select.ll
@@ -7,13 +7,13 @@
declare {i32, i32} @llvm.ct.select.sl_i32i32s(i1, {i32, i32}, {i32, i32})
declare [2 x i64] @llvm.ct.select.a2i64(i1, [2 x i64], [2 x i64])
-; CHECK: llvm.ct.select only supports integer, floating-point, or pointer types, or vectors of them
+; CHECK: llvm.ct.select only supports integer, byte, floating-point, or pointer types, or vectors of them
define {i32, i32} @ct_select_struct(i1 %c, {i32, i32} %x, {i32, i32} %y) {
%r = call {i32, i32} @llvm.ct.select.sl_i32i32s(i1 %c, {i32, i32} %x, {i32, i32} %y)
ret {i32, i32} %r
}
-; CHECK: llvm.ct.select only supports integer, floating-point, or pointer types, or vectors of them
+; CHECK: llvm.ct.select only supports integer, byte, floating-point, or pointer types, or vectors of them
define [2 x i64] @ct_select_array(i1 %c, [2 x i64] %x, [2 x i64] %y) {
%r = call [2 x i64] @llvm.ct.select.a2i64(i1 %c, [2 x i64] %x, [2 x i64] %y)
ret [2 x i64] %r
>From 5e39c74868829bb02dde22e31340580d16358307 Mon Sep 17 00:00:00 2001
From: wizardengineer <juliuswoosebert at gmail.com>
Date: Tue, 8 Sep 2026 08:10:17 -0400
Subject: [PATCH 09/12] [ConstantTime] Allow dead llvm.ct.select to be removed
An unused llvm.ct.select outlived the InstructionSimplify fold, which kept
both value arguments live and contradicted the LangRef text saying the unused
argument becomes dead code. Add the intrinsic to the wouldInstructionBeTriviallyDead
whitelist next to llvm.allow.runtime.check and llvm.allow.ubsan.check, which
carry IntrInaccessibleMemOnly for the same reason: to pin the call in place,
not to model a real memory access.
Keep IntrInaccessibleMemOnly rather than switching to IntrNoMem. As a pure
value the call is sunk into a conditionally-executed block by InstCombine
(gated on mayWriteToMemory) and split into one copy per branch arm by GVN PRE.
Also correct the undef condition semantics, trim the declaration list, and
drop claims from the definition comment that the memory effect does not
actually provide.
Co-Authored-By: Claude Opus 5 <noreply at anthropic.com>
---
llvm/docs/LangRef.md | 48 +++++------
llvm/include/llvm/IR/Intrinsics.td | 21 +++--
llvm/lib/Transforms/Utils/Local.cpp | 3 +-
.../test/Transforms/InstSimplify/ct-select.ll | 85 +++++++++++++++----
4 files changed, 107 insertions(+), 50 deletions(-)
diff --git a/llvm/docs/LangRef.md b/llvm/docs/LangRef.md
index 0457672ef496c..002a044ab4415 100644
--- a/llvm/docs/LangRef.md
+++ b/llvm/docs/LangRef.md
@@ -16055,29 +16055,20 @@ byte, floating-point, or pointer type, or on any vector of those types,
including scalable vectors. The declarations below are a representative sample:
```
-declare i8 @llvm.ct.select.i8(i1 <cond>, i8 <val1>, i8 <val2>)
-declare b8 @llvm.ct.select.b8(i1 <cond>, b8 <val1>, b8 <val2>)
declare i32 @llvm.ct.select.i32(i1 <cond>, i32 <val1>, i32 <val2>)
-declare i64 @llvm.ct.select.i64(i1 <cond>, i64 <val1>, i64 <val2>)
-declare half @llvm.ct.select.f16(i1 <cond>, half <val1>, half <val2>)
declare float @llvm.ct.select.f32(i1 <cond>, float <val1>, float <val2>)
-declare double @llvm.ct.select.f64(i1 <cond>, double <val1>, double <val2>)
-declare fp128 @llvm.ct.select.f128(i1 <cond>, fp128 <val1>, fp128 <val2>)
declare ptr @llvm.ct.select.p0(i1 <cond>, ptr <val1>, ptr <val2>)
declare <4 x i32> @llvm.ct.select.v4i32(i1 <cond>, <4 x i32> <val1>, <4 x i32> <val2>)
-declare <2 x double> @llvm.ct.select.v2f64(i1 <cond>, <2 x double> <val1>, <2 x double> <val2>)
-declare <2 x ptr> @llvm.ct.select.v2p0(i1 <cond>, <2 x ptr> <val1>, <2 x ptr> <val2>)
-declare <vscale x 4 x i32> @llvm.ct.select.nxv4i32(i1 <cond>, <vscale x 4 x i32> <val1>, <vscale x 4 x i32> <val2>)
```
##### Overview:
-The '`llvm.ct.select`' family of intrinsic functions selects one of two
-values based on a condition, like the standard {ref}`select <i_select>`
-instruction, but is lowered to branchless code whose execution time does not
-depend on the condition value. This keeps the condition from leaking through
-timing side channels; see the Semantics section for the exact guarantee and
-its platform requirements.
+The '`llvm.ct.select`' intrinsic selects one of two values based on a
+condition, like the standard {ref}`select <i_select>` instruction, but is
+lowered to branchless code whose execution time does not depend on the
+condition value. This keeps the condition from leaking through timing side
+channels; see the Semantics section for the exact guarantee and its platform
+requirements.
##### Arguments:
@@ -16100,9 +16091,9 @@ If the condition evaluates to true, the intrinsic returns the first value
argument; otherwise, it returns the second value argument.
Poison and undef propagate as for {ref}`select <i_select>`: a `poison`
-condition yields `poison`, an `undef` condition yields `undef`. Unlike
-`select`, both value arguments are always evaluated, so a `poison` value in
-either one yields `poison`.
+condition yields `poison`, and an `undef` condition returns either value
+argument. Unlike `select`, both value arguments are always evaluated, so a
+`poison` value in either one yields `poison`.
The key semantic difference from {ref}`select <i_select>` is the constant-time
code generation guarantee: the intrinsic must be lowered to machine code that:
@@ -16144,13 +16135,20 @@ with data-dependent timing, or, except as described below, optimizing away
either value argument before the selection completes.
The call folds to one of its value arguments when the condition operand is a
-constant `i1`, or when both value arguments are the same value; the unused
-argument then becomes ordinary dead code. Optimizers must not derive the fold
-by running value analyses (known-bits, range, or dominating conditions) on a
-non-constant condition. A condition that another pass has already proved to be
-a compile-time constant is no longer secret-dependent, so folding it is
-permitted. Rewrites that keep the `llvm.ct.select`, such as swapping the
-arguments to remove a negated condition, are allowed.
+constant `i1`, or when both value arguments are the same value. Optimizers must
+not derive the fold by running value analyses (known-bits, range, or dominating
+conditions) on a non-constant condition. A condition that another pass has
+already proved to be a compile-time constant is no longer secret-dependent, so
+folding it is permitted. Rewrites that keep the `llvm.ct.select`, such as
+swapping the arguments to remove a negated condition, are allowed.
+
+A call whose result has no users may be removed. Nothing observes the
+selection in that case, so deleting it cannot expose the condition. This is the
+only way an `llvm.ct.select` disappears: as long as the result is used, the
+call is neither deleted, duplicated, merged with another `llvm.ct.select`, nor
+moved to a different point in the control-flow graph. Keeping it in place is
+what the intrinsic's `inaccessiblemem` memory effect is for; the effect does
+not model any real access to memory, and passes must not read it as one.
Like other floating-point calls, `llvm.ct.select` may carry fast-math flags
when it returns a supported floating-point type; the poison-generating flags
diff --git a/llvm/include/llvm/IR/Intrinsics.td b/llvm/include/llvm/IR/Intrinsics.td
index 0533074a3f863..401e2a5a0a4a0 100644
--- a/llvm/include/llvm/IR/Intrinsics.td
+++ b/llvm/include/llvm/IR/Intrinsics.td
@@ -2068,14 +2068,19 @@ def int_coro_subfn_addr : DefaultAttrsIntrinsic<
///===---------------------- Constant Time Intrinsics ----------------------===//
//
-// Intrinsic to support constant-time select. The selected value is pure, but
-// the intrinsic is intentionally modeled like llvm.sideeffect: it only touches
-// inaccessible memory so it remains an opaque barrier to generic transforms.
-// This prevents passes from deleting it when unused, speculating it, or
-// rewriting it into an ordinary select, while avoiding claims about writes to
-// accessible program memory. The constant-time lowering contract is described
-// in LangRef. llvm_any_ty alone would also admit aggregates, which codegen
-// does not support; the Verifier restricts the value type to integer, byte,
+// Intrinsic to support constant-time select. The selected value is pure; the
+// inaccessible-memory effect claims no real access and exists only to pin the
+// call in place, the way llvm.allow.runtime.check and llvm.allow.ubsan.check
+// use it. Without it the call is a pure value that InstCombine will sink into
+// a conditionally-executed block and that GVN will partially redundify into
+// one copy per branch arm, neither of which a constant-time primitive should
+// be subject to. Being unmovable is all the effect buys: an unused call is
+// still trivially dead (see wouldInstructionBeTriviallyDead), and nothing here
+// stops a pass that matches on the intrinsic ID from rewriting the call into a
+// select. That contract is documented in LangRef and enforced in codegen by
+// ISD::CT_SELECT, which is opaque to the combiner and expands to bitwise ops.
+// llvm_any_ty alone would also admit aggregates, which codegen does not
+// support; the Verifier restricts the value type to integer, byte,
// floating-point, or pointer types, or vectors of them.
def int_ct_select
: DefaultAttrsIntrinsic<[llvm_any_ty],
diff --git a/llvm/lib/Transforms/Utils/Local.cpp b/llvm/lib/Transforms/Utils/Local.cpp
index 46c010e309449..1dfb085e5b560 100644
--- a/llvm/lib/Transforms/Utils/Local.cpp
+++ b/llvm/lib/Transforms/Utils/Local.cpp
@@ -466,7 +466,8 @@ bool llvm::wouldInstructionBeTriviallyDead(const Instruction *I,
// Intrinsics declare sideeffects to prevent them from moving, but they are
// nops without users.
if (II->getIntrinsicID() == Intrinsic::allow_runtime_check ||
- II->getIntrinsicID() == Intrinsic::allow_ubsan_check)
+ II->getIntrinsicID() == Intrinsic::allow_ubsan_check ||
+ II->getIntrinsicID() == Intrinsic::ct_select)
return true;
if (II->isLifetimeStartOrEnd()) {
diff --git a/llvm/test/Transforms/InstSimplify/ct-select.ll b/llvm/test/Transforms/InstSimplify/ct-select.ll
index b5802497cbed0..8286be53ca2fc 100644
--- a/llvm/test/Transforms/InstSimplify/ct-select.ll
+++ b/llvm/test/Transforms/InstSimplify/ct-select.ll
@@ -3,8 +3,7 @@
define i32 @ct_select_true(i32 %x, i32 %y) {
; CHECK-LABEL: @ct_select_true(
-; CHECK-NEXT: [[R:%.*]] = call i32 @llvm.ct.select.i32(i1 true, i32 [[X:%.*]], i32 [[Y:%.*]])
-; CHECK-NEXT: ret i32 [[X]]
+; CHECK-NEXT: ret i32 [[X:%.*]]
;
%r = call i32 @llvm.ct.select.i32(i1 true, i32 %x, i32 %y)
ret i32 %r
@@ -12,8 +11,7 @@ define i32 @ct_select_true(i32 %x, i32 %y) {
define i32 @ct_select_false(i32 %x, i32 %y) {
; CHECK-LABEL: @ct_select_false(
-; CHECK-NEXT: [[R:%.*]] = call i32 @llvm.ct.select.i32(i1 false, i32 [[X:%.*]], i32 [[Y:%.*]])
-; CHECK-NEXT: ret i32 [[Y]]
+; CHECK-NEXT: ret i32 [[Y:%.*]]
;
%r = call i32 @llvm.ct.select.i32(i1 false, i32 %x, i32 %y)
ret i32 %r
@@ -21,8 +19,7 @@ define i32 @ct_select_false(i32 %x, i32 %y) {
define i64 @ct_select_true_i64(i64 %x, i64 %y) {
; CHECK-LABEL: @ct_select_true_i64(
-; CHECK-NEXT: [[R:%.*]] = call i64 @llvm.ct.select.i64(i1 true, i64 [[X:%.*]], i64 [[Y:%.*]])
-; CHECK-NEXT: ret i64 [[X]]
+; CHECK-NEXT: ret i64 [[X:%.*]]
;
%r = call i64 @llvm.ct.select.i64(i1 true, i64 %x, i64 %y)
ret i64 %r
@@ -30,8 +27,7 @@ define i64 @ct_select_true_i64(i64 %x, i64 %y) {
define float @ct_select_false_f32(float %x, float %y) {
; CHECK-LABEL: @ct_select_false_f32(
-; CHECK-NEXT: [[R:%.*]] = call float @llvm.ct.select.f32(i1 false, float [[X:%.*]], float [[Y:%.*]])
-; CHECK-NEXT: ret float [[Y]]
+; CHECK-NEXT: ret float [[Y:%.*]]
;
%r = call float @llvm.ct.select.f32(i1 false, float %x, float %y)
ret float %r
@@ -39,8 +35,7 @@ define float @ct_select_false_f32(float %x, float %y) {
define ptr @ct_select_true_ptr(ptr %x, ptr %y) {
; CHECK-LABEL: @ct_select_true_ptr(
-; CHECK-NEXT: [[R:%.*]] = call ptr @llvm.ct.select.p0(i1 true, ptr [[X:%.*]], ptr [[Y:%.*]])
-; CHECK-NEXT: ret ptr [[X]]
+; CHECK-NEXT: ret ptr [[X:%.*]]
;
%r = call ptr @llvm.ct.select.p0(i1 true, ptr %x, ptr %y)
ret ptr %r
@@ -48,8 +43,7 @@ define ptr @ct_select_true_ptr(ptr %x, ptr %y) {
define <4 x i32> @ct_select_true_v4i32(<4 x i32> %x, <4 x i32> %y) {
; CHECK-LABEL: @ct_select_true_v4i32(
-; CHECK-NEXT: [[R:%.*]] = call <4 x i32> @llvm.ct.select.v4i32(i1 true, <4 x i32> [[X:%.*]], <4 x i32> [[Y:%.*]])
-; CHECK-NEXT: ret <4 x i32> [[X]]
+; CHECK-NEXT: ret <4 x i32> [[X:%.*]]
;
%r = call <4 x i32> @llvm.ct.select.v4i32(i1 true, <4 x i32> %x, <4 x i32> %y)
ret <4 x i32> %r
@@ -57,8 +51,7 @@ define <4 x i32> @ct_select_true_v4i32(<4 x i32> %x, <4 x i32> %y) {
define i32 @ct_select_same_arms(i1 %c, i32 %x) {
; CHECK-LABEL: @ct_select_same_arms(
-; CHECK-NEXT: [[R:%.*]] = call i32 @llvm.ct.select.i32(i1 [[C:%.*]], i32 [[X:%.*]], i32 [[X]])
-; CHECK-NEXT: ret i32 [[X]]
+; CHECK-NEXT: ret i32 [[X:%.*]]
;
%r = call i32 @llvm.ct.select.i32(i1 %c, i32 %x, i32 %x)
ret i32 %r
@@ -66,8 +59,7 @@ define i32 @ct_select_same_arms(i1 %c, i32 %x) {
define <4 x i32> @ct_select_same_arms_vec(i1 %c, <4 x i32> %x) {
; CHECK-LABEL: @ct_select_same_arms_vec(
-; CHECK-NEXT: [[R:%.*]] = call <4 x i32> @llvm.ct.select.v4i32(i1 [[C:%.*]], <4 x i32> [[X:%.*]], <4 x i32> [[X]])
-; CHECK-NEXT: ret <4 x i32> [[X]]
+; CHECK-NEXT: ret <4 x i32> [[X:%.*]]
;
%r = call <4 x i32> @llvm.ct.select.v4i32(i1 %c, <4 x i32> %x, <4 x i32> %x)
ret <4 x i32> %r
@@ -94,3 +86,64 @@ define i32 @ct_select_distinct_arms_no_fold(i1 %c, i32 %x, i32 %y) {
%r = call i32 @llvm.ct.select.i32(i1 %c, i32 %x, i32 %y)
ret i32 %r
}
+
+define <2 x double> @ct_select_false_v2f64(<2 x double> %x, <2 x double> %y) {
+; CHECK-LABEL: @ct_select_false_v2f64(
+; CHECK-NEXT: ret <2 x double> [[Y:%.*]]
+;
+ %r = call <2 x double> @llvm.ct.select.v2f64(i1 false, <2 x double> %x, <2 x double> %y)
+ ret <2 x double> %r
+}
+
+define <2 x ptr> @ct_select_true_v2p0(<2 x ptr> %x, <2 x ptr> %y) {
+; CHECK-LABEL: @ct_select_true_v2p0(
+; CHECK-NEXT: ret <2 x ptr> [[X:%.*]]
+;
+ %r = call <2 x ptr> @llvm.ct.select.v2p0(i1 true, <2 x ptr> %x, <2 x ptr> %y)
+ ret <2 x ptr> %r
+}
+
+define <vscale x 4 x i32> @ct_select_true_nxv4i32(<vscale x 4 x i32> %x, <vscale x 4 x i32> %y) {
+; CHECK-LABEL: @ct_select_true_nxv4i32(
+; CHECK-NEXT: ret <vscale x 4 x i32> [[X:%.*]]
+;
+ %r = call <vscale x 4 x i32> @llvm.ct.select.nxv4i32(i1 true, <vscale x 4 x i32> %x, <vscale x 4 x i32> %y)
+ ret <vscale x 4 x i32> %r
+}
+
+define <vscale x 2 x i64> @ct_select_same_arms_nxv2i64(i1 %c, <vscale x 2 x i64> %x) {
+; CHECK-LABEL: @ct_select_same_arms_nxv2i64(
+; CHECK-NEXT: ret <vscale x 2 x i64> [[X:%.*]]
+;
+ %r = call <vscale x 2 x i64> @llvm.ct.select.nxv2i64(i1 %c, <vscale x 2 x i64> %x, <vscale x 2 x i64> %x)
+ ret <vscale x 2 x i64> %r
+}
+
+; Negative test: vectors with a runtime condition must not fold either.
+define <4 x i32> @ct_select_distinct_arms_vec_no_fold(i1 %c, <4 x i32> %x, <4 x i32> %y) {
+; CHECK-LABEL: @ct_select_distinct_arms_vec_no_fold(
+; CHECK-NEXT: [[R:%.*]] = call <4 x i32> @llvm.ct.select.v4i32(i1 [[C:%.*]], <4 x i32> [[X:%.*]], <4 x i32> [[Y:%.*]])
+; CHECK-NEXT: ret <4 x i32> [[R]]
+;
+ %r = call <4 x i32> @llvm.ct.select.v4i32(i1 %c, <4 x i32> %x, <4 x i32> %y)
+ ret <4 x i32> %r
+}
+
+define <vscale x 4 x i32> @ct_select_distinct_arms_scalable_no_fold(i1 %c, <vscale x 4 x i32> %x, <vscale x 4 x i32> %y) {
+; CHECK-LABEL: @ct_select_distinct_arms_scalable_no_fold(
+; CHECK-NEXT: [[R:%.*]] = call <vscale x 4 x i32> @llvm.ct.select.nxv4i32(i1 [[C:%.*]], <vscale x 4 x i32> [[X:%.*]], <vscale x 4 x i32> [[Y:%.*]])
+; CHECK-NEXT: ret <vscale x 4 x i32> [[R]]
+;
+ %r = call <vscale x 4 x i32> @llvm.ct.select.nxv4i32(i1 %c, <vscale x 4 x i32> %x, <vscale x 4 x i32> %y)
+ ret <vscale x 4 x i32> %r
+}
+
+; An unused ct.select is trivially dead: nothing observes the selection, so
+; removing it cannot expose the condition.
+define void @ct_select_unused_is_dead(i1 %c, i32 %x, i32 %y) {
+; CHECK-LABEL: @ct_select_unused_is_dead(
+; CHECK-NEXT: ret void
+;
+ %r = call i32 @llvm.ct.select.i32(i1 %c, i32 %x, i32 %y)
+ ret void
+}
>From f571e75186293c65990c1da50491638b9a94a30a Mon Sep 17 00:00:00 2001
From: wizardengineer <juliuswoosebert at gmail.com>
Date: Tue, 8 Sep 2026 09:53:03 -0400
Subject: [PATCH 10/12] [ConstantTime] Move the CT_SELECT expansion into a
helper
The CT_SELECT case in ExpandNode was around 180 lines. Move it to
SelectionDAGLegalize::ExpandCTSELECT and name the operands, so the case is
three lines. Pure code motion: emitted assembly is unchanged for every RUN
line of the existing ct.select codegen tests.
Also use a temporary for the promoted SELECT so the call fits on one line
instead of wrapping after the assignment.
Co-Authored-By: Claude Opus 5 <noreply at anthropic.com>
---
llvm/lib/CodeGen/SelectionDAG/LegalizeDAG.cpp | 378 +++++++++---------
1 file changed, 193 insertions(+), 185 deletions(-)
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeDAG.cpp b/llvm/lib/CodeGen/SelectionDAG/LegalizeDAG.cpp
index 708d7c7df12a9..263b65a7e9e8a 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeDAG.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeDAG.cpp
@@ -202,6 +202,7 @@ class SelectionDAGLegalize {
SDValue ExpandInsertToVectorThroughStack(SDValue Op);
SDValue ExpandVectorBuildThroughStack(SDNode* Node);
SDValue ExpandConcatVectors(SDNode *Node);
+ SDValue ExpandCTSELECT(SDNode *Node);
SDValue ExpandConstantFP(ConstantFPSDNode *CFP, bool UseCP);
SDValue ExpandConstant(ConstantSDNode *CP);
@@ -3208,6 +3209,191 @@ SDValue SelectionDAGLegalize::PromoteReduction(SDNode *Node) {
DAG.getIntPtrConstant(0, DL, /*isTarget=*/true));
}
+/// Expand a constant-time select to F ^ ((T ^ F) & Mask), where
+/// Mask = 0 - (cond & 1). The sequence is bitwise only: no SELECT or cmov
+/// SDNode is built, so the condition never reaches a branch. Floating-point
+/// values are blended on the same-size integer type; vectors compute the mask
+/// as a scalar and splat it, which avoids an illegal vNi1.
+SDValue SelectionDAGLegalize::ExpandCTSELECT(SDNode *Node) {
+ SDLoc dl(Node);
+ // Constant-time select: F ^ ((T ^ F) & Mask), Mask = 0 - (cond & 1).
+ // Bitwise-only — no select/cmov SDNode is constructed, so the CT property
+ // holds against any combiner that targets those opcodes. FP types operate
+ // on the same-size integer; vectors build the mask as a scalar then splat
+ // (avoids illegal vNi1).
+ //
+ // The masked-diff is routed through a virtual register (CopyToReg /
+ // CopyFromReg) below as a forward-looking DAGCombine barrier. This is
+ // *not* required for correctness against any combiner in tree today —
+ // DAGCombiner has no rewrite that recognizes XOR/AND/XOR-with-sext-mask
+ // and reconstructs a SELECT. The chain edge is defense-in-depth against
+ // a hypothetical future fold of that form: the dependency partitions the
+ // bitwise sequence into a region the combiner can't see through. Cost is
+ // at most a coalesce-able MOV per call. ISD::ARITH_FENCE would serve the
+ // same purpose (ISel already selects it for any legal type); switching to
+ // it is left to a follow-up.
+ SDValue Cond = Node->getOperand(0);
+ SDValue T = Node->getOperand(1);
+ SDValue F = Node->getOperand(2);
+ EVT VT = T.getValueType();
+
+ // Memory-blend for FP scalars whose same-size integer isn't legal (f64 on
+ // i386 no-SSE, x86_fp80, fp128). The bitcast-to-int expansion below can't
+ // run since LegalizeDAG is the last legalization stage. Spill T/F, blend
+ // chunk-by-chunk at a legal int width in place over the T slot, reload it
+ // as the FP type. Fixed and scalable FP vectors fall through to the
+ // unified path below.
+ if (VT.isFloatingPoint() && !VT.isVector() &&
+ !TLI.isTypeLegal(VT.changeTypeToInteger())) {
+ const DataLayout &DL = DAG.getDataLayout();
+ Type *VTTy = VT.getTypeForEVT(*DAG.getContext());
+ unsigned StorageBytes = DL.getTypeStoreSize(VTTy);
+ assert(StorageBytes > 0 && "FP type with zero storage size");
+
+ // Blend at the widest legal scalar integer width. Chunks narrower than
+ // that (e.g. the 2-byte tail of x86_fp80) are zero-extended on load and
+ // truncated on store, so every value in the DAG has a legal type.
+ MVT BlendVT;
+ for (MVT MV : {MVT::i64, MVT::i32, MVT::i16, MVT::i8})
+ if (TLI.isTypeLegal(MV)) {
+ BlendVT = MV;
+ break;
+ }
+ assert(BlendVT.isValid() && "no legal scalar integer type");
+ unsigned BlendBytes = BlendVT.getSizeInBits() / 8;
+
+ MachineFunction &MF = DAG.getMachineFunction();
+ SDValue StackT = DAG.CreateStackTemporary(VT);
+ SDValue StackF = DAG.CreateStackTemporary(VT);
+ int FIT = cast<FrameIndexSDNode>(StackT.getNode())->getIndex();
+ int FIF = cast<FrameIndexSDNode>(StackF.getNode())->getIndex();
+ MachinePointerInfo PIT = MachinePointerInfo::getFixedStack(MF, FIT);
+ MachinePointerInfo PIF = MachinePointerInfo::getFixedStack(MF, FIF);
+
+ SDValue Chain = DAG.getEntryNode();
+ Chain = DAG.getStore(Chain, dl, T, StackT, PIT);
+ Chain = DAG.getStore(Chain, dl, F, StackF, PIF);
+
+ // Walk the storage in power-of-2 chunks, widest first, so every access
+ // stays naturally aligned within the max-aligned stack temporaries.
+ unsigned Offset = 0;
+ while (Offset < StorageBytes) {
+ unsigned ChunkBytes =
+ std::min(BlendBytes, llvm::bit_floor(StorageBytes - Offset));
+ MVT MemVT = MVT::getIntegerVT(ChunkBytes * 8);
+ TypeSize Off = TypeSize::getFixed(Offset);
+ SDValue TPtr = DAG.getMemBasePlusOffset(StackT, Off, dl);
+ SDValue FPtr = DAG.getMemBasePlusOffset(StackF, Off, dl);
+
+ SDValue Ti, Fi;
+ if (MemVT == BlendVT) {
+ Ti = DAG.getLoad(BlendVT, dl, Chain, TPtr, PIT.getWithOffset(Offset));
+ Chain = Ti.getValue(1);
+ Fi = DAG.getLoad(BlendVT, dl, Chain, FPtr, PIF.getWithOffset(Offset));
+ } else {
+ Ti = DAG.getExtLoad(ISD::ZEXTLOAD, dl, BlendVT, Chain, TPtr,
+ PIT.getWithOffset(Offset), MemVT);
+ Chain = Ti.getValue(1);
+ Fi = DAG.getExtLoad(ISD::ZEXTLOAD, dl, BlendVT, Chain, FPtr,
+ PIF.getWithOffset(Offset), MemVT);
+ }
+ Chain = Fi.getValue(1);
+
+ // Blend this chunk via CT_SELECT on the legal integer type (the
+ // recursive node is Expand'd by the scalar-int branch below), then
+ // store it back over the now-dead chunk of the T slot. The serial
+ // chain keeps each chunk's loads ahead of the store.
+ SDValue Ri = DAG.getCTSelect(dl, BlendVT, Cond, Ti, Fi);
+
+ if (MemVT == BlendVT)
+ Chain = DAG.getStore(Chain, dl, Ri, TPtr, PIT.getWithOffset(Offset));
+ else
+ Chain = DAG.getTruncStore(Chain, dl, Ri, TPtr,
+ PIT.getWithOffset(Offset), MemVT);
+ Offset += ChunkBytes;
+ }
+
+ return DAG.getLoad(VT, dl, Chain, StackT, PIT);
+ }
+
+ SDValue WorkingT = T;
+ SDValue WorkingF = F;
+ EVT WorkingVT = VT;
+
+ bool IsFP = VT.isVector() ? VT.getVectorElementType().isFloatingPoint()
+ : VT.isFloatingPoint();
+ if (IsFP) {
+ WorkingVT = VT.changeTypeToInteger();
+ // Scalars with an illegal integer twin took the memory path above; FP
+ // vector types are expected to have a legal integer counterpart.
+ assert(TLI.isTypeLegal(WorkingVT) &&
+ "no legal same-width integer type for CT_SELECT expansion");
+ WorkingT = DAG.getBitcast(WorkingVT, T);
+ WorkingF = DAG.getBitcast(WorkingVT, F);
+ }
+
+ // Compute the all-ones/all-zeros mask as a scalar, then splat for vectors.
+ // The element type of a legal vector is not always a legal scalar type
+ // (i32 on riscv64, i64 on riscv32), so build the scalar mask at a legal
+ // width: BUILD_VECTOR and SPLAT_VECTOR implicitly truncate a wider
+ // scalar, and if every legal scalar is narrower than the element, splat
+ // at that width and sign-extend the mask vector (0 and -1 survive both).
+ EVT MaskEltVT =
+ WorkingVT.isVector() ? WorkingVT.getVectorElementType() : WorkingVT;
+ EVT MaskSclVT = MaskEltVT;
+ if (!TLI.isTypeLegal(MaskSclVT)) {
+ assert(WorkingVT.isVector() && "scalar CT_SELECT type must be legal");
+ MaskSclVT = TLI.getLegalTypeToTransformTo(*DAG.getContext(), MaskSclVT);
+ assert(TLI.isTypeLegal(MaskSclVT) && "no legal scalar integer type");
+ }
+ SDValue ScalarCond = Cond;
+ if (ScalarCond.getValueType() != MaskSclVT)
+ ScalarCond = DAG.getAnyExtOrTrunc(ScalarCond, dl, MaskSclVT);
+ SDValue ScalarMask =
+ DAG.getNode(ISD::SUB, dl, MaskSclVT, DAG.getConstant(0, dl, MaskSclVT),
+ DAG.getNode(ISD::AND, dl, MaskSclVT, ScalarCond,
+ DAG.getConstant(1, dl, MaskSclVT)));
+ SDValue Mask;
+ if (!WorkingVT.isVector()) {
+ Mask = ScalarMask;
+ } else if (MaskSclVT.bitsGE(MaskEltVT)) {
+ Mask = DAG.getSplat(WorkingVT, dl, ScalarMask);
+ } else {
+ EVT NarrowVT =
+ WorkingVT.changeVectorElementType(*DAG.getContext(), MaskSclVT);
+ assert(TLI.isTypeLegal(NarrowVT) &&
+ "no legal vector type for the CT_SELECT mask splat");
+ Mask = DAG.getNode(ISD::SIGN_EXTEND, dl, WorkingVT,
+ DAG.getSplat(NarrowVT, dl, ScalarMask));
+ }
+
+ // F ^ ((T ^ F) & Mask)
+ SDValue XorTF = DAG.getNode(ISD::XOR, dl, WorkingVT, WorkingT, WorkingF);
+ SDValue TM = DAG.getNode(ISD::AND, dl, WorkingVT, XorTF, Mask);
+
+ // Forward-looking DAGCombine barrier (see header): route the masked-diff
+ // through a vreg so the chain edge partitions it from the surrounding
+ // bitwise ops, guarding against a future combiner that might fold
+ // XOR/AND/XOR-with-sext-mask back to SELECT. WorkingVT is legal here, so
+ // it always has a register class; only scalable vectors are skipped,
+ // conservatively.
+ if (!WorkingVT.isScalableVector()) {
+ const TargetRegisterClass *RC = TLI.getRegClassFor(WorkingVT.getSimpleVT());
+ MachineRegisterInfo &MRI = DAG.getMachineFunction().getRegInfo();
+ Register TMReg = MRI.createVirtualRegister(RC);
+ SDValue Chain = DAG.getEntryNode();
+ Chain = DAG.getCopyToReg(Chain, dl, TMReg, TM);
+ TM = DAG.getCopyFromReg(Chain, dl, TMReg, WorkingVT);
+ }
+
+ SDValue Res = DAG.getNode(ISD::XOR, dl, WorkingVT, WorkingF, TM);
+
+ if (WorkingVT != VT)
+ Res = DAG.getBitcast(VT, Res);
+
+ return Res;
+}
+
bool SelectionDAGLegalize::ExpandNode(SDNode *Node) {
LLVM_DEBUG(dbgs() << "Trying to expand node\n");
SmallVector<SDValue, 8> Results;
@@ -4337,188 +4523,9 @@ bool SelectionDAGLegalize::ExpandNode(SDNode *Node) {
}
Results.push_back(Tmp1);
break;
- case ISD::CT_SELECT: {
- // Constant-time select: F ^ ((T ^ F) & Mask), Mask = 0 - (cond & 1).
- // Bitwise-only — no select/cmov SDNode is constructed, so the CT property
- // holds against any combiner that targets those opcodes. FP types operate
- // on the same-size integer; vectors build the mask as a scalar then splat
- // (avoids illegal vNi1).
- //
- // The masked-diff is routed through a virtual register (CopyToReg /
- // CopyFromReg) below as a forward-looking DAGCombine barrier. This is
- // *not* required for correctness against any combiner in tree today —
- // DAGCombiner has no rewrite that recognizes XOR/AND/XOR-with-sext-mask
- // and reconstructs a SELECT. The chain edge is defense-in-depth against
- // a hypothetical future fold of that form: the dependency partitions the
- // bitwise sequence into a region the combiner can't see through. Cost is
- // at most a coalesce-able MOV per call. ISD::ARITH_FENCE would serve the
- // same purpose (ISel already selects it for any legal type); switching to
- // it is left to a follow-up.
- Tmp1 = Node->getOperand(0); // cond
- Tmp2 = Node->getOperand(1); // T
- Tmp3 = Node->getOperand(2); // F
- EVT VT = Tmp2.getValueType();
-
- // Memory-blend for FP scalars whose same-size integer isn't legal (f64 on
- // i386 no-SSE, x86_fp80, fp128). The bitcast-to-int expansion below can't
- // run since LegalizeDAG is the last legalization stage. Spill T/F, blend
- // chunk-by-chunk at a legal int width in place over the T slot, reload it
- // as the FP type. Fixed and scalable FP vectors fall through to the
- // unified path below.
- if (VT.isFloatingPoint() && !VT.isVector() &&
- !TLI.isTypeLegal(VT.changeTypeToInteger())) {
- const DataLayout &DL = DAG.getDataLayout();
- Type *VTTy = VT.getTypeForEVT(*DAG.getContext());
- unsigned StorageBytes = DL.getTypeStoreSize(VTTy);
- assert(StorageBytes > 0 && "FP type with zero storage size");
-
- // Blend at the widest legal scalar integer width. Chunks narrower than
- // that (e.g. the 2-byte tail of x86_fp80) are zero-extended on load and
- // truncated on store, so every value in the DAG has a legal type.
- MVT BlendVT;
- for (MVT MV : {MVT::i64, MVT::i32, MVT::i16, MVT::i8})
- if (TLI.isTypeLegal(MV)) {
- BlendVT = MV;
- break;
- }
- assert(BlendVT.isValid() && "no legal scalar integer type");
- unsigned BlendBytes = BlendVT.getSizeInBits() / 8;
-
- MachineFunction &MF = DAG.getMachineFunction();
- SDValue StackT = DAG.CreateStackTemporary(VT);
- SDValue StackF = DAG.CreateStackTemporary(VT);
- int FIT = cast<FrameIndexSDNode>(StackT.getNode())->getIndex();
- int FIF = cast<FrameIndexSDNode>(StackF.getNode())->getIndex();
- MachinePointerInfo PIT = MachinePointerInfo::getFixedStack(MF, FIT);
- MachinePointerInfo PIF = MachinePointerInfo::getFixedStack(MF, FIF);
-
- SDValue Chain = DAG.getEntryNode();
- Chain = DAG.getStore(Chain, dl, Tmp2, StackT, PIT);
- Chain = DAG.getStore(Chain, dl, Tmp3, StackF, PIF);
-
- // Walk the storage in power-of-2 chunks, widest first, so every access
- // stays naturally aligned within the max-aligned stack temporaries.
- unsigned Offset = 0;
- while (Offset < StorageBytes) {
- unsigned ChunkBytes =
- std::min(BlendBytes, llvm::bit_floor(StorageBytes - Offset));
- MVT MemVT = MVT::getIntegerVT(ChunkBytes * 8);
- TypeSize Off = TypeSize::getFixed(Offset);
- SDValue TPtr = DAG.getMemBasePlusOffset(StackT, Off, dl);
- SDValue FPtr = DAG.getMemBasePlusOffset(StackF, Off, dl);
-
- SDValue Ti, Fi;
- if (MemVT == BlendVT) {
- Ti = DAG.getLoad(BlendVT, dl, Chain, TPtr, PIT.getWithOffset(Offset));
- Chain = Ti.getValue(1);
- Fi = DAG.getLoad(BlendVT, dl, Chain, FPtr, PIF.getWithOffset(Offset));
- } else {
- Ti = DAG.getExtLoad(ISD::ZEXTLOAD, dl, BlendVT, Chain, TPtr,
- PIT.getWithOffset(Offset), MemVT);
- Chain = Ti.getValue(1);
- Fi = DAG.getExtLoad(ISD::ZEXTLOAD, dl, BlendVT, Chain, FPtr,
- PIF.getWithOffset(Offset), MemVT);
- }
- Chain = Fi.getValue(1);
-
- // Blend this chunk via CT_SELECT on the legal integer type (the
- // recursive node is Expand'd by the scalar-int branch below), then
- // store it back over the now-dead chunk of the T slot. The serial
- // chain keeps each chunk's loads ahead of the store.
- SDValue Ri = DAG.getCTSelect(dl, BlendVT, Tmp1, Ti, Fi);
-
- if (MemVT == BlendVT)
- Chain = DAG.getStore(Chain, dl, Ri, TPtr, PIT.getWithOffset(Offset));
- else
- Chain = DAG.getTruncStore(Chain, dl, Ri, TPtr,
- PIT.getWithOffset(Offset), MemVT);
- Offset += ChunkBytes;
- }
-
- Tmp1 = DAG.getLoad(VT, dl, Chain, StackT, PIT);
- Results.push_back(Tmp1);
- break;
- }
-
- SDValue WorkingT = Tmp2;
- SDValue WorkingF = Tmp3;
- EVT WorkingVT = VT;
-
- bool IsFP = VT.isVector() ? VT.getVectorElementType().isFloatingPoint()
- : VT.isFloatingPoint();
- if (IsFP) {
- WorkingVT = VT.changeTypeToInteger();
- // Scalars with an illegal integer twin took the memory path above; FP
- // vector types are expected to have a legal integer counterpart.
- assert(TLI.isTypeLegal(WorkingVT) &&
- "no legal same-width integer type for CT_SELECT expansion");
- WorkingT = DAG.getBitcast(WorkingVT, Tmp2);
- WorkingF = DAG.getBitcast(WorkingVT, Tmp3);
- }
-
- // Compute the all-ones/all-zeros mask as a scalar, then splat for vectors.
- // The element type of a legal vector is not always a legal scalar type
- // (i32 on riscv64, i64 on riscv32), so build the scalar mask at a legal
- // width: BUILD_VECTOR and SPLAT_VECTOR implicitly truncate a wider
- // scalar, and if every legal scalar is narrower than the element, splat
- // at that width and sign-extend the mask vector (0 and -1 survive both).
- EVT MaskEltVT =
- WorkingVT.isVector() ? WorkingVT.getVectorElementType() : WorkingVT;
- EVT MaskSclVT = MaskEltVT;
- if (!TLI.isTypeLegal(MaskSclVT)) {
- assert(WorkingVT.isVector() && "scalar CT_SELECT type must be legal");
- MaskSclVT = TLI.getLegalTypeToTransformTo(*DAG.getContext(), MaskSclVT);
- assert(TLI.isTypeLegal(MaskSclVT) && "no legal scalar integer type");
- }
- SDValue ScalarCond = Tmp1;
- if (ScalarCond.getValueType() != MaskSclVT)
- ScalarCond = DAG.getAnyExtOrTrunc(ScalarCond, dl, MaskSclVT);
- SDValue ScalarMask =
- DAG.getNode(ISD::SUB, dl, MaskSclVT, DAG.getConstant(0, dl, MaskSclVT),
- DAG.getNode(ISD::AND, dl, MaskSclVT, ScalarCond,
- DAG.getConstant(1, dl, MaskSclVT)));
- SDValue Mask;
- if (!WorkingVT.isVector()) {
- Mask = ScalarMask;
- } else if (MaskSclVT.bitsGE(MaskEltVT)) {
- Mask = DAG.getSplat(WorkingVT, dl, ScalarMask);
- } else {
- EVT NarrowVT =
- WorkingVT.changeVectorElementType(*DAG.getContext(), MaskSclVT);
- assert(TLI.isTypeLegal(NarrowVT) &&
- "no legal vector type for the CT_SELECT mask splat");
- Mask = DAG.getNode(ISD::SIGN_EXTEND, dl, WorkingVT,
- DAG.getSplat(NarrowVT, dl, ScalarMask));
- }
-
- // F ^ ((T ^ F) & Mask)
- SDValue XorTF = DAG.getNode(ISD::XOR, dl, WorkingVT, WorkingT, WorkingF);
- SDValue TM = DAG.getNode(ISD::AND, dl, WorkingVT, XorTF, Mask);
-
- // Forward-looking DAGCombine barrier (see header): route the masked-diff
- // through a vreg so the chain edge partitions it from the surrounding
- // bitwise ops, guarding against a future combiner that might fold
- // XOR/AND/XOR-with-sext-mask back to SELECT. WorkingVT is legal here, so
- // it always has a register class; only scalable vectors are skipped,
- // conservatively.
- if (!WorkingVT.isScalableVector()) {
- const TargetRegisterClass *RC =
- TLI.getRegClassFor(WorkingVT.getSimpleVT());
- MachineRegisterInfo &MRI = DAG.getMachineFunction().getRegInfo();
- Register TMReg = MRI.createVirtualRegister(RC);
- SDValue Chain = DAG.getEntryNode();
- Chain = DAG.getCopyToReg(Chain, dl, TMReg, TM);
- TM = DAG.getCopyFromReg(Chain, dl, TMReg, WorkingVT);
- }
-
- Tmp1 = DAG.getNode(ISD::XOR, dl, WorkingVT, WorkingF, TM);
-
- if (WorkingVT != VT)
- Tmp1 = DAG.getBitcast(VT, Tmp1);
-
- Results.push_back(Tmp1);
+ case ISD::CT_SELECT:
+ Results.push_back(ExpandCTSELECT(Node));
break;
- }
case ISD::BR_JT: {
SDValue Chain = Node->getOperand(0);
SDValue Table = Node->getOperand(1);
@@ -5864,11 +5871,12 @@ void SelectionDAGLegalize::PromoteNode(SDNode *Node) {
Tmp3 = DAG.getNode(ExtOp, dl, NVT, Node->getOperand(2));
// Perform the larger operation, then round down. CT_SELECT carries no
// SDNodeFlags (see ISDOpcodes.h); rebuild it through the helper.
- if (Node->getOpcode() == ISD::CT_SELECT)
+ if (Node->getOpcode() == ISD::CT_SELECT) {
Tmp1 = DAG.getCTSelect(dl, NVT, Tmp1, Tmp2, Tmp3);
- else
- Tmp1 =
- DAG.getNode(ISD::SELECT, dl, NVT, Tmp1, Tmp2, Tmp3, Node->getFlags());
+ } else {
+ SDNodeFlags Flags = Node->getFlags();
+ Tmp1 = DAG.getNode(ISD::SELECT, dl, NVT, Tmp1, Tmp2, Tmp3, Flags);
+ }
if (TruncOp != ISD::FP_ROUND)
Tmp1 = DAG.getNode(TruncOp, dl, Node->getValueType(0), Tmp1);
else
>From 6f2c3f2dd8fde35dab75c554d524deda9500f03b Mon Sep 17 00:00:00 2001
From: AkshayK <iit.akshay at gmail.com>
Date: Thu, 10 Sep 2026 12:43:24 -0400
Subject: [PATCH 11/12] [ConstantTime] Use ARITH_FENCE as the combine barrier
in CT_SELECT expansion
Replace the CopyToReg/CopyFromReg vreg barrier in ExpandCTSELECT with
ISD::ARITH_FENCE. The legalizer no longer creates vregs or picks register
classes via getRegClassFor, and scalable vectors are now covered too.
Codegen tests regenerated; output stays branchless.
---
llvm/lib/CodeGen/SelectionDAG/LegalizeDAG.cpp | 30 +-
.../CodeGen/RISCV/ctselect-fallback-gisel.ll | 2 +
.../RISCV/ctselect-fallback-scalable.ll | 8 +
llvm/test/CodeGen/RISCV/ctselect-fallback.ll | 376 +++--
llvm/test/CodeGen/X86/ctselect.ll | 1453 ++++++++++-------
5 files changed, 1095 insertions(+), 774 deletions(-)
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeDAG.cpp b/llvm/lib/CodeGen/SelectionDAG/LegalizeDAG.cpp
index 263b65a7e9e8a..513b0aaf49fbe 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeDAG.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeDAG.cpp
@@ -26,7 +26,6 @@
#include "llvm/CodeGen/MachineFunction.h"
#include "llvm/CodeGen/MachineJumpTableInfo.h"
#include "llvm/CodeGen/MachineMemOperand.h"
-#include "llvm/CodeGen/MachineRegisterInfo.h"
#include "llvm/CodeGen/RuntimeLibcallUtil.h"
#include "llvm/CodeGen/SelectionDAG.h"
#include "llvm/CodeGen/SelectionDAGNodes.h"
@@ -3222,16 +3221,9 @@ SDValue SelectionDAGLegalize::ExpandCTSELECT(SDNode *Node) {
// on the same-size integer; vectors build the mask as a scalar then splat
// (avoids illegal vNi1).
//
- // The masked-diff is routed through a virtual register (CopyToReg /
- // CopyFromReg) below as a forward-looking DAGCombine barrier. This is
- // *not* required for correctness against any combiner in tree today —
- // DAGCombiner has no rewrite that recognizes XOR/AND/XOR-with-sext-mask
- // and reconstructs a SELECT. The chain edge is defense-in-depth against
- // a hypothetical future fold of that form: the dependency partitions the
- // bitwise sequence into a region the combiner can't see through. Cost is
- // at most a coalesce-able MOV per call. ISD::ARITH_FENCE would serve the
- // same purpose (ISel already selects it for any legal type); switching to
- // it is left to a follow-up.
+ // The masked-diff passes through an ARITH_FENCE so no future combine can
+ // fold the XOR/AND/XOR sequence back into a SELECT. No in-tree combine does
+ // that today; the fence is defense-in-depth and emits no code.
SDValue Cond = Node->getOperand(0);
SDValue T = Node->getOperand(1);
SDValue F = Node->getOperand(2);
@@ -3371,20 +3363,8 @@ SDValue SelectionDAGLegalize::ExpandCTSELECT(SDNode *Node) {
SDValue XorTF = DAG.getNode(ISD::XOR, dl, WorkingVT, WorkingT, WorkingF);
SDValue TM = DAG.getNode(ISD::AND, dl, WorkingVT, XorTF, Mask);
- // Forward-looking DAGCombine barrier (see header): route the masked-diff
- // through a vreg so the chain edge partitions it from the surrounding
- // bitwise ops, guarding against a future combiner that might fold
- // XOR/AND/XOR-with-sext-mask back to SELECT. WorkingVT is legal here, so
- // it always has a register class; only scalable vectors are skipped,
- // conservatively.
- if (!WorkingVT.isScalableVector()) {
- const TargetRegisterClass *RC = TLI.getRegClassFor(WorkingVT.getSimpleVT());
- MachineRegisterInfo &MRI = DAG.getMachineFunction().getRegInfo();
- Register TMReg = MRI.createVirtualRegister(RC);
- SDValue Chain = DAG.getEntryNode();
- Chain = DAG.getCopyToReg(Chain, dl, TMReg, TM);
- TM = DAG.getCopyFromReg(Chain, dl, TMReg, WorkingVT);
- }
+ // DAGCombine barrier (see above).
+ TM = DAG.getNode(ISD::ARITH_FENCE, dl, WorkingVT, TM);
SDValue Res = DAG.getNode(ISD::XOR, dl, WorkingVT, WorkingF, TM);
diff --git a/llvm/test/CodeGen/RISCV/ctselect-fallback-gisel.ll b/llvm/test/CodeGen/RISCV/ctselect-fallback-gisel.ll
index 756d1154b5ef8..741de6444add3 100644
--- a/llvm/test/CodeGen/RISCV/ctselect-fallback-gisel.ll
+++ b/llvm/test/CodeGen/RISCV/ctselect-fallback-gisel.ll
@@ -15,6 +15,7 @@ define i32 @ctsel_fallback_i32(i1 %c, i32 %x, i32 %y) {
; CHECK-NEXT: slli a0, a0, 63
; CHECK-NEXT: srai a0, a0, 63
; CHECK-NEXT: and a0, a1, a0
+; CHECK-NEXT: #ARITH_FENCE
; CHECK-NEXT: xor a0, a2, a0
; CHECK-NEXT: ret
%r = call i32 @llvm.ct.select.i32(i1 %c, i32 %x, i32 %y)
@@ -28,6 +29,7 @@ define i64 @ctsel_fallback_i64(i1 %c, i64 %x, i64 %y) {
; CHECK-NEXT: slli a0, a0, 63
; CHECK-NEXT: srai a0, a0, 63
; CHECK-NEXT: and a0, a1, a0
+; CHECK-NEXT: #ARITH_FENCE
; CHECK-NEXT: xor a0, a2, a0
; CHECK-NEXT: ret
%r = call i64 @llvm.ct.select.i64(i1 %c, i64 %x, i64 %y)
diff --git a/llvm/test/CodeGen/RISCV/ctselect-fallback-scalable.ll b/llvm/test/CodeGen/RISCV/ctselect-fallback-scalable.ll
index d721ab1d76a40..d4aabab9935c7 100644
--- a/llvm/test/CodeGen/RISCV/ctselect-fallback-scalable.ll
+++ b/llvm/test/CodeGen/RISCV/ctselect-fallback-scalable.ll
@@ -20,6 +20,7 @@ define <vscale x 4 x i32> @ctsel_nxv4i32(i1 %c, <vscale x 4 x i32> %x, <vscale x
; RV64V-NEXT: slli a0, a0, 63
; RV64V-NEXT: srai a0, a0, 63
; RV64V-NEXT: vand.vx v8, v8, a0
+; RV64V-NEXT: #ARITH_FENCE
; RV64V-NEXT: vxor.vv v8, v10, v8
; RV64V-NEXT: ret
;
@@ -30,6 +31,7 @@ define <vscale x 4 x i32> @ctsel_nxv4i32(i1 %c, <vscale x 4 x i32> %x, <vscale x
; RV32V-NEXT: slli a0, a0, 31
; RV32V-NEXT: srai a0, a0, 31
; RV32V-NEXT: vand.vx v8, v8, a0
+; RV32V-NEXT: #ARITH_FENCE
; RV32V-NEXT: vxor.vv v8, v10, v8
; RV32V-NEXT: ret
%r = call <vscale x 4 x i32> @llvm.ct.select.nxv4i32(i1 %c, <vscale x 4 x i32> %x, <vscale x 4 x i32> %y)
@@ -44,6 +46,7 @@ define <vscale x 2 x i64> @ctsel_nxv2i64(i1 %c, <vscale x 2 x i64> %x, <vscale x
; RV64V-NEXT: slli a0, a0, 63
; RV64V-NEXT: srai a0, a0, 63
; RV64V-NEXT: vand.vx v8, v8, a0
+; RV64V-NEXT: #ARITH_FENCE
; RV64V-NEXT: vxor.vv v8, v10, v8
; RV64V-NEXT: ret
;
@@ -58,6 +61,7 @@ define <vscale x 2 x i64> @ctsel_nxv2i64(i1 %c, <vscale x 2 x i64> %x, <vscale x
; RV32V-NEXT: vsetvli zero, zero, e64, m2, ta, ma
; RV32V-NEXT: vsext.vf2 v12, v14
; RV32V-NEXT: vand.vv v8, v8, v12
+; RV32V-NEXT: #ARITH_FENCE
; RV32V-NEXT: vxor.vv v8, v10, v8
; RV32V-NEXT: ret
%r = call <vscale x 2 x i64> @llvm.ct.select.nxv2i64(i1 %c, <vscale x 2 x i64> %x, <vscale x 2 x i64> %y)
@@ -72,6 +76,7 @@ define <vscale x 4 x float> @ctsel_nxv4f32(i1 %c, <vscale x 4 x float> %x, <vsca
; RV64V-NEXT: slli a0, a0, 63
; RV64V-NEXT: srai a0, a0, 63
; RV64V-NEXT: vand.vx v8, v8, a0
+; RV64V-NEXT: #ARITH_FENCE
; RV64V-NEXT: vxor.vv v8, v10, v8
; RV64V-NEXT: ret
;
@@ -82,6 +87,7 @@ define <vscale x 4 x float> @ctsel_nxv4f32(i1 %c, <vscale x 4 x float> %x, <vsca
; RV32V-NEXT: slli a0, a0, 31
; RV32V-NEXT: srai a0, a0, 31
; RV32V-NEXT: vand.vx v8, v8, a0
+; RV32V-NEXT: #ARITH_FENCE
; RV32V-NEXT: vxor.vv v8, v10, v8
; RV32V-NEXT: ret
%r = call <vscale x 4 x float> @llvm.ct.select.nxv4f32(i1 %c, <vscale x 4 x float> %x, <vscale x 4 x float> %y)
@@ -96,6 +102,7 @@ define <vscale x 2 x double> @ctsel_nxv2f64(i1 %c, <vscale x 2 x double> %x, <vs
; RV64V-NEXT: slli a0, a0, 63
; RV64V-NEXT: srai a0, a0, 63
; RV64V-NEXT: vand.vx v8, v8, a0
+; RV64V-NEXT: #ARITH_FENCE
; RV64V-NEXT: vxor.vv v8, v10, v8
; RV64V-NEXT: ret
;
@@ -110,6 +117,7 @@ define <vscale x 2 x double> @ctsel_nxv2f64(i1 %c, <vscale x 2 x double> %x, <vs
; RV32V-NEXT: vsetvli zero, zero, e64, m2, ta, ma
; RV32V-NEXT: vsext.vf2 v12, v14
; RV32V-NEXT: vand.vv v8, v8, v12
+; RV32V-NEXT: #ARITH_FENCE
; RV32V-NEXT: vxor.vv v8, v10, v8
; RV32V-NEXT: ret
%r = call <vscale x 2 x double> @llvm.ct.select.nxv2f64(i1 %c, <vscale x 2 x double> %x, <vscale x 2 x double> %y)
diff --git a/llvm/test/CodeGen/RISCV/ctselect-fallback.ll b/llvm/test/CodeGen/RISCV/ctselect-fallback.ll
index 68d5489895a6a..6ad661cfad5a8 100644
--- a/llvm/test/CodeGen/RISCV/ctselect-fallback.ll
+++ b/llvm/test/CodeGen/RISCV/ctselect-fallback.ll
@@ -10,6 +10,7 @@ define i8 @test_ctselect_i8(i1 %cond, i8 %a, i8 %b) {
; RV64-NEXT: slli a0, a0, 63
; RV64-NEXT: srai a0, a0, 63
; RV64-NEXT: and a0, a1, a0
+; RV64-NEXT: #ARITH_FENCE
; RV64-NEXT: xor a0, a2, a0
; RV64-NEXT: ret
;
@@ -19,6 +20,7 @@ define i8 @test_ctselect_i8(i1 %cond, i8 %a, i8 %b) {
; RV32-NEXT: slli a0, a0, 31
; RV32-NEXT: srai a0, a0, 31
; RV32-NEXT: and a0, a1, a0
+; RV32-NEXT: #ARITH_FENCE
; RV32-NEXT: xor a0, a2, a0
; RV32-NEXT: ret
%result = call i8 @llvm.ct.select.i8(i1 %cond, i8 %a, i8 %b)
@@ -31,6 +33,7 @@ define i32 @test_ctselect_i32(i1 %cond, i32 %a, i32 %b) {
; RV64-NEXT: slli a0, a0, 63
; RV64-NEXT: srai a0, a0, 63
; RV64-NEXT: and a0, a1, a0
+; RV64-NEXT: #ARITH_FENCE
; RV64-NEXT: xor a0, a2, a0
; RV64-NEXT: ret
;
@@ -40,6 +43,7 @@ define i32 @test_ctselect_i32(i1 %cond, i32 %a, i32 %b) {
; RV32-NEXT: slli a0, a0, 31
; RV32-NEXT: srai a0, a0, 31
; RV32-NEXT: and a0, a1, a0
+; RV32-NEXT: #ARITH_FENCE
; RV32-NEXT: xor a0, a2, a0
; RV32-NEXT: ret
%result = call i32 @llvm.ct.select.i32(i1 %cond, i32 %a, i32 %b)
@@ -53,6 +57,7 @@ define i64 @test_ctselect_i64(i1 %cond, i64 %a, i64 %b) {
; RV64-NEXT: slli a0, a0, 63
; RV64-NEXT: srai a0, a0, 63
; RV64-NEXT: and a0, a1, a0
+; RV64-NEXT: #ARITH_FENCE
; RV64-NEXT: xor a0, a2, a0
; RV64-NEXT: ret
;
@@ -61,10 +66,12 @@ define i64 @test_ctselect_i64(i1 %cond, i64 %a, i64 %b) {
; RV32-NEXT: slli a0, a0, 31
; RV32-NEXT: xor a1, a1, a3
; RV32-NEXT: srai a0, a0, 31
-; RV32-NEXT: xor a2, a2, a4
; RV32-NEXT: and a1, a1, a0
+; RV32-NEXT: xor a2, a2, a4
+; RV32-NEXT: #ARITH_FENCE
; RV32-NEXT: and a2, a2, a0
; RV32-NEXT: xor a0, a3, a1
+; RV32-NEXT: #ARITH_FENCE
; RV32-NEXT: xor a1, a4, a2
; RV32-NEXT: ret
%result = call i64 @llvm.ct.select.i64(i1 %cond, i64 %a, i64 %b)
@@ -78,6 +85,7 @@ define ptr @test_ctselect_ptr(i1 %cond, ptr %a, ptr %b) {
; RV64-NEXT: slli a0, a0, 63
; RV64-NEXT: srai a0, a0, 63
; RV64-NEXT: and a0, a1, a0
+; RV64-NEXT: #ARITH_FENCE
; RV64-NEXT: xor a0, a2, a0
; RV64-NEXT: ret
;
@@ -87,6 +95,7 @@ define ptr @test_ctselect_ptr(i1 %cond, ptr %a, ptr %b) {
; RV32-NEXT: slli a0, a0, 31
; RV32-NEXT: srai a0, a0, 31
; RV32-NEXT: and a0, a1, a0
+; RV32-NEXT: #ARITH_FENCE
; RV32-NEXT: xor a0, a2, a0
; RV32-NEXT: ret
%result = call ptr @llvm.ct.select.p0(i1 %cond, ptr %a, ptr %b)
@@ -98,12 +107,14 @@ define i32 @test_ctselect_const_true(i32 %a, i32 %b) {
; RV64-LABEL: test_ctselect_const_true:
; RV64: # %bb.0:
; RV64-NEXT: xor a0, a0, a1
+; RV64-NEXT: #ARITH_FENCE
; RV64-NEXT: xor a0, a1, a0
; RV64-NEXT: ret
;
; RV32-LABEL: test_ctselect_const_true:
; RV32: # %bb.0:
; RV32-NEXT: xor a0, a0, a1
+; RV32-NEXT: #ARITH_FENCE
; RV32-NEXT: xor a0, a1, a0
; RV32-NEXT: ret
%result = call i32 @llvm.ct.select.i32(i1 true, i32 %a, i32 %b)
@@ -113,12 +124,16 @@ define i32 @test_ctselect_const_true(i32 %a, i32 %b) {
define i32 @test_ctselect_const_false(i32 %a, i32 %b) {
; RV64-LABEL: test_ctselect_const_false:
; RV64: # %bb.0:
-; RV64-NEXT: mv a0, a1
+; RV64-NEXT: li a0, 0
+; RV64-NEXT: #ARITH_FENCE
+; RV64-NEXT: xor a0, a1, a0
; RV64-NEXT: ret
;
; RV32-LABEL: test_ctselect_const_false:
; RV32: # %bb.0:
-; RV32-NEXT: mv a0, a1
+; RV32-NEXT: li a0, 0
+; RV32-NEXT: #ARITH_FENCE
+; RV32-NEXT: xor a0, a1, a0
; RV32-NEXT: ret
%result = call i32 @llvm.ct.select.i32(i1 false, i32 %a, i32 %b)
ret i32 %result
@@ -135,6 +150,7 @@ define i32 @test_ctselect_icmp_eq(i32 %x, i32 %y, i32 %a, i32 %b) {
; RV64-NEXT: xor a2, a2, a3
; RV64-NEXT: addi a0, a0, -1
; RV64-NEXT: and a0, a2, a0
+; RV64-NEXT: #ARITH_FENCE
; RV64-NEXT: xor a0, a3, a0
; RV64-NEXT: ret
;
@@ -145,6 +161,7 @@ define i32 @test_ctselect_icmp_eq(i32 %x, i32 %y, i32 %a, i32 %b) {
; RV32-NEXT: xor a2, a2, a3
; RV32-NEXT: addi a0, a0, -1
; RV32-NEXT: and a0, a2, a0
+; RV32-NEXT: #ARITH_FENCE
; RV32-NEXT: xor a0, a3, a0
; RV32-NEXT: ret
%cond = icmp eq i32 %x, %y
@@ -160,6 +177,7 @@ define i32 @test_ctselect_icmp_ult(i32 %x, i32 %y, i32 %a, i32 %b) {
; RV64-NEXT: xor a2, a2, a3
; RV64-NEXT: neg a0, a0
; RV64-NEXT: and a0, a2, a0
+; RV64-NEXT: #ARITH_FENCE
; RV64-NEXT: xor a0, a3, a0
; RV64-NEXT: ret
;
@@ -169,6 +187,7 @@ define i32 @test_ctselect_icmp_ult(i32 %x, i32 %y, i32 %a, i32 %b) {
; RV32-NEXT: xor a2, a2, a3
; RV32-NEXT: neg a0, a0
; RV32-NEXT: and a0, a2, a0
+; RV32-NEXT: #ARITH_FENCE
; RV32-NEXT: xor a0, a3, a0
; RV32-NEXT: ret
%cond = icmp ult i32 %x, %y
@@ -186,6 +205,7 @@ define i32 @test_ctselect_load(i1 %cond, ptr %p1, ptr %p2) {
; RV64-NEXT: xor a1, a1, a2
; RV64-NEXT: srai a0, a0, 63
; RV64-NEXT: and a0, a1, a0
+; RV64-NEXT: #ARITH_FENCE
; RV64-NEXT: xor a0, a2, a0
; RV64-NEXT: ret
;
@@ -197,6 +217,7 @@ define i32 @test_ctselect_load(i1 %cond, ptr %p1, ptr %p2) {
; RV32-NEXT: xor a1, a1, a2
; RV32-NEXT: srai a0, a0, 31
; RV32-NEXT: and a0, a1, a0
+; RV32-NEXT: #ARITH_FENCE
; RV32-NEXT: xor a0, a2, a0
; RV32-NEXT: ret
%a = load i32, ptr %p1
@@ -212,10 +233,12 @@ define i32 @test_ctselect_nested_and_i1_to_i32(i1 %c0, i1 %c1, i32 %x, i32 %y) {
; RV64: # %bb.0:
; RV64-NEXT: and a0, a0, a1
; RV64-NEXT: andi a0, a0, 1
+; RV64-NEXT: #ARITH_FENCE
; RV64-NEXT: slli a0, a0, 63
; RV64-NEXT: xor a2, a2, a3
; RV64-NEXT: srai a0, a0, 63
; RV64-NEXT: and a0, a2, a0
+; RV64-NEXT: #ARITH_FENCE
; RV64-NEXT: xor a0, a3, a0
; RV64-NEXT: ret
;
@@ -223,10 +246,12 @@ define i32 @test_ctselect_nested_and_i1_to_i32(i1 %c0, i1 %c1, i32 %x, i32 %y) {
; RV32: # %bb.0:
; RV32-NEXT: and a0, a0, a1
; RV32-NEXT: andi a0, a0, 1
+; RV32-NEXT: #ARITH_FENCE
; RV32-NEXT: slli a0, a0, 31
; RV32-NEXT: xor a2, a2, a3
; RV32-NEXT: srai a0, a0, 31
; RV32-NEXT: and a0, a2, a0
+; RV32-NEXT: #ARITH_FENCE
; RV32-NEXT: xor a0, a3, a0
; RV32-NEXT: ret
%inner = call i1 @llvm.ct.select.i1(i1 %c1, i1 true, i1 false)
@@ -242,10 +267,12 @@ define i32 @test_ctselect_nested_or_i1_to_i32(i1 %c0, i1 %c1, i32 %x, i32 %y) {
; RV64: # %bb.0:
; RV64-NEXT: or a0, a0, a1
; RV64-NEXT: andi a0, a0, 1
+; RV64-NEXT: #ARITH_FENCE
; RV64-NEXT: slli a0, a0, 63
; RV64-NEXT: xor a2, a2, a3
; RV64-NEXT: srai a0, a0, 63
; RV64-NEXT: and a0, a2, a0
+; RV64-NEXT: #ARITH_FENCE
; RV64-NEXT: xor a0, a3, a0
; RV64-NEXT: ret
;
@@ -253,10 +280,12 @@ define i32 @test_ctselect_nested_or_i1_to_i32(i1 %c0, i1 %c1, i32 %x, i32 %y) {
; RV32: # %bb.0:
; RV32-NEXT: or a0, a0, a1
; RV32-NEXT: andi a0, a0, 1
+; RV32-NEXT: #ARITH_FENCE
; RV32-NEXT: slli a0, a0, 31
; RV32-NEXT: xor a2, a2, a3
; RV32-NEXT: srai a0, a0, 31
; RV32-NEXT: and a0, a2, a0
+; RV32-NEXT: #ARITH_FENCE
; RV32-NEXT: xor a0, a3, a0
; RV32-NEXT: ret
%inner = call i1 @llvm.ct.select.i1(i1 %c1, i1 true, i1 false)
@@ -274,10 +303,12 @@ define i32 @test_ctselect_double_nested_and_i1(i1 %c0, i1 %c1, i1 %c2, i32 %x, i
; RV64-NEXT: and a0, a0, a1
; RV64-NEXT: and a0, a0, a2
; RV64-NEXT: andi a0, a0, 1
+; RV64-NEXT: #ARITH_FENCE
; RV64-NEXT: slli a0, a0, 63
; RV64-NEXT: xor a3, a3, a4
; RV64-NEXT: srai a0, a0, 63
; RV64-NEXT: and a0, a3, a0
+; RV64-NEXT: #ARITH_FENCE
; RV64-NEXT: xor a0, a4, a0
; RV64-NEXT: ret
;
@@ -286,10 +317,12 @@ define i32 @test_ctselect_double_nested_and_i1(i1 %c0, i1 %c1, i1 %c2, i32 %x, i
; RV32-NEXT: and a0, a0, a1
; RV32-NEXT: and a0, a0, a2
; RV32-NEXT: andi a0, a0, 1
+; RV32-NEXT: #ARITH_FENCE
; RV32-NEXT: slli a0, a0, 31
; RV32-NEXT: xor a3, a3, a4
; RV32-NEXT: srai a0, a0, 31
; RV32-NEXT: and a0, a3, a0
+; RV32-NEXT: #ARITH_FENCE
; RV32-NEXT: xor a0, a4, a0
; RV32-NEXT: ret
%inner2 = call i1 @llvm.ct.select.i1(i1 %c2, i1 true, i1 false)
@@ -305,15 +338,19 @@ define i32 @test_ctselect_double_nested_mixed_i1(i1 %c0, i1 %c1, i1 %c2, i32 %x,
; RV64: # %bb.0:
; RV64-NEXT: and a0, a0, a1
; RV64-NEXT: andi a0, a0, 1
+; RV64-NEXT: #ARITH_FENCE
; RV64-NEXT: or a0, a0, a2
; RV64-NEXT: andi a0, a0, 1
+; RV64-NEXT: #ARITH_FENCE
; RV64-NEXT: slli a0, a0, 63
; RV64-NEXT: xor a3, a3, a4
; RV64-NEXT: srai a0, a0, 63
; RV64-NEXT: and a3, a3, a0
; RV64-NEXT: xor a4, a4, a5
+; RV64-NEXT: #ARITH_FENCE
; RV64-NEXT: xor a3, a4, a3
; RV64-NEXT: and a0, a3, a0
+; RV64-NEXT: #ARITH_FENCE
; RV64-NEXT: xor a0, a5, a0
; RV64-NEXT: ret
;
@@ -321,15 +358,19 @@ define i32 @test_ctselect_double_nested_mixed_i1(i1 %c0, i1 %c1, i1 %c2, i32 %x,
; RV32: # %bb.0:
; RV32-NEXT: and a0, a0, a1
; RV32-NEXT: andi a0, a0, 1
+; RV32-NEXT: #ARITH_FENCE
; RV32-NEXT: or a0, a0, a2
; RV32-NEXT: andi a0, a0, 1
+; RV32-NEXT: #ARITH_FENCE
; RV32-NEXT: slli a0, a0, 31
; RV32-NEXT: xor a3, a3, a4
; RV32-NEXT: srai a0, a0, 31
; RV32-NEXT: and a3, a3, a0
; RV32-NEXT: xor a4, a4, a5
+; RV32-NEXT: #ARITH_FENCE
; RV32-NEXT: xor a3, a4, a3
; RV32-NEXT: and a0, a3, a0
+; RV32-NEXT: #ARITH_FENCE
; RV32-NEXT: xor a0, a5, a0
; RV32-NEXT: ret
%inner1 = call i1 @llvm.ct.select.i1(i1 %c1, i1 true, i1 false)
@@ -350,10 +391,12 @@ define i32 @test_ctselect_nested(i1 %cond1, i1 %cond2, i32 %a, i32 %b, i32 %c) {
; RV64-NEXT: srai a1, a1, 63
; RV64-NEXT: and a1, a2, a1
; RV64-NEXT: xor a3, a3, a4
+; RV64-NEXT: #ARITH_FENCE
; RV64-NEXT: slli a0, a0, 63
; RV64-NEXT: xor a1, a3, a1
; RV64-NEXT: srai a0, a0, 63
; RV64-NEXT: and a0, a1, a0
+; RV64-NEXT: #ARITH_FENCE
; RV64-NEXT: xor a0, a4, a0
; RV64-NEXT: ret
;
@@ -364,10 +407,12 @@ define i32 @test_ctselect_nested(i1 %cond1, i1 %cond2, i32 %a, i32 %b, i32 %c) {
; RV32-NEXT: srai a1, a1, 31
; RV32-NEXT: and a1, a2, a1
; RV32-NEXT: xor a3, a3, a4
+; RV32-NEXT: #ARITH_FENCE
; RV32-NEXT: slli a0, a0, 31
; RV32-NEXT: xor a1, a3, a1
; RV32-NEXT: srai a0, a0, 31
; RV32-NEXT: and a0, a1, a0
+; RV32-NEXT: #ARITH_FENCE
; RV32-NEXT: xor a0, a4, a0
; RV32-NEXT: ret
%inner = call i32 @llvm.ct.select.i32(i1 %cond2, i32 %a, i32 %b)
@@ -384,6 +429,7 @@ define float @test_ctselect_f32_nan_inf(i1 %cond) {
; RV64-NEXT: srai a0, a0, 63
; RV64-NEXT: and a0, a0, a1
; RV64-NEXT: lui a1, 522240
+; RV64-NEXT: #ARITH_FENCE
; RV64-NEXT: xor a0, a0, a1
; RV64-NEXT: ret
;
@@ -394,6 +440,7 @@ define float @test_ctselect_f32_nan_inf(i1 %cond) {
; RV32-NEXT: srai a0, a0, 31
; RV32-NEXT: and a0, a0, a1
; RV32-NEXT: lui a1, 522240
+; RV32-NEXT: #ARITH_FENCE
; RV32-NEXT: xor a0, a0, a1
; RV32-NEXT: ret
%result = call float @llvm.ct.select.f32(i1 %cond, float 0x7FF8000000000000, float 0x7FF0000000000000)
@@ -410,18 +457,22 @@ define double @test_ctselect_f64_nan_inf(i1 %cond) {
; RV64-NEXT: li a2, 2047
; RV64-NEXT: and a0, a0, a1
; RV64-NEXT: slli a2, a2, 52
+; RV64-NEXT: #ARITH_FENCE
; RV64-NEXT: xor a0, a0, a2
; RV64-NEXT: ret
;
; RV32-LABEL: test_ctselect_f64_nan_inf:
; RV32: # %bb.0:
-; RV32-NEXT: lui a1, 128
+; RV32-NEXT: li a2, 0
; RV32-NEXT: slli a0, a0, 31
; RV32-NEXT: srai a0, a0, 31
+; RV32-NEXT: lui a1, 128
; RV32-NEXT: and a0, a0, a1
; RV32-NEXT: lui a1, 524032
+; RV32-NEXT: #ARITH_FENCE
; RV32-NEXT: xor a1, a0, a1
-; RV32-NEXT: li a0, 0
+; RV32-NEXT: #ARITH_FENCE
+; RV32-NEXT: mv a0, a2
; RV32-NEXT: ret
%result = call double @llvm.ct.select.f64(i1 %cond, double 0x7FF8000000000000, double 0x7FF0000000000000)
ret double %result
@@ -435,6 +486,7 @@ define float @test_ctselect_f32(i1 %cond, float %a, float %b) {
; RV64-NEXT: slli a0, a0, 63
; RV64-NEXT: srai a0, a0, 63
; RV64-NEXT: and a0, a1, a0
+; RV64-NEXT: #ARITH_FENCE
; RV64-NEXT: xor a0, a2, a0
; RV64-NEXT: ret
;
@@ -444,6 +496,7 @@ define float @test_ctselect_f32(i1 %cond, float %a, float %b) {
; RV32-NEXT: slli a0, a0, 31
; RV32-NEXT: srai a0, a0, 31
; RV32-NEXT: and a0, a1, a0
+; RV32-NEXT: #ARITH_FENCE
; RV32-NEXT: xor a0, a2, a0
; RV32-NEXT: ret
%result = call float @llvm.ct.select.f32(i1 %cond, float %a, float %b)
@@ -457,6 +510,7 @@ define double @test_ctselect_f64(i1 %cond, double %a, double %b) {
; RV64-NEXT: slli a0, a0, 63
; RV64-NEXT: srai a0, a0, 63
; RV64-NEXT: and a0, a1, a0
+; RV64-NEXT: #ARITH_FENCE
; RV64-NEXT: xor a0, a2, a0
; RV64-NEXT: ret
;
@@ -465,10 +519,12 @@ define double @test_ctselect_f64(i1 %cond, double %a, double %b) {
; RV32-NEXT: slli a0, a0, 31
; RV32-NEXT: xor a1, a1, a3
; RV32-NEXT: srai a0, a0, 31
-; RV32-NEXT: xor a2, a2, a4
; RV32-NEXT: and a1, a1, a0
+; RV32-NEXT: xor a2, a2, a4
+; RV32-NEXT: #ARITH_FENCE
; RV32-NEXT: and a2, a2, a0
; RV32-NEXT: xor a0, a3, a1
+; RV32-NEXT: #ARITH_FENCE
; RV32-NEXT: xor a1, a4, a2
; RV32-NEXT: ret
%result = call double @llvm.ct.select.f64(i1 %cond, double %a, double %b)
@@ -479,61 +535,69 @@ define double @test_ctselect_f64(i1 %cond, double %a, double %b) {
define <4 x i32> @test_ctselect_v4i32(i1 %cond, <4 x i32> %a, <4 x i32> %b) {
; RV64-LABEL: test_ctselect_v4i32:
; RV64: # %bb.0:
-; RV64-NEXT: lw a4, 0(a3)
-; RV64-NEXT: lw a5, 0(a2)
-; RV64-NEXT: lw a6, 8(a2)
+; RV64-NEXT: lw a4, 0(a2)
+; RV64-NEXT: lw a5, 8(a2)
+; RV64-NEXT: lw a6, 0(a3)
; RV64-NEXT: lw a7, 8(a3)
; RV64-NEXT: lw t0, 16(a3)
; RV64-NEXT: lw a3, 24(a3)
; RV64-NEXT: lw t1, 16(a2)
; RV64-NEXT: lw a2, 24(a2)
-; RV64-NEXT: xor a5, a5, a4
; RV64-NEXT: slli a1, a1, 63
+; RV64-NEXT: xor a4, a4, a6
; RV64-NEXT: srai a1, a1, 63
-; RV64-NEXT: xor a6, a6, a7
+; RV64-NEXT: and a4, a4, a1
+; RV64-NEXT: xor a5, a5, a7
+; RV64-NEXT: #ARITH_FENCE
; RV64-NEXT: and a5, a5, a1
+; RV64-NEXT: xor a4, a6, a4
+; RV64-NEXT: #ARITH_FENCE
+; RV64-NEXT: xor a6, t1, t0
; RV64-NEXT: and a6, a6, a1
-; RV64-NEXT: xor t1, t1, t0
; RV64-NEXT: xor a2, a2, a3
-; RV64-NEXT: and t1, t1, a1
+; RV64-NEXT: xor a5, a7, a5
+; RV64-NEXT: #ARITH_FENCE
; RV64-NEXT: and a1, a2, a1
-; RV64-NEXT: xor a4, a4, a5
-; RV64-NEXT: xor a2, a7, a6
-; RV64-NEXT: xor a5, t0, t1
+; RV64-NEXT: xor a2, t0, a6
+; RV64-NEXT: #ARITH_FENCE
; RV64-NEXT: xor a1, a3, a1
; RV64-NEXT: sw a4, 0(a0)
-; RV64-NEXT: sw a2, 4(a0)
-; RV64-NEXT: sw a5, 8(a0)
+; RV64-NEXT: sw a5, 4(a0)
+; RV64-NEXT: sw a2, 8(a0)
; RV64-NEXT: sw a1, 12(a0)
; RV64-NEXT: ret
;
; RV32-LABEL: test_ctselect_v4i32:
; RV32: # %bb.0:
-; RV32-NEXT: lw a4, 0(a3)
-; RV32-NEXT: lw a5, 0(a2)
-; RV32-NEXT: lw a6, 4(a2)
+; RV32-NEXT: lw a4, 0(a2)
+; RV32-NEXT: lw a5, 4(a2)
+; RV32-NEXT: lw a6, 0(a3)
; RV32-NEXT: lw a7, 4(a3)
; RV32-NEXT: lw t0, 8(a3)
; RV32-NEXT: lw a3, 12(a3)
; RV32-NEXT: lw t1, 8(a2)
; RV32-NEXT: lw a2, 12(a2)
-; RV32-NEXT: xor a5, a5, a4
; RV32-NEXT: slli a1, a1, 31
+; RV32-NEXT: xor a4, a4, a6
; RV32-NEXT: srai a1, a1, 31
-; RV32-NEXT: xor a6, a6, a7
+; RV32-NEXT: and a4, a4, a1
+; RV32-NEXT: xor a5, a5, a7
+; RV32-NEXT: #ARITH_FENCE
; RV32-NEXT: and a5, a5, a1
+; RV32-NEXT: xor a4, a6, a4
+; RV32-NEXT: #ARITH_FENCE
+; RV32-NEXT: xor a6, t1, t0
; RV32-NEXT: and a6, a6, a1
-; RV32-NEXT: xor t1, t1, t0
; RV32-NEXT: xor a2, a2, a3
-; RV32-NEXT: and t1, t1, a1
+; RV32-NEXT: xor a5, a7, a5
+; RV32-NEXT: #ARITH_FENCE
; RV32-NEXT: and a1, a2, a1
-; RV32-NEXT: xor a4, a4, a5
-; RV32-NEXT: xor a2, a7, a6
-; RV32-NEXT: xor a5, t0, t1
+; RV32-NEXT: xor a2, t0, a6
+; RV32-NEXT: #ARITH_FENCE
; RV32-NEXT: xor a1, a3, a1
; RV32-NEXT: sw a4, 0(a0)
-; RV32-NEXT: sw a2, 4(a0)
-; RV32-NEXT: sw a5, 8(a0)
+; RV32-NEXT: sw a5, 4(a0)
+; RV32-NEXT: sw a2, 8(a0)
; RV32-NEXT: sw a1, 12(a0)
; RV32-NEXT: ret
%result = call <4 x i32> @llvm.ct.select.v4i32(i1 %cond, <4 x i32> %a, <4 x i32> %b)
@@ -542,61 +606,69 @@ define <4 x i32> @test_ctselect_v4i32(i1 %cond, <4 x i32> %a, <4 x i32> %b) {
define <4 x float> @test_ctselect_v4f32(i1 %cond, <4 x float> %a, <4 x float> %b) {
; RV64-LABEL: test_ctselect_v4f32:
; RV64: # %bb.0:
-; RV64-NEXT: lw a4, 0(a3)
-; RV64-NEXT: lw a5, 0(a2)
-; RV64-NEXT: lw a6, 8(a2)
+; RV64-NEXT: lw a4, 0(a2)
+; RV64-NEXT: lw a5, 8(a2)
+; RV64-NEXT: lw a6, 0(a3)
; RV64-NEXT: lw a7, 8(a3)
; RV64-NEXT: lw t0, 16(a3)
; RV64-NEXT: lw a3, 24(a3)
; RV64-NEXT: lw t1, 16(a2)
; RV64-NEXT: lw a2, 24(a2)
-; RV64-NEXT: xor a5, a5, a4
; RV64-NEXT: slli a1, a1, 63
+; RV64-NEXT: xor a4, a4, a6
; RV64-NEXT: srai a1, a1, 63
-; RV64-NEXT: xor a6, a6, a7
+; RV64-NEXT: and a4, a4, a1
+; RV64-NEXT: xor a5, a5, a7
+; RV64-NEXT: #ARITH_FENCE
; RV64-NEXT: and a5, a5, a1
+; RV64-NEXT: xor a4, a6, a4
+; RV64-NEXT: #ARITH_FENCE
+; RV64-NEXT: xor a6, t1, t0
; RV64-NEXT: and a6, a6, a1
-; RV64-NEXT: xor t1, t1, t0
; RV64-NEXT: xor a2, a2, a3
-; RV64-NEXT: and t1, t1, a1
+; RV64-NEXT: xor a5, a7, a5
+; RV64-NEXT: #ARITH_FENCE
; RV64-NEXT: and a1, a2, a1
-; RV64-NEXT: xor a4, a4, a5
-; RV64-NEXT: xor a2, a7, a6
-; RV64-NEXT: xor a5, t0, t1
+; RV64-NEXT: xor a2, t0, a6
+; RV64-NEXT: #ARITH_FENCE
; RV64-NEXT: xor a1, a3, a1
; RV64-NEXT: sw a4, 0(a0)
-; RV64-NEXT: sw a2, 4(a0)
-; RV64-NEXT: sw a5, 8(a0)
+; RV64-NEXT: sw a5, 4(a0)
+; RV64-NEXT: sw a2, 8(a0)
; RV64-NEXT: sw a1, 12(a0)
; RV64-NEXT: ret
;
; RV32-LABEL: test_ctselect_v4f32:
; RV32: # %bb.0:
-; RV32-NEXT: lw a4, 0(a3)
-; RV32-NEXT: lw a5, 0(a2)
-; RV32-NEXT: lw a6, 4(a2)
+; RV32-NEXT: lw a4, 0(a2)
+; RV32-NEXT: lw a5, 4(a2)
+; RV32-NEXT: lw a6, 0(a3)
; RV32-NEXT: lw a7, 4(a3)
; RV32-NEXT: lw t0, 8(a3)
; RV32-NEXT: lw a3, 12(a3)
; RV32-NEXT: lw t1, 8(a2)
; RV32-NEXT: lw a2, 12(a2)
-; RV32-NEXT: xor a5, a5, a4
; RV32-NEXT: slli a1, a1, 31
+; RV32-NEXT: xor a4, a4, a6
; RV32-NEXT: srai a1, a1, 31
-; RV32-NEXT: xor a6, a6, a7
+; RV32-NEXT: and a4, a4, a1
+; RV32-NEXT: xor a5, a5, a7
+; RV32-NEXT: #ARITH_FENCE
; RV32-NEXT: and a5, a5, a1
+; RV32-NEXT: xor a4, a6, a4
+; RV32-NEXT: #ARITH_FENCE
+; RV32-NEXT: xor a6, t1, t0
; RV32-NEXT: and a6, a6, a1
-; RV32-NEXT: xor t1, t1, t0
; RV32-NEXT: xor a2, a2, a3
-; RV32-NEXT: and t1, t1, a1
+; RV32-NEXT: xor a5, a7, a5
+; RV32-NEXT: #ARITH_FENCE
; RV32-NEXT: and a1, a2, a1
-; RV32-NEXT: xor a4, a4, a5
-; RV32-NEXT: xor a2, a7, a6
-; RV32-NEXT: xor a5, t0, t1
+; RV32-NEXT: xor a2, t0, a6
+; RV32-NEXT: #ARITH_FENCE
; RV32-NEXT: xor a1, a3, a1
; RV32-NEXT: sw a4, 0(a0)
-; RV32-NEXT: sw a2, 4(a0)
-; RV32-NEXT: sw a5, 8(a0)
+; RV32-NEXT: sw a5, 4(a0)
+; RV32-NEXT: sw a2, 8(a0)
; RV32-NEXT: sw a1, 12(a0)
; RV32-NEXT: ret
%result = call <4 x float> @llvm.ct.select.v4f32(i1 %cond, <4 x float> %a, <4 x float> %b)
@@ -613,56 +685,64 @@ define <8 x i32> @test_ctselect_v8i32(i1 %cond, <8 x i32> %a, <8 x i32> %b) {
; RV64-NEXT: .cfi_offset s0, -8
; RV64-NEXT: .cfi_offset s1, -16
; RV64-NEXT: .cfi_offset s2, -24
-; RV64-NEXT: lw a4, 32(a3)
-; RV64-NEXT: lw a5, 40(a3)
-; RV64-NEXT: lw a6, 48(a3)
-; RV64-NEXT: lw a7, 56(a3)
-; RV64-NEXT: lw t0, 32(a2)
-; RV64-NEXT: lw t1, 40(a2)
-; RV64-NEXT: lw t2, 48(a2)
-; RV64-NEXT: lw t3, 56(a2)
-; RV64-NEXT: lw t4, 0(a3)
-; RV64-NEXT: lw t5, 0(a2)
-; RV64-NEXT: lw t6, 8(a2)
-; RV64-NEXT: lw s0, 8(a3)
-; RV64-NEXT: lw s1, 16(a3)
-; RV64-NEXT: lw a3, 24(a3)
-; RV64-NEXT: lw s2, 16(a2)
-; RV64-NEXT: lw a2, 24(a2)
-; RV64-NEXT: xor t5, t5, t4
+; RV64-NEXT: lw a4, 0(a2)
+; RV64-NEXT: lw a5, 0(a3)
+; RV64-NEXT: lw a6, 8(a3)
+; RV64-NEXT: lw a7, 16(a3)
+; RV64-NEXT: lw t0, 24(a3)
+; RV64-NEXT: lw t1, 8(a2)
+; RV64-NEXT: lw t2, 16(a2)
+; RV64-NEXT: lw t3, 24(a2)
+; RV64-NEXT: xor a4, a4, a5
; RV64-NEXT: slli a1, a1, 63
; RV64-NEXT: srai a1, a1, 63
-; RV64-NEXT: xor t6, t6, s0
-; RV64-NEXT: and t5, t5, a1
-; RV64-NEXT: and t6, t6, a1
-; RV64-NEXT: xor s2, s2, s1
-; RV64-NEXT: xor a2, a2, a3
-; RV64-NEXT: and s2, s2, a1
-; RV64-NEXT: and a2, a2, a1
-; RV64-NEXT: xor t0, t0, a4
-; RV64-NEXT: xor t1, t1, a5
-; RV64-NEXT: and t0, t0, a1
+; RV64-NEXT: lw t4, 32(a3)
+; RV64-NEXT: lw t5, 40(a3)
+; RV64-NEXT: lw t6, 48(a3)
+; RV64-NEXT: lw a3, 56(a3)
+; RV64-NEXT: and a4, a4, a1
+; RV64-NEXT: xor t1, t1, a6
+; RV64-NEXT: lw s0, 32(a2)
+; RV64-NEXT: lw s1, 40(a2)
+; RV64-NEXT: lw s2, 48(a2)
+; RV64-NEXT: lw a2, 56(a2)
+; RV64-NEXT: #ARITH_FENCE
; RV64-NEXT: and t1, t1, a1
-; RV64-NEXT: xor t2, t2, a6
-; RV64-NEXT: xor t3, t3, a7
+; RV64-NEXT: xor a4, a5, a4
+; RV64-NEXT: #ARITH_FENCE
+; RV64-NEXT: xor a5, t2, a7
+; RV64-NEXT: and a5, a5, a1
+; RV64-NEXT: xor t2, t3, t0
+; RV64-NEXT: xor a6, a6, t1
+; RV64-NEXT: #ARITH_FENCE
+; RV64-NEXT: and t1, t2, a1
+; RV64-NEXT: xor a5, a7, a5
+; RV64-NEXT: #ARITH_FENCE
+; RV64-NEXT: xor a7, s0, t4
+; RV64-NEXT: and a7, a7, a1
+; RV64-NEXT: xor t2, s1, t5
+; RV64-NEXT: xor t0, t0, t1
+; RV64-NEXT: #ARITH_FENCE
+; RV64-NEXT: and t1, t2, a1
+; RV64-NEXT: xor a7, t4, a7
+; RV64-NEXT: #ARITH_FENCE
+; RV64-NEXT: xor t2, s2, t6
; RV64-NEXT: and t2, t2, a1
-; RV64-NEXT: and a1, t3, a1
-; RV64-NEXT: xor t3, t4, t5
-; RV64-NEXT: xor t4, s0, t6
-; RV64-NEXT: xor t5, s1, s2
-; RV64-NEXT: xor a2, a3, a2
-; RV64-NEXT: xor a3, a4, t0
-; RV64-NEXT: xor a4, a5, t1
-; RV64-NEXT: xor a5, a6, t2
-; RV64-NEXT: xor a1, a7, a1
-; RV64-NEXT: sw a3, 16(a0)
-; RV64-NEXT: sw a4, 20(a0)
-; RV64-NEXT: sw a5, 24(a0)
+; RV64-NEXT: xor a2, a2, a3
+; RV64-NEXT: xor t1, t5, t1
+; RV64-NEXT: #ARITH_FENCE
+; RV64-NEXT: and a1, a2, a1
+; RV64-NEXT: xor a2, t6, t2
+; RV64-NEXT: #ARITH_FENCE
+; RV64-NEXT: xor a1, a3, a1
+; RV64-NEXT: sw a7, 16(a0)
+; RV64-NEXT: sw t1, 20(a0)
+; RV64-NEXT: sw a2, 24(a0)
; RV64-NEXT: sw a1, 28(a0)
-; RV64-NEXT: sw t3, 0(a0)
-; RV64-NEXT: sw t4, 4(a0)
-; RV64-NEXT: sw t5, 8(a0)
-; RV64-NEXT: sw a2, 12(a0)
+; RV64-NEXT: sw a4, 0(a0)
+; RV64-NEXT: sw a6, 4(a0)
+; RV64-NEXT: sw a5, 8(a0)
+; RV64-NEXT: sw t0, 12(a0)
; RV64-NEXT: ld s0, 24(sp) # 8-byte Folded Reload
; RV64-NEXT: ld s1, 16(sp) # 8-byte Folded Reload
; RV64-NEXT: ld s2, 8(sp) # 8-byte Folded Reload
@@ -683,56 +763,64 @@ define <8 x i32> @test_ctselect_v8i32(i1 %cond, <8 x i32> %a, <8 x i32> %b) {
; RV32-NEXT: .cfi_offset s0, -4
; RV32-NEXT: .cfi_offset s1, -8
; RV32-NEXT: .cfi_offset s2, -12
-; RV32-NEXT: lw a4, 16(a3)
-; RV32-NEXT: lw a5, 20(a3)
-; RV32-NEXT: lw a6, 24(a3)
-; RV32-NEXT: lw a7, 28(a3)
-; RV32-NEXT: lw t0, 16(a2)
-; RV32-NEXT: lw t1, 20(a2)
-; RV32-NEXT: lw t2, 24(a2)
-; RV32-NEXT: lw t3, 28(a2)
-; RV32-NEXT: lw t4, 0(a3)
-; RV32-NEXT: lw t5, 0(a2)
-; RV32-NEXT: lw t6, 4(a2)
-; RV32-NEXT: lw s0, 4(a3)
-; RV32-NEXT: lw s1, 8(a3)
-; RV32-NEXT: lw a3, 12(a3)
-; RV32-NEXT: lw s2, 8(a2)
-; RV32-NEXT: lw a2, 12(a2)
-; RV32-NEXT: xor t5, t5, t4
+; RV32-NEXT: lw a4, 0(a2)
+; RV32-NEXT: lw a5, 0(a3)
+; RV32-NEXT: lw a6, 4(a3)
+; RV32-NEXT: lw a7, 8(a3)
+; RV32-NEXT: lw t0, 12(a3)
+; RV32-NEXT: lw t1, 4(a2)
+; RV32-NEXT: lw t2, 8(a2)
+; RV32-NEXT: lw t3, 12(a2)
+; RV32-NEXT: xor a4, a4, a5
; RV32-NEXT: slli a1, a1, 31
; RV32-NEXT: srai a1, a1, 31
-; RV32-NEXT: xor t6, t6, s0
-; RV32-NEXT: and t5, t5, a1
-; RV32-NEXT: and t6, t6, a1
-; RV32-NEXT: xor s2, s2, s1
-; RV32-NEXT: xor a2, a2, a3
-; RV32-NEXT: and s2, s2, a1
-; RV32-NEXT: and a2, a2, a1
-; RV32-NEXT: xor t0, t0, a4
-; RV32-NEXT: xor t1, t1, a5
-; RV32-NEXT: and t0, t0, a1
+; RV32-NEXT: lw t4, 16(a3)
+; RV32-NEXT: lw t5, 20(a3)
+; RV32-NEXT: lw t6, 24(a3)
+; RV32-NEXT: lw a3, 28(a3)
+; RV32-NEXT: and a4, a4, a1
+; RV32-NEXT: xor t1, t1, a6
+; RV32-NEXT: lw s0, 16(a2)
+; RV32-NEXT: lw s1, 20(a2)
+; RV32-NEXT: lw s2, 24(a2)
+; RV32-NEXT: lw a2, 28(a2)
+; RV32-NEXT: #ARITH_FENCE
; RV32-NEXT: and t1, t1, a1
-; RV32-NEXT: xor t2, t2, a6
-; RV32-NEXT: xor t3, t3, a7
+; RV32-NEXT: xor a4, a5, a4
+; RV32-NEXT: #ARITH_FENCE
+; RV32-NEXT: xor a5, t2, a7
+; RV32-NEXT: and a5, a5, a1
+; RV32-NEXT: xor t2, t3, t0
+; RV32-NEXT: xor a6, a6, t1
+; RV32-NEXT: #ARITH_FENCE
+; RV32-NEXT: and t1, t2, a1
+; RV32-NEXT: xor a5, a7, a5
+; RV32-NEXT: #ARITH_FENCE
+; RV32-NEXT: xor a7, s0, t4
+; RV32-NEXT: and a7, a7, a1
+; RV32-NEXT: xor t2, s1, t5
+; RV32-NEXT: xor t0, t0, t1
+; RV32-NEXT: #ARITH_FENCE
+; RV32-NEXT: and t1, t2, a1
+; RV32-NEXT: xor a7, t4, a7
+; RV32-NEXT: #ARITH_FENCE
+; RV32-NEXT: xor t2, s2, t6
; RV32-NEXT: and t2, t2, a1
-; RV32-NEXT: and a1, t3, a1
-; RV32-NEXT: xor t3, t4, t5
-; RV32-NEXT: xor t4, s0, t6
-; RV32-NEXT: xor t5, s1, s2
-; RV32-NEXT: xor a2, a3, a2
-; RV32-NEXT: xor a3, a4, t0
-; RV32-NEXT: xor a4, a5, t1
-; RV32-NEXT: xor a5, a6, t2
-; RV32-NEXT: xor a1, a7, a1
-; RV32-NEXT: sw a3, 16(a0)
-; RV32-NEXT: sw a4, 20(a0)
-; RV32-NEXT: sw a5, 24(a0)
+; RV32-NEXT: xor a2, a2, a3
+; RV32-NEXT: xor t1, t5, t1
+; RV32-NEXT: #ARITH_FENCE
+; RV32-NEXT: and a1, a2, a1
+; RV32-NEXT: xor a2, t6, t2
+; RV32-NEXT: #ARITH_FENCE
+; RV32-NEXT: xor a1, a3, a1
+; RV32-NEXT: sw a7, 16(a0)
+; RV32-NEXT: sw t1, 20(a0)
+; RV32-NEXT: sw a2, 24(a0)
; RV32-NEXT: sw a1, 28(a0)
-; RV32-NEXT: sw t3, 0(a0)
-; RV32-NEXT: sw t4, 4(a0)
-; RV32-NEXT: sw t5, 8(a0)
-; RV32-NEXT: sw a2, 12(a0)
+; RV32-NEXT: sw a4, 0(a0)
+; RV32-NEXT: sw a6, 4(a0)
+; RV32-NEXT: sw a5, 8(a0)
+; RV32-NEXT: sw t0, 12(a0)
; RV32-NEXT: lw s0, 12(sp) # 4-byte Folded Reload
; RV32-NEXT: lw s1, 8(sp) # 4-byte Folded Reload
; RV32-NEXT: lw s2, 4(sp) # 4-byte Folded Reload
diff --git a/llvm/test/CodeGen/X86/ctselect.ll b/llvm/test/CodeGen/X86/ctselect.ll
index 096b5213ec69e..7221b44d4dcc1 100644
--- a/llvm/test/CodeGen/X86/ctselect.ll
+++ b/llvm/test/CodeGen/X86/ctselect.ll
@@ -13,6 +13,7 @@ define i8 @test_ctselect_i8(i1 %cond, i8 %a, i8 %b) #0 {
; X64-NEXT: xorl %edx, %esi
; X64-NEXT: negb %al
; X64-NEXT: andb %sil, %al
+; X64-NEXT: #ARITH_FENCE
; X64-NEXT: xorb %dl, %al
; X64-NEXT: # kill: def $al killed $al killed $eax
; X64-NEXT: retq
@@ -26,6 +27,7 @@ define i8 @test_ctselect_i8(i1 %cond, i8 %a, i8 %b) #0 {
; X32-NEXT: xorb %cl, %dl
; X32-NEXT: negb %al
; X32-NEXT: andb %dl, %al
+; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorb %cl, %al
; X32-NEXT: retl
;
@@ -38,6 +40,7 @@ define i8 @test_ctselect_i8(i1 %cond, i8 %a, i8 %b) #0 {
; X32-NOCMOV-NEXT: xorb %cl, %dl
; X32-NOCMOV-NEXT: negb %al
; X32-NOCMOV-NEXT: andb %dl, %al
+; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorb %cl, %al
; X32-NOCMOV-NEXT: retl
%result = call i8 @llvm.ct.select.i8(i1 %cond, i8 %a, i8 %b)
@@ -52,6 +55,7 @@ define b8 @test_ctselect_b8(i1 %cond, b8 %a, b8 %b) #0 {
; X64-NEXT: xorl %edx, %esi
; X64-NEXT: negb %al
; X64-NEXT: andb %sil, %al
+; X64-NEXT: #ARITH_FENCE
; X64-NEXT: xorb %dl, %al
; X64-NEXT: # kill: def $al killed $al killed $eax
; X64-NEXT: retq
@@ -65,6 +69,7 @@ define b8 @test_ctselect_b8(i1 %cond, b8 %a, b8 %b) #0 {
; X32-NEXT: xorb %cl, %dl
; X32-NEXT: negb %al
; X32-NEXT: andb %dl, %al
+; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorb %cl, %al
; X32-NEXT: retl
;
@@ -77,6 +82,7 @@ define b8 @test_ctselect_b8(i1 %cond, b8 %a, b8 %b) #0 {
; X32-NOCMOV-NEXT: xorb %cl, %dl
; X32-NOCMOV-NEXT: negb %al
; X32-NOCMOV-NEXT: andb %dl, %al
+; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorb %cl, %al
; X32-NOCMOV-NEXT: retl
%result = call b8 @llvm.ct.select.b8(i1 %cond, b8 %a, b8 %b)
@@ -91,6 +97,7 @@ define i32 @test_ctselect_i32(i1 %cond, i32 %a, i32 %b) #0 {
; X64-NEXT: andl $1, %eax
; X64-NEXT: negl %eax
; X64-NEXT: andl %esi, %eax
+; X64-NEXT: #ARITH_FENCE
; X64-NEXT: xorl %edx, %eax
; X64-NEXT: retq
;
@@ -104,6 +111,7 @@ define i32 @test_ctselect_i32(i1 %cond, i32 %a, i32 %b) #0 {
; X32-NEXT: movzbl %al, %eax
; X32-NEXT: negl %eax
; X32-NEXT: andl %edx, %eax
+; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %ecx, %eax
; X32-NEXT: retl
;
@@ -117,6 +125,7 @@ define i32 @test_ctselect_i32(i1 %cond, i32 %a, i32 %b) #0 {
; X32-NOCMOV-NEXT: movzbl %al, %eax
; X32-NOCMOV-NEXT: negl %eax
; X32-NOCMOV-NEXT: andl %edx, %eax
+; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %ecx, %eax
; X32-NOCMOV-NEXT: retl
%result = call i32 @llvm.ct.select.i32(i1 %cond, i32 %a, i32 %b)
@@ -131,6 +140,7 @@ define i64 @test_ctselect_i64(i1 %cond, i64 %a, i64 %b) #0 {
; X64-NEXT: andl $1, %eax
; X64-NEXT: negq %rax
; X64-NEXT: andq %rsi, %rax
+; X64-NEXT: #ARITH_FENCE
; X64-NEXT: xorq %rdx, %rax
; X64-NEXT: retq
;
@@ -147,10 +157,12 @@ define i64 @test_ctselect_i64(i1 %cond, i64 %a, i64 %b) #0 {
; X32-NEXT: movzbl %dl, %edi
; X32-NEXT: negl %edi
; X32-NEXT: andl %edi, %eax
+; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %esi, %eax
; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
; X32-NEXT: xorl %ecx, %edx
; X32-NEXT: andl %edi, %edx
+; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %ecx, %edx
; X32-NEXT: popl %esi
; X32-NEXT: popl %edi
@@ -169,10 +181,12 @@ define i64 @test_ctselect_i64(i1 %cond, i64 %a, i64 %b) #0 {
; X32-NOCMOV-NEXT: movzbl %dl, %edi
; X32-NOCMOV-NEXT: negl %edi
; X32-NOCMOV-NEXT: andl %edi, %eax
+; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %esi, %eax
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
; X32-NOCMOV-NEXT: xorl %ecx, %edx
; X32-NOCMOV-NEXT: andl %edi, %edx
+; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %ecx, %edx
; X32-NOCMOV-NEXT: popl %esi
; X32-NOCMOV-NEXT: popl %edi
@@ -190,6 +204,7 @@ define float @test_ctselect_f32(i1 %cond, float %a, float %b) #0 {
; X64-NEXT: andl $1, %edi
; X64-NEXT: negl %edi
; X64-NEXT: andl %ecx, %edi
+; X64-NEXT: #ARITH_FENCE
; X64-NEXT: xorl %eax, %edi
; X64-NEXT: movd %edi, %xmm0
; X64-NEXT: retq
@@ -209,6 +224,7 @@ define float @test_ctselect_f32(i1 %cond, float %a, float %b) #0 {
; X32-NEXT: movl (%esp), %edx
; X32-NEXT: xorl %ecx, %edx
; X32-NEXT: andl %eax, %edx
+; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %ecx, %edx
; X32-NEXT: movl %edx, {{[0-9]+}}(%esp)
; X32-NEXT: flds {{[0-9]+}}(%esp)
@@ -230,6 +246,7 @@ define float @test_ctselect_f32(i1 %cond, float %a, float %b) #0 {
; X32-NOCMOV-NEXT: movl (%esp), %edx
; X32-NOCMOV-NEXT: xorl %ecx, %edx
; X32-NOCMOV-NEXT: andl %eax, %edx
+; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %ecx, %edx
; X32-NOCMOV-NEXT: movl %edx, {{[0-9]+}}(%esp)
; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
@@ -249,6 +266,7 @@ define double @test_ctselect_f64(i1 %cond, double %a, double %b) #0 {
; X64-NEXT: andl $1, %edi
; X64-NEXT: negq %rdi
; X64-NEXT: andq %rcx, %rdi
+; X64-NEXT: #ARITH_FENCE
; X64-NEXT: xorq %rax, %rdi
; X64-NEXT: movq %rdi, %xmm0
; X64-NEXT: retq
@@ -270,11 +288,13 @@ define double @test_ctselect_f64(i1 %cond, double %a, double %b) #0 {
; X32-NEXT: movl (%esp), %esi
; X32-NEXT: xorl %edx, %esi
; X32-NEXT: andl %eax, %esi
+; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %edx, %esi
; X32-NEXT: movl %esi, (%esp)
; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
; X32-NEXT: xorl %ecx, %edx
; X32-NEXT: andl %eax, %edx
+; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %ecx, %edx
; X32-NEXT: movl %edx, {{[0-9]+}}(%esp)
; X32-NEXT: fldl (%esp)
@@ -299,11 +319,13 @@ define double @test_ctselect_f64(i1 %cond, double %a, double %b) #0 {
; X32-NOCMOV-NEXT: movl (%esp), %esi
; X32-NOCMOV-NEXT: xorl %edx, %esi
; X32-NOCMOV-NEXT: andl %eax, %esi
+; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %edx, %esi
; X32-NOCMOV-NEXT: movl %esi, (%esp)
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
; X32-NOCMOV-NEXT: xorl %ecx, %edx
; X32-NOCMOV-NEXT: andl %eax, %edx
+; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %ecx, %edx
; X32-NOCMOV-NEXT: movl %edx, {{[0-9]+}}(%esp)
; X32-NOCMOV-NEXT: fldl (%esp)
@@ -323,6 +345,7 @@ define half @test_ctselect_f16(i1 %cond, half %a, half %b) #0 {
; X64-NEXT: andl $1, %edi
; X64-NEXT: negl %edi
; X64-NEXT: andl %ecx, %edi
+; X64-NEXT: #ARITH_FENCE
; X64-NEXT: xorl %eax, %edi
; X64-NEXT: pinsrw $0, %edi, %xmm0
; X64-NEXT: retq
@@ -337,6 +360,7 @@ define half @test_ctselect_f16(i1 %cond, half %a, half %b) #0 {
; X32-NEXT: movzbl %al, %eax
; X32-NEXT: negl %eax
; X32-NEXT: andl %edx, %eax
+; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %ecx, %eax
; X32-NEXT: # kill: def $ax killed $ax killed $eax
; X32-NEXT: retl
@@ -351,6 +375,7 @@ define half @test_ctselect_f16(i1 %cond, half %a, half %b) #0 {
; X32-NOCMOV-NEXT: movzbl %al, %eax
; X32-NOCMOV-NEXT: negl %eax
; X32-NOCMOV-NEXT: andl %edx, %eax
+; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %ecx, %eax
; X32-NOCMOV-NEXT: # kill: def $ax killed $ax killed $eax
; X32-NOCMOV-NEXT: retl
@@ -367,6 +392,7 @@ define bfloat @test_ctselect_bf16(i1 %cond, bfloat %a, bfloat %b) #0 {
; X64-NEXT: andl $1, %edi
; X64-NEXT: negl %edi
; X64-NEXT: andl %ecx, %edi
+; X64-NEXT: #ARITH_FENCE
; X64-NEXT: xorl %eax, %edi
; X64-NEXT: pinsrw $0, %edi, %xmm0
; X64-NEXT: retq
@@ -389,6 +415,7 @@ define bfloat @test_ctselect_bf16(i1 %cond, bfloat %a, bfloat %b) #0 {
; X32-NEXT: movzbl %cl, %ecx
; X32-NEXT: negl %ecx
; X32-NEXT: andl %eax, %ecx
+; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %esi, %ecx
; X32-NEXT: shll $16, %ecx
; X32-NEXT: movl %ecx, {{[0-9]+}}(%esp)
@@ -415,6 +442,7 @@ define bfloat @test_ctselect_bf16(i1 %cond, bfloat %a, bfloat %b) #0 {
; X32-NOCMOV-NEXT: movzbl %cl, %ecx
; X32-NOCMOV-NEXT: negl %ecx
; X32-NOCMOV-NEXT: andl %eax, %ecx
+; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %esi, %ecx
; X32-NOCMOV-NEXT: shll $16, %ecx
; X32-NOCMOV-NEXT: movl %ecx, {{[0-9]+}}(%esp)
@@ -439,11 +467,13 @@ define fp128 @test_ctselect_f128(i1 %cond, fp128 %a, fp128 %b) #0 {
; X64-NEXT: movq -{{[0-9]+}}(%rsp), %rdx
; X64-NEXT: xorq %rax, %rdx
; X64-NEXT: andq %rdi, %rdx
+; X64-NEXT: #ARITH_FENCE
; X64-NEXT: xorq %rax, %rdx
; X64-NEXT: movq %rdx, -{{[0-9]+}}(%rsp)
; X64-NEXT: movq -{{[0-9]+}}(%rsp), %rax
; X64-NEXT: xorq %rcx, %rax
; X64-NEXT: andq %rdi, %rax
+; X64-NEXT: #ARITH_FENCE
; X64-NEXT: xorq %rcx, %rax
; X64-NEXT: movq %rax, -{{[0-9]+}}(%rsp)
; X64-NEXT: movaps -{{[0-9]+}}(%rsp), %xmm0
@@ -458,32 +488,36 @@ define fp128 @test_ctselect_f128(i1 %cond, fp128 %a, fp128 %b) #0 {
; X32-NEXT: subl $12, %esp
; X32-NEXT: movzbl {{[0-9]+}}(%esp), %ebx
; X32-NEXT: andb $1, %bl
-; X32-NEXT: movl {{[0-9]+}}(%esp), %edi
-; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
-; X32-NEXT: xorl %edi, %ecx
-; X32-NEXT: movzbl %bl, %ebp
-; X32-NEXT: negl %ebp
-; X32-NEXT: andl %ebp, %ecx
-; X32-NEXT: movl {{[0-9]+}}(%esp), %esi
-; X32-NEXT: xorl {{[0-9]+}}(%esp), %esi
-; X32-NEXT: andl %ebp, %esi
-; X32-NEXT: movl {{[0-9]+}}(%esp), %ebx
-; X32-NEXT: xorl {{[0-9]+}}(%esp), %ebx
-; X32-NEXT: andl %ebp, %ebx
-; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
-; X32-NEXT: xorl %edx, %eax
-; X32-NEXT: andl %ebp, %eax
-; X32-NEXT: xorl %edi, %ecx
-; X32-NEXT: xorl {{[0-9]+}}(%esp), %esi
-; X32-NEXT: xorl {{[0-9]+}}(%esp), %ebx
-; X32-NEXT: xorl %edx, %eax
+; X32-NEXT: movl {{[0-9]+}}(%esp), %esi
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ebp
; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
-; X32-NEXT: movl %eax, 12(%edx)
-; X32-NEXT: movl %ebx, 8(%edx)
-; X32-NEXT: movl %esi, 4(%edx)
-; X32-NEXT: movl %ecx, (%edx)
-; X32-NEXT: movl %edx, %eax
+; X32-NEXT: xorl %ecx, %edx
+; X32-NEXT: movzbl %bl, %edi
+; X32-NEXT: negl %edi
+; X32-NEXT: andl %edi, %edx
+; X32-NEXT: #ARITH_FENCE
+; X32-NEXT: xorl %ecx, %edx
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ebx
+; X32-NEXT: xorl %ebp, %ebx
+; X32-NEXT: andl %edi, %ebx
+; X32-NEXT: #ARITH_FENCE
+; X32-NEXT: xorl %ebp, %ebx
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ebp
+; X32-NEXT: xorl %esi, %ebp
+; X32-NEXT: andl %edi, %ebp
+; X32-NEXT: #ARITH_FENCE
+; X32-NEXT: xorl %esi, %ebp
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: xorl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: andl %edi, %ecx
+; X32-NEXT: #ARITH_FENCE
+; X32-NEXT: xorl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: movl %ecx, 12(%eax)
+; X32-NEXT: movl %ebp, 8(%eax)
+; X32-NEXT: movl %ebx, 4(%eax)
+; X32-NEXT: movl %edx, (%eax)
; X32-NEXT: addl $12, %esp
; X32-NEXT: popl %esi
; X32-NEXT: popl %edi
@@ -500,32 +534,36 @@ define fp128 @test_ctselect_f128(i1 %cond, fp128 %a, fp128 %b) #0 {
; X32-NOCMOV-NEXT: subl $12, %esp
; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %ebx
; X32-NOCMOV-NEXT: andb $1, %bl
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edi
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
-; X32-NOCMOV-NEXT: xorl %edi, %ecx
-; X32-NOCMOV-NEXT: movzbl %bl, %ebp
-; X32-NOCMOV-NEXT: negl %ebp
-; X32-NOCMOV-NEXT: andl %ebp, %ecx
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %esi
-; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %esi
-; X32-NOCMOV-NEXT: andl %ebp, %esi
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebx
-; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %ebx
-; X32-NOCMOV-NEXT: andl %ebp, %ebx
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
-; X32-NOCMOV-NEXT: xorl %edx, %eax
-; X32-NOCMOV-NEXT: andl %ebp, %eax
-; X32-NOCMOV-NEXT: xorl %edi, %ecx
-; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %esi
-; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %ebx
-; X32-NOCMOV-NEXT: xorl %edx, %eax
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %esi
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebp
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
-; X32-NOCMOV-NEXT: movl %eax, 12(%edx)
-; X32-NOCMOV-NEXT: movl %ebx, 8(%edx)
-; X32-NOCMOV-NEXT: movl %esi, 4(%edx)
-; X32-NOCMOV-NEXT: movl %ecx, (%edx)
-; X32-NOCMOV-NEXT: movl %edx, %eax
+; X32-NOCMOV-NEXT: xorl %ecx, %edx
+; X32-NOCMOV-NEXT: movzbl %bl, %edi
+; X32-NOCMOV-NEXT: negl %edi
+; X32-NOCMOV-NEXT: andl %edi, %edx
+; X32-NOCMOV-NEXT: #ARITH_FENCE
+; X32-NOCMOV-NEXT: xorl %ecx, %edx
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebx
+; X32-NOCMOV-NEXT: xorl %ebp, %ebx
+; X32-NOCMOV-NEXT: andl %edi, %ebx
+; X32-NOCMOV-NEXT: #ARITH_FENCE
+; X32-NOCMOV-NEXT: xorl %ebp, %ebx
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebp
+; X32-NOCMOV-NEXT: xorl %esi, %ebp
+; X32-NOCMOV-NEXT: andl %edi, %ebp
+; X32-NOCMOV-NEXT: #ARITH_FENCE
+; X32-NOCMOV-NEXT: xorl %esi, %ebp
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: andl %edi, %ecx
+; X32-NOCMOV-NEXT: #ARITH_FENCE
+; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: movl %ecx, 12(%eax)
+; X32-NOCMOV-NEXT: movl %ebp, 8(%eax)
+; X32-NOCMOV-NEXT: movl %ebx, 4(%eax)
+; X32-NOCMOV-NEXT: movl %edx, (%eax)
; X32-NOCMOV-NEXT: addl $12, %esp
; X32-NOCMOV-NEXT: popl %esi
; X32-NOCMOV-NEXT: popl %edi
@@ -550,12 +588,14 @@ define x86_fp80 @test_ctselect_f80(i1 %cond, x86_fp80 %a, x86_fp80 %b) #0 {
; X64-NEXT: movq -{{[0-9]+}}(%rsp), %rcx
; X64-NEXT: xorq %rax, %rcx
; X64-NEXT: andq %rdi, %rcx
+; X64-NEXT: #ARITH_FENCE
; X64-NEXT: xorq %rax, %rcx
; X64-NEXT: movq %rcx, -{{[0-9]+}}(%rsp)
; X64-NEXT: movzwl -{{[0-9]+}}(%rsp), %eax
; X64-NEXT: movzwl -{{[0-9]+}}(%rsp), %ecx
; X64-NEXT: xorl %eax, %ecx
; X64-NEXT: andl %ecx, %edi
+; X64-NEXT: #ARITH_FENCE
; X64-NEXT: xorl %eax, %edi
; X64-NEXT: movw %di, -{{[0-9]+}}(%rsp)
; X64-NEXT: fldt -{{[0-9]+}}(%rsp)
@@ -578,17 +618,20 @@ define x86_fp80 @test_ctselect_f80(i1 %cond, x86_fp80 %a, x86_fp80 %b) #0 {
; X32-NEXT: movl (%esp), %esi
; X32-NEXT: xorl %edx, %esi
; X32-NEXT: andl %eax, %esi
+; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %edx, %esi
; X32-NEXT: movl %esi, (%esp)
; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
; X32-NEXT: xorl %ecx, %edx
; X32-NEXT: andl %eax, %edx
+; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %ecx, %edx
; X32-NEXT: movl %edx, {{[0-9]+}}(%esp)
; X32-NEXT: movzwl {{[0-9]+}}(%esp), %ecx
; X32-NEXT: movzwl {{[0-9]+}}(%esp), %edx
; X32-NEXT: xorl %ecx, %edx
; X32-NEXT: andl %eax, %edx
+; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %ecx, %edx
; X32-NEXT: movw %dx, {{[0-9]+}}(%esp)
; X32-NEXT: fldt (%esp)
@@ -613,17 +656,20 @@ define x86_fp80 @test_ctselect_f80(i1 %cond, x86_fp80 %a, x86_fp80 %b) #0 {
; X32-NOCMOV-NEXT: movl (%esp), %esi
; X32-NOCMOV-NEXT: xorl %edx, %esi
; X32-NOCMOV-NEXT: andl %eax, %esi
+; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %edx, %esi
; X32-NOCMOV-NEXT: movl %esi, (%esp)
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
; X32-NOCMOV-NEXT: xorl %ecx, %edx
; X32-NOCMOV-NEXT: andl %eax, %edx
+; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %ecx, %edx
; X32-NOCMOV-NEXT: movl %edx, {{[0-9]+}}(%esp)
; X32-NOCMOV-NEXT: movzwl {{[0-9]+}}(%esp), %ecx
; X32-NOCMOV-NEXT: movzwl {{[0-9]+}}(%esp), %edx
; X32-NOCMOV-NEXT: xorl %ecx, %edx
; X32-NOCMOV-NEXT: andl %eax, %edx
+; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %ecx, %edx
; X32-NOCMOV-NEXT: movw %dx, {{[0-9]+}}(%esp)
; X32-NOCMOV-NEXT: fldt (%esp)
@@ -642,6 +688,7 @@ define ptr @test_ctselect_ptr(i1 %cond, ptr %a, ptr %b) #0 {
; X64-NEXT: andl $1, %eax
; X64-NEXT: negq %rax
; X64-NEXT: andq %rsi, %rax
+; X64-NEXT: #ARITH_FENCE
; X64-NEXT: xorq %rdx, %rax
; X64-NEXT: retq
;
@@ -655,6 +702,7 @@ define ptr @test_ctselect_ptr(i1 %cond, ptr %a, ptr %b) #0 {
; X32-NEXT: movzbl %al, %eax
; X32-NEXT: negl %eax
; X32-NEXT: andl %edx, %eax
+; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %ecx, %eax
; X32-NEXT: retl
;
@@ -668,6 +716,7 @@ define ptr @test_ctselect_ptr(i1 %cond, ptr %a, ptr %b) #0 {
; X32-NOCMOV-NEXT: movzbl %al, %eax
; X32-NOCMOV-NEXT: negl %eax
; X32-NOCMOV-NEXT: andl %edx, %eax
+; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %ecx, %eax
; X32-NOCMOV-NEXT: retl
%result = call ptr @llvm.ct.select.p0(i1 %cond, ptr %a, ptr %b)
@@ -680,6 +729,7 @@ define i32 @test_ctselect_const_true(i32 %a, i32 %b) #0 {
; X64: # %bb.0:
; X64-NEXT: movl %edi, %eax
; X64-NEXT: xorl %esi, %eax
+; X64-NEXT: #ARITH_FENCE
; X64-NEXT: xorl %esi, %eax
; X64-NEXT: retq
;
@@ -688,6 +738,7 @@ define i32 @test_ctselect_const_true(i32 %a, i32 %b) #0 {
; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
; X32-NEXT: xorl %ecx, %eax
+; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %ecx, %eax
; X32-NEXT: retl
;
@@ -696,6 +747,7 @@ define i32 @test_ctselect_const_true(i32 %a, i32 %b) #0 {
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
; X32-NOCMOV-NEXT: xorl %ecx, %eax
+; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %ecx, %eax
; X32-NOCMOV-NEXT: retl
%result = call i32 @llvm.ct.select.i32(i1 true, i32 %a, i32 %b)
@@ -705,18 +757,22 @@ define i32 @test_ctselect_const_true(i32 %a, i32 %b) #0 {
define i32 @test_ctselect_const_false(i32 %a, i32 %b) #0 {
; X64-LABEL: test_ctselect_const_false:
; X64: # %bb.0:
-; X64-NEXT: movl %esi, %eax
+; X64-NEXT: xorl %eax, %eax
+; X64-NEXT: #ARITH_FENCE
+; X64-NEXT: xorl %esi, %eax
; X64-NEXT: retq
;
; X32-LABEL: test_ctselect_const_false:
; X32: # %bb.0:
; X32-NEXT: xorl %eax, %eax
+; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl {{[0-9]+}}(%esp), %eax
; X32-NEXT: retl
;
; X32-NOCMOV-LABEL: test_ctselect_const_false:
; X32-NOCMOV: # %bb.0:
; X32-NOCMOV-NEXT: xorl %eax, %eax
+; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %eax
; X32-NOCMOV-NEXT: retl
%result = call i32 @llvm.ct.select.i32(i1 false, i32 %a, i32 %b)
@@ -733,6 +789,7 @@ define i32 @test_ctselect_icmp_eq(i32 %x, i32 %y, i32 %a, i32 %b) #0 {
; X64-NEXT: xorl %ecx, %edx
; X64-NEXT: negl %eax
; X64-NEXT: andl %edx, %eax
+; X64-NEXT: #ARITH_FENCE
; X64-NEXT: xorl %ecx, %eax
; X64-NEXT: retq
;
@@ -747,6 +804,7 @@ define i32 @test_ctselect_icmp_eq(i32 %x, i32 %y, i32 %a, i32 %b) #0 {
; X32-NEXT: xorl %ecx, %edx
; X32-NEXT: negl %eax
; X32-NEXT: andl %edx, %eax
+; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %ecx, %eax
; X32-NEXT: retl
;
@@ -761,6 +819,7 @@ define i32 @test_ctselect_icmp_eq(i32 %x, i32 %y, i32 %a, i32 %b) #0 {
; X32-NOCMOV-NEXT: xorl %ecx, %edx
; X32-NOCMOV-NEXT: negl %eax
; X32-NOCMOV-NEXT: andl %edx, %eax
+; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %ecx, %eax
; X32-NOCMOV-NEXT: retl
%cond = icmp eq i32 %x, %y
@@ -776,6 +835,7 @@ define i32 @test_ctselect_icmp_ult(i32 %x, i32 %y, i32 %a, i32 %b) #0 {
; X64-NEXT: cmpl %esi, %edi
; X64-NEXT: sbbl %eax, %eax
; X64-NEXT: andl %edx, %eax
+; X64-NEXT: #ARITH_FENCE
; X64-NEXT: xorl %ecx, %eax
; X64-NEXT: retq
;
@@ -789,6 +849,7 @@ define i32 @test_ctselect_icmp_ult(i32 %x, i32 %y, i32 %a, i32 %b) #0 {
; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
; X32-NEXT: xorl %ecx, %eax
; X32-NEXT: andl %edx, %eax
+; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %ecx, %eax
; X32-NEXT: retl
;
@@ -802,6 +863,7 @@ define i32 @test_ctselect_icmp_ult(i32 %x, i32 %y, i32 %a, i32 %b) #0 {
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
; X32-NOCMOV-NEXT: xorl %ecx, %eax
; X32-NOCMOV-NEXT: andl %edx, %eax
+; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %ecx, %eax
; X32-NOCMOV-NEXT: retl
%cond = icmp ult i32 %x, %y
@@ -817,6 +879,7 @@ define float @test_ctselect_fcmp_oeq(float %x, float %y, float %a, float %b) #0
; X64-NEXT: pxor %xmm3, %xmm2
; X64-NEXT: pand %xmm0, %xmm2
; X64-NEXT: movd %xmm2, %ecx
+; X64-NEXT: #ARITH_FENCE
; X64-NEXT: xorl %eax, %ecx
; X64-NEXT: movd %ecx, %xmm0
; X64-NEXT: retq
@@ -841,6 +904,7 @@ define float @test_ctselect_fcmp_oeq(float %x, float %y, float %a, float %b) #0
; X32-NEXT: movl (%esp), %edx
; X32-NEXT: xorl %eax, %edx
; X32-NEXT: andl %ecx, %edx
+; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %eax, %edx
; X32-NEXT: movl %edx, {{[0-9]+}}(%esp)
; X32-NEXT: flds {{[0-9]+}}(%esp)
@@ -869,6 +933,7 @@ define float @test_ctselect_fcmp_oeq(float %x, float %y, float %a, float %b) #0
; X32-NOCMOV-NEXT: movl (%esp), %edx
; X32-NOCMOV-NEXT: xorl %ecx, %edx
; X32-NOCMOV-NEXT: andl %eax, %edx
+; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %ecx, %edx
; X32-NOCMOV-NEXT: movl %edx, {{[0-9]+}}(%esp)
; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
@@ -889,6 +954,7 @@ define i32 @test_ctselect_load(i1 %cond, ptr %p1, ptr %p2) #0 {
; X64-NEXT: andl $1, %edi
; X64-NEXT: negl %edi
; X64-NEXT: andl %edi, %eax
+; X64-NEXT: #ARITH_FENCE
; X64-NEXT: xorl %ecx, %eax
; X64-NEXT: retq
;
@@ -904,6 +970,7 @@ define i32 @test_ctselect_load(i1 %cond, ptr %p1, ptr %p2) #0 {
; X32-NEXT: movzbl %al, %eax
; X32-NEXT: negl %eax
; X32-NEXT: andl %ecx, %eax
+; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %edx, %eax
; X32-NEXT: retl
;
@@ -919,6 +986,7 @@ define i32 @test_ctselect_load(i1 %cond, ptr %p1, ptr %p2) #0 {
; X32-NOCMOV-NEXT: movzbl %al, %eax
; X32-NOCMOV-NEXT: negl %eax
; X32-NOCMOV-NEXT: andl %ecx, %eax
+; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %edx, %eax
; X32-NOCMOV-NEXT: retl
%a = load i32, ptr %p1
@@ -936,11 +1004,13 @@ define i32 @test_ctselect_nested(i1 %cond1, i1 %cond2, i32 %a, i32 %b, i32 %c) #
; X64-NEXT: andl $1, %esi
; X64-NEXT: negl %esi
; X64-NEXT: andl %edx, %esi
+; X64-NEXT: #ARITH_FENCE
; X64-NEXT: xorl %r8d, %ecx
; X64-NEXT: xorl %esi, %ecx
; X64-NEXT: andl $1, %eax
; X64-NEXT: negl %eax
; X64-NEXT: andl %ecx, %eax
+; X64-NEXT: #ARITH_FENCE
; X64-NEXT: xorl %r8d, %eax
; X64-NEXT: retq
;
@@ -959,11 +1029,13 @@ define i32 @test_ctselect_nested(i1 %cond1, i1 %cond2, i32 %a, i32 %b, i32 %c) #
; X32-NEXT: movzbl %ah, %edi
; X32-NEXT: negl %edi
; X32-NEXT: andl %esi, %edi
+; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %ecx, %edx
; X32-NEXT: xorl %edi, %edx
; X32-NEXT: movzbl %al, %eax
; X32-NEXT: negl %eax
; X32-NEXT: andl %edx, %eax
+; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %ecx, %eax
; X32-NEXT: popl %esi
; X32-NEXT: popl %edi
@@ -984,11 +1056,13 @@ define i32 @test_ctselect_nested(i1 %cond1, i1 %cond2, i32 %a, i32 %b, i32 %c) #
; X32-NOCMOV-NEXT: movzbl %ah, %edi
; X32-NOCMOV-NEXT: negl %edi
; X32-NOCMOV-NEXT: andl %esi, %edi
+; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %ecx, %edx
; X32-NOCMOV-NEXT: xorl %edi, %edx
; X32-NOCMOV-NEXT: movzbl %al, %eax
; X32-NOCMOV-NEXT: negl %eax
; X32-NOCMOV-NEXT: andl %edx, %eax
+; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %ecx, %eax
; X32-NOCMOV-NEXT: popl %esi
; X32-NOCMOV-NEXT: popl %edi
@@ -1006,11 +1080,13 @@ define i32 @test_ctselect_nested_and_i1_to_i32(i1 %c0, i1 %c1, i32 %x, i32 %y) #
; X64: # %bb.0:
; X64-NEXT: andl %esi, %edi
; X64-NEXT: andb $1, %dil
+; X64-NEXT: #ARITH_FENCE
; X64-NEXT: andb $1, %dil
; X64-NEXT: xorl %ecx, %edx
; X64-NEXT: movzbl %dil, %eax
; X64-NEXT: negl %eax
; X64-NEXT: andl %edx, %eax
+; X64-NEXT: #ARITH_FENCE
; X64-NEXT: xorl %ecx, %eax
; X64-NEXT: retq
;
@@ -1020,12 +1096,14 @@ define i32 @test_ctselect_nested_and_i1_to_i32(i1 %c0, i1 %c1, i32 %x, i32 %y) #
; X32-NEXT: movzbl {{[0-9]+}}(%esp), %eax
; X32-NEXT: andb {{[0-9]+}}(%esp), %al
; X32-NEXT: andb $1, %al
+; X32-NEXT: #ARITH_FENCE
; X32-NEXT: andb $1, %al
; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
; X32-NEXT: xorl %ecx, %edx
; X32-NEXT: movzbl %al, %eax
; X32-NEXT: negl %eax
; X32-NEXT: andl %edx, %eax
+; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %ecx, %eax
; X32-NEXT: retl
;
@@ -1035,12 +1113,14 @@ define i32 @test_ctselect_nested_and_i1_to_i32(i1 %c0, i1 %c1, i32 %x, i32 %y) #
; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %eax
; X32-NOCMOV-NEXT: andb {{[0-9]+}}(%esp), %al
; X32-NOCMOV-NEXT: andb $1, %al
+; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: andb $1, %al
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
; X32-NOCMOV-NEXT: xorl %ecx, %edx
; X32-NOCMOV-NEXT: movzbl %al, %eax
; X32-NOCMOV-NEXT: negl %eax
; X32-NOCMOV-NEXT: andl %edx, %eax
+; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %ecx, %eax
; X32-NOCMOV-NEXT: retl
%inner = call i1 @llvm.ct.select.i1(i1 %c1, i1 true, i1 false)
@@ -1057,11 +1137,13 @@ define i32 @test_ctselect_nested_or_i1_to_i32(i1 %c0, i1 %c1, i32 %x, i32 %y) #0
; X64: # %bb.0:
; X64-NEXT: orl %esi, %edi
; X64-NEXT: andb $1, %dil
+; X64-NEXT: #ARITH_FENCE
; X64-NEXT: andb $1, %dil
; X64-NEXT: xorl %ecx, %edx
; X64-NEXT: movzbl %dil, %eax
; X64-NEXT: negl %eax
; X64-NEXT: andl %edx, %eax
+; X64-NEXT: #ARITH_FENCE
; X64-NEXT: xorl %ecx, %eax
; X64-NEXT: retq
;
@@ -1071,12 +1153,14 @@ define i32 @test_ctselect_nested_or_i1_to_i32(i1 %c0, i1 %c1, i32 %x, i32 %y) #0
; X32-NEXT: movzbl {{[0-9]+}}(%esp), %eax
; X32-NEXT: orb {{[0-9]+}}(%esp), %al
; X32-NEXT: andb $1, %al
+; X32-NEXT: #ARITH_FENCE
; X32-NEXT: andb $1, %al
; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
; X32-NEXT: xorl %ecx, %edx
; X32-NEXT: movzbl %al, %eax
; X32-NEXT: negl %eax
; X32-NEXT: andl %edx, %eax
+; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %ecx, %eax
; X32-NEXT: retl
;
@@ -1086,12 +1170,14 @@ define i32 @test_ctselect_nested_or_i1_to_i32(i1 %c0, i1 %c1, i32 %x, i32 %y) #0
; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %eax
; X32-NOCMOV-NEXT: orb {{[0-9]+}}(%esp), %al
; X32-NOCMOV-NEXT: andb $1, %al
+; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: andb $1, %al
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
; X32-NOCMOV-NEXT: xorl %ecx, %edx
; X32-NOCMOV-NEXT: movzbl %al, %eax
; X32-NOCMOV-NEXT: negl %eax
; X32-NOCMOV-NEXT: andl %edx, %eax
+; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %ecx, %eax
; X32-NOCMOV-NEXT: retl
%inner = call i1 @llvm.ct.select.i1(i1 %c1, i1 true, i1 false)
@@ -1111,11 +1197,13 @@ define i32 @test_ctselect_double_nested_and_i1(i1 %c0, i1 %c1, i1 %c2, i32 %x, i
; X64-NEXT: andl %esi, %edi
; X64-NEXT: andl %edx, %edi
; X64-NEXT: andb $1, %dil
+; X64-NEXT: #ARITH_FENCE
; X64-NEXT: andb $1, %dil
; X64-NEXT: xorl %r8d, %ecx
; X64-NEXT: movzbl %dil, %eax
; X64-NEXT: negl %eax
; X64-NEXT: andl %ecx, %eax
+; X64-NEXT: #ARITH_FENCE
; X64-NEXT: xorl %r8d, %eax
; X64-NEXT: retq
;
@@ -1126,12 +1214,14 @@ define i32 @test_ctselect_double_nested_and_i1(i1 %c0, i1 %c1, i1 %c2, i32 %x, i
; X32-NEXT: andb {{[0-9]+}}(%esp), %al
; X32-NEXT: andb {{[0-9]+}}(%esp), %al
; X32-NEXT: andb $1, %al
+; X32-NEXT: #ARITH_FENCE
; X32-NEXT: andb $1, %al
; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
; X32-NEXT: xorl %ecx, %edx
; X32-NEXT: movzbl %al, %eax
; X32-NEXT: negl %eax
; X32-NEXT: andl %edx, %eax
+; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %ecx, %eax
; X32-NEXT: retl
;
@@ -1142,12 +1232,14 @@ define i32 @test_ctselect_double_nested_and_i1(i1 %c0, i1 %c1, i1 %c2, i32 %x, i
; X32-NOCMOV-NEXT: andb {{[0-9]+}}(%esp), %al
; X32-NOCMOV-NEXT: andb {{[0-9]+}}(%esp), %al
; X32-NOCMOV-NEXT: andb $1, %al
+; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: andb $1, %al
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
; X32-NOCMOV-NEXT: xorl %ecx, %edx
; X32-NOCMOV-NEXT: movzbl %al, %eax
; X32-NOCMOV-NEXT: negl %eax
; X32-NOCMOV-NEXT: andl %edx, %eax
+; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %ecx, %eax
; X32-NOCMOV-NEXT: retl
%inner2 = call i1 @llvm.ct.select.i1(i1 %c2, i1 true, i1 false)
@@ -1168,6 +1260,7 @@ define i32 @test_ctselect_negated_cond(i1 %c, i32 %a, i32 %b) #0 {
; X64-NEXT: andl $1, %eax
; X64-NEXT: negl %eax
; X64-NEXT: andl %edx, %eax
+; X64-NEXT: #ARITH_FENCE
; X64-NEXT: xorl %esi, %eax
; X64-NEXT: retq
;
@@ -1181,6 +1274,7 @@ define i32 @test_ctselect_negated_cond(i1 %c, i32 %a, i32 %b) #0 {
; X32-NEXT: movzbl %al, %eax
; X32-NEXT: negl %eax
; X32-NEXT: andl %edx, %eax
+; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %ecx, %eax
; X32-NEXT: retl
;
@@ -1194,6 +1288,7 @@ define i32 @test_ctselect_negated_cond(i1 %c, i32 %a, i32 %b) #0 {
; X32-NOCMOV-NEXT: movzbl %al, %eax
; X32-NOCMOV-NEXT: negl %eax
; X32-NOCMOV-NEXT: andl %edx, %eax
+; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %ecx, %eax
; X32-NOCMOV-NEXT: retl
%not = xor i1 %c, true
@@ -1217,256 +1312,295 @@ define <16 x b8> @test_ctselect_v16b8(i1 %cond, <16 x b8> %a, <16 x b8> %b) #0 {
; X64-NEXT: pshuflw {{.*#+}} xmm2 = xmm2[0,0,0,0,4,5,6,7]
; X64-NEXT: pshufd {{.*#+}} xmm2 = xmm2[0,1,0,1]
; X64-NEXT: pand %xmm2, %xmm0
+; X64-NEXT: #ARITH_FENCE
; X64-NEXT: pxor %xmm1, %xmm0
; X64-NEXT: retq
;
; X32-LABEL: test_ctselect_v16b8:
; X32: # %bb.0:
; X32-NEXT: pushl %ebx
+; X32-NEXT: pushl %esi
; X32-NEXT: subl $12, %esp
+; X32-NEXT: movzbl {{[0-9]+}}(%esp), %ebx
+; X32-NEXT: andb $1, %bl
+; X32-NEXT: movzbl {{[0-9]+}}(%esp), %edx
+; X32-NEXT: movb {{[0-9]+}}(%esp), %ch
+; X32-NEXT: movb {{[0-9]+}}(%esp), %dh
+; X32-NEXT: movb {{[0-9]+}}(%esp), %ah
+; X32-NEXT: movb {{[0-9]+}}(%esp), %bh
+; X32-NEXT: movb {{[0-9]+}}(%esp), %cl
+; X32-NEXT: movb {{[0-9]+}}(%esp), %al
+; X32-NEXT: xorb %cl, %al
+; X32-NEXT: negb %bl
+; X32-NEXT: andb %bl, %al
+; X32-NEXT: #ARITH_FENCE
+; X32-NEXT: xorb %cl, %al
+; X32-NEXT: movb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
+; X32-NEXT: movb {{[0-9]+}}(%esp), %al
+; X32-NEXT: xorb %bh, %al
+; X32-NEXT: andb %bl, %al
+; X32-NEXT: #ARITH_FENCE
+; X32-NEXT: xorb %bh, %al
+; X32-NEXT: movb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
+; X32-NEXT: movb {{[0-9]+}}(%esp), %al
+; X32-NEXT: xorb %ah, %al
+; X32-NEXT: andb %bl, %al
+; X32-NEXT: #ARITH_FENCE
+; X32-NEXT: xorb %ah, %al
+; X32-NEXT: movb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
+; X32-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: xorb %dh, %al
+; X32-NEXT: andb %bl, %al
+; X32-NEXT: #ARITH_FENCE
+; X32-NEXT: xorb %dh, %al
+; X32-NEXT: movb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
+; X32-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: xorb %ch, %al
+; X32-NEXT: andb %bl, %al
+; X32-NEXT: #ARITH_FENCE
+; X32-NEXT: xorb %ch, %al
+; X32-NEXT: movb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
+; X32-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: xorb %dl, %al
+; X32-NEXT: andb %bl, %al
+; X32-NEXT: #ARITH_FENCE
+; X32-NEXT: xorb %dl, %al
+; X32-NEXT: movb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
; X32-NEXT: movzbl {{[0-9]+}}(%esp), %eax
-; X32-NEXT: andb $1, %al
-; X32-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
-; X32-NEXT: xorb {{[0-9]+}}(%esp), %cl
-; X32-NEXT: negb %al
-; X32-NEXT: andb %al, %cl
-; X32-NEXT: movb %cl, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
-; X32-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
-; X32-NEXT: xorb {{[0-9]+}}(%esp), %cl
-; X32-NEXT: andb %al, %cl
-; X32-NEXT: movb %cl, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
-; X32-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
-; X32-NEXT: xorb {{[0-9]+}}(%esp), %cl
-; X32-NEXT: andb %al, %cl
-; X32-NEXT: movb %cl, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
-; X32-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
-; X32-NEXT: xorb {{[0-9]+}}(%esp), %cl
-; X32-NEXT: andb %al, %cl
-; X32-NEXT: movb %cl, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
-; X32-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
-; X32-NEXT: xorb {{[0-9]+}}(%esp), %cl
-; X32-NEXT: andb %al, %cl
-; X32-NEXT: movb %cl, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
-; X32-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
-; X32-NEXT: xorb {{[0-9]+}}(%esp), %cl
-; X32-NEXT: andb %al, %cl
-; X32-NEXT: movb %cl, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
-; X32-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
-; X32-NEXT: xorb {{[0-9]+}}(%esp), %cl
-; X32-NEXT: andb %al, %cl
-; X32-NEXT: movb %cl, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
; X32-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
-; X32-NEXT: xorb {{[0-9]+}}(%esp), %cl
-; X32-NEXT: andb %al, %cl
-; X32-NEXT: movb %cl, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
+; X32-NEXT: xorb %cl, %al
+; X32-NEXT: andb %bl, %al
+; X32-NEXT: #ARITH_FENCE
+; X32-NEXT: xorb %cl, %al
+; X32-NEXT: movb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
+; X32-NEXT: movzbl {{[0-9]+}}(%esp), %eax
; X32-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
-; X32-NEXT: xorb {{[0-9]+}}(%esp), %cl
-; X32-NEXT: andb %al, %cl
-; X32-NEXT: movb %cl, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
+; X32-NEXT: xorb %cl, %al
+; X32-NEXT: andb %bl, %al
+; X32-NEXT: #ARITH_FENCE
+; X32-NEXT: xorb %cl, %al
+; X32-NEXT: movb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
+; X32-NEXT: movzbl {{[0-9]+}}(%esp), %eax
; X32-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
-; X32-NEXT: xorb {{[0-9]+}}(%esp), %cl
-; X32-NEXT: andb %al, %cl
+; X32-NEXT: xorb %al, %cl
+; X32-NEXT: andb %bl, %cl
+; X32-NEXT: #ARITH_FENCE
+; X32-NEXT: xorb %al, %cl
; X32-NEXT: movb %cl, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
+; X32-NEXT: movzbl {{[0-9]+}}(%esp), %eax
; X32-NEXT: movb {{[0-9]+}}(%esp), %bh
-; X32-NEXT: xorb {{[0-9]+}}(%esp), %bh
-; X32-NEXT: andb %al, %bh
-; X32-NEXT: movb {{[0-9]+}}(%esp), %bl
-; X32-NEXT: xorb {{[0-9]+}}(%esp), %bl
-; X32-NEXT: andb %al, %bl
+; X32-NEXT: xorb %al, %bh
+; X32-NEXT: andb %bl, %bh
+; X32-NEXT: #ARITH_FENCE
+; X32-NEXT: xorb %al, %bh
+; X32-NEXT: movzbl {{[0-9]+}}(%esp), %eax
; X32-NEXT: movb {{[0-9]+}}(%esp), %dh
-; X32-NEXT: xorb {{[0-9]+}}(%esp), %dh
-; X32-NEXT: andb %al, %dh
+; X32-NEXT: xorb %al, %dh
+; X32-NEXT: andb %bl, %dh
+; X32-NEXT: #ARITH_FENCE
+; X32-NEXT: xorb %al, %dh
+; X32-NEXT: movzbl {{[0-9]+}}(%esp), %eax
; X32-NEXT: movb {{[0-9]+}}(%esp), %ch
-; X32-NEXT: xorb {{[0-9]+}}(%esp), %ch
-; X32-NEXT: andb %al, %ch
-; X32-NEXT: movb {{[0-9]+}}(%esp), %dl
-; X32-NEXT: xorb {{[0-9]+}}(%esp), %dl
-; X32-NEXT: andb %al, %dl
+; X32-NEXT: xorb %al, %ch
+; X32-NEXT: andb %bl, %ch
+; X32-NEXT: #ARITH_FENCE
+; X32-NEXT: xorb %al, %ch
+; X32-NEXT: movzbl {{[0-9]+}}(%esp), %eax
; X32-NEXT: movb {{[0-9]+}}(%esp), %ah
-; X32-NEXT: movb {{[0-9]+}}(%esp), %cl
-; X32-NEXT: xorb %ah, %cl
-; X32-NEXT: andb %al, %cl
-; X32-NEXT: movb {{[-0-9]+}}(%e{{[sb]}}p), %al # 1-byte Reload
-; X32-NEXT: xorb {{[0-9]+}}(%esp), %al
-; X32-NEXT: movb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
-; X32-NEXT: movb {{[0-9]+}}(%esp), %al
-; X32-NEXT: xorb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Folded Spill
-; X32-NEXT: movb {{[0-9]+}}(%esp), %al
-; X32-NEXT: xorb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Folded Spill
-; X32-NEXT: movb {{[0-9]+}}(%esp), %al
-; X32-NEXT: xorb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Folded Spill
-; X32-NEXT: movb {{[0-9]+}}(%esp), %al
-; X32-NEXT: xorb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Folded Spill
-; X32-NEXT: movb {{[0-9]+}}(%esp), %al
-; X32-NEXT: xorb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Folded Spill
+; X32-NEXT: xorb %al, %ah
+; X32-NEXT: andb %bl, %ah
+; X32-NEXT: #ARITH_FENCE
+; X32-NEXT: xorb %al, %ah
; X32-NEXT: movb {{[0-9]+}}(%esp), %al
-; X32-NEXT: xorb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Folded Spill
+; X32-NEXT: movb {{[0-9]+}}(%esp), %dl
+; X32-NEXT: xorb %al, %dl
+; X32-NEXT: andb %bl, %dl
+; X32-NEXT: #ARITH_FENCE
+; X32-NEXT: xorb %al, %dl
; X32-NEXT: movb {{[0-9]+}}(%esp), %al
-; X32-NEXT: xorb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Folded Spill
+; X32-NEXT: movb {{[0-9]+}}(%esp), %cl
+; X32-NEXT: xorb %al, %cl
+; X32-NEXT: andb %bl, %cl
+; X32-NEXT: #ARITH_FENCE
+; X32-NEXT: xorb %al, %cl
; X32-NEXT: movb {{[0-9]+}}(%esp), %al
-; X32-NEXT: xorb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Folded Spill
-; X32-NEXT: movb {{[-0-9]+}}(%e{{[sb]}}p), %al # 1-byte Reload
; X32-NEXT: xorb {{[0-9]+}}(%esp), %al
-; X32-NEXT: movb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
-; X32-NEXT: xorb {{[0-9]+}}(%esp), %bh
-; X32-NEXT: xorb {{[0-9]+}}(%esp), %bl
-; X32-NEXT: xorb {{[0-9]+}}(%esp), %dh
-; X32-NEXT: xorb {{[0-9]+}}(%esp), %ch
-; X32-NEXT: xorb {{[0-9]+}}(%esp), %dl
-; X32-NEXT: xorb %ah, %cl
-; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
-; X32-NEXT: movb %cl, 15(%eax)
-; X32-NEXT: movb %dl, 14(%eax)
-; X32-NEXT: movb %ch, 13(%eax)
-; X32-NEXT: movb %dh, 12(%eax)
-; X32-NEXT: movb %bl, 11(%eax)
-; X32-NEXT: movb %bh, 10(%eax)
-; X32-NEXT: movzbl {{[-0-9]+}}(%e{{[sb]}}p), %ecx # 1-byte Folded Reload
-; X32-NEXT: movb %cl, 9(%eax)
-; X32-NEXT: movzbl {{[-0-9]+}}(%e{{[sb]}}p), %ecx # 1-byte Folded Reload
-; X32-NEXT: movb %cl, 8(%eax)
-; X32-NEXT: movzbl {{[-0-9]+}}(%e{{[sb]}}p), %ecx # 1-byte Folded Reload
-; X32-NEXT: movb %cl, 7(%eax)
-; X32-NEXT: movzbl {{[-0-9]+}}(%e{{[sb]}}p), %ecx # 1-byte Folded Reload
-; X32-NEXT: movb %cl, 6(%eax)
-; X32-NEXT: movzbl {{[-0-9]+}}(%e{{[sb]}}p), %ecx # 1-byte Folded Reload
-; X32-NEXT: movb %cl, 5(%eax)
-; X32-NEXT: movzbl {{[-0-9]+}}(%e{{[sb]}}p), %ecx # 1-byte Folded Reload
-; X32-NEXT: movb %cl, 4(%eax)
-; X32-NEXT: movzbl {{[-0-9]+}}(%e{{[sb]}}p), %ecx # 1-byte Folded Reload
-; X32-NEXT: movb %cl, 3(%eax)
-; X32-NEXT: movzbl {{[-0-9]+}}(%e{{[sb]}}p), %ecx # 1-byte Folded Reload
-; X32-NEXT: movb %cl, 2(%eax)
-; X32-NEXT: movzbl {{[-0-9]+}}(%e{{[sb]}}p), %ecx # 1-byte Folded Reload
-; X32-NEXT: movb %cl, 1(%eax)
-; X32-NEXT: movzbl {{[-0-9]+}}(%e{{[sb]}}p), %ecx # 1-byte Folded Reload
-; X32-NEXT: movb %cl, (%eax)
+; X32-NEXT: andb %bl, %al
+; X32-NEXT: #ARITH_FENCE
+; X32-NEXT: xorb {{[0-9]+}}(%esp), %al
+; X32-NEXT: movl {{[0-9]+}}(%esp), %esi
+; X32-NEXT: movb %al, 15(%esi)
+; X32-NEXT: movb %cl, 14(%esi)
+; X32-NEXT: movb %dl, 13(%esi)
+; X32-NEXT: movb %ah, 12(%esi)
+; X32-NEXT: movb %ch, 11(%esi)
+; X32-NEXT: movb %dh, 10(%esi)
+; X32-NEXT: movb %bh, 9(%esi)
+; X32-NEXT: movzbl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 1-byte Folded Reload
+; X32-NEXT: movb %al, 8(%esi)
+; X32-NEXT: movzbl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 1-byte Folded Reload
+; X32-NEXT: movb %al, 7(%esi)
+; X32-NEXT: movzbl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 1-byte Folded Reload
+; X32-NEXT: movb %al, 6(%esi)
+; X32-NEXT: movzbl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 1-byte Folded Reload
+; X32-NEXT: movb %al, 5(%esi)
+; X32-NEXT: movzbl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 1-byte Folded Reload
+; X32-NEXT: movb %al, 4(%esi)
+; X32-NEXT: movzbl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 1-byte Folded Reload
+; X32-NEXT: movb %al, 3(%esi)
+; X32-NEXT: movzbl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 1-byte Folded Reload
+; X32-NEXT: movb %al, 2(%esi)
+; X32-NEXT: movzbl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 1-byte Folded Reload
+; X32-NEXT: movb %al, 1(%esi)
+; X32-NEXT: movzbl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 1-byte Folded Reload
+; X32-NEXT: movb %al, (%esi)
+; X32-NEXT: movl %esi, %eax
; X32-NEXT: addl $12, %esp
+; X32-NEXT: popl %esi
; X32-NEXT: popl %ebx
; X32-NEXT: retl $4
;
; X32-NOCMOV-LABEL: test_ctselect_v16b8:
; X32-NOCMOV: # %bb.0:
; X32-NOCMOV-NEXT: pushl %ebx
+; X32-NOCMOV-NEXT: pushl %esi
; X32-NOCMOV-NEXT: subl $12, %esp
+; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %ebx
+; X32-NOCMOV-NEXT: andb $1, %bl
+; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %edx
+; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %ch
+; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %dh
+; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %ah
+; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %bh
+; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %cl
+; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %al
+; X32-NOCMOV-NEXT: xorb %cl, %al
+; X32-NOCMOV-NEXT: negb %bl
+; X32-NOCMOV-NEXT: andb %bl, %al
+; X32-NOCMOV-NEXT: #ARITH_FENCE
+; X32-NOCMOV-NEXT: xorb %cl, %al
+; X32-NOCMOV-NEXT: movb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
+; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %al
+; X32-NOCMOV-NEXT: xorb %bh, %al
+; X32-NOCMOV-NEXT: andb %bl, %al
+; X32-NOCMOV-NEXT: #ARITH_FENCE
+; X32-NOCMOV-NEXT: xorb %bh, %al
+; X32-NOCMOV-NEXT: movb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
+; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %al
+; X32-NOCMOV-NEXT: xorb %ah, %al
+; X32-NOCMOV-NEXT: andb %bl, %al
+; X32-NOCMOV-NEXT: #ARITH_FENCE
+; X32-NOCMOV-NEXT: xorb %ah, %al
+; X32-NOCMOV-NEXT: movb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
+; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: xorb %dh, %al
+; X32-NOCMOV-NEXT: andb %bl, %al
+; X32-NOCMOV-NEXT: #ARITH_FENCE
+; X32-NOCMOV-NEXT: xorb %dh, %al
+; X32-NOCMOV-NEXT: movb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
+; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: xorb %ch, %al
+; X32-NOCMOV-NEXT: andb %bl, %al
+; X32-NOCMOV-NEXT: #ARITH_FENCE
+; X32-NOCMOV-NEXT: xorb %ch, %al
+; X32-NOCMOV-NEXT: movb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
+; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: xorb %dl, %al
+; X32-NOCMOV-NEXT: andb %bl, %al
+; X32-NOCMOV-NEXT: #ARITH_FENCE
+; X32-NOCMOV-NEXT: xorb %dl, %al
+; X32-NOCMOV-NEXT: movb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %eax
-; X32-NOCMOV-NEXT: andb $1, %al
-; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
-; X32-NOCMOV-NEXT: xorb {{[0-9]+}}(%esp), %cl
-; X32-NOCMOV-NEXT: negb %al
-; X32-NOCMOV-NEXT: andb %al, %cl
-; X32-NOCMOV-NEXT: movb %cl, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
-; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
-; X32-NOCMOV-NEXT: xorb {{[0-9]+}}(%esp), %cl
-; X32-NOCMOV-NEXT: andb %al, %cl
-; X32-NOCMOV-NEXT: movb %cl, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
-; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
-; X32-NOCMOV-NEXT: xorb {{[0-9]+}}(%esp), %cl
-; X32-NOCMOV-NEXT: andb %al, %cl
-; X32-NOCMOV-NEXT: movb %cl, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
-; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
-; X32-NOCMOV-NEXT: xorb {{[0-9]+}}(%esp), %cl
-; X32-NOCMOV-NEXT: andb %al, %cl
-; X32-NOCMOV-NEXT: movb %cl, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
-; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
-; X32-NOCMOV-NEXT: xorb {{[0-9]+}}(%esp), %cl
-; X32-NOCMOV-NEXT: andb %al, %cl
-; X32-NOCMOV-NEXT: movb %cl, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
-; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
-; X32-NOCMOV-NEXT: xorb {{[0-9]+}}(%esp), %cl
-; X32-NOCMOV-NEXT: andb %al, %cl
-; X32-NOCMOV-NEXT: movb %cl, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
-; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
-; X32-NOCMOV-NEXT: xorb {{[0-9]+}}(%esp), %cl
-; X32-NOCMOV-NEXT: andb %al, %cl
-; X32-NOCMOV-NEXT: movb %cl, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
-; X32-NOCMOV-NEXT: xorb {{[0-9]+}}(%esp), %cl
-; X32-NOCMOV-NEXT: andb %al, %cl
-; X32-NOCMOV-NEXT: movb %cl, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
+; X32-NOCMOV-NEXT: xorb %cl, %al
+; X32-NOCMOV-NEXT: andb %bl, %al
+; X32-NOCMOV-NEXT: #ARITH_FENCE
+; X32-NOCMOV-NEXT: xorb %cl, %al
+; X32-NOCMOV-NEXT: movb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
+; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %eax
; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
-; X32-NOCMOV-NEXT: xorb {{[0-9]+}}(%esp), %cl
-; X32-NOCMOV-NEXT: andb %al, %cl
-; X32-NOCMOV-NEXT: movb %cl, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
+; X32-NOCMOV-NEXT: xorb %cl, %al
+; X32-NOCMOV-NEXT: andb %bl, %al
+; X32-NOCMOV-NEXT: #ARITH_FENCE
+; X32-NOCMOV-NEXT: xorb %cl, %al
+; X32-NOCMOV-NEXT: movb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
+; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %eax
; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
-; X32-NOCMOV-NEXT: xorb {{[0-9]+}}(%esp), %cl
-; X32-NOCMOV-NEXT: andb %al, %cl
+; X32-NOCMOV-NEXT: xorb %al, %cl
+; X32-NOCMOV-NEXT: andb %bl, %cl
+; X32-NOCMOV-NEXT: #ARITH_FENCE
+; X32-NOCMOV-NEXT: xorb %al, %cl
; X32-NOCMOV-NEXT: movb %cl, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
+; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %eax
; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %bh
-; X32-NOCMOV-NEXT: xorb {{[0-9]+}}(%esp), %bh
-; X32-NOCMOV-NEXT: andb %al, %bh
-; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %bl
-; X32-NOCMOV-NEXT: xorb {{[0-9]+}}(%esp), %bl
-; X32-NOCMOV-NEXT: andb %al, %bl
+; X32-NOCMOV-NEXT: xorb %al, %bh
+; X32-NOCMOV-NEXT: andb %bl, %bh
+; X32-NOCMOV-NEXT: #ARITH_FENCE
+; X32-NOCMOV-NEXT: xorb %al, %bh
+; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %eax
; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %dh
-; X32-NOCMOV-NEXT: xorb {{[0-9]+}}(%esp), %dh
-; X32-NOCMOV-NEXT: andb %al, %dh
+; X32-NOCMOV-NEXT: xorb %al, %dh
+; X32-NOCMOV-NEXT: andb %bl, %dh
+; X32-NOCMOV-NEXT: #ARITH_FENCE
+; X32-NOCMOV-NEXT: xorb %al, %dh
+; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %eax
; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %ch
-; X32-NOCMOV-NEXT: xorb {{[0-9]+}}(%esp), %ch
-; X32-NOCMOV-NEXT: andb %al, %ch
-; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %dl
-; X32-NOCMOV-NEXT: xorb {{[0-9]+}}(%esp), %dl
-; X32-NOCMOV-NEXT: andb %al, %dl
+; X32-NOCMOV-NEXT: xorb %al, %ch
+; X32-NOCMOV-NEXT: andb %bl, %ch
+; X32-NOCMOV-NEXT: #ARITH_FENCE
+; X32-NOCMOV-NEXT: xorb %al, %ch
+; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %eax
; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %ah
-; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %cl
-; X32-NOCMOV-NEXT: xorb %ah, %cl
-; X32-NOCMOV-NEXT: andb %al, %cl
-; X32-NOCMOV-NEXT: movb {{[-0-9]+}}(%e{{[sb]}}p), %al # 1-byte Reload
-; X32-NOCMOV-NEXT: xorb {{[0-9]+}}(%esp), %al
-; X32-NOCMOV-NEXT: movb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
-; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %al
-; X32-NOCMOV-NEXT: xorb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Folded Spill
-; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %al
-; X32-NOCMOV-NEXT: xorb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Folded Spill
-; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %al
-; X32-NOCMOV-NEXT: xorb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Folded Spill
+; X32-NOCMOV-NEXT: xorb %al, %ah
+; X32-NOCMOV-NEXT: andb %bl, %ah
+; X32-NOCMOV-NEXT: #ARITH_FENCE
+; X32-NOCMOV-NEXT: xorb %al, %ah
; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %al
-; X32-NOCMOV-NEXT: xorb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Folded Spill
-; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %al
-; X32-NOCMOV-NEXT: xorb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Folded Spill
-; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %al
-; X32-NOCMOV-NEXT: xorb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Folded Spill
+; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %dl
+; X32-NOCMOV-NEXT: xorb %al, %dl
+; X32-NOCMOV-NEXT: andb %bl, %dl
+; X32-NOCMOV-NEXT: #ARITH_FENCE
+; X32-NOCMOV-NEXT: xorb %al, %dl
; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %al
-; X32-NOCMOV-NEXT: xorb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Folded Spill
+; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %cl
+; X32-NOCMOV-NEXT: xorb %al, %cl
+; X32-NOCMOV-NEXT: andb %bl, %cl
+; X32-NOCMOV-NEXT: #ARITH_FENCE
+; X32-NOCMOV-NEXT: xorb %al, %cl
; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %al
-; X32-NOCMOV-NEXT: xorb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Folded Spill
-; X32-NOCMOV-NEXT: movb {{[-0-9]+}}(%e{{[sb]}}p), %al # 1-byte Reload
; X32-NOCMOV-NEXT: xorb {{[0-9]+}}(%esp), %al
-; X32-NOCMOV-NEXT: movb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
-; X32-NOCMOV-NEXT: xorb {{[0-9]+}}(%esp), %bh
-; X32-NOCMOV-NEXT: xorb {{[0-9]+}}(%esp), %bl
-; X32-NOCMOV-NEXT: xorb {{[0-9]+}}(%esp), %dh
-; X32-NOCMOV-NEXT: xorb {{[0-9]+}}(%esp), %ch
-; X32-NOCMOV-NEXT: xorb {{[0-9]+}}(%esp), %dl
-; X32-NOCMOV-NEXT: xorb %ah, %cl
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
-; X32-NOCMOV-NEXT: movb %cl, 15(%eax)
-; X32-NOCMOV-NEXT: movb %dl, 14(%eax)
-; X32-NOCMOV-NEXT: movb %ch, 13(%eax)
-; X32-NOCMOV-NEXT: movb %dh, 12(%eax)
-; X32-NOCMOV-NEXT: movb %bl, 11(%eax)
-; X32-NOCMOV-NEXT: movb %bh, 10(%eax)
-; X32-NOCMOV-NEXT: movzbl {{[-0-9]+}}(%e{{[sb]}}p), %ecx # 1-byte Folded Reload
-; X32-NOCMOV-NEXT: movb %cl, 9(%eax)
-; X32-NOCMOV-NEXT: movzbl {{[-0-9]+}}(%e{{[sb]}}p), %ecx # 1-byte Folded Reload
-; X32-NOCMOV-NEXT: movb %cl, 8(%eax)
-; X32-NOCMOV-NEXT: movzbl {{[-0-9]+}}(%e{{[sb]}}p), %ecx # 1-byte Folded Reload
-; X32-NOCMOV-NEXT: movb %cl, 7(%eax)
-; X32-NOCMOV-NEXT: movzbl {{[-0-9]+}}(%e{{[sb]}}p), %ecx # 1-byte Folded Reload
-; X32-NOCMOV-NEXT: movb %cl, 6(%eax)
-; X32-NOCMOV-NEXT: movzbl {{[-0-9]+}}(%e{{[sb]}}p), %ecx # 1-byte Folded Reload
-; X32-NOCMOV-NEXT: movb %cl, 5(%eax)
-; X32-NOCMOV-NEXT: movzbl {{[-0-9]+}}(%e{{[sb]}}p), %ecx # 1-byte Folded Reload
-; X32-NOCMOV-NEXT: movb %cl, 4(%eax)
-; X32-NOCMOV-NEXT: movzbl {{[-0-9]+}}(%e{{[sb]}}p), %ecx # 1-byte Folded Reload
-; X32-NOCMOV-NEXT: movb %cl, 3(%eax)
-; X32-NOCMOV-NEXT: movzbl {{[-0-9]+}}(%e{{[sb]}}p), %ecx # 1-byte Folded Reload
-; X32-NOCMOV-NEXT: movb %cl, 2(%eax)
-; X32-NOCMOV-NEXT: movzbl {{[-0-9]+}}(%e{{[sb]}}p), %ecx # 1-byte Folded Reload
-; X32-NOCMOV-NEXT: movb %cl, 1(%eax)
-; X32-NOCMOV-NEXT: movzbl {{[-0-9]+}}(%e{{[sb]}}p), %ecx # 1-byte Folded Reload
-; X32-NOCMOV-NEXT: movb %cl, (%eax)
+; X32-NOCMOV-NEXT: andb %bl, %al
+; X32-NOCMOV-NEXT: #ARITH_FENCE
+; X32-NOCMOV-NEXT: xorb {{[0-9]+}}(%esp), %al
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %esi
+; X32-NOCMOV-NEXT: movb %al, 15(%esi)
+; X32-NOCMOV-NEXT: movb %cl, 14(%esi)
+; X32-NOCMOV-NEXT: movb %dl, 13(%esi)
+; X32-NOCMOV-NEXT: movb %ah, 12(%esi)
+; X32-NOCMOV-NEXT: movb %ch, 11(%esi)
+; X32-NOCMOV-NEXT: movb %dh, 10(%esi)
+; X32-NOCMOV-NEXT: movb %bh, 9(%esi)
+; X32-NOCMOV-NEXT: movzbl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 1-byte Folded Reload
+; X32-NOCMOV-NEXT: movb %al, 8(%esi)
+; X32-NOCMOV-NEXT: movzbl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 1-byte Folded Reload
+; X32-NOCMOV-NEXT: movb %al, 7(%esi)
+; X32-NOCMOV-NEXT: movzbl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 1-byte Folded Reload
+; X32-NOCMOV-NEXT: movb %al, 6(%esi)
+; X32-NOCMOV-NEXT: movzbl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 1-byte Folded Reload
+; X32-NOCMOV-NEXT: movb %al, 5(%esi)
+; X32-NOCMOV-NEXT: movzbl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 1-byte Folded Reload
+; X32-NOCMOV-NEXT: movb %al, 4(%esi)
+; X32-NOCMOV-NEXT: movzbl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 1-byte Folded Reload
+; X32-NOCMOV-NEXT: movb %al, 3(%esi)
+; X32-NOCMOV-NEXT: movzbl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 1-byte Folded Reload
+; X32-NOCMOV-NEXT: movb %al, 2(%esi)
+; X32-NOCMOV-NEXT: movzbl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 1-byte Folded Reload
+; X32-NOCMOV-NEXT: movb %al, 1(%esi)
+; X32-NOCMOV-NEXT: movzbl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 1-byte Folded Reload
+; X32-NOCMOV-NEXT: movb %al, (%esi)
+; X32-NOCMOV-NEXT: movl %esi, %eax
; X32-NOCMOV-NEXT: addl $12, %esp
+; X32-NOCMOV-NEXT: popl %esi
; X32-NOCMOV-NEXT: popl %ebx
; X32-NOCMOV-NEXT: retl $4
%result = call <16 x b8> @llvm.ct.select.v16b8(i1 %cond, <16 x b8> %a, <16 x b8> %b)
@@ -1484,6 +1618,7 @@ define <4 x i32> @test_ctselect_v4i32(i1 %cond, <4 x i32> %a, <4 x i32> %b) #0 {
; X64-NEXT: movd %edi, %xmm2
; X64-NEXT: pshufd {{.*#+}} xmm2 = xmm2[0,0,0,0]
; X64-NEXT: pand %xmm2, %xmm0
+; X64-NEXT: #ARITH_FENCE
; X64-NEXT: pxor %xmm1, %xmm0
; X64-NEXT: retq
;
@@ -1495,32 +1630,36 @@ define <4 x i32> @test_ctselect_v4i32(i1 %cond, <4 x i32> %a, <4 x i32> %b) #0 {
; X32-NEXT: pushl %esi
; X32-NEXT: movzbl {{[0-9]+}}(%esp), %ebx
; X32-NEXT: andb $1, %bl
-; X32-NEXT: movl {{[0-9]+}}(%esp), %edi
-; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
-; X32-NEXT: xorl %edi, %ecx
-; X32-NEXT: movzbl %bl, %ebp
-; X32-NEXT: negl %ebp
-; X32-NEXT: andl %ebp, %ecx
-; X32-NEXT: movl {{[0-9]+}}(%esp), %esi
-; X32-NEXT: xorl {{[0-9]+}}(%esp), %esi
-; X32-NEXT: andl %ebp, %esi
-; X32-NEXT: movl {{[0-9]+}}(%esp), %ebx
-; X32-NEXT: xorl {{[0-9]+}}(%esp), %ebx
-; X32-NEXT: andl %ebp, %ebx
-; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
-; X32-NEXT: xorl %edx, %eax
-; X32-NEXT: andl %ebp, %eax
-; X32-NEXT: xorl %edi, %ecx
-; X32-NEXT: xorl {{[0-9]+}}(%esp), %esi
-; X32-NEXT: xorl {{[0-9]+}}(%esp), %ebx
-; X32-NEXT: xorl %edx, %eax
+; X32-NEXT: movl {{[0-9]+}}(%esp), %esi
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ebp
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
-; X32-NEXT: movl %eax, 12(%edx)
-; X32-NEXT: movl %ebx, 8(%edx)
-; X32-NEXT: movl %esi, 4(%edx)
-; X32-NEXT: movl %ecx, (%edx)
-; X32-NEXT: movl %edx, %eax
+; X32-NEXT: xorl %ecx, %edx
+; X32-NEXT: movzbl %bl, %edi
+; X32-NEXT: negl %edi
+; X32-NEXT: andl %edi, %edx
+; X32-NEXT: #ARITH_FENCE
+; X32-NEXT: xorl %ecx, %edx
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ebx
+; X32-NEXT: xorl %ebp, %ebx
+; X32-NEXT: andl %edi, %ebx
+; X32-NEXT: #ARITH_FENCE
+; X32-NEXT: xorl %ebp, %ebx
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ebp
+; X32-NEXT: xorl %esi, %ebp
+; X32-NEXT: andl %edi, %ebp
+; X32-NEXT: #ARITH_FENCE
+; X32-NEXT: xorl %esi, %ebp
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: xorl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: andl %edi, %ecx
+; X32-NEXT: #ARITH_FENCE
+; X32-NEXT: xorl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: movl %ecx, 12(%eax)
+; X32-NEXT: movl %ebp, 8(%eax)
+; X32-NEXT: movl %ebx, 4(%eax)
+; X32-NEXT: movl %edx, (%eax)
; X32-NEXT: popl %esi
; X32-NEXT: popl %edi
; X32-NEXT: popl %ebx
@@ -1535,32 +1674,36 @@ define <4 x i32> @test_ctselect_v4i32(i1 %cond, <4 x i32> %a, <4 x i32> %b) #0 {
; X32-NOCMOV-NEXT: pushl %esi
; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %ebx
; X32-NOCMOV-NEXT: andb $1, %bl
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edi
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
-; X32-NOCMOV-NEXT: xorl %edi, %ecx
-; X32-NOCMOV-NEXT: movzbl %bl, %ebp
-; X32-NOCMOV-NEXT: negl %ebp
-; X32-NOCMOV-NEXT: andl %ebp, %ecx
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %esi
-; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %esi
-; X32-NOCMOV-NEXT: andl %ebp, %esi
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebx
-; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %ebx
-; X32-NOCMOV-NEXT: andl %ebp, %ebx
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
-; X32-NOCMOV-NEXT: xorl %edx, %eax
-; X32-NOCMOV-NEXT: andl %ebp, %eax
-; X32-NOCMOV-NEXT: xorl %edi, %ecx
-; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %esi
-; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %ebx
-; X32-NOCMOV-NEXT: xorl %edx, %eax
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %esi
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebp
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
-; X32-NOCMOV-NEXT: movl %eax, 12(%edx)
-; X32-NOCMOV-NEXT: movl %ebx, 8(%edx)
-; X32-NOCMOV-NEXT: movl %esi, 4(%edx)
-; X32-NOCMOV-NEXT: movl %ecx, (%edx)
-; X32-NOCMOV-NEXT: movl %edx, %eax
+; X32-NOCMOV-NEXT: xorl %ecx, %edx
+; X32-NOCMOV-NEXT: movzbl %bl, %edi
+; X32-NOCMOV-NEXT: negl %edi
+; X32-NOCMOV-NEXT: andl %edi, %edx
+; X32-NOCMOV-NEXT: #ARITH_FENCE
+; X32-NOCMOV-NEXT: xorl %ecx, %edx
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebx
+; X32-NOCMOV-NEXT: xorl %ebp, %ebx
+; X32-NOCMOV-NEXT: andl %edi, %ebx
+; X32-NOCMOV-NEXT: #ARITH_FENCE
+; X32-NOCMOV-NEXT: xorl %ebp, %ebx
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebp
+; X32-NOCMOV-NEXT: xorl %esi, %ebp
+; X32-NOCMOV-NEXT: andl %edi, %ebp
+; X32-NOCMOV-NEXT: #ARITH_FENCE
+; X32-NOCMOV-NEXT: xorl %esi, %ebp
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: andl %edi, %ecx
+; X32-NOCMOV-NEXT: #ARITH_FENCE
+; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: movl %ecx, 12(%eax)
+; X32-NOCMOV-NEXT: movl %ebp, 8(%eax)
+; X32-NOCMOV-NEXT: movl %ebx, 4(%eax)
+; X32-NOCMOV-NEXT: movl %edx, (%eax)
; X32-NOCMOV-NEXT: popl %esi
; X32-NOCMOV-NEXT: popl %edi
; X32-NOCMOV-NEXT: popl %ebx
@@ -1578,6 +1721,7 @@ define <4 x float> @test_ctselect_v4f32(i1 %cond, <4 x float> %a, <4 x float> %b
; X64-NEXT: movd %edi, %xmm2
; X64-NEXT: pshufd {{.*#+}} xmm2 = xmm2[0,0,0,0]
; X64-NEXT: pand %xmm2, %xmm0
+; X64-NEXT: #ARITH_FENCE
; X64-NEXT: pxor %xmm1, %xmm0
; X64-NEXT: retq
;
@@ -1616,21 +1760,25 @@ define <4 x float> @test_ctselect_v4f32(i1 %cond, <4 x float> %a, <4 x float> %b
; X32-NEXT: movl {{[0-9]+}}(%esp), %ebp
; X32-NEXT: xorl %ebx, %ebp
; X32-NEXT: andl %edx, %ebp
+; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %ebx, %ebp
; X32-NEXT: movl %ebp, {{[0-9]+}}(%esp)
; X32-NEXT: movl {{[0-9]+}}(%esp), %ebx
; X32-NEXT: xorl %edi, %ebx
; X32-NEXT: andl %edx, %ebx
+; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %edi, %ebx
; X32-NEXT: movl %ebx, {{[0-9]+}}(%esp)
; X32-NEXT: movl {{[0-9]+}}(%esp), %edi
; X32-NEXT: xorl %esi, %edi
; X32-NEXT: andl %edx, %edi
+; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %esi, %edi
; X32-NEXT: movl %edi, {{[0-9]+}}(%esp)
; X32-NEXT: movl (%esp), %esi
; X32-NEXT: xorl %ecx, %esi
; X32-NEXT: andl %edx, %esi
+; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %ecx, %esi
; X32-NEXT: movl %esi, {{[0-9]+}}(%esp)
; X32-NEXT: flds {{[0-9]+}}(%esp)
@@ -1683,21 +1831,25 @@ define <4 x float> @test_ctselect_v4f32(i1 %cond, <4 x float> %a, <4 x float> %b
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebp
; X32-NOCMOV-NEXT: xorl %ebx, %ebp
; X32-NOCMOV-NEXT: andl %edx, %ebp
+; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %ebx, %ebp
; X32-NOCMOV-NEXT: movl %ebp, {{[0-9]+}}(%esp)
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebx
; X32-NOCMOV-NEXT: xorl %edi, %ebx
; X32-NOCMOV-NEXT: andl %edx, %ebx
+; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %edi, %ebx
; X32-NOCMOV-NEXT: movl %ebx, {{[0-9]+}}(%esp)
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edi
; X32-NOCMOV-NEXT: xorl %esi, %edi
; X32-NOCMOV-NEXT: andl %edx, %edi
+; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %esi, %edi
; X32-NOCMOV-NEXT: movl %edi, {{[0-9]+}}(%esp)
; X32-NOCMOV-NEXT: movl (%esp), %esi
; X32-NOCMOV-NEXT: xorl %ecx, %esi
; X32-NOCMOV-NEXT: andl %edx, %esi
+; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %ecx, %esi
; X32-NOCMOV-NEXT: movl %esi, {{[0-9]+}}(%esp)
; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
@@ -1727,9 +1879,11 @@ define <8 x i32> @test_ctselect_v8i32_avx(i1 %cond, <8 x i32> %a, <8 x i32> %b)
; X64-NEXT: movd %edi, %xmm4
; X64-NEXT: pshufd {{.*#+}} xmm4 = xmm4[0,0,0,0]
; X64-NEXT: pand %xmm4, %xmm0
+; X64-NEXT: #ARITH_FENCE
; X64-NEXT: pxor %xmm2, %xmm0
; X64-NEXT: pxor %xmm3, %xmm1
; X64-NEXT: pand %xmm4, %xmm1
+; X64-NEXT: #ARITH_FENCE
; X64-NEXT: pxor %xmm3, %xmm1
; X64-NEXT: retq
;
@@ -1740,64 +1894,77 @@ define <8 x i32> @test_ctselect_v8i32_avx(i1 %cond, <8 x i32> %a, <8 x i32> %b)
; X32-NEXT: pushl %edi
; X32-NEXT: pushl %esi
; X32-NEXT: subl $8, %esp
-; X32-NEXT: movzbl {{[0-9]+}}(%esp), %eax
-; X32-NEXT: andb $1, %al
-; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
-; X32-NEXT: xorl {{[0-9]+}}(%esp), %edx
-; X32-NEXT: movzbl %al, %ecx
-; X32-NEXT: negl %ecx
-; X32-NEXT: andl %ecx, %edx
-; X32-NEXT: movl %edx, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
-; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
-; X32-NEXT: xorl {{[0-9]+}}(%esp), %eax
-; X32-NEXT: andl %ecx, %eax
-; X32-NEXT: movl %eax, (%esp) # 4-byte Spill
-; X32-NEXT: movl {{[0-9]+}}(%esp), %ebp
-; X32-NEXT: xorl {{[0-9]+}}(%esp), %ebp
-; X32-NEXT: andl %ecx, %ebp
+; X32-NEXT: movzbl {{[0-9]+}}(%esp), %edx
+; X32-NEXT: andb $1, %dl
; X32-NEXT: movl {{[0-9]+}}(%esp), %ebx
-; X32-NEXT: xorl {{[0-9]+}}(%esp), %ebx
-; X32-NEXT: andl %ecx, %ebx
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ebp
; X32-NEXT: movl {{[0-9]+}}(%esp), %edi
-; X32-NEXT: xorl {{[0-9]+}}(%esp), %edi
-; X32-NEXT: andl %ecx, %edi
; X32-NEXT: movl {{[0-9]+}}(%esp), %esi
-; X32-NEXT: xorl {{[0-9]+}}(%esp), %esi
-; X32-NEXT: andl %ecx, %esi
-; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
-; X32-NEXT: xorl {{[0-9]+}}(%esp), %edx
-; X32-NEXT: andl %ecx, %edx
-; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
-; X32-NEXT: xorl {{[0-9]+}}(%esp), %eax
-; X32-NEXT: andl %ecx, %eax
; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
-; X32-NEXT: xorl %ecx, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Folded Spill
-; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
-; X32-NEXT: xorl %ecx, (%esp) # 4-byte Folded Spill
-; X32-NEXT: xorl {{[0-9]+}}(%esp), %ebp
-; X32-NEXT: xorl {{[0-9]+}}(%esp), %ebx
-; X32-NEXT: xorl {{[0-9]+}}(%esp), %edi
-; X32-NEXT: xorl {{[0-9]+}}(%esp), %esi
-; X32-NEXT: xorl {{[0-9]+}}(%esp), %edx
-; X32-NEXT: xorl {{[0-9]+}}(%esp), %eax
-; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
-; X32-NEXT: movl %eax, 28(%ecx)
-; X32-NEXT: movl %edx, 24(%ecx)
-; X32-NEXT: movl %esi, 20(%ecx)
-; X32-NEXT: movl %edi, 16(%ecx)
-; X32-NEXT: movl %ebx, 12(%ecx)
-; X32-NEXT: movl %ebp, 8(%ecx)
-; X32-NEXT: movl (%esp), %eax # 4-byte Reload
-; X32-NEXT: movl %eax, 4(%ecx)
-; X32-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 4-byte Reload
-; X32-NEXT: movl %eax, (%ecx)
-; X32-NEXT: movl %ecx, %eax
-; X32-NEXT: addl $8, %esp
-; X32-NEXT: popl %esi
-; X32-NEXT: popl %edi
-; X32-NEXT: popl %ebx
-; X32-NEXT: popl %ebp
-; X32-NEXT: retl $4
+; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: xorl %ecx, %eax
+; X32-NEXT: movzbl %dl, %edx
+; X32-NEXT: negl %edx
+; X32-NEXT: andl %edx, %eax
+; X32-NEXT: #ARITH_FENCE
+; X32-NEXT: xorl %ecx, %eax
+; X32-NEXT: movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
+; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: xorl %esi, %eax
+; X32-NEXT: andl %edx, %eax
+; X32-NEXT: #ARITH_FENCE
+; X32-NEXT: xorl %esi, %eax
+; X32-NEXT: movl %eax, (%esp) # 4-byte Spill
+; X32-NEXT: movl {{[0-9]+}}(%esp), %esi
+; X32-NEXT: xorl %edi, %esi
+; X32-NEXT: andl %edx, %esi
+; X32-NEXT: #ARITH_FENCE
+; X32-NEXT: xorl %edi, %esi
+; X32-NEXT: movl {{[0-9]+}}(%esp), %edi
+; X32-NEXT: xorl %ebp, %edi
+; X32-NEXT: andl %edx, %edi
+; X32-NEXT: #ARITH_FENCE
+; X32-NEXT: xorl %ebp, %edi
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ebp
+; X32-NEXT: xorl %ebx, %ebp
+; X32-NEXT: andl %edx, %ebp
+; X32-NEXT: #ARITH_FENCE
+; X32-NEXT: xorl %ebx, %ebp
+; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ebx
+; X32-NEXT: xorl %eax, %ebx
+; X32-NEXT: andl %edx, %ebx
+; X32-NEXT: #ARITH_FENCE
+; X32-NEXT: xorl %eax, %ebx
+; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: xorl %eax, %ecx
+; X32-NEXT: andl %edx, %ecx
+; X32-NEXT: #ARITH_FENCE
+; X32-NEXT: xorl %eax, %ecx
+; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: xorl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: andl %edx, %eax
+; X32-NEXT: #ARITH_FENCE
+; X32-NEXT: xorl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NEXT: movl %eax, 28(%edx)
+; X32-NEXT: movl %ecx, 24(%edx)
+; X32-NEXT: movl %ebx, 20(%edx)
+; X32-NEXT: movl %ebp, 16(%edx)
+; X32-NEXT: movl %edi, 12(%edx)
+; X32-NEXT: movl %esi, 8(%edx)
+; X32-NEXT: movl (%esp), %eax # 4-byte Reload
+; X32-NEXT: movl %eax, 4(%edx)
+; X32-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 4-byte Reload
+; X32-NEXT: movl %eax, (%edx)
+; X32-NEXT: movl %edx, %eax
+; X32-NEXT: addl $8, %esp
+; X32-NEXT: popl %esi
+; X32-NEXT: popl %edi
+; X32-NEXT: popl %ebx
+; X32-NEXT: popl %ebp
+; X32-NEXT: retl $4
;
; X32-NOCMOV-LABEL: test_ctselect_v8i32_avx:
; X32-NOCMOV: # %bb.0:
@@ -1806,58 +1973,71 @@ define <8 x i32> @test_ctselect_v8i32_avx(i1 %cond, <8 x i32> %a, <8 x i32> %b)
; X32-NOCMOV-NEXT: pushl %edi
; X32-NOCMOV-NEXT: pushl %esi
; X32-NOCMOV-NEXT: subl $8, %esp
-; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %eax
-; X32-NOCMOV-NEXT: andb $1, %al
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
-; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %edx
-; X32-NOCMOV-NEXT: movzbl %al, %ecx
-; X32-NOCMOV-NEXT: negl %ecx
-; X32-NOCMOV-NEXT: andl %ecx, %edx
-; X32-NOCMOV-NEXT: movl %edx, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
+; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %edx
+; X32-NOCMOV-NEXT: andb $1, %dl
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebx
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebp
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edi
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %esi
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
-; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %eax
-; X32-NOCMOV-NEXT: andl %ecx, %eax
+; X32-NOCMOV-NEXT: xorl %ecx, %eax
+; X32-NOCMOV-NEXT: movzbl %dl, %edx
+; X32-NOCMOV-NEXT: negl %edx
+; X32-NOCMOV-NEXT: andl %edx, %eax
+; X32-NOCMOV-NEXT: #ARITH_FENCE
+; X32-NOCMOV-NEXT: xorl %ecx, %eax
+; X32-NOCMOV-NEXT: movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: xorl %esi, %eax
+; X32-NOCMOV-NEXT: andl %edx, %eax
+; X32-NOCMOV-NEXT: #ARITH_FENCE
+; X32-NOCMOV-NEXT: xorl %esi, %eax
; X32-NOCMOV-NEXT: movl %eax, (%esp) # 4-byte Spill
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %esi
+; X32-NOCMOV-NEXT: xorl %edi, %esi
+; X32-NOCMOV-NEXT: andl %edx, %esi
+; X32-NOCMOV-NEXT: #ARITH_FENCE
+; X32-NOCMOV-NEXT: xorl %edi, %esi
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edi
+; X32-NOCMOV-NEXT: xorl %ebp, %edi
+; X32-NOCMOV-NEXT: andl %edx, %edi
+; X32-NOCMOV-NEXT: #ARITH_FENCE
+; X32-NOCMOV-NEXT: xorl %ebp, %edi
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebp
-; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %ebp
-; X32-NOCMOV-NEXT: andl %ecx, %ebp
+; X32-NOCMOV-NEXT: xorl %ebx, %ebp
+; X32-NOCMOV-NEXT: andl %edx, %ebp
+; X32-NOCMOV-NEXT: #ARITH_FENCE
+; X32-NOCMOV-NEXT: xorl %ebx, %ebp
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebx
-; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %ebx
-; X32-NOCMOV-NEXT: andl %ecx, %ebx
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edi
-; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %edi
-; X32-NOCMOV-NEXT: andl %ecx, %edi
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %esi
-; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %esi
-; X32-NOCMOV-NEXT: andl %ecx, %esi
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
-; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %edx
-; X32-NOCMOV-NEXT: andl %ecx, %edx
+; X32-NOCMOV-NEXT: xorl %eax, %ebx
+; X32-NOCMOV-NEXT: andl %edx, %ebx
+; X32-NOCMOV-NEXT: #ARITH_FENCE
+; X32-NOCMOV-NEXT: xorl %eax, %ebx
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
-; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %eax
-; X32-NOCMOV-NEXT: andl %ecx, %eax
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
-; X32-NOCMOV-NEXT: xorl %ecx, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Folded Spill
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
-; X32-NOCMOV-NEXT: xorl %ecx, (%esp) # 4-byte Folded Spill
-; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %ebp
-; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %ebx
-; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %edi
-; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %esi
-; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %edx
+; X32-NOCMOV-NEXT: xorl %eax, %ecx
+; X32-NOCMOV-NEXT: andl %edx, %ecx
+; X32-NOCMOV-NEXT: #ARITH_FENCE
+; X32-NOCMOV-NEXT: xorl %eax, %ecx
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %eax
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
-; X32-NOCMOV-NEXT: movl %eax, 28(%ecx)
-; X32-NOCMOV-NEXT: movl %edx, 24(%ecx)
-; X32-NOCMOV-NEXT: movl %esi, 20(%ecx)
-; X32-NOCMOV-NEXT: movl %edi, 16(%ecx)
-; X32-NOCMOV-NEXT: movl %ebx, 12(%ecx)
-; X32-NOCMOV-NEXT: movl %ebp, 8(%ecx)
+; X32-NOCMOV-NEXT: andl %edx, %eax
+; X32-NOCMOV-NEXT: #ARITH_FENCE
+; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NOCMOV-NEXT: movl %eax, 28(%edx)
+; X32-NOCMOV-NEXT: movl %ecx, 24(%edx)
+; X32-NOCMOV-NEXT: movl %ebx, 20(%edx)
+; X32-NOCMOV-NEXT: movl %ebp, 16(%edx)
+; X32-NOCMOV-NEXT: movl %edi, 12(%edx)
+; X32-NOCMOV-NEXT: movl %esi, 8(%edx)
; X32-NOCMOV-NEXT: movl (%esp), %eax # 4-byte Reload
-; X32-NOCMOV-NEXT: movl %eax, 4(%ecx)
+; X32-NOCMOV-NEXT: movl %eax, 4(%edx)
; X32-NOCMOV-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 4-byte Reload
-; X32-NOCMOV-NEXT: movl %eax, (%ecx)
-; X32-NOCMOV-NEXT: movl %ecx, %eax
+; X32-NOCMOV-NEXT: movl %eax, (%edx)
+; X32-NOCMOV-NEXT: movl %edx, %eax
; X32-NOCMOV-NEXT: addl $8, %esp
; X32-NOCMOV-NEXT: popl %esi
; X32-NOCMOV-NEXT: popl %edi
@@ -1877,9 +2057,11 @@ define <8 x float> @test_ctselect_v8f32(i1 %cond, <8 x float> %a, <8 x float> %b
; X64-NEXT: movd %edi, %xmm4
; X64-NEXT: pshufd {{.*#+}} xmm4 = xmm4[0,0,0,0]
; X64-NEXT: pand %xmm4, %xmm0
+; X64-NEXT: #ARITH_FENCE
; X64-NEXT: pxor %xmm2, %xmm0
; X64-NEXT: pxor %xmm3, %xmm1
; X64-NEXT: pand %xmm4, %xmm1
+; X64-NEXT: #ARITH_FENCE
; X64-NEXT: pxor %xmm3, %xmm1
; X64-NEXT: retq
;
@@ -1936,6 +2118,7 @@ define <8 x float> @test_ctselect_v8f32(i1 %cond, <8 x float> %a, <8 x float> %b
; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
; X32-NEXT: xorl %ebx, %eax
; X32-NEXT: andl %edx, %eax
+; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %ebx, %eax
; X32-NEXT: movl {{[0-9]+}}(%esp), %ebx
; X32-NEXT: movl {{[0-9]+}}(%esp), %ebp
@@ -1944,38 +2127,45 @@ define <8 x float> @test_ctselect_v8f32(i1 %cond, <8 x float> %a, <8 x float> %b
; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
; X32-NEXT: xorl %ecx, %eax
; X32-NEXT: andl %edx, %eax
+; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %ecx, %eax
; X32-NEXT: movl %eax, {{[0-9]+}}(%esp)
; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
; X32-NEXT: xorl %ebp, %eax
; X32-NEXT: andl %edx, %eax
+; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %ebp, %eax
; X32-NEXT: movl %eax, {{[0-9]+}}(%esp)
; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
; X32-NEXT: xorl %ebx, %eax
; X32-NEXT: andl %edx, %eax
+; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %ebx, %eax
; X32-NEXT: movl %eax, {{[0-9]+}}(%esp)
; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
; X32-NEXT: xorl %edi, %eax
; X32-NEXT: andl %edx, %eax
+; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %edi, %eax
; X32-NEXT: movl %eax, {{[0-9]+}}(%esp)
; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
; X32-NEXT: xorl %esi, %eax
; X32-NEXT: andl %edx, %eax
+; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %esi, %eax
; X32-NEXT: movl %eax, {{[0-9]+}}(%esp)
; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
; X32-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %ecx # 4-byte Reload
; X32-NEXT: xorl %ecx, %eax
; X32-NEXT: andl %edx, %eax
+; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %ecx, %eax
; X32-NEXT: movl %eax, {{[0-9]+}}(%esp)
; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
; X32-NEXT: movl (%esp), %ecx # 4-byte Reload
; X32-NEXT: xorl %ecx, %eax
; X32-NEXT: andl %edx, %eax
+; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %ecx, %eax
; X32-NEXT: movl %eax, {{[0-9]+}}(%esp)
; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
@@ -2057,6 +2247,7 @@ define <8 x float> @test_ctselect_v8f32(i1 %cond, <8 x float> %a, <8 x float> %b
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
; X32-NOCMOV-NEXT: xorl %ebx, %eax
; X32-NOCMOV-NEXT: andl %edx, %eax
+; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %ebx, %eax
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebx
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebp
@@ -2065,38 +2256,45 @@ define <8 x float> @test_ctselect_v8f32(i1 %cond, <8 x float> %a, <8 x float> %b
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
; X32-NOCMOV-NEXT: xorl %ecx, %eax
; X32-NOCMOV-NEXT: andl %edx, %eax
+; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %ecx, %eax
; X32-NOCMOV-NEXT: movl %eax, {{[0-9]+}}(%esp)
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
; X32-NOCMOV-NEXT: xorl %ebp, %eax
; X32-NOCMOV-NEXT: andl %edx, %eax
+; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %ebp, %eax
; X32-NOCMOV-NEXT: movl %eax, {{[0-9]+}}(%esp)
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
; X32-NOCMOV-NEXT: xorl %ebx, %eax
; X32-NOCMOV-NEXT: andl %edx, %eax
+; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %ebx, %eax
; X32-NOCMOV-NEXT: movl %eax, {{[0-9]+}}(%esp)
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
; X32-NOCMOV-NEXT: xorl %edi, %eax
; X32-NOCMOV-NEXT: andl %edx, %eax
+; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %edi, %eax
; X32-NOCMOV-NEXT: movl %eax, {{[0-9]+}}(%esp)
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
; X32-NOCMOV-NEXT: xorl %esi, %eax
; X32-NOCMOV-NEXT: andl %edx, %eax
+; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %esi, %eax
; X32-NOCMOV-NEXT: movl %eax, {{[0-9]+}}(%esp)
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
; X32-NOCMOV-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %ecx # 4-byte Reload
; X32-NOCMOV-NEXT: xorl %ecx, %eax
; X32-NOCMOV-NEXT: andl %edx, %eax
+; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %ecx, %eax
; X32-NOCMOV-NEXT: movl %eax, {{[0-9]+}}(%esp)
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
; X32-NOCMOV-NEXT: movl (%esp), %ecx # 4-byte Reload
; X32-NOCMOV-NEXT: xorl %ecx, %eax
; X32-NOCMOV-NEXT: andl %edx, %eax
+; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %ecx, %eax
; X32-NOCMOV-NEXT: movl %eax, {{[0-9]+}}(%esp)
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
@@ -2138,6 +2336,7 @@ define <8 x half> @test_ctselect_v8f16(i1 %cond, <8 x half> %a, <8 x half> %b) #
; X64-NEXT: pshuflw {{.*#+}} xmm2 = xmm2[0,0,0,0,4,5,6,7]
; X64-NEXT: pshufd {{.*#+}} xmm2 = xmm2[0,1,0,1]
; X64-NEXT: pand %xmm2, %xmm0
+; X64-NEXT: #ARITH_FENCE
; X64-NEXT: pxor %xmm1, %xmm0
; X64-NEXT: retq
;
@@ -2148,67 +2347,74 @@ define <8 x half> @test_ctselect_v8f16(i1 %cond, <8 x half> %a, <8 x half> %b) #
; X32-NEXT: pushl %edi
; X32-NEXT: pushl %esi
; X32-NEXT: subl $12, %esp
-; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
-; X32-NEXT: movl {{[0-9]+}}(%esp), %esi
-; X32-NEXT: movzbl {{[0-9]+}}(%esp), %eax
-; X32-NEXT: andb $1, %al
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ebp
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ebx
; X32-NEXT: movl {{[0-9]+}}(%esp), %edi
-; X32-NEXT: movzwl {{[0-9]+}}(%esp), %ecx
-; X32-NEXT: xorw %di, %cx
-; X32-NEXT: movzbl %al, %eax
-; X32-NEXT: negl %eax
-; X32-NEXT: andl %eax, %ecx
-; X32-NEXT: movl %ecx, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
-; X32-NEXT: movzwl {{[0-9]+}}(%esp), %ecx
-; X32-NEXT: xorw %si, %cx
-; X32-NEXT: andl %eax, %ecx
-; X32-NEXT: movl %ecx, (%esp) # 4-byte Spill
-; X32-NEXT: movzwl {{[0-9]+}}(%esp), %ecx
-; X32-NEXT: xorw %dx, %cx
-; X32-NEXT: andl %eax, %ecx
-; X32-NEXT: movl %ecx, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
-; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: andb $1, %cl
+; X32-NEXT: movl {{[0-9]+}}(%esp), %esi
+; X32-NEXT: movzwl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: xorw %si, %ax
+; X32-NEXT: movzbl %cl, %edx
+; X32-NEXT: negl %edx
+; X32-NEXT: andl %edx, %eax
+; X32-NEXT: #ARITH_FENCE
+; X32-NEXT: xorl %esi, %eax
+; X32-NEXT: movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
+; X32-NEXT: movzwl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: xorw %di, %ax
+; X32-NEXT: andl %edx, %eax
+; X32-NEXT: #ARITH_FENCE
+; X32-NEXT: xorl %edi, %eax
+; X32-NEXT: movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
+; X32-NEXT: movzwl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: xorw %bx, %ax
+; X32-NEXT: andl %edx, %eax
+; X32-NEXT: #ARITH_FENCE
+; X32-NEXT: xorl %ebx, %eax
+; X32-NEXT: movl %eax, (%esp) # 4-byte Spill
; X32-NEXT: movzwl {{[0-9]+}}(%esp), %ebx
-; X32-NEXT: xorw %cx, %bx
-; X32-NEXT: andl %eax, %ebx
-; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: xorw %bp, %bx
+; X32-NEXT: andl %edx, %ebx
+; X32-NEXT: #ARITH_FENCE
+; X32-NEXT: xorl %ebp, %ebx
; X32-NEXT: movzwl {{[0-9]+}}(%esp), %ebp
-; X32-NEXT: xorw %cx, %bp
-; X32-NEXT: andl %eax, %ebp
-; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
-; X32-NEXT: movzwl {{[0-9]+}}(%esp), %esi
-; X32-NEXT: xorw %cx, %si
-; X32-NEXT: andl %eax, %esi
-; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
-; X32-NEXT: movzwl {{[0-9]+}}(%esp), %edx
-; X32-NEXT: xorw %cx, %dx
-; X32-NEXT: andl %eax, %edx
-; X32-NEXT: movzwl {{[0-9]+}}(%esp), %ecx
-; X32-NEXT: movl {{[0-9]+}}(%esp), %edi
-; X32-NEXT: xorw %di, %cx
-; X32-NEXT: andl %eax, %ecx
; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
-; X32-NEXT: xorl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Folded Spill
+; X32-NEXT: xorw %ax, %bp
+; X32-NEXT: andl %edx, %ebp
+; X32-NEXT: #ARITH_FENCE
+; X32-NEXT: xorl %eax, %ebp
; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
-; X32-NEXT: xorl %eax, (%esp) # 4-byte Folded Spill
-; X32-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %edi # 4-byte Reload
-; X32-NEXT: xorl {{[0-9]+}}(%esp), %edi
-; X32-NEXT: xorl {{[0-9]+}}(%esp), %ebx
-; X32-NEXT: xorl {{[0-9]+}}(%esp), %ebp
-; X32-NEXT: xorl {{[0-9]+}}(%esp), %esi
-; X32-NEXT: xorl {{[0-9]+}}(%esp), %edx
-; X32-NEXT: xorl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: movzwl {{[0-9]+}}(%esp), %edi
+; X32-NEXT: xorw %ax, %di
+; X32-NEXT: andl %edx, %edi
+; X32-NEXT: #ARITH_FENCE
+; X32-NEXT: xorl %eax, %edi
; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
-; X32-NEXT: movw %cx, 14(%eax)
-; X32-NEXT: movw %dx, 12(%eax)
-; X32-NEXT: movw %si, 10(%eax)
-; X32-NEXT: movw %bp, 8(%eax)
-; X32-NEXT: movw %bx, 6(%eax)
-; X32-NEXT: movw %di, 4(%eax)
-; X32-NEXT: movl (%esp), %ecx # 4-byte Reload
-; X32-NEXT: movw %cx, 2(%eax)
-; X32-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %ecx # 4-byte Reload
-; X32-NEXT: movw %cx, (%eax)
+; X32-NEXT: movzwl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: xorw %ax, %cx
+; X32-NEXT: andl %edx, %ecx
+; X32-NEXT: #ARITH_FENCE
+; X32-NEXT: xorl %eax, %ecx
+; X32-NEXT: movzwl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: movl {{[0-9]+}}(%esp), %esi
+; X32-NEXT: xorw %si, %ax
+; X32-NEXT: andl %edx, %eax
+; X32-NEXT: #ARITH_FENCE
+; X32-NEXT: xorl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NEXT: movw %ax, 14(%edx)
+; X32-NEXT: movw %cx, 12(%edx)
+; X32-NEXT: movw %di, 10(%edx)
+; X32-NEXT: movw %bp, 8(%edx)
+; X32-NEXT: movw %bx, 6(%edx)
+; X32-NEXT: movl (%esp), %eax # 4-byte Reload
+; X32-NEXT: movw %ax, 4(%edx)
+; X32-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 4-byte Reload
+; X32-NEXT: movw %ax, 2(%edx)
+; X32-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 4-byte Reload
+; X32-NEXT: movw %ax, (%edx)
+; X32-NEXT: movl %edx, %eax
; X32-NEXT: addl $12, %esp
; X32-NEXT: popl %esi
; X32-NEXT: popl %edi
@@ -2223,67 +2429,74 @@ define <8 x half> @test_ctselect_v8f16(i1 %cond, <8 x half> %a, <8 x half> %b) #
; X32-NOCMOV-NEXT: pushl %edi
; X32-NOCMOV-NEXT: pushl %esi
; X32-NOCMOV-NEXT: subl $12, %esp
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %esi
-; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %eax
-; X32-NOCMOV-NEXT: andb $1, %al
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebp
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebx
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edi
-; X32-NOCMOV-NEXT: movzwl {{[0-9]+}}(%esp), %ecx
-; X32-NOCMOV-NEXT: xorw %di, %cx
-; X32-NOCMOV-NEXT: movzbl %al, %eax
-; X32-NOCMOV-NEXT: negl %eax
-; X32-NOCMOV-NEXT: andl %eax, %ecx
-; X32-NOCMOV-NEXT: movl %ecx, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
-; X32-NOCMOV-NEXT: movzwl {{[0-9]+}}(%esp), %ecx
-; X32-NOCMOV-NEXT: xorw %si, %cx
-; X32-NOCMOV-NEXT: andl %eax, %ecx
-; X32-NOCMOV-NEXT: movl %ecx, (%esp) # 4-byte Spill
-; X32-NOCMOV-NEXT: movzwl {{[0-9]+}}(%esp), %ecx
-; X32-NOCMOV-NEXT: xorw %dx, %cx
-; X32-NOCMOV-NEXT: andl %eax, %ecx
-; X32-NOCMOV-NEXT: movl %ecx, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: andb $1, %cl
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %esi
+; X32-NOCMOV-NEXT: movzwl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: xorw %si, %ax
+; X32-NOCMOV-NEXT: movzbl %cl, %edx
+; X32-NOCMOV-NEXT: negl %edx
+; X32-NOCMOV-NEXT: andl %edx, %eax
+; X32-NOCMOV-NEXT: #ARITH_FENCE
+; X32-NOCMOV-NEXT: xorl %esi, %eax
+; X32-NOCMOV-NEXT: movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
+; X32-NOCMOV-NEXT: movzwl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: xorw %di, %ax
+; X32-NOCMOV-NEXT: andl %edx, %eax
+; X32-NOCMOV-NEXT: #ARITH_FENCE
+; X32-NOCMOV-NEXT: xorl %edi, %eax
+; X32-NOCMOV-NEXT: movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
+; X32-NOCMOV-NEXT: movzwl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: xorw %bx, %ax
+; X32-NOCMOV-NEXT: andl %edx, %eax
+; X32-NOCMOV-NEXT: #ARITH_FENCE
+; X32-NOCMOV-NEXT: xorl %ebx, %eax
+; X32-NOCMOV-NEXT: movl %eax, (%esp) # 4-byte Spill
; X32-NOCMOV-NEXT: movzwl {{[0-9]+}}(%esp), %ebx
-; X32-NOCMOV-NEXT: xorw %cx, %bx
-; X32-NOCMOV-NEXT: andl %eax, %ebx
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: xorw %bp, %bx
+; X32-NOCMOV-NEXT: andl %edx, %ebx
+; X32-NOCMOV-NEXT: #ARITH_FENCE
+; X32-NOCMOV-NEXT: xorl %ebp, %ebx
; X32-NOCMOV-NEXT: movzwl {{[0-9]+}}(%esp), %ebp
-; X32-NOCMOV-NEXT: xorw %cx, %bp
-; X32-NOCMOV-NEXT: andl %eax, %ebp
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
-; X32-NOCMOV-NEXT: movzwl {{[0-9]+}}(%esp), %esi
-; X32-NOCMOV-NEXT: xorw %cx, %si
-; X32-NOCMOV-NEXT: andl %eax, %esi
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
-; X32-NOCMOV-NEXT: movzwl {{[0-9]+}}(%esp), %edx
-; X32-NOCMOV-NEXT: xorw %cx, %dx
-; X32-NOCMOV-NEXT: andl %eax, %edx
-; X32-NOCMOV-NEXT: movzwl {{[0-9]+}}(%esp), %ecx
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edi
-; X32-NOCMOV-NEXT: xorw %di, %cx
-; X32-NOCMOV-NEXT: andl %eax, %ecx
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
-; X32-NOCMOV-NEXT: xorl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Folded Spill
+; X32-NOCMOV-NEXT: xorw %ax, %bp
+; X32-NOCMOV-NEXT: andl %edx, %ebp
+; X32-NOCMOV-NEXT: #ARITH_FENCE
+; X32-NOCMOV-NEXT: xorl %eax, %ebp
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
-; X32-NOCMOV-NEXT: xorl %eax, (%esp) # 4-byte Folded Spill
-; X32-NOCMOV-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %edi # 4-byte Reload
-; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %edi
-; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %ebx
-; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %ebp
-; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %esi
-; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %edx
-; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: movzwl {{[0-9]+}}(%esp), %edi
+; X32-NOCMOV-NEXT: xorw %ax, %di
+; X32-NOCMOV-NEXT: andl %edx, %edi
+; X32-NOCMOV-NEXT: #ARITH_FENCE
+; X32-NOCMOV-NEXT: xorl %eax, %edi
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
-; X32-NOCMOV-NEXT: movw %cx, 14(%eax)
-; X32-NOCMOV-NEXT: movw %dx, 12(%eax)
-; X32-NOCMOV-NEXT: movw %si, 10(%eax)
-; X32-NOCMOV-NEXT: movw %bp, 8(%eax)
-; X32-NOCMOV-NEXT: movw %bx, 6(%eax)
-; X32-NOCMOV-NEXT: movw %di, 4(%eax)
-; X32-NOCMOV-NEXT: movl (%esp), %ecx # 4-byte Reload
-; X32-NOCMOV-NEXT: movw %cx, 2(%eax)
-; X32-NOCMOV-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %ecx # 4-byte Reload
-; X32-NOCMOV-NEXT: movw %cx, (%eax)
+; X32-NOCMOV-NEXT: movzwl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: xorw %ax, %cx
+; X32-NOCMOV-NEXT: andl %edx, %ecx
+; X32-NOCMOV-NEXT: #ARITH_FENCE
+; X32-NOCMOV-NEXT: xorl %eax, %ecx
+; X32-NOCMOV-NEXT: movzwl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %esi
+; X32-NOCMOV-NEXT: xorw %si, %ax
+; X32-NOCMOV-NEXT: andl %edx, %eax
+; X32-NOCMOV-NEXT: #ARITH_FENCE
+; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
+; X32-NOCMOV-NEXT: movw %ax, 14(%edx)
+; X32-NOCMOV-NEXT: movw %cx, 12(%edx)
+; X32-NOCMOV-NEXT: movw %di, 10(%edx)
+; X32-NOCMOV-NEXT: movw %bp, 8(%edx)
+; X32-NOCMOV-NEXT: movw %bx, 6(%edx)
+; X32-NOCMOV-NEXT: movl (%esp), %eax # 4-byte Reload
+; X32-NOCMOV-NEXT: movw %ax, 4(%edx)
+; X32-NOCMOV-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 4-byte Reload
+; X32-NOCMOV-NEXT: movw %ax, 2(%edx)
+; X32-NOCMOV-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 4-byte Reload
+; X32-NOCMOV-NEXT: movw %ax, (%edx)
+; X32-NOCMOV-NEXT: movl %edx, %eax
; X32-NOCMOV-NEXT: addl $12, %esp
; X32-NOCMOV-NEXT: popl %esi
; X32-NOCMOV-NEXT: popl %edi
@@ -2312,11 +2525,13 @@ define <8 x bfloat> @test_ctselect_v8bf16(i1 %cond, <8 x bfloat> %a, <8 x bfloat
; X64-NEXT: andl $1, %edi
; X64-NEXT: negl %edi
; X64-NEXT: andl %edi, %r10d
+; X64-NEXT: #ARITH_FENCE
; X64-NEXT: xorl %r8d, %r10d
; X64-NEXT: movq %r9, %r8
+; X64-NEXT: shll $16, %r10d
; X64-NEXT: xorl %esi, %r9d
; X64-NEXT: andl %edi, %r9d
-; X64-NEXT: shll $16, %r10d
+; X64-NEXT: #ARITH_FENCE
; X64-NEXT: xorl %esi, %r9d
; X64-NEXT: movzwl %r9w, %r9d
; X64-NEXT: orl %r10d, %r9d
@@ -2325,11 +2540,13 @@ define <8 x bfloat> @test_ctselect_v8bf16(i1 %cond, <8 x bfloat> %a, <8 x bfloat
; X64-NEXT: shrq $48, %r8
; X64-NEXT: xorl %r10d, %r8d
; X64-NEXT: andl %edi, %r8d
+; X64-NEXT: #ARITH_FENCE
; X64-NEXT: xorl %r10d, %r8d
; X64-NEXT: shrq $32, %rsi
; X64-NEXT: shrq $32, %rdx
; X64-NEXT: xorl %esi, %edx
; X64-NEXT: andl %edi, %edx
+; X64-NEXT: #ARITH_FENCE
; X64-NEXT: xorl %esi, %edx
; X64-NEXT: movq %rcx, %rsi
; X64-NEXT: shll $16, %r8d
@@ -2343,24 +2560,28 @@ define <8 x bfloat> @test_ctselect_v8bf16(i1 %cond, <8 x bfloat> %a, <8 x bfloat
; X64-NEXT: shrl $16, %r9d
; X64-NEXT: xorl %r8d, %r9d
; X64-NEXT: andl %edi, %r9d
+; X64-NEXT: #ARITH_FENCE
; X64-NEXT: xorl %r8d, %r9d
; X64-NEXT: movq %rcx, %r8
+; X64-NEXT: shll $16, %r9d
; X64-NEXT: xorl %eax, %ecx
; X64-NEXT: andl %edi, %ecx
-; X64-NEXT: shll $16, %r9d
+; X64-NEXT: #ARITH_FENCE
; X64-NEXT: xorl %eax, %ecx
; X64-NEXT: movzwl %cx, %ecx
; X64-NEXT: orl %r9d, %ecx
; X64-NEXT: movq %rax, %r9
-; X64-NEXT: shrq $32, %rax
-; X64-NEXT: shrq $32, %rsi
; X64-NEXT: shrq $48, %r9
; X64-NEXT: shrq $48, %r8
; X64-NEXT: xorl %r9d, %r8d
; X64-NEXT: andl %edi, %r8d
+; X64-NEXT: #ARITH_FENCE
+; X64-NEXT: xorl %r9d, %r8d
+; X64-NEXT: shrq $32, %rax
+; X64-NEXT: shrq $32, %rsi
; X64-NEXT: xorl %eax, %esi
; X64-NEXT: andl %edi, %esi
-; X64-NEXT: xorl %r9d, %r8d
+; X64-NEXT: #ARITH_FENCE
; X64-NEXT: xorl %eax, %esi
; X64-NEXT: shll $16, %r8d
; X64-NEXT: movzwl %si, %eax
@@ -2387,7 +2608,8 @@ define <8 x bfloat> @test_ctselect_v8bf16(i1 %cond, <8 x bfloat> %a, <8 x bfloat
; X32-NEXT: flds {{[0-9]+}}(%esp)
; X32-NEXT: fstps (%esp)
; X32-NEXT: calll __truncsfbf2
-; X32-NEXT: movl %eax, %ebp
+; X32-NEXT: # kill: def $ax killed $ax def $eax
+; X32-NEXT: movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
; X32-NEXT: flds {{[0-9]+}}(%esp)
; X32-NEXT: fstps (%esp)
; X32-NEXT: calll __truncsfbf2
@@ -2426,8 +2648,7 @@ define <8 x bfloat> @test_ctselect_v8bf16(i1 %cond, <8 x bfloat> %a, <8 x bfloat
; X32-NEXT: flds {{[0-9]+}}(%esp)
; X32-NEXT: fstps (%esp)
; X32-NEXT: calll __truncsfbf2
-; X32-NEXT: # kill: def $ax killed $ax def $eax
-; X32-NEXT: movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
+; X32-NEXT: movl %eax, %ebp
; X32-NEXT: flds {{[0-9]+}}(%esp)
; X32-NEXT: fstps (%esp)
; X32-NEXT: calll __truncsfbf2
@@ -2445,7 +2666,7 @@ define <8 x bfloat> @test_ctselect_v8bf16(i1 %cond, <8 x bfloat> %a, <8 x bfloat
; X32-NEXT: flds {{[0-9]+}}(%esp)
; X32-NEXT: fstps (%esp)
; X32-NEXT: calll __truncsfbf2
-; X32-NEXT: movl %eax, %esi
+; X32-NEXT: movl %eax, %ebx
; X32-NEXT: flds {{[0-9]+}}(%esp)
; X32-NEXT: fstps (%esp)
; X32-NEXT: calll __truncsfbf2
@@ -2454,63 +2675,70 @@ define <8 x bfloat> @test_ctselect_v8bf16(i1 %cond, <8 x bfloat> %a, <8 x bfloat
; X32-NEXT: flds {{[0-9]+}}(%esp)
; X32-NEXT: fstps (%esp)
; X32-NEXT: calll __truncsfbf2
-; X32-NEXT: # kill: def $ax killed $ax def $eax
+; X32-NEXT: movl %eax, %esi
; X32-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
; X32-NEXT: andb $1, %cl
-; X32-NEXT: xorl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 4-byte Folded Reload
+; X32-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %edx # 4-byte Reload
+; X32-NEXT: xorl %edx, %esi
; X32-NEXT: movzbl %cl, %ecx
; X32-NEXT: negl %ecx
-; X32-NEXT: andl %ecx, %eax
-; X32-NEXT: movl %eax, %ebx
-; X32-NEXT: movl %ebp, %eax
-; X32-NEXT: xorl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 4-byte Folded Reload
-; X32-NEXT: andl %ecx, %eax
-; X32-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %ebp # 4-byte Reload
-; X32-NEXT: xorl {{[-0-9]+}}(%e{{[sb]}}p), %ebp # 4-byte Folded Reload
-; X32-NEXT: andl %ecx, %ebp
+; X32-NEXT: andl %ecx, %esi
+; X32-NEXT: #ARITH_FENCE
+; X32-NEXT: xorl %edx, %esi
; X32-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %edx # 4-byte Reload
-; X32-NEXT: xorl {{[-0-9]+}}(%e{{[sb]}}p), %edx # 4-byte Folded Reload
+; X32-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 4-byte Reload
+; X32-NEXT: xorl %eax, %edx
; X32-NEXT: andl %ecx, %edx
+; X32-NEXT: #ARITH_FENCE
+; X32-NEXT: xorl %eax, %edx
; X32-NEXT: movl %edx, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
; X32-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %edx # 4-byte Reload
-; X32-NEXT: xorl {{[-0-9]+}}(%e{{[sb]}}p), %edx # 4-byte Folded Reload
+; X32-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 4-byte Reload
+; X32-NEXT: xorl %eax, %edx
; X32-NEXT: andl %ecx, %edx
+; X32-NEXT: #ARITH_FENCE
+; X32-NEXT: xorl %eax, %edx
; X32-NEXT: movl %edx, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
; X32-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %edx # 4-byte Reload
-; X32-NEXT: xorl {{[-0-9]+}}(%e{{[sb]}}p), %edx # 4-byte Folded Reload
+; X32-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 4-byte Reload
+; X32-NEXT: xorl %eax, %edx
; X32-NEXT: andl %ecx, %edx
-; X32-NEXT: xorl {{[-0-9]+}}(%e{{[sb]}}p), %edi # 4-byte Folded Reload
-; X32-NEXT: andl %ecx, %edi
-; X32-NEXT: movl %edi, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
-; X32-NEXT: xorl {{[-0-9]+}}(%e{{[sb]}}p), %esi # 4-byte Folded Reload
-; X32-NEXT: andl %ecx, %esi
-; X32-NEXT: movl %esi, %ecx
-; X32-NEXT: xorl {{[-0-9]+}}(%e{{[sb]}}p), %ebx # 4-byte Folded Reload
-; X32-NEXT: movl %ebx, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
-; X32-NEXT: xorl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 4-byte Folded Reload
-; X32-NEXT: movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
-; X32-NEXT: xorl {{[-0-9]+}}(%e{{[sb]}}p), %ebp # 4-byte Folded Reload
-; X32-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %edi # 4-byte Reload
-; X32-NEXT: xorl {{[-0-9]+}}(%e{{[sb]}}p), %edi # 4-byte Folded Reload
-; X32-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %esi # 4-byte Reload
-; X32-NEXT: xorl {{[-0-9]+}}(%e{{[sb]}}p), %esi # 4-byte Folded Reload
-; X32-NEXT: xorl {{[-0-9]+}}(%e{{[sb]}}p), %edx # 4-byte Folded Reload
+; X32-NEXT: #ARITH_FENCE
+; X32-NEXT: xorl %eax, %edx
+; X32-NEXT: movl %edx, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
; X32-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 4-byte Reload
-; X32-NEXT: xorl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 4-byte Folded Reload
-; X32-NEXT: xorl {{[-0-9]+}}(%e{{[sb]}}p), %ecx # 4-byte Folded Reload
-; X32-NEXT: movl %ecx, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
+; X32-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %edx # 4-byte Reload
+; X32-NEXT: xorl %edx, %eax
+; X32-NEXT: andl %ecx, %eax
+; X32-NEXT: #ARITH_FENCE
+; X32-NEXT: xorl %edx, %eax
+; X32-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %edx # 4-byte Reload
+; X32-NEXT: xorl %edx, %ebp
+; X32-NEXT: andl %ecx, %ebp
+; X32-NEXT: #ARITH_FENCE
+; X32-NEXT: xorl %edx, %ebp
+; X32-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %edx # 4-byte Reload
+; X32-NEXT: xorl %edx, %edi
+; X32-NEXT: andl %ecx, %edi
+; X32-NEXT: #ARITH_FENCE
+; X32-NEXT: xorl %edx, %edi
+; X32-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %edx # 4-byte Reload
+; X32-NEXT: xorl %edx, %ebx
+; X32-NEXT: andl %ecx, %ebx
+; X32-NEXT: #ARITH_FENCE
+; X32-NEXT: xorl %edx, %ebx
; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
-; X32-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %ebx # 4-byte Reload
; X32-NEXT: movw %bx, 14(%ecx)
-; X32-NEXT: movw %ax, 12(%ecx)
-; X32-NEXT: movw %dx, 10(%ecx)
-; X32-NEXT: movw %si, 8(%ecx)
-; X32-NEXT: movw %di, 6(%ecx)
-; X32-NEXT: movw %bp, 4(%ecx)
+; X32-NEXT: movw %di, 12(%ecx)
+; X32-NEXT: movw %bp, 10(%ecx)
+; X32-NEXT: movw %ax, 8(%ecx)
; X32-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 4-byte Reload
-; X32-NEXT: movw %ax, 2(%ecx)
+; X32-NEXT: movw %ax, 6(%ecx)
; X32-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 4-byte Reload
-; X32-NEXT: movw %ax, (%ecx)
+; X32-NEXT: movw %ax, 4(%ecx)
+; X32-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 4-byte Reload
+; X32-NEXT: movw %ax, 2(%ecx)
+; X32-NEXT: movw %si, (%ecx)
; X32-NEXT: movl %ecx, %eax
; X32-NEXT: addl $60, %esp
; X32-NEXT: popl %esi
@@ -2534,7 +2762,8 @@ define <8 x bfloat> @test_ctselect_v8bf16(i1 %cond, <8 x bfloat> %a, <8 x bfloat
; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
; X32-NOCMOV-NEXT: fstps (%esp)
; X32-NOCMOV-NEXT: calll __truncsfbf2
-; X32-NOCMOV-NEXT: movl %eax, %ebp
+; X32-NOCMOV-NEXT: # kill: def $ax killed $ax def $eax
+; X32-NOCMOV-NEXT: movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
; X32-NOCMOV-NEXT: fstps (%esp)
; X32-NOCMOV-NEXT: calll __truncsfbf2
@@ -2573,8 +2802,7 @@ define <8 x bfloat> @test_ctselect_v8bf16(i1 %cond, <8 x bfloat> %a, <8 x bfloat
; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
; X32-NOCMOV-NEXT: fstps (%esp)
; X32-NOCMOV-NEXT: calll __truncsfbf2
-; X32-NOCMOV-NEXT: # kill: def $ax killed $ax def $eax
-; X32-NOCMOV-NEXT: movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
+; X32-NOCMOV-NEXT: movl %eax, %ebp
; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
; X32-NOCMOV-NEXT: fstps (%esp)
; X32-NOCMOV-NEXT: calll __truncsfbf2
@@ -2592,7 +2820,7 @@ define <8 x bfloat> @test_ctselect_v8bf16(i1 %cond, <8 x bfloat> %a, <8 x bfloat
; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
; X32-NOCMOV-NEXT: fstps (%esp)
; X32-NOCMOV-NEXT: calll __truncsfbf2
-; X32-NOCMOV-NEXT: movl %eax, %esi
+; X32-NOCMOV-NEXT: movl %eax, %ebx
; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
; X32-NOCMOV-NEXT: fstps (%esp)
; X32-NOCMOV-NEXT: calll __truncsfbf2
@@ -2601,63 +2829,70 @@ define <8 x bfloat> @test_ctselect_v8bf16(i1 %cond, <8 x bfloat> %a, <8 x bfloat
; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
; X32-NOCMOV-NEXT: fstps (%esp)
; X32-NOCMOV-NEXT: calll __truncsfbf2
-; X32-NOCMOV-NEXT: # kill: def $ax killed $ax def $eax
+; X32-NOCMOV-NEXT: movl %eax, %esi
; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
; X32-NOCMOV-NEXT: andb $1, %cl
-; X32-NOCMOV-NEXT: xorl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 4-byte Folded Reload
+; X32-NOCMOV-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %edx # 4-byte Reload
+; X32-NOCMOV-NEXT: xorl %edx, %esi
; X32-NOCMOV-NEXT: movzbl %cl, %ecx
; X32-NOCMOV-NEXT: negl %ecx
-; X32-NOCMOV-NEXT: andl %ecx, %eax
-; X32-NOCMOV-NEXT: movl %eax, %ebx
-; X32-NOCMOV-NEXT: movl %ebp, %eax
-; X32-NOCMOV-NEXT: xorl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 4-byte Folded Reload
-; X32-NOCMOV-NEXT: andl %ecx, %eax
-; X32-NOCMOV-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %ebp # 4-byte Reload
-; X32-NOCMOV-NEXT: xorl {{[-0-9]+}}(%e{{[sb]}}p), %ebp # 4-byte Folded Reload
-; X32-NOCMOV-NEXT: andl %ecx, %ebp
+; X32-NOCMOV-NEXT: andl %ecx, %esi
+; X32-NOCMOV-NEXT: #ARITH_FENCE
+; X32-NOCMOV-NEXT: xorl %edx, %esi
; X32-NOCMOV-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %edx # 4-byte Reload
-; X32-NOCMOV-NEXT: xorl {{[-0-9]+}}(%e{{[sb]}}p), %edx # 4-byte Folded Reload
+; X32-NOCMOV-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 4-byte Reload
+; X32-NOCMOV-NEXT: xorl %eax, %edx
; X32-NOCMOV-NEXT: andl %ecx, %edx
+; X32-NOCMOV-NEXT: #ARITH_FENCE
+; X32-NOCMOV-NEXT: xorl %eax, %edx
; X32-NOCMOV-NEXT: movl %edx, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
; X32-NOCMOV-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %edx # 4-byte Reload
-; X32-NOCMOV-NEXT: xorl {{[-0-9]+}}(%e{{[sb]}}p), %edx # 4-byte Folded Reload
+; X32-NOCMOV-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 4-byte Reload
+; X32-NOCMOV-NEXT: xorl %eax, %edx
; X32-NOCMOV-NEXT: andl %ecx, %edx
+; X32-NOCMOV-NEXT: #ARITH_FENCE
+; X32-NOCMOV-NEXT: xorl %eax, %edx
; X32-NOCMOV-NEXT: movl %edx, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
; X32-NOCMOV-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %edx # 4-byte Reload
-; X32-NOCMOV-NEXT: xorl {{[-0-9]+}}(%e{{[sb]}}p), %edx # 4-byte Folded Reload
+; X32-NOCMOV-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 4-byte Reload
+; X32-NOCMOV-NEXT: xorl %eax, %edx
; X32-NOCMOV-NEXT: andl %ecx, %edx
-; X32-NOCMOV-NEXT: xorl {{[-0-9]+}}(%e{{[sb]}}p), %edi # 4-byte Folded Reload
-; X32-NOCMOV-NEXT: andl %ecx, %edi
-; X32-NOCMOV-NEXT: movl %edi, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
-; X32-NOCMOV-NEXT: xorl {{[-0-9]+}}(%e{{[sb]}}p), %esi # 4-byte Folded Reload
-; X32-NOCMOV-NEXT: andl %ecx, %esi
-; X32-NOCMOV-NEXT: movl %esi, %ecx
-; X32-NOCMOV-NEXT: xorl {{[-0-9]+}}(%e{{[sb]}}p), %ebx # 4-byte Folded Reload
-; X32-NOCMOV-NEXT: movl %ebx, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
-; X32-NOCMOV-NEXT: xorl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 4-byte Folded Reload
-; X32-NOCMOV-NEXT: movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
-; X32-NOCMOV-NEXT: xorl {{[-0-9]+}}(%e{{[sb]}}p), %ebp # 4-byte Folded Reload
-; X32-NOCMOV-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %edi # 4-byte Reload
-; X32-NOCMOV-NEXT: xorl {{[-0-9]+}}(%e{{[sb]}}p), %edi # 4-byte Folded Reload
-; X32-NOCMOV-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %esi # 4-byte Reload
-; X32-NOCMOV-NEXT: xorl {{[-0-9]+}}(%e{{[sb]}}p), %esi # 4-byte Folded Reload
-; X32-NOCMOV-NEXT: xorl {{[-0-9]+}}(%e{{[sb]}}p), %edx # 4-byte Folded Reload
+; X32-NOCMOV-NEXT: #ARITH_FENCE
+; X32-NOCMOV-NEXT: xorl %eax, %edx
+; X32-NOCMOV-NEXT: movl %edx, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
; X32-NOCMOV-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 4-byte Reload
-; X32-NOCMOV-NEXT: xorl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 4-byte Folded Reload
-; X32-NOCMOV-NEXT: xorl {{[-0-9]+}}(%e{{[sb]}}p), %ecx # 4-byte Folded Reload
-; X32-NOCMOV-NEXT: movl %ecx, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
+; X32-NOCMOV-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %edx # 4-byte Reload
+; X32-NOCMOV-NEXT: xorl %edx, %eax
+; X32-NOCMOV-NEXT: andl %ecx, %eax
+; X32-NOCMOV-NEXT: #ARITH_FENCE
+; X32-NOCMOV-NEXT: xorl %edx, %eax
+; X32-NOCMOV-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %edx # 4-byte Reload
+; X32-NOCMOV-NEXT: xorl %edx, %ebp
+; X32-NOCMOV-NEXT: andl %ecx, %ebp
+; X32-NOCMOV-NEXT: #ARITH_FENCE
+; X32-NOCMOV-NEXT: xorl %edx, %ebp
+; X32-NOCMOV-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %edx # 4-byte Reload
+; X32-NOCMOV-NEXT: xorl %edx, %edi
+; X32-NOCMOV-NEXT: andl %ecx, %edi
+; X32-NOCMOV-NEXT: #ARITH_FENCE
+; X32-NOCMOV-NEXT: xorl %edx, %edi
+; X32-NOCMOV-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %edx # 4-byte Reload
+; X32-NOCMOV-NEXT: xorl %edx, %ebx
+; X32-NOCMOV-NEXT: andl %ecx, %ebx
+; X32-NOCMOV-NEXT: #ARITH_FENCE
+; X32-NOCMOV-NEXT: xorl %edx, %ebx
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
-; X32-NOCMOV-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %ebx # 4-byte Reload
; X32-NOCMOV-NEXT: movw %bx, 14(%ecx)
-; X32-NOCMOV-NEXT: movw %ax, 12(%ecx)
-; X32-NOCMOV-NEXT: movw %dx, 10(%ecx)
-; X32-NOCMOV-NEXT: movw %si, 8(%ecx)
-; X32-NOCMOV-NEXT: movw %di, 6(%ecx)
-; X32-NOCMOV-NEXT: movw %bp, 4(%ecx)
+; X32-NOCMOV-NEXT: movw %di, 12(%ecx)
+; X32-NOCMOV-NEXT: movw %bp, 10(%ecx)
+; X32-NOCMOV-NEXT: movw %ax, 8(%ecx)
; X32-NOCMOV-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 4-byte Reload
-; X32-NOCMOV-NEXT: movw %ax, 2(%ecx)
+; X32-NOCMOV-NEXT: movw %ax, 6(%ecx)
; X32-NOCMOV-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 4-byte Reload
-; X32-NOCMOV-NEXT: movw %ax, (%ecx)
+; X32-NOCMOV-NEXT: movw %ax, 4(%ecx)
+; X32-NOCMOV-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 4-byte Reload
+; X32-NOCMOV-NEXT: movw %ax, 2(%ecx)
+; X32-NOCMOV-NEXT: movw %si, (%ecx)
; X32-NOCMOV-NEXT: movl %ecx, %eax
; X32-NOCMOV-NEXT: addl $60, %esp
; X32-NOCMOV-NEXT: popl %esi
@@ -2675,6 +2910,7 @@ define float @test_ctselect_f32_nan_inf(i1 %cond) #0 {
; X64-NEXT: andl $1, %edi
; X64-NEXT: negl %edi
; X64-NEXT: andl $4194304, %edi # imm = 0x400000
+; X64-NEXT: #ARITH_FENCE
; X64-NEXT: xorl $2139095040, %edi # imm = 0x7F800000
; X64-NEXT: movd %edi, %xmm0
; X64-NEXT: retq
@@ -2687,6 +2923,7 @@ define float @test_ctselect_f32_nan_inf(i1 %cond) #0 {
; X32-NEXT: movzbl %al, %eax
; X32-NEXT: negl %eax
; X32-NEXT: andl $4194304, %eax # imm = 0x400000
+; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl $2139095040, %eax # imm = 0x7F800000
; X32-NEXT: movl %eax, (%esp)
; X32-NEXT: flds (%esp)
@@ -2701,6 +2938,7 @@ define float @test_ctselect_f32_nan_inf(i1 %cond) #0 {
; X32-NOCMOV-NEXT: movzbl %al, %eax
; X32-NOCMOV-NEXT: negl %eax
; X32-NOCMOV-NEXT: andl $4194304, %eax # imm = 0x400000
+; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl $2139095040, %eax # imm = 0x7F800000
; X32-NOCMOV-NEXT: movl %eax, (%esp)
; X32-NOCMOV-NEXT: flds (%esp)
@@ -2718,6 +2956,7 @@ define double @test_ctselect_f64_nan_inf(i1 %cond) #0 {
; X64-NEXT: negq %rdi
; X64-NEXT: movabsq $2251799813685248, %rax # imm = 0x8000000000000
; X64-NEXT: andq %rdi, %rax
+; X64-NEXT: #ARITH_FENCE
; X64-NEXT: movabsq $9218868437227405312, %rcx # imm = 0x7FF0000000000000
; X64-NEXT: xorq %rax, %rcx
; X64-NEXT: movq %rcx, %xmm0
@@ -2740,11 +2979,13 @@ define double @test_ctselect_f64_nan_inf(i1 %cond) #0 {
; X32-NEXT: movl (%esp), %esi
; X32-NEXT: xorl %edx, %esi
; X32-NEXT: andl %eax, %esi
+; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %edx, %esi
; X32-NEXT: movl %esi, (%esp)
; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
; X32-NEXT: xorl %ecx, %edx
; X32-NEXT: andl %eax, %edx
+; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %ecx, %edx
; X32-NEXT: movl %edx, {{[0-9]+}}(%esp)
; X32-NEXT: fldl (%esp)
@@ -2769,11 +3010,13 @@ define double @test_ctselect_f64_nan_inf(i1 %cond) #0 {
; X32-NOCMOV-NEXT: movl (%esp), %esi
; X32-NOCMOV-NEXT: xorl %edx, %esi
; X32-NOCMOV-NEXT: andl %eax, %esi
+; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %edx, %esi
; X32-NOCMOV-NEXT: movl %esi, (%esp)
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
; X32-NOCMOV-NEXT: xorl %ecx, %edx
; X32-NOCMOV-NEXT: andl %eax, %edx
+; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %ecx, %edx
; X32-NOCMOV-NEXT: movl %edx, {{[0-9]+}}(%esp)
; X32-NOCMOV-NEXT: fldl (%esp)
>From 98ed2f367b27763682404330bd361c11a47a11c3 Mon Sep 17 00:00:00 2001
From: AkshayK <iit.akshay at gmail.com>
Date: Wed, 30 Sep 2026 10:42:26 -0400
Subject: [PATCH 12/12] [ConstantTime] Drop the combine barrier from CT_SELECT
expansion
ARITH_FENCE is only specified for FP types, and the earlier vreg copy
had the legalizer picking register classes. No in-tree combine folds
the XOR/AND/XOR sequence back into a select, so emit it without a
barrier. A dedicated barrier op will come in a follow-up.
---
llvm/lib/CodeGen/SelectionDAG/LegalizeDAG.cpp | 9 +-
.../CodeGen/RISCV/ctselect-fallback-gisel.ll | 2 -
.../RISCV/ctselect-fallback-scalable.ll | 8 -
llvm/test/CodeGen/RISCV/ctselect-fallback.ll | 378 +++++-------
llvm/test/CodeGen/X86/ctselect.ll | 571 +++++-------------
5 files changed, 300 insertions(+), 668 deletions(-)
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeDAG.cpp b/llvm/lib/CodeGen/SelectionDAG/LegalizeDAG.cpp
index 513b0aaf49fbe..1a0ebeb54fa31 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeDAG.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeDAG.cpp
@@ -3221,9 +3221,8 @@ SDValue SelectionDAGLegalize::ExpandCTSELECT(SDNode *Node) {
// on the same-size integer; vectors build the mask as a scalar then splat
// (avoids illegal vNi1).
//
- // The masked-diff passes through an ARITH_FENCE so no future combine can
- // fold the XOR/AND/XOR sequence back into a SELECT. No in-tree combine does
- // that today; the fence is defense-in-depth and emits no code.
+ // No in-tree combine folds the XOR/AND/XOR sequence back into a SELECT.
+ // A dedicated combine barrier is left to a follow-up.
SDValue Cond = Node->getOperand(0);
SDValue T = Node->getOperand(1);
SDValue F = Node->getOperand(2);
@@ -3362,10 +3361,6 @@ SDValue SelectionDAGLegalize::ExpandCTSELECT(SDNode *Node) {
// F ^ ((T ^ F) & Mask)
SDValue XorTF = DAG.getNode(ISD::XOR, dl, WorkingVT, WorkingT, WorkingF);
SDValue TM = DAG.getNode(ISD::AND, dl, WorkingVT, XorTF, Mask);
-
- // DAGCombine barrier (see above).
- TM = DAG.getNode(ISD::ARITH_FENCE, dl, WorkingVT, TM);
-
SDValue Res = DAG.getNode(ISD::XOR, dl, WorkingVT, WorkingF, TM);
if (WorkingVT != VT)
diff --git a/llvm/test/CodeGen/RISCV/ctselect-fallback-gisel.ll b/llvm/test/CodeGen/RISCV/ctselect-fallback-gisel.ll
index 741de6444add3..756d1154b5ef8 100644
--- a/llvm/test/CodeGen/RISCV/ctselect-fallback-gisel.ll
+++ b/llvm/test/CodeGen/RISCV/ctselect-fallback-gisel.ll
@@ -15,7 +15,6 @@ define i32 @ctsel_fallback_i32(i1 %c, i32 %x, i32 %y) {
; CHECK-NEXT: slli a0, a0, 63
; CHECK-NEXT: srai a0, a0, 63
; CHECK-NEXT: and a0, a1, a0
-; CHECK-NEXT: #ARITH_FENCE
; CHECK-NEXT: xor a0, a2, a0
; CHECK-NEXT: ret
%r = call i32 @llvm.ct.select.i32(i1 %c, i32 %x, i32 %y)
@@ -29,7 +28,6 @@ define i64 @ctsel_fallback_i64(i1 %c, i64 %x, i64 %y) {
; CHECK-NEXT: slli a0, a0, 63
; CHECK-NEXT: srai a0, a0, 63
; CHECK-NEXT: and a0, a1, a0
-; CHECK-NEXT: #ARITH_FENCE
; CHECK-NEXT: xor a0, a2, a0
; CHECK-NEXT: ret
%r = call i64 @llvm.ct.select.i64(i1 %c, i64 %x, i64 %y)
diff --git a/llvm/test/CodeGen/RISCV/ctselect-fallback-scalable.ll b/llvm/test/CodeGen/RISCV/ctselect-fallback-scalable.ll
index d4aabab9935c7..d721ab1d76a40 100644
--- a/llvm/test/CodeGen/RISCV/ctselect-fallback-scalable.ll
+++ b/llvm/test/CodeGen/RISCV/ctselect-fallback-scalable.ll
@@ -20,7 +20,6 @@ define <vscale x 4 x i32> @ctsel_nxv4i32(i1 %c, <vscale x 4 x i32> %x, <vscale x
; RV64V-NEXT: slli a0, a0, 63
; RV64V-NEXT: srai a0, a0, 63
; RV64V-NEXT: vand.vx v8, v8, a0
-; RV64V-NEXT: #ARITH_FENCE
; RV64V-NEXT: vxor.vv v8, v10, v8
; RV64V-NEXT: ret
;
@@ -31,7 +30,6 @@ define <vscale x 4 x i32> @ctsel_nxv4i32(i1 %c, <vscale x 4 x i32> %x, <vscale x
; RV32V-NEXT: slli a0, a0, 31
; RV32V-NEXT: srai a0, a0, 31
; RV32V-NEXT: vand.vx v8, v8, a0
-; RV32V-NEXT: #ARITH_FENCE
; RV32V-NEXT: vxor.vv v8, v10, v8
; RV32V-NEXT: ret
%r = call <vscale x 4 x i32> @llvm.ct.select.nxv4i32(i1 %c, <vscale x 4 x i32> %x, <vscale x 4 x i32> %y)
@@ -46,7 +44,6 @@ define <vscale x 2 x i64> @ctsel_nxv2i64(i1 %c, <vscale x 2 x i64> %x, <vscale x
; RV64V-NEXT: slli a0, a0, 63
; RV64V-NEXT: srai a0, a0, 63
; RV64V-NEXT: vand.vx v8, v8, a0
-; RV64V-NEXT: #ARITH_FENCE
; RV64V-NEXT: vxor.vv v8, v10, v8
; RV64V-NEXT: ret
;
@@ -61,7 +58,6 @@ define <vscale x 2 x i64> @ctsel_nxv2i64(i1 %c, <vscale x 2 x i64> %x, <vscale x
; RV32V-NEXT: vsetvli zero, zero, e64, m2, ta, ma
; RV32V-NEXT: vsext.vf2 v12, v14
; RV32V-NEXT: vand.vv v8, v8, v12
-; RV32V-NEXT: #ARITH_FENCE
; RV32V-NEXT: vxor.vv v8, v10, v8
; RV32V-NEXT: ret
%r = call <vscale x 2 x i64> @llvm.ct.select.nxv2i64(i1 %c, <vscale x 2 x i64> %x, <vscale x 2 x i64> %y)
@@ -76,7 +72,6 @@ define <vscale x 4 x float> @ctsel_nxv4f32(i1 %c, <vscale x 4 x float> %x, <vsca
; RV64V-NEXT: slli a0, a0, 63
; RV64V-NEXT: srai a0, a0, 63
; RV64V-NEXT: vand.vx v8, v8, a0
-; RV64V-NEXT: #ARITH_FENCE
; RV64V-NEXT: vxor.vv v8, v10, v8
; RV64V-NEXT: ret
;
@@ -87,7 +82,6 @@ define <vscale x 4 x float> @ctsel_nxv4f32(i1 %c, <vscale x 4 x float> %x, <vsca
; RV32V-NEXT: slli a0, a0, 31
; RV32V-NEXT: srai a0, a0, 31
; RV32V-NEXT: vand.vx v8, v8, a0
-; RV32V-NEXT: #ARITH_FENCE
; RV32V-NEXT: vxor.vv v8, v10, v8
; RV32V-NEXT: ret
%r = call <vscale x 4 x float> @llvm.ct.select.nxv4f32(i1 %c, <vscale x 4 x float> %x, <vscale x 4 x float> %y)
@@ -102,7 +96,6 @@ define <vscale x 2 x double> @ctsel_nxv2f64(i1 %c, <vscale x 2 x double> %x, <vs
; RV64V-NEXT: slli a0, a0, 63
; RV64V-NEXT: srai a0, a0, 63
; RV64V-NEXT: vand.vx v8, v8, a0
-; RV64V-NEXT: #ARITH_FENCE
; RV64V-NEXT: vxor.vv v8, v10, v8
; RV64V-NEXT: ret
;
@@ -117,7 +110,6 @@ define <vscale x 2 x double> @ctsel_nxv2f64(i1 %c, <vscale x 2 x double> %x, <vs
; RV32V-NEXT: vsetvli zero, zero, e64, m2, ta, ma
; RV32V-NEXT: vsext.vf2 v12, v14
; RV32V-NEXT: vand.vv v8, v8, v12
-; RV32V-NEXT: #ARITH_FENCE
; RV32V-NEXT: vxor.vv v8, v10, v8
; RV32V-NEXT: ret
%r = call <vscale x 2 x double> @llvm.ct.select.nxv2f64(i1 %c, <vscale x 2 x double> %x, <vscale x 2 x double> %y)
diff --git a/llvm/test/CodeGen/RISCV/ctselect-fallback.ll b/llvm/test/CodeGen/RISCV/ctselect-fallback.ll
index 6ad661cfad5a8..437ec6f82c2db 100644
--- a/llvm/test/CodeGen/RISCV/ctselect-fallback.ll
+++ b/llvm/test/CodeGen/RISCV/ctselect-fallback.ll
@@ -10,7 +10,6 @@ define i8 @test_ctselect_i8(i1 %cond, i8 %a, i8 %b) {
; RV64-NEXT: slli a0, a0, 63
; RV64-NEXT: srai a0, a0, 63
; RV64-NEXT: and a0, a1, a0
-; RV64-NEXT: #ARITH_FENCE
; RV64-NEXT: xor a0, a2, a0
; RV64-NEXT: ret
;
@@ -20,7 +19,6 @@ define i8 @test_ctselect_i8(i1 %cond, i8 %a, i8 %b) {
; RV32-NEXT: slli a0, a0, 31
; RV32-NEXT: srai a0, a0, 31
; RV32-NEXT: and a0, a1, a0
-; RV32-NEXT: #ARITH_FENCE
; RV32-NEXT: xor a0, a2, a0
; RV32-NEXT: ret
%result = call i8 @llvm.ct.select.i8(i1 %cond, i8 %a, i8 %b)
@@ -33,7 +31,6 @@ define i32 @test_ctselect_i32(i1 %cond, i32 %a, i32 %b) {
; RV64-NEXT: slli a0, a0, 63
; RV64-NEXT: srai a0, a0, 63
; RV64-NEXT: and a0, a1, a0
-; RV64-NEXT: #ARITH_FENCE
; RV64-NEXT: xor a0, a2, a0
; RV64-NEXT: ret
;
@@ -43,7 +40,6 @@ define i32 @test_ctselect_i32(i1 %cond, i32 %a, i32 %b) {
; RV32-NEXT: slli a0, a0, 31
; RV32-NEXT: srai a0, a0, 31
; RV32-NEXT: and a0, a1, a0
-; RV32-NEXT: #ARITH_FENCE
; RV32-NEXT: xor a0, a2, a0
; RV32-NEXT: ret
%result = call i32 @llvm.ct.select.i32(i1 %cond, i32 %a, i32 %b)
@@ -57,7 +53,6 @@ define i64 @test_ctselect_i64(i1 %cond, i64 %a, i64 %b) {
; RV64-NEXT: slli a0, a0, 63
; RV64-NEXT: srai a0, a0, 63
; RV64-NEXT: and a0, a1, a0
-; RV64-NEXT: #ARITH_FENCE
; RV64-NEXT: xor a0, a2, a0
; RV64-NEXT: ret
;
@@ -66,12 +61,10 @@ define i64 @test_ctselect_i64(i1 %cond, i64 %a, i64 %b) {
; RV32-NEXT: slli a0, a0, 31
; RV32-NEXT: xor a1, a1, a3
; RV32-NEXT: srai a0, a0, 31
-; RV32-NEXT: and a1, a1, a0
; RV32-NEXT: xor a2, a2, a4
-; RV32-NEXT: #ARITH_FENCE
+; RV32-NEXT: and a1, a1, a0
; RV32-NEXT: and a2, a2, a0
; RV32-NEXT: xor a0, a3, a1
-; RV32-NEXT: #ARITH_FENCE
; RV32-NEXT: xor a1, a4, a2
; RV32-NEXT: ret
%result = call i64 @llvm.ct.select.i64(i1 %cond, i64 %a, i64 %b)
@@ -85,7 +78,6 @@ define ptr @test_ctselect_ptr(i1 %cond, ptr %a, ptr %b) {
; RV64-NEXT: slli a0, a0, 63
; RV64-NEXT: srai a0, a0, 63
; RV64-NEXT: and a0, a1, a0
-; RV64-NEXT: #ARITH_FENCE
; RV64-NEXT: xor a0, a2, a0
; RV64-NEXT: ret
;
@@ -95,7 +87,6 @@ define ptr @test_ctselect_ptr(i1 %cond, ptr %a, ptr %b) {
; RV32-NEXT: slli a0, a0, 31
; RV32-NEXT: srai a0, a0, 31
; RV32-NEXT: and a0, a1, a0
-; RV32-NEXT: #ARITH_FENCE
; RV32-NEXT: xor a0, a2, a0
; RV32-NEXT: ret
%result = call ptr @llvm.ct.select.p0(i1 %cond, ptr %a, ptr %b)
@@ -106,16 +97,10 @@ define ptr @test_ctselect_ptr(i1 %cond, ptr %a, ptr %b) {
define i32 @test_ctselect_const_true(i32 %a, i32 %b) {
; RV64-LABEL: test_ctselect_const_true:
; RV64: # %bb.0:
-; RV64-NEXT: xor a0, a0, a1
-; RV64-NEXT: #ARITH_FENCE
-; RV64-NEXT: xor a0, a1, a0
; RV64-NEXT: ret
;
; RV32-LABEL: test_ctselect_const_true:
; RV32: # %bb.0:
-; RV32-NEXT: xor a0, a0, a1
-; RV32-NEXT: #ARITH_FENCE
-; RV32-NEXT: xor a0, a1, a0
; RV32-NEXT: ret
%result = call i32 @llvm.ct.select.i32(i1 true, i32 %a, i32 %b)
ret i32 %result
@@ -124,16 +109,12 @@ define i32 @test_ctselect_const_true(i32 %a, i32 %b) {
define i32 @test_ctselect_const_false(i32 %a, i32 %b) {
; RV64-LABEL: test_ctselect_const_false:
; RV64: # %bb.0:
-; RV64-NEXT: li a0, 0
-; RV64-NEXT: #ARITH_FENCE
-; RV64-NEXT: xor a0, a1, a0
+; RV64-NEXT: mv a0, a1
; RV64-NEXT: ret
;
; RV32-LABEL: test_ctselect_const_false:
; RV32: # %bb.0:
-; RV32-NEXT: li a0, 0
-; RV32-NEXT: #ARITH_FENCE
-; RV32-NEXT: xor a0, a1, a0
+; RV32-NEXT: mv a0, a1
; RV32-NEXT: ret
%result = call i32 @llvm.ct.select.i32(i1 false, i32 %a, i32 %b)
ret i32 %result
@@ -150,7 +131,6 @@ define i32 @test_ctselect_icmp_eq(i32 %x, i32 %y, i32 %a, i32 %b) {
; RV64-NEXT: xor a2, a2, a3
; RV64-NEXT: addi a0, a0, -1
; RV64-NEXT: and a0, a2, a0
-; RV64-NEXT: #ARITH_FENCE
; RV64-NEXT: xor a0, a3, a0
; RV64-NEXT: ret
;
@@ -161,7 +141,6 @@ define i32 @test_ctselect_icmp_eq(i32 %x, i32 %y, i32 %a, i32 %b) {
; RV32-NEXT: xor a2, a2, a3
; RV32-NEXT: addi a0, a0, -1
; RV32-NEXT: and a0, a2, a0
-; RV32-NEXT: #ARITH_FENCE
; RV32-NEXT: xor a0, a3, a0
; RV32-NEXT: ret
%cond = icmp eq i32 %x, %y
@@ -177,7 +156,6 @@ define i32 @test_ctselect_icmp_ult(i32 %x, i32 %y, i32 %a, i32 %b) {
; RV64-NEXT: xor a2, a2, a3
; RV64-NEXT: neg a0, a0
; RV64-NEXT: and a0, a2, a0
-; RV64-NEXT: #ARITH_FENCE
; RV64-NEXT: xor a0, a3, a0
; RV64-NEXT: ret
;
@@ -187,7 +165,6 @@ define i32 @test_ctselect_icmp_ult(i32 %x, i32 %y, i32 %a, i32 %b) {
; RV32-NEXT: xor a2, a2, a3
; RV32-NEXT: neg a0, a0
; RV32-NEXT: and a0, a2, a0
-; RV32-NEXT: #ARITH_FENCE
; RV32-NEXT: xor a0, a3, a0
; RV32-NEXT: ret
%cond = icmp ult i32 %x, %y
@@ -205,7 +182,6 @@ define i32 @test_ctselect_load(i1 %cond, ptr %p1, ptr %p2) {
; RV64-NEXT: xor a1, a1, a2
; RV64-NEXT: srai a0, a0, 63
; RV64-NEXT: and a0, a1, a0
-; RV64-NEXT: #ARITH_FENCE
; RV64-NEXT: xor a0, a2, a0
; RV64-NEXT: ret
;
@@ -217,7 +193,6 @@ define i32 @test_ctselect_load(i1 %cond, ptr %p1, ptr %p2) {
; RV32-NEXT: xor a1, a1, a2
; RV32-NEXT: srai a0, a0, 31
; RV32-NEXT: and a0, a1, a0
-; RV32-NEXT: #ARITH_FENCE
; RV32-NEXT: xor a0, a2, a0
; RV32-NEXT: ret
%a = load i32, ptr %p1
@@ -232,26 +207,20 @@ define i32 @test_ctselect_nested_and_i1_to_i32(i1 %c0, i1 %c1, i32 %x, i32 %y) {
; RV64-LABEL: test_ctselect_nested_and_i1_to_i32:
; RV64: # %bb.0:
; RV64-NEXT: and a0, a0, a1
-; RV64-NEXT: andi a0, a0, 1
-; RV64-NEXT: #ARITH_FENCE
; RV64-NEXT: slli a0, a0, 63
; RV64-NEXT: xor a2, a2, a3
; RV64-NEXT: srai a0, a0, 63
; RV64-NEXT: and a0, a2, a0
-; RV64-NEXT: #ARITH_FENCE
; RV64-NEXT: xor a0, a3, a0
; RV64-NEXT: ret
;
; RV32-LABEL: test_ctselect_nested_and_i1_to_i32:
; RV32: # %bb.0:
; RV32-NEXT: and a0, a0, a1
-; RV32-NEXT: andi a0, a0, 1
-; RV32-NEXT: #ARITH_FENCE
; RV32-NEXT: slli a0, a0, 31
; RV32-NEXT: xor a2, a2, a3
; RV32-NEXT: srai a0, a0, 31
; RV32-NEXT: and a0, a2, a0
-; RV32-NEXT: #ARITH_FENCE
; RV32-NEXT: xor a0, a3, a0
; RV32-NEXT: ret
%inner = call i1 @llvm.ct.select.i1(i1 %c1, i1 true, i1 false)
@@ -266,26 +235,20 @@ define i32 @test_ctselect_nested_or_i1_to_i32(i1 %c0, i1 %c1, i32 %x, i32 %y) {
; RV64-LABEL: test_ctselect_nested_or_i1_to_i32:
; RV64: # %bb.0:
; RV64-NEXT: or a0, a0, a1
-; RV64-NEXT: andi a0, a0, 1
-; RV64-NEXT: #ARITH_FENCE
; RV64-NEXT: slli a0, a0, 63
; RV64-NEXT: xor a2, a2, a3
; RV64-NEXT: srai a0, a0, 63
; RV64-NEXT: and a0, a2, a0
-; RV64-NEXT: #ARITH_FENCE
; RV64-NEXT: xor a0, a3, a0
; RV64-NEXT: ret
;
; RV32-LABEL: test_ctselect_nested_or_i1_to_i32:
; RV32: # %bb.0:
; RV32-NEXT: or a0, a0, a1
-; RV32-NEXT: andi a0, a0, 1
-; RV32-NEXT: #ARITH_FENCE
; RV32-NEXT: slli a0, a0, 31
; RV32-NEXT: xor a2, a2, a3
; RV32-NEXT: srai a0, a0, 31
; RV32-NEXT: and a0, a2, a0
-; RV32-NEXT: #ARITH_FENCE
; RV32-NEXT: xor a0, a3, a0
; RV32-NEXT: ret
%inner = call i1 @llvm.ct.select.i1(i1 %c1, i1 true, i1 false)
@@ -302,13 +265,10 @@ define i32 @test_ctselect_double_nested_and_i1(i1 %c0, i1 %c1, i1 %c2, i32 %x, i
; RV64: # %bb.0:
; RV64-NEXT: and a0, a0, a1
; RV64-NEXT: and a0, a0, a2
-; RV64-NEXT: andi a0, a0, 1
-; RV64-NEXT: #ARITH_FENCE
; RV64-NEXT: slli a0, a0, 63
; RV64-NEXT: xor a3, a3, a4
; RV64-NEXT: srai a0, a0, 63
; RV64-NEXT: and a0, a3, a0
-; RV64-NEXT: #ARITH_FENCE
; RV64-NEXT: xor a0, a4, a0
; RV64-NEXT: ret
;
@@ -316,13 +276,10 @@ define i32 @test_ctselect_double_nested_and_i1(i1 %c0, i1 %c1, i1 %c2, i32 %x, i
; RV32: # %bb.0:
; RV32-NEXT: and a0, a0, a1
; RV32-NEXT: and a0, a0, a2
-; RV32-NEXT: andi a0, a0, 1
-; RV32-NEXT: #ARITH_FENCE
; RV32-NEXT: slli a0, a0, 31
; RV32-NEXT: xor a3, a3, a4
; RV32-NEXT: srai a0, a0, 31
; RV32-NEXT: and a0, a3, a0
-; RV32-NEXT: #ARITH_FENCE
; RV32-NEXT: xor a0, a4, a0
; RV32-NEXT: ret
%inner2 = call i1 @llvm.ct.select.i1(i1 %c2, i1 true, i1 false)
@@ -337,40 +294,28 @@ define i32 @test_ctselect_double_nested_mixed_i1(i1 %c0, i1 %c1, i1 %c2, i32 %x,
; RV64-LABEL: test_ctselect_double_nested_mixed_i1:
; RV64: # %bb.0:
; RV64-NEXT: and a0, a0, a1
-; RV64-NEXT: andi a0, a0, 1
-; RV64-NEXT: #ARITH_FENCE
; RV64-NEXT: or a0, a0, a2
-; RV64-NEXT: andi a0, a0, 1
-; RV64-NEXT: #ARITH_FENCE
; RV64-NEXT: slli a0, a0, 63
; RV64-NEXT: xor a3, a3, a4
; RV64-NEXT: srai a0, a0, 63
; RV64-NEXT: and a3, a3, a0
; RV64-NEXT: xor a4, a4, a5
-; RV64-NEXT: #ARITH_FENCE
; RV64-NEXT: xor a3, a4, a3
; RV64-NEXT: and a0, a3, a0
-; RV64-NEXT: #ARITH_FENCE
; RV64-NEXT: xor a0, a5, a0
; RV64-NEXT: ret
;
; RV32-LABEL: test_ctselect_double_nested_mixed_i1:
; RV32: # %bb.0:
; RV32-NEXT: and a0, a0, a1
-; RV32-NEXT: andi a0, a0, 1
-; RV32-NEXT: #ARITH_FENCE
; RV32-NEXT: or a0, a0, a2
-; RV32-NEXT: andi a0, a0, 1
-; RV32-NEXT: #ARITH_FENCE
; RV32-NEXT: slli a0, a0, 31
; RV32-NEXT: xor a3, a3, a4
; RV32-NEXT: srai a0, a0, 31
; RV32-NEXT: and a3, a3, a0
; RV32-NEXT: xor a4, a4, a5
-; RV32-NEXT: #ARITH_FENCE
; RV32-NEXT: xor a3, a4, a3
; RV32-NEXT: and a0, a3, a0
-; RV32-NEXT: #ARITH_FENCE
; RV32-NEXT: xor a0, a5, a0
; RV32-NEXT: ret
%inner1 = call i1 @llvm.ct.select.i1(i1 %c1, i1 true, i1 false)
@@ -391,12 +336,10 @@ define i32 @test_ctselect_nested(i1 %cond1, i1 %cond2, i32 %a, i32 %b, i32 %c) {
; RV64-NEXT: srai a1, a1, 63
; RV64-NEXT: and a1, a2, a1
; RV64-NEXT: xor a3, a3, a4
-; RV64-NEXT: #ARITH_FENCE
; RV64-NEXT: slli a0, a0, 63
; RV64-NEXT: xor a1, a3, a1
; RV64-NEXT: srai a0, a0, 63
; RV64-NEXT: and a0, a1, a0
-; RV64-NEXT: #ARITH_FENCE
; RV64-NEXT: xor a0, a4, a0
; RV64-NEXT: ret
;
@@ -407,12 +350,10 @@ define i32 @test_ctselect_nested(i1 %cond1, i1 %cond2, i32 %a, i32 %b, i32 %c) {
; RV32-NEXT: srai a1, a1, 31
; RV32-NEXT: and a1, a2, a1
; RV32-NEXT: xor a3, a3, a4
-; RV32-NEXT: #ARITH_FENCE
; RV32-NEXT: slli a0, a0, 31
; RV32-NEXT: xor a1, a3, a1
; RV32-NEXT: srai a0, a0, 31
; RV32-NEXT: and a0, a1, a0
-; RV32-NEXT: #ARITH_FENCE
; RV32-NEXT: xor a0, a4, a0
; RV32-NEXT: ret
%inner = call i32 @llvm.ct.select.i32(i1 %cond2, i32 %a, i32 %b)
@@ -429,8 +370,7 @@ define float @test_ctselect_f32_nan_inf(i1 %cond) {
; RV64-NEXT: srai a0, a0, 63
; RV64-NEXT: and a0, a0, a1
; RV64-NEXT: lui a1, 522240
-; RV64-NEXT: #ARITH_FENCE
-; RV64-NEXT: xor a0, a0, a1
+; RV64-NEXT: or a0, a0, a1
; RV64-NEXT: ret
;
; RV32-LABEL: test_ctselect_f32_nan_inf:
@@ -440,8 +380,7 @@ define float @test_ctselect_f32_nan_inf(i1 %cond) {
; RV32-NEXT: srai a0, a0, 31
; RV32-NEXT: and a0, a0, a1
; RV32-NEXT: lui a1, 522240
-; RV32-NEXT: #ARITH_FENCE
-; RV32-NEXT: xor a0, a0, a1
+; RV32-NEXT: or a0, a0, a1
; RV32-NEXT: ret
%result = call float @llvm.ct.select.f32(i1 %cond, float 0x7FF8000000000000, float 0x7FF0000000000000)
ret float %result
@@ -457,22 +396,18 @@ define double @test_ctselect_f64_nan_inf(i1 %cond) {
; RV64-NEXT: li a2, 2047
; RV64-NEXT: and a0, a0, a1
; RV64-NEXT: slli a2, a2, 52
-; RV64-NEXT: #ARITH_FENCE
-; RV64-NEXT: xor a0, a0, a2
+; RV64-NEXT: or a0, a0, a2
; RV64-NEXT: ret
;
; RV32-LABEL: test_ctselect_f64_nan_inf:
; RV32: # %bb.0:
-; RV32-NEXT: li a2, 0
+; RV32-NEXT: lui a1, 128
; RV32-NEXT: slli a0, a0, 31
; RV32-NEXT: srai a0, a0, 31
-; RV32-NEXT: lui a1, 128
; RV32-NEXT: and a0, a0, a1
; RV32-NEXT: lui a1, 524032
-; RV32-NEXT: #ARITH_FENCE
-; RV32-NEXT: xor a1, a0, a1
-; RV32-NEXT: #ARITH_FENCE
-; RV32-NEXT: mv a0, a2
+; RV32-NEXT: or a1, a0, a1
+; RV32-NEXT: li a0, 0
; RV32-NEXT: ret
%result = call double @llvm.ct.select.f64(i1 %cond, double 0x7FF8000000000000, double 0x7FF0000000000000)
ret double %result
@@ -486,7 +421,6 @@ define float @test_ctselect_f32(i1 %cond, float %a, float %b) {
; RV64-NEXT: slli a0, a0, 63
; RV64-NEXT: srai a0, a0, 63
; RV64-NEXT: and a0, a1, a0
-; RV64-NEXT: #ARITH_FENCE
; RV64-NEXT: xor a0, a2, a0
; RV64-NEXT: ret
;
@@ -496,7 +430,6 @@ define float @test_ctselect_f32(i1 %cond, float %a, float %b) {
; RV32-NEXT: slli a0, a0, 31
; RV32-NEXT: srai a0, a0, 31
; RV32-NEXT: and a0, a1, a0
-; RV32-NEXT: #ARITH_FENCE
; RV32-NEXT: xor a0, a2, a0
; RV32-NEXT: ret
%result = call float @llvm.ct.select.f32(i1 %cond, float %a, float %b)
@@ -510,7 +443,6 @@ define double @test_ctselect_f64(i1 %cond, double %a, double %b) {
; RV64-NEXT: slli a0, a0, 63
; RV64-NEXT: srai a0, a0, 63
; RV64-NEXT: and a0, a1, a0
-; RV64-NEXT: #ARITH_FENCE
; RV64-NEXT: xor a0, a2, a0
; RV64-NEXT: ret
;
@@ -519,12 +451,10 @@ define double @test_ctselect_f64(i1 %cond, double %a, double %b) {
; RV32-NEXT: slli a0, a0, 31
; RV32-NEXT: xor a1, a1, a3
; RV32-NEXT: srai a0, a0, 31
-; RV32-NEXT: and a1, a1, a0
; RV32-NEXT: xor a2, a2, a4
-; RV32-NEXT: #ARITH_FENCE
+; RV32-NEXT: and a1, a1, a0
; RV32-NEXT: and a2, a2, a0
; RV32-NEXT: xor a0, a3, a1
-; RV32-NEXT: #ARITH_FENCE
; RV32-NEXT: xor a1, a4, a2
; RV32-NEXT: ret
%result = call double @llvm.ct.select.f64(i1 %cond, double %a, double %b)
@@ -535,31 +465,27 @@ define double @test_ctselect_f64(i1 %cond, double %a, double %b) {
define <4 x i32> @test_ctselect_v4i32(i1 %cond, <4 x i32> %a, <4 x i32> %b) {
; RV64-LABEL: test_ctselect_v4i32:
; RV64: # %bb.0:
-; RV64-NEXT: lw a4, 0(a2)
-; RV64-NEXT: lw a5, 8(a2)
-; RV64-NEXT: lw a6, 0(a3)
+; RV64-NEXT: lw a4, 0(a3)
+; RV64-NEXT: lw a5, 0(a2)
+; RV64-NEXT: lw a6, 8(a2)
; RV64-NEXT: lw a7, 8(a3)
; RV64-NEXT: lw t0, 16(a3)
; RV64-NEXT: lw a3, 24(a3)
; RV64-NEXT: lw t1, 16(a2)
; RV64-NEXT: lw a2, 24(a2)
+; RV64-NEXT: xor a5, a5, a4
; RV64-NEXT: slli a1, a1, 63
-; RV64-NEXT: xor a4, a4, a6
; RV64-NEXT: srai a1, a1, 63
-; RV64-NEXT: and a4, a4, a1
-; RV64-NEXT: xor a5, a5, a7
-; RV64-NEXT: #ARITH_FENCE
+; RV64-NEXT: xor a6, a6, a7
; RV64-NEXT: and a5, a5, a1
-; RV64-NEXT: xor a4, a6, a4
-; RV64-NEXT: #ARITH_FENCE
-; RV64-NEXT: xor a6, t1, t0
; RV64-NEXT: and a6, a6, a1
+; RV64-NEXT: xor a4, a4, a5
+; RV64-NEXT: xor a5, a7, a6
+; RV64-NEXT: xor a6, t1, t0
; RV64-NEXT: xor a2, a2, a3
-; RV64-NEXT: xor a5, a7, a5
-; RV64-NEXT: #ARITH_FENCE
+; RV64-NEXT: and a6, a6, a1
; RV64-NEXT: and a1, a2, a1
; RV64-NEXT: xor a2, t0, a6
-; RV64-NEXT: #ARITH_FENCE
; RV64-NEXT: xor a1, a3, a1
; RV64-NEXT: sw a4, 0(a0)
; RV64-NEXT: sw a5, 4(a0)
@@ -569,31 +495,27 @@ define <4 x i32> @test_ctselect_v4i32(i1 %cond, <4 x i32> %a, <4 x i32> %b) {
;
; RV32-LABEL: test_ctselect_v4i32:
; RV32: # %bb.0:
-; RV32-NEXT: lw a4, 0(a2)
-; RV32-NEXT: lw a5, 4(a2)
-; RV32-NEXT: lw a6, 0(a3)
+; RV32-NEXT: lw a4, 0(a3)
+; RV32-NEXT: lw a5, 0(a2)
+; RV32-NEXT: lw a6, 4(a2)
; RV32-NEXT: lw a7, 4(a3)
; RV32-NEXT: lw t0, 8(a3)
; RV32-NEXT: lw a3, 12(a3)
; RV32-NEXT: lw t1, 8(a2)
; RV32-NEXT: lw a2, 12(a2)
+; RV32-NEXT: xor a5, a5, a4
; RV32-NEXT: slli a1, a1, 31
-; RV32-NEXT: xor a4, a4, a6
; RV32-NEXT: srai a1, a1, 31
-; RV32-NEXT: and a4, a4, a1
-; RV32-NEXT: xor a5, a5, a7
-; RV32-NEXT: #ARITH_FENCE
+; RV32-NEXT: xor a6, a6, a7
; RV32-NEXT: and a5, a5, a1
-; RV32-NEXT: xor a4, a6, a4
-; RV32-NEXT: #ARITH_FENCE
-; RV32-NEXT: xor a6, t1, t0
; RV32-NEXT: and a6, a6, a1
+; RV32-NEXT: xor a4, a4, a5
+; RV32-NEXT: xor a5, a7, a6
+; RV32-NEXT: xor a6, t1, t0
; RV32-NEXT: xor a2, a2, a3
-; RV32-NEXT: xor a5, a7, a5
-; RV32-NEXT: #ARITH_FENCE
+; RV32-NEXT: and a6, a6, a1
; RV32-NEXT: and a1, a2, a1
; RV32-NEXT: xor a2, t0, a6
-; RV32-NEXT: #ARITH_FENCE
; RV32-NEXT: xor a1, a3, a1
; RV32-NEXT: sw a4, 0(a0)
; RV32-NEXT: sw a5, 4(a0)
@@ -606,31 +528,27 @@ define <4 x i32> @test_ctselect_v4i32(i1 %cond, <4 x i32> %a, <4 x i32> %b) {
define <4 x float> @test_ctselect_v4f32(i1 %cond, <4 x float> %a, <4 x float> %b) {
; RV64-LABEL: test_ctselect_v4f32:
; RV64: # %bb.0:
-; RV64-NEXT: lw a4, 0(a2)
-; RV64-NEXT: lw a5, 8(a2)
-; RV64-NEXT: lw a6, 0(a3)
+; RV64-NEXT: lw a4, 0(a3)
+; RV64-NEXT: lw a5, 0(a2)
+; RV64-NEXT: lw a6, 8(a2)
; RV64-NEXT: lw a7, 8(a3)
; RV64-NEXT: lw t0, 16(a3)
; RV64-NEXT: lw a3, 24(a3)
; RV64-NEXT: lw t1, 16(a2)
; RV64-NEXT: lw a2, 24(a2)
+; RV64-NEXT: xor a5, a5, a4
; RV64-NEXT: slli a1, a1, 63
-; RV64-NEXT: xor a4, a4, a6
; RV64-NEXT: srai a1, a1, 63
-; RV64-NEXT: and a4, a4, a1
-; RV64-NEXT: xor a5, a5, a7
-; RV64-NEXT: #ARITH_FENCE
+; RV64-NEXT: xor a6, a6, a7
; RV64-NEXT: and a5, a5, a1
-; RV64-NEXT: xor a4, a6, a4
-; RV64-NEXT: #ARITH_FENCE
-; RV64-NEXT: xor a6, t1, t0
; RV64-NEXT: and a6, a6, a1
+; RV64-NEXT: xor a4, a4, a5
+; RV64-NEXT: xor a5, a7, a6
+; RV64-NEXT: xor a6, t1, t0
; RV64-NEXT: xor a2, a2, a3
-; RV64-NEXT: xor a5, a7, a5
-; RV64-NEXT: #ARITH_FENCE
+; RV64-NEXT: and a6, a6, a1
; RV64-NEXT: and a1, a2, a1
; RV64-NEXT: xor a2, t0, a6
-; RV64-NEXT: #ARITH_FENCE
; RV64-NEXT: xor a1, a3, a1
; RV64-NEXT: sw a4, 0(a0)
; RV64-NEXT: sw a5, 4(a0)
@@ -640,31 +558,27 @@ define <4 x float> @test_ctselect_v4f32(i1 %cond, <4 x float> %a, <4 x float> %b
;
; RV32-LABEL: test_ctselect_v4f32:
; RV32: # %bb.0:
-; RV32-NEXT: lw a4, 0(a2)
-; RV32-NEXT: lw a5, 4(a2)
-; RV32-NEXT: lw a6, 0(a3)
+; RV32-NEXT: lw a4, 0(a3)
+; RV32-NEXT: lw a5, 0(a2)
+; RV32-NEXT: lw a6, 4(a2)
; RV32-NEXT: lw a7, 4(a3)
; RV32-NEXT: lw t0, 8(a3)
; RV32-NEXT: lw a3, 12(a3)
; RV32-NEXT: lw t1, 8(a2)
; RV32-NEXT: lw a2, 12(a2)
+; RV32-NEXT: xor a5, a5, a4
; RV32-NEXT: slli a1, a1, 31
-; RV32-NEXT: xor a4, a4, a6
; RV32-NEXT: srai a1, a1, 31
-; RV32-NEXT: and a4, a4, a1
-; RV32-NEXT: xor a5, a5, a7
-; RV32-NEXT: #ARITH_FENCE
+; RV32-NEXT: xor a6, a6, a7
; RV32-NEXT: and a5, a5, a1
-; RV32-NEXT: xor a4, a6, a4
-; RV32-NEXT: #ARITH_FENCE
-; RV32-NEXT: xor a6, t1, t0
; RV32-NEXT: and a6, a6, a1
+; RV32-NEXT: xor a4, a4, a5
+; RV32-NEXT: xor a5, a7, a6
+; RV32-NEXT: xor a6, t1, t0
; RV32-NEXT: xor a2, a2, a3
-; RV32-NEXT: xor a5, a7, a5
-; RV32-NEXT: #ARITH_FENCE
+; RV32-NEXT: and a6, a6, a1
; RV32-NEXT: and a1, a2, a1
; RV32-NEXT: xor a2, t0, a6
-; RV32-NEXT: #ARITH_FENCE
; RV32-NEXT: xor a1, a3, a1
; RV32-NEXT: sw a4, 0(a0)
; RV32-NEXT: sw a5, 4(a0)
@@ -685,64 +599,56 @@ define <8 x i32> @test_ctselect_v8i32(i1 %cond, <8 x i32> %a, <8 x i32> %b) {
; RV64-NEXT: .cfi_offset s0, -8
; RV64-NEXT: .cfi_offset s1, -16
; RV64-NEXT: .cfi_offset s2, -24
-; RV64-NEXT: lw a4, 0(a2)
-; RV64-NEXT: lw a5, 0(a3)
-; RV64-NEXT: lw a6, 8(a3)
-; RV64-NEXT: lw a7, 16(a3)
-; RV64-NEXT: lw t0, 24(a3)
-; RV64-NEXT: lw t1, 8(a2)
-; RV64-NEXT: lw t2, 16(a2)
-; RV64-NEXT: lw t3, 24(a2)
-; RV64-NEXT: xor a4, a4, a5
+; RV64-NEXT: lw a4, 32(a3)
+; RV64-NEXT: lw a5, 40(a3)
+; RV64-NEXT: lw a6, 48(a3)
+; RV64-NEXT: lw a7, 56(a3)
+; RV64-NEXT: lw t0, 32(a2)
+; RV64-NEXT: lw t1, 40(a2)
+; RV64-NEXT: lw t2, 48(a2)
+; RV64-NEXT: lw t3, 56(a2)
+; RV64-NEXT: lw t4, 0(a3)
+; RV64-NEXT: lw t5, 0(a2)
+; RV64-NEXT: lw t6, 8(a2)
+; RV64-NEXT: lw s0, 8(a3)
+; RV64-NEXT: lw s1, 16(a3)
+; RV64-NEXT: lw a3, 24(a3)
+; RV64-NEXT: lw s2, 16(a2)
+; RV64-NEXT: lw a2, 24(a2)
+; RV64-NEXT: xor t5, t5, t4
; RV64-NEXT: slli a1, a1, 63
; RV64-NEXT: srai a1, a1, 63
-; RV64-NEXT: lw t4, 32(a3)
-; RV64-NEXT: lw t5, 40(a3)
-; RV64-NEXT: lw t6, 48(a3)
-; RV64-NEXT: lw a3, 56(a3)
-; RV64-NEXT: and a4, a4, a1
-; RV64-NEXT: xor t1, t1, a6
-; RV64-NEXT: lw s0, 32(a2)
-; RV64-NEXT: lw s1, 40(a2)
-; RV64-NEXT: lw s2, 48(a2)
-; RV64-NEXT: lw a2, 56(a2)
-; RV64-NEXT: #ARITH_FENCE
-; RV64-NEXT: and t1, t1, a1
-; RV64-NEXT: xor a4, a5, a4
-; RV64-NEXT: #ARITH_FENCE
-; RV64-NEXT: xor a5, t2, a7
-; RV64-NEXT: and a5, a5, a1
-; RV64-NEXT: xor t2, t3, t0
-; RV64-NEXT: xor a6, a6, t1
-; RV64-NEXT: #ARITH_FENCE
-; RV64-NEXT: and t1, t2, a1
-; RV64-NEXT: xor a5, a7, a5
-; RV64-NEXT: #ARITH_FENCE
-; RV64-NEXT: xor a7, s0, t4
-; RV64-NEXT: and a7, a7, a1
-; RV64-NEXT: xor t2, s1, t5
-; RV64-NEXT: xor t0, t0, t1
-; RV64-NEXT: #ARITH_FENCE
-; RV64-NEXT: and t1, t2, a1
-; RV64-NEXT: xor a7, t4, a7
-; RV64-NEXT: #ARITH_FENCE
-; RV64-NEXT: xor t2, s2, t6
-; RV64-NEXT: and t2, t2, a1
+; RV64-NEXT: xor t6, t6, s0
+; RV64-NEXT: and t5, t5, a1
+; RV64-NEXT: and t6, t6, a1
+; RV64-NEXT: xor t4, t4, t5
+; RV64-NEXT: xor t5, s0, t6
+; RV64-NEXT: xor t6, s2, s1
; RV64-NEXT: xor a2, a2, a3
-; RV64-NEXT: xor t1, t5, t1
-; RV64-NEXT: #ARITH_FENCE
-; RV64-NEXT: and a1, a2, a1
-; RV64-NEXT: xor a2, t6, t2
-; RV64-NEXT: #ARITH_FENCE
-; RV64-NEXT: xor a1, a3, a1
-; RV64-NEXT: sw a7, 16(a0)
-; RV64-NEXT: sw t1, 20(a0)
-; RV64-NEXT: sw a2, 24(a0)
+; RV64-NEXT: and t6, t6, a1
+; RV64-NEXT: and a2, a2, a1
+; RV64-NEXT: xor t6, s1, t6
+; RV64-NEXT: xor a2, a3, a2
+; RV64-NEXT: xor a3, t0, a4
+; RV64-NEXT: xor t0, t1, a5
+; RV64-NEXT: and a3, a3, a1
+; RV64-NEXT: and t0, t0, a1
+; RV64-NEXT: xor a3, a4, a3
+; RV64-NEXT: xor a4, a5, t0
+; RV64-NEXT: xor a5, t2, a6
+; RV64-NEXT: xor t0, t3, a7
+; RV64-NEXT: and a5, a5, a1
+; RV64-NEXT: and a1, t0, a1
+; RV64-NEXT: xor a5, a6, a5
+; RV64-NEXT: xor a1, a7, a1
+; RV64-NEXT: sw a3, 16(a0)
+; RV64-NEXT: sw a4, 20(a0)
+; RV64-NEXT: sw a5, 24(a0)
; RV64-NEXT: sw a1, 28(a0)
-; RV64-NEXT: sw a4, 0(a0)
-; RV64-NEXT: sw a6, 4(a0)
-; RV64-NEXT: sw a5, 8(a0)
-; RV64-NEXT: sw t0, 12(a0)
+; RV64-NEXT: sw t4, 0(a0)
+; RV64-NEXT: sw t5, 4(a0)
+; RV64-NEXT: sw t6, 8(a0)
+; RV64-NEXT: sw a2, 12(a0)
; RV64-NEXT: ld s0, 24(sp) # 8-byte Folded Reload
; RV64-NEXT: ld s1, 16(sp) # 8-byte Folded Reload
; RV64-NEXT: ld s2, 8(sp) # 8-byte Folded Reload
@@ -763,64 +669,56 @@ define <8 x i32> @test_ctselect_v8i32(i1 %cond, <8 x i32> %a, <8 x i32> %b) {
; RV32-NEXT: .cfi_offset s0, -4
; RV32-NEXT: .cfi_offset s1, -8
; RV32-NEXT: .cfi_offset s2, -12
-; RV32-NEXT: lw a4, 0(a2)
-; RV32-NEXT: lw a5, 0(a3)
-; RV32-NEXT: lw a6, 4(a3)
-; RV32-NEXT: lw a7, 8(a3)
-; RV32-NEXT: lw t0, 12(a3)
-; RV32-NEXT: lw t1, 4(a2)
-; RV32-NEXT: lw t2, 8(a2)
-; RV32-NEXT: lw t3, 12(a2)
-; RV32-NEXT: xor a4, a4, a5
+; RV32-NEXT: lw a4, 16(a3)
+; RV32-NEXT: lw a5, 20(a3)
+; RV32-NEXT: lw a6, 24(a3)
+; RV32-NEXT: lw a7, 28(a3)
+; RV32-NEXT: lw t0, 16(a2)
+; RV32-NEXT: lw t1, 20(a2)
+; RV32-NEXT: lw t2, 24(a2)
+; RV32-NEXT: lw t3, 28(a2)
+; RV32-NEXT: lw t4, 0(a3)
+; RV32-NEXT: lw t5, 0(a2)
+; RV32-NEXT: lw t6, 4(a2)
+; RV32-NEXT: lw s0, 4(a3)
+; RV32-NEXT: lw s1, 8(a3)
+; RV32-NEXT: lw a3, 12(a3)
+; RV32-NEXT: lw s2, 8(a2)
+; RV32-NEXT: lw a2, 12(a2)
+; RV32-NEXT: xor t5, t5, t4
; RV32-NEXT: slli a1, a1, 31
; RV32-NEXT: srai a1, a1, 31
-; RV32-NEXT: lw t4, 16(a3)
-; RV32-NEXT: lw t5, 20(a3)
-; RV32-NEXT: lw t6, 24(a3)
-; RV32-NEXT: lw a3, 28(a3)
-; RV32-NEXT: and a4, a4, a1
-; RV32-NEXT: xor t1, t1, a6
-; RV32-NEXT: lw s0, 16(a2)
-; RV32-NEXT: lw s1, 20(a2)
-; RV32-NEXT: lw s2, 24(a2)
-; RV32-NEXT: lw a2, 28(a2)
-; RV32-NEXT: #ARITH_FENCE
-; RV32-NEXT: and t1, t1, a1
-; RV32-NEXT: xor a4, a5, a4
-; RV32-NEXT: #ARITH_FENCE
-; RV32-NEXT: xor a5, t2, a7
-; RV32-NEXT: and a5, a5, a1
-; RV32-NEXT: xor t2, t3, t0
-; RV32-NEXT: xor a6, a6, t1
-; RV32-NEXT: #ARITH_FENCE
-; RV32-NEXT: and t1, t2, a1
-; RV32-NEXT: xor a5, a7, a5
-; RV32-NEXT: #ARITH_FENCE
-; RV32-NEXT: xor a7, s0, t4
-; RV32-NEXT: and a7, a7, a1
-; RV32-NEXT: xor t2, s1, t5
-; RV32-NEXT: xor t0, t0, t1
-; RV32-NEXT: #ARITH_FENCE
-; RV32-NEXT: and t1, t2, a1
-; RV32-NEXT: xor a7, t4, a7
-; RV32-NEXT: #ARITH_FENCE
-; RV32-NEXT: xor t2, s2, t6
-; RV32-NEXT: and t2, t2, a1
+; RV32-NEXT: xor t6, t6, s0
+; RV32-NEXT: and t5, t5, a1
+; RV32-NEXT: and t6, t6, a1
+; RV32-NEXT: xor t4, t4, t5
+; RV32-NEXT: xor t5, s0, t6
+; RV32-NEXT: xor t6, s2, s1
; RV32-NEXT: xor a2, a2, a3
-; RV32-NEXT: xor t1, t5, t1
-; RV32-NEXT: #ARITH_FENCE
-; RV32-NEXT: and a1, a2, a1
-; RV32-NEXT: xor a2, t6, t2
-; RV32-NEXT: #ARITH_FENCE
-; RV32-NEXT: xor a1, a3, a1
-; RV32-NEXT: sw a7, 16(a0)
-; RV32-NEXT: sw t1, 20(a0)
-; RV32-NEXT: sw a2, 24(a0)
+; RV32-NEXT: and t6, t6, a1
+; RV32-NEXT: and a2, a2, a1
+; RV32-NEXT: xor t6, s1, t6
+; RV32-NEXT: xor a2, a3, a2
+; RV32-NEXT: xor a3, t0, a4
+; RV32-NEXT: xor t0, t1, a5
+; RV32-NEXT: and a3, a3, a1
+; RV32-NEXT: and t0, t0, a1
+; RV32-NEXT: xor a3, a4, a3
+; RV32-NEXT: xor a4, a5, t0
+; RV32-NEXT: xor a5, t2, a6
+; RV32-NEXT: xor t0, t3, a7
+; RV32-NEXT: and a5, a5, a1
+; RV32-NEXT: and a1, t0, a1
+; RV32-NEXT: xor a5, a6, a5
+; RV32-NEXT: xor a1, a7, a1
+; RV32-NEXT: sw a3, 16(a0)
+; RV32-NEXT: sw a4, 20(a0)
+; RV32-NEXT: sw a5, 24(a0)
; RV32-NEXT: sw a1, 28(a0)
-; RV32-NEXT: sw a4, 0(a0)
-; RV32-NEXT: sw a6, 4(a0)
-; RV32-NEXT: sw a5, 8(a0)
-; RV32-NEXT: sw t0, 12(a0)
+; RV32-NEXT: sw t4, 0(a0)
+; RV32-NEXT: sw t5, 4(a0)
+; RV32-NEXT: sw t6, 8(a0)
+; RV32-NEXT: sw a2, 12(a0)
; RV32-NEXT: lw s0, 12(sp) # 4-byte Folded Reload
; RV32-NEXT: lw s1, 8(sp) # 4-byte Folded Reload
; RV32-NEXT: lw s2, 4(sp) # 4-byte Folded Reload
diff --git a/llvm/test/CodeGen/X86/ctselect.ll b/llvm/test/CodeGen/X86/ctselect.ll
index 7221b44d4dcc1..203d9a027d0ec 100644
--- a/llvm/test/CodeGen/X86/ctselect.ll
+++ b/llvm/test/CodeGen/X86/ctselect.ll
@@ -13,7 +13,6 @@ define i8 @test_ctselect_i8(i1 %cond, i8 %a, i8 %b) #0 {
; X64-NEXT: xorl %edx, %esi
; X64-NEXT: negb %al
; X64-NEXT: andb %sil, %al
-; X64-NEXT: #ARITH_FENCE
; X64-NEXT: xorb %dl, %al
; X64-NEXT: # kill: def $al killed $al killed $eax
; X64-NEXT: retq
@@ -27,7 +26,6 @@ define i8 @test_ctselect_i8(i1 %cond, i8 %a, i8 %b) #0 {
; X32-NEXT: xorb %cl, %dl
; X32-NEXT: negb %al
; X32-NEXT: andb %dl, %al
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorb %cl, %al
; X32-NEXT: retl
;
@@ -40,7 +38,6 @@ define i8 @test_ctselect_i8(i1 %cond, i8 %a, i8 %b) #0 {
; X32-NOCMOV-NEXT: xorb %cl, %dl
; X32-NOCMOV-NEXT: negb %al
; X32-NOCMOV-NEXT: andb %dl, %al
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorb %cl, %al
; X32-NOCMOV-NEXT: retl
%result = call i8 @llvm.ct.select.i8(i1 %cond, i8 %a, i8 %b)
@@ -55,7 +52,6 @@ define b8 @test_ctselect_b8(i1 %cond, b8 %a, b8 %b) #0 {
; X64-NEXT: xorl %edx, %esi
; X64-NEXT: negb %al
; X64-NEXT: andb %sil, %al
-; X64-NEXT: #ARITH_FENCE
; X64-NEXT: xorb %dl, %al
; X64-NEXT: # kill: def $al killed $al killed $eax
; X64-NEXT: retq
@@ -69,7 +65,6 @@ define b8 @test_ctselect_b8(i1 %cond, b8 %a, b8 %b) #0 {
; X32-NEXT: xorb %cl, %dl
; X32-NEXT: negb %al
; X32-NEXT: andb %dl, %al
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorb %cl, %al
; X32-NEXT: retl
;
@@ -82,7 +77,6 @@ define b8 @test_ctselect_b8(i1 %cond, b8 %a, b8 %b) #0 {
; X32-NOCMOV-NEXT: xorb %cl, %dl
; X32-NOCMOV-NEXT: negb %al
; X32-NOCMOV-NEXT: andb %dl, %al
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorb %cl, %al
; X32-NOCMOV-NEXT: retl
%result = call b8 @llvm.ct.select.b8(i1 %cond, b8 %a, b8 %b)
@@ -97,7 +91,6 @@ define i32 @test_ctselect_i32(i1 %cond, i32 %a, i32 %b) #0 {
; X64-NEXT: andl $1, %eax
; X64-NEXT: negl %eax
; X64-NEXT: andl %esi, %eax
-; X64-NEXT: #ARITH_FENCE
; X64-NEXT: xorl %edx, %eax
; X64-NEXT: retq
;
@@ -111,7 +104,6 @@ define i32 @test_ctselect_i32(i1 %cond, i32 %a, i32 %b) #0 {
; X32-NEXT: movzbl %al, %eax
; X32-NEXT: negl %eax
; X32-NEXT: andl %edx, %eax
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %ecx, %eax
; X32-NEXT: retl
;
@@ -125,7 +117,6 @@ define i32 @test_ctselect_i32(i1 %cond, i32 %a, i32 %b) #0 {
; X32-NOCMOV-NEXT: movzbl %al, %eax
; X32-NOCMOV-NEXT: negl %eax
; X32-NOCMOV-NEXT: andl %edx, %eax
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %ecx, %eax
; X32-NOCMOV-NEXT: retl
%result = call i32 @llvm.ct.select.i32(i1 %cond, i32 %a, i32 %b)
@@ -140,7 +131,6 @@ define i64 @test_ctselect_i64(i1 %cond, i64 %a, i64 %b) #0 {
; X64-NEXT: andl $1, %eax
; X64-NEXT: negq %rax
; X64-NEXT: andq %rsi, %rax
-; X64-NEXT: #ARITH_FENCE
; X64-NEXT: xorq %rdx, %rax
; X64-NEXT: retq
;
@@ -157,12 +147,10 @@ define i64 @test_ctselect_i64(i1 %cond, i64 %a, i64 %b) #0 {
; X32-NEXT: movzbl %dl, %edi
; X32-NEXT: negl %edi
; X32-NEXT: andl %edi, %eax
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %esi, %eax
; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
; X32-NEXT: xorl %ecx, %edx
; X32-NEXT: andl %edi, %edx
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %ecx, %edx
; X32-NEXT: popl %esi
; X32-NEXT: popl %edi
@@ -181,12 +169,10 @@ define i64 @test_ctselect_i64(i1 %cond, i64 %a, i64 %b) #0 {
; X32-NOCMOV-NEXT: movzbl %dl, %edi
; X32-NOCMOV-NEXT: negl %edi
; X32-NOCMOV-NEXT: andl %edi, %eax
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %esi, %eax
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
; X32-NOCMOV-NEXT: xorl %ecx, %edx
; X32-NOCMOV-NEXT: andl %edi, %edx
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %ecx, %edx
; X32-NOCMOV-NEXT: popl %esi
; X32-NOCMOV-NEXT: popl %edi
@@ -204,7 +190,6 @@ define float @test_ctselect_f32(i1 %cond, float %a, float %b) #0 {
; X64-NEXT: andl $1, %edi
; X64-NEXT: negl %edi
; X64-NEXT: andl %ecx, %edi
-; X64-NEXT: #ARITH_FENCE
; X64-NEXT: xorl %eax, %edi
; X64-NEXT: movd %edi, %xmm0
; X64-NEXT: retq
@@ -224,7 +209,6 @@ define float @test_ctselect_f32(i1 %cond, float %a, float %b) #0 {
; X32-NEXT: movl (%esp), %edx
; X32-NEXT: xorl %ecx, %edx
; X32-NEXT: andl %eax, %edx
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %ecx, %edx
; X32-NEXT: movl %edx, {{[0-9]+}}(%esp)
; X32-NEXT: flds {{[0-9]+}}(%esp)
@@ -246,7 +230,6 @@ define float @test_ctselect_f32(i1 %cond, float %a, float %b) #0 {
; X32-NOCMOV-NEXT: movl (%esp), %edx
; X32-NOCMOV-NEXT: xorl %ecx, %edx
; X32-NOCMOV-NEXT: andl %eax, %edx
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %ecx, %edx
; X32-NOCMOV-NEXT: movl %edx, {{[0-9]+}}(%esp)
; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
@@ -266,7 +249,6 @@ define double @test_ctselect_f64(i1 %cond, double %a, double %b) #0 {
; X64-NEXT: andl $1, %edi
; X64-NEXT: negq %rdi
; X64-NEXT: andq %rcx, %rdi
-; X64-NEXT: #ARITH_FENCE
; X64-NEXT: xorq %rax, %rdi
; X64-NEXT: movq %rdi, %xmm0
; X64-NEXT: retq
@@ -288,13 +270,11 @@ define double @test_ctselect_f64(i1 %cond, double %a, double %b) #0 {
; X32-NEXT: movl (%esp), %esi
; X32-NEXT: xorl %edx, %esi
; X32-NEXT: andl %eax, %esi
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %edx, %esi
; X32-NEXT: movl %esi, (%esp)
; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
; X32-NEXT: xorl %ecx, %edx
; X32-NEXT: andl %eax, %edx
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %ecx, %edx
; X32-NEXT: movl %edx, {{[0-9]+}}(%esp)
; X32-NEXT: fldl (%esp)
@@ -319,13 +299,11 @@ define double @test_ctselect_f64(i1 %cond, double %a, double %b) #0 {
; X32-NOCMOV-NEXT: movl (%esp), %esi
; X32-NOCMOV-NEXT: xorl %edx, %esi
; X32-NOCMOV-NEXT: andl %eax, %esi
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %edx, %esi
; X32-NOCMOV-NEXT: movl %esi, (%esp)
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
; X32-NOCMOV-NEXT: xorl %ecx, %edx
; X32-NOCMOV-NEXT: andl %eax, %edx
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %ecx, %edx
; X32-NOCMOV-NEXT: movl %edx, {{[0-9]+}}(%esp)
; X32-NOCMOV-NEXT: fldl (%esp)
@@ -345,7 +323,6 @@ define half @test_ctselect_f16(i1 %cond, half %a, half %b) #0 {
; X64-NEXT: andl $1, %edi
; X64-NEXT: negl %edi
; X64-NEXT: andl %ecx, %edi
-; X64-NEXT: #ARITH_FENCE
; X64-NEXT: xorl %eax, %edi
; X64-NEXT: pinsrw $0, %edi, %xmm0
; X64-NEXT: retq
@@ -360,7 +337,6 @@ define half @test_ctselect_f16(i1 %cond, half %a, half %b) #0 {
; X32-NEXT: movzbl %al, %eax
; X32-NEXT: negl %eax
; X32-NEXT: andl %edx, %eax
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %ecx, %eax
; X32-NEXT: # kill: def $ax killed $ax killed $eax
; X32-NEXT: retl
@@ -375,7 +351,6 @@ define half @test_ctselect_f16(i1 %cond, half %a, half %b) #0 {
; X32-NOCMOV-NEXT: movzbl %al, %eax
; X32-NOCMOV-NEXT: negl %eax
; X32-NOCMOV-NEXT: andl %edx, %eax
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %ecx, %eax
; X32-NOCMOV-NEXT: # kill: def $ax killed $ax killed $eax
; X32-NOCMOV-NEXT: retl
@@ -392,7 +367,6 @@ define bfloat @test_ctselect_bf16(i1 %cond, bfloat %a, bfloat %b) #0 {
; X64-NEXT: andl $1, %edi
; X64-NEXT: negl %edi
; X64-NEXT: andl %ecx, %edi
-; X64-NEXT: #ARITH_FENCE
; X64-NEXT: xorl %eax, %edi
; X64-NEXT: pinsrw $0, %edi, %xmm0
; X64-NEXT: retq
@@ -415,7 +389,6 @@ define bfloat @test_ctselect_bf16(i1 %cond, bfloat %a, bfloat %b) #0 {
; X32-NEXT: movzbl %cl, %ecx
; X32-NEXT: negl %ecx
; X32-NEXT: andl %eax, %ecx
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %esi, %ecx
; X32-NEXT: shll $16, %ecx
; X32-NEXT: movl %ecx, {{[0-9]+}}(%esp)
@@ -442,7 +415,6 @@ define bfloat @test_ctselect_bf16(i1 %cond, bfloat %a, bfloat %b) #0 {
; X32-NOCMOV-NEXT: movzbl %cl, %ecx
; X32-NOCMOV-NEXT: negl %ecx
; X32-NOCMOV-NEXT: andl %eax, %ecx
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %esi, %ecx
; X32-NOCMOV-NEXT: shll $16, %ecx
; X32-NOCMOV-NEXT: movl %ecx, {{[0-9]+}}(%esp)
@@ -467,13 +439,11 @@ define fp128 @test_ctselect_f128(i1 %cond, fp128 %a, fp128 %b) #0 {
; X64-NEXT: movq -{{[0-9]+}}(%rsp), %rdx
; X64-NEXT: xorq %rax, %rdx
; X64-NEXT: andq %rdi, %rdx
-; X64-NEXT: #ARITH_FENCE
; X64-NEXT: xorq %rax, %rdx
; X64-NEXT: movq %rdx, -{{[0-9]+}}(%rsp)
; X64-NEXT: movq -{{[0-9]+}}(%rsp), %rax
; X64-NEXT: xorq %rcx, %rax
; X64-NEXT: andq %rdi, %rax
-; X64-NEXT: #ARITH_FENCE
; X64-NEXT: xorq %rcx, %rax
; X64-NEXT: movq %rax, -{{[0-9]+}}(%rsp)
; X64-NEXT: movaps -{{[0-9]+}}(%rsp), %xmm0
@@ -497,22 +467,18 @@ define fp128 @test_ctselect_f128(i1 %cond, fp128 %a, fp128 %b) #0 {
; X32-NEXT: movzbl %bl, %edi
; X32-NEXT: negl %edi
; X32-NEXT: andl %edi, %edx
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %ecx, %edx
; X32-NEXT: movl {{[0-9]+}}(%esp), %ebx
; X32-NEXT: xorl %ebp, %ebx
; X32-NEXT: andl %edi, %ebx
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %ebp, %ebx
; X32-NEXT: movl {{[0-9]+}}(%esp), %ebp
; X32-NEXT: xorl %esi, %ebp
; X32-NEXT: andl %edi, %ebp
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %esi, %ebp
; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
; X32-NEXT: xorl {{[0-9]+}}(%esp), %ecx
; X32-NEXT: andl %edi, %ecx
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl {{[0-9]+}}(%esp), %ecx
; X32-NEXT: movl %ecx, 12(%eax)
; X32-NEXT: movl %ebp, 8(%eax)
@@ -543,22 +509,18 @@ define fp128 @test_ctselect_f128(i1 %cond, fp128 %a, fp128 %b) #0 {
; X32-NOCMOV-NEXT: movzbl %bl, %edi
; X32-NOCMOV-NEXT: negl %edi
; X32-NOCMOV-NEXT: andl %edi, %edx
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %ecx, %edx
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebx
; X32-NOCMOV-NEXT: xorl %ebp, %ebx
; X32-NOCMOV-NEXT: andl %edi, %ebx
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %ebp, %ebx
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebp
; X32-NOCMOV-NEXT: xorl %esi, %ebp
; X32-NOCMOV-NEXT: andl %edi, %ebp
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %esi, %ebp
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %ecx
; X32-NOCMOV-NEXT: andl %edi, %ecx
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %ecx
; X32-NOCMOV-NEXT: movl %ecx, 12(%eax)
; X32-NOCMOV-NEXT: movl %ebp, 8(%eax)
@@ -588,14 +550,12 @@ define x86_fp80 @test_ctselect_f80(i1 %cond, x86_fp80 %a, x86_fp80 %b) #0 {
; X64-NEXT: movq -{{[0-9]+}}(%rsp), %rcx
; X64-NEXT: xorq %rax, %rcx
; X64-NEXT: andq %rdi, %rcx
-; X64-NEXT: #ARITH_FENCE
; X64-NEXT: xorq %rax, %rcx
; X64-NEXT: movq %rcx, -{{[0-9]+}}(%rsp)
; X64-NEXT: movzwl -{{[0-9]+}}(%rsp), %eax
; X64-NEXT: movzwl -{{[0-9]+}}(%rsp), %ecx
; X64-NEXT: xorl %eax, %ecx
; X64-NEXT: andl %ecx, %edi
-; X64-NEXT: #ARITH_FENCE
; X64-NEXT: xorl %eax, %edi
; X64-NEXT: movw %di, -{{[0-9]+}}(%rsp)
; X64-NEXT: fldt -{{[0-9]+}}(%rsp)
@@ -618,20 +578,17 @@ define x86_fp80 @test_ctselect_f80(i1 %cond, x86_fp80 %a, x86_fp80 %b) #0 {
; X32-NEXT: movl (%esp), %esi
; X32-NEXT: xorl %edx, %esi
; X32-NEXT: andl %eax, %esi
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %edx, %esi
; X32-NEXT: movl %esi, (%esp)
; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
; X32-NEXT: xorl %ecx, %edx
; X32-NEXT: andl %eax, %edx
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %ecx, %edx
; X32-NEXT: movl %edx, {{[0-9]+}}(%esp)
; X32-NEXT: movzwl {{[0-9]+}}(%esp), %ecx
; X32-NEXT: movzwl {{[0-9]+}}(%esp), %edx
; X32-NEXT: xorl %ecx, %edx
; X32-NEXT: andl %eax, %edx
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %ecx, %edx
; X32-NEXT: movw %dx, {{[0-9]+}}(%esp)
; X32-NEXT: fldt (%esp)
@@ -656,20 +613,17 @@ define x86_fp80 @test_ctselect_f80(i1 %cond, x86_fp80 %a, x86_fp80 %b) #0 {
; X32-NOCMOV-NEXT: movl (%esp), %esi
; X32-NOCMOV-NEXT: xorl %edx, %esi
; X32-NOCMOV-NEXT: andl %eax, %esi
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %edx, %esi
; X32-NOCMOV-NEXT: movl %esi, (%esp)
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
; X32-NOCMOV-NEXT: xorl %ecx, %edx
; X32-NOCMOV-NEXT: andl %eax, %edx
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %ecx, %edx
; X32-NOCMOV-NEXT: movl %edx, {{[0-9]+}}(%esp)
; X32-NOCMOV-NEXT: movzwl {{[0-9]+}}(%esp), %ecx
; X32-NOCMOV-NEXT: movzwl {{[0-9]+}}(%esp), %edx
; X32-NOCMOV-NEXT: xorl %ecx, %edx
; X32-NOCMOV-NEXT: andl %eax, %edx
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %ecx, %edx
; X32-NOCMOV-NEXT: movw %dx, {{[0-9]+}}(%esp)
; X32-NOCMOV-NEXT: fldt (%esp)
@@ -688,7 +642,6 @@ define ptr @test_ctselect_ptr(i1 %cond, ptr %a, ptr %b) #0 {
; X64-NEXT: andl $1, %eax
; X64-NEXT: negq %rax
; X64-NEXT: andq %rsi, %rax
-; X64-NEXT: #ARITH_FENCE
; X64-NEXT: xorq %rdx, %rax
; X64-NEXT: retq
;
@@ -702,7 +655,6 @@ define ptr @test_ctselect_ptr(i1 %cond, ptr %a, ptr %b) #0 {
; X32-NEXT: movzbl %al, %eax
; X32-NEXT: negl %eax
; X32-NEXT: andl %edx, %eax
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %ecx, %eax
; X32-NEXT: retl
;
@@ -716,7 +668,6 @@ define ptr @test_ctselect_ptr(i1 %cond, ptr %a, ptr %b) #0 {
; X32-NOCMOV-NEXT: movzbl %al, %eax
; X32-NOCMOV-NEXT: negl %eax
; X32-NOCMOV-NEXT: andl %edx, %eax
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %ecx, %eax
; X32-NOCMOV-NEXT: retl
%result = call ptr @llvm.ct.select.p0(i1 %cond, ptr %a, ptr %b)
@@ -728,27 +679,16 @@ define i32 @test_ctselect_const_true(i32 %a, i32 %b) #0 {
; X64-LABEL: test_ctselect_const_true:
; X64: # %bb.0:
; X64-NEXT: movl %edi, %eax
-; X64-NEXT: xorl %esi, %eax
-; X64-NEXT: #ARITH_FENCE
-; X64-NEXT: xorl %esi, %eax
; X64-NEXT: retq
;
; X32-LABEL: test_ctselect_const_true:
; X32: # %bb.0:
-; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
-; X32-NEXT: xorl %ecx, %eax
-; X32-NEXT: #ARITH_FENCE
-; X32-NEXT: xorl %ecx, %eax
; X32-NEXT: retl
;
; X32-NOCMOV-LABEL: test_ctselect_const_true:
; X32-NOCMOV: # %bb.0:
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
-; X32-NOCMOV-NEXT: xorl %ecx, %eax
-; X32-NOCMOV-NEXT: #ARITH_FENCE
-; X32-NOCMOV-NEXT: xorl %ecx, %eax
; X32-NOCMOV-NEXT: retl
%result = call i32 @llvm.ct.select.i32(i1 true, i32 %a, i32 %b)
ret i32 %result
@@ -757,23 +697,17 @@ define i32 @test_ctselect_const_true(i32 %a, i32 %b) #0 {
define i32 @test_ctselect_const_false(i32 %a, i32 %b) #0 {
; X64-LABEL: test_ctselect_const_false:
; X64: # %bb.0:
-; X64-NEXT: xorl %eax, %eax
-; X64-NEXT: #ARITH_FENCE
-; X64-NEXT: xorl %esi, %eax
+; X64-NEXT: movl %esi, %eax
; X64-NEXT: retq
;
; X32-LABEL: test_ctselect_const_false:
; X32: # %bb.0:
-; X32-NEXT: xorl %eax, %eax
-; X32-NEXT: #ARITH_FENCE
-; X32-NEXT: xorl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
; X32-NEXT: retl
;
; X32-NOCMOV-LABEL: test_ctselect_const_false:
; X32-NOCMOV: # %bb.0:
-; X32-NOCMOV-NEXT: xorl %eax, %eax
-; X32-NOCMOV-NEXT: #ARITH_FENCE
-; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
; X32-NOCMOV-NEXT: retl
%result = call i32 @llvm.ct.select.i32(i1 false, i32 %a, i32 %b)
ret i32 %result
@@ -789,7 +723,6 @@ define i32 @test_ctselect_icmp_eq(i32 %x, i32 %y, i32 %a, i32 %b) #0 {
; X64-NEXT: xorl %ecx, %edx
; X64-NEXT: negl %eax
; X64-NEXT: andl %edx, %eax
-; X64-NEXT: #ARITH_FENCE
; X64-NEXT: xorl %ecx, %eax
; X64-NEXT: retq
;
@@ -804,7 +737,6 @@ define i32 @test_ctselect_icmp_eq(i32 %x, i32 %y, i32 %a, i32 %b) #0 {
; X32-NEXT: xorl %ecx, %edx
; X32-NEXT: negl %eax
; X32-NEXT: andl %edx, %eax
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %ecx, %eax
; X32-NEXT: retl
;
@@ -819,7 +751,6 @@ define i32 @test_ctselect_icmp_eq(i32 %x, i32 %y, i32 %a, i32 %b) #0 {
; X32-NOCMOV-NEXT: xorl %ecx, %edx
; X32-NOCMOV-NEXT: negl %eax
; X32-NOCMOV-NEXT: andl %edx, %eax
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %ecx, %eax
; X32-NOCMOV-NEXT: retl
%cond = icmp eq i32 %x, %y
@@ -835,7 +766,6 @@ define i32 @test_ctselect_icmp_ult(i32 %x, i32 %y, i32 %a, i32 %b) #0 {
; X64-NEXT: cmpl %esi, %edi
; X64-NEXT: sbbl %eax, %eax
; X64-NEXT: andl %edx, %eax
-; X64-NEXT: #ARITH_FENCE
; X64-NEXT: xorl %ecx, %eax
; X64-NEXT: retq
;
@@ -849,7 +779,6 @@ define i32 @test_ctselect_icmp_ult(i32 %x, i32 %y, i32 %a, i32 %b) #0 {
; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
; X32-NEXT: xorl %ecx, %eax
; X32-NEXT: andl %edx, %eax
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %ecx, %eax
; X32-NEXT: retl
;
@@ -863,7 +792,6 @@ define i32 @test_ctselect_icmp_ult(i32 %x, i32 %y, i32 %a, i32 %b) #0 {
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
; X32-NOCMOV-NEXT: xorl %ecx, %eax
; X32-NOCMOV-NEXT: andl %edx, %eax
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %ecx, %eax
; X32-NOCMOV-NEXT: retl
%cond = icmp ult i32 %x, %y
@@ -874,14 +802,10 @@ define i32 @test_ctselect_icmp_ult(i32 %x, i32 %y, i32 %a, i32 %b) #0 {
define float @test_ctselect_fcmp_oeq(float %x, float %y, float %a, float %b) #0 {
; X64-LABEL: test_ctselect_fcmp_oeq:
; X64: # %bb.0:
-; X64-NEXT: movd %xmm3, %eax
; X64-NEXT: cmpeqss %xmm1, %xmm0
-; X64-NEXT: pxor %xmm3, %xmm2
-; X64-NEXT: pand %xmm0, %xmm2
-; X64-NEXT: movd %xmm2, %ecx
-; X64-NEXT: #ARITH_FENCE
-; X64-NEXT: xorl %eax, %ecx
-; X64-NEXT: movd %ecx, %xmm0
+; X64-NEXT: xorps %xmm3, %xmm2
+; X64-NEXT: andps %xmm2, %xmm0
+; X64-NEXT: xorps %xmm3, %xmm0
; X64-NEXT: retq
;
; X32-LABEL: test_ctselect_fcmp_oeq:
@@ -904,7 +828,6 @@ define float @test_ctselect_fcmp_oeq(float %x, float %y, float %a, float %b) #0
; X32-NEXT: movl (%esp), %edx
; X32-NEXT: xorl %eax, %edx
; X32-NEXT: andl %ecx, %edx
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %eax, %edx
; X32-NEXT: movl %edx, {{[0-9]+}}(%esp)
; X32-NEXT: flds {{[0-9]+}}(%esp)
@@ -933,7 +856,6 @@ define float @test_ctselect_fcmp_oeq(float %x, float %y, float %a, float %b) #0
; X32-NOCMOV-NEXT: movl (%esp), %edx
; X32-NOCMOV-NEXT: xorl %ecx, %edx
; X32-NOCMOV-NEXT: andl %eax, %edx
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %ecx, %edx
; X32-NOCMOV-NEXT: movl %edx, {{[0-9]+}}(%esp)
; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
@@ -954,7 +876,6 @@ define i32 @test_ctselect_load(i1 %cond, ptr %p1, ptr %p2) #0 {
; X64-NEXT: andl $1, %edi
; X64-NEXT: negl %edi
; X64-NEXT: andl %edi, %eax
-; X64-NEXT: #ARITH_FENCE
; X64-NEXT: xorl %ecx, %eax
; X64-NEXT: retq
;
@@ -970,7 +891,6 @@ define i32 @test_ctselect_load(i1 %cond, ptr %p1, ptr %p2) #0 {
; X32-NEXT: movzbl %al, %eax
; X32-NEXT: negl %eax
; X32-NEXT: andl %ecx, %eax
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %edx, %eax
; X32-NEXT: retl
;
@@ -986,7 +906,6 @@ define i32 @test_ctselect_load(i1 %cond, ptr %p1, ptr %p2) #0 {
; X32-NOCMOV-NEXT: movzbl %al, %eax
; X32-NOCMOV-NEXT: negl %eax
; X32-NOCMOV-NEXT: andl %ecx, %eax
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %edx, %eax
; X32-NOCMOV-NEXT: retl
%a = load i32, ptr %p1
@@ -1004,13 +923,11 @@ define i32 @test_ctselect_nested(i1 %cond1, i1 %cond2, i32 %a, i32 %b, i32 %c) #
; X64-NEXT: andl $1, %esi
; X64-NEXT: negl %esi
; X64-NEXT: andl %edx, %esi
-; X64-NEXT: #ARITH_FENCE
; X64-NEXT: xorl %r8d, %ecx
; X64-NEXT: xorl %esi, %ecx
; X64-NEXT: andl $1, %eax
; X64-NEXT: negl %eax
; X64-NEXT: andl %ecx, %eax
-; X64-NEXT: #ARITH_FENCE
; X64-NEXT: xorl %r8d, %eax
; X64-NEXT: retq
;
@@ -1029,13 +946,11 @@ define i32 @test_ctselect_nested(i1 %cond1, i1 %cond2, i32 %a, i32 %b, i32 %c) #
; X32-NEXT: movzbl %ah, %edi
; X32-NEXT: negl %edi
; X32-NEXT: andl %esi, %edi
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %ecx, %edx
; X32-NEXT: xorl %edi, %edx
; X32-NEXT: movzbl %al, %eax
; X32-NEXT: negl %eax
; X32-NEXT: andl %edx, %eax
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %ecx, %eax
; X32-NEXT: popl %esi
; X32-NEXT: popl %edi
@@ -1056,13 +971,11 @@ define i32 @test_ctselect_nested(i1 %cond1, i1 %cond2, i32 %a, i32 %b, i32 %c) #
; X32-NOCMOV-NEXT: movzbl %ah, %edi
; X32-NOCMOV-NEXT: negl %edi
; X32-NOCMOV-NEXT: andl %esi, %edi
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %ecx, %edx
; X32-NOCMOV-NEXT: xorl %edi, %edx
; X32-NOCMOV-NEXT: movzbl %al, %eax
; X32-NOCMOV-NEXT: negl %eax
; X32-NOCMOV-NEXT: andl %edx, %eax
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %ecx, %eax
; X32-NOCMOV-NEXT: popl %esi
; X32-NOCMOV-NEXT: popl %edi
@@ -1078,15 +991,12 @@ define i32 @test_ctselect_nested(i1 %cond1, i1 %cond2, i32 %a, i32 %b, i32 %c) #
define i32 @test_ctselect_nested_and_i1_to_i32(i1 %c0, i1 %c1, i32 %x, i32 %y) #0 {
; X64-LABEL: test_ctselect_nested_and_i1_to_i32:
; X64: # %bb.0:
-; X64-NEXT: andl %esi, %edi
-; X64-NEXT: andb $1, %dil
-; X64-NEXT: #ARITH_FENCE
-; X64-NEXT: andb $1, %dil
+; X64-NEXT: movl %edi, %eax
+; X64-NEXT: andl %esi, %eax
; X64-NEXT: xorl %ecx, %edx
-; X64-NEXT: movzbl %dil, %eax
+; X64-NEXT: andl $1, %eax
; X64-NEXT: negl %eax
; X64-NEXT: andl %edx, %eax
-; X64-NEXT: #ARITH_FENCE
; X64-NEXT: xorl %ecx, %eax
; X64-NEXT: retq
;
@@ -1096,14 +1006,11 @@ define i32 @test_ctselect_nested_and_i1_to_i32(i1 %c0, i1 %c1, i32 %x, i32 %y) #
; X32-NEXT: movzbl {{[0-9]+}}(%esp), %eax
; X32-NEXT: andb {{[0-9]+}}(%esp), %al
; X32-NEXT: andb $1, %al
-; X32-NEXT: #ARITH_FENCE
-; X32-NEXT: andb $1, %al
; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
; X32-NEXT: xorl %ecx, %edx
; X32-NEXT: movzbl %al, %eax
; X32-NEXT: negl %eax
; X32-NEXT: andl %edx, %eax
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %ecx, %eax
; X32-NEXT: retl
;
@@ -1113,14 +1020,11 @@ define i32 @test_ctselect_nested_and_i1_to_i32(i1 %c0, i1 %c1, i32 %x, i32 %y) #
; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %eax
; X32-NOCMOV-NEXT: andb {{[0-9]+}}(%esp), %al
; X32-NOCMOV-NEXT: andb $1, %al
-; X32-NOCMOV-NEXT: #ARITH_FENCE
-; X32-NOCMOV-NEXT: andb $1, %al
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
; X32-NOCMOV-NEXT: xorl %ecx, %edx
; X32-NOCMOV-NEXT: movzbl %al, %eax
; X32-NOCMOV-NEXT: negl %eax
; X32-NOCMOV-NEXT: andl %edx, %eax
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %ecx, %eax
; X32-NOCMOV-NEXT: retl
%inner = call i1 @llvm.ct.select.i1(i1 %c1, i1 true, i1 false)
@@ -1135,15 +1039,12 @@ define i32 @test_ctselect_nested_and_i1_to_i32(i1 %c0, i1 %c1, i32 %x, i32 %y) #
define i32 @test_ctselect_nested_or_i1_to_i32(i1 %c0, i1 %c1, i32 %x, i32 %y) #0 {
; X64-LABEL: test_ctselect_nested_or_i1_to_i32:
; X64: # %bb.0:
-; X64-NEXT: orl %esi, %edi
-; X64-NEXT: andb $1, %dil
-; X64-NEXT: #ARITH_FENCE
-; X64-NEXT: andb $1, %dil
+; X64-NEXT: movl %edi, %eax
+; X64-NEXT: orl %esi, %eax
; X64-NEXT: xorl %ecx, %edx
-; X64-NEXT: movzbl %dil, %eax
+; X64-NEXT: andl $1, %eax
; X64-NEXT: negl %eax
; X64-NEXT: andl %edx, %eax
-; X64-NEXT: #ARITH_FENCE
; X64-NEXT: xorl %ecx, %eax
; X64-NEXT: retq
;
@@ -1153,14 +1054,11 @@ define i32 @test_ctselect_nested_or_i1_to_i32(i1 %c0, i1 %c1, i32 %x, i32 %y) #0
; X32-NEXT: movzbl {{[0-9]+}}(%esp), %eax
; X32-NEXT: orb {{[0-9]+}}(%esp), %al
; X32-NEXT: andb $1, %al
-; X32-NEXT: #ARITH_FENCE
-; X32-NEXT: andb $1, %al
; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
; X32-NEXT: xorl %ecx, %edx
; X32-NEXT: movzbl %al, %eax
; X32-NEXT: negl %eax
; X32-NEXT: andl %edx, %eax
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %ecx, %eax
; X32-NEXT: retl
;
@@ -1170,14 +1068,11 @@ define i32 @test_ctselect_nested_or_i1_to_i32(i1 %c0, i1 %c1, i32 %x, i32 %y) #0
; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %eax
; X32-NOCMOV-NEXT: orb {{[0-9]+}}(%esp), %al
; X32-NOCMOV-NEXT: andb $1, %al
-; X32-NOCMOV-NEXT: #ARITH_FENCE
-; X32-NOCMOV-NEXT: andb $1, %al
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
; X32-NOCMOV-NEXT: xorl %ecx, %edx
; X32-NOCMOV-NEXT: movzbl %al, %eax
; X32-NOCMOV-NEXT: negl %eax
; X32-NOCMOV-NEXT: andl %edx, %eax
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %ecx, %eax
; X32-NOCMOV-NEXT: retl
%inner = call i1 @llvm.ct.select.i1(i1 %c1, i1 true, i1 false)
@@ -1194,16 +1089,13 @@ define i32 @test_ctselect_nested_or_i1_to_i32(i1 %c0, i1 %c1, i32 %x, i32 %y) #0
define i32 @test_ctselect_double_nested_and_i1(i1 %c0, i1 %c1, i1 %c2, i32 %x, i32 %y) #0 {
; X64-LABEL: test_ctselect_double_nested_and_i1:
; X64: # %bb.0:
-; X64-NEXT: andl %esi, %edi
-; X64-NEXT: andl %edx, %edi
-; X64-NEXT: andb $1, %dil
-; X64-NEXT: #ARITH_FENCE
-; X64-NEXT: andb $1, %dil
+; X64-NEXT: movl %edi, %eax
+; X64-NEXT: andl %esi, %eax
+; X64-NEXT: andl %edx, %eax
; X64-NEXT: xorl %r8d, %ecx
-; X64-NEXT: movzbl %dil, %eax
+; X64-NEXT: andl $1, %eax
; X64-NEXT: negl %eax
; X64-NEXT: andl %ecx, %eax
-; X64-NEXT: #ARITH_FENCE
; X64-NEXT: xorl %r8d, %eax
; X64-NEXT: retq
;
@@ -1214,14 +1106,11 @@ define i32 @test_ctselect_double_nested_and_i1(i1 %c0, i1 %c1, i1 %c2, i32 %x, i
; X32-NEXT: andb {{[0-9]+}}(%esp), %al
; X32-NEXT: andb {{[0-9]+}}(%esp), %al
; X32-NEXT: andb $1, %al
-; X32-NEXT: #ARITH_FENCE
-; X32-NEXT: andb $1, %al
; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
; X32-NEXT: xorl %ecx, %edx
; X32-NEXT: movzbl %al, %eax
; X32-NEXT: negl %eax
; X32-NEXT: andl %edx, %eax
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %ecx, %eax
; X32-NEXT: retl
;
@@ -1232,14 +1121,11 @@ define i32 @test_ctselect_double_nested_and_i1(i1 %c0, i1 %c1, i1 %c2, i32 %x, i
; X32-NOCMOV-NEXT: andb {{[0-9]+}}(%esp), %al
; X32-NOCMOV-NEXT: andb {{[0-9]+}}(%esp), %al
; X32-NOCMOV-NEXT: andb $1, %al
-; X32-NOCMOV-NEXT: #ARITH_FENCE
-; X32-NOCMOV-NEXT: andb $1, %al
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
; X32-NOCMOV-NEXT: xorl %ecx, %edx
; X32-NOCMOV-NEXT: movzbl %al, %eax
; X32-NOCMOV-NEXT: negl %eax
; X32-NOCMOV-NEXT: andl %edx, %eax
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %ecx, %eax
; X32-NOCMOV-NEXT: retl
%inner2 = call i1 @llvm.ct.select.i1(i1 %c2, i1 true, i1 false)
@@ -1260,7 +1146,6 @@ define i32 @test_ctselect_negated_cond(i1 %c, i32 %a, i32 %b) #0 {
; X64-NEXT: andl $1, %eax
; X64-NEXT: negl %eax
; X64-NEXT: andl %edx, %eax
-; X64-NEXT: #ARITH_FENCE
; X64-NEXT: xorl %esi, %eax
; X64-NEXT: retq
;
@@ -1274,7 +1159,6 @@ define i32 @test_ctselect_negated_cond(i1 %c, i32 %a, i32 %b) #0 {
; X32-NEXT: movzbl %al, %eax
; X32-NEXT: negl %eax
; X32-NEXT: andl %edx, %eax
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %ecx, %eax
; X32-NEXT: retl
;
@@ -1288,7 +1172,6 @@ define i32 @test_ctselect_negated_cond(i1 %c, i32 %a, i32 %b) #0 {
; X32-NOCMOV-NEXT: movzbl %al, %eax
; X32-NOCMOV-NEXT: negl %eax
; X32-NOCMOV-NEXT: andl %edx, %eax
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %ecx, %eax
; X32-NOCMOV-NEXT: retl
%not = xor i1 %c, true
@@ -1304,7 +1187,6 @@ define <16 x b8> @test_ctselect_v16b8(i1 %cond, <16 x b8> %a, <16 x b8> %b) #0 {
; X64-LABEL: test_ctselect_v16b8:
; X64: # %bb.0:
; X64-NEXT: andb $1, %dil
-; X64-NEXT: pxor %xmm1, %xmm0
; X64-NEXT: negb %dil
; X64-NEXT: movzbl %dil, %eax
; X64-NEXT: movd %eax, %xmm2
@@ -1312,8 +1194,8 @@ define <16 x b8> @test_ctselect_v16b8(i1 %cond, <16 x b8> %a, <16 x b8> %b) #0 {
; X64-NEXT: pshuflw {{.*#+}} xmm2 = xmm2[0,0,0,0,4,5,6,7]
; X64-NEXT: pshufd {{.*#+}} xmm2 = xmm2[0,1,0,1]
; X64-NEXT: pand %xmm2, %xmm0
-; X64-NEXT: #ARITH_FENCE
-; X64-NEXT: pxor %xmm1, %xmm0
+; X64-NEXT: pandn %xmm1, %xmm2
+; X64-NEXT: por %xmm2, %xmm0
; X64-NEXT: retq
;
; X32-LABEL: test_ctselect_v16b8:
@@ -1321,120 +1203,104 @@ define <16 x b8> @test_ctselect_v16b8(i1 %cond, <16 x b8> %a, <16 x b8> %b) #0 {
; X32-NEXT: pushl %ebx
; X32-NEXT: pushl %esi
; X32-NEXT: subl $12, %esp
-; X32-NEXT: movzbl {{[0-9]+}}(%esp), %ebx
-; X32-NEXT: andb $1, %bl
+; X32-NEXT: movb {{[0-9]+}}(%esp), %ah
+; X32-NEXT: andb $1, %ah
; X32-NEXT: movzbl {{[0-9]+}}(%esp), %edx
; X32-NEXT: movb {{[0-9]+}}(%esp), %ch
; X32-NEXT: movb {{[0-9]+}}(%esp), %dh
-; X32-NEXT: movb {{[0-9]+}}(%esp), %ah
+; X32-NEXT: movzbl {{[0-9]+}}(%esp), %ebx
; X32-NEXT: movb {{[0-9]+}}(%esp), %bh
; X32-NEXT: movb {{[0-9]+}}(%esp), %cl
; X32-NEXT: movb {{[0-9]+}}(%esp), %al
; X32-NEXT: xorb %cl, %al
-; X32-NEXT: negb %bl
-; X32-NEXT: andb %bl, %al
-; X32-NEXT: #ARITH_FENCE
+; X32-NEXT: negb %ah
+; X32-NEXT: andb %ah, %al
; X32-NEXT: xorb %cl, %al
; X32-NEXT: movb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
; X32-NEXT: movb {{[0-9]+}}(%esp), %al
; X32-NEXT: xorb %bh, %al
-; X32-NEXT: andb %bl, %al
-; X32-NEXT: #ARITH_FENCE
+; X32-NEXT: andb %ah, %al
; X32-NEXT: xorb %bh, %al
; X32-NEXT: movb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
; X32-NEXT: movb {{[0-9]+}}(%esp), %al
-; X32-NEXT: xorb %ah, %al
-; X32-NEXT: andb %bl, %al
-; X32-NEXT: #ARITH_FENCE
-; X32-NEXT: xorb %ah, %al
+; X32-NEXT: xorb %bl, %al
+; X32-NEXT: andb %ah, %al
+; X32-NEXT: xorb %bl, %al
; X32-NEXT: movb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
-; X32-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: movb {{[0-9]+}}(%esp), %al
; X32-NEXT: xorb %dh, %al
-; X32-NEXT: andb %bl, %al
-; X32-NEXT: #ARITH_FENCE
+; X32-NEXT: andb %ah, %al
; X32-NEXT: xorb %dh, %al
; X32-NEXT: movb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
-; X32-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: movb {{[0-9]+}}(%esp), %al
; X32-NEXT: xorb %ch, %al
-; X32-NEXT: andb %bl, %al
-; X32-NEXT: #ARITH_FENCE
+; X32-NEXT: andb %ah, %al
; X32-NEXT: xorb %ch, %al
; X32-NEXT: movb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
-; X32-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: movb {{[0-9]+}}(%esp), %al
; X32-NEXT: xorb %dl, %al
-; X32-NEXT: andb %bl, %al
-; X32-NEXT: #ARITH_FENCE
+; X32-NEXT: andb %ah, %al
; X32-NEXT: xorb %dl, %al
; X32-NEXT: movb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
-; X32-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: movb {{[0-9]+}}(%esp), %al
; X32-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
; X32-NEXT: xorb %cl, %al
-; X32-NEXT: andb %bl, %al
-; X32-NEXT: #ARITH_FENCE
+; X32-NEXT: andb %ah, %al
; X32-NEXT: xorb %cl, %al
; X32-NEXT: movb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
-; X32-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: movb {{[0-9]+}}(%esp), %al
; X32-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
; X32-NEXT: xorb %cl, %al
-; X32-NEXT: andb %bl, %al
-; X32-NEXT: #ARITH_FENCE
+; X32-NEXT: andb %ah, %al
+; X32-NEXT: xorb %cl, %al
+; X32-NEXT: movb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
+; X32-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: movb {{[0-9]+}}(%esp), %al
+; X32-NEXT: xorb %cl, %al
+; X32-NEXT: andb %ah, %al
; X32-NEXT: xorb %cl, %al
; X32-NEXT: movb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
-; X32-NEXT: movzbl {{[0-9]+}}(%esp), %eax
; X32-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
-; X32-NEXT: xorb %al, %cl
-; X32-NEXT: andb %bl, %cl
-; X32-NEXT: #ARITH_FENCE
-; X32-NEXT: xorb %al, %cl
-; X32-NEXT: movb %cl, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
-; X32-NEXT: movzbl {{[0-9]+}}(%esp), %eax
; X32-NEXT: movb {{[0-9]+}}(%esp), %bh
-; X32-NEXT: xorb %al, %bh
-; X32-NEXT: andb %bl, %bh
-; X32-NEXT: #ARITH_FENCE
-; X32-NEXT: xorb %al, %bh
-; X32-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: xorb %cl, %bh
+; X32-NEXT: andb %ah, %bh
+; X32-NEXT: xorb %cl, %bh
+; X32-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
+; X32-NEXT: movb {{[0-9]+}}(%esp), %bl
+; X32-NEXT: xorb %cl, %bl
+; X32-NEXT: andb %ah, %bl
+; X32-NEXT: xorb %cl, %bl
+; X32-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
; X32-NEXT: movb {{[0-9]+}}(%esp), %dh
-; X32-NEXT: xorb %al, %dh
-; X32-NEXT: andb %bl, %dh
-; X32-NEXT: #ARITH_FENCE
-; X32-NEXT: xorb %al, %dh
-; X32-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NEXT: xorb %cl, %dh
+; X32-NEXT: andb %ah, %dh
+; X32-NEXT: xorb %cl, %dh
+; X32-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
; X32-NEXT: movb {{[0-9]+}}(%esp), %ch
-; X32-NEXT: xorb %al, %ch
-; X32-NEXT: andb %bl, %ch
-; X32-NEXT: #ARITH_FENCE
-; X32-NEXT: xorb %al, %ch
-; X32-NEXT: movzbl {{[0-9]+}}(%esp), %eax
-; X32-NEXT: movb {{[0-9]+}}(%esp), %ah
-; X32-NEXT: xorb %al, %ah
-; X32-NEXT: andb %bl, %ah
-; X32-NEXT: #ARITH_FENCE
-; X32-NEXT: xorb %al, %ah
-; X32-NEXT: movb {{[0-9]+}}(%esp), %al
+; X32-NEXT: xorb %cl, %ch
+; X32-NEXT: andb %ah, %ch
+; X32-NEXT: xorb %cl, %ch
+; X32-NEXT: movb {{[0-9]+}}(%esp), %cl
; X32-NEXT: movb {{[0-9]+}}(%esp), %dl
-; X32-NEXT: xorb %al, %dl
-; X32-NEXT: andb %bl, %dl
-; X32-NEXT: #ARITH_FENCE
-; X32-NEXT: xorb %al, %dl
+; X32-NEXT: xorb %cl, %dl
+; X32-NEXT: andb %ah, %dl
+; X32-NEXT: xorb %cl, %dl
; X32-NEXT: movb {{[0-9]+}}(%esp), %al
; X32-NEXT: movb {{[0-9]+}}(%esp), %cl
; X32-NEXT: xorb %al, %cl
-; X32-NEXT: andb %bl, %cl
-; X32-NEXT: #ARITH_FENCE
+; X32-NEXT: andb %ah, %cl
; X32-NEXT: xorb %al, %cl
; X32-NEXT: movb {{[0-9]+}}(%esp), %al
; X32-NEXT: xorb {{[0-9]+}}(%esp), %al
-; X32-NEXT: andb %bl, %al
-; X32-NEXT: #ARITH_FENCE
+; X32-NEXT: andb %ah, %al
; X32-NEXT: xorb {{[0-9]+}}(%esp), %al
; X32-NEXT: movl {{[0-9]+}}(%esp), %esi
; X32-NEXT: movb %al, 15(%esi)
; X32-NEXT: movb %cl, 14(%esi)
; X32-NEXT: movb %dl, 13(%esi)
-; X32-NEXT: movb %ah, 12(%esi)
-; X32-NEXT: movb %ch, 11(%esi)
-; X32-NEXT: movb %dh, 10(%esi)
+; X32-NEXT: movb %ch, 12(%esi)
+; X32-NEXT: movb %dh, 11(%esi)
+; X32-NEXT: movb %bl, 10(%esi)
; X32-NEXT: movb %bh, 9(%esi)
; X32-NEXT: movzbl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 1-byte Folded Reload
; X32-NEXT: movb %al, 8(%esi)
@@ -1465,120 +1331,104 @@ define <16 x b8> @test_ctselect_v16b8(i1 %cond, <16 x b8> %a, <16 x b8> %b) #0 {
; X32-NOCMOV-NEXT: pushl %ebx
; X32-NOCMOV-NEXT: pushl %esi
; X32-NOCMOV-NEXT: subl $12, %esp
-; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %ebx
-; X32-NOCMOV-NEXT: andb $1, %bl
+; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %ah
+; X32-NOCMOV-NEXT: andb $1, %ah
; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %edx
; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %ch
; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %dh
-; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %ah
+; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %ebx
; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %bh
; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %cl
; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %al
; X32-NOCMOV-NEXT: xorb %cl, %al
-; X32-NOCMOV-NEXT: negb %bl
-; X32-NOCMOV-NEXT: andb %bl, %al
-; X32-NOCMOV-NEXT: #ARITH_FENCE
+; X32-NOCMOV-NEXT: negb %ah
+; X32-NOCMOV-NEXT: andb %ah, %al
; X32-NOCMOV-NEXT: xorb %cl, %al
; X32-NOCMOV-NEXT: movb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %al
; X32-NOCMOV-NEXT: xorb %bh, %al
-; X32-NOCMOV-NEXT: andb %bl, %al
-; X32-NOCMOV-NEXT: #ARITH_FENCE
+; X32-NOCMOV-NEXT: andb %ah, %al
; X32-NOCMOV-NEXT: xorb %bh, %al
; X32-NOCMOV-NEXT: movb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %al
-; X32-NOCMOV-NEXT: xorb %ah, %al
-; X32-NOCMOV-NEXT: andb %bl, %al
-; X32-NOCMOV-NEXT: #ARITH_FENCE
-; X32-NOCMOV-NEXT: xorb %ah, %al
+; X32-NOCMOV-NEXT: xorb %bl, %al
+; X32-NOCMOV-NEXT: andb %ah, %al
+; X32-NOCMOV-NEXT: xorb %bl, %al
; X32-NOCMOV-NEXT: movb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
-; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %al
; X32-NOCMOV-NEXT: xorb %dh, %al
-; X32-NOCMOV-NEXT: andb %bl, %al
-; X32-NOCMOV-NEXT: #ARITH_FENCE
+; X32-NOCMOV-NEXT: andb %ah, %al
; X32-NOCMOV-NEXT: xorb %dh, %al
; X32-NOCMOV-NEXT: movb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
-; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %al
; X32-NOCMOV-NEXT: xorb %ch, %al
-; X32-NOCMOV-NEXT: andb %bl, %al
-; X32-NOCMOV-NEXT: #ARITH_FENCE
+; X32-NOCMOV-NEXT: andb %ah, %al
; X32-NOCMOV-NEXT: xorb %ch, %al
; X32-NOCMOV-NEXT: movb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
-; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %al
; X32-NOCMOV-NEXT: xorb %dl, %al
-; X32-NOCMOV-NEXT: andb %bl, %al
-; X32-NOCMOV-NEXT: #ARITH_FENCE
+; X32-NOCMOV-NEXT: andb %ah, %al
; X32-NOCMOV-NEXT: xorb %dl, %al
; X32-NOCMOV-NEXT: movb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
-; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %al
; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
; X32-NOCMOV-NEXT: xorb %cl, %al
-; X32-NOCMOV-NEXT: andb %bl, %al
-; X32-NOCMOV-NEXT: #ARITH_FENCE
+; X32-NOCMOV-NEXT: andb %ah, %al
; X32-NOCMOV-NEXT: xorb %cl, %al
; X32-NOCMOV-NEXT: movb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
-; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %al
; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
; X32-NOCMOV-NEXT: xorb %cl, %al
-; X32-NOCMOV-NEXT: andb %bl, %al
-; X32-NOCMOV-NEXT: #ARITH_FENCE
+; X32-NOCMOV-NEXT: andb %ah, %al
+; X32-NOCMOV-NEXT: xorb %cl, %al
+; X32-NOCMOV-NEXT: movb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
+; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %al
+; X32-NOCMOV-NEXT: xorb %cl, %al
+; X32-NOCMOV-NEXT: andb %ah, %al
; X32-NOCMOV-NEXT: xorb %cl, %al
; X32-NOCMOV-NEXT: movb %al, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
-; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %eax
; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
-; X32-NOCMOV-NEXT: xorb %al, %cl
-; X32-NOCMOV-NEXT: andb %bl, %cl
-; X32-NOCMOV-NEXT: #ARITH_FENCE
-; X32-NOCMOV-NEXT: xorb %al, %cl
-; X32-NOCMOV-NEXT: movb %cl, {{[-0-9]+}}(%e{{[sb]}}p) # 1-byte Spill
-; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %eax
; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %bh
-; X32-NOCMOV-NEXT: xorb %al, %bh
-; X32-NOCMOV-NEXT: andb %bl, %bh
-; X32-NOCMOV-NEXT: #ARITH_FENCE
-; X32-NOCMOV-NEXT: xorb %al, %bh
-; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: xorb %cl, %bh
+; X32-NOCMOV-NEXT: andb %ah, %bh
+; X32-NOCMOV-NEXT: xorb %cl, %bh
+; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
+; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %bl
+; X32-NOCMOV-NEXT: xorb %cl, %bl
+; X32-NOCMOV-NEXT: andb %ah, %bl
+; X32-NOCMOV-NEXT: xorb %cl, %bl
+; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %dh
-; X32-NOCMOV-NEXT: xorb %al, %dh
-; X32-NOCMOV-NEXT: andb %bl, %dh
-; X32-NOCMOV-NEXT: #ARITH_FENCE
-; X32-NOCMOV-NEXT: xorb %al, %dh
-; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X32-NOCMOV-NEXT: xorb %cl, %dh
+; X32-NOCMOV-NEXT: andb %ah, %dh
+; X32-NOCMOV-NEXT: xorb %cl, %dh
+; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %ch
-; X32-NOCMOV-NEXT: xorb %al, %ch
-; X32-NOCMOV-NEXT: andb %bl, %ch
-; X32-NOCMOV-NEXT: #ARITH_FENCE
-; X32-NOCMOV-NEXT: xorb %al, %ch
-; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %eax
-; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %ah
-; X32-NOCMOV-NEXT: xorb %al, %ah
-; X32-NOCMOV-NEXT: andb %bl, %ah
-; X32-NOCMOV-NEXT: #ARITH_FENCE
-; X32-NOCMOV-NEXT: xorb %al, %ah
-; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %al
+; X32-NOCMOV-NEXT: xorb %cl, %ch
+; X32-NOCMOV-NEXT: andb %ah, %ch
+; X32-NOCMOV-NEXT: xorb %cl, %ch
+; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %cl
; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %dl
-; X32-NOCMOV-NEXT: xorb %al, %dl
-; X32-NOCMOV-NEXT: andb %bl, %dl
-; X32-NOCMOV-NEXT: #ARITH_FENCE
-; X32-NOCMOV-NEXT: xorb %al, %dl
+; X32-NOCMOV-NEXT: xorb %cl, %dl
+; X32-NOCMOV-NEXT: andb %ah, %dl
+; X32-NOCMOV-NEXT: xorb %cl, %dl
; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %al
; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %cl
; X32-NOCMOV-NEXT: xorb %al, %cl
-; X32-NOCMOV-NEXT: andb %bl, %cl
-; X32-NOCMOV-NEXT: #ARITH_FENCE
+; X32-NOCMOV-NEXT: andb %ah, %cl
; X32-NOCMOV-NEXT: xorb %al, %cl
; X32-NOCMOV-NEXT: movb {{[0-9]+}}(%esp), %al
; X32-NOCMOV-NEXT: xorb {{[0-9]+}}(%esp), %al
-; X32-NOCMOV-NEXT: andb %bl, %al
-; X32-NOCMOV-NEXT: #ARITH_FENCE
+; X32-NOCMOV-NEXT: andb %ah, %al
; X32-NOCMOV-NEXT: xorb {{[0-9]+}}(%esp), %al
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %esi
; X32-NOCMOV-NEXT: movb %al, 15(%esi)
; X32-NOCMOV-NEXT: movb %cl, 14(%esi)
; X32-NOCMOV-NEXT: movb %dl, 13(%esi)
-; X32-NOCMOV-NEXT: movb %ah, 12(%esi)
-; X32-NOCMOV-NEXT: movb %ch, 11(%esi)
-; X32-NOCMOV-NEXT: movb %dh, 10(%esi)
+; X32-NOCMOV-NEXT: movb %ch, 12(%esi)
+; X32-NOCMOV-NEXT: movb %dh, 11(%esi)
+; X32-NOCMOV-NEXT: movb %bl, 10(%esi)
; X32-NOCMOV-NEXT: movb %bh, 9(%esi)
; X32-NOCMOV-NEXT: movzbl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 1-byte Folded Reload
; X32-NOCMOV-NEXT: movb %al, 8(%esi)
@@ -1612,14 +1462,13 @@ define <16 x b8> @test_ctselect_v16b8(i1 %cond, <16 x b8> %a, <16 x b8> %b) #0 {
define <4 x i32> @test_ctselect_v4i32(i1 %cond, <4 x i32> %a, <4 x i32> %b) #0 {
; X64-LABEL: test_ctselect_v4i32:
; X64: # %bb.0:
-; X64-NEXT: pxor %xmm1, %xmm0
; X64-NEXT: andl $1, %edi
; X64-NEXT: negl %edi
; X64-NEXT: movd %edi, %xmm2
; X64-NEXT: pshufd {{.*#+}} xmm2 = xmm2[0,0,0,0]
; X64-NEXT: pand %xmm2, %xmm0
-; X64-NEXT: #ARITH_FENCE
-; X64-NEXT: pxor %xmm1, %xmm0
+; X64-NEXT: pandn %xmm1, %xmm2
+; X64-NEXT: por %xmm2, %xmm0
; X64-NEXT: retq
;
; X32-LABEL: test_ctselect_v4i32:
@@ -1639,22 +1488,18 @@ define <4 x i32> @test_ctselect_v4i32(i1 %cond, <4 x i32> %a, <4 x i32> %b) #0 {
; X32-NEXT: movzbl %bl, %edi
; X32-NEXT: negl %edi
; X32-NEXT: andl %edi, %edx
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %ecx, %edx
; X32-NEXT: movl {{[0-9]+}}(%esp), %ebx
; X32-NEXT: xorl %ebp, %ebx
; X32-NEXT: andl %edi, %ebx
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %ebp, %ebx
; X32-NEXT: movl {{[0-9]+}}(%esp), %ebp
; X32-NEXT: xorl %esi, %ebp
; X32-NEXT: andl %edi, %ebp
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %esi, %ebp
; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
; X32-NEXT: xorl {{[0-9]+}}(%esp), %ecx
; X32-NEXT: andl %edi, %ecx
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl {{[0-9]+}}(%esp), %ecx
; X32-NEXT: movl %ecx, 12(%eax)
; X32-NEXT: movl %ebp, 8(%eax)
@@ -1683,22 +1528,18 @@ define <4 x i32> @test_ctselect_v4i32(i1 %cond, <4 x i32> %a, <4 x i32> %b) #0 {
; X32-NOCMOV-NEXT: movzbl %bl, %edi
; X32-NOCMOV-NEXT: negl %edi
; X32-NOCMOV-NEXT: andl %edi, %edx
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %ecx, %edx
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebx
; X32-NOCMOV-NEXT: xorl %ebp, %ebx
; X32-NOCMOV-NEXT: andl %edi, %ebx
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %ebp, %ebx
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebp
; X32-NOCMOV-NEXT: xorl %esi, %ebp
; X32-NOCMOV-NEXT: andl %edi, %ebp
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %esi, %ebp
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %ecx
; X32-NOCMOV-NEXT: andl %edi, %ecx
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %ecx
; X32-NOCMOV-NEXT: movl %ecx, 12(%eax)
; X32-NOCMOV-NEXT: movl %ebp, 8(%eax)
@@ -1715,14 +1556,13 @@ define <4 x i32> @test_ctselect_v4i32(i1 %cond, <4 x i32> %a, <4 x i32> %b) #0 {
define <4 x float> @test_ctselect_v4f32(i1 %cond, <4 x float> %a, <4 x float> %b) #0 {
; X64-LABEL: test_ctselect_v4f32:
; X64: # %bb.0:
-; X64-NEXT: pxor %xmm1, %xmm0
; X64-NEXT: andl $1, %edi
; X64-NEXT: negl %edi
; X64-NEXT: movd %edi, %xmm2
; X64-NEXT: pshufd {{.*#+}} xmm2 = xmm2[0,0,0,0]
; X64-NEXT: pand %xmm2, %xmm0
-; X64-NEXT: #ARITH_FENCE
-; X64-NEXT: pxor %xmm1, %xmm0
+; X64-NEXT: pandn %xmm1, %xmm2
+; X64-NEXT: por %xmm2, %xmm0
; X64-NEXT: retq
;
; X32-LABEL: test_ctselect_v4f32:
@@ -1760,25 +1600,21 @@ define <4 x float> @test_ctselect_v4f32(i1 %cond, <4 x float> %a, <4 x float> %b
; X32-NEXT: movl {{[0-9]+}}(%esp), %ebp
; X32-NEXT: xorl %ebx, %ebp
; X32-NEXT: andl %edx, %ebp
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %ebx, %ebp
; X32-NEXT: movl %ebp, {{[0-9]+}}(%esp)
; X32-NEXT: movl {{[0-9]+}}(%esp), %ebx
; X32-NEXT: xorl %edi, %ebx
; X32-NEXT: andl %edx, %ebx
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %edi, %ebx
; X32-NEXT: movl %ebx, {{[0-9]+}}(%esp)
; X32-NEXT: movl {{[0-9]+}}(%esp), %edi
; X32-NEXT: xorl %esi, %edi
; X32-NEXT: andl %edx, %edi
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %esi, %edi
; X32-NEXT: movl %edi, {{[0-9]+}}(%esp)
; X32-NEXT: movl (%esp), %esi
; X32-NEXT: xorl %ecx, %esi
; X32-NEXT: andl %edx, %esi
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %ecx, %esi
; X32-NEXT: movl %esi, {{[0-9]+}}(%esp)
; X32-NEXT: flds {{[0-9]+}}(%esp)
@@ -1831,25 +1667,21 @@ define <4 x float> @test_ctselect_v4f32(i1 %cond, <4 x float> %a, <4 x float> %b
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebp
; X32-NOCMOV-NEXT: xorl %ebx, %ebp
; X32-NOCMOV-NEXT: andl %edx, %ebp
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %ebx, %ebp
; X32-NOCMOV-NEXT: movl %ebp, {{[0-9]+}}(%esp)
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebx
; X32-NOCMOV-NEXT: xorl %edi, %ebx
; X32-NOCMOV-NEXT: andl %edx, %ebx
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %edi, %ebx
; X32-NOCMOV-NEXT: movl %ebx, {{[0-9]+}}(%esp)
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edi
; X32-NOCMOV-NEXT: xorl %esi, %edi
; X32-NOCMOV-NEXT: andl %edx, %edi
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %esi, %edi
; X32-NOCMOV-NEXT: movl %edi, {{[0-9]+}}(%esp)
; X32-NOCMOV-NEXT: movl (%esp), %esi
; X32-NOCMOV-NEXT: xorl %ecx, %esi
; X32-NOCMOV-NEXT: andl %edx, %esi
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %ecx, %esi
; X32-NOCMOV-NEXT: movl %esi, {{[0-9]+}}(%esp)
; X32-NOCMOV-NEXT: flds {{[0-9]+}}(%esp)
@@ -1873,18 +1705,17 @@ define <4 x float> @test_ctselect_v4f32(i1 %cond, <4 x float> %a, <4 x float> %b
define <8 x i32> @test_ctselect_v8i32_avx(i1 %cond, <8 x i32> %a, <8 x i32> %b) #0 {
; X64-LABEL: test_ctselect_v8i32_avx:
; X64: # %bb.0:
-; X64-NEXT: pxor %xmm2, %xmm0
; X64-NEXT: andl $1, %edi
; X64-NEXT: negl %edi
; X64-NEXT: movd %edi, %xmm4
; X64-NEXT: pshufd {{.*#+}} xmm4 = xmm4[0,0,0,0]
+; X64-NEXT: movdqa %xmm4, %xmm5
+; X64-NEXT: pandn %xmm2, %xmm5
; X64-NEXT: pand %xmm4, %xmm0
-; X64-NEXT: #ARITH_FENCE
-; X64-NEXT: pxor %xmm2, %xmm0
-; X64-NEXT: pxor %xmm3, %xmm1
+; X64-NEXT: por %xmm5, %xmm0
; X64-NEXT: pand %xmm4, %xmm1
-; X64-NEXT: #ARITH_FENCE
-; X64-NEXT: pxor %xmm3, %xmm1
+; X64-NEXT: pandn %xmm3, %xmm4
+; X64-NEXT: por %xmm4, %xmm1
; X64-NEXT: retq
;
; X32-LABEL: test_ctselect_v8i32_avx:
@@ -1906,46 +1737,38 @@ define <8 x i32> @test_ctselect_v8i32_avx(i1 %cond, <8 x i32> %a, <8 x i32> %b)
; X32-NEXT: movzbl %dl, %edx
; X32-NEXT: negl %edx
; X32-NEXT: andl %edx, %eax
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %ecx, %eax
; X32-NEXT: movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
; X32-NEXT: xorl %esi, %eax
; X32-NEXT: andl %edx, %eax
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %esi, %eax
; X32-NEXT: movl %eax, (%esp) # 4-byte Spill
; X32-NEXT: movl {{[0-9]+}}(%esp), %esi
; X32-NEXT: xorl %edi, %esi
; X32-NEXT: andl %edx, %esi
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %edi, %esi
; X32-NEXT: movl {{[0-9]+}}(%esp), %edi
; X32-NEXT: xorl %ebp, %edi
; X32-NEXT: andl %edx, %edi
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %ebp, %edi
; X32-NEXT: movl {{[0-9]+}}(%esp), %ebp
; X32-NEXT: xorl %ebx, %ebp
; X32-NEXT: andl %edx, %ebp
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %ebx, %ebp
; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
; X32-NEXT: movl {{[0-9]+}}(%esp), %ebx
; X32-NEXT: xorl %eax, %ebx
; X32-NEXT: andl %edx, %ebx
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %eax, %ebx
; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
; X32-NEXT: xorl %eax, %ecx
; X32-NEXT: andl %edx, %ecx
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %eax, %ecx
; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
; X32-NEXT: xorl {{[0-9]+}}(%esp), %eax
; X32-NEXT: andl %edx, %eax
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl {{[0-9]+}}(%esp), %eax
; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
; X32-NEXT: movl %eax, 28(%edx)
@@ -1985,46 +1808,38 @@ define <8 x i32> @test_ctselect_v8i32_avx(i1 %cond, <8 x i32> %a, <8 x i32> %b)
; X32-NOCMOV-NEXT: movzbl %dl, %edx
; X32-NOCMOV-NEXT: negl %edx
; X32-NOCMOV-NEXT: andl %edx, %eax
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %ecx, %eax
; X32-NOCMOV-NEXT: movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
; X32-NOCMOV-NEXT: xorl %esi, %eax
; X32-NOCMOV-NEXT: andl %edx, %eax
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %esi, %eax
; X32-NOCMOV-NEXT: movl %eax, (%esp) # 4-byte Spill
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %esi
; X32-NOCMOV-NEXT: xorl %edi, %esi
; X32-NOCMOV-NEXT: andl %edx, %esi
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %edi, %esi
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edi
; X32-NOCMOV-NEXT: xorl %ebp, %edi
; X32-NOCMOV-NEXT: andl %edx, %edi
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %ebp, %edi
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebp
; X32-NOCMOV-NEXT: xorl %ebx, %ebp
; X32-NOCMOV-NEXT: andl %edx, %ebp
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %ebx, %ebp
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebx
; X32-NOCMOV-NEXT: xorl %eax, %ebx
; X32-NOCMOV-NEXT: andl %edx, %ebx
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %eax, %ebx
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
; X32-NOCMOV-NEXT: xorl %eax, %ecx
; X32-NOCMOV-NEXT: andl %edx, %ecx
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %eax, %ecx
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %eax
; X32-NOCMOV-NEXT: andl %edx, %eax
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %eax
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
; X32-NOCMOV-NEXT: movl %eax, 28(%edx)
@@ -2051,18 +1866,17 @@ define <8 x i32> @test_ctselect_v8i32_avx(i1 %cond, <8 x i32> %a, <8 x i32> %b)
define <8 x float> @test_ctselect_v8f32(i1 %cond, <8 x float> %a, <8 x float> %b) #0 {
; X64-LABEL: test_ctselect_v8f32:
; X64: # %bb.0:
-; X64-NEXT: pxor %xmm2, %xmm0
; X64-NEXT: andl $1, %edi
; X64-NEXT: negl %edi
; X64-NEXT: movd %edi, %xmm4
; X64-NEXT: pshufd {{.*#+}} xmm4 = xmm4[0,0,0,0]
+; X64-NEXT: movdqa %xmm4, %xmm5
+; X64-NEXT: pandn %xmm2, %xmm5
; X64-NEXT: pand %xmm4, %xmm0
-; X64-NEXT: #ARITH_FENCE
-; X64-NEXT: pxor %xmm2, %xmm0
-; X64-NEXT: pxor %xmm3, %xmm1
+; X64-NEXT: por %xmm5, %xmm0
; X64-NEXT: pand %xmm4, %xmm1
-; X64-NEXT: #ARITH_FENCE
-; X64-NEXT: pxor %xmm3, %xmm1
+; X64-NEXT: pandn %xmm3, %xmm4
+; X64-NEXT: por %xmm4, %xmm1
; X64-NEXT: retq
;
; X32-LABEL: test_ctselect_v8f32:
@@ -2118,7 +1932,6 @@ define <8 x float> @test_ctselect_v8f32(i1 %cond, <8 x float> %a, <8 x float> %b
; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
; X32-NEXT: xorl %ebx, %eax
; X32-NEXT: andl %edx, %eax
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %ebx, %eax
; X32-NEXT: movl {{[0-9]+}}(%esp), %ebx
; X32-NEXT: movl {{[0-9]+}}(%esp), %ebp
@@ -2127,45 +1940,38 @@ define <8 x float> @test_ctselect_v8f32(i1 %cond, <8 x float> %a, <8 x float> %b
; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
; X32-NEXT: xorl %ecx, %eax
; X32-NEXT: andl %edx, %eax
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %ecx, %eax
; X32-NEXT: movl %eax, {{[0-9]+}}(%esp)
; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
; X32-NEXT: xorl %ebp, %eax
; X32-NEXT: andl %edx, %eax
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %ebp, %eax
; X32-NEXT: movl %eax, {{[0-9]+}}(%esp)
; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
; X32-NEXT: xorl %ebx, %eax
; X32-NEXT: andl %edx, %eax
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %ebx, %eax
; X32-NEXT: movl %eax, {{[0-9]+}}(%esp)
; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
; X32-NEXT: xorl %edi, %eax
; X32-NEXT: andl %edx, %eax
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %edi, %eax
; X32-NEXT: movl %eax, {{[0-9]+}}(%esp)
; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
; X32-NEXT: xorl %esi, %eax
; X32-NEXT: andl %edx, %eax
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %esi, %eax
; X32-NEXT: movl %eax, {{[0-9]+}}(%esp)
; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
; X32-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %ecx # 4-byte Reload
; X32-NEXT: xorl %ecx, %eax
; X32-NEXT: andl %edx, %eax
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %ecx, %eax
; X32-NEXT: movl %eax, {{[0-9]+}}(%esp)
; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
; X32-NEXT: movl (%esp), %ecx # 4-byte Reload
; X32-NEXT: xorl %ecx, %eax
; X32-NEXT: andl %edx, %eax
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %ecx, %eax
; X32-NEXT: movl %eax, {{[0-9]+}}(%esp)
; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
@@ -2247,7 +2053,6 @@ define <8 x float> @test_ctselect_v8f32(i1 %cond, <8 x float> %a, <8 x float> %b
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
; X32-NOCMOV-NEXT: xorl %ebx, %eax
; X32-NOCMOV-NEXT: andl %edx, %eax
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %ebx, %eax
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebx
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebp
@@ -2256,45 +2061,38 @@ define <8 x float> @test_ctselect_v8f32(i1 %cond, <8 x float> %a, <8 x float> %b
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
; X32-NOCMOV-NEXT: xorl %ecx, %eax
; X32-NOCMOV-NEXT: andl %edx, %eax
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %ecx, %eax
; X32-NOCMOV-NEXT: movl %eax, {{[0-9]+}}(%esp)
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
; X32-NOCMOV-NEXT: xorl %ebp, %eax
; X32-NOCMOV-NEXT: andl %edx, %eax
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %ebp, %eax
; X32-NOCMOV-NEXT: movl %eax, {{[0-9]+}}(%esp)
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
; X32-NOCMOV-NEXT: xorl %ebx, %eax
; X32-NOCMOV-NEXT: andl %edx, %eax
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %ebx, %eax
; X32-NOCMOV-NEXT: movl %eax, {{[0-9]+}}(%esp)
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
; X32-NOCMOV-NEXT: xorl %edi, %eax
; X32-NOCMOV-NEXT: andl %edx, %eax
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %edi, %eax
; X32-NOCMOV-NEXT: movl %eax, {{[0-9]+}}(%esp)
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
; X32-NOCMOV-NEXT: xorl %esi, %eax
; X32-NOCMOV-NEXT: andl %edx, %eax
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %esi, %eax
; X32-NOCMOV-NEXT: movl %eax, {{[0-9]+}}(%esp)
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
; X32-NOCMOV-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %ecx # 4-byte Reload
; X32-NOCMOV-NEXT: xorl %ecx, %eax
; X32-NOCMOV-NEXT: andl %edx, %eax
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %ecx, %eax
; X32-NOCMOV-NEXT: movl %eax, {{[0-9]+}}(%esp)
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
; X32-NOCMOV-NEXT: movl (%esp), %ecx # 4-byte Reload
; X32-NOCMOV-NEXT: xorl %ecx, %eax
; X32-NOCMOV-NEXT: andl %edx, %eax
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %ecx, %eax
; X32-NOCMOV-NEXT: movl %eax, {{[0-9]+}}(%esp)
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
@@ -2329,15 +2127,14 @@ define <8 x float> @test_ctselect_v8f32(i1 %cond, <8 x float> %a, <8 x float> %b
define <8 x half> @test_ctselect_v8f16(i1 %cond, <8 x half> %a, <8 x half> %b) #0 {
; X64-LABEL: test_ctselect_v8f16:
; X64: # %bb.0:
-; X64-NEXT: pxor %xmm1, %xmm0
; X64-NEXT: andl $1, %edi
; X64-NEXT: negl %edi
; X64-NEXT: movd %edi, %xmm2
; X64-NEXT: pshuflw {{.*#+}} xmm2 = xmm2[0,0,0,0,4,5,6,7]
; X64-NEXT: pshufd {{.*#+}} xmm2 = xmm2[0,1,0,1]
; X64-NEXT: pand %xmm2, %xmm0
-; X64-NEXT: #ARITH_FENCE
-; X64-NEXT: pxor %xmm1, %xmm0
+; X64-NEXT: pandn %xmm1, %xmm2
+; X64-NEXT: por %xmm2, %xmm0
; X64-NEXT: retq
;
; X32-LABEL: test_ctselect_v8f16:
@@ -2347,60 +2144,52 @@ define <8 x half> @test_ctselect_v8f16(i1 %cond, <8 x half> %a, <8 x half> %b) #
; X32-NEXT: pushl %edi
; X32-NEXT: pushl %esi
; X32-NEXT: subl $12, %esp
+; X32-NEXT: movl {{[0-9]+}}(%esp), %edi
; X32-NEXT: movl {{[0-9]+}}(%esp), %ebp
; X32-NEXT: movl {{[0-9]+}}(%esp), %ebx
-; X32-NEXT: movl {{[0-9]+}}(%esp), %edi
-; X32-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
-; X32-NEXT: andb $1, %cl
; X32-NEXT: movl {{[0-9]+}}(%esp), %esi
+; X32-NEXT: movzbl {{[0-9]+}}(%esp), %edx
+; X32-NEXT: andb $1, %dl
+; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
; X32-NEXT: movzwl {{[0-9]+}}(%esp), %eax
-; X32-NEXT: xorw %si, %ax
-; X32-NEXT: movzbl %cl, %edx
+; X32-NEXT: xorw %cx, %ax
+; X32-NEXT: movzbl %dl, %edx
; X32-NEXT: negl %edx
; X32-NEXT: andl %edx, %eax
-; X32-NEXT: #ARITH_FENCE
-; X32-NEXT: xorl %esi, %eax
+; X32-NEXT: xorl %ecx, %eax
; X32-NEXT: movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
; X32-NEXT: movzwl {{[0-9]+}}(%esp), %eax
-; X32-NEXT: xorw %di, %ax
+; X32-NEXT: xorw %si, %ax
; X32-NEXT: andl %edx, %eax
-; X32-NEXT: #ARITH_FENCE
-; X32-NEXT: xorl %edi, %eax
+; X32-NEXT: xorl %esi, %eax
; X32-NEXT: movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
; X32-NEXT: movzwl {{[0-9]+}}(%esp), %eax
; X32-NEXT: xorw %bx, %ax
; X32-NEXT: andl %edx, %eax
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %ebx, %eax
; X32-NEXT: movl %eax, (%esp) # 4-byte Spill
; X32-NEXT: movzwl {{[0-9]+}}(%esp), %ebx
; X32-NEXT: xorw %bp, %bx
; X32-NEXT: andl %edx, %ebx
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %ebp, %ebx
; X32-NEXT: movzwl {{[0-9]+}}(%esp), %ebp
-; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
-; X32-NEXT: xorw %ax, %bp
+; X32-NEXT: xorw %di, %bp
; X32-NEXT: andl %edx, %ebp
-; X32-NEXT: #ARITH_FENCE
-; X32-NEXT: xorl %eax, %ebp
+; X32-NEXT: xorl %edi, %ebp
; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
; X32-NEXT: movzwl {{[0-9]+}}(%esp), %edi
; X32-NEXT: xorw %ax, %di
; X32-NEXT: andl %edx, %edi
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %eax, %edi
; X32-NEXT: movl {{[0-9]+}}(%esp), %eax
; X32-NEXT: movzwl {{[0-9]+}}(%esp), %ecx
; X32-NEXT: xorw %ax, %cx
; X32-NEXT: andl %edx, %ecx
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %eax, %ecx
; X32-NEXT: movzwl {{[0-9]+}}(%esp), %eax
; X32-NEXT: movl {{[0-9]+}}(%esp), %esi
; X32-NEXT: xorw %si, %ax
; X32-NEXT: andl %edx, %eax
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl {{[0-9]+}}(%esp), %eax
; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
; X32-NEXT: movw %ax, 14(%edx)
@@ -2429,60 +2218,52 @@ define <8 x half> @test_ctselect_v8f16(i1 %cond, <8 x half> %a, <8 x half> %b) #
; X32-NOCMOV-NEXT: pushl %edi
; X32-NOCMOV-NEXT: pushl %esi
; X32-NOCMOV-NEXT: subl $12, %esp
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edi
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebp
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ebx
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edi
-; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
-; X32-NOCMOV-NEXT: andb $1, %cl
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %esi
+; X32-NOCMOV-NEXT: movzbl {{[0-9]+}}(%esp), %edx
+; X32-NOCMOV-NEXT: andb $1, %dl
+; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
; X32-NOCMOV-NEXT: movzwl {{[0-9]+}}(%esp), %eax
-; X32-NOCMOV-NEXT: xorw %si, %ax
-; X32-NOCMOV-NEXT: movzbl %cl, %edx
+; X32-NOCMOV-NEXT: xorw %cx, %ax
+; X32-NOCMOV-NEXT: movzbl %dl, %edx
; X32-NOCMOV-NEXT: negl %edx
; X32-NOCMOV-NEXT: andl %edx, %eax
-; X32-NOCMOV-NEXT: #ARITH_FENCE
-; X32-NOCMOV-NEXT: xorl %esi, %eax
+; X32-NOCMOV-NEXT: xorl %ecx, %eax
; X32-NOCMOV-NEXT: movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
; X32-NOCMOV-NEXT: movzwl {{[0-9]+}}(%esp), %eax
-; X32-NOCMOV-NEXT: xorw %di, %ax
+; X32-NOCMOV-NEXT: xorw %si, %ax
; X32-NOCMOV-NEXT: andl %edx, %eax
-; X32-NOCMOV-NEXT: #ARITH_FENCE
-; X32-NOCMOV-NEXT: xorl %edi, %eax
+; X32-NOCMOV-NEXT: xorl %esi, %eax
; X32-NOCMOV-NEXT: movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
; X32-NOCMOV-NEXT: movzwl {{[0-9]+}}(%esp), %eax
; X32-NOCMOV-NEXT: xorw %bx, %ax
; X32-NOCMOV-NEXT: andl %edx, %eax
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %ebx, %eax
; X32-NOCMOV-NEXT: movl %eax, (%esp) # 4-byte Spill
; X32-NOCMOV-NEXT: movzwl {{[0-9]+}}(%esp), %ebx
; X32-NOCMOV-NEXT: xorw %bp, %bx
; X32-NOCMOV-NEXT: andl %edx, %ebx
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %ebp, %ebx
; X32-NOCMOV-NEXT: movzwl {{[0-9]+}}(%esp), %ebp
-; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
-; X32-NOCMOV-NEXT: xorw %ax, %bp
+; X32-NOCMOV-NEXT: xorw %di, %bp
; X32-NOCMOV-NEXT: andl %edx, %ebp
-; X32-NOCMOV-NEXT: #ARITH_FENCE
-; X32-NOCMOV-NEXT: xorl %eax, %ebp
+; X32-NOCMOV-NEXT: xorl %edi, %ebp
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
; X32-NOCMOV-NEXT: movzwl {{[0-9]+}}(%esp), %edi
; X32-NOCMOV-NEXT: xorw %ax, %di
; X32-NOCMOV-NEXT: andl %edx, %edi
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %eax, %edi
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %eax
; X32-NOCMOV-NEXT: movzwl {{[0-9]+}}(%esp), %ecx
; X32-NOCMOV-NEXT: xorw %ax, %cx
; X32-NOCMOV-NEXT: andl %edx, %ecx
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %eax, %ecx
; X32-NOCMOV-NEXT: movzwl {{[0-9]+}}(%esp), %eax
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %esi
; X32-NOCMOV-NEXT: xorw %si, %ax
; X32-NOCMOV-NEXT: andl %edx, %eax
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl {{[0-9]+}}(%esp), %eax
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
; X32-NOCMOV-NEXT: movw %ax, 14(%edx)
@@ -2525,13 +2306,11 @@ define <8 x bfloat> @test_ctselect_v8bf16(i1 %cond, <8 x bfloat> %a, <8 x bfloat
; X64-NEXT: andl $1, %edi
; X64-NEXT: negl %edi
; X64-NEXT: andl %edi, %r10d
-; X64-NEXT: #ARITH_FENCE
; X64-NEXT: xorl %r8d, %r10d
; X64-NEXT: movq %r9, %r8
; X64-NEXT: shll $16, %r10d
; X64-NEXT: xorl %esi, %r9d
; X64-NEXT: andl %edi, %r9d
-; X64-NEXT: #ARITH_FENCE
; X64-NEXT: xorl %esi, %r9d
; X64-NEXT: movzwl %r9w, %r9d
; X64-NEXT: orl %r10d, %r9d
@@ -2540,13 +2319,11 @@ define <8 x bfloat> @test_ctselect_v8bf16(i1 %cond, <8 x bfloat> %a, <8 x bfloat
; X64-NEXT: shrq $48, %r8
; X64-NEXT: xorl %r10d, %r8d
; X64-NEXT: andl %edi, %r8d
-; X64-NEXT: #ARITH_FENCE
; X64-NEXT: xorl %r10d, %r8d
; X64-NEXT: shrq $32, %rsi
; X64-NEXT: shrq $32, %rdx
; X64-NEXT: xorl %esi, %edx
; X64-NEXT: andl %edi, %edx
-; X64-NEXT: #ARITH_FENCE
; X64-NEXT: xorl %esi, %edx
; X64-NEXT: movq %rcx, %rsi
; X64-NEXT: shll $16, %r8d
@@ -2560,13 +2337,11 @@ define <8 x bfloat> @test_ctselect_v8bf16(i1 %cond, <8 x bfloat> %a, <8 x bfloat
; X64-NEXT: shrl $16, %r9d
; X64-NEXT: xorl %r8d, %r9d
; X64-NEXT: andl %edi, %r9d
-; X64-NEXT: #ARITH_FENCE
; X64-NEXT: xorl %r8d, %r9d
; X64-NEXT: movq %rcx, %r8
; X64-NEXT: shll $16, %r9d
; X64-NEXT: xorl %eax, %ecx
; X64-NEXT: andl %edi, %ecx
-; X64-NEXT: #ARITH_FENCE
; X64-NEXT: xorl %eax, %ecx
; X64-NEXT: movzwl %cx, %ecx
; X64-NEXT: orl %r9d, %ecx
@@ -2575,13 +2350,11 @@ define <8 x bfloat> @test_ctselect_v8bf16(i1 %cond, <8 x bfloat> %a, <8 x bfloat
; X64-NEXT: shrq $48, %r8
; X64-NEXT: xorl %r9d, %r8d
; X64-NEXT: andl %edi, %r8d
-; X64-NEXT: #ARITH_FENCE
; X64-NEXT: xorl %r9d, %r8d
; X64-NEXT: shrq $32, %rax
; X64-NEXT: shrq $32, %rsi
; X64-NEXT: xorl %eax, %esi
; X64-NEXT: andl %edi, %esi
-; X64-NEXT: #ARITH_FENCE
; X64-NEXT: xorl %eax, %esi
; X64-NEXT: shll $16, %r8d
; X64-NEXT: movzwl %si, %eax
@@ -2683,49 +2456,41 @@ define <8 x bfloat> @test_ctselect_v8bf16(i1 %cond, <8 x bfloat> %a, <8 x bfloat
; X32-NEXT: movzbl %cl, %ecx
; X32-NEXT: negl %ecx
; X32-NEXT: andl %ecx, %esi
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %edx, %esi
; X32-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %edx # 4-byte Reload
; X32-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 4-byte Reload
; X32-NEXT: xorl %eax, %edx
; X32-NEXT: andl %ecx, %edx
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %eax, %edx
; X32-NEXT: movl %edx, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
; X32-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %edx # 4-byte Reload
; X32-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 4-byte Reload
; X32-NEXT: xorl %eax, %edx
; X32-NEXT: andl %ecx, %edx
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %eax, %edx
; X32-NEXT: movl %edx, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
; X32-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %edx # 4-byte Reload
; X32-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 4-byte Reload
; X32-NEXT: xorl %eax, %edx
; X32-NEXT: andl %ecx, %edx
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %eax, %edx
; X32-NEXT: movl %edx, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
; X32-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 4-byte Reload
; X32-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %edx # 4-byte Reload
; X32-NEXT: xorl %edx, %eax
; X32-NEXT: andl %ecx, %eax
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %edx, %eax
; X32-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %edx # 4-byte Reload
; X32-NEXT: xorl %edx, %ebp
; X32-NEXT: andl %ecx, %ebp
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %edx, %ebp
; X32-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %edx # 4-byte Reload
; X32-NEXT: xorl %edx, %edi
; X32-NEXT: andl %ecx, %edi
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %edx, %edi
; X32-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %edx # 4-byte Reload
; X32-NEXT: xorl %edx, %ebx
; X32-NEXT: andl %ecx, %ebx
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %edx, %ebx
; X32-NEXT: movl {{[0-9]+}}(%esp), %ecx
; X32-NEXT: movw %bx, 14(%ecx)
@@ -2837,49 +2602,41 @@ define <8 x bfloat> @test_ctselect_v8bf16(i1 %cond, <8 x bfloat> %a, <8 x bfloat
; X32-NOCMOV-NEXT: movzbl %cl, %ecx
; X32-NOCMOV-NEXT: negl %ecx
; X32-NOCMOV-NEXT: andl %ecx, %esi
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %edx, %esi
; X32-NOCMOV-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %edx # 4-byte Reload
; X32-NOCMOV-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 4-byte Reload
; X32-NOCMOV-NEXT: xorl %eax, %edx
; X32-NOCMOV-NEXT: andl %ecx, %edx
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %eax, %edx
; X32-NOCMOV-NEXT: movl %edx, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
; X32-NOCMOV-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %edx # 4-byte Reload
; X32-NOCMOV-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 4-byte Reload
; X32-NOCMOV-NEXT: xorl %eax, %edx
; X32-NOCMOV-NEXT: andl %ecx, %edx
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %eax, %edx
; X32-NOCMOV-NEXT: movl %edx, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
; X32-NOCMOV-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %edx # 4-byte Reload
; X32-NOCMOV-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 4-byte Reload
; X32-NOCMOV-NEXT: xorl %eax, %edx
; X32-NOCMOV-NEXT: andl %ecx, %edx
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %eax, %edx
; X32-NOCMOV-NEXT: movl %edx, {{[-0-9]+}}(%e{{[sb]}}p) # 4-byte Spill
; X32-NOCMOV-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %eax # 4-byte Reload
; X32-NOCMOV-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %edx # 4-byte Reload
; X32-NOCMOV-NEXT: xorl %edx, %eax
; X32-NOCMOV-NEXT: andl %ecx, %eax
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %edx, %eax
; X32-NOCMOV-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %edx # 4-byte Reload
; X32-NOCMOV-NEXT: xorl %edx, %ebp
; X32-NOCMOV-NEXT: andl %ecx, %ebp
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %edx, %ebp
; X32-NOCMOV-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %edx # 4-byte Reload
; X32-NOCMOV-NEXT: xorl %edx, %edi
; X32-NOCMOV-NEXT: andl %ecx, %edi
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %edx, %edi
; X32-NOCMOV-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %edx # 4-byte Reload
; X32-NOCMOV-NEXT: xorl %edx, %ebx
; X32-NOCMOV-NEXT: andl %ecx, %ebx
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %edx, %ebx
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %ecx
; X32-NOCMOV-NEXT: movw %bx, 14(%ecx)
@@ -2910,8 +2667,7 @@ define float @test_ctselect_f32_nan_inf(i1 %cond) #0 {
; X64-NEXT: andl $1, %edi
; X64-NEXT: negl %edi
; X64-NEXT: andl $4194304, %edi # imm = 0x400000
-; X64-NEXT: #ARITH_FENCE
-; X64-NEXT: xorl $2139095040, %edi # imm = 0x7F800000
+; X64-NEXT: orl $2139095040, %edi # imm = 0x7F800000
; X64-NEXT: movd %edi, %xmm0
; X64-NEXT: retq
;
@@ -2923,8 +2679,7 @@ define float @test_ctselect_f32_nan_inf(i1 %cond) #0 {
; X32-NEXT: movzbl %al, %eax
; X32-NEXT: negl %eax
; X32-NEXT: andl $4194304, %eax # imm = 0x400000
-; X32-NEXT: #ARITH_FENCE
-; X32-NEXT: xorl $2139095040, %eax # imm = 0x7F800000
+; X32-NEXT: orl $2139095040, %eax # imm = 0x7F800000
; X32-NEXT: movl %eax, (%esp)
; X32-NEXT: flds (%esp)
; X32-NEXT: popl %eax
@@ -2938,8 +2693,7 @@ define float @test_ctselect_f32_nan_inf(i1 %cond) #0 {
; X32-NOCMOV-NEXT: movzbl %al, %eax
; X32-NOCMOV-NEXT: negl %eax
; X32-NOCMOV-NEXT: andl $4194304, %eax # imm = 0x400000
-; X32-NOCMOV-NEXT: #ARITH_FENCE
-; X32-NOCMOV-NEXT: xorl $2139095040, %eax # imm = 0x7F800000
+; X32-NOCMOV-NEXT: orl $2139095040, %eax # imm = 0x7F800000
; X32-NOCMOV-NEXT: movl %eax, (%esp)
; X32-NOCMOV-NEXT: flds (%esp)
; X32-NOCMOV-NEXT: popl %eax
@@ -2956,9 +2710,8 @@ define double @test_ctselect_f64_nan_inf(i1 %cond) #0 {
; X64-NEXT: negq %rdi
; X64-NEXT: movabsq $2251799813685248, %rax # imm = 0x8000000000000
; X64-NEXT: andq %rdi, %rax
-; X64-NEXT: #ARITH_FENCE
; X64-NEXT: movabsq $9218868437227405312, %rcx # imm = 0x7FF0000000000000
-; X64-NEXT: xorq %rax, %rcx
+; X64-NEXT: orq %rax, %rcx
; X64-NEXT: movq %rcx, %xmm0
; X64-NEXT: retq
;
@@ -2979,13 +2732,11 @@ define double @test_ctselect_f64_nan_inf(i1 %cond) #0 {
; X32-NEXT: movl (%esp), %esi
; X32-NEXT: xorl %edx, %esi
; X32-NEXT: andl %eax, %esi
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %edx, %esi
; X32-NEXT: movl %esi, (%esp)
; X32-NEXT: movl {{[0-9]+}}(%esp), %edx
; X32-NEXT: xorl %ecx, %edx
; X32-NEXT: andl %eax, %edx
-; X32-NEXT: #ARITH_FENCE
; X32-NEXT: xorl %ecx, %edx
; X32-NEXT: movl %edx, {{[0-9]+}}(%esp)
; X32-NEXT: fldl (%esp)
@@ -3010,13 +2761,11 @@ define double @test_ctselect_f64_nan_inf(i1 %cond) #0 {
; X32-NOCMOV-NEXT: movl (%esp), %esi
; X32-NOCMOV-NEXT: xorl %edx, %esi
; X32-NOCMOV-NEXT: andl %eax, %esi
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %edx, %esi
; X32-NOCMOV-NEXT: movl %esi, (%esp)
; X32-NOCMOV-NEXT: movl {{[0-9]+}}(%esp), %edx
; X32-NOCMOV-NEXT: xorl %ecx, %edx
; X32-NOCMOV-NEXT: andl %eax, %edx
-; X32-NOCMOV-NEXT: #ARITH_FENCE
; X32-NOCMOV-NEXT: xorl %ecx, %edx
; X32-NOCMOV-NEXT: movl %edx, {{[0-9]+}}(%esp)
; X32-NOCMOV-NEXT: fldl (%esp)
More information about the llvm-commits
mailing list