[llvm] [X86][APX] Optimize usub.sat(X, 1) to cmp+adc with NDD (PR #208475)
via llvm-commits
llvm-commits at lists.llvm.org
Fri Jul 10 00:07:30 PDT 2026
https://github.com/AntonyCJ30 updated https://github.com/llvm/llvm-project/pull/208475
>From 2726d7973219552a8a8180348412704e69ef7730 Mon Sep 17 00:00:00 2001
From: AntonyCJ30 <cj6186609 at gmail@gmail.com>
Date: Thu, 9 Jul 2026 19:46:47 +0530
Subject: [PATCH] [X86][APX] Optimize usub.sat(X,1) to cmp+adc with NDD
---
llvm/lib/Target/X86/X86ISelLowering.cpp | 2823 +++++++++--------------
llvm/test/CodeGen/X86/apx/sub.ll | 383 ++-
2 files changed, 1287 insertions(+), 1919 deletions(-)
diff --git a/llvm/lib/Target/X86/X86ISelLowering.cpp b/llvm/lib/Target/X86/X86ISelLowering.cpp
index e97b4e6d84a9f..91cf7d3abab30 100644
--- a/llvm/lib/Target/X86/X86ISelLowering.cpp
+++ b/llvm/lib/Target/X86/X86ISelLowering.cpp
@@ -321,10 +321,6 @@ X86TargetLowering::X86TargetLowering(const X86TargetMachine &TM,
}
}
if (Subtarget.hasAVX10_2()) {
- for (MVT VT : {MVT::v8i8, MVT::v16i8, MVT::v32i8}) {
- setOperationAction(ISD::FP_TO_UINT_SAT, VT, Custom);
- setOperationAction(ISD::FP_TO_SINT_SAT, VT, Custom);
- }
setOperationAction(ISD::FP_TO_UINT_SAT, MVT::v2i32, Custom);
setOperationAction(ISD::FP_TO_SINT_SAT, MVT::v2i32, Custom);
setOperationAction(ISD::FP_TO_UINT_SAT, MVT::v8i64, Legal);
@@ -403,34 +399,34 @@ X86TargetLowering::X86TargetLowering(const X86TargetMachine &TM,
// Promote the i8 variants and force them on up to i32 which has a shorter
// encoding.
- setOperationPromotedToType(ISD::CTTZ, MVT::i8, MVT::i32);
- setOperationPromotedToType(ISD::CTTZ_ZERO_POISON, MVT::i8, MVT::i32);
+ setOperationPromotedToType(ISD::CTTZ , MVT::i8 , MVT::i32);
+ setOperationPromotedToType(ISD::CTTZ_ZERO_UNDEF, MVT::i8 , MVT::i32);
// Promoted i16. tzcntw has a false dependency on Intel CPUs. For BSF, we emit
// a REP prefix to encode it as TZCNT for modern CPUs so it makes sense to
// promote that too.
- setOperationPromotedToType(ISD::CTTZ, MVT::i16, MVT::i32);
- setOperationPromotedToType(ISD::CTTZ_ZERO_POISON, MVT::i16, MVT::i32);
+ setOperationPromotedToType(ISD::CTTZ , MVT::i16 , MVT::i32);
+ setOperationPromotedToType(ISD::CTTZ_ZERO_UNDEF, MVT::i16 , MVT::i32);
if (!Subtarget.hasBMI()) {
- setOperationAction(ISD::CTTZ, MVT::i32, Custom);
- setOperationAction(ISD::CTTZ_ZERO_POISON, MVT::i32, Legal);
+ setOperationAction(ISD::CTTZ , MVT::i32 , Custom);
+ setOperationAction(ISD::CTTZ_ZERO_UNDEF, MVT::i32 , Legal);
if (Subtarget.is64Bit()) {
- setOperationAction(ISD::CTTZ, MVT::i64, Custom);
- setOperationAction(ISD::CTTZ_ZERO_POISON, MVT::i64, Legal);
+ setOperationAction(ISD::CTTZ , MVT::i64 , Custom);
+ setOperationAction(ISD::CTTZ_ZERO_UNDEF, MVT::i64, Legal);
}
}
if (Subtarget.hasLZCNT()) {
// When promoting the i8 variants, force them to i32 for a shorter
// encoding.
- setOperationPromotedToType(ISD::CTLZ, MVT::i8, MVT::i32);
- setOperationPromotedToType(ISD::CTLZ_ZERO_POISON, MVT::i8, MVT::i32);
+ setOperationPromotedToType(ISD::CTLZ , MVT::i8 , MVT::i32);
+ setOperationPromotedToType(ISD::CTLZ_ZERO_UNDEF, MVT::i8 , MVT::i32);
} else {
for (auto VT : {MVT::i8, MVT::i16, MVT::i32, MVT::i64}) {
if (VT == MVT::i64 && !Subtarget.is64Bit())
continue;
- setOperationAction(ISD::CTLZ, VT, Custom);
- setOperationAction(ISD::CTLZ_ZERO_POISON, VT, Custom);
+ setOperationAction(ISD::CTLZ , VT, Custom);
+ setOperationAction(ISD::CTLZ_ZERO_UNDEF, VT, Custom);
}
}
@@ -480,14 +476,6 @@ X86TargetLowering::X86TargetLowering(const X86TargetMachine &TM,
setOperationAction(ISD::CTPOP , MVT::i64 , Custom);
}
- if (Subtarget.hasBMI2()) {
- setOperationAction({ISD::PEXT, ISD::PDEP}, MVT::i8, Promote);
- setOperationAction({ISD::PEXT, ISD::PDEP}, MVT::i16, Promote);
- setOperationAction({ISD::PEXT, ISD::PDEP}, MVT::i32, Legal);
- if (Subtarget.is64Bit())
- setOperationAction({ISD::PEXT, ISD::PDEP}, MVT::i64, Legal);
- }
-
setOperationAction(ISD::READCYCLECOUNTER , MVT::i64 , Custom);
if (!Subtarget.hasMOVBE())
@@ -1060,7 +1048,14 @@ X86TargetLowering::X86TargetLowering(const X86TargetMachine &TM,
setLoadExtAction(ISD::EXTLOAD, InnerVT, VT, Expand);
}
}
-
+ if (Subtarget.hasNDD()) {
+ // Enable custom lowering for scalar USUBSAT to optimize usub.sat(X,1)
+ // with cmp+adc when NDD is available.
+ setOperationAction(ISD::USUBSAT, MVT::i8, Custom);
+ setOperationAction(ISD::USUBSAT, MVT::i16, Custom);
+ setOperationAction(ISD::USUBSAT, MVT::i32, Custom);
+ setOperationAction(ISD::USUBSAT, MVT::i64, Custom);
+ }
// FIXME: In order to prevent SSE instructions being expanded to MMX ones
// with -msoft-float, disable use of MMX as well.
if (!Subtarget.useSoftFloat() && Subtarget.hasMMX()) {
@@ -1171,20 +1166,6 @@ X86TargetLowering::X86TargetLowering(const X86TargetMachine &TM,
setOperationAction(ISD::SMIN, VT, VT == MVT::v8i16 ? Legal : Custom);
setOperationAction(ISD::UMAX, VT, VT == MVT::v16i8 ? Legal : Custom);
setOperationAction(ISD::UMIN, VT, VT == MVT::v16i8 ? Legal : Custom);
- setOperationAction(ISD::VECREDUCE_AND, VT, Custom);
- setOperationAction(ISD::VECREDUCE_OR, VT, Custom);
- setOperationAction(ISD::VECREDUCE_XOR, VT, Custom);
- }
-
- // SSE2 can use basic vector unrolling.
- // SSE41 can use PHMINPOS to perform v16i8/v8i16 minmax reductions.
- // Fallback to ReplaceNodeResults for vXi64 reductions on 32-bit targets.
- for (auto VT : {MVT::v16i8, MVT::v8i16, MVT::v4i32, MVT::v2i64, MVT::i64}) {
- setOperationAction(ISD::VECREDUCE_MUL, VT, Custom);
- setOperationAction(ISD::VECREDUCE_SMAX, VT, Custom);
- setOperationAction(ISD::VECREDUCE_SMIN, VT, Custom);
- setOperationAction(ISD::VECREDUCE_UMAX, VT, Custom);
- setOperationAction(ISD::VECREDUCE_UMIN, VT, Custom);
}
setOperationAction(ISD::UADDSAT, MVT::v16i8, Legal);
@@ -1569,14 +1550,6 @@ X86TargetLowering::X86TargetLowering(const X86TargetMachine &TM,
setOperationAction(ISD::SRA, VT, Custom);
setOperationAction(ISD::ABDS, VT, Custom);
setOperationAction(ISD::ABDU, VT, Custom);
- setOperationAction(ISD::VECREDUCE_AND, VT, Custom);
- setOperationAction(ISD::VECREDUCE_OR, VT, Custom);
- setOperationAction(ISD::VECREDUCE_XOR, VT, Custom);
- setOperationAction(ISD::VECREDUCE_MUL, VT, Custom);
- setOperationAction(ISD::VECREDUCE_SMAX, VT, Custom);
- setOperationAction(ISD::VECREDUCE_SMIN, VT, Custom);
- setOperationAction(ISD::VECREDUCE_UMAX, VT, Custom);
- setOperationAction(ISD::VECREDUCE_UMIN, VT, Custom);
if (VT == MVT::v4i64) continue;
setOperationAction(ISD::ROTL, VT, Custom);
setOperationAction(ISD::ROTR, VT, Custom);
@@ -2046,14 +2019,6 @@ X86TargetLowering::X86TargetLowering(const X86TargetMachine &TM,
setOperationAction(ISD::ABDS, VT, Custom);
setOperationAction(ISD::ABDU, VT, Custom);
setOperationAction(ISD::BITREVERSE, VT, Custom);
- setOperationAction(ISD::VECREDUCE_AND, VT, Custom);
- setOperationAction(ISD::VECREDUCE_OR, VT, Custom);
- setOperationAction(ISD::VECREDUCE_XOR, VT, Custom);
- setOperationAction(ISD::VECREDUCE_MUL, VT, Custom);
- setOperationAction(ISD::VECREDUCE_SMAX, VT, Custom);
- setOperationAction(ISD::VECREDUCE_SMIN, VT, Custom);
- setOperationAction(ISD::VECREDUCE_UMAX, VT, Custom);
- setOperationAction(ISD::VECREDUCE_UMIN, VT, Custom);
// The condition codes aren't legal in SSE/AVX and under AVX512 we use
// setcc all the way to isel and prefer SETGT in some isel patterns.
@@ -2273,8 +2238,8 @@ X86TargetLowering::X86TargetLowering(const X86TargetMachine &TM,
continue;
setOperationAction(ISD::CTLZ, VT, Custom);
setOperationAction(ISD::CTTZ, VT, Custom);
- setOperationAction(ISD::CTLZ_ZERO_POISON, VT, Custom);
- setOperationAction(ISD::CTTZ_ZERO_POISON, VT, Custom);
+ setOperationAction(ISD::CTLZ_ZERO_UNDEF, VT, Custom);
+ setOperationAction(ISD::CTTZ_ZERO_UNDEF, VT, Custom);
}
for (auto VT : { MVT::v4i32, MVT::v8i32, MVT::v2i64, MVT::v4i64 }) {
setOperationAction(ISD::CTLZ, VT, Legal);
@@ -2358,11 +2323,6 @@ X86TargetLowering::X86TargetLowering(const X86TargetMachine &TM,
for (auto VT : { MVT::v16i8, MVT::v32i8, MVT::v8i16, MVT::v16i16 })
setOperationAction(ISD::CTPOP, VT, Legal);
}
-
- if (Subtarget.hasBMM()) {
- for (auto VT : {MVT::v16i8, MVT::v32i8, MVT::v64i8})
- setOperationAction(ISD::BITREVERSE, VT, Legal);
- }
}
if (!Subtarget.useSoftFloat() && Subtarget.hasFP16()) {
@@ -2765,10 +2725,6 @@ X86TargetLowering::X86TargetLowering(const X86TargetMachine &TM,
setOperationPromotedToType(ISD::ATOMIC_LOAD, MVT::f32, MVT::i32);
setOperationPromotedToType(ISD::ATOMIC_LOAD, MVT::f64, MVT::i64);
- setOperationPromotedToType(ISD::ATOMIC_STORE, MVT::f16, MVT::i16);
- setOperationPromotedToType(ISD::ATOMIC_STORE, MVT::f32, MVT::i32);
- setOperationPromotedToType(ISD::ATOMIC_STORE, MVT::f64, MVT::i64);
-
// We have target-specific dag combine patterns for the following nodes:
setTargetDAGCombine({ISD::VECTOR_SHUFFLE,
ISD::SCALAR_TO_VECTOR,
@@ -2801,7 +2757,6 @@ X86TargetLowering::X86TargetLowering(const X86TargetMachine &TM,
ISD::FMINNUM,
ISD::FMAXNUM,
ISD::SUB,
- ISD::ATOMIC_LOAD,
ISD::LOAD,
ISD::LRINT,
ISD::LLRINT,
@@ -2816,7 +2771,6 @@ X86TargetLowering::X86TargetLowering(const X86TargetMachine &TM,
ISD::ANY_EXTEND_VECTOR_INREG,
ISD::SIGN_EXTEND_VECTOR_INREG,
ISD::ZERO_EXTEND_VECTOR_INREG,
- ISD::VECREDUCE_MUL,
ISD::SINT_TO_FP,
ISD::UINT_TO_FP,
ISD::FP_TO_SINT,
@@ -2877,12 +2831,12 @@ bool X86TargetLowering::useLoadStackGuardNode(const Module &M) const {
return Subtarget.isTargetMachO() && Subtarget.is64Bit();
}
-bool X86TargetLowering::useStackGuardMixFP() const {
- // Currently only MSVC CRTs mix the frame pointer into the stack guard value.
+bool X86TargetLowering::useStackGuardXorFP() const {
+ // Currently only MSVC CRTs XOR the frame pointer into the stack guard value.
return Subtarget.getTargetTriple().isOSMSVCRT() && !Subtarget.isTargetMachO();
}
-SDValue X86TargetLowering::emitStackGuardMixFP(SelectionDAG &DAG, SDValue Val,
+SDValue X86TargetLowering::emitStackGuardXorFP(SelectionDAG &DAG, SDValue Val,
const SDLoc &DL) const {
EVT PtrTy = getPointerTy(DAG.getDataLayout());
unsigned XorOp = Subtarget.is64Bit() ? X86::XOR64_FP : X86::XOR32_FP;
@@ -2963,7 +2917,7 @@ bool X86::mayFoldIntoStore(SDValue Op) {
return false;
User = *User->user_begin();
}
- return ISD::isNormalStore(User) || User->getOpcode() == ISD::ATOMIC_STORE;
+ return ISD::isNormalStore(User);
}
bool X86::mayFoldIntoZeroExtend(SDValue Op) {
@@ -3389,7 +3343,7 @@ void X86TargetLowering::getTgtMemIntrinsic(
else if (IntrData->Type == TRUNCATE_TO_MEM_VI32)
ScalarVT = MVT::i32;
- Info.memVT = VT.changeElementType(ScalarVT);
+ Info.memVT = MVT::getVectorVT(ScalarVT, VT.getVectorNumElements());
Info.align = Align(1);
Info.flags |= MachineMemOperand::MOStore;
Infos.push_back(Info);
@@ -3585,9 +3539,8 @@ bool X86TargetLowering::isExtractSubvectorCheap(EVT ResVT, EVT SrcVT,
// Mask vectors support all subregister combinations and operations that
// extract half of vector.
if (ResVT.getVectorElementType() == MVT::i1)
- return Index == 0 ||
- ((ResVT.getSizeInBits() * 2 == SrcVT.getSizeInBits()) &&
- (Index == ResVT.getVectorNumElements()));
+ return Index == 0 || ((ResVT.getSizeInBits() == SrcVT.getSizeInBits()*2) &&
+ (Index == ResVT.getVectorNumElements()));
return (Index % ResVT.getVectorNumElements()) == 0;
}
@@ -4987,27 +4940,29 @@ static unsigned getTargetVShiftUniformOpcode(unsigned Opc, bool IsVariable) {
static SDValue getTargetVShiftByConstNode(unsigned Opc, const SDLoc &dl, MVT VT,
SDValue SrcOp, uint64_t ShiftAmt,
SelectionDAG &DAG) {
- assert(
- (Opc == X86ISD::VSHLI || Opc == X86ISD::VSRLI || Opc == X86ISD::VSRAI) &&
- "Unknown target vector shift-by-constant node");
+ MVT ElementType = VT.getVectorElementType();
// Bitcast the source vector to the output type, this is mainly necessary for
// vXi8/vXi64 shifts.
- SrcOp = DAG.getBitcast(VT, SrcOp);
+ if (VT != SrcOp.getSimpleValueType())
+ SrcOp = DAG.getBitcast(VT, SrcOp);
// Fold this packed shift into its first operand if ShiftAmt is 0.
if (ShiftAmt == 0)
return SrcOp;
// Check for ShiftAmt >= element width
- unsigned EltSizeInBits = VT.getScalarSizeInBits();
- if (ShiftAmt >= EltSizeInBits) {
+ if (ShiftAmt >= ElementType.getSizeInBits()) {
if (Opc == X86ISD::VSRAI)
- ShiftAmt = EltSizeInBits - 1;
+ ShiftAmt = ElementType.getSizeInBits() - 1;
else
return DAG.getConstant(0, dl, VT);
}
+ assert(
+ (Opc == X86ISD::VSHLI || Opc == X86ISD::VSRLI || Opc == X86ISD::VSRAI) &&
+ "Unknown target vector shift-by-constant node");
+
// Fold this packed vector shift into a build vector if SrcOp is a
// vector of Constants or UNDEFs.
if (ISD::isBuildVectorOfConstantSDNodes(SrcOp.getNode())) {
@@ -5458,13 +5413,11 @@ static bool getTargetConstantBitsFromNode(SDValue Op, unsigned EltSizeInBits,
return true;
}
if (auto *CInt = dyn_cast<ConstantInt>(Cst)) {
- Mask = APInt::getSplat(CInt->getType()->getPrimitiveSizeInBits(),
- CInt->getValue());
+ Mask = CInt->getValue();
return true;
}
if (auto *CFP = dyn_cast<ConstantFP>(Cst)) {
- Mask = APInt::getSplat(CFP->getType()->getPrimitiveSizeInBits(),
- CFP->getValueAPF().bitcastToAPInt());
+ Mask = CFP->getValueAPF().bitcastToAPInt();
return true;
}
if (auto *CDS = dyn_cast<ConstantDataSequential>(Cst)) {
@@ -7050,26 +7003,6 @@ static bool getFauxShuffleMask(SDValue N, const APInt &DemandedElts,
}
return true;
}
- case X86ISD::VSHLD:
- case X86ISD::VSHRD: {
- // We can only decode 'whole byte' bit funnel shifts as shuffles.
- uint64_t ShiftVal = N.getConstantOperandAPInt(2).urem(NumBitsPerElt);
- int Offset = ShiftVal / 8;
- if ((ShiftVal % 8) != 0 || Offset == 0)
- return false;
- Ops.push_back(N.getOperand(X86ISD::VSHRD == Opcode ? 1 : 0));
- Ops.push_back(N.getOperand(X86ISD::VSHRD == Opcode ? 0 : 1));
- Offset = X86ISD::VSHRD == Opcode ? (NumBytesPerElt - Offset) : Offset;
- for (int I = 0; I != (int)NumElts; ++I) {
- int BaseIdx = (I * NumBytesPerElt) - Offset;
- for (int J = 0; J != (int)NumBytesPerElt; ++J) {
- int MaskIdx = BaseIdx + J;
- MaskIdx += J < Offset ? (NumSizeInBytes + NumBytesPerElt) : 0;
- Mask.push_back(MaskIdx);
- }
- }
- return true;
- }
case X86ISD::VBROADCAST: {
SDValue Src = N.getOperand(0);
if (!Src.getSimpleValueType().isVector()) {
@@ -7157,7 +7090,7 @@ static void resolveTargetShuffleInputsAndMask(SmallVectorImpl<SDValue> &Inputs,
// Check for repeated inputs.
bool IsRepeat = false;
for (int j = 0, ue = UsedInputs.size(); j != ue; ++j) {
- if (peekThroughBitcasts(UsedInputs[j]) != peekThroughBitcasts(Inputs[i]))
+ if (UsedInputs[j] != Inputs[i])
continue;
for (int &M : Mask)
if (lo <= M)
@@ -7762,22 +7695,6 @@ static SDValue EltsFromConsecutiveLoads(EVT VT, ArrayRef<SDValue> Elts,
if ((VT.getScalarSizeInBits() % 8) != 0)
return SDValue();
- // If all of these are oneuse frozen loads, then attempt to create a frozen
- // consecutive load.
- if (all_of(Elts, [](SDValue Elt) {
- return Elt.getOpcode() == ISD::FREEZE &&
- ISD::isNormalLoad(Elt.getOperand(0).getNode()) &&
- Elt.hasOneUse();
- })) {
- SmallVector<SDValue, 16> SrcElts;
- for (SDValue Elt : Elts)
- SrcElts.push_back(peekThroughFreeze(Elt));
- if (SDValue LD = EltsFromConsecutiveLoads(VT, SrcElts, DL, DAG, Subtarget,
- IsAfterLegalize, Depth + 1))
- return DAG.getFreeze(LD);
- return SDValue();
- }
-
unsigned NumElems = Elts.size();
int LastLoadedElt = -1;
@@ -8545,7 +8462,7 @@ static SDValue LowerBUILD_VECTORvXbf16(SDValue Op, SelectionDAG &DAG,
return DAG.getBitcast(VT, Res);
}
-// Lower BUILD_VECTOR operation for vXi1 types.
+// Lower BUILD_VECTOR operation for v8i1 and v16i1 types.
static SDValue LowerBUILD_VECTORvXi1(SDValue Op, const SDLoc &dl,
SelectionDAG &DAG,
const X86Subtarget &Subtarget) {
@@ -8557,64 +8474,51 @@ static SDValue LowerBUILD_VECTORvXi1(SDValue Op, const SDLoc &dl,
ISD::isBuildVectorAllOnes(Op.getNode()))
return Op;
- uint64_t Undefs = 0;
uint64_t Immediate = 0;
- uint64_t NonConstMask = 0;
- SmallSet<SDValue, 16> NonConstElts;
+ SmallVector<unsigned, 16> NonConstIdx;
+ bool IsSplat = true;
bool HasConstElts = false;
+ int SplatIdx = -1;
for (unsigned idx = 0, e = Op.getNumOperands(); idx < e; ++idx) {
SDValue In = Op.getOperand(idx);
- if (In.isUndef()) {
- Undefs |= 1ULL << idx;
+ if (In.isUndef())
continue;
- }
if (auto *InC = dyn_cast<ConstantSDNode>(In)) {
Immediate |= (InC->getZExtValue() & 0x1) << idx;
HasConstElts = true;
} else {
- NonConstMask |= 1ULL << idx;
- NonConstElts.insert(In);
+ NonConstIdx.push_back(idx);
}
+ if (SplatIdx < 0)
+ SplatIdx = idx;
+ else if (In != Op.getOperand(SplatIdx))
+ IsSplat = false;
}
- // for single non-const use " (select i1 elt, imm | elt_mask, imm)"
- if (NonConstElts.size() == 1) {
+ // for splat use " (select i1 splat_elt, all-ones, all-zeroes)"
+ if (IsSplat) {
// The build_vector allows the scalar element to be larger than the vector
// element type. We need to mask it to use as a condition unless we know
// the upper bits are zero.
// FIXME: Use computeKnownBits instead of checking specific opcode?
- SDValue Cond = *NonConstElts.begin();
+ SDValue Cond = Op.getOperand(SplatIdx);
assert(Cond.getValueType() == MVT::i8 && "Unexpected VT!");
if (Cond.getOpcode() != ISD::SETCC)
Cond = DAG.getNode(ISD::AND, dl, MVT::i8, Cond,
DAG.getConstant(1, dl, MVT::i8));
- uint64_t TrueImm = NonConstMask | Immediate;
- uint64_t FalseImm = Immediate;
-
// Perform the select in the scalar domain so we can use cmov.
if (VT == MVT::v64i1 && !Subtarget.is64Bit()) {
- uint64_t TrueLo = (unsigned)TrueImm;
- uint64_t TrueHi = TrueImm >> 32;
- uint64_t FalseLo = (unsigned)FalseImm;
- uint64_t FalseHi = FalseImm >> 32;
- SDValue Lo = DAG.getSelect(dl, MVT::i32, Cond,
- DAG.getConstant(TrueLo, dl, MVT::i32),
- DAG.getConstant(FalseLo, dl, MVT::i32));
- SDValue Hi = DAG.getSelect(dl, MVT::i32, Cond,
- DAG.getConstant(TrueHi, dl, MVT::i32),
- DAG.getConstant(FalseHi, dl, MVT::i32));
- Lo = DAG.getBitcast(MVT::v32i1, Lo);
- Hi = DAG.getBitcast(MVT::v32i1, Hi);
- return DAG.getNode(ISD::CONCAT_VECTORS, dl, MVT::v64i1, Lo, Hi);
+ SDValue Select = DAG.getSelect(dl, MVT::i32, Cond,
+ DAG.getAllOnesConstant(dl, MVT::i32),
+ DAG.getConstant(0, dl, MVT::i32));
+ Select = DAG.getBitcast(MVT::v32i1, Select);
+ return DAG.getNode(ISD::CONCAT_VECTORS, dl, MVT::v64i1, Select, Select);
} else {
MVT ImmVT = MVT::getIntegerVT(std::max((unsigned)VT.getSizeInBits(), 8U));
- // Adjust extended value to -1 as it will improve folding.
- if ((TrueImm | Undefs) == (~0ULL >> (64 - VT.getSizeInBits())))
- TrueImm = ~0ULL >> (64 - ImmVT.getSizeInBits());
- SDValue Select =
- DAG.getSelect(dl, ImmVT, Cond, DAG.getConstant(TrueImm, dl, ImmVT),
- DAG.getConstant(FalseImm, dl, ImmVT));
+ SDValue Select = DAG.getSelect(dl, ImmVT, Cond,
+ DAG.getAllOnesConstant(dl, ImmVT),
+ DAG.getConstant(0, dl, ImmVT));
MVT VecVT = VT.getSizeInBits() >= 8 ? VT : MVT::v8i1;
Select = DAG.getBitcast(VecVT, Select);
return DAG.getNode(ISD::EXTRACT_SUBVECTOR, dl, VT, Select,
@@ -8622,32 +8526,6 @@ static SDValue LowerBUILD_VECTORvXi1(SDValue Op, const SDLoc &dl,
}
}
- // See if we can cheaply generate a vXi8 vector and convert to vXi1.
- MVT OpVT = Op.getOperand(0).getSimpleValueType();
- if (OpVT == MVT::i8 && NonConstMask != 0) {
- // On pre-BWI targets, we must extend to vXi32 instead.
- MVT ByteVT = VT.changeVectorElementType(MVT::i8);
- MVT WideSVT = Subtarget.hasBWI() ? MVT::i8 : MVT::i32;
- if (ByteVT.getSizeInBits() < 128) {
- WideSVT = ByteVT == MVT::v4i8 ? MVT::i32 : MVT::i64;
- ByteVT = MVT::v16i8;
- }
- MVT WideVT = VT.changeVectorElementType(WideSVT);
- if (DAG.getTargetLoweringInfo().isTypeLegal(ByteVT) &&
- DAG.getTargetLoweringInfo().isTypeLegal(WideVT)) {
- SmallVector<SDValue, 16> Elts(Op->op_values());
- Elts.append(ByteVT.getVectorNumElements() - Elts.size(),
- DAG.getPOISON(OpVT));
- SDValue ByteBV = DAG.getBuildVector(ByteVT, dl, Elts);
- SDValue WideBV =
- getEXTEND_VECTOR_INREG(ISD::ANY_EXTEND, dl, WideVT, ByteBV, DAG);
- WideBV = DAG.getNode(ISD::AND, dl, WideVT, WideBV,
- DAG.getConstant(1, dl, WideVT));
- return DAG.getSetCC(dl, VT, WideBV, DAG.getConstant(0, dl, WideVT),
- ISD::SETNE);
- }
- }
-
// insert elements one by one
SDValue DstVec;
if (HasConstElts) {
@@ -8668,10 +8546,11 @@ static SDValue LowerBUILD_VECTORvXi1(SDValue Op, const SDLoc &dl,
} else
DstVec = DAG.getUNDEF(VT);
- for (unsigned Idx = 0, E = Op.getNumOperands(); Idx != E; ++Idx)
- if (NonConstMask & (1ULL << Idx))
- DstVec = DAG.getInsertVectorElt(dl, DstVec, Op.getOperand(Idx), Idx);
-
+ for (unsigned InsertIdx : NonConstIdx) {
+ DstVec = DAG.getNode(ISD::INSERT_VECTOR_ELT, dl, VT, DstVec,
+ Op.getOperand(InsertIdx),
+ DAG.getVectorIdxConstant(InsertIdx, dl));
+ }
return DstVec;
}
@@ -8690,6 +8569,176 @@ static SDValue LowerBUILD_VECTORvXi1(SDValue Op, const SDLoc &dl,
return false;
}
+/// This is a helper function of LowerToHorizontalOp().
+/// This function checks that the build_vector \p N in input implements a
+/// 128-bit partial horizontal operation on a 256-bit vector, but that operation
+/// may not match the layout of an x86 256-bit horizontal instruction.
+/// In other words, if this returns true, then some extraction/insertion will
+/// be required to produce a valid horizontal instruction.
+///
+/// Parameter \p Opcode defines the kind of horizontal operation to match.
+/// For example, if \p Opcode is equal to ISD::ADD, then this function
+/// checks if \p N implements a horizontal arithmetic add; if instead \p Opcode
+/// is equal to ISD::SUB, then this function checks if this is a horizontal
+/// arithmetic sub.
+///
+/// This function only analyzes elements of \p N whose indices are
+/// in range [BaseIdx, LastIdx).
+///
+/// TODO: This function was originally used to match both real and fake partial
+/// horizontal operations, but the index-matching logic is incorrect for that.
+/// See the corrected implementation in isHopBuildVector(). Can we reduce this
+/// code because it is only used for partial h-op matching now?
+static bool isHorizontalBinOpPart(const BuildVectorSDNode *N, unsigned Opcode,
+ const SDLoc &DL, SelectionDAG &DAG,
+ unsigned BaseIdx, unsigned LastIdx,
+ SDValue &V0, SDValue &V1) {
+ EVT VT = N->getValueType(0);
+ assert(VT.is256BitVector() && "Only use for matching partial 256-bit h-ops");
+ assert(BaseIdx * 2 <= LastIdx && "Invalid Indices in input!");
+ assert(VT.isVector() && VT.getVectorNumElements() >= LastIdx &&
+ "Invalid Vector in input!");
+
+ bool IsCommutable = (Opcode == ISD::ADD || Opcode == ISD::FADD);
+ bool CanFold = true;
+ unsigned ExpectedVExtractIdx = BaseIdx;
+ unsigned NumElts = LastIdx - BaseIdx;
+ V0 = DAG.getUNDEF(VT);
+ V1 = DAG.getUNDEF(VT);
+
+ // Check if N implements a horizontal binop.
+ for (unsigned i = 0, e = NumElts; i != e && CanFold; ++i) {
+ SDValue Op = N->getOperand(i + BaseIdx);
+
+ // Skip UNDEFs.
+ if (Op->isUndef()) {
+ // Update the expected vector extract index.
+ if (i * 2 == NumElts)
+ ExpectedVExtractIdx = BaseIdx;
+ ExpectedVExtractIdx += 2;
+ continue;
+ }
+
+ CanFold = Op->getOpcode() == Opcode && Op->hasOneUse();
+
+ if (!CanFold)
+ break;
+
+ SDValue Op0 = Op.getOperand(0);
+ SDValue Op1 = Op.getOperand(1);
+
+ // Try to match the following pattern:
+ // (BINOP (extract_vector_elt A, I), (extract_vector_elt A, I+1))
+ CanFold = (Op0.getOpcode() == ISD::EXTRACT_VECTOR_ELT &&
+ Op1.getOpcode() == ISD::EXTRACT_VECTOR_ELT &&
+ Op0.getOperand(0) == Op1.getOperand(0) &&
+ isa<ConstantSDNode>(Op0.getOperand(1)) &&
+ isa<ConstantSDNode>(Op1.getOperand(1)));
+ if (!CanFold)
+ break;
+
+ unsigned I0 = Op0.getConstantOperandVal(1);
+ unsigned I1 = Op1.getConstantOperandVal(1);
+
+ if (i * 2 < NumElts) {
+ if (V0.isUndef()) {
+ V0 = Op0.getOperand(0);
+ if (V0.getValueType() != VT)
+ return false;
+ }
+ } else {
+ if (V1.isUndef()) {
+ V1 = Op0.getOperand(0);
+ if (V1.getValueType() != VT)
+ return false;
+ }
+ if (i * 2 == NumElts)
+ ExpectedVExtractIdx = BaseIdx;
+ }
+
+ SDValue Expected = (i * 2 < NumElts) ? V0 : V1;
+ if (I0 == ExpectedVExtractIdx)
+ CanFold = I1 == I0 + 1 && Op0.getOperand(0) == Expected;
+ else if (IsCommutable && I1 == ExpectedVExtractIdx) {
+ // Try to match the following dag sequence:
+ // (BINOP (extract_vector_elt A, I+1), (extract_vector_elt A, I))
+ CanFold = I0 == I1 + 1 && Op1.getOperand(0) == Expected;
+ } else
+ CanFold = false;
+
+ ExpectedVExtractIdx += 2;
+ }
+
+ return CanFold;
+}
+
+/// Emit a sequence of two 128-bit horizontal add/sub followed by
+/// a concat_vector.
+///
+/// This is a helper function of LowerToHorizontalOp().
+/// This function expects two 256-bit vectors called V0 and V1.
+/// At first, each vector is split into two separate 128-bit vectors.
+/// Then, the resulting 128-bit vectors are used to implement two
+/// horizontal binary operations.
+///
+/// The kind of horizontal binary operation is defined by \p X86Opcode.
+///
+/// \p Mode specifies how the 128-bit parts of V0 and V1 are passed in input to
+/// the two new horizontal binop.
+/// When Mode is set, the first horizontal binop dag node would take as input
+/// the lower 128-bit of V0 and the upper 128-bit of V0. The second
+/// horizontal binop dag node would take as input the lower 128-bit of V1
+/// and the upper 128-bit of V1.
+/// Example:
+/// HADD V0_LO, V0_HI
+/// HADD V1_LO, V1_HI
+///
+/// Otherwise, the first horizontal binop dag node takes as input the lower
+/// 128-bit of V0 and the lower 128-bit of V1, and the second horizontal binop
+/// dag node takes the upper 128-bit of V0 and the upper 128-bit of V1.
+/// Example:
+/// HADD V0_LO, V1_LO
+/// HADD V0_HI, V1_HI
+///
+/// If \p isUndefLO is set, then the algorithm propagates UNDEF to the lower
+/// 128-bits of the result. If \p isUndefHI is set, then UNDEF is propagated to
+/// the upper 128-bits of the result.
+static SDValue ExpandHorizontalBinOp(const SDValue &V0, const SDValue &V1,
+ const SDLoc &DL, SelectionDAG &DAG,
+ unsigned X86Opcode, bool Mode,
+ bool isUndefLO, bool isUndefHI) {
+ MVT VT = V0.getSimpleValueType();
+ assert(VT.is256BitVector() && VT == V1.getSimpleValueType() &&
+ "Invalid nodes in input!");
+
+ unsigned NumElts = VT.getVectorNumElements();
+ SDValue V0_LO = extract128BitVector(V0, 0, DAG, DL);
+ SDValue V0_HI = extract128BitVector(V0, NumElts/2, DAG, DL);
+ SDValue V1_LO = extract128BitVector(V1, 0, DAG, DL);
+ SDValue V1_HI = extract128BitVector(V1, NumElts/2, DAG, DL);
+ MVT NewVT = V0_LO.getSimpleValueType();
+
+ SDValue LO = DAG.getUNDEF(NewVT);
+ SDValue HI = DAG.getUNDEF(NewVT);
+
+ if (Mode) {
+ // Don't emit a horizontal binop if the result is expected to be UNDEF.
+ if (!isUndefLO && !V0->isUndef())
+ LO = DAG.getNode(X86Opcode, DL, NewVT, V0_LO, V0_HI);
+ if (!isUndefHI && !V1->isUndef())
+ HI = DAG.getNode(X86Opcode, DL, NewVT, V1_LO, V1_HI);
+ } else {
+ // Don't emit a horizontal binop if the result is expected to be UNDEF.
+ if (!isUndefLO && (!V0_LO->isUndef() || !V1_LO->isUndef()))
+ LO = DAG.getNode(X86Opcode, DL, NewVT, V0_LO, V1_LO);
+
+ if (!isUndefHI && (!V0_HI->isUndef() || !V1_HI->isUndef()))
+ HI = DAG.getNode(X86Opcode, DL, NewVT, V0_HI, V1_HI);
+ }
+
+ return DAG.getNode(ISD::CONCAT_VECTORS, DL, VT, LO, HI);
+}
+
/// Returns true iff \p BV builds a vector with the result equivalent to
/// the result of ADDSUB/SUBADD operation.
/// If true is returned then the operands of ADDSUB = Opnd0 +- Opnd1
@@ -8883,6 +8932,248 @@ static SDValue lowerToAddSubOrFMAddSub(const BuildVectorSDNode *BV,
return DAG.getNode(X86ISD::ADDSUB, DL, VT, Opnd0, Opnd1);
}
+static bool isHopBuildVector(const BuildVectorSDNode *BV, SelectionDAG &DAG,
+ unsigned &HOpcode, SDValue &V0, SDValue &V1) {
+ // Initialize outputs to known values.
+ MVT VT = BV->getSimpleValueType(0);
+ HOpcode = ISD::DELETED_NODE;
+ V0 = DAG.getUNDEF(VT);
+ V1 = DAG.getUNDEF(VT);
+
+ // x86 256-bit horizontal ops are defined in a non-obvious way. Each 128-bit
+ // half of the result is calculated independently from the 128-bit halves of
+ // the inputs, so that makes the index-checking logic below more complicated.
+ unsigned NumElts = VT.getVectorNumElements();
+ unsigned GenericOpcode = ISD::DELETED_NODE;
+ unsigned Num128BitChunks = VT.is256BitVector() ? 2 : 1;
+ unsigned NumEltsIn128Bits = NumElts / Num128BitChunks;
+ unsigned NumEltsIn64Bits = NumEltsIn128Bits / 2;
+ for (unsigned i = 0; i != Num128BitChunks; ++i) {
+ for (unsigned j = 0; j != NumEltsIn128Bits; ++j) {
+ // Ignore undef elements.
+ SDValue Op = BV->getOperand(i * NumEltsIn128Bits + j);
+ if (Op.isUndef())
+ continue;
+
+ // If there's an opcode mismatch, we're done.
+ if (HOpcode != ISD::DELETED_NODE && Op.getOpcode() != GenericOpcode)
+ return false;
+
+ // Initialize horizontal opcode.
+ if (HOpcode == ISD::DELETED_NODE) {
+ GenericOpcode = Op.getOpcode();
+ switch (GenericOpcode) {
+ // clang-format off
+ case ISD::ADD: HOpcode = X86ISD::HADD; break;
+ case ISD::SUB: HOpcode = X86ISD::HSUB; break;
+ case ISD::FADD: HOpcode = X86ISD::FHADD; break;
+ case ISD::FSUB: HOpcode = X86ISD::FHSUB; break;
+ default: return false;
+ // clang-format on
+ }
+ }
+
+ SDValue Op0 = Op.getOperand(0);
+ SDValue Op1 = Op.getOperand(1);
+ if (Op0.getOpcode() != ISD::EXTRACT_VECTOR_ELT ||
+ Op1.getOpcode() != ISD::EXTRACT_VECTOR_ELT ||
+ Op0.getOperand(0) != Op1.getOperand(0) ||
+ !isa<ConstantSDNode>(Op0.getOperand(1)) ||
+ !isa<ConstantSDNode>(Op1.getOperand(1)) || !Op.hasOneUse())
+ return false;
+
+ // The source vector is chosen based on which 64-bit half of the
+ // destination vector is being calculated.
+ if (j < NumEltsIn64Bits) {
+ if (V0.isUndef())
+ V0 = Op0.getOperand(0);
+ } else {
+ if (V1.isUndef())
+ V1 = Op0.getOperand(0);
+ }
+
+ SDValue SourceVec = (j < NumEltsIn64Bits) ? V0 : V1;
+ if (SourceVec != Op0.getOperand(0))
+ return false;
+
+ // op (extract_vector_elt A, I), (extract_vector_elt A, I+1)
+ unsigned ExtIndex0 = Op0.getConstantOperandVal(1);
+ unsigned ExtIndex1 = Op1.getConstantOperandVal(1);
+ unsigned ExpectedIndex = i * NumEltsIn128Bits +
+ (j % NumEltsIn64Bits) * 2;
+ if (ExpectedIndex == ExtIndex0 && ExtIndex1 == ExtIndex0 + 1)
+ continue;
+
+ // If this is not a commutative op, this does not match.
+ if (GenericOpcode != ISD::ADD && GenericOpcode != ISD::FADD)
+ return false;
+
+ // Addition is commutative, so try swapping the extract indexes.
+ // op (extract_vector_elt A, I+1), (extract_vector_elt A, I)
+ if (ExpectedIndex == ExtIndex1 && ExtIndex0 == ExtIndex1 + 1)
+ continue;
+
+ // Extract indexes do not match horizontal requirement.
+ return false;
+ }
+ }
+ // We matched. Opcode and operands are returned by reference as arguments.
+ return true;
+}
+
+static SDValue getHopForBuildVector(const BuildVectorSDNode *BV,
+ const SDLoc &DL, SelectionDAG &DAG,
+ unsigned HOpcode, SDValue V0, SDValue V1) {
+ // If either input vector is not the same size as the build vector,
+ // extract/insert the low bits to the correct size.
+ // This is free (examples: zmm --> xmm, xmm --> ymm).
+ MVT VT = BV->getSimpleValueType(0);
+ unsigned Width = VT.getSizeInBits();
+ if (V0.getValueSizeInBits() > Width)
+ V0 = extractSubVector(V0, 0, DAG, DL, Width);
+ else if (V0.getValueSizeInBits() < Width)
+ V0 = insertSubVector(DAG.getUNDEF(VT), V0, 0, DAG, DL, Width);
+
+ if (V1.getValueSizeInBits() > Width)
+ V1 = extractSubVector(V1, 0, DAG, DL, Width);
+ else if (V1.getValueSizeInBits() < Width)
+ V1 = insertSubVector(DAG.getUNDEF(VT), V1, 0, DAG, DL, Width);
+
+ unsigned NumElts = VT.getVectorNumElements();
+ APInt DemandedElts = APInt::getAllOnes(NumElts);
+ for (unsigned i = 0; i != NumElts; ++i)
+ if (BV->getOperand(i).isUndef())
+ DemandedElts.clearBit(i);
+
+ // If we don't need the upper xmm, then perform as a xmm hop.
+ unsigned HalfNumElts = NumElts / 2;
+ if (VT.is256BitVector() && DemandedElts.lshr(HalfNumElts) == 0) {
+ MVT HalfVT = VT.getHalfNumVectorElementsVT();
+ V0 = extractSubVector(V0, 0, DAG, DL, 128);
+ V1 = extractSubVector(V1, 0, DAG, DL, 128);
+ SDValue Half = DAG.getNode(HOpcode, DL, HalfVT, V0, V1);
+ return insertSubVector(DAG.getUNDEF(VT), Half, 0, DAG, DL, 256);
+ }
+
+ return DAG.getNode(HOpcode, DL, VT, V0, V1);
+}
+
+/// Lower BUILD_VECTOR to a horizontal add/sub operation if possible.
+static SDValue LowerToHorizontalOp(const BuildVectorSDNode *BV, const SDLoc &DL,
+ const X86Subtarget &Subtarget,
+ SelectionDAG &DAG) {
+ // We need at least 2 non-undef elements to make this worthwhile by default.
+ unsigned NumNonUndefs =
+ count_if(BV->op_values(), [](SDValue V) { return !V.isUndef(); });
+ if (NumNonUndefs < 2)
+ return SDValue();
+
+ // There are 4 sets of horizontal math operations distinguished by type:
+ // int/FP at 128-bit/256-bit. Each type was introduced with a different
+ // subtarget feature. Try to match those "native" patterns first.
+ MVT VT = BV->getSimpleValueType(0);
+ if (((VT == MVT::v4f32 || VT == MVT::v2f64) && Subtarget.hasSSE3()) ||
+ ((VT == MVT::v8i16 || VT == MVT::v4i32) && Subtarget.hasSSSE3()) ||
+ ((VT == MVT::v8f32 || VT == MVT::v4f64) && Subtarget.hasAVX()) ||
+ ((VT == MVT::v16i16 || VT == MVT::v8i32) && Subtarget.hasAVX2())) {
+ unsigned HOpcode;
+ SDValue V0, V1;
+ if (isHopBuildVector(BV, DAG, HOpcode, V0, V1))
+ return getHopForBuildVector(BV, DL, DAG, HOpcode, V0, V1);
+ }
+
+ // Try harder to match 256-bit ops by using extract/concat.
+ if (!Subtarget.hasAVX() || !VT.is256BitVector())
+ return SDValue();
+
+ // Count the number of UNDEF operands in the build_vector in input.
+ unsigned NumElts = VT.getVectorNumElements();
+ unsigned Half = NumElts / 2;
+ unsigned NumUndefsLO = 0;
+ unsigned NumUndefsHI = 0;
+ for (unsigned i = 0, e = Half; i != e; ++i)
+ if (BV->getOperand(i)->isUndef())
+ NumUndefsLO++;
+
+ for (unsigned i = Half, e = NumElts; i != e; ++i)
+ if (BV->getOperand(i)->isUndef())
+ NumUndefsHI++;
+
+ SDValue InVec0, InVec1;
+ if (VT == MVT::v8i32 || VT == MVT::v16i16) {
+ SDValue InVec2, InVec3;
+ unsigned X86Opcode;
+ bool CanFold = true;
+
+ if (isHorizontalBinOpPart(BV, ISD::ADD, DL, DAG, 0, Half, InVec0, InVec1) &&
+ isHorizontalBinOpPart(BV, ISD::ADD, DL, DAG, Half, NumElts, InVec2,
+ InVec3) &&
+ ((InVec0.isUndef() || InVec2.isUndef()) || InVec0 == InVec2) &&
+ ((InVec1.isUndef() || InVec3.isUndef()) || InVec1 == InVec3))
+ X86Opcode = X86ISD::HADD;
+ else if (isHorizontalBinOpPart(BV, ISD::SUB, DL, DAG, 0, Half, InVec0,
+ InVec1) &&
+ isHorizontalBinOpPart(BV, ISD::SUB, DL, DAG, Half, NumElts, InVec2,
+ InVec3) &&
+ ((InVec0.isUndef() || InVec2.isUndef()) || InVec0 == InVec2) &&
+ ((InVec1.isUndef() || InVec3.isUndef()) || InVec1 == InVec3))
+ X86Opcode = X86ISD::HSUB;
+ else
+ CanFold = false;
+
+ if (CanFold) {
+ // Do not try to expand this build_vector into a pair of horizontal
+ // add/sub if we can emit a pair of scalar add/sub.
+ if (NumUndefsLO + 1 == Half || NumUndefsHI + 1 == Half)
+ return SDValue();
+
+ // Convert this build_vector into a pair of horizontal binops followed by
+ // a concat vector. We must adjust the outputs from the partial horizontal
+ // matching calls above to account for undefined vector halves.
+ SDValue V0 = InVec0.isUndef() ? InVec2 : InVec0;
+ SDValue V1 = InVec1.isUndef() ? InVec3 : InVec1;
+ assert((!V0.isUndef() || !V1.isUndef()) && "Horizontal-op of undefs?");
+ bool isUndefLO = NumUndefsLO == Half;
+ bool isUndefHI = NumUndefsHI == Half;
+ return ExpandHorizontalBinOp(V0, V1, DL, DAG, X86Opcode, false, isUndefLO,
+ isUndefHI);
+ }
+ }
+
+ if (VT == MVT::v8f32 || VT == MVT::v4f64 || VT == MVT::v8i32 ||
+ VT == MVT::v16i16) {
+ unsigned X86Opcode;
+ if (isHorizontalBinOpPart(BV, ISD::ADD, DL, DAG, 0, NumElts, InVec0,
+ InVec1))
+ X86Opcode = X86ISD::HADD;
+ else if (isHorizontalBinOpPart(BV, ISD::SUB, DL, DAG, 0, NumElts, InVec0,
+ InVec1))
+ X86Opcode = X86ISD::HSUB;
+ else if (isHorizontalBinOpPart(BV, ISD::FADD, DL, DAG, 0, NumElts, InVec0,
+ InVec1))
+ X86Opcode = X86ISD::FHADD;
+ else if (isHorizontalBinOpPart(BV, ISD::FSUB, DL, DAG, 0, NumElts, InVec0,
+ InVec1))
+ X86Opcode = X86ISD::FHSUB;
+ else
+ return SDValue();
+
+ // Don't try to expand this build_vector into a pair of horizontal add/sub
+ // if we can simply emit a pair of scalar add/sub.
+ if (NumUndefsLO + 1 == Half || NumUndefsHI + 1 == Half)
+ return SDValue();
+
+ // Convert this build_vector into two horizontal add/sub followed by
+ // a concat vector.
+ bool isUndefLO = NumUndefsLO == Half;
+ bool isUndefHI = NumUndefsHI == Half;
+ return ExpandHorizontalBinOp(InVec0, InVec1, DL, DAG, X86Opcode, true,
+ isUndefLO, isUndefHI);
+ }
+
+ return SDValue();
+}
+
static SDValue LowerShift(SDValue Op, const X86Subtarget &Subtarget,
SelectionDAG &DAG);
@@ -9385,11 +9676,16 @@ LowerBUILD_VECTORAsVariablePermute(SDValue V, const SDLoc &DL,
const X86Subtarget &Subtarget) {
SDValue SrcVec, IndicesVec;
+ auto PeekThroughFreeze = [](SDValue N) {
+ if (N->getOpcode() == ISD::FREEZE && N.hasOneUse())
+ return N->getOperand(0);
+ return N;
+ };
// Check for a match of the permute source vector and permute index elements.
// This is done by checking that the i-th build_vector operand is of the form:
// (extract_elt SrcVec, (extract_elt IndicesVec, i)).
for (unsigned Idx = 0, E = V.getNumOperands(); Idx != E; ++Idx) {
- SDValue Op = peekThroughOneUseFreeze(V.getOperand(Idx));
+ SDValue Op = PeekThroughFreeze(V.getOperand(Idx));
if (Op.getOpcode() != ISD::EXTRACT_VECTOR_ELT)
return SDValue();
@@ -9540,6 +9836,8 @@ X86TargetLowering::LowerBUILD_VECTOR(SDValue Op, SelectionDAG &DAG) const {
if (SDValue AddSub = lowerToAddSubOrFMAddSub(BV, dl, Subtarget, DAG))
return AddSub;
+ if (SDValue HorizontalOp = LowerToHorizontalOp(BV, dl, Subtarget, DAG))
+ return HorizontalOp;
if (SDValue Broadcast = lowerBuildVectorAsBroadcast(BV, dl, Subtarget, DAG))
return Broadcast;
if (SDValue BitOp = lowerBuildVectorToBitOp(BV, dl, Subtarget, DAG))
@@ -10602,16 +10900,19 @@ static SDValue lowerShuffleWithPSHUFB(const SDLoc &DL, MVT VT,
(Subtarget.hasAVX2() && VT.is256BitVector()) ||
(Subtarget.hasBWI() && VT.is512BitVector()));
- SmallVector<int, 64> PSHUFBMask(NumBytes, -1);
+ SmallVector<SDValue, 64> PSHUFBMask(NumBytes);
+ // Sign bit set in i8 mask means zero element.
+ SDValue ZeroMask = DAG.getConstant(0x80, DL, MVT::i8);
+
SDValue V;
for (int i = 0; i < NumBytes; ++i) {
int M = Mask[i / NumEltBytes];
- if (M < 0)
+ if (M < 0) {
+ PSHUFBMask[i] = DAG.getUNDEF(MVT::i8);
continue;
-
+ }
if (Zeroable[i / NumEltBytes]) {
- // Sign bit set in i8 mask means zero element.
- PSHUFBMask[i] = 0x80;
+ PSHUFBMask[i] = ZeroMask;
continue;
}
@@ -10628,14 +10929,14 @@ static SDValue lowerShuffleWithPSHUFB(const SDLoc &DL, MVT VT,
M = M % LaneSize;
M = M * NumEltBytes + (i % NumEltBytes);
- PSHUFBMask[i] = M;
+ PSHUFBMask[i] = DAG.getConstant(M, DL, MVT::i8);
}
assert(V && "Failed to find a source input");
MVT I8VT = MVT::getVectorVT(MVT::i8, NumBytes);
- SDValue R = getConstVector(PSHUFBMask, I8VT, DAG, DL, /*IsMask=*/true);
- R = DAG.getNode(X86ISD::PSHUFB, DL, I8VT, DAG.getBitcast(I8VT, V), R);
- return DAG.getBitcast(VT, R);
+ return DAG.getBitcast(
+ VT, DAG.getNode(X86ISD::PSHUFB, DL, I8VT, DAG.getBitcast(I8VT, V),
+ DAG.getBuildVector(I8VT, DL, PSHUFBMask)));
}
/// Return Mask with the necessary casting or extending
@@ -11045,8 +11346,15 @@ static SDValue lowerShuffleAsVTRUNC(const SDLoc &DL, MVT VT, SDValue V1,
MVT SrcSVT = MVT::getIntegerVT(SrcEltBits);
MVT SrcVT = MVT::getVectorVT(SrcSVT, NumSrcElts);
- Src = getTargetVShiftByConstNode(X86ISD::VSRLI, DL, SrcVT, Src,
- Offset * EltSizeInBits, DAG);
+ Src = DAG.getBitcast(SrcVT, Src);
+
+ // Shift the offset'd elements into place for the truncation.
+ // TODO: Use getTargetVShiftByConstNode.
+ if (Offset)
+ Src = DAG.getNode(
+ X86ISD::VSRLI, DL, SrcVT, Src,
+ DAG.getTargetConstant(Offset * EltSizeInBits, DL, MVT::i8));
+
return getAVX512TruncNode(DL, VT, Src, Subtarget, DAG, !UndefUppers);
}
}
@@ -11253,30 +11561,48 @@ static SDValue lowerShuffleWithPACK(const SDLoc &DL, MVT VT, SDValue V1,
/// one of the inputs being zeroable.
static SDValue lowerShuffleAsBitMask(const SDLoc &DL, MVT VT, SDValue V1,
SDValue V2, ArrayRef<int> Mask,
- const APInt &Zeroable, SelectionDAG &DAG) {
- unsigned EltSizeInBIts = VT.getScalarSizeInBits();
- APInt Zero = APInt::getZero(EltSizeInBIts);
- APInt AllOnes = APInt::getAllOnes(EltSizeInBIts);
+ const APInt &Zeroable,
+ const X86Subtarget &Subtarget,
+ SelectionDAG &DAG) {
+ MVT MaskVT = VT;
+ MVT EltVT = VT.getVectorElementType();
+ SDValue Zero, AllOnes;
+ // Use f64 if i64 isn't legal.
+ if (EltVT == MVT::i64 && !Subtarget.is64Bit()) {
+ EltVT = MVT::f64;
+ MaskVT = MVT::getVectorVT(EltVT, Mask.size());
+ }
- SmallVector<APInt, 16> VMaskOps(Mask.size(), Zero);
+ MVT LogicVT = VT;
+ if (EltVT.isFloatingPoint()) {
+ Zero = DAG.getConstantFP(0.0, DL, EltVT);
+ APFloat AllOnesValue = APFloat::getAllOnesValue(EltVT.getFltSemantics());
+ AllOnes = DAG.getConstantFP(AllOnesValue, DL, EltVT);
+ LogicVT = MVT::getVectorVT(EltVT.changeTypeToInteger(), Mask.size());
+ } else {
+ Zero = DAG.getConstant(0, DL, EltVT);
+ AllOnes = DAG.getAllOnesConstant(DL, EltVT);
+ }
+
+ SmallVector<SDValue, 16> VMaskOps(Mask.size(), Zero);
SDValue V;
- for (int I = 0, Size = Mask.size(); I != Size; ++I) {
- if (Zeroable[I])
+ for (int i = 0, Size = Mask.size(); i < Size; ++i) {
+ if (Zeroable[i])
continue;
- if (Mask[I] % Size != I)
+ if (Mask[i] % Size != i)
return SDValue(); // Not a blend.
if (!V)
- V = Mask[I] < Size ? V1 : V2;
- else if (V != (Mask[I] < Size ? V1 : V2))
+ V = Mask[i] < Size ? V1 : V2;
+ else if (V != (Mask[i] < Size ? V1 : V2))
return SDValue(); // Can only let one input through the mask.
- VMaskOps[I] = AllOnes;
+ VMaskOps[i] = AllOnes;
}
if (!V)
return SDValue(); // No non-zeroable elements!
- MVT LogicVT = VT.changeTypeToInteger();
- SDValue VMask = getConstVector(VMaskOps, LogicVT, DAG, DL);
+ SDValue VMask = DAG.getBuildVector(MaskVT, DL, VMaskOps);
+ VMask = DAG.getBitcast(LogicVT, VMask);
V = DAG.getBitcast(LogicVT, V);
SDValue And = DAG.getNode(ISD::AND, DL, LogicVT, V, VMask);
return DAG.getBitcast(VT, And);
@@ -11291,16 +11617,17 @@ static SDValue lowerShuffleAsBitBlend(const SDLoc &DL, MVT VT, SDValue V1,
SDValue V2, ArrayRef<int> Mask,
SelectionDAG &DAG) {
assert(VT.isInteger() && "Only supports integer vector types!");
- unsigned EltSizeInBIts = VT.getScalarSizeInBits();
- APInt Zero = APInt::getZero(EltSizeInBIts);
- APInt AllOnes = APInt::getAllOnes(EltSizeInBIts);
- SmallVector<APInt, 16> MaskOps;
+ MVT EltVT = VT.getVectorElementType();
+ SDValue Zero = DAG.getConstant(0, DL, EltVT);
+ SDValue AllOnes = DAG.getAllOnesConstant(DL, EltVT);
+ SmallVector<SDValue, 16> MaskOps;
for (int i = 0, Size = Mask.size(); i < Size; ++i) {
if (Mask[i] >= 0 && Mask[i] != i && Mask[i] != i + Size)
return SDValue(); // Shuffled input!
MaskOps.push_back(Mask[i] < Size ? AllOnes : Zero);
}
- SDValue V1Mask = getConstVector(MaskOps, VT, DAG, DL);
+
+ SDValue V1Mask = DAG.getBuildVector(VT, DL, MaskOps);
return getBitSelect(DL, VT, V1, V2, V1Mask, DAG);
}
@@ -11466,8 +11793,8 @@ static SDValue lowerShuffleAsBlend(const SDLoc &DL, MVT VT, SDValue V1,
assert(Subtarget.hasSSE41() && "128-bit byte-blends require SSE41!");
// Attempt to lower to a bitmask if we can. VPAND is faster than VPBLENDVB.
- if (SDValue Masked =
- lowerShuffleAsBitMask(DL, VT, V1, V2, Mask, Zeroable, DAG))
+ if (SDValue Masked = lowerShuffleAsBitMask(DL, VT, V1, V2, Mask, Zeroable,
+ Subtarget, DAG))
return Masked;
if (Subtarget.hasBWI() && Subtarget.hasVLX()) {
@@ -11533,8 +11860,8 @@ static SDValue lowerShuffleAsBlend(const SDLoc &DL, MVT VT, SDValue V1,
// Attempt to lower to a bitmask if we can. Only if not optimizing for size.
bool OptForSize = DAG.shouldOptForSize();
if (!OptForSize) {
- if (SDValue Masked =
- lowerShuffleAsBitMask(DL, VT, V1, V2, Mask, Zeroable, DAG))
+ if (SDValue Masked = lowerShuffleAsBitMask(DL, VT, V1, V2, Mask, Zeroable,
+ Subtarget, DAG))
return Masked;
}
@@ -12080,12 +12407,14 @@ static SDValue lowerShuffleAsBitRotate(const SDLoc &DL, MVT VT, SDValue V1,
if (!IsLegal) {
if ((RotateAmt % 16) == 0)
return SDValue();
+ // TODO: Use getTargetVShiftByConstNode.
unsigned ShlAmt = RotateAmt;
unsigned SrlAmt = RotateVT.getScalarSizeInBits() - RotateAmt;
- SDValue SHL = getTargetVShiftByConstNode(X86ISD::VSHLI, DL, RotateVT, V1,
- ShlAmt, DAG);
- SDValue SRL = getTargetVShiftByConstNode(X86ISD::VSRLI, DL, RotateVT, V1,
- SrlAmt, DAG);
+ V1 = DAG.getBitcast(RotateVT, V1);
+ SDValue SHL = DAG.getNode(X86ISD::VSHLI, DL, RotateVT, V1,
+ DAG.getTargetConstant(ShlAmt, DL, MVT::i8));
+ SDValue SRL = DAG.getNode(X86ISD::VSRLI, DL, RotateVT, V1,
+ DAG.getTargetConstant(SrlAmt, DL, MVT::i8));
SDValue Rot = DAG.getNode(ISD::OR, DL, RotateVT, SHL, SRL);
return DAG.getBitcast(VT, Rot);
}
@@ -12268,29 +12597,18 @@ static SDValue lowerShuffleAsVALIGN(const SDLoc &DL, MVT VT, SDValue V1,
const APInt &Zeroable,
const X86Subtarget &Subtarget,
SelectionDAG &DAG) {
- unsigned EltBits = VT.getScalarSizeInBits();
- if (EltBits != 32 && EltBits != 64)
- return SDValue();
-
- MVT AlignVT = VT.changeVectorElementTypeToInteger();
+ assert((VT.getScalarType() == MVT::i32 || VT.getScalarType() == MVT::i64) &&
+ "Only 32-bit and 64-bit elements are supported!");
// 128/256-bit vectors are only supported with VLX.
- assert((Subtarget.hasVLX() ||
- (!AlignVT.is128BitVector() && !AlignVT.is256BitVector())) &&
- "VLX required for 128/256-bit vectors");
-
- auto emitVALIGN = [&](SDValue Lo, SDValue Hi, unsigned Imm) -> SDValue {
- SDValue AlignLo = VT.isFloatingPoint() ? DAG.getBitcast(AlignVT, Lo) : Lo;
- SDValue AlignHi = VT.isFloatingPoint() ? DAG.getBitcast(AlignVT, Hi) : Hi;
- SDValue Res = DAG.getNode(X86ISD::VALIGN, DL, AlignVT, AlignLo, AlignHi,
- DAG.getTargetConstant(Imm, DL, MVT::i8));
- return VT.isFloatingPoint() ? DAG.getBitcast(VT, Res) : Res;
- };
+ assert((Subtarget.hasVLX() || (!VT.is128BitVector() && !VT.is256BitVector()))
+ && "VLX required for 128/256-bit vectors");
SDValue Lo = V1, Hi = V2;
int Rotation = matchShuffleAsElementRotate(Lo, Hi, Mask);
if (0 < Rotation)
- return emitVALIGN(Lo, Hi, Rotation);
+ return DAG.getNode(X86ISD::VALIGN, DL, VT, Lo, Hi,
+ DAG.getTargetConstant(Rotation, DL, MVT::i8));
// See if we can use VALIGN as a cross-lane version of VSHLDQ/VSRLDQ.
// TODO: Pull this out as a matchShuffleAsElementShift helper?
@@ -12307,16 +12625,18 @@ static SDValue lowerShuffleAsVALIGN(const SDLoc &DL, MVT VT, SDValue V1,
SDValue Src = Mask[ZeroLo] < (int)NumElts ? V1 : V2;
int Low = Mask[ZeroLo] < (int)NumElts ? 0 : NumElts;
if (isSequentialOrUndefInRange(Mask, ZeroLo, NumElts - ZeroLo, Low))
- return emitVALIGN(Src, getZeroVector(AlignVT, Subtarget, DAG, DL),
- NumElts - ZeroLo);
+ return DAG.getNode(X86ISD::VALIGN, DL, VT, Src,
+ getZeroVector(VT, Subtarget, DAG, DL),
+ DAG.getTargetConstant(NumElts - ZeroLo, DL, MVT::i8));
}
if (ZeroHi) {
SDValue Src = Mask[0] < (int)NumElts ? V1 : V2;
int Low = Mask[0] < (int)NumElts ? 0 : NumElts;
if (isSequentialOrUndefInRange(Mask, 0, NumElts - ZeroHi, Low + ZeroHi))
- return emitVALIGN(getZeroVector(AlignVT, Subtarget, DAG, DL), Src,
- ZeroHi);
+ return DAG.getNode(X86ISD::VALIGN, DL, VT,
+ getZeroVector(VT, Subtarget, DAG, DL), Src,
+ DAG.getTargetConstant(ZeroHi, DL, MVT::i8));
}
return SDValue();
@@ -12509,42 +12829,6 @@ static SDValue lowerShuffleAsShift(const SDLoc &DL, MVT VT, SDValue V1,
return DAG.getBitcast(VT, V);
}
-/// Try to match a vector shuffle as a X86ISD::VSHLD funnel shift.
-static int matchShuffleAsVSHLD(MVT &ShiftVT, SDValue &V1, SDValue &V2,
- unsigned ScalarSizeInBits, ArrayRef<int> Mask) {
- assert(isPowerOf2_32(ScalarSizeInBits) && ScalarSizeInBits >= 8 &&
- "Unexpected element size");
- int Size = Mask.size();
- if (llvm::is_contained(Mask, SM_SentinelZero))
- return -1;
-
- SmallVector<int, 32> FunnelMask(Size);
- for (int Scale = 2; (Scale * ScalarSizeInBits) <= 64; Scale *= 2) {
- for (int Shift = 1; Shift != Scale; ++Shift) {
- for (int Elt = 0; Elt != Size; Elt += Scale) {
- std::iota(FunnelMask.begin() + Elt, FunnelMask.begin() + Elt + Shift,
- Elt + Size + (Scale - Shift));
- std::iota(FunnelMask.begin() + Elt + Shift,
- FunnelMask.begin() + Elt + Scale, Elt);
- }
- if (isShuffleEquivalent(Mask, FunnelMask)) {
- MVT ShiftSVT = MVT::getIntegerVT(ScalarSizeInBits * Scale);
- ShiftVT = MVT::getVectorVT(ShiftSVT, Size / Scale);
- return Shift * ScalarSizeInBits;
- }
- ShuffleVectorSDNode::commuteMask(FunnelMask);
- if (isShuffleEquivalent(Mask, FunnelMask)) {
- MVT ShiftSVT = MVT::getIntegerVT(ScalarSizeInBits * Scale);
- ShiftVT = MVT::getVectorVT(ShiftSVT, Size / Scale);
- std::swap(V1, V2);
- return Shift * ScalarSizeInBits;
- }
- }
- }
-
- return -1;
-}
-
// EXTRQ: Extract Len elements from lower half of source, starting at Idx.
// Remainder of lower half result is zero and upper half is all undef.
static bool matchShuffleAsEXTRQ(MVT VT, SDValue &V1, SDValue &V2,
@@ -13028,12 +13312,6 @@ static bool isSoftF16(T VT, const X86Subtarget &Subtarget) {
(EltVT == MVT::f16 && !Subtarget.hasFP16());
}
-template<typename T>
-static bool isBF16orSoftF16(T VT, const X86Subtarget &Subtarget) {
- T EltVT = VT.getScalarType();
- return EltVT == MVT::bf16 || (EltVT == MVT::f16 && !Subtarget.hasFP16());
-}
-
/// Try to lower insertion of a single element into a zero vector.
///
/// This is a common pattern that we have especially efficient patterns to lower
@@ -14106,8 +14384,8 @@ static SDValue lowerV4I32Shuffle(const SDLoc &DL, ArrayRef<int> Mask,
Zeroable, Subtarget, DAG))
return Blend;
- if (SDValue Masked =
- lowerShuffleAsBitMask(DL, MVT::v4i32, V1, V2, Mask, Zeroable, DAG))
+ if (SDValue Masked = lowerShuffleAsBitMask(DL, MVT::v4i32, V1, V2, Mask,
+ Zeroable, Subtarget, DAG))
return Masked;
// Use dedicated unpack instructions for masks that match their pattern.
@@ -14822,8 +15100,8 @@ static SDValue lowerV8I16Shuffle(const SDLoc &DL, ArrayRef<int> Mask,
Zeroable, Subtarget, DAG))
return Blend;
- if (SDValue Masked =
- lowerShuffleAsBitMask(DL, MVT::v8i16, V1, V2, Mask, Zeroable, DAG))
+ if (SDValue Masked = lowerShuffleAsBitMask(DL, MVT::v8i16, V1, V2, Mask,
+ Zeroable, Subtarget, DAG))
return Masked;
// Use dedicated unpack instructions for masks that match their pattern.
@@ -15178,8 +15456,8 @@ static SDValue lowerV16I8Shuffle(const SDLoc &DL, ArrayRef<int> Mask,
return V;
}
- if (SDValue Masked =
- lowerShuffleAsBitMask(DL, MVT::v16i8, V1, V2, Mask, Zeroable, DAG))
+ if (SDValue Masked = lowerShuffleAsBitMask(DL, MVT::v16i8, V1, V2, Mask,
+ Zeroable, Subtarget, DAG))
return Masked;
// Use dedicated unpack instructions for masks that match their pattern.
@@ -17517,8 +17795,8 @@ static SDValue lower256BitShuffle(const SDLoc &DL, ArrayRef<int> Mask, MVT VT,
if (ElementBits < 32) {
// No floating point type available, if we can't use the bit operations
// for masking/blending then decompose into 128-bit vectors.
- if (SDValue V =
- lowerShuffleAsBitMask(DL, VT, V1, V2, Mask, Zeroable, DAG))
+ if (SDValue V = lowerShuffleAsBitMask(DL, VT, V1, V2, Mask, Zeroable,
+ Subtarget, DAG))
return V;
if (SDValue V = lowerShuffleAsBitBlend(DL, VT, V1, V2, Mask, DAG))
return V;
@@ -17715,12 +17993,6 @@ static SDValue lowerV8F64Shuffle(const SDLoc &DL, ArrayRef<int> Mask,
Zeroable, Subtarget, DAG))
return Blend;
- // Try to use VALIGN via integer domain bitcast. Avoids VPERMPD which
- // requires an extra register for the index vector; VALIGNQ uses an immediate.
- if (SDValue Rotate = lowerShuffleAsVALIGN(DL, MVT::v8f64, V1, V2, Mask,
- Zeroable, Subtarget, DAG))
- return Rotate;
-
return lowerShuffleWithPERMV(DL, MVT::v8f64, Mask, V1, V2, Subtarget, DAG);
}
@@ -17788,12 +18060,6 @@ static SDValue lowerV16F32Shuffle(const SDLoc &DL, ArrayRef<int> Mask,
Zeroable, Subtarget, DAG))
return V;
- // Try to use VALIGN via integer domain bitcast. Avoids VPERMPS which
- // requires an extra register for the index vector; VALIGND uses an immediate.
- if (SDValue Rotate = lowerShuffleAsVALIGN(DL, MVT::v16f32, V1, V2, Mask,
- Zeroable, Subtarget, DAG))
- return Rotate;
-
return lowerShuffleWithPERMV(DL, MVT::v16f32, Mask, V1, V2, Subtarget, DAG);
}
@@ -18081,8 +18347,8 @@ static SDValue lowerV64I8Shuffle(const SDLoc &DL, ArrayRef<int> Mask,
return Rotate;
// Lower as AND if possible.
- if (SDValue Masked =
- lowerShuffleAsBitMask(DL, MVT::v64i8, V1, V2, Mask, Zeroable, DAG))
+ if (SDValue Masked = lowerShuffleAsBitMask(DL, MVT::v64i8, V1, V2, Mask,
+ Zeroable, Subtarget, DAG))
return Masked;
if (SDValue PSHUFB = lowerShuffleWithPSHUFB(DL, MVT::v64i8, Mask, V1, V2,
@@ -18176,7 +18442,8 @@ static SDValue lower512BitShuffle(const SDLoc &DL, ArrayRef<int> Mask,
if ((VT == MVT::v32i16 || VT == MVT::v64i8) && !Subtarget.hasBWI()) {
// Try using bit ops for masking and blending before falling back to
// splitting.
- if (SDValue V = lowerShuffleAsBitMask(DL, VT, V1, V2, Mask, Zeroable, DAG))
+ if (SDValue V = lowerShuffleAsBitMask(DL, VT, V1, V2, Mask, Zeroable,
+ Subtarget, DAG))
return V;
if (SDValue V = lowerShuffleAsBitBlend(DL, VT, V1, V2, Mask, DAG))
return V;
@@ -19520,18 +19787,10 @@ static SDValue LowerFLDEXP(SDValue Op, const X86Subtarget &Subtarget,
return splitVectorOp(Op, DAG, DL);
}
SDValue WideX = widenSubVector(X, true, Subtarget, DAG, DL, 512);
- // Widen Exp to the same *lane count* as WideX (not necessarily 512 bits) so
- // SINT_TO_FP has matching vector lengths. For wide f64 the int exponent
- // vector is narrower than 512 bits (e.g. v2i32 -> v8i32 to match v8f64).
- MVT WideExpVT =
- MVT::getVectorVT(Exp.getSimpleValueType().getVectorElementType(),
- WideX.getValueType().getVectorNumElements());
- SDValue WideExp = widenSubVector(WideExpVT, Exp, /*ZeroNewElements=*/true,
- Subtarget, DAG, DL);
- SDValue WideExpFp =
- DAG.getNode(ISD::SINT_TO_FP, DL, WideX.getValueType(), WideExp);
+ SDValue WideExp = widenSubVector(Exp, true, Subtarget, DAG, DL, 512);
+ Exp = DAG.getNode(ISD::SINT_TO_FP, DL, WideExp.getSimpleValueType(), Exp);
SDValue Scalef =
- DAG.getNode(X86ISD::SCALEF, DL, WideX.getValueType(), WideX, WideExpFp);
+ DAG.getNode(X86ISD::SCALEF, DL, WideX.getValueType(), WideX, WideExp);
SDValue Final =
DAG.getExtractSubvector(DL, X.getSimpleValueType(), Scalef, 0);
return DAG.getFPExtendOrRound(Final, DL, XTy);
@@ -20487,7 +20746,7 @@ static SDValue promoteXINT_TO_FP(SDValue Op, const SDLoc &dl,
SDValue Src = Op.getOperand(IsStrict ? 1 : 0);
SDValue Chain = IsStrict ? Op->getOperand(0) : DAG.getEntryNode();
MVT VT = Op.getSimpleValueType();
- MVT NVT = VT.changeElementType(MVT::f32);
+ MVT NVT = VT.isVector() ? VT.changeVectorElementType(MVT::f32) : MVT::f32;
SDValue Rnd = DAG.getIntPtrConstant(0, dl, /*isTarget=*/true);
if (IsStrict)
@@ -20534,7 +20793,7 @@ SDValue X86TargetLowering::LowerSINT_TO_FP(SDValue Op,
MVT VT = Op.getSimpleValueType();
SDLoc dl(Op);
- if (isBF16orSoftF16(VT, Subtarget))
+ if (isSoftF16(VT, Subtarget))
return promoteXINT_TO_FP(Op, dl, DAG);
else if (isLegalConversion(SrcVT, VT, true, Subtarget))
return Op;
@@ -20666,7 +20925,7 @@ std::pair<SDValue, SDValue> X86TargetLowering::BuildFILD(
/// Horizontal vector math instructions may be slower than normal math with
/// shuffles. Limit horizontal op codegen based on size/speed trade-offs, uarch
/// implementation, and likely shuffle complexity of the alternate sequence.
-static bool shouldUseHorizontalOp(bool IsSingleSource, const SelectionDAG &DAG,
+static bool shouldUseHorizontalOp(bool IsSingleSource, SelectionDAG &DAG,
const X86Subtarget &Subtarget) {
bool IsOptimizingSize = DAG.shouldOptForSize();
bool HasFastHOps = Subtarget.hasFastHorizontalOps();
@@ -21034,32 +21293,32 @@ SDValue X86TargetLowering::LowerUINT_TO_FP(SDValue Op,
bool IsStrict = Op->isStrictFPOpcode();
unsigned OpNo = IsStrict ? 1 : 0;
SDValue Src = Op.getOperand(OpNo);
- SDValue Chain = IsStrict ? Op.getOperand(0) : DAG.getEntryNode();
+ SDLoc dl(Op);
auto PtrVT = getPointerTy(DAG.getDataLayout());
MVT SrcVT = Src.getSimpleValueType();
MVT DstVT = Op->getSimpleValueType(0);
- SDLoc dl(Op);
+ SDValue Chain = IsStrict ? Op.getOperand(0) : DAG.getEntryNode();
+
+ // Bail out when we don't have native conversion instructions.
+ if (DstVT == MVT::f128)
+ return SDValue();
- if (isBF16orSoftF16(DstVT, Subtarget))
+ if (isSoftF16(DstVT, Subtarget))
return promoteXINT_TO_FP(Op, dl, DAG);
else if (isLegalConversion(SrcVT, DstVT, false, Subtarget))
return Op;
- if (Subtarget.isTargetWin64() && SrcVT == MVT::i128)
- return LowerWin64_INT128_TO_FP(Op, DAG);
-
- if (SDValue Extract = vectorizeExtractedCast(Op, dl, DAG, Subtarget))
- return Extract;
-
if (SDValue V = lowerFPToIntToFP(Op, dl, DAG, Subtarget))
return V;
if (DstVT.isVector())
return lowerUINT_TO_FP_vec(Op, dl, DAG, Subtarget);
- // Bail out when we don't have native conversion instructions.
- if (DstVT == MVT::f128)
- return SDValue();
+ if (Subtarget.isTargetWin64() && SrcVT == MVT::i128)
+ return LowerWin64_INT128_TO_FP(Op, DAG);
+
+ if (SDValue Extract = vectorizeExtractedCast(Op, dl, DAG, Subtarget))
+ return Extract;
if (Subtarget.hasAVX512() && isScalarFPTypeInSSEReg(DstVT) &&
(SrcVT == MVT::i32 || (SrcVT == MVT::i64 && Subtarget.is64Bit()))) {
@@ -22044,8 +22303,9 @@ static SDValue expandFP_TO_UINT_SSE(MVT VT, SDValue Src, const SDLoc &dl,
return DAG.getNode(X86ISD::BLENDV, dl, VT, Small, Overflow, Small);
}
- SDValue IsOverflown = getTargetVShiftByConstNode(X86ISD::VSRAI, dl, VT, Small,
- DstBits - 1, DAG);
+ SDValue IsOverflown =
+ DAG.getNode(X86ISD::VSRAI, dl, VT, Small,
+ DAG.getTargetConstant(DstBits - 1, dl, MVT::i8));
return DAG.getNode(ISD::OR, dl, VT, Small,
DAG.getNode(ISD::AND, dl, VT, Big, IsOverflown));
}
@@ -22062,8 +22322,8 @@ SDValue X86TargetLowering::LowerFP_TO_INT(SDValue Op, SelectionDAG &DAG) const {
SDLoc dl(Op);
SDValue Res;
- if (isBF16orSoftF16(SrcVT, Subtarget)) {
- MVT NVT = VT.changeElementType(MVT::f32);
+ if (isSoftF16(SrcVT, Subtarget)) {
+ MVT NVT = VT.isVector() ? VT.changeVectorElementType(MVT::f32) : MVT::f32;
if (IsStrict)
return DAG.getNode(Op.getOpcode(), dl, {VT, MVT::Other},
{Chain, DAG.getNode(ISD::STRICT_FP_EXTEND, dl,
@@ -22506,21 +22766,13 @@ X86TargetLowering::LowerFP_TO_INT_SAT(SDValue Op, SelectionDAG &DAG) const {
EVT SrcVT = Src.getValueType();
EVT DstVT = Node->getValueType(0);
EVT TmpVT = DstVT;
- EVT SatVT = cast<VTSDNode>(Node->getOperand(1))->getVT();
-
- if (Subtarget.hasAVX10_2() && SrcVT.isVector() &&
- SrcVT.getVectorElementType() == MVT::bf16 && SatVT == MVT::i8) {
- MVT VecI16VT = SrcVT.getSimpleVT().changeVectorElementType(MVT::i16);
- SDValue Res = DAG.getNode(IsSigned ? X86ISD::CVTTP2IBS : X86ISD::CVTTP2IUBS,
- dl, VecI16VT, Src);
- return DAG.getNode(ISD::TRUNCATE, dl, DstVT, Res);
- }
// This code is only for floats and doubles. Fall back to generic code for
// anything else.
- if (!isScalarFPTypeInSSEReg(SrcVT) || isBF16orSoftF16(SrcVT, Subtarget))
+ if (!isScalarFPTypeInSSEReg(SrcVT) || isSoftF16(SrcVT, Subtarget))
return SDValue();
+ EVT SatVT = cast<VTSDNode>(Node->getOperand(1))->getVT();
unsigned SatWidth = SatVT.getScalarSizeInBits();
unsigned DstWidth = DstVT.getScalarSizeInBits();
unsigned TmpWidth = TmpVT.getScalarSizeInBits();
@@ -23488,7 +23740,7 @@ static bool matchScalarReduction(SDValue Op, ISD::NodeType BinOp,
return false;
SDValue Src = I->getOperand(0);
- auto M = SrcOpMap.find(Src);
+ DenseMap<SDValue, APInt>::iterator M = SrcOpMap.find(Src);
if (M == SrcOpMap.end()) {
VT = Src.getValueType();
// Quit if not the same type.
@@ -24415,7 +24667,8 @@ static SDValue incDecVectorConstant(SDValue V, SelectionDAG &DAG, bool IsInc,
MVT VT = V.getSimpleValueType();
MVT EltVT = VT.getVectorElementType();
unsigned NumElts = VT.getVectorNumElements();
- SmallVector<APInt, 8> NewVecC;
+ SmallVector<SDValue, 8> NewVecC;
+ SDLoc DL(V);
for (unsigned i = 0; i < NumElts; ++i) {
auto *Elt = dyn_cast<ConstantSDNode>(BV->getOperand(i));
if (!Elt || Elt->isOpaque() || Elt->getSimpleValueType(0) != EltVT)
@@ -24429,9 +24682,10 @@ static SDValue incDecVectorConstant(SDValue V, SelectionDAG &DAG, bool IsInc,
(!IsInc && EltC.isMinSignedValue())))
return SDValue();
- NewVecC.push_back(EltC + (IsInc ? 1 : -1));
+ NewVecC.push_back(DAG.getConstant(EltC + (IsInc ? 1 : -1), DL, EltVT));
}
- return getConstVector(NewVecC, VT, DAG, SDLoc(V));
+
+ return DAG.getBuildVector(VT, DL, NewVecC);
}
/// As another special case, use PSUBUS[BW] when it's profitable. E.g. for
@@ -24568,12 +24822,6 @@ static SDValue LowerVSETCC(SDValue Op, const X86Subtarget &Subtarget,
SDValue Cmp;
bool IsAlwaysSignaling;
unsigned SSECC = translateX86FSETCC(Cond, Op0, Op1, IsAlwaysSignaling);
- if ((Cond == ISD::SETO || Cond == ISD::SETUO) && Op0 != Op1) {
- if (DAG.isKnownNeverNaN(Op1))
- Op1 = Op0;
- else if (DAG.isKnownNeverNaN(Op0))
- Op0 = Op1;
- }
if (!Subtarget.hasAVX()) {
// TODO: We could use following steps to handle a quiet compare with
// signaling encodings.
@@ -25565,12 +25813,12 @@ SDValue X86TargetLowering::LowerSELECT(SDValue Op, SelectionDAG &DAG) const {
unsigned CondCode = Cond.getConstantOperandVal(0);
// Special handling for __builtin_ffs(X) - 1 pattern which looks like
- // (select (seteq X, 0), -1, (cttz_zero_poison X)). Disable the special
+ // (select (seteq X, 0), -1, (cttz_zero_undef X)). Disable the special
// handle to keep the CMP with 0. This should be removed by
// optimizeCompareInst by using the flags from the BSR/TZCNT used for the
- // cttz_zero_poison.
+ // cttz_zero_undef.
auto MatchFFSMinus1 = [&](SDValue Op1, SDValue Op2) {
- return (Op1.getOpcode() == ISD::CTTZ_ZERO_POISON && Op1.hasOneUse() &&
+ return (Op1.getOpcode() == ISD::CTTZ_ZERO_UNDEF && Op1.hasOneUse() &&
Op1.getOperand(0) == CmpOp0 && isAllOnesConstant(Op2));
};
if (Subtarget.canUseCMOV() && (VT == MVT::i32 || VT == MVT::i64) &&
@@ -25899,8 +26147,8 @@ static SDValue LowerEXTEND_VECTOR_INREG(SDValue Op,
Curr = DAG.getBitcast(DestVT, Curr);
unsigned SignExtShift = DestWidth - InSVT.getSizeInBits();
- SignExt = getTargetVShiftByConstNode(X86ISD::VSRAI, dl, DestVT, Curr,
- SignExtShift, DAG);
+ SignExt = DAG.getNode(X86ISD::VSRAI, dl, DestVT, Curr,
+ DAG.getTargetConstant(SignExtShift, dl, MVT::i8));
}
if (VT == MVT::v2i64) {
@@ -26512,12 +26760,11 @@ static SDValue LowerVACOPY(SDValue Op, const X86Subtarget &Subtarget,
const Value *DstSV = cast<SrcValueSDNode>(Op.getOperand(3))->getValue();
const Value *SrcSV = cast<SrcValueSDNode>(Op.getOperand(4))->getValue();
SDLoc DL(Op);
- Align Alignment = Align(Subtarget.isTarget64BitLP64() ? 8 : 4);
return DAG.getMemcpy(
Chain, DL, DstPtr, SrcPtr,
DAG.getIntPtrConstant(Subtarget.isTarget64BitLP64() ? 24 : 16, DL),
- Alignment, Alignment, /*isVolatile*/ false, false,
+ Align(Subtarget.isTarget64BitLP64() ? 8 : 4), /*isVolatile*/ false, false,
/*CI=*/nullptr, std::nullopt, MachinePointerInfo(DstSV),
MachinePointerInfo(SrcSV));
}
@@ -26530,7 +26777,7 @@ static SDValue getVectorMaskingNode(SDValue Op, SDValue Mask,
const X86Subtarget &Subtarget,
SelectionDAG &DAG) {
MVT VT = Op.getSimpleValueType();
- MVT MaskVT = VT.changeElementType(MVT::i1);
+ MVT MaskVT = MVT::getVectorVT(MVT::i1, VT.getVectorNumElements());
unsigned OpcodeSelect = ISD::VSELECT;
SDLoc dl(Op);
@@ -27308,7 +27555,7 @@ SDValue X86TargetLowering::LowerINTRINSIC_WO_CHAIN(SDValue Op,
return DAG.getNode(IntrData->Opc0, dl, Op.getValueType(), Src);
MVT SrcVT = Src.getSimpleValueType();
- MVT MaskVT = SrcVT.changeElementType(MVT::i1);
+ MVT MaskVT = MVT::getVectorVT(MVT::i1, SrcVT.getVectorNumElements());
Mask = getMaskNode(Mask, MaskVT, Subtarget, DAG, dl);
return DAG.getNode(IntrData->Opc1, dl, Op.getValueType(),
{Src, PassThru, Mask});
@@ -27323,7 +27570,7 @@ SDValue X86TargetLowering::LowerINTRINSIC_WO_CHAIN(SDValue Op,
return DAG.getNode(IntrData->Opc0, dl, Op.getValueType(), {Src, Src2});
MVT Src2VT = Src2.getSimpleValueType();
- MVT MaskVT = Src2VT.changeElementType(MVT::i1);
+ MVT MaskVT = MVT::getVectorVT(MVT::i1, Src2VT.getVectorNumElements());
Mask = getMaskNode(Mask, MaskVT, Subtarget, DAG, dl);
return DAG.getNode(IntrData->Opc1, dl, Op.getValueType(),
{Src, Src2, PassThru, Mask});
@@ -27351,7 +27598,7 @@ SDValue X86TargetLowering::LowerINTRINSIC_WO_CHAIN(SDValue Op,
else
Opc = IntrData->Opc1;
MVT SrcVT = Src.getSimpleValueType();
- MVT MaskVT = SrcVT.changeElementType(MVT::i1);
+ MVT MaskVT = MVT::getVectorVT(MVT::i1, SrcVT.getVectorNumElements());
Mask = getMaskNode(Mask, MaskVT, Subtarget, DAG, dl);
return DAG.getNode(Opc, dl, Op.getValueType(), Src, Rnd, PassThru, Mask);
}
@@ -27983,7 +28230,7 @@ bool X86::isExtendedSwiftAsyncFrameSupported(const X86Subtarget &Subtarget,
return false;
// 64-bit targets support extended Swift async frame setup,
// except for targets that use the windows 64 prologue.
- return !MF.getTarget().getMCAsmInfo().usesWindowsCFI();
+ return !MF.getTarget().getMCAsmInfo()->usesWindowsCFI();
}
static SDValue LowerINTRINSIC_W_CHAIN(SDValue Op, const X86Subtarget &Subtarget,
@@ -28530,7 +28777,7 @@ SDValue X86TargetLowering::LowerFRAMEADDR(SDValue Op, SelectionDAG &DAG) const {
MFI.setFrameAddressIsTaken(true);
- if (MF.getTarget().getMCAsmInfo().usesWindowsCFI()) {
+ if (MF.getTarget().getMCAsmInfo()->usesWindowsCFI()) {
// Depth > 0 makes no sense on targets which use Windows unwind codes. It
// is not possible to crawl up the stack without looking at the unwind codes
// simultaneously.
@@ -29121,7 +29368,7 @@ SDValue X86TargetLowering::LowerRESET_FPENV(SDValue Op,
MachinePointerInfo MPI =
MachinePointerInfo::getConstantPool(DAG.getMachineFunction());
MachineMemOperand *MMO = MF.getMachineMemOperand(
- MPI, MachineMemOperand::MOLoad, X87StateSize, Align(4));
+ MPI, MachineMemOperand::MOStore, X87StateSize, Align(4));
return createSetFPEnvNodes(Env, Chain, DL, MVT::i32, MMO, DAG, Subtarget);
}
@@ -29166,25 +29413,6 @@ SDValue getGFNICtrlMask(unsigned Opcode, SelectionDAG &DAG, const SDLoc &DL,
return DAG.getBuildVector(VT, DL, MaskBits);
}
-static APInt getGFNIByteAffine(const APInt &ByteToAffine, const APInt &Matrix64,
- const APInt &Addend8) {
- assert(ByteToAffine.getBitWidth() == 8 && "Byte input unexpected size!");
- assert(Addend8.getBitWidth() == 8 && "8-bit addend input unexpected size!");
- assert(Matrix64.getBitWidth() == 64 &&
- "64-bit matrix input unexpected size!");
-
- APInt ByteSplat = APInt::getSplat(64, ByteToAffine);
- ByteSplat &= Matrix64.byteSwap();
-
- // Cumulative parity
- for (unsigned I = 0; I != 3; ++I)
- ByteSplat ^= ByteSplat.lshr(1 << I);
- ByteSplat &= 0x0101010101010101ull;
-
- APInt Affined = APIntOps::ScaleBitMask(ByteSplat, 8);
- return Affined ^ Addend8;
-}
-
/// Lower a vector CTLZ using native supported vector CTLZ instruction.
//
// i8/i16 vector implemented using dword LZCNT vector instruction
@@ -29253,7 +29481,7 @@ static SDValue LowerVectorCTLZInRegLUT(SDValue Op, const SDLoc &DL,
SDValue Hi = DAG.getNode(ISD::SRL, DL, CurrVT, Op0, NibbleShift);
SDValue HiZ;
if (CurrVT.is512BitVector()) {
- MVT MaskVT = CurrVT.changeElementType(MVT::i1);
+ MVT MaskVT = MVT::getVectorVT(MVT::i1, CurrVT.getVectorNumElements());
HiZ = DAG.getSetCC(DL, MaskVT, Hi, Zero, ISD::SETEQ);
HiZ = DAG.getNode(ISD::SIGN_EXTEND, DL, CurrVT, HiZ);
} else {
@@ -29279,7 +29507,7 @@ static SDValue LowerVectorCTLZInRegLUT(SDValue Op, const SDLoc &DL,
// Check if the upper half of the input element is zero.
if (CurrVT.is512BitVector()) {
- MVT MaskVT = CurrVT.changeElementType(MVT::i1);
+ MVT MaskVT = MVT::getVectorVT(MVT::i1, CurrVT.getVectorNumElements());
HiZ = DAG.getSetCC(DL, MaskVT, DAG.getBitcast(CurrVT, Op0),
DAG.getBitcast(CurrVT, Zero), ISD::SETEQ);
HiZ = DAG.getNode(ISD::SIGN_EXTEND, DL, CurrVT, HiZ);
@@ -29385,11 +29613,6 @@ static SDValue LowerCTTZ(SDValue Op, const X86Subtarget &Subtarget,
SDLoc dl(Op);
bool NonZeroSrc = DAG.isKnownNeverZero(N0);
- // Default to expansion if CTPOP is legal.
- if (VT.isVector() &&
- DAG.getTargetLoweringInfo().isOperationLegal(ISD::CTPOP, VT))
- return SDValue();
-
// GFNI - isolate LSB and perform GF2P8AFFINEQB lookup.
if (Subtarget.hasGFNI() && VT.isVector() &&
VT.getVectorElementType() == MVT::i8) {
@@ -29422,69 +29645,6 @@ static SDValue LowerCTTZ(SDValue Op, const X86Subtarget &Subtarget,
return DAG.getNode(X86ISD::CMOV, dl, VT, Ops);
}
-static SDValue peekThroughDemandedElts(SDValue V, const APInt &DemandedElts,
- SelectionDAG &DAG) {
- const TargetLowering &TLI = DAG.getTargetLoweringInfo();
- while (SDValue NewV =
- TLI.SimplifyMultipleUseDemandedVectorElts(V, DemandedElts, DAG))
- V = NewV;
- return V;
-}
-
-// Generic x86 vector reduction expansion.
-static SDValue LowerVECREDUCE(SDValue Op, const X86Subtarget &Subtarget,
- SelectionDAG &DAG, bool AllowScalarization) {
- ISD::NodeType BinOp = ISD::getVecReduceBaseOpcode(Op.getOpcode());
- assert(DAG.getTargetLoweringInfo().isBinOp(BinOp) &&
- "Only binops expected to be used by reductions");
-
- EVT ExtractVT = Op.getValueType();
- SDValue Src = Op.getOperand(0);
- EVT SrcVT = Src.getValueType();
- EVT SrcSVT = SrcVT.getScalarType();
- const TargetLowering &TLI = DAG.getTargetLoweringInfo();
- SDLoc DL(Op);
-
- // TODO: Pad non-pow2 vectors with identity constants.
- if (SrcSVT != ExtractVT || SrcVT.getSizeInBits() < 128 ||
- !isPowerOf2_32(SrcVT.getVectorNumElements()))
- return SDValue();
-
- // Split vector down to 128-bits, performing bin to lo/hi subvectors.
- while (SrcVT.getSizeInBits() > 128) {
- SDValue Lo, Hi;
- std::tie(Lo, Hi) = splitVector(Src, DAG, DL);
- SrcVT = Lo.getValueType();
- Src = DAG.getNode(BinOp, DL, SrcVT, Lo, Hi);
- }
- assert(SrcVT.is128BitVector() && "Unexpected value type");
-
- // Expand 128-bit shuffle tree + reduction binops.
- unsigned NumSrcElts = SrcVT.getVectorNumElements();
- for (unsigned NumElts = NumSrcElts; NumElts != 1; NumElts /= 2) {
- // Scalarize the last 2 elements if the vector binop isn't legal.
- if (NumElts == 2 && AllowScalarization &&
- !TLI.isOperationLegal(BinOp, SrcVT) && TLI.isTypeLegal(ExtractVT)) {
- return DAG.getNode(BinOp, DL, ExtractVT,
- DAG.getExtractVectorElt(DL, ExtractVT, Src, 0),
- DAG.getExtractVectorElt(DL, ExtractVT, Src, 1));
- }
-
- // Peek through identity elements added by legalisation padding.
- APInt HiElts = APInt::getBitsSet(NumSrcElts, NumElts / 2, NumElts);
- if (!DAG.isIdentityElement(BinOp, SDNodeFlags(), Src, HiElts, 1)) {
- APInt LoElts = APInt::getLowBitsSet(NumSrcElts, NumElts / 2);
- SDValue Lo = peekThroughDemandedElts(Src, LoElts, DAG);
- SmallVector<int, 16> Mask(NumSrcElts, -1);
- std::iota(Mask.begin(), Mask.begin() + (NumElts / 2), NumElts / 2);
- SDValue Hi =
- DAG.getVectorShuffle(SrcVT, DL, Src, DAG.getUNDEF(SrcVT), Mask);
- Src = DAG.getNode(BinOp, DL, SrcVT, Lo, Hi);
- }
- }
- return DAG.getExtractVectorElt(DL, ExtractVT, Src, 0);
-}
-
static SDValue lowerAddSub(SDValue Op, SelectionDAG &DAG,
const X86Subtarget &Subtarget) {
MVT VT = Op.getSimpleValueType();
@@ -29509,6 +29669,20 @@ static SDValue LowerADDSAT_SUBSAT(SDValue Op, SelectionDAG &DAG,
unsigned Opcode = Op.getOpcode();
SDLoc DL(Op);
+ if (Opcode == ISD::USUBSAT && !VT.isVector() && Subtarget.hasNDD()) {
+ if (isOneConstant(Y)) {
+ // usub.sat(X,1) == (X==0 ? 0 : X-1). Lower to cmp+adc with NDD.
+ SDValue Sub = DAG.getNode(X86ISD::SUB, DL, DAG.getVTList(VT, MVT::i32), X,
+ DAG.getConstant(1, DL, VT));
+ SDValue EFLAGS = Sub.getValue(1);
+ SDValue MinusOne = DAG.getAllOnesConstant(DL, VT);
+ return DAG.getNode(X86ISD::ADC, DL, DAG.getVTList(VT, MVT::i32), X,
+ MinusOne, EFLAGS);
+ }
+ // Scalar USUBSAT was previously Expand. Don't fall through to vector path.
+ return SDValue();
+ }
+
if (VT == MVT::v32i16 || VT == MVT::v64i8 ||
(VT.is256BitVector() && !Subtarget.hasInt256())) {
assert(Op.getSimpleValueType().isInteger() &&
@@ -29672,80 +29846,6 @@ static SDValue LowerMINMAX(SDValue Op, const X86Subtarget &Subtarget,
return SDValue();
}
-// Attempt to replace an min/max v8i16/v16i8 horizontal reduction with
-// PHMINPOSUW.
-static SDValue LowerMINMAX_REDUCE(SDValue Op, const X86Subtarget &Subtarget,
- SelectionDAG &DAG) {
- EVT ExtractVT = Op.getValueType();
- bool AllowScalarization = !Subtarget.hasAVX512();
- if (!Subtarget.hasSSE41() || (ExtractVT != MVT::i16 && ExtractVT != MVT::i8))
- return LowerVECREDUCE(Op, Subtarget, DAG, AllowScalarization);
-
- SDValue Src = Op.getOperand(0);
- EVT SrcVT = Src.getValueType();
- EVT SrcSVT = SrcVT.getScalarType();
- ISD::NodeType BinOp = ISD::getVecReduceBaseOpcode(Op.getOpcode());
- if (SrcSVT != ExtractVT || SrcVT.getSizeInBits() < 128 ||
- !isPowerOf2_32(SrcVT.getVectorNumElements()))
- return SDValue();
-
- // Bail if at least the upper half elements are identity.
- APInt HiElts = APInt::getHighBitsSet(SrcVT.getVectorNumElements(),
- SrcVT.getVectorNumElements() / 2);
- if (DAG.isIdentityElement(BinOp, SDNodeFlags(), Src, HiElts, 1))
- return LowerVECREDUCE(Op, Subtarget, DAG, AllowScalarization);
-
- SDLoc DL(Op);
- SDValue MinPos = Src;
-
- // First, reduce the source down to 128-bit, applying BinOp to lo/hi.
- while (SrcVT.getSizeInBits() > 128) {
- SDValue Lo, Hi;
- std::tie(Lo, Hi) = splitVector(MinPos, DAG, DL);
- SrcVT = Lo.getValueType();
- MinPos = DAG.getNode(BinOp, DL, SrcVT, Lo, Hi);
- }
- assert(((SrcVT == MVT::v8i16 && ExtractVT == MVT::i16) ||
- (SrcVT == MVT::v16i8 && ExtractVT == MVT::i8)) &&
- "Unexpected value type");
-
- // PHMINPOSUW applies to UMIN(v8i16), for SMIN/SMAX/UMAX we must apply a mask
- // to flip the value accordingly.
- SDValue Mask;
- unsigned MaskEltsBits = ExtractVT.getSizeInBits();
- if (BinOp == ISD::SMAX)
- Mask = DAG.getConstant(APInt::getSignedMaxValue(MaskEltsBits), DL, SrcVT);
- else if (BinOp == ISD::SMIN)
- Mask = DAG.getConstant(APInt::getSignedMinValue(MaskEltsBits), DL, SrcVT);
- else if (BinOp == ISD::UMAX)
- Mask = DAG.getAllOnesConstant(DL, SrcVT);
-
- if (Mask)
- MinPos = DAG.getNode(ISD::XOR, DL, SrcVT, Mask, MinPos);
-
- // For v16i8 cases we need to perform UMIN on pairs of byte elements,
- // shuffling each upper element down and insert zeros. This means that the
- // v16i8 UMIN will leave the upper element as zero, performing zero-extension
- // ready for the PHMINPOS.
- if (ExtractVT == MVT::i8) {
- SDValue Upper = DAG.getVectorShuffle(
- SrcVT, DL, MinPos, DAG.getConstant(0, DL, MVT::v16i8),
- {1, 16, 3, 16, 5, 16, 7, 16, 9, 16, 11, 16, 13, 16, 15, 16});
- MinPos = DAG.getNode(ISD::UMIN, DL, SrcVT, MinPos, Upper);
- }
-
- // Perform the PHMINPOS on a v8i16 vector,
- MinPos = DAG.getBitcast(MVT::v8i16, MinPos);
- MinPos = DAG.getNode(X86ISD::PHMINPOS, DL, MVT::v8i16, MinPos);
- MinPos = DAG.getBitcast(SrcVT, MinPos);
-
- if (Mask)
- MinPos = DAG.getNode(ISD::XOR, DL, SrcVT, Mask, MinPos);
-
- return DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, ExtractVT, MinPos,
- DAG.getVectorIdxConstant(0, DL));
-}
-
static SDValue LowerFMINIMUM_FMAXIMUM(SDValue Op, const X86Subtarget &Subtarget,
SelectionDAG &DAG) {
const TargetLowering &TLI = DAG.getTargetLoweringInfo();
@@ -29843,8 +29943,8 @@ static SDValue LowerFMINIMUM_FMAXIMUM(SDValue Op, const X86Subtarget &Subtarget,
bool IsXNeverNaN = DAG.isKnownNeverNaN(X);
bool IsYNeverNaN = DAG.isKnownNeverNaN(Y);
bool IgnoreSignedZero = Op->getFlags().hasNoSignedZeros() ||
- DAG.isKnownNeverLogicalZero(X) ||
- DAG.isKnownNeverLogicalZero(Y);
+ DAG.isKnownNeverZeroFloat(X) ||
+ DAG.isKnownNeverZeroFloat(Y);
bool ShouldHandleZeros = true;
SDValue NewX = X;
SDValue NewY = Y;
@@ -30821,16 +30921,6 @@ static SDValue LowerShiftByScalarImmediate(SDValue Op, SelectionDAG &DAG,
return DAG.getNode(ISD::AND, dl, VT, SRL, DAG.getConstant(Mask, dl, VT));
}
if (Op.getOpcode() == ISD::SRA) {
- // ashr(R, 1) === xor(avgceilu(R,-1),and(not(R),MIN_SIGNED))
- if (ShiftAmt == 1) {
- R = DAG.getFreeze(R);
- SDValue AllOnes = DAG.getAllOnesConstant(dl, VT);
- SDValue Avg = DAG.getNode(ISD::AVGCEILU, dl, VT, R, AllOnes);
- SDValue Not = DAG.getNode(ISD::XOR, dl, VT, R, AllOnes);
- SDValue Hi =
- DAG.getNode(ISD::AND, dl, VT, Not, DAG.getConstant(0x80, dl, VT));
- return DAG.getNode(ISD::XOR, dl, VT, Avg, Hi);
- }
// ashr(R, Amt) === sub(xor(lshr(R, Amt), Mask), Mask)
SDValue Res = DAG.getNode(ISD::SRL, dl, VT, R, Amt);
@@ -31996,25 +32086,6 @@ static SDValue LowerRotate(SDValue Op, const X86Subtarget &Subtarget,
if (VT.is256BitVector() && (Subtarget.hasXOP() || !Subtarget.hasAVX2()))
return splitVectorIntBinary(Op, DAG, DL);
- // rotl(x,7) -> pavgb(x, 0 - (x & 1))
- if (EltSizeInBits == 8 && IsCstSplat &&
- CstSplatValue.urem(EltSizeInBits) == 7 && IsROTL &&
- !Subtarget.hasAVX512()) {
- SDValue One = DAG.getConstant(1, DL, VT);
- SDValue LSB = DAG.getNode(ISD::AND, DL, VT, R, One);
- SDValue Neg = DAG.getNegative(LSB, DL, VT);
- return DAG.getNode(ISD::AVGCEILU, DL, VT, R, Neg);
- }
-
- // rotl(x,1) -> sub(add(x, x), icmp_slt(x, 0))
- if (IsROTL && EltSizeInBits == 8 && IsCstSplat &&
- CstSplatValue.urem(EltSizeInBits) == 1 && !Subtarget.hasAVX512()) {
- SDValue Double = DAG.getNode(ISD::ADD, DL, VT, R, R);
- SDValue Zero = DAG.getConstant(0, DL, VT);
- SDValue CmpNeg = DAG.getSetCC(DL, VT, R, Zero, ISD::SETLT);
- return DAG.getNode(ISD::SUB, DL, VT, Double, CmpNeg);
- }
-
// Rotate by an uniform constant - expand back to shifts.
// TODO: Can't use generic expansion as UNDEF amt elements can be converted
// to other values when folded to shift amounts, losing the splat.
@@ -32434,12 +32505,8 @@ X86TargetLowering::shouldExpandLogicAtomicRMWInIR(
}
void X86TargetLowering::emitBitTestAtomicRMWIntrinsic(AtomicRMWInst *AI) const {
- LLVMContext &Ctx = AI->getContext();
- IRBuilder<ConstantFolder, IRBuilderCallbackInserter> Builder(
- Ctx, ConstantFolder{}, IRBuilderCallbackInserter([&AI](Instruction *I) {
- I->copyMetadata(*AI, LLVMContext::MD_pcsections);
- }));
- Builder.SetInsertPoint(AI);
+ IRBuilder<> Builder(AI);
+ Builder.CollectMetadataToCopy(AI, {LLVMContext::MD_pcsections});
Intrinsic::ID IID_C = Intrinsic::not_intrinsic;
Intrinsic::ID IID_I = Intrinsic::not_intrinsic;
switch (AI->getOperation()) {
@@ -32459,6 +32526,7 @@ void X86TargetLowering::emitBitTestAtomicRMWIntrinsic(AtomicRMWInst *AI) const {
break;
}
Instruction *I = AI->user_back();
+ LLVMContext &Ctx = AI->getContext();
Value *Addr = Builder.CreatePointerCast(AI->getPointerOperand(),
PointerType::getUnqual(Ctx));
Value *Result = nullptr;
@@ -32515,188 +32583,100 @@ void X86TargetLowering::emitBitTestAtomicRMWIntrinsic(AtomicRMWInst *AI) const {
AI->eraseFromParent();
}
-static X86::CondCode matchSignedNewValueCC(const Instruction *I) {
- using namespace llvm::PatternMatch;
- if (match(I->user_back(),
- m_SpecificICmp(CmpInst::ICMP_SLT, m_Value(), m_ZeroInt())))
- return X86::COND_S;
- if (match(I->user_back(),
- m_SpecificICmp(CmpInst::ICMP_SGT, m_Value(), m_AllOnes())))
- return X86::COND_NS;
- return X86::COND_INVALID;
-}
-
-static X86::CondCode matchAddCC(const AtomicRMWInst *AI, const Instruction *I) {
- using namespace llvm::PatternMatch;
- Value *Op = AI->getOperand(1);
- CmpPredicate Pred;
-
- // Folded from icmp eq/ne (old + Op), 0 to icmp eq/ne old, -Op.
- if (match(I, m_c_ICmp(Pred, m_Neg(m_Specific(Op)), m_Value()))) {
- if (Pred == CmpInst::ICMP_EQ)
- return X86::COND_E;
- if (Pred == CmpInst::ICMP_NE)
- return X86::COND_NE;
- }
-
- // Non-folded SF form: %new = add %old, Op; icmp slt/sgt %new, 0/-1
- // lock add sets SF on the new value directly.
- if (match(I, m_OneUse(m_c_Add(m_Specific(Op), m_Value()))))
- return matchSignedNewValueCC(I);
-
- return X86::COND_INVALID;
-}
-
-static X86::CondCode matchSubCC(const AtomicRMWInst *AI, const Instruction *I) {
+static bool shouldExpandCmpArithRMWInIR(const AtomicRMWInst *AI) {
using namespace llvm::PatternMatch;
- Value *Op = AI->getOperand(1);
- CmpPredicate Pred;
-
- // Folded from icmp eq/ne (old - Op), 0 to icmp eq/ne old, Op.
- if (match(I, m_c_ICmp(Pred, m_Specific(Op), m_Value()))) {
- if (Pred == CmpInst::ICMP_EQ)
- return X86::COND_E;
- if (Pred == CmpInst::ICMP_NE)
- return X86::COND_NE;
- }
-
- // Non-folded SF form: %new = sub %old, Op; icmp slt/sgt %new, 0/-1
- // lock sub sets SF on the new value directly.
- if (match(I, m_OneUse(m_Sub(m_Value(), m_Specific(Op)))))
- return matchSignedNewValueCC(I);
-
- return X86::COND_INVALID;
-}
+ if (!AI->hasOneUse())
+ return false;
-static X86::CondCode matchOrCC(const AtomicRMWInst *AI, const Instruction *I) {
- using namespace llvm::PatternMatch;
Value *Op = AI->getOperand(1);
CmpPredicate Pred;
-
- // Non-folded form: %new = or %old, Op; icmp P %new, 0/-1
- // lock or sets ZF/SF on the new value directly.
- if (!match(I, m_OneUse(m_c_Or(m_Specific(Op), m_Value()))))
- return X86::COND_INVALID;
-
- if (match(I->user_back(), m_ICmp(Pred, m_Value(), m_ZeroInt()))) {
- if (Pred == CmpInst::ICMP_EQ)
- return X86::COND_E;
- if (Pred == CmpInst::ICMP_NE)
- return X86::COND_NE;
- if (Pred == CmpInst::ICMP_SLT)
- return X86::COND_S;
+ const Instruction *I = AI->user_back();
+ AtomicRMWInst::BinOp Opc = AI->getOperation();
+ if (Opc == AtomicRMWInst::Add) {
+ if (match(I, m_c_ICmp(Pred, m_Sub(m_ZeroInt(), m_Specific(Op)), m_Value())))
+ return Pred == CmpInst::ICMP_EQ || Pred == CmpInst::ICMP_NE;
+ if (match(I, m_OneUse(m_c_Add(m_Specific(Op), m_Value())))) {
+ if (match(I->user_back(),
+ m_SpecificICmp(CmpInst::ICMP_SLT, m_Value(), m_ZeroInt())))
+ return true;
+ if (match(I->user_back(),
+ m_SpecificICmp(CmpInst::ICMP_SGT, m_Value(), m_AllOnes())))
+ return true;
+ }
+ return false;
}
- if (match(I->user_back(),
- m_SpecificICmp(CmpInst::ICMP_SGT, m_Value(), m_AllOnes())))
- return X86::COND_NS;
-
- return X86::COND_INVALID;
-}
-
-static X86::CondCode matchAndCC(const AtomicRMWInst *AI, const Instruction *I) {
- using namespace llvm::PatternMatch;
- Value *Op = AI->getOperand(1);
- CmpPredicate Pred;
-
- // Non-folded form: %new = and %old, Op; icmp P %new, 0/-1
- // lock and sets ZF/SF on the new value directly.
- if (match(I, m_OneUse(m_c_And(m_Specific(Op), m_Value())))) {
- if (match(I->user_back(), m_ICmp(Pred, m_Value(), m_ZeroInt()))) {
- if (Pred == CmpInst::ICMP_EQ)
- return X86::COND_E;
- if (Pred == CmpInst::ICMP_NE)
- return X86::COND_NE;
- if (Pred == CmpInst::ICMP_SLT)
- return X86::COND_S;
+ if (Opc == AtomicRMWInst::Sub) {
+ if (match(I, m_c_ICmp(Pred, m_Specific(Op), m_Value())))
+ return Pred == CmpInst::ICMP_EQ || Pred == CmpInst::ICMP_NE;
+ if (match(I, m_OneUse(m_Sub(m_Value(), m_Specific(Op))))) {
+ if (match(I->user_back(),
+ m_SpecificICmp(CmpInst::ICMP_SLT, m_Value(), m_ZeroInt())))
+ return true;
+ if (match(I->user_back(),
+ m_SpecificICmp(CmpInst::ICMP_SGT, m_Value(), m_AllOnes())))
+ return true;
}
+ return false;
+ }
+ if ((Opc == AtomicRMWInst::Or &&
+ match(I, m_OneUse(m_c_Or(m_Specific(Op), m_Value())))) ||
+ (Opc == AtomicRMWInst::And &&
+ match(I, m_OneUse(m_c_And(m_Specific(Op), m_Value()))))) {
+ if (match(I->user_back(), m_ICmp(Pred, m_Value(), m_ZeroInt())))
+ return Pred == CmpInst::ICMP_EQ || Pred == CmpInst::ICMP_NE ||
+ Pred == CmpInst::ICMP_SLT;
if (match(I->user_back(),
m_SpecificICmp(CmpInst::ICMP_SGT, m_Value(), m_AllOnes())))
- return X86::COND_NS;
- return X86::COND_INVALID;
- }
-
- // If -C is a power of 2:
- // (old & C) == 0 <=> old ult -C
- // (old & C) != 0 <=> old ugt ~C
- auto *CI = dyn_cast<ConstantInt>(Op);
- if (!CI)
- return X86::COND_INVALID;
- const APInt &C = CI->getValue();
- const APInt *K;
- if (!(-C).isPowerOf2() ||
- !match(I, m_c_ICmp(Pred, m_Specific(AI), m_APInt(K))))
- return X86::COND_INVALID;
- if (Pred == ICmpInst::ICMP_ULT && *K == -C)
- return X86::COND_E;
- if (Pred == ICmpInst::ICMP_UGT && *K == ~C)
- return X86::COND_NE;
- return X86::COND_INVALID;
-}
-
-static X86::CondCode matchXorCC(const AtomicRMWInst *AI, const Instruction *I) {
- using namespace llvm::PatternMatch;
- Value *Op = AI->getOperand(1);
- CmpPredicate Pred;
-
- // Folded from icmp eq/ne (old ^ Op), 0 to icmp eq/ne old, Op.
- if (match(I, m_c_ICmp(Pred, m_Specific(Op), m_Value()))) {
- if (Pred == CmpInst::ICMP_EQ)
- return X86::COND_E;
- if (Pred == CmpInst::ICMP_NE)
- return X86::COND_NE;
+ return true;
+ return false;
}
-
- // Non-folded SF form: %new = xor %old, Op; icmp slt/sgt %new, 0/-1
- // lock xor sets SF on the new value directly.
- if (match(I, m_OneUse(m_c_Xor(m_Specific(Op), m_Value()))))
- return matchSignedNewValueCC(I);
-
- return X86::COND_INVALID;
-}
-
-static X86::CondCode getCmpArithCC(const AtomicRMWInst *AI) {
- if (!AI->hasOneUse())
- return X86::COND_INVALID;
-
- const Instruction *I = AI->user_back();
- switch (AI->getOperation()) {
- case AtomicRMWInst::Add:
- return matchAddCC(AI, I);
- case AtomicRMWInst::Sub:
- return matchSubCC(AI, I);
- case AtomicRMWInst::Or:
- return matchOrCC(AI, I);
- case AtomicRMWInst::And:
- return matchAndCC(AI, I);
- case AtomicRMWInst::Xor:
- return matchXorCC(AI, I);
- default:
- return X86::COND_INVALID;
+ if (Opc == AtomicRMWInst::Xor) {
+ if (match(I, m_c_ICmp(Pred, m_Specific(Op), m_Value())))
+ return Pred == CmpInst::ICMP_EQ || Pred == CmpInst::ICMP_NE;
+ if (match(I, m_OneUse(m_c_Xor(m_Specific(Op), m_Value())))) {
+ if (match(I->user_back(),
+ m_SpecificICmp(CmpInst::ICMP_SLT, m_Value(), m_ZeroInt())))
+ return true;
+ if (match(I->user_back(),
+ m_SpecificICmp(CmpInst::ICMP_SGT, m_Value(), m_AllOnes())))
+ return true;
+ }
+ return false;
}
-}
-static bool shouldExpandCmpArithRMWInIR(const AtomicRMWInst *AI) {
- return getCmpArithCC(AI) != X86::COND_INVALID;
+ return false;
}
void X86TargetLowering::emitCmpArithAtomicRMWIntrinsic(
AtomicRMWInst *AI) const {
- LLVMContext &Ctx = AI->getContext();
- IRBuilder<ConstantFolder, IRBuilderCallbackInserter> Builder(
- Ctx, ConstantFolder{}, IRBuilderCallbackInserter([&AI](Instruction *I) {
- I->copyMetadata(*AI, LLVMContext::MD_pcsections);
- }));
- Builder.SetInsertPoint(AI);
+ IRBuilder<> Builder(AI);
+ Builder.CollectMetadataToCopy(AI, {LLVMContext::MD_pcsections});
Instruction *TempI = nullptr;
+ LLVMContext &Ctx = AI->getContext();
ICmpInst *ICI = dyn_cast<ICmpInst>(AI->user_back());
if (!ICI) {
TempI = AI->user_back();
assert(TempI->hasOneUse() && "Must have one use");
ICI = cast<ICmpInst>(TempI->user_back());
}
- X86::CondCode CC = getCmpArithCC(AI);
- assert(CC != X86::COND_INVALID && "emitCmpArithAtomicRMWIntrinsic called "
- "without a recognised pattern");
+ X86::CondCode CC = X86::COND_INVALID;
+ ICmpInst::Predicate Pred = ICI->getPredicate();
+ switch (Pred) {
+ default:
+ llvm_unreachable("Not supported Pred");
+ case CmpInst::ICMP_EQ:
+ CC = X86::COND_E;
+ break;
+ case CmpInst::ICMP_NE:
+ CC = X86::COND_NE;
+ break;
+ case CmpInst::ICMP_SLT:
+ CC = X86::COND_S;
+ break;
+ case CmpInst::ICMP_SGT:
+ CC = X86::COND_NS;
+ break;
+ }
Intrinsic::ID IID = Intrinsic::not_intrinsic;
switch (AI->getOperation()) {
default:
@@ -32796,12 +32776,8 @@ X86TargetLowering::lowerIdempotentRMWIntoFencedLoad(AtomicRMWInst *AI) const {
AI->use_empty())
return nullptr;
- IRBuilder<ConstantFolder, IRBuilderCallbackInserter> Builder(
- AI->getContext(), ConstantFolder{},
- IRBuilderCallbackInserter([&AI](Instruction *I) {
- I->copyMetadata(*AI, LLVMContext::MD_pcsections);
- }));
- Builder.SetInsertPoint(AI);
+ IRBuilder<> Builder(AI);
+ Builder.CollectMetadataToCopy(AI, {LLVMContext::MD_pcsections});
auto SSID = AI->getSyncScopeID();
// We must restrict the ordering to avoid generating loads with Release or
// ReleaseAcquire orderings.
@@ -33395,15 +33371,11 @@ static SDValue LowerBITREVERSE(SDValue Op, const X86Subtarget &Subtarget,
Res = DAG.getNode(ISD::BITREVERSE, DL, ByteVT, Res);
return DAG.getBitcast(VT, Res);
}
- assert(VT.isVectorOf(MVT::i8) && "Only byte vector BITREVERSE supported");
+ assert(VT.isVector() && VT.getScalarType() == MVT::i8 &&
+ "Only byte vector BITREVERSE supported");
unsigned NumElts = VT.getVectorNumElements();
- // If we have BMM, BITREVERSE on vXi8 is marked Legal and will be handled
- // by TableGen pattern matching to VPBITREVB instruction. We should not
- // reach here in that case.
- assert(!Subtarget.hasBMM() && "BMM should use Legal operation action");
-
// If we have GFNI, we can use GF2P8AFFINEQB to reverse the bits.
if (Subtarget.hasGFNI()) {
SDValue Matrix = getGFNICtrlMask(ISD::BITREVERSE, DAG, DL, VT);
@@ -34315,9 +34287,9 @@ SDValue X86TargetLowering::LowerOperation(SDValue Op, SelectionDAG &DAG) const {
case ISD::SET_FPENV_MEM: return LowerSET_FPENV_MEM(Op, DAG);
case ISD::RESET_FPENV: return LowerRESET_FPENV(Op, DAG);
case ISD::CTLZ:
- case ISD::CTLZ_ZERO_POISON: return LowerCTLZ(Op, Subtarget, DAG);
+ case ISD::CTLZ_ZERO_UNDEF: return LowerCTLZ(Op, Subtarget, DAG);
case ISD::CTTZ:
- case ISD::CTTZ_ZERO_POISON: return LowerCTTZ(Op, Subtarget, DAG);
+ case ISD::CTTZ_ZERO_UNDEF: return LowerCTTZ(Op, Subtarget, DAG);
case ISD::MUL: return LowerMUL(Op, Subtarget, DAG);
case ISD::MULHS:
case ISD::MULHU: return LowerMULH(Op, Subtarget, DAG);
@@ -34348,14 +34320,6 @@ SDValue X86TargetLowering::LowerOperation(SDValue Op, SelectionDAG &DAG) const {
case ISD::SMIN:
case ISD::UMAX:
case ISD::UMIN: return LowerMINMAX(Op, Subtarget, DAG);
- case ISD::VECREDUCE_SMAX:
- case ISD::VECREDUCE_SMIN:
- case ISD::VECREDUCE_UMAX:
- case ISD::VECREDUCE_UMIN: return LowerMINMAX_REDUCE(Op, Subtarget, DAG);
- case ISD::VECREDUCE_AND:
- case ISD::VECREDUCE_OR:
- case ISD::VECREDUCE_XOR:
- case ISD::VECREDUCE_MUL: return LowerVECREDUCE(Op, Subtarget, DAG, true);
case ISD::FMINIMUM:
case ISD::FMAXIMUM:
case ISD::FMINIMUMNUM:
@@ -34795,8 +34759,8 @@ void X86TargetLowering::ReplaceNodeResults(SDNode *N,
}
case ISD::CTLZ:
case ISD::CTTZ:
- case ISD::CTLZ_ZERO_POISON:
- case ISD::CTTZ_ZERO_POISON: {
+ case ISD::CTLZ_ZERO_UNDEF:
+ case ISD::CTTZ_ZERO_UNDEF: {
// Fold i256/i512 CTLZ/CTTZ patterns to make use of AVX512
// vXi64 CTLZ/CTTZ and VECTOR_COMPRESS.
// Compute the CTLZ/CTTZ of each element, add the element's bit offset,
@@ -34825,7 +34789,7 @@ void X86TargetLowering::ReplaceNodeResults(SDNode *N,
// CTLZ - reverse the elements as we want the top non-zero element at the
// bottom for compression.
unsigned VecOpc = ISD::CTTZ;
- if (Opc == ISD::CTLZ || Opc == ISD::CTLZ_ZERO_POISON) {
+ if (Opc == ISD::CTLZ || Opc == ISD::CTLZ_ZERO_UNDEF) {
VecOpc = ISD::CTLZ;
Vec = DAG.getVectorShuffle(VecVT, dl, Vec, Vec, RevMask);
}
@@ -35158,21 +35122,6 @@ void X86TargetLowering::ReplaceNodeResults(SDNode *N,
}
return;
}
- case ISD::VECREDUCE_MUL: {
- assert(N->getValueType(0) == MVT::i64 && "Unexpected vector reduction");
- if (SDValue Res = LowerVECREDUCE(SDValue(N, 0), Subtarget, DAG, false))
- Results.push_back(Res);
- return;
- }
- case ISD::VECREDUCE_SMAX:
- case ISD::VECREDUCE_SMIN:
- case ISD::VECREDUCE_UMAX:
- case ISD::VECREDUCE_UMIN: {
- assert(N->getValueType(0) == MVT::i64 && "Unexpected vector reduction");
- if (SDValue Res = LowerMINMAX_REDUCE(SDValue(N, 0), Subtarget, DAG))
- Results.push_back(Res);
- return;
- }
case ISD::FP_TO_SINT_SAT:
case ISD::FP_TO_UINT_SAT: {
if (!Subtarget.hasAVX10_2())
@@ -35182,17 +35131,8 @@ void X86TargetLowering::ReplaceNodeResults(SDNode *N,
EVT VT = N->getValueType(0);
SDValue Op = N->getOperand(0);
EVT OpVT = Op.getValueType();
- EVT SatVT = cast<VTSDNode>(N->getOperand(1))->getVT();
SDValue Res;
- if (VT == MVT::v8i8 && OpVT == MVT::v8bf16 && SatVT == MVT::i8) {
- Res = DAG.getNode(IsSigned ? X86ISD::CVTTP2IBS : X86ISD::CVTTP2IUBS, dl,
- MVT::v8i16, Op);
- Res = DAG.getNode(ISD::TRUNCATE, dl, VT, Res);
- Results.push_back(Res);
- return;
- }
-
if (VT == MVT::v2i32 && OpVT == MVT::v2f64) {
if (IsSigned)
Res = DAG.getNode(X86ISD::FP_TO_SINT_SAT, dl, MVT::v4i32, Op);
@@ -35214,7 +35154,7 @@ void X86TargetLowering::ReplaceNodeResults(SDNode *N,
EVT SrcVT = Src.getValueType();
SDValue Res;
- if (isBF16orSoftF16(SrcVT, Subtarget)) {
+ if (isSoftF16(SrcVT, Subtarget)) {
EVT NVT = VT.changeElementType(*DAG.getContext(), MVT::f32);
if (IsStrict) {
Res =
@@ -37145,7 +37085,7 @@ X86TargetLowering::EmitLoweredProbedAlloca(MachineInstr &MI,
BuildMI(testMBB, MIMD, TII->get(X86::JCC_1))
.addMBB(tailMBB)
- .addImm(X86::COND_AE);
+ .addImm(X86::COND_GE);
testMBB->addSuccessor(blockMBB);
testMBB->addSuccessor(tailMBB);
@@ -38730,7 +38670,8 @@ X86TargetLowering::EmitInstrWithCustomInserter(MachineInstr &MI,
case X86::PTDPBF8PS:
case X86::PTDPBHF8PS:
case X86::PTDPHBF8PS:
- case X86::PTDPHF8PS: {
+ case X86::PTDPHF8PS:
+ case X86::PTMMULTF32PS: {
unsigned Opc;
switch (MI.getOpcode()) {
default: llvm_unreachable("illegal opcode!");
@@ -38747,6 +38688,7 @@ X86TargetLowering::EmitInstrWithCustomInserter(MachineInstr &MI,
case X86::PTDPBHF8PS: Opc = X86::TDPBHF8PS; break;
case X86::PTDPHBF8PS: Opc = X86::TDPHBF8PS; break;
case X86::PTDPHF8PS: Opc = X86::TDPHF8PS; break;
+ case X86::PTMMULTF32PS: Opc = X86::TMMULTF32PS; break;
// clang-format on
}
@@ -39399,6 +39341,25 @@ void X86TargetLowering::computeKnownBitsForTargetNode(const SDValue Op,
Known.One.clearAllBits();
break;
}
+ case X86ISD::PDEP: {
+ KnownBits Known2;
+ Known = DAG.computeKnownBits(Op.getOperand(1), DemandedElts, Depth + 1);
+ Known2 = DAG.computeKnownBits(Op.getOperand(0), DemandedElts, Depth + 1);
+ // Zeros are retained from the mask operand. But not ones.
+ Known.One.clearAllBits();
+ // The result will have at least as many trailing zeros as the non-mask
+ // operand since bits can only map to the same or higher bit position.
+ Known.Zero.setLowBits(Known2.countMinTrailingZeros());
+ break;
+ }
+ case X86ISD::PEXT: {
+ Known = DAG.computeKnownBits(Op.getOperand(1), DemandedElts, Depth + 1);
+ // The result has as many leading zeros as the number of zeroes in the mask.
+ unsigned Count = Known.Zero.popcount();
+ Known.Zero = APInt::getHighBitsSet(BitWidth, Count);
+ Known.One.clearAllBits();
+ break;
+ }
case X86ISD::VTRUNC:
case X86ISD::VTRUNCS:
case X86ISD::VTRUNCUS:
@@ -39450,14 +39411,6 @@ void X86TargetLowering::computeKnownBitsForTargetNode(const SDValue Op,
Known.setAllZero();
break;
}
- case X86ISD::VZEXT_LOAD: {
- // Upper elements are known zero.
- auto *LN = cast<MemIntrinsicSDNode>(Op);
- if (!DemandedElts[0] &&
- LN->getMemoryVT().getSizeInBits() == VT.getScalarSizeInBits())
- Known.setAllZero();
- break;
- }
case X86ISD::VBROADCAST_LOAD: {
APInt UndefElts;
SmallVector<APInt, 16> EltBits;
@@ -39805,16 +39758,19 @@ static bool matchUnaryShuffle(MVT MaskVT, ArrayRef<int> Mask,
unsigned NumMaskElts = Mask.size();
unsigned MaskEltSize = MaskVT.getScalarSizeInBits();
- // Match against a foldable vXi32/vXi16 VZEXT_MOVL zero-extending instruction.
- if ((MaskEltSize == 32 || (MaskEltSize == 16 && Subtarget.hasFP16())) &&
- (V1.getOpcode() == ISD::SCALAR_TO_VECTOR || isa<MemSDNode>(V1)) &&
- Mask[0] == 0 && isUndefOrZeroInRange(Mask, 1, NumMaskElts - 1)) {
- Shuffle = X86ISD::VZEXT_MOVL;
- if (MaskEltSize == 16)
- SrcVT = DstVT = MaskVT.changeVectorElementType(MVT::f16);
- else
- SrcVT = DstVT = !Subtarget.hasSSE2() ? MVT::v4f32 : MaskVT;
- return true;
+ // Match against a VZEXT_MOVL vXi32 and vXi16 zero-extending instruction.
+ if (Mask[0] == 0 &&
+ (MaskEltSize == 32 || (MaskEltSize == 16 && Subtarget.hasFP16()))) {
+ if ((isUndefOrZero(Mask[1]) && isUndefInRange(Mask, 2, NumMaskElts - 2)) ||
+ (V1.getOpcode() == ISD::SCALAR_TO_VECTOR &&
+ isUndefOrZeroInRange(Mask, 1, NumMaskElts - 1))) {
+ Shuffle = X86ISD::VZEXT_MOVL;
+ if (MaskEltSize == 16)
+ SrcVT = DstVT = MaskVT.changeVectorElementType(MVT::f16);
+ else
+ SrcVT = DstVT = !Subtarget.hasSSE2() ? MVT::v4f32 : MaskVT;
+ return true;
+ }
}
// Match against a ANY/SIGN/ZERO_EXTEND_VECTOR_INREG instruction.
@@ -39869,7 +39825,8 @@ static bool matchUnaryShuffle(MVT MaskVT, ArrayRef<int> Mask,
// Match against a VZEXT_MOVL instruction, SSE1 only supports 32-bits (MOVSS).
if (((MaskEltSize == 32) || (MaskEltSize == 64 && Subtarget.hasSSE2()) ||
(MaskEltSize == 16 && Subtarget.hasFP16())) &&
- Mask[0] == 0 && isUndefOrZeroInRange(Mask, 1, NumMaskElts - 1)) {
+ isUndefOrEqual(Mask[0], 0) &&
+ isUndefOrZeroInRange(Mask, 1, NumMaskElts - 1)) {
Shuffle = X86ISD::VZEXT_MOVL;
if (MaskEltSize == 16)
SrcVT = DstVT = MaskVT.changeVectorElementType(MVT::f16);
@@ -40363,16 +40320,6 @@ static bool matchBinaryPermuteShuffle(
}
}
- // Attempt to match against VSHLD funnel shift.
- if (AllowIntDomain && Subtarget.hasVBMI2()) {
- int ShiftAmt = matchShuffleAsVSHLD(ShuffleVT, V1, V2, EltSizeInBits, Mask);
- if (0 < ShiftAmt) {
- Shuffle = X86ISD::VSHLD;
- PermuteImm = (unsigned)ShiftAmt;
- return true;
- }
- }
-
// Attempt to combine to X86ISD::BLENDI.
if ((NumMaskElts <= 8 && ((Subtarget.hasSSE41() && MaskVT.is128BitVector()) ||
(Subtarget.hasAVX() && MaskVT.is256BitVector()))) ||
@@ -42114,20 +42061,7 @@ static SDValue combineX86ShufflesRecursively(
Op, OpScaledDemandedElts, DAG))
Op = NewOp;
}
-
- // Reresolve - we might have repeated subvector sources.
- resolveTargetShuffleInputsAndMask(Ops, Mask);
-
- // Handle the all undef/zero/ones cases.
- if (all_of(Mask, [](int Idx) { return Idx == SM_SentinelUndef; }))
- return DAG.getUNDEF(RootVT);
- if (all_of(Mask, [](int Idx) { return Idx < 0; }))
- return getZeroVector(RootVT, Subtarget, DAG, DL);
- if (Ops.size() == 1 && ISD::isBuildVectorAllOnes(Ops[0].getNode()) &&
- !llvm::is_contained(Mask, SM_SentinelZero))
- return getOnesVector(RootVT, DAG, DL);
-
- assert(!Ops.empty() && "Shuffle with no inputs detected");
+ // FIXME: should we rerun resolveTargetShuffleInputsAndMask() now?
// Widen any subvector shuffle inputs we've collected.
// TODO: Remove this to avoid generating temporary nodes, we should only
@@ -42139,8 +42073,21 @@ static SDValue combineX86ShufflesRecursively(
if (Op.getValueSizeInBits() < RootSizeInBits)
Op = widenSubVector(Op, false, Subtarget, DAG, SDLoc(Op),
RootSizeInBits);
+ // Reresolve - we might have repeated subvector sources.
+ resolveTargetShuffleInputsAndMask(Ops, Mask);
}
+ // Handle the all undef/zero/ones cases.
+ if (all_of(Mask, [](int Idx) { return Idx == SM_SentinelUndef; }))
+ return DAG.getUNDEF(RootVT);
+ if (all_of(Mask, [](int Idx) { return Idx < 0; }))
+ return getZeroVector(RootVT, Subtarget, DAG, DL);
+ if (Ops.size() == 1 && ISD::isBuildVectorAllOnes(Ops[0].getNode()) &&
+ !llvm::is_contained(Mask, SM_SentinelZero))
+ return getOnesVector(RootVT, DAG, DL);
+
+ assert(!Ops.empty() && "Shuffle with no inputs detected");
+
// We can only combine unary and binary shuffle mask cases.
if (Ops.size() <= 2) {
// Minor canonicalization of the accumulated shuffle mask to make it easier
@@ -42904,22 +42851,6 @@ static SDValue combineTargetShuffle(SDValue N, const SDLoc &DL,
TLI.isTypeLegal(Src.getOperand(0).getValueType()))
return DAG.getNode(X86ISD::VBROADCAST, DL, VT, Src.getOperand(0));
- // broadcast(truncate(extract_vector_elt(x, 0))) -> bitcast(broadcast(x)).
- if (Src.getOpcode() == ISD::TRUNCATE &&
- Src.getOperand(0).getOpcode() == ISD::EXTRACT_VECTOR_ELT &&
- isNullConstant(Src.getOperand(0).getOperand(1))) {
- SDValue NewSrc = Src.getOperand(0).getOperand(0);
- if (Src.getOperand(0).getValueType() ==
- NewSrc.getValueType().getScalarType() &&
- TLI.isTypeLegal(NewSrc.getValueType())) {
- MVT VecVT = MVT::getVectorVT(Src.getSimpleValueType(),
- NewSrc.getValueSizeInBits() /
- Src.getValueSizeInBits());
- return DAG.getNode(X86ISD::VBROADCAST, DL, VT,
- DAG.getBitcast(VecVT, NewSrc));
- }
- }
-
// Share broadcast with the longest vector and extract low subvector (free).
// Ensure the same SDValue from the SDNode use is being used.
for (SDNode *User : Src->users())
@@ -43292,25 +43223,11 @@ static SDValue combineTargetShuffle(SDValue N, const SDLoc &DL,
// If we're permuting the upper 256-bits subvectors of a concatenation, then
// see if we can peek through and access the subvector directly.
if (VT.is512BitVector()) {
+ // 512-bit mask uses 4 x i2 indices - if the msb is always set then only
+ // the upper subvector is used.
SDValue LHS = peekThroughBitcasts(N->getOperand(0));
SDValue RHS = peekThroughBitcasts(N->getOperand(1));
uint64_t Mask = N->getConstantOperandVal(2);
- // Attempt to concat directly instead of shuffling source together.
- // TODO: combineX86ShufflesRecursively should do this generically.
- if (Mask == 0x44 && LHS.getValueType() == RHS.getValueType() &&
- LHS.getValueType().isSimple()) {
- SmallVector<SDValue> LHSOps, RHSOps;
- if (collectConcatOps(LHS.getNode(), LHSOps, DAG) &&
- collectConcatOps(RHS.getNode(), RHSOps, DAG) &&
- LHSOps.size() == 2 && RHSOps.size() == 2) {
- if (SDValue Concat = combineConcatVectorOps(
- DL, LHS.getSimpleValueType(), {LHSOps[0], RHSOps[0]}, DAG,
- Subtarget))
- return DAG.getBitcast(VT, Concat);
- }
- }
- // 512-bit mask uses 4 x i2 indices - if the msb is always set then only
- // the upper subvector is used.
SmallVector<SDValue> LHSOps, RHSOps;
SDValue NewLHS, NewRHS;
if ((Mask & 0x0A) == 0x0A &&
@@ -43669,26 +43586,20 @@ static SDValue combineTargetShuffle(SDValue N, const SDLoc &DL,
return lowerShuffleWithPERMV(DL, VT, Mask, N.getOperand(0),
DAG.getUNDEF(VT), Subtarget, DAG);
}
- // If sources are widened, then concat and use VPERMV with adjusted mask.
+ // If sources are half width, then concat and use VPERMV with adjusted
+ // mask.
SDValue Ops[2];
+ MVT HalfVT = VT.getHalfNumVectorElementsVT();
if (sd_match(V1,
m_InsertSubvector(m_Undef(), m_Value(Ops[0]), m_Zero())) &&
sd_match(V2,
m_InsertSubvector(m_Undef(), m_Value(Ops[1]), m_Zero())) &&
- (Ops[0].getValueSizeInBits() % VT.getScalarSizeInBits()) == 0 &&
- Ops[0].getValueType() == Ops[1].getValueType() &&
- Ops[0].getValueType().isSimple()) {
- MVT SubVT = Ops[0].getSimpleValueType();
- MVT ConcatVT = SubVT.getDoubleNumVectorElementsVT();
- unsigned NumSubElts = SubVT.getSizeInBits() / VT.getScalarSizeInBits();
+ Ops[0].getValueType() == HalfVT && Ops[1].getValueType() == HalfVT) {
if (SDValue ConcatSrc =
- combineConcatVectorOps(DL, ConcatVT, Ops, DAG, Subtarget)) {
- ConcatSrc = widenSubVector(ConcatSrc, false, Subtarget, DAG, DL,
- VT.getSizeInBits());
+ combineConcatVectorOps(DL, VT, Ops, DAG, Subtarget)) {
for (int &M : Mask)
- M = (M < (int)NumElts ? M : (M - (NumElts - NumSubElts)));
- return lowerShuffleWithPERMV(DL, VT, Mask,
- DAG.getBitcast(VT, ConcatSrc),
+ M = (M < (int)NumElts ? M : (M - (NumElts / 2)));
+ return lowerShuffleWithPERMV(DL, VT, Mask, ConcatSrc,
DAG.getUNDEF(VT), Subtarget, DAG);
}
}
@@ -45659,6 +45570,34 @@ bool X86TargetLowering::SimplifyDemandedBitsForTargetNode(
break;
}
+ case X86ISD::PDEP: {
+ SDValue Op0 = Op.getOperand(0);
+ SDValue Op1 = Op.getOperand(1);
+
+ unsigned DemandedBitsLZ = OriginalDemandedBits.countl_zero();
+ APInt LoMask = APInt::getLowBitsSet(BitWidth, BitWidth - DemandedBitsLZ);
+
+ // If the demanded bits has leading zeroes, we don't demand those from the
+ // mask.
+ if (SimplifyDemandedBits(Op1, LoMask, Known, TLO, Depth + 1))
+ return true;
+
+ // The number of possible 1s in the mask determines the number of LSBs of
+ // operand 0 used. Undemanded bits from the mask don't matter so filter
+ // them before counting.
+ KnownBits Known2;
+ uint64_t Count = (~Known.Zero & LoMask).popcount();
+ APInt DemandedMask(APInt::getLowBitsSet(BitWidth, Count));
+ if (SimplifyDemandedBits(Op0, DemandedMask, Known2, TLO, Depth + 1))
+ return true;
+
+ // Zeroes are retained from the mask, but not ones.
+ Known.One.clearAllBits();
+ // The result will have at least as many trailing zeros as the non-mask
+ // operand since bits can only map to the same or higher bit position.
+ Known.Zero.setLowBits(Known2.countMinTrailingZeros());
+ return false;
+ }
case X86ISD::VPMADD52L:
case X86ISD::VPMADD52H: {
KnownBits KnownOp0, KnownOp1, KnownOp2;
@@ -45825,17 +45764,13 @@ SDValue X86TargetLowering::SimplifyMultipleUseDemandedBitsForTargetNode(
bool X86TargetLowering::isGuaranteedNotToBeUndefOrPoisonForTargetNode(
SDValue Op, const APInt &DemandedElts, const SelectionDAG &DAG,
- UndefPoisonKind Kind, unsigned Depth) const {
+ bool PoisonOnly, unsigned Depth) const {
unsigned NumElts = DemandedElts.getBitWidth();
switch (Op.getOpcode()) {
case X86ISD::GlobalBaseReg:
case X86ISD::Wrapper:
case X86ISD::WrapperRIP:
- // SETCC/SETCC_CARRY always produces a well-defined result based on
- // EFLAGS/carry flag.
- case X86ISD::SETCC:
- case X86ISD::SETCC_CARRY:
return true;
case X86ISD::PACKSS:
case X86ISD::PACKUS: {
@@ -45844,10 +45779,10 @@ bool X86TargetLowering::isGuaranteedNotToBeUndefOrPoisonForTargetNode(
DemandedRHS);
return (!DemandedLHS ||
DAG.isGuaranteedNotToBeUndefOrPoison(Op.getOperand(0), DemandedLHS,
- Kind, Depth + 1)) &&
+ PoisonOnly, Depth + 1)) &&
(!DemandedRHS ||
DAG.isGuaranteedNotToBeUndefOrPoison(Op.getOperand(1), DemandedRHS,
- Kind, Depth + 1));
+ PoisonOnly, Depth + 1));
}
case X86ISD::INSERTPS:
case X86ISD::BLENDI:
@@ -45881,7 +45816,7 @@ bool X86TargetLowering::isGuaranteedNotToBeUndefOrPoisonForTargetNode(
for (auto Op : enumerate(Ops))
if (!DemandedSrcElts[Op.index()].isZero() &&
!DAG.isGuaranteedNotToBeUndefOrPoison(
- Op.value(), DemandedSrcElts[Op.index()], Kind, Depth + 1))
+ Op.value(), DemandedSrcElts[Op.index()], PoisonOnly, Depth + 1))
return false;
return true;
}
@@ -45892,19 +45827,19 @@ bool X86TargetLowering::isGuaranteedNotToBeUndefOrPoisonForTargetNode(
MVT SrcVT = Src.getSimpleValueType();
if (SrcVT.isVector()) {
APInt DemandedSrc = APInt::getOneBitSet(SrcVT.getVectorNumElements(), 0);
- return DAG.isGuaranteedNotToBeUndefOrPoison(Src, DemandedSrc, Kind,
+ return DAG.isGuaranteedNotToBeUndefOrPoison(Src, DemandedSrc, PoisonOnly,
Depth + 1);
}
- return DAG.isGuaranteedNotToBeUndefOrPoison(Src, Kind, Depth + 1);
+ return DAG.isGuaranteedNotToBeUndefOrPoison(Src, PoisonOnly, Depth + 1);
}
}
return TargetLowering::isGuaranteedNotToBeUndefOrPoisonForTargetNode(
- Op, DemandedElts, DAG, Kind, Depth);
+ Op, DemandedElts, DAG, PoisonOnly, Depth);
}
bool X86TargetLowering::canCreateUndefOrPoisonForTargetNode(
SDValue Op, const APInt &DemandedElts, const SelectionDAG &DAG,
- UndefPoisonKind Kind, bool ConsiderFlags, unsigned Depth) const {
+ bool PoisonOnly, bool ConsiderFlags, unsigned Depth) const {
switch (Op.getOpcode()) {
// SSE bit logic.
@@ -46000,7 +45935,7 @@ bool X86TargetLowering::canCreateUndefOrPoisonForTargetNode(
}
}
return TargetLowering::canCreateUndefOrPoisonForTargetNode(
- Op, DemandedElts, DAG, Kind, ConsiderFlags, Depth);
+ Op, DemandedElts, DAG, PoisonOnly, ConsiderFlags, Depth);
}
bool X86TargetLowering::isSplatValueForTargetNode(SDValue Op,
@@ -46332,8 +46267,10 @@ static SDValue combineCastedMaskArithmetic(SDNode *N, SelectionDAG &DAG,
EVT SrcVT = Op.getValueType();
// Make sure we have a bitcast between mask registers and a scalar type.
- if (!(SrcVT.isVectorOf(MVT::i1) && DstVT.isScalarInteger()) &&
- !(DstVT.isVectorOf(MVT::i1) && SrcVT.isScalarInteger()))
+ if (!(SrcVT.isVector() && SrcVT.getVectorElementType() == MVT::i1 &&
+ DstVT.isScalarInteger()) &&
+ !(DstVT.isVector() && DstVT.getVectorElementType() == MVT::i1 &&
+ SrcVT.isScalarInteger()))
return SDValue();
SDValue LHS, RHS;
@@ -46592,8 +46529,8 @@ static SDValue combineBitcast(SDNode *N, SelectionDAG &DAG,
} else if (DCI.isAfterLegalizeDAG()) {
// If we're bitcasting from iX to vXi1, see if the integer originally
// began as a vXi1 and whether we can remove the bitcast entirely.
- if (VT.isVectorOf(MVT::i1) && SrcVT.isScalarInteger() &&
- TLI.isTypeLegal(VT)) {
+ if (VT.isVector() && VT.getScalarType() == MVT::i1 &&
+ SrcVT.isScalarInteger() && TLI.isTypeLegal(VT)) {
if (SDValue V =
combineBitcastToBoolVector(VT, N0, SDLoc(N), DAG, Subtarget))
return V;
@@ -46722,7 +46659,7 @@ static SDValue combineBitcast(SDNode *N, SelectionDAG &DAG,
// Try to remove a bitcast of constant vXi1 vector. We have to legalize
// most of these to scalar anyway.
if (Subtarget.hasAVX512() && VT.isScalarInteger() &&
- SrcVT.isVectorOf(MVT::i1) &&
+ SrcVT.isVector() && SrcVT.getVectorElementType() == MVT::i1 &&
ISD::isBuildVectorOfConstantSDNodes(N0.getNode())) {
return combinevXi1ConstantToInteger(N0, DAG);
}
@@ -46741,7 +46678,8 @@ static SDValue combineBitcast(SDNode *N, SelectionDAG &DAG,
// Turn it into a sign bit compare that produces a k-register. This avoids
// a trip through a GPR.
if (Subtarget.hasAVX512() && SrcVT.isScalarInteger() &&
- VT.isVectorOf(MVT::i1) && isPowerOf2_32(VT.getVectorNumElements())) {
+ VT.isVector() && VT.getVectorElementType() == MVT::i1 &&
+ isPowerOf2_32(VT.getVectorNumElements())) {
unsigned NumElts = VT.getVectorNumElements();
SDValue Src = N0;
@@ -46940,6 +46878,81 @@ static SDValue createPSADBW(SelectionDAG &DAG, SDValue N0, SDValue N1,
PSADBWBuilder);
}
+// Attempt to replace an min/max v8i16/v16i8 horizontal reduction with
+// PHMINPOSUW.
+static SDValue combineMinMaxReduction(SDNode *Extract, SelectionDAG &DAG,
+ const X86Subtarget &Subtarget) {
+ // Bail without SSE41.
+ if (!Subtarget.hasSSE41())
+ return SDValue();
+
+ EVT ExtractVT = Extract->getValueType(0);
+ if (ExtractVT != MVT::i16 && ExtractVT != MVT::i8)
+ return SDValue();
+
+ // Check for SMAX/SMIN/UMAX/UMIN horizontal reduction patterns.
+ ISD::NodeType BinOp;
+ SDValue Src = DAG.matchBinOpReduction(
+ Extract, BinOp, {ISD::SMAX, ISD::SMIN, ISD::UMAX, ISD::UMIN}, true);
+ if (!Src)
+ return SDValue();
+
+ EVT SrcVT = Src.getValueType();
+ EVT SrcSVT = SrcVT.getScalarType();
+ if (SrcSVT != ExtractVT || (SrcVT.getSizeInBits() % 128) != 0)
+ return SDValue();
+
+ SDLoc DL(Extract);
+ SDValue MinPos = Src;
+
+ // First, reduce the source down to 128-bit, applying BinOp to lo/hi.
+ while (SrcVT.getSizeInBits() > 128) {
+ SDValue Lo, Hi;
+ std::tie(Lo, Hi) = splitVector(MinPos, DAG, DL);
+ SrcVT = Lo.getValueType();
+ MinPos = DAG.getNode(BinOp, DL, SrcVT, Lo, Hi);
+ }
+ assert(((SrcVT == MVT::v8i16 && ExtractVT == MVT::i16) ||
+ (SrcVT == MVT::v16i8 && ExtractVT == MVT::i8)) &&
+ "Unexpected value type");
+
+ // PHMINPOSUW applies to UMIN(v8i16), for SMIN/SMAX/UMAX we must apply a mask
+ // to flip the value accordingly.
+ SDValue Mask;
+ unsigned MaskEltsBits = ExtractVT.getSizeInBits();
+ if (BinOp == ISD::SMAX)
+ Mask = DAG.getConstant(APInt::getSignedMaxValue(MaskEltsBits), DL, SrcVT);
+ else if (BinOp == ISD::SMIN)
+ Mask = DAG.getConstant(APInt::getSignedMinValue(MaskEltsBits), DL, SrcVT);
+ else if (BinOp == ISD::UMAX)
+ Mask = DAG.getAllOnesConstant(DL, SrcVT);
+
+ if (Mask)
+ MinPos = DAG.getNode(ISD::XOR, DL, SrcVT, Mask, MinPos);
+
+ // For v16i8 cases we need to perform UMIN on pairs of byte elements,
+ // shuffling each upper element down and insert zeros. This means that the
+ // v16i8 UMIN will leave the upper element as zero, performing zero-extension
+ // ready for the PHMINPOS.
+ if (ExtractVT == MVT::i8) {
+ SDValue Upper = DAG.getVectorShuffle(
+ SrcVT, DL, MinPos, DAG.getConstant(0, DL, MVT::v16i8),
+ {1, 16, 3, 16, 5, 16, 7, 16, 9, 16, 11, 16, 13, 16, 15, 16});
+ MinPos = DAG.getNode(ISD::UMIN, DL, SrcVT, MinPos, Upper);
+ }
+
+ // Perform the PHMINPOS on a v8i16 vector,
+ MinPos = DAG.getBitcast(MVT::v8i16, MinPos);
+ MinPos = DAG.getNode(X86ISD::PHMINPOS, DL, MVT::v8i16, MinPos);
+ MinPos = DAG.getBitcast(SrcVT, MinPos);
+
+ if (Mask)
+ MinPos = DAG.getNode(ISD::XOR, DL, SrcVT, Mask, MinPos);
+
+ return DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, ExtractVT, MinPos,
+ DAG.getVectorIdxConstant(0, DL));
+}
+
// Attempt to replace an all_of/any_of/parity style horizontal reduction with a MOVMSK.
static SDValue combinePredicateReduction(SDNode *Extract, SelectionDAG &DAG,
const X86Subtarget &Subtarget) {
@@ -47597,8 +47610,8 @@ static SDValue combineArithReduction(SDNode *ExtElt, SelectionDAG &DAG,
return SDValue();
ISD::NodeType Opc;
- SDValue Rdx =
- DAG.matchBinOpReduction(ExtElt, Opc, {ISD::ADD, ISD::FADD}, true);
+ SDValue Rdx = DAG.matchBinOpReduction(ExtElt, Opc,
+ {ISD::ADD, ISD::MUL, ISD::FADD}, true);
if (!Rdx)
return SDValue();
@@ -47615,10 +47628,10 @@ static SDValue combineArithReduction(SDNode *ExtElt, SelectionDAG &DAG,
unsigned NumElts = VecVT.getVectorNumElements();
unsigned EltSizeInBits = VecVT.getScalarSizeInBits();
- // ZeroExtend v4i8/v8i8 vector to v16i8, with undef upper 64-bits.
- auto WidenToV16I8 = [&](SDValue V) {
+ // Extend v4i8/v8i8 vector to v16i8, with undef upper 64-bits.
+ auto WidenToV16I8 = [&](SDValue V, bool ZeroExtend) {
if (V.getValueType() == MVT::v4i8) {
- if (Subtarget.hasSSE41()) {
+ if (ZeroExtend && Subtarget.hasSSE41()) {
V = DAG.getNode(ISD::INSERT_VECTOR_ELT, DL, MVT::v4i32,
DAG.getConstant(0, DL, MVT::v4i32),
DAG.getBitcast(MVT::i32, V),
@@ -47626,15 +47639,50 @@ static SDValue combineArithReduction(SDNode *ExtElt, SelectionDAG &DAG,
return DAG.getBitcast(MVT::v16i8, V);
}
V = DAG.getNode(ISD::CONCAT_VECTORS, DL, MVT::v8i8, V,
- DAG.getConstant(0, DL, MVT::v4i8));
+ ZeroExtend ? DAG.getConstant(0, DL, MVT::v4i8)
+ : DAG.getUNDEF(MVT::v4i8));
}
return DAG.getNode(ISD::CONCAT_VECTORS, DL, MVT::v16i8, V,
DAG.getUNDEF(MVT::v8i8));
};
+ // vXi8 mul reduction - promote to vXi16 mul reduction.
+ if (Opc == ISD::MUL) {
+ if (VT != MVT::i8 || NumElts < 4 || !isPowerOf2_32(NumElts))
+ return SDValue();
+ if (VecVT.getSizeInBits() >= 128) {
+ EVT WideVT = EVT::getVectorVT(*DAG.getContext(), MVT::i16, NumElts / 2);
+ SDValue Lo = getUnpackl(DAG, DL, VecVT, Rdx, DAG.getUNDEF(VecVT));
+ SDValue Hi = getUnpackh(DAG, DL, VecVT, Rdx, DAG.getUNDEF(VecVT));
+ Lo = DAG.getBitcast(WideVT, Lo);
+ Hi = DAG.getBitcast(WideVT, Hi);
+ Rdx = DAG.getNode(Opc, DL, WideVT, Lo, Hi);
+ while (Rdx.getValueSizeInBits() > 128) {
+ std::tie(Lo, Hi) = splitVector(Rdx, DAG, DL);
+ Rdx = DAG.getNode(Opc, DL, Lo.getValueType(), Lo, Hi);
+ }
+ } else {
+ Rdx = WidenToV16I8(Rdx, false);
+ Rdx = getUnpackl(DAG, DL, MVT::v16i8, Rdx, DAG.getUNDEF(MVT::v16i8));
+ Rdx = DAG.getBitcast(MVT::v8i16, Rdx);
+ }
+ if (NumElts >= 8)
+ Rdx = DAG.getNode(Opc, DL, MVT::v8i16, Rdx,
+ DAG.getVectorShuffle(MVT::v8i16, DL, Rdx, Rdx,
+ {4, 5, 6, 7, -1, -1, -1, -1}));
+ Rdx = DAG.getNode(Opc, DL, MVT::v8i16, Rdx,
+ DAG.getVectorShuffle(MVT::v8i16, DL, Rdx, Rdx,
+ {2, 3, -1, -1, -1, -1, -1, -1}));
+ Rdx = DAG.getNode(Opc, DL, MVT::v8i16, Rdx,
+ DAG.getVectorShuffle(MVT::v8i16, DL, Rdx, Rdx,
+ {1, -1, -1, -1, -1, -1, -1, -1}));
+ Rdx = DAG.getBitcast(MVT::v16i8, Rdx);
+ return DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, VT, Rdx, Index);
+ }
+
// vXi8 add reduction - sub 128-bit vector.
if (VecVT == MVT::v4i8 || VecVT == MVT::v8i8) {
- Rdx = WidenToV16I8(Rdx);
+ Rdx = WidenToV16I8(Rdx, true);
Rdx = DAG.getNode(X86ISD::PSADBW, DL, MVT::v2i64, Rdx,
DAG.getConstant(0, DL, MVT::v16i8));
Rdx = DAG.getBitcast(MVT::v16i8, Rdx);
@@ -47680,7 +47728,7 @@ static SDValue combineArithReduction(SDNode *ExtElt, SelectionDAG &DAG,
EVT ByteVT = VecVT.changeVectorElementType(*DAG.getContext(), MVT::i8);
Rdx = DAG.getNode(ISD::TRUNCATE, DL, ByteVT, Rdx);
if (ByteVT.getSizeInBits() < 128)
- Rdx = WidenToV16I8(Rdx);
+ Rdx = WidenToV16I8(Rdx, true);
}
// Build the PSADBW, split as 128/256/512 bits for SSE/AVX2/AVX512BW.
@@ -47818,17 +47866,6 @@ static SDValue combineExtractVectorElt(SDNode *N, SelectionDAG &DAG,
return SDValue();
}
- // Attempt to avoid multi-use src if we don't need anything from it.
- // TODO: Generlize this and move to DAGCombine.
- if (CIdx && ISD::isBitwiseLogicOp(InputVector.getOpcode())) {
- unsigned Idx = CIdx->getZExtValue();
- APInt DemandedElts = APInt::getOneBitSet(NumSrcElts, Idx);
- if (SDValue NewVector = TLI.SimplifyMultipleUseDemandedVectorElts(
- InputVector, DemandedElts, DAG))
- if (NewVector.getOpcode() == ISD::BUILD_VECTOR)
- return DAG.getNode(N->getOpcode(), dl, VT, NewVector, EltIdx);
- }
-
// Detect mmx extraction of all bits as a i64. It works better as a bitcast.
if (VT == MVT::i64 && SrcVT == MVT::v1i64 &&
InputVector.getOpcode() == ISD::BITCAST &&
@@ -47857,7 +47894,11 @@ static SDValue combineExtractVectorElt(SDNode *N, SelectionDAG &DAG,
if (SDValue Cmp = combinePredicateReduction(N, DAG, Subtarget))
return Cmp;
- // Attempt to optimize ADD/FADD reductions with HADD, promotion etc..
+ // Attempt to replace min/max v8i16/v16i8 reductions with PHMINPOSUW.
+ if (SDValue MinMax = combineMinMaxReduction(N, DAG, Subtarget))
+ return MinMax;
+
+ // Attempt to optimize ADD/FADD/MUL reductions with HADD, promotion etc..
if (SDValue V = combineArithReduction(N, DAG, Subtarget))
return V;
@@ -47946,38 +47987,6 @@ static SDValue combineExtractVectorElt(SDNode *N, SelectionDAG &DAG,
return SDValue();
}
-static SDValue combineVECREDUCE_MUL(SDNode *N, SelectionDAG &DAG,
- const X86Subtarget &Subtarget) {
- SDValue Src = N->getOperand(0);
- EVT SrcVT = Src.getValueType();
- unsigned NumElts = SrcVT.getVectorNumElements();
- SDLoc DL(N);
-
- // vXi8 mul reduction - promote to vXi16 mul reduction.
- if (!isPowerOf2_32(NumElts) || SrcVT.getScalarType() != MVT::i8)
- return SDValue();
-
- // Early out for v2i8 - no promotion necessary.
- if (NumElts == 2) {
- SDValue Hi = DAG.getVectorShuffle(SrcVT, DL, Src, Src, {1, -1});
- SDValue Rdx = DAG.getNode(ISD::MUL, DL, SrcVT, Src, Hi);
- return DAG.getExtractVectorElt(DL, N->getValueType(0), Rdx, 0);
- }
-
- if (SrcVT.getSizeInBits() >= 128) {
- EVT WideVT = EVT::getVectorVT(*DAG.getContext(), MVT::i16, NumElts / 2);
- SDValue Lo = getUnpackl(DAG, DL, SrcVT, Src, DAG.getUNDEF(SrcVT));
- SDValue Hi = getUnpackh(DAG, DL, SrcVT, Src, DAG.getUNDEF(SrcVT));
- Src = DAG.getNode(ISD::MUL, DL, WideVT, DAG.getBitcast(WideVT, Lo),
- DAG.getBitcast(WideVT, Hi));
- } else {
- EVT ExtVT = SrcVT.changeVectorElementType(*DAG.getContext(), MVT::i16);
- Src = DAG.getNode(ISD::ANY_EXTEND, DL, ExtVT, Src);
- }
- Src = DAG.getNode(ISD::VECREDUCE_MUL, DL, MVT::i16, Src);
- return DAG.getZExtOrTrunc(Src, DL, N->getValueType(0));
-}
-
// Convert (vXiY *ext(vXi1 bitcast(iX))) to extend_in_reg(broadcast(iX)).
// This is more or less the reverse of combineBitcastvxi1.
static SDValue combineToExtendBoolVectorInReg(
@@ -48405,8 +48414,8 @@ static SDValue combineSelectToMinMax(SelectionDAG &DAG,
case ISD::SETOLE:
// Converting this to a min would handle comparisons between positive
// and negative zero incorrectly.
- if (!N->getFlags().hasNoSignedZeros() &&
- !DAG.isKnownNeverLogicalZero(LHS) && !DAG.isKnownNeverLogicalZero(RHS))
+ if (!N->getFlags().hasNoSignedZeros() && !DAG.isKnownNeverZeroFloat(LHS) &&
+ !DAG.isKnownNeverZeroFloat(RHS))
break;
Opcode = X86ISD::FMIN;
break;
@@ -48422,8 +48431,8 @@ static SDValue combineSelectToMinMax(SelectionDAG &DAG,
case ISD::SETOGE:
// Converting this to a max would handle comparisons between positive
// and negative zero incorrectly.
- if (!N->getFlags().hasNoSignedZeros() &&
- !DAG.isKnownNeverLogicalZero(LHS) && !DAG.isKnownNeverLogicalZero(RHS))
+ if (!N->getFlags().hasNoSignedZeros() && !DAG.isKnownNeverZeroFloat(LHS) &&
+ !DAG.isKnownNeverZeroFloat(RHS))
break;
Opcode = X86ISD::FMAX;
break;
@@ -48582,15 +48591,6 @@ static SDValue combineSelect(SDNode *N, SelectionDAG &DAG,
CondVT.getVectorElementType() == MVT::i1 &&
(VT.getVectorElementType() == MVT::i8 ||
VT.getVectorElementType() == MVT::i16)) {
- // Handle AVX512F masked trunc patterns, which do have vXi8/vXi16 selects.
- if (LHS.getOpcode() == ISD::TRUNCATE) {
- SDValue TruncSrc = LHS.getOperand(0);
- EVT TruncSrcVT = TruncSrc.getValueType();
- if ((VT == MVT::v8i16 && TruncSrcVT == MVT::v8i64) ||
- (VT == MVT::v16i8 && TruncSrcVT == MVT::v16i32) ||
- (VT == MVT::v16i16 && TruncSrcVT == MVT::v16i32))
- return DAG.getNode(X86ISD::VMTRUNC, DL, VT, TruncSrc, RHS, Cond);
- }
Cond = DAG.getNode(ISD::SIGN_EXTEND, DL, VT, Cond);
return DAG.getNode(N->getOpcode(), DL, VT, Cond, LHS, RHS);
}
@@ -48778,9 +48778,7 @@ static SDValue combineSelect(SDNode *N, SelectionDAG &DAG,
return V;
// select(~Cond, X, Y) -> select(Cond, Y, X)
- // This is only valid for vector selects which use an all-bits mask semantic.
- // For scalar selects, ~Cond != 0 is not equivalent to Cond == 0.
- if (CondVT.isVector() && CondVT.getScalarType() != MVT::i1) {
+ if (CondVT.getScalarType() != MVT::i1) {
if (SDValue CondNot = IsNOT(Cond, DAG))
return DAG.getNode(N->getOpcode(), DL, VT,
DAG.getBitcast(CondVT, CondNot), RHS, LHS);
@@ -50008,7 +50006,7 @@ static SDValue combineCMov(SDNode *N, SelectionDAG &DAG,
// Ok, now make sure that Add is (add (cttz X), C2) and Const is a constant.
if (isa<ConstantSDNode>(Const) && Add.getOpcode() == ISD::ADD &&
Add.hasOneUse() && isa<ConstantSDNode>(Add.getOperand(1)) &&
- (Add.getOperand(0).getOpcode() == ISD::CTTZ_ZERO_POISON ||
+ (Add.getOperand(0).getOpcode() == ISD::CTTZ_ZERO_UNDEF ||
Add.getOperand(0).getOpcode() == ISD::CTTZ) &&
Add.getOperand(0).getOperand(0) == Cond.getOperand(0)) {
// This should constant fold.
@@ -50258,7 +50256,7 @@ static SDValue combineMulToPMADDWD(SDNode *N, const SDLoc &DL,
EVT VT = N->getValueType(0);
// Only support vXi32 vectors.
- if (!VT.isVectorOf(MVT::i32))
+ if (!VT.isVector() || VT.getVectorElementType() != MVT::i32)
return SDValue();
// Make sure the type is legal or can split/widen to a legal type.
@@ -50313,7 +50311,7 @@ static SDValue combineMulToPMADDWD(SDNode *N, const SDLoc &DL,
if (Op.getOpcode() == ISD::SIGN_EXTEND && N->isOnlyUserOf(Op.getNode())) {
SDValue Src = Op.getOperand(0);
// Convert sext(vXi16) to zext(vXi16).
- if (Src.getScalarValueSizeInBits() == 16)
+ if (Src.getScalarValueSizeInBits() == 16 && VT.getSizeInBits() <= 128)
return DAG.getNode(ISD::ZERO_EXTEND, DL, VT, Src);
// Convert sext(vXi8) to zext(vXi16 sext(vXi8)) on pre-SSE41 targets
// which will expand the extension.
@@ -50365,7 +50363,8 @@ static SDValue combineMulToPMULDQ(SDNode *N, const SDLoc &DL, SelectionDAG &DAG,
EVT VT = N->getValueType(0);
// Only support vXi64 vectors.
- if (!VT.isVectorOf(MVT::i64) || VT.getVectorNumElements() < 2 ||
+ if (!VT.isVector() || VT.getVectorElementType() != MVT::i64 ||
+ VT.getVectorNumElements() < 2 ||
!isPowerOf2_32(VT.getVectorNumElements()))
return SDValue();
@@ -50410,8 +50409,9 @@ static SDValue combineMulToPMADD52(SDNode *N, const SDLoc &DL,
// 128/256-bit vectors (v2i64/v4i64) require either AVX512-IFMA + VLX, or
// AVX-IFMA.
bool Supported512 = (VT == MVT::v8i64) && Subtarget.hasIFMA();
- bool SupportedSmall = (VT == MVT::v2i64 || VT == MVT::v4i64) &&
- (Subtarget.hasIFMA() || Subtarget.hasAVXIFMA());
+ bool SupportedSmall =
+ (VT == MVT::v2i64 || VT == MVT::v4i64) &&
+ ((Subtarget.hasIFMA() && Subtarget.hasVLX()) || Subtarget.hasAVXIFMA());
if (!Supported512 && !SupportedSmall)
return SDValue();
@@ -51398,9 +51398,6 @@ static SDValue combineVectorInsert(SDNode *N, SelectionDAG &DAG,
const X86Subtarget &Subtarget) {
EVT VT = N->getValueType(0);
unsigned Opcode = N->getOpcode();
- unsigned NumElts = VT.getVectorNumElements();
- unsigned NumBitsPerElt = VT.getScalarSizeInBits();
- const TargetLowering &TLI = DAG.getTargetLoweringInfo();
assert(((Opcode == X86ISD::PINSRB && VT == MVT::v16i8) ||
(Opcode == X86ISD::PINSRW && VT == MVT::v8i16) ||
Opcode == ISD::INSERT_VECTOR_ELT) &&
@@ -51410,48 +51407,13 @@ static SDValue combineVectorInsert(SDNode *N, SelectionDAG &DAG,
SDValue Scl = N->getOperand(1);
SDValue Idx = N->getOperand(2);
- if (Opcode == ISD::INSERT_VECTOR_ELT) {
- auto *CIdx = dyn_cast<ConstantSDNode>(Idx);
- if (CIdx && CIdx->getAPIntValue().uge(NumElts))
- return DAG.getPOISON(VT);
-
- // Fold insert_vector_elt(undef, elt, 0) --> scalar_to_vector(elt).
- if (Vec.isUndef() && isNullConstant(Idx))
- return DAG.getNode(ISD::SCALAR_TO_VECTOR, SDLoc(N), VT, Scl);
-
- // Attempt to fold neighboring pairs of inserted constants into a single
- // large constant insertion.
- // TODO: Consecutive loads might benefit as well?
- if (!DCI.isBeforeLegalize() && VT.isInteger() && CIdx &&
- (CIdx->getZExtValue() & 1) == 1 && TLI.isTypeLegal(VT)) {
- using namespace SDPatternMatch;
- auto *Cst = dyn_cast<ConstantSDNode>(Scl);
- MVT WideSVT = MVT::getIntegerVT(2 * NumBitsPerElt);
- MVT WideVT = MVT::getVectorVT(WideSVT, NumElts / 2);
- SDValue InnerVec, InnerScl;
- if (Cst && TLI.isTypeLegal(WideSVT) && TLI.isTypeLegal(WideVT) &&
- sd_match(Vec, m_OneUse(m_InsertElt(
- m_Value(InnerVec), m_Value(InnerScl),
- m_SpecificInt(CIdx->getZExtValue() - 1))))) {
- if (isa<ConstantSDNode>(InnerScl)) {
- SDLoc DL(N);
- SDValue Lo = DAG.getNode(ISD::ZERO_EXTEND, DL, WideSVT, InnerScl);
- SDValue Hi = DAG.getNode(ISD::ZERO_EXTEND, DL, WideSVT, Scl);
- Lo = DAG.getZeroExtendInReg(Lo, DL, VT.getScalarType());
- Hi = DAG.getNode(
- ISD::SHL, DL, WideSVT, Hi,
- DAG.getShiftAmountConstant(NumBitsPerElt, WideSVT, DL));
- unsigned NewIdx = (CIdx->getZExtValue() - 1) / 2;
- SDValue NewInsert = DAG.getInsertVectorElt(
- DL, DAG.getBitcast(WideVT, InnerVec),
- DAG.getNode(ISD::OR, DL, WideSVT, Lo, Hi), NewIdx);
- return DAG.getBitcast(VT, NewInsert);
- }
- }
- }
- }
+ // Fold insert_vector_elt(undef, elt, 0) --> scalar_to_vector(elt).
+ if (Opcode == ISD::INSERT_VECTOR_ELT && Vec.isUndef() && isNullConstant(Idx))
+ return DAG.getNode(ISD::SCALAR_TO_VECTOR, SDLoc(N), VT, Scl);
if (Opcode == X86ISD::PINSRB || Opcode == X86ISD::PINSRW) {
+ unsigned NumBitsPerElt = VT.getScalarSizeInBits();
+ const TargetLowering &TLI = DAG.getTargetLoweringInfo();
if (TLI.SimplifyDemandedBits(SDValue(N, 0),
APInt::getAllOnes(NumBitsPerElt), DCI))
return SDValue(N, 0);
@@ -52024,9 +51986,9 @@ static SDValue combineAndMaskToShift(SDNode *N, const SDLoc &DL,
if (EltBitWidth != DAG.ComputeNumSignBits(Op0))
return SDValue();
- unsigned ShiftVal = EltBitWidth - SplatVal.countr_one();
- SDValue Shift = getTargetVShiftByConstNode(
- X86ISD::VSRLI, DL, VT.getSimpleVT(), Op0, ShiftVal, DAG);
+ unsigned ShiftVal = SplatVal.countr_one();
+ SDValue ShAmt = DAG.getTargetConstant(EltBitWidth - ShiftVal, DL, MVT::i8);
+ SDValue Shift = DAG.getNode(X86ISD::VSRLI, DL, VT, Op0, ShAmt);
return DAG.getBitcast(N->getValueType(0), Shift);
}
@@ -52076,41 +52038,6 @@ static SDValue combineAndNotOrIntoAndNotAnd(SDNode *N, const SDLoc &DL,
return SDValue();
}
-// Fold vXi1 logicop(truncate(N0),truncate(N1)) -> truncate(logicop(X,Y))
-// Generic vector logicops are always quicker than predicate equivalents.
-static SDValue combineMaskBitOp(SDNode *N, const SDLoc &DL, SelectionDAG &DAG) {
- using namespace SDPatternMatch;
- unsigned Opc = N->getOpcode();
- assert(ISD::isBitwiseLogicOp(Opc) && "Unexpected opcode!");
- EVT VT = N->getValueType(0);
- const TargetLowering &TLI = DAG.getTargetLoweringInfo();
-
- if (!VT.isVector() || VT.getScalarType() != MVT::i1 || !TLI.isTypeLegal(VT))
- return SDValue();
-
- SDValue Src0, Src1;
- if (sd_match(
- N, m_BitwiseLogic(m_Trunc(m_Value(Src0)), m_Trunc(m_Value(Src1))))) {
- EVT SrcVT = Src0.getValueType();
- if (SrcVT == Src1.getValueType() && TLI.isOperationLegal(Opc, SrcVT)) {
- return DAG.getNode(ISD::TRUNCATE, DL, VT,
- DAG.getNode(Opc, DL, SrcVT, Src0, Src1));
- }
- }
-
- // Attempt to match expanded ANDNOT pattern (if AND is legal then ANDNP is).
- if (sd_match(N, m_And(m_OneUse(m_Not(m_Trunc(m_Value(Src0)))),
- m_Trunc(m_Value(Src1))))) {
- EVT SrcVT = Src0.getValueType();
- if (SrcVT == Src1.getValueType() && TLI.isOperationLegal(ISD::AND, SrcVT)) {
- return DAG.getNode(ISD::TRUNCATE, DL, VT,
- DAG.getNode(X86ISD::ANDNP, DL, SrcVT, Src0, Src1));
- }
- }
-
- return SDValue();
-}
-
// This function recognizes cases where X86 bzhi instruction can replace and
// 'and-load' sequence.
// In case of loading integer value from an array of constants which is defined
@@ -52229,7 +52156,8 @@ static SDValue combineScalarAndWithMaskSetcc(SDNode *N, SelectionDAG &DAG,
EVT SrcVT = Src.getValueType();
const TargetLowering &TLI = DAG.getTargetLoweringInfo();
- if (!SrcVT.isVectorOf(MVT::i1) || !TLI.isTypeLegal(SrcVT))
+ if (!SrcVT.isVector() || SrcVT.getVectorElementType() != MVT::i1 ||
+ !TLI.isTypeLegal(SrcVT))
return SDValue();
if (Src.getOpcode() != ISD::CONCAT_VECTORS)
@@ -52597,9 +52525,6 @@ static SDValue combineAnd(SDNode *N, SelectionDAG &DAG,
if (SDValue V = combineScalarAndWithMaskSetcc(N, DAG, Subtarget))
return V;
- if (SDValue R = combineMaskBitOp(N, dl, DAG))
- return R;
-
if (SDValue R = combineBitOpWithMOVMSK(N->getOpcode(), dl, N0, N1, DAG))
return R;
@@ -53267,70 +53192,6 @@ static SDValue combineAddOrSubToADCOrSBB(SDNode *N, const SDLoc &DL,
return SDValue();
}
-/// GF2P8AFFINEQB computes each output bit from one row of the
-/// 8x8 affine matrix, then xors the corresponding immediate bit.
-/// OR-ing the result with a constant byte splat forces selected output
-/// bits to 1, so the original affine result for those bits is irrelevant.
-///
-/// Fold:
-/// gf2p8affineqb(X, Matrix, Imm) | Mask
-/// into:
-/// gf2p8affineqb(X, NewMatrix, Imm | Mask)
-///
-static SDValue combineOrWithGF2P8AFFINEQB(SDNode *N, const SDLoc &DL,
- SelectionDAG &DAG, EVT VT) {
- using namespace SDPatternMatch;
- assert(N->getOpcode() == ISD::OR && "Expected OR node");
-
- SDValue LHS = N->getOperand(0), RHS = N->getOperand(1);
-
- SDValue X, Matrix, SplatOp;
- APInt Imm;
- if (!sd_match(N, m_Or(m_OneUse(m_TernaryOp(X86ISD::GF2P8AFFINEQB, m_Value(X),
- m_Value(Matrix), m_ConstInt(Imm))),
- m_Value(SplatOp))))
- return SDValue();
-
- APInt SplatVal;
- if (!X86::isConstantSplat(SplatOp, SplatVal, /*AllowPartialUndefs=*/false))
- return SDValue();
-
- if (N->getFlags().hasDisjoint() || DAG.haveNoCommonBitsSet(LHS, RHS)) {
- // Fold: (GF2P8AFFINEQB(X, Matrix, Imm) or_disjoint SplatVal)
- // -> GF2P8AFFINEQB(X, Matrix, Imm ^ SplatVal)
- // When OR is disjoint (no common bits), the splat constant can be folded
- // directly into the GF2P8AFFINEQB immediate via XOR.
- uint64_t NewImm = (Imm.getZExtValue() ^ SplatVal.getZExtValue()) & 0xFF;
- return DAG.getNode(X86ISD::GF2P8AFFINEQB, DL, VT, X, Matrix,
- DAG.getTargetConstant(NewImm, DL, MVT::i8));
- }
-
- APInt UndefElts;
- SmallVector<APInt, 16> OldMatrix;
- if (!getTargetConstantBitsFromNode(Matrix, 8, UndefElts, OldMatrix,
- /*AllowWholeUndefs=*/false,
- /*AllowPartialUndefs=*/false))
- return SDValue();
-
- uint8_t Mask8 = SplatVal.getZExtValue() & 0xFF;
- uint8_t NewImm = Imm.getZExtValue() | Mask8;
- // For each output bit selected by Mask, clear the corresponding matrix row
- // and set the same bit in the immediate. Rows are encoded in reverse bit
- // order within each byte: output bit B corresponds to row 7 - B.
- SmallVector<SDValue, 64> NewMatrixOps;
- for (unsigned I = 0, E = VT.getVectorNumElements(); I != E; ++I) {
- unsigned OutBit = 7 - (I & 7);
- uint8_t OldRow = OldMatrix[I].getZExtValue();
- uint8_t NewRow = ((Mask8 >> OutBit) & 1) ? 0x00 : OldRow;
- NewMatrixOps.push_back(DAG.getConstant(NewRow, DL, MVT::i8));
- }
-
- SDValue NewMatrix = DAG.getBuildVector(VT, DL, NewMatrixOps);
- SDValue NewGF = DAG.getNode(X86ISD::GF2P8AFFINEQB, DL, VT, X, NewMatrix,
- DAG.getTargetConstant(NewImm, DL, MVT::i8));
- return NewGF;
-}
-
static SDValue combineOrXorWithSETCC(unsigned Opc, const SDLoc &DL, EVT VT,
SDValue N0, SDValue N1,
SelectionDAG &DAG) {
@@ -53410,9 +53271,6 @@ static SDValue combineOr(SDNode *N, SelectionDAG &DAG,
if (SDValue SetCC = combineAndOrForCcmpCtest(N, DAG, DCI, Subtarget))
return SetCC;
- if (SDValue R = combineMaskBitOp(N, dl, DAG))
- return R;
-
if (SDValue R = combineBitOpWithMOVMSK(N->getOpcode(), dl, N0, N1, DAG))
return R;
@@ -53426,9 +53284,6 @@ static SDValue combineOr(SDNode *N, SelectionDAG &DAG,
DAG, DCI, Subtarget))
return FPLogic;
- if (SDValue R = combineOrWithGF2P8AFFINEQB(N, dl, DAG, VT))
- return R;
-
if (DCI.isBeforeLegalizeOps())
return SDValue();
@@ -53541,9 +53396,6 @@ static SDValue combineOr(SDNode *N, SelectionDAG &DAG,
if (SDValue R = combineOrXorWithSETCC(N->getOpcode(), dl, VT, N0, N1, DAG))
return R;
- if (SDValue R = combineOrWithGF2P8AFFINEQB(N, dl, DAG, VT))
- return R;
-
return SDValue();
}
@@ -53906,29 +53758,6 @@ static SDValue combineConstantPoolLoads(SDNode *N, const SDLoc &dl,
return SDValue();
}
-static SDValue combineAtomicLoad(SDNode *N, SelectionDAG &DAG,
- TargetLowering::DAGCombinerInfo &DCI) {
- if (!DCI.isBeforeLegalize())
- return SDValue();
-
- auto *AN = cast<AtomicSDNode>(N);
- EVT VT = AN->getValueType(0);
- if (!VT.getScalarType().isFloatingPoint())
- return SDValue();
-
- unsigned BitWidth = VT.getStoreSizeInBits();
- if (BitWidth != VT.getSizeInBits())
- return SDValue();
-
- SDLoc DL(N);
- EVT IntVT = EVT::getIntegerVT(*DAG.getContext(), BitWidth);
- SDValue IntLoad = DAG.getAtomic(
- ISD::ATOMIC_LOAD, DL, IntVT, DAG.getVTList(IntVT, MVT::Other),
- {AN->getChain(), AN->getBasePtr()}, AN->getMemOperand());
- SDValue Cast = DAG.getBitcast(VT, IntLoad);
- return DAG.getMergeValues({Cast, IntLoad.getValue(1)}, DL);
-}
-
static SDValue combineLoad(SDNode *N, SelectionDAG &DAG,
TargetLowering::DAGCombinerInfo &DCI,
const X86Subtarget &Subtarget) {
@@ -54916,8 +54745,8 @@ static SDValue combineVEXTRACT_STORE(SDNode *N, SelectionDAG &DAG,
/// A horizontal-op B, for some already available A and B, and if so then LHS is
/// set to A, RHS to B, and the routine returns 'true'.
static bool isHorizontalBinOp(unsigned HOpcode, SDValue &LHS, SDValue &RHS,
- const SelectionDAG &DAG,
- const X86Subtarget &Subtarget, bool IsCommutative,
+ SelectionDAG &DAG, const X86Subtarget &Subtarget,
+ bool IsCommutative,
SmallVectorImpl<int> &PostShuffleMask,
bool ForceHorizOp) {
// If either operand is undef, bail out. The binop should be simplified.
@@ -54962,11 +54791,8 @@ static bool isHorizontalBinOp(unsigned HOpcode, SDValue &LHS, SDValue &RHS,
ShuffleMask.assign(ScaledMask.begin(), ScaledMask.end());
}
if (UseSubVector && SrcOps.size() == 1 &&
- scaleShuffleElements(SrcMask, 2 * NumElts, ScaledMask) &&
- SrcOps[0].getOpcode() == ISD::CONCAT_VECTORS &&
- SrcOps[0].getNumOperands() == 2) {
- N0 = SrcOps[0].getOperand(0);
- N1 = SrcOps[0].getOperand(1);
+ scaleShuffleElements(SrcMask, 2 * NumElts, ScaledMask)) {
+ std::tie(N0, N1) = DAG.SplitVector(SrcOps[0], SDLoc(Op));
ArrayRef<int> Mask = ArrayRef<int>(ScaledMask).slice(0, NumElts);
ShuffleMask.assign(Mask.begin(), Mask.end());
}
@@ -55075,9 +54901,12 @@ static bool isHorizontalBinOp(unsigned HOpcode, SDValue &LHS, SDValue &RHS,
SDValue NewLHS = A.getNode() ? A : B; // If A is 'UNDEF', use B for it.
SDValue NewRHS = B.getNode() ? B : A; // If B is 'UNDEF', use A for it.
- // Avoid 128-bit multi lane shuffles if pre-AVX2 and FP (integer will split).
bool IsIdentityPostShuffle =
isSequentialOrUndefInRange(PostShuffleMask, 0, NumElts, 0);
+ if (IsIdentityPostShuffle)
+ PostShuffleMask.clear();
+
+ // Avoid 128-bit multi lane shuffles if pre-AVX2 and FP (integer will split).
if (!IsIdentityPostShuffle && !Subtarget.hasAVX2() && VT.isFloatingPoint() &&
isMultiLaneShuffleMask(128, VT.getScalarSizeInBits(), PostShuffleMask))
return false;
@@ -55099,8 +54928,8 @@ static bool isHorizontalBinOp(unsigned HOpcode, SDValue &LHS, SDValue &RHS,
DAG, Subtarget))
return false;
- LHS = NewLHS;
- RHS = NewRHS;
+ LHS = DAG.getBitcast(VT, NewLHS);
+ RHS = DAG.getBitcast(VT, NewRHS);
return true;
}
@@ -55121,21 +54950,6 @@ static SDValue combineToHorizontalAddSub(SDNode *N, SelectionDAG &DAG,
N->user_begin()->getOperand(1).getOpcode() == HorizOpcode);
};
- auto CanonicalizeHorizOps = [&](const SDLoc &DL, SDValue &LHS, SDValue &RHS) {
- assert(PostShuffleMask.size() == VT.getVectorNumElements() &&
- "Illegal shuffle mask");
- LHS = DAG.getBitcast(VT, LHS);
- RHS = DAG.getBitcast(VT, RHS);
- if (VT.is256BitVector() && isUndefUpperHalf(PostShuffleMask) &&
- !isLaneCrossingShuffleMask(128, VT.getScalarSizeInBits(),
- PostShuffleMask)) {
- LHS = extract128BitVector(LHS, 0, DAG, DL);
- RHS = extract128BitVector(RHS, 0, DAG, DL);
- PostShuffleMask.truncate(PostShuffleMask.size() / 2);
- }
- return LHS.getSimpleValueType();
- };
-
switch (Opcode) {
case ISD::FADD:
case ISD::FSUB:
@@ -55146,12 +54960,11 @@ static SDValue combineToHorizontalAddSub(SDNode *N, SelectionDAG &DAG,
auto HorizOpcode = IsAdd ? X86ISD::FHADD : X86ISD::FHSUB;
if (isHorizontalBinOp(HorizOpcode, LHS, RHS, DAG, Subtarget, IsAdd,
PostShuffleMask, MergableHorizOp(HorizOpcode))) {
- SDLoc DL(N);
- MVT HorizVT = CanonicalizeHorizOps(DL, LHS, RHS);
- SDValue HorizBinOp = DAG.getNode(HorizOpcode, DL, HorizVT, LHS, RHS);
- HorizBinOp = DAG.getVectorShuffle(HorizVT, DL, HorizBinOp, HorizBinOp,
- PostShuffleMask);
- return DAG.getInsertSubvector(DL, DAG.getUNDEF(VT), HorizBinOp, 0);
+ SDValue HorizBinOp = DAG.getNode(HorizOpcode, SDLoc(N), VT, LHS, RHS);
+ if (!PostShuffleMask.empty())
+ HorizBinOp = DAG.getVectorShuffle(VT, SDLoc(HorizBinOp), HorizBinOp,
+ DAG.getUNDEF(VT), PostShuffleMask);
+ return HorizBinOp;
}
}
break;
@@ -55163,6 +54976,7 @@ static SDValue combineToHorizontalAddSub(SDNode *N, SelectionDAG &DAG,
break;
if (VT == MVT::v8i16 || VT == MVT::v16i16 ||
(!IsSat && (VT == MVT::v4i32 || VT == MVT::v8i32))) {
+
SDValue LHS = N->getOperand(0);
SDValue RHS = N->getOperand(1);
auto HorizOpcode = IsSat ? (IsAdd ? X86ISD::HADDS : X86ISD::HSUBS)
@@ -55173,13 +54987,12 @@ static SDValue combineToHorizontalAddSub(SDNode *N, SelectionDAG &DAG,
ArrayRef<SDValue> Ops) {
return DAG.getNode(HorizOpcode, DL, Ops[0].getValueType(), Ops);
};
- SDLoc DL(N);
- MVT HorizVT = CanonicalizeHorizOps(DL, LHS, RHS);
- SDValue HorizBinOp = SplitOpsAndApply(DAG, Subtarget, DL, HorizVT,
+ SDValue HorizBinOp = SplitOpsAndApply(DAG, Subtarget, SDLoc(N), VT,
{LHS, RHS}, HOpBuilder);
- HorizBinOp = DAG.getVectorShuffle(HorizVT, DL, HorizBinOp, HorizBinOp,
- PostShuffleMask);
- return DAG.getInsertSubvector(DL, DAG.getUNDEF(VT), HorizBinOp, 0);
+ if (!PostShuffleMask.empty())
+ HorizBinOp = DAG.getVectorShuffle(VT, SDLoc(HorizBinOp), HorizBinOp,
+ DAG.getUNDEF(VT), PostShuffleMask);
+ return HorizBinOp;
}
}
break;
@@ -55493,7 +55306,7 @@ static SDValue combinePMULH(SDValue Src, EVT VT, const SDLoc &DL,
// Only handle vXi16 types that are at least 128-bits unless they will be
// widened.
- if (!VT.isVectorOf(MVT::i16))
+ if (!VT.isVector() || VT.getVectorElementType() != MVT::i16)
return SDValue();
// Input type should be at least vXi32.
@@ -55787,8 +55600,7 @@ static SDValue combineVTRUNC(SDNode *N, SelectionDAG &DAG,
}
static SDValue combineVTRUNCSAT(SDNode *N, SelectionDAG &DAG,
- TargetLowering::DAGCombinerInfo &DCI,
- const X86Subtarget &Subtarget) {
+ TargetLowering::DAGCombinerInfo &DCI) {
using namespace SDPatternMatch;
unsigned Opc = N->getOpcode();
EVT VT = N->getValueType(0);
@@ -55807,17 +55619,7 @@ static SDValue combineVTRUNCSAT(SDNode *N, SelectionDAG &DAG,
(EltSizeInBits * 2) == Src.getScalarValueSizeInBits() &&
isFreeToSplitVector(Src, DAG)) {
SDLoc DL(N);
- SDValue LHS, RHS;
- if (Src.getValueSizeInBits() == VT.getSizeInBits()) {
- assert(VT.is128BitVector() && "128-bit VTRUNC source expected");
- LHS = Src;
- RHS = getZeroVector(Src.getSimpleValueType(), Subtarget, DAG, DL);
- } else {
- std::tie(LHS, RHS) = splitVector(Src, DAG, DL);
- }
- assert(LHS.getValueSizeInBits() == VT.getSizeInBits() &&
- RHS.getValueSizeInBits() == VT.getSizeInBits() &&
- "PACK src/dst size mismatch");
+ auto [LHS, RHS] = splitVector(Src, DAG, DL);
unsigned PackOpc = Opc == X86ISD::VTRUNCS ? X86ISD::PACKSS : X86ISD::PACKUS;
SDValue Pack = DAG.getNode(PackOpc, DL, VT, LHS, RHS);
if (VT.is128BitVector())
@@ -56166,44 +55968,21 @@ static SDValue combineXorWithGF2P8AFFINEQB(SDNode *N, const SDLoc &DL,
SelectionDAG &DAG, EVT VT) {
using namespace SDPatternMatch;
- SDValue X, Y, XorOp;
- APInt Imm, ConstUndef;
-
+ SDValue X, Y, SplatOp;
+ APInt Imm;
+ // Use sd_match for structure matching - m_Xor handles commutation
if (!sd_match(N, m_Xor(m_OneUse(m_TernaryOp(X86ISD::GF2P8AFFINEQB, m_Value(X),
m_Value(Y), m_ConstInt(Imm))),
- m_Value(XorOp))))
+ m_Value(SplatOp))))
return SDValue();
// GF2P8AFFINEQB only operates on i8 vector types
assert((VT == MVT::v16i8 || VT == MVT::v32i8 || VT == MVT::v64i8) &&
"Unsupported GFNI type");
- unsigned NumElts = VT.getVectorNumElements();
-
- // Fold: GF2P8AFFINEQB(x, M) ^ x
- // => GF2P8AFFINEQB(x, M ^ IDENTITY)
- // The "identity" is a noop. It adds a XOR by the unmodified input.
- SmallVector<APInt> YEltBits;
- if (X == XorOp && getTargetConstantBitsFromNode(Y, 8, ConstUndef, YEltBits,
- /*AllowWholeUndefs=*/false)) {
- const uint8_t IdentityMatrix[] = {128, 64, 32, 16, 8, 4, 2, 1};
-
- SmallVector<SDValue> BuildMatrix;
- for (unsigned I = 0; I != NumElts; ++I) {
- APInt MatrixRow = YEltBits[I] ^ IdentityMatrix[I % 8];
- BuildMatrix.push_back(DAG.getConstant(MatrixRow, DL, MVT::i8));
- }
-
- SDValue NewMatrix = DAG.getBuildVector(VT, DL, BuildMatrix);
-
- return DAG.getNode(X86ISD::GF2P8AFFINEQB, DL, VT, X, NewMatrix,
- DAG.getTargetConstant(Imm, DL, MVT::i8));
- }
-
- // Fold: GF2P8AFFINEQB(x, m, Imm) ^ Splat(C)
- // => GF2P8AFFINEQB(x, m, Imm ^ C)
+ // Use X86::isConstantSplat for robust splat constant extraction
APInt SplatVal;
- if (!X86::isConstantSplat(XorOp, SplatVal, /*AllowPartialUndefs=*/false))
+ if (!X86::isConstantSplat(SplatOp, SplatVal, /*AllowPartialUndefs=*/false))
return SDValue();
uint64_t NewImm = Imm.getZExtValue() ^ SplatVal.getZExtValue();
@@ -56211,49 +55990,32 @@ static SDValue combineXorWithGF2P8AFFINEQB(SDNode *N, const SDLoc &DL,
DAG.getTargetConstant(NewImm, DL, MVT::i8));
}
-// Given that vgf2p8affineqb performs a XOR permutation, two affines that share
-// a operand can be reassociated through a standalone XOR.
+// Fold: vgf2p8affineqb(x, m1, i1) ^ vgf2p8affineqb(x, m2, i2)
+// => vgf2p8affineqb(x, m1 ^ m2, i1 ^ i2)
+// The matrix in vgf2p8affineqb determines which bits of the input are XORed
+// together. XORing two affine transformations of the same input can be folded
+// by XORing both their matrices and immediates together.
static SDValue combineXorWithTwoGF2P8AFFINEQB(SDNode *N, const SDLoc &DL,
SelectionDAG &DAG, EVT VT) {
using namespace SDPatternMatch;
- SDValue X0, X1, M0, M1;
+ SDValue X0, Y0, Y1;
APInt Imm0, Imm1;
// Use sd_match for structure matching - m_Xor handles commutation
- if (!sd_match(N,
- m_Xor(m_OneUse(m_TernaryOp(X86ISD::GF2P8AFFINEQB, m_Value(X0),
- m_Value(M0), m_ConstInt(Imm0))),
- m_OneUse(m_TernaryOp(X86ISD::GF2P8AFFINEQB, m_Value(X1),
- m_Value(M1), m_ConstInt(Imm1))))))
+ // Match: GF2P8AFFINEQB(x, m1, i1) ^ GF2P8AFFINEQB(x, m2, i2)
+ if (!sd_match(
+ N, m_Xor(m_OneUse(m_TernaryOp(X86ISD::GF2P8AFFINEQB, m_Value(X0),
+ m_Value(Y0), m_ConstInt(Imm0))),
+ m_OneUse(m_TernaryOp(X86ISD::GF2P8AFFINEQB, m_Deferred(X0),
+ m_Value(Y1), m_ConstInt(Imm1))))))
return SDValue();
assert((VT == MVT::v16i8 || VT == MVT::v32i8 || VT == MVT::v64i8) &&
"Unsupported GFNI type");
- // Fold: GF2P8AFFINEQB(x0, m, i1) ^ GF2P8AFFINEQB(x1, m, i2)
- // => GF2P8AFFINEQB(x0 ^ x1, m, i1 ^ i2)
- // This instruction performs an XOR permutation of the input, which is
- // associative. Therefore XORing before permuting is equivalent.
- if (M0 == M1) {
- uint64_t NewImm = Imm0.getZExtValue() ^ Imm1.getZExtValue();
-
- SDValue NewSrc = DAG.getNode(ISD::XOR, DL, VT, X0, X1);
-
- return DAG.getNode(X86ISD::GF2P8AFFINEQB, DL, VT, NewSrc, M0,
- DAG.getTargetConstant(NewImm, DL, MVT::i8));
- }
-
- // Fold: vgf2p8affineqb(x, m0, i1) ^ vgf2p8affineqb(x, m1, i2)
- // => vgf2p8affineqb(x, m0 ^ m1, i1 ^ i2)
- // The matrix in vgf2p8affineqb determines which bits of the input are XORed
- // together. XORing two affine transformations of the same input can be folded
- // by XORing both their matrices and immediates together.
- if (X0 != X1)
- return SDValue();
-
uint64_t NewImm = Imm0.getZExtValue() ^ Imm1.getZExtValue();
- SDValue NewMatrix = DAG.getNode(ISD::XOR, DL, VT, M0, M1);
+ SDValue NewMatrix = DAG.getNode(ISD::XOR, DL, VT, Y0, Y1);
return DAG.getNode(X86ISD::GF2P8AFFINEQB, DL, VT, X0, NewMatrix,
DAG.getTargetConstant(NewImm, DL, MVT::i8));
@@ -56274,14 +56036,14 @@ static SDValue combineXorSubCTLZ(SDNode *N, const SDLoc &DL, SelectionDAG &DAG,
SDValue N0 = N->getOperand(0);
SDValue N1 = N->getOperand(1);
- if (N0.getOpcode() != ISD::CTLZ_ZERO_POISON &&
- N1.getOpcode() != ISD::CTLZ_ZERO_POISON)
+ if (N0.getOpcode() != ISD::CTLZ_ZERO_UNDEF &&
+ N1.getOpcode() != ISD::CTLZ_ZERO_UNDEF)
return SDValue();
SDValue OpCTLZ;
SDValue OpSizeTM1;
- if (N1.getOpcode() == ISD::CTLZ_ZERO_POISON) {
+ if (N1.getOpcode() == ISD::CTLZ_ZERO_UNDEF) {
OpCTLZ = N1;
OpSizeTM1 = N0;
} else if (N->getOpcode() == ISD::SUB) {
@@ -56334,9 +56096,6 @@ static SDValue combineXor(SDNode *N, SelectionDAG &DAG,
if (SDValue Cmp = foldVectorXorShiftIntoCmp(N, DAG, Subtarget))
return Cmp;
- if (SDValue R = combineMaskBitOp(N, DL, DAG))
- return R;
-
if (SDValue R = combineBitOpWithMOVMSK(N->getOpcode(), DL, N0, N1, DAG))
return R;
@@ -56424,7 +56183,7 @@ static SDValue combineBITREVERSE(SDNode *N, SelectionDAG &DAG,
if (VT.isInteger() && N0.getOpcode() == ISD::BITCAST && N0.hasOneUse()) {
SDValue Src = N0.getOperand(0);
EVT SrcVT = Src.getValueType();
- if (SrcVT.isVectorOf(MVT::i1) &&
+ if (SrcVT.isVector() && SrcVT.getScalarType() == MVT::i1 &&
(DCI.isBeforeLegalize() ||
DAG.getTargetLoweringInfo().isTypeLegal(SrcVT)) &&
Subtarget.hasSSSE3()) {
@@ -56881,101 +56640,6 @@ static SDValue combineAndnp(SDNode *N, SelectionDAG &DAG,
return SDValue();
}
-// Strip ext/trunc/mask wrappers that do not change a bit index interpreted
-// modulo BW. Don't peek below log2(BW) bits (e.g. through a zext from i1):
-// a narrower value no longer determines the bit index on its own.
-static SDValue peekThroughBitPosExtTrunc(SDValue V, unsigned BW) {
- assert(V.getScalarValueSizeInBits() >= Log2_32(BW) &&
- "bit position type must cover the whole index range");
- APInt LowBits =
- APInt::getLowBitsSet(V.getScalarValueSizeInBits(), Log2_32(BW));
- for (;;) {
- unsigned Op = V.getOpcode();
- if ((Op == ISD::TRUNCATE || Op == ISD::ZERO_EXTEND ||
- Op == ISD::ANY_EXTEND) &&
- V.getOperand(0).getScalarValueSizeInBits() >= Log2_32(BW)) {
- V = V.getOperand(0);
- LowBits = LowBits.zextOrTrunc(V.getScalarValueSizeInBits());
- continue;
- }
- if (Op == ISD::AND) {
- // The mask must keep all bits the bit-test cares about set.
- auto *C = dyn_cast<ConstantSDNode>(V.getOperand(1));
- if (C && LowBits.isSubsetOf(C->getAPIntValue())) {
- V = V.getOperand(0);
- continue;
- }
- }
- return V;
- }
-}
-
-// Try to merge a (X86ISD::BT Src, BitNo) with a sibling bit-modifying op on
-// Src (AND(Src, rotl -2, X), OR(Src, shl 1, X), XOR(Src, shl 1, X)) into a
-// single flag-producing X86ISD::{BTR,BTS,BTC} node. Both BT and BTR/BTS/BTC
-// set CF from the pre-op bit value, so one instruction subsumes the other.
-static SDValue combineBTToBitOpFlag(SDNode *N, SelectionDAG &DAG) {
- using namespace SDPatternMatch;
- // Src is the BT source; matching against it with m_Specific requires
- // SDValue identity with the modify's operand, so no peek-through is needed
- // here (only the bit *position* may differ by ext/trunc/and-mask).
- SDValue Src = N->getOperand(0);
- SDValue BitNo = N->getOperand(1);
- EVT VT = Src.getValueType();
- SDLoc DL(N);
-
- // X86ISD::BT is only emitted for i32/i64 (smaller widths are promoted in
- // getBT before the node is created).
- assert((VT == MVT::i32 || VT == MVT::i64) &&
- "X86ISD::BT is only emitted for i32/i64");
-
- unsigned BW = VT.getScalarSizeInBits();
- SDValue PeeledBitNo = peekThroughBitPosExtTrunc(BitNo, BW);
-
- for (SDNode *User : Src->users()) {
- if (User == N)
- continue;
-
- unsigned FlagOp = 0;
- SDValue ShAmt;
- if (sd_match(User,
- m_And(m_Specific(Src),
- m_OneUse(m_Rotl(m_SpecificInt(APInt::getAllOnes(BW) - 1),
- m_Value(ShAmt)))))) {
- // (and Src, (rotl -2, X)): clears bit X.
- FlagOp = X86ISD::BTR;
- } else if (sd_match(User, m_Or(m_Specific(Src),
- m_OneUse(m_Shl(m_SpecificInt(1),
- m_Value(ShAmt)))))) {
- // (or Src, (shl 1, X)): sets bit X.
- FlagOp = X86ISD::BTS;
- } else if (sd_match(User, m_Xor(m_Specific(Src),
- m_OneUse(m_Shl(m_SpecificInt(1),
- m_Value(ShAmt)))))) {
- // (xor Src, (shl 1, X)): flips bit X.
- FlagOp = X86ISD::BTC;
- } else {
- continue;
- }
-
- // The BT and the bit-op must address the same bit. They can differ only
- // by truncation/extension or an AND that preserves the low log2(BW) bits.
- if (peekThroughBitPosExtTrunc(ShAmt, BW) != PeeledBitNo)
- continue;
-
- // BT's bit index is constrained to Src's type, so we can reuse it as-is
- // for BTR/BTS/BTC's bit index operand.
- assert(BitNo.getValueType() == VT && "BT bit index must match Src type");
- SDValue New =
- DAG.getNode(FlagOp, DL, DAG.getVTList(VT, MVT::i32), Src, BitNo);
- // Reroute the value output through User's consumers.
- DAG.ReplaceAllUsesOfValueWith(SDValue(User, 0), New.getValue(0));
- // Return the flags output so combineBT installs it as N's replacement.
- return New.getValue(1);
- }
- return SDValue();
-}
-
static SDValue combineBT(SDNode *N, SelectionDAG &DAG,
TargetLowering::DAGCombinerInfo &DCI) {
SDValue N1 = N->getOperand(1);
@@ -56989,9 +56653,6 @@ static SDValue combineBT(SDNode *N, SelectionDAG &DAG,
return SDValue(N, 0);
}
- if (SDValue V = combineBTToBitOpFlag(N, DAG))
- return V;
-
return SDValue();
}
@@ -57847,7 +57508,7 @@ static SDValue combineSetCC(SDNode *N, SelectionDAG &DAG,
// Both of these patterns can be better optimized in
// DAGCombiner::foldAndOrOfSETCC. Note this only applies for scalar
// integers which is checked above.
- if (ISD::isAbsOpcode(LHS.getOpcode()) && LHS.hasOneUse()) {
+ if (LHS.getOpcode() == ISD::ABS && LHS.hasOneUse()) {
if (auto *C = dyn_cast<ConstantSDNode>(RHS)) {
const APInt &CInt = C->getAPIntValue();
// We can better optimize this case in DAGCombiner::foldAndOrOfSETCC.
@@ -57867,7 +57528,7 @@ static SDValue combineSetCC(SDNode *N, SelectionDAG &DAG,
return V;
}
- if (VT.isVectorOf(MVT::i1) &&
+ if (VT.isVector() && VT.getVectorElementType() == MVT::i1 &&
(CC == ISD::SETNE || CC == ISD::SETEQ || ISD::isSignedIntSetCC(CC))) {
// Using temporaries to avoid messing up operand ordering for later
// transformations if this doesn't work.
@@ -59231,7 +58892,8 @@ static SDValue matchPMADDWD(SelectionDAG &DAG, SDNode *N,
if (!Subtarget.hasSSE2())
return SDValue();
- if (!VT.isVectorOf(MVT::i32) || VT.getVectorNumElements() < 4 ||
+ if (!VT.isVector() || VT.getVectorElementType() != MVT::i32 ||
+ VT.getVectorNumElements() < 4 ||
!isPowerOf2_32(VT.getVectorNumElements()))
return SDValue();
@@ -59283,12 +58945,12 @@ static SDValue matchPMADDWD(SelectionDAG &DAG, SDNode *N,
return SDValue();
if (!Mul) {
// First time an extract_elt's source vector is visited. Must be a MUL
- // with at least 2X number of vector elements than the BUILD_VECTOR.
+ // with 2X number of vector elements than the BUILD_VECTOR.
// Both extracts must be from same MUL.
Mul = Vec0L;
if ((Mul.getOpcode() != ISD::MUL && Mul.getOpcode() != ISD::SHL &&
Mul.getOpcode() != ISD::SIGN_EXTEND) ||
- Mul.getValueType().getVectorNumElements() < (2 * e))
+ Mul.getValueType().getVectorNumElements() != 2 * e)
return SDValue();
}
// Check that the extract is from the same MUL previously seen.
@@ -59298,7 +58960,6 @@ static SDValue matchPMADDWD(SelectionDAG &DAG, SDNode *N,
EVT TruncVT = EVT::getVectorVT(*DAG.getContext(), MVT::i16,
VT.getVectorNumElements() * 2);
- EVT MulVT = TruncVT.changeVectorElementType(*DAG.getContext(), MVT::i32);
SDValue N0, N1;
if (Mul.getOpcode() == ISD::MUL) {
@@ -59308,37 +58969,34 @@ static SDValue matchPMADDWD(SelectionDAG &DAG, SDNode *N,
Mode == ShrinkMode::MULU16)
return SDValue();
- N0 = DAG.getExtractSubvector(DL, MulVT, Mul.getOperand(0), 0);
- N1 = DAG.getExtractSubvector(DL, MulVT, Mul.getOperand(1), 0);
- N0 = DAG.getNode(ISD::TRUNCATE, DL, TruncVT, N0);
- N1 = DAG.getNode(ISD::TRUNCATE, DL, TruncVT, N1);
+ N0 = DAG.getNode(ISD::TRUNCATE, DL, TruncVT, Mul.getOperand(0));
+ N1 = DAG.getNode(ISD::TRUNCATE, DL, TruncVT, Mul.getOperand(1));
} else if (Mul.getOpcode() == ISD::SHL) {
SDValue ShVal = Mul.getOperand(0);
if (ShVal.getOpcode() != ISD::SIGN_EXTEND)
return SDValue();
N0 = ShVal.getOperand(0);
- if (N0.getValueType().getScalarType() != MVT::i16)
+ if (N0.getValueType() != TruncVT)
return SDValue();
- // A shift by more 15 or more would overflow a signed i16.
+ // A shift by more than 15 would overflow an i16.
if (!ISD::matchUnaryPredicate(Mul.getOperand(1), [](ConstantSDNode *C) {
- return C->getAPIntValue().ult(15);
+ return C->getAPIntValue().ule(15);
}))
return SDValue();
- N0 = DAG.getExtractSubvector(DL, TruncVT, N0, 0);
- N1 = DAG.getExtractSubvector(DL, MulVT, Mul.getOperand(1), 0);
N1 = DAG.getNode(ISD::SHL, DL, TruncVT, DAG.getConstant(1, DL, TruncVT),
- DAG.getZExtOrTrunc(N1, DL, TruncVT));
+ DAG.getZExtOrTrunc(Mul.getOperand(1), DL, TruncVT));
} else {
assert(Mul.getOpcode() == ISD::SIGN_EXTEND);
- if (Mul.getOperand(0).getValueType().getScalarType() != MVT::i16)
+ // Add a trivial multiplication with 1 so that we can make use of VPMADDWD.
+ N0 = Mul.getOperand(0);
+
+ if (N0.getValueType() != TruncVT)
return SDValue();
- // Add a trivial multiplication with 1 so that we can make use of VPMADDWD.
- N0 = DAG.getExtractSubvector(DL, TruncVT, Mul.getOperand(0), 0);
N1 = DAG.getConstant(1, DL, TruncVT);
}
@@ -59367,7 +59025,8 @@ static SDValue matchPMADDWD_2(SelectionDAG &DAG, SDNode *N,
if (!Subtarget.hasSSE2())
return SDValue();
- if (!VT.isVectorOf(MVT::i32) || VT.getVectorNumElements() < 4 ||
+ if (!VT.isVector() || VT.getVectorElementType() != MVT::i32 ||
+ VT.getVectorNumElements() < 4 ||
!isPowerOf2_32(VT.getVectorNumElements()))
return SDValue();
@@ -59484,12 +59143,15 @@ static SDValue combineAddOfPMADDWD(SelectionDAG &DAG, SDValue N0, SDValue N1,
unsigned NumElts = VT.getVectorNumElements();
MVT OpVT = N0.getOperand(0).getSimpleValueType();
+ APInt DemandedBits = APInt::getAllOnes(OpVT.getScalarSizeInBits());
APInt DemandedHiElts = APInt::getSplat(2 * NumElts, APInt(2, 2));
- bool Op0HiZero = DAG.MaskedVectorIsZero(N0.getOperand(0), DemandedHiElts) ||
- DAG.MaskedVectorIsZero(N0.getOperand(1), DemandedHiElts);
- bool Op1HiZero = DAG.MaskedVectorIsZero(N1.getOperand(0), DemandedHiElts) ||
- DAG.MaskedVectorIsZero(N1.getOperand(1), DemandedHiElts);
+ bool Op0HiZero =
+ DAG.MaskedValueIsZero(N0.getOperand(0), DemandedBits, DemandedHiElts) ||
+ DAG.MaskedValueIsZero(N0.getOperand(1), DemandedBits, DemandedHiElts);
+ bool Op1HiZero =
+ DAG.MaskedValueIsZero(N1.getOperand(0), DemandedBits, DemandedHiElts) ||
+ DAG.MaskedValueIsZero(N1.getOperand(1), DemandedBits, DemandedHiElts);
// TODO: Check for zero lower elements once we have actual codegen that
// creates them.
@@ -59588,6 +59250,11 @@ static SDValue matchVPMADD52(SDNode *N, SelectionDAG &DAG, const SDLoc &DL,
(!Subtarget.hasAVXIFMA() && !Subtarget.hasIFMA()))
return SDValue();
+ // Need AVX-512VL vector length extensions if operating on XMM/YMM registers
+ if (!Subtarget.hasAVXIFMA() && !Subtarget.hasVLX() &&
+ VT.getSizeInBits() < 512)
+ return SDValue();
+
const auto TotalSize = VT.getSizeInBits();
if (TotalSize < 128 || !isPowerOf2_64(TotalSize))
return SDValue();
@@ -60282,27 +59949,6 @@ static SDValue combineConcatVectorOps(const SDLoc &DL, MVT VT,
return DAG.getVectorShuffle(VT, DL, Concat0, Concat1, NewMask);
}
}
- // If we're concatenating 4 x 128-bit to 512-bit, see if we can
- // concat either 2 x 128-bit pair instead.
- // TODO: Can we do this generically?
- if (!IsSplat && NumOps == 4 && VT.is512BitVector() &&
- Subtarget.useAVX512Regs() &&
- (EltSizeInBits >= 32 || Subtarget.useBWIRegs())) {
- MVT HalfVT = VT.getHalfNumVectorElementsVT();
- SDValue Concat0 = combineConcatVectorOps(DL, HalfVT, Ops.slice(0, 2),
- DAG, Subtarget, Depth + 1);
- SDValue Concat1 = combineConcatVectorOps(DL, HalfVT, Ops.slice(2, 2),
- DAG, Subtarget, Depth + 1);
- if (Concat0 || Concat1) {
- Concat0 = Concat0 ? Concat0
- : DAG.getNode(ISD::CONCAT_VECTORS, DL, HalfVT,
- Ops.slice(0, 2));
- Concat1 = Concat1 ? Concat1
- : DAG.getNode(ISD::CONCAT_VECTORS, DL, HalfVT,
- Ops.slice(2, 2));
- return DAG.getNode(ISD::CONCAT_VECTORS, DL, VT, Concat0, Concat1);
- }
- }
break;
}
case X86ISD::VBROADCAST: {
@@ -60647,65 +60293,22 @@ static SDValue combineConcatVectorOps(const SDLoc &DL, MVT VT,
Op0.getOperand(1));
}
break;
- case X86ISD::VSHLDQ:
- case X86ISD::VSRLDQ:
- if (!IsSplat &&
- ((VT.is256BitVector() && Subtarget.hasInt256()) ||
- (VT.is512BitVector() && Subtarget.useBWIRegs())) &&
- llvm::all_of(Ops, [Op0](SDValue Op) {
- return Op0.getOperand(1) == Op.getOperand(1);
- })) {
- return DAG.getNode(Opcode, DL, VT, ConcatSubOperand(VT, Ops, 0),
- Op0.getOperand(1));
- }
- break;
+ case X86ISD::VPERMI:
case X86ISD::VROTLI:
case X86ISD::VROTRI:
if (!IsSplat &&
- ((VT.is256BitVector() && Subtarget.hasAVX512()) ||
+ ((VT.is256BitVector() && Subtarget.hasVLX()) ||
(VT.is512BitVector() && Subtarget.useAVX512Regs())) &&
llvm::all_of(Ops, [Op0](SDValue Op) {
return Op0.getOperand(1) == Op.getOperand(1);
})) {
+ assert(!(Opcode == X86ISD::VPERMI &&
+ Op0.getValueType().is128BitVector()) &&
+ "Illegal 128-bit X86ISD::VPERMI nodes");
return DAG.getNode(Opcode, DL, VT, ConcatSubOperand(VT, Ops, 0),
Op0.getOperand(1));
}
break;
- case ISD::ROTL:
- case ISD::ROTR:
- if (!IsSplat && ((VT.is256BitVector() && Subtarget.hasAVX512()) ||
- (VT.is512BitVector() && Subtarget.useAVX512Regs()))) {
- SDValue Concat0 = CombineSubOperand(VT, Ops, 0);
- SDValue Concat1 = CombineSubOperand(VT, Ops, 1);
- if (Concat0 || Concat1)
- return DAG.getNode(Opcode, DL, VT,
- Concat0 ? Concat0 : ConcatSubOperand(VT, Ops, 0),
- Concat1 ? Concat1 : ConcatSubOperand(VT, Ops, 1));
- }
- break;
- case X86ISD::VPERMI:
- if (!IsSplat && NumOps == 2 &&
- (VT.is512BitVector() && Subtarget.useAVX512Regs())) {
- // If both halves share the mask - then concat as VPERMI.
- if (Ops[0].getOperand(1) == Ops[1].getOperand(1))
- return DAG.getNode(Opcode, DL, VT, ConcatSubOperand(VT, Ops, 0),
- Op0.getOperand(1));
-
- // Fallback to a VPERMV3 variable mask shuffle.
- SmallVector<int, 8> Mask, HiMask;
- unsigned NumElts = VT.getVectorNumElements();
- DecodeVPERMMask(NumElts / 2, Ops[0].getConstantOperandVal(1), Mask);
- DecodeVPERMMask(NumElts / 2, Ops[1].getConstantOperandVal(1), HiMask);
- for (int M : HiMask)
- Mask.push_back(NumElts + M);
-
- return lowerShuffleWithPERMV(
- DL, VT, Mask,
- widenSubVector(VT, Ops[0].getOperand(0), false, Subtarget, DAG, DL),
- widenSubVector(VT, Ops[1].getOperand(0), false, Subtarget, DAG, DL),
- Subtarget, DAG);
- }
- break;
case ISD::AND:
case ISD::OR:
case ISD::XOR:
@@ -60732,6 +60335,7 @@ static SDValue combineConcatVectorOps(const SDLoc &DL, MVT VT,
break;
case X86ISD::PCMPEQ:
case X86ISD::PCMPGT:
+ // TODO: 512-bit PCMPEQ/PCMPGT -> VPCMP+VPMOVM2 handling.
if (!IsSplat && VT.is256BitVector() && Subtarget.hasInt256()) {
SDValue Concat0 = CombineSubOperand(VT, Ops, 0);
SDValue Concat1 = CombineSubOperand(VT, Ops, 1);
@@ -60741,19 +60345,6 @@ static SDValue combineConcatVectorOps(const SDLoc &DL, MVT VT,
Concat1 ? Concat1 : ConcatSubOperand(VT, Ops, 1));
break;
}
- if (!IsSplat && VT.is512BitVector() && Subtarget.useAVX512Regs() &&
- (EltSizeInBits >= 32 || Subtarget.useBWIRegs())) {
- if (IsConcatFree(VT, Ops, 0) && IsConcatFree(VT, Ops, 1)) {
- MVT BoolVT = VT.changeVectorElementType(MVT::i1);
- SDValue Cmp =
- DAG.getSetCC(DL, BoolVT, ConcatSubOperand(VT, Ops, 0),
- ConcatSubOperand(VT, Ops, 1),
- Opcode == X86ISD::PCMPEQ ? ISD::CondCode::SETEQ
- : ISD::CondCode::SETGT);
- return DAG.getNode(ISD::SIGN_EXTEND, DL, VT, Cmp);
- }
- break;
- }
if (!IsSplat && VT == MVT::v8i32) {
// Without AVX2, see if we can cast the values to v8f32 and use fcmp.
@@ -60825,8 +60416,8 @@ static SDValue combineConcatVectorOps(const SDLoc &DL, MVT VT,
case ISD::CTPOP:
case ISD::CTTZ:
case ISD::CTLZ:
- case ISD::CTTZ_ZERO_POISON:
- case ISD::CTLZ_ZERO_POISON:
+ case ISD::CTTZ_ZERO_UNDEF:
+ case ISD::CTLZ_ZERO_UNDEF:
if (!IsSplat && ((VT.is256BitVector() && Subtarget.hasInt256()) ||
(VT.is512BitVector() && Subtarget.useBWIRegs()))) {
return DAG.getNode(Opcode, DL, VT, ConcatSubOperand(VT, Ops, 0));
@@ -61328,7 +60919,6 @@ static SDValue combineINSERT_SUBVECTOR(SDNode *N, SelectionDAG &DAG,
MVT SubVecVT = SubVec.getSimpleValueType();
int VecNumElts = OpVT.getVectorNumElements();
int SubVecNumElts = SubVecVT.getVectorNumElements();
- const TargetLowering &TLI = DAG.getTargetLoweringInfo();
if (Vec.isUndef() && SubVec.isUndef())
return DAG.getUNDEF(OpVT);
@@ -61366,19 +60956,6 @@ static SDValue combineINSERT_SUBVECTOR(SDNode *N, SelectionDAG &DAG,
getZeroVector(OpVT, Subtarget, DAG, dl),
Ins.getOperand(1), N->getOperand(2));
}
-
- // See if were inserting into a zero vXi1 vector and the subvector was
- // bitcast from a gpr that could be zero-extended directly.
- if (IsI1Vector && TLI.isTypeLegal(OpVT) && SubVec.hasOneUse()) {
- SDValue SubInt = peekThroughOneUseBitcasts(SubVec);
- EVT IntVT = EVT::getIntegerVT(*DAG.getContext(), VecNumElts);
- if (TLI.isTypeLegal(IntVT) && SubInt.getValueType().isScalarInteger()) {
- SubInt = DAG.getNode(ISD::ZERO_EXTEND, dl, IntVT, SubInt);
- SubInt = DAG.getNode(ISD::SHL, dl, IntVT, SubInt,
- DAG.getShiftAmountConstant(IdxVal, IntVT, dl));
- return DAG.getBitcast(OpVT, SubInt);
- }
- }
}
// Stop here if this is an i1 vector.
@@ -61510,16 +61087,9 @@ static SDValue combineINSERT_SUBVECTOR(SDNode *N, SelectionDAG &DAG,
}
}
- auto peekThroughBitcastsAndExtracts = [](SDValue V) {
- while (V.getOpcode() == ISD::BITCAST ||
- V.getOpcode() == ISD::EXTRACT_SUBVECTOR)
- V = V.getOperand(0);
- return V;
- };
-
// Attempt to recursively combine to a shuffle.
if (isTargetShuffle(peekThroughBitcasts(Vec).getOpcode()) &&
- isTargetShuffle(peekThroughBitcastsAndExtracts(SubVec).getOpcode())) {
+ isTargetShuffle(peekThroughBitcasts(SubVec).getOpcode())) {
SDValue Op(N, 0);
if (SDValue Res = combineX86ShufflesRecursively(Op, DAG, Subtarget))
return Res;
@@ -61653,6 +61223,14 @@ static SDValue combineEXTRACT_SUBVECTOR(SDNode *N, SelectionDAG &DAG,
if (InVec.getOpcode() == ISD::BUILD_VECTOR)
return DAG.getBuildVector(VT, DL, InVec->ops().slice(IdxVal, NumSubElts));
+ // EXTRACT_SUBVECTOR(EXTRACT_SUBVECTOR(V,C1)),C2) - EXTRACT_SUBVECTOR(V,C1+C2)
+ if (IdxVal != 0 && InVec.getOpcode() == ISD::EXTRACT_SUBVECTOR &&
+ InVec.hasOneUse() && TLI.isTypeLegal(VT) &&
+ TLI.isTypeLegal(InVec.getOperand(0).getValueType())) {
+ unsigned NewIdx = IdxVal + InVec.getConstantOperandVal(1);
+ return extractSubVector(InVec.getOperand(0), NewIdx, DAG, DL, SizeInBits);
+ }
+
// EXTRACT_SUBVECTOR(INSERT_SUBVECTOR(SRC,SUB,C1),C2)
// --> INSERT_SUBVECTOR(EXTRACT_SUBVECTOR(SRC,C2),SUB,C1-C2)
// iff SUB is entirely contained in the extraction.
@@ -62237,62 +61815,28 @@ static SDValue combineEXTEND_VECTOR_INREG(SDNode *N, SelectionDAG &DAG,
static SDValue combineKSHIFT(SDNode *N, SelectionDAG &DAG,
TargetLowering::DAGCombinerInfo &DCI) {
EVT VT = N->getValueType(0);
- SDValue Src = N->getOperand(0);
- uint64_t Amt = N->getConstantOperandVal(1);
- unsigned Opcode = N->getOpcode();
const TargetLowering &TLI = DAG.getTargetLoweringInfo();
- SDLoc DL(N);
-
- if (ISD::isBuildVectorAllZeros(Src.getNode()))
- return DAG.getConstant(0, DL, VT);
+ if (ISD::isBuildVectorAllZeros(N->getOperand(0).getNode()))
+ return DAG.getConstant(0, SDLoc(N), VT);
- // Constant Fold.
- if (auto *SrcC = dyn_cast<ConstantSDNode>(peekThroughBitcasts(Src))) {
- APInt NewCst = Opcode == X86ISD::KSHIFTR ? SrcC->getAPIntValue().lshr(Amt)
- : SrcC->getAPIntValue().shl(Amt);
- return DAG.getBitcast(VT,
- DAG.getConstant(NewCst, DL, SrcC->getValueType(0)));
- }
-
- if (Opcode == X86ISD::KSHIFTR) {
- // Fold kshiftr(extract_subvector(X,C1),C2)
- // --> extract_subvector(kshiftr(X,C1+C2),0)
- // Fold kshiftr(kshiftr(X,C1),C2) --> kshiftr(X,C1+C2)
- if (Src.getOpcode() == ISD::EXTRACT_SUBVECTOR ||
- Src.getOpcode() == X86ISD::KSHIFTR) {
- SDValue Inner = Src.getOperand(0);
- EVT InnerVT = Inner.getValueType();
- uint64_t NewAmt = Amt + Src.getConstantOperandVal(1);
- if (TLI.isTypeLegal(InnerVT) && NewAmt < InnerVT.getVectorNumElements()) {
- SDValue Shift = DAG.getNode(X86ISD::KSHIFTR, DL, InnerVT, Inner,
- DAG.getTargetConstant(NewAmt, DL, MVT::i8));
+ // Fold kshiftr(extract_subvector(X,C1),C2)
+ // --> extract_subvector(kshiftr(X,C1+C2),0)
+ // Fold kshiftr(kshiftr(X,C1),C2) --> kshiftr(X,C1+C2)
+ if (N->getOpcode() == X86ISD::KSHIFTR) {
+ SDLoc DL(N);
+ if (N->getOperand(0).getOpcode() == ISD::EXTRACT_SUBVECTOR ||
+ N->getOperand(0).getOpcode() == X86ISD::KSHIFTR) {
+ SDValue Src = N->getOperand(0).getOperand(0);
+ uint64_t Amt = N->getConstantOperandVal(1) +
+ N->getOperand(0).getConstantOperandVal(1);
+ EVT SrcVT = Src.getValueType();
+ if (TLI.isTypeLegal(SrcVT) && Amt < SrcVT.getVectorNumElements()) {
+ SDValue Shift = DAG.getNode(X86ISD::KSHIFTR, DL, SrcVT, Src,
+ DAG.getTargetConstant(Amt, DL, MVT::i8));
return DAG.getNode(ISD::EXTRACT_SUBVECTOR, DL, VT, Shift,
DAG.getVectorIdxConstant(0, DL));
}
}
- // Fold kshiftr(concat_vectors(X,Y,Z,W),C)
- // --> concat_vectors(Z,W,0,0) iff amount is whole subvector shift.
- if (Src.getOpcode() == ISD::CONCAT_VECTORS &&
- (Amt % Src.getOperand(0).getValueType().getVectorNumElements()) == 0) {
- unsigned NumSubs = Src.getNumOperands();
- EVT SubVT = Src.getOperand(0).getValueType();
- unsigned Ofs = Amt / SubVT.getVectorNumElements();
- SmallVector<SDValue, 4> SubOps(NumSubs, DAG.getConstant(0, DL, SubVT));
- for (unsigned I = Ofs; I != NumSubs; ++I)
- SubOps[I - Ofs] = Src.getOperand(I);
- return DAG.getNode(ISD::CONCAT_VECTORS, DL, VT, SubOps);
- }
- }
-
- // Fold kshift(logicop(X,C1),C2)
- // --> logicop(kshift(X,C2),kshift(C1,C2))
- if (ISD::isBitwiseLogicOp(Src.getOpcode()) &&
- isa<ConstantSDNode>(peekThroughBitcasts(Src.getOperand(1)))) {
- SDValue LHS =
- DAG.getNode(Opcode, DL, VT, Src.getOperand(0), N->getOperand(1));
- SDValue RHS =
- DAG.getNode(Opcode, DL, VT, Src.getOperand(1), N->getOperand(1));
- return DAG.getNode(Src.getOpcode(), DL, VT, LHS, RHS);
}
APInt DemandedElts = APInt::getAllOnes(VT.getVectorNumElements());
@@ -62302,175 +61846,6 @@ static SDValue combineKSHIFT(SDNode *N, SelectionDAG &DAG,
return SDValue();
}
-// Reassociates AND by splat to other operand when profitable.
-// Equivalent as removing the same bit within each matrix's row acts like the
-// corresponding source bit is zero.
-static SDValue combineAndOnGF2P8AFFINEQBOperand(SDNode *N, const SDLoc &DL,
- SelectionDAG &DAG, EVT VT) {
- using namespace SDPatternMatch;
-
- SDValue X, Y, AndOp, SplatOp;
- APInt Imm, SplatVal, ConstUndef;
- SmallVector<APInt> ConstEltBits;
-
- // TODO: Add reverse fold when X is constant
- // Fold GF2P8AFFINEQB(x & Splat(C), M, Imm)
- // --> GF2P8AFFINEQB(x, M & Splat(C), Imm)
- if (sd_match(N, m_TernaryOp(X86ISD::GF2P8AFFINEQB, m_Value(AndOp), m_Value(Y),
- m_ConstInt(Imm))) &&
- sd_match(AndOp, m_And(m_Value(X), m_Value(SplatOp))) &&
- getTargetConstantBitsFromNode(Y, Y.getScalarValueSizeInBits(), ConstUndef,
- ConstEltBits, /*AllowWholeUndefs=*/false)) {
- bool SplatIsConst =
- X86::isConstantSplat(SplatOp, SplatVal, /*AllowPartialUndefs=*/false);
-
- // Can still shorten the chain when constant folded with the matrix
- if (!AndOp->hasOneUse() && !SplatIsConst)
- return SDValue();
-
- if (!(SplatIsConst || DAG.isSplatValue(SplatOp, /*AllowUndefs=*/false)) ||
- SplatOp.getScalarValueSizeInBits() != 8)
- return SDValue();
-
- // ANDs with constants are not folded away this far into lowering
- SDValue NewMatrix;
- if (!SplatIsConst) {
- NewMatrix = DAG.getNode(ISD::AND, DL, VT, SplatOp, Y);
- } else {
- SmallVector<SDValue> FoldedAnd;
- for (APInt &Elt : ConstEltBits)
- FoldedAnd.push_back(DAG.getConstant(SplatVal & Elt, DL, MVT::i8));
-
- NewMatrix = DAG.getBuildVector(VT, DL, FoldedAnd);
- }
-
- return DAG.getNode(X86ISD::GF2P8AFFINEQB, DL, VT, X, NewMatrix,
- DAG.getTargetConstant(Imm, DL, MVT::i8));
- }
-
- return SDValue();
-}
-
-// Fold: GF2P8AFFINEQB(GF2P8AFFINEQB(X, YSub), YSup)
-// => GF2P8AFFINEQB(X, YFolded)
-// Permuting the sub-matrix by the super-matrix at a byte, rather than bit,
-// granularity produces a matrix that performs both permutations at once.
-static SDValue combineNestedGF2P8AFFINEQB(SDNode *N, const SDLoc &DL,
- SelectionDAG &DAG, EVT VT) {
- using namespace SDPatternMatch;
-
- unsigned VecWidth = VT.getSizeInBits();
- unsigned NumElts = VT.getVectorNumElements();
- unsigned EltWidth = VT.getScalarSizeInBits();
-
- SDValue X, YSub, YSup;
- APInt ImmSub, ImmSup, ConstUndef;
- SmallVector<APInt> YSubEltBits, YSupEltBits;
-
- if (!(sd_match(N, m_TernaryOp(X86ISD::GF2P8AFFINEQB,
- m_TernaryOp(X86ISD::GF2P8AFFINEQB, m_Value(X),
- m_Value(YSub), m_ConstInt(ImmSub)),
- m_Value(YSup), m_ConstInt(ImmSup))) &&
- getTargetConstantBitsFromNode(YSub, EltWidth, ConstUndef, YSubEltBits,
- /*AllowWholeUndefs=*/false) &&
- getTargetConstantBitsFromNode(YSup, EltWidth, ConstUndef, YSupEltBits,
- /*AllowWholeUndefs=*/false)))
- return SDValue();
-
- APInt SubM(VecWidth, 0);
- APInt SupM(VecWidth, 0);
- for (unsigned I = 0; I != NumElts; ++I) {
- SubM.insertBits(YSubEltBits[I], I * EltWidth);
- SupM.insertBits(YSupEltBits[I], I * EltWidth);
- }
-
- // Immediate permute
- APInt FoldedImm;
- if (SupM.isSplat(64)) {
- FoldedImm = getGFNIByteAffine(ImmSub, SupM.trunc(64), ImmSup);
- } else {
- // Immediate is shared and needs to be permuted in the same manner
- if (ImmSub != 0)
- return SDValue();
- FoldedImm = ImmSup;
- }
-
- // Matrix permute
- APInt FoldedMatrix = APInt(VecWidth, 0);
- APInt LeastRowMask = APInt::getSplat(VecWidth, APInt(64, 0xFF));
- APInt LeastBitInByte = APInt::getSplat(VecWidth, APInt(8, 0x01));
- APInt RowSplatter = APInt(VecWidth, 0x0101010101010101ull);
-
- for (unsigned Row = 0; Row != 8; ++Row) {
- APInt RowSplat = (SubM & LeastRowMask) * RowSplatter;
- SubM = SubM.lshr(EltWidth);
-
- APInt ByteMaskIfSet = (SupM.lshr(7 - Row)) & LeastBitInByte;
- ByteMaskIfSet *= 0xFF;
-
- FoldedMatrix ^= RowSplat & ByteMaskIfSet;
- }
-
- SmallVector<SDValue> FoldedVector;
- for (unsigned I = 0; I < NumElts; ++I) {
- APInt FoldedElt = FoldedMatrix.extractBits(EltWidth, I * EltWidth);
- FoldedVector.push_back(DAG.getConstant(FoldedElt, DL, MVT::i8));
- }
- SDValue NewMatrix = DAG.getBuildVector(VT, DL, FoldedVector);
-
- return DAG.getNode(X86ISD::GF2P8AFFINEQB, DL, VT, X, NewMatrix,
- DAG.getTargetConstant(FoldedImm, DL, MVT::i8));
-}
-
-// Fold: GF2P8AFFINEQB(X ^ Splat(C), Y, Imm)
-// => GF2P8AFFINEQB(X, Y, (u8)GF2P8AFFINEQB(C, Y, Imm))
-// Reassociating a XOR (by permuting it in the same manner as the input) allows
-// it to be applied for free using the immediate.
-static SDValue combineXorOnGF2P8AFFINEQBOperand(SDNode *N, const SDLoc &DL,
- SelectionDAG &DAG, EVT VT) {
- using namespace SDPatternMatch;
-
- unsigned MatrixWidth = 64;
-
- SDValue X, Y, SplatOp;
- APInt Imm, SplatVal, ConstUndef;
- SmallVector<APInt> MatEltBits;
-
- if (sd_match(N, m_TernaryOp(X86ISD::GF2P8AFFINEQB,
- m_Xor(m_Value(X), m_Value(SplatOp)), m_Value(Y),
- m_ConstInt(Imm))) &&
- X86::isConstantSplat(SplatOp, SplatVal, /*AllowPartialUndefs=*/false) &&
- getTargetConstantBitsFromNode(Y, MatrixWidth, ConstUndef, MatEltBits,
- /*AllowWholeUndefs=*/false)) {
- // Immediate is shared, so all matrices need to permute it in the same way
- if (!llvm::all_equal(MatEltBits))
- return SDValue();
-
- APInt NewImm = getGFNIByteAffine(SplatVal, MatEltBits[0], Imm);
-
- return DAG.getNode(X86ISD::GF2P8AFFINEQB, DL, VT, X, Y,
- DAG.getTargetConstant(NewImm, DL, MVT::i8));
- }
-
- return SDValue();
-}
-
-static SDValue combineGF2P8AFFINEQB(SDNode *N, SelectionDAG &DAG) {
- EVT VT = N->getValueType(0);
- SDLoc dl(N);
-
- if (SDValue R = combineAndOnGF2P8AFFINEQBOperand(N, dl, DAG, VT))
- return R;
-
- if (SDValue R = combineXorOnGF2P8AFFINEQBOperand(N, dl, DAG, VT))
- return R;
-
- if (SDValue R = combineNestedGF2P8AFFINEQB(N, dl, DAG, VT))
- return R;
-
- return SDValue();
-}
-
// Optimize (fp16_to_fp (fp_to_fp16 X)) to VCVTPS2PH followed by VCVTPH2PS.
// Done as a combine because the lowering for fp16_to_fp and fp_to_fp16 produce
// extra instructions between the conversion due to going to scalar and back.
@@ -62534,7 +61909,7 @@ static SDValue combineFP_EXTEND(SDNode *N, SelectionDAG &DAG,
if (Subtarget.hasFP16())
return SDValue();
- if (!SrcVT.isVectorOf(MVT::f16))
+ if (!SrcVT.isVector() || SrcVT.getVectorElementType() != MVT::f16)
return SDValue();
if (VT.getVectorElementType() != MVT::f32 &&
@@ -62648,7 +62023,8 @@ static SDValue combineFP_ROUND(SDNode *N, SelectionDAG &DAG,
EVT SrcVT = Src.getValueType();
- if (!VT.isVectorOf(MVT::f16) || SrcVT.getVectorElementType() != MVT::f32)
+ if (!VT.isVector() || VT.getVectorElementType() != MVT::f16 ||
+ SrcVT.getVectorElementType() != MVT::f32)
return SDValue();
SDValue Cvt, Chain;
@@ -62927,7 +62303,6 @@ SDValue X86TargetLowering::PerformDAGCombine(SDNode *N,
case ISD::AVGCEILU:
case ISD::AVGFLOORS:
case ISD::AVGFLOORU: return combineAVG(N, DAG, DCI, Subtarget);
- case ISD::ATOMIC_LOAD: return combineAtomicLoad(N, DAG, DCI);
case ISD::LOAD: return combineLoad(N, DAG, DCI, Subtarget);
case ISD::MLOAD: return combineMaskedLoad(N, DAG, DCI, Subtarget);
case ISD::STORE: return combineStore(N, DAG, DCI, Subtarget);
@@ -62948,11 +62323,10 @@ SDValue X86TargetLowering::PerformDAGCombine(SDNode *N,
case X86ISD::VFCMULC:
case X86ISD::VFMULC: return combineFMulcFCMulc(N, DAG, Subtarget);
case ISD::FNEG: return combineFneg(N, DAG, DCI, Subtarget);
- case ISD::VECREDUCE_MUL: return combineVECREDUCE_MUL(N, DAG, Subtarget);
case ISD::TRUNCATE: return combineTruncate(N, DAG, Subtarget);
case X86ISD::VTRUNC: return combineVTRUNC(N, DAG, DCI);
case X86ISD::VTRUNCS:
- case X86ISD::VTRUNCUS: return combineVTRUNCSAT(N, DAG, DCI, Subtarget);
+ case X86ISD::VTRUNCUS: return combineVTRUNCSAT(N, DAG, DCI);
case X86ISD::ANDNP: return combineAndnp(N, DAG, DCI, Subtarget);
case X86ISD::FAND: return combineFAnd(N, DAG, Subtarget);
case X86ISD::FANDN: return combineFAndn(N, DAG, Subtarget);
@@ -63071,7 +62445,6 @@ SDValue X86TargetLowering::PerformDAGCombine(SDNode *N,
case X86ISD::VPMADD52H: return combineVPMADD52LH(N, DAG, DCI);
case X86ISD::KSHIFTL:
case X86ISD::KSHIFTR: return combineKSHIFT(N, DAG, DCI);
- case X86ISD::GF2P8AFFINEQB: return combineGF2P8AFFINEQB(N, DAG);
case ISD::FP16_TO_FP: return combineFP16_TO_FP(N, DAG, Subtarget);
case ISD::STRICT_FP_EXTEND:
case ISD::FP_EXTEND: return combineFP_EXTEND(N, DAG, DCI, Subtarget);
@@ -63082,7 +62455,8 @@ SDValue X86TargetLowering::PerformDAGCombine(SDNode *N,
case X86ISD::MOVDQ2Q: return combineMOVDQ2Q(N, DAG);
case X86ISD::BEXTR:
case X86ISD::BEXTRI:
- case X86ISD::BZHI: return combineBMI(N, DAG, DCI);
+ case X86ISD::BZHI:
+ case X86ISD::PDEP: return combineBMI(N, DAG, DCI);
case X86ISD::PCLMULQDQ: return combinePCLMULQDQ(N, DAG, DCI);
case ISD::INTRINSIC_WO_CHAIN: return combineINTRINSIC_WO_CHAIN(N, DAG, DCI);
case ISD::INTRINSIC_W_CHAIN: return combineINTRINSIC_W_CHAIN(N, DAG, DCI);
@@ -63112,7 +62486,7 @@ bool X86TargetLowering::isTypeDesirableForOp(unsigned Opc, EVT VT) const {
return false;
// There are no vXi8 shifts.
- if (Opc == ISD::SHL && VT.isVectorOf(MVT::i8))
+ if (Opc == ISD::SHL && VT.isVector() && VT.getVectorElementType() == MVT::i8)
return false;
// TODO: Almost no 8-bit ops are desirable because they have no actual
@@ -64453,7 +63827,7 @@ bool X86TargetLowering::hasStackProbeSymbol(const MachineFunction &MF) const {
bool X86TargetLowering::hasInlineStackProbe(const MachineFunction &MF) const {
// No inline stack probe for Windows, they have their own mechanism.
- if (Subtarget.isOSWindowsOrUEFI() ||
+ if (Subtarget.isOSWindows() || Subtarget.isUEFI() ||
MF.getFunction().hasFnAttribute("no-stack-arg-probe"))
return false;
@@ -64479,7 +63853,8 @@ X86TargetLowering::getStackProbeSymbolName(const MachineFunction &MF) const {
// Generally, if we aren't on Windows, the platform ABI does not include
// support for stack probes, so don't emit them.
- if (!Subtarget.isOSWindowsOrUEFI() || Subtarget.isTargetMachO() ||
+ if ((!Subtarget.isOSWindows() && !Subtarget.isUEFI()) ||
+ Subtarget.isTargetMachO() ||
MF.getFunction().hasFnAttribute("no-stack-arg-probe"))
return "";
diff --git a/llvm/test/CodeGen/X86/apx/sub.ll b/llvm/test/CodeGen/X86/apx/sub.ll
index 34af966465d93..bfc134ed1ed15 100644
--- a/llvm/test/CodeGen/X86/apx/sub.ll
+++ b/llvm/test/CodeGen/X86/apx/sub.ll
@@ -1,14 +1,19 @@
; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
; RUN: llc < %s -mtriple=x86_64-unknown -mattr=+ndd -verify-machineinstrs --show-mc-encoding | FileCheck %s --check-prefixes=CHECK,NDD
; RUN: llc < %s -mtriple=x86_64-unknown -mattr=+ndd,+prefer-ndd-mem -verify-machineinstrs --show-mc-encoding | FileCheck %s --check-prefixes=CHECK,MEM
-; RUN: llc < %s -mtriple=x86_64-unknown -mattr=+ndd,nf -verify-machineinstrs --show-mc-encoding | FileCheck --check-prefix=NF %s
-; RUN: llc < %s -mtriple=x86_64-unknown -mattr=+ndd,nf -x86-enable-apx-for-relocation=true -verify-machineinstrs --show-mc-encoding | FileCheck --check-prefix=NF %s
+; RUN: llc < %s -mtriple=x86_64-unknown -mattr=+ndd,nf -verify-machineinstrs --show-mc-encoding | FileCheck --check-prefixes=CHECK,NF %s
+; RUN: llc < %s -mtriple=x86_64-unknown -mattr=+ndd,nf -x86-enable-apx-for-relocation=true -verify-machineinstrs --show-mc-encoding | FileCheck --check-prefixes=CHECK,NF %s
define i8 @sub8rr(i8 noundef %a, i8 noundef %b) {
-; CHECK-LABEL: sub8rr:
-; CHECK: # %bb.0: # %entry
-; CHECK-NEXT: subb %sil, %dil, %al # encoding: [0x62,0xf4,0x7c,0x18,0x28,0xf7]
-; CHECK-NEXT: retq # encoding: [0xc3]
+; NDD-LABEL: sub8rr:
+; NDD: # %bb.0: # %entry
+; NDD-NEXT: subb %sil, %dil, %al # encoding: [0x62,0xf4,0x7c,0x18,0x28,0xf7]
+; NDD-NEXT: retq # encoding: [0xc3]
+;
+; MEM-LABEL: sub8rr:
+; MEM: # %bb.0: # %entry
+; MEM-NEXT: subb %sil, %dil, %al # encoding: [0x62,0xf4,0x7c,0x18,0x28,0xf7]
+; MEM-NEXT: retq # encoding: [0xc3]
;
; NF-LABEL: sub8rr:
; NF: # %bb.0: # %entry
@@ -20,10 +25,15 @@ entry:
}
define i16 @sub16rr(i16 noundef %a, i16 noundef %b) {
-; CHECK-LABEL: sub16rr:
-; CHECK: # %bb.0: # %entry
-; CHECK-NEXT: subw %si, %di, %ax # encoding: [0x62,0xf4,0x7d,0x18,0x29,0xf7]
-; CHECK-NEXT: retq # encoding: [0xc3]
+; NDD-LABEL: sub16rr:
+; NDD: # %bb.0: # %entry
+; NDD-NEXT: subw %si, %di, %ax # encoding: [0x62,0xf4,0x7d,0x18,0x29,0xf7]
+; NDD-NEXT: retq # encoding: [0xc3]
+;
+; MEM-LABEL: sub16rr:
+; MEM: # %bb.0: # %entry
+; MEM-NEXT: subw %si, %di, %ax # encoding: [0x62,0xf4,0x7d,0x18,0x29,0xf7]
+; MEM-NEXT: retq # encoding: [0xc3]
;
; NF-LABEL: sub16rr:
; NF: # %bb.0: # %entry
@@ -35,10 +45,15 @@ entry:
}
define i32 @sub32rr(i32 noundef %a, i32 noundef %b) {
-; CHECK-LABEL: sub32rr:
-; CHECK: # %bb.0: # %entry
-; CHECK-NEXT: subl %esi, %edi, %eax # encoding: [0x62,0xf4,0x7c,0x18,0x29,0xf7]
-; CHECK-NEXT: retq # encoding: [0xc3]
+; NDD-LABEL: sub32rr:
+; NDD: # %bb.0: # %entry
+; NDD-NEXT: subl %esi, %edi, %eax # encoding: [0x62,0xf4,0x7c,0x18,0x29,0xf7]
+; NDD-NEXT: retq # encoding: [0xc3]
+;
+; MEM-LABEL: sub32rr:
+; MEM: # %bb.0: # %entry
+; MEM-NEXT: subl %esi, %edi, %eax # encoding: [0x62,0xf4,0x7c,0x18,0x29,0xf7]
+; MEM-NEXT: retq # encoding: [0xc3]
;
; NF-LABEL: sub32rr:
; NF: # %bb.0: # %entry
@@ -50,10 +65,15 @@ entry:
}
define i64 @sub64rr(i64 noundef %a, i64 noundef %b) {
-; CHECK-LABEL: sub64rr:
-; CHECK: # %bb.0: # %entry
-; CHECK-NEXT: subq %rsi, %rdi, %rax # encoding: [0x62,0xf4,0xfc,0x18,0x29,0xf7]
-; CHECK-NEXT: retq # encoding: [0xc3]
+; NDD-LABEL: sub64rr:
+; NDD: # %bb.0: # %entry
+; NDD-NEXT: subq %rsi, %rdi, %rax # encoding: [0x62,0xf4,0xfc,0x18,0x29,0xf7]
+; NDD-NEXT: retq # encoding: [0xc3]
+;
+; MEM-LABEL: sub64rr:
+; MEM: # %bb.0: # %entry
+; MEM-NEXT: subq %rsi, %rdi, %rax # encoding: [0x62,0xf4,0xfc,0x18,0x29,0xf7]
+; MEM-NEXT: retq # encoding: [0xc3]
;
; NF-LABEL: sub64rr:
; NF: # %bb.0: # %entry
@@ -161,10 +181,15 @@ entry:
}
define i16 @sub16ri8(i16 noundef %a) {
-; CHECK-LABEL: sub16ri8:
-; CHECK: # %bb.0: # %entry
-; CHECK-NEXT: subw $-128, %di, %ax # encoding: [0x62,0xf4,0x7d,0x18,0x83,0xef,0x80]
-; CHECK-NEXT: retq # encoding: [0xc3]
+; NDD-LABEL: sub16ri8:
+; NDD: # %bb.0: # %entry
+; NDD-NEXT: subw $-128, %di, %ax # encoding: [0x62,0xf4,0x7d,0x18,0x83,0xef,0x80]
+; NDD-NEXT: retq # encoding: [0xc3]
+;
+; MEM-LABEL: sub16ri8:
+; MEM: # %bb.0: # %entry
+; MEM-NEXT: subw $-128, %di, %ax # encoding: [0x62,0xf4,0x7d,0x18,0x83,0xef,0x80]
+; MEM-NEXT: retq # encoding: [0xc3]
;
; NF-LABEL: sub16ri8:
; NF: # %bb.0: # %entry
@@ -176,10 +201,15 @@ entry:
}
define i32 @sub32ri8(i32 noundef %a) {
-; CHECK-LABEL: sub32ri8:
-; CHECK: # %bb.0: # %entry
-; CHECK-NEXT: subl $-128, %edi, %eax # encoding: [0x62,0xf4,0x7c,0x18,0x83,0xef,0x80]
-; CHECK-NEXT: retq # encoding: [0xc3]
+; NDD-LABEL: sub32ri8:
+; NDD: # %bb.0: # %entry
+; NDD-NEXT: subl $-128, %edi, %eax # encoding: [0x62,0xf4,0x7c,0x18,0x83,0xef,0x80]
+; NDD-NEXT: retq # encoding: [0xc3]
+;
+; MEM-LABEL: sub32ri8:
+; MEM: # %bb.0: # %entry
+; MEM-NEXT: subl $-128, %edi, %eax # encoding: [0x62,0xf4,0x7c,0x18,0x83,0xef,0x80]
+; MEM-NEXT: retq # encoding: [0xc3]
;
; NF-LABEL: sub32ri8:
; NF: # %bb.0: # %entry
@@ -191,10 +221,15 @@ entry:
}
define i64 @sub64ri8(i64 noundef %a) {
-; CHECK-LABEL: sub64ri8:
-; CHECK: # %bb.0: # %entry
-; CHECK-NEXT: subq $-128, %rdi, %rax # encoding: [0x62,0xf4,0xfc,0x18,0x83,0xef,0x80]
-; CHECK-NEXT: retq # encoding: [0xc3]
+; NDD-LABEL: sub64ri8:
+; NDD: # %bb.0: # %entry
+; NDD-NEXT: subq $-128, %rdi, %rax # encoding: [0x62,0xf4,0xfc,0x18,0x83,0xef,0x80]
+; NDD-NEXT: retq # encoding: [0xc3]
+;
+; MEM-LABEL: sub64ri8:
+; MEM: # %bb.0: # %entry
+; MEM-NEXT: subq $-128, %rdi, %rax # encoding: [0x62,0xf4,0xfc,0x18,0x83,0xef,0x80]
+; MEM-NEXT: retq # encoding: [0xc3]
;
; NF-LABEL: sub64ri8:
; NF: # %bb.0: # %entry
@@ -206,10 +241,15 @@ entry:
}
define i8 @sub8ri(i8 noundef %a) {
-; CHECK-LABEL: sub8ri:
-; CHECK: # %bb.0: # %entry
-; CHECK-NEXT: addb $-123, %dil, %al # encoding: [0x62,0xf4,0x7c,0x18,0x80,0xc7,0x85]
-; CHECK-NEXT: retq # encoding: [0xc3]
+; NDD-LABEL: sub8ri:
+; NDD: # %bb.0: # %entry
+; NDD-NEXT: addb $-123, %dil, %al # encoding: [0x62,0xf4,0x7c,0x18,0x80,0xc7,0x85]
+; NDD-NEXT: retq # encoding: [0xc3]
+;
+; MEM-LABEL: sub8ri:
+; MEM: # %bb.0: # %entry
+; MEM-NEXT: addb $-123, %dil, %al # encoding: [0x62,0xf4,0x7c,0x18,0x80,0xc7,0x85]
+; MEM-NEXT: retq # encoding: [0xc3]
;
; NF-LABEL: sub8ri:
; NF: # %bb.0: # %entry
@@ -221,11 +261,17 @@ entry:
}
define i16 @sub16ri(i16 noundef %a) {
-; CHECK-LABEL: sub16ri:
-; CHECK: # %bb.0: # %entry
-; CHECK-NEXT: addw $-1234, %di, %ax # encoding: [0x62,0xf4,0x7d,0x18,0x81,0xc7,0x2e,0xfb]
-; CHECK-NEXT: # imm = 0xFB2E
-; CHECK-NEXT: retq # encoding: [0xc3]
+; NDD-LABEL: sub16ri:
+; NDD: # %bb.0: # %entry
+; NDD-NEXT: addw $-1234, %di, %ax # encoding: [0x62,0xf4,0x7d,0x18,0x81,0xc7,0x2e,0xfb]
+; NDD-NEXT: # imm = 0xFB2E
+; NDD-NEXT: retq # encoding: [0xc3]
+;
+; MEM-LABEL: sub16ri:
+; MEM: # %bb.0: # %entry
+; MEM-NEXT: addw $-1234, %di, %ax # encoding: [0x62,0xf4,0x7d,0x18,0x81,0xc7,0x2e,0xfb]
+; MEM-NEXT: # imm = 0xFB2E
+; MEM-NEXT: retq # encoding: [0xc3]
;
; NF-LABEL: sub16ri:
; NF: # %bb.0: # %entry
@@ -242,22 +288,23 @@ define i32 @sub32ri(i32 noundef %a) {
; CHECK: # %bb.0: # %entry
; CHECK-NEXT: leal -123456(%rdi), %eax # encoding: [0x8d,0x87,0xc0,0x1d,0xfe,0xff]
; CHECK-NEXT: retq # encoding: [0xc3]
-;
-; NF-LABEL: sub32ri:
-; NF: # %bb.0: # %entry
-; NF-NEXT: leal -123456(%rdi), %eax # encoding: [0x8d,0x87,0xc0,0x1d,0xfe,0xff]
-; NF-NEXT: retq # encoding: [0xc3]
entry:
%sub = sub i32 %a, 123456
ret i32 %sub
}
define i64 @sub64ri(i64 noundef %a) {
-; CHECK-LABEL: sub64ri:
-; CHECK: # %bb.0: # %entry
-; CHECK-NEXT: subq $-2147483648, %rdi, %rax # encoding: [0x62,0xf4,0xfc,0x18,0x81,0xef,0x00,0x00,0x00,0x80]
-; CHECK-NEXT: # imm = 0x80000000
-; CHECK-NEXT: retq # encoding: [0xc3]
+; NDD-LABEL: sub64ri:
+; NDD: # %bb.0: # %entry
+; NDD-NEXT: subq $-2147483648, %rdi, %rax # encoding: [0x62,0xf4,0xfc,0x18,0x81,0xef,0x00,0x00,0x00,0x80]
+; NDD-NEXT: # imm = 0x80000000
+; NDD-NEXT: retq # encoding: [0xc3]
+;
+; MEM-LABEL: sub64ri:
+; MEM: # %bb.0: # %entry
+; MEM-NEXT: subq $-2147483648, %rdi, %rax # encoding: [0x62,0xf4,0xfc,0x18,0x81,0xef,0x00,0x00,0x00,0x80]
+; MEM-NEXT: # imm = 0x80000000
+; MEM-NEXT: retq # encoding: [0xc3]
;
; NF-LABEL: sub64ri:
; NF: # %bb.0: # %entry
@@ -545,15 +592,6 @@ define i8 @subflag8rr(i8 noundef %a, i8 noundef %b) {
; CHECK-NEXT: cmovael %ecx, %eax # EVEX TO LEGACY Compression encoding: [0x0f,0x43,0xc1]
; CHECK-NEXT: # kill: def $al killed $al killed $eax
; CHECK-NEXT: retq # encoding: [0xc3]
-;
-; NF-LABEL: subflag8rr:
-; NF: # %bb.0: # %entry
-; NF-NEXT: xorl %eax, %eax # encoding: [0x31,0xc0]
-; NF-NEXT: subb %sil, %dil, %cl # encoding: [0x62,0xf4,0x74,0x18,0x28,0xf7]
-; NF-NEXT: movzbl %cl, %ecx # encoding: [0x0f,0xb6,0xc9]
-; NF-NEXT: cmovael %ecx, %eax # EVEX TO LEGACY Compression encoding: [0x0f,0x43,0xc1]
-; NF-NEXT: # kill: def $al killed $al killed $eax
-; NF-NEXT: retq # encoding: [0xc3]
entry:
%sub = call i8 @llvm.usub.sat.i8(i8 %a, i8 %b)
ret i8 %sub
@@ -567,14 +605,6 @@ define i16 @subflag16rr(i16 noundef %a, i16 noundef %b) {
; CHECK-NEXT: cmovael %edi, %eax # EVEX TO LEGACY Compression encoding: [0x0f,0x43,0xc7]
; CHECK-NEXT: # kill: def $ax killed $ax killed $eax
; CHECK-NEXT: retq # encoding: [0xc3]
-;
-; NF-LABEL: subflag16rr:
-; NF: # %bb.0: # %entry
-; NF-NEXT: xorl %eax, %eax # encoding: [0x31,0xc0]
-; NF-NEXT: subw %si, %di # EVEX TO LEGACY Compression encoding: [0x66,0x29,0xf7]
-; NF-NEXT: cmovael %edi, %eax # EVEX TO LEGACY Compression encoding: [0x0f,0x43,0xc7]
-; NF-NEXT: # kill: def $ax killed $ax killed $eax
-; NF-NEXT: retq # encoding: [0xc3]
entry:
%sub = call i16 @llvm.usub.sat.i16(i16 %a, i16 %b)
ret i16 %sub
@@ -587,13 +617,6 @@ define i32 @subflag32rr(i32 noundef %a, i32 noundef %b) {
; CHECK-NEXT: subl %esi, %edi # EVEX TO LEGACY Compression encoding: [0x29,0xf7]
; CHECK-NEXT: cmovael %edi, %eax # EVEX TO LEGACY Compression encoding: [0x0f,0x43,0xc7]
; CHECK-NEXT: retq # encoding: [0xc3]
-;
-; NF-LABEL: subflag32rr:
-; NF: # %bb.0: # %entry
-; NF-NEXT: xorl %eax, %eax # encoding: [0x31,0xc0]
-; NF-NEXT: subl %esi, %edi # EVEX TO LEGACY Compression encoding: [0x29,0xf7]
-; NF-NEXT: cmovael %edi, %eax # EVEX TO LEGACY Compression encoding: [0x0f,0x43,0xc7]
-; NF-NEXT: retq # encoding: [0xc3]
entry:
%sub = call i32 @llvm.usub.sat.i32(i32 %a, i32 %b)
ret i32 %sub
@@ -606,13 +629,6 @@ define i64 @subflag64rr(i64 noundef %a, i64 noundef %b) {
; CHECK-NEXT: subq %rsi, %rdi # EVEX TO LEGACY Compression encoding: [0x48,0x29,0xf7]
; CHECK-NEXT: cmovaeq %rdi, %rax # EVEX TO LEGACY Compression encoding: [0x48,0x0f,0x43,0xc7]
; CHECK-NEXT: retq # encoding: [0xc3]
-;
-; NF-LABEL: subflag64rr:
-; NF: # %bb.0: # %entry
-; NF-NEXT: xorl %eax, %eax # encoding: [0x31,0xc0]
-; NF-NEXT: subq %rsi, %rdi # EVEX TO LEGACY Compression encoding: [0x48,0x29,0xf7]
-; NF-NEXT: cmovaeq %rdi, %rax # EVEX TO LEGACY Compression encoding: [0x48,0x0f,0x43,0xc7]
-; NF-NEXT: retq # encoding: [0xc3]
entry:
%sub = call i64 @llvm.usub.sat.i64(i64 %a, i64 %b)
ret i64 %sub
@@ -743,14 +759,6 @@ define i16 @subflag16ri8(i16 noundef %a) {
; CHECK-NEXT: cmovael %edi, %eax # EVEX TO LEGACY Compression encoding: [0x0f,0x43,0xc7]
; CHECK-NEXT: # kill: def $ax killed $ax killed $eax
; CHECK-NEXT: retq # encoding: [0xc3]
-;
-; NF-LABEL: subflag16ri8:
-; NF: # %bb.0: # %entry
-; NF-NEXT: xorl %eax, %eax # encoding: [0x31,0xc0]
-; NF-NEXT: subw $123, %di # EVEX TO LEGACY Compression encoding: [0x66,0x83,0xef,0x7b]
-; NF-NEXT: cmovael %edi, %eax # EVEX TO LEGACY Compression encoding: [0x0f,0x43,0xc7]
-; NF-NEXT: # kill: def $ax killed $ax killed $eax
-; NF-NEXT: retq # encoding: [0xc3]
entry:
%sub = call i16 @llvm.usub.sat.i16(i16 %a, i16 123)
ret i16 %sub
@@ -763,13 +771,6 @@ define i32 @subflag32ri8(i32 noundef %a) {
; CHECK-NEXT: subl $123, %edi # EVEX TO LEGACY Compression encoding: [0x83,0xef,0x7b]
; CHECK-NEXT: cmovael %edi, %eax # EVEX TO LEGACY Compression encoding: [0x0f,0x43,0xc7]
; CHECK-NEXT: retq # encoding: [0xc3]
-;
-; NF-LABEL: subflag32ri8:
-; NF: # %bb.0: # %entry
-; NF-NEXT: xorl %eax, %eax # encoding: [0x31,0xc0]
-; NF-NEXT: subl $123, %edi # EVEX TO LEGACY Compression encoding: [0x83,0xef,0x7b]
-; NF-NEXT: cmovael %edi, %eax # EVEX TO LEGACY Compression encoding: [0x0f,0x43,0xc7]
-; NF-NEXT: retq # encoding: [0xc3]
entry:
%sub = call i32 @llvm.usub.sat.i32(i32 %a, i32 123)
ret i32 %sub
@@ -782,13 +783,6 @@ define i64 @subflag64ri8(i64 noundef %a) {
; CHECK-NEXT: subq $123, %rdi # EVEX TO LEGACY Compression encoding: [0x48,0x83,0xef,0x7b]
; CHECK-NEXT: cmovaeq %rdi, %rax # EVEX TO LEGACY Compression encoding: [0x48,0x0f,0x43,0xc7]
; CHECK-NEXT: retq # encoding: [0xc3]
-;
-; NF-LABEL: subflag64ri8:
-; NF: # %bb.0: # %entry
-; NF-NEXT: xorl %eax, %eax # encoding: [0x31,0xc0]
-; NF-NEXT: subq $123, %rdi # EVEX TO LEGACY Compression encoding: [0x48,0x83,0xef,0x7b]
-; NF-NEXT: cmovaeq %rdi, %rax # EVEX TO LEGACY Compression encoding: [0x48,0x0f,0x43,0xc7]
-; NF-NEXT: retq # encoding: [0xc3]
entry:
%sub = call i64 @llvm.usub.sat.i64(i64 %a, i64 123)
ret i64 %sub
@@ -803,15 +797,6 @@ define i8 @subflag8ri(i8 noundef %a) {
; CHECK-NEXT: cmovael %ecx, %eax # EVEX TO LEGACY Compression encoding: [0x0f,0x43,0xc1]
; CHECK-NEXT: # kill: def $al killed $al killed $eax
; CHECK-NEXT: retq # encoding: [0xc3]
-;
-; NF-LABEL: subflag8ri:
-; NF: # %bb.0: # %entry
-; NF-NEXT: xorl %eax, %eax # encoding: [0x31,0xc0]
-; NF-NEXT: subb $123, %dil, %cl # encoding: [0x62,0xf4,0x74,0x18,0x80,0xef,0x7b]
-; NF-NEXT: movzbl %cl, %ecx # encoding: [0x0f,0xb6,0xc9]
-; NF-NEXT: cmovael %ecx, %eax # EVEX TO LEGACY Compression encoding: [0x0f,0x43,0xc1]
-; NF-NEXT: # kill: def $al killed $al killed $eax
-; NF-NEXT: retq # encoding: [0xc3]
entry:
%sub = call i8 @llvm.usub.sat.i8(i8 %a, i8 123)
ret i8 %sub
@@ -826,15 +811,6 @@ define i16 @subflag16ri(i16 noundef %a) {
; CHECK-NEXT: cmovael %edi, %eax # EVEX TO LEGACY Compression encoding: [0x0f,0x43,0xc7]
; CHECK-NEXT: # kill: def $ax killed $ax killed $eax
; CHECK-NEXT: retq # encoding: [0xc3]
-;
-; NF-LABEL: subflag16ri:
-; NF: # %bb.0: # %entry
-; NF-NEXT: xorl %eax, %eax # encoding: [0x31,0xc0]
-; NF-NEXT: subw $1234, %di # EVEX TO LEGACY Compression encoding: [0x66,0x81,0xef,0xd2,0x04]
-; NF-NEXT: # imm = 0x4D2
-; NF-NEXT: cmovael %edi, %eax # EVEX TO LEGACY Compression encoding: [0x0f,0x43,0xc7]
-; NF-NEXT: # kill: def $ax killed $ax killed $eax
-; NF-NEXT: retq # encoding: [0xc3]
entry:
%sub = call i16 @llvm.usub.sat.i16(i16 %a, i16 1234)
ret i16 %sub
@@ -848,14 +824,6 @@ define i32 @subflag32ri(i32 noundef %a) {
; CHECK-NEXT: # imm = 0x1E240
; CHECK-NEXT: cmovael %edi, %eax # EVEX TO LEGACY Compression encoding: [0x0f,0x43,0xc7]
; CHECK-NEXT: retq # encoding: [0xc3]
-;
-; NF-LABEL: subflag32ri:
-; NF: # %bb.0: # %entry
-; NF-NEXT: xorl %eax, %eax # encoding: [0x31,0xc0]
-; NF-NEXT: subl $123456, %edi # EVEX TO LEGACY Compression encoding: [0x81,0xef,0x40,0xe2,0x01,0x00]
-; NF-NEXT: # imm = 0x1E240
-; NF-NEXT: cmovael %edi, %eax # EVEX TO LEGACY Compression encoding: [0x0f,0x43,0xc7]
-; NF-NEXT: retq # encoding: [0xc3]
entry:
%sub = call i32 @llvm.usub.sat.i32(i32 %a, i32 123456)
ret i32 %sub
@@ -869,14 +837,6 @@ define i64 @subflag64ri(i64 noundef %a) {
; CHECK-NEXT: # imm = 0x1E240
; CHECK-NEXT: cmovaeq %rdi, %rax # EVEX TO LEGACY Compression encoding: [0x48,0x0f,0x43,0xc7]
; CHECK-NEXT: retq # encoding: [0xc3]
-;
-; NF-LABEL: subflag64ri:
-; NF: # %bb.0: # %entry
-; NF-NEXT: xorl %eax, %eax # encoding: [0x31,0xc0]
-; NF-NEXT: subq $123456, %rdi # EVEX TO LEGACY Compression encoding: [0x48,0x81,0xef,0x40,0xe2,0x01,0x00]
-; NF-NEXT: # imm = 0x1E240
-; NF-NEXT: cmovaeq %rdi, %rax # EVEX TO LEGACY Compression encoding: [0x48,0x0f,0x43,0xc7]
-; NF-NEXT: retq # encoding: [0xc3]
entry:
%sub = call i64 @llvm.usub.sat.i64(i64 %a, i64 123456)
ret i64 %sub
@@ -902,22 +862,6 @@ define void @sub64ri_reloc(i64 %val) {
; CHECK-NEXT: .cfi_def_cfa_offset 8
; CHECK-NEXT: .LBB41_2: # %f
; CHECK-NEXT: retq # encoding: [0xc3]
-;
-; NF-LABEL: sub64ri_reloc:
-; NF: # %bb.0:
-; NF-NEXT: cmpq $val, %rdi # encoding: [0x48,0x81,0xff,A,A,A,A]
-; NF-NEXT: # fixup A - offset: 3, value: val, kind: reloc_signed_4byte
-; NF-NEXT: jbe .LBB41_2 # encoding: [0x76,A]
-; NF-NEXT: # fixup A - offset: 1, value: .LBB41_2, kind: FK_PCRel_1
-; NF-NEXT: # %bb.1: # %t
-; NF-NEXT: pushq %rax # encoding: [0x50]
-; NF-NEXT: .cfi_def_cfa_offset 16
-; NF-NEXT: callq f at PLT # encoding: [0xe8,A,A,A,A]
-; NF-NEXT: # fixup A - offset: 1, value: f at PLT, kind: FK_PCRel_4
-; NF-NEXT: popq %rax # encoding: [0x58]
-; NF-NEXT: .cfi_def_cfa_offset 8
-; NF-NEXT: .LBB41_2: # %f
-; NF-NEXT: retq # encoding: [0xc3]
%cmp = icmp ugt i64 %val, ptrtoint (ptr @val to i64)
br i1 %cmp, label %t, label %f
@@ -934,11 +878,6 @@ define void @sub8mr_legacy(ptr %a, i8 noundef %b) {
; CHECK: # %bb.0: # %entry
; CHECK-NEXT: subb %sil, (%rdi) # encoding: [0x40,0x28,0x37]
; CHECK-NEXT: retq # encoding: [0xc3]
-;
-; NF-LABEL: sub8mr_legacy:
-; NF: # %bb.0: # %entry
-; NF-NEXT: subb %sil, (%rdi) # encoding: [0x40,0x28,0x37]
-; NF-NEXT: retq # encoding: [0xc3]
entry:
%t= load i8, ptr %a
%sub = sub i8 %t, %b
@@ -951,11 +890,6 @@ define void @sub16mr_legacy(ptr %a, i16 noundef %b) {
; CHECK: # %bb.0: # %entry
; CHECK-NEXT: subw %si, (%rdi) # encoding: [0x66,0x29,0x37]
; CHECK-NEXT: retq # encoding: [0xc3]
-;
-; NF-LABEL: sub16mr_legacy:
-; NF: # %bb.0: # %entry
-; NF-NEXT: subw %si, (%rdi) # encoding: [0x66,0x29,0x37]
-; NF-NEXT: retq # encoding: [0xc3]
entry:
%t= load i16, ptr %a
%sub = sub i16 %t, %b
@@ -968,11 +902,6 @@ define void @sub32mr_legacy(ptr %a, i32 noundef %b) {
; CHECK: # %bb.0: # %entry
; CHECK-NEXT: subl %esi, (%rdi) # encoding: [0x29,0x37]
; CHECK-NEXT: retq # encoding: [0xc3]
-;
-; NF-LABEL: sub32mr_legacy:
-; NF: # %bb.0: # %entry
-; NF-NEXT: subl %esi, (%rdi) # encoding: [0x29,0x37]
-; NF-NEXT: retq # encoding: [0xc3]
entry:
%t= load i32, ptr %a
%sub = sub i32 %t, %b
@@ -985,11 +914,6 @@ define void @sub64mr_legacy(ptr %a, i64 noundef %b) {
; CHECK: # %bb.0: # %entry
; CHECK-NEXT: subq %rsi, (%rdi) # encoding: [0x48,0x29,0x37]
; CHECK-NEXT: retq # encoding: [0xc3]
-;
-; NF-LABEL: sub64mr_legacy:
-; NF: # %bb.0: # %entry
-; NF-NEXT: subq %rsi, (%rdi) # encoding: [0x48,0x29,0x37]
-; NF-NEXT: retq # encoding: [0xc3]
entry:
%t= load i64, ptr %a
%sub = sub i64 %t, %b
@@ -1002,11 +926,6 @@ define void @sub8mi_legacy(ptr %a) {
; CHECK: # %bb.0: # %entry
; CHECK-NEXT: addb $-123, (%rdi) # encoding: [0x80,0x07,0x85]
; CHECK-NEXT: retq # encoding: [0xc3]
-;
-; NF-LABEL: sub8mi_legacy:
-; NF: # %bb.0: # %entry
-; NF-NEXT: addb $-123, (%rdi) # encoding: [0x80,0x07,0x85]
-; NF-NEXT: retq # encoding: [0xc3]
entry:
%t= load i8, ptr %a
%sub = sub nsw i8 %t, 123
@@ -1020,12 +939,6 @@ define void @sub16mi_legacy(ptr %a) {
; CHECK-NEXT: addw $-1234, (%rdi) # encoding: [0x66,0x81,0x07,0x2e,0xfb]
; CHECK-NEXT: # imm = 0xFB2E
; CHECK-NEXT: retq # encoding: [0xc3]
-;
-; NF-LABEL: sub16mi_legacy:
-; NF: # %bb.0: # %entry
-; NF-NEXT: addw $-1234, (%rdi) # encoding: [0x66,0x81,0x07,0x2e,0xfb]
-; NF-NEXT: # imm = 0xFB2E
-; NF-NEXT: retq # encoding: [0xc3]
entry:
%t= load i16, ptr %a
%sub = sub nsw i16 %t, 1234
@@ -1039,12 +952,6 @@ define void @sub32mi_legacy(ptr %a) {
; CHECK-NEXT: addl $-123456, (%rdi) # encoding: [0x81,0x07,0xc0,0x1d,0xfe,0xff]
; CHECK-NEXT: # imm = 0xFFFE1DC0
; CHECK-NEXT: retq # encoding: [0xc3]
-;
-; NF-LABEL: sub32mi_legacy:
-; NF: # %bb.0: # %entry
-; NF-NEXT: addl $-123456, (%rdi) # encoding: [0x81,0x07,0xc0,0x1d,0xfe,0xff]
-; NF-NEXT: # imm = 0xFFFE1DC0
-; NF-NEXT: retq # encoding: [0xc3]
entry:
%t= load i32, ptr %a
%sub = sub nsw i32 %t, 123456
@@ -1058,12 +965,6 @@ define void @sub64mi_legacy(ptr %a) {
; CHECK-NEXT: addq $-123456, (%rdi) # encoding: [0x48,0x81,0x07,0xc0,0x1d,0xfe,0xff]
; CHECK-NEXT: # imm = 0xFFFE1DC0
; CHECK-NEXT: retq # encoding: [0xc3]
-;
-; NF-LABEL: sub64mi_legacy:
-; NF: # %bb.0: # %entry
-; NF-NEXT: addq $-123456, (%rdi) # encoding: [0x48,0x81,0x07,0xc0,0x1d,0xfe,0xff]
-; NF-NEXT: # imm = 0xFFFE1DC0
-; NF-NEXT: retq # encoding: [0xc3]
entry:
%t= load i64, ptr %a
%sub = sub nsw i64 %t, 123456
@@ -1238,3 +1139,95 @@ bb2: ; preds = %bb2, %bb1
store ptr null, ptr %arg2, align 8
br i1 %arg3, label %bb1, label %bb2
}
+;
+define i8 @usubsat8_const1(i8 noundef %a) {
+; CHECK-LABEL: usubsat8_const1:
+; CHECK: # %bb.0: # %entry
+; CHECK-NEXT: cmpb $1, %dil # encoding: [0x40,0x80,0xff,0x01]
+; CHECK-NEXT: adcb $-1, %dil, %al # encoding: [0x62,0xf4,0x7c,0x18,0x80,0xd7,0xff]
+; CHECK-NEXT: retq # encoding: [0xc3]
+entry:
+ %sub = call i8 @llvm.usub.sat.i8(i8 %a, i8 1)
+ ret i8 %sub
+}
+
+define i16 @usubsat16_const1(i16 noundef %a) {
+; CHECK-LABEL: usubsat16_const1:
+; CHECK: # %bb.0: # %entry
+; CHECK-NEXT: cmpw $1, %di # encoding: [0x66,0x83,0xff,0x01]
+; CHECK-NEXT: adcw $-1, %di, %ax # encoding: [0x62,0xf4,0x7d,0x18,0x83,0xd7,0xff]
+; CHECK-NEXT: retq # encoding: [0xc3]
+entry:
+ %sub = call i16 @llvm.usub.sat.i16(i16 %a, i16 1)
+ ret i16 %sub
+}
+
+define i32 @usubsat32_const1(i32 noundef %a) {
+; CHECK-LABEL: usubsat32_const1:
+; CHECK: # %bb.0: # %entry
+; CHECK-NEXT: cmpl $1, %edi # encoding: [0x83,0xff,0x01]
+; CHECK-NEXT: adcl $-1, %edi, %eax # encoding: [0x62,0xf4,0x7c,0x18,0x83,0xd7,0xff]
+; CHECK-NEXT: retq # encoding: [0xc3]
+entry:
+ %sub = call i32 @llvm.usub.sat.i32(i32 %a, i32 1)
+ ret i32 %sub
+}
+
+define i64 @usubsat64_const1(i64 noundef %a) {
+; CHECK-LABEL: usubsat64_const1:
+; CHECK: # %bb.0: # %entry
+; CHECK-NEXT: cmpq $1, %rdi # encoding: [0x48,0x83,0xff,0x01]
+; CHECK-NEXT: adcq $-1, %rdi, %rax # encoding: [0x62,0xf4,0xfc,0x18,0x83,0xd7,0xff]
+; CHECK-NEXT: retq # encoding: [0xc3]
+entry:
+ %sub = call i64 @llvm.usub.sat.i64(i64 %a, i64 1)
+ ret i64 %sub
+}
+
+define i32 @usubsat32_var(i32 noundef %a, i32 noundef %b) {
+; CHECK-LABEL: usubsat32_var:
+; CHECK: # %bb.0: # %entry
+; CHECK-NEXT: xorl %eax, %eax # encoding: [0x31,0xc0]
+; CHECK-NEXT: subl %esi, %edi # EVEX TO LEGACY Compression encoding: [0x29,0xf7]
+; CHECK-NEXT: cmovael %edi, %eax # EVEX TO LEGACY Compression encoding: [0x0f,0x43,0xc7]
+; CHECK-NEXT: retq # encoding: [0xc3]
+entry:
+ %sub = call i32 @llvm.usub.sat.i32(i32 %a, i32 %b)
+ ret i32 %sub
+}
+
+define i32 @usubsat32_const123(i32 noundef %a) {
+; CHECK-LABEL: usubsat32_const123:
+; CHECK: # %bb.0: # %entry
+; CHECK-NEXT: xorl %eax, %eax # encoding: [0x31,0xc0]
+; CHECK-NEXT: subl $123, %edi # EVEX TO LEGACY Compression encoding: [0x83,0xef,0x7b]
+; CHECK-NEXT: cmovael %edi, %eax # EVEX TO LEGACY Compression encoding: [0x0f,0x43,0xc7]
+; CHECK-NEXT: retq # encoding: [0xc3]
+entry:
+ %sub = call i32 @llvm.usub.sat.i32(i32 %a, i32 123)
+ ret i32 %sub
+}
+
+define i32 @uaddsat32_const1(i32 noundef %a) {
+; CHECK-LABEL: uaddsat32_const1:
+; CHECK: # %bb.0: # %entry
+; CHECK-NEXT: incl %edi, %eax # encoding: [0x62,0xf4,0x7c,0x18,0xff,0xc7]
+; CHECK-NEXT: movl $-1, %ecx # encoding: [0xb9,0xff,0xff,0xff,0xff]
+; CHECK-NEXT: cmovel %ecx, %eax # EVEX TO LEGACY Compression encoding: [0x0f,0x44,0xc1]
+; CHECK-NEXT: retq # encoding: [0xc3]
+entry:
+ %add = call i32 @llvm.uadd.sat.i32(i32 %a, i32 1)
+ ret i32 %add
+}
+
+define i32 @uaddsat32_const_neg1(i32 noundef %a) {
+; CHECK-LABEL: uaddsat32_const_neg1:
+; CHECK: # %bb.0: # %entry
+; CHECK-NEXT: addl $-1, %edi, %eax # encoding: [0x62,0xf4,0x7c,0x18,0x83,0xc7,0xff]
+; CHECK-NEXT: movl $-1, %ecx # encoding: [0xb9,0xff,0xff,0xff,0xff]
+; CHECK-NEXT: cmovbl %ecx, %eax # EVEX TO LEGACY Compression encoding: [0x0f,0x42,0xc1]
+; CHECK-NEXT: retq # encoding: [0xc3]
+entry:
+ %add = call i32 @llvm.uadd.sat.i32(i32 %a, i32 -1)
+ ret i32 %add
+}
More information about the llvm-commits
mailing list