[llvm] [X86] Lower i64 vector division and remainder through float division (PR #215043)
Adam Scott via llvm-commits
llvm-commits at lists.llvm.org
Tue Aug 25 08:47:04 PDT 2026
================
@@ -50701,92 +50857,53 @@ static SDValue combineIntDivRem(SDNode *N, SelectionDAG &DAG,
auto BothFitFP = [&](const fltSemantics &Sem) {
return FitsFP(Dividend, Sem) && FitsFP(Divisor, Sem);
};
- // i8/i16/i32: the operands fit the float mantissa
- // exactly so one float divide recovers the exact quotient.
- if (EltBits > 32)
+
+ // i64 needs the qq converts which is AVX512DQ only.
+ bool NarrowI64 = EltBits == 64 && Subtarget.hasDQI() &&
+ Subtarget.useAVX512Regs() &&
+ BothFitFP(APFloat::IEEEdouble());
+
+ // i8/i16/i32 and narrow value i64 take one exact float divide.
+ bool UseExactFPDiv = EltBits <= 32 || NarrowI64;
+ if (!UseExactFPDiv && EltBits != 64)
return SDValue();
// f32 recovers the quotient exactly when both operands fit in 24 bits
MVT FPSclVT = MVT::f64;
- if (EltBits <= 16 || BothFitFP(APFloat::IEEEsingle()))
+ if (UseExactFPDiv && (EltBits <= 16 || BothFitFP(APFloat::IEEEsingle())))
FPSclVT = MVT::f32;
- EVT FPVT = VT.changeVectorElementType(*DAG.getContext(), FPSclVT);
bool IsStrict = DAG.getMachineFunction().getFunction().hasFnAttribute(
Attribute::StrictFP);
- if (IsStrict) {
- // The SAE forms are 512-bit only. Inputs widen into a zmm below, which
- // requires 512-bit types to be legal.
- if (!Subtarget.useAVX512Regs())
- return SDValue();
- // Widen a non-power-of-two lane count to get a machine type, but only
- // while it still fits one divide. Two chains lose to a chain plus a scalar.
- unsigned NumElts = VT.getVectorNumElements();
- if (!isPowerOf2_32(NumElts)) {
- if (NextPowerOf2(NumElts) * FPSclVT.getSizeInBits() > 512)
- return SDValue();
- SDValue WideDividend = DAG.WidenVector(Dividend, DL);
- EVT WideVT = WideDividend.getValueType();
- SDValue WideDivisor = DAG.WidenVector(Divisor, DL);
- SDValue Wide = DAG.getNode(Opc, DL, WideVT, WideDividend, WideDivisor);
- return DAG.getExtractSubvector(DL, VT, Wide, 0);
- }
- } else if (!IsSigned && VT.getScalarSizeInBits() == 32 &&
- !Subtarget.hasAVX2()) {
- // Unsigned i32 needs FP_TO_UINT(f64->u32) which is emulated and a loss
- // for latency and code size before AVX2.
- return SDValue();
- }
- // Nothing will split an illegal FP type after type legalization and the
- // strict SAE divide is 512-bit only.
- bool FPVTUsable = IsStrict
- ? FPVT.getSizeInBits() <= 512
- : DCI.isBeforeLegalize() ||
- DAG.getTargetLoweringInfo().isTypeLegal(FPVT);
+ // The SAE forms are 512-bit only. Inputs widen into a zmm below, which
+ // requires 512-bit types to be legal.
+ if (IsStrict && !Subtarget.useAVX512Regs())
+ return SDValue();
- // Halve the divide while the integer halves stay legal.
- if (!FPVTUsable) {
- if (VT.is256BitVector() || VT.is512BitVector()) {
- EVT HalfVT = VT.getHalfNumVectorElementsVT(*DAG.getContext());
- if (DAG.getTargetLoweringInfo().isTypeLegal(HalfVT))
- return splitVectorIntBinary(SDValue(N, 0), DAG, DL);
- }
+ if (!UseExactFPDiv && (!Subtarget.hasDQI() || !Subtarget.useAVX512Regs()))
return SDValue();
- }
- unsigned ToFP = IsSigned ? ISD::SINT_TO_FP : ISD::UINT_TO_FP;
- SDValue X = DAG.getNode(ToFP, DL, FPVT, Dividend);
- SDValue Y = DAG.getNode(ToFP, DL, FPVT, Divisor);
- SDValue Q;
- if (IsStrict) {
- // The converts are exact so only the divide and the truncate can
- // raise flags.
- unsigned WideElts = 512 / FPSclVT.getSizeInBits(); // 16 f32 or 8 f64
- MVT WideFP = MVT::getVectorVT(FPSclVT, WideElts);
- MVT WideIScl = MVT::i32;
- MVT WideI = MVT::getVectorVT(WideIScl, WideElts);
- SDValue RN = DAG.getTargetConstant(X86::STATIC_ROUNDING::TO_NEAREST_INT, DL,
- MVT::i32); // {rn-sae}
- SDValue Quot =
- DAG.getNode(X86ISD::FDIV_RND, DL, WideFP,
- widenSubVector(X, false, Subtarget, DAG, DL, 512),
- widenSubVector(Y, false, Subtarget, DAG, DL, 512), RN);
- unsigned FromFP = IsSigned ? X86ISD::CVTTP2SI_SAE : X86ISD::CVTTP2UI_SAE;
- Q = DAG.getNode(FromFP, DL, WideI, Quot); // vcvttp*2dq/qq {sae}
- MVT NarrowI = MVT::getVectorVT(WideIScl, VT.getVectorNumElements());
- Q = extractSubVector(Q, 0, DAG, DL, NarrowI.getSizeInBits());
- Q = IsSigned ? DAG.getSExtOrTrunc(Q, DL, VT)
- : DAG.getZExtOrTrunc(Q, DL, VT);
- } else {
- unsigned FromFP = IsSigned ? ISD::FP_TO_SINT : ISD::FP_TO_UINT;
- Q = DAG.getNode(FromFP, DL, VT, DAG.getNode(ISD::FDIV, DL, FPVT, X, Y));
+ // Widen a non-power-of-two lane count to get a machine type, but only
+ // while it still fits one divide. Two chains lose to a chain plus a scalar.
+ unsigned NumElts = VT.getVectorNumElements();
+ if ((IsStrict || !UseExactFPDiv) && !isPowerOf2_32(NumElts)) {
+ if (NextPowerOf2(NumElts) * FPSclVT.getSizeInBits() > 512)
+ return SDValue();
+ SDValue WideDividend = DAG.WidenVector(Dividend, DL);
+ EVT WideVT = WideDividend.getValueType();
+ SDValue WideDivisor = DAG.WidenVector(Divisor, DL);
+ SDValue Wide = DAG.getNode(Opc, DL, WideVT, WideDividend, WideDivisor);
+ return DAG.getExtractSubvector(DL, VT, Wide, 0);
----------------
as4230 wrote:
Yes thank you for catching this. I need to add non pow2 narrow strictfp tests too. WidenVector pads with poison so computeKnownBits can't see the range anymore and it drops off the narrow path. So we need to build the wide node and call the lowering function ourselves instead of tossing it back up.
https://github.com/llvm/llvm-project/pull/215043
More information about the llvm-commits
mailing list