[llvm] [X86] Lower vector integer division and remainder through float division (PR #205263)
Adam Scott via llvm-commits
llvm-commits at lists.llvm.org
Sun Jul 19 00:53:41 PDT 2026
================
@@ -50516,6 +50524,180 @@ static SDValue combineMulToPMADD52(SDNode *N, const SDLoc &DL,
return SDValue();
}
+// x86 has no vector integer divide instructions. Lower vector
+// UDIV/SDIV/UREM/SREM through float division instead of scalarizing into N
+// scalar hardware divides.
+static SDValue combineIntDivRem(SDNode *N, SelectionDAG &DAG,
+ TargetLowering::DAGCombinerInfo &DCI,
+ const X86Subtarget &Subtarget) {
+ EVT VT = N->getValueType(0);
+ SDLoc DL(N);
+
+ // Run before the legalizer expands the division.
+ if (!VT.isVector() || !Subtarget.hasSSE2() || !DCI.isBeforeLegalizeOps())
+ return SDValue();
+
+ SDValue Dividend = N->getOperand(0);
+ SDValue Divisor = N->getOperand(1);
+ unsigned Opc = N->getOpcode();
+
+ // Disabled lanes are poison and fdiv never traps, so ignore the mask.
+ if (Opc == ISD::MASKED_UDIV || Opc == ISD::MASKED_SDIV ||
+ Opc == ISD::MASKED_UREM || Opc == ISD::MASKED_SREM)
+ Opc = ISD::getUnmaskedBinOpOpcode(Opc);
+ bool IsRem = Opc == ISD::UREM || Opc == ISD::SREM;
+ bool IsSigned = Opc == ISD::SDIV || Opc == ISD::SREM;
+
+ // If the result is only read back as scalar extracts, scalarization computes
+ // just the demanded lanes.
+ if (all_of(N->users(), [](const SDNode *U) {
+ return U->getOpcode() == ISD::EXTRACT_VECTOR_ELT;
+ }))
+ return SDValue();
+
+ // Magic multiply lowers constant divisors cheaper than a divide.
+ if (DAG.isConstantIntBuildVectorOrConstantInt(Divisor))
+ return SDValue();
+
+ // i8/i16/i32: operands fit the float mantissa exactly (f32 for <=16-bit, f64
+ // for 32-bit) so one float divide recovers the exact quotient.
+ if (VT.getScalarSizeInBits() <= 32) {
+ MVT FPSclVT = VT.getScalarSizeInBits() <= 16 ? MVT::f32 : MVT::f64;
+ EVT FPVT = VT.changeVectorElementType(*DAG.getContext(), FPSclVT);
+
+ // Nothing will split an illegal FP type after type legalization.
+ if (!DCI.isBeforeLegalize() &&
+ !DAG.getTargetLoweringInfo().isTypeLegal(FPVT))
+ return SDValue();
+
+ bool IsStrict = DAG.getMachineFunction().getFunction().hasFnAttribute(
+ Attribute::StrictFP);
+ if (IsStrict) {
+ // No SAE below 512-bit AVX512
+ if (!Subtarget.useAVX512Regs())
+ return SDValue();
+ // More lanes than one zmm divide can hold so split the divide.
+ if (FPVT.getSizeInBits() > 512)
+ return splitVectorIntBinary(SDValue(N, 0), DAG, DL);
+ } else if (!IsSigned && VT.getScalarSizeInBits() == 32 &&
+ !Subtarget.hasAVX2()) {
+ // Unsigned i32 needs FP_TO_UINT(f64->u32) which is emulated and a loss
+ // for latency and code size before AVX2.
+ return SDValue();
+ }
+
+ unsigned ToFP = IsSigned ? ISD::SINT_TO_FP : ISD::UINT_TO_FP;
+ SDValue X = DAG.getNode(ToFP, DL, FPVT, Dividend);
+ SDValue Y = DAG.getNode(ToFP, DL, FPVT, Divisor);
+ SDValue Q;
+ if (IsStrict) {
+ // The converts are exact so only the divide and the truncate can
+ // raise flags.
+ unsigned WideElts = 512 / FPSclVT.getSizeInBits(); // 16 f32 or 8 f64
+ MVT WideFP = MVT::getVectorVT(FPSclVT, WideElts);
+ MVT WideI = MVT::getVectorVT(MVT::i32, WideElts);
+ SDValue RN = DAG.getTargetConstant(X86::STATIC_ROUNDING::TO_NEAREST_INT,
+ DL, MVT::i32); // {rn-sae}
+ SDValue Quot =
+ DAG.getNode(X86ISD::FDIV_RND, DL, WideFP,
+ widenSubVector(X, false, Subtarget, DAG, DL, 512),
+ widenSubVector(Y, false, Subtarget, DAG, DL, 512), RN);
+ unsigned FromFP = IsSigned ? X86ISD::CVTTP2SI_SAE : X86ISD::CVTTP2UI_SAE;
+ Q = DAG.getNode(FromFP, DL, WideI, Quot); // vcvttp*2dq {sae}
+ MVT NarrowI = MVT::getVectorVT(MVT::i32, VT.getVectorNumElements());
+ Q = extractSubVector(Q, 0, DAG, DL, NarrowI.getSizeInBits());
+ Q = DAG.getNode(ISD::TRUNCATE, DL, VT, Q);
+ } else {
+ unsigned FromFP = IsSigned ? ISD::FP_TO_SINT : ISD::FP_TO_UINT;
+ Q = DAG.getNode(FromFP, DL, VT, DAG.getNode(ISD::FDIV, DL, FPVT, X, Y));
+ }
----------------
as4230 wrote:
I agree a generic impl would duplicate some of the <=i32 logic. My leaning is that there should be a second target that wants this before lifting it into generic DAGCombine. With one consumer it's mostly a taste call.
https://github.com/llvm/llvm-project/pull/205263
More information about the llvm-commits
mailing list