[llvm] [CostModel][X86] Add variable divisor div/rem costs for scalar and <=i32 vectors (PR #215124)
via llvm-commits
llvm-commits at lists.llvm.org
Mon Sep 7 23:08:18 PDT 2026
================
@@ -753,6 +753,278 @@ InstructionCost X86TTIImpl::getArithmeticInstrCost(
if (auto KindCost = Entry->Cost[CostKind])
return LT.first * *KindCost;
+ // rem matches div because the divider returns the remainder for free.
+ static const CostKindTblEntry ScalarVarDivCostTable[] = {
+ { ISD::SDIV, MVT::i8, { 15, 20, 2, 4 } },
+ { ISD::UDIV, MVT::i8, { 15, 20, 2, 4 } },
+ { ISD::SREM, MVT::i8, { 15, 20, 2, 4 } },
+ { ISD::UREM, MVT::i8, { 15, 20, 2, 4 } },
+ { ISD::SDIV, MVT::i16, { 17, 20, 2, 4 } },
+ { ISD::UDIV, MVT::i16, { 17, 20, 2, 4 } },
+ { ISD::SREM, MVT::i16, { 17, 20, 2, 4 } },
+ { ISD::UREM, MVT::i16, { 17, 20, 2, 4 } },
+ { ISD::SDIV, MVT::i32, { 25, 22, 2, 4 } },
+ { ISD::UDIV, MVT::i32, { 25, 22, 2, 4 } },
+ { ISD::SREM, MVT::i32, { 25, 22, 2, 4 } },
+ { ISD::UREM, MVT::i32, { 25, 22, 2, 4 } },
+ { ISD::SDIV, MVT::i64, { 41, 24, 2, 4 } },
+ { ISD::UDIV, MVT::i64, { 41, 24, 2, 4 } },
+ { ISD::SREM, MVT::i64, { 41, 24, 2, 4 } },
+ { ISD::UREM, MVT::i64, { 41, 24, 2, 4 } },
+ };
+
+ if (!LT.second.isVector() && !Op2Info.isConstant())
+ if (const auto *Entry =
+ CostTableLookup(ScalarVarDivCostTable, ISD, LT.second))
+ if (auto KindCost = Entry->Cost[CostKind])
+ return LT.first * *KindCost;
+
+ // Variable divisors lower through a float divide. strictfp needs SAE
+ // rounding which is 512-bit only.
+ bool IsStrictFP =
+ CxtI && CxtI->getFunction()->hasFnAttribute(Attribute::StrictFP);
+ bool VarDivToFP =
+ !Op2Info.isConstant() && (!IsStrictFP || ST->useAVX512Regs());
+
+ // i64 needs the qq converts, which are AVX512DQ only. Two tables because the
+ // lowering picks by operand value and not by type.
+ static const CostKindTblEntry AVX512DQExactVarDivCostTable[] = {
+ { ISD::UDIV, MVT::v2i64, { 5 } }, // cvt+divpd sequence
+ { ISD::SDIV, MVT::v2i64, { 5 } },
+ { ISD::UREM, MVT::v2i64, { 5 } },
+ { ISD::SREM, MVT::v2i64, { 5 } },
+ { ISD::UDIV, MVT::v4i64, { 8 } },
+ { ISD::SDIV, MVT::v4i64, { 8 } },
+ { ISD::UREM, MVT::v4i64, { 8 } },
+ { ISD::SREM, MVT::v4i64, { 8 } },
+ { ISD::UDIV, MVT::v8i64, { 16 } },
+ { ISD::SDIV, MVT::v8i64, { 16 } },
+ { ISD::UREM, MVT::v8i64, { 16 } },
+ { ISD::SREM, MVT::v8i64, { 16 } },
+ };
+
+ static const CostKindTblEntry AVX512DQVarDivCostTable[] = {
+ { ISD::UDIV, MVT::v2i64, { 16 } },
+ { ISD::SDIV, MVT::v2i64, { 16 } },
+ { ISD::UREM, MVT::v2i64, { 16 } },
+ { ISD::SREM, MVT::v2i64, { 16 } },
+ { ISD::UDIV, MVT::v4i64, { 16 } },
+ { ISD::SDIV, MVT::v4i64, { 16 } },
+ { ISD::UREM, MVT::v4i64, { 16 } },
+ { ISD::SREM, MVT::v4i64, { 16 } },
+ { ISD::UDIV, MVT::v8i64, { 16 } },
+ { ISD::SDIV, MVT::v8i64, { 18 } },
+ { ISD::UREM, MVT::v8i64, { 16 } },
+ { ISD::SREM, MVT::v8i64, { 18 } },
+ };
+
+ if (VarDivToFP && ST->hasDQI() && ST->useAVX512Regs() &&
+ LT.second.getScalarType() == MVT::i64) {
+ // The DAG combine picks between the two sequences with these same two
+ // queries, so the cost cannot disagree with what codegen emits.
----------------
Andarwinux wrote:
Can the same method be used to provide the correct cost for IEEEfloat shrink as well?
https://github.com/llvm/llvm-project/pull/215124
More information about the llvm-commits
mailing list