[llvm] [CostModel][X86] Add variable divisor div/rem costs for scalar and <=i32 vectors (PR #215124)

via llvm-commits llvm-commits at lists.llvm.org
Mon Sep 7 23:08:18 PDT 2026


================
@@ -753,6 +753,278 @@ InstructionCost X86TTIImpl::getArithmeticInstrCost(
       if (auto KindCost = Entry->Cost[CostKind])
         return LT.first * *KindCost;
 
+  // rem matches div because the divider returns the remainder for free.
+  static const CostKindTblEntry ScalarVarDivCostTable[] = {
+    { ISD::SDIV, MVT::i8,  { 15, 20, 2, 4 } },
+    { ISD::UDIV, MVT::i8,  { 15, 20, 2, 4 } },
+    { ISD::SREM, MVT::i8,  { 15, 20, 2, 4 } },
+    { ISD::UREM, MVT::i8,  { 15, 20, 2, 4 } },
+    { ISD::SDIV, MVT::i16, { 17, 20, 2, 4 } },
+    { ISD::UDIV, MVT::i16, { 17, 20, 2, 4 } },
+    { ISD::SREM, MVT::i16, { 17, 20, 2, 4 } },
+    { ISD::UREM, MVT::i16, { 17, 20, 2, 4 } },
+    { ISD::SDIV, MVT::i32, { 25, 22, 2, 4 } },
+    { ISD::UDIV, MVT::i32, { 25, 22, 2, 4 } },
+    { ISD::SREM, MVT::i32, { 25, 22, 2, 4 } },
+    { ISD::UREM, MVT::i32, { 25, 22, 2, 4 } },
+    { ISD::SDIV, MVT::i64, { 41, 24, 2, 4 } },
+    { ISD::UDIV, MVT::i64, { 41, 24, 2, 4 } },
+    { ISD::SREM, MVT::i64, { 41, 24, 2, 4 } },
+    { ISD::UREM, MVT::i64, { 41, 24, 2, 4 } },
+  };
+
+  if (!LT.second.isVector() && !Op2Info.isConstant())
+    if (const auto *Entry =
+            CostTableLookup(ScalarVarDivCostTable, ISD, LT.second))
+      if (auto KindCost = Entry->Cost[CostKind])
+        return LT.first * *KindCost;
+
+  // Variable divisors lower through a float divide. strictfp needs SAE
+  // rounding which is 512-bit only.
+  bool IsStrictFP =
+      CxtI && CxtI->getFunction()->hasFnAttribute(Attribute::StrictFP);
+  bool VarDivToFP =
+      !Op2Info.isConstant() && (!IsStrictFP || ST->useAVX512Regs());
+
+  // i64 needs the qq converts, which are AVX512DQ only. Two tables because the
+  // lowering picks by operand value and not by type.
+  static const CostKindTblEntry AVX512DQExactVarDivCostTable[] = {
+    { ISD::UDIV, MVT::v2i64,  {   5 } }, // cvt+divpd sequence
+    { ISD::SDIV, MVT::v2i64,  {   5 } },
+    { ISD::UREM, MVT::v2i64,  {   5 } },
+    { ISD::SREM, MVT::v2i64,  {   5 } },
+    { ISD::UDIV, MVT::v4i64,  {   8 } },
+    { ISD::SDIV, MVT::v4i64,  {   8 } },
+    { ISD::UREM, MVT::v4i64,  {   8 } },
+    { ISD::SREM, MVT::v4i64,  {   8 } },
+    { ISD::UDIV, MVT::v8i64,  {  16 } },
+    { ISD::SDIV, MVT::v8i64,  {  16 } },
+    { ISD::UREM, MVT::v8i64,  {  16 } },
+    { ISD::SREM, MVT::v8i64,  {  16 } },
+  };
+
+  static const CostKindTblEntry AVX512DQVarDivCostTable[] = {
+    { ISD::UDIV, MVT::v2i64,  {  16 } },
+    { ISD::SDIV, MVT::v2i64,  {  16 } },
+    { ISD::UREM, MVT::v2i64,  {  16 } },
+    { ISD::SREM, MVT::v2i64,  {  16 } },
+    { ISD::UDIV, MVT::v4i64,  {  16 } },
+    { ISD::SDIV, MVT::v4i64,  {  16 } },
+    { ISD::UREM, MVT::v4i64,  {  16 } },
+    { ISD::SREM, MVT::v4i64,  {  16 } },
+    { ISD::UDIV, MVT::v8i64,  {  16 } },
+    { ISD::SDIV, MVT::v8i64,  {  18 } },
+    { ISD::UREM, MVT::v8i64,  {  16 } },
+    { ISD::SREM, MVT::v8i64,  {  18 } },
+  };
+
+  if (VarDivToFP && ST->hasDQI() && ST->useAVX512Regs() &&
+      LT.second.getScalarType() == MVT::i64) {
+    // The DAG combine picks between the two sequences with these same two
+    // queries, so the cost cannot disagree with what codegen emits.
----------------
Andarwinux wrote:

Can the same method be used to provide the correct cost for IEEEfloat shrink as well?

https://github.com/llvm/llvm-project/pull/215124


More information about the llvm-commits mailing list