[llvm] [X86] Add tuning for fast AVX2 vector division (PR #219873)

Xiaomeng Zhang via llvm-commits llvm-commits at lists.llvm.org
Sat Sep 12 01:21:27 PDT 2026


https://github.com/JacketPants updated https://github.com/llvm/llvm-project/pull/219873

>From a38e7ed6617c0b43c2033bd3f82294b656c9ef4d Mon Sep 17 00:00:00 2001
From: Xiaomeng Zhang <zhangxiaomeng at hygon.cn>
Date: Thu, 27 Aug 2026 17:40:31 +0800
Subject: [PATCH 1/2] [X86] Add tuning for fast AVX2 vector division

On some microarchitectures, 256-bit vector division has no significant
performance difference compared with 128-bit vector division. Add
TuningFastVectorFDIV to adjust the TTI cost estimates for these targets.
---
 llvm/lib/Target/X86/X86.td                          |  7 +++++++
 llvm/lib/Target/X86/X86TargetTransformInfo.cpp      | 13 +++++++++++++
 llvm/test/Analysis/CostModel/X86/arith-fp.ll        | 13 +++++++++++++
 .../SLPVectorizer/X86/alternate-fp-inseltpoison.ll  |  7 +++++++
 .../Transforms/SLPVectorizer/X86/alternate-fp.ll    |  7 +++++++
 5 files changed, 47 insertions(+)

diff --git a/llvm/lib/Target/X86/X86.td b/llvm/lib/Target/X86/X86.td
index 5786c659b0f98..e63f4fd43996e 100644
--- a/llvm/lib/Target/X86/X86.td
+++ b/llvm/lib/Target/X86/X86.td
@@ -741,6 +741,12 @@ def TuningFastVectorFSQRT
     : SubtargetFeature<"fast-vector-fsqrt", "HasFastVectorFSQRT",
                        "true", "Vector SQRT is fast (disable Newton-Raphson)",
                        [], InlineIgnore>;
+// True if 256-bit VDIVPS/VDIVPD instructions have no significant width penalty
+// compared with their 128-bit counterparts.
+def TuningFastVectorFDIV
+    : SubtargetFeature<"fast-vector-fdiv", "HasFastVectorFDIV",
+                       "true", "Vector FDIV is fast (enable fast costs)",
+                       [], InlineIgnore>;
 
 // If lzcnt has equivalent latency/throughput to most simple integer ops, it can
 // be used to replace test/set sequences.
@@ -1827,6 +1833,7 @@ def ProcessorFeatures {
                                           TuningFast15ByteNOP,
                                           TuningFastScalarFSQRT,
                                           TuningFastVectorFSQRT,
+                                          TuningFastVectorFDIV,
                                           TuningFastScalarShiftMasks,
                                           TuningFastVariablePerLaneShuffle,
                                           TuningFastMOVBE,
diff --git a/llvm/lib/Target/X86/X86TargetTransformInfo.cpp b/llvm/lib/Target/X86/X86TargetTransformInfo.cpp
index 8e0cf1fc5a153..8a4ac4af21058 100644
--- a/llvm/lib/Target/X86/X86TargetTransformInfo.cpp
+++ b/llvm/lib/Target/X86/X86TargetTransformInfo.cpp
@@ -1253,6 +1253,19 @@ InstructionCost X86TTIImpl::getArithmeticInstrCost(
     { ISD::FDIV, MVT::v4f64,   { 28, 35, 1, 3 } }, // vdivpd
   };
 
+  // Targets with fast 256-bit vector division use lower costs than the
+  // generic AVX2 table. This must be checked before the generic AVX2 lookup.
+  if (ST->hasAVX2() && ST->hasFastVectorFDIV()) {
+    static const CostKindTblEntry AVX2FastVectorFDIVCostTable[] = {
+      { ISD::FDIV, MVT::v8f32,   {  7, 13, 1, 3 } }, // vdivps
+      { ISD::FDIV, MVT::v4f64,   { 14, 20, 1, 3 } }, // vdivpd
+    };
+    if (const auto *Entry =
+            CostTableLookup(AVX2FastVectorFDIVCostTable, ISD, LT.second))
+      if (auto KindCost = Entry->Cost[CostKind])
+        return LT.first * *KindCost;
+  }
+
   // Look for AVX2 lowering tricks for custom cases.
   if (ST->hasAVX2())
     if (const auto *Entry = CostTableLookup(AVX2CostTable, ISD, LT.second))
diff --git a/llvm/test/Analysis/CostModel/X86/arith-fp.ll b/llvm/test/Analysis/CostModel/X86/arith-fp.ll
index e3eb3e60b844c..139212ef42501 100644
--- a/llvm/test/Analysis/CostModel/X86/arith-fp.ll
+++ b/llvm/test/Analysis/CostModel/X86/arith-fp.ll
@@ -4,6 +4,8 @@
 ; RUN: opt < %s  -passes="print<cost-model>" 2>&1 -disable-output -cost-kind=all -mtriple=x86_64-- -mattr=+sse4.2 | FileCheck %s --check-prefixes=SSE42
 ; RUN: opt < %s  -passes="print<cost-model>" 2>&1 -disable-output -cost-kind=all -mtriple=x86_64-- -mattr=+avx | FileCheck %s --check-prefixes=AVX,AVX1
 ; RUN: opt < %s  -passes="print<cost-model>" 2>&1 -disable-output -cost-kind=all -mtriple=x86_64-- -mattr=+avx2 | FileCheck %s --check-prefixes=AVX,AVX2
+; RUN: opt < %s  -passes="print<cost-model>" 2>&1 -disable-output -cost-kind=all -mtriple=x86_64-- -mattr=+avx2,+fast-vector-fdiv | FileCheck %s --check-prefixes=AVX2-FAST-FDIV
+; RUN: opt < %s  -passes="print<cost-model>" 2>&1 -disable-output -cost-kind=all -mtriple=x86_64-- -mcpu=c86-4g-m4 | FileCheck %s --check-prefixes=AVX2-FAST-FDIV
 ; RUN: opt < %s  -passes="print<cost-model>" 2>&1 -disable-output -cost-kind=all -mtriple=x86_64-- -mattr=+avx512f | FileCheck %s --check-prefixes=AVX512
 ; RUN: opt < %s  -passes="print<cost-model>" 2>&1 -disable-output -cost-kind=all -mtriple=x86_64-- -mattr=+avx512f,+avx512bw | FileCheck %s --check-prefixes=AVX512
 ;
@@ -577,6 +579,17 @@ define i32 @fdiv(i32 %arg) {
 ; AVX2-NEXT:  Cost Model: Found costs of RThru:56 CodeSize:2 Lat:70 SizeLat:6 for: %V8F64 = fdiv <8 x double> undef, undef
 ; AVX2-NEXT:  Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret i32 undef
 ;
+; AVX2-FAST-FDIV-LABEL: 'fdiv'
+; AVX2-FAST-FDIV-NEXT:  Cost Model: Found costs of RThru:7 CodeSize:1 Lat:13 SizeLat:1 for: %F32 = fdiv float undef, undef
+; AVX2-FAST-FDIV-NEXT:  Cost Model: Found costs of RThru:7 CodeSize:1 Lat:13 SizeLat:1 for: %V4F32 = fdiv <4 x float> undef, undef
+; AVX2-FAST-FDIV-NEXT:  Cost Model: Found costs of RThru:7 CodeSize:1 Lat:13 SizeLat:3 for: %V8F32 = fdiv <8 x float> undef, undef
+; AVX2-FAST-FDIV-NEXT:  Cost Model: Found costs of RThru:14 CodeSize:2 Lat:26 SizeLat:6 for: %V16F32 = fdiv <16 x float> undef, undef
+; AVX2-FAST-FDIV-NEXT:  Cost Model: Found costs of RThru:14 CodeSize:1 Lat:20 SizeLat:1 for: %F64 = fdiv double undef, undef
+; AVX2-FAST-FDIV-NEXT:  Cost Model: Found costs of RThru:14 CodeSize:1 Lat:20 SizeLat:1 for: %V2F64 = fdiv <2 x double> undef, undef
+; AVX2-FAST-FDIV-NEXT:  Cost Model: Found costs of RThru:14 CodeSize:1 Lat:20 SizeLat:3 for: %V4F64 = fdiv <4 x double> undef, undef
+; AVX2-FAST-FDIV-NEXT:  Cost Model: Found costs of RThru:28 CodeSize:2 Lat:40 SizeLat:6 for: %V8F64 = fdiv <8 x double> undef, undef
+; AVX2-FAST-FDIV-NEXT:  Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret i32 undef
+;
 ; AVX512-LABEL: 'fdiv'
 ; AVX512-NEXT:  Cost Model: Found costs of RThru:3 CodeSize:1 Lat:11 SizeLat:1 for: %F32 = fdiv float undef, undef
 ; AVX512-NEXT:  Cost Model: Found costs of RThru:3 CodeSize:1 Lat:11 SizeLat:1 for: %V4F32 = fdiv <4 x float> undef, undef
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/alternate-fp-inseltpoison.ll b/llvm/test/Transforms/SLPVectorizer/X86/alternate-fp-inseltpoison.ll
index 06498563a7d37..9a0f9cda167ae 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/alternate-fp-inseltpoison.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/alternate-fp-inseltpoison.ll
@@ -3,6 +3,7 @@
 ; RUN: opt < %s -mtriple=x86_64-unknown -mcpu=slm -passes=slp-vectorizer -S | FileCheck %s --check-prefix=SLM
 ; RUN: opt < %s -mtriple=x86_64-unknown -mcpu=corei7-avx -passes=slp-vectorizer -S | FileCheck %s --check-prefix=AVX
 ; RUN: opt < %s -mtriple=x86_64-unknown -mcpu=core-avx2 -passes=slp-vectorizer -S | FileCheck %s --check-prefix=AVX2
+; RUN: opt < %s -mtriple=x86_64-unknown -mattr=+avx2,+fast-vector-fdiv -passes=slp-vectorizer -S | FileCheck %s --check-prefix=AVX2-FAST-FDIV
 ; RUN: opt < %s -mtriple=x86_64-unknown -mcpu=knl -passes=slp-vectorizer -S | FileCheck %s --check-prefix=AVX512
 ; RUN: opt < %s -mtriple=x86_64-unknown -mcpu=skx -passes=slp-vectorizer -S | FileCheck %s --check-prefix=AVX512
 
@@ -146,6 +147,12 @@ define <8 x float> @fmul_fdiv_v8f32(<8 x float> %a, <8 x float> %b) {
 ; AVX2-NEXT:    [[TMP5:%.*]] = shufflevector <8 x float> [[TMP8]], <8 x float> poison, <8 x i32> <i32 0, i32 4, i32 5, i32 1, i32 2, i32 6, i32 7, i32 3>
 ; AVX2-NEXT:    ret <8 x float> [[TMP5]]
 ;
+; AVX2-FAST-FDIV-LABEL: @fmul_fdiv_v8f32(
+; AVX2-FAST-FDIV-NEXT:    [[TMP1:%.*]] = fmul <8 x float> [[A:%.*]], [[B:%.*]]
+; AVX2-FAST-FDIV-NEXT:    [[TMP2:%.*]] = fdiv <8 x float> [[A]], [[B]]
+; AVX2-FAST-FDIV-NEXT:    [[TMP3:%.*]] = shufflevector <8 x float> [[TMP1]], <8 x float> [[TMP2]], <8 x i32> <i32 0, i32 9, i32 10, i32 3, i32 4, i32 13, i32 14, i32 7>
+; AVX2-FAST-FDIV-NEXT:    ret <8 x float> [[TMP3]]
+;
 ; AVX512-LABEL: @fmul_fdiv_v8f32(
 ; AVX512-NEXT:    [[TMP1:%.*]] = fmul <8 x float> [[A:%.*]], [[B:%.*]]
 ; AVX512-NEXT:    [[TMP2:%.*]] = fdiv <8 x float> [[A]], [[B]]
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/alternate-fp.ll b/llvm/test/Transforms/SLPVectorizer/X86/alternate-fp.ll
index 6275d984295c0..82127f276389f 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/alternate-fp.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/alternate-fp.ll
@@ -3,6 +3,7 @@
 ; RUN: opt < %s -mtriple=x86_64-unknown -mcpu=slm -passes=slp-vectorizer -S | FileCheck %s --check-prefix=SLM
 ; RUN: opt < %s -mtriple=x86_64-unknown -mcpu=corei7-avx -passes=slp-vectorizer -S | FileCheck %s --check-prefix=AVX
 ; RUN: opt < %s -mtriple=x86_64-unknown -mcpu=core-avx2 -passes=slp-vectorizer -S | FileCheck %s --check-prefix=AVX2
+; RUN: opt < %s -mtriple=x86_64-unknown -mattr=+avx2,+fast-vector-fdiv -passes=slp-vectorizer -S | FileCheck %s --check-prefix=AVX2-FAST-FDIV
 ; RUN: opt < %s -mtriple=x86_64-unknown -mcpu=knl -passes=slp-vectorizer -S | FileCheck %s --check-prefix=AVX512
 ; RUN: opt < %s -mtriple=x86_64-unknown -mcpu=skx -passes=slp-vectorizer -S | FileCheck %s --check-prefix=AVX512
 
@@ -146,6 +147,12 @@ define <8 x float> @fmul_fdiv_v8f32(<8 x float> %a, <8 x float> %b) {
 ; AVX2-NEXT:    [[TMP5:%.*]] = shufflevector <8 x float> [[TMP8]], <8 x float> poison, <8 x i32> <i32 0, i32 4, i32 5, i32 1, i32 2, i32 6, i32 7, i32 3>
 ; AVX2-NEXT:    ret <8 x float> [[TMP5]]
 ;
+; AVX2-FAST-FDIV-LABEL: @fmul_fdiv_v8f32(
+; AVX2-FAST-FDIV-NEXT:    [[TMP1:%.*]] = fmul <8 x float> [[A:%.*]], [[B:%.*]]
+; AVX2-FAST-FDIV-NEXT:    [[TMP2:%.*]] = fdiv <8 x float> [[A]], [[B]]
+; AVX2-FAST-FDIV-NEXT:    [[TMP3:%.*]] = shufflevector <8 x float> [[TMP1]], <8 x float> [[TMP2]], <8 x i32> <i32 0, i32 9, i32 10, i32 3, i32 4, i32 13, i32 14, i32 7>
+; AVX2-FAST-FDIV-NEXT:    ret <8 x float> [[TMP3]]
+;
 ; AVX512-LABEL: @fmul_fdiv_v8f32(
 ; AVX512-NEXT:    [[TMP1:%.*]] = fmul <8 x float> [[A:%.*]], [[B:%.*]]
 ; AVX512-NEXT:    [[TMP2:%.*]] = fdiv <8 x float> [[A]], [[B]]

>From 34e22710431bca94075c26e0ef9ea083d48cfc24 Mon Sep 17 00:00:00 2001
From: Xiaomeng Zhang <zhangxiaomeng at hygon.cn>
Date: Sat, 12 Sep 2026 15:37:07 +0800
Subject: [PATCH 2/2] [X86] Fix SizeAndLatencyCost for fast vector division

---
 llvm/lib/Target/X86/X86.td                     | 4 ++--
 llvm/lib/Target/X86/X86TargetTransformInfo.cpp | 4 ++--
 llvm/test/Analysis/CostModel/X86/arith-fp.ll   | 8 ++++----
 3 files changed, 8 insertions(+), 8 deletions(-)

diff --git a/llvm/lib/Target/X86/X86.td b/llvm/lib/Target/X86/X86.td
index e63f4fd43996e..ecfed086f1a61 100644
--- a/llvm/lib/Target/X86/X86.td
+++ b/llvm/lib/Target/X86/X86.td
@@ -741,8 +741,8 @@ def TuningFastVectorFSQRT
     : SubtargetFeature<"fast-vector-fsqrt", "HasFastVectorFSQRT",
                        "true", "Vector SQRT is fast (disable Newton-Raphson)",
                        [], InlineIgnore>;
-// True if 256-bit VDIVPS/VDIVPD instructions have no significant width penalty
-// compared with their 128-bit counterparts.
+// True if 256-bit VDIVPS/VDIVPD instructions are nearly as fast as their
+// 128-bit counterparts.
 def TuningFastVectorFDIV
     : SubtargetFeature<"fast-vector-fdiv", "HasFastVectorFDIV",
                        "true", "Vector FDIV is fast (enable fast costs)",
diff --git a/llvm/lib/Target/X86/X86TargetTransformInfo.cpp b/llvm/lib/Target/X86/X86TargetTransformInfo.cpp
index 862e2b81bf883..c60dacefa25cf 100644
--- a/llvm/lib/Target/X86/X86TargetTransformInfo.cpp
+++ b/llvm/lib/Target/X86/X86TargetTransformInfo.cpp
@@ -1607,8 +1607,8 @@ InstructionCost X86TTIImpl::getArithmeticInstrCost(
   // generic AVX2 table. This must be checked before the generic AVX2 lookup.
   if (ST->hasAVX2() && ST->hasFastVectorFDIV()) {
     static const CostKindTblEntry AVX2FastVectorFDIVCostTable[] = {
-      { ISD::FDIV, MVT::v8f32,   {  7, 13, 1, 3 } }, // vdivps
-      { ISD::FDIV, MVT::v4f64,   { 14, 20, 1, 3 } }, // vdivpd
+      { ISD::FDIV, MVT::v8f32,   {  7, 13, 1, 1 } }, // vdivps
+      { ISD::FDIV, MVT::v4f64,   { 14, 20, 1, 1 } }, // vdivpd
     };
     if (const auto *Entry =
             CostTableLookup(AVX2FastVectorFDIVCostTable, ISD, LT.second))
diff --git a/llvm/test/Analysis/CostModel/X86/arith-fp.ll b/llvm/test/Analysis/CostModel/X86/arith-fp.ll
index 139212ef42501..2d1bf5bed42aa 100644
--- a/llvm/test/Analysis/CostModel/X86/arith-fp.ll
+++ b/llvm/test/Analysis/CostModel/X86/arith-fp.ll
@@ -582,12 +582,12 @@ define i32 @fdiv(i32 %arg) {
 ; AVX2-FAST-FDIV-LABEL: 'fdiv'
 ; AVX2-FAST-FDIV-NEXT:  Cost Model: Found costs of RThru:7 CodeSize:1 Lat:13 SizeLat:1 for: %F32 = fdiv float undef, undef
 ; AVX2-FAST-FDIV-NEXT:  Cost Model: Found costs of RThru:7 CodeSize:1 Lat:13 SizeLat:1 for: %V4F32 = fdiv <4 x float> undef, undef
-; AVX2-FAST-FDIV-NEXT:  Cost Model: Found costs of RThru:7 CodeSize:1 Lat:13 SizeLat:3 for: %V8F32 = fdiv <8 x float> undef, undef
-; AVX2-FAST-FDIV-NEXT:  Cost Model: Found costs of RThru:14 CodeSize:2 Lat:26 SizeLat:6 for: %V16F32 = fdiv <16 x float> undef, undef
+; AVX2-FAST-FDIV-NEXT:  Cost Model: Found costs of RThru:7 CodeSize:1 Lat:13 SizeLat:1 for: %V8F32 = fdiv <8 x float> undef, undef
+; AVX2-FAST-FDIV-NEXT:  Cost Model: Found costs of RThru:14 CodeSize:2 Lat:26 SizeLat:2 for: %V16F32 = fdiv <16 x float> undef, undef
 ; AVX2-FAST-FDIV-NEXT:  Cost Model: Found costs of RThru:14 CodeSize:1 Lat:20 SizeLat:1 for: %F64 = fdiv double undef, undef
 ; AVX2-FAST-FDIV-NEXT:  Cost Model: Found costs of RThru:14 CodeSize:1 Lat:20 SizeLat:1 for: %V2F64 = fdiv <2 x double> undef, undef
-; AVX2-FAST-FDIV-NEXT:  Cost Model: Found costs of RThru:14 CodeSize:1 Lat:20 SizeLat:3 for: %V4F64 = fdiv <4 x double> undef, undef
-; AVX2-FAST-FDIV-NEXT:  Cost Model: Found costs of RThru:28 CodeSize:2 Lat:40 SizeLat:6 for: %V8F64 = fdiv <8 x double> undef, undef
+; AVX2-FAST-FDIV-NEXT:  Cost Model: Found costs of RThru:14 CodeSize:1 Lat:20 SizeLat:1 for: %V4F64 = fdiv <4 x double> undef, undef
+; AVX2-FAST-FDIV-NEXT:  Cost Model: Found costs of RThru:28 CodeSize:2 Lat:40 SizeLat:2 for: %V8F64 = fdiv <8 x double> undef, undef
 ; AVX2-FAST-FDIV-NEXT:  Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret i32 undef
 ;
 ; AVX512-LABEL: 'fdiv'



More information about the llvm-commits mailing list