[llvm] [X86] Add tuning for fast AVX2 vector division (PR #219873)
Xiaomeng Zhang via llvm-commits
llvm-commits at lists.llvm.org
Sat Sep 12 01:21:27 PDT 2026
https://github.com/JacketPants updated https://github.com/llvm/llvm-project/pull/219873
>From a38e7ed6617c0b43c2033bd3f82294b656c9ef4d Mon Sep 17 00:00:00 2001
From: Xiaomeng Zhang <zhangxiaomeng at hygon.cn>
Date: Thu, 27 Aug 2026 17:40:31 +0800
Subject: [PATCH 1/2] [X86] Add tuning for fast AVX2 vector division
On some microarchitectures, 256-bit vector division has no significant
performance difference compared with 128-bit vector division. Add
TuningFastVectorFDIV to adjust the TTI cost estimates for these targets.
---
llvm/lib/Target/X86/X86.td | 7 +++++++
llvm/lib/Target/X86/X86TargetTransformInfo.cpp | 13 +++++++++++++
llvm/test/Analysis/CostModel/X86/arith-fp.ll | 13 +++++++++++++
.../SLPVectorizer/X86/alternate-fp-inseltpoison.ll | 7 +++++++
.../Transforms/SLPVectorizer/X86/alternate-fp.ll | 7 +++++++
5 files changed, 47 insertions(+)
diff --git a/llvm/lib/Target/X86/X86.td b/llvm/lib/Target/X86/X86.td
index 5786c659b0f98..e63f4fd43996e 100644
--- a/llvm/lib/Target/X86/X86.td
+++ b/llvm/lib/Target/X86/X86.td
@@ -741,6 +741,12 @@ def TuningFastVectorFSQRT
: SubtargetFeature<"fast-vector-fsqrt", "HasFastVectorFSQRT",
"true", "Vector SQRT is fast (disable Newton-Raphson)",
[], InlineIgnore>;
+// True if 256-bit VDIVPS/VDIVPD instructions have no significant width penalty
+// compared with their 128-bit counterparts.
+def TuningFastVectorFDIV
+ : SubtargetFeature<"fast-vector-fdiv", "HasFastVectorFDIV",
+ "true", "Vector FDIV is fast (enable fast costs)",
+ [], InlineIgnore>;
// If lzcnt has equivalent latency/throughput to most simple integer ops, it can
// be used to replace test/set sequences.
@@ -1827,6 +1833,7 @@ def ProcessorFeatures {
TuningFast15ByteNOP,
TuningFastScalarFSQRT,
TuningFastVectorFSQRT,
+ TuningFastVectorFDIV,
TuningFastScalarShiftMasks,
TuningFastVariablePerLaneShuffle,
TuningFastMOVBE,
diff --git a/llvm/lib/Target/X86/X86TargetTransformInfo.cpp b/llvm/lib/Target/X86/X86TargetTransformInfo.cpp
index 8e0cf1fc5a153..8a4ac4af21058 100644
--- a/llvm/lib/Target/X86/X86TargetTransformInfo.cpp
+++ b/llvm/lib/Target/X86/X86TargetTransformInfo.cpp
@@ -1253,6 +1253,19 @@ InstructionCost X86TTIImpl::getArithmeticInstrCost(
{ ISD::FDIV, MVT::v4f64, { 28, 35, 1, 3 } }, // vdivpd
};
+ // Targets with fast 256-bit vector division use lower costs than the
+ // generic AVX2 table. This must be checked before the generic AVX2 lookup.
+ if (ST->hasAVX2() && ST->hasFastVectorFDIV()) {
+ static const CostKindTblEntry AVX2FastVectorFDIVCostTable[] = {
+ { ISD::FDIV, MVT::v8f32, { 7, 13, 1, 3 } }, // vdivps
+ { ISD::FDIV, MVT::v4f64, { 14, 20, 1, 3 } }, // vdivpd
+ };
+ if (const auto *Entry =
+ CostTableLookup(AVX2FastVectorFDIVCostTable, ISD, LT.second))
+ if (auto KindCost = Entry->Cost[CostKind])
+ return LT.first * *KindCost;
+ }
+
// Look for AVX2 lowering tricks for custom cases.
if (ST->hasAVX2())
if (const auto *Entry = CostTableLookup(AVX2CostTable, ISD, LT.second))
diff --git a/llvm/test/Analysis/CostModel/X86/arith-fp.ll b/llvm/test/Analysis/CostModel/X86/arith-fp.ll
index e3eb3e60b844c..139212ef42501 100644
--- a/llvm/test/Analysis/CostModel/X86/arith-fp.ll
+++ b/llvm/test/Analysis/CostModel/X86/arith-fp.ll
@@ -4,6 +4,8 @@
; RUN: opt < %s -passes="print<cost-model>" 2>&1 -disable-output -cost-kind=all -mtriple=x86_64-- -mattr=+sse4.2 | FileCheck %s --check-prefixes=SSE42
; RUN: opt < %s -passes="print<cost-model>" 2>&1 -disable-output -cost-kind=all -mtriple=x86_64-- -mattr=+avx | FileCheck %s --check-prefixes=AVX,AVX1
; RUN: opt < %s -passes="print<cost-model>" 2>&1 -disable-output -cost-kind=all -mtriple=x86_64-- -mattr=+avx2 | FileCheck %s --check-prefixes=AVX,AVX2
+; RUN: opt < %s -passes="print<cost-model>" 2>&1 -disable-output -cost-kind=all -mtriple=x86_64-- -mattr=+avx2,+fast-vector-fdiv | FileCheck %s --check-prefixes=AVX2-FAST-FDIV
+; RUN: opt < %s -passes="print<cost-model>" 2>&1 -disable-output -cost-kind=all -mtriple=x86_64-- -mcpu=c86-4g-m4 | FileCheck %s --check-prefixes=AVX2-FAST-FDIV
; RUN: opt < %s -passes="print<cost-model>" 2>&1 -disable-output -cost-kind=all -mtriple=x86_64-- -mattr=+avx512f | FileCheck %s --check-prefixes=AVX512
; RUN: opt < %s -passes="print<cost-model>" 2>&1 -disable-output -cost-kind=all -mtriple=x86_64-- -mattr=+avx512f,+avx512bw | FileCheck %s --check-prefixes=AVX512
;
@@ -577,6 +579,17 @@ define i32 @fdiv(i32 %arg) {
; AVX2-NEXT: Cost Model: Found costs of RThru:56 CodeSize:2 Lat:70 SizeLat:6 for: %V8F64 = fdiv <8 x double> undef, undef
; AVX2-NEXT: Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret i32 undef
;
+; AVX2-FAST-FDIV-LABEL: 'fdiv'
+; AVX2-FAST-FDIV-NEXT: Cost Model: Found costs of RThru:7 CodeSize:1 Lat:13 SizeLat:1 for: %F32 = fdiv float undef, undef
+; AVX2-FAST-FDIV-NEXT: Cost Model: Found costs of RThru:7 CodeSize:1 Lat:13 SizeLat:1 for: %V4F32 = fdiv <4 x float> undef, undef
+; AVX2-FAST-FDIV-NEXT: Cost Model: Found costs of RThru:7 CodeSize:1 Lat:13 SizeLat:3 for: %V8F32 = fdiv <8 x float> undef, undef
+; AVX2-FAST-FDIV-NEXT: Cost Model: Found costs of RThru:14 CodeSize:2 Lat:26 SizeLat:6 for: %V16F32 = fdiv <16 x float> undef, undef
+; AVX2-FAST-FDIV-NEXT: Cost Model: Found costs of RThru:14 CodeSize:1 Lat:20 SizeLat:1 for: %F64 = fdiv double undef, undef
+; AVX2-FAST-FDIV-NEXT: Cost Model: Found costs of RThru:14 CodeSize:1 Lat:20 SizeLat:1 for: %V2F64 = fdiv <2 x double> undef, undef
+; AVX2-FAST-FDIV-NEXT: Cost Model: Found costs of RThru:14 CodeSize:1 Lat:20 SizeLat:3 for: %V4F64 = fdiv <4 x double> undef, undef
+; AVX2-FAST-FDIV-NEXT: Cost Model: Found costs of RThru:28 CodeSize:2 Lat:40 SizeLat:6 for: %V8F64 = fdiv <8 x double> undef, undef
+; AVX2-FAST-FDIV-NEXT: Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret i32 undef
+;
; AVX512-LABEL: 'fdiv'
; AVX512-NEXT: Cost Model: Found costs of RThru:3 CodeSize:1 Lat:11 SizeLat:1 for: %F32 = fdiv float undef, undef
; AVX512-NEXT: Cost Model: Found costs of RThru:3 CodeSize:1 Lat:11 SizeLat:1 for: %V4F32 = fdiv <4 x float> undef, undef
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/alternate-fp-inseltpoison.ll b/llvm/test/Transforms/SLPVectorizer/X86/alternate-fp-inseltpoison.ll
index 06498563a7d37..9a0f9cda167ae 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/alternate-fp-inseltpoison.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/alternate-fp-inseltpoison.ll
@@ -3,6 +3,7 @@
; RUN: opt < %s -mtriple=x86_64-unknown -mcpu=slm -passes=slp-vectorizer -S | FileCheck %s --check-prefix=SLM
; RUN: opt < %s -mtriple=x86_64-unknown -mcpu=corei7-avx -passes=slp-vectorizer -S | FileCheck %s --check-prefix=AVX
; RUN: opt < %s -mtriple=x86_64-unknown -mcpu=core-avx2 -passes=slp-vectorizer -S | FileCheck %s --check-prefix=AVX2
+; RUN: opt < %s -mtriple=x86_64-unknown -mattr=+avx2,+fast-vector-fdiv -passes=slp-vectorizer -S | FileCheck %s --check-prefix=AVX2-FAST-FDIV
; RUN: opt < %s -mtriple=x86_64-unknown -mcpu=knl -passes=slp-vectorizer -S | FileCheck %s --check-prefix=AVX512
; RUN: opt < %s -mtriple=x86_64-unknown -mcpu=skx -passes=slp-vectorizer -S | FileCheck %s --check-prefix=AVX512
@@ -146,6 +147,12 @@ define <8 x float> @fmul_fdiv_v8f32(<8 x float> %a, <8 x float> %b) {
; AVX2-NEXT: [[TMP5:%.*]] = shufflevector <8 x float> [[TMP8]], <8 x float> poison, <8 x i32> <i32 0, i32 4, i32 5, i32 1, i32 2, i32 6, i32 7, i32 3>
; AVX2-NEXT: ret <8 x float> [[TMP5]]
;
+; AVX2-FAST-FDIV-LABEL: @fmul_fdiv_v8f32(
+; AVX2-FAST-FDIV-NEXT: [[TMP1:%.*]] = fmul <8 x float> [[A:%.*]], [[B:%.*]]
+; AVX2-FAST-FDIV-NEXT: [[TMP2:%.*]] = fdiv <8 x float> [[A]], [[B]]
+; AVX2-FAST-FDIV-NEXT: [[TMP3:%.*]] = shufflevector <8 x float> [[TMP1]], <8 x float> [[TMP2]], <8 x i32> <i32 0, i32 9, i32 10, i32 3, i32 4, i32 13, i32 14, i32 7>
+; AVX2-FAST-FDIV-NEXT: ret <8 x float> [[TMP3]]
+;
; AVX512-LABEL: @fmul_fdiv_v8f32(
; AVX512-NEXT: [[TMP1:%.*]] = fmul <8 x float> [[A:%.*]], [[B:%.*]]
; AVX512-NEXT: [[TMP2:%.*]] = fdiv <8 x float> [[A]], [[B]]
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/alternate-fp.ll b/llvm/test/Transforms/SLPVectorizer/X86/alternate-fp.ll
index 6275d984295c0..82127f276389f 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/alternate-fp.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/alternate-fp.ll
@@ -3,6 +3,7 @@
; RUN: opt < %s -mtriple=x86_64-unknown -mcpu=slm -passes=slp-vectorizer -S | FileCheck %s --check-prefix=SLM
; RUN: opt < %s -mtriple=x86_64-unknown -mcpu=corei7-avx -passes=slp-vectorizer -S | FileCheck %s --check-prefix=AVX
; RUN: opt < %s -mtriple=x86_64-unknown -mcpu=core-avx2 -passes=slp-vectorizer -S | FileCheck %s --check-prefix=AVX2
+; RUN: opt < %s -mtriple=x86_64-unknown -mattr=+avx2,+fast-vector-fdiv -passes=slp-vectorizer -S | FileCheck %s --check-prefix=AVX2-FAST-FDIV
; RUN: opt < %s -mtriple=x86_64-unknown -mcpu=knl -passes=slp-vectorizer -S | FileCheck %s --check-prefix=AVX512
; RUN: opt < %s -mtriple=x86_64-unknown -mcpu=skx -passes=slp-vectorizer -S | FileCheck %s --check-prefix=AVX512
@@ -146,6 +147,12 @@ define <8 x float> @fmul_fdiv_v8f32(<8 x float> %a, <8 x float> %b) {
; AVX2-NEXT: [[TMP5:%.*]] = shufflevector <8 x float> [[TMP8]], <8 x float> poison, <8 x i32> <i32 0, i32 4, i32 5, i32 1, i32 2, i32 6, i32 7, i32 3>
; AVX2-NEXT: ret <8 x float> [[TMP5]]
;
+; AVX2-FAST-FDIV-LABEL: @fmul_fdiv_v8f32(
+; AVX2-FAST-FDIV-NEXT: [[TMP1:%.*]] = fmul <8 x float> [[A:%.*]], [[B:%.*]]
+; AVX2-FAST-FDIV-NEXT: [[TMP2:%.*]] = fdiv <8 x float> [[A]], [[B]]
+; AVX2-FAST-FDIV-NEXT: [[TMP3:%.*]] = shufflevector <8 x float> [[TMP1]], <8 x float> [[TMP2]], <8 x i32> <i32 0, i32 9, i32 10, i32 3, i32 4, i32 13, i32 14, i32 7>
+; AVX2-FAST-FDIV-NEXT: ret <8 x float> [[TMP3]]
+;
; AVX512-LABEL: @fmul_fdiv_v8f32(
; AVX512-NEXT: [[TMP1:%.*]] = fmul <8 x float> [[A:%.*]], [[B:%.*]]
; AVX512-NEXT: [[TMP2:%.*]] = fdiv <8 x float> [[A]], [[B]]
>From 34e22710431bca94075c26e0ef9ea083d48cfc24 Mon Sep 17 00:00:00 2001
From: Xiaomeng Zhang <zhangxiaomeng at hygon.cn>
Date: Sat, 12 Sep 2026 15:37:07 +0800
Subject: [PATCH 2/2] [X86] Fix SizeAndLatencyCost for fast vector division
---
llvm/lib/Target/X86/X86.td | 4 ++--
llvm/lib/Target/X86/X86TargetTransformInfo.cpp | 4 ++--
llvm/test/Analysis/CostModel/X86/arith-fp.ll | 8 ++++----
3 files changed, 8 insertions(+), 8 deletions(-)
diff --git a/llvm/lib/Target/X86/X86.td b/llvm/lib/Target/X86/X86.td
index e63f4fd43996e..ecfed086f1a61 100644
--- a/llvm/lib/Target/X86/X86.td
+++ b/llvm/lib/Target/X86/X86.td
@@ -741,8 +741,8 @@ def TuningFastVectorFSQRT
: SubtargetFeature<"fast-vector-fsqrt", "HasFastVectorFSQRT",
"true", "Vector SQRT is fast (disable Newton-Raphson)",
[], InlineIgnore>;
-// True if 256-bit VDIVPS/VDIVPD instructions have no significant width penalty
-// compared with their 128-bit counterparts.
+// True if 256-bit VDIVPS/VDIVPD instructions are nearly as fast as their
+// 128-bit counterparts.
def TuningFastVectorFDIV
: SubtargetFeature<"fast-vector-fdiv", "HasFastVectorFDIV",
"true", "Vector FDIV is fast (enable fast costs)",
diff --git a/llvm/lib/Target/X86/X86TargetTransformInfo.cpp b/llvm/lib/Target/X86/X86TargetTransformInfo.cpp
index 862e2b81bf883..c60dacefa25cf 100644
--- a/llvm/lib/Target/X86/X86TargetTransformInfo.cpp
+++ b/llvm/lib/Target/X86/X86TargetTransformInfo.cpp
@@ -1607,8 +1607,8 @@ InstructionCost X86TTIImpl::getArithmeticInstrCost(
// generic AVX2 table. This must be checked before the generic AVX2 lookup.
if (ST->hasAVX2() && ST->hasFastVectorFDIV()) {
static const CostKindTblEntry AVX2FastVectorFDIVCostTable[] = {
- { ISD::FDIV, MVT::v8f32, { 7, 13, 1, 3 } }, // vdivps
- { ISD::FDIV, MVT::v4f64, { 14, 20, 1, 3 } }, // vdivpd
+ { ISD::FDIV, MVT::v8f32, { 7, 13, 1, 1 } }, // vdivps
+ { ISD::FDIV, MVT::v4f64, { 14, 20, 1, 1 } }, // vdivpd
};
if (const auto *Entry =
CostTableLookup(AVX2FastVectorFDIVCostTable, ISD, LT.second))
diff --git a/llvm/test/Analysis/CostModel/X86/arith-fp.ll b/llvm/test/Analysis/CostModel/X86/arith-fp.ll
index 139212ef42501..2d1bf5bed42aa 100644
--- a/llvm/test/Analysis/CostModel/X86/arith-fp.ll
+++ b/llvm/test/Analysis/CostModel/X86/arith-fp.ll
@@ -582,12 +582,12 @@ define i32 @fdiv(i32 %arg) {
; AVX2-FAST-FDIV-LABEL: 'fdiv'
; AVX2-FAST-FDIV-NEXT: Cost Model: Found costs of RThru:7 CodeSize:1 Lat:13 SizeLat:1 for: %F32 = fdiv float undef, undef
; AVX2-FAST-FDIV-NEXT: Cost Model: Found costs of RThru:7 CodeSize:1 Lat:13 SizeLat:1 for: %V4F32 = fdiv <4 x float> undef, undef
-; AVX2-FAST-FDIV-NEXT: Cost Model: Found costs of RThru:7 CodeSize:1 Lat:13 SizeLat:3 for: %V8F32 = fdiv <8 x float> undef, undef
-; AVX2-FAST-FDIV-NEXT: Cost Model: Found costs of RThru:14 CodeSize:2 Lat:26 SizeLat:6 for: %V16F32 = fdiv <16 x float> undef, undef
+; AVX2-FAST-FDIV-NEXT: Cost Model: Found costs of RThru:7 CodeSize:1 Lat:13 SizeLat:1 for: %V8F32 = fdiv <8 x float> undef, undef
+; AVX2-FAST-FDIV-NEXT: Cost Model: Found costs of RThru:14 CodeSize:2 Lat:26 SizeLat:2 for: %V16F32 = fdiv <16 x float> undef, undef
; AVX2-FAST-FDIV-NEXT: Cost Model: Found costs of RThru:14 CodeSize:1 Lat:20 SizeLat:1 for: %F64 = fdiv double undef, undef
; AVX2-FAST-FDIV-NEXT: Cost Model: Found costs of RThru:14 CodeSize:1 Lat:20 SizeLat:1 for: %V2F64 = fdiv <2 x double> undef, undef
-; AVX2-FAST-FDIV-NEXT: Cost Model: Found costs of RThru:14 CodeSize:1 Lat:20 SizeLat:3 for: %V4F64 = fdiv <4 x double> undef, undef
-; AVX2-FAST-FDIV-NEXT: Cost Model: Found costs of RThru:28 CodeSize:2 Lat:40 SizeLat:6 for: %V8F64 = fdiv <8 x double> undef, undef
+; AVX2-FAST-FDIV-NEXT: Cost Model: Found costs of RThru:14 CodeSize:1 Lat:20 SizeLat:1 for: %V4F64 = fdiv <4 x double> undef, undef
+; AVX2-FAST-FDIV-NEXT: Cost Model: Found costs of RThru:28 CodeSize:2 Lat:40 SizeLat:2 for: %V8F64 = fdiv <8 x double> undef, undef
; AVX2-FAST-FDIV-NEXT: Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret i32 undef
;
; AVX512-LABEL: 'fdiv'
More information about the llvm-commits
mailing list