[llvm] 4f32e47 - [AArch64] Parameterize repeated FP divisor combine threshold by subtarget (#216930)
via llvm-commits
llvm-commits at lists.llvm.org
Fri Sep 11 00:59:50 PDT 2026
Author: yamash-fj
Date: 2026-09-11T08:59:45+01:00
New Revision: 4f32e473933d4a039d13fe2c0a4ccdde73c8b929
URL: https://github.com/llvm/llvm-project/commit/4f32e473933d4a039d13fe2c0a4ccdde73c8b929
DIFF: https://github.com/llvm/llvm-project/commit/4f32e473933d4a039d13fe2c0a4ccdde73c8b929.diff
LOG: [AArch64] Parameterize repeated FP divisor combine threshold by subtarget (#216930)
Changes:
This patch makes the threshold for combining repeated FP divisors
configurable per AArch64 subtarget, instead of using a single fixed
value 3 for all CPUs. Additionaly, set this value 2 for A64FX based on
observed profitability for this transform on A64FX.
Motivation:
Today, AArch64 uses a fixed threshold of 3 for reciprocal transform.
This means the transform runs when there are 3 FDIVs with the same
divisor (not 2). That is conservative for some subtargets, and it can
miss profitable opportunities on CPUs such as A64FX, where a lower
threshold appears to produce better code generation.
```
a / D; b / D; ...
=>
recip = 1.0 / D; a * recip; b * recip; ...
```
Note:
It does not change the semantics of the reciprocal transform itself.
Added:
Modified:
llvm/lib/Target/AArch64/AArch64Features.td
llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
llvm/lib/Target/AArch64/AArch64Processors.td
llvm/test/CodeGen/AArch64/fdiv-combine.ll
Removed:
################################################################################
diff --git a/llvm/lib/Target/AArch64/AArch64Features.td b/llvm/lib/Target/AArch64/AArch64Features.td
index 6eea064b257f6..ad1561c281703 100644
--- a/llvm/lib/Target/AArch64/AArch64Features.td
+++ b/llvm/lib/Target/AArch64/AArch64Features.td
@@ -1022,6 +1022,11 @@ def FeatureLimited64bitVectorMulBandwidth : SubtargetFeature<
"Has limited 64bit vector multiply bandwidth compared to scalar multiply",
[], InlineIgnore>;
+def FeatureUseReciprocalFDivCombineThreshold2 : SubtargetFeature<
+ "use-reciprocal-fdiv-combine-threshold-2", "UseReciprocalFDivCombineThreshold2", "true",
+ "Set the reciprocal FDiv combine threshold to 2 (from the default 3)",
+ [], InlineIgnore>;
+
//===----------------------------------------------------------------------===//
// Architectures.
//
diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index a3823f8552a3a..7dbf2ed45bf2c 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -32919,8 +32919,9 @@ SDValue AArch64TargetLowering::emitStackGuardMixFP(SelectionDAG &DAG,
unsigned AArch64TargetLowering::combineRepeatedFPDivisors() const {
// Combine multiple FDIVs with the same divisor into multiple FMULs by the
- // reciprocal if there are three or more FDIVs.
- return 3;
+ // reciprocal if there are enough FDIVs. The threshold is determined by the
+ // subtarget feature.
+ return Subtarget->useReciprocalFDivCombineThreshold2() ? 2 : 3;
}
TargetLoweringBase::LegalizeTypeAction
diff --git a/llvm/lib/Target/AArch64/AArch64Processors.td b/llvm/lib/Target/AArch64/AArch64Processors.td
index f996522ba4d09..1ed06ee68db49 100644
--- a/llvm/lib/Target/AArch64/AArch64Processors.td
+++ b/llvm/lib/Target/AArch64/AArch64Processors.td
@@ -362,7 +362,8 @@ def TuneA64FX : SubtargetFeature<"a64fx", "ARMProcFamily", "A64FX",
FeatureStorePairSuppress,
FeaturePredictableSelectIsExpensive,
FeatureDisableUnpredicatedLdStLower,
- FeatureMaxInterleaveFactor4]>;
+ FeatureMaxInterleaveFactor4,
+ FeatureUseReciprocalFDivCombineThreshold2]>;
def TuneMONAKA : SubtargetFeature<"fujitsu-monaka", "ARMProcFamily", "MONAKA",
"Fujitsu FUJITSU-MONAKA processors", [
diff --git a/llvm/test/CodeGen/AArch64/fdiv-combine.ll b/llvm/test/CodeGen/AArch64/fdiv-combine.ll
index 9eacb61eecd06..fc488b5ff856c 100644
--- a/llvm/test/CodeGen/AArch64/fdiv-combine.ll
+++ b/llvm/test/CodeGen/AArch64/fdiv-combine.ll
@@ -103,6 +103,36 @@ define void @two_fdiv_double(double %D, double %a, double %b) {
ret void
}
+; Following test cases check we combine two FDIVs if
+; the target cpu sets the threshold to 2.
+define void @two_fdiv_float_combine(float %D, float %a, float %b) #1 {
+; CHECK-LABEL: two_fdiv_float_combine:
+; CHECK: // %bb.0:
+; CHECK-NEXT: fmov s3, #1.00000000
+; CHECK-NEXT: fdiv s3, s3, s0
+; CHECK-NEXT: fmul s0, s1, s3
+; CHECK-NEXT: fmul s1, s2, s3
+; CHECK-NEXT: b foo_2f
+ %div = fdiv arcp float %a, %D
+ %div1 = fdiv arcp float %b, %D
+ tail call void @foo_2f(float %div, float %div1)
+ ret void
+}
+
+define void @two_fdiv_double_combine(double %D, double %a, double %b) #1 {
+; CHECK-LABEL: two_fdiv_double_combine:
+; CHECK: // %bb.0:
+; CHECK-NEXT: fmov d3, #1.00000000
+; CHECK-NEXT: fdiv d3, d3, d0
+; CHECK-NEXT: fmul d0, d1, d3
+; CHECK-NEXT: fmul d1, d2, d3
+; CHECK-NEXT: b foo_2d
+ %div = fdiv arcp double %a, %D
+ %div1 = fdiv arcp double %b, %D
+ tail call void @foo_2d(double %div, double %div1)
+ ret void
+}
+
define void @four_fdiv_multi_float(float %D, float %a, float %b, float %c) #0 {
; CHECK-SD-LABEL: four_fdiv_multi_float:
; CHECK-SD: // %bb.0:
@@ -255,3 +285,4 @@ declare void @foo_3_nxv4f32(<vscale x 4 x float>, <vscale x 4 x float>, <vscale
declare void @foo_2_nxv2f64(<vscale x 2 x double>, <vscale x 2 x double>)
attributes #0 = { "target-features"="+sve" }
+attributes #1 = { "target-cpu"="a64fx" }
More information about the llvm-commits
mailing list