[llvm] 4f32e47 - [AArch64] Parameterize repeated FP divisor combine threshold by subtarget (#216930)

via llvm-commits llvm-commits at lists.llvm.org
Fri Sep 11 00:59:50 PDT 2026


Author: yamash-fj
Date: 2026-09-11T08:59:45+01:00
New Revision: 4f32e473933d4a039d13fe2c0a4ccdde73c8b929

URL: https://github.com/llvm/llvm-project/commit/4f32e473933d4a039d13fe2c0a4ccdde73c8b929
DIFF: https://github.com/llvm/llvm-project/commit/4f32e473933d4a039d13fe2c0a4ccdde73c8b929.diff

LOG: [AArch64]  Parameterize repeated FP divisor combine threshold by subtarget (#216930)

Changes:
This patch makes the threshold for combining repeated FP divisors
configurable per AArch64 subtarget, instead of using a single fixed
value 3 for all CPUs. Additionaly, set this value 2 for A64FX based on
observed profitability for this transform on A64FX.

Motivation:
Today, AArch64 uses a fixed threshold of 3 for reciprocal transform.
This means the transform runs when there are 3 FDIVs with the same
divisor (not 2). That is conservative for some subtargets, and it can
miss profitable opportunities on CPUs such as A64FX, where a lower
threshold appears to produce better code generation.
```
a / D; b / D; ...
=>
recip = 1.0 / D; a * recip; b * recip; ...
```

Note:
It does not change the semantics of the reciprocal transform itself.

Added: 
    

Modified: 
    llvm/lib/Target/AArch64/AArch64Features.td
    llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
    llvm/lib/Target/AArch64/AArch64Processors.td
    llvm/test/CodeGen/AArch64/fdiv-combine.ll

Removed: 
    


################################################################################
diff  --git a/llvm/lib/Target/AArch64/AArch64Features.td b/llvm/lib/Target/AArch64/AArch64Features.td
index 6eea064b257f6..ad1561c281703 100644
--- a/llvm/lib/Target/AArch64/AArch64Features.td
+++ b/llvm/lib/Target/AArch64/AArch64Features.td
@@ -1022,6 +1022,11 @@ def FeatureLimited64bitVectorMulBandwidth : SubtargetFeature<
     "Has limited 64bit vector multiply bandwidth compared to scalar multiply",
     [], InlineIgnore>;
 
+def FeatureUseReciprocalFDivCombineThreshold2 : SubtargetFeature<
+    "use-reciprocal-fdiv-combine-threshold-2", "UseReciprocalFDivCombineThreshold2", "true",
+    "Set the reciprocal FDiv combine threshold to 2 (from the default 3)",
+    [], InlineIgnore>;
+
 //===----------------------------------------------------------------------===//
 // Architectures.
 //

diff  --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index a3823f8552a3a..7dbf2ed45bf2c 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -32919,8 +32919,9 @@ SDValue AArch64TargetLowering::emitStackGuardMixFP(SelectionDAG &DAG,
 
 unsigned AArch64TargetLowering::combineRepeatedFPDivisors() const {
   // Combine multiple FDIVs with the same divisor into multiple FMULs by the
-  // reciprocal if there are three or more FDIVs.
-  return 3;
+  // reciprocal if there are enough FDIVs. The threshold is determined by the
+  // subtarget feature.
+  return Subtarget->useReciprocalFDivCombineThreshold2() ? 2 : 3;
 }
 
 TargetLoweringBase::LegalizeTypeAction

diff  --git a/llvm/lib/Target/AArch64/AArch64Processors.td b/llvm/lib/Target/AArch64/AArch64Processors.td
index f996522ba4d09..1ed06ee68db49 100644
--- a/llvm/lib/Target/AArch64/AArch64Processors.td
+++ b/llvm/lib/Target/AArch64/AArch64Processors.td
@@ -362,7 +362,8 @@ def TuneA64FX : SubtargetFeature<"a64fx", "ARMProcFamily", "A64FX",
                                  FeatureStorePairSuppress,
                                  FeaturePredictableSelectIsExpensive,
                                  FeatureDisableUnpredicatedLdStLower,
-                                 FeatureMaxInterleaveFactor4]>;
+                                 FeatureMaxInterleaveFactor4,
+                                 FeatureUseReciprocalFDivCombineThreshold2]>;
 
 def TuneMONAKA : SubtargetFeature<"fujitsu-monaka", "ARMProcFamily", "MONAKA",
                                  "Fujitsu FUJITSU-MONAKA processors", [

diff  --git a/llvm/test/CodeGen/AArch64/fdiv-combine.ll b/llvm/test/CodeGen/AArch64/fdiv-combine.ll
index 9eacb61eecd06..fc488b5ff856c 100644
--- a/llvm/test/CodeGen/AArch64/fdiv-combine.ll
+++ b/llvm/test/CodeGen/AArch64/fdiv-combine.ll
@@ -103,6 +103,36 @@ define void @two_fdiv_double(double %D, double %a, double %b) {
   ret void
 }
 
+; Following test cases check we combine two FDIVs if
+; the target cpu sets the threshold to 2.
+define void @two_fdiv_float_combine(float %D, float %a, float %b) #1 {
+; CHECK-LABEL: two_fdiv_float_combine:
+; CHECK:       // %bb.0:
+; CHECK-NEXT:    fmov s3, #1.00000000
+; CHECK-NEXT:    fdiv s3, s3, s0
+; CHECK-NEXT:    fmul s0, s1, s3
+; CHECK-NEXT:    fmul s1, s2, s3
+; CHECK-NEXT:    b foo_2f
+  %div = fdiv arcp float %a, %D
+  %div1 = fdiv arcp float %b, %D
+  tail call void @foo_2f(float %div, float %div1)
+  ret void
+}
+
+define void @two_fdiv_double_combine(double %D, double %a, double %b) #1 {
+; CHECK-LABEL: two_fdiv_double_combine:
+; CHECK:       // %bb.0:
+; CHECK-NEXT:    fmov d3, #1.00000000
+; CHECK-NEXT:    fdiv d3, d3, d0
+; CHECK-NEXT:    fmul d0, d1, d3
+; CHECK-NEXT:    fmul d1, d2, d3
+; CHECK-NEXT:    b foo_2d
+  %div = fdiv arcp double %a, %D
+  %div1 = fdiv arcp double %b, %D
+  tail call void @foo_2d(double %div, double %div1)
+  ret void
+}
+
 define void @four_fdiv_multi_float(float %D, float %a, float %b, float %c) #0 {
 ; CHECK-SD-LABEL: four_fdiv_multi_float:
 ; CHECK-SD:       // %bb.0:
@@ -255,3 +285,4 @@ declare void @foo_3_nxv4f32(<vscale x 4 x float>, <vscale x 4 x float>, <vscale
 declare void @foo_2_nxv2f64(<vscale x 2 x double>, <vscale x 2 x double>)
 
 attributes #0 = { "target-features"="+sve" }
+attributes #1 = { "target-cpu"="a64fx" }


        


More information about the llvm-commits mailing list