[llvm] dd4b3b7 - [AMDGPU] Price only the fmul that fuses into an fadd/fsub as free (#226009)

via llvm-commits llvm-commits at lists.llvm.org
Thu Sep 24 07:46:29 PDT 2026


Author: Gheorghe-Teodor Bercea
Date: 2026-09-24T10:46:21-04:00
New Revision: dd4b3b7f0880c5bd15cea1867f0c4b30468fdd0c

URL: https://github.com/llvm/llvm-project/commit/dd4b3b7f0880c5bd15cea1867f0c4b30468fdd0c
DIFF: https://github.com/llvm/llvm-project/commit/dd4b3b7f0880c5bd15cea1867f0c4b30468fdd0c.diff

LOG: [AMDGPU] Price only the fmul that fuses into an fadd/fsub as free (#226009)

An fadd/fsub with two fmul operands can fuse with only one of them into
an FMA. Currently they are both priced as if they are fused (i.e. free).

Added: 
    llvm/test/Transforms/SLPVectorizer/AMDGPU/complex-mul-fma.ll

Modified: 
    llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
    llvm/test/Analysis/CostModel/AMDGPU/fused_costs.ll

Removed: 
    


################################################################################
diff  --git a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
index ccb0c7314dcef..f5119764bbb9d 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
@@ -546,6 +546,34 @@ static bool canFuseFMulWithFAddSub(const SITargetLowering &TLI, Type *Ty,
   return HasFMAD || (FAddSub->hasAllowContract() && FMul->hasAllowContract());
 }
 
+/// An fma holds one multiply, so only one fmul operand fuses with \p FAddSub.
+static const Instruction *getFusedFMul(const SITargetLowering &TLI, Type *Ty,
+                                       const Instruction *FAddSub) {
+  for (const Value *Op : FAddSub->operands()) {
+    const auto *FMul = dyn_cast<Instruction>(Op);
+    if (FMul && FMul->getOpcode() == Instruction::FMul && FMul->hasOneUse() &&
+        canFuseFMulWithFAddSub(TLI, Ty, FMul, FAddSub))
+      return FMul;
+  }
+  return nullptr;
+}
+
+static bool isFusedFMul(const SITargetLowering &TLI, Type *Ty,
+                        const Instruction *FMul, const Instruction *FAddSub) {
+  const Instruction *Fused = getFusedFMul(TLI, Ty, FAddSub);
+  if (Fused == FMul)
+    return true;
+  // (a * b + c * d) + e becomes fma(a, b, fma(c, d, e)) if the outer fadd has
+  // reassoc.
+  if (!Fused || FAddSub->getOpcode() != Instruction::FAdd ||
+      !FAddSub->hasOneUse())
+    return false;
+  const auto *Outer = dyn_cast<BinaryOperator>(*FAddSub->user_begin());
+  return Outer && Outer->getOpcode() == Instruction::FAdd &&
+         Outer->hasAllowReassoc() &&
+         canFuseFMulWithFAddSub(TLI, Ty, FMul, Outer);
+}
+
 InstructionCost GCNTTIImpl::getArithmeticInstrCost(
     unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind,
     TTI::OperandValueInfo Op1Info, TTI::OperandValueInfo Op2Info,
@@ -613,7 +641,7 @@ InstructionCost GCNTTIImpl::getArithmeticInstrCost(
       if (FAddSub &&
           (FAddSub->getOpcode() == Instruction::FAdd ||
            FAddSub->getOpcode() == Instruction::FSub) &&
-          canFuseFMulWithFAddSub(*TLI, Ty, CxtI, FAddSub))
+          isFusedFMul(*TLI, Ty, CxtI, FAddSub))
         return TargetTransformInfo::TCC_Free;
     }
     [[fallthrough]];
@@ -1533,17 +1561,9 @@ bool GCNTTIImpl::isProfitableToSinkOperands(Instruction *I,
   // so this stays a move.
   if (I->getOpcode() == Instruction::FAdd ||
       I->getOpcode() == Instruction::FSub) {
-    for (Use &Op : I->operands()) {
-      auto *FMul = dyn_cast<Instruction>(Op.get());
-      if (!FMul || FMul->getOpcode() != Instruction::FMul ||
-          !FMul->hasOneUse() ||
-          !canFuseFMulWithFAddSub(*TLI, I->getType(), FMul, I))
-        continue;
-      // The fused operand. Sink it when it sits in another block, then stop.
-      if (FMul->getParent() != I->getParent())
-        Ops.push_back(&Op);
-      break;
-    }
+    const Instruction *FMul = getFusedFMul(*TLI, I->getType(), I);
+    if (FMul && FMul->getParent() != I->getParent())
+      Ops.push_back(&I->getOperandUse(I->getOperand(0) == FMul ? 0 : 1));
   }
 
   for (auto &Op : I->operands()) {

diff  --git a/llvm/test/Analysis/CostModel/AMDGPU/fused_costs.ll b/llvm/test/Analysis/CostModel/AMDGPU/fused_costs.ll
index aeec6fd916514..3060d103bb1df 100644
--- a/llvm/test/Analysis/CostModel/AMDGPU/fused_costs.ll
+++ b/llvm/test/Analysis/CostModel/AMDGPU/fused_costs.ll
@@ -251,6 +251,112 @@ define void @fmul_fadd_f64(double %a, double %b, double %c, <2 x double> %va, <2
   ret void
 }
 
+define void @fmul_fmul_fadd_f32(float %a, float %b, float %c, float %d) #0 {
+; SLOWF32-LABEL: 'fmul_fmul_fadd_f32'
+; SLOWF32-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %add.m0 = fmul contract float %a, %b
+; SLOWF32-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %add.m1 = fmul contract float %c, %d
+; SLOWF32-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %add = fadd contract float %add.m0, %add.m1
+; SLOWF32-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %sub.m0 = fmul contract float %a, %b
+; SLOWF32-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %sub.m1 = fmul contract float %c, %d
+; SLOWF32-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %sub = fsub contract float %sub.m0, %sub.m1
+; SLOWF32-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %multi.m0 = fmul contract float %a, %b
+; SLOWF32-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %multi.m1 = fmul contract float %c, %d
+; SLOWF32-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %multi = fadd contract float %multi.m0, %multi.m1
+; SLOWF32-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %multi.use = fadd contract float %multi.m0, %c
+; SLOWF32-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %nc.m0 = fmul float %a, %b
+; SLOWF32-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %nc.m1 = fmul contract float %c, %d
+; SLOWF32-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %nc = fadd contract float %nc.m0, %nc.m1
+; SLOWF32-NEXT:  Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; FASTF32-LABEL: 'fmul_fmul_fadd_f32'
+; FASTF32-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %add.m0 = fmul contract float %a, %b
+; FASTF32-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %add.m1 = fmul contract float %c, %d
+; FASTF32-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %add = fadd contract float %add.m0, %add.m1
+; FASTF32-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %sub.m0 = fmul contract float %a, %b
+; FASTF32-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %sub.m1 = fmul contract float %c, %d
+; FASTF32-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %sub = fsub contract float %sub.m0, %sub.m1
+; FASTF32-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %multi.m0 = fmul contract float %a, %b
+; FASTF32-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %multi.m1 = fmul contract float %c, %d
+; FASTF32-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %multi = fadd contract float %multi.m0, %multi.m1
+; FASTF32-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %multi.use = fadd contract float %multi.m0, %c
+; FASTF32-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %nc.m0 = fmul float %a, %b
+; FASTF32-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %nc.m1 = fmul contract float %c, %d
+; FASTF32-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %nc = fadd contract float %nc.m0, %nc.m1
+; FASTF32-NEXT:  Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; SLOWF32-SIZE-LABEL: 'fmul_fmul_fadd_f32'
+; SLOWF32-SIZE-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %add.m0 = fmul contract float %a, %b
+; SLOWF32-SIZE-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %add.m1 = fmul contract float %c, %d
+; SLOWF32-SIZE-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %add = fadd contract float %add.m0, %add.m1
+; SLOWF32-SIZE-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %sub.m0 = fmul contract float %a, %b
+; SLOWF32-SIZE-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %sub.m1 = fmul contract float %c, %d
+; SLOWF32-SIZE-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %sub = fsub contract float %sub.m0, %sub.m1
+; SLOWF32-SIZE-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %multi.m0 = fmul contract float %a, %b
+; SLOWF32-SIZE-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %multi.m1 = fmul contract float %c, %d
+; SLOWF32-SIZE-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %multi = fadd contract float %multi.m0, %multi.m1
+; SLOWF32-SIZE-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %multi.use = fadd contract float %multi.m0, %c
+; SLOWF32-SIZE-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %nc.m0 = fmul float %a, %b
+; SLOWF32-SIZE-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %nc.m1 = fmul contract float %c, %d
+; SLOWF32-SIZE-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %nc = fadd contract float %nc.m0, %nc.m1
+; SLOWF32-SIZE-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; FASTF32-SIZE-LABEL: 'fmul_fmul_fadd_f32'
+; FASTF32-SIZE-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %add.m0 = fmul contract float %a, %b
+; FASTF32-SIZE-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %add.m1 = fmul contract float %c, %d
+; FASTF32-SIZE-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %add = fadd contract float %add.m0, %add.m1
+; FASTF32-SIZE-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %sub.m0 = fmul contract float %a, %b
+; FASTF32-SIZE-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %sub.m1 = fmul contract float %c, %d
+; FASTF32-SIZE-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %sub = fsub contract float %sub.m0, %sub.m1
+; FASTF32-SIZE-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %multi.m0 = fmul contract float %a, %b
+; FASTF32-SIZE-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %multi.m1 = fmul contract float %c, %d
+; FASTF32-SIZE-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %multi = fadd contract float %multi.m0, %multi.m1
+; FASTF32-SIZE-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %multi.use = fadd contract float %multi.m0, %c
+; FASTF32-SIZE-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %nc.m0 = fmul float %a, %b
+; FASTF32-SIZE-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %nc.m1 = fmul contract float %c, %d
+; FASTF32-SIZE-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %nc = fadd contract float %nc.m0, %nc.m1
+; FASTF32-SIZE-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+  %add.m0 = fmul contract float %a, %b
+  %add.m1 = fmul contract float %c, %d
+  %add = fadd contract float %add.m0, %add.m1
+
+  %sub.m0 = fmul contract float %a, %b
+  %sub.m1 = fmul contract float %c, %d
+  %sub = fsub contract float %sub.m0, %sub.m1
+
+  %multi.m0 = fmul contract float %a, %b
+  %multi.m1 = fmul contract float %c, %d
+  %multi = fadd contract float %multi.m0, %multi.m1
+  %multi.use = fadd contract float %multi.m0, %c
+
+  %nc.m0 = fmul float %a, %b
+  %nc.m1 = fmul contract float %c, %d
+  %nc = fadd contract float %nc.m0, %nc.m1
+  ret void
+}
+
+define float @fmul_fmul_fadd_reassoc_f32(float %a, float %b, float %c, float %d, float %val) #0 {
+; SLOWF64-LABEL: 'fmul_fmul_fadd_reassoc_f32'
+; SLOWF64-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %m0 = fmul contract float %a, %b
+; SLOWF64-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %m1 = fmul contract float %c, %d
+; SLOWF64-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %sum = fadd contract float %m0, %m1
+; SLOWF64-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %ret = fadd reassoc contract float %sum, %val
+; SLOWF64-NEXT:  Cost Model: Found an estimated cost of 10 for instruction: ret float %ret
+;
+; SLOWF64-SIZE-LABEL: 'fmul_fmul_fadd_reassoc_f32'
+; SLOWF64-SIZE-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %m0 = fmul contract float %a, %b
+; SLOWF64-SIZE-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %m1 = fmul contract float %c, %d
+; SLOWF64-SIZE-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %sum = fadd contract float %m0, %m1
+; SLOWF64-SIZE-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %ret = fadd reassoc contract float %sum, %val
+; SLOWF64-SIZE-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: ret float %ret
+;
+  %m0 = fmul contract float %a, %b
+  %m1 = fmul contract float %c, %d
+  %sum = fadd contract float %m0, %m1
+  %ret = fadd reassoc contract float %sum, %val
+  ret float %ret
+}
+
 attributes #0 = { nounwind }
 
 ;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:

diff  --git a/llvm/test/Transforms/SLPVectorizer/AMDGPU/complex-mul-fma.ll b/llvm/test/Transforms/SLPVectorizer/AMDGPU/complex-mul-fma.ll
new file mode 100644
index 0000000000000..3d3b2c151ecf0
--- /dev/null
+++ b/llvm/test/Transforms/SLPVectorizer/AMDGPU/complex-mul-fma.ll
@@ -0,0 +1,16 @@
+; RUN: opt -passes=slp-vectorizer -mtriple=amdgpu9.42-amd-amdhsa -pass-remarks=slp-vectorizer -disable-output < %s 2>&1 | FileCheck %s
+
+; The second fmul of each fsub/fadd does not fuse, so packing those pays off.
+; CHECK: Stores SLP vectorized with cost -2
+define void @cmul_store(ptr addrspace(1) %out, float %xr, float %xi, float %wr, float %wi) {
+  %xr.wr = fmul contract float %xr, %wr
+  %xi.wi = fmul contract float %xi, %wi
+  %re = fsub contract float %xr.wr, %xi.wi
+  %xi.wr = fmul contract float %xi, %wr
+  %xr.wi = fmul contract float %xr, %wi
+  %im = fadd contract float %xi.wr, %xr.wi
+  store float %re, ptr addrspace(1) %out, align 8
+  %out.im = getelementptr inbounds float, ptr addrspace(1) %out, i64 1
+  store float %im, ptr addrspace(1) %out.im, align 4
+  ret void
+}


        


More information about the llvm-commits mailing list