[llvm] dd4b3b7 - [AMDGPU] Price only the fmul that fuses into an fadd/fsub as free (#226009)
via llvm-commits
llvm-commits at lists.llvm.org
Thu Sep 24 07:46:29 PDT 2026
Author: Gheorghe-Teodor Bercea
Date: 2026-09-24T10:46:21-04:00
New Revision: dd4b3b7f0880c5bd15cea1867f0c4b30468fdd0c
URL: https://github.com/llvm/llvm-project/commit/dd4b3b7f0880c5bd15cea1867f0c4b30468fdd0c
DIFF: https://github.com/llvm/llvm-project/commit/dd4b3b7f0880c5bd15cea1867f0c4b30468fdd0c.diff
LOG: [AMDGPU] Price only the fmul that fuses into an fadd/fsub as free (#226009)
An fadd/fsub with two fmul operands can fuse with only one of them into
an FMA. Currently they are both priced as if they are fused (i.e. free).
Added:
llvm/test/Transforms/SLPVectorizer/AMDGPU/complex-mul-fma.ll
Modified:
llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
llvm/test/Analysis/CostModel/AMDGPU/fused_costs.ll
Removed:
################################################################################
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
index ccb0c7314dcef..f5119764bbb9d 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
@@ -546,6 +546,34 @@ static bool canFuseFMulWithFAddSub(const SITargetLowering &TLI, Type *Ty,
return HasFMAD || (FAddSub->hasAllowContract() && FMul->hasAllowContract());
}
+/// An fma holds one multiply, so only one fmul operand fuses with \p FAddSub.
+static const Instruction *getFusedFMul(const SITargetLowering &TLI, Type *Ty,
+ const Instruction *FAddSub) {
+ for (const Value *Op : FAddSub->operands()) {
+ const auto *FMul = dyn_cast<Instruction>(Op);
+ if (FMul && FMul->getOpcode() == Instruction::FMul && FMul->hasOneUse() &&
+ canFuseFMulWithFAddSub(TLI, Ty, FMul, FAddSub))
+ return FMul;
+ }
+ return nullptr;
+}
+
+static bool isFusedFMul(const SITargetLowering &TLI, Type *Ty,
+ const Instruction *FMul, const Instruction *FAddSub) {
+ const Instruction *Fused = getFusedFMul(TLI, Ty, FAddSub);
+ if (Fused == FMul)
+ return true;
+ // (a * b + c * d) + e becomes fma(a, b, fma(c, d, e)) if the outer fadd has
+ // reassoc.
+ if (!Fused || FAddSub->getOpcode() != Instruction::FAdd ||
+ !FAddSub->hasOneUse())
+ return false;
+ const auto *Outer = dyn_cast<BinaryOperator>(*FAddSub->user_begin());
+ return Outer && Outer->getOpcode() == Instruction::FAdd &&
+ Outer->hasAllowReassoc() &&
+ canFuseFMulWithFAddSub(TLI, Ty, FMul, Outer);
+}
+
InstructionCost GCNTTIImpl::getArithmeticInstrCost(
unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind,
TTI::OperandValueInfo Op1Info, TTI::OperandValueInfo Op2Info,
@@ -613,7 +641,7 @@ InstructionCost GCNTTIImpl::getArithmeticInstrCost(
if (FAddSub &&
(FAddSub->getOpcode() == Instruction::FAdd ||
FAddSub->getOpcode() == Instruction::FSub) &&
- canFuseFMulWithFAddSub(*TLI, Ty, CxtI, FAddSub))
+ isFusedFMul(*TLI, Ty, CxtI, FAddSub))
return TargetTransformInfo::TCC_Free;
}
[[fallthrough]];
@@ -1533,17 +1561,9 @@ bool GCNTTIImpl::isProfitableToSinkOperands(Instruction *I,
// so this stays a move.
if (I->getOpcode() == Instruction::FAdd ||
I->getOpcode() == Instruction::FSub) {
- for (Use &Op : I->operands()) {
- auto *FMul = dyn_cast<Instruction>(Op.get());
- if (!FMul || FMul->getOpcode() != Instruction::FMul ||
- !FMul->hasOneUse() ||
- !canFuseFMulWithFAddSub(*TLI, I->getType(), FMul, I))
- continue;
- // The fused operand. Sink it when it sits in another block, then stop.
- if (FMul->getParent() != I->getParent())
- Ops.push_back(&Op);
- break;
- }
+ const Instruction *FMul = getFusedFMul(*TLI, I->getType(), I);
+ if (FMul && FMul->getParent() != I->getParent())
+ Ops.push_back(&I->getOperandUse(I->getOperand(0) == FMul ? 0 : 1));
}
for (auto &Op : I->operands()) {
diff --git a/llvm/test/Analysis/CostModel/AMDGPU/fused_costs.ll b/llvm/test/Analysis/CostModel/AMDGPU/fused_costs.ll
index aeec6fd916514..3060d103bb1df 100644
--- a/llvm/test/Analysis/CostModel/AMDGPU/fused_costs.ll
+++ b/llvm/test/Analysis/CostModel/AMDGPU/fused_costs.ll
@@ -251,6 +251,112 @@ define void @fmul_fadd_f64(double %a, double %b, double %c, <2 x double> %va, <2
ret void
}
+define void @fmul_fmul_fadd_f32(float %a, float %b, float %c, float %d) #0 {
+; SLOWF32-LABEL: 'fmul_fmul_fadd_f32'
+; SLOWF32-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %add.m0 = fmul contract float %a, %b
+; SLOWF32-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %add.m1 = fmul contract float %c, %d
+; SLOWF32-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %add = fadd contract float %add.m0, %add.m1
+; SLOWF32-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %sub.m0 = fmul contract float %a, %b
+; SLOWF32-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sub.m1 = fmul contract float %c, %d
+; SLOWF32-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sub = fsub contract float %sub.m0, %sub.m1
+; SLOWF32-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %multi.m0 = fmul contract float %a, %b
+; SLOWF32-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %multi.m1 = fmul contract float %c, %d
+; SLOWF32-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %multi = fadd contract float %multi.m0, %multi.m1
+; SLOWF32-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %multi.use = fadd contract float %multi.m0, %c
+; SLOWF32-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %nc.m0 = fmul float %a, %b
+; SLOWF32-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %nc.m1 = fmul contract float %c, %d
+; SLOWF32-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %nc = fadd contract float %nc.m0, %nc.m1
+; SLOWF32-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; FASTF32-LABEL: 'fmul_fmul_fadd_f32'
+; FASTF32-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %add.m0 = fmul contract float %a, %b
+; FASTF32-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %add.m1 = fmul contract float %c, %d
+; FASTF32-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %add = fadd contract float %add.m0, %add.m1
+; FASTF32-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %sub.m0 = fmul contract float %a, %b
+; FASTF32-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sub.m1 = fmul contract float %c, %d
+; FASTF32-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sub = fsub contract float %sub.m0, %sub.m1
+; FASTF32-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %multi.m0 = fmul contract float %a, %b
+; FASTF32-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %multi.m1 = fmul contract float %c, %d
+; FASTF32-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %multi = fadd contract float %multi.m0, %multi.m1
+; FASTF32-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %multi.use = fadd contract float %multi.m0, %c
+; FASTF32-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %nc.m0 = fmul float %a, %b
+; FASTF32-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %nc.m1 = fmul contract float %c, %d
+; FASTF32-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %nc = fadd contract float %nc.m0, %nc.m1
+; FASTF32-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; SLOWF32-SIZE-LABEL: 'fmul_fmul_fadd_f32'
+; SLOWF32-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %add.m0 = fmul contract float %a, %b
+; SLOWF32-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %add.m1 = fmul contract float %c, %d
+; SLOWF32-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %add = fadd contract float %add.m0, %add.m1
+; SLOWF32-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %sub.m0 = fmul contract float %a, %b
+; SLOWF32-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sub.m1 = fmul contract float %c, %d
+; SLOWF32-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sub = fsub contract float %sub.m0, %sub.m1
+; SLOWF32-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %multi.m0 = fmul contract float %a, %b
+; SLOWF32-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %multi.m1 = fmul contract float %c, %d
+; SLOWF32-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %multi = fadd contract float %multi.m0, %multi.m1
+; SLOWF32-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %multi.use = fadd contract float %multi.m0, %c
+; SLOWF32-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %nc.m0 = fmul float %a, %b
+; SLOWF32-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %nc.m1 = fmul contract float %c, %d
+; SLOWF32-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %nc = fadd contract float %nc.m0, %nc.m1
+; SLOWF32-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; FASTF32-SIZE-LABEL: 'fmul_fmul_fadd_f32'
+; FASTF32-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %add.m0 = fmul contract float %a, %b
+; FASTF32-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %add.m1 = fmul contract float %c, %d
+; FASTF32-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %add = fadd contract float %add.m0, %add.m1
+; FASTF32-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %sub.m0 = fmul contract float %a, %b
+; FASTF32-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sub.m1 = fmul contract float %c, %d
+; FASTF32-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sub = fsub contract float %sub.m0, %sub.m1
+; FASTF32-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %multi.m0 = fmul contract float %a, %b
+; FASTF32-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %multi.m1 = fmul contract float %c, %d
+; FASTF32-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %multi = fadd contract float %multi.m0, %multi.m1
+; FASTF32-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %multi.use = fadd contract float %multi.m0, %c
+; FASTF32-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %nc.m0 = fmul float %a, %b
+; FASTF32-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %nc.m1 = fmul contract float %c, %d
+; FASTF32-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %nc = fadd contract float %nc.m0, %nc.m1
+; FASTF32-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %add.m0 = fmul contract float %a, %b
+ %add.m1 = fmul contract float %c, %d
+ %add = fadd contract float %add.m0, %add.m1
+
+ %sub.m0 = fmul contract float %a, %b
+ %sub.m1 = fmul contract float %c, %d
+ %sub = fsub contract float %sub.m0, %sub.m1
+
+ %multi.m0 = fmul contract float %a, %b
+ %multi.m1 = fmul contract float %c, %d
+ %multi = fadd contract float %multi.m0, %multi.m1
+ %multi.use = fadd contract float %multi.m0, %c
+
+ %nc.m0 = fmul float %a, %b
+ %nc.m1 = fmul contract float %c, %d
+ %nc = fadd contract float %nc.m0, %nc.m1
+ ret void
+}
+
+define float @fmul_fmul_fadd_reassoc_f32(float %a, float %b, float %c, float %d, float %val) #0 {
+; SLOWF64-LABEL: 'fmul_fmul_fadd_reassoc_f32'
+; SLOWF64-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %m0 = fmul contract float %a, %b
+; SLOWF64-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %m1 = fmul contract float %c, %d
+; SLOWF64-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sum = fadd contract float %m0, %m1
+; SLOWF64-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %ret = fadd reassoc contract float %sum, %val
+; SLOWF64-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret float %ret
+;
+; SLOWF64-SIZE-LABEL: 'fmul_fmul_fadd_reassoc_f32'
+; SLOWF64-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %m0 = fmul contract float %a, %b
+; SLOWF64-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %m1 = fmul contract float %c, %d
+; SLOWF64-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sum = fadd contract float %m0, %m1
+; SLOWF64-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %ret = fadd reassoc contract float %sum, %val
+; SLOWF64-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret float %ret
+;
+ %m0 = fmul contract float %a, %b
+ %m1 = fmul contract float %c, %d
+ %sum = fadd contract float %m0, %m1
+ %ret = fadd reassoc contract float %sum, %val
+ ret float %ret
+}
+
attributes #0 = { nounwind }
;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
diff --git a/llvm/test/Transforms/SLPVectorizer/AMDGPU/complex-mul-fma.ll b/llvm/test/Transforms/SLPVectorizer/AMDGPU/complex-mul-fma.ll
new file mode 100644
index 0000000000000..3d3b2c151ecf0
--- /dev/null
+++ b/llvm/test/Transforms/SLPVectorizer/AMDGPU/complex-mul-fma.ll
@@ -0,0 +1,16 @@
+; RUN: opt -passes=slp-vectorizer -mtriple=amdgpu9.42-amd-amdhsa -pass-remarks=slp-vectorizer -disable-output < %s 2>&1 | FileCheck %s
+
+; The second fmul of each fsub/fadd does not fuse, so packing those pays off.
+; CHECK: Stores SLP vectorized with cost -2
+define void @cmul_store(ptr addrspace(1) %out, float %xr, float %xi, float %wr, float %wi) {
+ %xr.wr = fmul contract float %xr, %wr
+ %xi.wi = fmul contract float %xi, %wi
+ %re = fsub contract float %xr.wr, %xi.wi
+ %xi.wr = fmul contract float %xi, %wr
+ %xr.wi = fmul contract float %xr, %wi
+ %im = fadd contract float %xi.wr, %xr.wi
+ store float %re, ptr addrspace(1) %out, align 8
+ %out.im = getelementptr inbounds float, ptr addrspace(1) %out, i64 1
+ store float %im, ptr addrspace(1) %out.im, align 4
+ ret void
+}
More information about the llvm-commits
mailing list