[llvm] [AMDGPU] Sink a single-use fmul into the block of its fadd/fsub user (PR #215810)
via llvm-commits
llvm-commits at lists.llvm.org
Wed Aug 12 07:46:00 PDT 2026
llvmorg-github-actions[bot] wrote:
<!--LLVM PR SUMMARY COMMENT-->
@llvm/pr-subscribers-backend-amdgpu
Author: Dmitry Sidorov (MrSidims)
<details>
<summary>Changes</summary>
The cost model prices an fmul as free when its only user is a contractable fadd/fsub, on the assumption the pair is selected as one fused instruction. That only holds when both sit in the same block, and a loop-invariant fmul hoisted out of its user's loop is left behind and never fused, while still priced as free.
The generic sinking pass handles the plain case but will not sink into a loop, which is exactly where the fmul is stranded. Sink it back into the user's block when the two would fuse, using the same condition the cost model does so they cannot disagree. Only the operand that would actually fuse is moved, and only when it has a single use so this stays a move rather than a copy.
Derive that shared condition from instruction selection's own mad legality check rather than re-deriving it here. On targets without a 16-bit mad the pair does not actually fuse, so the f16 fmul is no longer priced free and is not sunk, which is where the f16 test changes come from.
Contributes to https://github.com/llvm/llvm-project/issues/211092
Assisted-by: Claude Code Opus 5
---
Patch is 75.58 KiB, truncated to 20.00 KiB below, full version: https://github.com/llvm/llvm-project/pull/215810.diff
9 Files Affected:
- (modified) llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp (+41-17)
- (modified) llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.h (+6-1)
- (modified) llvm/test/Analysis/CostModel/AMDGPU/fused_costs.ll (+61-56)
- (modified) llvm/test/CodeGen/AMDGPU/global-atomicrmw-fadd-wrong-subtarget.ll (+3-4)
- (modified) llvm/test/CodeGen/AMDGPU/global_atomics_scan_fadd.ll (+18-26)
- (modified) llvm/test/CodeGen/AMDGPU/global_atomics_scan_fsub.ll (+18-26)
- (modified) llvm/test/CodeGen/AMDGPU/sink-fmul-fadd.ll (+91-104)
- (modified) llvm/test/Transforms/CodeGenPrepare/AMDGPU/sink-fmul-fadd-contract-fast.ll (+1-1)
- (modified) llvm/test/Transforms/CodeGenPrepare/AMDGPU/sink-fmul-fadd.ll (+117-49)
``````````diff
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
index 980e26082064f..e2f192a2c5bf8 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
@@ -290,8 +290,6 @@ GCNTTIImpl::GCNTTIImpl(const AMDGPUTargetMachine *TM, const Function &F)
IsGraphics(AMDGPU::isGraphics(F.getCallingConv())) {
SIModeRegisterDefaults Mode(F, *ST);
HasFP32Denormals = Mode.FP32Denormals != DenormalMode::getPreserveSign();
- HasFP64FP16Denormals =
- Mode.FP64FP16Denormals != DenormalMode::getPreserveSign();
}
bool GCNTTIImpl::hasBranchDivergence(const Function *F) const {
@@ -524,6 +522,24 @@ bool GCNTTIImpl::getTgtMemIntrinsic(IntrinsicInst *Inst,
}
}
+bool GCNTTIImpl::canFuseFMulWithFAddSub(MVT::SimpleValueType SLT,
+ const Instruction *FMul,
+ const Instruction *FAddSub) const {
+ const int OPC = TLI->InstructionOpcodeToISD(FAddSub->getOpcode());
+ if (OPC != ISD::FADD && OPC != ISD::FSUB)
+ return false;
+
+ // The mad forms fuse exactly without fast-math flags but flush denormals.
+ const DenormalFPEnv FPEnv = FAddSub->getFunction()->getDenormalFPEnv();
+ if (TLI->isFMADLegal(MVT(SLT), FPEnv))
+ return true;
+
+ // Other types fuse only with contract or fast.
+ const TargetOptions &Options = TLI->getTargetMachine().Options;
+ return Options.AllowFPOpFusion == FPOpFusion::Fast ||
+ (FAddSub->hasAllowContract() && FMul->hasAllowContract());
+}
+
InstructionCost GCNTTIImpl::getArithmeticInstrCost(
unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind,
TTI::OperandValueInfo Op1Info, TTI::OperandValueInfo Op2Info,
@@ -587,21 +603,9 @@ InstructionCost GCNTTIImpl::getArithmeticInstrCost(
// fmul(b,c) supposing the fadd|fsub will get estimated cost for the whole
// fused operation.
if (CxtI && CxtI->hasOneUse())
- if (const auto *FAdd = dyn_cast<BinaryOperator>(*CxtI->user_begin())) {
- const int OPC = TLI->InstructionOpcodeToISD(FAdd->getOpcode());
- if (OPC == ISD::FADD || OPC == ISD::FSUB) {
- if (ST->hasMadMacF32Insts() && SLT == MVT::f32 && !HasFP32Denormals)
- return TargetTransformInfo::TCC_Free;
- if (ST->has16BitInsts() && SLT == MVT::f16 && !HasFP64FP16Denormals)
- return TargetTransformInfo::TCC_Free;
-
- // Estimate all types may be fused with contract/unsafe flags
- const TargetOptions &Options = TLI->getTargetMachine().Options;
- if (Options.AllowFPOpFusion == FPOpFusion::Fast ||
- (FAdd->hasAllowContract() && CxtI->hasAllowContract()))
- return TargetTransformInfo::TCC_Free;
- }
- }
+ if (const auto *FAdd = dyn_cast<BinaryOperator>(*CxtI->user_begin()))
+ if (canFuseFMulWithFAddSub(SLT, CxtI, FAdd))
+ return TargetTransformInfo::TCC_Free;
[[fallthrough]];
case ISD::FADD:
case ISD::FSUB:
@@ -1483,6 +1487,26 @@ bool GCNTTIImpl::isProfitableToSinkOperands(Instruction *I,
SmallVectorImpl<Use *> &Ops) const {
using namespace PatternMatch;
+ // The cost model prices this fmul as free assuming it fuses with its
+ // fadd/fsub user, which needs them in one block. Sink a stranded
+ // loop-invariant fmul back to the user when they would fuse. Single use only,
+ // so this stays a move.
+ if (I->getOpcode() == Instruction::FAdd ||
+ I->getOpcode() == Instruction::FSub) {
+ MVT::SimpleValueType SLT =
+ getTypeLegalizationCost(I->getType()).second.getScalarType().SimpleTy;
+ for (Use &Op : I->operands()) {
+ auto *FMul = dyn_cast<Instruction>(Op.get());
+ if (!FMul || FMul->getOpcode() != Instruction::FMul ||
+ !FMul->hasOneUse() || !canFuseFMulWithFAddSub(SLT, FMul, I))
+ continue;
+ // The fused operand. Sink it when it sits in another block, then stop.
+ if (FMul->getParent() != I->getParent())
+ Ops.push_back(&Op);
+ break;
+ }
+ }
+
for (auto &Op : I->operands()) {
// Ensure we are not already sinking this operand.
if (any_of(Ops, [&](Use *U) { return U->get() == Op.get(); }))
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.h b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.h
index df7b6d339e6c2..155dcca92f4ac 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.h
+++ b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.h
@@ -71,7 +71,6 @@ class GCNTTIImpl final : public BasicTTIImplBase<GCNTTIImpl> {
AMDGPUTTIImpl CommonTTI;
bool IsGraphics;
bool HasFP32Denormals;
- bool HasFP64FP16Denormals;
static constexpr bool InlinerVectorBonusPercent = 0;
static const FeatureBitset InlineFeatureIgnoreList;
@@ -103,6 +102,12 @@ class GCNTTIImpl final : public BasicTTIImplBase<GCNTTIImpl> {
std::pair<InstructionCost, MVT> getTypeLegalizationCost(Type *Ty) const;
+ /// \returns true if \p FMul and its single fadd/fsub user \p FAddSub are
+ /// expected to fuse during instruction selection. \p SLT is the legalized
+ /// scalar type.
+ bool canFuseFMulWithFAddSub(MVT::SimpleValueType SLT, const Instruction *FMul,
+ const Instruction *FAddSub) const;
+
/// \returns true if V might be divergent even when all of its operands
/// are uniform.
bool isSourceOfDivergence(const Value *V) const;
diff --git a/llvm/test/Analysis/CostModel/AMDGPU/fused_costs.ll b/llvm/test/Analysis/CostModel/AMDGPU/fused_costs.ll
index 98773ba7292f4..858292b469fba 100644
--- a/llvm/test/Analysis/CostModel/AMDGPU/fused_costs.ll
+++ b/llvm/test/Analysis/CostModel/AMDGPU/fused_costs.ll
@@ -92,65 +92,65 @@ define void @fmul_fadd_f32() #0 {
}
define void @fmul_fadd_f16() #0 {
-; FUSED-LABEL: 'fmul_fadd_f16'
-; FUSED-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %f16 = fmul half undef, undef
-; FUSED-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f16add = fadd half %f16, undef
-; FUSED-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %f16c = fmul contract half undef, undef
-; FUSED-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f15cadd = fadd contract half %f16c, undef
-; FUSED-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %v2f16 = fmul <2 x half> undef, undef
-; FUSED-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %v2f16add = fadd <2 x half> %v2f16, undef
-; FUSED-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %f16_2 = fmul half undef, undef
-; FUSED-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f16sub = fsub half %f16_2, undef
-; FUSED-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %f16c_2 = fmul contract half undef, undef
-; FUSED-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f15csub = fsub contract half %f16c_2, undef
-; FUSED-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %v2f16_2 = fmul <2 x half> undef, undef
-; FUSED-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %v2f16sub = fsub <2 x half> %v2f16_2, undef
-; FUSED-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+; SLOWF32-LABEL: 'fmul_fadd_f16'
+; SLOWF32-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %f16 = fmul half undef, undef
+; SLOWF32-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f16add = fadd half %f16, undef
+; SLOWF32-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %f16c = fmul contract half undef, undef
+; SLOWF32-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f15cadd = fadd contract half %f16c, undef
+; SLOWF32-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %v2f16 = fmul <2 x half> undef, undef
+; SLOWF32-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %v2f16add = fadd <2 x half> %v2f16, undef
+; SLOWF32-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %f16_2 = fmul half undef, undef
+; SLOWF32-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f16sub = fsub half %f16_2, undef
+; SLOWF32-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %f16c_2 = fmul contract half undef, undef
+; SLOWF32-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f15csub = fsub contract half %f16c_2, undef
+; SLOWF32-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %v2f16_2 = fmul <2 x half> undef, undef
+; SLOWF32-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %v2f16sub = fsub <2 x half> %v2f16_2, undef
+; SLOWF32-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
;
-; GFX9SLOW-LABEL: 'fmul_fadd_f16'
-; GFX9SLOW-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f16 = fmul half undef, undef
-; GFX9SLOW-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f16add = fadd half %f16, undef
-; GFX9SLOW-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %f16c = fmul contract half undef, undef
-; GFX9SLOW-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f15cadd = fadd contract half %f16c, undef
-; GFX9SLOW-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %v2f16 = fmul <2 x half> undef, undef
-; GFX9SLOW-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %v2f16add = fadd <2 x half> %v2f16, undef
-; GFX9SLOW-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f16_2 = fmul half undef, undef
-; GFX9SLOW-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f16sub = fsub half %f16_2, undef
-; GFX9SLOW-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %f16c_2 = fmul contract half undef, undef
-; GFX9SLOW-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f15csub = fsub contract half %f16c_2, undef
-; GFX9SLOW-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %v2f16_2 = fmul <2 x half> undef, undef
-; GFX9SLOW-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %v2f16sub = fsub <2 x half> %v2f16_2, undef
-; GFX9SLOW-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+; FASTF32-LABEL: 'fmul_fadd_f16'
+; FASTF32-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f16 = fmul half undef, undef
+; FASTF32-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f16add = fadd half %f16, undef
+; FASTF32-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %f16c = fmul contract half undef, undef
+; FASTF32-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f15cadd = fadd contract half %f16c, undef
+; FASTF32-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %v2f16 = fmul <2 x half> undef, undef
+; FASTF32-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %v2f16add = fadd <2 x half> %v2f16, undef
+; FASTF32-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f16_2 = fmul half undef, undef
+; FASTF32-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f16sub = fsub half %f16_2, undef
+; FASTF32-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %f16c_2 = fmul contract half undef, undef
+; FASTF32-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f15csub = fsub contract half %f16c_2, undef
+; FASTF32-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %v2f16_2 = fmul <2 x half> undef, undef
+; FASTF32-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %v2f16sub = fsub <2 x half> %v2f16_2, undef
+; FASTF32-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
;
-; FUSED-SIZE-LABEL: 'fmul_fadd_f16'
-; FUSED-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %f16 = fmul half undef, undef
-; FUSED-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f16add = fadd half %f16, undef
-; FUSED-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %f16c = fmul contract half undef, undef
-; FUSED-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f15cadd = fadd contract half %f16c, undef
-; FUSED-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %v2f16 = fmul <2 x half> undef, undef
-; FUSED-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %v2f16add = fadd <2 x half> %v2f16, undef
-; FUSED-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %f16_2 = fmul half undef, undef
-; FUSED-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f16sub = fsub half %f16_2, undef
-; FUSED-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %f16c_2 = fmul contract half undef, undef
-; FUSED-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f15csub = fsub contract half %f16c_2, undef
-; FUSED-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %v2f16_2 = fmul <2 x half> undef, undef
-; FUSED-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %v2f16sub = fsub <2 x half> %v2f16_2, undef
-; FUSED-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+; SLOWF32-SIZE-LABEL: 'fmul_fadd_f16'
+; SLOWF32-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %f16 = fmul half undef, undef
+; SLOWF32-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f16add = fadd half %f16, undef
+; SLOWF32-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %f16c = fmul contract half undef, undef
+; SLOWF32-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f15cadd = fadd contract half %f16c, undef
+; SLOWF32-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %v2f16 = fmul <2 x half> undef, undef
+; SLOWF32-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %v2f16add = fadd <2 x half> %v2f16, undef
+; SLOWF32-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %f16_2 = fmul half undef, undef
+; SLOWF32-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f16sub = fsub half %f16_2, undef
+; SLOWF32-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %f16c_2 = fmul contract half undef, undef
+; SLOWF32-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f15csub = fsub contract half %f16c_2, undef
+; SLOWF32-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %v2f16_2 = fmul <2 x half> undef, undef
+; SLOWF32-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %v2f16sub = fsub <2 x half> %v2f16_2, undef
+; SLOWF32-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
-; GFX9SLOW-SIZE-LABEL: 'fmul_fadd_f16'
-; GFX9SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f16 = fmul half undef, undef
-; GFX9SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f16add = fadd half %f16, undef
-; GFX9SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %f16c = fmul contract half undef, undef
-; GFX9SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f15cadd = fadd contract half %f16c, undef
-; GFX9SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %v2f16 = fmul <2 x half> undef, undef
-; GFX9SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %v2f16add = fadd <2 x half> %v2f16, undef
-; GFX9SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f16_2 = fmul half undef, undef
-; GFX9SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f16sub = fsub half %f16_2, undef
-; GFX9SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %f16c_2 = fmul contract half undef, undef
-; GFX9SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f15csub = fsub contract half %f16c_2, undef
-; GFX9SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %v2f16_2 = fmul <2 x half> undef, undef
-; GFX9SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %v2f16sub = fsub <2 x half> %v2f16_2, undef
-; GFX9SLOW-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+; FASTF32-SIZE-LABEL: 'fmul_fadd_f16'
+; FASTF32-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f16 = fmul half undef, undef
+; FASTF32-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f16add = fadd half %f16, undef
+; FASTF32-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %f16c = fmul contract half undef, undef
+; FASTF32-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f15cadd = fadd contract half %f16c, undef
+; FASTF32-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %v2f16 = fmul <2 x half> undef, undef
+; FASTF32-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %v2f16add = fadd <2 x half> %v2f16, undef
+; FASTF32-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f16_2 = fmul half undef, undef
+; FASTF32-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f16sub = fsub half %f16_2, undef
+; FASTF32-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %f16c_2 = fmul contract half undef, undef
+; FASTF32-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %f15csub = fsub contract half %f16c_2, undef
+; FASTF32-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %v2f16_2 = fmul <2 x half> undef, undef
+; FASTF32-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %v2f16sub = fsub <2 x half> %v2f16_2, undef
+; FASTF32-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
;
%f16 = fmul half undef, undef
%f16add = fadd half %f16, undef
@@ -255,3 +255,8 @@ define void @fmul_fadd_f64() #0 {
attributes #0 = { nounwind }
+;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
+; FUSED: {{.*}}
+; FUSED-SIZE: {{.*}}
+; GFX9SLOW: {{.*}}
+; GFX9SLOW-SIZE: {{.*}}
diff --git a/llvm/test/CodeGen/AMDGPU/global-atomicrmw-fadd-wrong-subtarget.ll b/llvm/test/CodeGen/AMDGPU/global-atomicrmw-fadd-wrong-subtarget.ll
index a1da18969552a..d4870b7ca8a95 100644
--- a/llvm/test/CodeGen/AMDGPU/global-atomicrmw-fadd-wrong-subtarget.ll
+++ b/llvm/test/CodeGen/AMDGPU/global-atomicrmw-fadd-wrong-subtarget.ll
@@ -14,18 +14,17 @@ define amdgpu_kernel void @global_atomic_fadd_ret_f32_wrong_subtarget(ptr addrsp
; GCN-NEXT: ; %bb.1:
; GCN-NEXT: s_load_dwordx2 s[4:5], s[4:5], 0x0
; GCN-NEXT: s_bcnt1_i32_b64 s0, s[0:1]
-; GCN-NEXT: v_cvt_f32_ubyte0_e32 v1, s0
; GCN-NEXT: s_mov_b64 s[6:7], 0
-; GCN-NEXT: v_mul_f32_e32 v2, 4.0, v1
+; GCN-NEXT: v_cvt_f32_ubyte0_e32 v2, s0
+; GCN-NEXT: v_mov_b32_e32 v3, 0
; GCN-NEXT: s_waitcnt lgkmcnt(0)
; GCN-NEXT: s_load_dword s8, s[4:5], 0x0
-; GCN-NEXT: v_mov_b32_e32 v3, 0
; GCN-NEXT: s_waitcnt lgkmcnt(0)
; GCN-NEXT: v_mov_b32_e32 v1, s8
; GCN-NEXT: .LBB0_2: ; %atomicrmw.start
; GCN-NEXT: ; =>This Inner Loop Header: Depth=1
; GCN-NEXT: v_mov_b32_e32 v5, v1
-; GCN-NEXT: v_add_f32_e32 v4, v5, v2
+; GCN-NEXT: v_mad_f32 v4, 4.0, v2, v5
; GCN-NEXT: global_atomic_cmpswap v1, v3, v[4:5], s[4:5] glc
; GCN-NEXT: s_waitcnt vmcnt(0)
; GCN-NEXT: buffer_wbinvl1
diff --git a/llvm/test/CodeGen/AMDGPU/global_atomics_scan_fadd.ll b/llvm/test/CodeGen/AMDGPU/global_atomics_scan_fadd.ll
index d307ffaff5c63..8672b6eab90de 100644
--- a/llvm/test/CodeGen/AMDGPU/global_atomics_scan_fadd.ll
+++ b/llvm/test/CodeGen/AMDGPU/global_atomics_scan_fadd.ll
@@ -27,18 +27,17 @@ define amdgpu_kernel void @global_atomic_fadd_uni_address_uni_value_agent_scope_
; GFX7LESS-NEXT: ; %bb.1:
; GFX7LESS-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x9
; GFX7LESS-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
-; GFX7LESS-NEXT: v_cvt_f32_ubyte0_e32 v0, s2
; GFX7LESS-NEXT: s_mov_b64 s[4:5], 0
; GFX7LESS-NEXT: s_mov_b32 s3, 0xf000
+; GFX7LESS-NEXT: v_cvt_f32_ubyte0_e32 v2, s2
; GFX7LESS-NEXT: s_waitcnt lgkmcnt(0)
; GFX7LESS-NEX...
[truncated]
``````````
</details>
https://github.com/llvm/llvm-project/pull/215810
More information about the llvm-commits
mailing list