[llvm] [AMDGPU] Price only the fmul that fuses into an fadd/fsub as free (PR #226009)
Gheorghe-Teodor Bercea via llvm-commits
llvm-commits at lists.llvm.org
Thu Sep 24 06:14:20 PDT 2026
https://github.com/doru1004 updated https://github.com/llvm/llvm-project/pull/226009
>From 537155fdbb8029c40f3b040ef9f0146d5d28d0f2 Mon Sep 17 00:00:00 2001
From: Gheorghe-Teodor Bercea <dobercea at amd.com>
Date: Wed, 23 Sep 2026 23:17:33 -0400
Subject: [PATCH 1/3] Price only the fmul that fuses into an fadd/fsub as free
---
.../AMDGPU/AMDGPUTargetTransformInfo.cpp | 28 ++++---
.../Analysis/CostModel/AMDGPU/fused_costs.ll | 84 +++++++++++++++++++
.../SLPVectorizer/AMDGPU/complex-mul-fma.ll | 36 ++++++++
3 files changed, 136 insertions(+), 12 deletions(-)
create mode 100644 llvm/test/Transforms/SLPVectorizer/AMDGPU/complex-mul-fma.ll
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
index ccb0c7314dcef4..4a7f84f0baed5c 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
@@ -546,6 +546,18 @@ static bool canFuseFMulWithFAddSub(const SITargetLowering &TLI, Type *Ty,
return HasFMAD || (FAddSub->hasAllowContract() && FMul->hasAllowContract());
}
+/// An fma holds one multiply, so only one fmul operand fuses with \p FAddSub.
+static const Instruction *getFusedFMul(const SITargetLowering &TLI, Type *Ty,
+ const Instruction *FAddSub) {
+ for (const Value *Op : FAddSub->operands()) {
+ const auto *FMul = dyn_cast<Instruction>(Op);
+ if (FMul && FMul->getOpcode() == Instruction::FMul && FMul->hasOneUse() &&
+ canFuseFMulWithFAddSub(TLI, Ty, FMul, FAddSub))
+ return FMul;
+ }
+ return nullptr;
+}
+
InstructionCost GCNTTIImpl::getArithmeticInstrCost(
unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind,
TTI::OperandValueInfo Op1Info, TTI::OperandValueInfo Op2Info,
@@ -613,7 +625,7 @@ InstructionCost GCNTTIImpl::getArithmeticInstrCost(
if (FAddSub &&
(FAddSub->getOpcode() == Instruction::FAdd ||
FAddSub->getOpcode() == Instruction::FSub) &&
- canFuseFMulWithFAddSub(*TLI, Ty, CxtI, FAddSub))
+ getFusedFMul(*TLI, Ty, FAddSub) == CxtI)
return TargetTransformInfo::TCC_Free;
}
[[fallthrough]];
@@ -1533,17 +1545,9 @@ bool GCNTTIImpl::isProfitableToSinkOperands(Instruction *I,
// so this stays a move.
if (I->getOpcode() == Instruction::FAdd ||
I->getOpcode() == Instruction::FSub) {
- for (Use &Op : I->operands()) {
- auto *FMul = dyn_cast<Instruction>(Op.get());
- if (!FMul || FMul->getOpcode() != Instruction::FMul ||
- !FMul->hasOneUse() ||
- !canFuseFMulWithFAddSub(*TLI, I->getType(), FMul, I))
- continue;
- // The fused operand. Sink it when it sits in another block, then stop.
- if (FMul->getParent() != I->getParent())
- Ops.push_back(&Op);
- break;
- }
+ const Instruction *FMul = getFusedFMul(*TLI, I->getType(), I);
+ if (FMul && FMul->getParent() != I->getParent())
+ Ops.push_back(&I->getOperandUse(I->getOperand(0) == FMul ? 0 : 1));
}
for (auto &Op : I->operands()) {
diff --git a/llvm/test/Analysis/CostModel/AMDGPU/fused_costs.ll b/llvm/test/Analysis/CostModel/AMDGPU/fused_costs.ll
index aeec6fd9165142..1ca2d1e785bf53 100644
--- a/llvm/test/Analysis/CostModel/AMDGPU/fused_costs.ll
+++ b/llvm/test/Analysis/CostModel/AMDGPU/fused_costs.ll
@@ -251,6 +251,90 @@ define void @fmul_fadd_f64(double %a, double %b, double %c, <2 x double> %va, <2
ret void
}
+define void @fmul_fmul_fadd_f32(float %a, float %b, float %c, float %d) #0 {
+; SLOWF32-LABEL: 'fmul_fmul_fadd_f32'
+; SLOWF32-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %add.m0 = fmul contract float %a, %b
+; SLOWF32-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %add.m1 = fmul contract float %c, %d
+; SLOWF32-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %add = fadd contract float %add.m0, %add.m1
+; SLOWF32-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %sub.m0 = fmul contract float %a, %b
+; SLOWF32-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sub.m1 = fmul contract float %c, %d
+; SLOWF32-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sub = fsub contract float %sub.m0, %sub.m1
+; SLOWF32-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %multi.m0 = fmul contract float %a, %b
+; SLOWF32-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %multi.m1 = fmul contract float %c, %d
+; SLOWF32-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %multi = fadd contract float %multi.m0, %multi.m1
+; SLOWF32-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %multi.use = fadd contract float %multi.m0, %c
+; SLOWF32-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %nc.m0 = fmul float %a, %b
+; SLOWF32-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %nc.m1 = fmul contract float %c, %d
+; SLOWF32-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %nc = fadd contract float %nc.m0, %nc.m1
+; SLOWF32-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; FASTF32-LABEL: 'fmul_fmul_fadd_f32'
+; FASTF32-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %add.m0 = fmul contract float %a, %b
+; FASTF32-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %add.m1 = fmul contract float %c, %d
+; FASTF32-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %add = fadd contract float %add.m0, %add.m1
+; FASTF32-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %sub.m0 = fmul contract float %a, %b
+; FASTF32-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sub.m1 = fmul contract float %c, %d
+; FASTF32-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sub = fsub contract float %sub.m0, %sub.m1
+; FASTF32-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %multi.m0 = fmul contract float %a, %b
+; FASTF32-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %multi.m1 = fmul contract float %c, %d
+; FASTF32-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %multi = fadd contract float %multi.m0, %multi.m1
+; FASTF32-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %multi.use = fadd contract float %multi.m0, %c
+; FASTF32-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %nc.m0 = fmul float %a, %b
+; FASTF32-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %nc.m1 = fmul contract float %c, %d
+; FASTF32-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %nc = fadd contract float %nc.m0, %nc.m1
+; FASTF32-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret void
+;
+; SLOWF32-SIZE-LABEL: 'fmul_fmul_fadd_f32'
+; SLOWF32-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %add.m0 = fmul contract float %a, %b
+; SLOWF32-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %add.m1 = fmul contract float %c, %d
+; SLOWF32-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %add = fadd contract float %add.m0, %add.m1
+; SLOWF32-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %sub.m0 = fmul contract float %a, %b
+; SLOWF32-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sub.m1 = fmul contract float %c, %d
+; SLOWF32-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sub = fsub contract float %sub.m0, %sub.m1
+; SLOWF32-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %multi.m0 = fmul contract float %a, %b
+; SLOWF32-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %multi.m1 = fmul contract float %c, %d
+; SLOWF32-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %multi = fadd contract float %multi.m0, %multi.m1
+; SLOWF32-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %multi.use = fadd contract float %multi.m0, %c
+; SLOWF32-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %nc.m0 = fmul float %a, %b
+; SLOWF32-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %nc.m1 = fmul contract float %c, %d
+; SLOWF32-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %nc = fadd contract float %nc.m0, %nc.m1
+; SLOWF32-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+; FASTF32-SIZE-LABEL: 'fmul_fmul_fadd_f32'
+; FASTF32-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %add.m0 = fmul contract float %a, %b
+; FASTF32-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %add.m1 = fmul contract float %c, %d
+; FASTF32-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %add = fadd contract float %add.m0, %add.m1
+; FASTF32-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %sub.m0 = fmul contract float %a, %b
+; FASTF32-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sub.m1 = fmul contract float %c, %d
+; FASTF32-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sub = fsub contract float %sub.m0, %sub.m1
+; FASTF32-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %multi.m0 = fmul contract float %a, %b
+; FASTF32-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %multi.m1 = fmul contract float %c, %d
+; FASTF32-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %multi = fadd contract float %multi.m0, %multi.m1
+; FASTF32-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %multi.use = fadd contract float %multi.m0, %c
+; FASTF32-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %nc.m0 = fmul float %a, %b
+; FASTF32-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %nc.m1 = fmul contract float %c, %d
+; FASTF32-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %nc = fadd contract float %nc.m0, %nc.m1
+; FASTF32-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret void
+;
+ %add.m0 = fmul contract float %a, %b
+ %add.m1 = fmul contract float %c, %d
+ %add = fadd contract float %add.m0, %add.m1
+
+ %sub.m0 = fmul contract float %a, %b
+ %sub.m1 = fmul contract float %c, %d
+ %sub = fsub contract float %sub.m0, %sub.m1
+
+ %multi.m0 = fmul contract float %a, %b
+ %multi.m1 = fmul contract float %c, %d
+ %multi = fadd contract float %multi.m0, %multi.m1
+ %multi.use = fadd contract float %multi.m0, %c
+
+ %nc.m0 = fmul float %a, %b
+ %nc.m1 = fmul contract float %c, %d
+ %nc = fadd contract float %nc.m0, %nc.m1
+ ret void
+}
+
attributes #0 = { nounwind }
;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
diff --git a/llvm/test/Transforms/SLPVectorizer/AMDGPU/complex-mul-fma.ll b/llvm/test/Transforms/SLPVectorizer/AMDGPU/complex-mul-fma.ll
new file mode 100644
index 00000000000000..26e3b0e71d76e0
--- /dev/null
+++ b/llvm/test/Transforms/SLPVectorizer/AMDGPU/complex-mul-fma.ll
@@ -0,0 +1,36 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -passes=slp-vectorizer -S -mtriple=amdgcn-amd-amdhsa -mcpu=gfx90a -slp-threshold=1 < %s | FileCheck %s
+; RUN: opt -passes=slp-vectorizer -S -mtriple=amdgcn-amd-amdhsa -mcpu=gfx942 -slp-threshold=1 < %s | FileCheck %s
+; RUN: opt -passes=slp-vectorizer -S -mtriple=amdgcn-amd-amdhsa -mcpu=gfx950 -slp-threshold=1 < %s | FileCheck %s
+
+; The second fmul of each fsub/fadd does not fuse, so packing those pays off.
+
+define void @cmul_store(ptr addrspace(1) %out, float %xr, float %xi, float %wr, float %wi) {
+; CHECK-LABEL: define void @cmul_store(
+; CHECK-SAME: ptr addrspace(1) [[OUT:%.*]], float [[XR:%.*]], float [[XI:%.*]], float [[WR:%.*]], float [[WI:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT: [[TMP1:%.*]] = insertelement <2 x float> poison, float [[XR]], i64 0
+; CHECK-NEXT: [[TMP2:%.*]] = insertelement <2 x float> [[TMP1]], float [[XI]], i64 1
+; CHECK-NEXT: [[TMP3:%.*]] = insertelement <2 x float> poison, float [[WR]], i64 0
+; CHECK-NEXT: [[TMP4:%.*]] = shufflevector <2 x float> [[TMP3]], <2 x float> poison, <2 x i32> zeroinitializer
+; CHECK-NEXT: [[TMP5:%.*]] = fmul contract <2 x float> [[TMP2]], [[TMP4]]
+; CHECK-NEXT: [[TMP6:%.*]] = insertelement <2 x float> poison, float [[WI]], i64 0
+; CHECK-NEXT: [[TMP7:%.*]] = shufflevector <2 x float> [[TMP6]], <2 x float> poison, <2 x i32> zeroinitializer
+; CHECK-NEXT: [[TMP8:%.*]] = fmul contract <2 x float> [[TMP2]], [[TMP7]]
+; CHECK-NEXT: [[TMP9:%.*]] = shufflevector <2 x float> [[TMP8]], <2 x float> poison, <2 x i32> <i32 1, i32 0>
+; CHECK-NEXT: [[TMP10:%.*]] = fsub contract <2 x float> [[TMP5]], [[TMP9]]
+; CHECK-NEXT: [[TMP11:%.*]] = fadd contract <2 x float> [[TMP5]], [[TMP9]]
+; CHECK-NEXT: [[TMP12:%.*]] = shufflevector <2 x float> [[TMP10]], <2 x float> [[TMP11]], <2 x i32> <i32 0, i32 3>
+; CHECK-NEXT: store <2 x float> [[TMP12]], ptr addrspace(1) [[OUT]], align 8
+; CHECK-NEXT: ret void
+;
+ %xr.wr = fmul contract float %xr, %wr
+ %xi.wi = fmul contract float %xi, %wi
+ %re = fsub contract float %xr.wr, %xi.wi
+ %xi.wr = fmul contract float %xi, %wr
+ %xr.wi = fmul contract float %xr, %wi
+ %im = fadd contract float %xi.wr, %xr.wi
+ store float %re, ptr addrspace(1) %out, align 8
+ %out.im = getelementptr inbounds float, ptr addrspace(1) %out, i64 1
+ store float %im, ptr addrspace(1) %out.im, align 4
+ ret void
+}
>From e52a7fae4f25558ae2bd54c9c0997fecf376591b Mon Sep 17 00:00:00 2001
From: Gheorghe-Teodor Bercea <dobercea at amd.com>
Date: Thu, 24 Sep 2026 08:47:18 -0400
Subject: [PATCH 2/3] Fix the cost and the test.
---
.../AMDGPU/AMDGPUTargetTransformInfo.cpp | 18 ++++++++++++-
.../Analysis/CostModel/AMDGPU/fused_costs.ll | 22 ++++++++++++++++
.../SLPVectorizer/AMDGPU/complex-mul-fma.ll | 25 +++----------------
3 files changed, 42 insertions(+), 23 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
index 4a7f84f0baed5c..f5119764bbb9db 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
@@ -558,6 +558,22 @@ static const Instruction *getFusedFMul(const SITargetLowering &TLI, Type *Ty,
return nullptr;
}
+static bool isFusedFMul(const SITargetLowering &TLI, Type *Ty,
+ const Instruction *FMul, const Instruction *FAddSub) {
+ const Instruction *Fused = getFusedFMul(TLI, Ty, FAddSub);
+ if (Fused == FMul)
+ return true;
+ // (a * b + c * d) + e becomes fma(a, b, fma(c, d, e)) if the outer fadd has
+ // reassoc.
+ if (!Fused || FAddSub->getOpcode() != Instruction::FAdd ||
+ !FAddSub->hasOneUse())
+ return false;
+ const auto *Outer = dyn_cast<BinaryOperator>(*FAddSub->user_begin());
+ return Outer && Outer->getOpcode() == Instruction::FAdd &&
+ Outer->hasAllowReassoc() &&
+ canFuseFMulWithFAddSub(TLI, Ty, FMul, Outer);
+}
+
InstructionCost GCNTTIImpl::getArithmeticInstrCost(
unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind,
TTI::OperandValueInfo Op1Info, TTI::OperandValueInfo Op2Info,
@@ -625,7 +641,7 @@ InstructionCost GCNTTIImpl::getArithmeticInstrCost(
if (FAddSub &&
(FAddSub->getOpcode() == Instruction::FAdd ||
FAddSub->getOpcode() == Instruction::FSub) &&
- getFusedFMul(*TLI, Ty, FAddSub) == CxtI)
+ isFusedFMul(*TLI, Ty, CxtI, FAddSub))
return TargetTransformInfo::TCC_Free;
}
[[fallthrough]];
diff --git a/llvm/test/Analysis/CostModel/AMDGPU/fused_costs.ll b/llvm/test/Analysis/CostModel/AMDGPU/fused_costs.ll
index 1ca2d1e785bf53..3060d103bb1df6 100644
--- a/llvm/test/Analysis/CostModel/AMDGPU/fused_costs.ll
+++ b/llvm/test/Analysis/CostModel/AMDGPU/fused_costs.ll
@@ -335,6 +335,28 @@ define void @fmul_fmul_fadd_f32(float %a, float %b, float %c, float %d) #0 {
ret void
}
+define float @fmul_fmul_fadd_reassoc_f32(float %a, float %b, float %c, float %d, float %val) #0 {
+; SLOWF64-LABEL: 'fmul_fmul_fadd_reassoc_f32'
+; SLOWF64-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %m0 = fmul contract float %a, %b
+; SLOWF64-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %m1 = fmul contract float %c, %d
+; SLOWF64-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sum = fadd contract float %m0, %m1
+; SLOWF64-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %ret = fadd reassoc contract float %sum, %val
+; SLOWF64-NEXT: Cost Model: Found an estimated cost of 10 for instruction: ret float %ret
+;
+; SLOWF64-SIZE-LABEL: 'fmul_fmul_fadd_reassoc_f32'
+; SLOWF64-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %m0 = fmul contract float %a, %b
+; SLOWF64-SIZE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %m1 = fmul contract float %c, %d
+; SLOWF64-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %sum = fadd contract float %m0, %m1
+; SLOWF64-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %ret = fadd reassoc contract float %sum, %val
+; SLOWF64-SIZE-NEXT: Cost Model: Found an estimated cost of 1 for instruction: ret float %ret
+;
+ %m0 = fmul contract float %a, %b
+ %m1 = fmul contract float %c, %d
+ %sum = fadd contract float %m0, %m1
+ %ret = fadd reassoc contract float %sum, %val
+ ret float %ret
+}
+
attributes #0 = { nounwind }
;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
diff --git a/llvm/test/Transforms/SLPVectorizer/AMDGPU/complex-mul-fma.ll b/llvm/test/Transforms/SLPVectorizer/AMDGPU/complex-mul-fma.ll
index 26e3b0e71d76e0..b4c0de7fcf65cf 100644
--- a/llvm/test/Transforms/SLPVectorizer/AMDGPU/complex-mul-fma.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AMDGPU/complex-mul-fma.ll
@@ -1,28 +1,9 @@
-; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
-; RUN: opt -passes=slp-vectorizer -S -mtriple=amdgcn-amd-amdhsa -mcpu=gfx90a -slp-threshold=1 < %s | FileCheck %s
-; RUN: opt -passes=slp-vectorizer -S -mtriple=amdgcn-amd-amdhsa -mcpu=gfx942 -slp-threshold=1 < %s | FileCheck %s
-; RUN: opt -passes=slp-vectorizer -S -mtriple=amdgcn-amd-amdhsa -mcpu=gfx950 -slp-threshold=1 < %s | FileCheck %s
+; RUN: opt -passes=slp-vectorizer -mtriple=amdgcn-amd-amdhsa -mcpu=gfx942 -pass-remarks=slp-vectorizer -disable-output < %s 2>&1 | FileCheck %s
+; RUN: opt -passes=slp-vectorizer -mtriple=amdgcn-amd-amdhsa -mcpu=gfx950 -pass-remarks=slp-vectorizer -disable-output < %s 2>&1 | FileCheck %s
; The second fmul of each fsub/fadd does not fuse, so packing those pays off.
-
+; CHECK: Stores SLP vectorized with cost -2
define void @cmul_store(ptr addrspace(1) %out, float %xr, float %xi, float %wr, float %wi) {
-; CHECK-LABEL: define void @cmul_store(
-; CHECK-SAME: ptr addrspace(1) [[OUT:%.*]], float [[XR:%.*]], float [[XI:%.*]], float [[WR:%.*]], float [[WI:%.*]]) #[[ATTR0:[0-9]+]] {
-; CHECK-NEXT: [[TMP1:%.*]] = insertelement <2 x float> poison, float [[XR]], i64 0
-; CHECK-NEXT: [[TMP2:%.*]] = insertelement <2 x float> [[TMP1]], float [[XI]], i64 1
-; CHECK-NEXT: [[TMP3:%.*]] = insertelement <2 x float> poison, float [[WR]], i64 0
-; CHECK-NEXT: [[TMP4:%.*]] = shufflevector <2 x float> [[TMP3]], <2 x float> poison, <2 x i32> zeroinitializer
-; CHECK-NEXT: [[TMP5:%.*]] = fmul contract <2 x float> [[TMP2]], [[TMP4]]
-; CHECK-NEXT: [[TMP6:%.*]] = insertelement <2 x float> poison, float [[WI]], i64 0
-; CHECK-NEXT: [[TMP7:%.*]] = shufflevector <2 x float> [[TMP6]], <2 x float> poison, <2 x i32> zeroinitializer
-; CHECK-NEXT: [[TMP8:%.*]] = fmul contract <2 x float> [[TMP2]], [[TMP7]]
-; CHECK-NEXT: [[TMP9:%.*]] = shufflevector <2 x float> [[TMP8]], <2 x float> poison, <2 x i32> <i32 1, i32 0>
-; CHECK-NEXT: [[TMP10:%.*]] = fsub contract <2 x float> [[TMP5]], [[TMP9]]
-; CHECK-NEXT: [[TMP11:%.*]] = fadd contract <2 x float> [[TMP5]], [[TMP9]]
-; CHECK-NEXT: [[TMP12:%.*]] = shufflevector <2 x float> [[TMP10]], <2 x float> [[TMP11]], <2 x i32> <i32 0, i32 3>
-; CHECK-NEXT: store <2 x float> [[TMP12]], ptr addrspace(1) [[OUT]], align 8
-; CHECK-NEXT: ret void
-;
%xr.wr = fmul contract float %xr, %wr
%xi.wi = fmul contract float %xi, %wi
%re = fsub contract float %xr.wr, %xi.wi
>From ed38f91e967af1e0265be73ed532b2bcfef2e841 Mon Sep 17 00:00:00 2001
From: Gheorghe-Teodor Bercea <dobercea at amd.com>
Date: Thu, 24 Sep 2026 09:07:22 -0400
Subject: [PATCH 3/3] Use new triple
---
llvm/test/Transforms/SLPVectorizer/AMDGPU/complex-mul-fma.ll | 3 +--
1 file changed, 1 insertion(+), 2 deletions(-)
diff --git a/llvm/test/Transforms/SLPVectorizer/AMDGPU/complex-mul-fma.ll b/llvm/test/Transforms/SLPVectorizer/AMDGPU/complex-mul-fma.ll
index b4c0de7fcf65cf..3d3b2c151ecf0a 100644
--- a/llvm/test/Transforms/SLPVectorizer/AMDGPU/complex-mul-fma.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AMDGPU/complex-mul-fma.ll
@@ -1,5 +1,4 @@
-; RUN: opt -passes=slp-vectorizer -mtriple=amdgcn-amd-amdhsa -mcpu=gfx942 -pass-remarks=slp-vectorizer -disable-output < %s 2>&1 | FileCheck %s
-; RUN: opt -passes=slp-vectorizer -mtriple=amdgcn-amd-amdhsa -mcpu=gfx950 -pass-remarks=slp-vectorizer -disable-output < %s 2>&1 | FileCheck %s
+; RUN: opt -passes=slp-vectorizer -mtriple=amdgpu9.42-amd-amdhsa -pass-remarks=slp-vectorizer -disable-output < %s 2>&1 | FileCheck %s
; The second fmul of each fsub/fadd does not fuse, so packing those pays off.
; CHECK: Stores SLP vectorized with cost -2
More information about the llvm-commits
mailing list