[llvm] [SLP] Fix canConvertToFMA operand selection and fmul costing (PR #216425)
Dmitry Sidorov via llvm-commits
llvm-commits at lists.llvm.org
Mon Aug 17 02:59:29 PDT 2026
https://github.com/MrSidims updated https://github.com/llvm/llvm-project/pull/216425
>From 554fc8e30174ad10479dfcff6859da0b4f2f83b0 Mon Sep 17 00:00:00 2001
From: Dmitry Sidorov <Dmitry.Sidorov at amd.com>
Date: Fri, 14 Aug 2026 12:51:56 +0200
Subject: [PATCH 1/7] [SLP] Fix canConvertToFMA operand selection and fmul
costing
canConvertToFMA only looked for the fmul on operand 0 of the fadd, so the
accumulator shape fadd acc, a * b was never recognized. Check both operands
for chains that do not allow reassociation.
Also price the unfused fmul without a context instruction. Targets that
model fma fusion price a contractable fmul as free, which discounted the
scalar side of the comparison too and fmuladd never looked profitable.
---
.../Transforms/Vectorize/SLPVectorizer.cpp | 49 +++++--
.../AMDGPU/elementwise-fma-operand1.ll | 130 ++++++++++++++++++
2 files changed, 169 insertions(+), 10 deletions(-)
create mode 100644 llvm/test/Transforms/SLPVectorizer/AMDGPU/elementwise-fma-operand1.ll
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 53816d49de722..6e5628104f0f9 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -14382,18 +14382,44 @@ static InstructionCost canConvertToFMA(ArrayRef<Value *> VL,
InstructionsCompatibilityAnalysis Analysis(DT, DL, TTI, TLI);
SmallVector<BoUpSLP::ValueList> Operands = Analysis.buildOperands(S, VL);
- InstructionsState OpS = getSameOpcode(Operands.front(), TLI);
- if (!OpS.valid())
- return InstructionCost::getInvalid();
-
- if (OpS.isAltShuffle() || OpS.getOpcode() != Instruction::FMul)
- return InstructionCost::getInvalid();
- if (!CheckForContractable(Operands.front()))
+ // The fmul may sit on either side of the add/sub. Look past operand 0 only
+ // for chains that do not allow reassociation. A reassociative chain can be
+ // vectorized into a vector fmul feeding a reduction, which is usually
+ // better than the scalar fma chain this check protects.
+ bool AllowReassoc = any_of(VL, [](Value *V) {
+ auto *FPCI = dyn_cast<FPMathOperator>(V);
+ return FPCI && FPCI->getFastMathFlags().allowReassoc();
+ });
+ auto GetFMulOperandIdx = [&]() -> std::optional<unsigned> {
+ for (unsigned Idx : seq<unsigned>(0, AllowReassoc ? 1 : Operands.size())) {
+ InstructionsState CandS = getSameOpcode(Operands[Idx], TLI);
+ if (!CandS.valid() || CandS.isAltShuffle() ||
+ CandS.getOpcode() != Instruction::FMul)
+ continue;
+ if (!CheckForContractable(Operands[Idx]))
+ continue;
+ return Idx;
+ }
+ return std::nullopt;
+ };
+ std::optional<unsigned> FMulIdx = GetFMulOperandIdx();
+ if (!FMulIdx)
return InstructionCost::getInvalid();
+ InstructionsState OpS = getSameOpcode(Operands[*FMulIdx], TLI);
// Compare the costs.
InstructionCost FMulPlusFAddCost = 0;
InstructionCost FMACost = 0;
constexpr TTI::TargetCostKind CostKind = TTI::TCK_RecipThroughput;
+ // Price the fmul as not fused with its user. Passing a context instruction
+ // would let targets that model the fusion discount the unfused side of the
+ // comparison as well.
+ auto GetUnfusedFMulCost = [&](Instruction *I) {
+ TTI::OperandValueInfo Op1Info = TTI::getOperandInfo(I->getOperand(0));
+ TTI::OperandValueInfo Op2Info = TTI::getOperandInfo(I->getOperand(1));
+ return TTI.getArithmeticInstrCost(Instruction::FMul, I->getType(), CostKind,
+ Op1Info, Op2Info,
+ {I->getOperand(0), I->getOperand(1)});
+ };
FastMathFlags FMF;
FMF.set();
for (Value *V : VL) {
@@ -14406,7 +14432,7 @@ static InstructionCost canConvertToFMA(ArrayRef<Value *> VL,
FMulPlusFAddCost += TTI.getInstructionCost(I, CostKind);
}
unsigned NumOps = 0;
- for (auto [V, Op] : zip(VL, Operands.front())) {
+ for (auto [V, Op] : zip(VL, Operands[*FMulIdx])) {
if (S.isCopyableElement(V))
continue;
auto *I = dyn_cast<Instruction>(Op);
@@ -14420,7 +14446,9 @@ static InstructionCost canConvertToFMA(ArrayRef<Value *> VL,
++NumOps;
if (auto *FPCI = dyn_cast<FPMathOperator>(I))
FMF &= FPCI->getFastMathFlags();
- FMulPlusFAddCost += TTI.getInstructionCost(I, CostKind);
+ FMulPlusFAddCost += I->getOpcode() == Instruction::FMul
+ ? GetUnfusedFMulCost(I)
+ : TTI.getInstructionCost(I, CostKind);
}
Type *Ty = VL.front()->getType();
IntrinsicCostAttributes ICA(Intrinsic::fmuladd, Ty, {Ty, Ty, Ty}, FMF);
@@ -15225,7 +15253,8 @@ void BoUpSLP::transformNodes() {
break;
// This node is a fmuladd node.
E.CombinedOp = TreeEntry::FMulAdd;
- TreeEntry *FMulEntry = getOperandEntry(&E, 0);
+ TreeEntry *FMulEntry =
+ getOperandEntry(&E, IsOneUseVectorFMulOperand(LHS) ? 0 : 1);
if (FMulEntry->UserTreeIndex &&
FMulEntry->State == TreeEntry::Vectorize) {
// The FMul node is part of the combined fmuladd node.
diff --git a/llvm/test/Transforms/SLPVectorizer/AMDGPU/elementwise-fma-operand1.ll b/llvm/test/Transforms/SLPVectorizer/AMDGPU/elementwise-fma-operand1.ll
new file mode 100644
index 0000000000000..efa8f5c4dd930
--- /dev/null
+++ b/llvm/test/Transforms/SLPVectorizer/AMDGPU/elementwise-fma-operand1.ll
@@ -0,0 +1,130 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -passes=slp-vectorizer -S -mtriple=amdgcn-amd-amdhsa -mcpu=gfx90a < %s | FileCheck %s
+
+; Elementwise d = c + a * b where the fmul is operand 1 of the fadd. Marking
+; a hardcoded operand 0 turned the c load bundle into a CombinedVectorize
+; node and made vectorizeTree abort.
+
+define void @axpy4_contract(ptr noalias %d, ptr noalias %a, ptr noalias %b, ptr noalias %c) {
+; CHECK-LABEL: define void @axpy4_contract(
+; CHECK-SAME: ptr noalias [[D:%.*]], ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], ptr noalias [[C:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[TMP0:%.*]] = load <2 x float>, ptr [[C]], align 4
+; CHECK-NEXT: [[TMP1:%.*]] = load <2 x float>, ptr [[A]], align 4
+; CHECK-NEXT: [[TMP2:%.*]] = load <2 x float>, ptr [[B]], align 4
+; CHECK-NEXT: [[TMP3:%.*]] = fmul contract <2 x float> [[TMP1]], [[TMP2]]
+; CHECK-NEXT: [[TMP4:%.*]] = fadd contract <2 x float> [[TMP0]], [[TMP3]]
+; CHECK-NEXT: store <2 x float> [[TMP4]], ptr [[D]], align 4
+; CHECK-NEXT: [[CP2:%.*]] = getelementptr inbounds float, ptr [[C]], i64 2
+; CHECK-NEXT: [[AP2:%.*]] = getelementptr inbounds float, ptr [[A]], i64 2
+; CHECK-NEXT: [[BP2:%.*]] = getelementptr inbounds float, ptr [[B]], i64 2
+; CHECK-NEXT: [[DP2:%.*]] = getelementptr inbounds float, ptr [[D]], i64 2
+; CHECK-NEXT: [[TMP5:%.*]] = load <2 x float>, ptr [[CP2]], align 4
+; CHECK-NEXT: [[TMP6:%.*]] = load <2 x float>, ptr [[AP2]], align 4
+; CHECK-NEXT: [[TMP7:%.*]] = load <2 x float>, ptr [[BP2]], align 4
+; CHECK-NEXT: [[TMP8:%.*]] = fmul contract <2 x float> [[TMP6]], [[TMP7]]
+; CHECK-NEXT: [[TMP9:%.*]] = fadd contract <2 x float> [[TMP5]], [[TMP8]]
+; CHECK-NEXT: store <2 x float> [[TMP9]], ptr [[DP2]], align 4
+; CHECK-NEXT: ret void
+;
+entry:
+ %c0 = load float, ptr %c, align 4
+ %a0 = load float, ptr %a, align 4
+ %b0 = load float, ptr %b, align 4
+ %m0 = fmul contract float %a0, %b0
+ %r0 = fadd contract float %c0, %m0
+ store float %r0, ptr %d, align 4
+ %cp1 = getelementptr inbounds float, ptr %c, i64 1
+ %c1 = load float, ptr %cp1, align 4
+ %ap1 = getelementptr inbounds float, ptr %a, i64 1
+ %a1 = load float, ptr %ap1, align 4
+ %bp1 = getelementptr inbounds float, ptr %b, i64 1
+ %b1 = load float, ptr %bp1, align 4
+ %m1 = fmul contract float %a1, %b1
+ %r1 = fadd contract float %c1, %m1
+ %dp1 = getelementptr inbounds float, ptr %d, i64 1
+ store float %r1, ptr %dp1, align 4
+ %cp2 = getelementptr inbounds float, ptr %c, i64 2
+ %c2 = load float, ptr %cp2, align 4
+ %ap2 = getelementptr inbounds float, ptr %a, i64 2
+ %a2 = load float, ptr %ap2, align 4
+ %bp2 = getelementptr inbounds float, ptr %b, i64 2
+ %b2 = load float, ptr %bp2, align 4
+ %m2 = fmul contract float %a2, %b2
+ %r2 = fadd contract float %c2, %m2
+ %dp2 = getelementptr inbounds float, ptr %d, i64 2
+ store float %r2, ptr %dp2, align 4
+ %cp3 = getelementptr inbounds float, ptr %c, i64 3
+ %c3 = load float, ptr %cp3, align 4
+ %ap3 = getelementptr inbounds float, ptr %a, i64 3
+ %a3 = load float, ptr %ap3, align 4
+ %bp3 = getelementptr inbounds float, ptr %b, i64 3
+ %b3 = load float, ptr %bp3, align 4
+ %m3 = fmul contract float %a3, %b3
+ %r3 = fadd contract float %c3, %m3
+ %dp3 = getelementptr inbounds float, ptr %d, i64 3
+ store float %r3, ptr %dp3, align 4
+ ret void
+}
+
+define void @axpy4_reassoc(ptr noalias %d, ptr noalias %a, ptr noalias %b, ptr noalias %c) {
+; CHECK-LABEL: define void @axpy4_reassoc(
+; CHECK-SAME: ptr noalias [[D:%.*]], ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], ptr noalias [[C:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[TMP0:%.*]] = load <2 x float>, ptr [[C]], align 4
+; CHECK-NEXT: [[TMP1:%.*]] = load <2 x float>, ptr [[A]], align 4
+; CHECK-NEXT: [[TMP2:%.*]] = load <2 x float>, ptr [[B]], align 4
+; CHECK-NEXT: [[TMP3:%.*]] = fmul reassoc contract <2 x float> [[TMP1]], [[TMP2]]
+; CHECK-NEXT: [[TMP4:%.*]] = fadd reassoc contract <2 x float> [[TMP0]], [[TMP3]]
+; CHECK-NEXT: store <2 x float> [[TMP4]], ptr [[D]], align 4
+; CHECK-NEXT: [[CP2:%.*]] = getelementptr inbounds float, ptr [[C]], i64 2
+; CHECK-NEXT: [[AP2:%.*]] = getelementptr inbounds float, ptr [[A]], i64 2
+; CHECK-NEXT: [[BP2:%.*]] = getelementptr inbounds float, ptr [[B]], i64 2
+; CHECK-NEXT: [[DP2:%.*]] = getelementptr inbounds float, ptr [[D]], i64 2
+; CHECK-NEXT: [[TMP5:%.*]] = load <2 x float>, ptr [[CP2]], align 4
+; CHECK-NEXT: [[TMP6:%.*]] = load <2 x float>, ptr [[AP2]], align 4
+; CHECK-NEXT: [[TMP7:%.*]] = load <2 x float>, ptr [[BP2]], align 4
+; CHECK-NEXT: [[TMP8:%.*]] = fmul reassoc contract <2 x float> [[TMP6]], [[TMP7]]
+; CHECK-NEXT: [[TMP9:%.*]] = fadd reassoc contract <2 x float> [[TMP5]], [[TMP8]]
+; CHECK-NEXT: store <2 x float> [[TMP9]], ptr [[DP2]], align 4
+; CHECK-NEXT: ret void
+;
+entry:
+ %c0 = load float, ptr %c, align 4
+ %a0 = load float, ptr %a, align 4
+ %b0 = load float, ptr %b, align 4
+ %m0 = fmul contract reassoc float %a0, %b0
+ %r0 = fadd contract reassoc float %c0, %m0
+ store float %r0, ptr %d, align 4
+ %cp1 = getelementptr inbounds float, ptr %c, i64 1
+ %c1 = load float, ptr %cp1, align 4
+ %ap1 = getelementptr inbounds float, ptr %a, i64 1
+ %a1 = load float, ptr %ap1, align 4
+ %bp1 = getelementptr inbounds float, ptr %b, i64 1
+ %b1 = load float, ptr %bp1, align 4
+ %m1 = fmul contract reassoc float %a1, %b1
+ %r1 = fadd contract reassoc float %c1, %m1
+ %dp1 = getelementptr inbounds float, ptr %d, i64 1
+ store float %r1, ptr %dp1, align 4
+ %cp2 = getelementptr inbounds float, ptr %c, i64 2
+ %c2 = load float, ptr %cp2, align 4
+ %ap2 = getelementptr inbounds float, ptr %a, i64 2
+ %a2 = load float, ptr %ap2, align 4
+ %bp2 = getelementptr inbounds float, ptr %b, i64 2
+ %b2 = load float, ptr %bp2, align 4
+ %m2 = fmul contract reassoc float %a2, %b2
+ %r2 = fadd contract reassoc float %c2, %m2
+ %dp2 = getelementptr inbounds float, ptr %d, i64 2
+ store float %r2, ptr %dp2, align 4
+ %cp3 = getelementptr inbounds float, ptr %c, i64 3
+ %c3 = load float, ptr %cp3, align 4
+ %ap3 = getelementptr inbounds float, ptr %a, i64 3
+ %a3 = load float, ptr %ap3, align 4
+ %bp3 = getelementptr inbounds float, ptr %b, i64 3
+ %b3 = load float, ptr %bp3, align 4
+ %m3 = fmul contract reassoc float %a3, %b3
+ %r3 = fadd contract reassoc float %c3, %m3
+ %dp3 = getelementptr inbounds float, ptr %d, i64 3
+ store float %r3, ptr %dp3, align 4
+ ret void
+}
>From 6f88c8a36a58695280f6390846e80bbbbf580c13 Mon Sep 17 00:00:00 2001
From: Dmitry Sidorov <Dmitry.Sidorov at amd.com>
Date: Sat, 15 Aug 2026 16:48:53 +0200
Subject: [PATCH 2/7] apply comments
---
llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp | 11 ++++-------
1 file changed, 4 insertions(+), 7 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 6e5628104f0f9..67c52d4e8cc52 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -14386,10 +14386,8 @@ static InstructionCost canConvertToFMA(ArrayRef<Value *> VL,
// for chains that do not allow reassociation. A reassociative chain can be
// vectorized into a vector fmul feeding a reduction, which is usually
// better than the scalar fma chain this check protects.
- bool AllowReassoc = any_of(VL, [](Value *V) {
- auto *FPCI = dyn_cast<FPMathOperator>(V);
- return FPCI && FPCI->getFastMathFlags().allowReassoc();
- });
+ bool AllowReassoc = any_of(
+ VL, [](Value *V) { return match(V, m_AllowReassoc(m_Value())); });
auto GetFMulOperandIdx = [&]() -> std::optional<unsigned> {
for (unsigned Idx : seq<unsigned>(0, AllowReassoc ? 1 : Operands.size())) {
InstructionsState CandS = getSameOpcode(Operands[Idx], TLI);
@@ -14414,6 +14412,7 @@ static InstructionCost canConvertToFMA(ArrayRef<Value *> VL,
// would let targets that model the fusion discount the unfused side of the
// comparison as well.
auto GetUnfusedFMulCost = [&](Instruction *I) {
+ assert(I->getOpcode() == Instruction::FMul && "Expected an fmul");
TTI::OperandValueInfo Op1Info = TTI::getOperandInfo(I->getOperand(0));
TTI::OperandValueInfo Op2Info = TTI::getOperandInfo(I->getOperand(1));
return TTI.getArithmeticInstrCost(Instruction::FMul, I->getType(), CostKind,
@@ -14446,9 +14445,7 @@ static InstructionCost canConvertToFMA(ArrayRef<Value *> VL,
++NumOps;
if (auto *FPCI = dyn_cast<FPMathOperator>(I))
FMF &= FPCI->getFastMathFlags();
- FMulPlusFAddCost += I->getOpcode() == Instruction::FMul
- ? GetUnfusedFMulCost(I)
- : TTI.getInstructionCost(I, CostKind);
+ FMulPlusFAddCost += GetUnfusedFMulCost(I);
}
Type *Ty = VL.front()->getType();
IntrinsicCostAttributes ICA(Intrinsic::fmuladd, Ty, {Ty, Ty, Ty}, FMF);
>From 7f9561465cc4f8f72b5723d2fa201b4300ba245b Mon Sep 17 00:00:00 2001
From: Dmitry Sidorov <Dmitry.Sidorov at amd.com>
Date: Sat, 15 Aug 2026 23:28:39 +0200
Subject: [PATCH 3/7] extra fixes
---
.../Transforms/Vectorize/SLPVectorizer.cpp | 15 +++++---
.../AMDGPU/elementwise-fma-operand1.ll | 36 ++++++++++++++++---
2 files changed, 41 insertions(+), 10 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 67c52d4e8cc52..b74dd7f80f5b8 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -14383,13 +14383,17 @@ static InstructionCost canConvertToFMA(ArrayRef<Value *> VL,
SmallVector<BoUpSLP::ValueList> Operands = Analysis.buildOperands(S, VL);
// The fmul may sit on either side of the add/sub. Look past operand 0 only
- // for chains that do not allow reassociation. A reassociative chain can be
- // vectorized into a vector fmul feeding a reduction, which is usually
- // better than the scalar fma chain this check protects.
+ // for chains that do not allow reassociation. The gate is profitability, not
+ // correctness. A reassociative chain can be vectorized into a vector fmul
+ // feeding a reduction, which is usually better than the scalar fma chain
+ // this check protects. An fsub can only fold an fmul on its left, so stop at
+ // operand 0 there as well.
bool AllowReassoc = any_of(
VL, [](Value *V) { return match(V, m_AllowReassoc(m_Value())); });
+ bool OnlyFirstOperand = AllowReassoc || S.getOpcode() == Instruction::FSub;
auto GetFMulOperandIdx = [&]() -> std::optional<unsigned> {
- for (unsigned Idx : seq<unsigned>(0, AllowReassoc ? 1 : Operands.size())) {
+ for (unsigned Idx :
+ seq<unsigned>(0, OnlyFirstOperand ? 1 : Operands.size())) {
InstructionsState CandS = getSameOpcode(Operands[Idx], TLI);
if (!CandS.valid() || CandS.isAltShuffle() ||
CandS.getOpcode() != Instruction::FMul)
@@ -14428,7 +14432,8 @@ static InstructionCost canConvertToFMA(ArrayRef<Value *> VL,
if (!S.isCopyableElement(I))
if (auto *FPCI = dyn_cast<FPMathOperator>(I))
FMF &= FPCI->getFastMathFlags();
- FMulPlusFAddCost += TTI.getInstructionCost(I, CostKind);
+ FMulPlusFAddCost +=
+ TTI.getArithmeticInstrCost(S.getOpcode(), I->getType(), CostKind);
}
unsigned NumOps = 0;
for (auto [V, Op] : zip(VL, Operands[*FMulIdx])) {
diff --git a/llvm/test/Transforms/SLPVectorizer/AMDGPU/elementwise-fma-operand1.ll b/llvm/test/Transforms/SLPVectorizer/AMDGPU/elementwise-fma-operand1.ll
index 44f55e536179a..b43c0a8fdf91c 100644
--- a/llvm/test/Transforms/SLPVectorizer/AMDGPU/elementwise-fma-operand1.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AMDGPU/elementwise-fma-operand1.ll
@@ -2,12 +2,16 @@
; RUN: opt -passes=slp-vectorizer -S -mtriple=amdgcn-amd-amdhsa -mcpu=gfx90a -slp-threshold=14 < %s | FileCheck %s
; RUN: opt -passes=slp-vectorizer -S -mtriple=amdgcn-amd-amdhsa -mcpu=gfx942 -slp-threshold=14 < %s | FileCheck %s
; RUN: opt -passes=slp-vectorizer -S -mtriple=amdgcn-amd-amdhsa -mcpu=gfx950 -slp-threshold=14 < %s | FileCheck %s
+; RUN: opt -passes=slp-vectorizer -S -mtriple=amdgcn-amd-amdhsa -mcpu=gfx90a -slp-threshold=12 < %s | FileCheck %s --check-prefix=THR12
+; RUN: opt -passes=slp-vectorizer -S -mtriple=amdgcn-amd-amdhsa -mcpu=gfx942 -slp-threshold=12 < %s | FileCheck %s --check-prefix=THR12
+; RUN: opt -passes=slp-vectorizer -S -mtriple=amdgcn-amd-amdhsa -mcpu=gfx950 -slp-threshold=12 < %s | FileCheck %s --check-prefix=THR12
-; Elementwise d = c + a * b, where the fmul is operand 1 of the fadd. The
-; threshold puts the decision right at the cost boundary, so how the fmul and
-; fadd are priced against a fused fma is what decides it. These targets halve
-; the cost of a packed fmul, which is what tempts SLP into vectorizing and
-; breaking the scalar fma chain.
+; Elementwise d = c + a * b, where the fmul is operand 1 of the fadd. These
+; targets halve the cost of a packed fmul, so SLP is tempted to vectorize and
+; break the scalar fma chain. The 14 runs sit at the cost boundary. The 12 runs
+; vectorize either way and guard against the fmuladd marking landing on the load
+; at operand 0 after the fma detection picked the fmul at operand 1, which
+; asserts.
define void @axpy4_contract(ptr noalias %d, ptr noalias %a, ptr noalias %b, ptr noalias %c) {
; CHECK-LABEL: define void @axpy4_contract(
@@ -51,6 +55,17 @@ define void @axpy4_contract(ptr noalias %d, ptr noalias %a, ptr noalias %b, ptr
; CHECK-NEXT: store float [[R3]], ptr [[DP3]], align 4
; CHECK-NEXT: ret void
;
+; THR12-LABEL: define void @axpy4_contract(
+; THR12-SAME: ptr noalias [[D:%.*]], ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], ptr noalias [[C:%.*]]) #[[ATTR0:[0-9]+]] {
+; THR12-NEXT: [[ENTRY:.*:]]
+; THR12-NEXT: [[TMP0:%.*]] = load <4 x float>, ptr [[C]], align 4
+; THR12-NEXT: [[TMP1:%.*]] = load <4 x float>, ptr [[A]], align 4
+; THR12-NEXT: [[TMP2:%.*]] = load <4 x float>, ptr [[B]], align 4
+; THR12-NEXT: [[TMP3:%.*]] = fmul contract <4 x float> [[TMP1]], [[TMP2]]
+; THR12-NEXT: [[TMP4:%.*]] = fadd contract <4 x float> [[TMP0]], [[TMP3]]
+; THR12-NEXT: store <4 x float> [[TMP4]], ptr [[D]], align 4
+; THR12-NEXT: ret void
+;
entry:
%c0 = load float, ptr %c, align 4
%a0 = load float, ptr %a, align 4
@@ -103,6 +118,17 @@ define void @axpy4_reassoc(ptr noalias %d, ptr noalias %a, ptr noalias %b, ptr n
; CHECK-NEXT: store <4 x float> [[TMP4]], ptr [[D]], align 4
; CHECK-NEXT: ret void
;
+; THR12-LABEL: define void @axpy4_reassoc(
+; THR12-SAME: ptr noalias [[D:%.*]], ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], ptr noalias [[C:%.*]]) #[[ATTR0]] {
+; THR12-NEXT: [[ENTRY:.*:]]
+; THR12-NEXT: [[TMP0:%.*]] = load <4 x float>, ptr [[C]], align 4
+; THR12-NEXT: [[TMP1:%.*]] = load <4 x float>, ptr [[A]], align 4
+; THR12-NEXT: [[TMP2:%.*]] = load <4 x float>, ptr [[B]], align 4
+; THR12-NEXT: [[TMP3:%.*]] = fmul reassoc contract <4 x float> [[TMP1]], [[TMP2]]
+; THR12-NEXT: [[TMP4:%.*]] = fadd reassoc contract <4 x float> [[TMP0]], [[TMP3]]
+; THR12-NEXT: store <4 x float> [[TMP4]], ptr [[D]], align 4
+; THR12-NEXT: ret void
+;
entry:
%c0 = load float, ptr %c, align 4
%a0 = load float, ptr %a, align 4
>From 325b41fcbad6bbc3a1f1d3880d9498fcf717890f Mon Sep 17 00:00:00 2001
From: Dmitry Sidorov <Dmitry.Sidorov at amd.com>
Date: Sun, 16 Aug 2026 11:36:35 +0200
Subject: [PATCH 4/7] format
---
llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp | 4 ++--
1 file changed, 2 insertions(+), 2 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index b74dd7f80f5b8..fcf526bfecb75 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -14388,8 +14388,8 @@ static InstructionCost canConvertToFMA(ArrayRef<Value *> VL,
// feeding a reduction, which is usually better than the scalar fma chain
// this check protects. An fsub can only fold an fmul on its left, so stop at
// operand 0 there as well.
- bool AllowReassoc = any_of(
- VL, [](Value *V) { return match(V, m_AllowReassoc(m_Value())); });
+ bool AllowReassoc =
+ any_of(VL, [](Value *V) { return match(V, m_AllowReassoc(m_Value())); });
bool OnlyFirstOperand = AllowReassoc || S.getOpcode() == Instruction::FSub;
auto GetFMulOperandIdx = [&]() -> std::optional<unsigned> {
for (unsigned Idx :
>From f57929be7204d307d599b151f439bd5b269096c7 Mon Sep 17 00:00:00 2001
From: Dmitry Sidorov <Dmitry.Sidorov at amd.com>
Date: Sun, 16 Aug 2026 16:55:28 +0200
Subject: [PATCH 5/7] Require the whole bundle to be reassoc, and take the
operand count from the helper
---
llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp | 8 +++++---
1 file changed, 5 insertions(+), 3 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index fcf526bfecb75..0432a9deac6ea 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -14389,11 +14389,13 @@ static InstructionCost canConvertToFMA(ArrayRef<Value *> VL,
// this check protects. An fsub can only fold an fmul on its left, so stop at
// operand 0 there as well.
bool AllowReassoc =
- any_of(VL, [](Value *V) { return match(V, m_AllowReassoc(m_Value())); });
+ all_of(VL, [](Value *V) { return match(V, m_AllowReassoc(m_Value())); });
bool OnlyFirstOperand = AllowReassoc || S.getOpcode() == Instruction::FSub;
+ unsigned NumCandidateOps =
+ OnlyFirstOperand ? 1
+ : getNumberOfPotentiallyCommutativeOps(S.getMainOp());
auto GetFMulOperandIdx = [&]() -> std::optional<unsigned> {
- for (unsigned Idx :
- seq<unsigned>(0, OnlyFirstOperand ? 1 : Operands.size())) {
+ for (unsigned Idx : seq<unsigned>(0, NumCandidateOps)) {
InstructionsState CandS = getSameOpcode(Operands[Idx], TLI);
if (!CandS.valid() || CandS.isAltShuffle() ||
CandS.getOpcode() != Instruction::FMul)
>From 7221528e02cf4a6322cec42297137c21aac54126 Mon Sep 17 00:00:00 2001
From: Dmitry Sidorov <Dmitry.Sidorov at amd.com>
Date: Sun, 16 Aug 2026 18:22:10 +0200
Subject: [PATCH 6/7] Cover the mixed-reassoc bundle the all_of change fixes
---
.../AMDGPU/elementwise-fma-operand1.ll | 126 +++++++++++++++++-
1 file changed, 125 insertions(+), 1 deletion(-)
diff --git a/llvm/test/Transforms/SLPVectorizer/AMDGPU/elementwise-fma-operand1.ll b/llvm/test/Transforms/SLPVectorizer/AMDGPU/elementwise-fma-operand1.ll
index b43c0a8fdf91c..eed8216c5b1d2 100644
--- a/llvm/test/Transforms/SLPVectorizer/AMDGPU/elementwise-fma-operand1.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AMDGPU/elementwise-fma-operand1.ll
@@ -11,7 +11,8 @@
; break the scalar fma chain. The 14 runs sit at the cost boundary. The 12 runs
; vectorize either way and guard against the fmuladd marking landing on the load
; at operand 0 after the fma detection picked the fmul at operand 1, which
-; asserts.
+; asserts. axpy4_mixed_reassoc carries reassoc on one lane only, so the whole
+; bundle has to be reassociative before the search gives up on operand 1.
define void @axpy4_contract(ptr noalias %d, ptr noalias %a, ptr noalias %b, ptr noalias %c) {
; CHECK-LABEL: define void @axpy4_contract(
@@ -168,3 +169,126 @@ entry:
store float %r3, ptr %dp3, align 4
ret void
}
+
+define void @axpy4_mixed_reassoc(ptr noalias %d, ptr noalias %a, ptr noalias %b, ptr noalias %c) {
+; CHECK-LABEL: define void @axpy4_mixed_reassoc(
+; CHECK-SAME: ptr noalias [[D:%.*]], ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], ptr noalias [[C:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[C0:%.*]] = load float, ptr [[C]], align 4
+; CHECK-NEXT: [[A0:%.*]] = load float, ptr [[A]], align 4
+; CHECK-NEXT: [[B0:%.*]] = load float, ptr [[B]], align 4
+; CHECK-NEXT: [[M0:%.*]] = fmul contract float [[A0]], [[B0]]
+; CHECK-NEXT: [[R0:%.*]] = fadd reassoc contract float [[C0]], [[M0]]
+; CHECK-NEXT: store float [[R0]], ptr [[D]], align 4
+; CHECK-NEXT: [[CP1:%.*]] = getelementptr inbounds float, ptr [[C]], i64 1
+; CHECK-NEXT: [[C1:%.*]] = load float, ptr [[CP1]], align 4
+; CHECK-NEXT: [[AP1:%.*]] = getelementptr inbounds float, ptr [[A]], i64 1
+; CHECK-NEXT: [[A1:%.*]] = load float, ptr [[AP1]], align 4
+; CHECK-NEXT: [[BP1:%.*]] = getelementptr inbounds float, ptr [[B]], i64 1
+; CHECK-NEXT: [[B1:%.*]] = load float, ptr [[BP1]], align 4
+; CHECK-NEXT: [[M1:%.*]] = fmul contract float [[A1]], [[B1]]
+; CHECK-NEXT: [[R1:%.*]] = fadd contract float [[C1]], [[M1]]
+; CHECK-NEXT: [[DP1:%.*]] = getelementptr inbounds float, ptr [[D]], i64 1
+; CHECK-NEXT: store float [[R1]], ptr [[DP1]], align 4
+; CHECK-NEXT: [[CP2:%.*]] = getelementptr inbounds float, ptr [[C]], i64 2
+; CHECK-NEXT: [[C2:%.*]] = load float, ptr [[CP2]], align 4
+; CHECK-NEXT: [[AP2:%.*]] = getelementptr inbounds float, ptr [[A]], i64 2
+; CHECK-NEXT: [[A2:%.*]] = load float, ptr [[AP2]], align 4
+; CHECK-NEXT: [[BP2:%.*]] = getelementptr inbounds float, ptr [[B]], i64 2
+; CHECK-NEXT: [[B2:%.*]] = load float, ptr [[BP2]], align 4
+; CHECK-NEXT: [[M2:%.*]] = fmul contract float [[A2]], [[B2]]
+; CHECK-NEXT: [[R2:%.*]] = fadd contract float [[C2]], [[M2]]
+; CHECK-NEXT: [[DP2:%.*]] = getelementptr inbounds float, ptr [[D]], i64 2
+; CHECK-NEXT: store float [[R2]], ptr [[DP2]], align 4
+; CHECK-NEXT: [[CP3:%.*]] = getelementptr inbounds float, ptr [[C]], i64 3
+; CHECK-NEXT: [[C3:%.*]] = load float, ptr [[CP3]], align 4
+; CHECK-NEXT: [[AP3:%.*]] = getelementptr inbounds float, ptr [[A]], i64 3
+; CHECK-NEXT: [[A3:%.*]] = load float, ptr [[AP3]], align 4
+; CHECK-NEXT: [[BP3:%.*]] = getelementptr inbounds float, ptr [[B]], i64 3
+; CHECK-NEXT: [[B3:%.*]] = load float, ptr [[BP3]], align 4
+; CHECK-NEXT: [[M3:%.*]] = fmul contract float [[A3]], [[B3]]
+; CHECK-NEXT: [[R3:%.*]] = fadd contract float [[C3]], [[M3]]
+; CHECK-NEXT: [[DP3:%.*]] = getelementptr inbounds float, ptr [[D]], i64 3
+; CHECK-NEXT: store float [[R3]], ptr [[DP3]], align 4
+; CHECK-NEXT: ret void
+;
+; THR12-LABEL: define void @axpy4_mixed_reassoc(
+; THR12-SAME: ptr noalias [[D:%.*]], ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], ptr noalias [[C:%.*]]) #[[ATTR0]] {
+; THR12-NEXT: [[ENTRY:.*:]]
+; THR12-NEXT: [[C0:%.*]] = load float, ptr [[C]], align 4
+; THR12-NEXT: [[A0:%.*]] = load float, ptr [[A]], align 4
+; THR12-NEXT: [[B0:%.*]] = load float, ptr [[B]], align 4
+; THR12-NEXT: [[M0:%.*]] = fmul contract float [[A0]], [[B0]]
+; THR12-NEXT: [[R0:%.*]] = fadd reassoc contract float [[C0]], [[M0]]
+; THR12-NEXT: store float [[R0]], ptr [[D]], align 4
+; THR12-NEXT: [[CP1:%.*]] = getelementptr inbounds float, ptr [[C]], i64 1
+; THR12-NEXT: [[C1:%.*]] = load float, ptr [[CP1]], align 4
+; THR12-NEXT: [[AP1:%.*]] = getelementptr inbounds float, ptr [[A]], i64 1
+; THR12-NEXT: [[A1:%.*]] = load float, ptr [[AP1]], align 4
+; THR12-NEXT: [[BP1:%.*]] = getelementptr inbounds float, ptr [[B]], i64 1
+; THR12-NEXT: [[B1:%.*]] = load float, ptr [[BP1]], align 4
+; THR12-NEXT: [[M1:%.*]] = fmul contract float [[A1]], [[B1]]
+; THR12-NEXT: [[R1:%.*]] = fadd contract float [[C1]], [[M1]]
+; THR12-NEXT: [[DP1:%.*]] = getelementptr inbounds float, ptr [[D]], i64 1
+; THR12-NEXT: store float [[R1]], ptr [[DP1]], align 4
+; THR12-NEXT: [[CP2:%.*]] = getelementptr inbounds float, ptr [[C]], i64 2
+; THR12-NEXT: [[C2:%.*]] = load float, ptr [[CP2]], align 4
+; THR12-NEXT: [[AP2:%.*]] = getelementptr inbounds float, ptr [[A]], i64 2
+; THR12-NEXT: [[A2:%.*]] = load float, ptr [[AP2]], align 4
+; THR12-NEXT: [[BP2:%.*]] = getelementptr inbounds float, ptr [[B]], i64 2
+; THR12-NEXT: [[B2:%.*]] = load float, ptr [[BP2]], align 4
+; THR12-NEXT: [[M2:%.*]] = fmul contract float [[A2]], [[B2]]
+; THR12-NEXT: [[R2:%.*]] = fadd contract float [[C2]], [[M2]]
+; THR12-NEXT: [[DP2:%.*]] = getelementptr inbounds float, ptr [[D]], i64 2
+; THR12-NEXT: store float [[R2]], ptr [[DP2]], align 4
+; THR12-NEXT: [[CP3:%.*]] = getelementptr inbounds float, ptr [[C]], i64 3
+; THR12-NEXT: [[C3:%.*]] = load float, ptr [[CP3]], align 4
+; THR12-NEXT: [[AP3:%.*]] = getelementptr inbounds float, ptr [[A]], i64 3
+; THR12-NEXT: [[A3:%.*]] = load float, ptr [[AP3]], align 4
+; THR12-NEXT: [[BP3:%.*]] = getelementptr inbounds float, ptr [[B]], i64 3
+; THR12-NEXT: [[B3:%.*]] = load float, ptr [[BP3]], align 4
+; THR12-NEXT: [[M3:%.*]] = fmul contract float [[A3]], [[B3]]
+; THR12-NEXT: [[R3:%.*]] = fadd contract float [[C3]], [[M3]]
+; THR12-NEXT: [[DP3:%.*]] = getelementptr inbounds float, ptr [[D]], i64 3
+; THR12-NEXT: store float [[R3]], ptr [[DP3]], align 4
+; THR12-NEXT: ret void
+;
+entry:
+ %c0 = load float, ptr %c, align 4
+ %a0 = load float, ptr %a, align 4
+ %b0 = load float, ptr %b, align 4
+ %m0 = fmul contract float %a0, %b0
+ %r0 = fadd contract reassoc float %c0, %m0
+ store float %r0, ptr %d, align 4
+ %cp1 = getelementptr inbounds float, ptr %c, i64 1
+ %c1 = load float, ptr %cp1, align 4
+ %ap1 = getelementptr inbounds float, ptr %a, i64 1
+ %a1 = load float, ptr %ap1, align 4
+ %bp1 = getelementptr inbounds float, ptr %b, i64 1
+ %b1 = load float, ptr %bp1, align 4
+ %m1 = fmul contract float %a1, %b1
+ %r1 = fadd contract float %c1, %m1
+ %dp1 = getelementptr inbounds float, ptr %d, i64 1
+ store float %r1, ptr %dp1, align 4
+ %cp2 = getelementptr inbounds float, ptr %c, i64 2
+ %c2 = load float, ptr %cp2, align 4
+ %ap2 = getelementptr inbounds float, ptr %a, i64 2
+ %a2 = load float, ptr %ap2, align 4
+ %bp2 = getelementptr inbounds float, ptr %b, i64 2
+ %b2 = load float, ptr %bp2, align 4
+ %m2 = fmul contract float %a2, %b2
+ %r2 = fadd contract float %c2, %m2
+ %dp2 = getelementptr inbounds float, ptr %d, i64 2
+ store float %r2, ptr %dp2, align 4
+ %cp3 = getelementptr inbounds float, ptr %c, i64 3
+ %c3 = load float, ptr %cp3, align 4
+ %ap3 = getelementptr inbounds float, ptr %a, i64 3
+ %a3 = load float, ptr %ap3, align 4
+ %bp3 = getelementptr inbounds float, ptr %b, i64 3
+ %b3 = load float, ptr %bp3, align 4
+ %m3 = fmul contract float %a3, %b3
+ %r3 = fadd contract float %c3, %m3
+ %dp3 = getelementptr inbounds float, ptr %d, i64 3
+ store float %r3, ptr %dp3, align 4
+ ret void
+}
>From 8dfbc8435bfa00486938f82f3ce61df2b727df72 Mon Sep 17 00:00:00 2001
From: Dmitry Sidorov <Dmitry.Sidorov at amd.com>
Date: Mon, 17 Aug 2026 01:17:12 +0200
Subject: [PATCH 7/7] Skip undef and copyable lanes in the reassoc check
---
llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp | 8 ++++++--
1 file changed, 6 insertions(+), 2 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index e575918bf582c..bedea0c5ad873 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -14429,8 +14429,12 @@ static InstructionCost canConvertToFMA(ArrayRef<Value *> VL,
// feeding a reduction, which is usually better than the scalar fma chain
// this check protects. An fsub can only fold an fmul on its left, so stop at
// operand 0 there as well.
- bool AllowReassoc =
- all_of(VL, [](Value *V) { return match(V, m_AllowReassoc(m_Value())); });
+ bool AllowReassoc = all_of(VL, [&](Value *V) {
+ auto *I = dyn_cast<Instruction>(V);
+ if (!I || S.isCopyableElement(I))
+ return true;
+ return match(I, m_AllowReassoc(m_Value()));
+ });
bool OnlyFirstOperand = AllowReassoc || S.getOpcode() == Instruction::FSub;
unsigned NumCandidateOps =
OnlyFirstOperand ? 1
More information about the llvm-commits
mailing list