[llvm] [SLP] Let the caller tell canConvertToFMA which operand holds the fmul (PR #218754)
Dmitry Sidorov via llvm-commits
llvm-commits at lists.llvm.org
Wed Sep 9 06:53:45 PDT 2026
https://github.com/MrSidims updated https://github.com/llvm/llvm-project/pull/218754
>From 2eac97260692193b4e11071c05240e6feb47eea9 Mon Sep 17 00:00:00 2001
From: Dmitry Sidorov <Dmitry.Sidorov at amd.com>
Date: Sat, 22 Aug 2026 23:01:36 +0200
Subject: [PATCH 01/11] [SLP] Let the caller tell canConvertToFMA which operand
holds the fmul
canConvertToFMA only ever looked at operand 0 of the fadd/fsub, so a
tree whose fmul sits on the right hand side was never recognized as
fusable.
---
.../Transforms/Vectorize/SLPVectorizer.cpp | 39 +++--
.../SLPVectorizer/X86/fma-operand-index.ll | 26 ++--
.../X86/horizontal-fadd-with-sub.ll | 141 +++++++++++-------
3 files changed, 128 insertions(+), 78 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 50d81058968d4..5ffa5884c323a 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -13653,7 +13653,18 @@ static InstructionCost canConvertToFMA(ArrayRef<Value *> VL,
DominatorTree &DT, const DataLayout &DL,
TargetTransformInfo &TTI,
const TargetLibraryInfo &TLI,
- const TTI::TargetCostKind CostKind);
+ const TTI::TargetCostKind CostKind,
+ unsigned FMulOpIdx = 0);
+
+/// \returns the operand of \p I that is a one-use fmul, 0 if there is none.
+static unsigned getFMulOperandIdx(const Instruction *I) {
+ for (unsigned Idx : seq<unsigned>(0, I->getNumOperands())) {
+ auto *Op = dyn_cast<Instruction>(I->getOperand(Idx));
+ if (Op && Op->getOpcode() == Instruction::FMul && Op->hasOneUse())
+ return Idx;
+ }
+ return 0;
+}
uint64_t BoUpSLP::getNumScalarInsts(bool HasTreeLoop) {
uint64_t Total = 0;
@@ -13734,7 +13745,7 @@ uint64_t BoUpSLP::getNumScalarInsts(bool HasTreeLoop) {
I->getOpcode() != Instruction::FSub))
continue;
if (canConvertToFMA(I, InstructionsState(I, I), *DT, *DL, *TTI, *TLI,
- CostKind)
+ CostKind, getFMulOperandIdx(I))
.isValid()) {
assert(Count > 0 && "Underflow in scalar inst count (fma)");
--Count;
@@ -14457,7 +14468,8 @@ static InstructionCost canConvertToFMA(ArrayRef<Value *> VL,
DominatorTree &DT, const DataLayout &DL,
TargetTransformInfo &TTI,
const TargetLibraryInfo &TLI,
- const TTI::TargetCostKind CostKind) {
+ const TTI::TargetCostKind CostKind,
+ unsigned FMulOpIdx) {
assert(all_of(VL,
[](Value *V) {
return V->getType()->getScalarType()->isFloatingPointTy();
@@ -14489,13 +14501,15 @@ static InstructionCost canConvertToFMA(ArrayRef<Value *> VL,
InstructionsCompatibilityAnalysis Analysis(DT, DL, TTI, TLI);
SmallVector<BoUpSLP::ValueList> Operands = Analysis.buildOperands(S, VL);
- InstructionsState OpS = getSameOpcode(Operands.front(), TLI);
+ if (FMulOpIdx >= Operands.size())
+ return InstructionCost::getInvalid();
+ InstructionsState OpS = getSameOpcode(Operands[FMulOpIdx], TLI);
if (!OpS.valid())
return InstructionCost::getInvalid();
if (OpS.isAltShuffle() || OpS.getOpcode() != Instruction::FMul)
return InstructionCost::getInvalid();
- if (!CheckForContractable(Operands.front(), OpS))
+ if (!CheckForContractable(Operands[FMulOpIdx], OpS))
return InstructionCost::getInvalid();
// Compare the costs.
InstructionCost FMulPlusFAddCost = 0;
@@ -14536,7 +14550,7 @@ static InstructionCost canConvertToFMA(ArrayRef<Value *> VL,
{I->getOperand(0), I->getOperand(1)});
}
unsigned NumOps = 0;
- for (auto [V, Op] : zip(VL, Operands.front())) {
+ for (auto [V, Op] : zip(VL, Operands[FMulOpIdx])) {
if (S.isCopyableElement(V))
continue;
auto *I = dyn_cast<Instruction>(Op);
@@ -17068,9 +17082,10 @@ BoUpSLP::getEntryCost(const TreeEntry *E, ArrayRef<Value *> VectorizedVals,
return IntrinsicCost;
};
auto GetFMulAddCost = [&, &TTI = *TTI](const InstructionsState &S,
- Instruction *VI) {
+ Instruction *VI,
+ unsigned FMulOpIdx = 0) {
InstructionCost Cost =
- canConvertToFMA(VI, S, *DT, *DL, TTI, *TLI, CostKind);
+ canConvertToFMA(VI, S, *DT, *DL, TTI, *TLI, CostKind, FMulOpIdx);
return Cost;
};
switch (ShuffleOrOp) {
@@ -17708,7 +17723,8 @@ BoUpSLP::getEntryCost(const TreeEntry *E, ArrayRef<Value *> VectorizedVals,
if (auto *I = dyn_cast<Instruction>(UniqueValues[Idx]);
I && (ShuffleOrOp == Instruction::FAdd ||
ShuffleOrOp == Instruction::FSub)) {
- InstructionCost IntrinsicCost = GetFMulAddCost(E->getOperations(), I);
+ InstructionCost IntrinsicCost =
+ GetFMulAddCost(E->getOperations(), I, getFMulOperandIdx(I));
if (IntrinsicCost.isValid())
ScalarCost = IntrinsicCost;
}
@@ -33062,7 +33078,7 @@ bool SLPVectorizerPass::tryToVectorize(
(I->getOpcode() == Instruction::FAdd ||
I->getOpcode() == Instruction::FSub) &&
canConvertToFMA(I, getSameOpcode(I, *TLI), *DT, *DL, *TTI, *TLI,
- R.getCostKind())
+ R.getCostKind(), getFMulOperandIdx(I))
.isValid()) {
FMACandidates.insert(I);
return false;
@@ -34338,7 +34354,8 @@ bool SLPVectorizerPass::vectorizeOnceUsedSeeds(BasicBlock *BB, BoUpSLP &R) {
auto *U = cast<Instruction>(I.user_back());
if (InstructionsState S = getSameOpcode(U, *TLI);
S && S.isAddSubLikeOp() &&
- canConvertToFMA(U, S, *DT, *DL, *TTI, *TLI, R.getCostKind())
+ canConvertToFMA(U, S, *DT, *DL, *TTI, *TLI, R.getCostKind(),
+ U->getOperand(1) == &I ? 1 : 0)
.isValid())
continue;
}
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/fma-operand-index.ll b/llvm/test/Transforms/SLPVectorizer/X86/fma-operand-index.ll
index abf8abf106eee..a5b28dd341e7a 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/fma-operand-index.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/fma-operand-index.ll
@@ -10,16 +10,19 @@ define double @fmul_rhs_fadd(ptr %x, ptr %y, ptr %z) {
; CHECK-LABEL: define double @fmul_rhs_fadd(
; CHECK-SAME: ptr [[X:%.*]], ptr [[Y:%.*]], ptr [[Z:%.*]]) #[[ATTR0:[0-9]+]] {
; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[X8:%.*]] = getelementptr inbounds nuw i8, ptr [[X]], i64 8
+; CHECK-NEXT: [[Y8:%.*]] = getelementptr inbounds nuw i8, ptr [[Y]], i64 8
; CHECK-NEXT: [[Z8:%.*]] = getelementptr inbounds nuw i8, ptr [[Z]], i64 8
+; CHECK-NEXT: [[X0:%.*]] = load double, ptr [[X]], align 8
+; CHECK-NEXT: [[Y0:%.*]] = load double, ptr [[Y]], align 8
+; CHECK-NEXT: [[MUL0:%.*]] = fmul reassoc nsz contract double [[Y0]], [[X0]]
; CHECK-NEXT: [[Z0:%.*]] = load double, ptr [[Z]], align 8
-; CHECK-NEXT: [[TMP0:%.*]] = load <2 x double>, ptr [[X]], align 8
-; CHECK-NEXT: [[TMP1:%.*]] = load <2 x double>, ptr [[Y]], align 8
-; CHECK-NEXT: [[TMP2:%.*]] = fmul reassoc nsz contract <2 x double> [[TMP1]], [[TMP0]]
+; CHECK-NEXT: [[X1:%.*]] = load double, ptr [[X8]], align 8
+; CHECK-NEXT: [[Y1:%.*]] = load double, ptr [[Y8]], align 8
+; CHECK-NEXT: [[MUL1:%.*]] = fmul reassoc nsz contract double [[Y1]], [[X1]]
; CHECK-NEXT: [[Z1:%.*]] = load double, ptr [[Z8]], align 8
; CHECK-NEXT: [[ZSUM:%.*]] = fadd reassoc nsz contract double [[Z0]], [[Z1]]
-; CHECK-NEXT: [[MUL0:%.*]] = extractelement <2 x double> [[TMP2]], i64 0
; CHECK-NEXT: [[ADD0:%.*]] = fadd reassoc nsz contract double [[ZSUM]], [[MUL0]]
-; CHECK-NEXT: [[MUL1:%.*]] = extractelement <2 x double> [[TMP2]], i64 1
; CHECK-NEXT: [[ADD1:%.*]] = fadd reassoc nsz contract double [[ADD0]], [[MUL1]]
; CHECK-NEXT: ret double [[ADD1]]
;
@@ -46,16 +49,19 @@ define double @fmul_rhs_fsub(ptr %x, ptr %y, ptr %z) {
; CHECK-LABEL: define double @fmul_rhs_fsub(
; CHECK-SAME: ptr [[X:%.*]], ptr [[Y:%.*]], ptr [[Z:%.*]]) #[[ATTR0]] {
; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[X8:%.*]] = getelementptr inbounds nuw i8, ptr [[X]], i64 8
+; CHECK-NEXT: [[Y8:%.*]] = getelementptr inbounds nuw i8, ptr [[Y]], i64 8
; CHECK-NEXT: [[Z8:%.*]] = getelementptr inbounds nuw i8, ptr [[Z]], i64 8
+; CHECK-NEXT: [[X0:%.*]] = load double, ptr [[X]], align 8
+; CHECK-NEXT: [[Y0:%.*]] = load double, ptr [[Y]], align 8
+; CHECK-NEXT: [[MUL0:%.*]] = fmul reassoc nsz contract double [[Y0]], [[X0]]
; CHECK-NEXT: [[Z0:%.*]] = load double, ptr [[Z]], align 8
-; CHECK-NEXT: [[TMP0:%.*]] = load <2 x double>, ptr [[X]], align 8
-; CHECK-NEXT: [[TMP1:%.*]] = load <2 x double>, ptr [[Y]], align 8
-; CHECK-NEXT: [[TMP2:%.*]] = fmul reassoc nsz contract <2 x double> [[TMP1]], [[TMP0]]
+; CHECK-NEXT: [[X1:%.*]] = load double, ptr [[X8]], align 8
+; CHECK-NEXT: [[Y1:%.*]] = load double, ptr [[Y8]], align 8
+; CHECK-NEXT: [[MUL1:%.*]] = fmul reassoc nsz contract double [[Y1]], [[X1]]
; CHECK-NEXT: [[Z1:%.*]] = load double, ptr [[Z8]], align 8
; CHECK-NEXT: [[ZSUM:%.*]] = fadd reassoc nsz contract double [[Z0]], [[Z1]]
-; CHECK-NEXT: [[MUL0:%.*]] = extractelement <2 x double> [[TMP2]], i64 0
; CHECK-NEXT: [[SUB0:%.*]] = fsub reassoc nsz contract double [[ZSUM]], [[MUL0]]
-; CHECK-NEXT: [[MUL1:%.*]] = extractelement <2 x double> [[TMP2]], i64 1
; CHECK-NEXT: [[SUB1:%.*]] = fsub reassoc nsz contract double [[SUB0]], [[MUL1]]
; CHECK-NEXT: ret double [[SUB1]]
;
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/horizontal-fadd-with-sub.ll b/llvm/test/Transforms/SLPVectorizer/X86/horizontal-fadd-with-sub.ll
index f96ab87e86049..853ab156bed97 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/horizontal-fadd-with-sub.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/horizontal-fadd-with-sub.ll
@@ -9,16 +9,19 @@ define double @fsub_fmul_2(ptr %x, ptr %y, ptr %z) {
; CHECK-LABEL: define double @fsub_fmul_2(
; CHECK-SAME: ptr [[X:%.*]], ptr [[Y:%.*]], ptr [[Z:%.*]]) #[[ATTR0:[0-9]+]] {
; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[X8:%.*]] = getelementptr inbounds nuw i8, ptr [[X]], i64 8
+; CHECK-NEXT: [[Y8:%.*]] = getelementptr inbounds nuw i8, ptr [[Y]], i64 8
; CHECK-NEXT: [[Z8:%.*]] = getelementptr inbounds nuw i8, ptr [[Z]], i64 8
+; CHECK-NEXT: [[X0:%.*]] = load double, ptr [[X]], align 8
+; CHECK-NEXT: [[Y0:%.*]] = load double, ptr [[Y]], align 8
+; CHECK-NEXT: [[TMP5:%.*]] = fmul reassoc nsz contract double [[Y0]], [[X0]]
; CHECK-NEXT: [[Z0:%.*]] = load double, ptr [[Z]], align 8
-; CHECK-NEXT: [[TMP2:%.*]] = load <2 x double>, ptr [[X]], align 8
-; CHECK-NEXT: [[TMP1:%.*]] = load <2 x double>, ptr [[Y]], align 8
-; CHECK-NEXT: [[TMP3:%.*]] = fmul reassoc nsz contract <2 x double> [[TMP1]], [[TMP2]]
+; CHECK-NEXT: [[X1:%.*]] = load double, ptr [[X8]], align 8
+; CHECK-NEXT: [[Y1:%.*]] = load double, ptr [[Y8]], align 8
+; CHECK-NEXT: [[TMP4:%.*]] = fmul reassoc nsz contract double [[Y1]], [[X1]]
; CHECK-NEXT: [[Z1:%.*]] = load double, ptr [[Z8]], align 8
; CHECK-NEXT: [[ZSUM:%.*]] = fadd reassoc nsz contract double [[Z0]], [[Z1]]
-; CHECK-NEXT: [[TMP5:%.*]] = extractelement <2 x double> [[TMP3]], i64 0
; CHECK-NEXT: [[SUB:%.*]] = fsub reassoc nsz contract double [[TMP5]], [[ZSUM]]
-; CHECK-NEXT: [[TMP4:%.*]] = extractelement <2 x double> [[TMP3]], i64 1
; CHECK-NEXT: [[TMP6:%.*]] = fadd reassoc nsz contract double [[SUB]], [[TMP4]]
; CHECK-NEXT: ret double [[TMP6]]
;
@@ -46,25 +49,28 @@ define double @fsub_fmul_4(ptr %x, ptr %y, ptr %z) {
; CHECK-NEXT: [[ENTRY:.*:]]
; CHECK-NEXT: [[X8:%.*]] = getelementptr inbounds nuw i8, ptr [[X]], i64 8
; CHECK-NEXT: [[Y8:%.*]] = getelementptr inbounds nuw i8, ptr [[Y]], i64 8
+; CHECK-NEXT: [[X16:%.*]] = getelementptr inbounds nuw i8, ptr [[X]], i64 16
+; CHECK-NEXT: [[Y16:%.*]] = getelementptr inbounds nuw i8, ptr [[Y]], i64 16
; CHECK-NEXT: [[X24:%.*]] = getelementptr inbounds nuw i8, ptr [[X]], i64 24
; CHECK-NEXT: [[Y24:%.*]] = getelementptr inbounds nuw i8, ptr [[Y]], i64 24
; CHECK-NEXT: [[X0:%.*]] = load double, ptr [[X]], align 8
; CHECK-NEXT: [[Y0:%.*]] = load double, ptr [[Y]], align 8
; CHECK-NEXT: [[MUL:%.*]] = fmul reassoc nsz contract double [[Y0]], [[X0]]
-; CHECK-NEXT: [[TMP0:%.*]] = load <2 x double>, ptr [[X8]], align 8
-; CHECK-NEXT: [[TMP1:%.*]] = load <2 x double>, ptr [[Y8]], align 8
-; CHECK-NEXT: [[TMP2:%.*]] = fmul reassoc nsz contract <2 x double> [[TMP1]], [[TMP0]]
-; CHECK-NEXT: [[X3:%.*]] = load double, ptr [[X24]], align 8
-; CHECK-NEXT: [[Y3:%.*]] = load double, ptr [[Y24]], align 8
+; CHECK-NEXT: [[X3:%.*]] = load double, ptr [[X8]], align 8
+; CHECK-NEXT: [[Y3:%.*]] = load double, ptr [[Y8]], align 8
; CHECK-NEXT: [[MUL16:%.*]] = fmul reassoc nsz contract double [[Y3]], [[X3]]
+; CHECK-NEXT: [[X2:%.*]] = load double, ptr [[X16]], align 8
+; CHECK-NEXT: [[Y2:%.*]] = load double, ptr [[Y16]], align 8
+; CHECK-NEXT: [[TMP7:%.*]] = fmul reassoc nsz contract double [[Y2]], [[X2]]
+; CHECK-NEXT: [[X4:%.*]] = load double, ptr [[X24]], align 8
+; CHECK-NEXT: [[Y4:%.*]] = load double, ptr [[Y24]], align 8
+; CHECK-NEXT: [[MUL17:%.*]] = fmul reassoc nsz contract double [[Y4]], [[X4]]
; CHECK-NEXT: [[TMP3:%.*]] = load <4 x double>, ptr [[Z]], align 8
-; CHECK-NEXT: [[TMP4:%.*]] = extractelement <2 x double> [[TMP2]], i64 0
-; CHECK-NEXT: [[T1:%.*]] = fadd reassoc nsz contract double [[MUL]], [[TMP4]]
-; CHECK-NEXT: [[TMP7:%.*]] = extractelement <2 x double> [[TMP2]], i64 1
+; CHECK-NEXT: [[T1:%.*]] = fadd reassoc nsz contract double [[MUL]], [[MUL16]]
; CHECK-NEXT: [[T3:%.*]] = fadd reassoc nsz contract double [[T1]], [[TMP7]]
; CHECK-NEXT: [[TMP6:%.*]] = call reassoc nsz contract double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[TMP3]])
; CHECK-NEXT: [[ADD13:%.*]] = fsub reassoc nsz contract double [[T3]], [[TMP6]]
-; CHECK-NEXT: [[TMP5:%.*]] = fadd reassoc nsz contract double [[ADD13]], [[MUL16]]
+; CHECK-NEXT: [[TMP5:%.*]] = fadd reassoc nsz contract double [[ADD13]], [[MUL17]]
; CHECK-NEXT: ret double [[TMP5]]
;
entry:
@@ -390,30 +396,36 @@ define double @fsub_fmul_subtrahend_4(ptr %x, ptr %y, ptr %z) {
; CHECK-LABEL: define double @fsub_fmul_subtrahend_4(
; CHECK-SAME: ptr [[X:%.*]], ptr [[Y:%.*]], ptr [[Z:%.*]]) #[[ATTR0]] {
; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[X8:%.*]] = getelementptr inbounds nuw i8, ptr [[X]], i64 8
+; CHECK-NEXT: [[Y8:%.*]] = getelementptr inbounds nuw i8, ptr [[Y]], i64 8
; CHECK-NEXT: [[Z8:%.*]] = getelementptr inbounds nuw i8, ptr [[Z]], i64 8
; CHECK-NEXT: [[X16:%.*]] = getelementptr inbounds nuw i8, ptr [[X]], i64 16
; CHECK-NEXT: [[Y16:%.*]] = getelementptr inbounds nuw i8, ptr [[Y]], i64 16
; CHECK-NEXT: [[Z16:%.*]] = getelementptr inbounds nuw i8, ptr [[Z]], i64 16
+; CHECK-NEXT: [[X24:%.*]] = getelementptr inbounds nuw i8, ptr [[X]], i64 24
+; CHECK-NEXT: [[Y24:%.*]] = getelementptr inbounds nuw i8, ptr [[Y]], i64 24
; CHECK-NEXT: [[Z24:%.*]] = getelementptr inbounds nuw i8, ptr [[Z]], i64 24
+; CHECK-NEXT: [[X0:%.*]] = load double, ptr [[X]], align 8
+; CHECK-NEXT: [[Y0:%.*]] = load double, ptr [[Y]], align 8
+; CHECK-NEXT: [[TMP6:%.*]] = fmul reassoc nsz contract double [[Y0]], [[X0]]
; CHECK-NEXT: [[Z0:%.*]] = load double, ptr [[Z]], align 8
-; CHECK-NEXT: [[TMP0:%.*]] = load <2 x double>, ptr [[X]], align 8
-; CHECK-NEXT: [[TMP1:%.*]] = load <2 x double>, ptr [[Y]], align 8
-; CHECK-NEXT: [[TMP2:%.*]] = fmul reassoc nsz contract <2 x double> [[TMP1]], [[TMP0]]
+; CHECK-NEXT: [[X1:%.*]] = load double, ptr [[X8]], align 8
+; CHECK-NEXT: [[Y1:%.*]] = load double, ptr [[Y8]], align 8
+; CHECK-NEXT: [[TMP7:%.*]] = fmul reassoc nsz contract double [[Y1]], [[X1]]
; CHECK-NEXT: [[Z1:%.*]] = load double, ptr [[Z8]], align 8
+; CHECK-NEXT: [[X2:%.*]] = load double, ptr [[X16]], align 8
+; CHECK-NEXT: [[Y2:%.*]] = load double, ptr [[Y16]], align 8
+; CHECK-NEXT: [[TMP8:%.*]] = fmul reassoc nsz contract double [[Y2]], [[X2]]
; CHECK-NEXT: [[Z2:%.*]] = load double, ptr [[Z16]], align 8
-; CHECK-NEXT: [[TMP3:%.*]] = load <2 x double>, ptr [[X16]], align 8
-; CHECK-NEXT: [[TMP4:%.*]] = load <2 x double>, ptr [[Y16]], align 8
-; CHECK-NEXT: [[TMP5:%.*]] = fmul reassoc nsz contract <2 x double> [[TMP4]], [[TMP3]]
+; CHECK-NEXT: [[X3:%.*]] = load double, ptr [[X24]], align 8
+; CHECK-NEXT: [[Y3:%.*]] = load double, ptr [[Y24]], align 8
+; CHECK-NEXT: [[TMP9:%.*]] = fmul reassoc nsz contract double [[Y3]], [[X3]]
; CHECK-NEXT: [[Z3:%.*]] = load double, ptr [[Z24]], align 8
-; CHECK-NEXT: [[TMP6:%.*]] = extractelement <2 x double> [[TMP2]], i64 0
; CHECK-NEXT: [[S0:%.*]] = fsub reassoc nsz contract double [[Z0]], [[TMP6]]
; CHECK-NEXT: [[A1:%.*]] = fadd reassoc nsz contract double [[S0]], [[Z1]]
-; CHECK-NEXT: [[TMP7:%.*]] = extractelement <2 x double> [[TMP2]], i64 1
; CHECK-NEXT: [[S1:%.*]] = fsub reassoc nsz contract double [[A1]], [[TMP7]]
; CHECK-NEXT: [[A2:%.*]] = fadd reassoc nsz contract double [[S1]], [[Z2]]
-; CHECK-NEXT: [[TMP8:%.*]] = extractelement <2 x double> [[TMP5]], i64 0
; CHECK-NEXT: [[S2:%.*]] = fsub reassoc nsz contract double [[A2]], [[TMP8]]
-; CHECK-NEXT: [[TMP9:%.*]] = extractelement <2 x double> [[TMP5]], i64 1
; CHECK-NEXT: [[S3:%.*]] = fsub reassoc nsz contract double [[S2]], [[TMP9]]
; CHECK-NEXT: [[R:%.*]] = fadd reassoc nsz contract double [[S3]], [[Z3]]
; CHECK-NEXT: ret double [[R]]
@@ -499,50 +511,59 @@ define double @three_sign_groups(ptr %x, ptr %y, ptr %u, ptr %v, ptr %z) {
; CHECK-SAME: ptr [[X:%.*]], ptr [[Y:%.*]], ptr [[U:%.*]], ptr [[V:%.*]], ptr [[Z:%.*]]) #[[ATTR0]] {
; CHECK-NEXT: [[ENTRY:.*:]]
; CHECK-NEXT: [[X8:%.*]] = getelementptr inbounds nuw i8, ptr [[X]], i64 8
+; CHECK-NEXT: [[X16:%.*]] = getelementptr inbounds nuw i8, ptr [[X]], i64 16
; CHECK-NEXT: [[X24:%.*]] = getelementptr inbounds nuw i8, ptr [[X]], i64 24
; CHECK-NEXT: [[Y8:%.*]] = getelementptr inbounds nuw i8, ptr [[Y]], i64 8
+; CHECK-NEXT: [[Y16:%.*]] = getelementptr inbounds nuw i8, ptr [[Y]], i64 16
; CHECK-NEXT: [[Y24:%.*]] = getelementptr inbounds nuw i8, ptr [[Y]], i64 24
+; CHECK-NEXT: [[U8:%.*]] = getelementptr inbounds nuw i8, ptr [[U]], i64 8
; CHECK-NEXT: [[U16:%.*]] = getelementptr inbounds nuw i8, ptr [[U]], i64 16
+; CHECK-NEXT: [[U24:%.*]] = getelementptr inbounds nuw i8, ptr [[U]], i64 24
+; CHECK-NEXT: [[V8:%.*]] = getelementptr inbounds nuw i8, ptr [[V]], i64 8
; CHECK-NEXT: [[V16:%.*]] = getelementptr inbounds nuw i8, ptr [[V]], i64 16
+; CHECK-NEXT: [[V24:%.*]] = getelementptr inbounds nuw i8, ptr [[V]], i64 24
; CHECK-NEXT: [[Z8:%.*]] = getelementptr inbounds nuw i8, ptr [[Z]], i64 8
; CHECK-NEXT: [[Z16:%.*]] = getelementptr inbounds nuw i8, ptr [[Z]], i64 16
; CHECK-NEXT: [[Z24:%.*]] = getelementptr inbounds nuw i8, ptr [[Z]], i64 24
; CHECK-NEXT: [[X0:%.*]] = load double, ptr [[X]], align 8
-; CHECK-NEXT: [[X3:%.*]] = load double, ptr [[X24]], align 8
+; CHECK-NEXT: [[X3:%.*]] = load double, ptr [[X8]], align 8
+; CHECK-NEXT: [[X2:%.*]] = load double, ptr [[X16]], align 8
+; CHECK-NEXT: [[X4:%.*]] = load double, ptr [[X24]], align 8
; CHECK-NEXT: [[Y0:%.*]] = load double, ptr [[Y]], align 8
-; CHECK-NEXT: [[Y3:%.*]] = load double, ptr [[Y24]], align 8
+; CHECK-NEXT: [[Y3:%.*]] = load double, ptr [[Y8]], align 8
+; CHECK-NEXT: [[Y2:%.*]] = load double, ptr [[Y16]], align 8
+; CHECK-NEXT: [[Y4:%.*]] = load double, ptr [[Y24]], align 8
+; CHECK-NEXT: [[U0:%.*]] = load double, ptr [[U]], align 8
+; CHECK-NEXT: [[U1:%.*]] = load double, ptr [[U8]], align 8
+; CHECK-NEXT: [[U2:%.*]] = load double, ptr [[U16]], align 8
+; CHECK-NEXT: [[U3:%.*]] = load double, ptr [[U24]], align 8
+; CHECK-NEXT: [[V0:%.*]] = load double, ptr [[V]], align 8
+; CHECK-NEXT: [[V1:%.*]] = load double, ptr [[V8]], align 8
+; CHECK-NEXT: [[V2:%.*]] = load double, ptr [[V16]], align 8
+; CHECK-NEXT: [[V3:%.*]] = load double, ptr [[V24]], align 8
; CHECK-NEXT: [[Z0:%.*]] = load double, ptr [[Z]], align 8
; CHECK-NEXT: [[Z1:%.*]] = load double, ptr [[Z8]], align 8
; CHECK-NEXT: [[Z2:%.*]] = load double, ptr [[Z16]], align 8
; CHECK-NEXT: [[Z3:%.*]] = load double, ptr [[Z24]], align 8
; CHECK-NEXT: [[MXY0:%.*]] = fmul reassoc nsz contract double [[X0]], [[Y0]]
-; CHECK-NEXT: [[TMP0:%.*]] = load <2 x double>, ptr [[X8]], align 8
-; CHECK-NEXT: [[TMP1:%.*]] = load <2 x double>, ptr [[Y8]], align 8
-; CHECK-NEXT: [[TMP2:%.*]] = fmul reassoc nsz contract <2 x double> [[TMP0]], [[TMP1]]
; CHECK-NEXT: [[MXY3:%.*]] = fmul reassoc nsz contract double [[X3]], [[Y3]]
-; CHECK-NEXT: [[TMP3:%.*]] = load <2 x double>, ptr [[U]], align 8
-; CHECK-NEXT: [[TMP4:%.*]] = load <2 x double>, ptr [[V]], align 8
-; CHECK-NEXT: [[TMP5:%.*]] = fmul reassoc nsz contract <2 x double> [[TMP3]], [[TMP4]]
-; CHECK-NEXT: [[TMP6:%.*]] = load <2 x double>, ptr [[U16]], align 8
-; CHECK-NEXT: [[TMP7:%.*]] = load <2 x double>, ptr [[V16]], align 8
-; CHECK-NEXT: [[TMP8:%.*]] = fmul reassoc nsz contract <2 x double> [[TMP6]], [[TMP7]]
-; CHECK-NEXT: [[TMP9:%.*]] = extractelement <2 x double> [[TMP2]], i64 0
-; CHECK-NEXT: [[T0:%.*]] = fadd reassoc nsz contract double [[MXY0]], [[TMP9]]
-; CHECK-NEXT: [[TMP10:%.*]] = extractelement <2 x double> [[TMP2]], i64 1
+; CHECK-NEXT: [[TMP10:%.*]] = fmul reassoc nsz contract double [[X2]], [[Y2]]
+; CHECK-NEXT: [[MXY4:%.*]] = fmul reassoc nsz contract double [[X4]], [[Y4]]
+; CHECK-NEXT: [[TMP11:%.*]] = fmul reassoc nsz contract double [[U0]], [[V0]]
+; CHECK-NEXT: [[TMP12:%.*]] = fmul reassoc nsz contract double [[U1]], [[V1]]
+; CHECK-NEXT: [[TMP13:%.*]] = fmul reassoc nsz contract double [[U2]], [[V2]]
+; CHECK-NEXT: [[TMP14:%.*]] = fmul reassoc nsz contract double [[U3]], [[V3]]
+; CHECK-NEXT: [[T0:%.*]] = fadd reassoc nsz contract double [[MXY0]], [[MXY3]]
; CHECK-NEXT: [[T1:%.*]] = fadd reassoc nsz contract double [[T0]], [[TMP10]]
-; CHECK-NEXT: [[TMP11:%.*]] = extractelement <2 x double> [[TMP5]], i64 0
; CHECK-NEXT: [[S0:%.*]] = fsub reassoc nsz contract double [[T1]], [[TMP11]]
-; CHECK-NEXT: [[TMP12:%.*]] = extractelement <2 x double> [[TMP5]], i64 1
; CHECK-NEXT: [[S1:%.*]] = fsub reassoc nsz contract double [[S0]], [[TMP12]]
-; CHECK-NEXT: [[TMP13:%.*]] = extractelement <2 x double> [[TMP8]], i64 0
; CHECK-NEXT: [[S2:%.*]] = fsub reassoc nsz contract double [[S1]], [[TMP13]]
-; CHECK-NEXT: [[TMP14:%.*]] = extractelement <2 x double> [[TMP8]], i64 1
; CHECK-NEXT: [[S3:%.*]] = fsub reassoc nsz contract double [[S2]], [[TMP14]]
; CHECK-NEXT: [[W0:%.*]] = fsub reassoc nsz contract double [[S3]], [[Z0]]
; CHECK-NEXT: [[W1:%.*]] = fsub reassoc nsz contract double [[W0]], [[Z1]]
; CHECK-NEXT: [[W2:%.*]] = fsub reassoc nsz contract double [[W1]], [[Z2]]
; CHECK-NEXT: [[W3:%.*]] = fsub reassoc nsz contract double [[W2]], [[Z3]]
-; CHECK-NEXT: [[R:%.*]] = fadd reassoc nsz contract double [[W3]], [[MXY3]]
+; CHECK-NEXT: [[R:%.*]] = fadd reassoc nsz contract double [[W3]], [[MXY4]]
; CHECK-NEXT: ret double [[R]]
;
entry:
@@ -611,30 +632,33 @@ define double @negated_repeated_pair(ptr %x, ptr %y, ptr %z) {
; CHECK-SAME: ptr [[X:%.*]], ptr [[Y:%.*]], ptr [[Z:%.*]]) #[[ATTR0]] {
; CHECK-NEXT: [[ENTRY:.*:]]
; CHECK-NEXT: [[X8:%.*]] = getelementptr inbounds nuw i8, ptr [[X]], i64 8
+; CHECK-NEXT: [[X16:%.*]] = getelementptr inbounds nuw i8, ptr [[X]], i64 16
; CHECK-NEXT: [[X24:%.*]] = getelementptr inbounds nuw i8, ptr [[X]], i64 24
; CHECK-NEXT: [[Y8:%.*]] = getelementptr inbounds nuw i8, ptr [[Y]], i64 8
+; CHECK-NEXT: [[Y16:%.*]] = getelementptr inbounds nuw i8, ptr [[Y]], i64 16
; CHECK-NEXT: [[Y24:%.*]] = getelementptr inbounds nuw i8, ptr [[Y]], i64 24
; CHECK-NEXT: [[Z8:%.*]] = getelementptr inbounds nuw i8, ptr [[Z]], i64 8
; CHECK-NEXT: [[X0:%.*]] = load double, ptr [[X]], align 8
-; CHECK-NEXT: [[X3:%.*]] = load double, ptr [[X24]], align 8
+; CHECK-NEXT: [[X3:%.*]] = load double, ptr [[X8]], align 8
+; CHECK-NEXT: [[X2:%.*]] = load double, ptr [[X16]], align 8
+; CHECK-NEXT: [[X4:%.*]] = load double, ptr [[X24]], align 8
; CHECK-NEXT: [[Y0:%.*]] = load double, ptr [[Y]], align 8
-; CHECK-NEXT: [[Y3:%.*]] = load double, ptr [[Y24]], align 8
+; CHECK-NEXT: [[Y3:%.*]] = load double, ptr [[Y8]], align 8
+; CHECK-NEXT: [[Y2:%.*]] = load double, ptr [[Y16]], align 8
+; CHECK-NEXT: [[Y4:%.*]] = load double, ptr [[Y24]], align 8
; CHECK-NEXT: [[ZA:%.*]] = load double, ptr [[Z]], align 8
; CHECK-NEXT: [[ZB:%.*]] = load double, ptr [[Z8]], align 8
; CHECK-NEXT: [[M0:%.*]] = fmul reassoc nsz contract double [[X0]], [[Y0]]
-; CHECK-NEXT: [[TMP0:%.*]] = load <2 x double>, ptr [[X8]], align 8
-; CHECK-NEXT: [[TMP1:%.*]] = load <2 x double>, ptr [[Y8]], align 8
-; CHECK-NEXT: [[TMP2:%.*]] = fmul reassoc nsz contract <2 x double> [[TMP0]], [[TMP1]]
; CHECK-NEXT: [[M3:%.*]] = fmul reassoc nsz contract double [[X3]], [[Y3]]
-; CHECK-NEXT: [[TMP3:%.*]] = extractelement <2 x double> [[TMP2]], i64 0
-; CHECK-NEXT: [[T0:%.*]] = fadd reassoc nsz contract double [[M0]], [[TMP3]]
-; CHECK-NEXT: [[TMP4:%.*]] = extractelement <2 x double> [[TMP2]], i64 1
+; CHECK-NEXT: [[TMP4:%.*]] = fmul reassoc nsz contract double [[X2]], [[Y2]]
+; CHECK-NEXT: [[M4:%.*]] = fmul reassoc nsz contract double [[X4]], [[Y4]]
+; CHECK-NEXT: [[T0:%.*]] = fadd reassoc nsz contract double [[M0]], [[M3]]
; CHECK-NEXT: [[T1:%.*]] = fadd reassoc nsz contract double [[T0]], [[TMP4]]
; CHECK-NEXT: [[S0:%.*]] = fsub reassoc nsz contract double [[T1]], [[ZA]]
; CHECK-NEXT: [[S1:%.*]] = fsub reassoc nsz contract double [[S0]], [[ZB]]
; CHECK-NEXT: [[S2:%.*]] = fsub reassoc nsz contract double [[S1]], [[ZA]]
; CHECK-NEXT: [[S3:%.*]] = fsub reassoc nsz contract double [[S2]], [[ZB]]
-; CHECK-NEXT: [[R:%.*]] = fadd reassoc nsz contract double [[S3]], [[M3]]
+; CHECK-NEXT: [[R:%.*]] = fadd reassoc nsz contract double [[S3]], [[M4]]
; CHECK-NEXT: ret double [[R]]
;
entry:
@@ -707,17 +731,20 @@ define double @ordered_switch_restart(ptr %x, ptr %y, ptr %z, ptr %out) {
; CHECK-LABEL: define double @ordered_switch_restart(
; CHECK-SAME: ptr [[X:%.*]], ptr [[Y:%.*]], ptr [[Z:%.*]], ptr [[OUT:%.*]]) #[[ATTR0]] {
; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[X8:%.*]] = getelementptr inbounds nuw i8, ptr [[X]], i64 8
+; CHECK-NEXT: [[Y8:%.*]] = getelementptr inbounds nuw i8, ptr [[Y]], i64 8
; CHECK-NEXT: [[Z8:%.*]] = getelementptr inbounds nuw i8, ptr [[Z]], i64 8
+; CHECK-NEXT: [[X0:%.*]] = load double, ptr [[X]], align 8
+; CHECK-NEXT: [[Y0:%.*]] = load double, ptr [[Y]], align 8
+; CHECK-NEXT: [[TMP3:%.*]] = fmul reassoc nsz contract double [[Y0]], [[X0]]
; CHECK-NEXT: [[Z0:%.*]] = load double, ptr [[Z]], align 8
-; CHECK-NEXT: [[TMP0:%.*]] = load <2 x double>, ptr [[X]], align 8
-; CHECK-NEXT: [[TMP1:%.*]] = load <2 x double>, ptr [[Y]], align 8
-; CHECK-NEXT: [[TMP2:%.*]] = fmul reassoc nsz contract <2 x double> [[TMP1]], [[TMP0]]
+; CHECK-NEXT: [[X1:%.*]] = load double, ptr [[X8]], align 8
+; CHECK-NEXT: [[Y1:%.*]] = load double, ptr [[Y8]], align 8
+; CHECK-NEXT: [[TMP4:%.*]] = fmul reassoc nsz contract double [[Y1]], [[X1]]
; CHECK-NEXT: [[Z1:%.*]] = load double, ptr [[Z8]], align 8
; CHECK-NEXT: [[ZSUM:%.*]] = fadd reassoc nsz contract double [[Z0]], [[Z1]]
; CHECK-NEXT: store double [[ZSUM]], ptr [[OUT]], align 8
-; CHECK-NEXT: [[TMP3:%.*]] = extractelement <2 x double> [[TMP2]], i64 0
; CHECK-NEXT: [[SUB:%.*]] = fsub reassoc nsz contract double [[TMP3]], [[ZSUM]]
-; CHECK-NEXT: [[TMP4:%.*]] = extractelement <2 x double> [[TMP2]], i64 1
; CHECK-NEXT: [[ADD:%.*]] = fadd reassoc nsz contract double [[SUB]], [[TMP4]]
; CHECK-NEXT: ret double [[ADD]]
;
>From 8bd4e20ff0a3a3a5651e29abc65517d7d4741973 Mon Sep 17 00:00:00 2001
From: Dmitry Sidorov <Dmitry.Sidorov at amd.com>
Date: Tue, 25 Aug 2026 21:43:33 +0200
Subject: [PATCH 02/11] format
---
.../Transforms/Vectorize/SLPVectorizer.cpp | 24 ++++++++-----------
1 file changed, 10 insertions(+), 14 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 5ffa5884c323a..0a66731d45862 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -13648,13 +13648,11 @@ bool BoUpSLP::areAllUsersVectorized(
});
}
-static InstructionCost canConvertToFMA(ArrayRef<Value *> VL,
- const InstructionsState &S,
- DominatorTree &DT, const DataLayout &DL,
- TargetTransformInfo &TTI,
- const TargetLibraryInfo &TLI,
- const TTI::TargetCostKind CostKind,
- unsigned FMulOpIdx = 0);
+static InstructionCost
+canConvertToFMA(ArrayRef<Value *> VL, const InstructionsState &S,
+ DominatorTree &DT, const DataLayout &DL,
+ TargetTransformInfo &TTI, const TargetLibraryInfo &TLI,
+ const TTI::TargetCostKind CostKind, unsigned FMulOpIdx = 0);
/// \returns the operand of \p I that is a one-use fmul, 0 if there is none.
static unsigned getFMulOperandIdx(const Instruction *I) {
@@ -14463,13 +14461,11 @@ void BoUpSLP::reorderGatherNode(TreeEntry &TE) {
/// Check if we can convert fadd/fsub sequence to FMAD.
/// \returns Cost of the FMAD, if conversion is possible, invalid cost otherwise.
-static InstructionCost canConvertToFMA(ArrayRef<Value *> VL,
- const InstructionsState &S,
- DominatorTree &DT, const DataLayout &DL,
- TargetTransformInfo &TTI,
- const TargetLibraryInfo &TLI,
- const TTI::TargetCostKind CostKind,
- unsigned FMulOpIdx) {
+static InstructionCost
+canConvertToFMA(ArrayRef<Value *> VL, const InstructionsState &S,
+ DominatorTree &DT, const DataLayout &DL,
+ TargetTransformInfo &TTI, const TargetLibraryInfo &TLI,
+ const TTI::TargetCostKind CostKind, unsigned FMulOpIdx) {
assert(all_of(VL,
[](Value *V) {
return V->getType()->getScalarType()->isFloatingPointTy();
>From 2b708b28959cfb7ff9a2cb43e38ba77846654631 Mon Sep 17 00:00:00 2001
From: Dmitry Sidorov <Dmitry.Sidorov at amd.com>
Date: Sat, 29 Aug 2026 23:01:17 +0200
Subject: [PATCH 03/11] partially address comments
---
.../Transforms/Vectorize/SLPVectorizer.cpp | 25 ++++++-------------
.../Vectorize/SLPVectorizer/SLPUtils.cpp | 12 +++++++++
.../Vectorize/SLPVectorizer/SLPUtils.h | 4 +++
3 files changed, 23 insertions(+), 18 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 3478f5064b416..1779611454340 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -13756,17 +13756,7 @@ static InstructionCost
canConvertToFMA(ArrayRef<Value *> VL, const InstructionsState &S,
DominatorTree &DT, const DataLayout &DL,
TargetTransformInfo &TTI, const TargetLibraryInfo &TLI,
- const TTI::TargetCostKind CostKind, unsigned FMulOpIdx = 0);
-
-/// \returns the operand of \p I that is a one-use fmul, 0 if there is none.
-static unsigned getFMulOperandIdx(const Instruction *I) {
- for (unsigned Idx : seq<unsigned>(0, I->getNumOperands())) {
- auto *Op = dyn_cast<Instruction>(I->getOperand(Idx));
- if (Op && Op->getOpcode() == Instruction::FMul && Op->hasOneUse())
- return Idx;
- }
- return 0;
-}
+ const TTI::TargetCostKind CostKind, unsigned FMulOpIdx);
uint64_t BoUpSLP::getNumScalarInsts(bool HasTreeLoop) {
uint64_t Total = 0;
@@ -15461,7 +15451,7 @@ void BoUpSLP::transformNodes() {
!IsOneUseVectorFMulOperand(RHS)))
break;
if (!canConvertToFMA(E.Scalars, E.getOperations(), *DT, *DL, *TTI, *TLI,
- CostKind)
+ CostKind, /*FMulOpIdx=*/0)
.isValid())
break;
// This node is a fmuladd node.
@@ -17182,8 +17172,7 @@ BoUpSLP::getEntryCost(const TreeEntry *E, ArrayRef<Value *> VectorizedVals,
return IntrinsicCost;
};
auto GetFMulAddCost = [&, &TTI = *TTI](const InstructionsState &S,
- Instruction *VI,
- unsigned FMulOpIdx = 0) {
+ Instruction *VI, unsigned FMulOpIdx) {
InstructionCost Cost =
canConvertToFMA(VI, S, *DT, *DL, TTI, *TLI, CostKind, FMulOpIdx);
return Cost;
@@ -17649,8 +17638,8 @@ BoUpSLP::getEntryCost(const TreeEntry *E, ArrayRef<Value *> VectorizedVals,
auto GetScalarCost = [&](unsigned Idx) {
if (isa<PoisonValue>(UniqueValues[Idx]))
return InstructionCost(TTI::TCC_Free);
- return GetFMulAddCost(E->getOperations(),
- cast<Instruction>(UniqueValues[Idx]));
+ auto *VI = cast<Instruction>(UniqueValues[Idx]);
+ return GetFMulAddCost(E->getOperations(), VI, getFMulOperandIdx(VI));
};
auto GetVectorCost = [&, &TTI = *TTI](InstructionCost CommonCost) {
FastMathFlags FMF;
@@ -32281,7 +32270,7 @@ class HorizontalReduction {
if (RdxKind == RecurKind::FAdd) {
InstructionCost FMACost =
canConvertToFMA(RdxOp, getSameOpcode(RdxOp, TLI), DT, DL,
- *TTI, TLI, CostKind);
+ *TTI, TLI, CostKind, /*FMulOpIdx=*/0);
if (FMACost.isValid()) {
LLVM_DEBUG(dbgs() << "FMA cost: " << FMACost << "\n");
if (auto *I = dyn_cast<Instruction>(RdxVal)) {
@@ -32407,7 +32396,7 @@ class HorizontalReduction {
}
if (!Ops.empty()) {
FMACost = canConvertToFMA(Ops, getSameOpcode(Ops, TLI), DT, DL,
- *TTI, TLI, CostKind);
+ *TTI, TLI, CostKind, /*FMulOpIdx=*/0);
if (FMACost.isValid()) {
// Calculate actual FMAD cost.
IntrinsicCostAttributes ICA(Intrinsic::fmuladd, RVecTy,
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp
index af6f8032c7b0b..d8ab0d5f1d3c9 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp
@@ -771,6 +771,18 @@ Instruction *lookThroughCastRoundTrip(Value *V, bool MustBeElidable) {
return Narrow;
}
+unsigned getFMulOperandIdx(const Instruction *I) {
+ assert((I->getOpcode() == Instruction::Add ||
+ I->getOpcode() == Instruction::Sub ||
+ I->getOpcode() == Instruction::FAdd ||
+ I->getOpcode() == Instruction::FSub) &&
+ "Expected an add/sub-like instruction");
+ for (unsigned Idx : seq<unsigned>(I->getNumOperands()))
+ if (match(I->getOperand(Idx), m_OneUse(m_FMul(m_Value(), m_Value()))))
+ return Idx;
+ return 0;
+}
+
namespace {
/// Shifts and the mask accumulated from the narrow ops on the current path:
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.h b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.h
index e9aeff605eb91..774cd02ada4f7 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.h
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.h
@@ -337,6 +337,10 @@ bool isOnceUsedSeed(const Instruction *I);
/// produce nan/inf.
Instruction *lookThroughCastRoundTrip(Value *V, bool MustBeElidable);
+/// \returns the operand index of \p I that holds a one-use fmul, 0 if there is
+/// none. \p I must be an add/sub-like instruction.
+unsigned getFMulOperandIdx(const Instruction *I);
+
/// Narrow reduction leaf: the value, the shift applied after widening and
/// the mask applied in the narrow type before widening, clearing the bits
/// the absorbed narrow shls shift out and applying the absorbed narrow
>From 97c99be6ff7290a9671dc6452ec60a7e1483a2a1 Mon Sep 17 00:00:00 2001
From: Dmitry Sidorov <Dmitry.Sidorov at amd.com>
Date: Sun, 30 Aug 2026 19:33:30 +0200
Subject: [PATCH 04/11] wip
---
.../Transforms/Vectorize/SLPVectorizer.cpp | 24 ++++++++++---------
.../Vectorize/SLPVectorizer/SLPUtils.cpp | 6 ++---
.../Vectorize/SLPVectorizer/SLPUtils.h | 2 +-
3 files changed, 16 insertions(+), 16 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 1779611454340..e20bffb02800c 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -14591,8 +14591,7 @@ canConvertToFMA(ArrayRef<Value *> VL, const InstructionsState &S,
InstructionsCompatibilityAnalysis Analysis(DT, DL, TTI, TLI);
SmallVector<BoUpSLP::ValueList> Operands = Analysis.buildOperands(S, VL);
- if (FMulOpIdx >= Operands.size())
- return InstructionCost::getInvalid();
+ assert(FMulOpIdx < Operands.size() && "Expected a binary add/sub-like op");
InstructionsState OpS = getSameOpcode(Operands[FMulOpIdx], TLI);
if (!OpS.valid())
return InstructionCost::getInvalid();
@@ -15446,17 +15445,17 @@ void BoUpSLP::transformNodes() {
V->hasOneUse();
});
};
- if (!IsOneUseVectorFMulOperand(LHS) &&
- (E.getOpcode() == Instruction::FSub ||
- !IsOneUseVectorFMulOperand(RHS)))
+ const bool FMulOnLHS = IsOneUseVectorFMulOperand(LHS);
+ if (!FMulOnLHS && (E.getOpcode() == Instruction::FSub ||
+ !IsOneUseVectorFMulOperand(RHS)))
break;
if (!canConvertToFMA(E.Scalars, E.getOperations(), *DT, *DL, *TTI, *TLI,
- CostKind, /*FMulOpIdx=*/0)
+ CostKind, FMulOnLHS ? 0 : 1)
.isValid())
break;
// This node is a fmuladd node.
E.CombinedOp = TreeEntry::FMulAdd;
- TreeEntry *FMulEntry = getOperandEntry(&E, 0);
+ TreeEntry *FMulEntry = getOperandEntry(&E, FMulOnLHS ? 0 : 1);
if (FMulEntry->UserTreeIndex &&
FMulEntry->State == TreeEntry::Vectorize) {
// The FMul node is part of the combined fmuladd node.
@@ -17639,7 +17638,9 @@ BoUpSLP::getEntryCost(const TreeEntry *E, ArrayRef<Value *> VectorizedVals,
if (isa<PoisonValue>(UniqueValues[Idx]))
return InstructionCost(TTI::TCC_Free);
auto *VI = cast<Instruction>(UniqueValues[Idx]);
- return GetFMulAddCost(E->getOperations(), VI, getFMulOperandIdx(VI));
+ return GetFMulAddCost(E->getOperations(), VI,
+ E->isCopyableElement(VI) ? 0
+ : getFMulOperandIdx(VI));
};
auto GetVectorCost = [&, &TTI = *TTI](InstructionCost CommonCost) {
FastMathFlags FMF;
@@ -17810,8 +17811,9 @@ BoUpSLP::getEntryCost(const TreeEntry *E, ArrayRef<Value *> VectorizedVals,
InstructionCost ScalarCost = TTI->getArithmeticInstrCost(
ShuffleOrOp, OrigScalarTy, CostKind, Op1Info, Op2Info, Operands);
if (auto *I = dyn_cast<Instruction>(UniqueValues[Idx]);
- I && (ShuffleOrOp == Instruction::FAdd ||
- ShuffleOrOp == Instruction::FSub)) {
+ I && !E->isCopyableElement(I) &&
+ (ShuffleOrOp == Instruction::FAdd ||
+ ShuffleOrOp == Instruction::FSub)) {
InstructionCost IntrinsicCost =
GetFMulAddCost(E->getOperations(), I, getFMulOperandIdx(I));
if (IntrinsicCost.isValid())
@@ -34481,7 +34483,7 @@ bool SLPVectorizerPass::vectorizeOnceUsedSeeds(BasicBlock *BB, BoUpSLP &R) {
if (InstructionsState S = getSameOpcode(U, *TLI);
S && S.isAddSubLikeOp() &&
canConvertToFMA(U, S, *DT, *DL, *TTI, *TLI, R.getCostKind(),
- U->getOperand(1) == &I ? 1 : 0)
+ getFMulOperandIdx(U))
.isValid())
continue;
}
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp
index d8ab0d5f1d3c9..960d49866ad2c 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp
@@ -772,11 +772,9 @@ Instruction *lookThroughCastRoundTrip(Value *V, bool MustBeElidable) {
}
unsigned getFMulOperandIdx(const Instruction *I) {
- assert((I->getOpcode() == Instruction::Add ||
- I->getOpcode() == Instruction::Sub ||
- I->getOpcode() == Instruction::FAdd ||
+ assert((I->getOpcode() == Instruction::FAdd ||
I->getOpcode() == Instruction::FSub) &&
- "Expected an add/sub-like instruction");
+ "Expected an fadd/fsub-like instruction");
for (unsigned Idx : seq<unsigned>(I->getNumOperands()))
if (match(I->getOperand(Idx), m_OneUse(m_FMul(m_Value(), m_Value()))))
return Idx;
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.h b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.h
index 774cd02ada4f7..9edc8eda3cd8d 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.h
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.h
@@ -338,7 +338,7 @@ bool isOnceUsedSeed(const Instruction *I);
Instruction *lookThroughCastRoundTrip(Value *V, bool MustBeElidable);
/// \returns the operand index of \p I that holds a one-use fmul, 0 if there is
-/// none. \p I must be an add/sub-like instruction.
+/// none. \p I must be an fadd/fsub-like instruction.
unsigned getFMulOperandIdx(const Instruction *I);
/// Narrow reduction leaf: the value, the shift applied after widening and
>From 0f178028f239fca03d10a40ff25cfacadd47c95f Mon Sep 17 00:00:00 2001
From: Dmitry Sidorov <Dmitry.Sidorov at amd.com>
Date: Sun, 30 Aug 2026 22:58:20 +0200
Subject: [PATCH 05/11] update tests
---
.../AMDGPU/elementwise-fma-operand1.ll | 126 +++++++++++++++---
.../X86/select-logical-or-and-i1-vector.ll | 102 +++++++-------
2 files changed, 162 insertions(+), 66 deletions(-)
diff --git a/llvm/test/Transforms/SLPVectorizer/AMDGPU/elementwise-fma-operand1.ll b/llvm/test/Transforms/SLPVectorizer/AMDGPU/elementwise-fma-operand1.ll
index c4c97622ac726..c928eb4df3227 100644
--- a/llvm/test/Transforms/SLPVectorizer/AMDGPU/elementwise-fma-operand1.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AMDGPU/elementwise-fma-operand1.ll
@@ -18,12 +18,42 @@ define void @axpy4_contract(ptr noalias %d, ptr noalias %a, ptr noalias %b, ptr
; CHECK-LABEL: define void @axpy4_contract(
; CHECK-SAME: ptr noalias [[D:%.*]], ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], ptr noalias [[C:%.*]]) #[[ATTR0:[0-9]+]] {
; CHECK-NEXT: [[ENTRY:.*:]]
-; CHECK-NEXT: [[TMP0:%.*]] = load <4 x float>, ptr [[C]], align 4
-; CHECK-NEXT: [[TMP1:%.*]] = load <4 x float>, ptr [[A]], align 4
-; CHECK-NEXT: [[TMP2:%.*]] = load <4 x float>, ptr [[B]], align 4
-; CHECK-NEXT: [[TMP3:%.*]] = fmul contract <4 x float> [[TMP1]], [[TMP2]]
-; CHECK-NEXT: [[TMP4:%.*]] = fadd contract <4 x float> [[TMP0]], [[TMP3]]
-; CHECK-NEXT: store <4 x float> [[TMP4]], ptr [[D]], align 4
+; CHECK-NEXT: [[C0:%.*]] = load float, ptr [[C]], align 4
+; CHECK-NEXT: [[A0:%.*]] = load float, ptr [[A]], align 4
+; CHECK-NEXT: [[B0:%.*]] = load float, ptr [[B]], align 4
+; CHECK-NEXT: [[M0:%.*]] = fmul contract float [[A0]], [[B0]]
+; CHECK-NEXT: [[R0:%.*]] = fadd contract float [[C0]], [[M0]]
+; CHECK-NEXT: store float [[R0]], ptr [[D]], align 4
+; CHECK-NEXT: [[CP1:%.*]] = getelementptr inbounds float, ptr [[C]], i64 1
+; CHECK-NEXT: [[C1:%.*]] = load float, ptr [[CP1]], align 4
+; CHECK-NEXT: [[AP1:%.*]] = getelementptr inbounds float, ptr [[A]], i64 1
+; CHECK-NEXT: [[A1:%.*]] = load float, ptr [[AP1]], align 4
+; CHECK-NEXT: [[BP1:%.*]] = getelementptr inbounds float, ptr [[B]], i64 1
+; CHECK-NEXT: [[B1:%.*]] = load float, ptr [[BP1]], align 4
+; CHECK-NEXT: [[M1:%.*]] = fmul contract float [[A1]], [[B1]]
+; CHECK-NEXT: [[R1:%.*]] = fadd contract float [[C1]], [[M1]]
+; CHECK-NEXT: [[DP1:%.*]] = getelementptr inbounds float, ptr [[D]], i64 1
+; CHECK-NEXT: store float [[R1]], ptr [[DP1]], align 4
+; CHECK-NEXT: [[CP2:%.*]] = getelementptr inbounds float, ptr [[C]], i64 2
+; CHECK-NEXT: [[C2:%.*]] = load float, ptr [[CP2]], align 4
+; CHECK-NEXT: [[AP2:%.*]] = getelementptr inbounds float, ptr [[A]], i64 2
+; CHECK-NEXT: [[A2:%.*]] = load float, ptr [[AP2]], align 4
+; CHECK-NEXT: [[BP2:%.*]] = getelementptr inbounds float, ptr [[B]], i64 2
+; CHECK-NEXT: [[B2:%.*]] = load float, ptr [[BP2]], align 4
+; CHECK-NEXT: [[M2:%.*]] = fmul contract float [[A2]], [[B2]]
+; CHECK-NEXT: [[R2:%.*]] = fadd contract float [[C2]], [[M2]]
+; CHECK-NEXT: [[DP2:%.*]] = getelementptr inbounds float, ptr [[D]], i64 2
+; CHECK-NEXT: store float [[R2]], ptr [[DP2]], align 4
+; CHECK-NEXT: [[CP3:%.*]] = getelementptr inbounds float, ptr [[C]], i64 3
+; CHECK-NEXT: [[C3:%.*]] = load float, ptr [[CP3]], align 4
+; CHECK-NEXT: [[AP3:%.*]] = getelementptr inbounds float, ptr [[A]], i64 3
+; CHECK-NEXT: [[A3:%.*]] = load float, ptr [[AP3]], align 4
+; CHECK-NEXT: [[BP3:%.*]] = getelementptr inbounds float, ptr [[B]], i64 3
+; CHECK-NEXT: [[B3:%.*]] = load float, ptr [[BP3]], align 4
+; CHECK-NEXT: [[M3:%.*]] = fmul contract float [[A3]], [[B3]]
+; CHECK-NEXT: [[R3:%.*]] = fadd contract float [[C3]], [[M3]]
+; CHECK-NEXT: [[DP3:%.*]] = getelementptr inbounds float, ptr [[D]], i64 3
+; CHECK-NEXT: store float [[R3]], ptr [[DP3]], align 4
; CHECK-NEXT: ret void
;
; THR12-LABEL: define void @axpy4_contract(
@@ -81,12 +111,42 @@ define void @axpy4_reassoc(ptr noalias %d, ptr noalias %a, ptr noalias %b, ptr n
; CHECK-LABEL: define void @axpy4_reassoc(
; CHECK-SAME: ptr noalias [[D:%.*]], ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], ptr noalias [[C:%.*]]) #[[ATTR0]] {
; CHECK-NEXT: [[ENTRY:.*:]]
-; CHECK-NEXT: [[TMP0:%.*]] = load <4 x float>, ptr [[C]], align 4
-; CHECK-NEXT: [[TMP1:%.*]] = load <4 x float>, ptr [[A]], align 4
-; CHECK-NEXT: [[TMP2:%.*]] = load <4 x float>, ptr [[B]], align 4
-; CHECK-NEXT: [[TMP3:%.*]] = fmul reassoc contract <4 x float> [[TMP1]], [[TMP2]]
-; CHECK-NEXT: [[TMP4:%.*]] = fadd reassoc contract <4 x float> [[TMP0]], [[TMP3]]
-; CHECK-NEXT: store <4 x float> [[TMP4]], ptr [[D]], align 4
+; CHECK-NEXT: [[C0:%.*]] = load float, ptr [[C]], align 4
+; CHECK-NEXT: [[A0:%.*]] = load float, ptr [[A]], align 4
+; CHECK-NEXT: [[B0:%.*]] = load float, ptr [[B]], align 4
+; CHECK-NEXT: [[M0:%.*]] = fmul reassoc contract float [[A0]], [[B0]]
+; CHECK-NEXT: [[R0:%.*]] = fadd reassoc contract float [[C0]], [[M0]]
+; CHECK-NEXT: store float [[R0]], ptr [[D]], align 4
+; CHECK-NEXT: [[CP1:%.*]] = getelementptr inbounds float, ptr [[C]], i64 1
+; CHECK-NEXT: [[C1:%.*]] = load float, ptr [[CP1]], align 4
+; CHECK-NEXT: [[AP1:%.*]] = getelementptr inbounds float, ptr [[A]], i64 1
+; CHECK-NEXT: [[A1:%.*]] = load float, ptr [[AP1]], align 4
+; CHECK-NEXT: [[BP1:%.*]] = getelementptr inbounds float, ptr [[B]], i64 1
+; CHECK-NEXT: [[B1:%.*]] = load float, ptr [[BP1]], align 4
+; CHECK-NEXT: [[M1:%.*]] = fmul reassoc contract float [[A1]], [[B1]]
+; CHECK-NEXT: [[R1:%.*]] = fadd reassoc contract float [[C1]], [[M1]]
+; CHECK-NEXT: [[DP1:%.*]] = getelementptr inbounds float, ptr [[D]], i64 1
+; CHECK-NEXT: store float [[R1]], ptr [[DP1]], align 4
+; CHECK-NEXT: [[CP2:%.*]] = getelementptr inbounds float, ptr [[C]], i64 2
+; CHECK-NEXT: [[C2:%.*]] = load float, ptr [[CP2]], align 4
+; CHECK-NEXT: [[AP2:%.*]] = getelementptr inbounds float, ptr [[A]], i64 2
+; CHECK-NEXT: [[A2:%.*]] = load float, ptr [[AP2]], align 4
+; CHECK-NEXT: [[BP2:%.*]] = getelementptr inbounds float, ptr [[B]], i64 2
+; CHECK-NEXT: [[B2:%.*]] = load float, ptr [[BP2]], align 4
+; CHECK-NEXT: [[M2:%.*]] = fmul reassoc contract float [[A2]], [[B2]]
+; CHECK-NEXT: [[R2:%.*]] = fadd reassoc contract float [[C2]], [[M2]]
+; CHECK-NEXT: [[DP2:%.*]] = getelementptr inbounds float, ptr [[D]], i64 2
+; CHECK-NEXT: store float [[R2]], ptr [[DP2]], align 4
+; CHECK-NEXT: [[CP3:%.*]] = getelementptr inbounds float, ptr [[C]], i64 3
+; CHECK-NEXT: [[C3:%.*]] = load float, ptr [[CP3]], align 4
+; CHECK-NEXT: [[AP3:%.*]] = getelementptr inbounds float, ptr [[A]], i64 3
+; CHECK-NEXT: [[A3:%.*]] = load float, ptr [[AP3]], align 4
+; CHECK-NEXT: [[BP3:%.*]] = getelementptr inbounds float, ptr [[B]], i64 3
+; CHECK-NEXT: [[B3:%.*]] = load float, ptr [[BP3]], align 4
+; CHECK-NEXT: [[M3:%.*]] = fmul reassoc contract float [[A3]], [[B3]]
+; CHECK-NEXT: [[R3:%.*]] = fadd reassoc contract float [[C3]], [[M3]]
+; CHECK-NEXT: [[DP3:%.*]] = getelementptr inbounds float, ptr [[D]], i64 3
+; CHECK-NEXT: store float [[R3]], ptr [[DP3]], align 4
; CHECK-NEXT: ret void
;
; THR12-LABEL: define void @axpy4_reassoc(
@@ -144,12 +204,42 @@ define void @axpy4_mixed_reassoc(ptr noalias %d, ptr noalias %a, ptr noalias %b,
; CHECK-LABEL: define void @axpy4_mixed_reassoc(
; CHECK-SAME: ptr noalias [[D:%.*]], ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], ptr noalias [[C:%.*]]) #[[ATTR0]] {
; CHECK-NEXT: [[ENTRY:.*:]]
-; CHECK-NEXT: [[TMP0:%.*]] = load <4 x float>, ptr [[C]], align 4
-; CHECK-NEXT: [[TMP1:%.*]] = load <4 x float>, ptr [[A]], align 4
-; CHECK-NEXT: [[TMP2:%.*]] = load <4 x float>, ptr [[B]], align 4
-; CHECK-NEXT: [[TMP3:%.*]] = fmul contract <4 x float> [[TMP1]], [[TMP2]]
-; CHECK-NEXT: [[TMP4:%.*]] = fadd contract <4 x float> [[TMP0]], [[TMP3]]
-; CHECK-NEXT: store <4 x float> [[TMP4]], ptr [[D]], align 4
+; CHECK-NEXT: [[C0:%.*]] = load float, ptr [[C]], align 4
+; CHECK-NEXT: [[A0:%.*]] = load float, ptr [[A]], align 4
+; CHECK-NEXT: [[B0:%.*]] = load float, ptr [[B]], align 4
+; CHECK-NEXT: [[M0:%.*]] = fmul contract float [[A0]], [[B0]]
+; CHECK-NEXT: [[R0:%.*]] = fadd reassoc contract float [[C0]], [[M0]]
+; CHECK-NEXT: store float [[R0]], ptr [[D]], align 4
+; CHECK-NEXT: [[CP1:%.*]] = getelementptr inbounds float, ptr [[C]], i64 1
+; CHECK-NEXT: [[C1:%.*]] = load float, ptr [[CP1]], align 4
+; CHECK-NEXT: [[AP1:%.*]] = getelementptr inbounds float, ptr [[A]], i64 1
+; CHECK-NEXT: [[A1:%.*]] = load float, ptr [[AP1]], align 4
+; CHECK-NEXT: [[BP1:%.*]] = getelementptr inbounds float, ptr [[B]], i64 1
+; CHECK-NEXT: [[B1:%.*]] = load float, ptr [[BP1]], align 4
+; CHECK-NEXT: [[M1:%.*]] = fmul contract float [[A1]], [[B1]]
+; CHECK-NEXT: [[R1:%.*]] = fadd contract float [[C1]], [[M1]]
+; CHECK-NEXT: [[DP1:%.*]] = getelementptr inbounds float, ptr [[D]], i64 1
+; CHECK-NEXT: store float [[R1]], ptr [[DP1]], align 4
+; CHECK-NEXT: [[CP2:%.*]] = getelementptr inbounds float, ptr [[C]], i64 2
+; CHECK-NEXT: [[C2:%.*]] = load float, ptr [[CP2]], align 4
+; CHECK-NEXT: [[AP2:%.*]] = getelementptr inbounds float, ptr [[A]], i64 2
+; CHECK-NEXT: [[A2:%.*]] = load float, ptr [[AP2]], align 4
+; CHECK-NEXT: [[BP2:%.*]] = getelementptr inbounds float, ptr [[B]], i64 2
+; CHECK-NEXT: [[B2:%.*]] = load float, ptr [[BP2]], align 4
+; CHECK-NEXT: [[M2:%.*]] = fmul contract float [[A2]], [[B2]]
+; CHECK-NEXT: [[R2:%.*]] = fadd contract float [[C2]], [[M2]]
+; CHECK-NEXT: [[DP2:%.*]] = getelementptr inbounds float, ptr [[D]], i64 2
+; CHECK-NEXT: store float [[R2]], ptr [[DP2]], align 4
+; CHECK-NEXT: [[CP3:%.*]] = getelementptr inbounds float, ptr [[C]], i64 3
+; CHECK-NEXT: [[C3:%.*]] = load float, ptr [[CP3]], align 4
+; CHECK-NEXT: [[AP3:%.*]] = getelementptr inbounds float, ptr [[A]], i64 3
+; CHECK-NEXT: [[A3:%.*]] = load float, ptr [[AP3]], align 4
+; CHECK-NEXT: [[BP3:%.*]] = getelementptr inbounds float, ptr [[B]], i64 3
+; CHECK-NEXT: [[B3:%.*]] = load float, ptr [[BP3]], align 4
+; CHECK-NEXT: [[M3:%.*]] = fmul contract float [[A3]], [[B3]]
+; CHECK-NEXT: [[R3:%.*]] = fadd contract float [[C3]], [[M3]]
+; CHECK-NEXT: [[DP3:%.*]] = getelementptr inbounds float, ptr [[D]], i64 3
+; CHECK-NEXT: store float [[R3]], ptr [[DP3]], align 4
; CHECK-NEXT: ret void
;
; THR12-LABEL: define void @axpy4_mixed_reassoc(
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/select-logical-or-and-i1-vector.ll b/llvm/test/Transforms/SLPVectorizer/X86/select-logical-or-and-i1-vector.ll
index 979701609b499..bb10829b8cca2 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/select-logical-or-and-i1-vector.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/select-logical-or-and-i1-vector.ll
@@ -12,30 +12,33 @@ define void @select_logical_or_i1(ptr %dst,
; CHECK-LABEL: define void @select_logical_or_i1(
; CHECK-SAME: ptr [[DST:%.*]], float [[D0:%.*]], float [[D1:%.*]], float [[D2:%.*]], float [[D3:%.*]], float [[THRESHOLD:%.*]], float [[HPHB_VAL:%.*]], i1 [[SCALAR_COND:%.*]], float [[Y0:%.*]], float [[Y1:%.*]], float [[Y2:%.*]], float [[Y3:%.*]], float [[E0:%.*]], float [[E1:%.*]], float [[E2:%.*]], float [[E3:%.*]]) #[[ATTR0:[0-9]+]] {
; CHECK-NEXT: [[ENTRY:.*:]]
-; CHECK-NEXT: [[TMP0:%.*]] = insertelement <4 x float> poison, float [[D0]], i64 0
-; CHECK-NEXT: [[TMP1:%.*]] = insertelement <4 x float> [[TMP0]], float [[D1]], i64 1
-; CHECK-NEXT: [[TMP2:%.*]] = insertelement <4 x float> [[TMP1]], float [[D2]], i64 2
-; CHECK-NEXT: [[TMP3:%.*]] = insertelement <4 x float> [[TMP2]], float [[D3]], i64 3
-; CHECK-NEXT: [[TMP4:%.*]] = insertelement <4 x float> poison, float [[THRESHOLD]], i64 0
-; CHECK-NEXT: [[TMP5:%.*]] = shufflevector <4 x float> [[TMP4]], <4 x float> poison, <4 x i32> zeroinitializer
-; CHECK-NEXT: [[TMP6:%.*]] = fcmp fast uge <4 x float> [[TMP3]], [[TMP5]]
-; CHECK-NEXT: [[TMP7:%.*]] = insertelement <4 x i1> poison, i1 [[SCALAR_COND]], i64 0
-; CHECK-NEXT: [[TMP8:%.*]] = shufflevector <4 x i1> [[TMP7]], <4 x i1> poison, <4 x i32> zeroinitializer
-; CHECK-NEXT: [[TMP9:%.*]] = select <4 x i1> [[TMP6]], <4 x i1> splat (i1 true), <4 x i1> [[TMP8]]
-; CHECK-NEXT: [[TMP10:%.*]] = insertelement <4 x float> poison, float [[HPHB_VAL]], i64 0
-; CHECK-NEXT: [[TMP11:%.*]] = shufflevector <4 x float> [[TMP10]], <4 x float> poison, <4 x i32> zeroinitializer
-; CHECK-NEXT: [[TMP12:%.*]] = select <4 x i1> [[TMP9]], <4 x float> zeroinitializer, <4 x float> [[TMP11]]
-; CHECK-NEXT: [[TMP13:%.*]] = insertelement <4 x float> poison, float [[Y0]], i64 0
-; CHECK-NEXT: [[TMP14:%.*]] = insertelement <4 x float> [[TMP13]], float [[Y1]], i64 1
-; CHECK-NEXT: [[TMP15:%.*]] = insertelement <4 x float> [[TMP14]], float [[Y2]], i64 2
-; CHECK-NEXT: [[TMP16:%.*]] = insertelement <4 x float> [[TMP15]], float [[Y3]], i64 3
-; CHECK-NEXT: [[TMP17:%.*]] = fmul fast <4 x float> [[TMP12]], [[TMP16]]
-; CHECK-NEXT: [[TMP18:%.*]] = insertelement <4 x float> poison, float [[E0]], i64 0
-; CHECK-NEXT: [[TMP19:%.*]] = insertelement <4 x float> [[TMP18]], float [[E1]], i64 1
-; CHECK-NEXT: [[TMP20:%.*]] = insertelement <4 x float> [[TMP19]], float [[E2]], i64 2
-; CHECK-NEXT: [[TMP21:%.*]] = insertelement <4 x float> [[TMP20]], float [[E3]], i64 3
-; CHECK-NEXT: [[TMP22:%.*]] = fadd fast <4 x float> [[TMP21]], [[TMP17]]
-; CHECK-NEXT: store <4 x float> [[TMP22]], ptr [[DST]], align 4
+; CHECK-NEXT: [[CMP0:%.*]] = fcmp fast uge float [[D0]], [[THRESHOLD]]
+; CHECK-NEXT: [[CMP1:%.*]] = fcmp fast uge float [[D1]], [[THRESHOLD]]
+; CHECK-NEXT: [[CMP2:%.*]] = fcmp fast uge float [[D2]], [[THRESHOLD]]
+; CHECK-NEXT: [[CMP3:%.*]] = fcmp fast uge float [[D3]], [[THRESHOLD]]
+; CHECK-NEXT: [[OR0:%.*]] = select i1 [[CMP0]], i1 true, i1 [[SCALAR_COND]]
+; CHECK-NEXT: [[OR1:%.*]] = select i1 [[CMP1]], i1 true, i1 [[SCALAR_COND]]
+; CHECK-NEXT: [[OR2:%.*]] = select i1 [[CMP2]], i1 true, i1 [[SCALAR_COND]]
+; CHECK-NEXT: [[OR3:%.*]] = select i1 [[CMP3]], i1 true, i1 [[SCALAR_COND]]
+; CHECK-NEXT: [[SEL0:%.*]] = select i1 [[OR0]], float 0.000000e+00, float [[HPHB_VAL]]
+; CHECK-NEXT: [[SEL1:%.*]] = select i1 [[OR1]], float 0.000000e+00, float [[HPHB_VAL]]
+; CHECK-NEXT: [[SEL2:%.*]] = select i1 [[OR2]], float 0.000000e+00, float [[HPHB_VAL]]
+; CHECK-NEXT: [[SEL3:%.*]] = select i1 [[OR3]], float 0.000000e+00, float [[HPHB_VAL]]
+; CHECK-NEXT: [[MUL0:%.*]] = fmul fast float [[SEL0]], [[Y0]]
+; CHECK-NEXT: [[MUL1:%.*]] = fmul fast float [[SEL1]], [[Y1]]
+; CHECK-NEXT: [[MUL2:%.*]] = fmul fast float [[SEL2]], [[Y2]]
+; CHECK-NEXT: [[MUL3:%.*]] = fmul fast float [[SEL3]], [[Y3]]
+; CHECK-NEXT: [[RES0:%.*]] = fadd fast float [[E0]], [[MUL0]]
+; CHECK-NEXT: [[RES1:%.*]] = fadd fast float [[E1]], [[MUL1]]
+; CHECK-NEXT: [[RES2:%.*]] = fadd fast float [[E2]], [[MUL2]]
+; CHECK-NEXT: [[RES3:%.*]] = fadd fast float [[E3]], [[MUL3]]
+; CHECK-NEXT: store float [[RES0]], ptr [[DST]], align 4
+; CHECK-NEXT: [[P1:%.*]] = getelementptr inbounds float, ptr [[DST]], i64 1
+; CHECK-NEXT: store float [[RES1]], ptr [[P1]], align 4
+; CHECK-NEXT: [[P2:%.*]] = getelementptr inbounds float, ptr [[DST]], i64 2
+; CHECK-NEXT: store float [[RES2]], ptr [[P2]], align 4
+; CHECK-NEXT: [[P3:%.*]] = getelementptr inbounds float, ptr [[DST]], i64 3
+; CHECK-NEXT: store float [[RES3]], ptr [[P3]], align 4
; CHECK-NEXT: ret void
;
float %d0, float %d1, float %d2, float %d3,
@@ -84,30 +87,33 @@ define void @select_logical_and_i1(ptr %dst,
; CHECK-LABEL: define void @select_logical_and_i1(
; CHECK-SAME: ptr [[DST:%.*]], float [[D0:%.*]], float [[D1:%.*]], float [[D2:%.*]], float [[D3:%.*]], float [[THRESHOLD:%.*]], float [[HPHB_VAL:%.*]], i1 [[SCALAR_COND:%.*]], float [[Y0:%.*]], float [[Y1:%.*]], float [[Y2:%.*]], float [[Y3:%.*]], float [[E0:%.*]], float [[E1:%.*]], float [[E2:%.*]], float [[E3:%.*]]) #[[ATTR0]] {
; CHECK-NEXT: [[ENTRY:.*:]]
-; CHECK-NEXT: [[TMP0:%.*]] = insertelement <4 x float> poison, float [[D0]], i64 0
-; CHECK-NEXT: [[TMP1:%.*]] = insertelement <4 x float> [[TMP0]], float [[D1]], i64 1
-; CHECK-NEXT: [[TMP2:%.*]] = insertelement <4 x float> [[TMP1]], float [[D2]], i64 2
-; CHECK-NEXT: [[TMP3:%.*]] = insertelement <4 x float> [[TMP2]], float [[D3]], i64 3
-; CHECK-NEXT: [[TMP4:%.*]] = insertelement <4 x float> poison, float [[THRESHOLD]], i64 0
-; CHECK-NEXT: [[TMP5:%.*]] = shufflevector <4 x float> [[TMP4]], <4 x float> poison, <4 x i32> zeroinitializer
-; CHECK-NEXT: [[TMP6:%.*]] = fcmp fast uge <4 x float> [[TMP3]], [[TMP5]]
-; CHECK-NEXT: [[TMP7:%.*]] = insertelement <4 x i1> poison, i1 [[SCALAR_COND]], i64 0
-; CHECK-NEXT: [[TMP8:%.*]] = shufflevector <4 x i1> [[TMP7]], <4 x i1> poison, <4 x i32> zeroinitializer
-; CHECK-NEXT: [[TMP9:%.*]] = select <4 x i1> [[TMP6]], <4 x i1> [[TMP8]], <4 x i1> zeroinitializer
-; CHECK-NEXT: [[TMP10:%.*]] = insertelement <4 x float> poison, float [[HPHB_VAL]], i64 0
-; CHECK-NEXT: [[TMP11:%.*]] = shufflevector <4 x float> [[TMP10]], <4 x float> poison, <4 x i32> zeroinitializer
-; CHECK-NEXT: [[TMP12:%.*]] = select <4 x i1> [[TMP9]], <4 x float> zeroinitializer, <4 x float> [[TMP11]]
-; CHECK-NEXT: [[TMP13:%.*]] = insertelement <4 x float> poison, float [[Y0]], i64 0
-; CHECK-NEXT: [[TMP14:%.*]] = insertelement <4 x float> [[TMP13]], float [[Y1]], i64 1
-; CHECK-NEXT: [[TMP15:%.*]] = insertelement <4 x float> [[TMP14]], float [[Y2]], i64 2
-; CHECK-NEXT: [[TMP16:%.*]] = insertelement <4 x float> [[TMP15]], float [[Y3]], i64 3
-; CHECK-NEXT: [[TMP17:%.*]] = fmul fast <4 x float> [[TMP12]], [[TMP16]]
-; CHECK-NEXT: [[TMP18:%.*]] = insertelement <4 x float> poison, float [[E0]], i64 0
-; CHECK-NEXT: [[TMP19:%.*]] = insertelement <4 x float> [[TMP18]], float [[E1]], i64 1
-; CHECK-NEXT: [[TMP20:%.*]] = insertelement <4 x float> [[TMP19]], float [[E2]], i64 2
-; CHECK-NEXT: [[TMP21:%.*]] = insertelement <4 x float> [[TMP20]], float [[E3]], i64 3
-; CHECK-NEXT: [[TMP22:%.*]] = fadd fast <4 x float> [[TMP21]], [[TMP17]]
-; CHECK-NEXT: store <4 x float> [[TMP22]], ptr [[DST]], align 4
+; CHECK-NEXT: [[CMP0:%.*]] = fcmp fast uge float [[D0]], [[THRESHOLD]]
+; CHECK-NEXT: [[CMP1:%.*]] = fcmp fast uge float [[D1]], [[THRESHOLD]]
+; CHECK-NEXT: [[CMP2:%.*]] = fcmp fast uge float [[D2]], [[THRESHOLD]]
+; CHECK-NEXT: [[CMP3:%.*]] = fcmp fast uge float [[D3]], [[THRESHOLD]]
+; CHECK-NEXT: [[AND0:%.*]] = select i1 [[CMP0]], i1 [[SCALAR_COND]], i1 false
+; CHECK-NEXT: [[AND1:%.*]] = select i1 [[CMP1]], i1 [[SCALAR_COND]], i1 false
+; CHECK-NEXT: [[AND2:%.*]] = select i1 [[CMP2]], i1 [[SCALAR_COND]], i1 false
+; CHECK-NEXT: [[AND3:%.*]] = select i1 [[CMP3]], i1 [[SCALAR_COND]], i1 false
+; CHECK-NEXT: [[SEL0:%.*]] = select i1 [[AND0]], float 0.000000e+00, float [[HPHB_VAL]]
+; CHECK-NEXT: [[SEL1:%.*]] = select i1 [[AND1]], float 0.000000e+00, float [[HPHB_VAL]]
+; CHECK-NEXT: [[SEL2:%.*]] = select i1 [[AND2]], float 0.000000e+00, float [[HPHB_VAL]]
+; CHECK-NEXT: [[SEL3:%.*]] = select i1 [[AND3]], float 0.000000e+00, float [[HPHB_VAL]]
+; CHECK-NEXT: [[MUL0:%.*]] = fmul fast float [[SEL0]], [[Y0]]
+; CHECK-NEXT: [[MUL1:%.*]] = fmul fast float [[SEL1]], [[Y1]]
+; CHECK-NEXT: [[MUL2:%.*]] = fmul fast float [[SEL2]], [[Y2]]
+; CHECK-NEXT: [[MUL3:%.*]] = fmul fast float [[SEL3]], [[Y3]]
+; CHECK-NEXT: [[RES0:%.*]] = fadd fast float [[E0]], [[MUL0]]
+; CHECK-NEXT: [[RES1:%.*]] = fadd fast float [[E1]], [[MUL1]]
+; CHECK-NEXT: [[RES2:%.*]] = fadd fast float [[E2]], [[MUL2]]
+; CHECK-NEXT: [[RES3:%.*]] = fadd fast float [[E3]], [[MUL3]]
+; CHECK-NEXT: store float [[RES0]], ptr [[DST]], align 4
+; CHECK-NEXT: [[P1:%.*]] = getelementptr inbounds float, ptr [[DST]], i64 1
+; CHECK-NEXT: store float [[RES1]], ptr [[P1]], align 4
+; CHECK-NEXT: [[P2:%.*]] = getelementptr inbounds float, ptr [[DST]], i64 2
+; CHECK-NEXT: store float [[RES2]], ptr [[P2]], align 4
+; CHECK-NEXT: [[P3:%.*]] = getelementptr inbounds float, ptr [[DST]], i64 3
+; CHECK-NEXT: store float [[RES3]], ptr [[P3]], align 4
; CHECK-NEXT: ret void
;
float %d0, float %d1, float %d2, float %d3,
>From 6f8fbaa91013d053a96196f4527a4a6b56faa864 Mon Sep 17 00:00:00 2001
From: Dmitry Sidorov <Dmitry.Sidorov at amd.com>
Date: Sun, 30 Aug 2026 23:57:27 +0200
Subject: [PATCH 06/11] update test
---
.../AArch64/reassociate-fma-pairs.ll | 34 +++++++++----------
1 file changed, 17 insertions(+), 17 deletions(-)
diff --git a/llvm/test/Transforms/PhaseOrdering/AArch64/reassociate-fma-pairs.ll b/llvm/test/Transforms/PhaseOrdering/AArch64/reassociate-fma-pairs.ll
index 33df34fd80713..0a48a6a0f1015 100644
--- a/llvm/test/Transforms/PhaseOrdering/AArch64/reassociate-fma-pairs.ll
+++ b/llvm/test/Transforms/PhaseOrdering/AArch64/reassociate-fma-pairs.ll
@@ -14,49 +14,49 @@ define double @md_vdw_energy(ptr nocapture readonly %coeffs, double %energy, dou
; CHECK-NEXT: [[FACTOR_OP_FMUL2:%.*]] = fmul fast double [[TABLE_DELTA]], 5.000000e-01
; CHECK-NEXT: [[FACTOR_OP_FMUL3:%.*]] = fmul fast double [[FACTOR_OP_FMUL]], [[TABLE_DELTA]]
; CHECK-NEXT: [[FACTOR_OP_FMUL4:%.*]] = fmul fast double [[TMP0]], 2.500000e-01
-; CHECK-NEXT: [[TMP1:%.*]] = insertelement <2 x double> poison, double [[SCALE]], i64 0
-; CHECK-NEXT: [[TMP11:%.*]] = shufflevector <2 x double> [[TMP1]], <2 x double> poison, <2 x i32> zeroinitializer
; CHECK-NEXT: br label %[[LOOP:.*]]
; CHECK: [[LOOP]]:
; CHECK-NEXT: [[I:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[NEXT:%.*]], %[[LOOP]] ]
; CHECK-NEXT: [[ACC1:%.*]] = phi double [ [[ENERGY]], %[[ENTRY]] ], [ [[RESULT1:%.*]], %[[LOOP]] ]
; CHECK-NEXT: [[BASE:%.*]] = getelementptr inbounds [8 x i8], ptr [[COEFFS]], i64 [[I]]
+; CHECK-NEXT: [[A:%.*]] = load double, ptr [[BASE]], align 8
+; CHECK-NEXT: [[B_PTR:%.*]] = getelementptr inbounds nuw i8, ptr [[BASE]], i64 8
+; CHECK-NEXT: [[B:%.*]] = load double, ptr [[B_PTR]], align 8
+; CHECK-NEXT: [[TMP5:%.*]] = fmul fast double [[A]], [[SCALE]]
+; CHECK-NEXT: [[TMP6:%.*]] = fmul fast double [[B]], [[SCALE]]
; CHECK-NEXT: [[C0_PTR:%.*]] = getelementptr inbounds nuw i8, ptr [[BASE]], i64 16
; CHECK-NEXT: [[C0:%.*]] = load double, ptr [[C0_PTR]], align 8
; CHECK-NEXT: [[C1_PTR:%.*]] = getelementptr inbounds nuw i8, ptr [[BASE]], i64 24
; CHECK-NEXT: [[C1:%.*]] = load double, ptr [[C1_PTR]], align 8
+; CHECK-NEXT: [[P0:%.*]] = fmul fast double [[C0]], [[TMP5]]
+; CHECK-NEXT: [[Q0:%.*]] = fmul fast double [[C1]], [[TMP6]]
+; CHECK-NEXT: [[D2:%.*]] = fsub fast double [[P0]], [[Q0]]
; CHECK-NEXT: [[C2_PTR:%.*]] = getelementptr inbounds nuw i8, ptr [[BASE]], i64 32
; CHECK-NEXT: [[C2:%.*]] = load double, ptr [[C2_PTR]], align 8
; CHECK-NEXT: [[C3_PTR:%.*]] = getelementptr inbounds nuw i8, ptr [[BASE]], i64 40
; CHECK-NEXT: [[C3:%.*]] = load double, ptr [[C3_PTR]], align 8
+; CHECK-NEXT: [[P1:%.*]] = fmul fast double [[C2]], [[TMP5]]
+; CHECK-NEXT: [[Q1:%.*]] = fmul fast double [[C3]], [[TMP6]]
+; CHECK-NEXT: [[D1:%.*]] = fsub fast double [[P1]], [[Q1]]
; CHECK-NEXT: [[C4_PTR:%.*]] = getelementptr inbounds nuw i8, ptr [[BASE]], i64 48
; CHECK-NEXT: [[C4:%.*]] = load double, ptr [[C4_PTR]], align 8
; CHECK-NEXT: [[C5_PTR:%.*]] = getelementptr inbounds nuw i8, ptr [[BASE]], i64 56
; CHECK-NEXT: [[C5:%.*]] = load double, ptr [[C5_PTR]], align 8
-; CHECK-NEXT: [[C6_PTR:%.*]] = getelementptr inbounds nuw i8, ptr [[BASE]], i64 64
-; CHECK-NEXT: [[TMP3:%.*]] = load <2 x double>, ptr [[BASE]], align 8
-; CHECK-NEXT: [[TMP4:%.*]] = fmul fast <2 x double> [[TMP3]], [[TMP11]]
-; CHECK-NEXT: [[TMP5:%.*]] = extractelement <2 x double> [[TMP4]], i64 0
-; CHECK-NEXT: [[P2:%.*]] = fmul fast double [[C0]], [[TMP5]]
-; CHECK-NEXT: [[TMP6:%.*]] = extractelement <2 x double> [[TMP4]], i64 1
-; CHECK-NEXT: [[Q2:%.*]] = fmul fast double [[C1]], [[TMP6]]
-; CHECK-NEXT: [[D2:%.*]] = fsub fast double [[P2]], [[Q2]]
-; CHECK-NEXT: [[P1:%.*]] = fmul fast double [[C2]], [[TMP5]]
-; CHECK-NEXT: [[Q1:%.*]] = fmul fast double [[C3]], [[TMP6]]
-; CHECK-NEXT: [[D1:%.*]] = fsub fast double [[P1]], [[Q1]]
; CHECK-NEXT: [[ACC:%.*]] = fmul fast double [[C4]], [[TMP5]]
; CHECK-NEXT: [[TMP2:%.*]] = fmul fast double [[C5]], [[TMP6]]
; CHECK-NEXT: [[D3_NEG:%.*]] = fsub fast double [[ACC]], [[TMP2]]
-; CHECK-NEXT: [[TMP7:%.*]] = load <2 x double>, ptr [[C6_PTR]], align 8
+; CHECK-NEXT: [[C6_PTR:%.*]] = getelementptr inbounds nuw i8, ptr [[BASE]], i64 64
+; CHECK-NEXT: [[C6:%.*]] = load double, ptr [[C6_PTR]], align 8
+; CHECK-NEXT: [[C7_PTR:%.*]] = getelementptr inbounds nuw i8, ptr [[BASE]], i64 72
+; CHECK-NEXT: [[C7:%.*]] = load double, ptr [[C7_PTR]], align 8
; CHECK-NEXT: [[T1_REASS_REASS:%.*]] = fmul fast double [[FACTOR_OP_FMUL3]], [[D2]]
; CHECK-NEXT: [[T2_REASS_REASS:%.*]] = fmul fast double [[D1]], [[FACTOR_OP_FMUL4]]
; CHECK-NEXT: [[S1:%.*]] = fadd fast double [[T2_REASS_REASS]], [[T1_REASS_REASS]]
; CHECK-NEXT: [[S2_NEG:%.*]] = fmul fast double [[D3_NEG]], [[FACTOR_OP_FMUL2]]
; CHECK-NEXT: [[RESULT:%.*]] = fadd fast double [[S1]], [[S2_NEG]]
-; CHECK-NEXT: [[TMP8:%.*]] = fmul fast <2 x double> [[TMP7]], [[TMP4]]
-; CHECK-NEXT: [[TMP9:%.*]] = extractelement <2 x double> [[TMP8]], i64 0
+; CHECK-NEXT: [[TMP9:%.*]] = fmul fast double [[C6]], [[TMP5]]
+; CHECK-NEXT: [[TMP10:%.*]] = fmul fast double [[C7]], [[TMP6]]
; CHECK-NEXT: [[REASS_ADD:%.*]] = fadd fast double [[RESULT]], [[TMP9]]
-; CHECK-NEXT: [[TMP10:%.*]] = extractelement <2 x double> [[TMP8]], i64 1
; CHECK-NEXT: [[S2_NEG1:%.*]] = fadd fast double [[ACC1]], [[TMP10]]
; CHECK-NEXT: [[RESULT1]] = fsub fast double [[S2_NEG1]], [[REASS_ADD]]
; CHECK-NEXT: [[NEXT]] = add nuw i64 [[I]], 10
>From 71dc2c44588efea3080c4722184c989f09c13bfd Mon Sep 17 00:00:00 2001
From: Dmitry Sidorov <Dmitry.Sidorov at amd.com>
Date: Mon, 31 Aug 2026 17:42:11 +0200
Subject: [PATCH 07/11] update stale test comments
---
.../SLPVectorizer/X86/fma-operand-index.ll | 12 +++++++-----
.../SLPVectorizer/X86/horizontal-fadd-with-sub.ll | 5 +++--
.../X86/select-logical-or-and-i1-vector.ll | 4 ++++
3 files changed, 14 insertions(+), 7 deletions(-)
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/fma-operand-index.ll b/llvm/test/Transforms/SLPVectorizer/X86/fma-operand-index.ll
index a5b28dd341e7a..b2fa19d65e2e2 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/fma-operand-index.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/fma-operand-index.ll
@@ -1,11 +1,13 @@
; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
; RUN: opt -S --passes=slp-vectorizer -mtriple=x86_64-unknown-linux-gnu -mcpu=znver4 < %s | FileCheck %s
-; A one-use fmul is left scalar so the backend can fuse it, but only operand 0
-; of the fadd/fsub is checked. The same fmul is kept or gathered depending on
-; which side it sits on. To be addressed in a follow-up.
+; A one-use fmul is left scalar so the backend can fuse it, whichever operand
+; of the fadd/fsub it sits on.
+;
+; FIXME: The scalar form gives up the wide x and y loads. #218755 makes the
+; veto depend on whether the target always profits from a scalar fma.
-; Both fmuls are operand 1, so they gather and cannot fuse.
+; Both fmuls are operand 1 and stay scalar to fuse.
define double @fmul_rhs_fadd(ptr %x, ptr %y, ptr %z) {
; CHECK-LABEL: define double @fmul_rhs_fadd(
; CHECK-SAME: ptr [[X:%.*]], ptr [[Y:%.*]], ptr [[Z:%.*]]) #[[ATTR0:[0-9]+]] {
@@ -83,7 +85,7 @@ entry:
ret double %sub1
}
-; Operand 0 is the one checked, so these stay scalar and fuse.
+; Fmuls on operand 0 likewise stay scalar and fuse.
define double @fmul_lhs_fadd(ptr %x, ptr %y, ptr %z) {
; CHECK-LABEL: define double @fmul_lhs_fadd(
; CHECK-SAME: ptr [[X:%.*]], ptr [[Y:%.*]], ptr [[Z:%.*]]) #[[ATTR0]] {
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/horizontal-fadd-with-sub.ll b/llvm/test/Transforms/SLPVectorizer/X86/horizontal-fadd-with-sub.ll
index 853ab156bed97..119df94ee12c2 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/horizontal-fadd-with-sub.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/horizontal-fadd-with-sub.ll
@@ -2,8 +2,9 @@
; RUN: opt -S --passes=slp-vectorizer -mtriple=x86_64-unknown-linux-gnu -mcpu=znver4 < %s | FileCheck %s
; An fadd reduction over an fsub/fneg chain is flattened with per-operand
-; signs, so subtracted leaves are vectorized and subtracted in the final
-; combine, forming per-lane fma.
+; signs, so subtracted leaves can be vectorized and subtracted in the final
+; combine, forming per-lane fma. Where the leaves are one-use fmuls the
+; operand aware fma pricing may keep the chain scalar instead.
define double @fsub_fmul_2(ptr %x, ptr %y, ptr %z) {
; CHECK-LABEL: define double @fsub_fmul_2(
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/select-logical-or-and-i1-vector.ll b/llvm/test/Transforms/SLPVectorizer/X86/select-logical-or-and-i1-vector.ll
index bb10829b8cca2..c94193c7fc592 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/select-logical-or-and-i1-vector.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/select-logical-or-and-i1-vector.ll
@@ -7,6 +7,10 @@
; Reduced from a real-world molecular docking kernel where independent
; scalar fcmps are combined with a loop-invariant scalar condition via
; logical select, feeding into float selects stored to consecutive memory.
+;
+; FIXME: The operand aware fma pricing keeps the one-use fmuls scalar and the
+; rest of the tree with them, so nothing is vectorized right now. Recover the
+; vectorization without breaking the scalar fma pricing.
define void @select_logical_or_i1(ptr %dst,
; CHECK-LABEL: define void @select_logical_or_i1(
>From 0a69f7777b41a4f1f0bbc39148f0c300af894aa1 Mon Sep 17 00:00:00 2001
From: Dmitry Sidorov <Dmitry.Sidorov at amd.com>
Date: Mon, 31 Aug 2026 21:34:53 +0200
Subject: [PATCH 08/11] pass real fmul operand index at reduction sites
---
.../Transforms/Vectorize/SLPVectorizer.cpp | 21 +++--
.../X86/redux-feed-buildvector.ll | 92 ++++++++++++++++---
2 files changed, 96 insertions(+), 17 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index b5fdb50fb7879..37c7284eea118 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -32280,9 +32280,9 @@ class HorizontalReduction {
auto *RdxOp = cast<Instruction>(U);
if (hasRequiredNumberOfUses(IsCmpSelMinMax, RdxOp)) {
if (RdxKind == RecurKind::FAdd) {
- InstructionCost FMACost =
- canConvertToFMA(RdxOp, getSameOpcode(RdxOp, TLI), DT, DL,
- *TTI, TLI, CostKind, /*FMulOpIdx=*/0);
+ InstructionCost FMACost = canConvertToFMA(
+ RdxOp, getSameOpcode(RdxOp, TLI), DT, DL, *TTI, TLI,
+ CostKind, RdxOp->getOperand(1) == RdxVal ? 1 : 0);
if (FMACost.isValid()) {
LLVM_DEBUG(dbgs() << "FMA cost: " << FMACost << "\n");
if (auto *I = dyn_cast<Instruction>(RdxVal)) {
@@ -32397,18 +32397,27 @@ class HorizontalReduction {
SmallVector<Value *> Ops;
FastMathFlags FMF;
FMF.set();
- for (Value *RdxVal : ReducedVals) {
+ unsigned FMulOpIdx = 0;
+ for (auto [Idx, RdxVal] : enumerate(ReducedVals)) {
if (!RdxVal->hasOneUse()) {
Ops.clear();
break;
}
+ User *U = RdxVal->user_back();
+ unsigned OpIdx = U->getOperand(1) == RdxVal ? 1 : 0;
+ if (Idx == 0)
+ FMulOpIdx = OpIdx;
+ else if (FMulOpIdx != OpIdx) {
+ Ops.clear();
+ break;
+ }
if (auto *FPCI = dyn_cast<FPMathOperator>(RdxVal))
FMF &= FPCI->getFastMathFlags();
- Ops.push_back(RdxVal->user_back());
+ Ops.push_back(U);
}
if (!Ops.empty()) {
FMACost = canConvertToFMA(Ops, getSameOpcode(Ops, TLI), DT, DL,
- *TTI, TLI, CostKind, /*FMulOpIdx=*/0);
+ *TTI, TLI, CostKind, FMulOpIdx);
if (FMACost.isValid()) {
// Calculate actual FMAD cost.
IntrinsicCostAttributes ICA(Intrinsic::fmuladd, RVecTy,
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/redux-feed-buildvector.ll b/llvm/test/Transforms/SLPVectorizer/X86/redux-feed-buildvector.ll
index 67054d202d5a2..73670db825f2e 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/redux-feed-buildvector.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/redux-feed-buildvector.ll
@@ -4,23 +4,93 @@
; The test represents the case with multiple vectorization possibilities
; but the most effective way to vectorize it is to match both 8-way reductions
; feeding the insertelement vector build sequence.
+;
+; FIXME: The operand aware fma pricing values the scalar form as 16 fmas and
+; declines both reductions, giving up the wide loads. Recover the reduction
+; vectorization here.
declare void @llvm.masked.scatter.v2f64.v2p0(<2 x double>, <2 x ptr>, i32 immarg, <2 x i1>)
define void @test(ptr nocapture readonly %arg, ptr nocapture readonly %arg1, ptr nocapture %arg2) {
; CHECK-LABEL: @test(
; CHECK-NEXT: entry:
-; CHECK-NEXT: [[TMP0:%.*]] = insertelement <8 x ptr> poison, ptr [[ARG:%.*]], i64 0
-; CHECK-NEXT: [[TMP1:%.*]] = shufflevector <8 x ptr> [[TMP0]], <8 x ptr> poison, <8 x i32> zeroinitializer
-; CHECK-NEXT: [[TMP5:%.*]] = getelementptr inbounds double, <8 x ptr> [[TMP1]], <8 x i64> <i64 1, i64 3, i64 5, i64 7, i64 9, i64 11, i64 13, i64 15>
-; CHECK-NEXT: [[GEP2_0:%.*]] = getelementptr inbounds double, ptr [[ARG1:%.*]], i64 16
-; CHECK-NEXT: [[TMP2:%.*]] = call <8 x double> @llvm.masked.gather.v8f64.v8p0(<8 x ptr> align 8 [[TMP5]], <8 x i1> splat (i1 true), <8 x double> poison)
-; CHECK-NEXT: [[TMP3:%.*]] = load <8 x double>, ptr [[GEP2_0]], align 8
-; CHECK-NEXT: [[TMP4:%.*]] = fmul fast <8 x double> [[TMP3]], [[TMP2]]
-; CHECK-NEXT: [[TMP6:%.*]] = load <8 x double>, ptr [[ARG1]], align 8
-; CHECK-NEXT: [[TMP8:%.*]] = fmul fast <8 x double> [[TMP6]], [[TMP2]]
-; CHECK-NEXT: [[TMP7:%.*]] = call fast double @llvm.vector.reduce.fadd.v8f64(double 0.000000e+00, <8 x double> [[TMP8]])
-; CHECK-NEXT: [[TMP11:%.*]] = call fast double @llvm.vector.reduce.fadd.v8f64(double 0.000000e+00, <8 x double> [[TMP4]])
+; CHECK-NEXT: [[GEP1_0:%.*]] = getelementptr inbounds double, ptr [[ARG:%.*]], i64 1
+; CHECK-NEXT: [[LD1_0:%.*]] = load double, ptr [[GEP1_0]], align 8
+; CHECK-NEXT: [[LD0_0:%.*]] = load double, ptr [[ARG1:%.*]], align 8
+; CHECK-NEXT: [[MUL1_0:%.*]] = fmul fast double [[LD0_0]], [[LD1_0]]
+; CHECK-NEXT: [[GEP2_0:%.*]] = getelementptr inbounds double, ptr [[ARG1]], i64 16
+; CHECK-NEXT: [[LD2_0:%.*]] = load double, ptr [[GEP2_0]], align 8
+; CHECK-NEXT: [[MUL2_0:%.*]] = fmul fast double [[LD2_0]], [[LD1_0]]
+; CHECK-NEXT: [[GEP1_1:%.*]] = getelementptr inbounds double, ptr [[ARG]], i64 3
+; CHECK-NEXT: [[LD1_1:%.*]] = load double, ptr [[GEP1_1]], align 8
+; CHECK-NEXT: [[GEP0_1:%.*]] = getelementptr inbounds double, ptr [[ARG1]], i64 1
+; CHECK-NEXT: [[LD0_1:%.*]] = load double, ptr [[GEP0_1]], align 8
+; CHECK-NEXT: [[MUL1_1:%.*]] = fmul fast double [[LD0_1]], [[LD1_1]]
+; CHECK-NEXT: [[RDX1_0:%.*]] = fadd fast double [[MUL1_0]], [[MUL1_1]]
+; CHECK-NEXT: [[GEP2_1:%.*]] = getelementptr inbounds double, ptr [[ARG1]], i64 17
+; CHECK-NEXT: [[LD2_1:%.*]] = load double, ptr [[GEP2_1]], align 8
+; CHECK-NEXT: [[MUL2_1:%.*]] = fmul fast double [[LD2_1]], [[LD1_1]]
+; CHECK-NEXT: [[RDX2_0:%.*]] = fadd fast double [[MUL2_0]], [[MUL2_1]]
+; CHECK-NEXT: [[GEP1_2:%.*]] = getelementptr inbounds double, ptr [[ARG]], i64 5
+; CHECK-NEXT: [[LD1_2:%.*]] = load double, ptr [[GEP1_2]], align 8
+; CHECK-NEXT: [[GEP0_2:%.*]] = getelementptr inbounds double, ptr [[ARG1]], i64 2
+; CHECK-NEXT: [[LD0_2:%.*]] = load double, ptr [[GEP0_2]], align 8
+; CHECK-NEXT: [[MUL1_2:%.*]] = fmul fast double [[LD0_2]], [[LD1_2]]
+; CHECK-NEXT: [[RDX1_1:%.*]] = fadd fast double [[RDX1_0]], [[MUL1_2]]
+; CHECK-NEXT: [[GEP2_2:%.*]] = getelementptr inbounds double, ptr [[ARG1]], i64 18
+; CHECK-NEXT: [[LD2_2:%.*]] = load double, ptr [[GEP2_2]], align 8
+; CHECK-NEXT: [[MUL2_2:%.*]] = fmul fast double [[LD2_2]], [[LD1_2]]
+; CHECK-NEXT: [[RDX2_1:%.*]] = fadd fast double [[RDX2_0]], [[MUL2_2]]
+; CHECK-NEXT: [[GEP1_3:%.*]] = getelementptr inbounds double, ptr [[ARG]], i64 7
+; CHECK-NEXT: [[LD1_3:%.*]] = load double, ptr [[GEP1_3]], align 8
+; CHECK-NEXT: [[GEP0_3:%.*]] = getelementptr inbounds double, ptr [[ARG1]], i64 3
+; CHECK-NEXT: [[LD0_3:%.*]] = load double, ptr [[GEP0_3]], align 8
+; CHECK-NEXT: [[MUL1_3:%.*]] = fmul fast double [[LD0_3]], [[LD1_3]]
+; CHECK-NEXT: [[RDX1_2:%.*]] = fadd fast double [[RDX1_1]], [[MUL1_3]]
+; CHECK-NEXT: [[GEP2_3:%.*]] = getelementptr inbounds double, ptr [[ARG1]], i64 19
+; CHECK-NEXT: [[LD2_3:%.*]] = load double, ptr [[GEP2_3]], align 8
+; CHECK-NEXT: [[MUL2_3:%.*]] = fmul fast double [[LD2_3]], [[LD1_3]]
+; CHECK-NEXT: [[RDX2_2:%.*]] = fadd fast double [[RDX2_1]], [[MUL2_3]]
+; CHECK-NEXT: [[GEP1_4:%.*]] = getelementptr inbounds double, ptr [[ARG]], i64 9
+; CHECK-NEXT: [[LD1_4:%.*]] = load double, ptr [[GEP1_4]], align 8
+; CHECK-NEXT: [[GEP0_4:%.*]] = getelementptr inbounds double, ptr [[ARG1]], i64 4
+; CHECK-NEXT: [[LD0_4:%.*]] = load double, ptr [[GEP0_4]], align 8
+; CHECK-NEXT: [[MUL1_4:%.*]] = fmul fast double [[LD0_4]], [[LD1_4]]
+; CHECK-NEXT: [[RDX1_3:%.*]] = fadd fast double [[RDX1_2]], [[MUL1_4]]
+; CHECK-NEXT: [[GEP2_4:%.*]] = getelementptr inbounds double, ptr [[ARG1]], i64 20
+; CHECK-NEXT: [[LD2_4:%.*]] = load double, ptr [[GEP2_4]], align 8
+; CHECK-NEXT: [[MUL2_4:%.*]] = fmul fast double [[LD2_4]], [[LD1_4]]
+; CHECK-NEXT: [[RDX2_3:%.*]] = fadd fast double [[RDX2_2]], [[MUL2_4]]
+; CHECK-NEXT: [[GEP1_5:%.*]] = getelementptr inbounds double, ptr [[ARG]], i64 11
+; CHECK-NEXT: [[LD1_5:%.*]] = load double, ptr [[GEP1_5]], align 8
+; CHECK-NEXT: [[GEP0_5:%.*]] = getelementptr inbounds double, ptr [[ARG1]], i64 5
+; CHECK-NEXT: [[LD0_5:%.*]] = load double, ptr [[GEP0_5]], align 8
+; CHECK-NEXT: [[MUL1_5:%.*]] = fmul fast double [[LD0_5]], [[LD1_5]]
+; CHECK-NEXT: [[RDX1_4:%.*]] = fadd fast double [[RDX1_3]], [[MUL1_5]]
+; CHECK-NEXT: [[GEP2_5:%.*]] = getelementptr inbounds double, ptr [[ARG1]], i64 21
+; CHECK-NEXT: [[LD2_5:%.*]] = load double, ptr [[GEP2_5]], align 8
+; CHECK-NEXT: [[MUL2_5:%.*]] = fmul fast double [[LD2_5]], [[LD1_5]]
+; CHECK-NEXT: [[RDX2_4:%.*]] = fadd fast double [[RDX2_3]], [[MUL2_5]]
+; CHECK-NEXT: [[GEP1_6:%.*]] = getelementptr inbounds double, ptr [[ARG]], i64 13
+; CHECK-NEXT: [[LD1_6:%.*]] = load double, ptr [[GEP1_6]], align 8
+; CHECK-NEXT: [[GEP0_6:%.*]] = getelementptr inbounds double, ptr [[ARG1]], i64 6
+; CHECK-NEXT: [[LD0_6:%.*]] = load double, ptr [[GEP0_6]], align 8
+; CHECK-NEXT: [[MUL1_6:%.*]] = fmul fast double [[LD0_6]], [[LD1_6]]
+; CHECK-NEXT: [[RDX1_5:%.*]] = fadd fast double [[RDX1_4]], [[MUL1_6]]
+; CHECK-NEXT: [[GEP2_6:%.*]] = getelementptr inbounds double, ptr [[ARG1]], i64 22
+; CHECK-NEXT: [[LD2_6:%.*]] = load double, ptr [[GEP2_6]], align 8
+; CHECK-NEXT: [[MUL2_6:%.*]] = fmul fast double [[LD2_6]], [[LD1_6]]
+; CHECK-NEXT: [[RDX2_5:%.*]] = fadd fast double [[RDX2_4]], [[MUL2_6]]
+; CHECK-NEXT: [[GEP1_7:%.*]] = getelementptr inbounds double, ptr [[ARG]], i64 15
+; CHECK-NEXT: [[LD1_7:%.*]] = load double, ptr [[GEP1_7]], align 8
+; CHECK-NEXT: [[GEP0_7:%.*]] = getelementptr inbounds double, ptr [[ARG1]], i64 7
+; CHECK-NEXT: [[LD0_7:%.*]] = load double, ptr [[GEP0_7]], align 8
+; CHECK-NEXT: [[MUL1_7:%.*]] = fmul fast double [[LD0_7]], [[LD1_7]]
+; CHECK-NEXT: [[TMP7:%.*]] = fadd fast double [[RDX1_5]], [[MUL1_7]]
+; CHECK-NEXT: [[GEP2_7:%.*]] = getelementptr inbounds double, ptr [[ARG1]], i64 23
+; CHECK-NEXT: [[LD2_7:%.*]] = load double, ptr [[GEP2_7]], align 8
+; CHECK-NEXT: [[MUL2_7:%.*]] = fmul fast double [[LD2_7]], [[LD1_7]]
+; CHECK-NEXT: [[TMP11:%.*]] = fadd fast double [[RDX2_5]], [[MUL2_7]]
; CHECK-NEXT: [[I142:%.*]] = insertelement <2 x double> poison, double [[TMP7]], i64 0
; CHECK-NEXT: [[I143:%.*]] = insertelement <2 x double> [[I142]], double [[TMP11]], i64 1
; CHECK-NEXT: [[P:%.*]] = getelementptr inbounds double, ptr [[ARG2:%.*]], <2 x i64> <i64 0, i64 16>
>From e86a361377a2a462b5b52913e6933a4baf1e1928 Mon Sep 17 00:00:00 2001
From: Dmitry Sidorov <Dmitry.Sidorov at amd.com>
Date: Wed, 2 Sep 2026 00:15:39 +0200
Subject: [PATCH 09/11] apply review
---
.../Transforms/Vectorize/SLPVectorizer.cpp | 23 ++++++++++++-------
.../Vectorize/SLPVectorizer/SLPUtils.cpp | 4 ++--
2 files changed, 17 insertions(+), 10 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index d72e7dd3eef1f..3a0ba2c724919 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -32457,10 +32457,12 @@ class HorizontalReduction {
for (User *U : RdxVal->users()) {
auto *RdxOp = cast<Instruction>(U);
if (hasRequiredNumberOfUses(IsCmpSelMinMax, RdxOp)) {
- if (RdxKind == RecurKind::FAdd) {
- InstructionCost FMACost = canConvertToFMA(
- RdxOp, getSameOpcode(RdxOp, TLI), DT, DL, *TTI, TLI,
- CostKind, RdxOp->getOperand(1) == RdxVal ? 1 : 0);
+ auto *BO = dyn_cast<BinaryOperator>(RdxOp);
+ if (BO && RdxKind == RecurKind::FAdd) {
+ unsigned OpIdx = BO->getOperand(1) == RdxVal ? 1 : 0;
+ InstructionCost FMACost =
+ canConvertToFMA(BO, getSameOpcode(BO, TLI), DT, DL, *TTI,
+ TLI, CostKind, OpIdx);
if (FMACost.isValid()) {
LLVM_DEBUG(dbgs() << "FMA cost: " << FMACost << "\n");
if (auto *I = dyn_cast<Instruction>(RdxVal)) {
@@ -32582,13 +32584,18 @@ class HorizontalReduction {
break;
}
User *U = RdxVal->user_back();
- unsigned OpIdx = U->getOperand(1) == RdxVal ? 1 : 0;
- if (Idx == 0)
- FMulOpIdx = OpIdx;
- else if (FMulOpIdx != OpIdx) {
+ auto *BO = dyn_cast<BinaryOperator>(U);
+ if (!BO) {
Ops.clear();
break;
}
+ unsigned OpIdx = BO->getOperand(1) == RdxVal ? 1 : 0;
+ if (Idx != 0 && FMulOpIdx != OpIdx) {
+ Ops.clear();
+ break;
+ }
+ if (Idx == 0)
+ FMulOpIdx = OpIdx;
if (auto *FPCI = dyn_cast<FPMathOperator>(RdxVal))
FMF &= FPCI->getFastMathFlags();
Ops.push_back(U);
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp
index 960d49866ad2c..9c3da31db33c9 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp
@@ -775,8 +775,8 @@ unsigned getFMulOperandIdx(const Instruction *I) {
assert((I->getOpcode() == Instruction::FAdd ||
I->getOpcode() == Instruction::FSub) &&
"Expected an fadd/fsub-like instruction");
- for (unsigned Idx : seq<unsigned>(I->getNumOperands()))
- if (match(I->getOperand(Idx), m_OneUse(m_FMul(m_Value(), m_Value()))))
+ for (auto [Idx, Op] : enumerate(I->operand_values()))
+ if (match(Op, m_OneUse(m_FMul(m_Value(), m_Value()))))
return Idx;
return 0;
}
>From 82693e96d2a00f03a4ed0c85c49dbf3f2b4da2f3 Mon Sep 17 00:00:00 2001
From: Dmitry Sidorov <Dmitry.Sidorov at amd.com>
Date: Wed, 9 Sep 2026 13:21:07 +0200
Subject: [PATCH 10/11] [SLP] Guard the reduction FMA cost on a valid
add/sub-like state
The two reduction call sites built an InstructionsState and handed it straight
to canConvertToFMA. A reduction operand can be a non add/sub-like user, and the
state can be invalid, so check both before pricing the fusion.
Assisted-by: Claude Code Opus 5
---
llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp | 16 ++++++++++------
1 file changed, 10 insertions(+), 6 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 3a0ba2c724919..ecd6f418234bf 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -32458,11 +32458,13 @@ class HorizontalReduction {
auto *RdxOp = cast<Instruction>(U);
if (hasRequiredNumberOfUses(IsCmpSelMinMax, RdxOp)) {
auto *BO = dyn_cast<BinaryOperator>(RdxOp);
- if (BO && RdxKind == RecurKind::FAdd) {
+ InstructionsState RdxOpS = BO && RdxKind == RecurKind::FAdd
+ ? getSameOpcode(BO, TLI)
+ : InstructionsState::invalid();
+ if (RdxOpS && RdxOpS.isAddSubLikeOp()) {
unsigned OpIdx = BO->getOperand(1) == RdxVal ? 1 : 0;
- InstructionCost FMACost =
- canConvertToFMA(BO, getSameOpcode(BO, TLI), DT, DL, *TTI,
- TLI, CostKind, OpIdx);
+ InstructionCost FMACost = canConvertToFMA(
+ BO, RdxOpS, DT, DL, *TTI, TLI, CostKind, OpIdx);
if (FMACost.isValid()) {
LLVM_DEBUG(dbgs() << "FMA cost: " << FMACost << "\n");
if (auto *I = dyn_cast<Instruction>(RdxVal)) {
@@ -32601,8 +32603,10 @@ class HorizontalReduction {
Ops.push_back(U);
}
if (!Ops.empty()) {
- FMACost = canConvertToFMA(Ops, getSameOpcode(Ops, TLI), DT, DL,
- *TTI, TLI, CostKind, FMulOpIdx);
+ InstructionsState S = getSameOpcode(Ops, TLI);
+ if (S && S.isAddSubLikeOp())
+ FMACost = canConvertToFMA(Ops, S, DT, DL, *TTI, TLI, CostKind,
+ FMulOpIdx);
if (FMACost.isValid()) {
// Calculate actual FMAD cost.
IntrinsicCostAttributes ICA(Intrinsic::fmuladd, RVecTy,
>From f20af57e8910f159666558d56c50a76776f7c182 Mon Sep 17 00:00:00 2001
From: Dmitry Sidorov <Dmitry.Sidorov at amd.com>
Date: Wed, 9 Sep 2026 14:14:54 +0200
Subject: [PATCH 11/11] update test
---
.../AArch64/hor-fp-reduction-instcount.ll | 49 ++++++++++++-------
1 file changed, 31 insertions(+), 18 deletions(-)
diff --git a/llvm/test/Transforms/SLPVectorizer/AArch64/hor-fp-reduction-instcount.ll b/llvm/test/Transforms/SLPVectorizer/AArch64/hor-fp-reduction-instcount.ll
index 304db13bf6bb0..4a33c7cb55631 100644
--- a/llvm/test/Transforms/SLPVectorizer/AArch64/hor-fp-reduction-instcount.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AArch64/hor-fp-reduction-instcount.ll
@@ -2,46 +2,59 @@
; RUN: opt < %s -S -passes=slp-vectorizer -mtriple=aarch64-unknown-linux -mcpu=neoverse-v2 | FileCheck %s --check-prefix=NARROW
; RUN: opt < %s -S -passes=slp-vectorizer -mtriple=aarch64-unknown-linux -mcpu=neoverse-v1 | FileCheck %s --check-prefix=WIDE
-; A horizontal double reduction with gathered operands is not profitable on
-; targets whose vector register holds only 2 doubles: the operand gathering and
-; the reduction epilogue produce more instructions than the scalar reduction.
-; On wider targets (4+ doubles per register) it stays profitable.
+; A horizontal double reduction with gathered operands is not profitable. The
+; gathering and the reduction epilogue cost more than the scalar chain, whose
+; fadd links all fuse into fmadd.
define double @f64_red_gather(ptr %a, ptr %w) {
; NARROW-LABEL: define double @f64_red_gather(
; NARROW-SAME: ptr [[A:%.*]], ptr [[W:%.*]]) #[[ATTR0:[0-9]+]] {
; NARROW-NEXT: [[A0:%.*]] = load double, ptr [[A]], align 8
+; NARROW-NEXT: [[W0:%.*]] = load double, ptr [[W]], align 8
+; NARROW-NEXT: [[P0:%.*]] = fmul reassoc contract double [[A0]], [[W0]]
; NARROW-NEXT: [[A1P:%.*]] = getelementptr double, ptr [[A]], i64 3
; NARROW-NEXT: [[A1:%.*]] = load double, ptr [[A1P]], align 8
+; NARROW-NEXT: [[W1P:%.*]] = getelementptr double, ptr [[W]], i64 1
+; NARROW-NEXT: [[W1:%.*]] = load double, ptr [[W1P]], align 8
+; NARROW-NEXT: [[P1:%.*]] = fmul reassoc contract double [[A1]], [[W1]]
; NARROW-NEXT: [[A2P:%.*]] = getelementptr double, ptr [[A]], i64 6
; NARROW-NEXT: [[A2:%.*]] = load double, ptr [[A2P]], align 8
+; NARROW-NEXT: [[W2P:%.*]] = getelementptr double, ptr [[W]], i64 2
+; NARROW-NEXT: [[W2:%.*]] = load double, ptr [[W2P]], align 8
+; NARROW-NEXT: [[P2:%.*]] = fmul reassoc contract double [[A2]], [[W2]]
; NARROW-NEXT: [[A3P:%.*]] = getelementptr double, ptr [[A]], i64 9
; NARROW-NEXT: [[A3:%.*]] = load double, ptr [[A3P]], align 8
-; NARROW-NEXT: [[TMP1:%.*]] = load <4 x double>, ptr [[W]], align 8
-; NARROW-NEXT: [[TMP2:%.*]] = insertelement <4 x double> poison, double [[A0]], i64 0
-; NARROW-NEXT: [[TMP3:%.*]] = insertelement <4 x double> [[TMP2]], double [[A1]], i64 1
-; NARROW-NEXT: [[TMP4:%.*]] = insertelement <4 x double> [[TMP3]], double [[A2]], i64 2
-; NARROW-NEXT: [[TMP5:%.*]] = insertelement <4 x double> [[TMP4]], double [[A3]], i64 3
-; NARROW-NEXT: [[TMP6:%.*]] = fmul reassoc contract <4 x double> [[TMP5]], [[TMP1]]
-; NARROW-NEXT: [[TMP7:%.*]] = call reassoc contract double @llvm.vector.reduce.fadd.v4f64(double -0.000000e+00, <4 x double> [[TMP6]])
+; NARROW-NEXT: [[W3P:%.*]] = getelementptr double, ptr [[W]], i64 3
+; NARROW-NEXT: [[W3:%.*]] = load double, ptr [[W3P]], align 8
+; NARROW-NEXT: [[P3:%.*]] = fmul reassoc contract double [[A3]], [[W3]]
+; NARROW-NEXT: [[R0:%.*]] = fadd reassoc contract double [[P0]], [[P1]]
+; NARROW-NEXT: [[R1:%.*]] = fadd reassoc contract double [[R0]], [[P2]]
+; NARROW-NEXT: [[TMP7:%.*]] = fadd reassoc contract double [[R1]], [[P3]]
; NARROW-NEXT: ret double [[TMP7]]
;
; WIDE-LABEL: define double @f64_red_gather(
; WIDE-SAME: ptr [[A:%.*]], ptr [[W:%.*]]) #[[ATTR0:[0-9]+]] {
; WIDE-NEXT: [[A0:%.*]] = load double, ptr [[A]], align 8
+; WIDE-NEXT: [[W0:%.*]] = load double, ptr [[W]], align 8
+; WIDE-NEXT: [[P0:%.*]] = fmul reassoc contract double [[A0]], [[W0]]
; WIDE-NEXT: [[A1P:%.*]] = getelementptr double, ptr [[A]], i64 3
; WIDE-NEXT: [[A1:%.*]] = load double, ptr [[A1P]], align 8
+; WIDE-NEXT: [[W1P:%.*]] = getelementptr double, ptr [[W]], i64 1
+; WIDE-NEXT: [[W1:%.*]] = load double, ptr [[W1P]], align 8
+; WIDE-NEXT: [[P1:%.*]] = fmul reassoc contract double [[A1]], [[W1]]
; WIDE-NEXT: [[A2P:%.*]] = getelementptr double, ptr [[A]], i64 6
; WIDE-NEXT: [[A2:%.*]] = load double, ptr [[A2P]], align 8
+; WIDE-NEXT: [[W2P:%.*]] = getelementptr double, ptr [[W]], i64 2
+; WIDE-NEXT: [[W2:%.*]] = load double, ptr [[W2P]], align 8
+; WIDE-NEXT: [[P2:%.*]] = fmul reassoc contract double [[A2]], [[W2]]
; WIDE-NEXT: [[A3P:%.*]] = getelementptr double, ptr [[A]], i64 9
; WIDE-NEXT: [[A3:%.*]] = load double, ptr [[A3P]], align 8
-; WIDE-NEXT: [[TMP1:%.*]] = load <4 x double>, ptr [[W]], align 8
-; WIDE-NEXT: [[TMP2:%.*]] = insertelement <4 x double> poison, double [[A0]], i64 0
-; WIDE-NEXT: [[TMP3:%.*]] = insertelement <4 x double> [[TMP2]], double [[A1]], i64 1
-; WIDE-NEXT: [[TMP4:%.*]] = insertelement <4 x double> [[TMP3]], double [[A2]], i64 2
-; WIDE-NEXT: [[TMP5:%.*]] = insertelement <4 x double> [[TMP4]], double [[A3]], i64 3
-; WIDE-NEXT: [[TMP6:%.*]] = fmul reassoc contract <4 x double> [[TMP5]], [[TMP1]]
-; WIDE-NEXT: [[TMP7:%.*]] = call reassoc contract double @llvm.vector.reduce.fadd.v4f64(double -0.000000e+00, <4 x double> [[TMP6]])
+; WIDE-NEXT: [[W3P:%.*]] = getelementptr double, ptr [[W]], i64 3
+; WIDE-NEXT: [[W3:%.*]] = load double, ptr [[W3P]], align 8
+; WIDE-NEXT: [[P3:%.*]] = fmul reassoc contract double [[A3]], [[W3]]
+; WIDE-NEXT: [[R0:%.*]] = fadd reassoc contract double [[P0]], [[P1]]
+; WIDE-NEXT: [[R1:%.*]] = fadd reassoc contract double [[R0]], [[P2]]
+; WIDE-NEXT: [[TMP7:%.*]] = fadd reassoc contract double [[R1]], [[P3]]
; WIDE-NEXT: ret double [[TMP7]]
;
%a0 = load double, ptr %a
More information about the llvm-commits
mailing list