[llvm] c0bbc06 - [LV] Handle FSub Partial Reductions (#197134)
via llvm-commits
llvm-commits at lists.llvm.org
Wed May 13 02:31:59 PDT 2026
Author: Jacob Crawley
Date: 2026-05-13T09:31:54Z
New Revision: c0bbc06c692efac81c28a232f0f91198a36356b0
URL: https://github.com/llvm/llvm-project/commit/c0bbc06c692efac81c28a232f0f91198a36356b0
DIFF: https://github.com/llvm/llvm-project/commit/c0bbc06c692efac81c28a232f0f91198a36356b0.diff
LOG: [LV] Handle FSub Partial Reductions (#197134)
Reland #191186 after fixing up test failures
Introduces a new RecurKind value 'FSub' in order to handle partial
reductions of floating point values.
This is done by following the existing method for integer partial
reductions, doing a positive accumulation followed by a final
subtraction in the middle block.
Added:
llvm/test/Transforms/LoopVectorize/AArch64/partial-reduce-fsub-chained.ll
Modified:
llvm/include/llvm/Analysis/IVDescriptors.h
llvm/lib/Analysis/IVDescriptors.cpp
llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
llvm/lib/Transforms/Utils/LoopUtils.cpp
llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
llvm/test/Transforms/LoopVectorize/AArch64/partial-reduce-sub-epilogue-vec.ll
llvm/test/Transforms/LoopVectorize/AArch64/partial-reduce-sub.ll
Removed:
################################################################################
diff --git a/llvm/include/llvm/Analysis/IVDescriptors.h b/llvm/include/llvm/Analysis/IVDescriptors.h
index 05c58d1a20afb..2120eb8cd9914 100644
--- a/llvm/include/llvm/Analysis/IVDescriptors.h
+++ b/llvm/include/llvm/Analysis/IVDescriptors.h
@@ -46,6 +46,8 @@ enum class RecurKind {
UMin, ///< Unsigned integer min implemented in terms of select(cmp()).
UMax, ///< Unsigned integer max implemented in terms of select(cmp()).
FAdd, ///< Sum of floats.
+ FAddChainWithSubs, ///< A chain of fadds and fsubs.
+ FSub, ///< Subtraction of floats.
FMul, ///< Product of floats.
FMin, ///< FP min implemented in terms of select(cmp()).
FMax, ///< FP max implemented in terms of select(cmp()).
@@ -245,6 +247,9 @@ class RecurrenceDescriptor {
/// Returns true if the recurrence kind is a floating point kind.
LLVM_ABI static bool isFloatingPointRecurrenceKind(RecurKind Kind);
+ /// Returns true if the recurrence kind is for a sub operation.
+ LLVM_ABI static bool isSubRecurrenceKind(RecurKind Kind);
+
/// Returns true if the recurrence kind is an integer min/max kind.
static bool isIntMinMaxRecurrenceKind(RecurKind Kind) {
return Kind == RecurKind::UMin || Kind == RecurKind::UMax ||
diff --git a/llvm/lib/Analysis/IVDescriptors.cpp b/llvm/lib/Analysis/IVDescriptors.cpp
index 7549e14366d2c..22a519026f63f 100644
--- a/llvm/lib/Analysis/IVDescriptors.cpp
+++ b/llvm/lib/Analysis/IVDescriptors.cpp
@@ -92,6 +92,10 @@ static Instruction *lookThroughAnd(PHINode *Phi, Type *&RT,
return Phi;
}
+bool RecurrenceDescriptor::isSubRecurrenceKind(RecurKind Kind) {
+ return Kind == RecurKind::Sub || Kind == RecurKind::FSub;
+}
+
/// Compute the minimal bit width needed to represent a reduction whose exit
/// instruction is given by Exit.
static std::pair<Type *, bool> computeRecurrenceType(Instruction *Exit,
@@ -1009,13 +1013,18 @@ RecurrenceDescriptor::isRecurrenceInstr(Loop *L, PHINode *OrigPhi,
return InstDesc(Kind == RecurKind::FMul, I,
I->hasAllowReassoc() ? nullptr : I);
case Instruction::FSub:
+ return InstDesc(Kind == RecurKind::FSub ||
+ Kind == RecurKind::FAddChainWithSubs,
+ I, I->hasAllowReassoc() ? nullptr : I);
case Instruction::FAdd:
- return InstDesc(Kind == RecurKind::FAdd, I,
- I->hasAllowReassoc() ? nullptr : I);
+ return InstDesc(Kind == RecurKind::FAdd ||
+ Kind == RecurKind::FAddChainWithSubs,
+ I, I->hasAllowReassoc() ? nullptr : I);
case Instruction::Select:
- if (Kind == RecurKind::FAdd || Kind == RecurKind::FMul ||
- Kind == RecurKind::Add || Kind == RecurKind::Mul ||
- Kind == RecurKind::Sub || Kind == RecurKind::AddChainWithSubs)
+ if (isSubRecurrenceKind(Kind) || Kind == RecurKind::FAdd ||
+ Kind == RecurKind::FMul || Kind == RecurKind::Add ||
+ Kind == RecurKind::Mul || Kind == RecurKind::AddChainWithSubs ||
+ Kind == RecurKind::FAddChainWithSubs)
return isConditionalRdxPattern(I);
if (isFindRecurrenceKind(Kind) && SE)
return isFindPattern(L, OrigPhi, I, *SE);
@@ -1104,10 +1113,20 @@ bool RecurrenceDescriptor::isReductionPHI(PHINode *Phi, Loop *TheLoop,
LLVM_DEBUG(dbgs() << "Found an FMult reduction PHI." << *Phi << "\n");
return true;
}
+ if (AddReductionVar(Phi, RecurKind::FSub, TheLoop, RedDes, DB, AC, DT, SE)) {
+ LLVM_DEBUG(dbgs() << "Found an FSub reduction PHI." << *Phi << "\n");
+ return true;
+ }
if (AddReductionVar(Phi, RecurKind::FAdd, TheLoop, RedDes, DB, AC, DT, SE)) {
LLVM_DEBUG(dbgs() << "Found an FAdd reduction PHI." << *Phi << "\n");
return true;
}
+ if (AddReductionVar(Phi, RecurKind::FAddChainWithSubs, TheLoop, RedDes, DB,
+ AC, DT, SE)) {
+ LLVM_DEBUG(dbgs() << "Found a chained FADD-FSUB chained reduction PHI."
+ << *Phi << "\n");
+ return true;
+ }
if (AddReductionVar(Phi, RecurKind::FMulAdd, TheLoop, RedDes, DB, AC, DT,
SE)) {
LLVM_DEBUG(dbgs() << "Found an FMulAdd reduction PHI." << *Phi << "\n");
@@ -1224,8 +1243,11 @@ unsigned RecurrenceDescriptor::getOpcode(RecurKind Kind) {
case RecurKind::FMul:
return Instruction::FMul;
case RecurKind::FMulAdd:
+ case RecurKind::FAddChainWithSubs:
case RecurKind::FAdd:
return Instruction::FAdd;
+ case RecurKind::FSub:
+ return Instruction::FSub;
case RecurKind::SMax:
case RecurKind::SMin:
case RecurKind::UMax:
@@ -1302,6 +1324,10 @@ RecurrenceDescriptor::getReductionOpChain(PHINode *Phi, Loop *L) const {
Kind == RecurKind::AddChainWithSubs)
return true;
+ if (Cur->getOpcode() == Instruction::FSub &&
+ Kind == RecurKind::FAddChainWithSubs)
+ return true;
+
return Cur->getOpcode() == getOpcode();
};
diff --git a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
index 42cb253209495..2d6f11e4979f0 100644
--- a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
+++ b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
@@ -5668,6 +5668,7 @@ bool AArch64TTIImpl::isLegalToVectorizeReduction(
switch (RdxDesc.getRecurrenceKind()) {
case RecurKind::Sub:
case RecurKind::AddChainWithSubs:
+ case RecurKind::FAddChainWithSubs:
case RecurKind::Add:
case RecurKind::FAdd:
case RecurKind::And:
@@ -5989,7 +5990,7 @@ InstructionCost AArch64TTIImpl::getPartialReductionCost(
return Invalid;
if ((Opcode != Instruction::Add && Opcode != Instruction::Sub &&
- Opcode != Instruction::FAdd) ||
+ Opcode != Instruction::FAdd && Opcode != Instruction::FSub) ||
OpAExtend == TTI::PR_None)
return Invalid;
@@ -6056,7 +6057,7 @@ InstructionCost AArch64TTIImpl::getPartialReductionCost(
NEONPred);
};
- bool IsSub = Opcode == Instruction::Sub;
+ bool IsSub = (Opcode == Instruction::Sub) || (Opcode == Instruction::FSub);
InstructionCost Cost = InputLT.first * TTI::TCC_Basic;
// Integer partial sub-reductions that don't map to a specific instruction,
// carry an extra cost for implementing a double negation:
diff --git a/llvm/lib/Transforms/Utils/LoopUtils.cpp b/llvm/lib/Transforms/Utils/LoopUtils.cpp
index cc965cb36c36e..f0586e4f0f464 100644
--- a/llvm/lib/Transforms/Utils/LoopUtils.cpp
+++ b/llvm/lib/Transforms/Utils/LoopUtils.cpp
@@ -1110,6 +1110,8 @@ constexpr Intrinsic::ID llvm::getReductionIntrinsicID(RecurKind RK) {
case RecurKind::Xor:
return Intrinsic::vector_reduce_xor;
case RecurKind::FMulAdd:
+ case RecurKind::FAddChainWithSubs:
+ case RecurKind::FSub:
case RecurKind::FAdd:
return Intrinsic::vector_reduce_fadd;
case RecurKind::FMul:
@@ -1567,6 +1569,8 @@ Value *llvm::createSimpleReduction(IRBuilderBase &Builder, Value *Src,
case RecurKind::FMaximumNum:
return Builder.CreateUnaryIntrinsic(getReductionIntrinsicID(RdxKind), Src);
case RecurKind::FMulAdd:
+ case RecurKind::FAddChainWithSubs:
+ case RecurKind::FSub:
case RecurKind::FAdd:
return Builder.CreateFAddReduce(getIdentity(), Src);
case RecurKind::FMul:
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 9fec3b74d9630..7baf1ad299118 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -7781,16 +7781,28 @@ static SmallVector<Instruction *> preparePlanForEpilogueVectorLoop(
// reduction start value in a final subtraction. Update it to use the
// resume value from the main vector loop.
if (PhiR->getVFScaleFactor() > 1 &&
- PhiR->getRecurrenceKind() == RecurKind::Sub) {
+ RecurrenceDescriptor::isSubRecurrenceKind(
+ PhiR->getRecurrenceKind())) {
auto *Sub = cast<VPInstruction>(RdxResult->getSingleUser());
- assert(Sub->getOpcode() == Instruction::Sub && "Unexpected opcode");
+ assert((Sub->getOpcode() == Instruction::Sub ||
+ Sub->getOpcode() == Instruction::FSub) &&
+ "Unexpected opcode");
assert(isa<VPIRValue>(Sub->getOperand(0)) &&
"Expected operand to match the original start value of the "
"reduction");
- assert(VPlanPatternMatch::match(VPI->getOperand(0),
- VPlanPatternMatch::m_ZeroInt()) &&
- "Expected start value for partial sub-reduction to start at "
- "zero");
+ // For integer sub-reductions, verify start value is zero.
+ // For FP sub-reductions, verify start value is negative zero.
+ [[maybe_unused]] auto StartValueIsIdentity = [&] {
+ Value *IdentityValue = getRecurrenceIdentity(
+ PhiR->getRecurrenceKind(), ResumeV->getType(),
+ PhiR->getFastMathFlags());
+ auto *StartValue = dyn_cast<VPIRValue>(VPI->getOperand(0));
+ return StartValue && StartValue->getValue() == IdentityValue;
+ };
+ assert(StartValueIsIdentity() &&
+ "Expected start value for partial sub-reduction to be zero "
+ "(or negative zero)");
+
Sub->setOperand(0, StartVal);
} else
VPI->setOperand(0, StartVal);
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 2825a4c7acd73..9592771917995 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -29541,6 +29541,8 @@ class HorizontalReduction {
// res = vv
break;
case RecurKind::Sub:
+ case RecurKind::FSub:
+ case RecurKind::FAddChainWithSubs:
case RecurKind::AddChainWithSubs:
case RecurKind::Mul:
case RecurKind::FMul:
@@ -29692,6 +29694,8 @@ class HorizontalReduction {
// res = vv
return VectorizedValue;
case RecurKind::Sub:
+ case RecurKind::FSub:
+ case RecurKind::FAddChainWithSubs:
case RecurKind::AddChainWithSubs:
case RecurKind::Mul:
case RecurKind::FMul:
@@ -29795,6 +29799,8 @@ class HorizontalReduction {
return Builder.CreateFMul(VectorizedValue, Scale);
}
case RecurKind::Sub:
+ case RecurKind::FSub:
+ case RecurKind::FAddChainWithSubs:
case RecurKind::AddChainWithSubs:
case RecurKind::Mul:
case RecurKind::FMul:
diff --git a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
index 6d5db90436c79..7f3bb12ffdf6b 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
@@ -578,6 +578,14 @@ bool VPInstruction::canGenerateScalarForFirstLane() const {
}
}
+static Instruction::BinaryOps getSubRecurOpcode(RecurKind Kind) {
+ if (Kind == RecurKind::Sub)
+ return Instruction::Add;
+ if (Kind == RecurKind::FSub)
+ return Instruction::FAdd;
+ llvm_unreachable("RecurKind should be Sub/FSub.");
+}
+
Value *VPInstruction::generate(VPTransformState &State) {
IRBuilderBase &Builder = State.Builder;
@@ -791,8 +799,8 @@ Value *VPInstruction::generate(VPTransformState &State) {
// For sub-recurrences, each part's reduction variable is already
// negative, we need to do: reduce.add(-acc_uf0 + -acc_uf1)
Instruction::BinaryOps Opcode =
- RK == RecurKind::Sub
- ? Instruction::Add
+ RecurrenceDescriptor::isSubRecurrenceKind(RK)
+ ? getSubRecurOpcode(RK)
: (Instruction::BinaryOps)RecurrenceDescriptor::getOpcode(RK);
ReducedPartRdx =
Builder.CreateBinOp(Opcode, RdxPart, ReducedPartRdx, "bin.rdx");
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index 4050f9edd5b32..673355ffb1c96 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -5977,6 +5977,9 @@ static void transformToPartialReduction(const VPPartialReductionChain &Chain,
ExtendedOp = NegRecipe;
}
+ assert((Chain.RK != RecurKind::FAddChainWithSubs) &&
+ "FSub chain reduction isn't supported");
+
// FIXME: Do these transforms before invoking the cost-model.
ExtendedOp = optimizeExtendsForPartialReduction(ExtendedOp, TypeInfo);
@@ -6028,7 +6031,7 @@ static void transformToPartialReduction(const VPPartialReductionChain &Chain,
// If this is the last value in a sub-reduction chain, then update the PHI
// node to start at `0` and update the reduction-result to subtract from
// the PHI's start value.
- if (Chain.RK != RecurKind::Sub)
+ if (Chain.RK != RecurKind::Sub && Chain.RK != RecurKind::FSub)
return;
VPValue *OldStartValue = StartInst->getOperand(0);
@@ -6039,7 +6042,8 @@ static void transformToPartialReduction(const VPPartialReductionChain &Chain,
assert(RdxResult && "Could not find reduction result");
VPBuilder Builder = VPBuilder::getToInsertAfter(RdxResult);
- constexpr unsigned SubOpc = Instruction::BinaryOps::Sub;
+ unsigned SubOpc = Chain.RK == RecurKind::FSub ? Instruction::BinaryOps::FSub
+ : Instruction::BinaryOps::Sub;
VPInstruction *NewResult = Builder.createNaryOp(
SubOpc, {OldStartValue, RdxResult}, VPIRFlags::getDefaultFlags(SubOpc),
RdxPhi->getDebugLoc());
@@ -6064,12 +6068,20 @@ getPartialReductionLinkCost(VPCostContext &CostCtx,
if (RdxType->isFloatingPointTy())
Flags = Link.ReductionBinOp->getFastMathFlags();
- unsigned Opcode = Link.RK == RecurKind::Sub
- ? (unsigned)Instruction::Add
- : Link.ReductionBinOp->getOpcode();
+ auto GetLinkOpcode = [&Link]() -> unsigned {
+ switch (Link.RK) {
+ case RecurKind::Sub:
+ return Instruction::Add;
+ case RecurKind::FSub:
+ return Instruction::FAdd;
+ default:
+ return Link.ReductionBinOp->getOpcode();
+ }
+ };
+
return CostCtx.TTI.getPartialReductionCost(
- Opcode, ExtendedOp.ExtendA.SrcType, ExtendedOp.ExtendB.SrcType, RdxType,
- VF, ExtendedOp.ExtendA.Kind, ExtendedOp.ExtendB.Kind, BinOpc,
+ GetLinkOpcode(), ExtendedOp.ExtendA.SrcType, ExtendedOp.ExtendB.SrcType,
+ RdxType, VF, ExtendedOp.ExtendA.Kind, ExtendedOp.ExtendB.Kind, BinOpc,
CostCtx.CostKind, Flags);
}
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/partial-reduce-fsub-chained.ll b/llvm/test/Transforms/LoopVectorize/AArch64/partial-reduce-fsub-chained.ll
new file mode 100644
index 0000000000000..fb97f618b9acc
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/partial-reduce-fsub-chained.ll
@@ -0,0 +1,40 @@
+; REQUIRES: asserts
+; RUN: not --crash opt -passes=loop-vectorize -S %s 2>&1 | FileCheck %s --check-prefix=ASSERTION
+
+; Tests a partial reduction with an fadd->fsub chain.
+; There's an assertion preventing this type of partial reduction from
+; being generated as the current codegen for this case is incorrect.
+
+target datalayout = "e-m:e-i8:8:32-i16:16:32-i64:64-i128:128-n32:64-S128"
+target triple = "aarch64-none-unknown-elf"
+
+; ASSERTION: (Chain.RK != RecurKind::FAddChainWithSubs)
+define float @fadd_fsub_reduction(float %startval, ptr %src1, ptr %src2, ptr %src3) #0 {
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i32 [ 0, %entry ], [ %iv.next, %loop ]
+ %accum = phi float [ %startval, %entry ], [ %sub, %loop ]
+ %src1.gep = getelementptr half, ptr %src1, i32 %iv
+ %src1.load = load half, ptr %src1.gep, align 4
+ %src1.load.ext = fpext half %src1.load to float
+ %src2.gep = getelementptr half, ptr %src2, i32 %iv
+ %src2.load = load half, ptr %src2.gep, align 4
+ %src2.load.ext = fpext half %src2.load to float
+ %src3.gep = getelementptr half, ptr %src3, i32 %iv
+ %src3.load = load half, ptr %src3.gep, align 4
+ %src3.load.ext = fpext half %src3.load to float
+ %mul1 = fmul reassoc contract float %src1.load.ext, %src2.load.ext
+ %add = fadd reassoc contract float %accum, %mul1
+ %mul2 = fmul reassoc contract float %src3.load.ext, %src1.load.ext
+ %sub = fsub reassoc contract float %add, %mul2
+ %iv.next = add i32 %iv, 1
+ %exitcond.not = icmp eq i32 %iv, 1024
+ br i1 %exitcond.not, label %exit, label %loop
+
+exit:
+ ret float %sub
+}
+
+attributes #0 = { vscale_range(1,16) "target-features"="+sve2" }
\ No newline at end of file
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/partial-reduce-sub-epilogue-vec.ll b/llvm/test/Transforms/LoopVectorize/AArch64/partial-reduce-sub-epilogue-vec.ll
index 8db3a3294d7bc..29acde48c5d02 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/partial-reduce-sub-epilogue-vec.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/partial-reduce-sub-epilogue-vec.ll
@@ -185,4 +185,322 @@ exit:
ret i32 %sub
}
+define float @fsub_reduction(float %startval, ptr %src1, ptr %src2) #1 {
+; CHECK-EPI-LABEL: define float @fsub_reduction(
+; CHECK-EPI-SAME: float [[STARTVAL:%.*]], ptr [[SRC1:%.*]], ptr [[SRC2:%.*]]) #[[ATTR1:[0-9]+]] {
+; CHECK-EPI-NEXT: [[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]:
+; CHECK-EPI-NEXT: br i1 false, label %[[VEC_EPILOG_VECTOR_BODY:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK1:.*]]
+; CHECK-EPI: [[VECTOR_MAIN_LOOP_ITER_CHECK1]]:
+; CHECK-EPI-NEXT: br i1 false, label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_PH1:.*]]
+; CHECK-EPI: [[VECTOR_PH1]]:
+; CHECK-EPI-NEXT: br label %[[VECTOR_PH:.*]]
+; CHECK-EPI: [[VECTOR_PH]]:
+; CHECK-EPI-NEXT: [[INDEX1:%.*]] = phi i32 [ 0, %[[VECTOR_PH1]] ], [ [[INDEX_NEXT1:%.*]], %[[VECTOR_PH]] ]
+; CHECK-EPI-NEXT: [[VEC_PHI:%.*]] = phi <8 x float> [ splat (float -0.000000e+00), %[[VECTOR_PH1]] ], [ [[PARTIAL_REDUCE:%.*]], %[[VECTOR_PH]] ]
+; CHECK-EPI-NEXT: [[TMP0:%.*]] = getelementptr half, ptr [[SRC1]], i32 [[INDEX1]]
+; CHECK-EPI-NEXT: [[WIDE_LOAD:%.*]] = load <16 x half>, ptr [[TMP0]], align 4
+; CHECK-EPI-NEXT: [[TMP1:%.*]] = getelementptr half, ptr [[SRC2]], i32 [[INDEX1]]
+; CHECK-EPI-NEXT: [[WIDE_LOAD1:%.*]] = load <16 x half>, ptr [[TMP1]], align 4
+; CHECK-EPI-NEXT: [[TMP5:%.*]] = fpext <16 x half> [[WIDE_LOAD]] to <16 x float>
+; CHECK-EPI-NEXT: [[TMP10:%.*]] = fpext <16 x half> [[WIDE_LOAD1]] to <16 x float>
+; CHECK-EPI-NEXT: [[TMP11:%.*]] = fmul reassoc contract <16 x float> [[TMP5]], [[TMP10]]
+; CHECK-EPI-NEXT: [[PARTIAL_REDUCE]] = call reassoc contract <8 x float> @llvm.vector.partial.reduce.fadd.v8f32.v16f32(<8 x float> [[VEC_PHI]], <16 x float> [[TMP11]])
+; CHECK-EPI-NEXT: [[INDEX_NEXT1]] = add nuw i32 [[INDEX1]], 16
+; CHECK-EPI-NEXT: [[TMP13:%.*]] = icmp eq i32 [[INDEX_NEXT1]], 32
+; CHECK-EPI-NEXT: br i1 [[TMP13]], label %[[MIDDLE_BLOCK1:.*]], label %[[VECTOR_PH]], !llvm.loop [[LOOP6:![0-9]+]]
+; CHECK-EPI: [[MIDDLE_BLOCK1]]:
+; CHECK-EPI-NEXT: [[TMP6:%.*]] = call reassoc contract float @llvm.vector.reduce.fadd.v8f32(float -0.000000e+00, <8 x float> [[PARTIAL_REDUCE]])
+; CHECK-EPI-NEXT: [[TMP7:%.*]] = fsub float [[STARTVAL]], [[TMP6]]
+; CHECK-EPI-NEXT: br i1 false, label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
+; CHECK-EPI: [[VEC_EPILOG_ITER_CHECK]]:
+; CHECK-EPI-NEXT: br i1 false, label %[[VEC_EPILOG_VECTOR_BODY]], label %[[VEC_EPILOG_PH]], !prof [[PROF3]]
+; CHECK-EPI: [[VEC_EPILOG_PH]]:
+; CHECK-EPI-NEXT: [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i32 [ 32, %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK1]] ]
+; CHECK-EPI-NEXT: [[BC_MERGE_RDX:%.*]] = phi float [ [[TMP7]], %[[VEC_EPILOG_ITER_CHECK]] ], [ [[STARTVAL]], %[[VECTOR_MAIN_LOOP_ITER_CHECK1]] ]
+; CHECK-EPI-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK-EPI: [[VECTOR_BODY]]:
+; CHECK-EPI-NEXT: [[INDEX:%.*]] = phi i32 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-EPI-NEXT: [[VEC_PHI3:%.*]] = phi <2 x float> [ splat (float -0.000000e+00), %[[VEC_EPILOG_PH]] ], [ [[PARTIAL_REDUCE6:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-EPI-NEXT: [[TMP8:%.*]] = getelementptr half, ptr [[SRC1]], i32 [[INDEX]]
+; CHECK-EPI-NEXT: [[WIDE_LOAD4:%.*]] = load <4 x half>, ptr [[TMP8]], align 4
+; CHECK-EPI-NEXT: [[TMP9:%.*]] = getelementptr half, ptr [[SRC2]], i32 [[INDEX]]
+; CHECK-EPI-NEXT: [[WIDE_LOAD5:%.*]] = load <4 x half>, ptr [[TMP9]], align 4
+; CHECK-EPI-NEXT: [[TMP2:%.*]] = fpext <4 x half> [[WIDE_LOAD4]] to <4 x float>
+; CHECK-EPI-NEXT: [[TMP3:%.*]] = fpext <4 x half> [[WIDE_LOAD5]] to <4 x float>
+; CHECK-EPI-NEXT: [[TMP4:%.*]] = fmul reassoc contract <4 x float> [[TMP2]], [[TMP3]]
+; CHECK-EPI-NEXT: [[PARTIAL_REDUCE6]] = call reassoc contract <2 x float> @llvm.vector.partial.reduce.fadd.v2f32.v4f32(<2 x float> [[VEC_PHI3]], <4 x float> [[TMP4]])
+; CHECK-EPI-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 4
+; CHECK-EPI-NEXT: [[TMP16:%.*]] = icmp eq i32 [[INDEX_NEXT]], 40
+; CHECK-EPI-NEXT: br i1 [[TMP16]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP7:![0-9]+]]
+; CHECK-EPI: [[MIDDLE_BLOCK]]:
+; CHECK-EPI-NEXT: [[TMP14:%.*]] = call reassoc contract float @llvm.vector.reduce.fadd.v2f32(float -0.000000e+00, <2 x float> [[PARTIAL_REDUCE6]])
+; CHECK-EPI-NEXT: [[TMP15:%.*]] = fsub float [[BC_MERGE_RDX]], [[TMP14]]
+; CHECK-EPI-NEXT: br i1 false, label %[[EXIT]], label %[[VEC_EPILOG_VECTOR_BODY]]
+; CHECK-EPI: [[VEC_EPILOG_VECTOR_BODY]]:
+; CHECK-EPI-NEXT: [[BC_RESUME_VAL:%.*]] = phi i32 [ 40, %[[MIDDLE_BLOCK]] ], [ 32, %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-EPI-NEXT: [[BC_MERGE_RDX8:%.*]] = phi float [ [[TMP15]], %[[MIDDLE_BLOCK]] ], [ [[TMP7]], %[[VEC_EPILOG_ITER_CHECK]] ], [ [[STARTVAL]], %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-EPI-NEXT: br label %[[LOOP:.*]]
+; CHECK-EPI: [[LOOP]]:
+; CHECK-EPI-NEXT: [[IV:%.*]] = phi i32 [ [[BC_RESUME_VAL]], %[[VEC_EPILOG_VECTOR_BODY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-EPI-NEXT: [[ACCUM:%.*]] = phi float [ [[BC_MERGE_RDX8]], %[[VEC_EPILOG_VECTOR_BODY]] ], [ [[SUB:%.*]], %[[LOOP]] ]
+; CHECK-EPI-NEXT: [[SRC1_GEP:%.*]] = getelementptr half, ptr [[SRC1]], i32 [[IV]]
+; CHECK-EPI-NEXT: [[SRC1_LOAD:%.*]] = load half, ptr [[SRC1_GEP]], align 4
+; CHECK-EPI-NEXT: [[SRC1_LOAD_EXT:%.*]] = fpext half [[SRC1_LOAD]] to float
+; CHECK-EPI-NEXT: [[SRC2_GEP:%.*]] = getelementptr half, ptr [[SRC2]], i32 [[IV]]
+; CHECK-EPI-NEXT: [[SRC2_LOAD:%.*]] = load half, ptr [[SRC2_GEP]], align 4
+; CHECK-EPI-NEXT: [[SRC2_LOAD_EXT:%.*]] = fpext half [[SRC2_LOAD]] to float
+; CHECK-EPI-NEXT: [[MUL:%.*]] = fmul reassoc contract float [[SRC1_LOAD_EXT]], [[SRC2_LOAD_EXT]]
+; CHECK-EPI-NEXT: [[SUB]] = fsub reassoc contract float [[ACCUM]], [[MUL]]
+; CHECK-EPI-NEXT: [[IV_NEXT]] = add i32 [[IV]], 1
+; CHECK-EPI-NEXT: [[EXITCOND_NOT:%.*]] = icmp eq i32 [[IV]], 40
+; CHECK-EPI-NEXT: br i1 [[EXITCOND_NOT]], label %[[EXIT]], label %[[LOOP]], !llvm.loop [[LOOP8:![0-9]+]]
+; CHECK-EPI: [[EXIT]]:
+; CHECK-EPI-NEXT: [[SUB_LCSSA:%.*]] = phi float [ [[SUB]], %[[LOOP]] ], [ [[TMP7]], %[[MIDDLE_BLOCK1]] ], [ [[TMP15]], %[[MIDDLE_BLOCK]] ]
+; CHECK-EPI-NEXT: ret float [[SUB_LCSSA]]
+;
+; CHECK-PARTIAL-RED-EPI-LABEL: define float @fsub_reduction(
+; CHECK-PARTIAL-RED-EPI-SAME: float [[STARTVAL:%.*]], ptr [[SRC1:%.*]], ptr [[SRC2:%.*]]) #[[ATTR1:[0-9]+]] {
+; CHECK-PARTIAL-RED-EPI-NEXT: [[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]:
+; CHECK-PARTIAL-RED-EPI-NEXT: br i1 false, label %[[VEC_EPILOG_VECTOR_BODY:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK1:.*]]
+; CHECK-PARTIAL-RED-EPI: [[VECTOR_MAIN_LOOP_ITER_CHECK1]]:
+; CHECK-PARTIAL-RED-EPI-NEXT: br i1 false, label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_PH1:.*]]
+; CHECK-PARTIAL-RED-EPI: [[VECTOR_PH1]]:
+; CHECK-PARTIAL-RED-EPI-NEXT: br label %[[VECTOR_PH:.*]]
+; CHECK-PARTIAL-RED-EPI: [[VECTOR_PH]]:
+; CHECK-PARTIAL-RED-EPI-NEXT: [[INDEX1:%.*]] = phi i32 [ 0, %[[VECTOR_PH1]] ], [ [[INDEX_NEXT1:%.*]], %[[VECTOR_PH]] ]
+; CHECK-PARTIAL-RED-EPI-NEXT: [[VEC_PHI1:%.*]] = phi <8 x float> [ splat (float -0.000000e+00), %[[VECTOR_PH1]] ], [ [[PARTIAL_REDUCE1:%.*]], %[[VECTOR_PH]] ]
+; CHECK-PARTIAL-RED-EPI-NEXT: [[TMP8:%.*]] = getelementptr half, ptr [[SRC1]], i32 [[INDEX1]]
+; CHECK-PARTIAL-RED-EPI-NEXT: [[WIDE_LOAD2:%.*]] = load <16 x half>, ptr [[TMP8]], align 4
+; CHECK-PARTIAL-RED-EPI-NEXT: [[TMP9:%.*]] = getelementptr half, ptr [[SRC2]], i32 [[INDEX1]]
+; CHECK-PARTIAL-RED-EPI-NEXT: [[WIDE_LOAD3:%.*]] = load <16 x half>, ptr [[TMP9]], align 4
+; CHECK-PARTIAL-RED-EPI-NEXT: [[TMP10:%.*]] = fpext <16 x half> [[WIDE_LOAD2]] to <16 x float>
+; CHECK-PARTIAL-RED-EPI-NEXT: [[TMP11:%.*]] = fpext <16 x half> [[WIDE_LOAD3]] to <16 x float>
+; CHECK-PARTIAL-RED-EPI-NEXT: [[TMP12:%.*]] = fmul reassoc contract <16 x float> [[TMP10]], [[TMP11]]
+; CHECK-PARTIAL-RED-EPI-NEXT: [[PARTIAL_REDUCE1]] = call reassoc contract <8 x float> @llvm.vector.partial.reduce.fadd.v8f32.v16f32(<8 x float> [[VEC_PHI1]], <16 x float> [[TMP12]])
+; CHECK-PARTIAL-RED-EPI-NEXT: [[INDEX_NEXT1]] = add nuw i32 [[INDEX1]], 16
+; CHECK-PARTIAL-RED-EPI-NEXT: [[TMP5:%.*]] = icmp eq i32 [[INDEX_NEXT1]], 32
+; CHECK-PARTIAL-RED-EPI-NEXT: br i1 [[TMP5]], label %[[MIDDLE_BLOCK1:.*]], label %[[VECTOR_PH]], !llvm.loop [[LOOP6:![0-9]+]]
+; CHECK-PARTIAL-RED-EPI: [[MIDDLE_BLOCK1]]:
+; CHECK-PARTIAL-RED-EPI-NEXT: [[TMP14:%.*]] = call reassoc contract float @llvm.vector.reduce.fadd.v8f32(float -0.000000e+00, <8 x float> [[PARTIAL_REDUCE1]])
+; CHECK-PARTIAL-RED-EPI-NEXT: [[TMP7:%.*]] = fsub float [[STARTVAL]], [[TMP14]]
+; CHECK-PARTIAL-RED-EPI-NEXT: br i1 false, label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
+; CHECK-PARTIAL-RED-EPI: [[VEC_EPILOG_ITER_CHECK]]:
+; CHECK-PARTIAL-RED-EPI-NEXT: br i1 false, label %[[VEC_EPILOG_VECTOR_BODY]], label %[[VEC_EPILOG_PH]], !prof [[PROF3]]
+; CHECK-PARTIAL-RED-EPI: [[VEC_EPILOG_PH]]:
+; CHECK-PARTIAL-RED-EPI-NEXT: [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i32 [ 32, %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK1]] ]
+; CHECK-PARTIAL-RED-EPI-NEXT: [[BC_MERGE_RDX:%.*]] = phi float [ [[TMP7]], %[[VEC_EPILOG_ITER_CHECK]] ], [ [[STARTVAL]], %[[VECTOR_MAIN_LOOP_ITER_CHECK1]] ]
+; CHECK-PARTIAL-RED-EPI-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK-PARTIAL-RED-EPI: [[VECTOR_BODY]]:
+; CHECK-PARTIAL-RED-EPI-NEXT: [[INDEX:%.*]] = phi i32 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-PARTIAL-RED-EPI-NEXT: [[VEC_PHI:%.*]] = phi <4 x float> [ splat (float -0.000000e+00), %[[VEC_EPILOG_PH]] ], [ [[PARTIAL_REDUCE:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-PARTIAL-RED-EPI-NEXT: [[TMP0:%.*]] = getelementptr half, ptr [[SRC1]], i32 [[INDEX]]
+; CHECK-PARTIAL-RED-EPI-NEXT: [[WIDE_LOAD:%.*]] = load <8 x half>, ptr [[TMP0]], align 4
+; CHECK-PARTIAL-RED-EPI-NEXT: [[TMP1:%.*]] = getelementptr half, ptr [[SRC2]], i32 [[INDEX]]
+; CHECK-PARTIAL-RED-EPI-NEXT: [[WIDE_LOAD1:%.*]] = load <8 x half>, ptr [[TMP1]], align 4
+; CHECK-PARTIAL-RED-EPI-NEXT: [[TMP2:%.*]] = fpext <8 x half> [[WIDE_LOAD]] to <8 x float>
+; CHECK-PARTIAL-RED-EPI-NEXT: [[TMP3:%.*]] = fpext <8 x half> [[WIDE_LOAD1]] to <8 x float>
+; CHECK-PARTIAL-RED-EPI-NEXT: [[TMP4:%.*]] = fmul reassoc contract <8 x float> [[TMP2]], [[TMP3]]
+; CHECK-PARTIAL-RED-EPI-NEXT: [[PARTIAL_REDUCE]] = call reassoc contract <4 x float> @llvm.vector.partial.reduce.fadd.v4f32.v8f32(<4 x float> [[VEC_PHI]], <8 x float> [[TMP4]])
+; CHECK-PARTIAL-RED-EPI-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 8
+; CHECK-PARTIAL-RED-EPI-NEXT: [[TMP13:%.*]] = icmp eq i32 [[INDEX_NEXT]], 40
+; CHECK-PARTIAL-RED-EPI-NEXT: br i1 [[TMP13]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP7:![0-9]+]]
+; CHECK-PARTIAL-RED-EPI: [[MIDDLE_BLOCK]]:
+; CHECK-PARTIAL-RED-EPI-NEXT: [[TMP6:%.*]] = call reassoc contract float @llvm.vector.reduce.fadd.v4f32(float -0.000000e+00, <4 x float> [[PARTIAL_REDUCE]])
+; CHECK-PARTIAL-RED-EPI-NEXT: [[TMP15:%.*]] = fsub float [[BC_MERGE_RDX]], [[TMP6]]
+; CHECK-PARTIAL-RED-EPI-NEXT: br i1 false, label %[[EXIT]], label %[[VEC_EPILOG_VECTOR_BODY]]
+; CHECK-PARTIAL-RED-EPI: [[VEC_EPILOG_VECTOR_BODY]]:
+; CHECK-PARTIAL-RED-EPI-NEXT: [[BC_RESUME_VAL:%.*]] = phi i32 [ 40, %[[MIDDLE_BLOCK]] ], [ 32, %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-PARTIAL-RED-EPI-NEXT: [[BC_MERGE_RDX8:%.*]] = phi float [ [[TMP15]], %[[MIDDLE_BLOCK]] ], [ [[TMP7]], %[[VEC_EPILOG_ITER_CHECK]] ], [ [[STARTVAL]], %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-PARTIAL-RED-EPI-NEXT: br label %[[LOOP:.*]]
+; CHECK-PARTIAL-RED-EPI: [[LOOP]]:
+; CHECK-PARTIAL-RED-EPI-NEXT: [[IV:%.*]] = phi i32 [ [[BC_RESUME_VAL]], %[[VEC_EPILOG_VECTOR_BODY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-PARTIAL-RED-EPI-NEXT: [[ACCUM:%.*]] = phi float [ [[BC_MERGE_RDX8]], %[[VEC_EPILOG_VECTOR_BODY]] ], [ [[SUB:%.*]], %[[LOOP]] ]
+; CHECK-PARTIAL-RED-EPI-NEXT: [[SRC1_GEP:%.*]] = getelementptr half, ptr [[SRC1]], i32 [[IV]]
+; CHECK-PARTIAL-RED-EPI-NEXT: [[SRC1_LOAD:%.*]] = load half, ptr [[SRC1_GEP]], align 4
+; CHECK-PARTIAL-RED-EPI-NEXT: [[SRC1_LOAD_EXT:%.*]] = fpext half [[SRC1_LOAD]] to float
+; CHECK-PARTIAL-RED-EPI-NEXT: [[SRC2_GEP:%.*]] = getelementptr half, ptr [[SRC2]], i32 [[IV]]
+; CHECK-PARTIAL-RED-EPI-NEXT: [[SRC2_LOAD:%.*]] = load half, ptr [[SRC2_GEP]], align 4
+; CHECK-PARTIAL-RED-EPI-NEXT: [[SRC2_LOAD_EXT:%.*]] = fpext half [[SRC2_LOAD]] to float
+; CHECK-PARTIAL-RED-EPI-NEXT: [[MUL:%.*]] = fmul reassoc contract float [[SRC1_LOAD_EXT]], [[SRC2_LOAD_EXT]]
+; CHECK-PARTIAL-RED-EPI-NEXT: [[SUB]] = fsub reassoc contract float [[ACCUM]], [[MUL]]
+; CHECK-PARTIAL-RED-EPI-NEXT: [[IV_NEXT]] = add i32 [[IV]], 1
+; CHECK-PARTIAL-RED-EPI-NEXT: [[EXITCOND_NOT:%.*]] = icmp eq i32 [[IV]], 40
+; CHECK-PARTIAL-RED-EPI-NEXT: br i1 [[EXITCOND_NOT]], label %[[EXIT]], label %[[LOOP]], !llvm.loop [[LOOP8:![0-9]+]]
+; CHECK-PARTIAL-RED-EPI: [[EXIT]]:
+; CHECK-PARTIAL-RED-EPI-NEXT: [[SUB_LCSSA:%.*]] = phi float [ [[SUB]], %[[LOOP]] ], [ [[TMP7]], %[[MIDDLE_BLOCK1]] ], [ [[TMP15]], %[[MIDDLE_BLOCK]] ]
+; CHECK-PARTIAL-RED-EPI-NEXT: ret float [[SUB_LCSSA]]
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i32 [ 0, %entry ], [ %iv.next, %loop ]
+ %accum = phi float [ %startval, %entry ], [ %sub, %loop ]
+ %src1.gep = getelementptr half, ptr %src1, i32 %iv
+ %src1.load = load half, ptr %src1.gep, align 4
+ %src1.load.ext = fpext half %src1.load to float
+ %src2.gep = getelementptr half, ptr %src2, i32 %iv
+ %src2.load = load half, ptr %src2.gep, align 4
+ %src2.load.ext = fpext half %src2.load to float
+ %mul = fmul reassoc contract float %src1.load.ext, %src2.load.ext
+ %sub = fsub reassoc contract float %accum, %mul
+ %iv.next = add i32 %iv, 1
+ %exitcond.not = icmp eq i32 %iv, 40
+ br i1 %exitcond.not, label %exit, label %loop, !llvm.loop !0
+
+exit:
+ ret float %sub
+}
+
+define float @fsub_reduction_nsz(ptr %a, ptr %b, ptr %c, i64 %n) #1{
+; CHECK-EPI-LABEL: define float @fsub_reduction_nsz(
+; CHECK-EPI-SAME: ptr [[A:%.*]], ptr [[B:%.*]], ptr [[C:%.*]], i64 [[N:%.*]]) #[[ATTR1]] {
+; CHECK-EPI-NEXT: [[ITER_CHECK:.*]]:
+; CHECK-EPI-NEXT: br i1 false, label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; CHECK-EPI: [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; CHECK-EPI-NEXT: br i1 false, label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK-EPI: [[VECTOR_PH]]:
+; CHECK-EPI-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK-EPI: [[VECTOR_BODY]]:
+; CHECK-EPI-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-EPI-NEXT: [[VEC_PHI:%.*]] = phi <4 x float> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[PARTIAL_REDUCE3:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-EPI-NEXT: [[TMP0:%.*]] = getelementptr half, ptr [[A]], i64 [[INDEX]]
+; CHECK-EPI-NEXT: [[WIDE_LOAD:%.*]] = load <8 x half>, ptr [[TMP0]], align 2
+; CHECK-EPI-NEXT: [[TMP1:%.*]] = getelementptr half, ptr [[B]], i64 [[INDEX]]
+; CHECK-EPI-NEXT: [[WIDE_LOAD1:%.*]] = load <8 x half>, ptr [[TMP1]], align 2
+; CHECK-EPI-NEXT: [[TMP2:%.*]] = getelementptr half, ptr [[C]], i64 [[INDEX]]
+; CHECK-EPI-NEXT: [[WIDE_LOAD2:%.*]] = load <8 x half>, ptr [[TMP2]], align 2
+; CHECK-EPI-NEXT: [[TMP3:%.*]] = fpext <8 x half> [[WIDE_LOAD1]] to <8 x float>
+; CHECK-EPI-NEXT: [[TMP4:%.*]] = fpext <8 x half> [[WIDE_LOAD]] to <8 x float>
+; CHECK-EPI-NEXT: [[TMP5:%.*]] = fmul reassoc nsz contract <8 x float> [[TMP3]], [[TMP4]]
+; CHECK-EPI-NEXT: [[PARTIAL_REDUCE:%.*]] = call reassoc nsz contract <4 x float> @llvm.vector.partial.reduce.fadd.v4f32.v8f32(<4 x float> [[VEC_PHI]], <8 x float> [[TMP5]])
+; CHECK-EPI-NEXT: [[TMP6:%.*]] = fpext <8 x half> [[WIDE_LOAD2]] to <8 x float>
+; CHECK-EPI-NEXT: [[TMP7:%.*]] = fmul reassoc nsz contract <8 x float> [[TMP6]], [[TMP4]]
+; CHECK-EPI-NEXT: [[PARTIAL_REDUCE3]] = call reassoc nsz contract <4 x float> @llvm.vector.partial.reduce.fadd.v4f32.v8f32(<4 x float> [[PARTIAL_REDUCE]], <8 x float> [[TMP7]])
+; CHECK-EPI-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
+; CHECK-EPI-NEXT: [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024
+; CHECK-EPI-NEXT: br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP9:![0-9]+]]
+; CHECK-EPI: [[MIDDLE_BLOCK]]:
+; CHECK-EPI-NEXT: [[TMP9:%.*]] = call reassoc nsz contract float @llvm.vector.reduce.fadd.v4f32(float 0.000000e+00, <4 x float> [[PARTIAL_REDUCE3]])
+; CHECK-EPI-NEXT: [[TMP10:%.*]] = fsub float 0.000000e+00, [[TMP9]]
+; CHECK-EPI-NEXT: br i1 true, label %[[FOR_EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
+; CHECK-EPI: [[VEC_EPILOG_ITER_CHECK]]:
+; CHECK-EPI-NEXT: br i1 true, label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF10:![0-9]+]]
+; CHECK-EPI: [[VEC_EPILOG_PH]]:
+; CHECK-EPI-NEXT: [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ 1024, %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-EPI-NEXT: [[BC_MERGE_RDX:%.*]] = phi float [ [[TMP10]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0.000000e+00, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-EPI-NEXT: br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
+; CHECK-EPI: [[VEC_EPILOG_VECTOR_BODY]]:
+; CHECK-EPI-NEXT: [[INDEX4:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT11:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-EPI-NEXT: [[VEC_PHI5:%.*]] = phi <2 x float> [ zeroinitializer, %[[VEC_EPILOG_PH]] ], [ [[PARTIAL_REDUCE10:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-EPI-NEXT: [[TMP11:%.*]] = getelementptr half, ptr [[A]], i64 [[INDEX4]]
+; CHECK-EPI-NEXT: [[WIDE_LOAD6:%.*]] = load <4 x half>, ptr [[TMP11]], align 2
+; CHECK-EPI-NEXT: [[TMP12:%.*]] = getelementptr half, ptr [[B]], i64 [[INDEX4]]
+; CHECK-EPI-NEXT: [[WIDE_LOAD7:%.*]] = load <4 x half>, ptr [[TMP12]], align 2
+; CHECK-EPI-NEXT: [[TMP13:%.*]] = getelementptr half, ptr [[C]], i64 [[INDEX4]]
+; CHECK-EPI-NEXT: [[WIDE_LOAD8:%.*]] = load <4 x half>, ptr [[TMP13]], align 2
+; CHECK-EPI-NEXT: [[TMP14:%.*]] = fpext <4 x half> [[WIDE_LOAD7]] to <4 x float>
+; CHECK-EPI-NEXT: [[TMP15:%.*]] = fpext <4 x half> [[WIDE_LOAD6]] to <4 x float>
+; CHECK-EPI-NEXT: [[TMP16:%.*]] = fmul reassoc nsz contract <4 x float> [[TMP14]], [[TMP15]]
+; CHECK-EPI-NEXT: [[PARTIAL_REDUCE9:%.*]] = call reassoc nsz contract <2 x float> @llvm.vector.partial.reduce.fadd.v2f32.v4f32(<2 x float> [[VEC_PHI5]], <4 x float> [[TMP16]])
+; CHECK-EPI-NEXT: [[TMP17:%.*]] = fpext <4 x half> [[WIDE_LOAD8]] to <4 x float>
+; CHECK-EPI-NEXT: [[TMP18:%.*]] = fmul reassoc nsz contract <4 x float> [[TMP17]], [[TMP15]]
+; CHECK-EPI-NEXT: [[PARTIAL_REDUCE10]] = call reassoc nsz contract <2 x float> @llvm.vector.partial.reduce.fadd.v2f32.v4f32(<2 x float> [[PARTIAL_REDUCE9]], <4 x float> [[TMP18]])
+; CHECK-EPI-NEXT: [[INDEX_NEXT11]] = add nuw i64 [[INDEX4]], 4
+; CHECK-EPI-NEXT: [[TMP19:%.*]] = icmp eq i64 [[INDEX_NEXT11]], 1024
+; CHECK-EPI-NEXT: br i1 [[TMP19]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP11:![0-9]+]]
+; CHECK-EPI: [[VEC_EPILOG_MIDDLE_BLOCK]]:
+; CHECK-EPI-NEXT: [[TMP20:%.*]] = call reassoc nsz contract float @llvm.vector.reduce.fadd.v2f32(float 0.000000e+00, <2 x float> [[PARTIAL_REDUCE10]])
+; CHECK-EPI-NEXT: [[TMP21:%.*]] = fsub float [[BC_MERGE_RDX]], [[TMP20]]
+; CHECK-EPI-NEXT: br i1 true, label %[[FOR_EXIT]], label %[[VEC_EPILOG_SCALAR_PH]]
+; CHECK-EPI: [[VEC_EPILOG_SCALAR_PH]]:
+; CHECK-EPI-NEXT: [[BC_RESUME_VAL:%.*]] = phi i64 [ 1024, %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ 1024, %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ]
+; CHECK-EPI-NEXT: [[BC_MERGE_RDX12:%.*]] = phi float [ [[TMP21]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[TMP10]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0.000000e+00, %[[ITER_CHECK]] ]
+; CHECK-EPI-NEXT: br label %[[FOR_BODY:.*]]
+; CHECK-EPI: [[FOR_BODY]]:
+; CHECK-EPI-NEXT: [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[VEC_EPILOG_SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[FOR_BODY]] ]
+; CHECK-EPI-NEXT: [[ACCUM:%.*]] = phi float [ [[BC_MERGE_RDX12]], %[[VEC_EPILOG_SCALAR_PH]] ], [ [[SUB:%.*]], %[[FOR_BODY]] ]
+; CHECK-EPI-NEXT: [[GEP_A:%.*]] = getelementptr half, ptr [[A]], i64 [[IV]]
+; CHECK-EPI-NEXT: [[LOAD_A:%.*]] = load half, ptr [[GEP_A]], align 2
+; CHECK-EPI-NEXT: [[EXT_A:%.*]] = fpext half [[LOAD_A]] to float
+; CHECK-EPI-NEXT: [[GEP_B:%.*]] = getelementptr half, ptr [[B]], i64 [[IV]]
+; CHECK-EPI-NEXT: [[LOAD_B:%.*]] = load half, ptr [[GEP_B]], align 2
+; CHECK-EPI-NEXT: [[EXT_B:%.*]] = fpext half [[LOAD_B]] to float
+; CHECK-EPI-NEXT: [[GEP_C:%.*]] = getelementptr half, ptr [[C]], i64 [[IV]]
+; CHECK-EPI-NEXT: [[LOAD_C:%.*]] = load half, ptr [[GEP_C]], align 2
+; CHECK-EPI-NEXT: [[EXT_C:%.*]] = fpext half [[LOAD_C]] to float
+; CHECK-EPI-NEXT: [[MUL_AB:%.*]] = fmul reassoc nsz contract float [[EXT_B]], [[EXT_A]]
+; CHECK-EPI-NEXT: [[MUL_AC:%.*]] = fmul reassoc nsz contract float [[EXT_C]], [[EXT_A]]
+; CHECK-EPI-NEXT: [[SUB_AB:%.*]] = fsub reassoc nsz contract float [[ACCUM]], [[MUL_AB]]
+; CHECK-EPI-NEXT: [[SUB]] = fsub reassoc nsz contract float [[SUB_AB]], [[MUL_AC]]
+; CHECK-EPI-NEXT: [[IV_NEXT]] = add i64 [[IV]], 1
+; CHECK-EPI-NEXT: [[EXITCOND_NOT:%.*]] = icmp eq i64 [[IV_NEXT]], 1024
+; CHECK-EPI-NEXT: br i1 [[EXITCOND_NOT]], label %[[FOR_EXIT]], label %[[FOR_BODY]], !llvm.loop [[LOOP12:![0-9]+]]
+; CHECK-EPI: [[FOR_EXIT]]:
+; CHECK-EPI-NEXT: [[SUB_LCSSA:%.*]] = phi float [ [[SUB]], %[[FOR_BODY]] ], [ [[TMP10]], %[[MIDDLE_BLOCK]] ], [ [[TMP21]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ]
+; CHECK-EPI-NEXT: ret float [[SUB_LCSSA]]
+;
+; CHECK-PARTIAL-RED-EPI-LABEL: define float @fsub_reduction_nsz(
+; CHECK-PARTIAL-RED-EPI-SAME: ptr [[A:%.*]], ptr [[B:%.*]], ptr [[C:%.*]], i64 [[N:%.*]]) #[[ATTR1]] {
+; CHECK-PARTIAL-RED-EPI-NEXT: [[ENTRY:.*:]]
+; CHECK-PARTIAL-RED-EPI-NEXT: br label %[[VECTOR_PH:.*]]
+; CHECK-PARTIAL-RED-EPI: [[VECTOR_PH]]:
+; CHECK-PARTIAL-RED-EPI-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK-PARTIAL-RED-EPI: [[VECTOR_BODY]]:
+; CHECK-PARTIAL-RED-EPI-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-PARTIAL-RED-EPI-NEXT: [[VEC_PHI:%.*]] = phi <4 x float> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[PARTIAL_REDUCE3:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-PARTIAL-RED-EPI-NEXT: [[TMP0:%.*]] = getelementptr half, ptr [[A]], i64 [[INDEX]]
+; CHECK-PARTIAL-RED-EPI-NEXT: [[WIDE_LOAD:%.*]] = load <8 x half>, ptr [[TMP0]], align 2
+; CHECK-PARTIAL-RED-EPI-NEXT: [[TMP1:%.*]] = getelementptr half, ptr [[B]], i64 [[INDEX]]
+; CHECK-PARTIAL-RED-EPI-NEXT: [[WIDE_LOAD1:%.*]] = load <8 x half>, ptr [[TMP1]], align 2
+; CHECK-PARTIAL-RED-EPI-NEXT: [[TMP2:%.*]] = getelementptr half, ptr [[C]], i64 [[INDEX]]
+; CHECK-PARTIAL-RED-EPI-NEXT: [[WIDE_LOAD2:%.*]] = load <8 x half>, ptr [[TMP2]], align 2
+; CHECK-PARTIAL-RED-EPI-NEXT: [[TMP3:%.*]] = fpext <8 x half> [[WIDE_LOAD1]] to <8 x float>
+; CHECK-PARTIAL-RED-EPI-NEXT: [[TMP4:%.*]] = fpext <8 x half> [[WIDE_LOAD]] to <8 x float>
+; CHECK-PARTIAL-RED-EPI-NEXT: [[TMP5:%.*]] = fmul reassoc nsz contract <8 x float> [[TMP3]], [[TMP4]]
+; CHECK-PARTIAL-RED-EPI-NEXT: [[PARTIAL_REDUCE:%.*]] = call reassoc nsz contract <4 x float> @llvm.vector.partial.reduce.fadd.v4f32.v8f32(<4 x float> [[VEC_PHI]], <8 x float> [[TMP5]])
+; CHECK-PARTIAL-RED-EPI-NEXT: [[TMP6:%.*]] = fpext <8 x half> [[WIDE_LOAD2]] to <8 x float>
+; CHECK-PARTIAL-RED-EPI-NEXT: [[TMP7:%.*]] = fmul reassoc nsz contract <8 x float> [[TMP6]], [[TMP4]]
+; CHECK-PARTIAL-RED-EPI-NEXT: [[PARTIAL_REDUCE3]] = call reassoc nsz contract <4 x float> @llvm.vector.partial.reduce.fadd.v4f32.v8f32(<4 x float> [[PARTIAL_REDUCE]], <8 x float> [[TMP7]])
+; CHECK-PARTIAL-RED-EPI-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
+; CHECK-PARTIAL-RED-EPI-NEXT: [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024
+; CHECK-PARTIAL-RED-EPI-NEXT: br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP9:![0-9]+]]
+; CHECK-PARTIAL-RED-EPI: [[MIDDLE_BLOCK]]:
+; CHECK-PARTIAL-RED-EPI-NEXT: [[TMP9:%.*]] = call reassoc nsz contract float @llvm.vector.reduce.fadd.v4f32(float 0.000000e+00, <4 x float> [[PARTIAL_REDUCE3]])
+; CHECK-PARTIAL-RED-EPI-NEXT: [[TMP10:%.*]] = fsub float 0.000000e+00, [[TMP9]]
+; CHECK-PARTIAL-RED-EPI-NEXT: br label %[[FOR_EXIT:.*]]
+; CHECK-PARTIAL-RED-EPI: [[FOR_EXIT]]:
+; CHECK-PARTIAL-RED-EPI-NEXT: ret float [[TMP10]]
+;
+entry:
+ br label %for.body
+
+for.body:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %for.body ]
+ %accum = phi float [ 0.0, %entry ], [ %sub, %for.body ]
+ %gep.a = getelementptr half, ptr %a, i64 %iv
+ %load.a = load half, ptr %gep.a, align 2
+ %ext.a = fpext half %load.a to float
+ %gep.b = getelementptr half, ptr %b, i64 %iv
+ %load.b = load half, ptr %gep.b, align 2
+ %ext.b = fpext half %load.b to float
+ %gep.c = getelementptr half, ptr %c, i64 %iv
+ %load.c = load half, ptr %gep.c, align 2
+ %ext.c = fpext half %load.c to float
+ %mul.ab = fmul nsz reassoc contract float %ext.b, %ext.a
+ %mul.ac = fmul nsz reassoc contract float %ext.c, %ext.a
+ %sub.ab = fsub nsz reassoc contract float %accum, %mul.ab
+ %sub = fsub nsz reassoc contract float %sub.ab, %mul.ac
+ %iv.next = add i64 %iv, 1
+ %exitcond.not = icmp eq i64 %iv.next, 1024
+ br i1 %exitcond.not, label %for.exit, label %for.body
+
+for.exit:
+ ret float %sub
+}
+
attributes #0 = { vscale_range(1,16) "target-features"="+sve" }
+attributes #1 = { vscale_range(1,16) "target-features"="+sve2" }
+
+!0 = distinct !{!0, !1}
+!1 = !{!"llvm.loop.vectorize.width", i32 16}
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/partial-reduce-sub.ll b/llvm/test/Transforms/LoopVectorize/AArch64/partial-reduce-sub.ll
index 2aea67f6e5499..575a53f0bf373 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/partial-reduce-sub.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/partial-reduce-sub.ll
@@ -269,9 +269,150 @@ exit:
ret i64 %sub
}
+define float @fdotp_fsub(ptr %a, ptr %b, ptr %c) #2 {
+; CHECK-INTERLEAVE1-LABEL: define float @fdotp_fsub(
+; CHECK-INTERLEAVE1-SAME: ptr [[A:%.*]], ptr [[B:%.*]], ptr [[C:%.*]]) #[[ATTR2:[0-9]+]] {
+; CHECK-INTERLEAVE1-NEXT: entry:
+; CHECK-INTERLEAVE1-NEXT: br label [[VECTOR_PH:%.*]]
+; CHECK-INTERLEAVE1: vector.ph:
+; CHECK-INTERLEAVE1-NEXT: br label [[VECTOR_BODY:%.*]]
+; CHECK-INTERLEAVE1: vector.body:
+; CHECK-INTERLEAVE1-NEXT: [[INDEX:%.*]] = phi i64 [ 0, [[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], [[VECTOR_BODY]] ]
+; CHECK-INTERLEAVE1-NEXT: [[VEC_PHI:%.*]] = phi <4 x float> [ splat (float -0.000000e+00), [[VECTOR_PH]] ], [ [[PARTIAL_REDUCE:%.*]], [[VECTOR_BODY]] ]
+; CHECK-INTERLEAVE1-NEXT: [[TMP0:%.*]] = getelementptr half, ptr [[A]], i64 [[INDEX]]
+; CHECK-INTERLEAVE1-NEXT: [[WIDE_LOAD:%.*]] = load <8 x half>, ptr [[TMP0]], align 2
+; CHECK-INTERLEAVE1-NEXT: [[TMP1:%.*]] = getelementptr half, ptr [[B]], i64 [[INDEX]]
+; CHECK-INTERLEAVE1-NEXT: [[WIDE_LOAD1:%.*]] = load <8 x half>, ptr [[TMP1]], align 2
+; CHECK-INTERLEAVE1-NEXT: [[TMP8:%.*]] = getelementptr half, ptr [[C]], i64 [[INDEX]]
+; CHECK-INTERLEAVE1-NEXT: [[WIDE_LOAD2:%.*]] = load <8 x half>, ptr [[TMP8]], align 2
+; CHECK-INTERLEAVE1-NEXT: [[TMP2:%.*]] = fpext <8 x half> [[WIDE_LOAD1]] to <8 x float>
+; CHECK-INTERLEAVE1-NEXT: [[TMP3:%.*]] = fpext <8 x half> [[WIDE_LOAD]] to <8 x float>
+; CHECK-INTERLEAVE1-NEXT: [[TMP4:%.*]] = fmul reassoc contract <8 x float> [[TMP2]], [[TMP3]]
+; CHECK-INTERLEAVE1-NEXT: [[PARTIAL_REDUCE1:%.*]] = call reassoc contract <4 x float> @llvm.vector.partial.reduce.fadd.v4f32.v8f32(<4 x float> [[VEC_PHI]], <8 x float> [[TMP4]])
+; CHECK-INTERLEAVE1-NEXT: [[TMP9:%.*]] = fpext <8 x half> [[WIDE_LOAD2]] to <8 x float>
+; CHECK-INTERLEAVE1-NEXT: [[TMP10:%.*]] = fmul reassoc contract <8 x float> [[TMP9]], [[TMP3]]
+; CHECK-INTERLEAVE1-NEXT: [[PARTIAL_REDUCE]] = call reassoc contract <4 x float> @llvm.vector.partial.reduce.fadd.v4f32.v8f32(<4 x float> [[PARTIAL_REDUCE1]], <8 x float> [[TMP10]])
+; CHECK-INTERLEAVE1-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
+; CHECK-INTERLEAVE1-NEXT: [[TMP5:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024
+; CHECK-INTERLEAVE1-NEXT: br i1 [[TMP5]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP6:![0-9]+]]
+; CHECK-INTERLEAVE1: middle.block:
+; CHECK-INTERLEAVE1-NEXT: [[TMP6:%.*]] = call reassoc contract float @llvm.vector.reduce.fadd.v4f32(float -0.000000e+00, <4 x float> [[PARTIAL_REDUCE]])
+; CHECK-INTERLEAVE1-NEXT: [[TMP7:%.*]] = fsub float 0.000000e+00, [[TMP6]]
+; CHECK-INTERLEAVE1-NEXT: br label [[FOR_EXIT:%.*]]
+; CHECK-INTERLEAVE1: for.exit:
+; CHECK-INTERLEAVE1-NEXT: ret float [[TMP7]]
+;
+; CHECK-INTERLEAVED-LABEL: define float @fdotp_fsub(
+; CHECK-INTERLEAVED-SAME: ptr [[A:%.*]], ptr [[B:%.*]], ptr [[C:%.*]]) #[[ATTR2:[0-9]+]] {
+; CHECK-INTERLEAVED-NEXT: entry:
+; CHECK-INTERLEAVED-NEXT: br label [[VECTOR_PH:%.*]]
+; CHECK-INTERLEAVED: vector.ph:
+; CHECK-INTERLEAVED-NEXT: br label [[VECTOR_BODY:%.*]]
+; CHECK-INTERLEAVED: vector.body:
+; CHECK-INTERLEAVED-NEXT: [[INDEX:%.*]] = phi i64 [ 0, [[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], [[VECTOR_BODY]] ]
+; CHECK-INTERLEAVED-NEXT: [[VEC_PHI:%.*]] = phi <4 x float> [ splat (float -0.000000e+00), [[VECTOR_PH]] ], [ [[PARTIAL_REDUCE:%.*]], [[VECTOR_BODY]] ]
+; CHECK-INTERLEAVED-NEXT: [[VEC_PHI1:%.*]] = phi <4 x float> [ splat (float -0.000000e+00), [[VECTOR_PH]] ], [ [[PARTIAL_REDUCE5:%.*]], [[VECTOR_BODY]] ]
+; CHECK-INTERLEAVED-NEXT: [[TMP0:%.*]] = getelementptr half, ptr [[A]], i64 [[INDEX]]
+; CHECK-INTERLEAVED-NEXT: [[TMP1:%.*]] = getelementptr half, ptr [[TMP0]], i64 8
+; CHECK-INTERLEAVED-NEXT: [[WIDE_LOAD:%.*]] = load <8 x half>, ptr [[TMP0]], align 2
+; CHECK-INTERLEAVED-NEXT: [[WIDE_LOAD2:%.*]] = load <8 x half>, ptr [[TMP1]], align 2
+; CHECK-INTERLEAVED-NEXT: [[TMP2:%.*]] = getelementptr half, ptr [[B]], i64 [[INDEX]]
+; CHECK-INTERLEAVED-NEXT: [[TMP3:%.*]] = getelementptr half, ptr [[TMP2]], i64 8
+; CHECK-INTERLEAVED-NEXT: [[WIDE_LOAD3:%.*]] = load <8 x half>, ptr [[TMP2]], align 2
+; CHECK-INTERLEAVED-NEXT: [[WIDE_LOAD4:%.*]] = load <8 x half>, ptr [[TMP3]], align 2
+; CHECK-INTERLEAVED-NEXT: [[TMP16:%.*]] = getelementptr half, ptr [[C]], i64 [[INDEX]]
+; CHECK-INTERLEAVED-NEXT: [[TMP17:%.*]] = getelementptr half, ptr [[TMP16]], i64 8
+; CHECK-INTERLEAVED-NEXT: [[WIDE_LOAD5:%.*]] = load <8 x half>, ptr [[TMP16]], align 2
+; CHECK-INTERLEAVED-NEXT: [[WIDE_LOAD6:%.*]] = load <8 x half>, ptr [[TMP17]], align 2
+; CHECK-INTERLEAVED-NEXT: [[TMP4:%.*]] = fpext <8 x half> [[WIDE_LOAD3]] to <8 x float>
+; CHECK-INTERLEAVED-NEXT: [[TMP5:%.*]] = fpext <8 x half> [[WIDE_LOAD]] to <8 x float>
+; CHECK-INTERLEAVED-NEXT: [[TMP6:%.*]] = fmul reassoc contract <8 x float> [[TMP4]], [[TMP5]]
+; CHECK-INTERLEAVED-NEXT: [[PARTIAL_REDUCE1:%.*]] = call reassoc contract <4 x float> @llvm.vector.partial.reduce.fadd.v4f32.v8f32(<4 x float> [[VEC_PHI]], <8 x float> [[TMP6]])
+; CHECK-INTERLEAVED-NEXT: [[TMP7:%.*]] = fpext <8 x half> [[WIDE_LOAD4]] to <8 x float>
+; CHECK-INTERLEAVED-NEXT: [[TMP8:%.*]] = fpext <8 x half> [[WIDE_LOAD2]] to <8 x float>
+; CHECK-INTERLEAVED-NEXT: [[TMP9:%.*]] = fmul reassoc contract <8 x float> [[TMP7]], [[TMP8]]
+; CHECK-INTERLEAVED-NEXT: [[PARTIAL_REDUCE7:%.*]] = call reassoc contract <4 x float> @llvm.vector.partial.reduce.fadd.v4f32.v8f32(<4 x float> [[VEC_PHI1]], <8 x float> [[TMP9]])
+; CHECK-INTERLEAVED-NEXT: [[TMP18:%.*]] = fpext <8 x half> [[WIDE_LOAD5]] to <8 x float>
+; CHECK-INTERLEAVED-NEXT: [[TMP13:%.*]] = fmul reassoc contract <8 x float> [[TMP18]], [[TMP5]]
+; CHECK-INTERLEAVED-NEXT: [[PARTIAL_REDUCE]] = call reassoc contract <4 x float> @llvm.vector.partial.reduce.fadd.v4f32.v8f32(<4 x float> [[PARTIAL_REDUCE1]], <8 x float> [[TMP13]])
+; CHECK-INTERLEAVED-NEXT: [[TMP14:%.*]] = fpext <8 x half> [[WIDE_LOAD6]] to <8 x float>
+; CHECK-INTERLEAVED-NEXT: [[TMP15:%.*]] = fmul reassoc contract <8 x float> [[TMP14]], [[TMP8]]
+; CHECK-INTERLEAVED-NEXT: [[PARTIAL_REDUCE5]] = call reassoc contract <4 x float> @llvm.vector.partial.reduce.fadd.v4f32.v8f32(<4 x float> [[PARTIAL_REDUCE7]], <8 x float> [[TMP15]])
+; CHECK-INTERLEAVED-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
+; CHECK-INTERLEAVED-NEXT: [[TMP10:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024
+; CHECK-INTERLEAVED-NEXT: br i1 [[TMP10]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP6:![0-9]+]]
+; CHECK-INTERLEAVED: middle.block:
+; CHECK-INTERLEAVED-NEXT: [[BIN_RDX:%.*]] = fadd reassoc contract <4 x float> [[PARTIAL_REDUCE5]], [[PARTIAL_REDUCE]]
+; CHECK-INTERLEAVED-NEXT: [[TMP11:%.*]] = call reassoc contract float @llvm.vector.reduce.fadd.v4f32(float -0.000000e+00, <4 x float> [[BIN_RDX]])
+; CHECK-INTERLEAVED-NEXT: [[TMP12:%.*]] = fsub float 0.000000e+00, [[TMP11]]
+; CHECK-INTERLEAVED-NEXT: br label [[FOR_EXIT:%.*]]
+; CHECK-INTERLEAVED: for.exit:
+; CHECK-INTERLEAVED-NEXT: ret float [[TMP12]]
+;
+; CHECK-MAXBW-LABEL: define float @fdotp_fsub(
+; CHECK-MAXBW-SAME: ptr [[A:%.*]], ptr [[B:%.*]], ptr [[C:%.*]]) #[[ATTR2:[0-9]+]] {
+; CHECK-MAXBW-NEXT: entry:
+; CHECK-MAXBW-NEXT: br label [[VECTOR_PH:%.*]]
+; CHECK-MAXBW: vector.ph:
+; CHECK-MAXBW-NEXT: br label [[VECTOR_BODY:%.*]]
+; CHECK-MAXBW: vector.body:
+; CHECK-MAXBW-NEXT: [[INDEX:%.*]] = phi i64 [ 0, [[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], [[VECTOR_BODY]] ]
+; CHECK-MAXBW-NEXT: [[VEC_PHI:%.*]] = phi <4 x float> [ splat (float -0.000000e+00), [[VECTOR_PH]] ], [ [[PARTIAL_REDUCE:%.*]], [[VECTOR_BODY]] ]
+; CHECK-MAXBW-NEXT: [[TMP0:%.*]] = getelementptr half, ptr [[A]], i64 [[INDEX]]
+; CHECK-MAXBW-NEXT: [[WIDE_LOAD:%.*]] = load <8 x half>, ptr [[TMP0]], align 2
+; CHECK-MAXBW-NEXT: [[TMP1:%.*]] = getelementptr half, ptr [[B]], i64 [[INDEX]]
+; CHECK-MAXBW-NEXT: [[WIDE_LOAD1:%.*]] = load <8 x half>, ptr [[TMP1]], align 2
+; CHECK-MAXBW-NEXT: [[TMP8:%.*]] = getelementptr half, ptr [[C]], i64 [[INDEX]]
+; CHECK-MAXBW-NEXT: [[WIDE_LOAD2:%.*]] = load <8 x half>, ptr [[TMP8]], align 2
+; CHECK-MAXBW-NEXT: [[TMP2:%.*]] = fpext <8 x half> [[WIDE_LOAD1]] to <8 x float>
+; CHECK-MAXBW-NEXT: [[TMP3:%.*]] = fpext <8 x half> [[WIDE_LOAD]] to <8 x float>
+; CHECK-MAXBW-NEXT: [[TMP4:%.*]] = fmul reassoc contract <8 x float> [[TMP2]], [[TMP3]]
+; CHECK-MAXBW-NEXT: [[PARTIAL_REDUCE1:%.*]] = call reassoc contract <4 x float> @llvm.vector.partial.reduce.fadd.v4f32.v8f32(<4 x float> [[VEC_PHI]], <8 x float> [[TMP4]])
+; CHECK-MAXBW-NEXT: [[TMP9:%.*]] = fpext <8 x half> [[WIDE_LOAD2]] to <8 x float>
+; CHECK-MAXBW-NEXT: [[TMP10:%.*]] = fmul reassoc contract <8 x float> [[TMP9]], [[TMP3]]
+; CHECK-MAXBW-NEXT: [[PARTIAL_REDUCE]] = call reassoc contract <4 x float> @llvm.vector.partial.reduce.fadd.v4f32.v8f32(<4 x float> [[PARTIAL_REDUCE1]], <8 x float> [[TMP10]])
+; CHECK-MAXBW-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
+; CHECK-MAXBW-NEXT: [[TMP5:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024
+; CHECK-MAXBW-NEXT: br i1 [[TMP5]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP6:![0-9]+]]
+; CHECK-MAXBW: middle.block:
+; CHECK-MAXBW-NEXT: [[TMP6:%.*]] = call reassoc contract float @llvm.vector.reduce.fadd.v4f32(float -0.000000e+00, <4 x float> [[PARTIAL_REDUCE]])
+; CHECK-MAXBW-NEXT: [[TMP7:%.*]] = fsub float 0.000000e+00, [[TMP6]]
+; CHECK-MAXBW-NEXT: br label [[FOR_EXIT:%.*]]
+; CHECK-MAXBW: for.exit:
+; CHECK-MAXBW-NEXT: ret float [[TMP7]]
+;
+entry:
+ br label %for.body
+
+for.body:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %for.body ]
+ %accum = phi float [ 0.0, %entry ], [ %sub, %for.body ]
+ %gep.a = getelementptr half, ptr %a, i64 %iv
+ %load.a = load half, ptr %gep.a, align 2
+ %ext.a = fpext half %load.a to float
+ %gep.b = getelementptr half, ptr %b, i64 %iv
+ %load.b = load half, ptr %gep.b, align 2
+ %ext.b = fpext half %load.b to float
+ %gep.c = getelementptr half, ptr %c, i64 %iv
+ %load.c = load half, ptr %gep.c, align 2
+ %ext.c = fpext half %load.c to float
+ %mul.ab = fmul reassoc contract float %ext.b, %ext.a
+ %mul.ac = fmul reassoc contract float %ext.c, %ext.a
+ %sub.ab = fsub reassoc contract float %accum, %mul.ab
+ %sub = fsub reassoc contract float %sub.ab, %mul.ac
+ %iv.next = add i64 %iv, 1
+ %exitcond.not = icmp eq i64 %iv.next, 1024
+ br i1 %exitcond.not, label %for.exit, label %for.body
+
+for.exit:
+ ret float %sub
+}
+
!7 = distinct !{!7, !8, !9, !10}
!8 = !{!"llvm.loop.mustprogress"}
!9 = !{!"llvm.loop.vectorize.predicate.enable", i1 true}
!10 = !{!"llvm.loop.vectorize.enable", i1 true}
+
attributes #0 = { vscale_range(1,16) "target-features"="+sve" }
attributes #1 = { "target-cpu"="apple-m1" }
+attributes #2 = { vscale_range(1,16) "target-features"="+sve2" }
More information about the llvm-commits
mailing list