[llvm] ae1e3eb - [NFCI][VPlan] Split initial mem-widening into a separate transformation (#182592)
via llvm-commits
llvm-commits at lists.llvm.org
Tue Apr 14 10:14:58 PDT 2026
Author: Andrei Elovikov
Date: 2026-04-14T17:14:52Z
New Revision: ae1e3eb379cde6b5d2e31104b4dbbbcf42298f3e
URL: https://github.com/llvm/llvm-project/commit/ae1e3eb379cde6b5d2e31104b4dbbbcf42298f3e
DIFF: https://github.com/llvm/llvm-project/commit/ae1e3eb379cde6b5d2e31104b4dbbbcf42298f3e.diff
LOG: [NFCI][VPlan] Split initial mem-widening into a separate transformation (#182592)
Preparation change before implementing stride-multiversioning as a
VPlan-based transformation. Might help
https://github.com/llvm/llvm-project/pull/147297/ as well.
Added:
llvm/test/Transforms/LoopVectorize/AArch64/ordered-reduction-with-invariant-stores.ll
Modified:
llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
llvm/lib/Transforms/Vectorize/VPRecipeBuilder.h
llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
llvm/lib/Transforms/Vectorize/VPlanTransforms.h
llvm/test/Transforms/LoopVectorize/AArch64/predication_costs.ll
llvm/test/Transforms/LoopVectorize/VPlan/vplan-print-after-all.ll
Removed:
################################################################################
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 98bb12e8e3670..b3a5c39ec2999 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -7284,6 +7284,7 @@ VPRecipeBase *VPRecipeBuilder::tryToWidenMemory(VPInstruction *VPI,
: GEPNoWrapFlags::none(),
VPI->getDebugLoc());
}
+ Builder.setInsertPoint(VPI);
Builder.insert(VectorPtr);
Ptr = VectorPtr;
}
@@ -7496,8 +7497,16 @@ VPWidenRecipe *VPRecipeBuilder::tryToWiden(VPInstruction *VPI) {
};
}
-VPHistogramRecipe *VPRecipeBuilder::tryToWidenHistogram(const HistogramInfo *HI,
- VPInstruction *VPI) {
+VPHistogramRecipe *VPRecipeBuilder::widenIfHistogram(VPInstruction *VPI) {
+ if (VPI->getOpcode() != Instruction::Store)
+ return nullptr;
+
+ auto HistInfo =
+ Legal->getHistogramInfo(cast<StoreInst>(VPI->getUnderlyingInstr()));
+ if (!HistInfo)
+ return nullptr;
+
+ const HistogramInfo *HI = *HistInfo;
// FIXME: Support other operations.
unsigned Opcode = HI->Update->getOpcode();
assert((Opcode == Instruction::Add || Opcode == Instruction::Sub) &&
@@ -7517,6 +7526,25 @@ VPHistogramRecipe *VPRecipeBuilder::tryToWidenHistogram(const HistogramInfo *HI,
return new VPHistogramRecipe(Opcode, HGramOps, VPI->getDebugLoc());
}
+bool VPRecipeBuilder::replaceWithFinalIfReductionStore(
+ VPInstruction *VPI, VPBuilder &FinalRedStoresBuilder) {
+ StoreInst *SI;
+ if ((SI = dyn_cast<StoreInst>(VPI->getUnderlyingInstr())) &&
+ Legal->isInvariantAddressOfReduction(SI->getPointerOperand())) {
+ // Only create recipe for the final invariant store of the reduction.
+ if (Legal->isInvariantStoreOfReduction(SI)) {
+ auto *Recipe = new VPReplicateRecipe(
+ SI, VPI->operandsWithoutMask(), true /* IsUniform */,
+ nullptr /*Mask*/, *VPI, *VPI, VPI->getDebugLoc());
+ FinalRedStoresBuilder.insert(Recipe);
+ }
+ VPI->eraseFromParent();
+ return true;
+ }
+
+ return false;
+}
+
VPReplicateRecipe *VPRecipeBuilder::handleReplication(VPInstruction *VPI,
VFRange &Range) {
auto *I = VPI->getUnderlyingInstr();
@@ -7603,13 +7631,9 @@ VPRecipeBuilder::tryToCreateWidenNonPhiRecipe(VPSingleDefRecipe *R,
return tryToWidenCall(VPI, Range);
Instruction *Instr = R->getUnderlyingInstr();
- if (VPI->getOpcode() == Instruction::Store)
- if (auto HistInfo = Legal->getHistogramInfo(cast<StoreInst>(Instr)))
- return tryToWidenHistogram(*HistInfo, VPI);
-
- if (VPI->getOpcode() == Instruction::Load ||
- VPI->getOpcode() == Instruction::Store)
- return tryToWidenMemory(VPI, Range);
+ assert(!is_contained({Instruction::Load, Instruction::Store},
+ VPI->getOpcode()) &&
+ "Should have been handled prior to this!");
if (!shouldWiden(Instr, Range))
return nullptr;
@@ -7797,8 +7821,6 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlanWithVPRecipes(
ReversePostOrderTraversal<VPBlockShallowTraversalWrapper<VPBlockBase *>> RPOT(
HeaderVPBB);
- VPBasicBlock::iterator MBIP = MiddleVPBB->getFirstNonPhi();
-
// Collect blocks that need predication for in-loop reduction recipes.
DenseSet<BasicBlock *> BlocksNeedingPredication;
for (BasicBlock *BB : OrigLoop->blocks())
@@ -7808,13 +7830,23 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlanWithVPRecipes(
VPlanTransforms::createInLoopReductionRecipes(*Plan, BlocksNeedingPredication,
Range.Start);
+ VPCostContext CostCtx(CM.TTI, *CM.TLI, *Plan, CM, CM.CostKind, CM.PSE,
+ OrigLoop);
+
+ RUN_VPLAN_PASS_NO_VERIFY(VPlanTransforms::makeMemOpWideningDecisions, *Plan,
+ Range, RecipeBuilder);
+
// Now process all other blocks and instructions.
for (VPBasicBlock *VPBB : VPBlockUtils::blocksOnly<VPBasicBlock>(RPOT)) {
// Convert input VPInstructions to widened recipes.
for (VPRecipeBase &R : make_early_inc_range(
make_range(VPBB->getFirstNonPhi(), VPBB->end()))) {
- // Skip recipes that do not need transforming.
- if (isa<VPWidenCanonicalIVRecipe, VPBlendRecipe, VPReductionRecipe>(&R))
+ // Skip recipes that do not need transforming or have already been
+ // transformed.
+ if (isa<VPWidenCanonicalIVRecipe, VPBlendRecipe, VPReductionRecipe,
+ VPReplicateRecipe, VPWidenLoadRecipe, VPWidenStoreRecipe,
+ VPVectorPointerRecipe, VPVectorEndPointerRecipe,
+ VPHistogramRecipe>(&R))
continue;
auto *VPI = cast<VPInstruction>(&R);
if (!VPI->getUnderlyingValue())
@@ -7826,23 +7858,6 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlanWithVPRecipes(
Instruction *Instr = cast<Instruction>(VPI->getUnderlyingValue());
Builder.setInsertPoint(VPI);
- // The stores with invariant address inside the loop will be deleted, and
- // in the exit block, a uniform store recipe will be created for the final
- // invariant store of the reduction.
- StoreInst *SI;
- if ((SI = dyn_cast<StoreInst>(Instr)) &&
- Legal->isInvariantAddressOfReduction(SI->getPointerOperand())) {
- // Only create recipe for the final invariant store of the reduction.
- if (Legal->isInvariantStoreOfReduction(SI)) {
- auto *Recipe = new VPReplicateRecipe(
- SI, VPI->operandsWithoutMask(), true /* IsUniform */,
- nullptr /*Mask*/, *VPI, *VPI, VPI->getDebugLoc());
- Recipe->insertBefore(*MiddleVPBB, MBIP);
- }
- R.eraseFromParent();
- continue;
- }
-
VPRecipeBase *Recipe =
RecipeBuilder.tryToCreateWidenNonPhiRecipe(VPI, Range);
if (!Recipe)
@@ -7909,8 +7924,6 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlanWithVPRecipes(
// TODO: Enable following transform when the EVL-version of extended-reduction
// and mulacc-reduction are implemented.
if (!CM.foldTailWithEVL()) {
- VPCostContext CostCtx(CM.TTI, *CM.TLI, *Plan, CM, CM.CostKind, CM.PSE,
- OrigLoop);
RUN_VPLAN_PASS(VPlanTransforms::createPartialReductions, *Plan, CostCtx,
Range);
RUN_VPLAN_PASS(VPlanTransforms::convertToAbstractRecipes, *Plan, CostCtx,
diff --git a/llvm/lib/Transforms/Vectorize/VPRecipeBuilder.h b/llvm/lib/Transforms/Vectorize/VPRecipeBuilder.h
index 64315df74dda5..e7a295cd5fcb7 100644
--- a/llvm/lib/Transforms/Vectorize/VPRecipeBuilder.h
+++ b/llvm/lib/Transforms/Vectorize/VPRecipeBuilder.h
@@ -52,11 +52,6 @@ class VPRecipeBuilder {
/// Range. The function should not be called for memory instructions or calls.
bool shouldWiden(Instruction *I, VFRange &Range) const;
- /// Check if the load or store instruction \p VPI should widened for \p
- /// Range.Start and potentially masked. Such instructions are handled by a
- /// recipe that takes an additional VPInstruction for the mask.
- VPRecipeBase *tryToWidenMemory(VPInstruction *VPI, VFRange &Range);
-
/// Optimize the special case where the operand of \p VPI is a constant
/// integer induction variable.
VPWidenIntOrFpInductionRecipe *
@@ -72,13 +67,6 @@ class VPRecipeBuilder {
/// cost-model indicates that widening should be performed.
VPWidenRecipe *tryToWiden(VPInstruction *VPI);
- /// Makes Histogram count operations safe for vectorization, by emitting a
- /// llvm.experimental.vector.histogram.add intrinsic in place of the
- /// Load + Add|Sub + Store operations that perform the histogram in the
- /// original scalar loop.
- VPHistogramRecipe *tryToWidenHistogram(const HistogramInfo *HI,
- VPInstruction *VPI);
-
public:
VPRecipeBuilder(VPlan &Plan, const TargetLibraryInfo *TLI,
LoopVectorizationLegality *Legal,
@@ -90,6 +78,26 @@ class VPRecipeBuilder {
VPRecipeBase *tryToCreateWidenNonPhiRecipe(VPSingleDefRecipe *R,
VFRange &Range);
+ /// Check if the load or store instruction \p VPI should widened for \p
+ /// Range.Start and potentially masked. Such instructions are handled by a
+ /// recipe that takes an additional VPInstruction for the mask.
+ VPRecipeBase *tryToWidenMemory(VPInstruction *VPI, VFRange &Range);
+
+ /// If \p VPI represents a histogram operation (as determined by
+ /// LoopVectorizationLegality) make that safe for vectorization, by emitting a
+ /// llvm.experimental.vector.histogram.add intrinsic in place of the Load +
+ /// Add|Sub + Store operations that perform the histogram in the original
+ /// scalar loop.
+ VPHistogramRecipe *widenIfHistogram(VPInstruction *VPI);
+
+ /// If \p VPI is a store of a reduction into an invariant address, delete it.
+ /// If it is the final store of a reduction result, a uniform store recipe
+ /// will be created for it in the middle block. Returns `true` if replacement
+ /// took place. The order of stores must be preserved, hence \p
+ /// FinalRedStoresBuidler.
+ bool replaceWithFinalIfReductionStore(VPInstruction *VPI,
+ VPBuilder &FinalRedStoresBuilder);
+
/// Set the recipe created for given ingredient.
void setRecipe(Instruction *I, VPRecipeBase *R) {
assert(!Ingredient2Recipe.contains(I) &&
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index b2e8b6a85a35a..738f6bc92f541 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -6487,3 +6487,57 @@ void VPlanTransforms::createPartialReductions(VPlan &Plan,
for (const VPPartialReductionChain &Chain : Chains)
transformToPartialReduction(Chain, CostCtx.Types, Plan, Phi);
}
+
+void VPlanTransforms::makeMemOpWideningDecisions(
+ VPlan &Plan, VFRange &Range, VPRecipeBuilder &RecipeBuilder) {
+ // Collect all loads/stores first. We will start with ones having simpler
+ // decisions followed by more complex ones that are potentially
+ // guided/dependent on the simpler ones.
+ SmallVector<VPInstruction *> MemOps;
+ for (VPBasicBlock *VPBB :
+ VPBlockUtils::blocksOnly<VPBasicBlock>(vp_depth_first_shallow(
+ Plan.getVectorLoopRegion()->getEntryBasicBlock()))) {
+ for (VPRecipeBase &R : *VPBB) {
+ auto *VPI = dyn_cast<VPInstruction>(&R);
+ if (VPI && VPI->getUnderlyingValue() &&
+ is_contained({Instruction::Load, Instruction::Store},
+ VPI->getOpcode()))
+ MemOps.push_back(VPI);
+ }
+ }
+
+ VPBasicBlock *MiddleVPBB = Plan.getMiddleBlock();
+ VPBuilder FinalRedStoresBuilder(MiddleVPBB, MiddleVPBB->getFirstNonPhi());
+
+ for (VPInstruction *VPI : MemOps) {
+ auto ReplaceWith = [&](VPRecipeBase *New) {
+ RecipeBuilder.setRecipe(cast<Instruction>(VPI->getUnderlyingValue()),
+ New);
+ New->insertBefore(VPI);
+ if (VPI->getOpcode() == Instruction::Load)
+ VPI->replaceAllUsesWith(New->getVPSingleValue());
+ VPI->eraseFromParent();
+ };
+
+ // Note: we must do that for scalar VPlan as well.
+ if (RecipeBuilder.replaceWithFinalIfReductionStore(VPI,
+ FinalRedStoresBuilder))
+ continue;
+
+ // Filter out scalar VPlan for the remaining memory operations.
+ if (LoopVectorizationPlanner::getDecisionAndClampRange(
+ [](ElementCount VF) { return VF.isScalar(); }, Range))
+ continue;
+
+ if (VPHistogramRecipe *Histogram = RecipeBuilder.widenIfHistogram(VPI)) {
+ ReplaceWith(Histogram);
+ continue;
+ }
+
+ VPRecipeBase *Recipe = RecipeBuilder.tryToWidenMemory(VPI, Range);
+ if (!Recipe)
+ Recipe = RecipeBuilder.handleReplication(VPI, Range);
+
+ ReplaceWith(Recipe);
+ }
+}
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.h b/llvm/lib/Transforms/Vectorize/VPlanTransforms.h
index 6312b823e5d33..bebf7ff1e9262 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.h
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.h
@@ -534,6 +534,11 @@ struct VPlanTransforms {
/// are only valid for a subset of VFs in Range, Range.End is updated.
static void createPartialReductions(VPlan &Plan, VPCostContext &CostCtx,
VFRange &Range);
+
+ /// Convert load/store VPInstructions in \p Plan into widened or replicate
+ /// recipes. Non load/store input instructions are left unchanged.
+ static void makeMemOpWideningDecisions(VPlan &Plan, VFRange &Range,
+ VPRecipeBuilder &RecipeBuilder);
};
} // namespace llvm
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/ordered-reduction-with-invariant-stores.ll b/llvm/test/Transforms/LoopVectorize/AArch64/ordered-reduction-with-invariant-stores.ll
new file mode 100644
index 0000000000000..1179e83a50392
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/ordered-reduction-with-invariant-stores.ll
@@ -0,0 +1,107 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
+; RUN: opt -S -p loop-vectorize -mtriple=arm64 < %s | FileCheck %s
+
+; Crashed during refactoring if reduction store is not sunk out of the loop.
+define void @ordered_reduction(ptr %dst, float %a, float %b) {
+; CHECK-LABEL: define void @ordered_reduction(
+; CHECK-SAME: ptr [[DST:%.*]], float [[A:%.*]], float [[B:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: br label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <2 x float> poison, float [[B]], i64 0
+; CHECK-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <2 x float> [[BROADCAST_SPLATINSERT]], <2 x float> poison, <2 x i32> zeroinitializer
+; CHECK-NEXT: [[BROADCAST_SPLATINSERT1:%.*]] = insertelement <2 x float> poison, float [[A]], i64 0
+; CHECK-NEXT: [[BROADCAST_SPLAT2:%.*]] = shufflevector <2 x float> [[BROADCAST_SPLATINSERT1]], <2 x float> poison, <2 x i32> zeroinitializer
+; CHECK-NEXT: [[TMP0:%.*]] = fmul <2 x float> [[BROADCAST_SPLAT2]], [[BROADCAST_SPLAT]]
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[VEC_PHI:%.*]] = phi float [ 0.000000e+00, %[[VECTOR_PH]] ], [ [[TMP2:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP1:%.*]] = call float @llvm.vector.reduce.fadd.v2f32(float [[VEC_PHI]], <2 x float> [[TMP0]])
+; CHECK-NEXT: [[TMP2]] = call float @llvm.vector.reduce.fadd.v2f32(float [[TMP1]], <2 x float> [[TMP0]])
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT: [[TMP3:%.*]] = icmp eq i64 [[INDEX_NEXT]], 100
+; CHECK-NEXT: br i1 [[TMP3]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: store float [[TMP2]], ptr [[DST]], align 4
+; CHECK-NEXT: br label %[[SCALAR_PH:.*]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 100, %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[SUM:%.*]] = phi float [ [[TMP2]], %[[SCALAR_PH]] ], [ [[MULADD:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[MULADD]] = tail call float @llvm.fmuladd.f32(float [[A]], float [[B]], float [[SUM]])
+; CHECK-NEXT: store float [[MULADD]], ptr [[DST]], align 4
+; CHECK-NEXT: [[IV_NEXT]] = add i64 [[IV]], 1
+; CHECK-NEXT: [[DONE:%.*]] = icmp eq i64 [[IV]], 100
+; CHECK-NEXT: br i1 [[DONE]], label %[[EXIT:.*]], label %[[LOOP]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %sum = phi float [ 0.000000e+00, %entry ], [ %muladd, %loop ]
+ %muladd = tail call float @llvm.fmuladd.f32(float %a, float %b, float %sum)
+ store float %muladd, ptr %dst, align 4
+ %iv.next = add i64 %iv, 1
+ %done = icmp eq i64 %iv, 100
+ br i1 %done, label %exit, label %loop
+
+exit:
+ ret void
+}
+
+; Same as above but with an additional load (used to catch an error where "early
+; return" was used instead of "early continue" due to copy-paste error during
+; refactoring).
+define void @ordered_reduction2(ptr noalias %dst, ptr noalias %src) {
+; CHECK-LABEL: define void @ordered_reduction2(
+; CHECK-SAME: ptr noalias [[DST:%.*]], ptr noalias [[SRC:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: br label %[[VECTOR_MEMCHECK:.*]]
+; CHECK: [[VECTOR_MEMCHECK]]:
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_MEMCHECK]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[VEC_PHI:%.*]] = phi float [ 0.000000e+00, %[[VECTOR_MEMCHECK]] ], [ [[TMP1:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP0:%.*]] = call float @llvm.vector.reduce.fadd.v2f32(float [[VEC_PHI]], <2 x float> splat (float 6.000000e+00))
+; CHECK-NEXT: [[TMP1]] = call float @llvm.vector.reduce.fadd.v2f32(float [[TMP0]], <2 x float> splat (float 6.000000e+00))
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT: [[TMP2:%.*]] = icmp eq i64 [[INDEX_NEXT]], 96
+; CHECK-NEXT: br i1 [[TMP2]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: store float [[TMP1]], ptr [[DST]], align 4
+; CHECK-NEXT: br label %[[SCALAR_PH:.*]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 97, %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[TMP3:%.*]] = phi float [ [[TMP1]], %[[SCALAR_PH]] ], [ [[TMP5:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[TMP4:%.*]] = load float, ptr [[SRC]], align 4
+; CHECK-NEXT: [[TMP5]] = tail call float @llvm.fmuladd.f32(float 2.000000e+00, float 3.000000e+00, float [[TMP3]])
+; CHECK-NEXT: store float [[TMP5]], ptr [[DST]], align 4
+; CHECK-NEXT: [[IV_NEXT]] = add i64 [[IV]], 1
+; CHECK-NEXT: [[EXITCOND_NOT:%.*]] = icmp eq i64 [[IV_NEXT]], 100
+; CHECK-NEXT: br i1 [[EXITCOND_NOT]], label %[[EXIT:.*]], label %[[LOOP]], !llvm.loop [[LOOP5:![0-9]+]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 1, %entry ], [ %iv.next, %loop ]
+ %0 = phi float [ 0.000000e+00, %entry ], [ %2, %loop ]
+ %1 = load float, ptr %src, align 4
+ %2 = tail call float @llvm.fmuladd.f32(float 2.000000e+00, float 3.000000e+00, float %0)
+ store float %2, ptr %dst, align 4
+ %iv.next = add i64 %iv, 1
+ %exitcond.not = icmp eq i64 %iv.next, 100
+ br i1 %exitcond.not, label %exit, label %loop
+
+exit:
+ ret void
+}
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/predication_costs.ll b/llvm/test/Transforms/LoopVectorize/AArch64/predication_costs.ll
index b9b91be9b7a65..fdb22beb19695 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/predication_costs.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/predication_costs.ll
@@ -179,8 +179,8 @@ for.end:
; Cost of store:
; store(4) / 2 = 2
;
-; CHECK: Scalarizing: %tmp2 = add nsw i32 %tmp1, %x
; CHECK: Scalarizing and predicating: store i32 %tmp2, ptr %tmp0, align 4
+; CHECK: Scalarizing: %tmp2 = add nsw i32 %tmp1, %x
; CHECK: Cost of 2 for VF 2: profitable to scalarize store i32 %tmp2, ptr %tmp0, align 4
; CHECK: Cost of 3 for VF 2: profitable to scalarize %tmp2 = add nsw i32 %tmp1, %x
;
@@ -229,10 +229,11 @@ for.end:
; store(4) / 2 = 2
;
; CHECK-NOT: Scalarizing: %tmp2 = add i32 %tmp1, %x
+; CHECK: Scalarizing and predicating: store i32 %tmp5, ptr %tmp0, align 4
+; CHECK-NOT: Scalarizing: %tmp2 = add i32 %tmp1, %x
; CHECK: Scalarizing and predicating: %tmp3 = sdiv i32 %tmp1, %tmp2
; CHECK: Scalarizing and predicating: %tmp4 = udiv i32 %tmp3, %tmp2
; CHECK: Scalarizing: %tmp5 = sub i32 %tmp4, %x
-; CHECK: Scalarizing and predicating: store i32 %tmp5, ptr %tmp0, align 4
; CHECK: Cost of 2 for VF 2: profitable to scalarize store i32 %tmp5, ptr %tmp0, align 4
; CHECK: Cost of 3 for VF 2: profitable to scalarize %tmp5 = sub i32 %tmp4, %x
; CHECK: Cost of 1 for VF 2: WIDEN ir<%tmp2> = add ir<%tmp1>, ir<%x>
diff --git a/llvm/test/Transforms/LoopVectorize/VPlan/vplan-print-after-all.ll b/llvm/test/Transforms/LoopVectorize/VPlan/vplan-print-after-all.ll
index a698636aa2349..ca13a229e34a7 100644
--- a/llvm/test/Transforms/LoopVectorize/VPlan/vplan-print-after-all.ll
+++ b/llvm/test/Transforms/LoopVectorize/VPlan/vplan-print-after-all.ll
@@ -6,6 +6,7 @@
; CHECK: VPlan for loop in 'foo' after printAfterInitialConstruction
; CHECK: VPlan for loop in 'foo' after VPlanTransforms::createLoopRegions
; CHECK: VPlan for loop in 'foo' after VPlanTransforms::introduceMasksAndLinearize
+; CHECK: VPlan for loop in 'foo' after VPlanTransforms::makeMemOpWideningDecisions
; CHECK: VPlan for loop in 'foo' after VPlanTransforms::clearReductionWrapFlags
; CHECK: VPlan for loop in 'foo' after VPlanTransforms::optimizeFindIVReductions
; CHECK: VPlan for loop in 'foo' after VPlanTransforms::handleMultiUseReductions
More information about the llvm-commits
mailing list