[llvm] [SLP] Add -slp-use-vplan-codegen to emit vector code via VPlan. (POC) (PR #226843)
via llvm-commits
llvm-commits at lists.llvm.org
Sun Sep 27 15:11:23 PDT 2026
llvmorg-github-actions[bot] wrote:
<!--LLVM PR SUMMARY COMMENT-->
@llvm/pr-subscribers-backend-amdgpu
@llvm/pr-subscribers-llvm-analysis
Author: Florian Hahn (fhahn)
<details>
<summary>Changes</summary>
Add an off-by-default option to generate the vector code for an SLP tree
through VPlan instead of the hand-rolled emitter in vectorizeTree().
This patch is intended as proof-of-concept, showing how VPlan could be
used as codegen backend for SLP, following a similar path like the
initial VPlan bring-up in LoopVectorize.
Moving the codegen to VPlan could allow moving some SLP functionality to
be VPlan based (e.g. simplifications, codegen-optimizations), helping
modularizing the code.
I have bigger prototype, which can handle about 80% of cases in the
VPlan path on a large test set of workloads. Compile-time impact looks
neutral, so I would not expect that to become a blocker.
Aided by Opus 5
---
Patch is 92.93 KiB, truncated to 20.00 KiB below, full version: https://github.com/llvm/llvm-project/pull/226843.diff
61 Files Affected:
- (modified) llvm/include/llvm/Analysis/VectorUtils.h (+9)
- (modified) llvm/lib/Analysis/VectorUtils.cpp (+19-10)
- (modified) llvm/lib/Transforms/Vectorize/CMakeLists.txt (+1)
- (added) llvm/lib/Transforms/Vectorize/SLPVPlanCodegen.cpp (+83)
- (added) llvm/lib/Transforms/Vectorize/SLPVPlanCodegen.h (+51)
- (modified) llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp (+166-5)
- (modified) llvm/lib/Transforms/Vectorize/VPlan.h (+12-2)
- (modified) llvm/test/CodeGen/WebAssembly/simd-min-vec-reg-32.ll (+1)
- (modified) llvm/test/Transforms/PhaseOrdering/AArch64/interleave_vec.ll (+1)
- (modified) llvm/test/Transforms/SLPVectorizer/AArch64/32-bit.ll (+1)
- (modified) llvm/test/Transforms/SLPVectorizer/AArch64/masked-div-rem-non-pow2.ll (+1)
- (modified) llvm/test/Transforms/SLPVectorizer/AArch64/memory-runtime-checks.ll (+1)
- (modified) llvm/test/Transforms/SLPVectorizer/AArch64/nontemporal.ll (+1)
- (modified) llvm/test/Transforms/SLPVectorizer/AArch64/slp-frem.ll (+1)
- (modified) llvm/test/Transforms/SLPVectorizer/AArch64/spillcost-call-between-operands.ll (+1)
- (modified) llvm/test/Transforms/SLPVectorizer/AArch64/spillcost-di.ll (+1)
- (modified) llvm/test/Transforms/SLPVectorizer/AArch64/store-ptr.ll (+1)
- (modified) llvm/test/Transforms/SLPVectorizer/AArch64/vec3-base.ll (+2)
- (added) llvm/test/Transforms/SLPVectorizer/AArch64/vplan-codegen-load-sinking.ll (+133)
- (modified) llvm/test/Transforms/SLPVectorizer/AMDGPU/elementwise-fma-operand1.ll (+3)
- (modified) llvm/test/Transforms/SLPVectorizer/AMDGPU/fma-operand-contract-selection.ll (+2)
- (modified) llvm/test/Transforms/SLPVectorizer/AMDGPU/invariant-load-no-alias-store.ll (+1)
- (modified) llvm/test/Transforms/SLPVectorizer/AMDGPU/packed-math.ll (+2)
- (modified) llvm/test/Transforms/SLPVectorizer/AMDGPU/slp-v2f16.ll (+4)
- (modified) llvm/test/Transforms/SLPVectorizer/RISCV/basic-strided-loads.ll (+1)
- (modified) llvm/test/Transforms/SLPVectorizer/RISCV/basic-strided-stores.ll (+1)
- (modified) llvm/test/Transforms/SLPVectorizer/RISCV/external.ll (+1)
- (modified) llvm/test/Transforms/SLPVectorizer/RISCV/floating-point.ll (+5)
- (modified) llvm/test/Transforms/SLPVectorizer/RISCV/gep.ll (+2)
- (modified) llvm/test/Transforms/SLPVectorizer/RISCV/load-binop-store.ll (+3)
- (modified) llvm/test/Transforms/SLPVectorizer/RISCV/load-store.ll (+3)
- (modified) llvm/test/Transforms/SLPVectorizer/RISCV/rotated-strided-loads.ll (+1)
- (modified) llvm/test/Transforms/SLPVectorizer/RISCV/runtime-strided-stores.ll (+1)
- (modified) llvm/test/Transforms/SLPVectorizer/RISCV/test-delete-tree.ll (+1)
- (modified) llvm/test/Transforms/SLPVectorizer/RISCV/vec3-base.ll (+2)
- (modified) llvm/test/Transforms/SLPVectorizer/VE/disable_slp.ll (+1)
- (modified) llvm/test/Transforms/SLPVectorizer/X86/align.ll (+1)
- (modified) llvm/test/Transforms/SLPVectorizer/X86/arith-add-load.ll (+4)
- (modified) llvm/test/Transforms/SLPVectorizer/X86/arith-add.ll (+13)
- (modified) llvm/test/Transforms/SLPVectorizer/X86/arith-mul-load.ll (+4)
- (modified) llvm/test/Transforms/SLPVectorizer/X86/arith-mul.ll (+13)
- (modified) llvm/test/Transforms/SLPVectorizer/X86/arith-sub.ll (+13)
- (modified) llvm/test/Transforms/SLPVectorizer/X86/continue_vectorizing.ll (+1)
- (modified) llvm/test/Transforms/SLPVectorizer/X86/control-dependence.ll (+1)
- (modified) llvm/test/Transforms/SLPVectorizer/X86/fmuladd-copyable-add-part.ll (+2)
- (modified) llvm/test/Transforms/SLPVectorizer/X86/fsub-fmul-rhs-combine.ll (+1)
- (modified) llvm/test/Transforms/SLPVectorizer/X86/lookahead.ll (+2)
- (modified) llvm/test/Transforms/SLPVectorizer/X86/metadata.ll (+1)
- (modified) llvm/test/Transforms/SLPVectorizer/X86/operandorder.ll (+2)
- (modified) llvm/test/Transforms/SLPVectorizer/X86/opt.ll (+1)
- (modified) llvm/test/Transforms/SLPVectorizer/X86/reassociate-ops.ll (+1)
- (modified) llvm/test/Transforms/SLPVectorizer/X86/runtime-alias-checks.ll (+1)
- (modified) llvm/test/Transforms/SLPVectorizer/X86/schedule_budget.ll (+2)
- (modified) llvm/test/Transforms/SLPVectorizer/X86/schedule_budget_debug_info.ll (+2)
- (modified) llvm/test/Transforms/SLPVectorizer/X86/shift-ashr.ll (+11)
- (modified) llvm/test/Transforms/SLPVectorizer/X86/shift-lshr.ll (+11)
- (modified) llvm/test/Transforms/SLPVectorizer/X86/shift-shl.ll (+11)
- (modified) llvm/test/Transforms/SLPVectorizer/X86/simplebb.ll (+1)
- (modified) llvm/test/Transforms/SLPVectorizer/X86/tiny-tree.ll (+1)
- (modified) llvm/test/Transforms/SLPVectorizer/consecutive-access.ll (+2)
- (modified) llvm/test/Transforms/SLPVectorizer/int_sideeffect.ll (+1)
``````````diff
diff --git a/llvm/include/llvm/Analysis/VectorUtils.h b/llvm/include/llvm/Analysis/VectorUtils.h
index b177d9eec21896..d47bc503fe898c 100644
--- a/llvm/include/llvm/Analysis/VectorUtils.h
+++ b/llvm/include/llvm/Analysis/VectorUtils.h
@@ -378,6 +378,15 @@ LLVM_ABI void getMetadataToPropagate(
Instruction *Inst,
SmallVectorImpl<std::pair<unsigned, MDNode *>> &Metadata);
+/// Add the metadata that can be preserved when combining all of \p VL into a
+/// single instruction to \p Metadata. \p Repr is a representative for the
+/// combined instruction, used to determine which metadata kinds it can carry.
+/// Callers that already have the combined instruction should use
+/// propagateMetadata() instead.
+LLVM_ABI void getMetadataToPropagate(
+ const Instruction *Repr, ArrayRef<Value *> VL,
+ SmallVectorImpl<std::pair<unsigned, MDNode *>> &Metadata);
+
/// Specifically, let Kinds = [MD_tbaa, MD_alias_scope, MD_noalias, MD_fpmath,
/// MD_nontemporal, MD_access_group, MD_mmra].
/// For K in Kinds, we get the MDNode for K from each of the
diff --git a/llvm/lib/Analysis/VectorUtils.cpp b/llvm/lib/Analysis/VectorUtils.cpp
index 1c105ebb772b35..8b5a0cbf0220ea 100644
--- a/llvm/lib/Analysis/VectorUtils.cpp
+++ b/llvm/lib/Analysis/VectorUtils.cpp
@@ -1073,17 +1073,21 @@ void llvm::getMetadataToPropagate(
}
}
-/// \returns \p I after propagating metadata from \p VL.
-Instruction *llvm::propagateMetadata(Instruction *Inst, ArrayRef<Value *> VL) {
+/// Add metadata from all of \p VL to \p Metadata, if it can be preserved after
+/// combining them into \p Repr.
+void llvm::getMetadataToPropagate(
+ const Instruction *Repr, ArrayRef<Value *> VL,
+ SmallVectorImpl<std::pair<unsigned, MDNode *>> &Metadata) {
if (VL.empty())
- return Inst;
- SmallVector<std::pair<unsigned, MDNode *>> Metadata;
+ return;
getMetadataToPropagate(cast<Instruction>(VL[0]), Metadata);
for (auto &[Kind, MD] : Metadata) {
- // Skip MMRA metadata if the instruction cannot have it.
- if (Kind == LLVMContext::MD_mmra && !canInstructionHaveMMRAs(*Inst))
+ // Drop MMRA metadata if the combined instruction cannot have it.
+ if (Kind == LLVMContext::MD_mmra && !canInstructionHaveMMRAs(*Repr)) {
+ MD = nullptr;
continue;
+ }
for (int J = 1, E = VL.size(); MD && J != E; ++J) {
const Instruction *IJ = cast<Instruction>(VL[J]);
@@ -1091,7 +1095,7 @@ Instruction *llvm::propagateMetadata(Instruction *Inst, ArrayRef<Value *> VL) {
switch (Kind) {
case LLVMContext::MD_mmra: {
- MD = MMRAMetadata::combine(Inst->getContext(), MD, IMD);
+ MD = MMRAMetadata::combine(Repr->getContext(), MD, IMD);
break;
}
case LLVMContext::MD_tbaa:
@@ -1109,16 +1113,21 @@ Instruction *llvm::propagateMetadata(Instruction *Inst, ArrayRef<Value *> VL) {
MD = MDNode::intersect(MD, IMD);
break;
case LLVMContext::MD_access_group:
- MD = intersectAccessGroups(Inst, IJ);
+ MD = intersectAccessGroups(Repr, IJ);
break;
default:
llvm_unreachable("unhandled metadata");
}
}
-
- Inst->setMetadata(Kind, MD);
}
+}
+/// \returns \p Inst after propagating metadata from \p VL.
+Instruction *llvm::propagateMetadata(Instruction *Inst, ArrayRef<Value *> VL) {
+ SmallVector<std::pair<unsigned, MDNode *>> Metadata;
+ getMetadataToPropagate(Inst, VL, Metadata);
+ for (auto &[Kind, MD] : Metadata)
+ Inst->setMetadata(Kind, MD);
return Inst;
}
diff --git a/llvm/lib/Transforms/Vectorize/CMakeLists.txt b/llvm/lib/Transforms/Vectorize/CMakeLists.txt
index 90732112808861..80f7bb81b8a45c 100644
--- a/llvm/lib/Transforms/Vectorize/CMakeLists.txt
+++ b/llvm/lib/Transforms/Vectorize/CMakeLists.txt
@@ -30,6 +30,7 @@ add_llvm_component_library(LLVMVectorize
SLPVectorizer/SLPTypeUtils.cpp
SLPVectorizer/SLPUtils.cpp
SLPVectorizer.cpp
+ SLPVPlanCodegen.cpp
Vectorize.cpp
VectorCombine.cpp
VPlan.cpp
diff --git a/llvm/lib/Transforms/Vectorize/SLPVPlanCodegen.cpp b/llvm/lib/Transforms/Vectorize/SLPVPlanCodegen.cpp
new file mode 100644
index 00000000000000..1f7f5c30cdeae6
--- /dev/null
+++ b/llvm/lib/Transforms/Vectorize/SLPVPlanCodegen.cpp
@@ -0,0 +1,83 @@
+//===- SLPVPlanCodegen.cpp - VPlan-based codegen for SLP ------------------===//
+//
+// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
+// See https://llvm.org/LICENSE.txt for license information.
+// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
+//
+//===----------------------------------------------------------------------===//
+
+#include "SLPVPlanCodegen.h"
+#include "LoopVectorizationPlanner.h"
+#include "VPlan.h"
+#include "VPlanHelpers.h"
+#include "llvm/ADT/STLExtras.h"
+#include "llvm/IR/IRBuilder.h"
+#include "llvm/IR/Instructions.h"
+
+using namespace llvm;
+
+bool slpvectorizer::isLiveInOperand(unsigned Opcode, unsigned J) {
+ // Loads and stores take their pointer operand as a live-in.
+ return (Opcode == Instruction::Load &&
+ J == LoadInst::getPointerOperandIndex()) ||
+ (Opcode == Instruction::Store &&
+ J == StoreInst::getPointerOperandIndex());
+}
+
+bool slpvectorizer::isSupportedVPlanCodegenOpcode(unsigned Opcode) {
+ if (Instruction::isBinaryOp(Opcode))
+ return true;
+ switch (Opcode) {
+ case Instruction::FNeg:
+ case Instruction::Load:
+ case Instruction::Store:
+ return true;
+ default:
+ return false;
+ }
+}
+
+/// Returns the IR flags common to all of \p Scalars, seeded from \p MainOp.
+static VPIRFlags computeIntersectedFlags(Instruction *MainOp,
+ ArrayRef<Value *> Scalars) {
+ VPIRFlags Flags(*MainOp);
+ for (Value *V : Scalars) {
+ // Drop all flags if a lane is not an instruction, e.g. poison.
+ auto *I = dyn_cast<Instruction>(V);
+ if (!I)
+ return VPIRFlags();
+ Flags.intersectFlags(VPIRFlags(*I));
+ }
+ return Flags;
+}
+
+VPValue *slpvectorizer::createRecipeForBundle(VPlan &Plan, VPBuilder &VPB,
+ Instruction *MainOp,
+ ArrayRef<Value *> Scalars,
+ ArrayRef<VPValue *> Ops) {
+ VPIRFlags Flags = computeIntersectedFlags(MainOp, Scalars);
+ // Only instructions carry metadata, so other lanes, e.g. poison, are skipped.
+ VPIRMetadata Metadata(
+ *MainOp, to_vector(make_filter_range(Scalars, IsaPred<Instruction>)));
+ DebugLoc DL = MainOp->getDebugLoc();
+
+ if (auto *LI = dyn_cast<LoadInst>(MainOp))
+ return VPB.createWidenLoad(
+ *LI, Plan.getOrAddLiveIn(LI->getPointerOperand()),
+ /*Mask=*/nullptr, /*Consecutive=*/true, Metadata, DL);
+ if (auto *SI = dyn_cast<StoreInst>(MainOp)) {
+ VPB.createWidenStore(*SI, Plan.getOrAddLiveIn(SI->getPointerOperand()),
+ Ops[0], /*Mask=*/nullptr, /*Consecutive=*/true,
+ Metadata, DL);
+ // Stores do not define a value.
+ return nullptr;
+ }
+ return VPB.insert(new VPWidenRecipe(*MainOp, Ops, Flags, Metadata, DL));
+}
+
+void slpvectorizer::executeSLPPlan(VPlan &Plan, VPTransformState &State) {
+ for (VPRecipeBase &R : *Plan.getEntry()) {
+ State.Builder.SetCurrentDebugLocation(R.getDebugLoc());
+ R.execute(State);
+ }
+}
diff --git a/llvm/lib/Transforms/Vectorize/SLPVPlanCodegen.h b/llvm/lib/Transforms/Vectorize/SLPVPlanCodegen.h
new file mode 100644
index 00000000000000..39807f14a7ede8
--- /dev/null
+++ b/llvm/lib/Transforms/Vectorize/SLPVPlanCodegen.h
@@ -0,0 +1,51 @@
+//===- SLPVPlanCodegen.h - VPlan-based codegen for SLP ----------*- C++ -*-===//
+//
+// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
+// See https://llvm.org/LICENSE.txt for license information.
+// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
+//
+//===----------------------------------------------------------------------===//
+//
+// Helpers building and executing the VPlan for an SLP tree that do not depend
+// on BoUpSLP.
+//
+//===----------------------------------------------------------------------===//
+
+#ifndef LLVM_LIB_TRANSFORMS_VECTORIZE_SLPVPLANCODEGEN_H
+#define LLVM_LIB_TRANSFORMS_VECTORIZE_SLPVPLANCODEGEN_H
+
+#include "llvm/ADT/ArrayRef.h"
+
+namespace llvm {
+class Instruction;
+class Value;
+struct VPBuilderDefaultInserter;
+template <typename InserterTy> class VPBuilderBase;
+using VPBuilder = VPBuilderBase<VPBuilderDefaultInserter>;
+class VPValue;
+class VPlan;
+struct VPTransformState;
+
+namespace slpvectorizer {
+
+/// Returns true if operand \p J of a bundle with main opcode \p Opcode is a
+/// plan live-in rather than being defined by another tree entry.
+bool isLiveInOperand(unsigned Opcode, unsigned J);
+
+/// Returns true if VPlan-based codegen can emit a recipe for a bundle with
+/// main opcode \p Opcode. Keep in sync with createRecipeForBundle().
+bool isSupportedVPlanCodegenOpcode(unsigned Opcode);
+
+/// Creates the recipe for the bundle \p Scalars with main instruction \p MainOp
+/// and vector operands \p Ops. Returns its value, or nullptr for stores.
+VPValue *createRecipeForBundle(VPlan &Plan, VPBuilder &VPB, Instruction *MainOp,
+ ArrayRef<Value *> Scalars,
+ ArrayRef<VPValue *> Ops);
+
+/// Executes the recipes of \p Plan, which are already in execution order.
+void executeSLPPlan(VPlan &Plan, VPTransformState &State);
+
+} // namespace slpvectorizer
+} // namespace llvm
+
+#endif // LLVM_LIB_TRANSFORMS_VECTORIZE_SLPVPLANCODEGEN_H
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index f76a7544184222..1ff91a3f5863a6 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -17,6 +17,8 @@
//===----------------------------------------------------------------------===//
#include "llvm/Transforms/Vectorize/SLPVectorizer.h"
+#include "LoopVectorizationPlanner.h"
+#include "SLPVPlanCodegen.h"
#include "SLPVectorizer/SLPCompatibilityAnalysis.h"
#include "SLPVectorizer/SLPCostAnalysis.h"
#include "SLPVectorizer/SLPMemoryUtils.h"
@@ -24,6 +26,8 @@
#include "SLPVectorizer/SLPShuffleAnalysis.h"
#include "SLPVectorizer/SLPTypeUtils.h"
#include "SLPVectorizer/SLPUtils.h"
+#include "VPlan.h"
+#include "VPlanHelpers.h"
#include "llvm/ADT/DenseMap.h"
#include "llvm/ADT/DenseSet.h"
#include "llvm/ADT/PriorityQueue.h"
@@ -337,6 +341,10 @@ static cl::opt<unsigned> SLPRuntimeAliasChecksMaxScalarCostPercent(
"guarded scalar region cost, before versioning is rejected to "
"avoid pessimizing the scalar fallback path."));
+static cl::opt<bool>
+ SLPUseVPlanCodegen("slp-use-vplan-codegen", cl::init(false), cl::Hidden,
+ cl::desc("Use VPlan-based codegen in SLP vectorizer"));
+
// Limit the number of alias checks. The limit is chosen so that
// it has no negative effect on the llvm benchmarks.
static const unsigned AliasedCheckLimit = 10;
@@ -457,6 +465,16 @@ class slpvectorizer::BoUpSLP {
Instruction *ReductionRoot = nullptr,
ArrayRef<ReductionVectorPart> VectorValuesAndScales = {});
+ /// Returns true if the current SLP tree is eligible for VPlan-based codegen.
+ /// Must be called after scheduling.
+ bool isVPlanEligible();
+
+ /// Build a VPlan for the current SLP tree.
+ std::unique_ptr<VPlan> buildVPlanForTree();
+
+ /// Execute \p Plan, generating vector IR for the SLP tree.
+ void executeVPlanForTree(VPlan &Plan);
+
/// \returns the cost incurred by unwanted spills and fills, caused by
/// holding live values over call sites.
InstructionCost getSpillCost();
@@ -25722,6 +25740,139 @@ void BoUpSLP::versionBlocksForRuntimeChecks() {
CFGChanged = true;
}
+bool BoUpSLP::isVPlanEligible() {
+ // External uses require extracts, which are not supported yet.
+ if (!ExternalUses.empty())
+ return false;
+
+ BasicBlock *RootBB =
+ cast<Instruction>(getRootNode().Scalars.front())->getParent();
+ Instruction *FirstLoad = nullptr;
+ for (const std::unique_ptr<TreeEntry> &TE : VectorizableTree) {
+ if (DeletedNodes.contains(TE.get()))
+ continue;
+
+ if (MinBWs.contains(TE.get()))
+ return false;
+
+ // Only handle plain Vectorize entries, i.e. no gathers or combined entries,
+ // without any lane permutation.
+ if (TE->State != TreeEntry::Vectorize || !TE->hasState() ||
+ TE->CombinedOp != TreeEntry::NotCombinedOp || TE->isAltShuffle() ||
+ !TE->ReorderIndices.empty() || !TE->ReuseShuffleIndices.empty())
+ return false;
+
+ unsigned Opcode = TE->getOpcode();
+ if (!isSupportedVPlanCodegenOpcode(Opcode))
+ return false;
+
+ // All recipes are emitted into the root's block.
+ if (TE->getMainOp()->getParent() != RootBB)
+ return false;
+
+ // The recipes assume scalar element types, so revec is not supported.
+ if (TE->Scalars.front()->getType()->isVectorTy())
+ return false;
+
+ // Lanes with a different opcode, e.g. copyables, need their IR flags
+ // adjusted, which is not supported.
+ if (TE->hasCopyableElements() || any_of(TE->Scalars, [&](Value *V) {
+ auto *I = dyn_cast<Instruction>(V);
+ return I && I->getOpcode() != Opcode;
+ }))
+ return false;
+
+ // A commutative sub (feeding icmp eq/ne 0 or abs) may have its operands
+ // swapped, which invalidates nuw and nsw.
+ if (Opcode == Instruction::Sub && any_of(TE->Scalars, [](Value *V) {
+ auto *I = dyn_cast<Instruction>(V);
+ return !I || isCommutative(I);
+ }))
+ return false;
+
+ // Non-power-of-2 div/rem emitted as masked intrinsics is not supported.
+ if (getMaskedDivRemCost(*TTI, SLPReVec, Opcode,
+ getValueType(TE->Scalars.front(), SLPReVec),
+ TE->Scalars.size(), TTI::TCK_RecipThroughput)
+ .isValid())
+ return false;
+
+ // All operands must be defined by entries that are still in the tree.
+ for (unsigned J : seq<unsigned>(TE->getNumOperands())) {
+ if (isLiveInOperand(Opcode, J))
+ continue;
+ TreeEntry *OpTE = OperandsToTreeEntry.lookup({TE.get(), J});
+ if (!OpTE || DeletedNodes.contains(OpTE))
+ return false;
+ }
+
+ // Bundles with extra operands, e.g. reassociated ones, are not supported.
+ if (TE->getNumOperands() != TE->getMainOp()->getNumOperands())
+ return false;
+
+ if (Opcode == Instruction::Load) {
+ Instruction *LastInst = &getLastInstructionInBundle(TE.get());
+ if (!FirstLoad || LastInst->comesBefore(FirstLoad))
+ FirstLoad = LastInst;
+ }
+ }
+
+ // Recipes are emitted at the root, so loads must not be sunk past a write.
+ if (!FirstLoad)
+ return true;
+ Instruction *RootInst = &getLastInstructionInBundle(&getRootNode());
+ return none_of(make_range(FirstLoad->getIterator(), RootInst->getIterator()),
+ [](Instruction &I) { return I.mayWriteToMemory(); });
+}
+
+std::unique_ptr<VPlan> BoUpSLP::buildVPlanForTree() {
+ // All recipes use the root's lane count as VF.
+ assert(all_of(VectorizableTree,
+ [&](const std::unique_ptr<TreeEntry> &TE) {
+ return DeletedNodes.contains(TE.get()) ||
+ TE->Scalars.size() == getRootNode().Scalars.size();
+ }) &&
+ "all entries must have the same number of lanes as the root");
+
+ auto Plan = std::make_unique<VPlan>(getRootNode().getMainOp()->getParent(),
+ Type::getInt32Ty(F->getContext()));
+ VPBuilder VPB(Plan->getEntry());
+
+ // Create the recipes in scheduled order, like the existing codegen, with
+ // operands first. Operand entries are never deleted, see isVPlanEligible().
+ DenseMap<const TreeEntry *, VPValue *> EntryToVPValue;
+ auto AddEntry = [&](TreeEntry *E, auto &Self) -> VPValue * {
+ if (auto It = EntryToVPValue.find(E); It != EntryToVPValue.end())
+ return It->second;
+ SmallVector<VPValue *> Ops;
+ for (unsigned J : seq<unsigned>(E->getNumOperands()))
+ if (!isLiveInOperand(E->getOpcode(), J))
+ Ops.push_back(Self(getOperandEntry(E, J), Self));
+ return EntryToVPValue[E] = createRecipeForBundle(*Plan, VPB, E->getMainOp(),
+ E->Scalars, Ops);
+ };
+ SmallVector<TreeEntry *> Entries;
+ for (const std::unique_ptr<TreeEntry> &TE : VectorizableTree)
+ if (!DeletedNodes.contains(TE.get()))
+ Entries.push_back(TE.get());
+ stable_sort(Entries, [&](const TreeEntry *A, const TreeEntry *B) {
+ return getLastInstructionInBundle(A).comesBefore(
+ &getLastInstructionInBundle(B));
+ });
+ for (TreeEntry *TE : Entries)
+ AddEntry(TE, AddEntry);
+
+ return Plan;
+}
+
+void BoUpSLP::executeVPlanForTree(VPlan &Plan) {
+ VPTransformState State(
+ TTI, ElementCount::getFixed(getRootNode().Scalars.size()), LI, DT, AC,
+ Builder, &Plan, /*CurrentParentLoop=*/nullptr);
+
+ executeSLPPlan(Plan, State);
+}
+
Value *BoUpSLP::vectorizeTree() {
ExtraValueToDebugLocsMap ExternallyUsedValues;
return vectorizeTree(ExternallyUsedValues);
@@ -25768,6 +25919,9 @@ BoUpSLP::vectorizeTree(const ExtraValueToDebugLocsMap &ExternallyUsedValues,
else
Builder.SetInsertPoint(&F->getEntryBlock(), F->getEntryBlock().begin());
+ // Reductions need the root's VectorizedValue, which VPlan does not set.
+ bool UsedVPlan = SLPUseVPlanCodegen && !ReductionRoot && isVPlanEligible();
+
// Vectorize gather operands of the nodes with the external uses only.
SmallVector<std::pair<TreeEntry *, Instruction *>> GatherEntries;
// Multiple gather TEs may share the same UserTE - cache the per-UserTE
@@ -25805,14 +25959,14 @@ BoUpSLP::vectorizeTree(const ExtraValueToDebugLocsMap &ExternallyUsedValues,
// the splat gathers can be emitted as their broadcasts. They go before the
// gathered loads, which skip entries that already have a vector value.
for (TreeEntry *TE : SplatGatheredScalarsRoots) {
- if (DeletedNodes.contains(TE) || TE->VectorizedValue)
+ if (UsedVPlan || DeletedNodes.contains(TE) || TE->VectorizedValue)
continue;
(void)vectorizeTree(TE);
}
// Emit gathered loads first to emit better code for the users of those
// gathered loads.
for (const std::unique_ptr<TreeEntry> &TE : VectorizableTree) {
- if (DeletedNodes.contains(TE.get()))
+ if (UsedVPlan || DeletedNodes.contains(TE.get()))
continue;
if (GatheredLoadsEntriesFirst.has_value() &&
TE->Idx >= *GatheredLoadsEntriesFirst && !TE->VectorizedValue &&
@@ -25823,7 +25977,13 @@ BoUpSLP::vectorizeTree(const ExtraValueToDebugLocsMap &ExternallyUsedValues,
(void)vectorizeTree(TE.get());
}
}
- (void)vectorizeTree(&getRootNode());
+ if (UsedVPlan) {
+ setInsertPointAfterBundle(&getRootNode());
+ std::unique_ptr<VPlan> Plan = buildVPlanForTree();
+ executeVPlanForTree(*Plan);
+ } else {
+ (void)vectorizeTree(&getRootNode());
+ }
// Run through the list of postponed gathers and emit them, replacing the temp
// emitted allocas with actual vector instructions.
ArrayRef<const TreeEntry *> PostponedNodes = PostponedGathers.getArrayRef();
@@ -26453,7 +26613,8 @@ BoUpSLP::vectorizeTree(const ExtraValueToDebugLocsMap &ExternallyUsedValues,
continue;
}
- assert(Entry->VectorizedValue && "Can't find vectorizable value");
+ assert((Entry->VectorizedValue || UsedVPlan) &&
+ "Can't find vectorizable value");
// For each lane:
for (int Lane = 0, LE = Entry->Scalars.size(); Lane != LE; ++Lane) {
@@ -26558,7 +26719,7 @@ BoUpSLP::vectorizeTree(const ExtraValueToDebugLocsMap &ExternallyUsedValues,
// Merge the DIAssignIDs from the about-to-be-deleted instructions into the
// new vector instruction.
- if (auto *V = dyn_cast<Instruction>(getRootNode().VectorizedValue))
+ if (auto *V = dyn_cast_if_present<Instruction>(getRootNode().VectorizedValue))
V->mergeDIAssignID(RemovedInsts);
// Clear up reduction references, if any.
diff --git a/llvm/lib/Transforms/Vectorize/VPlan.h b/llvm/lib/Transforms/Vectorize/VPlan.h
index 746c0231f6fcf3..c19c7c3170d488 100644
--- a/llvm/...
[truncated]
``````````
</details>
https://github.com/llvm/llvm-project/pull/226843
More information about the llvm-commits
mailing list