[llvm] [NFCI][VPlan] Drop VPInstructionWithType (PR #222103)
via llvm-commits
llvm-commits at lists.llvm.org
Tue Sep 8 12:01:01 PDT 2026
llvmorg-github-actions[bot] wrote:
<!--LLVM PR SUMMARY COMMENT-->
@llvm/pr-subscribers-llvm-transforms
Author: Andrei Elovikov (eas)
<details>
<summary>Changes</summary>
Addresses a TODO added in https://github.com/llvm/llvm-project/pull/199572.
I've decided to add a new ctor overload to minimize diff, we can drop it later in a separate change if desired.
AI-assisted.
---
Patch is 27.60 KiB, truncated to 20.00 KiB below, full version: https://github.com/llvm/llvm-project/pull/222103.diff
8 Files Affected:
- (modified) llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h (+11-12)
- (modified) llvm/lib/Transforms/Vectorize/LoopVectorize.cpp (+3-3)
- (modified) llvm/lib/Transforms/Vectorize/VPlan.h (+10-74)
- (modified) llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp (+1-1)
- (modified) llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp (+73-109)
- (modified) llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp (+1-1)
- (modified) llvm/unittests/Transforms/Vectorize/VPDomTreeTest.cpp (+6-6)
- (modified) llvm/unittests/Transforms/Vectorize/VPlanTest.cpp (+20-36)
``````````diff
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
index 972f8f8e642e2..2d202088501a4 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
@@ -220,8 +220,8 @@ class VPBuilder {
Type *ResultTy, const VPIRFlags &Flags = {},
DebugLoc DL = DebugLoc::getUnknown(),
const Twine &Name = "") {
- return tryInsertInstruction(new VPInstructionWithType(
- Opcode, Operands, ResultTy, Flags, {}, DL, Name));
+ return tryInsertInstruction(
+ new VPInstruction(Opcode, Operands, ResultTy, Flags, {}, DL, Name));
}
VPInstruction *createFirstActiveLane(ArrayRef<VPValue *> Masks,
@@ -418,18 +418,17 @@ class VPBuilder {
new VPDerivedIVRecipe(Kind, FPBinOp, Start, Current, Step, Flags));
}
- VPInstructionWithType *createScalarLoad(Type *ResultTy, VPValue *Addr,
- DebugLoc DL,
- const VPIRMetadata &Metadata = {}) {
- return tryInsertInstruction(new VPInstructionWithType(
- Instruction::Load, Addr, ResultTy, {}, Metadata, DL));
+ VPInstruction *createScalarLoad(Type *ResultTy, VPValue *Addr, DebugLoc DL,
+ const VPIRMetadata &Metadata = {}) {
+ return tryInsertInstruction(
+ new VPInstruction(Instruction::Load, Addr, ResultTy, {}, Metadata, DL));
}
VPInstruction *createScalarCast(Instruction::CastOps Opcode, VPValue *Op,
Type *ResultTy, DebugLoc DL,
std::optional<VPIRFlags> Flags = std::nullopt,
const VPIRMetadata &Metadata = {}) {
- return tryInsertInstruction(new VPInstructionWithType(
+ return tryInsertInstruction(new VPInstruction(
Opcode, Op, ResultTy,
Flags.value_or(VPIRFlags::getDefaultFlags(Opcode)), Metadata, DL));
}
@@ -442,8 +441,8 @@ class VPBuilder {
VPlan &Plan = getPlan();
SmallVector<VPValue *, 2> Ops(Operands);
Ops.push_back(Plan.getConstantInt(8 * sizeof(IntrinsicID), IntrinsicID));
- return tryInsertInstruction(new VPInstructionWithType(
- VPInstruction::Intrinsic, Ops, ResultTy, {}, {}, DL));
+ return tryInsertInstruction(
+ new VPInstruction(VPInstruction::Intrinsic, Ops, ResultTy, {}, {}, DL));
}
/// Create a scalar llvm.vscale call.
@@ -495,8 +494,8 @@ class VPBuilder {
DebugLoc DL, Instruction *UV) {
if (Instruction::isCast(Opcode)) {
assert(!Mask && "Cast cannot be predicated");
- return new VPInstructionWithType(Opcode, Operands, UV->getType(), Flags,
- Metadata, DL, UV->getName(), UV);
+ return new VPInstruction(Opcode, Operands, UV->getType(), Flags, Metadata,
+ DL, UV->getName(), UV);
}
return new VPReplicateRecipe(UV, Operands, /*IsSingleScalar=*/true, Mask,
Flags, Metadata, DL);
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index e027dfa834759..4a3ce96261985 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -6280,7 +6280,7 @@ VPRecipeBuilder::tryToCreateWidenNonPhiRecipe(VPSingleDefRecipe *R,
if (Instruction::isCast(VPI->getOpcode())) {
auto *CI = cast<CastInst>(Instr);
- auto *CastR = cast<VPInstructionWithType>(VPI);
+ auto *CastR = cast<VPInstruction>(VPI);
return new VPWidenCastRecipe(CI->getOpcode(), VPI->getOperand(0),
CastR->getResultType(), CI, *VPI, *VPI,
VPI->getDebugLoc());
@@ -6620,8 +6620,8 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(VPlanPtr Plan,
VPReplicateRecipe, VPWidenLoadRecipe, VPWidenStoreRecipe,
VPWidenCallRecipe, VPWidenIntrinsicRecipe, VPVectorPointerRecipe,
VPVectorEndPointerRecipe, VPHistogramRecipe>(&R) ||
- (isa<VPInstructionWithType>(R) &&
- Instruction::isCast(cast<VPInstructionWithType>(R).getOpcode()) &&
+ (isa<VPInstruction>(R) &&
+ Instruction::isCast(cast<VPInstruction>(R).getOpcode()) &&
vputils::onlyFirstLaneUsed(R.getVPSingleValue())))
continue;
auto *VPI = cast<VPInstruction>(&R);
diff --git a/llvm/lib/Transforms/Vectorize/VPlan.h b/llvm/lib/Transforms/Vectorize/VPlan.h
index e9f18506af7c2..35eda87165970 100644
--- a/llvm/lib/Transforms/Vectorize/VPlan.h
+++ b/llvm/lib/Transforms/Vectorize/VPlan.h
@@ -1379,14 +1379,6 @@ class LLVM_ABI_FOR_TEST VPInstruction : public VPRecipeWithIRFlags,
/// backedge value). Has the wide induction recipe as operand.
ExitingIVValue,
MaskedCond,
-
- // The opcodes below are used for VPInstructionWithType.
- // NOTE: VPInstructionWithType classes are also used for:
- // 1. All CastInst variants - see createVPInstructionsForVPBB, and other
- // cases where createScalarCast, createScalarZExtOrTrunc and
- // createScalarSExtOrTrunc are invoked.
- // 2. Scalar load instructions - see createVPInstructionsForVPBB.
-
/// Scale the first operand (vector step) by the second operand
/// (scalar-step). Casts both operands to the result type if needed.
WideIVStep,
@@ -1441,6 +1433,13 @@ class LLVM_ABI_FOR_TEST VPInstruction : public VPRecipeWithIRFlags,
const VPIRFlags &Flags = {}, const VPIRMetadata &MD = {},
DebugLoc DL = DebugLoc::getUnknown(), const Twine &Name = "",
Type *ResultTy = nullptr);
+ VPInstruction(unsigned Opcode, ArrayRef<VPValue *> Operands, Type *ResultTy,
+ const VPIRFlags &Flags = {}, const VPIRMetadata &MD = {},
+ DebugLoc DL = DebugLoc::getUnknown(), const Twine &Name = "",
+ Value *UV = nullptr)
+ : VPInstruction(Opcode, Operands, Flags, MD, DL, Name, ResultTy) {
+ setUnderlyingValue(UV);
+ }
VP_CLASSOF_IMPL(VPRecipeBase::VPInstructionSC)
@@ -1564,81 +1563,18 @@ class LLVM_ABI_FOR_TEST VPInstruction : public VPRecipeWithIRFlags,
/// Set the symbolic name for the VPInstruction.
void setName(StringRef NewName) { Name = NewName.str(); }
-protected:
-#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
- /// Print the VPInstruction to \p O.
- void printRecipe(raw_ostream &O, const Twine &Indent,
- VPSlotTracker &SlotTracker) const override;
-#endif
-};
-
-/// A specialization of VPInstruction augmenting it with a dedicated result
-/// type, to be used when the opcode and operands of the VPInstruction don't
-/// directly determine the result type. Note that there is no separate recipe ID
-/// for VPInstructionWithType; it shares the same ID as VPInstruction and is
-/// distinguished purely by the opcode.
-/// TODO: Merge with VPInstruction, now that VPRecipeValue provides the type.
-class VPInstructionWithType : public VPInstruction {
-public:
- VPInstructionWithType(unsigned Opcode, ArrayRef<VPValue *> Operands,
- Type *ResultTy, const VPIRFlags &Flags = {},
- const VPIRMetadata &Metadata = {},
- DebugLoc DL = DebugLoc::getUnknown(),
- const Twine &Name = "", Value *UV = nullptr)
- : VPInstruction(Opcode, Operands, Flags, Metadata, DL, Name, ResultTy) {
- setUnderlyingValue(UV);
- }
-
- static inline bool classof(const VPRecipeBase *R) {
- // VPInstructionWithType are VPInstructions with specific opcodes requiring
- // type information.
- auto *VPI = dyn_cast<VPInstruction>(R);
- if (!VPI)
- return false;
- unsigned Opc = VPI->getOpcode();
- if (Instruction::isCast(Opc))
- return true;
- switch (Opc) {
- case VPInstruction::WideIVStep:
- case VPInstruction::StepVector:
- case VPInstruction::Intrinsic:
- case Instruction::Load:
- return true;
- default:
- return false;
- }
- }
-
- static inline bool classof(const VPUser *R) {
- return isa<VPInstructionWithType>(cast<VPRecipeBase>(R));
- }
-
- VPInstruction *clone() override {
- auto *New =
- new VPInstructionWithType(getOpcode(), operands(), getResultType(),
- *this, *this, getDebugLoc(), getName());
- New->setUnderlyingValue(getUnderlyingValue());
- return New;
- }
-
- void execute(VPTransformState &State) override;
-
- /// Return the cost of this VPInstruction.
- InstructionCost computeCost(ElementCount VF,
- VPCostContext &Ctx) const override;
-
Type *getResultType() const { return getScalarType(); }
/// Cast recipes always use scalars of their operand.
bool usesScalars(const VPValue *Op) const override {
if (Instruction::isCast(getOpcode()))
return true;
- return VPInstruction::usesScalars(Op);
+ return VPUser::usesScalars(Op);
}
protected:
#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
- /// Print the recipe.
+ /// Print the VPInstruction to \p O.
void printRecipe(raw_ostream &O, const Twine &Indent,
VPSlotTracker &SlotTracker) const override;
#endif
@@ -3450,7 +3386,7 @@ class LLVM_ABI_FOR_TEST VPReplicateRecipe : public VPRecipeWithIRFlags,
VPIRMetadata(Metadata), IsSingleScalar(IsSingleScalar),
IsPredicated(Mask) {
assert((!IsSingleScalar || !I->isCast()) &&
- "single-scalar casts should use VPInstructionWithType");
+ "single-scalar casts should use VPInstruction");
setUnderlyingValue(I);
if (Mask)
addOperand(Mask);
diff --git a/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp b/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
index a2006ebf4bb4f..20a05a2627dd6 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
@@ -1208,7 +1208,7 @@ bool VPlanTransforms::areAllLoadsDereferenceable(VPBasicBlock *HeaderVPBB,
const DataLayout &DL = TheLoop->getHeader()->getDataLayout();
for (VPBasicBlock *VPBB : vp_rpo_plain_cfg_loop_body(HeaderVPBB)) {
for (VPRecipeBase &R : *VPBB) {
- auto *VPI = dyn_cast<VPInstructionWithType>(&R);
+ auto *VPI = dyn_cast<VPInstruction>(&R);
if (!VPI || VPI->getOpcode() != Instruction::Load) {
assert(!R.mayReadFromMemory() && "unexpected recipe reading memory");
continue;
diff --git a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
index ff212663a3fac..2ce0258bee694 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
@@ -739,6 +739,17 @@ static Instruction::BinaryOps getSubRecurOpcode(RecurKind Kind) {
Value *VPInstruction::generate(VPTransformState &State) {
IRBuilderBase &Builder = State.Builder;
+ if (Instruction::isCast(getOpcode())) {
+ Value *Op = State.get(getOperand(0), VPLane(0));
+ Value *Cast = Builder.CreateCast(Instruction::CastOps(getOpcode()), Op,
+ getResultType());
+ if (auto *CastOp = dyn_cast<Instruction>(Cast)) {
+ applyFlags(*CastOp);
+ applyMetadata(*CastOp);
+ }
+ return Cast;
+ }
+
if (Instruction::isBinaryOp(getOpcode())) {
bool OnlyFirstLaneUsed = vputils::onlyFirstLaneUsed(this);
Value *A = State.get(getOperand(0), OnlyFirstLaneUsed);
@@ -751,6 +762,16 @@ Value *VPInstruction::generate(VPTransformState &State) {
}
switch (getOpcode()) {
+ case VPInstruction::StepVector:
+ return Builder.CreateStepVector(VectorType::get(getResultType(), State.VF));
+ case VPInstruction::Intrinsic: {
+ SmallVector<Value *, 2> Args;
+ for (VPValue *Op : drop_end(operands()))
+ Args.push_back(State.get(Op, /*IsSingleScalar=*/true));
+ return Builder.CreateIntrinsic(getResultType(),
+ vputils::getIntrinsicID(this), Args,
+ /*FMFSource=*/nullptr, getName());
+ }
case VPInstruction::Not: {
bool OnlyFirstLaneUsed = vputils::onlyFirstLaneUsed(this);
Value *A = State.get(getOperand(0), OnlyFirstLaneUsed);
@@ -1330,6 +1351,13 @@ InstructionCost VPRecipeWithIRFlags::getCostForRecipeWithOpcode(
InstructionCost VPInstruction::computeCost(ElementCount VF,
VPCostContext &Ctx) const {
+ if (Instruction::isCast(getOpcode()))
+ // NOTE: At the moment it seems only possible to expose this path for
+ // the trunc, zext and sext opcodes. However, isScalarCast also covers
+ // int<>fp conversions, bitcasts, ptr<>int conversions, etc.
+ return getCostForRecipeWithOpcode(getOpcode(), ElementCount::getFixed(1),
+ Ctx);
+
if (Instruction::isBinaryOp(getOpcode())) {
if (!getUnderlyingValue() && getOpcode() != Instruction::FMul) {
// TODO: Compute cost for VPInstructions without underlying values once
@@ -1346,6 +1374,26 @@ InstructionCost VPInstruction::computeCost(ElementCount VF,
}
switch (getOpcode()) {
+ case VPInstruction::StepVector:
+ // TODO: This isn't quite right since even if the step-vector is hoisted
+ // out of the loop it has a non-zero cost in the middle block, etc.
+ // Once the stepvector is correctly hoisted out of the vector loop by the
+ // licm transform we can add the cost here so that it doesn't incorrectly
+ // affect the choice of VF.
+ return 0;
+ case VPInstruction::WideIVStep: {
+ // Isn't currently possible to expose cases where this cost is queried.
+ llvm_unreachable("computeCost for WideIVStep is not implemented yet.");
+ return 0;
+ }
+ case VPInstruction::Intrinsic: {
+ Type *Ty = getScalarType();
+ SmallVector<Type *, 2> ArgTys;
+ for (const VPValue *Op : drop_end(operands()))
+ ArgTys.push_back(Op->getScalarType());
+ IntrinsicCostAttributes Attrs(vputils::getIntrinsicID(this), Ty, ArgTys);
+ return Ctx.TTI.getIntrinsicInstrCost(Attrs, Ctx.CostKind);
+ }
case Instruction::Select: {
llvm::CmpPredicate Pred = CmpInst::BAD_ICMP_PREDICATE;
match(getOperand(0), m_Cmp(Pred, m_VPValue(), m_VPValue()));
@@ -1763,7 +1811,32 @@ void VPInstruction::printRecipe(raw_ostream &O, const Twine &Indent,
O << " = ";
}
+ if (Instruction::isCast(getOpcode())) {
+ O << Instruction::getOpcodeName(getOpcode());
+ printFlags(O);
+ printOperands(O, SlotTracker);
+ O << " to " << *getResultType();
+ return;
+ }
+
switch (getOpcode()) {
+ case VPInstruction::Intrinsic:
+ O << "call " << *getResultType() << " @"
+ << Intrinsic::getBaseName(vputils::getIntrinsicID(this)) << "(";
+ interleaveComma(drop_end(operands()), O, [&O, &SlotTracker](VPValue *Op) {
+ Op->printAsOperand(O, SlotTracker);
+ });
+ O << ")";
+ return;
+ case VPInstruction::WideIVStep:
+ O << "wide-iv-step";
+ break;
+ case VPInstruction::StepVector:
+ O << "step-vector " << *getResultType();
+ break;
+ case Instruction::Load:
+ O << "load";
+ break;
case VPInstruction::Not:
O << "not";
break;
@@ -1875,115 +1948,6 @@ void VPInstruction::printRecipe(raw_ostream &O, const Twine &Indent,
}
#endif
-void VPInstructionWithType::execute(VPTransformState &State) {
- Type *ResultTy = getResultType();
- if (Instruction::isCast(getOpcode())) {
- Value *Op = State.get(getOperand(0), VPLane(0));
- Value *Cast = State.Builder.CreateCast(Instruction::CastOps(getOpcode()),
- Op, ResultTy);
- if (auto *CastOp = dyn_cast<Instruction>(Cast)) {
- applyFlags(*CastOp);
- applyMetadata(*CastOp);
- }
- State.set(this, Cast, VPLane(0));
- return;
- }
- switch (getOpcode()) {
- case VPInstruction::StepVector: {
- Value *StepVector =
- State.Builder.CreateStepVector(VectorType::get(ResultTy, State.VF));
- State.set(this, StepVector);
- break;
- }
- case VPInstruction::Intrinsic: {
- SmallVector<Value *, 2> Args;
- for (VPValue *Op : drop_end(operands()))
- Args.push_back(State.get(Op, /*IsSingleScalar=*/true));
- Value *Call =
- State.Builder.CreateIntrinsic(ResultTy, vputils::getIntrinsicID(this),
- Args, /*FMFSource=*/nullptr, getName());
- State.set(this, Call, true);
- break;
- }
-
- default:
- llvm_unreachable("opcode not implemented yet");
- }
-}
-
-InstructionCost VPInstructionWithType::computeCost(ElementCount VF,
- VPCostContext &Ctx) const {
- // NOTE: At the moment it seems only possible to expose this path for
- // the trunc, zext and sext opcodes. However, isScalarCast also covers
- // int<>fp conversions, bitcasts, ptr<>int conversions, etc.
- if (Instruction::isCast(getOpcode()))
- return getCostForRecipeWithOpcode(getOpcode(), ElementCount::getFixed(1),
- Ctx);
-
- switch (getOpcode()) {
- case VPInstruction::StepVector:
- // TODO: This isn't quite right since even if the step-vector is hoisted
- // out of the loop it has a non-zero cost in the middle block, etc.
- // Once the stepvector is correctly hoisted out of the vector loop by the
- // licm transform we can add the cost here so that it doesn't incorrectly
- // affect the choice of VF.
- return 0;
- case VPInstruction::Intrinsic: {
- Type *Ty = getScalarType();
- SmallVector<Type *, 2> ArgTys;
- for (const VPValue *Op : drop_end(operands()))
- ArgTys.push_back(Op->getScalarType());
- IntrinsicCostAttributes Attrs(vputils::getIntrinsicID(this), Ty, ArgTys);
- return Ctx.TTI.getIntrinsicInstrCost(Attrs, Ctx.CostKind);
- }
- default:
- // Although VPInstructionWithType is also used for
- // VPInstruction::WideIVStep it isn't currently possible to expose cases
- // where the cost is queried.
- llvm_unreachable("Unhandled opcode");
- }
- return 0;
-}
-
-#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
-void VPInstructionWithType::printRecipe(raw_ostream &O, const Twine &Indent,
- VPSlotTracker &SlotTracker) const {
- O << Indent << "EMIT" << (isSingleScalar() ? "-SCALAR" : "") << " ";
- printAsOperand(O, SlotTracker);
- O << " = ";
-
- Type *ResultTy = getResultType();
- switch (getOpcode()) {
- case VPInstruction::WideIVStep:
- O << "wide-iv-step ";
- printOperands(O, SlotTracker);
- break;
- case VPInstruction::StepVector:
- O << "step-vector " << *ResultTy;
- break;
- case VPInstruction::Intrinsic: {
- O << "call " << *ResultTy << " @"
- << Intrinsic::getBaseName(vputils::getIntrinsicID(this)) << "(";
- interleaveComma(drop_end(operands()), O, [&O, &SlotTracker](VPValue *Op) {
- Op->printAsOperand(O, SlotTracker);
- });
- O << ")";
- break;
- }
- case Instruction::Load:
- O << "load ";
- printOperands(O, SlotTracker);
- break;
- default:
- assert(Instruction::isCast(getOpcode()) && "unhandled opcode");
- O << Instruction::getOpcodeName(getOpcode());
- printFlags(O);
- printOperands(O, SlotTracker);
- O << " to " << *ResultTy;
- }
-}
-#endif
-
/// Shared execute logic for VPPhi and VPWidenPHIRecipe. Creates a PHI node,
/// adds incoming values, and stores the result in State. For header phis, only
/// the preheader incoming value is added; the backedge is fixed up later by
diff --git a/llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp b/llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp
index 0524f9c08be5b..dea1b0db70d2f 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp
@@ -993,7 +993,7 @@ void VPlanTransforms::replicateByVF(VPlan &Plan, ElementCount VF) {
DefR->replaceUsesWithIf(LaneDefs[0], [DefR](VPUser &U, unsigned) {
if (U.usesFirstLaneOnly(DefR))
return true;
- auto *VPI = dyn_cast<VPInstructionWithType>(&U);
+ auto *VPI = dyn_cast<VPInstruction>(&U);
return VPI && Instruction::isCast(VPI->getOpcode());
});
diff --git a/llvm/uni...
[truncated]
``````````
</details>
https://github.com/llvm/llvm-project/pull/222103
More information about the llvm-commits
mailing list