[llvm] [VPlan] Introduce distillation of widening semantics (NFC) (PR #196181)
Ramkumar Ramachandra via llvm-commits
llvm-commits at lists.llvm.org
Thu Sep 10 03:37:10 PDT 2026
https://github.com/artagnon updated https://github.com/llvm/llvm-project/pull/196181
>From 9064de1f59a44b4b2a69118559013e5426a39b8f Mon Sep 17 00:00:00 2001
From: Ramkumar Ramachandra <artagnon at tenstorrent.com>
Date: Wed, 6 May 2026 16:26:11 +0100
Subject: [PATCH 1/2] [VPlan] Introduce distillation of widening semantics
Introduce VPWideningInfo, a distillation of widening semantics of
recipes, and demonstrate its utility in vputils.
---
llvm/lib/Transforms/Vectorize/VPlanUtils.cpp | 243 +++++++++++++------
1 file changed, 167 insertions(+), 76 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp b/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
index 8327b30c7583e..e61896d5a906a 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
@@ -402,11 +402,40 @@ vputils::getOpcodeOrIntrinsicID(const VPValue *V) {
return {};
}
-/// Returns true if \p Opcode preserves uniformity, i.e., if all operands are
-/// uniform, the result will also be uniform.
-static bool preservesUniformity(unsigned Opcode) {
+/// A class keeping track of widening information of various recipes.
+/// A recipe necessarily produces a single scalar value if only the SingleScalar
+/// bit is set, a wide value if only the Wide bit is set, and scalar values for
+/// all VF lanes only the GenPerAllLanes bit is set. The SingleScalar bit can be
+/// set on Wide or GenPerAllLanes recipes, which indicates that the recipe could
+/// be narrowed to single-scalar if legal and profitable. For instructions not
+/// producing values, like an assume or store, the bits talk about the
+/// appropriate operands. Finally, there is a class of instructions that
+/// necessarily take vector operands and produce a scalar result termed
+/// VectorToScalar, or necessarily take a scalar values and produce a vector,
+/// termed ScalarToVector. These are marked with the Agnostic bit.
+class VPWideningInfo {
+ unsigned char Info : 4;
+
+public:
+ using VPWideningTy = enum {
+ SingleScalar = 1 << 0,
+ Wide = 1 << 1,
+ GenPerAllLanes = 1 << 2,
+ Agnostic = 1 << 3
+ };
+
+ VPWideningInfo(unsigned char Info) : Info(Info) {}
+ operator unsigned char() const { return Info; }
+ bool producesSingleScalarResult() const {
+ return Info == SingleScalar || Info == (SingleScalar | Agnostic);
+ }
+ bool couldProduceSingleScalarResult() const { return Info & SingleScalar; }
+};
+
+static VPWideningInfo getNarrowableWideningInfo(unsigned Opcode,
+ VPWideningInfo WideOrRep) {
if (Instruction::isBinaryOp(Opcode) || Instruction::isCast(Opcode))
- return true;
+ return WideOrRep | VPWideningInfo::SingleScalar;
switch (Opcode) {
case Instruction::Freeze:
case Instruction::GetElementPtr:
@@ -414,13 +443,107 @@ static bool preservesUniformity(unsigned Opcode) {
case Instruction::FCmp:
case Instruction::Select:
case VPInstruction::Not:
- case VPInstruction::Broadcast:
case VPInstruction::MaskedCond:
case VPInstruction::PtrAdd:
- return true;
+ return WideOrRep | VPWideningInfo::SingleScalar;
default:
- return false;
+ return WideOrRep;
+ }
+}
+
+static VPWideningInfo getWideningInfo(const VPRecipeBase &R) {
+ switch (R.getVPRecipeID()) {
+ case VPRecipeBase::VPVectorPointerSC:
+ case VPRecipeBase::VPVectorEndPointerSC:
+ case VPRecipeBase::VPDerivedIVSC:
+ case VPRecipeBase::VPExpandSCEVSC:
+ case VPRecipeBase::VPIRInstructionSC:
+ case VPRecipeBase::VPBranchOnMaskSC:
+ return VPWideningInfo::SingleScalar;
+ case VPRecipeBase::VPScalarIVStepsSC:
+ return VPWideningInfo::GenPerAllLanes;
+ case VPRecipeBase::VPWidenCastSC:
+ case VPRecipeBase::VPWidenGEPSC:
+ case VPRecipeBase::VPPredInstPHISC:
+ case VPRecipeBase::VPBlendSC:
+ return VPWideningInfo::Wide | VPWideningInfo::SingleScalar;
+ case VPRecipeBase::VPInstructionSC: {
+ auto *VPI = cast<VPInstruction>(&R);
+ // Broadcast is a special case of a vector-to-scalar.
+ if (VPI->isVectorToScalar() || VPI->getOpcode() == VPInstruction::Broadcast)
+ return VPWideningInfo::SingleScalar | VPWideningInfo::Agnostic;
+ // These opcodes take multiple scalars are produce a vector.
+ if (is_contained({VPInstruction::BuildStructVector,
+ VPInstruction::BuildVector,
+ VPInstruction::ActiveLaneMask},
+ VPI->getOpcode()))
+ return VPWideningInfo::Wide | VPWideningInfo::Agnostic;
+ if (VPI->isSingleScalar())
+ return VPWideningInfo::SingleScalar;
+ if (VPI->doesGeneratePerAllLanes())
+ return VPWideningInfo::GenPerAllLanes;
+ return getNarrowableWideningInfo(VPI->getOpcode(), VPWideningInfo::Wide);
+ }
+ case VPRecipeBase::VPExpressionSC: {
+ auto *Expr = cast<VPExpressionRecipe>(&R);
+ return Expr->isVectorToScalar()
+ ? (VPWideningInfo::SingleScalar | VPWideningInfo::Agnostic)
+ : VPWideningInfo::Wide;
+ }
+ case VPRecipeBase::VPReductionSC:
+ case VPRecipeBase::VPReductionEVLSC: {
+ auto *Red = cast<VPReductionRecipe>(&R);
+ return Red->isPartialReduction()
+ ? VPWideningInfo::Wide
+ : (VPWideningInfo::SingleScalar | VPWideningInfo::Agnostic);
+ }
+ case VPRecipeBase::VPReplicateSC: {
+ auto *Rep = cast<VPReplicateRecipe>(&R);
+ if (Rep->isSingleScalar())
+ return VPWideningInfo::SingleScalar;
+ return getNarrowableWideningInfo(Rep->getOpcode(),
+ VPWideningInfo::GenPerAllLanes);
+ }
+ case VPRecipeBase::VPWidenSC: {
+ auto *Wide = cast<VPWidenRecipe>(&R);
+ return getNarrowableWideningInfo(Wide->getOpcode(), VPWideningInfo::Wide);
+ }
+ case VPRecipeBase::VPWidenCanonicalIVSC:
+ case VPRecipeBase::VPWidenPHISC:
+ case VPRecipeBase::VPWidenCallSC:
+ case VPRecipeBase::VPWidenIntrinsicSC:
+ case VPRecipeBase::VPWidenMemIntrinsicSC:
+ case VPRecipeBase::VPWidenLoadSC:
+ case VPRecipeBase::VPWidenLoadEVLSC:
+ case VPRecipeBase::VPWidenStoreSC:
+ case VPRecipeBase::VPWidenStoreEVLSC:
+ case VPRecipeBase::VPInterleaveSC:
+ case VPRecipeBase::VPInterleaveEVLSC:
+ case VPRecipeBase::VPHistogramSC:
+ case VPRecipeBase::VPCurrentIterationPHISC:
+ case VPRecipeBase::VPActiveLaneMaskPHISC:
+ case VPRecipeBase::VPFirstOrderRecurrencePHISC:
+ case VPRecipeBase::VPWidenIntOrFpInductionSC:
+ case VPRecipeBase::VPWidenPointerInductionSC:
+ case VPRecipeBase::VPReductionPHISC:
+ return VPWideningInfo::Wide;
+ }
+ llvm_unreachable("Fell off end of switch: unknown recipe class");
+}
+
+static VPWideningInfo getWideningInfo(const VPValue *VPV) {
+ if (!VPV->hasDefiningRecipe()) {
+ // Only a CanonicalIV region value is single scalar.
+ if (auto *RV = dyn_cast<VPRegionValue>(VPV))
+ return RV == RV->getDefiningRegion()->getCanonicalIV()
+ ? VPWideningInfo::SingleScalar
+ : VPWideningInfo::Wide;
+ // A non-constant live-in may be introduce a Broadcast.
+ return isa<VPConstant>(VPV)
+ ? VPWideningInfo::SingleScalar
+ : VPWideningInfo::SingleScalar | VPWideningInfo::Agnostic;
}
+ return getWideningInfo(*VPV->getDefiningRecipe());
}
bool vputils::isElementwise(const VPValue *V) {
@@ -432,12 +555,6 @@ bool vputils::isElementwise(const VPValue *V) {
}
bool vputils::isSingleScalar(const VPValue *VPV) {
- // Live-in, symbolic and canonical-IV region values are single-scalar.
- if (auto *RV = dyn_cast<VPRegionValue>(VPV))
- return RV == RV->getDefiningRegion()->getCanonicalIV();
- if (isa<VPIRValue, VPSymbolicValue>(VPV))
- return true;
-
if (auto *Rep = dyn_cast<VPReplicateRecipe>(VPV)) {
const VPRegionBlock *RegionOfR = Rep->getRegion();
// Don't consider recipes in replicate regions as uniform yet; their first
@@ -445,29 +562,13 @@ bool vputils::isSingleScalar(const VPValue *VPV) {
// lanes.
if (RegionOfR && RegionOfR->isReplicator())
return false;
- return Rep->isSingleScalar() || (preservesUniformity(Rep->getOpcode()) &&
- all_of(Rep->operands(), isSingleScalar));
}
- if (isa<VPWidenGEPRecipe, VPBlendRecipe>(VPV))
- return all_of(VPV->getDefiningRecipe()->operands(), isSingleScalar);
- if (auto *WidenR = dyn_cast<VPWidenRecipe>(VPV)) {
- return preservesUniformity(WidenR->getOpcode()) &&
- all_of(WidenR->operands(), isSingleScalar);
- }
- if (auto *VPI = dyn_cast<VPInstruction>(VPV))
- return VPI->isSingleScalar() || VPI->isVectorToScalar() ||
- (preservesUniformity(VPI->getOpcode()) &&
- all_of(VPI->operands(), isSingleScalar));
- if (auto *RR = dyn_cast<VPReductionRecipe>(VPV))
- return !RR->isPartialReduction();
- if (isa<VPVectorPointerRecipe, VPVectorEndPointerRecipe, VPDerivedIVRecipe>(
- VPV))
- return true;
- if (auto *Expr = dyn_cast<VPExpressionRecipe>(VPV))
- return Expr->isVectorToScalar();
-
- // VPExpandSCEVRecipes must be placed in the entry and are always uniform.
- return isa<VPExpandSCEVRecipe>(VPV);
+ // FIXME: Marking WidenCast as a single-scalar leads to regressions.
+ VPWideningInfo Info = getWideningInfo(VPV);
+ return Info.producesSingleScalarResult() ||
+ (!isa<VPWidenCastRecipe>(VPV) &&
+ Info.couldProduceSingleScalarResult() &&
+ all_of(VPV->getDefiningRecipe()->operands(), isSingleScalar));
}
bool vputils::isUniformAcrossVFsAndUFs(const VPValue *V) {
@@ -477,50 +578,40 @@ bool vputils::isUniformAcrossVFsAndUFs(const VPValue *V) {
if (isa<VPIRValue, VPSymbolicValue>(V))
return true;
- const VPRecipeBase *R = V->getDefiningRecipe();
- const VPBasicBlock *VPBB = R ? R->getParent() : nullptr;
- const VPlan *Plan = VPBB ? VPBB->getPlan() : nullptr;
- if (VPBB &&
- (VPBB == Plan->getVectorPreheader() || VPBB == Plan->getEntry())) {
- if (match(R,
+ // Bail out on VPPhi, as we can end up in infinite cycles.
+ if (isa<VPPhi>(V))
+ return false;
+
+ if (const VPRecipeBase *R = V->getDefiningRecipe()) {
+ const VPBasicBlock *VPBB = R->getParent();
+ const VPlan *Plan = VPBB->getPlan();
+ if (VPBB == Plan->getVectorPreheader() || VPBB == Plan->getEntry()) {
+ if (match(
+ R,
m_VPInstruction<VPInstruction::CanonicalIVIncrementForPart>()) ||
- match(R, m_ExtractVectorForPart(m_VPValue(), m_VPValue())))
- return false;
- return all_of(R->operands(), isUniformAcrossVFsAndUFs);
+ match(R, m_ExtractVectorForPart(m_VPValue(), m_VPValue())))
+ return false;
+ return all_of(R->operands(), isUniformAcrossVFsAndUFs);
+ }
+ if (auto *RepR = dyn_cast<VPReplicateRecipe>(R)) {
+ // Be conservative about side-effects, except for the
+ // known-side-effecting assumes and stores, which we know will be
+ // uniform.
+ return RepR->isSingleScalar() &&
+ (!RepR->mayHaveSideEffects() ||
+ isa<AssumeInst, StoreInst>(RepR->getUnderlyingInstr())) &&
+ all_of(RepR->operands(), isUniformAcrossVFsAndUFs);
+ }
}
- return TypeSwitch<const VPRecipeBase *, bool>(R)
- .Case([](const VPDerivedIVRecipe *R) { return true; })
- .Case([](const VPReplicateRecipe *R) {
- // Be conservative about side-effects, except for the
- // known-side-effecting assumes and stores, which we know will be
- // uniform.
- return R->isSingleScalar() &&
- (!R->mayHaveSideEffects() ||
- isa<AssumeInst, StoreInst>(R->getUnderlyingInstr())) &&
- all_of(R->operands(), isUniformAcrossVFsAndUFs);
- })
- .Case([](const VPWidenRecipe *R) {
- return preservesUniformity(R->getOpcode()) &&
- all_of(R->operands(), isUniformAcrossVFsAndUFs);
- })
- .Case([](const VPPhi *) {
- // Bail out on VPPhi, as we can end up in infinite cycles.
- return false;
- })
- .Case([](const VPInstruction *VPI) {
- return (VPI->isSingleScalar() || VPI->isVectorToScalar() ||
- preservesUniformity(VPI->getOpcode())) &&
- all_of(VPI->operands(), isUniformAcrossVFsAndUFs);
- })
- .Case([](const VPWidenCastRecipe *R) {
- // A cast is uniform according to its operand.
- return isUniformAcrossVFsAndUFs(R->getOperand(0));
- })
- .Default([](const VPRecipeBase *) { // A value is considered non-uniform
- // unless proven otherwise.
- return false;
- });
+ // TODO: Match more recipes.
+ if (!isa<VPDerivedIVRecipe, VPWidenRecipe, VPWidenCastRecipe, VPInstruction>(
+ V))
+ return false;
+
+ VPWideningInfo Info = getWideningInfo(V);
+ return Info.couldProduceSingleScalarResult() &&
+ all_of(V->getDefiningRecipe()->operands(), isUniformAcrossVFsAndUFs);
}
bool vputils::doesGeneratePerAllLanes(const VPRecipeBase *R) {
>From 9fcbc356b457f88621be30b9498d4ff5e54e1816 Mon Sep 17 00:00:00 2001
From: Ramkumar Ramachandra <artagnon at tenstorrent.com>
Date: Thu, 10 Sep 2026 11:33:40 +0100
Subject: [PATCH 2/2] [VPlan] Introduce doesGenerateSingleScalar
Co-authored-by: Florian Hahn <flo at fhahn.com>
---
llvm/lib/Transforms/Vectorize/VPlanLowering.cpp | 2 +-
llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp | 2 +-
llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp | 12 +++++++-----
llvm/lib/Transforms/Vectorize/VPlanUtils.cpp | 5 ++++-
llvm/lib/Transforms/Vectorize/VPlanUtils.h | 4 ++++
5 files changed, 17 insertions(+), 8 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/VPlanLowering.cpp b/llvm/lib/Transforms/Vectorize/VPlanLowering.cpp
index ddde0cd83a704..ce8f06c561b1b 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanLowering.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanLowering.cpp
@@ -851,7 +851,7 @@ void VPlanTransforms::materializePacksAndUnpacks(VPlan &Plan) {
// TODO: The Defs skipped here may or may not be vector values.
// Introduce Unpacks, and remove them later, if they are guaranteed to
// produce scalar values.
- if (vputils::isSingleScalar(Def))
+ if (vputils::doesGenerateSingleScalar(Def))
continue;
// Only introduce an Unpack if some, but not all, users use the first
diff --git a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
index 2bf8cdcc8ca79..c97e145f758b8 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
@@ -3248,7 +3248,7 @@ void VPScalarIVStepsRecipe::printRecipe(raw_ostream &O, const Twine &Indent,
bool VPWidenGEPRecipe::usesFirstLaneOnly(const VPValue *Op) const {
assert(is_contained(operands(), Op) && "Op must be an operand of the recipe");
- return vputils::isSingleScalar(Op);
+ return vputils::doesGenerateSingleScalar(Op);
}
void VPWidenGEPRecipe::execute(VPTransformState &State) {
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index 37fa91fc97be5..f7b1f390e05c3 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -2264,10 +2264,11 @@ struct VPCSEDenseMapInfo : public DenseMapInfo<VPSingleDefRecipe *> {
/// Hash the underlying data of \p Def.
static unsigned getHashValue(const VPSingleDefRecipe *Def) {
- hash_code Result = hash_combine(
- Def->getVPRecipeID(), vputils::getOpcodeOrIntrinsicID(Def),
- getGEPSourceElementType(Def), Def->getScalarType(),
- vputils::isSingleScalar(Def), hash_combine_range(Def->operands()));
+ hash_code Result =
+ hash_combine(Def->getVPRecipeID(), vputils::getOpcodeOrIntrinsicID(Def),
+ getGEPSourceElementType(Def), Def->getScalarType(),
+ vputils::doesGenerateSingleScalar(Def),
+ hash_combine_range(Def->operands()));
if (auto *RFlags = dyn_cast<VPRecipeWithIRFlags>(Def))
if (RFlags->hasPredicate())
return hash_combine(Result, RFlags->getPredicate());
@@ -2286,7 +2287,8 @@ struct VPCSEDenseMapInfo : public DenseMapInfo<VPSingleDefRecipe *> {
vputils::getOpcodeOrIntrinsicID(L) !=
vputils::getOpcodeOrIntrinsicID(R) ||
getGEPSourceElementType(L) != getGEPSourceElementType(R) ||
- vputils::isSingleScalar(L) != vputils::isSingleScalar(R) ||
+ vputils::doesGenerateSingleScalar(L) !=
+ vputils::doesGenerateSingleScalar(R) ||
!equal(L->operands(), R->operands()))
return false;
assert(vputils::getOpcodeOrIntrinsicID(L) &&
diff --git a/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp b/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
index e61896d5a906a..7dd7ef3e7d529 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
@@ -528,7 +528,6 @@ static VPWideningInfo getWideningInfo(const VPRecipeBase &R) {
case VPRecipeBase::VPReductionPHISC:
return VPWideningInfo::Wide;
}
- llvm_unreachable("Fell off end of switch: unknown recipe class");
}
static VPWideningInfo getWideningInfo(const VPValue *VPV) {
@@ -624,6 +623,10 @@ bool vputils::doesGeneratePerAllLanes(const VPRecipeBase *R) {
return false;
}
+bool vputils::doesGenerateSingleScalar(const VPValue *V) {
+ return getWideningInfo(V).producesSingleScalarResult();
+}
+
VPBasicBlock *vputils::getFirstLoopHeader(VPlan &Plan, VPDominatorTree &VPDT) {
auto DepthFirst = vp_depth_first_shallow(Plan.getEntry());
auto I = find_if(DepthFirst, [&VPDT](VPBlockBase *VPB) {
diff --git a/llvm/lib/Transforms/Vectorize/VPlanUtils.h b/llvm/lib/Transforms/Vectorize/VPlanUtils.h
index 738a5bc8b8066..ceaf9c510ffad 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanUtils.h
+++ b/llvm/lib/Transforms/Vectorize/VPlanUtils.h
@@ -70,6 +70,10 @@ bool isElementwise(const VPValue *V);
/// Returns true if \p R produces scalar values for all VF lanes.
bool doesGeneratePerAllLanes(const VPRecipeBase *R);
+/// Returns true if \p V is defined by a recipe producing a single-scalar
+/// value or a live-in/symbolic value/single-scalar region value.
+bool doesGenerateSingleScalar(const VPValue *V);
+
/// Returns the header block of the first, top-level loop, or null if none
/// exist.
VPBasicBlock *getFirstLoopHeader(VPlan &Plan, VPDominatorTree &VPDT);
More information about the llvm-commits
mailing list