[llvm] [VPlan] Introduce distillation of widening semantics (NFC) (PR #196181)

Ramkumar Ramachandra via llvm-commits llvm-commits at lists.llvm.org
Thu Sep 10 03:37:10 PDT 2026


https://github.com/artagnon updated https://github.com/llvm/llvm-project/pull/196181

>From 9064de1f59a44b4b2a69118559013e5426a39b8f Mon Sep 17 00:00:00 2001
From: Ramkumar Ramachandra <artagnon at tenstorrent.com>
Date: Wed, 6 May 2026 16:26:11 +0100
Subject: [PATCH 1/2] [VPlan] Introduce distillation of widening semantics

Introduce VPWideningInfo, a distillation of widening semantics of
recipes, and demonstrate its utility in vputils.
---
 llvm/lib/Transforms/Vectorize/VPlanUtils.cpp | 243 +++++++++++++------
 1 file changed, 167 insertions(+), 76 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp b/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
index 8327b30c7583e..e61896d5a906a 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
@@ -402,11 +402,40 @@ vputils::getOpcodeOrIntrinsicID(const VPValue *V) {
   return {};
 }
 
-/// Returns true if \p Opcode preserves uniformity, i.e., if all operands are
-/// uniform, the result will also be uniform.
-static bool preservesUniformity(unsigned Opcode) {
+/// A class keeping track of widening information of various recipes.
+/// A recipe necessarily produces a single scalar value if only the SingleScalar
+/// bit is set, a wide value if only the Wide bit is set, and scalar values for
+/// all VF lanes only the GenPerAllLanes bit is set. The SingleScalar bit can be
+/// set on Wide or GenPerAllLanes recipes, which indicates that the recipe could
+/// be narrowed to single-scalar if legal and profitable. For instructions not
+/// producing values, like an assume or store, the bits talk about the
+/// appropriate operands. Finally, there is a class of instructions that
+/// necessarily take vector operands and produce a scalar result termed
+/// VectorToScalar, or necessarily take a scalar values and produce a vector,
+/// termed ScalarToVector. These are marked with the Agnostic bit.
+class VPWideningInfo {
+  unsigned char Info : 4;
+
+public:
+  using VPWideningTy = enum {
+    SingleScalar = 1 << 0,
+    Wide = 1 << 1,
+    GenPerAllLanes = 1 << 2,
+    Agnostic = 1 << 3
+  };
+
+  VPWideningInfo(unsigned char Info) : Info(Info) {}
+  operator unsigned char() const { return Info; }
+  bool producesSingleScalarResult() const {
+    return Info == SingleScalar || Info == (SingleScalar | Agnostic);
+  }
+  bool couldProduceSingleScalarResult() const { return Info & SingleScalar; }
+};
+
+static VPWideningInfo getNarrowableWideningInfo(unsigned Opcode,
+                                                VPWideningInfo WideOrRep) {
   if (Instruction::isBinaryOp(Opcode) || Instruction::isCast(Opcode))
-    return true;
+    return WideOrRep | VPWideningInfo::SingleScalar;
   switch (Opcode) {
   case Instruction::Freeze:
   case Instruction::GetElementPtr:
@@ -414,13 +443,107 @@ static bool preservesUniformity(unsigned Opcode) {
   case Instruction::FCmp:
   case Instruction::Select:
   case VPInstruction::Not:
-  case VPInstruction::Broadcast:
   case VPInstruction::MaskedCond:
   case VPInstruction::PtrAdd:
-    return true;
+    return WideOrRep | VPWideningInfo::SingleScalar;
   default:
-    return false;
+    return WideOrRep;
+  }
+}
+
+static VPWideningInfo getWideningInfo(const VPRecipeBase &R) {
+  switch (R.getVPRecipeID()) {
+  case VPRecipeBase::VPVectorPointerSC:
+  case VPRecipeBase::VPVectorEndPointerSC:
+  case VPRecipeBase::VPDerivedIVSC:
+  case VPRecipeBase::VPExpandSCEVSC:
+  case VPRecipeBase::VPIRInstructionSC:
+  case VPRecipeBase::VPBranchOnMaskSC:
+    return VPWideningInfo::SingleScalar;
+  case VPRecipeBase::VPScalarIVStepsSC:
+    return VPWideningInfo::GenPerAllLanes;
+  case VPRecipeBase::VPWidenCastSC:
+  case VPRecipeBase::VPWidenGEPSC:
+  case VPRecipeBase::VPPredInstPHISC:
+  case VPRecipeBase::VPBlendSC:
+    return VPWideningInfo::Wide | VPWideningInfo::SingleScalar;
+  case VPRecipeBase::VPInstructionSC: {
+    auto *VPI = cast<VPInstruction>(&R);
+    // Broadcast is a special case of a vector-to-scalar.
+    if (VPI->isVectorToScalar() || VPI->getOpcode() == VPInstruction::Broadcast)
+      return VPWideningInfo::SingleScalar | VPWideningInfo::Agnostic;
+    // These opcodes take multiple scalars are produce a vector.
+    if (is_contained({VPInstruction::BuildStructVector,
+                      VPInstruction::BuildVector,
+                      VPInstruction::ActiveLaneMask},
+                     VPI->getOpcode()))
+      return VPWideningInfo::Wide | VPWideningInfo::Agnostic;
+    if (VPI->isSingleScalar())
+      return VPWideningInfo::SingleScalar;
+    if (VPI->doesGeneratePerAllLanes())
+      return VPWideningInfo::GenPerAllLanes;
+    return getNarrowableWideningInfo(VPI->getOpcode(), VPWideningInfo::Wide);
+  }
+  case VPRecipeBase::VPExpressionSC: {
+    auto *Expr = cast<VPExpressionRecipe>(&R);
+    return Expr->isVectorToScalar()
+               ? (VPWideningInfo::SingleScalar | VPWideningInfo::Agnostic)
+               : VPWideningInfo::Wide;
+  }
+  case VPRecipeBase::VPReductionSC:
+  case VPRecipeBase::VPReductionEVLSC: {
+    auto *Red = cast<VPReductionRecipe>(&R);
+    return Red->isPartialReduction()
+               ? VPWideningInfo::Wide
+               : (VPWideningInfo::SingleScalar | VPWideningInfo::Agnostic);
+  }
+  case VPRecipeBase::VPReplicateSC: {
+    auto *Rep = cast<VPReplicateRecipe>(&R);
+    if (Rep->isSingleScalar())
+      return VPWideningInfo::SingleScalar;
+    return getNarrowableWideningInfo(Rep->getOpcode(),
+                                     VPWideningInfo::GenPerAllLanes);
+  }
+  case VPRecipeBase::VPWidenSC: {
+    auto *Wide = cast<VPWidenRecipe>(&R);
+    return getNarrowableWideningInfo(Wide->getOpcode(), VPWideningInfo::Wide);
+  }
+  case VPRecipeBase::VPWidenCanonicalIVSC:
+  case VPRecipeBase::VPWidenPHISC:
+  case VPRecipeBase::VPWidenCallSC:
+  case VPRecipeBase::VPWidenIntrinsicSC:
+  case VPRecipeBase::VPWidenMemIntrinsicSC:
+  case VPRecipeBase::VPWidenLoadSC:
+  case VPRecipeBase::VPWidenLoadEVLSC:
+  case VPRecipeBase::VPWidenStoreSC:
+  case VPRecipeBase::VPWidenStoreEVLSC:
+  case VPRecipeBase::VPInterleaveSC:
+  case VPRecipeBase::VPInterleaveEVLSC:
+  case VPRecipeBase::VPHistogramSC:
+  case VPRecipeBase::VPCurrentIterationPHISC:
+  case VPRecipeBase::VPActiveLaneMaskPHISC:
+  case VPRecipeBase::VPFirstOrderRecurrencePHISC:
+  case VPRecipeBase::VPWidenIntOrFpInductionSC:
+  case VPRecipeBase::VPWidenPointerInductionSC:
+  case VPRecipeBase::VPReductionPHISC:
+    return VPWideningInfo::Wide;
+  }
+  llvm_unreachable("Fell off end of switch: unknown recipe class");
+}
+
+static VPWideningInfo getWideningInfo(const VPValue *VPV) {
+  if (!VPV->hasDefiningRecipe()) {
+    // Only a CanonicalIV region value is single scalar.
+    if (auto *RV = dyn_cast<VPRegionValue>(VPV))
+      return RV == RV->getDefiningRegion()->getCanonicalIV()
+                 ? VPWideningInfo::SingleScalar
+                 : VPWideningInfo::Wide;
+    // A non-constant live-in may be introduce a Broadcast.
+    return isa<VPConstant>(VPV)
+               ? VPWideningInfo::SingleScalar
+               : VPWideningInfo::SingleScalar | VPWideningInfo::Agnostic;
   }
+  return getWideningInfo(*VPV->getDefiningRecipe());
 }
 
 bool vputils::isElementwise(const VPValue *V) {
@@ -432,12 +555,6 @@ bool vputils::isElementwise(const VPValue *V) {
 }
 
 bool vputils::isSingleScalar(const VPValue *VPV) {
-  // Live-in, symbolic and canonical-IV region values are single-scalar.
-  if (auto *RV = dyn_cast<VPRegionValue>(VPV))
-    return RV == RV->getDefiningRegion()->getCanonicalIV();
-  if (isa<VPIRValue, VPSymbolicValue>(VPV))
-    return true;
-
   if (auto *Rep = dyn_cast<VPReplicateRecipe>(VPV)) {
     const VPRegionBlock *RegionOfR = Rep->getRegion();
     // Don't consider recipes in replicate regions as uniform yet; their first
@@ -445,29 +562,13 @@ bool vputils::isSingleScalar(const VPValue *VPV) {
     // lanes.
     if (RegionOfR && RegionOfR->isReplicator())
       return false;
-    return Rep->isSingleScalar() || (preservesUniformity(Rep->getOpcode()) &&
-                                     all_of(Rep->operands(), isSingleScalar));
   }
-  if (isa<VPWidenGEPRecipe, VPBlendRecipe>(VPV))
-    return all_of(VPV->getDefiningRecipe()->operands(), isSingleScalar);
-  if (auto *WidenR = dyn_cast<VPWidenRecipe>(VPV)) {
-    return preservesUniformity(WidenR->getOpcode()) &&
-           all_of(WidenR->operands(), isSingleScalar);
-  }
-  if (auto *VPI = dyn_cast<VPInstruction>(VPV))
-    return VPI->isSingleScalar() || VPI->isVectorToScalar() ||
-           (preservesUniformity(VPI->getOpcode()) &&
-            all_of(VPI->operands(), isSingleScalar));
-  if (auto *RR = dyn_cast<VPReductionRecipe>(VPV))
-    return !RR->isPartialReduction();
-  if (isa<VPVectorPointerRecipe, VPVectorEndPointerRecipe, VPDerivedIVRecipe>(
-          VPV))
-    return true;
-  if (auto *Expr = dyn_cast<VPExpressionRecipe>(VPV))
-    return Expr->isVectorToScalar();
-
-  // VPExpandSCEVRecipes must be placed in the entry and are always uniform.
-  return isa<VPExpandSCEVRecipe>(VPV);
+  // FIXME: Marking WidenCast as a single-scalar leads to regressions.
+  VPWideningInfo Info = getWideningInfo(VPV);
+  return Info.producesSingleScalarResult() ||
+         (!isa<VPWidenCastRecipe>(VPV) &&
+          Info.couldProduceSingleScalarResult() &&
+          all_of(VPV->getDefiningRecipe()->operands(), isSingleScalar));
 }
 
 bool vputils::isUniformAcrossVFsAndUFs(const VPValue *V) {
@@ -477,50 +578,40 @@ bool vputils::isUniformAcrossVFsAndUFs(const VPValue *V) {
   if (isa<VPIRValue, VPSymbolicValue>(V))
     return true;
 
-  const VPRecipeBase *R = V->getDefiningRecipe();
-  const VPBasicBlock *VPBB = R ? R->getParent() : nullptr;
-  const VPlan *Plan = VPBB ? VPBB->getPlan() : nullptr;
-  if (VPBB &&
-      (VPBB == Plan->getVectorPreheader() || VPBB == Plan->getEntry())) {
-    if (match(R,
+  // Bail out on VPPhi, as we can end up in infinite cycles.
+  if (isa<VPPhi>(V))
+    return false;
+
+  if (const VPRecipeBase *R = V->getDefiningRecipe()) {
+    const VPBasicBlock *VPBB = R->getParent();
+    const VPlan *Plan = VPBB->getPlan();
+    if (VPBB == Plan->getVectorPreheader() || VPBB == Plan->getEntry()) {
+      if (match(
+              R,
               m_VPInstruction<VPInstruction::CanonicalIVIncrementForPart>()) ||
-        match(R, m_ExtractVectorForPart(m_VPValue(), m_VPValue())))
-      return false;
-    return all_of(R->operands(), isUniformAcrossVFsAndUFs);
+          match(R, m_ExtractVectorForPart(m_VPValue(), m_VPValue())))
+        return false;
+      return all_of(R->operands(), isUniformAcrossVFsAndUFs);
+    }
+    if (auto *RepR = dyn_cast<VPReplicateRecipe>(R)) {
+      // Be conservative about side-effects, except for the
+      // known-side-effecting assumes and stores, which we know will be
+      // uniform.
+      return RepR->isSingleScalar() &&
+             (!RepR->mayHaveSideEffects() ||
+              isa<AssumeInst, StoreInst>(RepR->getUnderlyingInstr())) &&
+             all_of(RepR->operands(), isUniformAcrossVFsAndUFs);
+    }
   }
 
-  return TypeSwitch<const VPRecipeBase *, bool>(R)
-      .Case([](const VPDerivedIVRecipe *R) { return true; })
-      .Case([](const VPReplicateRecipe *R) {
-        // Be conservative about side-effects, except for the
-        // known-side-effecting assumes and stores, which we know will be
-        // uniform.
-        return R->isSingleScalar() &&
-               (!R->mayHaveSideEffects() ||
-                isa<AssumeInst, StoreInst>(R->getUnderlyingInstr())) &&
-               all_of(R->operands(), isUniformAcrossVFsAndUFs);
-      })
-      .Case([](const VPWidenRecipe *R) {
-        return preservesUniformity(R->getOpcode()) &&
-               all_of(R->operands(), isUniformAcrossVFsAndUFs);
-      })
-      .Case([](const VPPhi *) {
-        // Bail out on VPPhi, as we can end up in infinite cycles.
-        return false;
-      })
-      .Case([](const VPInstruction *VPI) {
-        return (VPI->isSingleScalar() || VPI->isVectorToScalar() ||
-                preservesUniformity(VPI->getOpcode())) &&
-               all_of(VPI->operands(), isUniformAcrossVFsAndUFs);
-      })
-      .Case([](const VPWidenCastRecipe *R) {
-        // A cast is uniform according to its operand.
-        return isUniformAcrossVFsAndUFs(R->getOperand(0));
-      })
-      .Default([](const VPRecipeBase *) { // A value is considered non-uniform
-                                          // unless proven otherwise.
-        return false;
-      });
+  // TODO: Match more recipes.
+  if (!isa<VPDerivedIVRecipe, VPWidenRecipe, VPWidenCastRecipe, VPInstruction>(
+          V))
+    return false;
+
+  VPWideningInfo Info = getWideningInfo(V);
+  return Info.couldProduceSingleScalarResult() &&
+         all_of(V->getDefiningRecipe()->operands(), isUniformAcrossVFsAndUFs);
 }
 
 bool vputils::doesGeneratePerAllLanes(const VPRecipeBase *R) {

>From 9fcbc356b457f88621be30b9498d4ff5e54e1816 Mon Sep 17 00:00:00 2001
From: Ramkumar Ramachandra <artagnon at tenstorrent.com>
Date: Thu, 10 Sep 2026 11:33:40 +0100
Subject: [PATCH 2/2] [VPlan] Introduce doesGenerateSingleScalar

Co-authored-by: Florian Hahn <flo at fhahn.com>
---
 llvm/lib/Transforms/Vectorize/VPlanLowering.cpp   |  2 +-
 llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp    |  2 +-
 llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp | 12 +++++++-----
 llvm/lib/Transforms/Vectorize/VPlanUtils.cpp      |  5 ++++-
 llvm/lib/Transforms/Vectorize/VPlanUtils.h        |  4 ++++
 5 files changed, 17 insertions(+), 8 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/VPlanLowering.cpp b/llvm/lib/Transforms/Vectorize/VPlanLowering.cpp
index ddde0cd83a704..ce8f06c561b1b 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanLowering.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanLowering.cpp
@@ -851,7 +851,7 @@ void VPlanTransforms::materializePacksAndUnpacks(VPlan &Plan) {
         // TODO: The Defs skipped here may or may not be vector values.
         // Introduce Unpacks, and remove them later, if they are guaranteed to
         // produce scalar values.
-        if (vputils::isSingleScalar(Def))
+        if (vputils::doesGenerateSingleScalar(Def))
           continue;
 
         // Only introduce an Unpack if some, but not all, users use the first
diff --git a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
index 2bf8cdcc8ca79..c97e145f758b8 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
@@ -3248,7 +3248,7 @@ void VPScalarIVStepsRecipe::printRecipe(raw_ostream &O, const Twine &Indent,
 
 bool VPWidenGEPRecipe::usesFirstLaneOnly(const VPValue *Op) const {
   assert(is_contained(operands(), Op) && "Op must be an operand of the recipe");
-  return vputils::isSingleScalar(Op);
+  return vputils::doesGenerateSingleScalar(Op);
 }
 
 void VPWidenGEPRecipe::execute(VPTransformState &State) {
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index 37fa91fc97be5..f7b1f390e05c3 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -2264,10 +2264,11 @@ struct VPCSEDenseMapInfo : public DenseMapInfo<VPSingleDefRecipe *> {
 
   /// Hash the underlying data of \p Def.
   static unsigned getHashValue(const VPSingleDefRecipe *Def) {
-    hash_code Result = hash_combine(
-        Def->getVPRecipeID(), vputils::getOpcodeOrIntrinsicID(Def),
-        getGEPSourceElementType(Def), Def->getScalarType(),
-        vputils::isSingleScalar(Def), hash_combine_range(Def->operands()));
+    hash_code Result =
+        hash_combine(Def->getVPRecipeID(), vputils::getOpcodeOrIntrinsicID(Def),
+                     getGEPSourceElementType(Def), Def->getScalarType(),
+                     vputils::doesGenerateSingleScalar(Def),
+                     hash_combine_range(Def->operands()));
     if (auto *RFlags = dyn_cast<VPRecipeWithIRFlags>(Def))
       if (RFlags->hasPredicate())
         return hash_combine(Result, RFlags->getPredicate());
@@ -2286,7 +2287,8 @@ struct VPCSEDenseMapInfo : public DenseMapInfo<VPSingleDefRecipe *> {
         vputils::getOpcodeOrIntrinsicID(L) !=
             vputils::getOpcodeOrIntrinsicID(R) ||
         getGEPSourceElementType(L) != getGEPSourceElementType(R) ||
-        vputils::isSingleScalar(L) != vputils::isSingleScalar(R) ||
+        vputils::doesGenerateSingleScalar(L) !=
+            vputils::doesGenerateSingleScalar(R) ||
         !equal(L->operands(), R->operands()))
       return false;
     assert(vputils::getOpcodeOrIntrinsicID(L) &&
diff --git a/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp b/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
index e61896d5a906a..7dd7ef3e7d529 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
@@ -528,7 +528,6 @@ static VPWideningInfo getWideningInfo(const VPRecipeBase &R) {
   case VPRecipeBase::VPReductionPHISC:
     return VPWideningInfo::Wide;
   }
-  llvm_unreachable("Fell off end of switch: unknown recipe class");
 }
 
 static VPWideningInfo getWideningInfo(const VPValue *VPV) {
@@ -624,6 +623,10 @@ bool vputils::doesGeneratePerAllLanes(const VPRecipeBase *R) {
   return false;
 }
 
+bool vputils::doesGenerateSingleScalar(const VPValue *V) {
+  return getWideningInfo(V).producesSingleScalarResult();
+}
+
 VPBasicBlock *vputils::getFirstLoopHeader(VPlan &Plan, VPDominatorTree &VPDT) {
   auto DepthFirst = vp_depth_first_shallow(Plan.getEntry());
   auto I = find_if(DepthFirst, [&VPDT](VPBlockBase *VPB) {
diff --git a/llvm/lib/Transforms/Vectorize/VPlanUtils.h b/llvm/lib/Transforms/Vectorize/VPlanUtils.h
index 738a5bc8b8066..ceaf9c510ffad 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanUtils.h
+++ b/llvm/lib/Transforms/Vectorize/VPlanUtils.h
@@ -70,6 +70,10 @@ bool isElementwise(const VPValue *V);
 /// Returns true if \p R produces scalar values for all VF lanes.
 bool doesGeneratePerAllLanes(const VPRecipeBase *R);
 
+/// Returns true if \p V is defined by a recipe producing a single-scalar
+/// value or a live-in/symbolic value/single-scalar region value.
+bool doesGenerateSingleScalar(const VPValue *V);
+
 /// Returns the header block of the first, top-level loop, or null if none
 /// exist.
 VPBasicBlock *getFirstLoopHeader(VPlan &Plan, VPDominatorTree &VPDT);



More information about the llvm-commits mailing list