[llvm] [LV] Factor costInterleaveGatherScatter (NFC) (PR #215857)

Ramkumar Ramachandra via llvm-commits llvm-commits at lists.llvm.org
Thu Aug 27 03:35:47 PDT 2026


https://github.com/artagnon updated https://github.com/llvm/llvm-project/pull/215857

>From c4cfd0aaab4bbe7f028891fff5923de01fd5744a Mon Sep 17 00:00:00 2001
From: Ramkumar Ramachandra <artagnon at tenstorrent.com>
Date: Wed, 12 Aug 2026 18:14:21 +0100
Subject: [PATCH] [LV] Factor costInterleaveGatherScatter (NFC)

The motivation for factoring out a costInterleaveGatherScatter that
compares the cost of interleaving versus that of a gather-scatter is for
re-use in a follow-up doing VPlan-based gather-scatter-widening.
---
 .../Vectorize/LoopVectorizationPlanner.cpp    |  16 +-
 .../Vectorize/LoopVectorizationPlanner.h      |   8 +-
 .../Transforms/Vectorize/LoopVectorize.cpp    | 159 +++++++++---------
 3 files changed, 88 insertions(+), 95 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.cpp
index c464c7894ae5c..9c2f84b37ff11 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.cpp
@@ -147,19 +147,13 @@ bool VFSelectionContext::isLegalMaskedLoadOrStore(bool IsLoad, Type *ScalarTy,
                  : TTI.isLegalMaskedStore(ScalarTy, Alignment, AddressSpace));
 }
 
-bool VFSelectionContext::isLegalGatherOrScatter(Value *V,
+bool VFSelectionContext::isLegalGatherOrScatter(bool IsLoad, Type *ScalarTy,
+                                                Align Alignment,
                                                 ElementCount VF) const {
-  bool LI = isa<LoadInst>(V);
-  bool SI = isa<StoreInst>(V);
-  if (!LI && !SI)
-    return false;
-  auto *Ty = getLoadStoreType(V);
-  Align Align = getLoadStoreAlignment(V);
-  if (VF.isVector())
-    Ty = VectorType::get(Ty, VF);
+  Type *VectorTy = toVectorTy(ScalarTy, VF);
   return ForceTargetSupportsGatherScatterOps ||
-         (LI && TTI.isLegalMaskedGather(Ty, Align)) ||
-         (SI && TTI.isLegalMaskedScatter(Ty, Align));
+         (IsLoad ? TTI.isLegalMaskedGather(VectorTy, Alignment)
+                 : TTI.isLegalMaskedScatter(VectorTy, Alignment));
 }
 
 bool VFSelectionContext::supportsScalableVectors() const {
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
index 9ca869f5ebe88..d2c41e2280133 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
@@ -806,9 +806,11 @@ class VFSelectionContext {
   bool isLegalMaskedLoadOrStore(bool IsLoad, Type *ScalarTy, Align Alignment,
                                 unsigned AddressSpace) const;
 
-  /// Returns true if the target machine can represent \p V as a masked gather
-  /// or scatter operation.
-  bool isLegalGatherOrScatter(Value *V, ElementCount VF) const;
+  /// Returns true if the target machine supports a gather (if \p IsLoad)
+  /// or scatter of scalar type \p ScalarTy with \p Alignment for vectorization
+  /// factor \p VF.
+  bool isLegalGatherOrScatter(bool IsLoad, Type *ScalarTy, Align Alignment,
+                              ElementCount VF) const;
 
   /// Split reductions into those that happen in the loop, and those that
   /// happen outside. In-loop reductions are collected into InLoopReductions.
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index a5dff6d0267b0..d08e37cbbeae8 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -1062,6 +1062,10 @@ class LoopVectorizationCostModel {
   /// consecutive or part of an interleave group.
   bool isLegalMaskedLoadOrStore(Instruction *I, ElementCount VF) const;
 
+  /// Returns true if the target machine supports gather or scatter for \p I's
+  /// data type and alignment.
+  bool isLegalGatherOrScatter(Instruction *I, ElementCount VF) const;
+
   /// Check if \p Instr belongs to any interleaved access group.
   bool isAccessInterleaved(Instruction *Instr) const {
     return InterleaveInfo.isInterleaved(Instr);
@@ -1330,6 +1334,66 @@ class LoopVectorizationCostModel {
                                         : std::nullopt);
   }
 
+  bool isLegalToScalarize(Instruction *I, ElementCount VF) const {
+    if (!VF.isScalable())
+      // Scalarization of fixed length vectors "just works".
+      return true;
+
+    // We have dedicated lowering for unpredicated uniform loads and
+    // stores.  Note that even with tail folding we know that at least
+    // one lane is active (i.e. generalized predication is not possible
+    // here), and the logic below depends on this fact.
+    if (!foldTailByMasking())
+      return true;
+
+    // For scalable vectors, a uniform memop load is always
+    // uniform-by-parts  and we know how to scalarize that.
+    if (isa<LoadInst>(I))
+      return true;
+
+    // A uniform store isn't neccessarily uniform-by-part
+    // and we can't assume scalarization.
+    auto *SI = cast<StoreInst>(I);
+    return TheLoop->isLoopInvariant(SI->getValueOperand());
+  };
+
+  /// Pick between interleave and gather-scatter based on cost. Returns a pair
+  /// of widening decision along with corresponding cost.
+  std::pair<InstWidening, InstructionCost>
+  costInterleaveGatherScatter(Instruction *I, ElementCount VF) {
+    bool IsUniform = isUniformMemOp(*I, VF);
+    InstructionCost InterleaveCost = InstructionCost::getInvalid();
+    unsigned NumAccesses = 1;
+    if (!IsUniform && isAccessInterleaved(I)) {
+      const auto *Group = getInterleavedAccessGroup(I);
+      assert(Group && "Fail to get an interleaved access group.");
+
+      if (interleavedAccessCanBeWidened(I, VF)) {
+        NumAccesses = Group->getNumMembers();
+        InterleaveCost = getInterleaveGroupCost(I, VF);
+      }
+    }
+
+    InstructionCost GatherScatterCost =
+        isLegalGatherOrScatter(I, VF)
+            ? getGatherScatterCost(I, VF) * NumAccesses
+            : InstructionCost::getInvalid();
+
+    // FIXME: This cost is a significant under-estimate for tail folded
+    // memory ops.
+    InstructionCost ScalarizationCost =
+        IsUniform ? (isLegalToScalarize(I, VF) ? getUniformMemOpCost(I, VF)
+                                               : InstructionCost::getInvalid())
+                  : getMemInstScalarizationCost(I, VF) * NumAccesses;
+
+    if (!IsUniform && InterleaveCost <= GatherScatterCost &&
+        InterleaveCost < ScalarizationCost)
+      return {CM_Interleave, InterleaveCost};
+    if (GatherScatterCost < ScalarizationCost)
+      return {CM_GatherScatter, GatherScatterCost};
+    return {CM_Scalarize, ScalarizationCost};
+  }
+
   /// Calculate vectorization cost of memory instruction \p I.
   InstructionCost getMemoryInstructionCost(Instruction *I, ElementCount VF);
 
@@ -2380,6 +2444,13 @@ bool LoopVectorizationCostModel::isLegalMaskedLoadOrStore(
                                          getLoadStoreAddressSpace(I));
 }
 
+bool LoopVectorizationCostModel::isLegalGatherOrScatter(Instruction *I,
+                                                        ElementCount VF) const {
+  assert((isa<LoadInst, StoreInst>(I)));
+  return Config.isLegalGatherOrScatter(isa<LoadInst>(I), getLoadStoreType(I),
+                                       getLoadStoreAlignment(I), VF);
+}
+
 bool LoopVectorizationCostModel::isScalarWithPredication(Instruction *I,
                                                          ElementCount VF) {
   if (!isPredicatedInst(I))
@@ -2403,7 +2474,7 @@ bool LoopVectorizationCostModel::isScalarWithPredication(Instruction *I,
     bool IsConsecutive = Legal->isConsecutivePtr(getLoadStoreType(I),
                                                  getLoadStorePointerOperand(I));
     return !(IsConsecutive && isLegalMaskedLoadOrStore(I, VF)) &&
-           !Config.isLegalGatherOrScatter(I, VF);
+           !isLegalGatherOrScatter(I, VF);
   }
   case Instruction::UDiv:
   case Instruction::SDiv:
@@ -2563,8 +2634,6 @@ LoopVectorizationCostModel::getDivRemSpeculationCost(Instruction *I,
 bool LoopVectorizationCostModel::interleavedAccessCanBeWidened(
     Instruction *I, ElementCount VF) const {
   assert(isAccessInterleaved(I) && "Expecting interleaved access.");
-  assert(getWideningDecision(I, VF) == CM_Unknown &&
-         "Decision should not be set yet.");
   auto *Group = getInterleavedAccessGroup(I);
   assert(Group && "Must have a group.");
   unsigned InterleaveFactor = Group->getFactor();
@@ -4613,50 +4682,13 @@ void LoopVectorizationCostModel::setCostBasedWideningDecision(ElementCount VF) {
       if (!Ptr)
         continue;
 
+      // Choose between Interleaving, Gather/Scatter or Scalarization.
+      auto [Decision, Cost] = costInterleaveGatherScatter(&I, VF);
       if (isUniformMemOp(I, VF)) {
-        auto IsLegalToScalarize = [&]() {
-          if (!VF.isScalable())
-            // Scalarization of fixed length vectors "just works".
-            return true;
-
-          // We have dedicated lowering for unpredicated uniform loads and
-          // stores.  Note that even with tail folding we know that at least
-          // one lane is active (i.e. generalized predication is not possible
-          // here), and the logic below depends on this fact.
-          if (!foldTailByMasking())
-            return true;
-
-          // For scalable vectors, a uniform memop load is always
-          // uniform-by-parts  and we know how to scalarize that.
-          if (isa<LoadInst>(I))
-            return true;
-
-          // A uniform store isn't neccessarily uniform-by-part
-          // and we can't assume scalarization.
-          auto &SI = cast<StoreInst>(I);
-          return TheLoop->isLoopInvariant(SI.getValueOperand());
-        };
-
-        const InstructionCost GatherScatterCost =
-            Config.isLegalGatherOrScatter(&I, VF)
-                ? getGatherScatterCost(&I, VF)
-                : InstructionCost::getInvalid();
-
-        // Load: Scalar load + broadcast
-        // Store: Scalar store + isLoopInvariantStoreValue ? 0 : extract
-        // FIXME: This cost is a significant under-estimate for tail folded
-        // memory ops.
-        const InstructionCost ScalarizationCost =
-            IsLegalToScalarize() ? getUniformMemOpCost(&I, VF)
-                                 : InstructionCost::getInvalid();
-
         // Choose better solution for the current VF,  Note that Invalid
         // costs compare as maximumal large.  If both are invalid, we get
         // scalable invalid which signals a failure and a vectorization abort.
-        if (GatherScatterCost < ScalarizationCost)
-          setWideningDecision(&I, VF, CM_GatherScatter, GatherScatterCost);
-        else
-          setWideningDecision(&I, VF, CM_Scalarize, ScalarizationCost);
+        setWideningDecision(&I, VF, Decision, Cost);
         continue;
       }
 
@@ -4668,45 +4700,10 @@ void LoopVectorizationCostModel::setCostBasedWideningDecision(ElementCount VF) {
         continue;
       }
 
-      // Choose between Interleaving, Gather/Scatter or Scalarization.
-      InstructionCost InterleaveCost = InstructionCost::getInvalid();
-      unsigned NumAccesses = 1;
-      if (isAccessInterleaved(&I)) {
-        const auto *Group = getInterleavedAccessGroup(&I);
-        assert(Group && "Fail to get an interleaved access group.");
-
-        // Make one decision for the whole group.
-        if (getWideningDecision(&I, VF) != CM_Unknown)
-          continue;
-
-        NumAccesses = Group->getNumMembers();
-        if (interleavedAccessCanBeWidened(&I, VF))
-          InterleaveCost = getInterleaveGroupCost(&I, VF);
-      }
-
-      InstructionCost GatherScatterCost =
-          Config.isLegalGatherOrScatter(&I, VF)
-              ? getGatherScatterCost(&I, VF) * NumAccesses
-              : InstructionCost::getInvalid();
-
-      InstructionCost ScalarizationCost =
-          getMemInstScalarizationCost(&I, VF) * NumAccesses;
+      // Make one decision for the whole interleave group.
+      if (isAccessInterleaved(&I) && getWideningDecision(&I, VF) != CM_Unknown)
+        continue;
 
-      // Choose better solution for the current VF,
-      // write down this decision and use it during vectorization.
-      InstructionCost Cost;
-      InstWidening Decision;
-      if (InterleaveCost <= GatherScatterCost &&
-          InterleaveCost < ScalarizationCost) {
-        Decision = CM_Interleave;
-        Cost = InterleaveCost;
-      } else if (GatherScatterCost < ScalarizationCost) {
-        Decision = CM_GatherScatter;
-        Cost = GatherScatterCost;
-      } else {
-        Decision = CM_Scalarize;
-        Cost = ScalarizationCost;
-      }
       // If the instructions belongs to an interleave group, the whole group
       // receives the same decision. The whole group receives the cost, but
       // the cost will actually be assigned to one instruction.



More information about the llvm-commits mailing list