[llvm] [LV] Factor costInterleaveGatherScatter (NFC) (PR #215857)
Ramkumar Ramachandra via llvm-commits
llvm-commits at lists.llvm.org
Wed Sep 30 01:02:56 PDT 2026
https://github.com/artagnon updated https://github.com/llvm/llvm-project/pull/215857
>From 3a4903746b46df911c3b90bb3b84d1f084b790f2 Mon Sep 17 00:00:00 2001
From: Ramkumar Ramachandra <artagnon at tenstorrent.com>
Date: Wed, 12 Aug 2026 18:14:21 +0100
Subject: [PATCH] [LV] Factor costInterleaveGatherScatter (NFC)
The motivation for factoring out a costInterleaveGatherScatter that
compares the cost of interleaving versus that of a gather-scatter is for
re-use in a follow-up doing VPlan-based gather-scatter-widening.
---
.../Transforms/Vectorize/LoopVectorize.cpp | 144 ++++++++----------
1 file changed, 65 insertions(+), 79 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 7b0e80390379f..222381768aa07 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -1325,6 +1325,65 @@ class LoopVectorizationCostModel {
: std::nullopt);
}
+ bool isLegalToScalarize(Instruction *I, ElementCount VF) const {
+ if (!VF.isScalable())
+ // Scalarization of fixed length vectors "just works".
+ return true;
+
+ // We have dedicated lowering for unpredicated uniform loads and
+ // stores. Note that even with tail folding we know that at least
+ // one lane is active (i.e. generalized predication is not possible
+ // here), and the logic below depends on this fact.
+ if (!foldTailByMasking())
+ return true;
+
+ // For scalable vectors, a uniform memop load is always
+ // uniform-by-parts and we know how to scalarize that.
+ if (isa<LoadInst>(I))
+ return true;
+
+ // A uniform store isn't neccessarily uniform-by-part
+ // and we can't assume scalarization.
+ auto *SI = cast<StoreInst>(I);
+ return TheLoop->isLoopInvariant(SI->getValueOperand());
+ };
+
+ /// Pick between interleave and gather-scatter based on cost. Returns a pair
+ /// of widening decision along with corresponding cost.
+ std::pair<InstWidening, InstructionCost>
+ costInterleaveGatherScatter(Instruction *I, ElementCount VF) {
+ bool IsUniform = isUniformMemOp(*I, VF);
+ InstructionCost InterleaveCost = InstructionCost::getInvalid();
+ unsigned NumAccesses = 1;
+ if (!IsUniform && isAccessInterleaved(I)) {
+ const auto *Group = getInterleavedAccessGroup(I);
+ assert(Group && "Fail to get an interleaved access group.");
+
+ NumAccesses = Group->getNumMembers();
+ if (interleavedAccessCanBeWidened(I, VF))
+ InterleaveCost = getInterleaveGroupCost(I, VF);
+ }
+
+ InstructionCost GatherScatterCost =
+ isLegalGatherOrScatter(I, VF)
+ ? getGatherScatterCost(I, VF) * NumAccesses
+ : InstructionCost::getInvalid();
+
+ // FIXME: This cost is a significant under-estimate for tail folded
+ // memory ops.
+ InstructionCost ScalarizationCost =
+ IsUniform ? (isLegalToScalarize(I, VF) ? getUniformMemOpCost(I, VF)
+ : InstructionCost::getInvalid())
+ : getMemInstScalarizationCost(I, VF) * NumAccesses;
+
+ if (!IsUniform && InterleaveCost <= GatherScatterCost &&
+ InterleaveCost < ScalarizationCost)
+ return {CM_Interleave, InterleaveCost};
+ if (GatherScatterCost < ScalarizationCost)
+ return {CM_GatherScatter, GatherScatterCost};
+ return {CM_Scalarize, ScalarizationCost};
+ }
+
/// Calculate vectorization cost of memory instruction \p I.
InstructionCost getMemoryInstructionCost(Instruction *I, ElementCount VF);
@@ -2570,8 +2629,6 @@ LoopVectorizationCostModel::getDivRemSpeculationCost(Instruction *I,
bool LoopVectorizationCostModel::interleavedAccessCanBeWidened(
Instruction *I, ElementCount VF) const {
assert(isAccessInterleaved(I) && "Expecting interleaved access.");
- assert(getWideningDecision(I, VF) == CM_Unknown &&
- "Decision should not be set yet.");
auto *Group = getInterleavedAccessGroup(I);
assert(Group && "Must have a group.");
unsigned InterleaveFactor = Group->getFactor();
@@ -4561,49 +4618,13 @@ void LoopVectorizationCostModel::setCostBasedWideningDecision(ElementCount VF) {
if (!Ptr)
continue;
+ // Choose between Interleaving, Gather/Scatter or Scalarization.
+ auto [Decision, Cost] = costInterleaveGatherScatter(&I, VF);
if (isUniformMemOp(I, VF)) {
- auto IsLegalToScalarize = [&]() {
- if (!VF.isScalable())
- // Scalarization of fixed length vectors "just works".
- return true;
-
- // We have dedicated lowering for unpredicated uniform loads and
- // stores. Note that even with tail folding we know that at least
- // one lane is active (i.e. generalized predication is not possible
- // here), and the logic below depends on this fact.
- if (!foldTailByMasking())
- return true;
-
- // For scalable vectors, a uniform memop load is always
- // uniform-by-parts and we know how to scalarize that.
- if (isa<LoadInst>(I))
- return true;
-
- // A uniform store isn't neccessarily uniform-by-part
- // and we can't assume scalarization.
- auto &SI = cast<StoreInst>(I);
- return TheLoop->isLoopInvariant(SI.getValueOperand());
- };
-
- const InstructionCost GatherScatterCost =
- isLegalGatherOrScatter(&I, VF) ? getGatherScatterCost(&I, VF)
- : InstructionCost::getInvalid();
-
- // Load: Scalar load + broadcast
- // Store: Scalar store + isLoopInvariantStoreValue ? 0 : extract
- // FIXME: This cost is a significant under-estimate for tail folded
- // memory ops.
- const InstructionCost ScalarizationCost =
- IsLegalToScalarize() ? getUniformMemOpCost(&I, VF)
- : InstructionCost::getInvalid();
-
// Choose better solution for the current VF, Note that Invalid
// costs compare as maximumal large. If both are invalid, we get
// scalable invalid which signals a failure and a vectorization abort.
- if (GatherScatterCost < ScalarizationCost)
- setWideningDecision(&I, VF, CM_GatherScatter, GatherScatterCost);
- else
- setWideningDecision(&I, VF, CM_Scalarize, ScalarizationCost);
+ setWideningDecision(&I, VF, Decision, Cost);
continue;
}
@@ -4615,45 +4636,10 @@ void LoopVectorizationCostModel::setCostBasedWideningDecision(ElementCount VF) {
continue;
}
- // Choose between Interleaving, Gather/Scatter or Scalarization.
- InstructionCost InterleaveCost = InstructionCost::getInvalid();
- unsigned NumAccesses = 1;
- if (isAccessInterleaved(&I)) {
- const auto *Group = getInterleavedAccessGroup(&I);
- assert(Group && "Fail to get an interleaved access group.");
-
- // Make one decision for the whole group.
- if (getWideningDecision(&I, VF) != CM_Unknown)
- continue;
-
- NumAccesses = Group->getNumMembers();
- if (interleavedAccessCanBeWidened(&I, VF))
- InterleaveCost = getInterleaveGroupCost(&I, VF);
- }
-
- InstructionCost GatherScatterCost =
- isLegalGatherOrScatter(&I, VF)
- ? getGatherScatterCost(&I, VF) * NumAccesses
- : InstructionCost::getInvalid();
-
- InstructionCost ScalarizationCost =
- getMemInstScalarizationCost(&I, VF) * NumAccesses;
+ // Make one decision for the whole interleave group.
+ if (isAccessInterleaved(&I) && getWideningDecision(&I, VF) != CM_Unknown)
+ continue;
- // Choose better solution for the current VF,
- // write down this decision and use it during vectorization.
- InstructionCost Cost;
- InstWidening Decision;
- if (InterleaveCost <= GatherScatterCost &&
- InterleaveCost < ScalarizationCost) {
- Decision = CM_Interleave;
- Cost = InterleaveCost;
- } else if (GatherScatterCost < ScalarizationCost) {
- Decision = CM_GatherScatter;
- Cost = GatherScatterCost;
- } else {
- Decision = CM_Scalarize;
- Cost = ScalarizationCost;
- }
// If the instructions belongs to an interleave group, the whole group
// receives the same decision. The whole group receives the cost, but
// the cost will actually be assigned to one instruction.
More information about the llvm-commits
mailing list