[llvm] [SLP][modularisation][NFC] Move compress load/store helpers to SLPMemoryUtils (PR #222038)
Madhur Amilkanthwar via llvm-commits
llvm-commits at lists.llvm.org
Tue Sep 8 19:49:17 PDT 2026
https://github.com/madhur13490 updated https://github.com/llvm/llvm-project/pull/222038
>From 4fd4d764c9cacd8c775c7b35c2a8490c14f35ae6 Mon Sep 17 00:00:00 2001
From: Madhur Amilkanthwar <madhura at nvidia.com>
Date: Tue, 8 Sep 2026 08:48:36 -0700
Subject: [PATCH 1/2] [SLP][modularisation][NFC] Move compress load/store
helpers to SLPMemoryUtils
Move the following BoUpSLP-independent masked-load/store compress helpers
out of SLPVectorizer.cpp into SLPVectorizer/SLPMemoryUtils.{h,cpp}:
isMaskedLoadCompress (both overloads)
isMaskedStoreCompress
buildCompressMask, used only by isMaskedLoadCompress, moves alongside them
and stays file-local. isMaskedLoadCompress reads the file-local SLPReVec
cl::opt; the option stays static in SLPVectorizer.cpp and the moved helper
takes its value as an explicit bool parameter. Behavior is unchanged.
Part of the SLPVectorizer.cpp modularization effort:
https://discourse.llvm.org/t/modularizing-slpvectorizer-cpp/90922
Assisted by AI
---
.../Transforms/Vectorize/SLPVectorizer.cpp | 257 +-----------------
.../SLPVectorizer/SLPMemoryUtils.cpp | 255 +++++++++++++++++
.../Vectorize/SLPVectorizer/SLPMemoryUtils.h | 39 +++
3 files changed, 302 insertions(+), 249 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 950297eb31cc4..32bc120afbf7b 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -5700,249 +5700,6 @@ BoUpSLP::findReusedOrderedScalars(const BoUpSLP::TreeEntry &TE,
return std::move(CurrentOrder);
}
-/// Builds compress-like mask for shuffles for the given \p PointerOps, ordered
-/// with \p Order.
-/// \return true if the mask represents strided access, false - otherwise.
-static bool buildCompressMask(ArrayRef<Value *> PointerOps,
- ArrayRef<unsigned> Order, Type *ScalarTy,
- const DataLayout &DL, ScalarEvolution &SE,
- SmallVectorImpl<int> &CompressMask) {
- const unsigned Sz = PointerOps.size();
- CompressMask.assign(Sz, PoisonMaskElem);
- // The first element always set.
- CompressMask[0] = 0;
- // Check if the mask represents strided access.
- std::optional<unsigned> Stride = 0;
- Value *Ptr0 = Order.empty() ? PointerOps.front() : PointerOps[Order.front()];
- for (unsigned I : seq<unsigned>(1, Sz)) {
- Value *Ptr = Order.empty() ? PointerOps[I] : PointerOps[Order[I]];
- std::optional<int64_t> OptPos =
- getPointersDiff(ScalarTy, Ptr0, ScalarTy, Ptr, DL, SE);
- if (!OptPos || OptPos > std::numeric_limits<unsigned>::max())
- return false;
- unsigned Pos = static_cast<unsigned>(*OptPos);
- CompressMask[I] = Pos;
- if (!Stride)
- continue;
- if (*Stride == 0) {
- *Stride = Pos;
- continue;
- }
- if (Pos != *Stride * I)
- Stride.reset();
- }
- return Stride.has_value();
-}
-
-/// Checks if the \p VL can be transformed to a (masked)load + compress or
-/// (masked) interleaved load.
-static bool isMaskedLoadCompress(
- ArrayRef<Value *> VL, ArrayRef<Value *> PointerOps,
- ArrayRef<unsigned> Order, const TargetTransformInfo &TTI,
- const DataLayout &DL, ScalarEvolution &SE, AssumptionCache &AC,
- const DominatorTree &DT, const TargetLibraryInfo &TLI,
- const TTI::TargetCostKind CostKind,
- const function_ref<bool(Value *)> AreAllUsersVectorized, bool &IsMasked,
- unsigned &InterleaveFactor, SmallVectorImpl<int> &CompressMask,
- VectorType *&LoadVecTy) {
- InterleaveFactor = 0;
- Type *ScalarTy = VL.front()->getType();
- const size_t Sz = VL.size();
- auto *VecTy = cast<VectorType>(getWidenedType(ScalarTy, Sz));
- SmallVector<int> Mask;
- if (!Order.empty())
- inversePermutation(Order, Mask);
- // Check external uses.
- for (const auto [I, V] : enumerate(VL)) {
- if (AreAllUsersVectorized(V))
- continue;
- InstructionCost ExtractCost =
- TTI.getVectorInstrCost(Instruction::ExtractElement, VecTy, CostKind,
- Mask.empty() ? I : Mask[I]);
- InstructionCost ScalarCost =
- TTI.getInstructionCost(cast<Instruction>(V), CostKind);
- if (ExtractCost <= ScalarCost)
- return false;
- }
- Value *Ptr0;
- Value *PtrN;
- if (Order.empty()) {
- Ptr0 = PointerOps.front();
- PtrN = PointerOps.back();
- } else {
- Ptr0 = PointerOps[Order.front()];
- PtrN = PointerOps[Order.back()];
- }
- std::optional<int64_t> Diff =
- getPointersDiff(ScalarTy, Ptr0, ScalarTy, PtrN, DL, SE);
- if (!Diff)
- return false;
- const size_t MaxRegSize =
- TTI.getRegisterBitWidth(TargetTransformInfo::RGK_FixedWidthVector)
- .getFixedValue();
- // Check for very large distances between elements.
- if (*Diff / Sz >= MaxRegSize / 8)
- return false;
- LoadVecTy = cast<FixedVectorType>(getWidenedType(ScalarTy, *Diff + 1));
- auto *LI = cast<LoadInst>(Order.empty() ? VL.front() : VL[Order.front()]);
- Align CommonAlignment = LI->getAlign();
- SimplifyQuery SQ(
- DL, &TLI, &DT, &AC,
- cast<LoadInst>(Order.empty() ? VL.back() : VL[Order.back()]));
- IsMasked = !isSafeToLoadUnconditionally(Ptr0, LoadVecTy, CommonAlignment, SQ);
- if (IsMasked && !TTI.isLegalMaskedLoad(LoadVecTy, CommonAlignment,
- LI->getPointerAddressSpace()))
- return false;
- // TODO: perform the analysis of each scalar load for better
- // safe-load-unconditionally analysis.
- bool IsStrided =
- buildCompressMask(PointerOps, Order, ScalarTy, DL, SE, CompressMask);
- assert(CompressMask.size() >= 2 && "At least two elements are required");
- SmallVector<Value *> OrderedPointerOps(PointerOps);
- if (!Order.empty())
- reorderScalars(OrderedPointerOps, Mask);
- auto [ScalarGEPCost, VectorGEPCost] =
- getGEPCosts(TTI, OrderedPointerOps, OrderedPointerOps.front(),
- Instruction::Load, CostKind, ScalarTy, LoadVecTy);
- // The cost of scalar loads.
- InstructionCost ScalarLoadsCost =
- accumulate(VL, InstructionCost(),
- [&](InstructionCost C, Value *V) {
- return C + TTI.getInstructionCost(cast<Instruction>(V),
- CostKind);
- }) +
- ScalarGEPCost;
- APInt DemandedElts = APInt::getAllOnes(Sz);
- InstructionCost GatherCost =
- getScalarizationOverhead(TTI, SLPReVec, ScalarTy, VecTy, DemandedElts,
- /*Insert=*/true,
- /*Extract=*/false, CostKind) +
- ScalarLoadsCost;
- InstructionCost LoadCost = 0;
- if (IsMasked) {
- LoadCost = TTI.getMemIntrinsicInstrCost(
- MemIntrinsicCostAttributes(Intrinsic::masked_load, LoadVecTy,
- CommonAlignment,
- LI->getPointerAddressSpace()),
- CostKind);
- } else {
- LoadCost =
- TTI.getMemoryOpCost(Instruction::Load, LoadVecTy, CommonAlignment,
- LI->getPointerAddressSpace(), CostKind);
- }
- if (IsStrided && !IsMasked && Order.empty()) {
- // Check for potential segmented(interleaved) loads.
- VectorType *AlignedLoadVecTy = cast<VectorType>(getWidenedType(
- ScalarTy,
- getFullVectorNumberOfElements(TTI, ScalarTy, *Diff + 1, SLPReVec)));
- SimplifyQuery SQ(DL, &TLI, &DT, &AC, cast<LoadInst>(VL.back()));
- if (!isSafeToLoadUnconditionally(Ptr0, AlignedLoadVecTy, CommonAlignment,
- SQ))
- AlignedLoadVecTy = LoadVecTy;
- if (TTI.isLegalInterleavedAccessType(AlignedLoadVecTy, CompressMask[1],
- CommonAlignment,
- LI->getPointerAddressSpace())) {
- InstructionCost InterleavedCost =
- VectorGEPCost + TTI.getInterleavedMemoryOpCost(
- Instruction::Load, AlignedLoadVecTy,
- CompressMask[1], {}, CommonAlignment,
- LI->getPointerAddressSpace(), CostKind, IsMasked);
- if (InterleavedCost < GatherCost) {
- InterleaveFactor = CompressMask[1];
- LoadVecTy = AlignedLoadVecTy;
- return true;
- }
- }
- }
- // Estimating the compression shuffle cost below can be extremely expensive
- // for a very wide LoadVecTy, which is split into a large number of vector
- // registers (see processShuffleMasks). The shuffle cost is always
- // non-negative, so if the load cost alone already reaches the gather cost the
- // masked-load-compress cannot be profitable. Bail out before the costly
- // shuffle cost estimation in that case.
- if (VectorGEPCost + LoadCost >= GatherCost)
- return false;
- InstructionCost CompressCost = getShuffleCost(
- TTI, TTI::SK_PermuteSingleSrc, LoadVecTy, CostKind, CompressMask);
- if (!Order.empty()) {
- SmallVector<int> NewMask(Sz, PoisonMaskElem);
- for (unsigned I : seq<unsigned>(Sz)) {
- NewMask[I] = CompressMask[Mask[I]];
- }
- CompressMask.swap(NewMask);
- }
- InstructionCost TotalVecCost = VectorGEPCost + LoadCost + CompressCost;
- return TotalVecCost < GatherCost;
-}
-
-/// Checks if the \p VL can be transformed to a (masked)load + compress or
-/// (masked) interleaved load.
-static bool
-isMaskedLoadCompress(ArrayRef<Value *> VL, ArrayRef<Value *> PointerOps,
- ArrayRef<unsigned> Order, const TargetTransformInfo &TTI,
- const DataLayout &DL, ScalarEvolution &SE,
- AssumptionCache &AC, const DominatorTree &DT,
- const TargetLibraryInfo &TLI,
- const TTI::TargetCostKind CostKind,
- const function_ref<bool(Value *)> AreAllUsersVectorized) {
- bool IsMasked;
- unsigned InterleaveFactor;
- SmallVector<int> CompressMask;
- VectorType *LoadVecTy;
- return isMaskedLoadCompress(VL, PointerOps, Order, TTI, DL, SE, AC, DT, TLI,
- CostKind, AreAllUsersVectorized, IsMasked,
- InterleaveFactor, CompressMask, LoadVecTy);
-}
-
-/// Checks if the stores \p VL with pointers \p PointerOps can be lowered as a
-/// single masked store. On success \p StoreVecTy is the widened store type and
-/// \p ReuseShuffleIndices is the expand mask that places each stored value at
-/// its element offset from the base (poison in the gaps).
-static bool isMaskedStoreCompress(
- ArrayRef<Value *> VL, ArrayRef<Value *> PointerOps,
- ArrayRef<unsigned> Order, const TargetTransformInfo &TTI,
- const DataLayout &DL, ScalarEvolution &SE, Align CommonAlignment,
- SmallVectorImpl<int> &ReuseShuffleIndices, FixedVectorType *&StoreVecTy) {
- Type *ScalarTy = cast<StoreInst>(VL.front())->getValueOperand()->getType();
- const size_t Sz = VL.size();
- // Only simple scalar element types are supported.
- if (Sz < 2 || (!ScalarTy->isIntOrPtrTy() && !ScalarTy->isFloatingPointTy()))
- return false;
- Value *Ptr0 = Order.empty() ? PointerOps.front() : PointerOps[Order.front()];
- Value *PtrN = Order.empty() ? PointerOps.back() : PointerOps[Order.back()];
- std::optional<int64_t> Diff =
- getPointersDiff(ScalarTy, Ptr0, ScalarTy, PtrN, DL, SE);
- if (!Diff || *Diff <= 0)
- return false;
- // Avoid widened vectors with very large gaps between the stored elements.
- const unsigned MaxRegSize =
- TTI.getRegisterBitWidth(TargetTransformInfo::RGK_FixedWidthVector)
- .getFixedValue();
- const unsigned ScalarBits = DL.getTypeSizeInBits(ScalarTy).getFixedValue();
- if (ScalarBits == 0 ||
- static_cast<uint64_t>(*Diff) / Sz >= MaxRegSize / ScalarBits)
- return false;
- StoreVecTy = cast<FixedVectorType>(getWidenedType(ScalarTy, *Diff + 1));
- unsigned AS = cast<StoreInst>(VL.front())->getPointerAddressSpace();
- if (!TTI.isLegalMaskedStore(StoreVecTy, CommonAlignment, AS,
- TTI::ConstantMask))
- return false;
- // Build the expand mask: store I (in address-sorted order) is placed at its
- // element offset from the base, other widened lanes are poison.
- ReuseShuffleIndices.assign(*Diff + 1, PoisonMaskElem);
- int64_t Prev = -1;
- for (unsigned I : seq<unsigned>(Sz)) {
- Value *Ptr = Order.empty() ? PointerOps[I] : PointerOps[Order[I]];
- std::optional<int64_t> Off =
- getPointersDiff(ScalarTy, Ptr0, ScalarTy, Ptr, DL, SE);
- if (!Off || *Off <= Prev || *Off > *Diff)
- return false;
- ReuseShuffleIndices[*Off] = static_cast<int>(I);
- Prev = *Off;
- }
- return true;
-}
-
/// Checks if strided loads can be generated out of \p VL loads with pointers \p
/// PointerOps:
/// 1. Target with strided load support is detected.
@@ -6418,11 +6175,13 @@ BoUpSLP::LoadsState BoUpSLP::canVectorizeLoads(
// Check that the sorted loads are consecutive.
if (static_cast<uint64_t>(Diff) == Sz - 1)
return LoadsState::Vectorize;
- if (isMaskedLoadCompress(VL, PointerOps, Order, *TTI, *DL, *SE, *AC, *DT,
- *TLI, CostKind, [&](Value *V) {
- return areAllUsersVectorized(
- cast<Instruction>(V), UserIgnoreList);
- }))
+ if (isMaskedLoadCompress(
+ VL, PointerOps, Order, *TTI, *DL, *SE, *AC, *DT, *TLI, CostKind,
+ [&](Value *V) {
+ return areAllUsersVectorized(cast<Instruction>(V),
+ UserIgnoreList);
+ },
+ SLPReVec))
return LoadsState::CompressVectorize;
Align Alignment =
cast<LoadInst>(Order.empty() ? VL.front() : VL[Order.front()])
@@ -17429,7 +17188,7 @@ BoUpSLP::getEntryCost(const TreeEntry *E, ArrayRef<Value *> VectorizedVals,
PointerOps[I] = cast<LoadInst>(V)->getPointerOperand();
[[maybe_unused]] bool IsVectorized = isMaskedLoadCompress(
Scalars, PointerOps, E->ReorderIndices, *TTI, *DL, *SE, *AC, *DT,
- *TLI, CostKind, [](Value *) { return true; }, IsMasked,
+ *TLI, CostKind, [](Value *) { return true; }, SLPReVec, IsMasked,
InterleaveFactor, CompressMask, LoadVecTy);
CompressEntryToData.try_emplace(E, CompressMask, LoadVecTy,
InterleaveFactor, IsMasked);
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPMemoryUtils.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPMemoryUtils.cpp
index 7b2d5d1dd34ef..ced6baf782628 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPMemoryUtils.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPMemoryUtils.cpp
@@ -8,16 +8,28 @@
#include "SLPMemoryUtils.h"
#include "SLPCompatibilityAnalysis.h"
+#include "SLPCostAnalysis.h"
+#include "SLPTypeUtils.h"
#include "SLPUtils.h"
+#include "llvm/ADT/APInt.h"
#include "llvm/ADT/STLExtras.h"
+#include "llvm/ADT/Sequence.h"
+#include "llvm/Analysis/Loads.h"
+#include "llvm/Analysis/LoopAccessAnalysis.h"
#include "llvm/Analysis/ScalarEvolution.h"
#include "llvm/Analysis/ScalarEvolutionExpressions.h"
+#include "llvm/Analysis/TargetTransformInfo.h"
#include "llvm/Analysis/ValueTracking.h"
#include "llvm/IR/DataLayout.h"
+#include "llvm/IR/DerivedTypes.h"
#include "llvm/IR/Instructions.h"
+#include "llvm/IR/Intrinsics.h"
+#include "llvm/Support/InstructionCost.h"
#include <algorithm>
+#include <limits>
+#include <optional>
#include <set>
#include <utility>
@@ -154,4 +166,247 @@ const SCEV *calculateRtStride(ArrayRef<Value *> PointerOps, Type *ElemTy,
return Stride;
}
+/// Builds compress-like mask for shuffles for the given \p PointerOps, ordered
+/// with \p Order.
+/// \return true if the mask represents strided access, false - otherwise.
+static bool buildCompressMask(ArrayRef<Value *> PointerOps,
+ ArrayRef<unsigned> Order, Type *ScalarTy,
+ const DataLayout &DL, ScalarEvolution &SE,
+ SmallVectorImpl<int> &CompressMask) {
+ const unsigned Sz = PointerOps.size();
+ CompressMask.assign(Sz, PoisonMaskElem);
+ // The first element always set.
+ CompressMask[0] = 0;
+ // Check if the mask represents strided access.
+ std::optional<unsigned> Stride = 0;
+ Value *Ptr0 = Order.empty() ? PointerOps.front() : PointerOps[Order.front()];
+ for (unsigned I : seq<unsigned>(1, Sz)) {
+ Value *Ptr = Order.empty() ? PointerOps[I] : PointerOps[Order[I]];
+ std::optional<int64_t> OptPos =
+ getPointersDiff(ScalarTy, Ptr0, ScalarTy, Ptr, DL, SE);
+ if (!OptPos || OptPos > std::numeric_limits<unsigned>::max())
+ return false;
+ unsigned Pos = static_cast<unsigned>(*OptPos);
+ CompressMask[I] = Pos;
+ if (!Stride)
+ continue;
+ if (*Stride == 0) {
+ *Stride = Pos;
+ continue;
+ }
+ if (Pos != *Stride * I)
+ Stride.reset();
+ }
+ return Stride.has_value();
+}
+
+/// Checks if the \p VL can be transformed to a (masked)load + compress or
+/// (masked) interleaved load.
+bool isMaskedLoadCompress(
+ ArrayRef<Value *> VL, ArrayRef<Value *> PointerOps,
+ ArrayRef<unsigned> Order, const TargetTransformInfo &TTI,
+ const DataLayout &DL, ScalarEvolution &SE, AssumptionCache &AC,
+ const DominatorTree &DT, const TargetLibraryInfo &TLI,
+ const TargetTransformInfo::TargetCostKind CostKind,
+ const function_ref<bool(Value *)> AreAllUsersVectorized, bool ReVec,
+ bool &IsMasked, unsigned &InterleaveFactor,
+ SmallVectorImpl<int> &CompressMask, VectorType *&LoadVecTy) {
+ InterleaveFactor = 0;
+ Type *ScalarTy = VL.front()->getType();
+ const size_t Sz = VL.size();
+ auto *VecTy = cast<VectorType>(getWidenedType(ScalarTy, Sz));
+ SmallVector<int> Mask;
+ if (!Order.empty())
+ inversePermutation(Order, Mask);
+ // Check external uses.
+ for (const auto [I, V] : enumerate(VL)) {
+ if (AreAllUsersVectorized(V))
+ continue;
+ InstructionCost ExtractCost =
+ TTI.getVectorInstrCost(Instruction::ExtractElement, VecTy, CostKind,
+ Mask.empty() ? I : Mask[I]);
+ InstructionCost ScalarCost =
+ TTI.getInstructionCost(cast<Instruction>(V), CostKind);
+ if (ExtractCost <= ScalarCost)
+ return false;
+ }
+ Value *Ptr0;
+ Value *PtrN;
+ if (Order.empty()) {
+ Ptr0 = PointerOps.front();
+ PtrN = PointerOps.back();
+ } else {
+ Ptr0 = PointerOps[Order.front()];
+ PtrN = PointerOps[Order.back()];
+ }
+ std::optional<int64_t> Diff =
+ getPointersDiff(ScalarTy, Ptr0, ScalarTy, PtrN, DL, SE);
+ if (!Diff)
+ return false;
+ const size_t MaxRegSize =
+ TTI.getRegisterBitWidth(TargetTransformInfo::RGK_FixedWidthVector)
+ .getFixedValue();
+ // Check for very large distances between elements.
+ if (*Diff / Sz >= MaxRegSize / 8)
+ return false;
+ LoadVecTy = cast<FixedVectorType>(getWidenedType(ScalarTy, *Diff + 1));
+ auto *LI = cast<LoadInst>(Order.empty() ? VL.front() : VL[Order.front()]);
+ Align CommonAlignment = LI->getAlign();
+ SimplifyQuery SQ(
+ DL, &TLI, &DT, &AC,
+ cast<LoadInst>(Order.empty() ? VL.back() : VL[Order.back()]));
+ IsMasked = !isSafeToLoadUnconditionally(Ptr0, LoadVecTy, CommonAlignment, SQ);
+ if (IsMasked && !TTI.isLegalMaskedLoad(LoadVecTy, CommonAlignment,
+ LI->getPointerAddressSpace()))
+ return false;
+ // TODO: perform the analysis of each scalar load for better
+ // safe-load-unconditionally analysis.
+ bool IsStrided =
+ buildCompressMask(PointerOps, Order, ScalarTy, DL, SE, CompressMask);
+ assert(CompressMask.size() >= 2 && "At least two elements are required");
+ SmallVector<Value *> OrderedPointerOps(PointerOps);
+ if (!Order.empty())
+ reorderScalars(OrderedPointerOps, Mask);
+ auto [ScalarGEPCost, VectorGEPCost] =
+ getGEPCosts(TTI, OrderedPointerOps, OrderedPointerOps.front(),
+ Instruction::Load, CostKind, ScalarTy, LoadVecTy);
+ // The cost of scalar loads.
+ InstructionCost ScalarLoadsCost =
+ accumulate(VL, InstructionCost(),
+ [&](InstructionCost C, Value *V) {
+ return C + TTI.getInstructionCost(cast<Instruction>(V),
+ CostKind);
+ }) +
+ ScalarGEPCost;
+ APInt DemandedElts = APInt::getAllOnes(Sz);
+ InstructionCost GatherCost =
+ getScalarizationOverhead(TTI, ReVec, ScalarTy, VecTy, DemandedElts,
+ /*Insert=*/true,
+ /*Extract=*/false, CostKind) +
+ ScalarLoadsCost;
+ InstructionCost LoadCost = 0;
+ if (IsMasked) {
+ LoadCost = TTI.getMemIntrinsicInstrCost(
+ MemIntrinsicCostAttributes(Intrinsic::masked_load, LoadVecTy,
+ CommonAlignment,
+ LI->getPointerAddressSpace()),
+ CostKind);
+ } else {
+ LoadCost =
+ TTI.getMemoryOpCost(Instruction::Load, LoadVecTy, CommonAlignment,
+ LI->getPointerAddressSpace(), CostKind);
+ }
+ if (IsStrided && !IsMasked && Order.empty()) {
+ // Check for potential segmented(interleaved) loads.
+ VectorType *AlignedLoadVecTy = cast<VectorType>(getWidenedType(
+ ScalarTy,
+ getFullVectorNumberOfElements(TTI, ScalarTy, *Diff + 1, ReVec)));
+ SimplifyQuery SQ(DL, &TLI, &DT, &AC, cast<LoadInst>(VL.back()));
+ if (!isSafeToLoadUnconditionally(Ptr0, AlignedLoadVecTy, CommonAlignment,
+ SQ))
+ AlignedLoadVecTy = LoadVecTy;
+ if (TTI.isLegalInterleavedAccessType(AlignedLoadVecTy, CompressMask[1],
+ CommonAlignment,
+ LI->getPointerAddressSpace())) {
+ InstructionCost InterleavedCost =
+ VectorGEPCost + TTI.getInterleavedMemoryOpCost(
+ Instruction::Load, AlignedLoadVecTy,
+ CompressMask[1], {}, CommonAlignment,
+ LI->getPointerAddressSpace(), CostKind, IsMasked);
+ if (InterleavedCost < GatherCost) {
+ InterleaveFactor = CompressMask[1];
+ LoadVecTy = AlignedLoadVecTy;
+ return true;
+ }
+ }
+ }
+ // Estimating the compression shuffle cost below can be extremely expensive
+ // for a very wide LoadVecTy, which is split into a large number of vector
+ // registers (see processShuffleMasks). The shuffle cost is always
+ // non-negative, so if the load cost alone already reaches the gather cost the
+ // masked-load-compress cannot be profitable. Bail out before the costly
+ // shuffle cost estimation in that case.
+ if (VectorGEPCost + LoadCost >= GatherCost)
+ return false;
+ InstructionCost CompressCost = getShuffleCost(
+ TTI, TTI::SK_PermuteSingleSrc, LoadVecTy, CostKind, CompressMask);
+ if (!Order.empty()) {
+ SmallVector<int> NewMask(Sz, PoisonMaskElem);
+ for (unsigned I : seq<unsigned>(Sz)) {
+ NewMask[I] = CompressMask[Mask[I]];
+ }
+ CompressMask.swap(NewMask);
+ }
+ InstructionCost TotalVecCost = VectorGEPCost + LoadCost + CompressCost;
+ return TotalVecCost < GatherCost;
+}
+
+/// Checks if the \p VL can be transformed to a (masked)load + compress or
+/// (masked) interleaved load.
+bool isMaskedLoadCompress(
+ ArrayRef<Value *> VL, ArrayRef<Value *> PointerOps,
+ ArrayRef<unsigned> Order, const TargetTransformInfo &TTI,
+ const DataLayout &DL, ScalarEvolution &SE, AssumptionCache &AC,
+ const DominatorTree &DT, const TargetLibraryInfo &TLI,
+ const TargetTransformInfo::TargetCostKind CostKind,
+ const function_ref<bool(Value *)> AreAllUsersVectorized, bool ReVec) {
+ bool IsMasked;
+ unsigned InterleaveFactor;
+ SmallVector<int> CompressMask;
+ VectorType *LoadVecTy;
+ return isMaskedLoadCompress(VL, PointerOps, Order, TTI, DL, SE, AC, DT, TLI,
+ CostKind, AreAllUsersVectorized, ReVec, IsMasked,
+ InterleaveFactor, CompressMask, LoadVecTy);
+}
+
+/// Checks if the stores \p VL with pointers \p PointerOps can be lowered as a
+/// single masked store. On success \p StoreVecTy is the widened store type and
+/// \p ReuseShuffleIndices is the expand mask that places each stored value at
+/// its element offset from the base (poison in the gaps).
+bool isMaskedStoreCompress(ArrayRef<Value *> VL, ArrayRef<Value *> PointerOps,
+ ArrayRef<unsigned> Order,
+ const TargetTransformInfo &TTI, const DataLayout &DL,
+ ScalarEvolution &SE, Align CommonAlignment,
+ SmallVectorImpl<int> &ReuseShuffleIndices,
+ FixedVectorType *&StoreVecTy) {
+ Type *ScalarTy = cast<StoreInst>(VL.front())->getValueOperand()->getType();
+ const size_t Sz = VL.size();
+ // Only simple scalar element types are supported.
+ if (Sz < 2 || (!ScalarTy->isIntOrPtrTy() && !ScalarTy->isFloatingPointTy()))
+ return false;
+ Value *Ptr0 = Order.empty() ? PointerOps.front() : PointerOps[Order.front()];
+ Value *PtrN = Order.empty() ? PointerOps.back() : PointerOps[Order.back()];
+ std::optional<int64_t> Diff =
+ getPointersDiff(ScalarTy, Ptr0, ScalarTy, PtrN, DL, SE);
+ if (!Diff || *Diff <= 0)
+ return false;
+ // Avoid widened vectors with very large gaps between the stored elements.
+ const unsigned MaxRegSize =
+ TTI.getRegisterBitWidth(TargetTransformInfo::RGK_FixedWidthVector)
+ .getFixedValue();
+ const unsigned ScalarBits = DL.getTypeSizeInBits(ScalarTy).getFixedValue();
+ if (ScalarBits == 0 ||
+ static_cast<uint64_t>(*Diff) / Sz >= MaxRegSize / ScalarBits)
+ return false;
+ StoreVecTy = cast<FixedVectorType>(getWidenedType(ScalarTy, *Diff + 1));
+ unsigned AS = cast<StoreInst>(VL.front())->getPointerAddressSpace();
+ if (!TTI.isLegalMaskedStore(StoreVecTy, CommonAlignment, AS,
+ TTI::ConstantMask))
+ return false;
+ // Build the expand mask: store I (in address-sorted order) is placed at its
+ // element offset from the base, other widened lanes are poison.
+ ReuseShuffleIndices.assign(*Diff + 1, PoisonMaskElem);
+ int64_t Prev = -1;
+ for (unsigned I : seq<unsigned>(Sz)) {
+ Value *Ptr = Order.empty() ? PointerOps[I] : PointerOps[Order[I]];
+ std::optional<int64_t> Off =
+ getPointersDiff(ScalarTy, Ptr0, ScalarTy, Ptr, DL, SE);
+ if (!Off || *Off <= Prev || *Off > *Diff)
+ return false;
+ ReuseShuffleIndices[*Off] = static_cast<int>(I);
+ Prev = *Off;
+ }
+ return true;
+}
+
} // namespace llvm::slpvectorizer
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPMemoryUtils.h b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPMemoryUtils.h
index 75561b0b7b40b..96a9adf37bb8a 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPMemoryUtils.h
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPMemoryUtils.h
@@ -15,15 +15,21 @@
#define LLVM_LIB_TRANSFORMS_VECTORIZE_SLPVECTORIZER_SLPMEMORYUTILS_H
#include "llvm/ADT/ArrayRef.h"
+#include "llvm/ADT/STLExtras.h"
+#include "llvm/Analysis/TargetTransformInfo.h"
#include "llvm/Support/Alignment.h"
namespace llvm {
+class AssumptionCache;
class DataLayout;
+class DominatorTree;
+class FixedVectorType;
class SCEV;
class ScalarEvolution;
class TargetLibraryInfo;
class Type;
class Value;
+class VectorType;
} // namespace llvm
namespace llvm::slpvectorizer {
@@ -51,6 +57,39 @@ const SCEV *calculateRtStride(ArrayRef<Value *> PointerOps, Type *ElemTy,
const DataLayout &DL, ScalarEvolution &SE,
SmallVectorImpl<unsigned> &SortedIndices);
+/// Checks if the \p VL can be transformed to a (masked)load + compress or
+/// (masked) interleaved load.
+bool isMaskedLoadCompress(
+ ArrayRef<Value *> VL, ArrayRef<Value *> PointerOps,
+ ArrayRef<unsigned> Order, const TargetTransformInfo &TTI,
+ const DataLayout &DL, ScalarEvolution &SE, AssumptionCache &AC,
+ const DominatorTree &DT, const TargetLibraryInfo &TLI,
+ const TargetTransformInfo::TargetCostKind CostKind,
+ const function_ref<bool(Value *)> AreAllUsersVectorized, bool ReVec,
+ bool &IsMasked, unsigned &InterleaveFactor,
+ SmallVectorImpl<int> &CompressMask, VectorType *&LoadVecTy);
+
+/// Checks if the \p VL can be transformed to a (masked)load + compress or
+/// (masked) interleaved load.
+bool isMaskedLoadCompress(
+ ArrayRef<Value *> VL, ArrayRef<Value *> PointerOps,
+ ArrayRef<unsigned> Order, const TargetTransformInfo &TTI,
+ const DataLayout &DL, ScalarEvolution &SE, AssumptionCache &AC,
+ const DominatorTree &DT, const TargetLibraryInfo &TLI,
+ const TargetTransformInfo::TargetCostKind CostKind,
+ const function_ref<bool(Value *)> AreAllUsersVectorized, bool ReVec);
+
+/// Checks if the stores \p VL with pointers \p PointerOps can be lowered as a
+/// single masked store. On success \p StoreVecTy is the widened store type and
+/// \p ReuseShuffleIndices is the expand mask that places each stored value at
+/// its element offset from the base (poison in the gaps).
+bool isMaskedStoreCompress(ArrayRef<Value *> VL, ArrayRef<Value *> PointerOps,
+ ArrayRef<unsigned> Order,
+ const TargetTransformInfo &TTI, const DataLayout &DL,
+ ScalarEvolution &SE, Align CommonAlignment,
+ SmallVectorImpl<int> &ReuseShuffleIndices,
+ FixedVectorType *&StoreVecTy);
+
} // namespace llvm::slpvectorizer
#endif // LLVM_LIB_TRANSFORMS_VECTORIZE_SLPVECTORIZER_SLPMEMORYUTILS_H
>From 6f2b75b39652521200764a3721cb77260f15d467 Mon Sep 17 00:00:00 2001
From: Madhur Amilkanthwar <madhura at nvidia.com>
Date: Tue, 8 Sep 2026 19:46:53 -0700
Subject: [PATCH 2/2] fixup! change header
---
llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPMemoryUtils.h | 2 +-
1 file changed, 1 insertion(+), 1 deletion(-)
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPMemoryUtils.h b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPMemoryUtils.h
index 96a9adf37bb8a..fa2f5589345d7 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPMemoryUtils.h
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPMemoryUtils.h
@@ -15,7 +15,7 @@
#define LLVM_LIB_TRANSFORMS_VECTORIZE_SLPVECTORIZER_SLPMEMORYUTILS_H
#include "llvm/ADT/ArrayRef.h"
-#include "llvm/ADT/STLExtras.h"
+#include "llvm/ADT/STLFunctionalExtras.h"
#include "llvm/Analysis/TargetTransformInfo.h"
#include "llvm/Support/Alignment.h"
More information about the llvm-commits
mailing list