[llvm] [AArch64][LV] Cost low-VF interleaved access (PR #205844)
Jacob Crawley via llvm-commits
llvm-commits at lists.llvm.org
Tue Jul 14 05:47:00 PDT 2026
================
@@ -5389,18 +5389,47 @@ InstructionCost AArch64TTIImpl::getInterleavedMemoryOpCost(
return InstructionCost::getInvalid();
if (!UseMaskForGaps && Factor <= TLI->getMaxSupportedInterleaveFactor()) {
- unsigned MinElts = VecVTy->getElementCount().getKnownMinValue();
- auto *SubVecTy =
- VectorType::get(VecVTy->getElementType(),
- VecVTy->getElementCount().divideCoefficientBy(Factor));
+ ElementCount EC = VecVTy->getElementCount();
+ auto *SubVecTy = VectorType::get(VecVTy->getElementType(),
+ EC.divideCoefficientBy(Factor));
// ldN/stN only support legal vector types of size 64 or 128 in bits.
// Accesses having vector types that are a multiple of 128 bits can be
// matched to more than one ldN/stN instruction.
bool UseScalable;
- if (MinElts % Factor == 0 &&
+ if (EC.isKnownMultipleOf(Factor) &&
TLI->isLegalInterleavedAccessType(SubVecTy, DL, UseScalable))
return Factor * TLI->getNumInterleavedAccesses(SubVecTy, DL, UseScalable);
+
+ // Cost the alternative approach for scalable vectors where the interleave
+ // factor is larger than the VF: use a contiguous load/store of the full
+ // wide vector followed by deinterleave/interleave shuffles.
+ if (VecTy->isScalableTy() && EC.isKnownMultipleOf(Factor)) {
+ if (SubVecTy->getElementCount() == ElementCount::getScalable(1))
+ return InstructionCost::getInvalid();
+
+ // Cost of the contiguous memory operation on the wide vector.
+ InstructionCost MemCost;
+ if (UseMaskForCond) {
+ unsigned IID = Opcode == Instruction::Load ? Intrinsic::masked_load
+ : Intrinsic::masked_store;
+ MemCost = getMemIntrinsicInstrCost(
+ MemIntrinsicCostAttributes(IID, VecTy, Alignment, AddressSpace),
+ CostKind);
+ } else {
+ MemCost =
+ getMemoryOpCost(Opcode, VecTy, Alignment, AddressSpace, CostKind);
+ }
+
+ // llvm.vector.deinterleaveN is lowered as a binary tree of deinterleave2
+ // operations. A binary tree producing Factor leaf vectors has
+ // (Factor -1) inner deinterleave2 nodes. Each deinterleave2 on a pair of
+ // SVE registers emits one uzp1 + one uzp2.
+ // Total shuffle cost: (Factor - 1) deinterleave2 operations, each
+ // processing LT.first legal vector parts,with one uzp shuffle per part.
+ auto LT = getTypeLegalizationCost(VecTy);
+ return MemCost + (Factor - 1) * LT.first;
----------------
jacob-crawley wrote:
I've created https://github.com/llvm/llvm-project/pull/209441 as a quick fix to the cost model to address the regression.
I'll look at improving the sve codegen in a follow up patch.
https://github.com/llvm/llvm-project/pull/205844
More information about the llvm-commits
mailing list