[llvm] [LV] Add support for widening loads/stores to a VF multiple (PR #217670)
Benjamin Maxwell via llvm-commits
llvm-commits at lists.llvm.org
Wed Sep 30 09:56:40 PDT 2026
================
@@ -3989,6 +3989,76 @@ void VPlanTransforms::sinkPredicatedStores(VPlan &Plan,
}
}
+void VPlanTransforms::scaleMemoryAccessesByUF(VPlan &Plan, ElementCount VF,
+ unsigned UF,
+ const TargetTransformInfo &TTI) {
+ assert(UF > 1 && "Expected plan to have an UF > 1");
+
+ Type *IVTy = Plan.getVectorLoopRegion()->getCanonicalIVType();
+ for (VPBasicBlock *VPBB : VPBlockUtils::blocksOnly<VPBasicBlock>(
+ vp_depth_first_shallow(Plan.getVectorLoopRegion()->getEntry()))) {
+ for (VPRecipeBase &R : make_early_inc_range(*VPBB)) {
+ uint64_t Stride;
+ VPValue *StoredValue = nullptr;
+ auto m_ConstantStrideVecPtr =
+ m_VecPtr(m_VPValue(), m_ConstantInt(Stride));
+ if ((!match(&R, m_WidenLoad(m_ConstantStrideVecPtr)) &&
+ !match(&R, m_WidenStore(m_ConstantStrideVecPtr,
+ m_VPValue(StoredValue)))) ||
+ Stride != 1)
+ continue;
+
+ auto *MemOp = cast<VPWidenMemoryRecipe>(&R);
+ assert(MemOp->isConsecutive() && "Expected consecutive load/store");
+
+ // TODO: Support masked loads/stores. This requires widening the header
+ // mask to the same factor as the memory operation.
+ assert(!MemOp->isMasked() && "Masked accesses are not supported yet");
+
+ Type *AccessType = StoredValue ? StoredValue->getScalarType()
+ : R.getVPSingleValue()->getScalarType();
+ unsigned Opcode = isa<VPWidenLoadRecipe>(MemOp->getAsRecipe())
+ ? Instruction::Load
+ : Instruction::Store;
+
+ std::optional<Instruction::CastOps> CastHint;
+ VPUser *MaybeCast = Opcode == Instruction::Store
+ ? StoredValue->getDefiningRecipe()
+ : R.getVPSingleValue()->getSingleUser();
+ if (auto *Cast = dyn_cast_if_present<VPWidenCastRecipe>(MaybeCast))
+ CastHint = Cast->getOpcode();
+
+ unsigned VFMultiple = TTI.getPreferredVFMultipleForMemoryOp(
+ Opcode, AccessType, VF, UF, /*IsMasked=*/false, CastHint);
+ assert((VFMultiple != 0 && UF % VFMultiple == 0) &&
+ "VFMultiple must divide UF");
+
+ if (VFMultiple == 1)
+ continue;
+
+ VPValue *Ptr = MemOp->getAddr();
+ VPValue *VFMultipleVPV = Plan.getConstantInt(IVTy, VFMultiple);
+ VPValue *Align = Plan.getConstantInt(IVTy, MemOp->getAlign().value());
+
+ VPBuilder Builder(VPBB, R.getIterator());
+ if (Opcode == Instruction::Load) {
+ VPValue *OldLoad = R.getVPSingleValue();
+ VPValue *Load = Builder.createNaryOp(
+ VPInstruction::VFMultipleLoad, {VFMultipleVPV, Ptr, Align}, nullptr,
+ {}, {}, DebugLoc::getUnknown(), "", OldLoad->getScalarType());
+ OldLoad->replaceAllUsesWith(Load);
----------------
MacDue wrote:
I've tweaked the VPInstructions in [cad40b1](https://github.com/llvm/llvm-project/pull/217670/commits/cad40b17be4157e28b6edb5c6258d5eb8eb3049e) so that the transform only supplies a "PreferredVFMultiple", with the actual VFMultiple fixed to one before unrolling.
I think this means the VPlan is legal before/after unrolling (before the VFMultiple operations are still a single vector load/store).
https://github.com/llvm/llvm-project/pull/217670
More information about the llvm-commits
mailing list