[llvm] ec157b6 - [VPlan] Widen masked unit-stride consecutive accesses in VPlan. (#211315)
via llvm-commits
llvm-commits at lists.llvm.org
Sat Jul 25 02:15:27 PDT 2026
Author: Florian Hahn
Date: 2026-07-25T09:15:22Z
New Revision: ec157b60d8345b67be16fde4b96f85c8512fbbb6
URL: https://github.com/llvm/llvm-project/commit/ec157b60d8345b67be16fde4b96f85c8512fbbb6
DIFF: https://github.com/llvm/llvm-project/commit/ec157b60d8345b67be16fde4b96f85c8512fbbb6.diff
LOG: [VPlan] Widen masked unit-stride consecutive accesses in VPlan. (#211315)
Extend the widenConsecutiveMemOps sub-pass to widen masked/predicated
unit-stride consecutive accesses VPlan-natively.
This requires exposing an instruction-independent
isLegalMaskedLoadOrStore in VPSelectionContext, as well as adding it to
VPCostContext. Some of the members will probably be useful for other
changes as well.
PR: https://github.com/llvm/llvm-project/pull/211315
Added:
Modified:
llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.cpp
llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
llvm/lib/Transforms/Vectorize/VPlanHelpers.h
llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
llvm/test/Transforms/LoopVectorize/VPlan/AArch64/vplan-memory-op-decisions.ll
Removed:
################################################################################
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.cpp
index dbb5ad28fb4ed..fd7fd2e011a83 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.cpp
@@ -134,16 +134,12 @@ void LoopVectorizationUtils::reportVectorization(OptimizationRemarkEmitter *ORE,
});
}
-bool VFSelectionContext::isLegalMaskedLoadOrStore(Instruction *I,
- ElementCount VF) const {
- assert(isa<LoadInst>(I) || isa<StoreInst>(I));
- auto *Ty = getLoadStoreType(I);
- const unsigned AS = getLoadStoreAddressSpace(I);
- const Align Alignment = getLoadStoreAlignment(I);
-
+bool VFSelectionContext::isLegalMaskedLoadOrStore(bool IsLoad, Type *ScalarTy,
+ Align Alignment,
+ unsigned AddressSpace) const {
return ForceTargetSupportsMaskedMemoryOps ||
- (isa<LoadInst>(I) ? TTI.isLegalMaskedLoad(Ty, Alignment, AS)
- : TTI.isLegalMaskedStore(Ty, Alignment, AS));
+ (IsLoad ? TTI.isLegalMaskedLoad(ScalarTy, Alignment, AddressSpace)
+ : TTI.isLegalMaskedStore(ScalarTy, Alignment, AddressSpace));
}
bool VFSelectionContext::isLegalGatherOrScatter(Value *V,
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
index ba2b79199b8cd..0bef996f9147b 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
@@ -798,10 +798,12 @@ class VFSelectionContext {
/// of FP operations.
bool useOrderedReductions(const RecurrenceDescriptor &RdxDesc) const;
- /// Returns true if the target machine supports masked loads or stores
- /// for \p I's data type and alignment. The caller must ensure the access is
- /// consecutive or part of an interleave group.
- bool isLegalMaskedLoadOrStore(Instruction *I, ElementCount VF) const;
+ /// Returns true if the target machine supports a masked load (if \p IsLoad)
+ /// or masked store of scalar type \p ScalarTy with \p Alignment in address
+ /// space \p AddressSpace. The caller must ensure the access is consecutive or
+ /// part of an interleave group.
+ bool isLegalMaskedLoadOrStore(bool IsLoad, Type *ScalarTy, Align Alignment,
+ unsigned AddressSpace) const;
/// Returns true if the target machine can represent \p V as a masked gather
/// or scatter operation.
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 5f5767f836948..076e8e661df8b 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -1058,6 +1058,11 @@ class LoopVectorizationCostModel {
/// and shuffles.
bool interleavedAccessCanBeWidened(Instruction *I, ElementCount VF) const;
+ /// Returns true if the target machine supports masked loads or stores
+ /// for \p I's data type and alignment. The caller must ensure the access is
+ /// consecutive or part of an interleave group.
+ bool isLegalMaskedLoadOrStore(Instruction *I, ElementCount VF) const;
+
/// Check if \p Instr belongs to any interleaved access group.
bool isAccessInterleaved(Instruction *Instr) const {
return InterleaveInfo.isInterleaved(Instr);
@@ -2394,6 +2399,14 @@ void LoopVectorizationCostModel::collectLoopScalars(ElementCount VF) {
Scalars[VF].insert_range(Worklist);
}
+bool LoopVectorizationCostModel::isLegalMaskedLoadOrStore(
+ Instruction *I, ElementCount VF) const {
+ assert(isa<LoadInst>(I) || isa<StoreInst>(I));
+ return Config.isLegalMaskedLoadOrStore(isa<LoadInst>(I), getLoadStoreType(I),
+ getLoadStoreAlignment(I),
+ getLoadStoreAddressSpace(I));
+}
+
bool LoopVectorizationCostModel::isScalarWithPredication(Instruction *I,
ElementCount VF) {
if (!isPredicatedInst(I))
@@ -2416,7 +2429,7 @@ bool LoopVectorizationCostModel::isScalarWithPredication(Instruction *I,
case Instruction::Store: {
bool IsConsecutive = Legal->isConsecutivePtr(getLoadStoreType(I),
getLoadStorePointerOperand(I));
- return !(IsConsecutive && Config.isLegalMaskedLoadOrStore(I, VF)) &&
+ return !(IsConsecutive && isLegalMaskedLoadOrStore(I, VF)) &&
!Config.isLegalGatherOrScatter(I, VF);
}
case Instruction::UDiv:
@@ -2643,7 +2656,7 @@ bool LoopVectorizationCostModel::interleavedAccessCanBeWidened(
if (VF.isScalable() && NeedsMaskForGaps)
return false;
- return Config.isLegalMaskedLoadOrStore(I, VF);
+ return isLegalMaskedLoadOrStore(I, VF);
}
std::optional<LoopVectorizationCostModel::InstWidening>
@@ -5590,7 +5603,8 @@ VPCostContext::VPCostContext(const TargetLibraryInfo &TLI, const VPlan &Plan,
LoopVectorizationCostModel &CM,
VFSelectionContext &Config)
: TTI(Config.getTTI()), TLI(TLI), LLVMCtx(Plan.getContext()), CM(CM),
- CostKind(Config.CostKind), PSE(Config.getPSE()), L(Config.getLoop()) {}
+ Config(Config), CostKind(Config.CostKind), PSE(Config.getPSE()),
+ L(Config.getLoop()) {}
InstructionCost VPCostContext::getLegacyCost(Instruction *UI,
ElementCount VF) const {
diff --git a/llvm/lib/Transforms/Vectorize/VPlanHelpers.h b/llvm/lib/Transforms/Vectorize/VPlanHelpers.h
index aa7750a27273d..482a16ebdaaaf 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanHelpers.h
+++ b/llvm/lib/Transforms/Vectorize/VPlanHelpers.h
@@ -325,6 +325,7 @@ struct VPCostContext {
const TargetLibraryInfo &TLI;
LLVMContext &LLVMCtx;
LoopVectorizationCostModel &CM;
+ const VFSelectionContext &Config;
SmallPtrSet<Instruction *, 8> SkipCostComputation;
TargetTransformInfo::TargetCostKind CostKind;
PredicatedScalarEvolution &PSE;
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index e4f44ad899261..5921b9e88d9b5 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -5469,14 +5469,11 @@ void VPlanTransforms::makeMemOpWideningDecisions(VPlan &Plan, VFRange &Range,
});
}
- // Widen unmasked unit-stride consecutive accesses, matching the legacy CM.
- // Both forward (stride +1) and reverse (stride -1) accesses are handled.
+ // Widen unit-stride consecutive accesses, matching the legacy CM. Both
+ // forward (stride +1) and reverse (stride -1) accesses are handled.
VPlanTransforms::runPass(
"widenConsecutiveMemOps", ProcessSubset, Plan, [&](VPInstruction *VPI) {
Instruction *I = VPI->getUnderlyingInstr();
- if (RecipeBuilder.isPredicatedInst(I))
- return false;
-
bool IsLoad = VPI->getOpcode() == Instruction::Load;
VPValue *Ptr = VPI->getOperand(!IsLoad);
Type *ScalarTy =
@@ -5487,13 +5484,28 @@ void VPlanTransforms::makeMemOpWideningDecisions(VPlan &Plan, VFRange &Range,
return false;
bool Reverse = Stride == -1;
+ // A predicated access can only be widened (rather than scalarized) if
+ // the target supports a masked load/store for it.
+ // TODO: Determine if a load/store needs predication directly in VPlan.
+ bool IsPredicated = RecipeBuilder.isPredicatedInst(I);
+ if (IsPredicated && !CostCtx.Config.isLegalMaskedLoadOrStore(
+ IsLoad, ScalarTy, getLoadStoreAlignment(I),
+ getLoadStoreAddressSpace(I)))
+ return false;
+
VPBuilder Builder(VPI);
VPSingleDefRecipe *VectorPtr = Builder.createConsecutiveVectorPointer(
Ptr, ScalarTy, Reverse, VPI->getDebugLoc());
+
+ VPValue *Mask = IsPredicated ? VPI->getMask() : nullptr;
+ // Reverse the mask so it matches the reversed access order.
+ if (Reverse && Mask)
+ Mask = Builder.createNaryOp(VPInstruction::Reverse, Mask,
+ VPI->getDebugLoc());
+
if (IsLoad) {
VPSingleDefRecipe *Load = Builder.createWidenLoad(
- *cast<LoadInst>(I), VectorPtr,
- /*Mask=*/nullptr,
+ *cast<LoadInst>(I), VectorPtr, Mask,
/*Consecutive=*/true, *VPI, VPI->getDebugLoc());
// Reverse the loaded values back into program order.
if (Reverse)
@@ -5509,8 +5521,8 @@ void VPlanTransforms::makeMemOpWideningDecisions(VPlan &Plan, VFRange &Range,
VPI->getDebugLoc());
auto *StoreR = Builder.createWidenStore(
- *cast<StoreInst>(I), VectorPtr, StoredVal,
- /*Mask=*/nullptr, /*Consecutive=*/true, *VPI, VPI->getDebugLoc());
+ *cast<StoreInst>(I), VectorPtr, StoredVal, Mask,
+ /*Consecutive=*/true, *VPI, VPI->getDebugLoc());
return ReplaceWith(VPI, StoreR);
});
diff --git a/llvm/test/Transforms/LoopVectorize/VPlan/AArch64/vplan-memory-op-decisions.ll b/llvm/test/Transforms/LoopVectorize/VPlan/AArch64/vplan-memory-op-decisions.ll
index b396b09d187c7..b3971a73a010a 100644
--- a/llvm/test/Transforms/LoopVectorize/VPlan/AArch64/vplan-memory-op-decisions.ll
+++ b/llvm/test/Transforms/LoopVectorize/VPlan/AArch64/vplan-memory-op-decisions.ll
@@ -82,7 +82,8 @@ define void @load_feeding_only_mask_not_scalarized(ptr noalias %A, ptr noalias %
; CHECK-EMPTY:
; CHECK-NEXT: then:
; CHECK-NEXT: EMIT ir<%gep.B> = getelementptr ir<%B>, ir<%iv>
-; CHECK-NEXT: EMIT-SCALAR ir<%l.p> = load ir<%gep.B>, ir<%cmp>
+; CHECK-NEXT: vp<[[VP5:%[0-9]+]]> = vector-pointer ir<%gep.B>, ir<1>
+; CHECK-NEXT: WIDEN ir<%l.p> = load vp<[[VP5]]>, ir<%cmp>
; CHECK-NEXT: EMIT store ir<42>, ir<%l.p>, ir<%cmp>
; CHECK-NEXT: Successor(s): latch
; CHECK-EMPTY:
@@ -377,10 +378,12 @@ define void @cond_load_store(ptr noalias %a, ptr noalias %b, ptr noalias %cond,
; CHECK-EMPTY:
; CHECK-NEXT: then:
; CHECK-NEXT: EMIT ir<%gep.a> = getelementptr inbounds ir<%a>, ir<%iv>
-; CHECK-NEXT: EMIT-SCALAR ir<%lv> = load ir<%gep.a>, ir<%cmp>
+; CHECK-NEXT: vp<[[VP5:%[0-9]+]]> = vector-pointer inbounds ir<%gep.a>, ir<1>
+; CHECK-NEXT: WIDEN ir<%lv> = load vp<[[VP5]]>, ir<%cmp>
; CHECK-NEXT: EMIT ir<%add> = add ir<%lv>, ir<1>, ir<%cmp>
; CHECK-NEXT: EMIT ir<%gep.b> = getelementptr inbounds ir<%b>, ir<%iv>
-; CHECK-NEXT: EMIT store ir<%add>, ir<%gep.b>, ir<%cmp>
+; CHECK-NEXT: vp<[[VP6:%[0-9]+]]> = vector-pointer inbounds ir<%gep.b>, ir<1>
+; CHECK-NEXT: WIDEN store vp<[[VP6]]>, ir<%add>, ir<%cmp>
; CHECK-NEXT: Successor(s): latch
; CHECK-EMPTY:
; CHECK-NEXT: latch:
@@ -420,3 +423,83 @@ latch:
exit:
ret void
}
+
+; A reverse (stride -1) consecutive load and store guarded by a condition. The
+; mask must be reversed as well, to match the reversed access order.
+define void @cond_reverse_load_store(ptr noalias %a, ptr noalias %b, ptr noalias %cond) {
+; CHECK-LABEL: VPlan for loop in 'cond_reverse_load_store'
+; CHECK: VPlan ' for UF>=1' {
+; CHECK-NEXT: Live-in vp<[[VP0:%[0-9]+]]> = VF
+; CHECK-NEXT: Live-in vp<[[VP1:%[0-9]+]]> = VF * UF
+; CHECK-NEXT: Live-in vp<[[VP2:%[0-9]+]]> = vector-trip-count
+; CHECK-NEXT: Live-in ir<1023> = original trip-count
+; CHECK-EMPTY:
+; CHECK-NEXT: ir-bb<entry>:
+; CHECK-NEXT: Successor(s): scalar.ph, vector.ph
+; CHECK-EMPTY:
+; CHECK-NEXT: vector.ph:
+; CHECK-NEXT: Successor(s): vector loop
+; CHECK-EMPTY:
+; CHECK-NEXT: <x1> vector loop: {
+; CHECK-NEXT: vp<[[VP3:%[0-9]+]]> = CANONICAL-IV
+; CHECK-EMPTY:
+; CHECK-NEXT: vector.body:
+; CHECK-NEXT: ir<%iv> = WIDEN-INDUCTION nsw ir<1023>, ir<-1>, vp<[[VP0]]>
+; CHECK-NEXT: EMIT ir<%gep.cond> = getelementptr inbounds ir<%cond>, ir<%iv>
+; CHECK-NEXT: vp<[[VP4:%[0-9]+]]> = vector-end-pointer inbounds ir<%gep.cond>, vp<[[VP0]]>
+; CHECK-NEXT: WIDEN ir<%c> = load vp<[[VP4]]>
+; CHECK-NEXT: EMIT vp<[[VP5:%[0-9]+]]> = reverse ir<%c>
+; CHECK-NEXT: EMIT ir<%cmp> = icmp sgt vp<[[VP5]]>, ir<0>
+; CHECK-NEXT: Successor(s): then
+; CHECK-EMPTY:
+; CHECK-NEXT: then:
+; CHECK-NEXT: EMIT ir<%gep.a> = getelementptr inbounds ir<%a>, ir<%iv>
+; CHECK-NEXT: vp<[[VP6:%[0-9]+]]> = vector-end-pointer inbounds ir<%gep.a>, vp<[[VP0]]>
+; CHECK-NEXT: EMIT vp<[[VP7:%[0-9]+]]> = reverse ir<%cmp>
+; CHECK-NEXT: WIDEN ir<%lv> = load vp<[[VP6]]>, vp<[[VP7]]>
+; CHECK-NEXT: EMIT vp<[[VP8:%[0-9]+]]> = reverse ir<%lv>
+; CHECK-NEXT: EMIT ir<%add> = add vp<[[VP8]]>, ir<1>, ir<%cmp>
+; CHECK-NEXT: EMIT ir<%gep.b> = getelementptr inbounds ir<%b>, ir<%iv>
+; CHECK-NEXT: vp<[[VP9:%[0-9]+]]> = vector-end-pointer inbounds ir<%gep.b>, vp<[[VP0]]>
+; CHECK-NEXT: EMIT vp<[[VP10:%[0-9]+]]> = reverse ir<%cmp>
+; CHECK-NEXT: EMIT vp<[[VP11:%[0-9]+]]> = reverse ir<%add>
+; CHECK-NEXT: WIDEN store vp<[[VP9]]>, vp<[[VP11]]>, vp<[[VP10]]>
+; CHECK-NEXT: Successor(s): latch
+; CHECK-EMPTY:
+; CHECK-NEXT: latch:
+; CHECK-NEXT: EMIT ir<%iv.next> = add nsw ir<%iv>, ir<-1>
+; CHECK-NEXT: EMIT ir<%ec> = icmp eq ir<%iv.next>, ir<0>
+; CHECK-NEXT: EMIT vp<%index.next> = add nuw vp<[[VP3]]>, vp<[[VP1]]>
+; CHECK-NEXT: EMIT branch-on-count vp<%index.next>, vp<[[VP2]]>
+; CHECK-NEXT: No successors
+; CHECK-NEXT: }
+; CHECK-NEXT: Successor(s): middle.block
+; CHECK-EMPTY:
+; CHECK-NEXT: middle.block:
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 1023, %entry ], [ %iv.next, %latch ]
+ %gep.cond = getelementptr inbounds i32, ptr %cond, i64 %iv
+ %c = load i32, ptr %gep.cond, align 4
+ %cmp = icmp sgt i32 %c, 0
+ br i1 %cmp, label %then, label %latch
+
+then:
+ %gep.a = getelementptr inbounds i32, ptr %a, i64 %iv
+ %lv = load i32, ptr %gep.a, align 4
+ %add = add i32 %lv, 1
+ %gep.b = getelementptr inbounds i32, ptr %b, i64 %iv
+ store i32 %add, ptr %gep.b, align 4
+ br label %latch
+
+latch:
+ %iv.next = add nsw i64 %iv, -1
+ %ec = icmp eq i64 %iv.next, 0
+ br i1 %ec, label %exit, label %loop
+
+exit:
+ ret void
+}
More information about the llvm-commits
mailing list