[llvm] [LV] Vectorize predicated loop-invariant loads via masked load + broadcast (PR #200361)
Ramkumar Ramachandra via llvm-commits
llvm-commits at lists.llvm.org
Thu Sep 10 06:55:53 PDT 2026
================
@@ -3847,6 +3847,116 @@ void VPlanTransforms::hoistPredicatedLoads(VPlan &Plan,
}
}
+static cl::opt<bool> EnableMaskedInvariantLoad(
+ "enable-masked-invariant-load", cl::init(false), cl::ReallyHidden,
+ cl::desc("Rewrite predicated loads from loop-invariant addresses as a "
+ "single-lane masked load + broadcast"));
+
+void VPlanTransforms::widenPredicatedInvariantLoads(VPlan &Plan,
+ VPCostContext &CostCtx) {
+ if (!EnableMaskedInvariantLoad)
+ return;
+
+ VPRegionBlock *LoopRegion = Plan.getVectorLoopRegion();
+ if (!LoopRegion)
+ return;
+
+ // Collect masked, vector-producing replicate loads whose pointer is defined
+ // outside the loop region and which the target can lower as a masked load.
+ SmallVector<VPReplicateRecipe *> Worklist;
+ for (VPBasicBlock *VPBB : VPBlockUtils::blocksOnly<VPBasicBlock>(
+ vp_depth_first_shallow(LoopRegion->getEntry()))) {
+ for (VPRecipeBase &R : *VPBB) {
+ auto *RepR = dyn_cast<VPReplicateRecipe>(&R);
+ if (!RepR || !RepR->isPredicated() || RepR->isSingleScalar())
+ continue;
+ if (RepR->getOpcode() != Instruction::Load)
+ continue;
+ VPValue *Addr = RepR->getOperand(0);
+ if (!Addr->isDefinedOutsideLoopRegions())
+ continue;
+ // Prefer information from the recipe; alignment/address space are only
+ // available on the underlying load.
+ auto *LI = cast<LoadInst>(RepR->getUnderlyingInstr());
+ if (!ForceTargetSupportsMaskedMemoryOps &&
+ !CostCtx.TTI.isLegalMaskedLoad(RepR->getScalarType(), LI->getAlign(),
+ LI->getPointerAddressSpace()))
+ continue;
+ Worklist.push_back(RepR);
+ }
+ }
+
+ for (VPReplicateRecipe *RepR : Worklist) {
+ auto *LI = cast<LoadInst>(RepR->getUnderlyingInstr());
+ LLVMContext &Ctx = LI->getContext();
+ Type *LoadTy = RepR->getScalarType();
+ Align Alignment = LI->getAlign();
+ unsigned AS = LI->getPointerAddressSpace();
+ Type *I1Ty = Type::getInt1Ty(Ctx);
+
+ // Only rewrite when the masked-load + broadcast lowering is no more
+ // expensive than the scalarized predicated load for every vector VF.
+ bool Beneficial = true;
+ for (ElementCount VF : Plan.vectorFactors()) {
+ if (!VF.isVector())
+ continue;
+ auto *MaskTy = cast<VectorType>(toVectorTy(I1Ty, VF));
+ auto *VecTy = cast<VectorType>(toVectorTy(LoadTy, VF));
+ InstructionCost NewCost =
+ VPInstruction::computeAnyOfCost(MaskTy, CostCtx) +
+ VPWidenLoadRecipe::computeMaskedCost(Instruction::Load, VecTy,
+ Alignment, AS, CostCtx) +
+ VPInstruction::computeExtractElementCost(VecTy, CostCtx) +
+ VPInstruction::computeShuffleCost(TargetTransformInfo::SK_Broadcast,
+ VecTy, CostCtx);
+ InstructionCost OldCost = RepR->computeCost(VF, CostCtx);
+ if (NewCost > OldCost) {
+ Beneficial = false;
+ break;
+ }
+ }
+ if (!Beneficial)
+ continue;
+
+ DebugLoc DL = RepR->getDebugLoc();
+ VPValue *Addr = RepR->getOperand(0);
+ VPValue *Mask = RepR->getMask();
+
+ VPBuilder Builder(RepR);
+
+ // any = AnyOf(mask). Unroll appends per-part masks, so this becomes a
+ // single OR-reduction across all UF parts.
+ VPValue *AnyActive = Builder.createNaryOp(VPInstruction::AnyOf, {Mask}, DL);
+
+ // Build a mask with `any` in lane 0 and false elsewhere:
+ // vmsk = insertelement <VF x i1> zeroinitializer, any, 0
+ VPValue *FalseI1 = Plan.getOrAddLiveIn(ConstantInt::getFalse(I1Ty));
----------------
artagnon wrote:
```suggestion
VPValue *FalseI1 = Plan.getFalse();
```
https://github.com/llvm/llvm-project/pull/200361
More information about the llvm-commits
mailing list