[llvm] [LV] Vectorize early exit loops with stores using masking (PR #178454)
Graham Hunter via llvm-commits
llvm-commits at lists.llvm.org
Wed Jun 3 05:59:02 PDT 2026
================
@@ -4184,19 +4199,180 @@ void VPlanTransforms::convertToConcreteRecipes(VPlan &Plan) {
R->eraseFromParent();
}
-void VPlanTransforms::handleUncountableEarlyExits(VPlan &Plan,
- VPBasicBlock *HeaderVPBB,
- VPBasicBlock *LatchVPBB,
- VPBasicBlock *MiddleVPBB,
- UncountableExitStyle Style) {
- struct EarlyExitInfo {
- VPBasicBlock *EarlyExitingVPBB;
- VPIRBasicBlock *EarlyExitVPBB;
- VPValue *CondToExit;
- };
+struct EarlyExitInfo {
+ VPBasicBlock *EarlyExitingVPBB;
+ VPIRBasicBlock *EarlyExitVPBB;
+ VPValue *CondToExit;
+};
+
+/// Update \p Plan to mask memory operations in the loop based on whether the
+/// early exit is taken or not.
+static bool handleUncountableExitsWithSideEffects(
+ VPlan &Plan, SmallVectorImpl<EarlyExitInfo> &Exits,
+ VPBasicBlock *HeaderVPBB, VPBasicBlock *LatchVPBB, VPBasicBlock *MiddleVPBB,
+ Loop *TheLoop, PredicatedScalarEvolution &PSE, DominatorTree &DT,
+ AssumptionCache *AC, VPDominatorTree &VPDT) {
+
+ // Disconnect early exiting blocks from successors, remove branches. We
+ // currently don't support multiple uses for recipes involved in creating
+ // the uncountable exit condition.
+ for (auto &Exit : Exits) {
+ if (Exit.EarlyExitingVPBB == LatchVPBB)
+ continue;
+
+ for (VPRecipeBase &R : Exit.EarlyExitVPBB->phis())
+ cast<VPIRPhi>(&R)->removeIncomingValueFor(Exit.EarlyExitingVPBB);
+ Exit.EarlyExitingVPBB->getTerminator()->eraseFromParent();
+ VPBlockUtils::disconnectBlocks(Exit.EarlyExitingVPBB, Exit.EarlyExitVPBB);
+ }
+
+ // We can abandon a VPlan entirely if we return false here, so we shouldn't
+ // crash if some earlier assumptions on scalar IR don't hold for the vplan
+ // version of the loop.
+ SmallVector<VPInstruction *, 2> GEPs;
+ SmallVector<VPInstruction *, 8> ConditionRecipes;
+
+ std::optional<VPValue *> Cond =
+ vputils::getRecipesForUncountableExit(ConditionRecipes, GEPs, LatchVPBB);
+ if (!Cond)
+ return false;
+
+ // Find load contributing to condition.
+ VPRecipeBase *CondLoad = nullptr;
+ for (auto *Recipe : ConditionRecipes) {
+ if (match(Recipe, m_VPInstruction<Instruction::Load>(m_VPValue()))) {
+ // TODO: Support more than one load. Needs legality updates too.
+ assert(CondLoad == nullptr && "Too many condition loads");
+ CondLoad = Recipe;
+ }
+ }
+ assert(CondLoad && "Couldn't find load");
+
+ // Ensure that we are guaranteed to be able to dereference the memory used
+ // for determining the uncountable exit for the maximum possible number of
+ // scalar iterations of the loop.
+ //
+ // TODO: Support first-faulting loads in cases where we don't know whether
+ // all possible addresses are dereferenceable.
+ {
+ SmallVector<const SCEVPredicate *, 4> Predicates;
+ VPSingleDefRecipe *Load = cast<VPSingleDefRecipe>(CondLoad);
+ VPValue *Ptr = Load->getOperand(0);
+ const SCEV *PtrSCEV = vputils::getSCEVExprForVPValue(Ptr, PSE, TheLoop);
+ const DataLayout &DL = Plan.getDataLayout();
+ VPTypeAnalysis TypeInfo(Plan);
+ APInt EltSize(
+ DL.getIndexTypeSizeInBits(TypeInfo.inferScalarType(Ptr)),
+ DL.getTypeStoreSize(TypeInfo.inferScalarType(Load)).getFixedValue());
+ if (!isDereferenceableAndAlignedInLoop(
+ PtrSCEV, cast<LoadInst>(Load->getUnderlyingInstr())->getAlign(),
+ PSE.getSE()->getConstant(EltSize), TheLoop, *PSE.getSE(), DT, AC,
+ &Predicates))
+ return false;
+ }
+
+ // Check GEPs to see if we can link them to a widen IV recipe with a step of
+ // 1; we're only interested in contiguous accesses for the condition load
+ // right now.
+ for (auto *GEP : GEPs) {
+ VPValue *MaybeIV = nullptr;
+ if (!match(GEP, m_VPInstruction<Instruction::GetElementPtr>(
+ m_LiveIn(), m_VPValue(MaybeIV))))
+ return false;
+
+ auto *WIV = dyn_cast<VPWidenInductionRecipe>(MaybeIV);
+ if (!WIV)
+ return false;
+
+ if (!match(WIV->getStartValue(), m_SpecificInt(0)) ||
+ !match(WIV->getStepValue(), m_SpecificInt(1)))
+ return false;
+ }
+
+ // Find an insertion point. Default to the end of the header but override
+ // if we find a memory op that needs masking before the condition load.
+ auto InsertIt = HeaderVPBB->end();
+ VPRecipeBase *CondR = (*Cond)->getDefiningRecipe();
+ bool CondMoveNeeded = CondR->getParent() != HeaderVPBB;
+ for (VPRecipeBase &R : *HeaderVPBB) {
+ if (&R == CondLoad)
+ continue;
+
+ if (R.mayReadOrWriteMemory()) {
+ if (!VPDT.properlyDominates(CondR, &R)) {
+ CondMoveNeeded = true;
+ InsertIt = R.getIterator();
+ }
+ break;
+ }
+ }
+
+ // If another memory operation would take place before the comparison to
+ // determine whether to exit early or the comparison doesn't take place in
+ // the header, move the comparison (and supporting recipes).
+ if (CondMoveNeeded) {
+ for (auto *Recipe : reverse(ConditionRecipes))
+ Recipe->moveBefore(*HeaderVPBB, InsertIt);
+
+ CondR->moveBefore(*HeaderVPBB, InsertIt);
+ }
+
+ // Create a mask to represent all lanes that fully execute in the vector loop,
+ // stopping short of any early exit.
+ VPBuilder MaskBuilder(HeaderVPBB, InsertIt);
+ VPValue *FirstActive = MaskBuilder.createFirstActiveLane(*Cond);
+ VPTypeAnalysis TypeInfo(Plan);
+ VPValue *IV = cast<VPSingleDefRecipe>(&HeaderVPBB->front());
+ Type *IVScalarTy = TypeInfo.inferScalarType(IV);
+ Type *FirstActiveTy = TypeInfo.inferScalarType(FirstActive);
+ VPValue *ALMMultiplier = Plan.getConstantInt(IVScalarTy, 1);
+ VPValue *Zero = Plan.getZero(IVScalarTy);
+ FirstActive = MaskBuilder.createScalarZExtOrTrunc(FirstActive, IVScalarTy,
+ FirstActiveTy, DebugLoc());
+ VPValue *Mask = MaskBuilder.createNaryOp(VPInstruction::ActiveLaneMask,
+ {Zero, FirstActive, ALMMultiplier},
+ DebugLoc(), "uncountable.exit.mask");
+
+ // Convert all other memory operations to use the mask.
+ for (VPBasicBlock *VPBB : vp_rpo_plain_cfg_loop_body(HeaderVPBB))
+ for (VPRecipeBase &R : *VPBB)
+ if (R.mayReadOrWriteMemory() && &R != CondLoad) {
+ // TODO: Handle conditional memory operations in the loop.
+ if (!VPDT.dominates(R.getParent(), LatchVPBB))
+ return false;
+ cast<VPInstruction>(&R)->addMask(Mask);
+ }
+ // Update middle block branch to compare (IV + however many lanes were active)
+ // against the full trip count, since we may be exiting the vector loop early.
+ // If we didn't take an early exit, we should get the equivalent of VF from
+ // the FirstActiveLane.
+ VPBuilder MiddleBuilder(MiddleVPBB, MiddleVPBB->end());
+ VPValue *ScalarIV = MiddleBuilder.createNaryOp(VPInstruction::ExtractLane,
+ {Zero, IV}, DebugLoc());
+ VPValue *ExitIV = MiddleBuilder.createAdd(ScalarIV, FirstActive);
+ VPValue *FullTC =
+ MiddleBuilder.createICmp(CmpInst::ICMP_EQ, ExitIV, Plan.getTripCount());
+ MiddleBuilder.createNaryOp(VPInstruction::BranchOnCond, {FullTC});
+
+ // Update resume phi in scalar.ph.
+ VPBasicBlock *ScalarPH = Plan.getScalarPreheader();
+ auto Phis = ScalarPH->phis();
+ // TODO: Handle more than one Phi; re-derive from IV.
+ // TODO: Handle reductions.
+ if (range_size(Phis) != 1)
+ return false;
+ VPPhi *ContinueIV = cast<VPPhi>(Phis.begin());
+ ContinueIV->setOperand(0, ExitIV);
----------------
huntergr-arm wrote:
No; the matcher calls `VPWidenIntOrFpInductionRecipe::isCanonical()`, which in turn tries to match the type of the recipe against the canonical IV type from the region. Since the region is created after handling early exits, we get a null pointer then try to dereference it.
https://github.com/llvm/llvm-project/pull/178454
More information about the llvm-commits
mailing list