[llvm] [LV] Use wide lane masks as the canonical form when tail-folding & interleaving (PR #209484)
Kerry McLaughlin via llvm-commits
llvm-commits at lists.llvm.org
Tue Jul 28 03:33:41 PDT 2026
================
@@ -2035,103 +2035,41 @@ static bool isConditionTrueViaVFAndUF(VPValue *Cond, VPlan &Plan,
return SE.isKnownPredicate(CmpInst::ICMP_EQ, VectorTripCount, C);
}
-/// Try to replace multiple active lane masks used for control flow with
-/// a single, wide active lane mask instruction followed by multiple
-/// extract subvector intrinsics. This applies to the active lane mask
-/// instructions both in the loop and in the preheader.
-/// Incoming values of all ActiveLaneMaskPHIs are updated to use the
-/// new extracts from the first active lane mask, which has it's last
-/// operand (multiplier) set to UF.
-static bool tryToReplaceALMWithWideALM(VPlan &Plan, ElementCount VF,
- unsigned UF) {
- if (!EnableWideActiveLaneMask || !VF.isVector() || UF == 1)
+static bool replaceMaskWithCompare(VPlan &Plan, ElementCount BestVF) {
+ if (!BestVF.isScalar())
return false;
+ bool MadeChange = false;
+ VPBuilder Builder;
VPRegionBlock *VectorRegion = Plan.getVectorLoopRegion();
+ VPBasicBlock *PreheaderVPBB = Plan.getVectorPreheader();
VPBasicBlock *ExitingVPBB = VectorRegion->getExitingBasicBlock();
- auto *Term = &ExitingVPBB->back();
- using namespace llvm::VPlanPatternMatch;
- if (!match(Term, m_BranchOnCond(m_Not(m_ActiveLaneMask(
- m_VPValue(), m_VPValue(), m_VPValue())))))
- return false;
+ VPValue *Start, *TC;
+ uint64_t Idx;
+ for (VPBasicBlock *VPBB : {PreheaderVPBB, ExitingVPBB}) {
+ for (VPRecipeBase &R : make_early_inc_range(*VPBB)) {
+ if (!match(&R, m_ExtractVectorForPart(
+ m_WideActiveLaneMask(m_VPValue(Start), m_VPValue(TC),
+ m_VPValue()),
+ m_ConstantInt(Idx))))
+ continue;
- auto *Header = cast<VPBasicBlock>(VectorRegion->getEntry());
- LLVMContext &Ctx = Plan.getContext();
+ auto *Extract = cast<VPInstruction>(&R);
+ Builder.setInsertPoint(Extract);
- auto ExtractFromALM = [&](VPInstruction *ALM,
- SmallVectorImpl<VPValue *> &Extracts) {
- DebugLoc DL = ALM->getDebugLoc();
- for (unsigned Part = 0; Part < UF; ++Part) {
- SmallVector<VPValue *> Ops;
- Ops.append({ALM, Plan.getConstantInt(64, VF.getKnownMinValue() * Part)});
- auto *Ext =
- new VPWidenIntrinsicRecipe(Intrinsic::vector_extract, Ops,
- IntegerType::getInt1Ty(Ctx), {}, {}, DL);
- Extracts[Part] = Ext;
- Ext->insertAfter(ALM);
- }
- };
+ if (Idx > 0)
----------------
kmclaughlin-arm wrote:
Thanks for clarifying! I tried changing the transform to replace the extracts with `ActiveLaneMask` rather than a compare, however one of the tests seems to miss an optimisation after. When lowering directly to icmp, `simplifyRecipes` is able to fold one of the compares to true in pr73894.ll, which no longer happens with `ActiveLaneMask` i.e.
```
+; CHECK-NEXT: [[SCALAR_ACTIVE_LANE_MASK:%.*]] = icmp ult i64 0, [[UMAX]]
; CHECK-NEXT: [[ACTIVE_LANE_MASK_ENTRY1:%.*]] = icmp ult i64 1, [[UMAX]]
; CHECK-NEXT: br label [[VECTOR_BODY:%.*]]
; CHECK: vector.body:
; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, [[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], [[PRED_LOAD_CONTINUE5:%.*]] ]
-; CHECK-NEXT: [[ACTIVE_LANE_MASK:%.*]] = phi i1 [ true, [[VECTOR_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], [[PRED_LOAD_CONTINUE5]] ]
+; CHECK-NEXT: [[ACTIVE_LANE_MASK:%.*]] = phi i1 [ [[SCALAR_ACTIVE_LANE_MASK]], [[VECTOR_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], [[PRED_LOAD_CONTINUE5]] ]
```
https://github.com/llvm/llvm-project/pull/209484
More information about the llvm-commits
mailing list