[llvm] [LoopVectorize] Widen selected-base loads with masked loads (PR #213227)

via llvm-commits llvm-commits at lists.llvm.org
Tue Aug 4 00:03:39 PDT 2026


https://github.com/OHNope updated https://github.com/llvm/llvm-project/pull/213227

>From 39dc61fe9d233cc522882ae67fdb984bcf67fb0d Mon Sep 17 00:00:00 2001
From: OHNope <ohnopexboy at gmail.com>
Date: Fri, 31 Jul 2026 17:56:04 +0900
Subject: [PATCH 1/3] [LoopVectorize] Widen selected-base loads with masked
 loads

---
 .../Vectorize/LoopVectorizationPlanner.h      |  15 +-
 llvm/lib/Transforms/Vectorize/VPlan.h         |   8 +-
 .../Transforms/Vectorize/VPlanTransforms.cpp  | 189 ++++++++++++++
 .../X86/selected-base-masked-load.ll          | 235 ++++++++++++++++++
 4 files changed, 438 insertions(+), 9 deletions(-)
 create mode 100644 llvm/test/Transforms/LoopVectorize/X86/selected-base-masked-load.ll

diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
index fa317f290022e..9cf41bb3e7001 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
@@ -520,13 +520,14 @@ class VPBuilder {
   }
 
   /// Create a recipe widening \p Load, loading from \p Addr with \p Mask (may
-  /// be null).
-  VPWidenLoadRecipe *createWidenLoad(LoadInst &Load, VPValue *Addr,
-                                     VPValue *Mask, bool Consecutive,
-                                     const VPIRMetadata &Metadata,
-                                     DebugLoc DL) {
-    return tryInsertInstruction(
-        new VPWidenLoadRecipe(Load, Addr, Mask, Consecutive, Metadata, DL));
+  /// be null). Use \p Alignment instead of the ingredient's alignment when it
+  /// is provided.
+  VPWidenLoadRecipe *
+  createWidenLoad(LoadInst &Load, VPValue *Addr, VPValue *Mask,
+                  bool Consecutive, const VPIRMetadata &Metadata, DebugLoc DL,
+                  std::optional<Align> Alignment = std::nullopt) {
+    return tryInsertInstruction(new VPWidenLoadRecipe(
+        Load, Addr, Mask, Consecutive, Metadata, DL, Alignment));
   }
 
   /// Create a recipe widening \p Store, storing \p StoredVal to \p Addr with
diff --git a/llvm/lib/Transforms/Vectorize/VPlan.h b/llvm/lib/Transforms/Vectorize/VPlan.h
index 814b77a96e825..ed8d7ac2525a2 100644
--- a/llvm/lib/Transforms/Vectorize/VPlan.h
+++ b/llvm/lib/Transforms/Vectorize/VPlan.h
@@ -3798,16 +3798,20 @@ class LLVM_ABI_FOR_TEST VPWidenMemoryRecipe : public VPIRMetadata {
 struct LLVM_ABI_FOR_TEST VPWidenLoadRecipe final : public VPSingleDefRecipe,
                                                    public VPWidenMemoryRecipe {
   VPWidenLoadRecipe(LoadInst &Load, VPValue *Addr, VPValue *Mask,
-                    bool Consecutive, const VPIRMetadata &Metadata, DebugLoc DL)
+                    bool Consecutive, const VPIRMetadata &Metadata, DebugLoc DL,
+                    std::optional<Align> Alignment = std::nullopt)
       : VPSingleDefRecipe(VPRecipeBase::VPWidenLoadSC, {Addr}, Load.getType(),
                           &Load, DL),
         VPWidenMemoryRecipe(Load, Consecutive, Metadata) {
+    if (Alignment)
+      this->Alignment = *Alignment;
     setMask(Mask);
   }
 
   VPWidenLoadRecipe *clone() override {
     return new VPWidenLoadRecipe(cast<LoadInst>(Ingredient), getAddr(),
-                                 getMask(), Consecutive, *this, getDebugLoc());
+                                 getMask(), Consecutive, *this, getDebugLoc(),
+                                 getAlign());
   }
 
   VP_CLASSOF_IMPL(VPRecipeBase::VPWidenLoadSC);
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index 348dbe514a56e..7db8358a430f5 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -57,6 +57,93 @@ static std::optional<int64_t> getConstantStride(VPValue *Addr, Type *AccessTy,
   return getStrideFromAddRec(AddRec, L, AccessTy, /*Ptr=*/nullptr, PSE);
 }
 
+namespace {
+
+struct SelectedBaseLoadPattern {
+  VPInstruction *Addr;
+  SmallVector<VPValue *> GEPOperands;
+  VPValue *Cond;
+  VPValue *TrueBase;
+  VPValue *FalseBase;
+};
+
+/// Match a load through a GEP whose base pointer is selected per iteration
+/// between two loop-invariant candidates:
+///
+///   load (gep (select Cond, TrueBase, FalseBase), Indices...)
+///
+/// Legality, stride and profitability are checked separately by the caller.
+static std::optional<SelectedBaseLoadPattern>
+matchSelectedBaseLoad(VPInstruction *Load) {
+  if (Load->getOpcode() != Instruction::Load)
+    return std::nullopt;
+
+  // exclude volatile load, atomic load, other uneditable memory operations
+  auto *LI = dyn_cast_or_null<LoadInst>(Load->getUnderlyingInstr());
+  if (!LI || !LI->isSimple())
+    return std::nullopt;
+
+  VPValue *Addr = Load->getOperand(0);
+  auto *AddrI = dyn_cast<VPInstruction>(Addr);
+  if (!AddrI)
+    return std::nullopt;
+
+  Type *SourceElementTy = nullptr;
+  ArrayRef<VPValue *> GEPOperands;
+  // pattern with 3 params : cond + trueBase + falseBase
+  if (!match(Addr, m_GetElementPtr(SourceElementTy, GEPOperands)) ||
+      GEPOperands.size() < 2)
+    return std::nullopt;
+
+  VPValue *Cond = nullptr;
+  VPValue *TrueBase = nullptr;
+  VPValue *FalseBase = nullptr;
+  if (!match(GEPOperands.front(), m_Select(m_VPValue(Cond), m_VPValue(TrueBase),
+                                           m_VPValue(FalseBase))))
+    return std::nullopt;
+
+  if (!Cond->getScalarType()->isIntegerTy(1) || TrueBase == FalseBase)
+    return std::nullopt;
+
+  return SelectedBaseLoadPattern{AddrI, SmallVector<VPValue *>(GEPOperands),
+                                 Cond, TrueBase, FalseBase};
+}
+
+static InstructionCost getBlendedLoadCost(const TargetTransformInfo &TTI,
+                                          Type *VecTy, Align Alignment,
+                                          unsigned AddressSpace,
+                                          bool HasActiveMask,
+                                          TTI::TargetCostKind CostKind) {
+  Type *MaskTy = CmpInst::makeCmpResultType(VecTy);
+  InstructionCost Cost =
+      2 * TTI.getMemIntrinsicInstrCost(
+              MemIntrinsicCostAttributes(Intrinsic::masked_load, VecTy,
+                                         Alignment, AddressSpace),
+              CostKind) +
+      TTI.getArithmeticInstrCost(Instruction::Xor, MaskTy, CostKind) +
+      TTI.getCmpSelInstrCost(Instruction::Select, VecTy, MaskTy,
+                             CmpInst::BAD_ICMP_PREDICATE, CostKind);
+  if (HasActiveMask)
+    Cost += 2 * TTI.getArithmeticInstrCost(Instruction::And, MaskTy, CostKind);
+  return Cost;
+}
+
+static InstructionCost getGatherLoadCost(const LoadInst &LI, ElementCount VF,
+                                         bool HasActiveMask,
+                                         const VPCostContext &Ctx) {
+  Type *VecTy = VectorType::get(LI.getType(), VF);
+  Type *VecPtrTy = VectorType::get(LI.getPointerOperandType(), VF);
+  return Ctx.TTI.getAddressComputationCost(VecPtrTy, nullptr, nullptr,
+                                           Ctx.CostKind) +
+         Ctx.TTI.getMemIntrinsicInstrCost(
+             MemIntrinsicCostAttributes(Intrinsic::masked_gather, VecTy,
+                                        LI.getPointerOperand(), HasActiveMask,
+                                        LI.getAlign(), &LI),
+             Ctx.CostKind);
+}
+
+} // end anonymous namespace
+
 bool VPlanTransforms::tryToConvertVPInstructionsToVPRecipes(
     VPlan &Plan, const TargetLibraryInfo &TLI, PredicatedScalarEvolution &PSE,
     Loop *OuterLoop) {
@@ -5428,6 +5515,108 @@ void VPlanTransforms::makeMemOpWideningDecisions(VPlan &Plan, VFRange &Range,
         return false;
       });
 
+  // Recognize loads through a pointer selected per iteration between two
+  // loop-invariant bases. Although the combined address is not consecutive,
+  // each selected address chain is. When profitable, use two complementary
+  // masked consecutive loads and blend their results instead of gathering a
+  // vector of selected pointers.
+  VPlanTransforms::runPass(
+      "widenSelectedBaseLoads", ProcessSubset, Plan, [&](VPInstruction *VPI) {
+        std::optional<SelectedBaseLoadPattern> Pattern =
+            matchSelectedBaseLoad(VPI);
+        if (!Pattern)
+          return false;
+
+        if (!vputils::isUniformAcrossVFsAndUFs(Pattern->TrueBase) ||
+            !vputils::isUniformAcrossVFsAndUFs(Pattern->FalseBase) ||
+            vputils::isUniformAcrossVFsAndUFs(Pattern->Cond))
+          return false;
+
+        auto *LI = cast<LoadInst>(VPI->getUnderlyingInstr());
+        Type *ScalarTy = LI->getType();
+        // The ingredient's alignment only applies to the address selected by
+        // each scalar iteration. It does not imply that either candidate base
+        // is aligned when that candidate is masked off. Use a conservative
+        // alignment for the newly-speculated base pointers.
+        const Align SplitAlignment(1);
+        if (!CostCtx.Config.isLegalMaskedLoadOrStore(
+                /*IsLoad=*/true, ScalarTy, SplitAlignment,
+                LI->getPointerAddressSpace()))
+          return false;
+
+        VPBuilder Builder(VPI);
+        SmallVector<VPValue *> TrueOps = Pattern->GEPOperands;
+        SmallVector<VPValue *> FalseOps = Pattern->GEPOperands;
+        // replace the baseOp
+        TrueOps[0] = Pattern->TrueBase;
+        FalseOps[0] = Pattern->FalseBase;
+        auto *TrueAddr =
+            Builder.insert(Pattern->Addr->cloneWithOperands(TrueOps));
+        auto *FalseAddr =
+            Builder.insert(Pattern->Addr->cloneWithOperands(FalseOps));
+
+        // Splitting the selected base must expose two unit-stride accesses.
+        // Remove the speculative address recipes again if either proof fails.
+        if (getConstantStride(TrueAddr, ScalarTy, CostCtx.PSE, CostCtx.L) !=
+                1 ||
+            getConstantStride(FalseAddr, ScalarTy, CostCtx.PSE, CostCtx.L) !=
+                1) {
+          TrueAddr->eraseFromParent();
+          FalseAddr->eraseFromParent();
+          return false;
+        }
+
+        bool HasActiveMask = RecipeBuilder.isPredicatedInst(LI);
+        auto IsProfitable = [&](ElementCount VF) {
+          // Do not replace a scalarization decision without comparing against
+          // its cost and updating the legacy scalarization bookkeeping.
+          if (VF.isScalable() || CostCtx.willBeScalarized(LI, VF) ||
+              !CostCtx.Config.isLegalGatherOrScatter(LI, VF))
+            return false;
+          Type *VecTy = VectorType::get(ScalarTy, VF);
+          InstructionCost BlendedCost = getBlendedLoadCost(
+              CostCtx.TTI, VecTy, SplitAlignment, LI->getPointerAddressSpace(),
+              HasActiveMask, CostCtx.CostKind);
+          return BlendedCost <
+                 getGatherLoadCost(*LI, VF, HasActiveMask, CostCtx);
+        };
+        // if fail, then erase the tmp recipes
+        if (!LoopVectorizationPlanner::getDecisionAndClampRange(IsProfitable,
+                                                                Range)) {
+          TrueAddr->eraseFromParent();
+          FalseAddr->eraseFromParent();
+          return false;
+        }
+
+        VPSingleDefRecipe *TrueVectorPtr =
+            Builder.createConsecutiveVectorPointer(
+                TrueAddr, ScalarTy, /*Reverse=*/false, VPI->getDebugLoc());
+        VPSingleDefRecipe *FalseVectorPtr =
+            Builder.createConsecutiveVectorPointer(
+                FalseAddr, ScalarTy, /*Reverse=*/false, VPI->getDebugLoc());
+
+        VPValue *TrueMask = Pattern->Cond;
+        VPValue *FalseMask =
+            Builder.createNot(Pattern->Cond, VPI->getDebugLoc());
+        if (HasActiveMask) {
+          VPValue *ActiveMask = VPI->getMask();
+          TrueMask = Builder.createLogicalAnd(ActiveMask, TrueMask,
+                                              VPI->getDebugLoc());
+          FalseMask = Builder.createLogicalAnd(ActiveMask, FalseMask,
+                                               VPI->getDebugLoc());
+        }
+        // two load + merge
+        auto *TrueLoad = Builder.createWidenLoad(
+            *LI, TrueVectorPtr, TrueMask,
+            /*Consecutive=*/true, *VPI, VPI->getDebugLoc(), SplitAlignment);
+        auto *FalseLoad = Builder.createWidenLoad(
+            *LI, FalseVectorPtr, FalseMask, /*Consecutive=*/true, *VPI,
+            VPI->getDebugLoc(), SplitAlignment);
+        VPInstruction *Blend = Builder.createSelect(
+            Pattern->Cond, TrueLoad, FalseLoad, VPI->getDebugLoc());
+        return ReplaceWith(VPI, Blend);
+      });
+
   if (!RecipeBuilder.prefersVectorizedAddressing()) {
     VPlanTransforms::runPass(
         "makeVPlanMemOpDecision", ProcessSubset, Plan, [&](VPInstruction *VPI) {
diff --git a/llvm/test/Transforms/LoopVectorize/X86/selected-base-masked-load.ll b/llvm/test/Transforms/LoopVectorize/X86/selected-base-masked-load.ll
new file mode 100644
index 0000000000000..db925c80d529e
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/X86/selected-base-masked-load.ll
@@ -0,0 +1,235 @@
+; RUN: opt -passes=loop-vectorize -mtriple=x86_64-unknown-linux-gnu -mcpu=tigerlake \
+; RUN:   -force-vector-width=8 -force-vector-interleave=1 -S %s | FileCheck %s
+; RUN: opt -passes=loop-vectorize -mtriple=x86_64-unknown-linux-gnu -mcpu=tigerlake \
+; RUN:   -force-vector-width=8 -force-vector-interleave=1 \
+; RUN:   -tail-folding-policy=must-fold-tail -force-tail-folding-style=data \
+; RUN:   -S %s | FileCheck %s --check-prefix=TAIL
+; RUN: opt -passes=loop-vectorize -mtriple=x86_64-unknown-linux-gnu -mcpu=tigerlake \
+; RUN:   -force-vector-width=8 -force-vector-interleave=1 -disable-output \
+; RUN:   -vplan-print-after=widenSelectedBaseLoads %s 2>&1 | \
+; RUN:   FileCheck %s --check-prefix=VPLAN
+; RUN: opt -passes=loop-vectorize -mtriple=x86_64-unknown-linux-gnu -mcpu=tigerlake \
+; RUN:   -S %s | FileCheck %s --check-prefix=ISSUE
+; RUN: opt -passes=loop-vectorize -mtriple=x86_64-unknown-linux-gnu -mcpu=tigerlake \
+; RUN:   -force-vector-width=8 -force-vector-interleave=2 -S %s | \
+; RUN:   FileCheck %s --check-prefix=UNROLL
+; RUN: opt -passes=loop-vectorize -mtriple=x86_64-unknown-linux-gnu -mcpu=tigerlake \
+; RUN:   -force-vector-width=8 -force-vector-interleave=1 -disable-output \
+; RUN:   -vplan-print-after=dropPoisonGeneratingRecipes %s 2>&1 | \
+; RUN:   FileCheck %s --check-prefix=VPLAN-DROP
+
+
+; A load from a per-lane selected base and a unit-stride index is cheaper as
+; two complementary masked loads than as a gather on AVX-512 targets.
+
+define void @selected_base_unit_stride(ptr noalias readonly %conditions,
+                                      ptr noalias writeonly %out,
+                                      ptr noalias readonly %a,
+                                      ptr noalias readonly %b, i64 %n) {
+; CHECK-LABEL: @selected_base_unit_stride(
+; CHECK: vector.body:
+; CHECK: [[MASK:%.*]] = icmp sgt <8 x i32> {{.*}}, zeroinitializer
+; CHECK: [[A_PTR:%.*]] = getelementptr i32, ptr %a, i64 {{.*}}
+; CHECK: [[B_PTR:%.*]] = getelementptr i32, ptr %b, i64 {{.*}}
+; CHECK: [[NOT_MASK:%.*]] = xor <8 x i1> [[MASK]], splat (i1 true)
+; CHECK: [[FROM_A:%.*]] = call <8 x i32> @llvm.masked.load.v8i32.p0(ptr align 1 [[A_PTR]], <8 x i1> [[MASK]], <8 x i32> poison)
+; CHECK: [[FROM_B:%.*]] = call <8 x i32> @llvm.masked.load.v8i32.p0(ptr align 1 [[B_PTR]], <8 x i1> [[NOT_MASK]], <8 x i32> poison)
+; CHECK: [[RESULT:%.*]] = select <8 x i1> [[MASK]], <8 x i32> [[FROM_A]], <8 x i32> [[FROM_B]]
+; CHECK-NOT: @llvm.masked.gather
+;
+; TAIL-LABEL: @selected_base_unit_stride(
+; TAIL: [[ACTIVE:%.*]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(
+; TAIL: [[MASK:%.*]] = icmp sgt <8 x i32> {{.*}}, zeroinitializer
+; TAIL: [[NOT_MASK:%.*]] = xor <8 x i1> [[MASK]], splat (i1 true)
+; TAIL: [[ACTIVE_A:%.*]] = select <8 x i1> [[ACTIVE]], <8 x i1> [[MASK]], <8 x i1> zeroinitializer
+; TAIL: [[ACTIVE_B:%.*]] = select <8 x i1> [[ACTIVE]], <8 x i1> [[NOT_MASK]], <8 x i1> zeroinitializer
+; TAIL: [[FROM_A:%.*]] = call <8 x i32> @llvm.masked.load.v8i32.p0(ptr align 1 {{.*}}, <8 x i1> [[ACTIVE_A]], <8 x i32> poison)
+; TAIL: [[FROM_B:%.*]] = call <8 x i32> @llvm.masked.load.v8i32.p0(ptr align 1 {{.*}}, <8 x i1> [[ACTIVE_B]], <8 x i32> poison)
+; TAIL: select <8 x i1> [[MASK]], <8 x i32> [[FROM_A]], <8 x i32> [[FROM_B]]
+;
+; VPLAN-LABEL: VPlan for loop in 'selected_base_unit_stride' after widenSelectedBaseLoads
+; VPLAN: EMIT ir<%src.ptr>.1 = getelementptr inbounds ir<%a>, ir<%iv>
+; VPLAN: EMIT ir<%src.ptr>.2 = getelementptr inbounds ir<%b>, ir<%iv>
+; VPLAN: [[A_PTR:vp<%[0-9]+>]] = vector-pointer inbounds i32, ir<%src.ptr>.1, ir<1>
+; VPLAN: [[B_PTR:vp<%[0-9]+>]] = vector-pointer inbounds i32, ir<%src.ptr>.2, ir<1>
+; VPLAN: [[NOT_MASK:vp<%[0-9]+>]] = not ir<%cmp>
+; VPLAN: WIDEN ir<%value> = load [[A_PTR]], ir<%cmp>
+; VPLAN: WIDEN ir<%value>.1 = load [[B_PTR]], [[NOT_MASK]]
+; VPLAN: EMIT {{.*}} = select ir<%cmp>, ir<%value>, ir<%value>.1
+;
+; Cloning the recipes for multiple parts must preserve the conservative
+; alignment selected by this transform.
+; UNROLL-LABEL: @selected_base_unit_stride(
+; UNROLL-COUNT-4: call <8 x i32> @llvm.masked.load.v8i32.p0(ptr align 1
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %condition.ptr = getelementptr inbounds i32, ptr %conditions, i64 %iv
+  %condition = load i32, ptr %condition.ptr, align 4
+  %cmp = icmp sgt i32 %condition, 0
+  %base = select i1 %cmp, ptr %a, ptr %b
+  %src.ptr = getelementptr inbounds i32, ptr %base, i64 %iv
+  ; The selected address is known to be 8-byte aligned, but that does not imply
+  ; the unselected candidate base has the same alignment.
+  %value = load i32, ptr %src.ptr, align 8
+  %out.ptr = getelementptr inbounds i32, ptr %out, i64 %iv
+  store i32 %value, ptr %out.ptr, align 4
+  %iv.next = add nuw i64 %iv, 1
+  %exit.cond = icmp eq i64 %iv.next, %n
+  br i1 %exit.cond, label %exit, label %loop
+
+exit:
+  ret void
+}
+
+; Fixed-trip-count forms of the two examples from llvm.org/PR206384.
+
+define void @issue_i32(ptr noalias readonly %conditions,
+                       ptr noalias writeonly %out,
+                       ptr noalias readonly %a,
+                       ptr noalias readonly %b) {
+; ISSUE-LABEL: @issue_i32(
+; ISSUE: [[MASK:%.*]] = icmp sgt <8 x i32> {{.*}}, zeroinitializer
+; ISSUE: [[NOT_MASK:%.*]] = xor <8 x i1> [[MASK]], splat (i1 true)
+; ISSUE: call <8 x i32> @llvm.masked.load.v8i32.p0(ptr align 1 %a, <8 x i1> [[MASK]], <8 x i32> poison)
+; ISSUE: call <8 x i32> @llvm.masked.load.v8i32.p0(ptr align 1 %b, <8 x i1> [[NOT_MASK]], <8 x i32> poison)
+; ISSUE-NOT: @llvm.masked.gather
+; ISSUE: ret void
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %condition.ptr = getelementptr inbounds i32, ptr %conditions, i64 %iv
+  %condition = load i32, ptr %condition.ptr, align 4
+  %cmp = icmp sgt i32 %condition, 0
+  %base = select i1 %cmp, ptr %a, ptr %b
+  %src.ptr = getelementptr inbounds i32, ptr %base, i64 %iv
+  %value = load i32, ptr %src.ptr, align 4
+  %out.ptr = getelementptr inbounds i32, ptr %out, i64 %iv
+  store i32 %value, ptr %out.ptr, align 4
+  %iv.next = add nuw i64 %iv, 1
+  %exit.cond = icmp eq i64 %iv.next, 8
+  br i1 %exit.cond, label %exit, label %loop
+
+exit:
+  ret void
+}
+
+define void @issue_f64(ptr noalias readonly %conditions,
+                       ptr noalias writeonly %out,
+                       ptr noalias readonly %a,
+                       ptr noalias readonly %b) {
+; ISSUE-LABEL: @issue_f64(
+; ISSUE: [[MASK:%.*]] = fcmp ogt <4 x double> {{.*}}, zeroinitializer
+; ISSUE: [[NOT_MASK:%.*]] = xor <4 x i1> [[MASK]], splat (i1 true)
+; ISSUE: call <4 x double> @llvm.masked.load.v4f64.p0(ptr align 1 %a, <4 x i1> [[MASK]], <4 x double> poison)
+; ISSUE: call <4 x double> @llvm.masked.load.v4f64.p0(ptr align 1 %b, <4 x i1> [[NOT_MASK]], <4 x double> poison)
+; ISSUE-NOT: @llvm.masked.gather
+; ISSUE: ret void
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %condition.ptr = getelementptr inbounds double, ptr %conditions, i64 %iv
+  %condition = load double, ptr %condition.ptr, align 8
+  %cmp = fcmp ogt double %condition, 0.0
+  %base = select i1 %cmp, ptr %a, ptr %b
+  %src.ptr = getelementptr inbounds double, ptr %base, i64 %iv
+  %value = load double, ptr %src.ptr, align 8
+  %out.ptr = getelementptr inbounds double, ptr %out, i64 %iv
+  store double %value, ptr %out.ptr, align 8
+  %iv.next = add nuw i64 %iv, 1
+  %exit.cond = icmp eq i64 %iv.next, 4
+  br i1 %exit.cond, label %exit, label %loop
+
+exit:
+  ret void
+}
+
+; A non-unit index is not representable by two consecutive masked loads and
+; must keep the gather.
+
+define void @selected_base_stride_two(ptr noalias readonly %conditions,
+                                      ptr noalias writeonly %out,
+                                      ptr noalias readonly %a,
+                                      ptr noalias readonly %b, i64 %n) {
+; CHECK-LABEL: @selected_base_stride_two(
+; CHECK: vector.body:
+; CHECK: call <8 x i32> @llvm.masked.gather.v8i32.v8p0(
+; CHECK-NOT: @llvm.masked.load
+; CHECK: ret void
+;
+; VPLAN-LABEL: VPlan for loop in 'selected_base_stride_two' after widenSelectedBaseLoads
+; VPLAN: EMIT ir<%src.ptr> = getelementptr inbounds ir<%base>, ir<%index>
+; VPLAN-NOT: vector-pointer
+; VPLAN: EMIT-SCALAR ir<%value> = load ir<%src.ptr>
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %condition.ptr = getelementptr inbounds i32, ptr %conditions, i64 %iv
+  %condition = load i32, ptr %condition.ptr, align 4
+  %cmp = icmp sgt i32 %condition, 0
+  %base = select i1 %cmp, ptr %a, ptr %b
+  %index = shl nuw i64 %iv, 1
+  %src.ptr = getelementptr inbounds i32, ptr %base, i64 %index
+  %value = load i32, ptr %src.ptr, align 4
+  %out.ptr = getelementptr inbounds i32, ptr %out, i64 %iv
+  store i32 %value, ptr %out.ptr, align 4
+  %iv.next = add nuw i64 %iv, 1
+  %exit.cond = icmp eq i64 %iv.next, %n
+  br i1 %exit.cond, label %exit, label %loop
+
+exit:
+  ret void
+}
+
+define void @selected_base_inbounds_cleanup(
+    ptr noalias readonly dereferenceable(12) %a,
+    ptr noalias readonly dereferenceable(20) %b,
+    ptr noalias writeonly dereferenceable(32) %out) {
+; CHECK-LABEL: @selected_base_inbounds_cleanup(
+; CHECK: vector.body:
+; CHECK: call <8 x i32> @llvm.masked.load
+; CHECK: call <8 x i32> @llvm.masked.load
+; CHECK-NOT: @llvm.masked.gather
+; CHECK: ret void
+;
+; VPLAN-DROP-LABEL: VPlan for loop in 'selected_base_inbounds_cleanup' after VPlanTransforms::dropPoisonGeneratingRecipes
+; VPLAN-DROP: CLONE ir<%src.ptr>.1 = getelementptr ir<%a.end>, ir<%index>
+; VPLAN-DROP: CLONE ir<%src.ptr>.2 = getelementptr ir<%b>, ir<%index>
+; VPLAN-DROP: [[A_PTR:vp<%[0-9]+>]] = vector-pointer i32, ir<%src.ptr>.1, ir<1>
+; VPLAN-DROP: [[B_PTR:vp<%[0-9]+>]] = vector-pointer i32, ir<%src.ptr>.2, ir<1>
+
+entry:
+  %a.end = getelementptr inbounds i32, ptr %a, i64 3
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+
+  ; iv 0..2 use a.end + (-3..-1)
+  ; iv 3..7 use b + (0..4)
+  %use.a = icmp ult i64 %iv, 3
+  %base = select i1 %use.a, ptr %a.end, ptr %b
+  %index = add i64 %iv, -3
+
+  ; inbounds valid, only load the selected Addr
+  %src.ptr = getelementptr inbounds i32, ptr %base, i64 %index
+  %value = load i32, ptr %src.ptr, align 4
+
+  %out.ptr = getelementptr inbounds i32, ptr %out, i64 %iv
+  store i32 %value, ptr %out.ptr, align 4
+
+  %iv.next = add nuw i64 %iv, 1
+  %done = icmp eq i64 %iv.next, 8
+  br i1 %done, label %exit, label %loop
+
+exit:
+  ret void
+}
\ No newline at end of file

>From 9a01d9ad9768446011b442d2978b2fe20426d1e1 Mon Sep 17 00:00:00 2001
From: OHNope <ohnopexboy at gmail.com>
Date: Fri, 31 Jul 2026 20:48:31 +0900
Subject: [PATCH 2/3] [LoopVectorize] Update VPlan printing test for
 selected-base loads

---
 .../LoopVectorize/VPlan/vplan-print-before-after-all.ll          | 1 +
 1 file changed, 1 insertion(+)

diff --git a/llvm/test/Transforms/LoopVectorize/VPlan/vplan-print-before-after-all.ll b/llvm/test/Transforms/LoopVectorize/VPlan/vplan-print-before-after-all.ll
index a7271524f0191..97c0ed94f2946 100644
--- a/llvm/test/Transforms/LoopVectorize/VPlan/vplan-print-before-after-all.ll
+++ b/llvm/test/Transforms/LoopVectorize/VPlan/vplan-print-before-after-all.ll
@@ -20,6 +20,7 @@
 ; CHECK-BEFORE: VPlan for loop in 'foo' [[BEFORE_OR_AFTER]] VPlanTransforms::makeMemOpWideningDecisions
 ; CHECK: VPlan for loop in 'foo' [[BEFORE_OR_AFTER]] lowerMemoryIdioms
 ; CHECK: VPlan for loop in 'foo' [[BEFORE_OR_AFTER]] scalarizeMemOpsWithIrregularTypes
+; CHECK: VPlan for loop in 'foo' [[BEFORE_OR_AFTER]] widenSelectedBaseLoads
 ; CHECK: VPlan for loop in 'foo' [[BEFORE_OR_AFTER]] widenConsecutiveMemOps
 ; CHECK: VPlan for loop in 'foo' [[BEFORE_OR_AFTER]] delegateMemOpWideningToLegacyCM
 ; CHECK-AFTER: VPlan for loop in 'foo' [[BEFORE_OR_AFTER]] VPlanTransforms::makeMemOpWideningDecisions

>From f184b8b76a5e2c495d191e6bec469d6350838859 Mon Sep 17 00:00:00 2001
From: OHNope <ohnopexboy at gmail.com>
Date: Tue, 4 Aug 2026 15:31:44 +0900
Subject: [PATCH 3/3] [LoopVectorize] Reuse existing costs for selected-base
 loads

---
 .../Transforms/Vectorize/VPlanPatternMatch.h  |  56 +++++++
 .../Transforms/Vectorize/VPlanTransforms.cpp  | 140 ++++--------------
 2 files changed, 87 insertions(+), 109 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/VPlanPatternMatch.h b/llvm/lib/Transforms/Vectorize/VPlanPatternMatch.h
index d575d1f150e7a..b939147289687 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanPatternMatch.h
+++ b/llvm/lib/Transforms/Vectorize/VPlanPatternMatch.h
@@ -1122,6 +1122,62 @@ inline GetElementPtr_match m_GetElementPtr(Type *&SourceElementType,
   return GetElementPtr_match(SourceElementType, Operands);
 }
 
+/// The parts of a load through a pointer selected between two bases:
+///
+///   load (gep (select Cond, TrueBase, FalseBase), Indices...)
+struct SelectedBaseLoadPattern {
+  VPInstruction *Addr = nullptr;
+  SmallVector<VPValue *> GEPOperands;
+  VPValue *Cond = nullptr;
+  VPValue *TrueBase = nullptr;
+  VPValue *FalseBase = nullptr;
+};
+
+/// Match a simple load through a GEP whose base pointer is selected per
+/// iteration between two distinct candidates.
+struct SelectedBaseLoad_match {
+  SelectedBaseLoadPattern &Pattern;
+
+  bool match(const VPInstruction *Load) const {
+    if (Load->getOpcode() != Instruction::Load)
+      return false;
+
+    auto *LI = dyn_cast_or_null<LoadInst>(Load->getUnderlyingInstr());
+    if (!LI || !LI->isSimple())
+      return false;
+
+    Type *SourceElementTy = nullptr;
+    ArrayRef<VPValue *> GEPOperands;
+    if (!VPlanPatternMatch::match(
+            Load->getOperand(0),
+            m_GetElementPtr(SourceElementTy, GEPOperands)) ||
+        GEPOperands.size() < 2)
+      return false;
+
+    VPValue *Cond = nullptr;
+    VPValue *TrueBase = nullptr;
+    VPValue *FalseBase = nullptr;
+    if (!VPlanPatternMatch::match(GEPOperands.front(),
+                                  m_Select(m_VPValue(Cond), m_VPValue(TrueBase),
+                                           m_VPValue(FalseBase))) ||
+        !Cond->getScalarType()->isIntegerTy(1) || TrueBase == FalseBase)
+      return false;
+
+    auto *Addr = dyn_cast<VPInstruction>(Load->getOperand(0));
+    if (!Addr)
+      return false;
+
+    Pattern = {Addr, SmallVector<VPValue *>(GEPOperands), Cond, TrueBase,
+               FalseBase};
+    return true;
+  }
+};
+
+inline SelectedBaseLoad_match
+m_SelectedBaseLoad(SelectedBaseLoadPattern &Pattern) {
+  return {Pattern};
+}
+
 template <typename SubPattern_t> struct OneUse_match {
   SubPattern_t SubPattern;
 
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index a69f8496e2b3a..61a30bfab488b 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -12,6 +12,7 @@
 //===----------------------------------------------------------------------===//
 
 #include "VPlanTransforms.h"
+#include "SLPVectorizer/SLPCostAnalysis.h"
 #include "VPRecipeBuilder.h"
 #include "VPlan.h"
 #include "VPlanAnalysis.h"
@@ -57,93 +58,6 @@ static std::optional<int64_t> getConstantStride(VPValue *Addr, Type *AccessTy,
   return getStrideFromAddRec(AddRec, L, AccessTy, /*Ptr=*/nullptr, PSE);
 }
 
-namespace {
-
-struct SelectedBaseLoadPattern {
-  VPInstruction *Addr;
-  SmallVector<VPValue *> GEPOperands;
-  VPValue *Cond;
-  VPValue *TrueBase;
-  VPValue *FalseBase;
-};
-
-/// Match a load through a GEP whose base pointer is selected per iteration
-/// between two loop-invariant candidates:
-///
-///   load (gep (select Cond, TrueBase, FalseBase), Indices...)
-///
-/// Legality, stride and profitability are checked separately by the caller.
-static std::optional<SelectedBaseLoadPattern>
-matchSelectedBaseLoad(VPInstruction *Load) {
-  if (Load->getOpcode() != Instruction::Load)
-    return std::nullopt;
-
-  // exclude volatile load, atomic load, other uneditable memory operations
-  auto *LI = dyn_cast_or_null<LoadInst>(Load->getUnderlyingInstr());
-  if (!LI || !LI->isSimple())
-    return std::nullopt;
-
-  VPValue *Addr = Load->getOperand(0);
-  auto *AddrI = dyn_cast<VPInstruction>(Addr);
-  if (!AddrI)
-    return std::nullopt;
-
-  Type *SourceElementTy = nullptr;
-  ArrayRef<VPValue *> GEPOperands;
-  // pattern with 3 params : cond + trueBase + falseBase
-  if (!match(Addr, m_GetElementPtr(SourceElementTy, GEPOperands)) ||
-      GEPOperands.size() < 2)
-    return std::nullopt;
-
-  VPValue *Cond = nullptr;
-  VPValue *TrueBase = nullptr;
-  VPValue *FalseBase = nullptr;
-  if (!match(GEPOperands.front(), m_Select(m_VPValue(Cond), m_VPValue(TrueBase),
-                                           m_VPValue(FalseBase))))
-    return std::nullopt;
-
-  if (!Cond->getScalarType()->isIntegerTy(1) || TrueBase == FalseBase)
-    return std::nullopt;
-
-  return SelectedBaseLoadPattern{AddrI, SmallVector<VPValue *>(GEPOperands),
-                                 Cond, TrueBase, FalseBase};
-}
-
-static InstructionCost getBlendedLoadCost(const TargetTransformInfo &TTI,
-                                          Type *VecTy, Align Alignment,
-                                          unsigned AddressSpace,
-                                          bool HasActiveMask,
-                                          TTI::TargetCostKind CostKind) {
-  Type *MaskTy = CmpInst::makeCmpResultType(VecTy);
-  InstructionCost Cost =
-      2 * TTI.getMemIntrinsicInstrCost(
-              MemIntrinsicCostAttributes(Intrinsic::masked_load, VecTy,
-                                         Alignment, AddressSpace),
-              CostKind) +
-      TTI.getArithmeticInstrCost(Instruction::Xor, MaskTy, CostKind) +
-      TTI.getCmpSelInstrCost(Instruction::Select, VecTy, MaskTy,
-                             CmpInst::BAD_ICMP_PREDICATE, CostKind);
-  if (HasActiveMask)
-    Cost += 2 * TTI.getArithmeticInstrCost(Instruction::And, MaskTy, CostKind);
-  return Cost;
-}
-
-static InstructionCost getGatherLoadCost(const LoadInst &LI, ElementCount VF,
-                                         bool HasActiveMask,
-                                         const VPCostContext &Ctx) {
-  Type *VecTy = VectorType::get(LI.getType(), VF);
-  Type *VecPtrTy = VectorType::get(LI.getPointerOperandType(), VF);
-  return Ctx.TTI.getAddressComputationCost(VecPtrTy, nullptr, nullptr,
-                                           Ctx.CostKind) +
-         Ctx.TTI.getMemIntrinsicInstrCost(
-             MemIntrinsicCostAttributes(Intrinsic::masked_gather, VecTy,
-                                        LI.getPointerOperand(), HasActiveMask,
-                                        LI.getAlign(), &LI),
-             Ctx.CostKind);
-}
-
-} // end anonymous namespace
-
 bool VPlanTransforms::tryToConvertVPInstructionsToVPRecipes(
     VPlan &Plan, const TargetLibraryInfo &TLI, PredicatedScalarEvolution &PSE,
     Loop *OuterLoop) {
@@ -5533,14 +5447,13 @@ void VPlanTransforms::makeMemOpWideningDecisions(VPlan &Plan, VFRange &Range,
   // vector of selected pointers.
   VPlanTransforms::runPass(
       "widenSelectedBaseLoads", ProcessSubset, Plan, [&](VPInstruction *VPI) {
-        std::optional<SelectedBaseLoadPattern> Pattern =
-            matchSelectedBaseLoad(VPI);
-        if (!Pattern)
+        SelectedBaseLoadPattern Pattern;
+        if (!match(VPI, m_SelectedBaseLoad(Pattern)))
           return false;
 
-        if (!vputils::isUniformAcrossVFsAndUFs(Pattern->TrueBase) ||
-            !vputils::isUniformAcrossVFsAndUFs(Pattern->FalseBase) ||
-            vputils::isUniformAcrossVFsAndUFs(Pattern->Cond))
+        if (!vputils::isUniformAcrossVFsAndUFs(Pattern.TrueBase) ||
+            !vputils::isUniformAcrossVFsAndUFs(Pattern.FalseBase) ||
+            vputils::isUniformAcrossVFsAndUFs(Pattern.Cond))
           return false;
 
         auto *LI = cast<LoadInst>(VPI->getUnderlyingInstr());
@@ -5556,15 +5469,15 @@ void VPlanTransforms::makeMemOpWideningDecisions(VPlan &Plan, VFRange &Range,
           return false;
 
         VPBuilder Builder(VPI);
-        SmallVector<VPValue *> TrueOps = Pattern->GEPOperands;
-        SmallVector<VPValue *> FalseOps = Pattern->GEPOperands;
+        SmallVector<VPValue *> TrueOps = Pattern.GEPOperands;
+        SmallVector<VPValue *> FalseOps = Pattern.GEPOperands;
         // replace the baseOp
-        TrueOps[0] = Pattern->TrueBase;
-        FalseOps[0] = Pattern->FalseBase;
+        TrueOps[0] = Pattern.TrueBase;
+        FalseOps[0] = Pattern.FalseBase;
         auto *TrueAddr =
-            Builder.insert(Pattern->Addr->cloneWithOperands(TrueOps));
+            Builder.insert(Pattern.Addr->cloneWithOperands(TrueOps));
         auto *FalseAddr =
-            Builder.insert(Pattern->Addr->cloneWithOperands(FalseOps));
+            Builder.insert(Pattern.Addr->cloneWithOperands(FalseOps));
 
         // Splitting the selected base must expose two unit-stride accesses.
         // Remove the speculative address recipes again if either proof fails.
@@ -5579,19 +5492,28 @@ void VPlanTransforms::makeMemOpWideningDecisions(VPlan &Plan, VFRange &Range,
 
         bool HasActiveMask = RecipeBuilder.isPredicatedInst(LI);
         auto IsProfitable = [&](ElementCount VF) {
-          // Do not replace a scalarization decision without comparing against
-          // its cost and updating the legacy scalarization bookkeeping.
+          // Compare against the legacy decision for each VF so a scalarization
+          // choice is not replaced just because the gather is expensive.
           if (VF.isScalable() || CostCtx.willBeScalarized(LI, VF) ||
               !CostCtx.Config.isLegalGatherOrScatter(LI, VF))
             return false;
           Type *VecTy = VectorType::get(ScalarTy, VF);
-          InstructionCost BlendedCost = getBlendedLoadCost(
+          InstructionCost BlendedCost = slpvectorizer::getBlendedLoadCost(
               CostCtx.TTI, VecTy, SplitAlignment, LI->getPointerAddressSpace(),
-              HasActiveMask, CostCtx.CostKind);
-          return BlendedCost <
-                 getGatherLoadCost(*LI, VF, HasActiveMask, CostCtx);
+              CostCtx.CostKind);
+          if (HasActiveMask) {
+            Type *MaskTy = CmpInst::makeCmpResultType(VecTy);
+            BlendedCost += 2 * CostCtx.TTI.getArithmeticInstrCost(
+                                   Instruction::And, MaskTy, CostCtx.CostKind);
+          }
+          // Reuse the legacy decision's cost. This is the gather cost for the
+          // usual case, but also keeps the comparison correct if the legacy
+          // model selected scalarization.
+          InstructionCost LegacyCost = CostCtx.getLegacyCost(LI, VF);
+          return BlendedCost.isValid() && LegacyCost.isValid() &&
+                 BlendedCost < LegacyCost;
         };
-        // if fail, then erase the tmp recipes
+        // Erase the speculative address recipes if the cost comparison fails.
         if (!LoopVectorizationPlanner::getDecisionAndClampRange(IsProfitable,
                                                                 Range)) {
           TrueAddr->eraseFromParent();
@@ -5606,9 +5528,9 @@ void VPlanTransforms::makeMemOpWideningDecisions(VPlan &Plan, VFRange &Range,
             Builder.createConsecutiveVectorPointer(
                 FalseAddr, ScalarTy, /*Reverse=*/false, VPI->getDebugLoc());
 
-        VPValue *TrueMask = Pattern->Cond;
+        VPValue *TrueMask = Pattern.Cond;
         VPValue *FalseMask =
-            Builder.createNot(Pattern->Cond, VPI->getDebugLoc());
+            Builder.createNot(Pattern.Cond, VPI->getDebugLoc());
         if (HasActiveMask) {
           VPValue *ActiveMask = VPI->getMask();
           TrueMask = Builder.createLogicalAnd(ActiveMask, TrueMask,
@@ -5624,7 +5546,7 @@ void VPlanTransforms::makeMemOpWideningDecisions(VPlan &Plan, VFRange &Range,
             *LI, FalseVectorPtr, FalseMask, /*Consecutive=*/true, *VPI,
             VPI->getDebugLoc(), SplitAlignment);
         VPInstruction *Blend = Builder.createSelect(
-            Pattern->Cond, TrueLoad, FalseLoad, VPI->getDebugLoc());
+            Pattern.Cond, TrueLoad, FalseLoad, VPI->getDebugLoc());
         return ReplaceWith(VPI, Blend);
       });
 



More information about the llvm-commits mailing list