[llvm] [VPlan] Generalize vputils::getStrideExpr for known-sign (PR #219152)

via llvm-commits llvm-commits at lists.llvm.org
Thu Aug 27 02:23:20 PDT 2026


llvmorg-github-actions[bot] wrote:


<!--LLVM PR SUMMARY COMMENT-->

@llvm/pr-subscribers-llvm-transforms

Author: Ramkumar Ramachandra (artagnon)

<details>
<summary>Changes</summary>

Generalize vputils::getStrideExpr from matching an APInt to performing a UDiv when the Step is known-non-negative or known-negative. Generalizing this further needs SDiv and SRem expressions in ScalarEvolution.

-- 8< --
Based on https://github.com/llvm/llvm-project/pull/216760
See also https://github.com/llvm/llvm-project/pull/216862

---

Patch is 180.34 KiB, truncated to 20.00 KiB below, full version: https://github.com/llvm/llvm-project/pull/219152.diff


28 Files Affected:

- (modified) llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp (+20-22) 
- (modified) llvm/lib/Transforms/Vectorize/VPlanUtils.cpp (+42) 
- (modified) llvm/lib/Transforms/Vectorize/VPlanUtils.h (+7) 
- (modified) llvm/test/Transforms/LoopVectorize/AArch64/alias-mask.ll (+3-9) 
- (modified) llvm/test/Transforms/LoopVectorize/ARM/tail-folding-counting-down.ll (+27-57) 
- (modified) llvm/test/Transforms/LoopVectorize/RISCV/blocks-with-dead-instructions.ll (+15-15) 
- (modified) llvm/test/Transforms/LoopVectorize/RISCV/dead-ops-cost.ll (+7-7) 
- (modified) llvm/test/Transforms/LoopVectorize/RISCV/early-exit-live-out.ll (+4-4) 
- (modified) llvm/test/Transforms/LoopVectorize/RISCV/gather-scatter-cost.ll (+4-7) 
- (modified) llvm/test/Transforms/LoopVectorize/RISCV/interleaved-accesses.ll (+90-90) 
- (modified) llvm/test/Transforms/LoopVectorize/RISCV/masked_gather_scatter.ll (+11-11) 
- (modified) llvm/test/Transforms/LoopVectorize/RISCV/strided-access-wide-stride.ll (+2-2) 
- (modified) llvm/test/Transforms/LoopVectorize/RISCV/strided-accesses-i64-rv32.ll (+2-2) 
- (modified) llvm/test/Transforms/LoopVectorize/RISCV/strided-accesses-narrow-iv.ll (+16-16) 
- (modified) llvm/test/Transforms/LoopVectorize/RISCV/strided-accesses-unroll.ll (+4-4) 
- (modified) llvm/test/Transforms/LoopVectorize/RISCV/strided-accesses.ll (+32-32) 
- (modified) llvm/test/Transforms/LoopVectorize/RISCV/tail-folding-interleave.ll (+20-20) 
- (modified) llvm/test/Transforms/LoopVectorize/RISCV/tail-folding-reverse-load-store.ll (+9-23) 
- (modified) llvm/test/Transforms/LoopVectorize/VPlan/AArch64/vplan-memory-op-decisions.ll (+8-20) 
- (modified) llvm/test/Transforms/LoopVectorize/VPlan/vplan-sink-scalars-and-merge.ll (+2-3) 
- (modified) llvm/test/Transforms/LoopVectorize/X86/end-pointer-signed.ll (+2-3) 
- (modified) llvm/test/Transforms/LoopVectorize/X86/multi-exit-cost.ll (+9-15) 
- (modified) llvm/test/Transforms/LoopVectorize/bounded-load-multi-exit.ll (+10-15) 
- (modified) llvm/test/Transforms/LoopVectorize/induction-wrapflags.ll (+1-3) 
- (modified) llvm/test/Transforms/LoopVectorize/iv-select-cmp-decreasing.ll (+13-27) 
- (modified) llvm/test/Transforms/LoopVectorize/pointer-induction.ll (+1-3) 
- (modified) llvm/test/Transforms/LoopVectorize/single-early-exit-interleave.ll (+25-45) 
- (modified) llvm/test/Transforms/LoopVectorize/single_early_exit_live_outs.ll (+2-6) 


``````````diff
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index dc1a2697110b0..763a8a2495b04 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -46,17 +46,16 @@ using namespace SCEVPatternMatch;
 /// If the pointer operand \p Addr of a memory access is an affine AddRec
 /// w.r.t. \p L with a constant stride, return the stride in units of
 /// \p AccessTy. Otherwise return std::nullopt.
-static std::optional<int64_t> getConstantStride(VPValue *Addr, Type *AccessTy,
-                                                PredicatedScalarEvolution &PSE,
-                                                const Loop *L) {
+static std::optional<APInt> getConstantStride(VPValue *Addr, Type *AccessTy,
+                                              PredicatedScalarEvolution &PSE,
+                                              const Loop *L) {
   assert(!hasIrregularType(AccessTy, L->getHeader()->getDataLayout()) &&
          "should not try to widen irregular types");
-  const SCEV *AddrSCEV = vputils::getSCEVExprForVPValue(Addr, PSE, L);
-  auto *AddRec = dyn_cast<SCEVAddRecExpr>(AddrSCEV);
-  if (!AddRec)
+  auto Stride = vputils::getStrideExpr(Addr, PSE, *L, AccessTy);
+  const APInt *StrideC;
+  if (!Stride || !match(std::get<1>(*Stride), m_scev_APInt(StrideC)))
     return {};
-
-  return getStrideFromAddRec(AddRec, L, AccessTy, /*Ptr=*/nullptr, PSE);
+  return *StrideC;
 }
 
 bool VPlanTransforms::tryToConvertVPInstructionsToVPRecipes(
@@ -5481,7 +5480,7 @@ void VPlanTransforms::makeMemOpWideningDecisions(VPlan &Plan, VFRange &Range,
         VPValue *Ptr = VPI->getOperand(!IsLoad);
         Type *ScalarTy =
             IsLoad ? VPI->getScalarType() : VPI->getOperand(0)->getScalarType();
-        std::optional<int64_t> Stride =
+        std::optional<APInt> Stride =
             getConstantStride(Ptr, ScalarTy, CostCtx.PSE, CostCtx.L);
         if (Stride != 1 && Stride != -1)
           return false;
@@ -5774,14 +5773,16 @@ void VPlanTransforms::convertToStridedAccesses(VPlan &Plan,
       VPValue *Ptr = MemR->getAddr();
       // Check if this is a strided access by analyzing the address SCEV for an
       // affine addRec.
-      const SCEV *PtrSCEV = vputils::getSCEVExprForVPValue(Ptr, PSE, &L);
-      const SCEV *Start;
-      const SCEVConstant *Step;
+      auto StrideTup = vputils::getStrideExpr(
+          Ptr, PSE, L, Type::getInt8Ty(Plan.getContext()));
+      if (!StrideTup)
+        continue;
+      auto [Start, Stride, NW] = *StrideTup;
       // TODO: Support non-constant loop invariant stride.
-      if (!match(PtrSCEV,
-                 m_scev_AffineAddRec(m_SCEV(Start), m_SCEVConstant(Step),
-                                     m_SpecificLoop(&L))))
+      const APInt *StrideC;
+      if (!match(Stride, m_scev_APInt(StrideC)))
         continue;
+      bool HasNUW = any(NW & SCEV::FlagNUW);
 
       VPValue *StoredValue = nullptr;
       Type *DataTy;
@@ -5834,19 +5835,16 @@ void VPlanTransforms::convertToStridedAccesses(VPlan &Plan,
                               .tryToExpand(Start);
       if (!StartVPV)
         StartVPV = VPBuilder(Plan.getEntry()).createExpandSCEV(Start);
-      VPValue *StrideInBytes = Plan.getOrAddLiveIn(Step->getValue());
+      VPValue *StrideInBytes = Plan.getConstantInt(*StrideC);
       Type *IndexTy = Plan.getDataLayout().getIndexType(Ptr->getScalarType());
       assert(IndexTy == StrideInBytes->getScalarType() &&
              "Stride type from SCEV must match the index type");
       VPValue *CanIV = Builder.createScalarZExtOrTrunc(
           VectorLoop->getCanonicalIV(), IndexTy, DebugLoc::getUnknown());
-      auto *AddRecPtr = cast<SCEVAddRecExpr>(PtrSCEV);
       auto *Offset = Builder.createOverflowingOp(
-          Instruction::Mul, {CanIV, StrideInBytes},
-          {AddRecPtr->hasNoUnsignedWrap(), /*HasNSW=*/false});
-      GEPNoWrapFlags NWFlags = AddRecPtr->hasNoUnsignedWrap()
-                                   ? GEPNoWrapFlags::noUnsignedWrap()
-                                   : GEPNoWrapFlags::none();
+          Instruction::Mul, {CanIV, StrideInBytes}, {HasNUW, /*HasNSW=*/false});
+      GEPNoWrapFlags NWFlags =
+          HasNUW ? GEPNoWrapFlags::noUnsignedWrap() : GEPNoWrapFlags::none();
       VPValue *BasePtr = Builder.createNoWrapPtrAdd(StartVPV, Offset, NWFlags);
 
       // Create a new vector pointer for strided access.
diff --git a/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp b/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
index e6b267493d987..7eac7fd01e44d 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
@@ -339,6 +339,48 @@ const SCEV *vputils::getSCEVExprForVPValue(const VPValue *V,
   return PSE.getPredicatedSCEV(Expr);
 }
 
+std::optional<std::tuple<const SCEV *, const SCEV *, SCEVNoWrapFlags>>
+vputils::getStrideExpr(const VPValue *Ptr, PredicatedScalarEvolution &PSE,
+                       const Loop &L, Type *AccessTy) {
+  assert(Ptr->getScalarType()->isPointerTy() && "Ptr must be pointer type");
+  ScalarEvolution &SE = *PSE.getSE();
+  const SCEV *PtrSCEV = vputils::getSCEVExprForVPValue(Ptr, PSE, &L);
+  if (isa<SCEVCouldNotCompute>(PtrSCEV))
+    return std::nullopt;
+  const SCEV *PointerBase = SE.getPointerBase(PtrSCEV);
+  const SCEV *StrideExpr = SE.removePointerBase(PtrSCEV);
+  Type *StrideTy = StrideExpr->getType();
+  const SCEV *Start;
+  const SCEV *Step;
+  if (!match(StrideExpr, m_scev_AffineAddRec(m_SCEV(Start), m_SCEV(Step),
+                                             m_SpecificLoop(&L))))
+    return std::nullopt;
+  SCEVNoWrapFlags NWFlags = cast<SCEVAddRecExpr>(StrideExpr)->getNoWrapFlags();
+  const SCEV *Base =
+      SE.getAddExpr(PointerBase, SE.getNoopOrSignExtend(Start, StrideTy));
+  const DataLayout &DL = SE.getDataLayout();
+  TypeSize AllocSz = DL.getTypeAllocSize(AccessTy);
+  if (AllocSz.isScalable())
+    return std::nullopt;
+  const SCEV *StepExpr = SE.getNoopOrSignExtend(Step, StrideTy);
+  const SCEV *AllocSzExpr = SE.getConstant(StrideTy, AllocSz);
+  // TODO: Could extend to add a known-multiple-of predicate, once we have SRem
+  // expressions.
+  if (!SE.getURemExpr(StepExpr, AllocSzExpr)->isZero())
+    return std::nullopt;
+  // TODO: ScalarEvolution doesn't have SDiv expressions yet, so we resort to
+  // isKnownNonNegative/isKnownNegative, knowing that AllocSz is an unsigned
+  // quantity.
+  const SCEV *UDivExpr = SE.getUDivExactExpr(StepExpr, AllocSzExpr);
+  if (SE.isKnownNonNegative(StepExpr))
+    return std::make_tuple(Base, UDivExpr, NWFlags);
+  if (SE.isKnownNegative(StepExpr))
+    // SDiv(Negative(StepExpr), NonNegative(AllocSzExpr)) =
+    // -UDiv(NonNegative(StepExpr), NonNegative(AllocSzExpr)).
+    return std::make_tuple(Base, SE.getNegativeSCEV(UDivExpr), NWFlags);
+  return std::nullopt;
+}
+
 bool vputils::isAddressSCEVForCost(const SCEV *Addr, ScalarEvolution &SE,
                                    const Loop *L) {
   // If address is an SCEVAddExpr, we require that all operands must be either
diff --git a/llvm/lib/Transforms/Vectorize/VPlanUtils.h b/llvm/lib/Transforms/Vectorize/VPlanUtils.h
index 60a9a8013de66..3e0eaa0a1fcd9 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanUtils.h
+++ b/llvm/lib/Transforms/Vectorize/VPlanUtils.h
@@ -46,6 +46,13 @@ const SCEV *getSCEVExprForVPValue(const VPValue *V,
                                   PredicatedScalarEvolution &PSE,
                                   const Loop *L = nullptr);
 
+/// Get a stride expression, the AddRec's step found from \p Ptr divided by the
+/// alloc-size of \p AccessTy. Returns a tuple of the start SCEV expression, the
+/// stride SCEV expression, and the AddRec's wrap flags.
+std::optional<std::tuple<const SCEV *, const SCEV *, SCEVNoWrapFlags>>
+getStrideExpr(const VPValue *Ptr, PredicatedScalarEvolution &PSE, const Loop &L,
+              Type *AccessTy);
+
 /// Returns true if \p Addr is an address SCEV that can be passed to
 /// TTI::getAddressComputationCost, i.e. the address SCEV is loop invariant, an
 /// affine AddRec (i.e. induction ), or an add expression of such operands or a
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/alias-mask.ll b/llvm/test/Transforms/LoopVectorize/AArch64/alias-mask.ll
index f50a146db1a69..479c1fb4a2c67 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/alias-mask.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/alias-mask.ll
@@ -281,18 +281,12 @@ define void @alias_mask_reverse_iterate(ptr noalias %ptrA, ptr %ptrB, ptr %ptrC,
 ; CHECK-TF-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <vscale x 16 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VECTOR_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VECTOR_BODY]] ]
 ; CHECK-TF-NEXT:    [[OFFSET_IDX:%.*]] = sub i64 [[IV_START]], [[INDEX]]
 ; CHECK-TF-NEXT:    [[TMP8:%.*]] = getelementptr inbounds i8, ptr [[PTRA]], i64 [[OFFSET_IDX]]
-; CHECK-TF-NEXT:    [[TMP9:%.*]] = sub nuw nsw i64 [[TMP4]], 1
-; CHECK-TF-NEXT:    [[TMP10:%.*]] = sub i64 0, [[TMP9]]
-; CHECK-TF-NEXT:    [[TMP11:%.*]] = getelementptr i8, ptr [[TMP8]], i64 [[TMP10]]
-; CHECK-TF-NEXT:    [[REVERSE:%.*]] = call <vscale x 16 x i1> @llvm.vector.reverse.nxv16i1(<vscale x 16 x i1> [[ACTIVE_LANE_MASK]])
-; CHECK-TF-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <vscale x 16 x i8> @llvm.masked.load.nxv16i8.p0(ptr align 1 [[TMP11]], <vscale x 16 x i1> [[REVERSE]], <vscale x 16 x i8> poison)
+; CHECK-TF-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <vscale x 16 x i8> @llvm.masked.load.nxv16i8.p0(ptr align 1 [[TMP8]], <vscale x 16 x i1> [[ACTIVE_LANE_MASK]], <vscale x 16 x i8> poison)
 ; CHECK-TF-NEXT:    [[TMP12:%.*]] = getelementptr inbounds i8, ptr [[PTRB]], i64 [[OFFSET_IDX]]
-; CHECK-TF-NEXT:    [[TMP13:%.*]] = getelementptr i8, ptr [[TMP12]], i64 [[TMP10]]
-; CHECK-TF-NEXT:    [[WIDE_MASKED_LOAD5:%.*]] = call <vscale x 16 x i8> @llvm.masked.load.nxv16i8.p0(ptr align 1 [[TMP13]], <vscale x 16 x i1> [[REVERSE]], <vscale x 16 x i8> poison)
+; CHECK-TF-NEXT:    [[WIDE_MASKED_LOAD5:%.*]] = call <vscale x 16 x i8> @llvm.masked.load.nxv16i8.p0(ptr align 1 [[TMP12]], <vscale x 16 x i1> [[ACTIVE_LANE_MASK]], <vscale x 16 x i8> poison)
 ; CHECK-TF-NEXT:    [[REVERSE7:%.*]] = add <vscale x 16 x i8> [[WIDE_MASKED_LOAD5]], [[WIDE_MASKED_LOAD]]
 ; CHECK-TF-NEXT:    [[TMP15:%.*]] = getelementptr inbounds i8, ptr [[PTRC]], i64 [[OFFSET_IDX]]
-; CHECK-TF-NEXT:    [[TMP16:%.*]] = getelementptr i8, ptr [[TMP15]], i64 [[TMP10]]
-; CHECK-TF-NEXT:    call void @llvm.masked.store.nxv16i8.p0(<vscale x 16 x i8> [[REVERSE7]], ptr align 1 [[TMP16]], <vscale x 16 x i1> [[REVERSE]])
+; CHECK-TF-NEXT:    call void @llvm.masked.store.nxv16i8.p0(<vscale x 16 x i8> [[REVERSE7]], ptr align 1 [[TMP15]], <vscale x 16 x i1> [[ACTIVE_LANE_MASK]])
 ; CHECK-TF-NEXT:    [[INDEX_NEXT]] = add i64 [[INDEX]], [[TMP4]]
 ; CHECK-TF-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <vscale x 16 x i1> @llvm.get.active.lane.mask.nxv16i1.i64(i64 [[INDEX_NEXT]], i64 [[IV_START]])
 ; CHECK-TF-NEXT:    [[TMP17:%.*]] = extractelement <vscale x 16 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
diff --git a/llvm/test/Transforms/LoopVectorize/ARM/tail-folding-counting-down.ll b/llvm/test/Transforms/LoopVectorize/ARM/tail-folding-counting-down.ll
index 4621ddfd330eb..170a95e3b05dc 100644
--- a/llvm/test/Transforms/LoopVectorize/ARM/tail-folding-counting-down.ll
+++ b/llvm/test/Transforms/LoopVectorize/ARM/tail-folding-counting-down.ll
@@ -533,15 +533,12 @@ define void @sgt_for_loop(ptr noalias nocapture readonly %a, ptr noalias nocaptu
 ; DEFAULT-NEXT:    [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
 ; DEFAULT-NEXT:    [[TMP1:%.*]] = sub i32 [[N]], [[INDEX]]
 ; DEFAULT-NEXT:    [[TMP2:%.*]] = getelementptr inbounds i8, ptr [[A]], i32 [[TMP1]]
-; DEFAULT-NEXT:    [[TMP3:%.*]] = getelementptr inbounds i8, ptr [[TMP2]], i32 -15
-; DEFAULT-NEXT:    [[WIDE_LOAD:%.*]] = load <16 x i8>, ptr [[TMP3]], align 1
+; DEFAULT-NEXT:    [[WIDE_LOAD:%.*]] = load <16 x i8>, ptr [[TMP2]], align 1
 ; DEFAULT-NEXT:    [[TMP4:%.*]] = getelementptr inbounds i8, ptr [[B]], i32 [[TMP1]]
-; DEFAULT-NEXT:    [[TMP5:%.*]] = getelementptr inbounds i8, ptr [[TMP4]], i32 -15
-; DEFAULT-NEXT:    [[WIDE_LOAD1:%.*]] = load <16 x i8>, ptr [[TMP5]], align 1
+; DEFAULT-NEXT:    [[WIDE_LOAD1:%.*]] = load <16 x i8>, ptr [[TMP4]], align 1
 ; DEFAULT-NEXT:    [[REVERSE3:%.*]] = add <16 x i8> [[WIDE_LOAD1]], [[WIDE_LOAD]]
 ; DEFAULT-NEXT:    [[TMP7:%.*]] = getelementptr inbounds i8, ptr [[C]], i32 [[TMP1]]
-; DEFAULT-NEXT:    [[TMP8:%.*]] = getelementptr inbounds i8, ptr [[TMP7]], i32 -15
-; DEFAULT-NEXT:    store <16 x i8> [[REVERSE3]], ptr [[TMP8]], align 1
+; DEFAULT-NEXT:    store <16 x i8> [[REVERSE3]], ptr [[TMP7]], align 1
 ; DEFAULT-NEXT:    [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 16
 ; DEFAULT-NEXT:    [[TMP9:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
 ; DEFAULT-NEXT:    br i1 [[TMP9]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP8:![0-9]+]]
@@ -567,16 +564,12 @@ define void @sgt_for_loop(ptr noalias nocapture readonly %a, ptr noalias nocaptu
 ; CHECK-PREFER-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = call <16 x i1> @llvm.get.active.lane.mask.v16i1.i32(i32 [[INDEX]], i32 [[N]])
 ; CHECK-PREFER-NEXT:    [[TMP0:%.*]] = sub i32 [[N]], [[INDEX]]
 ; CHECK-PREFER-NEXT:    [[TMP1:%.*]] = getelementptr inbounds i8, ptr [[A]], i32 [[TMP0]]
-; CHECK-PREFER-NEXT:    [[TMP2:%.*]] = getelementptr i8, ptr [[TMP1]], i32 -15
-; CHECK-PREFER-NEXT:    [[REVERSE:%.*]] = shufflevector <16 x i1> [[ACTIVE_LANE_MASK]], <16 x i1> poison, <16 x i32> <i32 15, i32 14, i32 13, i32 12, i32 11, i32 10, i32 9, i32 8, i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
-; CHECK-PREFER-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <16 x i8> @llvm.masked.load.v16i8.p0(ptr align 1 [[TMP2]], <16 x i1> [[REVERSE]], <16 x i8> poison)
+; CHECK-PREFER-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <16 x i8> @llvm.masked.load.v16i8.p0(ptr align 1 [[TMP1]], <16 x i1> [[ACTIVE_LANE_MASK]], <16 x i8> poison)
 ; CHECK-PREFER-NEXT:    [[TMP3:%.*]] = getelementptr inbounds i8, ptr [[B]], i32 [[TMP0]]
-; CHECK-PREFER-NEXT:    [[TMP4:%.*]] = getelementptr i8, ptr [[TMP3]], i32 -15
-; CHECK-PREFER-NEXT:    [[WIDE_MASKED_LOAD2:%.*]] = call <16 x i8> @llvm.masked.load.v16i8.p0(ptr align 1 [[TMP4]], <16 x i1> [[REVERSE]], <16 x i8> poison)
+; CHECK-PREFER-NEXT:    [[WIDE_MASKED_LOAD2:%.*]] = call <16 x i8> @llvm.masked.load.v16i8.p0(ptr align 1 [[TMP3]], <16 x i1> [[ACTIVE_LANE_MASK]], <16 x i8> poison)
 ; CHECK-PREFER-NEXT:    [[REVERSE4:%.*]] = add <16 x i8> [[WIDE_MASKED_LOAD2]], [[WIDE_MASKED_LOAD]]
 ; CHECK-PREFER-NEXT:    [[TMP6:%.*]] = getelementptr inbounds i8, ptr [[C]], i32 [[TMP0]]
-; CHECK-PREFER-NEXT:    [[TMP7:%.*]] = getelementptr i8, ptr [[TMP6]], i32 -15
-; CHECK-PREFER-NEXT:    call void @llvm.masked.store.v16i8.p0(<16 x i8> [[REVERSE4]], ptr align 1 [[TMP7]], <16 x i1> [[REVERSE]])
+; CHECK-PREFER-NEXT:    call void @llvm.masked.store.v16i8.p0(<16 x i8> [[REVERSE4]], ptr align 1 [[TMP6]], <16 x i1> [[ACTIVE_LANE_MASK]])
 ; CHECK-PREFER-NEXT:    [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 16
 ; CHECK-PREFER-NEXT:    [[TMP8:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
 ; CHECK-PREFER-NEXT:    br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP8:![0-9]+]]
@@ -604,15 +597,12 @@ define void @sgt_for_loop(ptr noalias nocapture readonly %a, ptr noalias nocaptu
 ; CHECK-ENABLE-TP-NEXT:    [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
 ; CHECK-ENABLE-TP-NEXT:    [[TMP1:%.*]] = sub i32 [[N]], [[INDEX]]
 ; CHECK-ENABLE-TP-NEXT:    [[TMP2:%.*]] = getelementptr inbounds i8, ptr [[A]], i32 [[TMP1]]
-; CHECK-ENABLE-TP-NEXT:    [[TMP3:%.*]] = getelementptr inbounds i8, ptr [[TMP2]], i32 -15
-; CHECK-ENABLE-TP-NEXT:    [[WIDE_LOAD:%.*]] = load <16 x i8>, ptr [[TMP3]], align 1
+; CHECK-ENABLE-TP-NEXT:    [[WIDE_LOAD:%.*]] = load <16 x i8>, ptr [[TMP2]], align 1
 ; CHECK-ENABLE-TP-NEXT:    [[TMP4:%.*]] = getelementptr inbounds i8, ptr [[B]], i32 [[TMP1]]
-; CHECK-ENABLE-TP-NEXT:    [[TMP5:%.*]] = getelementptr inbounds i8, ptr [[TMP4]], i32 -15
-; CHECK-ENABLE-TP-NEXT:    [[WIDE_LOAD1:%.*]] = load <16 x i8>, ptr [[TMP5]], align 1
+; CHECK-ENABLE-TP-NEXT:    [[WIDE_LOAD1:%.*]] = load <16 x i8>, ptr [[TMP4]], align 1
 ; CHECK-ENABLE-TP-NEXT:    [[REVERSE3:%.*]] = add <16 x i8> [[WIDE_LOAD1]], [[WIDE_LOAD]]
 ; CHECK-ENABLE-TP-NEXT:    [[TMP7:%.*]] = getelementptr inbounds i8, ptr [[C]], i32 [[TMP1]]
-; CHECK-ENABLE-TP-NEXT:    [[TMP8:%.*]] = getelementptr inbounds i8, ptr [[TMP7]], i32 -15
-; CHECK-ENABLE-TP-NEXT:    store <16 x i8> [[REVERSE3]], ptr [[TMP8]], align 1
+; CHECK-ENABLE-TP-NEXT:    store <16 x i8> [[REVERSE3]], ptr [[TMP7]], align 1
 ; CHECK-ENABLE-TP-NEXT:    [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 16
 ; CHECK-ENABLE-TP-NEXT:    [[TMP9:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
 ; CHECK-ENABLE-TP-NEXT:    br i1 [[TMP9]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP8:![0-9]+]]
@@ -665,15 +655,12 @@ define void @sgt_for_loop_i64(ptr noalias nocapture readonly %a, ptr noalias noc
 ; DEFAULT-NEXT:    [[TMP1:%.*]] = sub i64 [[CONV16]], [[INDEX]]
 ; DEFAULT-NEXT:    [[TMP2:%.*]] = trunc i64 [[TMP1]] to i32
 ; DEFAULT-NEXT:    [[TMP3:%.*]] = getelementptr inbounds i8, ptr [[A]], i32 [[TMP2]]
-; DEFAULT-NEXT:    [[TMP4:%.*]] = getelementptr inbounds i8, ptr [[TMP3]], i32 -15
-; DEFAULT-NEXT:    [[WIDE_LOAD:%.*]] = load <16 x i8>, ptr [[TMP4]], align 1
+; DEFAULT-NEXT:    [[WIDE_LOAD:%.*]] = load <16 x i8>, ptr [[TMP3]], align 1
 ; DEFAULT-NEXT:    [[TMP5:%.*]] = getelementptr inbounds i8, ptr [[B]], i32 [[TMP2]]
-; DEFAULT-NEXT:    [[TMP6:%.*]] = getelementptr inbounds i8, ptr [[TMP5]], i32 -15
-; DEFAULT-NEXT:    [[WIDE_LOAD1:%.*]] = load <16 x i8>, ptr [[TMP6]], align 1
+; DEFAULT-NEXT:    [[WIDE_LOAD1:%.*]] = load <16 x i8>, ptr [[TMP5]], align 1
 ; DEFAULT-NEXT:    [[REVERSE3:%.*]] = add <16 x i8> [[WIDE_LOAD1]], [[WIDE_LOAD]]
 ; DEFAULT-NEXT:    [[TMP8:%.*]] = getelementptr inbounds i8, ptr [[C]], i32 [[TMP2]]
-; DEFAULT-NEXT:    [[TMP9:%.*]] = getelementptr inbounds i8, ptr [[TMP8]], i32 -15
-; DEFAULT-NEXT:    store <16 x i8> [[REVERSE3]], ptr [[TMP9]], align 1
+; DEFAULT-NEXT:    store <16 x i8> [[REVERSE3]], ptr [[TMP8]], align 1
 ; DEFAULT-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
 ; DEFAULT-NEXT:    [[TMP10:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
 ; DEFAULT-NEXT:    br i1 [[TMP10]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP10:![0-9]+]]
@@ -701,16 +688,12 @@ define void @sgt_for_loop_i64(ptr noalias nocapture readonly %a, ptr noalias noc
 ; CHECK-PREFER-NEXT:    [[TMP0:%.*]] = sub i64 [[CONV16]], [[INDEX]]
 ; CHECK-PREFER-NEXT:    [[TMP1:%.*]] = trunc i64 [[TMP0]] to i32
 ; CHECK-PREFER-NEXT:    [[TMP2:%.*]] = getelementptr inbounds i8, ptr [[A]], i32 [[TMP1]]
-; CHECK-PREFER-NEXT:    [[TMP3:%.*]] = getelementptr i8, ptr [[TMP2]], i32 -15
-; CHECK-PREFER-NEXT:    [[REVERSE:%.*]] = shufflevector <16 x i1> [[ACTIVE_LANE_MASK]], <16 x i1> poison, <16 x i32> <i32 15, i32 14, i32 13, i32 12, i32 11, i32 10, i32 9, i32 8, i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
-; CHECK-PREFER-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <16 x i8> @llvm.masked.load.v16i8.p0(ptr align 1 [[TMP3]], <16 x i1> [[REVERSE]], <16 x i8> poison)
+; CHECK-PREFER-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <16 x i8> @llvm.masked.load.v16i8.p0(ptr align 1 [[TMP2]], <16 x i1> [[ACTIVE_LANE_MASK]], <16 x i8> poison)
 ; CHECK-PREFER-NEXT:    [[TMP4:%.*]] = getelementptr inbounds i8, ptr [[B]], i32 [[TMP1]]
-; CHECK-PREFER-NEXT:    [[TMP5:%.*]] = getelementptr i8, ptr [[TMP4]], i32 -15
-; CHECK-PREFER-NEXT:    [[WIDE_MASKED_LOAD2:%.*]] = call <16 x i8> @llvm.masked.load.v16i8.p0(ptr align 1 [[TMP5]], <16 x i1> [[REVERSE]], <16 x i8> poison)
+; CHECK-PREFER-NEXT:    [[WIDE_MASKED_LOAD2:%.*]] = call <16 x i8> @llvm.masked.load.v16i8.p0(ptr align 1 [[TMP4]], <16 x i1> [[ACTIVE_LANE_MASK]], <16 x i8> poison)
 ; CHECK-PREFER-NEXT:    [[REVERSE4:%.*]] = add <16 x i8> [[WIDE_MASKED_LOAD2]], [[WIDE_MASKED_LOAD]]
 ; CHECK-PREFER-NEXT:    [[TMP7:%.*]] = getelementptr inbounds i8, ptr [[C]], i32 [[TMP1]]
-; CHECK-PREFER-NEXT:    [[TMP8:%.*]] = getelementptr i8, ptr [[TMP7]], i32 -15
-; CHECK-PREFER-NEXT:    call void @llvm.masked.store.v16i8....
[truncated]

``````````

</details>


https://github.com/llvm/llvm-project/pull/219152


More information about the llvm-commits mailing list