[llvm] [LV] Use index type for base pointer computation in convertToStridedAccesses (PR #201070)
via llvm-commits
llvm-commits at lists.llvm.org
Tue Jun 2 02:13:22 PDT 2026
llvmorg-github-actions[bot] wrote:
<!--LLVM PR SUMMARY COMMENT-->
@llvm/pr-subscribers-llvm-transforms
@llvm/pr-subscribers-backend-risc-v
Author: Mel Chen (Mel-Chen)
<details>
<summary>Changes</summary>
The base pointer for strided accesses was computed as:
```
offset = canonicalIV * stride
base_ptr = ptradd start, offset
```
On a 64-bit target, if the canonical IV type is i32, the GEP operation for ptradd will first sign-extend the offset to i64. Once the offset multiplication has already overflowed in i32, it will ultimately result in an incorrect base address.
This patch fixes this by extending the canonical IV to the index type before the offset multiplication.
Based on #<!-- -->199647
---
Patch is 29.55 KiB, truncated to 20.00 KiB below, full version: https://github.com/llvm/llvm-project/pull/201070.diff
6 Files Affected:
- (modified) llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp (+15-7)
- (modified) llvm/test/Transforms/LoopVectorize/RISCV/dead-ops-cost.ll (+4-3)
- (modified) llvm/test/Transforms/LoopVectorize/RISCV/masked_gather_scatter.ll (+7-6)
- (added) llvm/test/Transforms/LoopVectorize/RISCV/strided-accesses-narrow-iv.ll (+279)
- (modified) llvm/test/Transforms/LoopVectorize/RISCV/strided-accesses.ll (+8-6)
- (modified) llvm/test/Transforms/LoopVectorize/RISCV/truncate-to-minimal-bitwidth-cost.ll (+4-3)
``````````diff
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index 60d1c70393dc3..68d6f1908ea6d 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -7047,10 +7047,11 @@ void VPlanTransforms::convertToStridedAccesses(VPlan &Plan,
// affine addRec.
const SCEV *PtrSCEV = vputils::getSCEVExprForVPValue(Ptr, PSE, &L);
const SCEV *Start;
- const APInt *Step;
+ const SCEVConstant *Step;
// TODO: Support non-constant loop invariant stride.
- if (!match(PtrSCEV, m_scev_AffineAddRec(m_SCEV(Start), m_scev_APInt(Step),
- m_SpecificLoop(&L))))
+ if (!match(PtrSCEV,
+ m_scev_AffineAddRec(m_SCEV(Start), m_SCEVConstant(Step),
+ m_SpecificLoop(&L))))
continue;
Type *LoadTy = TypeInfo.inferScalarType(LoadR);
@@ -7087,13 +7088,20 @@ void VPlanTransforms::convertToStridedAccesses(VPlan &Plan,
VPBuilder Builder(LoadR);
// Create the base pointer of strided access.
+ // TODO: reuse VPDerivedIVRecipe for base pointer computation when it
+ // supports a general VPValue as the start value.
VPValue *StartVPV = vputils::getOrCreateVPValueForSCEVExpr(Plan, Start);
- VPValue *StrideInBytes =
- Plan.getConstantInt(VectorLoop->getCanonicalIVType(),
- Step->getSExtValue(), /*IsSigned=*/true);
+ VPValue *StrideInBytes = Plan.getOrAddLiveIn(Step->getValue());
+ Type *IndexTy =
+ Plan.getDataLayout().getIndexType(TypeInfo.inferScalarType(Ptr));
+ assert(IndexTy == TypeInfo.inferScalarType(StrideInBytes) &&
+ "Stride type from SCEV must match the index type");
+ VPValue *CanIV = Builder.createScalarSExtOrTrunc(
+ VectorLoop->getCanonicalIV(), IndexTy,
+ VectorLoop->getCanonicalIVType(), DebugLoc::getUnknown());
auto *AddRecPtr = cast<SCEVAddRecExpr>(PtrSCEV);
auto *Offset = Builder.createOverflowingOp(
- Instruction::Mul, {VectorLoop->getCanonicalIV(), StrideInBytes},
+ Instruction::Mul, {CanIV, StrideInBytes},
{AddRecPtr->hasNoUnsignedWrap(), AddRecPtr->hasNoSignedWrap()});
auto *BasePtr = Builder.createNoWrapPtrAdd(
StartVPV, Offset,
diff --git a/llvm/test/Transforms/LoopVectorize/RISCV/dead-ops-cost.ll b/llvm/test/Transforms/LoopVectorize/RISCV/dead-ops-cost.ll
index c9e5e2d2a6faa..a9db4458a34ff 100644
--- a/llvm/test/Transforms/LoopVectorize/RISCV/dead-ops-cost.ll
+++ b/llvm/test/Transforms/LoopVectorize/RISCV/dead-ops-cost.ll
@@ -91,9 +91,10 @@ define i8 @dead_live_out_due_to_scalar_epilogue_required(ptr %src, ptr %dst) {
; CHECK-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 16 x i32> poison, i32 [[TMP3]], i64 0
; CHECK-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <vscale x 16 x i32> [[BROADCAST_SPLATINSERT]], <vscale x 16 x i32> poison, <vscale x 16 x i32> zeroinitializer
; CHECK-NEXT: [[TMP9:%.*]] = sext <vscale x 16 x i32> [[VEC_IND]] to <vscale x 16 x i64>
-; CHECK-NEXT: [[TMP5:%.*]] = shl i32 [[INDEX]], 2
-; CHECK-NEXT: [[TMP6:%.*]] = getelementptr i8, ptr [[SRC]], i32 [[TMP5]]
-; CHECK-NEXT: [[WIDE_MASKED_GATHER:%.*]] = call <vscale x 16 x i8> @llvm.experimental.vp.strided.load.nxv16i8.p0.i32(ptr align 1 [[TMP6]], i32 4, <vscale x 16 x i1> splat (i1 true), i32 [[TMP2]]), !alias.scope [[META3:![0-9]+]]
+; CHECK-NEXT: [[TMP5:%.*]] = sext i32 [[INDEX]] to i64
+; CHECK-NEXT: [[TMP6:%.*]] = shl i64 [[TMP5]], 2
+; CHECK-NEXT: [[TMP12:%.*]] = getelementptr i8, ptr [[SRC]], i64 [[TMP6]]
+; CHECK-NEXT: [[WIDE_MASKED_GATHER:%.*]] = call <vscale x 16 x i8> @llvm.experimental.vp.strided.load.nxv16i8.p0.i64(ptr align 1 [[TMP12]], i64 4, <vscale x 16 x i1> splat (i1 true), i32 [[TMP2]]), !alias.scope [[META3:![0-9]+]]
; CHECK-NEXT: [[TMP7:%.*]] = getelementptr i8, ptr [[DST]], <vscale x 16 x i64> [[TMP9]]
; CHECK-NEXT: call void @llvm.vp.scatter.nxv16i8.nxv16p0(<vscale x 16 x i8> zeroinitializer, <vscale x 16 x ptr> align 1 [[TMP7]], <vscale x 16 x i1> splat (i1 true), i32 [[TMP2]]), !alias.scope [[META6:![0-9]+]], !noalias [[META3]]
; CHECK-NEXT: [[CURRENT_ITERATION_NEXT]] = add nuw i32 [[TMP2]], [[INDEX]]
diff --git a/llvm/test/Transforms/LoopVectorize/RISCV/masked_gather_scatter.ll b/llvm/test/Transforms/LoopVectorize/RISCV/masked_gather_scatter.ll
index 55af1a712fac0..1c67b243429b5 100644
--- a/llvm/test/Transforms/LoopVectorize/RISCV/masked_gather_scatter.ll
+++ b/llvm/test/Transforms/LoopVectorize/RISCV/masked_gather_scatter.ll
@@ -57,13 +57,14 @@ define void @foo4(ptr nocapture %A, ptr nocapture readonly %B, ptr nocapture rea
; RV32-NEXT: [[TMP11:%.*]] = shl nuw nsw i64 [[TMP8]], 4
; RV32-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 2 x i64> poison, i64 [[TMP11]], i64 0
; RV32-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <vscale x 2 x i64> [[BROADCAST_SPLATINSERT]], <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
-; RV32-NEXT: [[TMP12:%.*]] = shl i64 [[INDEX]], 6
-; RV32-NEXT: [[TMP13:%.*]] = getelementptr i8, ptr [[TRIGGER]], i64 [[TMP12]]
-; RV32-NEXT: [[WIDE_MASKED_GATHER:%.*]] = call <vscale x 2 x i32> @llvm.experimental.vp.strided.load.nxv2i32.p0.i64(ptr align 4 [[TMP13]], i64 64, <vscale x 2 x i1> splat (i1 true), i32 [[TMP10]]), !alias.scope [[META0:![0-9]+]], !noalias [[META3:![0-9]+]]
+; RV32-NEXT: [[TMP12:%.*]] = trunc i64 [[INDEX]] to i32
+; RV32-NEXT: [[TMP13:%.*]] = shl i32 [[TMP12]], 6
+; RV32-NEXT: [[TMP15:%.*]] = getelementptr i8, ptr [[TRIGGER]], i32 [[TMP13]]
+; RV32-NEXT: [[WIDE_MASKED_GATHER:%.*]] = call <vscale x 2 x i32> @llvm.experimental.vp.strided.load.nxv2i32.p0.i32(ptr align 4 [[TMP15]], i32 64, <vscale x 2 x i1> splat (i1 true), i32 [[TMP10]]), !alias.scope [[META0:![0-9]+]], !noalias [[META3:![0-9]+]]
; RV32-NEXT: [[TMP14:%.*]] = icmp slt <vscale x 2 x i32> [[WIDE_MASKED_GATHER]], splat (i32 100)
-; RV32-NEXT: [[TMP16:%.*]] = shl i64 [[INDEX]], 8
-; RV32-NEXT: [[TMP20:%.*]] = getelementptr i8, ptr [[B]], i64 [[TMP16]]
-; RV32-NEXT: [[WIDE_MASKED_GATHER6:%.*]] = call <vscale x 2 x double> @llvm.experimental.vp.strided.load.nxv2f64.p0.i64(ptr align 8 [[TMP20]], i64 256, <vscale x 2 x i1> [[TMP14]], i32 [[TMP10]]), !alias.scope [[META5:![0-9]+]]
+; RV32-NEXT: [[TMP20:%.*]] = shl i32 [[TMP12]], 8
+; RV32-NEXT: [[TMP25:%.*]] = getelementptr i8, ptr [[B]], i32 [[TMP20]]
+; RV32-NEXT: [[WIDE_MASKED_GATHER6:%.*]] = call <vscale x 2 x double> @llvm.experimental.vp.strided.load.nxv2f64.p0.i32(ptr align 8 [[TMP25]], i32 256, <vscale x 2 x i1> [[TMP14]], i32 [[TMP10]]), !alias.scope [[META5:![0-9]+]]
; RV32-NEXT: [[TMP17:%.*]] = sitofp <vscale x 2 x i32> [[WIDE_MASKED_GATHER]] to <vscale x 2 x double>
; RV32-NEXT: [[TMP18:%.*]] = fadd <vscale x 2 x double> [[WIDE_MASKED_GATHER6]], [[TMP17]]
; RV32-NEXT: [[TMP19:%.*]] = getelementptr inbounds double, ptr [[A]], <vscale x 2 x i64> [[VEC_IND]]
diff --git a/llvm/test/Transforms/LoopVectorize/RISCV/strided-accesses-narrow-iv.ll b/llvm/test/Transforms/LoopVectorize/RISCV/strided-accesses-narrow-iv.ll
new file mode 100644
index 0000000000000..cee73c92460b5
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/RISCV/strided-accesses-narrow-iv.ll
@@ -0,0 +1,279 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
+; RUN: opt < %s -passes=loop-vectorize -mtriple=riscv64 -mattr=+v -S | FileCheck --check-prefix=RV64 %s
+; RUN: opt < %s -passes=loop-vectorize -mtriple=riscv32 -mattr=+v -S | FileCheck --check-prefix=RV32 %s
+
+define void @narrow_iv_i8_sext_i64(ptr noalias %arr, ptr noalias %out) {
+; RV64-LABEL: define void @narrow_iv_i8_sext_i64(
+; RV64-SAME: ptr noalias [[ARR:%.*]], ptr noalias [[OUT:%.*]]) #[[ATTR0:[0-9]+]] {
+; RV64-NEXT: [[ENTRY:.*:]]
+; RV64-NEXT: br label %[[VECTOR_PH:.*]]
+; RV64: [[VECTOR_PH]]:
+; RV64-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 8 x ptr> poison, ptr [[OUT]], i64 0
+; RV64-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <vscale x 8 x ptr> [[BROADCAST_SPLATINSERT]], <vscale x 8 x ptr> poison, <vscale x 8 x i32> zeroinitializer
+; RV64-NEXT: br label %[[VECTOR_BODY:.*]]
+; RV64: [[VECTOR_BODY]]:
+; RV64-NEXT: [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[CURRENT_ITERATION_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; RV64-NEXT: [[AVL:%.*]] = phi i32 [ 16, %[[VECTOR_PH]] ], [ [[AVL_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; RV64-NEXT: [[TMP0:%.*]] = call i32 @llvm.experimental.get.vector.length.i32(i32 [[AVL]], i32 8, i1 true)
+; RV64-NEXT: [[TMP1:%.*]] = sext i32 [[INDEX]] to i64
+; RV64-NEXT: [[TMP5:%.*]] = shl i64 [[TMP1]], 12
+; RV64-NEXT: [[TMP2:%.*]] = getelementptr i8, ptr [[ARR]], i64 [[TMP5]]
+; RV64-NEXT: [[TMP3:%.*]] = call <vscale x 8 x i8> @llvm.experimental.vp.strided.load.nxv8i8.p0.i64(ptr align 1 [[TMP2]], i64 4096, <vscale x 8 x i1> splat (i1 true), i32 [[TMP0]])
+; RV64-NEXT: call void @llvm.vp.scatter.nxv8i8.nxv8p0(<vscale x 8 x i8> [[TMP3]], <vscale x 8 x ptr> align 1 [[BROADCAST_SPLAT]], <vscale x 8 x i1> splat (i1 true), i32 [[TMP0]])
+; RV64-NEXT: [[CURRENT_ITERATION_NEXT]] = add nuw i32 [[TMP0]], [[INDEX]]
+; RV64-NEXT: [[AVL_NEXT]] = sub nuw i32 [[AVL]], [[TMP0]]
+; RV64-NEXT: [[TMP4:%.*]] = icmp eq i32 [[AVL_NEXT]], 0
+; RV64-NEXT: br i1 [[TMP4]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; RV64: [[MIDDLE_BLOCK]]:
+; RV64-NEXT: br label %[[EXIT:.*]]
+; RV64: [[EXIT]]:
+; RV64-NEXT: ret void
+;
+; RV32-LABEL: define void @narrow_iv_i8_sext_i64(
+; RV32-SAME: ptr noalias [[ARR:%.*]], ptr noalias [[OUT:%.*]]) #[[ATTR0:[0-9]+]] {
+; RV32-NEXT: [[ENTRY:.*:]]
+; RV32-NEXT: br label %[[VECTOR_PH:.*]]
+; RV32: [[VECTOR_PH]]:
+; RV32-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 8 x ptr> poison, ptr [[OUT]], i64 0
+; RV32-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <vscale x 8 x ptr> [[BROADCAST_SPLATINSERT]], <vscale x 8 x ptr> poison, <vscale x 8 x i32> zeroinitializer
+; RV32-NEXT: br label %[[VECTOR_BODY:.*]]
+; RV32: [[VECTOR_BODY]]:
+; RV32-NEXT: [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[CURRENT_ITERATION_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; RV32-NEXT: [[AVL:%.*]] = phi i32 [ 16, %[[VECTOR_PH]] ], [ [[AVL_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; RV32-NEXT: [[TMP0:%.*]] = call i32 @llvm.experimental.get.vector.length.i32(i32 [[AVL]], i32 8, i1 true)
+; RV32-NEXT: [[TMP1:%.*]] = shl i32 [[INDEX]], 12
+; RV32-NEXT: [[TMP2:%.*]] = getelementptr i8, ptr [[ARR]], i32 [[TMP1]]
+; RV32-NEXT: [[TMP3:%.*]] = call <vscale x 8 x i8> @llvm.experimental.vp.strided.load.nxv8i8.p0.i32(ptr align 1 [[TMP2]], i32 4096, <vscale x 8 x i1> splat (i1 true), i32 [[TMP0]])
+; RV32-NEXT: call void @llvm.vp.scatter.nxv8i8.nxv8p0(<vscale x 8 x i8> [[TMP3]], <vscale x 8 x ptr> align 1 [[BROADCAST_SPLAT]], <vscale x 8 x i1> splat (i1 true), i32 [[TMP0]])
+; RV32-NEXT: [[CURRENT_ITERATION_NEXT]] = add nuw i32 [[TMP0]], [[INDEX]]
+; RV32-NEXT: [[AVL_NEXT]] = sub nuw i32 [[AVL]], [[TMP0]]
+; RV32-NEXT: [[TMP4:%.*]] = icmp eq i32 [[AVL_NEXT]], 0
+; RV32-NEXT: br i1 [[TMP4]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; RV32: [[MIDDLE_BLOCK]]:
+; RV32-NEXT: br label %[[EXIT:.*]]
+; RV32: [[EXIT]]:
+; RV32-NEXT: ret void
+;
+entry:
+ br label %loop
+
+loop:
+ %i = phi i8 [ 0, %entry ], [ %i.next, %loop ]
+ %idx = sext i8 %i to i64
+ %ptr = getelementptr [1024 x i8], ptr %arr, i64 %idx
+ %val = load i8, ptr %ptr, align 1
+ store i8 %val, ptr %out, align 1
+ %i.next = add i8 %i, 4
+ %cmp = icmp ult i8 %i.next, 64
+ br i1 %cmp, label %loop, label %exit
+
+exit:
+ ret void
+}
+
+define void @narrow_iv_i8_sext_i16(ptr noalias %arr, ptr noalias %out) {
+; RV64-LABEL: define void @narrow_iv_i8_sext_i16(
+; RV64-SAME: ptr noalias [[ARR:%.*]], ptr noalias [[OUT:%.*]]) #[[ATTR0]] {
+; RV64-NEXT: [[ENTRY:.*:]]
+; RV64-NEXT: br label %[[VECTOR_PH:.*]]
+; RV64: [[VECTOR_PH]]:
+; RV64-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 8 x ptr> poison, ptr [[OUT]], i64 0
+; RV64-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <vscale x 8 x ptr> [[BROADCAST_SPLATINSERT]], <vscale x 8 x ptr> poison, <vscale x 8 x i32> zeroinitializer
+; RV64-NEXT: br label %[[VECTOR_BODY:.*]]
+; RV64: [[VECTOR_BODY]]:
+; RV64-NEXT: [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[CURRENT_ITERATION_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; RV64-NEXT: [[AVL:%.*]] = phi i32 [ 16, %[[VECTOR_PH]] ], [ [[AVL_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; RV64-NEXT: [[TMP0:%.*]] = call i32 @llvm.experimental.get.vector.length.i32(i32 [[AVL]], i32 8, i1 true)
+; RV64-NEXT: [[TMP1:%.*]] = sext i32 [[INDEX]] to i64
+; RV64-NEXT: [[TMP5:%.*]] = shl i64 [[TMP1]], 12
+; RV64-NEXT: [[TMP2:%.*]] = getelementptr i8, ptr [[ARR]], i64 [[TMP5]]
+; RV64-NEXT: [[TMP3:%.*]] = call <vscale x 8 x i8> @llvm.experimental.vp.strided.load.nxv8i8.p0.i64(ptr align 1 [[TMP2]], i64 4096, <vscale x 8 x i1> splat (i1 true), i32 [[TMP0]])
+; RV64-NEXT: call void @llvm.vp.scatter.nxv8i8.nxv8p0(<vscale x 8 x i8> [[TMP3]], <vscale x 8 x ptr> align 1 [[BROADCAST_SPLAT]], <vscale x 8 x i1> splat (i1 true), i32 [[TMP0]])
+; RV64-NEXT: [[CURRENT_ITERATION_NEXT]] = add nuw i32 [[TMP0]], [[INDEX]]
+; RV64-NEXT: [[AVL_NEXT]] = sub nuw i32 [[AVL]], [[TMP0]]
+; RV64-NEXT: [[TMP4:%.*]] = icmp eq i32 [[AVL_NEXT]], 0
+; RV64-NEXT: br i1 [[TMP4]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP3:![0-9]+]]
+; RV64: [[MIDDLE_BLOCK]]:
+; RV64-NEXT: br label %[[EXIT:.*]]
+; RV64: [[EXIT]]:
+; RV64-NEXT: ret void
+;
+; RV32-LABEL: define void @narrow_iv_i8_sext_i16(
+; RV32-SAME: ptr noalias [[ARR:%.*]], ptr noalias [[OUT:%.*]]) #[[ATTR0]] {
+; RV32-NEXT: [[ENTRY:.*:]]
+; RV32-NEXT: br label %[[VECTOR_PH:.*]]
+; RV32: [[VECTOR_PH]]:
+; RV32-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 8 x ptr> poison, ptr [[OUT]], i64 0
+; RV32-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <vscale x 8 x ptr> [[BROADCAST_SPLATINSERT]], <vscale x 8 x ptr> poison, <vscale x 8 x i32> zeroinitializer
+; RV32-NEXT: br label %[[VECTOR_BODY:.*]]
+; RV32: [[VECTOR_BODY]]:
+; RV32-NEXT: [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[CURRENT_ITERATION_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; RV32-NEXT: [[AVL:%.*]] = phi i32 [ 16, %[[VECTOR_PH]] ], [ [[AVL_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; RV32-NEXT: [[TMP0:%.*]] = call i32 @llvm.experimental.get.vector.length.i32(i32 [[AVL]], i32 8, i1 true)
+; RV32-NEXT: [[TMP1:%.*]] = shl i32 [[INDEX]], 12
+; RV32-NEXT: [[TMP2:%.*]] = getelementptr i8, ptr [[ARR]], i32 [[TMP1]]
+; RV32-NEXT: [[TMP3:%.*]] = call <vscale x 8 x i8> @llvm.experimental.vp.strided.load.nxv8i8.p0.i32(ptr align 1 [[TMP2]], i32 4096, <vscale x 8 x i1> splat (i1 true), i32 [[TMP0]])
+; RV32-NEXT: call void @llvm.vp.scatter.nxv8i8.nxv8p0(<vscale x 8 x i8> [[TMP3]], <vscale x 8 x ptr> align 1 [[BROADCAST_SPLAT]], <vscale x 8 x i1> splat (i1 true), i32 [[TMP0]])
+; RV32-NEXT: [[CURRENT_ITERATION_NEXT]] = add nuw i32 [[TMP0]], [[INDEX]]
+; RV32-NEXT: [[AVL_NEXT]] = sub nuw i32 [[AVL]], [[TMP0]]
+; RV32-NEXT: [[TMP4:%.*]] = icmp eq i32 [[AVL_NEXT]], 0
+; RV32-NEXT: br i1 [[TMP4]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP3:![0-9]+]]
+; RV32: [[MIDDLE_BLOCK]]:
+; RV32-NEXT: br label %[[EXIT:.*]]
+; RV32: [[EXIT]]:
+; RV32-NEXT: ret void
+;
+entry:
+ br label %loop
+
+loop:
+ %i = phi i8 [ 0, %entry ], [ %i.next, %loop ]
+ %idx = sext i8 %i to i16
+ %ptr = getelementptr [1024 x i8], ptr %arr, i16 %idx
+ %val = load i8, ptr %ptr, align 1
+ store i8 %val, ptr %out, align 1
+ %i.next = add i8 %i, 4
+ %cmp = icmp ult i8 %i.next, 64
+ br i1 %cmp, label %loop, label %exit
+
+exit:
+ ret void
+}
+
+define void @narrow_iv_i8_zext_i64(ptr noalias %arr, ptr noalias %out) {
+; RV64-LABEL: define void @narrow_iv_i8_zext_i64(
+; RV64-SAME: ptr noalias [[ARR:%.*]], ptr noalias [[OUT:%.*]]) #[[ATTR0]] {
+; RV64-NEXT: [[ENTRY:.*:]]
+; RV64-NEXT: br label %[[VECTOR_PH:.*]]
+; RV64: [[VECTOR_PH]]:
+; RV64-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 8 x ptr> poison, ptr [[OUT]], i64 0
+; RV64-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <vscale x 8 x ptr> [[BROADCAST_SPLATINSERT]], <vscale x 8 x ptr> poison, <vscale x 8 x i32> zeroinitializer
+; RV64-NEXT: br label %[[VECTOR_BODY:.*]]
+; RV64: [[VECTOR_BODY]]:
+; RV64-NEXT: [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[CURRENT_ITERATION_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; RV64-NEXT: [[AVL:%.*]] = phi i32 [ 16, %[[VECTOR_PH]] ], [ [[AVL_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; RV64-NEXT: [[TMP0:%.*]] = call i32 @llvm.experimental.get.vector.length.i32(i32 [[AVL]], i32 8, i1 true)
+; RV64-NEXT: [[TMP1:%.*]] = sext i32 [[INDEX]] to i64
+; RV64-NEXT: [[TMP5:%.*]] = shl i64 [[TMP1]], 12
+; RV64-NEXT: [[TMP2:%.*]] = getelementptr i8, ptr [[ARR]], i64 [[TMP5]]
+; RV64-NEXT: [[TMP3:%.*]] = call <vscale x 8 x i8> @llvm.experimental.vp.strided.load.nxv8i8.p0.i64(ptr align 1 [[TMP2]], i64 4096, <vscale x 8 x i1> splat (i1 true), i32 [[TMP0]])
+; RV64-NEXT: call void @llvm.vp.scatter.nxv8i8.nxv8p0(<vscale x 8 x i8> [[TMP3]], <vscale x 8 x ptr> align 1 [[BROADCAST_SPLAT]], <vscale x 8 x i1> splat (i1 true), i32 [[TMP0]])
+; RV64-NEXT: [[CURRENT_ITERATION_NEXT]] = add nuw i32 [[TMP0]], [[INDEX]]
+; RV64-NEXT: [[AVL_NEXT]] = sub nuw i32 [[AVL]], [[TMP0]]
+; RV64-NEXT: [[TMP4:%.*]] = icmp eq i32 [[AVL_NEXT]], 0
+; RV64-NEXT: br i1 [[TMP4]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
+; RV64: [[MIDDLE_BLOCK]]:
+; RV64-NEXT: br label %[[EXIT:.*]]
+; RV64: [[EXIT]]:
+; RV64-NEXT: ret void
+;
+; RV32-LABEL: define void @narrow_iv_i8_zext_i64(
+; RV32-SAME: ptr noalias [[ARR:%.*]], ptr noalias [[OUT:%.*]]) #[[ATTR0]] {
+; RV32-NEXT: [[ENTRY:.*:]]
+; RV32-NEXT: br label %[[VECTOR_PH:.*]]
+; RV32: [[VECTOR_PH]]:
+; RV32-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 8 x ptr> poison, ptr [[OUT]], i64 0
+; RV32-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <vscale x 8 x ptr> [[BROADCAST_SPLATINSERT]], <vscale x 8 x ptr> poison, <vscale x 8 x i32> zeroinitializer
+; RV32-NEXT: br label %[[VECTOR_BODY:.*]]
+; RV32: [[VECTOR_BODY]]:
+; RV32-NEXT: [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[CURRENT_ITERATION_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; RV32-NEXT: [[AVL:%.*]] = phi i32 [ 16, %[[VECTOR_PH]] ], [ [[AVL_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; RV32-NEXT: [[TMP0:%.*]] = call i32 @llvm.experimental.get.vector.length.i32(i32 [[AVL]], i32 8, i1 true)
+; RV32-NEXT: [[TMP1:%.*]] = shl i32 [[INDEX]], 12
+; RV32-NEXT: [[TMP2:%.*]] = getelementptr i8, ptr [[ARR]], i32 [[TMP1]]
+; RV32-NEXT: [[TMP3:%.*]] = call <vscale x 8 x i8> @llvm.experimental.vp.strided.load.nxv8i8.p0.i32(ptr align 1 [[TMP2]], i32 4096, <vscale x 8 x i1> splat (i1 true), i32 [[TMP0]])
+; RV32-NEXT: call void @llvm.vp.scatter.nxv8i8.nxv8p0(<vscale x 8 x i8> [[TMP3]], <vscale x 8 x ptr> align 1 [[BROADCAST_SPLAT]], <vscale x 8 x i1> splat (i1 true), i32 [[TMP0]])
+; RV32-NEXT: [[CURRENT_ITERATION_NEXT]] = add nuw i32 [[TMP0]], [[INDEX]]
+; RV32-NEXT: [[AVL_NEXT]] = sub nuw i32 [[AVL]], [[TMP0]]
+; RV32-NEXT: [[TMP4:%.*]] = icmp eq i32 [[AVL_NEXT]], 0
+; RV32-NEXT: br i1 [[TMP4]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
+; RV32: [[MIDDLE_BLOCK]]:
+; RV32-NEXT: br label %[[EXIT:.*]]
+; RV32: [[EXIT]]:
+; RV32-NEXT: ret void
+;
+entry:
+ br label %loop
+
+loop:
+ %i = phi i8 [ 0, %entry ], [ %i.next, %loop ]
+ %idx = zext i8 %i to i64
+ %ptr = getelementptr [1024 x i8], ptr %arr, i64 %idx
+ %val = load i8, ptr %ptr, align 1
+ store i8 %val, ptr %out, align 1
+ %i.next = add i8 %i, 4
+ %cmp = icmp ult i8 %i.next, 64
+ br i1 %cmp, label %loop, label %exit
+
+exit:
+ ret void
+}
+
+define void @narrow_iv_i8_zext_i16(ptr noalias %arr, ptr noalias %out)...
[truncated]
``````````
</details>
https://github.com/llvm/llvm-project/pull/201070
More information about the llvm-commits
mailing list