[llvm] aeea5b1 - [LV] Add additional outer loop vectorization tests. (#225821)

via llvm-commits llvm-commits at lists.llvm.org
Wed Sep 23 11:49:41 PDT 2026


Author: Florian Hahn
Date: 2026-09-23T18:49:33Z
New Revision: aeea5b1050b9388a77cc749b70795fc8701af923

URL: https://github.com/llvm/llvm-project/commit/aeea5b1050b9388a77cc749b70795fc8701af923
DIFF: https://github.com/llvm/llvm-project/commit/aeea5b1050b9388a77cc749b70795fc8701af923.diff

LOG: [LV] Add additional outer loop vectorization tests. (#225821)

Add test more test coverage for outer loop vectorization path, focusing
on various legality aspects, like supported instructions and memory
checks.

Added: 
    llvm/test/Transforms/LoopVectorize/VPlan/vplan-printing-outer-loop-memchecks.ll
    llvm/test/Transforms/LoopVectorize/outer-loop-gep-nowrap-flags-scev.ll
    llvm/test/Transforms/LoopVectorize/outer-loop-memory-safety.ll

Modified: 
    llvm/test/Transforms/LoopVectorize/AArch64/outer_loop_prefer_scalable.ll
    llvm/test/Transforms/LoopVectorize/outer_loop_contiguous.ll

Removed: 
    


################################################################################
diff  --git a/llvm/test/Transforms/LoopVectorize/AArch64/outer_loop_prefer_scalable.ll b/llvm/test/Transforms/LoopVectorize/AArch64/outer_loop_prefer_scalable.ll
index 32b9f5e2337d6..ab198def82fa3 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/outer_loop_prefer_scalable.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/outer_loop_prefer_scalable.ll
@@ -1,5 +1,5 @@
 ; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
-; RUN: opt -S -mtriple aarch64 -mattr=+sve -passes=loop-vectorize -enable-vplan-native-path < %s | FileCheck %s
+; RUN: opt -mtriple aarch64 -mattr=+sve -passes=loop-vectorize -enable-vplan-native-path -S %s | FileCheck %s
 
 @A = external global [1024 x float], align 4
 @B = external global [512 x float], align 4
@@ -95,5 +95,102 @@ exit:
   ret void
 }
 
+; The lanes of the column-major store A[i + j*8] are only distinct for VF <= 8,
+; which a scalable VF cannot guarantee.
+define void @col_major_m8(ptr noalias %A, i64 %N) {
+; CHECK-LABEL: define void @col_major_m8(
+; CHECK-SAME: ptr noalias [[A:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-NEXT:    [[TMP1:%.*]] = shl nuw i64 [[TMP0]], 2
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], [[TMP1]]
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[N]], [[TMP1]]
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[N_MOD_VF]]
+; CHECK-NEXT:    [[TMP2:%.*]] = call <vscale x 4 x i64> @llvm.stepvector.nxv4i64()
+; CHECK-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 4 x i64> poison, i64 [[TMP1]], i64 0
+; CHECK-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <vscale x 4 x i64> [[BROADCAST_SPLATINSERT]], <vscale x 4 x i64> poison, <vscale x 4 x i32> zeroinitializer
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[OUTER_LATCH3:.*]] ]
+; CHECK-NEXT:    [[VEC_IND:%.*]] = phi <vscale x 4 x i64> [ [[TMP2]], %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[OUTER_LATCH3]] ]
+; CHECK-NEXT:    br label %[[INNER_BODY1:.*]]
+; CHECK:       [[INNER_BODY1]]:
+; CHECK-NEXT:    [[INNER_IV2:%.*]] = phi <vscale x 4 x i64> [ zeroinitializer, %[[VECTOR_BODY]] ], [ [[TMP7:%.*]], %[[INNER_BODY1]] ]
+; CHECK-NEXT:    [[TMP3:%.*]] = shl nsw <vscale x 4 x i64> [[INNER_IV2]], splat (i64 3)
+; CHECK-NEXT:    [[TMP4:%.*]] = add nsw <vscale x 4 x i64> [[VEC_IND]], [[TMP3]]
+; CHECK-NEXT:    [[WIDE_GEP:%.*]] = getelementptr inbounds float, ptr [[A]], <vscale x 4 x i64> [[TMP4]]
+; CHECK-NEXT:    [[WIDE_MASKED_GATHER:%.*]] = call <vscale x 4 x float> @llvm.masked.gather.nxv4f32.nxv4p0(<vscale x 4 x ptr> align 4 [[WIDE_GEP]], <vscale x 4 x i1> splat (i1 true), <vscale x 4 x float> poison)
+; CHECK-NEXT:    [[TMP5:%.*]] = sitofp <vscale x 4 x i64> [[VEC_IND]] to <vscale x 4 x float>
+; CHECK-NEXT:    [[TMP6:%.*]] = fsub <vscale x 4 x float> [[TMP5]], [[WIDE_MASKED_GATHER]]
+; CHECK-NEXT:    call void @llvm.masked.scatter.nxv4f32.nxv4p0(<vscale x 4 x float> [[TMP6]], <vscale x 4 x ptr> align 4 [[WIDE_GEP]], <vscale x 4 x i1> splat (i1 true))
+; CHECK-NEXT:    [[TMP7]] = add nuw nsw <vscale x 4 x i64> [[INNER_IV2]], splat (i64 1)
+; CHECK-NEXT:    [[TMP8:%.*]] = icmp eq <vscale x 4 x i64> [[TMP7]], splat (i64 8)
+; CHECK-NEXT:    [[TMP9:%.*]] = extractelement <vscale x 4 x i1> [[TMP8]], i64 0
+; CHECK-NEXT:    br i1 [[TMP9]], label %[[OUTER_LATCH3]], label %[[INNER_BODY1]]
+; CHECK:       [[OUTER_LATCH3]]:
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP1]]
+; CHECK-NEXT:    [[VEC_IND_NEXT]] = add nuw nsw <vscale x 4 x i64> [[VEC_IND]], [[BROADCAST_SPLAT]]
+; CHECK-NEXT:    [[TMP10:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP10]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; CHECK:       [[SCALAR_PH]]:
+; CHECK-NEXT:    [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT:    br label %[[OUTER_HEADER:.*]]
+; CHECK:       [[OUTER_HEADER]]:
+; CHECK-NEXT:    [[OUTER_IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[OUTER_IV_NEXT:%.*]], %[[OUTER_LATCH:.*]] ]
+; CHECK-NEXT:    br label %[[INNER_BODY:.*]]
+; CHECK:       [[INNER_BODY]]:
+; CHECK-NEXT:    [[INNER_IV:%.*]] = phi i64 [ 0, %[[OUTER_HEADER]] ], [ [[INNER_IV_NEXT:%.*]], %[[INNER_BODY]] ]
+; CHECK-NEXT:    [[INNER_IV_MUL_8:%.*]] = mul nsw i64 [[INNER_IV]], 8
+; CHECK-NEXT:    [[IDX:%.*]] = add nsw i64 [[OUTER_IV]], [[INNER_IV_MUL_8]]
+; CHECK-NEXT:    [[A_PTR:%.*]] = getelementptr inbounds float, ptr [[A]], i64 [[IDX]]
+; CHECK-NEXT:    [[A_VAL:%.*]] = load float, ptr [[A_PTR]], align 4
+; CHECK-NEXT:    [[OUTER_IV_FP:%.*]] = sitofp i64 [[OUTER_IV]] to float
+; CHECK-NEXT:    [[SUB:%.*]] = fsub float [[OUTER_IV_FP]], [[A_VAL]]
+; CHECK-NEXT:    store float [[SUB]], ptr [[A_PTR]], align 4
+; CHECK-NEXT:    [[INNER_IV_NEXT]] = add nuw nsw i64 [[INNER_IV]], 1
+; CHECK-NEXT:    [[INNER_IV_CMP:%.*]] = icmp eq i64 [[INNER_IV_NEXT]], 8
+; CHECK-NEXT:    br i1 [[INNER_IV_CMP]], label %[[OUTER_LATCH]], label %[[INNER_BODY]]
+; CHECK:       [[OUTER_LATCH]]:
+; CHECK-NEXT:    [[OUTER_IV_NEXT]] = add nuw nsw i64 [[OUTER_IV]], 1
+; CHECK-NEXT:    [[OUTER_IV_CMP:%.*]] = icmp eq i64 [[OUTER_IV_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[OUTER_IV_CMP]], label %[[EXIT]], label %[[OUTER_HEADER]], !llvm.loop [[LOOP5:![0-9]+]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %outer.header
+
+outer.header:
+  %outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
+  br label %inner.body
+
+inner.body:
+  %inner.iv = phi i64 [ 0, %outer.header ], [ %inner.iv.next, %inner.body ]
+  %inner.iv.mul.8 = mul nsw i64 %inner.iv, 8
+  %idx = add nsw i64 %outer.iv, %inner.iv.mul.8
+  %A.ptr = getelementptr inbounds float, ptr %A, i64 %idx
+  %A.val = load float, ptr %A.ptr, align 4
+  %outer.iv.fp = sitofp i64 %outer.iv to float
+  %sub = fsub float %outer.iv.fp, %A.val
+  store float %sub, ptr %A.ptr, align 4
+  %inner.iv.next = add nuw nsw i64 %inner.iv, 1
+  %inner.iv.cmp = icmp eq i64 %inner.iv.next, 8
+  br i1 %inner.iv.cmp, label %outer.latch, label %inner.body
+
+outer.latch:
+  %outer.iv.next = add nuw nsw i64 %outer.iv, 1
+  %outer.iv.cmp = icmp eq i64 %outer.iv.next, %N
+  br i1 %outer.iv.cmp, label %exit, label %outer.header, !llvm.loop !3
+
+exit:
+  ret void
+}
+
 !1 = distinct !{!1, !2}
 !2 = !{!"llvm.loop.vectorize.enable"}
+!3 = distinct !{!3, !2}

diff  --git a/llvm/test/Transforms/LoopVectorize/VPlan/vplan-printing-outer-loop-memchecks.ll b/llvm/test/Transforms/LoopVectorize/VPlan/vplan-printing-outer-loop-memchecks.ll
new file mode 100644
index 0000000000000..6e5e193c40d8f
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/VPlan/vplan-printing-outer-loop-memchecks.ll
@@ -0,0 +1,62 @@
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter-out-after "^vector.ph:" --version 6
+; RUN: opt -passes=loop-vectorize -enable-vplan-native-path -force-vector-width=4 -vplan-print-after=printOptimizedVPlan -disable-output %s 2>&1 | FileCheck %s --check-prefix=OPTIMIZED
+; RUN: opt -passes=loop-vectorize -enable-vplan-native-path -force-vector-width=4 -vplan-print-after=printFinalVPlan -disable-output %s 2>&1 | FileCheck %s --check-prefix=FINAL
+
+; The store may alias the load, but the memory both access over the whole nest
+; can be bounded.
+define void @may_alias_load(ptr %A, ptr %B, i64 %N, i64 %M) {
+; OPTIMIZED-LABEL: VPlan for loop in 'may_alias_load'
+; OPTIMIZED:  VPlan ' for VF={4},UF>=1' {
+; OPTIMIZED-NEXT:  Live-in vp<[[VP0:%[0-9]+]]> = VF
+; OPTIMIZED-NEXT:  Live-in vp<[[VP1:%[0-9]+]]> = VF * UF
+; OPTIMIZED-NEXT:  Live-in vp<[[VP2:%[0-9]+]]> = vector-trip-count
+; OPTIMIZED-NEXT:  Live-in ir<%N> = original trip-count
+; OPTIMIZED-EMPTY:
+; OPTIMIZED-NEXT:  ir-bb<entry>:
+; OPTIMIZED-NEXT:  Successor(s): scalar.ph, vector.ph
+; OPTIMIZED-EMPTY:
+; OPTIMIZED-NEXT:  vector.ph:
+;
+; FINAL-LABEL: VPlan for loop in 'may_alias_load'
+; FINAL:  VPlan 'Final VPlan for VF={4},UF={1}' {
+; FINAL-NEXT:  Live-in ir<%N> = original trip-count
+; FINAL-EMPTY:
+; FINAL-NEXT:  ir-bb<entry>:
+; FINAL-NEXT:    EMIT vp<%min.iters.check> = icmp ult ir<%N>, ir<4>
+; FINAL-NEXT:    EMIT branch-on-cond vp<%min.iters.check>
+; FINAL-NEXT:  Successor(s): ir-bb<scalar.ph>, vector.ph
+; FINAL-EMPTY:
+; FINAL-NEXT:  vector.ph:
+;
+entry:
+  br label %outer.header
+
+outer.header:
+  %outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
+  %outer.iv.mul.M = mul nsw i64 %outer.iv, %M
+  br label %inner.body
+
+inner.body:
+  %inner.iv = phi i64 [ 0, %outer.header ], [ %inner.iv.next, %inner.body ]
+  %sum = phi float [ 0.000000e+00, %outer.header ], [ %sum.next, %inner.body ]
+  %idx = add nsw i64 %outer.iv.mul.M, %inner.iv
+  %A.ptr = getelementptr inbounds float, ptr %A, i64 %idx
+  %A.val = load float, ptr %A.ptr, align 4
+  %sum.next = fadd float %sum, %A.val
+  %inner.iv.next = add nuw nsw i64 %inner.iv, 1
+  %inner.iv.cmp = icmp eq i64 %inner.iv.next, %M
+  br i1 %inner.iv.cmp, label %outer.latch, label %inner.body
+
+outer.latch:
+  %B.ptr = getelementptr inbounds float, ptr %B, i64 %outer.iv
+  store float %sum.next, ptr %B.ptr, align 4
+  %outer.iv.next = add nuw nsw i64 %outer.iv, 1
+  %outer.iv.cmp = icmp eq i64 %outer.iv.next, %N
+  br i1 %outer.iv.cmp, label %exit, label %outer.header, !llvm.loop !0
+
+exit:
+  ret void
+}
+
+!0 = distinct !{!0, !1}
+!1 = !{!"llvm.loop.vectorize.enable"}

diff  --git a/llvm/test/Transforms/LoopVectorize/outer-loop-gep-nowrap-flags-scev.ll b/llvm/test/Transforms/LoopVectorize/outer-loop-gep-nowrap-flags-scev.ll
new file mode 100644
index 0000000000000..59abff42fc089
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/outer-loop-gep-nowrap-flags-scev.ll
@@ -0,0 +1,147 @@
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter "getelementptr" --filter "A,\+,|B,\+,|A\),\+," --version 6
+; RUN: opt -passes='print<scalar-evolution>' -disable-output %s 2>&1 | FileCheck %s --check-prefix=BEFORE
+; RUN: opt -passes='loop-vectorize,print<scalar-evolution>' -enable-vplan-native-path -disable-output %s 2>&1 | FileCheck %s --check-prefix=AFTER
+
+; Make sure VPlan-based SCEV analysis does not leak any wrap flags, that are
+; not valid in the scalar loop.
+define void @cond_inbounds(ptr noalias %A, i64 %N, i64 %M, i1 %cond) {
+; BEFORE-LABEL: 'cond_inbounds'
+; BEFORE:    %A.ptr = getelementptr float, ptr %A, i64 %outer.iv
+; BEFORE:    --> {%A,+,4}<%outer.header> U: full-set S: full-set Exits: (-4 + (4 * %N) + %A) LoopDispositions: { %outer.header: Computable, %inner.body: Invariant }
+; BEFORE:    %A.cond.ptr = getelementptr inbounds float, ptr %A, i64 %outer.iv
+; BEFORE:    --> {%A,+,4}<%outer.header> U: full-set S: full-set Exits: (-4 + (4 * %N) + %A) LoopDispositions: { %outer.header: Computable, %inner.body: Invariant }
+;
+; AFTER-LABEL: 'cond_inbounds'
+; AFTER:    %1 = getelementptr float, ptr %A, i64 %index
+; AFTER:    --> {%A,+,16}<%vector.body> U: full-set S: full-set Exits: ((16 * ((-4 + (4 * (%N /u 4))<nuw>) /u 4)) + %A) LoopDispositions: { %vector.body: Computable, %inner.body3: Invariant }
+; AFTER:    %A.ptr = getelementptr float, ptr %A, i64 %outer.iv
+; AFTER:    --> {((4 * %bc.resume.val) + %A),+,4}<%outer.header> U: full-set S: full-set Exits: (-4 + (4 * %N) + %A) LoopDispositions: { %outer.header: Computable, %inner.body: Invariant }
+; AFTER:    %A.cond.ptr = getelementptr inbounds float, ptr %A, i64 %outer.iv
+; AFTER:    --> {((4 * %bc.resume.val) + %A),+,4}<%outer.header> U: full-set S: full-set Exits: (-4 + (4 * %N) + %A) LoopDispositions: { %outer.header: Computable, %inner.body: Invariant }
+;
+entry:
+  br label %outer.header
+
+outer.header:
+  %outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
+  %A.ptr = getelementptr float, ptr %A, i64 %outer.iv
+  store float 3.000000e+00, ptr %A.ptr, align 4, !llvm.access.group !3
+  br i1 %cond, label %outer.then, label %inner.ph
+
+outer.then:
+  %A.cond.ptr = getelementptr inbounds float, ptr %A, i64 %outer.iv
+  %A.val = load float, ptr %A.cond.ptr, align 4, !llvm.access.group !3
+  store float %A.val, ptr %A.ptr, align 4, !llvm.access.group !3
+  br label %inner.ph
+
+inner.ph:
+  br label %inner.body
+
+inner.body:
+  %inner.iv = phi i64 [ 0, %inner.ph ], [ %inner.iv.next, %inner.body ]
+  %inner.iv.next = add nuw nsw i64 %inner.iv, 1
+  %inner.iv.cmp = icmp eq i64 %inner.iv.next, %M
+  br i1 %inner.iv.cmp, label %outer.latch, label %inner.body
+
+outer.latch:
+  %outer.iv.next = add nuw nsw i64 %outer.iv, 1
+  %outer.iv.cmp = icmp eq i64 %outer.iv.next, %N
+  br i1 %outer.iv.cmp, label %exit, label %outer.header, !llvm.loop !0
+
+exit:
+  ret void
+}
+
+; Same as @cond_inbounds, but the inbounds GEP is executed unconditionally.
+define void @uncond_inbounds(ptr noalias %A, ptr noalias %B, i64 %N, i64 %M) {
+; BEFORE-LABEL: 'uncond_inbounds'
+; BEFORE:    %A.ptr = getelementptr inbounds float, ptr %A, i64 %outer.iv
+; BEFORE:    --> {%A,+,4}<nuw><%outer.header> U: full-set S: full-set Exits: (-4 + (4 * %N) + %A) LoopDispositions: { %outer.header: Computable, %inner.body: Invariant }
+; BEFORE:    %B.ptr = getelementptr inbounds float, ptr %B, i64 %inner.iv
+; BEFORE:    --> {%B,+,4}<nuw><%inner.body> U: full-set S: full-set Exits: (-4 + (4 * %M) + %B) LoopDispositions: { %inner.body: Computable, %outer.header: Uniform }
+;
+; AFTER-LABEL: 'uncond_inbounds'
+; AFTER:    %1 = getelementptr inbounds float, ptr %A, i64 %index
+; AFTER:    --> {%A,+,16}<%vector.body> U: full-set S: full-set Exits: ((16 * ((-4 + (4 * (%N /u 4))<nuw>) /u 4)) + %A) LoopDispositions: { %vector.body: Computable, %inner.body1: Invariant }
+; AFTER:    %A.ptr = getelementptr inbounds float, ptr %A, i64 %outer.iv
+; AFTER:    --> {((4 * %bc.resume.val) + %A),+,4}<%outer.header> U: full-set S: full-set Exits: (-4 + (4 * %N) + %A) LoopDispositions: { %outer.header: Computable, %inner.body: Invariant }
+; AFTER:    %B.ptr = getelementptr inbounds float, ptr %B, i64 %inner.iv
+; AFTER:    --> {%B,+,4}<nuw><%inner.body> U: full-set S: full-set Exits: (-4 + (4 * %M) + %B) LoopDispositions: { %inner.body: Computable, %outer.header: Uniform }
+;
+entry:
+  br label %outer.header
+
+outer.header:
+  %outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
+  %A.ptr = getelementptr inbounds float, ptr %A, i64 %outer.iv
+  store float 0.000000e+00, ptr %A.ptr, align 4, !llvm.access.group !3
+  br label %inner.body
+
+inner.body:
+  %inner.iv = phi i64 [ 0, %outer.header ], [ %inner.iv.next, %inner.body ]
+  %B.ptr = getelementptr inbounds float, ptr %B, i64 %inner.iv
+  %B.val = load float, ptr %B.ptr, align 4, !llvm.access.group !3
+  %A.val = load float, ptr %A.ptr, align 4, !llvm.access.group !3
+  %add = fadd float %A.val, %B.val
+  store float %add, ptr %A.ptr, align 4, !llvm.access.group !3
+  %inner.iv.next = add nuw nsw i64 %inner.iv, 1
+  %inner.iv.cmp = icmp eq i64 %inner.iv.next, %M
+  br i1 %inner.iv.cmp, label %outer.latch, label %inner.body
+
+outer.latch:
+  %outer.iv.next = add nuw nsw i64 %outer.iv, 1
+  %outer.iv.cmp = icmp eq i64 %outer.iv.next, %N
+  br i1 %outer.iv.cmp, label %exit, label %outer.header, !llvm.loop !0
+
+exit:
+  ret void
+}
+
+; The store address is a zero-offset GEP based on an inbounds GEP.
+define void @propagate_through_gep(ptr noalias %A, i64 %N, i64 %M) {
+; BEFORE-LABEL: 'propagate_through_gep'
+; BEFORE:    %A.ptr = getelementptr inbounds float, ptr %A, i64 %outer.iv
+; BEFORE:    --> {%A,+,4}<nuw><%outer.header> U: full-set S: full-set Exits: (-4 + (4 * %N) + %A) LoopDispositions: { %outer.header: Computable, %inner.body: Invariant }
+; BEFORE:    %A.ptr.0 = getelementptr i8, ptr %A.ptr, i64 0
+; BEFORE:    --> {%A,+,4}<nuw><%outer.header> U: full-set S: full-set Exits: (-4 + (4 * %N) + %A) LoopDispositions: { %outer.header: Computable, %inner.body: Invariant }
+;
+; AFTER-LABEL: 'propagate_through_gep'
+; AFTER:    %1 = getelementptr inbounds float, ptr %A, i64 %index
+; AFTER:    --> {%A,+,16}<%vector.body> U: full-set S: full-set Exits: ((16 * ((-4 + (4 * (%N /u 4))<nuw>) /u 4)) + %A) LoopDispositions: { %vector.body: Computable, %inner.body1: Invariant }
+; AFTER:    %2 = getelementptr i8, ptr %1, i64 0
+; AFTER:    --> {%A,+,16}<%vector.body> U: full-set S: full-set Exits: ((16 * ((-4 + (4 * (%N /u 4))<nuw>) /u 4)) + %A) LoopDispositions: { %vector.body: Computable, %inner.body1: Invariant }
+; AFTER:    %A.ptr = getelementptr inbounds float, ptr %A, i64 %outer.iv
+; AFTER:    --> {((4 * %bc.resume.val) + %A),+,4}<%outer.header> U: full-set S: full-set Exits: (-4 + (4 * %N) + %A) LoopDispositions: { %outer.header: Computable, %inner.body: Invariant }
+; AFTER:    %A.ptr.0 = getelementptr i8, ptr %A.ptr, i64 0
+; AFTER:    --> {((4 * %bc.resume.val) + %A),+,4}<%outer.header> U: full-set S: full-set Exits: (-4 + (4 * %N) + %A) LoopDispositions: { %outer.header: Computable, %inner.body: Invariant }
+;
+entry:
+  br label %outer.header
+
+outer.header:
+  %outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
+  %A.ptr = getelementptr inbounds float, ptr %A, i64 %outer.iv
+  %A.ptr.0 = getelementptr i8, ptr %A.ptr, i64 0
+  store float 0.000000e+00, ptr %A.ptr.0, align 4, !llvm.access.group !3
+  br label %inner.body
+
+inner.body:
+  %inner.iv = phi i64 [ 0, %outer.header ], [ %inner.iv.next, %inner.body ]
+  %inner.iv.next = add nuw nsw i64 %inner.iv, 1
+  %inner.iv.cmp = icmp eq i64 %inner.iv.next, %M
+  br i1 %inner.iv.cmp, label %outer.latch, label %inner.body
+
+outer.latch:
+  %outer.iv.next = add nuw nsw i64 %outer.iv, 1
+  %outer.iv.cmp = icmp eq i64 %outer.iv.next, %N
+  br i1 %outer.iv.cmp, label %exit, label %outer.header, !llvm.loop !0
+
+exit:
+  ret void
+}
+
+!0 = distinct !{!0, !1, !2, !4}
+!1 = !{!"llvm.loop.vectorize.width", i32 4}
+!2 = !{!"llvm.loop.vectorize.enable"}
+!3 = distinct !{}
+!4 = !{!"llvm.loop.parallel_accesses", !3}

diff  --git a/llvm/test/Transforms/LoopVectorize/outer-loop-memory-safety.ll b/llvm/test/Transforms/LoopVectorize/outer-loop-memory-safety.ll
new file mode 100644
index 0000000000000..b1337924219d4
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/outer-loop-memory-safety.ll
@@ -0,0 +1,1326 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --filter-out-after "^scalar.ph" --version 6
+; RUN: opt -passes=loop-vectorize -enable-vplan-native-path -force-vector-width=4 -S %s | FileCheck %s
+
+; The inner loop only reads, and the outer loop stores to a distinct object at a
+; unit-stride address.
+define void @inner_reads_outer_store(ptr noalias %A, ptr noalias %B, i64 %N, i64 %M) {
+; CHECK-LABEL: define void @inner_reads_outer_store(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], i64 [[N:%.*]], i64 [[M:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 4
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = and i64 [[N]], 3
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
+; CHECK-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x i64> poison, i64 [[M]], i64 0
+; CHECK-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i64> [[BROADCAST_SPLATINSERT]], <4 x i64> poison, <4 x i32> zeroinitializer
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[OUTER_LATCH4:.*]] ]
+; CHECK-NEXT:    [[VEC_IND:%.*]] = phi <4 x i64> [ <i64 0, i64 1, i64 2, i64 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[OUTER_LATCH4]] ]
+; CHECK-NEXT:    [[TMP1:%.*]] = mul nsw <4 x i64> [[VEC_IND]], [[BROADCAST_SPLAT]]
+; CHECK-NEXT:    br label %[[INNER_BODY1:.*]]
+; CHECK:       [[INNER_BODY1]]:
+; CHECK-NEXT:    [[INNER_IV2:%.*]] = phi <4 x i64> [ zeroinitializer, %[[VECTOR_BODY]] ], [ [[TMP4:%.*]], %[[INNER_BODY1]] ]
+; CHECK-NEXT:    [[SUM3:%.*]] = phi <4 x float> [ zeroinitializer, %[[VECTOR_BODY]] ], [ [[TMP3:%.*]], %[[INNER_BODY1]] ]
+; CHECK-NEXT:    [[TMP2:%.*]] = add nsw <4 x i64> [[TMP1]], [[INNER_IV2]]
+; CHECK-NEXT:    [[WIDE_GEP:%.*]] = getelementptr inbounds float, ptr [[A]], <4 x i64> [[TMP2]]
+; CHECK-NEXT:    [[WIDE_MASKED_GATHER:%.*]] = call <4 x float> @llvm.masked.gather.v4f32.v4p0(<4 x ptr> align 4 [[WIDE_GEP]], <4 x i1> splat (i1 true), <4 x float> poison)
+; CHECK-NEXT:    [[TMP3]] = fadd <4 x float> [[SUM3]], [[WIDE_MASKED_GATHER]]
+; CHECK-NEXT:    [[TMP4]] = add nuw nsw <4 x i64> [[INNER_IV2]], splat (i64 1)
+; CHECK-NEXT:    [[TMP5:%.*]] = icmp eq <4 x i64> [[TMP4]], [[BROADCAST_SPLAT]]
+; CHECK-NEXT:    [[TMP6:%.*]] = extractelement <4 x i1> [[TMP5]], i64 0
+; CHECK-NEXT:    br i1 [[TMP6]], label %[[OUTER_LATCH4]], label %[[INNER_BODY1]]
+; CHECK:       [[OUTER_LATCH4]]:
+; CHECK-NEXT:    [[TMP7:%.*]] = getelementptr inbounds float, ptr [[B]], i64 [[INDEX]]
+; CHECK-NEXT:    store <4 x float> [[TMP3]], ptr [[TMP7]], align 4
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT:    [[VEC_IND_NEXT]] = add nuw nsw <4 x i64> [[VEC_IND]], splat (i64 4)
+; CHECK-NEXT:    [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], [[EXIT:label %.*]], label %[[SCALAR_PH]]
+; CHECK:       [[SCALAR_PH]]:
+;
+entry:
+  br label %outer.header
+
+outer.header:
+  %outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
+  %outer.iv.mul.M = mul nsw i64 %outer.iv, %M
+  br label %inner.body
+
+inner.body:
+  %inner.iv = phi i64 [ 0, %outer.header ], [ %inner.iv.next, %inner.body ]
+  %sum = phi float [ 0.000000e+00, %outer.header ], [ %sum.next, %inner.body ]
+  %idx = add nsw i64 %outer.iv.mul.M, %inner.iv
+  %A.ptr = getelementptr inbounds float, ptr %A, i64 %idx
+  %A.val = load float, ptr %A.ptr, align 4
+  %sum.next = fadd float %sum, %A.val
+  %inner.iv.next = add nuw nsw i64 %inner.iv, 1
+  %inner.iv.cmp = icmp eq i64 %inner.iv.next, %M
+  br i1 %inner.iv.cmp, label %outer.latch, label %inner.body
+
+outer.latch:
+  %B.ptr = getelementptr inbounds float, ptr %B, i64 %outer.iv
+  store float %sum.next, ptr %B.ptr, align 4
+  %outer.iv.next = add nuw nsw i64 %outer.iv, 1
+  %outer.iv.cmp = icmp eq i64 %outer.iv.next, %N
+  br i1 %outer.iv.cmp, label %exit, label %outer.header, !llvm.loop !0
+
+exit:
+  ret void
+}
+
+; Each of the two stores writes a distinct object.
+define void @two_distinct_stores(ptr noalias %A, ptr noalias %B, ptr noalias %C, i64 %N, i64 %M) {
+; CHECK-LABEL: define void @two_distinct_stores(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]], i64 [[M:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 4
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = and i64 [[N]], 3
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
+; CHECK-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x i64> poison, i64 [[M]], i64 0
+; CHECK-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i64> [[BROADCAST_SPLATINSERT]], <4 x i64> poison, <4 x i32> zeroinitializer
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[OUTER_LATCH4:.*]] ]
+; CHECK-NEXT:    [[VEC_IND:%.*]] = phi <4 x i64> [ <i64 0, i64 1, i64 2, i64 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[OUTER_LATCH4]] ]
+; CHECK-NEXT:    [[TMP1:%.*]] = mul nsw <4 x i64> [[VEC_IND]], [[BROADCAST_SPLAT]]
+; CHECK-NEXT:    br label %[[INNER_BODY1:.*]]
+; CHECK:       [[INNER_BODY1]]:
+; CHECK-NEXT:    [[INNER_IV2:%.*]] = phi <4 x i64> [ zeroinitializer, %[[VECTOR_BODY]] ], [ [[TMP4:%.*]], %[[INNER_BODY1]] ]
+; CHECK-NEXT:    [[SUM3:%.*]] = phi <4 x float> [ zeroinitializer, %[[VECTOR_BODY]] ], [ [[TMP3:%.*]], %[[INNER_BODY1]] ]
+; CHECK-NEXT:    [[TMP2:%.*]] = add nsw <4 x i64> [[TMP1]], [[INNER_IV2]]
+; CHECK-NEXT:    [[WIDE_GEP:%.*]] = getelementptr inbounds float, ptr [[A]], <4 x i64> [[TMP2]]
+; CHECK-NEXT:    [[WIDE_MASKED_GATHER:%.*]] = call <4 x float> @llvm.masked.gather.v4f32.v4p0(<4 x ptr> align 4 [[WIDE_GEP]], <4 x i1> splat (i1 true), <4 x float> poison)
+; CHECK-NEXT:    [[TMP3]] = fadd <4 x float> [[SUM3]], [[WIDE_MASKED_GATHER]]
+; CHECK-NEXT:    [[TMP4]] = add nuw nsw <4 x i64> [[INNER_IV2]], splat (i64 1)
+; CHECK-NEXT:    [[TMP5:%.*]] = icmp eq <4 x i64> [[TMP4]], [[BROADCAST_SPLAT]]
+; CHECK-NEXT:    [[TMP6:%.*]] = extractelement <4 x i1> [[TMP5]], i64 0
+; CHECK-NEXT:    br i1 [[TMP6]], label %[[OUTER_LATCH4]], label %[[INNER_BODY1]]
+; CHECK:       [[OUTER_LATCH4]]:
+; CHECK-NEXT:    [[TMP7:%.*]] = getelementptr inbounds float, ptr [[B]], i64 [[INDEX]]
+; CHECK-NEXT:    store <4 x float> [[TMP3]], ptr [[TMP7]], align 4
+; CHECK-NEXT:    [[TMP8:%.*]] = getelementptr inbounds float, ptr [[C]], i64 [[INDEX]]
+; CHECK-NEXT:    store <4 x float> [[TMP3]], ptr [[TMP8]], align 4
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT:    [[VEC_IND_NEXT]] = add nuw nsw <4 x i64> [[VEC_IND]], splat (i64 4)
+; CHECK-NEXT:    [[TMP9:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP9]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], [[EXIT:label %.*]], label %[[SCALAR_PH]]
+; CHECK:       [[SCALAR_PH]]:
+;
+entry:
+  br label %outer.header
+
+outer.header:
+  %outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
+  %outer.iv.mul.M = mul nsw i64 %outer.iv, %M
+  br label %inner.body
+
+inner.body:
+  %inner.iv = phi i64 [ 0, %outer.header ], [ %inner.iv.next, %inner.body ]
+  %sum = phi float [ 0.000000e+00, %outer.header ], [ %sum.next, %inner.body ]
+  %idx = add nsw i64 %outer.iv.mul.M, %inner.iv
+  %A.ptr = getelementptr inbounds float, ptr %A, i64 %idx
+  %A.val = load float, ptr %A.ptr, align 4
+  %sum.next = fadd float %sum, %A.val
+  %inner.iv.next = add nuw nsw i64 %inner.iv, 1
+  %inner.iv.cmp = icmp eq i64 %inner.iv.next, %M
+  br i1 %inner.iv.cmp, label %outer.latch, label %inner.body
+
+outer.latch:
+  %B.ptr = getelementptr inbounds float, ptr %B, i64 %outer.iv
+  store float %sum.next, ptr %B.ptr, align 4
+  %C.ptr = getelementptr inbounds float, ptr %C, i64 %outer.iv
+  store float %sum.next, ptr %C.ptr, align 4
+  %outer.iv.next = add nuw nsw i64 %outer.iv, 1
+  %outer.iv.cmp = icmp eq i64 %outer.iv.next, %N
+  br i1 %outer.iv.cmp, label %exit, label %outer.header, !llvm.loop !0
+
+exit:
+  ret void
+}
+
+; The store advances by more than the bytes it writes, so the lanes write
+; disjoint elements.
+define void @store_strided(ptr noalias %A, ptr noalias %B, i64 %N, i64 %M) {
+; CHECK-LABEL: define void @store_strided(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], i64 [[N:%.*]], i64 [[M:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 4
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = and i64 [[N]], 3
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
+; CHECK-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x i64> poison, i64 [[M]], i64 0
+; CHECK-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i64> [[BROADCAST_SPLATINSERT]], <4 x i64> poison, <4 x i32> zeroinitializer
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[OUTER_LATCH4:.*]] ]
+; CHECK-NEXT:    [[VEC_IND:%.*]] = phi <4 x i64> [ <i64 0, i64 1, i64 2, i64 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[OUTER_LATCH4]] ]
+; CHECK-NEXT:    [[TMP1:%.*]] = mul nsw <4 x i64> [[VEC_IND]], [[BROADCAST_SPLAT]]
+; CHECK-NEXT:    br label %[[INNER_BODY1:.*]]
+; CHECK:       [[INNER_BODY1]]:
+; CHECK-NEXT:    [[INNER_IV2:%.*]] = phi <4 x i64> [ zeroinitializer, %[[VECTOR_BODY]] ], [ [[TMP4:%.*]], %[[INNER_BODY1]] ]
+; CHECK-NEXT:    [[SUM3:%.*]] = phi <4 x float> [ zeroinitializer, %[[VECTOR_BODY]] ], [ [[TMP3:%.*]], %[[INNER_BODY1]] ]
+; CHECK-NEXT:    [[TMP2:%.*]] = add nsw <4 x i64> [[TMP1]], [[INNER_IV2]]
+; CHECK-NEXT:    [[WIDE_GEP:%.*]] = getelementptr inbounds float, ptr [[A]], <4 x i64> [[TMP2]]
+; CHECK-NEXT:    [[WIDE_MASKED_GATHER:%.*]] = call <4 x float> @llvm.masked.gather.v4f32.v4p0(<4 x ptr> align 4 [[WIDE_GEP]], <4 x i1> splat (i1 true), <4 x float> poison)
+; CHECK-NEXT:    [[TMP3]] = fadd <4 x float> [[SUM3]], [[WIDE_MASKED_GATHER]]
+; CHECK-NEXT:    [[TMP4]] = add nuw nsw <4 x i64> [[INNER_IV2]], splat (i64 1)
+; CHECK-NEXT:    [[TMP5:%.*]] = icmp eq <4 x i64> [[TMP4]], [[BROADCAST_SPLAT]]
+; CHECK-NEXT:    [[TMP6:%.*]] = extractelement <4 x i1> [[TMP5]], i64 0
+; CHECK-NEXT:    br i1 [[TMP6]], label %[[OUTER_LATCH4]], label %[[INNER_BODY1]]
+; CHECK:       [[OUTER_LATCH4]]:
+; CHECK-NEXT:    [[TMP7:%.*]] = shl nsw <4 x i64> [[VEC_IND]], splat (i64 1)
+; CHECK-NEXT:    [[WIDE_GEP5:%.*]] = getelementptr inbounds float, ptr [[B]], <4 x i64> [[TMP7]]
+; CHECK-NEXT:    call void @llvm.masked.scatter.v4f32.v4p0(<4 x float> [[TMP3]], <4 x ptr> align 4 [[WIDE_GEP5]], <4 x i1> splat (i1 true))
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT:    [[VEC_IND_NEXT]] = add nuw nsw <4 x i64> [[VEC_IND]], splat (i64 4)
+; CHECK-NEXT:    [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP6:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], [[EXIT:label %.*]], label %[[SCALAR_PH]]
+; CHECK:       [[SCALAR_PH]]:
+;
+entry:
+  br label %outer.header
+
+outer.header:
+  %outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
+  %outer.iv.mul.M = mul nsw i64 %outer.iv, %M
+  br label %inner.body
+
+inner.body:
+  %inner.iv = phi i64 [ 0, %outer.header ], [ %inner.iv.next, %inner.body ]
+  %sum = phi float [ 0.000000e+00, %outer.header ], [ %sum.next, %inner.body ]
+  %idx = add nsw i64 %outer.iv.mul.M, %inner.iv
+  %A.ptr = getelementptr inbounds float, ptr %A, i64 %idx
+  %A.val = load float, ptr %A.ptr, align 4
+  %sum.next = fadd float %sum, %A.val
+  %inner.iv.next = add nuw nsw i64 %inner.iv, 1
+  %inner.iv.cmp = icmp eq i64 %inner.iv.next, %M
+  br i1 %inner.iv.cmp, label %outer.latch, label %inner.body
+
+outer.latch:
+  %outer.iv.mul.2 = mul nsw i64 %outer.iv, 2
+  %B.ptr = getelementptr inbounds float, ptr %B, i64 %outer.iv.mul.2
+  store float %sum.next, ptr %B.ptr, align 4
+  %outer.iv.next = add nuw nsw i64 %outer.iv, 1
+  %outer.iv.cmp = icmp eq i64 %outer.iv.next, %N
+  br i1 %outer.iv.cmp, label %exit, label %outer.header, !llvm.loop !0
+
+exit:
+  ret void
+}
+
+; The store may alias the load, but the memory both access over the whole nest
+; can be bounded.
+define void @may_alias_load(ptr %A, ptr %B, i64 %N, i64 %M) {
+; CHECK-LABEL: define void @may_alias_load(
+; CHECK-SAME: ptr [[A:%.*]], ptr [[B:%.*]], i64 [[N:%.*]], i64 [[M:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 4
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = and i64 [[N]], 3
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
+; CHECK-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x i64> poison, i64 [[M]], i64 0
+; CHECK-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i64> [[BROADCAST_SPLATINSERT]], <4 x i64> poison, <4 x i32> zeroinitializer
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[OUTER_LATCH4:.*]] ]
+; CHECK-NEXT:    [[VEC_IND:%.*]] = phi <4 x i64> [ <i64 0, i64 1, i64 2, i64 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[OUTER_LATCH4]] ]
+; CHECK-NEXT:    [[TMP1:%.*]] = mul nsw <4 x i64> [[VEC_IND]], [[BROADCAST_SPLAT]]
+; CHECK-NEXT:    br label %[[INNER_BODY1:.*]]
+; CHECK:       [[INNER_BODY1]]:
+; CHECK-NEXT:    [[INNER_IV2:%.*]] = phi <4 x i64> [ zeroinitializer, %[[VECTOR_BODY]] ], [ [[TMP4:%.*]], %[[INNER_BODY1]] ]
+; CHECK-NEXT:    [[SUM3:%.*]] = phi <4 x float> [ zeroinitializer, %[[VECTOR_BODY]] ], [ [[TMP3:%.*]], %[[INNER_BODY1]] ]
+; CHECK-NEXT:    [[TMP2:%.*]] = add nsw <4 x i64> [[TMP1]], [[INNER_IV2]]
+; CHECK-NEXT:    [[WIDE_GEP:%.*]] = getelementptr inbounds float, ptr [[A]], <4 x i64> [[TMP2]]
+; CHECK-NEXT:    [[WIDE_MASKED_GATHER:%.*]] = call <4 x float> @llvm.masked.gather.v4f32.v4p0(<4 x ptr> align 4 [[WIDE_GEP]], <4 x i1> splat (i1 true), <4 x float> poison)
+; CHECK-NEXT:    [[TMP3]] = fadd <4 x float> [[SUM3]], [[WIDE_MASKED_GATHER]]
+; CHECK-NEXT:    [[TMP4]] = add nuw nsw <4 x i64> [[INNER_IV2]], splat (i64 1)
+; CHECK-NEXT:    [[TMP5:%.*]] = icmp eq <4 x i64> [[TMP4]], [[BROADCAST_SPLAT]]
+; CHECK-NEXT:    [[TMP6:%.*]] = extractelement <4 x i1> [[TMP5]], i64 0
+; CHECK-NEXT:    br i1 [[TMP6]], label %[[OUTER_LATCH4]], label %[[INNER_BODY1]]
+; CHECK:       [[OUTER_LATCH4]]:
+; CHECK-NEXT:    [[TMP7:%.*]] = getelementptr inbounds float, ptr [[B]], i64 [[INDEX]]
+; CHECK-NEXT:    store <4 x float> [[TMP3]], ptr [[TMP7]], align 4
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT:    [[VEC_IND_NEXT]] = add nuw nsw <4 x i64> [[VEC_IND]], splat (i64 4)
+; CHECK-NEXT:    [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP8:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], [[EXIT:label %.*]], label %[[SCALAR_PH]]
+; CHECK:       [[SCALAR_PH]]:
+;
+entry:
+  br label %outer.header
+
+outer.header:
+  %outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
+  %outer.iv.mul.M = mul nsw i64 %outer.iv, %M
+  br label %inner.body
+
+inner.body:
+  %inner.iv = phi i64 [ 0, %outer.header ], [ %inner.iv.next, %inner.body ]
+  %sum = phi float [ 0.000000e+00, %outer.header ], [ %sum.next, %inner.body ]
+  %idx = add nsw i64 %outer.iv.mul.M, %inner.iv
+  %A.ptr = getelementptr inbounds float, ptr %A, i64 %idx
+  %A.val = load float, ptr %A.ptr, align 4
+  %sum.next = fadd float %sum, %A.val
+  %inner.iv.next = add nuw nsw i64 %inner.iv, 1
+  %inner.iv.cmp = icmp eq i64 %inner.iv.next, %M
+  br i1 %inner.iv.cmp, label %outer.latch, label %inner.body
+
+outer.latch:
+  %B.ptr = getelementptr inbounds float, ptr %B, i64 %outer.iv
+  store float %sum.next, ptr %B.ptr, align 4
+  %outer.iv.next = add nuw nsw i64 %outer.iv, 1
+  %outer.iv.cmp = icmp eq i64 %outer.iv.next, %N
+  br i1 %outer.iv.cmp, label %exit, label %outer.header, !llvm.loop !0
+
+exit:
+  ret void
+}
+
+; The store may alias both loads.
+define void @may_alias_two_loads(ptr %A, ptr %B, ptr %C, i64 %N, i64 %M) {
+; CHECK-LABEL: define void @may_alias_two_loads(
+; CHECK-SAME: ptr [[A:%.*]], ptr [[B:%.*]], ptr [[C:%.*]], i64 [[N:%.*]], i64 [[M:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 4
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = and i64 [[N]], 3
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
+; CHECK-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x i64> poison, i64 [[M]], i64 0
+; CHECK-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i64> [[BROADCAST_SPLATINSERT]], <4 x i64> poison, <4 x i32> zeroinitializer
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[OUTER_LATCH6:.*]] ]
+; CHECK-NEXT:    [[VEC_IND:%.*]] = phi <4 x i64> [ <i64 0, i64 1, i64 2, i64 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[OUTER_LATCH6]] ]
+; CHECK-NEXT:    [[TMP1:%.*]] = mul nsw <4 x i64> [[VEC_IND]], [[BROADCAST_SPLAT]]
+; CHECK-NEXT:    br label %[[INNER_BODY1:.*]]
+; CHECK:       [[INNER_BODY1]]:
+; CHECK-NEXT:    [[INNER_IV2:%.*]] = phi <4 x i64> [ zeroinitializer, %[[VECTOR_BODY]] ], [ [[TMP5:%.*]], %[[INNER_BODY1]] ]
+; CHECK-NEXT:    [[SUM3:%.*]] = phi <4 x float> [ zeroinitializer, %[[VECTOR_BODY]] ], [ [[TMP4:%.*]], %[[INNER_BODY1]] ]
+; CHECK-NEXT:    [[TMP2:%.*]] = add nsw <4 x i64> [[TMP1]], [[INNER_IV2]]
+; CHECK-NEXT:    [[WIDE_GEP:%.*]] = getelementptr inbounds float, ptr [[A]], <4 x i64> [[TMP2]]
+; CHECK-NEXT:    [[WIDE_MASKED_GATHER:%.*]] = call <4 x float> @llvm.masked.gather.v4f32.v4p0(<4 x ptr> align 4 [[WIDE_GEP]], <4 x i1> splat (i1 true), <4 x float> poison)
+; CHECK-NEXT:    [[WIDE_GEP4:%.*]] = getelementptr inbounds float, ptr [[C]], <4 x i64> [[TMP2]]
+; CHECK-NEXT:    [[WIDE_MASKED_GATHER5:%.*]] = call <4 x float> @llvm.masked.gather.v4f32.v4p0(<4 x ptr> align 4 [[WIDE_GEP4]], <4 x i1> splat (i1 true), <4 x float> poison)
+; CHECK-NEXT:    [[TMP3:%.*]] = fadd <4 x float> [[WIDE_MASKED_GATHER]], [[WIDE_MASKED_GATHER5]]
+; CHECK-NEXT:    [[TMP4]] = fadd <4 x float> [[SUM3]], [[TMP3]]
+; CHECK-NEXT:    [[TMP5]] = add nuw nsw <4 x i64> [[INNER_IV2]], splat (i64 1)
+; CHECK-NEXT:    [[TMP6:%.*]] = icmp eq <4 x i64> [[TMP5]], [[BROADCAST_SPLAT]]
+; CHECK-NEXT:    [[TMP7:%.*]] = extractelement <4 x i1> [[TMP6]], i64 0
+; CHECK-NEXT:    br i1 [[TMP7]], label %[[OUTER_LATCH6]], label %[[INNER_BODY1]]
+; CHECK:       [[OUTER_LATCH6]]:
+; CHECK-NEXT:    [[TMP8:%.*]] = getelementptr inbounds float, ptr [[B]], i64 [[INDEX]]
+; CHECK-NEXT:    store <4 x float> [[TMP4]], ptr [[TMP8]], align 4
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT:    [[VEC_IND_NEXT]] = add nuw nsw <4 x i64> [[VEC_IND]], splat (i64 4)
+; CHECK-NEXT:    [[TMP9:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP9]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP10:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], [[EXIT:label %.*]], label %[[SCALAR_PH]]
+; CHECK:       [[SCALAR_PH]]:
+;
+entry:
+  br label %outer.header
+
+outer.header:
+  %outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
+  %outer.iv.mul.M = mul nsw i64 %outer.iv, %M
+  br label %inner.body
+
+inner.body:
+  %inner.iv = phi i64 [ 0, %outer.header ], [ %inner.iv.next, %inner.body ]
+  %sum = phi float [ 0.000000e+00, %outer.header ], [ %sum.next, %inner.body ]
+  %idx = add nsw i64 %outer.iv.mul.M, %inner.iv
+  %A.ptr = getelementptr inbounds float, ptr %A, i64 %idx
+  %A.val = load float, ptr %A.ptr, align 4
+  %C.ptr = getelementptr inbounds float, ptr %C, i64 %idx
+  %C.val = load float, ptr %C.ptr, align 4
+  %add = fadd float %A.val, %C.val
+  %sum.next = fadd float %sum, %add
+  %inner.iv.next = add nuw nsw i64 %inner.iv, 1
+  %inner.iv.cmp = icmp eq i64 %inner.iv.next, %M
+  br i1 %inner.iv.cmp, label %outer.latch, label %inner.body
+
+outer.latch:
+  %B.ptr = getelementptr inbounds float, ptr %B, i64 %outer.iv
+  store float %sum.next, ptr %B.ptr, align 4
+  %outer.iv.next = add nuw nsw i64 %outer.iv, 1
+  %outer.iv.cmp = icmp eq i64 %outer.iv.next, %N
+  br i1 %outer.iv.cmp, label %exit, label %outer.header, !llvm.loop !0
+
+exit:
+  ret void
+}
+
+; The load may alias the store, and only dereferences the pointer of the last
+; inner iteration, so the inbounds GEPs of earlier iterations may be poison.
+define void @may_alias_load_after_inner_loop(ptr %A, ptr %B) {
+; CHECK-LABEL: define void @may_alias_load_after_inner_loop(
+; CHECK-SAME: ptr [[A:%.*]], ptr [[B:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    br label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    br label %[[INNER_BODY1:.*]]
+; CHECK:       [[INNER_BODY1]]:
+; CHECK-NEXT:    [[INNER_IV2:%.*]] = phi <4 x i64> [ zeroinitializer, %[[VECTOR_BODY]] ], [ [[TMP2:%.*]], %[[INNER_BODY1]] ]
+; CHECK-NEXT:    [[TMP0:%.*]] = mul <4 x i64> [[INNER_IV2]], splat (i64 9223372036854775807)
+; CHECK-NEXT:    [[TMP1:%.*]] = add <4 x i64> <i64 0, i64 1, i64 2, i64 3>, [[TMP0]]
+; CHECK-NEXT:    [[WIDE_GEP:%.*]] = getelementptr inbounds i8, ptr [[A]], <4 x i64> [[TMP1]]
+; CHECK-NEXT:    [[TMP2]] = add nuw nsw <4 x i64> [[INNER_IV2]], splat (i64 1)
+; CHECK-NEXT:    [[TMP3:%.*]] = icmp eq <4 x i64> [[TMP2]], splat (i64 3)
+; CHECK-NEXT:    [[TMP4:%.*]] = extractelement <4 x i1> [[TMP3]], i64 0
+; CHECK-NEXT:    br i1 [[TMP4]], label %[[OUTER_LATCH3:.*]], label %[[INNER_BODY1]], !llvm.loop [[LOOP12:![0-9]+]]
+; CHECK:       [[OUTER_LATCH3]]:
+; CHECK-NEXT:    [[WIDE_MASKED_GATHER:%.*]] = call <4 x i8> @llvm.masked.gather.v4i8.v4p0(<4 x ptr> align 1 [[WIDE_GEP]], <4 x i1> splat (i1 true), <4 x i8> poison)
+; CHECK-NEXT:    [[WIDE_GEP4:%.*]] = getelementptr inbounds i8, ptr [[B]], <4 x i64> <i64 0, i64 -1, i64 -2, i64 -3>
+; CHECK-NEXT:    call void @llvm.masked.scatter.v4i8.v4p0(<4 x i8> [[WIDE_MASKED_GATHER]], <4 x ptr> align 1 [[WIDE_GEP4]], <4 x i1> splat (i1 true))
+; CHECK-NEXT:    br label %[[MIDDLE_BLOCK:.*]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    br label %[[EXIT:.*]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %outer.header
+
+outer.header:
+  %outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
+  br label %inner.body
+
+inner.body:
+  %inner.iv = phi i64 [ 0, %outer.header ], [ %inner.iv.next, %inner.body ]
+  %inner.iv.mul.smax = mul i64 %inner.iv, 9223372036854775807
+  %idx = add i64 %outer.iv, %inner.iv.mul.smax
+  %A.ptr = getelementptr inbounds i8, ptr %A, i64 %idx
+  %inner.iv.next = add nuw nsw i64 %inner.iv, 1
+  %inner.iv.cmp = icmp eq i64 %inner.iv.next, 3
+  br i1 %inner.iv.cmp, label %outer.latch, label %inner.body
+
+outer.latch:
+  %A.ptr.lcssa = phi ptr [ %A.ptr, %inner.body ]
+  %A.val = load i8, ptr %A.ptr.lcssa, align 1
+  %outer.iv.neg = sub i64 0, %outer.iv
+  %B.ptr = getelementptr inbounds i8, ptr %B, i64 %outer.iv.neg
+  store i8 %A.val, ptr %B.ptr, align 1
+  %outer.iv.next = add nuw nsw i64 %outer.iv, 1
+  %outer.iv.cmp = icmp eq i64 %outer.iv.next, 4
+  br i1 %outer.iv.cmp, label %exit, label %outer.header, !llvm.loop !0
+
+exit:
+  ret void
+}
+
+; Same as @may_alias_load_after_inner_loop, but with noalias arguments.
+define void @load_after_inner_loop(ptr noalias %A, ptr noalias %B) {
+; CHECK-LABEL: define void @load_after_inner_loop(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    br label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    br label %[[INNER_BODY1:.*]]
+; CHECK:       [[INNER_BODY1]]:
+; CHECK-NEXT:    [[INNER_IV2:%.*]] = phi <4 x i64> [ zeroinitializer, %[[VECTOR_BODY]] ], [ [[TMP2:%.*]], %[[INNER_BODY1]] ]
+; CHECK-NEXT:    [[TMP0:%.*]] = mul <4 x i64> [[INNER_IV2]], splat (i64 9223372036854775807)
+; CHECK-NEXT:    [[TMP1:%.*]] = add <4 x i64> <i64 0, i64 1, i64 2, i64 3>, [[TMP0]]
+; CHECK-NEXT:    [[WIDE_GEP:%.*]] = getelementptr inbounds i8, ptr [[A]], <4 x i64> [[TMP1]]
+; CHECK-NEXT:    [[TMP2]] = add nuw nsw <4 x i64> [[INNER_IV2]], splat (i64 1)
+; CHECK-NEXT:    [[TMP3:%.*]] = icmp eq <4 x i64> [[TMP2]], splat (i64 3)
+; CHECK-NEXT:    [[TMP4:%.*]] = extractelement <4 x i1> [[TMP3]], i64 0
+; CHECK-NEXT:    br i1 [[TMP4]], label %[[OUTER_LATCH3:.*]], label %[[INNER_BODY1]], !llvm.loop [[LOOP13:![0-9]+]]
+; CHECK:       [[OUTER_LATCH3]]:
+; CHECK-NEXT:    [[WIDE_MASKED_GATHER:%.*]] = call <4 x i8> @llvm.masked.gather.v4i8.v4p0(<4 x ptr> align 1 [[WIDE_GEP]], <4 x i1> splat (i1 true), <4 x i8> poison)
+; CHECK-NEXT:    [[WIDE_GEP4:%.*]] = getelementptr inbounds i8, ptr [[B]], <4 x i64> <i64 0, i64 -1, i64 -2, i64 -3>
+; CHECK-NEXT:    call void @llvm.masked.scatter.v4i8.v4p0(<4 x i8> [[WIDE_MASKED_GATHER]], <4 x ptr> align 1 [[WIDE_GEP4]], <4 x i1> splat (i1 true))
+; CHECK-NEXT:    br label %[[MIDDLE_BLOCK:.*]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    br label %[[EXIT:.*]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %outer.header
+
+outer.header:
+  %outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
+  br label %inner.body
+
+inner.body:
+  %inner.iv = phi i64 [ 0, %outer.header ], [ %inner.iv.next, %inner.body ]
+  %inner.iv.mul.smax = mul i64 %inner.iv, 9223372036854775807
+  %idx = add i64 %outer.iv, %inner.iv.mul.smax
+  %A.ptr = getelementptr inbounds i8, ptr %A, i64 %idx
+  %inner.iv.next = add nuw nsw i64 %inner.iv, 1
+  %inner.iv.cmp = icmp eq i64 %inner.iv.next, 3
+  br i1 %inner.iv.cmp, label %outer.latch, label %inner.body
+
+outer.latch:
+  %A.ptr.lcssa = phi ptr [ %A.ptr, %inner.body ]
+  %A.val = load i8, ptr %A.ptr.lcssa, align 1
+  %outer.iv.neg = sub i64 0, %outer.iv
+  %B.ptr = getelementptr inbounds i8, ptr %B, i64 %outer.iv.neg
+  store i8 %A.val, ptr %B.ptr, align 1
+  %outer.iv.next = add nuw nsw i64 %outer.iv, 1
+  %outer.iv.cmp = icmp eq i64 %outer.iv.next, 4
+  br i1 %outer.iv.cmp, label %exit, label %outer.header, !llvm.loop !0
+
+exit:
+  ret void
+}
+
+; The load may alias the store, and its address is not formed by an inbounds
+; GEP.
+define void @may_alias_load_not_inbounds(ptr %A, ptr %B, i64 %N, i64 %M) {
+; CHECK-LABEL: define void @may_alias_load_not_inbounds(
+; CHECK-SAME: ptr [[A:%.*]], ptr [[B:%.*]], i64 [[N:%.*]], i64 [[M:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 4
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = and i64 [[N]], 3
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
+; CHECK-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x i64> poison, i64 [[M]], i64 0
+; CHECK-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i64> [[BROADCAST_SPLATINSERT]], <4 x i64> poison, <4 x i32> zeroinitializer
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[OUTER_LATCH4:.*]] ]
+; CHECK-NEXT:    [[VEC_IND:%.*]] = phi <4 x i64> [ <i64 0, i64 1, i64 2, i64 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[OUTER_LATCH4]] ]
+; CHECK-NEXT:    [[TMP1:%.*]] = mul nsw <4 x i64> [[VEC_IND]], [[BROADCAST_SPLAT]]
+; CHECK-NEXT:    br label %[[INNER_BODY1:.*]]
+; CHECK:       [[INNER_BODY1]]:
+; CHECK-NEXT:    [[INNER_IV2:%.*]] = phi <4 x i64> [ zeroinitializer, %[[VECTOR_BODY]] ], [ [[TMP4:%.*]], %[[INNER_BODY1]] ]
+; CHECK-NEXT:    [[SUM3:%.*]] = phi <4 x float> [ zeroinitializer, %[[VECTOR_BODY]] ], [ [[TMP3:%.*]], %[[INNER_BODY1]] ]
+; CHECK-NEXT:    [[TMP2:%.*]] = add nsw <4 x i64> [[TMP1]], [[INNER_IV2]]
+; CHECK-NEXT:    [[WIDE_GEP:%.*]] = getelementptr float, ptr [[A]], <4 x i64> [[TMP2]]
+; CHECK-NEXT:    [[WIDE_MASKED_GATHER:%.*]] = call <4 x float> @llvm.masked.gather.v4f32.v4p0(<4 x ptr> align 4 [[WIDE_GEP]], <4 x i1> splat (i1 true), <4 x float> poison)
+; CHECK-NEXT:    [[TMP3]] = fadd <4 x float> [[SUM3]], [[WIDE_MASKED_GATHER]]
+; CHECK-NEXT:    [[TMP4]] = add nuw nsw <4 x i64> [[INNER_IV2]], splat (i64 1)
+; CHECK-NEXT:    [[TMP5:%.*]] = icmp eq <4 x i64> [[TMP4]], [[BROADCAST_SPLAT]]
+; CHECK-NEXT:    [[TMP6:%.*]] = extractelement <4 x i1> [[TMP5]], i64 0
+; CHECK-NEXT:    br i1 [[TMP6]], label %[[OUTER_LATCH4]], label %[[INNER_BODY1]]
+; CHECK:       [[OUTER_LATCH4]]:
+; CHECK-NEXT:    [[TMP7:%.*]] = getelementptr inbounds float, ptr [[B]], i64 [[INDEX]]
+; CHECK-NEXT:    store <4 x float> [[TMP3]], ptr [[TMP7]], align 4
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT:    [[VEC_IND_NEXT]] = add nuw nsw <4 x i64> [[VEC_IND]], splat (i64 4)
+; CHECK-NEXT:    [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP14:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], [[EXIT:label %.*]], label %[[SCALAR_PH]]
+; CHECK:       [[SCALAR_PH]]:
+;
+entry:
+  br label %outer.header
+
+outer.header:
+  %outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
+  %outer.iv.mul.M = mul nsw i64 %outer.iv, %M
+  br label %inner.body
+
+inner.body:
+  %inner.iv = phi i64 [ 0, %outer.header ], [ %inner.iv.next, %inner.body ]
+  %sum = phi float [ 0.000000e+00, %outer.header ], [ %sum.next, %inner.body ]
+  %idx = add nsw i64 %outer.iv.mul.M, %inner.iv
+  %A.ptr = getelementptr float, ptr %A, i64 %idx
+  %A.val = load float, ptr %A.ptr, align 4
+  %sum.next = fadd float %sum, %A.val
+  %inner.iv.next = add nuw nsw i64 %inner.iv, 1
+  %inner.iv.cmp = icmp eq i64 %inner.iv.next, %M
+  br i1 %inner.iv.cmp, label %outer.latch, label %inner.body
+
+outer.latch:
+  %B.ptr = getelementptr inbounds float, ptr %B, i64 %outer.iv
+  store float %sum.next, ptr %B.ptr, align 4
+  %outer.iv.next = add nuw nsw i64 %outer.iv, 1
+  %outer.iv.cmp = icmp eq i64 %outer.iv.next, %N
+  br i1 %outer.iv.cmp, label %exit, label %outer.header, !llvm.loop !0
+
+exit:
+  ret void
+}
+
+; The load and the store may alias, but are in 
diff erent address spaces.
+define void @may_alias_
diff erent_address_spaces(ptr addrspace(1) %A, ptr %B, i64 %N, i64 %M) {
+; CHECK-LABEL: define void @may_alias_
diff erent_address_spaces(
+; CHECK-SAME: ptr addrspace(1) [[A:%.*]], ptr [[B:%.*]], i64 [[N:%.*]], i64 [[M:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 4
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = and i64 [[N]], 3
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
+; CHECK-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x i64> poison, i64 [[M]], i64 0
+; CHECK-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i64> [[BROADCAST_SPLATINSERT]], <4 x i64> poison, <4 x i32> zeroinitializer
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[OUTER_LATCH4:.*]] ]
+; CHECK-NEXT:    [[VEC_IND:%.*]] = phi <4 x i64> [ <i64 0, i64 1, i64 2, i64 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[OUTER_LATCH4]] ]
+; CHECK-NEXT:    [[TMP1:%.*]] = mul nsw <4 x i64> [[VEC_IND]], [[BROADCAST_SPLAT]]
+; CHECK-NEXT:    br label %[[INNER_BODY1:.*]]
+; CHECK:       [[INNER_BODY1]]:
+; CHECK-NEXT:    [[INNER_IV2:%.*]] = phi <4 x i64> [ zeroinitializer, %[[VECTOR_BODY]] ], [ [[TMP4:%.*]], %[[INNER_BODY1]] ]
+; CHECK-NEXT:    [[SUM3:%.*]] = phi <4 x float> [ zeroinitializer, %[[VECTOR_BODY]] ], [ [[TMP3:%.*]], %[[INNER_BODY1]] ]
+; CHECK-NEXT:    [[TMP2:%.*]] = add nsw <4 x i64> [[TMP1]], [[INNER_IV2]]
+; CHECK-NEXT:    [[WIDE_GEP:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[A]], <4 x i64> [[TMP2]]
+; CHECK-NEXT:    [[WIDE_MASKED_GATHER:%.*]] = call <4 x float> @llvm.masked.gather.v4f32.v4p1(<4 x ptr addrspace(1)> align 4 [[WIDE_GEP]], <4 x i1> splat (i1 true), <4 x float> poison)
+; CHECK-NEXT:    [[TMP3]] = fadd <4 x float> [[SUM3]], [[WIDE_MASKED_GATHER]]
+; CHECK-NEXT:    [[TMP4]] = add nuw nsw <4 x i64> [[INNER_IV2]], splat (i64 1)
+; CHECK-NEXT:    [[TMP5:%.*]] = icmp eq <4 x i64> [[TMP4]], [[BROADCAST_SPLAT]]
+; CHECK-NEXT:    [[TMP6:%.*]] = extractelement <4 x i1> [[TMP5]], i64 0
+; CHECK-NEXT:    br i1 [[TMP6]], label %[[OUTER_LATCH4]], label %[[INNER_BODY1]]
+; CHECK:       [[OUTER_LATCH4]]:
+; CHECK-NEXT:    [[TMP7:%.*]] = getelementptr inbounds float, ptr [[B]], i64 [[INDEX]]
+; CHECK-NEXT:    store <4 x float> [[TMP3]], ptr [[TMP7]], align 4
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT:    [[VEC_IND_NEXT]] = add nuw nsw <4 x i64> [[VEC_IND]], splat (i64 4)
+; CHECK-NEXT:    [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP16:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], [[EXIT:label %.*]], label %[[SCALAR_PH]]
+; CHECK:       [[SCALAR_PH]]:
+;
+entry:
+  br label %outer.header
+
+outer.header:
+  %outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
+  %outer.iv.mul.M = mul nsw i64 %outer.iv, %M
+  br label %inner.body
+
+inner.body:
+  %inner.iv = phi i64 [ 0, %outer.header ], [ %inner.iv.next, %inner.body ]
+  %sum = phi float [ 0.000000e+00, %outer.header ], [ %sum.next, %inner.body ]
+  %idx = add nsw i64 %outer.iv.mul.M, %inner.iv
+  %A.ptr = getelementptr inbounds float, ptr addrspace(1) %A, i64 %idx
+  %A.val = load float, ptr addrspace(1) %A.ptr, align 4
+  %sum.next = fadd float %sum, %A.val
+  %inner.iv.next = add nuw nsw i64 %inner.iv, 1
+  %inner.iv.cmp = icmp eq i64 %inner.iv.next, %M
+  br i1 %inner.iv.cmp, label %outer.latch, label %inner.body
+
+outer.latch:
+  %B.ptr = getelementptr inbounds float, ptr %B, i64 %outer.iv
+  store float %sum.next, ptr %B.ptr, align 4
+  %outer.iv.next = add nuw nsw i64 %outer.iv, 1
+  %outer.iv.cmp = icmp eq i64 %outer.iv.next, %N
+  br i1 %outer.iv.cmp, label %exit, label %outer.header, !llvm.loop !0
+
+exit:
+  ret void
+}
+
+; The load may alias the store, and is only executed conditionally.
+define void @may_alias_load_not_executed_every_iteration(ptr %A, ptr %B, i64 %N, i64 %M, i1 %cond) {
+; CHECK-LABEL: define void @may_alias_load_not_executed_every_iteration(
+; CHECK-SAME: ptr [[A:%.*]], ptr [[B:%.*]], i64 [[N:%.*]], i64 [[M:%.*]], i1 [[COND:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 4
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = and i64 [[N]], 3
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
+; CHECK-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x i64> poison, i64 [[M]], i64 0
+; CHECK-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i64> [[BROADCAST_SPLATINSERT]], <4 x i64> poison, <4 x i32> zeroinitializer
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[OUTER_LATCH7:.*]] ]
+; CHECK-NEXT:    [[VEC_IND:%.*]] = phi <4 x i64> [ <i64 0, i64 1, i64 2, i64 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[OUTER_LATCH7]] ]
+; CHECK-NEXT:    [[TMP1:%.*]] = mul nsw <4 x i64> [[VEC_IND]], [[BROADCAST_SPLAT]]
+; CHECK-NEXT:    br label %[[INNER_HEADER1:.*]]
+; CHECK:       [[INNER_HEADER1]]:
+; CHECK-NEXT:    [[INNER_IV2:%.*]] = phi <4 x i64> [ zeroinitializer, %[[VECTOR_BODY]] ], [ [[TMP4:%.*]], %[[INNER_LATCH5:.*]] ]
+; CHECK-NEXT:    [[SUM3:%.*]] = phi <4 x float> [ zeroinitializer, %[[VECTOR_BODY]] ], [ [[TMP3:%.*]], %[[INNER_LATCH5]] ]
+; CHECK-NEXT:    br i1 [[COND]], label %[[INNER_THEN4:.*]], label %[[INNER_LATCH5]]
+; CHECK:       [[INNER_THEN4]]:
+; CHECK-NEXT:    [[TMP2:%.*]] = add nsw <4 x i64> [[TMP1]], [[INNER_IV2]]
+; CHECK-NEXT:    [[WIDE_GEP:%.*]] = getelementptr inbounds float, ptr [[A]], <4 x i64> [[TMP2]]
+; CHECK-NEXT:    [[WIDE_MASKED_GATHER:%.*]] = call <4 x float> @llvm.masked.gather.v4f32.v4p0(<4 x ptr> align 4 [[WIDE_GEP]], <4 x i1> splat (i1 true), <4 x float> poison)
+; CHECK-NEXT:    br label %[[INNER_LATCH5]]
+; CHECK:       [[INNER_LATCH5]]:
+; CHECK-NEXT:    [[VAL6:%.*]] = phi <4 x float> [ [[WIDE_MASKED_GATHER]], %[[INNER_THEN4]] ], [ zeroinitializer, %[[INNER_HEADER1]] ]
+; CHECK-NEXT:    [[TMP3]] = fadd <4 x float> [[SUM3]], [[VAL6]]
+; CHECK-NEXT:    [[TMP4]] = add nuw nsw <4 x i64> [[INNER_IV2]], splat (i64 1)
+; CHECK-NEXT:    [[TMP5:%.*]] = icmp eq <4 x i64> [[TMP4]], [[BROADCAST_SPLAT]]
+; CHECK-NEXT:    [[TMP6:%.*]] = extractelement <4 x i1> [[TMP5]], i64 0
+; CHECK-NEXT:    br i1 [[TMP6]], label %[[OUTER_LATCH7]], label %[[INNER_HEADER1]]
+; CHECK:       [[OUTER_LATCH7]]:
+; CHECK-NEXT:    [[TMP7:%.*]] = getelementptr inbounds float, ptr [[B]], i64 [[INDEX]]
+; CHECK-NEXT:    store <4 x float> [[TMP3]], ptr [[TMP7]], align 4
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT:    [[VEC_IND_NEXT]] = add nuw nsw <4 x i64> [[VEC_IND]], splat (i64 4)
+; CHECK-NEXT:    [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP18:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], [[EXIT:label %.*]], label %[[SCALAR_PH]]
+; CHECK:       [[SCALAR_PH]]:
+;
+entry:
+  br label %outer.header
+
+outer.header:
+  %outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
+  %outer.iv.mul.M = mul nsw i64 %outer.iv, %M
+  br label %inner.header
+
+inner.header:
+  %inner.iv = phi i64 [ 0, %outer.header ], [ %inner.iv.next, %inner.latch ]
+  %sum = phi float [ 0.000000e+00, %outer.header ], [ %sum.next, %inner.latch ]
+  br i1 %cond, label %inner.then, label %inner.latch
+
+inner.then:
+  %idx = add nsw i64 %outer.iv.mul.M, %inner.iv
+  %A.ptr = getelementptr inbounds float, ptr %A, i64 %idx
+  %A.val = load float, ptr %A.ptr, align 4
+  br label %inner.latch
+
+inner.latch:
+  %val = phi float [ %A.val, %inner.then ], [ 0.000000e+00, %inner.header ]
+  %sum.next = fadd float %sum, %val
+  %inner.iv.next = add nuw nsw i64 %inner.iv, 1
+  %inner.iv.cmp = icmp eq i64 %inner.iv.next, %M
+  br i1 %inner.iv.cmp, label %outer.latch, label %inner.header
+
+outer.latch:
+  %B.ptr = getelementptr inbounds float, ptr %B, i64 %outer.iv
+  store float %sum.next, ptr %B.ptr, align 4
+  %outer.iv.next = add nuw nsw i64 %outer.iv, 1
+  %outer.iv.cmp = icmp eq i64 %outer.iv.next, %N
+  br i1 %outer.iv.cmp, label %exit, label %outer.header, !llvm.loop !0
+
+exit:
+  ret void
+}
+
+; The store address does not vary with the outer loop, so all lanes write the
+; same location.
+define void @store_outer_invariant(ptr noalias %A, ptr noalias %B, i64 %N, i64 %M) {
+; CHECK-LABEL: define void @store_outer_invariant(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], i64 [[N:%.*]], i64 [[M:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 4
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = and i64 [[N]], 3
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
+; CHECK-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x i64> poison, i64 [[M]], i64 0
+; CHECK-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i64> [[BROADCAST_SPLATINSERT]], <4 x i64> poison, <4 x i32> zeroinitializer
+; CHECK-NEXT:    [[BROADCAST_SPLATINSERT1:%.*]] = insertelement <4 x ptr> poison, ptr [[B]], i64 0
+; CHECK-NEXT:    [[BROADCAST_SPLAT2:%.*]] = shufflevector <4 x ptr> [[BROADCAST_SPLATINSERT1]], <4 x ptr> poison, <4 x i32> zeroinitializer
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[OUTER_LATCH6:.*]] ]
+; CHECK-NEXT:    [[VEC_IND:%.*]] = phi <4 x i64> [ <i64 0, i64 1, i64 2, i64 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[OUTER_LATCH6]] ]
+; CHECK-NEXT:    [[TMP1:%.*]] = mul nsw <4 x i64> [[VEC_IND]], [[BROADCAST_SPLAT]]
+; CHECK-NEXT:    br label %[[INNER_BODY3:.*]]
+; CHECK:       [[INNER_BODY3]]:
+; CHECK-NEXT:    [[INNER_IV4:%.*]] = phi <4 x i64> [ zeroinitializer, %[[VECTOR_BODY]] ], [ [[TMP4:%.*]], %[[INNER_BODY3]] ]
+; CHECK-NEXT:    [[SUM5:%.*]] = phi <4 x float> [ zeroinitializer, %[[VECTOR_BODY]] ], [ [[TMP3:%.*]], %[[INNER_BODY3]] ]
+; CHECK-NEXT:    [[TMP2:%.*]] = add nsw <4 x i64> [[TMP1]], [[INNER_IV4]]
+; CHECK-NEXT:    [[WIDE_GEP:%.*]] = getelementptr inbounds float, ptr [[A]], <4 x i64> [[TMP2]]
+; CHECK-NEXT:    [[WIDE_MASKED_GATHER:%.*]] = call <4 x float> @llvm.masked.gather.v4f32.v4p0(<4 x ptr> align 4 [[WIDE_GEP]], <4 x i1> splat (i1 true), <4 x float> poison)
+; CHECK-NEXT:    [[TMP3]] = fadd <4 x float> [[SUM5]], [[WIDE_MASKED_GATHER]]
+; CHECK-NEXT:    [[TMP4]] = add nuw nsw <4 x i64> [[INNER_IV4]], splat (i64 1)
+; CHECK-NEXT:    [[TMP5:%.*]] = icmp eq <4 x i64> [[TMP4]], [[BROADCAST_SPLAT]]
+; CHECK-NEXT:    [[TMP6:%.*]] = extractelement <4 x i1> [[TMP5]], i64 0
+; CHECK-NEXT:    br i1 [[TMP6]], label %[[OUTER_LATCH6]], label %[[INNER_BODY3]]
+; CHECK:       [[OUTER_LATCH6]]:
+; CHECK-NEXT:    call void @llvm.masked.scatter.v4f32.v4p0(<4 x float> [[TMP3]], <4 x ptr> align 4 [[BROADCAST_SPLAT2]], <4 x i1> splat (i1 true))
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT:    [[VEC_IND_NEXT]] = add nuw nsw <4 x i64> [[VEC_IND]], splat (i64 4)
+; CHECK-NEXT:    [[TMP7:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP7]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP20:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], [[EXIT:label %.*]], label %[[SCALAR_PH]]
+; CHECK:       [[SCALAR_PH]]:
+;
+entry:
+  br label %outer.header
+
+outer.header:
+  %outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
+  %outer.iv.mul.M = mul nsw i64 %outer.iv, %M
+  br label %inner.body
+
+inner.body:
+  %inner.iv = phi i64 [ 0, %outer.header ], [ %inner.iv.next, %inner.body ]
+  %sum = phi float [ 0.000000e+00, %outer.header ], [ %sum.next, %inner.body ]
+  %idx = add nsw i64 %outer.iv.mul.M, %inner.iv
+  %A.ptr = getelementptr inbounds float, ptr %A, i64 %idx
+  %A.val = load float, ptr %A.ptr, align 4
+  %sum.next = fadd float %sum, %A.val
+  %inner.iv.next = add nuw nsw i64 %inner.iv, 1
+  %inner.iv.cmp = icmp eq i64 %inner.iv.next, %M
+  br i1 %inner.iv.cmp, label %outer.latch, label %inner.body
+
+outer.latch:
+  store float %sum.next, ptr %B, align 4
+  %outer.iv.next = add nuw nsw i64 %outer.iv, 1
+  %outer.iv.cmp = icmp eq i64 %outer.iv.next, %N
+  br i1 %outer.iv.cmp, label %exit, label %outer.header, !llvm.loop !0
+
+exit:
+  ret void
+}
+
+; The store address is not formed by an inbounds GEP, so it may wrap and two
+; outer iterations may write the same location.
+define void @store_not_inbounds(ptr noalias %A, ptr noalias %B, i64 %N, i64 %M) {
+; CHECK-LABEL: define void @store_not_inbounds(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], i64 [[N:%.*]], i64 [[M:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 4
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = and i64 [[N]], 3
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
+; CHECK-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x i64> poison, i64 [[M]], i64 0
+; CHECK-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i64> [[BROADCAST_SPLATINSERT]], <4 x i64> poison, <4 x i32> zeroinitializer
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[OUTER_LATCH4:.*]] ]
+; CHECK-NEXT:    [[VEC_IND:%.*]] = phi <4 x i64> [ <i64 0, i64 1, i64 2, i64 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[OUTER_LATCH4]] ]
+; CHECK-NEXT:    [[TMP1:%.*]] = mul nsw <4 x i64> [[VEC_IND]], [[BROADCAST_SPLAT]]
+; CHECK-NEXT:    br label %[[INNER_BODY1:.*]]
+; CHECK:       [[INNER_BODY1]]:
+; CHECK-NEXT:    [[INNER_IV2:%.*]] = phi <4 x i64> [ zeroinitializer, %[[VECTOR_BODY]] ], [ [[TMP4:%.*]], %[[INNER_BODY1]] ]
+; CHECK-NEXT:    [[SUM3:%.*]] = phi <4 x float> [ zeroinitializer, %[[VECTOR_BODY]] ], [ [[TMP3:%.*]], %[[INNER_BODY1]] ]
+; CHECK-NEXT:    [[TMP2:%.*]] = add nsw <4 x i64> [[TMP1]], [[INNER_IV2]]
+; CHECK-NEXT:    [[WIDE_GEP:%.*]] = getelementptr inbounds float, ptr [[A]], <4 x i64> [[TMP2]]
+; CHECK-NEXT:    [[WIDE_MASKED_GATHER:%.*]] = call <4 x float> @llvm.masked.gather.v4f32.v4p0(<4 x ptr> align 4 [[WIDE_GEP]], <4 x i1> splat (i1 true), <4 x float> poison)
+; CHECK-NEXT:    [[TMP3]] = fadd <4 x float> [[SUM3]], [[WIDE_MASKED_GATHER]]
+; CHECK-NEXT:    [[TMP4]] = add nuw nsw <4 x i64> [[INNER_IV2]], splat (i64 1)
+; CHECK-NEXT:    [[TMP5:%.*]] = icmp eq <4 x i64> [[TMP4]], [[BROADCAST_SPLAT]]
+; CHECK-NEXT:    [[TMP6:%.*]] = extractelement <4 x i1> [[TMP5]], i64 0
+; CHECK-NEXT:    br i1 [[TMP6]], label %[[OUTER_LATCH4]], label %[[INNER_BODY1]]
+; CHECK:       [[OUTER_LATCH4]]:
+; CHECK-NEXT:    [[TMP7:%.*]] = getelementptr float, ptr [[B]], i64 [[INDEX]]
+; CHECK-NEXT:    store <4 x float> [[TMP3]], ptr [[TMP7]], align 4
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT:    [[VEC_IND_NEXT]] = add nuw nsw <4 x i64> [[VEC_IND]], splat (i64 4)
+; CHECK-NEXT:    [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP22:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], [[EXIT:label %.*]], label %[[SCALAR_PH]]
+; CHECK:       [[SCALAR_PH]]:
+;
+entry:
+  br label %outer.header
+
+outer.header:
+  %outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
+  %outer.iv.mul.M = mul nsw i64 %outer.iv, %M
+  br label %inner.body
+
+inner.body:
+  %inner.iv = phi i64 [ 0, %outer.header ], [ %inner.iv.next, %inner.body ]
+  %sum = phi float [ 0.000000e+00, %outer.header ], [ %sum.next, %inner.body ]
+  %idx = add nsw i64 %outer.iv.mul.M, %inner.iv
+  %A.ptr = getelementptr inbounds float, ptr %A, i64 %idx
+  %A.val = load float, ptr %A.ptr, align 4
+  %sum.next = fadd float %sum, %A.val
+  %inner.iv.next = add nuw nsw i64 %inner.iv, 1
+  %inner.iv.cmp = icmp eq i64 %inner.iv.next, %M
+  br i1 %inner.iv.cmp, label %outer.latch, label %inner.body
+
+outer.latch:
+  %B.ptr = getelementptr float, ptr %B, i64 %outer.iv
+  store float %sum.next, ptr %B.ptr, align 4
+  %outer.iv.next = add nuw nsw i64 %outer.iv, 1
+  %outer.iv.cmp = icmp eq i64 %outer.iv.next, %N
+  br i1 %outer.iv.cmp, label %exit, label %outer.header, !llvm.loop !0
+
+exit:
+  ret void
+}
+
+; The store advances by 4 bytes but writes 8, so adjacent lanes overlap.
+define void @store_wider_than_stride(ptr noalias %A, ptr noalias %B, i64 %N, i64 %M) {
+; CHECK-LABEL: define void @store_wider_than_stride(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], i64 [[N:%.*]], i64 [[M:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 4
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = and i64 [[N]], 3
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
+; CHECK-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x i64> poison, i64 [[M]], i64 0
+; CHECK-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i64> [[BROADCAST_SPLATINSERT]], <4 x i64> poison, <4 x i32> zeroinitializer
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[OUTER_LATCH4:.*]] ]
+; CHECK-NEXT:    [[VEC_IND:%.*]] = phi <4 x i64> [ <i64 0, i64 1, i64 2, i64 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[OUTER_LATCH4]] ]
+; CHECK-NEXT:    [[TMP1:%.*]] = mul nsw <4 x i64> [[VEC_IND]], [[BROADCAST_SPLAT]]
+; CHECK-NEXT:    br label %[[INNER_BODY1:.*]]
+; CHECK:       [[INNER_BODY1]]:
+; CHECK-NEXT:    [[INNER_IV2:%.*]] = phi <4 x i64> [ zeroinitializer, %[[VECTOR_BODY]] ], [ [[TMP4:%.*]], %[[INNER_BODY1]] ]
+; CHECK-NEXT:    [[SUM3:%.*]] = phi <4 x float> [ zeroinitializer, %[[VECTOR_BODY]] ], [ [[TMP3:%.*]], %[[INNER_BODY1]] ]
+; CHECK-NEXT:    [[TMP2:%.*]] = add nsw <4 x i64> [[TMP1]], [[INNER_IV2]]
+; CHECK-NEXT:    [[WIDE_GEP:%.*]] = getelementptr inbounds float, ptr [[A]], <4 x i64> [[TMP2]]
+; CHECK-NEXT:    [[WIDE_MASKED_GATHER:%.*]] = call <4 x float> @llvm.masked.gather.v4f32.v4p0(<4 x ptr> align 4 [[WIDE_GEP]], <4 x i1> splat (i1 true), <4 x float> poison)
+; CHECK-NEXT:    [[TMP3]] = fadd <4 x float> [[SUM3]], [[WIDE_MASKED_GATHER]]
+; CHECK-NEXT:    [[TMP4]] = add nuw nsw <4 x i64> [[INNER_IV2]], splat (i64 1)
+; CHECK-NEXT:    [[TMP5:%.*]] = icmp eq <4 x i64> [[TMP4]], [[BROADCAST_SPLAT]]
+; CHECK-NEXT:    [[TMP6:%.*]] = extractelement <4 x i1> [[TMP5]], i64 0
+; CHECK-NEXT:    br i1 [[TMP6]], label %[[OUTER_LATCH4]], label %[[INNER_BODY1]]
+; CHECK:       [[OUTER_LATCH4]]:
+; CHECK-NEXT:    [[WIDE_GEP5:%.*]] = getelementptr inbounds float, ptr [[B]], <4 x i64> [[VEC_IND]]
+; CHECK-NEXT:    [[TMP7:%.*]] = fpext <4 x float> [[TMP3]] to <4 x double>
+; CHECK-NEXT:    call void @llvm.masked.scatter.v4f64.v4p0(<4 x double> [[TMP7]], <4 x ptr> align 4 [[WIDE_GEP5]], <4 x i1> splat (i1 true))
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT:    [[VEC_IND_NEXT]] = add nuw nsw <4 x i64> [[VEC_IND]], splat (i64 4)
+; CHECK-NEXT:    [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP24:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], [[EXIT:label %.*]], label %[[SCALAR_PH]]
+; CHECK:       [[SCALAR_PH]]:
+;
+entry:
+  br label %outer.header
+
+outer.header:
+  %outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
+  %outer.iv.mul.M = mul nsw i64 %outer.iv, %M
+  br label %inner.body
+
+inner.body:
+  %inner.iv = phi i64 [ 0, %outer.header ], [ %inner.iv.next, %inner.body ]
+  %sum = phi float [ 0.000000e+00, %outer.header ], [ %sum.next, %inner.body ]
+  %idx = add nsw i64 %outer.iv.mul.M, %inner.iv
+  %A.ptr = getelementptr inbounds float, ptr %A, i64 %idx
+  %A.val = load float, ptr %A.ptr, align 4
+  %sum.next = fadd float %sum, %A.val
+  %inner.iv.next = add nuw nsw i64 %inner.iv, 1
+  %inner.iv.cmp = icmp eq i64 %inner.iv.next, %M
+  br i1 %inner.iv.cmp, label %outer.latch, label %inner.body
+
+outer.latch:
+  %B.ptr = getelementptr inbounds float, ptr %B, i64 %outer.iv
+  %sum.ext = fpext float %sum.next to double
+  store double %sum.ext, ptr %B.ptr, align 4
+  %outer.iv.next = add nuw nsw i64 %outer.iv, 1
+  %outer.iv.cmp = icmp eq i64 %outer.iv.next, %N
+  br i1 %outer.iv.cmp, label %exit, label %outer.header, !llvm.loop !0
+
+exit:
+  ret void
+}
+
+; The inner-loop store writes a distinct row on each outer iteration.
+define void @store_in_inner_loop(ptr noalias %A, ptr noalias %B, i64 %N, i64 %M) {
+; CHECK-LABEL: define void @store_in_inner_loop(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], i64 [[N:%.*]], i64 [[M:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 4
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = and i64 [[N]], 3
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
+; CHECK-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x i64> poison, i64 [[M]], i64 0
+; CHECK-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i64> [[BROADCAST_SPLATINSERT]], <4 x i64> poison, <4 x i32> zeroinitializer
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[OUTER_LATCH4:.*]] ]
+; CHECK-NEXT:    [[VEC_IND:%.*]] = phi <4 x i64> [ <i64 0, i64 1, i64 2, i64 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[OUTER_LATCH4]] ]
+; CHECK-NEXT:    [[TMP1:%.*]] = mul nsw <4 x i64> [[VEC_IND]], [[BROADCAST_SPLAT]]
+; CHECK-NEXT:    br label %[[INNER_BODY1:.*]]
+; CHECK:       [[INNER_BODY1]]:
+; CHECK-NEXT:    [[INNER_IV2:%.*]] = phi <4 x i64> [ zeroinitializer, %[[VECTOR_BODY]] ], [ [[TMP3:%.*]], %[[INNER_BODY1]] ]
+; CHECK-NEXT:    [[TMP2:%.*]] = add nsw <4 x i64> [[TMP1]], [[INNER_IV2]]
+; CHECK-NEXT:    [[WIDE_GEP:%.*]] = getelementptr inbounds float, ptr [[A]], <4 x i64> [[TMP2]]
+; CHECK-NEXT:    [[WIDE_MASKED_GATHER:%.*]] = call <4 x float> @llvm.masked.gather.v4f32.v4p0(<4 x ptr> align 4 [[WIDE_GEP]], <4 x i1> splat (i1 true), <4 x float> poison)
+; CHECK-NEXT:    [[WIDE_GEP3:%.*]] = getelementptr inbounds float, ptr [[B]], <4 x i64> [[TMP2]]
+; CHECK-NEXT:    call void @llvm.masked.scatter.v4f32.v4p0(<4 x float> [[WIDE_MASKED_GATHER]], <4 x ptr> align 4 [[WIDE_GEP3]], <4 x i1> splat (i1 true))
+; CHECK-NEXT:    [[TMP3]] = add nuw nsw <4 x i64> [[INNER_IV2]], splat (i64 1)
+; CHECK-NEXT:    [[TMP4:%.*]] = icmp eq <4 x i64> [[TMP3]], [[BROADCAST_SPLAT]]
+; CHECK-NEXT:    [[TMP5:%.*]] = extractelement <4 x i1> [[TMP4]], i64 0
+; CHECK-NEXT:    br i1 [[TMP5]], label %[[OUTER_LATCH4]], label %[[INNER_BODY1]]
+; CHECK:       [[OUTER_LATCH4]]:
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT:    [[VEC_IND_NEXT]] = add nuw nsw <4 x i64> [[VEC_IND]], splat (i64 4)
+; CHECK-NEXT:    [[TMP6:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP6]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP26:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], [[EXIT:label %.*]], label %[[SCALAR_PH]]
+; CHECK:       [[SCALAR_PH]]:
+;
+entry:
+  br label %outer.header
+
+outer.header:
+  %outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
+  %outer.iv.mul.M = mul nsw i64 %outer.iv, %M
+  br label %inner.body
+
+inner.body:
+  %inner.iv = phi i64 [ 0, %outer.header ], [ %inner.iv.next, %inner.body ]
+  %idx = add nsw i64 %outer.iv.mul.M, %inner.iv
+  %A.ptr = getelementptr inbounds float, ptr %A, i64 %idx
+  %A.val = load float, ptr %A.ptr, align 4
+  %B.ptr = getelementptr inbounds float, ptr %B, i64 %idx
+  store float %A.val, ptr %B.ptr, align 4
+  %inner.iv.next = add nuw nsw i64 %inner.iv, 1
+  %inner.iv.cmp = icmp eq i64 %inner.iv.next, %M
+  br i1 %inner.iv.cmp, label %outer.latch, label %inner.body
+
+outer.latch:
+  %outer.iv.next = add nuw nsw i64 %outer.iv, 1
+  %outer.iv.cmp = icmp eq i64 %outer.iv.next, %N
+  br i1 %outer.iv.cmp, label %exit, label %outer.header, !llvm.loop !0
+
+exit:
+  ret void
+}
+
+; The lanes of the column-major store A[i + j*8] are only distinct for VF <= 8.
+define void @col_major_m8(ptr noalias %A, i64 %N) {
+; CHECK-LABEL: define void @col_major_m8(
+; CHECK-SAME: ptr noalias [[A:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 4
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = and i64 [[N]], 3
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[OUTER_LATCH3:.*]] ]
+; CHECK-NEXT:    [[VEC_IND:%.*]] = phi <4 x i64> [ <i64 0, i64 1, i64 2, i64 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[OUTER_LATCH3]] ]
+; CHECK-NEXT:    br label %[[INNER_BODY1:.*]]
+; CHECK:       [[INNER_BODY1]]:
+; CHECK-NEXT:    [[INNER_IV2:%.*]] = phi <4 x i64> [ zeroinitializer, %[[VECTOR_BODY]] ], [ [[TMP5:%.*]], %[[INNER_BODY1]] ]
+; CHECK-NEXT:    [[TMP1:%.*]] = shl nsw <4 x i64> [[INNER_IV2]], splat (i64 3)
+; CHECK-NEXT:    [[TMP2:%.*]] = add nsw <4 x i64> [[VEC_IND]], [[TMP1]]
+; CHECK-NEXT:    [[WIDE_GEP:%.*]] = getelementptr inbounds float, ptr [[A]], <4 x i64> [[TMP2]]
+; CHECK-NEXT:    [[WIDE_MASKED_GATHER:%.*]] = call <4 x float> @llvm.masked.gather.v4f32.v4p0(<4 x ptr> align 4 [[WIDE_GEP]], <4 x i1> splat (i1 true), <4 x float> poison)
+; CHECK-NEXT:    [[TMP3:%.*]] = sitofp <4 x i64> [[VEC_IND]] to <4 x float>
+; CHECK-NEXT:    [[TMP4:%.*]] = fsub <4 x float> [[TMP3]], [[WIDE_MASKED_GATHER]]
+; CHECK-NEXT:    call void @llvm.masked.scatter.v4f32.v4p0(<4 x float> [[TMP4]], <4 x ptr> align 4 [[WIDE_GEP]], <4 x i1> splat (i1 true))
+; CHECK-NEXT:    [[TMP5]] = add nuw nsw <4 x i64> [[INNER_IV2]], splat (i64 1)
+; CHECK-NEXT:    [[TMP6:%.*]] = icmp eq <4 x i64> [[TMP5]], splat (i64 8)
+; CHECK-NEXT:    [[TMP7:%.*]] = extractelement <4 x i1> [[TMP6]], i64 0
+; CHECK-NEXT:    br i1 [[TMP7]], label %[[OUTER_LATCH3]], label %[[INNER_BODY1]]
+; CHECK:       [[OUTER_LATCH3]]:
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT:    [[VEC_IND_NEXT]] = add nuw nsw <4 x i64> [[VEC_IND]], splat (i64 4)
+; CHECK-NEXT:    [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP28:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], [[EXIT:label %.*]], label %[[SCALAR_PH]]
+; CHECK:       [[SCALAR_PH]]:
+;
+entry:
+  br label %outer.header
+
+outer.header:
+  %outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
+  br label %inner.body
+
+inner.body:
+  %inner.iv = phi i64 [ 0, %outer.header ], [ %inner.iv.next, %inner.body ]
+  %inner.iv.mul.8 = mul nsw i64 %inner.iv, 8
+  %idx = add nsw i64 %outer.iv, %inner.iv.mul.8
+  %A.ptr = getelementptr inbounds float, ptr %A, i64 %idx
+  %A.val = load float, ptr %A.ptr, align 4
+  %outer.iv.fp = sitofp i64 %outer.iv to float
+  %sub = fsub float %outer.iv.fp, %A.val
+  store float %sub, ptr %A.ptr, align 4
+  %inner.iv.next = add nuw nsw i64 %inner.iv, 1
+  %inner.iv.cmp = icmp eq i64 %inner.iv.next, 8
+  br i1 %inner.iv.cmp, label %outer.latch, label %inner.body
+
+outer.latch:
+  %outer.iv.next = add nuw nsw i64 %outer.iv, 1
+  %outer.iv.cmp = icmp eq i64 %outer.iv.next, %N
+  br i1 %outer.iv.cmp, label %exit, label %outer.header, !llvm.loop !0
+
+exit:
+  ret void
+}
+
+; The lanes of the column-major store A[i + j*2] are only distinct for VF <= 2.
+define void @col_major_m2(ptr noalias %A, i64 %N) {
+; CHECK-LABEL: define void @col_major_m2(
+; CHECK-SAME: ptr noalias [[A:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 4
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = and i64 [[N]], 3
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[OUTER_LATCH3:.*]] ]
+; CHECK-NEXT:    [[VEC_IND:%.*]] = phi <4 x i64> [ <i64 0, i64 1, i64 2, i64 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[OUTER_LATCH3]] ]
+; CHECK-NEXT:    br label %[[INNER_BODY1:.*]]
+; CHECK:       [[INNER_BODY1]]:
+; CHECK-NEXT:    [[INNER_IV2:%.*]] = phi <4 x i64> [ zeroinitializer, %[[VECTOR_BODY]] ], [ [[TMP5:%.*]], %[[INNER_BODY1]] ]
+; CHECK-NEXT:    [[TMP1:%.*]] = shl nsw <4 x i64> [[INNER_IV2]], splat (i64 1)
+; CHECK-NEXT:    [[TMP2:%.*]] = add nsw <4 x i64> [[VEC_IND]], [[TMP1]]
+; CHECK-NEXT:    [[WIDE_GEP:%.*]] = getelementptr inbounds float, ptr [[A]], <4 x i64> [[TMP2]]
+; CHECK-NEXT:    [[WIDE_MASKED_GATHER:%.*]] = call <4 x float> @llvm.masked.gather.v4f32.v4p0(<4 x ptr> align 4 [[WIDE_GEP]], <4 x i1> splat (i1 true), <4 x float> poison)
+; CHECK-NEXT:    [[TMP3:%.*]] = sitofp <4 x i64> [[VEC_IND]] to <4 x float>
+; CHECK-NEXT:    [[TMP4:%.*]] = fsub <4 x float> [[TMP3]], [[WIDE_MASKED_GATHER]]
+; CHECK-NEXT:    call void @llvm.masked.scatter.v4f32.v4p0(<4 x float> [[TMP4]], <4 x ptr> align 4 [[WIDE_GEP]], <4 x i1> splat (i1 true))
+; CHECK-NEXT:    [[TMP5]] = add nuw nsw <4 x i64> [[INNER_IV2]], splat (i64 1)
+; CHECK-NEXT:    [[TMP6:%.*]] = icmp eq <4 x i64> [[TMP5]], splat (i64 2)
+; CHECK-NEXT:    [[TMP7:%.*]] = extractelement <4 x i1> [[TMP6]], i64 0
+; CHECK-NEXT:    br i1 [[TMP7]], label %[[OUTER_LATCH3]], label %[[INNER_BODY1]]
+; CHECK:       [[OUTER_LATCH3]]:
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT:    [[VEC_IND_NEXT]] = add nuw nsw <4 x i64> [[VEC_IND]], splat (i64 4)
+; CHECK-NEXT:    [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP30:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], [[EXIT:label %.*]], label %[[SCALAR_PH]]
+; CHECK:       [[SCALAR_PH]]:
+;
+entry:
+  br label %outer.header
+
+outer.header:
+  %outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
+  br label %inner.body
+
+inner.body:
+  %inner.iv = phi i64 [ 0, %outer.header ], [ %inner.iv.next, %inner.body ]
+  %inner.iv.mul.2 = mul nsw i64 %inner.iv, 2
+  %idx = add nsw i64 %outer.iv, %inner.iv.mul.2
+  %A.ptr = getelementptr inbounds float, ptr %A, i64 %idx
+  %A.val = load float, ptr %A.ptr, align 4
+  %outer.iv.fp = sitofp i64 %outer.iv to float
+  %sub = fsub float %outer.iv.fp, %A.val
+  store float %sub, ptr %A.ptr, align 4
+  %inner.iv.next = add nuw nsw i64 %inner.iv, 1
+  %inner.iv.cmp = icmp eq i64 %inner.iv.next, 2
+  br i1 %inner.iv.cmp, label %outer.latch, label %inner.body
+
+outer.latch:
+  %outer.iv.next = add nuw nsw i64 %outer.iv, 1
+  %outer.iv.cmp = icmp eq i64 %outer.iv.next, %N
+  br i1 %outer.iv.cmp, label %exit, label %outer.header, !llvm.loop !0
+
+exit:
+  ret void
+}
+
+; The row-major store A[i*8 + j] writes a distinct row on each outer iteration.
+define void @row_major_m8(ptr noalias %A, i64 %N) {
+; CHECK-LABEL: define void @row_major_m8(
+; CHECK-SAME: ptr noalias [[A:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 4
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = and i64 [[N]], 3
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[OUTER_LATCH3:.*]] ]
+; CHECK-NEXT:    [[VEC_IND:%.*]] = phi <4 x i64> [ <i64 0, i64 1, i64 2, i64 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[OUTER_LATCH3]] ]
+; CHECK-NEXT:    [[TMP1:%.*]] = shl nsw <4 x i64> [[VEC_IND]], splat (i64 3)
+; CHECK-NEXT:    br label %[[INNER_BODY1:.*]]
+; CHECK:       [[INNER_BODY1]]:
+; CHECK-NEXT:    [[INNER_IV2:%.*]] = phi <4 x i64> [ zeroinitializer, %[[VECTOR_BODY]] ], [ [[TMP5:%.*]], %[[INNER_BODY1]] ]
+; CHECK-NEXT:    [[TMP2:%.*]] = add nsw <4 x i64> [[TMP1]], [[INNER_IV2]]
+; CHECK-NEXT:    [[WIDE_GEP:%.*]] = getelementptr inbounds float, ptr [[A]], <4 x i64> [[TMP2]]
+; CHECK-NEXT:    [[WIDE_MASKED_GATHER:%.*]] = call <4 x float> @llvm.masked.gather.v4f32.v4p0(<4 x ptr> align 4 [[WIDE_GEP]], <4 x i1> splat (i1 true), <4 x float> poison)
+; CHECK-NEXT:    [[TMP3:%.*]] = sitofp <4 x i64> [[VEC_IND]] to <4 x float>
+; CHECK-NEXT:    [[TMP4:%.*]] = fsub <4 x float> [[TMP3]], [[WIDE_MASKED_GATHER]]
+; CHECK-NEXT:    call void @llvm.masked.scatter.v4f32.v4p0(<4 x float> [[TMP4]], <4 x ptr> align 4 [[WIDE_GEP]], <4 x i1> splat (i1 true))
+; CHECK-NEXT:    [[TMP5]] = add nuw nsw <4 x i64> [[INNER_IV2]], splat (i64 1)
+; CHECK-NEXT:    [[TMP6:%.*]] = icmp eq <4 x i64> [[TMP5]], splat (i64 8)
+; CHECK-NEXT:    [[TMP7:%.*]] = extractelement <4 x i1> [[TMP6]], i64 0
+; CHECK-NEXT:    br i1 [[TMP7]], label %[[OUTER_LATCH3]], label %[[INNER_BODY1]]
+; CHECK:       [[OUTER_LATCH3]]:
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT:    [[VEC_IND_NEXT]] = add nuw nsw <4 x i64> [[VEC_IND]], splat (i64 4)
+; CHECK-NEXT:    [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP32:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], [[EXIT:label %.*]], label %[[SCALAR_PH]]
+; CHECK:       [[SCALAR_PH]]:
+;
+entry:
+  br label %outer.header
+
+outer.header:
+  %outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
+  %outer.iv.mul.8 = mul nsw i64 %outer.iv, 8
+  br label %inner.body
+
+inner.body:
+  %inner.iv = phi i64 [ 0, %outer.header ], [ %inner.iv.next, %inner.body ]
+  %idx = add nsw i64 %outer.iv.mul.8, %inner.iv
+  %A.ptr = getelementptr inbounds float, ptr %A, i64 %idx
+  %A.val = load float, ptr %A.ptr, align 4
+  %outer.iv.fp = sitofp i64 %outer.iv to float
+  %sub = fsub float %outer.iv.fp, %A.val
+  store float %sub, ptr %A.ptr, align 4
+  %inner.iv.next = add nuw nsw i64 %inner.iv, 1
+  %inner.iv.cmp = icmp eq i64 %inner.iv.next, 8
+  br i1 %inner.iv.cmp, label %outer.latch, label %inner.body
+
+outer.latch:
+  %outer.iv.next = add nuw nsw i64 %outer.iv, 1
+  %outer.iv.cmp = icmp eq i64 %outer.iv.next, %N
+  br i1 %outer.iv.cmp, label %exit, label %outer.header, !llvm.loop !0
+
+exit:
+  ret void
+}
+
+; The 4-byte accesses to A[i + j*4] advance by one byte per outer iteration, so
+; the lanes overlap.
+define void @access_wider_than_outer_stride(ptr noalias %A, i64 %N) {
+; CHECK-LABEL: define void @access_wider_than_outer_stride(
+; CHECK-SAME: ptr noalias [[A:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 4
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = and i64 [[N]], 3
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[OUTER_LATCH3:.*]] ]
+; CHECK-NEXT:    [[VEC_IND:%.*]] = phi <4 x i64> [ <i64 0, i64 1, i64 2, i64 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[OUTER_LATCH3]] ]
+; CHECK-NEXT:    br label %[[INNER_BODY1:.*]]
+; CHECK:       [[INNER_BODY1]]:
+; CHECK-NEXT:    [[INNER_IV2:%.*]] = phi <4 x i64> [ zeroinitializer, %[[VECTOR_BODY]] ], [ [[TMP5:%.*]], %[[INNER_BODY1]] ]
+; CHECK-NEXT:    [[TMP1:%.*]] = shl nsw <4 x i64> [[INNER_IV2]], splat (i64 2)
+; CHECK-NEXT:    [[TMP2:%.*]] = add nsw <4 x i64> [[VEC_IND]], [[TMP1]]
+; CHECK-NEXT:    [[WIDE_GEP:%.*]] = getelementptr inbounds i8, ptr [[A]], <4 x i64> [[TMP2]]
+; CHECK-NEXT:    [[WIDE_MASKED_GATHER:%.*]] = call <4 x float> @llvm.masked.gather.v4f32.v4p0(<4 x ptr> align 1 [[WIDE_GEP]], <4 x i1> splat (i1 true), <4 x float> poison)
+; CHECK-NEXT:    [[TMP3:%.*]] = sitofp <4 x i64> [[VEC_IND]] to <4 x float>
+; CHECK-NEXT:    [[TMP4:%.*]] = fsub <4 x float> [[TMP3]], [[WIDE_MASKED_GATHER]]
+; CHECK-NEXT:    call void @llvm.masked.scatter.v4f32.v4p0(<4 x float> [[TMP4]], <4 x ptr> align 1 [[WIDE_GEP]], <4 x i1> splat (i1 true))
+; CHECK-NEXT:    [[TMP5]] = add nuw nsw <4 x i64> [[INNER_IV2]], splat (i64 1)
+; CHECK-NEXT:    [[TMP6:%.*]] = icmp eq <4 x i64> [[TMP5]], splat (i64 4)
+; CHECK-NEXT:    [[TMP7:%.*]] = extractelement <4 x i1> [[TMP6]], i64 0
+; CHECK-NEXT:    br i1 [[TMP7]], label %[[OUTER_LATCH3]], label %[[INNER_BODY1]]
+; CHECK:       [[OUTER_LATCH3]]:
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT:    [[VEC_IND_NEXT]] = add nuw nsw <4 x i64> [[VEC_IND]], splat (i64 4)
+; CHECK-NEXT:    [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP34:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], [[EXIT:label %.*]], label %[[SCALAR_PH]]
+; CHECK:       [[SCALAR_PH]]:
+;
+entry:
+  br label %outer.header
+
+outer.header:
+  %outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
+  br label %inner.body
+
+inner.body:
+  %inner.iv = phi i64 [ 0, %outer.header ], [ %inner.iv.next, %inner.body ]
+  %inner.iv.mul.4 = mul nsw i64 %inner.iv, 4
+  %idx = add nsw i64 %outer.iv, %inner.iv.mul.4
+  %A.ptr = getelementptr inbounds i8, ptr %A, i64 %idx
+  %A.val = load float, ptr %A.ptr, align 1
+  %outer.iv.fp = sitofp i64 %outer.iv to float
+  %sub = fsub float %outer.iv.fp, %A.val
+  store float %sub, ptr %A.ptr, align 1
+  %inner.iv.next = add nuw nsw i64 %inner.iv, 1
+  %inner.iv.cmp = icmp eq i64 %inner.iv.next, 4
+  br i1 %inner.iv.cmp, label %outer.latch, label %inner.body
+
+outer.latch:
+  %outer.iv.next = add nuw nsw i64 %outer.iv, 1
+  %outer.iv.cmp = icmp eq i64 %outer.iv.next, %N
+  br i1 %outer.iv.cmp, label %exit, label %outer.header, !llvm.loop !0
+
+exit:
+  ret void
+}
+
+; The nest copies between distinct objects using a memory intrinsic.
+define void @memcpy_in_nest(ptr noalias %A, ptr noalias %B, i64 %N, i64 %M) {
+; CHECK-LABEL: define void @memcpy_in_nest(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], i64 [[N:%.*]], i64 [[M:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    br label %[[OUTER_HEADER:.*]]
+; CHECK:       [[OUTER_HEADER]]:
+; CHECK-NEXT:    [[OUTER_IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[OUTER_IV_NEXT:%.*]], %[[OUTER_LATCH:.*]] ]
+; CHECK-NEXT:    [[OUTER_IV_MUL_M:%.*]] = mul nsw i64 [[OUTER_IV]], [[M]]
+; CHECK-NEXT:    br label %[[INNER_BODY:.*]]
+; CHECK:       [[INNER_BODY]]:
+; CHECK-NEXT:    [[INNER_IV:%.*]] = phi i64 [ 0, %[[OUTER_HEADER]] ], [ [[INNER_IV_NEXT:%.*]], %[[INNER_BODY]] ]
+; CHECK-NEXT:    [[INNER_IV_NEXT]] = add nuw nsw i64 [[INNER_IV]], 1
+; CHECK-NEXT:    [[INNER_IV_CMP:%.*]] = icmp eq i64 [[INNER_IV_NEXT]], [[M]]
+; CHECK-NEXT:    br i1 [[INNER_IV_CMP]], label %[[OUTER_LATCH]], label %[[INNER_BODY]]
+; CHECK:       [[OUTER_LATCH]]:
+; CHECK-NEXT:    [[A_PTR:%.*]] = getelementptr inbounds float, ptr [[A]], i64 [[OUTER_IV_MUL_M]]
+; CHECK-NEXT:    [[B_PTR:%.*]] = getelementptr inbounds float, ptr [[B]], i64 [[OUTER_IV]]
+; CHECK-NEXT:    call void @llvm.memcpy.p0.p0.i64(ptr [[B_PTR]], ptr [[A_PTR]], i64 4, i1 false)
+; CHECK-NEXT:    [[OUTER_IV_NEXT]] = add nuw nsw i64 [[OUTER_IV]], 1
+; CHECK-NEXT:    [[OUTER_IV_CMP:%.*]] = icmp eq i64 [[OUTER_IV_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[OUTER_IV_CMP]], label %[[EXIT:.*]], label %[[OUTER_HEADER]], !llvm.loop [[LOOP36:![0-9]+]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %outer.header
+
+outer.header:
+  %outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
+  %outer.iv.mul.M = mul nsw i64 %outer.iv, %M
+  br label %inner.body
+
+inner.body:
+  %inner.iv = phi i64 [ 0, %outer.header ], [ %inner.iv.next, %inner.body ]
+  %inner.iv.next = add nuw nsw i64 %inner.iv, 1
+  %inner.iv.cmp = icmp eq i64 %inner.iv.next, %M
+  br i1 %inner.iv.cmp, label %outer.latch, label %inner.body
+
+outer.latch:
+  %A.ptr = getelementptr inbounds float, ptr %A, i64 %outer.iv.mul.M
+  %B.ptr = getelementptr inbounds float, ptr %B, i64 %outer.iv
+  call void @llvm.memcpy.p0.p0.i64(ptr %B.ptr, ptr %A.ptr, i64 4, i1 false)
+  %outer.iv.next = add nuw nsw i64 %outer.iv, 1
+  %outer.iv.cmp = icmp eq i64 %outer.iv.next, %N
+  br i1 %outer.iv.cmp, label %exit, label %outer.header, !llvm.loop !0
+
+exit:
+  ret void
+}
+
+declare void @llvm.memcpy.p0.p0.i64(ptr, ptr, i64, i1)
+
+!0 = distinct !{!0, !1}
+!1 = !{!"llvm.loop.vectorize.enable"}

diff  --git a/llvm/test/Transforms/LoopVectorize/outer_loop_contiguous.ll b/llvm/test/Transforms/LoopVectorize/outer_loop_contiguous.ll
index d7c27cafa04ac..fb6fbf0cc359a 100644
--- a/llvm/test/Transforms/LoopVectorize/outer_loop_contiguous.ll
+++ b/llvm/test/Transforms/LoopVectorize/outer_loop_contiguous.ll
@@ -1,5 +1,5 @@
 ; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --filter-out-after "^scalar.ph" --version 6
-; RUN: opt -S -passes=loop-vectorize -enable-vplan-native-path < %s | FileCheck %s
+; RUN: opt -passes=loop-vectorize -enable-vplan-native-path -S %s | FileCheck %s
 
 ; Test coverage for contiguous access detection in outer loop vectorization.
 ; Tests various stride and type combinations.
@@ -466,6 +466,482 @@ exit:
   ret void
 }
 
+; The column-major access A[i + j*N] is unit-stride with respect to the outer
+; loop.
+define void @col_major(ptr noalias %A, i64 %N, i64 %M) {
+; CHECK-LABEL: define void @col_major(
+; CHECK-SAME: ptr noalias [[A:%.*]], i64 [[N:%.*]], i64 [[M:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 4
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = and i64 [[N]], 3
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
+; CHECK-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x i64> poison, i64 [[N]], i64 0
+; CHECK-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i64> [[BROADCAST_SPLATINSERT]], <4 x i64> poison, <4 x i32> zeroinitializer
+; CHECK-NEXT:    [[BROADCAST_SPLATINSERT1:%.*]] = insertelement <4 x i64> poison, i64 [[M]], i64 0
+; CHECK-NEXT:    [[BROADCAST_SPLAT2:%.*]] = shufflevector <4 x i64> [[BROADCAST_SPLATINSERT1]], <4 x i64> poison, <4 x i32> zeroinitializer
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[OUTER_LATCH5:.*]] ]
+; CHECK-NEXT:    [[VEC_IND:%.*]] = phi <4 x i64> [ <i64 0, i64 1, i64 2, i64 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[OUTER_LATCH5]] ]
+; CHECK-NEXT:    br label %[[INNER_BODY3:.*]]
+; CHECK:       [[INNER_BODY3]]:
+; CHECK-NEXT:    [[INNER_IV4:%.*]] = phi <4 x i64> [ zeroinitializer, %[[VECTOR_BODY]] ], [ [[TMP4:%.*]], %[[INNER_BODY3]] ]
+; CHECK-NEXT:    [[TMP1:%.*]] = mul nsw <4 x i64> [[INNER_IV4]], [[BROADCAST_SPLAT]]
+; CHECK-NEXT:    [[TMP2:%.*]] = add nsw <4 x i64> [[VEC_IND]], [[TMP1]]
+; CHECK-NEXT:    [[WIDE_GEP:%.*]] = getelementptr inbounds float, ptr [[A]], <4 x i64> [[TMP2]]
+; CHECK-NEXT:    [[WIDE_MASKED_GATHER:%.*]] = call <4 x float> @llvm.masked.gather.v4f32.v4p0(<4 x ptr> align 4 [[WIDE_GEP]], <4 x i1> splat (i1 true), <4 x float> poison), !llvm.access.group [[ACC_GRP14:![0-9]+]]
+; CHECK-NEXT:    [[TMP3:%.*]] = fmul <4 x float> [[WIDE_MASKED_GATHER]], splat (float 2.000000e+00)
+; CHECK-NEXT:    call void @llvm.masked.scatter.v4f32.v4p0(<4 x float> [[TMP3]], <4 x ptr> align 4 [[WIDE_GEP]], <4 x i1> splat (i1 true)), !llvm.access.group [[ACC_GRP14]]
+; CHECK-NEXT:    [[TMP4]] = add nuw nsw <4 x i64> [[INNER_IV4]], splat (i64 1)
+; CHECK-NEXT:    [[TMP5:%.*]] = icmp eq <4 x i64> [[TMP4]], [[BROADCAST_SPLAT2]]
+; CHECK-NEXT:    [[TMP6:%.*]] = extractelement <4 x i1> [[TMP5]], i64 0
+; CHECK-NEXT:    br i1 [[TMP6]], label %[[OUTER_LATCH5]], label %[[INNER_BODY3]]
+; CHECK:       [[OUTER_LATCH5]]:
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT:    [[VEC_IND_NEXT]] = add nuw nsw <4 x i64> [[VEC_IND]], splat (i64 4)
+; CHECK-NEXT:    [[TMP7:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP7]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP15:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], [[EXIT:label %.*]], label %[[SCALAR_PH]]
+; CHECK:       [[SCALAR_PH]]:
+;
+entry:
+  br label %outer.header
+
+outer.header:
+  %outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
+  br label %inner.body
+
+inner.body:
+  %inner.iv = phi i64 [ 0, %outer.header ], [ %inner.iv.next, %inner.body ]
+  %inner.iv.mul.N = mul nsw i64 %inner.iv, %N
+  %idx = add nsw i64 %outer.iv, %inner.iv.mul.N
+  %A.ptr = getelementptr inbounds float, ptr %A, i64 %idx
+  %A.val = load float, ptr %A.ptr, align 4, !llvm.access.group !3
+  %mul = fmul float %A.val, 2.000000e+00
+  store float %mul, ptr %A.ptr, align 4, !llvm.access.group !3
+  %inner.iv.next = add nuw nsw i64 %inner.iv, 1
+  %inner.iv.cmp = icmp eq i64 %inner.iv.next, %M
+  br i1 %inner.iv.cmp, label %outer.latch, label %inner.body
+
+outer.latch:
+  %outer.iv.next = add nuw nsw i64 %outer.iv, 1
+  %outer.iv.cmp = icmp eq i64 %outer.iv.next, %N
+  br i1 %outer.iv.cmp, label %exit, label %outer.header, !llvm.loop !5
+
+exit:
+  ret void
+}
+
+; The row-major access A[i*M + j] has stride M with respect to the outer loop.
+define void @row_major(ptr noalias %A, i64 %N, i64 %M) {
+; CHECK-LABEL: define void @row_major(
+; CHECK-SAME: ptr noalias [[A:%.*]], i64 [[N:%.*]], i64 [[M:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 4
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = and i64 [[N]], 3
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
+; CHECK-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x i64> poison, i64 [[M]], i64 0
+; CHECK-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i64> [[BROADCAST_SPLATINSERT]], <4 x i64> poison, <4 x i32> zeroinitializer
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[OUTER_LATCH3:.*]] ]
+; CHECK-NEXT:    [[VEC_IND:%.*]] = phi <4 x i64> [ <i64 0, i64 1, i64 2, i64 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[OUTER_LATCH3]] ]
+; CHECK-NEXT:    [[TMP1:%.*]] = mul nsw <4 x i64> [[VEC_IND]], [[BROADCAST_SPLAT]]
+; CHECK-NEXT:    br label %[[INNER_BODY1:.*]]
+; CHECK:       [[INNER_BODY1]]:
+; CHECK-NEXT:    [[INNER_IV2:%.*]] = phi <4 x i64> [ zeroinitializer, %[[VECTOR_BODY]] ], [ [[TMP4:%.*]], %[[INNER_BODY1]] ]
+; CHECK-NEXT:    [[TMP2:%.*]] = add nsw <4 x i64> [[TMP1]], [[INNER_IV2]]
+; CHECK-NEXT:    [[WIDE_GEP:%.*]] = getelementptr inbounds float, ptr [[A]], <4 x i64> [[TMP2]]
+; CHECK-NEXT:    [[WIDE_MASKED_GATHER:%.*]] = call <4 x float> @llvm.masked.gather.v4f32.v4p0(<4 x ptr> align 4 [[WIDE_GEP]], <4 x i1> splat (i1 true), <4 x float> poison), !llvm.access.group [[ACC_GRP14]]
+; CHECK-NEXT:    [[TMP3:%.*]] = fmul <4 x float> [[WIDE_MASKED_GATHER]], splat (float 2.000000e+00)
+; CHECK-NEXT:    call void @llvm.masked.scatter.v4f32.v4p0(<4 x float> [[TMP3]], <4 x ptr> align 4 [[WIDE_GEP]], <4 x i1> splat (i1 true)), !llvm.access.group [[ACC_GRP14]]
+; CHECK-NEXT:    [[TMP4]] = add nuw nsw <4 x i64> [[INNER_IV2]], splat (i64 1)
+; CHECK-NEXT:    [[TMP5:%.*]] = icmp eq <4 x i64> [[TMP4]], [[BROADCAST_SPLAT]]
+; CHECK-NEXT:    [[TMP6:%.*]] = extractelement <4 x i1> [[TMP5]], i64 0
+; CHECK-NEXT:    br i1 [[TMP6]], label %[[OUTER_LATCH3]], label %[[INNER_BODY1]]
+; CHECK:       [[OUTER_LATCH3]]:
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT:    [[VEC_IND_NEXT]] = add nuw nsw <4 x i64> [[VEC_IND]], splat (i64 4)
+; CHECK-NEXT:    [[TMP7:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP7]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP18:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], [[EXIT:label %.*]], label %[[SCALAR_PH]]
+; CHECK:       [[SCALAR_PH]]:
+;
+entry:
+  br label %outer.header
+
+outer.header:
+  %outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
+  %outer.iv.mul.M = mul nsw i64 %outer.iv, %M
+  br label %inner.body
+
+inner.body:
+  %inner.iv = phi i64 [ 0, %outer.header ], [ %inner.iv.next, %inner.body ]
+  %idx = add nsw i64 %outer.iv.mul.M, %inner.iv
+  %A.ptr = getelementptr inbounds float, ptr %A, i64 %idx
+  %A.val = load float, ptr %A.ptr, align 4, !llvm.access.group !3
+  %mul = fmul float %A.val, 2.000000e+00
+  store float %mul, ptr %A.ptr, align 4, !llvm.access.group !3
+  %inner.iv.next = add nuw nsw i64 %inner.iv, 1
+  %inner.iv.cmp = icmp eq i64 %inner.iv.next, %M
+  br i1 %inner.iv.cmp, label %outer.latch, label %inner.body
+
+outer.latch:
+  %outer.iv.next = add nuw nsw i64 %outer.iv, 1
+  %outer.iv.cmp = icmp eq i64 %outer.iv.next, %N
+  br i1 %outer.iv.cmp, label %exit, label %outer.header, !llvm.loop !5
+
+exit:
+  ret void
+}
+
+; The inner-loop step of A[i + j*i] varies with the outer loop, so the access
+; has no constant stride with respect to the outer loop.
+define void @inner_step_varies_with_outer(ptr noalias %A, ptr noalias %B, i64 %N, i64 %M) {
+; CHECK-LABEL: define void @inner_step_varies_with_outer(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], i64 [[N:%.*]], i64 [[M:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 4
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = and i64 [[N]], 3
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
+; CHECK-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x i64> poison, i64 [[M]], i64 0
+; CHECK-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i64> [[BROADCAST_SPLATINSERT]], <4 x i64> poison, <4 x i32> zeroinitializer
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[OUTER_LATCH3:.*]] ]
+; CHECK-NEXT:    [[VEC_IND:%.*]] = phi <4 x i64> [ <i64 0, i64 1, i64 2, i64 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[OUTER_LATCH3]] ]
+; CHECK-NEXT:    [[TMP1:%.*]] = getelementptr inbounds float, ptr [[B]], i64 [[INDEX]]
+; CHECK-NEXT:    br label %[[INNER_BODY1:.*]]
+; CHECK:       [[INNER_BODY1]]:
+; CHECK-NEXT:    [[INNER_IV2:%.*]] = phi <4 x i64> [ zeroinitializer, %[[VECTOR_BODY]] ], [ [[TMP5:%.*]], %[[INNER_BODY1]] ]
+; CHECK-NEXT:    [[TMP2:%.*]] = mul nsw <4 x i64> [[INNER_IV2]], [[VEC_IND]]
+; CHECK-NEXT:    [[TMP3:%.*]] = add nsw <4 x i64> [[VEC_IND]], [[TMP2]]
+; CHECK-NEXT:    [[WIDE_GEP:%.*]] = getelementptr inbounds float, ptr [[A]], <4 x i64> [[TMP3]]
+; CHECK-NEXT:    [[WIDE_MASKED_GATHER:%.*]] = call <4 x float> @llvm.masked.gather.v4f32.v4p0(<4 x ptr> align 4 [[WIDE_GEP]], <4 x i1> splat (i1 true), <4 x float> poison)
+; CHECK-NEXT:    [[TMP4:%.*]] = fmul <4 x float> [[WIDE_MASKED_GATHER]], splat (float 2.000000e+00)
+; CHECK-NEXT:    store <4 x float> [[TMP4]], ptr [[TMP1]], align 4
+; CHECK-NEXT:    [[TMP5]] = add nuw nsw <4 x i64> [[INNER_IV2]], splat (i64 1)
+; CHECK-NEXT:    [[TMP6:%.*]] = icmp eq <4 x i64> [[TMP5]], [[BROADCAST_SPLAT]]
+; CHECK-NEXT:    [[TMP7:%.*]] = extractelement <4 x i1> [[TMP6]], i64 0
+; CHECK-NEXT:    br i1 [[TMP7]], label %[[OUTER_LATCH3]], label %[[INNER_BODY1]]
+; CHECK:       [[OUTER_LATCH3]]:
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT:    [[VEC_IND_NEXT]] = add nuw nsw <4 x i64> [[VEC_IND]], splat (i64 4)
+; CHECK-NEXT:    [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP20:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], [[EXIT:label %.*]], label %[[SCALAR_PH]]
+; CHECK:       [[SCALAR_PH]]:
+;
+entry:
+  br label %outer.header
+
+outer.header:
+  %outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
+  %B.ptr = getelementptr inbounds float, ptr %B, i64 %outer.iv
+  br label %inner.body
+
+inner.body:
+  %inner.iv = phi i64 [ 0, %outer.header ], [ %inner.iv.next, %inner.body ]
+  %inner.iv.mul.outer.iv = mul nsw i64 %inner.iv, %outer.iv
+  %idx = add nsw i64 %outer.iv, %inner.iv.mul.outer.iv
+  %A.ptr = getelementptr inbounds float, ptr %A, i64 %idx
+  %A.val = load float, ptr %A.ptr, align 4
+  %mul = fmul float %A.val, 2.000000e+00
+  store float %mul, ptr %B.ptr, align 4
+  %inner.iv.next = add nuw nsw i64 %inner.iv, 1
+  %inner.iv.cmp = icmp eq i64 %inner.iv.next, %M
+  br i1 %inner.iv.cmp, label %outer.latch, label %inner.body
+
+outer.latch:
+  %outer.iv.next = add nuw nsw i64 %outer.iv, 1
+  %outer.iv.cmp = icmp eq i64 %outer.iv.next, %N
+  br i1 %outer.iv.cmp, label %exit, label %outer.header, !llvm.loop !6
+
+exit:
+  ret void
+}
+
+; The inner-loop index doubles on each iteration, so the access is not an
+; affine recurrence.
+define void @inner_step_is_phi(ptr noalias %A, i64 %N, i64 %M) {
+; CHECK-LABEL: define void @inner_step_is_phi(
+; CHECK-SAME: ptr noalias [[A:%.*]], i64 [[N:%.*]], i64 [[M:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 4
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = and i64 [[N]], 3
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
+; CHECK-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x i64> poison, i64 [[M]], i64 0
+; CHECK-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i64> [[BROADCAST_SPLATINSERT]], <4 x i64> poison, <4 x i32> zeroinitializer
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[OUTER_LATCH4:.*]] ]
+; CHECK-NEXT:    [[VEC_IND:%.*]] = phi <4 x i64> [ <i64 0, i64 1, i64 2, i64 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[OUTER_LATCH4]] ]
+; CHECK-NEXT:    br label %[[INNER_BODY1:.*]]
+; CHECK:       [[INNER_BODY1]]:
+; CHECK-NEXT:    [[INNER_IV2:%.*]] = phi <4 x i64> [ zeroinitializer, %[[VECTOR_BODY]] ], [ [[TMP4:%.*]], %[[INNER_BODY1]] ]
+; CHECK-NEXT:    [[INNER_VALUE3:%.*]] = phi <4 x i64> [ splat (i64 1), %[[VECTOR_BODY]] ], [ [[TMP1:%.*]], %[[INNER_BODY1]] ]
+; CHECK-NEXT:    [[TMP1]] = add <4 x i64> [[INNER_VALUE3]], [[INNER_VALUE3]]
+; CHECK-NEXT:    [[TMP2:%.*]] = add nsw <4 x i64> [[INNER_VALUE3]], [[VEC_IND]]
+; CHECK-NEXT:    [[WIDE_GEP:%.*]] = getelementptr inbounds float, ptr [[A]], <4 x i64> [[TMP2]]
+; CHECK-NEXT:    [[WIDE_MASKED_GATHER:%.*]] = call <4 x float> @llvm.masked.gather.v4f32.v4p0(<4 x ptr> align 4 [[WIDE_GEP]], <4 x i1> splat (i1 true), <4 x float> poison)
+; CHECK-NEXT:    [[TMP3:%.*]] = fmul <4 x float> [[WIDE_MASKED_GATHER]], splat (float 2.000000e+00)
+; CHECK-NEXT:    call void @llvm.masked.scatter.v4f32.v4p0(<4 x float> [[TMP3]], <4 x ptr> align 4 [[WIDE_GEP]], <4 x i1> splat (i1 true))
+; CHECK-NEXT:    [[TMP4]] = add nuw nsw <4 x i64> [[INNER_IV2]], splat (i64 1)
+; CHECK-NEXT:    [[TMP5:%.*]] = icmp eq <4 x i64> [[TMP4]], [[BROADCAST_SPLAT]]
+; CHECK-NEXT:    [[TMP6:%.*]] = extractelement <4 x i1> [[TMP5]], i64 0
+; CHECK-NEXT:    br i1 [[TMP6]], label %[[OUTER_LATCH4]], label %[[INNER_BODY1]]
+; CHECK:       [[OUTER_LATCH4]]:
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT:    [[VEC_IND_NEXT]] = add nuw nsw <4 x i64> [[VEC_IND]], splat (i64 4)
+; CHECK-NEXT:    [[TMP7:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP7]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP22:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], [[EXIT:label %.*]], label %[[SCALAR_PH]]
+; CHECK:       [[SCALAR_PH]]:
+;
+entry:
+  br label %outer.header
+
+outer.header:
+  %outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
+  br label %inner.body
+
+inner.body:
+  %inner.iv = phi i64 [ 0, %outer.header ], [ %inner.iv.next, %inner.body ]
+  %inner.value = phi i64 [ 1, %outer.header ], [ %inner.value.next, %inner.body ]
+  %inner.value.next = add i64 %inner.value, %inner.value
+  %idx = add nsw i64 %inner.value, %outer.iv
+  %A.ptr = getelementptr inbounds float, ptr %A, i64 %idx
+  %A.val = load float, ptr %A.ptr, align 4
+  %mul = fmul float %A.val, 2.000000e+00
+  store float %mul, ptr %A.ptr, align 4
+  %inner.iv.next = add nuw nsw i64 %inner.iv, 1
+  %inner.iv.cmp = icmp eq i64 %inner.iv.next, %M
+  br i1 %inner.iv.cmp, label %outer.latch, label %inner.body
+
+outer.latch:
+  %outer.iv.next = add nuw nsw i64 %outer.iv, 1
+  %outer.iv.cmp = icmp eq i64 %outer.iv.next, %N
+  br i1 %outer.iv.cmp, label %exit, label %outer.header, !llvm.loop !5
+
+exit:
+  ret void
+}
+
+; Each of the two sibling inner loops accesses A[row*64 + i] with its own
+; recurrence row.
+define void @col_major_sibling_inner_loops(ptr %A) {
+; CHECK-LABEL: define void @col_major_sibling_inner_loops(
+; CHECK-SAME: ptr [[A:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    br label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[OUTER_LATCH10:.*]] ]
+; CHECK-NEXT:    [[VEC_IND:%.*]] = phi <4 x i64> [ <i64 0, i64 1, i64 2, i64 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[OUTER_LATCH10]] ]
+; CHECK-NEXT:    br label %[[INNER_BODY1:.*]]
+; CHECK:       [[INNER_BODY1]]:
+; CHECK-NEXT:    [[INNER_IV2:%.*]] = phi <4 x i64> [ zeroinitializer, %[[VECTOR_BODY]] ], [ [[TMP6:%.*]], %[[INNER_BODY1]] ]
+; CHECK-NEXT:    [[ROW3:%.*]] = phi <4 x i8> [ splat (i8 2), %[[VECTOR_BODY]] ], [ [[TMP5:%.*]], %[[INNER_BODY1]] ]
+; CHECK-NEXT:    [[TMP0:%.*]] = sext <4 x i8> [[ROW3]] to <4 x i64>
+; CHECK-NEXT:    [[TMP1:%.*]] = shl <4 x i64> [[TMP0]], splat (i64 6)
+; CHECK-NEXT:    [[TMP2:%.*]] = add <4 x i64> [[TMP1]], [[VEC_IND]]
+; CHECK-NEXT:    [[WIDE_GEP:%.*]] = getelementptr inbounds i32, ptr [[A]], <4 x i64> [[TMP2]]
+; CHECK-NEXT:    [[WIDE_MASKED_GATHER:%.*]] = call <4 x i32> @llvm.masked.gather.v4i32.v4p0(<4 x ptr> align 4 [[WIDE_GEP]], <4 x i1> splat (i1 true), <4 x i32> poison), !llvm.access.group [[ACC_GRP14]]
+; CHECK-NEXT:    [[TMP3:%.*]] = zext <4 x i8> [[ROW3]] to <4 x i32>
+; CHECK-NEXT:    [[TMP4:%.*]] = add <4 x i32> [[WIDE_MASKED_GATHER]], [[TMP3]]
+; CHECK-NEXT:    call void @llvm.masked.scatter.v4i32.v4p0(<4 x i32> [[TMP4]], <4 x ptr> align 4 [[WIDE_GEP]], <4 x i1> splat (i1 true)), !llvm.access.group [[ACC_GRP14]]
+; CHECK-NEXT:    [[TMP5]] = add <4 x i8> [[ROW3]], splat (i8 1)
+; CHECK-NEXT:    [[TMP6]] = add nuw nsw <4 x i64> [[INNER_IV2]], splat (i64 1)
+; CHECK-NEXT:    [[TMP7:%.*]] = icmp eq <4 x i64> [[TMP6]], splat (i64 8)
+; CHECK-NEXT:    [[TMP8:%.*]] = extractelement <4 x i1> [[TMP7]], i64 0
+; CHECK-NEXT:    br i1 [[TMP8]], label %[[INNER2_PH4:.*]], label %[[INNER_BODY1]]
+; CHECK:       [[INNER2_PH4]]:
+; CHECK-NEXT:    br label %[[INNER2_BODY5:.*]]
+; CHECK:       [[INNER2_BODY5]]:
+; CHECK-NEXT:    [[INNER2_IV6:%.*]] = phi <4 x i64> [ zeroinitializer, %[[INNER2_PH4]] ], [ [[TMP15:%.*]], %[[INNER2_BODY5]] ]
+; CHECK-NEXT:    [[ROW27:%.*]] = phi <4 x i8> [ splat (i8 2), %[[INNER2_PH4]] ], [ [[TMP14:%.*]], %[[INNER2_BODY5]] ]
+; CHECK-NEXT:    [[TMP9:%.*]] = sext <4 x i8> [[ROW27]] to <4 x i64>
+; CHECK-NEXT:    [[TMP10:%.*]] = shl <4 x i64> [[TMP9]], splat (i64 6)
+; CHECK-NEXT:    [[TMP11:%.*]] = add <4 x i64> [[TMP10]], [[VEC_IND]]
+; CHECK-NEXT:    [[WIDE_GEP8:%.*]] = getelementptr inbounds i32, ptr [[A]], <4 x i64> [[TMP11]]
+; CHECK-NEXT:    [[WIDE_MASKED_GATHER9:%.*]] = call <4 x i32> @llvm.masked.gather.v4i32.v4p0(<4 x ptr> align 4 [[WIDE_GEP8]], <4 x i1> splat (i1 true), <4 x i32> poison), !llvm.access.group [[ACC_GRP14]]
+; CHECK-NEXT:    [[TMP12:%.*]] = zext <4 x i8> [[ROW27]] to <4 x i32>
+; CHECK-NEXT:    [[TMP13:%.*]] = add <4 x i32> [[WIDE_MASKED_GATHER9]], [[TMP12]]
+; CHECK-NEXT:    call void @llvm.masked.scatter.v4i32.v4p0(<4 x i32> [[TMP13]], <4 x ptr> align 4 [[WIDE_GEP8]], <4 x i1> splat (i1 true)), !llvm.access.group [[ACC_GRP14]]
+; CHECK-NEXT:    [[TMP14]] = add <4 x i8> [[ROW27]], splat (i8 1)
+; CHECK-NEXT:    [[TMP15]] = add nuw nsw <4 x i64> [[INNER2_IV6]], splat (i64 1)
+; CHECK-NEXT:    [[TMP16:%.*]] = icmp eq <4 x i64> [[TMP15]], splat (i64 8)
+; CHECK-NEXT:    [[TMP17:%.*]] = extractelement <4 x i1> [[TMP16]], i64 0
+; CHECK-NEXT:    br i1 [[TMP17]], label %[[OUTER_LATCH10]], label %[[INNER2_BODY5]]
+; CHECK:       [[OUTER_LATCH10]]:
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT:    [[VEC_IND_NEXT]] = add nuw nsw <4 x i64> [[VEC_IND]], splat (i64 4)
+; CHECK-NEXT:    [[TMP18:%.*]] = icmp eq i64 [[INDEX_NEXT]], 64
+; CHECK-NEXT:    br i1 [[TMP18]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP24:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    br label %[[EXIT:.*]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %outer.header
+
+outer.header:
+  %outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
+  br label %inner.body
+
+inner.body:
+  %inner.iv = phi i64 [ 0, %outer.header ], [ %inner.iv.next, %inner.body ]
+  %row = phi i8 [ 2, %outer.header ], [ %row.next, %inner.body ]
+  %row.ext = sext i8 %row to i64
+  %row.mul.64 = mul i64 %row.ext, 64
+  %idx = add i64 %row.mul.64, %outer.iv
+  %A.ptr = getelementptr inbounds i32, ptr %A, i64 %idx
+  %A.val = load i32, ptr %A.ptr, align 4, !llvm.access.group !3
+  %row.i32 = zext i8 %row to i32
+  %add = add i32 %A.val, %row.i32
+  store i32 %add, ptr %A.ptr, align 4, !llvm.access.group !3
+  %row.next = add i8 %row, 1
+  %inner.iv.next = add nuw nsw i64 %inner.iv, 1
+  %inner.iv.cmp = icmp eq i64 %inner.iv.next, 8
+  br i1 %inner.iv.cmp, label %inner2.ph, label %inner.body
+
+inner2.ph:
+  br label %inner2.body
+
+inner2.body:
+  %inner2.iv = phi i64 [ 0, %inner2.ph ], [ %inner2.iv.next, %inner2.body ]
+  %row2 = phi i8 [ 2, %inner2.ph ], [ %row2.next, %inner2.body ]
+  %row2.ext = sext i8 %row2 to i64
+  %row2.mul.64 = mul i64 %row2.ext, 64
+  %idx2 = add i64 %row2.mul.64, %outer.iv
+  %A.ptr2 = getelementptr inbounds i32, ptr %A, i64 %idx2
+  %A.val2 = load i32, ptr %A.ptr2, align 4, !llvm.access.group !3
+  %row2.i32 = zext i8 %row2 to i32
+  %add2 = add i32 %A.val2, %row2.i32
+  store i32 %add2, ptr %A.ptr2, align 4, !llvm.access.group !3
+  %row2.next = add i8 %row2, 1
+  %inner2.iv.next = add nuw nsw i64 %inner2.iv, 1
+  %inner2.iv.cmp = icmp eq i64 %inner2.iv.next, 8
+  br i1 %inner2.iv.cmp, label %outer.latch, label %inner2.body
+
+outer.latch:
+  %outer.iv.next = add nuw nsw i64 %outer.iv, 1
+  %outer.iv.cmp = icmp eq i64 %outer.iv.next, 64
+  br i1 %outer.iv.cmp, label %exit, label %outer.header, !llvm.loop !5
+
+exit:
+  ret void
+}
+
+; The recurrence row of A[row*64 + i] is in a loop nested two levels deep.
+define void @col_major_nested_inner_loops(ptr %A) {
+; CHECK-LABEL: define void @col_major_nested_inner_loops(
+; CHECK-SAME: ptr [[A:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    br label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[OUTER_LATCH7:.*]] ]
+; CHECK-NEXT:    [[VEC_IND:%.*]] = phi <4 x i64> [ <i64 0, i64 1, i64 2, i64 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[OUTER_LATCH7]] ]
+; CHECK-NEXT:    br label %[[MID_HEADER1:.*]]
+; CHECK:       [[MID_HEADER1]]:
+; CHECK-NEXT:    [[MID_IV2:%.*]] = phi <4 x i64> [ zeroinitializer, %[[VECTOR_BODY]] ], [ [[TMP9:%.*]], %[[MID_LATCH6:.*]] ]
+; CHECK-NEXT:    br label %[[INNER_BODY3:.*]]
+; CHECK:       [[INNER_BODY3]]:
+; CHECK-NEXT:    [[INNER_IV4:%.*]] = phi <4 x i64> [ zeroinitializer, %[[MID_HEADER1]] ], [ [[TMP6:%.*]], %[[INNER_BODY3]] ]
+; CHECK-NEXT:    [[ROW5:%.*]] = phi <4 x i8> [ splat (i8 2), %[[MID_HEADER1]] ], [ [[TMP5:%.*]], %[[INNER_BODY3]] ]
+; CHECK-NEXT:    [[TMP0:%.*]] = sext <4 x i8> [[ROW5]] to <4 x i64>
+; CHECK-NEXT:    [[TMP1:%.*]] = shl <4 x i64> [[TMP0]], splat (i64 6)
+; CHECK-NEXT:    [[TMP2:%.*]] = add <4 x i64> [[TMP1]], [[VEC_IND]]
+; CHECK-NEXT:    [[WIDE_GEP:%.*]] = getelementptr inbounds i32, ptr [[A]], <4 x i64> [[TMP2]]
+; CHECK-NEXT:    [[WIDE_MASKED_GATHER:%.*]] = call <4 x i32> @llvm.masked.gather.v4i32.v4p0(<4 x ptr> align 4 [[WIDE_GEP]], <4 x i1> splat (i1 true), <4 x i32> poison), !llvm.access.group [[ACC_GRP14]]
+; CHECK-NEXT:    [[TMP3:%.*]] = zext <4 x i8> [[ROW5]] to <4 x i32>
+; CHECK-NEXT:    [[TMP4:%.*]] = add <4 x i32> [[WIDE_MASKED_GATHER]], [[TMP3]]
+; CHECK-NEXT:    call void @llvm.masked.scatter.v4i32.v4p0(<4 x i32> [[TMP4]], <4 x ptr> align 4 [[WIDE_GEP]], <4 x i1> splat (i1 true)), !llvm.access.group [[ACC_GRP14]]
+; CHECK-NEXT:    [[TMP5]] = add <4 x i8> [[ROW5]], splat (i8 1)
+; CHECK-NEXT:    [[TMP6]] = add nuw nsw <4 x i64> [[INNER_IV4]], splat (i64 1)
+; CHECK-NEXT:    [[TMP7:%.*]] = icmp eq <4 x i64> [[TMP6]], splat (i64 8)
+; CHECK-NEXT:    [[TMP8:%.*]] = extractelement <4 x i1> [[TMP7]], i64 0
+; CHECK-NEXT:    br i1 [[TMP8]], label %[[MID_LATCH6]], label %[[INNER_BODY3]]
+; CHECK:       [[MID_LATCH6]]:
+; CHECK-NEXT:    [[TMP9]] = add nuw nsw <4 x i64> [[MID_IV2]], splat (i64 1)
+; CHECK-NEXT:    [[TMP10:%.*]] = icmp eq <4 x i64> [[TMP9]], splat (i64 2)
+; CHECK-NEXT:    [[TMP11:%.*]] = extractelement <4 x i1> [[TMP10]], i64 0
+; CHECK-NEXT:    br i1 [[TMP11]], label %[[OUTER_LATCH7]], label %[[MID_HEADER1]]
+; CHECK:       [[OUTER_LATCH7]]:
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT:    [[VEC_IND_NEXT]] = add nuw nsw <4 x i64> [[VEC_IND]], splat (i64 4)
+; CHECK-NEXT:    [[TMP12:%.*]] = icmp eq i64 [[INDEX_NEXT]], 64
+; CHECK-NEXT:    br i1 [[TMP12]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP25:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    br label %[[EXIT:.*]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %outer.header
+
+outer.header:
+  %outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
+  br label %mid.header
+
+mid.header:
+  %mid.iv = phi i64 [ 0, %outer.header ], [ %mid.iv.next, %mid.latch ]
+  br label %inner.body
+
+inner.body:
+  %inner.iv = phi i64 [ 0, %mid.header ], [ %inner.iv.next, %inner.body ]
+  %row = phi i8 [ 2, %mid.header ], [ %row.next, %inner.body ]
+  %row.ext = sext i8 %row to i64
+  %row.mul.64 = mul i64 %row.ext, 64
+  %idx = add i64 %row.mul.64, %outer.iv
+  %A.ptr = getelementptr inbounds i32, ptr %A, i64 %idx
+  %A.val = load i32, ptr %A.ptr, align 4, !llvm.access.group !3
+  %row.i32 = zext i8 %row to i32
+  %add = add i32 %A.val, %row.i32
+  store i32 %add, ptr %A.ptr, align 4, !llvm.access.group !3
+  %row.next = add i8 %row, 1
+  %inner.iv.next = add nuw nsw i64 %inner.iv, 1
+  %inner.iv.cmp = icmp eq i64 %inner.iv.next, 8
+  br i1 %inner.iv.cmp, label %mid.latch, label %inner.body
+
+mid.latch:
+  %mid.iv.next = add nuw nsw i64 %mid.iv, 1
+  %mid.iv.cmp = icmp eq i64 %mid.iv.next, 2
+  br i1 %mid.iv.cmp, label %outer.latch, label %mid.header
+
+outer.latch:
+  %outer.iv.next = add nuw nsw i64 %outer.iv, 1
+  %outer.iv.cmp = icmp eq i64 %outer.iv.next, 64
+  br i1 %outer.iv.cmp, label %exit, label %outer.header, !llvm.loop !5
+
+exit:
+  ret void
+}
+
 !0 = distinct !{!0, !1, !2}
 !1 = !{!"llvm.loop.vectorize.width", i32 4}
 !2 = !{!"llvm.loop.vectorize.enable"}
+!3 = distinct !{}
+!4 = !{!"llvm.loop.parallel_accesses", !3}
+!5 = distinct !{!5, !1, !2, !4}
+!6 = distinct !{!6, !1, !2}


        


More information about the llvm-commits mailing list