[llvm] [LV] Add cross-part load overlap heuristic for interleaving (PR #214500)

Sergey Shcherbinin via llvm-commits llvm-commits at lists.llvm.org
Mon Sep 21 06:33:08 PDT 2026


https://github.com/SergeyShch01 updated https://github.com/llvm/llvm-project/pull/214500

>From d08f5cf781b3fb9d4f84ffd045a50b5a62227c7f Mon Sep 17 00:00:00 2001
From: Sergey Shcherbinin <sscherbinin at nvidia.com>
Date: Sat, 19 Sep 2026 23:10:56 -0700
Subject: [PATCH 1/2] [LV] Pre-commit cross-part load redundancy tests

Add fixtures that record how the loop vectorizer currently vectorizes loops in
which a widened load of one logical part of an interleaved vector loop covers
exactly the elements that another widened load covers in the next part. The
interleave-count heuristics do not model that redundancy, so every loop here
vectorizes with an interleave count of 1.
---
 .../cross-part-load-cse-fixed-cases.ll        | 776 ++++++++++++++++++
 .../cross-part-load-cse-opportunities.ll      | 167 ++++
 .../AArch64/cross-part-load-cse-scalable.ll   | 158 ++++
 .../AArch64/cross-part-load-cse.ll            |  81 ++
 4 files changed, 1182 insertions(+)
 create mode 100644 llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-fixed-cases.ll
 create mode 100644 llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-opportunities.ll
 create mode 100644 llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-scalable.ll
 create mode 100644 llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse.ll

diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-fixed-cases.ll b/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-fixed-cases.ll
new file mode 100644
index 0000000000000..15948347005af
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-fixed-cases.ll
@@ -0,0 +1,776 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
+; Pre-commit test for the properties that a cross-part load redundancy
+; analysis has to respect: loaded value type, exact address equality,
+; intervening memory writes, access stride, address provenance,
+; poison-generating metadata, multiple scalar loop blocks and reverse
+; accesses. Every loop below vectorizes with an interleave count of 1 today.
+; RUN: opt -passes=loop-vectorize -force-vector-width=4 \
+; RUN:     -force-target-max-vector-interleave=2 -small-loop-cost=0 \
+; RUN:     -S %s | FileCheck %s
+
+target triple = "aarch64-unknown-linux-gnu"
+
+declare i32 @write_memory(i32) #0
+declare <4 x i32> @write_memory_v4(<4 x i32>)
+
+; The part-shifted addresses are equal, but the two loads have different value
+; types and therefore cannot share a loaded result.
+define void @different_types(ptr noalias %a, ptr noalias %c, i64 %n) {
+; CHECK-LABEL: define void @different_types(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 4
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP1:%.*]] = and i64 [[TMP0]], 3
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4
+; CHECK-NEXT:    [[TMP3:%.*]] = add nuw nsw i64 [[INDEX]], 4
+; CHECK-NEXT:    [[TMP4:%.*]] = getelementptr inbounds float, ptr [[A]], i64 [[TMP3]]
+; CHECK-NEXT:    [[WIDE_LOAD1:%.*]] = load <4 x float>, ptr [[TMP4]], align 4
+; CHECK-NEXT:    [[TMP5:%.*]] = bitcast <4 x float> [[WIDE_LOAD1]] to <4 x i32>
+; CHECK-NEXT:    [[TMP6:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[TMP5]]
+; CHECK-NEXT:    [[TMP7:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
+; CHECK-NEXT:    store <4 x i32> [[TMP6]], ptr [[TMP7]], align 4
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT:    [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; CHECK:       [[SCALAR_PH]]:
+; CHECK-NEXT:    [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[A_IV:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT:    [[L1:%.*]] = load i32, ptr [[A_IV]], align 4
+; CHECK-NEXT:    [[IV_PLUS_4:%.*]] = add nuw nsw i64 [[IV]], 4
+; CHECK-NEXT:    [[A_IV_PLUS_4:%.*]] = getelementptr inbounds float, ptr [[A]], i64 [[IV_PLUS_4]]
+; CHECK-NEXT:    [[L2:%.*]] = load float, ptr [[A_IV_PLUS_4]], align 4
+; CHECK-NEXT:    [[L2_BITS:%.*]] = bitcast float [[L2]] to i32
+; CHECK-NEXT:    [[SUM:%.*]] = add i32 [[L1]], [[L2_BITS]]
+; CHECK-NEXT:    [[C_IV:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[IV]]
+; CHECK-NEXT:    store i32 [[SUM]], ptr [[C_IV]], align 4
+; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT:    [[CMP:%.*]] = icmp slt i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[CMP]], label %[[LOOP]], label %[[EXIT]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %a.iv = getelementptr inbounds i32, ptr %a, i64 %iv
+  %l1 = load i32, ptr %a.iv, align 4
+  %iv.plus.4 = add nuw nsw i64 %iv, 4
+  %a.iv.plus.4 = getelementptr inbounds float, ptr %a, i64 %iv.plus.4
+  %l2 = load float, ptr %a.iv.plus.4, align 4
+  %l2.bits = bitcast float %l2 to i32
+  %sum = add i32 %l1, %l2.bits
+  %c.iv = getelementptr inbounds i32, ptr %c, i64 %iv
+  store i32 %sum, ptr %c.iv, align 4
+  %iv.next = add nuw nsw i64 %iv, 1
+  %cmp = icmp slt i64 %iv.next, %n
+  br i1 %cmp, label %loop, label %exit
+
+exit:
+  ret void
+}
+
+; a[i] + a[i+3]: the source offset 3 is not a multiple of VF=4, so no
+; part-shifted address ever matches exactly.
+define void @inequality(ptr noalias %a, ptr noalias %c, i64 %n) {
+; CHECK-LABEL: define void @inequality(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 4
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP1:%.*]] = and i64 [[TMP0]], 3
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4
+; CHECK-NEXT:    [[TMP3:%.*]] = add nuw nsw i64 [[INDEX]], 3
+; CHECK-NEXT:    [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP3]]
+; CHECK-NEXT:    [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP4]], align 4
+; CHECK-NEXT:    [[TMP5:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[WIDE_LOAD1]]
+; CHECK-NEXT:    [[TMP6:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
+; CHECK-NEXT:    store <4 x i32> [[TMP5]], ptr [[TMP6]], align 4
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT:    [[TMP7:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP7]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; CHECK:       [[SCALAR_PH]]:
+; CHECK-NEXT:    [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[A_IV:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT:    [[L1:%.*]] = load i32, ptr [[A_IV]], align 4
+; CHECK-NEXT:    [[IV_PLUS_3:%.*]] = add nuw nsw i64 [[IV]], 3
+; CHECK-NEXT:    [[A_IV_PLUS_3:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV_PLUS_3]]
+; CHECK-NEXT:    [[L2:%.*]] = load i32, ptr [[A_IV_PLUS_3]], align 4
+; CHECK-NEXT:    [[SUM:%.*]] = add i32 [[L1]], [[L2]]
+; CHECK-NEXT:    [[C_IV:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[IV]]
+; CHECK-NEXT:    store i32 [[SUM]], ptr [[C_IV]], align 4
+; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT:    [[CMP:%.*]] = icmp slt i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[CMP]], label %[[LOOP]], label %[[EXIT]], !llvm.loop [[LOOP5:![0-9]+]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %a.iv = getelementptr inbounds i32, ptr %a, i64 %iv
+  %l1 = load i32, ptr %a.iv, align 4
+  %iv.plus.3 = add nuw nsw i64 %iv, 3
+  %a.iv.plus.3 = getelementptr inbounds i32, ptr %a, i64 %iv.plus.3
+  %l2 = load i32, ptr %a.iv.plus.3, align 4
+  %sum = add i32 %l1, %l2
+  %c.iv = getelementptr inbounds i32, ptr %c, i64 %iv
+  store i32 %sum, ptr %c.iv, align 4
+  %iv.next = add nuw nsw i64 %iv, 1
+  %cmp = icmp slt i64 %iv.next, %n
+  br i1 %cmp, label %loop, label %exit
+
+exit:
+  ret void
+}
+
+; A may-alias store occurs between the two matching logical-part loads, so their
+; values cannot be assumed to survive from one part to the other.
+define void @write_between(ptr %a, ptr %b, ptr noalias %c, i64 %n) {
+; CHECK-LABEL: define void @write_between(
+; CHECK-SAME: ptr [[A:%.*]], ptr [[B:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 4
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_MEMCHECK:.*]]
+; CHECK:       [[VECTOR_MEMCHECK]]:
+; CHECK-NEXT:    [[SMAX:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; CHECK-NEXT:    [[TMP1:%.*]] = shl i64 [[SMAX]], 2
+; CHECK-NEXT:    [[SCEVGEP:%.*]] = getelementptr i8, ptr [[B]], i64 [[TMP1]]
+; CHECK-NEXT:    [[TMP2:%.*]] = add i64 [[TMP1]], 16
+; CHECK-NEXT:    [[SCEVGEP1:%.*]] = getelementptr i8, ptr [[A]], i64 [[TMP2]]
+; CHECK-NEXT:    [[BOUND0:%.*]] = icmp ult ptr [[B]], [[SCEVGEP1]]
+; CHECK-NEXT:    [[BOUND1:%.*]] = icmp ult ptr [[A]], [[SCEVGEP]]
+; CHECK-NEXT:    [[FOUND_CONFLICT:%.*]] = and i1 [[BOUND0]], [[BOUND1]]
+; CHECK-NEXT:    br i1 [[FOUND_CONFLICT]], label %[[SCALAR_PH]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP3:%.*]] = and i64 [[TMP0]], 3
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP3]]
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP4]], align 4, !alias.scope [[META6:![0-9]+]]
+; CHECK-NEXT:    [[TMP5:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 [[INDEX]]
+; CHECK-NEXT:    store <4 x i32> [[WIDE_LOAD]], ptr [[TMP5]], align 4, !alias.scope [[META9:![0-9]+]], !noalias [[META6]]
+; CHECK-NEXT:    [[TMP6:%.*]] = add nuw nsw i64 [[INDEX]], 4
+; CHECK-NEXT:    [[TMP7:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP6]]
+; CHECK-NEXT:    [[WIDE_LOAD2:%.*]] = load <4 x i32>, ptr [[TMP7]], align 4, !alias.scope [[META6]]
+; CHECK-NEXT:    [[TMP8:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[WIDE_LOAD2]]
+; CHECK-NEXT:    [[TMP9:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
+; CHECK-NEXT:    store <4 x i32> [[TMP8]], ptr [[TMP9]], align 4
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT:    [[TMP10:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP10]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP11:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; CHECK:       [[SCALAR_PH]]:
+; CHECK-NEXT:    [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ], [ 0, %[[VECTOR_MEMCHECK]] ]
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[A_IV:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT:    [[L1:%.*]] = load i32, ptr [[A_IV]], align 4
+; CHECK-NEXT:    [[B_IV:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 [[IV]]
+; CHECK-NEXT:    store i32 [[L1]], ptr [[B_IV]], align 4
+; CHECK-NEXT:    [[IV_PLUS_4:%.*]] = add nuw nsw i64 [[IV]], 4
+; CHECK-NEXT:    [[A_IV_PLUS_4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV_PLUS_4]]
+; CHECK-NEXT:    [[L2:%.*]] = load i32, ptr [[A_IV_PLUS_4]], align 4
+; CHECK-NEXT:    [[SUM:%.*]] = add i32 [[L1]], [[L2]]
+; CHECK-NEXT:    [[C_IV:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[IV]]
+; CHECK-NEXT:    store i32 [[SUM]], ptr [[C_IV]], align 4
+; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT:    [[CMP:%.*]] = icmp slt i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[CMP]], label %[[LOOP]], label %[[EXIT]], !llvm.loop [[LOOP12:![0-9]+]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %a.iv = getelementptr inbounds i32, ptr %a, i64 %iv
+  %l1 = load i32, ptr %a.iv, align 4
+  %b.iv = getelementptr inbounds i32, ptr %b, i64 %iv
+  store i32 %l1, ptr %b.iv, align 4
+  %iv.plus.4 = add nuw nsw i64 %iv, 4
+  %a.iv.plus.4 = getelementptr inbounds i32, ptr %a, i64 %iv.plus.4
+  %l2 = load i32, ptr %a.iv.plus.4, align 4
+  %sum = add i32 %l1, %l2
+  %c.iv = getelementptr inbounds i32, ptr %c, i64 %iv
+  store i32 %sum, ptr %c.iv, align 4
+  %iv.next = add nuw nsw i64 %iv, 1
+  %cmp = icmp slt i64 %iv.next, %n
+  br i1 %cmp, label %loop, label %exit
+
+exit:
+  ret void
+}
+
+; A vector-mapped call with unknown memory effects occurs between the matching
+; loads. It may write through memory that its arguments do not describe, so no
+; loaded value may be assumed to survive across it.
+define void @writing_call_between(ptr noalias %a, ptr noalias %c, i64 %n) {
+; CHECK-LABEL: define void @writing_call_between(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 4
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP1:%.*]] = and i64 [[TMP0]], 3
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4
+; CHECK-NEXT:    [[TMP3:%.*]] = call <4 x i32> @write_memory_v4(<4 x i32> [[WIDE_LOAD]])
+; CHECK-NEXT:    [[TMP4:%.*]] = add nuw nsw i64 [[INDEX]], 4
+; CHECK-NEXT:    [[TMP5:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP4]]
+; CHECK-NEXT:    [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP5]], align 4
+; CHECK-NEXT:    [[TMP6:%.*]] = add <4 x i32> [[TMP3]], [[WIDE_LOAD1]]
+; CHECK-NEXT:    [[TMP7:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
+; CHECK-NEXT:    store <4 x i32> [[TMP6]], ptr [[TMP7]], align 4
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT:    [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP13:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; CHECK:       [[SCALAR_PH]]:
+; CHECK-NEXT:    [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[A_IV:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT:    [[L1:%.*]] = load i32, ptr [[A_IV]], align 4
+; CHECK-NEXT:    [[CALL:%.*]] = call i32 @write_memory(i32 [[L1]])
+; CHECK-NEXT:    [[IV_PLUS_4:%.*]] = add nuw nsw i64 [[IV]], 4
+; CHECK-NEXT:    [[A_IV_PLUS_4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV_PLUS_4]]
+; CHECK-NEXT:    [[L2:%.*]] = load i32, ptr [[A_IV_PLUS_4]], align 4
+; CHECK-NEXT:    [[SUM:%.*]] = add i32 [[CALL]], [[L2]]
+; CHECK-NEXT:    [[C_IV:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[IV]]
+; CHECK-NEXT:    store i32 [[SUM]], ptr [[C_IV]], align 4
+; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT:    [[CMP:%.*]] = icmp slt i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[CMP]], label %[[LOOP]], label %[[EXIT]], !llvm.loop [[LOOP14:![0-9]+]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %a.iv = getelementptr inbounds i32, ptr %a, i64 %iv
+  %l1 = load i32, ptr %a.iv, align 4
+  %call = call i32 @write_memory(i32 %l1)
+  %iv.plus.4 = add nuw nsw i64 %iv, 4
+  %a.iv.plus.4 = getelementptr inbounds i32, ptr %a, i64 %iv.plus.4
+  %l2 = load i32, ptr %a.iv.plus.4, align 4
+  %sum = add i32 %call, %l2
+  %c.iv = getelementptr inbounds i32, ptr %c, i64 %iv
+  store i32 %sum, ptr %c.iv, align 4
+  %iv.next = add nuw nsw i64 %iv, 1
+  %cmp = icmp slt i64 %iv.next, %n
+  br i1 %cmp, label %loop, label %exit
+
+exit:
+  ret void
+}
+
+; Stride-2 accesses are represented by interleave-group recipes rather than by
+; simple consecutive widened loads.
+define void @non_unit_stride(ptr noalias %a, ptr noalias %c, i64 %n) {
+; CHECK-LABEL: define void @non_unit_stride(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ule i64 [[TMP0]], 4
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP1:%.*]] = and i64 [[TMP0]], 3
+; CHECK-NEXT:    [[TMP2:%.*]] = icmp eq i64 [[TMP1]], 0
+; CHECK-NEXT:    [[TMP3:%.*]] = select i1 [[TMP2]], i64 4, i64 [[TMP1]]
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP3]]
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP4:%.*]] = shl nuw nsw i64 [[INDEX]], 1
+; CHECK-NEXT:    [[TMP5:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP4]]
+; CHECK-NEXT:    [[WIDE_VEC:%.*]] = load <8 x i32>, ptr [[TMP5]], align 4
+; CHECK-NEXT:    [[STRIDED_VEC:%.*]] = shufflevector <8 x i32> [[WIDE_VEC]], <8 x i32> poison, <4 x i32> <i32 0, i32 2, i32 4, i32 6>
+; CHECK-NEXT:    [[TMP6:%.*]] = add nuw nsw i64 [[TMP4]], 4
+; CHECK-NEXT:    [[TMP7:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP6]]
+; CHECK-NEXT:    [[WIDE_VEC1:%.*]] = load <8 x i32>, ptr [[TMP7]], align 4
+; CHECK-NEXT:    [[STRIDED_VEC2:%.*]] = shufflevector <8 x i32> [[WIDE_VEC1]], <8 x i32> poison, <4 x i32> <i32 0, i32 2, i32 4, i32 6>
+; CHECK-NEXT:    [[TMP8:%.*]] = add <4 x i32> [[STRIDED_VEC]], [[STRIDED_VEC2]]
+; CHECK-NEXT:    [[TMP9:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
+; CHECK-NEXT:    store <4 x i32> [[TMP8]], ptr [[TMP9]], align 4
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT:    [[TMP10:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP10]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP15:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    br label %[[SCALAR_PH]]
+; CHECK:       [[SCALAR_PH]]:
+; CHECK-NEXT:    [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[IV_TWICE:%.*]] = shl nuw nsw i64 [[IV]], 1
+; CHECK-NEXT:    [[A_IV:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV_TWICE]]
+; CHECK-NEXT:    [[L1:%.*]] = load i32, ptr [[A_IV]], align 4
+; CHECK-NEXT:    [[IV_TWICE_PLUS_4:%.*]] = add nuw nsw i64 [[IV_TWICE]], 4
+; CHECK-NEXT:    [[A_IV_PLUS_4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV_TWICE_PLUS_4]]
+; CHECK-NEXT:    [[L2:%.*]] = load i32, ptr [[A_IV_PLUS_4]], align 4
+; CHECK-NEXT:    [[SUM:%.*]] = add i32 [[L1]], [[L2]]
+; CHECK-NEXT:    [[C_IV:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[IV]]
+; CHECK-NEXT:    store i32 [[SUM]], ptr [[C_IV]], align 4
+; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT:    [[CMP:%.*]] = icmp slt i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[CMP]], label %[[LOOP]], label %[[EXIT:.*]], !llvm.loop [[LOOP16:![0-9]+]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %iv.twice = shl nuw nsw i64 %iv, 1
+  %a.iv = getelementptr inbounds i32, ptr %a, i64 %iv.twice
+  %l1 = load i32, ptr %a.iv, align 4
+  %iv.twice.plus.4 = add nuw nsw i64 %iv.twice, 4
+  %a.iv.plus.4 = getelementptr inbounds i32, ptr %a, i64 %iv.twice.plus.4
+  %l2 = load i32, ptr %a.iv.plus.4, align 4
+  %sum = add i32 %l1, %l2
+  %c.iv = getelementptr inbounds i32, ptr %c, i64 %iv
+  store i32 %sum, ptr %c.iv, align 4
+  %iv.next = add nuw nsw i64 %iv, 1
+  %cmp = icmp slt i64 %iv.next, %n
+  br i1 %cmp, label %loop, label %exit
+
+exit:
+  ret void
+}
+
+; VPlan folds each identical-arm select to its underlying GEP, while
+; ScalarEvolution keeps each nonconstant pointer select as a distinct unknown.
+; The address of these loads is therefore only exact in the folded VPlan value,
+; not in the scalar load ingredient.
+define void @folded_provenance(ptr noalias %a, ptr noalias %c, i64 %n, i1 %cond) {
+; CHECK-LABEL: define void @folded_provenance(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]], i1 [[COND:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 4
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP1:%.*]] = and i64 [[TMP0]], 3
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4
+; CHECK-NEXT:    [[TMP3:%.*]] = add nuw nsw i64 [[INDEX]], 4
+; CHECK-NEXT:    [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP3]]
+; CHECK-NEXT:    [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP4]], align 4
+; CHECK-NEXT:    [[TMP5:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[WIDE_LOAD1]]
+; CHECK-NEXT:    [[TMP6:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
+; CHECK-NEXT:    store <4 x i32> [[TMP5]], ptr [[TMP6]], align 4
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT:    [[TMP7:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP7]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP17:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; CHECK:       [[SCALAR_PH]]:
+; CHECK-NEXT:    [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[A_IV:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT:    [[P1:%.*]] = select i1 [[COND]], ptr [[A_IV]], ptr [[A_IV]]
+; CHECK-NEXT:    [[L1:%.*]] = load i32, ptr [[P1]], align 4
+; CHECK-NEXT:    [[IV_PLUS_4:%.*]] = add nuw nsw i64 [[IV]], 4
+; CHECK-NEXT:    [[A_IV_PLUS_4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV_PLUS_4]]
+; CHECK-NEXT:    [[P2:%.*]] = select i1 [[COND]], ptr [[A_IV_PLUS_4]], ptr [[A_IV_PLUS_4]]
+; CHECK-NEXT:    [[L2:%.*]] = load i32, ptr [[P2]], align 4
+; CHECK-NEXT:    [[SUM:%.*]] = add i32 [[L1]], [[L2]]
+; CHECK-NEXT:    [[C_IV:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[IV]]
+; CHECK-NEXT:    store i32 [[SUM]], ptr [[C_IV]], align 4
+; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT:    [[CMP:%.*]] = icmp slt i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[CMP]], label %[[LOOP]], label %[[EXIT]], !llvm.loop [[LOOP18:![0-9]+]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %a.iv = getelementptr inbounds i32, ptr %a, i64 %iv
+  %p1 = select i1 %cond, ptr %a.iv, ptr %a.iv
+  %l1 = load i32, ptr %p1, align 4
+  %iv.plus.4 = add nuw nsw i64 %iv, 4
+  %a.iv.plus.4 = getelementptr inbounds i32, ptr %a, i64 %iv.plus.4
+  %p2 = select i1 %cond, ptr %a.iv.plus.4, ptr %a.iv.plus.4
+  %l2 = load i32, ptr %p2, align 4
+  %sum = add i32 %l1, %l2
+  %c.iv = getelementptr inbounds i32, ptr %c, i64 %iv
+  store i32 %sum, ptr %c.iv, align 4
+  %iv.next = add nuw nsw i64 %iv, 1
+  %cmp = icmp slt i64 %iv.next, %n
+  br i1 %cmp, label %loop, label %exit
+
+exit:
+  ret void
+}
+
+; The two scalar loads carry different poison-generating !range metadata, which
+; is not propagated to the widened loads.
+define void @poison_annotations(ptr noalias %a, ptr noalias %c, i64 %n) {
+; CHECK-LABEL: define void @poison_annotations(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 4
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP1:%.*]] = and i64 [[TMP0]], 3
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4
+; CHECK-NEXT:    [[TMP3:%.*]] = add nuw nsw i64 [[INDEX]], 4
+; CHECK-NEXT:    [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP3]]
+; CHECK-NEXT:    [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP4]], align 4
+; CHECK-NEXT:    [[TMP5:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[WIDE_LOAD1]]
+; CHECK-NEXT:    [[TMP6:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
+; CHECK-NEXT:    store <4 x i32> [[TMP5]], ptr [[TMP6]], align 4
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT:    [[TMP7:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP7]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP19:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; CHECK:       [[SCALAR_PH]]:
+; CHECK-NEXT:    [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[A_IV:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT:    [[L1:%.*]] = load i32, ptr [[A_IV]], align 4, !range [[RNG20:![0-9]+]]
+; CHECK-NEXT:    [[IV_PLUS_4:%.*]] = add nuw nsw i64 [[IV]], 4
+; CHECK-NEXT:    [[A_IV_PLUS_4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV_PLUS_4]]
+; CHECK-NEXT:    [[L2:%.*]] = load i32, ptr [[A_IV_PLUS_4]], align 4, !range [[RNG21:![0-9]+]]
+; CHECK-NEXT:    [[SUM:%.*]] = add i32 [[L1]], [[L2]]
+; CHECK-NEXT:    [[C_IV:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[IV]]
+; CHECK-NEXT:    store i32 [[SUM]], ptr [[C_IV]], align 4
+; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT:    [[CMP:%.*]] = icmp slt i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[CMP]], label %[[LOOP]], label %[[EXIT]], !llvm.loop [[LOOP22:![0-9]+]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %a.iv = getelementptr inbounds i32, ptr %a, i64 %iv
+  %l1 = load i32, ptr %a.iv, align 4, !range !0
+  %iv.plus.4 = add nuw nsw i64 %iv, 4
+  %a.iv.plus.4 = getelementptr inbounds i32, ptr %a, i64 %iv.plus.4
+  %l2 = load i32, ptr %a.iv.plus.4, align 4, !range !1
+  %sum = add i32 %l1, %l2
+  %c.iv = getelementptr inbounds i32, ptr %c, i64 %iv
+  store i32 %sum, ptr %c.iv, align 4
+  %iv.next = add nuw nsw i64 %iv, 1
+  %cmp = icmp slt i64 %iv.next, %n
+  br i1 %cmp, label %loop, label %exit
+
+exit:
+  ret void
+}
+
+; The scalar loop has a separate latch, while VPlan places the two matching
+; loads in its single vector-loop block.
+define void @multi_block(ptr noalias %a, ptr noalias %c, i64 %n) {
+; CHECK-LABEL: define void @multi_block(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 4
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP1:%.*]] = and i64 [[TMP0]], 3
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4
+; CHECK-NEXT:    [[TMP3:%.*]] = add nuw nsw i64 [[INDEX]], 4
+; CHECK-NEXT:    [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP3]]
+; CHECK-NEXT:    [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP4]], align 4
+; CHECK-NEXT:    [[TMP5:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[WIDE_LOAD1]]
+; CHECK-NEXT:    [[TMP6:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
+; CHECK-NEXT:    store <4 x i32> [[TMP5]], ptr [[TMP6]], align 4
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT:    [[TMP7:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP7]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP23:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; CHECK:       [[SCALAR_PH]]:
+; CHECK-NEXT:    [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT:    br label %[[LOOP_HEADER:.*]]
+; CHECK:       [[LOOP_HEADER]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP_LATCH:.*]] ]
+; CHECK-NEXT:    [[A_IV:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT:    [[L1:%.*]] = load i32, ptr [[A_IV]], align 4
+; CHECK-NEXT:    [[IV_PLUS_4:%.*]] = add nuw nsw i64 [[IV]], 4
+; CHECK-NEXT:    [[A_IV_PLUS_4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV_PLUS_4]]
+; CHECK-NEXT:    [[L2:%.*]] = load i32, ptr [[A_IV_PLUS_4]], align 4
+; CHECK-NEXT:    [[SUM:%.*]] = add i32 [[L1]], [[L2]]
+; CHECK-NEXT:    [[C_IV:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[IV]]
+; CHECK-NEXT:    store i32 [[SUM]], ptr [[C_IV]], align 4
+; CHECK-NEXT:    br label %[[LOOP_LATCH]]
+; CHECK:       [[LOOP_LATCH]]:
+; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT:    [[CMP:%.*]] = icmp slt i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[CMP]], label %[[LOOP_HEADER]], label %[[EXIT]], !llvm.loop [[LOOP24:![0-9]+]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %loop.header
+
+loop.header:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop.latch ]
+  %a.iv = getelementptr inbounds i32, ptr %a, i64 %iv
+  %l1 = load i32, ptr %a.iv, align 4
+  %iv.plus.4 = add nuw nsw i64 %iv, 4
+  %a.iv.plus.4 = getelementptr inbounds i32, ptr %a, i64 %iv.plus.4
+  %l2 = load i32, ptr %a.iv.plus.4, align 4
+  %sum = add i32 %l1, %l2
+  %c.iv = getelementptr inbounds i32, ptr %c, i64 %iv
+  store i32 %sum, ptr %c.iv, align 4
+  br label %loop.latch
+
+loop.latch:
+  %iv.next = add nuw nsw i64 %iv, 1
+  %cmp = icmp slt i64 %iv.next, %n
+  br i1 %cmp, label %loop.header, label %exit
+
+exit:
+  ret void
+}
+
+; At VF=4, the first load in part 1 and the second load in part 0 both use the
+; reverse vector ending at a[last-iv-7].
+define void @reverse_redundancy(ptr noalias %a, ptr noalias %c, i64 %n) {
+; CHECK-LABEL: define void @reverse_redundancy(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[LAST:%.*]] = add i64 [[N]], -1
+; CHECK-NEXT:    [[TRIP_COUNT:%.*]] = add i64 [[N]], -4
+; CHECK-NEXT:    [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[TRIP_COUNT]], i64 1)
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 4
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP1:%.*]] = and i64 [[TMP0]], 3
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP2:%.*]] = sub i64 [[LAST]], [[INDEX]]
+; CHECK-NEXT:    [[TMP3:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP2]]
+; CHECK-NEXT:    [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[TMP3]], i64 -3
+; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP4]], align 4
+; CHECK-NEXT:    [[TMP5:%.*]] = add i64 [[TMP2]], -4
+; CHECK-NEXT:    [[TMP6:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP5]]
+; CHECK-NEXT:    [[TMP7:%.*]] = getelementptr inbounds i32, ptr [[TMP6]], i64 -3
+; CHECK-NEXT:    [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP7]], align 4
+; CHECK-NEXT:    [[TMP8:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[WIDE_LOAD1]]
+; CHECK-NEXT:    [[REVERSE:%.*]] = shufflevector <4 x i32> [[TMP8]], <4 x i32> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
+; CHECK-NEXT:    [[TMP9:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
+; CHECK-NEXT:    store <4 x i32> [[REVERSE]], ptr [[TMP9]], align 4
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT:    [[TMP10:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP10]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP25:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; CHECK:       [[SCALAR_PH]]:
+; CHECK-NEXT:    [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[REVERSE_IV:%.*]] = sub i64 [[LAST]], [[IV]]
+; CHECK-NEXT:    [[A_REVERSE:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[REVERSE_IV]]
+; CHECK-NEXT:    [[L1:%.*]] = load i32, ptr [[A_REVERSE]], align 4
+; CHECK-NEXT:    [[REVERSE_IV_MINUS_4:%.*]] = add i64 [[REVERSE_IV]], -4
+; CHECK-NEXT:    [[A_REVERSE_MINUS_4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[REVERSE_IV_MINUS_4]]
+; CHECK-NEXT:    [[L2:%.*]] = load i32, ptr [[A_REVERSE_MINUS_4]], align 4
+; CHECK-NEXT:    [[SUM:%.*]] = add i32 [[L1]], [[L2]]
+; CHECK-NEXT:    [[C_IV:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[IV]]
+; CHECK-NEXT:    store i32 [[SUM]], ptr [[C_IV]], align 4
+; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT:    [[CMP:%.*]] = icmp slt i64 [[IV_NEXT]], [[TRIP_COUNT]]
+; CHECK-NEXT:    br i1 [[CMP]], label %[[LOOP]], label %[[EXIT]], !llvm.loop [[LOOP26:![0-9]+]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  %last = add i64 %n, -1
+  %trip.count = add i64 %n, -4
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %reverse.iv = sub i64 %last, %iv
+  %a.reverse = getelementptr inbounds i32, ptr %a, i64 %reverse.iv
+  %l1 = load i32, ptr %a.reverse, align 4
+  %reverse.iv.minus.4 = add i64 %reverse.iv, -4
+  %a.reverse.minus.4 = getelementptr inbounds i32, ptr %a, i64 %reverse.iv.minus.4
+  %l2 = load i32, ptr %a.reverse.minus.4, align 4
+  %sum = add i32 %l1, %l2
+  %c.iv = getelementptr inbounds i32, ptr %c, i64 %iv
+  store i32 %sum, ptr %c.iv, align 4
+  %iv.next = add nuw nsw i64 %iv, 1
+  %cmp = icmp slt i64 %iv.next, %trip.count
+  br i1 %cmp, label %loop, label %exit
+
+exit:
+  ret void
+}
+
+; The reverse access sits physically between the two matching forward loads.
+; Its end-pointer address computation is pure and must not be mistaken for a
+; memory write.
+define void @reverse_between(ptr noalias %a, ptr noalias %c, i64 %n) {
+; CHECK-LABEL: define void @reverse_between(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[LAST:%.*]] = add i64 [[N]], -1
+; CHECK-NEXT:    [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 4
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP1:%.*]] = and i64 [[TMP0]], 3
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4
+; CHECK-NEXT:    [[TMP3:%.*]] = sub i64 [[LAST]], [[INDEX]]
+; CHECK-NEXT:    [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP3]]
+; CHECK-NEXT:    [[TMP5:%.*]] = getelementptr inbounds i32, ptr [[TMP4]], i64 -3
+; CHECK-NEXT:    [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP5]], align 4
+; CHECK-NEXT:    [[REVERSE2:%.*]] = shufflevector <4 x i32> [[WIDE_LOAD1]], <4 x i32> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
+; CHECK-NEXT:    [[TMP6:%.*]] = add nuw nsw i64 [[INDEX]], 4
+; CHECK-NEXT:    [[TMP7:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP6]]
+; CHECK-NEXT:    [[WIDE_LOAD3:%.*]] = load <4 x i32>, ptr [[TMP7]], align 4
+; CHECK-NEXT:    [[TMP8:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[WIDE_LOAD3]]
+; CHECK-NEXT:    [[TMP9:%.*]] = add <4 x i32> [[TMP8]], [[REVERSE2]]
+; CHECK-NEXT:    [[TMP10:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
+; CHECK-NEXT:    store <4 x i32> [[TMP9]], ptr [[TMP10]], align 4
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT:    [[TMP11:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP11]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP27:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; CHECK:       [[SCALAR_PH]]:
+; CHECK-NEXT:    [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[A_IV:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT:    [[L1:%.*]] = load i32, ptr [[A_IV]], align 4
+; CHECK-NEXT:    [[IV_REVERSE:%.*]] = sub i64 [[LAST]], [[IV]]
+; CHECK-NEXT:    [[A_REVERSE:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV_REVERSE]]
+; CHECK-NEXT:    [[REVERSE:%.*]] = load i32, ptr [[A_REVERSE]], align 4
+; CHECK-NEXT:    [[IV_PLUS_4:%.*]] = add nuw nsw i64 [[IV]], 4
+; CHECK-NEXT:    [[A_IV_PLUS_4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV_PLUS_4]]
+; CHECK-NEXT:    [[L2:%.*]] = load i32, ptr [[A_IV_PLUS_4]], align 4
+; CHECK-NEXT:    [[SUM_FORWARD:%.*]] = add i32 [[L1]], [[L2]]
+; CHECK-NEXT:    [[SUM:%.*]] = add i32 [[SUM_FORWARD]], [[REVERSE]]
+; CHECK-NEXT:    [[C_IV:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[IV]]
+; CHECK-NEXT:    store i32 [[SUM]], ptr [[C_IV]], align 4
+; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT:    [[CMP:%.*]] = icmp slt i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[CMP]], label %[[LOOP]], label %[[EXIT]], !llvm.loop [[LOOP28:![0-9]+]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  %last = add i64 %n, -1
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %a.iv = getelementptr inbounds i32, ptr %a, i64 %iv
+  %l1 = load i32, ptr %a.iv, align 4
+  %iv.reverse = sub i64 %last, %iv
+  %a.reverse = getelementptr inbounds i32, ptr %a, i64 %iv.reverse
+  %reverse = load i32, ptr %a.reverse, align 4
+  %iv.plus.4 = add nuw nsw i64 %iv, 4
+  %a.iv.plus.4 = getelementptr inbounds i32, ptr %a, i64 %iv.plus.4
+  %l2 = load i32, ptr %a.iv.plus.4, align 4
+  %sum.forward = add i32 %l1, %l2
+  %sum = add i32 %sum.forward, %reverse
+  %c.iv = getelementptr inbounds i32, ptr %c, i64 %iv
+  store i32 %sum, ptr %c.iv, align 4
+  %iv.next = add nuw nsw i64 %iv, 1
+  %cmp = icmp slt i64 %iv.next, %n
+  br i1 %cmp, label %loop, label %exit
+
+exit:
+  ret void
+}
+
+attributes #0 = { nounwind willreturn "vector-function-abi-variant"="_ZGVnN4v_write_memory(write_memory_v4)" }
+
+!0 = !{i32 0, i32 100}
+!1 = !{i32 0, i32 101}
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-opportunities.ll b/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-opportunities.ll
new file mode 100644
index 0000000000000..fe8e8c0b53880
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-opportunities.ll
@@ -0,0 +1,167 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
+; Pre-commit test for counting cross-part load redundancy opportunities. The
+; two loops below differ only in how many distinct opportunities they contain;
+; both vectorize with an interleave count of 1 today.
+; RUN: opt -passes=loop-vectorize -force-vector-width=4 \
+; RUN:     -force-target-max-vector-interleave=2 -small-loop-cost=0 \
+; RUN:     -S %s | FileCheck %s
+
+target triple = "aarch64-unknown-linux-gnu"
+
+; The first a[i+4] load forms a cross-part redundancy opportunity with a[i].
+; The second a[i+4] load is a duplicate within the same logical part: it already
+; exists without interleaving and therefore is no additional cross-part
+; opportunity, which leaves a single opportunity in this loop.
+define void @duplicate_after_cross_part(ptr noalias %a, ptr noalias %c, i64 %n) {
+; CHECK-LABEL: define void @duplicate_after_cross_part(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 4
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP1:%.*]] = and i64 [[TMP0]], 3
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4
+; CHECK-NEXT:    [[TMP3:%.*]] = add nuw nsw i64 [[INDEX]], 4
+; CHECK-NEXT:    [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP3]]
+; CHECK-NEXT:    [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP4]], align 4
+; CHECK-NEXT:    [[TMP5:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[WIDE_LOAD1]]
+; CHECK-NEXT:    [[TMP6:%.*]] = add <4 x i32> [[TMP5]], [[WIDE_LOAD1]]
+; CHECK-NEXT:    [[TMP7:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
+; CHECK-NEXT:    store <4 x i32> [[TMP6]], ptr [[TMP7]], align 4
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT:    [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; CHECK:       [[SCALAR_PH]]:
+; CHECK-NEXT:    [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[A_IV:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT:    [[L1:%.*]] = load i32, ptr [[A_IV]], align 4
+; CHECK-NEXT:    [[IV_PLUS_4:%.*]] = add nuw nsw i64 [[IV]], 4
+; CHECK-NEXT:    [[A_IV_PLUS_4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV_PLUS_4]]
+; CHECK-NEXT:    [[L2:%.*]] = load i32, ptr [[A_IV_PLUS_4]], align 4
+; CHECK-NEXT:    [[L3:%.*]] = load i32, ptr [[A_IV_PLUS_4]], align 4
+; CHECK-NEXT:    [[SUM_1:%.*]] = add i32 [[L1]], [[L2]]
+; CHECK-NEXT:    [[SUM_2:%.*]] = add i32 [[SUM_1]], [[L3]]
+; CHECK-NEXT:    [[C_IV:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[IV]]
+; CHECK-NEXT:    store i32 [[SUM_2]], ptr [[C_IV]], align 4
+; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT:    [[CMP:%.*]] = icmp slt i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[CMP]], label %[[LOOP]], label %[[EXIT]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %a.iv = getelementptr inbounds i32, ptr %a, i64 %iv
+  %l1 = load i32, ptr %a.iv, align 4
+  %iv.plus.4 = add nuw nsw i64 %iv, 4
+  %a.iv.plus.4 = getelementptr inbounds i32, ptr %a, i64 %iv.plus.4
+  %l2 = load i32, ptr %a.iv.plus.4, align 4
+  %l3 = load i32, ptr %a.iv.plus.4, align 4
+  %sum.1 = add i32 %l1, %l2
+  %sum.2 = add i32 %sum.1, %l3
+  %c.iv = getelementptr inbounds i32, ptr %c, i64 %iv
+  store i32 %sum.2, ptr %c.iv, align 4
+  %iv.next = add nuw nsw i64 %iv, 1
+  %cmp = icmp slt i64 %iv.next, %n
+  br i1 %cmp, label %loop, label %exit
+
+exit:
+  ret void
+}
+
+; At VF=4, a[i+4] in part 0 is redundant with a[i] in part 1, and a[i+8] in
+; part 0 is independently redundant with a[i+4] in part 1, which gives two
+; distinct opportunities in this loop.
+define void @two_opportunities(ptr noalias %a, ptr noalias %c, i64 %n) {
+; CHECK-LABEL: define void @two_opportunities(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 4
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP1:%.*]] = and i64 [[TMP0]], 3
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4
+; CHECK-NEXT:    [[TMP3:%.*]] = add nuw nsw i64 [[INDEX]], 4
+; CHECK-NEXT:    [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP3]]
+; CHECK-NEXT:    [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP4]], align 4
+; CHECK-NEXT:    [[TMP5:%.*]] = add nuw nsw i64 [[INDEX]], 8
+; CHECK-NEXT:    [[TMP6:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP5]]
+; CHECK-NEXT:    [[WIDE_LOAD2:%.*]] = load <4 x i32>, ptr [[TMP6]], align 4
+; CHECK-NEXT:    [[TMP7:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[WIDE_LOAD1]]
+; CHECK-NEXT:    [[TMP8:%.*]] = add <4 x i32> [[TMP7]], [[WIDE_LOAD2]]
+; CHECK-NEXT:    [[TMP9:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
+; CHECK-NEXT:    store <4 x i32> [[TMP8]], ptr [[TMP9]], align 4
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT:    [[TMP10:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP10]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; CHECK:       [[SCALAR_PH]]:
+; CHECK-NEXT:    [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[A_IV:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT:    [[L1:%.*]] = load i32, ptr [[A_IV]], align 4
+; CHECK-NEXT:    [[IV_PLUS_4:%.*]] = add nuw nsw i64 [[IV]], 4
+; CHECK-NEXT:    [[A_IV_PLUS_4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV_PLUS_4]]
+; CHECK-NEXT:    [[L2:%.*]] = load i32, ptr [[A_IV_PLUS_4]], align 4
+; CHECK-NEXT:    [[IV_PLUS_8:%.*]] = add nuw nsw i64 [[IV]], 8
+; CHECK-NEXT:    [[A_IV_PLUS_8:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV_PLUS_8]]
+; CHECK-NEXT:    [[L3:%.*]] = load i32, ptr [[A_IV_PLUS_8]], align 4
+; CHECK-NEXT:    [[SUM_1:%.*]] = add i32 [[L1]], [[L2]]
+; CHECK-NEXT:    [[SUM_2:%.*]] = add i32 [[SUM_1]], [[L3]]
+; CHECK-NEXT:    [[C_IV:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[IV]]
+; CHECK-NEXT:    store i32 [[SUM_2]], ptr [[C_IV]], align 4
+; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT:    [[CMP:%.*]] = icmp slt i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[CMP]], label %[[LOOP]], label %[[EXIT]], !llvm.loop [[LOOP5:![0-9]+]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %a.iv = getelementptr inbounds i32, ptr %a, i64 %iv
+  %l1 = load i32, ptr %a.iv, align 4
+  %iv.plus.4 = add nuw nsw i64 %iv, 4
+  %a.iv.plus.4 = getelementptr inbounds i32, ptr %a, i64 %iv.plus.4
+  %l2 = load i32, ptr %a.iv.plus.4, align 4
+  %iv.plus.8 = add nuw nsw i64 %iv, 8
+  %a.iv.plus.8 = getelementptr inbounds i32, ptr %a, i64 %iv.plus.8
+  %l3 = load i32, ptr %a.iv.plus.8, align 4
+  %sum.1 = add i32 %l1, %l2
+  %sum.2 = add i32 %sum.1, %l3
+  %c.iv = getelementptr inbounds i32, ptr %c, i64 %iv
+  store i32 %sum.2, ptr %c.iv, align 4
+  %iv.next = add nuw nsw i64 %iv, 1
+  %cmp = icmp slt i64 %iv.next, %n
+  br i1 %cmp, label %loop, label %exit
+
+exit:
+  ret void
+}
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-scalable.ll b/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-scalable.ll
new file mode 100644
index 0000000000000..2af48b332d9b2
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-scalable.ll
@@ -0,0 +1,158 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
+; Pre-commit test for cross-part load redundancy with scalable vectors, where
+; the distance between two logical parts is a runtime multiple of vscale. Both
+; loops below vectorize with an interleave count of 1 today.
+; RUN: opt -passes=loop-vectorize -mtriple=aarch64-none-linux-gnu -mattr=+sve \
+; RUN:     -force-vector-width="vscale x 2" \
+; RUN:     -force-target-max-vector-interleave=2 -small-loop-cost=0 \
+; RUN:     -S %s | FileCheck %s
+
+target triple = "aarch64-unknown-linux-gnu"
+
+declare i64 @llvm.vscale.i64()
+
+; A constant source offset cannot equal the scalable logical-part offset for
+; every runtime vscale, so the two addresses are not exactly equal.
+define void @constant_offset(ptr noalias %a, ptr noalias %c, i64 %n) {
+; CHECK-LABEL: define void @constant_offset(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) #[[ATTR1:[0-9]+]] {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; CHECK-NEXT:    [[TMP1:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-NEXT:    [[TMP2:%.*]] = shl nuw i64 [[TMP1]], 1
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], [[TMP2]]
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[TMP0]], [[TMP2]]
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[N_MOD_VF]]
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP3:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 2 x i32>, ptr [[TMP3]], align 4
+; CHECK-NEXT:    [[TMP4:%.*]] = add nuw nsw i64 [[INDEX]], 4
+; CHECK-NEXT:    [[TMP5:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP4]]
+; CHECK-NEXT:    [[WIDE_LOAD1:%.*]] = load <vscale x 2 x i32>, ptr [[TMP5]], align 4
+; CHECK-NEXT:    [[TMP6:%.*]] = add <vscale x 2 x i32> [[WIDE_LOAD]], [[WIDE_LOAD1]]
+; CHECK-NEXT:    [[TMP7:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
+; CHECK-NEXT:    store <vscale x 2 x i32> [[TMP6]], ptr [[TMP7]], align 4
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP2]]
+; CHECK-NEXT:    [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; CHECK:       [[SCALAR_PH]]:
+; CHECK-NEXT:    [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[A_IV:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT:    [[L1:%.*]] = load i32, ptr [[A_IV]], align 4
+; CHECK-NEXT:    [[IV_PLUS_4:%.*]] = add nuw nsw i64 [[IV]], 4
+; CHECK-NEXT:    [[A_IV_PLUS_4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV_PLUS_4]]
+; CHECK-NEXT:    [[L2:%.*]] = load i32, ptr [[A_IV_PLUS_4]], align 4
+; CHECK-NEXT:    [[SUM:%.*]] = add i32 [[L1]], [[L2]]
+; CHECK-NEXT:    [[C_IV:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[IV]]
+; CHECK-NEXT:    store i32 [[SUM]], ptr [[C_IV]], align 4
+; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT:    [[CMP:%.*]] = icmp slt i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[CMP]], label %[[LOOP]], label %[[EXIT]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %a.iv = getelementptr inbounds i32, ptr %a, i64 %iv
+  %l1 = load i32, ptr %a.iv, align 4
+  %iv.plus.4 = add nuw nsw i64 %iv, 4
+  %a.iv.plus.4 = getelementptr inbounds i32, ptr %a, i64 %iv.plus.4
+  %l2 = load i32, ptr %a.iv.plus.4, align 4
+  %sum = add i32 %l1, %l2
+  %c.iv = getelementptr inbounds i32, ptr %c, i64 %iv
+  store i32 %sum, ptr %c.iv, align 4
+  %iv.next = add nuw nsw i64 %iv, 1
+  %cmp = icmp slt i64 %iv.next, %n
+  br i1 %cmp, label %loop, label %exit
+
+exit:
+  ret void
+}
+
+; The second load starts exactly one scalable VF after the first, so its part-0
+; address equals the first load's part-1 address for every runtime vscale.
+define void @vscale_offset(ptr noalias %a, ptr noalias %c, i64 %n) {
+; CHECK-LABEL: define void @vscale_offset(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) #[[ATTR1]] {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[VSCALE:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-NEXT:    [[PART_OFFSET:%.*]] = mul nuw i64 [[VSCALE]], 2
+; CHECK-NEXT:    [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], [[PART_OFFSET]]
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP1:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-NEXT:    [[TMP2:%.*]] = shl nuw i64 [[TMP1]], 1
+; CHECK-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[TMP0]], [[TMP2]]
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[N_MOD_VF]]
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP3:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 2 x i32>, ptr [[TMP3]], align 4
+; CHECK-NEXT:    [[TMP4:%.*]] = add i64 [[INDEX]], [[PART_OFFSET]]
+; CHECK-NEXT:    [[TMP5:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP4]]
+; CHECK-NEXT:    [[WIDE_LOAD1:%.*]] = load <vscale x 2 x i32>, ptr [[TMP5]], align 4
+; CHECK-NEXT:    [[TMP6:%.*]] = add <vscale x 2 x i32> [[WIDE_LOAD]], [[WIDE_LOAD1]]
+; CHECK-NEXT:    [[TMP7:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
+; CHECK-NEXT:    store <vscale x 2 x i32> [[TMP6]], ptr [[TMP7]], align 4
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP2]]
+; CHECK-NEXT:    [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; CHECK:       [[SCALAR_PH]]:
+; CHECK-NEXT:    [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[A_IV:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT:    [[L1:%.*]] = load i32, ptr [[A_IV]], align 4
+; CHECK-NEXT:    [[IV_PART:%.*]] = add i64 [[IV]], [[PART_OFFSET]]
+; CHECK-NEXT:    [[A_IV_PART:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV_PART]]
+; CHECK-NEXT:    [[L2:%.*]] = load i32, ptr [[A_IV_PART]], align 4
+; CHECK-NEXT:    [[SUM:%.*]] = add i32 [[L1]], [[L2]]
+; CHECK-NEXT:    [[C_IV:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[IV]]
+; CHECK-NEXT:    store i32 [[SUM]], ptr [[C_IV]], align 4
+; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT:    [[CMP:%.*]] = icmp slt i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[CMP]], label %[[LOOP]], label %[[EXIT]], !llvm.loop [[LOOP5:![0-9]+]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  %vscale = call i64 @llvm.vscale.i64()
+  %part.offset = mul nuw i64 %vscale, 2
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %a.iv = getelementptr inbounds i32, ptr %a, i64 %iv
+  %l1 = load i32, ptr %a.iv, align 4
+  %iv.part = add i64 %iv, %part.offset
+  %a.iv.part = getelementptr inbounds i32, ptr %a, i64 %iv.part
+  %l2 = load i32, ptr %a.iv.part, align 4
+  %sum = add i32 %l1, %l2
+  %c.iv = getelementptr inbounds i32, ptr %c, i64 %iv
+  store i32 %sum, ptr %c.iv, align 4
+  %iv.next = add nuw nsw i64 %iv, 1
+  %cmp = icmp slt i64 %iv.next, %n
+  br i1 %cmp, label %loop, label %exit
+
+exit:
+  ret void
+}
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse.ll b/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse.ll
new file mode 100644
index 0000000000000..f3b3cafd67b37
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse.ll
@@ -0,0 +1,81 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
+; Pre-commit test for cross-part load redundancy. A widened load in one
+; logical part of an interleaved vector loop can cover exactly the elements
+; that another widened load covers in the next part. The interleave-count
+; heuristics do not model that redundancy yet, so this loop vectorizes with an
+; interleave count of 1.
+; RUN: opt -passes=loop-vectorize -force-vector-width=4 \
+; RUN:     -force-target-max-vector-interleave=2 -small-loop-cost=0 \
+; RUN:     -S %s | FileCheck %s
+
+target triple = "aarch64-unknown-linux-gnu"
+
+; The motivating shape for cross-part load redundancy: with VF=4, the widened
+; load of a[i+4] in logical part 0 would cover exactly the same elements as the
+; widened load of a[i] in logical part 1.
+define void @positive(ptr noalias %a, ptr noalias %c, i64 %n) {
+; CHECK-LABEL: define void @positive(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 4
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP1:%.*]] = and i64 [[TMP0]], 3
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4
+; CHECK-NEXT:    [[TMP3:%.*]] = add nuw nsw i64 [[INDEX]], 4
+; CHECK-NEXT:    [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP3]]
+; CHECK-NEXT:    [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP4]], align 4
+; CHECK-NEXT:    [[TMP5:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[WIDE_LOAD1]]
+; CHECK-NEXT:    [[TMP6:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
+; CHECK-NEXT:    store <4 x i32> [[TMP5]], ptr [[TMP6]], align 4
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT:    [[TMP7:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP7]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; CHECK:       [[SCALAR_PH]]:
+; CHECK-NEXT:    [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[A_IV:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT:    [[L1:%.*]] = load i32, ptr [[A_IV]], align 4
+; CHECK-NEXT:    [[IV_PLUS_4:%.*]] = add nuw nsw i64 [[IV]], 4
+; CHECK-NEXT:    [[A_IV_PLUS_4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV_PLUS_4]]
+; CHECK-NEXT:    [[L2:%.*]] = load i32, ptr [[A_IV_PLUS_4]], align 4
+; CHECK-NEXT:    [[SUM:%.*]] = add i32 [[L1]], [[L2]]
+; CHECK-NEXT:    [[C_IV:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[IV]]
+; CHECK-NEXT:    store i32 [[SUM]], ptr [[C_IV]], align 4
+; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT:    [[CMP:%.*]] = icmp slt i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[CMP]], label %[[LOOP]], label %[[EXIT]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %a.iv = getelementptr inbounds i32, ptr %a, i64 %iv
+  %l1 = load i32, ptr %a.iv, align 4
+  %iv.plus.4 = add nuw nsw i64 %iv, 4
+  %a.iv.plus.4 = getelementptr inbounds i32, ptr %a, i64 %iv.plus.4
+  %l2 = load i32, ptr %a.iv.plus.4, align 4
+  %sum = add i32 %l1, %l2
+  %c.iv = getelementptr inbounds i32, ptr %c, i64 %iv
+  store i32 %sum, ptr %c.iv, align 4
+  %iv.next = add nuw nsw i64 %iv, 1
+  %cmp = icmp slt i64 %iv.next, %n
+  br i1 %cmp, label %loop, label %exit
+
+exit:
+  ret void
+}

>From 8af706c315b72a632066a382d191a65054376f1e Mon Sep 17 00:00:00 2001
From: Sergey Shcherbinin <sscherbinin at nvidia.com>
Date: Mon, 21 Sep 2026 17:31:21 +0400
Subject: [PATCH 2/2] [LV] Add cross-part load redundancy heuristic for
 interleaving

Existing interleave heuristics can select IC=1 when load redundancies become visible only between logical VPlan parts.

Add a disabled-by-default, prediction-only analysis for eligible single-block VPlans. Model two logical parts with exact SCEV addresses, invalidate available loads at memory writes, and raise IC from 1 to 2 when the predicted saved load cost reaches the configured threshold. Handle forward and reverse accesses and fail closed for unsupported plan states.

The analysis neither mutates VPlan nor removes loads; existing downstream optimizations may realize the exposed fixed-width redundancies.

RFC: https://discourse.llvm.org/t/rfc-using-cross-part-cse-to-guide-loop-interleaving/91438
---
 llvm/lib/Transforms/Vectorize/CMakeLists.txt  |   1 +
 .../Transforms/Vectorize/LoopVectorize.cpp    |  46 +++
 .../Vectorize/VPlanCrossPartCSE.cpp           | 316 +++++++++++++++
 .../Transforms/Vectorize/VPlanCrossPartCSE.h  |  57 +++
 .../AArch64/cross-part-load-cse-debug.ll      | 263 ++++++++++++
 .../cross-part-load-cse-fixed-cases.ll        | 185 +++++----
 .../AArch64/cross-part-load-cse-narrowed.ll   |  50 +++
 .../cross-part-load-cse-opportunities.ll      |  34 +-
 .../AArch64/cross-part-load-cse-pipeline.ll   |  49 +++
 .../cross-part-load-cse-scalable-reverse.ll   | 188 +++++++++
 .../AArch64/cross-part-load-cse-scalable.ll   |  27 +-
 .../cross-part-load-cse-vplan-multi-block.ll  | 122 ++++++
 .../AArch64/cross-part-load-cse.ll            | 379 +++++++++++++++---
 .../llvm/lib/Transforms/Vectorize/BUILD.gn    |   1 +
 14 files changed, 1579 insertions(+), 139 deletions(-)
 create mode 100644 llvm/lib/Transforms/Vectorize/VPlanCrossPartCSE.cpp
 create mode 100644 llvm/lib/Transforms/Vectorize/VPlanCrossPartCSE.h
 create mode 100644 llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-debug.ll
 create mode 100644 llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-narrowed.ll
 create mode 100644 llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-pipeline.ll
 create mode 100644 llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-scalable-reverse.ll
 create mode 100644 llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-vplan-multi-block.ll

diff --git a/llvm/lib/Transforms/Vectorize/CMakeLists.txt b/llvm/lib/Transforms/Vectorize/CMakeLists.txt
index 9073211280886..f459cbc618557 100644
--- a/llvm/lib/Transforms/Vectorize/CMakeLists.txt
+++ b/llvm/lib/Transforms/Vectorize/CMakeLists.txt
@@ -35,6 +35,7 @@ add_llvm_component_library(LLVMVectorize
   VPlan.cpp
   VPlanAnalysis.cpp
   VPlanConstruction.cpp
+  VPlanCrossPartCSE.cpp
   VPlanDominatorTree.cpp
   VPlanEVLTailFolding.cpp
   VPlanLowering.cpp
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index e16e707066bdc..ebbdbefa69605 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -59,6 +59,7 @@
 #include "VPlan.h"
 #include "VPlanAnalysis.h"
 #include "VPlanCFG.h"
+#include "VPlanCrossPartCSE.h"
 #include "VPlanHelpers.h"
 #include "VPlanPatternMatch.h"
 #include "VPlanTransforms.h"
@@ -300,6 +301,18 @@ static cl::opt<bool> EnableLoadStoreRuntimeInterleave(
     cl::desc(
         "Enable runtime interleaving until load/store ports are saturated"));
 
+/// Enable cross-part load-redundancy analysis during IC selection.
+static cl::opt<bool> EnableInterleaveCSE(
+    "enable-interleave-cse", cl::init(false), cl::Hidden,
+    cl::desc("Raise heuristic IC=1 to IC=2 when exact cross-part load "
+             "redundancy predicts a downstream saving"));
+
+/// Minimum percentage of the modeled UF=2 body predicted to be saved.
+static cl::opt<unsigned> InterleaveCSEMinSavingPct(
+    "interleave-cse-min-pct", cl::init(5), cl::Hidden,
+    cl::desc("Minimum predicted downstream load saving as a percentage of the "
+             "modeled UF=2 vector loop body"));
+
 // TODO: Move size-based thresholds out of legality checking, make cost based
 // decisions instead of hard thresholds.
 static cl::opt<unsigned> VectorizeSCEVCheckThreshold(
@@ -3632,6 +3645,33 @@ std::unique_ptr<VPlan> LoopVectorizationPlanner::selectBestEpiloguePlan(
 unsigned
 LoopVectorizationPlanner::selectInterleaveCount(VPlan &Plan, ElementCount VF,
                                                 InstructionCost LoopCost) {
+  // Evaluate cross-part savings only when the ordinary heuristics are about to
+  // return IC=1. All cost state remains local to this UF-selection call.
+  auto shouldInterleaveForCrossPartCSE = [&](unsigned MaxIC) {
+    if (!EnableInterleaveCSE)
+      return false;
+
+    // Cross-part CSE only augments ordinary heuristic selection. These checks
+    // preserve target, trip-count, and user-hint restrictions. Loops vectorized
+    // with partial-alias masking are excluded explicitly because processLoop
+    // resets their interleave count to 1 after this selection; analyzing them
+    // would only spend cost-model queries and report a discarded decision.
+    if (MaxIC < CrossPartCSERequiredInterleaveCount || !VF.isVector() ||
+        !OrigLoop->isInnermost() || Config.getHints().getInterleave() != 0 ||
+        CM->maskPartialAliasing())
+      return false;
+
+    VPCostContext CostCtx(*TLI, Plan, *CM, Config);
+    CrossPartCSEOptions Options;
+    Options.MinSavingPct = InterleaveCSEMinSavingPct;
+    if (!isCrossPartCSEProfitable(Plan, VF, LoopCost, CostCtx, Options))
+      return false;
+
+    LLVM_DEBUG(dbgs() << "LV: Exact cross-part load redundancy predicts a "
+                         "downstream saving; raising IC to 2.\n");
+    return true;
+  };
+
   // -- The interleave heuristics --
   // We interleave the loop in order to expose ILP and reduce the loop overhead.
   // There are many micro-architectural considerations that we can't predict
@@ -3961,6 +4001,9 @@ LoopVectorizationPlanner::selectInterleaveCount(VPlan &Plan, ElementCount VF,
       return std::max(IC / 2, SmallIC);
     }
 
+    if (SmallIC == 1 && shouldInterleaveForCrossPartCSE(IC))
+      return CrossPartCSERequiredInterleaveCount;
+
     LLVM_DEBUG(dbgs() << "LV: Interleaving to reduce branch cost.\n");
     return SmallIC;
   }
@@ -3972,6 +4015,9 @@ LoopVectorizationPlanner::selectInterleaveCount(VPlan &Plan, ElementCount VF,
     return IC;
   }
 
+  if (shouldInterleaveForCrossPartCSE(IC))
+    return CrossPartCSERequiredInterleaveCount;
+
   LLVM_DEBUG(dbgs() << "LV: Not Interleaving.\n");
   return 1;
 }
diff --git a/llvm/lib/Transforms/Vectorize/VPlanCrossPartCSE.cpp b/llvm/lib/Transforms/Vectorize/VPlanCrossPartCSE.cpp
new file mode 100644
index 0000000000000..2912a2448de37
--- /dev/null
+++ b/llvm/lib/Transforms/Vectorize/VPlanCrossPartCSE.cpp
@@ -0,0 +1,316 @@
+//===- VPlanCrossPartCSE.cpp - Cross-part CSE for VPlan -------------------===//
+//
+// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
+// See https://llvm.org/LICENSE.txt for license information.
+// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
+//
+//===----------------------------------------------------------------------===//
+//
+// This file implements exact load-redundancy profitability analysis across two
+// logical VPlan parts.
+//
+//===----------------------------------------------------------------------===//
+
+#include "VPlanCrossPartCSE.h"
+#include "VPlan.h"
+#include "VPlanHelpers.h"
+#include "VPlanPatternMatch.h"
+#include "VPlanUtils.h"
+#include "llvm/ADT/DenseMap.h"
+#include "llvm/ADT/Hashing.h"
+#include "llvm/Analysis/LoopInfo.h"
+#include "llvm/Analysis/ScalarEvolution.h"
+#include "llvm/Support/Debug.h"
+#include "llvm/Support/raw_ostream.h"
+
+using namespace llvm;
+
+#define DEBUG_TYPE "loop-vectorize"
+
+namespace {
+
+/// Return an unmasked, non-EVL, consecutive widened load.
+static VPWidenLoadRecipe *getCrossPartSupportedLoad(VPRecipeBase &R) {
+  auto *Load = dyn_cast<VPWidenLoadRecipe>(&R);
+  if (!Load || Load->isMasked() || !Load->isConsecutive())
+    return nullptr;
+  return Load;
+}
+
+/// Build addresses only for provenance whose physical UF mapping is explicit.
+class CrossPartAddressBuilder {
+  /// Predicated SCEV state carrying vectorization assumptions.
+  PredicatedScalarEvolution &PSE;
+  /// ScalarEvolution used for canonical exact identities.
+  ScalarEvolution &SE;
+  /// Original loop used to interpret loop-varying VPlan values.
+  const Loop *OrigLoop;
+  /// Vector factor used to model the exact per-part offset.
+  const ElementCount VF;
+  /// Base SCEVs cached by VPlan value for reuse across loads and parts.
+  DenseMap<const VPValue *, const SCEV *> BaseSCEVs;
+
+  /// Return the SCEV represented by \p V, caching it after first construction.
+  const SCEV *getBaseSCEV(const VPValue *V) {
+    auto It = BaseSCEVs.find(V);
+    if (It != BaseSCEVs.end())
+      return It->second;
+
+    const SCEV *S = vputils::getSCEVExprForVPValue(V, PSE, OrigLoop);
+    BaseSCEVs.try_emplace(V, S);
+    return S;
+  }
+
+  /// Return a conservative GEP expression for \p Base + \p Offset.
+  const SCEV *getGEPAddress(const SCEV *Base, const SCEV *Offset,
+                            Type *SourceElementTy) {
+    // ScalarEvolution imports GEP nowrap facts only after accounting for their
+    // poison semantics. AddExpr uniquing still recognizes equal operands
+    // without adding those facts to the synthetic expression.
+    return SE.getGEPExpr(Base, {Offset}, SourceElementTy);
+  }
+
+  /// Return the physical address produced by \p VectorPtr for \p Part.
+  const SCEV *getForwardAddress(VPVectorPointerRecipe &VectorPtr,
+                                unsigned Part) {
+    // VPlanUnroll models a forward part as Base + Part * VF * Stride. Accept
+    // only unit stride until the analysis supports the complete expression.
+    using namespace VPlanPatternMatch;
+    if (!match(VectorPtr.getStride(), m_One()))
+      return SE.getCouldNotCompute();
+
+    const SCEV *Base = getBaseSCEV(VectorPtr.getOperand(0));
+    if (isa<SCEVCouldNotCompute>(Base))
+      return SE.getCouldNotCompute();
+    if (Part == 0)
+      return Base;
+
+    Type *IndexTy = SE.getDataLayout().getIndexType(VectorPtr.getScalarType());
+    const SCEV *Offset = SE.getElementCount(IndexTy, VF * Part);
+    return getGEPAddress(Base, Offset, VectorPtr.getSourceElementType());
+  }
+
+  /// Return the physical address produced by \p EndPtr for \p Part.
+  const SCEV *getReverseAddress(VPVectorEndPointerRecipe &EndPtr,
+                                unsigned Part) {
+    const SCEV *Base = getBaseSCEV(EndPtr.getPointer());
+    if (isa<SCEVCouldNotCompute>(Base))
+      return SE.getCouldNotCompute();
+
+    Type *IndexTy = SE.getDataLayout().getIndexType(EndPtr.getScalarType());
+    const SCEV *VFExpr = SE.getElementCount(IndexTy, VF);
+    const SCEV *Stride =
+        SE.getConstant(IndexTy, EndPtr.getStride(), /*isSigned=*/true);
+
+    // Mirror VPVectorEndPointerRecipe::materializeOffset:
+    //   Stride * (VF - 1) + Part * Stride * VF.
+    const SCEV *Offset0 =
+        SE.getMulExpr(SE.getMinusSCEV(VFExpr, SE.getOne(IndexTy)), Stride);
+    int64_t PartStride = static_cast<int64_t>(Part) * EndPtr.getStride();
+    const SCEV *PartOffset = SE.getMulExpr(
+        SE.getConstant(IndexTy, PartStride, /*isSigned=*/true), VFExpr);
+    const SCEV *Offset = SE.getAddExpr(Offset0, PartOffset);
+    return getGEPAddress(Base, Offset, EndPtr.getSourceElementType());
+  }
+
+public:
+  /// Bind the VF, original loop, and predicated SCEV state.
+  CrossPartAddressBuilder(ElementCount VF, PredicatedScalarEvolution &PSE,
+                          const Loop *OrigLoop)
+      : PSE(PSE), SE(*PSE.getSE()), OrigLoop(OrigLoop), VF(VF) {}
+
+  /// Return the exact address used by \p Load in logical part \p Part.
+  const SCEV *getAddress(VPWidenLoadRecipe &Load, unsigned Part) {
+    assert(Part < CrossPartCSERequiredInterleaveCount &&
+           "logical part must be zero or one");
+    VPValue *Addr = Load.getAddr();
+
+    // Reproduce only the recipe-specific physical rewrites performed by
+    // VPlanUnroll for consecutive forward and reverse accesses.
+    auto *VectorPtr = dyn_cast<VPVectorPointerRecipe>(Addr);
+    if (VectorPtr)
+      return getForwardAddress(*VectorPtr, Part);
+    auto *EndPtr = dyn_cast<VPVectorEndPointerRecipe>(Addr);
+    if (EndPtr)
+      return getReverseAddress(*EndPtr, Part);
+
+    return SE.getCouldNotCompute();
+  }
+};
+
+/// Key for exact value equality of two logical widened-load instances.
+/// Metadata and alignment are not part of value identity. A CSE implementation
+/// must intersect retained metadata and preserve an alignment sufficient for
+/// every replaced load.
+/// Poison-generating source metadata is not propagated to widened loads.
+struct CrossPartLoadKey {
+  /// Canonical SCEV address for this logical load instance.
+  const SCEV *Address;
+  /// Loaded scalar type required for value compatibility.
+  Type *ValueType;
+};
+
+/// DenseMap policy for exact canonical load keys.
+struct CrossPartLoadKeyInfo {
+  /// Hash every property required by exact load equality.
+  static unsigned getHashValue(const CrossPartLoadKey &Key) {
+    return hash_combine(Key.Address, Key.ValueType);
+  }
+
+  /// Compare every property required by exact load equality.
+  static bool isEqual(const CrossPartLoadKey &A, const CrossPartLoadKey &B) {
+    return A.Address == B.Address && A.ValueType == B.ValueType;
+  }
+};
+
+/// Return whether \p R may write memory during VPlan execution.
+static bool isCrossPartWrite(const VPRecipeBase &R) {
+  // VPVectorEndPointerRecipe is pure but inherits the conservative memory
+  // default. This local exception prevents its address computation from being
+  // mistaken for a write without changing global recipe memory behavior.
+  // TODO: Classify VPVectorEndPointerRecipe as non-memory in
+  // VPRecipeBase::mayReadFromMemory() and mayWriteToMemory(), then remove this
+  // exception. The shared fix can expose new VPlan CSE opportunities and needs
+  // dedicated code-generation tests.
+  switch (R.getVPRecipeID()) {
+  case VPRecipeBase::VPVectorEndPointerSC:
+    return false;
+  default:
+    return R.mayWriteToMemory();
+  }
+}
+
+/// Return whether \p Plan keeps the canonical IV increment in the symbolic
+/// VF * UF form required to model consecutive logical parts.
+static bool hasCanonicalIVIncrementForCrossPartCSE(VPlan &Plan) {
+  return vputils::findCanonicalIVIncrement(Plan);
+}
+
+} // namespace
+
+bool llvm::isCrossPartCSEProfitable(VPlan &Plan, ElementCount VF,
+                                    InstructionCost LoopCost,
+                                    VPCostContext &CostCtx,
+                                    const CrossPartCSEOptions &Options) {
+  assert(VF.isVector() && "cross-part analysis requires a vector VF");
+  assert(CostCtx.L && CostCtx.L->isInnermost() &&
+         "cross-part analysis requires an innermost loop");
+  assert(Plan.hasUF(CrossPartCSERequiredInterleaveCount) &&
+         "cross-part analysis requires support for UF=2");
+  assert(!Plan.isUnrolled() && "cross-part analysis requires symbolic UF");
+
+  // Narrowed plans replace symbolic VF * UF with a different effective step,
+  // so the selected VF no longer describes their physical per-part offset.
+  if (Plan.getVFxUF().isMaterialized())
+    return false;
+
+  // Reject an unspecified or impossible percentage before cost arithmetic.
+  if (Options.MinSavingPct == CrossPartCSEOptions::Unspecified ||
+      Options.MinSavingPct > 100)
+    return false;
+
+  VPRegionBlock *LoopRegion = Plan.getVectorLoopRegion();
+  if (!LoopRegion)
+    return false;
+
+  // Fail closed for every shape outside the exact single-block UF=2 model.
+  // TODO: Expand coverage by accepting additional plan shapes once their
+  // cross-part semantics can be modeled exactly.
+  if (!LoopCost.isValid() || LoopCost <= 0 ||
+      LoopRegion->getEntryBasicBlock() != LoopRegion->getExitingBasicBlock() ||
+      !hasCanonicalIVIncrementForCrossPartCSE(Plan))
+    return false;
+
+  using AvailableLoadMap =
+      DenseMap<CrossPartLoadKey, unsigned, CrossPartLoadKeyInfo>;
+  AvailableLoadMap AvailableLoadParts;
+  // A recipe may participate in multiple logical matches as supported shapes
+  // expand, but its local saving estimate is computed at most once.
+  DenseMap<const VPRecipeBase *, InstructionCost> SavingCosts;
+  CrossPartAddressBuilder Addresses(VF, CostCtx.PSE, CostCtx.L);
+#ifndef NDEBUG
+  // Count redundant-load opportunities only for diagnostics; profitability
+  // uses SavedCost.
+  unsigned NumOpportunities = 0;
+#endif
+  InstructionCost SavedCost = 0;
+
+  // Match VPlanUnroll's recipe-major UF=2 order. Clearing on every write
+  // enforces a strict no-write interval without alias disambiguation.
+  for (VPRecipeBase &R : *LoopRegion->getEntryBasicBlock()) {
+    if (isCrossPartWrite(R)) {
+      AvailableLoadParts.clear();
+      continue;
+    }
+
+    VPWidenLoadRecipe *Load = getCrossPartSupportedLoad(R);
+    if (!Load)
+      continue;
+
+    for (unsigned Part = 0; Part != CrossPartCSERequiredInterleaveCount;
+         ++Part) {
+      const SCEV *Address = Addresses.getAddress(*Load, Part);
+      if (isa<SCEVCouldNotCompute>(Address))
+        continue;
+
+      CrossPartLoadKey Key = {Address, Load->getScalarType()};
+      // Only reuse between different logical parts can justify raising IC from
+      // 1 to 2. A duplicate already seen in the same part also exists at IC=1
+      // and therefore provides no interleaving-specific saving.
+      unsigned PartBit = 1U << Part;
+      unsigned &AvailableParts = AvailableLoadParts[Key];
+      if (AvailableParts & PartBit)
+        continue;
+
+      bool HasOppositePart = (AvailableParts & ~PartBit) != 0;
+      AvailableParts |= PartBit;
+      if (!HasOppositePart) {
+        // Record the first occurrence in this part without assigning
+        // cross-part credit.
+        continue;
+      }
+
+      auto CostIt = SavingCosts.find(Load);
+      if (CostIt == SavingCosts.end())
+        CostIt = SavingCosts.try_emplace(Load, Load->cost(VF, CostCtx)).first;
+
+      // This estimate intentionally avoids retaining cost state from VF
+      // selection. It is exact for the directly costed widened loads supported
+      // here, but does not reproduce legacy attribution included in LoopCost.
+      // TODO: If measured profitability loses accuracy as supported recipes
+      // expand, consider passing cached costs from the selected-VF cost run.
+      // That would restore exact attribution at the cost of cross-phase state
+      // and recipe-lifetime management.
+      if (!CostIt->second.isValid() || CostIt->second <= 0)
+        continue;
+      SavedCost += CostIt->second;
+      LLVM_DEBUG(++NumOpportunities);
+    }
+  }
+
+  using CostType = InstructionCost::CostType;
+  bool Select = false;
+  if (SavedCost > 0) {
+    // Use InstructionCost arithmetic to preserve fractional cost units.
+    InstructionCost ScaledSavedCost = SavedCost * CostType(100);
+    InstructionCost RequiredCost =
+        LoopCost * CostType(CrossPartCSERequiredInterleaveCount);
+    RequiredCost *= CostType(Options.MinSavingPct);
+    Select = ScaledSavedCost >= RequiredCost;
+  }
+
+  LLVM_DEBUG({
+    CostType SavingPct = 0;
+    if (SavedCost.isValid() && SavedCost > 0 && LoopCost.isValid() &&
+        LoopCost > 0)
+      SavingPct = ((SavedCost * CostType(100)) /
+                   (LoopCost * CostType(CrossPartCSERequiredInterleaveCount)))
+                      .getValue();
+    dbgs() << "LV: Cross-part load redundancy estimate: opportunities="
+           << NumOpportunities << ", predicted-saved-cost=" << SavedCost
+           << ", loop-cost=" << LoopCost << ", saving=" << SavingPct
+           << "%, required=" << Options.MinSavingPct << "%; "
+           << (Select ? "selecting IC=2" : "skipping") << ".\n";
+  });
+  return Select;
+}
diff --git a/llvm/lib/Transforms/Vectorize/VPlanCrossPartCSE.h b/llvm/lib/Transforms/Vectorize/VPlanCrossPartCSE.h
new file mode 100644
index 0000000000000..391d32655b5c4
--- /dev/null
+++ b/llvm/lib/Transforms/Vectorize/VPlanCrossPartCSE.h
@@ -0,0 +1,57 @@
+//===- VPlanCrossPartCSE.h - Cross-part CSE for VPlan -----------*- C++ -*-===//
+//
+// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
+// See https://llvm.org/LICENSE.txt for license information.
+// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
+//
+//===----------------------------------------------------------------------===//
+//
+// This file declares prediction-only profitability analysis for exact load
+// redundancy across two modeled logical VPlan parts. It does not transform
+// VPlan or guarantee that a later pass will eliminate the redundant load.
+//
+//===----------------------------------------------------------------------===//
+
+#ifndef LLVM_TRANSFORMS_VECTORIZE_VPLANCROSSPARTCSE_H
+#define LLVM_TRANSFORMS_VECTORIZE_VPLANCROSSPARTCSE_H
+
+#include "llvm/Support/InstructionCost.h"
+#include "llvm/Support/TypeSize.h"
+#include <limits>
+
+namespace llvm {
+
+class VPlan;
+struct VPCostContext;
+
+/// The interleave count and logical unroll factor modeled by the analysis.
+constexpr unsigned CrossPartCSERequiredInterleaveCount = 2;
+
+/// Profitability criterion supplied by the caller.
+///
+/// The fail-closed default requires the caller to provide an explicit value.
+struct CrossPartCSEOptions {
+  /// Sentinel used until the caller supplies an explicit policy value.
+  static constexpr unsigned Unspecified = std::numeric_limits<unsigned>::max();
+
+  /// Minimum saving; the default rejects analysis until policy supplies it.
+  unsigned MinSavingPct = Unspecified;
+};
+
+/// Return whether exact cross-part load redundancy in \p Plan at \p VF meets
+/// \p Options.
+///
+/// The caller must establish that interleaving \p Plan is legal before using
+/// this opportunity estimate to raise its interleave count. A positive result
+/// does not guarantee that a later pass will eliminate the redundant load.
+/// The analysis reads \p Plan but takes a non-const reference because the VPlan
+/// query APIs it uses are not const-qualified.
+///
+/// \p CostCtx is local to the interleave decision and is not retained.
+bool isCrossPartCSEProfitable(VPlan &Plan, ElementCount VF,
+                              InstructionCost LoopCost, VPCostContext &CostCtx,
+                              const CrossPartCSEOptions &Options);
+
+} // namespace llvm
+
+#endif // LLVM_TRANSFORMS_VECTORIZE_VPLANCROSSPARTCSE_H
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-debug.ll b/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-debug.ll
new file mode 100644
index 0000000000000..a8cf27509988f
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-debug.ll
@@ -0,0 +1,263 @@
+; REQUIRES: asserts
+; RUN: split-file %s %t
+;
+; When cross-part analysis selects IC=2 after the ordinary heuristics decline
+; interleaving, it must not emit a contradictory non-interleaving diagnostic.
+; RUN: opt -passes=loop-vectorize -force-vector-width=4 \
+; RUN:     -force-target-max-vector-interleave=2 -small-loop-cost=0 \
+; RUN:     -enable-interleave-cse \
+; RUN:     -interleave-cse-min-pct=1 \
+; RUN:     -debug-only=loop-vectorize -disable-output %t/success.ll 2>&1 \
+; RUN:     | FileCheck %t/success.ll --check-prefix=SUCCESS
+;
+; A fixed-VF tail-folded plan is ineligible for interleaving. Cross-part
+; analysis must not override that policy or emit a profitability estimate.
+; The same holds when partial-alias masking additionally forces IC=1, which
+; requires runtime difference checks and therefore a second checked pointer.
+; RUN: opt -passes=loop-vectorize -force-vector-width=4 \
+; RUN:     -force-target-max-vector-interleave=2 -small-loop-cost=0 \
+; RUN:     -force-target-supports-masked-memory-ops \
+; RUN:     -force-tail-folding-style=data-and-control \
+; RUN:     -tail-folding-policy=must-fold-tail \
+; RUN:     -force-partial-aliasing-vectorization \
+; RUN:     -enable-interleave-cse \
+; RUN:     -interleave-cse-min-pct=1 -debug-only=loop-vectorize \
+; RUN:     -disable-output %t/success.ll 2>&1 \
+; RUN:     | FileCheck %t/success.ll --check-prefixes=MASKED,ALIAS
+;
+; When the ordinary branch-cost heuristic recommends IC=1, a successful
+; cross-part selection must return before emitting its baseline diagnostic.
+; RUN: opt -passes=loop-vectorize -force-vector-width=4 \
+; RUN:     -force-target-max-vector-interleave=2 \
+; RUN:     -force-target-instruction-cost=1 -small-loop-cost=12 \
+; RUN:     -enable-loadstore-runtime-interleave=false \
+; RUN:     -enable-interleave-cse \
+; RUN:     -interleave-cse-min-pct=1 \
+; RUN:     -debug-only=loop-vectorize -disable-output %t/success.ll 2>&1 \
+; RUN:     | FileCheck %t/success.ll --check-prefix=SUCCESS-SMALL
+;
+; The same-part duplicate after a genuine cross-part match must report exactly
+; one opportunity and must not double the saving past the 6% threshold.
+; RUN: opt -passes=loop-vectorize -force-vector-width=4 \
+; RUN:     -force-target-max-vector-interleave=2 -small-loop-cost=0 \
+; RUN:     -force-target-instruction-cost=1 \
+; RUN:     -enable-interleave-cse -interleave-cse-min-pct=6 \
+; RUN:     -debug-only=loop-vectorize \
+; RUN:     -disable-output %t/duplicate.ll 2>&1 \
+; RUN:     | FileCheck %t/duplicate.ll
+;
+; A predicted saving exactly equal to the configured threshold is sufficient,
+; preserving the inclusive >= comparison.
+; RUN: opt -passes=loop-vectorize -force-vector-width=4 \
+; RUN:     -force-target-max-vector-interleave=2 -small-loop-cost=0 \
+; RUN:     -force-target-instruction-cost=1 \
+; RUN:     -enable-interleave-cse -interleave-cse-min-pct=5 \
+; RUN:     -debug-only=loop-vectorize \
+; RUN:     -disable-output %t/threshold.ll 2>&1 \
+; RUN:     | FileCheck %t/threshold.ll
+;
+; With cross-part analysis disabled, the ordinary branch-cost diagnostic
+; remains unchanged.
+; RUN: opt -passes=loop-vectorize -force-vector-width=4 \
+; RUN:     -force-target-max-vector-interleave=2 \
+; RUN:     -force-target-instruction-cost=1 -small-loop-cost=12 \
+; RUN:     -enable-loadstore-runtime-interleave=false \
+; RUN:     -debug-only=loop-vectorize -disable-output %t/success.ll 2>&1 \
+; RUN:     | FileCheck %t/success.ll --check-prefix=DISABLED-SMALL
+;
+; A constant source offset cannot equal the scalable part offset for every
+; runtime vscale, so the exact analysis reports no opportunity and keeps UF=1.
+; RUN: opt -passes=loop-vectorize -mtriple=aarch64-none-linux-gnu -mattr=+sve \
+; RUN:     -force-vector-width="vscale x 2" \
+; RUN:     -force-target-max-vector-interleave=2 -small-loop-cost=0 \
+; RUN:     -enable-interleave-cse \
+; RUN:     -interleave-cse-min-pct=1 -debug-only=loop-vectorize \
+; RUN:     -disable-output %t/success.ll 2>&1 \
+; RUN:     | FileCheck %t/success.ll --check-prefix=SCALABLE
+;
+; A narrowed plan materializes VFxUF before IC selection. The redundancy
+; analysis must return before emitting an estimate for that unsupported shape.
+; RUN: opt -passes=loop-vectorize -mtriple=arm64-apple-macosx \
+; RUN:     -force-vector-width=2 -force-target-max-vector-interleave=2 \
+; RUN:     -force-target-num-vector-regs=1024 \
+; RUN:     -force-target-instruction-cost=1 -small-loop-cost=0 \
+; RUN:     -enable-interleave-cse -interleave-cse-min-pct=1 \
+; RUN:     -debug-only=loop-vectorize -disable-output \
+; RUN:     %S/cross-part-load-cse-narrowed.ll 2>&1 \
+; RUN:     | FileCheck %s --check-prefix=NARROWED
+;
+; NARROWED-LABEL: LV: Checking a loop in 'narrowed'
+; NARROWED-NOT: LV: Cross-part load redundancy estimate:
+; NARROWED: Executing best plan with VF=2, UF=1
+
+;--- success.ll
+; Every prefix below closes its 'positive' block with a second -LABEL line, so
+; that the checks cannot be satisfied by the 'partial_alias' log that follows.
+;
+; SUCCESS-LABEL: LV: Checking a loop in 'positive'
+; SUCCESS: LV: Cross-part load redundancy estimate: opportunities=1, predicted-saved-cost={{[^,]+}}, loop-cost={{[^,]+}}, saving={{[0-9]+}}%, required=1%; selecting IC=2.
+; SUCCESS-NEXT: LV: Exact cross-part load redundancy predicts a downstream saving; raising IC to 2.
+; SUCCESS-NOT: LV: Not Interleaving.
+; SUCCESS: LV: Found a vectorizable loop
+; SUCCESS: Executing best plan with VF=4, UF=2
+; SUCCESS-LABEL: LV: Checking a loop in 'partial_alias'
+;
+; MASKED-LABEL: LV: Checking a loop in 'positive'
+; MASKED-NOT: LV: Cross-part load redundancy estimate:
+; MASKED-NOT: Exact cross-part load redundancy predicts a downstream saving
+; MASKED-NOT: LV: Not interleaving due to partial aliasing vectorization.
+; MASKED: Executing best plan with VF=4, UF=1
+;
+; The ALIAS-LABEL line below also closes the MASKED block above, because
+; FileCheck partitions the input at the -LABEL lines of every active prefix.
+; ALIAS-LABEL: LV: Checking a loop in 'partial_alias'
+; ALIAS-NOT: LV: Cross-part load redundancy estimate:
+; ALIAS-NOT: Exact cross-part load redundancy predicts a downstream saving
+; ALIAS: LV: Not interleaving due to partial aliasing vectorization.
+; ALIAS: Executing best plan with VF=4, UF=1
+;
+; SUCCESS-SMALL-LABEL: LV: Checking a loop in 'positive'
+; SUCCESS-SMALL: LV: Cross-part load redundancy estimate: opportunities=1, predicted-saved-cost={{[^,]+}}, loop-cost={{[^,]+}}, saving={{[0-9]+}}%, required=1%; selecting IC=2.
+; SUCCESS-SMALL-NEXT: LV: Exact cross-part load redundancy predicts a downstream saving; raising IC to 2.
+; SUCCESS-SMALL-NOT: LV: Interleaving to reduce branch cost.
+; SUCCESS-SMALL: LV: Found a vectorizable loop
+; SUCCESS-SMALL: Executing best plan with VF=4, UF=2
+; SUCCESS-SMALL-LABEL: LV: Checking a loop in 'partial_alias'
+;
+; DISABLED-SMALL-LABEL: LV: Checking a loop in 'positive'
+; DISABLED-SMALL-NOT: Cross-part load redundancy
+; DISABLED-SMALL: LV: Interleaving to reduce branch cost.
+; DISABLED-SMALL-NOT: Cross-part load redundancy
+; DISABLED-SMALL: LV: Found a vectorizable loop
+; DISABLED-SMALL: Executing best plan with VF=4, UF=1
+; DISABLED-SMALL-LABEL: LV: Checking a loop in 'partial_alias'
+;
+; SCALABLE-LABEL: LV: Checking a loop in 'positive'
+; SCALABLE-NOT: Exact cross-part load redundancy predicts a downstream saving
+; SCALABLE: LV: VF is vscale x 2
+; SCALABLE-NEXT: LV: Cross-part load redundancy estimate: opportunities=0, predicted-saved-cost=0, loop-cost={{[^,]+}}, saving=0%, required=1%; skipping.
+; SCALABLE-NEXT: LV: Not Interleaving.
+; SCALABLE-NOT: Exact cross-part load redundancy predicts a downstream saving
+; SCALABLE: LV: Found a vectorizable loop (vscale x 2)
+; SCALABLE: Executing best plan with VF=vscale x 2, UF=1
+; SCALABLE-LABEL: LV: Checking a loop in 'partial_alias'
+
+target triple = "aarch64-unknown-linux-gnu"
+
+define void @positive(ptr noalias %a, ptr noalias %c, i64 %n) {
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %a.iv = getelementptr inbounds i32, ptr %a, i64 %iv
+  %l1 = load i32, ptr %a.iv, align 4
+  %iv.plus.4 = add nuw nsw i64 %iv, 4
+  %a.iv.plus.4 = getelementptr inbounds i32, ptr %a, i64 %iv.plus.4
+  %l2 = load i32, ptr %a.iv.plus.4, align 4
+  %sum = add i32 %l1, %l2
+  %c.iv = getelementptr inbounds i32, ptr %c, i64 %iv
+  store i32 %sum, ptr %c.iv, align 4
+  %iv.next = add nuw nsw i64 %iv, 1
+  %cmp = icmp slt i64 %iv.next, %n
+  br i1 %cmp, label %loop, label %exit
+
+exit:
+  ret void
+}
+
+; The %b/%c pair needs a runtime difference check, which enables partial-alias
+; masking, while the cross-part reuse candidate on %a stays present.
+define void @partial_alias(ptr noalias %a, ptr %b, ptr %c, i64 %n) {
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %a.iv = getelementptr inbounds i32, ptr %a, i64 %iv
+  %l1 = load i32, ptr %a.iv, align 4
+  %iv.plus.4 = add nuw nsw i64 %iv, 4
+  %a.iv.plus.4 = getelementptr inbounds i32, ptr %a, i64 %iv.plus.4
+  %l2 = load i32, ptr %a.iv.plus.4, align 4
+  %b.iv = getelementptr inbounds i32, ptr %b, i64 %iv
+  %l3 = load i32, ptr %b.iv, align 4
+  %sum.1 = add i32 %l1, %l2
+  %sum.2 = add i32 %sum.1, %l3
+  %c.iv = getelementptr inbounds i32, ptr %c, i64 %iv
+  store i32 %sum.2, ptr %c.iv, align 4
+  %iv.next = add nuw nsw i64 %iv, 1
+  %cmp = icmp slt i64 %iv.next, %n
+  br i1 %cmp, label %loop, label %exit
+
+exit:
+  ret void
+}
+
+;--- threshold.ll
+; CHECK-LABEL: LV: Checking a loop in 'threshold_equal'
+; CHECK: LV: Cross-part load redundancy estimate: opportunities=1, predicted-saved-cost=1, loop-cost=10, saving=5%, required=5%; selecting IC=2.
+; CHECK-NEXT: LV: Exact cross-part load redundancy predicts a downstream saving; raising IC to 2.
+; CHECK: Executing best plan with VF=4, UF=2
+
+target triple = "aarch64-unknown-linux-gnu"
+
+; With -force-target-instruction-cost=1 the modeled vector body costs exactly
+; 10: scalar steps, three address computations, two widened loads, one add, one
+; widened store, the backedge, and the canonical IV increment. The second
+; address is derived from a loop-invariant %a + 4 so that no in-loop index
+; arithmetic is costed. The single redundant load saves exactly 1, so the
+; comparison is 1 * 100 == 10 * 2 * 5, i.e. exact equality with the 5%
+; threshold.
+define void @threshold_equal(ptr noalias %a, ptr noalias %c, i64 %n) {
+entry:
+  %a.plus.4 = getelementptr inbounds i32, ptr %a, i64 4
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %a.iv = getelementptr inbounds i32, ptr %a, i64 %iv
+  %l1 = load i32, ptr %a.iv, align 4
+  %a.iv.plus.4 = getelementptr inbounds i32, ptr %a.plus.4, i64 %iv
+  %l2 = load i32, ptr %a.iv.plus.4, align 4
+  %sum = add i32 %l1, %l2
+  %c.iv = getelementptr inbounds i32, ptr %c, i64 %iv
+  store i32 %sum, ptr %c.iv, align 4
+  %iv.next = add nuw nsw i64 %iv, 1
+  %cmp = icmp slt i64 %iv.next, %n
+  br i1 %cmp, label %loop, label %exit
+
+exit:
+  ret void
+}
+
+;--- duplicate.ll
+; CHECK-LABEL: LV: Checking a loop in 'duplicate_after_cross_part'
+; CHECK: LV: Cross-part load redundancy estimate: opportunities=1, predicted-saved-cost={{[^,]+}}, loop-cost={{[^,]+}}, saving={{[0-9]+}}%, required=6%; skipping.
+; CHECK-NOT: Exact cross-part load redundancy predicts a downstream saving
+; CHECK: LV: Found a vectorizable loop
+
+target triple = "aarch64-unknown-linux-gnu"
+
+define void @duplicate_after_cross_part(ptr noalias %a, ptr noalias %c, i64 %n) {
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %a.iv = getelementptr inbounds i32, ptr %a, i64 %iv
+  %l1 = load i32, ptr %a.iv, align 4
+  %iv.plus.4 = add nuw nsw i64 %iv, 4
+  %a.iv.plus.4 = getelementptr inbounds i32, ptr %a, i64 %iv.plus.4
+  %l2 = load i32, ptr %a.iv.plus.4, align 4
+  %l3 = load i32, ptr %a.iv.plus.4, align 4
+  %sum.1 = add i32 %l1, %l2
+  %sum.2 = add i32 %sum.1, %l3
+  %c.iv = getelementptr inbounds i32, ptr %c, i64 %iv
+  store i32 %sum.2, ptr %c.iv, align 4
+  %iv.next = add nuw nsw i64 %iv, 1
+  %cmp = icmp slt i64 %iv.next, %n
+  br i1 %cmp, label %loop, label %exit
+
+exit:
+  ret void
+}
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-fixed-cases.ll b/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-fixed-cases.ll
index 15948347005af..1de6bbd58dde2 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-fixed-cases.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-fixed-cases.ll
@@ -1,11 +1,9 @@
 ; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
-; Pre-commit test for the properties that a cross-part load redundancy
-; analysis has to respect: loaded value type, exact address equality,
-; intervening memory writes, access stride, address provenance,
-; poison-generating metadata, multiple scalar loop blocks and reverse
-; accesses. Every loop below vectorizes with an interleave count of 1 today.
+; The minimum saving is lowered to 1% so that every interleave count of 1 below
+; is caused by the modeled property under test rather than by the threshold.
 ; RUN: opt -passes=loop-vectorize -force-vector-width=4 \
 ; RUN:     -force-target-max-vector-interleave=2 -small-loop-cost=0 \
+; RUN:     -enable-interleave-cse -interleave-cse-min-pct=1 \
 ; RUN:     -S %s | FileCheck %s
 
 target triple = "aarch64-unknown-linux-gnu"
@@ -14,7 +12,8 @@ declare i32 @write_memory(i32) #0
 declare <4 x i32> @write_memory_v4(<4 x i32>)
 
 ; The part-shifted addresses are equal, but the two loads have different value
-; types and therefore cannot share a loaded result.
+; types and therefore cannot share a loaded result. The value type is part of
+; the redundancy key, so the interleave count stays 1.
 define void @different_types(ptr noalias %a, ptr noalias %c, i64 %n) {
 ; CHECK-LABEL: define void @different_types(
 ; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) {
@@ -86,7 +85,7 @@ exit:
 }
 
 ; a[i] + a[i+3]: the source offset 3 is not a multiple of VF=4, so no
-; part-shifted address ever matches exactly.
+; part-shifted address ever matches exactly and the interleave count stays 1.
 define void @inequality(ptr noalias %a, ptr noalias %c, i64 %n) {
 ; CHECK-LABEL: define void @inequality(
 ; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) {
@@ -155,48 +154,49 @@ exit:
 }
 
 ; A may-alias store occurs between the two matching logical-part loads, so their
-; values cannot be assumed to survive from one part to the other.
+; values cannot be assumed to survive from one part to the other. No redundancy
+; opportunity is credited, and the interleave count stays 1.
 define void @write_between(ptr %a, ptr %b, ptr noalias %c, i64 %n) {
 ; CHECK-LABEL: define void @write_between(
 ; CHECK-SAME: ptr [[A:%.*]], ptr [[B:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) {
 ; CHECK-NEXT:  [[ENTRY:.*]]:
 ; CHECK-NEXT:    [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
 ; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 4
-; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_MEMCHECK:.*]]
-; CHECK:       [[VECTOR_MEMCHECK]]:
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
 ; CHECK-NEXT:    [[SMAX:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
-; CHECK-NEXT:    [[TMP1:%.*]] = shl i64 [[SMAX]], 2
-; CHECK-NEXT:    [[SCEVGEP:%.*]] = getelementptr i8, ptr [[B]], i64 [[TMP1]]
-; CHECK-NEXT:    [[TMP2:%.*]] = add i64 [[TMP1]], 16
-; CHECK-NEXT:    [[SCEVGEP1:%.*]] = getelementptr i8, ptr [[A]], i64 [[TMP2]]
+; CHECK-NEXT:    [[TMP9:%.*]] = shl i64 [[SMAX]], 2
+; CHECK-NEXT:    [[SCEVGEP:%.*]] = getelementptr i8, ptr [[B]], i64 [[TMP9]]
+; CHECK-NEXT:    [[TMP10:%.*]] = add i64 [[TMP9]], 16
+; CHECK-NEXT:    [[SCEVGEP1:%.*]] = getelementptr i8, ptr [[A]], i64 [[TMP10]]
 ; CHECK-NEXT:    [[BOUND0:%.*]] = icmp ult ptr [[B]], [[SCEVGEP1]]
 ; CHECK-NEXT:    [[BOUND1:%.*]] = icmp ult ptr [[A]], [[SCEVGEP]]
 ; CHECK-NEXT:    [[FOUND_CONFLICT:%.*]] = and i1 [[BOUND0]], [[BOUND1]]
-; CHECK-NEXT:    br i1 [[FOUND_CONFLICT]], label %[[SCALAR_PH]], label %[[VECTOR_PH:.*]]
-; CHECK:       [[VECTOR_PH]]:
-; CHECK-NEXT:    [[TMP3:%.*]] = and i64 [[TMP0]], 3
-; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP3]]
+; CHECK-NEXT:    br i1 [[FOUND_CONFLICT]], label %[[SCALAR_PH]], label %[[VECTOR_PH1:.*]]
+; CHECK:       [[VECTOR_PH1]]:
+; CHECK-NEXT:    [[TMP1:%.*]] = and i64 [[TMP0]], 3
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
 ; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
 ; CHECK:       [[VECTOR_BODY]]:
-; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT:    [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
-; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP4]], align 4, !alias.scope [[META6:![0-9]+]]
-; CHECK-NEXT:    [[TMP5:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 [[INDEX]]
-; CHECK-NEXT:    store <4 x i32> [[WIDE_LOAD]], ptr [[TMP5]], align 4, !alias.scope [[META9:![0-9]+]], !noalias [[META6]]
-; CHECK-NEXT:    [[TMP6:%.*]] = add nuw nsw i64 [[INDEX]], 4
-; CHECK-NEXT:    [[TMP7:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP6]]
-; CHECK-NEXT:    [[WIDE_LOAD2:%.*]] = load <4 x i32>, ptr [[TMP7]], align 4, !alias.scope [[META6]]
-; CHECK-NEXT:    [[TMP8:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[WIDE_LOAD2]]
-; CHECK-NEXT:    [[TMP9:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
-; CHECK-NEXT:    store <4 x i32> [[TMP8]], ptr [[TMP9]], align 4
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH1]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4, !alias.scope [[META6:![0-9]+]]
+; CHECK-NEXT:    [[TMP3:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 [[INDEX]]
+; CHECK-NEXT:    store <4 x i32> [[WIDE_LOAD]], ptr [[TMP3]], align 4, !alias.scope [[META9:![0-9]+]], !noalias [[META6]]
+; CHECK-NEXT:    [[TMP4:%.*]] = add nuw nsw i64 [[INDEX]], 4
+; CHECK-NEXT:    [[TMP5:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP4]]
+; CHECK-NEXT:    [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP5]], align 4, !alias.scope [[META6]]
+; CHECK-NEXT:    [[TMP6:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[WIDE_LOAD1]]
+; CHECK-NEXT:    [[TMP7:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
+; CHECK-NEXT:    store <4 x i32> [[TMP6]], ptr [[TMP7]], align 4
 ; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
-; CHECK-NEXT:    [[TMP10:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-NEXT:    br i1 [[TMP10]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP11:![0-9]+]]
+; CHECK-NEXT:    [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP11:![0-9]+]]
 ; CHECK:       [[MIDDLE_BLOCK]]:
 ; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
 ; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
 ; CHECK:       [[SCALAR_PH]]:
-; CHECK-NEXT:    [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ], [ 0, %[[VECTOR_MEMCHECK]] ]
+; CHECK-NEXT:    [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ], [ 0, %[[VECTOR_PH]] ]
 ; CHECK-NEXT:    br label %[[LOOP:.*]]
 ; CHECK:       [[LOOP]]:
 ; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
@@ -240,8 +240,8 @@ exit:
 }
 
 ; A vector-mapped call with unknown memory effects occurs between the matching
-; loads. It may write through memory that its arguments do not describe, so no
-; loaded value may be assumed to survive across it.
+; loads. It may write through memory not represented by its arguments, so it
+; must clear the available-load map and leave the interleave count at 1.
 define void @writing_call_between(ptr noalias %a, ptr noalias %c, i64 %n) {
 ; CHECK-LABEL: define void @writing_call_between(
 ; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) {
@@ -313,7 +313,8 @@ exit:
 }
 
 ; Stride-2 accesses are represented by interleave-group recipes rather than by
-; simple consecutive widened loads.
+; simple consecutive widened loads, so the analysis fails closed and leaves the
+; interleave count at 1.
 define void @non_unit_stride(ptr noalias %a, ptr noalias %c, i64 %n) {
 ; CHECK-LABEL: define void @non_unit_stride(
 ; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) {
@@ -387,32 +388,40 @@ exit:
   ret void
 }
 
-; VPlan folds each identical-arm select to its underlying GEP, while
-; ScalarEvolution keeps each nonconstant pointer select as a distinct unknown.
-; The address of these loads is therefore only exact in the folded VPlan value,
-; not in the scalar load ingredient.
+; VPlan folds each identical-arm select to its underlying GEP. ScalarEvolution
+; keeps each nonconstant pointer select as a distinct unknown, so deriving the
+; address from the scalar load ingredient would miss the exact match. Deriving
+; it from the folded VPlan value exposes cross-part load redundancy and raises
+; IC to 2.
 define void @folded_provenance(ptr noalias %a, ptr noalias %c, i64 %n, i1 %cond) {
 ; CHECK-LABEL: define void @folded_provenance(
 ; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]], i1 [[COND:%.*]]) {
 ; CHECK-NEXT:  [[ENTRY:.*]]:
 ; CHECK-NEXT:    [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
-; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 4
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 8
 ; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
 ; CHECK:       [[VECTOR_PH]]:
-; CHECK-NEXT:    [[TMP1:%.*]] = and i64 [[TMP0]], 3
+; CHECK-NEXT:    [[TMP1:%.*]] = and i64 [[TMP0]], 7
 ; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
 ; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
 ; CHECK:       [[VECTOR_BODY]]:
 ; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
 ; CHECK-NEXT:    [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT:    [[TMP9:%.*]] = getelementptr inbounds i32, ptr [[TMP2]], i64 4
 ; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4
+; CHECK-NEXT:    [[WIDE_LOAD2:%.*]] = load <4 x i32>, ptr [[TMP9]], align 4
 ; CHECK-NEXT:    [[TMP3:%.*]] = add nuw nsw i64 [[INDEX]], 4
 ; CHECK-NEXT:    [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP3]]
+; CHECK-NEXT:    [[TMP11:%.*]] = getelementptr inbounds i32, ptr [[TMP4]], i64 4
 ; CHECK-NEXT:    [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP4]], align 4
+; CHECK-NEXT:    [[WIDE_LOAD3:%.*]] = load <4 x i32>, ptr [[TMP11]], align 4
 ; CHECK-NEXT:    [[TMP5:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[WIDE_LOAD1]]
+; CHECK-NEXT:    [[TMP8:%.*]] = add <4 x i32> [[WIDE_LOAD2]], [[WIDE_LOAD3]]
 ; CHECK-NEXT:    [[TMP6:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
+; CHECK-NEXT:    [[TMP10:%.*]] = getelementptr inbounds i32, ptr [[TMP6]], i64 4
 ; CHECK-NEXT:    store <4 x i32> [[TMP5]], ptr [[TMP6]], align 4
-; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT:    store <4 x i32> [[TMP8]], ptr [[TMP10]], align 4
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
 ; CHECK-NEXT:    [[TMP7:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
 ; CHECK-NEXT:    br i1 [[TMP7]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP17:![0-9]+]]
 ; CHECK:       [[MIDDLE_BLOCK]]:
@@ -463,29 +472,38 @@ exit:
 }
 
 ; The two scalar loads carry different poison-generating !range metadata, which
-; is not propagated to the widened loads.
+; is not propagated to the widened loads. Exact VPlan address and type equality
+; therefore raises the interleave count to 2 without consulting the scalar
+; loads' metadata.
 define void @poison_annotations(ptr noalias %a, ptr noalias %c, i64 %n) {
 ; CHECK-LABEL: define void @poison_annotations(
 ; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) {
 ; CHECK-NEXT:  [[ENTRY:.*]]:
 ; CHECK-NEXT:    [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
-; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 4
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 8
 ; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
 ; CHECK:       [[VECTOR_PH]]:
-; CHECK-NEXT:    [[TMP1:%.*]] = and i64 [[TMP0]], 3
+; CHECK-NEXT:    [[TMP1:%.*]] = and i64 [[TMP0]], 7
 ; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
 ; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
 ; CHECK:       [[VECTOR_BODY]]:
 ; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
 ; CHECK-NEXT:    [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT:    [[TMP9:%.*]] = getelementptr inbounds i32, ptr [[TMP2]], i64 4
 ; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4
+; CHECK-NEXT:    [[WIDE_LOAD2:%.*]] = load <4 x i32>, ptr [[TMP9]], align 4
 ; CHECK-NEXT:    [[TMP3:%.*]] = add nuw nsw i64 [[INDEX]], 4
 ; CHECK-NEXT:    [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP3]]
+; CHECK-NEXT:    [[TMP11:%.*]] = getelementptr inbounds i32, ptr [[TMP4]], i64 4
 ; CHECK-NEXT:    [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP4]], align 4
+; CHECK-NEXT:    [[WIDE_LOAD3:%.*]] = load <4 x i32>, ptr [[TMP11]], align 4
 ; CHECK-NEXT:    [[TMP5:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[WIDE_LOAD1]]
+; CHECK-NEXT:    [[TMP8:%.*]] = add <4 x i32> [[WIDE_LOAD2]], [[WIDE_LOAD3]]
 ; CHECK-NEXT:    [[TMP6:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
+; CHECK-NEXT:    [[TMP10:%.*]] = getelementptr inbounds i32, ptr [[TMP6]], i64 4
 ; CHECK-NEXT:    store <4 x i32> [[TMP5]], ptr [[TMP6]], align 4
-; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT:    store <4 x i32> [[TMP8]], ptr [[TMP10]], align 4
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
 ; CHECK-NEXT:    [[TMP7:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
 ; CHECK-NEXT:    br i1 [[TMP7]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP19:![0-9]+]]
 ; CHECK:       [[MIDDLE_BLOCK]]:
@@ -531,30 +549,38 @@ exit:
   ret void
 }
 
-; The scalar loop has a separate latch, while VPlan places the two matching
-; loads in its single vector-loop block.
+; The scalar loop has a separate latch, but VPlan places the matching loads in
+; its single vector-loop block. The VPlan structural check therefore accepts
+; the loop and the redundancy raises the interleave count to 2.
 define void @multi_block(ptr noalias %a, ptr noalias %c, i64 %n) {
 ; CHECK-LABEL: define void @multi_block(
 ; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) {
 ; CHECK-NEXT:  [[ENTRY:.*]]:
 ; CHECK-NEXT:    [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
-; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 4
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 8
 ; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
 ; CHECK:       [[VECTOR_PH]]:
-; CHECK-NEXT:    [[TMP1:%.*]] = and i64 [[TMP0]], 3
+; CHECK-NEXT:    [[TMP1:%.*]] = and i64 [[TMP0]], 7
 ; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
 ; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
 ; CHECK:       [[VECTOR_BODY]]:
 ; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
 ; CHECK-NEXT:    [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT:    [[TMP9:%.*]] = getelementptr inbounds i32, ptr [[TMP2]], i64 4
 ; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4
+; CHECK-NEXT:    [[WIDE_LOAD2:%.*]] = load <4 x i32>, ptr [[TMP9]], align 4
 ; CHECK-NEXT:    [[TMP3:%.*]] = add nuw nsw i64 [[INDEX]], 4
 ; CHECK-NEXT:    [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP3]]
+; CHECK-NEXT:    [[TMP11:%.*]] = getelementptr inbounds i32, ptr [[TMP4]], i64 4
 ; CHECK-NEXT:    [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP4]], align 4
+; CHECK-NEXT:    [[WIDE_LOAD3:%.*]] = load <4 x i32>, ptr [[TMP11]], align 4
 ; CHECK-NEXT:    [[TMP5:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[WIDE_LOAD1]]
+; CHECK-NEXT:    [[TMP8:%.*]] = add <4 x i32> [[WIDE_LOAD2]], [[WIDE_LOAD3]]
 ; CHECK-NEXT:    [[TMP6:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
+; CHECK-NEXT:    [[TMP10:%.*]] = getelementptr inbounds i32, ptr [[TMP6]], i64 4
 ; CHECK-NEXT:    store <4 x i32> [[TMP5]], ptr [[TMP6]], align 4
-; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT:    store <4 x i32> [[TMP8]], ptr [[TMP10]], align 4
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
 ; CHECK-NEXT:    [[TMP7:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
 ; CHECK-NEXT:    br i1 [[TMP7]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP23:![0-9]+]]
 ; CHECK:       [[MIDDLE_BLOCK]]:
@@ -606,7 +632,8 @@ exit:
 }
 
 ; At VF=4, the first load in part 1 and the second load in part 0 both use the
-; reverse vector ending at a[last-iv-7].
+; reverse vector ending at a[last-iv-7]. Their exact redundancy raises the
+; interleave count to 2.
 define void @reverse_redundancy(ptr noalias %a, ptr noalias %c, i64 %n) {
 ; CHECK-LABEL: define void @reverse_redundancy(
 ; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) {
@@ -614,10 +641,10 @@ define void @reverse_redundancy(ptr noalias %a, ptr noalias %c, i64 %n) {
 ; CHECK-NEXT:    [[LAST:%.*]] = add i64 [[N]], -1
 ; CHECK-NEXT:    [[TRIP_COUNT:%.*]] = add i64 [[N]], -4
 ; CHECK-NEXT:    [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[TRIP_COUNT]], i64 1)
-; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 4
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 8
 ; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
 ; CHECK:       [[VECTOR_PH]]:
-; CHECK-NEXT:    [[TMP1:%.*]] = and i64 [[TMP0]], 3
+; CHECK-NEXT:    [[TMP1:%.*]] = and i64 [[TMP0]], 7
 ; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
 ; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
 ; CHECK:       [[VECTOR_BODY]]:
@@ -625,18 +652,26 @@ define void @reverse_redundancy(ptr noalias %a, ptr noalias %c, i64 %n) {
 ; CHECK-NEXT:    [[TMP2:%.*]] = sub i64 [[LAST]], [[INDEX]]
 ; CHECK-NEXT:    [[TMP3:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP2]]
 ; CHECK-NEXT:    [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[TMP3]], i64 -3
+; CHECK-NEXT:    [[TMP5:%.*]] = getelementptr inbounds i32, ptr [[TMP3]], i64 -7
 ; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP4]], align 4
-; CHECK-NEXT:    [[TMP5:%.*]] = add i64 [[TMP2]], -4
-; CHECK-NEXT:    [[TMP6:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP5]]
-; CHECK-NEXT:    [[TMP7:%.*]] = getelementptr inbounds i32, ptr [[TMP6]], i64 -3
-; CHECK-NEXT:    [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP7]], align 4
-; CHECK-NEXT:    [[TMP8:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[WIDE_LOAD1]]
-; CHECK-NEXT:    [[REVERSE:%.*]] = shufflevector <4 x i32> [[TMP8]], <4 x i32> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
-; CHECK-NEXT:    [[TMP9:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
-; CHECK-NEXT:    store <4 x i32> [[REVERSE]], ptr [[TMP9]], align 4
-; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
-; CHECK-NEXT:    [[TMP10:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-NEXT:    br i1 [[TMP10]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP25:![0-9]+]]
+; CHECK-NEXT:    [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP5]], align 4
+; CHECK-NEXT:    [[TMP6:%.*]] = add i64 [[TMP2]], -4
+; CHECK-NEXT:    [[TMP7:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP6]]
+; CHECK-NEXT:    [[TMP8:%.*]] = getelementptr inbounds i32, ptr [[TMP7]], i64 -3
+; CHECK-NEXT:    [[TMP9:%.*]] = getelementptr inbounds i32, ptr [[TMP7]], i64 -7
+; CHECK-NEXT:    [[WIDE_LOAD2:%.*]] = load <4 x i32>, ptr [[TMP8]], align 4
+; CHECK-NEXT:    [[WIDE_LOAD3:%.*]] = load <4 x i32>, ptr [[TMP9]], align 4
+; CHECK-NEXT:    [[TMP10:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[WIDE_LOAD2]]
+; CHECK-NEXT:    [[TMP11:%.*]] = add <4 x i32> [[WIDE_LOAD1]], [[WIDE_LOAD3]]
+; CHECK-NEXT:    [[REVERSE:%.*]] = shufflevector <4 x i32> [[TMP10]], <4 x i32> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
+; CHECK-NEXT:    [[REVERSE4:%.*]] = shufflevector <4 x i32> [[TMP11]], <4 x i32> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
+; CHECK-NEXT:    [[TMP12:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
+; CHECK-NEXT:    [[TMP13:%.*]] = getelementptr inbounds i32, ptr [[TMP12]], i64 4
+; CHECK-NEXT:    store <4 x i32> [[REVERSE]], ptr [[TMP12]], align 4
+; CHECK-NEXT:    store <4 x i32> [[REVERSE4]], ptr [[TMP13]], align 4
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
+; CHECK-NEXT:    [[TMP14:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP14]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP25:![0-9]+]]
 ; CHECK:       [[MIDDLE_BLOCK]]:
 ; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
 ; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
@@ -686,36 +721,48 @@ exit:
 
 ; The reverse access sits physically between the two matching forward loads.
 ; Its end-pointer address computation is pure and must not be mistaken for a
-; memory write.
+; memory write. The one forward equality therefore raises the interleave count
+; to 2.
 define void @reverse_between(ptr noalias %a, ptr noalias %c, i64 %n) {
 ; CHECK-LABEL: define void @reverse_between(
 ; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) {
 ; CHECK-NEXT:  [[ENTRY:.*]]:
 ; CHECK-NEXT:    [[LAST:%.*]] = add i64 [[N]], -1
 ; CHECK-NEXT:    [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
-; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 4
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 8
 ; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
 ; CHECK:       [[VECTOR_PH]]:
-; CHECK-NEXT:    [[TMP1:%.*]] = and i64 [[TMP0]], 3
+; CHECK-NEXT:    [[TMP1:%.*]] = and i64 [[TMP0]], 7
 ; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
 ; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
 ; CHECK:       [[VECTOR_BODY]]:
 ; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
 ; CHECK-NEXT:    [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT:    [[TMP13:%.*]] = getelementptr inbounds i32, ptr [[TMP2]], i64 4
 ; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4
+; CHECK-NEXT:    [[WIDE_LOAD2:%.*]] = load <4 x i32>, ptr [[TMP13]], align 4
 ; CHECK-NEXT:    [[TMP3:%.*]] = sub i64 [[LAST]], [[INDEX]]
 ; CHECK-NEXT:    [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP3]]
 ; CHECK-NEXT:    [[TMP5:%.*]] = getelementptr inbounds i32, ptr [[TMP4]], i64 -3
+; CHECK-NEXT:    [[TMP15:%.*]] = getelementptr inbounds i32, ptr [[TMP4]], i64 -7
 ; CHECK-NEXT:    [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP5]], align 4
+; CHECK-NEXT:    [[WIDE_LOAD4:%.*]] = load <4 x i32>, ptr [[TMP15]], align 4
 ; CHECK-NEXT:    [[REVERSE2:%.*]] = shufflevector <4 x i32> [[WIDE_LOAD1]], <4 x i32> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
+; CHECK-NEXT:    [[REVERSE5:%.*]] = shufflevector <4 x i32> [[WIDE_LOAD4]], <4 x i32> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
 ; CHECK-NEXT:    [[TMP6:%.*]] = add nuw nsw i64 [[INDEX]], 4
 ; CHECK-NEXT:    [[TMP7:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP6]]
+; CHECK-NEXT:    [[TMP17:%.*]] = getelementptr inbounds i32, ptr [[TMP7]], i64 4
 ; CHECK-NEXT:    [[WIDE_LOAD3:%.*]] = load <4 x i32>, ptr [[TMP7]], align 4
+; CHECK-NEXT:    [[WIDE_LOAD7:%.*]] = load <4 x i32>, ptr [[TMP17]], align 4
 ; CHECK-NEXT:    [[TMP8:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[WIDE_LOAD3]]
+; CHECK-NEXT:    [[TMP12:%.*]] = add <4 x i32> [[WIDE_LOAD2]], [[WIDE_LOAD7]]
 ; CHECK-NEXT:    [[TMP9:%.*]] = add <4 x i32> [[TMP8]], [[REVERSE2]]
+; CHECK-NEXT:    [[TMP14:%.*]] = add <4 x i32> [[TMP12]], [[REVERSE5]]
 ; CHECK-NEXT:    [[TMP10:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
+; CHECK-NEXT:    [[TMP16:%.*]] = getelementptr inbounds i32, ptr [[TMP10]], i64 4
 ; CHECK-NEXT:    store <4 x i32> [[TMP9]], ptr [[TMP10]], align 4
-; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT:    store <4 x i32> [[TMP14]], ptr [[TMP16]], align 4
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
 ; CHECK-NEXT:    [[TMP11:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
 ; CHECK-NEXT:    br i1 [[TMP11]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP27:![0-9]+]]
 ; CHECK:       [[MIDDLE_BLOCK]]:
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-narrowed.ll b/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-narrowed.ll
new file mode 100644
index 0000000000000..f26c7632c9f93
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-narrowed.ll
@@ -0,0 +1,50 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
+; Narrowing an interleave group materializes VFxUF before IC selection. The
+; redundancy analysis must fail closed rather than interpret the selected VF as
+; the narrowed plan's physical step.
+; RUN: opt -passes=loop-vectorize -mtriple=arm64-apple-macosx \
+; RUN:     -force-vector-width=2 -force-target-max-vector-interleave=2 \
+; RUN:     -force-target-num-vector-regs=1024 \
+; RUN:     -force-target-instruction-cost=1 -small-loop-cost=0 \
+; RUN:     -enable-interleave-cse -interleave-cse-min-pct=1 \
+; RUN:     -S %s | FileCheck %s
+
+target triple = "arm64-apple-macosx"
+
+define void @narrowed(ptr noalias %a, i64 %value) {
+; CHECK-LABEL: define void @narrowed(
+; CHECK-SAME: ptr noalias [[A:%.*]], i64 [[VALUE:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    br label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <2 x i64> poison, i64 [[VALUE]], i64 0
+; CHECK-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <2 x i64> [[BROADCAST_SPLATINSERT]], <2 x i64> poison, <2 x i32> zeroinitializer
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP0:%.*]] = getelementptr inbounds { i64, i64 }, ptr [[A]], i64 [[INDEX]], i32 0
+; CHECK-NEXT:    store <2 x i64> [[BROADCAST_SPLAT]], ptr [[TMP0]], align 8
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 1
+; CHECK-NEXT:    [[TMP1:%.*]] = icmp eq i64 [[INDEX_NEXT]], 100
+; CHECK-NEXT:    br i1 [[TMP1]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    br label %[[EXIT:.*]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %p0 = getelementptr inbounds { i64, i64 }, ptr %a, i64 %iv, i32 0
+  %p1 = getelementptr inbounds { i64, i64 }, ptr %a, i64 %iv, i32 1
+  store i64 %value, ptr %p0, align 8
+  store i64 %value, ptr %p1, align 8
+  %iv.next = add nuw nsw i64 %iv, 1
+  %done = icmp eq i64 %iv.next, 100
+  br i1 %done, label %exit, label %loop
+
+exit:
+  ret void
+}
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-opportunities.ll b/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-opportunities.ll
index fe8e8c0b53880..cf1ccc4aaac73 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-opportunities.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-opportunities.ll
@@ -1,17 +1,20 @@
 ; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
-; Pre-commit test for counting cross-part load redundancy opportunities. The
-; two loops below differ only in how many distinct opportunities they contain;
-; both vectorize with an interleave count of 1 today.
+; A minimum saving of 6% is above what one redundancy opportunity in this loop
+; can deliver and below what two opportunities deliver, so it separates the two
+; functions below.
 ; RUN: opt -passes=loop-vectorize -force-vector-width=4 \
 ; RUN:     -force-target-max-vector-interleave=2 -small-loop-cost=0 \
+; RUN:     -force-target-instruction-cost=1 \
+; RUN:     -enable-interleave-cse -interleave-cse-min-pct=6 \
 ; RUN:     -S %s | FileCheck %s
 
 target triple = "aarch64-unknown-linux-gnu"
 
 ; The first a[i+4] load forms a cross-part redundancy opportunity with a[i].
 ; The second a[i+4] load is a duplicate within the same logical part: it already
-; exists without interleaving and therefore is no additional cross-part
-; opportunity, which leaves a single opportunity in this loop.
+; exists without interleaving and therefore counts as no additional cross-part
+; opportunity. The single opportunity stays below the 6% threshold, so the
+; interleave count remains 1.
 define void @duplicate_after_cross_part(ptr noalias %a, ptr noalias %c, i64 %n) {
 ; CHECK-LABEL: define void @duplicate_after_cross_part(
 ; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) {
@@ -85,34 +88,45 @@ exit:
 }
 
 ; At VF=4, a[i+4] in part 0 is redundant with a[i] in part 1, and a[i+8] in
-; part 0 is independently redundant with a[i+4] in part 1, which gives two
-; distinct opportunities in this loop.
+; part 0 is independently redundant with a[i+4] in part 1, giving two distinct
+; opportunities. Their combined saving passes the 6% threshold and raises the
+; interleave count to 2.
 define void @two_opportunities(ptr noalias %a, ptr noalias %c, i64 %n) {
 ; CHECK-LABEL: define void @two_opportunities(
 ; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) {
 ; CHECK-NEXT:  [[ENTRY:.*]]:
 ; CHECK-NEXT:    [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
-; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 4
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 8
 ; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
 ; CHECK:       [[VECTOR_PH]]:
-; CHECK-NEXT:    [[TMP1:%.*]] = and i64 [[TMP0]], 3
+; CHECK-NEXT:    [[TMP1:%.*]] = and i64 [[TMP0]], 7
 ; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
 ; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
 ; CHECK:       [[VECTOR_BODY]]:
 ; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
 ; CHECK-NEXT:    [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT:    [[TMP12:%.*]] = getelementptr inbounds i32, ptr [[TMP2]], i64 4
 ; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4
+; CHECK-NEXT:    [[WIDE_LOAD4:%.*]] = load <4 x i32>, ptr [[TMP12]], align 4
 ; CHECK-NEXT:    [[TMP3:%.*]] = add nuw nsw i64 [[INDEX]], 4
 ; CHECK-NEXT:    [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP3]]
+; CHECK-NEXT:    [[TMP14:%.*]] = getelementptr inbounds i32, ptr [[TMP4]], i64 4
 ; CHECK-NEXT:    [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP4]], align 4
+; CHECK-NEXT:    [[WIDE_LOAD3:%.*]] = load <4 x i32>, ptr [[TMP14]], align 4
 ; CHECK-NEXT:    [[TMP5:%.*]] = add nuw nsw i64 [[INDEX]], 8
 ; CHECK-NEXT:    [[TMP6:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP5]]
+; CHECK-NEXT:    [[TMP16:%.*]] = getelementptr inbounds i32, ptr [[TMP6]], i64 4
 ; CHECK-NEXT:    [[WIDE_LOAD2:%.*]] = load <4 x i32>, ptr [[TMP6]], align 4
+; CHECK-NEXT:    [[WIDE_LOAD5:%.*]] = load <4 x i32>, ptr [[TMP16]], align 4
 ; CHECK-NEXT:    [[TMP7:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[WIDE_LOAD1]]
+; CHECK-NEXT:    [[TMP11:%.*]] = add <4 x i32> [[WIDE_LOAD4]], [[WIDE_LOAD3]]
 ; CHECK-NEXT:    [[TMP8:%.*]] = add <4 x i32> [[TMP7]], [[WIDE_LOAD2]]
+; CHECK-NEXT:    [[TMP13:%.*]] = add <4 x i32> [[TMP11]], [[WIDE_LOAD5]]
 ; CHECK-NEXT:    [[TMP9:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
+; CHECK-NEXT:    [[TMP15:%.*]] = getelementptr inbounds i32, ptr [[TMP9]], i64 4
 ; CHECK-NEXT:    store <4 x i32> [[TMP8]], ptr [[TMP9]], align 4
-; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT:    store <4 x i32> [[TMP13]], ptr [[TMP15]], align 4
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
 ; CHECK-NEXT:    [[TMP10:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
 ; CHECK-NEXT:    br i1 [[TMP10]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
 ; CHECK:       [[MIDDLE_BLOCK]]:
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-pipeline.ll b/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-pipeline.ll
new file mode 100644
index 0000000000000..a94d4685cc9ac
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-pipeline.ll
@@ -0,0 +1,49 @@
+; The redundancy analysis changes only the interleave count. This test
+; demonstrates that the standard O3 pipeline can realize the motivating
+; opportunity, without making downstream elimination part of the contract.
+;
+; RUN: opt -passes='default<O3>' -force-vector-width=4 \
+; RUN:     -force-target-max-vector-interleave=2 -small-loop-cost=0 \
+; RUN:     -enable-interleave-cse -interleave-cse-min-pct=1 \
+; RUN:     -S %s | FileCheck %s --check-prefix=ENABLED
+; RUN: opt -passes='default<O3>' -force-vector-width=4 \
+; RUN:     -force-target-max-vector-interleave=2 -small-loop-cost=0 \
+; RUN:     -S %s | FileCheck %s --check-prefix=DISABLED
+
+target triple = "aarch64-unknown-linux-gnu"
+
+; With the analysis enabled, IC=2 exposes one redundant vector load to the O3
+; pipeline. The vector body processes eight source iterations with three loads.
+; With the analysis disabled, IC=1 processes four iterations with two loads.
+define void @positive(ptr noalias %a, ptr noalias %c, i64 %n) {
+; ENABLED-LABEL: define void @positive(
+; ENABLED:       vector.body:
+; ENABLED-COUNT-3: load <4 x i32>
+; ENABLED-NOT:   load <4 x i32>
+; ENABLED:       add nuw i64 {{.*}}, 8
+;
+; DISABLED-LABEL: define void @positive(
+; DISABLED:       vector.body:
+; DISABLED-COUNT-2: load <4 x i32>
+; DISABLED-NOT:   load <4 x i32>
+; DISABLED:       add nuw i64 {{.*}}, 4
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %a.iv = getelementptr inbounds i32, ptr %a, i64 %iv
+  %l1 = load i32, ptr %a.iv, align 4
+  %iv.plus.4 = add nuw nsw i64 %iv, 4
+  %a.iv.plus.4 = getelementptr inbounds i32, ptr %a, i64 %iv.plus.4
+  %l2 = load i32, ptr %a.iv.plus.4, align 4
+  %sum = add i32 %l1, %l2
+  %c.iv = getelementptr inbounds i32, ptr %c, i64 %iv
+  store i32 %sum, ptr %c.iv, align 4
+  %iv.next = add nuw nsw i64 %iv, 1
+  %cmp = icmp slt i64 %iv.next, %n
+  br i1 %cmp, label %loop, label %exit
+
+exit:
+  ret void
+}
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-scalable-reverse.ll b/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-scalable-reverse.ll
new file mode 100644
index 0000000000000..0bd8906b48adc
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-scalable-reverse.ll
@@ -0,0 +1,188 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
+; RUN: opt -passes=loop-vectorize -mtriple=aarch64-none-linux-gnu -mattr=+sve \
+; RUN:     -force-vector-width="vscale x 2" \
+; RUN:     -force-target-max-vector-interleave=2 -small-loop-cost=0 \
+; RUN:     -enable-interleave-cse -interleave-cse-min-pct=1 \
+; RUN:     -S %s | FileCheck %s
+
+target triple = "aarch64-unknown-linux-gnu"
+
+declare i64 @llvm.vscale.i64()
+
+; A constant source offset cannot equal the scalable reverse part offset for
+; every runtime vscale, so the interleave count stays 1.
+define void @reverse_constant_offset(ptr noalias %a, ptr noalias %c, i64 %n) {
+; CHECK-LABEL: define void @reverse_constant_offset(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) #[[ATTR1:[0-9]+]] {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[LAST:%.*]] = add i64 [[N]], -1
+; CHECK-NEXT:    [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; CHECK-NEXT:    [[TMP1:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-NEXT:    [[TMP2:%.*]] = shl nuw i64 [[TMP1]], 1
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], [[TMP2]]
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[TMP0]], [[TMP2]]
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[N_MOD_VF]]
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP3:%.*]] = sub i64 [[LAST]], [[INDEX]]
+; CHECK-NEXT:    [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP3]]
+; CHECK-NEXT:    [[TMP5:%.*]] = sub nuw nsw i64 [[TMP2]], 1
+; CHECK-NEXT:    [[TMP6:%.*]] = sub i64 0, [[TMP5]]
+; CHECK-NEXT:    [[TMP7:%.*]] = getelementptr inbounds i32, ptr [[TMP4]], i64 [[TMP6]]
+; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 2 x i32>, ptr [[TMP7]], align 4
+; CHECK-NEXT:    [[TMP8:%.*]] = add i64 [[TMP3]], -2
+; CHECK-NEXT:    [[TMP9:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP8]]
+; CHECK-NEXT:    [[TMP10:%.*]] = getelementptr inbounds i32, ptr [[TMP9]], i64 [[TMP6]]
+; CHECK-NEXT:    [[WIDE_LOAD1:%.*]] = load <vscale x 2 x i32>, ptr [[TMP10]], align 4
+; CHECK-NEXT:    [[TMP11:%.*]] = add <vscale x 2 x i32> [[WIDE_LOAD]], [[WIDE_LOAD1]]
+; CHECK-NEXT:    [[REVERSE:%.*]] = call <vscale x 2 x i32> @llvm.vector.reverse.nxv2i32(<vscale x 2 x i32> [[TMP11]])
+; CHECK-NEXT:    [[TMP12:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
+; CHECK-NEXT:    store <vscale x 2 x i32> [[REVERSE]], ptr [[TMP12]], align 4
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP2]]
+; CHECK-NEXT:    [[TMP13:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP13]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; CHECK:       [[SCALAR_PH]]:
+; CHECK-NEXT:    [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[REVERSE_IV:%.*]] = sub i64 [[LAST]], [[IV]]
+; CHECK-NEXT:    [[A_REVERSE:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[REVERSE_IV]]
+; CHECK-NEXT:    [[L1:%.*]] = load i32, ptr [[A_REVERSE]], align 4
+; CHECK-NEXT:    [[REVERSE_IV_MINUS_2:%.*]] = add i64 [[REVERSE_IV]], -2
+; CHECK-NEXT:    [[A_REVERSE_MINUS_2:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[REVERSE_IV_MINUS_2]]
+; CHECK-NEXT:    [[L2:%.*]] = load i32, ptr [[A_REVERSE_MINUS_2]], align 4
+; CHECK-NEXT:    [[SUM:%.*]] = add i32 [[L1]], [[L2]]
+; CHECK-NEXT:    [[C_IV:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[IV]]
+; CHECK-NEXT:    store i32 [[SUM]], ptr [[C_IV]], align 4
+; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT:    [[CMP:%.*]] = icmp slt i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[CMP]], label %[[LOOP]], label %[[EXIT]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  %last = add i64 %n, -1
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %reverse.iv = sub i64 %last, %iv
+  %a.reverse = getelementptr inbounds i32, ptr %a, i64 %reverse.iv
+  %l1 = load i32, ptr %a.reverse, align 4
+  %reverse.iv.minus.2 = add i64 %reverse.iv, -2
+  %a.reverse.minus.2 = getelementptr inbounds i32, ptr %a, i64 %reverse.iv.minus.2
+  %l2 = load i32, ptr %a.reverse.minus.2, align 4
+  %sum = add i32 %l1, %l2
+  %c.iv = getelementptr inbounds i32, ptr %c, i64 %iv
+  store i32 %sum, ptr %c.iv, align 4
+  %iv.next = add nuw nsw i64 %iv, 1
+  %cmp = icmp slt i64 %iv.next, %n
+  br i1 %cmp, label %loop, label %exit
+
+exit:
+  ret void
+}
+
+; The second source address is one scalable VF behind the first. Its part-0
+; reverse vector therefore equals the first load's part-1 reverse vector and
+; raises the interleave count to 2.
+define void @reverse_vscale_offset(ptr noalias %a, ptr noalias %c, i64 %n) {
+; CHECK-LABEL: define void @reverse_vscale_offset(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) #[[ATTR1]] {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[LAST:%.*]] = add i64 [[N]], -1
+; CHECK-NEXT:    [[VSCALE:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-NEXT:    [[VF:%.*]] = shl nuw nsw i64 [[VSCALE]], 1
+; CHECK-NEXT:    [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; CHECK-NEXT:    [[TMP1:%.*]] = shl nuw i64 [[VSCALE]], 2
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], [[TMP1]]
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP2:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-NEXT:    [[TMP3:%.*]] = shl nuw i64 [[TMP2]], 1
+; CHECK-NEXT:    [[TMP4:%.*]] = shl nuw i64 [[TMP2]], 2
+; CHECK-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[TMP0]], [[TMP4]]
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[N_MOD_VF]]
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP5:%.*]] = sub i64 [[LAST]], [[INDEX]]
+; CHECK-NEXT:    [[TMP6:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP5]]
+; CHECK-NEXT:    [[TMP7:%.*]] = sub nuw nsw i64 [[TMP3]], 1
+; CHECK-NEXT:    [[TMP8:%.*]] = sub i64 0, [[TMP7]]
+; CHECK-NEXT:    [[TMP9:%.*]] = getelementptr inbounds i32, ptr [[TMP6]], i64 [[TMP8]]
+; CHECK-NEXT:    [[TMP10:%.*]] = sub i64 [[TMP8]], [[TMP3]]
+; CHECK-NEXT:    [[TMP11:%.*]] = getelementptr inbounds i32, ptr [[TMP6]], i64 [[TMP10]]
+; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 2 x i32>, ptr [[TMP9]], align 4
+; CHECK-NEXT:    [[WIDE_LOAD1:%.*]] = load <vscale x 2 x i32>, ptr [[TMP11]], align 4
+; CHECK-NEXT:    [[TMP12:%.*]] = sub i64 [[TMP5]], [[VF]]
+; CHECK-NEXT:    [[TMP13:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP12]]
+; CHECK-NEXT:    [[TMP14:%.*]] = getelementptr inbounds i32, ptr [[TMP13]], i64 [[TMP8]]
+; CHECK-NEXT:    [[TMP15:%.*]] = getelementptr inbounds i32, ptr [[TMP13]], i64 [[TMP10]]
+; CHECK-NEXT:    [[WIDE_LOAD2:%.*]] = load <vscale x 2 x i32>, ptr [[TMP14]], align 4
+; CHECK-NEXT:    [[WIDE_LOAD3:%.*]] = load <vscale x 2 x i32>, ptr [[TMP15]], align 4
+; CHECK-NEXT:    [[TMP16:%.*]] = add <vscale x 2 x i32> [[WIDE_LOAD]], [[WIDE_LOAD2]]
+; CHECK-NEXT:    [[TMP17:%.*]] = add <vscale x 2 x i32> [[WIDE_LOAD1]], [[WIDE_LOAD3]]
+; CHECK-NEXT:    [[REVERSE:%.*]] = call <vscale x 2 x i32> @llvm.vector.reverse.nxv2i32(<vscale x 2 x i32> [[TMP16]])
+; CHECK-NEXT:    [[REVERSE4:%.*]] = call <vscale x 2 x i32> @llvm.vector.reverse.nxv2i32(<vscale x 2 x i32> [[TMP17]])
+; CHECK-NEXT:    [[TMP18:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
+; CHECK-NEXT:    [[TMP19:%.*]] = getelementptr inbounds i32, ptr [[TMP18]], i64 [[TMP3]]
+; CHECK-NEXT:    store <vscale x 2 x i32> [[REVERSE]], ptr [[TMP18]], align 4
+; CHECK-NEXT:    store <vscale x 2 x i32> [[REVERSE4]], ptr [[TMP19]], align 4
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP4]]
+; CHECK-NEXT:    [[TMP20:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP20]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; CHECK:       [[SCALAR_PH]]:
+; CHECK-NEXT:    [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[REVERSE_IV:%.*]] = sub i64 [[LAST]], [[IV]]
+; CHECK-NEXT:    [[A_REVERSE:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[REVERSE_IV]]
+; CHECK-NEXT:    [[L1:%.*]] = load i32, ptr [[A_REVERSE]], align 4
+; CHECK-NEXT:    [[REVERSE_IV_MINUS_VF:%.*]] = sub i64 [[REVERSE_IV]], [[VF]]
+; CHECK-NEXT:    [[A_REVERSE_MINUS_VF:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[REVERSE_IV_MINUS_VF]]
+; CHECK-NEXT:    [[L2:%.*]] = load i32, ptr [[A_REVERSE_MINUS_VF]], align 4
+; CHECK-NEXT:    [[SUM:%.*]] = add i32 [[L1]], [[L2]]
+; CHECK-NEXT:    [[C_IV:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[IV]]
+; CHECK-NEXT:    store i32 [[SUM]], ptr [[C_IV]], align 4
+; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT:    [[CMP:%.*]] = icmp slt i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[CMP]], label %[[LOOP]], label %[[EXIT]], !llvm.loop [[LOOP5:![0-9]+]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  %last = add i64 %n, -1
+  %vscale = call i64 @llvm.vscale.i64()
+  %vf = shl nuw nsw i64 %vscale, 1
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %reverse.iv = sub i64 %last, %iv
+  %a.reverse = getelementptr inbounds i32, ptr %a, i64 %reverse.iv
+  %l1 = load i32, ptr %a.reverse, align 4
+  %reverse.iv.minus.vf = sub i64 %reverse.iv, %vf
+  %a.reverse.minus.vf = getelementptr inbounds i32, ptr %a, i64 %reverse.iv.minus.vf
+  %l2 = load i32, ptr %a.reverse.minus.vf, align 4
+  %sum = add i32 %l1, %l2
+  %c.iv = getelementptr inbounds i32, ptr %c, i64 %iv
+  store i32 %sum, ptr %c.iv, align 4
+  %iv.next = add nuw nsw i64 %iv, 1
+  %cmp = icmp slt i64 %iv.next, %n
+  br i1 %cmp, label %loop, label %exit
+
+exit:
+  ret void
+}
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-scalable.ll b/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-scalable.ll
index 2af48b332d9b2..f566c9e757c78 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-scalable.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-scalable.ll
@@ -1,10 +1,10 @@
 ; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
-; Pre-commit test for cross-part load redundancy with scalable vectors, where
-; the distance between two logical parts is a runtime multiple of vscale. Both
-; loops below vectorize with an interleave count of 1 today.
+; The minimum saving is lowered to 1% so that the difference between the two
+; functions below comes from exact scalable address equality alone.
 ; RUN: opt -passes=loop-vectorize -mtriple=aarch64-none-linux-gnu -mattr=+sve \
 ; RUN:     -force-vector-width="vscale x 2" \
 ; RUN:     -force-target-max-vector-interleave=2 -small-loop-cost=0 \
+; RUN:     -enable-interleave-cse -interleave-cse-min-pct=1 \
 ; RUN:     -S %s | FileCheck %s
 
 target triple = "aarch64-unknown-linux-gnu"
@@ -12,7 +12,8 @@ target triple = "aarch64-unknown-linux-gnu"
 declare i64 @llvm.vscale.i64()
 
 ; A constant source offset cannot equal the scalable logical-part offset for
-; every runtime vscale, so the two addresses are not exactly equal.
+; every runtime vscale, so the two addresses are not exactly equal and the
+; interleave count stays 1.
 define void @constant_offset(ptr noalias %a, ptr noalias %c, i64 %n) {
 ; CHECK-LABEL: define void @constant_offset(
 ; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) #[[ATTR1:[0-9]+]] {
@@ -83,7 +84,8 @@ exit:
 }
 
 ; The second load starts exactly one scalable VF after the first, so its part-0
-; address equals the first load's part-1 address for every runtime vscale.
+; address equals the first load's part-1 address for every runtime vscale, which
+; raises the interleave count to 2.
 define void @vscale_offset(ptr noalias %a, ptr noalias %c, i64 %n) {
 ; CHECK-LABEL: define void @vscale_offset(
 ; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) #[[ATTR1]] {
@@ -91,25 +93,34 @@ define void @vscale_offset(ptr noalias %a, ptr noalias %c, i64 %n) {
 ; CHECK-NEXT:    [[VSCALE:%.*]] = call i64 @llvm.vscale.i64()
 ; CHECK-NEXT:    [[PART_OFFSET:%.*]] = mul nuw i64 [[VSCALE]], 2
 ; CHECK-NEXT:    [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
-; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], [[PART_OFFSET]]
+; CHECK-NEXT:    [[TMP10:%.*]] = shl nuw i64 [[VSCALE]], 2
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], [[TMP10]]
 ; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
 ; CHECK:       [[VECTOR_PH]]:
 ; CHECK-NEXT:    [[TMP1:%.*]] = call i64 @llvm.vscale.i64()
 ; CHECK-NEXT:    [[TMP2:%.*]] = shl nuw i64 [[TMP1]], 1
-; CHECK-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[TMP0]], [[TMP2]]
+; CHECK-NEXT:    [[TMP12:%.*]] = shl nuw i64 [[TMP1]], 2
+; CHECK-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[TMP0]], [[TMP12]]
 ; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[N_MOD_VF]]
 ; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
 ; CHECK:       [[VECTOR_BODY]]:
 ; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
 ; CHECK-NEXT:    [[TMP3:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT:    [[TMP14:%.*]] = getelementptr inbounds i32, ptr [[TMP3]], i64 [[TMP2]]
 ; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 2 x i32>, ptr [[TMP3]], align 4
+; CHECK-NEXT:    [[WIDE_LOAD2:%.*]] = load <vscale x 2 x i32>, ptr [[TMP14]], align 4
 ; CHECK-NEXT:    [[TMP4:%.*]] = add i64 [[INDEX]], [[PART_OFFSET]]
 ; CHECK-NEXT:    [[TMP5:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP4]]
+; CHECK-NEXT:    [[TMP9:%.*]] = getelementptr inbounds i32, ptr [[TMP5]], i64 [[TMP2]]
 ; CHECK-NEXT:    [[WIDE_LOAD1:%.*]] = load <vscale x 2 x i32>, ptr [[TMP5]], align 4
+; CHECK-NEXT:    [[WIDE_LOAD3:%.*]] = load <vscale x 2 x i32>, ptr [[TMP9]], align 4
 ; CHECK-NEXT:    [[TMP6:%.*]] = add <vscale x 2 x i32> [[WIDE_LOAD]], [[WIDE_LOAD1]]
+; CHECK-NEXT:    [[TMP11:%.*]] = add <vscale x 2 x i32> [[WIDE_LOAD2]], [[WIDE_LOAD3]]
 ; CHECK-NEXT:    [[TMP7:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
+; CHECK-NEXT:    [[TMP13:%.*]] = getelementptr inbounds i32, ptr [[TMP7]], i64 [[TMP2]]
 ; CHECK-NEXT:    store <vscale x 2 x i32> [[TMP6]], ptr [[TMP7]], align 4
-; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP2]]
+; CHECK-NEXT:    store <vscale x 2 x i32> [[TMP11]], ptr [[TMP13]], align 4
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP12]]
 ; CHECK-NEXT:    [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
 ; CHECK-NEXT:    br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
 ; CHECK:       [[MIDDLE_BLOCK]]:
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-vplan-multi-block.ll b/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-vplan-multi-block.ll
new file mode 100644
index 0000000000000..10e1088e83013
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-vplan-multi-block.ll
@@ -0,0 +1,122 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
+; A predicated store creates a multi-block VPlan region after the matching
+; loads. The analysis scans only a single VPBasicBlock, so it must fail closed
+; and leave the interleave count at 1.
+; RUN: opt -passes=loop-vectorize -force-vector-width=4 \
+; RUN:     -force-target-max-vector-interleave=2 -small-loop-cost=0 \
+; RUN:     -enable-interleave-cse -interleave-cse-min-pct=1 \
+; RUN:     -S %s | FileCheck %s
+
+target triple = "aarch64-unknown-linux-gnu"
+
+define void @vplan_multi_block(ptr noalias %a, ptr noalias %c, i64 %n) {
+; CHECK-LABEL: define void @vplan_multi_block(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 4
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = and i64 [[N]], 3
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[PRED_STORE_CONTINUE7:.*]] ]
+; CHECK-NEXT:    [[TMP1:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP1]], align 4
+; CHECK-NEXT:    [[TMP2:%.*]] = add nuw nsw i64 [[INDEX]], 4
+; CHECK-NEXT:    [[TMP3:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP2]]
+; CHECK-NEXT:    [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP3]], align 4
+; CHECK-NEXT:    [[TMP4:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[WIDE_LOAD1]]
+; CHECK-NEXT:    [[TMP5:%.*]] = icmp eq <4 x i32> [[TMP4]], zeroinitializer
+; CHECK-NEXT:    [[TMP6:%.*]] = extractelement <4 x i1> [[TMP5]], i64 0
+; CHECK-NEXT:    br i1 [[TMP6]], label %[[PRED_STORE_IF:.*]], label %[[PRED_STORE_CONTINUE:.*]]
+; CHECK:       [[PRED_STORE_IF]]:
+; CHECK-NEXT:    [[TMP7:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
+; CHECK-NEXT:    [[TMP8:%.*]] = extractelement <4 x i32> [[TMP4]], i64 0
+; CHECK-NEXT:    store i32 [[TMP8]], ptr [[TMP7]], align 4
+; CHECK-NEXT:    br label %[[PRED_STORE_CONTINUE]]
+; CHECK:       [[PRED_STORE_CONTINUE]]:
+; CHECK-NEXT:    [[TMP9:%.*]] = extractelement <4 x i1> [[TMP5]], i64 1
+; CHECK-NEXT:    br i1 [[TMP9]], label %[[PRED_STORE_IF2:.*]], label %[[PRED_STORE_CONTINUE3:.*]]
+; CHECK:       [[PRED_STORE_IF2]]:
+; CHECK-NEXT:    [[TMP10:%.*]] = add i64 [[INDEX]], 1
+; CHECK-NEXT:    [[TMP11:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[TMP10]]
+; CHECK-NEXT:    [[TMP12:%.*]] = extractelement <4 x i32> [[TMP4]], i64 1
+; CHECK-NEXT:    store i32 [[TMP12]], ptr [[TMP11]], align 4
+; CHECK-NEXT:    br label %[[PRED_STORE_CONTINUE3]]
+; CHECK:       [[PRED_STORE_CONTINUE3]]:
+; CHECK-NEXT:    [[TMP13:%.*]] = extractelement <4 x i1> [[TMP5]], i64 2
+; CHECK-NEXT:    br i1 [[TMP13]], label %[[PRED_STORE_IF4:.*]], label %[[PRED_STORE_CONTINUE5:.*]]
+; CHECK:       [[PRED_STORE_IF4]]:
+; CHECK-NEXT:    [[TMP14:%.*]] = add i64 [[INDEX]], 2
+; CHECK-NEXT:    [[TMP15:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[TMP14]]
+; CHECK-NEXT:    [[TMP16:%.*]] = extractelement <4 x i32> [[TMP4]], i64 2
+; CHECK-NEXT:    store i32 [[TMP16]], ptr [[TMP15]], align 4
+; CHECK-NEXT:    br label %[[PRED_STORE_CONTINUE5]]
+; CHECK:       [[PRED_STORE_CONTINUE5]]:
+; CHECK-NEXT:    [[TMP17:%.*]] = extractelement <4 x i1> [[TMP5]], i64 3
+; CHECK-NEXT:    br i1 [[TMP17]], label %[[PRED_STORE_IF6:.*]], label %[[PRED_STORE_CONTINUE7]]
+; CHECK:       [[PRED_STORE_IF6]]:
+; CHECK-NEXT:    [[TMP18:%.*]] = add i64 [[INDEX]], 3
+; CHECK-NEXT:    [[TMP19:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[TMP18]]
+; CHECK-NEXT:    [[TMP20:%.*]] = extractelement <4 x i32> [[TMP4]], i64 3
+; CHECK-NEXT:    store i32 [[TMP20]], ptr [[TMP19]], align 4
+; CHECK-NEXT:    br label %[[PRED_STORE_CONTINUE7]]
+; CHECK:       [[PRED_STORE_CONTINUE7]]:
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT:    [[TMP21:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP21]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; CHECK:       [[SCALAR_PH]]:
+; CHECK-NEXT:    [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LATCH:.*]] ]
+; CHECK-NEXT:    [[A_IV:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT:    [[L1:%.*]] = load i32, ptr [[A_IV]], align 4
+; CHECK-NEXT:    [[IV_PLUS_4:%.*]] = add nuw nsw i64 [[IV]], 4
+; CHECK-NEXT:    [[A_IV_PLUS_4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV_PLUS_4]]
+; CHECK-NEXT:    [[L2:%.*]] = load i32, ptr [[A_IV_PLUS_4]], align 4
+; CHECK-NEXT:    [[SUM:%.*]] = add i32 [[L1]], [[L2]]
+; CHECK-NEXT:    [[PRED:%.*]] = icmp eq i32 [[SUM]], 0
+; CHECK-NEXT:    br i1 [[PRED]], label %[[STORE:.*]], label %[[LATCH]]
+; CHECK:       [[STORE]]:
+; CHECK-NEXT:    [[C_IV:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[IV]]
+; CHECK-NEXT:    store i32 [[SUM]], ptr [[C_IV]], align 4
+; CHECK-NEXT:    br label %[[LATCH]]
+; CHECK:       [[LATCH]]:
+; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT:    [[DONE:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[DONE]], label %[[EXIT]], label %[[LOOP]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %latch ]
+  %a.iv = getelementptr inbounds i32, ptr %a, i64 %iv
+  %l1 = load i32, ptr %a.iv, align 4
+  %iv.plus.4 = add nuw nsw i64 %iv, 4
+  %a.iv.plus.4 = getelementptr inbounds i32, ptr %a, i64 %iv.plus.4
+  %l2 = load i32, ptr %a.iv.plus.4, align 4
+  %sum = add i32 %l1, %l2
+  %pred = icmp eq i32 %sum, 0
+  br i1 %pred, label %store, label %latch
+
+store:
+  %c.iv = getelementptr inbounds i32, ptr %c, i64 %iv
+  store i32 %sum, ptr %c.iv, align 4
+  br label %latch
+
+latch:
+  %iv.next = add nuw nsw i64 %iv, 1
+  %done = icmp eq i64 %iv.next, %n
+  br i1 %done, label %exit, label %loop
+
+exit:
+  ret void
+}
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse.ll b/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse.ll
index f3b3cafd67b37..1775e9ea0cf73 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse.ll
@@ -1,63 +1,338 @@
 ; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
-; Pre-commit test for cross-part load redundancy. A widened load in one
-; logical part of an interleaved vector loop can cover exactly the elements
-; that another widened load covers in the next part. The interleave-count
-; heuristics do not model that redundancy yet, so this loop vectorizes with an
-; interleave count of 1.
+; The analysis only predicts a downstream saving in order to guide interleave
+; count selection. It never modifies the VPlan, so all widened loads remain.
+;
+; Vector width chosen by the cost model, cross-part analysis enabled:
+; RUN: opt -passes=loop-vectorize -force-target-max-vector-interleave=2 \
+; RUN:     -small-loop-cost=0 -enable-interleave-cse \
+; RUN:     -interleave-cse-min-pct=1 \
+; RUN:     -S %s | FileCheck %s --check-prefix=PRODUCTION
+;
+; Forced VF=4 with the default minimum saving percentage:
+; RUN: opt -passes=loop-vectorize -force-vector-width=4 \
+; RUN:     -force-target-max-vector-interleave=2 -small-loop-cost=0 \
+; RUN:     -enable-interleave-cse \
+; RUN:     -S %s | FileCheck %s --check-prefix=DEFAULT-PCT
+;
+; A minimum saving of 100% cannot be reached, so the interleave count stays 1:
+; RUN: opt -passes=loop-vectorize -force-vector-width=4 \
+; RUN:     -force-target-max-vector-interleave=2 -small-loop-cost=0 \
+; RUN:     -enable-interleave-cse -interleave-cse-min-pct=100 \
+; RUN:     -S %s | FileCheck %s --check-prefix=HIGH-PCT
+;
+; With the feature disabled the ordinary heuristics keep the interleave count:
+; RUN: opt -passes=loop-vectorize -force-vector-width=4 \
+; RUN:     -force-target-max-vector-interleave=2 -small-loop-cost=0 \
+; RUN:     -S %s | FileCheck %s --check-prefix=FEATURE-OFF
+;
+; An explicit false value must also dominate the saving threshold:
 ; RUN: opt -passes=loop-vectorize -force-vector-width=4 \
 ; RUN:     -force-target-max-vector-interleave=2 -small-loop-cost=0 \
-; RUN:     -S %s | FileCheck %s
+; RUN:     -enable-interleave-cse=false -interleave-cse-min-pct=0 \
+; RUN:     -S %s | FileCheck %s --check-prefix=FEATURE-OFF
+;
+; An explicit user interleave count must not be overridden:
+; RUN: opt -passes=loop-vectorize -force-vector-width=4 \
+; RUN:     -force-target-max-vector-interleave=2 -small-loop-cost=0 \
+; RUN:     -enable-interleave-cse -force-vector-interleave=1 \
+; RUN:     -S %s | FileCheck %s --check-prefix=USER-IC
+;
+; A target maximum interleave count of 1 must not be raised:
+; RUN: opt -passes=loop-vectorize -force-vector-width=4 \
+; RUN:     -force-target-max-vector-interleave=1 -small-loop-cost=0 \
+; RUN:     -enable-interleave-cse -interleave-cse-min-pct=1 \
+; RUN:     -S %s | FileCheck %s --check-prefix=MAX-IC
 
 target triple = "aarch64-unknown-linux-gnu"
 
 ; The motivating shape for cross-part load redundancy: with VF=4, the widened
-; load of a[i+4] in logical part 0 would cover exactly the same elements as the
-; widened load of a[i] in logical part 1.
+; load of a[i+4] in logical part 0 covers exactly the same elements as the
+; widened load of a[i] in logical part 1. The predicted saving raises the
+; interleave count to 2, while all four widened loads remain.
 define void @positive(ptr noalias %a, ptr noalias %c, i64 %n) {
-; CHECK-LABEL: define void @positive(
-; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) {
-; CHECK-NEXT:  [[ENTRY:.*]]:
-; CHECK-NEXT:    [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
-; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 4
-; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
-; CHECK:       [[VECTOR_PH]]:
-; CHECK-NEXT:    [[TMP1:%.*]] = and i64 [[TMP0]], 3
-; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
-; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
-; CHECK:       [[VECTOR_BODY]]:
-; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT:    [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
-; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4
-; CHECK-NEXT:    [[TMP3:%.*]] = add nuw nsw i64 [[INDEX]], 4
-; CHECK-NEXT:    [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP3]]
-; CHECK-NEXT:    [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP4]], align 4
-; CHECK-NEXT:    [[TMP5:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[WIDE_LOAD1]]
-; CHECK-NEXT:    [[TMP6:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
-; CHECK-NEXT:    store <4 x i32> [[TMP5]], ptr [[TMP6]], align 4
-; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
-; CHECK-NEXT:    [[TMP7:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-NEXT:    br i1 [[TMP7]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
-; CHECK:       [[MIDDLE_BLOCK]]:
-; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
-; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
-; CHECK:       [[SCALAR_PH]]:
-; CHECK-NEXT:    [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
-; CHECK-NEXT:    br label %[[LOOP:.*]]
-; CHECK:       [[LOOP]]:
-; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
-; CHECK-NEXT:    [[A_IV:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
-; CHECK-NEXT:    [[L1:%.*]] = load i32, ptr [[A_IV]], align 4
-; CHECK-NEXT:    [[IV_PLUS_4:%.*]] = add nuw nsw i64 [[IV]], 4
-; CHECK-NEXT:    [[A_IV_PLUS_4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV_PLUS_4]]
-; CHECK-NEXT:    [[L2:%.*]] = load i32, ptr [[A_IV_PLUS_4]], align 4
-; CHECK-NEXT:    [[SUM:%.*]] = add i32 [[L1]], [[L2]]
-; CHECK-NEXT:    [[C_IV:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[IV]]
-; CHECK-NEXT:    store i32 [[SUM]], ptr [[C_IV]], align 4
-; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
-; CHECK-NEXT:    [[CMP:%.*]] = icmp slt i64 [[IV_NEXT]], [[N]]
-; CHECK-NEXT:    br i1 [[CMP]], label %[[LOOP]], label %[[EXIT]], !llvm.loop [[LOOP3:![0-9]+]]
-; CHECK:       [[EXIT]]:
-; CHECK-NEXT:    ret void
+; PRODUCTION-LABEL: define void @positive(
+; PRODUCTION-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) {
+; PRODUCTION-NEXT:  [[ENTRY:.*]]:
+; PRODUCTION-NEXT:    [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; PRODUCTION-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 8
+; PRODUCTION-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; PRODUCTION:       [[VECTOR_PH]]:
+; PRODUCTION-NEXT:    [[TMP1:%.*]] = and i64 [[TMP0]], 7
+; PRODUCTION-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
+; PRODUCTION-NEXT:    br label %[[VECTOR_BODY:.*]]
+; PRODUCTION:       [[VECTOR_BODY]]:
+; PRODUCTION-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; PRODUCTION-NEXT:    [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; PRODUCTION-NEXT:    [[TMP3:%.*]] = getelementptr inbounds i32, ptr [[TMP2]], i64 4
+; PRODUCTION-NEXT:    [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4
+; PRODUCTION-NEXT:    [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP3]], align 4
+; PRODUCTION-NEXT:    [[TMP4:%.*]] = add nuw nsw i64 [[INDEX]], 4
+; PRODUCTION-NEXT:    [[TMP5:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP4]]
+; PRODUCTION-NEXT:    [[TMP6:%.*]] = getelementptr inbounds i32, ptr [[TMP5]], i64 4
+; PRODUCTION-NEXT:    [[WIDE_LOAD2:%.*]] = load <4 x i32>, ptr [[TMP5]], align 4
+; PRODUCTION-NEXT:    [[WIDE_LOAD3:%.*]] = load <4 x i32>, ptr [[TMP6]], align 4
+; PRODUCTION-NEXT:    [[TMP7:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[WIDE_LOAD2]]
+; PRODUCTION-NEXT:    [[TMP8:%.*]] = add <4 x i32> [[WIDE_LOAD1]], [[WIDE_LOAD3]]
+; PRODUCTION-NEXT:    [[TMP9:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
+; PRODUCTION-NEXT:    [[TMP10:%.*]] = getelementptr inbounds i32, ptr [[TMP9]], i64 4
+; PRODUCTION-NEXT:    store <4 x i32> [[TMP7]], ptr [[TMP9]], align 4
+; PRODUCTION-NEXT:    store <4 x i32> [[TMP8]], ptr [[TMP10]], align 4
+; PRODUCTION-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
+; PRODUCTION-NEXT:    [[TMP11:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; PRODUCTION-NEXT:    br i1 [[TMP11]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; PRODUCTION:       [[MIDDLE_BLOCK]]:
+; PRODUCTION-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
+; PRODUCTION-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; PRODUCTION:       [[SCALAR_PH]]:
+; PRODUCTION-NEXT:    [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; PRODUCTION-NEXT:    br label %[[LOOP:.*]]
+; PRODUCTION:       [[LOOP]]:
+; PRODUCTION-NEXT:    [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; PRODUCTION-NEXT:    [[A_IV:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
+; PRODUCTION-NEXT:    [[L1:%.*]] = load i32, ptr [[A_IV]], align 4
+; PRODUCTION-NEXT:    [[IV_PLUS_4:%.*]] = add nuw nsw i64 [[IV]], 4
+; PRODUCTION-NEXT:    [[A_IV_PLUS_4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV_PLUS_4]]
+; PRODUCTION-NEXT:    [[L2:%.*]] = load i32, ptr [[A_IV_PLUS_4]], align 4
+; PRODUCTION-NEXT:    [[SUM:%.*]] = add i32 [[L1]], [[L2]]
+; PRODUCTION-NEXT:    [[C_IV:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[IV]]
+; PRODUCTION-NEXT:    store i32 [[SUM]], ptr [[C_IV]], align 4
+; PRODUCTION-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; PRODUCTION-NEXT:    [[CMP:%.*]] = icmp slt i64 [[IV_NEXT]], [[N]]
+; PRODUCTION-NEXT:    br i1 [[CMP]], label %[[LOOP]], label %[[EXIT]], !llvm.loop [[LOOP3:![0-9]+]]
+; PRODUCTION:       [[EXIT]]:
+; PRODUCTION-NEXT:    ret void
+;
+; DEFAULT-PCT-LABEL: define void @positive(
+; DEFAULT-PCT-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) {
+; DEFAULT-PCT-NEXT:  [[ENTRY:.*]]:
+; DEFAULT-PCT-NEXT:    [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; DEFAULT-PCT-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 8
+; DEFAULT-PCT-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; DEFAULT-PCT:       [[VECTOR_PH]]:
+; DEFAULT-PCT-NEXT:    [[TMP1:%.*]] = and i64 [[TMP0]], 7
+; DEFAULT-PCT-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
+; DEFAULT-PCT-NEXT:    br label %[[VECTOR_BODY:.*]]
+; DEFAULT-PCT:       [[VECTOR_BODY]]:
+; DEFAULT-PCT-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; DEFAULT-PCT-NEXT:    [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; DEFAULT-PCT-NEXT:    [[TMP3:%.*]] = getelementptr inbounds i32, ptr [[TMP2]], i64 4
+; DEFAULT-PCT-NEXT:    [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4
+; DEFAULT-PCT-NEXT:    [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP3]], align 4
+; DEFAULT-PCT-NEXT:    [[TMP4:%.*]] = add nuw nsw i64 [[INDEX]], 4
+; DEFAULT-PCT-NEXT:    [[TMP5:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP4]]
+; DEFAULT-PCT-NEXT:    [[TMP6:%.*]] = getelementptr inbounds i32, ptr [[TMP5]], i64 4
+; DEFAULT-PCT-NEXT:    [[WIDE_LOAD2:%.*]] = load <4 x i32>, ptr [[TMP5]], align 4
+; DEFAULT-PCT-NEXT:    [[WIDE_LOAD3:%.*]] = load <4 x i32>, ptr [[TMP6]], align 4
+; DEFAULT-PCT-NEXT:    [[TMP7:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[WIDE_LOAD2]]
+; DEFAULT-PCT-NEXT:    [[TMP8:%.*]] = add <4 x i32> [[WIDE_LOAD1]], [[WIDE_LOAD3]]
+; DEFAULT-PCT-NEXT:    [[TMP9:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
+; DEFAULT-PCT-NEXT:    [[TMP10:%.*]] = getelementptr inbounds i32, ptr [[TMP9]], i64 4
+; DEFAULT-PCT-NEXT:    store <4 x i32> [[TMP7]], ptr [[TMP9]], align 4
+; DEFAULT-PCT-NEXT:    store <4 x i32> [[TMP8]], ptr [[TMP10]], align 4
+; DEFAULT-PCT-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
+; DEFAULT-PCT-NEXT:    [[TMP11:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; DEFAULT-PCT-NEXT:    br i1 [[TMP11]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; DEFAULT-PCT:       [[MIDDLE_BLOCK]]:
+; DEFAULT-PCT-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
+; DEFAULT-PCT-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; DEFAULT-PCT:       [[SCALAR_PH]]:
+; DEFAULT-PCT-NEXT:    [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; DEFAULT-PCT-NEXT:    br label %[[LOOP:.*]]
+; DEFAULT-PCT:       [[LOOP]]:
+; DEFAULT-PCT-NEXT:    [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; DEFAULT-PCT-NEXT:    [[A_IV:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
+; DEFAULT-PCT-NEXT:    [[L1:%.*]] = load i32, ptr [[A_IV]], align 4
+; DEFAULT-PCT-NEXT:    [[IV_PLUS_4:%.*]] = add nuw nsw i64 [[IV]], 4
+; DEFAULT-PCT-NEXT:    [[A_IV_PLUS_4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV_PLUS_4]]
+; DEFAULT-PCT-NEXT:    [[L2:%.*]] = load i32, ptr [[A_IV_PLUS_4]], align 4
+; DEFAULT-PCT-NEXT:    [[SUM:%.*]] = add i32 [[L1]], [[L2]]
+; DEFAULT-PCT-NEXT:    [[C_IV:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[IV]]
+; DEFAULT-PCT-NEXT:    store i32 [[SUM]], ptr [[C_IV]], align 4
+; DEFAULT-PCT-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; DEFAULT-PCT-NEXT:    [[CMP:%.*]] = icmp slt i64 [[IV_NEXT]], [[N]]
+; DEFAULT-PCT-NEXT:    br i1 [[CMP]], label %[[LOOP]], label %[[EXIT]], !llvm.loop [[LOOP3:![0-9]+]]
+; DEFAULT-PCT:       [[EXIT]]:
+; DEFAULT-PCT-NEXT:    ret void
+;
+; HIGH-PCT-LABEL: define void @positive(
+; HIGH-PCT-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) {
+; HIGH-PCT-NEXT:  [[ENTRY:.*]]:
+; HIGH-PCT-NEXT:    [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; HIGH-PCT-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 4
+; HIGH-PCT-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; HIGH-PCT:       [[VECTOR_PH]]:
+; HIGH-PCT-NEXT:    [[TMP1:%.*]] = and i64 [[TMP0]], 3
+; HIGH-PCT-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
+; HIGH-PCT-NEXT:    br label %[[VECTOR_BODY:.*]]
+; HIGH-PCT:       [[VECTOR_BODY]]:
+; HIGH-PCT-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; HIGH-PCT-NEXT:    [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; HIGH-PCT-NEXT:    [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4
+; HIGH-PCT-NEXT:    [[TMP3:%.*]] = add nuw nsw i64 [[INDEX]], 4
+; HIGH-PCT-NEXT:    [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP3]]
+; HIGH-PCT-NEXT:    [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP4]], align 4
+; HIGH-PCT-NEXT:    [[TMP5:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[WIDE_LOAD1]]
+; HIGH-PCT-NEXT:    [[TMP6:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
+; HIGH-PCT-NEXT:    store <4 x i32> [[TMP5]], ptr [[TMP6]], align 4
+; HIGH-PCT-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; HIGH-PCT-NEXT:    [[TMP7:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; HIGH-PCT-NEXT:    br i1 [[TMP7]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; HIGH-PCT:       [[MIDDLE_BLOCK]]:
+; HIGH-PCT-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
+; HIGH-PCT-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; HIGH-PCT:       [[SCALAR_PH]]:
+; HIGH-PCT-NEXT:    [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; HIGH-PCT-NEXT:    br label %[[LOOP:.*]]
+; HIGH-PCT:       [[LOOP]]:
+; HIGH-PCT-NEXT:    [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; HIGH-PCT-NEXT:    [[A_IV:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
+; HIGH-PCT-NEXT:    [[L1:%.*]] = load i32, ptr [[A_IV]], align 4
+; HIGH-PCT-NEXT:    [[IV_PLUS_4:%.*]] = add nuw nsw i64 [[IV]], 4
+; HIGH-PCT-NEXT:    [[A_IV_PLUS_4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV_PLUS_4]]
+; HIGH-PCT-NEXT:    [[L2:%.*]] = load i32, ptr [[A_IV_PLUS_4]], align 4
+; HIGH-PCT-NEXT:    [[SUM:%.*]] = add i32 [[L1]], [[L2]]
+; HIGH-PCT-NEXT:    [[C_IV:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[IV]]
+; HIGH-PCT-NEXT:    store i32 [[SUM]], ptr [[C_IV]], align 4
+; HIGH-PCT-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; HIGH-PCT-NEXT:    [[CMP:%.*]] = icmp slt i64 [[IV_NEXT]], [[N]]
+; HIGH-PCT-NEXT:    br i1 [[CMP]], label %[[LOOP]], label %[[EXIT]], !llvm.loop [[LOOP3:![0-9]+]]
+; HIGH-PCT:       [[EXIT]]:
+; HIGH-PCT-NEXT:    ret void
+;
+; FEATURE-OFF-LABEL: define void @positive(
+; FEATURE-OFF-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) {
+; FEATURE-OFF-NEXT:  [[ENTRY:.*]]:
+; FEATURE-OFF-NEXT:    [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; FEATURE-OFF-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 4
+; FEATURE-OFF-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; FEATURE-OFF:       [[VECTOR_PH]]:
+; FEATURE-OFF-NEXT:    [[TMP1:%.*]] = and i64 [[TMP0]], 3
+; FEATURE-OFF-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
+; FEATURE-OFF-NEXT:    br label %[[VECTOR_BODY:.*]]
+; FEATURE-OFF:       [[VECTOR_BODY]]:
+; FEATURE-OFF-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; FEATURE-OFF-NEXT:    [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; FEATURE-OFF-NEXT:    [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4
+; FEATURE-OFF-NEXT:    [[TMP3:%.*]] = add nuw nsw i64 [[INDEX]], 4
+; FEATURE-OFF-NEXT:    [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP3]]
+; FEATURE-OFF-NEXT:    [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP4]], align 4
+; FEATURE-OFF-NEXT:    [[TMP5:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[WIDE_LOAD1]]
+; FEATURE-OFF-NEXT:    [[TMP6:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
+; FEATURE-OFF-NEXT:    store <4 x i32> [[TMP5]], ptr [[TMP6]], align 4
+; FEATURE-OFF-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; FEATURE-OFF-NEXT:    [[TMP7:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; FEATURE-OFF-NEXT:    br i1 [[TMP7]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; FEATURE-OFF:       [[MIDDLE_BLOCK]]:
+; FEATURE-OFF-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
+; FEATURE-OFF-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; FEATURE-OFF:       [[SCALAR_PH]]:
+; FEATURE-OFF-NEXT:    [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; FEATURE-OFF-NEXT:    br label %[[LOOP:.*]]
+; FEATURE-OFF:       [[LOOP]]:
+; FEATURE-OFF-NEXT:    [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; FEATURE-OFF-NEXT:    [[A_IV:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
+; FEATURE-OFF-NEXT:    [[L1:%.*]] = load i32, ptr [[A_IV]], align 4
+; FEATURE-OFF-NEXT:    [[IV_PLUS_4:%.*]] = add nuw nsw i64 [[IV]], 4
+; FEATURE-OFF-NEXT:    [[A_IV_PLUS_4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV_PLUS_4]]
+; FEATURE-OFF-NEXT:    [[L2:%.*]] = load i32, ptr [[A_IV_PLUS_4]], align 4
+; FEATURE-OFF-NEXT:    [[SUM:%.*]] = add i32 [[L1]], [[L2]]
+; FEATURE-OFF-NEXT:    [[C_IV:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[IV]]
+; FEATURE-OFF-NEXT:    store i32 [[SUM]], ptr [[C_IV]], align 4
+; FEATURE-OFF-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; FEATURE-OFF-NEXT:    [[CMP:%.*]] = icmp slt i64 [[IV_NEXT]], [[N]]
+; FEATURE-OFF-NEXT:    br i1 [[CMP]], label %[[LOOP]], label %[[EXIT]], !llvm.loop [[LOOP3:![0-9]+]]
+; FEATURE-OFF:       [[EXIT]]:
+; FEATURE-OFF-NEXT:    ret void
+;
+; USER-IC-LABEL: define void @positive(
+; USER-IC-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) {
+; USER-IC-NEXT:  [[ENTRY:.*]]:
+; USER-IC-NEXT:    [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; USER-IC-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 4
+; USER-IC-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; USER-IC:       [[VECTOR_PH]]:
+; USER-IC-NEXT:    [[TMP1:%.*]] = and i64 [[TMP0]], 3
+; USER-IC-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
+; USER-IC-NEXT:    br label %[[VECTOR_BODY:.*]]
+; USER-IC:       [[VECTOR_BODY]]:
+; USER-IC-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; USER-IC-NEXT:    [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; USER-IC-NEXT:    [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4
+; USER-IC-NEXT:    [[TMP3:%.*]] = add nuw nsw i64 [[INDEX]], 4
+; USER-IC-NEXT:    [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP3]]
+; USER-IC-NEXT:    [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP4]], align 4
+; USER-IC-NEXT:    [[TMP5:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[WIDE_LOAD1]]
+; USER-IC-NEXT:    [[TMP6:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
+; USER-IC-NEXT:    store <4 x i32> [[TMP5]], ptr [[TMP6]], align 4
+; USER-IC-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; USER-IC-NEXT:    [[TMP7:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; USER-IC-NEXT:    br i1 [[TMP7]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; USER-IC:       [[MIDDLE_BLOCK]]:
+; USER-IC-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
+; USER-IC-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; USER-IC:       [[SCALAR_PH]]:
+; USER-IC-NEXT:    [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; USER-IC-NEXT:    br label %[[LOOP:.*]]
+; USER-IC:       [[LOOP]]:
+; USER-IC-NEXT:    [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; USER-IC-NEXT:    [[A_IV:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
+; USER-IC-NEXT:    [[L1:%.*]] = load i32, ptr [[A_IV]], align 4
+; USER-IC-NEXT:    [[IV_PLUS_4:%.*]] = add nuw nsw i64 [[IV]], 4
+; USER-IC-NEXT:    [[A_IV_PLUS_4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV_PLUS_4]]
+; USER-IC-NEXT:    [[L2:%.*]] = load i32, ptr [[A_IV_PLUS_4]], align 4
+; USER-IC-NEXT:    [[SUM:%.*]] = add i32 [[L1]], [[L2]]
+; USER-IC-NEXT:    [[C_IV:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[IV]]
+; USER-IC-NEXT:    store i32 [[SUM]], ptr [[C_IV]], align 4
+; USER-IC-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; USER-IC-NEXT:    [[CMP:%.*]] = icmp slt i64 [[IV_NEXT]], [[N]]
+; USER-IC-NEXT:    br i1 [[CMP]], label %[[LOOP]], label %[[EXIT]], !llvm.loop [[LOOP3:![0-9]+]]
+; USER-IC:       [[EXIT]]:
+; USER-IC-NEXT:    ret void
+;
+; MAX-IC-LABEL: define void @positive(
+; MAX-IC-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) {
+; MAX-IC-NEXT:  [[ENTRY:.*]]:
+; MAX-IC-NEXT:    [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; MAX-IC-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 4
+; MAX-IC-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; MAX-IC:       [[VECTOR_PH]]:
+; MAX-IC-NEXT:    [[TMP1:%.*]] = and i64 [[TMP0]], 3
+; MAX-IC-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
+; MAX-IC-NEXT:    br label %[[VECTOR_BODY:.*]]
+; MAX-IC:       [[VECTOR_BODY]]:
+; MAX-IC-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; MAX-IC-NEXT:    [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; MAX-IC-NEXT:    [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4
+; MAX-IC-NEXT:    [[TMP3:%.*]] = add nuw nsw i64 [[INDEX]], 4
+; MAX-IC-NEXT:    [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP3]]
+; MAX-IC-NEXT:    [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP4]], align 4
+; MAX-IC-NEXT:    [[TMP5:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[WIDE_LOAD1]]
+; MAX-IC-NEXT:    [[TMP6:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
+; MAX-IC-NEXT:    store <4 x i32> [[TMP5]], ptr [[TMP6]], align 4
+; MAX-IC-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; MAX-IC-NEXT:    [[TMP7:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; MAX-IC-NEXT:    br i1 [[TMP7]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; MAX-IC:       [[MIDDLE_BLOCK]]:
+; MAX-IC-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
+; MAX-IC-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; MAX-IC:       [[SCALAR_PH]]:
+; MAX-IC-NEXT:    [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; MAX-IC-NEXT:    br label %[[LOOP:.*]]
+; MAX-IC:       [[LOOP]]:
+; MAX-IC-NEXT:    [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; MAX-IC-NEXT:    [[A_IV:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
+; MAX-IC-NEXT:    [[L1:%.*]] = load i32, ptr [[A_IV]], align 4
+; MAX-IC-NEXT:    [[IV_PLUS_4:%.*]] = add nuw nsw i64 [[IV]], 4
+; MAX-IC-NEXT:    [[A_IV_PLUS_4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV_PLUS_4]]
+; MAX-IC-NEXT:    [[L2:%.*]] = load i32, ptr [[A_IV_PLUS_4]], align 4
+; MAX-IC-NEXT:    [[SUM:%.*]] = add i32 [[L1]], [[L2]]
+; MAX-IC-NEXT:    [[C_IV:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[IV]]
+; MAX-IC-NEXT:    store i32 [[SUM]], ptr [[C_IV]], align 4
+; MAX-IC-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; MAX-IC-NEXT:    [[CMP:%.*]] = icmp slt i64 [[IV_NEXT]], [[N]]
+; MAX-IC-NEXT:    br i1 [[CMP]], label %[[LOOP]], label %[[EXIT]], !llvm.loop [[LOOP3:![0-9]+]]
+; MAX-IC:       [[EXIT]]:
+; MAX-IC-NEXT:    ret void
 ;
 entry:
   br label %loop
diff --git a/llvm/utils/gn/secondary/llvm/lib/Transforms/Vectorize/BUILD.gn b/llvm/utils/gn/secondary/llvm/lib/Transforms/Vectorize/BUILD.gn
index 0ffc24ec855f7..6ceb8ce67b207 100644
--- a/llvm/utils/gn/secondary/llvm/lib/Transforms/Vectorize/BUILD.gn
+++ b/llvm/utils/gn/secondary/llvm/lib/Transforms/Vectorize/BUILD.gn
@@ -42,6 +42,7 @@ static_library("Vectorize") {
     "VPlan.cpp",
     "VPlanAnalysis.cpp",
     "VPlanConstruction.cpp",
+    "VPlanCrossPartCSE.cpp",
     "VPlanDominatorTree.cpp",
     "VPlanEVLTailFolding.cpp",
     "VPlanLowering.cpp",



More information about the llvm-commits mailing list