[llvm] [LV] Add cross-part load overlap heuristic for interleaving (PR #214500)
Sergey Shcherbinin via llvm-commits
llvm-commits at lists.llvm.org
Mon Sep 21 06:33:08 PDT 2026
https://github.com/SergeyShch01 updated https://github.com/llvm/llvm-project/pull/214500
>From d08f5cf781b3fb9d4f84ffd045a50b5a62227c7f Mon Sep 17 00:00:00 2001
From: Sergey Shcherbinin <sscherbinin at nvidia.com>
Date: Sat, 19 Sep 2026 23:10:56 -0700
Subject: [PATCH 1/2] [LV] Pre-commit cross-part load redundancy tests
Add fixtures that record how the loop vectorizer currently vectorizes loops in
which a widened load of one logical part of an interleaved vector loop covers
exactly the elements that another widened load covers in the next part. The
interleave-count heuristics do not model that redundancy, so every loop here
vectorizes with an interleave count of 1.
---
.../cross-part-load-cse-fixed-cases.ll | 776 ++++++++++++++++++
.../cross-part-load-cse-opportunities.ll | 167 ++++
.../AArch64/cross-part-load-cse-scalable.ll | 158 ++++
.../AArch64/cross-part-load-cse.ll | 81 ++
4 files changed, 1182 insertions(+)
create mode 100644 llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-fixed-cases.ll
create mode 100644 llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-opportunities.ll
create mode 100644 llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-scalable.ll
create mode 100644 llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse.ll
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-fixed-cases.ll b/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-fixed-cases.ll
new file mode 100644
index 0000000000000..15948347005af
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-fixed-cases.ll
@@ -0,0 +1,776 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
+; Pre-commit test for the properties that a cross-part load redundancy
+; analysis has to respect: loaded value type, exact address equality,
+; intervening memory writes, access stride, address provenance,
+; poison-generating metadata, multiple scalar loop blocks and reverse
+; accesses. Every loop below vectorizes with an interleave count of 1 today.
+; RUN: opt -passes=loop-vectorize -force-vector-width=4 \
+; RUN: -force-target-max-vector-interleave=2 -small-loop-cost=0 \
+; RUN: -S %s | FileCheck %s
+
+target triple = "aarch64-unknown-linux-gnu"
+
+declare i32 @write_memory(i32) #0
+declare <4 x i32> @write_memory_v4(<4 x i32>)
+
+; The part-shifted addresses are equal, but the two loads have different value
+; types and therefore cannot share a loaded result.
+define void @different_types(ptr noalias %a, ptr noalias %c, i64 %n) {
+; CHECK-LABEL: define void @different_types(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 4
+; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: [[TMP1:%.*]] = and i64 [[TMP0]], 3
+; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4
+; CHECK-NEXT: [[TMP3:%.*]] = add nuw nsw i64 [[INDEX]], 4
+; CHECK-NEXT: [[TMP4:%.*]] = getelementptr inbounds float, ptr [[A]], i64 [[TMP3]]
+; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <4 x float>, ptr [[TMP4]], align 4
+; CHECK-NEXT: [[TMP5:%.*]] = bitcast <4 x float> [[WIDE_LOAD1]] to <4 x i32>
+; CHECK-NEXT: [[TMP6:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[TMP5]]
+; CHECK-NEXT: [[TMP7:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
+; CHECK-NEXT: store <4 x i32> [[TMP6]], ptr [[TMP7]], align 4
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT: [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[A_IV:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT: [[L1:%.*]] = load i32, ptr [[A_IV]], align 4
+; CHECK-NEXT: [[IV_PLUS_4:%.*]] = add nuw nsw i64 [[IV]], 4
+; CHECK-NEXT: [[A_IV_PLUS_4:%.*]] = getelementptr inbounds float, ptr [[A]], i64 [[IV_PLUS_4]]
+; CHECK-NEXT: [[L2:%.*]] = load float, ptr [[A_IV_PLUS_4]], align 4
+; CHECK-NEXT: [[L2_BITS:%.*]] = bitcast float [[L2]] to i32
+; CHECK-NEXT: [[SUM:%.*]] = add i32 [[L1]], [[L2_BITS]]
+; CHECK-NEXT: [[C_IV:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[IV]]
+; CHECK-NEXT: store i32 [[SUM]], ptr [[C_IV]], align 4
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[CMP:%.*]] = icmp slt i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT: br i1 [[CMP]], label %[[LOOP]], label %[[EXIT]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %a.iv = getelementptr inbounds i32, ptr %a, i64 %iv
+ %l1 = load i32, ptr %a.iv, align 4
+ %iv.plus.4 = add nuw nsw i64 %iv, 4
+ %a.iv.plus.4 = getelementptr inbounds float, ptr %a, i64 %iv.plus.4
+ %l2 = load float, ptr %a.iv.plus.4, align 4
+ %l2.bits = bitcast float %l2 to i32
+ %sum = add i32 %l1, %l2.bits
+ %c.iv = getelementptr inbounds i32, ptr %c, i64 %iv
+ store i32 %sum, ptr %c.iv, align 4
+ %iv.next = add nuw nsw i64 %iv, 1
+ %cmp = icmp slt i64 %iv.next, %n
+ br i1 %cmp, label %loop, label %exit
+
+exit:
+ ret void
+}
+
+; a[i] + a[i+3]: the source offset 3 is not a multiple of VF=4, so no
+; part-shifted address ever matches exactly.
+define void @inequality(ptr noalias %a, ptr noalias %c, i64 %n) {
+; CHECK-LABEL: define void @inequality(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 4
+; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: [[TMP1:%.*]] = and i64 [[TMP0]], 3
+; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4
+; CHECK-NEXT: [[TMP3:%.*]] = add nuw nsw i64 [[INDEX]], 3
+; CHECK-NEXT: [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP3]]
+; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP4]], align 4
+; CHECK-NEXT: [[TMP5:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[WIDE_LOAD1]]
+; CHECK-NEXT: [[TMP6:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
+; CHECK-NEXT: store <4 x i32> [[TMP5]], ptr [[TMP6]], align 4
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT: [[TMP7:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[TMP7]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[A_IV:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT: [[L1:%.*]] = load i32, ptr [[A_IV]], align 4
+; CHECK-NEXT: [[IV_PLUS_3:%.*]] = add nuw nsw i64 [[IV]], 3
+; CHECK-NEXT: [[A_IV_PLUS_3:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV_PLUS_3]]
+; CHECK-NEXT: [[L2:%.*]] = load i32, ptr [[A_IV_PLUS_3]], align 4
+; CHECK-NEXT: [[SUM:%.*]] = add i32 [[L1]], [[L2]]
+; CHECK-NEXT: [[C_IV:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[IV]]
+; CHECK-NEXT: store i32 [[SUM]], ptr [[C_IV]], align 4
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[CMP:%.*]] = icmp slt i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT: br i1 [[CMP]], label %[[LOOP]], label %[[EXIT]], !llvm.loop [[LOOP5:![0-9]+]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %a.iv = getelementptr inbounds i32, ptr %a, i64 %iv
+ %l1 = load i32, ptr %a.iv, align 4
+ %iv.plus.3 = add nuw nsw i64 %iv, 3
+ %a.iv.plus.3 = getelementptr inbounds i32, ptr %a, i64 %iv.plus.3
+ %l2 = load i32, ptr %a.iv.plus.3, align 4
+ %sum = add i32 %l1, %l2
+ %c.iv = getelementptr inbounds i32, ptr %c, i64 %iv
+ store i32 %sum, ptr %c.iv, align 4
+ %iv.next = add nuw nsw i64 %iv, 1
+ %cmp = icmp slt i64 %iv.next, %n
+ br i1 %cmp, label %loop, label %exit
+
+exit:
+ ret void
+}
+
+; A may-alias store occurs between the two matching logical-part loads, so their
+; values cannot be assumed to survive from one part to the other.
+define void @write_between(ptr %a, ptr %b, ptr noalias %c, i64 %n) {
+; CHECK-LABEL: define void @write_between(
+; CHECK-SAME: ptr [[A:%.*]], ptr [[B:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 4
+; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_MEMCHECK:.*]]
+; CHECK: [[VECTOR_MEMCHECK]]:
+; CHECK-NEXT: [[SMAX:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; CHECK-NEXT: [[TMP1:%.*]] = shl i64 [[SMAX]], 2
+; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[B]], i64 [[TMP1]]
+; CHECK-NEXT: [[TMP2:%.*]] = add i64 [[TMP1]], 16
+; CHECK-NEXT: [[SCEVGEP1:%.*]] = getelementptr i8, ptr [[A]], i64 [[TMP2]]
+; CHECK-NEXT: [[BOUND0:%.*]] = icmp ult ptr [[B]], [[SCEVGEP1]]
+; CHECK-NEXT: [[BOUND1:%.*]] = icmp ult ptr [[A]], [[SCEVGEP]]
+; CHECK-NEXT: [[FOUND_CONFLICT:%.*]] = and i1 [[BOUND0]], [[BOUND1]]
+; CHECK-NEXT: br i1 [[FOUND_CONFLICT]], label %[[SCALAR_PH]], label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: [[TMP3:%.*]] = and i64 [[TMP0]], 3
+; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP3]]
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP4]], align 4, !alias.scope [[META6:![0-9]+]]
+; CHECK-NEXT: [[TMP5:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 [[INDEX]]
+; CHECK-NEXT: store <4 x i32> [[WIDE_LOAD]], ptr [[TMP5]], align 4, !alias.scope [[META9:![0-9]+]], !noalias [[META6]]
+; CHECK-NEXT: [[TMP6:%.*]] = add nuw nsw i64 [[INDEX]], 4
+; CHECK-NEXT: [[TMP7:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP6]]
+; CHECK-NEXT: [[WIDE_LOAD2:%.*]] = load <4 x i32>, ptr [[TMP7]], align 4, !alias.scope [[META6]]
+; CHECK-NEXT: [[TMP8:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[WIDE_LOAD2]]
+; CHECK-NEXT: [[TMP9:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
+; CHECK-NEXT: store <4 x i32> [[TMP8]], ptr [[TMP9]], align 4
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT: [[TMP10:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[TMP10]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP11:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ], [ 0, %[[VECTOR_MEMCHECK]] ]
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[A_IV:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT: [[L1:%.*]] = load i32, ptr [[A_IV]], align 4
+; CHECK-NEXT: [[B_IV:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 [[IV]]
+; CHECK-NEXT: store i32 [[L1]], ptr [[B_IV]], align 4
+; CHECK-NEXT: [[IV_PLUS_4:%.*]] = add nuw nsw i64 [[IV]], 4
+; CHECK-NEXT: [[A_IV_PLUS_4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV_PLUS_4]]
+; CHECK-NEXT: [[L2:%.*]] = load i32, ptr [[A_IV_PLUS_4]], align 4
+; CHECK-NEXT: [[SUM:%.*]] = add i32 [[L1]], [[L2]]
+; CHECK-NEXT: [[C_IV:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[IV]]
+; CHECK-NEXT: store i32 [[SUM]], ptr [[C_IV]], align 4
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[CMP:%.*]] = icmp slt i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT: br i1 [[CMP]], label %[[LOOP]], label %[[EXIT]], !llvm.loop [[LOOP12:![0-9]+]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %a.iv = getelementptr inbounds i32, ptr %a, i64 %iv
+ %l1 = load i32, ptr %a.iv, align 4
+ %b.iv = getelementptr inbounds i32, ptr %b, i64 %iv
+ store i32 %l1, ptr %b.iv, align 4
+ %iv.plus.4 = add nuw nsw i64 %iv, 4
+ %a.iv.plus.4 = getelementptr inbounds i32, ptr %a, i64 %iv.plus.4
+ %l2 = load i32, ptr %a.iv.plus.4, align 4
+ %sum = add i32 %l1, %l2
+ %c.iv = getelementptr inbounds i32, ptr %c, i64 %iv
+ store i32 %sum, ptr %c.iv, align 4
+ %iv.next = add nuw nsw i64 %iv, 1
+ %cmp = icmp slt i64 %iv.next, %n
+ br i1 %cmp, label %loop, label %exit
+
+exit:
+ ret void
+}
+
+; A vector-mapped call with unknown memory effects occurs between the matching
+; loads. It may write through memory that its arguments do not describe, so no
+; loaded value may be assumed to survive across it.
+define void @writing_call_between(ptr noalias %a, ptr noalias %c, i64 %n) {
+; CHECK-LABEL: define void @writing_call_between(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 4
+; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: [[TMP1:%.*]] = and i64 [[TMP0]], 3
+; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4
+; CHECK-NEXT: [[TMP3:%.*]] = call <4 x i32> @write_memory_v4(<4 x i32> [[WIDE_LOAD]])
+; CHECK-NEXT: [[TMP4:%.*]] = add nuw nsw i64 [[INDEX]], 4
+; CHECK-NEXT: [[TMP5:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP4]]
+; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP5]], align 4
+; CHECK-NEXT: [[TMP6:%.*]] = add <4 x i32> [[TMP3]], [[WIDE_LOAD1]]
+; CHECK-NEXT: [[TMP7:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
+; CHECK-NEXT: store <4 x i32> [[TMP6]], ptr [[TMP7]], align 4
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT: [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP13:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[A_IV:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT: [[L1:%.*]] = load i32, ptr [[A_IV]], align 4
+; CHECK-NEXT: [[CALL:%.*]] = call i32 @write_memory(i32 [[L1]])
+; CHECK-NEXT: [[IV_PLUS_4:%.*]] = add nuw nsw i64 [[IV]], 4
+; CHECK-NEXT: [[A_IV_PLUS_4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV_PLUS_4]]
+; CHECK-NEXT: [[L2:%.*]] = load i32, ptr [[A_IV_PLUS_4]], align 4
+; CHECK-NEXT: [[SUM:%.*]] = add i32 [[CALL]], [[L2]]
+; CHECK-NEXT: [[C_IV:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[IV]]
+; CHECK-NEXT: store i32 [[SUM]], ptr [[C_IV]], align 4
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[CMP:%.*]] = icmp slt i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT: br i1 [[CMP]], label %[[LOOP]], label %[[EXIT]], !llvm.loop [[LOOP14:![0-9]+]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %a.iv = getelementptr inbounds i32, ptr %a, i64 %iv
+ %l1 = load i32, ptr %a.iv, align 4
+ %call = call i32 @write_memory(i32 %l1)
+ %iv.plus.4 = add nuw nsw i64 %iv, 4
+ %a.iv.plus.4 = getelementptr inbounds i32, ptr %a, i64 %iv.plus.4
+ %l2 = load i32, ptr %a.iv.plus.4, align 4
+ %sum = add i32 %call, %l2
+ %c.iv = getelementptr inbounds i32, ptr %c, i64 %iv
+ store i32 %sum, ptr %c.iv, align 4
+ %iv.next = add nuw nsw i64 %iv, 1
+ %cmp = icmp slt i64 %iv.next, %n
+ br i1 %cmp, label %loop, label %exit
+
+exit:
+ ret void
+}
+
+; Stride-2 accesses are represented by interleave-group recipes rather than by
+; simple consecutive widened loads.
+define void @non_unit_stride(ptr noalias %a, ptr noalias %c, i64 %n) {
+; CHECK-LABEL: define void @non_unit_stride(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ule i64 [[TMP0]], 4
+; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: [[TMP1:%.*]] = and i64 [[TMP0]], 3
+; CHECK-NEXT: [[TMP2:%.*]] = icmp eq i64 [[TMP1]], 0
+; CHECK-NEXT: [[TMP3:%.*]] = select i1 [[TMP2]], i64 4, i64 [[TMP1]]
+; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP3]]
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP4:%.*]] = shl nuw nsw i64 [[INDEX]], 1
+; CHECK-NEXT: [[TMP5:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP4]]
+; CHECK-NEXT: [[WIDE_VEC:%.*]] = load <8 x i32>, ptr [[TMP5]], align 4
+; CHECK-NEXT: [[STRIDED_VEC:%.*]] = shufflevector <8 x i32> [[WIDE_VEC]], <8 x i32> poison, <4 x i32> <i32 0, i32 2, i32 4, i32 6>
+; CHECK-NEXT: [[TMP6:%.*]] = add nuw nsw i64 [[TMP4]], 4
+; CHECK-NEXT: [[TMP7:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP6]]
+; CHECK-NEXT: [[WIDE_VEC1:%.*]] = load <8 x i32>, ptr [[TMP7]], align 4
+; CHECK-NEXT: [[STRIDED_VEC2:%.*]] = shufflevector <8 x i32> [[WIDE_VEC1]], <8 x i32> poison, <4 x i32> <i32 0, i32 2, i32 4, i32 6>
+; CHECK-NEXT: [[TMP8:%.*]] = add <4 x i32> [[STRIDED_VEC]], [[STRIDED_VEC2]]
+; CHECK-NEXT: [[TMP9:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
+; CHECK-NEXT: store <4 x i32> [[TMP8]], ptr [[TMP9]], align 4
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT: [[TMP10:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[TMP10]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP15:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: br label %[[SCALAR_PH]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[IV_TWICE:%.*]] = shl nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[A_IV:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV_TWICE]]
+; CHECK-NEXT: [[L1:%.*]] = load i32, ptr [[A_IV]], align 4
+; CHECK-NEXT: [[IV_TWICE_PLUS_4:%.*]] = add nuw nsw i64 [[IV_TWICE]], 4
+; CHECK-NEXT: [[A_IV_PLUS_4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV_TWICE_PLUS_4]]
+; CHECK-NEXT: [[L2:%.*]] = load i32, ptr [[A_IV_PLUS_4]], align 4
+; CHECK-NEXT: [[SUM:%.*]] = add i32 [[L1]], [[L2]]
+; CHECK-NEXT: [[C_IV:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[IV]]
+; CHECK-NEXT: store i32 [[SUM]], ptr [[C_IV]], align 4
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[CMP:%.*]] = icmp slt i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT: br i1 [[CMP]], label %[[LOOP]], label %[[EXIT:.*]], !llvm.loop [[LOOP16:![0-9]+]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %iv.twice = shl nuw nsw i64 %iv, 1
+ %a.iv = getelementptr inbounds i32, ptr %a, i64 %iv.twice
+ %l1 = load i32, ptr %a.iv, align 4
+ %iv.twice.plus.4 = add nuw nsw i64 %iv.twice, 4
+ %a.iv.plus.4 = getelementptr inbounds i32, ptr %a, i64 %iv.twice.plus.4
+ %l2 = load i32, ptr %a.iv.plus.4, align 4
+ %sum = add i32 %l1, %l2
+ %c.iv = getelementptr inbounds i32, ptr %c, i64 %iv
+ store i32 %sum, ptr %c.iv, align 4
+ %iv.next = add nuw nsw i64 %iv, 1
+ %cmp = icmp slt i64 %iv.next, %n
+ br i1 %cmp, label %loop, label %exit
+
+exit:
+ ret void
+}
+
+; VPlan folds each identical-arm select to its underlying GEP, while
+; ScalarEvolution keeps each nonconstant pointer select as a distinct unknown.
+; The address of these loads is therefore only exact in the folded VPlan value,
+; not in the scalar load ingredient.
+define void @folded_provenance(ptr noalias %a, ptr noalias %c, i64 %n, i1 %cond) {
+; CHECK-LABEL: define void @folded_provenance(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]], i1 [[COND:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 4
+; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: [[TMP1:%.*]] = and i64 [[TMP0]], 3
+; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4
+; CHECK-NEXT: [[TMP3:%.*]] = add nuw nsw i64 [[INDEX]], 4
+; CHECK-NEXT: [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP3]]
+; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP4]], align 4
+; CHECK-NEXT: [[TMP5:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[WIDE_LOAD1]]
+; CHECK-NEXT: [[TMP6:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
+; CHECK-NEXT: store <4 x i32> [[TMP5]], ptr [[TMP6]], align 4
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT: [[TMP7:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[TMP7]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP17:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[A_IV:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT: [[P1:%.*]] = select i1 [[COND]], ptr [[A_IV]], ptr [[A_IV]]
+; CHECK-NEXT: [[L1:%.*]] = load i32, ptr [[P1]], align 4
+; CHECK-NEXT: [[IV_PLUS_4:%.*]] = add nuw nsw i64 [[IV]], 4
+; CHECK-NEXT: [[A_IV_PLUS_4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV_PLUS_4]]
+; CHECK-NEXT: [[P2:%.*]] = select i1 [[COND]], ptr [[A_IV_PLUS_4]], ptr [[A_IV_PLUS_4]]
+; CHECK-NEXT: [[L2:%.*]] = load i32, ptr [[P2]], align 4
+; CHECK-NEXT: [[SUM:%.*]] = add i32 [[L1]], [[L2]]
+; CHECK-NEXT: [[C_IV:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[IV]]
+; CHECK-NEXT: store i32 [[SUM]], ptr [[C_IV]], align 4
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[CMP:%.*]] = icmp slt i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT: br i1 [[CMP]], label %[[LOOP]], label %[[EXIT]], !llvm.loop [[LOOP18:![0-9]+]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %a.iv = getelementptr inbounds i32, ptr %a, i64 %iv
+ %p1 = select i1 %cond, ptr %a.iv, ptr %a.iv
+ %l1 = load i32, ptr %p1, align 4
+ %iv.plus.4 = add nuw nsw i64 %iv, 4
+ %a.iv.plus.4 = getelementptr inbounds i32, ptr %a, i64 %iv.plus.4
+ %p2 = select i1 %cond, ptr %a.iv.plus.4, ptr %a.iv.plus.4
+ %l2 = load i32, ptr %p2, align 4
+ %sum = add i32 %l1, %l2
+ %c.iv = getelementptr inbounds i32, ptr %c, i64 %iv
+ store i32 %sum, ptr %c.iv, align 4
+ %iv.next = add nuw nsw i64 %iv, 1
+ %cmp = icmp slt i64 %iv.next, %n
+ br i1 %cmp, label %loop, label %exit
+
+exit:
+ ret void
+}
+
+; The two scalar loads carry different poison-generating !range metadata, which
+; is not propagated to the widened loads.
+define void @poison_annotations(ptr noalias %a, ptr noalias %c, i64 %n) {
+; CHECK-LABEL: define void @poison_annotations(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 4
+; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: [[TMP1:%.*]] = and i64 [[TMP0]], 3
+; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4
+; CHECK-NEXT: [[TMP3:%.*]] = add nuw nsw i64 [[INDEX]], 4
+; CHECK-NEXT: [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP3]]
+; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP4]], align 4
+; CHECK-NEXT: [[TMP5:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[WIDE_LOAD1]]
+; CHECK-NEXT: [[TMP6:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
+; CHECK-NEXT: store <4 x i32> [[TMP5]], ptr [[TMP6]], align 4
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT: [[TMP7:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[TMP7]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP19:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[A_IV:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT: [[L1:%.*]] = load i32, ptr [[A_IV]], align 4, !range [[RNG20:![0-9]+]]
+; CHECK-NEXT: [[IV_PLUS_4:%.*]] = add nuw nsw i64 [[IV]], 4
+; CHECK-NEXT: [[A_IV_PLUS_4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV_PLUS_4]]
+; CHECK-NEXT: [[L2:%.*]] = load i32, ptr [[A_IV_PLUS_4]], align 4, !range [[RNG21:![0-9]+]]
+; CHECK-NEXT: [[SUM:%.*]] = add i32 [[L1]], [[L2]]
+; CHECK-NEXT: [[C_IV:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[IV]]
+; CHECK-NEXT: store i32 [[SUM]], ptr [[C_IV]], align 4
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[CMP:%.*]] = icmp slt i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT: br i1 [[CMP]], label %[[LOOP]], label %[[EXIT]], !llvm.loop [[LOOP22:![0-9]+]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %a.iv = getelementptr inbounds i32, ptr %a, i64 %iv
+ %l1 = load i32, ptr %a.iv, align 4, !range !0
+ %iv.plus.4 = add nuw nsw i64 %iv, 4
+ %a.iv.plus.4 = getelementptr inbounds i32, ptr %a, i64 %iv.plus.4
+ %l2 = load i32, ptr %a.iv.plus.4, align 4, !range !1
+ %sum = add i32 %l1, %l2
+ %c.iv = getelementptr inbounds i32, ptr %c, i64 %iv
+ store i32 %sum, ptr %c.iv, align 4
+ %iv.next = add nuw nsw i64 %iv, 1
+ %cmp = icmp slt i64 %iv.next, %n
+ br i1 %cmp, label %loop, label %exit
+
+exit:
+ ret void
+}
+
+; The scalar loop has a separate latch, while VPlan places the two matching
+; loads in its single vector-loop block.
+define void @multi_block(ptr noalias %a, ptr noalias %c, i64 %n) {
+; CHECK-LABEL: define void @multi_block(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 4
+; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: [[TMP1:%.*]] = and i64 [[TMP0]], 3
+; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4
+; CHECK-NEXT: [[TMP3:%.*]] = add nuw nsw i64 [[INDEX]], 4
+; CHECK-NEXT: [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP3]]
+; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP4]], align 4
+; CHECK-NEXT: [[TMP5:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[WIDE_LOAD1]]
+; CHECK-NEXT: [[TMP6:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
+; CHECK-NEXT: store <4 x i32> [[TMP5]], ptr [[TMP6]], align 4
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT: [[TMP7:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[TMP7]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP23:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT: br label %[[LOOP_HEADER:.*]]
+; CHECK: [[LOOP_HEADER]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP_LATCH:.*]] ]
+; CHECK-NEXT: [[A_IV:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT: [[L1:%.*]] = load i32, ptr [[A_IV]], align 4
+; CHECK-NEXT: [[IV_PLUS_4:%.*]] = add nuw nsw i64 [[IV]], 4
+; CHECK-NEXT: [[A_IV_PLUS_4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV_PLUS_4]]
+; CHECK-NEXT: [[L2:%.*]] = load i32, ptr [[A_IV_PLUS_4]], align 4
+; CHECK-NEXT: [[SUM:%.*]] = add i32 [[L1]], [[L2]]
+; CHECK-NEXT: [[C_IV:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[IV]]
+; CHECK-NEXT: store i32 [[SUM]], ptr [[C_IV]], align 4
+; CHECK-NEXT: br label %[[LOOP_LATCH]]
+; CHECK: [[LOOP_LATCH]]:
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[CMP:%.*]] = icmp slt i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT: br i1 [[CMP]], label %[[LOOP_HEADER]], label %[[EXIT]], !llvm.loop [[LOOP24:![0-9]+]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %loop.header
+
+loop.header:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop.latch ]
+ %a.iv = getelementptr inbounds i32, ptr %a, i64 %iv
+ %l1 = load i32, ptr %a.iv, align 4
+ %iv.plus.4 = add nuw nsw i64 %iv, 4
+ %a.iv.plus.4 = getelementptr inbounds i32, ptr %a, i64 %iv.plus.4
+ %l2 = load i32, ptr %a.iv.plus.4, align 4
+ %sum = add i32 %l1, %l2
+ %c.iv = getelementptr inbounds i32, ptr %c, i64 %iv
+ store i32 %sum, ptr %c.iv, align 4
+ br label %loop.latch
+
+loop.latch:
+ %iv.next = add nuw nsw i64 %iv, 1
+ %cmp = icmp slt i64 %iv.next, %n
+ br i1 %cmp, label %loop.header, label %exit
+
+exit:
+ ret void
+}
+
+; At VF=4, the first load in part 1 and the second load in part 0 both use the
+; reverse vector ending at a[last-iv-7].
+define void @reverse_redundancy(ptr noalias %a, ptr noalias %c, i64 %n) {
+; CHECK-LABEL: define void @reverse_redundancy(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: [[LAST:%.*]] = add i64 [[N]], -1
+; CHECK-NEXT: [[TRIP_COUNT:%.*]] = add i64 [[N]], -4
+; CHECK-NEXT: [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[TRIP_COUNT]], i64 1)
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 4
+; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: [[TMP1:%.*]] = and i64 [[TMP0]], 3
+; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP2:%.*]] = sub i64 [[LAST]], [[INDEX]]
+; CHECK-NEXT: [[TMP3:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP2]]
+; CHECK-NEXT: [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[TMP3]], i64 -3
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP4]], align 4
+; CHECK-NEXT: [[TMP5:%.*]] = add i64 [[TMP2]], -4
+; CHECK-NEXT: [[TMP6:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP5]]
+; CHECK-NEXT: [[TMP7:%.*]] = getelementptr inbounds i32, ptr [[TMP6]], i64 -3
+; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP7]], align 4
+; CHECK-NEXT: [[TMP8:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[WIDE_LOAD1]]
+; CHECK-NEXT: [[REVERSE:%.*]] = shufflevector <4 x i32> [[TMP8]], <4 x i32> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
+; CHECK-NEXT: [[TMP9:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
+; CHECK-NEXT: store <4 x i32> [[REVERSE]], ptr [[TMP9]], align 4
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT: [[TMP10:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[TMP10]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP25:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[REVERSE_IV:%.*]] = sub i64 [[LAST]], [[IV]]
+; CHECK-NEXT: [[A_REVERSE:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[REVERSE_IV]]
+; CHECK-NEXT: [[L1:%.*]] = load i32, ptr [[A_REVERSE]], align 4
+; CHECK-NEXT: [[REVERSE_IV_MINUS_4:%.*]] = add i64 [[REVERSE_IV]], -4
+; CHECK-NEXT: [[A_REVERSE_MINUS_4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[REVERSE_IV_MINUS_4]]
+; CHECK-NEXT: [[L2:%.*]] = load i32, ptr [[A_REVERSE_MINUS_4]], align 4
+; CHECK-NEXT: [[SUM:%.*]] = add i32 [[L1]], [[L2]]
+; CHECK-NEXT: [[C_IV:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[IV]]
+; CHECK-NEXT: store i32 [[SUM]], ptr [[C_IV]], align 4
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[CMP:%.*]] = icmp slt i64 [[IV_NEXT]], [[TRIP_COUNT]]
+; CHECK-NEXT: br i1 [[CMP]], label %[[LOOP]], label %[[EXIT]], !llvm.loop [[LOOP26:![0-9]+]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ %last = add i64 %n, -1
+ %trip.count = add i64 %n, -4
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %reverse.iv = sub i64 %last, %iv
+ %a.reverse = getelementptr inbounds i32, ptr %a, i64 %reverse.iv
+ %l1 = load i32, ptr %a.reverse, align 4
+ %reverse.iv.minus.4 = add i64 %reverse.iv, -4
+ %a.reverse.minus.4 = getelementptr inbounds i32, ptr %a, i64 %reverse.iv.minus.4
+ %l2 = load i32, ptr %a.reverse.minus.4, align 4
+ %sum = add i32 %l1, %l2
+ %c.iv = getelementptr inbounds i32, ptr %c, i64 %iv
+ store i32 %sum, ptr %c.iv, align 4
+ %iv.next = add nuw nsw i64 %iv, 1
+ %cmp = icmp slt i64 %iv.next, %trip.count
+ br i1 %cmp, label %loop, label %exit
+
+exit:
+ ret void
+}
+
+; The reverse access sits physically between the two matching forward loads.
+; Its end-pointer address computation is pure and must not be mistaken for a
+; memory write.
+define void @reverse_between(ptr noalias %a, ptr noalias %c, i64 %n) {
+; CHECK-LABEL: define void @reverse_between(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: [[LAST:%.*]] = add i64 [[N]], -1
+; CHECK-NEXT: [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 4
+; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: [[TMP1:%.*]] = and i64 [[TMP0]], 3
+; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4
+; CHECK-NEXT: [[TMP3:%.*]] = sub i64 [[LAST]], [[INDEX]]
+; CHECK-NEXT: [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP3]]
+; CHECK-NEXT: [[TMP5:%.*]] = getelementptr inbounds i32, ptr [[TMP4]], i64 -3
+; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP5]], align 4
+; CHECK-NEXT: [[REVERSE2:%.*]] = shufflevector <4 x i32> [[WIDE_LOAD1]], <4 x i32> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
+; CHECK-NEXT: [[TMP6:%.*]] = add nuw nsw i64 [[INDEX]], 4
+; CHECK-NEXT: [[TMP7:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP6]]
+; CHECK-NEXT: [[WIDE_LOAD3:%.*]] = load <4 x i32>, ptr [[TMP7]], align 4
+; CHECK-NEXT: [[TMP8:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[WIDE_LOAD3]]
+; CHECK-NEXT: [[TMP9:%.*]] = add <4 x i32> [[TMP8]], [[REVERSE2]]
+; CHECK-NEXT: [[TMP10:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
+; CHECK-NEXT: store <4 x i32> [[TMP9]], ptr [[TMP10]], align 4
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT: [[TMP11:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[TMP11]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP27:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[A_IV:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT: [[L1:%.*]] = load i32, ptr [[A_IV]], align 4
+; CHECK-NEXT: [[IV_REVERSE:%.*]] = sub i64 [[LAST]], [[IV]]
+; CHECK-NEXT: [[A_REVERSE:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV_REVERSE]]
+; CHECK-NEXT: [[REVERSE:%.*]] = load i32, ptr [[A_REVERSE]], align 4
+; CHECK-NEXT: [[IV_PLUS_4:%.*]] = add nuw nsw i64 [[IV]], 4
+; CHECK-NEXT: [[A_IV_PLUS_4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV_PLUS_4]]
+; CHECK-NEXT: [[L2:%.*]] = load i32, ptr [[A_IV_PLUS_4]], align 4
+; CHECK-NEXT: [[SUM_FORWARD:%.*]] = add i32 [[L1]], [[L2]]
+; CHECK-NEXT: [[SUM:%.*]] = add i32 [[SUM_FORWARD]], [[REVERSE]]
+; CHECK-NEXT: [[C_IV:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[IV]]
+; CHECK-NEXT: store i32 [[SUM]], ptr [[C_IV]], align 4
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[CMP:%.*]] = icmp slt i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT: br i1 [[CMP]], label %[[LOOP]], label %[[EXIT]], !llvm.loop [[LOOP28:![0-9]+]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ %last = add i64 %n, -1
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %a.iv = getelementptr inbounds i32, ptr %a, i64 %iv
+ %l1 = load i32, ptr %a.iv, align 4
+ %iv.reverse = sub i64 %last, %iv
+ %a.reverse = getelementptr inbounds i32, ptr %a, i64 %iv.reverse
+ %reverse = load i32, ptr %a.reverse, align 4
+ %iv.plus.4 = add nuw nsw i64 %iv, 4
+ %a.iv.plus.4 = getelementptr inbounds i32, ptr %a, i64 %iv.plus.4
+ %l2 = load i32, ptr %a.iv.plus.4, align 4
+ %sum.forward = add i32 %l1, %l2
+ %sum = add i32 %sum.forward, %reverse
+ %c.iv = getelementptr inbounds i32, ptr %c, i64 %iv
+ store i32 %sum, ptr %c.iv, align 4
+ %iv.next = add nuw nsw i64 %iv, 1
+ %cmp = icmp slt i64 %iv.next, %n
+ br i1 %cmp, label %loop, label %exit
+
+exit:
+ ret void
+}
+
+attributes #0 = { nounwind willreturn "vector-function-abi-variant"="_ZGVnN4v_write_memory(write_memory_v4)" }
+
+!0 = !{i32 0, i32 100}
+!1 = !{i32 0, i32 101}
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-opportunities.ll b/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-opportunities.ll
new file mode 100644
index 0000000000000..fe8e8c0b53880
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-opportunities.ll
@@ -0,0 +1,167 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
+; Pre-commit test for counting cross-part load redundancy opportunities. The
+; two loops below differ only in how many distinct opportunities they contain;
+; both vectorize with an interleave count of 1 today.
+; RUN: opt -passes=loop-vectorize -force-vector-width=4 \
+; RUN: -force-target-max-vector-interleave=2 -small-loop-cost=0 \
+; RUN: -S %s | FileCheck %s
+
+target triple = "aarch64-unknown-linux-gnu"
+
+; The first a[i+4] load forms a cross-part redundancy opportunity with a[i].
+; The second a[i+4] load is a duplicate within the same logical part: it already
+; exists without interleaving and therefore is no additional cross-part
+; opportunity, which leaves a single opportunity in this loop.
+define void @duplicate_after_cross_part(ptr noalias %a, ptr noalias %c, i64 %n) {
+; CHECK-LABEL: define void @duplicate_after_cross_part(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 4
+; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: [[TMP1:%.*]] = and i64 [[TMP0]], 3
+; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4
+; CHECK-NEXT: [[TMP3:%.*]] = add nuw nsw i64 [[INDEX]], 4
+; CHECK-NEXT: [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP3]]
+; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP4]], align 4
+; CHECK-NEXT: [[TMP5:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[WIDE_LOAD1]]
+; CHECK-NEXT: [[TMP6:%.*]] = add <4 x i32> [[TMP5]], [[WIDE_LOAD1]]
+; CHECK-NEXT: [[TMP7:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
+; CHECK-NEXT: store <4 x i32> [[TMP6]], ptr [[TMP7]], align 4
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT: [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[A_IV:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT: [[L1:%.*]] = load i32, ptr [[A_IV]], align 4
+; CHECK-NEXT: [[IV_PLUS_4:%.*]] = add nuw nsw i64 [[IV]], 4
+; CHECK-NEXT: [[A_IV_PLUS_4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV_PLUS_4]]
+; CHECK-NEXT: [[L2:%.*]] = load i32, ptr [[A_IV_PLUS_4]], align 4
+; CHECK-NEXT: [[L3:%.*]] = load i32, ptr [[A_IV_PLUS_4]], align 4
+; CHECK-NEXT: [[SUM_1:%.*]] = add i32 [[L1]], [[L2]]
+; CHECK-NEXT: [[SUM_2:%.*]] = add i32 [[SUM_1]], [[L3]]
+; CHECK-NEXT: [[C_IV:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[IV]]
+; CHECK-NEXT: store i32 [[SUM_2]], ptr [[C_IV]], align 4
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[CMP:%.*]] = icmp slt i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT: br i1 [[CMP]], label %[[LOOP]], label %[[EXIT]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %a.iv = getelementptr inbounds i32, ptr %a, i64 %iv
+ %l1 = load i32, ptr %a.iv, align 4
+ %iv.plus.4 = add nuw nsw i64 %iv, 4
+ %a.iv.plus.4 = getelementptr inbounds i32, ptr %a, i64 %iv.plus.4
+ %l2 = load i32, ptr %a.iv.plus.4, align 4
+ %l3 = load i32, ptr %a.iv.plus.4, align 4
+ %sum.1 = add i32 %l1, %l2
+ %sum.2 = add i32 %sum.1, %l3
+ %c.iv = getelementptr inbounds i32, ptr %c, i64 %iv
+ store i32 %sum.2, ptr %c.iv, align 4
+ %iv.next = add nuw nsw i64 %iv, 1
+ %cmp = icmp slt i64 %iv.next, %n
+ br i1 %cmp, label %loop, label %exit
+
+exit:
+ ret void
+}
+
+; At VF=4, a[i+4] in part 0 is redundant with a[i] in part 1, and a[i+8] in
+; part 0 is independently redundant with a[i+4] in part 1, which gives two
+; distinct opportunities in this loop.
+define void @two_opportunities(ptr noalias %a, ptr noalias %c, i64 %n) {
+; CHECK-LABEL: define void @two_opportunities(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 4
+; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: [[TMP1:%.*]] = and i64 [[TMP0]], 3
+; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4
+; CHECK-NEXT: [[TMP3:%.*]] = add nuw nsw i64 [[INDEX]], 4
+; CHECK-NEXT: [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP3]]
+; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP4]], align 4
+; CHECK-NEXT: [[TMP5:%.*]] = add nuw nsw i64 [[INDEX]], 8
+; CHECK-NEXT: [[TMP6:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP5]]
+; CHECK-NEXT: [[WIDE_LOAD2:%.*]] = load <4 x i32>, ptr [[TMP6]], align 4
+; CHECK-NEXT: [[TMP7:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[WIDE_LOAD1]]
+; CHECK-NEXT: [[TMP8:%.*]] = add <4 x i32> [[TMP7]], [[WIDE_LOAD2]]
+; CHECK-NEXT: [[TMP9:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
+; CHECK-NEXT: store <4 x i32> [[TMP8]], ptr [[TMP9]], align 4
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT: [[TMP10:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[TMP10]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[A_IV:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT: [[L1:%.*]] = load i32, ptr [[A_IV]], align 4
+; CHECK-NEXT: [[IV_PLUS_4:%.*]] = add nuw nsw i64 [[IV]], 4
+; CHECK-NEXT: [[A_IV_PLUS_4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV_PLUS_4]]
+; CHECK-NEXT: [[L2:%.*]] = load i32, ptr [[A_IV_PLUS_4]], align 4
+; CHECK-NEXT: [[IV_PLUS_8:%.*]] = add nuw nsw i64 [[IV]], 8
+; CHECK-NEXT: [[A_IV_PLUS_8:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV_PLUS_8]]
+; CHECK-NEXT: [[L3:%.*]] = load i32, ptr [[A_IV_PLUS_8]], align 4
+; CHECK-NEXT: [[SUM_1:%.*]] = add i32 [[L1]], [[L2]]
+; CHECK-NEXT: [[SUM_2:%.*]] = add i32 [[SUM_1]], [[L3]]
+; CHECK-NEXT: [[C_IV:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[IV]]
+; CHECK-NEXT: store i32 [[SUM_2]], ptr [[C_IV]], align 4
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[CMP:%.*]] = icmp slt i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT: br i1 [[CMP]], label %[[LOOP]], label %[[EXIT]], !llvm.loop [[LOOP5:![0-9]+]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %a.iv = getelementptr inbounds i32, ptr %a, i64 %iv
+ %l1 = load i32, ptr %a.iv, align 4
+ %iv.plus.4 = add nuw nsw i64 %iv, 4
+ %a.iv.plus.4 = getelementptr inbounds i32, ptr %a, i64 %iv.plus.4
+ %l2 = load i32, ptr %a.iv.plus.4, align 4
+ %iv.plus.8 = add nuw nsw i64 %iv, 8
+ %a.iv.plus.8 = getelementptr inbounds i32, ptr %a, i64 %iv.plus.8
+ %l3 = load i32, ptr %a.iv.plus.8, align 4
+ %sum.1 = add i32 %l1, %l2
+ %sum.2 = add i32 %sum.1, %l3
+ %c.iv = getelementptr inbounds i32, ptr %c, i64 %iv
+ store i32 %sum.2, ptr %c.iv, align 4
+ %iv.next = add nuw nsw i64 %iv, 1
+ %cmp = icmp slt i64 %iv.next, %n
+ br i1 %cmp, label %loop, label %exit
+
+exit:
+ ret void
+}
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-scalable.ll b/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-scalable.ll
new file mode 100644
index 0000000000000..2af48b332d9b2
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-scalable.ll
@@ -0,0 +1,158 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
+; Pre-commit test for cross-part load redundancy with scalable vectors, where
+; the distance between two logical parts is a runtime multiple of vscale. Both
+; loops below vectorize with an interleave count of 1 today.
+; RUN: opt -passes=loop-vectorize -mtriple=aarch64-none-linux-gnu -mattr=+sve \
+; RUN: -force-vector-width="vscale x 2" \
+; RUN: -force-target-max-vector-interleave=2 -small-loop-cost=0 \
+; RUN: -S %s | FileCheck %s
+
+target triple = "aarch64-unknown-linux-gnu"
+
+declare i64 @llvm.vscale.i64()
+
+; A constant source offset cannot equal the scalable logical-part offset for
+; every runtime vscale, so the two addresses are not exactly equal.
+define void @constant_offset(ptr noalias %a, ptr noalias %c, i64 %n) {
+; CHECK-LABEL: define void @constant_offset(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) #[[ATTR1:[0-9]+]] {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; CHECK-NEXT: [[TMP1:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-NEXT: [[TMP2:%.*]] = shl nuw i64 [[TMP1]], 1
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], [[TMP2]]
+; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: [[N_MOD_VF:%.*]] = urem i64 [[TMP0]], [[TMP2]]
+; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[TMP0]], [[N_MOD_VF]]
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP3:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <vscale x 2 x i32>, ptr [[TMP3]], align 4
+; CHECK-NEXT: [[TMP4:%.*]] = add nuw nsw i64 [[INDEX]], 4
+; CHECK-NEXT: [[TMP5:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP4]]
+; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <vscale x 2 x i32>, ptr [[TMP5]], align 4
+; CHECK-NEXT: [[TMP6:%.*]] = add <vscale x 2 x i32> [[WIDE_LOAD]], [[WIDE_LOAD1]]
+; CHECK-NEXT: [[TMP7:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
+; CHECK-NEXT: store <vscale x 2 x i32> [[TMP6]], ptr [[TMP7]], align 4
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP2]]
+; CHECK-NEXT: [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[A_IV:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT: [[L1:%.*]] = load i32, ptr [[A_IV]], align 4
+; CHECK-NEXT: [[IV_PLUS_4:%.*]] = add nuw nsw i64 [[IV]], 4
+; CHECK-NEXT: [[A_IV_PLUS_4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV_PLUS_4]]
+; CHECK-NEXT: [[L2:%.*]] = load i32, ptr [[A_IV_PLUS_4]], align 4
+; CHECK-NEXT: [[SUM:%.*]] = add i32 [[L1]], [[L2]]
+; CHECK-NEXT: [[C_IV:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[IV]]
+; CHECK-NEXT: store i32 [[SUM]], ptr [[C_IV]], align 4
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[CMP:%.*]] = icmp slt i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT: br i1 [[CMP]], label %[[LOOP]], label %[[EXIT]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %a.iv = getelementptr inbounds i32, ptr %a, i64 %iv
+ %l1 = load i32, ptr %a.iv, align 4
+ %iv.plus.4 = add nuw nsw i64 %iv, 4
+ %a.iv.plus.4 = getelementptr inbounds i32, ptr %a, i64 %iv.plus.4
+ %l2 = load i32, ptr %a.iv.plus.4, align 4
+ %sum = add i32 %l1, %l2
+ %c.iv = getelementptr inbounds i32, ptr %c, i64 %iv
+ store i32 %sum, ptr %c.iv, align 4
+ %iv.next = add nuw nsw i64 %iv, 1
+ %cmp = icmp slt i64 %iv.next, %n
+ br i1 %cmp, label %loop, label %exit
+
+exit:
+ ret void
+}
+
+; The second load starts exactly one scalable VF after the first, so its part-0
+; address equals the first load's part-1 address for every runtime vscale.
+define void @vscale_offset(ptr noalias %a, ptr noalias %c, i64 %n) {
+; CHECK-LABEL: define void @vscale_offset(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) #[[ATTR1]] {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: [[VSCALE:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-NEXT: [[PART_OFFSET:%.*]] = mul nuw i64 [[VSCALE]], 2
+; CHECK-NEXT: [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], [[PART_OFFSET]]
+; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: [[TMP1:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-NEXT: [[TMP2:%.*]] = shl nuw i64 [[TMP1]], 1
+; CHECK-NEXT: [[N_MOD_VF:%.*]] = urem i64 [[TMP0]], [[TMP2]]
+; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[TMP0]], [[N_MOD_VF]]
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP3:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <vscale x 2 x i32>, ptr [[TMP3]], align 4
+; CHECK-NEXT: [[TMP4:%.*]] = add i64 [[INDEX]], [[PART_OFFSET]]
+; CHECK-NEXT: [[TMP5:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP4]]
+; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <vscale x 2 x i32>, ptr [[TMP5]], align 4
+; CHECK-NEXT: [[TMP6:%.*]] = add <vscale x 2 x i32> [[WIDE_LOAD]], [[WIDE_LOAD1]]
+; CHECK-NEXT: [[TMP7:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
+; CHECK-NEXT: store <vscale x 2 x i32> [[TMP6]], ptr [[TMP7]], align 4
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP2]]
+; CHECK-NEXT: [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[A_IV:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT: [[L1:%.*]] = load i32, ptr [[A_IV]], align 4
+; CHECK-NEXT: [[IV_PART:%.*]] = add i64 [[IV]], [[PART_OFFSET]]
+; CHECK-NEXT: [[A_IV_PART:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV_PART]]
+; CHECK-NEXT: [[L2:%.*]] = load i32, ptr [[A_IV_PART]], align 4
+; CHECK-NEXT: [[SUM:%.*]] = add i32 [[L1]], [[L2]]
+; CHECK-NEXT: [[C_IV:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[IV]]
+; CHECK-NEXT: store i32 [[SUM]], ptr [[C_IV]], align 4
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[CMP:%.*]] = icmp slt i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT: br i1 [[CMP]], label %[[LOOP]], label %[[EXIT]], !llvm.loop [[LOOP5:![0-9]+]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ %vscale = call i64 @llvm.vscale.i64()
+ %part.offset = mul nuw i64 %vscale, 2
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %a.iv = getelementptr inbounds i32, ptr %a, i64 %iv
+ %l1 = load i32, ptr %a.iv, align 4
+ %iv.part = add i64 %iv, %part.offset
+ %a.iv.part = getelementptr inbounds i32, ptr %a, i64 %iv.part
+ %l2 = load i32, ptr %a.iv.part, align 4
+ %sum = add i32 %l1, %l2
+ %c.iv = getelementptr inbounds i32, ptr %c, i64 %iv
+ store i32 %sum, ptr %c.iv, align 4
+ %iv.next = add nuw nsw i64 %iv, 1
+ %cmp = icmp slt i64 %iv.next, %n
+ br i1 %cmp, label %loop, label %exit
+
+exit:
+ ret void
+}
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse.ll b/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse.ll
new file mode 100644
index 0000000000000..f3b3cafd67b37
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse.ll
@@ -0,0 +1,81 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
+; Pre-commit test for cross-part load redundancy. A widened load in one
+; logical part of an interleaved vector loop can cover exactly the elements
+; that another widened load covers in the next part. The interleave-count
+; heuristics do not model that redundancy yet, so this loop vectorizes with an
+; interleave count of 1.
+; RUN: opt -passes=loop-vectorize -force-vector-width=4 \
+; RUN: -force-target-max-vector-interleave=2 -small-loop-cost=0 \
+; RUN: -S %s | FileCheck %s
+
+target triple = "aarch64-unknown-linux-gnu"
+
+; The motivating shape for cross-part load redundancy: with VF=4, the widened
+; load of a[i+4] in logical part 0 would cover exactly the same elements as the
+; widened load of a[i] in logical part 1.
+define void @positive(ptr noalias %a, ptr noalias %c, i64 %n) {
+; CHECK-LABEL: define void @positive(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 4
+; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: [[TMP1:%.*]] = and i64 [[TMP0]], 3
+; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4
+; CHECK-NEXT: [[TMP3:%.*]] = add nuw nsw i64 [[INDEX]], 4
+; CHECK-NEXT: [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP3]]
+; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP4]], align 4
+; CHECK-NEXT: [[TMP5:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[WIDE_LOAD1]]
+; CHECK-NEXT: [[TMP6:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
+; CHECK-NEXT: store <4 x i32> [[TMP5]], ptr [[TMP6]], align 4
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT: [[TMP7:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[TMP7]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[A_IV:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT: [[L1:%.*]] = load i32, ptr [[A_IV]], align 4
+; CHECK-NEXT: [[IV_PLUS_4:%.*]] = add nuw nsw i64 [[IV]], 4
+; CHECK-NEXT: [[A_IV_PLUS_4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV_PLUS_4]]
+; CHECK-NEXT: [[L2:%.*]] = load i32, ptr [[A_IV_PLUS_4]], align 4
+; CHECK-NEXT: [[SUM:%.*]] = add i32 [[L1]], [[L2]]
+; CHECK-NEXT: [[C_IV:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[IV]]
+; CHECK-NEXT: store i32 [[SUM]], ptr [[C_IV]], align 4
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[CMP:%.*]] = icmp slt i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT: br i1 [[CMP]], label %[[LOOP]], label %[[EXIT]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %a.iv = getelementptr inbounds i32, ptr %a, i64 %iv
+ %l1 = load i32, ptr %a.iv, align 4
+ %iv.plus.4 = add nuw nsw i64 %iv, 4
+ %a.iv.plus.4 = getelementptr inbounds i32, ptr %a, i64 %iv.plus.4
+ %l2 = load i32, ptr %a.iv.plus.4, align 4
+ %sum = add i32 %l1, %l2
+ %c.iv = getelementptr inbounds i32, ptr %c, i64 %iv
+ store i32 %sum, ptr %c.iv, align 4
+ %iv.next = add nuw nsw i64 %iv, 1
+ %cmp = icmp slt i64 %iv.next, %n
+ br i1 %cmp, label %loop, label %exit
+
+exit:
+ ret void
+}
>From 8af706c315b72a632066a382d191a65054376f1e Mon Sep 17 00:00:00 2001
From: Sergey Shcherbinin <sscherbinin at nvidia.com>
Date: Mon, 21 Sep 2026 17:31:21 +0400
Subject: [PATCH 2/2] [LV] Add cross-part load redundancy heuristic for
interleaving
Existing interleave heuristics can select IC=1 when load redundancies become visible only between logical VPlan parts.
Add a disabled-by-default, prediction-only analysis for eligible single-block VPlans. Model two logical parts with exact SCEV addresses, invalidate available loads at memory writes, and raise IC from 1 to 2 when the predicted saved load cost reaches the configured threshold. Handle forward and reverse accesses and fail closed for unsupported plan states.
The analysis neither mutates VPlan nor removes loads; existing downstream optimizations may realize the exposed fixed-width redundancies.
RFC: https://discourse.llvm.org/t/rfc-using-cross-part-cse-to-guide-loop-interleaving/91438
---
llvm/lib/Transforms/Vectorize/CMakeLists.txt | 1 +
.../Transforms/Vectorize/LoopVectorize.cpp | 46 +++
.../Vectorize/VPlanCrossPartCSE.cpp | 316 +++++++++++++++
.../Transforms/Vectorize/VPlanCrossPartCSE.h | 57 +++
.../AArch64/cross-part-load-cse-debug.ll | 263 ++++++++++++
.../cross-part-load-cse-fixed-cases.ll | 185 +++++----
.../AArch64/cross-part-load-cse-narrowed.ll | 50 +++
.../cross-part-load-cse-opportunities.ll | 34 +-
.../AArch64/cross-part-load-cse-pipeline.ll | 49 +++
.../cross-part-load-cse-scalable-reverse.ll | 188 +++++++++
.../AArch64/cross-part-load-cse-scalable.ll | 27 +-
.../cross-part-load-cse-vplan-multi-block.ll | 122 ++++++
.../AArch64/cross-part-load-cse.ll | 379 +++++++++++++++---
.../llvm/lib/Transforms/Vectorize/BUILD.gn | 1 +
14 files changed, 1579 insertions(+), 139 deletions(-)
create mode 100644 llvm/lib/Transforms/Vectorize/VPlanCrossPartCSE.cpp
create mode 100644 llvm/lib/Transforms/Vectorize/VPlanCrossPartCSE.h
create mode 100644 llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-debug.ll
create mode 100644 llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-narrowed.ll
create mode 100644 llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-pipeline.ll
create mode 100644 llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-scalable-reverse.ll
create mode 100644 llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-vplan-multi-block.ll
diff --git a/llvm/lib/Transforms/Vectorize/CMakeLists.txt b/llvm/lib/Transforms/Vectorize/CMakeLists.txt
index 9073211280886..f459cbc618557 100644
--- a/llvm/lib/Transforms/Vectorize/CMakeLists.txt
+++ b/llvm/lib/Transforms/Vectorize/CMakeLists.txt
@@ -35,6 +35,7 @@ add_llvm_component_library(LLVMVectorize
VPlan.cpp
VPlanAnalysis.cpp
VPlanConstruction.cpp
+ VPlanCrossPartCSE.cpp
VPlanDominatorTree.cpp
VPlanEVLTailFolding.cpp
VPlanLowering.cpp
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index e16e707066bdc..ebbdbefa69605 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -59,6 +59,7 @@
#include "VPlan.h"
#include "VPlanAnalysis.h"
#include "VPlanCFG.h"
+#include "VPlanCrossPartCSE.h"
#include "VPlanHelpers.h"
#include "VPlanPatternMatch.h"
#include "VPlanTransforms.h"
@@ -300,6 +301,18 @@ static cl::opt<bool> EnableLoadStoreRuntimeInterleave(
cl::desc(
"Enable runtime interleaving until load/store ports are saturated"));
+/// Enable cross-part load-redundancy analysis during IC selection.
+static cl::opt<bool> EnableInterleaveCSE(
+ "enable-interleave-cse", cl::init(false), cl::Hidden,
+ cl::desc("Raise heuristic IC=1 to IC=2 when exact cross-part load "
+ "redundancy predicts a downstream saving"));
+
+/// Minimum percentage of the modeled UF=2 body predicted to be saved.
+static cl::opt<unsigned> InterleaveCSEMinSavingPct(
+ "interleave-cse-min-pct", cl::init(5), cl::Hidden,
+ cl::desc("Minimum predicted downstream load saving as a percentage of the "
+ "modeled UF=2 vector loop body"));
+
// TODO: Move size-based thresholds out of legality checking, make cost based
// decisions instead of hard thresholds.
static cl::opt<unsigned> VectorizeSCEVCheckThreshold(
@@ -3632,6 +3645,33 @@ std::unique_ptr<VPlan> LoopVectorizationPlanner::selectBestEpiloguePlan(
unsigned
LoopVectorizationPlanner::selectInterleaveCount(VPlan &Plan, ElementCount VF,
InstructionCost LoopCost) {
+ // Evaluate cross-part savings only when the ordinary heuristics are about to
+ // return IC=1. All cost state remains local to this UF-selection call.
+ auto shouldInterleaveForCrossPartCSE = [&](unsigned MaxIC) {
+ if (!EnableInterleaveCSE)
+ return false;
+
+ // Cross-part CSE only augments ordinary heuristic selection. These checks
+ // preserve target, trip-count, and user-hint restrictions. Loops vectorized
+ // with partial-alias masking are excluded explicitly because processLoop
+ // resets their interleave count to 1 after this selection; analyzing them
+ // would only spend cost-model queries and report a discarded decision.
+ if (MaxIC < CrossPartCSERequiredInterleaveCount || !VF.isVector() ||
+ !OrigLoop->isInnermost() || Config.getHints().getInterleave() != 0 ||
+ CM->maskPartialAliasing())
+ return false;
+
+ VPCostContext CostCtx(*TLI, Plan, *CM, Config);
+ CrossPartCSEOptions Options;
+ Options.MinSavingPct = InterleaveCSEMinSavingPct;
+ if (!isCrossPartCSEProfitable(Plan, VF, LoopCost, CostCtx, Options))
+ return false;
+
+ LLVM_DEBUG(dbgs() << "LV: Exact cross-part load redundancy predicts a "
+ "downstream saving; raising IC to 2.\n");
+ return true;
+ };
+
// -- The interleave heuristics --
// We interleave the loop in order to expose ILP and reduce the loop overhead.
// There are many micro-architectural considerations that we can't predict
@@ -3961,6 +4001,9 @@ LoopVectorizationPlanner::selectInterleaveCount(VPlan &Plan, ElementCount VF,
return std::max(IC / 2, SmallIC);
}
+ if (SmallIC == 1 && shouldInterleaveForCrossPartCSE(IC))
+ return CrossPartCSERequiredInterleaveCount;
+
LLVM_DEBUG(dbgs() << "LV: Interleaving to reduce branch cost.\n");
return SmallIC;
}
@@ -3972,6 +4015,9 @@ LoopVectorizationPlanner::selectInterleaveCount(VPlan &Plan, ElementCount VF,
return IC;
}
+ if (shouldInterleaveForCrossPartCSE(IC))
+ return CrossPartCSERequiredInterleaveCount;
+
LLVM_DEBUG(dbgs() << "LV: Not Interleaving.\n");
return 1;
}
diff --git a/llvm/lib/Transforms/Vectorize/VPlanCrossPartCSE.cpp b/llvm/lib/Transforms/Vectorize/VPlanCrossPartCSE.cpp
new file mode 100644
index 0000000000000..2912a2448de37
--- /dev/null
+++ b/llvm/lib/Transforms/Vectorize/VPlanCrossPartCSE.cpp
@@ -0,0 +1,316 @@
+//===- VPlanCrossPartCSE.cpp - Cross-part CSE for VPlan -------------------===//
+//
+// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
+// See https://llvm.org/LICENSE.txt for license information.
+// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
+//
+//===----------------------------------------------------------------------===//
+//
+// This file implements exact load-redundancy profitability analysis across two
+// logical VPlan parts.
+//
+//===----------------------------------------------------------------------===//
+
+#include "VPlanCrossPartCSE.h"
+#include "VPlan.h"
+#include "VPlanHelpers.h"
+#include "VPlanPatternMatch.h"
+#include "VPlanUtils.h"
+#include "llvm/ADT/DenseMap.h"
+#include "llvm/ADT/Hashing.h"
+#include "llvm/Analysis/LoopInfo.h"
+#include "llvm/Analysis/ScalarEvolution.h"
+#include "llvm/Support/Debug.h"
+#include "llvm/Support/raw_ostream.h"
+
+using namespace llvm;
+
+#define DEBUG_TYPE "loop-vectorize"
+
+namespace {
+
+/// Return an unmasked, non-EVL, consecutive widened load.
+static VPWidenLoadRecipe *getCrossPartSupportedLoad(VPRecipeBase &R) {
+ auto *Load = dyn_cast<VPWidenLoadRecipe>(&R);
+ if (!Load || Load->isMasked() || !Load->isConsecutive())
+ return nullptr;
+ return Load;
+}
+
+/// Build addresses only for provenance whose physical UF mapping is explicit.
+class CrossPartAddressBuilder {
+ /// Predicated SCEV state carrying vectorization assumptions.
+ PredicatedScalarEvolution &PSE;
+ /// ScalarEvolution used for canonical exact identities.
+ ScalarEvolution &SE;
+ /// Original loop used to interpret loop-varying VPlan values.
+ const Loop *OrigLoop;
+ /// Vector factor used to model the exact per-part offset.
+ const ElementCount VF;
+ /// Base SCEVs cached by VPlan value for reuse across loads and parts.
+ DenseMap<const VPValue *, const SCEV *> BaseSCEVs;
+
+ /// Return the SCEV represented by \p V, caching it after first construction.
+ const SCEV *getBaseSCEV(const VPValue *V) {
+ auto It = BaseSCEVs.find(V);
+ if (It != BaseSCEVs.end())
+ return It->second;
+
+ const SCEV *S = vputils::getSCEVExprForVPValue(V, PSE, OrigLoop);
+ BaseSCEVs.try_emplace(V, S);
+ return S;
+ }
+
+ /// Return a conservative GEP expression for \p Base + \p Offset.
+ const SCEV *getGEPAddress(const SCEV *Base, const SCEV *Offset,
+ Type *SourceElementTy) {
+ // ScalarEvolution imports GEP nowrap facts only after accounting for their
+ // poison semantics. AddExpr uniquing still recognizes equal operands
+ // without adding those facts to the synthetic expression.
+ return SE.getGEPExpr(Base, {Offset}, SourceElementTy);
+ }
+
+ /// Return the physical address produced by \p VectorPtr for \p Part.
+ const SCEV *getForwardAddress(VPVectorPointerRecipe &VectorPtr,
+ unsigned Part) {
+ // VPlanUnroll models a forward part as Base + Part * VF * Stride. Accept
+ // only unit stride until the analysis supports the complete expression.
+ using namespace VPlanPatternMatch;
+ if (!match(VectorPtr.getStride(), m_One()))
+ return SE.getCouldNotCompute();
+
+ const SCEV *Base = getBaseSCEV(VectorPtr.getOperand(0));
+ if (isa<SCEVCouldNotCompute>(Base))
+ return SE.getCouldNotCompute();
+ if (Part == 0)
+ return Base;
+
+ Type *IndexTy = SE.getDataLayout().getIndexType(VectorPtr.getScalarType());
+ const SCEV *Offset = SE.getElementCount(IndexTy, VF * Part);
+ return getGEPAddress(Base, Offset, VectorPtr.getSourceElementType());
+ }
+
+ /// Return the physical address produced by \p EndPtr for \p Part.
+ const SCEV *getReverseAddress(VPVectorEndPointerRecipe &EndPtr,
+ unsigned Part) {
+ const SCEV *Base = getBaseSCEV(EndPtr.getPointer());
+ if (isa<SCEVCouldNotCompute>(Base))
+ return SE.getCouldNotCompute();
+
+ Type *IndexTy = SE.getDataLayout().getIndexType(EndPtr.getScalarType());
+ const SCEV *VFExpr = SE.getElementCount(IndexTy, VF);
+ const SCEV *Stride =
+ SE.getConstant(IndexTy, EndPtr.getStride(), /*isSigned=*/true);
+
+ // Mirror VPVectorEndPointerRecipe::materializeOffset:
+ // Stride * (VF - 1) + Part * Stride * VF.
+ const SCEV *Offset0 =
+ SE.getMulExpr(SE.getMinusSCEV(VFExpr, SE.getOne(IndexTy)), Stride);
+ int64_t PartStride = static_cast<int64_t>(Part) * EndPtr.getStride();
+ const SCEV *PartOffset = SE.getMulExpr(
+ SE.getConstant(IndexTy, PartStride, /*isSigned=*/true), VFExpr);
+ const SCEV *Offset = SE.getAddExpr(Offset0, PartOffset);
+ return getGEPAddress(Base, Offset, EndPtr.getSourceElementType());
+ }
+
+public:
+ /// Bind the VF, original loop, and predicated SCEV state.
+ CrossPartAddressBuilder(ElementCount VF, PredicatedScalarEvolution &PSE,
+ const Loop *OrigLoop)
+ : PSE(PSE), SE(*PSE.getSE()), OrigLoop(OrigLoop), VF(VF) {}
+
+ /// Return the exact address used by \p Load in logical part \p Part.
+ const SCEV *getAddress(VPWidenLoadRecipe &Load, unsigned Part) {
+ assert(Part < CrossPartCSERequiredInterleaveCount &&
+ "logical part must be zero or one");
+ VPValue *Addr = Load.getAddr();
+
+ // Reproduce only the recipe-specific physical rewrites performed by
+ // VPlanUnroll for consecutive forward and reverse accesses.
+ auto *VectorPtr = dyn_cast<VPVectorPointerRecipe>(Addr);
+ if (VectorPtr)
+ return getForwardAddress(*VectorPtr, Part);
+ auto *EndPtr = dyn_cast<VPVectorEndPointerRecipe>(Addr);
+ if (EndPtr)
+ return getReverseAddress(*EndPtr, Part);
+
+ return SE.getCouldNotCompute();
+ }
+};
+
+/// Key for exact value equality of two logical widened-load instances.
+/// Metadata and alignment are not part of value identity. A CSE implementation
+/// must intersect retained metadata and preserve an alignment sufficient for
+/// every replaced load.
+/// Poison-generating source metadata is not propagated to widened loads.
+struct CrossPartLoadKey {
+ /// Canonical SCEV address for this logical load instance.
+ const SCEV *Address;
+ /// Loaded scalar type required for value compatibility.
+ Type *ValueType;
+};
+
+/// DenseMap policy for exact canonical load keys.
+struct CrossPartLoadKeyInfo {
+ /// Hash every property required by exact load equality.
+ static unsigned getHashValue(const CrossPartLoadKey &Key) {
+ return hash_combine(Key.Address, Key.ValueType);
+ }
+
+ /// Compare every property required by exact load equality.
+ static bool isEqual(const CrossPartLoadKey &A, const CrossPartLoadKey &B) {
+ return A.Address == B.Address && A.ValueType == B.ValueType;
+ }
+};
+
+/// Return whether \p R may write memory during VPlan execution.
+static bool isCrossPartWrite(const VPRecipeBase &R) {
+ // VPVectorEndPointerRecipe is pure but inherits the conservative memory
+ // default. This local exception prevents its address computation from being
+ // mistaken for a write without changing global recipe memory behavior.
+ // TODO: Classify VPVectorEndPointerRecipe as non-memory in
+ // VPRecipeBase::mayReadFromMemory() and mayWriteToMemory(), then remove this
+ // exception. The shared fix can expose new VPlan CSE opportunities and needs
+ // dedicated code-generation tests.
+ switch (R.getVPRecipeID()) {
+ case VPRecipeBase::VPVectorEndPointerSC:
+ return false;
+ default:
+ return R.mayWriteToMemory();
+ }
+}
+
+/// Return whether \p Plan keeps the canonical IV increment in the symbolic
+/// VF * UF form required to model consecutive logical parts.
+static bool hasCanonicalIVIncrementForCrossPartCSE(VPlan &Plan) {
+ return vputils::findCanonicalIVIncrement(Plan);
+}
+
+} // namespace
+
+bool llvm::isCrossPartCSEProfitable(VPlan &Plan, ElementCount VF,
+ InstructionCost LoopCost,
+ VPCostContext &CostCtx,
+ const CrossPartCSEOptions &Options) {
+ assert(VF.isVector() && "cross-part analysis requires a vector VF");
+ assert(CostCtx.L && CostCtx.L->isInnermost() &&
+ "cross-part analysis requires an innermost loop");
+ assert(Plan.hasUF(CrossPartCSERequiredInterleaveCount) &&
+ "cross-part analysis requires support for UF=2");
+ assert(!Plan.isUnrolled() && "cross-part analysis requires symbolic UF");
+
+ // Narrowed plans replace symbolic VF * UF with a different effective step,
+ // so the selected VF no longer describes their physical per-part offset.
+ if (Plan.getVFxUF().isMaterialized())
+ return false;
+
+ // Reject an unspecified or impossible percentage before cost arithmetic.
+ if (Options.MinSavingPct == CrossPartCSEOptions::Unspecified ||
+ Options.MinSavingPct > 100)
+ return false;
+
+ VPRegionBlock *LoopRegion = Plan.getVectorLoopRegion();
+ if (!LoopRegion)
+ return false;
+
+ // Fail closed for every shape outside the exact single-block UF=2 model.
+ // TODO: Expand coverage by accepting additional plan shapes once their
+ // cross-part semantics can be modeled exactly.
+ if (!LoopCost.isValid() || LoopCost <= 0 ||
+ LoopRegion->getEntryBasicBlock() != LoopRegion->getExitingBasicBlock() ||
+ !hasCanonicalIVIncrementForCrossPartCSE(Plan))
+ return false;
+
+ using AvailableLoadMap =
+ DenseMap<CrossPartLoadKey, unsigned, CrossPartLoadKeyInfo>;
+ AvailableLoadMap AvailableLoadParts;
+ // A recipe may participate in multiple logical matches as supported shapes
+ // expand, but its local saving estimate is computed at most once.
+ DenseMap<const VPRecipeBase *, InstructionCost> SavingCosts;
+ CrossPartAddressBuilder Addresses(VF, CostCtx.PSE, CostCtx.L);
+#ifndef NDEBUG
+ // Count redundant-load opportunities only for diagnostics; profitability
+ // uses SavedCost.
+ unsigned NumOpportunities = 0;
+#endif
+ InstructionCost SavedCost = 0;
+
+ // Match VPlanUnroll's recipe-major UF=2 order. Clearing on every write
+ // enforces a strict no-write interval without alias disambiguation.
+ for (VPRecipeBase &R : *LoopRegion->getEntryBasicBlock()) {
+ if (isCrossPartWrite(R)) {
+ AvailableLoadParts.clear();
+ continue;
+ }
+
+ VPWidenLoadRecipe *Load = getCrossPartSupportedLoad(R);
+ if (!Load)
+ continue;
+
+ for (unsigned Part = 0; Part != CrossPartCSERequiredInterleaveCount;
+ ++Part) {
+ const SCEV *Address = Addresses.getAddress(*Load, Part);
+ if (isa<SCEVCouldNotCompute>(Address))
+ continue;
+
+ CrossPartLoadKey Key = {Address, Load->getScalarType()};
+ // Only reuse between different logical parts can justify raising IC from
+ // 1 to 2. A duplicate already seen in the same part also exists at IC=1
+ // and therefore provides no interleaving-specific saving.
+ unsigned PartBit = 1U << Part;
+ unsigned &AvailableParts = AvailableLoadParts[Key];
+ if (AvailableParts & PartBit)
+ continue;
+
+ bool HasOppositePart = (AvailableParts & ~PartBit) != 0;
+ AvailableParts |= PartBit;
+ if (!HasOppositePart) {
+ // Record the first occurrence in this part without assigning
+ // cross-part credit.
+ continue;
+ }
+
+ auto CostIt = SavingCosts.find(Load);
+ if (CostIt == SavingCosts.end())
+ CostIt = SavingCosts.try_emplace(Load, Load->cost(VF, CostCtx)).first;
+
+ // This estimate intentionally avoids retaining cost state from VF
+ // selection. It is exact for the directly costed widened loads supported
+ // here, but does not reproduce legacy attribution included in LoopCost.
+ // TODO: If measured profitability loses accuracy as supported recipes
+ // expand, consider passing cached costs from the selected-VF cost run.
+ // That would restore exact attribution at the cost of cross-phase state
+ // and recipe-lifetime management.
+ if (!CostIt->second.isValid() || CostIt->second <= 0)
+ continue;
+ SavedCost += CostIt->second;
+ LLVM_DEBUG(++NumOpportunities);
+ }
+ }
+
+ using CostType = InstructionCost::CostType;
+ bool Select = false;
+ if (SavedCost > 0) {
+ // Use InstructionCost arithmetic to preserve fractional cost units.
+ InstructionCost ScaledSavedCost = SavedCost * CostType(100);
+ InstructionCost RequiredCost =
+ LoopCost * CostType(CrossPartCSERequiredInterleaveCount);
+ RequiredCost *= CostType(Options.MinSavingPct);
+ Select = ScaledSavedCost >= RequiredCost;
+ }
+
+ LLVM_DEBUG({
+ CostType SavingPct = 0;
+ if (SavedCost.isValid() && SavedCost > 0 && LoopCost.isValid() &&
+ LoopCost > 0)
+ SavingPct = ((SavedCost * CostType(100)) /
+ (LoopCost * CostType(CrossPartCSERequiredInterleaveCount)))
+ .getValue();
+ dbgs() << "LV: Cross-part load redundancy estimate: opportunities="
+ << NumOpportunities << ", predicted-saved-cost=" << SavedCost
+ << ", loop-cost=" << LoopCost << ", saving=" << SavingPct
+ << "%, required=" << Options.MinSavingPct << "%; "
+ << (Select ? "selecting IC=2" : "skipping") << ".\n";
+ });
+ return Select;
+}
diff --git a/llvm/lib/Transforms/Vectorize/VPlanCrossPartCSE.h b/llvm/lib/Transforms/Vectorize/VPlanCrossPartCSE.h
new file mode 100644
index 0000000000000..391d32655b5c4
--- /dev/null
+++ b/llvm/lib/Transforms/Vectorize/VPlanCrossPartCSE.h
@@ -0,0 +1,57 @@
+//===- VPlanCrossPartCSE.h - Cross-part CSE for VPlan -----------*- C++ -*-===//
+//
+// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
+// See https://llvm.org/LICENSE.txt for license information.
+// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
+//
+//===----------------------------------------------------------------------===//
+//
+// This file declares prediction-only profitability analysis for exact load
+// redundancy across two modeled logical VPlan parts. It does not transform
+// VPlan or guarantee that a later pass will eliminate the redundant load.
+//
+//===----------------------------------------------------------------------===//
+
+#ifndef LLVM_TRANSFORMS_VECTORIZE_VPLANCROSSPARTCSE_H
+#define LLVM_TRANSFORMS_VECTORIZE_VPLANCROSSPARTCSE_H
+
+#include "llvm/Support/InstructionCost.h"
+#include "llvm/Support/TypeSize.h"
+#include <limits>
+
+namespace llvm {
+
+class VPlan;
+struct VPCostContext;
+
+/// The interleave count and logical unroll factor modeled by the analysis.
+constexpr unsigned CrossPartCSERequiredInterleaveCount = 2;
+
+/// Profitability criterion supplied by the caller.
+///
+/// The fail-closed default requires the caller to provide an explicit value.
+struct CrossPartCSEOptions {
+ /// Sentinel used until the caller supplies an explicit policy value.
+ static constexpr unsigned Unspecified = std::numeric_limits<unsigned>::max();
+
+ /// Minimum saving; the default rejects analysis until policy supplies it.
+ unsigned MinSavingPct = Unspecified;
+};
+
+/// Return whether exact cross-part load redundancy in \p Plan at \p VF meets
+/// \p Options.
+///
+/// The caller must establish that interleaving \p Plan is legal before using
+/// this opportunity estimate to raise its interleave count. A positive result
+/// does not guarantee that a later pass will eliminate the redundant load.
+/// The analysis reads \p Plan but takes a non-const reference because the VPlan
+/// query APIs it uses are not const-qualified.
+///
+/// \p CostCtx is local to the interleave decision and is not retained.
+bool isCrossPartCSEProfitable(VPlan &Plan, ElementCount VF,
+ InstructionCost LoopCost, VPCostContext &CostCtx,
+ const CrossPartCSEOptions &Options);
+
+} // namespace llvm
+
+#endif // LLVM_TRANSFORMS_VECTORIZE_VPLANCROSSPARTCSE_H
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-debug.ll b/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-debug.ll
new file mode 100644
index 0000000000000..a8cf27509988f
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-debug.ll
@@ -0,0 +1,263 @@
+; REQUIRES: asserts
+; RUN: split-file %s %t
+;
+; When cross-part analysis selects IC=2 after the ordinary heuristics decline
+; interleaving, it must not emit a contradictory non-interleaving diagnostic.
+; RUN: opt -passes=loop-vectorize -force-vector-width=4 \
+; RUN: -force-target-max-vector-interleave=2 -small-loop-cost=0 \
+; RUN: -enable-interleave-cse \
+; RUN: -interleave-cse-min-pct=1 \
+; RUN: -debug-only=loop-vectorize -disable-output %t/success.ll 2>&1 \
+; RUN: | FileCheck %t/success.ll --check-prefix=SUCCESS
+;
+; A fixed-VF tail-folded plan is ineligible for interleaving. Cross-part
+; analysis must not override that policy or emit a profitability estimate.
+; The same holds when partial-alias masking additionally forces IC=1, which
+; requires runtime difference checks and therefore a second checked pointer.
+; RUN: opt -passes=loop-vectorize -force-vector-width=4 \
+; RUN: -force-target-max-vector-interleave=2 -small-loop-cost=0 \
+; RUN: -force-target-supports-masked-memory-ops \
+; RUN: -force-tail-folding-style=data-and-control \
+; RUN: -tail-folding-policy=must-fold-tail \
+; RUN: -force-partial-aliasing-vectorization \
+; RUN: -enable-interleave-cse \
+; RUN: -interleave-cse-min-pct=1 -debug-only=loop-vectorize \
+; RUN: -disable-output %t/success.ll 2>&1 \
+; RUN: | FileCheck %t/success.ll --check-prefixes=MASKED,ALIAS
+;
+; When the ordinary branch-cost heuristic recommends IC=1, a successful
+; cross-part selection must return before emitting its baseline diagnostic.
+; RUN: opt -passes=loop-vectorize -force-vector-width=4 \
+; RUN: -force-target-max-vector-interleave=2 \
+; RUN: -force-target-instruction-cost=1 -small-loop-cost=12 \
+; RUN: -enable-loadstore-runtime-interleave=false \
+; RUN: -enable-interleave-cse \
+; RUN: -interleave-cse-min-pct=1 \
+; RUN: -debug-only=loop-vectorize -disable-output %t/success.ll 2>&1 \
+; RUN: | FileCheck %t/success.ll --check-prefix=SUCCESS-SMALL
+;
+; The same-part duplicate after a genuine cross-part match must report exactly
+; one opportunity and must not double the saving past the 6% threshold.
+; RUN: opt -passes=loop-vectorize -force-vector-width=4 \
+; RUN: -force-target-max-vector-interleave=2 -small-loop-cost=0 \
+; RUN: -force-target-instruction-cost=1 \
+; RUN: -enable-interleave-cse -interleave-cse-min-pct=6 \
+; RUN: -debug-only=loop-vectorize \
+; RUN: -disable-output %t/duplicate.ll 2>&1 \
+; RUN: | FileCheck %t/duplicate.ll
+;
+; A predicted saving exactly equal to the configured threshold is sufficient,
+; preserving the inclusive >= comparison.
+; RUN: opt -passes=loop-vectorize -force-vector-width=4 \
+; RUN: -force-target-max-vector-interleave=2 -small-loop-cost=0 \
+; RUN: -force-target-instruction-cost=1 \
+; RUN: -enable-interleave-cse -interleave-cse-min-pct=5 \
+; RUN: -debug-only=loop-vectorize \
+; RUN: -disable-output %t/threshold.ll 2>&1 \
+; RUN: | FileCheck %t/threshold.ll
+;
+; With cross-part analysis disabled, the ordinary branch-cost diagnostic
+; remains unchanged.
+; RUN: opt -passes=loop-vectorize -force-vector-width=4 \
+; RUN: -force-target-max-vector-interleave=2 \
+; RUN: -force-target-instruction-cost=1 -small-loop-cost=12 \
+; RUN: -enable-loadstore-runtime-interleave=false \
+; RUN: -debug-only=loop-vectorize -disable-output %t/success.ll 2>&1 \
+; RUN: | FileCheck %t/success.ll --check-prefix=DISABLED-SMALL
+;
+; A constant source offset cannot equal the scalable part offset for every
+; runtime vscale, so the exact analysis reports no opportunity and keeps UF=1.
+; RUN: opt -passes=loop-vectorize -mtriple=aarch64-none-linux-gnu -mattr=+sve \
+; RUN: -force-vector-width="vscale x 2" \
+; RUN: -force-target-max-vector-interleave=2 -small-loop-cost=0 \
+; RUN: -enable-interleave-cse \
+; RUN: -interleave-cse-min-pct=1 -debug-only=loop-vectorize \
+; RUN: -disable-output %t/success.ll 2>&1 \
+; RUN: | FileCheck %t/success.ll --check-prefix=SCALABLE
+;
+; A narrowed plan materializes VFxUF before IC selection. The redundancy
+; analysis must return before emitting an estimate for that unsupported shape.
+; RUN: opt -passes=loop-vectorize -mtriple=arm64-apple-macosx \
+; RUN: -force-vector-width=2 -force-target-max-vector-interleave=2 \
+; RUN: -force-target-num-vector-regs=1024 \
+; RUN: -force-target-instruction-cost=1 -small-loop-cost=0 \
+; RUN: -enable-interleave-cse -interleave-cse-min-pct=1 \
+; RUN: -debug-only=loop-vectorize -disable-output \
+; RUN: %S/cross-part-load-cse-narrowed.ll 2>&1 \
+; RUN: | FileCheck %s --check-prefix=NARROWED
+;
+; NARROWED-LABEL: LV: Checking a loop in 'narrowed'
+; NARROWED-NOT: LV: Cross-part load redundancy estimate:
+; NARROWED: Executing best plan with VF=2, UF=1
+
+;--- success.ll
+; Every prefix below closes its 'positive' block with a second -LABEL line, so
+; that the checks cannot be satisfied by the 'partial_alias' log that follows.
+;
+; SUCCESS-LABEL: LV: Checking a loop in 'positive'
+; SUCCESS: LV: Cross-part load redundancy estimate: opportunities=1, predicted-saved-cost={{[^,]+}}, loop-cost={{[^,]+}}, saving={{[0-9]+}}%, required=1%; selecting IC=2.
+; SUCCESS-NEXT: LV: Exact cross-part load redundancy predicts a downstream saving; raising IC to 2.
+; SUCCESS-NOT: LV: Not Interleaving.
+; SUCCESS: LV: Found a vectorizable loop
+; SUCCESS: Executing best plan with VF=4, UF=2
+; SUCCESS-LABEL: LV: Checking a loop in 'partial_alias'
+;
+; MASKED-LABEL: LV: Checking a loop in 'positive'
+; MASKED-NOT: LV: Cross-part load redundancy estimate:
+; MASKED-NOT: Exact cross-part load redundancy predicts a downstream saving
+; MASKED-NOT: LV: Not interleaving due to partial aliasing vectorization.
+; MASKED: Executing best plan with VF=4, UF=1
+;
+; The ALIAS-LABEL line below also closes the MASKED block above, because
+; FileCheck partitions the input at the -LABEL lines of every active prefix.
+; ALIAS-LABEL: LV: Checking a loop in 'partial_alias'
+; ALIAS-NOT: LV: Cross-part load redundancy estimate:
+; ALIAS-NOT: Exact cross-part load redundancy predicts a downstream saving
+; ALIAS: LV: Not interleaving due to partial aliasing vectorization.
+; ALIAS: Executing best plan with VF=4, UF=1
+;
+; SUCCESS-SMALL-LABEL: LV: Checking a loop in 'positive'
+; SUCCESS-SMALL: LV: Cross-part load redundancy estimate: opportunities=1, predicted-saved-cost={{[^,]+}}, loop-cost={{[^,]+}}, saving={{[0-9]+}}%, required=1%; selecting IC=2.
+; SUCCESS-SMALL-NEXT: LV: Exact cross-part load redundancy predicts a downstream saving; raising IC to 2.
+; SUCCESS-SMALL-NOT: LV: Interleaving to reduce branch cost.
+; SUCCESS-SMALL: LV: Found a vectorizable loop
+; SUCCESS-SMALL: Executing best plan with VF=4, UF=2
+; SUCCESS-SMALL-LABEL: LV: Checking a loop in 'partial_alias'
+;
+; DISABLED-SMALL-LABEL: LV: Checking a loop in 'positive'
+; DISABLED-SMALL-NOT: Cross-part load redundancy
+; DISABLED-SMALL: LV: Interleaving to reduce branch cost.
+; DISABLED-SMALL-NOT: Cross-part load redundancy
+; DISABLED-SMALL: LV: Found a vectorizable loop
+; DISABLED-SMALL: Executing best plan with VF=4, UF=1
+; DISABLED-SMALL-LABEL: LV: Checking a loop in 'partial_alias'
+;
+; SCALABLE-LABEL: LV: Checking a loop in 'positive'
+; SCALABLE-NOT: Exact cross-part load redundancy predicts a downstream saving
+; SCALABLE: LV: VF is vscale x 2
+; SCALABLE-NEXT: LV: Cross-part load redundancy estimate: opportunities=0, predicted-saved-cost=0, loop-cost={{[^,]+}}, saving=0%, required=1%; skipping.
+; SCALABLE-NEXT: LV: Not Interleaving.
+; SCALABLE-NOT: Exact cross-part load redundancy predicts a downstream saving
+; SCALABLE: LV: Found a vectorizable loop (vscale x 2)
+; SCALABLE: Executing best plan with VF=vscale x 2, UF=1
+; SCALABLE-LABEL: LV: Checking a loop in 'partial_alias'
+
+target triple = "aarch64-unknown-linux-gnu"
+
+define void @positive(ptr noalias %a, ptr noalias %c, i64 %n) {
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %a.iv = getelementptr inbounds i32, ptr %a, i64 %iv
+ %l1 = load i32, ptr %a.iv, align 4
+ %iv.plus.4 = add nuw nsw i64 %iv, 4
+ %a.iv.plus.4 = getelementptr inbounds i32, ptr %a, i64 %iv.plus.4
+ %l2 = load i32, ptr %a.iv.plus.4, align 4
+ %sum = add i32 %l1, %l2
+ %c.iv = getelementptr inbounds i32, ptr %c, i64 %iv
+ store i32 %sum, ptr %c.iv, align 4
+ %iv.next = add nuw nsw i64 %iv, 1
+ %cmp = icmp slt i64 %iv.next, %n
+ br i1 %cmp, label %loop, label %exit
+
+exit:
+ ret void
+}
+
+; The %b/%c pair needs a runtime difference check, which enables partial-alias
+; masking, while the cross-part reuse candidate on %a stays present.
+define void @partial_alias(ptr noalias %a, ptr %b, ptr %c, i64 %n) {
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %a.iv = getelementptr inbounds i32, ptr %a, i64 %iv
+ %l1 = load i32, ptr %a.iv, align 4
+ %iv.plus.4 = add nuw nsw i64 %iv, 4
+ %a.iv.plus.4 = getelementptr inbounds i32, ptr %a, i64 %iv.plus.4
+ %l2 = load i32, ptr %a.iv.plus.4, align 4
+ %b.iv = getelementptr inbounds i32, ptr %b, i64 %iv
+ %l3 = load i32, ptr %b.iv, align 4
+ %sum.1 = add i32 %l1, %l2
+ %sum.2 = add i32 %sum.1, %l3
+ %c.iv = getelementptr inbounds i32, ptr %c, i64 %iv
+ store i32 %sum.2, ptr %c.iv, align 4
+ %iv.next = add nuw nsw i64 %iv, 1
+ %cmp = icmp slt i64 %iv.next, %n
+ br i1 %cmp, label %loop, label %exit
+
+exit:
+ ret void
+}
+
+;--- threshold.ll
+; CHECK-LABEL: LV: Checking a loop in 'threshold_equal'
+; CHECK: LV: Cross-part load redundancy estimate: opportunities=1, predicted-saved-cost=1, loop-cost=10, saving=5%, required=5%; selecting IC=2.
+; CHECK-NEXT: LV: Exact cross-part load redundancy predicts a downstream saving; raising IC to 2.
+; CHECK: Executing best plan with VF=4, UF=2
+
+target triple = "aarch64-unknown-linux-gnu"
+
+; With -force-target-instruction-cost=1 the modeled vector body costs exactly
+; 10: scalar steps, three address computations, two widened loads, one add, one
+; widened store, the backedge, and the canonical IV increment. The second
+; address is derived from a loop-invariant %a + 4 so that no in-loop index
+; arithmetic is costed. The single redundant load saves exactly 1, so the
+; comparison is 1 * 100 == 10 * 2 * 5, i.e. exact equality with the 5%
+; threshold.
+define void @threshold_equal(ptr noalias %a, ptr noalias %c, i64 %n) {
+entry:
+ %a.plus.4 = getelementptr inbounds i32, ptr %a, i64 4
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %a.iv = getelementptr inbounds i32, ptr %a, i64 %iv
+ %l1 = load i32, ptr %a.iv, align 4
+ %a.iv.plus.4 = getelementptr inbounds i32, ptr %a.plus.4, i64 %iv
+ %l2 = load i32, ptr %a.iv.plus.4, align 4
+ %sum = add i32 %l1, %l2
+ %c.iv = getelementptr inbounds i32, ptr %c, i64 %iv
+ store i32 %sum, ptr %c.iv, align 4
+ %iv.next = add nuw nsw i64 %iv, 1
+ %cmp = icmp slt i64 %iv.next, %n
+ br i1 %cmp, label %loop, label %exit
+
+exit:
+ ret void
+}
+
+;--- duplicate.ll
+; CHECK-LABEL: LV: Checking a loop in 'duplicate_after_cross_part'
+; CHECK: LV: Cross-part load redundancy estimate: opportunities=1, predicted-saved-cost={{[^,]+}}, loop-cost={{[^,]+}}, saving={{[0-9]+}}%, required=6%; skipping.
+; CHECK-NOT: Exact cross-part load redundancy predicts a downstream saving
+; CHECK: LV: Found a vectorizable loop
+
+target triple = "aarch64-unknown-linux-gnu"
+
+define void @duplicate_after_cross_part(ptr noalias %a, ptr noalias %c, i64 %n) {
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %a.iv = getelementptr inbounds i32, ptr %a, i64 %iv
+ %l1 = load i32, ptr %a.iv, align 4
+ %iv.plus.4 = add nuw nsw i64 %iv, 4
+ %a.iv.plus.4 = getelementptr inbounds i32, ptr %a, i64 %iv.plus.4
+ %l2 = load i32, ptr %a.iv.plus.4, align 4
+ %l3 = load i32, ptr %a.iv.plus.4, align 4
+ %sum.1 = add i32 %l1, %l2
+ %sum.2 = add i32 %sum.1, %l3
+ %c.iv = getelementptr inbounds i32, ptr %c, i64 %iv
+ store i32 %sum.2, ptr %c.iv, align 4
+ %iv.next = add nuw nsw i64 %iv, 1
+ %cmp = icmp slt i64 %iv.next, %n
+ br i1 %cmp, label %loop, label %exit
+
+exit:
+ ret void
+}
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-fixed-cases.ll b/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-fixed-cases.ll
index 15948347005af..1de6bbd58dde2 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-fixed-cases.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-fixed-cases.ll
@@ -1,11 +1,9 @@
; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
-; Pre-commit test for the properties that a cross-part load redundancy
-; analysis has to respect: loaded value type, exact address equality,
-; intervening memory writes, access stride, address provenance,
-; poison-generating metadata, multiple scalar loop blocks and reverse
-; accesses. Every loop below vectorizes with an interleave count of 1 today.
+; The minimum saving is lowered to 1% so that every interleave count of 1 below
+; is caused by the modeled property under test rather than by the threshold.
; RUN: opt -passes=loop-vectorize -force-vector-width=4 \
; RUN: -force-target-max-vector-interleave=2 -small-loop-cost=0 \
+; RUN: -enable-interleave-cse -interleave-cse-min-pct=1 \
; RUN: -S %s | FileCheck %s
target triple = "aarch64-unknown-linux-gnu"
@@ -14,7 +12,8 @@ declare i32 @write_memory(i32) #0
declare <4 x i32> @write_memory_v4(<4 x i32>)
; The part-shifted addresses are equal, but the two loads have different value
-; types and therefore cannot share a loaded result.
+; types and therefore cannot share a loaded result. The value type is part of
+; the redundancy key, so the interleave count stays 1.
define void @different_types(ptr noalias %a, ptr noalias %c, i64 %n) {
; CHECK-LABEL: define void @different_types(
; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) {
@@ -86,7 +85,7 @@ exit:
}
; a[i] + a[i+3]: the source offset 3 is not a multiple of VF=4, so no
-; part-shifted address ever matches exactly.
+; part-shifted address ever matches exactly and the interleave count stays 1.
define void @inequality(ptr noalias %a, ptr noalias %c, i64 %n) {
; CHECK-LABEL: define void @inequality(
; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) {
@@ -155,48 +154,49 @@ exit:
}
; A may-alias store occurs between the two matching logical-part loads, so their
-; values cannot be assumed to survive from one part to the other.
+; values cannot be assumed to survive from one part to the other. No redundancy
+; opportunity is credited, and the interleave count stays 1.
define void @write_between(ptr %a, ptr %b, ptr noalias %c, i64 %n) {
; CHECK-LABEL: define void @write_between(
; CHECK-SAME: ptr [[A:%.*]], ptr [[B:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) {
; CHECK-NEXT: [[ENTRY:.*]]:
; CHECK-NEXT: [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 4
-; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_MEMCHECK:.*]]
-; CHECK: [[VECTOR_MEMCHECK]]:
+; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
; CHECK-NEXT: [[SMAX:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
-; CHECK-NEXT: [[TMP1:%.*]] = shl i64 [[SMAX]], 2
-; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[B]], i64 [[TMP1]]
-; CHECK-NEXT: [[TMP2:%.*]] = add i64 [[TMP1]], 16
-; CHECK-NEXT: [[SCEVGEP1:%.*]] = getelementptr i8, ptr [[A]], i64 [[TMP2]]
+; CHECK-NEXT: [[TMP9:%.*]] = shl i64 [[SMAX]], 2
+; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[B]], i64 [[TMP9]]
+; CHECK-NEXT: [[TMP10:%.*]] = add i64 [[TMP9]], 16
+; CHECK-NEXT: [[SCEVGEP1:%.*]] = getelementptr i8, ptr [[A]], i64 [[TMP10]]
; CHECK-NEXT: [[BOUND0:%.*]] = icmp ult ptr [[B]], [[SCEVGEP1]]
; CHECK-NEXT: [[BOUND1:%.*]] = icmp ult ptr [[A]], [[SCEVGEP]]
; CHECK-NEXT: [[FOUND_CONFLICT:%.*]] = and i1 [[BOUND0]], [[BOUND1]]
-; CHECK-NEXT: br i1 [[FOUND_CONFLICT]], label %[[SCALAR_PH]], label %[[VECTOR_PH:.*]]
-; CHECK: [[VECTOR_PH]]:
-; CHECK-NEXT: [[TMP3:%.*]] = and i64 [[TMP0]], 3
-; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP3]]
+; CHECK-NEXT: br i1 [[FOUND_CONFLICT]], label %[[SCALAR_PH]], label %[[VECTOR_PH1:.*]]
+; CHECK: [[VECTOR_PH1]]:
+; CHECK-NEXT: [[TMP1:%.*]] = and i64 [[TMP0]], 3
+; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
-; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT: [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
-; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP4]], align 4, !alias.scope [[META6:![0-9]+]]
-; CHECK-NEXT: [[TMP5:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 [[INDEX]]
-; CHECK-NEXT: store <4 x i32> [[WIDE_LOAD]], ptr [[TMP5]], align 4, !alias.scope [[META9:![0-9]+]], !noalias [[META6]]
-; CHECK-NEXT: [[TMP6:%.*]] = add nuw nsw i64 [[INDEX]], 4
-; CHECK-NEXT: [[TMP7:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP6]]
-; CHECK-NEXT: [[WIDE_LOAD2:%.*]] = load <4 x i32>, ptr [[TMP7]], align 4, !alias.scope [[META6]]
-; CHECK-NEXT: [[TMP8:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[WIDE_LOAD2]]
-; CHECK-NEXT: [[TMP9:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
-; CHECK-NEXT: store <4 x i32> [[TMP8]], ptr [[TMP9]], align 4
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH1]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4, !alias.scope [[META6:![0-9]+]]
+; CHECK-NEXT: [[TMP3:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 [[INDEX]]
+; CHECK-NEXT: store <4 x i32> [[WIDE_LOAD]], ptr [[TMP3]], align 4, !alias.scope [[META9:![0-9]+]], !noalias [[META6]]
+; CHECK-NEXT: [[TMP4:%.*]] = add nuw nsw i64 [[INDEX]], 4
+; CHECK-NEXT: [[TMP5:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP4]]
+; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP5]], align 4, !alias.scope [[META6]]
+; CHECK-NEXT: [[TMP6:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[WIDE_LOAD1]]
+; CHECK-NEXT: [[TMP7:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
+; CHECK-NEXT: store <4 x i32> [[TMP6]], ptr [[TMP7]], align 4
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
-; CHECK-NEXT: [[TMP10:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-NEXT: br i1 [[TMP10]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP11:![0-9]+]]
+; CHECK-NEXT: [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP11:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
; CHECK-NEXT: br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
; CHECK: [[SCALAR_PH]]:
-; CHECK-NEXT: [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ], [ 0, %[[VECTOR_MEMCHECK]] ]
+; CHECK-NEXT: [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ], [ 0, %[[VECTOR_PH]] ]
; CHECK-NEXT: br label %[[LOOP:.*]]
; CHECK: [[LOOP]]:
; CHECK-NEXT: [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
@@ -240,8 +240,8 @@ exit:
}
; A vector-mapped call with unknown memory effects occurs between the matching
-; loads. It may write through memory that its arguments do not describe, so no
-; loaded value may be assumed to survive across it.
+; loads. It may write through memory not represented by its arguments, so it
+; must clear the available-load map and leave the interleave count at 1.
define void @writing_call_between(ptr noalias %a, ptr noalias %c, i64 %n) {
; CHECK-LABEL: define void @writing_call_between(
; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) {
@@ -313,7 +313,8 @@ exit:
}
; Stride-2 accesses are represented by interleave-group recipes rather than by
-; simple consecutive widened loads.
+; simple consecutive widened loads, so the analysis fails closed and leaves the
+; interleave count at 1.
define void @non_unit_stride(ptr noalias %a, ptr noalias %c, i64 %n) {
; CHECK-LABEL: define void @non_unit_stride(
; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) {
@@ -387,32 +388,40 @@ exit:
ret void
}
-; VPlan folds each identical-arm select to its underlying GEP, while
-; ScalarEvolution keeps each nonconstant pointer select as a distinct unknown.
-; The address of these loads is therefore only exact in the folded VPlan value,
-; not in the scalar load ingredient.
+; VPlan folds each identical-arm select to its underlying GEP. ScalarEvolution
+; keeps each nonconstant pointer select as a distinct unknown, so deriving the
+; address from the scalar load ingredient would miss the exact match. Deriving
+; it from the folded VPlan value exposes cross-part load redundancy and raises
+; IC to 2.
define void @folded_provenance(ptr noalias %a, ptr noalias %c, i64 %n, i1 %cond) {
; CHECK-LABEL: define void @folded_provenance(
; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]], i1 [[COND:%.*]]) {
; CHECK-NEXT: [[ENTRY:.*]]:
; CHECK-NEXT: [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
-; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 4
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 8
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
; CHECK: [[VECTOR_PH]]:
-; CHECK-NEXT: [[TMP1:%.*]] = and i64 [[TMP0]], 3
+; CHECK-NEXT: [[TMP1:%.*]] = and i64 [[TMP0]], 7
; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
; CHECK-NEXT: [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT: [[TMP9:%.*]] = getelementptr inbounds i32, ptr [[TMP2]], i64 4
; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4
+; CHECK-NEXT: [[WIDE_LOAD2:%.*]] = load <4 x i32>, ptr [[TMP9]], align 4
; CHECK-NEXT: [[TMP3:%.*]] = add nuw nsw i64 [[INDEX]], 4
; CHECK-NEXT: [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP3]]
+; CHECK-NEXT: [[TMP11:%.*]] = getelementptr inbounds i32, ptr [[TMP4]], i64 4
; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP4]], align 4
+; CHECK-NEXT: [[WIDE_LOAD3:%.*]] = load <4 x i32>, ptr [[TMP11]], align 4
; CHECK-NEXT: [[TMP5:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[WIDE_LOAD1]]
+; CHECK-NEXT: [[TMP8:%.*]] = add <4 x i32> [[WIDE_LOAD2]], [[WIDE_LOAD3]]
; CHECK-NEXT: [[TMP6:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
+; CHECK-NEXT: [[TMP10:%.*]] = getelementptr inbounds i32, ptr [[TMP6]], i64 4
; CHECK-NEXT: store <4 x i32> [[TMP5]], ptr [[TMP6]], align 4
-; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT: store <4 x i32> [[TMP8]], ptr [[TMP10]], align 4
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
; CHECK-NEXT: [[TMP7:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
; CHECK-NEXT: br i1 [[TMP7]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP17:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
@@ -463,29 +472,38 @@ exit:
}
; The two scalar loads carry different poison-generating !range metadata, which
-; is not propagated to the widened loads.
+; is not propagated to the widened loads. Exact VPlan address and type equality
+; therefore raises the interleave count to 2 without consulting the scalar
+; loads' metadata.
define void @poison_annotations(ptr noalias %a, ptr noalias %c, i64 %n) {
; CHECK-LABEL: define void @poison_annotations(
; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) {
; CHECK-NEXT: [[ENTRY:.*]]:
; CHECK-NEXT: [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
-; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 4
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 8
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
; CHECK: [[VECTOR_PH]]:
-; CHECK-NEXT: [[TMP1:%.*]] = and i64 [[TMP0]], 3
+; CHECK-NEXT: [[TMP1:%.*]] = and i64 [[TMP0]], 7
; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
; CHECK-NEXT: [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT: [[TMP9:%.*]] = getelementptr inbounds i32, ptr [[TMP2]], i64 4
; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4
+; CHECK-NEXT: [[WIDE_LOAD2:%.*]] = load <4 x i32>, ptr [[TMP9]], align 4
; CHECK-NEXT: [[TMP3:%.*]] = add nuw nsw i64 [[INDEX]], 4
; CHECK-NEXT: [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP3]]
+; CHECK-NEXT: [[TMP11:%.*]] = getelementptr inbounds i32, ptr [[TMP4]], i64 4
; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP4]], align 4
+; CHECK-NEXT: [[WIDE_LOAD3:%.*]] = load <4 x i32>, ptr [[TMP11]], align 4
; CHECK-NEXT: [[TMP5:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[WIDE_LOAD1]]
+; CHECK-NEXT: [[TMP8:%.*]] = add <4 x i32> [[WIDE_LOAD2]], [[WIDE_LOAD3]]
; CHECK-NEXT: [[TMP6:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
+; CHECK-NEXT: [[TMP10:%.*]] = getelementptr inbounds i32, ptr [[TMP6]], i64 4
; CHECK-NEXT: store <4 x i32> [[TMP5]], ptr [[TMP6]], align 4
-; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT: store <4 x i32> [[TMP8]], ptr [[TMP10]], align 4
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
; CHECK-NEXT: [[TMP7:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
; CHECK-NEXT: br i1 [[TMP7]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP19:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
@@ -531,30 +549,38 @@ exit:
ret void
}
-; The scalar loop has a separate latch, while VPlan places the two matching
-; loads in its single vector-loop block.
+; The scalar loop has a separate latch, but VPlan places the matching loads in
+; its single vector-loop block. The VPlan structural check therefore accepts
+; the loop and the redundancy raises the interleave count to 2.
define void @multi_block(ptr noalias %a, ptr noalias %c, i64 %n) {
; CHECK-LABEL: define void @multi_block(
; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) {
; CHECK-NEXT: [[ENTRY:.*]]:
; CHECK-NEXT: [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
-; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 4
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 8
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
; CHECK: [[VECTOR_PH]]:
-; CHECK-NEXT: [[TMP1:%.*]] = and i64 [[TMP0]], 3
+; CHECK-NEXT: [[TMP1:%.*]] = and i64 [[TMP0]], 7
; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
; CHECK-NEXT: [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT: [[TMP9:%.*]] = getelementptr inbounds i32, ptr [[TMP2]], i64 4
; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4
+; CHECK-NEXT: [[WIDE_LOAD2:%.*]] = load <4 x i32>, ptr [[TMP9]], align 4
; CHECK-NEXT: [[TMP3:%.*]] = add nuw nsw i64 [[INDEX]], 4
; CHECK-NEXT: [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP3]]
+; CHECK-NEXT: [[TMP11:%.*]] = getelementptr inbounds i32, ptr [[TMP4]], i64 4
; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP4]], align 4
+; CHECK-NEXT: [[WIDE_LOAD3:%.*]] = load <4 x i32>, ptr [[TMP11]], align 4
; CHECK-NEXT: [[TMP5:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[WIDE_LOAD1]]
+; CHECK-NEXT: [[TMP8:%.*]] = add <4 x i32> [[WIDE_LOAD2]], [[WIDE_LOAD3]]
; CHECK-NEXT: [[TMP6:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
+; CHECK-NEXT: [[TMP10:%.*]] = getelementptr inbounds i32, ptr [[TMP6]], i64 4
; CHECK-NEXT: store <4 x i32> [[TMP5]], ptr [[TMP6]], align 4
-; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT: store <4 x i32> [[TMP8]], ptr [[TMP10]], align 4
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
; CHECK-NEXT: [[TMP7:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
; CHECK-NEXT: br i1 [[TMP7]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP23:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
@@ -606,7 +632,8 @@ exit:
}
; At VF=4, the first load in part 1 and the second load in part 0 both use the
-; reverse vector ending at a[last-iv-7].
+; reverse vector ending at a[last-iv-7]. Their exact redundancy raises the
+; interleave count to 2.
define void @reverse_redundancy(ptr noalias %a, ptr noalias %c, i64 %n) {
; CHECK-LABEL: define void @reverse_redundancy(
; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) {
@@ -614,10 +641,10 @@ define void @reverse_redundancy(ptr noalias %a, ptr noalias %c, i64 %n) {
; CHECK-NEXT: [[LAST:%.*]] = add i64 [[N]], -1
; CHECK-NEXT: [[TRIP_COUNT:%.*]] = add i64 [[N]], -4
; CHECK-NEXT: [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[TRIP_COUNT]], i64 1)
-; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 4
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 8
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
; CHECK: [[VECTOR_PH]]:
-; CHECK-NEXT: [[TMP1:%.*]] = and i64 [[TMP0]], 3
+; CHECK-NEXT: [[TMP1:%.*]] = and i64 [[TMP0]], 7
; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
@@ -625,18 +652,26 @@ define void @reverse_redundancy(ptr noalias %a, ptr noalias %c, i64 %n) {
; CHECK-NEXT: [[TMP2:%.*]] = sub i64 [[LAST]], [[INDEX]]
; CHECK-NEXT: [[TMP3:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP2]]
; CHECK-NEXT: [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[TMP3]], i64 -3
+; CHECK-NEXT: [[TMP5:%.*]] = getelementptr inbounds i32, ptr [[TMP3]], i64 -7
; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP4]], align 4
-; CHECK-NEXT: [[TMP5:%.*]] = add i64 [[TMP2]], -4
-; CHECK-NEXT: [[TMP6:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP5]]
-; CHECK-NEXT: [[TMP7:%.*]] = getelementptr inbounds i32, ptr [[TMP6]], i64 -3
-; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP7]], align 4
-; CHECK-NEXT: [[TMP8:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[WIDE_LOAD1]]
-; CHECK-NEXT: [[REVERSE:%.*]] = shufflevector <4 x i32> [[TMP8]], <4 x i32> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
-; CHECK-NEXT: [[TMP9:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
-; CHECK-NEXT: store <4 x i32> [[REVERSE]], ptr [[TMP9]], align 4
-; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
-; CHECK-NEXT: [[TMP10:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-NEXT: br i1 [[TMP10]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP25:![0-9]+]]
+; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP5]], align 4
+; CHECK-NEXT: [[TMP6:%.*]] = add i64 [[TMP2]], -4
+; CHECK-NEXT: [[TMP7:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP6]]
+; CHECK-NEXT: [[TMP8:%.*]] = getelementptr inbounds i32, ptr [[TMP7]], i64 -3
+; CHECK-NEXT: [[TMP9:%.*]] = getelementptr inbounds i32, ptr [[TMP7]], i64 -7
+; CHECK-NEXT: [[WIDE_LOAD2:%.*]] = load <4 x i32>, ptr [[TMP8]], align 4
+; CHECK-NEXT: [[WIDE_LOAD3:%.*]] = load <4 x i32>, ptr [[TMP9]], align 4
+; CHECK-NEXT: [[TMP10:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[WIDE_LOAD2]]
+; CHECK-NEXT: [[TMP11:%.*]] = add <4 x i32> [[WIDE_LOAD1]], [[WIDE_LOAD3]]
+; CHECK-NEXT: [[REVERSE:%.*]] = shufflevector <4 x i32> [[TMP10]], <4 x i32> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
+; CHECK-NEXT: [[REVERSE4:%.*]] = shufflevector <4 x i32> [[TMP11]], <4 x i32> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
+; CHECK-NEXT: [[TMP12:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
+; CHECK-NEXT: [[TMP13:%.*]] = getelementptr inbounds i32, ptr [[TMP12]], i64 4
+; CHECK-NEXT: store <4 x i32> [[REVERSE]], ptr [[TMP12]], align 4
+; CHECK-NEXT: store <4 x i32> [[REVERSE4]], ptr [[TMP13]], align 4
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
+; CHECK-NEXT: [[TMP14:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[TMP14]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP25:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
; CHECK-NEXT: br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
@@ -686,36 +721,48 @@ exit:
; The reverse access sits physically between the two matching forward loads.
; Its end-pointer address computation is pure and must not be mistaken for a
-; memory write.
+; memory write. The one forward equality therefore raises the interleave count
+; to 2.
define void @reverse_between(ptr noalias %a, ptr noalias %c, i64 %n) {
; CHECK-LABEL: define void @reverse_between(
; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) {
; CHECK-NEXT: [[ENTRY:.*]]:
; CHECK-NEXT: [[LAST:%.*]] = add i64 [[N]], -1
; CHECK-NEXT: [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
-; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 4
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 8
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
; CHECK: [[VECTOR_PH]]:
-; CHECK-NEXT: [[TMP1:%.*]] = and i64 [[TMP0]], 3
+; CHECK-NEXT: [[TMP1:%.*]] = and i64 [[TMP0]], 7
; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
; CHECK-NEXT: [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT: [[TMP13:%.*]] = getelementptr inbounds i32, ptr [[TMP2]], i64 4
; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4
+; CHECK-NEXT: [[WIDE_LOAD2:%.*]] = load <4 x i32>, ptr [[TMP13]], align 4
; CHECK-NEXT: [[TMP3:%.*]] = sub i64 [[LAST]], [[INDEX]]
; CHECK-NEXT: [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP3]]
; CHECK-NEXT: [[TMP5:%.*]] = getelementptr inbounds i32, ptr [[TMP4]], i64 -3
+; CHECK-NEXT: [[TMP15:%.*]] = getelementptr inbounds i32, ptr [[TMP4]], i64 -7
; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP5]], align 4
+; CHECK-NEXT: [[WIDE_LOAD4:%.*]] = load <4 x i32>, ptr [[TMP15]], align 4
; CHECK-NEXT: [[REVERSE2:%.*]] = shufflevector <4 x i32> [[WIDE_LOAD1]], <4 x i32> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
+; CHECK-NEXT: [[REVERSE5:%.*]] = shufflevector <4 x i32> [[WIDE_LOAD4]], <4 x i32> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
; CHECK-NEXT: [[TMP6:%.*]] = add nuw nsw i64 [[INDEX]], 4
; CHECK-NEXT: [[TMP7:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP6]]
+; CHECK-NEXT: [[TMP17:%.*]] = getelementptr inbounds i32, ptr [[TMP7]], i64 4
; CHECK-NEXT: [[WIDE_LOAD3:%.*]] = load <4 x i32>, ptr [[TMP7]], align 4
+; CHECK-NEXT: [[WIDE_LOAD7:%.*]] = load <4 x i32>, ptr [[TMP17]], align 4
; CHECK-NEXT: [[TMP8:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[WIDE_LOAD3]]
+; CHECK-NEXT: [[TMP12:%.*]] = add <4 x i32> [[WIDE_LOAD2]], [[WIDE_LOAD7]]
; CHECK-NEXT: [[TMP9:%.*]] = add <4 x i32> [[TMP8]], [[REVERSE2]]
+; CHECK-NEXT: [[TMP14:%.*]] = add <4 x i32> [[TMP12]], [[REVERSE5]]
; CHECK-NEXT: [[TMP10:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
+; CHECK-NEXT: [[TMP16:%.*]] = getelementptr inbounds i32, ptr [[TMP10]], i64 4
; CHECK-NEXT: store <4 x i32> [[TMP9]], ptr [[TMP10]], align 4
-; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT: store <4 x i32> [[TMP14]], ptr [[TMP16]], align 4
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
; CHECK-NEXT: [[TMP11:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
; CHECK-NEXT: br i1 [[TMP11]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP27:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-narrowed.ll b/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-narrowed.ll
new file mode 100644
index 0000000000000..f26c7632c9f93
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-narrowed.ll
@@ -0,0 +1,50 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
+; Narrowing an interleave group materializes VFxUF before IC selection. The
+; redundancy analysis must fail closed rather than interpret the selected VF as
+; the narrowed plan's physical step.
+; RUN: opt -passes=loop-vectorize -mtriple=arm64-apple-macosx \
+; RUN: -force-vector-width=2 -force-target-max-vector-interleave=2 \
+; RUN: -force-target-num-vector-regs=1024 \
+; RUN: -force-target-instruction-cost=1 -small-loop-cost=0 \
+; RUN: -enable-interleave-cse -interleave-cse-min-pct=1 \
+; RUN: -S %s | FileCheck %s
+
+target triple = "arm64-apple-macosx"
+
+define void @narrowed(ptr noalias %a, i64 %value) {
+; CHECK-LABEL: define void @narrowed(
+; CHECK-SAME: ptr noalias [[A:%.*]], i64 [[VALUE:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: br label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <2 x i64> poison, i64 [[VALUE]], i64 0
+; CHECK-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <2 x i64> [[BROADCAST_SPLATINSERT]], <2 x i64> poison, <2 x i32> zeroinitializer
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds { i64, i64 }, ptr [[A]], i64 [[INDEX]], i32 0
+; CHECK-NEXT: store <2 x i64> [[BROADCAST_SPLAT]], ptr [[TMP0]], align 8
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 1
+; CHECK-NEXT: [[TMP1:%.*]] = icmp eq i64 [[INDEX_NEXT]], 100
+; CHECK-NEXT: br i1 [[TMP1]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: br label %[[EXIT:.*]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %p0 = getelementptr inbounds { i64, i64 }, ptr %a, i64 %iv, i32 0
+ %p1 = getelementptr inbounds { i64, i64 }, ptr %a, i64 %iv, i32 1
+ store i64 %value, ptr %p0, align 8
+ store i64 %value, ptr %p1, align 8
+ %iv.next = add nuw nsw i64 %iv, 1
+ %done = icmp eq i64 %iv.next, 100
+ br i1 %done, label %exit, label %loop
+
+exit:
+ ret void
+}
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-opportunities.ll b/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-opportunities.ll
index fe8e8c0b53880..cf1ccc4aaac73 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-opportunities.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-opportunities.ll
@@ -1,17 +1,20 @@
; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
-; Pre-commit test for counting cross-part load redundancy opportunities. The
-; two loops below differ only in how many distinct opportunities they contain;
-; both vectorize with an interleave count of 1 today.
+; A minimum saving of 6% is above what one redundancy opportunity in this loop
+; can deliver and below what two opportunities deliver, so it separates the two
+; functions below.
; RUN: opt -passes=loop-vectorize -force-vector-width=4 \
; RUN: -force-target-max-vector-interleave=2 -small-loop-cost=0 \
+; RUN: -force-target-instruction-cost=1 \
+; RUN: -enable-interleave-cse -interleave-cse-min-pct=6 \
; RUN: -S %s | FileCheck %s
target triple = "aarch64-unknown-linux-gnu"
; The first a[i+4] load forms a cross-part redundancy opportunity with a[i].
; The second a[i+4] load is a duplicate within the same logical part: it already
-; exists without interleaving and therefore is no additional cross-part
-; opportunity, which leaves a single opportunity in this loop.
+; exists without interleaving and therefore counts as no additional cross-part
+; opportunity. The single opportunity stays below the 6% threshold, so the
+; interleave count remains 1.
define void @duplicate_after_cross_part(ptr noalias %a, ptr noalias %c, i64 %n) {
; CHECK-LABEL: define void @duplicate_after_cross_part(
; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) {
@@ -85,34 +88,45 @@ exit:
}
; At VF=4, a[i+4] in part 0 is redundant with a[i] in part 1, and a[i+8] in
-; part 0 is independently redundant with a[i+4] in part 1, which gives two
-; distinct opportunities in this loop.
+; part 0 is independently redundant with a[i+4] in part 1, giving two distinct
+; opportunities. Their combined saving passes the 6% threshold and raises the
+; interleave count to 2.
define void @two_opportunities(ptr noalias %a, ptr noalias %c, i64 %n) {
; CHECK-LABEL: define void @two_opportunities(
; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) {
; CHECK-NEXT: [[ENTRY:.*]]:
; CHECK-NEXT: [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
-; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 4
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 8
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
; CHECK: [[VECTOR_PH]]:
-; CHECK-NEXT: [[TMP1:%.*]] = and i64 [[TMP0]], 3
+; CHECK-NEXT: [[TMP1:%.*]] = and i64 [[TMP0]], 7
; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
; CHECK-NEXT: [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT: [[TMP12:%.*]] = getelementptr inbounds i32, ptr [[TMP2]], i64 4
; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4
+; CHECK-NEXT: [[WIDE_LOAD4:%.*]] = load <4 x i32>, ptr [[TMP12]], align 4
; CHECK-NEXT: [[TMP3:%.*]] = add nuw nsw i64 [[INDEX]], 4
; CHECK-NEXT: [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP3]]
+; CHECK-NEXT: [[TMP14:%.*]] = getelementptr inbounds i32, ptr [[TMP4]], i64 4
; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP4]], align 4
+; CHECK-NEXT: [[WIDE_LOAD3:%.*]] = load <4 x i32>, ptr [[TMP14]], align 4
; CHECK-NEXT: [[TMP5:%.*]] = add nuw nsw i64 [[INDEX]], 8
; CHECK-NEXT: [[TMP6:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP5]]
+; CHECK-NEXT: [[TMP16:%.*]] = getelementptr inbounds i32, ptr [[TMP6]], i64 4
; CHECK-NEXT: [[WIDE_LOAD2:%.*]] = load <4 x i32>, ptr [[TMP6]], align 4
+; CHECK-NEXT: [[WIDE_LOAD5:%.*]] = load <4 x i32>, ptr [[TMP16]], align 4
; CHECK-NEXT: [[TMP7:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[WIDE_LOAD1]]
+; CHECK-NEXT: [[TMP11:%.*]] = add <4 x i32> [[WIDE_LOAD4]], [[WIDE_LOAD3]]
; CHECK-NEXT: [[TMP8:%.*]] = add <4 x i32> [[TMP7]], [[WIDE_LOAD2]]
+; CHECK-NEXT: [[TMP13:%.*]] = add <4 x i32> [[TMP11]], [[WIDE_LOAD5]]
; CHECK-NEXT: [[TMP9:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
+; CHECK-NEXT: [[TMP15:%.*]] = getelementptr inbounds i32, ptr [[TMP9]], i64 4
; CHECK-NEXT: store <4 x i32> [[TMP8]], ptr [[TMP9]], align 4
-; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT: store <4 x i32> [[TMP13]], ptr [[TMP15]], align 4
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
; CHECK-NEXT: [[TMP10:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
; CHECK-NEXT: br i1 [[TMP10]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-pipeline.ll b/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-pipeline.ll
new file mode 100644
index 0000000000000..a94d4685cc9ac
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-pipeline.ll
@@ -0,0 +1,49 @@
+; The redundancy analysis changes only the interleave count. This test
+; demonstrates that the standard O3 pipeline can realize the motivating
+; opportunity, without making downstream elimination part of the contract.
+;
+; RUN: opt -passes='default<O3>' -force-vector-width=4 \
+; RUN: -force-target-max-vector-interleave=2 -small-loop-cost=0 \
+; RUN: -enable-interleave-cse -interleave-cse-min-pct=1 \
+; RUN: -S %s | FileCheck %s --check-prefix=ENABLED
+; RUN: opt -passes='default<O3>' -force-vector-width=4 \
+; RUN: -force-target-max-vector-interleave=2 -small-loop-cost=0 \
+; RUN: -S %s | FileCheck %s --check-prefix=DISABLED
+
+target triple = "aarch64-unknown-linux-gnu"
+
+; With the analysis enabled, IC=2 exposes one redundant vector load to the O3
+; pipeline. The vector body processes eight source iterations with three loads.
+; With the analysis disabled, IC=1 processes four iterations with two loads.
+define void @positive(ptr noalias %a, ptr noalias %c, i64 %n) {
+; ENABLED-LABEL: define void @positive(
+; ENABLED: vector.body:
+; ENABLED-COUNT-3: load <4 x i32>
+; ENABLED-NOT: load <4 x i32>
+; ENABLED: add nuw i64 {{.*}}, 8
+;
+; DISABLED-LABEL: define void @positive(
+; DISABLED: vector.body:
+; DISABLED-COUNT-2: load <4 x i32>
+; DISABLED-NOT: load <4 x i32>
+; DISABLED: add nuw i64 {{.*}}, 4
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %a.iv = getelementptr inbounds i32, ptr %a, i64 %iv
+ %l1 = load i32, ptr %a.iv, align 4
+ %iv.plus.4 = add nuw nsw i64 %iv, 4
+ %a.iv.plus.4 = getelementptr inbounds i32, ptr %a, i64 %iv.plus.4
+ %l2 = load i32, ptr %a.iv.plus.4, align 4
+ %sum = add i32 %l1, %l2
+ %c.iv = getelementptr inbounds i32, ptr %c, i64 %iv
+ store i32 %sum, ptr %c.iv, align 4
+ %iv.next = add nuw nsw i64 %iv, 1
+ %cmp = icmp slt i64 %iv.next, %n
+ br i1 %cmp, label %loop, label %exit
+
+exit:
+ ret void
+}
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-scalable-reverse.ll b/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-scalable-reverse.ll
new file mode 100644
index 0000000000000..0bd8906b48adc
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-scalable-reverse.ll
@@ -0,0 +1,188 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
+; RUN: opt -passes=loop-vectorize -mtriple=aarch64-none-linux-gnu -mattr=+sve \
+; RUN: -force-vector-width="vscale x 2" \
+; RUN: -force-target-max-vector-interleave=2 -small-loop-cost=0 \
+; RUN: -enable-interleave-cse -interleave-cse-min-pct=1 \
+; RUN: -S %s | FileCheck %s
+
+target triple = "aarch64-unknown-linux-gnu"
+
+declare i64 @llvm.vscale.i64()
+
+; A constant source offset cannot equal the scalable reverse part offset for
+; every runtime vscale, so the interleave count stays 1.
+define void @reverse_constant_offset(ptr noalias %a, ptr noalias %c, i64 %n) {
+; CHECK-LABEL: define void @reverse_constant_offset(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) #[[ATTR1:[0-9]+]] {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: [[LAST:%.*]] = add i64 [[N]], -1
+; CHECK-NEXT: [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; CHECK-NEXT: [[TMP1:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-NEXT: [[TMP2:%.*]] = shl nuw i64 [[TMP1]], 1
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], [[TMP2]]
+; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: [[N_MOD_VF:%.*]] = urem i64 [[TMP0]], [[TMP2]]
+; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[TMP0]], [[N_MOD_VF]]
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP3:%.*]] = sub i64 [[LAST]], [[INDEX]]
+; CHECK-NEXT: [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP3]]
+; CHECK-NEXT: [[TMP5:%.*]] = sub nuw nsw i64 [[TMP2]], 1
+; CHECK-NEXT: [[TMP6:%.*]] = sub i64 0, [[TMP5]]
+; CHECK-NEXT: [[TMP7:%.*]] = getelementptr inbounds i32, ptr [[TMP4]], i64 [[TMP6]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <vscale x 2 x i32>, ptr [[TMP7]], align 4
+; CHECK-NEXT: [[TMP8:%.*]] = add i64 [[TMP3]], -2
+; CHECK-NEXT: [[TMP9:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP8]]
+; CHECK-NEXT: [[TMP10:%.*]] = getelementptr inbounds i32, ptr [[TMP9]], i64 [[TMP6]]
+; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <vscale x 2 x i32>, ptr [[TMP10]], align 4
+; CHECK-NEXT: [[TMP11:%.*]] = add <vscale x 2 x i32> [[WIDE_LOAD]], [[WIDE_LOAD1]]
+; CHECK-NEXT: [[REVERSE:%.*]] = call <vscale x 2 x i32> @llvm.vector.reverse.nxv2i32(<vscale x 2 x i32> [[TMP11]])
+; CHECK-NEXT: [[TMP12:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
+; CHECK-NEXT: store <vscale x 2 x i32> [[REVERSE]], ptr [[TMP12]], align 4
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP2]]
+; CHECK-NEXT: [[TMP13:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[TMP13]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[REVERSE_IV:%.*]] = sub i64 [[LAST]], [[IV]]
+; CHECK-NEXT: [[A_REVERSE:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[REVERSE_IV]]
+; CHECK-NEXT: [[L1:%.*]] = load i32, ptr [[A_REVERSE]], align 4
+; CHECK-NEXT: [[REVERSE_IV_MINUS_2:%.*]] = add i64 [[REVERSE_IV]], -2
+; CHECK-NEXT: [[A_REVERSE_MINUS_2:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[REVERSE_IV_MINUS_2]]
+; CHECK-NEXT: [[L2:%.*]] = load i32, ptr [[A_REVERSE_MINUS_2]], align 4
+; CHECK-NEXT: [[SUM:%.*]] = add i32 [[L1]], [[L2]]
+; CHECK-NEXT: [[C_IV:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[IV]]
+; CHECK-NEXT: store i32 [[SUM]], ptr [[C_IV]], align 4
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[CMP:%.*]] = icmp slt i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT: br i1 [[CMP]], label %[[LOOP]], label %[[EXIT]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ %last = add i64 %n, -1
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %reverse.iv = sub i64 %last, %iv
+ %a.reverse = getelementptr inbounds i32, ptr %a, i64 %reverse.iv
+ %l1 = load i32, ptr %a.reverse, align 4
+ %reverse.iv.minus.2 = add i64 %reverse.iv, -2
+ %a.reverse.minus.2 = getelementptr inbounds i32, ptr %a, i64 %reverse.iv.minus.2
+ %l2 = load i32, ptr %a.reverse.minus.2, align 4
+ %sum = add i32 %l1, %l2
+ %c.iv = getelementptr inbounds i32, ptr %c, i64 %iv
+ store i32 %sum, ptr %c.iv, align 4
+ %iv.next = add nuw nsw i64 %iv, 1
+ %cmp = icmp slt i64 %iv.next, %n
+ br i1 %cmp, label %loop, label %exit
+
+exit:
+ ret void
+}
+
+; The second source address is one scalable VF behind the first. Its part-0
+; reverse vector therefore equals the first load's part-1 reverse vector and
+; raises the interleave count to 2.
+define void @reverse_vscale_offset(ptr noalias %a, ptr noalias %c, i64 %n) {
+; CHECK-LABEL: define void @reverse_vscale_offset(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) #[[ATTR1]] {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: [[LAST:%.*]] = add i64 [[N]], -1
+; CHECK-NEXT: [[VSCALE:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-NEXT: [[VF:%.*]] = shl nuw nsw i64 [[VSCALE]], 1
+; CHECK-NEXT: [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; CHECK-NEXT: [[TMP1:%.*]] = shl nuw i64 [[VSCALE]], 2
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], [[TMP1]]
+; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: [[TMP2:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-NEXT: [[TMP3:%.*]] = shl nuw i64 [[TMP2]], 1
+; CHECK-NEXT: [[TMP4:%.*]] = shl nuw i64 [[TMP2]], 2
+; CHECK-NEXT: [[N_MOD_VF:%.*]] = urem i64 [[TMP0]], [[TMP4]]
+; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[TMP0]], [[N_MOD_VF]]
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP5:%.*]] = sub i64 [[LAST]], [[INDEX]]
+; CHECK-NEXT: [[TMP6:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP5]]
+; CHECK-NEXT: [[TMP7:%.*]] = sub nuw nsw i64 [[TMP3]], 1
+; CHECK-NEXT: [[TMP8:%.*]] = sub i64 0, [[TMP7]]
+; CHECK-NEXT: [[TMP9:%.*]] = getelementptr inbounds i32, ptr [[TMP6]], i64 [[TMP8]]
+; CHECK-NEXT: [[TMP10:%.*]] = sub i64 [[TMP8]], [[TMP3]]
+; CHECK-NEXT: [[TMP11:%.*]] = getelementptr inbounds i32, ptr [[TMP6]], i64 [[TMP10]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <vscale x 2 x i32>, ptr [[TMP9]], align 4
+; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <vscale x 2 x i32>, ptr [[TMP11]], align 4
+; CHECK-NEXT: [[TMP12:%.*]] = sub i64 [[TMP5]], [[VF]]
+; CHECK-NEXT: [[TMP13:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP12]]
+; CHECK-NEXT: [[TMP14:%.*]] = getelementptr inbounds i32, ptr [[TMP13]], i64 [[TMP8]]
+; CHECK-NEXT: [[TMP15:%.*]] = getelementptr inbounds i32, ptr [[TMP13]], i64 [[TMP10]]
+; CHECK-NEXT: [[WIDE_LOAD2:%.*]] = load <vscale x 2 x i32>, ptr [[TMP14]], align 4
+; CHECK-NEXT: [[WIDE_LOAD3:%.*]] = load <vscale x 2 x i32>, ptr [[TMP15]], align 4
+; CHECK-NEXT: [[TMP16:%.*]] = add <vscale x 2 x i32> [[WIDE_LOAD]], [[WIDE_LOAD2]]
+; CHECK-NEXT: [[TMP17:%.*]] = add <vscale x 2 x i32> [[WIDE_LOAD1]], [[WIDE_LOAD3]]
+; CHECK-NEXT: [[REVERSE:%.*]] = call <vscale x 2 x i32> @llvm.vector.reverse.nxv2i32(<vscale x 2 x i32> [[TMP16]])
+; CHECK-NEXT: [[REVERSE4:%.*]] = call <vscale x 2 x i32> @llvm.vector.reverse.nxv2i32(<vscale x 2 x i32> [[TMP17]])
+; CHECK-NEXT: [[TMP18:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
+; CHECK-NEXT: [[TMP19:%.*]] = getelementptr inbounds i32, ptr [[TMP18]], i64 [[TMP3]]
+; CHECK-NEXT: store <vscale x 2 x i32> [[REVERSE]], ptr [[TMP18]], align 4
+; CHECK-NEXT: store <vscale x 2 x i32> [[REVERSE4]], ptr [[TMP19]], align 4
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP4]]
+; CHECK-NEXT: [[TMP20:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[TMP20]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[REVERSE_IV:%.*]] = sub i64 [[LAST]], [[IV]]
+; CHECK-NEXT: [[A_REVERSE:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[REVERSE_IV]]
+; CHECK-NEXT: [[L1:%.*]] = load i32, ptr [[A_REVERSE]], align 4
+; CHECK-NEXT: [[REVERSE_IV_MINUS_VF:%.*]] = sub i64 [[REVERSE_IV]], [[VF]]
+; CHECK-NEXT: [[A_REVERSE_MINUS_VF:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[REVERSE_IV_MINUS_VF]]
+; CHECK-NEXT: [[L2:%.*]] = load i32, ptr [[A_REVERSE_MINUS_VF]], align 4
+; CHECK-NEXT: [[SUM:%.*]] = add i32 [[L1]], [[L2]]
+; CHECK-NEXT: [[C_IV:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[IV]]
+; CHECK-NEXT: store i32 [[SUM]], ptr [[C_IV]], align 4
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[CMP:%.*]] = icmp slt i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT: br i1 [[CMP]], label %[[LOOP]], label %[[EXIT]], !llvm.loop [[LOOP5:![0-9]+]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ %last = add i64 %n, -1
+ %vscale = call i64 @llvm.vscale.i64()
+ %vf = shl nuw nsw i64 %vscale, 1
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %reverse.iv = sub i64 %last, %iv
+ %a.reverse = getelementptr inbounds i32, ptr %a, i64 %reverse.iv
+ %l1 = load i32, ptr %a.reverse, align 4
+ %reverse.iv.minus.vf = sub i64 %reverse.iv, %vf
+ %a.reverse.minus.vf = getelementptr inbounds i32, ptr %a, i64 %reverse.iv.minus.vf
+ %l2 = load i32, ptr %a.reverse.minus.vf, align 4
+ %sum = add i32 %l1, %l2
+ %c.iv = getelementptr inbounds i32, ptr %c, i64 %iv
+ store i32 %sum, ptr %c.iv, align 4
+ %iv.next = add nuw nsw i64 %iv, 1
+ %cmp = icmp slt i64 %iv.next, %n
+ br i1 %cmp, label %loop, label %exit
+
+exit:
+ ret void
+}
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-scalable.ll b/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-scalable.ll
index 2af48b332d9b2..f566c9e757c78 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-scalable.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-scalable.ll
@@ -1,10 +1,10 @@
; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
-; Pre-commit test for cross-part load redundancy with scalable vectors, where
-; the distance between two logical parts is a runtime multiple of vscale. Both
-; loops below vectorize with an interleave count of 1 today.
+; The minimum saving is lowered to 1% so that the difference between the two
+; functions below comes from exact scalable address equality alone.
; RUN: opt -passes=loop-vectorize -mtriple=aarch64-none-linux-gnu -mattr=+sve \
; RUN: -force-vector-width="vscale x 2" \
; RUN: -force-target-max-vector-interleave=2 -small-loop-cost=0 \
+; RUN: -enable-interleave-cse -interleave-cse-min-pct=1 \
; RUN: -S %s | FileCheck %s
target triple = "aarch64-unknown-linux-gnu"
@@ -12,7 +12,8 @@ target triple = "aarch64-unknown-linux-gnu"
declare i64 @llvm.vscale.i64()
; A constant source offset cannot equal the scalable logical-part offset for
-; every runtime vscale, so the two addresses are not exactly equal.
+; every runtime vscale, so the two addresses are not exactly equal and the
+; interleave count stays 1.
define void @constant_offset(ptr noalias %a, ptr noalias %c, i64 %n) {
; CHECK-LABEL: define void @constant_offset(
; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) #[[ATTR1:[0-9]+]] {
@@ -83,7 +84,8 @@ exit:
}
; The second load starts exactly one scalable VF after the first, so its part-0
-; address equals the first load's part-1 address for every runtime vscale.
+; address equals the first load's part-1 address for every runtime vscale, which
+; raises the interleave count to 2.
define void @vscale_offset(ptr noalias %a, ptr noalias %c, i64 %n) {
; CHECK-LABEL: define void @vscale_offset(
; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) #[[ATTR1]] {
@@ -91,25 +93,34 @@ define void @vscale_offset(ptr noalias %a, ptr noalias %c, i64 %n) {
; CHECK-NEXT: [[VSCALE:%.*]] = call i64 @llvm.vscale.i64()
; CHECK-NEXT: [[PART_OFFSET:%.*]] = mul nuw i64 [[VSCALE]], 2
; CHECK-NEXT: [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
-; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], [[PART_OFFSET]]
+; CHECK-NEXT: [[TMP10:%.*]] = shl nuw i64 [[VSCALE]], 2
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], [[TMP10]]
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
; CHECK: [[VECTOR_PH]]:
; CHECK-NEXT: [[TMP1:%.*]] = call i64 @llvm.vscale.i64()
; CHECK-NEXT: [[TMP2:%.*]] = shl nuw i64 [[TMP1]], 1
-; CHECK-NEXT: [[N_MOD_VF:%.*]] = urem i64 [[TMP0]], [[TMP2]]
+; CHECK-NEXT: [[TMP12:%.*]] = shl nuw i64 [[TMP1]], 2
+; CHECK-NEXT: [[N_MOD_VF:%.*]] = urem i64 [[TMP0]], [[TMP12]]
; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[TMP0]], [[N_MOD_VF]]
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
; CHECK-NEXT: [[TMP3:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT: [[TMP14:%.*]] = getelementptr inbounds i32, ptr [[TMP3]], i64 [[TMP2]]
; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <vscale x 2 x i32>, ptr [[TMP3]], align 4
+; CHECK-NEXT: [[WIDE_LOAD2:%.*]] = load <vscale x 2 x i32>, ptr [[TMP14]], align 4
; CHECK-NEXT: [[TMP4:%.*]] = add i64 [[INDEX]], [[PART_OFFSET]]
; CHECK-NEXT: [[TMP5:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP4]]
+; CHECK-NEXT: [[TMP9:%.*]] = getelementptr inbounds i32, ptr [[TMP5]], i64 [[TMP2]]
; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <vscale x 2 x i32>, ptr [[TMP5]], align 4
+; CHECK-NEXT: [[WIDE_LOAD3:%.*]] = load <vscale x 2 x i32>, ptr [[TMP9]], align 4
; CHECK-NEXT: [[TMP6:%.*]] = add <vscale x 2 x i32> [[WIDE_LOAD]], [[WIDE_LOAD1]]
+; CHECK-NEXT: [[TMP11:%.*]] = add <vscale x 2 x i32> [[WIDE_LOAD2]], [[WIDE_LOAD3]]
; CHECK-NEXT: [[TMP7:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
+; CHECK-NEXT: [[TMP13:%.*]] = getelementptr inbounds i32, ptr [[TMP7]], i64 [[TMP2]]
; CHECK-NEXT: store <vscale x 2 x i32> [[TMP6]], ptr [[TMP7]], align 4
-; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP2]]
+; CHECK-NEXT: store <vscale x 2 x i32> [[TMP11]], ptr [[TMP13]], align 4
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP12]]
; CHECK-NEXT: [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
; CHECK-NEXT: br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-vplan-multi-block.ll b/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-vplan-multi-block.ll
new file mode 100644
index 0000000000000..10e1088e83013
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse-vplan-multi-block.ll
@@ -0,0 +1,122 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
+; A predicated store creates a multi-block VPlan region after the matching
+; loads. The analysis scans only a single VPBasicBlock, so it must fail closed
+; and leave the interleave count at 1.
+; RUN: opt -passes=loop-vectorize -force-vector-width=4 \
+; RUN: -force-target-max-vector-interleave=2 -small-loop-cost=0 \
+; RUN: -enable-interleave-cse -interleave-cse-min-pct=1 \
+; RUN: -S %s | FileCheck %s
+
+target triple = "aarch64-unknown-linux-gnu"
+
+define void @vplan_multi_block(ptr noalias %a, ptr noalias %c, i64 %n) {
+; CHECK-LABEL: define void @vplan_multi_block(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 4
+; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: [[TMP0:%.*]] = and i64 [[N]], 3
+; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[PRED_STORE_CONTINUE7:.*]] ]
+; CHECK-NEXT: [[TMP1:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP1]], align 4
+; CHECK-NEXT: [[TMP2:%.*]] = add nuw nsw i64 [[INDEX]], 4
+; CHECK-NEXT: [[TMP3:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP2]]
+; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP3]], align 4
+; CHECK-NEXT: [[TMP4:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[WIDE_LOAD1]]
+; CHECK-NEXT: [[TMP5:%.*]] = icmp eq <4 x i32> [[TMP4]], zeroinitializer
+; CHECK-NEXT: [[TMP6:%.*]] = extractelement <4 x i1> [[TMP5]], i64 0
+; CHECK-NEXT: br i1 [[TMP6]], label %[[PRED_STORE_IF:.*]], label %[[PRED_STORE_CONTINUE:.*]]
+; CHECK: [[PRED_STORE_IF]]:
+; CHECK-NEXT: [[TMP7:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
+; CHECK-NEXT: [[TMP8:%.*]] = extractelement <4 x i32> [[TMP4]], i64 0
+; CHECK-NEXT: store i32 [[TMP8]], ptr [[TMP7]], align 4
+; CHECK-NEXT: br label %[[PRED_STORE_CONTINUE]]
+; CHECK: [[PRED_STORE_CONTINUE]]:
+; CHECK-NEXT: [[TMP9:%.*]] = extractelement <4 x i1> [[TMP5]], i64 1
+; CHECK-NEXT: br i1 [[TMP9]], label %[[PRED_STORE_IF2:.*]], label %[[PRED_STORE_CONTINUE3:.*]]
+; CHECK: [[PRED_STORE_IF2]]:
+; CHECK-NEXT: [[TMP10:%.*]] = add i64 [[INDEX]], 1
+; CHECK-NEXT: [[TMP11:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[TMP10]]
+; CHECK-NEXT: [[TMP12:%.*]] = extractelement <4 x i32> [[TMP4]], i64 1
+; CHECK-NEXT: store i32 [[TMP12]], ptr [[TMP11]], align 4
+; CHECK-NEXT: br label %[[PRED_STORE_CONTINUE3]]
+; CHECK: [[PRED_STORE_CONTINUE3]]:
+; CHECK-NEXT: [[TMP13:%.*]] = extractelement <4 x i1> [[TMP5]], i64 2
+; CHECK-NEXT: br i1 [[TMP13]], label %[[PRED_STORE_IF4:.*]], label %[[PRED_STORE_CONTINUE5:.*]]
+; CHECK: [[PRED_STORE_IF4]]:
+; CHECK-NEXT: [[TMP14:%.*]] = add i64 [[INDEX]], 2
+; CHECK-NEXT: [[TMP15:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[TMP14]]
+; CHECK-NEXT: [[TMP16:%.*]] = extractelement <4 x i32> [[TMP4]], i64 2
+; CHECK-NEXT: store i32 [[TMP16]], ptr [[TMP15]], align 4
+; CHECK-NEXT: br label %[[PRED_STORE_CONTINUE5]]
+; CHECK: [[PRED_STORE_CONTINUE5]]:
+; CHECK-NEXT: [[TMP17:%.*]] = extractelement <4 x i1> [[TMP5]], i64 3
+; CHECK-NEXT: br i1 [[TMP17]], label %[[PRED_STORE_IF6:.*]], label %[[PRED_STORE_CONTINUE7]]
+; CHECK: [[PRED_STORE_IF6]]:
+; CHECK-NEXT: [[TMP18:%.*]] = add i64 [[INDEX]], 3
+; CHECK-NEXT: [[TMP19:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[TMP18]]
+; CHECK-NEXT: [[TMP20:%.*]] = extractelement <4 x i32> [[TMP4]], i64 3
+; CHECK-NEXT: store i32 [[TMP20]], ptr [[TMP19]], align 4
+; CHECK-NEXT: br label %[[PRED_STORE_CONTINUE7]]
+; CHECK: [[PRED_STORE_CONTINUE7]]:
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT: [[TMP21:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[TMP21]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LATCH:.*]] ]
+; CHECK-NEXT: [[A_IV:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT: [[L1:%.*]] = load i32, ptr [[A_IV]], align 4
+; CHECK-NEXT: [[IV_PLUS_4:%.*]] = add nuw nsw i64 [[IV]], 4
+; CHECK-NEXT: [[A_IV_PLUS_4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV_PLUS_4]]
+; CHECK-NEXT: [[L2:%.*]] = load i32, ptr [[A_IV_PLUS_4]], align 4
+; CHECK-NEXT: [[SUM:%.*]] = add i32 [[L1]], [[L2]]
+; CHECK-NEXT: [[PRED:%.*]] = icmp eq i32 [[SUM]], 0
+; CHECK-NEXT: br i1 [[PRED]], label %[[STORE:.*]], label %[[LATCH]]
+; CHECK: [[STORE]]:
+; CHECK-NEXT: [[C_IV:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[IV]]
+; CHECK-NEXT: store i32 [[SUM]], ptr [[C_IV]], align 4
+; CHECK-NEXT: br label %[[LATCH]]
+; CHECK: [[LATCH]]:
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[DONE:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT: br i1 [[DONE]], label %[[EXIT]], label %[[LOOP]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %latch ]
+ %a.iv = getelementptr inbounds i32, ptr %a, i64 %iv
+ %l1 = load i32, ptr %a.iv, align 4
+ %iv.plus.4 = add nuw nsw i64 %iv, 4
+ %a.iv.plus.4 = getelementptr inbounds i32, ptr %a, i64 %iv.plus.4
+ %l2 = load i32, ptr %a.iv.plus.4, align 4
+ %sum = add i32 %l1, %l2
+ %pred = icmp eq i32 %sum, 0
+ br i1 %pred, label %store, label %latch
+
+store:
+ %c.iv = getelementptr inbounds i32, ptr %c, i64 %iv
+ store i32 %sum, ptr %c.iv, align 4
+ br label %latch
+
+latch:
+ %iv.next = add nuw nsw i64 %iv, 1
+ %done = icmp eq i64 %iv.next, %n
+ br i1 %done, label %exit, label %loop
+
+exit:
+ ret void
+}
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse.ll b/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse.ll
index f3b3cafd67b37..1775e9ea0cf73 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/cross-part-load-cse.ll
@@ -1,63 +1,338 @@
; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
-; Pre-commit test for cross-part load redundancy. A widened load in one
-; logical part of an interleaved vector loop can cover exactly the elements
-; that another widened load covers in the next part. The interleave-count
-; heuristics do not model that redundancy yet, so this loop vectorizes with an
-; interleave count of 1.
+; The analysis only predicts a downstream saving in order to guide interleave
+; count selection. It never modifies the VPlan, so all widened loads remain.
+;
+; Vector width chosen by the cost model, cross-part analysis enabled:
+; RUN: opt -passes=loop-vectorize -force-target-max-vector-interleave=2 \
+; RUN: -small-loop-cost=0 -enable-interleave-cse \
+; RUN: -interleave-cse-min-pct=1 \
+; RUN: -S %s | FileCheck %s --check-prefix=PRODUCTION
+;
+; Forced VF=4 with the default minimum saving percentage:
+; RUN: opt -passes=loop-vectorize -force-vector-width=4 \
+; RUN: -force-target-max-vector-interleave=2 -small-loop-cost=0 \
+; RUN: -enable-interleave-cse \
+; RUN: -S %s | FileCheck %s --check-prefix=DEFAULT-PCT
+;
+; A minimum saving of 100% cannot be reached, so the interleave count stays 1:
+; RUN: opt -passes=loop-vectorize -force-vector-width=4 \
+; RUN: -force-target-max-vector-interleave=2 -small-loop-cost=0 \
+; RUN: -enable-interleave-cse -interleave-cse-min-pct=100 \
+; RUN: -S %s | FileCheck %s --check-prefix=HIGH-PCT
+;
+; With the feature disabled the ordinary heuristics keep the interleave count:
+; RUN: opt -passes=loop-vectorize -force-vector-width=4 \
+; RUN: -force-target-max-vector-interleave=2 -small-loop-cost=0 \
+; RUN: -S %s | FileCheck %s --check-prefix=FEATURE-OFF
+;
+; An explicit false value must also dominate the saving threshold:
; RUN: opt -passes=loop-vectorize -force-vector-width=4 \
; RUN: -force-target-max-vector-interleave=2 -small-loop-cost=0 \
-; RUN: -S %s | FileCheck %s
+; RUN: -enable-interleave-cse=false -interleave-cse-min-pct=0 \
+; RUN: -S %s | FileCheck %s --check-prefix=FEATURE-OFF
+;
+; An explicit user interleave count must not be overridden:
+; RUN: opt -passes=loop-vectorize -force-vector-width=4 \
+; RUN: -force-target-max-vector-interleave=2 -small-loop-cost=0 \
+; RUN: -enable-interleave-cse -force-vector-interleave=1 \
+; RUN: -S %s | FileCheck %s --check-prefix=USER-IC
+;
+; A target maximum interleave count of 1 must not be raised:
+; RUN: opt -passes=loop-vectorize -force-vector-width=4 \
+; RUN: -force-target-max-vector-interleave=1 -small-loop-cost=0 \
+; RUN: -enable-interleave-cse -interleave-cse-min-pct=1 \
+; RUN: -S %s | FileCheck %s --check-prefix=MAX-IC
target triple = "aarch64-unknown-linux-gnu"
; The motivating shape for cross-part load redundancy: with VF=4, the widened
-; load of a[i+4] in logical part 0 would cover exactly the same elements as the
-; widened load of a[i] in logical part 1.
+; load of a[i+4] in logical part 0 covers exactly the same elements as the
+; widened load of a[i] in logical part 1. The predicted saving raises the
+; interleave count to 2, while all four widened loads remain.
define void @positive(ptr noalias %a, ptr noalias %c, i64 %n) {
-; CHECK-LABEL: define void @positive(
-; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) {
-; CHECK-NEXT: [[ENTRY:.*]]:
-; CHECK-NEXT: [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
-; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 4
-; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
-; CHECK: [[VECTOR_PH]]:
-; CHECK-NEXT: [[TMP1:%.*]] = and i64 [[TMP0]], 3
-; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
-; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
-; CHECK: [[VECTOR_BODY]]:
-; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT: [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
-; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4
-; CHECK-NEXT: [[TMP3:%.*]] = add nuw nsw i64 [[INDEX]], 4
-; CHECK-NEXT: [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP3]]
-; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP4]], align 4
-; CHECK-NEXT: [[TMP5:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[WIDE_LOAD1]]
-; CHECK-NEXT: [[TMP6:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
-; CHECK-NEXT: store <4 x i32> [[TMP5]], ptr [[TMP6]], align 4
-; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
-; CHECK-NEXT: [[TMP7:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-NEXT: br i1 [[TMP7]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
-; CHECK: [[MIDDLE_BLOCK]]:
-; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
-; CHECK-NEXT: br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
-; CHECK: [[SCALAR_PH]]:
-; CHECK-NEXT: [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
-; CHECK-NEXT: br label %[[LOOP:.*]]
-; CHECK: [[LOOP]]:
-; CHECK-NEXT: [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
-; CHECK-NEXT: [[A_IV:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
-; CHECK-NEXT: [[L1:%.*]] = load i32, ptr [[A_IV]], align 4
-; CHECK-NEXT: [[IV_PLUS_4:%.*]] = add nuw nsw i64 [[IV]], 4
-; CHECK-NEXT: [[A_IV_PLUS_4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV_PLUS_4]]
-; CHECK-NEXT: [[L2:%.*]] = load i32, ptr [[A_IV_PLUS_4]], align 4
-; CHECK-NEXT: [[SUM:%.*]] = add i32 [[L1]], [[L2]]
-; CHECK-NEXT: [[C_IV:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[IV]]
-; CHECK-NEXT: store i32 [[SUM]], ptr [[C_IV]], align 4
-; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
-; CHECK-NEXT: [[CMP:%.*]] = icmp slt i64 [[IV_NEXT]], [[N]]
-; CHECK-NEXT: br i1 [[CMP]], label %[[LOOP]], label %[[EXIT]], !llvm.loop [[LOOP3:![0-9]+]]
-; CHECK: [[EXIT]]:
-; CHECK-NEXT: ret void
+; PRODUCTION-LABEL: define void @positive(
+; PRODUCTION-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) {
+; PRODUCTION-NEXT: [[ENTRY:.*]]:
+; PRODUCTION-NEXT: [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; PRODUCTION-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 8
+; PRODUCTION-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; PRODUCTION: [[VECTOR_PH]]:
+; PRODUCTION-NEXT: [[TMP1:%.*]] = and i64 [[TMP0]], 7
+; PRODUCTION-NEXT: [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
+; PRODUCTION-NEXT: br label %[[VECTOR_BODY:.*]]
+; PRODUCTION: [[VECTOR_BODY]]:
+; PRODUCTION-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; PRODUCTION-NEXT: [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; PRODUCTION-NEXT: [[TMP3:%.*]] = getelementptr inbounds i32, ptr [[TMP2]], i64 4
+; PRODUCTION-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4
+; PRODUCTION-NEXT: [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP3]], align 4
+; PRODUCTION-NEXT: [[TMP4:%.*]] = add nuw nsw i64 [[INDEX]], 4
+; PRODUCTION-NEXT: [[TMP5:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP4]]
+; PRODUCTION-NEXT: [[TMP6:%.*]] = getelementptr inbounds i32, ptr [[TMP5]], i64 4
+; PRODUCTION-NEXT: [[WIDE_LOAD2:%.*]] = load <4 x i32>, ptr [[TMP5]], align 4
+; PRODUCTION-NEXT: [[WIDE_LOAD3:%.*]] = load <4 x i32>, ptr [[TMP6]], align 4
+; PRODUCTION-NEXT: [[TMP7:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[WIDE_LOAD2]]
+; PRODUCTION-NEXT: [[TMP8:%.*]] = add <4 x i32> [[WIDE_LOAD1]], [[WIDE_LOAD3]]
+; PRODUCTION-NEXT: [[TMP9:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
+; PRODUCTION-NEXT: [[TMP10:%.*]] = getelementptr inbounds i32, ptr [[TMP9]], i64 4
+; PRODUCTION-NEXT: store <4 x i32> [[TMP7]], ptr [[TMP9]], align 4
+; PRODUCTION-NEXT: store <4 x i32> [[TMP8]], ptr [[TMP10]], align 4
+; PRODUCTION-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
+; PRODUCTION-NEXT: [[TMP11:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; PRODUCTION-NEXT: br i1 [[TMP11]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; PRODUCTION: [[MIDDLE_BLOCK]]:
+; PRODUCTION-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
+; PRODUCTION-NEXT: br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; PRODUCTION: [[SCALAR_PH]]:
+; PRODUCTION-NEXT: [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; PRODUCTION-NEXT: br label %[[LOOP:.*]]
+; PRODUCTION: [[LOOP]]:
+; PRODUCTION-NEXT: [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; PRODUCTION-NEXT: [[A_IV:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
+; PRODUCTION-NEXT: [[L1:%.*]] = load i32, ptr [[A_IV]], align 4
+; PRODUCTION-NEXT: [[IV_PLUS_4:%.*]] = add nuw nsw i64 [[IV]], 4
+; PRODUCTION-NEXT: [[A_IV_PLUS_4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV_PLUS_4]]
+; PRODUCTION-NEXT: [[L2:%.*]] = load i32, ptr [[A_IV_PLUS_4]], align 4
+; PRODUCTION-NEXT: [[SUM:%.*]] = add i32 [[L1]], [[L2]]
+; PRODUCTION-NEXT: [[C_IV:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[IV]]
+; PRODUCTION-NEXT: store i32 [[SUM]], ptr [[C_IV]], align 4
+; PRODUCTION-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; PRODUCTION-NEXT: [[CMP:%.*]] = icmp slt i64 [[IV_NEXT]], [[N]]
+; PRODUCTION-NEXT: br i1 [[CMP]], label %[[LOOP]], label %[[EXIT]], !llvm.loop [[LOOP3:![0-9]+]]
+; PRODUCTION: [[EXIT]]:
+; PRODUCTION-NEXT: ret void
+;
+; DEFAULT-PCT-LABEL: define void @positive(
+; DEFAULT-PCT-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) {
+; DEFAULT-PCT-NEXT: [[ENTRY:.*]]:
+; DEFAULT-PCT-NEXT: [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; DEFAULT-PCT-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 8
+; DEFAULT-PCT-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; DEFAULT-PCT: [[VECTOR_PH]]:
+; DEFAULT-PCT-NEXT: [[TMP1:%.*]] = and i64 [[TMP0]], 7
+; DEFAULT-PCT-NEXT: [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
+; DEFAULT-PCT-NEXT: br label %[[VECTOR_BODY:.*]]
+; DEFAULT-PCT: [[VECTOR_BODY]]:
+; DEFAULT-PCT-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; DEFAULT-PCT-NEXT: [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; DEFAULT-PCT-NEXT: [[TMP3:%.*]] = getelementptr inbounds i32, ptr [[TMP2]], i64 4
+; DEFAULT-PCT-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4
+; DEFAULT-PCT-NEXT: [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP3]], align 4
+; DEFAULT-PCT-NEXT: [[TMP4:%.*]] = add nuw nsw i64 [[INDEX]], 4
+; DEFAULT-PCT-NEXT: [[TMP5:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP4]]
+; DEFAULT-PCT-NEXT: [[TMP6:%.*]] = getelementptr inbounds i32, ptr [[TMP5]], i64 4
+; DEFAULT-PCT-NEXT: [[WIDE_LOAD2:%.*]] = load <4 x i32>, ptr [[TMP5]], align 4
+; DEFAULT-PCT-NEXT: [[WIDE_LOAD3:%.*]] = load <4 x i32>, ptr [[TMP6]], align 4
+; DEFAULT-PCT-NEXT: [[TMP7:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[WIDE_LOAD2]]
+; DEFAULT-PCT-NEXT: [[TMP8:%.*]] = add <4 x i32> [[WIDE_LOAD1]], [[WIDE_LOAD3]]
+; DEFAULT-PCT-NEXT: [[TMP9:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
+; DEFAULT-PCT-NEXT: [[TMP10:%.*]] = getelementptr inbounds i32, ptr [[TMP9]], i64 4
+; DEFAULT-PCT-NEXT: store <4 x i32> [[TMP7]], ptr [[TMP9]], align 4
+; DEFAULT-PCT-NEXT: store <4 x i32> [[TMP8]], ptr [[TMP10]], align 4
+; DEFAULT-PCT-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
+; DEFAULT-PCT-NEXT: [[TMP11:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; DEFAULT-PCT-NEXT: br i1 [[TMP11]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; DEFAULT-PCT: [[MIDDLE_BLOCK]]:
+; DEFAULT-PCT-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
+; DEFAULT-PCT-NEXT: br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; DEFAULT-PCT: [[SCALAR_PH]]:
+; DEFAULT-PCT-NEXT: [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; DEFAULT-PCT-NEXT: br label %[[LOOP:.*]]
+; DEFAULT-PCT: [[LOOP]]:
+; DEFAULT-PCT-NEXT: [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; DEFAULT-PCT-NEXT: [[A_IV:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
+; DEFAULT-PCT-NEXT: [[L1:%.*]] = load i32, ptr [[A_IV]], align 4
+; DEFAULT-PCT-NEXT: [[IV_PLUS_4:%.*]] = add nuw nsw i64 [[IV]], 4
+; DEFAULT-PCT-NEXT: [[A_IV_PLUS_4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV_PLUS_4]]
+; DEFAULT-PCT-NEXT: [[L2:%.*]] = load i32, ptr [[A_IV_PLUS_4]], align 4
+; DEFAULT-PCT-NEXT: [[SUM:%.*]] = add i32 [[L1]], [[L2]]
+; DEFAULT-PCT-NEXT: [[C_IV:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[IV]]
+; DEFAULT-PCT-NEXT: store i32 [[SUM]], ptr [[C_IV]], align 4
+; DEFAULT-PCT-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; DEFAULT-PCT-NEXT: [[CMP:%.*]] = icmp slt i64 [[IV_NEXT]], [[N]]
+; DEFAULT-PCT-NEXT: br i1 [[CMP]], label %[[LOOP]], label %[[EXIT]], !llvm.loop [[LOOP3:![0-9]+]]
+; DEFAULT-PCT: [[EXIT]]:
+; DEFAULT-PCT-NEXT: ret void
+;
+; HIGH-PCT-LABEL: define void @positive(
+; HIGH-PCT-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) {
+; HIGH-PCT-NEXT: [[ENTRY:.*]]:
+; HIGH-PCT-NEXT: [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; HIGH-PCT-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 4
+; HIGH-PCT-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; HIGH-PCT: [[VECTOR_PH]]:
+; HIGH-PCT-NEXT: [[TMP1:%.*]] = and i64 [[TMP0]], 3
+; HIGH-PCT-NEXT: [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
+; HIGH-PCT-NEXT: br label %[[VECTOR_BODY:.*]]
+; HIGH-PCT: [[VECTOR_BODY]]:
+; HIGH-PCT-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; HIGH-PCT-NEXT: [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; HIGH-PCT-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4
+; HIGH-PCT-NEXT: [[TMP3:%.*]] = add nuw nsw i64 [[INDEX]], 4
+; HIGH-PCT-NEXT: [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP3]]
+; HIGH-PCT-NEXT: [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP4]], align 4
+; HIGH-PCT-NEXT: [[TMP5:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[WIDE_LOAD1]]
+; HIGH-PCT-NEXT: [[TMP6:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
+; HIGH-PCT-NEXT: store <4 x i32> [[TMP5]], ptr [[TMP6]], align 4
+; HIGH-PCT-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; HIGH-PCT-NEXT: [[TMP7:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; HIGH-PCT-NEXT: br i1 [[TMP7]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; HIGH-PCT: [[MIDDLE_BLOCK]]:
+; HIGH-PCT-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
+; HIGH-PCT-NEXT: br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; HIGH-PCT: [[SCALAR_PH]]:
+; HIGH-PCT-NEXT: [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; HIGH-PCT-NEXT: br label %[[LOOP:.*]]
+; HIGH-PCT: [[LOOP]]:
+; HIGH-PCT-NEXT: [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; HIGH-PCT-NEXT: [[A_IV:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
+; HIGH-PCT-NEXT: [[L1:%.*]] = load i32, ptr [[A_IV]], align 4
+; HIGH-PCT-NEXT: [[IV_PLUS_4:%.*]] = add nuw nsw i64 [[IV]], 4
+; HIGH-PCT-NEXT: [[A_IV_PLUS_4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV_PLUS_4]]
+; HIGH-PCT-NEXT: [[L2:%.*]] = load i32, ptr [[A_IV_PLUS_4]], align 4
+; HIGH-PCT-NEXT: [[SUM:%.*]] = add i32 [[L1]], [[L2]]
+; HIGH-PCT-NEXT: [[C_IV:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[IV]]
+; HIGH-PCT-NEXT: store i32 [[SUM]], ptr [[C_IV]], align 4
+; HIGH-PCT-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; HIGH-PCT-NEXT: [[CMP:%.*]] = icmp slt i64 [[IV_NEXT]], [[N]]
+; HIGH-PCT-NEXT: br i1 [[CMP]], label %[[LOOP]], label %[[EXIT]], !llvm.loop [[LOOP3:![0-9]+]]
+; HIGH-PCT: [[EXIT]]:
+; HIGH-PCT-NEXT: ret void
+;
+; FEATURE-OFF-LABEL: define void @positive(
+; FEATURE-OFF-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) {
+; FEATURE-OFF-NEXT: [[ENTRY:.*]]:
+; FEATURE-OFF-NEXT: [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; FEATURE-OFF-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 4
+; FEATURE-OFF-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; FEATURE-OFF: [[VECTOR_PH]]:
+; FEATURE-OFF-NEXT: [[TMP1:%.*]] = and i64 [[TMP0]], 3
+; FEATURE-OFF-NEXT: [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
+; FEATURE-OFF-NEXT: br label %[[VECTOR_BODY:.*]]
+; FEATURE-OFF: [[VECTOR_BODY]]:
+; FEATURE-OFF-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; FEATURE-OFF-NEXT: [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; FEATURE-OFF-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4
+; FEATURE-OFF-NEXT: [[TMP3:%.*]] = add nuw nsw i64 [[INDEX]], 4
+; FEATURE-OFF-NEXT: [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP3]]
+; FEATURE-OFF-NEXT: [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP4]], align 4
+; FEATURE-OFF-NEXT: [[TMP5:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[WIDE_LOAD1]]
+; FEATURE-OFF-NEXT: [[TMP6:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
+; FEATURE-OFF-NEXT: store <4 x i32> [[TMP5]], ptr [[TMP6]], align 4
+; FEATURE-OFF-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; FEATURE-OFF-NEXT: [[TMP7:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; FEATURE-OFF-NEXT: br i1 [[TMP7]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; FEATURE-OFF: [[MIDDLE_BLOCK]]:
+; FEATURE-OFF-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
+; FEATURE-OFF-NEXT: br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; FEATURE-OFF: [[SCALAR_PH]]:
+; FEATURE-OFF-NEXT: [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; FEATURE-OFF-NEXT: br label %[[LOOP:.*]]
+; FEATURE-OFF: [[LOOP]]:
+; FEATURE-OFF-NEXT: [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; FEATURE-OFF-NEXT: [[A_IV:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
+; FEATURE-OFF-NEXT: [[L1:%.*]] = load i32, ptr [[A_IV]], align 4
+; FEATURE-OFF-NEXT: [[IV_PLUS_4:%.*]] = add nuw nsw i64 [[IV]], 4
+; FEATURE-OFF-NEXT: [[A_IV_PLUS_4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV_PLUS_4]]
+; FEATURE-OFF-NEXT: [[L2:%.*]] = load i32, ptr [[A_IV_PLUS_4]], align 4
+; FEATURE-OFF-NEXT: [[SUM:%.*]] = add i32 [[L1]], [[L2]]
+; FEATURE-OFF-NEXT: [[C_IV:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[IV]]
+; FEATURE-OFF-NEXT: store i32 [[SUM]], ptr [[C_IV]], align 4
+; FEATURE-OFF-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; FEATURE-OFF-NEXT: [[CMP:%.*]] = icmp slt i64 [[IV_NEXT]], [[N]]
+; FEATURE-OFF-NEXT: br i1 [[CMP]], label %[[LOOP]], label %[[EXIT]], !llvm.loop [[LOOP3:![0-9]+]]
+; FEATURE-OFF: [[EXIT]]:
+; FEATURE-OFF-NEXT: ret void
+;
+; USER-IC-LABEL: define void @positive(
+; USER-IC-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) {
+; USER-IC-NEXT: [[ENTRY:.*]]:
+; USER-IC-NEXT: [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; USER-IC-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 4
+; USER-IC-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; USER-IC: [[VECTOR_PH]]:
+; USER-IC-NEXT: [[TMP1:%.*]] = and i64 [[TMP0]], 3
+; USER-IC-NEXT: [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
+; USER-IC-NEXT: br label %[[VECTOR_BODY:.*]]
+; USER-IC: [[VECTOR_BODY]]:
+; USER-IC-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; USER-IC-NEXT: [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; USER-IC-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4
+; USER-IC-NEXT: [[TMP3:%.*]] = add nuw nsw i64 [[INDEX]], 4
+; USER-IC-NEXT: [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP3]]
+; USER-IC-NEXT: [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP4]], align 4
+; USER-IC-NEXT: [[TMP5:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[WIDE_LOAD1]]
+; USER-IC-NEXT: [[TMP6:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
+; USER-IC-NEXT: store <4 x i32> [[TMP5]], ptr [[TMP6]], align 4
+; USER-IC-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; USER-IC-NEXT: [[TMP7:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; USER-IC-NEXT: br i1 [[TMP7]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; USER-IC: [[MIDDLE_BLOCK]]:
+; USER-IC-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
+; USER-IC-NEXT: br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; USER-IC: [[SCALAR_PH]]:
+; USER-IC-NEXT: [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; USER-IC-NEXT: br label %[[LOOP:.*]]
+; USER-IC: [[LOOP]]:
+; USER-IC-NEXT: [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; USER-IC-NEXT: [[A_IV:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
+; USER-IC-NEXT: [[L1:%.*]] = load i32, ptr [[A_IV]], align 4
+; USER-IC-NEXT: [[IV_PLUS_4:%.*]] = add nuw nsw i64 [[IV]], 4
+; USER-IC-NEXT: [[A_IV_PLUS_4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV_PLUS_4]]
+; USER-IC-NEXT: [[L2:%.*]] = load i32, ptr [[A_IV_PLUS_4]], align 4
+; USER-IC-NEXT: [[SUM:%.*]] = add i32 [[L1]], [[L2]]
+; USER-IC-NEXT: [[C_IV:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[IV]]
+; USER-IC-NEXT: store i32 [[SUM]], ptr [[C_IV]], align 4
+; USER-IC-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; USER-IC-NEXT: [[CMP:%.*]] = icmp slt i64 [[IV_NEXT]], [[N]]
+; USER-IC-NEXT: br i1 [[CMP]], label %[[LOOP]], label %[[EXIT]], !llvm.loop [[LOOP3:![0-9]+]]
+; USER-IC: [[EXIT]]:
+; USER-IC-NEXT: ret void
+;
+; MAX-IC-LABEL: define void @positive(
+; MAX-IC-SAME: ptr noalias [[A:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) {
+; MAX-IC-NEXT: [[ENTRY:.*]]:
+; MAX-IC-NEXT: [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; MAX-IC-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 4
+; MAX-IC-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; MAX-IC: [[VECTOR_PH]]:
+; MAX-IC-NEXT: [[TMP1:%.*]] = and i64 [[TMP0]], 3
+; MAX-IC-NEXT: [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
+; MAX-IC-NEXT: br label %[[VECTOR_BODY:.*]]
+; MAX-IC: [[VECTOR_BODY]]:
+; MAX-IC-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; MAX-IC-NEXT: [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; MAX-IC-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4
+; MAX-IC-NEXT: [[TMP3:%.*]] = add nuw nsw i64 [[INDEX]], 4
+; MAX-IC-NEXT: [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP3]]
+; MAX-IC-NEXT: [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP4]], align 4
+; MAX-IC-NEXT: [[TMP5:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[WIDE_LOAD1]]
+; MAX-IC-NEXT: [[TMP6:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[INDEX]]
+; MAX-IC-NEXT: store <4 x i32> [[TMP5]], ptr [[TMP6]], align 4
+; MAX-IC-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; MAX-IC-NEXT: [[TMP7:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; MAX-IC-NEXT: br i1 [[TMP7]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; MAX-IC: [[MIDDLE_BLOCK]]:
+; MAX-IC-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
+; MAX-IC-NEXT: br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; MAX-IC: [[SCALAR_PH]]:
+; MAX-IC-NEXT: [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; MAX-IC-NEXT: br label %[[LOOP:.*]]
+; MAX-IC: [[LOOP]]:
+; MAX-IC-NEXT: [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; MAX-IC-NEXT: [[A_IV:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
+; MAX-IC-NEXT: [[L1:%.*]] = load i32, ptr [[A_IV]], align 4
+; MAX-IC-NEXT: [[IV_PLUS_4:%.*]] = add nuw nsw i64 [[IV]], 4
+; MAX-IC-NEXT: [[A_IV_PLUS_4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV_PLUS_4]]
+; MAX-IC-NEXT: [[L2:%.*]] = load i32, ptr [[A_IV_PLUS_4]], align 4
+; MAX-IC-NEXT: [[SUM:%.*]] = add i32 [[L1]], [[L2]]
+; MAX-IC-NEXT: [[C_IV:%.*]] = getelementptr inbounds i32, ptr [[C]], i64 [[IV]]
+; MAX-IC-NEXT: store i32 [[SUM]], ptr [[C_IV]], align 4
+; MAX-IC-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; MAX-IC-NEXT: [[CMP:%.*]] = icmp slt i64 [[IV_NEXT]], [[N]]
+; MAX-IC-NEXT: br i1 [[CMP]], label %[[LOOP]], label %[[EXIT]], !llvm.loop [[LOOP3:![0-9]+]]
+; MAX-IC: [[EXIT]]:
+; MAX-IC-NEXT: ret void
;
entry:
br label %loop
diff --git a/llvm/utils/gn/secondary/llvm/lib/Transforms/Vectorize/BUILD.gn b/llvm/utils/gn/secondary/llvm/lib/Transforms/Vectorize/BUILD.gn
index 0ffc24ec855f7..6ceb8ce67b207 100644
--- a/llvm/utils/gn/secondary/llvm/lib/Transforms/Vectorize/BUILD.gn
+++ b/llvm/utils/gn/secondary/llvm/lib/Transforms/Vectorize/BUILD.gn
@@ -42,6 +42,7 @@ static_library("Vectorize") {
"VPlan.cpp",
"VPlanAnalysis.cpp",
"VPlanConstruction.cpp",
+ "VPlanCrossPartCSE.cpp",
"VPlanDominatorTree.cpp",
"VPlanEVLTailFolding.cpp",
"VPlanLowering.cpp",
More information about the llvm-commits
mailing list