[llvm] [LV] Add cross-part load overlap heuristic for interleaving (PR #214500)

Sergey Shcherbinin via llvm-commits llvm-commits at lists.llvm.org
Mon Sep 21 06:52:26 PDT 2026


================
@@ -0,0 +1,479 @@
+; RUN: split-file %s %t
+;
+; The analysis only predicts downstream savings to guide IC selection. It does
+; not modify VPlan or eliminate any widened loads.
+;
+; RUN: opt -passes=loop-vectorize -force-vector-width=4 \
+; RUN:     -force-target-max-vector-interleave=2 -small-loop-cost=0 \
+; RUN:     -enable-interleave-cse -interleave-cse-min-ops=1 \
+; RUN:     -interleave-cse-min-pct=1 \
+; RUN:     -S %t/positive.ll | FileCheck %t/positive.ll --check-prefix=FORCED
+; RUN: opt -passes=loop-vectorize -force-target-max-vector-interleave=2 \
+; RUN:     -small-loop-cost=0 -enable-interleave-cse \
+; RUN:     -interleave-cse-min-ops=1 -interleave-cse-min-pct=1 \
+; RUN:     -S %t/positive.ll \
+; RUN:     | FileCheck %t/positive.ll --check-prefix=PRODUCTION
+; RUN: opt -passes=loop-vectorize -force-vector-width=4 \
+; RUN:     -force-target-max-vector-interleave=2 -small-loop-cost=0 \
+; RUN:     -enable-interleave-cse -S %t/positive.ll \
+; RUN:     | FileCheck %t/positive.ll --check-prefix=THRESHOLD
+; RUN: opt -passes=loop-vectorize -force-vector-width=4 \
+; RUN:     -force-target-max-vector-interleave=2 -small-loop-cost=0 \
+; RUN:     -enable-interleave-cse -interleave-cse-min-ops=1 \
+; RUN:     -interleave-cse-min-pct=100 -S %t/positive.ll \
+; RUN:     | FileCheck %t/positive.ll --check-prefix=PERCENT
+; Force deterministic costs so the ordinary branch-cost heuristic recommends
+; IC=1, allowing this run to verify that cross-part analysis can raise it to 2.
+; RUN: opt -passes=loop-vectorize -force-vector-width=4 \
+; RUN:     -force-target-max-vector-interleave=2 \
+; RUN:     -force-target-instruction-cost=1 -small-loop-cost=12 \
+; RUN:     -enable-loadstore-runtime-interleave=false \
+; RUN:     -enable-interleave-cse -interleave-cse-min-ops=1 \
+; RUN:     -interleave-cse-min-pct=1 \
+; RUN:     -S %t/positive.ll | FileCheck %t/positive.ll --check-prefix=SMALL-IC
+; RUN: opt -passes=loop-vectorize -force-vector-width=4 \
+; RUN:     -force-target-max-vector-interleave=2 -small-loop-cost=0 \
+; RUN:     -S %t/positive.ll | FileCheck %t/positive.ll --check-prefix=DISABLED
+; RUN: opt -passes=loop-vectorize -force-vector-width=4 \
+; RUN:     -force-target-max-vector-interleave=2 -small-loop-cost=0 \
+; RUN:     -enable-interleave-cse -force-vector-interleave=1 \
+; RUN:     -S %t/positive.ll | FileCheck %t/positive.ll --check-prefix=USERIC
+; RUN: opt -passes=loop-vectorize -force-vector-width=4 \
+; RUN:     -force-target-max-vector-interleave=1 -small-loop-cost=0 \
+; RUN:     -enable-interleave-cse -interleave-cse-min-ops=1 \
+; RUN:     -interleave-cse-min-pct=1 \
+; RUN:     -S %t/positive.ll | FileCheck %t/positive.ll --check-prefix=MAXIC
+; RUN: opt -passes=loop-vectorize -mtriple=aarch64-none-linux-gnu -mattr=+sve \
+; RUN:     -force-vector-width="vscale x 2" \
+; RUN:     -force-target-max-vector-interleave=2 -small-loop-cost=0 \
+; RUN:     -enable-interleave-cse -interleave-cse-min-ops=1 \
+; RUN:     -interleave-cse-min-pct=1 -S %t/positive.ll \
+; RUN:     | FileCheck %t/positive.ll --check-prefix=SCALABLE
+;
+; RUN: opt -passes=loop-vectorize -force-vector-width=4 \
+; RUN:     -force-target-max-vector-interleave=2 -small-loop-cost=0 \
+; RUN:     -enable-interleave-cse -interleave-cse-min-ops=2 \
+; RUN:     -S %t/duplicate.ll \
+; RUN:     | FileCheck %t/duplicate.ll
+; RUN: opt -passes=loop-vectorize -force-vector-width=4 \
+; RUN:     -force-target-max-vector-interleave=2 -small-loop-cost=0 \
+; RUN:     -enable-interleave-cse -interleave-cse-min-ops=2 \
+; RUN:     -interleave-cse-min-pct=6 -S %t/two-opportunities.ll \
+; RUN:     | FileCheck %t/two-opportunities.ll
+; RUN: opt -passes=loop-vectorize -force-vector-width=4 \
+; RUN:     -force-target-max-vector-interleave=2 -small-loop-cost=0 \
+; RUN:     -enable-interleave-cse -interleave-cse-min-ops=1 \
+; RUN:     -interleave-cse-min-pct=1 -S %t/type-mismatch.ll \
+; RUN:     | FileCheck %t/type-mismatch.ll
+; RUN: opt -passes=loop-vectorize -force-vector-width=4 \
+; RUN:     -force-target-max-vector-interleave=2 -small-loop-cost=0 \
+; RUN:     -enable-interleave-cse -interleave-cse-min-ops=1 \
+; RUN:     -interleave-cse-min-pct=1 -S %t/simple-negatives.ll \
+; RUN:     | FileCheck %t/simple-negatives.ll
+; RUN: opt -passes=loop-vectorize -force-vector-width=4 \
+; RUN:     -force-target-max-vector-interleave=2 -small-loop-cost=0 \
+; RUN:     -enable-interleave-cse -interleave-cse-min-ops=1 \
+; RUN:     -interleave-cse-min-pct=1 -S %t/provenance.ll \
+; RUN:     | FileCheck %t/provenance.ll
+; RUN: opt -passes=loop-vectorize -force-vector-width=4 \
+; RUN:     -force-target-max-vector-interleave=2 -small-loop-cost=0 \
+; RUN:     -enable-interleave-cse -interleave-cse-min-ops=1 \
+; RUN:     -interleave-cse-min-pct=1 -S %t/poison.ll \
+; RUN:     | FileCheck %t/poison.ll
+; RUN: opt -passes=loop-vectorize -force-vector-width=4 \
+; RUN:     -force-target-max-vector-interleave=2 -small-loop-cost=0 \
+; RUN:     -enable-interleave-cse -interleave-cse-min-ops=1 \
+; RUN:     -interleave-cse-min-pct=1 -S %t/multi-block.ll \
+; RUN:     | FileCheck %t/multi-block.ll
+; RUN: opt -passes=loop-vectorize -force-vector-width=4 \
+; RUN:     -force-target-max-vector-interleave=2 -small-loop-cost=0 \
+; RUN:     -enable-interleave-cse -interleave-cse-min-ops=1 \
+; RUN:     -interleave-cse-min-pct=1 -S %t/reverse.ll \
+; RUN:     | FileCheck %t/reverse.ll
+
+;--- positive.ll
+target triple = "aarch64-unknown-linux-gnu"
+
+; With VF=4, a[i+4] in part 0 has the same modeled vector address as a[i] in
+; part 1. The predicted downstream saving raises IC to 2, while all four
+; widened loads remain because the analysis does not realize the overlap.
+;
+; FORCED-LABEL: @positive(
+; FORCED:       vector.body:
+; FORCED-COUNT-4: load <4 x i32>
+; FORCED-NOT:   load <4 x i32>
+; FORCED:       add nuw i64 %index, 8
+;
+; PRODUCTION-LABEL: @positive(
+; PRODUCTION:       vector.body:
+; PRODUCTION-COUNT-4: load <4 x i32>
+; PRODUCTION-NOT:   load <4 x i32>
+; PRODUCTION:       add nuw i64 %index, 8
+;
+; THRESHOLD-LABEL: @positive(
+; THRESHOLD:       add nuw i64 %index, 4
+;
+; PERCENT-LABEL: @positive(
+; PERCENT:       add nuw i64 %index, 4
+;
+; SMALL-IC-LABEL: @positive(
+; SMALL-IC:       add nuw i64 %index, 8
+;
+; DISABLED-LABEL: @positive(
+; DISABLED:       add nuw i64 %index, 4
+;
+; USERIC-LABEL: @positive(
+; USERIC:       add nuw i64 %index, 4
+;
+; MAXIC-LABEL: @positive(
+; MAXIC:       add nuw i64 %index, 4
+;
+; SCALABLE-LABEL: @positive(
+; SCALABLE:       call i64 @llvm.vscale.i64()
+; SCALABLE:       vector.body:
+; SCALABLE-COUNT-2: load <vscale x 2 x i32>
+; SCALABLE-NOT:   load <vscale x 2 x i32>
+; SCALABLE:       %index.next = add nuw i64 %index, %
+define void @positive(ptr noalias %a, ptr noalias %c, i64 %n) {
----------------
SergeyShch01 wrote:

Done. I restructured the PR into two commits:

1. d08f5cf pre-commits the test inputs with the feature absent and full update_test_checks.py output, showing the baseline IC=1 behavior.
2. 8af706c adds the heuristic and regenerates the checks, making the IC=1 → IC=2 change explicit.

https://github.com/llvm/llvm-project/pull/214500


More information about the llvm-commits mailing list