[llvm] [SLP][NFC] Pre-commit test for AArch64 store-to-load forwarding bail-out (PR #218116)
Milin Bhade via llvm-commits
llvm-commits at lists.llvm.org
Sat Aug 22 04:43:21 PDT 2026
https://github.com/mbhade-amd updated https://github.com/llvm/llvm-project/pull/218116
>From b3c0123c6d310c1b4691fbcaded78afc22a3e9f3 Mon Sep 17 00:00:00 2001
From: mbhade <mbhade at amd.com>
Date: Sat, 22 Aug 2026 15:56:42 +0530
Subject: [PATCH] [SLP][NFC] Pre-commit test for AArch64 store-to-load
forwarding bail-out
Add an AArch64 SLP test that locks in the current (pre-feature) vectorization
of a widened backward load that aliases a widened store, plus a narrow-load
control case that never straddles two widened stores. The follow-up patch
adding the store-to-load forwarding cost-model bail-out will update these
checks, making its effect visible as a diff on a non-X86 target.
Co-authored-by: Cursor <cursoragent at cursor.com>
---
.../AArch64/store-load-forward-conflict.ll | 135 ++++++++++++++++++
1 file changed, 135 insertions(+)
create mode 100644 llvm/test/Transforms/SLPVectorizer/AArch64/store-load-forward-conflict.ll
diff --git a/llvm/test/Transforms/SLPVectorizer/AArch64/store-load-forward-conflict.ll b/llvm/test/Transforms/SLPVectorizer/AArch64/store-load-forward-conflict.ll
new file mode 100644
index 0000000000000..e81b2d941148f
--- /dev/null
+++ b/llvm/test/Transforms/SLPVectorizer/AArch64/store-load-forward-conflict.ll
@@ -0,0 +1,135 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt < %s -passes=slp-vectorizer -S -mtriple=aarch64-- -mcpu=neoverse-v2 | FileCheck %s
+
+; Pre-commit test for the SLP store-to-load forwarding (STLF) cost model on a
+; non-X86 target. This locks in the current (pre-feature) AArch64 behavior so
+; that the follow-up patch adding the STLF bail-out shows its effect as a diff.
+
+; Widened backward load at a misaligned distance: the four consecutive loads
+; A[i-5..i-2] widen to a 16-byte <4 x i32> load, 20 bytes behind the 16-byte
+; <4 x i32> store at A[i] (20 % 16 = 4 -> misaligned; the 16-byte load overruns
+; the 4 bytes left before the boundary and straddles two widened stores). Today
+; SLP widens this chain; the STLF patch will make it stay scalar.
+define void @stlf_conflict_widened_misaligned(ptr noalias %A, i64 %n) {
+; CHECK-LABEL: define void @stlf_conflict_widened_misaligned(
+; CHECK-SAME: ptr noalias [[A:%.*]], i64 [[N:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: br label %[[FOR_BODY:.*]]
+; CHECK: [[FOR_BODY]]:
+; CHECK-NEXT: [[I:%.*]] = phi i64 [ 8, %[[ENTRY]] ], [ [[I_NEXT:%.*]], %[[FOR_BODY]] ]
+; CHECK-NEXT: [[B0:%.*]] = add i64 [[I]], -5
+; CHECK-NEXT: [[P0:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[B0]]
+; CHECK-NEXT: [[GEP0:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[I]]
+; CHECK-NEXT: [[TMP0:%.*]] = load <4 x i32>, ptr [[P0]], align 4
+; CHECK-NEXT: [[TMP1:%.*]] = add nsw <4 x i32> [[TMP0]], <i32 1, i32 2, i32 3, i32 4>
+; CHECK-NEXT: store <4 x i32> [[TMP1]], ptr [[GEP0]], align 4
+; CHECK-NEXT: [[I_NEXT]] = add nuw nsw i64 [[I]], 4
+; CHECK-NEXT: [[CMP:%.*]] = icmp slt i64 [[I_NEXT]], [[N]]
+; CHECK-NEXT: br i1 [[CMP]], label %[[FOR_BODY]], label %[[FOR_END:.*]]
+; CHECK: [[FOR_END]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %for.body
+
+for.body:
+ %i = phi i64 [ 8, %entry ], [ %i.next, %for.body ]
+
+ %b0 = add i64 %i, -5
+ %b1 = add i64 %i, -4
+ %b2 = add i64 %i, -3
+ %b3 = add i64 %i, -2
+ %p0 = getelementptr inbounds i32, ptr %A, i64 %b0
+ %p1 = getelementptr inbounds i32, ptr %A, i64 %b1
+ %p2 = getelementptr inbounds i32, ptr %A, i64 %b2
+ %p3 = getelementptr inbounds i32, ptr %A, i64 %b3
+ %l0 = load i32, ptr %p0, align 4
+ %l1 = load i32, ptr %p1, align 4
+ %l2 = load i32, ptr %p2, align 4
+ %l3 = load i32, ptr %p3, align 4
+
+ %t1 = add nsw i32 %l0, 1
+ %t2 = add nsw i32 %l1, 2
+ %t3 = add nsw i32 %l2, 3
+ %t4 = add nsw i32 %l3, 4
+
+ %i1 = add nuw nsw i64 %i, 1
+ %i2 = add nuw nsw i64 %i, 2
+ %i3 = add nuw nsw i64 %i, 3
+ %gep0 = getelementptr inbounds i32, ptr %A, i64 %i
+ %gep1 = getelementptr inbounds i32, ptr %A, i64 %i1
+ %gep2 = getelementptr inbounds i32, ptr %A, i64 %i2
+ %gep3 = getelementptr inbounds i32, ptr %A, i64 %i3
+ store i32 %t1, ptr %gep0, align 4
+ store i32 %t2, ptr %gep1, align 4
+ store i32 %t3, ptr %gep2, align 4
+ store i32 %t4, ptr %gep3, align 4
+
+ %i.next = add nuw nsw i64 %i, 4
+ %cmp = icmp slt i64 %i.next, %n
+ br i1 %cmp, label %for.body, label %for.end
+
+for.end:
+ ret void
+}
+
+; Precision control: a single narrow scalar i32 load A[i-1], 4 bytes behind
+; (4 % 16 = 4 -> misaligned) but only 4 bytes wide, so it fits inside the bytes
+; left before the window boundary and never straddles two widened stores. No
+; forwarding conflict exists, so the chain widens both before and after the
+; STLF patch.
+define void @stlf_no_conflict_narrow_scalar_load(ptr noalias %A, i64 %n) {
+; CHECK-LABEL: define void @stlf_no_conflict_narrow_scalar_load(
+; CHECK-SAME: ptr noalias [[A:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: br label %[[FOR_BODY:.*]]
+; CHECK: [[FOR_BODY]]:
+; CHECK-NEXT: [[I:%.*]] = phi i64 [ 1, %[[ENTRY]] ], [ [[I_NEXT:%.*]], %[[FOR_BODY]] ]
+; CHECK-NEXT: [[BACK_IDX:%.*]] = sub i64 [[I]], 1
+; CHECK-NEXT: [[BACK_GEP:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[BACK_IDX]]
+; CHECK-NEXT: [[T:%.*]] = load i32, ptr [[BACK_GEP]], align 4
+; CHECK-NEXT: [[GEP0:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[I]]
+; CHECK-NEXT: [[TMP0:%.*]] = insertelement <4 x i32> poison, i32 [[T]], i64 0
+; CHECK-NEXT: [[TMP1:%.*]] = shufflevector <4 x i32> [[TMP0]], <4 x i32> poison, <4 x i32> zeroinitializer
+; CHECK-NEXT: [[TMP2:%.*]] = add nsw <4 x i32> [[TMP1]], <i32 1, i32 2, i32 3, i32 4>
+; CHECK-NEXT: store <4 x i32> [[TMP2]], ptr [[GEP0]], align 4
+; CHECK-NEXT: [[I_NEXT]] = add nuw nsw i64 [[I]], 4
+; CHECK-NEXT: [[CMP:%.*]] = icmp slt i64 [[I_NEXT]], [[N]]
+; CHECK-NEXT: br i1 [[CMP]], label %[[FOR_BODY]], label %[[FOR_END:.*]]
+; CHECK: [[FOR_END]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %for.body
+
+for.body:
+ %i = phi i64 [ 1, %entry ], [ %i.next, %for.body ]
+
+ %back.idx = sub i64 %i, 1
+ %back.gep = getelementptr inbounds i32, ptr %A, i64 %back.idx
+ %t = load i32, ptr %back.gep, align 4
+
+ %t1 = add nsw i32 %t, 1
+ %t2 = add nsw i32 %t, 2
+ %t3 = add nsw i32 %t, 3
+ %t4 = add nsw i32 %t, 4
+
+ %i1 = add nuw nsw i64 %i, 1
+ %i2 = add nuw nsw i64 %i, 2
+ %i3 = add nuw nsw i64 %i, 3
+ %gep0 = getelementptr inbounds i32, ptr %A, i64 %i
+ %gep1 = getelementptr inbounds i32, ptr %A, i64 %i1
+ %gep2 = getelementptr inbounds i32, ptr %A, i64 %i2
+ %gep3 = getelementptr inbounds i32, ptr %A, i64 %i3
+ store i32 %t1, ptr %gep0, align 4
+ store i32 %t2, ptr %gep1, align 4
+ store i32 %t3, ptr %gep2, align 4
+ store i32 %t4, ptr %gep3, align 4
+
+ %i.next = add nuw nsw i64 %i, 4
+ %cmp = icmp slt i64 %i.next, %n
+ br i1 %cmp, label %for.body, label %for.end
+
+for.end:
+ ret void
+}
More information about the llvm-commits
mailing list