[llvm] bf6129b - [SLP][NFC] Pre-commit test for AArch64 store-to-load forwarding bail-out (#218116)

via llvm-commits llvm-commits at lists.llvm.org
Mon Aug 31 23:52:24 PDT 2026


Author: Milin Bhade
Date: 2026-09-01T12:22:19+05:30
New Revision: bf6129be1de2f393e8ee031a1315c01626dc552e

URL: https://github.com/llvm/llvm-project/commit/bf6129be1de2f393e8ee031a1315c01626dc552e
DIFF: https://github.com/llvm/llvm-project/commit/bf6129be1de2f393e8ee031a1315c01626dc552e.diff

LOG: [SLP][NFC] Pre-commit test for AArch64 store-to-load forwarding bail-out (#218116)

Add an AArch64 SLP test that locks in the current (pre-feature)
vectorization of a widened backward load that aliases a widened store,
plus a narrow-load control case that never straddles two widened stores.
The follow-up patch adding the store-to-load forwarding cost-model
bail-out will update these checks, making its effect visible as a diff
on a non-X86 target.

---------

Co-authored-by: Cursor <cursoragent at cursor.com>

Added: 
    llvm/test/Transforms/SLPVectorizer/AArch64/store-load-forward-conflict.ll

Modified: 
    

Removed: 
    


################################################################################
diff  --git a/llvm/test/Transforms/SLPVectorizer/AArch64/store-load-forward-conflict.ll b/llvm/test/Transforms/SLPVectorizer/AArch64/store-load-forward-conflict.ll
new file mode 100644
index 0000000000000..d62d3153f7480
--- /dev/null
+++ b/llvm/test/Transforms/SLPVectorizer/AArch64/store-load-forward-conflict.ll
@@ -0,0 +1,131 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt < %s -passes=slp-vectorizer -S -mtriple=aarch64-- -mcpu=neoverse-v2 | FileCheck %s
+
+; Widened backward load at a misaligned distance: the four consecutive loads
+; A[i-5..i-2] widen to a 16-byte <4 x i32> load, 20 bytes behind the 16-byte
+; <4 x i32> store at A[i] (20 % 16 = 4 -> misaligned; the 16-byte load overruns
+; the 4 bytes left before the boundary and straddles two widened stores). Today
+; SLP widens this chain; the STLF patch will make it stay scalar.
+define void @stlf_conflict_widened_misaligned(ptr noalias %A, i64 %n) {
+; CHECK-LABEL: define void @stlf_conflict_widened_misaligned(
+; CHECK-SAME: ptr noalias [[A:%.*]], i64 [[N:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    br label %[[FOR_BODY:.*]]
+; CHECK:       [[FOR_BODY]]:
+; CHECK-NEXT:    [[I:%.*]] = phi i64 [ 8, %[[ENTRY]] ], [ [[I_NEXT:%.*]], %[[FOR_BODY]] ]
+; CHECK-NEXT:    [[B0:%.*]] = add i64 [[I]], -5
+; CHECK-NEXT:    [[P0:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[B0]]
+; CHECK-NEXT:    [[GEP0:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[I]]
+; CHECK-NEXT:    [[TMP0:%.*]] = load <4 x i32>, ptr [[P0]], align 4
+; CHECK-NEXT:    [[TMP1:%.*]] = add nsw <4 x i32> [[TMP0]], <i32 1, i32 2, i32 3, i32 4>
+; CHECK-NEXT:    store <4 x i32> [[TMP1]], ptr [[GEP0]], align 4
+; CHECK-NEXT:    [[I_NEXT]] = add nuw nsw i64 [[I]], 4
+; CHECK-NEXT:    [[CMP:%.*]] = icmp slt i64 [[I_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[CMP]], label %[[FOR_BODY]], label %[[FOR_END:.*]]
+; CHECK:       [[FOR_END]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %for.body
+
+for.body:
+  %i = phi i64 [ 8, %entry ], [ %i.next, %for.body ]
+
+  %b0 = add i64 %i, -5
+  %b1 = add i64 %i, -4
+  %b2 = add i64 %i, -3
+  %b3 = add i64 %i, -2
+  %p0 = getelementptr inbounds i32, ptr %A, i64 %b0
+  %p1 = getelementptr inbounds i32, ptr %A, i64 %b1
+  %p2 = getelementptr inbounds i32, ptr %A, i64 %b2
+  %p3 = getelementptr inbounds i32, ptr %A, i64 %b3
+  %l0 = load i32, ptr %p0, align 4
+  %l1 = load i32, ptr %p1, align 4
+  %l2 = load i32, ptr %p2, align 4
+  %l3 = load i32, ptr %p3, align 4
+
+  %t1 = add nsw i32 %l0, 1
+  %t2 = add nsw i32 %l1, 2
+  %t3 = add nsw i32 %l2, 3
+  %t4 = add nsw i32 %l3, 4
+
+  %i1 = add nuw nsw i64 %i, 1
+  %i2 = add nuw nsw i64 %i, 2
+  %i3 = add nuw nsw i64 %i, 3
+  %gep0 = getelementptr inbounds i32, ptr %A, i64 %i
+  %gep1 = getelementptr inbounds i32, ptr %A, i64 %i1
+  %gep2 = getelementptr inbounds i32, ptr %A, i64 %i2
+  %gep3 = getelementptr inbounds i32, ptr %A, i64 %i3
+  store i32 %t1, ptr %gep0, align 4
+  store i32 %t2, ptr %gep1, align 4
+  store i32 %t3, ptr %gep2, align 4
+  store i32 %t4, ptr %gep3, align 4
+
+  %i.next = add nuw nsw i64 %i, 4
+  %cmp = icmp slt i64 %i.next, %n
+  br i1 %cmp, label %for.body, label %for.end
+
+for.end:
+  ret void
+}
+
+; Precision control: a single narrow scalar i32 load A[i-1], 4 bytes behind
+; (4 % 16 = 4 -> misaligned) but only 4 bytes wide, so it fits inside the bytes
+; left before the window boundary and never straddles two widened stores. No
+; forwarding conflict exists, so the chain widens both before and after the
+; STLF patch.
+define void @stlf_no_conflict_narrow_scalar_load(ptr noalias %A, i64 %n) {
+; CHECK-LABEL: define void @stlf_no_conflict_narrow_scalar_load(
+; CHECK-SAME: ptr noalias [[A:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    br label %[[FOR_BODY:.*]]
+; CHECK:       [[FOR_BODY]]:
+; CHECK-NEXT:    [[I:%.*]] = phi i64 [ 1, %[[ENTRY]] ], [ [[I_NEXT:%.*]], %[[FOR_BODY]] ]
+; CHECK-NEXT:    [[BACK_IDX:%.*]] = sub i64 [[I]], 1
+; CHECK-NEXT:    [[BACK_GEP:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[BACK_IDX]]
+; CHECK-NEXT:    [[T:%.*]] = load i32, ptr [[BACK_GEP]], align 4
+; CHECK-NEXT:    [[GEP0:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[I]]
+; CHECK-NEXT:    [[TMP0:%.*]] = insertelement <4 x i32> poison, i32 [[T]], i64 0
+; CHECK-NEXT:    [[TMP1:%.*]] = shufflevector <4 x i32> [[TMP0]], <4 x i32> poison, <4 x i32> zeroinitializer
+; CHECK-NEXT:    [[TMP2:%.*]] = add nsw <4 x i32> [[TMP1]], <i32 1, i32 2, i32 3, i32 4>
+; CHECK-NEXT:    store <4 x i32> [[TMP2]], ptr [[GEP0]], align 4
+; CHECK-NEXT:    [[I_NEXT]] = add nuw nsw i64 [[I]], 4
+; CHECK-NEXT:    [[CMP:%.*]] = icmp slt i64 [[I_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[CMP]], label %[[FOR_BODY]], label %[[FOR_END:.*]]
+; CHECK:       [[FOR_END]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %for.body
+
+for.body:
+  %i = phi i64 [ 1, %entry ], [ %i.next, %for.body ]
+
+  %back.idx = sub i64 %i, 1
+  %back.gep = getelementptr inbounds i32, ptr %A, i64 %back.idx
+  %t = load i32, ptr %back.gep, align 4
+
+  %t1 = add nsw i32 %t, 1
+  %t2 = add nsw i32 %t, 2
+  %t3 = add nsw i32 %t, 3
+  %t4 = add nsw i32 %t, 4
+
+  %i1 = add nuw nsw i64 %i, 1
+  %i2 = add nuw nsw i64 %i, 2
+  %i3 = add nuw nsw i64 %i, 3
+  %gep0 = getelementptr inbounds i32, ptr %A, i64 %i
+  %gep1 = getelementptr inbounds i32, ptr %A, i64 %i1
+  %gep2 = getelementptr inbounds i32, ptr %A, i64 %i2
+  %gep3 = getelementptr inbounds i32, ptr %A, i64 %i3
+  store i32 %t1, ptr %gep0, align 4
+  store i32 %t2, ptr %gep1, align 4
+  store i32 %t3, ptr %gep2, align 4
+  store i32 %t4, ptr %gep3, align 4
+
+  %i.next = add nuw nsw i64 %i, 4
+  %cmp = icmp slt i64 %i.next, %n
+  br i1 %cmp, label %for.body, label %for.end
+
+for.end:
+  ret void
+}


        


More information about the llvm-commits mailing list