[llvm] [AArch64] Fix hasNearbyPairedStore to handle non-inbounds GEPs (PR #199137)

Kunal Pathak via llvm-commits llvm-commits at lists.llvm.org
Fri May 22 07:46:24 PDT 2026


https://github.com/kunalspathak updated https://github.com/llvm/llvm-project/pull/199137

>From 601286979d8e6b2fedee879f3dc461a8a2f34528 Mon Sep 17 00:00:00 2001
From: Kunal Pathak <kupathak at fb.com>
Date: Tue, 19 May 2026 08:46:10 -0700
Subject: [PATCH 1/2] [AArch64] Fix hasNearbyPairedStore to handle non-inbounds
 GEPs

Summary:
`hasNearbyPairedStore`` uses `stripAndAccumulateInBoundsConstantOffsets`` to
decompose store pointers into (base, offset) pairs and check whether
two stores are 16 bytes apart. This fails when LSR has
rewritten pointer arithmetic into non-inbounds GEPs because the function
refuses to look through them. The two stores then appear to have
different base pointers and the check returns false.

When this happens, lowerInterleavedStore proceeds to emit ST2 for a
pattern that would be more profitable as zip+stp, since the load-store
optimizer can pair adjacent stores into STP but cannot merge ST2 with
anything. On a bf16-to-fp32 NEON conversion loop this causes a
regression from 11 to 17 instructions per iteration.

Switch to stripAndAccumulateConstantOffsets with AllowNonInbounds=true.
The function is a bail-out heuristic doing pure address arithmetic, so
the inbounds semantic guarantee is not needed for correctness.

Test Plan:

Added llvm/test/CodeGen/AArch64/interleaved-store-noninbounds-gep.ll

Reviewers:

Subscribers:

Tasks:

Tags:

Differential Revision: https://phabricator.intern.facebook.com/D105490334
---
 .../Target/AArch64/AArch64ISelLowering.cpp    |  7 +--
 .../interleaved-store-noninbounds-gep.ll      | 44 +++++++++++++++++++
 2 files changed, 48 insertions(+), 3 deletions(-)
 create mode 100644 llvm/test/CodeGen/AArch64/interleaved-store-noninbounds-gep.ll

diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index fddd3f96ba66c..97efc3a143c14 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -18779,7 +18779,8 @@ bool hasNearbyPairedStore(Iter It, Iter End, Value *Ptr, const DataLayout &DL) {
   unsigned IdxWidth = DL.getIndexSizeInBits(0);
   APInt OffsetA(IdxWidth, 0), OffsetB(IdxWidth, 0);
   const Value *PtrA1 =
-      Ptr->stripAndAccumulateInBoundsConstantOffsets(DL, OffsetA);
+      Ptr->stripAndAccumulateConstantOffsets(DL, OffsetA,
+                                             /*AllowNonInbounds=*/ true);
 
   while (++It != End) {
     if (It->isDebugOrPseudoInst())
@@ -18788,8 +18789,8 @@ bool hasNearbyPairedStore(Iter It, Iter End, Value *Ptr, const DataLayout &DL) {
       break;
     if (const auto *SI = dyn_cast<StoreInst>(&*It)) {
       const Value *PtrB1 =
-          SI->getPointerOperand()->stripAndAccumulateInBoundsConstantOffsets(
-              DL, OffsetB);
+          SI->getPointerOperand()->stripAndAccumulateConstantOffsets(
+              DL, OffsetB, /*AllowNonInbounds=*/ true);
       if (PtrA1 == PtrB1 &&
           (OffsetA.sextOrTrunc(IdxWidth) - OffsetB.sextOrTrunc(IdxWidth))
                   .abs() == 16)
diff --git a/llvm/test/CodeGen/AArch64/interleaved-store-noninbounds-gep.ll b/llvm/test/CodeGen/AArch64/interleaved-store-noninbounds-gep.ll
new file mode 100644
index 0000000000000..a40e38d02abd7
--- /dev/null
+++ b/llvm/test/CodeGen/AArch64/interleaved-store-noninbounds-gep.ll
@@ -0,0 +1,44 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 2
+; RUN: llc -mtriple aarch64-none-linux-gnu < %s | FileCheck %s
+
+; Verify that hasNearbyPairedStore sees through non-inbounds GEPs
+; (as produced by LoopStrengthReduce) and avoids unprofitable ST2
+; lowering in favor of zip+stp.
+
+define void @st2_noninbounds_gep(ptr %buf, <8 x i16> %a) {
+; CHECK-LABEL: st2_noninbounds_gep:
+; CHECK:       // %bb.0: // %entry
+; CHECK-NEXT:    movi v1.2d, #0000000000000000
+; CHECK-NEXT:    zip1 v2.8h, v0.8h, v1.8h
+; CHECK-NEXT:    zip2 v0.8h, v0.8h, v1.8h
+; CHECK-NEXT:    stp q2, q0, [x0]
+; CHECK-NEXT:    ret
+entry:
+  %vzip.i = shufflevector <8 x i16> %a, <8 x i16> <i16 0, i16 0, i16 0, i16 0, i16 poison, i16 poison, i16 poison, i16 poison>, <8 x i32> <i32 0, i32 8, i32 1, i32 9, i32 2, i32 10, i32 3, i32 11>
+  %vzip1.i = shufflevector <8 x i16> %a, <8 x i16> <i16 poison, i16 poison, i16 poison, i16 poison, i16 0, i16 0, i16 0, i16 0>, <8 x i32> <i32 4, i32 12, i32 5, i32 13, i32 6, i32 14, i32 7, i32 15>
+  store <8 x i16> %vzip.i, ptr %buf, align 4
+  %add.ptr = getelementptr i32, ptr %buf, i64 4
+  store <8 x i16> %vzip1.i, ptr %add.ptr, align 4
+  ret void
+}
+
+; Same pattern but with a negative non-inbounds offset, mimicking LSR output
+; where the base pointer is advanced past the stores and offsets go negative.
+define void @st2_noninbounds_negative_offset(ptr %buf, <8 x i16> %a) {
+; CHECK-LABEL: st2_noninbounds_negative_offset:
+; CHECK:       // %bb.0: // %entry
+; CHECK-NEXT:    movi v1.2d, #0000000000000000
+; CHECK-NEXT:    zip1 v2.8h, v0.8h, v1.8h
+; CHECK-NEXT:    zip2 v0.8h, v0.8h, v1.8h
+; CHECK-NEXT:    stp q2, q0, [x0]
+; CHECK-NEXT:    ret
+entry:
+  %advanced = getelementptr i8, ptr %buf, i64 32
+  %vzip.i = shufflevector <8 x i16> %a, <8 x i16> <i16 0, i16 0, i16 0, i16 0, i16 poison, i16 poison, i16 poison, i16 poison>, <8 x i32> <i32 0, i32 8, i32 1, i32 9, i32 2, i32 10, i32 3, i32 11>
+  %vzip1.i = shufflevector <8 x i16> %a, <8 x i16> <i16 poison, i16 poison, i16 poison, i16 poison, i16 0, i16 0, i16 0, i16 0>, <8 x i32> <i32 4, i32 12, i32 5, i32 13, i32 6, i32 14, i32 7, i32 15>
+  %ptr0 = getelementptr i8, ptr %advanced, i64 -32
+  store <8 x i16> %vzip.i, ptr %ptr0, align 4
+  %ptr1 = getelementptr i8, ptr %advanced, i64 -16
+  store <8 x i16> %vzip1.i, ptr %ptr1, align 4
+  ret void
+}

>From 67f16b028ec61cc7b1203151994fda08f7f497fd Mon Sep 17 00:00:00 2001
From: Kunal Pathak <kupathak at fb.com>
Date: Fri, 22 May 2026 07:45:04 -0700
Subject: [PATCH 2/2] fix clang formatting

---
 llvm/lib/Target/AArch64/AArch64ISelLowering.cpp | 4 ++--
 1 file changed, 2 insertions(+), 2 deletions(-)

diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index 97efc3a143c14..084ba27d104fd 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -18780,7 +18780,7 @@ bool hasNearbyPairedStore(Iter It, Iter End, Value *Ptr, const DataLayout &DL) {
   APInt OffsetA(IdxWidth, 0), OffsetB(IdxWidth, 0);
   const Value *PtrA1 =
       Ptr->stripAndAccumulateConstantOffsets(DL, OffsetA,
-                                             /*AllowNonInbounds=*/ true);
+                                             /*AllowNonInbounds=*/true);
 
   while (++It != End) {
     if (It->isDebugOrPseudoInst())
@@ -18790,7 +18790,7 @@ bool hasNearbyPairedStore(Iter It, Iter End, Value *Ptr, const DataLayout &DL) {
     if (const auto *SI = dyn_cast<StoreInst>(&*It)) {
       const Value *PtrB1 =
           SI->getPointerOperand()->stripAndAccumulateConstantOffsets(
-              DL, OffsetB, /*AllowNonInbounds=*/ true);
+              DL, OffsetB, /*AllowNonInbounds=*/true);
       if (PtrA1 == PtrB1 &&
           (OffsetA.sextOrTrunc(IdxWidth) - OffsetB.sextOrTrunc(IdxWidth))
                   .abs() == 16)



More information about the llvm-commits mailing list