[llvm] [AArch64][DAG] Split stores of merged i32 pairs so they can form STP (PR #227179)

Jerry Dang via llvm-commits llvm-commits at lists.llvm.org
Tue Sep 29 17:07:19 PDT 2026


https://github.com/kuroyukiasuna updated https://github.com/llvm/llvm-project/pull/227179

>From 7fb034054145055efc1942cd3675aeb661a0cfa4 Mon Sep 17 00:00:00 2001
From: Jerry Dang <kuroyukiasuna at gmail.com>
Date: Mon, 28 Sep 2026 14:23:50 -0400
Subject: [PATCH 1/2] [AArch64][DAG] Split stores of merged i32 pairs so they
 can form STP

---
 llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp |  20 +-
 llvm/lib/Target/AArch64/AArch64ISelLowering.h |   7 +
 llvm/test/CodeGen/AArch64/split-store.ll      | 213 ++++++++++++++++++
 3 files changed, 232 insertions(+), 8 deletions(-)
 create mode 100644 llvm/test/CodeGen/AArch64/split-store.ll

diff --git a/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp b/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp
index e86514aed9410..678b1863616e9 100644
--- a/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp
@@ -25055,14 +25055,15 @@ SDValue DAGCombiner::splitMergedValStore(StoreSDNode *ST) {
       Hi.getOperand(0).getValueSizeInBits() > HalfValBitSize)
     return SDValue();
 
-  // Use the EVT of low and high parts before bitcast as the input
-  // of target query.
-  EVT LowTy = (Lo.getOperand(0).getOpcode() == ISD::BITCAST)
-                  ? Lo.getOperand(0).getValueType()
-                  : Lo.getValueType();
-  EVT HighTy = (Hi.getOperand(0).getOpcode() == ISD::BITCAST)
-                   ? Hi.getOperand(0).getValueType()
-                   : Hi.getValueType();
+  // Use the EVT of low and high parts before zext and bitcast as the input
+  // of target query, matching the CodeGenPrepare version of this transform.
+  auto GetPartTy = [](SDValue Part) {
+    SDValue V = Part.getOperand(0);
+    return V.getOpcode() == ISD::BITCAST ? V.getOperand(0).getValueType()
+                                         : V.getValueType();
+  };
+  EVT LowTy = GetPartTy(Lo);
+  EVT HighTy = GetPartTy(Hi);
   if (!TLI.isMultiStoresCheaperThanBitsMerge(LowTy, HighTy))
     return SDValue();
 
@@ -25074,6 +25075,9 @@ SDValue DAGCombiner::splitMergedValStore(StoreSDNode *ST) {
   EVT VT = EVT::getIntegerVT(*DAG.getContext(), HalfValBitSize);
   Lo = DAG.getNode(ISD::ZERO_EXTEND, DL, VT, Lo.getOperand(0));
   Hi = DAG.getNode(ISD::ZERO_EXTEND, DL, VT, Hi.getOperand(0));
+  // On big-endian targets the high part goes at the lower address.
+  if (DAG.getDataLayout().isBigEndian())
+    std::swap(Lo, Hi);
 
   SDValue Chain = ST->getChain();
   SDValue Ptr = ST->getBasePtr();
diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.h b/llvm/lib/Target/AArch64/AArch64ISelLowering.h
index b68e06dede580..5fe5e4c121c4f 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.h
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.h
@@ -163,6 +163,13 @@ class AArch64TargetLowering : public TargetLowering {
   /// shuffle mask can be codegen'd directly.
   bool isVectorClearMaskLegal(ArrayRef<int> M, EVT VT) const override;
 
+  bool isMultiStoresCheaperThanBitsMerge(EVT LTy, EVT HTy) const override {
+    // AArch64LoadStoreOptimizer merges the two stores into a single STP, so the
+    // ORR (and any zero-extend) goes away without adding a store. Narrower
+    // halves have no STP form and would only trade a BFI for an extra store.
+    return LTy == MVT::i32 && HTy == MVT::i32;
+  }
+
   /// Return the ISD::SETCC ValueType.
   EVT getSetCCResultType(const DataLayout &DL, LLVMContext &Context,
                          EVT VT) const override;
diff --git a/llvm/test/CodeGen/AArch64/split-store.ll b/llvm/test/CodeGen/AArch64/split-store.ll
new file mode 100644
index 0000000000000..18352a3a194d2
--- /dev/null
+++ b/llvm/test/CodeGen/AArch64/split-store.ll
@@ -0,0 +1,213 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc -mtriple=aarch64 < %s | FileCheck %s --check-prefixes=CHECK,CHECK-LE,CHECK-LE-CGP
+; RUN: llc -mtriple=aarch64_be < %s | FileCheck %s --check-prefixes=CHECK,CHECK-BE
+; RUN: llc -mtriple=aarch64 -disable-cgp < %s | FileCheck %s --check-prefixes=CHECK,CHECK-LE,CHECK-LE-NOCGP
+; RUN: llc -mtriple=aarch64_be -disable-cgp < %s | FileCheck %s --check-prefixes=CHECK,CHECK-BE
+
+; A store of an i64 merged from two i32 halves is split into two i32 stores,
+; which are then combined into a single STP.
+
+define void @int32_int32_loads(ptr %p, ptr %a, ptr %b) {
+; CHECK-LE-CGP-LABEL: int32_int32_loads:
+; CHECK-LE-CGP:       // %bb.0:
+; CHECK-LE-CGP-NEXT:    ldr w8, [x1]
+; CHECK-LE-CGP-NEXT:    ldr w9, [x2]
+; CHECK-LE-CGP-NEXT:    stp w8, w9, [x0]
+; CHECK-LE-CGP-NEXT:    ret
+;
+; CHECK-BE-LABEL: int32_int32_loads:
+; CHECK-BE:       // %bb.0:
+; CHECK-BE-NEXT:    ldr w8, [x1]
+; CHECK-BE-NEXT:    ldr w9, [x2]
+; CHECK-BE-NEXT:    stp w9, w8, [x0]
+; CHECK-BE-NEXT:    ret
+;
+; CHECK-LE-NOCGP-LABEL: int32_int32_loads:
+; CHECK-LE-NOCGP:       // %bb.0:
+; CHECK-LE-NOCGP-NEXT:    ldr w8, [x2]
+; CHECK-LE-NOCGP-NEXT:    ldr w9, [x1]
+; CHECK-LE-NOCGP-NEXT:    stp w9, w8, [x0]
+; CHECK-LE-NOCGP-NEXT:    ret
+  %x = load i32, ptr %a, align 4
+  %y = load i32, ptr %b, align 4
+  %hi.ext = zext i32 %y to i64
+  %hi = shl nuw i64 %hi.ext, 32
+  %lo = zext i32 %x to i64
+  %v = or disjoint i64 %hi, %lo
+  store i64 %v, ptr %p, align 4
+  ret void
+}
+
+define void @int32_int32(ptr %p, i32 %x, i32 %y) {
+; CHECK-LE-LABEL: int32_int32:
+; CHECK-LE:       // %bb.0:
+; CHECK-LE-NEXT:    stp w1, w2, [x0]
+; CHECK-LE-NEXT:    ret
+;
+; CHECK-BE-LABEL: int32_int32:
+; CHECK-BE:       // %bb.0:
+; CHECK-BE-NEXT:    stp w2, w1, [x0]
+; CHECK-BE-NEXT:    ret
+  %lo = zext i32 %x to i64
+  %hi.ext = zext i32 %y to i64
+  %hi = shl i64 %hi.ext, 32
+  %v = or i64 %hi, %lo
+  store i64 %v, ptr %p
+  ret void
+}
+
+define void @int32_int32_commuted(ptr %p, i32 %x, i32 %y) {
+; CHECK-LE-LABEL: int32_int32_commuted:
+; CHECK-LE:       // %bb.0:
+; CHECK-LE-NEXT:    stp w1, w2, [x0]
+; CHECK-LE-NEXT:    ret
+;
+; CHECK-BE-LABEL: int32_int32_commuted:
+; CHECK-BE:       // %bb.0:
+; CHECK-BE-NEXT:    stp w2, w1, [x0]
+; CHECK-BE-NEXT:    ret
+  %lo = zext i32 %x to i64
+  %hi.ext = zext i32 %y to i64
+  %hi = shl i64 %hi.ext, 32
+  %v = or i64 %lo, %hi
+  store i64 %v, ptr %p
+  ret void
+}
+
+; The merged value has another use, so the ORR stays but the store is split.
+define i64 @int32_int32_multiuse(ptr %p, i32 %x, i32 %y) {
+; CHECK-LE-LABEL: int32_int32_multiuse:
+; CHECK-LE:       // %bb.0:
+; CHECK-LE-NEXT:    mov w9, w1
+; CHECK-LE-NEXT:    // kill: def $w2 killed $w2 def $x2
+; CHECK-LE-NEXT:    mov x8, x0
+; CHECK-LE-NEXT:    orr x0, x9, x2, lsl #32
+; CHECK-LE-NEXT:    stp w1, w2, [x8]
+; CHECK-LE-NEXT:    ret
+;
+; CHECK-BE-LABEL: int32_int32_multiuse:
+; CHECK-BE:       // %bb.0:
+; CHECK-BE-NEXT:    mov w9, w1
+; CHECK-BE-NEXT:    // kill: def $w2 killed $w2 def $x2
+; CHECK-BE-NEXT:    mov x8, x0
+; CHECK-BE-NEXT:    orr x0, x9, x2, lsl #32
+; CHECK-BE-NEXT:    stp w2, w1, [x8]
+; CHECK-BE-NEXT:    ret
+  %lo = zext i32 %x to i64
+  %hi.ext = zext i32 %y to i64
+  %hi = shl i64 %hi.ext, 32
+  %v = or i64 %hi, %lo
+  store i64 %v, ptr %p
+  ret i64 %v
+}
+
+; The offset is out of range for STP, so the halves stay as two STRs.
+define void @int32_int32_far_offset(ptr %p, i32 %x, i32 %y) {
+; CHECK-LE-CGP-LABEL: int32_int32_far_offset:
+; CHECK-LE-CGP:       // %bb.0:
+; CHECK-LE-CGP-NEXT:    str w1, [x0, #4000]
+; CHECK-LE-CGP-NEXT:    str w2, [x0, #4004]
+; CHECK-LE-CGP-NEXT:    ret
+;
+; CHECK-BE-LABEL: int32_int32_far_offset:
+; CHECK-BE:       // %bb.0:
+; CHECK-BE-NEXT:    str w1, [x0, #4004]
+; CHECK-BE-NEXT:    str w2, [x0, #4000]
+; CHECK-BE-NEXT:    ret
+;
+; CHECK-LE-NOCGP-LABEL: int32_int32_far_offset:
+; CHECK-LE-NOCGP:       // %bb.0:
+; CHECK-LE-NOCGP-NEXT:    str w2, [x0, #4004]
+; CHECK-LE-NOCGP-NEXT:    str w1, [x0, #4000]
+; CHECK-LE-NOCGP-NEXT:    ret
+  %lo = zext i32 %x to i64
+  %hi.ext = zext i32 %y to i64
+  %hi = shl i64 %hi.ext, 32
+  %v = or i64 %hi, %lo
+  %addr = getelementptr inbounds i8, ptr %p, i64 4000
+  store i64 %v, ptr %addr
+  ret void
+}
+
+; Narrower halves have no STP form, so the merged store is kept.
+define void @int16_int16(ptr %p, i16 %x, i16 %y) {
+; CHECK-LABEL: int16_int16:
+; CHECK:       // %bb.0:
+; CHECK-NEXT:    bfi w1, w2, #16, #16
+; CHECK-NEXT:    str w1, [x0]
+; CHECK-NEXT:    ret
+  %lo = zext i16 %x to i32
+  %hi.ext = zext i16 %y to i32
+  %hi = shl i32 %hi.ext, 16
+  %v = or i32 %hi, %lo
+  store i32 %v, ptr %p
+  ret void
+}
+
+define void @int8_int8(ptr %p, i8 %x, i8 %y) {
+; CHECK-LABEL: int8_int8:
+; CHECK:       // %bb.0:
+; CHECK-NEXT:    bfi w1, w2, #8, #24
+; CHECK-NEXT:    strh w1, [x0]
+; CHECK-NEXT:    ret
+  %lo = zext i8 %x to i16
+  %hi.ext = zext i8 %y to i16
+  %hi = shl i16 %hi.ext, 8
+  %v = or i16 %hi, %lo
+  store i16 %v, ptr %p
+  ret void
+}
+
+; A volatile store must not be split.
+define void @int32_int32_volatile(ptr %p, i32 %x, i32 %y) {
+; CHECK-LABEL: int32_int32_volatile:
+; CHECK:       // %bb.0:
+; CHECK-NEXT:    mov w8, w1
+; CHECK-NEXT:    // kill: def $w2 killed $w2 def $x2
+; CHECK-NEXT:    orr x8, x8, x2, lsl #32
+; CHECK-NEXT:    str x8, [x0]
+; CHECK-NEXT:    ret
+  %lo = zext i32 %x to i64
+  %hi.ext = zext i32 %y to i64
+  %hi = shl i64 %hi.ext, 32
+  %v = or i64 %hi, %lo
+  store volatile i64 %v, ptr %p
+  ret void
+}
+
+; An int/FP pair can't be stored with one STP, so it is not split.
+define void @int32_float(ptr %p, i32 %x, float %y) {
+; CHECK-LABEL: int32_float:
+; CHECK:       // %bb.0:
+; CHECK-NEXT:    fmov w8, s0
+; CHECK-NEXT:    mov w9, w1
+; CHECK-NEXT:    orr x8, x9, x8, lsl #32
+; CHECK-NEXT:    str x8, [x0]
+; CHECK-NEXT:    ret
+  %y.int = bitcast float %y to i32
+  %lo = zext i32 %x to i64
+  %hi.ext = zext i32 %y.int to i64
+  %hi = shl i64 %hi.ext, 32
+  %v = or i64 %hi, %lo
+  store i64 %v, ptr %p
+  ret void
+}
+
+; FP pairs are not split either.
+define void @float_float(ptr %p, float %x, float %y) {
+; CHECK-LABEL: float_float:
+; CHECK:       // %bb.0:
+; CHECK-NEXT:    fmov w8, s0
+; CHECK-NEXT:    fmov w9, s1
+; CHECK-NEXT:    orr x8, x8, x9, lsl #32
+; CHECK-NEXT:    str x8, [x0]
+; CHECK-NEXT:    ret
+  %x.int = bitcast float %x to i32
+  %y.int = bitcast float %y to i32
+  %lo = zext i32 %x.int to i64
+  %hi.ext = zext i32 %y.int to i64
+  %hi = shl i64 %hi.ext, 32
+  %v = or i64 %hi, %lo
+  store i64 %v, ptr %p
+  ret void
+}

>From 57079c1fff96b90d90d5eaebb1cace3ae5177f21 Mon Sep 17 00:00:00 2001
From: Jerry Dang <kuroyukiasuna at gmail.com>
Date: Tue, 29 Sep 2026 20:00:10 -0400
Subject: [PATCH 2/2] Add adjacent-load guard in splitMergedValStore for both
 CGP and DAG

---
 llvm/lib/CodeGen/CodeGenPrepare.cpp           | 16 ++++++
 llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp | 11 ++++
 llvm/test/CodeGen/AArch64/split-store.ll      | 56 +++++++++++++++++++
 3 files changed, 83 insertions(+)

diff --git a/llvm/lib/CodeGen/CodeGenPrepare.cpp b/llvm/lib/CodeGen/CodeGenPrepare.cpp
index 3e0b8a956ca81..c847474349c1f 100644
--- a/llvm/lib/CodeGen/CodeGenPrepare.cpp
+++ b/llvm/lib/CodeGen/CodeGenPrepare.cpp
@@ -8664,6 +8664,22 @@ static bool splitMergedValStore(StoreInst &SI, const DataLayout &DL,
   if (!ForceSplitStore && !TLI.isMultiStoresCheaperThanBitsMerge(LowTy, HighTy))
     return false;
 
+  // If both halves are loaded from adjacent memory, the merged value is really
+  // a single wide load, which may also pair with a neighboring load.
+  auto *LLoad = dyn_cast<LoadInst>(LValue);
+  auto *HLoad = dyn_cast<LoadInst>(HValue);
+  if (!ForceSplitStore && LLoad && HLoad && LLoad->isSimple() &&
+      HLoad->isSimple() && LLoad->getParent() == HLoad->getParent() &&
+      DL.getTypeStoreSizeInBits(LLoad->getType()) == HalfValBitSize &&
+      DL.getTypeStoreSizeInBits(HLoad->getType()) == HalfValBitSize) {
+    int64_t HalfBytes = HalfValBitSize / 8;
+    std::optional<int64_t> Offset =
+        HLoad->getPointerOperand()->getPointerOffsetFrom(
+            LLoad->getPointerOperand(), DL);
+    if (Offset && *Offset == (DL.isLittleEndian() ? HalfBytes : -HalfBytes))
+      return false;
+  }
+
   // Start to split store.
   IRBuilder<> Builder(SI.getContext());
   Builder.SetInsertPoint(&SI);
diff --git a/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp b/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp
index 678b1863616e9..a2fdc05a8e0bc 100644
--- a/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp
@@ -25067,6 +25067,17 @@ SDValue DAGCombiner::splitMergedValStore(StoreSDNode *ST) {
   if (!TLI.isMultiStoresCheaperThanBitsMerge(LowTy, HighTy))
     return SDValue();
 
+  // If both halves are loaded from adjacent memory, leave the value for load
+  // combining to turn into a single wide load.
+  auto *LoLd = dyn_cast<LoadSDNode>(Lo.getOperand(0));
+  auto *HiLd = dyn_cast<LoadSDNode>(Hi.getOperand(0));
+  if (LoLd && HiLd) {
+    bool IsLE = DAG.getDataLayout().isLittleEndian();
+    if (DAG.areNonVolatileConsecutiveLoads(
+            IsLE ? HiLd : LoLd, IsLE ? LoLd : HiLd, HalfValBitSize / 8, 1))
+      return SDValue();
+  }
+
   // Start to split store.
   MachineMemOperand::Flags MMOFlags = ST->getMemOperand()->getFlags();
   AAMDNodes AAInfo = ST->getAAInfo();
diff --git a/llvm/test/CodeGen/AArch64/split-store.ll b/llvm/test/CodeGen/AArch64/split-store.ll
index 18352a3a194d2..5f6c6b063ad5a 100644
--- a/llvm/test/CodeGen/AArch64/split-store.ll
+++ b/llvm/test/CodeGen/AArch64/split-store.ll
@@ -211,3 +211,59 @@ define void @float_float(ptr %p, float %x, float %y) {
   store i64 %v, ptr %p
   ret void
 }
+
+; The halves are loaded from adjacent memory, so on little-endian the merged
+; value is a single i64 load that pairs with the neighboring i64 load, and the
+; store is not split. On big-endian the halves are in the wrong order for that.
+define void @int32_int32_adjacent_loads(ptr %p, ptr %a) {
+; CHECK-LE-LABEL: int32_int32_adjacent_loads:
+; CHECK-LE:       // %bb.0:
+; CHECK-LE-NEXT:    ldp x8, x9, [x1]
+; CHECK-LE-NEXT:    stp x8, x9, [x0]
+; CHECK-LE-NEXT:    ret
+;
+; CHECK-BE-LABEL: int32_int32_adjacent_loads:
+; CHECK-BE:       // %bb.0:
+; CHECK-BE-NEXT:    ldp w8, w9, [x1]
+; CHECK-BE-NEXT:    ldr x10, [x1, #8]
+; CHECK-BE-NEXT:    stp w9, w8, [x0]
+; CHECK-BE-NEXT:    str x10, [x0, #8]
+; CHECK-BE-NEXT:    ret
+  %x = load i32, ptr %a, align 4
+  %a4 = getelementptr inbounds i8, ptr %a, i64 4
+  %y = load i32, ptr %a4, align 4
+  %a8 = getelementptr inbounds i8, ptr %a, i64 8
+  %z = load i64, ptr %a8, align 8
+  %lo = zext i32 %x to i64
+  %hi.ext = zext i32 %y to i64
+  %hi = shl nuw i64 %hi.ext, 32
+  %v = or disjoint i64 %hi, %lo
+  store i64 %v, ptr %p, align 8
+  %p8 = getelementptr inbounds i8, ptr %p, i64 8
+  store i64 %z, ptr %p8, align 8
+  ret void
+}
+
+; A neighboring i64 store no longer pairs with the merged store, but the split
+; still saves the ORR.
+define void @int32_int32_adjacent_i64_store(ptr %p, i32 %x, i32 %y, i64 %z) {
+; CHECK-LE-LABEL: int32_int32_adjacent_i64_store:
+; CHECK-LE:       // %bb.0:
+; CHECK-LE-NEXT:    stp w1, w2, [x0]
+; CHECK-LE-NEXT:    str x3, [x0, #8]
+; CHECK-LE-NEXT:    ret
+;
+; CHECK-BE-LABEL: int32_int32_adjacent_i64_store:
+; CHECK-BE:       // %bb.0:
+; CHECK-BE-NEXT:    stp w2, w1, [x0]
+; CHECK-BE-NEXT:    str x3, [x0, #8]
+; CHECK-BE-NEXT:    ret
+  %lo = zext i32 %x to i64
+  %hi.ext = zext i32 %y to i64
+  %hi = shl nuw i64 %hi.ext, 32
+  %v = or disjoint i64 %hi, %lo
+  store i64 %v, ptr %p, align 8
+  %p8 = getelementptr inbounds i8, ptr %p, i64 8
+  store i64 %z, ptr %p8, align 8
+  ret void
+}



More information about the llvm-commits mailing list