[llvm] [AArch64][DAG] Split stores of merged i32 pairs so they can form STP (PR #227179)
Jerry Dang via llvm-commits
llvm-commits at lists.llvm.org
Mon Sep 28 19:56:09 PDT 2026
https://github.com/kuroyukiasuna created https://github.com/llvm/llvm-project/pull/227179
SROA turns a copy of struct with 2 i32 fields into a single i64 store of `(or (zext lo), (shl (zext hi), 32))`, which AArch64 emits as ORR + STR. This PR implements the `isMultiStoresCheaperThanBitsMerge` for 2 i32 which enables `AArch64LoadStoreOptimizer` to combine the two stores into a single STP.
Assisted-by: Claude Code
>From 7fb034054145055efc1942cd3675aeb661a0cfa4 Mon Sep 17 00:00:00 2001
From: Jerry Dang <kuroyukiasuna at gmail.com>
Date: Mon, 28 Sep 2026 14:23:50 -0400
Subject: [PATCH] [AArch64][DAG] Split stores of merged i32 pairs so they can
form STP
---
llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp | 20 +-
llvm/lib/Target/AArch64/AArch64ISelLowering.h | 7 +
llvm/test/CodeGen/AArch64/split-store.ll | 213 ++++++++++++++++++
3 files changed, 232 insertions(+), 8 deletions(-)
create mode 100644 llvm/test/CodeGen/AArch64/split-store.ll
diff --git a/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp b/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp
index e86514aed9410..678b1863616e9 100644
--- a/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp
@@ -25055,14 +25055,15 @@ SDValue DAGCombiner::splitMergedValStore(StoreSDNode *ST) {
Hi.getOperand(0).getValueSizeInBits() > HalfValBitSize)
return SDValue();
- // Use the EVT of low and high parts before bitcast as the input
- // of target query.
- EVT LowTy = (Lo.getOperand(0).getOpcode() == ISD::BITCAST)
- ? Lo.getOperand(0).getValueType()
- : Lo.getValueType();
- EVT HighTy = (Hi.getOperand(0).getOpcode() == ISD::BITCAST)
- ? Hi.getOperand(0).getValueType()
- : Hi.getValueType();
+ // Use the EVT of low and high parts before zext and bitcast as the input
+ // of target query, matching the CodeGenPrepare version of this transform.
+ auto GetPartTy = [](SDValue Part) {
+ SDValue V = Part.getOperand(0);
+ return V.getOpcode() == ISD::BITCAST ? V.getOperand(0).getValueType()
+ : V.getValueType();
+ };
+ EVT LowTy = GetPartTy(Lo);
+ EVT HighTy = GetPartTy(Hi);
if (!TLI.isMultiStoresCheaperThanBitsMerge(LowTy, HighTy))
return SDValue();
@@ -25074,6 +25075,9 @@ SDValue DAGCombiner::splitMergedValStore(StoreSDNode *ST) {
EVT VT = EVT::getIntegerVT(*DAG.getContext(), HalfValBitSize);
Lo = DAG.getNode(ISD::ZERO_EXTEND, DL, VT, Lo.getOperand(0));
Hi = DAG.getNode(ISD::ZERO_EXTEND, DL, VT, Hi.getOperand(0));
+ // On big-endian targets the high part goes at the lower address.
+ if (DAG.getDataLayout().isBigEndian())
+ std::swap(Lo, Hi);
SDValue Chain = ST->getChain();
SDValue Ptr = ST->getBasePtr();
diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.h b/llvm/lib/Target/AArch64/AArch64ISelLowering.h
index b68e06dede580..5fe5e4c121c4f 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.h
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.h
@@ -163,6 +163,13 @@ class AArch64TargetLowering : public TargetLowering {
/// shuffle mask can be codegen'd directly.
bool isVectorClearMaskLegal(ArrayRef<int> M, EVT VT) const override;
+ bool isMultiStoresCheaperThanBitsMerge(EVT LTy, EVT HTy) const override {
+ // AArch64LoadStoreOptimizer merges the two stores into a single STP, so the
+ // ORR (and any zero-extend) goes away without adding a store. Narrower
+ // halves have no STP form and would only trade a BFI for an extra store.
+ return LTy == MVT::i32 && HTy == MVT::i32;
+ }
+
/// Return the ISD::SETCC ValueType.
EVT getSetCCResultType(const DataLayout &DL, LLVMContext &Context,
EVT VT) const override;
diff --git a/llvm/test/CodeGen/AArch64/split-store.ll b/llvm/test/CodeGen/AArch64/split-store.ll
new file mode 100644
index 0000000000000..18352a3a194d2
--- /dev/null
+++ b/llvm/test/CodeGen/AArch64/split-store.ll
@@ -0,0 +1,213 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc -mtriple=aarch64 < %s | FileCheck %s --check-prefixes=CHECK,CHECK-LE,CHECK-LE-CGP
+; RUN: llc -mtriple=aarch64_be < %s | FileCheck %s --check-prefixes=CHECK,CHECK-BE
+; RUN: llc -mtriple=aarch64 -disable-cgp < %s | FileCheck %s --check-prefixes=CHECK,CHECK-LE,CHECK-LE-NOCGP
+; RUN: llc -mtriple=aarch64_be -disable-cgp < %s | FileCheck %s --check-prefixes=CHECK,CHECK-BE
+
+; A store of an i64 merged from two i32 halves is split into two i32 stores,
+; which are then combined into a single STP.
+
+define void @int32_int32_loads(ptr %p, ptr %a, ptr %b) {
+; CHECK-LE-CGP-LABEL: int32_int32_loads:
+; CHECK-LE-CGP: // %bb.0:
+; CHECK-LE-CGP-NEXT: ldr w8, [x1]
+; CHECK-LE-CGP-NEXT: ldr w9, [x2]
+; CHECK-LE-CGP-NEXT: stp w8, w9, [x0]
+; CHECK-LE-CGP-NEXT: ret
+;
+; CHECK-BE-LABEL: int32_int32_loads:
+; CHECK-BE: // %bb.0:
+; CHECK-BE-NEXT: ldr w8, [x1]
+; CHECK-BE-NEXT: ldr w9, [x2]
+; CHECK-BE-NEXT: stp w9, w8, [x0]
+; CHECK-BE-NEXT: ret
+;
+; CHECK-LE-NOCGP-LABEL: int32_int32_loads:
+; CHECK-LE-NOCGP: // %bb.0:
+; CHECK-LE-NOCGP-NEXT: ldr w8, [x2]
+; CHECK-LE-NOCGP-NEXT: ldr w9, [x1]
+; CHECK-LE-NOCGP-NEXT: stp w9, w8, [x0]
+; CHECK-LE-NOCGP-NEXT: ret
+ %x = load i32, ptr %a, align 4
+ %y = load i32, ptr %b, align 4
+ %hi.ext = zext i32 %y to i64
+ %hi = shl nuw i64 %hi.ext, 32
+ %lo = zext i32 %x to i64
+ %v = or disjoint i64 %hi, %lo
+ store i64 %v, ptr %p, align 4
+ ret void
+}
+
+define void @int32_int32(ptr %p, i32 %x, i32 %y) {
+; CHECK-LE-LABEL: int32_int32:
+; CHECK-LE: // %bb.0:
+; CHECK-LE-NEXT: stp w1, w2, [x0]
+; CHECK-LE-NEXT: ret
+;
+; CHECK-BE-LABEL: int32_int32:
+; CHECK-BE: // %bb.0:
+; CHECK-BE-NEXT: stp w2, w1, [x0]
+; CHECK-BE-NEXT: ret
+ %lo = zext i32 %x to i64
+ %hi.ext = zext i32 %y to i64
+ %hi = shl i64 %hi.ext, 32
+ %v = or i64 %hi, %lo
+ store i64 %v, ptr %p
+ ret void
+}
+
+define void @int32_int32_commuted(ptr %p, i32 %x, i32 %y) {
+; CHECK-LE-LABEL: int32_int32_commuted:
+; CHECK-LE: // %bb.0:
+; CHECK-LE-NEXT: stp w1, w2, [x0]
+; CHECK-LE-NEXT: ret
+;
+; CHECK-BE-LABEL: int32_int32_commuted:
+; CHECK-BE: // %bb.0:
+; CHECK-BE-NEXT: stp w2, w1, [x0]
+; CHECK-BE-NEXT: ret
+ %lo = zext i32 %x to i64
+ %hi.ext = zext i32 %y to i64
+ %hi = shl i64 %hi.ext, 32
+ %v = or i64 %lo, %hi
+ store i64 %v, ptr %p
+ ret void
+}
+
+; The merged value has another use, so the ORR stays but the store is split.
+define i64 @int32_int32_multiuse(ptr %p, i32 %x, i32 %y) {
+; CHECK-LE-LABEL: int32_int32_multiuse:
+; CHECK-LE: // %bb.0:
+; CHECK-LE-NEXT: mov w9, w1
+; CHECK-LE-NEXT: // kill: def $w2 killed $w2 def $x2
+; CHECK-LE-NEXT: mov x8, x0
+; CHECK-LE-NEXT: orr x0, x9, x2, lsl #32
+; CHECK-LE-NEXT: stp w1, w2, [x8]
+; CHECK-LE-NEXT: ret
+;
+; CHECK-BE-LABEL: int32_int32_multiuse:
+; CHECK-BE: // %bb.0:
+; CHECK-BE-NEXT: mov w9, w1
+; CHECK-BE-NEXT: // kill: def $w2 killed $w2 def $x2
+; CHECK-BE-NEXT: mov x8, x0
+; CHECK-BE-NEXT: orr x0, x9, x2, lsl #32
+; CHECK-BE-NEXT: stp w2, w1, [x8]
+; CHECK-BE-NEXT: ret
+ %lo = zext i32 %x to i64
+ %hi.ext = zext i32 %y to i64
+ %hi = shl i64 %hi.ext, 32
+ %v = or i64 %hi, %lo
+ store i64 %v, ptr %p
+ ret i64 %v
+}
+
+; The offset is out of range for STP, so the halves stay as two STRs.
+define void @int32_int32_far_offset(ptr %p, i32 %x, i32 %y) {
+; CHECK-LE-CGP-LABEL: int32_int32_far_offset:
+; CHECK-LE-CGP: // %bb.0:
+; CHECK-LE-CGP-NEXT: str w1, [x0, #4000]
+; CHECK-LE-CGP-NEXT: str w2, [x0, #4004]
+; CHECK-LE-CGP-NEXT: ret
+;
+; CHECK-BE-LABEL: int32_int32_far_offset:
+; CHECK-BE: // %bb.0:
+; CHECK-BE-NEXT: str w1, [x0, #4004]
+; CHECK-BE-NEXT: str w2, [x0, #4000]
+; CHECK-BE-NEXT: ret
+;
+; CHECK-LE-NOCGP-LABEL: int32_int32_far_offset:
+; CHECK-LE-NOCGP: // %bb.0:
+; CHECK-LE-NOCGP-NEXT: str w2, [x0, #4004]
+; CHECK-LE-NOCGP-NEXT: str w1, [x0, #4000]
+; CHECK-LE-NOCGP-NEXT: ret
+ %lo = zext i32 %x to i64
+ %hi.ext = zext i32 %y to i64
+ %hi = shl i64 %hi.ext, 32
+ %v = or i64 %hi, %lo
+ %addr = getelementptr inbounds i8, ptr %p, i64 4000
+ store i64 %v, ptr %addr
+ ret void
+}
+
+; Narrower halves have no STP form, so the merged store is kept.
+define void @int16_int16(ptr %p, i16 %x, i16 %y) {
+; CHECK-LABEL: int16_int16:
+; CHECK: // %bb.0:
+; CHECK-NEXT: bfi w1, w2, #16, #16
+; CHECK-NEXT: str w1, [x0]
+; CHECK-NEXT: ret
+ %lo = zext i16 %x to i32
+ %hi.ext = zext i16 %y to i32
+ %hi = shl i32 %hi.ext, 16
+ %v = or i32 %hi, %lo
+ store i32 %v, ptr %p
+ ret void
+}
+
+define void @int8_int8(ptr %p, i8 %x, i8 %y) {
+; CHECK-LABEL: int8_int8:
+; CHECK: // %bb.0:
+; CHECK-NEXT: bfi w1, w2, #8, #24
+; CHECK-NEXT: strh w1, [x0]
+; CHECK-NEXT: ret
+ %lo = zext i8 %x to i16
+ %hi.ext = zext i8 %y to i16
+ %hi = shl i16 %hi.ext, 8
+ %v = or i16 %hi, %lo
+ store i16 %v, ptr %p
+ ret void
+}
+
+; A volatile store must not be split.
+define void @int32_int32_volatile(ptr %p, i32 %x, i32 %y) {
+; CHECK-LABEL: int32_int32_volatile:
+; CHECK: // %bb.0:
+; CHECK-NEXT: mov w8, w1
+; CHECK-NEXT: // kill: def $w2 killed $w2 def $x2
+; CHECK-NEXT: orr x8, x8, x2, lsl #32
+; CHECK-NEXT: str x8, [x0]
+; CHECK-NEXT: ret
+ %lo = zext i32 %x to i64
+ %hi.ext = zext i32 %y to i64
+ %hi = shl i64 %hi.ext, 32
+ %v = or i64 %hi, %lo
+ store volatile i64 %v, ptr %p
+ ret void
+}
+
+; An int/FP pair can't be stored with one STP, so it is not split.
+define void @int32_float(ptr %p, i32 %x, float %y) {
+; CHECK-LABEL: int32_float:
+; CHECK: // %bb.0:
+; CHECK-NEXT: fmov w8, s0
+; CHECK-NEXT: mov w9, w1
+; CHECK-NEXT: orr x8, x9, x8, lsl #32
+; CHECK-NEXT: str x8, [x0]
+; CHECK-NEXT: ret
+ %y.int = bitcast float %y to i32
+ %lo = zext i32 %x to i64
+ %hi.ext = zext i32 %y.int to i64
+ %hi = shl i64 %hi.ext, 32
+ %v = or i64 %hi, %lo
+ store i64 %v, ptr %p
+ ret void
+}
+
+; FP pairs are not split either.
+define void @float_float(ptr %p, float %x, float %y) {
+; CHECK-LABEL: float_float:
+; CHECK: // %bb.0:
+; CHECK-NEXT: fmov w8, s0
+; CHECK-NEXT: fmov w9, s1
+; CHECK-NEXT: orr x8, x8, x9, lsl #32
+; CHECK-NEXT: str x8, [x0]
+; CHECK-NEXT: ret
+ %x.int = bitcast float %x to i32
+ %y.int = bitcast float %y to i32
+ %lo = zext i32 %x.int to i64
+ %hi.ext = zext i32 %y.int to i64
+ %hi = shl i64 %hi.ext, 32
+ %v = or i64 %hi, %lo
+ store i64 %v, ptr %p
+ ret void
+}
More information about the llvm-commits
mailing list