[llvm] [RISCV][GlobalISel] Fold large constant offsets in selectAddrRegImm (PR #219161)
via llvm-commits
llvm-commits at lists.llvm.org
Thu Aug 27 02:52:41 PDT 2026
llvmorg-github-actions[bot] wrote:
<!--LLVM PR SUMMARY COMMENT-->
@llvm/pr-subscribers-backend-risc-v
Author: Kane Wang (ReVe1uv)
<details>
<summary>Changes</summary>
Fold ADDI adjustment (AddiPair) for offsets in [-4096, 4094] and split larger constants into materialized Hi + Lo12 offset, matching SDAG. Add isWorthFoldingAdd to guard the split and extract the shared ADDI renderer into renderAddiPair.
Assisted-by: Claude
---
Patch is 25.57 KiB, truncated to 20.00 KiB below, full version: https://github.com/llvm/llvm-project/pull/219161.diff
2 Files Affected:
- (modified) llvm/lib/Target/RISCV/GISel/RISCVInstructionSelector.cpp (+67-16)
- (added) llvm/test/CodeGen/RISCV/GlobalISel/load-store.ll (+641)
``````````diff
diff --git a/llvm/lib/Target/RISCV/GISel/RISCVInstructionSelector.cpp b/llvm/lib/Target/RISCV/GISel/RISCVInstructionSelector.cpp
index bf31c9a9d1e9e..fdd6814020211 100644
--- a/llvm/lib/Target/RISCV/GISel/RISCVInstructionSelector.cpp
+++ b/llvm/lib/Target/RISCV/GISel/RISCVInstructionSelector.cpp
@@ -67,6 +67,7 @@ class RISCVInstructionSelector : public InstructionSelector {
bool isRegInGprb(Register Reg) const;
bool isRegInFprb(Register Reg) const;
+ bool isWorthFoldingAdd(Register AddResult) const;
// tblgen-erated 'select' implementation, used as the initial selector for
// the patterns that don't require complex C++.
@@ -155,7 +156,8 @@ class RISCVInstructionSelector : public InstructionSelector {
}
ComplexRendererFns renderVLOp(MachineOperand &Root) const;
-
+ ComplexRendererFns renderAddiPair(Register BaseReg, int64_t AddiImm,
+ int64_t OffsetImm) const;
// Custom renderers for tablegen
void renderNegImm(MachineInstrBuilder &MIB, const MachineInstr &MI,
int OpIdx) const;
@@ -573,6 +575,22 @@ RISCVInstructionSelector::renderVLOp(MachineOperand &Root) const {
return {{[=](MachineInstrBuilder &MIB) { MIB.addReg(Root.getReg()); }}};
}
+InstructionSelector::ComplexRendererFns
+RISCVInstructionSelector::renderAddiPair(Register BaseReg, int64_t AddiImm,
+ int64_t OffsetImm) const {
+ return {{[=](MachineInstrBuilder &MIB) {
+ Register Tmp = MRI->createVirtualRegister(&RISCV::GPRRegClass);
+ MachineInstr *Addi =
+ BuildMI(*MIB->getParent(), *MIB.getInstr(), MIB->getDebugLoc(),
+ TII.get(RISCV::ADDI), Tmp)
+ .addReg(BaseReg)
+ .addImm(AddiImm);
+ constrainSelectedInstRegOperands(*Addi, TII, TRI, RBI);
+ MIB.addReg(Tmp);
+ },
+ [=](MachineInstrBuilder &MIB) { MIB.addImm(OffsetImm); }}};
+}
+
InstructionSelector::ComplexRendererFns
RISCVInstructionSelector::selectAddrRegImm(MachineOperand &Root) const {
if (!Root.isReg())
@@ -603,10 +621,32 @@ RISCVInstructionSelector::selectAddrRegImm(MachineOperand &Root) const {
return {{[=](MachineInstrBuilder &MIB) { MIB.add(LHS); },
[=](MachineInstrBuilder &MIB) { MIB.addImm(RHSC); }}};
}
+
+ // Large constant offset. Fold a -2048/2047 adjustment so the whole
+ // constant can be split across an ADDI and the load/store offset.
+ if (RHSC >= -4096 && RHSC <= 4094) {
+ int64_t Adj = RHSC < 0 ? -2048 : 2047;
+ return renderAddiPair(LHS.getReg(), Adj, RHSC - Adj);
+ }
+
+ if (isWorthFoldingAdd(Root.getReg()))
+ if (auto Fns = computeConstAddr(RHSC, /*IsPrefetch=*/false, LHS.getReg()))
+ return Fns;
+ }
+
+ // Bare constant address. IRTranslator lowers inttoptr(C) to
+ // G_INTTOPTR(G_CONSTANT); look through it to reach the constant.
+ if (RootDef->getOpcode() == TargetOpcode::G_INTTOPTR) {
+ MachineInstr *SrcDef = MRI->getVRegDef(RootDef->getOperand(1).getReg());
+ if (SrcDef && SrcDef->getOpcode() == TargetOpcode::G_CONSTANT)
+ RootDef = SrcDef;
+ }
+ if (RootDef->getOpcode() == TargetOpcode::G_CONSTANT) {
+ int64_t CVal = RootDef->getOperand(1).getCImm()->getSExtValue();
+ if (auto Fns = computeConstAddr(CVal, /*IsPrefetch=*/false, Register()))
+ return Fns;
}
- // TODO: Need to get the immediate from a G_PTR_ADD. Should this be done in
- // the combiner?
return {{[=](MachineInstrBuilder &MIB) { MIB.addReg(Root.getReg()); },
[=](MachineInstrBuilder &MIB) { MIB.addImm(0); }}};
}
@@ -650,19 +690,7 @@ RISCVInstructionSelector::selectAddrRegImmLsb00000(MachineOperand &Root) const {
// Large constant: fold a -2048/2016 adjustment to save an instruction.
if ((-2049 >= RHSC && RHSC >= -4096) || (4063 >= RHSC && RHSC >= 2017)) {
int64_t Adj = RHSC < 0 ? -2048 : 2016;
- int64_t AdjustedOffset = RHSC - Adj;
- Register BaseReg = LHS.getReg();
- return {{[=](MachineInstrBuilder &MIB) {
- Register Tmp = MRI->createVirtualRegister(&RISCV::GPRRegClass);
- MachineInstr *Addi =
- BuildMI(*MIB->getParent(), *MIB.getInstr(),
- MIB->getDebugLoc(), TII.get(RISCV::ADDI), Tmp)
- .addReg(BaseReg)
- .addImm(AdjustedOffset);
- constrainSelectedInstRegOperands(*Addi, TII, TRI, RBI);
- MIB.addReg(Tmp);
- },
- [=](MachineInstrBuilder &MIB) { MIB.addImm(Adj); }}};
+ return renderAddiPair(LHS.getReg(), RHSC - Adj, Adj);
}
// Otherwise split the constant into Hi (materialized + added to the base)
@@ -1687,6 +1715,29 @@ bool RISCVInstructionSelector::isRegInFprb(Register Reg) const {
return RBI.getRegBank(Reg, *MRI, TRI)->getID() == RISCV::FPRBRegBankID;
}
+// A G_PTR_ADD result is worth splitting into Hi (materialized) +
+// Lo12 (folded offset) only if every user is a plain scalar load/store
+// using it as the address. Otherwise the ADD is selected on its own with
+// the full materialized constant, making the Hi materialization here redundant.
+bool RISCVInstructionSelector::isWorthFoldingAdd(Register AddResult) const {
+ for (const MachineOperand &Use : MRI->use_operands(AddResult)) {
+ const MachineInstr *User = Use.getParent();
+ auto *LdSt = dyn_cast<GLoadStore>(User);
+ if (!LdSt)
+ return false;
+ // Must be used as the pointer, not the stored value.
+ if (LdSt->getPointerReg() != AddResult)
+ return false;
+ if (isStrongerThanMonotonic(LdSt->getMMO().getSuccessOrdering()))
+ return false;
+ // Only scalar integer/f16/f32/f64 memory (exclude vectors, f128, ...).
+ LLT Ty = MRI->getType(User->getOperand(0).getReg());
+ if (!Ty.isScalar() || Ty.getSizeInBits() > 64)
+ return false;
+ }
+ return true;
+}
+
bool RISCVInstructionSelector::selectCopy(MachineInstr &MI) const {
MachineOperand Dst = MI.getOperand(0);
Register DstReg = MI.getOperand(0).getReg();
diff --git a/llvm/test/CodeGen/RISCV/GlobalISel/load-store.ll b/llvm/test/CodeGen/RISCV/GlobalISel/load-store.ll
new file mode 100644
index 0000000000000..e44636adc4c4e
--- /dev/null
+++ b/llvm/test/CodeGen/RISCV/GlobalISel/load-store.ll
@@ -0,0 +1,641 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
+; RUN: llc -global-isel -global-isel-abort=1 -mtriple=riscv32 < %s | FileCheck %s --check-prefix=RV32I
+; RUN: llc -global-isel -global-isel-abort=1 -mtriple=riscv64 < %s | FileCheck %s --check-prefix=RV64I
+
+; ============================================================================
+; Basic simm12 addressing
+; ============================================================================
+
+define i32 @load_offset_positive(ptr %base) nounwind {
+; RV32I-LABEL: load_offset_positive:
+; RV32I: # %bb.0:
+; RV32I-NEXT: lw a0, 8(a0)
+; RV32I-NEXT: ret
+;
+; RV64I-LABEL: load_offset_positive:
+; RV64I: # %bb.0:
+; RV64I-NEXT: lw a0, 8(a0)
+; RV64I-NEXT: ret
+ %ptr = getelementptr i8, ptr %base, i32 8
+ %val = load i32, ptr %ptr
+ ret i32 %val
+}
+
+define i32 @load_offset_negative(ptr %base) nounwind {
+; RV32I-LABEL: load_offset_negative:
+; RV32I: # %bb.0:
+; RV32I-NEXT: lw a0, -8(a0)
+; RV32I-NEXT: ret
+;
+; RV64I-LABEL: load_offset_negative:
+; RV64I: # %bb.0:
+; RV64I-NEXT: lw a0, -8(a0)
+; RV64I-NEXT: ret
+ %ptr = getelementptr i8, ptr %base, i32 -8
+ %val = load i32, ptr %ptr
+ ret i32 %val
+}
+
+define i32 @load_offset_max(ptr %base) nounwind {
+; RV32I-LABEL: load_offset_max:
+; RV32I: # %bb.0:
+; RV32I-NEXT: lw a0, 2047(a0)
+; RV32I-NEXT: ret
+;
+; RV64I-LABEL: load_offset_max:
+; RV64I: # %bb.0:
+; RV64I-NEXT: lw a0, 2047(a0)
+; RV64I-NEXT: ret
+ %ptr = getelementptr i8, ptr %base, i32 2047
+ %val = load i32, ptr %ptr
+ ret i32 %val
+}
+
+define i32 @load_offset_min(ptr %base) nounwind {
+; RV32I-LABEL: load_offset_min:
+; RV32I: # %bb.0:
+; RV32I-NEXT: lw a0, -2048(a0)
+; RV32I-NEXT: ret
+;
+; RV64I-LABEL: load_offset_min:
+; RV64I: # %bb.0:
+; RV64I-NEXT: lw a0, -2048(a0)
+; RV64I-NEXT: ret
+ %ptr = getelementptr i8, ptr %base, i32 -2048
+ %val = load i32, ptr %ptr
+ ret i32 %val
+}
+
+define void @store_offset_positive(ptr %base, i32 %val) nounwind {
+; RV32I-LABEL: store_offset_positive:
+; RV32I: # %bb.0:
+; RV32I-NEXT: sw a1, 8(a0)
+; RV32I-NEXT: ret
+;
+; RV64I-LABEL: store_offset_positive:
+; RV64I: # %bb.0:
+; RV64I-NEXT: sw a1, 8(a0)
+; RV64I-NEXT: ret
+ %ptr = getelementptr i8, ptr %base, i32 8
+ store i32 %val, ptr %ptr
+ ret void
+}
+
+define void @store_offset_negative(ptr %base, i32 %val) nounwind {
+; RV32I-LABEL: store_offset_negative:
+; RV32I: # %bb.0:
+; RV32I-NEXT: sw a1, -8(a0)
+; RV32I-NEXT: ret
+;
+; RV64I-LABEL: store_offset_negative:
+; RV64I: # %bb.0:
+; RV64I-NEXT: sw a1, -8(a0)
+; RV64I-NEXT: ret
+ %ptr = getelementptr i8, ptr %base, i32 -8
+ store i32 %val, ptr %ptr
+ ret void
+}
+
+; ============================================================================
+; AddiPair: positive offsets in [-4096, 4094]
+; ============================================================================
+
+define i32 @load_addi_pair_positive(ptr %base) nounwind {
+; RV32I-LABEL: load_addi_pair_positive:
+; RV32I: # %bb.0:
+; RV32I-NEXT: addi a0, a0, 2047
+; RV32I-NEXT: lw a0, 953(a0)
+; RV32I-NEXT: ret
+;
+; RV64I-LABEL: load_addi_pair_positive:
+; RV64I: # %bb.0:
+; RV64I-NEXT: addi a0, a0, 2047
+; RV64I-NEXT: lw a0, 953(a0)
+; RV64I-NEXT: ret
+ %ptr = getelementptr i8, ptr %base, i32 3000
+ %val = load i32, ptr %ptr
+ ret i32 %val
+}
+
+define void @store_addi_pair_positive(ptr %base, i32 %val) nounwind {
+; RV32I-LABEL: store_addi_pair_positive:
+; RV32I: # %bb.0:
+; RV32I-NEXT: addi a0, a0, 2047
+; RV32I-NEXT: sw a1, 953(a0)
+; RV32I-NEXT: ret
+;
+; RV64I-LABEL: store_addi_pair_positive:
+; RV64I: # %bb.0:
+; RV64I-NEXT: addi a0, a0, 2047
+; RV64I-NEXT: sw a1, 953(a0)
+; RV64I-NEXT: ret
+ %ptr = getelementptr i8, ptr %base, i32 3000
+ store i32 %val, ptr %ptr
+ ret void
+}
+
+define i32 @load_addi_pair_positive_max(ptr %base) nounwind {
+; RV32I-LABEL: load_addi_pair_positive_max:
+; RV32I: # %bb.0:
+; RV32I-NEXT: addi a0, a0, 2047
+; RV32I-NEXT: lw a0, 2047(a0)
+; RV32I-NEXT: ret
+;
+; RV64I-LABEL: load_addi_pair_positive_max:
+; RV64I: # %bb.0:
+; RV64I-NEXT: addi a0, a0, 2047
+; RV64I-NEXT: lw a0, 2047(a0)
+; RV64I-NEXT: ret
+ %ptr = getelementptr i8, ptr %base, i32 4094
+ %val = load i32, ptr %ptr
+ ret i32 %val
+}
+
+define i32 @load_addi_pair_positive_min(ptr %base) nounwind {
+; RV32I-LABEL: load_addi_pair_positive_min:
+; RV32I: # %bb.0:
+; RV32I-NEXT: addi a0, a0, 2047
+; RV32I-NEXT: lw a0, 1(a0)
+; RV32I-NEXT: ret
+;
+; RV64I-LABEL: load_addi_pair_positive_min:
+; RV64I: # %bb.0:
+; RV64I-NEXT: addi a0, a0, 2047
+; RV64I-NEXT: lw a0, 1(a0)
+; RV64I-NEXT: ret
+ %ptr = getelementptr i8, ptr %base, i32 2048
+ %val = load i32, ptr %ptr
+ ret i32 %val
+}
+
+; ============================================================================
+; AddiPair: negative offsets in [-4096, -2049]
+; ============================================================================
+
+define i32 @load_addi_pair_negative(ptr %base) nounwind {
+; RV32I-LABEL: load_addi_pair_negative:
+; RV32I: # %bb.0:
+; RV32I-NEXT: addi a0, a0, -2048
+; RV32I-NEXT: lw a0, -952(a0)
+; RV32I-NEXT: ret
+;
+; RV64I-LABEL: load_addi_pair_negative:
+; RV64I: # %bb.0:
+; RV64I-NEXT: addi a0, a0, -2048
+; RV64I-NEXT: lw a0, -952(a0)
+; RV64I-NEXT: ret
+ %ptr = getelementptr i8, ptr %base, i32 -3000
+ %val = load i32, ptr %ptr
+ ret i32 %val
+}
+
+define void @store_addi_pair_negative(ptr %base, i32 %val) nounwind {
+; RV32I-LABEL: store_addi_pair_negative:
+; RV32I: # %bb.0:
+; RV32I-NEXT: addi a0, a0, -2048
+; RV32I-NEXT: sw a1, -952(a0)
+; RV32I-NEXT: ret
+;
+; RV64I-LABEL: store_addi_pair_negative:
+; RV64I: # %bb.0:
+; RV64I-NEXT: addi a0, a0, -2048
+; RV64I-NEXT: sw a1, -952(a0)
+; RV64I-NEXT: ret
+ %ptr = getelementptr i8, ptr %base, i32 -3000
+ store i32 %val, ptr %ptr
+ ret void
+}
+
+define i32 @load_addi_pair_negative_min(ptr %base) nounwind {
+; RV32I-LABEL: load_addi_pair_negative_min:
+; RV32I: # %bb.0:
+; RV32I-NEXT: addi a0, a0, -2048
+; RV32I-NEXT: lw a0, -2048(a0)
+; RV32I-NEXT: ret
+;
+; RV64I-LABEL: load_addi_pair_negative_min:
+; RV64I: # %bb.0:
+; RV64I-NEXT: addi a0, a0, -2048
+; RV64I-NEXT: lw a0, -2048(a0)
+; RV64I-NEXT: ret
+ %ptr = getelementptr i8, ptr %base, i32 -4096
+ %val = load i32, ptr %ptr
+ ret i32 %val
+}
+
+define i32 @load_addi_pair_negative_max(ptr %base) nounwind {
+; RV32I-LABEL: load_addi_pair_negative_max:
+; RV32I: # %bb.0:
+; RV32I-NEXT: addi a0, a0, -2048
+; RV32I-NEXT: lw a0, -1(a0)
+; RV32I-NEXT: ret
+;
+; RV64I-LABEL: load_addi_pair_negative_max:
+; RV64I: # %bb.0:
+; RV64I-NEXT: addi a0, a0, -2048
+; RV64I-NEXT: lw a0, -1(a0)
+; RV64I-NEXT: ret
+ %ptr = getelementptr i8, ptr %base, i32 -2049
+ %val = load i32, ptr %ptr
+ ret i32 %val
+}
+
+; ============================================================================
+; Just outside the AddiPair range.
+;
+; These should not enter the simple:
+;
+; [-4096, 4094]
+;
+; ADDI + load/store-offset path.
+; ============================================================================
+
+define i32 @load_offset_above_addi_pair_range(ptr %base) nounwind {
+; RV32I-LABEL: load_offset_above_addi_pair_range:
+; RV32I: # %bb.0:
+; RV32I-NEXT: lui a1, 1
+; RV32I-NEXT: add a0, a0, a1
+; RV32I-NEXT: lw a0, -1(a0)
+; RV32I-NEXT: ret
+;
+; RV64I-LABEL: load_offset_above_addi_pair_range:
+; RV64I: # %bb.0:
+; RV64I-NEXT: lui a1, 1
+; RV64I-NEXT: add a0, a0, a1
+; RV64I-NEXT: lw a0, -1(a0)
+; RV64I-NEXT: ret
+ %ptr = getelementptr i8, ptr %base, i32 4095
+ %val = load i32, ptr %ptr
+ ret i32 %val
+}
+
+define i32 @load_offset_below_addi_pair_range(ptr %base) nounwind {
+; RV32I-LABEL: load_offset_below_addi_pair_range:
+; RV32I: # %bb.0:
+; RV32I-NEXT: lui a1, 1048575
+; RV32I-NEXT: add a0, a0, a1
+; RV32I-NEXT: lw a0, -1(a0)
+; RV32I-NEXT: ret
+;
+; RV64I-LABEL: load_offset_below_addi_pair_range:
+; RV64I: # %bb.0:
+; RV64I-NEXT: lui a1, 1048575
+; RV64I-NEXT: add a0, a0, a1
+; RV64I-NEXT: lw a0, -1(a0)
+; RV64I-NEXT: ret
+ %ptr = getelementptr i8, ptr %base, i32 -4097
+ %val = load i32, ptr %ptr
+ ret i32 %val
+}
+
+; ============================================================================
+; Large constant address folding.
+;
+; The address is only used by the load/store, so the ADD can be folded and
+; computeConstAddr() may materialize the high part while folding the low
+; part into the memory instruction.
+; ============================================================================
+
+define i32 @load_large_offset(ptr %base) nounwind {
+; RV32I-LABEL: load_large_offset:
+; RV32I: # %bb.0:
+; RV32I-NEXT: lui a1, 24
+; RV32I-NEXT: add a0, a0, a1
+; RV32I-NEXT: lw a0, 1696(a0)
+; RV32I-NEXT: ret
+;
+; RV64I-LABEL: load_large_offset:
+; RV64I: # %bb.0:
+; RV64I-NEXT: lui a1, 24
+; RV64I-NEXT: add a0, a0, a1
+; RV64I-NEXT: lw a0, 1696(a0)
+; RV64I-NEXT: ret
+ %ptr = getelementptr i8, ptr %base, i32 100000
+ %val = load i32, ptr %ptr
+ ret i32 %val
+}
+
+define void @store_large_offset(ptr %base, i32 %val) nounwind {
+; RV32I-LABEL: store_large_offset:
+; RV32I: # %bb.0:
+; RV32I-NEXT: lui a2, 24
+; RV32I-NEXT: add a0, a0, a2
+; RV32I-NEXT: sw a1, 1696(a0)
+; RV32I-NEXT: ret
+;
+; RV64I-LABEL: store_large_offset:
+; RV64I: # %bb.0:
+; RV64I-NEXT: lui a2, 24
+; RV64I-NEXT: add a0, a0, a2
+; RV64I-NEXT: sw a1, 1696(a0)
+; RV64I-NEXT: ret
+ %ptr = getelementptr i8, ptr %base, i32 100000
+ store i32 %val, ptr %ptr
+ ret void
+}
+
+define i32 @load_large_negative_offset(ptr %base) nounwind {
+; RV32I-LABEL: load_large_negative_offset:
+; RV32I: # %bb.0:
+; RV32I-NEXT: lui a1, 1048552
+; RV32I-NEXT: add a0, a0, a1
+; RV32I-NEXT: lw a0, -1696(a0)
+; RV32I-NEXT: ret
+;
+; RV64I-LABEL: load_large_negative_offset:
+; RV64I: # %bb.0:
+; RV64I-NEXT: lui a1, 1048552
+; RV64I-NEXT: add a0, a0, a1
+; RV64I-NEXT: lw a0, -1696(a0)
+; RV64I-NEXT: ret
+ %ptr = getelementptr i8, ptr %base, i32 -100000
+ %val = load i32, ptr %ptr
+ ret i32 %val
+}
+
+; ============================================================================
+; Large constant with a non-memory user.
+;
+; The computed address must remain available for the call, so the ADD should
+; not be folded solely into the load.
+; ============================================================================
+
+declare void @use_pointer(ptr)
+
+define i32 @load_large_offset_multi_use(ptr %base) nounwind {
+; RV32I-LABEL: load_large_offset_multi_use:
+; RV32I: # %bb.0:
+; RV32I-NEXT: addi sp, sp, -16
+; RV32I-NEXT: sw ra, 12(sp) # 4-byte Folded Spill
+; RV32I-NEXT: sw s0, 8(sp) # 4-byte Folded Spill
+; RV32I-NEXT: lui a1, 24
+; RV32I-NEXT: addi a1, a1, 1696
+; RV32I-NEXT: add a0, a0, a1
+; RV32I-NEXT: lw s0, 0(a0)
+; RV32I-NEXT: call use_pointer
+; RV32I-NEXT: mv a0, s0
+; RV32I-NEXT: lw ra, 12(sp) # 4-byte Folded Reload
+; RV32I-NEXT: lw s0, 8(sp) # 4-byte Folded Reload
+; RV32I-NEXT: addi sp, sp, 16
+; RV32I-NEXT: ret
+;
+; RV64I-LABEL: load_large_offset_multi_use:
+; RV64I: # %bb.0:
+; RV64I-NEXT: addi sp, sp, -16
+; RV64I-NEXT: sd ra, 8(sp) # 8-byte Folded Spill
+; RV64I-NEXT: sd s0, 0(sp) # 8-byte Folded Spill
+; RV64I-NEXT: lui a1, 24
+; RV64I-NEXT: addi a1, a1, 1696
+; RV64I-NEXT: add a0, a0, a1
+; RV64I-NEXT: lw s0, 0(a0)
+; RV64I-NEXT: call use_pointer
+; RV64I-NEXT: mv a0, s0
+; RV64I-NEXT: ld ra, 8(sp) # 8-byte Folded Reload
+; RV64I-NEXT: ld s0, 0(sp) # 8-byte Folded Reload
+; RV64I-NEXT: addi sp, sp, 16
+; RV64I-NEXT: ret
+ %ptr = getelementptr i8, ptr %base, i32 100000
+ %val = load i32, ptr %ptr
+ call void @use_pointer(ptr %ptr)
+ ret i32 %val
+}
+
+define void @store_large_offset_multi_use(ptr %base, i32 %val) nounwind {
+; RV32I-LABEL: store_large_offset_multi_use:
+; RV32I: # %bb.0:
+; RV32I-NEXT: addi sp, sp, -16
+; RV32I-NEXT: sw ra, 12(sp) # 4-byte Folded Spill
+; RV32I-NEXT: lui a2, 24
+; RV32I-NEXT: addi a2, a2, 1696
+; RV32I-NEXT: add a0, a0, a2
+; RV32I-NEXT: sw a1, 0(a0)
+; RV32I-NEXT: call use_pointer
+; RV32I-NEXT: lw ra, 12(sp) # 4-byte Folded Reload
+; RV32I-NEXT: addi sp, sp, 16
+; RV32I-NEXT: ret
+;
+; RV64I-LABEL: store_large_offset_multi_use:
+; RV64I: # %bb.0:
+; RV64I-NEXT: addi sp, sp, -16
+; RV64I-NEXT: sd ra, 8(sp) # 8-byte Folded Spill
+; RV64I-NEXT: lui a2, 24
+; RV64I-NEXT: addi a2, a2, 1696
+; RV64I-NEXT: add a0, a0, a2
+; RV64I-NEXT: sw a1, 0(a0)
+; RV64I-NEXT: call use_pointer
+; RV64I-NEXT: ld ra, 8(sp) # 8-byte Folded Reload
+; RV64I-NEXT: addi sp, sp, 16
+; RV64I-NEXT: ret
+ %ptr = getelementptr i8, ptr %base, i32 100000
+ store i32 %val, ptr %ptr
+ call void @use_pointer(ptr %ptr)
+ ret void
+}
+
+; ============================================================================
+; The address has multiple load/store users.
+;
+; Folding the address into both memory operations should still be possible
+; when the address itself has no non-memory users.
+; ============================================================================
+
+define i32 @load_large_offset_two_loads(ptr %base) nounwind {
+; RV32I-LABEL: load_large_offset_two_loads:
+; RV32I: # %bb.0:
+; RV32I-NEXT: lui a1, 24
+; RV32I-NEXT: add a0, a0, a1
+; RV32I-NEXT: lw a1, 1696(a0)
+; RV32I-NEXT: lw a0, 1696(a0)
+; RV32I-NEXT: add a0, a1, a0
+; RV32I-NEXT: ret
+;
+; RV64I-LABEL: load_large_offset_two_loads:
+; RV64I: # %bb.0:
+; RV64I-NEXT: lui a1, 24
+; RV64I-NEXT: add a0, a0, a1
+;...
[truncated]
``````````
</details>
https://github.com/llvm/llvm-project/pull/219161
More information about the llvm-commits
mailing list