[llvm] [RISCV][GlobalISel] Fold large constant offsets in selectAddrRegImm (PR #219161)

via llvm-commits llvm-commits at lists.llvm.org
Thu Aug 27 02:52:41 PDT 2026


llvmorg-github-actions[bot] wrote:


<!--LLVM PR SUMMARY COMMENT-->

@llvm/pr-subscribers-backend-risc-v

Author: Kane Wang (ReVe1uv)

<details>
<summary>Changes</summary>

Fold ADDI adjustment (AddiPair) for offsets in [-4096, 4094] and split larger constants into materialized Hi + Lo12 offset, matching SDAG. Add isWorthFoldingAdd to guard the split and extract the shared ADDI renderer into renderAddiPair.

Assisted-by: Claude

---

Patch is 25.57 KiB, truncated to 20.00 KiB below, full version: https://github.com/llvm/llvm-project/pull/219161.diff


2 Files Affected:

- (modified) llvm/lib/Target/RISCV/GISel/RISCVInstructionSelector.cpp (+67-16) 
- (added) llvm/test/CodeGen/RISCV/GlobalISel/load-store.ll (+641) 


``````````diff
diff --git a/llvm/lib/Target/RISCV/GISel/RISCVInstructionSelector.cpp b/llvm/lib/Target/RISCV/GISel/RISCVInstructionSelector.cpp
index bf31c9a9d1e9e..fdd6814020211 100644
--- a/llvm/lib/Target/RISCV/GISel/RISCVInstructionSelector.cpp
+++ b/llvm/lib/Target/RISCV/GISel/RISCVInstructionSelector.cpp
@@ -67,6 +67,7 @@ class RISCVInstructionSelector : public InstructionSelector {
 
   bool isRegInGprb(Register Reg) const;
   bool isRegInFprb(Register Reg) const;
+  bool isWorthFoldingAdd(Register AddResult) const;
 
   // tblgen-erated 'select' implementation, used as the initial selector for
   // the patterns that don't require complex C++.
@@ -155,7 +156,8 @@ class RISCVInstructionSelector : public InstructionSelector {
   }
 
   ComplexRendererFns renderVLOp(MachineOperand &Root) const;
-
+  ComplexRendererFns renderAddiPair(Register BaseReg, int64_t AddiImm,
+                                   int64_t OffsetImm) const;
   // Custom renderers for tablegen
   void renderNegImm(MachineInstrBuilder &MIB, const MachineInstr &MI,
                     int OpIdx) const;
@@ -573,6 +575,22 @@ RISCVInstructionSelector::renderVLOp(MachineOperand &Root) const {
   return {{[=](MachineInstrBuilder &MIB) { MIB.addReg(Root.getReg()); }}};
 }
 
+InstructionSelector::ComplexRendererFns
+RISCVInstructionSelector::renderAddiPair(Register BaseReg, int64_t AddiImm,
+                                         int64_t OffsetImm) const {
+  return {{[=](MachineInstrBuilder &MIB) {
+             Register Tmp = MRI->createVirtualRegister(&RISCV::GPRRegClass);
+             MachineInstr *Addi =
+                 BuildMI(*MIB->getParent(), *MIB.getInstr(), MIB->getDebugLoc(),
+                         TII.get(RISCV::ADDI), Tmp)
+                     .addReg(BaseReg)
+                     .addImm(AddiImm);
+             constrainSelectedInstRegOperands(*Addi, TII, TRI, RBI);
+             MIB.addReg(Tmp);
+           },
+           [=](MachineInstrBuilder &MIB) { MIB.addImm(OffsetImm); }}};
+}
+
 InstructionSelector::ComplexRendererFns
 RISCVInstructionSelector::selectAddrRegImm(MachineOperand &Root) const {
   if (!Root.isReg())
@@ -603,10 +621,32 @@ RISCVInstructionSelector::selectAddrRegImm(MachineOperand &Root) const {
       return {{[=](MachineInstrBuilder &MIB) { MIB.add(LHS); },
                [=](MachineInstrBuilder &MIB) { MIB.addImm(RHSC); }}};
     }
+
+    // Large constant offset. Fold a -2048/2047 adjustment so the whole
+    // constant can be split across an ADDI and the load/store offset. 
+    if (RHSC >= -4096 && RHSC <= 4094) {
+      int64_t Adj = RHSC < 0 ? -2048 : 2047;
+      return renderAddiPair(LHS.getReg(), Adj, RHSC - Adj);
+    }
+
+    if (isWorthFoldingAdd(Root.getReg()))
+      if (auto Fns = computeConstAddr(RHSC, /*IsPrefetch=*/false, LHS.getReg()))
+        return Fns;
+  }
+
+  // Bare constant address. IRTranslator lowers inttoptr(C) to
+  // G_INTTOPTR(G_CONSTANT); look through it to reach the constant.
+  if (RootDef->getOpcode() == TargetOpcode::G_INTTOPTR) {
+    MachineInstr *SrcDef = MRI->getVRegDef(RootDef->getOperand(1).getReg());
+    if (SrcDef && SrcDef->getOpcode() == TargetOpcode::G_CONSTANT)
+      RootDef = SrcDef;
+  }
+  if (RootDef->getOpcode() == TargetOpcode::G_CONSTANT) {
+    int64_t CVal = RootDef->getOperand(1).getCImm()->getSExtValue();
+    if (auto Fns = computeConstAddr(CVal, /*IsPrefetch=*/false, Register()))
+      return Fns;
   }
 
-  // TODO: Need to get the immediate from a G_PTR_ADD. Should this be done in
-  // the combiner?
   return {{[=](MachineInstrBuilder &MIB) { MIB.addReg(Root.getReg()); },
            [=](MachineInstrBuilder &MIB) { MIB.addImm(0); }}};
 }
@@ -650,19 +690,7 @@ RISCVInstructionSelector::selectAddrRegImmLsb00000(MachineOperand &Root) const {
     // Large constant: fold a -2048/2016 adjustment to save an instruction.
     if ((-2049 >= RHSC && RHSC >= -4096) || (4063 >= RHSC && RHSC >= 2017)) {
       int64_t Adj = RHSC < 0 ? -2048 : 2016;
-      int64_t AdjustedOffset = RHSC - Adj;
-      Register BaseReg = LHS.getReg();
-      return {{[=](MachineInstrBuilder &MIB) {
-                 Register Tmp = MRI->createVirtualRegister(&RISCV::GPRRegClass);
-                 MachineInstr *Addi =
-                     BuildMI(*MIB->getParent(), *MIB.getInstr(),
-                             MIB->getDebugLoc(), TII.get(RISCV::ADDI), Tmp)
-                         .addReg(BaseReg)
-                         .addImm(AdjustedOffset);
-                 constrainSelectedInstRegOperands(*Addi, TII, TRI, RBI);
-                 MIB.addReg(Tmp);
-               },
-               [=](MachineInstrBuilder &MIB) { MIB.addImm(Adj); }}};
+      return renderAddiPair(LHS.getReg(), RHSC - Adj, Adj);
     }
 
     // Otherwise split the constant into Hi (materialized + added to the base)
@@ -1687,6 +1715,29 @@ bool RISCVInstructionSelector::isRegInFprb(Register Reg) const {
   return RBI.getRegBank(Reg, *MRI, TRI)->getID() == RISCV::FPRBRegBankID;
 }
 
+// A G_PTR_ADD result is worth splitting into Hi (materialized) +
+// Lo12 (folded offset) only if every user is a plain scalar load/store
+// using it as the address. Otherwise the ADD is selected on its own with
+// the full materialized constant, making the Hi materialization here redundant.
+bool RISCVInstructionSelector::isWorthFoldingAdd(Register AddResult) const {
+  for (const MachineOperand &Use : MRI->use_operands(AddResult)) {
+    const MachineInstr *User = Use.getParent();
+    auto *LdSt = dyn_cast<GLoadStore>(User);
+    if (!LdSt)
+      return false;
+    // Must be used as the pointer, not the stored value.
+    if (LdSt->getPointerReg() != AddResult)
+      return false;
+    if (isStrongerThanMonotonic(LdSt->getMMO().getSuccessOrdering()))
+      return false;
+    // Only scalar integer/f16/f32/f64 memory (exclude vectors, f128, ...).
+    LLT Ty = MRI->getType(User->getOperand(0).getReg());
+    if (!Ty.isScalar() || Ty.getSizeInBits() > 64)
+      return false;
+  }
+  return true;
+}
+
 bool RISCVInstructionSelector::selectCopy(MachineInstr &MI) const {
   MachineOperand Dst = MI.getOperand(0);
   Register DstReg = MI.getOperand(0).getReg();
diff --git a/llvm/test/CodeGen/RISCV/GlobalISel/load-store.ll b/llvm/test/CodeGen/RISCV/GlobalISel/load-store.ll
new file mode 100644
index 0000000000000..e44636adc4c4e
--- /dev/null
+++ b/llvm/test/CodeGen/RISCV/GlobalISel/load-store.ll
@@ -0,0 +1,641 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
+; RUN: llc -global-isel -global-isel-abort=1 -mtriple=riscv32 < %s | FileCheck %s --check-prefix=RV32I
+; RUN: llc -global-isel -global-isel-abort=1 -mtriple=riscv64 < %s | FileCheck %s --check-prefix=RV64I
+
+; ============================================================================
+; Basic simm12 addressing
+; ============================================================================
+
+define i32 @load_offset_positive(ptr %base) nounwind {
+; RV32I-LABEL: load_offset_positive:
+; RV32I:       # %bb.0:
+; RV32I-NEXT:    lw a0, 8(a0)
+; RV32I-NEXT:    ret
+;
+; RV64I-LABEL: load_offset_positive:
+; RV64I:       # %bb.0:
+; RV64I-NEXT:    lw a0, 8(a0)
+; RV64I-NEXT:    ret
+  %ptr = getelementptr i8, ptr %base, i32 8
+  %val = load i32, ptr %ptr
+  ret i32 %val
+}
+
+define i32 @load_offset_negative(ptr %base) nounwind {
+; RV32I-LABEL: load_offset_negative:
+; RV32I:       # %bb.0:
+; RV32I-NEXT:    lw a0, -8(a0)
+; RV32I-NEXT:    ret
+;
+; RV64I-LABEL: load_offset_negative:
+; RV64I:       # %bb.0:
+; RV64I-NEXT:    lw a0, -8(a0)
+; RV64I-NEXT:    ret
+  %ptr = getelementptr i8, ptr %base, i32 -8
+  %val = load i32, ptr %ptr
+  ret i32 %val
+}
+
+define i32 @load_offset_max(ptr %base) nounwind {
+; RV32I-LABEL: load_offset_max:
+; RV32I:       # %bb.0:
+; RV32I-NEXT:    lw a0, 2047(a0)
+; RV32I-NEXT:    ret
+;
+; RV64I-LABEL: load_offset_max:
+; RV64I:       # %bb.0:
+; RV64I-NEXT:    lw a0, 2047(a0)
+; RV64I-NEXT:    ret
+  %ptr = getelementptr i8, ptr %base, i32 2047
+  %val = load i32, ptr %ptr
+  ret i32 %val
+}
+
+define i32 @load_offset_min(ptr %base) nounwind {
+; RV32I-LABEL: load_offset_min:
+; RV32I:       # %bb.0:
+; RV32I-NEXT:    lw a0, -2048(a0)
+; RV32I-NEXT:    ret
+;
+; RV64I-LABEL: load_offset_min:
+; RV64I:       # %bb.0:
+; RV64I-NEXT:    lw a0, -2048(a0)
+; RV64I-NEXT:    ret
+  %ptr = getelementptr i8, ptr %base, i32 -2048
+  %val = load i32, ptr %ptr
+  ret i32 %val
+}
+
+define void @store_offset_positive(ptr %base, i32 %val) nounwind {
+; RV32I-LABEL: store_offset_positive:
+; RV32I:       # %bb.0:
+; RV32I-NEXT:    sw a1, 8(a0)
+; RV32I-NEXT:    ret
+;
+; RV64I-LABEL: store_offset_positive:
+; RV64I:       # %bb.0:
+; RV64I-NEXT:    sw a1, 8(a0)
+; RV64I-NEXT:    ret
+  %ptr = getelementptr i8, ptr %base, i32 8
+  store i32 %val, ptr %ptr
+  ret void
+}
+
+define void @store_offset_negative(ptr %base, i32 %val) nounwind {
+; RV32I-LABEL: store_offset_negative:
+; RV32I:       # %bb.0:
+; RV32I-NEXT:    sw a1, -8(a0)
+; RV32I-NEXT:    ret
+;
+; RV64I-LABEL: store_offset_negative:
+; RV64I:       # %bb.0:
+; RV64I-NEXT:    sw a1, -8(a0)
+; RV64I-NEXT:    ret
+  %ptr = getelementptr i8, ptr %base, i32 -8
+  store i32 %val, ptr %ptr
+  ret void
+}
+
+; ============================================================================
+; AddiPair: positive offsets in [-4096, 4094]
+; ============================================================================
+
+define i32 @load_addi_pair_positive(ptr %base) nounwind {
+; RV32I-LABEL: load_addi_pair_positive:
+; RV32I:       # %bb.0:
+; RV32I-NEXT:    addi a0, a0, 2047
+; RV32I-NEXT:    lw a0, 953(a0)
+; RV32I-NEXT:    ret
+;
+; RV64I-LABEL: load_addi_pair_positive:
+; RV64I:       # %bb.0:
+; RV64I-NEXT:    addi a0, a0, 2047
+; RV64I-NEXT:    lw a0, 953(a0)
+; RV64I-NEXT:    ret
+  %ptr = getelementptr i8, ptr %base, i32 3000
+  %val = load i32, ptr %ptr
+  ret i32 %val
+}
+
+define void @store_addi_pair_positive(ptr %base, i32 %val) nounwind {
+; RV32I-LABEL: store_addi_pair_positive:
+; RV32I:       # %bb.0:
+; RV32I-NEXT:    addi a0, a0, 2047
+; RV32I-NEXT:    sw a1, 953(a0)
+; RV32I-NEXT:    ret
+;
+; RV64I-LABEL: store_addi_pair_positive:
+; RV64I:       # %bb.0:
+; RV64I-NEXT:    addi a0, a0, 2047
+; RV64I-NEXT:    sw a1, 953(a0)
+; RV64I-NEXT:    ret
+  %ptr = getelementptr i8, ptr %base, i32 3000
+  store i32 %val, ptr %ptr
+  ret void
+}
+
+define i32 @load_addi_pair_positive_max(ptr %base) nounwind {
+; RV32I-LABEL: load_addi_pair_positive_max:
+; RV32I:       # %bb.0:
+; RV32I-NEXT:    addi a0, a0, 2047
+; RV32I-NEXT:    lw a0, 2047(a0)
+; RV32I-NEXT:    ret
+;
+; RV64I-LABEL: load_addi_pair_positive_max:
+; RV64I:       # %bb.0:
+; RV64I-NEXT:    addi a0, a0, 2047
+; RV64I-NEXT:    lw a0, 2047(a0)
+; RV64I-NEXT:    ret
+  %ptr = getelementptr i8, ptr %base, i32 4094
+  %val = load i32, ptr %ptr
+  ret i32 %val
+}
+
+define i32 @load_addi_pair_positive_min(ptr %base) nounwind {
+; RV32I-LABEL: load_addi_pair_positive_min:
+; RV32I:       # %bb.0:
+; RV32I-NEXT:    addi a0, a0, 2047
+; RV32I-NEXT:    lw a0, 1(a0)
+; RV32I-NEXT:    ret
+;
+; RV64I-LABEL: load_addi_pair_positive_min:
+; RV64I:       # %bb.0:
+; RV64I-NEXT:    addi a0, a0, 2047
+; RV64I-NEXT:    lw a0, 1(a0)
+; RV64I-NEXT:    ret
+  %ptr = getelementptr i8, ptr %base, i32 2048
+  %val = load i32, ptr %ptr
+  ret i32 %val
+}
+
+; ============================================================================
+; AddiPair: negative offsets in [-4096, -2049]
+; ============================================================================
+
+define i32 @load_addi_pair_negative(ptr %base) nounwind {
+; RV32I-LABEL: load_addi_pair_negative:
+; RV32I:       # %bb.0:
+; RV32I-NEXT:    addi a0, a0, -2048
+; RV32I-NEXT:    lw a0, -952(a0)
+; RV32I-NEXT:    ret
+;
+; RV64I-LABEL: load_addi_pair_negative:
+; RV64I:       # %bb.0:
+; RV64I-NEXT:    addi a0, a0, -2048
+; RV64I-NEXT:    lw a0, -952(a0)
+; RV64I-NEXT:    ret
+  %ptr = getelementptr i8, ptr %base, i32 -3000
+  %val = load i32, ptr %ptr
+  ret i32 %val
+}
+
+define void @store_addi_pair_negative(ptr %base, i32 %val) nounwind {
+; RV32I-LABEL: store_addi_pair_negative:
+; RV32I:       # %bb.0:
+; RV32I-NEXT:    addi a0, a0, -2048
+; RV32I-NEXT:    sw a1, -952(a0)
+; RV32I-NEXT:    ret
+;
+; RV64I-LABEL: store_addi_pair_negative:
+; RV64I:       # %bb.0:
+; RV64I-NEXT:    addi a0, a0, -2048
+; RV64I-NEXT:    sw a1, -952(a0)
+; RV64I-NEXT:    ret
+  %ptr = getelementptr i8, ptr %base, i32 -3000
+  store i32 %val, ptr %ptr
+  ret void
+}
+
+define i32 @load_addi_pair_negative_min(ptr %base) nounwind {
+; RV32I-LABEL: load_addi_pair_negative_min:
+; RV32I:       # %bb.0:
+; RV32I-NEXT:    addi a0, a0, -2048
+; RV32I-NEXT:    lw a0, -2048(a0)
+; RV32I-NEXT:    ret
+;
+; RV64I-LABEL: load_addi_pair_negative_min:
+; RV64I:       # %bb.0:
+; RV64I-NEXT:    addi a0, a0, -2048
+; RV64I-NEXT:    lw a0, -2048(a0)
+; RV64I-NEXT:    ret
+  %ptr = getelementptr i8, ptr %base, i32 -4096
+  %val = load i32, ptr %ptr
+  ret i32 %val
+}
+
+define i32 @load_addi_pair_negative_max(ptr %base) nounwind {
+; RV32I-LABEL: load_addi_pair_negative_max:
+; RV32I:       # %bb.0:
+; RV32I-NEXT:    addi a0, a0, -2048
+; RV32I-NEXT:    lw a0, -1(a0)
+; RV32I-NEXT:    ret
+;
+; RV64I-LABEL: load_addi_pair_negative_max:
+; RV64I:       # %bb.0:
+; RV64I-NEXT:    addi a0, a0, -2048
+; RV64I-NEXT:    lw a0, -1(a0)
+; RV64I-NEXT:    ret
+  %ptr = getelementptr i8, ptr %base, i32 -2049
+  %val = load i32, ptr %ptr
+  ret i32 %val
+}
+
+; ============================================================================
+; Just outside the AddiPair range.
+;
+; These should not enter the simple:
+;
+;   [-4096, 4094]
+;
+; ADDI + load/store-offset path.
+; ============================================================================
+
+define i32 @load_offset_above_addi_pair_range(ptr %base) nounwind {
+; RV32I-LABEL: load_offset_above_addi_pair_range:
+; RV32I:       # %bb.0:
+; RV32I-NEXT:    lui a1, 1
+; RV32I-NEXT:    add a0, a0, a1
+; RV32I-NEXT:    lw a0, -1(a0)
+; RV32I-NEXT:    ret
+;
+; RV64I-LABEL: load_offset_above_addi_pair_range:
+; RV64I:       # %bb.0:
+; RV64I-NEXT:    lui a1, 1
+; RV64I-NEXT:    add a0, a0, a1
+; RV64I-NEXT:    lw a0, -1(a0)
+; RV64I-NEXT:    ret
+  %ptr = getelementptr i8, ptr %base, i32 4095
+  %val = load i32, ptr %ptr
+  ret i32 %val
+}
+
+define i32 @load_offset_below_addi_pair_range(ptr %base) nounwind {
+; RV32I-LABEL: load_offset_below_addi_pair_range:
+; RV32I:       # %bb.0:
+; RV32I-NEXT:    lui a1, 1048575
+; RV32I-NEXT:    add a0, a0, a1
+; RV32I-NEXT:    lw a0, -1(a0)
+; RV32I-NEXT:    ret
+;
+; RV64I-LABEL: load_offset_below_addi_pair_range:
+; RV64I:       # %bb.0:
+; RV64I-NEXT:    lui a1, 1048575
+; RV64I-NEXT:    add a0, a0, a1
+; RV64I-NEXT:    lw a0, -1(a0)
+; RV64I-NEXT:    ret
+  %ptr = getelementptr i8, ptr %base, i32 -4097
+  %val = load i32, ptr %ptr
+  ret i32 %val
+}
+
+; ============================================================================
+; Large constant address folding.
+;
+; The address is only used by the load/store, so the ADD can be folded and
+; computeConstAddr() may materialize the high part while folding the low
+; part into the memory instruction.
+; ============================================================================
+
+define i32 @load_large_offset(ptr %base) nounwind {
+; RV32I-LABEL: load_large_offset:
+; RV32I:       # %bb.0:
+; RV32I-NEXT:    lui a1, 24
+; RV32I-NEXT:    add a0, a0, a1
+; RV32I-NEXT:    lw a0, 1696(a0)
+; RV32I-NEXT:    ret
+;
+; RV64I-LABEL: load_large_offset:
+; RV64I:       # %bb.0:
+; RV64I-NEXT:    lui a1, 24
+; RV64I-NEXT:    add a0, a0, a1
+; RV64I-NEXT:    lw a0, 1696(a0)
+; RV64I-NEXT:    ret
+  %ptr = getelementptr i8, ptr %base, i32 100000
+  %val = load i32, ptr %ptr
+  ret i32 %val
+}
+
+define void @store_large_offset(ptr %base, i32 %val) nounwind {
+; RV32I-LABEL: store_large_offset:
+; RV32I:       # %bb.0:
+; RV32I-NEXT:    lui a2, 24
+; RV32I-NEXT:    add a0, a0, a2
+; RV32I-NEXT:    sw a1, 1696(a0)
+; RV32I-NEXT:    ret
+;
+; RV64I-LABEL: store_large_offset:
+; RV64I:       # %bb.0:
+; RV64I-NEXT:    lui a2, 24
+; RV64I-NEXT:    add a0, a0, a2
+; RV64I-NEXT:    sw a1, 1696(a0)
+; RV64I-NEXT:    ret
+  %ptr = getelementptr i8, ptr %base, i32 100000
+  store i32 %val, ptr %ptr
+  ret void
+}
+
+define i32 @load_large_negative_offset(ptr %base) nounwind {
+; RV32I-LABEL: load_large_negative_offset:
+; RV32I:       # %bb.0:
+; RV32I-NEXT:    lui a1, 1048552
+; RV32I-NEXT:    add a0, a0, a1
+; RV32I-NEXT:    lw a0, -1696(a0)
+; RV32I-NEXT:    ret
+;
+; RV64I-LABEL: load_large_negative_offset:
+; RV64I:       # %bb.0:
+; RV64I-NEXT:    lui a1, 1048552
+; RV64I-NEXT:    add a0, a0, a1
+; RV64I-NEXT:    lw a0, -1696(a0)
+; RV64I-NEXT:    ret
+  %ptr = getelementptr i8, ptr %base, i32 -100000
+  %val = load i32, ptr %ptr
+  ret i32 %val
+}
+
+; ============================================================================
+; Large constant with a non-memory user.
+;
+; The computed address must remain available for the call, so the ADD should
+; not be folded solely into the load.
+; ============================================================================
+
+declare void @use_pointer(ptr)
+
+define i32 @load_large_offset_multi_use(ptr %base) nounwind {
+; RV32I-LABEL: load_large_offset_multi_use:
+; RV32I:       # %bb.0:
+; RV32I-NEXT:    addi sp, sp, -16
+; RV32I-NEXT:    sw ra, 12(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s0, 8(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    lui a1, 24
+; RV32I-NEXT:    addi a1, a1, 1696
+; RV32I-NEXT:    add a0, a0, a1
+; RV32I-NEXT:    lw s0, 0(a0)
+; RV32I-NEXT:    call use_pointer
+; RV32I-NEXT:    mv a0, s0
+; RV32I-NEXT:    lw ra, 12(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s0, 8(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    addi sp, sp, 16
+; RV32I-NEXT:    ret
+;
+; RV64I-LABEL: load_large_offset_multi_use:
+; RV64I:       # %bb.0:
+; RV64I-NEXT:    addi sp, sp, -16
+; RV64I-NEXT:    sd ra, 8(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    sd s0, 0(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    lui a1, 24
+; RV64I-NEXT:    addi a1, a1, 1696
+; RV64I-NEXT:    add a0, a0, a1
+; RV64I-NEXT:    lw s0, 0(a0)
+; RV64I-NEXT:    call use_pointer
+; RV64I-NEXT:    mv a0, s0
+; RV64I-NEXT:    ld ra, 8(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s0, 0(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    addi sp, sp, 16
+; RV64I-NEXT:    ret
+  %ptr = getelementptr i8, ptr %base, i32 100000
+  %val = load i32, ptr %ptr
+  call void @use_pointer(ptr %ptr)
+  ret i32 %val
+}
+
+define void @store_large_offset_multi_use(ptr %base, i32 %val) nounwind {
+; RV32I-LABEL: store_large_offset_multi_use:
+; RV32I:       # %bb.0:
+; RV32I-NEXT:    addi sp, sp, -16
+; RV32I-NEXT:    sw ra, 12(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    lui a2, 24
+; RV32I-NEXT:    addi a2, a2, 1696
+; RV32I-NEXT:    add a0, a0, a2
+; RV32I-NEXT:    sw a1, 0(a0)
+; RV32I-NEXT:    call use_pointer
+; RV32I-NEXT:    lw ra, 12(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    addi sp, sp, 16
+; RV32I-NEXT:    ret
+;
+; RV64I-LABEL: store_large_offset_multi_use:
+; RV64I:       # %bb.0:
+; RV64I-NEXT:    addi sp, sp, -16
+; RV64I-NEXT:    sd ra, 8(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    lui a2, 24
+; RV64I-NEXT:    addi a2, a2, 1696
+; RV64I-NEXT:    add a0, a0, a2
+; RV64I-NEXT:    sw a1, 0(a0)
+; RV64I-NEXT:    call use_pointer
+; RV64I-NEXT:    ld ra, 8(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    addi sp, sp, 16
+; RV64I-NEXT:    ret
+  %ptr = getelementptr i8, ptr %base, i32 100000
+  store i32 %val, ptr %ptr
+  call void @use_pointer(ptr %ptr)
+  ret void
+}
+
+; ============================================================================
+; The address has multiple load/store users.
+;
+; Folding the address into both memory operations should still be possible
+; when the address itself has no non-memory users.
+; ============================================================================
+
+define i32 @load_large_offset_two_loads(ptr %base) nounwind {
+; RV32I-LABEL: load_large_offset_two_loads:
+; RV32I:       # %bb.0:
+; RV32I-NEXT:    lui a1, 24
+; RV32I-NEXT:    add a0, a0, a1
+; RV32I-NEXT:    lw a1, 1696(a0)
+; RV32I-NEXT:    lw a0, 1696(a0)
+; RV32I-NEXT:    add a0, a1, a0
+; RV32I-NEXT:    ret
+;
+; RV64I-LABEL: load_large_offset_two_loads:
+; RV64I:       # %bb.0:
+; RV64I-NEXT:    lui a1, 24
+; RV64I-NEXT:    add a0, a0, a1
+;...
[truncated]

``````````

</details>


https://github.com/llvm/llvm-project/pull/219161


More information about the llvm-commits mailing list