[llvm] [RISCV][GlobalISel] Fold large constant offsets in selectAddrRegImm (PR #219161)

Kane Wang via llvm-commits llvm-commits at lists.llvm.org
Tue Sep 1 00:31:01 PDT 2026


https://github.com/ReVe1uv updated https://github.com/llvm/llvm-project/pull/219161

>From 7516d44abcc5b99ee98cd784260d3c0afec9d26b Mon Sep 17 00:00:00 2001
From: Kane Wang <wangqiang1 at kylinos.cn>
Date: Thu, 27 Aug 2026 17:49:36 +0800
Subject: [PATCH 1/3] [RISCV][GlobalISel] Fold large constant offsets in
 selectAddrRegImm

Fold ADDI adjustment (AddiPair) for offsets in [-4096, 4094] and split
larger constants into materialized Hi + Lo12 offset, matching SDAG.
Add isWorthFoldingAdd to guard the split and extract the shared ADDI
renderer into renderAddiPair.

Assisted-by: Claude
---
 .../RISCV/GISel/RISCVInstructionSelector.cpp  |  83 ++-
 .../CodeGen/RISCV/GlobalISel/load-store.ll    | 641 ++++++++++++++++++
 2 files changed, 708 insertions(+), 16 deletions(-)
 create mode 100644 llvm/test/CodeGen/RISCV/GlobalISel/load-store.ll

diff --git a/llvm/lib/Target/RISCV/GISel/RISCVInstructionSelector.cpp b/llvm/lib/Target/RISCV/GISel/RISCVInstructionSelector.cpp
index 8af102776b3d3..d3d98d92aac44 100644
--- a/llvm/lib/Target/RISCV/GISel/RISCVInstructionSelector.cpp
+++ b/llvm/lib/Target/RISCV/GISel/RISCVInstructionSelector.cpp
@@ -67,6 +67,7 @@ class RISCVInstructionSelector : public InstructionSelector {
 
   bool isRegInGprb(Register Reg) const;
   bool isRegInFprb(Register Reg) const;
+  bool isWorthFoldingAdd(Register AddResult) const;
 
   // tblgen-erated 'select' implementation, used as the initial selector for
   // the patterns that don't require complex C++.
@@ -155,7 +156,8 @@ class RISCVInstructionSelector : public InstructionSelector {
   }
 
   ComplexRendererFns renderVLOp(MachineOperand &Root) const;
-
+  ComplexRendererFns renderAddiPair(Register BaseReg, int64_t AddiImm,
+                                    int64_t OffsetImm) const;
   // Custom renderers for tablegen
   void renderNegImm(MachineInstrBuilder &MIB, const MachineInstr &MI,
                     int OpIdx) const;
@@ -573,6 +575,22 @@ RISCVInstructionSelector::renderVLOp(MachineOperand &Root) const {
   return {{[=](MachineInstrBuilder &MIB) { MIB.addReg(Root.getReg()); }}};
 }
 
+InstructionSelector::ComplexRendererFns
+RISCVInstructionSelector::renderAddiPair(Register BaseReg, int64_t AddiImm,
+                                         int64_t OffsetImm) const {
+  return {{[=](MachineInstrBuilder &MIB) {
+             Register Tmp = MRI->createVirtualRegister(&RISCV::GPRRegClass);
+             MachineInstr *Addi =
+                 BuildMI(*MIB->getParent(), *MIB.getInstr(), MIB->getDebugLoc(),
+                         TII.get(RISCV::ADDI), Tmp)
+                     .addReg(BaseReg)
+                     .addImm(AddiImm);
+             constrainSelectedInstRegOperands(*Addi, TII, TRI, RBI);
+             MIB.addReg(Tmp);
+           },
+           [=](MachineInstrBuilder &MIB) { MIB.addImm(OffsetImm); }}};
+}
+
 InstructionSelector::ComplexRendererFns
 RISCVInstructionSelector::selectAddrRegImm(MachineOperand &Root) const {
   if (!Root.isReg())
@@ -603,10 +621,32 @@ RISCVInstructionSelector::selectAddrRegImm(MachineOperand &Root) const {
       return {{[=](MachineInstrBuilder &MIB) { MIB.add(LHS); },
                [=](MachineInstrBuilder &MIB) { MIB.addImm(RHSC); }}};
     }
+
+    // Large constant offset. Fold a -2048/2047 adjustment so the whole
+    // constant can be split across an ADDI and the load/store offset.
+    if (RHSC >= -4096 && RHSC <= 4094) {
+      int64_t Adj = RHSC < 0 ? -2048 : 2047;
+      return renderAddiPair(LHS.getReg(), Adj, RHSC - Adj);
+    }
+
+    if (isWorthFoldingAdd(Root.getReg()))
+      if (auto Fns = computeConstAddr(RHSC, /*IsPrefetch=*/false, LHS.getReg()))
+        return Fns;
+  }
+
+  // Bare constant address. IRTranslator lowers inttoptr(C) to
+  // G_INTTOPTR(G_CONSTANT); look through it to reach the constant.
+  if (RootDef->getOpcode() == TargetOpcode::G_INTTOPTR) {
+    MachineInstr *SrcDef = MRI->getVRegDef(RootDef->getOperand(1).getReg());
+    if (SrcDef && SrcDef->getOpcode() == TargetOpcode::G_CONSTANT)
+      RootDef = SrcDef;
+  }
+  if (RootDef->getOpcode() == TargetOpcode::G_CONSTANT) {
+    int64_t CVal = RootDef->getOperand(1).getCImm()->getSExtValue();
+    if (auto Fns = computeConstAddr(CVal, /*IsPrefetch=*/false, Register()))
+      return Fns;
   }
 
-  // TODO: Need to get the immediate from a G_PTR_ADD. Should this be done in
-  // the combiner?
   return {{[=](MachineInstrBuilder &MIB) { MIB.addReg(Root.getReg()); },
            [=](MachineInstrBuilder &MIB) { MIB.addImm(0); }}};
 }
@@ -650,19 +690,7 @@ RISCVInstructionSelector::selectAddrRegImmLsb00000(MachineOperand &Root) const {
     // Large constant: fold a -2048/2016 adjustment to save an instruction.
     if ((-2049 >= RHSC && RHSC >= -4096) || (4063 >= RHSC && RHSC >= 2017)) {
       int64_t Adj = RHSC < 0 ? -2048 : 2016;
-      int64_t AdjustedOffset = RHSC - Adj;
-      Register BaseReg = LHS.getReg();
-      return {{[=](MachineInstrBuilder &MIB) {
-                 Register Tmp = MRI->createVirtualRegister(&RISCV::GPRRegClass);
-                 MachineInstr *Addi =
-                     BuildMI(*MIB->getParent(), *MIB.getInstr(),
-                             MIB->getDebugLoc(), TII.get(RISCV::ADDI), Tmp)
-                         .addReg(BaseReg)
-                         .addImm(AdjustedOffset);
-                 constrainSelectedInstRegOperands(*Addi, TII, TRI, RBI);
-                 MIB.addReg(Tmp);
-               },
-               [=](MachineInstrBuilder &MIB) { MIB.addImm(Adj); }}};
+      return renderAddiPair(LHS.getReg(), RHSC - Adj, Adj);
     }
 
     // Otherwise split the constant into Hi (materialized + added to the base)
@@ -1687,6 +1715,29 @@ bool RISCVInstructionSelector::isRegInFprb(Register Reg) const {
   return RBI.getRegBank(Reg, *MRI, TRI)->getID() == RISCV::FPRBRegBankID;
 }
 
+// A G_PTR_ADD result is worth splitting into Hi (materialized) +
+// Lo12 (folded offset) only if every user is a plain scalar load/store
+// using it as the address. Otherwise the ADD is selected on its own with
+// the full materialized constant, making the Hi materialization here redundant.
+bool RISCVInstructionSelector::isWorthFoldingAdd(Register AddResult) const {
+  for (const MachineOperand &Use : MRI->use_operands(AddResult)) {
+    const MachineInstr *User = Use.getParent();
+    auto *LdSt = dyn_cast<GLoadStore>(User);
+    if (!LdSt)
+      return false;
+    // Must be used as the pointer, not the stored value.
+    if (LdSt->getPointerReg() != AddResult)
+      return false;
+    if (isStrongerThanMonotonic(LdSt->getMMO().getSuccessOrdering()))
+      return false;
+    // Only scalar integer/f16/f32/f64 memory (exclude vectors, f128, ...).
+    LLT Ty = MRI->getType(User->getOperand(0).getReg());
+    if (!Ty.isScalar() || Ty.getSizeInBits() > 64)
+      return false;
+  }
+  return true;
+}
+
 bool RISCVInstructionSelector::selectCopy(MachineInstr &MI) const {
   Register DstReg = MI.getOperand(0).getReg();
 
diff --git a/llvm/test/CodeGen/RISCV/GlobalISel/load-store.ll b/llvm/test/CodeGen/RISCV/GlobalISel/load-store.ll
new file mode 100644
index 0000000000000..e44636adc4c4e
--- /dev/null
+++ b/llvm/test/CodeGen/RISCV/GlobalISel/load-store.ll
@@ -0,0 +1,641 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
+; RUN: llc -global-isel -global-isel-abort=1 -mtriple=riscv32 < %s | FileCheck %s --check-prefix=RV32I
+; RUN: llc -global-isel -global-isel-abort=1 -mtriple=riscv64 < %s | FileCheck %s --check-prefix=RV64I
+
+; ============================================================================
+; Basic simm12 addressing
+; ============================================================================
+
+define i32 @load_offset_positive(ptr %base) nounwind {
+; RV32I-LABEL: load_offset_positive:
+; RV32I:       # %bb.0:
+; RV32I-NEXT:    lw a0, 8(a0)
+; RV32I-NEXT:    ret
+;
+; RV64I-LABEL: load_offset_positive:
+; RV64I:       # %bb.0:
+; RV64I-NEXT:    lw a0, 8(a0)
+; RV64I-NEXT:    ret
+  %ptr = getelementptr i8, ptr %base, i32 8
+  %val = load i32, ptr %ptr
+  ret i32 %val
+}
+
+define i32 @load_offset_negative(ptr %base) nounwind {
+; RV32I-LABEL: load_offset_negative:
+; RV32I:       # %bb.0:
+; RV32I-NEXT:    lw a0, -8(a0)
+; RV32I-NEXT:    ret
+;
+; RV64I-LABEL: load_offset_negative:
+; RV64I:       # %bb.0:
+; RV64I-NEXT:    lw a0, -8(a0)
+; RV64I-NEXT:    ret
+  %ptr = getelementptr i8, ptr %base, i32 -8
+  %val = load i32, ptr %ptr
+  ret i32 %val
+}
+
+define i32 @load_offset_max(ptr %base) nounwind {
+; RV32I-LABEL: load_offset_max:
+; RV32I:       # %bb.0:
+; RV32I-NEXT:    lw a0, 2047(a0)
+; RV32I-NEXT:    ret
+;
+; RV64I-LABEL: load_offset_max:
+; RV64I:       # %bb.0:
+; RV64I-NEXT:    lw a0, 2047(a0)
+; RV64I-NEXT:    ret
+  %ptr = getelementptr i8, ptr %base, i32 2047
+  %val = load i32, ptr %ptr
+  ret i32 %val
+}
+
+define i32 @load_offset_min(ptr %base) nounwind {
+; RV32I-LABEL: load_offset_min:
+; RV32I:       # %bb.0:
+; RV32I-NEXT:    lw a0, -2048(a0)
+; RV32I-NEXT:    ret
+;
+; RV64I-LABEL: load_offset_min:
+; RV64I:       # %bb.0:
+; RV64I-NEXT:    lw a0, -2048(a0)
+; RV64I-NEXT:    ret
+  %ptr = getelementptr i8, ptr %base, i32 -2048
+  %val = load i32, ptr %ptr
+  ret i32 %val
+}
+
+define void @store_offset_positive(ptr %base, i32 %val) nounwind {
+; RV32I-LABEL: store_offset_positive:
+; RV32I:       # %bb.0:
+; RV32I-NEXT:    sw a1, 8(a0)
+; RV32I-NEXT:    ret
+;
+; RV64I-LABEL: store_offset_positive:
+; RV64I:       # %bb.0:
+; RV64I-NEXT:    sw a1, 8(a0)
+; RV64I-NEXT:    ret
+  %ptr = getelementptr i8, ptr %base, i32 8
+  store i32 %val, ptr %ptr
+  ret void
+}
+
+define void @store_offset_negative(ptr %base, i32 %val) nounwind {
+; RV32I-LABEL: store_offset_negative:
+; RV32I:       # %bb.0:
+; RV32I-NEXT:    sw a1, -8(a0)
+; RV32I-NEXT:    ret
+;
+; RV64I-LABEL: store_offset_negative:
+; RV64I:       # %bb.0:
+; RV64I-NEXT:    sw a1, -8(a0)
+; RV64I-NEXT:    ret
+  %ptr = getelementptr i8, ptr %base, i32 -8
+  store i32 %val, ptr %ptr
+  ret void
+}
+
+; ============================================================================
+; AddiPair: positive offsets in [-4096, 4094]
+; ============================================================================
+
+define i32 @load_addi_pair_positive(ptr %base) nounwind {
+; RV32I-LABEL: load_addi_pair_positive:
+; RV32I:       # %bb.0:
+; RV32I-NEXT:    addi a0, a0, 2047
+; RV32I-NEXT:    lw a0, 953(a0)
+; RV32I-NEXT:    ret
+;
+; RV64I-LABEL: load_addi_pair_positive:
+; RV64I:       # %bb.0:
+; RV64I-NEXT:    addi a0, a0, 2047
+; RV64I-NEXT:    lw a0, 953(a0)
+; RV64I-NEXT:    ret
+  %ptr = getelementptr i8, ptr %base, i32 3000
+  %val = load i32, ptr %ptr
+  ret i32 %val
+}
+
+define void @store_addi_pair_positive(ptr %base, i32 %val) nounwind {
+; RV32I-LABEL: store_addi_pair_positive:
+; RV32I:       # %bb.0:
+; RV32I-NEXT:    addi a0, a0, 2047
+; RV32I-NEXT:    sw a1, 953(a0)
+; RV32I-NEXT:    ret
+;
+; RV64I-LABEL: store_addi_pair_positive:
+; RV64I:       # %bb.0:
+; RV64I-NEXT:    addi a0, a0, 2047
+; RV64I-NEXT:    sw a1, 953(a0)
+; RV64I-NEXT:    ret
+  %ptr = getelementptr i8, ptr %base, i32 3000
+  store i32 %val, ptr %ptr
+  ret void
+}
+
+define i32 @load_addi_pair_positive_max(ptr %base) nounwind {
+; RV32I-LABEL: load_addi_pair_positive_max:
+; RV32I:       # %bb.0:
+; RV32I-NEXT:    addi a0, a0, 2047
+; RV32I-NEXT:    lw a0, 2047(a0)
+; RV32I-NEXT:    ret
+;
+; RV64I-LABEL: load_addi_pair_positive_max:
+; RV64I:       # %bb.0:
+; RV64I-NEXT:    addi a0, a0, 2047
+; RV64I-NEXT:    lw a0, 2047(a0)
+; RV64I-NEXT:    ret
+  %ptr = getelementptr i8, ptr %base, i32 4094
+  %val = load i32, ptr %ptr
+  ret i32 %val
+}
+
+define i32 @load_addi_pair_positive_min(ptr %base) nounwind {
+; RV32I-LABEL: load_addi_pair_positive_min:
+; RV32I:       # %bb.0:
+; RV32I-NEXT:    addi a0, a0, 2047
+; RV32I-NEXT:    lw a0, 1(a0)
+; RV32I-NEXT:    ret
+;
+; RV64I-LABEL: load_addi_pair_positive_min:
+; RV64I:       # %bb.0:
+; RV64I-NEXT:    addi a0, a0, 2047
+; RV64I-NEXT:    lw a0, 1(a0)
+; RV64I-NEXT:    ret
+  %ptr = getelementptr i8, ptr %base, i32 2048
+  %val = load i32, ptr %ptr
+  ret i32 %val
+}
+
+; ============================================================================
+; AddiPair: negative offsets in [-4096, -2049]
+; ============================================================================
+
+define i32 @load_addi_pair_negative(ptr %base) nounwind {
+; RV32I-LABEL: load_addi_pair_negative:
+; RV32I:       # %bb.0:
+; RV32I-NEXT:    addi a0, a0, -2048
+; RV32I-NEXT:    lw a0, -952(a0)
+; RV32I-NEXT:    ret
+;
+; RV64I-LABEL: load_addi_pair_negative:
+; RV64I:       # %bb.0:
+; RV64I-NEXT:    addi a0, a0, -2048
+; RV64I-NEXT:    lw a0, -952(a0)
+; RV64I-NEXT:    ret
+  %ptr = getelementptr i8, ptr %base, i32 -3000
+  %val = load i32, ptr %ptr
+  ret i32 %val
+}
+
+define void @store_addi_pair_negative(ptr %base, i32 %val) nounwind {
+; RV32I-LABEL: store_addi_pair_negative:
+; RV32I:       # %bb.0:
+; RV32I-NEXT:    addi a0, a0, -2048
+; RV32I-NEXT:    sw a1, -952(a0)
+; RV32I-NEXT:    ret
+;
+; RV64I-LABEL: store_addi_pair_negative:
+; RV64I:       # %bb.0:
+; RV64I-NEXT:    addi a0, a0, -2048
+; RV64I-NEXT:    sw a1, -952(a0)
+; RV64I-NEXT:    ret
+  %ptr = getelementptr i8, ptr %base, i32 -3000
+  store i32 %val, ptr %ptr
+  ret void
+}
+
+define i32 @load_addi_pair_negative_min(ptr %base) nounwind {
+; RV32I-LABEL: load_addi_pair_negative_min:
+; RV32I:       # %bb.0:
+; RV32I-NEXT:    addi a0, a0, -2048
+; RV32I-NEXT:    lw a0, -2048(a0)
+; RV32I-NEXT:    ret
+;
+; RV64I-LABEL: load_addi_pair_negative_min:
+; RV64I:       # %bb.0:
+; RV64I-NEXT:    addi a0, a0, -2048
+; RV64I-NEXT:    lw a0, -2048(a0)
+; RV64I-NEXT:    ret
+  %ptr = getelementptr i8, ptr %base, i32 -4096
+  %val = load i32, ptr %ptr
+  ret i32 %val
+}
+
+define i32 @load_addi_pair_negative_max(ptr %base) nounwind {
+; RV32I-LABEL: load_addi_pair_negative_max:
+; RV32I:       # %bb.0:
+; RV32I-NEXT:    addi a0, a0, -2048
+; RV32I-NEXT:    lw a0, -1(a0)
+; RV32I-NEXT:    ret
+;
+; RV64I-LABEL: load_addi_pair_negative_max:
+; RV64I:       # %bb.0:
+; RV64I-NEXT:    addi a0, a0, -2048
+; RV64I-NEXT:    lw a0, -1(a0)
+; RV64I-NEXT:    ret
+  %ptr = getelementptr i8, ptr %base, i32 -2049
+  %val = load i32, ptr %ptr
+  ret i32 %val
+}
+
+; ============================================================================
+; Just outside the AddiPair range.
+;
+; These should not enter the simple:
+;
+;   [-4096, 4094]
+;
+; ADDI + load/store-offset path.
+; ============================================================================
+
+define i32 @load_offset_above_addi_pair_range(ptr %base) nounwind {
+; RV32I-LABEL: load_offset_above_addi_pair_range:
+; RV32I:       # %bb.0:
+; RV32I-NEXT:    lui a1, 1
+; RV32I-NEXT:    add a0, a0, a1
+; RV32I-NEXT:    lw a0, -1(a0)
+; RV32I-NEXT:    ret
+;
+; RV64I-LABEL: load_offset_above_addi_pair_range:
+; RV64I:       # %bb.0:
+; RV64I-NEXT:    lui a1, 1
+; RV64I-NEXT:    add a0, a0, a1
+; RV64I-NEXT:    lw a0, -1(a0)
+; RV64I-NEXT:    ret
+  %ptr = getelementptr i8, ptr %base, i32 4095
+  %val = load i32, ptr %ptr
+  ret i32 %val
+}
+
+define i32 @load_offset_below_addi_pair_range(ptr %base) nounwind {
+; RV32I-LABEL: load_offset_below_addi_pair_range:
+; RV32I:       # %bb.0:
+; RV32I-NEXT:    lui a1, 1048575
+; RV32I-NEXT:    add a0, a0, a1
+; RV32I-NEXT:    lw a0, -1(a0)
+; RV32I-NEXT:    ret
+;
+; RV64I-LABEL: load_offset_below_addi_pair_range:
+; RV64I:       # %bb.0:
+; RV64I-NEXT:    lui a1, 1048575
+; RV64I-NEXT:    add a0, a0, a1
+; RV64I-NEXT:    lw a0, -1(a0)
+; RV64I-NEXT:    ret
+  %ptr = getelementptr i8, ptr %base, i32 -4097
+  %val = load i32, ptr %ptr
+  ret i32 %val
+}
+
+; ============================================================================
+; Large constant address folding.
+;
+; The address is only used by the load/store, so the ADD can be folded and
+; computeConstAddr() may materialize the high part while folding the low
+; part into the memory instruction.
+; ============================================================================
+
+define i32 @load_large_offset(ptr %base) nounwind {
+; RV32I-LABEL: load_large_offset:
+; RV32I:       # %bb.0:
+; RV32I-NEXT:    lui a1, 24
+; RV32I-NEXT:    add a0, a0, a1
+; RV32I-NEXT:    lw a0, 1696(a0)
+; RV32I-NEXT:    ret
+;
+; RV64I-LABEL: load_large_offset:
+; RV64I:       # %bb.0:
+; RV64I-NEXT:    lui a1, 24
+; RV64I-NEXT:    add a0, a0, a1
+; RV64I-NEXT:    lw a0, 1696(a0)
+; RV64I-NEXT:    ret
+  %ptr = getelementptr i8, ptr %base, i32 100000
+  %val = load i32, ptr %ptr
+  ret i32 %val
+}
+
+define void @store_large_offset(ptr %base, i32 %val) nounwind {
+; RV32I-LABEL: store_large_offset:
+; RV32I:       # %bb.0:
+; RV32I-NEXT:    lui a2, 24
+; RV32I-NEXT:    add a0, a0, a2
+; RV32I-NEXT:    sw a1, 1696(a0)
+; RV32I-NEXT:    ret
+;
+; RV64I-LABEL: store_large_offset:
+; RV64I:       # %bb.0:
+; RV64I-NEXT:    lui a2, 24
+; RV64I-NEXT:    add a0, a0, a2
+; RV64I-NEXT:    sw a1, 1696(a0)
+; RV64I-NEXT:    ret
+  %ptr = getelementptr i8, ptr %base, i32 100000
+  store i32 %val, ptr %ptr
+  ret void
+}
+
+define i32 @load_large_negative_offset(ptr %base) nounwind {
+; RV32I-LABEL: load_large_negative_offset:
+; RV32I:       # %bb.0:
+; RV32I-NEXT:    lui a1, 1048552
+; RV32I-NEXT:    add a0, a0, a1
+; RV32I-NEXT:    lw a0, -1696(a0)
+; RV32I-NEXT:    ret
+;
+; RV64I-LABEL: load_large_negative_offset:
+; RV64I:       # %bb.0:
+; RV64I-NEXT:    lui a1, 1048552
+; RV64I-NEXT:    add a0, a0, a1
+; RV64I-NEXT:    lw a0, -1696(a0)
+; RV64I-NEXT:    ret
+  %ptr = getelementptr i8, ptr %base, i32 -100000
+  %val = load i32, ptr %ptr
+  ret i32 %val
+}
+
+; ============================================================================
+; Large constant with a non-memory user.
+;
+; The computed address must remain available for the call, so the ADD should
+; not be folded solely into the load.
+; ============================================================================
+
+declare void @use_pointer(ptr)
+
+define i32 @load_large_offset_multi_use(ptr %base) nounwind {
+; RV32I-LABEL: load_large_offset_multi_use:
+; RV32I:       # %bb.0:
+; RV32I-NEXT:    addi sp, sp, -16
+; RV32I-NEXT:    sw ra, 12(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    sw s0, 8(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    lui a1, 24
+; RV32I-NEXT:    addi a1, a1, 1696
+; RV32I-NEXT:    add a0, a0, a1
+; RV32I-NEXT:    lw s0, 0(a0)
+; RV32I-NEXT:    call use_pointer
+; RV32I-NEXT:    mv a0, s0
+; RV32I-NEXT:    lw ra, 12(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    lw s0, 8(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    addi sp, sp, 16
+; RV32I-NEXT:    ret
+;
+; RV64I-LABEL: load_large_offset_multi_use:
+; RV64I:       # %bb.0:
+; RV64I-NEXT:    addi sp, sp, -16
+; RV64I-NEXT:    sd ra, 8(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    sd s0, 0(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    lui a1, 24
+; RV64I-NEXT:    addi a1, a1, 1696
+; RV64I-NEXT:    add a0, a0, a1
+; RV64I-NEXT:    lw s0, 0(a0)
+; RV64I-NEXT:    call use_pointer
+; RV64I-NEXT:    mv a0, s0
+; RV64I-NEXT:    ld ra, 8(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    ld s0, 0(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    addi sp, sp, 16
+; RV64I-NEXT:    ret
+  %ptr = getelementptr i8, ptr %base, i32 100000
+  %val = load i32, ptr %ptr
+  call void @use_pointer(ptr %ptr)
+  ret i32 %val
+}
+
+define void @store_large_offset_multi_use(ptr %base, i32 %val) nounwind {
+; RV32I-LABEL: store_large_offset_multi_use:
+; RV32I:       # %bb.0:
+; RV32I-NEXT:    addi sp, sp, -16
+; RV32I-NEXT:    sw ra, 12(sp) # 4-byte Folded Spill
+; RV32I-NEXT:    lui a2, 24
+; RV32I-NEXT:    addi a2, a2, 1696
+; RV32I-NEXT:    add a0, a0, a2
+; RV32I-NEXT:    sw a1, 0(a0)
+; RV32I-NEXT:    call use_pointer
+; RV32I-NEXT:    lw ra, 12(sp) # 4-byte Folded Reload
+; RV32I-NEXT:    addi sp, sp, 16
+; RV32I-NEXT:    ret
+;
+; RV64I-LABEL: store_large_offset_multi_use:
+; RV64I:       # %bb.0:
+; RV64I-NEXT:    addi sp, sp, -16
+; RV64I-NEXT:    sd ra, 8(sp) # 8-byte Folded Spill
+; RV64I-NEXT:    lui a2, 24
+; RV64I-NEXT:    addi a2, a2, 1696
+; RV64I-NEXT:    add a0, a0, a2
+; RV64I-NEXT:    sw a1, 0(a0)
+; RV64I-NEXT:    call use_pointer
+; RV64I-NEXT:    ld ra, 8(sp) # 8-byte Folded Reload
+; RV64I-NEXT:    addi sp, sp, 16
+; RV64I-NEXT:    ret
+  %ptr = getelementptr i8, ptr %base, i32 100000
+  store i32 %val, ptr %ptr
+  call void @use_pointer(ptr %ptr)
+  ret void
+}
+
+; ============================================================================
+; The address has multiple load/store users.
+;
+; Folding the address into both memory operations should still be possible
+; when the address itself has no non-memory users.
+; ============================================================================
+
+define i32 @load_large_offset_two_loads(ptr %base) nounwind {
+; RV32I-LABEL: load_large_offset_two_loads:
+; RV32I:       # %bb.0:
+; RV32I-NEXT:    lui a1, 24
+; RV32I-NEXT:    add a0, a0, a1
+; RV32I-NEXT:    lw a1, 1696(a0)
+; RV32I-NEXT:    lw a0, 1696(a0)
+; RV32I-NEXT:    add a0, a1, a0
+; RV32I-NEXT:    ret
+;
+; RV64I-LABEL: load_large_offset_two_loads:
+; RV64I:       # %bb.0:
+; RV64I-NEXT:    lui a1, 24
+; RV64I-NEXT:    add a0, a0, a1
+; RV64I-NEXT:    lw a1, 1696(a0)
+; RV64I-NEXT:    lw a0, 1696(a0)
+; RV64I-NEXT:    addw a0, a1, a0
+; RV64I-NEXT:    ret
+  %ptr = getelementptr i8, ptr %base, i32 100000
+  %v1 = load i32, ptr %ptr
+  %v2 = load i32, ptr %ptr
+  %sum = add i32 %v1, %v2
+  ret i32 %sum
+}
+
+define void @store_large_offset_two_stores(ptr %base, i32 %v1, i32 %v2) nounwind {
+; RV32I-LABEL: store_large_offset_two_stores:
+; RV32I:       # %bb.0:
+; RV32I-NEXT:    lui a3, 24
+; RV32I-NEXT:    add a0, a0, a3
+; RV32I-NEXT:    sw a1, 1696(a0)
+; RV32I-NEXT:    sw a2, 1696(a0)
+; RV32I-NEXT:    ret
+;
+; RV64I-LABEL: store_large_offset_two_stores:
+; RV64I:       # %bb.0:
+; RV64I-NEXT:    lui a3, 24
+; RV64I-NEXT:    add a0, a0, a3
+; RV64I-NEXT:    sw a1, 1696(a0)
+; RV64I-NEXT:    sw a2, 1696(a0)
+; RV64I-NEXT:    ret
+  %ptr = getelementptr i8, ptr %base, i32 100000
+  store i32 %v1, ptr %ptr
+  store i32 %v2, ptr %ptr
+  ret void
+}
+
+; ============================================================================
+; Bare constant address.
+; ============================================================================
+
+define i32 @load_absolute_address() nounwind {
+; RV32I-LABEL: load_absolute_address:
+; RV32I:       # %bb.0:
+; RV32I-NEXT:    lui a0, 74564
+; RV32I-NEXT:    lw a0, -1024(a0)
+; RV32I-NEXT:    ret
+;
+; RV64I-LABEL: load_absolute_address:
+; RV64I:       # %bb.0:
+; RV64I-NEXT:    lui a0, 74564
+; RV64I-NEXT:    lw a0, -1024(a0)
+; RV64I-NEXT:    ret
+  %ptr = inttoptr i64 305413120 to ptr
+  %val = load i32, ptr %ptr
+  ret i32 %val
+}
+
+define i32 @load_absolute_small_address() nounwind {
+; RV32I-LABEL: load_absolute_small_address:
+; RV32I:       # %bb.0:
+; RV32I-NEXT:    lui a0, 1
+; RV32I-NEXT:    lw a0, 0(a0)
+; RV32I-NEXT:    ret
+;
+; RV64I-LABEL: load_absolute_small_address:
+; RV64I:       # %bb.0:
+; RV64I-NEXT:    lui a0, 1
+; RV64I-NEXT:    lw a0, 0(a0)
+; RV64I-NEXT:    ret
+  %ptr = inttoptr i64 4096 to ptr
+  %val = load i32, ptr %ptr
+  ret i32 %val
+}
+
+; ============================================================================
+; Absolute address used by a store.
+; ============================================================================
+
+define void @store_absolute_address(i32 %val) nounwind {
+; RV32I-LABEL: store_absolute_address:
+; RV32I:       # %bb.0:
+; RV32I-NEXT:    lui a1, 74564
+; RV32I-NEXT:    sw a0, -1024(a1)
+; RV32I-NEXT:    ret
+;
+; RV64I-LABEL: store_absolute_address:
+; RV64I:       # %bb.0:
+; RV64I-NEXT:    lui a1, 74564
+; RV64I-NEXT:    sw a0, -1024(a1)
+; RV64I-NEXT:    ret
+  %ptr = inttoptr i64 305413120 to ptr
+  store i32 %val, ptr %ptr
+  ret void
+}
+
+; ============================================================================
+; Large constants whose low 12 bits require signed-immediate handling.
+;
+; These provide additional coverage for computeConstAddr() rather than
+; relying only on 100000.
+; ============================================================================
+
+define i32 @load_large_offset_lo_positive(ptr %base) nounwind {
+; RV32I-LABEL: load_large_offset_lo_positive:
+; RV32I:       # %bb.0:
+; RV32I-NEXT:    lui a1, 24
+; RV32I-NEXT:    add a0, a0, a1
+; RV32I-NEXT:    lw a0, 0(a0)
+; RV32I-NEXT:    ret
+;
+; RV64I-LABEL: load_large_offset_lo_positive:
+; RV64I:       # %bb.0:
+; RV64I-NEXT:    lui a1, 24
+; RV64I-NEXT:    add a0, a0, a1
+; RV64I-NEXT:    lw a0, 0(a0)
+; RV64I-NEXT:    ret
+  %ptr = getelementptr i8, ptr %base, i32 98304
+  %val = load i32, ptr %ptr
+  ret i32 %val
+}
+
+define i32 @load_large_offset_lo_negative(ptr %base) nounwind {
+; RV32I-LABEL: load_large_offset_lo_negative:
+; RV32I:       # %bb.0:
+; RV32I-NEXT:    lui a1, 24
+; RV32I-NEXT:    add a0, a0, a1
+; RV32I-NEXT:    lw a0, 1(a0)
+; RV32I-NEXT:    ret
+;
+; RV64I-LABEL: load_large_offset_lo_negative:
+; RV64I:       # %bb.0:
+; RV64I-NEXT:    lui a1, 24
+; RV64I-NEXT:    add a0, a0, a1
+; RV64I-NEXT:    lw a0, 1(a0)
+; RV64I-NEXT:    ret
+  %ptr = getelementptr i8, ptr %base, i32 98305
+  %val = load i32, ptr %ptr
+  ret i32 %val
+}
+
+define i32 @load_large_offset_cross_signed_boundary(ptr %base) nounwind {
+; RV32I-LABEL: load_large_offset_cross_signed_boundary:
+; RV32I:       # %bb.0:
+; RV32I-NEXT:    lui a1, 25
+; RV32I-NEXT:    add a0, a0, a1
+; RV32I-NEXT:    lw a0, -2048(a0)
+; RV32I-NEXT:    ret
+;
+; RV64I-LABEL: load_large_offset_cross_signed_boundary:
+; RV64I:       # %bb.0:
+; RV64I-NEXT:    lui a1, 25
+; RV64I-NEXT:    add a0, a0, a1
+; RV64I-NEXT:    lw a0, -2048(a0)
+; RV64I-NEXT:    ret
+  %ptr = getelementptr i8, ptr %base, i32 100352
+  %val = load i32, ptr %ptr
+  ret i32 %val
+}
+
+; ============================================================================
+; Address reused by memory operations and a pointer comparison.
+;
+; This ensures the address has a non-memory use other than a call.
+; ============================================================================
+
+define i1 @compare_large_offset_address(ptr %base, ptr %other) nounwind {
+; RV32I-LABEL: compare_large_offset_address:
+; RV32I:       # %bb.0:
+; RV32I-NEXT:    lui a2, 24
+; RV32I-NEXT:    addi a2, a2, 1696
+; RV32I-NEXT:    add a0, a0, a2
+; RV32I-NEXT:    xor a0, a0, a1
+; RV32I-NEXT:    seqz a0, a0
+; RV32I-NEXT:    ret
+;
+; RV64I-LABEL: compare_large_offset_address:
+; RV64I:       # %bb.0:
+; RV64I-NEXT:    lui a2, 24
+; RV64I-NEXT:    addi a2, a2, 1696
+; RV64I-NEXT:    add a0, a0, a2
+; RV64I-NEXT:    xor a0, a0, a1
+; RV64I-NEXT:    seqz a0, a0
+; RV64I-NEXT:    ret
+  %ptr = getelementptr i8, ptr %base, i32 100000
+  %val = load i32, ptr %ptr
+  %cmp = icmp eq ptr %ptr, %other
+  %unused = add i32 %val, 0
+  ret i1 %cmp
+}

>From f9c43aa8b4d00b98c87c2d17f89b2b646aff05a7 Mon Sep 17 00:00:00 2001
From: Kane Wang <wangqiang1 at kylinos.cn>
Date: Tue, 1 Sep 2026 14:46:38 +0800
Subject: [PATCH 2/3] [RISCV][GlobalISel] Use mi_match in selectAddrRegImm
 (NFC)

Replace direct getVRegDef calls with mi_matchi.
---
 .../RISCV/GISel/RISCVInstructionSelector.cpp  | 45 +++++++++----------
 1 file changed, 21 insertions(+), 24 deletions(-)

diff --git a/llvm/lib/Target/RISCV/GISel/RISCVInstructionSelector.cpp b/llvm/lib/Target/RISCV/GISel/RISCVInstructionSelector.cpp
index d3d98d92aac44..62107beffb0cc 100644
--- a/llvm/lib/Target/RISCV/GISel/RISCVInstructionSelector.cpp
+++ b/llvm/lib/Target/RISCV/GISel/RISCVInstructionSelector.cpp
@@ -596,29 +596,30 @@ RISCVInstructionSelector::selectAddrRegImm(MachineOperand &Root) const {
   if (!Root.isReg())
     return std::nullopt;
 
-  MachineInstr *RootDef = MRI->getVRegDef(Root.getReg());
-  if (RootDef->getOpcode() == TargetOpcode::G_FRAME_INDEX) {
+  Register RootReg = Root.getReg();
+
+  // Frame index.
+  int FI;
+  if (mi_match(RootReg, *MRI, m_GFrameIndex(FI))) {
     return {{
-        [=](MachineInstrBuilder &MIB) { MIB.add(RootDef->getOperand(1)); },
+        [=](MachineInstrBuilder &MIB) { MIB.addFrameIndex(FI); },
         [=](MachineInstrBuilder &MIB) { MIB.addImm(0); },
     }};
   }
 
-  if (isBaseWithConstantOffset(Root, *MRI)) {
-    MachineOperand &LHS = RootDef->getOperand(1);
-    MachineOperand &RHS = RootDef->getOperand(2);
-    MachineInstr *LHSDef = MRI->getVRegDef(LHS.getReg());
-    MachineInstr *RHSDef = MRI->getVRegDef(RHS.getReg());
-
-    int64_t RHSC = RHSDef->getOperand(1).getCImm()->getSExtValue();
+  // base + constant offset (G_PTR_ADD).
+  Register BaseReg;
+  int64_t RHSC;
+  if (mi_match(RootReg, *MRI, m_GPtrAdd(m_Reg(BaseReg), m_ICst(RHSC)))) {
     if (isInt<12>(RHSC)) {
-      if (LHSDef->getOpcode() == TargetOpcode::G_FRAME_INDEX)
+      int BaseFI;
+      if (mi_match(BaseReg, *MRI, m_GFrameIndex(BaseFI)))
         return {{
-            [=](MachineInstrBuilder &MIB) { MIB.add(LHSDef->getOperand(1)); },
+            [=](MachineInstrBuilder &MIB) { MIB.addFrameIndex(BaseFI); },
             [=](MachineInstrBuilder &MIB) { MIB.addImm(RHSC); },
         }};
 
-      return {{[=](MachineInstrBuilder &MIB) { MIB.add(LHS); },
+      return {{[=](MachineInstrBuilder &MIB) { MIB.addReg(BaseReg); },
                [=](MachineInstrBuilder &MIB) { MIB.addImm(RHSC); }}};
     }
 
@@ -626,28 +627,24 @@ RISCVInstructionSelector::selectAddrRegImm(MachineOperand &Root) const {
     // constant can be split across an ADDI and the load/store offset.
     if (RHSC >= -4096 && RHSC <= 4094) {
       int64_t Adj = RHSC < 0 ? -2048 : 2047;
-      return renderAddiPair(LHS.getReg(), Adj, RHSC - Adj);
+      return renderAddiPair(BaseReg, Adj, RHSC - Adj);
     }
 
-    if (isWorthFoldingAdd(Root.getReg()))
-      if (auto Fns = computeConstAddr(RHSC, /*IsPrefetch=*/false, LHS.getReg()))
+    if (isWorthFoldingAdd(RootReg))
+      if (auto Fns = computeConstAddr(RHSC, /*IsPrefetch=*/false, BaseReg))
         return Fns;
   }
 
   // Bare constant address. IRTranslator lowers inttoptr(C) to
   // G_INTTOPTR(G_CONSTANT); look through it to reach the constant.
-  if (RootDef->getOpcode() == TargetOpcode::G_INTTOPTR) {
-    MachineInstr *SrcDef = MRI->getVRegDef(RootDef->getOperand(1).getReg());
-    if (SrcDef && SrcDef->getOpcode() == TargetOpcode::G_CONSTANT)
-      RootDef = SrcDef;
-  }
-  if (RootDef->getOpcode() == TargetOpcode::G_CONSTANT) {
-    int64_t CVal = RootDef->getOperand(1).getCImm()->getSExtValue();
+  int64_t CVal;
+  if (mi_match(RootReg, *MRI, m_GIntToPtr(m_ICst(CVal))) ||
+      mi_match(RootReg, *MRI, m_ICst(CVal))) {
     if (auto Fns = computeConstAddr(CVal, /*IsPrefetch=*/false, Register()))
       return Fns;
   }
 
-  return {{[=](MachineInstrBuilder &MIB) { MIB.addReg(Root.getReg()); },
+  return {{[=](MachineInstrBuilder &MIB) { MIB.addReg(RootReg); },
            [=](MachineInstrBuilder &MIB) { MIB.addImm(0); }}};
 }
 

>From 33f12859300c233e8335f62a6f1ed47301767ba1 Mon Sep 17 00:00:00 2001
From: Kane Wang <wangqiang1 at kylinos.cn>
Date: Tue, 1 Sep 2026 15:30:41 +0800
Subject: [PATCH 3/3] [RISCV][GlobalISel] Use mi_match in
 selectAddrRegImmLsb00000 (NFC)

---
 .../RISCV/GISel/RISCVInstructionSelector.cpp  | 45 +++++++++----------
 1 file changed, 21 insertions(+), 24 deletions(-)

diff --git a/llvm/lib/Target/RISCV/GISel/RISCVInstructionSelector.cpp b/llvm/lib/Target/RISCV/GISel/RISCVInstructionSelector.cpp
index 62107beffb0cc..a1158c5322bb4 100644
--- a/llvm/lib/Target/RISCV/GISel/RISCVInstructionSelector.cpp
+++ b/llvm/lib/Target/RISCV/GISel/RISCVInstructionSelector.cpp
@@ -653,63 +653,60 @@ RISCVInstructionSelector::selectAddrRegImmLsb00000(MachineOperand &Root) const {
   if (!Root.isReg())
     return std::nullopt;
 
-  MachineInstr *RootDef = MRI->getVRegDef(Root.getReg());
-  if (RootDef->getOpcode() == TargetOpcode::G_FRAME_INDEX) {
+  Register RootReg = Root.getReg();
+
+  // Frame index.
+  int FI;
+  if (mi_match(RootReg, *MRI, m_GFrameIndex(FI))) {
     return {{
-        [=](MachineInstrBuilder &MIB) { MIB.add(RootDef->getOperand(1)); },
+        [=](MachineInstrBuilder &MIB) { MIB.addFrameIndex(FI); },
         [=](MachineInstrBuilder &MIB) { MIB.addImm(0); },
     }};
   }
 
-  if (isBaseWithConstantOffset(Root, *MRI)) {
-    MachineOperand &LHS = RootDef->getOperand(1);
-    MachineOperand &RHS = RootDef->getOperand(2);
-    MachineInstr *LHSDef = MRI->getVRegDef(LHS.getReg());
-    MachineInstr *RHSDef = MRI->getVRegDef(RHS.getReg());
-    int64_t RHSC = RHSDef->getOperand(1).getCImm()->getSExtValue();
-
+  // base + constant offset (G_PTR_ADD).
+  Register BaseReg;
+  int64_t RHSC;
+  if (mi_match(RootReg, *MRI, m_GPtrAdd(m_Reg(BaseReg), m_ICst(RHSC)))) {
     if (isInt<12>(RHSC)) {
       // Not a multiple of 32: can't encode, use the address as-is.
       if ((RHSC & 0b11111) != 0) {
-        return {{[=](MachineInstrBuilder &MIB) { MIB.addReg(Root.getReg()); },
+        return {{[=](MachineInstrBuilder &MIB) { MIB.addReg(RootReg); },
                  [=](MachineInstrBuilder &MIB) { MIB.addImm(0); }}};
       }
       // Fold the offset.
-      if (LHSDef->getOpcode() == TargetOpcode::G_FRAME_INDEX)
+      int BaseFI;
+      if (mi_match(BaseReg, *MRI, m_GFrameIndex(BaseFI)))
         return {{
-            [=](MachineInstrBuilder &MIB) { MIB.add(LHSDef->getOperand(1)); },
+            [=](MachineInstrBuilder &MIB) { MIB.addFrameIndex(BaseFI); },
             [=](MachineInstrBuilder &MIB) { MIB.addImm(RHSC); },
         }};
-      return {{[=](MachineInstrBuilder &MIB) { MIB.add(LHS); },
+      return {{[=](MachineInstrBuilder &MIB) { MIB.addReg(BaseReg); },
                [=](MachineInstrBuilder &MIB) { MIB.addImm(RHSC); }}};
     }
 
     // Large constant: fold a -2048/2016 adjustment to save an instruction.
     if ((-2049 >= RHSC && RHSC >= -4096) || (4063 >= RHSC && RHSC >= 2017)) {
       int64_t Adj = RHSC < 0 ? -2048 : 2016;
-      return renderAddiPair(LHS.getReg(), RHSC - Adj, Adj);
+      return renderAddiPair(BaseReg, RHSC - Adj, Adj);
     }
 
     // Otherwise split the constant into Hi (materialized + added to the base)
     // and Lo12 (folded offset).
-    if (auto Fns = computeConstAddr(RHSC, /*IsPrefetch=*/true, LHS.getReg()))
+    if (auto Fns = computeConstAddr(RHSC, /*IsPrefetch=*/true, BaseReg))
       return Fns;
   }
 
   // Bare constant address. IRTranslator emits inttoptr(C) as
   // G_INTTOPTR(G_CONSTANT); look through the G_INTTOPTR to reach the constant.
-  if (RootDef->getOpcode() == TargetOpcode::G_INTTOPTR) {
-    MachineInstr *SrcDef = MRI->getVRegDef(RootDef->getOperand(1).getReg());
-    if (SrcDef->getOpcode() == TargetOpcode::G_CONSTANT)
-      RootDef = SrcDef;
-  }
-  if (RootDef->getOpcode() == TargetOpcode::G_CONSTANT) {
-    int64_t CVal = RootDef->getOperand(1).getCImm()->getSExtValue();
+  int64_t CVal;
+  if (mi_match(RootReg, *MRI, m_GIntToPtr(m_ICst(CVal))) ||
+      mi_match(RootReg, *MRI, m_ICst(CVal))) {
     if (auto Fns = computeConstAddr(CVal, /*IsPrefetch=*/true, Register()))
       return Fns;
   }
 
-  return {{[=](MachineInstrBuilder &MIB) { MIB.addReg(Root.getReg()); },
+  return {{[=](MachineInstrBuilder &MIB) { MIB.addReg(RootReg); },
            [=](MachineInstrBuilder &MIB) { MIB.addImm(0); }}};
 }
 



More information about the llvm-commits mailing list