[llvm] [RISCV][GlobalISel] Fold large constant offsets in selectAddrRegImm (PR #219161)
Kane Wang via llvm-commits
llvm-commits at lists.llvm.org
Tue Sep 1 00:31:01 PDT 2026
https://github.com/ReVe1uv updated https://github.com/llvm/llvm-project/pull/219161
>From 7516d44abcc5b99ee98cd784260d3c0afec9d26b Mon Sep 17 00:00:00 2001
From: Kane Wang <wangqiang1 at kylinos.cn>
Date: Thu, 27 Aug 2026 17:49:36 +0800
Subject: [PATCH 1/3] [RISCV][GlobalISel] Fold large constant offsets in
selectAddrRegImm
Fold ADDI adjustment (AddiPair) for offsets in [-4096, 4094] and split
larger constants into materialized Hi + Lo12 offset, matching SDAG.
Add isWorthFoldingAdd to guard the split and extract the shared ADDI
renderer into renderAddiPair.
Assisted-by: Claude
---
.../RISCV/GISel/RISCVInstructionSelector.cpp | 83 ++-
.../CodeGen/RISCV/GlobalISel/load-store.ll | 641 ++++++++++++++++++
2 files changed, 708 insertions(+), 16 deletions(-)
create mode 100644 llvm/test/CodeGen/RISCV/GlobalISel/load-store.ll
diff --git a/llvm/lib/Target/RISCV/GISel/RISCVInstructionSelector.cpp b/llvm/lib/Target/RISCV/GISel/RISCVInstructionSelector.cpp
index 8af102776b3d3..d3d98d92aac44 100644
--- a/llvm/lib/Target/RISCV/GISel/RISCVInstructionSelector.cpp
+++ b/llvm/lib/Target/RISCV/GISel/RISCVInstructionSelector.cpp
@@ -67,6 +67,7 @@ class RISCVInstructionSelector : public InstructionSelector {
bool isRegInGprb(Register Reg) const;
bool isRegInFprb(Register Reg) const;
+ bool isWorthFoldingAdd(Register AddResult) const;
// tblgen-erated 'select' implementation, used as the initial selector for
// the patterns that don't require complex C++.
@@ -155,7 +156,8 @@ class RISCVInstructionSelector : public InstructionSelector {
}
ComplexRendererFns renderVLOp(MachineOperand &Root) const;
-
+ ComplexRendererFns renderAddiPair(Register BaseReg, int64_t AddiImm,
+ int64_t OffsetImm) const;
// Custom renderers for tablegen
void renderNegImm(MachineInstrBuilder &MIB, const MachineInstr &MI,
int OpIdx) const;
@@ -573,6 +575,22 @@ RISCVInstructionSelector::renderVLOp(MachineOperand &Root) const {
return {{[=](MachineInstrBuilder &MIB) { MIB.addReg(Root.getReg()); }}};
}
+InstructionSelector::ComplexRendererFns
+RISCVInstructionSelector::renderAddiPair(Register BaseReg, int64_t AddiImm,
+ int64_t OffsetImm) const {
+ return {{[=](MachineInstrBuilder &MIB) {
+ Register Tmp = MRI->createVirtualRegister(&RISCV::GPRRegClass);
+ MachineInstr *Addi =
+ BuildMI(*MIB->getParent(), *MIB.getInstr(), MIB->getDebugLoc(),
+ TII.get(RISCV::ADDI), Tmp)
+ .addReg(BaseReg)
+ .addImm(AddiImm);
+ constrainSelectedInstRegOperands(*Addi, TII, TRI, RBI);
+ MIB.addReg(Tmp);
+ },
+ [=](MachineInstrBuilder &MIB) { MIB.addImm(OffsetImm); }}};
+}
+
InstructionSelector::ComplexRendererFns
RISCVInstructionSelector::selectAddrRegImm(MachineOperand &Root) const {
if (!Root.isReg())
@@ -603,10 +621,32 @@ RISCVInstructionSelector::selectAddrRegImm(MachineOperand &Root) const {
return {{[=](MachineInstrBuilder &MIB) { MIB.add(LHS); },
[=](MachineInstrBuilder &MIB) { MIB.addImm(RHSC); }}};
}
+
+ // Large constant offset. Fold a -2048/2047 adjustment so the whole
+ // constant can be split across an ADDI and the load/store offset.
+ if (RHSC >= -4096 && RHSC <= 4094) {
+ int64_t Adj = RHSC < 0 ? -2048 : 2047;
+ return renderAddiPair(LHS.getReg(), Adj, RHSC - Adj);
+ }
+
+ if (isWorthFoldingAdd(Root.getReg()))
+ if (auto Fns = computeConstAddr(RHSC, /*IsPrefetch=*/false, LHS.getReg()))
+ return Fns;
+ }
+
+ // Bare constant address. IRTranslator lowers inttoptr(C) to
+ // G_INTTOPTR(G_CONSTANT); look through it to reach the constant.
+ if (RootDef->getOpcode() == TargetOpcode::G_INTTOPTR) {
+ MachineInstr *SrcDef = MRI->getVRegDef(RootDef->getOperand(1).getReg());
+ if (SrcDef && SrcDef->getOpcode() == TargetOpcode::G_CONSTANT)
+ RootDef = SrcDef;
+ }
+ if (RootDef->getOpcode() == TargetOpcode::G_CONSTANT) {
+ int64_t CVal = RootDef->getOperand(1).getCImm()->getSExtValue();
+ if (auto Fns = computeConstAddr(CVal, /*IsPrefetch=*/false, Register()))
+ return Fns;
}
- // TODO: Need to get the immediate from a G_PTR_ADD. Should this be done in
- // the combiner?
return {{[=](MachineInstrBuilder &MIB) { MIB.addReg(Root.getReg()); },
[=](MachineInstrBuilder &MIB) { MIB.addImm(0); }}};
}
@@ -650,19 +690,7 @@ RISCVInstructionSelector::selectAddrRegImmLsb00000(MachineOperand &Root) const {
// Large constant: fold a -2048/2016 adjustment to save an instruction.
if ((-2049 >= RHSC && RHSC >= -4096) || (4063 >= RHSC && RHSC >= 2017)) {
int64_t Adj = RHSC < 0 ? -2048 : 2016;
- int64_t AdjustedOffset = RHSC - Adj;
- Register BaseReg = LHS.getReg();
- return {{[=](MachineInstrBuilder &MIB) {
- Register Tmp = MRI->createVirtualRegister(&RISCV::GPRRegClass);
- MachineInstr *Addi =
- BuildMI(*MIB->getParent(), *MIB.getInstr(),
- MIB->getDebugLoc(), TII.get(RISCV::ADDI), Tmp)
- .addReg(BaseReg)
- .addImm(AdjustedOffset);
- constrainSelectedInstRegOperands(*Addi, TII, TRI, RBI);
- MIB.addReg(Tmp);
- },
- [=](MachineInstrBuilder &MIB) { MIB.addImm(Adj); }}};
+ return renderAddiPair(LHS.getReg(), RHSC - Adj, Adj);
}
// Otherwise split the constant into Hi (materialized + added to the base)
@@ -1687,6 +1715,29 @@ bool RISCVInstructionSelector::isRegInFprb(Register Reg) const {
return RBI.getRegBank(Reg, *MRI, TRI)->getID() == RISCV::FPRBRegBankID;
}
+// A G_PTR_ADD result is worth splitting into Hi (materialized) +
+// Lo12 (folded offset) only if every user is a plain scalar load/store
+// using it as the address. Otherwise the ADD is selected on its own with
+// the full materialized constant, making the Hi materialization here redundant.
+bool RISCVInstructionSelector::isWorthFoldingAdd(Register AddResult) const {
+ for (const MachineOperand &Use : MRI->use_operands(AddResult)) {
+ const MachineInstr *User = Use.getParent();
+ auto *LdSt = dyn_cast<GLoadStore>(User);
+ if (!LdSt)
+ return false;
+ // Must be used as the pointer, not the stored value.
+ if (LdSt->getPointerReg() != AddResult)
+ return false;
+ if (isStrongerThanMonotonic(LdSt->getMMO().getSuccessOrdering()))
+ return false;
+ // Only scalar integer/f16/f32/f64 memory (exclude vectors, f128, ...).
+ LLT Ty = MRI->getType(User->getOperand(0).getReg());
+ if (!Ty.isScalar() || Ty.getSizeInBits() > 64)
+ return false;
+ }
+ return true;
+}
+
bool RISCVInstructionSelector::selectCopy(MachineInstr &MI) const {
Register DstReg = MI.getOperand(0).getReg();
diff --git a/llvm/test/CodeGen/RISCV/GlobalISel/load-store.ll b/llvm/test/CodeGen/RISCV/GlobalISel/load-store.ll
new file mode 100644
index 0000000000000..e44636adc4c4e
--- /dev/null
+++ b/llvm/test/CodeGen/RISCV/GlobalISel/load-store.ll
@@ -0,0 +1,641 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
+; RUN: llc -global-isel -global-isel-abort=1 -mtriple=riscv32 < %s | FileCheck %s --check-prefix=RV32I
+; RUN: llc -global-isel -global-isel-abort=1 -mtriple=riscv64 < %s | FileCheck %s --check-prefix=RV64I
+
+; ============================================================================
+; Basic simm12 addressing
+; ============================================================================
+
+define i32 @load_offset_positive(ptr %base) nounwind {
+; RV32I-LABEL: load_offset_positive:
+; RV32I: # %bb.0:
+; RV32I-NEXT: lw a0, 8(a0)
+; RV32I-NEXT: ret
+;
+; RV64I-LABEL: load_offset_positive:
+; RV64I: # %bb.0:
+; RV64I-NEXT: lw a0, 8(a0)
+; RV64I-NEXT: ret
+ %ptr = getelementptr i8, ptr %base, i32 8
+ %val = load i32, ptr %ptr
+ ret i32 %val
+}
+
+define i32 @load_offset_negative(ptr %base) nounwind {
+; RV32I-LABEL: load_offset_negative:
+; RV32I: # %bb.0:
+; RV32I-NEXT: lw a0, -8(a0)
+; RV32I-NEXT: ret
+;
+; RV64I-LABEL: load_offset_negative:
+; RV64I: # %bb.0:
+; RV64I-NEXT: lw a0, -8(a0)
+; RV64I-NEXT: ret
+ %ptr = getelementptr i8, ptr %base, i32 -8
+ %val = load i32, ptr %ptr
+ ret i32 %val
+}
+
+define i32 @load_offset_max(ptr %base) nounwind {
+; RV32I-LABEL: load_offset_max:
+; RV32I: # %bb.0:
+; RV32I-NEXT: lw a0, 2047(a0)
+; RV32I-NEXT: ret
+;
+; RV64I-LABEL: load_offset_max:
+; RV64I: # %bb.0:
+; RV64I-NEXT: lw a0, 2047(a0)
+; RV64I-NEXT: ret
+ %ptr = getelementptr i8, ptr %base, i32 2047
+ %val = load i32, ptr %ptr
+ ret i32 %val
+}
+
+define i32 @load_offset_min(ptr %base) nounwind {
+; RV32I-LABEL: load_offset_min:
+; RV32I: # %bb.0:
+; RV32I-NEXT: lw a0, -2048(a0)
+; RV32I-NEXT: ret
+;
+; RV64I-LABEL: load_offset_min:
+; RV64I: # %bb.0:
+; RV64I-NEXT: lw a0, -2048(a0)
+; RV64I-NEXT: ret
+ %ptr = getelementptr i8, ptr %base, i32 -2048
+ %val = load i32, ptr %ptr
+ ret i32 %val
+}
+
+define void @store_offset_positive(ptr %base, i32 %val) nounwind {
+; RV32I-LABEL: store_offset_positive:
+; RV32I: # %bb.0:
+; RV32I-NEXT: sw a1, 8(a0)
+; RV32I-NEXT: ret
+;
+; RV64I-LABEL: store_offset_positive:
+; RV64I: # %bb.0:
+; RV64I-NEXT: sw a1, 8(a0)
+; RV64I-NEXT: ret
+ %ptr = getelementptr i8, ptr %base, i32 8
+ store i32 %val, ptr %ptr
+ ret void
+}
+
+define void @store_offset_negative(ptr %base, i32 %val) nounwind {
+; RV32I-LABEL: store_offset_negative:
+; RV32I: # %bb.0:
+; RV32I-NEXT: sw a1, -8(a0)
+; RV32I-NEXT: ret
+;
+; RV64I-LABEL: store_offset_negative:
+; RV64I: # %bb.0:
+; RV64I-NEXT: sw a1, -8(a0)
+; RV64I-NEXT: ret
+ %ptr = getelementptr i8, ptr %base, i32 -8
+ store i32 %val, ptr %ptr
+ ret void
+}
+
+; ============================================================================
+; AddiPair: positive offsets in [-4096, 4094]
+; ============================================================================
+
+define i32 @load_addi_pair_positive(ptr %base) nounwind {
+; RV32I-LABEL: load_addi_pair_positive:
+; RV32I: # %bb.0:
+; RV32I-NEXT: addi a0, a0, 2047
+; RV32I-NEXT: lw a0, 953(a0)
+; RV32I-NEXT: ret
+;
+; RV64I-LABEL: load_addi_pair_positive:
+; RV64I: # %bb.0:
+; RV64I-NEXT: addi a0, a0, 2047
+; RV64I-NEXT: lw a0, 953(a0)
+; RV64I-NEXT: ret
+ %ptr = getelementptr i8, ptr %base, i32 3000
+ %val = load i32, ptr %ptr
+ ret i32 %val
+}
+
+define void @store_addi_pair_positive(ptr %base, i32 %val) nounwind {
+; RV32I-LABEL: store_addi_pair_positive:
+; RV32I: # %bb.0:
+; RV32I-NEXT: addi a0, a0, 2047
+; RV32I-NEXT: sw a1, 953(a0)
+; RV32I-NEXT: ret
+;
+; RV64I-LABEL: store_addi_pair_positive:
+; RV64I: # %bb.0:
+; RV64I-NEXT: addi a0, a0, 2047
+; RV64I-NEXT: sw a1, 953(a0)
+; RV64I-NEXT: ret
+ %ptr = getelementptr i8, ptr %base, i32 3000
+ store i32 %val, ptr %ptr
+ ret void
+}
+
+define i32 @load_addi_pair_positive_max(ptr %base) nounwind {
+; RV32I-LABEL: load_addi_pair_positive_max:
+; RV32I: # %bb.0:
+; RV32I-NEXT: addi a0, a0, 2047
+; RV32I-NEXT: lw a0, 2047(a0)
+; RV32I-NEXT: ret
+;
+; RV64I-LABEL: load_addi_pair_positive_max:
+; RV64I: # %bb.0:
+; RV64I-NEXT: addi a0, a0, 2047
+; RV64I-NEXT: lw a0, 2047(a0)
+; RV64I-NEXT: ret
+ %ptr = getelementptr i8, ptr %base, i32 4094
+ %val = load i32, ptr %ptr
+ ret i32 %val
+}
+
+define i32 @load_addi_pair_positive_min(ptr %base) nounwind {
+; RV32I-LABEL: load_addi_pair_positive_min:
+; RV32I: # %bb.0:
+; RV32I-NEXT: addi a0, a0, 2047
+; RV32I-NEXT: lw a0, 1(a0)
+; RV32I-NEXT: ret
+;
+; RV64I-LABEL: load_addi_pair_positive_min:
+; RV64I: # %bb.0:
+; RV64I-NEXT: addi a0, a0, 2047
+; RV64I-NEXT: lw a0, 1(a0)
+; RV64I-NEXT: ret
+ %ptr = getelementptr i8, ptr %base, i32 2048
+ %val = load i32, ptr %ptr
+ ret i32 %val
+}
+
+; ============================================================================
+; AddiPair: negative offsets in [-4096, -2049]
+; ============================================================================
+
+define i32 @load_addi_pair_negative(ptr %base) nounwind {
+; RV32I-LABEL: load_addi_pair_negative:
+; RV32I: # %bb.0:
+; RV32I-NEXT: addi a0, a0, -2048
+; RV32I-NEXT: lw a0, -952(a0)
+; RV32I-NEXT: ret
+;
+; RV64I-LABEL: load_addi_pair_negative:
+; RV64I: # %bb.0:
+; RV64I-NEXT: addi a0, a0, -2048
+; RV64I-NEXT: lw a0, -952(a0)
+; RV64I-NEXT: ret
+ %ptr = getelementptr i8, ptr %base, i32 -3000
+ %val = load i32, ptr %ptr
+ ret i32 %val
+}
+
+define void @store_addi_pair_negative(ptr %base, i32 %val) nounwind {
+; RV32I-LABEL: store_addi_pair_negative:
+; RV32I: # %bb.0:
+; RV32I-NEXT: addi a0, a0, -2048
+; RV32I-NEXT: sw a1, -952(a0)
+; RV32I-NEXT: ret
+;
+; RV64I-LABEL: store_addi_pair_negative:
+; RV64I: # %bb.0:
+; RV64I-NEXT: addi a0, a0, -2048
+; RV64I-NEXT: sw a1, -952(a0)
+; RV64I-NEXT: ret
+ %ptr = getelementptr i8, ptr %base, i32 -3000
+ store i32 %val, ptr %ptr
+ ret void
+}
+
+define i32 @load_addi_pair_negative_min(ptr %base) nounwind {
+; RV32I-LABEL: load_addi_pair_negative_min:
+; RV32I: # %bb.0:
+; RV32I-NEXT: addi a0, a0, -2048
+; RV32I-NEXT: lw a0, -2048(a0)
+; RV32I-NEXT: ret
+;
+; RV64I-LABEL: load_addi_pair_negative_min:
+; RV64I: # %bb.0:
+; RV64I-NEXT: addi a0, a0, -2048
+; RV64I-NEXT: lw a0, -2048(a0)
+; RV64I-NEXT: ret
+ %ptr = getelementptr i8, ptr %base, i32 -4096
+ %val = load i32, ptr %ptr
+ ret i32 %val
+}
+
+define i32 @load_addi_pair_negative_max(ptr %base) nounwind {
+; RV32I-LABEL: load_addi_pair_negative_max:
+; RV32I: # %bb.0:
+; RV32I-NEXT: addi a0, a0, -2048
+; RV32I-NEXT: lw a0, -1(a0)
+; RV32I-NEXT: ret
+;
+; RV64I-LABEL: load_addi_pair_negative_max:
+; RV64I: # %bb.0:
+; RV64I-NEXT: addi a0, a0, -2048
+; RV64I-NEXT: lw a0, -1(a0)
+; RV64I-NEXT: ret
+ %ptr = getelementptr i8, ptr %base, i32 -2049
+ %val = load i32, ptr %ptr
+ ret i32 %val
+}
+
+; ============================================================================
+; Just outside the AddiPair range.
+;
+; These should not enter the simple:
+;
+; [-4096, 4094]
+;
+; ADDI + load/store-offset path.
+; ============================================================================
+
+define i32 @load_offset_above_addi_pair_range(ptr %base) nounwind {
+; RV32I-LABEL: load_offset_above_addi_pair_range:
+; RV32I: # %bb.0:
+; RV32I-NEXT: lui a1, 1
+; RV32I-NEXT: add a0, a0, a1
+; RV32I-NEXT: lw a0, -1(a0)
+; RV32I-NEXT: ret
+;
+; RV64I-LABEL: load_offset_above_addi_pair_range:
+; RV64I: # %bb.0:
+; RV64I-NEXT: lui a1, 1
+; RV64I-NEXT: add a0, a0, a1
+; RV64I-NEXT: lw a0, -1(a0)
+; RV64I-NEXT: ret
+ %ptr = getelementptr i8, ptr %base, i32 4095
+ %val = load i32, ptr %ptr
+ ret i32 %val
+}
+
+define i32 @load_offset_below_addi_pair_range(ptr %base) nounwind {
+; RV32I-LABEL: load_offset_below_addi_pair_range:
+; RV32I: # %bb.0:
+; RV32I-NEXT: lui a1, 1048575
+; RV32I-NEXT: add a0, a0, a1
+; RV32I-NEXT: lw a0, -1(a0)
+; RV32I-NEXT: ret
+;
+; RV64I-LABEL: load_offset_below_addi_pair_range:
+; RV64I: # %bb.0:
+; RV64I-NEXT: lui a1, 1048575
+; RV64I-NEXT: add a0, a0, a1
+; RV64I-NEXT: lw a0, -1(a0)
+; RV64I-NEXT: ret
+ %ptr = getelementptr i8, ptr %base, i32 -4097
+ %val = load i32, ptr %ptr
+ ret i32 %val
+}
+
+; ============================================================================
+; Large constant address folding.
+;
+; The address is only used by the load/store, so the ADD can be folded and
+; computeConstAddr() may materialize the high part while folding the low
+; part into the memory instruction.
+; ============================================================================
+
+define i32 @load_large_offset(ptr %base) nounwind {
+; RV32I-LABEL: load_large_offset:
+; RV32I: # %bb.0:
+; RV32I-NEXT: lui a1, 24
+; RV32I-NEXT: add a0, a0, a1
+; RV32I-NEXT: lw a0, 1696(a0)
+; RV32I-NEXT: ret
+;
+; RV64I-LABEL: load_large_offset:
+; RV64I: # %bb.0:
+; RV64I-NEXT: lui a1, 24
+; RV64I-NEXT: add a0, a0, a1
+; RV64I-NEXT: lw a0, 1696(a0)
+; RV64I-NEXT: ret
+ %ptr = getelementptr i8, ptr %base, i32 100000
+ %val = load i32, ptr %ptr
+ ret i32 %val
+}
+
+define void @store_large_offset(ptr %base, i32 %val) nounwind {
+; RV32I-LABEL: store_large_offset:
+; RV32I: # %bb.0:
+; RV32I-NEXT: lui a2, 24
+; RV32I-NEXT: add a0, a0, a2
+; RV32I-NEXT: sw a1, 1696(a0)
+; RV32I-NEXT: ret
+;
+; RV64I-LABEL: store_large_offset:
+; RV64I: # %bb.0:
+; RV64I-NEXT: lui a2, 24
+; RV64I-NEXT: add a0, a0, a2
+; RV64I-NEXT: sw a1, 1696(a0)
+; RV64I-NEXT: ret
+ %ptr = getelementptr i8, ptr %base, i32 100000
+ store i32 %val, ptr %ptr
+ ret void
+}
+
+define i32 @load_large_negative_offset(ptr %base) nounwind {
+; RV32I-LABEL: load_large_negative_offset:
+; RV32I: # %bb.0:
+; RV32I-NEXT: lui a1, 1048552
+; RV32I-NEXT: add a0, a0, a1
+; RV32I-NEXT: lw a0, -1696(a0)
+; RV32I-NEXT: ret
+;
+; RV64I-LABEL: load_large_negative_offset:
+; RV64I: # %bb.0:
+; RV64I-NEXT: lui a1, 1048552
+; RV64I-NEXT: add a0, a0, a1
+; RV64I-NEXT: lw a0, -1696(a0)
+; RV64I-NEXT: ret
+ %ptr = getelementptr i8, ptr %base, i32 -100000
+ %val = load i32, ptr %ptr
+ ret i32 %val
+}
+
+; ============================================================================
+; Large constant with a non-memory user.
+;
+; The computed address must remain available for the call, so the ADD should
+; not be folded solely into the load.
+; ============================================================================
+
+declare void @use_pointer(ptr)
+
+define i32 @load_large_offset_multi_use(ptr %base) nounwind {
+; RV32I-LABEL: load_large_offset_multi_use:
+; RV32I: # %bb.0:
+; RV32I-NEXT: addi sp, sp, -16
+; RV32I-NEXT: sw ra, 12(sp) # 4-byte Folded Spill
+; RV32I-NEXT: sw s0, 8(sp) # 4-byte Folded Spill
+; RV32I-NEXT: lui a1, 24
+; RV32I-NEXT: addi a1, a1, 1696
+; RV32I-NEXT: add a0, a0, a1
+; RV32I-NEXT: lw s0, 0(a0)
+; RV32I-NEXT: call use_pointer
+; RV32I-NEXT: mv a0, s0
+; RV32I-NEXT: lw ra, 12(sp) # 4-byte Folded Reload
+; RV32I-NEXT: lw s0, 8(sp) # 4-byte Folded Reload
+; RV32I-NEXT: addi sp, sp, 16
+; RV32I-NEXT: ret
+;
+; RV64I-LABEL: load_large_offset_multi_use:
+; RV64I: # %bb.0:
+; RV64I-NEXT: addi sp, sp, -16
+; RV64I-NEXT: sd ra, 8(sp) # 8-byte Folded Spill
+; RV64I-NEXT: sd s0, 0(sp) # 8-byte Folded Spill
+; RV64I-NEXT: lui a1, 24
+; RV64I-NEXT: addi a1, a1, 1696
+; RV64I-NEXT: add a0, a0, a1
+; RV64I-NEXT: lw s0, 0(a0)
+; RV64I-NEXT: call use_pointer
+; RV64I-NEXT: mv a0, s0
+; RV64I-NEXT: ld ra, 8(sp) # 8-byte Folded Reload
+; RV64I-NEXT: ld s0, 0(sp) # 8-byte Folded Reload
+; RV64I-NEXT: addi sp, sp, 16
+; RV64I-NEXT: ret
+ %ptr = getelementptr i8, ptr %base, i32 100000
+ %val = load i32, ptr %ptr
+ call void @use_pointer(ptr %ptr)
+ ret i32 %val
+}
+
+define void @store_large_offset_multi_use(ptr %base, i32 %val) nounwind {
+; RV32I-LABEL: store_large_offset_multi_use:
+; RV32I: # %bb.0:
+; RV32I-NEXT: addi sp, sp, -16
+; RV32I-NEXT: sw ra, 12(sp) # 4-byte Folded Spill
+; RV32I-NEXT: lui a2, 24
+; RV32I-NEXT: addi a2, a2, 1696
+; RV32I-NEXT: add a0, a0, a2
+; RV32I-NEXT: sw a1, 0(a0)
+; RV32I-NEXT: call use_pointer
+; RV32I-NEXT: lw ra, 12(sp) # 4-byte Folded Reload
+; RV32I-NEXT: addi sp, sp, 16
+; RV32I-NEXT: ret
+;
+; RV64I-LABEL: store_large_offset_multi_use:
+; RV64I: # %bb.0:
+; RV64I-NEXT: addi sp, sp, -16
+; RV64I-NEXT: sd ra, 8(sp) # 8-byte Folded Spill
+; RV64I-NEXT: lui a2, 24
+; RV64I-NEXT: addi a2, a2, 1696
+; RV64I-NEXT: add a0, a0, a2
+; RV64I-NEXT: sw a1, 0(a0)
+; RV64I-NEXT: call use_pointer
+; RV64I-NEXT: ld ra, 8(sp) # 8-byte Folded Reload
+; RV64I-NEXT: addi sp, sp, 16
+; RV64I-NEXT: ret
+ %ptr = getelementptr i8, ptr %base, i32 100000
+ store i32 %val, ptr %ptr
+ call void @use_pointer(ptr %ptr)
+ ret void
+}
+
+; ============================================================================
+; The address has multiple load/store users.
+;
+; Folding the address into both memory operations should still be possible
+; when the address itself has no non-memory users.
+; ============================================================================
+
+define i32 @load_large_offset_two_loads(ptr %base) nounwind {
+; RV32I-LABEL: load_large_offset_two_loads:
+; RV32I: # %bb.0:
+; RV32I-NEXT: lui a1, 24
+; RV32I-NEXT: add a0, a0, a1
+; RV32I-NEXT: lw a1, 1696(a0)
+; RV32I-NEXT: lw a0, 1696(a0)
+; RV32I-NEXT: add a0, a1, a0
+; RV32I-NEXT: ret
+;
+; RV64I-LABEL: load_large_offset_two_loads:
+; RV64I: # %bb.0:
+; RV64I-NEXT: lui a1, 24
+; RV64I-NEXT: add a0, a0, a1
+; RV64I-NEXT: lw a1, 1696(a0)
+; RV64I-NEXT: lw a0, 1696(a0)
+; RV64I-NEXT: addw a0, a1, a0
+; RV64I-NEXT: ret
+ %ptr = getelementptr i8, ptr %base, i32 100000
+ %v1 = load i32, ptr %ptr
+ %v2 = load i32, ptr %ptr
+ %sum = add i32 %v1, %v2
+ ret i32 %sum
+}
+
+define void @store_large_offset_two_stores(ptr %base, i32 %v1, i32 %v2) nounwind {
+; RV32I-LABEL: store_large_offset_two_stores:
+; RV32I: # %bb.0:
+; RV32I-NEXT: lui a3, 24
+; RV32I-NEXT: add a0, a0, a3
+; RV32I-NEXT: sw a1, 1696(a0)
+; RV32I-NEXT: sw a2, 1696(a0)
+; RV32I-NEXT: ret
+;
+; RV64I-LABEL: store_large_offset_two_stores:
+; RV64I: # %bb.0:
+; RV64I-NEXT: lui a3, 24
+; RV64I-NEXT: add a0, a0, a3
+; RV64I-NEXT: sw a1, 1696(a0)
+; RV64I-NEXT: sw a2, 1696(a0)
+; RV64I-NEXT: ret
+ %ptr = getelementptr i8, ptr %base, i32 100000
+ store i32 %v1, ptr %ptr
+ store i32 %v2, ptr %ptr
+ ret void
+}
+
+; ============================================================================
+; Bare constant address.
+; ============================================================================
+
+define i32 @load_absolute_address() nounwind {
+; RV32I-LABEL: load_absolute_address:
+; RV32I: # %bb.0:
+; RV32I-NEXT: lui a0, 74564
+; RV32I-NEXT: lw a0, -1024(a0)
+; RV32I-NEXT: ret
+;
+; RV64I-LABEL: load_absolute_address:
+; RV64I: # %bb.0:
+; RV64I-NEXT: lui a0, 74564
+; RV64I-NEXT: lw a0, -1024(a0)
+; RV64I-NEXT: ret
+ %ptr = inttoptr i64 305413120 to ptr
+ %val = load i32, ptr %ptr
+ ret i32 %val
+}
+
+define i32 @load_absolute_small_address() nounwind {
+; RV32I-LABEL: load_absolute_small_address:
+; RV32I: # %bb.0:
+; RV32I-NEXT: lui a0, 1
+; RV32I-NEXT: lw a0, 0(a0)
+; RV32I-NEXT: ret
+;
+; RV64I-LABEL: load_absolute_small_address:
+; RV64I: # %bb.0:
+; RV64I-NEXT: lui a0, 1
+; RV64I-NEXT: lw a0, 0(a0)
+; RV64I-NEXT: ret
+ %ptr = inttoptr i64 4096 to ptr
+ %val = load i32, ptr %ptr
+ ret i32 %val
+}
+
+; ============================================================================
+; Absolute address used by a store.
+; ============================================================================
+
+define void @store_absolute_address(i32 %val) nounwind {
+; RV32I-LABEL: store_absolute_address:
+; RV32I: # %bb.0:
+; RV32I-NEXT: lui a1, 74564
+; RV32I-NEXT: sw a0, -1024(a1)
+; RV32I-NEXT: ret
+;
+; RV64I-LABEL: store_absolute_address:
+; RV64I: # %bb.0:
+; RV64I-NEXT: lui a1, 74564
+; RV64I-NEXT: sw a0, -1024(a1)
+; RV64I-NEXT: ret
+ %ptr = inttoptr i64 305413120 to ptr
+ store i32 %val, ptr %ptr
+ ret void
+}
+
+; ============================================================================
+; Large constants whose low 12 bits require signed-immediate handling.
+;
+; These provide additional coverage for computeConstAddr() rather than
+; relying only on 100000.
+; ============================================================================
+
+define i32 @load_large_offset_lo_positive(ptr %base) nounwind {
+; RV32I-LABEL: load_large_offset_lo_positive:
+; RV32I: # %bb.0:
+; RV32I-NEXT: lui a1, 24
+; RV32I-NEXT: add a0, a0, a1
+; RV32I-NEXT: lw a0, 0(a0)
+; RV32I-NEXT: ret
+;
+; RV64I-LABEL: load_large_offset_lo_positive:
+; RV64I: # %bb.0:
+; RV64I-NEXT: lui a1, 24
+; RV64I-NEXT: add a0, a0, a1
+; RV64I-NEXT: lw a0, 0(a0)
+; RV64I-NEXT: ret
+ %ptr = getelementptr i8, ptr %base, i32 98304
+ %val = load i32, ptr %ptr
+ ret i32 %val
+}
+
+define i32 @load_large_offset_lo_negative(ptr %base) nounwind {
+; RV32I-LABEL: load_large_offset_lo_negative:
+; RV32I: # %bb.0:
+; RV32I-NEXT: lui a1, 24
+; RV32I-NEXT: add a0, a0, a1
+; RV32I-NEXT: lw a0, 1(a0)
+; RV32I-NEXT: ret
+;
+; RV64I-LABEL: load_large_offset_lo_negative:
+; RV64I: # %bb.0:
+; RV64I-NEXT: lui a1, 24
+; RV64I-NEXT: add a0, a0, a1
+; RV64I-NEXT: lw a0, 1(a0)
+; RV64I-NEXT: ret
+ %ptr = getelementptr i8, ptr %base, i32 98305
+ %val = load i32, ptr %ptr
+ ret i32 %val
+}
+
+define i32 @load_large_offset_cross_signed_boundary(ptr %base) nounwind {
+; RV32I-LABEL: load_large_offset_cross_signed_boundary:
+; RV32I: # %bb.0:
+; RV32I-NEXT: lui a1, 25
+; RV32I-NEXT: add a0, a0, a1
+; RV32I-NEXT: lw a0, -2048(a0)
+; RV32I-NEXT: ret
+;
+; RV64I-LABEL: load_large_offset_cross_signed_boundary:
+; RV64I: # %bb.0:
+; RV64I-NEXT: lui a1, 25
+; RV64I-NEXT: add a0, a0, a1
+; RV64I-NEXT: lw a0, -2048(a0)
+; RV64I-NEXT: ret
+ %ptr = getelementptr i8, ptr %base, i32 100352
+ %val = load i32, ptr %ptr
+ ret i32 %val
+}
+
+; ============================================================================
+; Address reused by memory operations and a pointer comparison.
+;
+; This ensures the address has a non-memory use other than a call.
+; ============================================================================
+
+define i1 @compare_large_offset_address(ptr %base, ptr %other) nounwind {
+; RV32I-LABEL: compare_large_offset_address:
+; RV32I: # %bb.0:
+; RV32I-NEXT: lui a2, 24
+; RV32I-NEXT: addi a2, a2, 1696
+; RV32I-NEXT: add a0, a0, a2
+; RV32I-NEXT: xor a0, a0, a1
+; RV32I-NEXT: seqz a0, a0
+; RV32I-NEXT: ret
+;
+; RV64I-LABEL: compare_large_offset_address:
+; RV64I: # %bb.0:
+; RV64I-NEXT: lui a2, 24
+; RV64I-NEXT: addi a2, a2, 1696
+; RV64I-NEXT: add a0, a0, a2
+; RV64I-NEXT: xor a0, a0, a1
+; RV64I-NEXT: seqz a0, a0
+; RV64I-NEXT: ret
+ %ptr = getelementptr i8, ptr %base, i32 100000
+ %val = load i32, ptr %ptr
+ %cmp = icmp eq ptr %ptr, %other
+ %unused = add i32 %val, 0
+ ret i1 %cmp
+}
>From f9c43aa8b4d00b98c87c2d17f89b2b646aff05a7 Mon Sep 17 00:00:00 2001
From: Kane Wang <wangqiang1 at kylinos.cn>
Date: Tue, 1 Sep 2026 14:46:38 +0800
Subject: [PATCH 2/3] [RISCV][GlobalISel] Use mi_match in selectAddrRegImm
(NFC)
Replace direct getVRegDef calls with mi_matchi.
---
.../RISCV/GISel/RISCVInstructionSelector.cpp | 45 +++++++++----------
1 file changed, 21 insertions(+), 24 deletions(-)
diff --git a/llvm/lib/Target/RISCV/GISel/RISCVInstructionSelector.cpp b/llvm/lib/Target/RISCV/GISel/RISCVInstructionSelector.cpp
index d3d98d92aac44..62107beffb0cc 100644
--- a/llvm/lib/Target/RISCV/GISel/RISCVInstructionSelector.cpp
+++ b/llvm/lib/Target/RISCV/GISel/RISCVInstructionSelector.cpp
@@ -596,29 +596,30 @@ RISCVInstructionSelector::selectAddrRegImm(MachineOperand &Root) const {
if (!Root.isReg())
return std::nullopt;
- MachineInstr *RootDef = MRI->getVRegDef(Root.getReg());
- if (RootDef->getOpcode() == TargetOpcode::G_FRAME_INDEX) {
+ Register RootReg = Root.getReg();
+
+ // Frame index.
+ int FI;
+ if (mi_match(RootReg, *MRI, m_GFrameIndex(FI))) {
return {{
- [=](MachineInstrBuilder &MIB) { MIB.add(RootDef->getOperand(1)); },
+ [=](MachineInstrBuilder &MIB) { MIB.addFrameIndex(FI); },
[=](MachineInstrBuilder &MIB) { MIB.addImm(0); },
}};
}
- if (isBaseWithConstantOffset(Root, *MRI)) {
- MachineOperand &LHS = RootDef->getOperand(1);
- MachineOperand &RHS = RootDef->getOperand(2);
- MachineInstr *LHSDef = MRI->getVRegDef(LHS.getReg());
- MachineInstr *RHSDef = MRI->getVRegDef(RHS.getReg());
-
- int64_t RHSC = RHSDef->getOperand(1).getCImm()->getSExtValue();
+ // base + constant offset (G_PTR_ADD).
+ Register BaseReg;
+ int64_t RHSC;
+ if (mi_match(RootReg, *MRI, m_GPtrAdd(m_Reg(BaseReg), m_ICst(RHSC)))) {
if (isInt<12>(RHSC)) {
- if (LHSDef->getOpcode() == TargetOpcode::G_FRAME_INDEX)
+ int BaseFI;
+ if (mi_match(BaseReg, *MRI, m_GFrameIndex(BaseFI)))
return {{
- [=](MachineInstrBuilder &MIB) { MIB.add(LHSDef->getOperand(1)); },
+ [=](MachineInstrBuilder &MIB) { MIB.addFrameIndex(BaseFI); },
[=](MachineInstrBuilder &MIB) { MIB.addImm(RHSC); },
}};
- return {{[=](MachineInstrBuilder &MIB) { MIB.add(LHS); },
+ return {{[=](MachineInstrBuilder &MIB) { MIB.addReg(BaseReg); },
[=](MachineInstrBuilder &MIB) { MIB.addImm(RHSC); }}};
}
@@ -626,28 +627,24 @@ RISCVInstructionSelector::selectAddrRegImm(MachineOperand &Root) const {
// constant can be split across an ADDI and the load/store offset.
if (RHSC >= -4096 && RHSC <= 4094) {
int64_t Adj = RHSC < 0 ? -2048 : 2047;
- return renderAddiPair(LHS.getReg(), Adj, RHSC - Adj);
+ return renderAddiPair(BaseReg, Adj, RHSC - Adj);
}
- if (isWorthFoldingAdd(Root.getReg()))
- if (auto Fns = computeConstAddr(RHSC, /*IsPrefetch=*/false, LHS.getReg()))
+ if (isWorthFoldingAdd(RootReg))
+ if (auto Fns = computeConstAddr(RHSC, /*IsPrefetch=*/false, BaseReg))
return Fns;
}
// Bare constant address. IRTranslator lowers inttoptr(C) to
// G_INTTOPTR(G_CONSTANT); look through it to reach the constant.
- if (RootDef->getOpcode() == TargetOpcode::G_INTTOPTR) {
- MachineInstr *SrcDef = MRI->getVRegDef(RootDef->getOperand(1).getReg());
- if (SrcDef && SrcDef->getOpcode() == TargetOpcode::G_CONSTANT)
- RootDef = SrcDef;
- }
- if (RootDef->getOpcode() == TargetOpcode::G_CONSTANT) {
- int64_t CVal = RootDef->getOperand(1).getCImm()->getSExtValue();
+ int64_t CVal;
+ if (mi_match(RootReg, *MRI, m_GIntToPtr(m_ICst(CVal))) ||
+ mi_match(RootReg, *MRI, m_ICst(CVal))) {
if (auto Fns = computeConstAddr(CVal, /*IsPrefetch=*/false, Register()))
return Fns;
}
- return {{[=](MachineInstrBuilder &MIB) { MIB.addReg(Root.getReg()); },
+ return {{[=](MachineInstrBuilder &MIB) { MIB.addReg(RootReg); },
[=](MachineInstrBuilder &MIB) { MIB.addImm(0); }}};
}
>From 33f12859300c233e8335f62a6f1ed47301767ba1 Mon Sep 17 00:00:00 2001
From: Kane Wang <wangqiang1 at kylinos.cn>
Date: Tue, 1 Sep 2026 15:30:41 +0800
Subject: [PATCH 3/3] [RISCV][GlobalISel] Use mi_match in
selectAddrRegImmLsb00000 (NFC)
---
.../RISCV/GISel/RISCVInstructionSelector.cpp | 45 +++++++++----------
1 file changed, 21 insertions(+), 24 deletions(-)
diff --git a/llvm/lib/Target/RISCV/GISel/RISCVInstructionSelector.cpp b/llvm/lib/Target/RISCV/GISel/RISCVInstructionSelector.cpp
index 62107beffb0cc..a1158c5322bb4 100644
--- a/llvm/lib/Target/RISCV/GISel/RISCVInstructionSelector.cpp
+++ b/llvm/lib/Target/RISCV/GISel/RISCVInstructionSelector.cpp
@@ -653,63 +653,60 @@ RISCVInstructionSelector::selectAddrRegImmLsb00000(MachineOperand &Root) const {
if (!Root.isReg())
return std::nullopt;
- MachineInstr *RootDef = MRI->getVRegDef(Root.getReg());
- if (RootDef->getOpcode() == TargetOpcode::G_FRAME_INDEX) {
+ Register RootReg = Root.getReg();
+
+ // Frame index.
+ int FI;
+ if (mi_match(RootReg, *MRI, m_GFrameIndex(FI))) {
return {{
- [=](MachineInstrBuilder &MIB) { MIB.add(RootDef->getOperand(1)); },
+ [=](MachineInstrBuilder &MIB) { MIB.addFrameIndex(FI); },
[=](MachineInstrBuilder &MIB) { MIB.addImm(0); },
}};
}
- if (isBaseWithConstantOffset(Root, *MRI)) {
- MachineOperand &LHS = RootDef->getOperand(1);
- MachineOperand &RHS = RootDef->getOperand(2);
- MachineInstr *LHSDef = MRI->getVRegDef(LHS.getReg());
- MachineInstr *RHSDef = MRI->getVRegDef(RHS.getReg());
- int64_t RHSC = RHSDef->getOperand(1).getCImm()->getSExtValue();
-
+ // base + constant offset (G_PTR_ADD).
+ Register BaseReg;
+ int64_t RHSC;
+ if (mi_match(RootReg, *MRI, m_GPtrAdd(m_Reg(BaseReg), m_ICst(RHSC)))) {
if (isInt<12>(RHSC)) {
// Not a multiple of 32: can't encode, use the address as-is.
if ((RHSC & 0b11111) != 0) {
- return {{[=](MachineInstrBuilder &MIB) { MIB.addReg(Root.getReg()); },
+ return {{[=](MachineInstrBuilder &MIB) { MIB.addReg(RootReg); },
[=](MachineInstrBuilder &MIB) { MIB.addImm(0); }}};
}
// Fold the offset.
- if (LHSDef->getOpcode() == TargetOpcode::G_FRAME_INDEX)
+ int BaseFI;
+ if (mi_match(BaseReg, *MRI, m_GFrameIndex(BaseFI)))
return {{
- [=](MachineInstrBuilder &MIB) { MIB.add(LHSDef->getOperand(1)); },
+ [=](MachineInstrBuilder &MIB) { MIB.addFrameIndex(BaseFI); },
[=](MachineInstrBuilder &MIB) { MIB.addImm(RHSC); },
}};
- return {{[=](MachineInstrBuilder &MIB) { MIB.add(LHS); },
+ return {{[=](MachineInstrBuilder &MIB) { MIB.addReg(BaseReg); },
[=](MachineInstrBuilder &MIB) { MIB.addImm(RHSC); }}};
}
// Large constant: fold a -2048/2016 adjustment to save an instruction.
if ((-2049 >= RHSC && RHSC >= -4096) || (4063 >= RHSC && RHSC >= 2017)) {
int64_t Adj = RHSC < 0 ? -2048 : 2016;
- return renderAddiPair(LHS.getReg(), RHSC - Adj, Adj);
+ return renderAddiPair(BaseReg, RHSC - Adj, Adj);
}
// Otherwise split the constant into Hi (materialized + added to the base)
// and Lo12 (folded offset).
- if (auto Fns = computeConstAddr(RHSC, /*IsPrefetch=*/true, LHS.getReg()))
+ if (auto Fns = computeConstAddr(RHSC, /*IsPrefetch=*/true, BaseReg))
return Fns;
}
// Bare constant address. IRTranslator emits inttoptr(C) as
// G_INTTOPTR(G_CONSTANT); look through the G_INTTOPTR to reach the constant.
- if (RootDef->getOpcode() == TargetOpcode::G_INTTOPTR) {
- MachineInstr *SrcDef = MRI->getVRegDef(RootDef->getOperand(1).getReg());
- if (SrcDef->getOpcode() == TargetOpcode::G_CONSTANT)
- RootDef = SrcDef;
- }
- if (RootDef->getOpcode() == TargetOpcode::G_CONSTANT) {
- int64_t CVal = RootDef->getOperand(1).getCImm()->getSExtValue();
+ int64_t CVal;
+ if (mi_match(RootReg, *MRI, m_GIntToPtr(m_ICst(CVal))) ||
+ mi_match(RootReg, *MRI, m_ICst(CVal))) {
if (auto Fns = computeConstAddr(CVal, /*IsPrefetch=*/true, Register()))
return Fns;
}
- return {{[=](MachineInstrBuilder &MIB) { MIB.addReg(Root.getReg()); },
+ return {{[=](MachineInstrBuilder &MIB) { MIB.addReg(RootReg); },
[=](MachineInstrBuilder &MIB) { MIB.addImm(0); }}};
}
More information about the llvm-commits
mailing list