[llvm] [RegAlloc] Change the computation of CSRCost (PR #177226)
via llvm-commits
llvm-commits at lists.llvm.org
Fri Jan 30 14:58:50 PST 2026
https://github.com/weiguozhi updated https://github.com/llvm/llvm-project/pull/177226
>From 551d8656d751f549c3a12db7b115f737c0db7086 Mon Sep 17 00:00:00 2001
From: Guozhi Wei <carrot at google.com>
Date: Wed, 21 Jan 2026 09:28:04 -0800
Subject: [PATCH 1/4] [RegAlloc] Change the computation of CSRCost
The original computed CSRCost is too small, so the optimization of
spilling instead of CSR usage is rarely triggered.
Also the original cost model is too difficult to be understood and too
hard to be tuned by backend developer and user.
So this patch changes the CSRCost to be
CSRCost = TRI->getCSRFirstUseCost() * EntryFreq * Scale
TRI->getCSRFirstUseCost() is the raw cost of save/restore a CSR. We
don't need to tune this number.
EntryFreq is the BlockFrequency of the entry block.
Scale is used to scale down the CSRCost, because we usually prefer a CSR
register instead of spilling if we have similar CSRCost and spill cost,
so it should be less than 100%. We usually tune this number.
This patch also implements a correct RAGreedy::calcSpillCost() function.
---
.../include/llvm/CodeGen/TargetRegisterInfo.h | 2 +
llvm/lib/CodeGen/RegAllocGreedy.cpp | 100 +++++++++++++-----
llvm/lib/CodeGen/RegAllocGreedy.h | 3 +-
llvm/lib/Target/AArch64/AArch64RegisterInfo.h | 6 +-
llvm/lib/Target/AMDGPU/SIRegisterInfo.h | 2 +-
llvm/lib/Target/RISCV/RISCVRegisterInfo.h | 2 +-
6 files changed, 82 insertions(+), 33 deletions(-)
diff --git a/llvm/include/llvm/CodeGen/TargetRegisterInfo.h b/llvm/include/llvm/CodeGen/TargetRegisterInfo.h
index 35b14e8b8fd30..b69a91651e300 100644
--- a/llvm/include/llvm/CodeGen/TargetRegisterInfo.h
+++ b/llvm/include/llvm/CodeGen/TargetRegisterInfo.h
@@ -1029,6 +1029,8 @@ class LLVM_ABI TargetRegisterInfo : public MCRegisterInfo {
/// the first time. Default value of 0 means we will use a callee-saved
/// register if it is available.
virtual unsigned getCSRFirstUseCost() const { return 0; }
+ /// FIXME: We should deprecate this usage.
+ virtual unsigned getCSRCost() const { return 0; }
/// Returns true if the target requires (and can make use of) the register
/// scavenger.
diff --git a/llvm/lib/CodeGen/RegAllocGreedy.cpp b/llvm/lib/CodeGen/RegAllocGreedy.cpp
index 26149f8137cac..b2c62c29ca8a8 100644
--- a/llvm/lib/CodeGen/RegAllocGreedy.cpp
+++ b/llvm/lib/CodeGen/RegAllocGreedy.cpp
@@ -107,12 +107,18 @@ static cl::opt<bool> ExhaustiveSearch(
"and interference cutoffs of last chance recoloring"),
cl::Hidden);
+// This option should be deprecated!
// FIXME: Find a good default for this flag and remove the flag.
static cl::opt<unsigned>
CSRFirstTimeCost("regalloc-csr-first-time-cost",
cl::desc("Cost for first time use of callee-saved register."),
cl::init(0), cl::Hidden);
+static cl::opt<unsigned>
+CSRCostScale("regalloc-csr-cost-scale",
+ cl::desc("Scale for the callee-saved register cost, in percentage."),
+ cl::init(80), cl::Hidden);
+
static cl::opt<unsigned long> GrowRegionComplexityBudget(
"grow-region-complexity-budget",
cl::desc("growRegion() does not scale with the number of BB edges, so "
@@ -977,9 +983,9 @@ bool RAGreedy::calcCompactRegion(GlobalSplitCandidate &Cand) {
return true;
}
-/// calcSpillCost - Compute how expensive it would be to split the live range in
-/// SA around all use blocks instead of forming bundle regions.
-BlockFrequency RAGreedy::calcSpillCost() {
+/// calcBlockSplitCost - Compute how expensive it would be to split the live
+/// range in SA around all use blocks instead of forming bundle regions.
+BlockFrequency RAGreedy::calcBlockSplitCost() {
BlockFrequency Cost = BlockFrequency(0);
ArrayRef<SplitAnalysis::BlockInfo> UseBlocks = SA->getUseBlocks();
for (const SplitAnalysis::BlockInfo &BI : UseBlocks) {
@@ -1197,7 +1203,7 @@ MCRegister RAGreedy::tryRegionSplit(const LiveInterval &VirtReg,
if (!TRI->shouldRegionSplitForVirtReg(*MF, VirtReg))
return MCRegister::NoRegister;
unsigned NumCands = 0;
- BlockFrequency SpillCost = calcSpillCost();
+ BlockFrequency SpillCost = calcBlockSplitCost();
BlockFrequency BestCost;
// Check if we can split this live range around a compact region.
@@ -2339,6 +2345,31 @@ MCRegister RAGreedy::selectOrSplit(const LiveInterval &VirtReg,
return Reg;
}
+/// calcSpillCost - Compute how expensive it would be to spill the live range in
+/// LI into memory.
+BlockFrequency RAGreedy::calcSpillCost(const LiveInterval &LI) {
+ uint64_t SpillCost = 0;
+ SmallPtrSet<MachineInstr *, 8> Visited;
+
+ for (MachineRegisterInfo::reg_instr_nodbg_iterator
+ I = MRI->reg_instr_nodbg_begin(LI.reg()),
+ E = MRI->reg_instr_nodbg_end();
+ I != E;) {
+ MachineInstr *MI = &*(I++);
+ if (MI->isImplicitDef())
+ continue;
+ if (!Visited.insert(MI).second)
+ continue;
+
+ bool Reads, Writes;
+ std::tie(Reads, Writes) = MI->readsWritesVirtualRegister(LI.reg());
+ auto MBBFreq = SpillPlacer->getBlockFrequency(MI->getParent()->getNumber());
+ SpillCost += (Reads + Writes) * MBBFreq.getFrequency();
+ }
+
+ return BlockFrequency(SpillCost);
+}
+
/// Using a CSR for the first time has a cost because it causes push|pop
/// to be added to prologue|epilogue. Splitting a cold section of the live
/// range can have lower cost than using the CSR for the first time;
@@ -2352,7 +2383,7 @@ MCRegister RAGreedy::tryAssignCSRFirstTime(
// We choose spill over using the CSR for the first time if the spill cost
// is lower than CSRCost.
SA->analyze(&VirtReg);
- if (calcSpillCost() >= CSRCost)
+ if (calcSpillCost(VirtReg) >= CSRCost)
return PhysReg;
// We are going to spill, set CostPerUseLimit to 1 to make sure that
@@ -2385,31 +2416,42 @@ void RAGreedy::aboutToRemoveInterval(const LiveInterval &LI) {
}
void RAGreedy::initializeCSRCost() {
- // We use the command-line option if it is explicitly set, otherwise use the
- // larger one out of the command-line option and the value reported by TRI.
- CSRCost = BlockFrequency(
- CSRFirstTimeCost.getNumOccurrences()
- ? CSRFirstTimeCost
- : std::max((unsigned)CSRFirstTimeCost, TRI->getCSRFirstUseCost()));
- if (!CSRCost.getFrequency())
- return;
-
- // Raw cost is relative to Entry == 2^14; scale it appropriately.
- uint64_t ActualEntry = MBFI->getEntryFreq().getFrequency();
- if (!ActualEntry) {
- CSRCost = BlockFrequency(0);
- return;
+ if (!CSRCostScale.getNumOccurrences() &&
+ (CSRFirstTimeCost.getNumOccurrences() || TRI->getCSRCost())) {
+ // We should deprecate the usage of CSRFirstTimeCost!
+ // We use the command-line option if it is explicitly set, otherwise use the
+ // larger one out of the command-line option and the value reported by TRI.
+ CSRCost = BlockFrequency(
+ CSRFirstTimeCost.getNumOccurrences()
+ ? CSRFirstTimeCost
+ : std::max((unsigned)CSRFirstTimeCost, TRI->getCSRCost()));
+ if (!CSRCost.getFrequency())
+ return;
+
+ // Raw cost is relative to Entry == 2^14; scale it appropriately.
+ uint64_t ActualEntry = MBFI->getEntryFreq().getFrequency();
+ if (!ActualEntry) {
+ CSRCost = BlockFrequency(0);
+ return;
+ }
+ uint64_t FixedEntry = 1 << 14;
+ if (ActualEntry < FixedEntry)
+ CSRCost *= BranchProbability(ActualEntry, FixedEntry);
+ else if (ActualEntry <= UINT32_MAX)
+ // Invert the fraction and divide.
+ CSRCost /= BranchProbability(FixedEntry, ActualEntry);
+ else
+ // Can't use BranchProbability in general, since it takes 32-bit numbers.
+ CSRCost =
+ BlockFrequency(CSRCost.getFrequency() * (ActualEntry / FixedEntry));
+ } else {
+ uint64_t EntryFreq = MBFI->getEntryFreq().getFrequency();
+ CSRCost = BlockFrequency(TRI->getCSRFirstUseCost() * EntryFreq);
+ if (CSRCostScale < 100)
+ CSRCost *= BranchProbability(CSRCostScale, 100);
+ else
+ CSRCost /= BranchProbability(100, CSRCostScale);
}
- uint64_t FixedEntry = 1 << 14;
- if (ActualEntry < FixedEntry)
- CSRCost *= BranchProbability(ActualEntry, FixedEntry);
- else if (ActualEntry <= UINT32_MAX)
- // Invert the fraction and divide.
- CSRCost /= BranchProbability(FixedEntry, ActualEntry);
- else
- // Can't use BranchProbability in general, since it takes 32-bit numbers.
- CSRCost =
- BlockFrequency(CSRCost.getFrequency() * (ActualEntry / FixedEntry));
}
/// Collect the hint info for \p Reg.
diff --git a/llvm/lib/CodeGen/RegAllocGreedy.h b/llvm/lib/CodeGen/RegAllocGreedy.h
index 081ea368f505e..465be0d76809e 100644
--- a/llvm/lib/CodeGen/RegAllocGreedy.h
+++ b/llvm/lib/CodeGen/RegAllocGreedy.h
@@ -312,7 +312,7 @@ class LLVM_LIBRARY_VISIBILITY RAGreedy : public RegAllocBase,
const LiveInterval *dequeue(PQueue &CurQueue);
bool hasVirtRegAlloc();
- BlockFrequency calcSpillCost();
+ BlockFrequency calcBlockSplitCost();
bool addSplitConstraints(InterferenceCache::Cursor, BlockFrequency &);
bool addThroughConstraints(InterferenceCache::Cursor, ArrayRef<unsigned>);
bool growRegion(GlobalSplitCandidate &Cand);
@@ -360,6 +360,7 @@ class LLVM_LIBRARY_VISIBILITY RAGreedy : public RegAllocBase,
AllocationOrder &Order, MCRegister PhysReg,
uint8_t &CostPerUseLimit,
SmallVectorImpl<Register> &NewVRegs);
+ BlockFrequency calcSpillCost(const LiveInterval &LI);
void initializeCSRCost();
MCRegister tryBlockSplit(const LiveInterval &, AllocationOrder &,
SmallVectorImpl<Register> &);
diff --git a/llvm/lib/Target/AArch64/AArch64RegisterInfo.h b/llvm/lib/Target/AArch64/AArch64RegisterInfo.h
index 89d1802ab98d5..ac58d8d6b1cc7 100644
--- a/llvm/lib/Target/AArch64/AArch64RegisterInfo.h
+++ b/llvm/lib/Target/AArch64/AArch64RegisterInfo.h
@@ -53,12 +53,16 @@ class AArch64RegisterInfo final : public AArch64GenRegisterInfo {
const uint32_t *getDarwinCallPreservedMask(const MachineFunction &MF,
CallingConv::ID) const;
- unsigned getCSRFirstUseCost() const override {
+ unsigned getCSRCost() const override {
// The cost will be compared against BlockFrequency where entry has the
// value of 1 << 14. A value of 5 will choose to spill or split really
// cold path instead of using a callee-saved register.
return 5;
}
+ unsigned getCSRFirstUseCost() const override {
+ // The cost of 2 means push and pop for each CSR.
+ return 2;
+ }
const TargetRegisterClass *
getSubClassWithSubReg(const TargetRegisterClass *RC,
diff --git a/llvm/lib/Target/AMDGPU/SIRegisterInfo.h b/llvm/lib/Target/AMDGPU/SIRegisterInfo.h
index 2e2916f68f584..2dd0413eb054a 100644
--- a/llvm/lib/Target/AMDGPU/SIRegisterInfo.h
+++ b/llvm/lib/Target/AMDGPU/SIRegisterInfo.h
@@ -107,7 +107,7 @@ class SIRegisterInfo final : public AMDGPUGenRegisterInfo {
// Stack access is very expensive. CSRs are also the high registers, and we
// want to minimize the number of used registers.
- unsigned getCSRFirstUseCost() const override {
+ unsigned getCSRCost() const override {
return 100;
}
diff --git a/llvm/lib/Target/RISCV/RISCVRegisterInfo.h b/llvm/lib/Target/RISCV/RISCVRegisterInfo.h
index f29f85e4987f6..df8a1b072b21c 100644
--- a/llvm/lib/Target/RISCV/RISCVRegisterInfo.h
+++ b/llvm/lib/Target/RISCV/RISCVRegisterInfo.h
@@ -68,7 +68,7 @@ struct RISCVRegisterInfo : public RISCVGenRegisterInfo {
const uint32_t *getCallPreservedMask(const MachineFunction &MF,
CallingConv::ID) const override;
- unsigned getCSRFirstUseCost() const override {
+ unsigned getCSRCost() const override {
// The cost will be compared against BlockFrequency where entry has the
// value of 1 << 14. A value of 5 will choose to spill or split cold
// path instead of using a callee-saved register.
>From 1cd2a05ea7b0e5d770983e8c4b0674bfe008b346 Mon Sep 17 00:00:00 2001
From: Guozhi Wei <carrot at google.com>
Date: Wed, 21 Jan 2026 11:18:49 -0800
Subject: [PATCH 2/4] Add a test case for the CSR optimization.
---
llvm/test/CodeGen/AArch64/ragreedy-csr2.ll | 131 +++++++++++++++++++++
1 file changed, 131 insertions(+)
create mode 100644 llvm/test/CodeGen/AArch64/ragreedy-csr2.ll
diff --git a/llvm/test/CodeGen/AArch64/ragreedy-csr2.ll b/llvm/test/CodeGen/AArch64/ragreedy-csr2.ll
new file mode 100644
index 0000000000000..59fd73d497309
--- /dev/null
+++ b/llvm/test/CodeGen/AArch64/ragreedy-csr2.ll
@@ -0,0 +1,131 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc < %s -regalloc-csr-cost-scale=80 | FileCheck %s
+
+target datalayout = "e-m:e-p270:32:32-p271:32:32-p272:64:64-i8:8:32-i16:16:32-i64:64-i128:128-n32:64-S128-Fn32"
+target triple = "aarch64"
+
+; In cold basic blocks we should spill a register instead of using callee saved
+; register.
+
+ at Func = external global ptr
+
+define void @foo(ptr %param) #0 {
+; CHECK-LABEL: foo:
+; CHECK: // %bb.0: // %entry
+; CHECK-NEXT: ldr x8, [x0]
+; CHECK-NEXT: cbz x8, .LBB0_4
+; CHECK-NEXT: // %bb.1: // %if.end
+; CHECK-NEXT: sub sp, sp, #32
+; CHECK-NEXT: .cfi_def_cfa_offset 32
+; CHECK-NEXT: stp x29, x30, [sp, #16] // 16-byte Folded Spill
+; CHECK-NEXT: add x29, sp, #16
+; CHECK-NEXT: .cfi_def_cfa w29, 16
+; CHECK-NEXT: .cfi_offset w30, -8
+; CHECK-NEXT: .cfi_offset w29, -16
+; CHECK-NEXT: .cfi_remember_state
+; CHECK-NEXT: ldr w1, [x0, #8]
+; CHECK-NEXT: cbnz w1, .LBB0_5
+; CHECK-NEXT: // %bb.2: // %if.end6
+; CHECK-NEXT: ldr w8, [x0, #12]
+; CHECK-NEXT: cbnz w8, .LBB0_6
+; CHECK-NEXT: .LBB0_3: // %if.end24
+; CHECK-NEXT: ldrb w8, [x0, #16]
+; CHECK-NEXT: .cfi_def_cfa wsp, 32
+; CHECK-NEXT: ldp x29, x30, [sp, #16] // 16-byte Folded Reload
+; CHECK-NEXT: add sp, sp, #32
+; CHECK-NEXT: .cfi_def_cfa_offset 0
+; CHECK-NEXT: .cfi_restore w30
+; CHECK-NEXT: .cfi_restore w29
+; CHECK-NEXT: tbnz w8, #0, .LBB0_8
+; CHECK-NEXT: .LBB0_4: // %common.ret
+; CHECK-NEXT: ret
+; CHECK-NEXT: .LBB0_5: // %cold1
+; CHECK-NEXT: .cfi_restore_state
+; CHECK-NEXT: str x0, [sp, #8] // 8-byte Spill
+; CHECK-NEXT: mov x0, x8
+; CHECK-NEXT: bl fuzz
+; CHECK-NEXT: ldr x0, [sp, #8] // 8-byte Reload
+; CHECK-NEXT: ldr w8, [x0, #12]
+; CHECK-NEXT: cbz w8, .LBB0_3
+; CHECK-NEXT: .LBB0_6: // %cold2
+; CHECK-NEXT: adrp x8, :got:Func
+; CHECK-NEXT: ldr x8, [x8, :got_lo12:Func]
+; CHECK-NEXT: str x0, [sp, #8] // 8-byte Spill
+; CHECK-NEXT: ldr x8, [x8]
+; CHECK-NEXT: blr x8
+; CHECK-NEXT: mov w8, w0
+; CHECK-NEXT: ldr x0, [sp, #8] // 8-byte Reload
+; CHECK-NEXT: tbz w8, #0, .LBB0_3
+; CHECK-NEXT: // %bb.7: // %cold3
+; CHECK-NEXT: ldr x8, [x0]
+; CHECK-NEXT: ldr w1, [x0, #12]
+; CHECK-NEXT: mov x0, x8
+; CHECK-NEXT: bl bar
+; CHECK-NEXT: ldr x0, [sp, #8] // 8-byte Reload
+; CHECK-NEXT: b .LBB0_3
+; CHECK-NEXT: .LBB0_8: // %if.then35
+; CHECK-NEXT: .cfi_def_cfa wsp, 0
+; CHECK-NEXT: .cfi_same_value w30
+; CHECK-NEXT: .cfi_same_value w29
+; CHECK-NEXT: ldr x0, [x0]
+; CHECK-NEXT: b bob
+entry:
+ %0 = load ptr, ptr %param, align 8
+ %cmp = icmp eq ptr %0, null
+ br i1 %cmp, label %return, label %if.end
+
+if.end: ; preds = %entry
+ %b = getelementptr inbounds nuw i8, ptr %param, i64 8
+ %1 = load i32, ptr %b, align 8
+ %cmp1.not = icmp eq i32 %1, 0
+ br i1 %cmp1.not, label %if.end6, label %cold1, !prof !12
+
+cold1: ; preds = %if.end
+ %call = tail call i64 @fuzz(ptr %0, i32 %1)
+ br label %if.end6
+
+if.end6: ; preds = %cold1, %if.end
+ %c = getelementptr inbounds nuw i8, ptr %param, i64 12
+ %2 = load i32, ptr %c, align 4
+ %cmp7.not = icmp eq i32 %2, 0
+ br i1 %cmp7.not, label %if.end24, label %cold2, !prof !12
+
+cold2: ; preds = %if.end6
+ %3 = load ptr, ptr @Func, align 8
+ %call17 = tail call i1 %3()
+ br i1 %call17, label %cold3, label %if.end24
+
+cold3: ; preds = %cold2
+ %4 = load ptr, ptr %param, align 8
+ %sunkaddr = getelementptr inbounds i8, ptr %param, i64 12
+ %5 = load i32, ptr %sunkaddr, align 4
+ %conv21 = trunc i32 %5 to i16
+ %call22 = tail call i64 @bar(ptr %4, i16 %conv21)
+ br label %if.end24
+
+if.end24: ; preds = %cold2, %cold3, %if.end6
+ %d = getelementptr inbounds nuw i8, ptr %param, i64 16
+ %bf.load = load i8, ptr %d, align 8
+ %6 = zext i8 %bf.load to i32
+ %bf.clear = and i32 %6, 1
+ %cmp26.not = icmp eq i32 %bf.clear, 0
+ br i1 %cmp26.not, label %return, label %if.then35, !prof !12
+
+if.then35: ; preds = %if.end24
+ %7 = load ptr, ptr %param, align 8
+ tail call void @bob(ptr %7)
+ ret void
+
+return: ; preds = %if.end24, %entry
+ ret void
+}
+
+declare i64 @fuzz(ptr, i32)
+
+declare i64 @bar(ptr, i16)
+
+declare void @bob(ptr)
+
+attributes #0 = { nounwind uwtable "frame-pointer"="non-leaf-no-reserve" "no-trapping-math"="true" "stack-protector-buffer-size"="8" "target-cpu"="generic" "target-features"="+fp-armv8,+neon,+v8a,-fmv" }
+
+!12 = !{!"branch_weights", !"expected", i32 2000, i32 1}
>From 52775f02fea56a8dfa7f3f11ef73f4c067b4c6fa Mon Sep 17 00:00:00 2001
From: Guozhi Wei <carrot at google.com>
Date: Thu, 22 Jan 2026 20:51:55 -0800
Subject: [PATCH 3/4] Address review comments and reformat.
---
llvm/lib/CodeGen/RegAllocGreedy.cpp | 10 +++++-----
llvm/lib/Target/AMDGPU/SIRegisterInfo.h | 4 +---
llvm/test/CodeGen/AArch64/ragreedy-csr2.ll | 3 +--
3 files changed, 7 insertions(+), 10 deletions(-)
diff --git a/llvm/lib/CodeGen/RegAllocGreedy.cpp b/llvm/lib/CodeGen/RegAllocGreedy.cpp
index b2c62c29ca8a8..836be7bdb3645 100644
--- a/llvm/lib/CodeGen/RegAllocGreedy.cpp
+++ b/llvm/lib/CodeGen/RegAllocGreedy.cpp
@@ -114,10 +114,10 @@ CSRFirstTimeCost("regalloc-csr-first-time-cost",
cl::desc("Cost for first time use of callee-saved register."),
cl::init(0), cl::Hidden);
-static cl::opt<unsigned>
-CSRCostScale("regalloc-csr-cost-scale",
- cl::desc("Scale for the callee-saved register cost, in percentage."),
- cl::init(80), cl::Hidden);
+static cl::opt<unsigned> CSRCostScale(
+ "regalloc-csr-cost-scale",
+ cl::desc("Scale for the callee-saved register cost, in percentage."),
+ cl::init(80), cl::Hidden);
static cl::opt<unsigned long> GrowRegionComplexityBudget(
"grow-region-complexity-budget",
@@ -2356,7 +2356,7 @@ BlockFrequency RAGreedy::calcSpillCost(const LiveInterval &LI) {
E = MRI->reg_instr_nodbg_end();
I != E;) {
MachineInstr *MI = &*(I++);
- if (MI->isImplicitDef())
+ if (MI->isMetaInstruction())
continue;
if (!Visited.insert(MI).second)
continue;
diff --git a/llvm/lib/Target/AMDGPU/SIRegisterInfo.h b/llvm/lib/Target/AMDGPU/SIRegisterInfo.h
index 2dd0413eb054a..9d1a9eae75020 100644
--- a/llvm/lib/Target/AMDGPU/SIRegisterInfo.h
+++ b/llvm/lib/Target/AMDGPU/SIRegisterInfo.h
@@ -107,9 +107,7 @@ class SIRegisterInfo final : public AMDGPUGenRegisterInfo {
// Stack access is very expensive. CSRs are also the high registers, and we
// want to minimize the number of used registers.
- unsigned getCSRCost() const override {
- return 100;
- }
+ unsigned getCSRCost() const override { return 100; }
// When building a block VGPR load, we only really transfer a subset of the
// registers in the block, based on a mask. Liveness analysis is not aware of
diff --git a/llvm/test/CodeGen/AArch64/ragreedy-csr2.ll b/llvm/test/CodeGen/AArch64/ragreedy-csr2.ll
index 59fd73d497309..d32b3a3ab8ec0 100644
--- a/llvm/test/CodeGen/AArch64/ragreedy-csr2.ll
+++ b/llvm/test/CodeGen/AArch64/ragreedy-csr2.ll
@@ -1,7 +1,6 @@
; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
; RUN: llc < %s -regalloc-csr-cost-scale=80 | FileCheck %s
-target datalayout = "e-m:e-p270:32:32-p271:32:32-p272:64:64-i8:8:32-i16:16:32-i64:64-i128:128-n32:64-S128-Fn32"
target triple = "aarch64"
; In cold basic blocks we should spill a register instead of using callee saved
@@ -126,6 +125,6 @@ declare i64 @bar(ptr, i16)
declare void @bob(ptr)
-attributes #0 = { nounwind uwtable "frame-pointer"="non-leaf-no-reserve" "no-trapping-math"="true" "stack-protector-buffer-size"="8" "target-cpu"="generic" "target-features"="+fp-armv8,+neon,+v8a,-fmv" }
+attributes #0 = { uwtable "frame-pointer"="non-leaf-no-reserve" }
!12 = !{!"branch_weights", !"expected", i32 2000, i32 1}
>From 08787767ad85d5823dd7d581f1a7e9a3b9832f1a Mon Sep 17 00:00:00 2001
From: Guozhi Wei <carrot at google.com>
Date: Fri, 30 Jan 2026 14:57:53 -0800
Subject: [PATCH 4/4] Reformat.
---
llvm/lib/CodeGen/RegAllocGreedy.cpp | 8 ++---
llvm/test/CodeGen/AArch64/ragreedy-csr2.ll | 34 +++++++++++-----------
2 files changed, 21 insertions(+), 21 deletions(-)
diff --git a/llvm/lib/CodeGen/RegAllocGreedy.cpp b/llvm/lib/CodeGen/RegAllocGreedy.cpp
index 836be7bdb3645..74fe37327f549 100644
--- a/llvm/lib/CodeGen/RegAllocGreedy.cpp
+++ b/llvm/lib/CodeGen/RegAllocGreedy.cpp
@@ -2361,8 +2361,7 @@ BlockFrequency RAGreedy::calcSpillCost(const LiveInterval &LI) {
if (!Visited.insert(MI).second)
continue;
- bool Reads, Writes;
- std::tie(Reads, Writes) = MI->readsWritesVirtualRegister(LI.reg());
+ auto [Reads, Writes] = MI->readsWritesVirtualRegister(LI.reg());
auto MBBFreq = SpillPlacer->getBlockFrequency(MI->getParent()->getNumber());
SpillCost += (Reads + Writes) * MBBFreq.getFrequency();
}
@@ -2437,13 +2436,14 @@ void RAGreedy::initializeCSRCost() {
uint64_t FixedEntry = 1 << 14;
if (ActualEntry < FixedEntry)
CSRCost *= BranchProbability(ActualEntry, FixedEntry);
- else if (ActualEntry <= UINT32_MAX)
+ else if (ActualEntry <= UINT32_MAX) {
// Invert the fraction and divide.
CSRCost /= BranchProbability(FixedEntry, ActualEntry);
- else
+ } else {
// Can't use BranchProbability in general, since it takes 32-bit numbers.
CSRCost =
BlockFrequency(CSRCost.getFrequency() * (ActualEntry / FixedEntry));
+ }
} else {
uint64_t EntryFreq = MBFI->getEntryFreq().getFrequency();
CSRCost = BlockFrequency(TRI->getCSRFirstUseCost() * EntryFreq);
diff --git a/llvm/test/CodeGen/AArch64/ragreedy-csr2.ll b/llvm/test/CodeGen/AArch64/ragreedy-csr2.ll
index d32b3a3ab8ec0..2d5f5dbbf8f07 100644
--- a/llvm/test/CodeGen/AArch64/ragreedy-csr2.ll
+++ b/llvm/test/CodeGen/AArch64/ragreedy-csr2.ll
@@ -69,50 +69,50 @@ define void @foo(ptr %param) #0 {
; CHECK-NEXT: ldr x0, [x0]
; CHECK-NEXT: b bob
entry:
- %0 = load ptr, ptr %param, align 8
- %cmp = icmp eq ptr %0, null
+ %load0 = load ptr, ptr %param, align 8
+ %cmp = icmp eq ptr %load0, null
br i1 %cmp, label %return, label %if.end
if.end: ; preds = %entry
%b = getelementptr inbounds nuw i8, ptr %param, i64 8
- %1 = load i32, ptr %b, align 8
- %cmp1.not = icmp eq i32 %1, 0
+ %load1 = load i32, ptr %b, align 8
+ %cmp1.not = icmp eq i32 %load1, 0
br i1 %cmp1.not, label %if.end6, label %cold1, !prof !12
cold1: ; preds = %if.end
- %call = tail call i64 @fuzz(ptr %0, i32 %1)
+ %call = tail call i64 @fuzz(ptr %load0, i32 %load1)
br label %if.end6
if.end6: ; preds = %cold1, %if.end
%c = getelementptr inbounds nuw i8, ptr %param, i64 12
- %2 = load i32, ptr %c, align 4
- %cmp7.not = icmp eq i32 %2, 0
+ %load2 = load i32, ptr %c, align 4
+ %cmp7.not = icmp eq i32 %load2, 0
br i1 %cmp7.not, label %if.end24, label %cold2, !prof !12
cold2: ; preds = %if.end6
- %3 = load ptr, ptr @Func, align 8
- %call17 = tail call i1 %3()
+ %func = load ptr, ptr @Func, align 8
+ %call17 = tail call i1 %func()
br i1 %call17, label %cold3, label %if.end24
cold3: ; preds = %cold2
- %4 = load ptr, ptr %param, align 8
+ %load4 = load ptr, ptr %param, align 8
%sunkaddr = getelementptr inbounds i8, ptr %param, i64 12
- %5 = load i32, ptr %sunkaddr, align 4
- %conv21 = trunc i32 %5 to i16
- %call22 = tail call i64 @bar(ptr %4, i16 %conv21)
+ %load5 = load i32, ptr %sunkaddr, align 4
+ %conv21 = trunc i32 %load5 to i16
+ %call22 = tail call i64 @bar(ptr %load4, i16 %conv21)
br label %if.end24
if.end24: ; preds = %cold2, %cold3, %if.end6
%d = getelementptr inbounds nuw i8, ptr %param, i64 16
%bf.load = load i8, ptr %d, align 8
- %6 = zext i8 %bf.load to i32
- %bf.clear = and i32 %6, 1
+ %zext = zext i8 %bf.load to i32
+ %bf.clear = and i32 %zext, 1
%cmp26.not = icmp eq i32 %bf.clear, 0
br i1 %cmp26.not, label %return, label %if.then35, !prof !12
if.then35: ; preds = %if.end24
- %7 = load ptr, ptr %param, align 8
- tail call void @bob(ptr %7)
+ %load6 = load ptr, ptr %param, align 8
+ tail call void @bob(ptr %load6)
ret void
return: ; preds = %if.end24, %entry
More information about the llvm-commits
mailing list