[llvm] [RegAlloc] Recolor blockers of physical register hints (PR #224263)
Pengcheng Wang via llvm-commits
llvm-commits at lists.llvm.org
Thu Sep 17 04:08:27 PDT 2026
https://github.com/wangpc-pp created https://github.com/llvm/llvm-project/pull/224263
Recolor virtual intervals that occupy a broken physical register
hint. Support both assigning a complete interval to a lower-cost hint
and freeing a local COPY-source range for later copy optimization.
Keep recoloring transactional and reject choices that increase
register cost, introduce a callee-saved register, or exceed the copy
frequency benefit.
This also enables shrink wrapping when an entry COPY is otherwise
pinned by a short-lived use of its physical source.
Fixes #223960.
Assisted-by: TRAE CLI (GPT-5)
>From 3245ea9ab32f846798c330229d72f08897bcc09f Mon Sep 17 00:00:00 2001
From: Pengcheng Wang <wangpengcheng.pp at bytedance.com>
Date: Thu, 17 Sep 2026 15:47:21 +0800
Subject: [PATCH] [RegAlloc] Recolor blockers of physical register hints
Recolor virtual intervals that occupy a broken physical register
hint. Support both assigning a complete interval to a lower-cost hint
and freeing a local COPY-source range for later copy optimization.
Keep recoloring transactional and reject choices that increase
register cost, introduce a callee-saved register, or exceed the copy
frequency benefit.
This also enables shrink wrapping when an entry COPY is otherwise
pinned by a short-lived use of its physical source.
Fixes #223960.
Assisted-by: TRAE CLI (GPT-5)
---
llvm/lib/CodeGen/RegAllocGreedy.cpp | 200 ++++++++++++++++++
llvm/lib/CodeGen/RegAllocGreedy.h | 3 +
llvm/test/CodeGen/RISCV/ctlz-cttz-ctpop.ll | 13 +-
.../test/CodeGen/RISCV/overflow-intrinsics.ll | 20 +-
.../RISCV/regalloc-physical-hint-recolor.mir | 39 ++++
llvm/test/CodeGen/RISCV/rv32xtheadbb.ll | 13 +-
llvm/test/CodeGen/RISCV/rv32zbb.ll | 13 +-
7 files changed, 273 insertions(+), 28 deletions(-)
create mode 100644 llvm/test/CodeGen/RISCV/regalloc-physical-hint-recolor.mir
diff --git a/llvm/lib/CodeGen/RegAllocGreedy.cpp b/llvm/lib/CodeGen/RegAllocGreedy.cpp
index d534a71a38d0d..6403e8b4f053d 100644
--- a/llvm/lib/CodeGen/RegAllocGreedy.cpp
+++ b/llvm/lib/CodeGen/RegAllocGreedy.cpp
@@ -82,6 +82,8 @@ using namespace llvm;
STATISTIC(NumGlobalSplits, "Number of split global live ranges");
STATISTIC(NumLocalSplits, "Number of split local live ranges");
STATISTIC(NumEvicted, "Number of interferences evicted");
+STATISTIC(NumPhysicalHintRecolors,
+ "Number of broken physical hints repaired by recoloring");
static cl::opt<SplitEditor::ComplementSpillMode> SplitSpillMode(
"split-spill-mode", cl::Hidden,
@@ -2321,6 +2323,194 @@ bool RAGreedy::tryRecoloringCandidates(PQueue &RecoloringQueue,
return true;
}
+/// Recolor the virtual intervals that occupy \p PhysHint over \p HintRange.
+/// \p MaxCostIncrease is the copy cost that satisfying the hint would save.
+/// Return true when at least one interference was recolored.
+bool RAGreedy::recolorPhysicalHintInterferences(
+ const LiveRange &HintRange, MCRegister PhysHint,
+ BlockFrequency MaxCostIncrease) {
+ SmallLISet RecoloringCandidates;
+ for (MCRegUnit Unit : TRI->regunits(PhysHint)) {
+ LiveIntervalUnion::Query Query;
+ Query.reset(0, HintRange, Matrix->getLiveUnions()[unsigned(Unit)]);
+ if (Query.interferingVRegs(LastChanceRecoloringMaxInterference).size() >=
+ LastChanceRecoloringMaxInterference &&
+ !ExhaustiveSearch)
+ return false;
+ RecoloringCandidates.insert_range(Query.interferingVRegs());
+ }
+ if (RecoloringCandidates.empty())
+ return false;
+
+ SmallVector<std::pair<const LiveInterval *, MCRegister>, 4> OldAssignments;
+ SmallVector<HintsInfo, 4> HintInfos;
+ BlockFrequency OldCost(0);
+ for (const LiveInterval *Intf : RecoloringCandidates) {
+ MCRegister OldPhys = VRM->getPhys(Intf->reg());
+ if (!OldPhys)
+ return false;
+ OldAssignments.emplace_back(Intf, OldPhys);
+ HintInfos.emplace_back();
+ collectHintInfo(Intf->reg(), HintInfos.back());
+ OldCost += getBrokenHintFreq(HintInfos.back(), OldPhys);
+ }
+ for (const auto &[Intf, OldPhys] : OldAssignments)
+ Matrix->unassign(*Intf);
+
+ bool Success = true;
+ BlockFrequency NewCost(0);
+ for (const auto &[Index, Assignment] : llvm::enumerate(OldAssignments)) {
+ const LiveInterval *Intf = Assignment.first;
+ MCRegister OldPhys = Assignment.second;
+ MCRegister NewPhys;
+ BlockFrequency BestCost = BlockFrequency::max();
+ auto Order =
+ AllocationOrder::create(Intf->reg(), *VRM, RegClassInfo, Matrix);
+ for (MCRegister Candidate : Order) {
+ if (TRI->regsOverlap(Candidate, PhysHint) ||
+ RegCosts[Candidate.id()] > RegCosts[OldPhys.id()] ||
+ RegClassInfo.getLastCalleeSavedAlias(Candidate) ||
+ Matrix->checkInterference(*Intf, Candidate) != LiveRegMatrix::IK_Free)
+ continue;
+ BlockFrequency Cost = getBrokenHintFreq(HintInfos[Index], Candidate);
+ if (!NewPhys || Cost < BestCost) {
+ NewPhys = Candidate;
+ BestCost = Cost;
+ }
+ }
+ if (!NewPhys) {
+ Success = false;
+ break;
+ }
+ Matrix->assign(*Intf, NewPhys);
+ }
+
+ if (Success)
+ for (const auto &Assignment : OldAssignments) {
+ const LiveInterval *Intf = Assignment.first;
+ HintsInfo NewHintInfo;
+ collectHintInfo(Intf->reg(), NewHintInfo);
+ NewCost += getBrokenHintFreq(NewHintInfo, VRM->getPhys(Intf->reg()));
+ }
+
+ if (Success && NewCost <= OldCost + MaxCostIncrease) {
+ LLVM_DEBUG(dbgs() << "Freed physical hint " << printReg(PhysHint, TRI)
+ << " over [" << HintRange.beginIndex() << ','
+ << HintRange.endIndex() << ")\n");
+ return true;
+ }
+
+ for (auto [Intf, OldPhys] : llvm::reverse(OldAssignments))
+ if (VRM->hasPhys(Intf->reg()))
+ Matrix->unassign(*Intf);
+ for (auto [Intf, OldPhys] : OldAssignments)
+ Matrix->assign(*Intf, OldPhys);
+ return false;
+}
+
+/// Try to repair a broken physical register hint. Prefer assigning the whole
+/// live interval to the hint when that lowers register cost. When whole-range
+/// reassignment is not possible, try to make a physical-source copy
+/// optimizable by freeing its local range.
+void RAGreedy::tryRecoloringForPhysicalHint(const LiveInterval &VirtReg,
+ MCRegister PhysHint) {
+ MCRegister CurrPhys = VRM->getPhys(VirtReg.reg());
+ const TargetRegisterClass *RC = MRI->getRegClass(VirtReg.reg());
+ if (!CurrPhys || TRI->regsOverlap(CurrPhys, PhysHint) ||
+ !RC->contains(PhysHint) || MRI->isReserved(PhysHint))
+ return;
+
+ MCRegister CurrCSR = RegClassInfo.getLastCalleeSavedAlias(CurrPhys);
+ MCRegister HintCSR = RegClassInfo.getLastCalleeSavedAlias(PhysHint);
+ bool ReducesCSRCost = false;
+ if (CurrCSR && !HintCSR) {
+ Matrix->unassign(VirtReg);
+ ReducesCSRCost = !Matrix->isPhysRegUsed(CurrPhys);
+ Matrix->assign(VirtReg, CurrPhys);
+ }
+ // Reassigning an entire interval solely for an equal-cost copy hint can
+ // cause widespread register churn. Only do so when it removes a CSR use.
+ // Local copy repair below can still operate on an equal-cost hint when it
+ // enables a concrete copy optimization.
+ if (ReducesCSRCost) {
+ HintsInfo VirtRegHints;
+ collectHintInfo(VirtReg.reg(), VirtRegHints);
+ BlockFrequency OldVirtRegCost = getBrokenHintFreq(VirtRegHints, CurrPhys);
+ BlockFrequency NewVirtRegCost = getBrokenHintFreq(VirtRegHints, PhysHint);
+ if (NewVirtRegCost <= OldVirtRegCost) {
+ LiveRegMatrix::InterferenceKind IK =
+ Matrix->checkInterference(VirtReg, PhysHint);
+ if (IK == LiveRegMatrix::IK_Free ||
+ (IK == LiveRegMatrix::IK_VirtReg &&
+ recolorPhysicalHintInterferences(VirtReg, PhysHint,
+ OldVirtRegCost - NewVirtRegCost))) {
+ Matrix->unassign(VirtReg);
+ Matrix->assign(VirtReg, PhysHint);
+ LLVM_DEBUG(dbgs() << "Recolored " << printReg(VirtReg.reg(), TRI)
+ << " to physical hint " << printReg(PhysHint, TRI)
+ << '\n');
+ ++NumPhysicalHintRecolors;
+ return;
+ }
+ }
+ }
+
+ // Restrict partial repair to a callee-saved interval whose physical hint
+ // is call-clobbered. Other equal-cost moves tend to cause widespread
+ // register churn without reducing save/restore cost.
+ if (!CurrCSR || HintCSR)
+ return;
+
+ MachineInstr *Copy = MRI->getUniqueVRegDef(VirtReg.reg());
+ if (!Copy || !Copy->isCopy() || !TII->shouldPostRASink(*Copy) ||
+ Copy->getNumOperands() < 2 || !Copy->getOperand(0).isReg() ||
+ !Copy->getOperand(1).isReg() ||
+ Copy->getOperand(0).getReg() != VirtReg.reg() ||
+ Copy->getOperand(0).getSubReg() ||
+ Copy->getOperand(1).getReg() != PhysHint ||
+ Copy->getOperand(1).getSubReg() || Copy->getParent()->succ_size() < 2)
+ return;
+
+ MachineBasicBlock *CopyMBB = Copy->getParent();
+ MachineBasicBlock *LiveInSucc = nullptr;
+ for (MachineBasicBlock *Succ : CopyMBB->successors()) {
+ if (!VirtReg.liveAt(Indexes->getMBBStartIdx(Succ)))
+ continue;
+ if (LiveInSucc || Succ->pred_size() != 1)
+ return;
+ LiveInSucc = Succ;
+ }
+ if (!LiveInSucc)
+ return;
+
+ SlotIndex LastLocalUse = LIS->getInstructionIndex(*Copy).getDeadSlot();
+ for (const MachineOperand &MO : MRI->use_nodbg_operands(VirtReg.reg())) {
+ const MachineInstr &MI = *MO.getParent();
+ if (MI.getParent() != CopyMBB)
+ continue;
+ const TargetRegisterClass *UseRC =
+ MI.getRegClassConstraint(MO.getOperandNo(), TII, TRI);
+ if (MO.isImplicit() || MO.isTied() || MO.isUndef() ||
+ MI.hasExtraSrcRegAllocReq(MachineInstr::IgnoreBundle) || !UseRC ||
+ !UseRC->contains(PhysHint))
+ return;
+ LastLocalUse =
+ std::max(LastLocalUse, LIS->getInstructionIndex(MI).getDeadSlot());
+ }
+
+ for (auto I = std::next(Copy->getIterator()), E = CopyMBB->instr_end();
+ I != E; ++I)
+ if (I->isCall() || I->modifiesRegister(PhysHint, TRI))
+ return;
+
+ LiveRange HintRange;
+ SlotIndex MBBEnd = Indexes->getMBBEndIdx(CopyMBB);
+ VNInfo HintValue(0, LastLocalUse);
+ HintRange.addSegment(LiveRange::Segment(LastLocalUse, MBBEnd, &HintValue));
+ if (recolorPhysicalHintInterferences(HintRange, PhysHint, BlockFrequency(0)))
+ ++NumPhysicalHintRecolors;
+}
+
//===----------------------------------------------------------------------===//
// Main Entry Point
//===----------------------------------------------------------------------===//
@@ -2650,6 +2840,16 @@ void RAGreedy::tryHintsRecoloring() {
continue;
tryHintRecoloring(*LI);
}
+
+ // Try to repair physical hints after normal hint reconciliation so later
+ // recoloring cannot undo the result.
+ for (const LiveInterval *LI : SetOfBrokenHints) {
+ if (!VRM->hasPhys(LI->reg()))
+ continue;
+ Register Hint = MRI->getSimpleHint(LI->reg());
+ if (Hint && Hint.isPhysical())
+ tryRecoloringForPhysicalHint(*LI, Hint.asMCReg());
+ }
}
MCRegister RAGreedy::selectOrSplitImpl(const LiveInterval &VirtReg,
diff --git a/llvm/lib/CodeGen/RegAllocGreedy.h b/llvm/lib/CodeGen/RegAllocGreedy.h
index 465be0d76809e..5f76e259f18b9 100644
--- a/llvm/lib/CodeGen/RegAllocGreedy.h
+++ b/llvm/lib/CodeGen/RegAllocGreedy.h
@@ -396,6 +396,9 @@ class LLVM_LIBRARY_VISIBILITY RAGreedy : public RegAllocBase,
BlockFrequency getBrokenHintFreq(const HintsInfo &, MCRegister);
void collectHintInfo(Register, HintsInfo &);
+ bool recolorPhysicalHintInterferences(const LiveRange &, MCRegister,
+ BlockFrequency);
+ void tryRecoloringForPhysicalHint(const LiveInterval &, MCRegister);
/// Greedy RA statistic to remark.
struct RAGreedyStats {
diff --git a/llvm/test/CodeGen/RISCV/ctlz-cttz-ctpop.ll b/llvm/test/CodeGen/RISCV/ctlz-cttz-ctpop.ll
index 45b29fd970ef3..56c825acdba39 100644
--- a/llvm/test/CodeGen/RISCV/ctlz-cttz-ctpop.ll
+++ b/llvm/test/CodeGen/RISCV/ctlz-cttz-ctpop.ll
@@ -355,6 +355,9 @@ define i32 @test_cttz_i32(i32 %a) nounwind {
define i64 @test_cttz_i64(i64 %a) nounwind {
; RV32I-LABEL: test_cttz_i64:
; RV32I: # %bb.0:
+; RV32I-NEXT: or a2, a0, a1
+; RV32I-NEXT: beqz a2, .LBB3_3
+; RV32I-NEXT: # %bb.1: # %cond.false
; RV32I-NEXT: addi sp, sp, -32
; RV32I-NEXT: sw ra, 28(sp) # 4-byte Folded Spill
; RV32I-NEXT: sw s0, 24(sp) # 4-byte Folded Spill
@@ -363,9 +366,6 @@ define i64 @test_cttz_i64(i64 %a) nounwind {
; RV32I-NEXT: sw s3, 12(sp) # 4-byte Folded Spill
; RV32I-NEXT: sw s4, 8(sp) # 4-byte Folded Spill
; RV32I-NEXT: mv s0, a1
-; RV32I-NEXT: or a1, a0, a1
-; RV32I-NEXT: beqz a1, .LBB3_3
-; RV32I-NEXT: # %bb.1: # %cond.false
; RV32I-NEXT: neg a1, a0
; RV32I-NEXT: and a1, a0, a1
; RV32I-NEXT: lui s2, 30667
@@ -390,13 +390,13 @@ define i64 @test_cttz_i64(i64 %a) nounwind {
; RV32I-NEXT: j .LBB3_5
; RV32I-NEXT: .LBB3_3:
; RV32I-NEXT: li a0, 64
-; RV32I-NEXT: j .LBB3_5
+; RV32I-NEXT: li a1, 0
+; RV32I-NEXT: ret
; RV32I-NEXT: .LBB3_4:
; RV32I-NEXT: srli s1, s1, 27
; RV32I-NEXT: add s1, s3, s1
; RV32I-NEXT: lbu a0, 0(s1)
-; RV32I-NEXT: .LBB3_5: # %cond.end
-; RV32I-NEXT: li a1, 0
+; RV32I-NEXT: .LBB3_5: # %cond.false
; RV32I-NEXT: lw ra, 28(sp) # 4-byte Folded Reload
; RV32I-NEXT: lw s0, 24(sp) # 4-byte Folded Reload
; RV32I-NEXT: lw s1, 20(sp) # 4-byte Folded Reload
@@ -404,6 +404,7 @@ define i64 @test_cttz_i64(i64 %a) nounwind {
; RV32I-NEXT: lw s3, 12(sp) # 4-byte Folded Reload
; RV32I-NEXT: lw s4, 8(sp) # 4-byte Folded Reload
; RV32I-NEXT: addi sp, sp, 32
+; RV32I-NEXT: li a1, 0
; RV32I-NEXT: ret
;
; RV64I-LABEL: test_cttz_i64:
diff --git a/llvm/test/CodeGen/RISCV/overflow-intrinsics.ll b/llvm/test/CodeGen/RISCV/overflow-intrinsics.ll
index aef7b82eb78b4..1e7a5c2817644 100644
--- a/llvm/test/CodeGen/RISCV/overflow-intrinsics.ll
+++ b/llvm/test/CodeGen/RISCV/overflow-intrinsics.ll
@@ -1090,15 +1090,15 @@ define i1 @usubo_ult_cmp_dominates_i64(i64 %x, i64 %y, ptr %p, i1 %cond) {
; RV32-NEXT: .cfi_offset s5, -28
; RV32-NEXT: .cfi_offset s6, -32
; RV32-NEXT: mv s1, a5
-; RV32-NEXT: mv s4, a1
-; RV32-NEXT: andi a1, a5, 1
-; RV32-NEXT: beqz a1, .LBB32_6
+; RV32-NEXT: mv s2, a2
+; RV32-NEXT: andi a2, a5, 1
+; RV32-NEXT: beqz a2, .LBB32_6
; RV32-NEXT: # %bb.1: # %t
; RV32-NEXT: mv s0, a4
; RV32-NEXT: mv s3, a3
-; RV32-NEXT: mv s2, a2
+; RV32-NEXT: mv s4, a1
; RV32-NEXT: mv s5, a0
-; RV32-NEXT: beq s4, a3, .LBB32_3
+; RV32-NEXT: beq a1, a3, .LBB32_3
; RV32-NEXT: # %bb.2: # %t
; RV32-NEXT: sltu s6, s4, s3
; RV32-NEXT: j .LBB32_4
@@ -1157,13 +1157,13 @@ define i1 @usubo_ult_cmp_dominates_i64(i64 %x, i64 %y, ptr %p, i1 %cond) {
; RV64-NEXT: .cfi_offset s3, -40
; RV64-NEXT: .cfi_offset s4, -48
; RV64-NEXT: mv s1, a3
-; RV64-NEXT: mv s2, a1
-; RV64-NEXT: andi a1, a3, 1
-; RV64-NEXT: beqz a1, .LBB32_3
-; RV64-NEXT: # %bb.1: # %t
; RV64-NEXT: mv s0, a2
+; RV64-NEXT: andi a2, a3, 1
+; RV64-NEXT: beqz a2, .LBB32_3
+; RV64-NEXT: # %bb.1: # %t
+; RV64-NEXT: mv s2, a1
; RV64-NEXT: mv s3, a0
-; RV64-NEXT: sltu s4, a0, s2
+; RV64-NEXT: sltu s4, a0, a1
; RV64-NEXT: mv a0, s4
; RV64-NEXT: call call
; RV64-NEXT: bgeu s3, s2, .LBB32_3
diff --git a/llvm/test/CodeGen/RISCV/regalloc-physical-hint-recolor.mir b/llvm/test/CodeGen/RISCV/regalloc-physical-hint-recolor.mir
new file mode 100644
index 0000000000000..6a3264149daae
--- /dev/null
+++ b/llvm/test/CodeGen/RISCV/regalloc-physical-hint-recolor.mir
@@ -0,0 +1,39 @@
+# RUN: llc -mtriple=riscv64 -mattr=+reserve-x9,+reserve-x11,+reserve-x12,+reserve-x13,+reserve-x14,+reserve-x15 -run-pass=greedy,virtregrewriter -regalloc-enable-priority-advisor=dummy -verify-machineinstrs -o - %s | FileCheck %s
+
+# Verify that the final broken-hint reconciliation can move virtual
+# interferences out of a physical hint and assign the complete hinted interval
+# to that register. This test does not depend on post-RA copy sinking.
+
+---
+name: direct_physical_hint_recolor
+tracksRegLiveness: true
+registers:
+ - { id: 0, class: gpr }
+ - { id: 1, class: gprc }
+ - { id: 2, class: gpr }
+ - { id: 3, class: gpr }
+ - { id: 4, class: gpr }
+body: |
+ bb.0:
+ liveins: $x10
+
+ %0:gpr = COPY $x10
+ %1:gprc = COPY $x10
+ %2:gpr = ADD %0, %0
+ %3:gpr = ADD %2, %0
+ %4:gpr = ADD %3, %1
+ $x10 = COPY %1
+ $x11 = COPY %4
+ PseudoRET implicit $x10, implicit $x11
+
+ ; CHECK-LABEL: name: direct_physical_hint_recolor
+ ; CHECK: bb.0:
+ ; CHECK-NEXT: liveins: $x10
+ ; CHECK-NEXT: {{ $}}
+ ; CHECK-NEXT: renamable $x17 = COPY $x10
+ ; CHECK-NEXT: renamable $x16 = ADD renamable $x17, renamable $x17
+ ; CHECK-NEXT: renamable $x16 = ADD killed renamable $x16, killed renamable $x17
+ ; CHECK-NEXT: renamable $x16 = ADD killed renamable $x16, renamable $x10
+ ; CHECK-NEXT: $x11 = COPY killed renamable $x16
+ ; CHECK-NEXT: PseudoRET implicit $x10, implicit $x11
+...
diff --git a/llvm/test/CodeGen/RISCV/rv32xtheadbb.ll b/llvm/test/CodeGen/RISCV/rv32xtheadbb.ll
index 3ba5312f97626..ec01e26388838 100644
--- a/llvm/test/CodeGen/RISCV/rv32xtheadbb.ll
+++ b/llvm/test/CodeGen/RISCV/rv32xtheadbb.ll
@@ -215,6 +215,9 @@ define i32 @cttz_i32(i32 %a) nounwind {
define i64 @cttz_i64(i64 %a) nounwind {
; RV32I-LABEL: cttz_i64:
; RV32I: # %bb.0:
+; RV32I-NEXT: or a2, a0, a1
+; RV32I-NEXT: beqz a2, .LBB3_3
+; RV32I-NEXT: # %bb.1: # %cond.false
; RV32I-NEXT: addi sp, sp, -32
; RV32I-NEXT: sw ra, 28(sp) # 4-byte Folded Spill
; RV32I-NEXT: sw s0, 24(sp) # 4-byte Folded Spill
@@ -223,9 +226,6 @@ define i64 @cttz_i64(i64 %a) nounwind {
; RV32I-NEXT: sw s3, 12(sp) # 4-byte Folded Spill
; RV32I-NEXT: sw s4, 8(sp) # 4-byte Folded Spill
; RV32I-NEXT: mv s0, a1
-; RV32I-NEXT: or a1, a0, a1
-; RV32I-NEXT: beqz a1, .LBB3_3
-; RV32I-NEXT: # %bb.1: # %cond.false
; RV32I-NEXT: neg a1, a0
; RV32I-NEXT: and a1, a0, a1
; RV32I-NEXT: lui s2, 30667
@@ -250,13 +250,13 @@ define i64 @cttz_i64(i64 %a) nounwind {
; RV32I-NEXT: j .LBB3_5
; RV32I-NEXT: .LBB3_3:
; RV32I-NEXT: li a0, 64
-; RV32I-NEXT: j .LBB3_5
+; RV32I-NEXT: li a1, 0
+; RV32I-NEXT: ret
; RV32I-NEXT: .LBB3_4:
; RV32I-NEXT: srli s1, s1, 27
; RV32I-NEXT: add s1, s3, s1
; RV32I-NEXT: lbu a0, 0(s1)
-; RV32I-NEXT: .LBB3_5: # %cond.end
-; RV32I-NEXT: li a1, 0
+; RV32I-NEXT: .LBB3_5: # %cond.false
; RV32I-NEXT: lw ra, 28(sp) # 4-byte Folded Reload
; RV32I-NEXT: lw s0, 24(sp) # 4-byte Folded Reload
; RV32I-NEXT: lw s1, 20(sp) # 4-byte Folded Reload
@@ -264,6 +264,7 @@ define i64 @cttz_i64(i64 %a) nounwind {
; RV32I-NEXT: lw s3, 12(sp) # 4-byte Folded Reload
; RV32I-NEXT: lw s4, 8(sp) # 4-byte Folded Reload
; RV32I-NEXT: addi sp, sp, 32
+; RV32I-NEXT: li a1, 0
; RV32I-NEXT: ret
;
; RV32XTHEADBB-NOB-LABEL: cttz_i64:
diff --git a/llvm/test/CodeGen/RISCV/rv32zbb.ll b/llvm/test/CodeGen/RISCV/rv32zbb.ll
index 6164065d02d94..2274c30b65df9 100644
--- a/llvm/test/CodeGen/RISCV/rv32zbb.ll
+++ b/llvm/test/CodeGen/RISCV/rv32zbb.ll
@@ -182,6 +182,9 @@ define i32 @cttz_i32(i32 %a) nounwind {
define i64 @cttz_i64(i64 %a) nounwind {
; RV32I-LABEL: cttz_i64:
; RV32I: # %bb.0:
+; RV32I-NEXT: or a2, a0, a1
+; RV32I-NEXT: beqz a2, .LBB3_3
+; RV32I-NEXT: # %bb.1: # %cond.false
; RV32I-NEXT: addi sp, sp, -32
; RV32I-NEXT: sw ra, 28(sp) # 4-byte Folded Spill
; RV32I-NEXT: sw s0, 24(sp) # 4-byte Folded Spill
@@ -190,9 +193,6 @@ define i64 @cttz_i64(i64 %a) nounwind {
; RV32I-NEXT: sw s3, 12(sp) # 4-byte Folded Spill
; RV32I-NEXT: sw s4, 8(sp) # 4-byte Folded Spill
; RV32I-NEXT: mv s0, a1
-; RV32I-NEXT: or a1, a0, a1
-; RV32I-NEXT: beqz a1, .LBB3_3
-; RV32I-NEXT: # %bb.1: # %cond.false
; RV32I-NEXT: neg a1, a0
; RV32I-NEXT: and a1, a0, a1
; RV32I-NEXT: lui s2, 30667
@@ -217,13 +217,13 @@ define i64 @cttz_i64(i64 %a) nounwind {
; RV32I-NEXT: j .LBB3_5
; RV32I-NEXT: .LBB3_3:
; RV32I-NEXT: li a0, 64
-; RV32I-NEXT: j .LBB3_5
+; RV32I-NEXT: li a1, 0
+; RV32I-NEXT: ret
; RV32I-NEXT: .LBB3_4:
; RV32I-NEXT: srli s1, s1, 27
; RV32I-NEXT: add s1, s3, s1
; RV32I-NEXT: lbu a0, 0(s1)
-; RV32I-NEXT: .LBB3_5: # %cond.end
-; RV32I-NEXT: li a1, 0
+; RV32I-NEXT: .LBB3_5: # %cond.false
; RV32I-NEXT: lw ra, 28(sp) # 4-byte Folded Reload
; RV32I-NEXT: lw s0, 24(sp) # 4-byte Folded Reload
; RV32I-NEXT: lw s1, 20(sp) # 4-byte Folded Reload
@@ -231,6 +231,7 @@ define i64 @cttz_i64(i64 %a) nounwind {
; RV32I-NEXT: lw s3, 12(sp) # 4-byte Folded Reload
; RV32I-NEXT: lw s4, 8(sp) # 4-byte Folded Reload
; RV32I-NEXT: addi sp, sp, 32
+; RV32I-NEXT: li a1, 0
; RV32I-NEXT: ret
;
; RV32ZBB-LABEL: cttz_i64:
More information about the llvm-commits
mailing list