[llvm] [RegAlloc] Recolor blockers of physical register hints (PR #224263)

Pengcheng Wang via llvm-commits llvm-commits at lists.llvm.org
Thu Sep 17 04:08:27 PDT 2026


https://github.com/wangpc-pp created https://github.com/llvm/llvm-project/pull/224263

Recolor virtual intervals that occupy a broken physical register
hint. Support both assigning a complete interval to a lower-cost hint
and freeing a local COPY-source range for later copy optimization.

Keep recoloring transactional and reject choices that increase
register cost, introduce a callee-saved register, or exceed the copy
frequency benefit.

This also enables shrink wrapping when an entry COPY is otherwise
pinned by a short-lived use of its physical source.

Fixes #223960.

Assisted-by: TRAE CLI (GPT-5)


>From 3245ea9ab32f846798c330229d72f08897bcc09f Mon Sep 17 00:00:00 2001
From: Pengcheng Wang <wangpengcheng.pp at bytedance.com>
Date: Thu, 17 Sep 2026 15:47:21 +0800
Subject: [PATCH] [RegAlloc] Recolor blockers of physical register hints

Recolor virtual intervals that occupy a broken physical register
hint. Support both assigning a complete interval to a lower-cost hint
and freeing a local COPY-source range for later copy optimization.

Keep recoloring transactional and reject choices that increase
register cost, introduce a callee-saved register, or exceed the copy
frequency benefit.

This also enables shrink wrapping when an entry COPY is otherwise
pinned by a short-lived use of its physical source.

Fixes #223960.

Assisted-by: TRAE CLI (GPT-5)
---
 llvm/lib/CodeGen/RegAllocGreedy.cpp           | 200 ++++++++++++++++++
 llvm/lib/CodeGen/RegAllocGreedy.h             |   3 +
 llvm/test/CodeGen/RISCV/ctlz-cttz-ctpop.ll    |  13 +-
 .../test/CodeGen/RISCV/overflow-intrinsics.ll |  20 +-
 .../RISCV/regalloc-physical-hint-recolor.mir  |  39 ++++
 llvm/test/CodeGen/RISCV/rv32xtheadbb.ll       |  13 +-
 llvm/test/CodeGen/RISCV/rv32zbb.ll            |  13 +-
 7 files changed, 273 insertions(+), 28 deletions(-)
 create mode 100644 llvm/test/CodeGen/RISCV/regalloc-physical-hint-recolor.mir

diff --git a/llvm/lib/CodeGen/RegAllocGreedy.cpp b/llvm/lib/CodeGen/RegAllocGreedy.cpp
index d534a71a38d0d..6403e8b4f053d 100644
--- a/llvm/lib/CodeGen/RegAllocGreedy.cpp
+++ b/llvm/lib/CodeGen/RegAllocGreedy.cpp
@@ -82,6 +82,8 @@ using namespace llvm;
 STATISTIC(NumGlobalSplits, "Number of split global live ranges");
 STATISTIC(NumLocalSplits,  "Number of split local live ranges");
 STATISTIC(NumEvicted,      "Number of interferences evicted");
+STATISTIC(NumPhysicalHintRecolors,
+          "Number of broken physical hints repaired by recoloring");
 
 static cl::opt<SplitEditor::ComplementSpillMode> SplitSpillMode(
     "split-spill-mode", cl::Hidden,
@@ -2321,6 +2323,194 @@ bool RAGreedy::tryRecoloringCandidates(PQueue &RecoloringQueue,
   return true;
 }
 
+/// Recolor the virtual intervals that occupy \p PhysHint over \p HintRange.
+/// \p MaxCostIncrease is the copy cost that satisfying the hint would save.
+/// Return true when at least one interference was recolored.
+bool RAGreedy::recolorPhysicalHintInterferences(
+    const LiveRange &HintRange, MCRegister PhysHint,
+    BlockFrequency MaxCostIncrease) {
+  SmallLISet RecoloringCandidates;
+  for (MCRegUnit Unit : TRI->regunits(PhysHint)) {
+    LiveIntervalUnion::Query Query;
+    Query.reset(0, HintRange, Matrix->getLiveUnions()[unsigned(Unit)]);
+    if (Query.interferingVRegs(LastChanceRecoloringMaxInterference).size() >=
+            LastChanceRecoloringMaxInterference &&
+        !ExhaustiveSearch)
+      return false;
+    RecoloringCandidates.insert_range(Query.interferingVRegs());
+  }
+  if (RecoloringCandidates.empty())
+    return false;
+
+  SmallVector<std::pair<const LiveInterval *, MCRegister>, 4> OldAssignments;
+  SmallVector<HintsInfo, 4> HintInfos;
+  BlockFrequency OldCost(0);
+  for (const LiveInterval *Intf : RecoloringCandidates) {
+    MCRegister OldPhys = VRM->getPhys(Intf->reg());
+    if (!OldPhys)
+      return false;
+    OldAssignments.emplace_back(Intf, OldPhys);
+    HintInfos.emplace_back();
+    collectHintInfo(Intf->reg(), HintInfos.back());
+    OldCost += getBrokenHintFreq(HintInfos.back(), OldPhys);
+  }
+  for (const auto &[Intf, OldPhys] : OldAssignments)
+    Matrix->unassign(*Intf);
+
+  bool Success = true;
+  BlockFrequency NewCost(0);
+  for (const auto &[Index, Assignment] : llvm::enumerate(OldAssignments)) {
+    const LiveInterval *Intf = Assignment.first;
+    MCRegister OldPhys = Assignment.second;
+    MCRegister NewPhys;
+    BlockFrequency BestCost = BlockFrequency::max();
+    auto Order =
+        AllocationOrder::create(Intf->reg(), *VRM, RegClassInfo, Matrix);
+    for (MCRegister Candidate : Order) {
+      if (TRI->regsOverlap(Candidate, PhysHint) ||
+          RegCosts[Candidate.id()] > RegCosts[OldPhys.id()] ||
+          RegClassInfo.getLastCalleeSavedAlias(Candidate) ||
+          Matrix->checkInterference(*Intf, Candidate) != LiveRegMatrix::IK_Free)
+        continue;
+      BlockFrequency Cost = getBrokenHintFreq(HintInfos[Index], Candidate);
+      if (!NewPhys || Cost < BestCost) {
+        NewPhys = Candidate;
+        BestCost = Cost;
+      }
+    }
+    if (!NewPhys) {
+      Success = false;
+      break;
+    }
+    Matrix->assign(*Intf, NewPhys);
+  }
+
+  if (Success)
+    for (const auto &Assignment : OldAssignments) {
+      const LiveInterval *Intf = Assignment.first;
+      HintsInfo NewHintInfo;
+      collectHintInfo(Intf->reg(), NewHintInfo);
+      NewCost += getBrokenHintFreq(NewHintInfo, VRM->getPhys(Intf->reg()));
+    }
+
+  if (Success && NewCost <= OldCost + MaxCostIncrease) {
+    LLVM_DEBUG(dbgs() << "Freed physical hint " << printReg(PhysHint, TRI)
+                      << " over [" << HintRange.beginIndex() << ','
+                      << HintRange.endIndex() << ")\n");
+    return true;
+  }
+
+  for (auto [Intf, OldPhys] : llvm::reverse(OldAssignments))
+    if (VRM->hasPhys(Intf->reg()))
+      Matrix->unassign(*Intf);
+  for (auto [Intf, OldPhys] : OldAssignments)
+    Matrix->assign(*Intf, OldPhys);
+  return false;
+}
+
+/// Try to repair a broken physical register hint. Prefer assigning the whole
+/// live interval to the hint when that lowers register cost. When whole-range
+/// reassignment is not possible, try to make a physical-source copy
+/// optimizable by freeing its local range.
+void RAGreedy::tryRecoloringForPhysicalHint(const LiveInterval &VirtReg,
+                                            MCRegister PhysHint) {
+  MCRegister CurrPhys = VRM->getPhys(VirtReg.reg());
+  const TargetRegisterClass *RC = MRI->getRegClass(VirtReg.reg());
+  if (!CurrPhys || TRI->regsOverlap(CurrPhys, PhysHint) ||
+      !RC->contains(PhysHint) || MRI->isReserved(PhysHint))
+    return;
+
+  MCRegister CurrCSR = RegClassInfo.getLastCalleeSavedAlias(CurrPhys);
+  MCRegister HintCSR = RegClassInfo.getLastCalleeSavedAlias(PhysHint);
+  bool ReducesCSRCost = false;
+  if (CurrCSR && !HintCSR) {
+    Matrix->unassign(VirtReg);
+    ReducesCSRCost = !Matrix->isPhysRegUsed(CurrPhys);
+    Matrix->assign(VirtReg, CurrPhys);
+  }
+  // Reassigning an entire interval solely for an equal-cost copy hint can
+  // cause widespread register churn. Only do so when it removes a CSR use.
+  // Local copy repair below can still operate on an equal-cost hint when it
+  // enables a concrete copy optimization.
+  if (ReducesCSRCost) {
+    HintsInfo VirtRegHints;
+    collectHintInfo(VirtReg.reg(), VirtRegHints);
+    BlockFrequency OldVirtRegCost = getBrokenHintFreq(VirtRegHints, CurrPhys);
+    BlockFrequency NewVirtRegCost = getBrokenHintFreq(VirtRegHints, PhysHint);
+    if (NewVirtRegCost <= OldVirtRegCost) {
+      LiveRegMatrix::InterferenceKind IK =
+          Matrix->checkInterference(VirtReg, PhysHint);
+      if (IK == LiveRegMatrix::IK_Free ||
+          (IK == LiveRegMatrix::IK_VirtReg &&
+           recolorPhysicalHintInterferences(VirtReg, PhysHint,
+                                            OldVirtRegCost - NewVirtRegCost))) {
+        Matrix->unassign(VirtReg);
+        Matrix->assign(VirtReg, PhysHint);
+        LLVM_DEBUG(dbgs() << "Recolored " << printReg(VirtReg.reg(), TRI)
+                          << " to physical hint " << printReg(PhysHint, TRI)
+                          << '\n');
+        ++NumPhysicalHintRecolors;
+        return;
+      }
+    }
+  }
+
+  // Restrict partial repair to a callee-saved interval whose physical hint
+  // is call-clobbered. Other equal-cost moves tend to cause widespread
+  // register churn without reducing save/restore cost.
+  if (!CurrCSR || HintCSR)
+    return;
+
+  MachineInstr *Copy = MRI->getUniqueVRegDef(VirtReg.reg());
+  if (!Copy || !Copy->isCopy() || !TII->shouldPostRASink(*Copy) ||
+      Copy->getNumOperands() < 2 || !Copy->getOperand(0).isReg() ||
+      !Copy->getOperand(1).isReg() ||
+      Copy->getOperand(0).getReg() != VirtReg.reg() ||
+      Copy->getOperand(0).getSubReg() ||
+      Copy->getOperand(1).getReg() != PhysHint ||
+      Copy->getOperand(1).getSubReg() || Copy->getParent()->succ_size() < 2)
+    return;
+
+  MachineBasicBlock *CopyMBB = Copy->getParent();
+  MachineBasicBlock *LiveInSucc = nullptr;
+  for (MachineBasicBlock *Succ : CopyMBB->successors()) {
+    if (!VirtReg.liveAt(Indexes->getMBBStartIdx(Succ)))
+      continue;
+    if (LiveInSucc || Succ->pred_size() != 1)
+      return;
+    LiveInSucc = Succ;
+  }
+  if (!LiveInSucc)
+    return;
+
+  SlotIndex LastLocalUse = LIS->getInstructionIndex(*Copy).getDeadSlot();
+  for (const MachineOperand &MO : MRI->use_nodbg_operands(VirtReg.reg())) {
+    const MachineInstr &MI = *MO.getParent();
+    if (MI.getParent() != CopyMBB)
+      continue;
+    const TargetRegisterClass *UseRC =
+        MI.getRegClassConstraint(MO.getOperandNo(), TII, TRI);
+    if (MO.isImplicit() || MO.isTied() || MO.isUndef() ||
+        MI.hasExtraSrcRegAllocReq(MachineInstr::IgnoreBundle) || !UseRC ||
+        !UseRC->contains(PhysHint))
+      return;
+    LastLocalUse =
+        std::max(LastLocalUse, LIS->getInstructionIndex(MI).getDeadSlot());
+  }
+
+  for (auto I = std::next(Copy->getIterator()), E = CopyMBB->instr_end();
+       I != E; ++I)
+    if (I->isCall() || I->modifiesRegister(PhysHint, TRI))
+      return;
+
+  LiveRange HintRange;
+  SlotIndex MBBEnd = Indexes->getMBBEndIdx(CopyMBB);
+  VNInfo HintValue(0, LastLocalUse);
+  HintRange.addSegment(LiveRange::Segment(LastLocalUse, MBBEnd, &HintValue));
+  if (recolorPhysicalHintInterferences(HintRange, PhysHint, BlockFrequency(0)))
+    ++NumPhysicalHintRecolors;
+}
+
 //===----------------------------------------------------------------------===//
 //                            Main Entry Point
 //===----------------------------------------------------------------------===//
@@ -2650,6 +2840,16 @@ void RAGreedy::tryHintsRecoloring() {
       continue;
     tryHintRecoloring(*LI);
   }
+
+  // Try to repair physical hints after normal hint reconciliation so later
+  // recoloring cannot undo the result.
+  for (const LiveInterval *LI : SetOfBrokenHints) {
+    if (!VRM->hasPhys(LI->reg()))
+      continue;
+    Register Hint = MRI->getSimpleHint(LI->reg());
+    if (Hint && Hint.isPhysical())
+      tryRecoloringForPhysicalHint(*LI, Hint.asMCReg());
+  }
 }
 
 MCRegister RAGreedy::selectOrSplitImpl(const LiveInterval &VirtReg,
diff --git a/llvm/lib/CodeGen/RegAllocGreedy.h b/llvm/lib/CodeGen/RegAllocGreedy.h
index 465be0d76809e..5f76e259f18b9 100644
--- a/llvm/lib/CodeGen/RegAllocGreedy.h
+++ b/llvm/lib/CodeGen/RegAllocGreedy.h
@@ -396,6 +396,9 @@ class LLVM_LIBRARY_VISIBILITY RAGreedy : public RegAllocBase,
 
   BlockFrequency getBrokenHintFreq(const HintsInfo &, MCRegister);
   void collectHintInfo(Register, HintsInfo &);
+  bool recolorPhysicalHintInterferences(const LiveRange &, MCRegister,
+                                        BlockFrequency);
+  void tryRecoloringForPhysicalHint(const LiveInterval &, MCRegister);
 
   /// Greedy RA statistic to remark.
   struct RAGreedyStats {
diff --git a/llvm/test/CodeGen/RISCV/ctlz-cttz-ctpop.ll b/llvm/test/CodeGen/RISCV/ctlz-cttz-ctpop.ll
index 45b29fd970ef3..56c825acdba39 100644
--- a/llvm/test/CodeGen/RISCV/ctlz-cttz-ctpop.ll
+++ b/llvm/test/CodeGen/RISCV/ctlz-cttz-ctpop.ll
@@ -355,6 +355,9 @@ define i32 @test_cttz_i32(i32 %a) nounwind {
 define i64 @test_cttz_i64(i64 %a) nounwind {
 ; RV32I-LABEL: test_cttz_i64:
 ; RV32I:       # %bb.0:
+; RV32I-NEXT:    or a2, a0, a1
+; RV32I-NEXT:    beqz a2, .LBB3_3
+; RV32I-NEXT:  # %bb.1: # %cond.false
 ; RV32I-NEXT:    addi sp, sp, -32
 ; RV32I-NEXT:    sw ra, 28(sp) # 4-byte Folded Spill
 ; RV32I-NEXT:    sw s0, 24(sp) # 4-byte Folded Spill
@@ -363,9 +366,6 @@ define i64 @test_cttz_i64(i64 %a) nounwind {
 ; RV32I-NEXT:    sw s3, 12(sp) # 4-byte Folded Spill
 ; RV32I-NEXT:    sw s4, 8(sp) # 4-byte Folded Spill
 ; RV32I-NEXT:    mv s0, a1
-; RV32I-NEXT:    or a1, a0, a1
-; RV32I-NEXT:    beqz a1, .LBB3_3
-; RV32I-NEXT:  # %bb.1: # %cond.false
 ; RV32I-NEXT:    neg a1, a0
 ; RV32I-NEXT:    and a1, a0, a1
 ; RV32I-NEXT:    lui s2, 30667
@@ -390,13 +390,13 @@ define i64 @test_cttz_i64(i64 %a) nounwind {
 ; RV32I-NEXT:    j .LBB3_5
 ; RV32I-NEXT:  .LBB3_3:
 ; RV32I-NEXT:    li a0, 64
-; RV32I-NEXT:    j .LBB3_5
+; RV32I-NEXT:    li a1, 0
+; RV32I-NEXT:    ret
 ; RV32I-NEXT:  .LBB3_4:
 ; RV32I-NEXT:    srli s1, s1, 27
 ; RV32I-NEXT:    add s1, s3, s1
 ; RV32I-NEXT:    lbu a0, 0(s1)
-; RV32I-NEXT:  .LBB3_5: # %cond.end
-; RV32I-NEXT:    li a1, 0
+; RV32I-NEXT:  .LBB3_5: # %cond.false
 ; RV32I-NEXT:    lw ra, 28(sp) # 4-byte Folded Reload
 ; RV32I-NEXT:    lw s0, 24(sp) # 4-byte Folded Reload
 ; RV32I-NEXT:    lw s1, 20(sp) # 4-byte Folded Reload
@@ -404,6 +404,7 @@ define i64 @test_cttz_i64(i64 %a) nounwind {
 ; RV32I-NEXT:    lw s3, 12(sp) # 4-byte Folded Reload
 ; RV32I-NEXT:    lw s4, 8(sp) # 4-byte Folded Reload
 ; RV32I-NEXT:    addi sp, sp, 32
+; RV32I-NEXT:    li a1, 0
 ; RV32I-NEXT:    ret
 ;
 ; RV64I-LABEL: test_cttz_i64:
diff --git a/llvm/test/CodeGen/RISCV/overflow-intrinsics.ll b/llvm/test/CodeGen/RISCV/overflow-intrinsics.ll
index aef7b82eb78b4..1e7a5c2817644 100644
--- a/llvm/test/CodeGen/RISCV/overflow-intrinsics.ll
+++ b/llvm/test/CodeGen/RISCV/overflow-intrinsics.ll
@@ -1090,15 +1090,15 @@ define i1 @usubo_ult_cmp_dominates_i64(i64 %x, i64 %y, ptr %p, i1 %cond) {
 ; RV32-NEXT:    .cfi_offset s5, -28
 ; RV32-NEXT:    .cfi_offset s6, -32
 ; RV32-NEXT:    mv s1, a5
-; RV32-NEXT:    mv s4, a1
-; RV32-NEXT:    andi a1, a5, 1
-; RV32-NEXT:    beqz a1, .LBB32_6
+; RV32-NEXT:    mv s2, a2
+; RV32-NEXT:    andi a2, a5, 1
+; RV32-NEXT:    beqz a2, .LBB32_6
 ; RV32-NEXT:  # %bb.1: # %t
 ; RV32-NEXT:    mv s0, a4
 ; RV32-NEXT:    mv s3, a3
-; RV32-NEXT:    mv s2, a2
+; RV32-NEXT:    mv s4, a1
 ; RV32-NEXT:    mv s5, a0
-; RV32-NEXT:    beq s4, a3, .LBB32_3
+; RV32-NEXT:    beq a1, a3, .LBB32_3
 ; RV32-NEXT:  # %bb.2: # %t
 ; RV32-NEXT:    sltu s6, s4, s3
 ; RV32-NEXT:    j .LBB32_4
@@ -1157,13 +1157,13 @@ define i1 @usubo_ult_cmp_dominates_i64(i64 %x, i64 %y, ptr %p, i1 %cond) {
 ; RV64-NEXT:    .cfi_offset s3, -40
 ; RV64-NEXT:    .cfi_offset s4, -48
 ; RV64-NEXT:    mv s1, a3
-; RV64-NEXT:    mv s2, a1
-; RV64-NEXT:    andi a1, a3, 1
-; RV64-NEXT:    beqz a1, .LBB32_3
-; RV64-NEXT:  # %bb.1: # %t
 ; RV64-NEXT:    mv s0, a2
+; RV64-NEXT:    andi a2, a3, 1
+; RV64-NEXT:    beqz a2, .LBB32_3
+; RV64-NEXT:  # %bb.1: # %t
+; RV64-NEXT:    mv s2, a1
 ; RV64-NEXT:    mv s3, a0
-; RV64-NEXT:    sltu s4, a0, s2
+; RV64-NEXT:    sltu s4, a0, a1
 ; RV64-NEXT:    mv a0, s4
 ; RV64-NEXT:    call call
 ; RV64-NEXT:    bgeu s3, s2, .LBB32_3
diff --git a/llvm/test/CodeGen/RISCV/regalloc-physical-hint-recolor.mir b/llvm/test/CodeGen/RISCV/regalloc-physical-hint-recolor.mir
new file mode 100644
index 0000000000000..6a3264149daae
--- /dev/null
+++ b/llvm/test/CodeGen/RISCV/regalloc-physical-hint-recolor.mir
@@ -0,0 +1,39 @@
+# RUN: llc -mtriple=riscv64 -mattr=+reserve-x9,+reserve-x11,+reserve-x12,+reserve-x13,+reserve-x14,+reserve-x15 -run-pass=greedy,virtregrewriter -regalloc-enable-priority-advisor=dummy -verify-machineinstrs -o - %s | FileCheck %s
+
+# Verify that the final broken-hint reconciliation can move virtual
+# interferences out of a physical hint and assign the complete hinted interval
+# to that register. This test does not depend on post-RA copy sinking.
+
+---
+name: direct_physical_hint_recolor
+tracksRegLiveness: true
+registers:
+  - { id: 0, class: gpr }
+  - { id: 1, class: gprc }
+  - { id: 2, class: gpr }
+  - { id: 3, class: gpr }
+  - { id: 4, class: gpr }
+body: |
+  bb.0:
+    liveins: $x10
+
+    %0:gpr = COPY $x10
+    %1:gprc = COPY $x10
+    %2:gpr = ADD %0, %0
+    %3:gpr = ADD %2, %0
+    %4:gpr = ADD %3, %1
+    $x10 = COPY %1
+    $x11 = COPY %4
+    PseudoRET implicit $x10, implicit $x11
+
+    ; CHECK-LABEL: name: direct_physical_hint_recolor
+    ; CHECK: bb.0:
+    ; CHECK-NEXT: liveins: $x10
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: renamable $x17 = COPY $x10
+    ; CHECK-NEXT: renamable $x16 = ADD renamable $x17, renamable $x17
+    ; CHECK-NEXT: renamable $x16 = ADD killed renamable $x16, killed renamable $x17
+    ; CHECK-NEXT: renamable $x16 = ADD killed renamable $x16, renamable $x10
+    ; CHECK-NEXT: $x11 = COPY killed renamable $x16
+    ; CHECK-NEXT: PseudoRET implicit $x10, implicit $x11
+...
diff --git a/llvm/test/CodeGen/RISCV/rv32xtheadbb.ll b/llvm/test/CodeGen/RISCV/rv32xtheadbb.ll
index 3ba5312f97626..ec01e26388838 100644
--- a/llvm/test/CodeGen/RISCV/rv32xtheadbb.ll
+++ b/llvm/test/CodeGen/RISCV/rv32xtheadbb.ll
@@ -215,6 +215,9 @@ define i32 @cttz_i32(i32 %a) nounwind {
 define i64 @cttz_i64(i64 %a) nounwind {
 ; RV32I-LABEL: cttz_i64:
 ; RV32I:       # %bb.0:
+; RV32I-NEXT:    or a2, a0, a1
+; RV32I-NEXT:    beqz a2, .LBB3_3
+; RV32I-NEXT:  # %bb.1: # %cond.false
 ; RV32I-NEXT:    addi sp, sp, -32
 ; RV32I-NEXT:    sw ra, 28(sp) # 4-byte Folded Spill
 ; RV32I-NEXT:    sw s0, 24(sp) # 4-byte Folded Spill
@@ -223,9 +226,6 @@ define i64 @cttz_i64(i64 %a) nounwind {
 ; RV32I-NEXT:    sw s3, 12(sp) # 4-byte Folded Spill
 ; RV32I-NEXT:    sw s4, 8(sp) # 4-byte Folded Spill
 ; RV32I-NEXT:    mv s0, a1
-; RV32I-NEXT:    or a1, a0, a1
-; RV32I-NEXT:    beqz a1, .LBB3_3
-; RV32I-NEXT:  # %bb.1: # %cond.false
 ; RV32I-NEXT:    neg a1, a0
 ; RV32I-NEXT:    and a1, a0, a1
 ; RV32I-NEXT:    lui s2, 30667
@@ -250,13 +250,13 @@ define i64 @cttz_i64(i64 %a) nounwind {
 ; RV32I-NEXT:    j .LBB3_5
 ; RV32I-NEXT:  .LBB3_3:
 ; RV32I-NEXT:    li a0, 64
-; RV32I-NEXT:    j .LBB3_5
+; RV32I-NEXT:    li a1, 0
+; RV32I-NEXT:    ret
 ; RV32I-NEXT:  .LBB3_4:
 ; RV32I-NEXT:    srli s1, s1, 27
 ; RV32I-NEXT:    add s1, s3, s1
 ; RV32I-NEXT:    lbu a0, 0(s1)
-; RV32I-NEXT:  .LBB3_5: # %cond.end
-; RV32I-NEXT:    li a1, 0
+; RV32I-NEXT:  .LBB3_5: # %cond.false
 ; RV32I-NEXT:    lw ra, 28(sp) # 4-byte Folded Reload
 ; RV32I-NEXT:    lw s0, 24(sp) # 4-byte Folded Reload
 ; RV32I-NEXT:    lw s1, 20(sp) # 4-byte Folded Reload
@@ -264,6 +264,7 @@ define i64 @cttz_i64(i64 %a) nounwind {
 ; RV32I-NEXT:    lw s3, 12(sp) # 4-byte Folded Reload
 ; RV32I-NEXT:    lw s4, 8(sp) # 4-byte Folded Reload
 ; RV32I-NEXT:    addi sp, sp, 32
+; RV32I-NEXT:    li a1, 0
 ; RV32I-NEXT:    ret
 ;
 ; RV32XTHEADBB-NOB-LABEL: cttz_i64:
diff --git a/llvm/test/CodeGen/RISCV/rv32zbb.ll b/llvm/test/CodeGen/RISCV/rv32zbb.ll
index 6164065d02d94..2274c30b65df9 100644
--- a/llvm/test/CodeGen/RISCV/rv32zbb.ll
+++ b/llvm/test/CodeGen/RISCV/rv32zbb.ll
@@ -182,6 +182,9 @@ define i32 @cttz_i32(i32 %a) nounwind {
 define i64 @cttz_i64(i64 %a) nounwind {
 ; RV32I-LABEL: cttz_i64:
 ; RV32I:       # %bb.0:
+; RV32I-NEXT:    or a2, a0, a1
+; RV32I-NEXT:    beqz a2, .LBB3_3
+; RV32I-NEXT:  # %bb.1: # %cond.false
 ; RV32I-NEXT:    addi sp, sp, -32
 ; RV32I-NEXT:    sw ra, 28(sp) # 4-byte Folded Spill
 ; RV32I-NEXT:    sw s0, 24(sp) # 4-byte Folded Spill
@@ -190,9 +193,6 @@ define i64 @cttz_i64(i64 %a) nounwind {
 ; RV32I-NEXT:    sw s3, 12(sp) # 4-byte Folded Spill
 ; RV32I-NEXT:    sw s4, 8(sp) # 4-byte Folded Spill
 ; RV32I-NEXT:    mv s0, a1
-; RV32I-NEXT:    or a1, a0, a1
-; RV32I-NEXT:    beqz a1, .LBB3_3
-; RV32I-NEXT:  # %bb.1: # %cond.false
 ; RV32I-NEXT:    neg a1, a0
 ; RV32I-NEXT:    and a1, a0, a1
 ; RV32I-NEXT:    lui s2, 30667
@@ -217,13 +217,13 @@ define i64 @cttz_i64(i64 %a) nounwind {
 ; RV32I-NEXT:    j .LBB3_5
 ; RV32I-NEXT:  .LBB3_3:
 ; RV32I-NEXT:    li a0, 64
-; RV32I-NEXT:    j .LBB3_5
+; RV32I-NEXT:    li a1, 0
+; RV32I-NEXT:    ret
 ; RV32I-NEXT:  .LBB3_4:
 ; RV32I-NEXT:    srli s1, s1, 27
 ; RV32I-NEXT:    add s1, s3, s1
 ; RV32I-NEXT:    lbu a0, 0(s1)
-; RV32I-NEXT:  .LBB3_5: # %cond.end
-; RV32I-NEXT:    li a1, 0
+; RV32I-NEXT:  .LBB3_5: # %cond.false
 ; RV32I-NEXT:    lw ra, 28(sp) # 4-byte Folded Reload
 ; RV32I-NEXT:    lw s0, 24(sp) # 4-byte Folded Reload
 ; RV32I-NEXT:    lw s1, 20(sp) # 4-byte Folded Reload
@@ -231,6 +231,7 @@ define i64 @cttz_i64(i64 %a) nounwind {
 ; RV32I-NEXT:    lw s3, 12(sp) # 4-byte Folded Reload
 ; RV32I-NEXT:    lw s4, 8(sp) # 4-byte Folded Reload
 ; RV32I-NEXT:    addi sp, sp, 32
+; RV32I-NEXT:    li a1, 0
 ; RV32I-NEXT:    ret
 ;
 ; RV32ZBB-LABEL: cttz_i64:



More information about the llvm-commits mailing list