[llvm] [RegisterCoalescer] Narrow full COPYs with dead destination lanes (PR #199631)
Yuyang Zhang via llvm-commits
llvm-commits at lists.llvm.org
Tue Jun 30 02:03:57 PDT 2026
https://github.com/yuyzhang512 updated https://github.com/llvm/llvm-project/pull/199631
>From 70a9b203ccfe0fda3a9af97986b680d3796010c8 Mon Sep 17 00:00:00 2001
From: yuyzhang512 <yuyzhang at amd.com>
Date: Tue, 30 Jun 2026 08:27:01 +0000
Subject: [PATCH] [RegisterCoalescer] Narrow full COPYs with dead destination
lanes
A full `%dst = COPY %src` can define lanes that are dead at the copy:
overwritten by a later subreg def before any use. The full-width COPY def
still interferes on those lanes, which blocks coalescing the overwriting
defs and leaves redundant copies behind.
Narrow such a COPY to the surviving lanes so the overwriting defs can
coalesce on a later attempt:
%dst = COPY %src
%dst.sub0 = ... ; overwrites sub0 before use
-->
%dst.sub1 = COPY %src.sub1
The pattern is common with SelectionDAG vector lowering. insertelement on a
shared base vector expands to a full COPY of the base followed by a subreg
def that overwrites the inserted lane, so the copied base lane is dead. On a
gfx1250 tensor-load kernel this removes 5 s_mov (29 -> 24).
GlobalISel does not hit this: it builds each vector from scalar operands, so
the dead-lane copies never form.
---
llvm/lib/CodeGen/RegisterCoalescer.cpp | 46 +++++++++++++++++++
.../AMDGPU/coalesce-narrow-dead-lane-copy.mir | 27 +++++++++++
2 files changed, 73 insertions(+)
create mode 100644 llvm/test/CodeGen/AMDGPU/coalesce-narrow-dead-lane-copy.mir
diff --git a/llvm/lib/CodeGen/RegisterCoalescer.cpp b/llvm/lib/CodeGen/RegisterCoalescer.cpp
index 4b4ba2144f964..1e1479bab1054 100644
--- a/llvm/lib/CodeGen/RegisterCoalescer.cpp
+++ b/llvm/lib/CodeGen/RegisterCoalescer.cpp
@@ -2220,6 +2220,52 @@ bool RegisterCoalescer::joinCopy(
if (removePartialRedundancy(CP, *CopyMI))
return true;
+ // Narrow a full `%dst = COPY %src` whose destination has lanes that are
+ // dead at the copy (overwritten before use): the dead-lane def causes false
+ // interference, so narrow to the surviving lanes and let the overwriting
+ // defs coalesce.
+ if (!CP.isPartial() && !CP.isPhys()) {
+ Register CopySrcReg = CopyMI->getOperand(1).getReg();
+ Register CopyDstReg = CopyMI->getOperand(0).getReg();
+ LiveInterval &SrcLI = LIS->getInterval(CopySrcReg);
+ LiveInterval &DstLI = LIS->getInterval(CopyDstReg);
+ if (SrcLI.hasSubRanges() && DstLI.hasSubRanges()) {
+ SlotIndex CopyIdx = LIS->getInstructionIndex(*CopyMI).getRegSlot();
+ // Lanes the source provides a value for.
+ LaneBitmask SrcLanes = LaneBitmask::getNone();
+ for (auto &SR : SrcLI.subranges())
+ SrcLanes |= SR.LaneMask;
+ // Drop those whose destination def is dead (overwritten before use).
+ LaneBitmask SurvivingLanes = SrcLanes;
+ for (auto &SR : DstLI.subranges())
+ if (SR.Query(CopyIdx).isDeadDef())
+ SurvivingLanes &= ~SR.LaneMask;
+
+ if (SurvivingLanes.any() && SurvivingLanes != SrcLanes) {
+ const TargetRegisterClass *SrcRC = MRI->getRegClass(CopySrcReg);
+ const TargetRegisterClass *DstRC = MRI->getRegClass(CopyDstReg);
+ SmallVector<unsigned, 4> SubRegIdxs;
+ if (TRI->getCoveringSubRegIndexes(SrcRC, SurvivingLanes,
+ SubRegIdxs) &&
+ SubRegIdxs.size() == 1 &&
+ TRI->getSubClassWithSubReg(DstRC, SubRegIdxs[0])) {
+ unsigned SubIdx = SubRegIdxs[0];
+ LLVM_DEBUG(dbgs() << "\tNarrowing COPY to "
+ << TRI->getSubRegIndexName(SubIdx) << '\n');
+ CopyMI->getOperand(0).setSubReg(SubIdx);
+ CopyMI->getOperand(0).setIsUndef(true);
+ CopyMI->getOperand(1).setSubReg(SubIdx);
+ LIS->removeInterval(CopyDstReg);
+ LIS->createAndComputeVirtRegInterval(CopyDstReg);
+ LIS->removeInterval(CopySrcReg);
+ LIS->createAndComputeVirtRegInterval(CopySrcReg);
+ Again = true;
+ return false;
+ }
+ }
+ }
+ }
+
// Otherwise, we are unable to join the intervals.
LLVM_DEBUG(dbgs() << "\tInterference!\n");
Again = true; // May be possible to coalesce later.
diff --git a/llvm/test/CodeGen/AMDGPU/coalesce-narrow-dead-lane-copy.mir b/llvm/test/CodeGen/AMDGPU/coalesce-narrow-dead-lane-copy.mir
new file mode 100644
index 0000000000000..87d7f9d2bbc6b
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/coalesce-narrow-dead-lane-copy.mir
@@ -0,0 +1,27 @@
+# NOTE: Assertions have been autogenerated by utils/update_mir_test_checks.py UTC_ARGS: --version 6
+# RUN: llc -mtriple=amdgcn -mcpu=gfx1250 -run-pass=register-coalescer -o - %s | FileCheck %s
+
+# %4 = COPY %3 copies both lanes, but %4.sub0 = COPY %2 overwrites sub0 before
+# any use, so the COPY's sub0 def is dead. The full-width COPY blocks coalescing
+# %2 into %4.sub0. Check the COPY is narrowed to its surviving lane (sub1), which
+# lets %4.sub0 = COPY %2 coalesce away: the four input COPYs become one.
+---
+name: narrow_copy_dead_dst_lane
+tracksRegLiveness: true
+body: |
+ bb.0:
+ ; CHECK-LABEL: name: narrow_copy_dead_dst_lane
+ ; CHECK: undef [[S_MOV_B32_:%[0-9]+]].sub0:sgpr_128 = S_MOV_B32 1
+ ; CHECK-NEXT: [[S_MOV_B32_:%[0-9]+]].sub1:sgpr_128 = S_MOV_B32 2
+ ; CHECK-NEXT: undef [[S_MOV_B32_1:%[0-9]+]].sub0:sgpr_128 = S_MOV_B32 3
+ ; CHECK-NEXT: [[S_MOV_B32_1:%[0-9]+]].sub1:sgpr_128 = COPY [[S_MOV_B32_]].sub1
+ ; CHECK-NEXT: S_ENDPGM 0, implicit [[S_MOV_B32_]], implicit [[S_MOV_B32_1]]
+ %0:sreg_32 = S_MOV_B32 1
+ %1:sreg_32 = S_MOV_B32 2
+ %2:sreg_32 = S_MOV_B32 3
+ undef %3.sub0:sgpr_128 = COPY %0
+ %3.sub1:sgpr_128 = COPY %1
+ %4:sgpr_128 = COPY %3
+ %4.sub0:sgpr_128 = COPY %2
+ S_ENDPGM 0, implicit %3, implicit %4
+...
More information about the llvm-commits
mailing list