[llvm] [RegisterCoalescer] Narrow full COPYs with dead destination lanes (PR #199631)

Yuyang Zhang via llvm-commits llvm-commits at lists.llvm.org
Tue Jun 30 02:03:57 PDT 2026


https://github.com/yuyzhang512 updated https://github.com/llvm/llvm-project/pull/199631

>From 70a9b203ccfe0fda3a9af97986b680d3796010c8 Mon Sep 17 00:00:00 2001
From: yuyzhang512 <yuyzhang at amd.com>
Date: Tue, 30 Jun 2026 08:27:01 +0000
Subject: [PATCH] [RegisterCoalescer] Narrow full COPYs with dead destination
 lanes

A full `%dst = COPY %src` can define lanes that are dead at the copy:
overwritten by a later subreg def before any use. The full-width COPY def
still interferes on those lanes, which blocks coalescing the overwriting
defs and leaves redundant copies behind.

Narrow such a COPY to the surviving lanes so the overwriting defs can
coalesce on a later attempt:

  %dst = COPY %src
  %dst.sub0 = ...        ; overwrites sub0 before use
  -->
  %dst.sub1 = COPY %src.sub1

The pattern is common with SelectionDAG vector lowering. insertelement on a
shared base vector expands to a full COPY of the base followed by a subreg
def that overwrites the inserted lane, so the copied base lane is dead. On a
gfx1250 tensor-load kernel this removes 5 s_mov (29 -> 24).

GlobalISel does not hit this: it builds each vector from scalar operands, so
the dead-lane copies never form.
---
 llvm/lib/CodeGen/RegisterCoalescer.cpp        | 46 +++++++++++++++++++
 .../AMDGPU/coalesce-narrow-dead-lane-copy.mir | 27 +++++++++++
 2 files changed, 73 insertions(+)
 create mode 100644 llvm/test/CodeGen/AMDGPU/coalesce-narrow-dead-lane-copy.mir

diff --git a/llvm/lib/CodeGen/RegisterCoalescer.cpp b/llvm/lib/CodeGen/RegisterCoalescer.cpp
index 4b4ba2144f964..1e1479bab1054 100644
--- a/llvm/lib/CodeGen/RegisterCoalescer.cpp
+++ b/llvm/lib/CodeGen/RegisterCoalescer.cpp
@@ -2220,6 +2220,52 @@ bool RegisterCoalescer::joinCopy(
       if (removePartialRedundancy(CP, *CopyMI))
         return true;
 
+    // Narrow a full `%dst = COPY %src` whose destination has lanes that are
+    // dead at the copy (overwritten before use): the dead-lane def causes false
+    // interference, so narrow to the surviving lanes and let the overwriting
+    // defs coalesce.
+    if (!CP.isPartial() && !CP.isPhys()) {
+      Register CopySrcReg = CopyMI->getOperand(1).getReg();
+      Register CopyDstReg = CopyMI->getOperand(0).getReg();
+      LiveInterval &SrcLI = LIS->getInterval(CopySrcReg);
+      LiveInterval &DstLI = LIS->getInterval(CopyDstReg);
+      if (SrcLI.hasSubRanges() && DstLI.hasSubRanges()) {
+        SlotIndex CopyIdx = LIS->getInstructionIndex(*CopyMI).getRegSlot();
+        // Lanes the source provides a value for.
+        LaneBitmask SrcLanes = LaneBitmask::getNone();
+        for (auto &SR : SrcLI.subranges())
+          SrcLanes |= SR.LaneMask;
+        // Drop those whose destination def is dead (overwritten before use).
+        LaneBitmask SurvivingLanes = SrcLanes;
+        for (auto &SR : DstLI.subranges())
+          if (SR.Query(CopyIdx).isDeadDef())
+            SurvivingLanes &= ~SR.LaneMask;
+
+        if (SurvivingLanes.any() && SurvivingLanes != SrcLanes) {
+          const TargetRegisterClass *SrcRC = MRI->getRegClass(CopySrcReg);
+          const TargetRegisterClass *DstRC = MRI->getRegClass(CopyDstReg);
+          SmallVector<unsigned, 4> SubRegIdxs;
+          if (TRI->getCoveringSubRegIndexes(SrcRC, SurvivingLanes,
+                                            SubRegIdxs) &&
+              SubRegIdxs.size() == 1 &&
+              TRI->getSubClassWithSubReg(DstRC, SubRegIdxs[0])) {
+            unsigned SubIdx = SubRegIdxs[0];
+            LLVM_DEBUG(dbgs() << "\tNarrowing COPY to "
+                              << TRI->getSubRegIndexName(SubIdx) << '\n');
+            CopyMI->getOperand(0).setSubReg(SubIdx);
+            CopyMI->getOperand(0).setIsUndef(true);
+            CopyMI->getOperand(1).setSubReg(SubIdx);
+            LIS->removeInterval(CopyDstReg);
+            LIS->createAndComputeVirtRegInterval(CopyDstReg);
+            LIS->removeInterval(CopySrcReg);
+            LIS->createAndComputeVirtRegInterval(CopySrcReg);
+            Again = true;
+            return false;
+          }
+        }
+      }
+    }
+
     // Otherwise, we are unable to join the intervals.
     LLVM_DEBUG(dbgs() << "\tInterference!\n");
     Again = true; // May be possible to coalesce later.
diff --git a/llvm/test/CodeGen/AMDGPU/coalesce-narrow-dead-lane-copy.mir b/llvm/test/CodeGen/AMDGPU/coalesce-narrow-dead-lane-copy.mir
new file mode 100644
index 0000000000000..87d7f9d2bbc6b
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/coalesce-narrow-dead-lane-copy.mir
@@ -0,0 +1,27 @@
+# NOTE: Assertions have been autogenerated by utils/update_mir_test_checks.py UTC_ARGS: --version 6
+# RUN: llc -mtriple=amdgcn -mcpu=gfx1250 -run-pass=register-coalescer -o - %s | FileCheck %s
+
+# %4 = COPY %3 copies both lanes, but %4.sub0 = COPY %2 overwrites sub0 before
+# any use, so the COPY's sub0 def is dead. The full-width COPY blocks coalescing
+# %2 into %4.sub0. Check the COPY is narrowed to its surviving lane (sub1), which
+# lets %4.sub0 = COPY %2 coalesce away: the four input COPYs become one.
+---
+name:            narrow_copy_dead_dst_lane
+tracksRegLiveness: true
+body:             |
+  bb.0:
+    ; CHECK-LABEL: name: narrow_copy_dead_dst_lane
+    ; CHECK: undef [[S_MOV_B32_:%[0-9]+]].sub0:sgpr_128 = S_MOV_B32 1
+    ; CHECK-NEXT: [[S_MOV_B32_:%[0-9]+]].sub1:sgpr_128 = S_MOV_B32 2
+    ; CHECK-NEXT: undef [[S_MOV_B32_1:%[0-9]+]].sub0:sgpr_128 = S_MOV_B32 3
+    ; CHECK-NEXT: [[S_MOV_B32_1:%[0-9]+]].sub1:sgpr_128 = COPY [[S_MOV_B32_]].sub1
+    ; CHECK-NEXT: S_ENDPGM 0, implicit [[S_MOV_B32_]], implicit [[S_MOV_B32_1]]
+    %0:sreg_32 = S_MOV_B32 1
+    %1:sreg_32 = S_MOV_B32 2
+    %2:sreg_32 = S_MOV_B32 3
+    undef %3.sub0:sgpr_128 = COPY %0
+    %3.sub1:sgpr_128 = COPY %1
+    %4:sgpr_128 = COPY %3
+    %4.sub0:sgpr_128 = COPY %2
+    S_ENDPGM 0, implicit %3, implicit %4
+...



More information about the llvm-commits mailing list