[llvm] [AMDGPU] Fix DPP combine when a mov feeds several REG_SEQUENCE lanes (PR #217902)

Arseniy Obolenskiy via llvm-commits llvm-commits at lists.llvm.org
Sun Sep 13 13:13:24 PDT 2026


https://github.com/aobolensk updated https://github.com/llvm/llvm-project/pull/217902

>From 2e3aca7e4ce662a05be1de15b98a4e439c392b92 Mon Sep 17 00:00:00 2001
From: Arseniy Obolenskiy <arseniy.obolenskiy at amd.com>
Date: Fri, 21 Aug 2026 14:27:20 +0200
Subject: [PATCH 1/2] [AMDGPU] Fix DPP combine when a mov feeds several
 REG_SEQUENCE lanes

The subreg index came from the first operand reading the mov, so the wrong operand was marked undef

Also skip lanes that other reads still use
---
 llvm/lib/Target/AMDGPU/GCNDPPCombine.cpp      |  48 ++++--
 .../test/CodeGen/AMDGPU/dpp_combine_gfx11.mir | 143 ++++++++++++++++++
 2 files changed, 180 insertions(+), 11 deletions(-)

diff --git a/llvm/lib/Target/AMDGPU/GCNDPPCombine.cpp b/llvm/lib/Target/AMDGPU/GCNDPPCombine.cpp
index 9d22757b4514a..502f82b0ad7f9 100644
--- a/llvm/lib/Target/AMDGPU/GCNDPPCombine.cpp
+++ b/llvm/lib/Target/AMDGPU/GCNDPPCombine.cpp
@@ -670,6 +670,8 @@ bool GCNDPPCombine::combineDPPMov(MachineInstr &MovMI) const {
 
   OrigMIs.push_back(&MovMI);
   bool Rollback = true;
+  bool AnyCombined = false;
+  const SIRegisterInfo *TRI = ST->getRegisterInfo();
   SmallVector<MachineOperand *, 16> Uses(
       llvm::make_pointer_range(MRI->use_nodbg_operands(DPPMovReg)));
 
@@ -684,8 +686,13 @@ bool GCNDPPCombine::combineDPPMov(MachineInstr &MovMI) const {
     assert((TII->get(OrigOp).getSize() != 4 || !AMDGPU::isTrue16Inst(OrigOp)) &&
            "There should not be e32 True16 instructions pre-RA");
     if (OrigOp == AMDGPU::REG_SEQUENCE) {
+      // A different reg is a forwarded lane of a nested REG_SEQUENCE.
+      if (Use->getReg() != DPPMovReg) {
+        LLVM_DEBUG(dbgs() << "  failed: nested REG_SEQUENCE\n");
+        break;
+      }
+
       Register FwdReg = OrigMI.getOperand(0).getReg();
-      unsigned FwdSubReg = 0;
 
       if (execMayBeModifiedBeforeAnyUse(*MRI, FwdReg, OrigMI)) {
         LLVM_DEBUG(dbgs() << "  failed: EXEC mask should remain the same"
@@ -693,21 +700,37 @@ bool GCNDPPCombine::combineDPPMov(MachineInstr &MovMI) const {
         break;
       }
 
-      unsigned OpNo, E = OrigMI.getNumOperands();
-      for (OpNo = 1; OpNo < E; OpNo += 2) {
-        if (OrigMI.getOperand(OpNo).getReg() == DPPMovReg) {
-          FwdSubReg = OrigMI.getOperand(OpNo + 1).getImm();
-          break;
-        }
+      // The DPP mov reg can appear in several operands.
+      unsigned OpNo = OrigMI.getOperandNo(Use);
+      unsigned FwdSubReg = OrigMI.getOperand(OpNo + 1).getImm();
+      auto It = RegSeqWithOpNos.find(&OrigMI);
+
+      // Duplicate subreg indices pass the verifier. Lane already queued.
+      if (It != RegSeqWithOpNos.end() &&
+          llvm::any_of(It->second, [&](unsigned N) {
+            return OrigMI.getOperand(N + 1).getImm() == FwdSubReg;
+          })) {
+        It->second.push_back(OpNo);
+        Rollback = false;
+        continue;
       }
 
-      if (!FwdSubReg)
+      // Marking the operand undef is unsound if another read consumes the lane.
+      LaneBitmask FwdLanes = TRI->getSubRegIndexLaneMask(FwdSubReg);
+      if (llvm::any_of(
+              MRI->use_nodbg_operands(FwdReg), [&](const MachineOperand &Op) {
+                return Op.getSubReg() != FwdSubReg &&
+                       (TRI->getSubRegIndexLaneMask(Op.getSubReg()) & FwdLanes)
+                           .any();
+              })) {
+        LLVM_DEBUG(dbgs() << "  failed: REG_SEQUENCE lane has other reads\n");
         break;
+      }
 
-      for (auto &Op : MRI->use_nodbg_operands(FwdReg)) {
+      for (MachineOperand &Op : MRI->use_nodbg_operands(FwdReg))
         if (Op.getSubReg() == FwdSubReg)
           Uses.push_back(&Op);
-      }
+
       RegSeqWithOpNos[&OrigMI].push_back(OpNo);
       continue;
     }
@@ -766,6 +789,7 @@ bool GCNDPPCombine::combineDPPMov(MachineInstr &MovMI) const {
                                         OldOpndValue, CombBCZ, IsShrinkable)) {
         DPPMIs.push_back(DPPInst);
         Rollback = false;
+        AnyCombined = true;
       }
     } else {
       assert(Use == Src1 && OrigMI.isCommutable()); // by check [1]
@@ -779,6 +803,7 @@ bool GCNDPPCombine::combineDPPMov(MachineInstr &MovMI) const {
                               IsShrinkable)) {
           DPPMIs.push_back(DPPInst);
           Rollback = false;
+          AnyCombined = true;
         }
       } else
         LLVM_DEBUG(dbgs() << "  failed: cannot be commuted\n");
@@ -789,7 +814,8 @@ bool GCNDPPCombine::combineDPPMov(MachineInstr &MovMI) const {
     OrigMIs.push_back(&OrigMI);
   }
 
-  Rollback |= !Uses.empty();
+  // A REG_SEQUENCE lane with no readers must not commit the erase of MovMI.
+  Rollback |= !Uses.empty() || !AnyCombined;
 
   for (auto *MI : *(Rollback? &DPPMIs : &OrigMIs))
     MI->eraseFromParent();
diff --git a/llvm/test/CodeGen/AMDGPU/dpp_combine_gfx11.mir b/llvm/test/CodeGen/AMDGPU/dpp_combine_gfx11.mir
index b7c884c7995e7..4f7009e421db0 100644
--- a/llvm/test/CodeGen/AMDGPU/dpp_combine_gfx11.mir
+++ b/llvm/test/CodeGen/AMDGPU/dpp_combine_gfx11.mir
@@ -731,6 +731,149 @@ body: |
     %7:vgpr_32 = V_ADDC_U32_e32 %5.sub1, %8, implicit-def $vcc, implicit $vcc, implicit $exec
 ...
 
+# the dpp mov feeds both lanes, each lane read via its own index
+# GCN-LABEL: name: dpp_reg_sequence_dup_combined
+# GCN: %5:vgpr_32 = V_ADD_U32_dpp %2, %0, %1, 1, 15, 15, 1, implicit $exec
+# GCN: %6:vgpr_32 = V_ADD_U32_dpp %2, %0, %1, 1, 15, 15, 1, implicit $exec
+# GCN-NOT: V_MOV_B32_dpp
+# GCN-NOT: REG_SEQUENCE
+name: dpp_reg_sequence_dup_combined
+tracksRegLiveness: true
+body: |
+  bb.0:
+    liveins: $vgpr0, $vgpr1
+
+    %0:vgpr_32 = COPY $vgpr0
+    %1:vgpr_32 = COPY $vgpr1
+    %2:vgpr_32 = IMPLICIT_DEF
+    %3:vgpr_32 = V_MOV_B32_dpp %2, %0, 1, 15, 15, 1, implicit $exec
+    %4:vreg_64 = REG_SEQUENCE %3, %subreg.sub0, %3, %subreg.sub1
+    %5:vgpr_32 = V_ADD_U32_e32 %4.sub0, %1, implicit $exec
+    %6:vgpr_32 = V_ADD_U32_e32 %4.sub1, %1, implicit $exec
+    S_ENDPGM 0, implicit %5, implicit %6
+...
+
+# duplicate subreg indices in one REG_SEQUENCE pass the verifier, so the lane
+# reaches the worklist through two operands - combine the use only once
+# GCN-LABEL: name: dpp_reg_sequence_same_subreg_dup
+# GCN: %5:vgpr_32 = V_ADD_U32_dpp %2, %0, %1, 1, 15, 15, 1, implicit $exec
+# GCN-NOT: V_MOV_B32_dpp
+# GCN-NOT: REG_SEQUENCE
+name: dpp_reg_sequence_same_subreg_dup
+tracksRegLiveness: true
+body: |
+  bb.0:
+    liveins: $vgpr0, $vgpr1
+
+    %0:vgpr_32 = COPY $vgpr0
+    %1:vgpr_32 = COPY $vgpr1
+    %2:vgpr_32 = IMPLICIT_DEF
+    %3:vgpr_32 = V_MOV_B32_dpp %2, %0, 1, 15, 15, 1, implicit $exec
+    %4:vreg_64 = REG_SEQUENCE %3, %subreg.sub0, %3, %subreg.sub0
+    %5:vgpr_32 = V_ADD_U32_e32 %4.sub0, %1, implicit $exec
+    S_ENDPGM 0, implicit %5
+...
+
+# REG_SEQUENCE fed by both lanes is read as a whole - do not combine
+# GCN-LABEL: name: dpp_reg_sequence_repeated_reg
+# GCN: %3:vgpr_32 = V_MOV_B32_dpp %2, %0, 1, 15, 15, 1, implicit $exec
+# GCN: %4:vgpr_32 = V_ADD_U32_e32 %3, %1, implicit $exec
+# GCN: %5:vreg_64 = REG_SEQUENCE %3, %subreg.sub0, %3, %subreg.sub1
+name: dpp_reg_sequence_repeated_reg
+tracksRegLiveness: true
+body: |
+  bb.0:
+    liveins: $vgpr0, $vgpr1
+
+    %0:vgpr_32 = COPY $vgpr0
+    %1:vgpr_32 = COPY $vgpr1
+    %2:vgpr_32 = IMPLICIT_DEF
+    %3:vgpr_32 = V_MOV_B32_dpp %2, %0, 1, 15, 15, 1, implicit $exec
+    %4:vgpr_32 = V_ADD_U32_e32 %3, %1, implicit $exec
+    %5:vreg_64 = REG_SEQUENCE %3, %subreg.sub0, %3, %subreg.sub1
+    S_ENDPGM 0, implicit %4, implicit %5
+...
+
+# forwarded lane is read via the full 64-bit reg - do not combine
+# GCN-LABEL: name: dpp_reg_sequence_wide_read
+# GCN: %3:vgpr_32 = V_MOV_B32_dpp %2, %0, 1, 15, 15, 1, implicit $exec
+# GCN: %4:vgpr_32 = V_ADD_U32_e32 %3, %1, implicit $exec
+# GCN: %5:vreg_64 = REG_SEQUENCE %3, %subreg.sub0, %1, %subreg.sub1
+name: dpp_reg_sequence_wide_read
+tracksRegLiveness: true
+body: |
+  bb.0:
+    liveins: $vgpr0, $vgpr1
+
+    %0:vgpr_32 = COPY $vgpr0
+    %1:vgpr_32 = COPY $vgpr1
+    %2:vgpr_32 = IMPLICIT_DEF
+    %3:vgpr_32 = V_MOV_B32_dpp %2, %0, 1, 15, 15, 1, implicit $exec
+    %4:vgpr_32 = V_ADD_U32_e32 %3, %1, implicit $exec
+    %5:vreg_64 = REG_SEQUENCE %3, %subreg.sub0, %1, %subreg.sub1
+    S_ENDPGM 0, implicit %4, implicit %5
+...
+
+# both lanes feed one instruction, which used to be combined twice and erased
+# twice - do not combine
+# GCN-LABEL: name: dpp_reg_sequence_both_lanes_one_use
+# GCN: %2:vgpr_32 = V_MOV_B32_dpp %1, %0, 1, 15, 15, 1, implicit $exec
+# GCN: %3:vreg_64 = REG_SEQUENCE %2, %subreg.sub0, %2, %subreg.sub1
+# GCN: %4:vgpr_32 = V_ADD_U32_e32 %3.sub0, %3.sub1, implicit $exec
+name: dpp_reg_sequence_both_lanes_one_use
+tracksRegLiveness: true
+body: |
+  bb.0:
+    liveins: $vgpr0
+
+    %0:vgpr_32 = COPY $vgpr0
+    %1:vgpr_32 = IMPLICIT_DEF
+    %2:vgpr_32 = V_MOV_B32_dpp %1, %0, 1, 15, 15, 1, implicit $exec
+    %3:vreg_64 = REG_SEQUENCE %2, %subreg.sub0, %2, %subreg.sub1
+    %4:vgpr_32 = V_ADD_U32_e32 %3.sub0, %3.sub1, implicit $exec
+    S_ENDPGM 0, implicit %4
+...
+
+# forwarded lane sub0 partially overlaps a sub0_sub1 read - do not combine
+# GCN-LABEL: name: dpp_reg_sequence_partial_overlap
+# GCN: %3:vgpr_32 = V_MOV_B32_dpp %2, %0, 1, 15, 15, 1, implicit $exec
+# GCN: %5:vgpr_32 = V_ADD_U32_e32 %4.sub0, %1, implicit $exec
+name: dpp_reg_sequence_partial_overlap
+tracksRegLiveness: true
+body: |
+  bb.0:
+    liveins: $vgpr0, $vgpr1
+
+    %0:vgpr_32 = COPY $vgpr0
+    %1:vgpr_32 = COPY $vgpr1
+    %2:vgpr_32 = IMPLICIT_DEF
+    %3:vgpr_32 = V_MOV_B32_dpp %2, %0, 1, 15, 15, 1, implicit $exec
+    %4:vreg_128 = REG_SEQUENCE %3, %subreg.sub0, %1, %subreg.sub1, %1, %subreg.sub2, %1, %subreg.sub3
+    %5:vgpr_32 = V_ADD_U32_e32 %4.sub0, %1, implicit $exec
+    %6:vreg_64 = COPY %4.sub0_sub1
+    S_ENDPGM 0, implicit %5, implicit %6
+...
+
+# the duplicated lane has no reader, so nothing is combined - do not erase the
+# dpp mov
+# GCN-LABEL: name: dpp_reg_sequence_dup_unread_lane
+# GCN: %3:vgpr_32 = V_MOV_B32_dpp %2, %0, 1, 15, 15, 1, implicit $exec
+# GCN: %4:vreg_64 = REG_SEQUENCE %3, %subreg.sub0, %3, %subreg.sub0
+name: dpp_reg_sequence_dup_unread_lane
+tracksRegLiveness: true
+body: |
+  bb.0:
+    liveins: $vgpr0, $vgpr1
+
+    %0:vgpr_32 = COPY $vgpr0
+    %1:vgpr_32 = COPY $vgpr1
+    %2:vgpr_32 = IMPLICIT_DEF
+    %3:vgpr_32 = V_MOV_B32_dpp %2, %0, 1, 15, 15, 1, implicit $exec
+    %4:vreg_64 = REG_SEQUENCE %3, %subreg.sub0, %3, %subreg.sub0
+    %5:vgpr_32 = V_ADD_U32_e32 %4.sub1, %1, implicit $exec
+    S_ENDPGM 0, implicit %5
+...
+
 # GCN-LABEL: name: dpp_reg_sequence_src2_reject
 #GCN: %2:vgpr_32 = V_MOV_B32_dpp %0.sub0, %1.sub0, 1, 15, 15, 1, implicit $exec
 #GCN: %3:vgpr_32 = V_MOV_B32_dpp %0.sub1, %1.sub1, 1, 15, 15, 1, implicit $exec

>From 25af19d0ba52ae10c2e8a99840459f1618426cdd Mon Sep 17 00:00:00 2001
From: Arseniy Obolenskiy <arseniy.obolenskiy at amd.com>
Date: Sun, 13 Sep 2026 22:13:13 +0200
Subject: [PATCH 2/2] address comments

---
 llvm/lib/Target/AMDGPU/GCNDPPCombine.cpp      | 20 +----
 .../test/CodeGen/AMDGPU/dpp_combine_gfx11.mir | 81 ++++++-------------
 2 files changed, 28 insertions(+), 73 deletions(-)

diff --git a/llvm/lib/Target/AMDGPU/GCNDPPCombine.cpp b/llvm/lib/Target/AMDGPU/GCNDPPCombine.cpp
index 502f82b0ad7f9..1b8517428ec81 100644
--- a/llvm/lib/Target/AMDGPU/GCNDPPCombine.cpp
+++ b/llvm/lib/Target/AMDGPU/GCNDPPCombine.cpp
@@ -670,7 +670,6 @@ bool GCNDPPCombine::combineDPPMov(MachineInstr &MovMI) const {
 
   OrigMIs.push_back(&MovMI);
   bool Rollback = true;
-  bool AnyCombined = false;
   const SIRegisterInfo *TRI = ST->getRegisterInfo();
   SmallVector<MachineOperand *, 16> Uses(
       llvm::make_pointer_range(MRI->use_nodbg_operands(DPPMovReg)));
@@ -700,20 +699,10 @@ bool GCNDPPCombine::combineDPPMov(MachineInstr &MovMI) const {
         break;
       }
 
-      // The DPP mov reg can appear in several operands.
+      // The DPP mov reg can appear in several operands, each with its own
+      // subreg index: REG_SEQUENCE inputs never overlap.
       unsigned OpNo = OrigMI.getOperandNo(Use);
       unsigned FwdSubReg = OrigMI.getOperand(OpNo + 1).getImm();
-      auto It = RegSeqWithOpNos.find(&OrigMI);
-
-      // Duplicate subreg indices pass the verifier. Lane already queued.
-      if (It != RegSeqWithOpNos.end() &&
-          llvm::any_of(It->second, [&](unsigned N) {
-            return OrigMI.getOperand(N + 1).getImm() == FwdSubReg;
-          })) {
-        It->second.push_back(OpNo);
-        Rollback = false;
-        continue;
-      }
 
       // Marking the operand undef is unsound if another read consumes the lane.
       LaneBitmask FwdLanes = TRI->getSubRegIndexLaneMask(FwdSubReg);
@@ -789,7 +778,6 @@ bool GCNDPPCombine::combineDPPMov(MachineInstr &MovMI) const {
                                         OldOpndValue, CombBCZ, IsShrinkable)) {
         DPPMIs.push_back(DPPInst);
         Rollback = false;
-        AnyCombined = true;
       }
     } else {
       assert(Use == Src1 && OrigMI.isCommutable()); // by check [1]
@@ -803,7 +791,6 @@ bool GCNDPPCombine::combineDPPMov(MachineInstr &MovMI) const {
                               IsShrinkable)) {
           DPPMIs.push_back(DPPInst);
           Rollback = false;
-          AnyCombined = true;
         }
       } else
         LLVM_DEBUG(dbgs() << "  failed: cannot be commuted\n");
@@ -814,8 +801,7 @@ bool GCNDPPCombine::combineDPPMov(MachineInstr &MovMI) const {
     OrigMIs.push_back(&OrigMI);
   }
 
-  // A REG_SEQUENCE lane with no readers must not commit the erase of MovMI.
-  Rollback |= !Uses.empty() || !AnyCombined;
+  Rollback |= !Uses.empty();
 
   for (auto *MI : *(Rollback? &DPPMIs : &OrigMIs))
     MI->eraseFromParent();
diff --git a/llvm/test/CodeGen/AMDGPU/dpp_combine_gfx11.mir b/llvm/test/CodeGen/AMDGPU/dpp_combine_gfx11.mir
index 4f7009e421db0..a158c437562cb 100644
--- a/llvm/test/CodeGen/AMDGPU/dpp_combine_gfx11.mir
+++ b/llvm/test/CodeGen/AMDGPU/dpp_combine_gfx11.mir
@@ -733,10 +733,10 @@ body: |
 
 # the dpp mov feeds both lanes, each lane read via its own index
 # GCN-LABEL: name: dpp_reg_sequence_dup_combined
-# GCN: %5:vgpr_32 = V_ADD_U32_dpp %2, %0, %1, 1, 15, 15, 1, implicit $exec
-# GCN: %6:vgpr_32 = V_ADD_U32_dpp %2, %0, %1, 1, 15, 15, 1, implicit $exec
-# GCN-NOT: V_MOV_B32_dpp
-# GCN-NOT: REG_SEQUENCE
+# GCN: %2:vgpr_32 = IMPLICIT_DEF
+# GCN-NEXT: %5:vgpr_32 = V_ADD_U32_dpp %2, %0, %1, 1, 15, 15, 1, implicit $exec
+# GCN-NEXT: %6:vgpr_32 = V_ADD_U32_dpp %2, %0, %1, 1, 15, 15, 1, implicit $exec
+# GCN-NEXT: S_ENDPGM 0, implicit %5, implicit %6
 name: dpp_reg_sequence_dup_combined
 tracksRegLiveness: true
 body: |
@@ -753,32 +753,13 @@ body: |
     S_ENDPGM 0, implicit %5, implicit %6
 ...
 
-# duplicate subreg indices in one REG_SEQUENCE pass the verifier, so the lane
-# reaches the worklist through two operands - combine the use only once
-# GCN-LABEL: name: dpp_reg_sequence_same_subreg_dup
-# GCN: %5:vgpr_32 = V_ADD_U32_dpp %2, %0, %1, 1, 15, 15, 1, implicit $exec
-# GCN-NOT: V_MOV_B32_dpp
-# GCN-NOT: REG_SEQUENCE
-name: dpp_reg_sequence_same_subreg_dup
-tracksRegLiveness: true
-body: |
-  bb.0:
-    liveins: $vgpr0, $vgpr1
-
-    %0:vgpr_32 = COPY $vgpr0
-    %1:vgpr_32 = COPY $vgpr1
-    %2:vgpr_32 = IMPLICIT_DEF
-    %3:vgpr_32 = V_MOV_B32_dpp %2, %0, 1, 15, 15, 1, implicit $exec
-    %4:vreg_64 = REG_SEQUENCE %3, %subreg.sub0, %3, %subreg.sub0
-    %5:vgpr_32 = V_ADD_U32_e32 %4.sub0, %1, implicit $exec
-    S_ENDPGM 0, implicit %5
-...
-
 # REG_SEQUENCE fed by both lanes is read as a whole - do not combine
 # GCN-LABEL: name: dpp_reg_sequence_repeated_reg
-# GCN: %3:vgpr_32 = V_MOV_B32_dpp %2, %0, 1, 15, 15, 1, implicit $exec
-# GCN: %4:vgpr_32 = V_ADD_U32_e32 %3, %1, implicit $exec
-# GCN: %5:vreg_64 = REG_SEQUENCE %3, %subreg.sub0, %3, %subreg.sub1
+# GCN: %2:vgpr_32 = IMPLICIT_DEF
+# GCN-NEXT: %3:vgpr_32 = V_MOV_B32_dpp %2, %0, 1, 15, 15, 1, implicit $exec
+# GCN-NEXT: %4:vgpr_32 = V_ADD_U32_e32 %3, %1, implicit $exec
+# GCN-NEXT: %5:vreg_64 = REG_SEQUENCE %3, %subreg.sub0, %3, %subreg.sub1
+# GCN-NEXT: S_ENDPGM 0, implicit %4, implicit %5
 name: dpp_reg_sequence_repeated_reg
 tracksRegLiveness: true
 body: |
@@ -796,9 +777,11 @@ body: |
 
 # forwarded lane is read via the full 64-bit reg - do not combine
 # GCN-LABEL: name: dpp_reg_sequence_wide_read
-# GCN: %3:vgpr_32 = V_MOV_B32_dpp %2, %0, 1, 15, 15, 1, implicit $exec
-# GCN: %4:vgpr_32 = V_ADD_U32_e32 %3, %1, implicit $exec
-# GCN: %5:vreg_64 = REG_SEQUENCE %3, %subreg.sub0, %1, %subreg.sub1
+# GCN: %2:vgpr_32 = IMPLICIT_DEF
+# GCN-NEXT: %3:vgpr_32 = V_MOV_B32_dpp %2, %0, 1, 15, 15, 1, implicit $exec
+# GCN-NEXT: %4:vgpr_32 = V_ADD_U32_e32 %3, %1, implicit $exec
+# GCN-NEXT: %5:vreg_64 = REG_SEQUENCE %3, %subreg.sub0, %1, %subreg.sub1
+# GCN-NEXT: S_ENDPGM 0, implicit %4, implicit %5
 name: dpp_reg_sequence_wide_read
 tracksRegLiveness: true
 body: |
@@ -817,9 +800,11 @@ body: |
 # both lanes feed one instruction, which used to be combined twice and erased
 # twice - do not combine
 # GCN-LABEL: name: dpp_reg_sequence_both_lanes_one_use
-# GCN: %2:vgpr_32 = V_MOV_B32_dpp %1, %0, 1, 15, 15, 1, implicit $exec
-# GCN: %3:vreg_64 = REG_SEQUENCE %2, %subreg.sub0, %2, %subreg.sub1
-# GCN: %4:vgpr_32 = V_ADD_U32_e32 %3.sub0, %3.sub1, implicit $exec
+# GCN: %1:vgpr_32 = IMPLICIT_DEF
+# GCN-NEXT: %2:vgpr_32 = V_MOV_B32_dpp %1, %0, 1, 15, 15, 1, implicit $exec
+# GCN-NEXT: %3:vreg_64 = REG_SEQUENCE %2, %subreg.sub0, %2, %subreg.sub1
+# GCN-NEXT: %4:vgpr_32 = V_ADD_U32_e32 %3.sub0, %3.sub1, implicit $exec
+# GCN-NEXT: S_ENDPGM 0, implicit %4
 name: dpp_reg_sequence_both_lanes_one_use
 tracksRegLiveness: true
 body: |
@@ -836,8 +821,12 @@ body: |
 
 # forwarded lane sub0 partially overlaps a sub0_sub1 read - do not combine
 # GCN-LABEL: name: dpp_reg_sequence_partial_overlap
-# GCN: %3:vgpr_32 = V_MOV_B32_dpp %2, %0, 1, 15, 15, 1, implicit $exec
-# GCN: %5:vgpr_32 = V_ADD_U32_e32 %4.sub0, %1, implicit $exec
+# GCN: %2:vgpr_32 = IMPLICIT_DEF
+# GCN-NEXT: %3:vgpr_32 = V_MOV_B32_dpp %2, %0, 1, 15, 15, 1, implicit $exec
+# GCN-NEXT: %4:vreg_128 = REG_SEQUENCE %3, %subreg.sub0, %1, %subreg.sub1, %1, %subreg.sub2, %1, %subreg.sub3
+# GCN-NEXT: %5:vgpr_32 = V_ADD_U32_e32 %4.sub0, %1, implicit $exec
+# GCN-NEXT: %6:vreg_64 = COPY %4.sub0_sub1
+# GCN-NEXT: S_ENDPGM 0, implicit %5, implicit %6
 name: dpp_reg_sequence_partial_overlap
 tracksRegLiveness: true
 body: |
@@ -854,26 +843,6 @@ body: |
     S_ENDPGM 0, implicit %5, implicit %6
 ...
 
-# the duplicated lane has no reader, so nothing is combined - do not erase the
-# dpp mov
-# GCN-LABEL: name: dpp_reg_sequence_dup_unread_lane
-# GCN: %3:vgpr_32 = V_MOV_B32_dpp %2, %0, 1, 15, 15, 1, implicit $exec
-# GCN: %4:vreg_64 = REG_SEQUENCE %3, %subreg.sub0, %3, %subreg.sub0
-name: dpp_reg_sequence_dup_unread_lane
-tracksRegLiveness: true
-body: |
-  bb.0:
-    liveins: $vgpr0, $vgpr1
-
-    %0:vgpr_32 = COPY $vgpr0
-    %1:vgpr_32 = COPY $vgpr1
-    %2:vgpr_32 = IMPLICIT_DEF
-    %3:vgpr_32 = V_MOV_B32_dpp %2, %0, 1, 15, 15, 1, implicit $exec
-    %4:vreg_64 = REG_SEQUENCE %3, %subreg.sub0, %3, %subreg.sub0
-    %5:vgpr_32 = V_ADD_U32_e32 %4.sub1, %1, implicit $exec
-    S_ENDPGM 0, implicit %5
-...
-
 # GCN-LABEL: name: dpp_reg_sequence_src2_reject
 #GCN: %2:vgpr_32 = V_MOV_B32_dpp %0.sub0, %1.sub0, 1, 15, 15, 1, implicit $exec
 #GCN: %3:vgpr_32 = V_MOV_B32_dpp %0.sub1, %1.sub1, 1, 15, 15, 1, implicit $exec



More information about the llvm-commits mailing list