[llvm] 6594e82 - [AMDGPU] Reject DPP combine when old is narrower than dst (#217894)

via llvm-commits llvm-commits at lists.llvm.org
Wed Sep 9 21:05:03 PDT 2026


Author: Arseniy Obolenskiy
Date: 2026-09-10T06:04:58+02:00
New Revision: 6594e8248e7d9dae26b7909d8c5eec4784d55000

URL: https://github.com/llvm/llvm-project/commit/6594e8248e7d9dae26b7909d8c5eec4784d55000
DIFF: https://github.com/llvm/llvm-project/commit/6594e8248e7d9dae26b7909d8c5eec4784d55000.diff

LOG: [AMDGPU] Reject DPP combine when old is narrower than dst (#217894)

Old is copied from the mov dst class, but the combined dst can be wider
(e.g. V_CVT_F64_I32), making old illegal for it

Added: 
    

Modified: 
    llvm/lib/Target/AMDGPU/GCNDPPCombine.cpp
    llvm/test/CodeGen/AMDGPU/dpp_combine_gfx11.mir

Removed: 
    


################################################################################
diff  --git a/llvm/lib/Target/AMDGPU/GCNDPPCombine.cpp b/llvm/lib/Target/AMDGPU/GCNDPPCombine.cpp
index 2f0e8c61ffcbd..4fb9abb707040 100644
--- a/llvm/lib/Target/AMDGPU/GCNDPPCombine.cpp
+++ b/llvm/lib/Target/AMDGPU/GCNDPPCombine.cpp
@@ -215,6 +215,16 @@ MachineInstr *GCNDPPCombine::createDPPInst(MachineInstr &OrigMI,
     LLVM_DEBUG(dbgs() << "  failed: no DPP opcode\n");
     return nullptr;
   }
+
+  const int OldIdx = AMDGPU::getNamedOperandIdx(DPPOp, AMDGPU::OpName::old);
+  const int MovDstIdx =
+      AMDGPU::getNamedOperandIdx(MovMI.getOpcode(), AMDGPU::OpName::vdst);
+  if (OldIdx != -1 &&
+      TII->getOpSize(DPPOp, OldIdx) != TII->getOpSize(MovMI, MovDstIdx)) {
+    LLVM_DEBUG(dbgs() << "  failed: old operand size 
diff ers from dst\n");
+    return nullptr;
+  }
+
   int OrigOpE32 = AMDGPU::getVOPe32(OrigOp);
   // Prior checks cover Mask with VOPC condition, but not on purpose
   auto *RowMaskOpnd = TII->getNamedOperand(MovMI, AMDGPU::OpName::row_mask);
@@ -248,7 +258,6 @@ MachineInstr *GCNDPPCombine::createDPPInst(MachineInstr &OrigMI,
       // If we shrunk a 64bit vop3b to 32bits, just ignore the sdst
     }
 
-    const int OldIdx = AMDGPU::getNamedOperandIdx(DPPOp, AMDGPU::OpName::old);
     if (OldIdx != -1) {
       assert(OldIdx == NumOperands);
       assert(isOfRegClass(

diff  --git a/llvm/test/CodeGen/AMDGPU/dpp_combine_gfx11.mir b/llvm/test/CodeGen/AMDGPU/dpp_combine_gfx11.mir
index 4fbcc9581f2bc..321414a81bdc8 100644
--- a/llvm/test/CodeGen/AMDGPU/dpp_combine_gfx11.mir
+++ b/llvm/test/CodeGen/AMDGPU/dpp_combine_gfx11.mir
@@ -958,3 +958,21 @@ body:             |
     %10:vgpr_32 = V_MOV_B32_dpp %3, %0, 1, 15, 15, 1, implicit $exec
     %11:vgpr_32 = V_FMA_MIX_F32 12, %10, 12, %1, 12, %2, 0, 0, 0, implicit $mode, implicit $exec
 ...
+
+# GCN-LABEL: name: dpp_64bit_dst_32bit_old
+# GCN: %2:vgpr_32 = V_MOV_B32_dpp %1, %0, 85, 15, 15, 1, implicit $exec
+# GCN: %3:vreg_64 = V_CVT_F64_I32_e64 %2, 0, 0, implicit $mode, implicit $exec
+
+name: dpp_64bit_dst_32bit_old
+tracksRegLiveness: true
+body:             |
+  bb.0:
+    liveins: $vgpr0
+
+    %0:vgpr_32 = COPY $vgpr0
+    %1:vgpr_32 = IMPLICIT_DEF
+    ; should not be combined: old is 32-bit but V_CVT_F64_I32 dst is 64-bit
+    %2:vgpr_32 = V_MOV_B32_dpp %1, %0, 85, 15, 15, 1, implicit $exec
+    %3:vreg_64 = V_CVT_F64_I32_e64 %2, 0, 0, implicit $mode, implicit $exec
+    S_ENDPGM 0, implicit %3
+...


        


More information about the llvm-commits mailing list