[llvm] 6594e82 - [AMDGPU] Reject DPP combine when old is narrower than dst (#217894)
via llvm-commits
llvm-commits at lists.llvm.org
Wed Sep 9 21:05:03 PDT 2026
Author: Arseniy Obolenskiy
Date: 2026-09-10T06:04:58+02:00
New Revision: 6594e8248e7d9dae26b7909d8c5eec4784d55000
URL: https://github.com/llvm/llvm-project/commit/6594e8248e7d9dae26b7909d8c5eec4784d55000
DIFF: https://github.com/llvm/llvm-project/commit/6594e8248e7d9dae26b7909d8c5eec4784d55000.diff
LOG: [AMDGPU] Reject DPP combine when old is narrower than dst (#217894)
Old is copied from the mov dst class, but the combined dst can be wider
(e.g. V_CVT_F64_I32), making old illegal for it
Added:
Modified:
llvm/lib/Target/AMDGPU/GCNDPPCombine.cpp
llvm/test/CodeGen/AMDGPU/dpp_combine_gfx11.mir
Removed:
################################################################################
diff --git a/llvm/lib/Target/AMDGPU/GCNDPPCombine.cpp b/llvm/lib/Target/AMDGPU/GCNDPPCombine.cpp
index 2f0e8c61ffcbd..4fb9abb707040 100644
--- a/llvm/lib/Target/AMDGPU/GCNDPPCombine.cpp
+++ b/llvm/lib/Target/AMDGPU/GCNDPPCombine.cpp
@@ -215,6 +215,16 @@ MachineInstr *GCNDPPCombine::createDPPInst(MachineInstr &OrigMI,
LLVM_DEBUG(dbgs() << " failed: no DPP opcode\n");
return nullptr;
}
+
+ const int OldIdx = AMDGPU::getNamedOperandIdx(DPPOp, AMDGPU::OpName::old);
+ const int MovDstIdx =
+ AMDGPU::getNamedOperandIdx(MovMI.getOpcode(), AMDGPU::OpName::vdst);
+ if (OldIdx != -1 &&
+ TII->getOpSize(DPPOp, OldIdx) != TII->getOpSize(MovMI, MovDstIdx)) {
+ LLVM_DEBUG(dbgs() << " failed: old operand size
diff ers from dst\n");
+ return nullptr;
+ }
+
int OrigOpE32 = AMDGPU::getVOPe32(OrigOp);
// Prior checks cover Mask with VOPC condition, but not on purpose
auto *RowMaskOpnd = TII->getNamedOperand(MovMI, AMDGPU::OpName::row_mask);
@@ -248,7 +258,6 @@ MachineInstr *GCNDPPCombine::createDPPInst(MachineInstr &OrigMI,
// If we shrunk a 64bit vop3b to 32bits, just ignore the sdst
}
- const int OldIdx = AMDGPU::getNamedOperandIdx(DPPOp, AMDGPU::OpName::old);
if (OldIdx != -1) {
assert(OldIdx == NumOperands);
assert(isOfRegClass(
diff --git a/llvm/test/CodeGen/AMDGPU/dpp_combine_gfx11.mir b/llvm/test/CodeGen/AMDGPU/dpp_combine_gfx11.mir
index 4fbcc9581f2bc..321414a81bdc8 100644
--- a/llvm/test/CodeGen/AMDGPU/dpp_combine_gfx11.mir
+++ b/llvm/test/CodeGen/AMDGPU/dpp_combine_gfx11.mir
@@ -958,3 +958,21 @@ body: |
%10:vgpr_32 = V_MOV_B32_dpp %3, %0, 1, 15, 15, 1, implicit $exec
%11:vgpr_32 = V_FMA_MIX_F32 12, %10, 12, %1, 12, %2, 0, 0, 0, implicit $mode, implicit $exec
...
+
+# GCN-LABEL: name: dpp_64bit_dst_32bit_old
+# GCN: %2:vgpr_32 = V_MOV_B32_dpp %1, %0, 85, 15, 15, 1, implicit $exec
+# GCN: %3:vreg_64 = V_CVT_F64_I32_e64 %2, 0, 0, implicit $mode, implicit $exec
+
+name: dpp_64bit_dst_32bit_old
+tracksRegLiveness: true
+body: |
+ bb.0:
+ liveins: $vgpr0
+
+ %0:vgpr_32 = COPY $vgpr0
+ %1:vgpr_32 = IMPLICIT_DEF
+ ; should not be combined: old is 32-bit but V_CVT_F64_I32 dst is 64-bit
+ %2:vgpr_32 = V_MOV_B32_dpp %1, %0, 85, 15, 15, 1, implicit $exec
+ %3:vreg_64 = V_CVT_F64_I32_e64 %2, 0, 0, implicit $mode, implicit $exec
+ S_ENDPGM 0, implicit %3
+...
More information about the llvm-commits
mailing list