[llvm] [AMDGPU] Restrict DPP combine from performing bad transformations when handling certain REV subtraction insts that use Src1 as the DPP operand (PR #216835)
via llvm-commits
llvm-commits at lists.llvm.org
Mon Aug 17 17:08:57 PDT 2026
llvmorg-github-actions[bot] wrote:
<!--LLVM PR SUMMARY COMMENT-->
@llvm/pr-subscribers-backend-amdgpu
Author: Domenic Nutile (saxlungs)
<details>
<summary>Changes</summary>
This issue was originally discovered due to a benchmark failure in the downstream. Internal documentation says this applies for GFX12, but testing revealed it also applies for at least GFX11 as well.
---
Patch is 25.47 KiB, truncated to 20.00 KiB below, full version: https://github.com/llvm/llvm-project/pull/216835.diff
6 Files Affected:
- (modified) llvm/lib/Target/AMDGPU/GCNDPPCombine.cpp (+40)
- (modified) llvm/test/CodeGen/AMDGPU/GlobalISel/llvm.amdgcn.wqm.demote.ll (+16-16)
- (modified) llvm/test/CodeGen/AMDGPU/dpp_combine.mir (+10-7)
- (modified) llvm/test/CodeGen/AMDGPU/dpp_combine_gfx11.mir (+19-11)
- (added) llvm/test/CodeGen/AMDGPU/dpp_combine_rev_opcode.mir (+105)
- (modified) llvm/test/CodeGen/AMDGPU/llvm.amdgcn.wqm.demote.ll (+16-16)
``````````diff
diff --git a/llvm/lib/Target/AMDGPU/GCNDPPCombine.cpp b/llvm/lib/Target/AMDGPU/GCNDPPCombine.cpp
index 9d22757b4514a..ec504e376f740 100644
--- a/llvm/lib/Target/AMDGPU/GCNDPPCombine.cpp
+++ b/llvm/lib/Target/AMDGPU/GCNDPPCombine.cpp
@@ -117,6 +117,34 @@ FunctionPass *llvm::createGCNDPPCombinePass() {
return new GCNDPPCombineLegacy();
}
+// Some opcodes use Src1 for DPP instead of Src0, because the sequencer
+// transforms them and reverse the order of their operands at runtime.
+//
+// Listed as target-independent pseudos; the per-subtarget MC opcodes
+// (V_SUBREV_NC_U32_e32_gfx11 and friends) are all reached through these.
+static bool isSrc1DPPRevOpcode(unsigned Opc) {
+ switch (Opc) {
+ // v_subrev_u32 (gfx9) / v_subrev_nc_u32 (gfx10+)
+ case AMDGPU::V_SUBREV_U32_e32:
+ case AMDGPU::V_SUBREV_U32_e64:
+ // v_subrev_co_u32
+ case AMDGPU::V_SUBREV_CO_U32_e32:
+ case AMDGPU::V_SUBREV_CO_U32_e64:
+ // v_subbrev_u32 (gfx9) / v_subrev_co_ci_u32 (gfx10+)
+ case AMDGPU::V_SUBBREV_U32_e32:
+ case AMDGPU::V_SUBBREV_U32_e64:
+ // v_subrev_f32
+ case AMDGPU::V_SUBREV_F32_e32:
+ case AMDGPU::V_SUBREV_F32_e64:
+ // v_subrev_f16
+ case AMDGPU::V_SUBREV_F16_e32:
+ case AMDGPU::V_SUBREV_F16_e64:
+ return true;
+ default:
+ return false;
+ }
+}
+
bool GCNDPPCombine::isShrinkable(MachineInstr &MI) const {
unsigned Op = MI.getOpcode();
if (!TII->isVOP3(Op)) {
@@ -746,6 +774,18 @@ bool GCNDPPCombine::combineDPPMov(MachineInstr &MovMI) const {
break;
}
+ // We have to be careful to prevent trying to fold into the first source
+ // operand of instructions that apply DPP to the second source operand.
+ // This could be directly, or when folding into an instruction that will
+ // get commuted into one.
+ int FoldedOp =
+ (Use == Src0) ? static_cast<int>(OrigOp) : TII->commuteOpcode(OrigOp);
+ if (FoldedOp < 0 || isSrc1DPPRevOpcode(FoldedOp)) {
+ LLVM_DEBUG(
+ dbgs() << " failed: Use operand cannot have DPP applied to it\n");
+ break;
+ }
+
if (!ST->hasFeature(AMDGPU::FeatureDPALU_DPP) &&
AMDGPU::isDPALU_DPP32BitOpc(OrigOp)) {
LLVM_DEBUG(dbgs() << " " << OrigMI
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/llvm.amdgcn.wqm.demote.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/llvm.amdgcn.wqm.demote.ll
index 02b68822047c3..72773edfbd615 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/llvm.amdgcn.wqm.demote.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/llvm.amdgcn.wqm.demote.ll
@@ -690,9 +690,9 @@ define amdgpu_ps void @wqm_deriv(<2 x float> %input, float %arg, i32 %index) {
; SI-NEXT: v_mov_b32_e32 v1, v0
; SI-NEXT: s_mov_b64 s[2:3], exec
; SI-NEXT: s_nop 0
-; SI-NEXT: v_mov_b32_dpp v1, v1 quad_perm:[1,1,1,1] row_mask:0xf bank_mask:0xf bound_ctrl:1
+; SI-NEXT: v_mov_b32_dpp v1, v1 quad_perm:[0,0,0,0] row_mask:0xf bank_mask:0xf bound_ctrl:1
; SI-NEXT: s_nop 1
-; SI-NEXT: v_subrev_f32_dpp v0, v0, v1 quad_perm:[0,0,0,0] row_mask:0xf bank_mask:0xf bound_ctrl:1
+; SI-NEXT: v_sub_f32_dpp v0, v0, v1 quad_perm:[1,1,1,1] row_mask:0xf bank_mask:0xf bound_ctrl:1
; SI-NEXT: ; kill: def $vgpr0 killed $vgpr0 killed $exec
; SI-NEXT: s_and_b64 exec, exec, s[0:1]
; SI-NEXT: v_cmp_eq_f32_e32 vcc, 0, v0
@@ -739,9 +739,9 @@ define amdgpu_ps void @wqm_deriv(<2 x float> %input, float %arg, i32 %index) {
; GFX9-NEXT: v_mov_b32_e32 v1, v0
; GFX9-NEXT: s_mov_b64 s[2:3], exec
; GFX9-NEXT: s_nop 0
-; GFX9-NEXT: v_mov_b32_dpp v1, v1 quad_perm:[1,1,1,1] row_mask:0xf bank_mask:0xf bound_ctrl:1
+; GFX9-NEXT: v_mov_b32_dpp v1, v1 quad_perm:[0,0,0,0] row_mask:0xf bank_mask:0xf bound_ctrl:1
; GFX9-NEXT: s_nop 1
-; GFX9-NEXT: v_subrev_f32_dpp v0, v0, v1 quad_perm:[0,0,0,0] row_mask:0xf bank_mask:0xf bound_ctrl:1
+; GFX9-NEXT: v_sub_f32_dpp v0, v0, v1 quad_perm:[1,1,1,1] row_mask:0xf bank_mask:0xf bound_ctrl:1
; GFX9-NEXT: ; kill: def $vgpr0 killed $vgpr0 killed $exec
; GFX9-NEXT: s_and_b64 exec, exec, s[0:1]
; GFX9-NEXT: v_cmp_eq_f32_e32 vcc, 0, v0
@@ -787,8 +787,8 @@ define amdgpu_ps void @wqm_deriv(<2 x float> %input, float %arg, i32 %index) {
; GFX10-32-NEXT: s_mov_b32 s1, exec_lo
; GFX10-32-NEXT: v_cndmask_b32_e64 v0, 1.0, 0, s2
; GFX10-32-NEXT: v_mov_b32_e32 v1, v0
-; GFX10-32-NEXT: v_mov_b32_dpp v1, v1 quad_perm:[1,1,1,1] row_mask:0xf bank_mask:0xf bound_ctrl:1
-; GFX10-32-NEXT: v_subrev_f32_dpp v0, v0, v1 quad_perm:[0,0,0,0] row_mask:0xf bank_mask:0xf bound_ctrl:1
+; GFX10-32-NEXT: v_mov_b32_dpp v1, v1 quad_perm:[0,0,0,0] row_mask:0xf bank_mask:0xf bound_ctrl:1
+; GFX10-32-NEXT: v_sub_f32_dpp v0, v0, v1 quad_perm:[1,1,1,1] row_mask:0xf bank_mask:0xf bound_ctrl:1
; GFX10-32-NEXT: ; kill: def $vgpr0 killed $vgpr0 killed $exec
; GFX10-32-NEXT: s_and_b32 exec_lo, exec_lo, s0
; GFX10-32-NEXT: v_cmp_eq_f32_e32 vcc_lo, 0, v0
@@ -834,8 +834,8 @@ define amdgpu_ps void @wqm_deriv(<2 x float> %input, float %arg, i32 %index) {
; GFX10-64-NEXT: s_mov_b64 s[2:3], exec
; GFX10-64-NEXT: v_cndmask_b32_e64 v0, 1.0, 0, s[4:5]
; GFX10-64-NEXT: v_mov_b32_e32 v1, v0
-; GFX10-64-NEXT: v_mov_b32_dpp v1, v1 quad_perm:[1,1,1,1] row_mask:0xf bank_mask:0xf bound_ctrl:1
-; GFX10-64-NEXT: v_subrev_f32_dpp v0, v0, v1 quad_perm:[0,0,0,0] row_mask:0xf bank_mask:0xf bound_ctrl:1
+; GFX10-64-NEXT: v_mov_b32_dpp v1, v1 quad_perm:[0,0,0,0] row_mask:0xf bank_mask:0xf bound_ctrl:1
+; GFX10-64-NEXT: v_sub_f32_dpp v0, v0, v1 quad_perm:[1,1,1,1] row_mask:0xf bank_mask:0xf bound_ctrl:1
; GFX10-64-NEXT: ; kill: def $vgpr0 killed $vgpr0 killed $exec
; GFX10-64-NEXT: s_and_b64 exec, exec, s[0:1]
; GFX10-64-NEXT: v_cmp_eq_f32_e32 vcc, 0, v0
@@ -938,9 +938,9 @@ define amdgpu_ps void @wqm_deriv_loop(<2 x float> %input, float %arg, i32 %index
; SI-NEXT: v_mov_b32_e32 v2, v0
; SI-NEXT: s_mov_b64 s[4:5], exec
; SI-NEXT: s_nop 0
-; SI-NEXT: v_mov_b32_dpp v2, v2 quad_perm:[1,1,1,1] row_mask:0xf bank_mask:0xf bound_ctrl:1
+; SI-NEXT: v_mov_b32_dpp v2, v2 quad_perm:[0,0,0,0] row_mask:0xf bank_mask:0xf bound_ctrl:1
; SI-NEXT: s_nop 1
-; SI-NEXT: v_subrev_f32_dpp v0, v0, v2 quad_perm:[0,0,0,0] row_mask:0xf bank_mask:0xf bound_ctrl:1
+; SI-NEXT: v_sub_f32_dpp v0, v0, v2 quad_perm:[1,1,1,1] row_mask:0xf bank_mask:0xf bound_ctrl:1
; SI-NEXT: ; kill: def $vgpr0 killed $vgpr0 killed $exec
; SI-NEXT: v_cmp_eq_f32_e32 vcc, 0, v0
; SI-NEXT: s_and_b64 s[10:11], s[0:1], vcc
@@ -1005,9 +1005,9 @@ define amdgpu_ps void @wqm_deriv_loop(<2 x float> %input, float %arg, i32 %index
; GFX9-NEXT: v_mov_b32_e32 v2, v0
; GFX9-NEXT: s_mov_b64 s[4:5], exec
; GFX9-NEXT: s_nop 0
-; GFX9-NEXT: v_mov_b32_dpp v2, v2 quad_perm:[1,1,1,1] row_mask:0xf bank_mask:0xf bound_ctrl:1
+; GFX9-NEXT: v_mov_b32_dpp v2, v2 quad_perm:[0,0,0,0] row_mask:0xf bank_mask:0xf bound_ctrl:1
; GFX9-NEXT: s_nop 1
-; GFX9-NEXT: v_subrev_f32_dpp v0, v0, v2 quad_perm:[0,0,0,0] row_mask:0xf bank_mask:0xf bound_ctrl:1
+; GFX9-NEXT: v_sub_f32_dpp v0, v0, v2 quad_perm:[1,1,1,1] row_mask:0xf bank_mask:0xf bound_ctrl:1
; GFX9-NEXT: ; kill: def $vgpr0 killed $vgpr0 killed $exec
; GFX9-NEXT: v_cmp_eq_f32_e32 vcc, 0, v0
; GFX9-NEXT: s_and_b64 s[8:9], s[0:1], vcc
@@ -1070,8 +1070,8 @@ define amdgpu_ps void @wqm_deriv_loop(<2 x float> %input, float %arg, i32 %index
; GFX10-32-NEXT: s_mov_b32 s3, exec_lo
; GFX10-32-NEXT: v_cndmask_b32_e64 v0, s2, 0, s4
; GFX10-32-NEXT: v_mov_b32_e32 v2, v0
-; GFX10-32-NEXT: v_mov_b32_dpp v2, v2 quad_perm:[1,1,1,1] row_mask:0xf bank_mask:0xf bound_ctrl:1
-; GFX10-32-NEXT: v_subrev_f32_dpp v0, v0, v2 quad_perm:[0,0,0,0] row_mask:0xf bank_mask:0xf bound_ctrl:1
+; GFX10-32-NEXT: v_mov_b32_dpp v2, v2 quad_perm:[0,0,0,0] row_mask:0xf bank_mask:0xf bound_ctrl:1
+; GFX10-32-NEXT: v_sub_f32_dpp v0, v0, v2 quad_perm:[1,1,1,1] row_mask:0xf bank_mask:0xf bound_ctrl:1
; GFX10-32-NEXT: ; kill: def $vgpr0 killed $vgpr0 killed $exec
; GFX10-32-NEXT: v_cmp_eq_f32_e32 vcc_lo, 0, v0
; GFX10-32-NEXT: s_and_b32 s4, s0, vcc_lo
@@ -1134,8 +1134,8 @@ define amdgpu_ps void @wqm_deriv_loop(<2 x float> %input, float %arg, i32 %index
; GFX10-64-NEXT: s_mov_b64 s[4:5], exec
; GFX10-64-NEXT: v_cndmask_b32_e64 v0, s6, 0, s[8:9]
; GFX10-64-NEXT: v_mov_b32_e32 v2, v0
-; GFX10-64-NEXT: v_mov_b32_dpp v2, v2 quad_perm:[1,1,1,1] row_mask:0xf bank_mask:0xf bound_ctrl:1
-; GFX10-64-NEXT: v_subrev_f32_dpp v0, v0, v2 quad_perm:[0,0,0,0] row_mask:0xf bank_mask:0xf bound_ctrl:1
+; GFX10-64-NEXT: v_mov_b32_dpp v2, v2 quad_perm:[0,0,0,0] row_mask:0xf bank_mask:0xf bound_ctrl:1
+; GFX10-64-NEXT: v_sub_f32_dpp v0, v0, v2 quad_perm:[1,1,1,1] row_mask:0xf bank_mask:0xf bound_ctrl:1
; GFX10-64-NEXT: ; kill: def $vgpr0 killed $vgpr0 killed $exec
; GFX10-64-NEXT: v_cmp_eq_f32_e32 vcc, 0, v0
; GFX10-64-NEXT: s_and_b64 s[8:9], s[0:1], vcc
diff --git a/llvm/test/CodeGen/AMDGPU/dpp_combine.mir b/llvm/test/CodeGen/AMDGPU/dpp_combine.mir
index 140a509bcbc34..fe9d6864c5d52 100644
--- a/llvm/test/CodeGen/AMDGPU/dpp_combine.mir
+++ b/llvm/test/CodeGen/AMDGPU/dpp_combine.mir
@@ -257,7 +257,8 @@ body: |
# GCN: %7:vgpr_32 = V_AND_B32_dpp %1, %0, %1, 1, 15, 14, 0, implicit $exec
# GCN: %10:vgpr_32 = V_MAX_I32_dpp %1, %0, %1, 1, 14, 15, 0, implicit $exec
# GCN: %13:vgpr_32 = V_MIN_I32_dpp %1, %0, %1, 1, 15, 14, 0, implicit $exec
-# GCN: %16:vgpr_32 = V_SUBREV_CO_U32_dpp %1, %0, %1, 1, 14, 15, 0, implicit-def $vcc, implicit $exec
+# GCN: %15:vgpr_32 = V_MOV_B32_dpp %14, %0, 1, 14, 15, 0, implicit $exec
+# GCN-NEXT: %16:vgpr_32 = V_SUB_CO_U32_e32 %1, %15, implicit-def $vcc, implicit $exec
# GCN: %19:vgpr_32 = V_ADD_CO_U32_e32 5, %18, implicit-def $vcc, implicit $exec
name: dpp_commute
tracksRegLiveness: true
@@ -377,9 +378,10 @@ body: |
# tests on sequences of dpp consumers
# GCN-LABEL: name: dpp_seq
-# GCN: %4:vgpr_32 = V_ADD_CO_U32_dpp %1, %0, %1, 1, 14, 15, 0, implicit-def $vcc, implicit $exec
-# GCN: %5:vgpr_32 = V_SUBREV_CO_U32_dpp %1, %0, %1, 1, 14, 15, 0, implicit-def $vcc, implicit $exec
-# GCN: %6:vgpr_32 = V_OR_B32_dpp %1, %0, %1, 1, 14, 15, 0, implicit $exec
+# GCN: %3:vgpr_32 = V_MOV_B32_dpp %2, %0, 1, 14, 15, 0, implicit $exec
+# GCN-NEXT: %4:vgpr_32 = V_ADD_CO_U32_e32 %3, %1, implicit-def $vcc, implicit $exec
+# GCN-NEXT: %5:vgpr_32 = V_SUB_CO_U32_e32 %1, %3, implicit-def $vcc, implicit $exec
+# GCN-NEXT: %6:vgpr_32 = V_OR_B32_e32 %3, %1, implicit $exec
# broken sequence:
# GCN: %7:vgpr_32 = V_MOV_B32_dpp %2, %0, 1, 14, 15, 0, implicit $exec
@@ -405,9 +407,10 @@ body: |
# tests on sequences of dpp consumers followed by control flow
# GCN-LABEL: name: dpp_seq_cf
-# GCN: %4:vgpr_32 = V_ADD_CO_U32_dpp %1, %0, %1, 1, 14, 15, 0, implicit-def $vcc, implicit $exec
-# GCN: %5:vgpr_32 = V_SUBREV_CO_U32_dpp %1, %0, %1, 1, 14, 15, 0, implicit-def $vcc, implicit $exec
-# GCN: %6:vgpr_32 = V_OR_B32_dpp %1, %0, %1, 1, 14, 15, 0, implicit $exec
+# GCN: %3:vgpr_32 = V_MOV_B32_dpp %2, %0, 1, 14, 15, 0, implicit $exec
+# GCN-NEXT: %4:vgpr_32 = V_ADD_CO_U32_e32 %3, %1, implicit-def $vcc, implicit $exec
+# GCN-NEXT: %5:vgpr_32 = V_SUB_CO_U32_e32 %1, %3, implicit-def $vcc, implicit $exec
+# GCN-NEXT: %6:vgpr_32 = V_OR_B32_e32 %3, %1, implicit $exec
name: dpp_seq_cf
tracksRegLiveness: true
diff --git a/llvm/test/CodeGen/AMDGPU/dpp_combine_gfx11.mir b/llvm/test/CodeGen/AMDGPU/dpp_combine_gfx11.mir
index b7c884c7995e7..4fbcc9581f2bc 100644
--- a/llvm/test/CodeGen/AMDGPU/dpp_combine_gfx11.mir
+++ b/llvm/test/CodeGen/AMDGPU/dpp_combine_gfx11.mir
@@ -5,9 +5,13 @@
---
# GCN-LABEL: name: vop3
-# GCN: %6:vgpr_32, %7:sreg_32_xm0_xexec = V_SUBBREV_U32_e64_dpp %3, %0, %1, %5, 1, 1, 15, 15, 1, implicit $exec
-# GCN: %8:vgpr_32 = V_CVT_PK_U8_F32_e64_dpp %3, 4, %0, 2, %2, 2, %1, 1, 1, 15, 15, 1, implicit $mode, implicit $exec
-# GCN: %10:vgpr_32 = V_MED3_F32_e64 0, %9, 0, %0, 0, 12345678, 0, 0, implicit $mode, implicit $exec
+# V_SUBBREV is a Src1-DPP opcode, so %4 cannot be folded into it. That rolls
+# back the whole group sharing %4, including the V_CVT_PK_U8_F32 use.
+# GCN: %4:vgpr_32 = V_MOV_B32_dpp %3, %0, 1, 15, 15, 1, implicit $exec
+# GCN: %6:vgpr_32, %7:sreg_32_xm0_xexec = V_SUBBREV_U32_e64 %4, %1, %5, 1, implicit $exec
+# GCN-NEXT: %8:vgpr_32 = V_CVT_PK_U8_F32_e64 4, %4, 2, %2, 2, %1, 1, implicit $mode, implicit $exec
+# GCN: %9:vgpr_32 = V_MOV_B32_dpp %3, %1, 1, 15, 15, 1, implicit $exec
+# GCN-NEXT: %10:vgpr_32 = V_MED3_F32_e64 0, %9, 0, %0, 0, 12345678, 0, 0, implicit $mode, implicit $exec
# GFX_NO_SRC1_SGPR: %12:vgpr_32 = V_MED3_F32_e64 0, %11, 0, 2, 0, %7, 0, 0, implicit $mode, implicit $exec
# GFX_SRC1_SGPR: %12:vgpr_32 = V_MED3_F32_e64_dpp %3, 0, %1, 0, 2, 0, %7, 0, 0, 1, 15, 15, 1, implicit $mode, implicit $exec
name: vop3
@@ -172,7 +176,8 @@ body: |
# GCN: %7:vgpr_32 = V_AND_B32_dpp %1, %0, %1, 1, 15, 14, 0, implicit $exec
# GCN: %10:vgpr_32 = V_MAX_I32_dpp %1, %0, %1, 1, 14, 15, 0, implicit $exec
# GCN: %13:vgpr_32 = V_MIN_I32_dpp %1, %0, %1, 1, 15, 14, 0, implicit $exec
-# GCN: %16:vgpr_32 = V_SUBREV_U32_dpp %1, %0, %1, 1, 14, 15, 0, implicit $exec
+# GCN: %15:vgpr_32 = V_MOV_B32_dpp %14, %0, 1, 14, 15, 0, implicit $exec
+# GCN-NEXT: %16:vgpr_32 = V_SUB_U32_e64 %1, %15, 0, implicit $exec
name: dpp_commute_shrink
tracksRegLiveness: true
body: |
@@ -234,7 +239,8 @@ body: |
# GCN-LABEL: name: dpp_commute_e64
# GCN: %4:vgpr_32 = V_MUL_U32_U24_e64_dpp %1, %0, %1, 1, 1, 14, 15, 0, implicit $exec
# GCN: %7:vgpr_32 = V_FMA_F32_e64_dpp %5, 2, %0, 1, %1, 2, %1, 1, 2, 1, 15, 15, 1, implicit $mode, implicit $exec
-# GCN: %10:vgpr_32 = V_SUBREV_U32_e64_dpp %1, %0, %1, 1, 1, 14, 15, 0, implicit $exec
+# GCN: %9:vgpr_32 = V_MOV_B32_dpp %8, %0, 1, 14, 15, 0, implicit $exec
+# GCN-NEXT: %10:vgpr_32 = V_SUB_U32_e64 %1, %9, 1, implicit $exec
# GCN: %13:vgpr_32, %14:sreg_32_xm0_xexec = V_ADD_CO_U32_e64_dpp %1, %0, %1, 0, 1, 14, 15, 0, implicit $exec
# GCN: %17:vgpr_32, %18:sreg_32_xm0_xexec = V_ADD_CO_U32_e64 5, %16, 0, implicit $exec
name: dpp_commute_e64
@@ -330,9 +336,10 @@ body: |
# tests on sequences of dpp consumers
# GCN-LABEL: name: dpp_seq
-# GCN: %4:vgpr_32 = V_ADD_U32_dpp %1, %0, %1, 1, 14, 15, 0, implicit $exec
-# GCN: %5:vgpr_32 = V_SUBREV_U32_dpp %1, %0, %1, 1, 14, 15, 0, implicit $exec
-# GCN: %6:vgpr_32 = V_OR_B32_dpp %1, %0, %1, 1, 14, 15, 0, implicit $exec
+# GCN: %3:vgpr_32 = V_MOV_B32_dpp %2, %0, 1, 14, 15, 0, implicit $exec
+# GCN-NEXT: %4:vgpr_32 = V_ADD_U32_e32 %3, %1, implicit $exec
+# GCN-NEXT: %5:vgpr_32 = V_SUB_U32_e32 %1, %3, implicit $exec
+# GCN-NEXT: %6:vgpr_32 = V_OR_B32_e32 %3, %1, implicit $exec
# broken sequence:
# GCN: %7:vgpr_32 = V_MOV_B32_dpp %2, %0, 1, 14, 15, 0, implicit $exec
@@ -358,9 +365,10 @@ body: |
# tests on sequences of dpp consumers followed by control flow
# GCN-LABEL: name: dpp_seq_cf
-# GCN: %4:vgpr_32 = V_ADD_U32_dpp %1, %0, %1, 1, 14, 15, 0, implicit $exec
-# GCN: %5:vgpr_32 = V_SUBREV_U32_dpp %1, %0, %1, 1, 14, 15, 0, implicit $exec
-# GCN: %6:vgpr_32 = V_OR_B32_dpp %1, %0, %1, 1, 14, 15, 0, implicit $exec
+# GCN: %3:vgpr_32 = V_MOV_B32_dpp %2, %0, 1, 14, 15, 0, implicit $exec
+# GCN-NEXT: %4:vgpr_32 = V_ADD_U32_e32 %3, %1, implicit $exec
+# GCN-NEXT: %5:vgpr_32 = V_SUB_U32_e32 %1, %3, implicit $exec
+# GCN-NEXT: %6:vgpr_32 = V_OR_B32_e32 %3, %1, implicit $exec
name: dpp_seq_cf
tracksRegLiveness: true
diff --git a/llvm/test/CodeGen/AMDGPU/dpp_combine_rev_opcode.mir b/llvm/test/CodeGen/AMDGPU/dpp_combine_rev_opcode.mir
new file mode 100644
index 0000000000000..769f12cc31468
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/dpp_combine_rev_opcode.mir
@@ -0,0 +1,105 @@
+# RUN: llc -mtriple=amdgpu10.30 -run-pass=gcn-dpp-combine -verify-machineinstrs -o - %s | FileCheck %s
+# RUN: llc -mtriple=amdgpu11.00 -run-pass=gcn-dpp-combine -verify-machineinstrs -o - %s | FileCheck %s
+# RUN: llc -mtriple=amdgpu12.00 -run-pass=gcn-dpp-combine -verify-machineinstrs -o - %s | FileCheck %s
+
+---
+# The DPP value is src1 of a subtract, so folding requires commuting
+# V_SUB_U32_e32 into V_SUBREV_U32_dpp, which is not allowed.
+
+# CHECK-LABEL: name: dpp_commute_to_rev_sub
+# CHECK: %3:vgpr_32 = V_MOV_B32_dpp %2, %0, 1, 15, 15, 1, implicit $exec
+# CHECK-NEXT: %4:vgpr_32 = V_SUB_U32_e32 %1, %3, implicit $exec
+
+name: dpp_commute_to_rev_sub
+tracksRegLiveness: true
+body: |
+ bb.0:
+ liveins: $vgpr0, $vgpr1
+ %0:vgpr_32 = COPY $vgpr0
+ %1:vgpr_32 = COPY $vgpr1
+ %2:vgpr_32 = IMPLICIT_DEF
+ %3:vgpr_32 = V_MOV_B32_dpp %2, %0, 1, 15, 15, 1, implicit $exec
+ %4:vgpr_32 = V_SUB_U32_e32 %1, %3, implicit $exec
+ S_NOP 0, implicit %4
+...
+
+---
+# Both sources of the subtract are the same register, so a mis-routed DPP
+# silently yields the negated result rather than an obviously wrong instruction.
+
+# CHECK-LABEL: name: dpp_commute_to_rev_sub_same_reg
+# CHECK: %2:vgpr_32 = V_MOV_B32_dpp %1, %0, 0, 15, 15, 1, implicit $exec
+# CHECK-NEXT: %3:vgpr_32 = V_SUB_U32_e32 %0, %2, implicit $exec
+
+name: dpp_commute_to_rev_sub_same_reg
+tracksRegLiveness: true
+body: |
+ bb.0:
+ liveins: $vgpr0
+ %0:vgpr_32 = COPY $vgpr0
+ %1:vgpr_32 = IMPLICIT_DEF
+ %2:vgpr_32 = V_MOV_B32_dpp %1, %0, 0, 15, 15, 1, implicit $exec
+ %3:vgpr_32 = V_SUB_U32_e32 %0, %2, implicit $exec
+ S_NOP 0, implicit %3
+...
+
+---
+# CHECK-LABEL: name: dpp_commute_to_rev_sub_f32
+# CHECK: %3:vgpr_32 = V_MOV_B32_dpp %2, %0, 1, 15, 15, 1, implicit $exec
+# CHECK-NEXT: %4:vgpr_32 = V_SUB_F32_e32 %1, %3, implicit $mode, implicit $exec
+
+name: dpp_commute_to_rev_sub_f32
+tracksRegLiveness: true
+body: |
+ bb.0:
+ liveins: $vgpr0, $vgpr1
+ %0:vgpr_32 = COPY $vgpr0
+ %1:vgpr_32 = COPY $vgpr1
+ %2:vgpr_32 = IMPLICIT_DEF
+ %3:vgpr_32 = V_MOV_B32_dpp %2, %0, 1, 15, 15, 1, implicit $exec
+ %4:vgpr_32 = V_SUB_F32_e32 %1, %3, implicit $mode, implicit $exec
+ S_NOP 0, implicit %4
+...
+
+---
+# Positive control: the DPP value is already src0 of a non-REV opcode, so no
+# commute is needed and the fold is legal.
+
+# CHECK-LABEL: name: dpp_no_commute_needed
+# CHECK: %4:vgpr_32 = V_SUB_U32_dpp %2, %0, %1, 1, 15, 15, 1, implicit $exec
+
+name: dpp_no_commute_needed
+tracksRegLiveness: true
+body: |
+ bb.0:
+ liveins: $vgpr0, $vgpr1
+ %0:vgpr_32 = COPY $vgpr0
+ %1:vgpr_32 = COPY $vgpr1
+ %2:vgpr_32 = IMPLICIT_DEF
+ %3:vgpr_32 = V_MOV_B32_dpp %2, %0, 1, 15, 15, 1, implicit $exec
+ %4:vgpr_32 = V_SUB_U32_e32 %3, %1, implicit $exec
+ S_NOP 0, implicit %4
+...
+
+---
+# Positive control: commuting an opcode that is its own reverse keeps the same
+# opcode, so it still folds.
+
+# CHECK-LABEL: name: dpp_commute_symmetric
+# CHECK: %4:vgpr_32 = V_ADD_U32_dpp %2, %0, %1, 1, 15, 15, 1, implicit $exec
+# CHECK: %6:vgpr_32 = V_OR_B32_dpp %2, %0, %1, 1, 15, 15, 1, implicit $exec
+
+name: dpp_commute_symmetric
+tracksRegLiveness: true
+body: |
+ bb.0:
+ liveins: $vgpr0, $vgpr1
+ %0:vgpr_32 = COPY $vgpr0
+ %1:vgpr_32 = COPY $vgpr1
+ %2:vgpr_32 = IMPLICIT_DEF
+ %3:vgpr_32 = V_MOV_B32_dpp %2, %0, 1, 15, 15, 1, implicit $exec
+ %4:vgpr_32 = V_ADD_U32_e32 %1, %3, implicit $exec
+ %5:vgpr_32 = V_MOV_B32_dpp %2, %0, 1, 15, 15, 1, implicit $exec
+ %6:vgpr_32 = V_OR_B32_e32 %1, %5, implicit $exec
+ S_NOP 0, implicit %4, implicit %6
+...
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.wqm.demote.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.wqm.demote.ll
index cc1cc6453fa05..f3cb95cdd8697 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.wqm.demote.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.wqm.demote.ll
@@ -681,9 +681,9 @@ define amdgpu_ps void @wqm_deriv(<2 x float> %input, float %arg, i32 %index) {
; SI-NEXT: v_mov_b32_e32 v1, v0
; SI-NEXT: s_xor_b64 s[2:3], s[0:1], -1
; SI-NEXT: s_nop 0
-; SI-NEXT: v_mov_b32_dpp v1, v1 quad_perm:[1,1,1,1] row_mask:0xf bank_mask:0xf bound_ctrl:1
+; SI-NEXT: v_mov_b32_dpp v1, v1 quad_perm:[0,0,0,0] row_mask:0xf bank_mask:0xf bound_ctrl:1
; SI-NEXT: s_nop 1
-; SI-NEXT: v_subrev_f32_dpp v0, v0, v1 quad_perm:[0,0,0,0] row_mask:0xf bank_mask:0xf bound_ctrl:1
+; SI-NEXT: v_sub_f32_dpp v0, v0, v1 quad_perm:[1,1,1,1] row_mask:0xf bank_mask:0xf bound_ctrl:1
; SI-NEXT: ; kill: def $vgpr0 killed $vgpr0 killed $exec
; SI-NEXT: s_and_b64 exec, exec, s[0:1]
; SI-NEXT: v_cmp_neq_f32_e32 vcc, 0, v0
@@ -729,9 +729,9 @@ define amdgpu_ps void @wqm_deriv(<2 x float> %input,...
[truncated]
``````````
</details>
https://github.com/llvm/llvm-project/pull/216835
More information about the llvm-commits
mailing list