[llvm] [AMDGPU] Insert exec-forced V_NOP after V_PERM_PK16 (gfx1250 hazard) (PR #214406)

Jay Foad via llvm-commits llvm-commits at lists.llvm.org
Thu Sep 24 01:21:37 PDT 2026


================
@@ -1199,6 +1199,60 @@ class SIInstrInfo final : public AMDGPUGenInstrInfo {
            Opc == AMDGPU::GLOBAL_WBINV;
   }
 
+  static bool isF16PseudoScalarTrans(unsigned Opcode) {
+    return Opcode == AMDGPU::V_S_EXP_F16_e64 ||
+           Opcode == AMDGPU::V_S_LOG_F16_e64 ||
+           Opcode == AMDGPU::V_S_RCP_F16_e64 ||
+           Opcode == AMDGPU::V_S_RSQ_F16_e64 ||
+           Opcode == AMDGPU::V_S_SQRT_F16_e64;
+  }
+
+  static bool isPseudoScalarTrans(unsigned Opcode) {
+    return isF16PseudoScalarTrans(Opcode) ||
+           Opcode == AMDGPU::V_S_EXP_F32_e64 ||
+           Opcode == AMDGPU::V_S_LOG_F32_e64 ||
+           Opcode == AMDGPU::V_S_RCP_F32_e64 ||
+           Opcode == AMDGPU::V_S_RSQ_F32_e64 ||
+           Opcode == AMDGPU::V_S_SQRT_F32_e64;
+  }
+
+  static bool isF64Trans(unsigned Opcode) {
+    return Opcode == AMDGPU::V_RCP_F64_e32 || Opcode == AMDGPU::V_RCP_F64_e64 ||
+           Opcode == AMDGPU::V_RSQ_F64_e32 || Opcode == AMDGPU::V_RSQ_F64_e64 ||
+           Opcode == AMDGPU::V_SQRT_F64_e32 || Opcode == AMDGPU::V_SQRT_F64_e64;
+  }
+
+  static bool isVPermPk16(unsigned Opcode) {
+    return Opcode == AMDGPU::V_PERM_PK16_B4_U4_e64 ||
+           Opcode == AMDGPU::V_PERM_PK16_B6_U4_e64 ||
+           Opcode == AMDGPU::V_PERM_PK16_B8_U4_e64;
+  }
+
+  // \returns true if \p MI clears the V_PERM_PK16 hazard when it immediately
+  // follows a V_PERM_PK16 (i.e. \p MI is a "safe" instruction).
+  bool isVPermPk16SafeInstr(const MachineInstr &MI) const {
+    unsigned Opc = MI.getOpcode();
+
+    // Only VALU ops issue on the pipe that clears the V_PERM_PK16 hazard.
+    if (!isVALU(MI, /*AllowLDSDMA=*/false))
+      return false;
+    // OP_XDL: matrix (WMMA/SWMMAC/DOT) ops clear the hazard.
+    if (isXDL(MI))
+      return true;
+    // Pseudo-scalar transcendentals (OP32_SCL_T) do NOT clear the hazard.
+    if (isPseudoScalarTrans(Opc))
+      return false;
+    // OP_32_T: genuine transcendentals clear the hazard, except the F64
+    // transcendentals (which belong to the multi-pass FP64 class).
+    if (isTRANS(MI))
+      return !isF64Trans(Opc);
----------------
jayfoad wrote:

Do we really need a specific check for isF64Trans here, or could it be handled by the more general getBlockingCycles check below?

https://github.com/llvm/llvm-project/pull/214406


More information about the llvm-commits mailing list