[llvm] [AMDGPU][MC] Diagnose bf16 inline constants without op_sel in the ass… (PR #223035)
Joe Nash via llvm-commits
llvm-commits at lists.llvm.org
Fri Sep 11 13:04:52 PDT 2026
https://github.com/Sisyph updated https://github.com/llvm/llvm-project/pull/223035
>From 1d3f97664d481f98c30fa7d793f20c8d61886f09 Mon Sep 17 00:00:00 2001
From: Joseph Nash <joseph.nash at amd.com>
Date: Fri, 11 Sep 2026 14:45:16 -0400
Subject: [PATCH] [AMDGPU][MC] Diagnose bf16 inline constants without op_sel in
the assembler
GFX1250 generates a bf16 inline constant in the high half of the
corresponding fp32 inline constant. The VOP1 bf16 opcodes therefore only
read it correctly in the VOP3 encoding with op_sel[0] set:
v_cvt_f32_bf16, v_rcp_bf16, v_sqrt_bf16, v_rsq_bf16, v_log_bf16,
v_exp_bf16, v_sin_bf16, v_cos_bf16 and v_tanh_bf16.
Codegen has applied this workaround since 57a9c4f939e5, but hand-written
assembly and inline asm had no protection at all and silently assembled
to instructions that read the wrong half. Reject those forms in the
assembler.
Only single-source bf16 opcodes are affected. Multi-source and packed
bf16 instructions such as v_fma_mix*_bf16, which also have a scalar bf16
src0, are explicitly excluded.
GFX1310 has the same FeatureBF16InlineConstFromUpperFP32 behaviour, so
the gfx1250 and gfx13 MC tests are updated together.
Assisted-by: Claude Code:claude-opus-5[1m]
---
.../AMDGPU/AsmParser/AMDGPUAsmParser.cpp | 55 ++++++++++++++
llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h | 6 ++
...gfx1250_asm_bf16_inline_const_err-fake16.s | 72 +++++++++++++++++++
.../gfx1250_asm_bf16_inline_const_err.s | 72 +++++++++++++++++++
llvm/test/MC/AMDGPU/gfx1250_asm_vop1-fake16.s | 54 --------------
llvm/test/MC/AMDGPU/gfx1250_asm_vop1.s | 54 --------------
.../gfx1250_asm_vop3_from_vop1-fake16.s | 36 +++++-----
.../MC/AMDGPU/gfx1250_asm_vop3_from_vop1.s | 36 +++++-----
.../gfx13_asm_bf16_inline_const_err-fake16.s | 72 +++++++++++++++++++
.../AMDGPU/gfx13_asm_bf16_inline_const_err.s | 72 +++++++++++++++++++
llvm/test/MC/AMDGPU/gfx13_asm_vop1.s | 54 --------------
.../AMDGPU/gfx13_asm_vop3_from_vop1-fake16.s | 36 +++++-----
.../test/MC/AMDGPU/gfx13_asm_vop3_from_vop1.s | 36 +++++-----
llvm/test/MC/AMDGPU/literals.s | 8 +--
14 files changed, 425 insertions(+), 238 deletions(-)
create mode 100644 llvm/test/MC/AMDGPU/gfx1250_asm_bf16_inline_const_err-fake16.s
create mode 100644 llvm/test/MC/AMDGPU/gfx1250_asm_bf16_inline_const_err.s
create mode 100644 llvm/test/MC/AMDGPU/gfx13_asm_bf16_inline_const_err-fake16.s
create mode 100644 llvm/test/MC/AMDGPU/gfx13_asm_bf16_inline_const_err.s
diff --git a/llvm/lib/Target/AMDGPU/AsmParser/AMDGPUAsmParser.cpp b/llvm/lib/Target/AMDGPU/AsmParser/AMDGPUAsmParser.cpp
index 50733d573b623..cf52274e65189 100644
--- a/llvm/lib/Target/AMDGPU/AsmParser/AMDGPUAsmParser.cpp
+++ b/llvm/lib/Target/AMDGPU/AsmParser/AMDGPUAsmParser.cpp
@@ -1826,6 +1826,8 @@ class AMDGPUAsmParser : public MCTargetAsmParser {
bool validateOffset(const MCInst &Inst, const OperandVector &Operands);
bool validateFlatOffset(const MCInst &Inst, const OperandVector &Operands);
bool validateSMEMOffset(const MCInst &Inst, const OperandVector &Operands);
+ bool validateBF16InlineConst(const MCInst &Inst,
+ const OperandVector &Operands);
bool validateSOPLiteral(const MCInst &Inst, const OperandVector &Operands);
bool validateConstantBusLimitations(const MCInst &Inst,
const OperandVector &Operands);
@@ -4866,6 +4868,56 @@ bool AMDGPUAsmParser::validateSMEMOffset(const MCInst &Inst,
return false;
}
+// On subtargets with FeatureBF16InlineConstFromUpperFP32 the hardware generates
+// a bf16 inline constant in the high half of the corresponding fp32 inline
+// constant. A VOP1 bf16 opcode (v_cvt_f32_bf16 and the bf16 transcendentals)
+// reads the low half of its source, so it must use the VOP3 encoding with
+// op_sel[0] set in order to see the constant at all.
+bool AMDGPUAsmParser::validateBF16InlineConst(const MCInst &Inst,
+ const OperandVector &Operands) {
+ if (!getFeatureBits()[AMDGPU::FeatureBF16InlineConstFromUpperFP32])
+ return true;
+
+ const unsigned Opc = Inst.getOpcode();
+ const MCInstrDesc &Desc = MII.get(Opc);
+ const bool IsVOP3 =
+ SIInstrFlags::isVOP3(Desc) && !SIInstrFlags::isVOP3P(Desc);
+ if (!SIInstrFlags::isVOP1(Desc) && !IsVOP3)
+ return true;
+
+ // Only the single-source VOP1 bf16 opcodes are affected. Multi-source and
+ // packed bf16 instructions such as v_fma_mix*_bf16 are not.
+ if (AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::src1))
+ return true;
+
+ const int Src0Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src0);
+ if (Src0Idx == -1)
+ return true;
+
+ const MCOperandInfo &Src0Info = Desc.operands()[Src0Idx];
+ if (!AMDGPU::isBF16SrcOperand(Src0Info))
+ return true;
+
+ const MCOperand &Src0 = Inst.getOperand(Src0Idx);
+ if (!Src0.isImm() ||
+ !AMDGPU::isInlinableLiteralBF16(static_cast<int16_t>(Src0.getImm()),
+ hasInv2PiInlineImm()))
+ return true;
+
+ if (IsVOP3) {
+ const int ModsIdx =
+ AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src0_modifiers);
+ if (ModsIdx != -1 &&
+ (Inst.getOperand(ModsIdx).getImm() & SISrcMods::OP_SEL_0))
+ return true;
+ }
+
+ Error(getOperandLoc(Operands, Src0Idx),
+ "bf16 inline constant is read from the high half of the fp32 inline "
+ "constant on this GPU; use the e64 encoding with op_sel:[1,0]");
+ return false;
+}
+
bool AMDGPUAsmParser::validateSOPLiteral(const MCInst &Inst,
const OperandVector &Operands) {
unsigned Opcode = Inst.getOpcode();
@@ -5749,6 +5801,9 @@ bool AMDGPUAsmParser::validateInstruction(const MCInst &Inst, SMLoc IDLoc,
if (!validateOffset(Inst, Operands)) {
return false;
}
+ if (!validateBF16InlineConst(Inst, Operands)) {
+ return false;
+ }
if (!validateMAIAccWrite(Inst, Operands)) {
return false;
}
diff --git a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
index 6f875db59917e..22968c499e019 100644
--- a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
+++ b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
@@ -1546,6 +1546,12 @@ inline bool isSISrcOperand(const MCInstrDesc &Desc, unsigned OpNo) {
return isSISrcOperand(Desc.operands()[OpNo]);
}
+/// Is this a scalar (i.e. not packed) bf16 source operand?
+constexpr bool isBF16SrcOperand(const MCOperandInfo &OpInfo) {
+ return OpInfo.OperandType == AMDGPU::OPERAND_REG_IMM_BF16 ||
+ OpInfo.OperandType == AMDGPU::OPERAND_REG_INLINE_C_BF16;
+}
+
/// Is this a KImm operand?
bool isKImmOperand(const MCInstrDesc &Desc, unsigned OpNo);
diff --git a/llvm/test/MC/AMDGPU/gfx1250_asm_bf16_inline_const_err-fake16.s b/llvm/test/MC/AMDGPU/gfx1250_asm_bf16_inline_const_err-fake16.s
new file mode 100644
index 0000000000000..7103e7af7d591
--- /dev/null
+++ b/llvm/test/MC/AMDGPU/gfx1250_asm_bf16_inline_const_err-fake16.s
@@ -0,0 +1,72 @@
+// NOTE: Assertions have been autogenerated by utils/update_mc_test_checks.py UTC_ARGS: --version 6
+// RUN: not llvm-mc -triple=amdgpu12.50 -mattr=-real-true16 -filetype=null %s 2>&1 | FileCheck --check-prefix=ERR --implicit-check-not=error: --strict-whitespace %s
+
+// The hardware generates a bf16 inline constant in the high half of the
+// corresponding fp32 inline constant, so these opcodes must use the VOP3
+// encoding with op_sel[0] set. Anything else silently reads the wrong half.
+
+v_cvt_f32_bf16 v1, 0.5
+// ERR: :[[@LINE-1]]:20: error: bf16 inline constant is read from the high half of the fp32 inline constant on this GPU; use the e64 encoding with op_sel:[1,0]
+
+v_rcp_bf16 v1, 0.5
+// ERR: :[[@LINE-1]]:16: error: bf16 inline constant is read from the high half of the fp32 inline constant on this GPU; use the e64 encoding with op_sel:[1,0]
+
+v_sqrt_bf16 v1, 0.5
+// ERR: :[[@LINE-1]]:17: error: bf16 inline constant is read from the high half of the fp32 inline constant on this GPU; use the e64 encoding with op_sel:[1,0]
+
+v_rsq_bf16 v1, 0.5
+// ERR: :[[@LINE-1]]:16: error: bf16 inline constant is read from the high half of the fp32 inline constant on this GPU; use the e64 encoding with op_sel:[1,0]
+
+v_log_bf16 v1, 0.5
+// ERR: :[[@LINE-1]]:16: error: bf16 inline constant is read from the high half of the fp32 inline constant on this GPU; use the e64 encoding with op_sel:[1,0]
+
+v_exp_bf16 v1, 0.5
+// ERR: :[[@LINE-1]]:16: error: bf16 inline constant is read from the high half of the fp32 inline constant on this GPU; use the e64 encoding with op_sel:[1,0]
+
+v_sin_bf16 v1, 0.5
+// ERR: :[[@LINE-1]]:16: error: bf16 inline constant is read from the high half of the fp32 inline constant on this GPU; use the e64 encoding with op_sel:[1,0]
+
+v_cos_bf16 v1, 0.5
+// ERR: :[[@LINE-1]]:16: error: bf16 inline constant is read from the high half of the fp32 inline constant on this GPU; use the e64 encoding with op_sel:[1,0]
+
+v_tanh_bf16 v1, 0.5
+// ERR: :[[@LINE-1]]:17: error: bf16 inline constant is read from the high half of the fp32 inline constant on this GPU; use the e64 encoding with op_sel:[1,0]
+
+// The VOP1 encoding cannot be fixed up, and the VOP3 encoding is only correct
+// with op_sel[0] set.
+
+v_rcp_bf16_e32 v1, 0.5
+// ERR: :[[@LINE-1]]:20: error: bf16 inline constant is read from the high half of the fp32 inline constant on this GPU; use the e64 encoding with op_sel:[1,0]
+
+v_rcp_bf16_e64 v1, 0.5
+// ERR: :[[@LINE-1]]:20: error: bf16 inline constant is read from the high half of the fp32 inline constant on this GPU; use the e64 encoding with op_sel:[1,0]
+
+// Integer inline constants are affected just like the floating-point ones.
+
+v_rcp_bf16 v1, -1
+// ERR: :[[@LINE-1]]:16: error: bf16 inline constant is read from the high half of the fp32 inline constant on this GPU; use the e64 encoding with op_sel:[1,0]
+
+v_rcp_bf16 v1, 64
+// ERR: :[[@LINE-1]]:16: error: bf16 inline constant is read from the high half of the fp32 inline constant on this GPU; use the e64 encoding with op_sel:[1,0]
+
+// These are fine: a literal is not an inline constant (lit() forces one for a
+// value that would otherwise be inlinable), a register source is unaffected,
+// and the VOP3 encoding with op_sel[0] reads the correct half.
+
+v_rcp_bf16 v1, 0x4049
+
+v_rcp_bf16 v1, lit(0.5)
+
+v_rcp_bf16 v1, v2
+
+v_rcp_bf16 v1, s2
+
+v_rcp_bf16 v1, src_scc
+
+v_rcp_bf16_e64 v1, 0.5 op_sel:[1,0]
+
+v_cvt_f32_bf16_e64 v1, 0.5 op_sel:[1,0]
+
+// f16 opcodes are not affected.
+
+v_rcp_f16 v1, 0.5
diff --git a/llvm/test/MC/AMDGPU/gfx1250_asm_bf16_inline_const_err.s b/llvm/test/MC/AMDGPU/gfx1250_asm_bf16_inline_const_err.s
new file mode 100644
index 0000000000000..54923138c2301
--- /dev/null
+++ b/llvm/test/MC/AMDGPU/gfx1250_asm_bf16_inline_const_err.s
@@ -0,0 +1,72 @@
+// NOTE: Assertions have been autogenerated by utils/update_mc_test_checks.py UTC_ARGS: --version 6
+// RUN: not llvm-mc -triple=amdgpu12.50 -mattr=+real-true16 -filetype=null %s 2>&1 | FileCheck --check-prefix=ERR --implicit-check-not=error: --strict-whitespace %s
+
+// The hardware generates a bf16 inline constant in the high half of the
+// corresponding fp32 inline constant, so these opcodes must use the VOP3
+// encoding with op_sel[0] set. Anything else silently reads the wrong half.
+
+v_cvt_f32_bf16 v1, 0.5
+// ERR: :[[@LINE-1]]:20: error: bf16 inline constant is read from the high half of the fp32 inline constant on this GPU; use the e64 encoding with op_sel:[1,0]
+
+v_rcp_bf16 v1.l, 0.5
+// ERR: :[[@LINE-1]]:18: error: bf16 inline constant is read from the high half of the fp32 inline constant on this GPU; use the e64 encoding with op_sel:[1,0]
+
+v_sqrt_bf16 v1.l, 0.5
+// ERR: :[[@LINE-1]]:19: error: bf16 inline constant is read from the high half of the fp32 inline constant on this GPU; use the e64 encoding with op_sel:[1,0]
+
+v_rsq_bf16 v1.l, 0.5
+// ERR: :[[@LINE-1]]:18: error: bf16 inline constant is read from the high half of the fp32 inline constant on this GPU; use the e64 encoding with op_sel:[1,0]
+
+v_log_bf16 v1.l, 0.5
+// ERR: :[[@LINE-1]]:18: error: bf16 inline constant is read from the high half of the fp32 inline constant on this GPU; use the e64 encoding with op_sel:[1,0]
+
+v_exp_bf16 v1.l, 0.5
+// ERR: :[[@LINE-1]]:18: error: bf16 inline constant is read from the high half of the fp32 inline constant on this GPU; use the e64 encoding with op_sel:[1,0]
+
+v_sin_bf16 v1.l, 0.5
+// ERR: :[[@LINE-1]]:18: error: bf16 inline constant is read from the high half of the fp32 inline constant on this GPU; use the e64 encoding with op_sel:[1,0]
+
+v_cos_bf16 v1.l, 0.5
+// ERR: :[[@LINE-1]]:18: error: bf16 inline constant is read from the high half of the fp32 inline constant on this GPU; use the e64 encoding with op_sel:[1,0]
+
+v_tanh_bf16 v1.l, 0.5
+// ERR: :[[@LINE-1]]:19: error: bf16 inline constant is read from the high half of the fp32 inline constant on this GPU; use the e64 encoding with op_sel:[1,0]
+
+// The VOP1 encoding cannot be fixed up, and the VOP3 encoding is only correct
+// with op_sel[0] set.
+
+v_rcp_bf16_e32 v1.l, 0.5
+// ERR: :[[@LINE-1]]:22: error: bf16 inline constant is read from the high half of the fp32 inline constant on this GPU; use the e64 encoding with op_sel:[1,0]
+
+v_rcp_bf16_e64 v1.l, 0.5
+// ERR: :[[@LINE-1]]:22: error: bf16 inline constant is read from the high half of the fp32 inline constant on this GPU; use the e64 encoding with op_sel:[1,0]
+
+// Integer inline constants are affected just like the floating-point ones.
+
+v_rcp_bf16 v1.l, -1
+// ERR: :[[@LINE-1]]:18: error: bf16 inline constant is read from the high half of the fp32 inline constant on this GPU; use the e64 encoding with op_sel:[1,0]
+
+v_rcp_bf16 v1.l, 64
+// ERR: :[[@LINE-1]]:18: error: bf16 inline constant is read from the high half of the fp32 inline constant on this GPU; use the e64 encoding with op_sel:[1,0]
+
+// These are fine: a literal is not an inline constant (lit() forces one for a
+// value that would otherwise be inlinable), a register source is unaffected,
+// and the VOP3 encoding with op_sel[0] reads the correct half.
+
+v_rcp_bf16 v1.l, 0x4049
+
+v_rcp_bf16 v1.l, lit(0.5)
+
+v_rcp_bf16 v1.l, v2.l
+
+v_rcp_bf16 v1.l, s2
+
+v_rcp_bf16 v1.l, src_scc
+
+v_rcp_bf16_e64 v1.l, 0.5 op_sel:[1,0]
+
+v_cvt_f32_bf16_e64 v1, 0.5 op_sel:[1,0]
+
+// f16 opcodes are not affected.
+
+v_rcp_f16 v1.l, 0.5
diff --git a/llvm/test/MC/AMDGPU/gfx1250_asm_vop1-fake16.s b/llvm/test/MC/AMDGPU/gfx1250_asm_vop1-fake16.s
index 53f9ccff58aa6..d0476d439ba4e 100644
--- a/llvm/test/MC/AMDGPU/gfx1250_asm_vop1-fake16.s
+++ b/llvm/test/MC/AMDGPU/gfx1250_asm_vop1-fake16.s
@@ -151,12 +151,6 @@ v_tanh_bf16 v5, exec_hi
v_tanh_bf16 v5, null
// GFX1250: v_tanh_bf16_e32 v5, null ; encoding: [0x7c,0x94,0x0a,0x7e]
-v_tanh_bf16 v5, -1
-// GFX1250: v_tanh_bf16_e32 v5, -1 ; encoding: [0xc1,0x94,0x0a,0x7e]
-
-v_tanh_bf16 v5, 0.5
-// GFX1250: v_tanh_bf16_e32 v5, 0.5 ; encoding: [0xf0,0x94,0x0a,0x7e]
-
v_tanh_bf16 v5, src_scc
// GFX1250: v_tanh_bf16_e32 v5, src_scc ; encoding: [0xfd,0x94,0x0a,0x7e]
@@ -241,12 +235,6 @@ v_rcp_bf16 v5, exec_hi
v_rcp_bf16 v5, null
// GFX1250: v_rcp_bf16_e32 v5, null ; encoding: [0x7c,0xf2,0x0a,0x7e]
-v_rcp_bf16 v5, -1
-// GFX1250: v_rcp_bf16_e32 v5, -1 ; encoding: [0xc1,0xf2,0x0a,0x7e]
-
-v_rcp_bf16 v5, 0.5
-// GFX1250: v_rcp_bf16_e32 v5, 0.5 ; encoding: [0xf0,0xf2,0x0a,0x7e]
-
v_rcp_bf16 v5, src_scc
// GFX1250: v_rcp_bf16_e32 v5, src_scc ; encoding: [0xfd,0xf2,0x0a,0x7e]
@@ -286,12 +274,6 @@ v_sqrt_bf16 v5, exec_hi
v_sqrt_bf16 v5, null
// GFX1250: v_sqrt_bf16_e32 v5, null ; encoding: [0x7c,0xf4,0x0a,0x7e]
-v_sqrt_bf16 v5, -1
-// GFX1250: v_sqrt_bf16_e32 v5, -1 ; encoding: [0xc1,0xf4,0x0a,0x7e]
-
-v_sqrt_bf16 v5, 0.5
-// GFX1250: v_sqrt_bf16_e32 v5, 0.5 ; encoding: [0xf0,0xf4,0x0a,0x7e]
-
v_sqrt_bf16 v5, src_scc
// GFX1250: v_sqrt_bf16_e32 v5, src_scc ; encoding: [0xfd,0xf4,0x0a,0x7e]
@@ -331,12 +313,6 @@ v_rsq_bf16 v5, exec_hi
v_rsq_bf16 v5, null
// GFX1250: v_rsq_bf16_e32 v5, null ; encoding: [0x7c,0xf6,0x0a,0x7e]
-v_rsq_bf16 v5, -1
-// GFX1250: v_rsq_bf16_e32 v5, -1 ; encoding: [0xc1,0xf6,0x0a,0x7e]
-
-v_rsq_bf16 v5, 0.5
-// GFX1250: v_rsq_bf16_e32 v5, 0.5 ; encoding: [0xf0,0xf6,0x0a,0x7e]
-
v_rsq_bf16 v5, src_scc
// GFX1250: v_rsq_bf16_e32 v5, src_scc ; encoding: [0xfd,0xf6,0x0a,0x7e]
@@ -376,12 +352,6 @@ v_log_bf16 v5, exec_hi
v_log_bf16 v5, null
// GFX1250: v_log_bf16_e32 v5, null ; encoding: [0x7c,0xf8,0x0a,0x7e]
-v_log_bf16 v5, -1
-// GFX1250: v_log_bf16_e32 v5, -1 ; encoding: [0xc1,0xf8,0x0a,0x7e]
-
-v_log_bf16 v5, 0.5
-// GFX1250: v_log_bf16_e32 v5, 0.5 ; encoding: [0xf0,0xf8,0x0a,0x7e]
-
v_log_bf16 v5, src_scc
// GFX1250: v_log_bf16_e32 v5, src_scc ; encoding: [0xfd,0xf8,0x0a,0x7e]
@@ -421,12 +391,6 @@ v_exp_bf16 v5, exec_hi
v_exp_bf16 v5, null
// GFX1250: v_exp_bf16_e32 v5, null ; encoding: [0x7c,0xfa,0x0a,0x7e]
-v_exp_bf16 v5, -1
-// GFX1250: v_exp_bf16_e32 v5, -1 ; encoding: [0xc1,0xfa,0x0a,0x7e]
-
-v_exp_bf16 v5, 0.5
-// GFX1250: v_exp_bf16_e32 v5, 0.5 ; encoding: [0xf0,0xfa,0x0a,0x7e]
-
v_exp_bf16 v5, src_scc
// GFX1250: v_exp_bf16_e32 v5, src_scc ; encoding: [0xfd,0xfa,0x0a,0x7e]
@@ -466,12 +430,6 @@ v_sin_bf16 v5, exec_hi
v_sin_bf16 v5, null
// GFX1250: v_sin_bf16_e32 v5, null ; encoding: [0x7c,0xfc,0x0a,0x7e]
-v_sin_bf16 v5, -1
-// GFX1250: v_sin_bf16_e32 v5, -1 ; encoding: [0xc1,0xfc,0x0a,0x7e]
-
-v_sin_bf16 v5, 0.5
-// GFX1250: v_sin_bf16_e32 v5, 0.5 ; encoding: [0xf0,0xfc,0x0a,0x7e]
-
v_sin_bf16 v5, src_scc
// GFX1250: v_sin_bf16_e32 v5, src_scc ; encoding: [0xfd,0xfc,0x0a,0x7e]
@@ -511,12 +469,6 @@ v_cos_bf16 v5, exec_hi
v_cos_bf16 v5, null
// GFX1250: v_cos_bf16_e32 v5, null ; encoding: [0x7c,0xfe,0x0a,0x7e]
-v_cos_bf16 v5, -1
-// GFX1250: v_cos_bf16_e32 v5, -1 ; encoding: [0xc1,0xfe,0x0a,0x7e]
-
-v_cos_bf16 v5, 0.5
-// GFX1250: v_cos_bf16_e32 v5, 0.5 ; encoding: [0xf0,0xfe,0x0a,0x7e]
-
v_cos_bf16 v5, src_scc
// GFX1250: v_cos_bf16_e32 v5, src_scc ; encoding: [0xfd,0xfe,0x0a,0x7e]
@@ -556,12 +508,6 @@ v_cvt_f32_bf16 v5, exec_hi
v_cvt_f32_bf16 v5, null
// GFX1250: v_cvt_f32_bf16_e32 v5, null ; encoding: [0x7c,0xe4,0x0a,0x7e]
-v_cvt_f32_bf16 v5, -1
-// GFX1250: v_cvt_f32_bf16_e32 v5, -1 ; encoding: [0xc1,0xe4,0x0a,0x7e]
-
-v_cvt_f32_bf16 v5, 0.5
-// GFX1250: v_cvt_f32_bf16_e32 v5, 0.5 ; encoding: [0xf0,0xe4,0x0a,0x7e]
-
v_cvt_f32_bf16 v5, src_scc
// GFX1250: v_cvt_f32_bf16_e32 v5, src_scc ; encoding: [0xfd,0xe4,0x0a,0x7e]
diff --git a/llvm/test/MC/AMDGPU/gfx1250_asm_vop1.s b/llvm/test/MC/AMDGPU/gfx1250_asm_vop1.s
index 09e4719c0c8cd..131432652055b 100644
--- a/llvm/test/MC/AMDGPU/gfx1250_asm_vop1.s
+++ b/llvm/test/MC/AMDGPU/gfx1250_asm_vop1.s
@@ -157,12 +157,6 @@ v_tanh_bf16 v5.l, exec_hi
v_tanh_bf16 v5.l, null
// GFX1250: v_tanh_bf16_e32 v5.l, null ; encoding: [0x7c,0x94,0x0a,0x7e]
-v_tanh_bf16 v5.l, -1
-// GFX1250: v_tanh_bf16_e32 v5.l, -1 ; encoding: [0xc1,0x94,0x0a,0x7e]
-
-v_tanh_bf16 v5.l, 0.5
-// GFX1250: v_tanh_bf16_e32 v5.l, 0.5 ; encoding: [0xf0,0x94,0x0a,0x7e]
-
v_tanh_bf16 v5.l, src_scc
// GFX1250: v_tanh_bf16_e32 v5.l, src_scc ; encoding: [0xfd,0x94,0x0a,0x7e]
@@ -250,12 +244,6 @@ v_rcp_bf16 v5.l, exec_hi
v_rcp_bf16 v5.l, null
// GFX1250: v_rcp_bf16_e32 v5.l, null ; encoding: [0x7c,0xf2,0x0a,0x7e]
-v_rcp_bf16 v5.l, -1
-// GFX1250: v_rcp_bf16_e32 v5.l, -1 ; encoding: [0xc1,0xf2,0x0a,0x7e]
-
-v_rcp_bf16 v5.l, 0.5
-// GFX1250: v_rcp_bf16_e32 v5.l, 0.5 ; encoding: [0xf0,0xf2,0x0a,0x7e]
-
v_rcp_bf16 v5.l, src_scc
// GFX1250: v_rcp_bf16_e32 v5.l, src_scc ; encoding: [0xfd,0xf2,0x0a,0x7e]
@@ -298,12 +286,6 @@ v_sqrt_bf16 v5.l, exec_hi
v_sqrt_bf16 v5.l, null
// GFX1250: v_sqrt_bf16_e32 v5.l, null ; encoding: [0x7c,0xf4,0x0a,0x7e]
-v_sqrt_bf16 v5.l, -1
-// GFX1250: v_sqrt_bf16_e32 v5.l, -1 ; encoding: [0xc1,0xf4,0x0a,0x7e]
-
-v_sqrt_bf16 v5.l, 0.5
-// GFX1250: v_sqrt_bf16_e32 v5.l, 0.5 ; encoding: [0xf0,0xf4,0x0a,0x7e]
-
v_sqrt_bf16 v5.l, src_scc
// GFX1250: v_sqrt_bf16_e32 v5.l, src_scc ; encoding: [0xfd,0xf4,0x0a,0x7e]
@@ -346,12 +328,6 @@ v_rsq_bf16 v5.l, exec_hi
v_rsq_bf16 v5.l, null
// GFX1250: v_rsq_bf16_e32 v5.l, null ; encoding: [0x7c,0xf6,0x0a,0x7e]
-v_rsq_bf16 v5.l, -1
-// GFX1250: v_rsq_bf16_e32 v5.l, -1 ; encoding: [0xc1,0xf6,0x0a,0x7e]
-
-v_rsq_bf16 v5.l, 0.5
-// GFX1250: v_rsq_bf16_e32 v5.l, 0.5 ; encoding: [0xf0,0xf6,0x0a,0x7e]
-
v_rsq_bf16 v5.l, src_scc
// GFX1250: v_rsq_bf16_e32 v5.l, src_scc ; encoding: [0xfd,0xf6,0x0a,0x7e]
@@ -394,12 +370,6 @@ v_log_bf16 v5.l, exec_hi
v_log_bf16 v5.l, null
// GFX1250: v_log_bf16_e32 v5.l, null ; encoding: [0x7c,0xf8,0x0a,0x7e]
-v_log_bf16 v5.l, -1
-// GFX1250: v_log_bf16_e32 v5.l, -1 ; encoding: [0xc1,0xf8,0x0a,0x7e]
-
-v_log_bf16 v5.l, 0.5
-// GFX1250: v_log_bf16_e32 v5.l, 0.5 ; encoding: [0xf0,0xf8,0x0a,0x7e]
-
v_log_bf16 v5.l, src_scc
// GFX1250: v_log_bf16_e32 v5.l, src_scc ; encoding: [0xfd,0xf8,0x0a,0x7e]
@@ -442,12 +412,6 @@ v_exp_bf16 v5.l, exec_hi
v_exp_bf16 v5.l, null
// GFX1250: v_exp_bf16_e32 v5.l, null ; encoding: [0x7c,0xfa,0x0a,0x7e]
-v_exp_bf16 v5.l, -1
-// GFX1250: v_exp_bf16_e32 v5.l, -1 ; encoding: [0xc1,0xfa,0x0a,0x7e]
-
-v_exp_bf16 v5.l, 0.5
-// GFX1250: v_exp_bf16_e32 v5.l, 0.5 ; encoding: [0xf0,0xfa,0x0a,0x7e]
-
v_exp_bf16 v5.l, src_scc
// GFX1250: v_exp_bf16_e32 v5.l, src_scc ; encoding: [0xfd,0xfa,0x0a,0x7e]
@@ -490,12 +454,6 @@ v_sin_bf16 v5.l, exec_hi
v_sin_bf16 v5.l, null
// GFX1250: v_sin_bf16_e32 v5.l, null ; encoding: [0x7c,0xfc,0x0a,0x7e]
-v_sin_bf16 v5.l, -1
-// GFX1250: v_sin_bf16_e32 v5.l, -1 ; encoding: [0xc1,0xfc,0x0a,0x7e]
-
-v_sin_bf16 v5.l, 0.5
-// GFX1250: v_sin_bf16_e32 v5.l, 0.5 ; encoding: [0xf0,0xfc,0x0a,0x7e]
-
v_sin_bf16 v5.l, src_scc
// GFX1250: v_sin_bf16_e32 v5.l, src_scc ; encoding: [0xfd,0xfc,0x0a,0x7e]
@@ -538,12 +496,6 @@ v_cos_bf16 v5.l, exec_hi
v_cos_bf16 v5.l, null
// GFX1250: v_cos_bf16_e32 v5.l, null ; encoding: [0x7c,0xfe,0x0a,0x7e]
-v_cos_bf16 v5.l, -1
-// GFX1250: v_cos_bf16_e32 v5.l, -1 ; encoding: [0xc1,0xfe,0x0a,0x7e]
-
-v_cos_bf16 v5.l, 0.5
-// GFX1250: v_cos_bf16_e32 v5.l, 0.5 ; encoding: [0xf0,0xfe,0x0a,0x7e]
-
v_cos_bf16 v5.l, src_scc
// GFX1250: v_cos_bf16_e32 v5.l, src_scc ; encoding: [0xfd,0xfe,0x0a,0x7e]
@@ -586,12 +538,6 @@ v_cvt_f32_bf16 v5, exec_hi
v_cvt_f32_bf16 v5, null
// GFX1250: v_cvt_f32_bf16_e32 v5, null ; encoding: [0x7c,0xe4,0x0a,0x7e]
-v_cvt_f32_bf16 v5, -1
-// GFX1250: v_cvt_f32_bf16_e32 v5, -1 ; encoding: [0xc1,0xe4,0x0a,0x7e]
-
-v_cvt_f32_bf16 v5, 0.5
-// GFX1250: v_cvt_f32_bf16_e32 v5, 0.5 ; encoding: [0xf0,0xe4,0x0a,0x7e]
-
v_cvt_f32_bf16 v5, src_scc
// GFX1250: v_cvt_f32_bf16_e32 v5, src_scc ; encoding: [0xfd,0xe4,0x0a,0x7e]
diff --git a/llvm/test/MC/AMDGPU/gfx1250_asm_vop3_from_vop1-fake16.s b/llvm/test/MC/AMDGPU/gfx1250_asm_vop3_from_vop1-fake16.s
index 4ec6599374ac3..9c04ac56ee143 100644
--- a/llvm/test/MC/AMDGPU/gfx1250_asm_vop3_from_vop1-fake16.s
+++ b/llvm/test/MC/AMDGPU/gfx1250_asm_vop3_from_vop1-fake16.s
@@ -3784,8 +3784,8 @@ v_tanh_bf16_e64 v5, exec_hi
v_tanh_bf16_e64 v5, null
// GFX1250: v_tanh_bf16_e64 v5, null ; encoding: [0x05,0x00,0xca,0xd5,0x7c,0x00,0x01,0x02]
-v_tanh_bf16_e64 v5, -1
-// GFX1250: v_tanh_bf16_e64 v5, -1 ; encoding: [0x05,0x00,0xca,0xd5,0xc1,0x00,0x01,0x02]
+v_tanh_bf16_e64 v5, -1 op_sel:[1,0]
+// GFX1250: v_tanh_bf16_e64 v5, -1 op_sel:[1,0] ; encoding: [0x05,0x08,0xca,0xd5,0xc1,0x00,0x01,0x02]
v_tanh_bf16_e64 v5, v128 op_sel:[1,0]
// GFX1250: v_tanh_bf16_e64 v5, v128 op_sel:[1,0] ; encoding: [0x05,0x08,0xca,0xd5,0x80,0x01,0x01,0x02]
@@ -3865,8 +3865,8 @@ v_rcp_bf16_e64 v5, exec_hi
v_rcp_bf16_e64 v5, null
// GFX1250: v_rcp_bf16_e64 v5, null ; encoding: [0x05,0x00,0xf9,0xd5,0x7c,0x00,0x01,0x02]
-v_rcp_bf16_e64 v5, -1
-// GFX1250: v_rcp_bf16_e64 v5, -1 ; encoding: [0x05,0x00,0xf9,0xd5,0xc1,0x00,0x01,0x02]
+v_rcp_bf16_e64 v5, -1 op_sel:[1,0]
+// GFX1250: v_rcp_bf16_e64 v5, -1 op_sel:[1,0] ; encoding: [0x05,0x08,0xf9,0xd5,0xc1,0x00,0x01,0x02]
v_rcp_bf16_e64 v5, v128 op_sel:[1,1]
// GFX1250: v_rcp_bf16_e64 v5, v128 op_sel:[1,1] ; encoding: [0x05,0x48,0xf9,0xd5,0x80,0x01,0x01,0x02]
@@ -3910,8 +3910,8 @@ v_sqrt_bf16_e64 v5, exec_hi
v_sqrt_bf16_e64 v5, null
// GFX1250: v_sqrt_bf16_e64 v5, null ; encoding: [0x05,0x00,0xfa,0xd5,0x7c,0x00,0x01,0x02]
-v_sqrt_bf16_e64 v5, -1
-// GFX1250: v_sqrt_bf16_e64 v5, -1 ; encoding: [0x05,0x00,0xfa,0xd5,0xc1,0x00,0x01,0x02]
+v_sqrt_bf16_e64 v5, -1 op_sel:[1,0]
+// GFX1250: v_sqrt_bf16_e64 v5, -1 op_sel:[1,0] ; encoding: [0x05,0x08,0xfa,0xd5,0xc1,0x00,0x01,0x02]
v_sqrt_bf16_e64 v5, v128 op_sel:[1,1]
// GFX1250: v_sqrt_bf16_e64 v5, v128 op_sel:[1,1] ; encoding: [0x05,0x48,0xfa,0xd5,0x80,0x01,0x01,0x02]
@@ -3955,8 +3955,8 @@ v_rsq_bf16_e64 v5, exec_hi
v_rsq_bf16_e64 v5, null
// GFX1250: v_rsq_bf16_e64 v5, null ; encoding: [0x05,0x00,0xfb,0xd5,0x7c,0x00,0x01,0x02]
-v_rsq_bf16_e64 v5, -1
-// GFX1250: v_rsq_bf16_e64 v5, -1 ; encoding: [0x05,0x00,0xfb,0xd5,0xc1,0x00,0x01,0x02]
+v_rsq_bf16_e64 v5, -1 op_sel:[1,0]
+// GFX1250: v_rsq_bf16_e64 v5, -1 op_sel:[1,0] ; encoding: [0x05,0x08,0xfb,0xd5,0xc1,0x00,0x01,0x02]
v_rsq_bf16_e64 v5, v128 op_sel:[1,1]
// GFX1250: v_rsq_bf16_e64 v5, v128 op_sel:[1,1] ; encoding: [0x05,0x48,0xfb,0xd5,0x80,0x01,0x01,0x02]
@@ -4000,8 +4000,8 @@ v_log_bf16_e64 v5, exec_hi
v_log_bf16_e64 v5, null
// GFX1250: v_log_bf16_e64 v5, null ; encoding: [0x05,0x00,0xfc,0xd5,0x7c,0x00,0x01,0x02]
-v_log_bf16_e64 v5, -1
-// GFX1250: v_log_bf16_e64 v5, -1 ; encoding: [0x05,0x00,0xfc,0xd5,0xc1,0x00,0x01,0x02]
+v_log_bf16_e64 v5, -1 op_sel:[1,0]
+// GFX1250: v_log_bf16_e64 v5, -1 op_sel:[1,0] ; encoding: [0x05,0x08,0xfc,0xd5,0xc1,0x00,0x01,0x02]
v_log_bf16_e64 v5, v128 op_sel:[1,1]
// GFX1250: v_log_bf16_e64 v5, v128 op_sel:[1,1] ; encoding: [0x05,0x48,0xfc,0xd5,0x80,0x01,0x01,0x02]
@@ -4045,8 +4045,8 @@ v_exp_bf16_e64 v5, exec_hi
v_exp_bf16_e64 v5, null
// GFX1250: v_exp_bf16_e64 v5, null ; encoding: [0x05,0x00,0xfd,0xd5,0x7c,0x00,0x01,0x02]
-v_exp_bf16_e64 v5, -1
-// GFX1250: v_exp_bf16_e64 v5, -1 ; encoding: [0x05,0x00,0xfd,0xd5,0xc1,0x00,0x01,0x02]
+v_exp_bf16_e64 v5, -1 op_sel:[1,0]
+// GFX1250: v_exp_bf16_e64 v5, -1 op_sel:[1,0] ; encoding: [0x05,0x08,0xfd,0xd5,0xc1,0x00,0x01,0x02]
v_exp_bf16_e64 v5, v128 op_sel:[1,1]
// GFX1250: v_exp_bf16_e64 v5, v128 op_sel:[1,1] ; encoding: [0x05,0x48,0xfd,0xd5,0x80,0x01,0x01,0x02]
@@ -4090,8 +4090,8 @@ v_sin_bf16_e64 v5, exec_hi
v_sin_bf16_e64 v5, null
// GFX1250: v_sin_bf16_e64 v5, null ; encoding: [0x05,0x00,0xfe,0xd5,0x7c,0x00,0x01,0x02]
-v_sin_bf16_e64 v5, -1
-// GFX1250: v_sin_bf16_e64 v5, -1 ; encoding: [0x05,0x00,0xfe,0xd5,0xc1,0x00,0x01,0x02]
+v_sin_bf16_e64 v5, -1 op_sel:[1,0]
+// GFX1250: v_sin_bf16_e64 v5, -1 op_sel:[1,0] ; encoding: [0x05,0x08,0xfe,0xd5,0xc1,0x00,0x01,0x02]
v_sin_bf16_e64 v5, v128 op_sel:[1,1]
// GFX1250: v_sin_bf16_e64 v5, v128 op_sel:[1,1] ; encoding: [0x05,0x48,0xfe,0xd5,0x80,0x01,0x01,0x02]
@@ -4135,8 +4135,8 @@ v_cos_bf16_e64 v5, exec_hi
v_cos_bf16_e64 v5, null
// GFX1250: v_cos_bf16_e64 v5, null ; encoding: [0x05,0x00,0xff,0xd5,0x7c,0x00,0x01,0x02]
-v_cos_bf16_e64 v5, -1
-// GFX1250: v_cos_bf16_e64 v5, -1 ; encoding: [0x05,0x00,0xff,0xd5,0xc1,0x00,0x01,0x02]
+v_cos_bf16_e64 v5, -1 op_sel:[1,0]
+// GFX1250: v_cos_bf16_e64 v5, -1 op_sel:[1,0] ; encoding: [0x05,0x08,0xff,0xd5,0xc1,0x00,0x01,0x02]
v_cos_bf16_e64 v5, v128 op_sel:[1,1]
// GFX1250: v_cos_bf16_e64 v5, v128 op_sel:[1,1] ; encoding: [0x05,0x48,0xff,0xd5,0x80,0x01,0x01,0x02]
@@ -4174,8 +4174,8 @@ v_cvt_f32_bf16_e64 v5, exec_hi
v_cvt_f32_bf16_e64 v5, null
// GFX1250: v_cvt_f32_bf16_e64 v5, null ; encoding: [0x05,0x00,0xf2,0xd5,0x7c,0x00,0x01,0x02]
-v_cvt_f32_bf16_e64 v5, -1
-// GFX1250: v_cvt_f32_bf16_e64 v5, -1 ; encoding: [0x05,0x00,0xf2,0xd5,0xc1,0x00,0x01,0x02]
+v_cvt_f32_bf16_e64 v5, -1 op_sel:[1,0]
+// GFX1250: v_cvt_f32_bf16_e64 v5, -1 op_sel:[1,0] ; encoding: [0x05,0x08,0xf2,0xd5,0xc1,0x00,0x01,0x02]
v_cvt_f32_bf16_e64 v5, v1 op_sel:[1]
// GFX1250: v_cvt_f32_bf16_e64 v5, v1 op_sel:[1,0] ; encoding: [0x05,0x08,0xf2,0xd5,0x01,0x01,0x01,0x02]
diff --git a/llvm/test/MC/AMDGPU/gfx1250_asm_vop3_from_vop1.s b/llvm/test/MC/AMDGPU/gfx1250_asm_vop3_from_vop1.s
index 9588f067667cf..feb90c9bd7958 100644
--- a/llvm/test/MC/AMDGPU/gfx1250_asm_vop3_from_vop1.s
+++ b/llvm/test/MC/AMDGPU/gfx1250_asm_vop3_from_vop1.s
@@ -3955,8 +3955,8 @@ v_tanh_bf16_e64 v5.l, exec_hi
v_tanh_bf16_e64 v5.l, null
// GFX1250: v_tanh_bf16_e64 v5.l, null ; encoding: [0x05,0x00,0xca,0xd5,0x7c,0x00,0x01,0x02]
-v_tanh_bf16_e64 v5.l, -1
-// GFX1250: v_tanh_bf16_e64 v5.l, -1 ; encoding: [0x05,0x00,0xca,0xd5,0xc1,0x00,0x01,0x02]
+v_tanh_bf16_e64 v5.l, -1 op_sel:[1,0]
+// GFX1250: v_tanh_bf16_e64 v5.l, -1 op_sel:[1,0] ; encoding: [0x05,0x08,0xca,0xd5,0xc1,0x00,0x01,0x02]
v_tanh_bf16 v5.l, v128.h
// GFX1250: v_tanh_bf16_e64 v5.l, v128.h op_sel:[1,0] ; encoding: [0x05,0x08,0xca,0xd5,0x80,0x01,0x01,0x02]
@@ -4036,8 +4036,8 @@ v_rcp_bf16_e64 v5.l, exec_hi
v_rcp_bf16_e64 v5.l, null
// GFX1250: v_rcp_bf16_e64 v5.l, null ; encoding: [0x05,0x00,0xf9,0xd5,0x7c,0x00,0x01,0x02]
-v_rcp_bf16_e64 v5.l, -1
-// GFX1250: v_rcp_bf16_e64 v5.l, -1 ; encoding: [0x05,0x00,0xf9,0xd5,0xc1,0x00,0x01,0x02]
+v_rcp_bf16_e64 v5.l, -1 op_sel:[1,0]
+// GFX1250: v_rcp_bf16_e64 v5.l, -1 op_sel:[1,0] ; encoding: [0x05,0x08,0xf9,0xd5,0xc1,0x00,0x01,0x02]
v_rcp_bf16 v5.h, v128.h
// GFX1250: v_rcp_bf16_e64 v5.h, v128.h op_sel:[1,1] ; encoding: [0x05,0x48,0xf9,0xd5,0x80,0x01,0x01,0x02]
@@ -4081,8 +4081,8 @@ v_sqrt_bf16_e64 v5.l, exec_hi
v_sqrt_bf16_e64 v5.l, null
// GFX1250: v_sqrt_bf16_e64 v5.l, null ; encoding: [0x05,0x00,0xfa,0xd5,0x7c,0x00,0x01,0x02]
-v_sqrt_bf16_e64 v5.l, -1
-// GFX1250: v_sqrt_bf16_e64 v5.l, -1 ; encoding: [0x05,0x00,0xfa,0xd5,0xc1,0x00,0x01,0x02]
+v_sqrt_bf16_e64 v5.l, -1 op_sel:[1,0]
+// GFX1250: v_sqrt_bf16_e64 v5.l, -1 op_sel:[1,0] ; encoding: [0x05,0x08,0xfa,0xd5,0xc1,0x00,0x01,0x02]
v_sqrt_bf16 v5.h, v128.h
// GFX1250: v_sqrt_bf16_e64 v5.h, v128.h op_sel:[1,1] ; encoding: [0x05,0x48,0xfa,0xd5,0x80,0x01,0x01,0x02]
@@ -4126,8 +4126,8 @@ v_rsq_bf16_e64 v5.l, exec_hi
v_rsq_bf16_e64 v5.l, null
// GFX1250: v_rsq_bf16_e64 v5.l, null ; encoding: [0x05,0x00,0xfb,0xd5,0x7c,0x00,0x01,0x02]
-v_rsq_bf16_e64 v5.l, -1
-// GFX1250: v_rsq_bf16_e64 v5.l, -1 ; encoding: [0x05,0x00,0xfb,0xd5,0xc1,0x00,0x01,0x02]
+v_rsq_bf16_e64 v5.l, -1 op_sel:[1,0]
+// GFX1250: v_rsq_bf16_e64 v5.l, -1 op_sel:[1,0] ; encoding: [0x05,0x08,0xfb,0xd5,0xc1,0x00,0x01,0x02]
v_rsq_bf16 v5.h, v128.h
// GFX1250: v_rsq_bf16_e64 v5.h, v128.h op_sel:[1,1] ; encoding: [0x05,0x48,0xfb,0xd5,0x80,0x01,0x01,0x02]
@@ -4171,8 +4171,8 @@ v_log_bf16_e64 v5.l, exec_hi
v_log_bf16_e64 v5.l, null
// GFX1250: v_log_bf16_e64 v5.l, null ; encoding: [0x05,0x00,0xfc,0xd5,0x7c,0x00,0x01,0x02]
-v_log_bf16_e64 v5.l, -1
-// GFX1250: v_log_bf16_e64 v5.l, -1 ; encoding: [0x05,0x00,0xfc,0xd5,0xc1,0x00,0x01,0x02]
+v_log_bf16_e64 v5.l, -1 op_sel:[1,0]
+// GFX1250: v_log_bf16_e64 v5.l, -1 op_sel:[1,0] ; encoding: [0x05,0x08,0xfc,0xd5,0xc1,0x00,0x01,0x02]
v_log_bf16 v5.h, v128.h
// GFX1250: v_log_bf16_e64 v5.h, v128.h op_sel:[1,1] ; encoding: [0x05,0x48,0xfc,0xd5,0x80,0x01,0x01,0x02]
@@ -4216,8 +4216,8 @@ v_exp_bf16_e64 v5.l, exec_hi
v_exp_bf16_e64 v5.l, null
// GFX1250: v_exp_bf16_e64 v5.l, null ; encoding: [0x05,0x00,0xfd,0xd5,0x7c,0x00,0x01,0x02]
-v_exp_bf16_e64 v5.l, -1
-// GFX1250: v_exp_bf16_e64 v5.l, -1 ; encoding: [0x05,0x00,0xfd,0xd5,0xc1,0x00,0x01,0x02]
+v_exp_bf16_e64 v5.l, -1 op_sel:[1,0]
+// GFX1250: v_exp_bf16_e64 v5.l, -1 op_sel:[1,0] ; encoding: [0x05,0x08,0xfd,0xd5,0xc1,0x00,0x01,0x02]
v_exp_bf16 v5.h, v128.h
// GFX1250: v_exp_bf16_e64 v5.h, v128.h op_sel:[1,1] ; encoding: [0x05,0x48,0xfd,0xd5,0x80,0x01,0x01,0x02]
@@ -4261,8 +4261,8 @@ v_sin_bf16_e64 v5.l, exec_hi
v_sin_bf16_e64 v5.l, null
// GFX1250: v_sin_bf16_e64 v5.l, null ; encoding: [0x05,0x00,0xfe,0xd5,0x7c,0x00,0x01,0x02]
-v_sin_bf16_e64 v5.l, -1
-// GFX1250: v_sin_bf16_e64 v5.l, -1 ; encoding: [0x05,0x00,0xfe,0xd5,0xc1,0x00,0x01,0x02]
+v_sin_bf16_e64 v5.l, -1 op_sel:[1,0]
+// GFX1250: v_sin_bf16_e64 v5.l, -1 op_sel:[1,0] ; encoding: [0x05,0x08,0xfe,0xd5,0xc1,0x00,0x01,0x02]
v_sin_bf16 v5.h, v128.h
// GFX1250: v_sin_bf16_e64 v5.h, v128.h op_sel:[1,1] ; encoding: [0x05,0x48,0xfe,0xd5,0x80,0x01,0x01,0x02]
@@ -4306,8 +4306,8 @@ v_cos_bf16_e64 v5.l, exec_hi
v_cos_bf16_e64 v5.l, null
// GFX1250: v_cos_bf16_e64 v5.l, null ; encoding: [0x05,0x00,0xff,0xd5,0x7c,0x00,0x01,0x02]
-v_cos_bf16_e64 v5.l, -1
-// GFX1250: v_cos_bf16_e64 v5.l, -1 ; encoding: [0x05,0x00,0xff,0xd5,0xc1,0x00,0x01,0x02]
+v_cos_bf16_e64 v5.l, -1 op_sel:[1,0]
+// GFX1250: v_cos_bf16_e64 v5.l, -1 op_sel:[1,0] ; encoding: [0x05,0x08,0xff,0xd5,0xc1,0x00,0x01,0x02]
v_cos_bf16_e64 v5.h, v128.h
// GFX1250: v_cos_bf16_e64 v5.h, v128.h op_sel:[1,1] ; encoding: [0x05,0x48,0xff,0xd5,0x80,0x01,0x01,0x02]
@@ -4345,8 +4345,8 @@ v_cvt_f32_bf16_e64 v5, exec_hi
v_cvt_f32_bf16_e64 v5, null
// GFX1250: v_cvt_f32_bf16_e64 v5, null ; encoding: [0x05,0x00,0xf2,0xd5,0x7c,0x00,0x01,0x02]
-v_cvt_f32_bf16_e64 v5, -1
-// GFX1250: v_cvt_f32_bf16_e64 v5, -1 ; encoding: [0x05,0x00,0xf2,0xd5,0xc1,0x00,0x01,0x02]
+v_cvt_f32_bf16_e64 v5, -1 op_sel:[1,0]
+// GFX1250: v_cvt_f32_bf16_e64 v5, -1 op_sel:[1,0] ; encoding: [0x05,0x08,0xf2,0xd5,0xc1,0x00,0x01,0x02]
v_cvt_f32_bf16_e64 v5, v1.h op_sel:[1,0]
// GFX1250: v_cvt_f32_bf16_e64 v5, v1.h op_sel:[1,0] ; encoding: [0x05,0x08,0xf2,0xd5,0x01,0x01,0x01,0x02]
diff --git a/llvm/test/MC/AMDGPU/gfx13_asm_bf16_inline_const_err-fake16.s b/llvm/test/MC/AMDGPU/gfx13_asm_bf16_inline_const_err-fake16.s
new file mode 100644
index 0000000000000..1d8c4b790010d
--- /dev/null
+++ b/llvm/test/MC/AMDGPU/gfx13_asm_bf16_inline_const_err-fake16.s
@@ -0,0 +1,72 @@
+// NOTE: Assertions have been autogenerated by utils/update_mc_test_checks.py UTC_ARGS: --version 6
+// RUN: not llvm-mc -triple=amdgpu13.10 -mattr=-real-true16 -filetype=null %s 2>&1 | FileCheck --check-prefix=ERR --implicit-check-not=error: --strict-whitespace %s
+
+// The hardware generates a bf16 inline constant in the high half of the
+// corresponding fp32 inline constant, so these opcodes must use the VOP3
+// encoding with op_sel[0] set. Anything else silently reads the wrong half.
+
+v_cvt_f32_bf16 v1, 0.5
+// ERR: :[[@LINE-1]]:20: error: bf16 inline constant is read from the high half of the fp32 inline constant on this GPU; use the e64 encoding with op_sel:[1,0]
+
+v_rcp_bf16 v1, 0.5
+// ERR: :[[@LINE-1]]:16: error: bf16 inline constant is read from the high half of the fp32 inline constant on this GPU; use the e64 encoding with op_sel:[1,0]
+
+v_sqrt_bf16 v1, 0.5
+// ERR: :[[@LINE-1]]:17: error: bf16 inline constant is read from the high half of the fp32 inline constant on this GPU; use the e64 encoding with op_sel:[1,0]
+
+v_rsq_bf16 v1, 0.5
+// ERR: :[[@LINE-1]]:16: error: bf16 inline constant is read from the high half of the fp32 inline constant on this GPU; use the e64 encoding with op_sel:[1,0]
+
+v_log_bf16 v1, 0.5
+// ERR: :[[@LINE-1]]:16: error: bf16 inline constant is read from the high half of the fp32 inline constant on this GPU; use the e64 encoding with op_sel:[1,0]
+
+v_exp_bf16 v1, 0.5
+// ERR: :[[@LINE-1]]:16: error: bf16 inline constant is read from the high half of the fp32 inline constant on this GPU; use the e64 encoding with op_sel:[1,0]
+
+v_sin_bf16 v1, 0.5
+// ERR: :[[@LINE-1]]:16: error: bf16 inline constant is read from the high half of the fp32 inline constant on this GPU; use the e64 encoding with op_sel:[1,0]
+
+v_cos_bf16 v1, 0.5
+// ERR: :[[@LINE-1]]:16: error: bf16 inline constant is read from the high half of the fp32 inline constant on this GPU; use the e64 encoding with op_sel:[1,0]
+
+v_tanh_bf16 v1, 0.5
+// ERR: :[[@LINE-1]]:17: error: bf16 inline constant is read from the high half of the fp32 inline constant on this GPU; use the e64 encoding with op_sel:[1,0]
+
+// The VOP1 encoding cannot be fixed up, and the VOP3 encoding is only correct
+// with op_sel[0] set.
+
+v_rcp_bf16_e32 v1, 0.5
+// ERR: :[[@LINE-1]]:20: error: bf16 inline constant is read from the high half of the fp32 inline constant on this GPU; use the e64 encoding with op_sel:[1,0]
+
+v_rcp_bf16_e64 v1, 0.5
+// ERR: :[[@LINE-1]]:20: error: bf16 inline constant is read from the high half of the fp32 inline constant on this GPU; use the e64 encoding with op_sel:[1,0]
+
+// Integer inline constants are affected just like the floating-point ones.
+
+v_rcp_bf16 v1, -1
+// ERR: :[[@LINE-1]]:16: error: bf16 inline constant is read from the high half of the fp32 inline constant on this GPU; use the e64 encoding with op_sel:[1,0]
+
+v_rcp_bf16 v1, 64
+// ERR: :[[@LINE-1]]:16: error: bf16 inline constant is read from the high half of the fp32 inline constant on this GPU; use the e64 encoding with op_sel:[1,0]
+
+// These are fine: a literal is not an inline constant (lit() forces one for a
+// value that would otherwise be inlinable), a register source is unaffected,
+// and the VOP3 encoding with op_sel[0] reads the correct half.
+
+v_rcp_bf16 v1, 0x4049
+
+v_rcp_bf16 v1, lit(0.5)
+
+v_rcp_bf16 v1, v2
+
+v_rcp_bf16 v1, s2
+
+v_rcp_bf16 v1, src_scc
+
+v_rcp_bf16_e64 v1, 0.5 op_sel:[1,0]
+
+v_cvt_f32_bf16_e64 v1, 0.5 op_sel:[1,0]
+
+// f16 opcodes are not affected.
+
+v_rcp_f16 v1, 0.5
diff --git a/llvm/test/MC/AMDGPU/gfx13_asm_bf16_inline_const_err.s b/llvm/test/MC/AMDGPU/gfx13_asm_bf16_inline_const_err.s
new file mode 100644
index 0000000000000..c88cb762e0710
--- /dev/null
+++ b/llvm/test/MC/AMDGPU/gfx13_asm_bf16_inline_const_err.s
@@ -0,0 +1,72 @@
+// NOTE: Assertions have been autogenerated by utils/update_mc_test_checks.py UTC_ARGS: --version 6
+// RUN: not llvm-mc -triple=amdgpu13.10 -mattr=+real-true16 -filetype=null %s 2>&1 | FileCheck --check-prefix=ERR --implicit-check-not=error: --strict-whitespace %s
+
+// The hardware generates a bf16 inline constant in the high half of the
+// corresponding fp32 inline constant, so these opcodes must use the VOP3
+// encoding with op_sel[0] set. Anything else silently reads the wrong half.
+
+v_cvt_f32_bf16 v1, 0.5
+// ERR: :[[@LINE-1]]:20: error: bf16 inline constant is read from the high half of the fp32 inline constant on this GPU; use the e64 encoding with op_sel:[1,0]
+
+v_rcp_bf16 v1.l, 0.5
+// ERR: :[[@LINE-1]]:18: error: bf16 inline constant is read from the high half of the fp32 inline constant on this GPU; use the e64 encoding with op_sel:[1,0]
+
+v_sqrt_bf16 v1.l, 0.5
+// ERR: :[[@LINE-1]]:19: error: bf16 inline constant is read from the high half of the fp32 inline constant on this GPU; use the e64 encoding with op_sel:[1,0]
+
+v_rsq_bf16 v1.l, 0.5
+// ERR: :[[@LINE-1]]:18: error: bf16 inline constant is read from the high half of the fp32 inline constant on this GPU; use the e64 encoding with op_sel:[1,0]
+
+v_log_bf16 v1.l, 0.5
+// ERR: :[[@LINE-1]]:18: error: bf16 inline constant is read from the high half of the fp32 inline constant on this GPU; use the e64 encoding with op_sel:[1,0]
+
+v_exp_bf16 v1.l, 0.5
+// ERR: :[[@LINE-1]]:18: error: bf16 inline constant is read from the high half of the fp32 inline constant on this GPU; use the e64 encoding with op_sel:[1,0]
+
+v_sin_bf16 v1.l, 0.5
+// ERR: :[[@LINE-1]]:18: error: bf16 inline constant is read from the high half of the fp32 inline constant on this GPU; use the e64 encoding with op_sel:[1,0]
+
+v_cos_bf16 v1.l, 0.5
+// ERR: :[[@LINE-1]]:18: error: bf16 inline constant is read from the high half of the fp32 inline constant on this GPU; use the e64 encoding with op_sel:[1,0]
+
+v_tanh_bf16 v1.l, 0.5
+// ERR: :[[@LINE-1]]:19: error: bf16 inline constant is read from the high half of the fp32 inline constant on this GPU; use the e64 encoding with op_sel:[1,0]
+
+// The VOP1 encoding cannot be fixed up, and the VOP3 encoding is only correct
+// with op_sel[0] set.
+
+v_rcp_bf16_e32 v1.l, 0.5
+// ERR: :[[@LINE-1]]:22: error: bf16 inline constant is read from the high half of the fp32 inline constant on this GPU; use the e64 encoding with op_sel:[1,0]
+
+v_rcp_bf16_e64 v1.l, 0.5
+// ERR: :[[@LINE-1]]:22: error: bf16 inline constant is read from the high half of the fp32 inline constant on this GPU; use the e64 encoding with op_sel:[1,0]
+
+// Integer inline constants are affected just like the floating-point ones.
+
+v_rcp_bf16 v1.l, -1
+// ERR: :[[@LINE-1]]:18: error: bf16 inline constant is read from the high half of the fp32 inline constant on this GPU; use the e64 encoding with op_sel:[1,0]
+
+v_rcp_bf16 v1.l, 64
+// ERR: :[[@LINE-1]]:18: error: bf16 inline constant is read from the high half of the fp32 inline constant on this GPU; use the e64 encoding with op_sel:[1,0]
+
+// These are fine: a literal is not an inline constant (lit() forces one for a
+// value that would otherwise be inlinable), a register source is unaffected,
+// and the VOP3 encoding with op_sel[0] reads the correct half.
+
+v_rcp_bf16 v1.l, 0x4049
+
+v_rcp_bf16 v1.l, lit(0.5)
+
+v_rcp_bf16 v1.l, v2.l
+
+v_rcp_bf16 v1.l, s2
+
+v_rcp_bf16 v1.l, src_scc
+
+v_rcp_bf16_e64 v1.l, 0.5 op_sel:[1,0]
+
+v_cvt_f32_bf16_e64 v1, 0.5 op_sel:[1,0]
+
+// f16 opcodes are not affected.
+
+v_rcp_f16 v1.l, 0.5
diff --git a/llvm/test/MC/AMDGPU/gfx13_asm_vop1.s b/llvm/test/MC/AMDGPU/gfx13_asm_vop1.s
index 0cc6fd15adb45..632baa4e566de 100644
--- a/llvm/test/MC/AMDGPU/gfx13_asm_vop1.s
+++ b/llvm/test/MC/AMDGPU/gfx13_asm_vop1.s
@@ -305,12 +305,6 @@ v_cos_bf16 v5.l, exec_hi
v_cos_bf16 v5.l, null
// GFX13: v_cos_bf16_e32 v5.l, null ; encoding: [0x7c,0xfe,0x0a,0x7e]
-v_cos_bf16 v5.l, -1
-// GFX13: v_cos_bf16_e32 v5.l, -1 ; encoding: [0xc1,0xfe,0x0a,0x7e]
-
-v_cos_bf16 v5.l, 0.5
-// GFX13: v_cos_bf16_e32 v5.l, 0.5 ; encoding: [0xf0,0xfe,0x0a,0x7e]
-
v_cos_bf16 v5.l, src_scc
// GFX13: v_cos_bf16_e32 v5.l, src_scc ; encoding: [0xfd,0xfe,0x0a,0x7e]
@@ -512,12 +506,6 @@ v_cvt_f32_bf16 v5, exec_hi
v_cvt_f32_bf16 v5, null
// GFX13: v_cvt_f32_bf16_e32 v5, null ; encoding: [0x7c,0xe4,0x0a,0x7e]
-v_cvt_f32_bf16 v5, -1
-// GFX13: v_cvt_f32_bf16_e32 v5, -1 ; encoding: [0xc1,0xe4,0x0a,0x7e]
-
-v_cvt_f32_bf16 v5, 0.5
-// GFX13: v_cvt_f32_bf16_e32 v5, 0.5 ; encoding: [0xf0,0xe4,0x0a,0x7e]
-
v_cvt_f32_bf16 v5, src_scc
// GFX13: v_cvt_f32_bf16_e32 v5, src_scc ; encoding: [0xfd,0xe4,0x0a,0x7e]
@@ -1922,12 +1910,6 @@ v_exp_bf16 v5.l, exec_hi
v_exp_bf16 v5.l, null
// GFX13: v_exp_bf16_e32 v5.l, null ; encoding: [0x7c,0xfa,0x0a,0x7e]
-v_exp_bf16 v5.l, -1
-// GFX13: v_exp_bf16_e32 v5.l, -1 ; encoding: [0xc1,0xfa,0x0a,0x7e]
-
-v_exp_bf16 v5.l, 0.5
-// GFX13: v_exp_bf16_e32 v5.l, 0.5 ; encoding: [0xf0,0xfa,0x0a,0x7e]
-
v_exp_bf16 v5.l, src_scc
// GFX13: v_exp_bf16_e32 v5.l, src_scc ; encoding: [0xfd,0xfa,0x0a,0x7e]
@@ -2720,12 +2702,6 @@ v_log_bf16 v5.l, exec_hi
v_log_bf16 v5.l, null
// GFX13: v_log_bf16_e32 v5.l, null ; encoding: [0x7c,0xf8,0x0a,0x7e]
-v_log_bf16 v5.l, -1
-// GFX13: v_log_bf16_e32 v5.l, -1 ; encoding: [0xc1,0xf8,0x0a,0x7e]
-
-v_log_bf16 v5.l, 0.5
-// GFX13: v_log_bf16_e32 v5.l, 0.5 ; encoding: [0xf0,0xf8,0x0a,0x7e]
-
v_log_bf16 v5.l, src_scc
// GFX13: v_log_bf16_e32 v5.l, src_scc ; encoding: [0xfd,0xf8,0x0a,0x7e]
@@ -3116,12 +3092,6 @@ v_rcp_bf16 v5.l, exec_hi
v_rcp_bf16 v5.l, null
// GFX13: v_rcp_bf16_e32 v5.l, null ; encoding: [0x7c,0xf2,0x0a,0x7e]
-v_rcp_bf16 v5.l, -1
-// GFX13: v_rcp_bf16_e32 v5.l, -1 ; encoding: [0xc1,0xf2,0x0a,0x7e]
-
-v_rcp_bf16 v5.l, 0.5
-// GFX13: v_rcp_bf16_e32 v5.l, 0.5 ; encoding: [0xf0,0xf2,0x0a,0x7e]
-
v_rcp_bf16 v5.l, src_scc
// GFX13: v_rcp_bf16_e32 v5.l, src_scc ; encoding: [0xfd,0xf2,0x0a,0x7e]
@@ -3491,12 +3461,6 @@ v_rsq_bf16 v5.l, exec_hi
v_rsq_bf16 v5.l, null
// GFX13: v_rsq_bf16_e32 v5.l, null ; encoding: [0x7c,0xf6,0x0a,0x7e]
-v_rsq_bf16 v5.l, -1
-// GFX13: v_rsq_bf16_e32 v5.l, -1 ; encoding: [0xc1,0xf6,0x0a,0x7e]
-
-v_rsq_bf16 v5.l, 0.5
-// GFX13: v_rsq_bf16_e32 v5.l, 0.5 ; encoding: [0xf0,0xf6,0x0a,0x7e]
-
v_rsq_bf16 v5.l, src_scc
// GFX13: v_rsq_bf16_e32 v5.l, src_scc ; encoding: [0xfd,0xf6,0x0a,0x7e]
@@ -3749,12 +3713,6 @@ v_sin_bf16 v5.l, exec_hi
v_sin_bf16 v5.l, null
// GFX13: v_sin_bf16_e32 v5.l, null ; encoding: [0x7c,0xfc,0x0a,0x7e]
-v_sin_bf16 v5.l, -1
-// GFX13: v_sin_bf16_e32 v5.l, -1 ; encoding: [0xc1,0xfc,0x0a,0x7e]
-
-v_sin_bf16 v5.l, 0.5
-// GFX13: v_sin_bf16_e32 v5.l, 0.5 ; encoding: [0xf0,0xfc,0x0a,0x7e]
-
v_sin_bf16 v5.l, src_scc
// GFX13: v_sin_bf16_e32 v5.l, src_scc ; encoding: [0xfd,0xfc,0x0a,0x7e]
@@ -3887,12 +3845,6 @@ v_sqrt_bf16 v5.l, exec_hi
v_sqrt_bf16 v5.l, null
// GFX13: v_sqrt_bf16_e32 v5.l, null ; encoding: [0x7c,0xf4,0x0a,0x7e]
-v_sqrt_bf16 v5.l, -1
-// GFX13: v_sqrt_bf16_e32 v5.l, -1 ; encoding: [0xc1,0xf4,0x0a,0x7e]
-
-v_sqrt_bf16 v5.l, 0.5
-// GFX13: v_sqrt_bf16_e32 v5.l, 0.5 ; encoding: [0xf0,0xf4,0x0a,0x7e]
-
v_sqrt_bf16 v5.l, src_scc
// GFX13: v_sqrt_bf16_e32 v5.l, src_scc ; encoding: [0xfd,0xf4,0x0a,0x7e]
@@ -4088,12 +4040,6 @@ v_tanh_bf16 v5.l, exec_hi
v_tanh_bf16 v5.l, null
// GFX13: v_tanh_bf16_e32 v5.l, null ; encoding: [0x7c,0x94,0x0a,0x7e]
-v_tanh_bf16 v5.l, -1
-// GFX13: v_tanh_bf16_e32 v5.l, -1 ; encoding: [0xc1,0x94,0x0a,0x7e]
-
-v_tanh_bf16 v5.l, 0.5
-// GFX13: v_tanh_bf16_e32 v5.l, 0.5 ; encoding: [0xf0,0x94,0x0a,0x7e]
-
v_tanh_bf16 v5.l, src_scc
// GFX13: v_tanh_bf16_e32 v5.l, src_scc ; encoding: [0xfd,0x94,0x0a,0x7e]
diff --git a/llvm/test/MC/AMDGPU/gfx13_asm_vop3_from_vop1-fake16.s b/llvm/test/MC/AMDGPU/gfx13_asm_vop3_from_vop1-fake16.s
index fcd087f247ac6..25ece42570501 100644
--- a/llvm/test/MC/AMDGPU/gfx13_asm_vop3_from_vop1-fake16.s
+++ b/llvm/test/MC/AMDGPU/gfx13_asm_vop3_from_vop1-fake16.s
@@ -263,8 +263,8 @@ v_clz_i32_u32_e64 v5, src_scc
v_clz_i32_u32_e64 v255, 0xaf123456
// GFX13: v_clz_i32_u32_e64 v255, 0xaf123456 ; encoding: [0xff,0x00,0xb9,0xd5,0xff,0x00,0x01,0x02,0x56,0x34,0x12,0xaf]
-v_cos_bf16_e64 v5, -1
-// GFX13: v_cos_bf16_e64 v5, -1 ; encoding: [0x05,0x00,0xff,0xd5,0xc1,0x00,0x01,0x02]
+v_cos_bf16_e64 v5, -1 op_sel:[1,0]
+// GFX13: v_cos_bf16_e64 v5, -1 op_sel:[1,0] ; encoding: [0x05,0x08,0xff,0xd5,0xc1,0x00,0x01,0x02]
v_cos_bf16_e64 v5, exec_hi
// GFX13: v_cos_bf16_e64 v5, exec_hi ; encoding: [0x05,0x00,0xff,0xd5,0x7f,0x00,0x01,0x02]
@@ -476,8 +476,8 @@ v_cvt_f16_fp8 v150, s2
v_cvt_f16_fp8 v150, v2
// GFX13: v_cvt_f16_fp8_e64 v150, v2 ; encoding: [0x96,0x00,0xf7,0xd5,0x02,0x01,0x01,0x02]
-v_cvt_f32_bf16_e64 v5, -1
-// GFX13: v_cvt_f32_bf16_e64 v5, -1 ; encoding: [0x05,0x00,0xf2,0xd5,0xc1,0x00,0x01,0x02]
+v_cvt_f32_bf16_e64 v5, -1 op_sel:[1,0]
+// GFX13: v_cvt_f32_bf16_e64 v5, -1 op_sel:[1,0] ; encoding: [0x05,0x08,0xf2,0xd5,0xc1,0x00,0x01,0x02]
v_cvt_f32_bf16_e64 v5, exec_hi
// GFX13: v_cvt_f32_bf16_e64 v5, exec_hi ; encoding: [0x05,0x00,0xf2,0xd5,0x7f,0x00,0x01,0x02]
@@ -1907,8 +1907,8 @@ v_cvt_u32_u16_e64 v5, src_scc
v_cvt_u32_u16_e64 v255, 0xfe0b
// GFX13: v_cvt_u32_u16_e64 v255, 0xfe0b ; encoding: [0xff,0x00,0xeb,0xd5,0xff,0x00,0x01,0x02,0x0b,0xfe,0x00,0x00]
-v_exp_bf16_e64 v5, -1
-// GFX13: v_exp_bf16_e64 v5, -1 ; encoding: [0x05,0x00,0xfd,0xd5,0xc1,0x00,0x01,0x02]
+v_exp_bf16_e64 v5, -1 op_sel:[1,0]
+// GFX13: v_exp_bf16_e64 v5, -1 op_sel:[1,0] ; encoding: [0x05,0x08,0xfd,0xd5,0xc1,0x00,0x01,0x02]
v_exp_bf16_e64 v5, exec_hi
// GFX13: v_exp_bf16_e64 v5, exec_hi ; encoding: [0x05,0x00,0xfd,0xd5,0x7f,0x00,0x01,0x02]
@@ -2672,8 +2672,8 @@ v_frexp_mant_f64_e64 v[5:6], -|src_scc| mul:4
v_frexp_mant_f64_e64 v[254:255], 0xaf123456 clamp div:2
// GFX13: v_frexp_mant_f64_e64 v[254:255], 0xaf123456 clamp div:2 ; encoding: [0xfe,0x80,0xbd,0xd5,0xff,0x00,0x01,0x1a,0x56,0x34,0x12,0xaf]
-v_log_bf16_e64 v5, -1
-// GFX13: v_log_bf16_e64 v5, -1 ; encoding: [0x05,0x00,0xfc,0xd5,0xc1,0x00,0x01,0x02]
+v_log_bf16_e64 v5, -1 op_sel:[1,0]
+// GFX13: v_log_bf16_e64 v5, -1 op_sel:[1,0] ; encoding: [0x05,0x08,0xfc,0xd5,0xc1,0x00,0x01,0x02]
v_log_bf16_e64 v5, exec_hi
// GFX13: v_log_bf16_e64 v5, exec_hi ; encoding: [0x05,0x00,0xfc,0xd5,0x7f,0x00,0x01,0x02]
@@ -2996,8 +2996,8 @@ v_prng_b32_e64 v5, vcc_hi
v_prng_b32_e64 v5, vcc_lo
// GFX13: v_prng_b32_e64 v5, vcc_lo ; encoding: [0x05,0x00,0xcb,0xd5,0x6a,0x00,0x01,0x02]
-v_rcp_bf16_e64 v5, -1
-// GFX13: v_rcp_bf16_e64 v5, -1 ; encoding: [0x05,0x00,0xf9,0xd5,0xc1,0x00,0x01,0x02]
+v_rcp_bf16_e64 v5, -1 op_sel:[1,0]
+// GFX13: v_rcp_bf16_e64 v5, -1 op_sel:[1,0] ; encoding: [0x05,0x08,0xf9,0xd5,0xc1,0x00,0x01,0x02]
v_rcp_bf16_e64 v5, exec_hi
// GFX13: v_rcp_bf16_e64 v5, exec_hi ; encoding: [0x05,0x00,0xf9,0xd5,0x7f,0x00,0x01,0x02]
@@ -3329,8 +3329,8 @@ v_rndne_f64_e64 v[5:6], -|src_scc| mul:4
v_rndne_f64_e64 v[254:255], 0xaf123456 clamp div:2
// GFX13: v_rndne_f64_e64 v[254:255], 0xaf123456 clamp div:2 ; encoding: [0xfe,0x80,0x99,0xd5,0xff,0x00,0x01,0x1a,0x56,0x34,0x12,0xaf]
-v_rsq_bf16_e64 v5, -1
-// GFX13: v_rsq_bf16_e64 v5, -1 ; encoding: [0x05,0x00,0xfb,0xd5,0xc1,0x00,0x01,0x02]
+v_rsq_bf16_e64 v5, -1 op_sel:[1,0]
+// GFX13: v_rsq_bf16_e64 v5, -1 op_sel:[1,0] ; encoding: [0x05,0x08,0xfb,0xd5,0xc1,0x00,0x01,0x02]
v_rsq_bf16_e64 v5, exec_hi
// GFX13: v_rsq_bf16_e64 v5, exec_hi ; encoding: [0x05,0x00,0xfb,0xd5,0x7f,0x00,0x01,0x02]
@@ -3560,8 +3560,8 @@ v_sat_pk_u8_i16_e64 v5, src_scc
v_sat_pk_u8_i16_e64 v255, 0xfe0b
// GFX13: v_sat_pk_u8_i16_e64 v255, 0xfe0b ; encoding: [0xff,0x00,0xe2,0xd5,0xff,0x00,0x01,0x02,0x0b,0xfe,0x00,0x00]
-v_sin_bf16_e64 v5, -1
-// GFX13: v_sin_bf16_e64 v5, -1 ; encoding: [0x05,0x00,0xfe,0xd5,0xc1,0x00,0x01,0x02]
+v_sin_bf16_e64 v5, -1 op_sel:[1,0]
+// GFX13: v_sin_bf16_e64 v5, -1 op_sel:[1,0] ; encoding: [0x05,0x08,0xfe,0xd5,0xc1,0x00,0x01,0x02]
v_sin_bf16_e64 v5, exec_hi
// GFX13: v_sin_bf16_e64 v5, exec_hi ; encoding: [0x05,0x00,0xfe,0xd5,0x7f,0x00,0x01,0x02]
@@ -3686,8 +3686,8 @@ v_sin_f32_e64 v5, src_scc mul:4
v_sin_f32_e64 v255, -|0xaf123456| clamp div:2
// GFX13: v_sin_f32_e64 v255, -|0xaf123456| clamp div:2 ; encoding: [0xff,0x81,0xb5,0xd5,0xff,0x00,0x01,0x3a,0x56,0x34,0x12,0xaf]
-v_sqrt_bf16_e64 v5, -1
-// GFX13: v_sqrt_bf16_e64 v5, -1 ; encoding: [0x05,0x00,0xfa,0xd5,0xc1,0x00,0x01,0x02]
+v_sqrt_bf16_e64 v5, -1 op_sel:[1,0]
+// GFX13: v_sqrt_bf16_e64 v5, -1 op_sel:[1,0] ; encoding: [0x05,0x08,0xfa,0xd5,0xc1,0x00,0x01,0x02]
v_sqrt_bf16_e64 v5, exec_hi
// GFX13: v_sqrt_bf16_e64 v5, exec_hi ; encoding: [0x05,0x00,0xfa,0xd5,0x7f,0x00,0x01,0x02]
@@ -3848,8 +3848,8 @@ v_sqrt_f64_e64 v[5:6], -|src_scc| mul:4
v_sqrt_f64_e64 v[254:255], 0xaf123456 clamp div:2
// GFX13: v_sqrt_f64_e64 v[254:255], 0xaf123456 clamp div:2 ; encoding: [0xfe,0x80,0xb4,0xd5,0xff,0x00,0x01,0x1a,0x56,0x34,0x12,0xaf]
-v_tanh_bf16_e64 v5, -1
-// GFX13: v_tanh_bf16_e64 v5, -1 ; encoding: [0x05,0x00,0xca,0xd5,0xc1,0x00,0x01,0x02]
+v_tanh_bf16_e64 v5, -1 op_sel:[1,0]
+// GFX13: v_tanh_bf16_e64 v5, -1 op_sel:[1,0] ; encoding: [0x05,0x08,0xca,0xd5,0xc1,0x00,0x01,0x02]
v_tanh_bf16_e64 v5, exec_hi
// GFX13: v_tanh_bf16_e64 v5, exec_hi ; encoding: [0x05,0x00,0xca,0xd5,0x7f,0x00,0x01,0x02]
diff --git a/llvm/test/MC/AMDGPU/gfx13_asm_vop3_from_vop1.s b/llvm/test/MC/AMDGPU/gfx13_asm_vop3_from_vop1.s
index fa113d853139e..0e92904d00447 100644
--- a/llvm/test/MC/AMDGPU/gfx13_asm_vop3_from_vop1.s
+++ b/llvm/test/MC/AMDGPU/gfx13_asm_vop3_from_vop1.s
@@ -308,8 +308,8 @@ v_clz_i32_u32_e64 v5, src_scc
v_clz_i32_u32_e64 v255, 0xaf123456
// GFX13: v_clz_i32_u32_e64 v255, 0xaf123456 ; encoding: [0xff,0x00,0xb9,0xd5,0xff,0x00,0x01,0x02,0x56,0x34,0x12,0xaf]
-v_cos_bf16_e64 v5.l, -1
-// GFX13: v_cos_bf16_e64 v5.l, -1 ; encoding: [0x05,0x00,0xff,0xd5,0xc1,0x00,0x01,0x02]
+v_cos_bf16_e64 v5.l, -1 op_sel:[1,0]
+// GFX13: v_cos_bf16_e64 v5.l, -1 op_sel:[1,0] ; encoding: [0x05,0x08,0xff,0xd5,0xc1,0x00,0x01,0x02]
v_cos_bf16_e64 v5.l, exec_hi
// GFX13: v_cos_bf16_e64 v5.l, exec_hi ; encoding: [0x05,0x00,0xff,0xd5,0x7f,0x00,0x01,0x02]
@@ -524,8 +524,8 @@ v_cvt_f16_fp8 v150.l, s2
v_cvt_f16_fp8 v150.l, v2
// GFX13: v_cvt_f16_fp8_e64 v150.l, v2 ; encoding: [0x96,0x00,0xf7,0xd5,0x02,0x01,0x01,0x02]
-v_cvt_f32_bf16_e64 v5, -1
-// GFX13: v_cvt_f32_bf16_e64 v5, -1 ; encoding: [0x05,0x00,0xf2,0xd5,0xc1,0x00,0x01,0x02]
+v_cvt_f32_bf16_e64 v5, -1 op_sel:[1,0]
+// GFX13: v_cvt_f32_bf16_e64 v5, -1 op_sel:[1,0] ; encoding: [0x05,0x08,0xf2,0xd5,0xc1,0x00,0x01,0x02]
v_cvt_f32_bf16_e64 v5, exec_hi
// GFX13: v_cvt_f32_bf16_e64 v5, exec_hi ; encoding: [0x05,0x00,0xf2,0xd5,0x7f,0x00,0x01,0x02]
@@ -1979,8 +1979,8 @@ v_cvt_u32_u16_e64 v5, src_scc
v_cvt_u32_u16_e64 v255, 0xfe0b
// GFX13: v_cvt_u32_u16_e64 v255, 0xfe0b ; encoding: [0xff,0x00,0xeb,0xd5,0xff,0x00,0x01,0x02,0x0b,0xfe,0x00,0x00]
-v_exp_bf16_e64 v5.l, -1
-// GFX13: v_exp_bf16_e64 v5.l, -1 ; encoding: [0x05,0x00,0xfd,0xd5,0xc1,0x00,0x01,0x02]
+v_exp_bf16_e64 v5.l, -1 op_sel:[1,0]
+// GFX13: v_exp_bf16_e64 v5.l, -1 op_sel:[1,0] ; encoding: [0x05,0x08,0xfd,0xd5,0xc1,0x00,0x01,0x02]
v_exp_bf16_e64 v5.l, exec_hi
// GFX13: v_exp_bf16_e64 v5.l, exec_hi ; encoding: [0x05,0x00,0xfd,0xd5,0x7f,0x00,0x01,0x02]
@@ -2747,8 +2747,8 @@ v_frexp_mant_f64_e64 v[5:6], -|src_scc| mul:4
v_frexp_mant_f64_e64 v[254:255], 0xaf123456 clamp div:2
// GFX13: v_frexp_mant_f64_e64 v[254:255], 0xaf123456 clamp div:2 ; encoding: [0xfe,0x80,0xbd,0xd5,0xff,0x00,0x01,0x1a,0x56,0x34,0x12,0xaf]
-v_log_bf16_e64 v5.l, -1
-// GFX13: v_log_bf16_e64 v5.l, -1 ; encoding: [0x05,0x00,0xfc,0xd5,0xc1,0x00,0x01,0x02]
+v_log_bf16_e64 v5.l, -1 op_sel:[1,0]
+// GFX13: v_log_bf16_e64 v5.l, -1 op_sel:[1,0] ; encoding: [0x05,0x08,0xfc,0xd5,0xc1,0x00,0x01,0x02]
v_log_bf16_e64 v5.l, exec_hi
// GFX13: v_log_bf16_e64 v5.l, exec_hi ; encoding: [0x05,0x00,0xfc,0xd5,0x7f,0x00,0x01,0x02]
@@ -3128,8 +3128,8 @@ v_prng_b32_e64 v5, vcc_hi
v_prng_b32_e64 v5, vcc_lo
// GFX13: v_prng_b32_e64 v5, vcc_lo ; encoding: [0x05,0x00,0xcb,0xd5,0x6a,0x00,0x01,0x02]
-v_rcp_bf16_e64 v5.l, -1
-// GFX13: v_rcp_bf16_e64 v5.l, -1 ; encoding: [0x05,0x00,0xf9,0xd5,0xc1,0x00,0x01,0x02]
+v_rcp_bf16_e64 v5.l, -1 op_sel:[1,0]
+// GFX13: v_rcp_bf16_e64 v5.l, -1 op_sel:[1,0] ; encoding: [0x05,0x08,0xf9,0xd5,0xc1,0x00,0x01,0x02]
v_rcp_bf16_e64 v5.l, exec_hi
// GFX13: v_rcp_bf16_e64 v5.l, exec_hi ; encoding: [0x05,0x00,0xf9,0xd5,0x7f,0x00,0x01,0x02]
@@ -3464,8 +3464,8 @@ v_rndne_f64_e64 v[5:6], -|src_scc| mul:4
v_rndne_f64_e64 v[254:255], 0xaf123456 clamp div:2
// GFX13: v_rndne_f64_e64 v[254:255], 0xaf123456 clamp div:2 ; encoding: [0xfe,0x80,0x99,0xd5,0xff,0x00,0x01,0x1a,0x56,0x34,0x12,0xaf]
-v_rsq_bf16_e64 v5.l, -1
-// GFX13: v_rsq_bf16_e64 v5.l, -1 ; encoding: [0x05,0x00,0xfb,0xd5,0xc1,0x00,0x01,0x02]
+v_rsq_bf16_e64 v5.l, -1 op_sel:[1,0]
+// GFX13: v_rsq_bf16_e64 v5.l, -1 op_sel:[1,0] ; encoding: [0x05,0x08,0xfb,0xd5,0xc1,0x00,0x01,0x02]
v_rsq_bf16_e64 v5.l, exec_hi
// GFX13: v_rsq_bf16_e64 v5.l, exec_hi ; encoding: [0x05,0x00,0xfb,0xd5,0x7f,0x00,0x01,0x02]
@@ -3704,8 +3704,8 @@ v_sat_pk_u8_i16_e64 v5.l, src_scc
v_sat_pk_u8_i16_e64 v255.l, 0xfe0b
// GFX13: v_sat_pk_u8_i16_e64 v255.l, 0xfe0b ; encoding: [0xff,0x00,0xe2,0xd5,0xff,0x00,0x01,0x02,0x0b,0xfe,0x00,0x00]
-v_sin_bf16_e64 v5.l, -1
-// GFX13: v_sin_bf16_e64 v5.l, -1 ; encoding: [0x05,0x00,0xfe,0xd5,0xc1,0x00,0x01,0x02]
+v_sin_bf16_e64 v5.l, -1 op_sel:[1,0]
+// GFX13: v_sin_bf16_e64 v5.l, -1 op_sel:[1,0] ; encoding: [0x05,0x08,0xfe,0xd5,0xc1,0x00,0x01,0x02]
v_sin_bf16_e64 v5.l, exec_hi
// GFX13: v_sin_bf16_e64 v5.l, exec_hi ; encoding: [0x05,0x00,0xfe,0xd5,0x7f,0x00,0x01,0x02]
@@ -3833,8 +3833,8 @@ v_sin_f32_e64 v5, src_scc mul:4
v_sin_f32_e64 v255, -|0xaf123456| clamp div:2
// GFX13: v_sin_f32_e64 v255, -|0xaf123456| clamp div:2 ; encoding: [0xff,0x81,0xb5,0xd5,0xff,0x00,0x01,0x3a,0x56,0x34,0x12,0xaf]
-v_sqrt_bf16_e64 v5.l, -1
-// GFX13: v_sqrt_bf16_e64 v5.l, -1 ; encoding: [0x05,0x00,0xfa,0xd5,0xc1,0x00,0x01,0x02]
+v_sqrt_bf16_e64 v5.l, -1 op_sel:[1,0]
+// GFX13: v_sqrt_bf16_e64 v5.l, -1 op_sel:[1,0] ; encoding: [0x05,0x08,0xfa,0xd5,0xc1,0x00,0x01,0x02]
v_sqrt_bf16_e64 v5.l, exec_hi
// GFX13: v_sqrt_bf16_e64 v5.l, exec_hi ; encoding: [0x05,0x00,0xfa,0xd5,0x7f,0x00,0x01,0x02]
@@ -3998,8 +3998,8 @@ v_sqrt_f64_e64 v[5:6], -|src_scc| mul:4
v_sqrt_f64_e64 v[254:255], 0xaf123456 clamp div:2
// GFX13: v_sqrt_f64_e64 v[254:255], 0xaf123456 clamp div:2 ; encoding: [0xfe,0x80,0xb4,0xd5,0xff,0x00,0x01,0x1a,0x56,0x34,0x12,0xaf]
-v_tanh_bf16_e64 v5.l, -1
-// GFX13: v_tanh_bf16_e64 v5.l, -1 ; encoding: [0x05,0x00,0xca,0xd5,0xc1,0x00,0x01,0x02]
+v_tanh_bf16_e64 v5.l, -1 op_sel:[1,0]
+// GFX13: v_tanh_bf16_e64 v5.l, -1 op_sel:[1,0] ; encoding: [0x05,0x08,0xca,0xd5,0xc1,0x00,0x01,0x02]
v_tanh_bf16_e64 v5.l, exec_hi
// GFX13: v_tanh_bf16_e64 v5.l, exec_hi ; encoding: [0x05,0x00,0xca,0xd5,0x7f,0x00,0x01,0x02]
diff --git a/llvm/test/MC/AMDGPU/literals.s b/llvm/test/MC/AMDGPU/literals.s
index 471d484a4df3b..4908872ede84a 100644
--- a/llvm/test/MC/AMDGPU/literals.s
+++ b/llvm/test/MC/AMDGPU/literals.s
@@ -237,8 +237,8 @@ v_cos_f16_e32 v5.l, lit(1.0)
// NOGFX89: :[[@LINE-4]]:1: error: operands are not valid for this GPU or mode
// NOSI: :[[@LINE-5]]:1: error: instruction not supported on this GPU (gfx600): v_cos_f16
-v_tanh_bf16 v5.l, 1.0
-// GFX1250: v_tanh_bf16_e32 v5.l, 1.0 ; encoding: [0xf2,0x94,0x0a,0x7e]
+v_tanh_bf16_e64 v5.l, 1.0 op_sel:[1,0]
+// GFX1250: v_tanh_bf16_e64 v5.l, 1.0 op_sel:[1,0] ; encoding: [0x05,0x08,0xca,0xd5,0xf2,0x00,0x01,0x02]
// NOCI: :[[@LINE-2]]:1: error: instruction not supported on this GPU (gfx704): v_tanh_bf16
// NOGFX11: :[[@LINE-3]]:1: error: instruction not supported on this GPU (gfx1100): v_tanh_bf16
// NOGFX12: :[[@LINE-4]]:1: error: instruction not supported on this GPU (gfx1200): v_tanh_bf16
@@ -713,8 +713,8 @@ v_cos_f16_e32 v5.l, lit(1)
// NOGFX89: :[[@LINE-4]]:1: error: operands are not valid for this GPU or mode
// NOSI: :[[@LINE-5]]:1: error: instruction not supported on this GPU (gfx600): v_cos_f16
-v_tanh_bf16 v5.l, 1
-// GFX1250: v_tanh_bf16_e32 v5.l, 1 ; encoding: [0x81,0x94,0x0a,0x7e]
+v_tanh_bf16_e64 v5.l, 1 op_sel:[1,0]
+// GFX1250: v_tanh_bf16_e64 v5.l, 1 op_sel:[1,0] ; encoding: [0x05,0x08,0xca,0xd5,0x81,0x00,0x01,0x02]
// NOCI: :[[@LINE-2]]:1: error: instruction not supported on this GPU (gfx704): v_tanh_bf16
// NOGFX11: :[[@LINE-3]]:1: error: instruction not supported on this GPU (gfx1100): v_tanh_bf16
// NOGFX12: :[[@LINE-4]]:1: error: instruction not supported on this GPU (gfx1200): v_tanh_bf16
More information about the llvm-commits
mailing list