[clang] [compiler-rt] [llvm] [X86] AVX10_V2_AUX Implementation (PR #206888)

via llvm-commits llvm-commits at lists.llvm.org
Mon Sep 21 10:54:00 PDT 2026


https://github.com/ganeshgit updated https://github.com/llvm/llvm-project/pull/206888

>From 09d35e0917a8efa42fc626606a2b2e262b9f97f5 Mon Sep 17 00:00:00 2001
From: Ganesh Gopalasubramanian <Ganesh.Gopalasubramanian at amd.com>
Date: Wed, 1 Jul 2026 10:01:13 +0530
Subject: [PATCH 01/16] AVX10_V2_AUX Implementation

 - AVX10_V2_AUX extends AVX10.2 with FP8/FP4/FP6 format conversions optimized
 for AI/ML inference workloads, enabling efficient low-precision arithmetic

 - Narrowing conversions from single-precision to FP8: VCVTPS2BF8, VCVTPS2HF8,
 VCVTPS2BF8S, VCVTPS2HF8S with optional saturation, biasing
 (VCVTBIASPS2BF8/HF8), and round-to-odd (VCVTROPS2HF8) variants

 - Expanding conversions from FP8 to single-precision: VCVTBF82PS, VCVTHF82PS
 that widen 8-bit FP formats with masking support

 - FP4/FP6 conversions: VCVTBF82BF4S, VCVTHF82BF4S for truncating,
 VCVTBF82BF6S, VCVTHF82HF6S for same-size, and expanding VCVTBF42HF8,
 VCVTBF62BF8, VCVTHF62HF8

 - Support includes instruction definitions, intrinsics, and Clang headers

 - Tests covering AT&T and Intel syntax assembly for both 32-bit and 64-bit
 modes, plus disassembler tests verifying round-trip encoding and decoding
---
 clang/include/clang/Basic/BuiltinsX86.td      |  247 +++
 clang/lib/Basic/Targets/X86.cpp               |    6 +
 clang/lib/Basic/Targets/X86.h                 |    1 +
 clang/lib/Headers/CMakeLists.txt              |    1 +
 clang/lib/Headers/avx10_2_v2auxintrin.h       | 1067 +++++++++
 clang/lib/Headers/cpuid.h                     |    3 +
 clang/lib/Headers/immintrin.h                 |    4 +
 .../test/CodeGen/X86/avx10_2_v2aux-builtins.c | 1056 +++++++++
 clang/test/CodeGen/attr-target-x86.c          |    4 +-
 llvm/include/llvm/IR/IntrinsicsX86.td         |  197 ++
 .../llvm/TargetParser/X86TargetParser.def     |    1 +
 llvm/lib/Target/X86/X86.td                    |    3 +
 llvm/lib/Target/X86/X86InstrAVX10_V2_AUX.td   |  588 +++++
 llvm/lib/Target/X86/X86InstrFragmentsSIMD.td  |   57 +
 llvm/lib/Target/X86/X86InstrInfo.td           |    3 +
 llvm/lib/Target/X86/X86InstrPredicates.td     |    1 +
 llvm/lib/Target/X86/X86IntrinsicsInfo.h       |   72 +
 llvm/lib/TargetParser/X86TargetParser.cpp     |    1 +
 .../CodeGen/X86/avx10_2_v2aux-intrinsics.ll   | 1931 +++++++++++++++++
 .../MC/Disassembler/X86/avx10_v2_aux-32.txt   |  460 ++++
 .../MC/Disassembler/X86/avx10_v2_aux-64.txt   |  460 ++++
 llvm/test/MC/X86/avx10_v2_aux-att-32.s        |  463 ++++
 llvm/test/MC/X86/avx10_v2_aux-att-64.s        |  687 ++++++
 llvm/test/MC/X86/avx10_v2_aux-intel-32.s      |  331 +++
 llvm/test/MC/X86/avx10_v2_aux-intel-64.s      |  687 ++++++
 llvm/test/TableGen/x86-fold-tables.inc        |  224 ++
 26 files changed, 8553 insertions(+), 2 deletions(-)
 create mode 100644 clang/lib/Headers/avx10_2_v2auxintrin.h
 create mode 100644 clang/test/CodeGen/X86/avx10_2_v2aux-builtins.c
 create mode 100644 llvm/lib/Target/X86/X86InstrAVX10_V2_AUX.td
 create mode 100644 llvm/test/CodeGen/X86/avx10_2_v2aux-intrinsics.ll
 create mode 100644 llvm/test/MC/Disassembler/X86/avx10_v2_aux-32.txt
 create mode 100644 llvm/test/MC/Disassembler/X86/avx10_v2_aux-64.txt
 create mode 100644 llvm/test/MC/X86/avx10_v2_aux-att-32.s
 create mode 100644 llvm/test/MC/X86/avx10_v2_aux-att-64.s
 create mode 100644 llvm/test/MC/X86/avx10_v2_aux-intel-32.s
 create mode 100644 llvm/test/MC/X86/avx10_v2_aux-intel-64.s

diff --git a/clang/include/clang/Basic/BuiltinsX86.td b/clang/include/clang/Basic/BuiltinsX86.td
index b0f95d98b84719..82fd8d80cc82dc 100644
--- a/clang/include/clang/Basic/BuiltinsX86.td
+++ b/clang/include/clang/Basic/BuiltinsX86.td
@@ -5045,3 +5045,250 @@ let Features = "avx10.2", Attributes = [NoThrow, Const, RequiredVectorWidth<256>
 let Features = "avx10.2", Attributes = [NoThrow, Const, RequiredVectorWidth<512>] in {
   def vgetmantbf16512_mask : X86Builtin<"_Vector<32, __bf16>(_Vector<32, __bf16>, _Constant int, _Vector<32, __bf16>, unsigned int)">;
 }
+
+// AVX10 V2 AUX - Convert instructions
+
+// Group A: PS(f32) -> i8 truncating conversions (quarter-size: output always v16i8)
+
+// VCVTPS2BF8
+let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
+  def vcvtps2bf8_128_mask : X86Builtin<"_Vector<16, char>(_Vector<4, float>, _Vector<16, char>, unsigned char)">;
+}
+
+let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
+  def vcvtps2bf8_256_mask : X86Builtin<"_Vector<16, char>(_Vector<8, float>, _Vector<16, char>, unsigned char)">;
+}
+
+let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
+  def vcvtps2bf8_512_mask : X86Builtin<"_Vector<16, char>(_Vector<16, float>, _Vector<16, char>, unsigned short)">;
+}
+
+// VCVTPS2BF8S
+let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
+  def vcvtps2bf8s_128_mask : X86Builtin<"_Vector<16, char>(_Vector<4, float>, _Vector<16, char>, unsigned char)">;
+}
+
+let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
+  def vcvtps2bf8s_256_mask : X86Builtin<"_Vector<16, char>(_Vector<8, float>, _Vector<16, char>, unsigned char)">;
+}
+
+let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
+  def vcvtps2bf8s_512_mask : X86Builtin<"_Vector<16, char>(_Vector<16, float>, _Vector<16, char>, unsigned short)">;
+}
+
+// VCVTPS2HF8
+let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
+  def vcvtps2hf8_128_mask : X86Builtin<"_Vector<16, char>(_Vector<4, float>, _Vector<16, char>, unsigned char)">;
+}
+
+let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
+  def vcvtps2hf8_256_mask : X86Builtin<"_Vector<16, char>(_Vector<8, float>, _Vector<16, char>, unsigned char)">;
+}
+
+let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
+  def vcvtps2hf8_512_mask : X86Builtin<"_Vector<16, char>(_Vector<16, float>, _Vector<16, char>, unsigned short)">;
+}
+
+// VCVTPS2HF8S
+let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
+  def vcvtps2hf8s_128_mask : X86Builtin<"_Vector<16, char>(_Vector<4, float>, _Vector<16, char>, unsigned char)">;
+}
+
+let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
+  def vcvtps2hf8s_256_mask : X86Builtin<"_Vector<16, char>(_Vector<8, float>, _Vector<16, char>, unsigned char)">;
+}
+
+let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
+  def vcvtps2hf8s_512_mask : X86Builtin<"_Vector<16, char>(_Vector<16, float>, _Vector<16, char>, unsigned short)">;
+}
+
+// VCVTROPS2HF8
+let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
+  def vcvtrops2hf8_128_mask : X86Builtin<"_Vector<16, char>(_Vector<4, float>, _Vector<16, char>, unsigned char)">;
+}
+
+let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
+  def vcvtrops2hf8_256_mask : X86Builtin<"_Vector<16, char>(_Vector<8, float>, _Vector<16, char>, unsigned char)">;
+}
+
+let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
+  def vcvtrops2hf8_512_mask : X86Builtin<"_Vector<16, char>(_Vector<16, float>, _Vector<16, char>, unsigned short)">;
+}
+
+// VCVTROPS2HF8S
+let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
+  def vcvtrops2hf8s_128_mask : X86Builtin<"_Vector<16, char>(_Vector<4, float>, _Vector<16, char>, unsigned char)">;
+}
+
+let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
+  def vcvtrops2hf8s_256_mask : X86Builtin<"_Vector<16, char>(_Vector<8, float>, _Vector<16, char>, unsigned char)">;
+}
+
+let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
+  def vcvtrops2hf8s_512_mask : X86Builtin<"_Vector<16, char>(_Vector<16, float>, _Vector<16, char>, unsigned short)">;
+}
+
+// Group B: Bias PS -> i8 conversions (3-operand: bias + f32 source -> i8 dest)
+
+// VCVTBIASPS2BF8
+let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
+  def vcvtbiasps2bf8_128_mask : X86Builtin<"_Vector<16, char>(_Vector<16, char>, _Vector<4, float>, _Vector<16, char>, unsigned char)">;
+}
+
+let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
+  def vcvtbiasps2bf8_256_mask : X86Builtin<"_Vector<16, char>(_Vector<32, char>, _Vector<8, float>, _Vector<16, char>, unsigned char)">;
+}
+
+let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
+  def vcvtbiasps2bf8_512_mask : X86Builtin<"_Vector<16, char>(_Vector<64, char>, _Vector<16, float>, _Vector<16, char>, unsigned short)">;
+}
+
+// VCVTBIASPS2BF8S
+let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
+  def vcvtbiasps2bf8s_128_mask : X86Builtin<"_Vector<16, char>(_Vector<16, char>, _Vector<4, float>, _Vector<16, char>, unsigned char)">;
+}
+
+let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
+  def vcvtbiasps2bf8s_256_mask : X86Builtin<"_Vector<16, char>(_Vector<32, char>, _Vector<8, float>, _Vector<16, char>, unsigned char)">;
+}
+
+let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
+  def vcvtbiasps2bf8s_512_mask : X86Builtin<"_Vector<16, char>(_Vector<64, char>, _Vector<16, float>, _Vector<16, char>, unsigned short)">;
+}
+
+// VCVTBIASPS2HF8
+let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
+  def vcvtbiasps2hf8_128_mask : X86Builtin<"_Vector<16, char>(_Vector<16, char>, _Vector<4, float>, _Vector<16, char>, unsigned char)">;
+}
+
+let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
+  def vcvtbiasps2hf8_256_mask : X86Builtin<"_Vector<16, char>(_Vector<32, char>, _Vector<8, float>, _Vector<16, char>, unsigned char)">;
+}
+
+let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
+  def vcvtbiasps2hf8_512_mask : X86Builtin<"_Vector<16, char>(_Vector<64, char>, _Vector<16, float>, _Vector<16, char>, unsigned short)">;
+}
+
+// VCVTBIASPS2HF8S
+let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
+  def vcvtbiasps2hf8s_128_mask : X86Builtin<"_Vector<16, char>(_Vector<16, char>, _Vector<4, float>, _Vector<16, char>, unsigned char)">;
+}
+
+let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
+  def vcvtbiasps2hf8s_256_mask : X86Builtin<"_Vector<16, char>(_Vector<32, char>, _Vector<8, float>, _Vector<16, char>, unsigned char)">;
+}
+
+let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
+  def vcvtbiasps2hf8s_512_mask : X86Builtin<"_Vector<16, char>(_Vector<64, char>, _Vector<16, float>, _Vector<16, char>, unsigned short)">;
+}
+
+// Group C: 8bit -> PS expanding conversions
+
+// VCVTBF82PS
+let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
+  def vcvtbf8_2ps128_mask : X86Builtin<"_Vector<4, float>(_Vector<16, char>, _Vector<4, float>, unsigned char)">;
+}
+
+let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
+  def vcvtbf8_2ps256_mask : X86Builtin<"_Vector<8, float>(_Vector<16, char>, _Vector<8, float>, unsigned char)">;
+}
+
+let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
+  def vcvtbf8_2ps512_mask : X86Builtin<"_Vector<16, float>(_Vector<16, char>, _Vector<16, float>, unsigned short)">;
+}
+
+// VCVTHF82PS
+let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
+  def vcvthf8_2ps128_mask : X86Builtin<"_Vector<4, float>(_Vector<16, char>, _Vector<4, float>, unsigned char)">;
+}
+
+let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
+  def vcvthf8_2ps256_mask : X86Builtin<"_Vector<8, float>(_Vector<16, char>, _Vector<8, float>, unsigned char)">;
+}
+
+let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
+  def vcvthf8_2ps512_mask : X86Builtin<"_Vector<16, float>(_Vector<16, char>, _Vector<16, float>, unsigned short)">;
+}
+
+// Group E: Same-size reg-only conversions (no masking)
+
+// VCVTBF82BF6S
+let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
+  def vcvtbf82bf6s128 : X86Builtin<"_Vector<16, char>(_Vector<16, char>)">;
+}
+
+let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
+  def vcvtbf82bf6s256 : X86Builtin<"_Vector<32, char>(_Vector<32, char>)">;
+}
+
+let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
+  def vcvtbf82bf6s512 : X86Builtin<"_Vector<64, char>(_Vector<64, char>)">;
+}
+
+// VCVTHF82HF6S
+let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
+  def vcvthf82hf6s128 : X86Builtin<"_Vector<16, char>(_Vector<16, char>)">;
+}
+
+let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
+  def vcvthf82hf6s256 : X86Builtin<"_Vector<32, char>(_Vector<32, char>)">;
+}
+
+let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
+  def vcvthf82hf6s512 : X86Builtin<"_Vector<64, char>(_Vector<64, char>)">;
+}
+
+// Group F: Expanding/same-size conversions (no masking in intrinsic; use selectb for masking)
+
+// VCVTBF42HF8
+let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
+  def vcvtbf42hf8128 : X86Builtin<"_Vector<16, char>(_Vector<16, char>)">;
+}
+
+let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
+  def vcvtbf42hf8256 : X86Builtin<"_Vector<32, char>(_Vector<16, char>)">;
+}
+
+let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
+  def vcvtbf42hf8512 : X86Builtin<"_Vector<64, char>(_Vector<32, char>)">;
+}
+
+// VCVTBF62HF8
+let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
+  def vcvtbf62hf8128 : X86Builtin<"_Vector<16, char>(_Vector<16, char>)">;
+}
+
+let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
+  def vcvtbf62hf8256 : X86Builtin<"_Vector<32, char>(_Vector<32, char>)">;
+}
+
+let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
+  def vcvtbf62hf8512 : X86Builtin<"_Vector<64, char>(_Vector<64, char>)">;
+}
+
+// VCVTHF62HF8
+let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
+  def vcvthf62hf8128 : X86Builtin<"_Vector<16, char>(_Vector<16, char>)">;
+}
+
+let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
+  def vcvthf62hf8256 : X86Builtin<"_Vector<32, char>(_Vector<32, char>)">;
+}
+
+let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
+  def vcvthf62hf8512 : X86Builtin<"_Vector<64, char>(_Vector<64, char>)">;
+}
+
+// Group H: VUNPACKB - Byte unpack with immediate (no masking in intrinsic)
+
+let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
+  def vunpackb128 : X86Builtin<"_Vector<16, char>(_Vector<16, char>, _Constant unsigned char)">;
+}
+
+let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
+  def vunpackb256 : X86Builtin<"_Vector<32, char>(_Vector<32, char>, _Constant unsigned char)">;
+}
+
+let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
+  def vunpackb512 : X86Builtin<"_Vector<64, char>(_Vector<64, char>, _Constant unsigned char)">;
+}
diff --git a/clang/lib/Basic/Targets/X86.cpp b/clang/lib/Basic/Targets/X86.cpp
index 2943604f8c8ff4..9482e179dee5c4 100644
--- a/clang/lib/Basic/Targets/X86.cpp
+++ b/clang/lib/Basic/Targets/X86.cpp
@@ -281,6 +281,8 @@ bool X86TargetInfo::handleTargetFeatures(std::vector<std::string> &Features,
     } else if (Feature == "+avx10.2") {
       HasAVX10_2 = true;
       HasFullBFloat16 = true;
+    } else if (Feature == "+avx10-v2-aux") {
+      HasAVX10_V2_AUX = true;
     } else if (Feature == "+avx512cd") {
       HasAVX512CD = true;
     } else if (Feature == "+avx512vpopcntdq") {
@@ -836,6 +838,8 @@ void X86TargetInfo::getTargetDefines(const LangOptions &Opts,
     Builder.defineMacro("__AVX10_2__");
     Builder.defineMacro("__AVX10_2_512__");
   }
+  if (HasAVX10_V2_AUX)
+    Builder.defineMacro("__AVX10_V2_AUX__");
   if (HasAVX512CD)
     Builder.defineMacro("__AVX512CD__");
   if (HasAVX512VPOPCNTDQ)
@@ -1087,6 +1091,7 @@ bool X86TargetInfo::isValidFeatureName(StringRef Name) const {
       .Case("avx", true)
       .Case("avx10.1", true)
       .Case("avx10.2", true)
+      .Case("avx10-v2-aux", true)
       .Case("avx2", true)
       .Case("avx512f", true)
       .Case("avx512cd", true)
@@ -1208,6 +1213,7 @@ bool X86TargetInfo::hasFeature(StringRef Feature) const {
       .Case("avx", SSELevel >= AVX)
       .Case("avx10.1", HasAVX10_1)
       .Case("avx10.2", HasAVX10_2)
+      .Case("avx10-v2-aux", HasAVX10_V2_AUX)
       .Case("avx2", SSELevel >= AVX2)
       .Case("avx512f", SSELevel >= AVX512F)
       .Case("avx512cd", HasAVX512CD)
diff --git a/clang/lib/Basic/Targets/X86.h b/clang/lib/Basic/Targets/X86.h
index f9c39b31f5e089..02f7e8c5807ca4 100644
--- a/clang/lib/Basic/Targets/X86.h
+++ b/clang/lib/Basic/Targets/X86.h
@@ -98,6 +98,7 @@ class LLVM_LIBRARY_VISIBILITY X86TargetInfo : public TargetInfo {
   bool HasF16C = false;
   bool HasAVX10_1 = false;
   bool HasAVX10_2 = false;
+  bool HasAVX10_V2_AUX = false;
   bool HasAVX512CD = false;
   bool HasAVX512VPOPCNTDQ = false;
   bool HasAVX512VNNI = false;
diff --git a/clang/lib/Headers/CMakeLists.txt b/clang/lib/Headers/CMakeLists.txt
index 439f2725168ba1..b4891961a45986 100644
--- a/clang/lib/Headers/CMakeLists.txt
+++ b/clang/lib/Headers/CMakeLists.txt
@@ -183,6 +183,7 @@ set(x86_files
   avx10_2_512niintrin.h
   avx10_2_512satcvtdsintrin.h
   avx10_2_512satcvtintrin.h
+  avx10_2_v2auxintrin.h
   avx10_2bf16intrin.h
   avx10_2convertintrin.h
   avx10_2copyintrin.h
diff --git a/clang/lib/Headers/avx10_2_v2auxintrin.h b/clang/lib/Headers/avx10_2_v2auxintrin.h
new file mode 100644
index 00000000000000..5e54085c104c52
--- /dev/null
+++ b/clang/lib/Headers/avx10_2_v2auxintrin.h
@@ -0,0 +1,1067 @@
+/*===------------ avx10_2_v2auxintrin.h - AVX10_2_V2AUX -------------------===
+ *
+ * Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
+ * See https://llvm.org/LICENSE.txt for license information.
+ * SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
+ *
+ *===-----------------------------------------------------------------------===
+ */
+#ifndef __IMMINTRIN_H
+#error                                                                         \
+    "Never use <avx10_2_v2auxintrin.h> directly; include <immintrin.h> instead."
+#endif // __IMMINTRIN_H
+
+#ifdef __SSE2__
+
+#ifndef __AVX10_2_V2AUXINTRIN_H
+#define __AVX10_2_V2AUXINTRIN_H
+
+/* Define the default attributes for the functions in this file. */
+#define __DEFAULT_FN_ATTRS128                                                  \
+  __attribute__((__always_inline__, __nodebug__, __target__("avx10-v2-aux"),   \
+                 __min_vector_width__(128)))
+#define __DEFAULT_FN_ATTRS256                                                  \
+  __attribute__((__always_inline__, __nodebug__, __target__("avx10-v2-aux"),   \
+                 __min_vector_width__(256)))
+#define __DEFAULT_FN_ATTRS512                                                  \
+  __attribute__((__always_inline__, __nodebug__, __target__("avx10-v2-aux"),   \
+                 __min_vector_width__(512)))
+
+// clang-format off
+
+//===----------------------------------------------------------------------===//
+// Group A: VCVTPS2BF8 / VCVTPS2BF8S / VCVTPS2HF8 / VCVTPS2HF8S /
+//          VCVTROPS2HF8 / VCVTROPS2HF8S
+// Convert packed single-precision to FP8. Output is always __m128i.
+//===----------------------------------------------------------------------===//
+
+// VCVTPS2BF8 - 128-bit
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS128
+_mm_cvtps_bf8(__m128 __A) {
+  return (__m128i)__builtin_ia32_vcvtps2bf8_128_mask(
+      (__v4sf)__A, (__v16qi)_mm_undefined_si128(), (__mmask8)-1);
+}
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS128
+_mm_mask_cvtps_bf8(__m128i __W, __mmask8 __U, __m128 __A) {
+  return (__m128i)__builtin_ia32_vcvtps2bf8_128_mask(
+      (__v4sf)__A, (__v16qi)(__m128i)__W, (__mmask8)__U);
+}
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS128
+_mm_maskz_cvtps_bf8(__mmask8 __U, __m128 __A) {
+  return (__m128i)__builtin_ia32_vcvtps2bf8_128_mask(
+      (__v4sf)__A, (__v16qi)(__m128i)_mm_setzero_si128(), (__mmask8)__U);
+}
+
+// VCVTPS2BF8 - 256-bit
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS256
+_mm256_cvtps_bf8(__m256 __A) {
+  return (__m128i)__builtin_ia32_vcvtps2bf8_256_mask(
+      (__v8sf)__A, (__v16qi)_mm_undefined_si128(), (__mmask8)-1);
+}
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS256
+_mm256_mask_cvtps_bf8(__m128i __W, __mmask8 __U, __m256 __A) {
+  return (__m128i)__builtin_ia32_vcvtps2bf8_256_mask(
+      (__v8sf)__A, (__v16qi)(__m128i)__W, (__mmask8)__U);
+}
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS256
+_mm256_maskz_cvtps_bf8(__mmask8 __U, __m256 __A) {
+  return (__m128i)__builtin_ia32_vcvtps2bf8_256_mask(
+      (__v8sf)__A, (__v16qi)(__m128i)_mm_setzero_si128(), (__mmask8)__U);
+}
+
+// VCVTPS2BF8 - 512-bit
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS512
+_mm512_cvtps_bf8(__m512 __A) {
+  return (__m128i)__builtin_ia32_vcvtps2bf8_512_mask(
+      (__v16sf)__A, (__v16qi)_mm_undefined_si128(), (__mmask16)-1);
+}
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS512
+_mm512_mask_cvtps_bf8(__m128i __W, __mmask16 __U, __m512 __A) {
+  return (__m128i)__builtin_ia32_vcvtps2bf8_512_mask(
+      (__v16sf)__A, (__v16qi)(__m128i)__W, (__mmask16)__U);
+}
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS512
+_mm512_maskz_cvtps_bf8(__mmask16 __U, __m512 __A) {
+  return (__m128i)__builtin_ia32_vcvtps2bf8_512_mask(
+      (__v16sf)__A, (__v16qi)(__m128i)_mm_setzero_si128(), (__mmask16)__U);
+}
+
+// VCVTPS2BF8S - 128-bit
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS128
+_mm_cvts_ps_bf8(__m128 __A) {
+  return (__m128i)__builtin_ia32_vcvtps2bf8s_128_mask(
+      (__v4sf)__A, (__v16qi)_mm_undefined_si128(), (__mmask8)-1);
+}
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS128
+_mm_mask_cvts_ps_bf8(__m128i __W, __mmask8 __U, __m128 __A) {
+  return (__m128i)__builtin_ia32_vcvtps2bf8s_128_mask(
+      (__v4sf)__A, (__v16qi)(__m128i)__W, (__mmask8)__U);
+}
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS128
+_mm_maskz_cvts_ps_bf8(__mmask8 __U, __m128 __A) {
+  return (__m128i)__builtin_ia32_vcvtps2bf8s_128_mask(
+      (__v4sf)__A, (__v16qi)(__m128i)_mm_setzero_si128(), (__mmask8)__U);
+}
+
+// VCVTPS2BF8S - 256-bit
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS256
+_mm256_cvts_ps_bf8(__m256 __A) {
+  return (__m128i)__builtin_ia32_vcvtps2bf8s_256_mask(
+      (__v8sf)__A, (__v16qi)_mm_undefined_si128(), (__mmask8)-1);
+}
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS256
+_mm256_mask_cvts_ps_bf8(__m128i __W, __mmask8 __U, __m256 __A) {
+  return (__m128i)__builtin_ia32_vcvtps2bf8s_256_mask(
+      (__v8sf)__A, (__v16qi)(__m128i)__W, (__mmask8)__U);
+}
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS256
+_mm256_maskz_cvts_ps_bf8(__mmask8 __U, __m256 __A) {
+  return (__m128i)__builtin_ia32_vcvtps2bf8s_256_mask(
+      (__v8sf)__A, (__v16qi)(__m128i)_mm_setzero_si128(), (__mmask8)__U);
+}
+
+// VCVTPS2BF8S - 512-bit
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS512
+_mm512_cvts_ps_bf8(__m512 __A) {
+  return (__m128i)__builtin_ia32_vcvtps2bf8s_512_mask(
+      (__v16sf)__A, (__v16qi)_mm_undefined_si128(), (__mmask16)-1);
+}
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS512
+_mm512_mask_cvts_ps_bf8(__m128i __W, __mmask16 __U, __m512 __A) {
+  return (__m128i)__builtin_ia32_vcvtps2bf8s_512_mask(
+      (__v16sf)__A, (__v16qi)(__m128i)__W, (__mmask16)__U);
+}
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS512
+_mm512_maskz_cvts_ps_bf8(__mmask16 __U, __m512 __A) {
+  return (__m128i)__builtin_ia32_vcvtps2bf8s_512_mask(
+      (__v16sf)__A, (__v16qi)(__m128i)_mm_setzero_si128(), (__mmask16)__U);
+}
+
+// VCVTPS2HF8 - 128-bit
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS128
+_mm_cvtps_hf8(__m128 __A) {
+  return (__m128i)__builtin_ia32_vcvtps2hf8_128_mask(
+      (__v4sf)__A, (__v16qi)_mm_undefined_si128(), (__mmask8)-1);
+}
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS128
+_mm_mask_cvtps_hf8(__m128i __W, __mmask8 __U, __m128 __A) {
+  return (__m128i)__builtin_ia32_vcvtps2hf8_128_mask(
+      (__v4sf)__A, (__v16qi)(__m128i)__W, (__mmask8)__U);
+}
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS128
+_mm_maskz_cvtps_hf8(__mmask8 __U, __m128 __A) {
+  return (__m128i)__builtin_ia32_vcvtps2hf8_128_mask(
+      (__v4sf)__A, (__v16qi)(__m128i)_mm_setzero_si128(), (__mmask8)__U);
+}
+
+// VCVTPS2HF8 - 256-bit
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS256
+_mm256_cvtps_hf8(__m256 __A) {
+  return (__m128i)__builtin_ia32_vcvtps2hf8_256_mask(
+      (__v8sf)__A, (__v16qi)_mm_undefined_si128(), (__mmask8)-1);
+}
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS256
+_mm256_mask_cvtps_hf8(__m128i __W, __mmask8 __U, __m256 __A) {
+  return (__m128i)__builtin_ia32_vcvtps2hf8_256_mask(
+      (__v8sf)__A, (__v16qi)(__m128i)__W, (__mmask8)__U);
+}
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS256
+_mm256_maskz_cvtps_hf8(__mmask8 __U, __m256 __A) {
+  return (__m128i)__builtin_ia32_vcvtps2hf8_256_mask(
+      (__v8sf)__A, (__v16qi)(__m128i)_mm_setzero_si128(), (__mmask8)__U);
+}
+
+// VCVTPS2HF8 - 512-bit
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS512
+_mm512_cvtps_hf8(__m512 __A) {
+  return (__m128i)__builtin_ia32_vcvtps2hf8_512_mask(
+      (__v16sf)__A, (__v16qi)_mm_undefined_si128(), (__mmask16)-1);
+}
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS512
+_mm512_mask_cvtps_hf8(__m128i __W, __mmask16 __U, __m512 __A) {
+  return (__m128i)__builtin_ia32_vcvtps2hf8_512_mask(
+      (__v16sf)__A, (__v16qi)(__m128i)__W, (__mmask16)__U);
+}
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS512
+_mm512_maskz_cvtps_hf8(__mmask16 __U, __m512 __A) {
+  return (__m128i)__builtin_ia32_vcvtps2hf8_512_mask(
+      (__v16sf)__A, (__v16qi)(__m128i)_mm_setzero_si128(), (__mmask16)__U);
+}
+
+// VCVTPS2HF8S - 128-bit
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS128
+_mm_cvts_ps_hf8(__m128 __A) {
+  return (__m128i)__builtin_ia32_vcvtps2hf8s_128_mask(
+      (__v4sf)__A, (__v16qi)_mm_undefined_si128(), (__mmask8)-1);
+}
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS128
+_mm_mask_cvts_ps_hf8(__m128i __W, __mmask8 __U, __m128 __A) {
+  return (__m128i)__builtin_ia32_vcvtps2hf8s_128_mask(
+      (__v4sf)__A, (__v16qi)(__m128i)__W, (__mmask8)__U);
+}
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS128
+_mm_maskz_cvts_ps_hf8(__mmask8 __U, __m128 __A) {
+  return (__m128i)__builtin_ia32_vcvtps2hf8s_128_mask(
+      (__v4sf)__A, (__v16qi)(__m128i)_mm_setzero_si128(), (__mmask8)__U);
+}
+
+// VCVTPS2HF8S - 256-bit
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS256
+_mm256_cvts_ps_hf8(__m256 __A) {
+  return (__m128i)__builtin_ia32_vcvtps2hf8s_256_mask(
+      (__v8sf)__A, (__v16qi)_mm_undefined_si128(), (__mmask8)-1);
+}
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS256
+_mm256_mask_cvts_ps_hf8(__m128i __W, __mmask8 __U, __m256 __A) {
+  return (__m128i)__builtin_ia32_vcvtps2hf8s_256_mask(
+      (__v8sf)__A, (__v16qi)(__m128i)__W, (__mmask8)__U);
+}
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS256
+_mm256_maskz_cvts_ps_hf8(__mmask8 __U, __m256 __A) {
+  return (__m128i)__builtin_ia32_vcvtps2hf8s_256_mask(
+      (__v8sf)__A, (__v16qi)(__m128i)_mm_setzero_si128(), (__mmask8)__U);
+}
+
+// VCVTPS2HF8S - 512-bit
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS512
+_mm512_cvts_ps_hf8(__m512 __A) {
+  return (__m128i)__builtin_ia32_vcvtps2hf8s_512_mask(
+      (__v16sf)__A, (__v16qi)_mm_undefined_si128(), (__mmask16)-1);
+}
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS512
+_mm512_mask_cvts_ps_hf8(__m128i __W, __mmask16 __U, __m512 __A) {
+  return (__m128i)__builtin_ia32_vcvtps2hf8s_512_mask(
+      (__v16sf)__A, (__v16qi)(__m128i)__W, (__mmask16)__U);
+}
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS512
+_mm512_maskz_cvts_ps_hf8(__mmask16 __U, __m512 __A) {
+  return (__m128i)__builtin_ia32_vcvtps2hf8s_512_mask(
+      (__v16sf)__A, (__v16qi)(__m128i)_mm_setzero_si128(), (__mmask16)__U);
+}
+
+// VCVTROPS2HF8 - 128-bit
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS128
+_mm_cvtrops_hf8(__m128 __A) {
+  return (__m128i)__builtin_ia32_vcvtrops2hf8_128_mask(
+      (__v4sf)__A, (__v16qi)_mm_undefined_si128(), (__mmask8)-1);
+}
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS128
+_mm_mask_cvtrops_hf8(__m128i __W, __mmask8 __U, __m128 __A) {
+  return (__m128i)__builtin_ia32_vcvtrops2hf8_128_mask(
+      (__v4sf)__A, (__v16qi)(__m128i)__W, (__mmask8)__U);
+}
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS128
+_mm_maskz_cvtrops_hf8(__mmask8 __U, __m128 __A) {
+  return (__m128i)__builtin_ia32_vcvtrops2hf8_128_mask(
+      (__v4sf)__A, (__v16qi)(__m128i)_mm_setzero_si128(), (__mmask8)__U);
+}
+
+// VCVTROPS2HF8 - 256-bit
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS256
+_mm256_cvtrops_hf8(__m256 __A) {
+  return (__m128i)__builtin_ia32_vcvtrops2hf8_256_mask(
+      (__v8sf)__A, (__v16qi)_mm_undefined_si128(), (__mmask8)-1);
+}
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS256
+_mm256_mask_cvtrops_hf8(__m128i __W, __mmask8 __U, __m256 __A) {
+  return (__m128i)__builtin_ia32_vcvtrops2hf8_256_mask(
+      (__v8sf)__A, (__v16qi)(__m128i)__W, (__mmask8)__U);
+}
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS256
+_mm256_maskz_cvtrops_hf8(__mmask8 __U, __m256 __A) {
+  return (__m128i)__builtin_ia32_vcvtrops2hf8_256_mask(
+      (__v8sf)__A, (__v16qi)(__m128i)_mm_setzero_si128(), (__mmask8)__U);
+}
+
+// VCVTROPS2HF8 - 512-bit
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS512
+_mm512_cvtrops_hf8(__m512 __A) {
+  return (__m128i)__builtin_ia32_vcvtrops2hf8_512_mask(
+      (__v16sf)__A, (__v16qi)_mm_undefined_si128(), (__mmask16)-1);
+}
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS512
+_mm512_mask_cvtrops_hf8(__m128i __W, __mmask16 __U, __m512 __A) {
+  return (__m128i)__builtin_ia32_vcvtrops2hf8_512_mask(
+      (__v16sf)__A, (__v16qi)(__m128i)__W, (__mmask16)__U);
+}
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS512
+_mm512_maskz_cvtrops_hf8(__mmask16 __U, __m512 __A) {
+  return (__m128i)__builtin_ia32_vcvtrops2hf8_512_mask(
+      (__v16sf)__A, (__v16qi)(__m128i)_mm_setzero_si128(), (__mmask16)__U);
+}
+
+// VCVTROPS2HF8S - 128-bit
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS128
+_mm_cvts_rops_hf8(__m128 __A) {
+  return (__m128i)__builtin_ia32_vcvtrops2hf8s_128_mask(
+      (__v4sf)__A, (__v16qi)_mm_undefined_si128(), (__mmask8)-1);
+}
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS128
+_mm_mask_cvts_rops_hf8(__m128i __W, __mmask8 __U, __m128 __A) {
+  return (__m128i)__builtin_ia32_vcvtrops2hf8s_128_mask(
+      (__v4sf)__A, (__v16qi)(__m128i)__W, (__mmask8)__U);
+}
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS128
+_mm_maskz_cvts_rops_hf8(__mmask8 __U, __m128 __A) {
+  return (__m128i)__builtin_ia32_vcvtrops2hf8s_128_mask(
+      (__v4sf)__A, (__v16qi)(__m128i)_mm_setzero_si128(), (__mmask8)__U);
+}
+
+// VCVTROPS2HF8S - 256-bit
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS256
+_mm256_cvts_rops_hf8(__m256 __A) {
+  return (__m128i)__builtin_ia32_vcvtrops2hf8s_256_mask(
+      (__v8sf)__A, (__v16qi)_mm_undefined_si128(), (__mmask8)-1);
+}
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS256
+_mm256_mask_cvts_rops_hf8(__m128i __W, __mmask8 __U, __m256 __A) {
+  return (__m128i)__builtin_ia32_vcvtrops2hf8s_256_mask(
+      (__v8sf)__A, (__v16qi)(__m128i)__W, (__mmask8)__U);
+}
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS256
+_mm256_maskz_cvts_rops_hf8(__mmask8 __U, __m256 __A) {
+  return (__m128i)__builtin_ia32_vcvtrops2hf8s_256_mask(
+      (__v8sf)__A, (__v16qi)(__m128i)_mm_setzero_si128(), (__mmask8)__U);
+}
+
+// VCVTROPS2HF8S - 512-bit
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS512
+_mm512_cvts_rops_hf8(__m512 __A) {
+  return (__m128i)__builtin_ia32_vcvtrops2hf8s_512_mask(
+      (__v16sf)__A, (__v16qi)_mm_undefined_si128(), (__mmask16)-1);
+}
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS512
+_mm512_mask_cvts_rops_hf8(__m128i __W, __mmask16 __U, __m512 __A) {
+  return (__m128i)__builtin_ia32_vcvtrops2hf8s_512_mask(
+      (__v16sf)__A, (__v16qi)(__m128i)__W, (__mmask16)__U);
+}
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS512
+_mm512_maskz_cvts_rops_hf8(__mmask16 __U, __m512 __A) {
+  return (__m128i)__builtin_ia32_vcvtrops2hf8s_512_mask(
+      (__v16sf)__A, (__v16qi)(__m128i)_mm_setzero_si128(), (__mmask16)__U);
+}
+
+//===----------------------------------------------------------------------===//
+// Group B: VCVTBIASPS2BF8 / VCVTBIASPS2BF8S / VCVTBIASPS2HF8 /
+//          VCVTBIASPS2HF8S
+// Convert packed single-precision with bias to FP8. Output is always __m128i.
+//===----------------------------------------------------------------------===//
+
+// VCVTBIASPS2BF8 - 128-bit
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS128
+_mm_cvtbiasps_bf8(__m128i __A, __m128 __B) {
+  return (__m128i)__builtin_ia32_vcvtbiasps2bf8_128_mask(
+      (__v16qi)__A, (__v4sf)__B, (__v16qi)_mm_undefined_si128(), (__mmask8)-1);
+}
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS128
+_mm_mask_cvtbiasps_bf8(__m128i __W, __mmask8 __U, __m128i __A, __m128 __B) {
+  return (__m128i)__builtin_ia32_vcvtbiasps2bf8_128_mask(
+      (__v16qi)__A, (__v4sf)__B, (__v16qi)(__m128i)__W, (__mmask8)__U);
+}
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS128
+_mm_maskz_cvtbiasps_bf8(__mmask8 __U, __m128i __A, __m128 __B) {
+  return (__m128i)__builtin_ia32_vcvtbiasps2bf8_128_mask(
+      (__v16qi)__A, (__v4sf)__B, (__v16qi)(__m128i)_mm_setzero_si128(),
+      (__mmask8)__U);
+}
+
+// VCVTBIASPS2BF8 - 256-bit
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS256
+_mm256_cvtbiasps_bf8(__m256i __A, __m256 __B) {
+  return (__m128i)__builtin_ia32_vcvtbiasps2bf8_256_mask(
+      (__v32qi)__A, (__v8sf)__B, (__v16qi)_mm_undefined_si128(), (__mmask8)-1);
+}
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS256
+_mm256_mask_cvtbiasps_bf8(__m128i __W, __mmask8 __U, __m256i __A, __m256 __B) {
+  return (__m128i)__builtin_ia32_vcvtbiasps2bf8_256_mask(
+      (__v32qi)__A, (__v8sf)__B, (__v16qi)(__m128i)__W, (__mmask8)__U);
+}
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS256
+_mm256_maskz_cvtbiasps_bf8(__mmask8 __U, __m256i __A, __m256 __B) {
+  return (__m128i)__builtin_ia32_vcvtbiasps2bf8_256_mask(
+      (__v32qi)__A, (__v8sf)__B, (__v16qi)(__m128i)_mm_setzero_si128(),
+      (__mmask8)__U);
+}
+
+// VCVTBIASPS2BF8 - 512-bit
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS512
+_mm512_cvtbiasps_bf8(__m512i __A, __m512 __B) {
+  return (__m128i)__builtin_ia32_vcvtbiasps2bf8_512_mask(
+      (__v64qi)__A, (__v16sf)__B, (__v16qi)_mm_undefined_si128(),
+      (__mmask16)-1);
+}
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS512
+_mm512_mask_cvtbiasps_bf8(__m128i __W, __mmask16 __U, __m512i __A,
+                          __m512 __B) {
+  return (__m128i)__builtin_ia32_vcvtbiasps2bf8_512_mask(
+      (__v64qi)__A, (__v16sf)__B, (__v16qi)(__m128i)__W, (__mmask16)__U);
+}
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS512
+_mm512_maskz_cvtbiasps_bf8(__mmask16 __U, __m512i __A, __m512 __B) {
+  return (__m128i)__builtin_ia32_vcvtbiasps2bf8_512_mask(
+      (__v64qi)__A, (__v16sf)__B, (__v16qi)(__m128i)_mm_setzero_si128(),
+      (__mmask16)__U);
+}
+
+// VCVTBIASPS2BF8S - 128-bit
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS128
+_mm_cvts_biasps_bf8(__m128i __A, __m128 __B) {
+  return (__m128i)__builtin_ia32_vcvtbiasps2bf8s_128_mask(
+      (__v16qi)__A, (__v4sf)__B, (__v16qi)_mm_undefined_si128(), (__mmask8)-1);
+}
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS128
+_mm_mask_cvts_biasps_bf8(__m128i __W, __mmask8 __U, __m128i __A, __m128 __B) {
+  return (__m128i)__builtin_ia32_vcvtbiasps2bf8s_128_mask(
+      (__v16qi)__A, (__v4sf)__B, (__v16qi)(__m128i)__W, (__mmask8)__U);
+}
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS128
+_mm_maskz_cvts_biasps_bf8(__mmask8 __U, __m128i __A, __m128 __B) {
+  return (__m128i)__builtin_ia32_vcvtbiasps2bf8s_128_mask(
+      (__v16qi)__A, (__v4sf)__B, (__v16qi)(__m128i)_mm_setzero_si128(),
+      (__mmask8)__U);
+}
+
+// VCVTBIASPS2BF8S - 256-bit
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS256
+_mm256_cvts_biasps_bf8(__m256i __A, __m256 __B) {
+  return (__m128i)__builtin_ia32_vcvtbiasps2bf8s_256_mask(
+      (__v32qi)__A, (__v8sf)__B, (__v16qi)_mm_undefined_si128(), (__mmask8)-1);
+}
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS256 _mm256_mask_cvts_biasps_bf8(
+    __m128i __W, __mmask8 __U, __m256i __A, __m256 __B) {
+  return (__m128i)__builtin_ia32_vcvtbiasps2bf8s_256_mask(
+      (__v32qi)__A, (__v8sf)__B, (__v16qi)(__m128i)__W, (__mmask8)__U);
+}
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS256
+_mm256_maskz_cvts_biasps_bf8(__mmask8 __U, __m256i __A, __m256 __B) {
+  return (__m128i)__builtin_ia32_vcvtbiasps2bf8s_256_mask(
+      (__v32qi)__A, (__v8sf)__B, (__v16qi)(__m128i)_mm_setzero_si128(),
+      (__mmask8)__U);
+}
+
+// VCVTBIASPS2BF8S - 512-bit
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS512
+_mm512_cvts_biasps_bf8(__m512i __A, __m512 __B) {
+  return (__m128i)__builtin_ia32_vcvtbiasps2bf8s_512_mask(
+      (__v64qi)__A, (__v16sf)__B, (__v16qi)_mm_undefined_si128(),
+      (__mmask16)-1);
+}
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS512
+_mm512_mask_cvts_biasps_bf8(__m128i __W, __mmask16 __U, __m512i __A,
+                           __m512 __B) {
+  return (__m128i)__builtin_ia32_vcvtbiasps2bf8s_512_mask(
+      (__v64qi)__A, (__v16sf)__B, (__v16qi)(__m128i)__W, (__mmask16)__U);
+}
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS512
+_mm512_maskz_cvts_biasps_bf8(__mmask16 __U, __m512i __A, __m512 __B) {
+  return (__m128i)__builtin_ia32_vcvtbiasps2bf8s_512_mask(
+      (__v64qi)__A, (__v16sf)__B, (__v16qi)(__m128i)_mm_setzero_si128(),
+      (__mmask16)__U);
+}
+
+// VCVTBIASPS2HF8 - 128-bit
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS128
+_mm_cvtbiasps_hf8(__m128i __A, __m128 __B) {
+  return (__m128i)__builtin_ia32_vcvtbiasps2hf8_128_mask(
+      (__v16qi)__A, (__v4sf)__B, (__v16qi)_mm_undefined_si128(), (__mmask8)-1);
+}
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS128
+_mm_mask_cvtbiasps_hf8(__m128i __W, __mmask8 __U, __m128i __A, __m128 __B) {
+  return (__m128i)__builtin_ia32_vcvtbiasps2hf8_128_mask(
+      (__v16qi)__A, (__v4sf)__B, (__v16qi)(__m128i)__W, (__mmask8)__U);
+}
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS128
+_mm_maskz_cvtbiasps_hf8(__mmask8 __U, __m128i __A, __m128 __B) {
+  return (__m128i)__builtin_ia32_vcvtbiasps2hf8_128_mask(
+      (__v16qi)__A, (__v4sf)__B, (__v16qi)(__m128i)_mm_setzero_si128(),
+      (__mmask8)__U);
+}
+
+// VCVTBIASPS2HF8 - 256-bit
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS256
+_mm256_cvtbiasps_hf8(__m256i __A, __m256 __B) {
+  return (__m128i)__builtin_ia32_vcvtbiasps2hf8_256_mask(
+      (__v32qi)__A, (__v8sf)__B, (__v16qi)_mm_undefined_si128(), (__mmask8)-1);
+}
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS256
+_mm256_mask_cvtbiasps_hf8(__m128i __W, __mmask8 __U, __m256i __A, __m256 __B) {
+  return (__m128i)__builtin_ia32_vcvtbiasps2hf8_256_mask(
+      (__v32qi)__A, (__v8sf)__B, (__v16qi)(__m128i)__W, (__mmask8)__U);
+}
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS256
+_mm256_maskz_cvtbiasps_hf8(__mmask8 __U, __m256i __A, __m256 __B) {
+  return (__m128i)__builtin_ia32_vcvtbiasps2hf8_256_mask(
+      (__v32qi)__A, (__v8sf)__B, (__v16qi)(__m128i)_mm_setzero_si128(),
+      (__mmask8)__U);
+}
+
+// VCVTBIASPS2HF8 - 512-bit
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS512
+_mm512_cvtbiasps_hf8(__m512i __A, __m512 __B) {
+  return (__m128i)__builtin_ia32_vcvtbiasps2hf8_512_mask(
+      (__v64qi)__A, (__v16sf)__B, (__v16qi)_mm_undefined_si128(),
+      (__mmask16)-1);
+}
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS512
+_mm512_mask_cvtbiasps_hf8(__m128i __W, __mmask16 __U, __m512i __A,
+                          __m512 __B) {
+  return (__m128i)__builtin_ia32_vcvtbiasps2hf8_512_mask(
+      (__v64qi)__A, (__v16sf)__B, (__v16qi)(__m128i)__W, (__mmask16)__U);
+}
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS512
+_mm512_maskz_cvtbiasps_hf8(__mmask16 __U, __m512i __A, __m512 __B) {
+  return (__m128i)__builtin_ia32_vcvtbiasps2hf8_512_mask(
+      (__v64qi)__A, (__v16sf)__B, (__v16qi)(__m128i)_mm_setzero_si128(),
+      (__mmask16)__U);
+}
+
+// VCVTBIASPS2HF8S - 128-bit
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS128
+_mm_cvts_biasps_hf8(__m128i __A, __m128 __B) {
+  return (__m128i)__builtin_ia32_vcvtbiasps2hf8s_128_mask(
+      (__v16qi)__A, (__v4sf)__B, (__v16qi)_mm_undefined_si128(), (__mmask8)-1);
+}
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS128
+_mm_mask_cvts_biasps_hf8(__m128i __W, __mmask8 __U, __m128i __A, __m128 __B) {
+  return (__m128i)__builtin_ia32_vcvtbiasps2hf8s_128_mask(
+      (__v16qi)__A, (__v4sf)__B, (__v16qi)(__m128i)__W, (__mmask8)__U);
+}
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS128
+_mm_maskz_cvts_biasps_hf8(__mmask8 __U, __m128i __A, __m128 __B) {
+  return (__m128i)__builtin_ia32_vcvtbiasps2hf8s_128_mask(
+      (__v16qi)__A, (__v4sf)__B, (__v16qi)(__m128i)_mm_setzero_si128(),
+      (__mmask8)__U);
+}
+
+// VCVTBIASPS2HF8S - 256-bit
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS256
+_mm256_cvts_biasps_hf8(__m256i __A, __m256 __B) {
+  return (__m128i)__builtin_ia32_vcvtbiasps2hf8s_256_mask(
+      (__v32qi)__A, (__v8sf)__B, (__v16qi)_mm_undefined_si128(), (__mmask8)-1);
+}
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS256 _mm256_mask_cvts_biasps_hf8(
+    __m128i __W, __mmask8 __U, __m256i __A, __m256 __B) {
+  return (__m128i)__builtin_ia32_vcvtbiasps2hf8s_256_mask(
+      (__v32qi)__A, (__v8sf)__B, (__v16qi)(__m128i)__W, (__mmask8)__U);
+}
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS256
+_mm256_maskz_cvts_biasps_hf8(__mmask8 __U, __m256i __A, __m256 __B) {
+  return (__m128i)__builtin_ia32_vcvtbiasps2hf8s_256_mask(
+      (__v32qi)__A, (__v8sf)__B, (__v16qi)(__m128i)_mm_setzero_si128(),
+      (__mmask8)__U);
+}
+
+// VCVTBIASPS2HF8S - 512-bit
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS512
+_mm512_cvts_biasps_hf8(__m512i __A, __m512 __B) {
+  return (__m128i)__builtin_ia32_vcvtbiasps2hf8s_512_mask(
+      (__v64qi)__A, (__v16sf)__B, (__v16qi)_mm_undefined_si128(),
+      (__mmask16)-1);
+}
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS512
+_mm512_mask_cvts_biasps_hf8(__m128i __W, __mmask16 __U, __m512i __A,
+                           __m512 __B) {
+  return (__m128i)__builtin_ia32_vcvtbiasps2hf8s_512_mask(
+      (__v64qi)__A, (__v16sf)__B, (__v16qi)(__m128i)__W, (__mmask16)__U);
+}
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS512
+_mm512_maskz_cvts_biasps_hf8(__mmask16 __U, __m512i __A, __m512 __B) {
+  return (__m128i)__builtin_ia32_vcvtbiasps2hf8s_512_mask(
+      (__v64qi)__A, (__v16sf)__B, (__v16qi)(__m128i)_mm_setzero_si128(),
+      (__mmask16)__U);
+}
+
+//===----------------------------------------------------------------------===//
+// Group C: VCVTBF82PS / VCVTHF82PS
+// Convert packed FP8 to single-precision. Input is __m128i, output varies.
+//===----------------------------------------------------------------------===//
+
+// VCVTBF82PS - 128-bit
+
+static __inline__ __m128 __DEFAULT_FN_ATTRS128
+_mm_cvtbf8_ps(__m128i __A) {
+  return (__m128)__builtin_ia32_vcvtbf8_2ps128_mask(
+      (__v16qi)__A, (__v4sf)_mm_undefined_ps(), (__mmask8)-1);
+}
+
+static __inline__ __m128 __DEFAULT_FN_ATTRS128
+_mm_mask_cvtbf8_ps(__m128 __W, __mmask8 __U, __m128i __A) {
+  return (__m128)__builtin_ia32_vcvtbf8_2ps128_mask(
+      (__v16qi)__A, (__v4sf)__W, (__mmask8)__U);
+}
+
+static __inline__ __m128 __DEFAULT_FN_ATTRS128
+_mm_maskz_cvtbf8_ps(__mmask8 __U, __m128i __A) {
+  return (__m128)__builtin_ia32_vcvtbf8_2ps128_mask(
+      (__v16qi)__A, (__v4sf)_mm_setzero_ps(), (__mmask8)__U);
+}
+
+// VCVTBF82PS - 256-bit
+
+static __inline__ __m256 __DEFAULT_FN_ATTRS256
+_mm256_cvtbf8_ps(__m128i __A) {
+  return (__m256)__builtin_ia32_vcvtbf8_2ps256_mask(
+      (__v16qi)__A, (__v8sf)_mm256_undefined_ps(), (__mmask8)-1);
+}
+
+static __inline__ __m256 __DEFAULT_FN_ATTRS256
+_mm256_mask_cvtbf8_ps(__m256 __W, __mmask8 __U, __m128i __A) {
+  return (__m256)__builtin_ia32_vcvtbf8_2ps256_mask(
+      (__v16qi)__A, (__v8sf)__W, (__mmask8)__U);
+}
+
+static __inline__ __m256 __DEFAULT_FN_ATTRS256
+_mm256_maskz_cvtbf8_ps(__mmask8 __U, __m128i __A) {
+  return (__m256)__builtin_ia32_vcvtbf8_2ps256_mask(
+      (__v16qi)__A, (__v8sf)_mm256_setzero_ps(), (__mmask8)__U);
+}
+
+// VCVTBF82PS - 512-bit
+
+static __inline__ __m512 __DEFAULT_FN_ATTRS512
+_mm512_cvtbf8_ps(__m128i __A) {
+  return (__m512)__builtin_ia32_vcvtbf8_2ps512_mask(
+      (__v16qi)__A, (__v16sf)_mm512_undefined_ps(), (__mmask16)-1);
+}
+
+static __inline__ __m512 __DEFAULT_FN_ATTRS512
+_mm512_mask_cvtbf8_ps(__m512 __W, __mmask16 __U, __m128i __A) {
+  return (__m512)__builtin_ia32_vcvtbf8_2ps512_mask(
+      (__v16qi)__A, (__v16sf)__W, (__mmask16)__U);
+}
+
+static __inline__ __m512 __DEFAULT_FN_ATTRS512
+_mm512_maskz_cvtbf8_ps(__mmask16 __U, __m128i __A) {
+  return (__m512)__builtin_ia32_vcvtbf8_2ps512_mask(
+      (__v16qi)__A, (__v16sf)_mm512_setzero_ps(), (__mmask16)__U);
+}
+
+// VCVTHF82PS - 128-bit
+
+static __inline__ __m128 __DEFAULT_FN_ATTRS128
+_mm_cvthf8_ps(__m128i __A) {
+  return (__m128)__builtin_ia32_vcvthf8_2ps128_mask(
+      (__v16qi)__A, (__v4sf)_mm_undefined_ps(), (__mmask8)-1);
+}
+
+static __inline__ __m128 __DEFAULT_FN_ATTRS128
+_mm_mask_cvthf8_ps(__m128 __W, __mmask8 __U, __m128i __A) {
+  return (__m128)__builtin_ia32_vcvthf8_2ps128_mask(
+      (__v16qi)__A, (__v4sf)__W, (__mmask8)__U);
+}
+
+static __inline__ __m128 __DEFAULT_FN_ATTRS128
+_mm_maskz_cvthf8_ps(__mmask8 __U, __m128i __A) {
+  return (__m128)__builtin_ia32_vcvthf8_2ps128_mask(
+      (__v16qi)__A, (__v4sf)_mm_setzero_ps(), (__mmask8)__U);
+}
+
+// VCVTHF82PS - 256-bit
+
+static __inline__ __m256 __DEFAULT_FN_ATTRS256
+_mm256_cvthf8_ps(__m128i __A) {
+  return (__m256)__builtin_ia32_vcvthf8_2ps256_mask(
+      (__v16qi)__A, (__v8sf)_mm256_undefined_ps(), (__mmask8)-1);
+}
+
+static __inline__ __m256 __DEFAULT_FN_ATTRS256
+_mm256_mask_cvthf8_ps(__m256 __W, __mmask8 __U, __m128i __A) {
+  return (__m256)__builtin_ia32_vcvthf8_2ps256_mask(
+      (__v16qi)__A, (__v8sf)__W, (__mmask8)__U);
+}
+
+static __inline__ __m256 __DEFAULT_FN_ATTRS256
+_mm256_maskz_cvthf8_ps(__mmask8 __U, __m128i __A) {
+  return (__m256)__builtin_ia32_vcvthf8_2ps256_mask(
+      (__v16qi)__A, (__v8sf)_mm256_setzero_ps(), (__mmask8)__U);
+}
+
+// VCVTHF82PS - 512-bit
+
+static __inline__ __m512 __DEFAULT_FN_ATTRS512
+_mm512_cvthf8_ps(__m128i __A) {
+  return (__m512)__builtin_ia32_vcvthf8_2ps512_mask(
+      (__v16qi)__A, (__v16sf)_mm512_undefined_ps(), (__mmask16)-1);
+}
+
+static __inline__ __m512 __DEFAULT_FN_ATTRS512
+_mm512_mask_cvthf8_ps(__m512 __W, __mmask16 __U, __m128i __A) {
+  return (__m512)__builtin_ia32_vcvthf8_2ps512_mask(
+      (__v16qi)__A, (__v16sf)__W, (__mmask16)__U);
+}
+
+static __inline__ __m512 __DEFAULT_FN_ATTRS512
+_mm512_maskz_cvthf8_ps(__mmask16 __U, __m128i __A) {
+  return (__m512)__builtin_ia32_vcvthf8_2ps512_mask(
+      (__v16qi)__A, (__v16sf)_mm512_setzero_ps(), (__mmask16)__U);
+}
+
+//===----------------------------------------------------------------------===//
+// Group E: VCVTBF82BF6S / VCVTHF82HF6S
+// Same-size reg-only conversions (no masking support)
+//===----------------------------------------------------------------------===//
+
+// VCVTBF82BF6S
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS128
+_mm_cvtbf8_bf6s(__m128i __A) {
+  return (__m128i)__builtin_ia32_vcvtbf82bf6s128((__v16qi)__A);
+}
+
+static __inline__ __m256i __DEFAULT_FN_ATTRS256
+_mm256_cvtbf8_bf6s(__m256i __A) {
+  return (__m256i)__builtin_ia32_vcvtbf82bf6s256((__v32qi)__A);
+}
+
+static __inline__ __m512i __DEFAULT_FN_ATTRS512
+_mm512_cvtbf8_bf6s(__m512i __A) {
+  return (__m512i)__builtin_ia32_vcvtbf82bf6s512((__v64qi)__A);
+}
+
+// VCVTHF82HF6S
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS128
+_mm_cvthf8_hf6s(__m128i __A) {
+  return (__m128i)__builtin_ia32_vcvthf82hf6s128((__v16qi)__A);
+}
+
+static __inline__ __m256i __DEFAULT_FN_ATTRS256
+_mm256_cvthf8_hf6s(__m256i __A) {
+  return (__m256i)__builtin_ia32_vcvthf82hf6s256((__v32qi)__A);
+}
+
+static __inline__ __m512i __DEFAULT_FN_ATTRS512
+_mm512_cvthf8_hf6s(__m512i __A) {
+  return (__m512i)__builtin_ia32_vcvthf82hf6s512((__v64qi)__A);
+}
+
+//===----------------------------------------------------------------------===//
+// Group F: VCVTBF42HF8 / VCVTBF62HF8 / VCVTHF62HF8
+// Expanding/same-size conversions with masking support
+//===----------------------------------------------------------------------===//
+
+// VCVTBF42HF8 - 128-bit
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS128
+_mm_cvtbf4_hf8(__m128i __A) {
+  return (__m128i)__builtin_ia32_vcvtbf42hf8128((__v16qi)__A);
+}
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS128
+_mm_mask_cvtbf4_hf8(__m128i __W, __mmask16 __U, __m128i __A) {
+  return (__m128i)__builtin_ia32_selectb_128(
+      __U, (__v16qi)_mm_cvtbf4_hf8(__A), (__v16qi)__W);
+}
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS128
+_mm_maskz_cvtbf4_hf8(__mmask16 __U, __m128i __A) {
+  return (__m128i)__builtin_ia32_selectb_128(
+      __U, (__v16qi)_mm_cvtbf4_hf8(__A), (__v16qi)_mm_setzero_si128());
+}
+
+// VCVTBF42HF8 - 256-bit
+
+static __inline__ __m256i __DEFAULT_FN_ATTRS256
+_mm256_cvtbf4_hf8(__m128i __A) {
+  return (__m256i)__builtin_ia32_vcvtbf42hf8256((__v16qi)__A);
+}
+
+static __inline__ __m256i __DEFAULT_FN_ATTRS256
+_mm256_mask_cvtbf4_hf8(__m256i __W, __mmask32 __U, __m128i __A) {
+  return (__m256i)__builtin_ia32_selectb_256(
+      __U, (__v32qi)_mm256_cvtbf4_hf8(__A), (__v32qi)__W);
+}
+
+static __inline__ __m256i __DEFAULT_FN_ATTRS256
+_mm256_maskz_cvtbf4_hf8(__mmask32 __U, __m128i __A) {
+  return (__m256i)__builtin_ia32_selectb_256(
+      __U, (__v32qi)_mm256_cvtbf4_hf8(__A), (__v32qi)_mm256_setzero_si256());
+}
+
+// VCVTBF42HF8 - 512-bit
+
+static __inline__ __m512i __DEFAULT_FN_ATTRS512
+_mm512_cvtbf4_hf8(__m256i __A) {
+  return (__m512i)__builtin_ia32_vcvtbf42hf8512((__v32qi)__A);
+}
+
+static __inline__ __m512i __DEFAULT_FN_ATTRS512
+_mm512_mask_cvtbf4_hf8(__m512i __W, __mmask64 __U, __m256i __A) {
+  return (__m512i)__builtin_ia32_selectb_512(
+      __U, (__v64qi)_mm512_cvtbf4_hf8(__A), (__v64qi)__W);
+}
+
+static __inline__ __m512i __DEFAULT_FN_ATTRS512
+_mm512_maskz_cvtbf4_hf8(__mmask64 __U, __m256i __A) {
+  return (__m512i)__builtin_ia32_selectb_512(
+      __U, (__v64qi)_mm512_cvtbf4_hf8(__A), (__v64qi)_mm512_setzero_si512());
+}
+
+// VCVTBF62HF8 - 128-bit
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS128
+_mm_cvtbf6_hf8(__m128i __A) {
+  return (__m128i)__builtin_ia32_vcvtbf62hf8128((__v16qi)__A);
+}
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS128
+_mm_mask_cvtbf6_hf8(__m128i __W, __mmask16 __U, __m128i __A) {
+  return (__m128i)__builtin_ia32_selectb_128(
+      __U, (__v16qi)_mm_cvtbf6_hf8(__A), (__v16qi)__W);
+}
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS128
+_mm_maskz_cvtbf6_hf8(__mmask16 __U, __m128i __A) {
+  return (__m128i)__builtin_ia32_selectb_128(
+      __U, (__v16qi)_mm_cvtbf6_hf8(__A), (__v16qi)_mm_setzero_si128());
+}
+
+// VCVTBF62HF8 - 256-bit
+
+static __inline__ __m256i __DEFAULT_FN_ATTRS256
+_mm256_cvtbf6_hf8(__m256i __A) {
+  return (__m256i)__builtin_ia32_vcvtbf62hf8256((__v32qi)__A);
+}
+
+static __inline__ __m256i __DEFAULT_FN_ATTRS256
+_mm256_mask_cvtbf6_hf8(__m256i __W, __mmask32 __U, __m256i __A) {
+  return (__m256i)__builtin_ia32_selectb_256(
+      __U, (__v32qi)_mm256_cvtbf6_hf8(__A), (__v32qi)__W);
+}
+
+static __inline__ __m256i __DEFAULT_FN_ATTRS256
+_mm256_maskz_cvtbf6_hf8(__mmask32 __U, __m256i __A) {
+  return (__m256i)__builtin_ia32_selectb_256(
+      __U, (__v32qi)_mm256_cvtbf6_hf8(__A), (__v32qi)_mm256_setzero_si256());
+}
+
+// VCVTBF62HF8 - 512-bit
+
+static __inline__ __m512i __DEFAULT_FN_ATTRS512
+_mm512_cvtbf6_hf8(__m512i __A) {
+  return (__m512i)__builtin_ia32_vcvtbf62hf8512((__v64qi)__A);
+}
+
+static __inline__ __m512i __DEFAULT_FN_ATTRS512
+_mm512_mask_cvtbf6_hf8(__m512i __W, __mmask64 __U, __m512i __A) {
+  return (__m512i)__builtin_ia32_selectb_512(
+      __U, (__v64qi)_mm512_cvtbf6_hf8(__A), (__v64qi)__W);
+}
+
+static __inline__ __m512i __DEFAULT_FN_ATTRS512
+_mm512_maskz_cvtbf6_hf8(__mmask64 __U, __m512i __A) {
+  return (__m512i)__builtin_ia32_selectb_512(
+      __U, (__v64qi)_mm512_cvtbf6_hf8(__A), (__v64qi)_mm512_setzero_si512());
+}
+
+// VCVTHF62HF8 - 128-bit
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS128
+_mm_cvthf6_hf8(__m128i __A) {
+  return (__m128i)__builtin_ia32_vcvthf62hf8128((__v16qi)__A);
+}
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS128
+_mm_mask_cvthf6_hf8(__m128i __W, __mmask16 __U, __m128i __A) {
+  return (__m128i)__builtin_ia32_selectb_128(
+      __U, (__v16qi)_mm_cvthf6_hf8(__A), (__v16qi)__W);
+}
+
+static __inline__ __m128i __DEFAULT_FN_ATTRS128
+_mm_maskz_cvthf6_hf8(__mmask16 __U, __m128i __A) {
+  return (__m128i)__builtin_ia32_selectb_128(
+      __U, (__v16qi)_mm_cvthf6_hf8(__A), (__v16qi)_mm_setzero_si128());
+}
+
+// VCVTHF62HF8 - 256-bit
+
+static __inline__ __m256i __DEFAULT_FN_ATTRS256
+_mm256_cvthf6_hf8(__m256i __A) {
+  return (__m256i)__builtin_ia32_vcvthf62hf8256((__v32qi)__A);
+}
+
+static __inline__ __m256i __DEFAULT_FN_ATTRS256
+_mm256_mask_cvthf6_hf8(__m256i __W, __mmask32 __U, __m256i __A) {
+  return (__m256i)__builtin_ia32_selectb_256(
+      __U, (__v32qi)_mm256_cvthf6_hf8(__A), (__v32qi)__W);
+}
+
+static __inline__ __m256i __DEFAULT_FN_ATTRS256
+_mm256_maskz_cvthf6_hf8(__mmask32 __U, __m256i __A) {
+  return (__m256i)__builtin_ia32_selectb_256(
+      __U, (__v32qi)_mm256_cvthf6_hf8(__A), (__v32qi)_mm256_setzero_si256());
+}
+
+// VCVTHF62HF8 - 512-bit
+
+static __inline__ __m512i __DEFAULT_FN_ATTRS512
+_mm512_cvthf6_hf8(__m512i __A) {
+  return (__m512i)__builtin_ia32_vcvthf62hf8512((__v64qi)__A);
+}
+
+static __inline__ __m512i __DEFAULT_FN_ATTRS512
+_mm512_mask_cvthf6_hf8(__m512i __W, __mmask64 __U, __m512i __A) {
+  return (__m512i)__builtin_ia32_selectb_512(
+      __U, (__v64qi)_mm512_cvthf6_hf8(__A), (__v64qi)__W);
+}
+
+static __inline__ __m512i __DEFAULT_FN_ATTRS512
+_mm512_maskz_cvthf6_hf8(__mmask64 __U, __m512i __A) {
+  return (__m512i)__builtin_ia32_selectb_512(
+      __U, (__v64qi)_mm512_cvthf6_hf8(__A), (__v64qi)_mm512_setzero_si512());
+}
+
+//===----------------------------------------------------------------------===//
+// Group H: VUNPACKB
+// Byte unpack with immediate
+//===----------------------------------------------------------------------===//
+
+// VUNPACKB - 128-bit
+
+#define _mm_unpackb_epi8(A, imm)                                               \
+  ((__m128i)__builtin_ia32_vunpackb128((__v16qi)(__m128i)(A), (int)(imm)))
+
+#define _mm_mask_unpackb_epi8(W, U, A, imm)                                    \
+  ((__m128i)__builtin_ia32_selectb_128(                                         \
+      (__mmask16)(U),                                                           \
+      (__v16qi)_mm_unpackb_epi8((A), (imm)),                                    \
+      (__v16qi)(__m128i)(W)))
+
+#define _mm_maskz_unpackb_epi8(U, A, imm)                                      \
+  ((__m128i)__builtin_ia32_selectb_128(                                         \
+      (__mmask16)(U),                                                           \
+      (__v16qi)_mm_unpackb_epi8((A), (imm)),                                    \
+      (__v16qi)_mm_setzero_si128()))
+
+// VUNPACKB - 256-bit
+
+#define _mm256_unpackb_epi8(A, imm)                                            \
+  ((__m256i)__builtin_ia32_vunpackb256((__v32qi)(__m256i)(A), (int)(imm)))
+
+#define _mm256_mask_unpackb_epi8(W, U, A, imm)                                 \
+  ((__m256i)__builtin_ia32_selectb_256(                                         \
+      (__mmask32)(U),                                                           \
+      (__v32qi)_mm256_unpackb_epi8((A), (imm)),                                 \
+      (__v32qi)(__m256i)(W)))
+
+#define _mm256_maskz_unpackb_epi8(U, A, imm)                                   \
+  ((__m256i)__builtin_ia32_selectb_256(                                         \
+      (__mmask32)(U),                                                           \
+      (__v32qi)_mm256_unpackb_epi8((A), (imm)),                                 \
+      (__v32qi)_mm256_setzero_si256()))
+
+// VUNPACKB - 512-bit
+
+#define _mm512_unpackb_epi8(A, imm)                                            \
+  ((__m512i)__builtin_ia32_vunpackb512((__v64qi)(__m512i)(A), (int)(imm)))
+
+#define _mm512_mask_unpackb_epi8(W, U, A, imm)                                 \
+  ((__m512i)__builtin_ia32_selectb_512(                                         \
+      (__mmask64)(U),                                                           \
+      (__v64qi)_mm512_unpackb_epi8((A), (imm)),                                 \
+      (__v64qi)(__m512i)(W)))
+
+#define _mm512_maskz_unpackb_epi8(U, A, imm)                                   \
+  ((__m512i)__builtin_ia32_selectb_512(                                         \
+      (__mmask64)(U),                                                           \
+      (__v64qi)_mm512_unpackb_epi8((A), (imm)),                                 \
+      (__v64qi)_mm512_setzero_si512()))
+
+// clang-format on
+
+#undef __DEFAULT_FN_ATTRS128
+#undef __DEFAULT_FN_ATTRS256
+#undef __DEFAULT_FN_ATTRS512
+
+#endif // __AVX10_2_V2AUXINTRIN_H
+#endif // __SSE2__
diff --git a/clang/lib/Headers/cpuid.h b/clang/lib/Headers/cpuid.h
index 156425c7561bb6..a8c6adbc3e2dc8 100644
--- a/clang/lib/Headers/cpuid.h
+++ b/clang/lib/Headers/cpuid.h
@@ -222,6 +222,9 @@
 #define bit_AVX10         0x00080000
 #define bit_APXF          0x00200000
 
+/* Features in %ecx for leaf 24 sub-leaf 1 */
+#define bit_AVX10_V2_AUX 0x00000008
+
 /* Features in %eax for leaf 13 sub-leaf 1 */
 #define bit_XSAVEOPT    0x00000001
 #define bit_XSAVEC      0x00000002
diff --git a/clang/lib/Headers/immintrin.h b/clang/lib/Headers/immintrin.h
index 19064a4ff5cea3..bd474d65079351 100644
--- a/clang/lib/Headers/immintrin.h
+++ b/clang/lib/Headers/immintrin.h
@@ -500,6 +500,10 @@ _storebe_i64(void * __P, long long __D) {
 #include <avx10_2_512satcvtdsintrin.h>
 #include <avx10_2_512satcvtintrin.h>
 
+#ifdef __AVX10_V2_AUX__
+#include <avx10_2_v2auxintrin.h>
+#endif
+
 #include <sm4evexintrin.h>
 
 #include <enqcmdintrin.h>
diff --git a/clang/test/CodeGen/X86/avx10_2_v2aux-builtins.c b/clang/test/CodeGen/X86/avx10_2_v2aux-builtins.c
new file mode 100644
index 00000000000000..c4d076b27b21eb
--- /dev/null
+++ b/clang/test/CodeGen/X86/avx10_2_v2aux-builtins.c
@@ -0,0 +1,1056 @@
+// RUN: %clang_cc1 %s -flax-vector-conversions=none -ffreestanding -triple=x86_64 -target-feature +avx10-v2-aux \
+// RUN: -emit-llvm -o - -Wno-invalid-feature-combination -Wall -Werror | FileCheck %s
+// RUN: %clang_cc1 %s -flax-vector-conversions=none -ffreestanding -triple=i386 -target-feature +avx10-v2-aux \
+// RUN: -emit-llvm -o - -Wno-invalid-feature-combination -Wall -Werror | FileCheck %s
+
+#include <immintrin.h>
+
+//
+// Group A: VCVTPS2BF8 / VCVTPS2BF8S / VCVTPS2HF8 / VCVTPS2HF8S /
+//          VCVTROPS2HF8 / VCVTROPS2HF8S
+//
+
+// VCVTPS2BF8 - 128-bit
+
+__m128i test_mm_cvtps_bf8(__m128 __A) {
+  // CHECK-LABEL: @test_mm_cvtps_bf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.128(
+  return _mm_cvtps_bf8(__A);
+}
+
+__m128i test_mm_mask_cvtps_bf8(__m128i __W, __mmask8 __U, __m128 __A) {
+  // CHECK-LABEL: @test_mm_mask_cvtps_bf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.128(
+  return _mm_mask_cvtps_bf8(__W, __U, __A);
+}
+
+__m128i test_mm_maskz_cvtps_bf8(__mmask8 __U, __m128 __A) {
+  // CHECK-LABEL: @test_mm_maskz_cvtps_bf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.128(
+  return _mm_maskz_cvtps_bf8(__U, __A);
+}
+
+// VCVTPS2BF8 - 256-bit
+
+__m128i test_mm256_cvtps_bf8(__m256 __A) {
+  // CHECK-LABEL: @test_mm256_cvtps_bf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.256(
+  return _mm256_cvtps_bf8(__A);
+}
+
+__m128i test_mm256_mask_cvtps_bf8(__m128i __W, __mmask8 __U, __m256 __A) {
+  // CHECK-LABEL: @test_mm256_mask_cvtps_bf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.256(
+  return _mm256_mask_cvtps_bf8(__W, __U, __A);
+}
+
+__m128i test_mm256_maskz_cvtps_bf8(__mmask8 __U, __m256 __A) {
+  // CHECK-LABEL: @test_mm256_maskz_cvtps_bf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.256(
+  return _mm256_maskz_cvtps_bf8(__U, __A);
+}
+
+// VCVTPS2BF8 - 512-bit
+
+__m128i test_mm512_cvtps_bf8(__m512 __A) {
+  // CHECK-LABEL: @test_mm512_cvtps_bf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.512(
+  return _mm512_cvtps_bf8(__A);
+}
+
+__m128i test_mm512_mask_cvtps_bf8(__m128i __W, __mmask16 __U, __m512 __A) {
+  // CHECK-LABEL: @test_mm512_mask_cvtps_bf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.512(
+  return _mm512_mask_cvtps_bf8(__W, __U, __A);
+}
+
+__m128i test_mm512_maskz_cvtps_bf8(__mmask16 __U, __m512 __A) {
+  // CHECK-LABEL: @test_mm512_maskz_cvtps_bf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.512(
+  return _mm512_maskz_cvtps_bf8(__U, __A);
+}
+
+// VCVTPS2BF8S - 128-bit
+
+__m128i test_mm_cvts_ps_bf8(__m128 __A) {
+  // CHECK-LABEL: @test_mm_cvts_ps_bf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.128(
+  return _mm_cvts_ps_bf8(__A);
+}
+
+__m128i test_mm_mask_cvts_ps_bf8(__m128i __W, __mmask8 __U, __m128 __A) {
+  // CHECK-LABEL: @test_mm_mask_cvts_ps_bf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.128(
+  return _mm_mask_cvts_ps_bf8(__W, __U, __A);
+}
+
+__m128i test_mm_maskz_cvts_ps_bf8(__mmask8 __U, __m128 __A) {
+  // CHECK-LABEL: @test_mm_maskz_cvts_ps_bf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.128(
+  return _mm_maskz_cvts_ps_bf8(__U, __A);
+}
+
+// VCVTPS2BF8S - 256-bit
+
+__m128i test_mm256_cvts_ps_bf8(__m256 __A) {
+  // CHECK-LABEL: @test_mm256_cvts_ps_bf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.256(
+  return _mm256_cvts_ps_bf8(__A);
+}
+
+__m128i test_mm256_mask_cvts_ps_bf8(__m128i __W, __mmask8 __U, __m256 __A) {
+  // CHECK-LABEL: @test_mm256_mask_cvts_ps_bf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.256(
+  return _mm256_mask_cvts_ps_bf8(__W, __U, __A);
+}
+
+__m128i test_mm256_maskz_cvts_ps_bf8(__mmask8 __U, __m256 __A) {
+  // CHECK-LABEL: @test_mm256_maskz_cvts_ps_bf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.256(
+  return _mm256_maskz_cvts_ps_bf8(__U, __A);
+}
+
+// VCVTPS2BF8S - 512-bit
+
+__m128i test_mm512_cvts_ps_bf8(__m512 __A) {
+  // CHECK-LABEL: @test_mm512_cvts_ps_bf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.512(
+  return _mm512_cvts_ps_bf8(__A);
+}
+
+__m128i test_mm512_mask_cvts_ps_bf8(__m128i __W, __mmask16 __U, __m512 __A) {
+  // CHECK-LABEL: @test_mm512_mask_cvts_ps_bf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.512(
+  return _mm512_mask_cvts_ps_bf8(__W, __U, __A);
+}
+
+__m128i test_mm512_maskz_cvts_ps_bf8(__mmask16 __U, __m512 __A) {
+  // CHECK-LABEL: @test_mm512_maskz_cvts_ps_bf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.512(
+  return _mm512_maskz_cvts_ps_bf8(__U, __A);
+}
+
+// VCVTPS2HF8 - 128-bit
+
+__m128i test_mm_cvtps_hf8(__m128 __A) {
+  // CHECK-LABEL: @test_mm_cvtps_hf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.128(
+  return _mm_cvtps_hf8(__A);
+}
+
+__m128i test_mm_mask_cvtps_hf8(__m128i __W, __mmask8 __U, __m128 __A) {
+  // CHECK-LABEL: @test_mm_mask_cvtps_hf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.128(
+  return _mm_mask_cvtps_hf8(__W, __U, __A);
+}
+
+__m128i test_mm_maskz_cvtps_hf8(__mmask8 __U, __m128 __A) {
+  // CHECK-LABEL: @test_mm_maskz_cvtps_hf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.128(
+  return _mm_maskz_cvtps_hf8(__U, __A);
+}
+
+// VCVTPS2HF8 - 256-bit
+
+__m128i test_mm256_cvtps_hf8(__m256 __A) {
+  // CHECK-LABEL: @test_mm256_cvtps_hf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.256(
+  return _mm256_cvtps_hf8(__A);
+}
+
+__m128i test_mm256_mask_cvtps_hf8(__m128i __W, __mmask8 __U, __m256 __A) {
+  // CHECK-LABEL: @test_mm256_mask_cvtps_hf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.256(
+  return _mm256_mask_cvtps_hf8(__W, __U, __A);
+}
+
+__m128i test_mm256_maskz_cvtps_hf8(__mmask8 __U, __m256 __A) {
+  // CHECK-LABEL: @test_mm256_maskz_cvtps_hf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.256(
+  return _mm256_maskz_cvtps_hf8(__U, __A);
+}
+
+// VCVTPS2HF8 - 512-bit
+
+__m128i test_mm512_cvtps_hf8(__m512 __A) {
+  // CHECK-LABEL: @test_mm512_cvtps_hf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.512(
+  return _mm512_cvtps_hf8(__A);
+}
+
+__m128i test_mm512_mask_cvtps_hf8(__m128i __W, __mmask16 __U, __m512 __A) {
+  // CHECK-LABEL: @test_mm512_mask_cvtps_hf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.512(
+  return _mm512_mask_cvtps_hf8(__W, __U, __A);
+}
+
+__m128i test_mm512_maskz_cvtps_hf8(__mmask16 __U, __m512 __A) {
+  // CHECK-LABEL: @test_mm512_maskz_cvtps_hf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.512(
+  return _mm512_maskz_cvtps_hf8(__U, __A);
+}
+
+// VCVTPS2HF8S - 128-bit
+
+__m128i test_mm_cvts_ps_hf8(__m128 __A) {
+  // CHECK-LABEL: @test_mm_cvts_ps_hf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.128(
+  return _mm_cvts_ps_hf8(__A);
+}
+
+__m128i test_mm_mask_cvts_ps_hf8(__m128i __W, __mmask8 __U, __m128 __A) {
+  // CHECK-LABEL: @test_mm_mask_cvts_ps_hf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.128(
+  return _mm_mask_cvts_ps_hf8(__W, __U, __A);
+}
+
+__m128i test_mm_maskz_cvts_ps_hf8(__mmask8 __U, __m128 __A) {
+  // CHECK-LABEL: @test_mm_maskz_cvts_ps_hf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.128(
+  return _mm_maskz_cvts_ps_hf8(__U, __A);
+}
+
+// VCVTPS2HF8S - 256-bit
+
+__m128i test_mm256_cvts_ps_hf8(__m256 __A) {
+  // CHECK-LABEL: @test_mm256_cvts_ps_hf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.256(
+  return _mm256_cvts_ps_hf8(__A);
+}
+
+__m128i test_mm256_mask_cvts_ps_hf8(__m128i __W, __mmask8 __U, __m256 __A) {
+  // CHECK-LABEL: @test_mm256_mask_cvts_ps_hf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.256(
+  return _mm256_mask_cvts_ps_hf8(__W, __U, __A);
+}
+
+__m128i test_mm256_maskz_cvts_ps_hf8(__mmask8 __U, __m256 __A) {
+  // CHECK-LABEL: @test_mm256_maskz_cvts_ps_hf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.256(
+  return _mm256_maskz_cvts_ps_hf8(__U, __A);
+}
+
+// VCVTPS2HF8S - 512-bit
+
+__m128i test_mm512_cvts_ps_hf8(__m512 __A) {
+  // CHECK-LABEL: @test_mm512_cvts_ps_hf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.512(
+  return _mm512_cvts_ps_hf8(__A);
+}
+
+__m128i test_mm512_mask_cvts_ps_hf8(__m128i __W, __mmask16 __U, __m512 __A) {
+  // CHECK-LABEL: @test_mm512_mask_cvts_ps_hf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.512(
+  return _mm512_mask_cvts_ps_hf8(__W, __U, __A);
+}
+
+__m128i test_mm512_maskz_cvts_ps_hf8(__mmask16 __U, __m512 __A) {
+  // CHECK-LABEL: @test_mm512_maskz_cvts_ps_hf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.512(
+  return _mm512_maskz_cvts_ps_hf8(__U, __A);
+}
+
+// VCVTROPS2HF8 - 128-bit
+
+__m128i test_mm_cvtrops_hf8(__m128 __A) {
+  // CHECK-LABEL: @test_mm_cvtrops_hf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.128(
+  return _mm_cvtrops_hf8(__A);
+}
+
+__m128i test_mm_mask_cvtrops_hf8(__m128i __W, __mmask8 __U, __m128 __A) {
+  // CHECK-LABEL: @test_mm_mask_cvtrops_hf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.128(
+  return _mm_mask_cvtrops_hf8(__W, __U, __A);
+}
+
+__m128i test_mm_maskz_cvtrops_hf8(__mmask8 __U, __m128 __A) {
+  // CHECK-LABEL: @test_mm_maskz_cvtrops_hf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.128(
+  return _mm_maskz_cvtrops_hf8(__U, __A);
+}
+
+// VCVTROPS2HF8 - 256-bit
+
+__m128i test_mm256_cvtrops_hf8(__m256 __A) {
+  // CHECK-LABEL: @test_mm256_cvtrops_hf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.256(
+  return _mm256_cvtrops_hf8(__A);
+}
+
+__m128i test_mm256_mask_cvtrops_hf8(__m128i __W, __mmask8 __U, __m256 __A) {
+  // CHECK-LABEL: @test_mm256_mask_cvtrops_hf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.256(
+  return _mm256_mask_cvtrops_hf8(__W, __U, __A);
+}
+
+__m128i test_mm256_maskz_cvtrops_hf8(__mmask8 __U, __m256 __A) {
+  // CHECK-LABEL: @test_mm256_maskz_cvtrops_hf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.256(
+  return _mm256_maskz_cvtrops_hf8(__U, __A);
+}
+
+// VCVTROPS2HF8 - 512-bit
+
+__m128i test_mm512_cvtrops_hf8(__m512 __A) {
+  // CHECK-LABEL: @test_mm512_cvtrops_hf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.512(
+  return _mm512_cvtrops_hf8(__A);
+}
+
+__m128i test_mm512_mask_cvtrops_hf8(__m128i __W, __mmask16 __U, __m512 __A) {
+  // CHECK-LABEL: @test_mm512_mask_cvtrops_hf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.512(
+  return _mm512_mask_cvtrops_hf8(__W, __U, __A);
+}
+
+__m128i test_mm512_maskz_cvtrops_hf8(__mmask16 __U, __m512 __A) {
+  // CHECK-LABEL: @test_mm512_maskz_cvtrops_hf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.512(
+  return _mm512_maskz_cvtrops_hf8(__U, __A);
+}
+
+// VCVTROPS2HF8S - 128-bit
+
+__m128i test_mm_cvts_rops_hf8(__m128 __A) {
+  // CHECK-LABEL: @test_mm_cvts_rops_hf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.128(
+  return _mm_cvts_rops_hf8(__A);
+}
+
+__m128i test_mm_mask_cvts_rops_hf8(__m128i __W, __mmask8 __U, __m128 __A) {
+  // CHECK-LABEL: @test_mm_mask_cvts_rops_hf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.128(
+  return _mm_mask_cvts_rops_hf8(__W, __U, __A);
+}
+
+__m128i test_mm_maskz_cvts_rops_hf8(__mmask8 __U, __m128 __A) {
+  // CHECK-LABEL: @test_mm_maskz_cvts_rops_hf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.128(
+  return _mm_maskz_cvts_rops_hf8(__U, __A);
+}
+
+// VCVTROPS2HF8S - 256-bit
+
+__m128i test_mm256_cvts_rops_hf8(__m256 __A) {
+  // CHECK-LABEL: @test_mm256_cvts_rops_hf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.256(
+  return _mm256_cvts_rops_hf8(__A);
+}
+
+__m128i test_mm256_mask_cvts_rops_hf8(__m128i __W, __mmask8 __U, __m256 __A) {
+  // CHECK-LABEL: @test_mm256_mask_cvts_rops_hf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.256(
+  return _mm256_mask_cvts_rops_hf8(__W, __U, __A);
+}
+
+__m128i test_mm256_maskz_cvts_rops_hf8(__mmask8 __U, __m256 __A) {
+  // CHECK-LABEL: @test_mm256_maskz_cvts_rops_hf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.256(
+  return _mm256_maskz_cvts_rops_hf8(__U, __A);
+}
+
+// VCVTROPS2HF8S - 512-bit
+
+__m128i test_mm512_cvts_rops_hf8(__m512 __A) {
+  // CHECK-LABEL: @test_mm512_cvts_rops_hf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.512(
+  return _mm512_cvts_rops_hf8(__A);
+}
+
+__m128i test_mm512_mask_cvts_rops_hf8(__m128i __W, __mmask16 __U, __m512 __A) {
+  // CHECK-LABEL: @test_mm512_mask_cvts_rops_hf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.512(
+  return _mm512_mask_cvts_rops_hf8(__W, __U, __A);
+}
+
+__m128i test_mm512_maskz_cvts_rops_hf8(__mmask16 __U, __m512 __A) {
+  // CHECK-LABEL: @test_mm512_maskz_cvts_rops_hf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.512(
+  return _mm512_maskz_cvts_rops_hf8(__U, __A);
+}
+
+//
+// Group B: VCVTBIASPS2BF8 / VCVTBIASPS2BF8S / VCVTBIASPS2HF8 /
+//          VCVTBIASPS2HF8S
+//
+
+// VCVTBIASPS2BF8 - 128-bit
+
+__m128i test_mm_cvtbiasps_bf8(__m128i __A, __m128 __B) {
+  // CHECK-LABEL: @test_mm_cvtbiasps_bf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.128(
+  return _mm_cvtbiasps_bf8(__A, __B);
+}
+
+__m128i test_mm_mask_cvtbiasps_bf8(__m128i __W, __mmask8 __U, __m128i __A, __m128 __B) {
+  // CHECK-LABEL: @test_mm_mask_cvtbiasps_bf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.128(
+  return _mm_mask_cvtbiasps_bf8(__W, __U, __A, __B);
+}
+
+__m128i test_mm_maskz_cvtbiasps_bf8(__mmask8 __U, __m128i __A, __m128 __B) {
+  // CHECK-LABEL: @test_mm_maskz_cvtbiasps_bf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.128(
+  return _mm_maskz_cvtbiasps_bf8(__U, __A, __B);
+}
+
+// VCVTBIASPS2BF8 - 256-bit
+
+__m128i test_mm256_cvtbiasps_bf8(__m256i __A, __m256 __B) {
+  // CHECK-LABEL: @test_mm256_cvtbiasps_bf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.256(
+  return _mm256_cvtbiasps_bf8(__A, __B);
+}
+
+__m128i test_mm256_mask_cvtbiasps_bf8(__m128i __W, __mmask8 __U, __m256i __A, __m256 __B) {
+  // CHECK-LABEL: @test_mm256_mask_cvtbiasps_bf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.256(
+  return _mm256_mask_cvtbiasps_bf8(__W, __U, __A, __B);
+}
+
+__m128i test_mm256_maskz_cvtbiasps_bf8(__mmask8 __U, __m256i __A, __m256 __B) {
+  // CHECK-LABEL: @test_mm256_maskz_cvtbiasps_bf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.256(
+  return _mm256_maskz_cvtbiasps_bf8(__U, __A, __B);
+}
+
+// VCVTBIASPS2BF8 - 512-bit
+
+__m128i test_mm512_cvtbiasps_bf8(__m512i __A, __m512 __B) {
+  // CHECK-LABEL: @test_mm512_cvtbiasps_bf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.512(
+  return _mm512_cvtbiasps_bf8(__A, __B);
+}
+
+__m128i test_mm512_mask_cvtbiasps_bf8(__m128i __W, __mmask16 __U, __m512i __A, __m512 __B) {
+  // CHECK-LABEL: @test_mm512_mask_cvtbiasps_bf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.512(
+  return _mm512_mask_cvtbiasps_bf8(__W, __U, __A, __B);
+}
+
+__m128i test_mm512_maskz_cvtbiasps_bf8(__mmask16 __U, __m512i __A, __m512 __B) {
+  // CHECK-LABEL: @test_mm512_maskz_cvtbiasps_bf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.512(
+  return _mm512_maskz_cvtbiasps_bf8(__U, __A, __B);
+}
+
+// VCVTBIASPS2BF8S - 128-bit
+
+__m128i test_mm_cvts_biasps_bf8(__m128i __A, __m128 __B) {
+  // CHECK-LABEL: @test_mm_cvts_biasps_bf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.128(
+  return _mm_cvts_biasps_bf8(__A, __B);
+}
+
+__m128i test_mm_mask_cvts_biasps_bf8(__m128i __W, __mmask8 __U, __m128i __A, __m128 __B) {
+  // CHECK-LABEL: @test_mm_mask_cvts_biasps_bf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.128(
+  return _mm_mask_cvts_biasps_bf8(__W, __U, __A, __B);
+}
+
+__m128i test_mm_maskz_cvts_biasps_bf8(__mmask8 __U, __m128i __A, __m128 __B) {
+  // CHECK-LABEL: @test_mm_maskz_cvts_biasps_bf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.128(
+  return _mm_maskz_cvts_biasps_bf8(__U, __A, __B);
+}
+
+// VCVTBIASPS2BF8S - 256-bit
+
+__m128i test_mm256_cvts_biasps_bf8(__m256i __A, __m256 __B) {
+  // CHECK-LABEL: @test_mm256_cvts_biasps_bf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.256(
+  return _mm256_cvts_biasps_bf8(__A, __B);
+}
+
+__m128i test_mm256_mask_cvts_biasps_bf8(__m128i __W, __mmask8 __U, __m256i __A, __m256 __B) {
+  // CHECK-LABEL: @test_mm256_mask_cvts_biasps_bf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.256(
+  return _mm256_mask_cvts_biasps_bf8(__W, __U, __A, __B);
+}
+
+__m128i test_mm256_maskz_cvts_biasps_bf8(__mmask8 __U, __m256i __A, __m256 __B) {
+  // CHECK-LABEL: @test_mm256_maskz_cvts_biasps_bf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.256(
+  return _mm256_maskz_cvts_biasps_bf8(__U, __A, __B);
+}
+
+// VCVTBIASPS2BF8S - 512-bit
+
+__m128i test_mm512_cvts_biasps_bf8(__m512i __A, __m512 __B) {
+  // CHECK-LABEL: @test_mm512_cvts_biasps_bf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.512(
+  return _mm512_cvts_biasps_bf8(__A, __B);
+}
+
+__m128i test_mm512_mask_cvts_biasps_bf8(__m128i __W, __mmask16 __U, __m512i __A, __m512 __B) {
+  // CHECK-LABEL: @test_mm512_mask_cvts_biasps_bf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.512(
+  return _mm512_mask_cvts_biasps_bf8(__W, __U, __A, __B);
+}
+
+__m128i test_mm512_maskz_cvts_biasps_bf8(__mmask16 __U, __m512i __A, __m512 __B) {
+  // CHECK-LABEL: @test_mm512_maskz_cvts_biasps_bf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.512(
+  return _mm512_maskz_cvts_biasps_bf8(__U, __A, __B);
+}
+
+// VCVTBIASPS2HF8 - 128-bit
+
+__m128i test_mm_cvtbiasps_hf8(__m128i __A, __m128 __B) {
+  // CHECK-LABEL: @test_mm_cvtbiasps_hf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.128(
+  return _mm_cvtbiasps_hf8(__A, __B);
+}
+
+__m128i test_mm_mask_cvtbiasps_hf8(__m128i __W, __mmask8 __U, __m128i __A, __m128 __B) {
+  // CHECK-LABEL: @test_mm_mask_cvtbiasps_hf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.128(
+  return _mm_mask_cvtbiasps_hf8(__W, __U, __A, __B);
+}
+
+__m128i test_mm_maskz_cvtbiasps_hf8(__mmask8 __U, __m128i __A, __m128 __B) {
+  // CHECK-LABEL: @test_mm_maskz_cvtbiasps_hf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.128(
+  return _mm_maskz_cvtbiasps_hf8(__U, __A, __B);
+}
+
+// VCVTBIASPS2HF8 - 256-bit
+
+__m128i test_mm256_cvtbiasps_hf8(__m256i __A, __m256 __B) {
+  // CHECK-LABEL: @test_mm256_cvtbiasps_hf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.256(
+  return _mm256_cvtbiasps_hf8(__A, __B);
+}
+
+__m128i test_mm256_mask_cvtbiasps_hf8(__m128i __W, __mmask8 __U, __m256i __A, __m256 __B) {
+  // CHECK-LABEL: @test_mm256_mask_cvtbiasps_hf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.256(
+  return _mm256_mask_cvtbiasps_hf8(__W, __U, __A, __B);
+}
+
+__m128i test_mm256_maskz_cvtbiasps_hf8(__mmask8 __U, __m256i __A, __m256 __B) {
+  // CHECK-LABEL: @test_mm256_maskz_cvtbiasps_hf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.256(
+  return _mm256_maskz_cvtbiasps_hf8(__U, __A, __B);
+}
+
+// VCVTBIASPS2HF8 - 512-bit
+
+__m128i test_mm512_cvtbiasps_hf8(__m512i __A, __m512 __B) {
+  // CHECK-LABEL: @test_mm512_cvtbiasps_hf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.512(
+  return _mm512_cvtbiasps_hf8(__A, __B);
+}
+
+__m128i test_mm512_mask_cvtbiasps_hf8(__m128i __W, __mmask16 __U, __m512i __A, __m512 __B) {
+  // CHECK-LABEL: @test_mm512_mask_cvtbiasps_hf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.512(
+  return _mm512_mask_cvtbiasps_hf8(__W, __U, __A, __B);
+}
+
+__m128i test_mm512_maskz_cvtbiasps_hf8(__mmask16 __U, __m512i __A, __m512 __B) {
+  // CHECK-LABEL: @test_mm512_maskz_cvtbiasps_hf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.512(
+  return _mm512_maskz_cvtbiasps_hf8(__U, __A, __B);
+}
+
+// VCVTBIASPS2HF8S - 128-bit
+
+__m128i test_mm_cvts_biasps_hf8(__m128i __A, __m128 __B) {
+  // CHECK-LABEL: @test_mm_cvts_biasps_hf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.128(
+  return _mm_cvts_biasps_hf8(__A, __B);
+}
+
+__m128i test_mm_mask_cvts_biasps_hf8(__m128i __W, __mmask8 __U, __m128i __A, __m128 __B) {
+  // CHECK-LABEL: @test_mm_mask_cvts_biasps_hf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.128(
+  return _mm_mask_cvts_biasps_hf8(__W, __U, __A, __B);
+}
+
+__m128i test_mm_maskz_cvts_biasps_hf8(__mmask8 __U, __m128i __A, __m128 __B) {
+  // CHECK-LABEL: @test_mm_maskz_cvts_biasps_hf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.128(
+  return _mm_maskz_cvts_biasps_hf8(__U, __A, __B);
+}
+
+// VCVTBIASPS2HF8S - 256-bit
+
+__m128i test_mm256_cvts_biasps_hf8(__m256i __A, __m256 __B) {
+  // CHECK-LABEL: @test_mm256_cvts_biasps_hf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.256(
+  return _mm256_cvts_biasps_hf8(__A, __B);
+}
+
+__m128i test_mm256_mask_cvts_biasps_hf8(__m128i __W, __mmask8 __U, __m256i __A, __m256 __B) {
+  // CHECK-LABEL: @test_mm256_mask_cvts_biasps_hf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.256(
+  return _mm256_mask_cvts_biasps_hf8(__W, __U, __A, __B);
+}
+
+__m128i test_mm256_maskz_cvts_biasps_hf8(__mmask8 __U, __m256i __A, __m256 __B) {
+  // CHECK-LABEL: @test_mm256_maskz_cvts_biasps_hf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.256(
+  return _mm256_maskz_cvts_biasps_hf8(__U, __A, __B);
+}
+
+// VCVTBIASPS2HF8S - 512-bit
+
+__m128i test_mm512_cvts_biasps_hf8(__m512i __A, __m512 __B) {
+  // CHECK-LABEL: @test_mm512_cvts_biasps_hf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.512(
+  return _mm512_cvts_biasps_hf8(__A, __B);
+}
+
+__m128i test_mm512_mask_cvts_biasps_hf8(__m128i __W, __mmask16 __U, __m512i __A, __m512 __B) {
+  // CHECK-LABEL: @test_mm512_mask_cvts_biasps_hf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.512(
+  return _mm512_mask_cvts_biasps_hf8(__W, __U, __A, __B);
+}
+
+__m128i test_mm512_maskz_cvts_biasps_hf8(__mmask16 __U, __m512i __A, __m512 __B) {
+  // CHECK-LABEL: @test_mm512_maskz_cvts_biasps_hf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.512(
+  return _mm512_maskz_cvts_biasps_hf8(__U, __A, __B);
+}
+
+//
+// Group C: VCVTBF82PS / VCVTHF82PS
+//
+
+// VCVTBF82PS - 128-bit
+
+__m128 test_mm_cvtbf8_ps(__m128i __A) {
+  // CHECK-LABEL: @test_mm_cvtbf8_ps(
+  // CHECK: call <4 x float> @llvm.x86.avx10.mask.vcvtbf82ps.128(
+  return _mm_cvtbf8_ps(__A);
+}
+
+__m128 test_mm_mask_cvtbf8_ps(__m128 __W, __mmask8 __U, __m128i __A) {
+  // CHECK-LABEL: @test_mm_mask_cvtbf8_ps(
+  // CHECK: call <4 x float> @llvm.x86.avx10.mask.vcvtbf82ps.128(
+  return _mm_mask_cvtbf8_ps(__W, __U, __A);
+}
+
+__m128 test_mm_maskz_cvtbf8_ps(__mmask8 __U, __m128i __A) {
+  // CHECK-LABEL: @test_mm_maskz_cvtbf8_ps(
+  // CHECK: call <4 x float> @llvm.x86.avx10.mask.vcvtbf82ps.128(
+  return _mm_maskz_cvtbf8_ps(__U, __A);
+}
+
+// VCVTBF82PS - 256-bit
+
+__m256 test_mm256_cvtbf8_ps(__m128i __A) {
+  // CHECK-LABEL: @test_mm256_cvtbf8_ps(
+  // CHECK: call <8 x float> @llvm.x86.avx10.mask.vcvtbf82ps.256(
+  return _mm256_cvtbf8_ps(__A);
+}
+
+__m256 test_mm256_mask_cvtbf8_ps(__m256 __W, __mmask8 __U, __m128i __A) {
+  // CHECK-LABEL: @test_mm256_mask_cvtbf8_ps(
+  // CHECK: call <8 x float> @llvm.x86.avx10.mask.vcvtbf82ps.256(
+  return _mm256_mask_cvtbf8_ps(__W, __U, __A);
+}
+
+__m256 test_mm256_maskz_cvtbf8_ps(__mmask8 __U, __m128i __A) {
+  // CHECK-LABEL: @test_mm256_maskz_cvtbf8_ps(
+  // CHECK: call <8 x float> @llvm.x86.avx10.mask.vcvtbf82ps.256(
+  return _mm256_maskz_cvtbf8_ps(__U, __A);
+}
+
+// VCVTBF82PS - 512-bit
+
+__m512 test_mm512_cvtbf8_ps(__m128i __A) {
+  // CHECK-LABEL: @test_mm512_cvtbf8_ps(
+  // CHECK: call <16 x float> @llvm.x86.avx10.mask.vcvtbf82ps.512(
+  return _mm512_cvtbf8_ps(__A);
+}
+
+__m512 test_mm512_mask_cvtbf8_ps(__m512 __W, __mmask16 __U, __m128i __A) {
+  // CHECK-LABEL: @test_mm512_mask_cvtbf8_ps(
+  // CHECK: call <16 x float> @llvm.x86.avx10.mask.vcvtbf82ps.512(
+  return _mm512_mask_cvtbf8_ps(__W, __U, __A);
+}
+
+__m512 test_mm512_maskz_cvtbf8_ps(__mmask16 __U, __m128i __A) {
+  // CHECK-LABEL: @test_mm512_maskz_cvtbf8_ps(
+  // CHECK: call <16 x float> @llvm.x86.avx10.mask.vcvtbf82ps.512(
+  return _mm512_maskz_cvtbf8_ps(__U, __A);
+}
+
+// VCVTHF82PS - 128-bit
+
+__m128 test_mm_cvthf8_ps(__m128i __A) {
+  // CHECK-LABEL: @test_mm_cvthf8_ps(
+  // CHECK: call <4 x float> @llvm.x86.avx10.mask.vcvthf82ps.128(
+  return _mm_cvthf8_ps(__A);
+}
+
+__m128 test_mm_mask_cvthf8_ps(__m128 __W, __mmask8 __U, __m128i __A) {
+  // CHECK-LABEL: @test_mm_mask_cvthf8_ps(
+  // CHECK: call <4 x float> @llvm.x86.avx10.mask.vcvthf82ps.128(
+  return _mm_mask_cvthf8_ps(__W, __U, __A);
+}
+
+__m128 test_mm_maskz_cvthf8_ps(__mmask8 __U, __m128i __A) {
+  // CHECK-LABEL: @test_mm_maskz_cvthf8_ps(
+  // CHECK: call <4 x float> @llvm.x86.avx10.mask.vcvthf82ps.128(
+  return _mm_maskz_cvthf8_ps(__U, __A);
+}
+
+// VCVTHF82PS - 256-bit
+
+__m256 test_mm256_cvthf8_ps(__m128i __A) {
+  // CHECK-LABEL: @test_mm256_cvthf8_ps(
+  // CHECK: call <8 x float> @llvm.x86.avx10.mask.vcvthf82ps.256(
+  return _mm256_cvthf8_ps(__A);
+}
+
+__m256 test_mm256_mask_cvthf8_ps(__m256 __W, __mmask8 __U, __m128i __A) {
+  // CHECK-LABEL: @test_mm256_mask_cvthf8_ps(
+  // CHECK: call <8 x float> @llvm.x86.avx10.mask.vcvthf82ps.256(
+  return _mm256_mask_cvthf8_ps(__W, __U, __A);
+}
+
+__m256 test_mm256_maskz_cvthf8_ps(__mmask8 __U, __m128i __A) {
+  // CHECK-LABEL: @test_mm256_maskz_cvthf8_ps(
+  // CHECK: call <8 x float> @llvm.x86.avx10.mask.vcvthf82ps.256(
+  return _mm256_maskz_cvthf8_ps(__U, __A);
+}
+
+// VCVTHF82PS - 512-bit
+
+__m512 test_mm512_cvthf8_ps(__m128i __A) {
+  // CHECK-LABEL: @test_mm512_cvthf8_ps(
+  // CHECK: call <16 x float> @llvm.x86.avx10.mask.vcvthf82ps.512(
+  return _mm512_cvthf8_ps(__A);
+}
+
+__m512 test_mm512_mask_cvthf8_ps(__m512 __W, __mmask16 __U, __m128i __A) {
+  // CHECK-LABEL: @test_mm512_mask_cvthf8_ps(
+  // CHECK: call <16 x float> @llvm.x86.avx10.mask.vcvthf82ps.512(
+  return _mm512_mask_cvthf8_ps(__W, __U, __A);
+}
+
+__m512 test_mm512_maskz_cvthf8_ps(__mmask16 __U, __m128i __A) {
+  // CHECK-LABEL: @test_mm512_maskz_cvthf8_ps(
+  // CHECK: call <16 x float> @llvm.x86.avx10.mask.vcvthf82ps.512(
+  return _mm512_maskz_cvthf8_ps(__U, __A);
+}
+
+//
+// Group E: VCVTBF82BF6S / VCVTHF82HF6S
+//
+
+// VCVTBF82BF6S
+
+__m128i test_mm_cvtbf8_bf6s(__m128i __A) {
+  // CHECK-LABEL: @test_mm_cvtbf8_bf6s(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbf82bf6s.128(
+  return _mm_cvtbf8_bf6s(__A);
+}
+
+__m256i test_mm256_cvtbf8_bf6s(__m256i __A) {
+  // CHECK-LABEL: @test_mm256_cvtbf8_bf6s(
+  // CHECK: call <32 x i8> @llvm.x86.avx10.vcvtbf82bf6s.256(
+  return _mm256_cvtbf8_bf6s(__A);
+}
+
+__m512i test_mm512_cvtbf8_bf6s(__m512i __A) {
+  // CHECK-LABEL: @test_mm512_cvtbf8_bf6s(
+  // CHECK: call <64 x i8> @llvm.x86.avx10.vcvtbf82bf6s.512(
+  return _mm512_cvtbf8_bf6s(__A);
+}
+
+// VCVTHF82HF6S
+
+__m128i test_mm_cvthf8_hf6s(__m128i __A) {
+  // CHECK-LABEL: @test_mm_cvthf8_hf6s(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvthf82hf6s.128(
+  return _mm_cvthf8_hf6s(__A);
+}
+
+__m256i test_mm256_cvthf8_hf6s(__m256i __A) {
+  // CHECK-LABEL: @test_mm256_cvthf8_hf6s(
+  // CHECK: call <32 x i8> @llvm.x86.avx10.vcvthf82hf6s.256(
+  return _mm256_cvthf8_hf6s(__A);
+}
+
+__m512i test_mm512_cvthf8_hf6s(__m512i __A) {
+  // CHECK-LABEL: @test_mm512_cvthf8_hf6s(
+  // CHECK: call <64 x i8> @llvm.x86.avx10.vcvthf82hf6s.512(
+  return _mm512_cvthf8_hf6s(__A);
+}
+
+//
+// Group F: VCVTBF42HF8 / VCVTBF62HF8 / VCVTHF62HF8
+//
+
+// VCVTBF42HF8 - 128-bit
+
+__m128i test_mm_cvtbf4_hf8(__m128i __A) {
+  // CHECK-LABEL: @test_mm_cvtbf4_hf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbf42hf8.128(
+  return _mm_cvtbf4_hf8(__A);
+}
+
+__m128i test_mm_mask_cvtbf4_hf8(__m128i __W, __mmask16 __U, __m128i __A) {
+  // CHECK-LABEL: @test_mm_mask_cvtbf4_hf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbf42hf8.128(
+  // CHECK: select <16 x i1>
+  return _mm_mask_cvtbf4_hf8(__W, __U, __A);
+}
+
+__m128i test_mm_maskz_cvtbf4_hf8(__mmask16 __U, __m128i __A) {
+  // CHECK-LABEL: @test_mm_maskz_cvtbf4_hf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbf42hf8.128(
+  // CHECK: select <16 x i1>
+  return _mm_maskz_cvtbf4_hf8(__U, __A);
+}
+
+// VCVTBF42HF8 - 256-bit
+
+__m256i test_mm256_cvtbf4_hf8(__m128i __A) {
+  // CHECK-LABEL: @test_mm256_cvtbf4_hf8(
+  // CHECK: call <32 x i8> @llvm.x86.avx10.vcvtbf42hf8.256(
+  return _mm256_cvtbf4_hf8(__A);
+}
+
+__m256i test_mm256_mask_cvtbf4_hf8(__m256i __W, __mmask32 __U, __m128i __A) {
+  // CHECK-LABEL: @test_mm256_mask_cvtbf4_hf8(
+  // CHECK: call <32 x i8> @llvm.x86.avx10.vcvtbf42hf8.256(
+  // CHECK: select <32 x i1>
+  return _mm256_mask_cvtbf4_hf8(__W, __U, __A);
+}
+
+__m256i test_mm256_maskz_cvtbf4_hf8(__mmask32 __U, __m128i __A) {
+  // CHECK-LABEL: @test_mm256_maskz_cvtbf4_hf8(
+  // CHECK: call <32 x i8> @llvm.x86.avx10.vcvtbf42hf8.256(
+  // CHECK: select <32 x i1>
+  return _mm256_maskz_cvtbf4_hf8(__U, __A);
+}
+
+// VCVTBF42HF8 - 512-bit
+
+__m512i test_mm512_cvtbf4_hf8(__m256i __A) {
+  // CHECK-LABEL: @test_mm512_cvtbf4_hf8(
+  // CHECK: call <64 x i8> @llvm.x86.avx10.vcvtbf42hf8.512(
+  return _mm512_cvtbf4_hf8(__A);
+}
+
+__m512i test_mm512_mask_cvtbf4_hf8(__m512i __W, __mmask64 __U, __m256i __A) {
+  // CHECK-LABEL: @test_mm512_mask_cvtbf4_hf8(
+  // CHECK: call <64 x i8> @llvm.x86.avx10.vcvtbf42hf8.512(
+  // CHECK: select <64 x i1>
+  return _mm512_mask_cvtbf4_hf8(__W, __U, __A);
+}
+
+__m512i test_mm512_maskz_cvtbf4_hf8(__mmask64 __U, __m256i __A) {
+  // CHECK-LABEL: @test_mm512_maskz_cvtbf4_hf8(
+  // CHECK: call <64 x i8> @llvm.x86.avx10.vcvtbf42hf8.512(
+  // CHECK: select <64 x i1>
+  return _mm512_maskz_cvtbf4_hf8(__U, __A);
+}
+
+// VCVTBF62HF8 - 128-bit
+
+__m128i test_mm_cvtbf6_hf8(__m128i __A) {
+  // CHECK-LABEL: @test_mm_cvtbf6_hf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbf62hf8.128(
+  return _mm_cvtbf6_hf8(__A);
+}
+
+__m128i test_mm_mask_cvtbf6_hf8(__m128i __W, __mmask16 __U, __m128i __A) {
+  // CHECK-LABEL: @test_mm_mask_cvtbf6_hf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbf62hf8.128(
+  // CHECK: select <16 x i1>
+  return _mm_mask_cvtbf6_hf8(__W, __U, __A);
+}
+
+__m128i test_mm_maskz_cvtbf6_hf8(__mmask16 __U, __m128i __A) {
+  // CHECK-LABEL: @test_mm_maskz_cvtbf6_hf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbf62hf8.128(
+  // CHECK: select <16 x i1>
+  return _mm_maskz_cvtbf6_hf8(__U, __A);
+}
+
+// VCVTBF62HF8 - 256-bit
+
+__m256i test_mm256_cvtbf6_hf8(__m256i __A) {
+  // CHECK-LABEL: @test_mm256_cvtbf6_hf8(
+  // CHECK: call <32 x i8> @llvm.x86.avx10.vcvtbf62hf8.256(
+  return _mm256_cvtbf6_hf8(__A);
+}
+
+__m256i test_mm256_mask_cvtbf6_hf8(__m256i __W, __mmask32 __U, __m256i __A) {
+  // CHECK-LABEL: @test_mm256_mask_cvtbf6_hf8(
+  // CHECK: call <32 x i8> @llvm.x86.avx10.vcvtbf62hf8.256(
+  // CHECK: select <32 x i1>
+  return _mm256_mask_cvtbf6_hf8(__W, __U, __A);
+}
+
+__m256i test_mm256_maskz_cvtbf6_hf8(__mmask32 __U, __m256i __A) {
+  // CHECK-LABEL: @test_mm256_maskz_cvtbf6_hf8(
+  // CHECK: call <32 x i8> @llvm.x86.avx10.vcvtbf62hf8.256(
+  // CHECK: select <32 x i1>
+  return _mm256_maskz_cvtbf6_hf8(__U, __A);
+}
+
+// VCVTBF62HF8 - 512-bit
+
+__m512i test_mm512_cvtbf6_hf8(__m512i __A) {
+  // CHECK-LABEL: @test_mm512_cvtbf6_hf8(
+  // CHECK: call <64 x i8> @llvm.x86.avx10.vcvtbf62hf8.512(
+  return _mm512_cvtbf6_hf8(__A);
+}
+
+__m512i test_mm512_mask_cvtbf6_hf8(__m512i __W, __mmask64 __U, __m512i __A) {
+  // CHECK-LABEL: @test_mm512_mask_cvtbf6_hf8(
+  // CHECK: call <64 x i8> @llvm.x86.avx10.vcvtbf62hf8.512(
+  // CHECK: select <64 x i1>
+  return _mm512_mask_cvtbf6_hf8(__W, __U, __A);
+}
+
+__m512i test_mm512_maskz_cvtbf6_hf8(__mmask64 __U, __m512i __A) {
+  // CHECK-LABEL: @test_mm512_maskz_cvtbf6_hf8(
+  // CHECK: call <64 x i8> @llvm.x86.avx10.vcvtbf62hf8.512(
+  // CHECK: select <64 x i1>
+  return _mm512_maskz_cvtbf6_hf8(__U, __A);
+}
+
+// VCVTHF62HF8 - 128-bit
+
+__m128i test_mm_cvthf6_hf8(__m128i __A) {
+  // CHECK-LABEL: @test_mm_cvthf6_hf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvthf62hf8.128(
+  return _mm_cvthf6_hf8(__A);
+}
+
+__m128i test_mm_mask_cvthf6_hf8(__m128i __W, __mmask16 __U, __m128i __A) {
+  // CHECK-LABEL: @test_mm_mask_cvthf6_hf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvthf62hf8.128(
+  // CHECK: select <16 x i1>
+  return _mm_mask_cvthf6_hf8(__W, __U, __A);
+}
+
+__m128i test_mm_maskz_cvthf6_hf8(__mmask16 __U, __m128i __A) {
+  // CHECK-LABEL: @test_mm_maskz_cvthf6_hf8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvthf62hf8.128(
+  // CHECK: select <16 x i1>
+  return _mm_maskz_cvthf6_hf8(__U, __A);
+}
+
+// VCVTHF62HF8 - 256-bit
+
+__m256i test_mm256_cvthf6_hf8(__m256i __A) {
+  // CHECK-LABEL: @test_mm256_cvthf6_hf8(
+  // CHECK: call <32 x i8> @llvm.x86.avx10.vcvthf62hf8.256(
+  return _mm256_cvthf6_hf8(__A);
+}
+
+__m256i test_mm256_mask_cvthf6_hf8(__m256i __W, __mmask32 __U, __m256i __A) {
+  // CHECK-LABEL: @test_mm256_mask_cvthf6_hf8(
+  // CHECK: call <32 x i8> @llvm.x86.avx10.vcvthf62hf8.256(
+  // CHECK: select <32 x i1>
+  return _mm256_mask_cvthf6_hf8(__W, __U, __A);
+}
+
+__m256i test_mm256_maskz_cvthf6_hf8(__mmask32 __U, __m256i __A) {
+  // CHECK-LABEL: @test_mm256_maskz_cvthf6_hf8(
+  // CHECK: call <32 x i8> @llvm.x86.avx10.vcvthf62hf8.256(
+  // CHECK: select <32 x i1>
+  return _mm256_maskz_cvthf6_hf8(__U, __A);
+}
+
+// VCVTHF62HF8 - 512-bit
+
+__m512i test_mm512_cvthf6_hf8(__m512i __A) {
+  // CHECK-LABEL: @test_mm512_cvthf6_hf8(
+  // CHECK: call <64 x i8> @llvm.x86.avx10.vcvthf62hf8.512(
+  return _mm512_cvthf6_hf8(__A);
+}
+
+__m512i test_mm512_mask_cvthf6_hf8(__m512i __W, __mmask64 __U, __m512i __A) {
+  // CHECK-LABEL: @test_mm512_mask_cvthf6_hf8(
+  // CHECK: call <64 x i8> @llvm.x86.avx10.vcvthf62hf8.512(
+  // CHECK: select <64 x i1>
+  return _mm512_mask_cvthf6_hf8(__W, __U, __A);
+}
+
+__m512i test_mm512_maskz_cvthf6_hf8(__mmask64 __U, __m512i __A) {
+  // CHECK-LABEL: @test_mm512_maskz_cvthf6_hf8(
+  // CHECK: call <64 x i8> @llvm.x86.avx10.vcvthf62hf8.512(
+  // CHECK: select <64 x i1>
+  return _mm512_maskz_cvthf6_hf8(__U, __A);
+}
+
+//
+// Group H: VUNPACKB
+//
+
+// VUNPACKB - 128-bit
+
+__m128i test_mm_unpackb_epi8(__m128i __A) {
+  // CHECK-LABEL: @test_mm_unpackb_epi8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vunpackb.128(
+  return _mm_unpackb_epi8(__A, 1);
+}
+
+__m128i test_mm_mask_unpackb_epi8(__m128i __W, __mmask16 __U, __m128i __A) {
+  // CHECK-LABEL: @test_mm_mask_unpackb_epi8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vunpackb.128(
+  // CHECK: select <16 x i1>
+  return _mm_mask_unpackb_epi8(__W, __U, __A, 1);
+}
+
+__m128i test_mm_maskz_unpackb_epi8(__mmask16 __U, __m128i __A) {
+  // CHECK-LABEL: @test_mm_maskz_unpackb_epi8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vunpackb.128(
+  // CHECK: select <16 x i1>
+  return _mm_maskz_unpackb_epi8(__U, __A, 1);
+}
+
+// VUNPACKB - 256-bit
+
+__m256i test_mm256_unpackb_epi8(__m256i __A) {
+  // CHECK-LABEL: @test_mm256_unpackb_epi8(
+  // CHECK: call <32 x i8> @llvm.x86.avx10.vunpackb.256(
+  return _mm256_unpackb_epi8(__A, 2);
+}
+
+__m256i test_mm256_mask_unpackb_epi8(__m256i __W, __mmask32 __U, __m256i __A) {
+  // CHECK-LABEL: @test_mm256_mask_unpackb_epi8(
+  // CHECK: call <32 x i8> @llvm.x86.avx10.vunpackb.256(
+  // CHECK: select <32 x i1>
+  return _mm256_mask_unpackb_epi8(__W, __U, __A, 2);
+}
+
+__m256i test_mm256_maskz_unpackb_epi8(__mmask32 __U, __m256i __A) {
+  // CHECK-LABEL: @test_mm256_maskz_unpackb_epi8(
+  // CHECK: call <32 x i8> @llvm.x86.avx10.vunpackb.256(
+  // CHECK: select <32 x i1>
+  return _mm256_maskz_unpackb_epi8(__U, __A, 2);
+}
+
+// VUNPACKB - 512-bit
+
+__m512i test_mm512_unpackb_epi8(__m512i __A) {
+  // CHECK-LABEL: @test_mm512_unpackb_epi8(
+  // CHECK: call <64 x i8> @llvm.x86.avx10.vunpackb.512(
+  return _mm512_unpackb_epi8(__A, 3);
+}
+
+__m512i test_mm512_mask_unpackb_epi8(__m512i __W, __mmask64 __U, __m512i __A) {
+  // CHECK-LABEL: @test_mm512_mask_unpackb_epi8(
+  // CHECK: call <64 x i8> @llvm.x86.avx10.vunpackb.512(
+  // CHECK: select <64 x i1>
+  return _mm512_mask_unpackb_epi8(__W, __U, __A, 3);
+}
+
+__m512i test_mm512_maskz_unpackb_epi8(__mmask64 __U, __m512i __A) {
+  // CHECK-LABEL: @test_mm512_maskz_unpackb_epi8(
+  // CHECK: call <64 x i8> @llvm.x86.avx10.vunpackb.512(
+  // CHECK: select <64 x i1>
+  return _mm512_maskz_unpackb_epi8(__U, __A, 3);
+}
diff --git a/clang/test/CodeGen/attr-target-x86.c b/clang/test/CodeGen/attr-target-x86.c
index 474fa93629d897..3105517711e5e9 100644
--- a/clang/test/CodeGen/attr-target-x86.c
+++ b/clang/test/CodeGen/attr-target-x86.c
@@ -33,7 +33,7 @@ __attribute__((target("fpmath=387")))
 void f_fpmath_387(void) {}
 
 // CHECK-NOT: tune-cpu
-// CHECK: [[f_no_sse2]] = {{.*}}"target-cpu"="i686" "target-features"="+cmov,+cx8,+x87,-aes,-amx-avx512,-avx,-avx10.1,-avx10.2,-avx2,-avx512bf16,-avx512bitalg,-avx512bw,-avx512cd,-avx512dq,-avx512f,-avx512fp16,-avx512ifma,-avx512vbmi,-avx512vbmi2,-avx512vl,-avx512vnni,-avx512vp2intersect,-avx512vpopcntdq,-avxifma,-avxneconvert,-avxvnni,-avxvnniint16,-avxvnniint8,-f16c,-fma,-fma4,-gfni,-kl,-pclmul,-sha,-sha512,-sm3,-sm4,-sse2,-sse3,-sse4.1,-sse4.2,-sse4a,-ssse3,-vaes,-vpclmulqdq,-widekl,-xop" "tune-cpu"="i686"
+// CHECK: [[f_no_sse2]] = {{.*}}"target-cpu"="i686" "target-features"="+cmov,+cx8,+x87,-aes,-amx-avx512,-avx,-avx10-v2-aux,-avx10.1,-avx10.2,-avx2,-avx512bf16,-avx512bitalg,-avx512bw,-avx512cd,-avx512dq,-avx512f,-avx512fp16,-avx512ifma,-avx512vbmi,-avx512vbmi2,-avx512vl,-avx512vnni,-avx512vp2intersect,-avx512vpopcntdq,-avxifma,-avxneconvert,-avxvnni,-avxvnniint16,-avxvnniint8,-f16c,-fma,-fma4,-gfni,-kl,-pclmul,-sha,-sha512,-sm3,-sm4,-sse2,-sse3,-sse4.1,-sse4.2,-sse4a,-ssse3,-vaes,-vpclmulqdq,-widekl,-xop" "tune-cpu"="i686"
 __attribute__((target("no-sse2")))
 void f_no_sse2(void) {}
 
@@ -41,7 +41,7 @@ void f_no_sse2(void) {}
 __attribute__((target("sse4")))
 void f_sse4(void) {}
 
-// CHECK: [[f_no_sse4]] = {{.*}}"target-cpu"="i686" "target-features"="+cmov,+cx8,+x87,-amx-avx512,-avx,-avx10.1,-avx10.2,-avx2,-avx512bf16,-avx512bitalg,-avx512bw,-avx512cd,-avx512dq,-avx512f,-avx512fp16,-avx512ifma,-avx512vbmi,-avx512vbmi2,-avx512vl,-avx512vnni,-avx512vp2intersect,-avx512vpopcntdq,-avxifma,-avxneconvert,-avxvnni,-avxvnniint16,-avxvnniint8,-f16c,-fma,-fma4,-sha512,-sm3,-sm4,-sse4.1,-sse4.2,-vaes,-vpclmulqdq,-xop" "tune-cpu"="i686"
+// CHECK: [[f_no_sse4]] = {{.*}}"target-cpu"="i686" "target-features"="+cmov,+cx8,+x87,-amx-avx512,-avx,-avx10-v2-aux,-avx10.1,-avx10.2,-avx2,-avx512bf16,-avx512bitalg,-avx512bw,-avx512cd,-avx512dq,-avx512f,-avx512fp16,-avx512ifma,-avx512vbmi,-avx512vbmi2,-avx512vl,-avx512vnni,-avx512vp2intersect,-avx512vpopcntdq,-avxifma,-avxneconvert,-avxvnni,-avxvnniint16,-avxvnniint8,-f16c,-fma,-fma4,-sha512,-sm3,-sm4,-sse4.1,-sse4.2,-vaes,-vpclmulqdq,-xop" "tune-cpu"="i686"
 __attribute__((target("no-sse4")))
 void f_no_sse4(void) {}
 
diff --git a/llvm/include/llvm/IR/IntrinsicsX86.td b/llvm/include/llvm/IR/IntrinsicsX86.td
index 5c7785731111cf..b0bbd6f35359e1 100644
--- a/llvm/include/llvm/IR/IntrinsicsX86.td
+++ b/llvm/include/llvm/IR/IntrinsicsX86.td
@@ -7022,6 +7022,203 @@ def int_x86_avx10_mask_vcvtph2hf8s512 : ClangBuiltin<"__builtin_ia32_vcvtph2hf8s
                               [IntrNoMem]>;
 }
 
+// AVX10 V2 AUX - Convert instructions
+let TargetPrefix = "x86" in {
+// Group A: PS(f32) -> i8 truncating conversions (quarter-size: output always v16i8)
+
+// VCVTPS2BF8
+def int_x86_avx10_mask_vcvtps2bf8_128 : ClangBuiltin<"__builtin_ia32_vcvtps2bf8_128_mask">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v4f32_ty, llvm_v16i8_ty, llvm_i8_ty],
+                              [IntrNoMem]>;
+def int_x86_avx10_mask_vcvtps2bf8_256 : ClangBuiltin<"__builtin_ia32_vcvtps2bf8_256_mask">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v8f32_ty, llvm_v16i8_ty, llvm_i8_ty],
+                              [IntrNoMem]>;
+def int_x86_avx10_mask_vcvtps2bf8_512 : ClangBuiltin<"__builtin_ia32_vcvtps2bf8_512_mask">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16f32_ty, llvm_v16i8_ty, llvm_i16_ty],
+                              [IntrNoMem]>;
+
+// VCVTPS2BF8S
+def int_x86_avx10_mask_vcvtps2bf8s_128 : ClangBuiltin<"__builtin_ia32_vcvtps2bf8s_128_mask">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v4f32_ty, llvm_v16i8_ty, llvm_i8_ty],
+                              [IntrNoMem]>;
+def int_x86_avx10_mask_vcvtps2bf8s_256 : ClangBuiltin<"__builtin_ia32_vcvtps2bf8s_256_mask">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v8f32_ty, llvm_v16i8_ty, llvm_i8_ty],
+                              [IntrNoMem]>;
+def int_x86_avx10_mask_vcvtps2bf8s_512 : ClangBuiltin<"__builtin_ia32_vcvtps2bf8s_512_mask">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16f32_ty, llvm_v16i8_ty, llvm_i16_ty],
+                              [IntrNoMem]>;
+
+// VCVTPS2HF8
+def int_x86_avx10_mask_vcvtps2hf8_128 : ClangBuiltin<"__builtin_ia32_vcvtps2hf8_128_mask">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v4f32_ty, llvm_v16i8_ty, llvm_i8_ty],
+                              [IntrNoMem]>;
+def int_x86_avx10_mask_vcvtps2hf8_256 : ClangBuiltin<"__builtin_ia32_vcvtps2hf8_256_mask">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v8f32_ty, llvm_v16i8_ty, llvm_i8_ty],
+                              [IntrNoMem]>;
+def int_x86_avx10_mask_vcvtps2hf8_512 : ClangBuiltin<"__builtin_ia32_vcvtps2hf8_512_mask">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16f32_ty, llvm_v16i8_ty, llvm_i16_ty],
+                              [IntrNoMem]>;
+
+// VCVTPS2HF8S
+def int_x86_avx10_mask_vcvtps2hf8s_128 : ClangBuiltin<"__builtin_ia32_vcvtps2hf8s_128_mask">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v4f32_ty, llvm_v16i8_ty, llvm_i8_ty],
+                              [IntrNoMem]>;
+def int_x86_avx10_mask_vcvtps2hf8s_256 : ClangBuiltin<"__builtin_ia32_vcvtps2hf8s_256_mask">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v8f32_ty, llvm_v16i8_ty, llvm_i8_ty],
+                              [IntrNoMem]>;
+def int_x86_avx10_mask_vcvtps2hf8s_512 : ClangBuiltin<"__builtin_ia32_vcvtps2hf8s_512_mask">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16f32_ty, llvm_v16i8_ty, llvm_i16_ty],
+                              [IntrNoMem]>;
+
+// VCVTROPS2HF8
+def int_x86_avx10_mask_vcvtrops2hf8_128 : ClangBuiltin<"__builtin_ia32_vcvtrops2hf8_128_mask">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v4f32_ty, llvm_v16i8_ty, llvm_i8_ty],
+                              [IntrNoMem]>;
+def int_x86_avx10_mask_vcvtrops2hf8_256 : ClangBuiltin<"__builtin_ia32_vcvtrops2hf8_256_mask">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v8f32_ty, llvm_v16i8_ty, llvm_i8_ty],
+                              [IntrNoMem]>;
+def int_x86_avx10_mask_vcvtrops2hf8_512 : ClangBuiltin<"__builtin_ia32_vcvtrops2hf8_512_mask">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16f32_ty, llvm_v16i8_ty, llvm_i16_ty],
+                              [IntrNoMem]>;
+
+// VCVTROPS2HF8S
+def int_x86_avx10_mask_vcvtrops2hf8s_128 : ClangBuiltin<"__builtin_ia32_vcvtrops2hf8s_128_mask">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v4f32_ty, llvm_v16i8_ty, llvm_i8_ty],
+                              [IntrNoMem]>;
+def int_x86_avx10_mask_vcvtrops2hf8s_256 : ClangBuiltin<"__builtin_ia32_vcvtrops2hf8s_256_mask">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v8f32_ty, llvm_v16i8_ty, llvm_i8_ty],
+                              [IntrNoMem]>;
+def int_x86_avx10_mask_vcvtrops2hf8s_512 : ClangBuiltin<"__builtin_ia32_vcvtrops2hf8s_512_mask">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16f32_ty, llvm_v16i8_ty, llvm_i16_ty],
+                              [IntrNoMem]>;
+
+// Group B: Bias PS -> i8 conversions (3-operand: bias + f32 source -> i8 dest)
+
+// VCVTBIASPS2BF8
+def int_x86_avx10_mask_vcvtbiasps2bf8_128 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2bf8_128_mask">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16i8_ty, llvm_v4f32_ty, llvm_v16i8_ty, llvm_i8_ty],
+                              [IntrNoMem]>;
+def int_x86_avx10_mask_vcvtbiasps2bf8_256 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2bf8_256_mask">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v32i8_ty, llvm_v8f32_ty, llvm_v16i8_ty, llvm_i8_ty],
+                              [IntrNoMem]>;
+def int_x86_avx10_mask_vcvtbiasps2bf8_512 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2bf8_512_mask">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v64i8_ty, llvm_v16f32_ty, llvm_v16i8_ty, llvm_i16_ty],
+                              [IntrNoMem]>;
+
+// VCVTBIASPS2BF8S
+def int_x86_avx10_mask_vcvtbiasps2bf8s_128 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2bf8s_128_mask">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16i8_ty, llvm_v4f32_ty, llvm_v16i8_ty, llvm_i8_ty],
+                              [IntrNoMem]>;
+def int_x86_avx10_mask_vcvtbiasps2bf8s_256 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2bf8s_256_mask">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v32i8_ty, llvm_v8f32_ty, llvm_v16i8_ty, llvm_i8_ty],
+                              [IntrNoMem]>;
+def int_x86_avx10_mask_vcvtbiasps2bf8s_512 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2bf8s_512_mask">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v64i8_ty, llvm_v16f32_ty, llvm_v16i8_ty, llvm_i16_ty],
+                              [IntrNoMem]>;
+
+// VCVTBIASPS2HF8
+def int_x86_avx10_mask_vcvtbiasps2hf8_128 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2hf8_128_mask">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16i8_ty, llvm_v4f32_ty, llvm_v16i8_ty, llvm_i8_ty],
+                              [IntrNoMem]>;
+def int_x86_avx10_mask_vcvtbiasps2hf8_256 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2hf8_256_mask">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v32i8_ty, llvm_v8f32_ty, llvm_v16i8_ty, llvm_i8_ty],
+                              [IntrNoMem]>;
+def int_x86_avx10_mask_vcvtbiasps2hf8_512 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2hf8_512_mask">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v64i8_ty, llvm_v16f32_ty, llvm_v16i8_ty, llvm_i16_ty],
+                              [IntrNoMem]>;
+
+// VCVTBIASPS2HF8S
+def int_x86_avx10_mask_vcvtbiasps2hf8s_128 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2hf8s_128_mask">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16i8_ty, llvm_v4f32_ty, llvm_v16i8_ty, llvm_i8_ty],
+                              [IntrNoMem]>;
+def int_x86_avx10_mask_vcvtbiasps2hf8s_256 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2hf8s_256_mask">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v32i8_ty, llvm_v8f32_ty, llvm_v16i8_ty, llvm_i8_ty],
+                              [IntrNoMem]>;
+def int_x86_avx10_mask_vcvtbiasps2hf8s_512 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2hf8s_512_mask">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v64i8_ty, llvm_v16f32_ty, llvm_v16i8_ty, llvm_i16_ty],
+                              [IntrNoMem]>;
+
+// Group C: 8bit -> PS expanding conversions
+
+// VCVTBF82PS
+def int_x86_avx10_mask_vcvtbf82ps_128 : ClangBuiltin<"__builtin_ia32_vcvtbf8_2ps128_mask">,
+        DefaultAttrsIntrinsic<[llvm_v4f32_ty], [llvm_v16i8_ty, llvm_v4f32_ty, llvm_i8_ty],
+                              [IntrNoMem]>;
+def int_x86_avx10_mask_vcvtbf82ps_256 : ClangBuiltin<"__builtin_ia32_vcvtbf8_2ps256_mask">,
+        DefaultAttrsIntrinsic<[llvm_v8f32_ty], [llvm_v16i8_ty, llvm_v8f32_ty, llvm_i8_ty],
+                              [IntrNoMem]>;
+def int_x86_avx10_mask_vcvtbf82ps_512 : ClangBuiltin<"__builtin_ia32_vcvtbf8_2ps512_mask">,
+        DefaultAttrsIntrinsic<[llvm_v16f32_ty], [llvm_v16i8_ty, llvm_v16f32_ty, llvm_i16_ty],
+                              [IntrNoMem]>;
+
+// VCVTHF82PS
+def int_x86_avx10_mask_vcvthf82ps_128 : ClangBuiltin<"__builtin_ia32_vcvthf8_2ps128_mask">,
+        DefaultAttrsIntrinsic<[llvm_v4f32_ty], [llvm_v16i8_ty, llvm_v4f32_ty, llvm_i8_ty],
+                              [IntrNoMem]>;
+def int_x86_avx10_mask_vcvthf82ps_256 : ClangBuiltin<"__builtin_ia32_vcvthf8_2ps256_mask">,
+        DefaultAttrsIntrinsic<[llvm_v8f32_ty], [llvm_v16i8_ty, llvm_v8f32_ty, llvm_i8_ty],
+                              [IntrNoMem]>;
+def int_x86_avx10_mask_vcvthf82ps_512 : ClangBuiltin<"__builtin_ia32_vcvthf8_2ps512_mask">,
+        DefaultAttrsIntrinsic<[llvm_v16f32_ty], [llvm_v16i8_ty, llvm_v16f32_ty, llvm_i16_ty],
+                              [IntrNoMem]>;
+
+// Group E: Same-size reg-only conversions (no masking)
+
+// VCVTBF82BF6S
+def int_x86_avx10_vcvtbf82bf6s_128 : ClangBuiltin<"__builtin_ia32_vcvtbf82bf6s128">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16i8_ty], [IntrNoMem]>;
+def int_x86_avx10_vcvtbf82bf6s_256 : ClangBuiltin<"__builtin_ia32_vcvtbf82bf6s256">,
+        DefaultAttrsIntrinsic<[llvm_v32i8_ty], [llvm_v32i8_ty], [IntrNoMem]>;
+def int_x86_avx10_vcvtbf82bf6s_512 : ClangBuiltin<"__builtin_ia32_vcvtbf82bf6s512">,
+        DefaultAttrsIntrinsic<[llvm_v64i8_ty], [llvm_v64i8_ty], [IntrNoMem]>;
+
+// VCVTHF82HF6S
+def int_x86_avx10_vcvthf82hf6s_128 : ClangBuiltin<"__builtin_ia32_vcvthf82hf6s128">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16i8_ty], [IntrNoMem]>;
+def int_x86_avx10_vcvthf82hf6s_256 : ClangBuiltin<"__builtin_ia32_vcvthf82hf6s256">,
+        DefaultAttrsIntrinsic<[llvm_v32i8_ty], [llvm_v32i8_ty], [IntrNoMem]>;
+def int_x86_avx10_vcvthf82hf6s_512 : ClangBuiltin<"__builtin_ia32_vcvthf82hf6s512">,
+        DefaultAttrsIntrinsic<[llvm_v64i8_ty], [llvm_v64i8_ty], [IntrNoMem]>;
+
+// Group F: Expanding/same-size conversions with masking
+
+// VCVTBF42HF8: expanding (half-size input -> full output)
+def int_x86_avx10_vcvtbf42hf8_128 : ClangBuiltin<"__builtin_ia32_vcvtbf42hf8128">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16i8_ty], [IntrNoMem]>;
+def int_x86_avx10_vcvtbf42hf8_256 : ClangBuiltin<"__builtin_ia32_vcvtbf42hf8256">,
+        DefaultAttrsIntrinsic<[llvm_v32i8_ty], [llvm_v16i8_ty], [IntrNoMem]>;
+def int_x86_avx10_vcvtbf42hf8_512 : ClangBuiltin<"__builtin_ia32_vcvtbf42hf8512">,
+        DefaultAttrsIntrinsic<[llvm_v64i8_ty], [llvm_v32i8_ty], [IntrNoMem]>;
+
+// VCVTBF62HF8: same-size, reg-only
+def int_x86_avx10_vcvtbf62hf8_128 : ClangBuiltin<"__builtin_ia32_vcvtbf62hf8128">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16i8_ty], [IntrNoMem]>;
+def int_x86_avx10_vcvtbf62hf8_256 : ClangBuiltin<"__builtin_ia32_vcvtbf62hf8256">,
+        DefaultAttrsIntrinsic<[llvm_v32i8_ty], [llvm_v32i8_ty], [IntrNoMem]>;
+def int_x86_avx10_vcvtbf62hf8_512 : ClangBuiltin<"__builtin_ia32_vcvtbf62hf8512">,
+        DefaultAttrsIntrinsic<[llvm_v64i8_ty], [llvm_v64i8_ty], [IntrNoMem]>;
+
+// VCVTHF62HF8: same-size, reg-only
+def int_x86_avx10_vcvthf62hf8_128 : ClangBuiltin<"__builtin_ia32_vcvthf62hf8128">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16i8_ty], [IntrNoMem]>;
+def int_x86_avx10_vcvthf62hf8_256 : ClangBuiltin<"__builtin_ia32_vcvthf62hf8256">,
+        DefaultAttrsIntrinsic<[llvm_v32i8_ty], [llvm_v32i8_ty], [IntrNoMem]>;
+def int_x86_avx10_vcvthf62hf8_512 : ClangBuiltin<"__builtin_ia32_vcvthf62hf8512">,
+        DefaultAttrsIntrinsic<[llvm_v64i8_ty], [llvm_v64i8_ty], [IntrNoMem]>;
+
+// Group H: VUNPACKB - Byte unpack with immediate
+
+def int_x86_avx10_vunpackb_128 : ClangBuiltin<"__builtin_ia32_vunpackb128">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16i8_ty, llvm_i8_ty],
+                              [IntrNoMem, ImmArg<ArgIndex<1>>]>;
+def int_x86_avx10_vunpackb_256 : ClangBuiltin<"__builtin_ia32_vunpackb256">,
+        DefaultAttrsIntrinsic<[llvm_v32i8_ty], [llvm_v32i8_ty, llvm_i8_ty],
+                              [IntrNoMem, ImmArg<ArgIndex<1>>]>;
+def int_x86_avx10_vunpackb_512 : ClangBuiltin<"__builtin_ia32_vunpackb512">,
+        DefaultAttrsIntrinsic<[llvm_v64i8_ty], [llvm_v64i8_ty, llvm_i8_ty],
+                              [IntrNoMem, ImmArg<ArgIndex<1>>]>;
+}
+
 //===----------------------------------------------------------------------===//
 let TargetPrefix = "x86" in {
 def int_x86_avx10_vaddbf16512 : ClangBuiltin<"__builtin_ia32_vaddbf16512">,
diff --git a/llvm/include/llvm/TargetParser/X86TargetParser.def b/llvm/include/llvm/TargetParser/X86TargetParser.def
index bf1b6c894d9591..916f90a4bfd750 100644
--- a/llvm/include/llvm/TargetParser/X86TargetParser.def
+++ b/llvm/include/llvm/TargetParser/X86TargetParser.def
@@ -274,6 +274,7 @@ X86_FEATURE       (NDD,                "ndd")
 X86_FEATURE       (EGPR,               "egpr")
 X86_FEATURE       (ZU,                 "zu")
 X86_FEATURE       (JMPABS,             "jmpabs")
+X86_FEATURE       (AVX10_V2_AUX,       "avx10-v2-aux")
 
 // These features aren't really CPU features, but the frontend can set them.
 X86_FEATURE       (RETPOLINE_EXTERNAL_THUNK,    "retpoline-external-thunk")
diff --git a/llvm/lib/Target/X86/X86.td b/llvm/lib/Target/X86/X86.td
index 5797171fef23de..443f5bf3ff3729 100644
--- a/llvm/lib/Target/X86/X86.td
+++ b/llvm/lib/Target/X86/X86.td
@@ -361,6 +361,9 @@ def FeatureAVX10_2 : SubtargetFeature<"avx10.2", "HasAVX10_2", "true",
 def FeatureAVX10_2_512 : SubtargetFeature<"avx10.2-512", "HasAVX10_2_512", "true",
                                           "Support AVX10.2 instruction",
                                           [FeatureAVX10_2]>;
+def FeatureAVX10_V2_AUX : SubtargetFeature<"avx10-v2-aux", "HasAVX10_V2_AUX", "true",
+                                          "Support AVX10 V2 AUX instructions",
+                                          [FeatureAVX10_2]>;
 def FeatureEGPR : SubtargetFeature<"egpr", "HasEGPR", "true",
                                    "Support extended general purpose register">;
 def FeaturePush2Pop2 : SubtargetFeature<"push2pop2", "HasPush2Pop2", "true",
diff --git a/llvm/lib/Target/X86/X86InstrAVX10_V2_AUX.td b/llvm/lib/Target/X86/X86InstrAVX10_V2_AUX.td
new file mode 100644
index 00000000000000..29777281440cef
--- /dev/null
+++ b/llvm/lib/Target/X86/X86InstrAVX10_V2_AUX.td
@@ -0,0 +1,588 @@
+//===-- X86InstrAVX10_V2_AUX.td - AVX10 V2 AUX Instructions --*- tablegen -*-===//
+//
+// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
+// See https://llvm.org/LICENSE.txt for license information.
+// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
+//
+//===----------------------------------------------------------------------===//
+//
+// This file describes the X86 AVX10 V2 AUX instruction set, defining the
+// instructions and their encoding.
+//
+//===----------------------------------------------------------------------===//
+
+//===----------------------------------------------------------------------===//
+// AVX10 V2 AUX Multiclass Definitions
+//===----------------------------------------------------------------------===//
+
+// Group A multiclass: PS(f32) -> i8 truncating conversion (quarter-size output)
+// Output is always xmm for all VL variants.
+// Adapts avx512_vcvt_fp for each VL level with explicit dest/src type mappings.
+multiclass avx10_v2aux_cvt_trunc_ps2i8<bits<8> opc, string OpcodeStr,
+                                        SDPatternOperator OpNode,
+                                        SDPatternOperator MaskOpNode> {
+  let Predicates = [HasAVX10_V2_AUX] in {
+    let ExeDomain = SSEPackedSingle in {
+      let Uses = []<Register>, mayRaiseFPException = 0 in {
+        defm Z : avx512_vcvt_fp<opc, OpcodeStr, v16i8x_info, v16f32_info,
+                                OpNode, OpNode, WriteCvtPH2PSZ>, EVEX_V512;
+        // Z256/Z128: use null_frag because element count mismatch between
+        // dest (v16i8) and source (v8f32/v4f32) prevents avx512_vcvt_fp from
+        // generating correct masked patterns. Explicit Pat patterns below.
+        defm Z256 : avx512_vcvt_fp<opc, OpcodeStr, v16i8x_info, v8f32x_info,
+                                   null_frag, null_frag,
+                                   WriteCvtPH2PSZ, v8f32x_info.BroadcastStr,
+                                   "{y}", v8f32x_info.MemOp,
+                                   v8f32x_info.KRCWM>, EVEX_V256;
+        defm Z128 : avx512_vcvt_fp<opc, OpcodeStr, v16i8x_info, v4f32x_info,
+                                   null_frag, null_frag,
+                                   WriteCvtPH2PSZ, v4f32x_info.BroadcastStr,
+                                   "{x}", f128mem,
+                                   v4f32x_info.KRCWM>, EVEX_V128;
+      }
+    }
+
+    // InstAliases for x/y suffixes
+    def : InstAlias<OpcodeStr#"x\t{$src, $dst|$dst, $src}",
+                    (!cast<Instruction>(NAME # "Z128rr") VR128X:$dst,
+                    VR128X:$src), 0>;
+    def : InstAlias<OpcodeStr#"x\t{$src, $dst|$dst, $src}",
+                    (!cast<Instruction>(NAME # "Z128rm") VR128X:$dst,
+                    f128mem:$src), 0, "intel">;
+    def : InstAlias<OpcodeStr#"y\t{$src, $dst|$dst, $src}",
+                    (!cast<Instruction>(NAME # "Z256rr") VR128X:$dst,
+                    VR256X:$src), 0>;
+    def : InstAlias<OpcodeStr#"y\t{$src, $dst|$dst, $src}",
+                    (!cast<Instruction>(NAME # "Z256rm") VR128X:$dst,
+                    f256mem:$src), 0, "intel">;
+
+    // Explicit patterns for Z256 (8 source elements, VK8WM mask)
+    // Unmasked
+    def : Pat<(v16i8 (OpNode (v8f32 VR256X:$src))),
+              (!cast<Instruction>(NAME # "Z256rr") VR256X:$src)>;
+    // Masked (merge)
+    def : Pat<(MaskOpNode (v8f32 VR256X:$src), (v16i8 VR128X:$src0),
+                           VK8WM:$mask),
+              (!cast<Instruction>(NAME # "Z256rrk") VR128X:$src0, VK8WM:$mask,
+                                  VR256X:$src)>;
+    // Masked (zero)
+    def : Pat<(MaskOpNode (v8f32 VR256X:$src), v16i8x_info.ImmAllZerosV,
+                           VK8WM:$mask),
+              (!cast<Instruction>(NAME # "Z256rrkz") VK8WM:$mask,
+                                  VR256X:$src)>;
+    // Memory
+    def : Pat<(v16i8 (OpNode (loadv8f32 addr:$src))),
+              (!cast<Instruction>(NAME # "Z256rm") addr:$src)>;
+    def : Pat<(MaskOpNode (loadv8f32 addr:$src), (v16i8 VR128X:$src0),
+                           VK8WM:$mask),
+              (!cast<Instruction>(NAME # "Z256rmk") VR128X:$src0, VK8WM:$mask,
+                                  addr:$src)>;
+    def : Pat<(MaskOpNode (loadv8f32 addr:$src), v16i8x_info.ImmAllZerosV,
+                           VK8WM:$mask),
+              (!cast<Instruction>(NAME # "Z256rmkz") VK8WM:$mask, addr:$src)>;
+    // Broadcast
+    def : Pat<(v16i8 (OpNode (v8f32 (X86VBroadcastld32 addr:$src)))),
+              (!cast<Instruction>(NAME # "Z256rmb") addr:$src)>;
+    def : Pat<(MaskOpNode (v8f32 (X86VBroadcastld32 addr:$src)),
+                            (v16i8 VR128X:$src0), VK8WM:$mask),
+              (!cast<Instruction>(NAME # "Z256rmbk") VR128X:$src0, VK8WM:$mask,
+                                  addr:$src)>;
+    def : Pat<(MaskOpNode (v8f32 (X86VBroadcastld32 addr:$src)),
+                            v16i8x_info.ImmAllZerosV, VK8WM:$mask),
+              (!cast<Instruction>(NAME # "Z256rmbkz") VK8WM:$mask, addr:$src)>;
+
+    // Explicit patterns for Z128 (4 source elements, VK4WM mask)
+    // Unmasked
+    def : Pat<(v16i8 (OpNode (v4f32 VR128X:$src))),
+              (!cast<Instruction>(NAME # "Z128rr") VR128X:$src)>;
+    // Masked (merge)
+    def : Pat<(MaskOpNode (v4f32 VR128X:$src), (v16i8 VR128X:$src0),
+                           VK4WM:$mask),
+              (!cast<Instruction>(NAME # "Z128rrk") VR128X:$src0, VK4WM:$mask,
+                                  VR128X:$src)>;
+    // Masked (zero)
+    def : Pat<(MaskOpNode (v4f32 VR128X:$src), v16i8x_info.ImmAllZerosV,
+                           VK4WM:$mask),
+              (!cast<Instruction>(NAME # "Z128rrkz") VK4WM:$mask,
+                                  VR128X:$src)>;
+    // Memory
+    def : Pat<(v16i8 (OpNode (loadv4f32 addr:$src))),
+              (!cast<Instruction>(NAME # "Z128rm") addr:$src)>;
+    def : Pat<(MaskOpNode (loadv4f32 addr:$src), (v16i8 VR128X:$src0),
+                           VK4WM:$mask),
+              (!cast<Instruction>(NAME # "Z128rmk") VR128X:$src0, VK4WM:$mask,
+                                  addr:$src)>;
+    def : Pat<(MaskOpNode (loadv4f32 addr:$src), v16i8x_info.ImmAllZerosV,
+                           VK4WM:$mask),
+              (!cast<Instruction>(NAME # "Z128rmkz") VK4WM:$mask, addr:$src)>;
+    // Broadcast
+    def : Pat<(v16i8 (OpNode (v4f32 (X86VBroadcastld32 addr:$src)))),
+              (!cast<Instruction>(NAME # "Z128rmb") addr:$src)>;
+    def : Pat<(MaskOpNode (v4f32 (X86VBroadcastld32 addr:$src)),
+                            (v16i8 VR128X:$src0), VK4WM:$mask),
+              (!cast<Instruction>(NAME # "Z128rmbk") VR128X:$src0, VK4WM:$mask,
+                                  addr:$src)>;
+    def : Pat<(MaskOpNode (v4f32 (X86VBroadcastld32 addr:$src)),
+                            v16i8x_info.ImmAllZerosV, VK4WM:$mask),
+              (!cast<Instruction>(NAME # "Z128rmbkz") VK4WM:$mask, addr:$src)>;
+  }
+}
+
+// Group B multiclass: 3-operand bias PS(f32) -> i8 conversion (quarter-size output)
+// bias + f32 source -> i8 dest
+// Reuses avx10_convert_3op_packed from X86InstrAVX10.td for each VL variant.
+multiclass avx10_v2aux_convert_3op_ps<bits<8> OpCode, string OpcodeStr,
+                                       SDPatternOperator OpNode,
+                                       SDPatternOperator MaskOpNode> {
+  let Predicates = [HasAVX10_V2_AUX] in {
+    // Z (512-bit): bias=v64i8(zmm), src=v16f32(zmm), dst=v16i8(xmm)
+    // Element counts match (16), so vselect_mask works directly.
+    defm Z : avx10_convert_3op_packed<OpCode, OpcodeStr, v16i8x_info,
+               v64i8_info, v16f32_info, OpNode, OpNode, WriteCvtPH2PSZ>,
+               EVEX_V512, EVEX_CD8<32, CD8VF>;
+    // Z256/Z128: use null_frag because element count mismatch between
+    // dest (v16i8) and source (v8f32/v4f32) prevents vselect_mask from
+    // generating correct masked patterns. Explicit Pat patterns below.
+    defm Z256 : avx10_convert_3op_packed<OpCode, OpcodeStr, v16i8x_info,
+                  v32i8x_info, v8f32x_info,
+                  null_frag, null_frag, WriteCvtPH2PSZ>,
+                  EVEX_V256, EVEX_CD8<32, CD8VF>;
+    defm Z128 : avx10_convert_3op_packed<OpCode, OpcodeStr, v16i8x_info,
+                  v16i8x_info, v4f32x_info,
+                  null_frag, null_frag, WriteCvtPH2PSZ>,
+                  EVEX_V128, EVEX_CD8<32, CD8VF>;
+
+    // Explicit patterns for Z256 (8 source elements, VK8WM mask)
+    def : Pat<(v16i8 (OpNode (v32i8 VR256X:$src1), (v8f32 VR256X:$src2))),
+              (!cast<Instruction>(NAME # "Z256rr") VR256X:$src1, VR256X:$src2)>;
+    def : Pat<(MaskOpNode (v32i8 VR256X:$src1), (v8f32 VR256X:$src2),
+                           (v16i8 VR128X:$src0), VK8WM:$mask),
+              (!cast<Instruction>(NAME # "Z256rrk") VR128X:$src0, VK8WM:$mask,
+                                  VR256X:$src1, VR256X:$src2)>;
+    def : Pat<(MaskOpNode (v32i8 VR256X:$src1), (v8f32 VR256X:$src2),
+                           v16i8x_info.ImmAllZerosV, VK8WM:$mask),
+              (!cast<Instruction>(NAME # "Z256rrkz") VK8WM:$mask,
+                                  VR256X:$src1, VR256X:$src2)>;
+    // Memory
+    def : Pat<(v16i8 (OpNode (v32i8 VR256X:$src1), (loadv8f32 addr:$src2))),
+              (!cast<Instruction>(NAME # "Z256rm") VR256X:$src1, addr:$src2)>;
+    def : Pat<(MaskOpNode (v32i8 VR256X:$src1), (loadv8f32 addr:$src2),
+                           (v16i8 VR128X:$src0), VK8WM:$mask),
+              (!cast<Instruction>(NAME # "Z256rmk") VR128X:$src0, VK8WM:$mask,
+                                  VR256X:$src1, addr:$src2)>;
+    def : Pat<(MaskOpNode (v32i8 VR256X:$src1), (loadv8f32 addr:$src2),
+                           v16i8x_info.ImmAllZerosV, VK8WM:$mask),
+              (!cast<Instruction>(NAME # "Z256rmkz") VK8WM:$mask,
+                                  VR256X:$src1, addr:$src2)>;
+    // Broadcast
+    def : Pat<(v16i8 (OpNode (v32i8 VR256X:$src1),
+                              (v8f32 (X86VBroadcastld32 addr:$src2)))),
+              (!cast<Instruction>(NAME # "Z256rmb") VR256X:$src1, addr:$src2)>;
+    def : Pat<(MaskOpNode (v32i8 VR256X:$src1),
+                            (v8f32 (X86VBroadcastld32 addr:$src2)),
+                            (v16i8 VR128X:$src0), VK8WM:$mask),
+              (!cast<Instruction>(NAME # "Z256rmbk") VR128X:$src0, VK8WM:$mask,
+                                  VR256X:$src1, addr:$src2)>;
+    def : Pat<(MaskOpNode (v32i8 VR256X:$src1),
+                            (v8f32 (X86VBroadcastld32 addr:$src2)),
+                            v16i8x_info.ImmAllZerosV, VK8WM:$mask),
+              (!cast<Instruction>(NAME # "Z256rmbkz") VK8WM:$mask,
+                                  VR256X:$src1, addr:$src2)>;
+
+    // Explicit patterns for Z128 (4 source elements, VK4WM mask)
+    def : Pat<(v16i8 (OpNode (v16i8 VR128X:$src1), (v4f32 VR128X:$src2))),
+              (!cast<Instruction>(NAME # "Z128rr") VR128X:$src1, VR128X:$src2)>;
+    def : Pat<(MaskOpNode (v16i8 VR128X:$src1), (v4f32 VR128X:$src2),
+                           (v16i8 VR128X:$src0), VK4WM:$mask),
+              (!cast<Instruction>(NAME # "Z128rrk") VR128X:$src0, VK4WM:$mask,
+                                  VR128X:$src1, VR128X:$src2)>;
+    def : Pat<(MaskOpNode (v16i8 VR128X:$src1), (v4f32 VR128X:$src2),
+                           v16i8x_info.ImmAllZerosV, VK4WM:$mask),
+              (!cast<Instruction>(NAME # "Z128rrkz") VK4WM:$mask,
+                                  VR128X:$src1, VR128X:$src2)>;
+    // Memory
+    def : Pat<(v16i8 (OpNode (v16i8 VR128X:$src1), (loadv4f32 addr:$src2))),
+              (!cast<Instruction>(NAME # "Z128rm") VR128X:$src1, addr:$src2)>;
+    def : Pat<(MaskOpNode (v16i8 VR128X:$src1), (loadv4f32 addr:$src2),
+                           (v16i8 VR128X:$src0), VK4WM:$mask),
+              (!cast<Instruction>(NAME # "Z128rmk") VR128X:$src0, VK4WM:$mask,
+                                  VR128X:$src1, addr:$src2)>;
+    def : Pat<(MaskOpNode (v16i8 VR128X:$src1), (loadv4f32 addr:$src2),
+                           v16i8x_info.ImmAllZerosV, VK4WM:$mask),
+              (!cast<Instruction>(NAME # "Z128rmkz") VK4WM:$mask,
+                                  VR128X:$src1, addr:$src2)>;
+    // Broadcast
+    def : Pat<(v16i8 (OpNode (v16i8 VR128X:$src1),
+                              (v4f32 (X86VBroadcastld32 addr:$src2)))),
+              (!cast<Instruction>(NAME # "Z128rmb") VR128X:$src1, addr:$src2)>;
+    def : Pat<(MaskOpNode (v16i8 VR128X:$src1),
+                            (v4f32 (X86VBroadcastld32 addr:$src2)),
+                            (v16i8 VR128X:$src0), VK4WM:$mask),
+              (!cast<Instruction>(NAME # "Z128rmbk") VR128X:$src0, VK4WM:$mask,
+                                  VR128X:$src1, addr:$src2)>;
+    def : Pat<(MaskOpNode (v16i8 VR128X:$src1),
+                            (v4f32 (X86VBroadcastld32 addr:$src2)),
+                            v16i8x_info.ImmAllZerosV, VK4WM:$mask),
+              (!cast<Instruction>(NAME # "Z128rmbkz") VK4WM:$mask,
+                                  VR128X:$src1, addr:$src2)>;
+  }
+}
+
+// Group C multiclass: i8 -> f32 expanding conversion (4x expansion, no broadcast)
+multiclass avx10_v2aux_convert_2op_i8_to_f32<string OpcodeStr, bits<8> opc,
+                                              SDNode OpNode> {
+  let Predicates = [HasAVX10_V2_AUX] in {
+    defm Z : avx10_convert_2op_nomb_packed<opc, OpcodeStr, v16f32_info,
+                                           v16i8x_info, OpNode, f128mem,
+                                           WriteCvtPH2PSZ>, EVEX_V512;
+    defm Z128 : avx10_convert_2op_nomb_packed<opc, OpcodeStr, v4f32x_info,
+                                              v16i8x_info, OpNode, f32mem,
+                                              WriteCvtPH2PSZ>, EVEX_V128;
+    defm Z256 : avx10_convert_2op_nomb_packed<opc, OpcodeStr, v8f32x_info,
+                                              v16i8x_info, OpNode, f64mem,
+                                              WriteCvtPH2PSZ>, EVEX_V256;
+  }
+}
+
+// Group D multiclass: Store-like truncation (reg/mem dest, no masking)
+// Uses MRMDestReg/MRMDestMem since destination can be memory operand.
+// Source is in reg field, destination is in r/m field.
+multiclass avx10_v2aux_trunc_store<bits<8> opc, string OpcodeStr,
+                                    X86VectorVTInfo SrcInfo,
+                                    X86VectorVTInfo DestInfo,
+                                    X86MemOperand x86memop> {
+  let hasSideEffects = 0 in {
+    def rr : I<opc, MRMDestReg, (outs DestInfo.RC:$dst),
+               (ins SrcInfo.RC:$src),
+               OpcodeStr # "\t{$src, $dst|$dst, $src}", []>,
+               EVEX, Sched<[WriteCvtPH2PSZ]>;
+    let mayStore = 1 in
+    def mr : I<opc, MRMDestMem, (outs),
+               (ins x86memop:$dst, SrcInfo.RC:$src),
+               OpcodeStr # "\t{$src, $dst|$dst, $src}", []>,
+               EVEX, Sched<[WriteCvtPH2PSZ.Folded]>;
+  }
+}
+
+// Group F helper: expanding conversion with masking, no broadcast (reg+mem)
+multiclass avx10_v2aux_convert_expand_masked<bits<8> opc, string OpcodeStr,
+                                              X86VectorVTInfo _dest,
+                                              X86VectorVTInfo _src,
+                                              X86MemOperand x86memop> {
+  let ExeDomain = _dest.ExeDomain in {
+    defm rr : AVX512_maskable_custom<opc, MRMSrcReg,
+                (outs _dest.RC:$dst),
+                (ins _src.RC:$src),
+                (ins _dest.RC:$src0, _dest.KRCWM:$mask, _src.RC:$src),
+                (ins _dest.KRCWM:$mask, _src.RC:$src),
+                OpcodeStr, "$src", "$src", [], [], [],
+                "$src0 = $dst">,
+               EVEX, Sched<[WriteCvtPH2PSZ]>;
+    let mayLoad = 1 in
+    defm rm : AVX512_maskable_custom<opc, MRMSrcMem,
+                (outs _dest.RC:$dst),
+                (ins x86memop:$src),
+                (ins _dest.RC:$src0, _dest.KRCWM:$mask, x86memop:$src),
+                (ins _dest.KRCWM:$mask, x86memop:$src),
+                OpcodeStr, "$src", "$src", [], [], [],
+                "$src0 = $dst">,
+               EVEX, Sched<[WriteCvtPH2PSZ.Folded]>;
+  }
+}
+
+// Group F helper: same-size conversion with masking (reg-only)
+multiclass avx10_v2aux_convert_samesize_masked<bits<8> opc, string OpcodeStr,
+                                                X86VectorVTInfo _> {
+  let ExeDomain = _.ExeDomain in {
+    defm rr : AVX512_maskable_custom<opc, MRMSrcReg,
+                (outs _.RC:$dst),
+                (ins _.RC:$src),
+                (ins _.RC:$src0, _.KRCWM:$mask, _.RC:$src),
+                (ins _.KRCWM:$mask, _.RC:$src),
+                OpcodeStr, "$src", "$src", [], [], [],
+                "$src0 = $dst">,
+               EVEX, Sched<[WriteCvtPH2PSZ]>;
+  }
+}
+
+// Group H multiclass: Byte unpack with immediate
+multiclass avx10_v2aux_unpackb<bits<8> opc, string OpcodeStr,
+                                X86VectorVTInfo _> {
+  let ImmT = Imm8 in {
+    defm ri : AVX512_maskable_custom<opc, MRMSrcReg,
+                    (outs _.RC:$dst),
+                    (ins _.RC:$src1, u8imm:$src2),
+                    (ins _.RC:$src0, _.KRCWM:$mask, _.RC:$src1, u8imm:$src2),
+                    (ins _.KRCWM:$mask, _.RC:$src1, u8imm:$src2),
+                    OpcodeStr, "$src2, $src1", "$src1, $src2",
+                    [], [], [],
+                    "$src0 = $dst">,
+                    Sched<[WriteShuffle]>;
+    let mayLoad = 1 in
+    defm mi : AVX512_maskable_custom<opc, MRMSrcMem,
+                    (outs _.RC:$dst),
+                    (ins _.MemOp:$src1, u8imm:$src2),
+                    (ins _.RC:$src0, _.KRCWM:$mask, _.MemOp:$src1, u8imm:$src2),
+                    (ins _.KRCWM:$mask, _.MemOp:$src1, u8imm:$src2),
+                    OpcodeStr, "$src2, $src1", "$src1, $src2",
+                    [], [], [],
+                    "$src0 = $dst">,
+                    Sched<[WriteShuffle.Folded]>;
+  }
+}
+
+//===----------------------------------------------------------------------===//
+// AVX10 V2 AUX Instruction Definitions
+//===----------------------------------------------------------------------===//
+
+//-------------------------------------------------
+// Group A: PS->8bit truncating conversions
+//-------------------------------------------------
+
+defm VCVTPS2BF8 : avx10_v2aux_cvt_trunc_ps2i8<0x39, "vcvtps2bf8",
+                                                X86vcvtps2bf8, X86vmcvtps2bf8>,
+                  T_MAP5, XS, EVEX_CD8<32, CD8VF>;
+defm VCVTPS2BF8S : avx10_v2aux_cvt_trunc_ps2i8<0x3B, "vcvtps2bf8s",
+                                                 X86vcvtps2bf8s, X86vmcvtps2bf8s>,
+                   T_MAP5, XS, EVEX_CD8<32, CD8VF>;
+defm VCVTPS2HF8 : avx10_v2aux_cvt_trunc_ps2i8<0x38, "vcvtps2hf8",
+                                                X86vcvtps2hf8, X86vmcvtps2hf8>,
+                  T_MAP5, XS, EVEX_CD8<32, CD8VF>;
+defm VCVTPS2HF8S : avx10_v2aux_cvt_trunc_ps2i8<0x3A, "vcvtps2hf8s",
+                                                  X86vcvtps2hf8s, X86vmcvtps2hf8s>,
+                   T_MAP5, XS, EVEX_CD8<32, CD8VF>;
+defm VCVTROPS2HF8 : avx10_v2aux_cvt_trunc_ps2i8<0x38, "vcvtrops2hf8",
+                                                   X86vcvtrops2hf8, X86vmcvtrops2hf8>,
+                    T_MAP5, PD, EVEX_CD8<32, CD8VF>;
+defm VCVTROPS2HF8S : avx10_v2aux_cvt_trunc_ps2i8<0x3A, "vcvtrops2hf8s",
+                                                    X86vcvtrops2hf8s, X86vmcvtrops2hf8s>,
+                     T_MAP5, PD, EVEX_CD8<32, CD8VF>;
+
+//-------------------------------------------------
+// Group B: Bias PS->8bit conversions (3-operand)
+//-------------------------------------------------
+
+defm VCVTBIASPS2BF8 : avx10_v2aux_convert_3op_ps<0x39, "vcvtbiasps2bf8",
+                                                   X86vcvtbiasps2bf8,
+                                                   X86vmcvtbiasps2bf8>,
+                      T_MAP5, PS;
+defm VCVTBIASPS2BF8S : avx10_v2aux_convert_3op_ps<0x3B, "vcvtbiasps2bf8s",
+                                                    X86vcvtbiasps2bf8s,
+                                                    X86vmcvtbiasps2bf8s>,
+                       T_MAP5, PS;
+defm VCVTBIASPS2HF8 : avx10_v2aux_convert_3op_ps<0x38, "vcvtbiasps2hf8",
+                                                   X86vcvtbiasps2hf8,
+                                                   X86vmcvtbiasps2hf8>,
+                      T_MAP5, PS;
+defm VCVTBIASPS2HF8S : avx10_v2aux_convert_3op_ps<0x3A, "vcvtbiasps2hf8s",
+                                                    X86vcvtbiasps2hf8s,
+                                                    X86vmcvtbiasps2hf8s>,
+                       T_MAP5, PS;
+
+//-------------------------------------------------
+// Group C: 8bit->PS expanding conversions
+//-------------------------------------------------
+
+defm VCVTBF82PS : avx10_v2aux_convert_2op_i8_to_f32<"vcvtbf82ps", 0x36,
+                                                      X86vcvtbf82ps>,
+                  PS, T_MAP5, EVEX, EVEX_CD8<32, CD8VQ>, REX_W;
+defm VCVTHF82PS : avx10_v2aux_convert_2op_i8_to_f32<"vcvthf82ps", 0x36,
+                                                      X86vcvthf82ps>,
+                  PS, T_MAP5, EVEX, EVEX_CD8<32, CD8VQ>;
+
+// X //-------------------------------------------------
+// Group D: BF8/HF8->BF4S store-like truncations
+//-------------------------------------------------
+
+let Predicates = [HasAVX10_V2_AUX] in {
+  defm VCVTBF82BF4SZ    : avx10_v2aux_trunc_store<0x3D, "vcvtbf82bf4s",
+                             v64i8_info, v32i8x_info, i256mem>,
+                           T_MAP5, XS, REX_W, EVEX_V512, EVEX_CD8<8, CD8VH>;
+  defm VCVTBF82BF4SZ256 : avx10_v2aux_trunc_store<0x3D, "vcvtbf82bf4s",
+                             v32i8x_info, v16i8x_info, i128mem>,
+                           T_MAP5, XS, REX_W, EVEX_V256, EVEX_CD8<8, CD8VH>;
+  defm VCVTBF82BF4SZ128 : avx10_v2aux_trunc_store<0x3D, "vcvtbf82bf4s",
+                             v16i8x_info, v16i8x_info, i64mem>,
+                           T_MAP5, XS, REX_W, EVEX_V128, EVEX_CD8<8, CD8VH>;
+
+  defm VCVTHF82BF4SZ    : avx10_v2aux_trunc_store<0x3D, "vcvthf82bf4s",
+                             v64i8_info, v32i8x_info, i256mem>,
+                           T_MAP5, XS, EVEX_V512, EVEX_CD8<8, CD8VH>;
+  defm VCVTHF82BF4SZ256 : avx10_v2aux_trunc_store<0x3D, "vcvthf82bf4s",
+                             v32i8x_info, v16i8x_info, i128mem>,
+                           T_MAP5, XS, EVEX_V256, EVEX_CD8<8, CD8VH>;
+  defm VCVTHF82BF4SZ128 : avx10_v2aux_trunc_store<0x3D, "vcvthf82bf4s",
+                             v16i8x_info, v16i8x_info, i64mem>,
+                           T_MAP5, XS, EVEX_V128, EVEX_CD8<8, CD8VH>;
+}
+
+//-------------------------------------------------
+// Group E: Same-size reg-only conversions (no masking)
+//-------------------------------------------------
+
+let Predicates = [HasAVX10_V2_AUX], hasSideEffects = 0 in {
+  def VCVTBF82BF6SZrr : I<0x3E, MRMSrcReg, (outs VR512:$dst),
+                           (ins VR512:$src),
+                           "vcvtbf82bf6s\t{$src, $dst|$dst, $src}", []>,
+                         EVEX, T_MAP5, XS, REX_W, EVEX_V512,
+                         Sched<[WriteCvtPH2PSZ]>;
+  def VCVTBF82BF6SZ256rr : I<0x3E, MRMSrcReg, (outs VR256X:$dst),
+                              (ins VR256X:$src),
+                              "vcvtbf82bf6s\t{$src, $dst|$dst, $src}", []>,
+                            EVEX, T_MAP5, XS, REX_W, EVEX_V256,
+                            Sched<[WriteCvtPH2PSZ]>;
+  def VCVTBF82BF6SZ128rr : I<0x3E, MRMSrcReg, (outs VR128X:$dst),
+                              (ins VR128X:$src),
+                              "vcvtbf82bf6s\t{$src, $dst|$dst, $src}", []>,
+                            EVEX, T_MAP5, XS, REX_W, EVEX_V128,
+                            Sched<[WriteCvtPH2PSZ]>;
+
+  def VCVTHF82HF6SZrr : I<0x3C, MRMSrcReg, (outs VR512:$dst),
+                           (ins VR512:$src),
+                           "vcvthf82hf6s\t{$src, $dst|$dst, $src}", []>,
+                         EVEX, T_MAP5, XS, EVEX_V512,
+                         Sched<[WriteCvtPH2PSZ]>;
+  def VCVTHF82HF6SZ256rr : I<0x3C, MRMSrcReg, (outs VR256X:$dst),
+                              (ins VR256X:$src),
+                              "vcvthf82hf6s\t{$src, $dst|$dst, $src}", []>,
+                            EVEX, T_MAP5, XS, EVEX_V256,
+                            Sched<[WriteCvtPH2PSZ]>;
+  def VCVTHF82HF6SZ128rr : I<0x3C, MRMSrcReg, (outs VR128X:$dst),
+                              (ins VR128X:$src),
+                              "vcvthf82hf6s\t{$src, $dst|$dst, $src}", []>,
+                            EVEX, T_MAP5, XS, EVEX_V128,
+                            Sched<[WriteCvtPH2PSZ]>;
+}
+
+// Intrinsic patterns for Group E
+let Predicates = [HasAVX10_V2_AUX] in {
+  def : Pat<(v16i8 (int_x86_avx10_vcvtbf82bf6s_128 (v16i8 VR128X:$src))),
+            (VCVTBF82BF6SZ128rr VR128X:$src)>;
+  def : Pat<(v32i8 (int_x86_avx10_vcvtbf82bf6s_256 (v32i8 VR256X:$src))),
+            (VCVTBF82BF6SZ256rr VR256X:$src)>;
+  def : Pat<(v64i8 (int_x86_avx10_vcvtbf82bf6s_512 (v64i8 VR512:$src))),
+            (VCVTBF82BF6SZrr VR512:$src)>;
+
+  def : Pat<(v16i8 (int_x86_avx10_vcvthf82hf6s_128 (v16i8 VR128X:$src))),
+            (VCVTHF82HF6SZ128rr VR128X:$src)>;
+  def : Pat<(v32i8 (int_x86_avx10_vcvthf82hf6s_256 (v32i8 VR256X:$src))),
+            (VCVTHF82HF6SZ256rr VR256X:$src)>;
+  def : Pat<(v64i8 (int_x86_avx10_vcvthf82hf6s_512 (v64i8 VR512:$src))),
+            (VCVTHF82HF6SZrr VR512:$src)>;
+}
+
+//-------------------------------------------------
+// Group F: Expanding/same-size conversions with masking
+//-------------------------------------------------
+
+// VCVTBF42HF8: expanding (2x), with masking, no broadcast
+// Z128: xmm{k}{z}, xmm/m64  Z256: ymm{k}{z}, xmm/m128  Z: zmm{k}{z}, ymm/m256
+let Predicates = [HasAVX10_V2_AUX] in {
+  defm VCVTBF42HF8Z : avx10_v2aux_convert_expand_masked<0x37, "vcvtbf42hf8",
+                         v64i8_info, v32i8x_info, f256mem>,
+                       T_MAP5, PS, EVEX_V512, EVEX_CD8<8, CD8VH>;
+  defm VCVTBF42HF8Z128 : avx10_v2aux_convert_expand_masked<0x37, "vcvtbf42hf8",
+                            v16i8x_info, v16i8x_info, f64mem>,
+                          T_MAP5, PS, EVEX_V128, EVEX_CD8<8, CD8VH>;
+  defm VCVTBF42HF8Z256 : avx10_v2aux_convert_expand_masked<0x37, "vcvtbf42hf8",
+                            v32i8x_info, v16i8x_info, f128mem>,
+                          T_MAP5, PS, EVEX_V256, EVEX_CD8<8, CD8VH>;
+}
+let Predicates = [HasAVX10_V2_AUX] in {
+  let ExeDomain = SSEPackedInt in {
+    // VCVTBF62HF8: same-size with masking, reg-only (66.MAP5.W1)
+    defm VCVTBF62HF8Z    : avx10_v2aux_convert_samesize_masked<0x37,
+                              "vcvtbf62hf8", v64i8_info>,
+                            T_MAP5, PD, REX_W, EVEX_V512;
+    defm VCVTBF62HF8Z256 : avx10_v2aux_convert_samesize_masked<0x37,
+                              "vcvtbf62hf8", v32i8x_info>,
+                            T_MAP5, PD, REX_W, EVEX_V256;
+    defm VCVTBF62HF8Z128 : avx10_v2aux_convert_samesize_masked<0x37,
+                              "vcvtbf62hf8", v16i8x_info>,
+                            T_MAP5, PD, REX_W, EVEX_V128;
+
+    // VCVTHF62HF8: same-size with masking, reg-only (66.MAP5.W0)
+    defm VCVTHF62HF8Z    : avx10_v2aux_convert_samesize_masked<0x37,
+                              "vcvthf62hf8", v64i8_info>,
+                            T_MAP5, PD, EVEX_V512;
+    defm VCVTHF62HF8Z256 : avx10_v2aux_convert_samesize_masked<0x37,
+                              "vcvthf62hf8", v32i8x_info>,
+                            T_MAP5, PD, EVEX_V256;
+    defm VCVTHF62HF8Z128 : avx10_v2aux_convert_samesize_masked<0x37,
+                              "vcvthf62hf8", v16i8x_info>,
+                            T_MAP5, PD, EVEX_V128;
+  }
+}
+
+// Intrinsic patterns for Group F (unmasked only; masking via selectb in C header)
+
+let Predicates = [HasAVX10_V2_AUX] in {
+  // VCVTBF42HF8
+  def : Pat<(v16i8 (int_x86_avx10_vcvtbf42hf8_128 (v16i8 VR128X:$src))),
+            (VCVTBF42HF8Z128rr VR128X:$src)>;
+  def : Pat<(v32i8 (int_x86_avx10_vcvtbf42hf8_256 (v16i8 VR128X:$src))),
+            (VCVTBF42HF8Z256rr VR128X:$src)>;
+  def : Pat<(v64i8 (int_x86_avx10_vcvtbf42hf8_512 (v32i8 VR256X:$src))),
+            (VCVTBF42HF8Zrr VR256X:$src)>;
+
+  // VCVTBF62HF8
+  def : Pat<(v16i8 (int_x86_avx10_vcvtbf62hf8_128 (v16i8 VR128X:$src))),
+            (VCVTBF62HF8Z128rr VR128X:$src)>;
+  def : Pat<(v32i8 (int_x86_avx10_vcvtbf62hf8_256 (v32i8 VR256X:$src))),
+            (VCVTBF62HF8Z256rr VR256X:$src)>;
+  def : Pat<(v64i8 (int_x86_avx10_vcvtbf62hf8_512 (v64i8 VR512:$src))),
+            (VCVTBF62HF8Zrr VR512:$src)>;
+
+  // VCVTHF62HF8
+  def : Pat<(v16i8 (int_x86_avx10_vcvthf62hf8_128 (v16i8 VR128X:$src))),
+            (VCVTHF62HF8Z128rr VR128X:$src)>;
+  def : Pat<(v32i8 (int_x86_avx10_vcvthf62hf8_256 (v32i8 VR256X:$src))),
+            (VCVTHF62HF8Z256rr VR256X:$src)>;
+  def : Pat<(v64i8 (int_x86_avx10_vcvthf62hf8_512 (v64i8 VR512:$src))),
+            (VCVTHF62HF8Zrr VR512:$src)>;
+}
+
+//-------------------------------------------------
+// Group G: VPMOVSSDB - Integer DWord->Byte signed saturation
+// F3.0F38.W0 0x41
+//-------------------------------------------------
+
+let Predicates = [HasAVX10_V2_AUX] in
+  defm VPMOVSSDZ : avx512_trunc_common<0x41, "vpmovssdb", X86vtruncs,
+                     X86vmtruncs, SchedWriteVecTruncate.ZMM,
+                     v16i32_info, v16i8x_info, i128mem>,
+                   EVEX_V512, EVEX_CD8<8, CD8VQ>;
+let Predicates = [HasVLX, HasAVX10_V2_AUX] in {
+  defm VPMOVSSDZ256 : avx512_trunc_common<0x41, "vpmovssdb", X86vtruncs,
+                        X86vmtruncs, SchedWriteVecTruncate.YMM,
+                        v8i32x_info, v16i8x_info, i64mem>,
+                      EVEX_V256, EVEX_CD8<8, CD8VQ>;
+  defm VPMOVSSDZ128 : avx512_trunc_common<0x41, "vpmovssdb", X86vtruncs,
+                        X86vmtruncs, SchedWriteVecTruncate.XMM,
+                        v4i32x_info, v16i8x_info, i32mem>,
+                      EVEX_V128, EVEX_CD8<8, CD8VQ>;
+}
+
+//-------------------------------------------------
+// Group H: VUNPACKB - Byte unpack with immediate
+//-------------------------------------------------
+
+let Predicates = [HasAVX10_V2_AUX] in {
+  defm VUNPACKBZ    : avx10_v2aux_unpackb<0x3D, "vunpackb", v64i8_info>,
+                      EVEX_V512, TA, PS, EVEX, EVEX_CD8<8, CD8VF>;
+  defm VUNPACKBZ256 : avx10_v2aux_unpackb<0x3D, "vunpackb", v32i8x_info>,
+                      EVEX_V256, TA, PS, EVEX, EVEX_CD8<8, CD8VF>;
+  defm VUNPACKBZ128 : avx10_v2aux_unpackb<0x3D, "vunpackb", v16i8x_info>,
+                      EVEX_V128, TA, PS, EVEX, EVEX_CD8<8, CD8VF>;
+
+  // Intrinsic patterns for Group H (unmasked only; masking via selectb in C header)
+  def : Pat<(v16i8 (int_x86_avx10_vunpackb_128
+              (v16i8 VR128X:$src1), timm:$src2)),
+            (VUNPACKBZ128ri VR128X:$src1, timm:$src2)>;
+  def : Pat<(v32i8 (int_x86_avx10_vunpackb_256
+              (v32i8 VR256X:$src1), timm:$src2)),
+            (VUNPACKBZ256ri VR256X:$src1, timm:$src2)>;
+  def : Pat<(v64i8 (int_x86_avx10_vunpackb_512
+              (v64i8 VR512:$src1), timm:$src2)),
+            (VUNPACKBZri VR512:$src1, timm:$src2)>;
+}
diff --git a/llvm/lib/Target/X86/X86InstrFragmentsSIMD.td b/llvm/lib/Target/X86/X86InstrFragmentsSIMD.td
index 1a75381aaaa24c..c109d682dd9156 100644
--- a/llvm/lib/Target/X86/X86InstrFragmentsSIMD.td
+++ b/llvm/lib/Target/X86/X86InstrFragmentsSIMD.td
@@ -1211,6 +1211,63 @@ def X86vmcvtph2hf8 : SDNode<"X86ISD::VMCVTPH2HF8",
 def X86vmcvtph2hf8s : SDNode<"X86ISD::VMCVTPH2HF8S",
                         SDTAVX10CONVERT_I8F16_MASK>;
 
+// AVX10 V2 AUX convert SDTypeProfiles
+def SDTAVX10CONVERT_I8F32 : SDTypeProfile<1, 1, [
+  SDTCVecEltisVT<0, i8>, SDTCVecEltisVT<1, f32>
+]>;
+
+def SDTAVX10CONVERT_I8F32_MASK : SDTypeProfile<1, 3, [
+  SDTCVecEltisVT<0, i8>, SDTCVecEltisVT<1, f32>,
+  SDTCisSameAs<0, 2>, SDTCVecEltisVT<3, i1>,
+  SDTCisSameNumEltsAs<1, 3>
+]>;
+
+def SDTAVX10CONVERT_2I8F32 : SDTypeProfile<1, 2, [
+  SDTCVecEltisVT<0, i8>, SDTCVecEltisVT<1, i8>, SDTCVecEltisVT<2, f32>
+]>;
+
+def SDTAVX10CONVERT_2I8F32_MASK : SDTypeProfile<1, 4, [
+  SDTCVecEltisVT<0, i8>, SDTCVecEltisVT<1, i8>,
+  SDTCVecEltisVT<2, f32>, SDTCisSameAs<0, 3>, SDTCVecEltisVT<4, i1>,
+  SDTCisSameNumEltsAs<2, 4>
+]>;
+
+def SDTAVX10CONVERT_F32I8 : SDTypeProfile<1, 1, [
+  SDTCVecEltisVT<0, f32>, SDTCVecEltisVT<1, i8>
+]>;
+
+// AVX10 V2 AUX convert SDNodes
+// Group A: PS->8bit truncating (single source)
+def X86vcvtps2bf8  : SDNode<"X86ISD::VCVTPS2BF8",  SDTAVX10CONVERT_I8F32>;
+def X86vcvtps2bf8s : SDNode<"X86ISD::VCVTPS2BF8S", SDTAVX10CONVERT_I8F32>;
+def X86vcvtps2hf8  : SDNode<"X86ISD::VCVTPS2HF8",  SDTAVX10CONVERT_I8F32>;
+def X86vcvtps2hf8s : SDNode<"X86ISD::VCVTPS2HF8S", SDTAVX10CONVERT_I8F32>;
+def X86vcvtrops2hf8  : SDNode<"X86ISD::VCVTROPS2HF8",  SDTAVX10CONVERT_I8F32>;
+def X86vcvtrops2hf8s : SDNode<"X86ISD::VCVTROPS2HF8S", SDTAVX10CONVERT_I8F32>;
+
+// Masked variants
+def X86vmcvtps2bf8  : SDNode<"X86ISD::VMCVTPS2BF8",  SDTAVX10CONVERT_I8F32_MASK>;
+def X86vmcvtps2bf8s : SDNode<"X86ISD::VMCVTPS2BF8S", SDTAVX10CONVERT_I8F32_MASK>;
+def X86vmcvtps2hf8  : SDNode<"X86ISD::VMCVTPS2HF8",  SDTAVX10CONVERT_I8F32_MASK>;
+def X86vmcvtps2hf8s : SDNode<"X86ISD::VMCVTPS2HF8S", SDTAVX10CONVERT_I8F32_MASK>;
+def X86vmcvtrops2hf8  : SDNode<"X86ISD::VMCVTROPS2HF8",  SDTAVX10CONVERT_I8F32_MASK>;
+def X86vmcvtrops2hf8s : SDNode<"X86ISD::VMCVTROPS2HF8S", SDTAVX10CONVERT_I8F32_MASK>;
+
+// Group B: Bias PS->8bit (3 operand)
+def X86vcvtbiasps2bf8  : SDNode<"X86ISD::VCVTBIASPS2BF8",  SDTAVX10CONVERT_2I8F32>;
+def X86vcvtbiasps2bf8s : SDNode<"X86ISD::VCVTBIASPS2BF8S", SDTAVX10CONVERT_2I8F32>;
+def X86vcvtbiasps2hf8  : SDNode<"X86ISD::VCVTBIASPS2HF8",  SDTAVX10CONVERT_2I8F32>;
+def X86vcvtbiasps2hf8s : SDNode<"X86ISD::VCVTBIASPS2HF8S", SDTAVX10CONVERT_2I8F32>;
+
+def X86vmcvtbiasps2bf8  : SDNode<"X86ISD::VMCVTBIASPS2BF8",  SDTAVX10CONVERT_2I8F32_MASK>;
+def X86vmcvtbiasps2bf8s : SDNode<"X86ISD::VMCVTBIASPS2BF8S", SDTAVX10CONVERT_2I8F32_MASK>;
+def X86vmcvtbiasps2hf8  : SDNode<"X86ISD::VMCVTBIASPS2HF8",  SDTAVX10CONVERT_2I8F32_MASK>;
+def X86vmcvtbiasps2hf8s : SDNode<"X86ISD::VMCVTBIASPS2HF8S", SDTAVX10CONVERT_2I8F32_MASK>;
+
+// Group C: 8bit->PS expanding
+def X86vcvtbf82ps  : SDNode<"X86ISD::VCVTBF82PS",  SDTAVX10CONVERT_F32I8>;
+def X86vcvthf82ps  : SDNode<"X86ISD::VCVTHF82PS",  SDTAVX10CONVERT_F32I8>;
+
 //===----------------------------------------------------------------------===//
 // SSE pattern fragments
 //===----------------------------------------------------------------------===//
diff --git a/llvm/lib/Target/X86/X86InstrInfo.td b/llvm/lib/Target/X86/X86InstrInfo.td
index 0c4abc2c400f6f..caf8344dc664ce 100644
--- a/llvm/lib/Target/X86/X86InstrInfo.td
+++ b/llvm/lib/Target/X86/X86InstrInfo.td
@@ -85,6 +85,9 @@ include "X86InstrKL.td"
 // AMX instructions
 include "X86InstrAMX.td"
 
+// AVX10 V2 AUX instructions
+include "X86InstrAVX10_V2_AUX.td"
+
 // RAO-INT instructions
 include "X86InstrRAOINT.td"
 
diff --git a/llvm/lib/Target/X86/X86InstrPredicates.td b/llvm/lib/Target/X86/X86InstrPredicates.td
index afca2e6eafd2c5..54633b451e41b2 100644
--- a/llvm/lib/Target/X86/X86InstrPredicates.td
+++ b/llvm/lib/Target/X86/X86InstrPredicates.td
@@ -77,6 +77,7 @@ def HasAVX1Only  : Predicate<"Subtarget->hasAVX() && !Subtarget->hasAVX2()">;
 def HasAVX10_1   : Predicate<"Subtarget->hasAVX10_1()">;
 def HasAVX10_2   : Predicate<"Subtarget->hasAVX10_2()">;
 def NoAVX10_2    : Predicate<"!Subtarget->hasAVX10_2()">;
+def HasAVX10_V2_AUX : Predicate<"Subtarget->hasAVX10_V2_AUX()">;
 def HasAVX512    : Predicate<"Subtarget->hasAVX512()">;
 def UseAVX       : Predicate<"Subtarget->hasAVX() && !Subtarget->hasAVX512()">;
 def UseAVX2      : Predicate<"Subtarget->hasAVX2() && !Subtarget->hasAVX512()">;
diff --git a/llvm/lib/Target/X86/X86IntrinsicsInfo.h b/llvm/lib/Target/X86/X86IntrinsicsInfo.h
index a6b0db0230cf3f..d92834aa5d5c5d 100644
--- a/llvm/lib/Target/X86/X86IntrinsicsInfo.h
+++ b/llvm/lib/Target/X86/X86IntrinsicsInfo.h
@@ -443,6 +443,12 @@ static const IntrinsicData IntrinsicsWithoutChain[] = {
                        X86ISD::VFPROUND2, 0),
     X86_INTRINSIC_DATA(avx10_mask_vcvt2ps2phx_512, INTR_TYPE_2OP_MASK,
                        X86ISD::VFPROUND2, X86ISD::VFPROUND2_RND),
+    X86_INTRINSIC_DATA(avx10_mask_vcvtbf82ps_128, INTR_TYPE_1OP_MASK,
+                       X86ISD::VCVTBF82PS, 0),
+    X86_INTRINSIC_DATA(avx10_mask_vcvtbf82ps_256, INTR_TYPE_1OP_MASK,
+                       X86ISD::VCVTBF82PS, 0),
+    X86_INTRINSIC_DATA(avx10_mask_vcvtbf82ps_512, INTR_TYPE_1OP_MASK,
+                       X86ISD::VCVTBF82PS, 0),
     X86_INTRINSIC_DATA(avx10_mask_vcvtbiasph2bf8128, TRUNCATE2_TO_REG,
                        X86ISD::VCVTBIASPH2BF8, X86ISD::VMCVTBIASPH2BF8),
     X86_INTRINSIC_DATA(avx10_mask_vcvtbiasph2bf8256, INTR_TYPE_2OP_MASK,
@@ -467,12 +473,42 @@ static const IntrinsicData IntrinsicsWithoutChain[] = {
                        X86ISD::VCVTBIASPH2HF8S, 0),
     X86_INTRINSIC_DATA(avx10_mask_vcvtbiasph2hf8s512, INTR_TYPE_2OP_MASK,
                        X86ISD::VCVTBIASPH2HF8S, 0),
+    X86_INTRINSIC_DATA(avx10_mask_vcvtbiasps2bf8_128, TRUNCATE2_TO_REG,
+                       X86ISD::VCVTBIASPS2BF8, X86ISD::VMCVTBIASPS2BF8),
+    X86_INTRINSIC_DATA(avx10_mask_vcvtbiasps2bf8_256, TRUNCATE2_TO_REG,
+                       X86ISD::VCVTBIASPS2BF8, X86ISD::VMCVTBIASPS2BF8),
+    X86_INTRINSIC_DATA(avx10_mask_vcvtbiasps2bf8_512, INTR_TYPE_2OP_MASK,
+                       X86ISD::VCVTBIASPS2BF8, 0),
+    X86_INTRINSIC_DATA(avx10_mask_vcvtbiasps2bf8s_128, TRUNCATE2_TO_REG,
+                       X86ISD::VCVTBIASPS2BF8S, X86ISD::VMCVTBIASPS2BF8S),
+    X86_INTRINSIC_DATA(avx10_mask_vcvtbiasps2bf8s_256, TRUNCATE2_TO_REG,
+                       X86ISD::VCVTBIASPS2BF8S, X86ISD::VMCVTBIASPS2BF8S),
+    X86_INTRINSIC_DATA(avx10_mask_vcvtbiasps2bf8s_512, INTR_TYPE_2OP_MASK,
+                       X86ISD::VCVTBIASPS2BF8S, 0),
+    X86_INTRINSIC_DATA(avx10_mask_vcvtbiasps2hf8_128, TRUNCATE2_TO_REG,
+                       X86ISD::VCVTBIASPS2HF8, X86ISD::VMCVTBIASPS2HF8),
+    X86_INTRINSIC_DATA(avx10_mask_vcvtbiasps2hf8_256, TRUNCATE2_TO_REG,
+                       X86ISD::VCVTBIASPS2HF8, X86ISD::VMCVTBIASPS2HF8),
+    X86_INTRINSIC_DATA(avx10_mask_vcvtbiasps2hf8_512, INTR_TYPE_2OP_MASK,
+                       X86ISD::VCVTBIASPS2HF8, 0),
+    X86_INTRINSIC_DATA(avx10_mask_vcvtbiasps2hf8s_128, TRUNCATE2_TO_REG,
+                       X86ISD::VCVTBIASPS2HF8S, X86ISD::VMCVTBIASPS2HF8S),
+    X86_INTRINSIC_DATA(avx10_mask_vcvtbiasps2hf8s_256, TRUNCATE2_TO_REG,
+                       X86ISD::VCVTBIASPS2HF8S, X86ISD::VMCVTBIASPS2HF8S),
+    X86_INTRINSIC_DATA(avx10_mask_vcvtbiasps2hf8s_512, INTR_TYPE_2OP_MASK,
+                       X86ISD::VCVTBIASPS2HF8S, 0),
     X86_INTRINSIC_DATA(avx10_mask_vcvthf82ph128, INTR_TYPE_1OP_MASK,
                        X86ISD::VCVTHF82PH, 0),
     X86_INTRINSIC_DATA(avx10_mask_vcvthf82ph256, INTR_TYPE_1OP_MASK,
                        X86ISD::VCVTHF82PH, 0),
     X86_INTRINSIC_DATA(avx10_mask_vcvthf82ph512, INTR_TYPE_1OP_MASK,
                        X86ISD::VCVTHF82PH, 0),
+    X86_INTRINSIC_DATA(avx10_mask_vcvthf82ps_128, INTR_TYPE_1OP_MASK,
+                       X86ISD::VCVTHF82PS, 0),
+    X86_INTRINSIC_DATA(avx10_mask_vcvthf82ps_256, INTR_TYPE_1OP_MASK,
+                       X86ISD::VCVTHF82PS, 0),
+    X86_INTRINSIC_DATA(avx10_mask_vcvthf82ps_512, INTR_TYPE_1OP_MASK,
+                       X86ISD::VCVTHF82PS, 0),
     X86_INTRINSIC_DATA(avx10_mask_vcvtph2bf8128, TRUNCATE_TO_REG,
                        X86ISD::VCVTPH2BF8, X86ISD::VMCVTPH2BF8),
     X86_INTRINSIC_DATA(avx10_mask_vcvtph2bf8256, INTR_TYPE_1OP_MASK,
@@ -509,6 +545,30 @@ static const IntrinsicData IntrinsicsWithoutChain[] = {
                        X86ISD::CVTP2IUBS, 0),
     X86_INTRINSIC_DATA(avx10_mask_vcvtph2iubs512, INTR_TYPE_1OP_MASK,
                        X86ISD::CVTP2IUBS, X86ISD::CVTP2IUBS_RND),
+    X86_INTRINSIC_DATA(avx10_mask_vcvtps2bf8_128, TRUNCATE_TO_REG,
+                       X86ISD::VCVTPS2BF8, X86ISD::VMCVTPS2BF8),
+    X86_INTRINSIC_DATA(avx10_mask_vcvtps2bf8_256, TRUNCATE_TO_REG,
+                       X86ISD::VCVTPS2BF8, X86ISD::VMCVTPS2BF8),
+    X86_INTRINSIC_DATA(avx10_mask_vcvtps2bf8_512, INTR_TYPE_1OP_MASK,
+                       X86ISD::VCVTPS2BF8, 0),
+    X86_INTRINSIC_DATA(avx10_mask_vcvtps2bf8s_128, TRUNCATE_TO_REG,
+                       X86ISD::VCVTPS2BF8S, X86ISD::VMCVTPS2BF8S),
+    X86_INTRINSIC_DATA(avx10_mask_vcvtps2bf8s_256, TRUNCATE_TO_REG,
+                       X86ISD::VCVTPS2BF8S, X86ISD::VMCVTPS2BF8S),
+    X86_INTRINSIC_DATA(avx10_mask_vcvtps2bf8s_512, INTR_TYPE_1OP_MASK,
+                       X86ISD::VCVTPS2BF8S, 0),
+    X86_INTRINSIC_DATA(avx10_mask_vcvtps2hf8_128, TRUNCATE_TO_REG,
+                       X86ISD::VCVTPS2HF8, X86ISD::VMCVTPS2HF8),
+    X86_INTRINSIC_DATA(avx10_mask_vcvtps2hf8_256, TRUNCATE_TO_REG,
+                       X86ISD::VCVTPS2HF8, X86ISD::VMCVTPS2HF8),
+    X86_INTRINSIC_DATA(avx10_mask_vcvtps2hf8_512, INTR_TYPE_1OP_MASK,
+                       X86ISD::VCVTPS2HF8, 0),
+    X86_INTRINSIC_DATA(avx10_mask_vcvtps2hf8s_128, TRUNCATE_TO_REG,
+                       X86ISD::VCVTPS2HF8S, X86ISD::VMCVTPS2HF8S),
+    X86_INTRINSIC_DATA(avx10_mask_vcvtps2hf8s_256, TRUNCATE_TO_REG,
+                       X86ISD::VCVTPS2HF8S, X86ISD::VMCVTPS2HF8S),
+    X86_INTRINSIC_DATA(avx10_mask_vcvtps2hf8s_512, INTR_TYPE_1OP_MASK,
+                       X86ISD::VCVTPS2HF8S, 0),
     X86_INTRINSIC_DATA(avx10_mask_vcvtps2ibs128, INTR_TYPE_1OP_MASK,
                        X86ISD::CVTP2IBS, 0),
     X86_INTRINSIC_DATA(avx10_mask_vcvtps2ibs256, INTR_TYPE_1OP_MASK,
@@ -521,6 +581,18 @@ static const IntrinsicData IntrinsicsWithoutChain[] = {
                        X86ISD::CVTP2IUBS, 0),
     X86_INTRINSIC_DATA(avx10_mask_vcvtps2iubs512, INTR_TYPE_1OP_MASK,
                        X86ISD::CVTP2IUBS, X86ISD::CVTP2IUBS_RND),
+    X86_INTRINSIC_DATA(avx10_mask_vcvtrops2hf8_128, TRUNCATE_TO_REG,
+                       X86ISD::VCVTROPS2HF8, X86ISD::VMCVTROPS2HF8),
+    X86_INTRINSIC_DATA(avx10_mask_vcvtrops2hf8_256, TRUNCATE_TO_REG,
+                       X86ISD::VCVTROPS2HF8, X86ISD::VMCVTROPS2HF8),
+    X86_INTRINSIC_DATA(avx10_mask_vcvtrops2hf8_512, INTR_TYPE_1OP_MASK,
+                       X86ISD::VCVTROPS2HF8, 0),
+    X86_INTRINSIC_DATA(avx10_mask_vcvtrops2hf8s_128, TRUNCATE_TO_REG,
+                       X86ISD::VCVTROPS2HF8S, X86ISD::VMCVTROPS2HF8S),
+    X86_INTRINSIC_DATA(avx10_mask_vcvtrops2hf8s_256, TRUNCATE_TO_REG,
+                       X86ISD::VCVTROPS2HF8S, X86ISD::VMCVTROPS2HF8S),
+    X86_INTRINSIC_DATA(avx10_mask_vcvtrops2hf8s_512, INTR_TYPE_1OP_MASK,
+                       X86ISD::VCVTROPS2HF8S, 0),
     X86_INTRINSIC_DATA(avx10_mask_vcvttpd2dqs_128, CVTPD2DQ_MASK,
                        X86ISD::CVTTP2SIS, X86ISD::MCVTTP2SIS),
     X86_INTRINSIC_DATA(avx10_mask_vcvttpd2dqs_256, INTR_TYPE_1OP_MASK,
diff --git a/llvm/lib/TargetParser/X86TargetParser.cpp b/llvm/lib/TargetParser/X86TargetParser.cpp
index 5d62d4bb8603ff..a8f7c01c694270 100644
--- a/llvm/lib/TargetParser/X86TargetParser.cpp
+++ b/llvm/lib/TargetParser/X86TargetParser.cpp
@@ -669,6 +669,7 @@ constexpr FeatureBitset ImpliedFeaturesAVX10_1 =
     FeatureAVX512VBMI2 | FeatureAVX512BITALG | FeatureAVX512FP16 |
     FeatureAVX512DQ | FeatureAVX512VL;
 constexpr FeatureBitset ImpliedFeaturesAVX10_2 = FeatureAVX10_1;
+constexpr FeatureBitset ImpliedFeaturesAVX10_V2_AUX = FeatureAVX10_2;
 
 // APX Features
 constexpr FeatureBitset ImpliedFeaturesEGPR = {};
diff --git a/llvm/test/CodeGen/X86/avx10_2_v2aux-intrinsics.ll b/llvm/test/CodeGen/X86/avx10_2_v2aux-intrinsics.ll
new file mode 100644
index 00000000000000..980ba1447d5f37
--- /dev/null
+++ b/llvm/test/CodeGen/X86/avx10_2_v2aux-intrinsics.ll
@@ -0,0 +1,1931 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
+; RUN: llc < %s -verify-machineinstrs -mtriple=x86_64-unknown-unknown --show-mc-encoding -mattr=+avx10-v2-aux | FileCheck %s --check-prefixes=CHECK,X64
+; RUN: llc < %s -verify-machineinstrs -mtriple=i686-unknown-unknown --show-mc-encoding -mattr=+avx10-v2-aux | FileCheck %s --check-prefixes=CHECK,X86
+
+; ===== Group A: 1-operand truncating conversions (PS->i8) =====
+
+; --- vcvtps2bf8 ---
+
+define <16 x i8> @test_int_x86_avx10_vcvtps2bf8_128(<4 x float> %A) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtps2bf8_128:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtps2bf8 %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x39,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.128(<4 x float> %A, <16 x i8> zeroinitializer, i8 -1)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtps2bf8_256(<8 x float> %A) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtps2bf8_256:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtps2bf8 %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x39,0xc0]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.256(<8 x float> %A, <16 x i8> zeroinitializer, i8 -1)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtps2bf8_512(<16 x float> %A) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtps2bf8_512:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtps2bf8 %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x39,0xc0]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.512(<16 x float> %A, <16 x i8> zeroinitializer, i16 -1)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtps2bf8_128(<4 x float> %A, <16 x i8> %B, i8 %C) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtps2bf8_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtps2bf8 %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x39,0xc8]
+; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtps2bf8_128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtps2bf8 %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x39,0xc8]
+; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.128(<4 x float> %A, <16 x i8> %B, i8 %C)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtps2bf8_256(<8 x float> %A, <16 x i8> %B, i8 %C) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtps2bf8_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtps2bf8 %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x39,0xc8]
+; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtps2bf8_256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtps2bf8 %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x39,0xc8]
+; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.256(<8 x float> %A, <16 x i8> %B, i8 %C)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtps2bf8_512(<16 x float> %A, <16 x i8> %B, i16 %C) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtps2bf8_512:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtps2bf8 %zmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x49,0x39,0xc8]
+; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtps2bf8_512:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtps2bf8 %zmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x49,0x39,0xc8]
+; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.512(<16 x float> %A, <16 x i8> %B, i16 %C)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2bf8_128(<4 x float> %A, i8 %B) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtps2bf8 %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x89,0x39,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8_128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtps2bf8 %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x89,0x39,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.128(<4 x float> %A, <16 x i8> zeroinitializer, i8 %B)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2bf8_256(<8 x float> %A, i8 %B) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtps2bf8 %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xa9,0x39,0xc0]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8_256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtps2bf8 %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xa9,0x39,0xc0]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.256(<8 x float> %A, <16 x i8> zeroinitializer, i8 %B)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2bf8_512(<16 x float> %A, i16 %B) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8_512:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtps2bf8 %zmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xc9,0x39,0xc0]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8_512:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtps2bf8 %zmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xc9,0x39,0xc0]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.512(<16 x float> %A, <16 x i8> zeroinitializer, i16 %B)
+  ret <16 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.128(<4 x float>, <16 x i8>, i8)
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.256(<8 x float>, <16 x i8>, i8)
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.512(<16 x float>, <16 x i8>, i16)
+
+; --- vcvtps2bf8s ---
+
+define <16 x i8> @test_int_x86_avx10_vcvtps2bf8s_128(<4 x float> %A) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtps2bf8s_128:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtps2bf8s %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x3b,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.128(<4 x float> %A, <16 x i8> zeroinitializer, i8 -1)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtps2bf8s_256(<8 x float> %A) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtps2bf8s_256:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtps2bf8s %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x3b,0xc0]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.256(<8 x float> %A, <16 x i8> zeroinitializer, i8 -1)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtps2bf8s_512(<16 x float> %A) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtps2bf8s_512:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtps2bf8s %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x3b,0xc0]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.512(<16 x float> %A, <16 x i8> zeroinitializer, i16 -1)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtps2bf8s_128(<4 x float> %A, <16 x i8> %B, i8 %C) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtps2bf8s_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtps2bf8s %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x3b,0xc8]
+; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtps2bf8s_128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtps2bf8s %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x3b,0xc8]
+; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.128(<4 x float> %A, <16 x i8> %B, i8 %C)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtps2bf8s_256(<8 x float> %A, <16 x i8> %B, i8 %C) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtps2bf8s_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtps2bf8s %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x3b,0xc8]
+; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtps2bf8s_256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtps2bf8s %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x3b,0xc8]
+; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.256(<8 x float> %A, <16 x i8> %B, i8 %C)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtps2bf8s_512(<16 x float> %A, <16 x i8> %B, i16 %C) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtps2bf8s_512:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtps2bf8s %zmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x49,0x3b,0xc8]
+; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtps2bf8s_512:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtps2bf8s %zmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x49,0x3b,0xc8]
+; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.512(<16 x float> %A, <16 x i8> %B, i16 %C)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2bf8s_128(<4 x float> %A, i8 %B) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8s_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtps2bf8s %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x89,0x3b,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8s_128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtps2bf8s %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x89,0x3b,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.128(<4 x float> %A, <16 x i8> zeroinitializer, i8 %B)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2bf8s_256(<8 x float> %A, i8 %B) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8s_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtps2bf8s %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xa9,0x3b,0xc0]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8s_256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtps2bf8s %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xa9,0x3b,0xc0]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.256(<8 x float> %A, <16 x i8> zeroinitializer, i8 %B)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2bf8s_512(<16 x float> %A, i16 %B) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8s_512:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtps2bf8s %zmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xc9,0x3b,0xc0]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8s_512:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtps2bf8s %zmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xc9,0x3b,0xc0]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.512(<16 x float> %A, <16 x i8> zeroinitializer, i16 %B)
+  ret <16 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.128(<4 x float>, <16 x i8>, i8)
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.256(<8 x float>, <16 x i8>, i8)
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.512(<16 x float>, <16 x i8>, i16)
+
+; --- vcvtps2hf8 ---
+
+define <16 x i8> @test_int_x86_avx10_vcvtps2hf8_128(<4 x float> %A) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtps2hf8_128:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtps2hf8 %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x38,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.128(<4 x float> %A, <16 x i8> zeroinitializer, i8 -1)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtps2hf8_256(<8 x float> %A) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtps2hf8_256:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtps2hf8 %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x38,0xc0]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.256(<8 x float> %A, <16 x i8> zeroinitializer, i8 -1)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtps2hf8_512(<16 x float> %A) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtps2hf8_512:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtps2hf8 %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x38,0xc0]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.512(<16 x float> %A, <16 x i8> zeroinitializer, i16 -1)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtps2hf8_128(<4 x float> %A, <16 x i8> %B, i8 %C) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtps2hf8_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtps2hf8 %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x38,0xc8]
+; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtps2hf8_128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtps2hf8 %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x38,0xc8]
+; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.128(<4 x float> %A, <16 x i8> %B, i8 %C)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtps2hf8_256(<8 x float> %A, <16 x i8> %B, i8 %C) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtps2hf8_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtps2hf8 %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x38,0xc8]
+; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtps2hf8_256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtps2hf8 %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x38,0xc8]
+; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.256(<8 x float> %A, <16 x i8> %B, i8 %C)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtps2hf8_512(<16 x float> %A, <16 x i8> %B, i16 %C) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtps2hf8_512:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtps2hf8 %zmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x49,0x38,0xc8]
+; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtps2hf8_512:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtps2hf8 %zmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x49,0x38,0xc8]
+; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.512(<16 x float> %A, <16 x i8> %B, i16 %C)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2hf8_128(<4 x float> %A, i8 %B) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtps2hf8 %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x89,0x38,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8_128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtps2hf8 %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x89,0x38,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.128(<4 x float> %A, <16 x i8> zeroinitializer, i8 %B)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2hf8_256(<8 x float> %A, i8 %B) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtps2hf8 %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xa9,0x38,0xc0]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8_256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtps2hf8 %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xa9,0x38,0xc0]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.256(<8 x float> %A, <16 x i8> zeroinitializer, i8 %B)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2hf8_512(<16 x float> %A, i16 %B) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8_512:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtps2hf8 %zmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xc9,0x38,0xc0]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8_512:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtps2hf8 %zmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xc9,0x38,0xc0]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.512(<16 x float> %A, <16 x i8> zeroinitializer, i16 %B)
+  ret <16 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.128(<4 x float>, <16 x i8>, i8)
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.256(<8 x float>, <16 x i8>, i8)
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.512(<16 x float>, <16 x i8>, i16)
+
+; --- vcvtps2hf8s ---
+
+define <16 x i8> @test_int_x86_avx10_vcvtps2hf8s_128(<4 x float> %A) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtps2hf8s_128:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtps2hf8s %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x3a,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.128(<4 x float> %A, <16 x i8> zeroinitializer, i8 -1)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtps2hf8s_256(<8 x float> %A) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtps2hf8s_256:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtps2hf8s %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x3a,0xc0]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.256(<8 x float> %A, <16 x i8> zeroinitializer, i8 -1)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtps2hf8s_512(<16 x float> %A) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtps2hf8s_512:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtps2hf8s %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x3a,0xc0]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.512(<16 x float> %A, <16 x i8> zeroinitializer, i16 -1)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtps2hf8s_128(<4 x float> %A, <16 x i8> %B, i8 %C) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtps2hf8s_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtps2hf8s %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x3a,0xc8]
+; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtps2hf8s_128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtps2hf8s %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x3a,0xc8]
+; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.128(<4 x float> %A, <16 x i8> %B, i8 %C)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtps2hf8s_256(<8 x float> %A, <16 x i8> %B, i8 %C) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtps2hf8s_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtps2hf8s %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x3a,0xc8]
+; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtps2hf8s_256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtps2hf8s %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x3a,0xc8]
+; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.256(<8 x float> %A, <16 x i8> %B, i8 %C)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtps2hf8s_512(<16 x float> %A, <16 x i8> %B, i16 %C) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtps2hf8s_512:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtps2hf8s %zmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x49,0x3a,0xc8]
+; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtps2hf8s_512:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtps2hf8s %zmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x49,0x3a,0xc8]
+; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.512(<16 x float> %A, <16 x i8> %B, i16 %C)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2hf8s_128(<4 x float> %A, i8 %B) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8s_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtps2hf8s %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x89,0x3a,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8s_128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtps2hf8s %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x89,0x3a,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.128(<4 x float> %A, <16 x i8> zeroinitializer, i8 %B)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2hf8s_256(<8 x float> %A, i8 %B) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8s_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtps2hf8s %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xa9,0x3a,0xc0]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8s_256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtps2hf8s %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xa9,0x3a,0xc0]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.256(<8 x float> %A, <16 x i8> zeroinitializer, i8 %B)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2hf8s_512(<16 x float> %A, i16 %B) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8s_512:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtps2hf8s %zmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xc9,0x3a,0xc0]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8s_512:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtps2hf8s %zmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xc9,0x3a,0xc0]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.512(<16 x float> %A, <16 x i8> zeroinitializer, i16 %B)
+  ret <16 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.128(<4 x float>, <16 x i8>, i8)
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.256(<8 x float>, <16 x i8>, i8)
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.512(<16 x float>, <16 x i8>, i16)
+
+; --- vcvtrops2hf8 ---
+
+define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8_128(<4 x float> %A) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtrops2hf8_128:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtrops2hf8 %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7d,0x08,0x38,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.128(<4 x float> %A, <16 x i8> zeroinitializer, i8 -1)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8_256(<8 x float> %A) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtrops2hf8_256:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtrops2hf8 %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7d,0x28,0x38,0xc0]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.256(<8 x float> %A, <16 x i8> zeroinitializer, i8 -1)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8_512(<16 x float> %A) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtrops2hf8_512:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtrops2hf8 %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7d,0x48,0x38,0xc0]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.512(<16 x float> %A, <16 x i8> zeroinitializer, i16 -1)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtrops2hf8_128(<4 x float> %A, <16 x i8> %B, i8 %C) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtrops2hf8 %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7d,0x09,0x38,0xc8]
+; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8_128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtrops2hf8 %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7d,0x09,0x38,0xc8]
+; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.128(<4 x float> %A, <16 x i8> %B, i8 %C)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtrops2hf8_256(<8 x float> %A, <16 x i8> %B, i8 %C) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtrops2hf8 %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7d,0x29,0x38,0xc8]
+; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8_256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtrops2hf8 %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7d,0x29,0x38,0xc8]
+; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.256(<8 x float> %A, <16 x i8> %B, i8 %C)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtrops2hf8_512(<16 x float> %A, <16 x i8> %B, i16 %C) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8_512:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtrops2hf8 %zmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7d,0x49,0x38,0xc8]
+; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8_512:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtrops2hf8 %zmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7d,0x49,0x38,0xc8]
+; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.512(<16 x float> %A, <16 x i8> %B, i16 %C)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtrops2hf8_128(<4 x float> %A, i8 %B) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtrops2hf8 %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0x89,0x38,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8_128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtrops2hf8 %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0x89,0x38,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.128(<4 x float> %A, <16 x i8> zeroinitializer, i8 %B)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtrops2hf8_256(<8 x float> %A, i8 %B) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtrops2hf8 %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0xa9,0x38,0xc0]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8_256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtrops2hf8 %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0xa9,0x38,0xc0]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.256(<8 x float> %A, <16 x i8> zeroinitializer, i8 %B)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtrops2hf8_512(<16 x float> %A, i16 %B) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8_512:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtrops2hf8 %zmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0xc9,0x38,0xc0]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8_512:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtrops2hf8 %zmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0xc9,0x38,0xc0]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.512(<16 x float> %A, <16 x i8> zeroinitializer, i16 %B)
+  ret <16 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.128(<4 x float>, <16 x i8>, i8)
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.256(<8 x float>, <16 x i8>, i8)
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.512(<16 x float>, <16 x i8>, i16)
+
+; --- vcvtrops2hf8s ---
+
+define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8s_128(<4 x float> %A) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtrops2hf8s_128:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtrops2hf8s %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7d,0x08,0x3a,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.128(<4 x float> %A, <16 x i8> zeroinitializer, i8 -1)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8s_256(<8 x float> %A) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtrops2hf8s_256:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtrops2hf8s %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7d,0x28,0x3a,0xc0]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.256(<8 x float> %A, <16 x i8> zeroinitializer, i8 -1)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8s_512(<16 x float> %A) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtrops2hf8s_512:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtrops2hf8s %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7d,0x48,0x3a,0xc0]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.512(<16 x float> %A, <16 x i8> zeroinitializer, i16 -1)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtrops2hf8s_128(<4 x float> %A, <16 x i8> %B, i8 %C) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8s_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtrops2hf8s %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7d,0x09,0x3a,0xc8]
+; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8s_128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtrops2hf8s %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7d,0x09,0x3a,0xc8]
+; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.128(<4 x float> %A, <16 x i8> %B, i8 %C)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtrops2hf8s_256(<8 x float> %A, <16 x i8> %B, i8 %C) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8s_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtrops2hf8s %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7d,0x29,0x3a,0xc8]
+; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8s_256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtrops2hf8s %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7d,0x29,0x3a,0xc8]
+; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.256(<8 x float> %A, <16 x i8> %B, i8 %C)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtrops2hf8s_512(<16 x float> %A, <16 x i8> %B, i16 %C) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8s_512:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtrops2hf8s %zmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7d,0x49,0x3a,0xc8]
+; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8s_512:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtrops2hf8s %zmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7d,0x49,0x3a,0xc8]
+; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.512(<16 x float> %A, <16 x i8> %B, i16 %C)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtrops2hf8s_128(<4 x float> %A, i8 %B) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8s_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtrops2hf8s %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0x89,0x3a,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8s_128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtrops2hf8s %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0x89,0x3a,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.128(<4 x float> %A, <16 x i8> zeroinitializer, i8 %B)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtrops2hf8s_256(<8 x float> %A, i8 %B) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8s_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtrops2hf8s %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0xa9,0x3a,0xc0]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8s_256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtrops2hf8s %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0xa9,0x3a,0xc0]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.256(<8 x float> %A, <16 x i8> zeroinitializer, i8 %B)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtrops2hf8s_512(<16 x float> %A, i16 %B) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8s_512:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtrops2hf8s %zmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0xc9,0x3a,0xc0]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8s_512:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtrops2hf8s %zmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0xc9,0x3a,0xc0]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.512(<16 x float> %A, <16 x i8> zeroinitializer, i16 %B)
+  ret <16 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.128(<4 x float>, <16 x i8>, i8)
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.256(<8 x float>, <16 x i8>, i8)
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.512(<16 x float>, <16 x i8>, i16)
+
+; ===== Group B: 3-operand bias conversions (bias+PS->i8) =====
+
+; --- vcvtbiasps2bf8 ---
+
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8_128(<16 x i8> %A, <4 x float> %B) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2bf8_128:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbiasps2bf8 %xmm1, %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x39,0xc1]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.128(<16 x i8> %A, <4 x float> %B, <16 x i8> zeroinitializer, i8 -1)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8_256(<32 x i8> %A, <8 x float> %B) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2bf8_256:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbiasps2bf8 %ymm1, %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x28,0x39,0xc1]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.256(<32 x i8> %A, <8 x float> %B, <16 x i8> zeroinitializer, i8 -1)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8_512(<64 x i8> %A, <16 x float> %B) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2bf8_512:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbiasps2bf8 %zmm1, %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x39,0xc1]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.512(<64 x i8> %A, <16 x float> %B, <16 x i8> zeroinitializer, i16 -1)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2bf8_128(<16 x i8> %A, <4 x float> %B, <16 x i8> %C, i8 %D) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbiasps2bf8 %xmm1, %xmm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x39,0xd1]
+; X64-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8_128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2bf8 %xmm1, %xmm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x39,0xd1]
+; X86-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.128(<16 x i8> %A, <4 x float> %B, <16 x i8> %C, i8 %D)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2bf8_256(<32 x i8> %A, <8 x float> %B, <16 x i8> %C, i8 %D) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbiasps2bf8 %ymm1, %ymm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x39,0xd1]
+; X64-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8_256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2bf8 %ymm1, %ymm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x39,0xd1]
+; X86-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.256(<32 x i8> %A, <8 x float> %B, <16 x i8> %C, i8 %D)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2bf8_512(<64 x i8> %A, <16 x float> %B, <16 x i8> %C, i16 %D) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8_512:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbiasps2bf8 %zmm1, %zmm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x49,0x39,0xd1]
+; X64-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8_512:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2bf8 %zmm1, %zmm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x49,0x39,0xd1]
+; X86-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.512(<64 x i8> %A, <16 x float> %B, <16 x i8> %C, i16 %D)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2bf8_128(<16 x i8> %A, <4 x float> %B, i8 %C) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbiasps2bf8 %xmm1, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x39,0xc1]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8_128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2bf8 %xmm1, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x39,0xc1]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.128(<16 x i8> %A, <4 x float> %B, <16 x i8> zeroinitializer, i8 %C)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2bf8_256(<32 x i8> %A, <8 x float> %B, i8 %C) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbiasps2bf8 %ymm1, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xa9,0x39,0xc1]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8_256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2bf8 %ymm1, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xa9,0x39,0xc1]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.256(<32 x i8> %A, <8 x float> %B, <16 x i8> zeroinitializer, i8 %C)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2bf8_512(<64 x i8> %A, <16 x float> %B, i16 %C) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8_512:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbiasps2bf8 %zmm1, %zmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xc9,0x39,0xc1]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8_512:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2bf8 %zmm1, %zmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xc9,0x39,0xc1]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.512(<64 x i8> %A, <16 x float> %B, <16 x i8> zeroinitializer, i16 %C)
+  ret <16 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.128(<16 x i8>, <4 x float>, <16 x i8>, i8)
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.256(<32 x i8>, <8 x float>, <16 x i8>, i8)
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.512(<64 x i8>, <16 x float>, <16 x i8>, i16)
+
+; --- vcvtbiasps2bf8s ---
+
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8s_128(<16 x i8> %A, <4 x float> %B) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2bf8s_128:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbiasps2bf8s %xmm1, %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x3b,0xc1]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.128(<16 x i8> %A, <4 x float> %B, <16 x i8> zeroinitializer, i8 -1)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8s_256(<32 x i8> %A, <8 x float> %B) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2bf8s_256:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbiasps2bf8s %ymm1, %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x28,0x3b,0xc1]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.256(<32 x i8> %A, <8 x float> %B, <16 x i8> zeroinitializer, i8 -1)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8s_512(<64 x i8> %A, <16 x float> %B) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2bf8s_512:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbiasps2bf8s %zmm1, %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x3b,0xc1]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.512(<64 x i8> %A, <16 x float> %B, <16 x i8> zeroinitializer, i16 -1)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2bf8s_128(<16 x i8> %A, <4 x float> %B, <16 x i8> %C, i8 %D) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8s_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbiasps2bf8s %xmm1, %xmm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x3b,0xd1]
+; X64-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8s_128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2bf8s %xmm1, %xmm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x3b,0xd1]
+; X86-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.128(<16 x i8> %A, <4 x float> %B, <16 x i8> %C, i8 %D)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2bf8s_256(<32 x i8> %A, <8 x float> %B, <16 x i8> %C, i8 %D) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8s_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbiasps2bf8s %ymm1, %ymm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x3b,0xd1]
+; X64-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8s_256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2bf8s %ymm1, %ymm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x3b,0xd1]
+; X86-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.256(<32 x i8> %A, <8 x float> %B, <16 x i8> %C, i8 %D)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2bf8s_512(<64 x i8> %A, <16 x float> %B, <16 x i8> %C, i16 %D) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8s_512:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbiasps2bf8s %zmm1, %zmm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x49,0x3b,0xd1]
+; X64-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8s_512:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2bf8s %zmm1, %zmm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x49,0x3b,0xd1]
+; X86-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.512(<64 x i8> %A, <16 x float> %B, <16 x i8> %C, i16 %D)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2bf8s_128(<16 x i8> %A, <4 x float> %B, i8 %C) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8s_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbiasps2bf8s %xmm1, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x3b,0xc1]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8s_128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2bf8s %xmm1, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x3b,0xc1]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.128(<16 x i8> %A, <4 x float> %B, <16 x i8> zeroinitializer, i8 %C)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2bf8s_256(<32 x i8> %A, <8 x float> %B, i8 %C) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8s_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbiasps2bf8s %ymm1, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xa9,0x3b,0xc1]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8s_256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2bf8s %ymm1, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xa9,0x3b,0xc1]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.256(<32 x i8> %A, <8 x float> %B, <16 x i8> zeroinitializer, i8 %C)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2bf8s_512(<64 x i8> %A, <16 x float> %B, i16 %C) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8s_512:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbiasps2bf8s %zmm1, %zmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xc9,0x3b,0xc1]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8s_512:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2bf8s %zmm1, %zmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xc9,0x3b,0xc1]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.512(<64 x i8> %A, <16 x float> %B, <16 x i8> zeroinitializer, i16 %C)
+  ret <16 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.128(<16 x i8>, <4 x float>, <16 x i8>, i8)
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.256(<32 x i8>, <8 x float>, <16 x i8>, i8)
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.512(<64 x i8>, <16 x float>, <16 x i8>, i16)
+
+; --- vcvtbiasps2hf8 ---
+
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8_128(<16 x i8> %A, <4 x float> %B) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2hf8_128:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbiasps2hf8 %xmm1, %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x38,0xc1]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.128(<16 x i8> %A, <4 x float> %B, <16 x i8> zeroinitializer, i8 -1)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8_256(<32 x i8> %A, <8 x float> %B) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2hf8_256:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbiasps2hf8 %ymm1, %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x28,0x38,0xc1]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.256(<32 x i8> %A, <8 x float> %B, <16 x i8> zeroinitializer, i8 -1)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8_512(<64 x i8> %A, <16 x float> %B) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2hf8_512:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbiasps2hf8 %zmm1, %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x38,0xc1]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.512(<64 x i8> %A, <16 x float> %B, <16 x i8> zeroinitializer, i16 -1)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2hf8_128(<16 x i8> %A, <4 x float> %B, <16 x i8> %C, i8 %D) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbiasps2hf8 %xmm1, %xmm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x38,0xd1]
+; X64-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8_128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2hf8 %xmm1, %xmm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x38,0xd1]
+; X86-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.128(<16 x i8> %A, <4 x float> %B, <16 x i8> %C, i8 %D)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2hf8_256(<32 x i8> %A, <8 x float> %B, <16 x i8> %C, i8 %D) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbiasps2hf8 %ymm1, %ymm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x38,0xd1]
+; X64-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8_256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2hf8 %ymm1, %ymm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x38,0xd1]
+; X86-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.256(<32 x i8> %A, <8 x float> %B, <16 x i8> %C, i8 %D)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2hf8_512(<64 x i8> %A, <16 x float> %B, <16 x i8> %C, i16 %D) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8_512:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbiasps2hf8 %zmm1, %zmm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x49,0x38,0xd1]
+; X64-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8_512:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2hf8 %zmm1, %zmm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x49,0x38,0xd1]
+; X86-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.512(<64 x i8> %A, <16 x float> %B, <16 x i8> %C, i16 %D)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2hf8_128(<16 x i8> %A, <4 x float> %B, i8 %C) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbiasps2hf8 %xmm1, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x38,0xc1]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8_128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2hf8 %xmm1, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x38,0xc1]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.128(<16 x i8> %A, <4 x float> %B, <16 x i8> zeroinitializer, i8 %C)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2hf8_256(<32 x i8> %A, <8 x float> %B, i8 %C) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbiasps2hf8 %ymm1, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xa9,0x38,0xc1]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8_256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2hf8 %ymm1, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xa9,0x38,0xc1]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.256(<32 x i8> %A, <8 x float> %B, <16 x i8> zeroinitializer, i8 %C)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2hf8_512(<64 x i8> %A, <16 x float> %B, i16 %C) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8_512:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbiasps2hf8 %zmm1, %zmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xc9,0x38,0xc1]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8_512:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2hf8 %zmm1, %zmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xc9,0x38,0xc1]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.512(<64 x i8> %A, <16 x float> %B, <16 x i8> zeroinitializer, i16 %C)
+  ret <16 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.128(<16 x i8>, <4 x float>, <16 x i8>, i8)
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.256(<32 x i8>, <8 x float>, <16 x i8>, i8)
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.512(<64 x i8>, <16 x float>, <16 x i8>, i16)
+
+; --- vcvtbiasps2hf8s ---
+
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8s_128(<16 x i8> %A, <4 x float> %B) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2hf8s_128:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbiasps2hf8s %xmm1, %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x3a,0xc1]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.128(<16 x i8> %A, <4 x float> %B, <16 x i8> zeroinitializer, i8 -1)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8s_256(<32 x i8> %A, <8 x float> %B) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2hf8s_256:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbiasps2hf8s %ymm1, %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x28,0x3a,0xc1]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.256(<32 x i8> %A, <8 x float> %B, <16 x i8> zeroinitializer, i8 -1)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8s_512(<64 x i8> %A, <16 x float> %B) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2hf8s_512:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbiasps2hf8s %zmm1, %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x3a,0xc1]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.512(<64 x i8> %A, <16 x float> %B, <16 x i8> zeroinitializer, i16 -1)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2hf8s_128(<16 x i8> %A, <4 x float> %B, <16 x i8> %C, i8 %D) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8s_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbiasps2hf8s %xmm1, %xmm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x3a,0xd1]
+; X64-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8s_128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2hf8s %xmm1, %xmm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x3a,0xd1]
+; X86-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.128(<16 x i8> %A, <4 x float> %B, <16 x i8> %C, i8 %D)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2hf8s_256(<32 x i8> %A, <8 x float> %B, <16 x i8> %C, i8 %D) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8s_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbiasps2hf8s %ymm1, %ymm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x3a,0xd1]
+; X64-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8s_256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2hf8s %ymm1, %ymm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x3a,0xd1]
+; X86-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.256(<32 x i8> %A, <8 x float> %B, <16 x i8> %C, i8 %D)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2hf8s_512(<64 x i8> %A, <16 x float> %B, <16 x i8> %C, i16 %D) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8s_512:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbiasps2hf8s %zmm1, %zmm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x49,0x3a,0xd1]
+; X64-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8s_512:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2hf8s %zmm1, %zmm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x49,0x3a,0xd1]
+; X86-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.512(<64 x i8> %A, <16 x float> %B, <16 x i8> %C, i16 %D)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2hf8s_128(<16 x i8> %A, <4 x float> %B, i8 %C) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8s_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbiasps2hf8s %xmm1, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x3a,0xc1]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8s_128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2hf8s %xmm1, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x3a,0xc1]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.128(<16 x i8> %A, <4 x float> %B, <16 x i8> zeroinitializer, i8 %C)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2hf8s_256(<32 x i8> %A, <8 x float> %B, i8 %C) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8s_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbiasps2hf8s %ymm1, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xa9,0x3a,0xc1]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8s_256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2hf8s %ymm1, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xa9,0x3a,0xc1]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.256(<32 x i8> %A, <8 x float> %B, <16 x i8> zeroinitializer, i8 %C)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2hf8s_512(<64 x i8> %A, <16 x float> %B, i16 %C) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8s_512:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbiasps2hf8s %zmm1, %zmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xc9,0x3a,0xc1]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8s_512:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2hf8s %zmm1, %zmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xc9,0x3a,0xc1]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.512(<64 x i8> %A, <16 x float> %B, <16 x i8> zeroinitializer, i16 %C)
+  ret <16 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.128(<16 x i8>, <4 x float>, <16 x i8>, i8)
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.256(<32 x i8>, <8 x float>, <16 x i8>, i8)
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.512(<64 x i8>, <16 x float>, <16 x i8>, i16)
+
+; ===== Group C: 8bit->PS expanding conversions =====
+
+; --- vcvtbf82ps ---
+
+define <4 x float> @test_int_x86_avx10_vcvtbf82ps_128(<16 x i8> %A) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbf82ps_128:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbf82ps %xmm0, %xmm0 # encoding: [0x62,0xf5,0xfc,0x08,0x36,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <4 x float> @llvm.x86.avx10.mask.vcvtbf82ps.128(<16 x i8> %A, <4 x float> zeroinitializer, i8 -1)
+  ret <4 x float> %ret
+}
+
+define <8 x float> @test_int_x86_avx10_vcvtbf82ps_256(<16 x i8> %A) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbf82ps_256:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbf82ps %xmm0, %ymm0 # encoding: [0x62,0xf5,0xfc,0x28,0x36,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <8 x float> @llvm.x86.avx10.mask.vcvtbf82ps.256(<16 x i8> %A, <8 x float> zeroinitializer, i8 -1)
+  ret <8 x float> %ret
+}
+
+define <16 x float> @test_int_x86_avx10_vcvtbf82ps_512(<16 x i8> %A) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbf82ps_512:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbf82ps %xmm0, %zmm0 # encoding: [0x62,0xf5,0xfc,0x48,0x36,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x float> @llvm.x86.avx10.mask.vcvtbf82ps.512(<16 x i8> %A, <16 x float> zeroinitializer, i16 -1)
+  ret <16 x float> %ret
+}
+
+define <4 x float> @test_int_x86_avx10_mask_vcvtbf82ps_128(<16 x i8> %A, <4 x float> %B, i8 %C) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtbf82ps_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbf82ps %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0xfc,0x09,0x36,0xc8]
+; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtbf82ps_128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbf82ps %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0xfc,0x09,0x36,0xc8]
+; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <4 x float> @llvm.x86.avx10.mask.vcvtbf82ps.128(<16 x i8> %A, <4 x float> %B, i8 %C)
+  ret <4 x float> %ret
+}
+
+define <8 x float> @test_int_x86_avx10_mask_vcvtbf82ps_256(<16 x i8> %A, <8 x float> %B, i8 %C) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtbf82ps_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbf82ps %xmm0, %ymm1 {%k1} # encoding: [0x62,0xf5,0xfc,0x29,0x36,0xc8]
+; X64-NEXT:    vmovaps %ymm1, %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfc,0x28,0xc1]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtbf82ps_256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbf82ps %xmm0, %ymm1 {%k1} # encoding: [0x62,0xf5,0xfc,0x29,0x36,0xc8]
+; X86-NEXT:    vmovaps %ymm1, %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfc,0x28,0xc1]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <8 x float> @llvm.x86.avx10.mask.vcvtbf82ps.256(<16 x i8> %A, <8 x float> %B, i8 %C)
+  ret <8 x float> %ret
+}
+
+define <16 x float> @test_int_x86_avx10_mask_vcvtbf82ps_512(<16 x i8> %A, <16 x float> %B, i16 %C) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtbf82ps_512:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbf82ps %xmm0, %zmm1 {%k1} # encoding: [0x62,0xf5,0xfc,0x49,0x36,0xc8]
+; X64-NEXT:    vmovaps %zmm1, %zmm0 # encoding: [0x62,0xf1,0x7c,0x48,0x28,0xc1]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtbf82ps_512:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbf82ps %xmm0, %zmm1 {%k1} # encoding: [0x62,0xf5,0xfc,0x49,0x36,0xc8]
+; X86-NEXT:    vmovaps %zmm1, %zmm0 # encoding: [0x62,0xf1,0x7c,0x48,0x28,0xc1]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x float> @llvm.x86.avx10.mask.vcvtbf82ps.512(<16 x i8> %A, <16 x float> %B, i16 %C)
+  ret <16 x float> %ret
+}
+
+define <4 x float> @test_int_x86_avx10_maskz_vcvtbf82ps_128(<16 x i8> %A, i8 %B) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtbf82ps_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbf82ps %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0xfc,0x89,0x36,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtbf82ps_128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbf82ps %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0xfc,0x89,0x36,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <4 x float> @llvm.x86.avx10.mask.vcvtbf82ps.128(<16 x i8> %A, <4 x float> zeroinitializer, i8 %B)
+  ret <4 x float> %ret
+}
+
+define <8 x float> @test_int_x86_avx10_maskz_vcvtbf82ps_256(<16 x i8> %A, i8 %B) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtbf82ps_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbf82ps %xmm0, %ymm0 {%k1} {z} # encoding: [0x62,0xf5,0xfc,0xa9,0x36,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtbf82ps_256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbf82ps %xmm0, %ymm0 {%k1} {z} # encoding: [0x62,0xf5,0xfc,0xa9,0x36,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <8 x float> @llvm.x86.avx10.mask.vcvtbf82ps.256(<16 x i8> %A, <8 x float> zeroinitializer, i8 %B)
+  ret <8 x float> %ret
+}
+
+define <16 x float> @test_int_x86_avx10_maskz_vcvtbf82ps_512(<16 x i8> %A, i16 %B) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtbf82ps_512:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbf82ps %xmm0, %zmm0 {%k1} {z} # encoding: [0x62,0xf5,0xfc,0xc9,0x36,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtbf82ps_512:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbf82ps %xmm0, %zmm0 {%k1} {z} # encoding: [0x62,0xf5,0xfc,0xc9,0x36,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x float> @llvm.x86.avx10.mask.vcvtbf82ps.512(<16 x i8> %A, <16 x float> zeroinitializer, i16 %B)
+  ret <16 x float> %ret
+}
+
+declare <4 x float> @llvm.x86.avx10.mask.vcvtbf82ps.128(<16 x i8>, <4 x float>, i8)
+declare <8 x float> @llvm.x86.avx10.mask.vcvtbf82ps.256(<16 x i8>, <8 x float>, i8)
+declare <16 x float> @llvm.x86.avx10.mask.vcvtbf82ps.512(<16 x i8>, <16 x float>, i16)
+
+; --- vcvthf82ps ---
+
+define <4 x float> @test_int_x86_avx10_vcvthf82ps_128(<16 x i8> %A) {
+; CHECK-LABEL: test_int_x86_avx10_vcvthf82ps_128:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvthf82ps %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x36,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <4 x float> @llvm.x86.avx10.mask.vcvthf82ps.128(<16 x i8> %A, <4 x float> zeroinitializer, i8 -1)
+  ret <4 x float> %ret
+}
+
+define <8 x float> @test_int_x86_avx10_vcvthf82ps_256(<16 x i8> %A) {
+; CHECK-LABEL: test_int_x86_avx10_vcvthf82ps_256:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvthf82ps %xmm0, %ymm0 # encoding: [0x62,0xf5,0x7c,0x28,0x36,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <8 x float> @llvm.x86.avx10.mask.vcvthf82ps.256(<16 x i8> %A, <8 x float> zeroinitializer, i8 -1)
+  ret <8 x float> %ret
+}
+
+define <16 x float> @test_int_x86_avx10_vcvthf82ps_512(<16 x i8> %A) {
+; CHECK-LABEL: test_int_x86_avx10_vcvthf82ps_512:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvthf82ps %xmm0, %zmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x36,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x float> @llvm.x86.avx10.mask.vcvthf82ps.512(<16 x i8> %A, <16 x float> zeroinitializer, i16 -1)
+  ret <16 x float> %ret
+}
+
+define <4 x float> @test_int_x86_avx10_mask_vcvthf82ps_128(<16 x i8> %A, <4 x float> %B, i8 %C) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvthf82ps_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvthf82ps %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x36,0xc8]
+; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvthf82ps_128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvthf82ps %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x36,0xc8]
+; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <4 x float> @llvm.x86.avx10.mask.vcvthf82ps.128(<16 x i8> %A, <4 x float> %B, i8 %C)
+  ret <4 x float> %ret
+}
+
+define <8 x float> @test_int_x86_avx10_mask_vcvthf82ps_256(<16 x i8> %A, <8 x float> %B, i8 %C) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvthf82ps_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvthf82ps %xmm0, %ymm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x36,0xc8]
+; X64-NEXT:    vmovaps %ymm1, %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfc,0x28,0xc1]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvthf82ps_256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvthf82ps %xmm0, %ymm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x36,0xc8]
+; X86-NEXT:    vmovaps %ymm1, %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfc,0x28,0xc1]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <8 x float> @llvm.x86.avx10.mask.vcvthf82ps.256(<16 x i8> %A, <8 x float> %B, i8 %C)
+  ret <8 x float> %ret
+}
+
+define <16 x float> @test_int_x86_avx10_mask_vcvthf82ps_512(<16 x i8> %A, <16 x float> %B, i16 %C) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvthf82ps_512:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvthf82ps %xmm0, %zmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x49,0x36,0xc8]
+; X64-NEXT:    vmovaps %zmm1, %zmm0 # encoding: [0x62,0xf1,0x7c,0x48,0x28,0xc1]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvthf82ps_512:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvthf82ps %xmm0, %zmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x49,0x36,0xc8]
+; X86-NEXT:    vmovaps %zmm1, %zmm0 # encoding: [0x62,0xf1,0x7c,0x48,0x28,0xc1]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x float> @llvm.x86.avx10.mask.vcvthf82ps.512(<16 x i8> %A, <16 x float> %B, i16 %C)
+  ret <16 x float> %ret
+}
+
+define <4 x float> @test_int_x86_avx10_maskz_vcvthf82ps_128(<16 x i8> %A, i8 %B) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvthf82ps_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvthf82ps %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x36,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvthf82ps_128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvthf82ps %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x36,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <4 x float> @llvm.x86.avx10.mask.vcvthf82ps.128(<16 x i8> %A, <4 x float> zeroinitializer, i8 %B)
+  ret <4 x float> %ret
+}
+
+define <8 x float> @test_int_x86_avx10_maskz_vcvthf82ps_256(<16 x i8> %A, i8 %B) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvthf82ps_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvthf82ps %xmm0, %ymm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xa9,0x36,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvthf82ps_256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvthf82ps %xmm0, %ymm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xa9,0x36,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <8 x float> @llvm.x86.avx10.mask.vcvthf82ps.256(<16 x i8> %A, <8 x float> zeroinitializer, i8 %B)
+  ret <8 x float> %ret
+}
+
+define <16 x float> @test_int_x86_avx10_maskz_vcvthf82ps_512(<16 x i8> %A, i16 %B) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvthf82ps_512:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvthf82ps %xmm0, %zmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xc9,0x36,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvthf82ps_512:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvthf82ps %xmm0, %zmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xc9,0x36,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x float> @llvm.x86.avx10.mask.vcvthf82ps.512(<16 x i8> %A, <16 x float> zeroinitializer, i16 %B)
+  ret <16 x float> %ret
+}
+
+declare <4 x float> @llvm.x86.avx10.mask.vcvthf82ps.128(<16 x i8>, <4 x float>, i8)
+declare <8 x float> @llvm.x86.avx10.mask.vcvthf82ps.256(<16 x i8>, <8 x float>, i8)
+declare <16 x float> @llvm.x86.avx10.mask.vcvthf82ps.512(<16 x i8>, <16 x float>, i16)
+
+; ===== Group E: Same-size reg-only conversions =====
+
+; --- vcvtbf82bf6s ---
+
+define <16 x i8> @test_int_x86_avx10_vcvtbf82bf6s_128(<16 x i8> %A) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbf82bf6s_128:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbf82bf6s %xmm0, %xmm0 # encoding: [0x62,0xf5,0xfe,0x08,0x3e,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf82bf6s.128(<16 x i8> %A)
+  ret <16 x i8> %ret
+}
+
+define <32 x i8> @test_int_x86_avx10_vcvtbf82bf6s_256(<32 x i8> %A) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbf82bf6s_256:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbf82bf6s %ymm0, %ymm0 # encoding: [0x62,0xf5,0xfe,0x28,0x3e,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <32 x i8> @llvm.x86.avx10.vcvtbf82bf6s.256(<32 x i8> %A)
+  ret <32 x i8> %ret
+}
+
+define <64 x i8> @test_int_x86_avx10_vcvtbf82bf6s_512(<64 x i8> %A) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbf82bf6s_512:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbf82bf6s %zmm0, %zmm0 # encoding: [0x62,0xf5,0xfe,0x48,0x3e,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <64 x i8> @llvm.x86.avx10.vcvtbf82bf6s.512(<64 x i8> %A)
+  ret <64 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.vcvtbf82bf6s.128(<16 x i8>)
+declare <32 x i8> @llvm.x86.avx10.vcvtbf82bf6s.256(<32 x i8>)
+declare <64 x i8> @llvm.x86.avx10.vcvtbf82bf6s.512(<64 x i8>)
+
+; --- vcvthf82hf6s ---
+
+define <16 x i8> @test_int_x86_avx10_vcvthf82hf6s_128(<16 x i8> %A) {
+; CHECK-LABEL: test_int_x86_avx10_vcvthf82hf6s_128:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvthf82hf6s %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x3c,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvthf82hf6s.128(<16 x i8> %A)
+  ret <16 x i8> %ret
+}
+
+define <32 x i8> @test_int_x86_avx10_vcvthf82hf6s_256(<32 x i8> %A) {
+; CHECK-LABEL: test_int_x86_avx10_vcvthf82hf6s_256:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvthf82hf6s %ymm0, %ymm0 # encoding: [0x62,0xf5,0x7e,0x28,0x3c,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <32 x i8> @llvm.x86.avx10.vcvthf82hf6s.256(<32 x i8> %A)
+  ret <32 x i8> %ret
+}
+
+define <64 x i8> @test_int_x86_avx10_vcvthf82hf6s_512(<64 x i8> %A) {
+; CHECK-LABEL: test_int_x86_avx10_vcvthf82hf6s_512:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvthf82hf6s %zmm0, %zmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x3c,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <64 x i8> @llvm.x86.avx10.vcvthf82hf6s.512(<64 x i8> %A)
+  ret <64 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.vcvthf82hf6s.128(<16 x i8>)
+declare <32 x i8> @llvm.x86.avx10.vcvthf82hf6s.256(<32 x i8>)
+declare <64 x i8> @llvm.x86.avx10.vcvthf82hf6s.512(<64 x i8>)
+
+; ===== Group F: Expanding/same-size conversions =====
+
+; --- vcvtbf42hf8 ---
+
+define <16 x i8> @test_int_x86_avx10_vcvtbf42hf8_128(<16 x i8> %A) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbf42hf8_128:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbf42hf8 %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x37,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf42hf8.128(<16 x i8> %A)
+  ret <16 x i8> %ret
+}
+
+define <32 x i8> @test_int_x86_avx10_vcvtbf42hf8_256(<16 x i8> %A) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbf42hf8_256:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbf42hf8 %xmm0, %ymm0 # encoding: [0x62,0xf5,0x7c,0x28,0x37,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <32 x i8> @llvm.x86.avx10.vcvtbf42hf8.256(<16 x i8> %A)
+  ret <32 x i8> %ret
+}
+
+define <64 x i8> @test_int_x86_avx10_vcvtbf42hf8_512(<32 x i8> %A) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbf42hf8_512:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbf42hf8 %ymm0, %zmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x37,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <64 x i8> @llvm.x86.avx10.vcvtbf42hf8.512(<32 x i8> %A)
+  ret <64 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.vcvtbf42hf8.128(<16 x i8>)
+declare <32 x i8> @llvm.x86.avx10.vcvtbf42hf8.256(<16 x i8>)
+declare <64 x i8> @llvm.x86.avx10.vcvtbf42hf8.512(<32 x i8>)
+
+; --- vcvtbf62hf8 ---
+
+define <16 x i8> @test_int_x86_avx10_vcvtbf62hf8_128(<16 x i8> %A) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbf62hf8_128:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbf62hf8 %xmm0, %xmm0 # encoding: [0x62,0xf5,0xfd,0x08,0x37,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf62hf8.128(<16 x i8> %A)
+  ret <16 x i8> %ret
+}
+
+define <32 x i8> @test_int_x86_avx10_vcvtbf62hf8_256(<32 x i8> %A) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbf62hf8_256:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbf62hf8 %ymm0, %ymm0 # encoding: [0x62,0xf5,0xfd,0x28,0x37,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <32 x i8> @llvm.x86.avx10.vcvtbf62hf8.256(<32 x i8> %A)
+  ret <32 x i8> %ret
+}
+
+define <64 x i8> @test_int_x86_avx10_vcvtbf62hf8_512(<64 x i8> %A) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbf62hf8_512:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbf62hf8 %zmm0, %zmm0 # encoding: [0x62,0xf5,0xfd,0x48,0x37,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <64 x i8> @llvm.x86.avx10.vcvtbf62hf8.512(<64 x i8> %A)
+  ret <64 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.vcvtbf62hf8.128(<16 x i8>)
+declare <32 x i8> @llvm.x86.avx10.vcvtbf62hf8.256(<32 x i8>)
+declare <64 x i8> @llvm.x86.avx10.vcvtbf62hf8.512(<64 x i8>)
+
+; --- vcvthf62hf8 ---
+
+define <16 x i8> @test_int_x86_avx10_vcvthf62hf8_128(<16 x i8> %A) {
+; CHECK-LABEL: test_int_x86_avx10_vcvthf62hf8_128:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvthf62hf8 %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7d,0x08,0x37,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvthf62hf8.128(<16 x i8> %A)
+  ret <16 x i8> %ret
+}
+
+define <32 x i8> @test_int_x86_avx10_vcvthf62hf8_256(<32 x i8> %A) {
+; CHECK-LABEL: test_int_x86_avx10_vcvthf62hf8_256:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvthf62hf8 %ymm0, %ymm0 # encoding: [0x62,0xf5,0x7d,0x28,0x37,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <32 x i8> @llvm.x86.avx10.vcvthf62hf8.256(<32 x i8> %A)
+  ret <32 x i8> %ret
+}
+
+define <64 x i8> @test_int_x86_avx10_vcvthf62hf8_512(<64 x i8> %A) {
+; CHECK-LABEL: test_int_x86_avx10_vcvthf62hf8_512:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvthf62hf8 %zmm0, %zmm0 # encoding: [0x62,0xf5,0x7d,0x48,0x37,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <64 x i8> @llvm.x86.avx10.vcvthf62hf8.512(<64 x i8> %A)
+  ret <64 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.vcvthf62hf8.128(<16 x i8>)
+declare <32 x i8> @llvm.x86.avx10.vcvthf62hf8.256(<32 x i8>)
+declare <64 x i8> @llvm.x86.avx10.vcvthf62hf8.512(<64 x i8>)
+
+; ===== Group H: VUNPACKB =====
+
+define <16 x i8> @test_int_x86_avx10_vunpackb_128(<16 x i8> %A) {
+; CHECK-LABEL: test_int_x86_avx10_vunpackb_128:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vunpackb $1, %xmm0, %xmm0 # encoding: [0x62,0xf3,0x7c,0x08,0x3d,0xc0,0x01]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vunpackb.128(<16 x i8> %A, i8 1)
+  ret <16 x i8> %ret
+}
+
+define <32 x i8> @test_int_x86_avx10_vunpackb_256(<32 x i8> %A) {
+; CHECK-LABEL: test_int_x86_avx10_vunpackb_256:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vunpackb $2, %ymm0, %ymm0 # encoding: [0x62,0xf3,0x7c,0x28,0x3d,0xc0,0x02]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <32 x i8> @llvm.x86.avx10.vunpackb.256(<32 x i8> %A, i8 2)
+  ret <32 x i8> %ret
+}
+
+define <64 x i8> @test_int_x86_avx10_vunpackb_512(<64 x i8> %A) {
+; CHECK-LABEL: test_int_x86_avx10_vunpackb_512:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vunpackb $3, %zmm0, %zmm0 # encoding: [0x62,0xf3,0x7c,0x48,0x3d,0xc0,0x03]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <64 x i8> @llvm.x86.avx10.vunpackb.512(<64 x i8> %A, i8 3)
+  ret <64 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.vunpackb.128(<16 x i8>, i8)
+declare <32 x i8> @llvm.x86.avx10.vunpackb.256(<32 x i8>, i8)
+declare <64 x i8> @llvm.x86.avx10.vunpackb.512(<64 x i8>, i8)
diff --git a/llvm/test/MC/Disassembler/X86/avx10_v2_aux-32.txt b/llvm/test/MC/Disassembler/X86/avx10_v2_aux-32.txt
new file mode 100644
index 00000000000000..2f301e02e1d4f2
--- /dev/null
+++ b/llvm/test/MC/Disassembler/X86/avx10_v2_aux-32.txt
@@ -0,0 +1,460 @@
+# RUN: llvm-mc --disassemble %s -triple=i386 | FileCheck %s --check-prefixes=ATT
+# RUN: llvm-mc --disassemble %s -triple=i386 --output-asm-variant=1 | FileCheck %s --check-prefixes=INTEL
+
+#
+# Group A: PS->8bit truncating conversions
+#
+
+# vcvtps2bf8
+
+# ATT:   vcvtps2bf8 %zmm1, %xmm0
+# INTEL: vcvtps2bf8 xmm0, zmm1
+0x62,0xf5,0x7e,0x48,0x39,0xc1
+
+# ATT:   vcvtps2bf8 %ymm1, %xmm0
+# INTEL: vcvtps2bf8 xmm0, ymm1
+0x62,0xf5,0x7e,0x28,0x39,0xc1
+
+# ATT:   vcvtps2bf8 %xmm1, %xmm0
+# INTEL: vcvtps2bf8 xmm0, xmm1
+0x62,0xf5,0x7e,0x08,0x39,0xc1
+
+# ATT:   vcvtps2bf8 %zmm1, %xmm0 {%k1}
+# INTEL: vcvtps2bf8 xmm0 {k1}, zmm1
+0x62,0xf5,0x7e,0x49,0x39,0xc1
+
+# ATT:   vcvtps2bf8 %zmm1, %xmm0 {%k1} {z}
+# INTEL: vcvtps2bf8 xmm0 {k1} {z}, zmm1
+0x62,0xf5,0x7e,0xc9,0x39,0xc1
+
+# vcvtps2bf8s
+
+# ATT:   vcvtps2bf8s %zmm1, %xmm0
+# INTEL: vcvtps2bf8s xmm0, zmm1
+0x62,0xf5,0x7e,0x48,0x3b,0xc1
+
+# ATT:   vcvtps2bf8s %ymm1, %xmm0
+# INTEL: vcvtps2bf8s xmm0, ymm1
+0x62,0xf5,0x7e,0x28,0x3b,0xc1
+
+# ATT:   vcvtps2bf8s %xmm1, %xmm0
+# INTEL: vcvtps2bf8s xmm0, xmm1
+0x62,0xf5,0x7e,0x08,0x3b,0xc1
+
+# ATT:   vcvtps2bf8s %zmm1, %xmm0 {%k1}
+# INTEL: vcvtps2bf8s xmm0 {k1}, zmm1
+0x62,0xf5,0x7e,0x49,0x3b,0xc1
+
+# ATT:   vcvtps2bf8s %zmm1, %xmm0 {%k1} {z}
+# INTEL: vcvtps2bf8s xmm0 {k1} {z}, zmm1
+0x62,0xf5,0x7e,0xc9,0x3b,0xc1
+
+# vcvtps2hf8
+
+# ATT:   vcvtps2hf8 %zmm1, %xmm0
+# INTEL: vcvtps2hf8 xmm0, zmm1
+0x62,0xf5,0x7e,0x48,0x38,0xc1
+
+# ATT:   vcvtps2hf8 %ymm1, %xmm0
+# INTEL: vcvtps2hf8 xmm0, ymm1
+0x62,0xf5,0x7e,0x28,0x38,0xc1
+
+# ATT:   vcvtps2hf8 %xmm1, %xmm0
+# INTEL: vcvtps2hf8 xmm0, xmm1
+0x62,0xf5,0x7e,0x08,0x38,0xc1
+
+# ATT:   vcvtps2hf8 %zmm1, %xmm0 {%k1}
+# INTEL: vcvtps2hf8 xmm0 {k1}, zmm1
+0x62,0xf5,0x7e,0x49,0x38,0xc1
+
+# ATT:   vcvtps2hf8 %zmm1, %xmm0 {%k1} {z}
+# INTEL: vcvtps2hf8 xmm0 {k1} {z}, zmm1
+0x62,0xf5,0x7e,0xc9,0x38,0xc1
+
+# vcvtps2hf8s
+
+# ATT:   vcvtps2hf8s %zmm1, %xmm0
+# INTEL: vcvtps2hf8s xmm0, zmm1
+0x62,0xf5,0x7e,0x48,0x3a,0xc1
+
+# ATT:   vcvtps2hf8s %ymm1, %xmm0
+# INTEL: vcvtps2hf8s xmm0, ymm1
+0x62,0xf5,0x7e,0x28,0x3a,0xc1
+
+# ATT:   vcvtps2hf8s %xmm1, %xmm0
+# INTEL: vcvtps2hf8s xmm0, xmm1
+0x62,0xf5,0x7e,0x08,0x3a,0xc1
+
+# ATT:   vcvtps2hf8s %zmm1, %xmm0 {%k1}
+# INTEL: vcvtps2hf8s xmm0 {k1}, zmm1
+0x62,0xf5,0x7e,0x49,0x3a,0xc1
+
+# ATT:   vcvtps2hf8s %zmm1, %xmm0 {%k1} {z}
+# INTEL: vcvtps2hf8s xmm0 {k1} {z}, zmm1
+0x62,0xf5,0x7e,0xc9,0x3a,0xc1
+
+# vcvtrops2hf8
+
+# ATT:   vcvtrops2hf8 %zmm1, %xmm0
+# INTEL: vcvtrops2hf8 xmm0, zmm1
+0x62,0xf5,0x7d,0x48,0x38,0xc1
+
+# ATT:   vcvtrops2hf8 %ymm1, %xmm0
+# INTEL: vcvtrops2hf8 xmm0, ymm1
+0x62,0xf5,0x7d,0x28,0x38,0xc1
+
+# ATT:   vcvtrops2hf8 %xmm1, %xmm0
+# INTEL: vcvtrops2hf8 xmm0, xmm1
+0x62,0xf5,0x7d,0x08,0x38,0xc1
+
+# ATT:   vcvtrops2hf8 %zmm1, %xmm0 {%k1}
+# INTEL: vcvtrops2hf8 xmm0 {k1}, zmm1
+0x62,0xf5,0x7d,0x49,0x38,0xc1
+
+# ATT:   vcvtrops2hf8 %zmm1, %xmm0 {%k1} {z}
+# INTEL: vcvtrops2hf8 xmm0 {k1} {z}, zmm1
+0x62,0xf5,0x7d,0xc9,0x38,0xc1
+
+# vcvtrops2hf8s
+
+# ATT:   vcvtrops2hf8s %zmm1, %xmm0
+# INTEL: vcvtrops2hf8s xmm0, zmm1
+0x62,0xf5,0x7d,0x48,0x3a,0xc1
+
+# ATT:   vcvtrops2hf8s %ymm1, %xmm0
+# INTEL: vcvtrops2hf8s xmm0, ymm1
+0x62,0xf5,0x7d,0x28,0x3a,0xc1
+
+# ATT:   vcvtrops2hf8s %xmm1, %xmm0
+# INTEL: vcvtrops2hf8s xmm0, xmm1
+0x62,0xf5,0x7d,0x08,0x3a,0xc1
+
+# ATT:   vcvtrops2hf8s %zmm1, %xmm0 {%k1}
+# INTEL: vcvtrops2hf8s xmm0 {k1}, zmm1
+0x62,0xf5,0x7d,0x49,0x3a,0xc1
+
+# ATT:   vcvtrops2hf8s %zmm1, %xmm0 {%k1} {z}
+# INTEL: vcvtrops2hf8s xmm0 {k1} {z}, zmm1
+0x62,0xf5,0x7d,0xc9,0x3a,0xc1
+
+#
+# Group B: Bias PS->8bit conversions (3-operand)
+#
+
+# vcvtbiasps2bf8
+
+# ATT:   vcvtbiasps2bf8 %zmm2, %zmm1, %xmm0
+# INTEL: vcvtbiasps2bf8 xmm0, zmm1, zmm2
+0x62,0xf5,0x74,0x48,0x39,0xc2
+
+# ATT:   vcvtbiasps2bf8 %ymm2, %ymm1, %xmm0
+# INTEL: vcvtbiasps2bf8 xmm0, ymm1, ymm2
+0x62,0xf5,0x74,0x28,0x39,0xc2
+
+# ATT:   vcvtbiasps2bf8 %xmm2, %xmm1, %xmm0
+# INTEL: vcvtbiasps2bf8 xmm0, xmm1, xmm2
+0x62,0xf5,0x74,0x08,0x39,0xc2
+
+# ATT:   vcvtbiasps2bf8 %zmm2, %zmm1, %xmm0 {%k1}
+# INTEL: vcvtbiasps2bf8 xmm0 {k1}, zmm1, zmm2
+0x62,0xf5,0x74,0x49,0x39,0xc2
+
+# ATT:   vcvtbiasps2bf8 %zmm2, %zmm1, %xmm0 {%k1} {z}
+# INTEL: vcvtbiasps2bf8 xmm0 {k1} {z}, zmm1, zmm2
+0x62,0xf5,0x74,0xc9,0x39,0xc2
+
+# vcvtbiasps2bf8s
+
+# ATT:   vcvtbiasps2bf8s %zmm2, %zmm1, %xmm0
+# INTEL: vcvtbiasps2bf8s xmm0, zmm1, zmm2
+0x62,0xf5,0x74,0x48,0x3b,0xc2
+
+# ATT:   vcvtbiasps2bf8s %ymm2, %ymm1, %xmm0
+# INTEL: vcvtbiasps2bf8s xmm0, ymm1, ymm2
+0x62,0xf5,0x74,0x28,0x3b,0xc2
+
+# ATT:   vcvtbiasps2bf8s %xmm2, %xmm1, %xmm0
+# INTEL: vcvtbiasps2bf8s xmm0, xmm1, xmm2
+0x62,0xf5,0x74,0x08,0x3b,0xc2
+
+# ATT:   vcvtbiasps2bf8s %zmm2, %zmm1, %xmm0 {%k1}
+# INTEL: vcvtbiasps2bf8s xmm0 {k1}, zmm1, zmm2
+0x62,0xf5,0x74,0x49,0x3b,0xc2
+
+# ATT:   vcvtbiasps2bf8s %zmm2, %zmm1, %xmm0 {%k1} {z}
+# INTEL: vcvtbiasps2bf8s xmm0 {k1} {z}, zmm1, zmm2
+0x62,0xf5,0x74,0xc9,0x3b,0xc2
+
+# vcvtbiasps2hf8
+
+# ATT:   vcvtbiasps2hf8 %zmm2, %zmm1, %xmm0
+# INTEL: vcvtbiasps2hf8 xmm0, zmm1, zmm2
+0x62,0xf5,0x74,0x48,0x38,0xc2
+
+# ATT:   vcvtbiasps2hf8 %ymm2, %ymm1, %xmm0
+# INTEL: vcvtbiasps2hf8 xmm0, ymm1, ymm2
+0x62,0xf5,0x74,0x28,0x38,0xc2
+
+# ATT:   vcvtbiasps2hf8 %xmm2, %xmm1, %xmm0
+# INTEL: vcvtbiasps2hf8 xmm0, xmm1, xmm2
+0x62,0xf5,0x74,0x08,0x38,0xc2
+
+# ATT:   vcvtbiasps2hf8 %zmm2, %zmm1, %xmm0 {%k1}
+# INTEL: vcvtbiasps2hf8 xmm0 {k1}, zmm1, zmm2
+0x62,0xf5,0x74,0x49,0x38,0xc2
+
+# ATT:   vcvtbiasps2hf8 %zmm2, %zmm1, %xmm0 {%k1} {z}
+# INTEL: vcvtbiasps2hf8 xmm0 {k1} {z}, zmm1, zmm2
+0x62,0xf5,0x74,0xc9,0x38,0xc2
+
+# vcvtbiasps2hf8s
+
+# ATT:   vcvtbiasps2hf8s %zmm2, %zmm1, %xmm0
+# INTEL: vcvtbiasps2hf8s xmm0, zmm1, zmm2
+0x62,0xf5,0x74,0x48,0x3a,0xc2
+
+# ATT:   vcvtbiasps2hf8s %ymm2, %ymm1, %xmm0
+# INTEL: vcvtbiasps2hf8s xmm0, ymm1, ymm2
+0x62,0xf5,0x74,0x28,0x3a,0xc2
+
+# ATT:   vcvtbiasps2hf8s %xmm2, %xmm1, %xmm0
+# INTEL: vcvtbiasps2hf8s xmm0, xmm1, xmm2
+0x62,0xf5,0x74,0x08,0x3a,0xc2
+
+# ATT:   vcvtbiasps2hf8s %zmm2, %zmm1, %xmm0 {%k1}
+# INTEL: vcvtbiasps2hf8s xmm0 {k1}, zmm1, zmm2
+0x62,0xf5,0x74,0x49,0x3a,0xc2
+
+# ATT:   vcvtbiasps2hf8s %zmm2, %zmm1, %xmm0 {%k1} {z}
+# INTEL: vcvtbiasps2hf8s xmm0 {k1} {z}, zmm1, zmm2
+0x62,0xf5,0x74,0xc9,0x3a,0xc2
+
+#
+# Group C: 8bit->PS expanding conversions
+#
+
+# vcvtbf82ps
+
+# ATT:   vcvtbf82ps %xmm1, %zmm0
+# INTEL: vcvtbf82ps zmm0, xmm1
+0x62,0xf5,0xfc,0x48,0x36,0xc1
+
+# ATT:   vcvtbf82ps %xmm1, %ymm0
+# INTEL: vcvtbf82ps ymm0, xmm1
+0x62,0xf5,0xfc,0x28,0x36,0xc1
+
+# ATT:   vcvtbf82ps %xmm1, %xmm0
+# INTEL: vcvtbf82ps xmm0, xmm1
+0x62,0xf5,0xfc,0x08,0x36,0xc1
+
+# ATT:   vcvtbf82ps %xmm1, %zmm0 {%k1}
+# INTEL: vcvtbf82ps zmm0 {k1}, xmm1
+0x62,0xf5,0xfc,0x49,0x36,0xc1
+
+# ATT:   vcvtbf82ps %xmm1, %zmm0 {%k1} {z}
+# INTEL: vcvtbf82ps zmm0 {k1} {z}, xmm1
+0x62,0xf5,0xfc,0xc9,0x36,0xc1
+
+# vcvthf82ps
+
+# ATT:   vcvthf82ps %xmm1, %zmm0
+# INTEL: vcvthf82ps zmm0, xmm1
+0x62,0xf5,0x7c,0x48,0x36,0xc1
+
+# ATT:   vcvthf82ps %xmm1, %ymm0
+# INTEL: vcvthf82ps ymm0, xmm1
+0x62,0xf5,0x7c,0x28,0x36,0xc1
+
+# ATT:   vcvthf82ps %xmm1, %xmm0
+# INTEL: vcvthf82ps xmm0, xmm1
+0x62,0xf5,0x7c,0x08,0x36,0xc1
+
+# ATT:   vcvthf82ps %xmm1, %zmm0 {%k1}
+# INTEL: vcvthf82ps zmm0 {k1}, xmm1
+0x62,0xf5,0x7c,0x49,0x36,0xc1
+
+# ATT:   vcvthf82ps %xmm1, %zmm0 {%k1} {z}
+# INTEL: vcvthf82ps zmm0 {k1} {z}, xmm1
+0x62,0xf5,0x7c,0xc9,0x36,0xc1
+
+#
+# Group D: BF8/HF8->BF4S truncations
+#
+
+# vcvtbf82bf4s
+
+# ATT:   vcvtbf82bf4s %zmm1, %ymm0
+# INTEL: vcvtbf82bf4s ymm0, zmm1
+0x62,0xf5,0xfe,0x48,0x3d,0xc8
+
+# ATT:   vcvtbf82bf4s %ymm1, %xmm0
+# INTEL: vcvtbf82bf4s xmm0, ymm1
+0x62,0xf5,0xfe,0x28,0x3d,0xc8
+
+# ATT:   vcvtbf82bf4s %xmm1, %xmm0
+# INTEL: vcvtbf82bf4s xmm0, xmm1
+0x62,0xf5,0xfe,0x08,0x3d,0xc8
+
+# vcvthf82bf4s
+
+# ATT:   vcvthf82bf4s %zmm1, %ymm0
+# INTEL: vcvthf82bf4s ymm0, zmm1
+0x62,0xf5,0x7e,0x48,0x3d,0xc8
+
+# ATT:   vcvthf82bf4s %ymm1, %xmm0
+# INTEL: vcvthf82bf4s xmm0, ymm1
+0x62,0xf5,0x7e,0x28,0x3d,0xc8
+
+# ATT:   vcvthf82bf4s %xmm1, %xmm0
+# INTEL: vcvthf82bf4s xmm0, xmm1
+0x62,0xf5,0x7e,0x08,0x3d,0xc8
+
+#
+# Group E: Same-size reg-only conversions (no masking)
+#
+
+# vcvtbf82bf6s
+
+# ATT:   vcvtbf82bf6s %zmm1, %zmm0
+# INTEL: vcvtbf82bf6s zmm0, zmm1
+0x62,0xf5,0xfe,0x48,0x3e,0xc1
+
+# ATT:   vcvtbf82bf6s %ymm1, %ymm0
+# INTEL: vcvtbf82bf6s ymm0, ymm1
+0x62,0xf5,0xfe,0x28,0x3e,0xc1
+
+# ATT:   vcvtbf82bf6s %xmm1, %xmm0
+# INTEL: vcvtbf82bf6s xmm0, xmm1
+0x62,0xf5,0xfe,0x08,0x3e,0xc1
+
+# vcvthf82hf6s
+
+# ATT:   vcvthf82hf6s %zmm1, %zmm0
+# INTEL: vcvthf82hf6s zmm0, zmm1
+0x62,0xf5,0x7e,0x48,0x3c,0xc1
+
+# ATT:   vcvthf82hf6s %ymm1, %ymm0
+# INTEL: vcvthf82hf6s ymm0, ymm1
+0x62,0xf5,0x7e,0x28,0x3c,0xc1
+
+# ATT:   vcvthf82hf6s %xmm1, %xmm0
+# INTEL: vcvthf82hf6s xmm0, xmm1
+0x62,0xf5,0x7e,0x08,0x3c,0xc1
+
+#
+# Group F: Expanding/same-size conversions with masking
+#
+
+# vcvtbf42hf8
+
+# ATT:   vcvtbf42hf8 %ymm1, %zmm0
+# INTEL: vcvtbf42hf8 zmm0, ymm1
+0x62,0xf5,0x7c,0x48,0x37,0xc1
+
+# ATT:   vcvtbf42hf8 %xmm1, %ymm0
+# INTEL: vcvtbf42hf8 ymm0, xmm1
+0x62,0xf5,0x7c,0x28,0x37,0xc1
+
+# ATT:   vcvtbf42hf8 %xmm1, %xmm0
+# INTEL: vcvtbf42hf8 xmm0, xmm1
+0x62,0xf5,0x7c,0x08,0x37,0xc1
+
+# ATT:   vcvtbf42hf8 %ymm1, %zmm0 {%k1}
+# INTEL: vcvtbf42hf8 zmm0 {k1}, ymm1
+0x62,0xf5,0x7c,0x49,0x37,0xc1
+
+# ATT:   vcvtbf42hf8 %ymm1, %zmm0 {%k1} {z}
+# INTEL: vcvtbf42hf8 zmm0 {k1} {z}, ymm1
+0x62,0xf5,0x7c,0xc9,0x37,0xc1
+
+# vcvtbf62hf8
+
+# ATT:   vcvtbf62hf8 %zmm1, %zmm0
+# INTEL: vcvtbf62hf8 zmm0, zmm1
+0x62,0xf5,0xfd,0x48,0x37,0xc1
+
+# ATT:   vcvtbf62hf8 %ymm1, %ymm0
+# INTEL: vcvtbf62hf8 ymm0, ymm1
+0x62,0xf5,0xfd,0x28,0x37,0xc1
+
+# ATT:   vcvtbf62hf8 %xmm1, %xmm0
+# INTEL: vcvtbf62hf8 xmm0, xmm1
+0x62,0xf5,0xfd,0x08,0x37,0xc1
+
+# ATT:   vcvtbf62hf8 %zmm1, %zmm0 {%k1}
+# INTEL: vcvtbf62hf8 zmm0 {k1}, zmm1
+0x62,0xf5,0xfd,0x49,0x37,0xc1
+
+# ATT:   vcvtbf62hf8 %zmm1, %zmm0 {%k1} {z}
+# INTEL: vcvtbf62hf8 zmm0 {k1} {z}, zmm1
+0x62,0xf5,0xfd,0xc9,0x37,0xc1
+
+# vcvthf62hf8
+
+# ATT:   vcvthf62hf8 %zmm1, %zmm0
+# INTEL: vcvthf62hf8 zmm0, zmm1
+0x62,0xf5,0x7d,0x48,0x37,0xc1
+
+# ATT:   vcvthf62hf8 %ymm1, %ymm0
+# INTEL: vcvthf62hf8 ymm0, ymm1
+0x62,0xf5,0x7d,0x28,0x37,0xc1
+
+# ATT:   vcvthf62hf8 %xmm1, %xmm0
+# INTEL: vcvthf62hf8 xmm0, xmm1
+0x62,0xf5,0x7d,0x08,0x37,0xc1
+
+# ATT:   vcvthf62hf8 %zmm1, %zmm0 {%k1}
+# INTEL: vcvthf62hf8 zmm0 {k1}, zmm1
+0x62,0xf5,0x7d,0x49,0x37,0xc1
+
+# ATT:   vcvthf62hf8 %zmm1, %zmm0 {%k1} {z}
+# INTEL: vcvthf62hf8 zmm0 {k1} {z}, zmm1
+0x62,0xf5,0x7d,0xc9,0x37,0xc1
+
+#
+# Group G: VPMOVSSDB - Integer DWord->Byte signed saturation
+#
+
+# ATT:   vpmovssdb %zmm1, %xmm0
+# INTEL: vpmovssdb xmm0, zmm1
+0x62,0xf2,0x7e,0x48,0x41,0xc8
+
+# ATT:   vpmovssdb %ymm1, %xmm0
+# INTEL: vpmovssdb xmm0, ymm1
+0x62,0xf2,0x7e,0x28,0x41,0xc8
+
+# ATT:   vpmovssdb %xmm1, %xmm0
+# INTEL: vpmovssdb xmm0, xmm1
+0x62,0xf2,0x7e,0x08,0x41,0xc8
+
+# ATT:   vpmovssdb %zmm1, %xmm0 {%k1}
+# INTEL: vpmovssdb xmm0 {k1}, zmm1
+0x62,0xf2,0x7e,0x49,0x41,0xc8
+
+# ATT:   vpmovssdb %zmm1, %xmm0 {%k1} {z}
+# INTEL: vpmovssdb xmm0 {k1} {z}, zmm1
+0x62,0xf2,0x7e,0xc9,0x41,0xc8
+
+#
+# Group H: VUNPACKB - Byte unpack with immediate
+#
+
+# ATT:   vunpackb $1, %zmm1, %zmm0
+# INTEL: vunpackb zmm0, zmm1, 1
+0x62,0xf3,0x7c,0x48,0x3d,0xc1,0x01
+
+# ATT:   vunpackb $1, %ymm1, %ymm0
+# INTEL: vunpackb ymm0, ymm1, 1
+0x62,0xf3,0x7c,0x28,0x3d,0xc1,0x01
+
+# ATT:   vunpackb $1, %xmm1, %xmm0
+# INTEL: vunpackb xmm0, xmm1, 1
+0x62,0xf3,0x7c,0x08,0x3d,0xc1,0x01
+
+# ATT:   vunpackb $1, %zmm1, %zmm0 {%k1}
+# INTEL: vunpackb zmm0 {k1}, zmm1, 1
+0x62,0xf3,0x7c,0x49,0x3d,0xc1,0x01
+
+# ATT:   vunpackb $1, %zmm1, %zmm0 {%k1} {z}
+# INTEL: vunpackb zmm0 {k1} {z}, zmm1, 1
+0x62,0xf3,0x7c,0xc9,0x3d,0xc1,0x01
diff --git a/llvm/test/MC/Disassembler/X86/avx10_v2_aux-64.txt b/llvm/test/MC/Disassembler/X86/avx10_v2_aux-64.txt
new file mode 100644
index 00000000000000..40a7a366e34f9a
--- /dev/null
+++ b/llvm/test/MC/Disassembler/X86/avx10_v2_aux-64.txt
@@ -0,0 +1,460 @@
+# RUN: llvm-mc --disassemble %s -triple=x86_64 | FileCheck %s --check-prefixes=ATT
+# RUN: llvm-mc --disassemble %s -triple=x86_64 --output-asm-variant=1 | FileCheck %s --check-prefixes=INTEL
+
+#
+# Group A: PS->8bit truncating conversions
+#
+
+# vcvtps2bf8
+
+# ATT:   vcvtps2bf8 %zmm1, %xmm0
+# INTEL: vcvtps2bf8 xmm0, zmm1
+0x62,0xf5,0x7e,0x48,0x39,0xc1
+
+# ATT:   vcvtps2bf8 %ymm1, %xmm0
+# INTEL: vcvtps2bf8 xmm0, ymm1
+0x62,0xf5,0x7e,0x28,0x39,0xc1
+
+# ATT:   vcvtps2bf8 %xmm1, %xmm0
+# INTEL: vcvtps2bf8 xmm0, xmm1
+0x62,0xf5,0x7e,0x08,0x39,0xc1
+
+# ATT:   vcvtps2bf8 %zmm1, %xmm0 {%k1}
+# INTEL: vcvtps2bf8 xmm0 {k1}, zmm1
+0x62,0xf5,0x7e,0x49,0x39,0xc1
+
+# ATT:   vcvtps2bf8 %zmm1, %xmm0 {%k1} {z}
+# INTEL: vcvtps2bf8 xmm0 {k1} {z}, zmm1
+0x62,0xf5,0x7e,0xc9,0x39,0xc1
+
+# vcvtps2bf8s
+
+# ATT:   vcvtps2bf8s %zmm1, %xmm0
+# INTEL: vcvtps2bf8s xmm0, zmm1
+0x62,0xf5,0x7e,0x48,0x3b,0xc1
+
+# ATT:   vcvtps2bf8s %ymm1, %xmm0
+# INTEL: vcvtps2bf8s xmm0, ymm1
+0x62,0xf5,0x7e,0x28,0x3b,0xc1
+
+# ATT:   vcvtps2bf8s %xmm1, %xmm0
+# INTEL: vcvtps2bf8s xmm0, xmm1
+0x62,0xf5,0x7e,0x08,0x3b,0xc1
+
+# ATT:   vcvtps2bf8s %zmm1, %xmm0 {%k1}
+# INTEL: vcvtps2bf8s xmm0 {k1}, zmm1
+0x62,0xf5,0x7e,0x49,0x3b,0xc1
+
+# ATT:   vcvtps2bf8s %zmm1, %xmm0 {%k1} {z}
+# INTEL: vcvtps2bf8s xmm0 {k1} {z}, zmm1
+0x62,0xf5,0x7e,0xc9,0x3b,0xc1
+
+# vcvtps2hf8
+
+# ATT:   vcvtps2hf8 %zmm1, %xmm0
+# INTEL: vcvtps2hf8 xmm0, zmm1
+0x62,0xf5,0x7e,0x48,0x38,0xc1
+
+# ATT:   vcvtps2hf8 %ymm1, %xmm0
+# INTEL: vcvtps2hf8 xmm0, ymm1
+0x62,0xf5,0x7e,0x28,0x38,0xc1
+
+# ATT:   vcvtps2hf8 %xmm1, %xmm0
+# INTEL: vcvtps2hf8 xmm0, xmm1
+0x62,0xf5,0x7e,0x08,0x38,0xc1
+
+# ATT:   vcvtps2hf8 %zmm1, %xmm0 {%k1}
+# INTEL: vcvtps2hf8 xmm0 {k1}, zmm1
+0x62,0xf5,0x7e,0x49,0x38,0xc1
+
+# ATT:   vcvtps2hf8 %zmm1, %xmm0 {%k1} {z}
+# INTEL: vcvtps2hf8 xmm0 {k1} {z}, zmm1
+0x62,0xf5,0x7e,0xc9,0x38,0xc1
+
+# vcvtps2hf8s
+
+# ATT:   vcvtps2hf8s %zmm1, %xmm0
+# INTEL: vcvtps2hf8s xmm0, zmm1
+0x62,0xf5,0x7e,0x48,0x3a,0xc1
+
+# ATT:   vcvtps2hf8s %ymm1, %xmm0
+# INTEL: vcvtps2hf8s xmm0, ymm1
+0x62,0xf5,0x7e,0x28,0x3a,0xc1
+
+# ATT:   vcvtps2hf8s %xmm1, %xmm0
+# INTEL: vcvtps2hf8s xmm0, xmm1
+0x62,0xf5,0x7e,0x08,0x3a,0xc1
+
+# ATT:   vcvtps2hf8s %zmm1, %xmm0 {%k1}
+# INTEL: vcvtps2hf8s xmm0 {k1}, zmm1
+0x62,0xf5,0x7e,0x49,0x3a,0xc1
+
+# ATT:   vcvtps2hf8s %zmm1, %xmm0 {%k1} {z}
+# INTEL: vcvtps2hf8s xmm0 {k1} {z}, zmm1
+0x62,0xf5,0x7e,0xc9,0x3a,0xc1
+
+# vcvtrops2hf8
+
+# ATT:   vcvtrops2hf8 %zmm1, %xmm0
+# INTEL: vcvtrops2hf8 xmm0, zmm1
+0x62,0xf5,0x7d,0x48,0x38,0xc1
+
+# ATT:   vcvtrops2hf8 %ymm1, %xmm0
+# INTEL: vcvtrops2hf8 xmm0, ymm1
+0x62,0xf5,0x7d,0x28,0x38,0xc1
+
+# ATT:   vcvtrops2hf8 %xmm1, %xmm0
+# INTEL: vcvtrops2hf8 xmm0, xmm1
+0x62,0xf5,0x7d,0x08,0x38,0xc1
+
+# ATT:   vcvtrops2hf8 %zmm1, %xmm0 {%k1}
+# INTEL: vcvtrops2hf8 xmm0 {k1}, zmm1
+0x62,0xf5,0x7d,0x49,0x38,0xc1
+
+# ATT:   vcvtrops2hf8 %zmm1, %xmm0 {%k1} {z}
+# INTEL: vcvtrops2hf8 xmm0 {k1} {z}, zmm1
+0x62,0xf5,0x7d,0xc9,0x38,0xc1
+
+# vcvtrops2hf8s
+
+# ATT:   vcvtrops2hf8s %zmm1, %xmm0
+# INTEL: vcvtrops2hf8s xmm0, zmm1
+0x62,0xf5,0x7d,0x48,0x3a,0xc1
+
+# ATT:   vcvtrops2hf8s %ymm1, %xmm0
+# INTEL: vcvtrops2hf8s xmm0, ymm1
+0x62,0xf5,0x7d,0x28,0x3a,0xc1
+
+# ATT:   vcvtrops2hf8s %xmm1, %xmm0
+# INTEL: vcvtrops2hf8s xmm0, xmm1
+0x62,0xf5,0x7d,0x08,0x3a,0xc1
+
+# ATT:   vcvtrops2hf8s %zmm1, %xmm0 {%k1}
+# INTEL: vcvtrops2hf8s xmm0 {k1}, zmm1
+0x62,0xf5,0x7d,0x49,0x3a,0xc1
+
+# ATT:   vcvtrops2hf8s %zmm1, %xmm0 {%k1} {z}
+# INTEL: vcvtrops2hf8s xmm0 {k1} {z}, zmm1
+0x62,0xf5,0x7d,0xc9,0x3a,0xc1
+
+#
+# Group B: Bias PS->8bit conversions (3-operand)
+#
+
+# vcvtbiasps2bf8
+
+# ATT:   vcvtbiasps2bf8 %zmm2, %zmm1, %xmm0
+# INTEL: vcvtbiasps2bf8 xmm0, zmm1, zmm2
+0x62,0xf5,0x74,0x48,0x39,0xc2
+
+# ATT:   vcvtbiasps2bf8 %ymm2, %ymm1, %xmm0
+# INTEL: vcvtbiasps2bf8 xmm0, ymm1, ymm2
+0x62,0xf5,0x74,0x28,0x39,0xc2
+
+# ATT:   vcvtbiasps2bf8 %xmm2, %xmm1, %xmm0
+# INTEL: vcvtbiasps2bf8 xmm0, xmm1, xmm2
+0x62,0xf5,0x74,0x08,0x39,0xc2
+
+# ATT:   vcvtbiasps2bf8 %zmm2, %zmm1, %xmm0 {%k1}
+# INTEL: vcvtbiasps2bf8 xmm0 {k1}, zmm1, zmm2
+0x62,0xf5,0x74,0x49,0x39,0xc2
+
+# ATT:   vcvtbiasps2bf8 %zmm2, %zmm1, %xmm0 {%k1} {z}
+# INTEL: vcvtbiasps2bf8 xmm0 {k1} {z}, zmm1, zmm2
+0x62,0xf5,0x74,0xc9,0x39,0xc2
+
+# vcvtbiasps2bf8s
+
+# ATT:   vcvtbiasps2bf8s %zmm2, %zmm1, %xmm0
+# INTEL: vcvtbiasps2bf8s xmm0, zmm1, zmm2
+0x62,0xf5,0x74,0x48,0x3b,0xc2
+
+# ATT:   vcvtbiasps2bf8s %ymm2, %ymm1, %xmm0
+# INTEL: vcvtbiasps2bf8s xmm0, ymm1, ymm2
+0x62,0xf5,0x74,0x28,0x3b,0xc2
+
+# ATT:   vcvtbiasps2bf8s %xmm2, %xmm1, %xmm0
+# INTEL: vcvtbiasps2bf8s xmm0, xmm1, xmm2
+0x62,0xf5,0x74,0x08,0x3b,0xc2
+
+# ATT:   vcvtbiasps2bf8s %zmm2, %zmm1, %xmm0 {%k1}
+# INTEL: vcvtbiasps2bf8s xmm0 {k1}, zmm1, zmm2
+0x62,0xf5,0x74,0x49,0x3b,0xc2
+
+# ATT:   vcvtbiasps2bf8s %zmm2, %zmm1, %xmm0 {%k1} {z}
+# INTEL: vcvtbiasps2bf8s xmm0 {k1} {z}, zmm1, zmm2
+0x62,0xf5,0x74,0xc9,0x3b,0xc2
+
+# vcvtbiasps2hf8
+
+# ATT:   vcvtbiasps2hf8 %zmm2, %zmm1, %xmm0
+# INTEL: vcvtbiasps2hf8 xmm0, zmm1, zmm2
+0x62,0xf5,0x74,0x48,0x38,0xc2
+
+# ATT:   vcvtbiasps2hf8 %ymm2, %ymm1, %xmm0
+# INTEL: vcvtbiasps2hf8 xmm0, ymm1, ymm2
+0x62,0xf5,0x74,0x28,0x38,0xc2
+
+# ATT:   vcvtbiasps2hf8 %xmm2, %xmm1, %xmm0
+# INTEL: vcvtbiasps2hf8 xmm0, xmm1, xmm2
+0x62,0xf5,0x74,0x08,0x38,0xc2
+
+# ATT:   vcvtbiasps2hf8 %zmm2, %zmm1, %xmm0 {%k1}
+# INTEL: vcvtbiasps2hf8 xmm0 {k1}, zmm1, zmm2
+0x62,0xf5,0x74,0x49,0x38,0xc2
+
+# ATT:   vcvtbiasps2hf8 %zmm2, %zmm1, %xmm0 {%k1} {z}
+# INTEL: vcvtbiasps2hf8 xmm0 {k1} {z}, zmm1, zmm2
+0x62,0xf5,0x74,0xc9,0x38,0xc2
+
+# vcvtbiasps2hf8s
+
+# ATT:   vcvtbiasps2hf8s %zmm2, %zmm1, %xmm0
+# INTEL: vcvtbiasps2hf8s xmm0, zmm1, zmm2
+0x62,0xf5,0x74,0x48,0x3a,0xc2
+
+# ATT:   vcvtbiasps2hf8s %ymm2, %ymm1, %xmm0
+# INTEL: vcvtbiasps2hf8s xmm0, ymm1, ymm2
+0x62,0xf5,0x74,0x28,0x3a,0xc2
+
+# ATT:   vcvtbiasps2hf8s %xmm2, %xmm1, %xmm0
+# INTEL: vcvtbiasps2hf8s xmm0, xmm1, xmm2
+0x62,0xf5,0x74,0x08,0x3a,0xc2
+
+# ATT:   vcvtbiasps2hf8s %zmm2, %zmm1, %xmm0 {%k1}
+# INTEL: vcvtbiasps2hf8s xmm0 {k1}, zmm1, zmm2
+0x62,0xf5,0x74,0x49,0x3a,0xc2
+
+# ATT:   vcvtbiasps2hf8s %zmm2, %zmm1, %xmm0 {%k1} {z}
+# INTEL: vcvtbiasps2hf8s xmm0 {k1} {z}, zmm1, zmm2
+0x62,0xf5,0x74,0xc9,0x3a,0xc2
+
+#
+# Group C: 8bit->PS expanding conversions
+#
+
+# vcvtbf82ps
+
+# ATT:   vcvtbf82ps %xmm1, %zmm0
+# INTEL: vcvtbf82ps zmm0, xmm1
+0x62,0xf5,0xfc,0x48,0x36,0xc1
+
+# ATT:   vcvtbf82ps %xmm1, %ymm0
+# INTEL: vcvtbf82ps ymm0, xmm1
+0x62,0xf5,0xfc,0x28,0x36,0xc1
+
+# ATT:   vcvtbf82ps %xmm1, %xmm0
+# INTEL: vcvtbf82ps xmm0, xmm1
+0x62,0xf5,0xfc,0x08,0x36,0xc1
+
+# ATT:   vcvtbf82ps %xmm1, %zmm0 {%k1}
+# INTEL: vcvtbf82ps zmm0 {k1}, xmm1
+0x62,0xf5,0xfc,0x49,0x36,0xc1
+
+# ATT:   vcvtbf82ps %xmm1, %zmm0 {%k1} {z}
+# INTEL: vcvtbf82ps zmm0 {k1} {z}, xmm1
+0x62,0xf5,0xfc,0xc9,0x36,0xc1
+
+# vcvthf82ps
+
+# ATT:   vcvthf82ps %xmm1, %zmm0
+# INTEL: vcvthf82ps zmm0, xmm1
+0x62,0xf5,0x7c,0x48,0x36,0xc1
+
+# ATT:   vcvthf82ps %xmm1, %ymm0
+# INTEL: vcvthf82ps ymm0, xmm1
+0x62,0xf5,0x7c,0x28,0x36,0xc1
+
+# ATT:   vcvthf82ps %xmm1, %xmm0
+# INTEL: vcvthf82ps xmm0, xmm1
+0x62,0xf5,0x7c,0x08,0x36,0xc1
+
+# ATT:   vcvthf82ps %xmm1, %zmm0 {%k1}
+# INTEL: vcvthf82ps zmm0 {k1}, xmm1
+0x62,0xf5,0x7c,0x49,0x36,0xc1
+
+# ATT:   vcvthf82ps %xmm1, %zmm0 {%k1} {z}
+# INTEL: vcvthf82ps zmm0 {k1} {z}, xmm1
+0x62,0xf5,0x7c,0xc9,0x36,0xc1
+
+#
+# Group D: BF8/HF8->BF4S truncations
+#
+
+# vcvtbf82bf4s
+
+# ATT:   vcvtbf82bf4s %zmm1, %ymm0
+# INTEL: vcvtbf82bf4s ymm0, zmm1
+0x62,0xf5,0xfe,0x48,0x3d,0xc8
+
+# ATT:   vcvtbf82bf4s %ymm1, %xmm0
+# INTEL: vcvtbf82bf4s xmm0, ymm1
+0x62,0xf5,0xfe,0x28,0x3d,0xc8
+
+# ATT:   vcvtbf82bf4s %xmm1, %xmm0
+# INTEL: vcvtbf82bf4s xmm0, xmm1
+0x62,0xf5,0xfe,0x08,0x3d,0xc8
+
+# vcvthf82bf4s
+
+# ATT:   vcvthf82bf4s %zmm1, %ymm0
+# INTEL: vcvthf82bf4s ymm0, zmm1
+0x62,0xf5,0x7e,0x48,0x3d,0xc8
+
+# ATT:   vcvthf82bf4s %ymm1, %xmm0
+# INTEL: vcvthf82bf4s xmm0, ymm1
+0x62,0xf5,0x7e,0x28,0x3d,0xc8
+
+# ATT:   vcvthf82bf4s %xmm1, %xmm0
+# INTEL: vcvthf82bf4s xmm0, xmm1
+0x62,0xf5,0x7e,0x08,0x3d,0xc8
+
+#
+# Group E: Same-size reg-only conversions (no masking)
+#
+
+# vcvtbf82bf6s
+
+# ATT:   vcvtbf82bf6s %zmm1, %zmm0
+# INTEL: vcvtbf82bf6s zmm0, zmm1
+0x62,0xf5,0xfe,0x48,0x3e,0xc1
+
+# ATT:   vcvtbf82bf6s %ymm1, %ymm0
+# INTEL: vcvtbf82bf6s ymm0, ymm1
+0x62,0xf5,0xfe,0x28,0x3e,0xc1
+
+# ATT:   vcvtbf82bf6s %xmm1, %xmm0
+# INTEL: vcvtbf82bf6s xmm0, xmm1
+0x62,0xf5,0xfe,0x08,0x3e,0xc1
+
+# vcvthf82hf6s
+
+# ATT:   vcvthf82hf6s %zmm1, %zmm0
+# INTEL: vcvthf82hf6s zmm0, zmm1
+0x62,0xf5,0x7e,0x48,0x3c,0xc1
+
+# ATT:   vcvthf82hf6s %ymm1, %ymm0
+# INTEL: vcvthf82hf6s ymm0, ymm1
+0x62,0xf5,0x7e,0x28,0x3c,0xc1
+
+# ATT:   vcvthf82hf6s %xmm1, %xmm0
+# INTEL: vcvthf82hf6s xmm0, xmm1
+0x62,0xf5,0x7e,0x08,0x3c,0xc1
+
+#
+# Group F: Expanding/same-size conversions with masking
+#
+
+# vcvtbf42hf8
+
+# ATT:   vcvtbf42hf8 %ymm1, %zmm0
+# INTEL: vcvtbf42hf8 zmm0, ymm1
+0x62,0xf5,0x7c,0x48,0x37,0xc1
+
+# ATT:   vcvtbf42hf8 %xmm1, %ymm0
+# INTEL: vcvtbf42hf8 ymm0, xmm1
+0x62,0xf5,0x7c,0x28,0x37,0xc1
+
+# ATT:   vcvtbf42hf8 %xmm1, %xmm0
+# INTEL: vcvtbf42hf8 xmm0, xmm1
+0x62,0xf5,0x7c,0x08,0x37,0xc1
+
+# ATT:   vcvtbf42hf8 %ymm1, %zmm0 {%k1}
+# INTEL: vcvtbf42hf8 zmm0 {k1}, ymm1
+0x62,0xf5,0x7c,0x49,0x37,0xc1
+
+# ATT:   vcvtbf42hf8 %ymm1, %zmm0 {%k1} {z}
+# INTEL: vcvtbf42hf8 zmm0 {k1} {z}, ymm1
+0x62,0xf5,0x7c,0xc9,0x37,0xc1
+
+# vcvtbf62hf8
+
+# ATT:   vcvtbf62hf8 %zmm1, %zmm0
+# INTEL: vcvtbf62hf8 zmm0, zmm1
+0x62,0xf5,0xfd,0x48,0x37,0xc1
+
+# ATT:   vcvtbf62hf8 %ymm1, %ymm0
+# INTEL: vcvtbf62hf8 ymm0, ymm1
+0x62,0xf5,0xfd,0x28,0x37,0xc1
+
+# ATT:   vcvtbf62hf8 %xmm1, %xmm0
+# INTEL: vcvtbf62hf8 xmm0, xmm1
+0x62,0xf5,0xfd,0x08,0x37,0xc1
+
+# ATT:   vcvtbf62hf8 %zmm1, %zmm0 {%k1}
+# INTEL: vcvtbf62hf8 zmm0 {k1}, zmm1
+0x62,0xf5,0xfd,0x49,0x37,0xc1
+
+# ATT:   vcvtbf62hf8 %zmm1, %zmm0 {%k1} {z}
+# INTEL: vcvtbf62hf8 zmm0 {k1} {z}, zmm1
+0x62,0xf5,0xfd,0xc9,0x37,0xc1
+
+# vcvthf62hf8
+
+# ATT:   vcvthf62hf8 %zmm1, %zmm0
+# INTEL: vcvthf62hf8 zmm0, zmm1
+0x62,0xf5,0x7d,0x48,0x37,0xc1
+
+# ATT:   vcvthf62hf8 %ymm1, %ymm0
+# INTEL: vcvthf62hf8 ymm0, ymm1
+0x62,0xf5,0x7d,0x28,0x37,0xc1
+
+# ATT:   vcvthf62hf8 %xmm1, %xmm0
+# INTEL: vcvthf62hf8 xmm0, xmm1
+0x62,0xf5,0x7d,0x08,0x37,0xc1
+
+# ATT:   vcvthf62hf8 %zmm1, %zmm0 {%k1}
+# INTEL: vcvthf62hf8 zmm0 {k1}, zmm1
+0x62,0xf5,0x7d,0x49,0x37,0xc1
+
+# ATT:   vcvthf62hf8 %zmm1, %zmm0 {%k1} {z}
+# INTEL: vcvthf62hf8 zmm0 {k1} {z}, zmm1
+0x62,0xf5,0x7d,0xc9,0x37,0xc1
+
+#
+# Group G: VPMOVSSDB - Integer DWord->Byte signed saturation
+#
+
+# ATT:   vpmovssdb %zmm1, %xmm0
+# INTEL: vpmovssdb xmm0, zmm1
+0x62,0xf2,0x7e,0x48,0x41,0xc8
+
+# ATT:   vpmovssdb %ymm1, %xmm0
+# INTEL: vpmovssdb xmm0, ymm1
+0x62,0xf2,0x7e,0x28,0x41,0xc8
+
+# ATT:   vpmovssdb %xmm1, %xmm0
+# INTEL: vpmovssdb xmm0, xmm1
+0x62,0xf2,0x7e,0x08,0x41,0xc8
+
+# ATT:   vpmovssdb %zmm1, %xmm0 {%k1}
+# INTEL: vpmovssdb xmm0 {k1}, zmm1
+0x62,0xf2,0x7e,0x49,0x41,0xc8
+
+# ATT:   vpmovssdb %zmm1, %xmm0 {%k1} {z}
+# INTEL: vpmovssdb xmm0 {k1} {z}, zmm1
+0x62,0xf2,0x7e,0xc9,0x41,0xc8
+
+#
+# Group H: VUNPACKB - Byte unpack with immediate
+#
+
+# ATT:   vunpackb $1, %zmm1, %zmm0
+# INTEL: vunpackb zmm0, zmm1, 1
+0x62,0xf3,0x7c,0x48,0x3d,0xc1,0x01
+
+# ATT:   vunpackb $1, %ymm1, %ymm0
+# INTEL: vunpackb ymm0, ymm1, 1
+0x62,0xf3,0x7c,0x28,0x3d,0xc1,0x01
+
+# ATT:   vunpackb $1, %xmm1, %xmm0
+# INTEL: vunpackb xmm0, xmm1, 1
+0x62,0xf3,0x7c,0x08,0x3d,0xc1,0x01
+
+# ATT:   vunpackb $1, %zmm1, %zmm0 {%k1}
+# INTEL: vunpackb zmm0 {k1}, zmm1, 1
+0x62,0xf3,0x7c,0x49,0x3d,0xc1,0x01
+
+# ATT:   vunpackb $1, %zmm1, %zmm0 {%k1} {z}
+# INTEL: vunpackb zmm0 {k1} {z}, zmm1, 1
+0x62,0xf3,0x7c,0xc9,0x3d,0xc1,0x01
diff --git a/llvm/test/MC/X86/avx10_v2_aux-att-32.s b/llvm/test/MC/X86/avx10_v2_aux-att-32.s
new file mode 100644
index 00000000000000..10c4ed5bd32592
--- /dev/null
+++ b/llvm/test/MC/X86/avx10_v2_aux-att-32.s
@@ -0,0 +1,463 @@
+// RUN: llvm-mc -triple i386 --show-encoding -mattr=+avx10-v2-aux,+avx512vl %s | FileCheck %s
+
+//
+// Group A: PS->8bit truncating conversions
+//
+
+// vcvtps2bf8
+
+// CHECK: vcvtps2bf8 %zmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x39,0xc1]
+          vcvtps2bf8 %zmm1, %xmm0
+
+// CHECK: vcvtps2bf8 %ymm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x28,0x39,0xc1]
+          vcvtps2bf8 %ymm1, %xmm0
+
+// CHECK: vcvtps2bf8 %xmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x39,0xc1]
+          vcvtps2bf8 %xmm1, %xmm0
+
+// CHECK: vcvtps2bf8 (%edi), %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x39,0x07]
+          vcvtps2bf8 (%edi), %xmm0
+
+// CHECK: vcvtps2bf8 %zmm1, %xmm0 {%k1}
+// CHECK: encoding: [0x62,0xf5,0x7e,0x49,0x39,0xc1]
+          vcvtps2bf8 %zmm1, %xmm0 {%k1}
+
+// CHECK: vcvtps2bf8 %zmm1, %xmm0 {%k1} {z}
+// CHECK: encoding: [0x62,0xf5,0x7e,0xc9,0x39,0xc1]
+          vcvtps2bf8 %zmm1, %xmm0 {%k1} {z}
+
+// vcvtps2bf8s
+
+// CHECK: vcvtps2bf8s %zmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x3b,0xc1]
+          vcvtps2bf8s %zmm1, %xmm0
+
+// CHECK: vcvtps2bf8s %ymm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x28,0x3b,0xc1]
+          vcvtps2bf8s %ymm1, %xmm0
+
+// CHECK: vcvtps2bf8s %xmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x3b,0xc1]
+          vcvtps2bf8s %xmm1, %xmm0
+
+// CHECK: vcvtps2bf8s %zmm1, %xmm0 {%k1}
+// CHECK: encoding: [0x62,0xf5,0x7e,0x49,0x3b,0xc1]
+          vcvtps2bf8s %zmm1, %xmm0 {%k1}
+
+// CHECK: vcvtps2bf8s %zmm1, %xmm0 {%k1} {z}
+// CHECK: encoding: [0x62,0xf5,0x7e,0xc9,0x3b,0xc1]
+          vcvtps2bf8s %zmm1, %xmm0 {%k1} {z}
+
+// vcvtps2hf8
+
+// CHECK: vcvtps2hf8 %zmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x38,0xc1]
+          vcvtps2hf8 %zmm1, %xmm0
+
+// CHECK: vcvtps2hf8 %ymm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x28,0x38,0xc1]
+          vcvtps2hf8 %ymm1, %xmm0
+
+// CHECK: vcvtps2hf8 %xmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x38,0xc1]
+          vcvtps2hf8 %xmm1, %xmm0
+
+// CHECK: vcvtps2hf8 %zmm1, %xmm0 {%k1}
+// CHECK: encoding: [0x62,0xf5,0x7e,0x49,0x38,0xc1]
+          vcvtps2hf8 %zmm1, %xmm0 {%k1}
+
+// CHECK: vcvtps2hf8 %zmm1, %xmm0 {%k1} {z}
+// CHECK: encoding: [0x62,0xf5,0x7e,0xc9,0x38,0xc1]
+          vcvtps2hf8 %zmm1, %xmm0 {%k1} {z}
+
+// vcvtps2hf8s
+
+// CHECK: vcvtps2hf8s %zmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x3a,0xc1]
+          vcvtps2hf8s %zmm1, %xmm0
+
+// CHECK: vcvtps2hf8s %ymm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x28,0x3a,0xc1]
+          vcvtps2hf8s %ymm1, %xmm0
+
+// CHECK: vcvtps2hf8s %xmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x3a,0xc1]
+          vcvtps2hf8s %xmm1, %xmm0
+
+// CHECK: vcvtps2hf8s %zmm1, %xmm0 {%k1}
+// CHECK: encoding: [0x62,0xf5,0x7e,0x49,0x3a,0xc1]
+          vcvtps2hf8s %zmm1, %xmm0 {%k1}
+
+// CHECK: vcvtps2hf8s %zmm1, %xmm0 {%k1} {z}
+// CHECK: encoding: [0x62,0xf5,0x7e,0xc9,0x3a,0xc1]
+          vcvtps2hf8s %zmm1, %xmm0 {%k1} {z}
+
+// vcvtrops2hf8
+
+// CHECK: vcvtrops2hf8 %zmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7d,0x48,0x38,0xc1]
+          vcvtrops2hf8 %zmm1, %xmm0
+
+// CHECK: vcvtrops2hf8 %ymm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7d,0x28,0x38,0xc1]
+          vcvtrops2hf8 %ymm1, %xmm0
+
+// CHECK: vcvtrops2hf8 %xmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7d,0x08,0x38,0xc1]
+          vcvtrops2hf8 %xmm1, %xmm0
+
+// CHECK: vcvtrops2hf8 %zmm1, %xmm0 {%k1}
+// CHECK: encoding: [0x62,0xf5,0x7d,0x49,0x38,0xc1]
+          vcvtrops2hf8 %zmm1, %xmm0 {%k1}
+
+// CHECK: vcvtrops2hf8 %zmm1, %xmm0 {%k1} {z}
+// CHECK: encoding: [0x62,0xf5,0x7d,0xc9,0x38,0xc1]
+          vcvtrops2hf8 %zmm1, %xmm0 {%k1} {z}
+
+// vcvtrops2hf8s
+
+// CHECK: vcvtrops2hf8s %zmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7d,0x48,0x3a,0xc1]
+          vcvtrops2hf8s %zmm1, %xmm0
+
+// CHECK: vcvtrops2hf8s %ymm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7d,0x28,0x3a,0xc1]
+          vcvtrops2hf8s %ymm1, %xmm0
+
+// CHECK: vcvtrops2hf8s %xmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7d,0x08,0x3a,0xc1]
+          vcvtrops2hf8s %xmm1, %xmm0
+
+// CHECK: vcvtrops2hf8s %zmm1, %xmm0 {%k1}
+// CHECK: encoding: [0x62,0xf5,0x7d,0x49,0x3a,0xc1]
+          vcvtrops2hf8s %zmm1, %xmm0 {%k1}
+
+// CHECK: vcvtrops2hf8s %zmm1, %xmm0 {%k1} {z}
+// CHECK: encoding: [0x62,0xf5,0x7d,0xc9,0x3a,0xc1]
+          vcvtrops2hf8s %zmm1, %xmm0 {%k1} {z}
+
+//
+// Group B: Bias PS->8bit conversions (3-operand)
+//
+
+// vcvtbiasps2bf8
+
+// CHECK: vcvtbiasps2bf8 %zmm2, %zmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x48,0x39,0xc2]
+          vcvtbiasps2bf8 %zmm2, %zmm1, %xmm0
+
+// CHECK: vcvtbiasps2bf8 %ymm2, %ymm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x28,0x39,0xc2]
+          vcvtbiasps2bf8 %ymm2, %ymm1, %xmm0
+
+// CHECK: vcvtbiasps2bf8 %xmm2, %xmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x08,0x39,0xc2]
+          vcvtbiasps2bf8 %xmm2, %xmm1, %xmm0
+
+// CHECK: vcvtbiasps2bf8 %zmm2, %zmm1, %xmm0 {%k1}
+// CHECK: encoding: [0x62,0xf5,0x74,0x49,0x39,0xc2]
+          vcvtbiasps2bf8 %zmm2, %zmm1, %xmm0 {%k1}
+
+// CHECK: vcvtbiasps2bf8 %zmm2, %zmm1, %xmm0 {%k1} {z}
+// CHECK: encoding: [0x62,0xf5,0x74,0xc9,0x39,0xc2]
+          vcvtbiasps2bf8 %zmm2, %zmm1, %xmm0 {%k1} {z}
+
+// vcvtbiasps2bf8s
+
+// CHECK: vcvtbiasps2bf8s %zmm2, %zmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x48,0x3b,0xc2]
+          vcvtbiasps2bf8s %zmm2, %zmm1, %xmm0
+
+// CHECK: vcvtbiasps2bf8s %ymm2, %ymm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x28,0x3b,0xc2]
+          vcvtbiasps2bf8s %ymm2, %ymm1, %xmm0
+
+// CHECK: vcvtbiasps2bf8s %xmm2, %xmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x08,0x3b,0xc2]
+          vcvtbiasps2bf8s %xmm2, %xmm1, %xmm0
+
+// CHECK: vcvtbiasps2bf8s %zmm2, %zmm1, %xmm0 {%k1}
+// CHECK: encoding: [0x62,0xf5,0x74,0x49,0x3b,0xc2]
+          vcvtbiasps2bf8s %zmm2, %zmm1, %xmm0 {%k1}
+
+// CHECK: vcvtbiasps2bf8s %zmm2, %zmm1, %xmm0 {%k1} {z}
+// CHECK: encoding: [0x62,0xf5,0x74,0xc9,0x3b,0xc2]
+          vcvtbiasps2bf8s %zmm2, %zmm1, %xmm0 {%k1} {z}
+
+// vcvtbiasps2hf8
+
+// CHECK: vcvtbiasps2hf8 %zmm2, %zmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x48,0x38,0xc2]
+          vcvtbiasps2hf8 %zmm2, %zmm1, %xmm0
+
+// CHECK: vcvtbiasps2hf8 %ymm2, %ymm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x28,0x38,0xc2]
+          vcvtbiasps2hf8 %ymm2, %ymm1, %xmm0
+
+// CHECK: vcvtbiasps2hf8 %xmm2, %xmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x08,0x38,0xc2]
+          vcvtbiasps2hf8 %xmm2, %xmm1, %xmm0
+
+// CHECK: vcvtbiasps2hf8 %zmm2, %zmm1, %xmm0 {%k1}
+// CHECK: encoding: [0x62,0xf5,0x74,0x49,0x38,0xc2]
+          vcvtbiasps2hf8 %zmm2, %zmm1, %xmm0 {%k1}
+
+// CHECK: vcvtbiasps2hf8 %zmm2, %zmm1, %xmm0 {%k1} {z}
+// CHECK: encoding: [0x62,0xf5,0x74,0xc9,0x38,0xc2]
+          vcvtbiasps2hf8 %zmm2, %zmm1, %xmm0 {%k1} {z}
+
+// vcvtbiasps2hf8s
+
+// CHECK: vcvtbiasps2hf8s %zmm2, %zmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x48,0x3a,0xc2]
+          vcvtbiasps2hf8s %zmm2, %zmm1, %xmm0
+
+// CHECK: vcvtbiasps2hf8s %ymm2, %ymm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x28,0x3a,0xc2]
+          vcvtbiasps2hf8s %ymm2, %ymm1, %xmm0
+
+// CHECK: vcvtbiasps2hf8s %xmm2, %xmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x08,0x3a,0xc2]
+          vcvtbiasps2hf8s %xmm2, %xmm1, %xmm0
+
+// CHECK: vcvtbiasps2hf8s %zmm2, %zmm1, %xmm0 {%k1}
+// CHECK: encoding: [0x62,0xf5,0x74,0x49,0x3a,0xc2]
+          vcvtbiasps2hf8s %zmm2, %zmm1, %xmm0 {%k1}
+
+// CHECK: vcvtbiasps2hf8s %zmm2, %zmm1, %xmm0 {%k1} {z}
+// CHECK: encoding: [0x62,0xf5,0x74,0xc9,0x3a,0xc2]
+          vcvtbiasps2hf8s %zmm2, %zmm1, %xmm0 {%k1} {z}
+
+//
+// Group C: 8bit->PS expanding conversions
+//
+
+// vcvtbf82ps
+
+// CHECK: vcvtbf82ps %xmm1, %zmm0
+// CHECK: encoding: [0x62,0xf5,0xfc,0x48,0x36,0xc1]
+          vcvtbf82ps %xmm1, %zmm0
+
+// CHECK: vcvtbf82ps %xmm1, %ymm0
+// CHECK: encoding: [0x62,0xf5,0xfc,0x28,0x36,0xc1]
+          vcvtbf82ps %xmm1, %ymm0
+
+// CHECK: vcvtbf82ps %xmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0xfc,0x08,0x36,0xc1]
+          vcvtbf82ps %xmm1, %xmm0
+
+// CHECK: vcvtbf82ps %xmm1, %zmm0 {%k1}
+// CHECK: encoding: [0x62,0xf5,0xfc,0x49,0x36,0xc1]
+          vcvtbf82ps %xmm1, %zmm0 {%k1}
+
+// CHECK: vcvtbf82ps %xmm1, %zmm0 {%k1} {z}
+// CHECK: encoding: [0x62,0xf5,0xfc,0xc9,0x36,0xc1]
+          vcvtbf82ps %xmm1, %zmm0 {%k1} {z}
+
+// vcvthf82ps
+
+// CHECK: vcvthf82ps %xmm1, %zmm0
+// CHECK: encoding: [0x62,0xf5,0x7c,0x48,0x36,0xc1]
+          vcvthf82ps %xmm1, %zmm0
+
+// CHECK: vcvthf82ps %xmm1, %ymm0
+// CHECK: encoding: [0x62,0xf5,0x7c,0x28,0x36,0xc1]
+          vcvthf82ps %xmm1, %ymm0
+
+// CHECK: vcvthf82ps %xmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7c,0x08,0x36,0xc1]
+          vcvthf82ps %xmm1, %xmm0
+
+// CHECK: vcvthf82ps %xmm1, %zmm0 {%k1}
+// CHECK: encoding: [0x62,0xf5,0x7c,0x49,0x36,0xc1]
+          vcvthf82ps %xmm1, %zmm0 {%k1}
+
+// CHECK: vcvthf82ps %xmm1, %zmm0 {%k1} {z}
+// CHECK: encoding: [0x62,0xf5,0x7c,0xc9,0x36,0xc1]
+          vcvthf82ps %xmm1, %zmm0 {%k1} {z}
+
+//
+// Group D: BF8/HF8->BF4S truncations
+//
+
+// vcvtbf82bf4s
+
+// CHECK: vcvtbf82bf4s %zmm1, %ymm0
+// CHECK: encoding: [0x62,0xf5,0xfe,0x48,0x3d,0xc8]
+          vcvtbf82bf4s %zmm1, %ymm0
+
+// CHECK: vcvtbf82bf4s %ymm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0xfe,0x28,0x3d,0xc8]
+          vcvtbf82bf4s %ymm1, %xmm0
+
+// CHECK: vcvtbf82bf4s %xmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0xfe,0x08,0x3d,0xc8]
+          vcvtbf82bf4s %xmm1, %xmm0
+
+// vcvthf82bf4s
+
+// CHECK: vcvthf82bf4s %zmm1, %ymm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x3d,0xc8]
+          vcvthf82bf4s %zmm1, %ymm0
+
+// CHECK: vcvthf82bf4s %ymm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x28,0x3d,0xc8]
+          vcvthf82bf4s %ymm1, %xmm0
+
+// CHECK: vcvthf82bf4s %xmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x3d,0xc8]
+          vcvthf82bf4s %xmm1, %xmm0
+
+//
+// Group E: Same-size reg-only conversions (no masking)
+//
+
+// vcvtbf82bf6s
+
+// CHECK: vcvtbf82bf6s %zmm1, %zmm0
+// CHECK: encoding: [0x62,0xf5,0xfe,0x48,0x3e,0xc1]
+          vcvtbf82bf6s %zmm1, %zmm0
+
+// CHECK: vcvtbf82bf6s %ymm1, %ymm0
+// CHECK: encoding: [0x62,0xf5,0xfe,0x28,0x3e,0xc1]
+          vcvtbf82bf6s %ymm1, %ymm0
+
+// CHECK: vcvtbf82bf6s %xmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0xfe,0x08,0x3e,0xc1]
+          vcvtbf82bf6s %xmm1, %xmm0
+
+// vcvthf82hf6s
+
+// CHECK: vcvthf82hf6s %zmm1, %zmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x3c,0xc1]
+          vcvthf82hf6s %zmm1, %zmm0
+
+// CHECK: vcvthf82hf6s %ymm1, %ymm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x28,0x3c,0xc1]
+          vcvthf82hf6s %ymm1, %ymm0
+
+// CHECK: vcvthf82hf6s %xmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x3c,0xc1]
+          vcvthf82hf6s %xmm1, %xmm0
+
+//
+// Group F: Expanding/same-size conversions with masking
+//
+
+// vcvtbf42hf8
+
+// CHECK: vcvtbf42hf8 %ymm1, %zmm0
+// CHECK: encoding: [0x62,0xf5,0x7c,0x48,0x37,0xc1]
+          vcvtbf42hf8 %ymm1, %zmm0
+
+// CHECK: vcvtbf42hf8 %xmm1, %ymm0
+// CHECK: encoding: [0x62,0xf5,0x7c,0x28,0x37,0xc1]
+          vcvtbf42hf8 %xmm1, %ymm0
+
+// CHECK: vcvtbf42hf8 %xmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7c,0x08,0x37,0xc1]
+          vcvtbf42hf8 %xmm1, %xmm0
+
+// CHECK: vcvtbf42hf8 %ymm1, %zmm0 {%k1}
+// CHECK: encoding: [0x62,0xf5,0x7c,0x49,0x37,0xc1]
+          vcvtbf42hf8 %ymm1, %zmm0 {%k1}
+
+// CHECK: vcvtbf42hf8 %ymm1, %zmm0 {%k1} {z}
+// CHECK: encoding: [0x62,0xf5,0x7c,0xc9,0x37,0xc1]
+          vcvtbf42hf8 %ymm1, %zmm0 {%k1} {z}
+
+// vcvtbf62hf8
+
+// CHECK: vcvtbf62hf8 %zmm1, %zmm0
+// CHECK: encoding: [0x62,0xf5,0xfd,0x48,0x37,0xc1]
+          vcvtbf62hf8 %zmm1, %zmm0
+
+// CHECK: vcvtbf62hf8 %ymm1, %ymm0
+// CHECK: encoding: [0x62,0xf5,0xfd,0x28,0x37,0xc1]
+          vcvtbf62hf8 %ymm1, %ymm0
+
+// CHECK: vcvtbf62hf8 %xmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0xfd,0x08,0x37,0xc1]
+          vcvtbf62hf8 %xmm1, %xmm0
+
+// CHECK: vcvtbf62hf8 %zmm1, %zmm0 {%k1}
+// CHECK: encoding: [0x62,0xf5,0xfd,0x49,0x37,0xc1]
+          vcvtbf62hf8 %zmm1, %zmm0 {%k1}
+
+// CHECK: vcvtbf62hf8 %zmm1, %zmm0 {%k1} {z}
+// CHECK: encoding: [0x62,0xf5,0xfd,0xc9,0x37,0xc1]
+          vcvtbf62hf8 %zmm1, %zmm0 {%k1} {z}
+
+// vcvthf62hf8
+
+// CHECK: vcvthf62hf8 %zmm1, %zmm0
+// CHECK: encoding: [0x62,0xf5,0x7d,0x48,0x37,0xc1]
+          vcvthf62hf8 %zmm1, %zmm0
+
+// CHECK: vcvthf62hf8 %ymm1, %ymm0
+// CHECK: encoding: [0x62,0xf5,0x7d,0x28,0x37,0xc1]
+          vcvthf62hf8 %ymm1, %ymm0
+
+// CHECK: vcvthf62hf8 %xmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7d,0x08,0x37,0xc1]
+          vcvthf62hf8 %xmm1, %xmm0
+
+// CHECK: vcvthf62hf8 %zmm1, %zmm0 {%k1}
+// CHECK: encoding: [0x62,0xf5,0x7d,0x49,0x37,0xc1]
+          vcvthf62hf8 %zmm1, %zmm0 {%k1}
+
+// CHECK: vcvthf62hf8 %zmm1, %zmm0 {%k1} {z}
+// CHECK: encoding: [0x62,0xf5,0x7d,0xc9,0x37,0xc1]
+          vcvthf62hf8 %zmm1, %zmm0 {%k1} {z}
+
+//
+// Group G: VPMOVSSDB - Integer DWord->Byte signed saturation
+//
+
+// CHECK: vpmovssdb %zmm1, %xmm0
+// CHECK: encoding: [0x62,0xf2,0x7e,0x48,0x41,0xc8]
+          vpmovssdb %zmm1, %xmm0
+
+// CHECK: vpmovssdb %ymm1, %xmm0
+// CHECK: encoding: [0x62,0xf2,0x7e,0x28,0x41,0xc8]
+          vpmovssdb %ymm1, %xmm0
+
+// CHECK: vpmovssdb %xmm1, %xmm0
+// CHECK: encoding: [0x62,0xf2,0x7e,0x08,0x41,0xc8]
+          vpmovssdb %xmm1, %xmm0
+
+// CHECK: vpmovssdb %zmm1, %xmm0 {%k1}
+// CHECK: encoding: [0x62,0xf2,0x7e,0x49,0x41,0xc8]
+          vpmovssdb %zmm1, %xmm0 {%k1}
+
+// CHECK: vpmovssdb %zmm1, %xmm0 {%k1} {z}
+// CHECK: encoding: [0x62,0xf2,0x7e,0xc9,0x41,0xc8]
+          vpmovssdb %zmm1, %xmm0 {%k1} {z}
+
+//
+// Group H: VUNPACKB - Byte unpack with immediate
+//
+
+// CHECK: vunpackb $1, %zmm1, %zmm0
+// CHECK: encoding: [0x62,0xf3,0x7c,0x48,0x3d,0xc1,0x01]
+          vunpackb $1, %zmm1, %zmm0
+
+// CHECK: vunpackb $1, %ymm1, %ymm0
+// CHECK: encoding: [0x62,0xf3,0x7c,0x28,0x3d,0xc1,0x01]
+          vunpackb $1, %ymm1, %ymm0
+
+// CHECK: vunpackb $1, %xmm1, %xmm0
+// CHECK: encoding: [0x62,0xf3,0x7c,0x08,0x3d,0xc1,0x01]
+          vunpackb $1, %xmm1, %xmm0
+
+// CHECK: vunpackb $1, %zmm1, %zmm0 {%k1}
+// CHECK: encoding: [0x62,0xf3,0x7c,0x49,0x3d,0xc1,0x01]
+          vunpackb $1, %zmm1, %zmm0 {%k1}
+
+// CHECK: vunpackb $1, %zmm1, %zmm0 {%k1} {z}
+// CHECK: encoding: [0x62,0xf3,0x7c,0xc9,0x3d,0xc1,0x01]
+          vunpackb $1, %zmm1, %zmm0 {%k1} {z}
diff --git a/llvm/test/MC/X86/avx10_v2_aux-att-64.s b/llvm/test/MC/X86/avx10_v2_aux-att-64.s
new file mode 100644
index 00000000000000..81154f3d8b33c6
--- /dev/null
+++ b/llvm/test/MC/X86/avx10_v2_aux-att-64.s
@@ -0,0 +1,687 @@
+// RUN: llvm-mc -triple x86_64 --show-encoding -mattr=+avx10-v2-aux,+avx512vl %s | FileCheck %s
+
+//
+// Group A: PS->8bit truncating conversions
+//
+
+// vcvtps2bf8
+
+// CHECK: vcvtps2bf8 %zmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x39,0xc1]
+          vcvtps2bf8 %zmm1, %xmm0
+
+// CHECK: vcvtps2bf8 %ymm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x28,0x39,0xc1]
+          vcvtps2bf8 %ymm1, %xmm0
+
+// CHECK: vcvtps2bf8 %xmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x39,0xc1]
+          vcvtps2bf8 %xmm1, %xmm0
+
+// CHECK: vcvtps2bf8 (%rdi), %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x39,0x07]
+          vcvtps2bf8 (%rdi), %xmm0
+
+// CHECK: vcvtps2bf8y (%rdi), %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x28,0x39,0x07]
+          vcvtps2bf8y (%rdi), %xmm0
+
+// CHECK: vcvtps2bf8x (%rdi), %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x39,0x07]
+          vcvtps2bf8x (%rdi), %xmm0
+
+// CHECK: vcvtps2bf8 %zmm1, %xmm0 {%k1}
+// CHECK: encoding: [0x62,0xf5,0x7e,0x49,0x39,0xc1]
+          vcvtps2bf8 %zmm1, %xmm0 {%k1}
+
+// CHECK: vcvtps2bf8 %zmm1, %xmm0 {%k1} {z}
+// CHECK: encoding: [0x62,0xf5,0x7e,0xc9,0x39,0xc1]
+          vcvtps2bf8 %zmm1, %xmm0 {%k1} {z}
+
+// CHECK: vcvtps2bf8 (%rdi){1to16}, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x58,0x39,0x07]
+          vcvtps2bf8 (%rdi){1to16}, %xmm0
+
+// vcvtps2bf8s
+
+// CHECK: vcvtps2bf8s %zmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x3b,0xc1]
+          vcvtps2bf8s %zmm1, %xmm0
+
+// CHECK: vcvtps2bf8s %ymm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x28,0x3b,0xc1]
+          vcvtps2bf8s %ymm1, %xmm0
+
+// CHECK: vcvtps2bf8s %xmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x3b,0xc1]
+          vcvtps2bf8s %xmm1, %xmm0
+
+// CHECK: vcvtps2bf8s (%rdi), %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x3b,0x07]
+          vcvtps2bf8s (%rdi), %xmm0
+
+// CHECK: vcvtps2bf8sy (%rdi), %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x28,0x3b,0x07]
+          vcvtps2bf8sy (%rdi), %xmm0
+
+// CHECK: vcvtps2bf8sx (%rdi), %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x3b,0x07]
+          vcvtps2bf8sx (%rdi), %xmm0
+
+// CHECK: vcvtps2bf8s %zmm1, %xmm0 {%k1}
+// CHECK: encoding: [0x62,0xf5,0x7e,0x49,0x3b,0xc1]
+          vcvtps2bf8s %zmm1, %xmm0 {%k1}
+
+// CHECK: vcvtps2bf8s %zmm1, %xmm0 {%k1} {z}
+// CHECK: encoding: [0x62,0xf5,0x7e,0xc9,0x3b,0xc1]
+          vcvtps2bf8s %zmm1, %xmm0 {%k1} {z}
+
+// CHECK: vcvtps2bf8s (%rdi){1to16}, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x58,0x3b,0x07]
+          vcvtps2bf8s (%rdi){1to16}, %xmm0
+
+// vcvtps2hf8
+
+// CHECK: vcvtps2hf8 %zmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x38,0xc1]
+          vcvtps2hf8 %zmm1, %xmm0
+
+// CHECK: vcvtps2hf8 %ymm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x28,0x38,0xc1]
+          vcvtps2hf8 %ymm1, %xmm0
+
+// CHECK: vcvtps2hf8 %xmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x38,0xc1]
+          vcvtps2hf8 %xmm1, %xmm0
+
+// CHECK: vcvtps2hf8 (%rdi), %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x38,0x07]
+          vcvtps2hf8 (%rdi), %xmm0
+
+// CHECK: vcvtps2hf8y (%rdi), %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x28,0x38,0x07]
+          vcvtps2hf8y (%rdi), %xmm0
+
+// CHECK: vcvtps2hf8x (%rdi), %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x38,0x07]
+          vcvtps2hf8x (%rdi), %xmm0
+
+// CHECK: vcvtps2hf8 %zmm1, %xmm0 {%k1}
+// CHECK: encoding: [0x62,0xf5,0x7e,0x49,0x38,0xc1]
+          vcvtps2hf8 %zmm1, %xmm0 {%k1}
+
+// CHECK: vcvtps2hf8 %zmm1, %xmm0 {%k1} {z}
+// CHECK: encoding: [0x62,0xf5,0x7e,0xc9,0x38,0xc1]
+          vcvtps2hf8 %zmm1, %xmm0 {%k1} {z}
+
+// CHECK: vcvtps2hf8 (%rdi){1to16}, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x58,0x38,0x07]
+          vcvtps2hf8 (%rdi){1to16}, %xmm0
+
+// vcvtps2hf8s
+
+// CHECK: vcvtps2hf8s %zmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x3a,0xc1]
+          vcvtps2hf8s %zmm1, %xmm0
+
+// CHECK: vcvtps2hf8s %ymm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x28,0x3a,0xc1]
+          vcvtps2hf8s %ymm1, %xmm0
+
+// CHECK: vcvtps2hf8s %xmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x3a,0xc1]
+          vcvtps2hf8s %xmm1, %xmm0
+
+// CHECK: vcvtps2hf8s (%rdi), %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x3a,0x07]
+          vcvtps2hf8s (%rdi), %xmm0
+
+// CHECK: vcvtps2hf8sy (%rdi), %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x28,0x3a,0x07]
+          vcvtps2hf8sy (%rdi), %xmm0
+
+// CHECK: vcvtps2hf8sx (%rdi), %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x3a,0x07]
+          vcvtps2hf8sx (%rdi), %xmm0
+
+// CHECK: vcvtps2hf8s %zmm1, %xmm0 {%k1}
+// CHECK: encoding: [0x62,0xf5,0x7e,0x49,0x3a,0xc1]
+          vcvtps2hf8s %zmm1, %xmm0 {%k1}
+
+// CHECK: vcvtps2hf8s %zmm1, %xmm0 {%k1} {z}
+// CHECK: encoding: [0x62,0xf5,0x7e,0xc9,0x3a,0xc1]
+          vcvtps2hf8s %zmm1, %xmm0 {%k1} {z}
+
+// CHECK: vcvtps2hf8s (%rdi){1to16}, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x58,0x3a,0x07]
+          vcvtps2hf8s (%rdi){1to16}, %xmm0
+
+// vcvtrops2hf8
+
+// CHECK: vcvtrops2hf8 %zmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7d,0x48,0x38,0xc1]
+          vcvtrops2hf8 %zmm1, %xmm0
+
+// CHECK: vcvtrops2hf8 %ymm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7d,0x28,0x38,0xc1]
+          vcvtrops2hf8 %ymm1, %xmm0
+
+// CHECK: vcvtrops2hf8 %xmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7d,0x08,0x38,0xc1]
+          vcvtrops2hf8 %xmm1, %xmm0
+
+// CHECK: vcvtrops2hf8 (%rdi), %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7d,0x48,0x38,0x07]
+          vcvtrops2hf8 (%rdi), %xmm0
+
+// CHECK: vcvtrops2hf8y (%rdi), %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7d,0x28,0x38,0x07]
+          vcvtrops2hf8y (%rdi), %xmm0
+
+// CHECK: vcvtrops2hf8x (%rdi), %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7d,0x08,0x38,0x07]
+          vcvtrops2hf8x (%rdi), %xmm0
+
+// CHECK: vcvtrops2hf8 %zmm1, %xmm0 {%k1}
+// CHECK: encoding: [0x62,0xf5,0x7d,0x49,0x38,0xc1]
+          vcvtrops2hf8 %zmm1, %xmm0 {%k1}
+
+// CHECK: vcvtrops2hf8 %zmm1, %xmm0 {%k1} {z}
+// CHECK: encoding: [0x62,0xf5,0x7d,0xc9,0x38,0xc1]
+          vcvtrops2hf8 %zmm1, %xmm0 {%k1} {z}
+
+// CHECK: vcvtrops2hf8 (%rdi){1to16}, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7d,0x58,0x38,0x07]
+          vcvtrops2hf8 (%rdi){1to16}, %xmm0
+
+// vcvtrops2hf8s
+
+// CHECK: vcvtrops2hf8s %zmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7d,0x48,0x3a,0xc1]
+          vcvtrops2hf8s %zmm1, %xmm0
+
+// CHECK: vcvtrops2hf8s %ymm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7d,0x28,0x3a,0xc1]
+          vcvtrops2hf8s %ymm1, %xmm0
+
+// CHECK: vcvtrops2hf8s %xmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7d,0x08,0x3a,0xc1]
+          vcvtrops2hf8s %xmm1, %xmm0
+
+// CHECK: vcvtrops2hf8s (%rdi), %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7d,0x48,0x3a,0x07]
+          vcvtrops2hf8s (%rdi), %xmm0
+
+// CHECK: vcvtrops2hf8sy (%rdi), %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7d,0x28,0x3a,0x07]
+          vcvtrops2hf8sy (%rdi), %xmm0
+
+// CHECK: vcvtrops2hf8sx (%rdi), %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7d,0x08,0x3a,0x07]
+          vcvtrops2hf8sx (%rdi), %xmm0
+
+// CHECK: vcvtrops2hf8s %zmm1, %xmm0 {%k1}
+// CHECK: encoding: [0x62,0xf5,0x7d,0x49,0x3a,0xc1]
+          vcvtrops2hf8s %zmm1, %xmm0 {%k1}
+
+// CHECK: vcvtrops2hf8s %zmm1, %xmm0 {%k1} {z}
+// CHECK: encoding: [0x62,0xf5,0x7d,0xc9,0x3a,0xc1]
+          vcvtrops2hf8s %zmm1, %xmm0 {%k1} {z}
+
+// CHECK: vcvtrops2hf8s (%rdi){1to16}, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7d,0x58,0x3a,0x07]
+          vcvtrops2hf8s (%rdi){1to16}, %xmm0
+
+//
+// Group B: Bias PS->8bit conversions (3-operand)
+//
+
+// vcvtbiasps2bf8
+
+// CHECK: vcvtbiasps2bf8 %zmm2, %zmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x48,0x39,0xc2]
+          vcvtbiasps2bf8 %zmm2, %zmm1, %xmm0
+
+// CHECK: vcvtbiasps2bf8 %ymm2, %ymm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x28,0x39,0xc2]
+          vcvtbiasps2bf8 %ymm2, %ymm1, %xmm0
+
+// CHECK: vcvtbiasps2bf8 %xmm2, %xmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x08,0x39,0xc2]
+          vcvtbiasps2bf8 %xmm2, %xmm1, %xmm0
+
+// CHECK: vcvtbiasps2bf8 (%rdi), %zmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x48,0x39,0x07]
+          vcvtbiasps2bf8 (%rdi), %zmm1, %xmm0
+
+// CHECK: vcvtbiasps2bf8 (%rdi), %ymm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x28,0x39,0x07]
+          vcvtbiasps2bf8 (%rdi), %ymm1, %xmm0
+
+// CHECK: vcvtbiasps2bf8 (%rdi), %xmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x08,0x39,0x07]
+          vcvtbiasps2bf8 (%rdi), %xmm1, %xmm0
+
+// CHECK: vcvtbiasps2bf8 %zmm2, %zmm1, %xmm0 {%k1}
+// CHECK: encoding: [0x62,0xf5,0x74,0x49,0x39,0xc2]
+          vcvtbiasps2bf8 %zmm2, %zmm1, %xmm0 {%k1}
+
+// CHECK: vcvtbiasps2bf8 %zmm2, %zmm1, %xmm0 {%k1} {z}
+// CHECK: encoding: [0x62,0xf5,0x74,0xc9,0x39,0xc2]
+          vcvtbiasps2bf8 %zmm2, %zmm1, %xmm0 {%k1} {z}
+
+// vcvtbiasps2bf8s
+
+// CHECK: vcvtbiasps2bf8s %zmm2, %zmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x48,0x3b,0xc2]
+          vcvtbiasps2bf8s %zmm2, %zmm1, %xmm0
+
+// CHECK: vcvtbiasps2bf8s %ymm2, %ymm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x28,0x3b,0xc2]
+          vcvtbiasps2bf8s %ymm2, %ymm1, %xmm0
+
+// CHECK: vcvtbiasps2bf8s %xmm2, %xmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x08,0x3b,0xc2]
+          vcvtbiasps2bf8s %xmm2, %xmm1, %xmm0
+
+// CHECK: vcvtbiasps2bf8s (%rdi), %zmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x48,0x3b,0x07]
+          vcvtbiasps2bf8s (%rdi), %zmm1, %xmm0
+
+// CHECK: vcvtbiasps2bf8s (%rdi), %ymm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x28,0x3b,0x07]
+          vcvtbiasps2bf8s (%rdi), %ymm1, %xmm0
+
+// CHECK: vcvtbiasps2bf8s (%rdi), %xmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x08,0x3b,0x07]
+          vcvtbiasps2bf8s (%rdi), %xmm1, %xmm0
+
+// CHECK: vcvtbiasps2bf8s %zmm2, %zmm1, %xmm0 {%k1}
+// CHECK: encoding: [0x62,0xf5,0x74,0x49,0x3b,0xc2]
+          vcvtbiasps2bf8s %zmm2, %zmm1, %xmm0 {%k1}
+
+// CHECK: vcvtbiasps2bf8s %zmm2, %zmm1, %xmm0 {%k1} {z}
+// CHECK: encoding: [0x62,0xf5,0x74,0xc9,0x3b,0xc2]
+          vcvtbiasps2bf8s %zmm2, %zmm1, %xmm0 {%k1} {z}
+
+// vcvtbiasps2hf8
+
+// CHECK: vcvtbiasps2hf8 %zmm2, %zmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x48,0x38,0xc2]
+          vcvtbiasps2hf8 %zmm2, %zmm1, %xmm0
+
+// CHECK: vcvtbiasps2hf8 %ymm2, %ymm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x28,0x38,0xc2]
+          vcvtbiasps2hf8 %ymm2, %ymm1, %xmm0
+
+// CHECK: vcvtbiasps2hf8 %xmm2, %xmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x08,0x38,0xc2]
+          vcvtbiasps2hf8 %xmm2, %xmm1, %xmm0
+
+// CHECK: vcvtbiasps2hf8 (%rdi), %zmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x48,0x38,0x07]
+          vcvtbiasps2hf8 (%rdi), %zmm1, %xmm0
+
+// CHECK: vcvtbiasps2hf8 (%rdi), %ymm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x28,0x38,0x07]
+          vcvtbiasps2hf8 (%rdi), %ymm1, %xmm0
+
+// CHECK: vcvtbiasps2hf8 (%rdi), %xmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x08,0x38,0x07]
+          vcvtbiasps2hf8 (%rdi), %xmm1, %xmm0
+
+// CHECK: vcvtbiasps2hf8 %zmm2, %zmm1, %xmm0 {%k1}
+// CHECK: encoding: [0x62,0xf5,0x74,0x49,0x38,0xc2]
+          vcvtbiasps2hf8 %zmm2, %zmm1, %xmm0 {%k1}
+
+// CHECK: vcvtbiasps2hf8 %zmm2, %zmm1, %xmm0 {%k1} {z}
+// CHECK: encoding: [0x62,0xf5,0x74,0xc9,0x38,0xc2]
+          vcvtbiasps2hf8 %zmm2, %zmm1, %xmm0 {%k1} {z}
+
+// vcvtbiasps2hf8s
+
+// CHECK: vcvtbiasps2hf8s %zmm2, %zmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x48,0x3a,0xc2]
+          vcvtbiasps2hf8s %zmm2, %zmm1, %xmm0
+
+// CHECK: vcvtbiasps2hf8s %ymm2, %ymm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x28,0x3a,0xc2]
+          vcvtbiasps2hf8s %ymm2, %ymm1, %xmm0
+
+// CHECK: vcvtbiasps2hf8s %xmm2, %xmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x08,0x3a,0xc2]
+          vcvtbiasps2hf8s %xmm2, %xmm1, %xmm0
+
+// CHECK: vcvtbiasps2hf8s (%rdi), %zmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x48,0x3a,0x07]
+          vcvtbiasps2hf8s (%rdi), %zmm1, %xmm0
+
+// CHECK: vcvtbiasps2hf8s (%rdi), %ymm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x28,0x3a,0x07]
+          vcvtbiasps2hf8s (%rdi), %ymm1, %xmm0
+
+// CHECK: vcvtbiasps2hf8s (%rdi), %xmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x08,0x3a,0x07]
+          vcvtbiasps2hf8s (%rdi), %xmm1, %xmm0
+
+// CHECK: vcvtbiasps2hf8s %zmm2, %zmm1, %xmm0 {%k1}
+// CHECK: encoding: [0x62,0xf5,0x74,0x49,0x3a,0xc2]
+          vcvtbiasps2hf8s %zmm2, %zmm1, %xmm0 {%k1}
+
+// CHECK: vcvtbiasps2hf8s %zmm2, %zmm1, %xmm0 {%k1} {z}
+// CHECK: encoding: [0x62,0xf5,0x74,0xc9,0x3a,0xc2]
+          vcvtbiasps2hf8s %zmm2, %zmm1, %xmm0 {%k1} {z}
+
+//
+// Group C: 8bit->PS expanding conversions
+//
+
+// vcvtbf82ps
+
+// CHECK: vcvtbf82ps %xmm1, %zmm0
+// CHECK: encoding: [0x62,0xf5,0xfc,0x48,0x36,0xc1]
+          vcvtbf82ps %xmm1, %zmm0
+
+// CHECK: vcvtbf82ps %xmm1, %ymm0
+// CHECK: encoding: [0x62,0xf5,0xfc,0x28,0x36,0xc1]
+          vcvtbf82ps %xmm1, %ymm0
+
+// CHECK: vcvtbf82ps %xmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0xfc,0x08,0x36,0xc1]
+          vcvtbf82ps %xmm1, %xmm0
+
+// CHECK: vcvtbf82ps (%rdi), %zmm0
+// CHECK: encoding: [0x62,0xf5,0xfc,0x48,0x36,0x07]
+          vcvtbf82ps (%rdi), %zmm0
+
+// CHECK: vcvtbf82ps (%rdi), %ymm0
+// CHECK: encoding: [0x62,0xf5,0xfc,0x28,0x36,0x07]
+          vcvtbf82ps (%rdi), %ymm0
+
+// CHECK: vcvtbf82ps (%rdi), %xmm0
+// CHECK: encoding: [0x62,0xf5,0xfc,0x08,0x36,0x07]
+          vcvtbf82ps (%rdi), %xmm0
+
+// CHECK: vcvtbf82ps %xmm1, %zmm0 {%k1}
+// CHECK: encoding: [0x62,0xf5,0xfc,0x49,0x36,0xc1]
+          vcvtbf82ps %xmm1, %zmm0 {%k1}
+
+// CHECK: vcvtbf82ps %xmm1, %zmm0 {%k1} {z}
+// CHECK: encoding: [0x62,0xf5,0xfc,0xc9,0x36,0xc1]
+          vcvtbf82ps %xmm1, %zmm0 {%k1} {z}
+
+// vcvthf82ps
+
+// CHECK: vcvthf82ps %xmm1, %zmm0
+// CHECK: encoding: [0x62,0xf5,0x7c,0x48,0x36,0xc1]
+          vcvthf82ps %xmm1, %zmm0
+
+// CHECK: vcvthf82ps %xmm1, %ymm0
+// CHECK: encoding: [0x62,0xf5,0x7c,0x28,0x36,0xc1]
+          vcvthf82ps %xmm1, %ymm0
+
+// CHECK: vcvthf82ps %xmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7c,0x08,0x36,0xc1]
+          vcvthf82ps %xmm1, %xmm0
+
+// CHECK: vcvthf82ps (%rdi), %zmm0
+// CHECK: encoding: [0x62,0xf5,0x7c,0x48,0x36,0x07]
+          vcvthf82ps (%rdi), %zmm0
+
+// CHECK: vcvthf82ps (%rdi), %ymm0
+// CHECK: encoding: [0x62,0xf5,0x7c,0x28,0x36,0x07]
+          vcvthf82ps (%rdi), %ymm0
+
+// CHECK: vcvthf82ps (%rdi), %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7c,0x08,0x36,0x07]
+          vcvthf82ps (%rdi), %xmm0
+
+// CHECK: vcvthf82ps %xmm1, %zmm0 {%k1}
+// CHECK: encoding: [0x62,0xf5,0x7c,0x49,0x36,0xc1]
+          vcvthf82ps %xmm1, %zmm0 {%k1}
+
+// CHECK: vcvthf82ps %xmm1, %zmm0 {%k1} {z}
+// CHECK: encoding: [0x62,0xf5,0x7c,0xc9,0x36,0xc1]
+          vcvthf82ps %xmm1, %zmm0 {%k1} {z}
+
+//
+// Group D: BF8/HF8->BF4S store-like truncations
+//
+
+// vcvtbf82bf4s
+
+// CHECK: vcvtbf82bf4s %zmm1, %ymm0
+// CHECK: encoding: [0x62,0xf5,0xfe,0x48,0x3d,0xc8]
+          vcvtbf82bf4s %zmm1, %ymm0
+
+// CHECK: vcvtbf82bf4s %ymm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0xfe,0x28,0x3d,0xc8]
+          vcvtbf82bf4s %ymm1, %xmm0
+
+// CHECK: vcvtbf82bf4s %xmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0xfe,0x08,0x3d,0xc8]
+          vcvtbf82bf4s %xmm1, %xmm0
+
+// CHECK: vcvtbf82bf4s %zmm1, (%rdi)
+// CHECK: encoding: [0x62,0xf5,0xfe,0x48,0x3d,0x0f]
+          vcvtbf82bf4s %zmm1, (%rdi)
+
+// CHECK: vcvtbf82bf4s %ymm1, (%rdi)
+// CHECK: encoding: [0x62,0xf5,0xfe,0x28,0x3d,0x0f]
+          vcvtbf82bf4s %ymm1, (%rdi)
+
+// CHECK: vcvtbf82bf4s %xmm1, (%rdi)
+// CHECK: encoding: [0x62,0xf5,0xfe,0x08,0x3d,0x0f]
+          vcvtbf82bf4s %xmm1, (%rdi)
+
+// vcvthf82bf4s
+
+// CHECK: vcvthf82bf4s %zmm1, %ymm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x3d,0xc8]
+          vcvthf82bf4s %zmm1, %ymm0
+
+// CHECK: vcvthf82bf4s %ymm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x28,0x3d,0xc8]
+          vcvthf82bf4s %ymm1, %xmm0
+
+// CHECK: vcvthf82bf4s %xmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x3d,0xc8]
+          vcvthf82bf4s %xmm1, %xmm0
+
+// CHECK: vcvthf82bf4s %zmm1, (%rdi)
+// CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x3d,0x0f]
+          vcvthf82bf4s %zmm1, (%rdi)
+
+// CHECK: vcvthf82bf4s %ymm1, (%rdi)
+// CHECK: encoding: [0x62,0xf5,0x7e,0x28,0x3d,0x0f]
+          vcvthf82bf4s %ymm1, (%rdi)
+
+// CHECK: vcvthf82bf4s %xmm1, (%rdi)
+// CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x3d,0x0f]
+          vcvthf82bf4s %xmm1, (%rdi)
+
+//
+// Group E: Same-size reg-only conversions (no masking)
+//
+
+// vcvtbf82bf6s
+
+// CHECK: vcvtbf82bf6s %zmm1, %zmm0
+// CHECK: encoding: [0x62,0xf5,0xfe,0x48,0x3e,0xc1]
+          vcvtbf82bf6s %zmm1, %zmm0
+
+// CHECK: vcvtbf82bf6s %ymm1, %ymm0
+// CHECK: encoding: [0x62,0xf5,0xfe,0x28,0x3e,0xc1]
+          vcvtbf82bf6s %ymm1, %ymm0
+
+// CHECK: vcvtbf82bf6s %xmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0xfe,0x08,0x3e,0xc1]
+          vcvtbf82bf6s %xmm1, %xmm0
+
+// vcvthf82hf6s
+
+// CHECK: vcvthf82hf6s %zmm1, %zmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x3c,0xc1]
+          vcvthf82hf6s %zmm1, %zmm0
+
+// CHECK: vcvthf82hf6s %ymm1, %ymm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x28,0x3c,0xc1]
+          vcvthf82hf6s %ymm1, %ymm0
+
+// CHECK: vcvthf82hf6s %xmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x3c,0xc1]
+          vcvthf82hf6s %xmm1, %xmm0
+
+//
+// Group F: Expanding/same-size conversions with masking
+//
+
+// vcvtbf42hf8
+
+// CHECK: vcvtbf42hf8 %ymm1, %zmm0
+// CHECK: encoding: [0x62,0xf5,0x7c,0x48,0x37,0xc1]
+          vcvtbf42hf8 %ymm1, %zmm0
+
+// CHECK: vcvtbf42hf8 %xmm1, %ymm0
+// CHECK: encoding: [0x62,0xf5,0x7c,0x28,0x37,0xc1]
+          vcvtbf42hf8 %xmm1, %ymm0
+
+// CHECK: vcvtbf42hf8 %xmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7c,0x08,0x37,0xc1]
+          vcvtbf42hf8 %xmm1, %xmm0
+
+// CHECK: vcvtbf42hf8 (%rdi), %zmm0
+// CHECK: encoding: [0x62,0xf5,0x7c,0x48,0x37,0x07]
+          vcvtbf42hf8 (%rdi), %zmm0
+
+// CHECK: vcvtbf42hf8 (%rdi), %ymm0
+// CHECK: encoding: [0x62,0xf5,0x7c,0x28,0x37,0x07]
+          vcvtbf42hf8 (%rdi), %ymm0
+
+// CHECK: vcvtbf42hf8 (%rdi), %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7c,0x08,0x37,0x07]
+          vcvtbf42hf8 (%rdi), %xmm0
+
+// CHECK: vcvtbf42hf8 %ymm1, %zmm0 {%k1}
+// CHECK: encoding: [0x62,0xf5,0x7c,0x49,0x37,0xc1]
+          vcvtbf42hf8 %ymm1, %zmm0 {%k1}
+
+// CHECK: vcvtbf42hf8 %ymm1, %zmm0 {%k1} {z}
+// CHECK: encoding: [0x62,0xf5,0x7c,0xc9,0x37,0xc1]
+          vcvtbf42hf8 %ymm1, %zmm0 {%k1} {z}
+
+// vcvtbf62hf8
+
+// CHECK: vcvtbf62hf8 %zmm1, %zmm0
+// CHECK: encoding: [0x62,0xf5,0xfd,0x48,0x37,0xc1]
+          vcvtbf62hf8 %zmm1, %zmm0
+
+// CHECK: vcvtbf62hf8 %ymm1, %ymm0
+// CHECK: encoding: [0x62,0xf5,0xfd,0x28,0x37,0xc1]
+          vcvtbf62hf8 %ymm1, %ymm0
+
+// CHECK: vcvtbf62hf8 %xmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0xfd,0x08,0x37,0xc1]
+          vcvtbf62hf8 %xmm1, %xmm0
+
+// CHECK: vcvtbf62hf8 %zmm1, %zmm0 {%k1}
+// CHECK: encoding: [0x62,0xf5,0xfd,0x49,0x37,0xc1]
+          vcvtbf62hf8 %zmm1, %zmm0 {%k1}
+
+// CHECK: vcvtbf62hf8 %zmm1, %zmm0 {%k1} {z}
+// CHECK: encoding: [0x62,0xf5,0xfd,0xc9,0x37,0xc1]
+          vcvtbf62hf8 %zmm1, %zmm0 {%k1} {z}
+
+// vcvthf62hf8
+
+// CHECK: vcvthf62hf8 %zmm1, %zmm0
+// CHECK: encoding: [0x62,0xf5,0x7d,0x48,0x37,0xc1]
+          vcvthf62hf8 %zmm1, %zmm0
+
+// CHECK: vcvthf62hf8 %ymm1, %ymm0
+// CHECK: encoding: [0x62,0xf5,0x7d,0x28,0x37,0xc1]
+          vcvthf62hf8 %ymm1, %ymm0
+
+// CHECK: vcvthf62hf8 %xmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7d,0x08,0x37,0xc1]
+          vcvthf62hf8 %xmm1, %xmm0
+
+// CHECK: vcvthf62hf8 %zmm1, %zmm0 {%k1}
+// CHECK: encoding: [0x62,0xf5,0x7d,0x49,0x37,0xc1]
+          vcvthf62hf8 %zmm1, %zmm0 {%k1}
+
+// CHECK: vcvthf62hf8 %zmm1, %zmm0 {%k1} {z}
+// CHECK: encoding: [0x62,0xf5,0x7d,0xc9,0x37,0xc1]
+          vcvthf62hf8 %zmm1, %zmm0 {%k1} {z}
+
+//
+// Group G: VPMOVSSDB - Integer DWord->Byte signed saturation
+//
+
+// CHECK: vpmovssdb %zmm1, %xmm0
+// CHECK: encoding: [0x62,0xf2,0x7e,0x48,0x41,0xc8]
+          vpmovssdb %zmm1, %xmm0
+
+// CHECK: vpmovssdb %ymm1, %xmm0
+// CHECK: encoding: [0x62,0xf2,0x7e,0x28,0x41,0xc8]
+          vpmovssdb %ymm1, %xmm0
+
+// CHECK: vpmovssdb %xmm1, %xmm0
+// CHECK: encoding: [0x62,0xf2,0x7e,0x08,0x41,0xc8]
+          vpmovssdb %xmm1, %xmm0
+
+// CHECK: vpmovssdb %zmm1, (%rdi)
+// CHECK: encoding: [0x62,0xf2,0x7e,0x48,0x41,0x0f]
+          vpmovssdb %zmm1, (%rdi)
+
+// CHECK: vpmovssdb %ymm1, (%rdi)
+// CHECK: encoding: [0x62,0xf2,0x7e,0x28,0x41,0x0f]
+          vpmovssdb %ymm1, (%rdi)
+
+// CHECK: vpmovssdb %xmm1, (%rdi)
+// CHECK: encoding: [0x62,0xf2,0x7e,0x08,0x41,0x0f]
+          vpmovssdb %xmm1, (%rdi)
+
+// CHECK: vpmovssdb %zmm1, %xmm0 {%k1}
+// CHECK: encoding: [0x62,0xf2,0x7e,0x49,0x41,0xc8]
+          vpmovssdb %zmm1, %xmm0 {%k1}
+
+// CHECK: vpmovssdb %zmm1, %xmm0 {%k1} {z}
+// CHECK: encoding: [0x62,0xf2,0x7e,0xc9,0x41,0xc8]
+          vpmovssdb %zmm1, %xmm0 {%k1} {z}
+
+//
+// Group H: VUNPACKB - Byte unpack with immediate
+//
+
+// CHECK: vunpackb $1, %zmm1, %zmm0
+// CHECK: encoding: [0x62,0xf3,0x7c,0x48,0x3d,0xc1,0x01]
+          vunpackb $1, %zmm1, %zmm0
+
+// CHECK: vunpackb $1, %ymm1, %ymm0
+// CHECK: encoding: [0x62,0xf3,0x7c,0x28,0x3d,0xc1,0x01]
+          vunpackb $1, %ymm1, %ymm0
+
+// CHECK: vunpackb $1, %xmm1, %xmm0
+// CHECK: encoding: [0x62,0xf3,0x7c,0x08,0x3d,0xc1,0x01]
+          vunpackb $1, %xmm1, %xmm0
+
+// CHECK: vunpackb $1, (%rdi), %zmm0
+// CHECK: encoding: [0x62,0xf3,0x7c,0x48,0x3d,0x07,0x01]
+          vunpackb $1, (%rdi), %zmm0
+
+// CHECK: vunpackb $1, (%rdi), %ymm0
+// CHECK: encoding: [0x62,0xf3,0x7c,0x28,0x3d,0x07,0x01]
+          vunpackb $1, (%rdi), %ymm0
+
+// CHECK: vunpackb $1, (%rdi), %xmm0
+// CHECK: encoding: [0x62,0xf3,0x7c,0x08,0x3d,0x07,0x01]
+          vunpackb $1, (%rdi), %xmm0
+
+// CHECK: vunpackb $1, %zmm1, %zmm0 {%k1}
+// CHECK: encoding: [0x62,0xf3,0x7c,0x49,0x3d,0xc1,0x01]
+          vunpackb $1, %zmm1, %zmm0 {%k1}
+
+// CHECK: vunpackb $1, %zmm1, %zmm0 {%k1} {z}
+// CHECK: encoding: [0x62,0xf3,0x7c,0xc9,0x3d,0xc1,0x01]
+          vunpackb $1, %zmm1, %zmm0 {%k1} {z}
diff --git a/llvm/test/MC/X86/avx10_v2_aux-intel-32.s b/llvm/test/MC/X86/avx10_v2_aux-intel-32.s
new file mode 100644
index 00000000000000..7a06743ed04b8d
--- /dev/null
+++ b/llvm/test/MC/X86/avx10_v2_aux-intel-32.s
@@ -0,0 +1,331 @@
+// RUN: llvm-mc -triple i386 -x86-asm-syntax=intel -output-asm-variant=1 --show-encoding -mattr=+avx10-v2-aux,+avx512vl %s | FileCheck %s
+
+//
+// Group A: PS->8bit truncating conversions
+//
+
+// vcvtps2bf8
+
+// CHECK: vcvtps2bf8 xmm0, zmm1
+// CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x39,0xc1]
+          vcvtps2bf8 xmm0, zmm1
+
+// CHECK: vcvtps2bf8 xmm0, ymm1
+// CHECK: encoding: [0x62,0xf5,0x7e,0x28,0x39,0xc1]
+          vcvtps2bf8 xmm0, ymm1
+
+// CHECK: vcvtps2bf8 xmm0, xmm1
+// CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x39,0xc1]
+          vcvtps2bf8 xmm0, xmm1
+
+// CHECK: vcvtps2bf8 xmm0 {k1}, zmm1
+// CHECK: encoding: [0x62,0xf5,0x7e,0x49,0x39,0xc1]
+          vcvtps2bf8 xmm0 {k1}, zmm1
+
+// CHECK: vcvtps2bf8 xmm0 {k1} {z}, zmm1
+// CHECK: encoding: [0x62,0xf5,0x7e,0xc9,0x39,0xc1]
+          vcvtps2bf8 xmm0 {k1} {z}, zmm1
+
+// vcvtps2bf8s
+
+// CHECK: vcvtps2bf8s xmm0, zmm1
+// CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x3b,0xc1]
+          vcvtps2bf8s xmm0, zmm1
+
+// CHECK: vcvtps2bf8s xmm0, ymm1
+// CHECK: encoding: [0x62,0xf5,0x7e,0x28,0x3b,0xc1]
+          vcvtps2bf8s xmm0, ymm1
+
+// CHECK: vcvtps2bf8s xmm0, xmm1
+// CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x3b,0xc1]
+          vcvtps2bf8s xmm0, xmm1
+
+// vcvtps2hf8
+
+// CHECK: vcvtps2hf8 xmm0, zmm1
+// CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x38,0xc1]
+          vcvtps2hf8 xmm0, zmm1
+
+// CHECK: vcvtps2hf8 xmm0, ymm1
+// CHECK: encoding: [0x62,0xf5,0x7e,0x28,0x38,0xc1]
+          vcvtps2hf8 xmm0, ymm1
+
+// CHECK: vcvtps2hf8 xmm0, xmm1
+// CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x38,0xc1]
+          vcvtps2hf8 xmm0, xmm1
+
+// vcvtps2hf8s
+
+// CHECK: vcvtps2hf8s xmm0, zmm1
+// CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x3a,0xc1]
+          vcvtps2hf8s xmm0, zmm1
+
+// CHECK: vcvtps2hf8s xmm0, ymm1
+// CHECK: encoding: [0x62,0xf5,0x7e,0x28,0x3a,0xc1]
+          vcvtps2hf8s xmm0, ymm1
+
+// CHECK: vcvtps2hf8s xmm0, xmm1
+// CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x3a,0xc1]
+          vcvtps2hf8s xmm0, xmm1
+
+// vcvtrops2hf8
+
+// CHECK: vcvtrops2hf8 xmm0, zmm1
+// CHECK: encoding: [0x62,0xf5,0x7d,0x48,0x38,0xc1]
+          vcvtrops2hf8 xmm0, zmm1
+
+// CHECK: vcvtrops2hf8 xmm0, ymm1
+// CHECK: encoding: [0x62,0xf5,0x7d,0x28,0x38,0xc1]
+          vcvtrops2hf8 xmm0, ymm1
+
+// CHECK: vcvtrops2hf8 xmm0, xmm1
+// CHECK: encoding: [0x62,0xf5,0x7d,0x08,0x38,0xc1]
+          vcvtrops2hf8 xmm0, xmm1
+
+// vcvtrops2hf8s
+
+// CHECK: vcvtrops2hf8s xmm0, zmm1
+// CHECK: encoding: [0x62,0xf5,0x7d,0x48,0x3a,0xc1]
+          vcvtrops2hf8s xmm0, zmm1
+
+// CHECK: vcvtrops2hf8s xmm0, ymm1
+// CHECK: encoding: [0x62,0xf5,0x7d,0x28,0x3a,0xc1]
+          vcvtrops2hf8s xmm0, ymm1
+
+// CHECK: vcvtrops2hf8s xmm0, xmm1
+// CHECK: encoding: [0x62,0xf5,0x7d,0x08,0x3a,0xc1]
+          vcvtrops2hf8s xmm0, xmm1
+
+//
+// Group B: Bias PS->8bit conversions (3-operand)
+//
+
+// vcvtbiasps2bf8
+
+// CHECK: vcvtbiasps2bf8 xmm0, zmm1, zmm2
+// CHECK: encoding: [0x62,0xf5,0x74,0x48,0x39,0xc2]
+          vcvtbiasps2bf8 xmm0, zmm1, zmm2
+
+// CHECK: vcvtbiasps2bf8 xmm0, ymm1, ymm2
+// CHECK: encoding: [0x62,0xf5,0x74,0x28,0x39,0xc2]
+          vcvtbiasps2bf8 xmm0, ymm1, ymm2
+
+// CHECK: vcvtbiasps2bf8 xmm0, xmm1, xmm2
+// CHECK: encoding: [0x62,0xf5,0x74,0x08,0x39,0xc2]
+          vcvtbiasps2bf8 xmm0, xmm1, xmm2
+
+// vcvtbiasps2bf8s
+
+// CHECK: vcvtbiasps2bf8s xmm0, zmm1, zmm2
+// CHECK: encoding: [0x62,0xf5,0x74,0x48,0x3b,0xc2]
+          vcvtbiasps2bf8s xmm0, zmm1, zmm2
+
+// CHECK: vcvtbiasps2bf8s xmm0, ymm1, ymm2
+// CHECK: encoding: [0x62,0xf5,0x74,0x28,0x3b,0xc2]
+          vcvtbiasps2bf8s xmm0, ymm1, ymm2
+
+// CHECK: vcvtbiasps2bf8s xmm0, xmm1, xmm2
+// CHECK: encoding: [0x62,0xf5,0x74,0x08,0x3b,0xc2]
+          vcvtbiasps2bf8s xmm0, xmm1, xmm2
+
+// vcvtbiasps2hf8
+
+// CHECK: vcvtbiasps2hf8 xmm0, zmm1, zmm2
+// CHECK: encoding: [0x62,0xf5,0x74,0x48,0x38,0xc2]
+          vcvtbiasps2hf8 xmm0, zmm1, zmm2
+
+// CHECK: vcvtbiasps2hf8 xmm0, ymm1, ymm2
+// CHECK: encoding: [0x62,0xf5,0x74,0x28,0x38,0xc2]
+          vcvtbiasps2hf8 xmm0, ymm1, ymm2
+
+// CHECK: vcvtbiasps2hf8 xmm0, xmm1, xmm2
+// CHECK: encoding: [0x62,0xf5,0x74,0x08,0x38,0xc2]
+          vcvtbiasps2hf8 xmm0, xmm1, xmm2
+
+// vcvtbiasps2hf8s
+
+// CHECK: vcvtbiasps2hf8s xmm0, zmm1, zmm2
+// CHECK: encoding: [0x62,0xf5,0x74,0x48,0x3a,0xc2]
+          vcvtbiasps2hf8s xmm0, zmm1, zmm2
+
+// CHECK: vcvtbiasps2hf8s xmm0, ymm1, ymm2
+// CHECK: encoding: [0x62,0xf5,0x74,0x28,0x3a,0xc2]
+          vcvtbiasps2hf8s xmm0, ymm1, ymm2
+
+// CHECK: vcvtbiasps2hf8s xmm0, xmm1, xmm2
+// CHECK: encoding: [0x62,0xf5,0x74,0x08,0x3a,0xc2]
+          vcvtbiasps2hf8s xmm0, xmm1, xmm2
+
+//
+// Group C: 8bit->PS expanding conversions
+//
+
+// vcvtbf82ps
+
+// CHECK: vcvtbf82ps zmm0, xmm1
+// CHECK: encoding: [0x62,0xf5,0xfc,0x48,0x36,0xc1]
+          vcvtbf82ps zmm0, xmm1
+
+// CHECK: vcvtbf82ps ymm0, xmm1
+// CHECK: encoding: [0x62,0xf5,0xfc,0x28,0x36,0xc1]
+          vcvtbf82ps ymm0, xmm1
+
+// CHECK: vcvtbf82ps xmm0, xmm1
+// CHECK: encoding: [0x62,0xf5,0xfc,0x08,0x36,0xc1]
+          vcvtbf82ps xmm0, xmm1
+
+// vcvthf82ps
+
+// CHECK: vcvthf82ps zmm0, xmm1
+// CHECK: encoding: [0x62,0xf5,0x7c,0x48,0x36,0xc1]
+          vcvthf82ps zmm0, xmm1
+
+// CHECK: vcvthf82ps ymm0, xmm1
+// CHECK: encoding: [0x62,0xf5,0x7c,0x28,0x36,0xc1]
+          vcvthf82ps ymm0, xmm1
+
+// CHECK: vcvthf82ps xmm0, xmm1
+// CHECK: encoding: [0x62,0xf5,0x7c,0x08,0x36,0xc1]
+          vcvthf82ps xmm0, xmm1
+
+//
+// Group D: BF8/HF8->BF4S truncations
+//
+
+// vcvtbf82bf4s
+
+// CHECK: vcvtbf82bf4s ymm0, zmm1
+// CHECK: encoding: [0x62,0xf5,0xfe,0x48,0x3d,0xc8]
+          vcvtbf82bf4s ymm0, zmm1
+
+// CHECK: vcvtbf82bf4s xmm0, ymm1
+// CHECK: encoding: [0x62,0xf5,0xfe,0x28,0x3d,0xc8]
+          vcvtbf82bf4s xmm0, ymm1
+
+// CHECK: vcvtbf82bf4s xmm0, xmm1
+// CHECK: encoding: [0x62,0xf5,0xfe,0x08,0x3d,0xc8]
+          vcvtbf82bf4s xmm0, xmm1
+
+// vcvthf82bf4s
+
+// CHECK: vcvthf82bf4s ymm0, zmm1
+// CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x3d,0xc8]
+          vcvthf82bf4s ymm0, zmm1
+
+// CHECK: vcvthf82bf4s xmm0, ymm1
+// CHECK: encoding: [0x62,0xf5,0x7e,0x28,0x3d,0xc8]
+          vcvthf82bf4s xmm0, ymm1
+
+// CHECK: vcvthf82bf4s xmm0, xmm1
+// CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x3d,0xc8]
+          vcvthf82bf4s xmm0, xmm1
+
+//
+// Group E: Same-size reg-only conversions (no masking)
+//
+
+// vcvtbf82bf6s
+
+// CHECK: vcvtbf82bf6s zmm0, zmm1
+// CHECK: encoding: [0x62,0xf5,0xfe,0x48,0x3e,0xc1]
+          vcvtbf82bf6s zmm0, zmm1
+
+// CHECK: vcvtbf82bf6s ymm0, ymm1
+// CHECK: encoding: [0x62,0xf5,0xfe,0x28,0x3e,0xc1]
+          vcvtbf82bf6s ymm0, ymm1
+
+// CHECK: vcvtbf82bf6s xmm0, xmm1
+// CHECK: encoding: [0x62,0xf5,0xfe,0x08,0x3e,0xc1]
+          vcvtbf82bf6s xmm0, xmm1
+
+// vcvthf82hf6s
+
+// CHECK: vcvthf82hf6s zmm0, zmm1
+// CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x3c,0xc1]
+          vcvthf82hf6s zmm0, zmm1
+
+// CHECK: vcvthf82hf6s ymm0, ymm1
+// CHECK: encoding: [0x62,0xf5,0x7e,0x28,0x3c,0xc1]
+          vcvthf82hf6s ymm0, ymm1
+
+// CHECK: vcvthf82hf6s xmm0, xmm1
+// CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x3c,0xc1]
+          vcvthf82hf6s xmm0, xmm1
+
+//
+// Group F: Expanding/same-size conversions with masking
+//
+
+// vcvtbf42hf8
+
+// CHECK: vcvtbf42hf8 zmm0, ymm1
+// CHECK: encoding: [0x62,0xf5,0x7c,0x48,0x37,0xc1]
+          vcvtbf42hf8 zmm0, ymm1
+
+// CHECK: vcvtbf42hf8 ymm0, xmm1
+// CHECK: encoding: [0x62,0xf5,0x7c,0x28,0x37,0xc1]
+          vcvtbf42hf8 ymm0, xmm1
+
+// CHECK: vcvtbf42hf8 xmm0, xmm1
+// CHECK: encoding: [0x62,0xf5,0x7c,0x08,0x37,0xc1]
+          vcvtbf42hf8 xmm0, xmm1
+
+// vcvtbf62hf8
+
+// CHECK: vcvtbf62hf8 zmm0, zmm1
+// CHECK: encoding: [0x62,0xf5,0xfd,0x48,0x37,0xc1]
+          vcvtbf62hf8 zmm0, zmm1
+
+// CHECK: vcvtbf62hf8 ymm0, ymm1
+// CHECK: encoding: [0x62,0xf5,0xfd,0x28,0x37,0xc1]
+          vcvtbf62hf8 ymm0, ymm1
+
+// CHECK: vcvtbf62hf8 xmm0, xmm1
+// CHECK: encoding: [0x62,0xf5,0xfd,0x08,0x37,0xc1]
+          vcvtbf62hf8 xmm0, xmm1
+
+// vcvthf62hf8
+
+// CHECK: vcvthf62hf8 zmm0, zmm1
+// CHECK: encoding: [0x62,0xf5,0x7d,0x48,0x37,0xc1]
+          vcvthf62hf8 zmm0, zmm1
+
+// CHECK: vcvthf62hf8 ymm0, ymm1
+// CHECK: encoding: [0x62,0xf5,0x7d,0x28,0x37,0xc1]
+          vcvthf62hf8 ymm0, ymm1
+
+// CHECK: vcvthf62hf8 xmm0, xmm1
+// CHECK: encoding: [0x62,0xf5,0x7d,0x08,0x37,0xc1]
+          vcvthf62hf8 xmm0, xmm1
+
+//
+// Group G: VPMOVSSDB - Integer DWord->Byte signed saturation
+//
+
+// CHECK: vpmovssdb xmm0, zmm1
+// CHECK: encoding: [0x62,0xf2,0x7e,0x48,0x41,0xc8]
+          vpmovssdb xmm0, zmm1
+
+// CHECK: vpmovssdb xmm0, ymm1
+// CHECK: encoding: [0x62,0xf2,0x7e,0x28,0x41,0xc8]
+          vpmovssdb xmm0, ymm1
+
+// CHECK: vpmovssdb xmm0, xmm1
+// CHECK: encoding: [0x62,0xf2,0x7e,0x08,0x41,0xc8]
+          vpmovssdb xmm0, xmm1
+
+//
+// Group H: VUNPACKB - Byte unpack with immediate
+//
+
+// CHECK: vunpackb zmm0, zmm1, 1
+// CHECK: encoding: [0x62,0xf3,0x7c,0x48,0x3d,0xc1,0x01]
+          vunpackb zmm0, zmm1, 1
+
+// CHECK: vunpackb ymm0, ymm1, 1
+// CHECK: encoding: [0x62,0xf3,0x7c,0x28,0x3d,0xc1,0x01]
+          vunpackb ymm0, ymm1, 1
+
+// CHECK: vunpackb xmm0, xmm1, 1
+// CHECK: encoding: [0x62,0xf3,0x7c,0x08,0x3d,0xc1,0x01]
+          vunpackb xmm0, xmm1, 1
diff --git a/llvm/test/MC/X86/avx10_v2_aux-intel-64.s b/llvm/test/MC/X86/avx10_v2_aux-intel-64.s
new file mode 100644
index 00000000000000..679a8ddc3afe57
--- /dev/null
+++ b/llvm/test/MC/X86/avx10_v2_aux-intel-64.s
@@ -0,0 +1,687 @@
+// RUN: llvm-mc -triple x86_64 -x86-asm-syntax=intel -output-asm-variant=1 --show-encoding -mattr=+avx10-v2-aux,+avx512vl %s | FileCheck %s
+
+//
+// Group A: PS->8bit truncating conversions
+//
+
+// vcvtps2bf8
+
+// CHECK: vcvtps2bf8 xmm0, zmm1
+// CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x39,0xc1]
+          vcvtps2bf8 xmm0, zmm1
+
+// CHECK: vcvtps2bf8 xmm0, ymm1
+// CHECK: encoding: [0x62,0xf5,0x7e,0x28,0x39,0xc1]
+          vcvtps2bf8 xmm0, ymm1
+
+// CHECK: vcvtps2bf8 xmm0, xmm1
+// CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x39,0xc1]
+          vcvtps2bf8 xmm0, xmm1
+
+// CHECK: vcvtps2bf8 xmm0, zmmword ptr [rdi]
+// CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x39,0x07]
+          vcvtps2bf8 xmm0, zmmword ptr [rdi]
+
+// CHECK: vcvtps2bf8 xmm0, ymmword ptr [rdi]
+// CHECK: encoding: [0x62,0xf5,0x7e,0x28,0x39,0x07]
+          vcvtps2bf8 xmm0, ymmword ptr [rdi]
+
+// CHECK: vcvtps2bf8 xmm0, xmmword ptr [rdi]
+// CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x39,0x07]
+          vcvtps2bf8 xmm0, xmmword ptr [rdi]
+
+// CHECK: vcvtps2bf8 xmm0 {k1}, zmm1
+// CHECK: encoding: [0x62,0xf5,0x7e,0x49,0x39,0xc1]
+          vcvtps2bf8 xmm0 {k1}, zmm1
+
+// CHECK: vcvtps2bf8 xmm0 {k1} {z}, zmm1
+// CHECK: encoding: [0x62,0xf5,0x7e,0xc9,0x39,0xc1]
+          vcvtps2bf8 xmm0 {k1} {z}, zmm1
+
+// CHECK: vcvtps2bf8 xmm0, dword ptr [rdi]{1to16}
+// CHECK: encoding: [0x62,0xf5,0x7e,0x58,0x39,0x07]
+          vcvtps2bf8 xmm0, dword ptr [rdi]{1to16}
+
+// vcvtps2bf8s
+
+// CHECK: vcvtps2bf8s xmm0, zmm1
+// CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x3b,0xc1]
+          vcvtps2bf8s xmm0, zmm1
+
+// CHECK: vcvtps2bf8s xmm0, ymm1
+// CHECK: encoding: [0x62,0xf5,0x7e,0x28,0x3b,0xc1]
+          vcvtps2bf8s xmm0, ymm1
+
+// CHECK: vcvtps2bf8s xmm0, xmm1
+// CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x3b,0xc1]
+          vcvtps2bf8s xmm0, xmm1
+
+// CHECK: vcvtps2bf8s xmm0, zmmword ptr [rdi]
+// CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x3b,0x07]
+          vcvtps2bf8s xmm0, zmmword ptr [rdi]
+
+// CHECK: vcvtps2bf8s xmm0, ymmword ptr [rdi]
+// CHECK: encoding: [0x62,0xf5,0x7e,0x28,0x3b,0x07]
+          vcvtps2bf8s xmm0, ymmword ptr [rdi]
+
+// CHECK: vcvtps2bf8s xmm0, xmmword ptr [rdi]
+// CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x3b,0x07]
+          vcvtps2bf8s xmm0, xmmword ptr [rdi]
+
+// CHECK: vcvtps2bf8s xmm0 {k1}, zmm1
+// CHECK: encoding: [0x62,0xf5,0x7e,0x49,0x3b,0xc1]
+          vcvtps2bf8s xmm0 {k1}, zmm1
+
+// CHECK: vcvtps2bf8s xmm0 {k1} {z}, zmm1
+// CHECK: encoding: [0x62,0xf5,0x7e,0xc9,0x3b,0xc1]
+          vcvtps2bf8s xmm0 {k1} {z}, zmm1
+
+// CHECK: vcvtps2bf8s xmm0, dword ptr [rdi]{1to16}
+// CHECK: encoding: [0x62,0xf5,0x7e,0x58,0x3b,0x07]
+          vcvtps2bf8s xmm0, dword ptr [rdi]{1to16}
+
+// vcvtps2hf8
+
+// CHECK: vcvtps2hf8 xmm0, zmm1
+// CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x38,0xc1]
+          vcvtps2hf8 xmm0, zmm1
+
+// CHECK: vcvtps2hf8 xmm0, ymm1
+// CHECK: encoding: [0x62,0xf5,0x7e,0x28,0x38,0xc1]
+          vcvtps2hf8 xmm0, ymm1
+
+// CHECK: vcvtps2hf8 xmm0, xmm1
+// CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x38,0xc1]
+          vcvtps2hf8 xmm0, xmm1
+
+// CHECK: vcvtps2hf8 xmm0, zmmword ptr [rdi]
+// CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x38,0x07]
+          vcvtps2hf8 xmm0, zmmword ptr [rdi]
+
+// CHECK: vcvtps2hf8 xmm0, ymmword ptr [rdi]
+// CHECK: encoding: [0x62,0xf5,0x7e,0x28,0x38,0x07]
+          vcvtps2hf8 xmm0, ymmword ptr [rdi]
+
+// CHECK: vcvtps2hf8 xmm0, xmmword ptr [rdi]
+// CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x38,0x07]
+          vcvtps2hf8 xmm0, xmmword ptr [rdi]
+
+// CHECK: vcvtps2hf8 xmm0 {k1}, zmm1
+// CHECK: encoding: [0x62,0xf5,0x7e,0x49,0x38,0xc1]
+          vcvtps2hf8 xmm0 {k1}, zmm1
+
+// CHECK: vcvtps2hf8 xmm0 {k1} {z}, zmm1
+// CHECK: encoding: [0x62,0xf5,0x7e,0xc9,0x38,0xc1]
+          vcvtps2hf8 xmm0 {k1} {z}, zmm1
+
+// CHECK: vcvtps2hf8 xmm0, dword ptr [rdi]{1to16}
+// CHECK: encoding: [0x62,0xf5,0x7e,0x58,0x38,0x07]
+          vcvtps2hf8 xmm0, dword ptr [rdi]{1to16}
+
+// vcvtps2hf8s
+
+// CHECK: vcvtps2hf8s xmm0, zmm1
+// CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x3a,0xc1]
+          vcvtps2hf8s xmm0, zmm1
+
+// CHECK: vcvtps2hf8s xmm0, ymm1
+// CHECK: encoding: [0x62,0xf5,0x7e,0x28,0x3a,0xc1]
+          vcvtps2hf8s xmm0, ymm1
+
+// CHECK: vcvtps2hf8s xmm0, xmm1
+// CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x3a,0xc1]
+          vcvtps2hf8s xmm0, xmm1
+
+// CHECK: vcvtps2hf8s xmm0, zmmword ptr [rdi]
+// CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x3a,0x07]
+          vcvtps2hf8s xmm0, zmmword ptr [rdi]
+
+// CHECK: vcvtps2hf8s xmm0, ymmword ptr [rdi]
+// CHECK: encoding: [0x62,0xf5,0x7e,0x28,0x3a,0x07]
+          vcvtps2hf8s xmm0, ymmword ptr [rdi]
+
+// CHECK: vcvtps2hf8s xmm0, xmmword ptr [rdi]
+// CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x3a,0x07]
+          vcvtps2hf8s xmm0, xmmword ptr [rdi]
+
+// CHECK: vcvtps2hf8s xmm0 {k1}, zmm1
+// CHECK: encoding: [0x62,0xf5,0x7e,0x49,0x3a,0xc1]
+          vcvtps2hf8s xmm0 {k1}, zmm1
+
+// CHECK: vcvtps2hf8s xmm0 {k1} {z}, zmm1
+// CHECK: encoding: [0x62,0xf5,0x7e,0xc9,0x3a,0xc1]
+          vcvtps2hf8s xmm0 {k1} {z}, zmm1
+
+// CHECK: vcvtps2hf8s xmm0, dword ptr [rdi]{1to16}
+// CHECK: encoding: [0x62,0xf5,0x7e,0x58,0x3a,0x07]
+          vcvtps2hf8s xmm0, dword ptr [rdi]{1to16}
+
+// vcvtrops2hf8
+
+// CHECK: vcvtrops2hf8 xmm0, zmm1
+// CHECK: encoding: [0x62,0xf5,0x7d,0x48,0x38,0xc1]
+          vcvtrops2hf8 xmm0, zmm1
+
+// CHECK: vcvtrops2hf8 xmm0, ymm1
+// CHECK: encoding: [0x62,0xf5,0x7d,0x28,0x38,0xc1]
+          vcvtrops2hf8 xmm0, ymm1
+
+// CHECK: vcvtrops2hf8 xmm0, xmm1
+// CHECK: encoding: [0x62,0xf5,0x7d,0x08,0x38,0xc1]
+          vcvtrops2hf8 xmm0, xmm1
+
+// CHECK: vcvtrops2hf8 xmm0, zmmword ptr [rdi]
+// CHECK: encoding: [0x62,0xf5,0x7d,0x48,0x38,0x07]
+          vcvtrops2hf8 xmm0, zmmword ptr [rdi]
+
+// CHECK: vcvtrops2hf8 xmm0, ymmword ptr [rdi]
+// CHECK: encoding: [0x62,0xf5,0x7d,0x28,0x38,0x07]
+          vcvtrops2hf8 xmm0, ymmword ptr [rdi]
+
+// CHECK: vcvtrops2hf8 xmm0, xmmword ptr [rdi]
+// CHECK: encoding: [0x62,0xf5,0x7d,0x08,0x38,0x07]
+          vcvtrops2hf8 xmm0, xmmword ptr [rdi]
+
+// CHECK: vcvtrops2hf8 xmm0 {k1}, zmm1
+// CHECK: encoding: [0x62,0xf5,0x7d,0x49,0x38,0xc1]
+          vcvtrops2hf8 xmm0 {k1}, zmm1
+
+// CHECK: vcvtrops2hf8 xmm0 {k1} {z}, zmm1
+// CHECK: encoding: [0x62,0xf5,0x7d,0xc9,0x38,0xc1]
+          vcvtrops2hf8 xmm0 {k1} {z}, zmm1
+
+// CHECK: vcvtrops2hf8 xmm0, dword ptr [rdi]{1to16}
+// CHECK: encoding: [0x62,0xf5,0x7d,0x58,0x38,0x07]
+          vcvtrops2hf8 xmm0, dword ptr [rdi]{1to16}
+
+// vcvtrops2hf8s
+
+// CHECK: vcvtrops2hf8s xmm0, zmm1
+// CHECK: encoding: [0x62,0xf5,0x7d,0x48,0x3a,0xc1]
+          vcvtrops2hf8s xmm0, zmm1
+
+// CHECK: vcvtrops2hf8s xmm0, ymm1
+// CHECK: encoding: [0x62,0xf5,0x7d,0x28,0x3a,0xc1]
+          vcvtrops2hf8s xmm0, ymm1
+
+// CHECK: vcvtrops2hf8s xmm0, xmm1
+// CHECK: encoding: [0x62,0xf5,0x7d,0x08,0x3a,0xc1]
+          vcvtrops2hf8s xmm0, xmm1
+
+// CHECK: vcvtrops2hf8s xmm0, zmmword ptr [rdi]
+// CHECK: encoding: [0x62,0xf5,0x7d,0x48,0x3a,0x07]
+          vcvtrops2hf8s xmm0, zmmword ptr [rdi]
+
+// CHECK: vcvtrops2hf8s xmm0, ymmword ptr [rdi]
+// CHECK: encoding: [0x62,0xf5,0x7d,0x28,0x3a,0x07]
+          vcvtrops2hf8s xmm0, ymmword ptr [rdi]
+
+// CHECK: vcvtrops2hf8s xmm0, xmmword ptr [rdi]
+// CHECK: encoding: [0x62,0xf5,0x7d,0x08,0x3a,0x07]
+          vcvtrops2hf8s xmm0, xmmword ptr [rdi]
+
+// CHECK: vcvtrops2hf8s xmm0 {k1}, zmm1
+// CHECK: encoding: [0x62,0xf5,0x7d,0x49,0x3a,0xc1]
+          vcvtrops2hf8s xmm0 {k1}, zmm1
+
+// CHECK: vcvtrops2hf8s xmm0 {k1} {z}, zmm1
+// CHECK: encoding: [0x62,0xf5,0x7d,0xc9,0x3a,0xc1]
+          vcvtrops2hf8s xmm0 {k1} {z}, zmm1
+
+// CHECK: vcvtrops2hf8s xmm0, dword ptr [rdi]{1to16}
+// CHECK: encoding: [0x62,0xf5,0x7d,0x58,0x3a,0x07]
+          vcvtrops2hf8s xmm0, dword ptr [rdi]{1to16}
+
+//
+// Group B: Bias PS->8bit conversions (3-operand)
+//
+
+// vcvtbiasps2bf8
+
+// CHECK: vcvtbiasps2bf8 xmm0, zmm1, zmm2
+// CHECK: encoding: [0x62,0xf5,0x74,0x48,0x39,0xc2]
+          vcvtbiasps2bf8 xmm0, zmm1, zmm2
+
+// CHECK: vcvtbiasps2bf8 xmm0, ymm1, ymm2
+// CHECK: encoding: [0x62,0xf5,0x74,0x28,0x39,0xc2]
+          vcvtbiasps2bf8 xmm0, ymm1, ymm2
+
+// CHECK: vcvtbiasps2bf8 xmm0, xmm1, xmm2
+// CHECK: encoding: [0x62,0xf5,0x74,0x08,0x39,0xc2]
+          vcvtbiasps2bf8 xmm0, xmm1, xmm2
+
+// CHECK: vcvtbiasps2bf8 xmm0, zmm1, zmmword ptr [rdi]
+// CHECK: encoding: [0x62,0xf5,0x74,0x48,0x39,0x07]
+          vcvtbiasps2bf8 xmm0, zmm1, zmmword ptr [rdi]
+
+// CHECK: vcvtbiasps2bf8 xmm0, ymm1, ymmword ptr [rdi]
+// CHECK: encoding: [0x62,0xf5,0x74,0x28,0x39,0x07]
+          vcvtbiasps2bf8 xmm0, ymm1, ymmword ptr [rdi]
+
+// CHECK: vcvtbiasps2bf8 xmm0, xmm1, xmmword ptr [rdi]
+// CHECK: encoding: [0x62,0xf5,0x74,0x08,0x39,0x07]
+          vcvtbiasps2bf8 xmm0, xmm1, xmmword ptr [rdi]
+
+// CHECK: vcvtbiasps2bf8 xmm0 {k1}, zmm1, zmm2
+// CHECK: encoding: [0x62,0xf5,0x74,0x49,0x39,0xc2]
+          vcvtbiasps2bf8 xmm0 {k1}, zmm1, zmm2
+
+// CHECK: vcvtbiasps2bf8 xmm0 {k1} {z}, zmm1, zmm2
+// CHECK: encoding: [0x62,0xf5,0x74,0xc9,0x39,0xc2]
+          vcvtbiasps2bf8 xmm0 {k1} {z}, zmm1, zmm2
+
+// vcvtbiasps2bf8s
+
+// CHECK: vcvtbiasps2bf8s xmm0, zmm1, zmm2
+// CHECK: encoding: [0x62,0xf5,0x74,0x48,0x3b,0xc2]
+          vcvtbiasps2bf8s xmm0, zmm1, zmm2
+
+// CHECK: vcvtbiasps2bf8s xmm0, ymm1, ymm2
+// CHECK: encoding: [0x62,0xf5,0x74,0x28,0x3b,0xc2]
+          vcvtbiasps2bf8s xmm0, ymm1, ymm2
+
+// CHECK: vcvtbiasps2bf8s xmm0, xmm1, xmm2
+// CHECK: encoding: [0x62,0xf5,0x74,0x08,0x3b,0xc2]
+          vcvtbiasps2bf8s xmm0, xmm1, xmm2
+
+// CHECK: vcvtbiasps2bf8s xmm0, zmm1, zmmword ptr [rdi]
+// CHECK: encoding: [0x62,0xf5,0x74,0x48,0x3b,0x07]
+          vcvtbiasps2bf8s xmm0, zmm1, zmmword ptr [rdi]
+
+// CHECK: vcvtbiasps2bf8s xmm0, ymm1, ymmword ptr [rdi]
+// CHECK: encoding: [0x62,0xf5,0x74,0x28,0x3b,0x07]
+          vcvtbiasps2bf8s xmm0, ymm1, ymmword ptr [rdi]
+
+// CHECK: vcvtbiasps2bf8s xmm0, xmm1, xmmword ptr [rdi]
+// CHECK: encoding: [0x62,0xf5,0x74,0x08,0x3b,0x07]
+          vcvtbiasps2bf8s xmm0, xmm1, xmmword ptr [rdi]
+
+// CHECK: vcvtbiasps2bf8s xmm0 {k1}, zmm1, zmm2
+// CHECK: encoding: [0x62,0xf5,0x74,0x49,0x3b,0xc2]
+          vcvtbiasps2bf8s xmm0 {k1}, zmm1, zmm2
+
+// CHECK: vcvtbiasps2bf8s xmm0 {k1} {z}, zmm1, zmm2
+// CHECK: encoding: [0x62,0xf5,0x74,0xc9,0x3b,0xc2]
+          vcvtbiasps2bf8s xmm0 {k1} {z}, zmm1, zmm2
+
+// vcvtbiasps2hf8
+
+// CHECK: vcvtbiasps2hf8 xmm0, zmm1, zmm2
+// CHECK: encoding: [0x62,0xf5,0x74,0x48,0x38,0xc2]
+          vcvtbiasps2hf8 xmm0, zmm1, zmm2
+
+// CHECK: vcvtbiasps2hf8 xmm0, ymm1, ymm2
+// CHECK: encoding: [0x62,0xf5,0x74,0x28,0x38,0xc2]
+          vcvtbiasps2hf8 xmm0, ymm1, ymm2
+
+// CHECK: vcvtbiasps2hf8 xmm0, xmm1, xmm2
+// CHECK: encoding: [0x62,0xf5,0x74,0x08,0x38,0xc2]
+          vcvtbiasps2hf8 xmm0, xmm1, xmm2
+
+// CHECK: vcvtbiasps2hf8 xmm0, zmm1, zmmword ptr [rdi]
+// CHECK: encoding: [0x62,0xf5,0x74,0x48,0x38,0x07]
+          vcvtbiasps2hf8 xmm0, zmm1, zmmword ptr [rdi]
+
+// CHECK: vcvtbiasps2hf8 xmm0, ymm1, ymmword ptr [rdi]
+// CHECK: encoding: [0x62,0xf5,0x74,0x28,0x38,0x07]
+          vcvtbiasps2hf8 xmm0, ymm1, ymmword ptr [rdi]
+
+// CHECK: vcvtbiasps2hf8 xmm0, xmm1, xmmword ptr [rdi]
+// CHECK: encoding: [0x62,0xf5,0x74,0x08,0x38,0x07]
+          vcvtbiasps2hf8 xmm0, xmm1, xmmword ptr [rdi]
+
+// CHECK: vcvtbiasps2hf8 xmm0 {k1}, zmm1, zmm2
+// CHECK: encoding: [0x62,0xf5,0x74,0x49,0x38,0xc2]
+          vcvtbiasps2hf8 xmm0 {k1}, zmm1, zmm2
+
+// CHECK: vcvtbiasps2hf8 xmm0 {k1} {z}, zmm1, zmm2
+// CHECK: encoding: [0x62,0xf5,0x74,0xc9,0x38,0xc2]
+          vcvtbiasps2hf8 xmm0 {k1} {z}, zmm1, zmm2
+
+// vcvtbiasps2hf8s
+
+// CHECK: vcvtbiasps2hf8s xmm0, zmm1, zmm2
+// CHECK: encoding: [0x62,0xf5,0x74,0x48,0x3a,0xc2]
+          vcvtbiasps2hf8s xmm0, zmm1, zmm2
+
+// CHECK: vcvtbiasps2hf8s xmm0, ymm1, ymm2
+// CHECK: encoding: [0x62,0xf5,0x74,0x28,0x3a,0xc2]
+          vcvtbiasps2hf8s xmm0, ymm1, ymm2
+
+// CHECK: vcvtbiasps2hf8s xmm0, xmm1, xmm2
+// CHECK: encoding: [0x62,0xf5,0x74,0x08,0x3a,0xc2]
+          vcvtbiasps2hf8s xmm0, xmm1, xmm2
+
+// CHECK: vcvtbiasps2hf8s xmm0, zmm1, zmmword ptr [rdi]
+// CHECK: encoding: [0x62,0xf5,0x74,0x48,0x3a,0x07]
+          vcvtbiasps2hf8s xmm0, zmm1, zmmword ptr [rdi]
+
+// CHECK: vcvtbiasps2hf8s xmm0, ymm1, ymmword ptr [rdi]
+// CHECK: encoding: [0x62,0xf5,0x74,0x28,0x3a,0x07]
+          vcvtbiasps2hf8s xmm0, ymm1, ymmword ptr [rdi]
+
+// CHECK: vcvtbiasps2hf8s xmm0, xmm1, xmmword ptr [rdi]
+// CHECK: encoding: [0x62,0xf5,0x74,0x08,0x3a,0x07]
+          vcvtbiasps2hf8s xmm0, xmm1, xmmword ptr [rdi]
+
+// CHECK: vcvtbiasps2hf8s xmm0 {k1}, zmm1, zmm2
+// CHECK: encoding: [0x62,0xf5,0x74,0x49,0x3a,0xc2]
+          vcvtbiasps2hf8s xmm0 {k1}, zmm1, zmm2
+
+// CHECK: vcvtbiasps2hf8s xmm0 {k1} {z}, zmm1, zmm2
+// CHECK: encoding: [0x62,0xf5,0x74,0xc9,0x3a,0xc2]
+          vcvtbiasps2hf8s xmm0 {k1} {z}, zmm1, zmm2
+
+//
+// Group C: 8bit->PS expanding conversions
+//
+
+// vcvtbf82ps
+
+// CHECK: vcvtbf82ps zmm0, xmm1
+// CHECK: encoding: [0x62,0xf5,0xfc,0x48,0x36,0xc1]
+          vcvtbf82ps zmm0, xmm1
+
+// CHECK: vcvtbf82ps ymm0, xmm1
+// CHECK: encoding: [0x62,0xf5,0xfc,0x28,0x36,0xc1]
+          vcvtbf82ps ymm0, xmm1
+
+// CHECK: vcvtbf82ps xmm0, xmm1
+// CHECK: encoding: [0x62,0xf5,0xfc,0x08,0x36,0xc1]
+          vcvtbf82ps xmm0, xmm1
+
+// CHECK: vcvtbf82ps zmm0, xmmword ptr [rdi]
+// CHECK: encoding: [0x62,0xf5,0xfc,0x48,0x36,0x07]
+          vcvtbf82ps zmm0, xmmword ptr [rdi]
+
+// CHECK: vcvtbf82ps ymm0, qword ptr [rdi]
+// CHECK: encoding: [0x62,0xf5,0xfc,0x28,0x36,0x07]
+          vcvtbf82ps ymm0, qword ptr [rdi]
+
+// CHECK: vcvtbf82ps xmm0, dword ptr [rdi]
+// CHECK: encoding: [0x62,0xf5,0xfc,0x08,0x36,0x07]
+          vcvtbf82ps xmm0, dword ptr [rdi]
+
+// CHECK: vcvtbf82ps zmm0 {k1}, xmm1
+// CHECK: encoding: [0x62,0xf5,0xfc,0x49,0x36,0xc1]
+          vcvtbf82ps zmm0 {k1}, xmm1
+
+// CHECK: vcvtbf82ps zmm0 {k1} {z}, xmm1
+// CHECK: encoding: [0x62,0xf5,0xfc,0xc9,0x36,0xc1]
+          vcvtbf82ps zmm0 {k1} {z}, xmm1
+
+// vcvthf82ps
+
+// CHECK: vcvthf82ps zmm0, xmm1
+// CHECK: encoding: [0x62,0xf5,0x7c,0x48,0x36,0xc1]
+          vcvthf82ps zmm0, xmm1
+
+// CHECK: vcvthf82ps ymm0, xmm1
+// CHECK: encoding: [0x62,0xf5,0x7c,0x28,0x36,0xc1]
+          vcvthf82ps ymm0, xmm1
+
+// CHECK: vcvthf82ps xmm0, xmm1
+// CHECK: encoding: [0x62,0xf5,0x7c,0x08,0x36,0xc1]
+          vcvthf82ps xmm0, xmm1
+
+// CHECK: vcvthf82ps zmm0, xmmword ptr [rdi]
+// CHECK: encoding: [0x62,0xf5,0x7c,0x48,0x36,0x07]
+          vcvthf82ps zmm0, xmmword ptr [rdi]
+
+// CHECK: vcvthf82ps ymm0, qword ptr [rdi]
+// CHECK: encoding: [0x62,0xf5,0x7c,0x28,0x36,0x07]
+          vcvthf82ps ymm0, qword ptr [rdi]
+
+// CHECK: vcvthf82ps xmm0, dword ptr [rdi]
+// CHECK: encoding: [0x62,0xf5,0x7c,0x08,0x36,0x07]
+          vcvthf82ps xmm0, dword ptr [rdi]
+
+// CHECK: vcvthf82ps zmm0 {k1}, xmm1
+// CHECK: encoding: [0x62,0xf5,0x7c,0x49,0x36,0xc1]
+          vcvthf82ps zmm0 {k1}, xmm1
+
+// CHECK: vcvthf82ps zmm0 {k1} {z}, xmm1
+// CHECK: encoding: [0x62,0xf5,0x7c,0xc9,0x36,0xc1]
+          vcvthf82ps zmm0 {k1} {z}, xmm1
+
+//
+// Group D: BF8/HF8->BF4S store-like truncations
+//
+
+// vcvtbf82bf4s
+
+// CHECK: vcvtbf82bf4s ymm0, zmm1
+// CHECK: encoding: [0x62,0xf5,0xfe,0x48,0x3d,0xc8]
+          vcvtbf82bf4s ymm0, zmm1
+
+// CHECK: vcvtbf82bf4s xmm0, ymm1
+// CHECK: encoding: [0x62,0xf5,0xfe,0x28,0x3d,0xc8]
+          vcvtbf82bf4s xmm0, ymm1
+
+// CHECK: vcvtbf82bf4s xmm0, xmm1
+// CHECK: encoding: [0x62,0xf5,0xfe,0x08,0x3d,0xc8]
+          vcvtbf82bf4s xmm0, xmm1
+
+// CHECK: vcvtbf82bf4s ymmword ptr [rdi], zmm1
+// CHECK: encoding: [0x62,0xf5,0xfe,0x48,0x3d,0x0f]
+          vcvtbf82bf4s ymmword ptr [rdi], zmm1
+
+// CHECK: vcvtbf82bf4s xmmword ptr [rdi], ymm1
+// CHECK: encoding: [0x62,0xf5,0xfe,0x28,0x3d,0x0f]
+          vcvtbf82bf4s xmmword ptr [rdi], ymm1
+
+// CHECK: vcvtbf82bf4s qword ptr [rdi], xmm1
+// CHECK: encoding: [0x62,0xf5,0xfe,0x08,0x3d,0x0f]
+          vcvtbf82bf4s qword ptr [rdi], xmm1
+
+// vcvthf82bf4s
+
+// CHECK: vcvthf82bf4s ymm0, zmm1
+// CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x3d,0xc8]
+          vcvthf82bf4s ymm0, zmm1
+
+// CHECK: vcvthf82bf4s xmm0, ymm1
+// CHECK: encoding: [0x62,0xf5,0x7e,0x28,0x3d,0xc8]
+          vcvthf82bf4s xmm0, ymm1
+
+// CHECK: vcvthf82bf4s xmm0, xmm1
+// CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x3d,0xc8]
+          vcvthf82bf4s xmm0, xmm1
+
+// CHECK: vcvthf82bf4s ymmword ptr [rdi], zmm1
+// CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x3d,0x0f]
+          vcvthf82bf4s ymmword ptr [rdi], zmm1
+
+// CHECK: vcvthf82bf4s xmmword ptr [rdi], ymm1
+// CHECK: encoding: [0x62,0xf5,0x7e,0x28,0x3d,0x0f]
+          vcvthf82bf4s xmmword ptr [rdi], ymm1
+
+// CHECK: vcvthf82bf4s qword ptr [rdi], xmm1
+// CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x3d,0x0f]
+          vcvthf82bf4s qword ptr [rdi], xmm1
+
+//
+// Group E: Same-size reg-only conversions (no masking)
+//
+
+// vcvtbf82bf6s
+
+// CHECK: vcvtbf82bf6s zmm0, zmm1
+// CHECK: encoding: [0x62,0xf5,0xfe,0x48,0x3e,0xc1]
+          vcvtbf82bf6s zmm0, zmm1
+
+// CHECK: vcvtbf82bf6s ymm0, ymm1
+// CHECK: encoding: [0x62,0xf5,0xfe,0x28,0x3e,0xc1]
+          vcvtbf82bf6s ymm0, ymm1
+
+// CHECK: vcvtbf82bf6s xmm0, xmm1
+// CHECK: encoding: [0x62,0xf5,0xfe,0x08,0x3e,0xc1]
+          vcvtbf82bf6s xmm0, xmm1
+
+// vcvthf82hf6s
+
+// CHECK: vcvthf82hf6s zmm0, zmm1
+// CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x3c,0xc1]
+          vcvthf82hf6s zmm0, zmm1
+
+// CHECK: vcvthf82hf6s ymm0, ymm1
+// CHECK: encoding: [0x62,0xf5,0x7e,0x28,0x3c,0xc1]
+          vcvthf82hf6s ymm0, ymm1
+
+// CHECK: vcvthf82hf6s xmm0, xmm1
+// CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x3c,0xc1]
+          vcvthf82hf6s xmm0, xmm1
+
+//
+// Group F: Expanding/same-size conversions with masking
+//
+
+// vcvtbf42hf8
+
+// CHECK: vcvtbf42hf8 zmm0, ymm1
+// CHECK: encoding: [0x62,0xf5,0x7c,0x48,0x37,0xc1]
+          vcvtbf42hf8 zmm0, ymm1
+
+// CHECK: vcvtbf42hf8 ymm0, xmm1
+// CHECK: encoding: [0x62,0xf5,0x7c,0x28,0x37,0xc1]
+          vcvtbf42hf8 ymm0, xmm1
+
+// CHECK: vcvtbf42hf8 xmm0, xmm1
+// CHECK: encoding: [0x62,0xf5,0x7c,0x08,0x37,0xc1]
+          vcvtbf42hf8 xmm0, xmm1
+
+// CHECK: vcvtbf42hf8 zmm0, ymmword ptr [rdi]
+// CHECK: encoding: [0x62,0xf5,0x7c,0x48,0x37,0x07]
+          vcvtbf42hf8 zmm0, ymmword ptr [rdi]
+
+// CHECK: vcvtbf42hf8 ymm0, xmmword ptr [rdi]
+// CHECK: encoding: [0x62,0xf5,0x7c,0x28,0x37,0x07]
+          vcvtbf42hf8 ymm0, xmmword ptr [rdi]
+
+// CHECK: vcvtbf42hf8 xmm0, qword ptr [rdi]
+// CHECK: encoding: [0x62,0xf5,0x7c,0x08,0x37,0x07]
+          vcvtbf42hf8 xmm0, qword ptr [rdi]
+
+// CHECK: vcvtbf42hf8 zmm0 {k1}, ymm1
+// CHECK: encoding: [0x62,0xf5,0x7c,0x49,0x37,0xc1]
+          vcvtbf42hf8 zmm0 {k1}, ymm1
+
+// CHECK: vcvtbf42hf8 zmm0 {k1} {z}, ymm1
+// CHECK: encoding: [0x62,0xf5,0x7c,0xc9,0x37,0xc1]
+          vcvtbf42hf8 zmm0 {k1} {z}, ymm1
+
+// vcvtbf62hf8
+
+// CHECK: vcvtbf62hf8 zmm0, zmm1
+// CHECK: encoding: [0x62,0xf5,0xfd,0x48,0x37,0xc1]
+          vcvtbf62hf8 zmm0, zmm1
+
+// CHECK: vcvtbf62hf8 ymm0, ymm1
+// CHECK: encoding: [0x62,0xf5,0xfd,0x28,0x37,0xc1]
+          vcvtbf62hf8 ymm0, ymm1
+
+// CHECK: vcvtbf62hf8 xmm0, xmm1
+// CHECK: encoding: [0x62,0xf5,0xfd,0x08,0x37,0xc1]
+          vcvtbf62hf8 xmm0, xmm1
+
+// CHECK: vcvtbf62hf8 zmm0 {k1}, zmm1
+// CHECK: encoding: [0x62,0xf5,0xfd,0x49,0x37,0xc1]
+          vcvtbf62hf8 zmm0 {k1}, zmm1
+
+// CHECK: vcvtbf62hf8 zmm0 {k1} {z}, zmm1
+// CHECK: encoding: [0x62,0xf5,0xfd,0xc9,0x37,0xc1]
+          vcvtbf62hf8 zmm0 {k1} {z}, zmm1
+
+// vcvthf62hf8
+
+// CHECK: vcvthf62hf8 zmm0, zmm1
+// CHECK: encoding: [0x62,0xf5,0x7d,0x48,0x37,0xc1]
+          vcvthf62hf8 zmm0, zmm1
+
+// CHECK: vcvthf62hf8 ymm0, ymm1
+// CHECK: encoding: [0x62,0xf5,0x7d,0x28,0x37,0xc1]
+          vcvthf62hf8 ymm0, ymm1
+
+// CHECK: vcvthf62hf8 xmm0, xmm1
+// CHECK: encoding: [0x62,0xf5,0x7d,0x08,0x37,0xc1]
+          vcvthf62hf8 xmm0, xmm1
+
+// CHECK: vcvthf62hf8 zmm0 {k1}, zmm1
+// CHECK: encoding: [0x62,0xf5,0x7d,0x49,0x37,0xc1]
+          vcvthf62hf8 zmm0 {k1}, zmm1
+
+// CHECK: vcvthf62hf8 zmm0 {k1} {z}, zmm1
+// CHECK: encoding: [0x62,0xf5,0x7d,0xc9,0x37,0xc1]
+          vcvthf62hf8 zmm0 {k1} {z}, zmm1
+
+//
+// Group G: VPMOVSSDB - Integer DWord->Byte signed saturation
+//
+
+// CHECK: vpmovssdb xmm0, zmm1
+// CHECK: encoding: [0x62,0xf2,0x7e,0x48,0x41,0xc8]
+          vpmovssdb xmm0, zmm1
+
+// CHECK: vpmovssdb xmm0, ymm1
+// CHECK: encoding: [0x62,0xf2,0x7e,0x28,0x41,0xc8]
+          vpmovssdb xmm0, ymm1
+
+// CHECK: vpmovssdb xmm0, xmm1
+// CHECK: encoding: [0x62,0xf2,0x7e,0x08,0x41,0xc8]
+          vpmovssdb xmm0, xmm1
+
+// CHECK: vpmovssdb xmmword ptr [rdi], zmm1
+// CHECK: encoding: [0x62,0xf2,0x7e,0x48,0x41,0x0f]
+          vpmovssdb xmmword ptr [rdi], zmm1
+
+// CHECK: vpmovssdb qword ptr [rdi], ymm1
+// CHECK: encoding: [0x62,0xf2,0x7e,0x28,0x41,0x0f]
+          vpmovssdb qword ptr [rdi], ymm1
+
+// CHECK: vpmovssdb dword ptr [rdi], xmm1
+// CHECK: encoding: [0x62,0xf2,0x7e,0x08,0x41,0x0f]
+          vpmovssdb dword ptr [rdi], xmm1
+
+// CHECK: vpmovssdb xmm0 {k1}, zmm1
+// CHECK: encoding: [0x62,0xf2,0x7e,0x49,0x41,0xc8]
+          vpmovssdb xmm0 {k1}, zmm1
+
+// CHECK: vpmovssdb xmm0 {k1} {z}, zmm1
+// CHECK: encoding: [0x62,0xf2,0x7e,0xc9,0x41,0xc8]
+          vpmovssdb xmm0 {k1} {z}, zmm1
+
+//
+// Group H: VUNPACKB - Byte unpack with immediate
+//
+
+// CHECK: vunpackb zmm0, zmm1, 1
+// CHECK: encoding: [0x62,0xf3,0x7c,0x48,0x3d,0xc1,0x01]
+          vunpackb zmm0, zmm1, 1
+
+// CHECK: vunpackb ymm0, ymm1, 1
+// CHECK: encoding: [0x62,0xf3,0x7c,0x28,0x3d,0xc1,0x01]
+          vunpackb ymm0, ymm1, 1
+
+// CHECK: vunpackb xmm0, xmm1, 1
+// CHECK: encoding: [0x62,0xf3,0x7c,0x08,0x3d,0xc1,0x01]
+          vunpackb xmm0, xmm1, 1
+
+// CHECK: vunpackb zmm0, zmmword ptr [rdi], 1
+// CHECK: encoding: [0x62,0xf3,0x7c,0x48,0x3d,0x07,0x01]
+          vunpackb zmm0, zmmword ptr [rdi], 1
+
+// CHECK: vunpackb ymm0, ymmword ptr [rdi], 1
+// CHECK: encoding: [0x62,0xf3,0x7c,0x28,0x3d,0x07,0x01]
+          vunpackb ymm0, ymmword ptr [rdi], 1
+
+// CHECK: vunpackb xmm0, xmmword ptr [rdi], 1
+// CHECK: encoding: [0x62,0xf3,0x7c,0x08,0x3d,0x07,0x01]
+          vunpackb xmm0, xmmword ptr [rdi], 1
+
+// CHECK: vunpackb zmm0 {k1}, zmm1, 1
+// CHECK: encoding: [0x62,0xf3,0x7c,0x49,0x3d,0xc1,0x01]
+          vunpackb zmm0 {k1}, zmm1, 1
+
+// CHECK: vunpackb zmm0 {k1} {z}, zmm1, 1
+// CHECK: encoding: [0x62,0xf3,0x7c,0xc9,0x3d,0xc1,0x01]
+          vunpackb zmm0 {k1} {z}, zmm1, 1
diff --git a/llvm/test/TableGen/x86-fold-tables.inc b/llvm/test/TableGen/x86-fold-tables.inc
index b99b6ef60eac75..64f7ff432e83aa 100644
--- a/llvm/test/TableGen/x86-fold-tables.inc
+++ b/llvm/test/TableGen/x86-fold-tables.inc
@@ -328,6 +328,9 @@ static const X86FoldTableEntry Table2Addr[] = {
   {X86::SUB8ri_NF, X86::SUB8mi_NF, TB_NO_REVERSE},
   {X86::SUB8rr, X86::SUB8mr, TB_NO_REVERSE},
   {X86::SUB8rr_NF, X86::SUB8mr_NF, TB_NO_REVERSE},
+  {X86::VPMOVSSDZ128rrk, X86::VPMOVSSDZ128mrk, TB_NO_REVERSE},
+  {X86::VPMOVSSDZ256rrk, X86::VPMOVSSDZ256mrk, TB_NO_REVERSE},
+  {X86::VPMOVSSDZrrk, X86::VPMOVSSDZmrk, TB_NO_REVERSE},
   {X86::XOR16ri, X86::XOR16mi, TB_NO_REVERSE},
   {X86::XOR16ri8, X86::XOR16mi8, TB_NO_REVERSE},
   {X86::XOR16ri8_NF, X86::XOR16mi8_NF, TB_NO_REVERSE},
@@ -483,6 +486,10 @@ static const X86FoldTableEntry Table0[] = {
   {X86::TEST64rr, X86::TEST64mr, TB_FOLDED_LOAD},
   {X86::TEST8ri, X86::TEST8mi, TB_FOLDED_LOAD},
   {X86::TEST8rr, X86::TEST8mr, TB_FOLDED_LOAD},
+  {X86::VCVTBF82BF4SZ256rr, X86::VCVTBF82BF4SZ256mr, TB_FOLDED_STORE},
+  {X86::VCVTBF82BF4SZrr, X86::VCVTBF82BF4SZmr, TB_FOLDED_STORE},
+  {X86::VCVTHF82BF4SZ256rr, X86::VCVTHF82BF4SZ256mr, TB_FOLDED_STORE},
+  {X86::VCVTHF82BF4SZrr, X86::VCVTHF82BF4SZmr, TB_FOLDED_STORE},
   {X86::VCVTPS2PHYrr, X86::VCVTPS2PHYmr, TB_FOLDED_STORE},
   {X86::VCVTPS2PHZ256rr, X86::VCVTPS2PHZ256mr, TB_FOLDED_STORE},
   {X86::VCVTPS2PHZrr, X86::VCVTPS2PHZmr, TB_FOLDED_STORE},
@@ -572,6 +579,7 @@ static const X86FoldTableEntry Table0[] = {
   {X86::VPMOVSQDZ256rr, X86::VPMOVSQDZ256mr, TB_FOLDED_STORE},
   {X86::VPMOVSQDZrr, X86::VPMOVSQDZmr, TB_FOLDED_STORE},
   {X86::VPMOVSQWZrr, X86::VPMOVSQWZmr, TB_FOLDED_STORE},
+  {X86::VPMOVSSDZrr, X86::VPMOVSSDZmr, TB_FOLDED_STORE},
   {X86::VPMOVSWBZ256rr, X86::VPMOVSWBZ256mr, TB_FOLDED_STORE},
   {X86::VPMOVSWBZrr, X86::VPMOVSWBZmr, TB_FOLDED_STORE},
   {X86::VPMOVUSDBZrr, X86::VPMOVUSDBZmr, TB_FOLDED_STORE},
@@ -1183,6 +1191,12 @@ static const X86FoldTableEntry Table1[] = {
   {X86::VCVTBF162IUBSZ128rr, X86::VCVTBF162IUBSZ128rm, 0},
   {X86::VCVTBF162IUBSZ256rr, X86::VCVTBF162IUBSZ256rm, 0},
   {X86::VCVTBF162IUBSZrr, X86::VCVTBF162IUBSZrm, 0},
+  {X86::VCVTBF42HF8Z128rr, X86::VCVTBF42HF8Z128rm, TB_NO_REVERSE},
+  {X86::VCVTBF42HF8Z256rr, X86::VCVTBF42HF8Z256rm, 0},
+  {X86::VCVTBF42HF8Zrr, X86::VCVTBF42HF8Zrm, 0},
+  {X86::VCVTBF82PSZ128rr, X86::VCVTBF82PSZ128rm, TB_NO_REVERSE},
+  {X86::VCVTBF82PSZ256rr, X86::VCVTBF82PSZ256rm, TB_NO_REVERSE},
+  {X86::VCVTBF82PSZrr, X86::VCVTBF82PSZrm, 0},
   {X86::VCVTDQ2PDYrr, X86::VCVTDQ2PDYrm, 0},
   {X86::VCVTDQ2PDZ128rr, X86::VCVTDQ2PDZ128rm, TB_NO_REVERSE},
   {X86::VCVTDQ2PDZ256rr, X86::VCVTDQ2PDZ256rm, 0},
@@ -1199,6 +1213,9 @@ static const X86FoldTableEntry Table1[] = {
   {X86::VCVTHF82PHZ128rr, X86::VCVTHF82PHZ128rm, TB_NO_REVERSE},
   {X86::VCVTHF82PHZ256rr, X86::VCVTHF82PHZ256rm, 0},
   {X86::VCVTHF82PHZrr, X86::VCVTHF82PHZrm, 0},
+  {X86::VCVTHF82PSZ128rr, X86::VCVTHF82PSZ128rm, TB_NO_REVERSE},
+  {X86::VCVTHF82PSZ256rr, X86::VCVTHF82PSZ256rm, TB_NO_REVERSE},
+  {X86::VCVTHF82PSZrr, X86::VCVTHF82PSZrm, 0},
   {X86::VCVTNEPS2BF16Yrr, X86::VCVTNEPS2BF16Yrm, 0},
   {X86::VCVTNEPS2BF16Z128rr, X86::VCVTNEPS2BF16Z128rm, 0},
   {X86::VCVTNEPS2BF16Z256rr, X86::VCVTNEPS2BF16Z256rm, 0},
@@ -1273,11 +1290,23 @@ static const X86FoldTableEntry Table1[] = {
   {X86::VCVTPH2WZ128rr, X86::VCVTPH2WZ128rm, 0},
   {X86::VCVTPH2WZ256rr, X86::VCVTPH2WZ256rm, 0},
   {X86::VCVTPH2WZrr, X86::VCVTPH2WZrm, 0},
+  {X86::VCVTPS2BF8SZ128rr, X86::VCVTPS2BF8SZ128rm, 0},
+  {X86::VCVTPS2BF8SZ256rr, X86::VCVTPS2BF8SZ256rm, 0},
+  {X86::VCVTPS2BF8SZrr, X86::VCVTPS2BF8SZrm, 0},
+  {X86::VCVTPS2BF8Z128rr, X86::VCVTPS2BF8Z128rm, 0},
+  {X86::VCVTPS2BF8Z256rr, X86::VCVTPS2BF8Z256rm, 0},
+  {X86::VCVTPS2BF8Zrr, X86::VCVTPS2BF8Zrm, 0},
   {X86::VCVTPS2DQYrr, X86::VCVTPS2DQYrm, 0},
   {X86::VCVTPS2DQZ128rr, X86::VCVTPS2DQZ128rm, 0},
   {X86::VCVTPS2DQZ256rr, X86::VCVTPS2DQZ256rm, 0},
   {X86::VCVTPS2DQZrr, X86::VCVTPS2DQZrm, 0},
   {X86::VCVTPS2DQrr, X86::VCVTPS2DQrm, 0},
+  {X86::VCVTPS2HF8SZ128rr, X86::VCVTPS2HF8SZ128rm, 0},
+  {X86::VCVTPS2HF8SZ256rr, X86::VCVTPS2HF8SZ256rm, 0},
+  {X86::VCVTPS2HF8SZrr, X86::VCVTPS2HF8SZrm, 0},
+  {X86::VCVTPS2HF8Z128rr, X86::VCVTPS2HF8Z128rm, 0},
+  {X86::VCVTPS2HF8Z256rr, X86::VCVTPS2HF8Z256rm, 0},
+  {X86::VCVTPS2HF8Zrr, X86::VCVTPS2HF8Zrm, 0},
   {X86::VCVTPS2IBSZ128rr, X86::VCVTPS2IBSZ128rm, 0},
   {X86::VCVTPS2IBSZ256rr, X86::VCVTPS2IBSZ256rm, 0},
   {X86::VCVTPS2IBSZrr, X86::VCVTPS2IBSZrm, 0},
@@ -1310,6 +1339,12 @@ static const X86FoldTableEntry Table1[] = {
   {X86::VCVTQQ2PSZ128rr, X86::VCVTQQ2PSZ128rm, 0},
   {X86::VCVTQQ2PSZ256rr, X86::VCVTQQ2PSZ256rm, 0},
   {X86::VCVTQQ2PSZrr, X86::VCVTQQ2PSZrm, 0},
+  {X86::VCVTROPS2HF8SZ128rr, X86::VCVTROPS2HF8SZ128rm, 0},
+  {X86::VCVTROPS2HF8SZ256rr, X86::VCVTROPS2HF8SZ256rm, 0},
+  {X86::VCVTROPS2HF8SZrr, X86::VCVTROPS2HF8SZrm, 0},
+  {X86::VCVTROPS2HF8Z128rr, X86::VCVTROPS2HF8Z128rm, 0},
+  {X86::VCVTROPS2HF8Z256rr, X86::VCVTROPS2HF8Z256rm, 0},
+  {X86::VCVTROPS2HF8Zrr, X86::VCVTROPS2HF8Zrm, 0},
   {X86::VCVTSD2SI64Zrr, X86::VCVTSD2SI64Zrm, 0},
   {X86::VCVTSD2SI64Zrr_Int, X86::VCVTSD2SI64Zrm_Int, TB_NO_REVERSE},
   {X86::VCVTSD2SI64rr, X86::VCVTSD2SI64rm, 0},
@@ -1963,6 +1998,9 @@ static const X86FoldTableEntry Table1[] = {
   {X86::VUCOMXSHZrr_Int, X86::VUCOMXSHZrm_Int, TB_NO_REVERSE},
   {X86::VUCOMXSSZrr, X86::VUCOMXSSZrm, 0},
   {X86::VUCOMXSSZrr_Int, X86::VUCOMXSSZrm_Int, TB_NO_REVERSE},
+  {X86::VUNPACKBZ128ri, X86::VUNPACKBZ128mi, 0},
+  {X86::VUNPACKBZ256ri, X86::VUNPACKBZ256mi, 0},
+  {X86::VUNPACKBZri, X86::VUNPACKBZmi, 0},
   {X86::XOR16ri8_ND, X86::XOR16mi8_ND, 0},
   {X86::XOR16ri8_NF_ND, X86::XOR16mi8_NF_ND, 0},
   {X86::XOR16ri_ND, X86::XOR16mi_ND, 0},
@@ -2563,6 +2601,12 @@ static const X86FoldTableEntry Table2[] = {
   {X86::VCVTBF162IUBSZ128rrkz, X86::VCVTBF162IUBSZ128rmkz, 0},
   {X86::VCVTBF162IUBSZ256rrkz, X86::VCVTBF162IUBSZ256rmkz, 0},
   {X86::VCVTBF162IUBSZrrkz, X86::VCVTBF162IUBSZrmkz, 0},
+  {X86::VCVTBF42HF8Z128rrkz, X86::VCVTBF42HF8Z128rmkz, TB_NO_REVERSE},
+  {X86::VCVTBF42HF8Z256rrkz, X86::VCVTBF42HF8Z256rmkz, 0},
+  {X86::VCVTBF42HF8Zrrkz, X86::VCVTBF42HF8Zrmkz, 0},
+  {X86::VCVTBF82PSZ128rrkz, X86::VCVTBF82PSZ128rmkz, TB_NO_REVERSE},
+  {X86::VCVTBF82PSZ256rrkz, X86::VCVTBF82PSZ256rmkz, TB_NO_REVERSE},
+  {X86::VCVTBF82PSZrrkz, X86::VCVTBF82PSZrmkz, 0},
   {X86::VCVTBIASPH2BF8SZ128rr, X86::VCVTBIASPH2BF8SZ128rm, 0},
   {X86::VCVTBIASPH2BF8SZ256rr, X86::VCVTBIASPH2BF8SZ256rm, 0},
   {X86::VCVTBIASPH2BF8SZrr, X86::VCVTBIASPH2BF8SZrm, 0},
@@ -2575,6 +2619,18 @@ static const X86FoldTableEntry Table2[] = {
   {X86::VCVTBIASPH2HF8Z128rr, X86::VCVTBIASPH2HF8Z128rm, 0},
   {X86::VCVTBIASPH2HF8Z256rr, X86::VCVTBIASPH2HF8Z256rm, 0},
   {X86::VCVTBIASPH2HF8Zrr, X86::VCVTBIASPH2HF8Zrm, 0},
+  {X86::VCVTBIASPS2BF8SZ128rr, X86::VCVTBIASPS2BF8SZ128rm, 0},
+  {X86::VCVTBIASPS2BF8SZ256rr, X86::VCVTBIASPS2BF8SZ256rm, 0},
+  {X86::VCVTBIASPS2BF8SZrr, X86::VCVTBIASPS2BF8SZrm, 0},
+  {X86::VCVTBIASPS2BF8Z128rr, X86::VCVTBIASPS2BF8Z128rm, 0},
+  {X86::VCVTBIASPS2BF8Z256rr, X86::VCVTBIASPS2BF8Z256rm, 0},
+  {X86::VCVTBIASPS2BF8Zrr, X86::VCVTBIASPS2BF8Zrm, 0},
+  {X86::VCVTBIASPS2HF8SZ128rr, X86::VCVTBIASPS2HF8SZ128rm, 0},
+  {X86::VCVTBIASPS2HF8SZ256rr, X86::VCVTBIASPS2HF8SZ256rm, 0},
+  {X86::VCVTBIASPS2HF8SZrr, X86::VCVTBIASPS2HF8SZrm, 0},
+  {X86::VCVTBIASPS2HF8Z128rr, X86::VCVTBIASPS2HF8Z128rm, 0},
+  {X86::VCVTBIASPS2HF8Z256rr, X86::VCVTBIASPS2HF8Z256rm, 0},
+  {X86::VCVTBIASPS2HF8Zrr, X86::VCVTBIASPS2HF8Zrm, 0},
   {X86::VCVTDQ2PDZ128rrkz, X86::VCVTDQ2PDZ128rmkz, TB_NO_REVERSE},
   {X86::VCVTDQ2PDZ256rrkz, X86::VCVTDQ2PDZ256rmkz, 0},
   {X86::VCVTDQ2PDZrrkz, X86::VCVTDQ2PDZrmkz, 0},
@@ -2587,6 +2643,9 @@ static const X86FoldTableEntry Table2[] = {
   {X86::VCVTHF82PHZ128rrkz, X86::VCVTHF82PHZ128rmkz, TB_NO_REVERSE},
   {X86::VCVTHF82PHZ256rrkz, X86::VCVTHF82PHZ256rmkz, 0},
   {X86::VCVTHF82PHZrrkz, X86::VCVTHF82PHZrmkz, 0},
+  {X86::VCVTHF82PSZ128rrkz, X86::VCVTHF82PSZ128rmkz, TB_NO_REVERSE},
+  {X86::VCVTHF82PSZ256rrkz, X86::VCVTHF82PSZ256rmkz, TB_NO_REVERSE},
+  {X86::VCVTHF82PSZrrkz, X86::VCVTHF82PSZrmkz, 0},
   {X86::VCVTNE2PS2BF16Z128rr, X86::VCVTNE2PS2BF16Z128rm, 0},
   {X86::VCVTNE2PS2BF16Z256rr, X86::VCVTNE2PS2BF16Z256rm, 0},
   {X86::VCVTNE2PS2BF16Zrr, X86::VCVTNE2PS2BF16Zrm, 0},
@@ -2656,9 +2715,21 @@ static const X86FoldTableEntry Table2[] = {
   {X86::VCVTPH2WZ128rrkz, X86::VCVTPH2WZ128rmkz, 0},
   {X86::VCVTPH2WZ256rrkz, X86::VCVTPH2WZ256rmkz, 0},
   {X86::VCVTPH2WZrrkz, X86::VCVTPH2WZrmkz, 0},
+  {X86::VCVTPS2BF8SZ128rrkz, X86::VCVTPS2BF8SZ128rmkz, 0},
+  {X86::VCVTPS2BF8SZ256rrkz, X86::VCVTPS2BF8SZ256rmkz, 0},
+  {X86::VCVTPS2BF8SZrrkz, X86::VCVTPS2BF8SZrmkz, 0},
+  {X86::VCVTPS2BF8Z128rrkz, X86::VCVTPS2BF8Z128rmkz, 0},
+  {X86::VCVTPS2BF8Z256rrkz, X86::VCVTPS2BF8Z256rmkz, 0},
+  {X86::VCVTPS2BF8Zrrkz, X86::VCVTPS2BF8Zrmkz, 0},
   {X86::VCVTPS2DQZ128rrkz, X86::VCVTPS2DQZ128rmkz, 0},
   {X86::VCVTPS2DQZ256rrkz, X86::VCVTPS2DQZ256rmkz, 0},
   {X86::VCVTPS2DQZrrkz, X86::VCVTPS2DQZrmkz, 0},
+  {X86::VCVTPS2HF8SZ128rrkz, X86::VCVTPS2HF8SZ128rmkz, 0},
+  {X86::VCVTPS2HF8SZ256rrkz, X86::VCVTPS2HF8SZ256rmkz, 0},
+  {X86::VCVTPS2HF8SZrrkz, X86::VCVTPS2HF8SZrmkz, 0},
+  {X86::VCVTPS2HF8Z128rrkz, X86::VCVTPS2HF8Z128rmkz, 0},
+  {X86::VCVTPS2HF8Z256rrkz, X86::VCVTPS2HF8Z256rmkz, 0},
+  {X86::VCVTPS2HF8Zrrkz, X86::VCVTPS2HF8Zrmkz, 0},
   {X86::VCVTPS2IBSZ128rrkz, X86::VCVTPS2IBSZ128rmkz, 0},
   {X86::VCVTPS2IBSZ256rrkz, X86::VCVTPS2IBSZ256rmkz, 0},
   {X86::VCVTPS2IBSZrrkz, X86::VCVTPS2IBSZrmkz, 0},
@@ -2689,6 +2760,12 @@ static const X86FoldTableEntry Table2[] = {
   {X86::VCVTQQ2PSZ128rrkz, X86::VCVTQQ2PSZ128rmkz, 0},
   {X86::VCVTQQ2PSZ256rrkz, X86::VCVTQQ2PSZ256rmkz, 0},
   {X86::VCVTQQ2PSZrrkz, X86::VCVTQQ2PSZrmkz, 0},
+  {X86::VCVTROPS2HF8SZ128rrkz, X86::VCVTROPS2HF8SZ128rmkz, 0},
+  {X86::VCVTROPS2HF8SZ256rrkz, X86::VCVTROPS2HF8SZ256rmkz, 0},
+  {X86::VCVTROPS2HF8SZrrkz, X86::VCVTROPS2HF8SZrmkz, 0},
+  {X86::VCVTROPS2HF8Z128rrkz, X86::VCVTROPS2HF8Z128rmkz, 0},
+  {X86::VCVTROPS2HF8Z256rrkz, X86::VCVTROPS2HF8Z256rmkz, 0},
+  {X86::VCVTROPS2HF8Zrrkz, X86::VCVTROPS2HF8Zrmkz, 0},
   {X86::VCVTSD2SHZrr, X86::VCVTSD2SHZrm, 0},
   {X86::VCVTSD2SHZrr_Int, X86::VCVTSD2SHZrm_Int, TB_NO_REVERSE},
   {X86::VCVTSD2SSZrr, X86::VCVTSD2SSZrm, 0},
@@ -4180,6 +4257,9 @@ static const X86FoldTableEntry Table2[] = {
   {X86::VSUBSSZrr_Int, X86::VSUBSSZrm_Int, TB_NO_REVERSE},
   {X86::VSUBSSrr, X86::VSUBSSrm, 0},
   {X86::VSUBSSrr_Int, X86::VSUBSSrm_Int, TB_NO_REVERSE},
+  {X86::VUNPACKBZ128rikz, X86::VUNPACKBZ128mikz, 0},
+  {X86::VUNPACKBZ256rikz, X86::VUNPACKBZ256mikz, 0},
+  {X86::VUNPACKBZrikz, X86::VUNPACKBZmikz, 0},
   {X86::VUNPCKHPDYrr, X86::VUNPCKHPDYrm, 0},
   {X86::VUNPCKHPDZ128rr, X86::VUNPCKHPDZ128rm, 0},
   {X86::VUNPCKHPDZ256rr, X86::VUNPCKHPDZ256rm, 0},
@@ -4316,6 +4396,12 @@ static const X86FoldTableEntry Table3[] = {
   {X86::VCVTBF162IUBSZ128rrk, X86::VCVTBF162IUBSZ128rmk, 0},
   {X86::VCVTBF162IUBSZ256rrk, X86::VCVTBF162IUBSZ256rmk, 0},
   {X86::VCVTBF162IUBSZrrk, X86::VCVTBF162IUBSZrmk, 0},
+  {X86::VCVTBF42HF8Z128rrk, X86::VCVTBF42HF8Z128rmk, TB_NO_REVERSE},
+  {X86::VCVTBF42HF8Z256rrk, X86::VCVTBF42HF8Z256rmk, 0},
+  {X86::VCVTBF42HF8Zrrk, X86::VCVTBF42HF8Zrmk, 0},
+  {X86::VCVTBF82PSZ128rrk, X86::VCVTBF82PSZ128rmk, TB_NO_REVERSE},
+  {X86::VCVTBF82PSZ256rrk, X86::VCVTBF82PSZ256rmk, TB_NO_REVERSE},
+  {X86::VCVTBF82PSZrrk, X86::VCVTBF82PSZrmk, 0},
   {X86::VCVTBIASPH2BF8SZ128rrkz, X86::VCVTBIASPH2BF8SZ128rmkz, 0},
   {X86::VCVTBIASPH2BF8SZ256rrkz, X86::VCVTBIASPH2BF8SZ256rmkz, 0},
   {X86::VCVTBIASPH2BF8SZrrkz, X86::VCVTBIASPH2BF8SZrmkz, 0},
@@ -4328,6 +4414,18 @@ static const X86FoldTableEntry Table3[] = {
   {X86::VCVTBIASPH2HF8Z128rrkz, X86::VCVTBIASPH2HF8Z128rmkz, 0},
   {X86::VCVTBIASPH2HF8Z256rrkz, X86::VCVTBIASPH2HF8Z256rmkz, 0},
   {X86::VCVTBIASPH2HF8Zrrkz, X86::VCVTBIASPH2HF8Zrmkz, 0},
+  {X86::VCVTBIASPS2BF8SZ128rrkz, X86::VCVTBIASPS2BF8SZ128rmkz, 0},
+  {X86::VCVTBIASPS2BF8SZ256rrkz, X86::VCVTBIASPS2BF8SZ256rmkz, 0},
+  {X86::VCVTBIASPS2BF8SZrrkz, X86::VCVTBIASPS2BF8SZrmkz, 0},
+  {X86::VCVTBIASPS2BF8Z128rrkz, X86::VCVTBIASPS2BF8Z128rmkz, 0},
+  {X86::VCVTBIASPS2BF8Z256rrkz, X86::VCVTBIASPS2BF8Z256rmkz, 0},
+  {X86::VCVTBIASPS2BF8Zrrkz, X86::VCVTBIASPS2BF8Zrmkz, 0},
+  {X86::VCVTBIASPS2HF8SZ128rrkz, X86::VCVTBIASPS2HF8SZ128rmkz, 0},
+  {X86::VCVTBIASPS2HF8SZ256rrkz, X86::VCVTBIASPS2HF8SZ256rmkz, 0},
+  {X86::VCVTBIASPS2HF8SZrrkz, X86::VCVTBIASPS2HF8SZrmkz, 0},
+  {X86::VCVTBIASPS2HF8Z128rrkz, X86::VCVTBIASPS2HF8Z128rmkz, 0},
+  {X86::VCVTBIASPS2HF8Z256rrkz, X86::VCVTBIASPS2HF8Z256rmkz, 0},
+  {X86::VCVTBIASPS2HF8Zrrkz, X86::VCVTBIASPS2HF8Zrmkz, 0},
   {X86::VCVTDQ2PDZ128rrk, X86::VCVTDQ2PDZ128rmk, TB_NO_REVERSE},
   {X86::VCVTDQ2PDZ256rrk, X86::VCVTDQ2PDZ256rmk, 0},
   {X86::VCVTDQ2PDZrrk, X86::VCVTDQ2PDZrmk, 0},
@@ -4340,6 +4438,9 @@ static const X86FoldTableEntry Table3[] = {
   {X86::VCVTHF82PHZ128rrk, X86::VCVTHF82PHZ128rmk, TB_NO_REVERSE},
   {X86::VCVTHF82PHZ256rrk, X86::VCVTHF82PHZ256rmk, 0},
   {X86::VCVTHF82PHZrrk, X86::VCVTHF82PHZrmk, 0},
+  {X86::VCVTHF82PSZ128rrk, X86::VCVTHF82PSZ128rmk, TB_NO_REVERSE},
+  {X86::VCVTHF82PSZ256rrk, X86::VCVTHF82PSZ256rmk, TB_NO_REVERSE},
+  {X86::VCVTHF82PSZrrk, X86::VCVTHF82PSZrmk, 0},
   {X86::VCVTNE2PS2BF16Z128rrkz, X86::VCVTNE2PS2BF16Z128rmkz, 0},
   {X86::VCVTNE2PS2BF16Z256rrkz, X86::VCVTNE2PS2BF16Z256rmkz, 0},
   {X86::VCVTNE2PS2BF16Zrrkz, X86::VCVTNE2PS2BF16Zrmkz, 0},
@@ -4409,9 +4510,21 @@ static const X86FoldTableEntry Table3[] = {
   {X86::VCVTPH2WZ128rrk, X86::VCVTPH2WZ128rmk, 0},
   {X86::VCVTPH2WZ256rrk, X86::VCVTPH2WZ256rmk, 0},
   {X86::VCVTPH2WZrrk, X86::VCVTPH2WZrmk, 0},
+  {X86::VCVTPS2BF8SZ128rrk, X86::VCVTPS2BF8SZ128rmk, 0},
+  {X86::VCVTPS2BF8SZ256rrk, X86::VCVTPS2BF8SZ256rmk, 0},
+  {X86::VCVTPS2BF8SZrrk, X86::VCVTPS2BF8SZrmk, 0},
+  {X86::VCVTPS2BF8Z128rrk, X86::VCVTPS2BF8Z128rmk, 0},
+  {X86::VCVTPS2BF8Z256rrk, X86::VCVTPS2BF8Z256rmk, 0},
+  {X86::VCVTPS2BF8Zrrk, X86::VCVTPS2BF8Zrmk, 0},
   {X86::VCVTPS2DQZ128rrk, X86::VCVTPS2DQZ128rmk, 0},
   {X86::VCVTPS2DQZ256rrk, X86::VCVTPS2DQZ256rmk, 0},
   {X86::VCVTPS2DQZrrk, X86::VCVTPS2DQZrmk, 0},
+  {X86::VCVTPS2HF8SZ128rrk, X86::VCVTPS2HF8SZ128rmk, 0},
+  {X86::VCVTPS2HF8SZ256rrk, X86::VCVTPS2HF8SZ256rmk, 0},
+  {X86::VCVTPS2HF8SZrrk, X86::VCVTPS2HF8SZrmk, 0},
+  {X86::VCVTPS2HF8Z128rrk, X86::VCVTPS2HF8Z128rmk, 0},
+  {X86::VCVTPS2HF8Z256rrk, X86::VCVTPS2HF8Z256rmk, 0},
+  {X86::VCVTPS2HF8Zrrk, X86::VCVTPS2HF8Zrmk, 0},
   {X86::VCVTPS2IBSZ128rrk, X86::VCVTPS2IBSZ128rmk, 0},
   {X86::VCVTPS2IBSZ256rrk, X86::VCVTPS2IBSZ256rmk, 0},
   {X86::VCVTPS2IBSZrrk, X86::VCVTPS2IBSZrmk, 0},
@@ -4442,6 +4555,12 @@ static const X86FoldTableEntry Table3[] = {
   {X86::VCVTQQ2PSZ128rrk, X86::VCVTQQ2PSZ128rmk, 0},
   {X86::VCVTQQ2PSZ256rrk, X86::VCVTQQ2PSZ256rmk, 0},
   {X86::VCVTQQ2PSZrrk, X86::VCVTQQ2PSZrmk, 0},
+  {X86::VCVTROPS2HF8SZ128rrk, X86::VCVTROPS2HF8SZ128rmk, 0},
+  {X86::VCVTROPS2HF8SZ256rrk, X86::VCVTROPS2HF8SZ256rmk, 0},
+  {X86::VCVTROPS2HF8SZrrk, X86::VCVTROPS2HF8SZrmk, 0},
+  {X86::VCVTROPS2HF8Z128rrk, X86::VCVTROPS2HF8Z128rmk, 0},
+  {X86::VCVTROPS2HF8Z256rrk, X86::VCVTROPS2HF8Z256rmk, 0},
+  {X86::VCVTROPS2HF8Zrrk, X86::VCVTROPS2HF8Zrmk, 0},
   {X86::VCVTSD2SHZrrkz_Int, X86::VCVTSD2SHZrmkz_Int, TB_NO_REVERSE},
   {X86::VCVTSD2SSZrrkz_Int, X86::VCVTSD2SSZrmkz_Int, TB_NO_REVERSE},
   {X86::VCVTSH2SDZrrkz_Int, X86::VCVTSH2SDZrmkz_Int, TB_NO_REVERSE},
@@ -6060,6 +6179,9 @@ static const X86FoldTableEntry Table3[] = {
   {X86::VSUBSDZrrkz_Int, X86::VSUBSDZrmkz_Int, TB_NO_REVERSE},
   {X86::VSUBSHZrrkz_Int, X86::VSUBSHZrmkz_Int, TB_NO_REVERSE},
   {X86::VSUBSSZrrkz_Int, X86::VSUBSSZrmkz_Int, TB_NO_REVERSE},
+  {X86::VUNPACKBZ128rik, X86::VUNPACKBZ128mik, 0},
+  {X86::VUNPACKBZ256rik, X86::VUNPACKBZ256mik, 0},
+  {X86::VUNPACKBZrik, X86::VUNPACKBZmik, 0},
   {X86::VUNPCKHPDZ128rrkz, X86::VUNPCKHPDZ128rmkz, 0},
   {X86::VUNPCKHPDZ256rrkz, X86::VUNPCKHPDZ256rmkz, 0},
   {X86::VUNPCKHPDZrrkz, X86::VUNPCKHPDZrmkz, 0},
@@ -6141,6 +6263,18 @@ static const X86FoldTableEntry Table4[] = {
   {X86::VCVTBIASPH2HF8Z128rrk, X86::VCVTBIASPH2HF8Z128rmk, 0},
   {X86::VCVTBIASPH2HF8Z256rrk, X86::VCVTBIASPH2HF8Z256rmk, 0},
   {X86::VCVTBIASPH2HF8Zrrk, X86::VCVTBIASPH2HF8Zrmk, 0},
+  {X86::VCVTBIASPS2BF8SZ128rrk, X86::VCVTBIASPS2BF8SZ128rmk, 0},
+  {X86::VCVTBIASPS2BF8SZ256rrk, X86::VCVTBIASPS2BF8SZ256rmk, 0},
+  {X86::VCVTBIASPS2BF8SZrrk, X86::VCVTBIASPS2BF8SZrmk, 0},
+  {X86::VCVTBIASPS2BF8Z128rrk, X86::VCVTBIASPS2BF8Z128rmk, 0},
+  {X86::VCVTBIASPS2BF8Z256rrk, X86::VCVTBIASPS2BF8Z256rmk, 0},
+  {X86::VCVTBIASPS2BF8Zrrk, X86::VCVTBIASPS2BF8Zrmk, 0},
+  {X86::VCVTBIASPS2HF8SZ128rrk, X86::VCVTBIASPS2HF8SZ128rmk, 0},
+  {X86::VCVTBIASPS2HF8SZ256rrk, X86::VCVTBIASPS2HF8SZ256rmk, 0},
+  {X86::VCVTBIASPS2HF8SZrrk, X86::VCVTBIASPS2HF8SZrmk, 0},
+  {X86::VCVTBIASPS2HF8Z128rrk, X86::VCVTBIASPS2HF8Z128rmk, 0},
+  {X86::VCVTBIASPS2HF8Z256rrk, X86::VCVTBIASPS2HF8Z256rmk, 0},
+  {X86::VCVTBIASPS2HF8Zrrk, X86::VCVTBIASPS2HF8Zrmk, 0},
   {X86::VCVTNE2PS2BF16Z128rrk, X86::VCVTNE2PS2BF16Z128rmk, 0},
   {X86::VCVTNE2PS2BF16Z256rrk, X86::VCVTNE2PS2BF16Z256rmk, 0},
   {X86::VCVTNE2PS2BF16Zrrk, X86::VCVTNE2PS2BF16Zrmk, 0},
@@ -7505,9 +7639,21 @@ static const X86FoldTableEntry BroadcastTable1[] = {
   {X86::VCVTPH2WZ128rr, X86::VCVTPH2WZ128rmb, TB_BCAST_SH},
   {X86::VCVTPH2WZ256rr, X86::VCVTPH2WZ256rmb, TB_BCAST_SH},
   {X86::VCVTPH2WZrr, X86::VCVTPH2WZrmb, TB_BCAST_SH},
+  {X86::VCVTPS2BF8SZ128rr, X86::VCVTPS2BF8SZ128rmb, TB_BCAST_SS},
+  {X86::VCVTPS2BF8SZ256rr, X86::VCVTPS2BF8SZ256rmb, TB_BCAST_SS},
+  {X86::VCVTPS2BF8SZrr, X86::VCVTPS2BF8SZrmb, TB_BCAST_SS},
+  {X86::VCVTPS2BF8Z128rr, X86::VCVTPS2BF8Z128rmb, TB_BCAST_SS},
+  {X86::VCVTPS2BF8Z256rr, X86::VCVTPS2BF8Z256rmb, TB_BCAST_SS},
+  {X86::VCVTPS2BF8Zrr, X86::VCVTPS2BF8Zrmb, TB_BCAST_SS},
   {X86::VCVTPS2DQZ128rr, X86::VCVTPS2DQZ128rmb, TB_BCAST_SS},
   {X86::VCVTPS2DQZ256rr, X86::VCVTPS2DQZ256rmb, TB_BCAST_SS},
   {X86::VCVTPS2DQZrr, X86::VCVTPS2DQZrmb, TB_BCAST_SS},
+  {X86::VCVTPS2HF8SZ128rr, X86::VCVTPS2HF8SZ128rmb, TB_BCAST_SS},
+  {X86::VCVTPS2HF8SZ256rr, X86::VCVTPS2HF8SZ256rmb, TB_BCAST_SS},
+  {X86::VCVTPS2HF8SZrr, X86::VCVTPS2HF8SZrmb, TB_BCAST_SS},
+  {X86::VCVTPS2HF8Z128rr, X86::VCVTPS2HF8Z128rmb, TB_BCAST_SS},
+  {X86::VCVTPS2HF8Z256rr, X86::VCVTPS2HF8Z256rmb, TB_BCAST_SS},
+  {X86::VCVTPS2HF8Zrr, X86::VCVTPS2HF8Zrmb, TB_BCAST_SS},
   {X86::VCVTPS2IBSZ128rr, X86::VCVTPS2IBSZ128rmb, TB_BCAST_SS},
   {X86::VCVTPS2IBSZ256rr, X86::VCVTPS2IBSZ256rmb, TB_BCAST_SS},
   {X86::VCVTPS2IBSZrr, X86::VCVTPS2IBSZrmb, TB_BCAST_SS},
@@ -7538,6 +7684,12 @@ static const X86FoldTableEntry BroadcastTable1[] = {
   {X86::VCVTQQ2PSZ128rr, X86::VCVTQQ2PSZ128rmb, TB_BCAST_Q},
   {X86::VCVTQQ2PSZ256rr, X86::VCVTQQ2PSZ256rmb, TB_BCAST_Q},
   {X86::VCVTQQ2PSZrr, X86::VCVTQQ2PSZrmb, TB_BCAST_Q},
+  {X86::VCVTROPS2HF8SZ128rr, X86::VCVTROPS2HF8SZ128rmb, TB_BCAST_SS},
+  {X86::VCVTROPS2HF8SZ256rr, X86::VCVTROPS2HF8SZ256rmb, TB_BCAST_SS},
+  {X86::VCVTROPS2HF8SZrr, X86::VCVTROPS2HF8SZrmb, TB_BCAST_SS},
+  {X86::VCVTROPS2HF8Z128rr, X86::VCVTROPS2HF8Z128rmb, TB_BCAST_SS},
+  {X86::VCVTROPS2HF8Z256rr, X86::VCVTROPS2HF8Z256rmb, TB_BCAST_SS},
+  {X86::VCVTROPS2HF8Zrr, X86::VCVTROPS2HF8Zrmb, TB_BCAST_SS},
   {X86::VCVTTBF162IBSZ128rr, X86::VCVTTBF162IBSZ128rmb, TB_BCAST_SH},
   {X86::VCVTTBF162IBSZ256rr, X86::VCVTTBF162IBSZ256rmb, TB_BCAST_SH},
   {X86::VCVTTBF162IBSZrr, X86::VCVTTBF162IBSZrmb, TB_BCAST_SH},
@@ -7899,6 +8051,18 @@ static const X86FoldTableEntry BroadcastTable2[] = {
   {X86::VCVTBIASPH2HF8Z128rr, X86::VCVTBIASPH2HF8Z128rmb, TB_BCAST_SH},
   {X86::VCVTBIASPH2HF8Z256rr, X86::VCVTBIASPH2HF8Z256rmb, TB_BCAST_SH},
   {X86::VCVTBIASPH2HF8Zrr, X86::VCVTBIASPH2HF8Zrmb, TB_BCAST_SH},
+  {X86::VCVTBIASPS2BF8SZ128rr, X86::VCVTBIASPS2BF8SZ128rmb, TB_BCAST_SS},
+  {X86::VCVTBIASPS2BF8SZ256rr, X86::VCVTBIASPS2BF8SZ256rmb, TB_BCAST_SS},
+  {X86::VCVTBIASPS2BF8SZrr, X86::VCVTBIASPS2BF8SZrmb, TB_BCAST_SS},
+  {X86::VCVTBIASPS2BF8Z128rr, X86::VCVTBIASPS2BF8Z128rmb, TB_BCAST_SS},
+  {X86::VCVTBIASPS2BF8Z256rr, X86::VCVTBIASPS2BF8Z256rmb, TB_BCAST_SS},
+  {X86::VCVTBIASPS2BF8Zrr, X86::VCVTBIASPS2BF8Zrmb, TB_BCAST_SS},
+  {X86::VCVTBIASPS2HF8SZ128rr, X86::VCVTBIASPS2HF8SZ128rmb, TB_BCAST_SS},
+  {X86::VCVTBIASPS2HF8SZ256rr, X86::VCVTBIASPS2HF8SZ256rmb, TB_BCAST_SS},
+  {X86::VCVTBIASPS2HF8SZrr, X86::VCVTBIASPS2HF8SZrmb, TB_BCAST_SS},
+  {X86::VCVTBIASPS2HF8Z128rr, X86::VCVTBIASPS2HF8Z128rmb, TB_BCAST_SS},
+  {X86::VCVTBIASPS2HF8Z256rr, X86::VCVTBIASPS2HF8Z256rmb, TB_BCAST_SS},
+  {X86::VCVTBIASPS2HF8Zrr, X86::VCVTBIASPS2HF8Zrmb, TB_BCAST_SS},
   {X86::VCVTDQ2PDZ128rrkz, X86::VCVTDQ2PDZ128rmbkz, TB_BCAST_D},
   {X86::VCVTDQ2PDZ256rrkz, X86::VCVTDQ2PDZ256rmbkz, TB_BCAST_D},
   {X86::VCVTDQ2PDZrrkz, X86::VCVTDQ2PDZrmbkz, TB_BCAST_D},
@@ -7974,9 +8138,21 @@ static const X86FoldTableEntry BroadcastTable2[] = {
   {X86::VCVTPH2WZ128rrkz, X86::VCVTPH2WZ128rmbkz, TB_BCAST_SH},
   {X86::VCVTPH2WZ256rrkz, X86::VCVTPH2WZ256rmbkz, TB_BCAST_SH},
   {X86::VCVTPH2WZrrkz, X86::VCVTPH2WZrmbkz, TB_BCAST_SH},
+  {X86::VCVTPS2BF8SZ128rrkz, X86::VCVTPS2BF8SZ128rmbkz, TB_BCAST_SS},
+  {X86::VCVTPS2BF8SZ256rrkz, X86::VCVTPS2BF8SZ256rmbkz, TB_BCAST_SS},
+  {X86::VCVTPS2BF8SZrrkz, X86::VCVTPS2BF8SZrmbkz, TB_BCAST_SS},
+  {X86::VCVTPS2BF8Z128rrkz, X86::VCVTPS2BF8Z128rmbkz, TB_BCAST_SS},
+  {X86::VCVTPS2BF8Z256rrkz, X86::VCVTPS2BF8Z256rmbkz, TB_BCAST_SS},
+  {X86::VCVTPS2BF8Zrrkz, X86::VCVTPS2BF8Zrmbkz, TB_BCAST_SS},
   {X86::VCVTPS2DQZ128rrkz, X86::VCVTPS2DQZ128rmbkz, TB_BCAST_SS},
   {X86::VCVTPS2DQZ256rrkz, X86::VCVTPS2DQZ256rmbkz, TB_BCAST_SS},
   {X86::VCVTPS2DQZrrkz, X86::VCVTPS2DQZrmbkz, TB_BCAST_SS},
+  {X86::VCVTPS2HF8SZ128rrkz, X86::VCVTPS2HF8SZ128rmbkz, TB_BCAST_SS},
+  {X86::VCVTPS2HF8SZ256rrkz, X86::VCVTPS2HF8SZ256rmbkz, TB_BCAST_SS},
+  {X86::VCVTPS2HF8SZrrkz, X86::VCVTPS2HF8SZrmbkz, TB_BCAST_SS},
+  {X86::VCVTPS2HF8Z128rrkz, X86::VCVTPS2HF8Z128rmbkz, TB_BCAST_SS},
+  {X86::VCVTPS2HF8Z256rrkz, X86::VCVTPS2HF8Z256rmbkz, TB_BCAST_SS},
+  {X86::VCVTPS2HF8Zrrkz, X86::VCVTPS2HF8Zrmbkz, TB_BCAST_SS},
   {X86::VCVTPS2IBSZ128rrkz, X86::VCVTPS2IBSZ128rmbkz, TB_BCAST_SS},
   {X86::VCVTPS2IBSZ256rrkz, X86::VCVTPS2IBSZ256rmbkz, TB_BCAST_SS},
   {X86::VCVTPS2IBSZrrkz, X86::VCVTPS2IBSZrmbkz, TB_BCAST_SS},
@@ -8007,6 +8183,12 @@ static const X86FoldTableEntry BroadcastTable2[] = {
   {X86::VCVTQQ2PSZ128rrkz, X86::VCVTQQ2PSZ128rmbkz, TB_BCAST_Q},
   {X86::VCVTQQ2PSZ256rrkz, X86::VCVTQQ2PSZ256rmbkz, TB_BCAST_Q},
   {X86::VCVTQQ2PSZrrkz, X86::VCVTQQ2PSZrmbkz, TB_BCAST_Q},
+  {X86::VCVTROPS2HF8SZ128rrkz, X86::VCVTROPS2HF8SZ128rmbkz, TB_BCAST_SS},
+  {X86::VCVTROPS2HF8SZ256rrkz, X86::VCVTROPS2HF8SZ256rmbkz, TB_BCAST_SS},
+  {X86::VCVTROPS2HF8SZrrkz, X86::VCVTROPS2HF8SZrmbkz, TB_BCAST_SS},
+  {X86::VCVTROPS2HF8Z128rrkz, X86::VCVTROPS2HF8Z128rmbkz, TB_BCAST_SS},
+  {X86::VCVTROPS2HF8Z256rrkz, X86::VCVTROPS2HF8Z256rmbkz, TB_BCAST_SS},
+  {X86::VCVTROPS2HF8Zrrkz, X86::VCVTROPS2HF8Zrmbkz, TB_BCAST_SS},
   {X86::VCVTTBF162IBSZ128rrkz, X86::VCVTTBF162IBSZ128rmbkz, TB_BCAST_SH},
   {X86::VCVTTBF162IBSZ256rrkz, X86::VCVTTBF162IBSZ256rmbkz, TB_BCAST_SH},
   {X86::VCVTTBF162IBSZrrkz, X86::VCVTTBF162IBSZrmbkz, TB_BCAST_SH},
@@ -8723,6 +8905,18 @@ static const X86FoldTableEntry BroadcastTable3[] = {
   {X86::VCVTBIASPH2HF8Z128rrkz, X86::VCVTBIASPH2HF8Z128rmbkz, TB_BCAST_SH},
   {X86::VCVTBIASPH2HF8Z256rrkz, X86::VCVTBIASPH2HF8Z256rmbkz, TB_BCAST_SH},
   {X86::VCVTBIASPH2HF8Zrrkz, X86::VCVTBIASPH2HF8Zrmbkz, TB_BCAST_SH},
+  {X86::VCVTBIASPS2BF8SZ128rrkz, X86::VCVTBIASPS2BF8SZ128rmbkz, TB_BCAST_SS},
+  {X86::VCVTBIASPS2BF8SZ256rrkz, X86::VCVTBIASPS2BF8SZ256rmbkz, TB_BCAST_SS},
+  {X86::VCVTBIASPS2BF8SZrrkz, X86::VCVTBIASPS2BF8SZrmbkz, TB_BCAST_SS},
+  {X86::VCVTBIASPS2BF8Z128rrkz, X86::VCVTBIASPS2BF8Z128rmbkz, TB_BCAST_SS},
+  {X86::VCVTBIASPS2BF8Z256rrkz, X86::VCVTBIASPS2BF8Z256rmbkz, TB_BCAST_SS},
+  {X86::VCVTBIASPS2BF8Zrrkz, X86::VCVTBIASPS2BF8Zrmbkz, TB_BCAST_SS},
+  {X86::VCVTBIASPS2HF8SZ128rrkz, X86::VCVTBIASPS2HF8SZ128rmbkz, TB_BCAST_SS},
+  {X86::VCVTBIASPS2HF8SZ256rrkz, X86::VCVTBIASPS2HF8SZ256rmbkz, TB_BCAST_SS},
+  {X86::VCVTBIASPS2HF8SZrrkz, X86::VCVTBIASPS2HF8SZrmbkz, TB_BCAST_SS},
+  {X86::VCVTBIASPS2HF8Z128rrkz, X86::VCVTBIASPS2HF8Z128rmbkz, TB_BCAST_SS},
+  {X86::VCVTBIASPS2HF8Z256rrkz, X86::VCVTBIASPS2HF8Z256rmbkz, TB_BCAST_SS},
+  {X86::VCVTBIASPS2HF8Zrrkz, X86::VCVTBIASPS2HF8Zrmbkz, TB_BCAST_SS},
   {X86::VCVTDQ2PDZ128rrk, X86::VCVTDQ2PDZ128rmbk, TB_BCAST_D},
   {X86::VCVTDQ2PDZ256rrk, X86::VCVTDQ2PDZ256rmbk, TB_BCAST_D},
   {X86::VCVTDQ2PDZrrk, X86::VCVTDQ2PDZrmbk, TB_BCAST_D},
@@ -8798,9 +8992,21 @@ static const X86FoldTableEntry BroadcastTable3[] = {
   {X86::VCVTPH2WZ128rrk, X86::VCVTPH2WZ128rmbk, TB_BCAST_SH},
   {X86::VCVTPH2WZ256rrk, X86::VCVTPH2WZ256rmbk, TB_BCAST_SH},
   {X86::VCVTPH2WZrrk, X86::VCVTPH2WZrmbk, TB_BCAST_SH},
+  {X86::VCVTPS2BF8SZ128rrk, X86::VCVTPS2BF8SZ128rmbk, TB_BCAST_SS},
+  {X86::VCVTPS2BF8SZ256rrk, X86::VCVTPS2BF8SZ256rmbk, TB_BCAST_SS},
+  {X86::VCVTPS2BF8SZrrk, X86::VCVTPS2BF8SZrmbk, TB_BCAST_SS},
+  {X86::VCVTPS2BF8Z128rrk, X86::VCVTPS2BF8Z128rmbk, TB_BCAST_SS},
+  {X86::VCVTPS2BF8Z256rrk, X86::VCVTPS2BF8Z256rmbk, TB_BCAST_SS},
+  {X86::VCVTPS2BF8Zrrk, X86::VCVTPS2BF8Zrmbk, TB_BCAST_SS},
   {X86::VCVTPS2DQZ128rrk, X86::VCVTPS2DQZ128rmbk, TB_BCAST_SS},
   {X86::VCVTPS2DQZ256rrk, X86::VCVTPS2DQZ256rmbk, TB_BCAST_SS},
   {X86::VCVTPS2DQZrrk, X86::VCVTPS2DQZrmbk, TB_BCAST_SS},
+  {X86::VCVTPS2HF8SZ128rrk, X86::VCVTPS2HF8SZ128rmbk, TB_BCAST_SS},
+  {X86::VCVTPS2HF8SZ256rrk, X86::VCVTPS2HF8SZ256rmbk, TB_BCAST_SS},
+  {X86::VCVTPS2HF8SZrrk, X86::VCVTPS2HF8SZrmbk, TB_BCAST_SS},
+  {X86::VCVTPS2HF8Z128rrk, X86::VCVTPS2HF8Z128rmbk, TB_BCAST_SS},
+  {X86::VCVTPS2HF8Z256rrk, X86::VCVTPS2HF8Z256rmbk, TB_BCAST_SS},
+  {X86::VCVTPS2HF8Zrrk, X86::VCVTPS2HF8Zrmbk, TB_BCAST_SS},
   {X86::VCVTPS2IBSZ128rrk, X86::VCVTPS2IBSZ128rmbk, TB_BCAST_SS},
   {X86::VCVTPS2IBSZ256rrk, X86::VCVTPS2IBSZ256rmbk, TB_BCAST_SS},
   {X86::VCVTPS2IBSZrrk, X86::VCVTPS2IBSZrmbk, TB_BCAST_SS},
@@ -8831,6 +9037,12 @@ static const X86FoldTableEntry BroadcastTable3[] = {
   {X86::VCVTQQ2PSZ128rrk, X86::VCVTQQ2PSZ128rmbk, TB_BCAST_Q},
   {X86::VCVTQQ2PSZ256rrk, X86::VCVTQQ2PSZ256rmbk, TB_BCAST_Q},
   {X86::VCVTQQ2PSZrrk, X86::VCVTQQ2PSZrmbk, TB_BCAST_Q},
+  {X86::VCVTROPS2HF8SZ128rrk, X86::VCVTROPS2HF8SZ128rmbk, TB_BCAST_SS},
+  {X86::VCVTROPS2HF8SZ256rrk, X86::VCVTROPS2HF8SZ256rmbk, TB_BCAST_SS},
+  {X86::VCVTROPS2HF8SZrrk, X86::VCVTROPS2HF8SZrmbk, TB_BCAST_SS},
+  {X86::VCVTROPS2HF8Z128rrk, X86::VCVTROPS2HF8Z128rmbk, TB_BCAST_SS},
+  {X86::VCVTROPS2HF8Z256rrk, X86::VCVTROPS2HF8Z256rmbk, TB_BCAST_SS},
+  {X86::VCVTROPS2HF8Zrrk, X86::VCVTROPS2HF8Zrmbk, TB_BCAST_SS},
   {X86::VCVTTBF162IBSZ128rrk, X86::VCVTTBF162IBSZ128rmbk, TB_BCAST_SH},
   {X86::VCVTTBF162IBSZ256rrk, X86::VCVTTBF162IBSZ256rmbk, TB_BCAST_SH},
   {X86::VCVTTBF162IBSZrrk, X86::VCVTTBF162IBSZrmbk, TB_BCAST_SH},
@@ -9817,6 +10029,18 @@ static const X86FoldTableEntry BroadcastTable4[] = {
   {X86::VCVTBIASPH2HF8Z128rrk, X86::VCVTBIASPH2HF8Z128rmbk, TB_BCAST_SH},
   {X86::VCVTBIASPH2HF8Z256rrk, X86::VCVTBIASPH2HF8Z256rmbk, TB_BCAST_SH},
   {X86::VCVTBIASPH2HF8Zrrk, X86::VCVTBIASPH2HF8Zrmbk, TB_BCAST_SH},
+  {X86::VCVTBIASPS2BF8SZ128rrk, X86::VCVTBIASPS2BF8SZ128rmbk, TB_BCAST_SS},
+  {X86::VCVTBIASPS2BF8SZ256rrk, X86::VCVTBIASPS2BF8SZ256rmbk, TB_BCAST_SS},
+  {X86::VCVTBIASPS2BF8SZrrk, X86::VCVTBIASPS2BF8SZrmbk, TB_BCAST_SS},
+  {X86::VCVTBIASPS2BF8Z128rrk, X86::VCVTBIASPS2BF8Z128rmbk, TB_BCAST_SS},
+  {X86::VCVTBIASPS2BF8Z256rrk, X86::VCVTBIASPS2BF8Z256rmbk, TB_BCAST_SS},
+  {X86::VCVTBIASPS2BF8Zrrk, X86::VCVTBIASPS2BF8Zrmbk, TB_BCAST_SS},
+  {X86::VCVTBIASPS2HF8SZ128rrk, X86::VCVTBIASPS2HF8SZ128rmbk, TB_BCAST_SS},
+  {X86::VCVTBIASPS2HF8SZ256rrk, X86::VCVTBIASPS2HF8SZ256rmbk, TB_BCAST_SS},
+  {X86::VCVTBIASPS2HF8SZrrk, X86::VCVTBIASPS2HF8SZrmbk, TB_BCAST_SS},
+  {X86::VCVTBIASPS2HF8Z128rrk, X86::VCVTBIASPS2HF8Z128rmbk, TB_BCAST_SS},
+  {X86::VCVTBIASPS2HF8Z256rrk, X86::VCVTBIASPS2HF8Z256rmbk, TB_BCAST_SS},
+  {X86::VCVTBIASPS2HF8Zrrk, X86::VCVTBIASPS2HF8Zrmbk, TB_BCAST_SS},
   {X86::VCVTNE2PS2BF16Z128rrk, X86::VCVTNE2PS2BF16Z128rmbk, TB_BCAST_SS},
   {X86::VCVTNE2PS2BF16Z256rrk, X86::VCVTNE2PS2BF16Z256rmbk, TB_BCAST_SS},
   {X86::VCVTNE2PS2BF16Zrrk, X86::VCVTNE2PS2BF16Zrmbk, TB_BCAST_SS},

>From d7fdb50ae4f2b50947dee5cf33bbc6722843a97f Mon Sep 17 00:00:00 2001
From: Ganesh Gopalasubramanian <Ganesh.Gopalasubramanian at amd.com>
Date: Wed, 15 Jul 2026 12:51:35 +0530
Subject: [PATCH 02/16] Address upstream review comments for AVX10.2 V2AUX
 support

- Split header file into non-512 and 512 versions
- Change option name to 'avx10v2aux'
- Remove mask variants of builtins and use select with basic intrinsic instead
- Fix tests for non-mask version after removal of builtins
- Add args to tests
- Fix Predicates and alignment comments
- Remove unnecessary comment from test file
- Merge multiclass pattern and instantiations
- Fix operand suffix and change convert to cvt
- Use compact notation
- Fix MOVSSDB record, fix SDNode for X86vmtruncss
- Add MOVSSDB intrinsic
- Add missing intrinsics
- Add semantic check and test for vunpackb
- Handle arguments in tests
- Check selects and arguments in tests
- Add store form of MOVSSDB
- Use avx512_trunc_db for vpmovssdb
- Fix naming convention widen/narrowing
---
 clang/include/clang/Basic/BuiltinsX86.td      |  257 +-
 clang/lib/Basic/Targets/X86.cpp               |    6 +-
 clang/lib/Headers/CMakeLists.txt              |    1 +
 clang/lib/Headers/avx10_2_512v2auxintrin.h    | 1183 +++++++
 clang/lib/Headers/avx10_2_v2auxintrin.h       | 2802 ++++++++++++-----
 clang/lib/Headers/immintrin.h                 |    3 +-
 clang/lib/Sema/SemaX86.cpp                    |    7 +
 .../test/CodeGen/X86/avx10_2_v2aux-builtins.c |  530 ++--
 clang/test/CodeGen/attr-target-x86.c          |    9 +-
 clang/test/Sema/builtins-x86.c                |   12 +
 llvm/include/llvm/IR/IntrinsicsX86.td         |  257 +-
 .../llvm/TargetParser/X86TargetParser.def     |    4 +-
 llvm/lib/Target/X86/X86.td                    |    2 +-
 llvm/lib/Target/X86/X86ISelLowering.cpp       |   14 +
 llvm/lib/Target/X86/X86InstrAVX10_V2_AUX.td   |  855 +++--
 llvm/lib/Target/X86/X86InstrFragmentsSIMD.td  |   25 +
 llvm/lib/Target/X86/X86IntrinsicsInfo.h       |  157 +-
 .../CodeGen/X86/avx10_2_v2aux-intrinsics.ll   | 1931 ------------
 .../CodeGen/X86/avx10_v2aux-intrinsics.ll     | 1773 +++++++++++
 .../MC/Disassembler/X86/avx10_v2_aux-32.txt   |   85 -
 .../MC/Disassembler/X86/avx10_v2_aux-64.txt   |   85 -
 llvm/test/MC/X86/avx10_v2_aux-att-32.s        |  226 +-
 llvm/test/MC/X86/avx10_v2_aux-att-64.s        |    2 +-
 llvm/test/MC/X86/avx10_v2_aux-intel-32.s      |  358 ++-
 llvm/test/MC/X86/avx10_v2_aux-intel-64.s      |    2 +-
 llvm/test/TableGen/x86-fold-tables.inc        |   26 +-
 26 files changed, 6835 insertions(+), 3777 deletions(-)
 create mode 100644 clang/lib/Headers/avx10_2_512v2auxintrin.h
 delete mode 100644 llvm/test/CodeGen/X86/avx10_2_v2aux-intrinsics.ll
 create mode 100644 llvm/test/CodeGen/X86/avx10_v2aux-intrinsics.ll

diff --git a/clang/include/clang/Basic/BuiltinsX86.td b/clang/include/clang/Basic/BuiltinsX86.td
index 82fd8d80cc82dc..5bd7909cad3c6c 100644
--- a/clang/include/clang/Basic/BuiltinsX86.td
+++ b/clang/include/clang/Basic/BuiltinsX86.td
@@ -2625,6 +2625,27 @@ let Features = "avx512f", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
   def pmovusdb512mem_mask : X86Builtin<"void(_Vector<16, char *>, _Vector<16, int>, unsigned short)">;
 }
 
+// VPMOVSSDB - Symmetric signed saturation DWord to Byte
+let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<128>] in {
+  def pmovssdb128_mask : X86Builtin<"_Vector<16, char>(_Vector<4, int>, _Vector<16, char>, unsigned char)">;
+}
+let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<256>] in {
+  def pmovssdb256_mask : X86Builtin<"_Vector<16, char>(_Vector<8, int>, _Vector<16, char>, unsigned char)">;
+}
+let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<512>] in {
+  def pmovssdb512_mask : X86Builtin<"_Vector<16, char>(_Vector<16, int>, _Vector<16, char>, unsigned short)">;
+}
+// VPMOVSSDB memory store
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
+  def pmovssdb128mem_mask : X86Builtin<"void(_Vector<16, char *>, _Vector<4, int>, unsigned char)">;
+}
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
+  def pmovssdb256mem_mask : X86Builtin<"void(_Vector<16, char *>, _Vector<8, int>, unsigned char)">;
+}
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
+  def pmovssdb512mem_mask : X86Builtin<"void(_Vector<16, char *>, _Vector<16, int>, unsigned short)">;
+}
+
 let Features = "avx512bw", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
   def pmovuswb512mem_mask : X86Builtin<"void(_Vector<32, char *>, _Vector<32, short>, unsigned int)">;
 }
@@ -5051,244 +5072,298 @@ let Features = "avx10.2", Attributes = [NoThrow, Const, RequiredVectorWidth<512>
 // Group A: PS(f32) -> i8 truncating conversions (quarter-size: output always v16i8)
 
 // VCVTPS2BF8
-let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
-  def vcvtps2bf8_128_mask : X86Builtin<"_Vector<16, char>(_Vector<4, float>, _Vector<16, char>, unsigned char)">;
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
+  def vcvtps2bf8_128 : X86Builtin<"_Vector<16, char>(_Vector<4, float>)">;
 }
 
-let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
-  def vcvtps2bf8_256_mask : X86Builtin<"_Vector<16, char>(_Vector<8, float>, _Vector<16, char>, unsigned char)">;
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
+  def vcvtps2bf8_256 : X86Builtin<"_Vector<16, char>(_Vector<8, float>)">;
 }
 
-let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
-  def vcvtps2bf8_512_mask : X86Builtin<"_Vector<16, char>(_Vector<16, float>, _Vector<16, char>, unsigned short)">;
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
+  def vcvtps2bf8_512 : X86Builtin<"_Vector<16, char>(_Vector<16, float>)">;
 }
 
 // VCVTPS2BF8S
-let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
-  def vcvtps2bf8s_128_mask : X86Builtin<"_Vector<16, char>(_Vector<4, float>, _Vector<16, char>, unsigned char)">;
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
+  def vcvtps2bf8s_128 : X86Builtin<"_Vector<16, char>(_Vector<4, float>)">;
 }
 
-let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
-  def vcvtps2bf8s_256_mask : X86Builtin<"_Vector<16, char>(_Vector<8, float>, _Vector<16, char>, unsigned char)">;
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
+  def vcvtps2bf8s_256 : X86Builtin<"_Vector<16, char>(_Vector<8, float>)">;
 }
 
-let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
-  def vcvtps2bf8s_512_mask : X86Builtin<"_Vector<16, char>(_Vector<16, float>, _Vector<16, char>, unsigned short)">;
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
+  def vcvtps2bf8s_512 : X86Builtin<"_Vector<16, char>(_Vector<16, float>)">;
 }
 
 // VCVTPS2HF8
-let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
-  def vcvtps2hf8_128_mask : X86Builtin<"_Vector<16, char>(_Vector<4, float>, _Vector<16, char>, unsigned char)">;
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
+  def vcvtps2hf8_128 : X86Builtin<"_Vector<16, char>(_Vector<4, float>)">;
 }
 
-let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
-  def vcvtps2hf8_256_mask : X86Builtin<"_Vector<16, char>(_Vector<8, float>, _Vector<16, char>, unsigned char)">;
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
+  def vcvtps2hf8_256 : X86Builtin<"_Vector<16, char>(_Vector<8, float>)">;
 }
 
-let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
-  def vcvtps2hf8_512_mask : X86Builtin<"_Vector<16, char>(_Vector<16, float>, _Vector<16, char>, unsigned short)">;
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
+  def vcvtps2hf8_512 : X86Builtin<"_Vector<16, char>(_Vector<16, float>)">;
 }
 
 // VCVTPS2HF8S
-let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
-  def vcvtps2hf8s_128_mask : X86Builtin<"_Vector<16, char>(_Vector<4, float>, _Vector<16, char>, unsigned char)">;
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
+  def vcvtps2hf8s_128 : X86Builtin<"_Vector<16, char>(_Vector<4, float>)">;
 }
 
-let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
-  def vcvtps2hf8s_256_mask : X86Builtin<"_Vector<16, char>(_Vector<8, float>, _Vector<16, char>, unsigned char)">;
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
+  def vcvtps2hf8s_256 : X86Builtin<"_Vector<16, char>(_Vector<8, float>)">;
 }
 
-let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
-  def vcvtps2hf8s_512_mask : X86Builtin<"_Vector<16, char>(_Vector<16, float>, _Vector<16, char>, unsigned short)">;
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
+  def vcvtps2hf8s_512 : X86Builtin<"_Vector<16, char>(_Vector<16, float>)">;
 }
 
 // VCVTROPS2HF8
-let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
-  def vcvtrops2hf8_128_mask : X86Builtin<"_Vector<16, char>(_Vector<4, float>, _Vector<16, char>, unsigned char)">;
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
+  def vcvtrops2hf8_128 : X86Builtin<"_Vector<16, char>(_Vector<4, float>)">;
 }
 
-let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
-  def vcvtrops2hf8_256_mask : X86Builtin<"_Vector<16, char>(_Vector<8, float>, _Vector<16, char>, unsigned char)">;
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
+  def vcvtrops2hf8_256 : X86Builtin<"_Vector<16, char>(_Vector<8, float>)">;
 }
 
-let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
-  def vcvtrops2hf8_512_mask : X86Builtin<"_Vector<16, char>(_Vector<16, float>, _Vector<16, char>, unsigned short)">;
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
+  def vcvtrops2hf8_512 : X86Builtin<"_Vector<16, char>(_Vector<16, float>)">;
 }
 
 // VCVTROPS2HF8S
-let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
-  def vcvtrops2hf8s_128_mask : X86Builtin<"_Vector<16, char>(_Vector<4, float>, _Vector<16, char>, unsigned char)">;
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
+  def vcvtrops2hf8s_128 : X86Builtin<"_Vector<16, char>(_Vector<4, float>)">;
 }
 
-let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
-  def vcvtrops2hf8s_256_mask : X86Builtin<"_Vector<16, char>(_Vector<8, float>, _Vector<16, char>, unsigned char)">;
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
+  def vcvtrops2hf8s_256 : X86Builtin<"_Vector<16, char>(_Vector<8, float>)">;
 }
 
-let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
-  def vcvtrops2hf8s_512_mask : X86Builtin<"_Vector<16, char>(_Vector<16, float>, _Vector<16, char>, unsigned short)">;
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
+  def vcvtrops2hf8s_512 : X86Builtin<"_Vector<16, char>(_Vector<16, float>)">;
 }
 
-// Group B: Bias PS -> i8 conversions (3-operand: bias + f32 source -> i8 dest)
+// Group B: Bias PS -> i8 conversions (2-operand: bias + f32 source -> i8 dest)
 
 // VCVTBIASPS2BF8
-let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
-  def vcvtbiasps2bf8_128_mask : X86Builtin<"_Vector<16, char>(_Vector<16, char>, _Vector<4, float>, _Vector<16, char>, unsigned char)">;
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
+  def vcvtbiasps2bf8_128 : X86Builtin<"_Vector<16, char>(_Vector<16, char>, _Vector<4, float>)">;
 }
 
-let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
-  def vcvtbiasps2bf8_256_mask : X86Builtin<"_Vector<16, char>(_Vector<32, char>, _Vector<8, float>, _Vector<16, char>, unsigned char)">;
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
+  def vcvtbiasps2bf8_256 : X86Builtin<"_Vector<16, char>(_Vector<32, char>, _Vector<8, float>)">;
 }
 
-let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
-  def vcvtbiasps2bf8_512_mask : X86Builtin<"_Vector<16, char>(_Vector<64, char>, _Vector<16, float>, _Vector<16, char>, unsigned short)">;
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
+  def vcvtbiasps2bf8_512 : X86Builtin<"_Vector<16, char>(_Vector<64, char>, _Vector<16, float>)">;
 }
 
 // VCVTBIASPS2BF8S
-let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
-  def vcvtbiasps2bf8s_128_mask : X86Builtin<"_Vector<16, char>(_Vector<16, char>, _Vector<4, float>, _Vector<16, char>, unsigned char)">;
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
+  def vcvtbiasps2bf8s_128 : X86Builtin<"_Vector<16, char>(_Vector<16, char>, _Vector<4, float>)">;
 }
 
-let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
-  def vcvtbiasps2bf8s_256_mask : X86Builtin<"_Vector<16, char>(_Vector<32, char>, _Vector<8, float>, _Vector<16, char>, unsigned char)">;
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
+  def vcvtbiasps2bf8s_256 : X86Builtin<"_Vector<16, char>(_Vector<32, char>, _Vector<8, float>)">;
 }
 
-let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
-  def vcvtbiasps2bf8s_512_mask : X86Builtin<"_Vector<16, char>(_Vector<64, char>, _Vector<16, float>, _Vector<16, char>, unsigned short)">;
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
+  def vcvtbiasps2bf8s_512 : X86Builtin<"_Vector<16, char>(_Vector<64, char>, _Vector<16, float>)">;
 }
 
 // VCVTBIASPS2HF8
-let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
-  def vcvtbiasps2hf8_128_mask : X86Builtin<"_Vector<16, char>(_Vector<16, char>, _Vector<4, float>, _Vector<16, char>, unsigned char)">;
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
+  def vcvtbiasps2hf8_128 : X86Builtin<"_Vector<16, char>(_Vector<16, char>, _Vector<4, float>)">;
 }
 
-let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
-  def vcvtbiasps2hf8_256_mask : X86Builtin<"_Vector<16, char>(_Vector<32, char>, _Vector<8, float>, _Vector<16, char>, unsigned char)">;
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
+  def vcvtbiasps2hf8_256 : X86Builtin<"_Vector<16, char>(_Vector<32, char>, _Vector<8, float>)">;
 }
 
-let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
-  def vcvtbiasps2hf8_512_mask : X86Builtin<"_Vector<16, char>(_Vector<64, char>, _Vector<16, float>, _Vector<16, char>, unsigned short)">;
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
+  def vcvtbiasps2hf8_512 : X86Builtin<"_Vector<16, char>(_Vector<64, char>, _Vector<16, float>)">;
 }
 
 // VCVTBIASPS2HF8S
-let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
-  def vcvtbiasps2hf8s_128_mask : X86Builtin<"_Vector<16, char>(_Vector<16, char>, _Vector<4, float>, _Vector<16, char>, unsigned char)">;
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
+  def vcvtbiasps2hf8s_128 : X86Builtin<"_Vector<16, char>(_Vector<16, char>, _Vector<4, float>)">;
 }
 
-let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
-  def vcvtbiasps2hf8s_256_mask : X86Builtin<"_Vector<16, char>(_Vector<32, char>, _Vector<8, float>, _Vector<16, char>, unsigned char)">;
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
+  def vcvtbiasps2hf8s_256 : X86Builtin<"_Vector<16, char>(_Vector<32, char>, _Vector<8, float>)">;
 }
 
-let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
-  def vcvtbiasps2hf8s_512_mask : X86Builtin<"_Vector<16, char>(_Vector<64, char>, _Vector<16, float>, _Vector<16, char>, unsigned short)">;
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
+  def vcvtbiasps2hf8s_512 : X86Builtin<"_Vector<16, char>(_Vector<64, char>, _Vector<16, float>)">;
 }
 
 // Group C: 8bit -> PS expanding conversions
 
 // VCVTBF82PS
-let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
-  def vcvtbf8_2ps128_mask : X86Builtin<"_Vector<4, float>(_Vector<16, char>, _Vector<4, float>, unsigned char)">;
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
+  def vcvtbf8_2ps128 : X86Builtin<"_Vector<4, float>(_Vector<16, char>)">;
 }
 
-let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
-  def vcvtbf8_2ps256_mask : X86Builtin<"_Vector<8, float>(_Vector<16, char>, _Vector<8, float>, unsigned char)">;
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
+  def vcvtbf8_2ps256 : X86Builtin<"_Vector<8, float>(_Vector<16, char>)">;
 }
 
-let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
-  def vcvtbf8_2ps512_mask : X86Builtin<"_Vector<16, float>(_Vector<16, char>, _Vector<16, float>, unsigned short)">;
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
+  def vcvtbf8_2ps512 : X86Builtin<"_Vector<16, float>(_Vector<16, char>)">;
 }
 
 // VCVTHF82PS
-let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
-  def vcvthf8_2ps128_mask : X86Builtin<"_Vector<4, float>(_Vector<16, char>, _Vector<4, float>, unsigned char)">;
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
+  def vcvthf8_2ps128 : X86Builtin<"_Vector<4, float>(_Vector<16, char>)">;
 }
 
-let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
-  def vcvthf8_2ps256_mask : X86Builtin<"_Vector<8, float>(_Vector<16, char>, _Vector<8, float>, unsigned char)">;
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
+  def vcvthf8_2ps256 : X86Builtin<"_Vector<8, float>(_Vector<16, char>)">;
 }
 
-let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
-  def vcvthf8_2ps512_mask : X86Builtin<"_Vector<16, float>(_Vector<16, char>, _Vector<16, float>, unsigned short)">;
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
+  def vcvthf8_2ps512 : X86Builtin<"_Vector<16, float>(_Vector<16, char>)">;
 }
 
 // Group E: Same-size reg-only conversions (no masking)
 
 // VCVTBF82BF6S
-let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
   def vcvtbf82bf6s128 : X86Builtin<"_Vector<16, char>(_Vector<16, char>)">;
 }
 
-let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
   def vcvtbf82bf6s256 : X86Builtin<"_Vector<32, char>(_Vector<32, char>)">;
 }
 
-let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
   def vcvtbf82bf6s512 : X86Builtin<"_Vector<64, char>(_Vector<64, char>)">;
 }
 
 // VCVTHF82HF6S
-let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
   def vcvthf82hf6s128 : X86Builtin<"_Vector<16, char>(_Vector<16, char>)">;
 }
 
-let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
   def vcvthf82hf6s256 : X86Builtin<"_Vector<32, char>(_Vector<32, char>)">;
 }
 
-let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
   def vcvthf82hf6s512 : X86Builtin<"_Vector<64, char>(_Vector<64, char>)">;
 }
 
 // Group F: Expanding/same-size conversions (no masking in intrinsic; use selectb for masking)
 
 // VCVTBF42HF8
-let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
   def vcvtbf42hf8128 : X86Builtin<"_Vector<16, char>(_Vector<16, char>)">;
 }
 
-let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
   def vcvtbf42hf8256 : X86Builtin<"_Vector<32, char>(_Vector<16, char>)">;
 }
 
-let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
   def vcvtbf42hf8512 : X86Builtin<"_Vector<64, char>(_Vector<32, char>)">;
 }
 
 // VCVTBF62HF8
-let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
   def vcvtbf62hf8128 : X86Builtin<"_Vector<16, char>(_Vector<16, char>)">;
 }
 
-let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
   def vcvtbf62hf8256 : X86Builtin<"_Vector<32, char>(_Vector<32, char>)">;
 }
 
-let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
   def vcvtbf62hf8512 : X86Builtin<"_Vector<64, char>(_Vector<64, char>)">;
 }
 
 // VCVTHF62HF8
-let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
   def vcvthf62hf8128 : X86Builtin<"_Vector<16, char>(_Vector<16, char>)">;
 }
 
-let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
   def vcvthf62hf8256 : X86Builtin<"_Vector<32, char>(_Vector<32, char>)">;
 }
 
-let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
   def vcvthf62hf8512 : X86Builtin<"_Vector<64, char>(_Vector<64, char>)">;
 }
 
+// Group D: VCVTBF82BF4S / VCVTHF82BF4S - FP8 to FP4 truncating conversions
+
+// VCVTBF82BF4S - FP8 E5M2 to FP4 E2M1 with saturation (output is half size)
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
+  def vcvtbf82bf4s128 : X86Builtin<"_Vector<16, char>(_Vector<16, char>)">;
+}
+
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
+  def vcvtbf82bf4s256 : X86Builtin<"_Vector<16, char>(_Vector<32, char>)">;
+}
+
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
+  def vcvtbf82bf4s512 : X86Builtin<"_Vector<32, char>(_Vector<64, char>)">;
+}
+
+// VCVTBF82BF4S memory store variants (no masking - spec does not support masks)
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
+  def vcvtbf82bf4s128mem : X86Builtin<"void(void *, _Vector<16, char>)">;
+}
+
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
+  def vcvtbf82bf4s256mem : X86Builtin<"void(void *, _Vector<32, char>)">;
+}
+
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
+  def vcvtbf82bf4s512mem : X86Builtin<"void(void *, _Vector<64, char>)">;
+}
+
+// VCVTHF82BF4S - FP8 E4M3 to FP4 E2M1 with saturation (output is half size)
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
+  def vcvthf82bf4s128 : X86Builtin<"_Vector<16, char>(_Vector<16, char>)">;
+}
+
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
+  def vcvthf82bf4s256 : X86Builtin<"_Vector<16, char>(_Vector<32, char>)">;
+}
+
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
+  def vcvthf82bf4s512 : X86Builtin<"_Vector<32, char>(_Vector<64, char>)">;
+}
+
+// VCVTHF82BF4S memory store variants (no masking - spec does not support masks)
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
+  def vcvthf82bf4s128mem : X86Builtin<"void(void *, _Vector<16, char>)">;
+}
+
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
+  def vcvthf82bf4s256mem : X86Builtin<"void(void *, _Vector<32, char>)">;
+}
+
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
+  def vcvthf82bf4s512mem : X86Builtin<"void(void *, _Vector<64, char>)">;
+}
+
 // Group H: VUNPACKB - Byte unpack with immediate (no masking in intrinsic)
 
-let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
   def vunpackb128 : X86Builtin<"_Vector<16, char>(_Vector<16, char>, _Constant unsigned char)">;
 }
 
-let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
   def vunpackb256 : X86Builtin<"_Vector<32, char>(_Vector<32, char>, _Constant unsigned char)">;
 }
 
-let Features = "avx10-v2-aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
   def vunpackb512 : X86Builtin<"_Vector<64, char>(_Vector<64, char>, _Constant unsigned char)">;
 }
diff --git a/clang/lib/Basic/Targets/X86.cpp b/clang/lib/Basic/Targets/X86.cpp
index 9482e179dee5c4..0046eb45cfd8be 100644
--- a/clang/lib/Basic/Targets/X86.cpp
+++ b/clang/lib/Basic/Targets/X86.cpp
@@ -281,7 +281,7 @@ bool X86TargetInfo::handleTargetFeatures(std::vector<std::string> &Features,
     } else if (Feature == "+avx10.2") {
       HasAVX10_2 = true;
       HasFullBFloat16 = true;
-    } else if (Feature == "+avx10-v2-aux") {
+    } else if (Feature == "+avx10v2aux") {
       HasAVX10_V2_AUX = true;
     } else if (Feature == "+avx512cd") {
       HasAVX512CD = true;
@@ -1091,7 +1091,7 @@ bool X86TargetInfo::isValidFeatureName(StringRef Name) const {
       .Case("avx", true)
       .Case("avx10.1", true)
       .Case("avx10.2", true)
-      .Case("avx10-v2-aux", true)
+      .Case("avx10v2aux", true)
       .Case("avx2", true)
       .Case("avx512f", true)
       .Case("avx512cd", true)
@@ -1213,7 +1213,7 @@ bool X86TargetInfo::hasFeature(StringRef Feature) const {
       .Case("avx", SSELevel >= AVX)
       .Case("avx10.1", HasAVX10_1)
       .Case("avx10.2", HasAVX10_2)
-      .Case("avx10-v2-aux", HasAVX10_V2_AUX)
+      .Case("avx10v2aux", HasAVX10_V2_AUX)
       .Case("avx2", SSELevel >= AVX2)
       .Case("avx512f", SSELevel >= AVX512F)
       .Case("avx512cd", HasAVX512CD)
diff --git a/clang/lib/Headers/CMakeLists.txt b/clang/lib/Headers/CMakeLists.txt
index b4891961a45986..73885595e80650 100644
--- a/clang/lib/Headers/CMakeLists.txt
+++ b/clang/lib/Headers/CMakeLists.txt
@@ -183,6 +183,7 @@ set(x86_files
   avx10_2_512niintrin.h
   avx10_2_512satcvtdsintrin.h
   avx10_2_512satcvtintrin.h
+  avx10_2_512v2auxintrin.h
   avx10_2_v2auxintrin.h
   avx10_2bf16intrin.h
   avx10_2convertintrin.h
diff --git a/clang/lib/Headers/avx10_2_512v2auxintrin.h b/clang/lib/Headers/avx10_2_512v2auxintrin.h
new file mode 100644
index 00000000000000..02bda8a948a778
--- /dev/null
+++ b/clang/lib/Headers/avx10_2_512v2auxintrin.h
@@ -0,0 +1,1183 @@
+/*===--------- avx10_2_512v2auxintrin.h - AVX10_2_512V2AUX ---------------===
+ *
+ * Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
+ * See https://llvm.org/LICENSE.txt for license information.
+ * SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
+ *
+ *===-----------------------------------------------------------------------===
+ */
+#ifndef __IMMINTRIN_H
+#error                                                                         \
+    "Never use <avx10_2_512v2auxintrin.h> directly; include <immintrin.h> instead."
+#endif // __IMMINTRIN_H
+
+#ifdef __SSE2__
+
+#ifndef __AVX10_2_512V2AUXINTRIN_H
+#define __AVX10_2_512V2AUXINTRIN_H
+
+/* Define the default attributes for the functions in this file. */
+#define __DEFAULT_FN_ATTRS512                                                  \
+  __attribute__((__always_inline__, __nodebug__, __target__("avx10v2aux"),     \
+                 __min_vector_width__(512)))
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __A
+///    to packed BF8 (8-bit) floating-point elements, and store the results in
+///    a 128-bit vector.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTPS2BF8 instruction.
+///
+/// \param __A
+///    A 512-bit vector of [16 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
+static __inline__ __m128i __DEFAULT_FN_ATTRS512 _mm512_cvtps_bf8(__m512 __A) {
+  return (__m128i)__builtin_ia32_vcvtps2bf8_512((__v16sf)__A);
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __A
+///    to packed BF8 (8-bit) floating-point elements, and store the results in
+///    a 128-bit vector using writemask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTPS2BF8 instruction.
+///
+/// \param __W
+///    A 128-bit vector of [16 x i8] used for writemask.
+/// \param __U
+///    A 16-bit mask indicating which elements to write.
+/// \param __A
+///    A 512-bit vector of [16 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
+static __inline__ __m128i __DEFAULT_FN_ATTRS512
+_mm512_mask_cvtps_bf8(__m128i __W, __mmask16 __U, __m512 __A) {
+  return (__m128i)__builtin_ia32_selectb_128(
+      (__mmask16)__U, (__v16qi)_mm512_cvtps_bf8(__A), (__v16qi)__W);
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __A
+///    to packed BF8 (8-bit) floating-point elements, and store the results in
+///    a 128-bit vector using zeromask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTPS2BF8 instruction.
+///
+/// \param __U
+///    A 16-bit mask indicating which elements to write (zero otherwise).
+/// \param __A
+///    A 512-bit vector of [16 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
+static __inline__ __m128i __DEFAULT_FN_ATTRS512
+_mm512_maskz_cvtps_bf8(__mmask16 __U, __m512 __A) {
+  return (__m128i)__builtin_ia32_selectb_128((__mmask16)__U,
+                                             (__v16qi)_mm512_cvtps_bf8(__A),
+                                             (__v16qi)_mm_setzero_si128());
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __A
+///    to packed BF8 (8-bit) floating-point elements with saturation, and store
+///    the results in a 128-bit vector.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTPS2BF8S instruction.
+///
+/// \param __A
+///    A 512-bit vector of [16 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
+static __inline__ __m128i __DEFAULT_FN_ATTRS512 _mm512_cvts_ps_bf8(__m512 __A) {
+  return (__m128i)__builtin_ia32_vcvtps2bf8s_512((__v16sf)__A);
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __A
+///    to packed BF8 (8-bit) floating-point elements with saturation, and store
+///    the results in a 128-bit vector using writemask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTPS2BF8S instruction.
+///
+/// \param __W
+///    A 128-bit vector of [16 x i8] used for writemask.
+/// \param __U
+///    A 16-bit mask indicating which elements to write.
+/// \param __A
+///    A 512-bit vector of [16 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
+static __inline__ __m128i __DEFAULT_FN_ATTRS512
+_mm512_mask_cvts_ps_bf8(__m128i __W, __mmask16 __U, __m512 __A) {
+  return (__m128i)__builtin_ia32_selectb_128(
+      (__mmask16)__U, (__v16qi)_mm512_cvts_ps_bf8(__A), (__v16qi)__W);
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __A
+///    to packed BF8 (8-bit) floating-point elements with saturation, and store
+///    the results in a 128-bit vector using zeromask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTPS2BF8S instruction.
+///
+/// \param __U
+///    A 16-bit mask indicating which elements to write (zero otherwise).
+/// \param __A
+///    A 512-bit vector of [16 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
+static __inline__ __m128i __DEFAULT_FN_ATTRS512
+_mm512_maskz_cvts_ps_bf8(__mmask16 __U, __m512 __A) {
+  return (__m128i)__builtin_ia32_selectb_128((__mmask16)__U,
+                                             (__v16qi)_mm512_cvts_ps_bf8(__A),
+                                             (__v16qi)_mm_setzero_si128());
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __A
+///    to packed HF8 (8-bit) floating-point elements, and store the results in
+///    a 128-bit vector.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTPS2HF8 instruction.
+///
+/// \param __A
+///    A 512-bit vector of [16 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
+static __inline__ __m128i __DEFAULT_FN_ATTRS512 _mm512_cvtps_hf8(__m512 __A) {
+  return (__m128i)__builtin_ia32_vcvtps2hf8_512((__v16sf)__A);
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __A
+///    to packed HF8 (8-bit) floating-point elements, and store the results in
+///    a 128-bit vector using writemask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTPS2HF8 instruction.
+///
+/// \param __W
+///    A 128-bit vector of [16 x i8] used for writemask.
+/// \param __U
+///    A 16-bit mask indicating which elements to write.
+/// \param __A
+///    A 512-bit vector of [16 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
+static __inline__ __m128i __DEFAULT_FN_ATTRS512
+_mm512_mask_cvtps_hf8(__m128i __W, __mmask16 __U, __m512 __A) {
+  return (__m128i)__builtin_ia32_selectb_128(
+      (__mmask16)__U, (__v16qi)_mm512_cvtps_hf8(__A), (__v16qi)__W);
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __A
+///    to packed HF8 (8-bit) floating-point elements, and store the results in
+///    a 128-bit vector using zeromask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTPS2HF8 instruction.
+///
+/// \param __U
+///    A 16-bit mask indicating which elements to write (zero otherwise).
+/// \param __A
+///    A 512-bit vector of [16 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
+static __inline__ __m128i __DEFAULT_FN_ATTRS512
+_mm512_maskz_cvtps_hf8(__mmask16 __U, __m512 __A) {
+  return (__m128i)__builtin_ia32_selectb_128((__mmask16)__U,
+                                             (__v16qi)_mm512_cvtps_hf8(__A),
+                                             (__v16qi)_mm_setzero_si128());
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __A
+///    to packed HF8 (8-bit) floating-point elements with saturation, and store
+///    the results in a 128-bit vector.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTPS2HF8S instruction.
+///
+/// \param __A
+///    A 512-bit vector of [16 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
+static __inline__ __m128i __DEFAULT_FN_ATTRS512 _mm512_cvts_ps_hf8(__m512 __A) {
+  return (__m128i)__builtin_ia32_vcvtps2hf8s_512((__v16sf)__A);
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __A
+///    to packed HF8 (8-bit) floating-point elements with saturation, and store
+///    the results in a 128-bit vector using writemask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTPS2HF8S instruction.
+///
+/// \param __W
+///    A 128-bit vector of [16 x i8] used for writemask.
+/// \param __U
+///    A 16-bit mask indicating which elements to write.
+/// \param __A
+///    A 512-bit vector of [16 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
+static __inline__ __m128i __DEFAULT_FN_ATTRS512
+_mm512_mask_cvts_ps_hf8(__m128i __W, __mmask16 __U, __m512 __A) {
+  return (__m128i)__builtin_ia32_selectb_128(
+      (__mmask16)__U, (__v16qi)_mm512_cvts_ps_hf8(__A), (__v16qi)__W);
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __A
+///    to packed HF8 (8-bit) floating-point elements with saturation, and store
+///    the results in a 128-bit vector using zeromask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTPS2HF8S instruction.
+///
+/// \param __U
+///    A 16-bit mask indicating which elements to write (zero otherwise).
+/// \param __A
+///    A 512-bit vector of [16 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
+static __inline__ __m128i __DEFAULT_FN_ATTRS512
+_mm512_maskz_cvts_ps_hf8(__mmask16 __U, __m512 __A) {
+  return (__m128i)__builtin_ia32_selectb_128((__mmask16)__U,
+                                             (__v16qi)_mm512_cvts_ps_hf8(__A),
+                                             (__v16qi)_mm_setzero_si128());
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __A
+///    to packed HF8 (8-bit) floating-point elements using round-to-odd, and
+///    store the results in a 128-bit vector.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTROPS2HF8 instruction.
+///
+/// \param __A
+///    A 512-bit vector of [16 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
+static __inline__ __m128i __DEFAULT_FN_ATTRS512 _mm512_cvtrops_hf8(__m512 __A) {
+  return (__m128i)__builtin_ia32_vcvtrops2hf8_512((__v16sf)__A);
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __A
+///    to packed HF8 (8-bit) floating-point elements using round-to-odd, and
+///    store the results in a 128-bit vector using writemask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTROPS2HF8 instruction.
+///
+/// \param __W
+///    A 128-bit vector of [16 x i8] used for writemask.
+/// \param __U
+///    A 16-bit mask indicating which elements to write.
+/// \param __A
+///    A 512-bit vector of [16 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
+static __inline__ __m128i __DEFAULT_FN_ATTRS512
+_mm512_mask_cvtrops_hf8(__m128i __W, __mmask16 __U, __m512 __A) {
+  return (__m128i)__builtin_ia32_selectb_128(
+      (__mmask16)__U, (__v16qi)_mm512_cvtrops_hf8(__A), (__v16qi)__W);
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __A
+///    to packed HF8 (8-bit) floating-point elements using round-to-odd, and
+///    store the results in a 128-bit vector using zeromask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTROPS2HF8 instruction.
+///
+/// \param __U
+///    A 16-bit mask indicating which elements to write (zero otherwise).
+/// \param __A
+///    A 512-bit vector of [16 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
+static __inline__ __m128i __DEFAULT_FN_ATTRS512
+_mm512_maskz_cvtrops_hf8(__mmask16 __U, __m512 __A) {
+  return (__m128i)__builtin_ia32_selectb_128((__mmask16)__U,
+                                             (__v16qi)_mm512_cvtrops_hf8(__A),
+                                             (__v16qi)_mm_setzero_si128());
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __A
+///    to packed HF8 (8-bit) floating-point elements using round-to-odd with
+///    saturation, and store the results in a 128-bit vector.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTROPS2HF8S instruction.
+///
+/// \param __A
+///    A 512-bit vector of [16 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
+static __inline__ __m128i __DEFAULT_FN_ATTRS512
+_mm512_cvts_rops_hf8(__m512 __A) {
+  return (__m128i)__builtin_ia32_vcvtrops2hf8s_512((__v16sf)__A);
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __A
+///    to packed HF8 (8-bit) floating-point elements using round-to-odd with
+///    saturation, and store the results in a 128-bit vector using writemask
+///    \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTROPS2HF8S instruction.
+///
+/// \param __W
+///    A 128-bit vector of [16 x i8] used for writemask.
+/// \param __U
+///    A 16-bit mask indicating which elements to write.
+/// \param __A
+///    A 512-bit vector of [16 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
+static __inline__ __m128i __DEFAULT_FN_ATTRS512
+_mm512_mask_cvts_rops_hf8(__m128i __W, __mmask16 __U, __m512 __A) {
+  return (__m128i)__builtin_ia32_selectb_128(
+      (__mmask16)__U, (__v16qi)_mm512_cvts_rops_hf8(__A), (__v16qi)__W);
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __A
+///    to packed HF8 (8-bit) floating-point elements using round-to-odd with
+///    saturation, and store the results in a 128-bit vector using zeromask
+///    \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTROPS2HF8S instruction.
+///
+/// \param __U
+///    A 16-bit mask indicating which elements to write (zero otherwise).
+/// \param __A
+///    A 512-bit vector of [16 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
+static __inline__ __m128i __DEFAULT_FN_ATTRS512
+_mm512_maskz_cvts_rops_hf8(__mmask16 __U, __m512 __A) {
+  return (__m128i)__builtin_ia32_selectb_128((__mmask16)__U,
+                                             (__v16qi)_mm512_cvts_rops_hf8(__A),
+                                             (__v16qi)_mm_setzero_si128());
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __B
+///    to packed BF8 (8-bit) floating-point elements using bias values from
+///    \a __A, and store the results in a 128-bit vector.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBIASPS2BF8 instruction.
+///
+/// \param __A
+///    A 512-bit vector of [64 x i8] containing bias values.
+/// \param __B
+///    A 512-bit vector of [16 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
+static __inline__ __m128i __DEFAULT_FN_ATTRS512
+_mm512_cvtbiasps_bf8(__m512i __A, __m512 __B) {
+  return (__m128i)__builtin_ia32_vcvtbiasps2bf8_512((__v64qi)__A, (__v16sf)__B);
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __B
+///    to packed BF8 (8-bit) floating-point elements using bias values from
+///    \a __A, and store the results in a 128-bit vector using writemask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBIASPS2BF8 instruction.
+///
+/// \param __W
+///    A 128-bit vector of [16 x i8] used for writemask.
+/// \param __U
+///    A 16-bit mask indicating which elements to write.
+/// \param __A
+///    A 512-bit vector of [64 x i8] containing bias values.
+/// \param __B
+///    A 512-bit vector of [16 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
+static __inline__ __m128i __DEFAULT_FN_ATTRS512
+_mm512_mask_cvtbiasps_bf8(__m128i __W, __mmask16 __U, __m512i __A, __m512 __B) {
+  return (__m128i)__builtin_ia32_selectb_128(
+      (__mmask16)__U, (__v16qi)_mm512_cvtbiasps_bf8(__A, __B), (__v16qi)__W);
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __B
+///    to packed BF8 (8-bit) floating-point elements using bias values from
+///    \a __A, and store the results in a 128-bit vector using zeromask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBIASPS2BF8 instruction.
+///
+/// \param __U
+///    A 16-bit mask indicating which elements to write (zero otherwise).
+/// \param __A
+///    A 512-bit vector of [64 x i8] containing bias values.
+/// \param __B
+///    A 512-bit vector of [16 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
+static __inline__ __m128i __DEFAULT_FN_ATTRS512
+_mm512_maskz_cvtbiasps_bf8(__mmask16 __U, __m512i __A, __m512 __B) {
+  return (__m128i)__builtin_ia32_selectb_128(
+      (__mmask16)__U, (__v16qi)_mm512_cvtbiasps_bf8(__A, __B),
+      (__v16qi)_mm_setzero_si128());
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __B
+///    to packed BF8 (8-bit) floating-point elements with saturation using bias
+///    values from \a __A, and store the results in a 128-bit vector.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBIASPS2BF8S instruction.
+///
+/// \param __A
+///    A 512-bit vector of [64 x i8] containing bias values.
+/// \param __B
+///    A 512-bit vector of [16 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
+static __inline__ __m128i __DEFAULT_FN_ATTRS512
+_mm512_cvts_biasps_bf8(__m512i __A, __m512 __B) {
+  return (__m128i)__builtin_ia32_vcvtbiasps2bf8s_512((__v64qi)__A,
+                                                     (__v16sf)__B);
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __B
+///    to packed BF8 (8-bit) floating-point elements with saturation using bias
+///    values from \a __A, and store the results using writemask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBIASPS2BF8S instruction.
+///
+/// \param __W
+///    A 128-bit vector of [16 x i8] used for writemask.
+/// \param __U
+///    A 16-bit mask indicating which elements to write.
+/// \param __A
+///    A 512-bit vector of [64 x i8] containing bias values.
+/// \param __B
+///    A 512-bit vector of [16 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
+static __inline__ __m128i __DEFAULT_FN_ATTRS512 _mm512_mask_cvts_biasps_bf8(
+    __m128i __W, __mmask16 __U, __m512i __A, __m512 __B) {
+  return (__m128i)__builtin_ia32_selectb_128(
+      (__mmask16)__U, (__v16qi)_mm512_cvts_biasps_bf8(__A, __B), (__v16qi)__W);
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __B
+///    to packed BF8 (8-bit) floating-point elements with saturation using bias
+///    values from \a __A, and store the results using zeromask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBIASPS2BF8S instruction.
+///
+/// \param __U
+///    A 16-bit mask indicating which elements to write (zero otherwise).
+/// \param __A
+///    A 512-bit vector of [64 x i8] containing bias values.
+/// \param __B
+///    A 512-bit vector of [16 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
+static __inline__ __m128i __DEFAULT_FN_ATTRS512
+_mm512_maskz_cvts_biasps_bf8(__mmask16 __U, __m512i __A, __m512 __B) {
+  return (__m128i)__builtin_ia32_selectb_128(
+      (__mmask16)__U, (__v16qi)_mm512_cvts_biasps_bf8(__A, __B),
+      (__v16qi)_mm_setzero_si128());
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __B
+///    to packed HF8 (8-bit) floating-point elements using bias values from
+///    \a __A, and store the results in a 128-bit vector.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBIASPS2HF8 instruction.
+///
+/// \param __A
+///    A 512-bit vector of [64 x i8] containing bias values.
+/// \param __B
+///    A 512-bit vector of [16 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
+static __inline__ __m128i __DEFAULT_FN_ATTRS512
+_mm512_cvtbiasps_hf8(__m512i __A, __m512 __B) {
+  return (__m128i)__builtin_ia32_vcvtbiasps2hf8_512((__v64qi)__A, (__v16sf)__B);
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __B
+///    to packed HF8 (8-bit) floating-point elements using bias values from
+///    \a __A, and store the results in a 128-bit vector using writemask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBIASPS2HF8 instruction.
+///
+/// \param __W
+///    A 128-bit vector of [16 x i8] used for writemask.
+/// \param __U
+///    A 16-bit mask indicating which elements to write.
+/// \param __A
+///    A 512-bit vector of [64 x i8] containing bias values.
+/// \param __B
+///    A 512-bit vector of [16 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
+static __inline__ __m128i __DEFAULT_FN_ATTRS512
+_mm512_mask_cvtbiasps_hf8(__m128i __W, __mmask16 __U, __m512i __A, __m512 __B) {
+  return (__m128i)__builtin_ia32_selectb_128(
+      (__mmask16)__U, (__v16qi)_mm512_cvtbiasps_hf8(__A, __B), (__v16qi)__W);
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __B
+///    to packed HF8 (8-bit) floating-point elements using bias values from
+///    \a __A, and store the results in a 128-bit vector using zeromask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBIASPS2HF8 instruction.
+///
+/// \param __U
+///    A 16-bit mask indicating which elements to write (zero otherwise).
+/// \param __A
+///    A 512-bit vector of [64 x i8] containing bias values.
+/// \param __B
+///    A 512-bit vector of [16 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
+static __inline__ __m128i __DEFAULT_FN_ATTRS512
+_mm512_maskz_cvtbiasps_hf8(__mmask16 __U, __m512i __A, __m512 __B) {
+  return (__m128i)__builtin_ia32_selectb_128(
+      (__mmask16)__U, (__v16qi)_mm512_cvtbiasps_hf8(__A, __B),
+      (__v16qi)_mm_setzero_si128());
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __B
+///    to packed HF8 (8-bit) floating-point elements with saturation using bias
+///    values from \a __A, and store the results in a 128-bit vector.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBIASPS2HF8S instruction.
+///
+/// \param __A
+///    A 512-bit vector of [64 x i8] containing bias values.
+/// \param __B
+///    A 512-bit vector of [16 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
+static __inline__ __m128i __DEFAULT_FN_ATTRS512
+_mm512_cvts_biasps_hf8(__m512i __A, __m512 __B) {
+  return (__m128i)__builtin_ia32_vcvtbiasps2hf8s_512((__v64qi)__A,
+                                                     (__v16sf)__B);
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __B
+///    to packed HF8 (8-bit) floating-point elements with saturation using bias
+///    values from \a __A, and store the results using writemask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBIASPS2HF8S instruction.
+///
+/// \param __W
+///    A 128-bit vector of [16 x i8] used for writemask.
+/// \param __U
+///    A 16-bit mask indicating which elements to write.
+/// \param __A
+///    A 512-bit vector of [64 x i8] containing bias values.
+/// \param __B
+///    A 512-bit vector of [16 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
+static __inline__ __m128i __DEFAULT_FN_ATTRS512 _mm512_mask_cvts_biasps_hf8(
+    __m128i __W, __mmask16 __U, __m512i __A, __m512 __B) {
+  return (__m128i)__builtin_ia32_selectb_128(
+      (__mmask16)__U, (__v16qi)_mm512_cvts_biasps_hf8(__A, __B), (__v16qi)__W);
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __B
+///    to packed HF8 (8-bit) floating-point elements with saturation using bias
+///    values from \a __A, and store the results using zeromask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBIASPS2HF8S instruction.
+///
+/// \param __U
+///    A 16-bit mask indicating which elements to write (zero otherwise).
+/// \param __A
+///    A 512-bit vector of [64 x i8] containing bias values.
+/// \param __B
+///    A 512-bit vector of [16 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
+static __inline__ __m128i __DEFAULT_FN_ATTRS512
+_mm512_maskz_cvts_biasps_hf8(__mmask16 __U, __m512i __A, __m512 __B) {
+  return (__m128i)__builtin_ia32_selectb_128(
+      (__mmask16)__U, (__v16qi)_mm512_cvts_biasps_hf8(__A, __B),
+      (__v16qi)_mm_setzero_si128());
+}
+
+/// Convert packed BF8 (8-bit) floating-point elements in \a __A to packed
+///    single-precision (32-bit) floating-point elements, and store the results
+///    in a 512-bit vector.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBF82PS instruction.
+///
+/// \param __A
+///    A 128-bit vector of [16 x i8] containing BF8 values.
+/// \returns
+///    A 512-bit vector of [16 x float] containing the converted values.
+static __inline__ __m512 __DEFAULT_FN_ATTRS512 _mm512_cvtbf8_ps(__m128i __A) {
+  return (__m512)__builtin_ia32_vcvtbf8_2ps512((__v16qi)__A);
+}
+
+/// Convert packed BF8 (8-bit) floating-point elements in \a __A to packed
+///    single-precision (32-bit) floating-point elements, and store the results
+///    in a 512-bit vector using writemask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBF82PS instruction.
+///
+/// \param __W
+///    A 512-bit vector of [16 x float] used for writemask.
+/// \param __U
+///    A 16-bit mask indicating which elements to write.
+/// \param __A
+///    A 128-bit vector of [16 x i8] containing BF8 values.
+/// \returns
+///    A 512-bit vector of [16 x float] containing the converted values.
+static __inline__ __m512 __DEFAULT_FN_ATTRS512
+_mm512_mask_cvtbf8_ps(__m512 __W, __mmask16 __U, __m128i __A) {
+  return (__m512)__builtin_ia32_selectps_512(
+      (__mmask16)__U, (__v16sf)_mm512_cvtbf8_ps(__A), (__v16sf)__W);
+}
+
+/// Convert packed BF8 (8-bit) floating-point elements in \a __A to packed
+///    single-precision (32-bit) floating-point elements, and store the results
+///    in a 512-bit vector using zeromask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBF82PS instruction.
+///
+/// \param __U
+///    A 16-bit mask indicating which elements to write (zero otherwise).
+/// \param __A
+///    A 128-bit vector of [16 x i8] containing BF8 values.
+/// \returns
+///    A 512-bit vector of [16 x float] containing the converted values.
+static __inline__ __m512 __DEFAULT_FN_ATTRS512
+_mm512_maskz_cvtbf8_ps(__mmask16 __U, __m128i __A) {
+  return (__m512)__builtin_ia32_selectps_512((__mmask16)__U,
+                                             (__v16sf)_mm512_cvtbf8_ps(__A),
+                                             (__v16sf)_mm512_setzero_ps());
+}
+
+/// Convert packed HF8 (8-bit) floating-point elements in \a __A to packed
+///    single-precision (32-bit) floating-point elements, and store the results
+///    in a 512-bit vector.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTHF82PS instruction.
+///
+/// \param __A
+///    A 128-bit vector of [16 x i8] containing HF8 values.
+/// \returns
+///    A 512-bit vector of [16 x float] containing the converted values.
+static __inline__ __m512 __DEFAULT_FN_ATTRS512 _mm512_cvthf8_ps(__m128i __A) {
+  return (__m512)__builtin_ia32_vcvthf8_2ps512((__v16qi)__A);
+}
+
+/// Convert packed HF8 (8-bit) floating-point elements in \a __A to packed
+///    single-precision (32-bit) floating-point elements, and store the results
+///    in a 512-bit vector using writemask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTHF82PS instruction.
+///
+/// \param __W
+///    A 512-bit vector of [16 x float] used for writemask.
+/// \param __U
+///    A 16-bit mask indicating which elements to write.
+/// \param __A
+///    A 128-bit vector of [16 x i8] containing HF8 values.
+/// \returns
+///    A 512-bit vector of [16 x float] containing the converted values.
+static __inline__ __m512 __DEFAULT_FN_ATTRS512
+_mm512_mask_cvthf8_ps(__m512 __W, __mmask16 __U, __m128i __A) {
+  return (__m512)__builtin_ia32_selectps_512(
+      (__mmask16)__U, (__v16sf)_mm512_cvthf8_ps(__A), (__v16sf)__W);
+}
+
+/// Convert packed HF8 (8-bit) floating-point elements in \a __A to packed
+///    single-precision (32-bit) floating-point elements, and store the results
+///    in a 512-bit vector using zeromask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTHF82PS instruction.
+///
+/// \param __U
+///    A 16-bit mask indicating which elements to write (zero otherwise).
+/// \param __A
+///    A 128-bit vector of [16 x i8] containing HF8 values.
+/// \returns
+///    A 512-bit vector of [16 x float] containing the converted values.
+static __inline__ __m512 __DEFAULT_FN_ATTRS512
+_mm512_maskz_cvthf8_ps(__mmask16 __U, __m128i __A) {
+  return (__m512)__builtin_ia32_selectps_512((__mmask16)__U,
+                                             (__v16sf)_mm512_cvthf8_ps(__A),
+                                             (__v16sf)_mm512_setzero_ps());
+}
+
+/// Convert packed BF8 (8-bit) floating-point elements in \a __A to packed
+///    BF6 (6-bit) floating-point elements with saturation, and store the
+///    results in a 512-bit vector.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBF82BF6S instruction.
+///
+/// \param __A
+///    A 512-bit vector of [64 x i8] containing BF8 values.
+/// \returns
+///    A 512-bit vector of [64 x i8] containing the converted BF6 values.
+static __inline__ __m512i __DEFAULT_FN_ATTRS512
+_mm512_cvtbf8_bf6s(__m512i __A) {
+  return (__m512i)__builtin_ia32_vcvtbf82bf6s512((__v64qi)__A);
+}
+
+/// Convert packed HF8 (8-bit) floating-point elements in \a __A to packed
+///    HF6 (6-bit) floating-point elements with saturation, and store the
+///    results in a 512-bit vector.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTHF82HF6S instruction.
+///
+/// \param __A
+///    A 512-bit vector of [64 x i8] containing HF8 values.
+/// \returns
+///    A 512-bit vector of [64 x i8] containing the converted HF6 values.
+static __inline__ __m512i __DEFAULT_FN_ATTRS512
+_mm512_cvthf8_hf6s(__m512i __A) {
+  return (__m512i)__builtin_ia32_vcvthf82hf6s512((__v64qi)__A);
+}
+
+/// Convert packed BF8 (8-bit) floating-point elements in \a __A to packed
+///    BF4 (4-bit) floating-point elements with saturation, and store the
+///    results in a 256-bit vector.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBF82BF4S instruction.
+///
+/// \param __A
+///    A 512-bit vector of [64 x i8] containing BF8 values.
+/// \returns
+///    A 256-bit vector of [32 x i8] containing the converted BF4 values.
+static __inline__ __m256i __DEFAULT_FN_ATTRS512
+_mm512_cvtbf8_bf4s(__m512i __A) {
+  return (__m256i)__builtin_ia32_vcvtbf82bf4s512((__v64qi)__A);
+}
+
+/// Convert packed HF8 (8-bit) floating-point elements in \a __A to packed
+///    BF4 (4-bit) floating-point elements with saturation, and store the
+///    results in a 256-bit vector.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTHF82BF4S instruction.
+///
+/// \param __A
+///    A 512-bit vector of [64 x i8] containing HF8 values.
+/// \returns
+///    A 256-bit vector of [32 x i8] containing the converted BF4 values.
+static __inline__ __m256i __DEFAULT_FN_ATTRS512
+_mm512_cvthf8_bf4s(__m512i __A) {
+  return (__m256i)__builtin_ia32_vcvthf82bf4s512((__v64qi)__A);
+}
+
+/// Convert packed BF8 (8-bit) floating-point elements in \a __A to packed
+///    BF4 (4-bit) floating-point elements with saturation, and store the
+///    results to memory at \a __P.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBF82BF4S instruction.
+///
+/// \param __P
+///    A pointer to a 256-bit memory location. The address does not need to be
+///    aligned.
+/// \param __A
+///    A 512-bit vector of [64 x i8] containing BF8 values.
+static __inline__ void __DEFAULT_FN_ATTRS512
+_mm512_cvtbf8_bf4s_storeu(void *__P, __m512i __A) {
+  __builtin_ia32_vcvtbf82bf4s512mem(__P, (__v64qi)__A);
+}
+
+/// Convert packed HF8 (8-bit) floating-point elements in \a __A to packed
+///    BF4 (4-bit) floating-point elements with saturation, and store the
+///    results to memory at \a __P.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTHF82BF4S instruction.
+///
+/// \param __P
+///    A pointer to a 256-bit memory location. The address does not need to be
+///    aligned.
+/// \param __A
+///    A 512-bit vector of [64 x i8] containing HF8 values.
+static __inline__ void __DEFAULT_FN_ATTRS512
+_mm512_cvthf8_bf4s_storeu(void *__P, __m512i __A) {
+  __builtin_ia32_vcvthf82bf4s512mem(__P, (__v64qi)__A);
+}
+
+/// Convert packed BF4 (4-bit) floating-point elements in \a __A to packed
+///    HF8 (8-bit) floating-point elements, and store the results in a 512-bit
+///    vector.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBF42HF8 instruction.
+///
+/// \param __A
+///    A 256-bit vector of [32 x i8] containing BF4 values.
+/// \returns
+///    A 512-bit vector of [64 x i8] containing the converted HF8 values.
+static __inline__ __m512i __DEFAULT_FN_ATTRS512 _mm512_cvtbf4_hf8(__m256i __A) {
+  return (__m512i)__builtin_ia32_vcvtbf42hf8512((__v32qi)__A);
+}
+
+/// Convert packed BF4 (4-bit) floating-point elements in \a __A to packed
+///    HF8 (8-bit) floating-point elements, and store the results in a 512-bit
+///    vector using writemask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBF42HF8 instruction.
+///
+/// \param __W
+///    A 512-bit vector of [64 x i8] used for writemask.
+/// \param __U
+///    A 64-bit mask indicating which elements to write.
+/// \param __A
+///    A 256-bit vector of [32 x i8] containing BF4 values.
+/// \returns
+///    A 512-bit vector of [64 x i8] containing the converted HF8 values.
+static __inline__ __m512i __DEFAULT_FN_ATTRS512
+_mm512_mask_cvtbf4_hf8(__m512i __W, __mmask64 __U, __m256i __A) {
+  return (__m512i)__builtin_ia32_selectb_512(
+      __U, (__v64qi)_mm512_cvtbf4_hf8(__A), (__v64qi)__W);
+}
+
+/// Convert packed BF4 (4-bit) floating-point elements in \a __A to packed
+///    HF8 (8-bit) floating-point elements, and store the results in a 512-bit
+///    vector using zeromask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBF42HF8 instruction.
+///
+/// \param __U
+///    A 64-bit mask indicating which elements to write (zero otherwise).
+/// \param __A
+///    A 256-bit vector of [32 x i8] containing BF4 values.
+/// \returns
+///    A 512-bit vector of [64 x i8] containing the converted HF8 values.
+static __inline__ __m512i __DEFAULT_FN_ATTRS512
+_mm512_maskz_cvtbf4_hf8(__mmask64 __U, __m256i __A) {
+  return (__m512i)__builtin_ia32_selectb_512(
+      __U, (__v64qi)_mm512_cvtbf4_hf8(__A), (__v64qi)_mm512_setzero_si512());
+}
+
+/// Convert packed BF6 (6-bit) floating-point elements in \a __A to packed
+///    HF8 (8-bit) floating-point elements, and store the results in a 512-bit
+///    vector.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBF62HF8 instruction.
+///
+/// \param __A
+///    A 512-bit vector of [64 x i8] containing BF6 values.
+/// \returns
+///    A 512-bit vector of [64 x i8] containing the converted HF8 values.
+static __inline__ __m512i __DEFAULT_FN_ATTRS512 _mm512_cvtbf6_hf8(__m512i __A) {
+  return (__m512i)__builtin_ia32_vcvtbf62hf8512((__v64qi)__A);
+}
+
+/// Convert packed BF6 (6-bit) floating-point elements in \a __A to packed
+///    HF8 (8-bit) floating-point elements, and store the results in a 512-bit
+///    vector using writemask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBF62HF8 instruction.
+///
+/// \param __W
+///    A 512-bit vector of [64 x i8] used for writemask.
+/// \param __U
+///    A 64-bit mask indicating which elements to write.
+/// \param __A
+///    A 512-bit vector of [64 x i8] containing BF6 values.
+/// \returns
+///    A 512-bit vector of [64 x i8] containing the converted HF8 values.
+static __inline__ __m512i __DEFAULT_FN_ATTRS512
+_mm512_mask_cvtbf6_hf8(__m512i __W, __mmask64 __U, __m512i __A) {
+  return (__m512i)__builtin_ia32_selectb_512(
+      __U, (__v64qi)_mm512_cvtbf6_hf8(__A), (__v64qi)__W);
+}
+
+/// Convert packed BF6 (6-bit) floating-point elements in \a __A to packed
+///    HF8 (8-bit) floating-point elements, and store the results in a 512-bit
+///    vector using zeromask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBF62HF8 instruction.
+///
+/// \param __U
+///    A 64-bit mask indicating which elements to write (zero otherwise).
+/// \param __A
+///    A 512-bit vector of [64 x i8] containing BF6 values.
+/// \returns
+///    A 512-bit vector of [64 x i8] containing the converted HF8 values.
+static __inline__ __m512i __DEFAULT_FN_ATTRS512
+_mm512_maskz_cvtbf6_hf8(__mmask64 __U, __m512i __A) {
+  return (__m512i)__builtin_ia32_selectb_512(
+      __U, (__v64qi)_mm512_cvtbf6_hf8(__A), (__v64qi)_mm512_setzero_si512());
+}
+
+/// Convert packed HF6 (6-bit) floating-point elements in \a __A to packed
+///    HF8 (8-bit) floating-point elements, and store the results in a 512-bit
+///    vector.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTHF62HF8 instruction.
+///
+/// \param __A
+///    A 512-bit vector of [64 x i8] containing HF6 values.
+/// \returns
+///    A 512-bit vector of [64 x i8] containing the converted HF8 values.
+static __inline__ __m512i __DEFAULT_FN_ATTRS512 _mm512_cvthf6_hf8(__m512i __A) {
+  return (__m512i)__builtin_ia32_vcvthf62hf8512((__v64qi)__A);
+}
+
+/// Convert packed HF6 (6-bit) floating-point elements in \a __A to packed
+///    HF8 (8-bit) floating-point elements, and store the results in a 512-bit
+///    vector using writemask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTHF62HF8 instruction.
+///
+/// \param __W
+///    A 512-bit vector of [64 x i8] used for writemask.
+/// \param __U
+///    A 64-bit mask indicating which elements to write.
+/// \param __A
+///    A 512-bit vector of [64 x i8] containing HF6 values.
+/// \returns
+///    A 512-bit vector of [64 x i8] containing the converted HF8 values.
+static __inline__ __m512i __DEFAULT_FN_ATTRS512
+_mm512_mask_cvthf6_hf8(__m512i __W, __mmask64 __U, __m512i __A) {
+  return (__m512i)__builtin_ia32_selectb_512(
+      __U, (__v64qi)_mm512_cvthf6_hf8(__A), (__v64qi)__W);
+}
+
+/// Convert packed HF6 (6-bit) floating-point elements in \a __A to packed
+///    HF8 (8-bit) floating-point elements, and store the results in a 512-bit
+///    vector using zeromask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTHF62HF8 instruction.
+///
+/// \param __U
+///    A 64-bit mask indicating which elements to write (zero otherwise).
+/// \param __A
+///    A 512-bit vector of [64 x i8] containing HF6 values.
+/// \returns
+///    A 512-bit vector of [64 x i8] containing the converted HF8 values.
+static __inline__ __m512i __DEFAULT_FN_ATTRS512
+_mm512_maskz_cvthf6_hf8(__mmask64 __U, __m512i __A) {
+  return (__m512i)__builtin_ia32_selectb_512(
+      __U, (__v64qi)_mm512_cvthf6_hf8(__A), (__v64qi)_mm512_setzero_si512());
+}
+
+/// Unpack bytes from \a A according to the immediate value \a imm, and store
+///    the results in a 512-bit vector.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VUNPACKB instruction.
+///
+/// \param A
+///    A 512-bit vector of [64 x i8].
+/// \param imm
+///    An immediate value specifying the unpack operation.
+/// \returns
+///    A 512-bit vector of [64 x i8] containing the unpacked values.
+#define _mm512_unpackb_epi8(A, imm)                                            \
+  ((__m512i)__builtin_ia32_vunpackb512((__v64qi)(__m512i)(A), (int)(imm)))
+
+/// Unpack bytes from \a A according to the immediate value \a imm, and store
+///    the results in a 512-bit vector using writemask \a U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VUNPACKB instruction.
+///
+/// \param W
+///    A 512-bit vector of [64 x i8] used for writemask.
+/// \param U
+///    A 64-bit mask indicating which elements to write.
+/// \param A
+///    A 512-bit vector of [64 x i8].
+/// \param imm
+///    An immediate value specifying the unpack operation.
+/// \returns
+///    A 512-bit vector of [64 x i8] containing the unpacked values.
+#define _mm512_mask_unpackb_epi8(W, U, A, imm)                                 \
+  ((__m512i)__builtin_ia32_selectb_512(                                        \
+      (__mmask64)(U), (__v64qi)_mm512_unpackb_epi8((A), (imm)),                \
+      (__v64qi)(__m512i)(W)))
+
+/// Unpack bytes from \a A according to the immediate value \a imm, and store
+///    the results in a 512-bit vector using zeromask \a U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VUNPACKB instruction.
+///
+/// \param U
+///    A 64-bit mask indicating which elements to write (zero otherwise).
+/// \param A
+///    A 512-bit vector of [64 x i8].
+/// \param imm
+///    An immediate value specifying the unpack operation.
+/// \returns
+///    A 512-bit vector of [64 x i8] containing the unpacked values.
+#define _mm512_maskz_unpackb_epi8(U, A, imm)                                   \
+  ((__m512i)__builtin_ia32_selectb_512(                                        \
+      (__mmask64)(U), (__v64qi)_mm512_unpackb_epi8((A), (imm)),                \
+      (__v64qi)_mm512_setzero_si512()))
+
+/* VPMOVSSDB - Symmetric Signed Saturation DWord to Byte */
+
+/// Convert packed signed 32-bit integers in \a __A to packed 8-bit integers
+///    with symmetric signed saturation (clamp to [-127, +127]), and store
+///    the results in a 128-bit vector.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VPMOVSSDB instruction.
+///
+/// \param __A
+///    A 512-bit vector of [16 x i32].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
+static __inline__ __m128i __DEFAULT_FN_ATTRS512
+_mm512_cvtss_epi32_epi8(__m512i __A) {
+  return (__m128i)__builtin_ia32_pmovssdb512_mask(
+      (__v16si)__A, (__v16qi)_mm_setzero_si128(), (__mmask16)-1);
+}
+
+/// Convert packed signed 32-bit integers in \a __A to packed 8-bit integers
+///    with symmetric signed saturation, using writemask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VPMOVSSDB instruction.
+///
+/// \param __W
+///    A 128-bit vector of [16 x i8] used for writemask.
+/// \param __U
+///    A 16-bit mask indicating which elements to write.
+/// \param __A
+///    A 512-bit vector of [16 x i32].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
+static __inline__ __m128i __DEFAULT_FN_ATTRS512
+_mm512_mask_cvtss_epi32_epi8(__m128i __W, __mmask16 __U, __m512i __A) {
+  return (__m128i)__builtin_ia32_pmovssdb512_mask((__v16si)__A, (__v16qi)__W,
+                                                  __U);
+}
+
+/// Convert packed signed 32-bit integers in \a __A to packed 8-bit integers
+///    with symmetric signed saturation, using zeromask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VPMOVSSDB instruction.
+///
+/// \param __U
+///    A 16-bit mask indicating which elements to write (zero otherwise).
+/// \param __A
+///    A 512-bit vector of [16 x i32].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
+static __inline__ __m128i __DEFAULT_FN_ATTRS512
+_mm512_maskz_cvtss_epi32_epi8(__mmask16 __U, __m512i __A) {
+  return (__m128i)__builtin_ia32_pmovssdb512_mask(
+      (__v16si)__A, (__v16qi)_mm_setzero_si128(), __U);
+}
+
+/// Truncate packed 32-bit integers in \a __A to packed 8-bit integers with
+/// symmetric signed saturation, and store the results to memory at \a __P
+/// using writemask \a __M (elements not selected by the mask are not written).
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VPMOVSSDB instruction.
+///
+/// \param __P
+///    Pointer to the destination memory.
+/// \param __M
+///    A 16-bit mask indicating which elements to write.
+/// \param __A
+///    A 512-bit vector of [16 x i32].
+static __inline__ void __DEFAULT_FN_ATTRS512
+_mm512_mask_cvtss_epi32_storeu_epi8(void *__P, __mmask16 __M, __m512i __A) {
+  __builtin_ia32_pmovssdb512mem_mask((__v16qi *)__P, (__v16si)__A, __M);
+}
+
+#undef __DEFAULT_FN_ATTRS512
+
+#endif // __AVX10_2_512V2AUXINTRIN_H
+#endif // __SSE2__
diff --git a/clang/lib/Headers/avx10_2_v2auxintrin.h b/clang/lib/Headers/avx10_2_v2auxintrin.h
index 5e54085c104c52..2214b8cead4c25 100644
--- a/clang/lib/Headers/avx10_2_v2auxintrin.h
+++ b/clang/lib/Headers/avx10_2_v2auxintrin.h
@@ -18,1050 +18,2338 @@
 
 /* Define the default attributes for the functions in this file. */
 #define __DEFAULT_FN_ATTRS128                                                  \
-  __attribute__((__always_inline__, __nodebug__, __target__("avx10-v2-aux"),   \
+  __attribute__((__always_inline__, __nodebug__, __target__("avx10v2aux"),     \
                  __min_vector_width__(128)))
 #define __DEFAULT_FN_ATTRS256                                                  \
-  __attribute__((__always_inline__, __nodebug__, __target__("avx10-v2-aux"),   \
+  __attribute__((__always_inline__, __nodebug__, __target__("avx10v2aux"),     \
                  __min_vector_width__(256)))
-#define __DEFAULT_FN_ATTRS512                                                  \
-  __attribute__((__always_inline__, __nodebug__, __target__("avx10-v2-aux"),   \
-                 __min_vector_width__(512)))
-
-// clang-format off
-
-//===----------------------------------------------------------------------===//
-// Group A: VCVTPS2BF8 / VCVTPS2BF8S / VCVTPS2HF8 / VCVTPS2HF8S /
-//          VCVTROPS2HF8 / VCVTROPS2HF8S
-// Convert packed single-precision to FP8. Output is always __m128i.
-//===----------------------------------------------------------------------===//
-
-// VCVTPS2BF8 - 128-bit
 
+/// Convert packed single-precision (32-bit) floating-point elements in \a __A
+///    to packed BF8 (8-bit) floating-point elements, and store the results in
+///    a 128-bit vector.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTPS2BF8 instruction.
+///
+/// \param __A
+///    A 128-bit vector of [4 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8]. The lower 4 bytes contain the converted
+///    values; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_cvtps_bf8(__m128 __A) {
-  return (__m128i)__builtin_ia32_vcvtps2bf8_128_mask(
-      (__v4sf)__A, (__v16qi)_mm_undefined_si128(), (__mmask8)-1);
-}
-
+  return (__m128i)__builtin_ia32_vcvtps2bf8_128((__v4sf)__A);
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __A
+///    to packed BF8 (8-bit) floating-point elements, and store the results in
+///    a 128-bit vector using writemask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTPS2BF8 instruction.
+///
+/// \param __W
+///    A 128-bit vector of [16 x i8] used for writemask.
+/// \param __U
+///    A 8-bit mask indicating which elements to write.
+/// \param __A
+///    A 128-bit vector of [4 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_mask_cvtps_bf8(__m128i __W, __mmask8 __U, __m128 __A) {
-  return (__m128i)__builtin_ia32_vcvtps2bf8_128_mask(
-      (__v4sf)__A, (__v16qi)(__m128i)__W, (__mmask8)__U);
-}
-
+  return (__m128i)__builtin_ia32_selectb_128(
+      (__mmask16)__U, (__v16qi)_mm_cvtps_bf8(__A), (__v16qi)__W);
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __A
+///    to packed BF8 (8-bit) floating-point elements, and store the results in
+///    a 128-bit vector using zeromask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTPS2BF8 instruction.
+///
+/// \param __U
+///    A 8-bit mask indicating which elements to write (zero otherwise).
+/// \param __A
+///    A 128-bit vector of [4 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_maskz_cvtps_bf8(__mmask8 __U, __m128 __A) {
-  return (__m128i)__builtin_ia32_vcvtps2bf8_128_mask(
-      (__v4sf)__A, (__v16qi)(__m128i)_mm_setzero_si128(), (__mmask8)__U);
-}
-
-// VCVTPS2BF8 - 256-bit
-
+  return (__m128i)__builtin_ia32_selectb_128((__mmask16)__U,
+                                             (__v16qi)_mm_cvtps_bf8(__A),
+                                             (__v16qi)_mm_setzero_si128());
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __A
+///    to packed BF8 (8-bit) floating-point elements, and store the results in
+///    a 128-bit vector.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTPS2BF8 instruction.
+///
+/// \param __A
+///    A 256-bit vector of [8 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8]. The lower 8 bytes contain the converted
+///    values; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_cvtps_bf8(__m256 __A) {
-  return (__m128i)__builtin_ia32_vcvtps2bf8_256_mask(
-      (__v8sf)__A, (__v16qi)_mm_undefined_si128(), (__mmask8)-1);
-}
-
+  return (__m128i)__builtin_ia32_vcvtps2bf8_256((__v8sf)__A);
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __A
+///    to packed BF8 (8-bit) floating-point elements, and store the results in
+///    a 128-bit vector using writemask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTPS2BF8 instruction.
+///
+/// \param __W
+///    A 128-bit vector of [16 x i8] used for writemask.
+/// \param __U
+///    A 8-bit mask indicating which elements to write.
+/// \param __A
+///    A 256-bit vector of [8 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_mask_cvtps_bf8(__m128i __W, __mmask8 __U, __m256 __A) {
-  return (__m128i)__builtin_ia32_vcvtps2bf8_256_mask(
-      (__v8sf)__A, (__v16qi)(__m128i)__W, (__mmask8)__U);
-}
-
+  return (__m128i)__builtin_ia32_selectb_128(
+      (__mmask16)__U, (__v16qi)_mm256_cvtps_bf8(__A), (__v16qi)__W);
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __A
+///    to packed BF8 (8-bit) floating-point elements, and store the results in
+///    a 128-bit vector using zeromask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTPS2BF8 instruction.
+///
+/// \param __U
+///    A 8-bit mask indicating which elements to write (zero otherwise).
+/// \param __A
+///    A 256-bit vector of [8 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_maskz_cvtps_bf8(__mmask8 __U, __m256 __A) {
-  return (__m128i)__builtin_ia32_vcvtps2bf8_256_mask(
-      (__v8sf)__A, (__v16qi)(__m128i)_mm_setzero_si128(), (__mmask8)__U);
-}
-
-// VCVTPS2BF8 - 512-bit
-
-static __inline__ __m128i __DEFAULT_FN_ATTRS512
-_mm512_cvtps_bf8(__m512 __A) {
-  return (__m128i)__builtin_ia32_vcvtps2bf8_512_mask(
-      (__v16sf)__A, (__v16qi)_mm_undefined_si128(), (__mmask16)-1);
-}
-
-static __inline__ __m128i __DEFAULT_FN_ATTRS512
-_mm512_mask_cvtps_bf8(__m128i __W, __mmask16 __U, __m512 __A) {
-  return (__m128i)__builtin_ia32_vcvtps2bf8_512_mask(
-      (__v16sf)__A, (__v16qi)(__m128i)__W, (__mmask16)__U);
-}
-
-static __inline__ __m128i __DEFAULT_FN_ATTRS512
-_mm512_maskz_cvtps_bf8(__mmask16 __U, __m512 __A) {
-  return (__m128i)__builtin_ia32_vcvtps2bf8_512_mask(
-      (__v16sf)__A, (__v16qi)(__m128i)_mm_setzero_si128(), (__mmask16)__U);
-}
-
-// VCVTPS2BF8S - 128-bit
-
+  return (__m128i)__builtin_ia32_selectb_128((__mmask16)__U,
+                                             (__v16qi)_mm256_cvtps_bf8(__A),
+                                             (__v16qi)_mm_setzero_si128());
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __A
+///    to packed BF8 (8-bit) floating-point elements with saturation, and store
+///    the results in a 128-bit vector.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTPS2BF8S instruction.
+///
+/// \param __A
+///    A 128-bit vector of [4 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_cvts_ps_bf8(__m128 __A) {
-  return (__m128i)__builtin_ia32_vcvtps2bf8s_128_mask(
-      (__v4sf)__A, (__v16qi)_mm_undefined_si128(), (__mmask8)-1);
-}
-
+  return (__m128i)__builtin_ia32_vcvtps2bf8s_128((__v4sf)__A);
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __A
+///    to packed BF8 (8-bit) floating-point elements with saturation, and store
+///    the results in a 128-bit vector using writemask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTPS2BF8S instruction.
+///
+/// \param __W
+///    A 128-bit vector of [16 x i8] used for writemask.
+/// \param __U
+///    A 8-bit mask indicating which elements to write.
+/// \param __A
+///    A 128-bit vector of [4 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_mask_cvts_ps_bf8(__m128i __W, __mmask8 __U, __m128 __A) {
-  return (__m128i)__builtin_ia32_vcvtps2bf8s_128_mask(
-      (__v4sf)__A, (__v16qi)(__m128i)__W, (__mmask8)__U);
-}
-
+  return (__m128i)__builtin_ia32_selectb_128(
+      (__mmask16)__U, (__v16qi)_mm_cvts_ps_bf8(__A), (__v16qi)__W);
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __A
+///    to packed BF8 (8-bit) floating-point elements with saturation, and store
+///    the results in a 128-bit vector using zeromask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTPS2BF8S instruction.
+///
+/// \param __U
+///    A 8-bit mask indicating which elements to write (zero otherwise).
+/// \param __A
+///    A 128-bit vector of [4 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_maskz_cvts_ps_bf8(__mmask8 __U, __m128 __A) {
-  return (__m128i)__builtin_ia32_vcvtps2bf8s_128_mask(
-      (__v4sf)__A, (__v16qi)(__m128i)_mm_setzero_si128(), (__mmask8)__U);
-}
-
-// VCVTPS2BF8S - 256-bit
-
+  return (__m128i)__builtin_ia32_selectb_128((__mmask16)__U,
+                                             (__v16qi)_mm_cvts_ps_bf8(__A),
+                                             (__v16qi)_mm_setzero_si128());
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __A
+///    to packed BF8 (8-bit) floating-point elements with saturation, and store
+///    the results in a 128-bit vector.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTPS2BF8S instruction.
+///
+/// \param __A
+///    A 256-bit vector of [8 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_cvts_ps_bf8(__m256 __A) {
-  return (__m128i)__builtin_ia32_vcvtps2bf8s_256_mask(
-      (__v8sf)__A, (__v16qi)_mm_undefined_si128(), (__mmask8)-1);
-}
-
+  return (__m128i)__builtin_ia32_vcvtps2bf8s_256((__v8sf)__A);
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __A
+///    to packed BF8 (8-bit) floating-point elements with saturation, and store
+///    the results in a 128-bit vector using writemask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTPS2BF8S instruction.
+///
+/// \param __W
+///    A 128-bit vector of [16 x i8] used for writemask.
+/// \param __U
+///    A 8-bit mask indicating which elements to write.
+/// \param __A
+///    A 256-bit vector of [8 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_mask_cvts_ps_bf8(__m128i __W, __mmask8 __U, __m256 __A) {
-  return (__m128i)__builtin_ia32_vcvtps2bf8s_256_mask(
-      (__v8sf)__A, (__v16qi)(__m128i)__W, (__mmask8)__U);
-}
-
+  return (__m128i)__builtin_ia32_selectb_128(
+      (__mmask16)__U, (__v16qi)_mm256_cvts_ps_bf8(__A), (__v16qi)__W);
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __A
+///    to packed BF8 (8-bit) floating-point elements with saturation, and store
+///    the results in a 128-bit vector using zeromask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTPS2BF8S instruction.
+///
+/// \param __U
+///    A 8-bit mask indicating which elements to write (zero otherwise).
+/// \param __A
+///    A 256-bit vector of [8 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_maskz_cvts_ps_bf8(__mmask8 __U, __m256 __A) {
-  return (__m128i)__builtin_ia32_vcvtps2bf8s_256_mask(
-      (__v8sf)__A, (__v16qi)(__m128i)_mm_setzero_si128(), (__mmask8)__U);
-}
-
-// VCVTPS2BF8S - 512-bit
-
-static __inline__ __m128i __DEFAULT_FN_ATTRS512
-_mm512_cvts_ps_bf8(__m512 __A) {
-  return (__m128i)__builtin_ia32_vcvtps2bf8s_512_mask(
-      (__v16sf)__A, (__v16qi)_mm_undefined_si128(), (__mmask16)-1);
-}
-
-static __inline__ __m128i __DEFAULT_FN_ATTRS512
-_mm512_mask_cvts_ps_bf8(__m128i __W, __mmask16 __U, __m512 __A) {
-  return (__m128i)__builtin_ia32_vcvtps2bf8s_512_mask(
-      (__v16sf)__A, (__v16qi)(__m128i)__W, (__mmask16)__U);
-}
-
-static __inline__ __m128i __DEFAULT_FN_ATTRS512
-_mm512_maskz_cvts_ps_bf8(__mmask16 __U, __m512 __A) {
-  return (__m128i)__builtin_ia32_vcvtps2bf8s_512_mask(
-      (__v16sf)__A, (__v16qi)(__m128i)_mm_setzero_si128(), (__mmask16)__U);
-}
-
-// VCVTPS2HF8 - 128-bit
-
+  return (__m128i)__builtin_ia32_selectb_128((__mmask16)__U,
+                                             (__v16qi)_mm256_cvts_ps_bf8(__A),
+                                             (__v16qi)_mm_setzero_si128());
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __A
+///    to packed HF8 (8-bit) floating-point elements, and store the results in
+///    a 128-bit vector.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTPS2HF8 instruction.
+///
+/// \param __A
+///    A 128-bit vector of [4 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_cvtps_hf8(__m128 __A) {
-  return (__m128i)__builtin_ia32_vcvtps2hf8_128_mask(
-      (__v4sf)__A, (__v16qi)_mm_undefined_si128(), (__mmask8)-1);
-}
-
+  return (__m128i)__builtin_ia32_vcvtps2hf8_128((__v4sf)__A);
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __A
+///    to packed HF8 (8-bit) floating-point elements, and store the results in
+///    a 128-bit vector using writemask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTPS2HF8 instruction.
+///
+/// \param __W
+///    A 128-bit vector of [16 x i8] used for writemask.
+/// \param __U
+///    A 8-bit mask indicating which elements to write.
+/// \param __A
+///    A 128-bit vector of [4 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_mask_cvtps_hf8(__m128i __W, __mmask8 __U, __m128 __A) {
-  return (__m128i)__builtin_ia32_vcvtps2hf8_128_mask(
-      (__v4sf)__A, (__v16qi)(__m128i)__W, (__mmask8)__U);
-}
-
+  return (__m128i)__builtin_ia32_selectb_128(
+      (__mmask16)__U, (__v16qi)_mm_cvtps_hf8(__A), (__v16qi)__W);
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __A
+///    to packed HF8 (8-bit) floating-point elements, and store the results in
+///    a 128-bit vector using zeromask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTPS2HF8 instruction.
+///
+/// \param __U
+///    A 8-bit mask indicating which elements to write (zero otherwise).
+/// \param __A
+///    A 128-bit vector of [4 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_maskz_cvtps_hf8(__mmask8 __U, __m128 __A) {
-  return (__m128i)__builtin_ia32_vcvtps2hf8_128_mask(
-      (__v4sf)__A, (__v16qi)(__m128i)_mm_setzero_si128(), (__mmask8)__U);
-}
-
-// VCVTPS2HF8 - 256-bit
-
+  return (__m128i)__builtin_ia32_selectb_128((__mmask16)__U,
+                                             (__v16qi)_mm_cvtps_hf8(__A),
+                                             (__v16qi)_mm_setzero_si128());
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __A
+///    to packed HF8 (8-bit) floating-point elements, and store the results in
+///    a 128-bit vector.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTPS2HF8 instruction.
+///
+/// \param __A
+///    A 256-bit vector of [8 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_cvtps_hf8(__m256 __A) {
-  return (__m128i)__builtin_ia32_vcvtps2hf8_256_mask(
-      (__v8sf)__A, (__v16qi)_mm_undefined_si128(), (__mmask8)-1);
-}
-
+  return (__m128i)__builtin_ia32_vcvtps2hf8_256((__v8sf)__A);
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __A
+///    to packed HF8 (8-bit) floating-point elements, and store the results in
+///    a 128-bit vector using writemask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTPS2HF8 instruction.
+///
+/// \param __W
+///    A 128-bit vector of [16 x i8] used for writemask.
+/// \param __U
+///    A 8-bit mask indicating which elements to write.
+/// \param __A
+///    A 256-bit vector of [8 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_mask_cvtps_hf8(__m128i __W, __mmask8 __U, __m256 __A) {
-  return (__m128i)__builtin_ia32_vcvtps2hf8_256_mask(
-      (__v8sf)__A, (__v16qi)(__m128i)__W, (__mmask8)__U);
-}
-
+  return (__m128i)__builtin_ia32_selectb_128(
+      (__mmask16)__U, (__v16qi)_mm256_cvtps_hf8(__A), (__v16qi)__W);
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __A
+///    to packed HF8 (8-bit) floating-point elements, and store the results in
+///    a 128-bit vector using zeromask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTPS2HF8 instruction.
+///
+/// \param __U
+///    A 8-bit mask indicating which elements to write (zero otherwise).
+/// \param __A
+///    A 256-bit vector of [8 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_maskz_cvtps_hf8(__mmask8 __U, __m256 __A) {
-  return (__m128i)__builtin_ia32_vcvtps2hf8_256_mask(
-      (__v8sf)__A, (__v16qi)(__m128i)_mm_setzero_si128(), (__mmask8)__U);
-}
-
-// VCVTPS2HF8 - 512-bit
-
-static __inline__ __m128i __DEFAULT_FN_ATTRS512
-_mm512_cvtps_hf8(__m512 __A) {
-  return (__m128i)__builtin_ia32_vcvtps2hf8_512_mask(
-      (__v16sf)__A, (__v16qi)_mm_undefined_si128(), (__mmask16)-1);
-}
-
-static __inline__ __m128i __DEFAULT_FN_ATTRS512
-_mm512_mask_cvtps_hf8(__m128i __W, __mmask16 __U, __m512 __A) {
-  return (__m128i)__builtin_ia32_vcvtps2hf8_512_mask(
-      (__v16sf)__A, (__v16qi)(__m128i)__W, (__mmask16)__U);
-}
-
-static __inline__ __m128i __DEFAULT_FN_ATTRS512
-_mm512_maskz_cvtps_hf8(__mmask16 __U, __m512 __A) {
-  return (__m128i)__builtin_ia32_vcvtps2hf8_512_mask(
-      (__v16sf)__A, (__v16qi)(__m128i)_mm_setzero_si128(), (__mmask16)__U);
-}
-
-// VCVTPS2HF8S - 128-bit
-
+  return (__m128i)__builtin_ia32_selectb_128((__mmask16)__U,
+                                             (__v16qi)_mm256_cvtps_hf8(__A),
+                                             (__v16qi)_mm_setzero_si128());
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __A
+///    to packed HF8 (8-bit) floating-point elements with saturation, and store
+///    the results in a 128-bit vector.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTPS2HF8S instruction.
+///
+/// \param __A
+///    A 128-bit vector of [4 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_cvts_ps_hf8(__m128 __A) {
-  return (__m128i)__builtin_ia32_vcvtps2hf8s_128_mask(
-      (__v4sf)__A, (__v16qi)_mm_undefined_si128(), (__mmask8)-1);
-}
-
+  return (__m128i)__builtin_ia32_vcvtps2hf8s_128((__v4sf)__A);
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __A
+///    to packed HF8 (8-bit) floating-point elements with saturation, and store
+///    the results in a 128-bit vector using writemask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTPS2HF8S instruction.
+///
+/// \param __W
+///    A 128-bit vector of [16 x i8] used for writemask.
+/// \param __U
+///    A 8-bit mask indicating which elements to write.
+/// \param __A
+///    A 128-bit vector of [4 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_mask_cvts_ps_hf8(__m128i __W, __mmask8 __U, __m128 __A) {
-  return (__m128i)__builtin_ia32_vcvtps2hf8s_128_mask(
-      (__v4sf)__A, (__v16qi)(__m128i)__W, (__mmask8)__U);
-}
-
+  return (__m128i)__builtin_ia32_selectb_128(
+      (__mmask16)__U, (__v16qi)_mm_cvts_ps_hf8(__A), (__v16qi)__W);
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __A
+///    to packed HF8 (8-bit) floating-point elements with saturation, and store
+///    the results in a 128-bit vector using zeromask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTPS2HF8S instruction.
+///
+/// \param __U
+///    A 8-bit mask indicating which elements to write (zero otherwise).
+/// \param __A
+///    A 128-bit vector of [4 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_maskz_cvts_ps_hf8(__mmask8 __U, __m128 __A) {
-  return (__m128i)__builtin_ia32_vcvtps2hf8s_128_mask(
-      (__v4sf)__A, (__v16qi)(__m128i)_mm_setzero_si128(), (__mmask8)__U);
-}
-
-// VCVTPS2HF8S - 256-bit
-
+  return (__m128i)__builtin_ia32_selectb_128((__mmask16)__U,
+                                             (__v16qi)_mm_cvts_ps_hf8(__A),
+                                             (__v16qi)_mm_setzero_si128());
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __A
+///    to packed HF8 (8-bit) floating-point elements with saturation, and store
+///    the results in a 128-bit vector.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTPS2HF8S instruction.
+///
+/// \param __A
+///    A 256-bit vector of [8 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_cvts_ps_hf8(__m256 __A) {
-  return (__m128i)__builtin_ia32_vcvtps2hf8s_256_mask(
-      (__v8sf)__A, (__v16qi)_mm_undefined_si128(), (__mmask8)-1);
-}
-
+  return (__m128i)__builtin_ia32_vcvtps2hf8s_256((__v8sf)__A);
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __A
+///    to packed HF8 (8-bit) floating-point elements with saturation, and store
+///    the results in a 128-bit vector using writemask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTPS2HF8S instruction.
+///
+/// \param __W
+///    A 128-bit vector of [16 x i8] used for writemask.
+/// \param __U
+///    A 8-bit mask indicating which elements to write.
+/// \param __A
+///    A 256-bit vector of [8 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_mask_cvts_ps_hf8(__m128i __W, __mmask8 __U, __m256 __A) {
-  return (__m128i)__builtin_ia32_vcvtps2hf8s_256_mask(
-      (__v8sf)__A, (__v16qi)(__m128i)__W, (__mmask8)__U);
-}
-
+  return (__m128i)__builtin_ia32_selectb_128(
+      (__mmask16)__U, (__v16qi)_mm256_cvts_ps_hf8(__A), (__v16qi)__W);
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __A
+///    to packed HF8 (8-bit) floating-point elements with saturation, and store
+///    the results in a 128-bit vector using zeromask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTPS2HF8S instruction.
+///
+/// \param __U
+///    A 8-bit mask indicating which elements to write (zero otherwise).
+/// \param __A
+///    A 256-bit vector of [8 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_maskz_cvts_ps_hf8(__mmask8 __U, __m256 __A) {
-  return (__m128i)__builtin_ia32_vcvtps2hf8s_256_mask(
-      (__v8sf)__A, (__v16qi)(__m128i)_mm_setzero_si128(), (__mmask8)__U);
-}
-
-// VCVTPS2HF8S - 512-bit
-
-static __inline__ __m128i __DEFAULT_FN_ATTRS512
-_mm512_cvts_ps_hf8(__m512 __A) {
-  return (__m128i)__builtin_ia32_vcvtps2hf8s_512_mask(
-      (__v16sf)__A, (__v16qi)_mm_undefined_si128(), (__mmask16)-1);
-}
-
-static __inline__ __m128i __DEFAULT_FN_ATTRS512
-_mm512_mask_cvts_ps_hf8(__m128i __W, __mmask16 __U, __m512 __A) {
-  return (__m128i)__builtin_ia32_vcvtps2hf8s_512_mask(
-      (__v16sf)__A, (__v16qi)(__m128i)__W, (__mmask16)__U);
-}
-
-static __inline__ __m128i __DEFAULT_FN_ATTRS512
-_mm512_maskz_cvts_ps_hf8(__mmask16 __U, __m512 __A) {
-  return (__m128i)__builtin_ia32_vcvtps2hf8s_512_mask(
-      (__v16sf)__A, (__v16qi)(__m128i)_mm_setzero_si128(), (__mmask16)__U);
-}
-
-// VCVTROPS2HF8 - 128-bit
-
+  return (__m128i)__builtin_ia32_selectb_128((__mmask16)__U,
+                                             (__v16qi)_mm256_cvts_ps_hf8(__A),
+                                             (__v16qi)_mm_setzero_si128());
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __A
+///    to packed HF8 (8-bit) floating-point elements using round-to-odd, and
+///    store the results in a 128-bit vector.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTROPS2HF8 instruction.
+///
+/// \param __A
+///    A 128-bit vector of [4 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_cvtrops_hf8(__m128 __A) {
-  return (__m128i)__builtin_ia32_vcvtrops2hf8_128_mask(
-      (__v4sf)__A, (__v16qi)_mm_undefined_si128(), (__mmask8)-1);
-}
-
+  return (__m128i)__builtin_ia32_vcvtrops2hf8_128((__v4sf)__A);
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __A
+///    to packed HF8 (8-bit) floating-point elements using round-to-odd, and
+///    store the results in a 128-bit vector using writemask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTROPS2HF8 instruction.
+///
+/// \param __W
+///    A 128-bit vector of [16 x i8] used for writemask.
+/// \param __U
+///    A 8-bit mask indicating which elements to write.
+/// \param __A
+///    A 128-bit vector of [4 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_mask_cvtrops_hf8(__m128i __W, __mmask8 __U, __m128 __A) {
-  return (__m128i)__builtin_ia32_vcvtrops2hf8_128_mask(
-      (__v4sf)__A, (__v16qi)(__m128i)__W, (__mmask8)__U);
-}
-
+  return (__m128i)__builtin_ia32_selectb_128(
+      (__mmask16)__U, (__v16qi)_mm_cvtrops_hf8(__A), (__v16qi)__W);
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __A
+///    to packed HF8 (8-bit) floating-point elements using round-to-odd, and
+///    store the results in a 128-bit vector using zeromask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTROPS2HF8 instruction.
+///
+/// \param __U
+///    A 8-bit mask indicating which elements to write (zero otherwise).
+/// \param __A
+///    A 128-bit vector of [4 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_maskz_cvtrops_hf8(__mmask8 __U, __m128 __A) {
-  return (__m128i)__builtin_ia32_vcvtrops2hf8_128_mask(
-      (__v4sf)__A, (__v16qi)(__m128i)_mm_setzero_si128(), (__mmask8)__U);
-}
-
-// VCVTROPS2HF8 - 256-bit
-
+  return (__m128i)__builtin_ia32_selectb_128((__mmask16)__U,
+                                             (__v16qi)_mm_cvtrops_hf8(__A),
+                                             (__v16qi)_mm_setzero_si128());
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __A
+///    to packed HF8 (8-bit) floating-point elements using round-to-odd, and
+///    store the results in a 128-bit vector.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTROPS2HF8 instruction.
+///
+/// \param __A
+///    A 256-bit vector of [8 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_cvtrops_hf8(__m256 __A) {
-  return (__m128i)__builtin_ia32_vcvtrops2hf8_256_mask(
-      (__v8sf)__A, (__v16qi)_mm_undefined_si128(), (__mmask8)-1);
-}
-
+  return (__m128i)__builtin_ia32_vcvtrops2hf8_256((__v8sf)__A);
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __A
+///    to packed HF8 (8-bit) floating-point elements using round-to-odd, and
+///    store the results in a 128-bit vector using writemask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTROPS2HF8 instruction.
+///
+/// \param __W
+///    A 128-bit vector of [16 x i8] used for writemask.
+/// \param __U
+///    A 8-bit mask indicating which elements to write.
+/// \param __A
+///    A 256-bit vector of [8 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_mask_cvtrops_hf8(__m128i __W, __mmask8 __U, __m256 __A) {
-  return (__m128i)__builtin_ia32_vcvtrops2hf8_256_mask(
-      (__v8sf)__A, (__v16qi)(__m128i)__W, (__mmask8)__U);
-}
-
+  return (__m128i)__builtin_ia32_selectb_128(
+      (__mmask16)__U, (__v16qi)_mm256_cvtrops_hf8(__A), (__v16qi)__W);
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __A
+///    to packed HF8 (8-bit) floating-point elements using round-to-odd, and
+///    store the results in a 128-bit vector using zeromask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTROPS2HF8 instruction.
+///
+/// \param __U
+///    A 8-bit mask indicating which elements to write (zero otherwise).
+/// \param __A
+///    A 256-bit vector of [8 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_maskz_cvtrops_hf8(__mmask8 __U, __m256 __A) {
-  return (__m128i)__builtin_ia32_vcvtrops2hf8_256_mask(
-      (__v8sf)__A, (__v16qi)(__m128i)_mm_setzero_si128(), (__mmask8)__U);
-}
-
-// VCVTROPS2HF8 - 512-bit
-
-static __inline__ __m128i __DEFAULT_FN_ATTRS512
-_mm512_cvtrops_hf8(__m512 __A) {
-  return (__m128i)__builtin_ia32_vcvtrops2hf8_512_mask(
-      (__v16sf)__A, (__v16qi)_mm_undefined_si128(), (__mmask16)-1);
-}
-
-static __inline__ __m128i __DEFAULT_FN_ATTRS512
-_mm512_mask_cvtrops_hf8(__m128i __W, __mmask16 __U, __m512 __A) {
-  return (__m128i)__builtin_ia32_vcvtrops2hf8_512_mask(
-      (__v16sf)__A, (__v16qi)(__m128i)__W, (__mmask16)__U);
-}
-
-static __inline__ __m128i __DEFAULT_FN_ATTRS512
-_mm512_maskz_cvtrops_hf8(__mmask16 __U, __m512 __A) {
-  return (__m128i)__builtin_ia32_vcvtrops2hf8_512_mask(
-      (__v16sf)__A, (__v16qi)(__m128i)_mm_setzero_si128(), (__mmask16)__U);
-}
-
-// VCVTROPS2HF8S - 128-bit
-
+  return (__m128i)__builtin_ia32_selectb_128((__mmask16)__U,
+                                             (__v16qi)_mm256_cvtrops_hf8(__A),
+                                             (__v16qi)_mm_setzero_si128());
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __A
+///    to packed HF8 (8-bit) floating-point elements using round-to-odd with
+///    saturation, and store the results in a 128-bit vector.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTROPS2HF8S instruction.
+///
+/// \param __A
+///    A 128-bit vector of [4 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_cvts_rops_hf8(__m128 __A) {
-  return (__m128i)__builtin_ia32_vcvtrops2hf8s_128_mask(
-      (__v4sf)__A, (__v16qi)_mm_undefined_si128(), (__mmask8)-1);
-}
-
+  return (__m128i)__builtin_ia32_vcvtrops2hf8s_128((__v4sf)__A);
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __A
+///    to packed HF8 (8-bit) floating-point elements using round-to-odd with
+///    saturation, and store the results using writemask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTROPS2HF8S instruction.
+///
+/// \param __W
+///    A 128-bit vector of [16 x i8] used for writemask.
+/// \param __U
+///    A 8-bit mask indicating which elements to write.
+/// \param __A
+///    A 128-bit vector of [4 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_mask_cvts_rops_hf8(__m128i __W, __mmask8 __U, __m128 __A) {
-  return (__m128i)__builtin_ia32_vcvtrops2hf8s_128_mask(
-      (__v4sf)__A, (__v16qi)(__m128i)__W, (__mmask8)__U);
-}
-
+  return (__m128i)__builtin_ia32_selectb_128(
+      (__mmask16)__U, (__v16qi)_mm_cvts_rops_hf8(__A), (__v16qi)__W);
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __A
+///    to packed HF8 (8-bit) floating-point elements using round-to-odd with
+///    saturation, and store the results using zeromask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTROPS2HF8S instruction.
+///
+/// \param __U
+///    A 8-bit mask indicating which elements to write (zero otherwise).
+/// \param __A
+///    A 128-bit vector of [4 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_maskz_cvts_rops_hf8(__mmask8 __U, __m128 __A) {
-  return (__m128i)__builtin_ia32_vcvtrops2hf8s_128_mask(
-      (__v4sf)__A, (__v16qi)(__m128i)_mm_setzero_si128(), (__mmask8)__U);
-}
-
-// VCVTROPS2HF8S - 256-bit
-
+  return (__m128i)__builtin_ia32_selectb_128((__mmask16)__U,
+                                             (__v16qi)_mm_cvts_rops_hf8(__A),
+                                             (__v16qi)_mm_setzero_si128());
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __A
+///    to packed HF8 (8-bit) floating-point elements using round-to-odd with
+///    saturation, and store the results in a 128-bit vector.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTROPS2HF8S instruction.
+///
+/// \param __A
+///    A 256-bit vector of [8 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_cvts_rops_hf8(__m256 __A) {
-  return (__m128i)__builtin_ia32_vcvtrops2hf8s_256_mask(
-      (__v8sf)__A, (__v16qi)_mm_undefined_si128(), (__mmask8)-1);
-}
-
+  return (__m128i)__builtin_ia32_vcvtrops2hf8s_256((__v8sf)__A);
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __A
+///    to packed HF8 (8-bit) floating-point elements using round-to-odd with
+///    saturation, and store the results using writemask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTROPS2HF8S instruction.
+///
+/// \param __W
+///    A 128-bit vector of [16 x i8] used for writemask.
+/// \param __U
+///    A 8-bit mask indicating which elements to write.
+/// \param __A
+///    A 256-bit vector of [8 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_mask_cvts_rops_hf8(__m128i __W, __mmask8 __U, __m256 __A) {
-  return (__m128i)__builtin_ia32_vcvtrops2hf8s_256_mask(
-      (__v8sf)__A, (__v16qi)(__m128i)__W, (__mmask8)__U);
-}
-
+  return (__m128i)__builtin_ia32_selectb_128(
+      (__mmask16)__U, (__v16qi)_mm256_cvts_rops_hf8(__A), (__v16qi)__W);
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __A
+///    to packed HF8 (8-bit) floating-point elements using round-to-odd with
+///    saturation, and store the results using zeromask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTROPS2HF8S instruction.
+///
+/// \param __U
+///    A 8-bit mask indicating which elements to write (zero otherwise).
+/// \param __A
+///    A 256-bit vector of [8 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_maskz_cvts_rops_hf8(__mmask8 __U, __m256 __A) {
-  return (__m128i)__builtin_ia32_vcvtrops2hf8s_256_mask(
-      (__v8sf)__A, (__v16qi)(__m128i)_mm_setzero_si128(), (__mmask8)__U);
-}
-
-// VCVTROPS2HF8S - 512-bit
-
-static __inline__ __m128i __DEFAULT_FN_ATTRS512
-_mm512_cvts_rops_hf8(__m512 __A) {
-  return (__m128i)__builtin_ia32_vcvtrops2hf8s_512_mask(
-      (__v16sf)__A, (__v16qi)_mm_undefined_si128(), (__mmask16)-1);
-}
-
-static __inline__ __m128i __DEFAULT_FN_ATTRS512
-_mm512_mask_cvts_rops_hf8(__m128i __W, __mmask16 __U, __m512 __A) {
-  return (__m128i)__builtin_ia32_vcvtrops2hf8s_512_mask(
-      (__v16sf)__A, (__v16qi)(__m128i)__W, (__mmask16)__U);
-}
-
-static __inline__ __m128i __DEFAULT_FN_ATTRS512
-_mm512_maskz_cvts_rops_hf8(__mmask16 __U, __m512 __A) {
-  return (__m128i)__builtin_ia32_vcvtrops2hf8s_512_mask(
-      (__v16sf)__A, (__v16qi)(__m128i)_mm_setzero_si128(), (__mmask16)__U);
-}
-
-//===----------------------------------------------------------------------===//
-// Group B: VCVTBIASPS2BF8 / VCVTBIASPS2BF8S / VCVTBIASPS2HF8 /
-//          VCVTBIASPS2HF8S
-// Convert packed single-precision with bias to FP8. Output is always __m128i.
-//===----------------------------------------------------------------------===//
-
-// VCVTBIASPS2BF8 - 128-bit
-
+  return (__m128i)__builtin_ia32_selectb_128((__mmask16)__U,
+                                             (__v16qi)_mm256_cvts_rops_hf8(__A),
+                                             (__v16qi)_mm_setzero_si128());
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __B
+///    to packed BF8 (8-bit) floating-point elements using bias values from
+///    \a __A, and store the results in a 128-bit vector.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBIASPS2BF8 instruction.
+///
+/// \param __A
+///    A 128-bit vector of [16 x i8] containing bias values.
+/// \param __B
+///    A 128-bit vector of [4 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_cvtbiasps_bf8(__m128i __A, __m128 __B) {
-  return (__m128i)__builtin_ia32_vcvtbiasps2bf8_128_mask(
-      (__v16qi)__A, (__v4sf)__B, (__v16qi)_mm_undefined_si128(), (__mmask8)-1);
-}
-
+  return (__m128i)__builtin_ia32_vcvtbiasps2bf8_128((__v16qi)__A, (__v4sf)__B);
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __B
+///    to packed BF8 (8-bit) floating-point elements using bias values from
+///    \a __A, and store the results using writemask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBIASPS2BF8 instruction.
+///
+/// \param __W
+///    A 128-bit vector of [16 x i8] used for writemask.
+/// \param __U
+///    A 8-bit mask indicating which elements to write.
+/// \param __A
+///    A 128-bit vector of [16 x i8] containing bias values.
+/// \param __B
+///    A 128-bit vector of [4 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_mask_cvtbiasps_bf8(__m128i __W, __mmask8 __U, __m128i __A, __m128 __B) {
-  return (__m128i)__builtin_ia32_vcvtbiasps2bf8_128_mask(
-      (__v16qi)__A, (__v4sf)__B, (__v16qi)(__m128i)__W, (__mmask8)__U);
-}
-
+  return (__m128i)__builtin_ia32_selectb_128(
+      (__mmask16)__U, (__v16qi)_mm_cvtbiasps_bf8(__A, __B), (__v16qi)__W);
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __B
+///    to packed BF8 (8-bit) floating-point elements using bias values from
+///    \a __A, and store the results using zeromask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBIASPS2BF8 instruction.
+///
+/// \param __U
+///    A 8-bit mask indicating which elements to write (zero otherwise).
+/// \param __A
+///    A 128-bit vector of [16 x i8] containing bias values.
+/// \param __B
+///    A 128-bit vector of [4 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_maskz_cvtbiasps_bf8(__mmask8 __U, __m128i __A, __m128 __B) {
-  return (__m128i)__builtin_ia32_vcvtbiasps2bf8_128_mask(
-      (__v16qi)__A, (__v4sf)__B, (__v16qi)(__m128i)_mm_setzero_si128(),
-      (__mmask8)__U);
-}
-
-// VCVTBIASPS2BF8 - 256-bit
-
+  return (__m128i)__builtin_ia32_selectb_128(
+      (__mmask16)__U, (__v16qi)_mm_cvtbiasps_bf8(__A, __B),
+      (__v16qi)_mm_setzero_si128());
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __B
+///    to packed BF8 (8-bit) floating-point elements using bias values from
+///    \a __A, and store the results in a 128-bit vector.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBIASPS2BF8 instruction.
+///
+/// \param __A
+///    A 256-bit vector of [32 x i8] containing bias values.
+/// \param __B
+///    A 256-bit vector of [8 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_cvtbiasps_bf8(__m256i __A, __m256 __B) {
-  return (__m128i)__builtin_ia32_vcvtbiasps2bf8_256_mask(
-      (__v32qi)__A, (__v8sf)__B, (__v16qi)_mm_undefined_si128(), (__mmask8)-1);
-}
-
+  return (__m128i)__builtin_ia32_vcvtbiasps2bf8_256((__v32qi)__A, (__v8sf)__B);
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __B
+///    to packed BF8 (8-bit) floating-point elements using bias values from
+///    \a __A, and store the results using writemask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBIASPS2BF8 instruction.
+///
+/// \param __W
+///    A 128-bit vector of [16 x i8] used for writemask.
+/// \param __U
+///    A 8-bit mask indicating which elements to write.
+/// \param __A
+///    A 256-bit vector of [32 x i8] containing bias values.
+/// \param __B
+///    A 256-bit vector of [8 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_mask_cvtbiasps_bf8(__m128i __W, __mmask8 __U, __m256i __A, __m256 __B) {
-  return (__m128i)__builtin_ia32_vcvtbiasps2bf8_256_mask(
-      (__v32qi)__A, (__v8sf)__B, (__v16qi)(__m128i)__W, (__mmask8)__U);
-}
-
+  return (__m128i)__builtin_ia32_selectb_128(
+      (__mmask16)__U, (__v16qi)_mm256_cvtbiasps_bf8(__A, __B), (__v16qi)__W);
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __B
+///    to packed BF8 (8-bit) floating-point elements using bias values from
+///    \a __A, and store the results using zeromask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBIASPS2BF8 instruction.
+///
+/// \param __U
+///    A 8-bit mask indicating which elements to write (zero otherwise).
+/// \param __A
+///    A 256-bit vector of [32 x i8] containing bias values.
+/// \param __B
+///    A 256-bit vector of [8 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_maskz_cvtbiasps_bf8(__mmask8 __U, __m256i __A, __m256 __B) {
-  return (__m128i)__builtin_ia32_vcvtbiasps2bf8_256_mask(
-      (__v32qi)__A, (__v8sf)__B, (__v16qi)(__m128i)_mm_setzero_si128(),
-      (__mmask8)__U);
-}
-
-// VCVTBIASPS2BF8 - 512-bit
-
-static __inline__ __m128i __DEFAULT_FN_ATTRS512
-_mm512_cvtbiasps_bf8(__m512i __A, __m512 __B) {
-  return (__m128i)__builtin_ia32_vcvtbiasps2bf8_512_mask(
-      (__v64qi)__A, (__v16sf)__B, (__v16qi)_mm_undefined_si128(),
-      (__mmask16)-1);
-}
-
-static __inline__ __m128i __DEFAULT_FN_ATTRS512
-_mm512_mask_cvtbiasps_bf8(__m128i __W, __mmask16 __U, __m512i __A,
-                          __m512 __B) {
-  return (__m128i)__builtin_ia32_vcvtbiasps2bf8_512_mask(
-      (__v64qi)__A, (__v16sf)__B, (__v16qi)(__m128i)__W, (__mmask16)__U);
-}
-
-static __inline__ __m128i __DEFAULT_FN_ATTRS512
-_mm512_maskz_cvtbiasps_bf8(__mmask16 __U, __m512i __A, __m512 __B) {
-  return (__m128i)__builtin_ia32_vcvtbiasps2bf8_512_mask(
-      (__v64qi)__A, (__v16sf)__B, (__v16qi)(__m128i)_mm_setzero_si128(),
-      (__mmask16)__U);
-}
-
-// VCVTBIASPS2BF8S - 128-bit
-
+  return (__m128i)__builtin_ia32_selectb_128(
+      (__mmask16)__U, (__v16qi)_mm256_cvtbiasps_bf8(__A, __B),
+      (__v16qi)_mm_setzero_si128());
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __B
+///    to packed BF8 (8-bit) floating-point elements with saturation using bias
+///    values from \a __A, and store the results in a 128-bit vector.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBIASPS2BF8S instruction.
+///
+/// \param __A
+///    A 128-bit vector of [16 x i8] containing bias values.
+/// \param __B
+///    A 128-bit vector of [4 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_cvts_biasps_bf8(__m128i __A, __m128 __B) {
-  return (__m128i)__builtin_ia32_vcvtbiasps2bf8s_128_mask(
-      (__v16qi)__A, (__v4sf)__B, (__v16qi)_mm_undefined_si128(), (__mmask8)-1);
-}
-
+  return (__m128i)__builtin_ia32_vcvtbiasps2bf8s_128((__v16qi)__A, (__v4sf)__B);
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __B
+///    to packed BF8 (8-bit) floating-point elements with saturation using bias
+///    values from \a __A, and store the results using writemask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBIASPS2BF8S instruction.
+///
+/// \param __W
+///    A 128-bit vector of [16 x i8] used for writemask.
+/// \param __U
+///    A 8-bit mask indicating which elements to write.
+/// \param __A
+///    A 128-bit vector of [16 x i8] containing bias values.
+/// \param __B
+///    A 128-bit vector of [4 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_mask_cvts_biasps_bf8(__m128i __W, __mmask8 __U, __m128i __A, __m128 __B) {
-  return (__m128i)__builtin_ia32_vcvtbiasps2bf8s_128_mask(
-      (__v16qi)__A, (__v4sf)__B, (__v16qi)(__m128i)__W, (__mmask8)__U);
-}
-
+  return (__m128i)__builtin_ia32_selectb_128(
+      (__mmask16)__U, (__v16qi)_mm_cvts_biasps_bf8(__A, __B), (__v16qi)__W);
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __B
+///    to packed BF8 (8-bit) floating-point elements with saturation using bias
+///    values from \a __A, and store the results using zeromask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBIASPS2BF8S instruction.
+///
+/// \param __U
+///    A 8-bit mask indicating which elements to write (zero otherwise).
+/// \param __A
+///    A 128-bit vector of [16 x i8] containing bias values.
+/// \param __B
+///    A 128-bit vector of [4 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_maskz_cvts_biasps_bf8(__mmask8 __U, __m128i __A, __m128 __B) {
-  return (__m128i)__builtin_ia32_vcvtbiasps2bf8s_128_mask(
-      (__v16qi)__A, (__v4sf)__B, (__v16qi)(__m128i)_mm_setzero_si128(),
-      (__mmask8)__U);
-}
-
-// VCVTBIASPS2BF8S - 256-bit
-
+  return (__m128i)__builtin_ia32_selectb_128(
+      (__mmask16)__U, (__v16qi)_mm_cvts_biasps_bf8(__A, __B),
+      (__v16qi)_mm_setzero_si128());
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __B
+///    to packed BF8 (8-bit) floating-point elements with saturation using bias
+///    values from \a __A, and store the results in a 128-bit vector.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBIASPS2BF8S instruction.
+///
+/// \param __A
+///    A 256-bit vector of [32 x i8] containing bias values.
+/// \param __B
+///    A 256-bit vector of [8 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_cvts_biasps_bf8(__m256i __A, __m256 __B) {
-  return (__m128i)__builtin_ia32_vcvtbiasps2bf8s_256_mask(
-      (__v32qi)__A, (__v8sf)__B, (__v16qi)_mm_undefined_si128(), (__mmask8)-1);
-}
-
+  return (__m128i)__builtin_ia32_vcvtbiasps2bf8s_256((__v32qi)__A, (__v8sf)__B);
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __B
+///    to packed BF8 (8-bit) floating-point elements with saturation using bias
+///    values from \a __A, and store the results using writemask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBIASPS2BF8S instruction.
+///
+/// \param __W
+///    A 128-bit vector of [16 x i8] used for writemask.
+/// \param __U
+///    A 8-bit mask indicating which elements to write.
+/// \param __A
+///    A 256-bit vector of [32 x i8] containing bias values.
+/// \param __B
+///    A 256-bit vector of [8 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256 _mm256_mask_cvts_biasps_bf8(
     __m128i __W, __mmask8 __U, __m256i __A, __m256 __B) {
-  return (__m128i)__builtin_ia32_vcvtbiasps2bf8s_256_mask(
-      (__v32qi)__A, (__v8sf)__B, (__v16qi)(__m128i)__W, (__mmask8)__U);
-}
-
+  return (__m128i)__builtin_ia32_selectb_128(
+      (__mmask16)__U, (__v16qi)_mm256_cvts_biasps_bf8(__A, __B), (__v16qi)__W);
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __B
+///    to packed BF8 (8-bit) floating-point elements with saturation using bias
+///    values from \a __A, and store the results using zeromask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBIASPS2BF8S instruction.
+///
+/// \param __U
+///    A 8-bit mask indicating which elements to write (zero otherwise).
+/// \param __A
+///    A 256-bit vector of [32 x i8] containing bias values.
+/// \param __B
+///    A 256-bit vector of [8 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_maskz_cvts_biasps_bf8(__mmask8 __U, __m256i __A, __m256 __B) {
-  return (__m128i)__builtin_ia32_vcvtbiasps2bf8s_256_mask(
-      (__v32qi)__A, (__v8sf)__B, (__v16qi)(__m128i)_mm_setzero_si128(),
-      (__mmask8)__U);
-}
-
-// VCVTBIASPS2BF8S - 512-bit
-
-static __inline__ __m128i __DEFAULT_FN_ATTRS512
-_mm512_cvts_biasps_bf8(__m512i __A, __m512 __B) {
-  return (__m128i)__builtin_ia32_vcvtbiasps2bf8s_512_mask(
-      (__v64qi)__A, (__v16sf)__B, (__v16qi)_mm_undefined_si128(),
-      (__mmask16)-1);
-}
-
-static __inline__ __m128i __DEFAULT_FN_ATTRS512
-_mm512_mask_cvts_biasps_bf8(__m128i __W, __mmask16 __U, __m512i __A,
-                           __m512 __B) {
-  return (__m128i)__builtin_ia32_vcvtbiasps2bf8s_512_mask(
-      (__v64qi)__A, (__v16sf)__B, (__v16qi)(__m128i)__W, (__mmask16)__U);
-}
-
-static __inline__ __m128i __DEFAULT_FN_ATTRS512
-_mm512_maskz_cvts_biasps_bf8(__mmask16 __U, __m512i __A, __m512 __B) {
-  return (__m128i)__builtin_ia32_vcvtbiasps2bf8s_512_mask(
-      (__v64qi)__A, (__v16sf)__B, (__v16qi)(__m128i)_mm_setzero_si128(),
-      (__mmask16)__U);
-}
-
-// VCVTBIASPS2HF8 - 128-bit
-
+  return (__m128i)__builtin_ia32_selectb_128(
+      (__mmask16)__U, (__v16qi)_mm256_cvts_biasps_bf8(__A, __B),
+      (__v16qi)_mm_setzero_si128());
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __B
+///    to packed HF8 (8-bit) floating-point elements using bias values from
+///    \a __A, and store the results in a 128-bit vector.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBIASPS2HF8 instruction.
+///
+/// \param __A
+///    A 128-bit vector of [16 x i8] containing bias values.
+/// \param __B
+///    A 128-bit vector of [4 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_cvtbiasps_hf8(__m128i __A, __m128 __B) {
-  return (__m128i)__builtin_ia32_vcvtbiasps2hf8_128_mask(
-      (__v16qi)__A, (__v4sf)__B, (__v16qi)_mm_undefined_si128(), (__mmask8)-1);
-}
-
+  return (__m128i)__builtin_ia32_vcvtbiasps2hf8_128((__v16qi)__A, (__v4sf)__B);
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __B
+///    to packed HF8 (8-bit) floating-point elements using bias values from
+///    \a __A, and store the results using writemask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBIASPS2HF8 instruction.
+///
+/// \param __W
+///    A 128-bit vector of [16 x i8] used for writemask.
+/// \param __U
+///    A 8-bit mask indicating which elements to write.
+/// \param __A
+///    A 128-bit vector of [16 x i8] containing bias values.
+/// \param __B
+///    A 128-bit vector of [4 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_mask_cvtbiasps_hf8(__m128i __W, __mmask8 __U, __m128i __A, __m128 __B) {
-  return (__m128i)__builtin_ia32_vcvtbiasps2hf8_128_mask(
-      (__v16qi)__A, (__v4sf)__B, (__v16qi)(__m128i)__W, (__mmask8)__U);
-}
-
+  return (__m128i)__builtin_ia32_selectb_128(
+      (__mmask16)__U, (__v16qi)_mm_cvtbiasps_hf8(__A, __B), (__v16qi)__W);
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __B
+///    to packed HF8 (8-bit) floating-point elements using bias values from
+///    \a __A, and store the results using zeromask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBIASPS2HF8 instruction.
+///
+/// \param __U
+///    A 8-bit mask indicating which elements to write (zero otherwise).
+/// \param __A
+///    A 128-bit vector of [16 x i8] containing bias values.
+/// \param __B
+///    A 128-bit vector of [4 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_maskz_cvtbiasps_hf8(__mmask8 __U, __m128i __A, __m128 __B) {
-  return (__m128i)__builtin_ia32_vcvtbiasps2hf8_128_mask(
-      (__v16qi)__A, (__v4sf)__B, (__v16qi)(__m128i)_mm_setzero_si128(),
-      (__mmask8)__U);
-}
-
-// VCVTBIASPS2HF8 - 256-bit
-
+  return (__m128i)__builtin_ia32_selectb_128(
+      (__mmask16)__U, (__v16qi)_mm_cvtbiasps_hf8(__A, __B),
+      (__v16qi)_mm_setzero_si128());
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __B
+///    to packed HF8 (8-bit) floating-point elements using bias values from
+///    \a __A, and store the results in a 128-bit vector.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBIASPS2HF8 instruction.
+///
+/// \param __A
+///    A 256-bit vector of [32 x i8] containing bias values.
+/// \param __B
+///    A 256-bit vector of [8 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_cvtbiasps_hf8(__m256i __A, __m256 __B) {
-  return (__m128i)__builtin_ia32_vcvtbiasps2hf8_256_mask(
-      (__v32qi)__A, (__v8sf)__B, (__v16qi)_mm_undefined_si128(), (__mmask8)-1);
-}
-
+  return (__m128i)__builtin_ia32_vcvtbiasps2hf8_256((__v32qi)__A, (__v8sf)__B);
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __B
+///    to packed HF8 (8-bit) floating-point elements using bias values from
+///    \a __A, and store the results using writemask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBIASPS2HF8 instruction.
+///
+/// \param __W
+///    A 128-bit vector of [16 x i8] used for writemask.
+/// \param __U
+///    A 8-bit mask indicating which elements to write.
+/// \param __A
+///    A 256-bit vector of [32 x i8] containing bias values.
+/// \param __B
+///    A 256-bit vector of [8 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_mask_cvtbiasps_hf8(__m128i __W, __mmask8 __U, __m256i __A, __m256 __B) {
-  return (__m128i)__builtin_ia32_vcvtbiasps2hf8_256_mask(
-      (__v32qi)__A, (__v8sf)__B, (__v16qi)(__m128i)__W, (__mmask8)__U);
-}
-
+  return (__m128i)__builtin_ia32_selectb_128(
+      (__mmask16)__U, (__v16qi)_mm256_cvtbiasps_hf8(__A, __B), (__v16qi)__W);
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __B
+///    to packed HF8 (8-bit) floating-point elements using bias values from
+///    \a __A, and store the results using zeromask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBIASPS2HF8 instruction.
+///
+/// \param __U
+///    A 8-bit mask indicating which elements to write (zero otherwise).
+/// \param __A
+///    A 256-bit vector of [32 x i8] containing bias values.
+/// \param __B
+///    A 256-bit vector of [8 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_maskz_cvtbiasps_hf8(__mmask8 __U, __m256i __A, __m256 __B) {
-  return (__m128i)__builtin_ia32_vcvtbiasps2hf8_256_mask(
-      (__v32qi)__A, (__v8sf)__B, (__v16qi)(__m128i)_mm_setzero_si128(),
-      (__mmask8)__U);
-}
-
-// VCVTBIASPS2HF8 - 512-bit
-
-static __inline__ __m128i __DEFAULT_FN_ATTRS512
-_mm512_cvtbiasps_hf8(__m512i __A, __m512 __B) {
-  return (__m128i)__builtin_ia32_vcvtbiasps2hf8_512_mask(
-      (__v64qi)__A, (__v16sf)__B, (__v16qi)_mm_undefined_si128(),
-      (__mmask16)-1);
-}
-
-static __inline__ __m128i __DEFAULT_FN_ATTRS512
-_mm512_mask_cvtbiasps_hf8(__m128i __W, __mmask16 __U, __m512i __A,
-                          __m512 __B) {
-  return (__m128i)__builtin_ia32_vcvtbiasps2hf8_512_mask(
-      (__v64qi)__A, (__v16sf)__B, (__v16qi)(__m128i)__W, (__mmask16)__U);
-}
-
-static __inline__ __m128i __DEFAULT_FN_ATTRS512
-_mm512_maskz_cvtbiasps_hf8(__mmask16 __U, __m512i __A, __m512 __B) {
-  return (__m128i)__builtin_ia32_vcvtbiasps2hf8_512_mask(
-      (__v64qi)__A, (__v16sf)__B, (__v16qi)(__m128i)_mm_setzero_si128(),
-      (__mmask16)__U);
-}
-
-// VCVTBIASPS2HF8S - 128-bit
-
+  return (__m128i)__builtin_ia32_selectb_128(
+      (__mmask16)__U, (__v16qi)_mm256_cvtbiasps_hf8(__A, __B),
+      (__v16qi)_mm_setzero_si128());
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __B
+///    to packed HF8 (8-bit) floating-point elements with saturation using bias
+///    values from \a __A, and store the results in a 128-bit vector.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBIASPS2HF8S instruction.
+///
+/// \param __A
+///    A 128-bit vector of [16 x i8] containing bias values.
+/// \param __B
+///    A 128-bit vector of [4 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_cvts_biasps_hf8(__m128i __A, __m128 __B) {
-  return (__m128i)__builtin_ia32_vcvtbiasps2hf8s_128_mask(
-      (__v16qi)__A, (__v4sf)__B, (__v16qi)_mm_undefined_si128(), (__mmask8)-1);
-}
-
+  return (__m128i)__builtin_ia32_vcvtbiasps2hf8s_128((__v16qi)__A, (__v4sf)__B);
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __B
+///    to packed HF8 (8-bit) floating-point elements with saturation using bias
+///    values from \a __A, and store the results using writemask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBIASPS2HF8S instruction.
+///
+/// \param __W
+///    A 128-bit vector of [16 x i8] used for writemask.
+/// \param __U
+///    A 8-bit mask indicating which elements to write.
+/// \param __A
+///    A 128-bit vector of [16 x i8] containing bias values.
+/// \param __B
+///    A 128-bit vector of [4 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_mask_cvts_biasps_hf8(__m128i __W, __mmask8 __U, __m128i __A, __m128 __B) {
-  return (__m128i)__builtin_ia32_vcvtbiasps2hf8s_128_mask(
-      (__v16qi)__A, (__v4sf)__B, (__v16qi)(__m128i)__W, (__mmask8)__U);
-}
-
+  return (__m128i)__builtin_ia32_selectb_128(
+      (__mmask16)__U, (__v16qi)_mm_cvts_biasps_hf8(__A, __B), (__v16qi)__W);
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __B
+///    to packed HF8 (8-bit) floating-point elements with saturation using bias
+///    values from \a __A, and store the results using zeromask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBIASPS2HF8S instruction.
+///
+/// \param __U
+///    A 8-bit mask indicating which elements to write (zero otherwise).
+/// \param __A
+///    A 128-bit vector of [16 x i8] containing bias values.
+/// \param __B
+///    A 128-bit vector of [4 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_maskz_cvts_biasps_hf8(__mmask8 __U, __m128i __A, __m128 __B) {
-  return (__m128i)__builtin_ia32_vcvtbiasps2hf8s_128_mask(
-      (__v16qi)__A, (__v4sf)__B, (__v16qi)(__m128i)_mm_setzero_si128(),
-      (__mmask8)__U);
-}
-
-// VCVTBIASPS2HF8S - 256-bit
-
+  return (__m128i)__builtin_ia32_selectb_128(
+      (__mmask16)__U, (__v16qi)_mm_cvts_biasps_hf8(__A, __B),
+      (__v16qi)_mm_setzero_si128());
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __B
+///    to packed HF8 (8-bit) floating-point elements with saturation using bias
+///    values from \a __A, and store the results in a 128-bit vector.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBIASPS2HF8S instruction.
+///
+/// \param __A
+///    A 256-bit vector of [32 x i8] containing bias values.
+/// \param __B
+///    A 256-bit vector of [8 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_cvts_biasps_hf8(__m256i __A, __m256 __B) {
-  return (__m128i)__builtin_ia32_vcvtbiasps2hf8s_256_mask(
-      (__v32qi)__A, (__v8sf)__B, (__v16qi)_mm_undefined_si128(), (__mmask8)-1);
-}
-
+  return (__m128i)__builtin_ia32_vcvtbiasps2hf8s_256((__v32qi)__A, (__v8sf)__B);
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __B
+///    to packed HF8 (8-bit) floating-point elements with saturation using bias
+///    values from \a __A, and store the results using writemask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBIASPS2HF8S instruction.
+///
+/// \param __W
+///    A 128-bit vector of [16 x i8] used for writemask.
+/// \param __U
+///    A 8-bit mask indicating which elements to write.
+/// \param __A
+///    A 256-bit vector of [32 x i8] containing bias values.
+/// \param __B
+///    A 256-bit vector of [8 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256 _mm256_mask_cvts_biasps_hf8(
     __m128i __W, __mmask8 __U, __m256i __A, __m256 __B) {
-  return (__m128i)__builtin_ia32_vcvtbiasps2hf8s_256_mask(
-      (__v32qi)__A, (__v8sf)__B, (__v16qi)(__m128i)__W, (__mmask8)__U);
-}
-
+  return (__m128i)__builtin_ia32_selectb_128(
+      (__mmask16)__U, (__v16qi)_mm256_cvts_biasps_hf8(__A, __B), (__v16qi)__W);
+}
+
+/// Convert packed single-precision (32-bit) floating-point elements in \a __B
+///    to packed HF8 (8-bit) floating-point elements with saturation using bias
+///    values from \a __A, and store the results using zeromask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBIASPS2HF8S instruction.
+///
+/// \param __U
+///    A 8-bit mask indicating which elements to write (zero otherwise).
+/// \param __A
+///    A 256-bit vector of [32 x i8] containing bias values.
+/// \param __B
+///    A 256-bit vector of [8 x float].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_maskz_cvts_biasps_hf8(__mmask8 __U, __m256i __A, __m256 __B) {
-  return (__m128i)__builtin_ia32_vcvtbiasps2hf8s_256_mask(
-      (__v32qi)__A, (__v8sf)__B, (__v16qi)(__m128i)_mm_setzero_si128(),
-      (__mmask8)__U);
-}
-
-// VCVTBIASPS2HF8S - 512-bit
-
-static __inline__ __m128i __DEFAULT_FN_ATTRS512
-_mm512_cvts_biasps_hf8(__m512i __A, __m512 __B) {
-  return (__m128i)__builtin_ia32_vcvtbiasps2hf8s_512_mask(
-      (__v64qi)__A, (__v16sf)__B, (__v16qi)_mm_undefined_si128(),
-      (__mmask16)-1);
-}
-
-static __inline__ __m128i __DEFAULT_FN_ATTRS512
-_mm512_mask_cvts_biasps_hf8(__m128i __W, __mmask16 __U, __m512i __A,
-                           __m512 __B) {
-  return (__m128i)__builtin_ia32_vcvtbiasps2hf8s_512_mask(
-      (__v64qi)__A, (__v16sf)__B, (__v16qi)(__m128i)__W, (__mmask16)__U);
-}
-
-static __inline__ __m128i __DEFAULT_FN_ATTRS512
-_mm512_maskz_cvts_biasps_hf8(__mmask16 __U, __m512i __A, __m512 __B) {
-  return (__m128i)__builtin_ia32_vcvtbiasps2hf8s_512_mask(
-      (__v64qi)__A, (__v16sf)__B, (__v16qi)(__m128i)_mm_setzero_si128(),
-      (__mmask16)__U);
-}
-
-//===----------------------------------------------------------------------===//
-// Group C: VCVTBF82PS / VCVTHF82PS
-// Convert packed FP8 to single-precision. Input is __m128i, output varies.
-//===----------------------------------------------------------------------===//
-
-// VCVTBF82PS - 128-bit
-
+  return (__m128i)__builtin_ia32_selectb_128(
+      (__mmask16)__U, (__v16qi)_mm256_cvts_biasps_hf8(__A, __B),
+      (__v16qi)_mm_setzero_si128());
+}
+
+/// Convert packed BF8 (8-bit) floating-point elements in \a __A to packed
+///    single-precision (32-bit) floating-point elements, and store the results
+///    in a 128-bit vector.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBF82PS instruction.
+///
+/// \param __A
+///    A 128-bit vector of [16 x i8] containing BF8 values.
+/// \returns
+///    A 128-bit vector of [4 x float] containing the converted values.
 static __inline__ __m128 __DEFAULT_FN_ATTRS128
 _mm_cvtbf8_ps(__m128i __A) {
-  return (__m128)__builtin_ia32_vcvtbf8_2ps128_mask(
-      (__v16qi)__A, (__v4sf)_mm_undefined_ps(), (__mmask8)-1);
-}
-
+  return (__m128)__builtin_ia32_vcvtbf8_2ps128((__v16qi)__A);
+}
+
+/// Convert packed BF8 (8-bit) floating-point elements in \a __A to packed
+///    single-precision (32-bit) floating-point elements, and store the results
+///    using writemask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBF82PS instruction.
+///
+/// \param __W
+///    A 128-bit vector of [4 x float] used for writemask.
+/// \param __U
+///    A 8-bit mask indicating which elements to write.
+/// \param __A
+///    A 128-bit vector of [16 x i8] containing BF8 values.
+/// \returns
+///    A 128-bit vector of [4 x float] containing the converted values.
 static __inline__ __m128 __DEFAULT_FN_ATTRS128
 _mm_mask_cvtbf8_ps(__m128 __W, __mmask8 __U, __m128i __A) {
-  return (__m128)__builtin_ia32_vcvtbf8_2ps128_mask(
-      (__v16qi)__A, (__v4sf)__W, (__mmask8)__U);
-}
-
+  return (__m128)__builtin_ia32_selectps_128(
+      (__mmask8)__U, (__v4sf)_mm_cvtbf8_ps(__A), (__v4sf)__W);
+}
+
+/// Convert packed BF8 (8-bit) floating-point elements in \a __A to packed
+///    single-precision (32-bit) floating-point elements, and store the results
+///    using zeromask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBF82PS instruction.
+///
+/// \param __U
+///    A 8-bit mask indicating which elements to write (zero otherwise).
+/// \param __A
+///    A 128-bit vector of [16 x i8] containing BF8 values.
+/// \returns
+///    A 128-bit vector of [4 x float] containing the converted values.
 static __inline__ __m128 __DEFAULT_FN_ATTRS128
 _mm_maskz_cvtbf8_ps(__mmask8 __U, __m128i __A) {
-  return (__m128)__builtin_ia32_vcvtbf8_2ps128_mask(
-      (__v16qi)__A, (__v4sf)_mm_setzero_ps(), (__mmask8)__U);
-}
-
-// VCVTBF82PS - 256-bit
-
+  return (__m128)__builtin_ia32_selectps_128(
+      (__mmask8)__U, (__v4sf)_mm_cvtbf8_ps(__A), (__v4sf)_mm_setzero_ps());
+}
+
+/// Convert packed BF8 (8-bit) floating-point elements in \a __A to packed
+///    single-precision (32-bit) floating-point elements, and store the results
+///    in a 256-bit vector.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBF82PS instruction.
+///
+/// \param __A
+///    A 128-bit vector of [16 x i8] containing BF8 values.
+/// \returns
+///    A 256-bit vector of [8 x float] containing the converted values.
 static __inline__ __m256 __DEFAULT_FN_ATTRS256
 _mm256_cvtbf8_ps(__m128i __A) {
-  return (__m256)__builtin_ia32_vcvtbf8_2ps256_mask(
-      (__v16qi)__A, (__v8sf)_mm256_undefined_ps(), (__mmask8)-1);
-}
-
+  return (__m256)__builtin_ia32_vcvtbf8_2ps256((__v16qi)__A);
+}
+
+/// Convert packed BF8 (8-bit) floating-point elements in \a __A to packed
+///    single-precision (32-bit) floating-point elements, and store the results
+///    using writemask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBF82PS instruction.
+///
+/// \param __W
+///    A 256-bit vector of [8 x float] used for writemask.
+/// \param __U
+///    A 8-bit mask indicating which elements to write.
+/// \param __A
+///    A 128-bit vector of [16 x i8] containing BF8 values.
+/// \returns
+///    A 256-bit vector of [8 x float] containing the converted values.
 static __inline__ __m256 __DEFAULT_FN_ATTRS256
 _mm256_mask_cvtbf8_ps(__m256 __W, __mmask8 __U, __m128i __A) {
-  return (__m256)__builtin_ia32_vcvtbf8_2ps256_mask(
-      (__v16qi)__A, (__v8sf)__W, (__mmask8)__U);
-}
-
+  return (__m256)__builtin_ia32_selectps_256(
+      (__mmask8)__U, (__v8sf)_mm256_cvtbf8_ps(__A), (__v8sf)__W);
+}
+
+/// Convert packed BF8 (8-bit) floating-point elements in \a __A to packed
+///    single-precision (32-bit) floating-point elements, and store the results
+///    using zeromask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBF82PS instruction.
+///
+/// \param __U
+///    A 8-bit mask indicating which elements to write (zero otherwise).
+/// \param __A
+///    A 128-bit vector of [16 x i8] containing BF8 values.
+/// \returns
+///    A 256-bit vector of [8 x float] containing the converted values.
 static __inline__ __m256 __DEFAULT_FN_ATTRS256
 _mm256_maskz_cvtbf8_ps(__mmask8 __U, __m128i __A) {
-  return (__m256)__builtin_ia32_vcvtbf8_2ps256_mask(
-      (__v16qi)__A, (__v8sf)_mm256_setzero_ps(), (__mmask8)__U);
-}
-
-// VCVTBF82PS - 512-bit
-
-static __inline__ __m512 __DEFAULT_FN_ATTRS512
-_mm512_cvtbf8_ps(__m128i __A) {
-  return (__m512)__builtin_ia32_vcvtbf8_2ps512_mask(
-      (__v16qi)__A, (__v16sf)_mm512_undefined_ps(), (__mmask16)-1);
-}
-
-static __inline__ __m512 __DEFAULT_FN_ATTRS512
-_mm512_mask_cvtbf8_ps(__m512 __W, __mmask16 __U, __m128i __A) {
-  return (__m512)__builtin_ia32_vcvtbf8_2ps512_mask(
-      (__v16qi)__A, (__v16sf)__W, (__mmask16)__U);
-}
-
-static __inline__ __m512 __DEFAULT_FN_ATTRS512
-_mm512_maskz_cvtbf8_ps(__mmask16 __U, __m128i __A) {
-  return (__m512)__builtin_ia32_vcvtbf8_2ps512_mask(
-      (__v16qi)__A, (__v16sf)_mm512_setzero_ps(), (__mmask16)__U);
-}
-
-// VCVTHF82PS - 128-bit
-
+  return (__m256)__builtin_ia32_selectps_256((__mmask8)__U,
+                                             (__v8sf)_mm256_cvtbf8_ps(__A),
+                                             (__v8sf)_mm256_setzero_ps());
+}
+
+/// Convert packed HF8 (8-bit) floating-point elements in \a __A to packed
+///    single-precision (32-bit) floating-point elements, and store the results
+///    in a 128-bit vector.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTHF82PS instruction.
+///
+/// \param __A
+///    A 128-bit vector of [16 x i8] containing HF8 values.
+/// \returns
+///    A 128-bit vector of [4 x float] containing the converted values.
 static __inline__ __m128 __DEFAULT_FN_ATTRS128
 _mm_cvthf8_ps(__m128i __A) {
-  return (__m128)__builtin_ia32_vcvthf8_2ps128_mask(
-      (__v16qi)__A, (__v4sf)_mm_undefined_ps(), (__mmask8)-1);
-}
-
+  return (__m128)__builtin_ia32_vcvthf8_2ps128((__v16qi)__A);
+}
+
+/// Convert packed HF8 (8-bit) floating-point elements in \a __A to packed
+///    single-precision (32-bit) floating-point elements, and store the results
+///    using writemask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTHF82PS instruction.
+///
+/// \param __W
+///    A 128-bit vector of [4 x float] used for writemask.
+/// \param __U
+///    A 8-bit mask indicating which elements to write.
+/// \param __A
+///    A 128-bit vector of [16 x i8] containing HF8 values.
+/// \returns
+///    A 128-bit vector of [4 x float] containing the converted values.
 static __inline__ __m128 __DEFAULT_FN_ATTRS128
 _mm_mask_cvthf8_ps(__m128 __W, __mmask8 __U, __m128i __A) {
-  return (__m128)__builtin_ia32_vcvthf8_2ps128_mask(
-      (__v16qi)__A, (__v4sf)__W, (__mmask8)__U);
-}
-
+  return (__m128)__builtin_ia32_selectps_128(
+      (__mmask8)__U, (__v4sf)_mm_cvthf8_ps(__A), (__v4sf)__W);
+}
+
+/// Convert packed HF8 (8-bit) floating-point elements in \a __A to packed
+///    single-precision (32-bit) floating-point elements, and store the results
+///    using zeromask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTHF82PS instruction.
+///
+/// \param __U
+///    A 8-bit mask indicating which elements to write (zero otherwise).
+/// \param __A
+///    A 128-bit vector of [16 x i8] containing HF8 values.
+/// \returns
+///    A 128-bit vector of [4 x float] containing the converted values.
 static __inline__ __m128 __DEFAULT_FN_ATTRS128
 _mm_maskz_cvthf8_ps(__mmask8 __U, __m128i __A) {
-  return (__m128)__builtin_ia32_vcvthf8_2ps128_mask(
-      (__v16qi)__A, (__v4sf)_mm_setzero_ps(), (__mmask8)__U);
-}
-
-// VCVTHF82PS - 256-bit
-
+  return (__m128)__builtin_ia32_selectps_128(
+      (__mmask8)__U, (__v4sf)_mm_cvthf8_ps(__A), (__v4sf)_mm_setzero_ps());
+}
+
+/// Convert packed HF8 (8-bit) floating-point elements in \a __A to packed
+///    single-precision (32-bit) floating-point elements, and store the results
+///    in a 256-bit vector.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTHF82PS instruction.
+///
+/// \param __A
+///    A 128-bit vector of [16 x i8] containing HF8 values.
+/// \returns
+///    A 256-bit vector of [8 x float] containing the converted values.
 static __inline__ __m256 __DEFAULT_FN_ATTRS256
 _mm256_cvthf8_ps(__m128i __A) {
-  return (__m256)__builtin_ia32_vcvthf8_2ps256_mask(
-      (__v16qi)__A, (__v8sf)_mm256_undefined_ps(), (__mmask8)-1);
-}
-
+  return (__m256)__builtin_ia32_vcvthf8_2ps256((__v16qi)__A);
+}
+
+/// Convert packed HF8 (8-bit) floating-point elements in \a __A to packed
+///    single-precision (32-bit) floating-point elements, and store the results
+///    using writemask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTHF82PS instruction.
+///
+/// \param __W
+///    A 256-bit vector of [8 x float] used for writemask.
+/// \param __U
+///    A 8-bit mask indicating which elements to write.
+/// \param __A
+///    A 128-bit vector of [16 x i8] containing HF8 values.
+/// \returns
+///    A 256-bit vector of [8 x float] containing the converted values.
 static __inline__ __m256 __DEFAULT_FN_ATTRS256
 _mm256_mask_cvthf8_ps(__m256 __W, __mmask8 __U, __m128i __A) {
-  return (__m256)__builtin_ia32_vcvthf8_2ps256_mask(
-      (__v16qi)__A, (__v8sf)__W, (__mmask8)__U);
-}
-
+  return (__m256)__builtin_ia32_selectps_256(
+      (__mmask8)__U, (__v8sf)_mm256_cvthf8_ps(__A), (__v8sf)__W);
+}
+
+/// Convert packed HF8 (8-bit) floating-point elements in \a __A to packed
+///    single-precision (32-bit) floating-point elements, and store the results
+///    using zeromask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTHF82PS instruction.
+///
+/// \param __U
+///    A 8-bit mask indicating which elements to write (zero otherwise).
+/// \param __A
+///    A 128-bit vector of [16 x i8] containing HF8 values.
+/// \returns
+///    A 256-bit vector of [8 x float] containing the converted values.
 static __inline__ __m256 __DEFAULT_FN_ATTRS256
 _mm256_maskz_cvthf8_ps(__mmask8 __U, __m128i __A) {
-  return (__m256)__builtin_ia32_vcvthf8_2ps256_mask(
-      (__v16qi)__A, (__v8sf)_mm256_setzero_ps(), (__mmask8)__U);
-}
-
-// VCVTHF82PS - 512-bit
-
-static __inline__ __m512 __DEFAULT_FN_ATTRS512
-_mm512_cvthf8_ps(__m128i __A) {
-  return (__m512)__builtin_ia32_vcvthf8_2ps512_mask(
-      (__v16qi)__A, (__v16sf)_mm512_undefined_ps(), (__mmask16)-1);
-}
-
-static __inline__ __m512 __DEFAULT_FN_ATTRS512
-_mm512_mask_cvthf8_ps(__m512 __W, __mmask16 __U, __m128i __A) {
-  return (__m512)__builtin_ia32_vcvthf8_2ps512_mask(
-      (__v16qi)__A, (__v16sf)__W, (__mmask16)__U);
-}
-
-static __inline__ __m512 __DEFAULT_FN_ATTRS512
-_mm512_maskz_cvthf8_ps(__mmask16 __U, __m128i __A) {
-  return (__m512)__builtin_ia32_vcvthf8_2ps512_mask(
-      (__v16qi)__A, (__v16sf)_mm512_setzero_ps(), (__mmask16)__U);
-}
-
-//===----------------------------------------------------------------------===//
-// Group E: VCVTBF82BF6S / VCVTHF82HF6S
-// Same-size reg-only conversions (no masking support)
-//===----------------------------------------------------------------------===//
-
-// VCVTBF82BF6S
-
+  return (__m256)__builtin_ia32_selectps_256((__mmask8)__U,
+                                             (__v8sf)_mm256_cvthf8_ps(__A),
+                                             (__v8sf)_mm256_setzero_ps());
+}
+
+/// Convert packed BF8 (8-bit) floating-point elements in \a __A to packed
+///    BF6 (6-bit) floating-point elements with saturation, and store the
+///    results in a 128-bit vector.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBF82BF6S instruction.
+///
+/// \param __A
+///    A 128-bit vector of [16 x i8] containing BF8 values.
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted BF6 values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_cvtbf8_bf6s(__m128i __A) {
   return (__m128i)__builtin_ia32_vcvtbf82bf6s128((__v16qi)__A);
 }
 
+/// Convert packed BF8 (8-bit) floating-point elements in \a __A to packed
+///    BF6 (6-bit) floating-point elements with saturation, and store the
+///    results in a 256-bit vector.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBF82BF6S instruction.
+///
+/// \param __A
+///    A 256-bit vector of [32 x i8] containing BF8 values.
+/// \returns
+///    A 256-bit vector of [32 x i8] containing the converted BF6 values.
 static __inline__ __m256i __DEFAULT_FN_ATTRS256
 _mm256_cvtbf8_bf6s(__m256i __A) {
   return (__m256i)__builtin_ia32_vcvtbf82bf6s256((__v32qi)__A);
 }
 
-static __inline__ __m512i __DEFAULT_FN_ATTRS512
-_mm512_cvtbf8_bf6s(__m512i __A) {
-  return (__m512i)__builtin_ia32_vcvtbf82bf6s512((__v64qi)__A);
-}
-
-// VCVTHF82HF6S
-
+/// Convert packed HF8 (8-bit) floating-point elements in \a __A to packed
+///    HF6 (6-bit) floating-point elements with saturation, and store the
+///    results in a 128-bit vector.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTHF82HF6S instruction.
+///
+/// \param __A
+///    A 128-bit vector of [16 x i8] containing HF8 values.
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted HF6 values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_cvthf8_hf6s(__m128i __A) {
   return (__m128i)__builtin_ia32_vcvthf82hf6s128((__v16qi)__A);
 }
 
+/// Convert packed HF8 (8-bit) floating-point elements in \a __A to packed
+///    HF6 (6-bit) floating-point elements with saturation, and store the
+///    results in a 256-bit vector.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTHF82HF6S instruction.
+///
+/// \param __A
+///    A 256-bit vector of [32 x i8] containing HF8 values.
+/// \returns
+///    A 256-bit vector of [32 x i8] containing the converted HF6 values.
 static __inline__ __m256i __DEFAULT_FN_ATTRS256
 _mm256_cvthf8_hf6s(__m256i __A) {
   return (__m256i)__builtin_ia32_vcvthf82hf6s256((__v32qi)__A);
 }
 
-static __inline__ __m512i __DEFAULT_FN_ATTRS512
-_mm512_cvthf8_hf6s(__m512i __A) {
-  return (__m512i)__builtin_ia32_vcvthf82hf6s512((__v64qi)__A);
-}
-
-//===----------------------------------------------------------------------===//
-// Group F: VCVTBF42HF8 / VCVTBF62HF8 / VCVTHF62HF8
-// Expanding/same-size conversions with masking support
-//===----------------------------------------------------------------------===//
-
-// VCVTBF42HF8 - 128-bit
-
+/// Convert packed BF8 (8-bit) floating-point elements in \a __A to packed
+///    BF4 (4-bit) floating-point elements with saturation, and store the
+///    results in a 128-bit vector.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBF82BF4S instruction.
+///
+/// \param __A
+///    A 128-bit vector of [16 x i8] containing BF8 values.
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted BF4 values
+///    (lower 8 bytes used).
+static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvtbf8_bf4s(__m128i __A) {
+  return (__m128i)__builtin_ia32_vcvtbf82bf4s128((__v16qi)__A);
+}
+
+/// Convert packed BF8 (8-bit) floating-point elements in \a __A to packed
+///    BF4 (4-bit) floating-point elements with saturation, and store the
+///    results in a 128-bit vector.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBF82BF4S instruction.
+///
+/// \param __A
+///    A 256-bit vector of [32 x i8] containing BF8 values.
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted BF4 values.
+static __inline__ __m128i __DEFAULT_FN_ATTRS256
+_mm256_cvtbf8_bf4s(__m256i __A) {
+  return (__m128i)__builtin_ia32_vcvtbf82bf4s256((__v32qi)__A);
+}
+
+/// Convert packed HF8 (8-bit) floating-point elements in \a __A to packed
+///    BF4 (4-bit) floating-point elements with saturation, and store the
+///    results in a 128-bit vector.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTHF82BF4S instruction.
+///
+/// \param __A
+///    A 128-bit vector of [16 x i8] containing HF8 values.
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted BF4 values
+///    (lower 8 bytes used).
+static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvthf8_bf4s(__m128i __A) {
+  return (__m128i)__builtin_ia32_vcvthf82bf4s128((__v16qi)__A);
+}
+
+/// Convert packed HF8 (8-bit) floating-point elements in \a __A to packed
+///    BF4 (4-bit) floating-point elements with saturation, and store the
+///    results in a 128-bit vector.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTHF82BF4S instruction.
+///
+/// \param __A
+///    A 256-bit vector of [32 x i8] containing HF8 values.
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted BF4 values.
+static __inline__ __m128i __DEFAULT_FN_ATTRS256
+_mm256_cvthf8_bf4s(__m256i __A) {
+  return (__m128i)__builtin_ia32_vcvthf82bf4s256((__v32qi)__A);
+}
+
+/// Convert packed BF8 (8-bit) floating-point elements in \a __A to packed
+///    BF4 (4-bit) floating-point elements with saturation, and store the
+///    results to memory at \a __P.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBF82BF4S instruction.
+///
+/// \param __P
+///    A pointer to a 64-bit memory location. The address does not need to be
+///    aligned.
+/// \param __A
+///    A 128-bit vector of [16 x i8] containing BF8 values.
+static __inline__ void __DEFAULT_FN_ATTRS128
+_mm_cvtbf8_bf4s_storeu(void *__P, __m128i __A) {
+  __builtin_ia32_vcvtbf82bf4s128mem(__P, (__v16qi)__A);
+}
+
+/// Convert packed BF8 (8-bit) floating-point elements in \a __A to packed
+///    BF4 (4-bit) floating-point elements with saturation, and store the
+///    results to memory at \a __P.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBF82BF4S instruction.
+///
+/// \param __P
+///    A pointer to a 128-bit memory location. The address does not need to be
+///    aligned.
+/// \param __A
+///    A 256-bit vector of [32 x i8] containing BF8 values.
+static __inline__ void __DEFAULT_FN_ATTRS256
+_mm256_cvtbf8_bf4s_storeu(void *__P, __m256i __A) {
+  __builtin_ia32_vcvtbf82bf4s256mem(__P, (__v32qi)__A);
+}
+
+/// Convert packed HF8 (8-bit) floating-point elements in \a __A to packed
+///    BF4 (4-bit) floating-point elements with saturation, and store the
+///    results to memory at \a __P.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTHF82BF4S instruction.
+///
+/// \param __P
+///    A pointer to a 64-bit memory location. The address does not need to be
+///    aligned.
+/// \param __A
+///    A 128-bit vector of [16 x i8] containing HF8 values.
+static __inline__ void __DEFAULT_FN_ATTRS128
+_mm_cvthf8_bf4s_storeu(void *__P, __m128i __A) {
+  __builtin_ia32_vcvthf82bf4s128mem(__P, (__v16qi)__A);
+}
+
+/// Convert packed HF8 (8-bit) floating-point elements in \a __A to packed
+///    BF4 (4-bit) floating-point elements with saturation, and store the
+///    results to memory at \a __P.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTHF82BF4S instruction.
+///
+/// \param __P
+///    A pointer to a 128-bit memory location. The address does not need to be
+///    aligned.
+/// \param __A
+///    A 256-bit vector of [32 x i8] containing HF8 values.
+static __inline__ void __DEFAULT_FN_ATTRS256
+_mm256_cvthf8_bf4s_storeu(void *__P, __m256i __A) {
+  __builtin_ia32_vcvthf82bf4s256mem(__P, (__v32qi)__A);
+}
+
+/// Convert packed BF4 (4-bit) floating-point elements in \a __A to packed
+///    HF8 (8-bit) floating-point elements, and store the results in a 128-bit
+///    vector.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBF42HF8 instruction.
+///
+/// \param __A
+///    A 128-bit vector of [16 x i8] containing BF4 values.
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted HF8 values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_cvtbf4_hf8(__m128i __A) {
   return (__m128i)__builtin_ia32_vcvtbf42hf8128((__v16qi)__A);
 }
 
+/// Convert packed BF4 (4-bit) floating-point elements in \a __A to packed
+///    HF8 (8-bit) floating-point elements, and store the results using
+///    writemask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBF42HF8 instruction.
+///
+/// \param __W
+///    A 128-bit vector of [16 x i8] used for writemask.
+/// \param __U
+///    A 16-bit mask indicating which elements to write.
+/// \param __A
+///    A 128-bit vector of [16 x i8] containing BF4 values.
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted HF8 values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_mask_cvtbf4_hf8(__m128i __W, __mmask16 __U, __m128i __A) {
   return (__m128i)__builtin_ia32_selectb_128(
       __U, (__v16qi)_mm_cvtbf4_hf8(__A), (__v16qi)__W);
 }
 
+/// Convert packed BF4 (4-bit) floating-point elements in \a __A to packed
+///    HF8 (8-bit) floating-point elements, and store the results using
+///    zeromask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBF42HF8 instruction.
+///
+/// \param __U
+///    A 16-bit mask indicating which elements to write (zero otherwise).
+/// \param __A
+///    A 128-bit vector of [16 x i8] containing BF4 values.
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted HF8 values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_maskz_cvtbf4_hf8(__mmask16 __U, __m128i __A) {
   return (__m128i)__builtin_ia32_selectb_128(
       __U, (__v16qi)_mm_cvtbf4_hf8(__A), (__v16qi)_mm_setzero_si128());
 }
 
-// VCVTBF42HF8 - 256-bit
-
+/// Convert packed BF4 (4-bit) floating-point elements in \a __A to packed
+///    HF8 (8-bit) floating-point elements, and store the results in a 256-bit
+///    vector.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBF42HF8 instruction.
+///
+/// \param __A
+///    A 128-bit vector of [16 x i8] containing BF4 values.
+/// \returns
+///    A 256-bit vector of [32 x i8] containing the converted HF8 values.
 static __inline__ __m256i __DEFAULT_FN_ATTRS256
 _mm256_cvtbf4_hf8(__m128i __A) {
   return (__m256i)__builtin_ia32_vcvtbf42hf8256((__v16qi)__A);
 }
 
+/// Convert packed BF4 (4-bit) floating-point elements in \a __A to packed
+///    HF8 (8-bit) floating-point elements, and store the results using
+///    writemask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBF42HF8 instruction.
+///
+/// \param __W
+///    A 256-bit vector of [32 x i8] used for writemask.
+/// \param __U
+///    A 32-bit mask indicating which elements to write.
+/// \param __A
+///    A 128-bit vector of [16 x i8] containing BF4 values.
+/// \returns
+///    A 256-bit vector of [32 x i8] containing the converted HF8 values.
 static __inline__ __m256i __DEFAULT_FN_ATTRS256
 _mm256_mask_cvtbf4_hf8(__m256i __W, __mmask32 __U, __m128i __A) {
   return (__m256i)__builtin_ia32_selectb_256(
       __U, (__v32qi)_mm256_cvtbf4_hf8(__A), (__v32qi)__W);
 }
 
+/// Convert packed BF4 (4-bit) floating-point elements in \a __A to packed
+///    HF8 (8-bit) floating-point elements, and store the results using
+///    zeromask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBF42HF8 instruction.
+///
+/// \param __U
+///    A 32-bit mask indicating which elements to write (zero otherwise).
+/// \param __A
+///    A 128-bit vector of [16 x i8] containing BF4 values.
+/// \returns
+///    A 256-bit vector of [32 x i8] containing the converted HF8 values.
 static __inline__ __m256i __DEFAULT_FN_ATTRS256
 _mm256_maskz_cvtbf4_hf8(__mmask32 __U, __m128i __A) {
   return (__m256i)__builtin_ia32_selectb_256(
       __U, (__v32qi)_mm256_cvtbf4_hf8(__A), (__v32qi)_mm256_setzero_si256());
 }
 
-// VCVTBF42HF8 - 512-bit
-
-static __inline__ __m512i __DEFAULT_FN_ATTRS512
-_mm512_cvtbf4_hf8(__m256i __A) {
-  return (__m512i)__builtin_ia32_vcvtbf42hf8512((__v32qi)__A);
-}
-
-static __inline__ __m512i __DEFAULT_FN_ATTRS512
-_mm512_mask_cvtbf4_hf8(__m512i __W, __mmask64 __U, __m256i __A) {
-  return (__m512i)__builtin_ia32_selectb_512(
-      __U, (__v64qi)_mm512_cvtbf4_hf8(__A), (__v64qi)__W);
-}
-
-static __inline__ __m512i __DEFAULT_FN_ATTRS512
-_mm512_maskz_cvtbf4_hf8(__mmask64 __U, __m256i __A) {
-  return (__m512i)__builtin_ia32_selectb_512(
-      __U, (__v64qi)_mm512_cvtbf4_hf8(__A), (__v64qi)_mm512_setzero_si512());
-}
-
-// VCVTBF62HF8 - 128-bit
-
+/// Convert packed BF6 (6-bit) floating-point elements in \a __A to packed
+///    HF8 (8-bit) floating-point elements, and store the results in a 128-bit
+///    vector.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBF62HF8 instruction.
+///
+/// \param __A
+///    A 128-bit vector of [16 x i8] containing BF6 values.
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted HF8 values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_cvtbf6_hf8(__m128i __A) {
   return (__m128i)__builtin_ia32_vcvtbf62hf8128((__v16qi)__A);
 }
 
+/// Convert packed BF6 (6-bit) floating-point elements in \a __A to packed
+///    HF8 (8-bit) floating-point elements, and store the results using
+///    writemask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBF62HF8 instruction.
+///
+/// \param __W
+///    A 128-bit vector of [16 x i8] used for writemask.
+/// \param __U
+///    A 16-bit mask indicating which elements to write.
+/// \param __A
+///    A 128-bit vector of [16 x i8] containing BF6 values.
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted HF8 values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_mask_cvtbf6_hf8(__m128i __W, __mmask16 __U, __m128i __A) {
   return (__m128i)__builtin_ia32_selectb_128(
       __U, (__v16qi)_mm_cvtbf6_hf8(__A), (__v16qi)__W);
 }
 
+/// Convert packed BF6 (6-bit) floating-point elements in \a __A to packed
+///    HF8 (8-bit) floating-point elements, and store the results using
+///    zeromask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBF62HF8 instruction.
+///
+/// \param __U
+///    A 16-bit mask indicating which elements to write (zero otherwise).
+/// \param __A
+///    A 128-bit vector of [16 x i8] containing BF6 values.
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted HF8 values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_maskz_cvtbf6_hf8(__mmask16 __U, __m128i __A) {
   return (__m128i)__builtin_ia32_selectb_128(
       __U, (__v16qi)_mm_cvtbf6_hf8(__A), (__v16qi)_mm_setzero_si128());
 }
 
-// VCVTBF62HF8 - 256-bit
-
+/// Convert packed BF6 (6-bit) floating-point elements in \a __A to packed
+///    HF8 (8-bit) floating-point elements, and store the results in a 256-bit
+///    vector.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBF62HF8 instruction.
+///
+/// \param __A
+///    A 256-bit vector of [32 x i8] containing BF6 values.
+/// \returns
+///    A 256-bit vector of [32 x i8] containing the converted HF8 values.
 static __inline__ __m256i __DEFAULT_FN_ATTRS256
 _mm256_cvtbf6_hf8(__m256i __A) {
   return (__m256i)__builtin_ia32_vcvtbf62hf8256((__v32qi)__A);
 }
 
+/// Convert packed BF6 (6-bit) floating-point elements in \a __A to packed
+///    HF8 (8-bit) floating-point elements, and store the results using
+///    writemask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBF62HF8 instruction.
+///
+/// \param __W
+///    A 256-bit vector of [32 x i8] used for writemask.
+/// \param __U
+///    A 32-bit mask indicating which elements to write.
+/// \param __A
+///    A 256-bit vector of [32 x i8] containing BF6 values.
+/// \returns
+///    A 256-bit vector of [32 x i8] containing the converted HF8 values.
 static __inline__ __m256i __DEFAULT_FN_ATTRS256
 _mm256_mask_cvtbf6_hf8(__m256i __W, __mmask32 __U, __m256i __A) {
   return (__m256i)__builtin_ia32_selectb_256(
       __U, (__v32qi)_mm256_cvtbf6_hf8(__A), (__v32qi)__W);
 }
 
+/// Convert packed BF6 (6-bit) floating-point elements in \a __A to packed
+///    HF8 (8-bit) floating-point elements, and store the results using
+///    zeromask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTBF62HF8 instruction.
+///
+/// \param __U
+///    A 32-bit mask indicating which elements to write (zero otherwise).
+/// \param __A
+///    A 256-bit vector of [32 x i8] containing BF6 values.
+/// \returns
+///    A 256-bit vector of [32 x i8] containing the converted HF8 values.
 static __inline__ __m256i __DEFAULT_FN_ATTRS256
 _mm256_maskz_cvtbf6_hf8(__mmask32 __U, __m256i __A) {
   return (__m256i)__builtin_ia32_selectb_256(
       __U, (__v32qi)_mm256_cvtbf6_hf8(__A), (__v32qi)_mm256_setzero_si256());
 }
 
-// VCVTBF62HF8 - 512-bit
-
-static __inline__ __m512i __DEFAULT_FN_ATTRS512
-_mm512_cvtbf6_hf8(__m512i __A) {
-  return (__m512i)__builtin_ia32_vcvtbf62hf8512((__v64qi)__A);
-}
-
-static __inline__ __m512i __DEFAULT_FN_ATTRS512
-_mm512_mask_cvtbf6_hf8(__m512i __W, __mmask64 __U, __m512i __A) {
-  return (__m512i)__builtin_ia32_selectb_512(
-      __U, (__v64qi)_mm512_cvtbf6_hf8(__A), (__v64qi)__W);
-}
-
-static __inline__ __m512i __DEFAULT_FN_ATTRS512
-_mm512_maskz_cvtbf6_hf8(__mmask64 __U, __m512i __A) {
-  return (__m512i)__builtin_ia32_selectb_512(
-      __U, (__v64qi)_mm512_cvtbf6_hf8(__A), (__v64qi)_mm512_setzero_si512());
-}
-
-// VCVTHF62HF8 - 128-bit
-
+/// Convert packed HF6 (6-bit) floating-point elements in \a __A to packed
+///    HF8 (8-bit) floating-point elements, and store the results in a 128-bit
+///    vector.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTHF62HF8 instruction.
+///
+/// \param __A
+///    A 128-bit vector of [16 x i8] containing HF6 values.
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted HF8 values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_cvthf6_hf8(__m128i __A) {
   return (__m128i)__builtin_ia32_vcvthf62hf8128((__v16qi)__A);
 }
 
+/// Convert packed HF6 (6-bit) floating-point elements in \a __A to packed
+///    HF8 (8-bit) floating-point elements, and store the results using
+///    writemask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTHF62HF8 instruction.
+///
+/// \param __W
+///    A 128-bit vector of [16 x i8] used for writemask.
+/// \param __U
+///    A 16-bit mask indicating which elements to write.
+/// \param __A
+///    A 128-bit vector of [16 x i8] containing HF6 values.
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted HF8 values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_mask_cvthf6_hf8(__m128i __W, __mmask16 __U, __m128i __A) {
   return (__m128i)__builtin_ia32_selectb_128(
       __U, (__v16qi)_mm_cvthf6_hf8(__A), (__v16qi)__W);
 }
 
+/// Convert packed HF6 (6-bit) floating-point elements in \a __A to packed
+///    HF8 (8-bit) floating-point elements, and store the results using
+///    zeromask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTHF62HF8 instruction.
+///
+/// \param __U
+///    A 16-bit mask indicating which elements to write (zero otherwise).
+/// \param __A
+///    A 128-bit vector of [16 x i8] containing HF6 values.
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted HF8 values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_maskz_cvthf6_hf8(__mmask16 __U, __m128i __A) {
   return (__m128i)__builtin_ia32_selectb_128(
       __U, (__v16qi)_mm_cvthf6_hf8(__A), (__v16qi)_mm_setzero_si128());
 }
 
-// VCVTHF62HF8 - 256-bit
-
+/// Convert packed HF6 (6-bit) floating-point elements in \a __A to packed
+///    HF8 (8-bit) floating-point elements, and store the results in a 256-bit
+///    vector.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTHF62HF8 instruction.
+///
+/// \param __A
+///    A 256-bit vector of [32 x i8] containing HF6 values.
+/// \returns
+///    A 256-bit vector of [32 x i8] containing the converted HF8 values.
 static __inline__ __m256i __DEFAULT_FN_ATTRS256
 _mm256_cvthf6_hf8(__m256i __A) {
   return (__m256i)__builtin_ia32_vcvthf62hf8256((__v32qi)__A);
 }
 
+/// Convert packed HF6 (6-bit) floating-point elements in \a __A to packed
+///    HF8 (8-bit) floating-point elements, and store the results using
+///    writemask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTHF62HF8 instruction.
+///
+/// \param __W
+///    A 256-bit vector of [32 x i8] used for writemask.
+/// \param __U
+///    A 32-bit mask indicating which elements to write.
+/// \param __A
+///    A 256-bit vector of [32 x i8] containing HF6 values.
+/// \returns
+///    A 256-bit vector of [32 x i8] containing the converted HF8 values.
 static __inline__ __m256i __DEFAULT_FN_ATTRS256
 _mm256_mask_cvthf6_hf8(__m256i __W, __mmask32 __U, __m256i __A) {
   return (__m256i)__builtin_ia32_selectb_256(
       __U, (__v32qi)_mm256_cvthf6_hf8(__A), (__v32qi)__W);
 }
 
+/// Convert packed HF6 (6-bit) floating-point elements in \a __A to packed
+///    HF8 (8-bit) floating-point elements, and store the results using
+///    zeromask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VCVTHF62HF8 instruction.
+///
+/// \param __U
+///    A 32-bit mask indicating which elements to write (zero otherwise).
+/// \param __A
+///    A 256-bit vector of [32 x i8] containing HF6 values.
+/// \returns
+///    A 256-bit vector of [32 x i8] containing the converted HF8 values.
 static __inline__ __m256i __DEFAULT_FN_ATTRS256
 _mm256_maskz_cvthf6_hf8(__mmask32 __U, __m256i __A) {
   return (__m256i)__builtin_ia32_selectb_256(
       __U, (__v32qi)_mm256_cvthf6_hf8(__A), (__v32qi)_mm256_setzero_si256());
 }
 
-// VCVTHF62HF8 - 512-bit
-
-static __inline__ __m512i __DEFAULT_FN_ATTRS512
-_mm512_cvthf6_hf8(__m512i __A) {
-  return (__m512i)__builtin_ia32_vcvthf62hf8512((__v64qi)__A);
-}
-
-static __inline__ __m512i __DEFAULT_FN_ATTRS512
-_mm512_mask_cvthf6_hf8(__m512i __W, __mmask64 __U, __m512i __A) {
-  return (__m512i)__builtin_ia32_selectb_512(
-      __U, (__v64qi)_mm512_cvthf6_hf8(__A), (__v64qi)__W);
-}
-
-static __inline__ __m512i __DEFAULT_FN_ATTRS512
-_mm512_maskz_cvthf6_hf8(__mmask64 __U, __m512i __A) {
-  return (__m512i)__builtin_ia32_selectb_512(
-      __U, (__v64qi)_mm512_cvthf6_hf8(__A), (__v64qi)_mm512_setzero_si512());
-}
-
-//===----------------------------------------------------------------------===//
-// Group H: VUNPACKB
-// Byte unpack with immediate
-//===----------------------------------------------------------------------===//
-
-// VUNPACKB - 128-bit
-
+/// Unpack bytes from \a A according to the immediate value \a imm, and store
+///    the results in a 128-bit vector.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VUNPACKB instruction.
+///
+/// \param A
+///    A 128-bit vector of [16 x i8].
+/// \param imm
+///    An immediate value specifying the unpack operation.
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the unpacked values.
 #define _mm_unpackb_epi8(A, imm)                                               \
   ((__m128i)__builtin_ia32_vunpackb128((__v16qi)(__m128i)(A), (int)(imm)))
 
+/// Unpack bytes from \a A according to the immediate value \a imm, and store
+///    the results in a 128-bit vector using writemask \a U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VUNPACKB instruction.
+///
+/// \param W
+///    A 128-bit vector of [16 x i8] used for writemask.
+/// \param U
+///    A 16-bit mask indicating which elements to write.
+/// \param A
+///    A 128-bit vector of [16 x i8].
+/// \param imm
+///    An immediate value specifying the unpack operation.
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the unpacked values.
 #define _mm_mask_unpackb_epi8(W, U, A, imm)                                    \
-  ((__m128i)__builtin_ia32_selectb_128(                                         \
-      (__mmask16)(U),                                                           \
-      (__v16qi)_mm_unpackb_epi8((A), (imm)),                                    \
-      (__v16qi)(__m128i)(W)))
-
+  ((__m128i)__builtin_ia32_selectb_128((__mmask16)(U),                         \
+                                       (__v16qi)_mm_unpackb_epi8((A), (imm)),  \
+                                       (__v16qi)(__m128i)(W)))
+
+/// Unpack bytes from \a A according to the immediate value \a imm, and store
+///    the results in a 128-bit vector using zeromask \a U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VUNPACKB instruction.
+///
+/// \param U
+///    A 16-bit mask indicating which elements to write (zero otherwise).
+/// \param A
+///    A 128-bit vector of [16 x i8].
+/// \param imm
+///    An immediate value specifying the unpack operation.
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the unpacked values.
 #define _mm_maskz_unpackb_epi8(U, A, imm)                                      \
-  ((__m128i)__builtin_ia32_selectb_128(                                         \
-      (__mmask16)(U),                                                           \
-      (__v16qi)_mm_unpackb_epi8((A), (imm)),                                    \
-      (__v16qi)_mm_setzero_si128()))
-
-// VUNPACKB - 256-bit
-
+  ((__m128i)__builtin_ia32_selectb_128((__mmask16)(U),                         \
+                                       (__v16qi)_mm_unpackb_epi8((A), (imm)),  \
+                                       (__v16qi)_mm_setzero_si128()))
+
+/// Unpack bytes from \a A according to the immediate value \a imm, and store
+///    the results in a 256-bit vector.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VUNPACKB instruction.
+///
+/// \param A
+///    A 256-bit vector of [32 x i8].
+/// \param imm
+///    An immediate value specifying the unpack operation.
+/// \returns
+///    A 256-bit vector of [32 x i8] containing the unpacked values.
 #define _mm256_unpackb_epi8(A, imm)                                            \
   ((__m256i)__builtin_ia32_vunpackb256((__v32qi)(__m256i)(A), (int)(imm)))
 
+/// Unpack bytes from \a A according to the immediate value \a imm, and store
+///    the results in a 256-bit vector using writemask \a U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VUNPACKB instruction.
+///
+/// \param W
+///    A 256-bit vector of [32 x i8] used for writemask.
+/// \param U
+///    A 32-bit mask indicating which elements to write.
+/// \param A
+///    A 256-bit vector of [32 x i8].
+/// \param imm
+///    An immediate value specifying the unpack operation.
+/// \returns
+///    A 256-bit vector of [32 x i8] containing the unpacked values.
 #define _mm256_mask_unpackb_epi8(W, U, A, imm)                                 \
-  ((__m256i)__builtin_ia32_selectb_256(                                         \
-      (__mmask32)(U),                                                           \
-      (__v32qi)_mm256_unpackb_epi8((A), (imm)),                                 \
+  ((__m256i)__builtin_ia32_selectb_256(                                        \
+      (__mmask32)(U), (__v32qi)_mm256_unpackb_epi8((A), (imm)),                \
       (__v32qi)(__m256i)(W)))
 
+/// Unpack bytes from \a A according to the immediate value \a imm, and store
+///    the results in a 256-bit vector using zeromask \a U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VUNPACKB instruction.
+///
+/// \param U
+///    A 32-bit mask indicating which elements to write (zero otherwise).
+/// \param A
+///    A 256-bit vector of [32 x i8].
+/// \param imm
+///    An immediate value specifying the unpack operation.
+/// \returns
+///    A 256-bit vector of [32 x i8] containing the unpacked values.
 #define _mm256_maskz_unpackb_epi8(U, A, imm)                                   \
-  ((__m256i)__builtin_ia32_selectb_256(                                         \
-      (__mmask32)(U),                                                           \
-      (__v32qi)_mm256_unpackb_epi8((A), (imm)),                                 \
+  ((__m256i)__builtin_ia32_selectb_256(                                        \
+      (__mmask32)(U), (__v32qi)_mm256_unpackb_epi8((A), (imm)),                \
       (__v32qi)_mm256_setzero_si256()))
 
-// VUNPACKB - 512-bit
-
-#define _mm512_unpackb_epi8(A, imm)                                            \
-  ((__m512i)__builtin_ia32_vunpackb512((__v64qi)(__m512i)(A), (int)(imm)))
-
-#define _mm512_mask_unpackb_epi8(W, U, A, imm)                                 \
-  ((__m512i)__builtin_ia32_selectb_512(                                         \
-      (__mmask64)(U),                                                           \
-      (__v64qi)_mm512_unpackb_epi8((A), (imm)),                                 \
-      (__v64qi)(__m512i)(W)))
-
-#define _mm512_maskz_unpackb_epi8(U, A, imm)                                   \
-  ((__m512i)__builtin_ia32_selectb_512(                                         \
-      (__mmask64)(U),                                                           \
-      (__v64qi)_mm512_unpackb_epi8((A), (imm)),                                 \
-      (__v64qi)_mm512_setzero_si512()))
-
-// clang-format on
+/* VPMOVSSDB - Symmetric Signed Saturation DWord to Byte */
+
+/// Convert packed signed 32-bit integers in \a __A to packed 8-bit integers
+///    with symmetric signed saturation (clamp to [-127, +127]), and store
+///    the results in a 128-bit vector.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VPMOVSSDB instruction.
+///
+/// \param __A
+///    A 128-bit vector of [4 x i32].
+/// \returns
+///    A 128-bit vector of [16 x i8]. The lower 4 bytes contain the converted
+///    values; the upper bytes are zeroed.
+static __inline__ __m128i __DEFAULT_FN_ATTRS128
+_mm_cvtss_epi32_epi8(__m128i __A) {
+  return (__m128i)__builtin_ia32_pmovssdb128_mask(
+      (__v4si)__A, (__v16qi)_mm_setzero_si128(), (__mmask8)-1);
+}
+
+/// Convert packed signed 32-bit integers in \a __A to packed 8-bit integers
+///    with symmetric signed saturation, using writemask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VPMOVSSDB instruction.
+///
+/// \param __W
+///    A 128-bit vector of [16 x i8] used for writemask.
+/// \param __U
+///    A 8-bit mask indicating which elements to write.
+/// \param __A
+///    A 128-bit vector of [4 x i32].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
+static __inline__ __m128i __DEFAULT_FN_ATTRS128
+_mm_mask_cvtss_epi32_epi8(__m128i __W, __mmask8 __U, __m128i __A) {
+  return (__m128i)__builtin_ia32_pmovssdb128_mask((__v4si)__A, (__v16qi)__W,
+                                                  __U);
+}
+
+/// Convert packed signed 32-bit integers in \a __A to packed 8-bit integers
+///    with symmetric signed saturation, using zeromask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VPMOVSSDB instruction.
+///
+/// \param __U
+///    A 8-bit mask indicating which elements to write (zero otherwise).
+/// \param __A
+///    A 128-bit vector of [4 x i32].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
+static __inline__ __m128i __DEFAULT_FN_ATTRS128
+_mm_maskz_cvtss_epi32_epi8(__mmask8 __U, __m128i __A) {
+  return (__m128i)__builtin_ia32_pmovssdb128_mask(
+      (__v4si)__A, (__v16qi)_mm_setzero_si128(), __U);
+}
+
+/// Convert packed signed 32-bit integers in \a __A to packed 8-bit integers
+///    with symmetric signed saturation (clamp to [-127, +127]), and store
+///    the results in a 128-bit vector.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VPMOVSSDB instruction.
+///
+/// \param __A
+///    A 256-bit vector of [8 x i32].
+/// \returns
+///    A 128-bit vector of [16 x i8]. The lower 8 bytes contain the converted
+///    values; the upper bytes are zeroed.
+static __inline__ __m128i __DEFAULT_FN_ATTRS256
+_mm256_cvtss_epi32_epi8(__m256i __A) {
+  return (__m128i)__builtin_ia32_pmovssdb256_mask(
+      (__v8si)__A, (__v16qi)_mm_setzero_si128(), (__mmask8)-1);
+}
+
+/// Convert packed signed 32-bit integers in \a __A to packed 8-bit integers
+///    with symmetric signed saturation, using writemask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VPMOVSSDB instruction.
+///
+/// \param __W
+///    A 128-bit vector of [16 x i8] used for writemask.
+/// \param __U
+///    A 8-bit mask indicating which elements to write.
+/// \param __A
+///    A 256-bit vector of [8 x i32].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
+static __inline__ __m128i __DEFAULT_FN_ATTRS256
+_mm256_mask_cvtss_epi32_epi8(__m128i __W, __mmask8 __U, __m256i __A) {
+  return (__m128i)__builtin_ia32_pmovssdb256_mask((__v8si)__A, (__v16qi)__W,
+                                                  __U);
+}
+
+/// Convert packed signed 32-bit integers in \a __A to packed 8-bit integers
+///    with symmetric signed saturation, using zeromask \a __U.
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VPMOVSSDB instruction.
+///
+/// \param __U
+///    A 8-bit mask indicating which elements to write (zero otherwise).
+/// \param __A
+///    A 256-bit vector of [8 x i32].
+/// \returns
+///    A 128-bit vector of [16 x i8] containing the converted values.
+static __inline__ __m128i __DEFAULT_FN_ATTRS256
+_mm256_maskz_cvtss_epi32_epi8(__mmask8 __U, __m256i __A) {
+  return (__m128i)__builtin_ia32_pmovssdb256_mask(
+      (__v8si)__A, (__v16qi)_mm_setzero_si128(), __U);
+}
+
+/// Truncate packed 32-bit integers in \a __A to packed 8-bit integers with
+/// symmetric signed saturation, and store the results to memory at \a __P
+/// using writemask \a __M (elements not selected by the mask are not written).
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VPMOVSSDB instruction.
+///
+/// \param __P
+///    Pointer to the destination memory.
+/// \param __M
+///    A 8-bit mask indicating which elements to write.
+/// \param __A
+///    A 128-bit vector of [4 x i32].
+static __inline__ void __DEFAULT_FN_ATTRS128
+_mm_mask_cvtss_epi32_storeu_epi8(void *__P, __mmask8 __M, __m128i __A) {
+  __builtin_ia32_pmovssdb128mem_mask((__v16qi *)__P, (__v4si)__A, __M);
+}
+
+/// Truncate packed 32-bit integers in \a __A to packed 8-bit integers with
+/// symmetric signed saturation, and store the results to memory at \a __P
+/// using writemask \a __M (elements not selected by the mask are not written).
+///
+/// \headerfile <immintrin.h>
+///
+/// This intrinsic corresponds to the \c VPMOVSSDB instruction.
+///
+/// \param __P
+///    Pointer to the destination memory.
+/// \param __M
+///    A 8-bit mask indicating which elements to write.
+/// \param __A
+///    A 256-bit vector of [8 x i32].
+static __inline__ void __DEFAULT_FN_ATTRS256
+_mm256_mask_cvtss_epi32_storeu_epi8(void *__P, __mmask8 __M, __m256i __A) {
+  __builtin_ia32_pmovssdb256mem_mask((__v16qi *)__P, (__v8si)__A, __M);
+}
 
 #undef __DEFAULT_FN_ATTRS128
 #undef __DEFAULT_FN_ATTRS256
-#undef __DEFAULT_FN_ATTRS512
 
 #endif // __AVX10_2_V2AUXINTRIN_H
 #endif // __SSE2__
diff --git a/clang/lib/Headers/immintrin.h b/clang/lib/Headers/immintrin.h
index bd474d65079351..149967b3817bdd 100644
--- a/clang/lib/Headers/immintrin.h
+++ b/clang/lib/Headers/immintrin.h
@@ -500,9 +500,8 @@ _storebe_i64(void * __P, long long __D) {
 #include <avx10_2_512satcvtdsintrin.h>
 #include <avx10_2_512satcvtintrin.h>
 
-#ifdef __AVX10_V2_AUX__
+#include <avx10_2_512v2auxintrin.h>
 #include <avx10_2_v2auxintrin.h>
-#endif
 
 #include <sm4evexintrin.h>
 
diff --git a/clang/lib/Sema/SemaX86.cpp b/clang/lib/Sema/SemaX86.cpp
index fa92d6ff122eef..1728484a5bff14 100644
--- a/clang/lib/Sema/SemaX86.cpp
+++ b/clang/lib/Sema/SemaX86.cpp
@@ -722,6 +722,13 @@ bool SemaX86::CheckBuiltinFunctionCall(const TargetInfo &TI, unsigned BuiltinID,
     l = 0;
     u = 31;
     break;
+  case X86::BI__builtin_ia32_vunpackb128:
+  case X86::BI__builtin_ia32_vunpackb256:
+  case X86::BI__builtin_ia32_vunpackb512:
+    i = 1;
+    l = 0;
+    u = 63;
+    break;
   case X86::BI__builtin_ia32_cmpps:
   case X86::BI__builtin_ia32_cmpss:
   case X86::BI__builtin_ia32_cmppd:
diff --git a/clang/test/CodeGen/X86/avx10_2_v2aux-builtins.c b/clang/test/CodeGen/X86/avx10_2_v2aux-builtins.c
index c4d076b27b21eb..c652af1e4d238d 100644
--- a/clang/test/CodeGen/X86/avx10_2_v2aux-builtins.c
+++ b/clang/test/CodeGen/X86/avx10_2_v2aux-builtins.c
@@ -1,6 +1,6 @@
-// RUN: %clang_cc1 %s -flax-vector-conversions=none -ffreestanding -triple=x86_64 -target-feature +avx10-v2-aux \
+// RUN: %clang_cc1 %s -flax-vector-conversions=none -ffreestanding -triple=x86_64 -target-feature +avx10v2aux \
 // RUN: -emit-llvm -o - -Wno-invalid-feature-combination -Wall -Werror | FileCheck %s
-// RUN: %clang_cc1 %s -flax-vector-conversions=none -ffreestanding -triple=i386 -target-feature +avx10-v2-aux \
+// RUN: %clang_cc1 %s -flax-vector-conversions=none -ffreestanding -triple=i386 -target-feature +avx10v2aux \
 // RUN: -emit-llvm -o - -Wno-invalid-feature-combination -Wall -Werror | FileCheck %s
 
 #include <immintrin.h>
@@ -14,19 +14,21 @@
 
 __m128i test_mm_cvtps_bf8(__m128 __A) {
   // CHECK-LABEL: @test_mm_cvtps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.128(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtps2bf8.128(<4 x float> %{{.*}})
   return _mm_cvtps_bf8(__A);
 }
 
 __m128i test_mm_mask_cvtps_bf8(__m128i __W, __mmask8 __U, __m128 __A) {
   // CHECK-LABEL: @test_mm_mask_cvtps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.128(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8.128(<4 x float> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm_mask_cvtps_bf8(__W, __U, __A);
 }
 
 __m128i test_mm_maskz_cvtps_bf8(__mmask8 __U, __m128 __A) {
   // CHECK-LABEL: @test_mm_maskz_cvtps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.128(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8.128(<4 x float> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm_maskz_cvtps_bf8(__U, __A);
 }
 
@@ -34,19 +36,21 @@ __m128i test_mm_maskz_cvtps_bf8(__mmask8 __U, __m128 __A) {
 
 __m128i test_mm256_cvtps_bf8(__m256 __A) {
   // CHECK-LABEL: @test_mm256_cvtps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.256(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtps2bf8.256(<8 x float> %{{.*}})
   return _mm256_cvtps_bf8(__A);
 }
 
 __m128i test_mm256_mask_cvtps_bf8(__m128i __W, __mmask8 __U, __m256 __A) {
   // CHECK-LABEL: @test_mm256_mask_cvtps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.256(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8.256(<8 x float> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm256_mask_cvtps_bf8(__W, __U, __A);
 }
 
 __m128i test_mm256_maskz_cvtps_bf8(__mmask8 __U, __m256 __A) {
   // CHECK-LABEL: @test_mm256_maskz_cvtps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.256(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8.256(<8 x float> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm256_maskz_cvtps_bf8(__U, __A);
 }
 
@@ -54,19 +58,21 @@ __m128i test_mm256_maskz_cvtps_bf8(__mmask8 __U, __m256 __A) {
 
 __m128i test_mm512_cvtps_bf8(__m512 __A) {
   // CHECK-LABEL: @test_mm512_cvtps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.512(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtps2bf8.512(<16 x float> %{{.*}})
   return _mm512_cvtps_bf8(__A);
 }
 
 __m128i test_mm512_mask_cvtps_bf8(__m128i __W, __mmask16 __U, __m512 __A) {
   // CHECK-LABEL: @test_mm512_mask_cvtps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.512(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8.512(<16 x float> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm512_mask_cvtps_bf8(__W, __U, __A);
 }
 
 __m128i test_mm512_maskz_cvtps_bf8(__mmask16 __U, __m512 __A) {
   // CHECK-LABEL: @test_mm512_maskz_cvtps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.512(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8.512(<16 x float> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm512_maskz_cvtps_bf8(__U, __A);
 }
 
@@ -74,19 +80,21 @@ __m128i test_mm512_maskz_cvtps_bf8(__mmask16 __U, __m512 __A) {
 
 __m128i test_mm_cvts_ps_bf8(__m128 __A) {
   // CHECK-LABEL: @test_mm_cvts_ps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.128(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtps2bf8s.128(<4 x float> %{{.*}})
   return _mm_cvts_ps_bf8(__A);
 }
 
 __m128i test_mm_mask_cvts_ps_bf8(__m128i __W, __mmask8 __U, __m128 __A) {
   // CHECK-LABEL: @test_mm_mask_cvts_ps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.128(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8s.128(<4 x float> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm_mask_cvts_ps_bf8(__W, __U, __A);
 }
 
 __m128i test_mm_maskz_cvts_ps_bf8(__mmask8 __U, __m128 __A) {
   // CHECK-LABEL: @test_mm_maskz_cvts_ps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.128(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8s.128(<4 x float> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm_maskz_cvts_ps_bf8(__U, __A);
 }
 
@@ -94,19 +102,21 @@ __m128i test_mm_maskz_cvts_ps_bf8(__mmask8 __U, __m128 __A) {
 
 __m128i test_mm256_cvts_ps_bf8(__m256 __A) {
   // CHECK-LABEL: @test_mm256_cvts_ps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.256(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtps2bf8s.256(<8 x float> %{{.*}})
   return _mm256_cvts_ps_bf8(__A);
 }
 
 __m128i test_mm256_mask_cvts_ps_bf8(__m128i __W, __mmask8 __U, __m256 __A) {
   // CHECK-LABEL: @test_mm256_mask_cvts_ps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.256(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8s.256(<8 x float> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm256_mask_cvts_ps_bf8(__W, __U, __A);
 }
 
 __m128i test_mm256_maskz_cvts_ps_bf8(__mmask8 __U, __m256 __A) {
   // CHECK-LABEL: @test_mm256_maskz_cvts_ps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.256(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8s.256(<8 x float> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm256_maskz_cvts_ps_bf8(__U, __A);
 }
 
@@ -114,19 +124,21 @@ __m128i test_mm256_maskz_cvts_ps_bf8(__mmask8 __U, __m256 __A) {
 
 __m128i test_mm512_cvts_ps_bf8(__m512 __A) {
   // CHECK-LABEL: @test_mm512_cvts_ps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.512(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtps2bf8s.512(<16 x float> %{{.*}})
   return _mm512_cvts_ps_bf8(__A);
 }
 
 __m128i test_mm512_mask_cvts_ps_bf8(__m128i __W, __mmask16 __U, __m512 __A) {
   // CHECK-LABEL: @test_mm512_mask_cvts_ps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.512(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8s.512(<16 x float> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm512_mask_cvts_ps_bf8(__W, __U, __A);
 }
 
 __m128i test_mm512_maskz_cvts_ps_bf8(__mmask16 __U, __m512 __A) {
   // CHECK-LABEL: @test_mm512_maskz_cvts_ps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.512(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8s.512(<16 x float> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm512_maskz_cvts_ps_bf8(__U, __A);
 }
 
@@ -134,19 +146,21 @@ __m128i test_mm512_maskz_cvts_ps_bf8(__mmask16 __U, __m512 __A) {
 
 __m128i test_mm_cvtps_hf8(__m128 __A) {
   // CHECK-LABEL: @test_mm_cvtps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.128(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtps2hf8.128(<4 x float> %{{.*}})
   return _mm_cvtps_hf8(__A);
 }
 
 __m128i test_mm_mask_cvtps_hf8(__m128i __W, __mmask8 __U, __m128 __A) {
   // CHECK-LABEL: @test_mm_mask_cvtps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.128(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8.128(<4 x float> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm_mask_cvtps_hf8(__W, __U, __A);
 }
 
 __m128i test_mm_maskz_cvtps_hf8(__mmask8 __U, __m128 __A) {
   // CHECK-LABEL: @test_mm_maskz_cvtps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.128(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8.128(<4 x float> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm_maskz_cvtps_hf8(__U, __A);
 }
 
@@ -154,19 +168,21 @@ __m128i test_mm_maskz_cvtps_hf8(__mmask8 __U, __m128 __A) {
 
 __m128i test_mm256_cvtps_hf8(__m256 __A) {
   // CHECK-LABEL: @test_mm256_cvtps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.256(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtps2hf8.256(<8 x float> %{{.*}})
   return _mm256_cvtps_hf8(__A);
 }
 
 __m128i test_mm256_mask_cvtps_hf8(__m128i __W, __mmask8 __U, __m256 __A) {
   // CHECK-LABEL: @test_mm256_mask_cvtps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.256(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8.256(<8 x float> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm256_mask_cvtps_hf8(__W, __U, __A);
 }
 
 __m128i test_mm256_maskz_cvtps_hf8(__mmask8 __U, __m256 __A) {
   // CHECK-LABEL: @test_mm256_maskz_cvtps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.256(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8.256(<8 x float> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm256_maskz_cvtps_hf8(__U, __A);
 }
 
@@ -174,19 +190,21 @@ __m128i test_mm256_maskz_cvtps_hf8(__mmask8 __U, __m256 __A) {
 
 __m128i test_mm512_cvtps_hf8(__m512 __A) {
   // CHECK-LABEL: @test_mm512_cvtps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.512(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtps2hf8.512(<16 x float> %{{.*}})
   return _mm512_cvtps_hf8(__A);
 }
 
 __m128i test_mm512_mask_cvtps_hf8(__m128i __W, __mmask16 __U, __m512 __A) {
   // CHECK-LABEL: @test_mm512_mask_cvtps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.512(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8.512(<16 x float> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm512_mask_cvtps_hf8(__W, __U, __A);
 }
 
 __m128i test_mm512_maskz_cvtps_hf8(__mmask16 __U, __m512 __A) {
   // CHECK-LABEL: @test_mm512_maskz_cvtps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.512(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8.512(<16 x float> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm512_maskz_cvtps_hf8(__U, __A);
 }
 
@@ -194,19 +212,21 @@ __m128i test_mm512_maskz_cvtps_hf8(__mmask16 __U, __m512 __A) {
 
 __m128i test_mm_cvts_ps_hf8(__m128 __A) {
   // CHECK-LABEL: @test_mm_cvts_ps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.128(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtps2hf8s.128(<4 x float> %{{.*}})
   return _mm_cvts_ps_hf8(__A);
 }
 
 __m128i test_mm_mask_cvts_ps_hf8(__m128i __W, __mmask8 __U, __m128 __A) {
   // CHECK-LABEL: @test_mm_mask_cvts_ps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.128(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8s.128(<4 x float> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm_mask_cvts_ps_hf8(__W, __U, __A);
 }
 
 __m128i test_mm_maskz_cvts_ps_hf8(__mmask8 __U, __m128 __A) {
   // CHECK-LABEL: @test_mm_maskz_cvts_ps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.128(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8s.128(<4 x float> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm_maskz_cvts_ps_hf8(__U, __A);
 }
 
@@ -214,19 +234,21 @@ __m128i test_mm_maskz_cvts_ps_hf8(__mmask8 __U, __m128 __A) {
 
 __m128i test_mm256_cvts_ps_hf8(__m256 __A) {
   // CHECK-LABEL: @test_mm256_cvts_ps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.256(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtps2hf8s.256(<8 x float> %{{.*}})
   return _mm256_cvts_ps_hf8(__A);
 }
 
 __m128i test_mm256_mask_cvts_ps_hf8(__m128i __W, __mmask8 __U, __m256 __A) {
   // CHECK-LABEL: @test_mm256_mask_cvts_ps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.256(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8s.256(<8 x float> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm256_mask_cvts_ps_hf8(__W, __U, __A);
 }
 
 __m128i test_mm256_maskz_cvts_ps_hf8(__mmask8 __U, __m256 __A) {
   // CHECK-LABEL: @test_mm256_maskz_cvts_ps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.256(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8s.256(<8 x float> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm256_maskz_cvts_ps_hf8(__U, __A);
 }
 
@@ -234,19 +256,21 @@ __m128i test_mm256_maskz_cvts_ps_hf8(__mmask8 __U, __m256 __A) {
 
 __m128i test_mm512_cvts_ps_hf8(__m512 __A) {
   // CHECK-LABEL: @test_mm512_cvts_ps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.512(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtps2hf8s.512(<16 x float> %{{.*}})
   return _mm512_cvts_ps_hf8(__A);
 }
 
 __m128i test_mm512_mask_cvts_ps_hf8(__m128i __W, __mmask16 __U, __m512 __A) {
   // CHECK-LABEL: @test_mm512_mask_cvts_ps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.512(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8s.512(<16 x float> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm512_mask_cvts_ps_hf8(__W, __U, __A);
 }
 
 __m128i test_mm512_maskz_cvts_ps_hf8(__mmask16 __U, __m512 __A) {
   // CHECK-LABEL: @test_mm512_maskz_cvts_ps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.512(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8s.512(<16 x float> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm512_maskz_cvts_ps_hf8(__U, __A);
 }
 
@@ -254,19 +278,21 @@ __m128i test_mm512_maskz_cvts_ps_hf8(__mmask16 __U, __m512 __A) {
 
 __m128i test_mm_cvtrops_hf8(__m128 __A) {
   // CHECK-LABEL: @test_mm_cvtrops_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.128(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8.128(<4 x float> %{{.*}})
   return _mm_cvtrops_hf8(__A);
 }
 
 __m128i test_mm_mask_cvtrops_hf8(__m128i __W, __mmask8 __U, __m128 __A) {
   // CHECK-LABEL: @test_mm_mask_cvtrops_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.128(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8.128(<4 x float> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm_mask_cvtrops_hf8(__W, __U, __A);
 }
 
 __m128i test_mm_maskz_cvtrops_hf8(__mmask8 __U, __m128 __A) {
   // CHECK-LABEL: @test_mm_maskz_cvtrops_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.128(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8.128(<4 x float> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm_maskz_cvtrops_hf8(__U, __A);
 }
 
@@ -274,19 +300,21 @@ __m128i test_mm_maskz_cvtrops_hf8(__mmask8 __U, __m128 __A) {
 
 __m128i test_mm256_cvtrops_hf8(__m256 __A) {
   // CHECK-LABEL: @test_mm256_cvtrops_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.256(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8.256(<8 x float> %{{.*}})
   return _mm256_cvtrops_hf8(__A);
 }
 
 __m128i test_mm256_mask_cvtrops_hf8(__m128i __W, __mmask8 __U, __m256 __A) {
   // CHECK-LABEL: @test_mm256_mask_cvtrops_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.256(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8.256(<8 x float> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm256_mask_cvtrops_hf8(__W, __U, __A);
 }
 
 __m128i test_mm256_maskz_cvtrops_hf8(__mmask8 __U, __m256 __A) {
   // CHECK-LABEL: @test_mm256_maskz_cvtrops_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.256(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8.256(<8 x float> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm256_maskz_cvtrops_hf8(__U, __A);
 }
 
@@ -294,19 +322,21 @@ __m128i test_mm256_maskz_cvtrops_hf8(__mmask8 __U, __m256 __A) {
 
 __m128i test_mm512_cvtrops_hf8(__m512 __A) {
   // CHECK-LABEL: @test_mm512_cvtrops_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.512(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8.512(<16 x float> %{{.*}})
   return _mm512_cvtrops_hf8(__A);
 }
 
 __m128i test_mm512_mask_cvtrops_hf8(__m128i __W, __mmask16 __U, __m512 __A) {
   // CHECK-LABEL: @test_mm512_mask_cvtrops_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.512(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8.512(<16 x float> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm512_mask_cvtrops_hf8(__W, __U, __A);
 }
 
 __m128i test_mm512_maskz_cvtrops_hf8(__mmask16 __U, __m512 __A) {
   // CHECK-LABEL: @test_mm512_maskz_cvtrops_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.512(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8.512(<16 x float> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm512_maskz_cvtrops_hf8(__U, __A);
 }
 
@@ -314,19 +344,21 @@ __m128i test_mm512_maskz_cvtrops_hf8(__mmask16 __U, __m512 __A) {
 
 __m128i test_mm_cvts_rops_hf8(__m128 __A) {
   // CHECK-LABEL: @test_mm_cvts_rops_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.128(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s.128(<4 x float> %{{.*}})
   return _mm_cvts_rops_hf8(__A);
 }
 
 __m128i test_mm_mask_cvts_rops_hf8(__m128i __W, __mmask8 __U, __m128 __A) {
   // CHECK-LABEL: @test_mm_mask_cvts_rops_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.128(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s.128(<4 x float> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm_mask_cvts_rops_hf8(__W, __U, __A);
 }
 
 __m128i test_mm_maskz_cvts_rops_hf8(__mmask8 __U, __m128 __A) {
   // CHECK-LABEL: @test_mm_maskz_cvts_rops_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.128(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s.128(<4 x float> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm_maskz_cvts_rops_hf8(__U, __A);
 }
 
@@ -334,19 +366,21 @@ __m128i test_mm_maskz_cvts_rops_hf8(__mmask8 __U, __m128 __A) {
 
 __m128i test_mm256_cvts_rops_hf8(__m256 __A) {
   // CHECK-LABEL: @test_mm256_cvts_rops_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.256(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s.256(<8 x float> %{{.*}})
   return _mm256_cvts_rops_hf8(__A);
 }
 
 __m128i test_mm256_mask_cvts_rops_hf8(__m128i __W, __mmask8 __U, __m256 __A) {
   // CHECK-LABEL: @test_mm256_mask_cvts_rops_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.256(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s.256(<8 x float> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm256_mask_cvts_rops_hf8(__W, __U, __A);
 }
 
 __m128i test_mm256_maskz_cvts_rops_hf8(__mmask8 __U, __m256 __A) {
   // CHECK-LABEL: @test_mm256_maskz_cvts_rops_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.256(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s.256(<8 x float> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm256_maskz_cvts_rops_hf8(__U, __A);
 }
 
@@ -354,19 +388,21 @@ __m128i test_mm256_maskz_cvts_rops_hf8(__mmask8 __U, __m256 __A) {
 
 __m128i test_mm512_cvts_rops_hf8(__m512 __A) {
   // CHECK-LABEL: @test_mm512_cvts_rops_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.512(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s.512(<16 x float> %{{.*}})
   return _mm512_cvts_rops_hf8(__A);
 }
 
 __m128i test_mm512_mask_cvts_rops_hf8(__m128i __W, __mmask16 __U, __m512 __A) {
   // CHECK-LABEL: @test_mm512_mask_cvts_rops_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.512(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s.512(<16 x float> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm512_mask_cvts_rops_hf8(__W, __U, __A);
 }
 
 __m128i test_mm512_maskz_cvts_rops_hf8(__mmask16 __U, __m512 __A) {
   // CHECK-LABEL: @test_mm512_maskz_cvts_rops_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.512(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s.512(<16 x float> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm512_maskz_cvts_rops_hf8(__U, __A);
 }
 
@@ -379,19 +415,21 @@ __m128i test_mm512_maskz_cvts_rops_hf8(__mmask16 __U, __m512 __A) {
 
 __m128i test_mm_cvtbiasps_bf8(__m128i __A, __m128 __B) {
   // CHECK-LABEL: @test_mm_cvtbiasps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.128(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8.128(<16 x i8> %{{.*}}, <4 x float> %{{.*}})
   return _mm_cvtbiasps_bf8(__A, __B);
 }
 
 __m128i test_mm_mask_cvtbiasps_bf8(__m128i __W, __mmask8 __U, __m128i __A, __m128 __B) {
   // CHECK-LABEL: @test_mm_mask_cvtbiasps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.128(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8.128(<16 x i8> %{{.*}}, <4 x float> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm_mask_cvtbiasps_bf8(__W, __U, __A, __B);
 }
 
 __m128i test_mm_maskz_cvtbiasps_bf8(__mmask8 __U, __m128i __A, __m128 __B) {
   // CHECK-LABEL: @test_mm_maskz_cvtbiasps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.128(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8.128(<16 x i8> %{{.*}}, <4 x float> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm_maskz_cvtbiasps_bf8(__U, __A, __B);
 }
 
@@ -399,19 +437,21 @@ __m128i test_mm_maskz_cvtbiasps_bf8(__mmask8 __U, __m128i __A, __m128 __B) {
 
 __m128i test_mm256_cvtbiasps_bf8(__m256i __A, __m256 __B) {
   // CHECK-LABEL: @test_mm256_cvtbiasps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.256(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8.256(<32 x i8> %{{.*}}, <8 x float> %{{.*}})
   return _mm256_cvtbiasps_bf8(__A, __B);
 }
 
 __m128i test_mm256_mask_cvtbiasps_bf8(__m128i __W, __mmask8 __U, __m256i __A, __m256 __B) {
   // CHECK-LABEL: @test_mm256_mask_cvtbiasps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.256(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8.256(<32 x i8> %{{.*}}, <8 x float> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm256_mask_cvtbiasps_bf8(__W, __U, __A, __B);
 }
 
 __m128i test_mm256_maskz_cvtbiasps_bf8(__mmask8 __U, __m256i __A, __m256 __B) {
   // CHECK-LABEL: @test_mm256_maskz_cvtbiasps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.256(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8.256(<32 x i8> %{{.*}}, <8 x float> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm256_maskz_cvtbiasps_bf8(__U, __A, __B);
 }
 
@@ -419,19 +459,21 @@ __m128i test_mm256_maskz_cvtbiasps_bf8(__mmask8 __U, __m256i __A, __m256 __B) {
 
 __m128i test_mm512_cvtbiasps_bf8(__m512i __A, __m512 __B) {
   // CHECK-LABEL: @test_mm512_cvtbiasps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.512(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8.512(<64 x i8> %{{.*}}, <16 x float> %{{.*}})
   return _mm512_cvtbiasps_bf8(__A, __B);
 }
 
 __m128i test_mm512_mask_cvtbiasps_bf8(__m128i __W, __mmask16 __U, __m512i __A, __m512 __B) {
   // CHECK-LABEL: @test_mm512_mask_cvtbiasps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.512(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8.512(<64 x i8> %{{.*}}, <16 x float> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm512_mask_cvtbiasps_bf8(__W, __U, __A, __B);
 }
 
 __m128i test_mm512_maskz_cvtbiasps_bf8(__mmask16 __U, __m512i __A, __m512 __B) {
   // CHECK-LABEL: @test_mm512_maskz_cvtbiasps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.512(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8.512(<64 x i8> %{{.*}}, <16 x float> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm512_maskz_cvtbiasps_bf8(__U, __A, __B);
 }
 
@@ -439,19 +481,21 @@ __m128i test_mm512_maskz_cvtbiasps_bf8(__mmask16 __U, __m512i __A, __m512 __B) {
 
 __m128i test_mm_cvts_biasps_bf8(__m128i __A, __m128 __B) {
   // CHECK-LABEL: @test_mm_cvts_biasps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.128(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s.128(<16 x i8> %{{.*}}, <4 x float> %{{.*}})
   return _mm_cvts_biasps_bf8(__A, __B);
 }
 
 __m128i test_mm_mask_cvts_biasps_bf8(__m128i __W, __mmask8 __U, __m128i __A, __m128 __B) {
   // CHECK-LABEL: @test_mm_mask_cvts_biasps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.128(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s.128(<16 x i8> %{{.*}}, <4 x float> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm_mask_cvts_biasps_bf8(__W, __U, __A, __B);
 }
 
 __m128i test_mm_maskz_cvts_biasps_bf8(__mmask8 __U, __m128i __A, __m128 __B) {
   // CHECK-LABEL: @test_mm_maskz_cvts_biasps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.128(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s.128(<16 x i8> %{{.*}}, <4 x float> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm_maskz_cvts_biasps_bf8(__U, __A, __B);
 }
 
@@ -459,19 +503,21 @@ __m128i test_mm_maskz_cvts_biasps_bf8(__mmask8 __U, __m128i __A, __m128 __B) {
 
 __m128i test_mm256_cvts_biasps_bf8(__m256i __A, __m256 __B) {
   // CHECK-LABEL: @test_mm256_cvts_biasps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.256(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s.256(<32 x i8> %{{.*}}, <8 x float> %{{.*}})
   return _mm256_cvts_biasps_bf8(__A, __B);
 }
 
 __m128i test_mm256_mask_cvts_biasps_bf8(__m128i __W, __mmask8 __U, __m256i __A, __m256 __B) {
   // CHECK-LABEL: @test_mm256_mask_cvts_biasps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.256(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s.256(<32 x i8> %{{.*}}, <8 x float> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm256_mask_cvts_biasps_bf8(__W, __U, __A, __B);
 }
 
 __m128i test_mm256_maskz_cvts_biasps_bf8(__mmask8 __U, __m256i __A, __m256 __B) {
   // CHECK-LABEL: @test_mm256_maskz_cvts_biasps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.256(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s.256(<32 x i8> %{{.*}}, <8 x float> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm256_maskz_cvts_biasps_bf8(__U, __A, __B);
 }
 
@@ -479,19 +525,21 @@ __m128i test_mm256_maskz_cvts_biasps_bf8(__mmask8 __U, __m256i __A, __m256 __B)
 
 __m128i test_mm512_cvts_biasps_bf8(__m512i __A, __m512 __B) {
   // CHECK-LABEL: @test_mm512_cvts_biasps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.512(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s.512(<64 x i8> %{{.*}}, <16 x float> %{{.*}})
   return _mm512_cvts_biasps_bf8(__A, __B);
 }
 
 __m128i test_mm512_mask_cvts_biasps_bf8(__m128i __W, __mmask16 __U, __m512i __A, __m512 __B) {
   // CHECK-LABEL: @test_mm512_mask_cvts_biasps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.512(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s.512(<64 x i8> %{{.*}}, <16 x float> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm512_mask_cvts_biasps_bf8(__W, __U, __A, __B);
 }
 
 __m128i test_mm512_maskz_cvts_biasps_bf8(__mmask16 __U, __m512i __A, __m512 __B) {
   // CHECK-LABEL: @test_mm512_maskz_cvts_biasps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.512(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s.512(<64 x i8> %{{.*}}, <16 x float> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm512_maskz_cvts_biasps_bf8(__U, __A, __B);
 }
 
@@ -499,19 +547,21 @@ __m128i test_mm512_maskz_cvts_biasps_bf8(__mmask16 __U, __m512i __A, __m512 __B)
 
 __m128i test_mm_cvtbiasps_hf8(__m128i __A, __m128 __B) {
   // CHECK-LABEL: @test_mm_cvtbiasps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.128(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8.128(<16 x i8> %{{.*}}, <4 x float> %{{.*}})
   return _mm_cvtbiasps_hf8(__A, __B);
 }
 
 __m128i test_mm_mask_cvtbiasps_hf8(__m128i __W, __mmask8 __U, __m128i __A, __m128 __B) {
   // CHECK-LABEL: @test_mm_mask_cvtbiasps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.128(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8.128(<16 x i8> %{{.*}}, <4 x float> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm_mask_cvtbiasps_hf8(__W, __U, __A, __B);
 }
 
 __m128i test_mm_maskz_cvtbiasps_hf8(__mmask8 __U, __m128i __A, __m128 __B) {
   // CHECK-LABEL: @test_mm_maskz_cvtbiasps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.128(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8.128(<16 x i8> %{{.*}}, <4 x float> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm_maskz_cvtbiasps_hf8(__U, __A, __B);
 }
 
@@ -519,19 +569,21 @@ __m128i test_mm_maskz_cvtbiasps_hf8(__mmask8 __U, __m128i __A, __m128 __B) {
 
 __m128i test_mm256_cvtbiasps_hf8(__m256i __A, __m256 __B) {
   // CHECK-LABEL: @test_mm256_cvtbiasps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.256(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8.256(<32 x i8> %{{.*}}, <8 x float> %{{.*}})
   return _mm256_cvtbiasps_hf8(__A, __B);
 }
 
 __m128i test_mm256_mask_cvtbiasps_hf8(__m128i __W, __mmask8 __U, __m256i __A, __m256 __B) {
   // CHECK-LABEL: @test_mm256_mask_cvtbiasps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.256(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8.256(<32 x i8> %{{.*}}, <8 x float> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm256_mask_cvtbiasps_hf8(__W, __U, __A, __B);
 }
 
 __m128i test_mm256_maskz_cvtbiasps_hf8(__mmask8 __U, __m256i __A, __m256 __B) {
   // CHECK-LABEL: @test_mm256_maskz_cvtbiasps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.256(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8.256(<32 x i8> %{{.*}}, <8 x float> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm256_maskz_cvtbiasps_hf8(__U, __A, __B);
 }
 
@@ -539,19 +591,21 @@ __m128i test_mm256_maskz_cvtbiasps_hf8(__mmask8 __U, __m256i __A, __m256 __B) {
 
 __m128i test_mm512_cvtbiasps_hf8(__m512i __A, __m512 __B) {
   // CHECK-LABEL: @test_mm512_cvtbiasps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.512(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8.512(<64 x i8> %{{.*}}, <16 x float> %{{.*}})
   return _mm512_cvtbiasps_hf8(__A, __B);
 }
 
 __m128i test_mm512_mask_cvtbiasps_hf8(__m128i __W, __mmask16 __U, __m512i __A, __m512 __B) {
   // CHECK-LABEL: @test_mm512_mask_cvtbiasps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.512(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8.512(<64 x i8> %{{.*}}, <16 x float> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm512_mask_cvtbiasps_hf8(__W, __U, __A, __B);
 }
 
 __m128i test_mm512_maskz_cvtbiasps_hf8(__mmask16 __U, __m512i __A, __m512 __B) {
   // CHECK-LABEL: @test_mm512_maskz_cvtbiasps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.512(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8.512(<64 x i8> %{{.*}}, <16 x float> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm512_maskz_cvtbiasps_hf8(__U, __A, __B);
 }
 
@@ -559,19 +613,21 @@ __m128i test_mm512_maskz_cvtbiasps_hf8(__mmask16 __U, __m512i __A, __m512 __B) {
 
 __m128i test_mm_cvts_biasps_hf8(__m128i __A, __m128 __B) {
   // CHECK-LABEL: @test_mm_cvts_biasps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.128(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s.128(<16 x i8> %{{.*}}, <4 x float> %{{.*}})
   return _mm_cvts_biasps_hf8(__A, __B);
 }
 
 __m128i test_mm_mask_cvts_biasps_hf8(__m128i __W, __mmask8 __U, __m128i __A, __m128 __B) {
   // CHECK-LABEL: @test_mm_mask_cvts_biasps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.128(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s.128(<16 x i8> %{{.*}}, <4 x float> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm_mask_cvts_biasps_hf8(__W, __U, __A, __B);
 }
 
 __m128i test_mm_maskz_cvts_biasps_hf8(__mmask8 __U, __m128i __A, __m128 __B) {
   // CHECK-LABEL: @test_mm_maskz_cvts_biasps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.128(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s.128(<16 x i8> %{{.*}}, <4 x float> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm_maskz_cvts_biasps_hf8(__U, __A, __B);
 }
 
@@ -579,19 +635,21 @@ __m128i test_mm_maskz_cvts_biasps_hf8(__mmask8 __U, __m128i __A, __m128 __B) {
 
 __m128i test_mm256_cvts_biasps_hf8(__m256i __A, __m256 __B) {
   // CHECK-LABEL: @test_mm256_cvts_biasps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.256(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s.256(<32 x i8> %{{.*}}, <8 x float> %{{.*}})
   return _mm256_cvts_biasps_hf8(__A, __B);
 }
 
 __m128i test_mm256_mask_cvts_biasps_hf8(__m128i __W, __mmask8 __U, __m256i __A, __m256 __B) {
   // CHECK-LABEL: @test_mm256_mask_cvts_biasps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.256(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s.256(<32 x i8> %{{.*}}, <8 x float> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm256_mask_cvts_biasps_hf8(__W, __U, __A, __B);
 }
 
 __m128i test_mm256_maskz_cvts_biasps_hf8(__mmask8 __U, __m256i __A, __m256 __B) {
   // CHECK-LABEL: @test_mm256_maskz_cvts_biasps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.256(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s.256(<32 x i8> %{{.*}}, <8 x float> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm256_maskz_cvts_biasps_hf8(__U, __A, __B);
 }
 
@@ -599,19 +657,21 @@ __m128i test_mm256_maskz_cvts_biasps_hf8(__mmask8 __U, __m256i __A, __m256 __B)
 
 __m128i test_mm512_cvts_biasps_hf8(__m512i __A, __m512 __B) {
   // CHECK-LABEL: @test_mm512_cvts_biasps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.512(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s.512(<64 x i8> %{{.*}}, <16 x float> %{{.*}})
   return _mm512_cvts_biasps_hf8(__A, __B);
 }
 
 __m128i test_mm512_mask_cvts_biasps_hf8(__m128i __W, __mmask16 __U, __m512i __A, __m512 __B) {
   // CHECK-LABEL: @test_mm512_mask_cvts_biasps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.512(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s.512(<64 x i8> %{{.*}}, <16 x float> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm512_mask_cvts_biasps_hf8(__W, __U, __A, __B);
 }
 
 __m128i test_mm512_maskz_cvts_biasps_hf8(__mmask16 __U, __m512i __A, __m512 __B) {
   // CHECK-LABEL: @test_mm512_maskz_cvts_biasps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.512(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s.512(<64 x i8> %{{.*}}, <16 x float> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm512_maskz_cvts_biasps_hf8(__U, __A, __B);
 }
 
@@ -623,19 +683,21 @@ __m128i test_mm512_maskz_cvts_biasps_hf8(__mmask16 __U, __m512i __A, __m512 __B)
 
 __m128 test_mm_cvtbf8_ps(__m128i __A) {
   // CHECK-LABEL: @test_mm_cvtbf8_ps(
-  // CHECK: call <4 x float> @llvm.x86.avx10.mask.vcvtbf82ps.128(
+  // CHECK: call <4 x float> @llvm.x86.avx10.vcvtbf82ps.128(<16 x i8> %{{.*}})
   return _mm_cvtbf8_ps(__A);
 }
 
 __m128 test_mm_mask_cvtbf8_ps(__m128 __W, __mmask8 __U, __m128i __A) {
   // CHECK-LABEL: @test_mm_mask_cvtbf8_ps(
-  // CHECK: call <4 x float> @llvm.x86.avx10.mask.vcvtbf82ps.128(
+  // CHECK: [[RES:%.*]] = call <4 x float> @llvm.x86.avx10.vcvtbf82ps.128(<16 x i8> %{{.*}})
+  // CHECK: select <4 x i1> %{{.*}}, <4 x float> [[RES]], <4 x float> %{{.*}}
   return _mm_mask_cvtbf8_ps(__W, __U, __A);
 }
 
 __m128 test_mm_maskz_cvtbf8_ps(__mmask8 __U, __m128i __A) {
   // CHECK-LABEL: @test_mm_maskz_cvtbf8_ps(
-  // CHECK: call <4 x float> @llvm.x86.avx10.mask.vcvtbf82ps.128(
+  // CHECK: [[RES:%.*]] = call <4 x float> @llvm.x86.avx10.vcvtbf82ps.128(<16 x i8> %{{.*}})
+  // CHECK: select <4 x i1> %{{.*}}, <4 x float> [[RES]], <4 x float> %{{.*}}
   return _mm_maskz_cvtbf8_ps(__U, __A);
 }
 
@@ -643,19 +705,21 @@ __m128 test_mm_maskz_cvtbf8_ps(__mmask8 __U, __m128i __A) {
 
 __m256 test_mm256_cvtbf8_ps(__m128i __A) {
   // CHECK-LABEL: @test_mm256_cvtbf8_ps(
-  // CHECK: call <8 x float> @llvm.x86.avx10.mask.vcvtbf82ps.256(
+  // CHECK: call <8 x float> @llvm.x86.avx10.vcvtbf82ps.256(<16 x i8> %{{.*}})
   return _mm256_cvtbf8_ps(__A);
 }
 
 __m256 test_mm256_mask_cvtbf8_ps(__m256 __W, __mmask8 __U, __m128i __A) {
   // CHECK-LABEL: @test_mm256_mask_cvtbf8_ps(
-  // CHECK: call <8 x float> @llvm.x86.avx10.mask.vcvtbf82ps.256(
+  // CHECK: [[RES:%.*]] = call <8 x float> @llvm.x86.avx10.vcvtbf82ps.256(<16 x i8> %{{.*}})
+  // CHECK: select <8 x i1> %{{.*}}, <8 x float> [[RES]], <8 x float> %{{.*}}
   return _mm256_mask_cvtbf8_ps(__W, __U, __A);
 }
 
 __m256 test_mm256_maskz_cvtbf8_ps(__mmask8 __U, __m128i __A) {
   // CHECK-LABEL: @test_mm256_maskz_cvtbf8_ps(
-  // CHECK: call <8 x float> @llvm.x86.avx10.mask.vcvtbf82ps.256(
+  // CHECK: [[RES:%.*]] = call <8 x float> @llvm.x86.avx10.vcvtbf82ps.256(<16 x i8> %{{.*}})
+  // CHECK: select <8 x i1> %{{.*}}, <8 x float> [[RES]], <8 x float> %{{.*}}
   return _mm256_maskz_cvtbf8_ps(__U, __A);
 }
 
@@ -663,19 +727,21 @@ __m256 test_mm256_maskz_cvtbf8_ps(__mmask8 __U, __m128i __A) {
 
 __m512 test_mm512_cvtbf8_ps(__m128i __A) {
   // CHECK-LABEL: @test_mm512_cvtbf8_ps(
-  // CHECK: call <16 x float> @llvm.x86.avx10.mask.vcvtbf82ps.512(
+  // CHECK: call <16 x float> @llvm.x86.avx10.vcvtbf82ps.512(<16 x i8> %{{.*}})
   return _mm512_cvtbf8_ps(__A);
 }
 
 __m512 test_mm512_mask_cvtbf8_ps(__m512 __W, __mmask16 __U, __m128i __A) {
   // CHECK-LABEL: @test_mm512_mask_cvtbf8_ps(
-  // CHECK: call <16 x float> @llvm.x86.avx10.mask.vcvtbf82ps.512(
+  // CHECK: [[RES:%.*]] = call <16 x float> @llvm.x86.avx10.vcvtbf82ps.512(<16 x i8> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x float> [[RES]], <16 x float> %{{.*}}
   return _mm512_mask_cvtbf8_ps(__W, __U, __A);
 }
 
 __m512 test_mm512_maskz_cvtbf8_ps(__mmask16 __U, __m128i __A) {
   // CHECK-LABEL: @test_mm512_maskz_cvtbf8_ps(
-  // CHECK: call <16 x float> @llvm.x86.avx10.mask.vcvtbf82ps.512(
+  // CHECK: [[RES:%.*]] = call <16 x float> @llvm.x86.avx10.vcvtbf82ps.512(<16 x i8> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x float> [[RES]], <16 x float> %{{.*}}
   return _mm512_maskz_cvtbf8_ps(__U, __A);
 }
 
@@ -683,19 +749,21 @@ __m512 test_mm512_maskz_cvtbf8_ps(__mmask16 __U, __m128i __A) {
 
 __m128 test_mm_cvthf8_ps(__m128i __A) {
   // CHECK-LABEL: @test_mm_cvthf8_ps(
-  // CHECK: call <4 x float> @llvm.x86.avx10.mask.vcvthf82ps.128(
+  // CHECK: call <4 x float> @llvm.x86.avx10.vcvthf82ps.128(<16 x i8> %{{.*}})
   return _mm_cvthf8_ps(__A);
 }
 
 __m128 test_mm_mask_cvthf8_ps(__m128 __W, __mmask8 __U, __m128i __A) {
   // CHECK-LABEL: @test_mm_mask_cvthf8_ps(
-  // CHECK: call <4 x float> @llvm.x86.avx10.mask.vcvthf82ps.128(
+  // CHECK: [[RES:%.*]] = call <4 x float> @llvm.x86.avx10.vcvthf82ps.128(<16 x i8> %{{.*}})
+  // CHECK: select <4 x i1> %{{.*}}, <4 x float> [[RES]], <4 x float> %{{.*}}
   return _mm_mask_cvthf8_ps(__W, __U, __A);
 }
 
 __m128 test_mm_maskz_cvthf8_ps(__mmask8 __U, __m128i __A) {
   // CHECK-LABEL: @test_mm_maskz_cvthf8_ps(
-  // CHECK: call <4 x float> @llvm.x86.avx10.mask.vcvthf82ps.128(
+  // CHECK: [[RES:%.*]] = call <4 x float> @llvm.x86.avx10.vcvthf82ps.128(<16 x i8> %{{.*}})
+  // CHECK: select <4 x i1> %{{.*}}, <4 x float> [[RES]], <4 x float> %{{.*}}
   return _mm_maskz_cvthf8_ps(__U, __A);
 }
 
@@ -703,19 +771,21 @@ __m128 test_mm_maskz_cvthf8_ps(__mmask8 __U, __m128i __A) {
 
 __m256 test_mm256_cvthf8_ps(__m128i __A) {
   // CHECK-LABEL: @test_mm256_cvthf8_ps(
-  // CHECK: call <8 x float> @llvm.x86.avx10.mask.vcvthf82ps.256(
+  // CHECK: call <8 x float> @llvm.x86.avx10.vcvthf82ps.256(<16 x i8> %{{.*}})
   return _mm256_cvthf8_ps(__A);
 }
 
 __m256 test_mm256_mask_cvthf8_ps(__m256 __W, __mmask8 __U, __m128i __A) {
   // CHECK-LABEL: @test_mm256_mask_cvthf8_ps(
-  // CHECK: call <8 x float> @llvm.x86.avx10.mask.vcvthf82ps.256(
+  // CHECK: [[RES:%.*]] = call <8 x float> @llvm.x86.avx10.vcvthf82ps.256(<16 x i8> %{{.*}})
+  // CHECK: select <8 x i1> %{{.*}}, <8 x float> [[RES]], <8 x float> %{{.*}}
   return _mm256_mask_cvthf8_ps(__W, __U, __A);
 }
 
 __m256 test_mm256_maskz_cvthf8_ps(__mmask8 __U, __m128i __A) {
   // CHECK-LABEL: @test_mm256_maskz_cvthf8_ps(
-  // CHECK: call <8 x float> @llvm.x86.avx10.mask.vcvthf82ps.256(
+  // CHECK: [[RES:%.*]] = call <8 x float> @llvm.x86.avx10.vcvthf82ps.256(<16 x i8> %{{.*}})
+  // CHECK: select <8 x i1> %{{.*}}, <8 x float> [[RES]], <8 x float> %{{.*}}
   return _mm256_maskz_cvthf8_ps(__U, __A);
 }
 
@@ -723,22 +793,108 @@ __m256 test_mm256_maskz_cvthf8_ps(__mmask8 __U, __m128i __A) {
 
 __m512 test_mm512_cvthf8_ps(__m128i __A) {
   // CHECK-LABEL: @test_mm512_cvthf8_ps(
-  // CHECK: call <16 x float> @llvm.x86.avx10.mask.vcvthf82ps.512(
+  // CHECK: call <16 x float> @llvm.x86.avx10.vcvthf82ps.512(<16 x i8> %{{.*}})
   return _mm512_cvthf8_ps(__A);
 }
 
 __m512 test_mm512_mask_cvthf8_ps(__m512 __W, __mmask16 __U, __m128i __A) {
   // CHECK-LABEL: @test_mm512_mask_cvthf8_ps(
-  // CHECK: call <16 x float> @llvm.x86.avx10.mask.vcvthf82ps.512(
+  // CHECK: [[RES:%.*]] = call <16 x float> @llvm.x86.avx10.vcvthf82ps.512(<16 x i8> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x float> [[RES]], <16 x float> %{{.*}}
   return _mm512_mask_cvthf8_ps(__W, __U, __A);
 }
 
 __m512 test_mm512_maskz_cvthf8_ps(__mmask16 __U, __m128i __A) {
   // CHECK-LABEL: @test_mm512_maskz_cvthf8_ps(
-  // CHECK: call <16 x float> @llvm.x86.avx10.mask.vcvthf82ps.512(
+  // CHECK: [[RES:%.*]] = call <16 x float> @llvm.x86.avx10.vcvthf82ps.512(<16 x i8> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x float> [[RES]], <16 x float> %{{.*}}
   return _mm512_maskz_cvthf8_ps(__U, __A);
 }
 
+//
+// Group D: VCVTBF82BF4S / VCVTHF82BF4S (FP8 to FP4 truncating conversions)
+//
+
+// VCVTBF82BF4S - register forms
+
+__m128i test_mm_cvtbf8_bf4s(__m128i __A) {
+  // CHECK-LABEL: @test_mm_cvtbf8_bf4s(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbf82bf4s.128(<16 x i8> %{{.*}})
+  return _mm_cvtbf8_bf4s(__A);
+}
+
+__m128i test_mm256_cvtbf8_bf4s(__m256i __A) {
+  // CHECK-LABEL: @test_mm256_cvtbf8_bf4s(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbf82bf4s.256(<32 x i8> %{{.*}})
+  return _mm256_cvtbf8_bf4s(__A);
+}
+
+__m256i test_mm512_cvtbf8_bf4s(__m512i __A) {
+  // CHECK-LABEL: @test_mm512_cvtbf8_bf4s(
+  // CHECK: call <32 x i8> @llvm.x86.avx10.vcvtbf82bf4s.512(<64 x i8> %{{.*}})
+  return _mm512_cvtbf8_bf4s(__A);
+}
+
+// VCVTHF82BF4S - register forms
+
+__m128i test_mm_cvthf8_bf4s(__m128i __A) {
+  // CHECK-LABEL: @test_mm_cvthf8_bf4s(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvthf82bf4s.128(<16 x i8> %{{.*}})
+  return _mm_cvthf8_bf4s(__A);
+}
+
+__m128i test_mm256_cvthf8_bf4s(__m256i __A) {
+  // CHECK-LABEL: @test_mm256_cvthf8_bf4s(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvthf82bf4s.256(<32 x i8> %{{.*}})
+  return _mm256_cvthf8_bf4s(__A);
+}
+
+__m256i test_mm512_cvthf8_bf4s(__m512i __A) {
+  // CHECK-LABEL: @test_mm512_cvthf8_bf4s(
+  // CHECK: call <32 x i8> @llvm.x86.avx10.vcvthf82bf4s.512(<64 x i8> %{{.*}})
+  return _mm512_cvthf8_bf4s(__A);
+}
+
+// VCVTBF82BF4S - memory store forms
+
+void test_mm_cvtbf8_bf4s_storeu(void *__P, __m128i __A) {
+  // CHECK-LABEL: @test_mm_cvtbf8_bf4s_storeu(
+  // CHECK: call void @llvm.x86.avx10.vcvtbf82bf4s.128.mem(ptr %{{.*}}, <16 x i8> %{{.*}})
+  _mm_cvtbf8_bf4s_storeu(__P, __A);
+}
+
+void test_mm256_cvtbf8_bf4s_storeu(void *__P, __m256i __A) {
+  // CHECK-LABEL: @test_mm256_cvtbf8_bf4s_storeu(
+  // CHECK: call void @llvm.x86.avx10.vcvtbf82bf4s.256.mem(ptr %{{.*}}, <32 x i8> %{{.*}})
+  _mm256_cvtbf8_bf4s_storeu(__P, __A);
+}
+
+void test_mm512_cvtbf8_bf4s_storeu(void *__P, __m512i __A) {
+  // CHECK-LABEL: @test_mm512_cvtbf8_bf4s_storeu(
+  // CHECK: call void @llvm.x86.avx10.vcvtbf82bf4s.512.mem(ptr %{{.*}}, <64 x i8> %{{.*}})
+  _mm512_cvtbf8_bf4s_storeu(__P, __A);
+}
+
+// VCVTHF82BF4S - memory store forms
+
+void test_mm_cvthf8_bf4s_storeu(void *__P, __m128i __A) {
+  // CHECK-LABEL: @test_mm_cvthf8_bf4s_storeu(
+  // CHECK: call void @llvm.x86.avx10.vcvthf82bf4s.128.mem(ptr %{{.*}}, <16 x i8> %{{.*}})
+  _mm_cvthf8_bf4s_storeu(__P, __A);
+}
+
+void test_mm256_cvthf8_bf4s_storeu(void *__P, __m256i __A) {
+  // CHECK-LABEL: @test_mm256_cvthf8_bf4s_storeu(
+  // CHECK: call void @llvm.x86.avx10.vcvthf82bf4s.256.mem(ptr %{{.*}}, <32 x i8> %{{.*}})
+  _mm256_cvthf8_bf4s_storeu(__P, __A);
+}
+
+void test_mm512_cvthf8_bf4s_storeu(void *__P, __m512i __A) {
+  // CHECK-LABEL: @test_mm512_cvthf8_bf4s_storeu(
+  // CHECK: call void @llvm.x86.avx10.vcvthf82bf4s.512.mem(ptr %{{.*}}, <64 x i8> %{{.*}})
+  _mm512_cvthf8_bf4s_storeu(__P, __A);
+}
+
 //
 // Group E: VCVTBF82BF6S / VCVTHF82HF6S
 //
@@ -747,19 +903,19 @@ __m512 test_mm512_maskz_cvthf8_ps(__mmask16 __U, __m128i __A) {
 
 __m128i test_mm_cvtbf8_bf6s(__m128i __A) {
   // CHECK-LABEL: @test_mm_cvtbf8_bf6s(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbf82bf6s.128(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbf82bf6s.128(<16 x i8> %{{.*}})
   return _mm_cvtbf8_bf6s(__A);
 }
 
 __m256i test_mm256_cvtbf8_bf6s(__m256i __A) {
   // CHECK-LABEL: @test_mm256_cvtbf8_bf6s(
-  // CHECK: call <32 x i8> @llvm.x86.avx10.vcvtbf82bf6s.256(
+  // CHECK: call <32 x i8> @llvm.x86.avx10.vcvtbf82bf6s.256(<32 x i8> %{{.*}})
   return _mm256_cvtbf8_bf6s(__A);
 }
 
 __m512i test_mm512_cvtbf8_bf6s(__m512i __A) {
   // CHECK-LABEL: @test_mm512_cvtbf8_bf6s(
-  // CHECK: call <64 x i8> @llvm.x86.avx10.vcvtbf82bf6s.512(
+  // CHECK: call <64 x i8> @llvm.x86.avx10.vcvtbf82bf6s.512(<64 x i8> %{{.*}})
   return _mm512_cvtbf8_bf6s(__A);
 }
 
@@ -767,19 +923,19 @@ __m512i test_mm512_cvtbf8_bf6s(__m512i __A) {
 
 __m128i test_mm_cvthf8_hf6s(__m128i __A) {
   // CHECK-LABEL: @test_mm_cvthf8_hf6s(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvthf82hf6s.128(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvthf82hf6s.128(<16 x i8> %{{.*}})
   return _mm_cvthf8_hf6s(__A);
 }
 
 __m256i test_mm256_cvthf8_hf6s(__m256i __A) {
   // CHECK-LABEL: @test_mm256_cvthf8_hf6s(
-  // CHECK: call <32 x i8> @llvm.x86.avx10.vcvthf82hf6s.256(
+  // CHECK: call <32 x i8> @llvm.x86.avx10.vcvthf82hf6s.256(<32 x i8> %{{.*}})
   return _mm256_cvthf8_hf6s(__A);
 }
 
 __m512i test_mm512_cvthf8_hf6s(__m512i __A) {
   // CHECK-LABEL: @test_mm512_cvthf8_hf6s(
-  // CHECK: call <64 x i8> @llvm.x86.avx10.vcvthf82hf6s.512(
+  // CHECK: call <64 x i8> @llvm.x86.avx10.vcvthf82hf6s.512(<64 x i8> %{{.*}})
   return _mm512_cvthf8_hf6s(__A);
 }
 
@@ -791,21 +947,21 @@ __m512i test_mm512_cvthf8_hf6s(__m512i __A) {
 
 __m128i test_mm_cvtbf4_hf8(__m128i __A) {
   // CHECK-LABEL: @test_mm_cvtbf4_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbf42hf8.128(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbf42hf8.128(<16 x i8> %{{.*}})
   return _mm_cvtbf4_hf8(__A);
 }
 
 __m128i test_mm_mask_cvtbf4_hf8(__m128i __W, __mmask16 __U, __m128i __A) {
   // CHECK-LABEL: @test_mm_mask_cvtbf4_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbf42hf8.128(
-  // CHECK: select <16 x i1>
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbf42hf8.128(<16 x i8> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm_mask_cvtbf4_hf8(__W, __U, __A);
 }
 
 __m128i test_mm_maskz_cvtbf4_hf8(__mmask16 __U, __m128i __A) {
   // CHECK-LABEL: @test_mm_maskz_cvtbf4_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbf42hf8.128(
-  // CHECK: select <16 x i1>
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbf42hf8.128(<16 x i8> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm_maskz_cvtbf4_hf8(__U, __A);
 }
 
@@ -813,21 +969,21 @@ __m128i test_mm_maskz_cvtbf4_hf8(__mmask16 __U, __m128i __A) {
 
 __m256i test_mm256_cvtbf4_hf8(__m128i __A) {
   // CHECK-LABEL: @test_mm256_cvtbf4_hf8(
-  // CHECK: call <32 x i8> @llvm.x86.avx10.vcvtbf42hf8.256(
+  // CHECK: call <32 x i8> @llvm.x86.avx10.vcvtbf42hf8.256(<16 x i8> %{{.*}})
   return _mm256_cvtbf4_hf8(__A);
 }
 
 __m256i test_mm256_mask_cvtbf4_hf8(__m256i __W, __mmask32 __U, __m128i __A) {
   // CHECK-LABEL: @test_mm256_mask_cvtbf4_hf8(
-  // CHECK: call <32 x i8> @llvm.x86.avx10.vcvtbf42hf8.256(
-  // CHECK: select <32 x i1>
+  // CHECK: [[RES:%.*]] = call <32 x i8> @llvm.x86.avx10.vcvtbf42hf8.256(<16 x i8> %{{.*}})
+  // CHECK: select <32 x i1> %{{.*}}, <32 x i8> [[RES]], <32 x i8> %{{.*}}
   return _mm256_mask_cvtbf4_hf8(__W, __U, __A);
 }
 
 __m256i test_mm256_maskz_cvtbf4_hf8(__mmask32 __U, __m128i __A) {
   // CHECK-LABEL: @test_mm256_maskz_cvtbf4_hf8(
-  // CHECK: call <32 x i8> @llvm.x86.avx10.vcvtbf42hf8.256(
-  // CHECK: select <32 x i1>
+  // CHECK: [[RES:%.*]] = call <32 x i8> @llvm.x86.avx10.vcvtbf42hf8.256(<16 x i8> %{{.*}})
+  // CHECK: select <32 x i1> %{{.*}}, <32 x i8> [[RES]], <32 x i8> %{{.*}}
   return _mm256_maskz_cvtbf4_hf8(__U, __A);
 }
 
@@ -835,21 +991,21 @@ __m256i test_mm256_maskz_cvtbf4_hf8(__mmask32 __U, __m128i __A) {
 
 __m512i test_mm512_cvtbf4_hf8(__m256i __A) {
   // CHECK-LABEL: @test_mm512_cvtbf4_hf8(
-  // CHECK: call <64 x i8> @llvm.x86.avx10.vcvtbf42hf8.512(
+  // CHECK: call <64 x i8> @llvm.x86.avx10.vcvtbf42hf8.512(<32 x i8> %{{.*}})
   return _mm512_cvtbf4_hf8(__A);
 }
 
 __m512i test_mm512_mask_cvtbf4_hf8(__m512i __W, __mmask64 __U, __m256i __A) {
   // CHECK-LABEL: @test_mm512_mask_cvtbf4_hf8(
-  // CHECK: call <64 x i8> @llvm.x86.avx10.vcvtbf42hf8.512(
-  // CHECK: select <64 x i1>
+  // CHECK: [[RES:%.*]] = call <64 x i8> @llvm.x86.avx10.vcvtbf42hf8.512(<32 x i8> %{{.*}})
+  // CHECK: select <64 x i1> %{{.*}}, <64 x i8> [[RES]], <64 x i8> %{{.*}}
   return _mm512_mask_cvtbf4_hf8(__W, __U, __A);
 }
 
 __m512i test_mm512_maskz_cvtbf4_hf8(__mmask64 __U, __m256i __A) {
   // CHECK-LABEL: @test_mm512_maskz_cvtbf4_hf8(
-  // CHECK: call <64 x i8> @llvm.x86.avx10.vcvtbf42hf8.512(
-  // CHECK: select <64 x i1>
+  // CHECK: [[RES:%.*]] = call <64 x i8> @llvm.x86.avx10.vcvtbf42hf8.512(<32 x i8> %{{.*}})
+  // CHECK: select <64 x i1> %{{.*}}, <64 x i8> [[RES]], <64 x i8> %{{.*}}
   return _mm512_maskz_cvtbf4_hf8(__U, __A);
 }
 
@@ -857,21 +1013,21 @@ __m512i test_mm512_maskz_cvtbf4_hf8(__mmask64 __U, __m256i __A) {
 
 __m128i test_mm_cvtbf6_hf8(__m128i __A) {
   // CHECK-LABEL: @test_mm_cvtbf6_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbf62hf8.128(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbf62hf8.128(<16 x i8> %{{.*}})
   return _mm_cvtbf6_hf8(__A);
 }
 
 __m128i test_mm_mask_cvtbf6_hf8(__m128i __W, __mmask16 __U, __m128i __A) {
   // CHECK-LABEL: @test_mm_mask_cvtbf6_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbf62hf8.128(
-  // CHECK: select <16 x i1>
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbf62hf8.128(<16 x i8> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm_mask_cvtbf6_hf8(__W, __U, __A);
 }
 
 __m128i test_mm_maskz_cvtbf6_hf8(__mmask16 __U, __m128i __A) {
   // CHECK-LABEL: @test_mm_maskz_cvtbf6_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbf62hf8.128(
-  // CHECK: select <16 x i1>
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbf62hf8.128(<16 x i8> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm_maskz_cvtbf6_hf8(__U, __A);
 }
 
@@ -879,21 +1035,21 @@ __m128i test_mm_maskz_cvtbf6_hf8(__mmask16 __U, __m128i __A) {
 
 __m256i test_mm256_cvtbf6_hf8(__m256i __A) {
   // CHECK-LABEL: @test_mm256_cvtbf6_hf8(
-  // CHECK: call <32 x i8> @llvm.x86.avx10.vcvtbf62hf8.256(
+  // CHECK: call <32 x i8> @llvm.x86.avx10.vcvtbf62hf8.256(<32 x i8> %{{.*}})
   return _mm256_cvtbf6_hf8(__A);
 }
 
 __m256i test_mm256_mask_cvtbf6_hf8(__m256i __W, __mmask32 __U, __m256i __A) {
   // CHECK-LABEL: @test_mm256_mask_cvtbf6_hf8(
-  // CHECK: call <32 x i8> @llvm.x86.avx10.vcvtbf62hf8.256(
-  // CHECK: select <32 x i1>
+  // CHECK: [[RES:%.*]] = call <32 x i8> @llvm.x86.avx10.vcvtbf62hf8.256(<32 x i8> %{{.*}})
+  // CHECK: select <32 x i1> %{{.*}}, <32 x i8> [[RES]], <32 x i8> %{{.*}}
   return _mm256_mask_cvtbf6_hf8(__W, __U, __A);
 }
 
 __m256i test_mm256_maskz_cvtbf6_hf8(__mmask32 __U, __m256i __A) {
   // CHECK-LABEL: @test_mm256_maskz_cvtbf6_hf8(
-  // CHECK: call <32 x i8> @llvm.x86.avx10.vcvtbf62hf8.256(
-  // CHECK: select <32 x i1>
+  // CHECK: [[RES:%.*]] = call <32 x i8> @llvm.x86.avx10.vcvtbf62hf8.256(<32 x i8> %{{.*}})
+  // CHECK: select <32 x i1> %{{.*}}, <32 x i8> [[RES]], <32 x i8> %{{.*}}
   return _mm256_maskz_cvtbf6_hf8(__U, __A);
 }
 
@@ -901,21 +1057,21 @@ __m256i test_mm256_maskz_cvtbf6_hf8(__mmask32 __U, __m256i __A) {
 
 __m512i test_mm512_cvtbf6_hf8(__m512i __A) {
   // CHECK-LABEL: @test_mm512_cvtbf6_hf8(
-  // CHECK: call <64 x i8> @llvm.x86.avx10.vcvtbf62hf8.512(
+  // CHECK: call <64 x i8> @llvm.x86.avx10.vcvtbf62hf8.512(<64 x i8> %{{.*}})
   return _mm512_cvtbf6_hf8(__A);
 }
 
 __m512i test_mm512_mask_cvtbf6_hf8(__m512i __W, __mmask64 __U, __m512i __A) {
   // CHECK-LABEL: @test_mm512_mask_cvtbf6_hf8(
-  // CHECK: call <64 x i8> @llvm.x86.avx10.vcvtbf62hf8.512(
-  // CHECK: select <64 x i1>
+  // CHECK: [[RES:%.*]] = call <64 x i8> @llvm.x86.avx10.vcvtbf62hf8.512(<64 x i8> %{{.*}})
+  // CHECK: select <64 x i1> %{{.*}}, <64 x i8> [[RES]], <64 x i8> %{{.*}}
   return _mm512_mask_cvtbf6_hf8(__W, __U, __A);
 }
 
 __m512i test_mm512_maskz_cvtbf6_hf8(__mmask64 __U, __m512i __A) {
   // CHECK-LABEL: @test_mm512_maskz_cvtbf6_hf8(
-  // CHECK: call <64 x i8> @llvm.x86.avx10.vcvtbf62hf8.512(
-  // CHECK: select <64 x i1>
+  // CHECK: [[RES:%.*]] = call <64 x i8> @llvm.x86.avx10.vcvtbf62hf8.512(<64 x i8> %{{.*}})
+  // CHECK: select <64 x i1> %{{.*}}, <64 x i8> [[RES]], <64 x i8> %{{.*}}
   return _mm512_maskz_cvtbf6_hf8(__U, __A);
 }
 
@@ -923,21 +1079,21 @@ __m512i test_mm512_maskz_cvtbf6_hf8(__mmask64 __U, __m512i __A) {
 
 __m128i test_mm_cvthf6_hf8(__m128i __A) {
   // CHECK-LABEL: @test_mm_cvthf6_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvthf62hf8.128(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvthf62hf8.128(<16 x i8> %{{.*}})
   return _mm_cvthf6_hf8(__A);
 }
 
 __m128i test_mm_mask_cvthf6_hf8(__m128i __W, __mmask16 __U, __m128i __A) {
   // CHECK-LABEL: @test_mm_mask_cvthf6_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvthf62hf8.128(
-  // CHECK: select <16 x i1>
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvthf62hf8.128(<16 x i8> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm_mask_cvthf6_hf8(__W, __U, __A);
 }
 
 __m128i test_mm_maskz_cvthf6_hf8(__mmask16 __U, __m128i __A) {
   // CHECK-LABEL: @test_mm_maskz_cvthf6_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvthf62hf8.128(
-  // CHECK: select <16 x i1>
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvthf62hf8.128(<16 x i8> %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm_maskz_cvthf6_hf8(__U, __A);
 }
 
@@ -945,21 +1101,21 @@ __m128i test_mm_maskz_cvthf6_hf8(__mmask16 __U, __m128i __A) {
 
 __m256i test_mm256_cvthf6_hf8(__m256i __A) {
   // CHECK-LABEL: @test_mm256_cvthf6_hf8(
-  // CHECK: call <32 x i8> @llvm.x86.avx10.vcvthf62hf8.256(
+  // CHECK: call <32 x i8> @llvm.x86.avx10.vcvthf62hf8.256(<32 x i8> %{{.*}})
   return _mm256_cvthf6_hf8(__A);
 }
 
 __m256i test_mm256_mask_cvthf6_hf8(__m256i __W, __mmask32 __U, __m256i __A) {
   // CHECK-LABEL: @test_mm256_mask_cvthf6_hf8(
-  // CHECK: call <32 x i8> @llvm.x86.avx10.vcvthf62hf8.256(
-  // CHECK: select <32 x i1>
+  // CHECK: [[RES:%.*]] = call <32 x i8> @llvm.x86.avx10.vcvthf62hf8.256(<32 x i8> %{{.*}})
+  // CHECK: select <32 x i1> %{{.*}}, <32 x i8> [[RES]], <32 x i8> %{{.*}}
   return _mm256_mask_cvthf6_hf8(__W, __U, __A);
 }
 
 __m256i test_mm256_maskz_cvthf6_hf8(__mmask32 __U, __m256i __A) {
   // CHECK-LABEL: @test_mm256_maskz_cvthf6_hf8(
-  // CHECK: call <32 x i8> @llvm.x86.avx10.vcvthf62hf8.256(
-  // CHECK: select <32 x i1>
+  // CHECK: [[RES:%.*]] = call <32 x i8> @llvm.x86.avx10.vcvthf62hf8.256(<32 x i8> %{{.*}})
+  // CHECK: select <32 x i1> %{{.*}}, <32 x i8> [[RES]], <32 x i8> %{{.*}}
   return _mm256_maskz_cvthf6_hf8(__U, __A);
 }
 
@@ -967,21 +1123,21 @@ __m256i test_mm256_maskz_cvthf6_hf8(__mmask32 __U, __m256i __A) {
 
 __m512i test_mm512_cvthf6_hf8(__m512i __A) {
   // CHECK-LABEL: @test_mm512_cvthf6_hf8(
-  // CHECK: call <64 x i8> @llvm.x86.avx10.vcvthf62hf8.512(
+  // CHECK: call <64 x i8> @llvm.x86.avx10.vcvthf62hf8.512(<64 x i8> %{{.*}})
   return _mm512_cvthf6_hf8(__A);
 }
 
 __m512i test_mm512_mask_cvthf6_hf8(__m512i __W, __mmask64 __U, __m512i __A) {
   // CHECK-LABEL: @test_mm512_mask_cvthf6_hf8(
-  // CHECK: call <64 x i8> @llvm.x86.avx10.vcvthf62hf8.512(
-  // CHECK: select <64 x i1>
+  // CHECK: [[RES:%.*]] = call <64 x i8> @llvm.x86.avx10.vcvthf62hf8.512(<64 x i8> %{{.*}})
+  // CHECK: select <64 x i1> %{{.*}}, <64 x i8> [[RES]], <64 x i8> %{{.*}}
   return _mm512_mask_cvthf6_hf8(__W, __U, __A);
 }
 
 __m512i test_mm512_maskz_cvthf6_hf8(__mmask64 __U, __m512i __A) {
   // CHECK-LABEL: @test_mm512_maskz_cvthf6_hf8(
-  // CHECK: call <64 x i8> @llvm.x86.avx10.vcvthf62hf8.512(
-  // CHECK: select <64 x i1>
+  // CHECK: [[RES:%.*]] = call <64 x i8> @llvm.x86.avx10.vcvthf62hf8.512(<64 x i8> %{{.*}})
+  // CHECK: select <64 x i1> %{{.*}}, <64 x i8> [[RES]], <64 x i8> %{{.*}}
   return _mm512_maskz_cvthf6_hf8(__U, __A);
 }
 
@@ -993,21 +1149,21 @@ __m512i test_mm512_maskz_cvthf6_hf8(__mmask64 __U, __m512i __A) {
 
 __m128i test_mm_unpackb_epi8(__m128i __A) {
   // CHECK-LABEL: @test_mm_unpackb_epi8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vunpackb.128(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vunpackb.128(<16 x i8> %{{.*}}, i8 1)
   return _mm_unpackb_epi8(__A, 1);
 }
 
 __m128i test_mm_mask_unpackb_epi8(__m128i __W, __mmask16 __U, __m128i __A) {
   // CHECK-LABEL: @test_mm_mask_unpackb_epi8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vunpackb.128(
-  // CHECK: select <16 x i1>
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vunpackb.128(<16 x i8> %{{.*}}, i8 1)
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> %{{.*}}, <16 x i8> %{{.*}}
   return _mm_mask_unpackb_epi8(__W, __U, __A, 1);
 }
 
 __m128i test_mm_maskz_unpackb_epi8(__mmask16 __U, __m128i __A) {
   // CHECK-LABEL: @test_mm_maskz_unpackb_epi8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vunpackb.128(
-  // CHECK: select <16 x i1>
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vunpackb.128(<16 x i8> %{{.*}}, i8 1)
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> %{{.*}}, <16 x i8> %{{.*}}
   return _mm_maskz_unpackb_epi8(__U, __A, 1);
 }
 
@@ -1015,21 +1171,21 @@ __m128i test_mm_maskz_unpackb_epi8(__mmask16 __U, __m128i __A) {
 
 __m256i test_mm256_unpackb_epi8(__m256i __A) {
   // CHECK-LABEL: @test_mm256_unpackb_epi8(
-  // CHECK: call <32 x i8> @llvm.x86.avx10.vunpackb.256(
+  // CHECK: call <32 x i8> @llvm.x86.avx10.vunpackb.256(<32 x i8> %{{.*}}, i8 2)
   return _mm256_unpackb_epi8(__A, 2);
 }
 
 __m256i test_mm256_mask_unpackb_epi8(__m256i __W, __mmask32 __U, __m256i __A) {
   // CHECK-LABEL: @test_mm256_mask_unpackb_epi8(
-  // CHECK: call <32 x i8> @llvm.x86.avx10.vunpackb.256(
-  // CHECK: select <32 x i1>
+  // CHECK: call <32 x i8> @llvm.x86.avx10.vunpackb.256(<32 x i8> %{{.*}}, i8 2)
+  // CHECK: select <32 x i1> %{{.*}}, <32 x i8> %{{.*}}, <32 x i8> %{{.*}}
   return _mm256_mask_unpackb_epi8(__W, __U, __A, 2);
 }
 
 __m256i test_mm256_maskz_unpackb_epi8(__mmask32 __U, __m256i __A) {
   // CHECK-LABEL: @test_mm256_maskz_unpackb_epi8(
-  // CHECK: call <32 x i8> @llvm.x86.avx10.vunpackb.256(
-  // CHECK: select <32 x i1>
+  // CHECK: call <32 x i8> @llvm.x86.avx10.vunpackb.256(<32 x i8> %{{.*}}, i8 2)
+  // CHECK: select <32 x i1> %{{.*}}, <32 x i8> %{{.*}}, <32 x i8> %{{.*}}
   return _mm256_maskz_unpackb_epi8(__U, __A, 2);
 }
 
@@ -1037,20 +1193,42 @@ __m256i test_mm256_maskz_unpackb_epi8(__mmask32 __U, __m256i __A) {
 
 __m512i test_mm512_unpackb_epi8(__m512i __A) {
   // CHECK-LABEL: @test_mm512_unpackb_epi8(
-  // CHECK: call <64 x i8> @llvm.x86.avx10.vunpackb.512(
+  // CHECK: call <64 x i8> @llvm.x86.avx10.vunpackb.512(<64 x i8> %{{.*}}, i8 3)
   return _mm512_unpackb_epi8(__A, 3);
 }
 
 __m512i test_mm512_mask_unpackb_epi8(__m512i __W, __mmask64 __U, __m512i __A) {
   // CHECK-LABEL: @test_mm512_mask_unpackb_epi8(
-  // CHECK: call <64 x i8> @llvm.x86.avx10.vunpackb.512(
-  // CHECK: select <64 x i1>
+  // CHECK: call <64 x i8> @llvm.x86.avx10.vunpackb.512(<64 x i8> %{{.*}}, i8 3)
+  // CHECK: select <64 x i1> %{{.*}}, <64 x i8> %{{.*}}, <64 x i8> %{{.*}}
   return _mm512_mask_unpackb_epi8(__W, __U, __A, 3);
 }
 
 __m512i test_mm512_maskz_unpackb_epi8(__mmask64 __U, __m512i __A) {
   // CHECK-LABEL: @test_mm512_maskz_unpackb_epi8(
-  // CHECK: call <64 x i8> @llvm.x86.avx10.vunpackb.512(
-  // CHECK: select <64 x i1>
+  // CHECK: call <64 x i8> @llvm.x86.avx10.vunpackb.512(<64 x i8> %{{.*}}, i8 3)
+  // CHECK: select <64 x i1> %{{.*}}, <64 x i8> %{{.*}}, <64 x i8> %{{.*}}
   return _mm512_maskz_unpackb_epi8(__U, __A, 3);
 }
+
+//
+// VPMOVSSDB - Symmetric Signed Saturation DWord to Byte (memory store)
+//
+
+void test_mm_mask_cvtss_epi32_storeu_epi8(void *__P, __mmask8 __M, __m128i __A) {
+  // CHECK-LABEL: @test_mm_mask_cvtss_epi32_storeu_epi8(
+  // CHECK: call void @llvm.x86.avx10.mask.pmovss.db.mem.128(ptr %{{.*}}, <4 x i32> %{{.*}}, i8 %{{.*}})
+  _mm_mask_cvtss_epi32_storeu_epi8(__P, __M, __A);
+}
+
+void test_mm256_mask_cvtss_epi32_storeu_epi8(void *__P, __mmask8 __M, __m256i __A) {
+  // CHECK-LABEL: @test_mm256_mask_cvtss_epi32_storeu_epi8(
+  // CHECK: call void @llvm.x86.avx10.mask.pmovss.db.mem.256(ptr %{{.*}}, <8 x i32> %{{.*}}, i8 %{{.*}})
+  _mm256_mask_cvtss_epi32_storeu_epi8(__P, __M, __A);
+}
+
+void test_mm512_mask_cvtss_epi32_storeu_epi8(void *__P, __mmask16 __M, __m512i __A) {
+  // CHECK-LABEL: @test_mm512_mask_cvtss_epi32_storeu_epi8(
+  // CHECK: call void @llvm.x86.avx10.mask.pmovss.db.mem.512(ptr %{{.*}}, <16 x i32> %{{.*}}, i16 %{{.*}})
+  _mm512_mask_cvtss_epi32_storeu_epi8(__P, __M, __A);
+}
diff --git a/clang/test/CodeGen/attr-target-x86.c b/clang/test/CodeGen/attr-target-x86.c
index 3105517711e5e9..8c052ee4d67475 100644
--- a/clang/test/CodeGen/attr-target-x86.c
+++ b/clang/test/CodeGen/attr-target-x86.c
@@ -17,6 +17,7 @@
 // CHECK: define {{.*}}@f_x86_64_v3({{.*}} [[f_x86_64_v3:#[0-9]+]]
 // CHECK: define {{.*}}@f_x86_64_v4({{.*}} [[f_x86_64_v4:#[0-9]+]]
 // CHECK: define {{.*}}@f_avx10_1{{.*}} [[f_avx10_1:#[0-9]+]]
+// CHECK: define {{.*}}@f_avx10_v2_aux{{.*}} [[f_avx10_v2_aux:#[0-9]+]]
 // CHECK: define {{.*}}@f_prefer_256_bit({{.*}} [[f_prefer_256_bit:#[0-9]+]]
 // CHECK: define {{.*}}@f_no_prefer_256_bit({{.*}} [[f_no_prefer_256_bit:#[0-9]+]]
 
@@ -33,7 +34,7 @@ __attribute__((target("fpmath=387")))
 void f_fpmath_387(void) {}
 
 // CHECK-NOT: tune-cpu
-// CHECK: [[f_no_sse2]] = {{.*}}"target-cpu"="i686" "target-features"="+cmov,+cx8,+x87,-aes,-amx-avx512,-avx,-avx10-v2-aux,-avx10.1,-avx10.2,-avx2,-avx512bf16,-avx512bitalg,-avx512bw,-avx512cd,-avx512dq,-avx512f,-avx512fp16,-avx512ifma,-avx512vbmi,-avx512vbmi2,-avx512vl,-avx512vnni,-avx512vp2intersect,-avx512vpopcntdq,-avxifma,-avxneconvert,-avxvnni,-avxvnniint16,-avxvnniint8,-f16c,-fma,-fma4,-gfni,-kl,-pclmul,-sha,-sha512,-sm3,-sm4,-sse2,-sse3,-sse4.1,-sse4.2,-sse4a,-ssse3,-vaes,-vpclmulqdq,-widekl,-xop" "tune-cpu"="i686"
+// CHECK: [[f_no_sse2]] = {{.*}}"target-cpu"="i686" "target-features"="+cmov,+cx8,+x87,-aes,-amx-avx512,-avx,-avx10.1,-avx10.2,-avx10v2aux,-avx2,-avx512bf16,-avx512bitalg,-avx512bw,-avx512cd,-avx512dq,-avx512f,-avx512fp16,-avx512ifma,-avx512vbmi,-avx512vbmi2,-avx512vl,-avx512vnni,-avx512vp2intersect,-avx512vpopcntdq,-avxifma,-avxneconvert,-avxvnni,-avxvnniint16,-avxvnniint8,-f16c,-fma,-fma4,-gfni,-kl,-pclmul,-sha,-sha512,-sm3,-sm4,-sse2,-sse3,-sse4.1,-sse4.2,-sse4a,-ssse3,-vaes,-vpclmulqdq,-widekl,-xop" "tune-cpu"="i686"
 __attribute__((target("no-sse2")))
 void f_no_sse2(void) {}
 
@@ -41,7 +42,7 @@ void f_no_sse2(void) {}
 __attribute__((target("sse4")))
 void f_sse4(void) {}
 
-// CHECK: [[f_no_sse4]] = {{.*}}"target-cpu"="i686" "target-features"="+cmov,+cx8,+x87,-amx-avx512,-avx,-avx10-v2-aux,-avx10.1,-avx10.2,-avx2,-avx512bf16,-avx512bitalg,-avx512bw,-avx512cd,-avx512dq,-avx512f,-avx512fp16,-avx512ifma,-avx512vbmi,-avx512vbmi2,-avx512vl,-avx512vnni,-avx512vp2intersect,-avx512vpopcntdq,-avxifma,-avxneconvert,-avxvnni,-avxvnniint16,-avxvnniint8,-f16c,-fma,-fma4,-sha512,-sm3,-sm4,-sse4.1,-sse4.2,-vaes,-vpclmulqdq,-xop" "tune-cpu"="i686"
+// CHECK: [[f_no_sse4]] = {{.*}}"target-cpu"="i686" "target-features"="+cmov,+cx8,+x87,-amx-avx512,-avx,-avx10.1,-avx10.2,-avx10v2aux,-avx2,-avx512bf16,-avx512bitalg,-avx512bw,-avx512cd,-avx512dq,-avx512f,-avx512fp16,-avx512ifma,-avx512vbmi,-avx512vbmi2,-avx512vl,-avx512vnni,-avx512vp2intersect,-avx512vpopcntdq,-avxifma,-avxneconvert,-avxvnni,-avxvnniint16,-avxvnniint8,-f16c,-fma,-fma4,-sha512,-sm3,-sm4,-sse4.1,-sse4.2,-vaes,-vpclmulqdq,-xop" "tune-cpu"="i686"
 __attribute__((target("no-sse4")))
 void f_no_sse4(void) {}
 
@@ -101,6 +102,10 @@ void f_x86_64_v4(void) {}
 __attribute__((target("avx10.1")))
 void f_avx10_1(void) {}
 
+// CHECK: [[f_avx10_v2_aux]] = {{.*}}"target-cpu"="i686" "target-features"="{{.*}}+avx10v2aux{{.*}}"
+__attribute__((target("avx10v2aux")))
+void f_avx10_v2_aux(void) {}
+
 // CHECK: [[f_prefer_256_bit]] = {{.*}}"target-features"="{{.*}}+prefer-256-bit
 __attribute__((target("prefer-256-bit")))
 void f_prefer_256_bit(void) {}
diff --git a/clang/test/Sema/builtins-x86.c b/clang/test/Sema/builtins-x86.c
index 7d9cdce3d78948..51417ccdd733a7 100644
--- a/clang/test/Sema/builtins-x86.c
+++ b/clang/test/Sema/builtins-x86.c
@@ -193,3 +193,15 @@ unsigned char test_lwpins64(unsigned long long data2, unsigned long long data1,
 void test_lwpval64(unsigned long long data2, unsigned long long data1, unsigned int flags) {
   __builtin_ia32_lwpval64(data2, data1, flags); // expected-error {{argument to '__builtin_ia32_lwpval64' must be a constant integer}}
 }
+
+__m128i test__builtin_ia32_vunpackb128(__m128i __a) {
+  return __builtin_ia32_vunpackb128(__a, 64); // expected-error {{argument value 64 is outside the valid range [0, 63]}}
+}
+
+__m256i test__builtin_ia32_vunpackb256(__m256i __a) {
+  return __builtin_ia32_vunpackb256(__a, 64); // expected-error {{argument value 64 is outside the valid range [0, 63]}}
+}
+
+__m512i test__builtin_ia32_vunpackb512(__m512i __a) {
+  return __builtin_ia32_vunpackb512(__a, 64); // expected-error {{argument value 64 is outside the valid range [0, 63]}}
+}
diff --git a/llvm/include/llvm/IR/IntrinsicsX86.td b/llvm/include/llvm/IR/IntrinsicsX86.td
index b0bbd6f35359e1..5e3e602e6c2627 100644
--- a/llvm/include/llvm/IR/IntrinsicsX86.td
+++ b/llvm/include/llvm/IR/IntrinsicsX86.td
@@ -4679,6 +4679,41 @@ let TargetPrefix = "x86" in {
       DefaultAttrsIntrinsic<[],
                             [llvm_ptr_ty, llvm_v16i32_ty, llvm_i16_ty],
                             [IntrArgMemOnly]>;
+
+  // VPMOVSSDB - Symmetric signed saturation DWord to Byte
+  def int_x86_avx10_mask_pmovss_db_128 :
+      ClangBuiltin<"__builtin_ia32_pmovssdb128_mask">,
+      DefaultAttrsIntrinsic<[llvm_v16i8_ty],
+                            [llvm_v4i32_ty, llvm_v16i8_ty, llvm_i8_ty],
+                            [IntrNoMem]>;
+  def int_x86_avx10_mask_pmovss_db_256 :
+      ClangBuiltin<"__builtin_ia32_pmovssdb256_mask">,
+      DefaultAttrsIntrinsic<[llvm_v16i8_ty],
+                            [llvm_v8i32_ty, llvm_v16i8_ty, llvm_i8_ty],
+                            [IntrNoMem]>;
+  def int_x86_avx10_mask_pmovss_db_512 :
+      ClangBuiltin<"__builtin_ia32_pmovssdb512_mask">,
+      DefaultAttrsIntrinsic<[llvm_v16i8_ty],
+                            [llvm_v16i32_ty, llvm_v16i8_ty, llvm_i16_ty],
+                            [IntrNoMem]>;
+
+  // VPMOVSSDB memory store intrinsics
+  def int_x86_avx10_mask_pmovss_db_mem_128 :
+      ClangBuiltin<"__builtin_ia32_pmovssdb128mem_mask">,
+      DefaultAttrsIntrinsic<[],
+                            [llvm_ptr_ty, llvm_v4i32_ty, llvm_i8_ty],
+                            [IntrArgMemOnly]>;
+  def int_x86_avx10_mask_pmovss_db_mem_256 :
+      ClangBuiltin<"__builtin_ia32_pmovssdb256mem_mask">,
+      DefaultAttrsIntrinsic<[],
+                            [llvm_ptr_ty, llvm_v8i32_ty, llvm_i8_ty],
+                            [IntrArgMemOnly]>;
+  def int_x86_avx10_mask_pmovss_db_mem_512 :
+      ClangBuiltin<"__builtin_ia32_pmovssdb512mem_mask">,
+      DefaultAttrsIntrinsic<[],
+                            [llvm_ptr_ty, llvm_v16i32_ty, llvm_i16_ty],
+                            [IntrArgMemOnly]>;
+
   def int_x86_avx512_mask_pmov_dw_128 :
       ClangBuiltin<"__builtin_ia32_pmovdw128_mask">,
       DefaultAttrsIntrinsic<[llvm_v8i16_ty],
@@ -7027,140 +7062,104 @@ let TargetPrefix = "x86" in {
 // Group A: PS(f32) -> i8 truncating conversions (quarter-size: output always v16i8)
 
 // VCVTPS2BF8
-def int_x86_avx10_mask_vcvtps2bf8_128 : ClangBuiltin<"__builtin_ia32_vcvtps2bf8_128_mask">,
-        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v4f32_ty, llvm_v16i8_ty, llvm_i8_ty],
-                              [IntrNoMem]>;
-def int_x86_avx10_mask_vcvtps2bf8_256 : ClangBuiltin<"__builtin_ia32_vcvtps2bf8_256_mask">,
-        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v8f32_ty, llvm_v16i8_ty, llvm_i8_ty],
-                              [IntrNoMem]>;
-def int_x86_avx10_mask_vcvtps2bf8_512 : ClangBuiltin<"__builtin_ia32_vcvtps2bf8_512_mask">,
-        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16f32_ty, llvm_v16i8_ty, llvm_i16_ty],
-                              [IntrNoMem]>;
+def int_x86_avx10_vcvtps2bf8_128 : ClangBuiltin<"__builtin_ia32_vcvtps2bf8_128">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v4f32_ty], [IntrNoMem]>;
+def int_x86_avx10_vcvtps2bf8_256 : ClangBuiltin<"__builtin_ia32_vcvtps2bf8_256">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v8f32_ty], [IntrNoMem]>;
+def int_x86_avx10_vcvtps2bf8_512 : ClangBuiltin<"__builtin_ia32_vcvtps2bf8_512">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16f32_ty], [IntrNoMem]>;
 
 // VCVTPS2BF8S
-def int_x86_avx10_mask_vcvtps2bf8s_128 : ClangBuiltin<"__builtin_ia32_vcvtps2bf8s_128_mask">,
-        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v4f32_ty, llvm_v16i8_ty, llvm_i8_ty],
-                              [IntrNoMem]>;
-def int_x86_avx10_mask_vcvtps2bf8s_256 : ClangBuiltin<"__builtin_ia32_vcvtps2bf8s_256_mask">,
-        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v8f32_ty, llvm_v16i8_ty, llvm_i8_ty],
-                              [IntrNoMem]>;
-def int_x86_avx10_mask_vcvtps2bf8s_512 : ClangBuiltin<"__builtin_ia32_vcvtps2bf8s_512_mask">,
-        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16f32_ty, llvm_v16i8_ty, llvm_i16_ty],
-                              [IntrNoMem]>;
+def int_x86_avx10_vcvtps2bf8s_128 : ClangBuiltin<"__builtin_ia32_vcvtps2bf8s_128">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v4f32_ty], [IntrNoMem]>;
+def int_x86_avx10_vcvtps2bf8s_256 : ClangBuiltin<"__builtin_ia32_vcvtps2bf8s_256">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v8f32_ty], [IntrNoMem]>;
+def int_x86_avx10_vcvtps2bf8s_512 : ClangBuiltin<"__builtin_ia32_vcvtps2bf8s_512">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16f32_ty], [IntrNoMem]>;
 
 // VCVTPS2HF8
-def int_x86_avx10_mask_vcvtps2hf8_128 : ClangBuiltin<"__builtin_ia32_vcvtps2hf8_128_mask">,
-        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v4f32_ty, llvm_v16i8_ty, llvm_i8_ty],
-                              [IntrNoMem]>;
-def int_x86_avx10_mask_vcvtps2hf8_256 : ClangBuiltin<"__builtin_ia32_vcvtps2hf8_256_mask">,
-        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v8f32_ty, llvm_v16i8_ty, llvm_i8_ty],
-                              [IntrNoMem]>;
-def int_x86_avx10_mask_vcvtps2hf8_512 : ClangBuiltin<"__builtin_ia32_vcvtps2hf8_512_mask">,
-        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16f32_ty, llvm_v16i8_ty, llvm_i16_ty],
-                              [IntrNoMem]>;
+def int_x86_avx10_vcvtps2hf8_128 : ClangBuiltin<"__builtin_ia32_vcvtps2hf8_128">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v4f32_ty], [IntrNoMem]>;
+def int_x86_avx10_vcvtps2hf8_256 : ClangBuiltin<"__builtin_ia32_vcvtps2hf8_256">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v8f32_ty], [IntrNoMem]>;
+def int_x86_avx10_vcvtps2hf8_512 : ClangBuiltin<"__builtin_ia32_vcvtps2hf8_512">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16f32_ty], [IntrNoMem]>;
 
 // VCVTPS2HF8S
-def int_x86_avx10_mask_vcvtps2hf8s_128 : ClangBuiltin<"__builtin_ia32_vcvtps2hf8s_128_mask">,
-        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v4f32_ty, llvm_v16i8_ty, llvm_i8_ty],
-                              [IntrNoMem]>;
-def int_x86_avx10_mask_vcvtps2hf8s_256 : ClangBuiltin<"__builtin_ia32_vcvtps2hf8s_256_mask">,
-        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v8f32_ty, llvm_v16i8_ty, llvm_i8_ty],
-                              [IntrNoMem]>;
-def int_x86_avx10_mask_vcvtps2hf8s_512 : ClangBuiltin<"__builtin_ia32_vcvtps2hf8s_512_mask">,
-        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16f32_ty, llvm_v16i8_ty, llvm_i16_ty],
-                              [IntrNoMem]>;
+def int_x86_avx10_vcvtps2hf8s_128 : ClangBuiltin<"__builtin_ia32_vcvtps2hf8s_128">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v4f32_ty], [IntrNoMem]>;
+def int_x86_avx10_vcvtps2hf8s_256 : ClangBuiltin<"__builtin_ia32_vcvtps2hf8s_256">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v8f32_ty], [IntrNoMem]>;
+def int_x86_avx10_vcvtps2hf8s_512 : ClangBuiltin<"__builtin_ia32_vcvtps2hf8s_512">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16f32_ty], [IntrNoMem]>;
 
 // VCVTROPS2HF8
-def int_x86_avx10_mask_vcvtrops2hf8_128 : ClangBuiltin<"__builtin_ia32_vcvtrops2hf8_128_mask">,
-        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v4f32_ty, llvm_v16i8_ty, llvm_i8_ty],
-                              [IntrNoMem]>;
-def int_x86_avx10_mask_vcvtrops2hf8_256 : ClangBuiltin<"__builtin_ia32_vcvtrops2hf8_256_mask">,
-        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v8f32_ty, llvm_v16i8_ty, llvm_i8_ty],
-                              [IntrNoMem]>;
-def int_x86_avx10_mask_vcvtrops2hf8_512 : ClangBuiltin<"__builtin_ia32_vcvtrops2hf8_512_mask">,
-        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16f32_ty, llvm_v16i8_ty, llvm_i16_ty],
-                              [IntrNoMem]>;
+def int_x86_avx10_vcvtrops2hf8_128 : ClangBuiltin<"__builtin_ia32_vcvtrops2hf8_128">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v4f32_ty], [IntrNoMem]>;
+def int_x86_avx10_vcvtrops2hf8_256 : ClangBuiltin<"__builtin_ia32_vcvtrops2hf8_256">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v8f32_ty], [IntrNoMem]>;
+def int_x86_avx10_vcvtrops2hf8_512 : ClangBuiltin<"__builtin_ia32_vcvtrops2hf8_512">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16f32_ty], [IntrNoMem]>;
 
 // VCVTROPS2HF8S
-def int_x86_avx10_mask_vcvtrops2hf8s_128 : ClangBuiltin<"__builtin_ia32_vcvtrops2hf8s_128_mask">,
-        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v4f32_ty, llvm_v16i8_ty, llvm_i8_ty],
-                              [IntrNoMem]>;
-def int_x86_avx10_mask_vcvtrops2hf8s_256 : ClangBuiltin<"__builtin_ia32_vcvtrops2hf8s_256_mask">,
-        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v8f32_ty, llvm_v16i8_ty, llvm_i8_ty],
-                              [IntrNoMem]>;
-def int_x86_avx10_mask_vcvtrops2hf8s_512 : ClangBuiltin<"__builtin_ia32_vcvtrops2hf8s_512_mask">,
-        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16f32_ty, llvm_v16i8_ty, llvm_i16_ty],
-                              [IntrNoMem]>;
+def int_x86_avx10_vcvtrops2hf8s_128 : ClangBuiltin<"__builtin_ia32_vcvtrops2hf8s_128">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v4f32_ty], [IntrNoMem]>;
+def int_x86_avx10_vcvtrops2hf8s_256 : ClangBuiltin<"__builtin_ia32_vcvtrops2hf8s_256">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v8f32_ty], [IntrNoMem]>;
+def int_x86_avx10_vcvtrops2hf8s_512 : ClangBuiltin<"__builtin_ia32_vcvtrops2hf8s_512">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16f32_ty], [IntrNoMem]>;
 
-// Group B: Bias PS -> i8 conversions (3-operand: bias + f32 source -> i8 dest)
+// Group B: Bias PS -> i8 conversions (2-operand: bias + f32 source -> i8 dest)
 
 // VCVTBIASPS2BF8
-def int_x86_avx10_mask_vcvtbiasps2bf8_128 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2bf8_128_mask">,
-        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16i8_ty, llvm_v4f32_ty, llvm_v16i8_ty, llvm_i8_ty],
-                              [IntrNoMem]>;
-def int_x86_avx10_mask_vcvtbiasps2bf8_256 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2bf8_256_mask">,
-        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v32i8_ty, llvm_v8f32_ty, llvm_v16i8_ty, llvm_i8_ty],
-                              [IntrNoMem]>;
-def int_x86_avx10_mask_vcvtbiasps2bf8_512 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2bf8_512_mask">,
-        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v64i8_ty, llvm_v16f32_ty, llvm_v16i8_ty, llvm_i16_ty],
-                              [IntrNoMem]>;
+def int_x86_avx10_vcvtbiasps2bf8_128 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2bf8_128">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16i8_ty, llvm_v4f32_ty], [IntrNoMem]>;
+def int_x86_avx10_vcvtbiasps2bf8_256 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2bf8_256">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v32i8_ty, llvm_v8f32_ty], [IntrNoMem]>;
+def int_x86_avx10_vcvtbiasps2bf8_512 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2bf8_512">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v64i8_ty, llvm_v16f32_ty], [IntrNoMem]>;
 
 // VCVTBIASPS2BF8S
-def int_x86_avx10_mask_vcvtbiasps2bf8s_128 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2bf8s_128_mask">,
-        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16i8_ty, llvm_v4f32_ty, llvm_v16i8_ty, llvm_i8_ty],
-                              [IntrNoMem]>;
-def int_x86_avx10_mask_vcvtbiasps2bf8s_256 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2bf8s_256_mask">,
-        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v32i8_ty, llvm_v8f32_ty, llvm_v16i8_ty, llvm_i8_ty],
-                              [IntrNoMem]>;
-def int_x86_avx10_mask_vcvtbiasps2bf8s_512 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2bf8s_512_mask">,
-        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v64i8_ty, llvm_v16f32_ty, llvm_v16i8_ty, llvm_i16_ty],
-                              [IntrNoMem]>;
+def int_x86_avx10_vcvtbiasps2bf8s_128 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2bf8s_128">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16i8_ty, llvm_v4f32_ty], [IntrNoMem]>;
+def int_x86_avx10_vcvtbiasps2bf8s_256 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2bf8s_256">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v32i8_ty, llvm_v8f32_ty], [IntrNoMem]>;
+def int_x86_avx10_vcvtbiasps2bf8s_512 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2bf8s_512">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v64i8_ty, llvm_v16f32_ty], [IntrNoMem]>;
 
 // VCVTBIASPS2HF8
-def int_x86_avx10_mask_vcvtbiasps2hf8_128 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2hf8_128_mask">,
-        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16i8_ty, llvm_v4f32_ty, llvm_v16i8_ty, llvm_i8_ty],
-                              [IntrNoMem]>;
-def int_x86_avx10_mask_vcvtbiasps2hf8_256 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2hf8_256_mask">,
-        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v32i8_ty, llvm_v8f32_ty, llvm_v16i8_ty, llvm_i8_ty],
-                              [IntrNoMem]>;
-def int_x86_avx10_mask_vcvtbiasps2hf8_512 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2hf8_512_mask">,
-        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v64i8_ty, llvm_v16f32_ty, llvm_v16i8_ty, llvm_i16_ty],
-                              [IntrNoMem]>;
+def int_x86_avx10_vcvtbiasps2hf8_128 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2hf8_128">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16i8_ty, llvm_v4f32_ty], [IntrNoMem]>;
+def int_x86_avx10_vcvtbiasps2hf8_256 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2hf8_256">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v32i8_ty, llvm_v8f32_ty], [IntrNoMem]>;
+def int_x86_avx10_vcvtbiasps2hf8_512 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2hf8_512">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v64i8_ty, llvm_v16f32_ty], [IntrNoMem]>;
 
 // VCVTBIASPS2HF8S
-def int_x86_avx10_mask_vcvtbiasps2hf8s_128 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2hf8s_128_mask">,
-        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16i8_ty, llvm_v4f32_ty, llvm_v16i8_ty, llvm_i8_ty],
-                              [IntrNoMem]>;
-def int_x86_avx10_mask_vcvtbiasps2hf8s_256 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2hf8s_256_mask">,
-        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v32i8_ty, llvm_v8f32_ty, llvm_v16i8_ty, llvm_i8_ty],
-                              [IntrNoMem]>;
-def int_x86_avx10_mask_vcvtbiasps2hf8s_512 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2hf8s_512_mask">,
-        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v64i8_ty, llvm_v16f32_ty, llvm_v16i8_ty, llvm_i16_ty],
-                              [IntrNoMem]>;
+def int_x86_avx10_vcvtbiasps2hf8s_128 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2hf8s_128">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16i8_ty, llvm_v4f32_ty], [IntrNoMem]>;
+def int_x86_avx10_vcvtbiasps2hf8s_256 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2hf8s_256">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v32i8_ty, llvm_v8f32_ty], [IntrNoMem]>;
+def int_x86_avx10_vcvtbiasps2hf8s_512 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2hf8s_512">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v64i8_ty, llvm_v16f32_ty], [IntrNoMem]>;
 
 // Group C: 8bit -> PS expanding conversions
 
 // VCVTBF82PS
-def int_x86_avx10_mask_vcvtbf82ps_128 : ClangBuiltin<"__builtin_ia32_vcvtbf8_2ps128_mask">,
-        DefaultAttrsIntrinsic<[llvm_v4f32_ty], [llvm_v16i8_ty, llvm_v4f32_ty, llvm_i8_ty],
-                              [IntrNoMem]>;
-def int_x86_avx10_mask_vcvtbf82ps_256 : ClangBuiltin<"__builtin_ia32_vcvtbf8_2ps256_mask">,
-        DefaultAttrsIntrinsic<[llvm_v8f32_ty], [llvm_v16i8_ty, llvm_v8f32_ty, llvm_i8_ty],
-                              [IntrNoMem]>;
-def int_x86_avx10_mask_vcvtbf82ps_512 : ClangBuiltin<"__builtin_ia32_vcvtbf8_2ps512_mask">,
-        DefaultAttrsIntrinsic<[llvm_v16f32_ty], [llvm_v16i8_ty, llvm_v16f32_ty, llvm_i16_ty],
-                              [IntrNoMem]>;
+def int_x86_avx10_vcvtbf82ps_128 : ClangBuiltin<"__builtin_ia32_vcvtbf8_2ps128">,
+        DefaultAttrsIntrinsic<[llvm_v4f32_ty], [llvm_v16i8_ty], [IntrNoMem]>;
+def int_x86_avx10_vcvtbf82ps_256 : ClangBuiltin<"__builtin_ia32_vcvtbf8_2ps256">,
+        DefaultAttrsIntrinsic<[llvm_v8f32_ty], [llvm_v16i8_ty], [IntrNoMem]>;
+def int_x86_avx10_vcvtbf82ps_512 : ClangBuiltin<"__builtin_ia32_vcvtbf8_2ps512">,
+        DefaultAttrsIntrinsic<[llvm_v16f32_ty], [llvm_v16i8_ty], [IntrNoMem]>;
 
 // VCVTHF82PS
-def int_x86_avx10_mask_vcvthf82ps_128 : ClangBuiltin<"__builtin_ia32_vcvthf8_2ps128_mask">,
-        DefaultAttrsIntrinsic<[llvm_v4f32_ty], [llvm_v16i8_ty, llvm_v4f32_ty, llvm_i8_ty],
-                              [IntrNoMem]>;
-def int_x86_avx10_mask_vcvthf82ps_256 : ClangBuiltin<"__builtin_ia32_vcvthf8_2ps256_mask">,
-        DefaultAttrsIntrinsic<[llvm_v8f32_ty], [llvm_v16i8_ty, llvm_v8f32_ty, llvm_i8_ty],
-                              [IntrNoMem]>;
-def int_x86_avx10_mask_vcvthf82ps_512 : ClangBuiltin<"__builtin_ia32_vcvthf8_2ps512_mask">,
-        DefaultAttrsIntrinsic<[llvm_v16f32_ty], [llvm_v16i8_ty, llvm_v16f32_ty, llvm_i16_ty],
-                              [IntrNoMem]>;
+def int_x86_avx10_vcvthf82ps_128 : ClangBuiltin<"__builtin_ia32_vcvthf8_2ps128">,
+        DefaultAttrsIntrinsic<[llvm_v4f32_ty], [llvm_v16i8_ty], [IntrNoMem]>;
+def int_x86_avx10_vcvthf82ps_256 : ClangBuiltin<"__builtin_ia32_vcvthf8_2ps256">,
+        DefaultAttrsIntrinsic<[llvm_v8f32_ty], [llvm_v16i8_ty], [IntrNoMem]>;
+def int_x86_avx10_vcvthf82ps_512 : ClangBuiltin<"__builtin_ia32_vcvthf8_2ps512">,
+        DefaultAttrsIntrinsic<[llvm_v16f32_ty], [llvm_v16i8_ty], [IntrNoMem]>;
 
 // Group E: Same-size reg-only conversions (no masking)
 
@@ -7206,6 +7205,46 @@ def int_x86_avx10_vcvthf62hf8_256 : ClangBuiltin<"__builtin_ia32_vcvthf62hf8256"
 def int_x86_avx10_vcvthf62hf8_512 : ClangBuiltin<"__builtin_ia32_vcvthf62hf8512">,
         DefaultAttrsIntrinsic<[llvm_v64i8_ty], [llvm_v64i8_ty], [IntrNoMem]>;
 
+// Group D: VCVTBF82BF4S / VCVTHF82BF4S - FP8 to FP4 truncating conversions
+
+// VCVTBF82BF4S: FP8 E5M2 to FP4 E2M1 with saturation (truncating, output half size)
+def int_x86_avx10_vcvtbf82bf4s_128 : ClangBuiltin<"__builtin_ia32_vcvtbf82bf4s128">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16i8_ty], [IntrNoMem]>;
+def int_x86_avx10_vcvtbf82bf4s_256 : ClangBuiltin<"__builtin_ia32_vcvtbf82bf4s256">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v32i8_ty], [IntrNoMem]>;
+def int_x86_avx10_vcvtbf82bf4s_512 : ClangBuiltin<"__builtin_ia32_vcvtbf82bf4s512">,
+        DefaultAttrsIntrinsic<[llvm_v32i8_ty], [llvm_v64i8_ty], [IntrNoMem]>;
+
+// VCVTHF82BF4S: FP8 E4M3 to FP4 E2M1 with saturation (truncating, output half size)
+def int_x86_avx10_vcvthf82bf4s_128 : ClangBuiltin<"__builtin_ia32_vcvthf82bf4s128">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16i8_ty], [IntrNoMem]>;
+def int_x86_avx10_vcvthf82bf4s_256 : ClangBuiltin<"__builtin_ia32_vcvthf82bf4s256">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v32i8_ty], [IntrNoMem]>;
+def int_x86_avx10_vcvthf82bf4s_512 : ClangBuiltin<"__builtin_ia32_vcvthf82bf4s512">,
+        DefaultAttrsIntrinsic<[llvm_v32i8_ty], [llvm_v64i8_ty], [IntrNoMem]>;
+
+// VCVTBF82BF4S memory store: FP8 E5M2 to FP4 E2M1, store to memory
+def int_x86_avx10_vcvtbf82bf4s_128_mem : ClangBuiltin<"__builtin_ia32_vcvtbf82bf4s128mem">,
+        DefaultAttrsIntrinsic<[], [llvm_ptr_ty, llvm_v16i8_ty],
+                              [IntrWriteMem, IntrArgMemOnly, NoCapture<ArgIndex<0>>]>;
+def int_x86_avx10_vcvtbf82bf4s_256_mem : ClangBuiltin<"__builtin_ia32_vcvtbf82bf4s256mem">,
+        DefaultAttrsIntrinsic<[], [llvm_ptr_ty, llvm_v32i8_ty],
+                              [IntrWriteMem, IntrArgMemOnly, NoCapture<ArgIndex<0>>]>;
+def int_x86_avx10_vcvtbf82bf4s_512_mem : ClangBuiltin<"__builtin_ia32_vcvtbf82bf4s512mem">,
+        DefaultAttrsIntrinsic<[], [llvm_ptr_ty, llvm_v64i8_ty],
+                              [IntrWriteMem, IntrArgMemOnly, NoCapture<ArgIndex<0>>]>;
+
+// VCVTHF82BF4S memory store: FP8 E4M3 to FP4 E2M1, store to memory
+def int_x86_avx10_vcvthf82bf4s_128_mem : ClangBuiltin<"__builtin_ia32_vcvthf82bf4s128mem">,
+        DefaultAttrsIntrinsic<[], [llvm_ptr_ty, llvm_v16i8_ty],
+                              [IntrWriteMem, IntrArgMemOnly, NoCapture<ArgIndex<0>>]>;
+def int_x86_avx10_vcvthf82bf4s_256_mem : ClangBuiltin<"__builtin_ia32_vcvthf82bf4s256mem">,
+        DefaultAttrsIntrinsic<[], [llvm_ptr_ty, llvm_v32i8_ty],
+                              [IntrWriteMem, IntrArgMemOnly, NoCapture<ArgIndex<0>>]>;
+def int_x86_avx10_vcvthf82bf4s_512_mem : ClangBuiltin<"__builtin_ia32_vcvthf82bf4s512mem">,
+        DefaultAttrsIntrinsic<[], [llvm_ptr_ty, llvm_v64i8_ty],
+                              [IntrWriteMem, IntrArgMemOnly, NoCapture<ArgIndex<0>>]>;
+
 // Group H: VUNPACKB - Byte unpack with immediate
 
 def int_x86_avx10_vunpackb_128 : ClangBuiltin<"__builtin_ia32_vunpackb128">,
diff --git a/llvm/include/llvm/TargetParser/X86TargetParser.def b/llvm/include/llvm/TargetParser/X86TargetParser.def
index 916f90a4bfd750..d276215ec1cceb 100644
--- a/llvm/include/llvm/TargetParser/X86TargetParser.def
+++ b/llvm/include/llvm/TargetParser/X86TargetParser.def
@@ -274,7 +274,7 @@ X86_FEATURE       (NDD,                "ndd")
 X86_FEATURE       (EGPR,               "egpr")
 X86_FEATURE       (ZU,                 "zu")
 X86_FEATURE       (JMPABS,             "jmpabs")
-X86_FEATURE       (AVX10_V2_AUX,       "avx10-v2-aux")
+X86_FEATURE_COMPAT(AVX10_V2_AUX,       "avx10v2aux",          36, 123)
 
 // These features aren't really CPU features, but the frontend can set them.
 X86_FEATURE       (RETPOLINE_EXTERNAL_THUNK,    "retpoline-external-thunk")
@@ -284,7 +284,7 @@ X86_FEATURE       (LVI_CFI,                     "lvi-cfi")
 X86_FEATURE       (LVI_LOAD_HARDENING,          "lvi-load-hardening")
 
 // Max number of priorities. Priorities form a consecutive range
-#define MAX_PRIORITY 35
+#define MAX_PRIORITY 36
 
 #undef X86_FEATURE_COMPAT
 #undef X86_FEATURE
diff --git a/llvm/lib/Target/X86/X86.td b/llvm/lib/Target/X86/X86.td
index 443f5bf3ff3729..00ebf885c9cb32 100644
--- a/llvm/lib/Target/X86/X86.td
+++ b/llvm/lib/Target/X86/X86.td
@@ -361,7 +361,7 @@ def FeatureAVX10_2 : SubtargetFeature<"avx10.2", "HasAVX10_2", "true",
 def FeatureAVX10_2_512 : SubtargetFeature<"avx10.2-512", "HasAVX10_2_512", "true",
                                           "Support AVX10.2 instruction",
                                           [FeatureAVX10_2]>;
-def FeatureAVX10_V2_AUX : SubtargetFeature<"avx10-v2-aux", "HasAVX10_V2_AUX", "true",
+def FeatureAVX10_V2_AUX : SubtargetFeature<"avx10v2aux", "HasAVX10_V2_AUX", "true",
                                           "Support AVX10 V2 AUX instructions",
                                           [FeatureAVX10_2]>;
 def FeatureEGPR : SubtargetFeature<"egpr", "HasEGPR", "true",
diff --git a/llvm/lib/Target/X86/X86ISelLowering.cpp b/llvm/lib/Target/X86/X86ISelLowering.cpp
index 03cad8457484e0..58913896aa2ee0 100644
--- a/llvm/lib/Target/X86/X86ISelLowering.cpp
+++ b/llvm/lib/Target/X86/X86ISelLowering.cpp
@@ -28881,6 +28881,20 @@ static SDValue LowerINTRINSIC_W_CHAIN(SDValue Op, const X86Subtarget &Subtarget,
       return EmitMaskedTruncSStore(IsSigned, Chain, dl, DataToTruncate, Addr,
                                    VMask, MemVT, MemIntr->getMemOperand(), DAG);
     }
+    case X86ISD::VTRUNCSS: {
+      SDVTList VTs = DAG.getVTList(MVT::Other);
+      if (isAllOnesConstant(Mask)) {
+        SDValue Ops[] = {Chain, DataToTruncate, Addr};
+        return DAG.getMemIntrinsicNode(X86ISD::VTRUNCSTORSS, dl, VTs, Ops,
+                                       MemVT, MemIntr->getMemOperand());
+      }
+
+      MVT MaskVT = MVT::getVectorVT(MVT::i1, MemVT.getVectorNumElements());
+      SDValue VMask = getMaskNode(Mask, MaskVT, Subtarget, DAG, dl);
+      SDValue Ops[] = {Chain, DataToTruncate, Addr, VMask};
+      return DAG.getMemIntrinsicNode(X86ISD::VMTRUNCSTORSS, dl, VTs, Ops, MemVT,
+                                     MemIntr->getMemOperand());
+    }
     default:
       llvm_unreachable("Unsupported truncstore intrinsic");
     }
diff --git a/llvm/lib/Target/X86/X86InstrAVX10_V2_AUX.td b/llvm/lib/Target/X86/X86InstrAVX10_V2_AUX.td
index 29777281440cef..c38094dec6df6e 100644
--- a/llvm/lib/Target/X86/X86InstrAVX10_V2_AUX.td
+++ b/llvm/lib/Target/X86/X86InstrAVX10_V2_AUX.td
@@ -21,236 +21,231 @@
 multiclass avx10_v2aux_cvt_trunc_ps2i8<bits<8> opc, string OpcodeStr,
                                         SDPatternOperator OpNode,
                                         SDPatternOperator MaskOpNode> {
-  let Predicates = [HasAVX10_V2_AUX] in {
-    let ExeDomain = SSEPackedSingle in {
-      let Uses = []<Register>, mayRaiseFPException = 0 in {
-        defm Z : avx512_vcvt_fp<opc, OpcodeStr, v16i8x_info, v16f32_info,
-                                OpNode, OpNode, WriteCvtPH2PSZ>, EVEX_V512;
-        // Z256/Z128: use null_frag because element count mismatch between
-        // dest (v16i8) and source (v8f32/v4f32) prevents avx512_vcvt_fp from
-        // generating correct masked patterns. Explicit Pat patterns below.
-        defm Z256 : avx512_vcvt_fp<opc, OpcodeStr, v16i8x_info, v8f32x_info,
-                                   null_frag, null_frag,
-                                   WriteCvtPH2PSZ, v8f32x_info.BroadcastStr,
-                                   "{y}", v8f32x_info.MemOp,
-                                   v8f32x_info.KRCWM>, EVEX_V256;
-        defm Z128 : avx512_vcvt_fp<opc, OpcodeStr, v16i8x_info, v4f32x_info,
-                                   null_frag, null_frag,
-                                   WriteCvtPH2PSZ, v4f32x_info.BroadcastStr,
-                                   "{x}", f128mem,
-                                   v4f32x_info.KRCWM>, EVEX_V128;
-      }
+  let ExeDomain = SSEPackedSingle in {
+    let Uses = []<Register>, mayRaiseFPException = 0 in {
+      defm Z : avx512_vcvt_fp<opc, OpcodeStr, v16i8x_info, v16f32_info,
+                              OpNode, OpNode, WriteCvtPH2PSZ>, EVEX_V512;
+      // Z256/Z128: use null_frag because element count mismatch between
+      // dest (v16i8) and source (v8f32/v4f32) prevents avx512_vcvt_fp from
+      // generating correct masked patterns. Explicit Pat patterns below.
+      defm Z256 : avx512_vcvt_fp<opc, OpcodeStr, v16i8x_info, v8f32x_info,
+                                 null_frag, null_frag,
+                                 WriteCvtPH2PSZ, v8f32x_info.BroadcastStr,
+                                 "{y}", v8f32x_info.MemOp,
+                                 v8f32x_info.KRCWM>, EVEX_V256;
+      defm Z128 : avx512_vcvt_fp<opc, OpcodeStr, v16i8x_info, v4f32x_info,
+                                 null_frag, null_frag,
+                                 WriteCvtPH2PSZ, v4f32x_info.BroadcastStr,
+                                 "{x}", f128mem,
+                                 v4f32x_info.KRCWM>, EVEX_V128;
     }
-
-    // InstAliases for x/y suffixes
-    def : InstAlias<OpcodeStr#"x\t{$src, $dst|$dst, $src}",
-                    (!cast<Instruction>(NAME # "Z128rr") VR128X:$dst,
-                    VR128X:$src), 0>;
-    def : InstAlias<OpcodeStr#"x\t{$src, $dst|$dst, $src}",
-                    (!cast<Instruction>(NAME # "Z128rm") VR128X:$dst,
-                    f128mem:$src), 0, "intel">;
-    def : InstAlias<OpcodeStr#"y\t{$src, $dst|$dst, $src}",
-                    (!cast<Instruction>(NAME # "Z256rr") VR128X:$dst,
-                    VR256X:$src), 0>;
-    def : InstAlias<OpcodeStr#"y\t{$src, $dst|$dst, $src}",
-                    (!cast<Instruction>(NAME # "Z256rm") VR128X:$dst,
-                    f256mem:$src), 0, "intel">;
-
-    // Explicit patterns for Z256 (8 source elements, VK8WM mask)
-    // Unmasked
-    def : Pat<(v16i8 (OpNode (v8f32 VR256X:$src))),
-              (!cast<Instruction>(NAME # "Z256rr") VR256X:$src)>;
-    // Masked (merge)
-    def : Pat<(MaskOpNode (v8f32 VR256X:$src), (v16i8 VR128X:$src0),
-                           VK8WM:$mask),
-              (!cast<Instruction>(NAME # "Z256rrk") VR128X:$src0, VK8WM:$mask,
-                                  VR256X:$src)>;
-    // Masked (zero)
-    def : Pat<(MaskOpNode (v8f32 VR256X:$src), v16i8x_info.ImmAllZerosV,
-                           VK8WM:$mask),
-              (!cast<Instruction>(NAME # "Z256rrkz") VK8WM:$mask,
-                                  VR256X:$src)>;
-    // Memory
-    def : Pat<(v16i8 (OpNode (loadv8f32 addr:$src))),
-              (!cast<Instruction>(NAME # "Z256rm") addr:$src)>;
-    def : Pat<(MaskOpNode (loadv8f32 addr:$src), (v16i8 VR128X:$src0),
-                           VK8WM:$mask),
-              (!cast<Instruction>(NAME # "Z256rmk") VR128X:$src0, VK8WM:$mask,
-                                  addr:$src)>;
-    def : Pat<(MaskOpNode (loadv8f32 addr:$src), v16i8x_info.ImmAllZerosV,
-                           VK8WM:$mask),
-              (!cast<Instruction>(NAME # "Z256rmkz") VK8WM:$mask, addr:$src)>;
-    // Broadcast
-    def : Pat<(v16i8 (OpNode (v8f32 (X86VBroadcastld32 addr:$src)))),
-              (!cast<Instruction>(NAME # "Z256rmb") addr:$src)>;
-    def : Pat<(MaskOpNode (v8f32 (X86VBroadcastld32 addr:$src)),
-                            (v16i8 VR128X:$src0), VK8WM:$mask),
-              (!cast<Instruction>(NAME # "Z256rmbk") VR128X:$src0, VK8WM:$mask,
-                                  addr:$src)>;
-    def : Pat<(MaskOpNode (v8f32 (X86VBroadcastld32 addr:$src)),
-                            v16i8x_info.ImmAllZerosV, VK8WM:$mask),
-              (!cast<Instruction>(NAME # "Z256rmbkz") VK8WM:$mask, addr:$src)>;
-
-    // Explicit patterns for Z128 (4 source elements, VK4WM mask)
-    // Unmasked
-    def : Pat<(v16i8 (OpNode (v4f32 VR128X:$src))),
-              (!cast<Instruction>(NAME # "Z128rr") VR128X:$src)>;
-    // Masked (merge)
-    def : Pat<(MaskOpNode (v4f32 VR128X:$src), (v16i8 VR128X:$src0),
-                           VK4WM:$mask),
-              (!cast<Instruction>(NAME # "Z128rrk") VR128X:$src0, VK4WM:$mask,
-                                  VR128X:$src)>;
-    // Masked (zero)
-    def : Pat<(MaskOpNode (v4f32 VR128X:$src), v16i8x_info.ImmAllZerosV,
-                           VK4WM:$mask),
-              (!cast<Instruction>(NAME # "Z128rrkz") VK4WM:$mask,
-                                  VR128X:$src)>;
-    // Memory
-    def : Pat<(v16i8 (OpNode (loadv4f32 addr:$src))),
-              (!cast<Instruction>(NAME # "Z128rm") addr:$src)>;
-    def : Pat<(MaskOpNode (loadv4f32 addr:$src), (v16i8 VR128X:$src0),
-                           VK4WM:$mask),
-              (!cast<Instruction>(NAME # "Z128rmk") VR128X:$src0, VK4WM:$mask,
-                                  addr:$src)>;
-    def : Pat<(MaskOpNode (loadv4f32 addr:$src), v16i8x_info.ImmAllZerosV,
-                           VK4WM:$mask),
-              (!cast<Instruction>(NAME # "Z128rmkz") VK4WM:$mask, addr:$src)>;
-    // Broadcast
-    def : Pat<(v16i8 (OpNode (v4f32 (X86VBroadcastld32 addr:$src)))),
-              (!cast<Instruction>(NAME # "Z128rmb") addr:$src)>;
-    def : Pat<(MaskOpNode (v4f32 (X86VBroadcastld32 addr:$src)),
-                            (v16i8 VR128X:$src0), VK4WM:$mask),
-              (!cast<Instruction>(NAME # "Z128rmbk") VR128X:$src0, VK4WM:$mask,
-                                  addr:$src)>;
-    def : Pat<(MaskOpNode (v4f32 (X86VBroadcastld32 addr:$src)),
-                            v16i8x_info.ImmAllZerosV, VK4WM:$mask),
-              (!cast<Instruction>(NAME # "Z128rmbkz") VK4WM:$mask, addr:$src)>;
   }
+
+  // InstAliases for x/y suffixes
+  def : InstAlias<OpcodeStr#"x\t{$src, $dst|$dst, $src}",
+                  (!cast<Instruction>(NAME # "Z128rr") VR128X:$dst,
+                  VR128X:$src), 0>;
+  def : InstAlias<OpcodeStr#"x\t{$src, $dst|$dst, $src}",
+                  (!cast<Instruction>(NAME # "Z128rm") VR128X:$dst,
+                  f128mem:$src), 0, "intel">;
+  def : InstAlias<OpcodeStr#"y\t{$src, $dst|$dst, $src}",
+                  (!cast<Instruction>(NAME # "Z256rr") VR128X:$dst,
+                  VR256X:$src), 0>;
+  def : InstAlias<OpcodeStr#"y\t{$src, $dst|$dst, $src}",
+                  (!cast<Instruction>(NAME # "Z256rm") VR128X:$dst,
+                  f256mem:$src), 0, "intel">;
+
+  // Explicit patterns for Z256 (8 source elements, VK8WM mask)
+  // Unmasked
+  def : Pat<(v16i8 (OpNode (v8f32 VR256X:$src))),
+            (!cast<Instruction>(NAME # "Z256rr") VR256X:$src)>;
+  // Masked (merge)
+  def : Pat<(MaskOpNode (v8f32 VR256X:$src), (v16i8 VR128X:$src0),
+                         VK8WM:$mask),
+            (!cast<Instruction>(NAME # "Z256rrk") VR128X:$src0, VK8WM:$mask,
+                                VR256X:$src)>;
+  // Masked (zero)
+  def : Pat<(MaskOpNode (v8f32 VR256X:$src), v16i8x_info.ImmAllZerosV,
+                         VK8WM:$mask),
+            (!cast<Instruction>(NAME # "Z256rrkz") VK8WM:$mask,
+                                VR256X:$src)>;
+  // Memory
+  def : Pat<(v16i8 (OpNode (loadv8f32 addr:$src))),
+            (!cast<Instruction>(NAME # "Z256rm") addr:$src)>;
+  def : Pat<(MaskOpNode (loadv8f32 addr:$src), (v16i8 VR128X:$src0),
+                         VK8WM:$mask),
+            (!cast<Instruction>(NAME # "Z256rmk") VR128X:$src0, VK8WM:$mask,
+                                addr:$src)>;
+  def : Pat<(MaskOpNode (loadv8f32 addr:$src), v16i8x_info.ImmAllZerosV,
+                         VK8WM:$mask),
+            (!cast<Instruction>(NAME # "Z256rmkz") VK8WM:$mask, addr:$src)>;
+  // Broadcast
+  def : Pat<(v16i8 (OpNode (v8f32 (X86VBroadcastld32 addr:$src)))),
+            (!cast<Instruction>(NAME # "Z256rmb") addr:$src)>;
+  def : Pat<(MaskOpNode (v8f32 (X86VBroadcastld32 addr:$src)),
+                          (v16i8 VR128X:$src0), VK8WM:$mask),
+            (!cast<Instruction>(NAME # "Z256rmbk") VR128X:$src0, VK8WM:$mask,
+                                addr:$src)>;
+  def : Pat<(MaskOpNode (v8f32 (X86VBroadcastld32 addr:$src)),
+                          v16i8x_info.ImmAllZerosV, VK8WM:$mask),
+            (!cast<Instruction>(NAME # "Z256rmbkz") VK8WM:$mask, addr:$src)>;
+
+  // Explicit patterns for Z128 (4 source elements, VK4WM mask)
+  // Unmasked
+  def : Pat<(v16i8 (OpNode (v4f32 VR128X:$src))),
+            (!cast<Instruction>(NAME # "Z128rr") VR128X:$src)>;
+  // Masked (merge)
+  def : Pat<(MaskOpNode (v4f32 VR128X:$src), (v16i8 VR128X:$src0),
+                         VK4WM:$mask),
+            (!cast<Instruction>(NAME # "Z128rrk") VR128X:$src0, VK4WM:$mask,
+                                VR128X:$src)>;
+  // Masked (zero)
+  def : Pat<(MaskOpNode (v4f32 VR128X:$src), v16i8x_info.ImmAllZerosV,
+                         VK4WM:$mask),
+            (!cast<Instruction>(NAME # "Z128rrkz") VK4WM:$mask,
+                                VR128X:$src)>;
+  // Memory
+  def : Pat<(v16i8 (OpNode (loadv4f32 addr:$src))),
+            (!cast<Instruction>(NAME # "Z128rm") addr:$src)>;
+  def : Pat<(MaskOpNode (loadv4f32 addr:$src), (v16i8 VR128X:$src0),
+                         VK4WM:$mask),
+            (!cast<Instruction>(NAME # "Z128rmk") VR128X:$src0, VK4WM:$mask,
+                                addr:$src)>;
+  def : Pat<(MaskOpNode (loadv4f32 addr:$src), v16i8x_info.ImmAllZerosV,
+                         VK4WM:$mask),
+            (!cast<Instruction>(NAME # "Z128rmkz") VK4WM:$mask, addr:$src)>;
+  // Broadcast
+  def : Pat<(v16i8 (OpNode (v4f32 (X86VBroadcastld32 addr:$src)))),
+            (!cast<Instruction>(NAME # "Z128rmb") addr:$src)>;
+  def : Pat<(MaskOpNode (v4f32 (X86VBroadcastld32 addr:$src)),
+                          (v16i8 VR128X:$src0), VK4WM:$mask),
+            (!cast<Instruction>(NAME # "Z128rmbk") VR128X:$src0, VK4WM:$mask,
+                                addr:$src)>;
+  def : Pat<(MaskOpNode (v4f32 (X86VBroadcastld32 addr:$src)),
+                          v16i8x_info.ImmAllZerosV, VK4WM:$mask),
+            (!cast<Instruction>(NAME # "Z128rmbkz") VK4WM:$mask, addr:$src)>;
 }
 
 // Group B multiclass: 3-operand bias PS(f32) -> i8 conversion (quarter-size output)
 // bias + f32 source -> i8 dest
 // Reuses avx10_convert_3op_packed from X86InstrAVX10.td for each VL variant.
-multiclass avx10_v2aux_convert_3op_ps<bits<8> OpCode, string OpcodeStr,
-                                       SDPatternOperator OpNode,
-                                       SDPatternOperator MaskOpNode> {
-  let Predicates = [HasAVX10_V2_AUX] in {
-    // Z (512-bit): bias=v64i8(zmm), src=v16f32(zmm), dst=v16i8(xmm)
-    // Element counts match (16), so vselect_mask works directly.
-    defm Z : avx10_convert_3op_packed<OpCode, OpcodeStr, v16i8x_info,
-               v64i8_info, v16f32_info, OpNode, OpNode, WriteCvtPH2PSZ>,
-               EVEX_V512, EVEX_CD8<32, CD8VF>;
-    // Z256/Z128: use null_frag because element count mismatch between
-    // dest (v16i8) and source (v8f32/v4f32) prevents vselect_mask from
-    // generating correct masked patterns. Explicit Pat patterns below.
-    defm Z256 : avx10_convert_3op_packed<OpCode, OpcodeStr, v16i8x_info,
-                  v32i8x_info, v8f32x_info,
-                  null_frag, null_frag, WriteCvtPH2PSZ>,
-                  EVEX_V256, EVEX_CD8<32, CD8VF>;
-    defm Z128 : avx10_convert_3op_packed<OpCode, OpcodeStr, v16i8x_info,
-                  v16i8x_info, v4f32x_info,
-                  null_frag, null_frag, WriteCvtPH2PSZ>,
-                  EVEX_V128, EVEX_CD8<32, CD8VF>;
-
-    // Explicit patterns for Z256 (8 source elements, VK8WM mask)
-    def : Pat<(v16i8 (OpNode (v32i8 VR256X:$src1), (v8f32 VR256X:$src2))),
-              (!cast<Instruction>(NAME # "Z256rr") VR256X:$src1, VR256X:$src2)>;
-    def : Pat<(MaskOpNode (v32i8 VR256X:$src1), (v8f32 VR256X:$src2),
-                           (v16i8 VR128X:$src0), VK8WM:$mask),
-              (!cast<Instruction>(NAME # "Z256rrk") VR128X:$src0, VK8WM:$mask,
-                                  VR256X:$src1, VR256X:$src2)>;
-    def : Pat<(MaskOpNode (v32i8 VR256X:$src1), (v8f32 VR256X:$src2),
-                           v16i8x_info.ImmAllZerosV, VK8WM:$mask),
-              (!cast<Instruction>(NAME # "Z256rrkz") VK8WM:$mask,
-                                  VR256X:$src1, VR256X:$src2)>;
-    // Memory
-    def : Pat<(v16i8 (OpNode (v32i8 VR256X:$src1), (loadv8f32 addr:$src2))),
-              (!cast<Instruction>(NAME # "Z256rm") VR256X:$src1, addr:$src2)>;
-    def : Pat<(MaskOpNode (v32i8 VR256X:$src1), (loadv8f32 addr:$src2),
-                           (v16i8 VR128X:$src0), VK8WM:$mask),
-              (!cast<Instruction>(NAME # "Z256rmk") VR128X:$src0, VK8WM:$mask,
-                                  VR256X:$src1, addr:$src2)>;
-    def : Pat<(MaskOpNode (v32i8 VR256X:$src1), (loadv8f32 addr:$src2),
-                           v16i8x_info.ImmAllZerosV, VK8WM:$mask),
-              (!cast<Instruction>(NAME # "Z256rmkz") VK8WM:$mask,
-                                  VR256X:$src1, addr:$src2)>;
-    // Broadcast
-    def : Pat<(v16i8 (OpNode (v32i8 VR256X:$src1),
-                              (v8f32 (X86VBroadcastld32 addr:$src2)))),
-              (!cast<Instruction>(NAME # "Z256rmb") VR256X:$src1, addr:$src2)>;
-    def : Pat<(MaskOpNode (v32i8 VR256X:$src1),
-                            (v8f32 (X86VBroadcastld32 addr:$src2)),
-                            (v16i8 VR128X:$src0), VK8WM:$mask),
-              (!cast<Instruction>(NAME # "Z256rmbk") VR128X:$src0, VK8WM:$mask,
-                                  VR256X:$src1, addr:$src2)>;
-    def : Pat<(MaskOpNode (v32i8 VR256X:$src1),
-                            (v8f32 (X86VBroadcastld32 addr:$src2)),
-                            v16i8x_info.ImmAllZerosV, VK8WM:$mask),
-              (!cast<Instruction>(NAME # "Z256rmbkz") VK8WM:$mask,
-                                  VR256X:$src1, addr:$src2)>;
-
-    // Explicit patterns for Z128 (4 source elements, VK4WM mask)
-    def : Pat<(v16i8 (OpNode (v16i8 VR128X:$src1), (v4f32 VR128X:$src2))),
-              (!cast<Instruction>(NAME # "Z128rr") VR128X:$src1, VR128X:$src2)>;
-    def : Pat<(MaskOpNode (v16i8 VR128X:$src1), (v4f32 VR128X:$src2),
-                           (v16i8 VR128X:$src0), VK4WM:$mask),
-              (!cast<Instruction>(NAME # "Z128rrk") VR128X:$src0, VK4WM:$mask,
-                                  VR128X:$src1, VR128X:$src2)>;
-    def : Pat<(MaskOpNode (v16i8 VR128X:$src1), (v4f32 VR128X:$src2),
-                           v16i8x_info.ImmAllZerosV, VK4WM:$mask),
-              (!cast<Instruction>(NAME # "Z128rrkz") VK4WM:$mask,
-                                  VR128X:$src1, VR128X:$src2)>;
-    // Memory
-    def : Pat<(v16i8 (OpNode (v16i8 VR128X:$src1), (loadv4f32 addr:$src2))),
-              (!cast<Instruction>(NAME # "Z128rm") VR128X:$src1, addr:$src2)>;
-    def : Pat<(MaskOpNode (v16i8 VR128X:$src1), (loadv4f32 addr:$src2),
-                           (v16i8 VR128X:$src0), VK4WM:$mask),
-              (!cast<Instruction>(NAME # "Z128rmk") VR128X:$src0, VK4WM:$mask,
-                                  VR128X:$src1, addr:$src2)>;
-    def : Pat<(MaskOpNode (v16i8 VR128X:$src1), (loadv4f32 addr:$src2),
-                           v16i8x_info.ImmAllZerosV, VK4WM:$mask),
-              (!cast<Instruction>(NAME # "Z128rmkz") VK4WM:$mask,
-                                  VR128X:$src1, addr:$src2)>;
-    // Broadcast
-    def : Pat<(v16i8 (OpNode (v16i8 VR128X:$src1),
-                              (v4f32 (X86VBroadcastld32 addr:$src2)))),
-              (!cast<Instruction>(NAME # "Z128rmb") VR128X:$src1, addr:$src2)>;
-    def : Pat<(MaskOpNode (v16i8 VR128X:$src1),
-                            (v4f32 (X86VBroadcastld32 addr:$src2)),
-                            (v16i8 VR128X:$src0), VK4WM:$mask),
-              (!cast<Instruction>(NAME # "Z128rmbk") VR128X:$src0, VK4WM:$mask,
-                                  VR128X:$src1, addr:$src2)>;
-    def : Pat<(MaskOpNode (v16i8 VR128X:$src1),
-                            (v4f32 (X86VBroadcastld32 addr:$src2)),
-                            v16i8x_info.ImmAllZerosV, VK4WM:$mask),
-              (!cast<Instruction>(NAME # "Z128rmbkz") VK4WM:$mask,
-                                  VR128X:$src1, addr:$src2)>;
-  }
+multiclass avx10_v2aux_cvt_3op_ps<bits<8> opc, string OpcodeStr,
+                                  SDPatternOperator OpNode,
+                                  SDPatternOperator MaskOpNode> {
+  // Z (512-bit): bias=v64i8(zmm), src=v16f32(zmm), dst=v16i8(xmm)
+  // Element counts match (16), so vselect_mask works directly.
+  defm Z : avx10_convert_3op_packed<opc, OpcodeStr, v16i8x_info,
+             v64i8_info, v16f32_info, OpNode, OpNode, WriteCvtPH2PSZ>,
+             EVEX_V512, EVEX_CD8<32, CD8VF>;
+  // Z256/Z128: use null_frag because element count mismatch between
+  // dest (v16i8) and source (v8f32/v4f32) prevents vselect_mask from
+  // generating correct masked patterns. Explicit Pat patterns below.
+  defm Z256 : avx10_convert_3op_packed<opc, OpcodeStr, v16i8x_info,
+                v32i8x_info, v8f32x_info,
+                null_frag, null_frag, WriteCvtPH2PSZ>,
+                EVEX_V256, EVEX_CD8<32, CD8VF>;
+  defm Z128 : avx10_convert_3op_packed<opc, OpcodeStr, v16i8x_info,
+                v16i8x_info, v4f32x_info,
+                null_frag, null_frag, WriteCvtPH2PSZ>,
+                EVEX_V128, EVEX_CD8<32, CD8VF>;
+
+  // Explicit patterns for Z256 (8 source elements, VK8WM mask)
+  def : Pat<(v16i8 (OpNode (v32i8 VR256X:$src1), (v8f32 VR256X:$src2))),
+            (!cast<Instruction>(NAME # "Z256rr") VR256X:$src1, VR256X:$src2)>;
+  def : Pat<(MaskOpNode (v32i8 VR256X:$src1), (v8f32 VR256X:$src2),
+                         (v16i8 VR128X:$src0), VK8WM:$mask),
+            (!cast<Instruction>(NAME # "Z256rrk") VR128X:$src0, VK8WM:$mask,
+                                VR256X:$src1, VR256X:$src2)>;
+  def : Pat<(MaskOpNode (v32i8 VR256X:$src1), (v8f32 VR256X:$src2),
+                         v16i8x_info.ImmAllZerosV, VK8WM:$mask),
+            (!cast<Instruction>(NAME # "Z256rrkz") VK8WM:$mask,
+                                VR256X:$src1, VR256X:$src2)>;
+  // Memory
+  def : Pat<(v16i8 (OpNode (v32i8 VR256X:$src1), (loadv8f32 addr:$src2))),
+            (!cast<Instruction>(NAME # "Z256rm") VR256X:$src1, addr:$src2)>;
+  def : Pat<(MaskOpNode (v32i8 VR256X:$src1), (loadv8f32 addr:$src2),
+                         (v16i8 VR128X:$src0), VK8WM:$mask),
+            (!cast<Instruction>(NAME # "Z256rmk") VR128X:$src0, VK8WM:$mask,
+                                VR256X:$src1, addr:$src2)>;
+  def : Pat<(MaskOpNode (v32i8 VR256X:$src1), (loadv8f32 addr:$src2),
+                         v16i8x_info.ImmAllZerosV, VK8WM:$mask),
+            (!cast<Instruction>(NAME # "Z256rmkz") VK8WM:$mask,
+                                VR256X:$src1, addr:$src2)>;
+  // Broadcast
+  def : Pat<(v16i8 (OpNode (v32i8 VR256X:$src1),
+                            (v8f32 (X86VBroadcastld32 addr:$src2)))),
+            (!cast<Instruction>(NAME # "Z256rmb") VR256X:$src1, addr:$src2)>;
+  def : Pat<(MaskOpNode (v32i8 VR256X:$src1),
+                          (v8f32 (X86VBroadcastld32 addr:$src2)),
+                          (v16i8 VR128X:$src0), VK8WM:$mask),
+            (!cast<Instruction>(NAME # "Z256rmbk") VR128X:$src0, VK8WM:$mask,
+                                VR256X:$src1, addr:$src2)>;
+  def : Pat<(MaskOpNode (v32i8 VR256X:$src1),
+                          (v8f32 (X86VBroadcastld32 addr:$src2)),
+                          v16i8x_info.ImmAllZerosV, VK8WM:$mask),
+            (!cast<Instruction>(NAME # "Z256rmbkz") VK8WM:$mask,
+                                VR256X:$src1, addr:$src2)>;
+
+  // Explicit patterns for Z128 (4 source elements, VK4WM mask)
+  def : Pat<(v16i8 (OpNode (v16i8 VR128X:$src1), (v4f32 VR128X:$src2))),
+            (!cast<Instruction>(NAME # "Z128rr") VR128X:$src1, VR128X:$src2)>;
+  def : Pat<(MaskOpNode (v16i8 VR128X:$src1), (v4f32 VR128X:$src2),
+                         (v16i8 VR128X:$src0), VK4WM:$mask),
+            (!cast<Instruction>(NAME # "Z128rrk") VR128X:$src0, VK4WM:$mask,
+                                VR128X:$src1, VR128X:$src2)>;
+  def : Pat<(MaskOpNode (v16i8 VR128X:$src1), (v4f32 VR128X:$src2),
+                         v16i8x_info.ImmAllZerosV, VK4WM:$mask),
+            (!cast<Instruction>(NAME # "Z128rrkz") VK4WM:$mask,
+                                VR128X:$src1, VR128X:$src2)>;
+  // Memory
+  def : Pat<(v16i8 (OpNode (v16i8 VR128X:$src1), (loadv4f32 addr:$src2))),
+            (!cast<Instruction>(NAME # "Z128rm") VR128X:$src1, addr:$src2)>;
+  def : Pat<(MaskOpNode (v16i8 VR128X:$src1), (loadv4f32 addr:$src2),
+                         (v16i8 VR128X:$src0), VK4WM:$mask),
+            (!cast<Instruction>(NAME # "Z128rmk") VR128X:$src0, VK4WM:$mask,
+                                VR128X:$src1, addr:$src2)>;
+  def : Pat<(MaskOpNode (v16i8 VR128X:$src1), (loadv4f32 addr:$src2),
+                         v16i8x_info.ImmAllZerosV, VK4WM:$mask),
+            (!cast<Instruction>(NAME # "Z128rmkz") VK4WM:$mask,
+                                VR128X:$src1, addr:$src2)>;
+  // Broadcast
+  def : Pat<(v16i8 (OpNode (v16i8 VR128X:$src1),
+                            (v4f32 (X86VBroadcastld32 addr:$src2)))),
+            (!cast<Instruction>(NAME # "Z128rmb") VR128X:$src1, addr:$src2)>;
+  def : Pat<(MaskOpNode (v16i8 VR128X:$src1),
+                          (v4f32 (X86VBroadcastld32 addr:$src2)),
+                          (v16i8 VR128X:$src0), VK4WM:$mask),
+            (!cast<Instruction>(NAME # "Z128rmbk") VR128X:$src0, VK4WM:$mask,
+                                VR128X:$src1, addr:$src2)>;
+  def : Pat<(MaskOpNode (v16i8 VR128X:$src1),
+                          (v4f32 (X86VBroadcastld32 addr:$src2)),
+                          v16i8x_info.ImmAllZerosV, VK4WM:$mask),
+            (!cast<Instruction>(NAME # "Z128rmbkz") VK4WM:$mask,
+                                VR128X:$src1, addr:$src2)>;
 }
 
 // Group C multiclass: i8 -> f32 expanding conversion (4x expansion, no broadcast)
-multiclass avx10_v2aux_convert_2op_i8_to_f32<string OpcodeStr, bits<8> opc,
-                                              SDNode OpNode> {
-  let Predicates = [HasAVX10_V2_AUX] in {
-    defm Z : avx10_convert_2op_nomb_packed<opc, OpcodeStr, v16f32_info,
-                                           v16i8x_info, OpNode, f128mem,
-                                           WriteCvtPH2PSZ>, EVEX_V512;
-    defm Z128 : avx10_convert_2op_nomb_packed<opc, OpcodeStr, v4f32x_info,
-                                              v16i8x_info, OpNode, f32mem,
-                                              WriteCvtPH2PSZ>, EVEX_V128;
-    defm Z256 : avx10_convert_2op_nomb_packed<opc, OpcodeStr, v8f32x_info,
-                                              v16i8x_info, OpNode, f64mem,
-                                              WriteCvtPH2PSZ>, EVEX_V256;
-  }
+multiclass avx10_v2aux_cvt_2op_i8_to_f32<bits<8> opc, string OpcodeStr,
+                                         SDNode OpNode> {
+  defm Z : avx10_convert_2op_nomb_packed<opc, OpcodeStr, v16f32_info,
+                                         v16i8x_info, OpNode, f128mem,
+                                         WriteCvtPH2PSZ>, EVEX_V512;
+  defm Z128 : avx10_convert_2op_nomb_packed<opc, OpcodeStr, v4f32x_info,
+                                            v16i8x_info, OpNode, f32mem,
+                                            WriteCvtPH2PSZ>, EVEX_V128;
+  defm Z256 : avx10_convert_2op_nomb_packed<opc, OpcodeStr, v8f32x_info,
+                                            v16i8x_info, OpNode, f64mem,
+                                            WriteCvtPH2PSZ>, EVEX_V256;
 }
 
-// Group D multiclass: Store-like truncation (reg/mem dest, no masking)
+// Group D multiclass: Truncating conversion (reg/mem dest, no masking)
 // Uses MRMDestReg/MRMDestMem since destination can be memory operand.
 // Source is in reg field, destination is in r/m field.
-multiclass avx10_v2aux_trunc_store<bits<8> opc, string OpcodeStr,
-                                    X86VectorVTInfo SrcInfo,
-                                    X86VectorVTInfo DestInfo,
-                                    X86MemOperand x86memop> {
+multiclass avx10_v2aux_cvt_trunc<bits<8> opc, string OpcodeStr,
+                                  X86VectorVTInfo SrcInfo,
+                                  X86VectorVTInfo DestInfo,
+                                  X86MemOperand x86memop,
+                                  SDPatternOperator OpNode> {
   let hasSideEffects = 0 in {
     def rr : I<opc, MRMDestReg, (outs DestInfo.RC:$dst),
                (ins SrcInfo.RC:$src),
@@ -262,71 +257,66 @@ multiclass avx10_v2aux_trunc_store<bits<8> opc, string OpcodeStr,
                OpcodeStr # "\t{$src, $dst|$dst, $src}", []>,
                EVEX, Sched<[WriteCvtPH2PSZ.Folded]>;
   }
+
+  // Intrinsic pattern
+  def : Pat<(DestInfo.VT (OpNode (SrcInfo.VT SrcInfo.RC:$src))),
+            (!cast<Instruction>(NAME # "rr") SrcInfo.RC:$src)>;
 }
 
 // Group F helper: expanding conversion with masking, no broadcast (reg+mem)
-multiclass avx10_v2aux_convert_expand_masked<bits<8> opc, string OpcodeStr,
-                                              X86VectorVTInfo _dest,
-                                              X86VectorVTInfo _src,
-                                              X86MemOperand x86memop> {
+multiclass avx10_v2aux_cvt_expand_masked<bits<8> opc, string OpcodeStr,
+                                         X86VectorVTInfo _dest,
+                                         X86VectorVTInfo _src,
+                                         X86MemOperand x86memop,
+                                         SDPatternOperator OpNode> {
   let ExeDomain = _dest.ExeDomain in {
-    defm rr : AVX512_maskable_custom<opc, MRMSrcReg,
+    defm rr : AVX512_maskable<opc, MRMSrcReg, _dest,
                 (outs _dest.RC:$dst),
                 (ins _src.RC:$src),
-                (ins _dest.RC:$src0, _dest.KRCWM:$mask, _src.RC:$src),
-                (ins _dest.KRCWM:$mask, _src.RC:$src),
-                OpcodeStr, "$src", "$src", [], [], [],
-                "$src0 = $dst">,
+                OpcodeStr, "$src", "$src",
+                (_dest.VT (OpNode (_src.VT _src.RC:$src)))>,
                EVEX, Sched<[WriteCvtPH2PSZ]>;
     let mayLoad = 1 in
-    defm rm : AVX512_maskable_custom<opc, MRMSrcMem,
+    defm rm : AVX512_maskable<opc, MRMSrcMem, _dest,
                 (outs _dest.RC:$dst),
                 (ins x86memop:$src),
-                (ins _dest.RC:$src0, _dest.KRCWM:$mask, x86memop:$src),
-                (ins _dest.KRCWM:$mask, x86memop:$src),
-                OpcodeStr, "$src", "$src", [], [], [],
-                "$src0 = $dst">,
+                OpcodeStr, "$src", "$src",
+                (_dest.VT (OpNode (_src.VT (load addr:$src))))>,
                EVEX, Sched<[WriteCvtPH2PSZ.Folded]>;
   }
 }
 
-// Group F helper: same-size conversion with masking (reg-only)
-multiclass avx10_v2aux_convert_samesize_masked<bits<8> opc, string OpcodeStr,
-                                                X86VectorVTInfo _> {
+// Group F helper: widening conversion with masking (reg-only, 6-bit to 8-bit)
+multiclass avx10_v2aux_cvt_widen_masked<bits<8> opc, string OpcodeStr,
+                                        X86VectorVTInfo _,
+                                        SDPatternOperator OpNode> {
   let ExeDomain = _.ExeDomain in {
-    defm rr : AVX512_maskable_custom<opc, MRMSrcReg,
+    defm rr : AVX512_maskable<opc, MRMSrcReg, _,
                 (outs _.RC:$dst),
                 (ins _.RC:$src),
-                (ins _.RC:$src0, _.KRCWM:$mask, _.RC:$src),
-                (ins _.KRCWM:$mask, _.RC:$src),
-                OpcodeStr, "$src", "$src", [], [], [],
-                "$src0 = $dst">,
+                OpcodeStr, "$src", "$src",
+                (_.VT (OpNode (_.VT _.RC:$src)))>,
                EVEX, Sched<[WriteCvtPH2PSZ]>;
   }
 }
 
 // Group H multiclass: Byte unpack with immediate
 multiclass avx10_v2aux_unpackb<bits<8> opc, string OpcodeStr,
-                                X86VectorVTInfo _> {
+                                X86VectorVTInfo _,
+                                SDPatternOperator OpNode> {
   let ImmT = Imm8 in {
-    defm ri : AVX512_maskable_custom<opc, MRMSrcReg,
+    defm rri : AVX512_maskable<opc, MRMSrcReg, _,
                     (outs _.RC:$dst),
                     (ins _.RC:$src1, u8imm:$src2),
-                    (ins _.RC:$src0, _.KRCWM:$mask, _.RC:$src1, u8imm:$src2),
-                    (ins _.KRCWM:$mask, _.RC:$src1, u8imm:$src2),
                     OpcodeStr, "$src2, $src1", "$src1, $src2",
-                    [], [], [],
-                    "$src0 = $dst">,
+                    (_.VT (OpNode (_.VT _.RC:$src1), (i8 timm:$src2)))>,
                     Sched<[WriteShuffle]>;
     let mayLoad = 1 in
-    defm mi : AVX512_maskable_custom<opc, MRMSrcMem,
+    defm rmi : AVX512_maskable<opc, MRMSrcMem, _,
                     (outs _.RC:$dst),
                     (ins _.MemOp:$src1, u8imm:$src2),
-                    (ins _.RC:$src0, _.KRCWM:$mask, _.MemOp:$src1, u8imm:$src2),
-                    (ins _.KRCWM:$mask, _.MemOp:$src1, u8imm:$src2),
                     OpcodeStr, "$src2, $src1", "$src1, $src2",
-                    [], [], [],
-                    "$src0 = $dst">,
+                    (_.VT (OpNode (_.VT (load addr:$src1)), (i8 timm:$src2)))>,
                     Sched<[WriteShuffle.Folded]>;
   }
 }
@@ -339,228 +329,213 @@ multiclass avx10_v2aux_unpackb<bits<8> opc, string OpcodeStr,
 // Group A: PS->8bit truncating conversions
 //-------------------------------------------------
 
-defm VCVTPS2BF8 : avx10_v2aux_cvt_trunc_ps2i8<0x39, "vcvtps2bf8",
+let Predicates = [HasAVX10_V2_AUX] in {
+  defm VCVTPS2BF8 : avx10_v2aux_cvt_trunc_ps2i8<0x39, "vcvtps2bf8",
                                                 X86vcvtps2bf8, X86vmcvtps2bf8>,
-                  T_MAP5, XS, EVEX_CD8<32, CD8VF>;
-defm VCVTPS2BF8S : avx10_v2aux_cvt_trunc_ps2i8<0x3B, "vcvtps2bf8s",
+                    T_MAP5, XS, EVEX_CD8<32, CD8VF>;
+  defm VCVTPS2BF8S : avx10_v2aux_cvt_trunc_ps2i8<0x3B, "vcvtps2bf8s",
                                                  X86vcvtps2bf8s, X86vmcvtps2bf8s>,
-                   T_MAP5, XS, EVEX_CD8<32, CD8VF>;
-defm VCVTPS2HF8 : avx10_v2aux_cvt_trunc_ps2i8<0x38, "vcvtps2hf8",
+                     T_MAP5, XS, EVEX_CD8<32, CD8VF>;
+  defm VCVTPS2HF8 : avx10_v2aux_cvt_trunc_ps2i8<0x38, "vcvtps2hf8",
                                                 X86vcvtps2hf8, X86vmcvtps2hf8>,
-                  T_MAP5, XS, EVEX_CD8<32, CD8VF>;
-defm VCVTPS2HF8S : avx10_v2aux_cvt_trunc_ps2i8<0x3A, "vcvtps2hf8s",
-                                                  X86vcvtps2hf8s, X86vmcvtps2hf8s>,
-                   T_MAP5, XS, EVEX_CD8<32, CD8VF>;
-defm VCVTROPS2HF8 : avx10_v2aux_cvt_trunc_ps2i8<0x38, "vcvtrops2hf8",
-                                                   X86vcvtrops2hf8, X86vmcvtrops2hf8>,
-                    T_MAP5, PD, EVEX_CD8<32, CD8VF>;
-defm VCVTROPS2HF8S : avx10_v2aux_cvt_trunc_ps2i8<0x3A, "vcvtrops2hf8s",
-                                                    X86vcvtrops2hf8s, X86vmcvtrops2hf8s>,
-                     T_MAP5, PD, EVEX_CD8<32, CD8VF>;
+                    T_MAP5, XS, EVEX_CD8<32, CD8VF>;
+  defm VCVTPS2HF8S : avx10_v2aux_cvt_trunc_ps2i8<0x3A, "vcvtps2hf8s",
+                                                 X86vcvtps2hf8s, X86vmcvtps2hf8s>,
+                     T_MAP5, XS, EVEX_CD8<32, CD8VF>;
+  defm VCVTROPS2HF8 : avx10_v2aux_cvt_trunc_ps2i8<0x38, "vcvtrops2hf8",
+                                                  X86vcvtrops2hf8, X86vmcvtrops2hf8>,
+                      T_MAP5, PD, EVEX_CD8<32, CD8VF>;
+  defm VCVTROPS2HF8S : avx10_v2aux_cvt_trunc_ps2i8<0x3A, "vcvtrops2hf8s",
+                                                   X86vcvtrops2hf8s, X86vmcvtrops2hf8s>,
+                       T_MAP5, PD, EVEX_CD8<32, CD8VF>;
+}
 
 //-------------------------------------------------
 // Group B: Bias PS->8bit conversions (3-operand)
 //-------------------------------------------------
 
-defm VCVTBIASPS2BF8 : avx10_v2aux_convert_3op_ps<0x39, "vcvtbiasps2bf8",
-                                                   X86vcvtbiasps2bf8,
-                                                   X86vmcvtbiasps2bf8>,
-                      T_MAP5, PS;
-defm VCVTBIASPS2BF8S : avx10_v2aux_convert_3op_ps<0x3B, "vcvtbiasps2bf8s",
-                                                    X86vcvtbiasps2bf8s,
-                                                    X86vmcvtbiasps2bf8s>,
-                       T_MAP5, PS;
-defm VCVTBIASPS2HF8 : avx10_v2aux_convert_3op_ps<0x38, "vcvtbiasps2hf8",
-                                                   X86vcvtbiasps2hf8,
-                                                   X86vmcvtbiasps2hf8>,
-                      T_MAP5, PS;
-defm VCVTBIASPS2HF8S : avx10_v2aux_convert_3op_ps<0x3A, "vcvtbiasps2hf8s",
-                                                    X86vcvtbiasps2hf8s,
-                                                    X86vmcvtbiasps2hf8s>,
-                       T_MAP5, PS;
+let Predicates = [HasAVX10_V2_AUX] in {
+  defm VCVTBIASPS2BF8 : avx10_v2aux_cvt_3op_ps<0x39, "vcvtbiasps2bf8",
+                                               X86vcvtbiasps2bf8,
+                                               X86vmcvtbiasps2bf8>,
+                        T_MAP5, PS;
+  defm VCVTBIASPS2BF8S : avx10_v2aux_cvt_3op_ps<0x3B, "vcvtbiasps2bf8s",
+                                                X86vcvtbiasps2bf8s,
+                                                X86vmcvtbiasps2bf8s>,
+                         T_MAP5, PS;
+  defm VCVTBIASPS2HF8 : avx10_v2aux_cvt_3op_ps<0x38, "vcvtbiasps2hf8",
+                                               X86vcvtbiasps2hf8,
+                                               X86vmcvtbiasps2hf8>,
+                        T_MAP5, PS;
+  defm VCVTBIASPS2HF8S : avx10_v2aux_cvt_3op_ps<0x3A, "vcvtbiasps2hf8s",
+                                                X86vcvtbiasps2hf8s,
+                                                X86vmcvtbiasps2hf8s>,
+                         T_MAP5, PS;
+}
 
 //-------------------------------------------------
 // Group C: 8bit->PS expanding conversions
 //-------------------------------------------------
 
-defm VCVTBF82PS : avx10_v2aux_convert_2op_i8_to_f32<"vcvtbf82ps", 0x36,
-                                                      X86vcvtbf82ps>,
-                  PS, T_MAP5, EVEX, EVEX_CD8<32, CD8VQ>, REX_W;
-defm VCVTHF82PS : avx10_v2aux_convert_2op_i8_to_f32<"vcvthf82ps", 0x36,
-                                                      X86vcvthf82ps>,
-                  PS, T_MAP5, EVEX, EVEX_CD8<32, CD8VQ>;
+let Predicates = [HasAVX10_V2_AUX] in {
+  defm VCVTBF82PS : avx10_v2aux_cvt_2op_i8_to_f32<0x36, "vcvtbf82ps",
+                                                  X86vcvtbf82ps>,
+                    PS, T_MAP5, EVEX, EVEX_CD8<32, CD8VQ>, REX_W;
+  defm VCVTHF82PS : avx10_v2aux_cvt_2op_i8_to_f32<0x36, "vcvthf82ps",
+                                                  X86vcvthf82ps>,
+                    PS, T_MAP5, EVEX, EVEX_CD8<32, CD8VQ>;
+}
 
-// X //-------------------------------------------------
-// Group D: BF8/HF8->BF4S store-like truncations
+//-------------------------------------------------
+// Group D: BF8/HF8->BF4S truncating conversions
 //-------------------------------------------------
 
 let Predicates = [HasAVX10_V2_AUX] in {
-  defm VCVTBF82BF4SZ    : avx10_v2aux_trunc_store<0x3D, "vcvtbf82bf4s",
-                             v64i8_info, v32i8x_info, i256mem>,
-                           T_MAP5, XS, REX_W, EVEX_V512, EVEX_CD8<8, CD8VH>;
-  defm VCVTBF82BF4SZ256 : avx10_v2aux_trunc_store<0x3D, "vcvtbf82bf4s",
-                             v32i8x_info, v16i8x_info, i128mem>,
-                           T_MAP5, XS, REX_W, EVEX_V256, EVEX_CD8<8, CD8VH>;
-  defm VCVTBF82BF4SZ128 : avx10_v2aux_trunc_store<0x3D, "vcvtbf82bf4s",
-                             v16i8x_info, v16i8x_info, i64mem>,
-                           T_MAP5, XS, REX_W, EVEX_V128, EVEX_CD8<8, CD8VH>;
-
-  defm VCVTHF82BF4SZ    : avx10_v2aux_trunc_store<0x3D, "vcvthf82bf4s",
-                             v64i8_info, v32i8x_info, i256mem>,
-                           T_MAP5, XS, EVEX_V512, EVEX_CD8<8, CD8VH>;
-  defm VCVTHF82BF4SZ256 : avx10_v2aux_trunc_store<0x3D, "vcvthf82bf4s",
-                             v32i8x_info, v16i8x_info, i128mem>,
-                           T_MAP5, XS, EVEX_V256, EVEX_CD8<8, CD8VH>;
-  defm VCVTHF82BF4SZ128 : avx10_v2aux_trunc_store<0x3D, "vcvthf82bf4s",
-                             v16i8x_info, v16i8x_info, i64mem>,
-                           T_MAP5, XS, EVEX_V128, EVEX_CD8<8, CD8VH>;
+  defm VCVTBF82BF4SZ    : avx10_v2aux_cvt_trunc<0x3D, "vcvtbf82bf4s",
+                                                v64i8_info, v32i8x_info, i256mem,
+                                                int_x86_avx10_vcvtbf82bf4s_512>,
+                          T_MAP5, XS, REX_W, EVEX_V512, EVEX_CD8<8, CD8VH>;
+  defm VCVTBF82BF4SZ256 : avx10_v2aux_cvt_trunc<0x3D, "vcvtbf82bf4s",
+                                                v32i8x_info, v16i8x_info, i128mem,
+                                                int_x86_avx10_vcvtbf82bf4s_256>,
+                          T_MAP5, XS, REX_W, EVEX_V256, EVEX_CD8<8, CD8VH>;
+  defm VCVTBF82BF4SZ128 : avx10_v2aux_cvt_trunc<0x3D, "vcvtbf82bf4s",
+                                                v16i8x_info, v16i8x_info, i64mem,
+                                                int_x86_avx10_vcvtbf82bf4s_128>,
+                          T_MAP5, XS, REX_W, EVEX_V128, EVEX_CD8<8, CD8VH>;
+
+  defm VCVTHF82BF4SZ    : avx10_v2aux_cvt_trunc<0x3D, "vcvthf82bf4s",
+                                                v64i8_info, v32i8x_info, i256mem,
+                                                int_x86_avx10_vcvthf82bf4s_512>,
+                          T_MAP5, XS, EVEX_V512, EVEX_CD8<8, CD8VH>;
+  defm VCVTHF82BF4SZ256 : avx10_v2aux_cvt_trunc<0x3D, "vcvthf82bf4s",
+                                                v32i8x_info, v16i8x_info, i128mem,
+                                                int_x86_avx10_vcvthf82bf4s_256>,
+                          T_MAP5, XS, EVEX_V256, EVEX_CD8<8, CD8VH>;
+  defm VCVTHF82BF4SZ128 : avx10_v2aux_cvt_trunc<0x3D, "vcvthf82bf4s",
+                                                v16i8x_info, v16i8x_info, i64mem,
+                                                int_x86_avx10_vcvthf82bf4s_128>,
+                          T_MAP5, XS, EVEX_V128, EVEX_CD8<8, CD8VH>;
+
+  // Memory store patterns for VCVTBF82BF4S
+  def : Pat<(int_x86_avx10_vcvtbf82bf4s_512_mem addr:$dst, VR512:$src),
+            (VCVTBF82BF4SZmr addr:$dst, VR512:$src)>;
+  def : Pat<(int_x86_avx10_vcvtbf82bf4s_256_mem addr:$dst, VR256X:$src),
+            (VCVTBF82BF4SZ256mr addr:$dst, VR256X:$src)>;
+  def : Pat<(int_x86_avx10_vcvtbf82bf4s_128_mem addr:$dst, VR128X:$src),
+            (VCVTBF82BF4SZ128mr addr:$dst, VR128X:$src)>;
+
+  // Memory store patterns for VCVTHF82BF4S
+  def : Pat<(int_x86_avx10_vcvthf82bf4s_512_mem addr:$dst, VR512:$src),
+            (VCVTHF82BF4SZmr addr:$dst, VR512:$src)>;
+  def : Pat<(int_x86_avx10_vcvthf82bf4s_256_mem addr:$dst, VR256X:$src),
+            (VCVTHF82BF4SZ256mr addr:$dst, VR256X:$src)>;
+  def : Pat<(int_x86_avx10_vcvthf82bf4s_128_mem addr:$dst, VR128X:$src),
+            (VCVTHF82BF4SZ128mr addr:$dst, VR128X:$src)>;
 }
 
 //-------------------------------------------------
-// Group E: Same-size reg-only conversions (no masking)
+// Group E: Narrowing reg-only conversions (8-bit to 6-bit, no masking)
 //-------------------------------------------------
 
-let Predicates = [HasAVX10_V2_AUX], hasSideEffects = 0 in {
-  def VCVTBF82BF6SZrr : I<0x3E, MRMSrcReg, (outs VR512:$dst),
-                           (ins VR512:$src),
-                           "vcvtbf82bf6s\t{$src, $dst|$dst, $src}", []>,
-                         EVEX, T_MAP5, XS, REX_W, EVEX_V512,
-                         Sched<[WriteCvtPH2PSZ]>;
-  def VCVTBF82BF6SZ256rr : I<0x3E, MRMSrcReg, (outs VR256X:$dst),
-                              (ins VR256X:$src),
-                              "vcvtbf82bf6s\t{$src, $dst|$dst, $src}", []>,
-                            EVEX, T_MAP5, XS, REX_W, EVEX_V256,
-                            Sched<[WriteCvtPH2PSZ]>;
-  def VCVTBF82BF6SZ128rr : I<0x3E, MRMSrcReg, (outs VR128X:$dst),
-                              (ins VR128X:$src),
-                              "vcvtbf82bf6s\t{$src, $dst|$dst, $src}", []>,
-                            EVEX, T_MAP5, XS, REX_W, EVEX_V128,
-                            Sched<[WriteCvtPH2PSZ]>;
-
-  def VCVTHF82HF6SZrr : I<0x3C, MRMSrcReg, (outs VR512:$dst),
-                           (ins VR512:$src),
-                           "vcvthf82hf6s\t{$src, $dst|$dst, $src}", []>,
-                         EVEX, T_MAP5, XS, EVEX_V512,
-                         Sched<[WriteCvtPH2PSZ]>;
-  def VCVTHF82HF6SZ256rr : I<0x3C, MRMSrcReg, (outs VR256X:$dst),
-                              (ins VR256X:$src),
-                              "vcvthf82hf6s\t{$src, $dst|$dst, $src}", []>,
-                            EVEX, T_MAP5, XS, EVEX_V256,
-                            Sched<[WriteCvtPH2PSZ]>;
-  def VCVTHF82HF6SZ128rr : I<0x3C, MRMSrcReg, (outs VR128X:$dst),
-                              (ins VR128X:$src),
-                              "vcvthf82hf6s\t{$src, $dst|$dst, $src}", []>,
-                            EVEX, T_MAP5, XS, EVEX_V128,
-                            Sched<[WriteCvtPH2PSZ]>;
+// Group E multiclass: Narrowing conversion reg-only (no masking, 8-bit to 6-bit)
+multiclass avx10_v2aux_cvt_narrow<bits<8> opc, string OpcodeStr,
+                                  X86VectorVTInfo Info,
+                                  SDPatternOperator OpNode> {
+  let hasSideEffects = 0 in {
+    def rr : I<opc, MRMSrcReg, (outs Info.RC:$dst),
+               (ins Info.RC:$src),
+               OpcodeStr # "\t{$src, $dst|$dst, $src}", []>,
+             EVEX, Sched<[WriteCvtPH2PSZ]>;
+  }
+
+  // Intrinsic pattern
+  def : Pat<(Info.VT (OpNode (Info.VT Info.RC:$src))),
+            (!cast<Instruction>(NAME # "rr") Info.RC:$src)>;
 }
 
-// Intrinsic patterns for Group E
 let Predicates = [HasAVX10_V2_AUX] in {
-  def : Pat<(v16i8 (int_x86_avx10_vcvtbf82bf6s_128 (v16i8 VR128X:$src))),
-            (VCVTBF82BF6SZ128rr VR128X:$src)>;
-  def : Pat<(v32i8 (int_x86_avx10_vcvtbf82bf6s_256 (v32i8 VR256X:$src))),
-            (VCVTBF82BF6SZ256rr VR256X:$src)>;
-  def : Pat<(v64i8 (int_x86_avx10_vcvtbf82bf6s_512 (v64i8 VR512:$src))),
-            (VCVTBF82BF6SZrr VR512:$src)>;
-
-  def : Pat<(v16i8 (int_x86_avx10_vcvthf82hf6s_128 (v16i8 VR128X:$src))),
-            (VCVTHF82HF6SZ128rr VR128X:$src)>;
-  def : Pat<(v32i8 (int_x86_avx10_vcvthf82hf6s_256 (v32i8 VR256X:$src))),
-            (VCVTHF82HF6SZ256rr VR256X:$src)>;
-  def : Pat<(v64i8 (int_x86_avx10_vcvthf82hf6s_512 (v64i8 VR512:$src))),
-            (VCVTHF82HF6SZrr VR512:$src)>;
+  defm VCVTBF82BF6SZ    : avx10_v2aux_cvt_narrow<0x3E, "vcvtbf82bf6s",
+                                                 v64i8_info, int_x86_avx10_vcvtbf82bf6s_512>,
+                          T_MAP5, XS, REX_W, EVEX_V512;
+  defm VCVTBF82BF6SZ256 : avx10_v2aux_cvt_narrow<0x3E, "vcvtbf82bf6s",
+                                                 v32i8x_info, int_x86_avx10_vcvtbf82bf6s_256>,
+                          T_MAP5, XS, REX_W, EVEX_V256;
+  defm VCVTBF82BF6SZ128 : avx10_v2aux_cvt_narrow<0x3E, "vcvtbf82bf6s",
+                                                 v16i8x_info, int_x86_avx10_vcvtbf82bf6s_128>,
+                          T_MAP5, XS, REX_W, EVEX_V128;
+
+  defm VCVTHF82HF6SZ    : avx10_v2aux_cvt_narrow<0x3C, "vcvthf82hf6s",
+                                                 v64i8_info, int_x86_avx10_vcvthf82hf6s_512>,
+                          T_MAP5, XS, EVEX_V512;
+  defm VCVTHF82HF6SZ256 : avx10_v2aux_cvt_narrow<0x3C, "vcvthf82hf6s",
+                                                 v32i8x_info, int_x86_avx10_vcvthf82hf6s_256>,
+                          T_MAP5, XS, EVEX_V256;
+  defm VCVTHF82HF6SZ128 : avx10_v2aux_cvt_narrow<0x3C, "vcvthf82hf6s",
+                                                 v16i8x_info, int_x86_avx10_vcvthf82hf6s_128>,
+                          T_MAP5, XS, EVEX_V128;
 }
 
 //-------------------------------------------------
-// Group F: Expanding/same-size conversions with masking
+// Group F: Expanding/widening conversions with masking
 //-------------------------------------------------
 
 // VCVTBF42HF8: expanding (2x), with masking, no broadcast
 // Z128: xmm{k}{z}, xmm/m64  Z256: ymm{k}{z}, xmm/m128  Z: zmm{k}{z}, ymm/m256
 let Predicates = [HasAVX10_V2_AUX] in {
-  defm VCVTBF42HF8Z : avx10_v2aux_convert_expand_masked<0x37, "vcvtbf42hf8",
-                         v64i8_info, v32i8x_info, f256mem>,
-                       T_MAP5, PS, EVEX_V512, EVEX_CD8<8, CD8VH>;
-  defm VCVTBF42HF8Z128 : avx10_v2aux_convert_expand_masked<0x37, "vcvtbf42hf8",
-                            v16i8x_info, v16i8x_info, f64mem>,
-                          T_MAP5, PS, EVEX_V128, EVEX_CD8<8, CD8VH>;
-  defm VCVTBF42HF8Z256 : avx10_v2aux_convert_expand_masked<0x37, "vcvtbf42hf8",
-                            v32i8x_info, v16i8x_info, f128mem>,
-                          T_MAP5, PS, EVEX_V256, EVEX_CD8<8, CD8VH>;
+  defm VCVTBF42HF8Z : avx10_v2aux_cvt_expand_masked<0x37, "vcvtbf42hf8",
+                                                    v64i8_info, v32i8x_info, f256mem,
+                                                    int_x86_avx10_vcvtbf42hf8_512>,
+                      T_MAP5, PS, EVEX_V512, EVEX_CD8<8, CD8VH>;
+  defm VCVTBF42HF8Z128 : avx10_v2aux_cvt_expand_masked<0x37, "vcvtbf42hf8",
+                                                       v16i8x_info, v16i8x_info, f64mem,
+                                                       int_x86_avx10_vcvtbf42hf8_128>,
+                         T_MAP5, PS, EVEX_V128, EVEX_CD8<8, CD8VH>;
+  defm VCVTBF42HF8Z256 : avx10_v2aux_cvt_expand_masked<0x37, "vcvtbf42hf8",
+                                                       v32i8x_info, v16i8x_info, f128mem,
+                                                       int_x86_avx10_vcvtbf42hf8_256>,
+                         T_MAP5, PS, EVEX_V256, EVEX_CD8<8, CD8VH>;
 }
 let Predicates = [HasAVX10_V2_AUX] in {
-  let ExeDomain = SSEPackedInt in {
-    // VCVTBF62HF8: same-size with masking, reg-only (66.MAP5.W1)
-    defm VCVTBF62HF8Z    : avx10_v2aux_convert_samesize_masked<0x37,
-                              "vcvtbf62hf8", v64i8_info>,
-                            T_MAP5, PD, REX_W, EVEX_V512;
-    defm VCVTBF62HF8Z256 : avx10_v2aux_convert_samesize_masked<0x37,
-                              "vcvtbf62hf8", v32i8x_info>,
-                            T_MAP5, PD, REX_W, EVEX_V256;
-    defm VCVTBF62HF8Z128 : avx10_v2aux_convert_samesize_masked<0x37,
-                              "vcvtbf62hf8", v16i8x_info>,
-                            T_MAP5, PD, REX_W, EVEX_V128;
-
-    // VCVTHF62HF8: same-size with masking, reg-only (66.MAP5.W0)
-    defm VCVTHF62HF8Z    : avx10_v2aux_convert_samesize_masked<0x37,
-                              "vcvthf62hf8", v64i8_info>,
-                            T_MAP5, PD, EVEX_V512;
-    defm VCVTHF62HF8Z256 : avx10_v2aux_convert_samesize_masked<0x37,
-                              "vcvthf62hf8", v32i8x_info>,
-                            T_MAP5, PD, EVEX_V256;
-    defm VCVTHF62HF8Z128 : avx10_v2aux_convert_samesize_masked<0x37,
-                              "vcvthf62hf8", v16i8x_info>,
-                            T_MAP5, PD, EVEX_V128;
-  }
-}
-
-// Intrinsic patterns for Group F (unmasked only; masking via selectb in C header)
-
-let Predicates = [HasAVX10_V2_AUX] in {
-  // VCVTBF42HF8
-  def : Pat<(v16i8 (int_x86_avx10_vcvtbf42hf8_128 (v16i8 VR128X:$src))),
-            (VCVTBF42HF8Z128rr VR128X:$src)>;
-  def : Pat<(v32i8 (int_x86_avx10_vcvtbf42hf8_256 (v16i8 VR128X:$src))),
-            (VCVTBF42HF8Z256rr VR128X:$src)>;
-  def : Pat<(v64i8 (int_x86_avx10_vcvtbf42hf8_512 (v32i8 VR256X:$src))),
-            (VCVTBF42HF8Zrr VR256X:$src)>;
-
-  // VCVTBF62HF8
-  def : Pat<(v16i8 (int_x86_avx10_vcvtbf62hf8_128 (v16i8 VR128X:$src))),
-            (VCVTBF62HF8Z128rr VR128X:$src)>;
-  def : Pat<(v32i8 (int_x86_avx10_vcvtbf62hf8_256 (v32i8 VR256X:$src))),
-            (VCVTBF62HF8Z256rr VR256X:$src)>;
-  def : Pat<(v64i8 (int_x86_avx10_vcvtbf62hf8_512 (v64i8 VR512:$src))),
-            (VCVTBF62HF8Zrr VR512:$src)>;
-
-  // VCVTHF62HF8
-  def : Pat<(v16i8 (int_x86_avx10_vcvthf62hf8_128 (v16i8 VR128X:$src))),
-            (VCVTHF62HF8Z128rr VR128X:$src)>;
-  def : Pat<(v32i8 (int_x86_avx10_vcvthf62hf8_256 (v32i8 VR256X:$src))),
-            (VCVTHF62HF8Z256rr VR256X:$src)>;
-  def : Pat<(v64i8 (int_x86_avx10_vcvthf62hf8_512 (v64i8 VR512:$src))),
-            (VCVTHF62HF8Zrr VR512:$src)>;
+  // VCVTBF62HF8: widening (6-bit to 8-bit) with masking, reg-only (66.MAP5.W1)
+  defm VCVTBF62HF8Z    : avx10_v2aux_cvt_widen_masked<0x37, "vcvtbf62hf8",
+                                                      v64i8_info,
+                                                      int_x86_avx10_vcvtbf62hf8_512>,
+                         T_MAP5, PD, REX_W, EVEX_V512;
+  defm VCVTBF62HF8Z256 : avx10_v2aux_cvt_widen_masked<0x37, "vcvtbf62hf8",
+                                                      v32i8x_info,
+                                                      int_x86_avx10_vcvtbf62hf8_256>,
+                         T_MAP5, PD, REX_W, EVEX_V256;
+  defm VCVTBF62HF8Z128 : avx10_v2aux_cvt_widen_masked<0x37, "vcvtbf62hf8",
+                                                      v16i8x_info,
+                                                      int_x86_avx10_vcvtbf62hf8_128>,
+                         T_MAP5, PD, REX_W, EVEX_V128;
+
+  // VCVTHF62HF8: widening (6-bit to 8-bit) with masking, reg-only (66.MAP5.W0)
+  defm VCVTHF62HF8Z    : avx10_v2aux_cvt_widen_masked<0x37, "vcvthf62hf8",
+                                                      v64i8_info,
+                                                      int_x86_avx10_vcvthf62hf8_512>,
+                         T_MAP5, PD, EVEX_V512;
+  defm VCVTHF62HF8Z256 : avx10_v2aux_cvt_widen_masked<0x37, "vcvthf62hf8",
+                                                      v32i8x_info,
+                                                      int_x86_avx10_vcvthf62hf8_256>,
+                         T_MAP5, PD, EVEX_V256;
+  defm VCVTHF62HF8Z128 : avx10_v2aux_cvt_widen_masked<0x37, "vcvthf62hf8",
+                                                      v16i8x_info,
+                                                      int_x86_avx10_vcvthf62hf8_128>,
+                         T_MAP5, PD, EVEX_V128;
 }
 
 //-------------------------------------------------
-// Group G: VPMOVSSDB - Integer DWord->Byte signed saturation
+// Group G: VPMOVSSDB - Integer DWord->Byte symmetric signed saturation
 // F3.0F38.W0 0x41
 //-------------------------------------------------
 
-let Predicates = [HasAVX10_V2_AUX] in
-  defm VPMOVSSDZ : avx512_trunc_common<0x41, "vpmovssdb", X86vtruncs,
-                     X86vmtruncs, SchedWriteVecTruncate.ZMM,
-                     v16i32_info, v16i8x_info, i128mem>,
-                   EVEX_V512, EVEX_CD8<8, CD8VQ>;
-let Predicates = [HasVLX, HasAVX10_V2_AUX] in {
-  defm VPMOVSSDZ256 : avx512_trunc_common<0x41, "vpmovssdb", X86vtruncs,
-                        X86vmtruncs, SchedWriteVecTruncate.YMM,
-                        v8i32x_info, v16i8x_info, i64mem>,
-                      EVEX_V256, EVEX_CD8<8, CD8VQ>;
-  defm VPMOVSSDZ128 : avx512_trunc_common<0x41, "vpmovssdb", X86vtruncs,
-                        X86vmtruncs, SchedWriteVecTruncate.XMM,
-                        v4i32x_info, v16i8x_info, i32mem>,
-                      EVEX_V128, EVEX_CD8<8, CD8VQ>;
+let Predicates = [HasAVX10_V2_AUX] in {
+  defm VPMOVSSDB : avx512_trunc_db<0x41, "vpmovssdb", X86vtruncss, select_truncss,
+                                   SchedWriteVecTruncate, truncstore_ss_vi8,
+                                   masked_truncstore_ss_vi8, X86vtruncss,
+                                   X86vmtruncss>;
 }
 
 //-------------------------------------------------
@@ -568,21 +543,13 @@ let Predicates = [HasVLX, HasAVX10_V2_AUX] in {
 //-------------------------------------------------
 
 let Predicates = [HasAVX10_V2_AUX] in {
-  defm VUNPACKBZ    : avx10_v2aux_unpackb<0x3D, "vunpackb", v64i8_info>,
+  defm VUNPACKBZ    : avx10_v2aux_unpackb<0x3D, "vunpackb", v64i8_info,
+                                          int_x86_avx10_vunpackb_512>,
                       EVEX_V512, TA, PS, EVEX, EVEX_CD8<8, CD8VF>;
-  defm VUNPACKBZ256 : avx10_v2aux_unpackb<0x3D, "vunpackb", v32i8x_info>,
+  defm VUNPACKBZ256 : avx10_v2aux_unpackb<0x3D, "vunpackb", v32i8x_info,
+                                          int_x86_avx10_vunpackb_256>,
                       EVEX_V256, TA, PS, EVEX, EVEX_CD8<8, CD8VF>;
-  defm VUNPACKBZ128 : avx10_v2aux_unpackb<0x3D, "vunpackb", v16i8x_info>,
+  defm VUNPACKBZ128 : avx10_v2aux_unpackb<0x3D, "vunpackb", v16i8x_info,
+                                          int_x86_avx10_vunpackb_128>,
                       EVEX_V128, TA, PS, EVEX, EVEX_CD8<8, CD8VF>;
-
-  // Intrinsic patterns for Group H (unmasked only; masking via selectb in C header)
-  def : Pat<(v16i8 (int_x86_avx10_vunpackb_128
-              (v16i8 VR128X:$src1), timm:$src2)),
-            (VUNPACKBZ128ri VR128X:$src1, timm:$src2)>;
-  def : Pat<(v32i8 (int_x86_avx10_vunpackb_256
-              (v32i8 VR256X:$src1), timm:$src2)),
-            (VUNPACKBZ256ri VR256X:$src1, timm:$src2)>;
-  def : Pat<(v64i8 (int_x86_avx10_vunpackb_512
-              (v64i8 VR512:$src1), timm:$src2)),
-            (VUNPACKBZri VR512:$src1, timm:$src2)>;
 }
diff --git a/llvm/lib/Target/X86/X86InstrFragmentsSIMD.td b/llvm/lib/Target/X86/X86InstrFragmentsSIMD.td
index c109d682dd9156..2ee055cd28834f 100644
--- a/llvm/lib/Target/X86/X86InstrFragmentsSIMD.td
+++ b/llvm/lib/Target/X86/X86InstrFragmentsSIMD.td
@@ -214,6 +214,9 @@ def X86vtrunc    : SDNode<"X86ISD::VTRUNC",   SDTVtrunc>;
 def X86vtruncs   : SDNode<"X86ISD::VTRUNCS",  SDTVtrunc>;
 def X86vtruncus  : SDNode<"X86ISD::VTRUNCUS", SDTVtrunc>;
 
+// Vector integer truncate with symmetric signed saturation.
+def X86vtruncss  : SDNode<"X86ISD::VTRUNCSS", SDTVtrunc>;
+
 // Masked version of the above. Used when less than a 128-bit result is
 // produced since the mask only applies to the lower elements and can't
 // be represented by a select.
@@ -221,6 +224,7 @@ def X86vtruncus  : SDNode<"X86ISD::VTRUNCUS", SDTVtrunc>;
 def X86vmtrunc   : SDNode<"X86ISD::VMTRUNC",   SDTVmtrunc>;
 def X86vmtruncs  : SDNode<"X86ISD::VMTRUNCS",  SDTVmtrunc>;
 def X86vmtruncus : SDNode<"X86ISD::VMTRUNCUS", SDTVmtrunc>;
+def X86vmtruncss : SDNode<"X86ISD::VMTRUNCSS", SDTVmtrunc>;
 
 // Vector FP extend.
 def X86vfpext  : SDNode<"X86ISD::VFPEXT",
@@ -1670,6 +1674,14 @@ def X86MTruncSStore : SDNode<"X86ISD::VMTRUNCSTORES",  SDTX86MaskedStore,
 def X86MTruncUSStore : SDNode<"X86ISD::VMTRUNCSTOREUS",  SDTX86MaskedStore,
                        [SDNPHasChain, SDNPMayStore, SDNPMemOperand]>;
 
+// Vector truncating store with symmetric signed saturation
+def X86TruncSSStore : SDNode<"X86ISD::VTRUNCSTORSS",  SDTStore,
+                       [SDNPHasChain, SDNPMayStore, SDNPMemOperand]>;
+
+// Vector truncating masked store with symmetric signed saturation
+def X86MTruncSSStore : SDNode<"X86ISD::VMTRUNCSTORSS",  SDTX86MaskedStore,
+                       [SDNPHasChain, SDNPMayStore, SDNPMemOperand]>;
+
 def truncstore_s_vi8 : PatFrag<(ops node:$val, node:$ptr),
                                (X86TruncSStore node:$val, node:$ptr), [{
   return cast<MemIntrinsicSDNode>(N)->getMemoryVT().getScalarType() == MVT::i8;
@@ -1680,6 +1692,11 @@ def truncstore_us_vi8 : PatFrag<(ops node:$val, node:$ptr),
   return cast<MemIntrinsicSDNode>(N)->getMemoryVT().getScalarType() == MVT::i8;
 }]>;
 
+def truncstore_ss_vi8 : PatFrag<(ops node:$val, node:$ptr),
+                               (X86TruncSSStore node:$val, node:$ptr), [{
+  return cast<MemIntrinsicSDNode>(N)->getMemoryVT().getScalarType() == MVT::i8;
+}]>;
+
 def truncstore_s_vi16 : PatFrag<(ops node:$val, node:$ptr),
                                (X86TruncSStore node:$val, node:$ptr), [{
   return cast<MemIntrinsicSDNode>(N)->getMemoryVT().getScalarType() == MVT::i16;
@@ -1710,6 +1727,11 @@ def masked_truncstore_us_vi8 : PatFrag<(ops node:$src1, node:$src2, node:$src3),
   return cast<MemIntrinsicSDNode>(N)->getMemoryVT().getScalarType() == MVT::i8;
 }]>;
 
+def masked_truncstore_ss_vi8 : PatFrag<(ops node:$src1, node:$src2, node:$src3),
+                               (X86MTruncSSStore node:$src1, node:$src2, node:$src3), [{
+  return cast<MemIntrinsicSDNode>(N)->getMemoryVT().getScalarType() == MVT::i8;
+}]>;
+
 def masked_truncstore_s_vi16 : PatFrag<(ops node:$src1, node:$src2, node:$src3),
                                (X86MTruncSStore node:$src1, node:$src2, node:$src3), [{
   return cast<MemIntrinsicSDNode>(N)->getMemoryVT().getScalarType() == MVT::i16;
@@ -1777,6 +1799,9 @@ def select_truncs : PatFrag<(ops node:$src, node:$src0, node:$mask),
 def select_truncus : PatFrag<(ops node:$src, node:$src0, node:$mask),
                              (vselect_mask node:$mask,
                                            (X86vtruncus node:$src), node:$src0)>;
+def select_truncss : PatFrag<(ops node:$src, node:$src0, node:$mask),
+                             (vselect_mask node:$mask,
+                                           (X86vtruncss node:$src), node:$src0)>;
 
 def X86Vpshufbitqmb_su : PatFrag<(ops node:$src1, node:$src2),
                                  (X86Vpshufbitqmb node:$src1, node:$src2), [{
diff --git a/llvm/lib/Target/X86/X86IntrinsicsInfo.h b/llvm/lib/Target/X86/X86IntrinsicsInfo.h
index d92834aa5d5c5d..fe34b6da9e563f 100644
--- a/llvm/lib/Target/X86/X86IntrinsicsInfo.h
+++ b/llvm/lib/Target/X86/X86IntrinsicsInfo.h
@@ -100,6 +100,13 @@ struct IntrinsicData {
  * the alphabetical order.
  */
 static const IntrinsicData IntrinsicsWithChain[] = {
+    X86_INTRINSIC_DATA(avx10_mask_pmovss_db_mem_128, TRUNCATE_TO_MEM_VI8,
+                       X86ISD::VTRUNCSS, 0),
+    X86_INTRINSIC_DATA(avx10_mask_pmovss_db_mem_256, TRUNCATE_TO_MEM_VI8,
+                       X86ISD::VTRUNCSS, 0),
+    X86_INTRINSIC_DATA(avx10_mask_pmovss_db_mem_512, TRUNCATE_TO_MEM_VI8,
+                       X86ISD::VTRUNCSS, 0),
+
     X86_INTRINSIC_DATA(avx2_gather_d_d, GATHER_AVX2, 0, 0),
     X86_INTRINSIC_DATA(avx2_gather_d_d_256, GATHER_AVX2, 0, 0),
     X86_INTRINSIC_DATA(avx2_gather_d_pd, GATHER_AVX2, 0, 0),
@@ -407,6 +414,12 @@ static const IntrinsicData IntrinsicsWithoutChain[] = {
                        X86ISD::VGETMANT, 0),
     X86_INTRINSIC_DATA(avx10_mask_getmant_bf16_512, INTR_TYPE_2OP_MASK,
                        X86ISD::VGETMANT, 0),
+    X86_INTRINSIC_DATA(avx10_mask_pmovss_db_128, TRUNCATE_TO_REG,
+                       X86ISD::VTRUNCSS, X86ISD::VMTRUNCSS),
+    X86_INTRINSIC_DATA(avx10_mask_pmovss_db_256, TRUNCATE_TO_REG,
+                       X86ISD::VTRUNCSS, X86ISD::VMTRUNCSS),
+    X86_INTRINSIC_DATA(avx10_mask_pmovss_db_512, TRUNCATE_TO_REG,
+                       X86ISD::VTRUNCSS, X86ISD::VMTRUNCSS),
     X86_INTRINSIC_DATA(avx10_mask_rcp_bf16_128, INTR_TYPE_1OP_MASK,
                        X86ISD::RCP14, 0),
     X86_INTRINSIC_DATA(avx10_mask_rcp_bf16_256, INTR_TYPE_1OP_MASK,
@@ -443,12 +456,6 @@ static const IntrinsicData IntrinsicsWithoutChain[] = {
                        X86ISD::VFPROUND2, 0),
     X86_INTRINSIC_DATA(avx10_mask_vcvt2ps2phx_512, INTR_TYPE_2OP_MASK,
                        X86ISD::VFPROUND2, X86ISD::VFPROUND2_RND),
-    X86_INTRINSIC_DATA(avx10_mask_vcvtbf82ps_128, INTR_TYPE_1OP_MASK,
-                       X86ISD::VCVTBF82PS, 0),
-    X86_INTRINSIC_DATA(avx10_mask_vcvtbf82ps_256, INTR_TYPE_1OP_MASK,
-                       X86ISD::VCVTBF82PS, 0),
-    X86_INTRINSIC_DATA(avx10_mask_vcvtbf82ps_512, INTR_TYPE_1OP_MASK,
-                       X86ISD::VCVTBF82PS, 0),
     X86_INTRINSIC_DATA(avx10_mask_vcvtbiasph2bf8128, TRUNCATE2_TO_REG,
                        X86ISD::VCVTBIASPH2BF8, X86ISD::VMCVTBIASPH2BF8),
     X86_INTRINSIC_DATA(avx10_mask_vcvtbiasph2bf8256, INTR_TYPE_2OP_MASK,
@@ -473,42 +480,12 @@ static const IntrinsicData IntrinsicsWithoutChain[] = {
                        X86ISD::VCVTBIASPH2HF8S, 0),
     X86_INTRINSIC_DATA(avx10_mask_vcvtbiasph2hf8s512, INTR_TYPE_2OP_MASK,
                        X86ISD::VCVTBIASPH2HF8S, 0),
-    X86_INTRINSIC_DATA(avx10_mask_vcvtbiasps2bf8_128, TRUNCATE2_TO_REG,
-                       X86ISD::VCVTBIASPS2BF8, X86ISD::VMCVTBIASPS2BF8),
-    X86_INTRINSIC_DATA(avx10_mask_vcvtbiasps2bf8_256, TRUNCATE2_TO_REG,
-                       X86ISD::VCVTBIASPS2BF8, X86ISD::VMCVTBIASPS2BF8),
-    X86_INTRINSIC_DATA(avx10_mask_vcvtbiasps2bf8_512, INTR_TYPE_2OP_MASK,
-                       X86ISD::VCVTBIASPS2BF8, 0),
-    X86_INTRINSIC_DATA(avx10_mask_vcvtbiasps2bf8s_128, TRUNCATE2_TO_REG,
-                       X86ISD::VCVTBIASPS2BF8S, X86ISD::VMCVTBIASPS2BF8S),
-    X86_INTRINSIC_DATA(avx10_mask_vcvtbiasps2bf8s_256, TRUNCATE2_TO_REG,
-                       X86ISD::VCVTBIASPS2BF8S, X86ISD::VMCVTBIASPS2BF8S),
-    X86_INTRINSIC_DATA(avx10_mask_vcvtbiasps2bf8s_512, INTR_TYPE_2OP_MASK,
-                       X86ISD::VCVTBIASPS2BF8S, 0),
-    X86_INTRINSIC_DATA(avx10_mask_vcvtbiasps2hf8_128, TRUNCATE2_TO_REG,
-                       X86ISD::VCVTBIASPS2HF8, X86ISD::VMCVTBIASPS2HF8),
-    X86_INTRINSIC_DATA(avx10_mask_vcvtbiasps2hf8_256, TRUNCATE2_TO_REG,
-                       X86ISD::VCVTBIASPS2HF8, X86ISD::VMCVTBIASPS2HF8),
-    X86_INTRINSIC_DATA(avx10_mask_vcvtbiasps2hf8_512, INTR_TYPE_2OP_MASK,
-                       X86ISD::VCVTBIASPS2HF8, 0),
-    X86_INTRINSIC_DATA(avx10_mask_vcvtbiasps2hf8s_128, TRUNCATE2_TO_REG,
-                       X86ISD::VCVTBIASPS2HF8S, X86ISD::VMCVTBIASPS2HF8S),
-    X86_INTRINSIC_DATA(avx10_mask_vcvtbiasps2hf8s_256, TRUNCATE2_TO_REG,
-                       X86ISD::VCVTBIASPS2HF8S, X86ISD::VMCVTBIASPS2HF8S),
-    X86_INTRINSIC_DATA(avx10_mask_vcvtbiasps2hf8s_512, INTR_TYPE_2OP_MASK,
-                       X86ISD::VCVTBIASPS2HF8S, 0),
     X86_INTRINSIC_DATA(avx10_mask_vcvthf82ph128, INTR_TYPE_1OP_MASK,
                        X86ISD::VCVTHF82PH, 0),
     X86_INTRINSIC_DATA(avx10_mask_vcvthf82ph256, INTR_TYPE_1OP_MASK,
                        X86ISD::VCVTHF82PH, 0),
     X86_INTRINSIC_DATA(avx10_mask_vcvthf82ph512, INTR_TYPE_1OP_MASK,
                        X86ISD::VCVTHF82PH, 0),
-    X86_INTRINSIC_DATA(avx10_mask_vcvthf82ps_128, INTR_TYPE_1OP_MASK,
-                       X86ISD::VCVTHF82PS, 0),
-    X86_INTRINSIC_DATA(avx10_mask_vcvthf82ps_256, INTR_TYPE_1OP_MASK,
-                       X86ISD::VCVTHF82PS, 0),
-    X86_INTRINSIC_DATA(avx10_mask_vcvthf82ps_512, INTR_TYPE_1OP_MASK,
-                       X86ISD::VCVTHF82PS, 0),
     X86_INTRINSIC_DATA(avx10_mask_vcvtph2bf8128, TRUNCATE_TO_REG,
                        X86ISD::VCVTPH2BF8, X86ISD::VMCVTPH2BF8),
     X86_INTRINSIC_DATA(avx10_mask_vcvtph2bf8256, INTR_TYPE_1OP_MASK,
@@ -545,30 +522,6 @@ static const IntrinsicData IntrinsicsWithoutChain[] = {
                        X86ISD::CVTP2IUBS, 0),
     X86_INTRINSIC_DATA(avx10_mask_vcvtph2iubs512, INTR_TYPE_1OP_MASK,
                        X86ISD::CVTP2IUBS, X86ISD::CVTP2IUBS_RND),
-    X86_INTRINSIC_DATA(avx10_mask_vcvtps2bf8_128, TRUNCATE_TO_REG,
-                       X86ISD::VCVTPS2BF8, X86ISD::VMCVTPS2BF8),
-    X86_INTRINSIC_DATA(avx10_mask_vcvtps2bf8_256, TRUNCATE_TO_REG,
-                       X86ISD::VCVTPS2BF8, X86ISD::VMCVTPS2BF8),
-    X86_INTRINSIC_DATA(avx10_mask_vcvtps2bf8_512, INTR_TYPE_1OP_MASK,
-                       X86ISD::VCVTPS2BF8, 0),
-    X86_INTRINSIC_DATA(avx10_mask_vcvtps2bf8s_128, TRUNCATE_TO_REG,
-                       X86ISD::VCVTPS2BF8S, X86ISD::VMCVTPS2BF8S),
-    X86_INTRINSIC_DATA(avx10_mask_vcvtps2bf8s_256, TRUNCATE_TO_REG,
-                       X86ISD::VCVTPS2BF8S, X86ISD::VMCVTPS2BF8S),
-    X86_INTRINSIC_DATA(avx10_mask_vcvtps2bf8s_512, INTR_TYPE_1OP_MASK,
-                       X86ISD::VCVTPS2BF8S, 0),
-    X86_INTRINSIC_DATA(avx10_mask_vcvtps2hf8_128, TRUNCATE_TO_REG,
-                       X86ISD::VCVTPS2HF8, X86ISD::VMCVTPS2HF8),
-    X86_INTRINSIC_DATA(avx10_mask_vcvtps2hf8_256, TRUNCATE_TO_REG,
-                       X86ISD::VCVTPS2HF8, X86ISD::VMCVTPS2HF8),
-    X86_INTRINSIC_DATA(avx10_mask_vcvtps2hf8_512, INTR_TYPE_1OP_MASK,
-                       X86ISD::VCVTPS2HF8, 0),
-    X86_INTRINSIC_DATA(avx10_mask_vcvtps2hf8s_128, TRUNCATE_TO_REG,
-                       X86ISD::VCVTPS2HF8S, X86ISD::VMCVTPS2HF8S),
-    X86_INTRINSIC_DATA(avx10_mask_vcvtps2hf8s_256, TRUNCATE_TO_REG,
-                       X86ISD::VCVTPS2HF8S, X86ISD::VMCVTPS2HF8S),
-    X86_INTRINSIC_DATA(avx10_mask_vcvtps2hf8s_512, INTR_TYPE_1OP_MASK,
-                       X86ISD::VCVTPS2HF8S, 0),
     X86_INTRINSIC_DATA(avx10_mask_vcvtps2ibs128, INTR_TYPE_1OP_MASK,
                        X86ISD::CVTP2IBS, 0),
     X86_INTRINSIC_DATA(avx10_mask_vcvtps2ibs256, INTR_TYPE_1OP_MASK,
@@ -581,18 +534,6 @@ static const IntrinsicData IntrinsicsWithoutChain[] = {
                        X86ISD::CVTP2IUBS, 0),
     X86_INTRINSIC_DATA(avx10_mask_vcvtps2iubs512, INTR_TYPE_1OP_MASK,
                        X86ISD::CVTP2IUBS, X86ISD::CVTP2IUBS_RND),
-    X86_INTRINSIC_DATA(avx10_mask_vcvtrops2hf8_128, TRUNCATE_TO_REG,
-                       X86ISD::VCVTROPS2HF8, X86ISD::VMCVTROPS2HF8),
-    X86_INTRINSIC_DATA(avx10_mask_vcvtrops2hf8_256, TRUNCATE_TO_REG,
-                       X86ISD::VCVTROPS2HF8, X86ISD::VMCVTROPS2HF8),
-    X86_INTRINSIC_DATA(avx10_mask_vcvtrops2hf8_512, INTR_TYPE_1OP_MASK,
-                       X86ISD::VCVTROPS2HF8, 0),
-    X86_INTRINSIC_DATA(avx10_mask_vcvtrops2hf8s_128, TRUNCATE_TO_REG,
-                       X86ISD::VCVTROPS2HF8S, X86ISD::VMCVTROPS2HF8S),
-    X86_INTRINSIC_DATA(avx10_mask_vcvtrops2hf8s_256, TRUNCATE_TO_REG,
-                       X86ISD::VCVTROPS2HF8S, X86ISD::VMCVTROPS2HF8S),
-    X86_INTRINSIC_DATA(avx10_mask_vcvtrops2hf8s_512, INTR_TYPE_1OP_MASK,
-                       X86ISD::VCVTROPS2HF8S, 0),
     X86_INTRINSIC_DATA(avx10_mask_vcvttpd2dqs_128, CVTPD2DQ_MASK,
                        X86ISD::CVTTP2SIS, X86ISD::MCVTTP2SIS),
     X86_INTRINSIC_DATA(avx10_mask_vcvttpd2dqs_256, INTR_TYPE_1OP_MASK,
@@ -734,6 +675,78 @@ static const IntrinsicData IntrinsicsWithoutChain[] = {
                        0),
     X86_INTRINSIC_DATA(avx10_vcvtbf162iubs512, INTR_TYPE_1OP, X86ISD::CVTP2IUBS,
                        0),
+    X86_INTRINSIC_DATA(avx10_vcvtbf82ps_128, INTR_TYPE_1OP, X86ISD::VCVTBF82PS,
+                       0),
+    X86_INTRINSIC_DATA(avx10_vcvtbf82ps_256, INTR_TYPE_1OP, X86ISD::VCVTBF82PS,
+                       0),
+    X86_INTRINSIC_DATA(avx10_vcvtbf82ps_512, INTR_TYPE_1OP, X86ISD::VCVTBF82PS,
+                       0),
+    X86_INTRINSIC_DATA(avx10_vcvtbiasps2bf8_128, INTR_TYPE_2OP,
+                       X86ISD::VCVTBIASPS2BF8, 0),
+    X86_INTRINSIC_DATA(avx10_vcvtbiasps2bf8_256, INTR_TYPE_2OP,
+                       X86ISD::VCVTBIASPS2BF8, 0),
+    X86_INTRINSIC_DATA(avx10_vcvtbiasps2bf8_512, INTR_TYPE_2OP,
+                       X86ISD::VCVTBIASPS2BF8, 0),
+    X86_INTRINSIC_DATA(avx10_vcvtbiasps2bf8s_128, INTR_TYPE_2OP,
+                       X86ISD::VCVTBIASPS2BF8S, 0),
+    X86_INTRINSIC_DATA(avx10_vcvtbiasps2bf8s_256, INTR_TYPE_2OP,
+                       X86ISD::VCVTBIASPS2BF8S, 0),
+    X86_INTRINSIC_DATA(avx10_vcvtbiasps2bf8s_512, INTR_TYPE_2OP,
+                       X86ISD::VCVTBIASPS2BF8S, 0),
+    X86_INTRINSIC_DATA(avx10_vcvtbiasps2hf8_128, INTR_TYPE_2OP,
+                       X86ISD::VCVTBIASPS2HF8, 0),
+    X86_INTRINSIC_DATA(avx10_vcvtbiasps2hf8_256, INTR_TYPE_2OP,
+                       X86ISD::VCVTBIASPS2HF8, 0),
+    X86_INTRINSIC_DATA(avx10_vcvtbiasps2hf8_512, INTR_TYPE_2OP,
+                       X86ISD::VCVTBIASPS2HF8, 0),
+    X86_INTRINSIC_DATA(avx10_vcvtbiasps2hf8s_128, INTR_TYPE_2OP,
+                       X86ISD::VCVTBIASPS2HF8S, 0),
+    X86_INTRINSIC_DATA(avx10_vcvtbiasps2hf8s_256, INTR_TYPE_2OP,
+                       X86ISD::VCVTBIASPS2HF8S, 0),
+    X86_INTRINSIC_DATA(avx10_vcvtbiasps2hf8s_512, INTR_TYPE_2OP,
+                       X86ISD::VCVTBIASPS2HF8S, 0),
+    X86_INTRINSIC_DATA(avx10_vcvthf82ps_128, INTR_TYPE_1OP, X86ISD::VCVTHF82PS,
+                       0),
+    X86_INTRINSIC_DATA(avx10_vcvthf82ps_256, INTR_TYPE_1OP, X86ISD::VCVTHF82PS,
+                       0),
+    X86_INTRINSIC_DATA(avx10_vcvthf82ps_512, INTR_TYPE_1OP, X86ISD::VCVTHF82PS,
+                       0),
+    X86_INTRINSIC_DATA(avx10_vcvtps2bf8_128, INTR_TYPE_1OP, X86ISD::VCVTPS2BF8,
+                       0),
+    X86_INTRINSIC_DATA(avx10_vcvtps2bf8_256, INTR_TYPE_1OP, X86ISD::VCVTPS2BF8,
+                       0),
+    X86_INTRINSIC_DATA(avx10_vcvtps2bf8_512, INTR_TYPE_1OP, X86ISD::VCVTPS2BF8,
+                       0),
+    X86_INTRINSIC_DATA(avx10_vcvtps2bf8s_128, INTR_TYPE_1OP,
+                       X86ISD::VCVTPS2BF8S, 0),
+    X86_INTRINSIC_DATA(avx10_vcvtps2bf8s_256, INTR_TYPE_1OP,
+                       X86ISD::VCVTPS2BF8S, 0),
+    X86_INTRINSIC_DATA(avx10_vcvtps2bf8s_512, INTR_TYPE_1OP,
+                       X86ISD::VCVTPS2BF8S, 0),
+    X86_INTRINSIC_DATA(avx10_vcvtps2hf8_128, INTR_TYPE_1OP, X86ISD::VCVTPS2HF8,
+                       0),
+    X86_INTRINSIC_DATA(avx10_vcvtps2hf8_256, INTR_TYPE_1OP, X86ISD::VCVTPS2HF8,
+                       0),
+    X86_INTRINSIC_DATA(avx10_vcvtps2hf8_512, INTR_TYPE_1OP, X86ISD::VCVTPS2HF8,
+                       0),
+    X86_INTRINSIC_DATA(avx10_vcvtps2hf8s_128, INTR_TYPE_1OP,
+                       X86ISD::VCVTPS2HF8S, 0),
+    X86_INTRINSIC_DATA(avx10_vcvtps2hf8s_256, INTR_TYPE_1OP,
+                       X86ISD::VCVTPS2HF8S, 0),
+    X86_INTRINSIC_DATA(avx10_vcvtps2hf8s_512, INTR_TYPE_1OP,
+                       X86ISD::VCVTPS2HF8S, 0),
+    X86_INTRINSIC_DATA(avx10_vcvtrops2hf8_128, INTR_TYPE_1OP,
+                       X86ISD::VCVTROPS2HF8, 0),
+    X86_INTRINSIC_DATA(avx10_vcvtrops2hf8_256, INTR_TYPE_1OP,
+                       X86ISD::VCVTROPS2HF8, 0),
+    X86_INTRINSIC_DATA(avx10_vcvtrops2hf8_512, INTR_TYPE_1OP,
+                       X86ISD::VCVTROPS2HF8, 0),
+    X86_INTRINSIC_DATA(avx10_vcvtrops2hf8s_128, INTR_TYPE_1OP,
+                       X86ISD::VCVTROPS2HF8S, 0),
+    X86_INTRINSIC_DATA(avx10_vcvtrops2hf8s_256, INTR_TYPE_1OP,
+                       X86ISD::VCVTROPS2HF8S, 0),
+    X86_INTRINSIC_DATA(avx10_vcvtrops2hf8s_512, INTR_TYPE_1OP,
+                       X86ISD::VCVTROPS2HF8S, 0),
     X86_INTRINSIC_DATA(avx10_vcvttbf162ibs128, INTR_TYPE_1OP, X86ISD::CVTTP2IBS,
                        0),
     X86_INTRINSIC_DATA(avx10_vcvttbf162ibs256, INTR_TYPE_1OP, X86ISD::CVTTP2IBS,
diff --git a/llvm/test/CodeGen/X86/avx10_2_v2aux-intrinsics.ll b/llvm/test/CodeGen/X86/avx10_2_v2aux-intrinsics.ll
deleted file mode 100644
index 980ba1447d5f37..00000000000000
--- a/llvm/test/CodeGen/X86/avx10_2_v2aux-intrinsics.ll
+++ /dev/null
@@ -1,1931 +0,0 @@
-; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
-; RUN: llc < %s -verify-machineinstrs -mtriple=x86_64-unknown-unknown --show-mc-encoding -mattr=+avx10-v2-aux | FileCheck %s --check-prefixes=CHECK,X64
-; RUN: llc < %s -verify-machineinstrs -mtriple=i686-unknown-unknown --show-mc-encoding -mattr=+avx10-v2-aux | FileCheck %s --check-prefixes=CHECK,X86
-
-; ===== Group A: 1-operand truncating conversions (PS->i8) =====
-
-; --- vcvtps2bf8 ---
-
-define <16 x i8> @test_int_x86_avx10_vcvtps2bf8_128(<4 x float> %A) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtps2bf8_128:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtps2bf8 %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x39,0xc0]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.128(<4 x float> %A, <16 x i8> zeroinitializer, i8 -1)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtps2bf8_256(<8 x float> %A) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtps2bf8_256:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtps2bf8 %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x39,0xc0]
-; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.256(<8 x float> %A, <16 x i8> zeroinitializer, i8 -1)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtps2bf8_512(<16 x float> %A) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtps2bf8_512:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtps2bf8 %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x39,0xc0]
-; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.512(<16 x float> %A, <16 x i8> zeroinitializer, i16 -1)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtps2bf8_128(<4 x float> %A, <16 x i8> %B, i8 %C) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtps2bf8_128:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtps2bf8 %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x39,0xc8]
-; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtps2bf8_128:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtps2bf8 %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x39,0xc8]
-; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.128(<4 x float> %A, <16 x i8> %B, i8 %C)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtps2bf8_256(<8 x float> %A, <16 x i8> %B, i8 %C) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtps2bf8_256:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtps2bf8 %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x39,0xc8]
-; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtps2bf8_256:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtps2bf8 %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x39,0xc8]
-; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.256(<8 x float> %A, <16 x i8> %B, i8 %C)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtps2bf8_512(<16 x float> %A, <16 x i8> %B, i16 %C) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtps2bf8_512:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtps2bf8 %zmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x49,0x39,0xc8]
-; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtps2bf8_512:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtps2bf8 %zmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x49,0x39,0xc8]
-; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.512(<16 x float> %A, <16 x i8> %B, i16 %C)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2bf8_128(<4 x float> %A, i8 %B) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8_128:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtps2bf8 %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x89,0x39,0xc0]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8_128:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtps2bf8 %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x89,0x39,0xc0]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.128(<4 x float> %A, <16 x i8> zeroinitializer, i8 %B)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2bf8_256(<8 x float> %A, i8 %B) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8_256:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtps2bf8 %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xa9,0x39,0xc0]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8_256:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtps2bf8 %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xa9,0x39,0xc0]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.256(<8 x float> %A, <16 x i8> zeroinitializer, i8 %B)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2bf8_512(<16 x float> %A, i16 %B) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8_512:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtps2bf8 %zmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xc9,0x39,0xc0]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8_512:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtps2bf8 %zmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xc9,0x39,0xc0]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.512(<16 x float> %A, <16 x i8> zeroinitializer, i16 %B)
-  ret <16 x i8> %ret
-}
-
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.128(<4 x float>, <16 x i8>, i8)
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.256(<8 x float>, <16 x i8>, i8)
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.512(<16 x float>, <16 x i8>, i16)
-
-; --- vcvtps2bf8s ---
-
-define <16 x i8> @test_int_x86_avx10_vcvtps2bf8s_128(<4 x float> %A) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtps2bf8s_128:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtps2bf8s %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x3b,0xc0]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.128(<4 x float> %A, <16 x i8> zeroinitializer, i8 -1)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtps2bf8s_256(<8 x float> %A) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtps2bf8s_256:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtps2bf8s %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x3b,0xc0]
-; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.256(<8 x float> %A, <16 x i8> zeroinitializer, i8 -1)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtps2bf8s_512(<16 x float> %A) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtps2bf8s_512:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtps2bf8s %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x3b,0xc0]
-; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.512(<16 x float> %A, <16 x i8> zeroinitializer, i16 -1)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtps2bf8s_128(<4 x float> %A, <16 x i8> %B, i8 %C) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtps2bf8s_128:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtps2bf8s %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x3b,0xc8]
-; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtps2bf8s_128:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtps2bf8s %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x3b,0xc8]
-; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.128(<4 x float> %A, <16 x i8> %B, i8 %C)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtps2bf8s_256(<8 x float> %A, <16 x i8> %B, i8 %C) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtps2bf8s_256:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtps2bf8s %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x3b,0xc8]
-; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtps2bf8s_256:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtps2bf8s %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x3b,0xc8]
-; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.256(<8 x float> %A, <16 x i8> %B, i8 %C)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtps2bf8s_512(<16 x float> %A, <16 x i8> %B, i16 %C) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtps2bf8s_512:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtps2bf8s %zmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x49,0x3b,0xc8]
-; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtps2bf8s_512:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtps2bf8s %zmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x49,0x3b,0xc8]
-; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.512(<16 x float> %A, <16 x i8> %B, i16 %C)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2bf8s_128(<4 x float> %A, i8 %B) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8s_128:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtps2bf8s %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x89,0x3b,0xc0]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8s_128:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtps2bf8s %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x89,0x3b,0xc0]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.128(<4 x float> %A, <16 x i8> zeroinitializer, i8 %B)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2bf8s_256(<8 x float> %A, i8 %B) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8s_256:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtps2bf8s %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xa9,0x3b,0xc0]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8s_256:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtps2bf8s %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xa9,0x3b,0xc0]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.256(<8 x float> %A, <16 x i8> zeroinitializer, i8 %B)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2bf8s_512(<16 x float> %A, i16 %B) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8s_512:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtps2bf8s %zmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xc9,0x3b,0xc0]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8s_512:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtps2bf8s %zmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xc9,0x3b,0xc0]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.512(<16 x float> %A, <16 x i8> zeroinitializer, i16 %B)
-  ret <16 x i8> %ret
-}
-
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.128(<4 x float>, <16 x i8>, i8)
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.256(<8 x float>, <16 x i8>, i8)
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.512(<16 x float>, <16 x i8>, i16)
-
-; --- vcvtps2hf8 ---
-
-define <16 x i8> @test_int_x86_avx10_vcvtps2hf8_128(<4 x float> %A) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtps2hf8_128:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtps2hf8 %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x38,0xc0]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.128(<4 x float> %A, <16 x i8> zeroinitializer, i8 -1)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtps2hf8_256(<8 x float> %A) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtps2hf8_256:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtps2hf8 %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x38,0xc0]
-; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.256(<8 x float> %A, <16 x i8> zeroinitializer, i8 -1)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtps2hf8_512(<16 x float> %A) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtps2hf8_512:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtps2hf8 %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x38,0xc0]
-; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.512(<16 x float> %A, <16 x i8> zeroinitializer, i16 -1)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtps2hf8_128(<4 x float> %A, <16 x i8> %B, i8 %C) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtps2hf8_128:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtps2hf8 %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x38,0xc8]
-; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtps2hf8_128:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtps2hf8 %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x38,0xc8]
-; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.128(<4 x float> %A, <16 x i8> %B, i8 %C)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtps2hf8_256(<8 x float> %A, <16 x i8> %B, i8 %C) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtps2hf8_256:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtps2hf8 %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x38,0xc8]
-; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtps2hf8_256:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtps2hf8 %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x38,0xc8]
-; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.256(<8 x float> %A, <16 x i8> %B, i8 %C)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtps2hf8_512(<16 x float> %A, <16 x i8> %B, i16 %C) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtps2hf8_512:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtps2hf8 %zmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x49,0x38,0xc8]
-; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtps2hf8_512:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtps2hf8 %zmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x49,0x38,0xc8]
-; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.512(<16 x float> %A, <16 x i8> %B, i16 %C)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2hf8_128(<4 x float> %A, i8 %B) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8_128:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtps2hf8 %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x89,0x38,0xc0]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8_128:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtps2hf8 %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x89,0x38,0xc0]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.128(<4 x float> %A, <16 x i8> zeroinitializer, i8 %B)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2hf8_256(<8 x float> %A, i8 %B) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8_256:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtps2hf8 %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xa9,0x38,0xc0]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8_256:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtps2hf8 %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xa9,0x38,0xc0]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.256(<8 x float> %A, <16 x i8> zeroinitializer, i8 %B)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2hf8_512(<16 x float> %A, i16 %B) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8_512:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtps2hf8 %zmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xc9,0x38,0xc0]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8_512:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtps2hf8 %zmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xc9,0x38,0xc0]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.512(<16 x float> %A, <16 x i8> zeroinitializer, i16 %B)
-  ret <16 x i8> %ret
-}
-
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.128(<4 x float>, <16 x i8>, i8)
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.256(<8 x float>, <16 x i8>, i8)
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.512(<16 x float>, <16 x i8>, i16)
-
-; --- vcvtps2hf8s ---
-
-define <16 x i8> @test_int_x86_avx10_vcvtps2hf8s_128(<4 x float> %A) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtps2hf8s_128:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtps2hf8s %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x3a,0xc0]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.128(<4 x float> %A, <16 x i8> zeroinitializer, i8 -1)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtps2hf8s_256(<8 x float> %A) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtps2hf8s_256:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtps2hf8s %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x3a,0xc0]
-; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.256(<8 x float> %A, <16 x i8> zeroinitializer, i8 -1)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtps2hf8s_512(<16 x float> %A) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtps2hf8s_512:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtps2hf8s %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x3a,0xc0]
-; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.512(<16 x float> %A, <16 x i8> zeroinitializer, i16 -1)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtps2hf8s_128(<4 x float> %A, <16 x i8> %B, i8 %C) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtps2hf8s_128:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtps2hf8s %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x3a,0xc8]
-; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtps2hf8s_128:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtps2hf8s %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x3a,0xc8]
-; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.128(<4 x float> %A, <16 x i8> %B, i8 %C)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtps2hf8s_256(<8 x float> %A, <16 x i8> %B, i8 %C) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtps2hf8s_256:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtps2hf8s %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x3a,0xc8]
-; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtps2hf8s_256:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtps2hf8s %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x3a,0xc8]
-; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.256(<8 x float> %A, <16 x i8> %B, i8 %C)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtps2hf8s_512(<16 x float> %A, <16 x i8> %B, i16 %C) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtps2hf8s_512:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtps2hf8s %zmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x49,0x3a,0xc8]
-; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtps2hf8s_512:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtps2hf8s %zmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x49,0x3a,0xc8]
-; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.512(<16 x float> %A, <16 x i8> %B, i16 %C)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2hf8s_128(<4 x float> %A, i8 %B) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8s_128:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtps2hf8s %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x89,0x3a,0xc0]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8s_128:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtps2hf8s %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x89,0x3a,0xc0]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.128(<4 x float> %A, <16 x i8> zeroinitializer, i8 %B)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2hf8s_256(<8 x float> %A, i8 %B) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8s_256:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtps2hf8s %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xa9,0x3a,0xc0]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8s_256:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtps2hf8s %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xa9,0x3a,0xc0]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.256(<8 x float> %A, <16 x i8> zeroinitializer, i8 %B)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2hf8s_512(<16 x float> %A, i16 %B) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8s_512:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtps2hf8s %zmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xc9,0x3a,0xc0]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8s_512:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtps2hf8s %zmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xc9,0x3a,0xc0]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.512(<16 x float> %A, <16 x i8> zeroinitializer, i16 %B)
-  ret <16 x i8> %ret
-}
-
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.128(<4 x float>, <16 x i8>, i8)
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.256(<8 x float>, <16 x i8>, i8)
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.512(<16 x float>, <16 x i8>, i16)
-
-; --- vcvtrops2hf8 ---
-
-define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8_128(<4 x float> %A) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtrops2hf8_128:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtrops2hf8 %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7d,0x08,0x38,0xc0]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.128(<4 x float> %A, <16 x i8> zeroinitializer, i8 -1)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8_256(<8 x float> %A) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtrops2hf8_256:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtrops2hf8 %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7d,0x28,0x38,0xc0]
-; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.256(<8 x float> %A, <16 x i8> zeroinitializer, i8 -1)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8_512(<16 x float> %A) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtrops2hf8_512:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtrops2hf8 %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7d,0x48,0x38,0xc0]
-; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.512(<16 x float> %A, <16 x i8> zeroinitializer, i16 -1)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtrops2hf8_128(<4 x float> %A, <16 x i8> %B, i8 %C) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8_128:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtrops2hf8 %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7d,0x09,0x38,0xc8]
-; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8_128:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtrops2hf8 %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7d,0x09,0x38,0xc8]
-; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.128(<4 x float> %A, <16 x i8> %B, i8 %C)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtrops2hf8_256(<8 x float> %A, <16 x i8> %B, i8 %C) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8_256:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtrops2hf8 %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7d,0x29,0x38,0xc8]
-; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8_256:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtrops2hf8 %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7d,0x29,0x38,0xc8]
-; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.256(<8 x float> %A, <16 x i8> %B, i8 %C)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtrops2hf8_512(<16 x float> %A, <16 x i8> %B, i16 %C) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8_512:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtrops2hf8 %zmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7d,0x49,0x38,0xc8]
-; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8_512:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtrops2hf8 %zmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7d,0x49,0x38,0xc8]
-; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.512(<16 x float> %A, <16 x i8> %B, i16 %C)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtrops2hf8_128(<4 x float> %A, i8 %B) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8_128:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtrops2hf8 %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0x89,0x38,0xc0]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8_128:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtrops2hf8 %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0x89,0x38,0xc0]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.128(<4 x float> %A, <16 x i8> zeroinitializer, i8 %B)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtrops2hf8_256(<8 x float> %A, i8 %B) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8_256:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtrops2hf8 %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0xa9,0x38,0xc0]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8_256:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtrops2hf8 %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0xa9,0x38,0xc0]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.256(<8 x float> %A, <16 x i8> zeroinitializer, i8 %B)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtrops2hf8_512(<16 x float> %A, i16 %B) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8_512:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtrops2hf8 %zmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0xc9,0x38,0xc0]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8_512:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtrops2hf8 %zmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0xc9,0x38,0xc0]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.512(<16 x float> %A, <16 x i8> zeroinitializer, i16 %B)
-  ret <16 x i8> %ret
-}
-
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.128(<4 x float>, <16 x i8>, i8)
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.256(<8 x float>, <16 x i8>, i8)
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.512(<16 x float>, <16 x i8>, i16)
-
-; --- vcvtrops2hf8s ---
-
-define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8s_128(<4 x float> %A) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtrops2hf8s_128:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtrops2hf8s %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7d,0x08,0x3a,0xc0]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.128(<4 x float> %A, <16 x i8> zeroinitializer, i8 -1)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8s_256(<8 x float> %A) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtrops2hf8s_256:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtrops2hf8s %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7d,0x28,0x3a,0xc0]
-; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.256(<8 x float> %A, <16 x i8> zeroinitializer, i8 -1)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8s_512(<16 x float> %A) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtrops2hf8s_512:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtrops2hf8s %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7d,0x48,0x3a,0xc0]
-; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.512(<16 x float> %A, <16 x i8> zeroinitializer, i16 -1)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtrops2hf8s_128(<4 x float> %A, <16 x i8> %B, i8 %C) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8s_128:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtrops2hf8s %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7d,0x09,0x3a,0xc8]
-; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8s_128:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtrops2hf8s %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7d,0x09,0x3a,0xc8]
-; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.128(<4 x float> %A, <16 x i8> %B, i8 %C)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtrops2hf8s_256(<8 x float> %A, <16 x i8> %B, i8 %C) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8s_256:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtrops2hf8s %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7d,0x29,0x3a,0xc8]
-; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8s_256:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtrops2hf8s %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7d,0x29,0x3a,0xc8]
-; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.256(<8 x float> %A, <16 x i8> %B, i8 %C)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtrops2hf8s_512(<16 x float> %A, <16 x i8> %B, i16 %C) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8s_512:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtrops2hf8s %zmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7d,0x49,0x3a,0xc8]
-; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8s_512:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtrops2hf8s %zmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7d,0x49,0x3a,0xc8]
-; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.512(<16 x float> %A, <16 x i8> %B, i16 %C)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtrops2hf8s_128(<4 x float> %A, i8 %B) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8s_128:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtrops2hf8s %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0x89,0x3a,0xc0]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8s_128:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtrops2hf8s %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0x89,0x3a,0xc0]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.128(<4 x float> %A, <16 x i8> zeroinitializer, i8 %B)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtrops2hf8s_256(<8 x float> %A, i8 %B) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8s_256:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtrops2hf8s %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0xa9,0x3a,0xc0]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8s_256:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtrops2hf8s %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0xa9,0x3a,0xc0]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.256(<8 x float> %A, <16 x i8> zeroinitializer, i8 %B)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtrops2hf8s_512(<16 x float> %A, i16 %B) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8s_512:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtrops2hf8s %zmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0xc9,0x3a,0xc0]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8s_512:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtrops2hf8s %zmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0xc9,0x3a,0xc0]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.512(<16 x float> %A, <16 x i8> zeroinitializer, i16 %B)
-  ret <16 x i8> %ret
-}
-
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.128(<4 x float>, <16 x i8>, i8)
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.256(<8 x float>, <16 x i8>, i8)
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.512(<16 x float>, <16 x i8>, i16)
-
-; ===== Group B: 3-operand bias conversions (bias+PS->i8) =====
-
-; --- vcvtbiasps2bf8 ---
-
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8_128(<16 x i8> %A, <4 x float> %B) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2bf8_128:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtbiasps2bf8 %xmm1, %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x39,0xc1]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.128(<16 x i8> %A, <4 x float> %B, <16 x i8> zeroinitializer, i8 -1)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8_256(<32 x i8> %A, <8 x float> %B) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2bf8_256:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtbiasps2bf8 %ymm1, %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x28,0x39,0xc1]
-; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.256(<32 x i8> %A, <8 x float> %B, <16 x i8> zeroinitializer, i8 -1)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8_512(<64 x i8> %A, <16 x float> %B) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2bf8_512:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtbiasps2bf8 %zmm1, %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x39,0xc1]
-; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.512(<64 x i8> %A, <16 x float> %B, <16 x i8> zeroinitializer, i16 -1)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2bf8_128(<16 x i8> %A, <4 x float> %B, <16 x i8> %C, i8 %D) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8_128:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtbiasps2bf8 %xmm1, %xmm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x39,0xd1]
-; X64-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8_128:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtbiasps2bf8 %xmm1, %xmm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x39,0xd1]
-; X86-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.128(<16 x i8> %A, <4 x float> %B, <16 x i8> %C, i8 %D)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2bf8_256(<32 x i8> %A, <8 x float> %B, <16 x i8> %C, i8 %D) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8_256:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtbiasps2bf8 %ymm1, %ymm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x39,0xd1]
-; X64-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8_256:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtbiasps2bf8 %ymm1, %ymm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x39,0xd1]
-; X86-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.256(<32 x i8> %A, <8 x float> %B, <16 x i8> %C, i8 %D)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2bf8_512(<64 x i8> %A, <16 x float> %B, <16 x i8> %C, i16 %D) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8_512:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtbiasps2bf8 %zmm1, %zmm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x49,0x39,0xd1]
-; X64-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8_512:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtbiasps2bf8 %zmm1, %zmm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x49,0x39,0xd1]
-; X86-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.512(<64 x i8> %A, <16 x float> %B, <16 x i8> %C, i16 %D)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2bf8_128(<16 x i8> %A, <4 x float> %B, i8 %C) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8_128:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtbiasps2bf8 %xmm1, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x39,0xc1]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8_128:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtbiasps2bf8 %xmm1, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x39,0xc1]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.128(<16 x i8> %A, <4 x float> %B, <16 x i8> zeroinitializer, i8 %C)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2bf8_256(<32 x i8> %A, <8 x float> %B, i8 %C) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8_256:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtbiasps2bf8 %ymm1, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xa9,0x39,0xc1]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8_256:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtbiasps2bf8 %ymm1, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xa9,0x39,0xc1]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.256(<32 x i8> %A, <8 x float> %B, <16 x i8> zeroinitializer, i8 %C)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2bf8_512(<64 x i8> %A, <16 x float> %B, i16 %C) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8_512:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtbiasps2bf8 %zmm1, %zmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xc9,0x39,0xc1]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8_512:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtbiasps2bf8 %zmm1, %zmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xc9,0x39,0xc1]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.512(<64 x i8> %A, <16 x float> %B, <16 x i8> zeroinitializer, i16 %C)
-  ret <16 x i8> %ret
-}
-
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.128(<16 x i8>, <4 x float>, <16 x i8>, i8)
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.256(<32 x i8>, <8 x float>, <16 x i8>, i8)
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.512(<64 x i8>, <16 x float>, <16 x i8>, i16)
-
-; --- vcvtbiasps2bf8s ---
-
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8s_128(<16 x i8> %A, <4 x float> %B) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2bf8s_128:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtbiasps2bf8s %xmm1, %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x3b,0xc1]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.128(<16 x i8> %A, <4 x float> %B, <16 x i8> zeroinitializer, i8 -1)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8s_256(<32 x i8> %A, <8 x float> %B) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2bf8s_256:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtbiasps2bf8s %ymm1, %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x28,0x3b,0xc1]
-; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.256(<32 x i8> %A, <8 x float> %B, <16 x i8> zeroinitializer, i8 -1)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8s_512(<64 x i8> %A, <16 x float> %B) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2bf8s_512:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtbiasps2bf8s %zmm1, %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x3b,0xc1]
-; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.512(<64 x i8> %A, <16 x float> %B, <16 x i8> zeroinitializer, i16 -1)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2bf8s_128(<16 x i8> %A, <4 x float> %B, <16 x i8> %C, i8 %D) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8s_128:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtbiasps2bf8s %xmm1, %xmm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x3b,0xd1]
-; X64-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8s_128:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtbiasps2bf8s %xmm1, %xmm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x3b,0xd1]
-; X86-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.128(<16 x i8> %A, <4 x float> %B, <16 x i8> %C, i8 %D)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2bf8s_256(<32 x i8> %A, <8 x float> %B, <16 x i8> %C, i8 %D) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8s_256:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtbiasps2bf8s %ymm1, %ymm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x3b,0xd1]
-; X64-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8s_256:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtbiasps2bf8s %ymm1, %ymm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x3b,0xd1]
-; X86-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.256(<32 x i8> %A, <8 x float> %B, <16 x i8> %C, i8 %D)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2bf8s_512(<64 x i8> %A, <16 x float> %B, <16 x i8> %C, i16 %D) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8s_512:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtbiasps2bf8s %zmm1, %zmm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x49,0x3b,0xd1]
-; X64-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8s_512:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtbiasps2bf8s %zmm1, %zmm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x49,0x3b,0xd1]
-; X86-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.512(<64 x i8> %A, <16 x float> %B, <16 x i8> %C, i16 %D)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2bf8s_128(<16 x i8> %A, <4 x float> %B, i8 %C) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8s_128:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtbiasps2bf8s %xmm1, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x3b,0xc1]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8s_128:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtbiasps2bf8s %xmm1, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x3b,0xc1]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.128(<16 x i8> %A, <4 x float> %B, <16 x i8> zeroinitializer, i8 %C)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2bf8s_256(<32 x i8> %A, <8 x float> %B, i8 %C) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8s_256:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtbiasps2bf8s %ymm1, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xa9,0x3b,0xc1]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8s_256:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtbiasps2bf8s %ymm1, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xa9,0x3b,0xc1]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.256(<32 x i8> %A, <8 x float> %B, <16 x i8> zeroinitializer, i8 %C)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2bf8s_512(<64 x i8> %A, <16 x float> %B, i16 %C) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8s_512:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtbiasps2bf8s %zmm1, %zmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xc9,0x3b,0xc1]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8s_512:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtbiasps2bf8s %zmm1, %zmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xc9,0x3b,0xc1]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.512(<64 x i8> %A, <16 x float> %B, <16 x i8> zeroinitializer, i16 %C)
-  ret <16 x i8> %ret
-}
-
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.128(<16 x i8>, <4 x float>, <16 x i8>, i8)
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.256(<32 x i8>, <8 x float>, <16 x i8>, i8)
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.512(<64 x i8>, <16 x float>, <16 x i8>, i16)
-
-; --- vcvtbiasps2hf8 ---
-
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8_128(<16 x i8> %A, <4 x float> %B) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2hf8_128:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtbiasps2hf8 %xmm1, %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x38,0xc1]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.128(<16 x i8> %A, <4 x float> %B, <16 x i8> zeroinitializer, i8 -1)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8_256(<32 x i8> %A, <8 x float> %B) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2hf8_256:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtbiasps2hf8 %ymm1, %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x28,0x38,0xc1]
-; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.256(<32 x i8> %A, <8 x float> %B, <16 x i8> zeroinitializer, i8 -1)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8_512(<64 x i8> %A, <16 x float> %B) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2hf8_512:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtbiasps2hf8 %zmm1, %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x38,0xc1]
-; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.512(<64 x i8> %A, <16 x float> %B, <16 x i8> zeroinitializer, i16 -1)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2hf8_128(<16 x i8> %A, <4 x float> %B, <16 x i8> %C, i8 %D) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8_128:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtbiasps2hf8 %xmm1, %xmm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x38,0xd1]
-; X64-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8_128:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtbiasps2hf8 %xmm1, %xmm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x38,0xd1]
-; X86-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.128(<16 x i8> %A, <4 x float> %B, <16 x i8> %C, i8 %D)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2hf8_256(<32 x i8> %A, <8 x float> %B, <16 x i8> %C, i8 %D) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8_256:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtbiasps2hf8 %ymm1, %ymm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x38,0xd1]
-; X64-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8_256:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtbiasps2hf8 %ymm1, %ymm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x38,0xd1]
-; X86-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.256(<32 x i8> %A, <8 x float> %B, <16 x i8> %C, i8 %D)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2hf8_512(<64 x i8> %A, <16 x float> %B, <16 x i8> %C, i16 %D) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8_512:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtbiasps2hf8 %zmm1, %zmm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x49,0x38,0xd1]
-; X64-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8_512:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtbiasps2hf8 %zmm1, %zmm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x49,0x38,0xd1]
-; X86-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.512(<64 x i8> %A, <16 x float> %B, <16 x i8> %C, i16 %D)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2hf8_128(<16 x i8> %A, <4 x float> %B, i8 %C) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8_128:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtbiasps2hf8 %xmm1, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x38,0xc1]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8_128:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtbiasps2hf8 %xmm1, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x38,0xc1]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.128(<16 x i8> %A, <4 x float> %B, <16 x i8> zeroinitializer, i8 %C)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2hf8_256(<32 x i8> %A, <8 x float> %B, i8 %C) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8_256:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtbiasps2hf8 %ymm1, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xa9,0x38,0xc1]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8_256:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtbiasps2hf8 %ymm1, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xa9,0x38,0xc1]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.256(<32 x i8> %A, <8 x float> %B, <16 x i8> zeroinitializer, i8 %C)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2hf8_512(<64 x i8> %A, <16 x float> %B, i16 %C) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8_512:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtbiasps2hf8 %zmm1, %zmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xc9,0x38,0xc1]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8_512:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtbiasps2hf8 %zmm1, %zmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xc9,0x38,0xc1]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.512(<64 x i8> %A, <16 x float> %B, <16 x i8> zeroinitializer, i16 %C)
-  ret <16 x i8> %ret
-}
-
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.128(<16 x i8>, <4 x float>, <16 x i8>, i8)
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.256(<32 x i8>, <8 x float>, <16 x i8>, i8)
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.512(<64 x i8>, <16 x float>, <16 x i8>, i16)
-
-; --- vcvtbiasps2hf8s ---
-
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8s_128(<16 x i8> %A, <4 x float> %B) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2hf8s_128:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtbiasps2hf8s %xmm1, %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x3a,0xc1]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.128(<16 x i8> %A, <4 x float> %B, <16 x i8> zeroinitializer, i8 -1)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8s_256(<32 x i8> %A, <8 x float> %B) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2hf8s_256:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtbiasps2hf8s %ymm1, %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x28,0x3a,0xc1]
-; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.256(<32 x i8> %A, <8 x float> %B, <16 x i8> zeroinitializer, i8 -1)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8s_512(<64 x i8> %A, <16 x float> %B) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2hf8s_512:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtbiasps2hf8s %zmm1, %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x3a,0xc1]
-; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.512(<64 x i8> %A, <16 x float> %B, <16 x i8> zeroinitializer, i16 -1)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2hf8s_128(<16 x i8> %A, <4 x float> %B, <16 x i8> %C, i8 %D) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8s_128:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtbiasps2hf8s %xmm1, %xmm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x3a,0xd1]
-; X64-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8s_128:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtbiasps2hf8s %xmm1, %xmm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x3a,0xd1]
-; X86-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.128(<16 x i8> %A, <4 x float> %B, <16 x i8> %C, i8 %D)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2hf8s_256(<32 x i8> %A, <8 x float> %B, <16 x i8> %C, i8 %D) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8s_256:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtbiasps2hf8s %ymm1, %ymm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x3a,0xd1]
-; X64-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8s_256:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtbiasps2hf8s %ymm1, %ymm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x3a,0xd1]
-; X86-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.256(<32 x i8> %A, <8 x float> %B, <16 x i8> %C, i8 %D)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2hf8s_512(<64 x i8> %A, <16 x float> %B, <16 x i8> %C, i16 %D) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8s_512:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtbiasps2hf8s %zmm1, %zmm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x49,0x3a,0xd1]
-; X64-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8s_512:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtbiasps2hf8s %zmm1, %zmm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x49,0x3a,0xd1]
-; X86-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.512(<64 x i8> %A, <16 x float> %B, <16 x i8> %C, i16 %D)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2hf8s_128(<16 x i8> %A, <4 x float> %B, i8 %C) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8s_128:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtbiasps2hf8s %xmm1, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x3a,0xc1]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8s_128:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtbiasps2hf8s %xmm1, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x3a,0xc1]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.128(<16 x i8> %A, <4 x float> %B, <16 x i8> zeroinitializer, i8 %C)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2hf8s_256(<32 x i8> %A, <8 x float> %B, i8 %C) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8s_256:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtbiasps2hf8s %ymm1, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xa9,0x3a,0xc1]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8s_256:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtbiasps2hf8s %ymm1, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xa9,0x3a,0xc1]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.256(<32 x i8> %A, <8 x float> %B, <16 x i8> zeroinitializer, i8 %C)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2hf8s_512(<64 x i8> %A, <16 x float> %B, i16 %C) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8s_512:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtbiasps2hf8s %zmm1, %zmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xc9,0x3a,0xc1]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8s_512:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtbiasps2hf8s %zmm1, %zmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xc9,0x3a,0xc1]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.512(<64 x i8> %A, <16 x float> %B, <16 x i8> zeroinitializer, i16 %C)
-  ret <16 x i8> %ret
-}
-
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.128(<16 x i8>, <4 x float>, <16 x i8>, i8)
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.256(<32 x i8>, <8 x float>, <16 x i8>, i8)
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.512(<64 x i8>, <16 x float>, <16 x i8>, i16)
-
-; ===== Group C: 8bit->PS expanding conversions =====
-
-; --- vcvtbf82ps ---
-
-define <4 x float> @test_int_x86_avx10_vcvtbf82ps_128(<16 x i8> %A) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbf82ps_128:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtbf82ps %xmm0, %xmm0 # encoding: [0x62,0xf5,0xfc,0x08,0x36,0xc0]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <4 x float> @llvm.x86.avx10.mask.vcvtbf82ps.128(<16 x i8> %A, <4 x float> zeroinitializer, i8 -1)
-  ret <4 x float> %ret
-}
-
-define <8 x float> @test_int_x86_avx10_vcvtbf82ps_256(<16 x i8> %A) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbf82ps_256:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtbf82ps %xmm0, %ymm0 # encoding: [0x62,0xf5,0xfc,0x28,0x36,0xc0]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <8 x float> @llvm.x86.avx10.mask.vcvtbf82ps.256(<16 x i8> %A, <8 x float> zeroinitializer, i8 -1)
-  ret <8 x float> %ret
-}
-
-define <16 x float> @test_int_x86_avx10_vcvtbf82ps_512(<16 x i8> %A) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbf82ps_512:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtbf82ps %xmm0, %zmm0 # encoding: [0x62,0xf5,0xfc,0x48,0x36,0xc0]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x float> @llvm.x86.avx10.mask.vcvtbf82ps.512(<16 x i8> %A, <16 x float> zeroinitializer, i16 -1)
-  ret <16 x float> %ret
-}
-
-define <4 x float> @test_int_x86_avx10_mask_vcvtbf82ps_128(<16 x i8> %A, <4 x float> %B, i8 %C) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtbf82ps_128:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtbf82ps %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0xfc,0x09,0x36,0xc8]
-; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtbf82ps_128:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtbf82ps %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0xfc,0x09,0x36,0xc8]
-; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <4 x float> @llvm.x86.avx10.mask.vcvtbf82ps.128(<16 x i8> %A, <4 x float> %B, i8 %C)
-  ret <4 x float> %ret
-}
-
-define <8 x float> @test_int_x86_avx10_mask_vcvtbf82ps_256(<16 x i8> %A, <8 x float> %B, i8 %C) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtbf82ps_256:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtbf82ps %xmm0, %ymm1 {%k1} # encoding: [0x62,0xf5,0xfc,0x29,0x36,0xc8]
-; X64-NEXT:    vmovaps %ymm1, %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfc,0x28,0xc1]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtbf82ps_256:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtbf82ps %xmm0, %ymm1 {%k1} # encoding: [0x62,0xf5,0xfc,0x29,0x36,0xc8]
-; X86-NEXT:    vmovaps %ymm1, %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfc,0x28,0xc1]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <8 x float> @llvm.x86.avx10.mask.vcvtbf82ps.256(<16 x i8> %A, <8 x float> %B, i8 %C)
-  ret <8 x float> %ret
-}
-
-define <16 x float> @test_int_x86_avx10_mask_vcvtbf82ps_512(<16 x i8> %A, <16 x float> %B, i16 %C) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtbf82ps_512:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtbf82ps %xmm0, %zmm1 {%k1} # encoding: [0x62,0xf5,0xfc,0x49,0x36,0xc8]
-; X64-NEXT:    vmovaps %zmm1, %zmm0 # encoding: [0x62,0xf1,0x7c,0x48,0x28,0xc1]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtbf82ps_512:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtbf82ps %xmm0, %zmm1 {%k1} # encoding: [0x62,0xf5,0xfc,0x49,0x36,0xc8]
-; X86-NEXT:    vmovaps %zmm1, %zmm0 # encoding: [0x62,0xf1,0x7c,0x48,0x28,0xc1]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x float> @llvm.x86.avx10.mask.vcvtbf82ps.512(<16 x i8> %A, <16 x float> %B, i16 %C)
-  ret <16 x float> %ret
-}
-
-define <4 x float> @test_int_x86_avx10_maskz_vcvtbf82ps_128(<16 x i8> %A, i8 %B) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtbf82ps_128:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtbf82ps %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0xfc,0x89,0x36,0xc0]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtbf82ps_128:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtbf82ps %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0xfc,0x89,0x36,0xc0]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <4 x float> @llvm.x86.avx10.mask.vcvtbf82ps.128(<16 x i8> %A, <4 x float> zeroinitializer, i8 %B)
-  ret <4 x float> %ret
-}
-
-define <8 x float> @test_int_x86_avx10_maskz_vcvtbf82ps_256(<16 x i8> %A, i8 %B) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtbf82ps_256:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtbf82ps %xmm0, %ymm0 {%k1} {z} # encoding: [0x62,0xf5,0xfc,0xa9,0x36,0xc0]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtbf82ps_256:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtbf82ps %xmm0, %ymm0 {%k1} {z} # encoding: [0x62,0xf5,0xfc,0xa9,0x36,0xc0]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <8 x float> @llvm.x86.avx10.mask.vcvtbf82ps.256(<16 x i8> %A, <8 x float> zeroinitializer, i8 %B)
-  ret <8 x float> %ret
-}
-
-define <16 x float> @test_int_x86_avx10_maskz_vcvtbf82ps_512(<16 x i8> %A, i16 %B) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtbf82ps_512:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtbf82ps %xmm0, %zmm0 {%k1} {z} # encoding: [0x62,0xf5,0xfc,0xc9,0x36,0xc0]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtbf82ps_512:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtbf82ps %xmm0, %zmm0 {%k1} {z} # encoding: [0x62,0xf5,0xfc,0xc9,0x36,0xc0]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x float> @llvm.x86.avx10.mask.vcvtbf82ps.512(<16 x i8> %A, <16 x float> zeroinitializer, i16 %B)
-  ret <16 x float> %ret
-}
-
-declare <4 x float> @llvm.x86.avx10.mask.vcvtbf82ps.128(<16 x i8>, <4 x float>, i8)
-declare <8 x float> @llvm.x86.avx10.mask.vcvtbf82ps.256(<16 x i8>, <8 x float>, i8)
-declare <16 x float> @llvm.x86.avx10.mask.vcvtbf82ps.512(<16 x i8>, <16 x float>, i16)
-
-; --- vcvthf82ps ---
-
-define <4 x float> @test_int_x86_avx10_vcvthf82ps_128(<16 x i8> %A) {
-; CHECK-LABEL: test_int_x86_avx10_vcvthf82ps_128:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvthf82ps %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x36,0xc0]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <4 x float> @llvm.x86.avx10.mask.vcvthf82ps.128(<16 x i8> %A, <4 x float> zeroinitializer, i8 -1)
-  ret <4 x float> %ret
-}
-
-define <8 x float> @test_int_x86_avx10_vcvthf82ps_256(<16 x i8> %A) {
-; CHECK-LABEL: test_int_x86_avx10_vcvthf82ps_256:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvthf82ps %xmm0, %ymm0 # encoding: [0x62,0xf5,0x7c,0x28,0x36,0xc0]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <8 x float> @llvm.x86.avx10.mask.vcvthf82ps.256(<16 x i8> %A, <8 x float> zeroinitializer, i8 -1)
-  ret <8 x float> %ret
-}
-
-define <16 x float> @test_int_x86_avx10_vcvthf82ps_512(<16 x i8> %A) {
-; CHECK-LABEL: test_int_x86_avx10_vcvthf82ps_512:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvthf82ps %xmm0, %zmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x36,0xc0]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x float> @llvm.x86.avx10.mask.vcvthf82ps.512(<16 x i8> %A, <16 x float> zeroinitializer, i16 -1)
-  ret <16 x float> %ret
-}
-
-define <4 x float> @test_int_x86_avx10_mask_vcvthf82ps_128(<16 x i8> %A, <4 x float> %B, i8 %C) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvthf82ps_128:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvthf82ps %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x36,0xc8]
-; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvthf82ps_128:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvthf82ps %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x36,0xc8]
-; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <4 x float> @llvm.x86.avx10.mask.vcvthf82ps.128(<16 x i8> %A, <4 x float> %B, i8 %C)
-  ret <4 x float> %ret
-}
-
-define <8 x float> @test_int_x86_avx10_mask_vcvthf82ps_256(<16 x i8> %A, <8 x float> %B, i8 %C) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvthf82ps_256:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvthf82ps %xmm0, %ymm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x36,0xc8]
-; X64-NEXT:    vmovaps %ymm1, %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfc,0x28,0xc1]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvthf82ps_256:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvthf82ps %xmm0, %ymm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x36,0xc8]
-; X86-NEXT:    vmovaps %ymm1, %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfc,0x28,0xc1]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <8 x float> @llvm.x86.avx10.mask.vcvthf82ps.256(<16 x i8> %A, <8 x float> %B, i8 %C)
-  ret <8 x float> %ret
-}
-
-define <16 x float> @test_int_x86_avx10_mask_vcvthf82ps_512(<16 x i8> %A, <16 x float> %B, i16 %C) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvthf82ps_512:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvthf82ps %xmm0, %zmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x49,0x36,0xc8]
-; X64-NEXT:    vmovaps %zmm1, %zmm0 # encoding: [0x62,0xf1,0x7c,0x48,0x28,0xc1]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvthf82ps_512:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvthf82ps %xmm0, %zmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x49,0x36,0xc8]
-; X86-NEXT:    vmovaps %zmm1, %zmm0 # encoding: [0x62,0xf1,0x7c,0x48,0x28,0xc1]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x float> @llvm.x86.avx10.mask.vcvthf82ps.512(<16 x i8> %A, <16 x float> %B, i16 %C)
-  ret <16 x float> %ret
-}
-
-define <4 x float> @test_int_x86_avx10_maskz_vcvthf82ps_128(<16 x i8> %A, i8 %B) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvthf82ps_128:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvthf82ps %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x36,0xc0]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvthf82ps_128:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvthf82ps %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x36,0xc0]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <4 x float> @llvm.x86.avx10.mask.vcvthf82ps.128(<16 x i8> %A, <4 x float> zeroinitializer, i8 %B)
-  ret <4 x float> %ret
-}
-
-define <8 x float> @test_int_x86_avx10_maskz_vcvthf82ps_256(<16 x i8> %A, i8 %B) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvthf82ps_256:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvthf82ps %xmm0, %ymm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xa9,0x36,0xc0]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvthf82ps_256:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvthf82ps %xmm0, %ymm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xa9,0x36,0xc0]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <8 x float> @llvm.x86.avx10.mask.vcvthf82ps.256(<16 x i8> %A, <8 x float> zeroinitializer, i8 %B)
-  ret <8 x float> %ret
-}
-
-define <16 x float> @test_int_x86_avx10_maskz_vcvthf82ps_512(<16 x i8> %A, i16 %B) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvthf82ps_512:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvthf82ps %xmm0, %zmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xc9,0x36,0xc0]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvthf82ps_512:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvthf82ps %xmm0, %zmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xc9,0x36,0xc0]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x float> @llvm.x86.avx10.mask.vcvthf82ps.512(<16 x i8> %A, <16 x float> zeroinitializer, i16 %B)
-  ret <16 x float> %ret
-}
-
-declare <4 x float> @llvm.x86.avx10.mask.vcvthf82ps.128(<16 x i8>, <4 x float>, i8)
-declare <8 x float> @llvm.x86.avx10.mask.vcvthf82ps.256(<16 x i8>, <8 x float>, i8)
-declare <16 x float> @llvm.x86.avx10.mask.vcvthf82ps.512(<16 x i8>, <16 x float>, i16)
-
-; ===== Group E: Same-size reg-only conversions =====
-
-; --- vcvtbf82bf6s ---
-
-define <16 x i8> @test_int_x86_avx10_vcvtbf82bf6s_128(<16 x i8> %A) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbf82bf6s_128:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtbf82bf6s %xmm0, %xmm0 # encoding: [0x62,0xf5,0xfe,0x08,0x3e,0xc0]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf82bf6s.128(<16 x i8> %A)
-  ret <16 x i8> %ret
-}
-
-define <32 x i8> @test_int_x86_avx10_vcvtbf82bf6s_256(<32 x i8> %A) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbf82bf6s_256:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtbf82bf6s %ymm0, %ymm0 # encoding: [0x62,0xf5,0xfe,0x28,0x3e,0xc0]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <32 x i8> @llvm.x86.avx10.vcvtbf82bf6s.256(<32 x i8> %A)
-  ret <32 x i8> %ret
-}
-
-define <64 x i8> @test_int_x86_avx10_vcvtbf82bf6s_512(<64 x i8> %A) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbf82bf6s_512:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtbf82bf6s %zmm0, %zmm0 # encoding: [0x62,0xf5,0xfe,0x48,0x3e,0xc0]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <64 x i8> @llvm.x86.avx10.vcvtbf82bf6s.512(<64 x i8> %A)
-  ret <64 x i8> %ret
-}
-
-declare <16 x i8> @llvm.x86.avx10.vcvtbf82bf6s.128(<16 x i8>)
-declare <32 x i8> @llvm.x86.avx10.vcvtbf82bf6s.256(<32 x i8>)
-declare <64 x i8> @llvm.x86.avx10.vcvtbf82bf6s.512(<64 x i8>)
-
-; --- vcvthf82hf6s ---
-
-define <16 x i8> @test_int_x86_avx10_vcvthf82hf6s_128(<16 x i8> %A) {
-; CHECK-LABEL: test_int_x86_avx10_vcvthf82hf6s_128:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvthf82hf6s %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x3c,0xc0]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvthf82hf6s.128(<16 x i8> %A)
-  ret <16 x i8> %ret
-}
-
-define <32 x i8> @test_int_x86_avx10_vcvthf82hf6s_256(<32 x i8> %A) {
-; CHECK-LABEL: test_int_x86_avx10_vcvthf82hf6s_256:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvthf82hf6s %ymm0, %ymm0 # encoding: [0x62,0xf5,0x7e,0x28,0x3c,0xc0]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <32 x i8> @llvm.x86.avx10.vcvthf82hf6s.256(<32 x i8> %A)
-  ret <32 x i8> %ret
-}
-
-define <64 x i8> @test_int_x86_avx10_vcvthf82hf6s_512(<64 x i8> %A) {
-; CHECK-LABEL: test_int_x86_avx10_vcvthf82hf6s_512:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvthf82hf6s %zmm0, %zmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x3c,0xc0]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <64 x i8> @llvm.x86.avx10.vcvthf82hf6s.512(<64 x i8> %A)
-  ret <64 x i8> %ret
-}
-
-declare <16 x i8> @llvm.x86.avx10.vcvthf82hf6s.128(<16 x i8>)
-declare <32 x i8> @llvm.x86.avx10.vcvthf82hf6s.256(<32 x i8>)
-declare <64 x i8> @llvm.x86.avx10.vcvthf82hf6s.512(<64 x i8>)
-
-; ===== Group F: Expanding/same-size conversions =====
-
-; --- vcvtbf42hf8 ---
-
-define <16 x i8> @test_int_x86_avx10_vcvtbf42hf8_128(<16 x i8> %A) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbf42hf8_128:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtbf42hf8 %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x37,0xc0]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf42hf8.128(<16 x i8> %A)
-  ret <16 x i8> %ret
-}
-
-define <32 x i8> @test_int_x86_avx10_vcvtbf42hf8_256(<16 x i8> %A) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbf42hf8_256:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtbf42hf8 %xmm0, %ymm0 # encoding: [0x62,0xf5,0x7c,0x28,0x37,0xc0]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <32 x i8> @llvm.x86.avx10.vcvtbf42hf8.256(<16 x i8> %A)
-  ret <32 x i8> %ret
-}
-
-define <64 x i8> @test_int_x86_avx10_vcvtbf42hf8_512(<32 x i8> %A) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbf42hf8_512:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtbf42hf8 %ymm0, %zmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x37,0xc0]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <64 x i8> @llvm.x86.avx10.vcvtbf42hf8.512(<32 x i8> %A)
-  ret <64 x i8> %ret
-}
-
-declare <16 x i8> @llvm.x86.avx10.vcvtbf42hf8.128(<16 x i8>)
-declare <32 x i8> @llvm.x86.avx10.vcvtbf42hf8.256(<16 x i8>)
-declare <64 x i8> @llvm.x86.avx10.vcvtbf42hf8.512(<32 x i8>)
-
-; --- vcvtbf62hf8 ---
-
-define <16 x i8> @test_int_x86_avx10_vcvtbf62hf8_128(<16 x i8> %A) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbf62hf8_128:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtbf62hf8 %xmm0, %xmm0 # encoding: [0x62,0xf5,0xfd,0x08,0x37,0xc0]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf62hf8.128(<16 x i8> %A)
-  ret <16 x i8> %ret
-}
-
-define <32 x i8> @test_int_x86_avx10_vcvtbf62hf8_256(<32 x i8> %A) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbf62hf8_256:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtbf62hf8 %ymm0, %ymm0 # encoding: [0x62,0xf5,0xfd,0x28,0x37,0xc0]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <32 x i8> @llvm.x86.avx10.vcvtbf62hf8.256(<32 x i8> %A)
-  ret <32 x i8> %ret
-}
-
-define <64 x i8> @test_int_x86_avx10_vcvtbf62hf8_512(<64 x i8> %A) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbf62hf8_512:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtbf62hf8 %zmm0, %zmm0 # encoding: [0x62,0xf5,0xfd,0x48,0x37,0xc0]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <64 x i8> @llvm.x86.avx10.vcvtbf62hf8.512(<64 x i8> %A)
-  ret <64 x i8> %ret
-}
-
-declare <16 x i8> @llvm.x86.avx10.vcvtbf62hf8.128(<16 x i8>)
-declare <32 x i8> @llvm.x86.avx10.vcvtbf62hf8.256(<32 x i8>)
-declare <64 x i8> @llvm.x86.avx10.vcvtbf62hf8.512(<64 x i8>)
-
-; --- vcvthf62hf8 ---
-
-define <16 x i8> @test_int_x86_avx10_vcvthf62hf8_128(<16 x i8> %A) {
-; CHECK-LABEL: test_int_x86_avx10_vcvthf62hf8_128:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvthf62hf8 %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7d,0x08,0x37,0xc0]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvthf62hf8.128(<16 x i8> %A)
-  ret <16 x i8> %ret
-}
-
-define <32 x i8> @test_int_x86_avx10_vcvthf62hf8_256(<32 x i8> %A) {
-; CHECK-LABEL: test_int_x86_avx10_vcvthf62hf8_256:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvthf62hf8 %ymm0, %ymm0 # encoding: [0x62,0xf5,0x7d,0x28,0x37,0xc0]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <32 x i8> @llvm.x86.avx10.vcvthf62hf8.256(<32 x i8> %A)
-  ret <32 x i8> %ret
-}
-
-define <64 x i8> @test_int_x86_avx10_vcvthf62hf8_512(<64 x i8> %A) {
-; CHECK-LABEL: test_int_x86_avx10_vcvthf62hf8_512:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvthf62hf8 %zmm0, %zmm0 # encoding: [0x62,0xf5,0x7d,0x48,0x37,0xc0]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <64 x i8> @llvm.x86.avx10.vcvthf62hf8.512(<64 x i8> %A)
-  ret <64 x i8> %ret
-}
-
-declare <16 x i8> @llvm.x86.avx10.vcvthf62hf8.128(<16 x i8>)
-declare <32 x i8> @llvm.x86.avx10.vcvthf62hf8.256(<32 x i8>)
-declare <64 x i8> @llvm.x86.avx10.vcvthf62hf8.512(<64 x i8>)
-
-; ===== Group H: VUNPACKB =====
-
-define <16 x i8> @test_int_x86_avx10_vunpackb_128(<16 x i8> %A) {
-; CHECK-LABEL: test_int_x86_avx10_vunpackb_128:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vunpackb $1, %xmm0, %xmm0 # encoding: [0x62,0xf3,0x7c,0x08,0x3d,0xc0,0x01]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vunpackb.128(<16 x i8> %A, i8 1)
-  ret <16 x i8> %ret
-}
-
-define <32 x i8> @test_int_x86_avx10_vunpackb_256(<32 x i8> %A) {
-; CHECK-LABEL: test_int_x86_avx10_vunpackb_256:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vunpackb $2, %ymm0, %ymm0 # encoding: [0x62,0xf3,0x7c,0x28,0x3d,0xc0,0x02]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <32 x i8> @llvm.x86.avx10.vunpackb.256(<32 x i8> %A, i8 2)
-  ret <32 x i8> %ret
-}
-
-define <64 x i8> @test_int_x86_avx10_vunpackb_512(<64 x i8> %A) {
-; CHECK-LABEL: test_int_x86_avx10_vunpackb_512:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vunpackb $3, %zmm0, %zmm0 # encoding: [0x62,0xf3,0x7c,0x48,0x3d,0xc0,0x03]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <64 x i8> @llvm.x86.avx10.vunpackb.512(<64 x i8> %A, i8 3)
-  ret <64 x i8> %ret
-}
-
-declare <16 x i8> @llvm.x86.avx10.vunpackb.128(<16 x i8>, i8)
-declare <32 x i8> @llvm.x86.avx10.vunpackb.256(<32 x i8>, i8)
-declare <64 x i8> @llvm.x86.avx10.vunpackb.512(<64 x i8>, i8)
diff --git a/llvm/test/CodeGen/X86/avx10_v2aux-intrinsics.ll b/llvm/test/CodeGen/X86/avx10_v2aux-intrinsics.ll
new file mode 100644
index 00000000000000..53cfea524ce7f6
--- /dev/null
+++ b/llvm/test/CodeGen/X86/avx10_v2aux-intrinsics.ll
@@ -0,0 +1,1773 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
+; RUN: llc < %s -verify-machineinstrs -mtriple=x86_64-unknown-unknown --show-mc-encoding -mattr=+avx10v2aux | FileCheck %s --check-prefixes=CHECK,X64
+; RUN: llc < %s -verify-machineinstrs -mtriple=i686-unknown-unknown --show-mc-encoding -mattr=+avx10v2aux | FileCheck %s --check-prefixes=CHECK,X86
+
+define <16 x i8> @test_int_x86_avx10_vcvtps2bf8_128(<4 x float> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtps2bf8_128:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtps2bf8 %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x39,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8.128(<4 x float> %a)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtps2bf8_256(<8 x float> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtps2bf8_256:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtps2bf8 %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x39,0xc0]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8.256(<8 x float> %a)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtps2bf8_512(<16 x float> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtps2bf8_512:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtps2bf8 %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x39,0xc0]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8.512(<16 x float> %a)
+  ret <16 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.vcvtps2bf8.128(<4 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtps2bf8.256(<8 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtps2bf8.512(<16 x float>)
+
+; Memory folding tests for vcvtps2bf8
+define <16 x i8> @test_int_x86_avx10_vcvtps2bf8_mem_128(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtps2bf8_mem_128:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtps2bf8x (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x39,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtps2bf8_mem_128:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtps2bf8x (%eax), %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x39,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <4 x float>, ptr %ptr_a
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8.128(<4 x float> %a)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtps2bf8_mem_256(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtps2bf8_mem_256:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtps2bf8y (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x39,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtps2bf8_mem_256:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtps2bf8y (%eax), %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x39,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <8 x float>, ptr %ptr_a
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8.256(<8 x float> %a)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtps2bf8_mem_512(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtps2bf8_mem_512:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtps2bf8 (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x39,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtps2bf8_mem_512:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtps2bf8 (%eax), %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x39,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <16 x float>, ptr %ptr_a
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8.512(<16 x float> %a)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtps2bf8s_128(<4 x float> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtps2bf8s_128:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtps2bf8s %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x3b,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8s.128(<4 x float> %a)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtps2bf8s_256(<8 x float> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtps2bf8s_256:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtps2bf8s %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x3b,0xc0]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8s.256(<8 x float> %a)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtps2bf8s_512(<16 x float> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtps2bf8s_512:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtps2bf8s %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x3b,0xc0]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8s.512(<16 x float> %a)
+  ret <16 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.vcvtps2bf8s.128(<4 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtps2bf8s.256(<8 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtps2bf8s.512(<16 x float>)
+
+; Memory folding tests for vcvtps2bf8s
+define <16 x i8> @test_int_x86_avx10_vcvtps2bf8s_mem_128(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtps2bf8s_mem_128:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtps2bf8sx (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x3b,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtps2bf8s_mem_128:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtps2bf8sx (%eax), %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x3b,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <4 x float>, ptr %ptr_a
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8s.128(<4 x float> %a)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtps2bf8s_mem_256(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtps2bf8s_mem_256:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtps2bf8sy (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x3b,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtps2bf8s_mem_256:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtps2bf8sy (%eax), %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x3b,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <8 x float>, ptr %ptr_a
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8s.256(<8 x float> %a)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtps2bf8s_mem_512(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtps2bf8s_mem_512:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtps2bf8s (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x3b,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtps2bf8s_mem_512:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtps2bf8s (%eax), %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x3b,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <16 x float>, ptr %ptr_a
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8s.512(<16 x float> %a)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtps2hf8_128(<4 x float> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtps2hf8_128:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtps2hf8 %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x38,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8.128(<4 x float> %a)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtps2hf8_256(<8 x float> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtps2hf8_256:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtps2hf8 %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x38,0xc0]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8.256(<8 x float> %a)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtps2hf8_512(<16 x float> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtps2hf8_512:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtps2hf8 %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x38,0xc0]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8.512(<16 x float> %a)
+  ret <16 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.vcvtps2hf8.128(<4 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtps2hf8.256(<8 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtps2hf8.512(<16 x float>)
+
+; Memory folding tests for vcvtps2hf8
+define <16 x i8> @test_int_x86_avx10_vcvtps2hf8_mem_128(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtps2hf8_mem_128:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtps2hf8x (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x38,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtps2hf8_mem_128:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtps2hf8x (%eax), %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x38,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <4 x float>, ptr %ptr_a
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8.128(<4 x float> %a)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtps2hf8_mem_256(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtps2hf8_mem_256:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtps2hf8y (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x38,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtps2hf8_mem_256:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtps2hf8y (%eax), %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x38,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <8 x float>, ptr %ptr_a
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8.256(<8 x float> %a)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtps2hf8_mem_512(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtps2hf8_mem_512:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtps2hf8 (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x38,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtps2hf8_mem_512:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtps2hf8 (%eax), %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x38,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <16 x float>, ptr %ptr_a
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8.512(<16 x float> %a)
+  ret <16 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.vcvtps2hf8s.128(<4 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtps2hf8s.256(<8 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtps2hf8s.512(<16 x float>)
+
+define <16 x i8> @test_int_x86_avx10_vcvtps2hf8s_128(<4 x float> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtps2hf8s_128:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtps2hf8s %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x3a,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8s.128(<4 x float> %a)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtps2hf8s_256(<8 x float> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtps2hf8s_256:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtps2hf8s %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x3a,0xc0]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8s.256(<8 x float> %a)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtps2hf8s_512(<16 x float> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtps2hf8s_512:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtps2hf8s %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x3a,0xc0]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8s.512(<16 x float> %a)
+  ret <16 x i8> %ret
+}
+
+; Memory folding tests for vcvtps2hf8s
+define <16 x i8> @test_int_x86_avx10_vcvtps2hf8s_mem_128(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtps2hf8s_mem_128:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtps2hf8sx (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x3a,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtps2hf8s_mem_128:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtps2hf8sx (%eax), %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x3a,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <4 x float>, ptr %ptr_a
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8s.128(<4 x float> %a)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtps2hf8s_mem_256(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtps2hf8s_mem_256:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtps2hf8sy (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x3a,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtps2hf8s_mem_256:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtps2hf8sy (%eax), %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x3a,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <8 x float>, ptr %ptr_a
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8s.256(<8 x float> %a)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtps2hf8s_mem_512(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtps2hf8s_mem_512:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtps2hf8s (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x3a,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtps2hf8s_mem_512:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtps2hf8s (%eax), %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x3a,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <16 x float>, ptr %ptr_a
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8s.512(<16 x float> %a)
+  ret <16 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.vcvtrops2hf8.128(<4 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtrops2hf8.256(<8 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtrops2hf8.512(<16 x float>)
+
+define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8_128(<4 x float> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtrops2hf8_128:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtrops2hf8 %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7d,0x08,0x38,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8.128(<4 x float> %a)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8_256(<8 x float> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtrops2hf8_256:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtrops2hf8 %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7d,0x28,0x38,0xc0]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8.256(<8 x float> %a)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8_512(<16 x float> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtrops2hf8_512:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtrops2hf8 %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7d,0x48,0x38,0xc0]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8.512(<16 x float> %a)
+  ret <16 x i8> %ret
+}
+
+; Memory folding tests for vcvtrops2hf8
+define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8_mem_128(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtrops2hf8_mem_128:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtrops2hf8x (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7d,0x08,0x38,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtrops2hf8_mem_128:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtrops2hf8x (%eax), %xmm0 # encoding: [0x62,0xf5,0x7d,0x08,0x38,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <4 x float>, ptr %ptr_a
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8.128(<4 x float> %a)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8_mem_256(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtrops2hf8_mem_256:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtrops2hf8y (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7d,0x28,0x38,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtrops2hf8_mem_256:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtrops2hf8y (%eax), %xmm0 # encoding: [0x62,0xf5,0x7d,0x28,0x38,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <8 x float>, ptr %ptr_a
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8.256(<8 x float> %a)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8_mem_512(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtrops2hf8_mem_512:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtrops2hf8 (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7d,0x48,0x38,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtrops2hf8_mem_512:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtrops2hf8 (%eax), %xmm0 # encoding: [0x62,0xf5,0x7d,0x48,0x38,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <16 x float>, ptr %ptr_a
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8.512(<16 x float> %a)
+  ret <16 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s.128(<4 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s.256(<8 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s.512(<16 x float>)
+
+define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8s_128(<4 x float> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtrops2hf8s_128:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtrops2hf8s %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7d,0x08,0x3a,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s.128(<4 x float> %a)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8s_256(<8 x float> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtrops2hf8s_256:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtrops2hf8s %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7d,0x28,0x3a,0xc0]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s.256(<8 x float> %a)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8s_512(<16 x float> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtrops2hf8s_512:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtrops2hf8s %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7d,0x48,0x3a,0xc0]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s.512(<16 x float> %a)
+  ret <16 x i8> %ret
+}
+
+; Memory folding tests for vcvtrops2hf8s
+define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8s_mem_128(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtrops2hf8s_mem_128:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtrops2hf8sx (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7d,0x08,0x3a,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtrops2hf8s_mem_128:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtrops2hf8sx (%eax), %xmm0 # encoding: [0x62,0xf5,0x7d,0x08,0x3a,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <4 x float>, ptr %ptr_a
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s.128(<4 x float> %a)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8s_mem_256(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtrops2hf8s_mem_256:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtrops2hf8sy (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7d,0x28,0x3a,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtrops2hf8s_mem_256:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtrops2hf8sy (%eax), %xmm0 # encoding: [0x62,0xf5,0x7d,0x28,0x3a,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <8 x float>, ptr %ptr_a
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s.256(<8 x float> %a)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8s_mem_512(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtrops2hf8s_mem_512:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtrops2hf8s (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7d,0x48,0x3a,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtrops2hf8s_mem_512:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtrops2hf8s (%eax), %xmm0 # encoding: [0x62,0xf5,0x7d,0x48,0x3a,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <16 x float>, ptr %ptr_a
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s.512(<16 x float> %a)
+  ret <16 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8.128(<16 x i8>, <4 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8.256(<32 x i8>, <8 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8.512(<64 x i8>, <16 x float>)
+
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8_128(<16 x i8> %A, <4 x float> %b) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2bf8_128:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbiasps2bf8 %xmm1, %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x39,0xc1]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8.128(<16 x i8> %A, <4 x float> %b)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8_256(<32 x i8> %A, <8 x float> %b) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2bf8_256:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbiasps2bf8 %ymm1, %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x28,0x39,0xc1]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8.256(<32 x i8> %A, <8 x float> %b)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8_512(<64 x i8> %A, <16 x float> %b) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2bf8_512:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbiasps2bf8 %zmm1, %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x39,0xc1]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8.512(<64 x i8> %A, <16 x float> %b)
+  ret <16 x i8> %ret
+}
+
+; Memory folding tests for vcvtbiasps2bf8 (second operand from memory)
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8_mem_128(<16 x i8> %A, ptr %ptr_b) {
+; X64-LABEL: test_int_x86_avx10_vcvtbiasps2bf8_mem_128:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtbiasps2bf8 (%rdi), %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x39,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbiasps2bf8_mem_128:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2bf8 (%eax), %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x39,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %b = load <4 x float>, ptr %ptr_b
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8.128(<16 x i8> %A, <4 x float> %b)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8_mem_256(<32 x i8> %A, ptr %ptr_b) {
+; X64-LABEL: test_int_x86_avx10_vcvtbiasps2bf8_mem_256:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtbiasps2bf8 (%rdi), %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x28,0x39,0x07]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbiasps2bf8_mem_256:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2bf8 (%eax), %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x28,0x39,0x00]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %b = load <8 x float>, ptr %ptr_b
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8.256(<32 x i8> %A, <8 x float> %b)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8_mem_512(<64 x i8> %A, ptr %ptr_b) {
+; X64-LABEL: test_int_x86_avx10_vcvtbiasps2bf8_mem_512:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtbiasps2bf8 (%rdi), %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x39,0x07]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbiasps2bf8_mem_512:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2bf8 (%eax), %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x39,0x00]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %b = load <16 x float>, ptr %ptr_b
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8.512(<64 x i8> %A, <16 x float> %b)
+  ret <16 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s.128(<16 x i8>, <4 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s.256(<32 x i8>, <8 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s.512(<64 x i8>, <16 x float>)
+
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8s_128(<16 x i8> %A, <4 x float> %b) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2bf8s_128:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbiasps2bf8s %xmm1, %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x3b,0xc1]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s.128(<16 x i8> %A, <4 x float> %b)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8s_256(<32 x i8> %A, <8 x float> %b) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2bf8s_256:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbiasps2bf8s %ymm1, %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x28,0x3b,0xc1]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s.256(<32 x i8> %A, <8 x float> %b)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8s_512(<64 x i8> %A, <16 x float> %b) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2bf8s_512:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbiasps2bf8s %zmm1, %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x3b,0xc1]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s.512(<64 x i8> %A, <16 x float> %b)
+  ret <16 x i8> %ret
+}
+
+; Memory folding tests for vcvtbiasps2bf8s
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8s_mem_128(<16 x i8> %A, ptr %ptr_b) {
+; X64-LABEL: test_int_x86_avx10_vcvtbiasps2bf8s_mem_128:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtbiasps2bf8s (%rdi), %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x3b,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbiasps2bf8s_mem_128:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2bf8s (%eax), %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x3b,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %b = load <4 x float>, ptr %ptr_b
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s.128(<16 x i8> %A, <4 x float> %b)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8s_mem_256(<32 x i8> %A, ptr %ptr_b) {
+; X64-LABEL: test_int_x86_avx10_vcvtbiasps2bf8s_mem_256:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtbiasps2bf8s (%rdi), %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x28,0x3b,0x07]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbiasps2bf8s_mem_256:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2bf8s (%eax), %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x28,0x3b,0x00]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %b = load <8 x float>, ptr %ptr_b
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s.256(<32 x i8> %A, <8 x float> %b)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8s_mem_512(<64 x i8> %A, ptr %ptr_b) {
+; X64-LABEL: test_int_x86_avx10_vcvtbiasps2bf8s_mem_512:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtbiasps2bf8s (%rdi), %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x3b,0x07]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbiasps2bf8s_mem_512:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2bf8s (%eax), %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x3b,0x00]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %b = load <16 x float>, ptr %ptr_b
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s.512(<64 x i8> %A, <16 x float> %b)
+  ret <16 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8.128(<16 x i8>, <4 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8.256(<32 x i8>, <8 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8.512(<64 x i8>, <16 x float>)
+
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8_128(<16 x i8> %A, <4 x float> %b) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2hf8_128:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbiasps2hf8 %xmm1, %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x38,0xc1]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8.128(<16 x i8> %A, <4 x float> %b)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8_256(<32 x i8> %A, <8 x float> %b) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2hf8_256:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbiasps2hf8 %ymm1, %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x28,0x38,0xc1]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8.256(<32 x i8> %A, <8 x float> %b)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8_512(<64 x i8> %A, <16 x float> %b) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2hf8_512:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbiasps2hf8 %zmm1, %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x38,0xc1]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8.512(<64 x i8> %A, <16 x float> %b)
+  ret <16 x i8> %ret
+}
+
+; Memory folding tests for vcvtbiasps2hf8
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8_mem_128(<16 x i8> %A, ptr %ptr_b) {
+; X64-LABEL: test_int_x86_avx10_vcvtbiasps2hf8_mem_128:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtbiasps2hf8 (%rdi), %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x38,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbiasps2hf8_mem_128:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2hf8 (%eax), %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x38,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %b = load <4 x float>, ptr %ptr_b
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8.128(<16 x i8> %A, <4 x float> %b)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8_mem_256(<32 x i8> %A, ptr %ptr_b) {
+; X64-LABEL: test_int_x86_avx10_vcvtbiasps2hf8_mem_256:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtbiasps2hf8 (%rdi), %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x28,0x38,0x07]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbiasps2hf8_mem_256:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2hf8 (%eax), %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x28,0x38,0x00]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %b = load <8 x float>, ptr %ptr_b
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8.256(<32 x i8> %A, <8 x float> %b)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8_mem_512(<64 x i8> %A, ptr %ptr_b) {
+; X64-LABEL: test_int_x86_avx10_vcvtbiasps2hf8_mem_512:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtbiasps2hf8 (%rdi), %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x38,0x07]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbiasps2hf8_mem_512:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2hf8 (%eax), %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x38,0x00]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %b = load <16 x float>, ptr %ptr_b
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8.512(<64 x i8> %A, <16 x float> %b)
+  ret <16 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s.128(<16 x i8>, <4 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s.256(<32 x i8>, <8 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s.512(<64 x i8>, <16 x float>)
+
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8s_128(<16 x i8> %A, <4 x float> %b) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2hf8s_128:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbiasps2hf8s %xmm1, %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x3a,0xc1]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s.128(<16 x i8> %A, <4 x float> %b)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8s_256(<32 x i8> %A, <8 x float> %b) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2hf8s_256:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbiasps2hf8s %ymm1, %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x28,0x3a,0xc1]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s.256(<32 x i8> %A, <8 x float> %b)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8s_512(<64 x i8> %A, <16 x float> %b) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2hf8s_512:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbiasps2hf8s %zmm1, %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x3a,0xc1]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s.512(<64 x i8> %A, <16 x float> %b)
+  ret <16 x i8> %ret
+}
+
+; Memory folding tests for vcvtbiasps2hf8s
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8s_mem_128(<16 x i8> %A, ptr %ptr_b) {
+; X64-LABEL: test_int_x86_avx10_vcvtbiasps2hf8s_mem_128:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtbiasps2hf8s (%rdi), %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x3a,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbiasps2hf8s_mem_128:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2hf8s (%eax), %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x3a,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %b = load <4 x float>, ptr %ptr_b
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s.128(<16 x i8> %A, <4 x float> %b)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8s_mem_256(<32 x i8> %A, ptr %ptr_b) {
+; X64-LABEL: test_int_x86_avx10_vcvtbiasps2hf8s_mem_256:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtbiasps2hf8s (%rdi), %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x28,0x3a,0x07]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbiasps2hf8s_mem_256:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2hf8s (%eax), %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x28,0x3a,0x00]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %b = load <8 x float>, ptr %ptr_b
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s.256(<32 x i8> %A, <8 x float> %b)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8s_mem_512(<64 x i8> %A, ptr %ptr_b) {
+; X64-LABEL: test_int_x86_avx10_vcvtbiasps2hf8s_mem_512:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtbiasps2hf8s (%rdi), %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x3a,0x07]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbiasps2hf8s_mem_512:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2hf8s (%eax), %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x3a,0x00]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %b = load <16 x float>, ptr %ptr_b
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s.512(<64 x i8> %A, <16 x float> %b)
+  ret <16 x i8> %ret
+}
+
+declare <4 x float> @llvm.x86.avx10.vcvtbf82ps.128(<16 x i8>)
+declare <8 x float> @llvm.x86.avx10.vcvtbf82ps.256(<16 x i8>)
+declare <16 x float> @llvm.x86.avx10.vcvtbf82ps.512(<16 x i8>)
+
+define <4 x float> @test_int_x86_avx10_vcvtbf82ps_128(<16 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbf82ps_128:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbf82ps %xmm0, %xmm0 # encoding: [0x62,0xf5,0xfc,0x08,0x36,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <4 x float> @llvm.x86.avx10.vcvtbf82ps.128(<16 x i8> %a)
+  ret <4 x float> %ret
+}
+
+define <8 x float> @test_int_x86_avx10_vcvtbf82ps_256(<16 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbf82ps_256:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbf82ps %xmm0, %ymm0 # encoding: [0x62,0xf5,0xfc,0x28,0x36,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <8 x float> @llvm.x86.avx10.vcvtbf82ps.256(<16 x i8> %a)
+  ret <8 x float> %ret
+}
+
+define <16 x float> @test_int_x86_avx10_vcvtbf82ps_512(<16 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbf82ps_512:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbf82ps %xmm0, %zmm0 # encoding: [0x62,0xf5,0xfc,0x48,0x36,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x float> @llvm.x86.avx10.vcvtbf82ps.512(<16 x i8> %a)
+  ret <16 x float> %ret
+}
+
+; Memory folding tests for vcvtbf82ps
+define <4 x float> @test_int_x86_avx10_vcvtbf82ps_mem_128(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf82ps_mem_128:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtbf82ps (%rdi), %xmm0 # encoding: [0x62,0xf5,0xfc,0x08,0x36,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbf82ps_mem_128:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtbf82ps (%eax), %xmm0 # encoding: [0x62,0xf5,0xfc,0x08,0x36,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <16 x i8>, ptr %ptr_a
+  %ret = call <4 x float> @llvm.x86.avx10.vcvtbf82ps.128(<16 x i8> %a)
+  ret <4 x float> %ret
+}
+
+define <8 x float> @test_int_x86_avx10_vcvtbf82ps_mem_256(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf82ps_mem_256:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtbf82ps (%rdi), %ymm0 # encoding: [0x62,0xf5,0xfc,0x28,0x36,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbf82ps_mem_256:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtbf82ps (%eax), %ymm0 # encoding: [0x62,0xf5,0xfc,0x28,0x36,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <16 x i8>, ptr %ptr_a
+  %ret = call <8 x float> @llvm.x86.avx10.vcvtbf82ps.256(<16 x i8> %a)
+  ret <8 x float> %ret
+}
+
+define <16 x float> @test_int_x86_avx10_vcvtbf82ps_mem_512(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf82ps_mem_512:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtbf82ps (%rdi), %zmm0 # encoding: [0x62,0xf5,0xfc,0x48,0x36,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbf82ps_mem_512:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtbf82ps (%eax), %zmm0 # encoding: [0x62,0xf5,0xfc,0x48,0x36,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <16 x i8>, ptr %ptr_a
+  %ret = call <16 x float> @llvm.x86.avx10.vcvtbf82ps.512(<16 x i8> %a)
+  ret <16 x float> %ret
+}
+
+declare <4 x float> @llvm.x86.avx10.vcvthf82ps.128(<16 x i8>)
+declare <8 x float> @llvm.x86.avx10.vcvthf82ps.256(<16 x i8>)
+declare <16 x float> @llvm.x86.avx10.vcvthf82ps.512(<16 x i8>)
+
+define <4 x float> @test_int_x86_avx10_vcvthf82ps_128(<16 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvthf82ps_128:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvthf82ps %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x36,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <4 x float> @llvm.x86.avx10.vcvthf82ps.128(<16 x i8> %a)
+  ret <4 x float> %ret
+}
+
+define <8 x float> @test_int_x86_avx10_vcvthf82ps_256(<16 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvthf82ps_256:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvthf82ps %xmm0, %ymm0 # encoding: [0x62,0xf5,0x7c,0x28,0x36,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <8 x float> @llvm.x86.avx10.vcvthf82ps.256(<16 x i8> %a)
+  ret <8 x float> %ret
+}
+
+define <16 x float> @test_int_x86_avx10_vcvthf82ps_512(<16 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvthf82ps_512:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvthf82ps %xmm0, %zmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x36,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x float> @llvm.x86.avx10.vcvthf82ps.512(<16 x i8> %a)
+  ret <16 x float> %ret
+}
+
+; Memory folding tests for vcvthf82ps
+define <4 x float> @test_int_x86_avx10_vcvthf82ps_mem_128(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvthf82ps_mem_128:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvthf82ps (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x36,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvthf82ps_mem_128:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvthf82ps (%eax), %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x36,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <16 x i8>, ptr %ptr_a
+  %ret = call <4 x float> @llvm.x86.avx10.vcvthf82ps.128(<16 x i8> %a)
+  ret <4 x float> %ret
+}
+
+define <8 x float> @test_int_x86_avx10_vcvthf82ps_mem_256(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvthf82ps_mem_256:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvthf82ps (%rdi), %ymm0 # encoding: [0x62,0xf5,0x7c,0x28,0x36,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvthf82ps_mem_256:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvthf82ps (%eax), %ymm0 # encoding: [0x62,0xf5,0x7c,0x28,0x36,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <16 x i8>, ptr %ptr_a
+  %ret = call <8 x float> @llvm.x86.avx10.vcvthf82ps.256(<16 x i8> %a)
+  ret <8 x float> %ret
+}
+
+define <16 x float> @test_int_x86_avx10_vcvthf82ps_mem_512(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvthf82ps_mem_512:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvthf82ps (%rdi), %zmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x36,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvthf82ps_mem_512:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvthf82ps (%eax), %zmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x36,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <16 x i8>, ptr %ptr_a
+  %ret = call <16 x float> @llvm.x86.avx10.vcvthf82ps.512(<16 x i8> %a)
+  ret <16 x float> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtbf82bf4s_128(<16 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbf82bf4s_128:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbf82bf4s %xmm0, %xmm0 # encoding: [0x62,0xf5,0xfe,0x08,0x3d,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf82bf4s.128(<16 x i8> %a)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtbf82bf4s_256(<32 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbf82bf4s_256:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbf82bf4s %ymm0, %xmm0 # encoding: [0x62,0xf5,0xfe,0x28,0x3d,0xc0]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf82bf4s.256(<32 x i8> %a)
+  ret <16 x i8> %ret
+}
+
+define <32 x i8> @test_int_x86_avx10_vcvtbf82bf4s_512(<64 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbf82bf4s_512:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbf82bf4s %zmm0, %ymm0 # encoding: [0x62,0xf5,0xfe,0x48,0x3d,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <32 x i8> @llvm.x86.avx10.vcvtbf82bf4s.512(<64 x i8> %a)
+  ret <32 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.vcvtbf82bf4s.128(<16 x i8>)
+declare <16 x i8> @llvm.x86.avx10.vcvtbf82bf4s.256(<32 x i8>)
+declare <32 x i8> @llvm.x86.avx10.vcvtbf82bf4s.512(<64 x i8>)
+
+; Memory folding tests for vcvtbf82bf4s
+define <16 x i8> @test_int_x86_avx10_vcvtbf82bf4s_mem_128(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf82bf4s_mem_128:
+; X64:       # %bb.0:
+; X64-NEXT:    vmovaps (%rdi), %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0x07]
+; X64-NEXT:    vcvtbf82bf4s %xmm0, %xmm0 # encoding: [0x62,0xf5,0xfe,0x08,0x3d,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbf82bf4s_mem_128:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vmovaps (%eax), %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0x00]
+; X86-NEXT:    vcvtbf82bf4s %xmm0, %xmm0 # encoding: [0x62,0xf5,0xfe,0x08,0x3d,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <16 x i8>, ptr %ptr_a
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf82bf4s.128(<16 x i8> %a)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtbf82bf4s_mem_256(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf82bf4s_mem_256:
+; X64:       # %bb.0:
+; X64-NEXT:    vmovaps (%rdi), %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfc,0x28,0x07]
+; X64-NEXT:    vcvtbf82bf4s %ymm0, %xmm0 # encoding: [0x62,0xf5,0xfe,0x28,0x3d,0xc0]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbf82bf4s_mem_256:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vmovaps (%eax), %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfc,0x28,0x00]
+; X86-NEXT:    vcvtbf82bf4s %ymm0, %xmm0 # encoding: [0x62,0xf5,0xfe,0x28,0x3d,0xc0]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <32 x i8>, ptr %ptr_a
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf82bf4s.256(<32 x i8> %a)
+  ret <16 x i8> %ret
+}
+
+define <32 x i8> @test_int_x86_avx10_vcvtbf82bf4s_mem_512(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf82bf4s_mem_512:
+; X64:       # %bb.0:
+; X64-NEXT:    vmovaps (%rdi), %zmm0 # encoding: [0x62,0xf1,0x7c,0x48,0x28,0x07]
+; X64-NEXT:    vcvtbf82bf4s %zmm0, %ymm0 # encoding: [0x62,0xf5,0xfe,0x48,0x3d,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbf82bf4s_mem_512:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vmovaps (%eax), %zmm0 # encoding: [0x62,0xf1,0x7c,0x48,0x28,0x00]
+; X86-NEXT:    vcvtbf82bf4s %zmm0, %ymm0 # encoding: [0x62,0xf5,0xfe,0x48,0x3d,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <64 x i8>, ptr %ptr_a
+  %ret = call <32 x i8> @llvm.x86.avx10.vcvtbf82bf4s.512(<64 x i8> %a)
+  ret <32 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvthf82bf4s_128(<16 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvthf82bf4s_128:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvthf82bf4s %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x3d,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvthf82bf4s.128(<16 x i8> %a)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvthf82bf4s_256(<32 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvthf82bf4s_256:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvthf82bf4s %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x3d,0xc0]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvthf82bf4s.256(<32 x i8> %a)
+  ret <16 x i8> %ret
+}
+
+define <32 x i8> @test_int_x86_avx10_vcvthf82bf4s_512(<64 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvthf82bf4s_512:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvthf82bf4s %zmm0, %ymm0 # encoding: [0x62,0xf5,0x7e,0x48,0x3d,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <32 x i8> @llvm.x86.avx10.vcvthf82bf4s.512(<64 x i8> %a)
+  ret <32 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.vcvthf82bf4s.128(<16 x i8>)
+declare <16 x i8> @llvm.x86.avx10.vcvthf82bf4s.256(<32 x i8>)
+declare <32 x i8> @llvm.x86.avx10.vcvthf82bf4s.512(<64 x i8>)
+
+; Memory folding tests for vcvthf82bf4s
+define <16 x i8> @test_int_x86_avx10_vcvthf82bf4s_mem_128(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvthf82bf4s_mem_128:
+; X64:       # %bb.0:
+; X64-NEXT:    vmovaps (%rdi), %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0x07]
+; X64-NEXT:    vcvthf82bf4s %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x3d,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvthf82bf4s_mem_128:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vmovaps (%eax), %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0x00]
+; X86-NEXT:    vcvthf82bf4s %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x3d,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <16 x i8>, ptr %ptr_a
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvthf82bf4s.128(<16 x i8> %a)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvthf82bf4s_mem_256(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvthf82bf4s_mem_256:
+; X64:       # %bb.0:
+; X64-NEXT:    vmovaps (%rdi), %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfc,0x28,0x07]
+; X64-NEXT:    vcvthf82bf4s %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x3d,0xc0]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvthf82bf4s_mem_256:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vmovaps (%eax), %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfc,0x28,0x00]
+; X86-NEXT:    vcvthf82bf4s %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x3d,0xc0]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <32 x i8>, ptr %ptr_a
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvthf82bf4s.256(<32 x i8> %a)
+  ret <16 x i8> %ret
+}
+
+define <32 x i8> @test_int_x86_avx10_vcvthf82bf4s_mem_512(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvthf82bf4s_mem_512:
+; X64:       # %bb.0:
+; X64-NEXT:    vmovaps (%rdi), %zmm0 # encoding: [0x62,0xf1,0x7c,0x48,0x28,0x07]
+; X64-NEXT:    vcvthf82bf4s %zmm0, %ymm0 # encoding: [0x62,0xf5,0x7e,0x48,0x3d,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvthf82bf4s_mem_512:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vmovaps (%eax), %zmm0 # encoding: [0x62,0xf1,0x7c,0x48,0x28,0x00]
+; X86-NEXT:    vcvthf82bf4s %zmm0, %ymm0 # encoding: [0x62,0xf5,0x7e,0x48,0x3d,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <64 x i8>, ptr %ptr_a
+  %ret = call <32 x i8> @llvm.x86.avx10.vcvthf82bf4s.512(<64 x i8> %a)
+  ret <32 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtbf82bf6s_128(<16 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbf82bf6s_128:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbf82bf6s %xmm0, %xmm0 # encoding: [0x62,0xf5,0xfe,0x08,0x3e,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf82bf6s.128(<16 x i8> %a)
+  ret <16 x i8> %ret
+}
+
+define <32 x i8> @test_int_x86_avx10_vcvtbf82bf6s_256(<32 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbf82bf6s_256:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbf82bf6s %ymm0, %ymm0 # encoding: [0x62,0xf5,0xfe,0x28,0x3e,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <32 x i8> @llvm.x86.avx10.vcvtbf82bf6s.256(<32 x i8> %a)
+  ret <32 x i8> %ret
+}
+
+define <64 x i8> @test_int_x86_avx10_vcvtbf82bf6s_512(<64 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbf82bf6s_512:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbf82bf6s %zmm0, %zmm0 # encoding: [0x62,0xf5,0xfe,0x48,0x3e,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <64 x i8> @llvm.x86.avx10.vcvtbf82bf6s.512(<64 x i8> %a)
+  ret <64 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.vcvtbf82bf6s.128(<16 x i8>)
+declare <32 x i8> @llvm.x86.avx10.vcvtbf82bf6s.256(<32 x i8>)
+declare <64 x i8> @llvm.x86.avx10.vcvtbf82bf6s.512(<64 x i8>)
+
+; Memory folding tests for vcvtbf82bf6s
+define <16 x i8> @test_int_x86_avx10_vcvtbf82bf6s_mem_128(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf82bf6s_mem_128:
+; X64:       # %bb.0:
+; X64-NEXT:    vmovaps (%rdi), %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0x07]
+; X64-NEXT:    vcvtbf82bf6s %xmm0, %xmm0 # encoding: [0x62,0xf5,0xfe,0x08,0x3e,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbf82bf6s_mem_128:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vmovaps (%eax), %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0x00]
+; X86-NEXT:    vcvtbf82bf6s %xmm0, %xmm0 # encoding: [0x62,0xf5,0xfe,0x08,0x3e,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <16 x i8>, ptr %ptr_a
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf82bf6s.128(<16 x i8> %a)
+  ret <16 x i8> %ret
+}
+
+define <32 x i8> @test_int_x86_avx10_vcvtbf82bf6s_mem_256(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf82bf6s_mem_256:
+; X64:       # %bb.0:
+; X64-NEXT:    vmovaps (%rdi), %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfc,0x28,0x07]
+; X64-NEXT:    vcvtbf82bf6s %ymm0, %ymm0 # encoding: [0x62,0xf5,0xfe,0x28,0x3e,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbf82bf6s_mem_256:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vmovaps (%eax), %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfc,0x28,0x00]
+; X86-NEXT:    vcvtbf82bf6s %ymm0, %ymm0 # encoding: [0x62,0xf5,0xfe,0x28,0x3e,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <32 x i8>, ptr %ptr_a
+  %ret = call <32 x i8> @llvm.x86.avx10.vcvtbf82bf6s.256(<32 x i8> %a)
+  ret <32 x i8> %ret
+}
+
+define <64 x i8> @test_int_x86_avx10_vcvtbf82bf6s_mem_512(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf82bf6s_mem_512:
+; X64:       # %bb.0:
+; X64-NEXT:    vmovaps (%rdi), %zmm0 # encoding: [0x62,0xf1,0x7c,0x48,0x28,0x07]
+; X64-NEXT:    vcvtbf82bf6s %zmm0, %zmm0 # encoding: [0x62,0xf5,0xfe,0x48,0x3e,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbf82bf6s_mem_512:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vmovaps (%eax), %zmm0 # encoding: [0x62,0xf1,0x7c,0x48,0x28,0x00]
+; X86-NEXT:    vcvtbf82bf6s %zmm0, %zmm0 # encoding: [0x62,0xf5,0xfe,0x48,0x3e,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <64 x i8>, ptr %ptr_a
+  %ret = call <64 x i8> @llvm.x86.avx10.vcvtbf82bf6s.512(<64 x i8> %a)
+  ret <64 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvthf82hf6s_128(<16 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvthf82hf6s_128:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvthf82hf6s %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x3c,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvthf82hf6s.128(<16 x i8> %a)
+  ret <16 x i8> %ret
+}
+
+define <32 x i8> @test_int_x86_avx10_vcvthf82hf6s_256(<32 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvthf82hf6s_256:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvthf82hf6s %ymm0, %ymm0 # encoding: [0x62,0xf5,0x7e,0x28,0x3c,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <32 x i8> @llvm.x86.avx10.vcvthf82hf6s.256(<32 x i8> %a)
+  ret <32 x i8> %ret
+}
+
+define <64 x i8> @test_int_x86_avx10_vcvthf82hf6s_512(<64 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvthf82hf6s_512:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvthf82hf6s %zmm0, %zmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x3c,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <64 x i8> @llvm.x86.avx10.vcvthf82hf6s.512(<64 x i8> %a)
+  ret <64 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.vcvthf82hf6s.128(<16 x i8>)
+declare <32 x i8> @llvm.x86.avx10.vcvthf82hf6s.256(<32 x i8>)
+declare <64 x i8> @llvm.x86.avx10.vcvthf82hf6s.512(<64 x i8>)
+
+; Memory folding tests for vcvthf82hf6s
+define <16 x i8> @test_int_x86_avx10_vcvthf82hf6s_mem_128(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvthf82hf6s_mem_128:
+; X64:       # %bb.0:
+; X64-NEXT:    vmovaps (%rdi), %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0x07]
+; X64-NEXT:    vcvthf82hf6s %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x3c,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvthf82hf6s_mem_128:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vmovaps (%eax), %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0x00]
+; X86-NEXT:    vcvthf82hf6s %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x3c,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <16 x i8>, ptr %ptr_a
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvthf82hf6s.128(<16 x i8> %a)
+  ret <16 x i8> %ret
+}
+
+define <32 x i8> @test_int_x86_avx10_vcvthf82hf6s_mem_256(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvthf82hf6s_mem_256:
+; X64:       # %bb.0:
+; X64-NEXT:    vmovaps (%rdi), %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfc,0x28,0x07]
+; X64-NEXT:    vcvthf82hf6s %ymm0, %ymm0 # encoding: [0x62,0xf5,0x7e,0x28,0x3c,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvthf82hf6s_mem_256:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vmovaps (%eax), %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfc,0x28,0x00]
+; X86-NEXT:    vcvthf82hf6s %ymm0, %ymm0 # encoding: [0x62,0xf5,0x7e,0x28,0x3c,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <32 x i8>, ptr %ptr_a
+  %ret = call <32 x i8> @llvm.x86.avx10.vcvthf82hf6s.256(<32 x i8> %a)
+  ret <32 x i8> %ret
+}
+
+define <64 x i8> @test_int_x86_avx10_vcvthf82hf6s_mem_512(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvthf82hf6s_mem_512:
+; X64:       # %bb.0:
+; X64-NEXT:    vmovaps (%rdi), %zmm0 # encoding: [0x62,0xf1,0x7c,0x48,0x28,0x07]
+; X64-NEXT:    vcvthf82hf6s %zmm0, %zmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x3c,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvthf82hf6s_mem_512:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vmovaps (%eax), %zmm0 # encoding: [0x62,0xf1,0x7c,0x48,0x28,0x00]
+; X86-NEXT:    vcvthf82hf6s %zmm0, %zmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x3c,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <64 x i8>, ptr %ptr_a
+  %ret = call <64 x i8> @llvm.x86.avx10.vcvthf82hf6s.512(<64 x i8> %a)
+  ret <64 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtbf42hf8_128(<16 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbf42hf8_128:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbf42hf8 %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x37,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf42hf8.128(<16 x i8> %a)
+  ret <16 x i8> %ret
+}
+
+define <32 x i8> @test_int_x86_avx10_vcvtbf42hf8_256(<16 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbf42hf8_256:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbf42hf8 %xmm0, %ymm0 # encoding: [0x62,0xf5,0x7c,0x28,0x37,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <32 x i8> @llvm.x86.avx10.vcvtbf42hf8.256(<16 x i8> %a)
+  ret <32 x i8> %ret
+}
+
+define <64 x i8> @test_int_x86_avx10_vcvtbf42hf8_512(<32 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbf42hf8_512:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbf42hf8 %ymm0, %zmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x37,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <64 x i8> @llvm.x86.avx10.vcvtbf42hf8.512(<32 x i8> %a)
+  ret <64 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.vcvtbf42hf8.128(<16 x i8>)
+declare <32 x i8> @llvm.x86.avx10.vcvtbf42hf8.256(<16 x i8>)
+declare <64 x i8> @llvm.x86.avx10.vcvtbf42hf8.512(<32 x i8>)
+
+; Memory folding tests for vcvtbf42hf8
+define <16 x i8> @test_int_x86_avx10_vcvtbf42hf8_mem_128(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf42hf8_mem_128:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtbf42hf8 (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x37,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbf42hf8_mem_128:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtbf42hf8 (%eax), %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x37,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <16 x i8>, ptr %ptr_a
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf42hf8.128(<16 x i8> %a)
+  ret <16 x i8> %ret
+}
+
+define <32 x i8> @test_int_x86_avx10_vcvtbf42hf8_mem_256(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf42hf8_mem_256:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtbf42hf8 (%rdi), %ymm0 # encoding: [0x62,0xf5,0x7c,0x28,0x37,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbf42hf8_mem_256:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtbf42hf8 (%eax), %ymm0 # encoding: [0x62,0xf5,0x7c,0x28,0x37,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <16 x i8>, ptr %ptr_a
+  %ret = call <32 x i8> @llvm.x86.avx10.vcvtbf42hf8.256(<16 x i8> %a)
+  ret <32 x i8> %ret
+}
+
+define <64 x i8> @test_int_x86_avx10_vcvtbf42hf8_mem_512(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf42hf8_mem_512:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtbf42hf8 (%rdi), %zmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x37,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbf42hf8_mem_512:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtbf42hf8 (%eax), %zmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x37,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <32 x i8>, ptr %ptr_a
+  %ret = call <64 x i8> @llvm.x86.avx10.vcvtbf42hf8.512(<32 x i8> %a)
+  ret <64 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtbf62hf8_128(<16 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbf62hf8_128:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbf62hf8 %xmm0, %xmm0 # encoding: [0x62,0xf5,0xfd,0x08,0x37,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf62hf8.128(<16 x i8> %a)
+  ret <16 x i8> %ret
+}
+
+define <32 x i8> @test_int_x86_avx10_vcvtbf62hf8_256(<32 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbf62hf8_256:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbf62hf8 %ymm0, %ymm0 # encoding: [0x62,0xf5,0xfd,0x28,0x37,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <32 x i8> @llvm.x86.avx10.vcvtbf62hf8.256(<32 x i8> %a)
+  ret <32 x i8> %ret
+}
+
+define <64 x i8> @test_int_x86_avx10_vcvtbf62hf8_512(<64 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbf62hf8_512:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbf62hf8 %zmm0, %zmm0 # encoding: [0x62,0xf5,0xfd,0x48,0x37,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <64 x i8> @llvm.x86.avx10.vcvtbf62hf8.512(<64 x i8> %a)
+  ret <64 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.vcvtbf62hf8.128(<16 x i8>)
+declare <32 x i8> @llvm.x86.avx10.vcvtbf62hf8.256(<32 x i8>)
+declare <64 x i8> @llvm.x86.avx10.vcvtbf62hf8.512(<64 x i8>)
+
+; Memory folding tests for vcvtbf62hf8
+define <16 x i8> @test_int_x86_avx10_vcvtbf62hf8_mem_128(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf62hf8_mem_128:
+; X64:       # %bb.0:
+; X64-NEXT:    vmovdqa (%rdi), %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf9,0x6f,0x07]
+; X64-NEXT:    vcvtbf62hf8 %xmm0, %xmm0 # encoding: [0x62,0xf5,0xfd,0x08,0x37,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbf62hf8_mem_128:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vmovdqa (%eax), %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf9,0x6f,0x00]
+; X86-NEXT:    vcvtbf62hf8 %xmm0, %xmm0 # encoding: [0x62,0xf5,0xfd,0x08,0x37,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <16 x i8>, ptr %ptr_a
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf62hf8.128(<16 x i8> %a)
+  ret <16 x i8> %ret
+}
+
+define <32 x i8> @test_int_x86_avx10_vcvtbf62hf8_mem_256(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf62hf8_mem_256:
+; X64:       # %bb.0:
+; X64-NEXT:    vmovdqa (%rdi), %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfd,0x6f,0x07]
+; X64-NEXT:    vcvtbf62hf8 %ymm0, %ymm0 # encoding: [0x62,0xf5,0xfd,0x28,0x37,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbf62hf8_mem_256:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vmovdqa (%eax), %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfd,0x6f,0x00]
+; X86-NEXT:    vcvtbf62hf8 %ymm0, %ymm0 # encoding: [0x62,0xf5,0xfd,0x28,0x37,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <32 x i8>, ptr %ptr_a
+  %ret = call <32 x i8> @llvm.x86.avx10.vcvtbf62hf8.256(<32 x i8> %a)
+  ret <32 x i8> %ret
+}
+
+define <64 x i8> @test_int_x86_avx10_vcvtbf62hf8_mem_512(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf62hf8_mem_512:
+; X64:       # %bb.0:
+; X64-NEXT:    vmovdqa64 (%rdi), %zmm0 # encoding: [0x62,0xf1,0xfd,0x48,0x6f,0x07]
+; X64-NEXT:    vcvtbf62hf8 %zmm0, %zmm0 # encoding: [0x62,0xf5,0xfd,0x48,0x37,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbf62hf8_mem_512:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vmovdqa64 (%eax), %zmm0 # encoding: [0x62,0xf1,0xfd,0x48,0x6f,0x00]
+; X86-NEXT:    vcvtbf62hf8 %zmm0, %zmm0 # encoding: [0x62,0xf5,0xfd,0x48,0x37,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <64 x i8>, ptr %ptr_a
+  %ret = call <64 x i8> @llvm.x86.avx10.vcvtbf62hf8.512(<64 x i8> %a)
+  ret <64 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvthf62hf8_128(<16 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvthf62hf8_128:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvthf62hf8 %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7d,0x08,0x37,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvthf62hf8.128(<16 x i8> %a)
+  ret <16 x i8> %ret
+}
+
+define <32 x i8> @test_int_x86_avx10_vcvthf62hf8_256(<32 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvthf62hf8_256:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvthf62hf8 %ymm0, %ymm0 # encoding: [0x62,0xf5,0x7d,0x28,0x37,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <32 x i8> @llvm.x86.avx10.vcvthf62hf8.256(<32 x i8> %a)
+  ret <32 x i8> %ret
+}
+
+define <64 x i8> @test_int_x86_avx10_vcvthf62hf8_512(<64 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvthf62hf8_512:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvthf62hf8 %zmm0, %zmm0 # encoding: [0x62,0xf5,0x7d,0x48,0x37,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <64 x i8> @llvm.x86.avx10.vcvthf62hf8.512(<64 x i8> %a)
+  ret <64 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.vcvthf62hf8.128(<16 x i8>)
+declare <32 x i8> @llvm.x86.avx10.vcvthf62hf8.256(<32 x i8>)
+declare <64 x i8> @llvm.x86.avx10.vcvthf62hf8.512(<64 x i8>)
+
+; Memory folding tests for vcvthf62hf8
+define <16 x i8> @test_int_x86_avx10_vcvthf62hf8_mem_128(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvthf62hf8_mem_128:
+; X64:       # %bb.0:
+; X64-NEXT:    vmovdqa (%rdi), %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf9,0x6f,0x07]
+; X64-NEXT:    vcvthf62hf8 %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7d,0x08,0x37,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvthf62hf8_mem_128:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vmovdqa (%eax), %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf9,0x6f,0x00]
+; X86-NEXT:    vcvthf62hf8 %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7d,0x08,0x37,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <16 x i8>, ptr %ptr_a
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvthf62hf8.128(<16 x i8> %a)
+  ret <16 x i8> %ret
+}
+
+define <32 x i8> @test_int_x86_avx10_vcvthf62hf8_mem_256(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvthf62hf8_mem_256:
+; X64:       # %bb.0:
+; X64-NEXT:    vmovdqa (%rdi), %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfd,0x6f,0x07]
+; X64-NEXT:    vcvthf62hf8 %ymm0, %ymm0 # encoding: [0x62,0xf5,0x7d,0x28,0x37,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvthf62hf8_mem_256:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vmovdqa (%eax), %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfd,0x6f,0x00]
+; X86-NEXT:    vcvthf62hf8 %ymm0, %ymm0 # encoding: [0x62,0xf5,0x7d,0x28,0x37,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <32 x i8>, ptr %ptr_a
+  %ret = call <32 x i8> @llvm.x86.avx10.vcvthf62hf8.256(<32 x i8> %a)
+  ret <32 x i8> %ret
+}
+
+define <64 x i8> @test_int_x86_avx10_vcvthf62hf8_mem_512(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvthf62hf8_mem_512:
+; X64:       # %bb.0:
+; X64-NEXT:    vmovdqa64 (%rdi), %zmm0 # encoding: [0x62,0xf1,0xfd,0x48,0x6f,0x07]
+; X64-NEXT:    vcvthf62hf8 %zmm0, %zmm0 # encoding: [0x62,0xf5,0x7d,0x48,0x37,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvthf62hf8_mem_512:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vmovdqa64 (%eax), %zmm0 # encoding: [0x62,0xf1,0xfd,0x48,0x6f,0x00]
+; X86-NEXT:    vcvthf62hf8 %zmm0, %zmm0 # encoding: [0x62,0xf5,0x7d,0x48,0x37,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <64 x i8>, ptr %ptr_a
+  %ret = call <64 x i8> @llvm.x86.avx10.vcvthf62hf8.512(<64 x i8> %a)
+  ret <64 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vunpackb_128(<16 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vunpackb_128:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vunpackb $1, %xmm0, %xmm0 # encoding: [0x62,0xf3,0x7c,0x08,0x3d,0xc0,0x01]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vunpackb.128(<16 x i8> %a, i8 1)
+  ret <16 x i8> %ret
+}
+
+define <32 x i8> @test_int_x86_avx10_vunpackb_256(<32 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vunpackb_256:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vunpackb $2, %ymm0, %ymm0 # encoding: [0x62,0xf3,0x7c,0x28,0x3d,0xc0,0x02]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <32 x i8> @llvm.x86.avx10.vunpackb.256(<32 x i8> %a, i8 2)
+  ret <32 x i8> %ret
+}
+
+define <64 x i8> @test_int_x86_avx10_vunpackb_512(<64 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vunpackb_512:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vunpackb $3, %zmm0, %zmm0 # encoding: [0x62,0xf3,0x7c,0x48,0x3d,0xc0,0x03]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <64 x i8> @llvm.x86.avx10.vunpackb.512(<64 x i8> %a, i8 3)
+  ret <64 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.vunpackb.128(<16 x i8>, i8)
+declare <32 x i8> @llvm.x86.avx10.vunpackb.256(<32 x i8>, i8)
+declare <64 x i8> @llvm.x86.avx10.vunpackb.512(<64 x i8>, i8)
+
+; Memory folding tests for vunpackb
+define <16 x i8> @test_int_x86_avx10_vunpackb_mem_128(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vunpackb_mem_128:
+; X64:       # %bb.0:
+; X64-NEXT:    vunpackb $1, (%rdi), %xmm0 # encoding: [0x62,0xf3,0x7c,0x08,0x3d,0x07,0x01]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vunpackb_mem_128:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vunpackb $1, (%eax), %xmm0 # encoding: [0x62,0xf3,0x7c,0x08,0x3d,0x00,0x01]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <16 x i8>, ptr %ptr_a
+  %ret = call <16 x i8> @llvm.x86.avx10.vunpackb.128(<16 x i8> %a, i8 1)
+  ret <16 x i8> %ret
+}
+
+define <32 x i8> @test_int_x86_avx10_vunpackb_mem_256(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vunpackb_mem_256:
+; X64:       # %bb.0:
+; X64-NEXT:    vunpackb $2, (%rdi), %ymm0 # encoding: [0x62,0xf3,0x7c,0x28,0x3d,0x07,0x02]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vunpackb_mem_256:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vunpackb $2, (%eax), %ymm0 # encoding: [0x62,0xf3,0x7c,0x28,0x3d,0x00,0x02]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <32 x i8>, ptr %ptr_a
+  %ret = call <32 x i8> @llvm.x86.avx10.vunpackb.256(<32 x i8> %a, i8 2)
+  ret <32 x i8> %ret
+}
+
+define <64 x i8> @test_int_x86_avx10_vunpackb_mem_512(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vunpackb_mem_512:
+; X64:       # %bb.0:
+; X64-NEXT:    vunpackb $3, (%rdi), %zmm0 # encoding: [0x62,0xf3,0x7c,0x48,0x3d,0x07,0x03]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vunpackb_mem_512:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vunpackb $3, (%eax), %zmm0 # encoding: [0x62,0xf3,0x7c,0x48,0x3d,0x00,0x03]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <64 x i8>, ptr %ptr_a
+  %ret = call <64 x i8> @llvm.x86.avx10.vunpackb.512(<64 x i8> %a, i8 3)
+  ret <64 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_pmovssdb_128(<4 x i32> %a) {
+; CHECK-LABEL: test_int_x86_avx10_pmovssdb_128:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vpmovssdb %xmm0, %xmm0 # encoding: [0x62,0xf2,0x7e,0x08,0x41,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.128(<4 x i32> %a, <16 x i8> zeroinitializer, i8 -1)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_pmovssdb_256(<8 x i32> %a) {
+; CHECK-LABEL: test_int_x86_avx10_pmovssdb_256:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vpmovssdb %ymm0, %xmm0 # encoding: [0x62,0xf2,0x7e,0x28,0x41,0xc0]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.256(<8 x i32> %a, <16 x i8> zeroinitializer, i8 -1)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_pmovssdb_512(<16 x i32> %a) {
+; CHECK-LABEL: test_int_x86_avx10_pmovssdb_512:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vpmovssdb %zmm0, %xmm0 # encoding: [0x62,0xf2,0x7e,0x48,0x41,0xc0]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.512(<16 x i32> %a, <16 x i8> zeroinitializer, i16 -1)
+  ret <16 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.mask.pmovss.db.128(<4 x i32>, <16 x i8>, i8)
+declare <16 x i8> @llvm.x86.avx10.mask.pmovss.db.256(<8 x i32>, <16 x i8>, i8)
+declare <16 x i8> @llvm.x86.avx10.mask.pmovss.db.512(<16 x i32>, <16 x i8>, i16)
+
+; Memory folding tests for vpmovssdb
+define <16 x i8> @test_int_x86_avx10_pmovssdb_mem_128(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_pmovssdb_mem_128:
+; X64:       # %bb.0:
+; X64-NEXT:    vmovdqa (%rdi), %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf9,0x6f,0x07]
+; X64-NEXT:    vpmovssdb %xmm0, %xmm0 # encoding: [0x62,0xf2,0x7e,0x08,0x41,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_pmovssdb_mem_128:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vmovdqa (%eax), %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf9,0x6f,0x00]
+; X86-NEXT:    vpmovssdb %xmm0, %xmm0 # encoding: [0x62,0xf2,0x7e,0x08,0x41,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <4 x i32>, ptr %ptr_a
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.128(<4 x i32> %a, <16 x i8> zeroinitializer, i8 -1)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_pmovssdb_mem_256(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_pmovssdb_mem_256:
+; X64:       # %bb.0:
+; X64-NEXT:    vmovdqa (%rdi), %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfd,0x6f,0x07]
+; X64-NEXT:    vpmovssdb %ymm0, %xmm0 # encoding: [0x62,0xf2,0x7e,0x28,0x41,0xc0]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_pmovssdb_mem_256:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vmovdqa (%eax), %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfd,0x6f,0x00]
+; X86-NEXT:    vpmovssdb %ymm0, %xmm0 # encoding: [0x62,0xf2,0x7e,0x28,0x41,0xc0]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <8 x i32>, ptr %ptr_a
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.256(<8 x i32> %a, <16 x i8> zeroinitializer, i8 -1)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_pmovssdb_mem_512(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_pmovssdb_mem_512:
+; X64:       # %bb.0:
+; X64-NEXT:    vmovdqa64 (%rdi), %zmm0 # encoding: [0x62,0xf1,0xfd,0x48,0x6f,0x07]
+; X64-NEXT:    vpmovssdb %zmm0, %xmm0 # encoding: [0x62,0xf2,0x7e,0x48,0x41,0xc0]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_pmovssdb_mem_512:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vmovdqa64 (%eax), %zmm0 # encoding: [0x62,0xf1,0xfd,0x48,0x6f,0x00]
+; X86-NEXT:    vpmovssdb %zmm0, %xmm0 # encoding: [0x62,0xf2,0x7e,0x48,0x41,0xc0]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <16 x i32>, ptr %ptr_a
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.512(<16 x i32> %a, <16 x i8> zeroinitializer, i16 -1)
+  ret <16 x i8> %ret
+}
diff --git a/llvm/test/MC/Disassembler/X86/avx10_v2_aux-32.txt b/llvm/test/MC/Disassembler/X86/avx10_v2_aux-32.txt
index 2f301e02e1d4f2..6dacc4e33bde00 100644
--- a/llvm/test/MC/Disassembler/X86/avx10_v2_aux-32.txt
+++ b/llvm/test/MC/Disassembler/X86/avx10_v2_aux-32.txt
@@ -1,12 +1,6 @@
 # RUN: llvm-mc --disassemble %s -triple=i386 | FileCheck %s --check-prefixes=ATT
 # RUN: llvm-mc --disassemble %s -triple=i386 --output-asm-variant=1 | FileCheck %s --check-prefixes=INTEL
 
-#
-# Group A: PS->8bit truncating conversions
-#
-
-# vcvtps2bf8
-
 # ATT:   vcvtps2bf8 %zmm1, %xmm0
 # INTEL: vcvtps2bf8 xmm0, zmm1
 0x62,0xf5,0x7e,0x48,0x39,0xc1
@@ -26,9 +20,6 @@
 # ATT:   vcvtps2bf8 %zmm1, %xmm0 {%k1} {z}
 # INTEL: vcvtps2bf8 xmm0 {k1} {z}, zmm1
 0x62,0xf5,0x7e,0xc9,0x39,0xc1
-
-# vcvtps2bf8s
-
 # ATT:   vcvtps2bf8s %zmm1, %xmm0
 # INTEL: vcvtps2bf8s xmm0, zmm1
 0x62,0xf5,0x7e,0x48,0x3b,0xc1
@@ -48,9 +39,6 @@
 # ATT:   vcvtps2bf8s %zmm1, %xmm0 {%k1} {z}
 # INTEL: vcvtps2bf8s xmm0 {k1} {z}, zmm1
 0x62,0xf5,0x7e,0xc9,0x3b,0xc1
-
-# vcvtps2hf8
-
 # ATT:   vcvtps2hf8 %zmm1, %xmm0
 # INTEL: vcvtps2hf8 xmm0, zmm1
 0x62,0xf5,0x7e,0x48,0x38,0xc1
@@ -70,9 +58,6 @@
 # ATT:   vcvtps2hf8 %zmm1, %xmm0 {%k1} {z}
 # INTEL: vcvtps2hf8 xmm0 {k1} {z}, zmm1
 0x62,0xf5,0x7e,0xc9,0x38,0xc1
-
-# vcvtps2hf8s
-
 # ATT:   vcvtps2hf8s %zmm1, %xmm0
 # INTEL: vcvtps2hf8s xmm0, zmm1
 0x62,0xf5,0x7e,0x48,0x3a,0xc1
@@ -92,9 +77,6 @@
 # ATT:   vcvtps2hf8s %zmm1, %xmm0 {%k1} {z}
 # INTEL: vcvtps2hf8s xmm0 {k1} {z}, zmm1
 0x62,0xf5,0x7e,0xc9,0x3a,0xc1
-
-# vcvtrops2hf8
-
 # ATT:   vcvtrops2hf8 %zmm1, %xmm0
 # INTEL: vcvtrops2hf8 xmm0, zmm1
 0x62,0xf5,0x7d,0x48,0x38,0xc1
@@ -114,9 +96,6 @@
 # ATT:   vcvtrops2hf8 %zmm1, %xmm0 {%k1} {z}
 # INTEL: vcvtrops2hf8 xmm0 {k1} {z}, zmm1
 0x62,0xf5,0x7d,0xc9,0x38,0xc1
-
-# vcvtrops2hf8s
-
 # ATT:   vcvtrops2hf8s %zmm1, %xmm0
 # INTEL: vcvtrops2hf8s xmm0, zmm1
 0x62,0xf5,0x7d,0x48,0x3a,0xc1
@@ -137,12 +116,6 @@
 # INTEL: vcvtrops2hf8s xmm0 {k1} {z}, zmm1
 0x62,0xf5,0x7d,0xc9,0x3a,0xc1
 
-#
-# Group B: Bias PS->8bit conversions (3-operand)
-#
-
-# vcvtbiasps2bf8
-
 # ATT:   vcvtbiasps2bf8 %zmm2, %zmm1, %xmm0
 # INTEL: vcvtbiasps2bf8 xmm0, zmm1, zmm2
 0x62,0xf5,0x74,0x48,0x39,0xc2
@@ -162,9 +135,6 @@
 # ATT:   vcvtbiasps2bf8 %zmm2, %zmm1, %xmm0 {%k1} {z}
 # INTEL: vcvtbiasps2bf8 xmm0 {k1} {z}, zmm1, zmm2
 0x62,0xf5,0x74,0xc9,0x39,0xc2
-
-# vcvtbiasps2bf8s
-
 # ATT:   vcvtbiasps2bf8s %zmm2, %zmm1, %xmm0
 # INTEL: vcvtbiasps2bf8s xmm0, zmm1, zmm2
 0x62,0xf5,0x74,0x48,0x3b,0xc2
@@ -184,9 +154,6 @@
 # ATT:   vcvtbiasps2bf8s %zmm2, %zmm1, %xmm0 {%k1} {z}
 # INTEL: vcvtbiasps2bf8s xmm0 {k1} {z}, zmm1, zmm2
 0x62,0xf5,0x74,0xc9,0x3b,0xc2
-
-# vcvtbiasps2hf8
-
 # ATT:   vcvtbiasps2hf8 %zmm2, %zmm1, %xmm0
 # INTEL: vcvtbiasps2hf8 xmm0, zmm1, zmm2
 0x62,0xf5,0x74,0x48,0x38,0xc2
@@ -206,9 +173,6 @@
 # ATT:   vcvtbiasps2hf8 %zmm2, %zmm1, %xmm0 {%k1} {z}
 # INTEL: vcvtbiasps2hf8 xmm0 {k1} {z}, zmm1, zmm2
 0x62,0xf5,0x74,0xc9,0x38,0xc2
-
-# vcvtbiasps2hf8s
-
 # ATT:   vcvtbiasps2hf8s %zmm2, %zmm1, %xmm0
 # INTEL: vcvtbiasps2hf8s xmm0, zmm1, zmm2
 0x62,0xf5,0x74,0x48,0x3a,0xc2
@@ -229,12 +193,6 @@
 # INTEL: vcvtbiasps2hf8s xmm0 {k1} {z}, zmm1, zmm2
 0x62,0xf5,0x74,0xc9,0x3a,0xc2
 
-#
-# Group C: 8bit->PS expanding conversions
-#
-
-# vcvtbf82ps
-
 # ATT:   vcvtbf82ps %xmm1, %zmm0
 # INTEL: vcvtbf82ps zmm0, xmm1
 0x62,0xf5,0xfc,0x48,0x36,0xc1
@@ -254,9 +212,6 @@
 # ATT:   vcvtbf82ps %xmm1, %zmm0 {%k1} {z}
 # INTEL: vcvtbf82ps zmm0 {k1} {z}, xmm1
 0x62,0xf5,0xfc,0xc9,0x36,0xc1
-
-# vcvthf82ps
-
 # ATT:   vcvthf82ps %xmm1, %zmm0
 # INTEL: vcvthf82ps zmm0, xmm1
 0x62,0xf5,0x7c,0x48,0x36,0xc1
@@ -277,12 +232,6 @@
 # INTEL: vcvthf82ps zmm0 {k1} {z}, xmm1
 0x62,0xf5,0x7c,0xc9,0x36,0xc1
 
-#
-# Group D: BF8/HF8->BF4S truncations
-#
-
-# vcvtbf82bf4s
-
 # ATT:   vcvtbf82bf4s %zmm1, %ymm0
 # INTEL: vcvtbf82bf4s ymm0, zmm1
 0x62,0xf5,0xfe,0x48,0x3d,0xc8
@@ -294,9 +243,6 @@
 # ATT:   vcvtbf82bf4s %xmm1, %xmm0
 # INTEL: vcvtbf82bf4s xmm0, xmm1
 0x62,0xf5,0xfe,0x08,0x3d,0xc8
-
-# vcvthf82bf4s
-
 # ATT:   vcvthf82bf4s %zmm1, %ymm0
 # INTEL: vcvthf82bf4s ymm0, zmm1
 0x62,0xf5,0x7e,0x48,0x3d,0xc8
@@ -309,12 +255,6 @@
 # INTEL: vcvthf82bf4s xmm0, xmm1
 0x62,0xf5,0x7e,0x08,0x3d,0xc8
 
-#
-# Group E: Same-size reg-only conversions (no masking)
-#
-
-# vcvtbf82bf6s
-
 # ATT:   vcvtbf82bf6s %zmm1, %zmm0
 # INTEL: vcvtbf82bf6s zmm0, zmm1
 0x62,0xf5,0xfe,0x48,0x3e,0xc1
@@ -326,9 +266,6 @@
 # ATT:   vcvtbf82bf6s %xmm1, %xmm0
 # INTEL: vcvtbf82bf6s xmm0, xmm1
 0x62,0xf5,0xfe,0x08,0x3e,0xc1
-
-# vcvthf82hf6s
-
 # ATT:   vcvthf82hf6s %zmm1, %zmm0
 # INTEL: vcvthf82hf6s zmm0, zmm1
 0x62,0xf5,0x7e,0x48,0x3c,0xc1
@@ -341,12 +278,6 @@
 # INTEL: vcvthf82hf6s xmm0, xmm1
 0x62,0xf5,0x7e,0x08,0x3c,0xc1
 
-#
-# Group F: Expanding/same-size conversions with masking
-#
-
-# vcvtbf42hf8
-
 # ATT:   vcvtbf42hf8 %ymm1, %zmm0
 # INTEL: vcvtbf42hf8 zmm0, ymm1
 0x62,0xf5,0x7c,0x48,0x37,0xc1
@@ -366,9 +297,6 @@
 # ATT:   vcvtbf42hf8 %ymm1, %zmm0 {%k1} {z}
 # INTEL: vcvtbf42hf8 zmm0 {k1} {z}, ymm1
 0x62,0xf5,0x7c,0xc9,0x37,0xc1
-
-# vcvtbf62hf8
-
 # ATT:   vcvtbf62hf8 %zmm1, %zmm0
 # INTEL: vcvtbf62hf8 zmm0, zmm1
 0x62,0xf5,0xfd,0x48,0x37,0xc1
@@ -388,9 +316,6 @@
 # ATT:   vcvtbf62hf8 %zmm1, %zmm0 {%k1} {z}
 # INTEL: vcvtbf62hf8 zmm0 {k1} {z}, zmm1
 0x62,0xf5,0xfd,0xc9,0x37,0xc1
-
-# vcvthf62hf8
-
 # ATT:   vcvthf62hf8 %zmm1, %zmm0
 # INTEL: vcvthf62hf8 zmm0, zmm1
 0x62,0xf5,0x7d,0x48,0x37,0xc1
@@ -410,11 +335,6 @@
 # ATT:   vcvthf62hf8 %zmm1, %zmm0 {%k1} {z}
 # INTEL: vcvthf62hf8 zmm0 {k1} {z}, zmm1
 0x62,0xf5,0x7d,0xc9,0x37,0xc1
-
-#
-# Group G: VPMOVSSDB - Integer DWord->Byte signed saturation
-#
-
 # ATT:   vpmovssdb %zmm1, %xmm0
 # INTEL: vpmovssdb xmm0, zmm1
 0x62,0xf2,0x7e,0x48,0x41,0xc8
@@ -434,11 +354,6 @@
 # ATT:   vpmovssdb %zmm1, %xmm0 {%k1} {z}
 # INTEL: vpmovssdb xmm0 {k1} {z}, zmm1
 0x62,0xf2,0x7e,0xc9,0x41,0xc8
-
-#
-# Group H: VUNPACKB - Byte unpack with immediate
-#
-
 # ATT:   vunpackb $1, %zmm1, %zmm0
 # INTEL: vunpackb zmm0, zmm1, 1
 0x62,0xf3,0x7c,0x48,0x3d,0xc1,0x01
diff --git a/llvm/test/MC/Disassembler/X86/avx10_v2_aux-64.txt b/llvm/test/MC/Disassembler/X86/avx10_v2_aux-64.txt
index 40a7a366e34f9a..498b10303f855e 100644
--- a/llvm/test/MC/Disassembler/X86/avx10_v2_aux-64.txt
+++ b/llvm/test/MC/Disassembler/X86/avx10_v2_aux-64.txt
@@ -1,12 +1,6 @@
 # RUN: llvm-mc --disassemble %s -triple=x86_64 | FileCheck %s --check-prefixes=ATT
 # RUN: llvm-mc --disassemble %s -triple=x86_64 --output-asm-variant=1 | FileCheck %s --check-prefixes=INTEL
 
-#
-# Group A: PS->8bit truncating conversions
-#
-
-# vcvtps2bf8
-
 # ATT:   vcvtps2bf8 %zmm1, %xmm0
 # INTEL: vcvtps2bf8 xmm0, zmm1
 0x62,0xf5,0x7e,0x48,0x39,0xc1
@@ -26,9 +20,6 @@
 # ATT:   vcvtps2bf8 %zmm1, %xmm0 {%k1} {z}
 # INTEL: vcvtps2bf8 xmm0 {k1} {z}, zmm1
 0x62,0xf5,0x7e,0xc9,0x39,0xc1
-
-# vcvtps2bf8s
-
 # ATT:   vcvtps2bf8s %zmm1, %xmm0
 # INTEL: vcvtps2bf8s xmm0, zmm1
 0x62,0xf5,0x7e,0x48,0x3b,0xc1
@@ -48,9 +39,6 @@
 # ATT:   vcvtps2bf8s %zmm1, %xmm0 {%k1} {z}
 # INTEL: vcvtps2bf8s xmm0 {k1} {z}, zmm1
 0x62,0xf5,0x7e,0xc9,0x3b,0xc1
-
-# vcvtps2hf8
-
 # ATT:   vcvtps2hf8 %zmm1, %xmm0
 # INTEL: vcvtps2hf8 xmm0, zmm1
 0x62,0xf5,0x7e,0x48,0x38,0xc1
@@ -70,9 +58,6 @@
 # ATT:   vcvtps2hf8 %zmm1, %xmm0 {%k1} {z}
 # INTEL: vcvtps2hf8 xmm0 {k1} {z}, zmm1
 0x62,0xf5,0x7e,0xc9,0x38,0xc1
-
-# vcvtps2hf8s
-
 # ATT:   vcvtps2hf8s %zmm1, %xmm0
 # INTEL: vcvtps2hf8s xmm0, zmm1
 0x62,0xf5,0x7e,0x48,0x3a,0xc1
@@ -92,9 +77,6 @@
 # ATT:   vcvtps2hf8s %zmm1, %xmm0 {%k1} {z}
 # INTEL: vcvtps2hf8s xmm0 {k1} {z}, zmm1
 0x62,0xf5,0x7e,0xc9,0x3a,0xc1
-
-# vcvtrops2hf8
-
 # ATT:   vcvtrops2hf8 %zmm1, %xmm0
 # INTEL: vcvtrops2hf8 xmm0, zmm1
 0x62,0xf5,0x7d,0x48,0x38,0xc1
@@ -114,9 +96,6 @@
 # ATT:   vcvtrops2hf8 %zmm1, %xmm0 {%k1} {z}
 # INTEL: vcvtrops2hf8 xmm0 {k1} {z}, zmm1
 0x62,0xf5,0x7d,0xc9,0x38,0xc1
-
-# vcvtrops2hf8s
-
 # ATT:   vcvtrops2hf8s %zmm1, %xmm0
 # INTEL: vcvtrops2hf8s xmm0, zmm1
 0x62,0xf5,0x7d,0x48,0x3a,0xc1
@@ -137,12 +116,6 @@
 # INTEL: vcvtrops2hf8s xmm0 {k1} {z}, zmm1
 0x62,0xf5,0x7d,0xc9,0x3a,0xc1
 
-#
-# Group B: Bias PS->8bit conversions (3-operand)
-#
-
-# vcvtbiasps2bf8
-
 # ATT:   vcvtbiasps2bf8 %zmm2, %zmm1, %xmm0
 # INTEL: vcvtbiasps2bf8 xmm0, zmm1, zmm2
 0x62,0xf5,0x74,0x48,0x39,0xc2
@@ -162,9 +135,6 @@
 # ATT:   vcvtbiasps2bf8 %zmm2, %zmm1, %xmm0 {%k1} {z}
 # INTEL: vcvtbiasps2bf8 xmm0 {k1} {z}, zmm1, zmm2
 0x62,0xf5,0x74,0xc9,0x39,0xc2
-
-# vcvtbiasps2bf8s
-
 # ATT:   vcvtbiasps2bf8s %zmm2, %zmm1, %xmm0
 # INTEL: vcvtbiasps2bf8s xmm0, zmm1, zmm2
 0x62,0xf5,0x74,0x48,0x3b,0xc2
@@ -184,9 +154,6 @@
 # ATT:   vcvtbiasps2bf8s %zmm2, %zmm1, %xmm0 {%k1} {z}
 # INTEL: vcvtbiasps2bf8s xmm0 {k1} {z}, zmm1, zmm2
 0x62,0xf5,0x74,0xc9,0x3b,0xc2
-
-# vcvtbiasps2hf8
-
 # ATT:   vcvtbiasps2hf8 %zmm2, %zmm1, %xmm0
 # INTEL: vcvtbiasps2hf8 xmm0, zmm1, zmm2
 0x62,0xf5,0x74,0x48,0x38,0xc2
@@ -206,9 +173,6 @@
 # ATT:   vcvtbiasps2hf8 %zmm2, %zmm1, %xmm0 {%k1} {z}
 # INTEL: vcvtbiasps2hf8 xmm0 {k1} {z}, zmm1, zmm2
 0x62,0xf5,0x74,0xc9,0x38,0xc2
-
-# vcvtbiasps2hf8s
-
 # ATT:   vcvtbiasps2hf8s %zmm2, %zmm1, %xmm0
 # INTEL: vcvtbiasps2hf8s xmm0, zmm1, zmm2
 0x62,0xf5,0x74,0x48,0x3a,0xc2
@@ -229,12 +193,6 @@
 # INTEL: vcvtbiasps2hf8s xmm0 {k1} {z}, zmm1, zmm2
 0x62,0xf5,0x74,0xc9,0x3a,0xc2
 
-#
-# Group C: 8bit->PS expanding conversions
-#
-
-# vcvtbf82ps
-
 # ATT:   vcvtbf82ps %xmm1, %zmm0
 # INTEL: vcvtbf82ps zmm0, xmm1
 0x62,0xf5,0xfc,0x48,0x36,0xc1
@@ -254,9 +212,6 @@
 # ATT:   vcvtbf82ps %xmm1, %zmm0 {%k1} {z}
 # INTEL: vcvtbf82ps zmm0 {k1} {z}, xmm1
 0x62,0xf5,0xfc,0xc9,0x36,0xc1
-
-# vcvthf82ps
-
 # ATT:   vcvthf82ps %xmm1, %zmm0
 # INTEL: vcvthf82ps zmm0, xmm1
 0x62,0xf5,0x7c,0x48,0x36,0xc1
@@ -277,12 +232,6 @@
 # INTEL: vcvthf82ps zmm0 {k1} {z}, xmm1
 0x62,0xf5,0x7c,0xc9,0x36,0xc1
 
-#
-# Group D: BF8/HF8->BF4S truncations
-#
-
-# vcvtbf82bf4s
-
 # ATT:   vcvtbf82bf4s %zmm1, %ymm0
 # INTEL: vcvtbf82bf4s ymm0, zmm1
 0x62,0xf5,0xfe,0x48,0x3d,0xc8
@@ -294,9 +243,6 @@
 # ATT:   vcvtbf82bf4s %xmm1, %xmm0
 # INTEL: vcvtbf82bf4s xmm0, xmm1
 0x62,0xf5,0xfe,0x08,0x3d,0xc8
-
-# vcvthf82bf4s
-
 # ATT:   vcvthf82bf4s %zmm1, %ymm0
 # INTEL: vcvthf82bf4s ymm0, zmm1
 0x62,0xf5,0x7e,0x48,0x3d,0xc8
@@ -309,12 +255,6 @@
 # INTEL: vcvthf82bf4s xmm0, xmm1
 0x62,0xf5,0x7e,0x08,0x3d,0xc8
 
-#
-# Group E: Same-size reg-only conversions (no masking)
-#
-
-# vcvtbf82bf6s
-
 # ATT:   vcvtbf82bf6s %zmm1, %zmm0
 # INTEL: vcvtbf82bf6s zmm0, zmm1
 0x62,0xf5,0xfe,0x48,0x3e,0xc1
@@ -326,9 +266,6 @@
 # ATT:   vcvtbf82bf6s %xmm1, %xmm0
 # INTEL: vcvtbf82bf6s xmm0, xmm1
 0x62,0xf5,0xfe,0x08,0x3e,0xc1
-
-# vcvthf82hf6s
-
 # ATT:   vcvthf82hf6s %zmm1, %zmm0
 # INTEL: vcvthf82hf6s zmm0, zmm1
 0x62,0xf5,0x7e,0x48,0x3c,0xc1
@@ -341,12 +278,6 @@
 # INTEL: vcvthf82hf6s xmm0, xmm1
 0x62,0xf5,0x7e,0x08,0x3c,0xc1
 
-#
-# Group F: Expanding/same-size conversions with masking
-#
-
-# vcvtbf42hf8
-
 # ATT:   vcvtbf42hf8 %ymm1, %zmm0
 # INTEL: vcvtbf42hf8 zmm0, ymm1
 0x62,0xf5,0x7c,0x48,0x37,0xc1
@@ -366,9 +297,6 @@
 # ATT:   vcvtbf42hf8 %ymm1, %zmm0 {%k1} {z}
 # INTEL: vcvtbf42hf8 zmm0 {k1} {z}, ymm1
 0x62,0xf5,0x7c,0xc9,0x37,0xc1
-
-# vcvtbf62hf8
-
 # ATT:   vcvtbf62hf8 %zmm1, %zmm0
 # INTEL: vcvtbf62hf8 zmm0, zmm1
 0x62,0xf5,0xfd,0x48,0x37,0xc1
@@ -388,9 +316,6 @@
 # ATT:   vcvtbf62hf8 %zmm1, %zmm0 {%k1} {z}
 # INTEL: vcvtbf62hf8 zmm0 {k1} {z}, zmm1
 0x62,0xf5,0xfd,0xc9,0x37,0xc1
-
-# vcvthf62hf8
-
 # ATT:   vcvthf62hf8 %zmm1, %zmm0
 # INTEL: vcvthf62hf8 zmm0, zmm1
 0x62,0xf5,0x7d,0x48,0x37,0xc1
@@ -410,11 +335,6 @@
 # ATT:   vcvthf62hf8 %zmm1, %zmm0 {%k1} {z}
 # INTEL: vcvthf62hf8 zmm0 {k1} {z}, zmm1
 0x62,0xf5,0x7d,0xc9,0x37,0xc1
-
-#
-# Group G: VPMOVSSDB - Integer DWord->Byte signed saturation
-#
-
 # ATT:   vpmovssdb %zmm1, %xmm0
 # INTEL: vpmovssdb xmm0, zmm1
 0x62,0xf2,0x7e,0x48,0x41,0xc8
@@ -434,11 +354,6 @@
 # ATT:   vpmovssdb %zmm1, %xmm0 {%k1} {z}
 # INTEL: vpmovssdb xmm0 {k1} {z}, zmm1
 0x62,0xf2,0x7e,0xc9,0x41,0xc8
-
-#
-# Group H: VUNPACKB - Byte unpack with immediate
-#
-
 # ATT:   vunpackb $1, %zmm1, %zmm0
 # INTEL: vunpackb zmm0, zmm1, 1
 0x62,0xf3,0x7c,0x48,0x3d,0xc1,0x01
diff --git a/llvm/test/MC/X86/avx10_v2_aux-att-32.s b/llvm/test/MC/X86/avx10_v2_aux-att-32.s
index 10c4ed5bd32592..3ac02c13f2587b 100644
--- a/llvm/test/MC/X86/avx10_v2_aux-att-32.s
+++ b/llvm/test/MC/X86/avx10_v2_aux-att-32.s
@@ -1,4 +1,4 @@
-// RUN: llvm-mc -triple i386 --show-encoding -mattr=+avx10-v2-aux,+avx512vl %s | FileCheck %s
+// RUN: llvm-mc -triple i386 --show-encoding -mattr=+avx10v2aux,+avx512vl %s | FileCheck %s
 
 //
 // Group A: PS->8bit truncating conversions
@@ -22,6 +22,14 @@
 // CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x39,0x07]
           vcvtps2bf8 (%edi), %xmm0
 
+// CHECK: vcvtps2bf8y (%edi), %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x28,0x39,0x07]
+          vcvtps2bf8y (%edi), %xmm0
+
+// CHECK: vcvtps2bf8x (%edi), %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x39,0x07]
+          vcvtps2bf8x (%edi), %xmm0
+
 // CHECK: vcvtps2bf8 %zmm1, %xmm0 {%k1}
 // CHECK: encoding: [0x62,0xf5,0x7e,0x49,0x39,0xc1]
           vcvtps2bf8 %zmm1, %xmm0 {%k1}
@@ -30,6 +38,10 @@
 // CHECK: encoding: [0x62,0xf5,0x7e,0xc9,0x39,0xc1]
           vcvtps2bf8 %zmm1, %xmm0 {%k1} {z}
 
+// CHECK: vcvtps2bf8 (%edi){1to16}, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x58,0x39,0x07]
+          vcvtps2bf8 (%edi){1to16}, %xmm0
+
 // vcvtps2bf8s
 
 // CHECK: vcvtps2bf8s %zmm1, %xmm0
@@ -44,6 +56,18 @@
 // CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x3b,0xc1]
           vcvtps2bf8s %xmm1, %xmm0
 
+// CHECK: vcvtps2bf8s (%edi), %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x3b,0x07]
+          vcvtps2bf8s (%edi), %xmm0
+
+// CHECK: vcvtps2bf8sy (%edi), %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x28,0x3b,0x07]
+          vcvtps2bf8sy (%edi), %xmm0
+
+// CHECK: vcvtps2bf8sx (%edi), %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x3b,0x07]
+          vcvtps2bf8sx (%edi), %xmm0
+
 // CHECK: vcvtps2bf8s %zmm1, %xmm0 {%k1}
 // CHECK: encoding: [0x62,0xf5,0x7e,0x49,0x3b,0xc1]
           vcvtps2bf8s %zmm1, %xmm0 {%k1}
@@ -52,6 +76,10 @@
 // CHECK: encoding: [0x62,0xf5,0x7e,0xc9,0x3b,0xc1]
           vcvtps2bf8s %zmm1, %xmm0 {%k1} {z}
 
+// CHECK: vcvtps2bf8s (%edi){1to16}, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x58,0x3b,0x07]
+          vcvtps2bf8s (%edi){1to16}, %xmm0
+
 // vcvtps2hf8
 
 // CHECK: vcvtps2hf8 %zmm1, %xmm0
@@ -66,6 +94,18 @@
 // CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x38,0xc1]
           vcvtps2hf8 %xmm1, %xmm0
 
+// CHECK: vcvtps2hf8 (%edi), %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x38,0x07]
+          vcvtps2hf8 (%edi), %xmm0
+
+// CHECK: vcvtps2hf8y (%edi), %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x28,0x38,0x07]
+          vcvtps2hf8y (%edi), %xmm0
+
+// CHECK: vcvtps2hf8x (%edi), %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x38,0x07]
+          vcvtps2hf8x (%edi), %xmm0
+
 // CHECK: vcvtps2hf8 %zmm1, %xmm0 {%k1}
 // CHECK: encoding: [0x62,0xf5,0x7e,0x49,0x38,0xc1]
           vcvtps2hf8 %zmm1, %xmm0 {%k1}
@@ -74,6 +114,10 @@
 // CHECK: encoding: [0x62,0xf5,0x7e,0xc9,0x38,0xc1]
           vcvtps2hf8 %zmm1, %xmm0 {%k1} {z}
 
+// CHECK: vcvtps2hf8 (%edi){1to16}, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x58,0x38,0x07]
+          vcvtps2hf8 (%edi){1to16}, %xmm0
+
 // vcvtps2hf8s
 
 // CHECK: vcvtps2hf8s %zmm1, %xmm0
@@ -88,6 +132,18 @@
 // CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x3a,0xc1]
           vcvtps2hf8s %xmm1, %xmm0
 
+// CHECK: vcvtps2hf8s (%edi), %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x3a,0x07]
+          vcvtps2hf8s (%edi), %xmm0
+
+// CHECK: vcvtps2hf8sy (%edi), %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x28,0x3a,0x07]
+          vcvtps2hf8sy (%edi), %xmm0
+
+// CHECK: vcvtps2hf8sx (%edi), %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x3a,0x07]
+          vcvtps2hf8sx (%edi), %xmm0
+
 // CHECK: vcvtps2hf8s %zmm1, %xmm0 {%k1}
 // CHECK: encoding: [0x62,0xf5,0x7e,0x49,0x3a,0xc1]
           vcvtps2hf8s %zmm1, %xmm0 {%k1}
@@ -96,6 +152,10 @@
 // CHECK: encoding: [0x62,0xf5,0x7e,0xc9,0x3a,0xc1]
           vcvtps2hf8s %zmm1, %xmm0 {%k1} {z}
 
+// CHECK: vcvtps2hf8s (%edi){1to16}, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x58,0x3a,0x07]
+          vcvtps2hf8s (%edi){1to16}, %xmm0
+
 // vcvtrops2hf8
 
 // CHECK: vcvtrops2hf8 %zmm1, %xmm0
@@ -110,6 +170,18 @@
 // CHECK: encoding: [0x62,0xf5,0x7d,0x08,0x38,0xc1]
           vcvtrops2hf8 %xmm1, %xmm0
 
+// CHECK: vcvtrops2hf8 (%edi), %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7d,0x48,0x38,0x07]
+          vcvtrops2hf8 (%edi), %xmm0
+
+// CHECK: vcvtrops2hf8y (%edi), %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7d,0x28,0x38,0x07]
+          vcvtrops2hf8y (%edi), %xmm0
+
+// CHECK: vcvtrops2hf8x (%edi), %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7d,0x08,0x38,0x07]
+          vcvtrops2hf8x (%edi), %xmm0
+
 // CHECK: vcvtrops2hf8 %zmm1, %xmm0 {%k1}
 // CHECK: encoding: [0x62,0xf5,0x7d,0x49,0x38,0xc1]
           vcvtrops2hf8 %zmm1, %xmm0 {%k1}
@@ -118,6 +190,10 @@
 // CHECK: encoding: [0x62,0xf5,0x7d,0xc9,0x38,0xc1]
           vcvtrops2hf8 %zmm1, %xmm0 {%k1} {z}
 
+// CHECK: vcvtrops2hf8 (%edi){1to16}, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7d,0x58,0x38,0x07]
+          vcvtrops2hf8 (%edi){1to16}, %xmm0
+
 // vcvtrops2hf8s
 
 // CHECK: vcvtrops2hf8s %zmm1, %xmm0
@@ -132,6 +208,18 @@
 // CHECK: encoding: [0x62,0xf5,0x7d,0x08,0x3a,0xc1]
           vcvtrops2hf8s %xmm1, %xmm0
 
+// CHECK: vcvtrops2hf8s (%edi), %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7d,0x48,0x3a,0x07]
+          vcvtrops2hf8s (%edi), %xmm0
+
+// CHECK: vcvtrops2hf8sy (%edi), %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7d,0x28,0x3a,0x07]
+          vcvtrops2hf8sy (%edi), %xmm0
+
+// CHECK: vcvtrops2hf8sx (%edi), %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7d,0x08,0x3a,0x07]
+          vcvtrops2hf8sx (%edi), %xmm0
+
 // CHECK: vcvtrops2hf8s %zmm1, %xmm0 {%k1}
 // CHECK: encoding: [0x62,0xf5,0x7d,0x49,0x3a,0xc1]
           vcvtrops2hf8s %zmm1, %xmm0 {%k1}
@@ -140,6 +228,10 @@
 // CHECK: encoding: [0x62,0xf5,0x7d,0xc9,0x3a,0xc1]
           vcvtrops2hf8s %zmm1, %xmm0 {%k1} {z}
 
+// CHECK: vcvtrops2hf8s (%edi){1to16}, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7d,0x58,0x3a,0x07]
+          vcvtrops2hf8s (%edi){1to16}, %xmm0
+
 //
 // Group B: Bias PS->8bit conversions (3-operand)
 //
@@ -166,6 +258,18 @@
 // CHECK: encoding: [0x62,0xf5,0x74,0xc9,0x39,0xc2]
           vcvtbiasps2bf8 %zmm2, %zmm1, %xmm0 {%k1} {z}
 
+// CHECK: vcvtbiasps2bf8 (%edi), %zmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x48,0x39,0x07]
+          vcvtbiasps2bf8 (%edi), %zmm1, %xmm0
+
+// CHECK: vcvtbiasps2bf8 (%edi), %ymm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x28,0x39,0x07]
+          vcvtbiasps2bf8 (%edi), %ymm1, %xmm0
+
+// CHECK: vcvtbiasps2bf8 (%edi), %xmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x08,0x39,0x07]
+          vcvtbiasps2bf8 (%edi), %xmm1, %xmm0
+
 // vcvtbiasps2bf8s
 
 // CHECK: vcvtbiasps2bf8s %zmm2, %zmm1, %xmm0
@@ -188,6 +292,18 @@
 // CHECK: encoding: [0x62,0xf5,0x74,0xc9,0x3b,0xc2]
           vcvtbiasps2bf8s %zmm2, %zmm1, %xmm0 {%k1} {z}
 
+// CHECK: vcvtbiasps2bf8s (%edi), %zmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x48,0x3b,0x07]
+          vcvtbiasps2bf8s (%edi), %zmm1, %xmm0
+
+// CHECK: vcvtbiasps2bf8s (%edi), %ymm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x28,0x3b,0x07]
+          vcvtbiasps2bf8s (%edi), %ymm1, %xmm0
+
+// CHECK: vcvtbiasps2bf8s (%edi), %xmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x08,0x3b,0x07]
+          vcvtbiasps2bf8s (%edi), %xmm1, %xmm0
+
 // vcvtbiasps2hf8
 
 // CHECK: vcvtbiasps2hf8 %zmm2, %zmm1, %xmm0
@@ -210,6 +326,18 @@
 // CHECK: encoding: [0x62,0xf5,0x74,0xc9,0x38,0xc2]
           vcvtbiasps2hf8 %zmm2, %zmm1, %xmm0 {%k1} {z}
 
+// CHECK: vcvtbiasps2hf8 (%edi), %zmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x48,0x38,0x07]
+          vcvtbiasps2hf8 (%edi), %zmm1, %xmm0
+
+// CHECK: vcvtbiasps2hf8 (%edi), %ymm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x28,0x38,0x07]
+          vcvtbiasps2hf8 (%edi), %ymm1, %xmm0
+
+// CHECK: vcvtbiasps2hf8 (%edi), %xmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x08,0x38,0x07]
+          vcvtbiasps2hf8 (%edi), %xmm1, %xmm0
+
 // vcvtbiasps2hf8s
 
 // CHECK: vcvtbiasps2hf8s %zmm2, %zmm1, %xmm0
@@ -232,6 +360,18 @@
 // CHECK: encoding: [0x62,0xf5,0x74,0xc9,0x3a,0xc2]
           vcvtbiasps2hf8s %zmm2, %zmm1, %xmm0 {%k1} {z}
 
+// CHECK: vcvtbiasps2hf8s (%edi), %zmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x48,0x3a,0x07]
+          vcvtbiasps2hf8s (%edi), %zmm1, %xmm0
+
+// CHECK: vcvtbiasps2hf8s (%edi), %ymm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x28,0x3a,0x07]
+          vcvtbiasps2hf8s (%edi), %ymm1, %xmm0
+
+// CHECK: vcvtbiasps2hf8s (%edi), %xmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x08,0x3a,0x07]
+          vcvtbiasps2hf8s (%edi), %xmm1, %xmm0
+
 //
 // Group C: 8bit->PS expanding conversions
 //
@@ -258,6 +398,18 @@
 // CHECK: encoding: [0x62,0xf5,0xfc,0xc9,0x36,0xc1]
           vcvtbf82ps %xmm1, %zmm0 {%k1} {z}
 
+// CHECK: vcvtbf82ps (%edi), %zmm0
+// CHECK: encoding: [0x62,0xf5,0xfc,0x48,0x36,0x07]
+          vcvtbf82ps (%edi), %zmm0
+
+// CHECK: vcvtbf82ps (%edi), %ymm0
+// CHECK: encoding: [0x62,0xf5,0xfc,0x28,0x36,0x07]
+          vcvtbf82ps (%edi), %ymm0
+
+// CHECK: vcvtbf82ps (%edi), %xmm0
+// CHECK: encoding: [0x62,0xf5,0xfc,0x08,0x36,0x07]
+          vcvtbf82ps (%edi), %xmm0
+
 // vcvthf82ps
 
 // CHECK: vcvthf82ps %xmm1, %zmm0
@@ -280,6 +432,18 @@
 // CHECK: encoding: [0x62,0xf5,0x7c,0xc9,0x36,0xc1]
           vcvthf82ps %xmm1, %zmm0 {%k1} {z}
 
+// CHECK: vcvthf82ps (%edi), %zmm0
+// CHECK: encoding: [0x62,0xf5,0x7c,0x48,0x36,0x07]
+          vcvthf82ps (%edi), %zmm0
+
+// CHECK: vcvthf82ps (%edi), %ymm0
+// CHECK: encoding: [0x62,0xf5,0x7c,0x28,0x36,0x07]
+          vcvthf82ps (%edi), %ymm0
+
+// CHECK: vcvthf82ps (%edi), %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7c,0x08,0x36,0x07]
+          vcvthf82ps (%edi), %xmm0
+
 //
 // Group D: BF8/HF8->BF4S truncations
 //
@@ -298,6 +462,18 @@
 // CHECK: encoding: [0x62,0xf5,0xfe,0x08,0x3d,0xc8]
           vcvtbf82bf4s %xmm1, %xmm0
 
+// CHECK: vcvtbf82bf4s %zmm1, (%edi)
+// CHECK: encoding: [0x62,0xf5,0xfe,0x48,0x3d,0x0f]
+          vcvtbf82bf4s %zmm1, (%edi)
+
+// CHECK: vcvtbf82bf4s %ymm1, (%edi)
+// CHECK: encoding: [0x62,0xf5,0xfe,0x28,0x3d,0x0f]
+          vcvtbf82bf4s %ymm1, (%edi)
+
+// CHECK: vcvtbf82bf4s %xmm1, (%edi)
+// CHECK: encoding: [0x62,0xf5,0xfe,0x08,0x3d,0x0f]
+          vcvtbf82bf4s %xmm1, (%edi)
+
 // vcvthf82bf4s
 
 // CHECK: vcvthf82bf4s %zmm1, %ymm0
@@ -312,6 +488,18 @@
 // CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x3d,0xc8]
           vcvthf82bf4s %xmm1, %xmm0
 
+// CHECK: vcvthf82bf4s %zmm1, (%edi)
+// CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x3d,0x0f]
+          vcvthf82bf4s %zmm1, (%edi)
+
+// CHECK: vcvthf82bf4s %ymm1, (%edi)
+// CHECK: encoding: [0x62,0xf5,0x7e,0x28,0x3d,0x0f]
+          vcvthf82bf4s %ymm1, (%edi)
+
+// CHECK: vcvthf82bf4s %xmm1, (%edi)
+// CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x3d,0x0f]
+          vcvthf82bf4s %xmm1, (%edi)
+
 //
 // Group E: Same-size reg-only conversions (no masking)
 //
@@ -370,6 +558,18 @@
 // CHECK: encoding: [0x62,0xf5,0x7c,0xc9,0x37,0xc1]
           vcvtbf42hf8 %ymm1, %zmm0 {%k1} {z}
 
+// CHECK: vcvtbf42hf8 (%edi), %zmm0
+// CHECK: encoding: [0x62,0xf5,0x7c,0x48,0x37,0x07]
+          vcvtbf42hf8 (%edi), %zmm0
+
+// CHECK: vcvtbf42hf8 (%edi), %ymm0
+// CHECK: encoding: [0x62,0xf5,0x7c,0x28,0x37,0x07]
+          vcvtbf42hf8 (%edi), %ymm0
+
+// CHECK: vcvtbf42hf8 (%edi), %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7c,0x08,0x37,0x07]
+          vcvtbf42hf8 (%edi), %xmm0
+
 // vcvtbf62hf8
 
 // CHECK: vcvtbf62hf8 %zmm1, %zmm0
@@ -438,6 +638,18 @@
 // CHECK: encoding: [0x62,0xf2,0x7e,0xc9,0x41,0xc8]
           vpmovssdb %zmm1, %xmm0 {%k1} {z}
 
+// CHECK: vpmovssdb %zmm1, (%edi)
+// CHECK: encoding: [0x62,0xf2,0x7e,0x48,0x41,0x0f]
+          vpmovssdb %zmm1, (%edi)
+
+// CHECK: vpmovssdb %ymm1, (%edi)
+// CHECK: encoding: [0x62,0xf2,0x7e,0x28,0x41,0x0f]
+          vpmovssdb %ymm1, (%edi)
+
+// CHECK: vpmovssdb %xmm1, (%edi)
+// CHECK: encoding: [0x62,0xf2,0x7e,0x08,0x41,0x0f]
+          vpmovssdb %xmm1, (%edi)
+
 //
 // Group H: VUNPACKB - Byte unpack with immediate
 //
@@ -461,3 +673,15 @@
 // CHECK: vunpackb $1, %zmm1, %zmm0 {%k1} {z}
 // CHECK: encoding: [0x62,0xf3,0x7c,0xc9,0x3d,0xc1,0x01]
           vunpackb $1, %zmm1, %zmm0 {%k1} {z}
+
+// CHECK: vunpackb $1, (%edi), %zmm0
+// CHECK: encoding: [0x62,0xf3,0x7c,0x48,0x3d,0x07,0x01]
+          vunpackb $1, (%edi), %zmm0
+
+// CHECK: vunpackb $1, (%edi), %ymm0
+// CHECK: encoding: [0x62,0xf3,0x7c,0x28,0x3d,0x07,0x01]
+          vunpackb $1, (%edi), %ymm0
+
+// CHECK: vunpackb $1, (%edi), %xmm0
+// CHECK: encoding: [0x62,0xf3,0x7c,0x08,0x3d,0x07,0x01]
+          vunpackb $1, (%edi), %xmm0
diff --git a/llvm/test/MC/X86/avx10_v2_aux-att-64.s b/llvm/test/MC/X86/avx10_v2_aux-att-64.s
index 81154f3d8b33c6..fb84b55a896cf0 100644
--- a/llvm/test/MC/X86/avx10_v2_aux-att-64.s
+++ b/llvm/test/MC/X86/avx10_v2_aux-att-64.s
@@ -1,4 +1,4 @@
-// RUN: llvm-mc -triple x86_64 --show-encoding -mattr=+avx10-v2-aux,+avx512vl %s | FileCheck %s
+// RUN: llvm-mc -triple x86_64 --show-encoding -mattr=+avx10v2aux,+avx512vl %s | FileCheck %s
 
 //
 // Group A: PS->8bit truncating conversions
diff --git a/llvm/test/MC/X86/avx10_v2_aux-intel-32.s b/llvm/test/MC/X86/avx10_v2_aux-intel-32.s
index 7a06743ed04b8d..b415359ac87818 100644
--- a/llvm/test/MC/X86/avx10_v2_aux-intel-32.s
+++ b/llvm/test/MC/X86/avx10_v2_aux-intel-32.s
@@ -1,4 +1,4 @@
-// RUN: llvm-mc -triple i386 -x86-asm-syntax=intel -output-asm-variant=1 --show-encoding -mattr=+avx10-v2-aux,+avx512vl %s | FileCheck %s
+// RUN: llvm-mc -triple i386 -x86-asm-syntax=intel -output-asm-variant=1 --show-encoding -mattr=+avx10v2aux,+avx512vl %s | FileCheck %s
 
 //
 // Group A: PS->8bit truncating conversions
@@ -26,6 +26,22 @@
 // CHECK: encoding: [0x62,0xf5,0x7e,0xc9,0x39,0xc1]
           vcvtps2bf8 xmm0 {k1} {z}, zmm1
 
+// CHECK: vcvtps2bf8 xmm0, zmmword ptr [edi]
+// CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x39,0x07]
+          vcvtps2bf8 xmm0, zmmword ptr [edi]
+
+// CHECK: vcvtps2bf8 xmm0, ymmword ptr [edi]
+// CHECK: encoding: [0x62,0xf5,0x7e,0x28,0x39,0x07]
+          vcvtps2bf8 xmm0, ymmword ptr [edi]
+
+// CHECK: vcvtps2bf8 xmm0, xmmword ptr [edi]
+// CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x39,0x07]
+          vcvtps2bf8 xmm0, xmmword ptr [edi]
+
+// CHECK: vcvtps2bf8 xmm0, dword ptr [edi]{1to16}
+// CHECK: encoding: [0x62,0xf5,0x7e,0x58,0x39,0x07]
+          vcvtps2bf8 xmm0, dword ptr [edi]{1to16}
+
 // vcvtps2bf8s
 
 // CHECK: vcvtps2bf8s xmm0, zmm1
@@ -40,6 +56,30 @@
 // CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x3b,0xc1]
           vcvtps2bf8s xmm0, xmm1
 
+// CHECK: vcvtps2bf8s xmm0 {k1}, zmm1
+// CHECK: encoding: [0x62,0xf5,0x7e,0x49,0x3b,0xc1]
+          vcvtps2bf8s xmm0 {k1}, zmm1
+
+// CHECK: vcvtps2bf8s xmm0 {k1} {z}, zmm1
+// CHECK: encoding: [0x62,0xf5,0x7e,0xc9,0x3b,0xc1]
+          vcvtps2bf8s xmm0 {k1} {z}, zmm1
+
+// CHECK: vcvtps2bf8s xmm0, zmmword ptr [edi]
+// CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x3b,0x07]
+          vcvtps2bf8s xmm0, zmmword ptr [edi]
+
+// CHECK: vcvtps2bf8s xmm0, ymmword ptr [edi]
+// CHECK: encoding: [0x62,0xf5,0x7e,0x28,0x3b,0x07]
+          vcvtps2bf8s xmm0, ymmword ptr [edi]
+
+// CHECK: vcvtps2bf8s xmm0, xmmword ptr [edi]
+// CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x3b,0x07]
+          vcvtps2bf8s xmm0, xmmword ptr [edi]
+
+// CHECK: vcvtps2bf8s xmm0, dword ptr [edi]{1to16}
+// CHECK: encoding: [0x62,0xf5,0x7e,0x58,0x3b,0x07]
+          vcvtps2bf8s xmm0, dword ptr [edi]{1to16}
+
 // vcvtps2hf8
 
 // CHECK: vcvtps2hf8 xmm0, zmm1
@@ -54,6 +94,30 @@
 // CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x38,0xc1]
           vcvtps2hf8 xmm0, xmm1
 
+// CHECK: vcvtps2hf8 xmm0 {k1}, zmm1
+// CHECK: encoding: [0x62,0xf5,0x7e,0x49,0x38,0xc1]
+          vcvtps2hf8 xmm0 {k1}, zmm1
+
+// CHECK: vcvtps2hf8 xmm0 {k1} {z}, zmm1
+// CHECK: encoding: [0x62,0xf5,0x7e,0xc9,0x38,0xc1]
+          vcvtps2hf8 xmm0 {k1} {z}, zmm1
+
+// CHECK: vcvtps2hf8 xmm0, zmmword ptr [edi]
+// CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x38,0x07]
+          vcvtps2hf8 xmm0, zmmword ptr [edi]
+
+// CHECK: vcvtps2hf8 xmm0, ymmword ptr [edi]
+// CHECK: encoding: [0x62,0xf5,0x7e,0x28,0x38,0x07]
+          vcvtps2hf8 xmm0, ymmword ptr [edi]
+
+// CHECK: vcvtps2hf8 xmm0, xmmword ptr [edi]
+// CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x38,0x07]
+          vcvtps2hf8 xmm0, xmmword ptr [edi]
+
+// CHECK: vcvtps2hf8 xmm0, dword ptr [edi]{1to16}
+// CHECK: encoding: [0x62,0xf5,0x7e,0x58,0x38,0x07]
+          vcvtps2hf8 xmm0, dword ptr [edi]{1to16}
+
 // vcvtps2hf8s
 
 // CHECK: vcvtps2hf8s xmm0, zmm1
@@ -68,6 +132,30 @@
 // CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x3a,0xc1]
           vcvtps2hf8s xmm0, xmm1
 
+// CHECK: vcvtps2hf8s xmm0 {k1}, zmm1
+// CHECK: encoding: [0x62,0xf5,0x7e,0x49,0x3a,0xc1]
+          vcvtps2hf8s xmm0 {k1}, zmm1
+
+// CHECK: vcvtps2hf8s xmm0 {k1} {z}, zmm1
+// CHECK: encoding: [0x62,0xf5,0x7e,0xc9,0x3a,0xc1]
+          vcvtps2hf8s xmm0 {k1} {z}, zmm1
+
+// CHECK: vcvtps2hf8s xmm0, zmmword ptr [edi]
+// CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x3a,0x07]
+          vcvtps2hf8s xmm0, zmmword ptr [edi]
+
+// CHECK: vcvtps2hf8s xmm0, ymmword ptr [edi]
+// CHECK: encoding: [0x62,0xf5,0x7e,0x28,0x3a,0x07]
+          vcvtps2hf8s xmm0, ymmword ptr [edi]
+
+// CHECK: vcvtps2hf8s xmm0, xmmword ptr [edi]
+// CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x3a,0x07]
+          vcvtps2hf8s xmm0, xmmword ptr [edi]
+
+// CHECK: vcvtps2hf8s xmm0, dword ptr [edi]{1to16}
+// CHECK: encoding: [0x62,0xf5,0x7e,0x58,0x3a,0x07]
+          vcvtps2hf8s xmm0, dword ptr [edi]{1to16}
+
 // vcvtrops2hf8
 
 // CHECK: vcvtrops2hf8 xmm0, zmm1
@@ -82,6 +170,30 @@
 // CHECK: encoding: [0x62,0xf5,0x7d,0x08,0x38,0xc1]
           vcvtrops2hf8 xmm0, xmm1
 
+// CHECK: vcvtrops2hf8 xmm0 {k1}, zmm1
+// CHECK: encoding: [0x62,0xf5,0x7d,0x49,0x38,0xc1]
+          vcvtrops2hf8 xmm0 {k1}, zmm1
+
+// CHECK: vcvtrops2hf8 xmm0 {k1} {z}, zmm1
+// CHECK: encoding: [0x62,0xf5,0x7d,0xc9,0x38,0xc1]
+          vcvtrops2hf8 xmm0 {k1} {z}, zmm1
+
+// CHECK: vcvtrops2hf8 xmm0, zmmword ptr [edi]
+// CHECK: encoding: [0x62,0xf5,0x7d,0x48,0x38,0x07]
+          vcvtrops2hf8 xmm0, zmmword ptr [edi]
+
+// CHECK: vcvtrops2hf8 xmm0, ymmword ptr [edi]
+// CHECK: encoding: [0x62,0xf5,0x7d,0x28,0x38,0x07]
+          vcvtrops2hf8 xmm0, ymmword ptr [edi]
+
+// CHECK: vcvtrops2hf8 xmm0, xmmword ptr [edi]
+// CHECK: encoding: [0x62,0xf5,0x7d,0x08,0x38,0x07]
+          vcvtrops2hf8 xmm0, xmmword ptr [edi]
+
+// CHECK: vcvtrops2hf8 xmm0, dword ptr [edi]{1to16}
+// CHECK: encoding: [0x62,0xf5,0x7d,0x58,0x38,0x07]
+          vcvtrops2hf8 xmm0, dword ptr [edi]{1to16}
+
 // vcvtrops2hf8s
 
 // CHECK: vcvtrops2hf8s xmm0, zmm1
@@ -96,6 +208,30 @@
 // CHECK: encoding: [0x62,0xf5,0x7d,0x08,0x3a,0xc1]
           vcvtrops2hf8s xmm0, xmm1
 
+// CHECK: vcvtrops2hf8s xmm0 {k1}, zmm1
+// CHECK: encoding: [0x62,0xf5,0x7d,0x49,0x3a,0xc1]
+          vcvtrops2hf8s xmm0 {k1}, zmm1
+
+// CHECK: vcvtrops2hf8s xmm0 {k1} {z}, zmm1
+// CHECK: encoding: [0x62,0xf5,0x7d,0xc9,0x3a,0xc1]
+          vcvtrops2hf8s xmm0 {k1} {z}, zmm1
+
+// CHECK: vcvtrops2hf8s xmm0, zmmword ptr [edi]
+// CHECK: encoding: [0x62,0xf5,0x7d,0x48,0x3a,0x07]
+          vcvtrops2hf8s xmm0, zmmword ptr [edi]
+
+// CHECK: vcvtrops2hf8s xmm0, ymmword ptr [edi]
+// CHECK: encoding: [0x62,0xf5,0x7d,0x28,0x3a,0x07]
+          vcvtrops2hf8s xmm0, ymmword ptr [edi]
+
+// CHECK: vcvtrops2hf8s xmm0, xmmword ptr [edi]
+// CHECK: encoding: [0x62,0xf5,0x7d,0x08,0x3a,0x07]
+          vcvtrops2hf8s xmm0, xmmword ptr [edi]
+
+// CHECK: vcvtrops2hf8s xmm0, dword ptr [edi]{1to16}
+// CHECK: encoding: [0x62,0xf5,0x7d,0x58,0x3a,0x07]
+          vcvtrops2hf8s xmm0, dword ptr [edi]{1to16}
+
 //
 // Group B: Bias PS->8bit conversions (3-operand)
 //
@@ -114,6 +250,26 @@
 // CHECK: encoding: [0x62,0xf5,0x74,0x08,0x39,0xc2]
           vcvtbiasps2bf8 xmm0, xmm1, xmm2
 
+// CHECK: vcvtbiasps2bf8 xmm0 {k1}, zmm1, zmm2
+// CHECK: encoding: [0x62,0xf5,0x74,0x49,0x39,0xc2]
+          vcvtbiasps2bf8 xmm0 {k1}, zmm1, zmm2
+
+// CHECK: vcvtbiasps2bf8 xmm0 {k1} {z}, zmm1, zmm2
+// CHECK: encoding: [0x62,0xf5,0x74,0xc9,0x39,0xc2]
+          vcvtbiasps2bf8 xmm0 {k1} {z}, zmm1, zmm2
+
+// CHECK: vcvtbiasps2bf8 xmm0, zmm1, zmmword ptr [edi]
+// CHECK: encoding: [0x62,0xf5,0x74,0x48,0x39,0x07]
+          vcvtbiasps2bf8 xmm0, zmm1, zmmword ptr [edi]
+
+// CHECK: vcvtbiasps2bf8 xmm0, ymm1, ymmword ptr [edi]
+// CHECK: encoding: [0x62,0xf5,0x74,0x28,0x39,0x07]
+          vcvtbiasps2bf8 xmm0, ymm1, ymmword ptr [edi]
+
+// CHECK: vcvtbiasps2bf8 xmm0, xmm1, xmmword ptr [edi]
+// CHECK: encoding: [0x62,0xf5,0x74,0x08,0x39,0x07]
+          vcvtbiasps2bf8 xmm0, xmm1, xmmword ptr [edi]
+
 // vcvtbiasps2bf8s
 
 // CHECK: vcvtbiasps2bf8s xmm0, zmm1, zmm2
@@ -128,6 +284,26 @@
 // CHECK: encoding: [0x62,0xf5,0x74,0x08,0x3b,0xc2]
           vcvtbiasps2bf8s xmm0, xmm1, xmm2
 
+// CHECK: vcvtbiasps2bf8s xmm0 {k1}, zmm1, zmm2
+// CHECK: encoding: [0x62,0xf5,0x74,0x49,0x3b,0xc2]
+          vcvtbiasps2bf8s xmm0 {k1}, zmm1, zmm2
+
+// CHECK: vcvtbiasps2bf8s xmm0 {k1} {z}, zmm1, zmm2
+// CHECK: encoding: [0x62,0xf5,0x74,0xc9,0x3b,0xc2]
+          vcvtbiasps2bf8s xmm0 {k1} {z}, zmm1, zmm2
+
+// CHECK: vcvtbiasps2bf8s xmm0, zmm1, zmmword ptr [edi]
+// CHECK: encoding: [0x62,0xf5,0x74,0x48,0x3b,0x07]
+          vcvtbiasps2bf8s xmm0, zmm1, zmmword ptr [edi]
+
+// CHECK: vcvtbiasps2bf8s xmm0, ymm1, ymmword ptr [edi]
+// CHECK: encoding: [0x62,0xf5,0x74,0x28,0x3b,0x07]
+          vcvtbiasps2bf8s xmm0, ymm1, ymmword ptr [edi]
+
+// CHECK: vcvtbiasps2bf8s xmm0, xmm1, xmmword ptr [edi]
+// CHECK: encoding: [0x62,0xf5,0x74,0x08,0x3b,0x07]
+          vcvtbiasps2bf8s xmm0, xmm1, xmmword ptr [edi]
+
 // vcvtbiasps2hf8
 
 // CHECK: vcvtbiasps2hf8 xmm0, zmm1, zmm2
@@ -142,6 +318,26 @@
 // CHECK: encoding: [0x62,0xf5,0x74,0x08,0x38,0xc2]
           vcvtbiasps2hf8 xmm0, xmm1, xmm2
 
+// CHECK: vcvtbiasps2hf8 xmm0 {k1}, zmm1, zmm2
+// CHECK: encoding: [0x62,0xf5,0x74,0x49,0x38,0xc2]
+          vcvtbiasps2hf8 xmm0 {k1}, zmm1, zmm2
+
+// CHECK: vcvtbiasps2hf8 xmm0 {k1} {z}, zmm1, zmm2
+// CHECK: encoding: [0x62,0xf5,0x74,0xc9,0x38,0xc2]
+          vcvtbiasps2hf8 xmm0 {k1} {z}, zmm1, zmm2
+
+// CHECK: vcvtbiasps2hf8 xmm0, zmm1, zmmword ptr [edi]
+// CHECK: encoding: [0x62,0xf5,0x74,0x48,0x38,0x07]
+          vcvtbiasps2hf8 xmm0, zmm1, zmmword ptr [edi]
+
+// CHECK: vcvtbiasps2hf8 xmm0, ymm1, ymmword ptr [edi]
+// CHECK: encoding: [0x62,0xf5,0x74,0x28,0x38,0x07]
+          vcvtbiasps2hf8 xmm0, ymm1, ymmword ptr [edi]
+
+// CHECK: vcvtbiasps2hf8 xmm0, xmm1, xmmword ptr [edi]
+// CHECK: encoding: [0x62,0xf5,0x74,0x08,0x38,0x07]
+          vcvtbiasps2hf8 xmm0, xmm1, xmmword ptr [edi]
+
 // vcvtbiasps2hf8s
 
 // CHECK: vcvtbiasps2hf8s xmm0, zmm1, zmm2
@@ -156,6 +352,26 @@
 // CHECK: encoding: [0x62,0xf5,0x74,0x08,0x3a,0xc2]
           vcvtbiasps2hf8s xmm0, xmm1, xmm2
 
+// CHECK: vcvtbiasps2hf8s xmm0 {k1}, zmm1, zmm2
+// CHECK: encoding: [0x62,0xf5,0x74,0x49,0x3a,0xc2]
+          vcvtbiasps2hf8s xmm0 {k1}, zmm1, zmm2
+
+// CHECK: vcvtbiasps2hf8s xmm0 {k1} {z}, zmm1, zmm2
+// CHECK: encoding: [0x62,0xf5,0x74,0xc9,0x3a,0xc2]
+          vcvtbiasps2hf8s xmm0 {k1} {z}, zmm1, zmm2
+
+// CHECK: vcvtbiasps2hf8s xmm0, zmm1, zmmword ptr [edi]
+// CHECK: encoding: [0x62,0xf5,0x74,0x48,0x3a,0x07]
+          vcvtbiasps2hf8s xmm0, zmm1, zmmword ptr [edi]
+
+// CHECK: vcvtbiasps2hf8s xmm0, ymm1, ymmword ptr [edi]
+// CHECK: encoding: [0x62,0xf5,0x74,0x28,0x3a,0x07]
+          vcvtbiasps2hf8s xmm0, ymm1, ymmword ptr [edi]
+
+// CHECK: vcvtbiasps2hf8s xmm0, xmm1, xmmword ptr [edi]
+// CHECK: encoding: [0x62,0xf5,0x74,0x08,0x3a,0x07]
+          vcvtbiasps2hf8s xmm0, xmm1, xmmword ptr [edi]
+
 //
 // Group C: 8bit->PS expanding conversions
 //
@@ -174,6 +390,26 @@
 // CHECK: encoding: [0x62,0xf5,0xfc,0x08,0x36,0xc1]
           vcvtbf82ps xmm0, xmm1
 
+// CHECK: vcvtbf82ps zmm0 {k1}, xmm1
+// CHECK: encoding: [0x62,0xf5,0xfc,0x49,0x36,0xc1]
+          vcvtbf82ps zmm0 {k1}, xmm1
+
+// CHECK: vcvtbf82ps zmm0 {k1} {z}, xmm1
+// CHECK: encoding: [0x62,0xf5,0xfc,0xc9,0x36,0xc1]
+          vcvtbf82ps zmm0 {k1} {z}, xmm1
+
+// CHECK: vcvtbf82ps zmm0, xmmword ptr [edi]
+// CHECK: encoding: [0x62,0xf5,0xfc,0x48,0x36,0x07]
+          vcvtbf82ps zmm0, xmmword ptr [edi]
+
+// CHECK: vcvtbf82ps ymm0, qword ptr [edi]
+// CHECK: encoding: [0x62,0xf5,0xfc,0x28,0x36,0x07]
+          vcvtbf82ps ymm0, qword ptr [edi]
+
+// CHECK: vcvtbf82ps xmm0, dword ptr [edi]
+// CHECK: encoding: [0x62,0xf5,0xfc,0x08,0x36,0x07]
+          vcvtbf82ps xmm0, dword ptr [edi]
+
 // vcvthf82ps
 
 // CHECK: vcvthf82ps zmm0, xmm1
@@ -188,6 +424,26 @@
 // CHECK: encoding: [0x62,0xf5,0x7c,0x08,0x36,0xc1]
           vcvthf82ps xmm0, xmm1
 
+// CHECK: vcvthf82ps zmm0 {k1}, xmm1
+// CHECK: encoding: [0x62,0xf5,0x7c,0x49,0x36,0xc1]
+          vcvthf82ps zmm0 {k1}, xmm1
+
+// CHECK: vcvthf82ps zmm0 {k1} {z}, xmm1
+// CHECK: encoding: [0x62,0xf5,0x7c,0xc9,0x36,0xc1]
+          vcvthf82ps zmm0 {k1} {z}, xmm1
+
+// CHECK: vcvthf82ps zmm0, xmmword ptr [edi]
+// CHECK: encoding: [0x62,0xf5,0x7c,0x48,0x36,0x07]
+          vcvthf82ps zmm0, xmmword ptr [edi]
+
+// CHECK: vcvthf82ps ymm0, qword ptr [edi]
+// CHECK: encoding: [0x62,0xf5,0x7c,0x28,0x36,0x07]
+          vcvthf82ps ymm0, qword ptr [edi]
+
+// CHECK: vcvthf82ps xmm0, dword ptr [edi]
+// CHECK: encoding: [0x62,0xf5,0x7c,0x08,0x36,0x07]
+          vcvthf82ps xmm0, dword ptr [edi]
+
 //
 // Group D: BF8/HF8->BF4S truncations
 //
@@ -206,6 +462,18 @@
 // CHECK: encoding: [0x62,0xf5,0xfe,0x08,0x3d,0xc8]
           vcvtbf82bf4s xmm0, xmm1
 
+// CHECK: vcvtbf82bf4s ymmword ptr [edi], zmm1
+// CHECK: encoding: [0x62,0xf5,0xfe,0x48,0x3d,0x0f]
+          vcvtbf82bf4s ymmword ptr [edi], zmm1
+
+// CHECK: vcvtbf82bf4s xmmword ptr [edi], ymm1
+// CHECK: encoding: [0x62,0xf5,0xfe,0x28,0x3d,0x0f]
+          vcvtbf82bf4s xmmword ptr [edi], ymm1
+
+// CHECK: vcvtbf82bf4s qword ptr [edi], xmm1
+// CHECK: encoding: [0x62,0xf5,0xfe,0x08,0x3d,0x0f]
+          vcvtbf82bf4s qword ptr [edi], xmm1
+
 // vcvthf82bf4s
 
 // CHECK: vcvthf82bf4s ymm0, zmm1
@@ -220,6 +488,18 @@
 // CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x3d,0xc8]
           vcvthf82bf4s xmm0, xmm1
 
+// CHECK: vcvthf82bf4s ymmword ptr [edi], zmm1
+// CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x3d,0x0f]
+          vcvthf82bf4s ymmword ptr [edi], zmm1
+
+// CHECK: vcvthf82bf4s xmmword ptr [edi], ymm1
+// CHECK: encoding: [0x62,0xf5,0x7e,0x28,0x3d,0x0f]
+          vcvthf82bf4s xmmword ptr [edi], ymm1
+
+// CHECK: vcvthf82bf4s qword ptr [edi], xmm1
+// CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x3d,0x0f]
+          vcvthf82bf4s qword ptr [edi], xmm1
+
 //
 // Group E: Same-size reg-only conversions (no masking)
 //
@@ -270,6 +550,26 @@
 // CHECK: encoding: [0x62,0xf5,0x7c,0x08,0x37,0xc1]
           vcvtbf42hf8 xmm0, xmm1
 
+// CHECK: vcvtbf42hf8 zmm0 {k1}, ymm1
+// CHECK: encoding: [0x62,0xf5,0x7c,0x49,0x37,0xc1]
+          vcvtbf42hf8 zmm0 {k1}, ymm1
+
+// CHECK: vcvtbf42hf8 zmm0 {k1} {z}, ymm1
+// CHECK: encoding: [0x62,0xf5,0x7c,0xc9,0x37,0xc1]
+          vcvtbf42hf8 zmm0 {k1} {z}, ymm1
+
+// CHECK: vcvtbf42hf8 zmm0, ymmword ptr [edi]
+// CHECK: encoding: [0x62,0xf5,0x7c,0x48,0x37,0x07]
+          vcvtbf42hf8 zmm0, ymmword ptr [edi]
+
+// CHECK: vcvtbf42hf8 ymm0, xmmword ptr [edi]
+// CHECK: encoding: [0x62,0xf5,0x7c,0x28,0x37,0x07]
+          vcvtbf42hf8 ymm0, xmmword ptr [edi]
+
+// CHECK: vcvtbf42hf8 xmm0, qword ptr [edi]
+// CHECK: encoding: [0x62,0xf5,0x7c,0x08,0x37,0x07]
+          vcvtbf42hf8 xmm0, qword ptr [edi]
+
 // vcvtbf62hf8
 
 // CHECK: vcvtbf62hf8 zmm0, zmm1
@@ -284,6 +584,14 @@
 // CHECK: encoding: [0x62,0xf5,0xfd,0x08,0x37,0xc1]
           vcvtbf62hf8 xmm0, xmm1
 
+// CHECK: vcvtbf62hf8 zmm0 {k1}, zmm1
+// CHECK: encoding: [0x62,0xf5,0xfd,0x49,0x37,0xc1]
+          vcvtbf62hf8 zmm0 {k1}, zmm1
+
+// CHECK: vcvtbf62hf8 zmm0 {k1} {z}, zmm1
+// CHECK: encoding: [0x62,0xf5,0xfd,0xc9,0x37,0xc1]
+          vcvtbf62hf8 zmm0 {k1} {z}, zmm1
+
 // vcvthf62hf8
 
 // CHECK: vcvthf62hf8 zmm0, zmm1
@@ -298,6 +606,14 @@
 // CHECK: encoding: [0x62,0xf5,0x7d,0x08,0x37,0xc1]
           vcvthf62hf8 xmm0, xmm1
 
+// CHECK: vcvthf62hf8 zmm0 {k1}, zmm1
+// CHECK: encoding: [0x62,0xf5,0x7d,0x49,0x37,0xc1]
+          vcvthf62hf8 zmm0 {k1}, zmm1
+
+// CHECK: vcvthf62hf8 zmm0 {k1} {z}, zmm1
+// CHECK: encoding: [0x62,0xf5,0x7d,0xc9,0x37,0xc1]
+          vcvthf62hf8 zmm0 {k1} {z}, zmm1
+
 //
 // Group G: VPMOVSSDB - Integer DWord->Byte signed saturation
 //
@@ -314,6 +630,26 @@
 // CHECK: encoding: [0x62,0xf2,0x7e,0x08,0x41,0xc8]
           vpmovssdb xmm0, xmm1
 
+// CHECK: vpmovssdb xmm0 {k1}, zmm1
+// CHECK: encoding: [0x62,0xf2,0x7e,0x49,0x41,0xc8]
+          vpmovssdb xmm0 {k1}, zmm1
+
+// CHECK: vpmovssdb xmm0 {k1} {z}, zmm1
+// CHECK: encoding: [0x62,0xf2,0x7e,0xc9,0x41,0xc8]
+          vpmovssdb xmm0 {k1} {z}, zmm1
+
+// CHECK: vpmovssdb xmmword ptr [edi], zmm1
+// CHECK: encoding: [0x62,0xf2,0x7e,0x48,0x41,0x0f]
+          vpmovssdb xmmword ptr [edi], zmm1
+
+// CHECK: vpmovssdb qword ptr [edi], ymm1
+// CHECK: encoding: [0x62,0xf2,0x7e,0x28,0x41,0x0f]
+          vpmovssdb qword ptr [edi], ymm1
+
+// CHECK: vpmovssdb dword ptr [edi], xmm1
+// CHECK: encoding: [0x62,0xf2,0x7e,0x08,0x41,0x0f]
+          vpmovssdb dword ptr [edi], xmm1
+
 //
 // Group H: VUNPACKB - Byte unpack with immediate
 //
@@ -329,3 +665,23 @@
 // CHECK: vunpackb xmm0, xmm1, 1
 // CHECK: encoding: [0x62,0xf3,0x7c,0x08,0x3d,0xc1,0x01]
           vunpackb xmm0, xmm1, 1
+
+// CHECK: vunpackb zmm0 {k1}, zmm1, 1
+// CHECK: encoding: [0x62,0xf3,0x7c,0x49,0x3d,0xc1,0x01]
+          vunpackb zmm0 {k1}, zmm1, 1
+
+// CHECK: vunpackb zmm0 {k1} {z}, zmm1, 1
+// CHECK: encoding: [0x62,0xf3,0x7c,0xc9,0x3d,0xc1,0x01]
+          vunpackb zmm0 {k1} {z}, zmm1, 1
+
+// CHECK: vunpackb zmm0, zmmword ptr [edi], 1
+// CHECK: encoding: [0x62,0xf3,0x7c,0x48,0x3d,0x07,0x01]
+          vunpackb zmm0, zmmword ptr [edi], 1
+
+// CHECK: vunpackb ymm0, ymmword ptr [edi], 1
+// CHECK: encoding: [0x62,0xf3,0x7c,0x28,0x3d,0x07,0x01]
+          vunpackb ymm0, ymmword ptr [edi], 1
+
+// CHECK: vunpackb xmm0, xmmword ptr [edi], 1
+// CHECK: encoding: [0x62,0xf3,0x7c,0x08,0x3d,0x07,0x01]
+          vunpackb xmm0, xmmword ptr [edi], 1
diff --git a/llvm/test/MC/X86/avx10_v2_aux-intel-64.s b/llvm/test/MC/X86/avx10_v2_aux-intel-64.s
index 679a8ddc3afe57..b5bc7360ffffa5 100644
--- a/llvm/test/MC/X86/avx10_v2_aux-intel-64.s
+++ b/llvm/test/MC/X86/avx10_v2_aux-intel-64.s
@@ -1,4 +1,4 @@
-// RUN: llvm-mc -triple x86_64 -x86-asm-syntax=intel -output-asm-variant=1 --show-encoding -mattr=+avx10-v2-aux,+avx512vl %s | FileCheck %s
+// RUN: llvm-mc -triple x86_64 -x86-asm-syntax=intel -output-asm-variant=1 --show-encoding -mattr=+avx10v2aux,+avx512vl %s | FileCheck %s
 
 //
 // Group A: PS->8bit truncating conversions
diff --git a/llvm/test/TableGen/x86-fold-tables.inc b/llvm/test/TableGen/x86-fold-tables.inc
index 64f7ff432e83aa..acdd7b7af9f5dd 100644
--- a/llvm/test/TableGen/x86-fold-tables.inc
+++ b/llvm/test/TableGen/x86-fold-tables.inc
@@ -328,9 +328,9 @@ static const X86FoldTableEntry Table2Addr[] = {
   {X86::SUB8ri_NF, X86::SUB8mi_NF, TB_NO_REVERSE},
   {X86::SUB8rr, X86::SUB8mr, TB_NO_REVERSE},
   {X86::SUB8rr_NF, X86::SUB8mr_NF, TB_NO_REVERSE},
-  {X86::VPMOVSSDZ128rrk, X86::VPMOVSSDZ128mrk, TB_NO_REVERSE},
-  {X86::VPMOVSSDZ256rrk, X86::VPMOVSSDZ256mrk, TB_NO_REVERSE},
-  {X86::VPMOVSSDZrrk, X86::VPMOVSSDZmrk, TB_NO_REVERSE},
+  {X86::VPMOVSSDBZ128rrk, X86::VPMOVSSDBZ128mrk, TB_NO_REVERSE},
+  {X86::VPMOVSSDBZ256rrk, X86::VPMOVSSDBZ256mrk, TB_NO_REVERSE},
+  {X86::VPMOVSSDBZrrk, X86::VPMOVSSDBZmrk, TB_NO_REVERSE},
   {X86::XOR16ri, X86::XOR16mi, TB_NO_REVERSE},
   {X86::XOR16ri8, X86::XOR16mi8, TB_NO_REVERSE},
   {X86::XOR16ri8_NF, X86::XOR16mi8_NF, TB_NO_REVERSE},
@@ -579,7 +579,7 @@ static const X86FoldTableEntry Table0[] = {
   {X86::VPMOVSQDZ256rr, X86::VPMOVSQDZ256mr, TB_FOLDED_STORE},
   {X86::VPMOVSQDZrr, X86::VPMOVSQDZmr, TB_FOLDED_STORE},
   {X86::VPMOVSQWZrr, X86::VPMOVSQWZmr, TB_FOLDED_STORE},
-  {X86::VPMOVSSDZrr, X86::VPMOVSSDZmr, TB_FOLDED_STORE},
+  {X86::VPMOVSSDBZrr, X86::VPMOVSSDBZmr, TB_FOLDED_STORE},
   {X86::VPMOVSWBZ256rr, X86::VPMOVSWBZ256mr, TB_FOLDED_STORE},
   {X86::VPMOVSWBZrr, X86::VPMOVSWBZmr, TB_FOLDED_STORE},
   {X86::VPMOVUSDBZrr, X86::VPMOVUSDBZmr, TB_FOLDED_STORE},
@@ -1998,9 +1998,9 @@ static const X86FoldTableEntry Table1[] = {
   {X86::VUCOMXSHZrr_Int, X86::VUCOMXSHZrm_Int, TB_NO_REVERSE},
   {X86::VUCOMXSSZrr, X86::VUCOMXSSZrm, 0},
   {X86::VUCOMXSSZrr_Int, X86::VUCOMXSSZrm_Int, TB_NO_REVERSE},
-  {X86::VUNPACKBZ128ri, X86::VUNPACKBZ128mi, 0},
-  {X86::VUNPACKBZ256ri, X86::VUNPACKBZ256mi, 0},
-  {X86::VUNPACKBZri, X86::VUNPACKBZmi, 0},
+  {X86::VUNPACKBZ128rri, X86::VUNPACKBZ128rmi, 0},
+  {X86::VUNPACKBZ256rri, X86::VUNPACKBZ256rmi, 0},
+  {X86::VUNPACKBZrri, X86::VUNPACKBZrmi, 0},
   {X86::XOR16ri8_ND, X86::XOR16mi8_ND, 0},
   {X86::XOR16ri8_NF_ND, X86::XOR16mi8_NF_ND, 0},
   {X86::XOR16ri_ND, X86::XOR16mi_ND, 0},
@@ -4257,9 +4257,9 @@ static const X86FoldTableEntry Table2[] = {
   {X86::VSUBSSZrr_Int, X86::VSUBSSZrm_Int, TB_NO_REVERSE},
   {X86::VSUBSSrr, X86::VSUBSSrm, 0},
   {X86::VSUBSSrr_Int, X86::VSUBSSrm_Int, TB_NO_REVERSE},
-  {X86::VUNPACKBZ128rikz, X86::VUNPACKBZ128mikz, 0},
-  {X86::VUNPACKBZ256rikz, X86::VUNPACKBZ256mikz, 0},
-  {X86::VUNPACKBZrikz, X86::VUNPACKBZmikz, 0},
+  {X86::VUNPACKBZ128rrikz, X86::VUNPACKBZ128rmikz, 0},
+  {X86::VUNPACKBZ256rrikz, X86::VUNPACKBZ256rmikz, 0},
+  {X86::VUNPACKBZrrikz, X86::VUNPACKBZrmikz, 0},
   {X86::VUNPCKHPDYrr, X86::VUNPCKHPDYrm, 0},
   {X86::VUNPCKHPDZ128rr, X86::VUNPCKHPDZ128rm, 0},
   {X86::VUNPCKHPDZ256rr, X86::VUNPCKHPDZ256rm, 0},
@@ -6179,9 +6179,9 @@ static const X86FoldTableEntry Table3[] = {
   {X86::VSUBSDZrrkz_Int, X86::VSUBSDZrmkz_Int, TB_NO_REVERSE},
   {X86::VSUBSHZrrkz_Int, X86::VSUBSHZrmkz_Int, TB_NO_REVERSE},
   {X86::VSUBSSZrrkz_Int, X86::VSUBSSZrmkz_Int, TB_NO_REVERSE},
-  {X86::VUNPACKBZ128rik, X86::VUNPACKBZ128mik, 0},
-  {X86::VUNPACKBZ256rik, X86::VUNPACKBZ256mik, 0},
-  {X86::VUNPACKBZrik, X86::VUNPACKBZmik, 0},
+  {X86::VUNPACKBZ128rrik, X86::VUNPACKBZ128rmik, 0},
+  {X86::VUNPACKBZ256rrik, X86::VUNPACKBZ256rmik, 0},
+  {X86::VUNPACKBZrrik, X86::VUNPACKBZrmik, 0},
   {X86::VUNPCKHPDZ128rrkz, X86::VUNPCKHPDZ128rmkz, 0},
   {X86::VUNPCKHPDZ256rrkz, X86::VUNPCKHPDZ256rmkz, 0},
   {X86::VUNPCKHPDZrrkz, X86::VUNPCKHPDZrmkz, 0},

>From c526d95964756ebd7d1b4b6dc7cad8d6d993ff74 Mon Sep 17 00:00:00 2001
From: Ganesh Gopalasubramanian <Ganesh.Gopalasubramanian at amd.com>
Date: Wed, 15 Jul 2026 22:08:47 +0530
Subject: [PATCH 03/16] Use VMTRUNCSS 512 patterns for VPMOVSSDB and add tests.
 Fix format errors.

---
 clang/lib/Headers/avx10_2_v2auxintrin.h       | 121 ++++++++----------
 llvm/lib/Target/X86/X86InstrAVX10_V2_AUX.td   |   9 ++
 .../CodeGen/X86/avx10_v2aux-intrinsics.ll     |  89 +++++++++++++
 3 files changed, 149 insertions(+), 70 deletions(-)

diff --git a/clang/lib/Headers/avx10_2_v2auxintrin.h b/clang/lib/Headers/avx10_2_v2auxintrin.h
index 2214b8cead4c25..e852ab365a5d88 100644
--- a/clang/lib/Headers/avx10_2_v2auxintrin.h
+++ b/clang/lib/Headers/avx10_2_v2auxintrin.h
@@ -37,8 +37,7 @@
 /// \returns
 ///    A 128-bit vector of [16 x i8]. The lower 4 bytes contain the converted
 ///    values; the upper bytes are zeroed.
-static __inline__ __m128i __DEFAULT_FN_ATTRS128
-_mm_cvtps_bf8(__m128 __A) {
+static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvtps_bf8(__m128 __A) {
   return (__m128i)__builtin_ia32_vcvtps2bf8_128((__v4sf)__A);
 }
 
@@ -58,8 +57,9 @@ _mm_cvtps_bf8(__m128 __A) {
 ///    A 128-bit vector of [4 x float].
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted values.
-static __inline__ __m128i __DEFAULT_FN_ATTRS128
-_mm_mask_cvtps_bf8(__m128i __W, __mmask8 __U, __m128 __A) {
+static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_mask_cvtps_bf8(__m128i __W,
+                                                                   __mmask8 __U,
+                                                                   __m128 __A) {
   return (__m128i)__builtin_ia32_selectb_128(
       (__mmask16)__U, (__v16qi)_mm_cvtps_bf8(__A), (__v16qi)__W);
 }
@@ -98,8 +98,7 @@ _mm_maskz_cvtps_bf8(__mmask8 __U, __m128 __A) {
 /// \returns
 ///    A 128-bit vector of [16 x i8]. The lower 8 bytes contain the converted
 ///    values; the upper bytes are zeroed.
-static __inline__ __m128i __DEFAULT_FN_ATTRS256
-_mm256_cvtps_bf8(__m256 __A) {
+static __inline__ __m128i __DEFAULT_FN_ATTRS256 _mm256_cvtps_bf8(__m256 __A) {
   return (__m128i)__builtin_ia32_vcvtps2bf8_256((__v8sf)__A);
 }
 
@@ -158,8 +157,7 @@ _mm256_maskz_cvtps_bf8(__mmask8 __U, __m256 __A) {
 ///    A 128-bit vector of [4 x float].
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted values.
-static __inline__ __m128i __DEFAULT_FN_ATTRS128
-_mm_cvts_ps_bf8(__m128 __A) {
+static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvts_ps_bf8(__m128 __A) {
   return (__m128i)__builtin_ia32_vcvtps2bf8s_128((__v4sf)__A);
 }
 
@@ -218,8 +216,7 @@ _mm_maskz_cvts_ps_bf8(__mmask8 __U, __m128 __A) {
 ///    A 256-bit vector of [8 x float].
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted values.
-static __inline__ __m128i __DEFAULT_FN_ATTRS256
-_mm256_cvts_ps_bf8(__m256 __A) {
+static __inline__ __m128i __DEFAULT_FN_ATTRS256 _mm256_cvts_ps_bf8(__m256 __A) {
   return (__m128i)__builtin_ia32_vcvtps2bf8s_256((__v8sf)__A);
 }
 
@@ -278,8 +275,7 @@ _mm256_maskz_cvts_ps_bf8(__mmask8 __U, __m256 __A) {
 ///    A 128-bit vector of [4 x float].
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted values.
-static __inline__ __m128i __DEFAULT_FN_ATTRS128
-_mm_cvtps_hf8(__m128 __A) {
+static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvtps_hf8(__m128 __A) {
   return (__m128i)__builtin_ia32_vcvtps2hf8_128((__v4sf)__A);
 }
 
@@ -299,8 +295,9 @@ _mm_cvtps_hf8(__m128 __A) {
 ///    A 128-bit vector of [4 x float].
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted values.
-static __inline__ __m128i __DEFAULT_FN_ATTRS128
-_mm_mask_cvtps_hf8(__m128i __W, __mmask8 __U, __m128 __A) {
+static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_mask_cvtps_hf8(__m128i __W,
+                                                                   __mmask8 __U,
+                                                                   __m128 __A) {
   return (__m128i)__builtin_ia32_selectb_128(
       (__mmask16)__U, (__v16qi)_mm_cvtps_hf8(__A), (__v16qi)__W);
 }
@@ -338,8 +335,7 @@ _mm_maskz_cvtps_hf8(__mmask8 __U, __m128 __A) {
 ///    A 256-bit vector of [8 x float].
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted values.
-static __inline__ __m128i __DEFAULT_FN_ATTRS256
-_mm256_cvtps_hf8(__m256 __A) {
+static __inline__ __m128i __DEFAULT_FN_ATTRS256 _mm256_cvtps_hf8(__m256 __A) {
   return (__m128i)__builtin_ia32_vcvtps2hf8_256((__v8sf)__A);
 }
 
@@ -398,8 +394,7 @@ _mm256_maskz_cvtps_hf8(__mmask8 __U, __m256 __A) {
 ///    A 128-bit vector of [4 x float].
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted values.
-static __inline__ __m128i __DEFAULT_FN_ATTRS128
-_mm_cvts_ps_hf8(__m128 __A) {
+static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvts_ps_hf8(__m128 __A) {
   return (__m128i)__builtin_ia32_vcvtps2hf8s_128((__v4sf)__A);
 }
 
@@ -458,8 +453,7 @@ _mm_maskz_cvts_ps_hf8(__mmask8 __U, __m128 __A) {
 ///    A 256-bit vector of [8 x float].
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted values.
-static __inline__ __m128i __DEFAULT_FN_ATTRS256
-_mm256_cvts_ps_hf8(__m256 __A) {
+static __inline__ __m128i __DEFAULT_FN_ATTRS256 _mm256_cvts_ps_hf8(__m256 __A) {
   return (__m128i)__builtin_ia32_vcvtps2hf8s_256((__v8sf)__A);
 }
 
@@ -518,8 +512,7 @@ _mm256_maskz_cvts_ps_hf8(__mmask8 __U, __m256 __A) {
 ///    A 128-bit vector of [4 x float].
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted values.
-static __inline__ __m128i __DEFAULT_FN_ATTRS128
-_mm_cvtrops_hf8(__m128 __A) {
+static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvtrops_hf8(__m128 __A) {
   return (__m128i)__builtin_ia32_vcvtrops2hf8_128((__v4sf)__A);
 }
 
@@ -578,8 +571,7 @@ _mm_maskz_cvtrops_hf8(__mmask8 __U, __m128 __A) {
 ///    A 256-bit vector of [8 x float].
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted values.
-static __inline__ __m128i __DEFAULT_FN_ATTRS256
-_mm256_cvtrops_hf8(__m256 __A) {
+static __inline__ __m128i __DEFAULT_FN_ATTRS256 _mm256_cvtrops_hf8(__m256 __A) {
   return (__m128i)__builtin_ia32_vcvtrops2hf8_256((__v8sf)__A);
 }
 
@@ -638,8 +630,7 @@ _mm256_maskz_cvtrops_hf8(__mmask8 __U, __m256 __A) {
 ///    A 128-bit vector of [4 x float].
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted values.
-static __inline__ __m128i __DEFAULT_FN_ATTRS128
-_mm_cvts_rops_hf8(__m128 __A) {
+static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvts_rops_hf8(__m128 __A) {
   return (__m128i)__builtin_ia32_vcvtrops2hf8s_128((__v4sf)__A);
 }
 
@@ -760,8 +751,8 @@ _mm256_maskz_cvts_rops_hf8(__mmask8 __U, __m256 __A) {
 ///    A 128-bit vector of [4 x float].
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted values.
-static __inline__ __m128i __DEFAULT_FN_ATTRS128
-_mm_cvtbiasps_bf8(__m128i __A, __m128 __B) {
+static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvtbiasps_bf8(__m128i __A,
+                                                                  __m128 __B) {
   return (__m128i)__builtin_ia32_vcvtbiasps2bf8_128((__v16qi)__A, (__v4sf)__B);
 }
 
@@ -1024,8 +1015,8 @@ _mm256_maskz_cvts_biasps_bf8(__mmask8 __U, __m256i __A, __m256 __B) {
 ///    A 128-bit vector of [4 x float].
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted values.
-static __inline__ __m128i __DEFAULT_FN_ATTRS128
-_mm_cvtbiasps_hf8(__m128i __A, __m128 __B) {
+static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvtbiasps_hf8(__m128i __A,
+                                                                  __m128 __B) {
   return (__m128i)__builtin_ia32_vcvtbiasps2hf8_128((__v16qi)__A, (__v4sf)__B);
 }
 
@@ -1286,8 +1277,7 @@ _mm256_maskz_cvts_biasps_hf8(__mmask8 __U, __m256i __A, __m256 __B) {
 ///    A 128-bit vector of [16 x i8] containing BF8 values.
 /// \returns
 ///    A 128-bit vector of [4 x float] containing the converted values.
-static __inline__ __m128 __DEFAULT_FN_ATTRS128
-_mm_cvtbf8_ps(__m128i __A) {
+static __inline__ __m128 __DEFAULT_FN_ATTRS128 _mm_cvtbf8_ps(__m128i __A) {
   return (__m128)__builtin_ia32_vcvtbf8_2ps128((__v16qi)__A);
 }
 
@@ -1307,8 +1297,9 @@ _mm_cvtbf8_ps(__m128i __A) {
 ///    A 128-bit vector of [16 x i8] containing BF8 values.
 /// \returns
 ///    A 128-bit vector of [4 x float] containing the converted values.
-static __inline__ __m128 __DEFAULT_FN_ATTRS128
-_mm_mask_cvtbf8_ps(__m128 __W, __mmask8 __U, __m128i __A) {
+static __inline__ __m128 __DEFAULT_FN_ATTRS128 _mm_mask_cvtbf8_ps(__m128 __W,
+                                                                  __mmask8 __U,
+                                                                  __m128i __A) {
   return (__m128)__builtin_ia32_selectps_128(
       (__mmask8)__U, (__v4sf)_mm_cvtbf8_ps(__A), (__v4sf)__W);
 }
@@ -1345,8 +1336,7 @@ _mm_maskz_cvtbf8_ps(__mmask8 __U, __m128i __A) {
 ///    A 128-bit vector of [16 x i8] containing BF8 values.
 /// \returns
 ///    A 256-bit vector of [8 x float] containing the converted values.
-static __inline__ __m256 __DEFAULT_FN_ATTRS256
-_mm256_cvtbf8_ps(__m128i __A) {
+static __inline__ __m256 __DEFAULT_FN_ATTRS256 _mm256_cvtbf8_ps(__m128i __A) {
   return (__m256)__builtin_ia32_vcvtbf8_2ps256((__v16qi)__A);
 }
 
@@ -1405,8 +1395,7 @@ _mm256_maskz_cvtbf8_ps(__mmask8 __U, __m128i __A) {
 ///    A 128-bit vector of [16 x i8] containing HF8 values.
 /// \returns
 ///    A 128-bit vector of [4 x float] containing the converted values.
-static __inline__ __m128 __DEFAULT_FN_ATTRS128
-_mm_cvthf8_ps(__m128i __A) {
+static __inline__ __m128 __DEFAULT_FN_ATTRS128 _mm_cvthf8_ps(__m128i __A) {
   return (__m128)__builtin_ia32_vcvthf8_2ps128((__v16qi)__A);
 }
 
@@ -1426,8 +1415,9 @@ _mm_cvthf8_ps(__m128i __A) {
 ///    A 128-bit vector of [16 x i8] containing HF8 values.
 /// \returns
 ///    A 128-bit vector of [4 x float] containing the converted values.
-static __inline__ __m128 __DEFAULT_FN_ATTRS128
-_mm_mask_cvthf8_ps(__m128 __W, __mmask8 __U, __m128i __A) {
+static __inline__ __m128 __DEFAULT_FN_ATTRS128 _mm_mask_cvthf8_ps(__m128 __W,
+                                                                  __mmask8 __U,
+                                                                  __m128i __A) {
   return (__m128)__builtin_ia32_selectps_128(
       (__mmask8)__U, (__v4sf)_mm_cvthf8_ps(__A), (__v4sf)__W);
 }
@@ -1464,8 +1454,7 @@ _mm_maskz_cvthf8_ps(__mmask8 __U, __m128i __A) {
 ///    A 128-bit vector of [16 x i8] containing HF8 values.
 /// \returns
 ///    A 256-bit vector of [8 x float] containing the converted values.
-static __inline__ __m256 __DEFAULT_FN_ATTRS256
-_mm256_cvthf8_ps(__m128i __A) {
+static __inline__ __m256 __DEFAULT_FN_ATTRS256 _mm256_cvthf8_ps(__m128i __A) {
   return (__m256)__builtin_ia32_vcvthf8_2ps256((__v16qi)__A);
 }
 
@@ -1524,8 +1513,7 @@ _mm256_maskz_cvthf8_ps(__mmask8 __U, __m128i __A) {
 ///    A 128-bit vector of [16 x i8] containing BF8 values.
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted BF6 values.
-static __inline__ __m128i __DEFAULT_FN_ATTRS128
-_mm_cvtbf8_bf6s(__m128i __A) {
+static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvtbf8_bf6s(__m128i __A) {
   return (__m128i)__builtin_ia32_vcvtbf82bf6s128((__v16qi)__A);
 }
 
@@ -1558,8 +1546,7 @@ _mm256_cvtbf8_bf6s(__m256i __A) {
 ///    A 128-bit vector of [16 x i8] containing HF8 values.
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted HF6 values.
-static __inline__ __m128i __DEFAULT_FN_ATTRS128
-_mm_cvthf8_hf6s(__m128i __A) {
+static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvthf8_hf6s(__m128i __A) {
   return (__m128i)__builtin_ia32_vcvthf82hf6s128((__v16qi)__A);
 }
 
@@ -1732,8 +1719,7 @@ _mm256_cvthf8_bf4s_storeu(void *__P, __m256i __A) {
 ///    A 128-bit vector of [16 x i8] containing BF4 values.
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted HF8 values.
-static __inline__ __m128i __DEFAULT_FN_ATTRS128
-_mm_cvtbf4_hf8(__m128i __A) {
+static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvtbf4_hf8(__m128i __A) {
   return (__m128i)__builtin_ia32_vcvtbf42hf8128((__v16qi)__A);
 }
 
@@ -1755,8 +1741,8 @@ _mm_cvtbf4_hf8(__m128i __A) {
 ///    A 128-bit vector of [16 x i8] containing the converted HF8 values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_mask_cvtbf4_hf8(__m128i __W, __mmask16 __U, __m128i __A) {
-  return (__m128i)__builtin_ia32_selectb_128(
-      __U, (__v16qi)_mm_cvtbf4_hf8(__A), (__v16qi)__W);
+  return (__m128i)__builtin_ia32_selectb_128(__U, (__v16qi)_mm_cvtbf4_hf8(__A),
+                                             (__v16qi)__W);
 }
 
 /// Convert packed BF4 (4-bit) floating-point elements in \a __A to packed
@@ -1775,8 +1761,8 @@ _mm_mask_cvtbf4_hf8(__m128i __W, __mmask16 __U, __m128i __A) {
 ///    A 128-bit vector of [16 x i8] containing the converted HF8 values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_maskz_cvtbf4_hf8(__mmask16 __U, __m128i __A) {
-  return (__m128i)__builtin_ia32_selectb_128(
-      __U, (__v16qi)_mm_cvtbf4_hf8(__A), (__v16qi)_mm_setzero_si128());
+  return (__m128i)__builtin_ia32_selectb_128(__U, (__v16qi)_mm_cvtbf4_hf8(__A),
+                                             (__v16qi)_mm_setzero_si128());
 }
 
 /// Convert packed BF4 (4-bit) floating-point elements in \a __A to packed
@@ -1791,8 +1777,7 @@ _mm_maskz_cvtbf4_hf8(__mmask16 __U, __m128i __A) {
 ///    A 128-bit vector of [16 x i8] containing BF4 values.
 /// \returns
 ///    A 256-bit vector of [32 x i8] containing the converted HF8 values.
-static __inline__ __m256i __DEFAULT_FN_ATTRS256
-_mm256_cvtbf4_hf8(__m128i __A) {
+static __inline__ __m256i __DEFAULT_FN_ATTRS256 _mm256_cvtbf4_hf8(__m128i __A) {
   return (__m256i)__builtin_ia32_vcvtbf42hf8256((__v16qi)__A);
 }
 
@@ -1850,8 +1835,7 @@ _mm256_maskz_cvtbf4_hf8(__mmask32 __U, __m128i __A) {
 ///    A 128-bit vector of [16 x i8] containing BF6 values.
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted HF8 values.
-static __inline__ __m128i __DEFAULT_FN_ATTRS128
-_mm_cvtbf6_hf8(__m128i __A) {
+static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvtbf6_hf8(__m128i __A) {
   return (__m128i)__builtin_ia32_vcvtbf62hf8128((__v16qi)__A);
 }
 
@@ -1873,8 +1857,8 @@ _mm_cvtbf6_hf8(__m128i __A) {
 ///    A 128-bit vector of [16 x i8] containing the converted HF8 values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_mask_cvtbf6_hf8(__m128i __W, __mmask16 __U, __m128i __A) {
-  return (__m128i)__builtin_ia32_selectb_128(
-      __U, (__v16qi)_mm_cvtbf6_hf8(__A), (__v16qi)__W);
+  return (__m128i)__builtin_ia32_selectb_128(__U, (__v16qi)_mm_cvtbf6_hf8(__A),
+                                             (__v16qi)__W);
 }
 
 /// Convert packed BF6 (6-bit) floating-point elements in \a __A to packed
@@ -1893,8 +1877,8 @@ _mm_mask_cvtbf6_hf8(__m128i __W, __mmask16 __U, __m128i __A) {
 ///    A 128-bit vector of [16 x i8] containing the converted HF8 values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_maskz_cvtbf6_hf8(__mmask16 __U, __m128i __A) {
-  return (__m128i)__builtin_ia32_selectb_128(
-      __U, (__v16qi)_mm_cvtbf6_hf8(__A), (__v16qi)_mm_setzero_si128());
+  return (__m128i)__builtin_ia32_selectb_128(__U, (__v16qi)_mm_cvtbf6_hf8(__A),
+                                             (__v16qi)_mm_setzero_si128());
 }
 
 /// Convert packed BF6 (6-bit) floating-point elements in \a __A to packed
@@ -1909,8 +1893,7 @@ _mm_maskz_cvtbf6_hf8(__mmask16 __U, __m128i __A) {
 ///    A 256-bit vector of [32 x i8] containing BF6 values.
 /// \returns
 ///    A 256-bit vector of [32 x i8] containing the converted HF8 values.
-static __inline__ __m256i __DEFAULT_FN_ATTRS256
-_mm256_cvtbf6_hf8(__m256i __A) {
+static __inline__ __m256i __DEFAULT_FN_ATTRS256 _mm256_cvtbf6_hf8(__m256i __A) {
   return (__m256i)__builtin_ia32_vcvtbf62hf8256((__v32qi)__A);
 }
 
@@ -1968,8 +1951,7 @@ _mm256_maskz_cvtbf6_hf8(__mmask32 __U, __m256i __A) {
 ///    A 128-bit vector of [16 x i8] containing HF6 values.
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted HF8 values.
-static __inline__ __m128i __DEFAULT_FN_ATTRS128
-_mm_cvthf6_hf8(__m128i __A) {
+static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvthf6_hf8(__m128i __A) {
   return (__m128i)__builtin_ia32_vcvthf62hf8128((__v16qi)__A);
 }
 
@@ -1991,8 +1973,8 @@ _mm_cvthf6_hf8(__m128i __A) {
 ///    A 128-bit vector of [16 x i8] containing the converted HF8 values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_mask_cvthf6_hf8(__m128i __W, __mmask16 __U, __m128i __A) {
-  return (__m128i)__builtin_ia32_selectb_128(
-      __U, (__v16qi)_mm_cvthf6_hf8(__A), (__v16qi)__W);
+  return (__m128i)__builtin_ia32_selectb_128(__U, (__v16qi)_mm_cvthf6_hf8(__A),
+                                             (__v16qi)__W);
 }
 
 /// Convert packed HF6 (6-bit) floating-point elements in \a __A to packed
@@ -2011,8 +1993,8 @@ _mm_mask_cvthf6_hf8(__m128i __W, __mmask16 __U, __m128i __A) {
 ///    A 128-bit vector of [16 x i8] containing the converted HF8 values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_maskz_cvthf6_hf8(__mmask16 __U, __m128i __A) {
-  return (__m128i)__builtin_ia32_selectb_128(
-      __U, (__v16qi)_mm_cvthf6_hf8(__A), (__v16qi)_mm_setzero_si128());
+  return (__m128i)__builtin_ia32_selectb_128(__U, (__v16qi)_mm_cvthf6_hf8(__A),
+                                             (__v16qi)_mm_setzero_si128());
 }
 
 /// Convert packed HF6 (6-bit) floating-point elements in \a __A to packed
@@ -2027,8 +2009,7 @@ _mm_maskz_cvthf6_hf8(__mmask16 __U, __m128i __A) {
 ///    A 256-bit vector of [32 x i8] containing HF6 values.
 /// \returns
 ///    A 256-bit vector of [32 x i8] containing the converted HF8 values.
-static __inline__ __m256i __DEFAULT_FN_ATTRS256
-_mm256_cvthf6_hf8(__m256i __A) {
+static __inline__ __m256i __DEFAULT_FN_ATTRS256 _mm256_cvthf6_hf8(__m256i __A) {
   return (__m256i)__builtin_ia32_vcvthf62hf8256((__v32qi)__A);
 }
 
diff --git a/llvm/lib/Target/X86/X86InstrAVX10_V2_AUX.td b/llvm/lib/Target/X86/X86InstrAVX10_V2_AUX.td
index c38094dec6df6e..f148cd552661cf 100644
--- a/llvm/lib/Target/X86/X86InstrAVX10_V2_AUX.td
+++ b/llvm/lib/Target/X86/X86InstrAVX10_V2_AUX.td
@@ -536,6 +536,15 @@ let Predicates = [HasAVX10_V2_AUX] in {
                                    SchedWriteVecTruncate, truncstore_ss_vi8,
                                    masked_truncstore_ss_vi8, X86vtruncss,
                                    X86vmtruncss>;
+
+  // Explicit patterns for 512-bit VMTRUNCSS (intrinsic lowering produces this
+  // SDNode directly, but avx512_trunc_db generates vselect_mask patterns for Z)
+  def : Pat<(v16i8 (X86vmtruncss (v16i32 VR512:$src), (v16i8 VR128X:$src0),
+                                  VK16WM:$mask)),
+            (VPMOVSSDBZrrk VR128X:$src0, VK16WM:$mask, VR512:$src)>;
+  def : Pat<(v16i8 (X86vmtruncss (v16i32 VR512:$src), v16i8x_info.ImmAllZerosV,
+                                  VK16WM:$mask)),
+            (VPMOVSSDBZrrkz VK16WM:$mask, VR512:$src)>;
 }
 
 //-------------------------------------------------
diff --git a/llvm/test/CodeGen/X86/avx10_v2aux-intrinsics.ll b/llvm/test/CodeGen/X86/avx10_v2aux-intrinsics.ll
index 53cfea524ce7f6..046ed29628a57d 100644
--- a/llvm/test/CodeGen/X86/avx10_v2aux-intrinsics.ll
+++ b/llvm/test/CodeGen/X86/avx10_v2aux-intrinsics.ll
@@ -1713,6 +1713,95 @@ declare <16 x i8> @llvm.x86.avx10.mask.pmovss.db.128(<4 x i32>, <16 x i8>, i8)
 declare <16 x i8> @llvm.x86.avx10.mask.pmovss.db.256(<8 x i32>, <16 x i8>, i8)
 declare <16 x i8> @llvm.x86.avx10.mask.pmovss.db.512(<16 x i32>, <16 x i8>, i16)
 
+; Masked tests for vpmovssdb
+define <16 x i8> @test_int_x86_avx10_mask_pmovssdb_128(<4 x i32> %a, <16 x i8> %passthru, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_pmovssdb_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vpmovssdb %xmm0, %xmm2 # encoding: [0x62,0xf2,0x7e,0x08,0x41,0xc2]
+; X64-NEXT:    vpmovssdb %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf2,0x7e,0x09,0x41,0xc1]
+; X64-NEXT:    vpaddb %xmm1, %xmm2, %xmm1 # EVEX TO VEX Compression encoding: [0xc5,0xe9,0xfc,0xc9]
+; X64-NEXT:    vpmovssdb %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf2,0x7e,0x89,0x41,0xc0]
+; X64-NEXT:    vpaddb %xmm0, %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf1,0xfc,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_pmovssdb_128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vpmovssdb %xmm0, %xmm2 # encoding: [0x62,0xf2,0x7e,0x08,0x41,0xc2]
+; X86-NEXT:    vpmovssdb %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf2,0x7e,0x09,0x41,0xc1]
+; X86-NEXT:    vpaddb %xmm1, %xmm2, %xmm1 # EVEX TO VEX Compression encoding: [0xc5,0xe9,0xfc,0xc9]
+; X86-NEXT:    vpmovssdb %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf2,0x7e,0x89,0x41,0xc0]
+; X86-NEXT:    vpaddb %xmm0, %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf1,0xfc,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %res0 = call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.128(<4 x i32> %a, <16 x i8> %passthru, i8 -1)
+  %res1 = call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.128(<4 x i32> %a, <16 x i8> %passthru, i8 %mask)
+  %res2 = call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.128(<4 x i32> %a, <16 x i8> zeroinitializer, i8 %mask)
+  %add1 = add <16 x i8> %res0, %res1
+  %add2 = add <16 x i8> %add1, %res2
+  ret <16 x i8> %add2
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_pmovssdb_256(<8 x i32> %a, <16 x i8> %passthru, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_pmovssdb_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vpmovssdb %ymm0, %xmm2 # encoding: [0x62,0xf2,0x7e,0x28,0x41,0xc2]
+; X64-NEXT:    vpmovssdb %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf2,0x7e,0x29,0x41,0xc1]
+; X64-NEXT:    vpaddb %xmm1, %xmm2, %xmm1 # EVEX TO VEX Compression encoding: [0xc5,0xe9,0xfc,0xc9]
+; X64-NEXT:    vpmovssdb %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf2,0x7e,0xa9,0x41,0xc0]
+; X64-NEXT:    vpaddb %xmm0, %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf1,0xfc,0xc0]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_pmovssdb_256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vpmovssdb %ymm0, %xmm2 # encoding: [0x62,0xf2,0x7e,0x28,0x41,0xc2]
+; X86-NEXT:    vpmovssdb %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf2,0x7e,0x29,0x41,0xc1]
+; X86-NEXT:    vpaddb %xmm1, %xmm2, %xmm1 # EVEX TO VEX Compression encoding: [0xc5,0xe9,0xfc,0xc9]
+; X86-NEXT:    vpmovssdb %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf2,0x7e,0xa9,0x41,0xc0]
+; X86-NEXT:    vpaddb %xmm0, %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf1,0xfc,0xc0]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %res0 = call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.256(<8 x i32> %a, <16 x i8> %passthru, i8 -1)
+  %res1 = call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.256(<8 x i32> %a, <16 x i8> %passthru, i8 %mask)
+  %res2 = call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.256(<8 x i32> %a, <16 x i8> zeroinitializer, i8 %mask)
+  %add1 = add <16 x i8> %res0, %res1
+  %add2 = add <16 x i8> %add1, %res2
+  ret <16 x i8> %add2
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_pmovssdb_512(<16 x i32> %a, <16 x i8> %passthru, i16 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_pmovssdb_512:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vpmovssdb %zmm0, %xmm2 # encoding: [0x62,0xf2,0x7e,0x48,0x41,0xc2]
+; X64-NEXT:    vpmovssdb %zmm0, %xmm1 {%k1} # encoding: [0x62,0xf2,0x7e,0x49,0x41,0xc1]
+; X64-NEXT:    vpaddb %xmm1, %xmm2, %xmm1 # EVEX TO VEX Compression encoding: [0xc5,0xe9,0xfc,0xc9]
+; X64-NEXT:    vpmovssdb %zmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf2,0x7e,0xc9,0x41,0xc0]
+; X64-NEXT:    vpaddb %xmm0, %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf1,0xfc,0xc0]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_pmovssdb_512:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vpmovssdb %zmm0, %xmm2 # encoding: [0x62,0xf2,0x7e,0x48,0x41,0xc2]
+; X86-NEXT:    vpmovssdb %zmm0, %xmm1 {%k1} # encoding: [0x62,0xf2,0x7e,0x49,0x41,0xc1]
+; X86-NEXT:    vpaddb %xmm1, %xmm2, %xmm1 # EVEX TO VEX Compression encoding: [0xc5,0xe9,0xfc,0xc9]
+; X86-NEXT:    vpmovssdb %zmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf2,0x7e,0xc9,0x41,0xc0]
+; X86-NEXT:    vpaddb %xmm0, %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf1,0xfc,0xc0]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %res0 = call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.512(<16 x i32> %a, <16 x i8> %passthru, i16 -1)
+  %res1 = call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.512(<16 x i32> %a, <16 x i8> %passthru, i16 %mask)
+  %res2 = call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.512(<16 x i32> %a, <16 x i8> zeroinitializer, i16 %mask)
+  %add1 = add <16 x i8> %res0, %res1
+  %add2 = add <16 x i8> %add1, %res2
+  ret <16 x i8> %add2
+}
+
 ; Memory folding tests for vpmovssdb
 define <16 x i8> @test_int_x86_avx10_pmovssdb_mem_128(ptr %ptr_a) {
 ; X64-LABEL: test_int_x86_avx10_pmovssdb_mem_128:

>From 745f84b3c1bd160c17d183136831f60ea72d3775 Mon Sep 17 00:00:00 2001
From: Ganesh Gopalasubramanian <Ganesh.Gopalasubramanian at amd.com>
Date: Wed, 22 Jul 2026 20:31:54 +0530
Subject: [PATCH 04/16] Add v prefix for the movssdb builtins and patterns
 using it

---
 clang/include/clang/Basic/BuiltinsX86.td   | 12 ++++++------
 clang/lib/Headers/avx10_2_512v2auxintrin.h |  8 ++++----
 clang/lib/Headers/avx10_2_v2auxintrin.h    | 16 ++++++++--------
 llvm/include/llvm/IR/IntrinsicsX86.td      | 12 ++++++------
 4 files changed, 24 insertions(+), 24 deletions(-)

diff --git a/clang/include/clang/Basic/BuiltinsX86.td b/clang/include/clang/Basic/BuiltinsX86.td
index eb155776420009..41fa98054ad6e9 100644
--- a/clang/include/clang/Basic/BuiltinsX86.td
+++ b/clang/include/clang/Basic/BuiltinsX86.td
@@ -2623,23 +2623,23 @@ let Features = "avx512f", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
 
 // VPMOVSSDB - Symmetric signed saturation DWord to Byte
 let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<128>] in {
-  def pmovssdb128_mask : X86Builtin<"_Vector<16, char>(_Vector<4, int>, _Vector<16, char>, unsigned char)">;
+  def vpmovssdb128_mask : X86Builtin<"_Vector<16, char>(_Vector<4, int>, _Vector<16, char>, unsigned char)">;
 }
 let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<256>] in {
-  def pmovssdb256_mask : X86Builtin<"_Vector<16, char>(_Vector<8, int>, _Vector<16, char>, unsigned char)">;
+  def vpmovssdb256_mask : X86Builtin<"_Vector<16, char>(_Vector<8, int>, _Vector<16, char>, unsigned char)">;
 }
 let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<512>] in {
-  def pmovssdb512_mask : X86Builtin<"_Vector<16, char>(_Vector<16, int>, _Vector<16, char>, unsigned short)">;
+  def vpmovssdb512_mask : X86Builtin<"_Vector<16, char>(_Vector<16, int>, _Vector<16, char>, unsigned short)">;
 }
 // VPMOVSSDB memory store
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
-  def pmovssdb128mem_mask : X86Builtin<"void(_Vector<16, char *>, _Vector<4, int>, unsigned char)">;
+  def vpmovssdb128mem_mask : X86Builtin<"void(_Vector<16, char *>, _Vector<4, int>, unsigned char)">;
 }
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
-  def pmovssdb256mem_mask : X86Builtin<"void(_Vector<16, char *>, _Vector<8, int>, unsigned char)">;
+  def vpmovssdb256mem_mask : X86Builtin<"void(_Vector<16, char *>, _Vector<8, int>, unsigned char)">;
 }
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
-  def pmovssdb512mem_mask : X86Builtin<"void(_Vector<16, char *>, _Vector<16, int>, unsigned short)">;
+  def vpmovssdb512mem_mask : X86Builtin<"void(_Vector<16, char *>, _Vector<16, int>, unsigned short)">;
 }
 
 let Features = "avx512bw", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
diff --git a/clang/lib/Headers/avx10_2_512v2auxintrin.h b/clang/lib/Headers/avx10_2_512v2auxintrin.h
index 02bda8a948a778..adf697371ec0b8 100644
--- a/clang/lib/Headers/avx10_2_512v2auxintrin.h
+++ b/clang/lib/Headers/avx10_2_512v2auxintrin.h
@@ -1114,7 +1114,7 @@ _mm512_maskz_cvthf6_hf8(__mmask64 __U, __m512i __A) {
 ///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS512
 _mm512_cvtss_epi32_epi8(__m512i __A) {
-  return (__m128i)__builtin_ia32_pmovssdb512_mask(
+  return (__m128i)__builtin_ia32_vpmovssdb512_mask(
       (__v16si)__A, (__v16qi)_mm_setzero_si128(), (__mmask16)-1);
 }
 
@@ -1135,7 +1135,7 @@ _mm512_cvtss_epi32_epi8(__m512i __A) {
 ///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS512
 _mm512_mask_cvtss_epi32_epi8(__m128i __W, __mmask16 __U, __m512i __A) {
-  return (__m128i)__builtin_ia32_pmovssdb512_mask((__v16si)__A, (__v16qi)__W,
+  return (__m128i)__builtin_ia32_vpmovssdb512_mask((__v16si)__A, (__v16qi)__W,
                                                   __U);
 }
 
@@ -1154,7 +1154,7 @@ _mm512_mask_cvtss_epi32_epi8(__m128i __W, __mmask16 __U, __m512i __A) {
 ///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS512
 _mm512_maskz_cvtss_epi32_epi8(__mmask16 __U, __m512i __A) {
-  return (__m128i)__builtin_ia32_pmovssdb512_mask(
+  return (__m128i)__builtin_ia32_vpmovssdb512_mask(
       (__v16si)__A, (__v16qi)_mm_setzero_si128(), __U);
 }
 
@@ -1174,7 +1174,7 @@ _mm512_maskz_cvtss_epi32_epi8(__mmask16 __U, __m512i __A) {
 ///    A 512-bit vector of [16 x i32].
 static __inline__ void __DEFAULT_FN_ATTRS512
 _mm512_mask_cvtss_epi32_storeu_epi8(void *__P, __mmask16 __M, __m512i __A) {
-  __builtin_ia32_pmovssdb512mem_mask((__v16qi *)__P, (__v16si)__A, __M);
+  __builtin_ia32_vpmovssdb512mem_mask((__v16qi *)__P, (__v16si)__A, __M);
 }
 
 #undef __DEFAULT_FN_ATTRS512
diff --git a/clang/lib/Headers/avx10_2_v2auxintrin.h b/clang/lib/Headers/avx10_2_v2auxintrin.h
index e852ab365a5d88..9da0209b1e77cc 100644
--- a/clang/lib/Headers/avx10_2_v2auxintrin.h
+++ b/clang/lib/Headers/avx10_2_v2auxintrin.h
@@ -2188,7 +2188,7 @@ _mm256_maskz_cvthf6_hf8(__mmask32 __U, __m256i __A) {
 ///    values; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_cvtss_epi32_epi8(__m128i __A) {
-  return (__m128i)__builtin_ia32_pmovssdb128_mask(
+  return (__m128i)__builtin_ia32_vpmovssdb128_mask(
       (__v4si)__A, (__v16qi)_mm_setzero_si128(), (__mmask8)-1);
 }
 
@@ -2209,7 +2209,7 @@ _mm_cvtss_epi32_epi8(__m128i __A) {
 ///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_mask_cvtss_epi32_epi8(__m128i __W, __mmask8 __U, __m128i __A) {
-  return (__m128i)__builtin_ia32_pmovssdb128_mask((__v4si)__A, (__v16qi)__W,
+  return (__m128i)__builtin_ia32_vpmovssdb128_mask((__v4si)__A, (__v16qi)__W,
                                                   __U);
 }
 
@@ -2228,7 +2228,7 @@ _mm_mask_cvtss_epi32_epi8(__m128i __W, __mmask8 __U, __m128i __A) {
 ///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_maskz_cvtss_epi32_epi8(__mmask8 __U, __m128i __A) {
-  return (__m128i)__builtin_ia32_pmovssdb128_mask(
+  return (__m128i)__builtin_ia32_vpmovssdb128_mask(
       (__v4si)__A, (__v16qi)_mm_setzero_si128(), __U);
 }
 
@@ -2247,7 +2247,7 @@ _mm_maskz_cvtss_epi32_epi8(__mmask8 __U, __m128i __A) {
 ///    values; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_cvtss_epi32_epi8(__m256i __A) {
-  return (__m128i)__builtin_ia32_pmovssdb256_mask(
+  return (__m128i)__builtin_ia32_vpmovssdb256_mask(
       (__v8si)__A, (__v16qi)_mm_setzero_si128(), (__mmask8)-1);
 }
 
@@ -2268,7 +2268,7 @@ _mm256_cvtss_epi32_epi8(__m256i __A) {
 ///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_mask_cvtss_epi32_epi8(__m128i __W, __mmask8 __U, __m256i __A) {
-  return (__m128i)__builtin_ia32_pmovssdb256_mask((__v8si)__A, (__v16qi)__W,
+  return (__m128i)__builtin_ia32_vpmovssdb256_mask((__v8si)__A, (__v16qi)__W,
                                                   __U);
 }
 
@@ -2287,7 +2287,7 @@ _mm256_mask_cvtss_epi32_epi8(__m128i __W, __mmask8 __U, __m256i __A) {
 ///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_maskz_cvtss_epi32_epi8(__mmask8 __U, __m256i __A) {
-  return (__m128i)__builtin_ia32_pmovssdb256_mask(
+  return (__m128i)__builtin_ia32_vpmovssdb256_mask(
       (__v8si)__A, (__v16qi)_mm_setzero_si128(), __U);
 }
 
@@ -2307,7 +2307,7 @@ _mm256_maskz_cvtss_epi32_epi8(__mmask8 __U, __m256i __A) {
 ///    A 128-bit vector of [4 x i32].
 static __inline__ void __DEFAULT_FN_ATTRS128
 _mm_mask_cvtss_epi32_storeu_epi8(void *__P, __mmask8 __M, __m128i __A) {
-  __builtin_ia32_pmovssdb128mem_mask((__v16qi *)__P, (__v4si)__A, __M);
+  __builtin_ia32_vpmovssdb128mem_mask((__v16qi *)__P, (__v4si)__A, __M);
 }
 
 /// Truncate packed 32-bit integers in \a __A to packed 8-bit integers with
@@ -2326,7 +2326,7 @@ _mm_mask_cvtss_epi32_storeu_epi8(void *__P, __mmask8 __M, __m128i __A) {
 ///    A 256-bit vector of [8 x i32].
 static __inline__ void __DEFAULT_FN_ATTRS256
 _mm256_mask_cvtss_epi32_storeu_epi8(void *__P, __mmask8 __M, __m256i __A) {
-  __builtin_ia32_pmovssdb256mem_mask((__v16qi *)__P, (__v8si)__A, __M);
+  __builtin_ia32_vpmovssdb256mem_mask((__v16qi *)__P, (__v8si)__A, __M);
 }
 
 #undef __DEFAULT_FN_ATTRS128
diff --git a/llvm/include/llvm/IR/IntrinsicsX86.td b/llvm/include/llvm/IR/IntrinsicsX86.td
index 35a57cd47659e4..79180301dff1d0 100644
--- a/llvm/include/llvm/IR/IntrinsicsX86.td
+++ b/llvm/include/llvm/IR/IntrinsicsX86.td
@@ -4682,34 +4682,34 @@ let TargetPrefix = "x86" in {
 
   // VPMOVSSDB - Symmetric signed saturation DWord to Byte
   def int_x86_avx10_mask_pmovss_db_128 :
-      ClangBuiltin<"__builtin_ia32_pmovssdb128_mask">,
+      ClangBuiltin<"__builtin_ia32_vpmovssdb128_mask">,
       DefaultAttrsIntrinsic<[llvm_v16i8_ty],
                             [llvm_v4i32_ty, llvm_v16i8_ty, llvm_i8_ty],
                             [IntrNoMem]>;
   def int_x86_avx10_mask_pmovss_db_256 :
-      ClangBuiltin<"__builtin_ia32_pmovssdb256_mask">,
+      ClangBuiltin<"__builtin_ia32_vpmovssdb256_mask">,
       DefaultAttrsIntrinsic<[llvm_v16i8_ty],
                             [llvm_v8i32_ty, llvm_v16i8_ty, llvm_i8_ty],
                             [IntrNoMem]>;
   def int_x86_avx10_mask_pmovss_db_512 :
-      ClangBuiltin<"__builtin_ia32_pmovssdb512_mask">,
+      ClangBuiltin<"__builtin_ia32_vpmovssdb512_mask">,
       DefaultAttrsIntrinsic<[llvm_v16i8_ty],
                             [llvm_v16i32_ty, llvm_v16i8_ty, llvm_i16_ty],
                             [IntrNoMem]>;
 
   // VPMOVSSDB memory store intrinsics
   def int_x86_avx10_mask_pmovss_db_mem_128 :
-      ClangBuiltin<"__builtin_ia32_pmovssdb128mem_mask">,
+      ClangBuiltin<"__builtin_ia32_vpmovssdb128mem_mask">,
       DefaultAttrsIntrinsic<[],
                             [llvm_ptr_ty, llvm_v4i32_ty, llvm_i8_ty],
                             [IntrArgMemOnly]>;
   def int_x86_avx10_mask_pmovss_db_mem_256 :
-      ClangBuiltin<"__builtin_ia32_pmovssdb256mem_mask">,
+      ClangBuiltin<"__builtin_ia32_vpmovssdb256mem_mask">,
       DefaultAttrsIntrinsic<[],
                             [llvm_ptr_ty, llvm_v8i32_ty, llvm_i8_ty],
                             [IntrArgMemOnly]>;
   def int_x86_avx10_mask_pmovss_db_mem_512 :
-      ClangBuiltin<"__builtin_ia32_pmovssdb512mem_mask">,
+      ClangBuiltin<"__builtin_ia32_vpmovssdb512mem_mask">,
       DefaultAttrsIntrinsic<[],
                             [llvm_ptr_ty, llvm_v16i32_ty, llvm_i16_ty],
                             [IntrArgMemOnly]>;

>From f5663dff7504e6ea454b91f656250e765a082956 Mon Sep 17 00:00:00 2001
From: "Attarde, Mahesh" <mahesh.attarde at intel.com>
Date: Thu, 16 Jul 2026 04:13:58 -0700
Subject: [PATCH 05/16] fix unpack

---
 llvm/lib/Target/X86/X86InstrAVX10_V2_AUX.td | 45 +++++++++++----------
 1 file changed, 23 insertions(+), 22 deletions(-)

diff --git a/llvm/lib/Target/X86/X86InstrAVX10_V2_AUX.td b/llvm/lib/Target/X86/X86InstrAVX10_V2_AUX.td
index f148cd552661cf..ba77c5fd240d2e 100644
--- a/llvm/lib/Target/X86/X86InstrAVX10_V2_AUX.td
+++ b/llvm/lib/Target/X86/X86InstrAVX10_V2_AUX.td
@@ -301,26 +301,35 @@ multiclass avx10_v2aux_cvt_widen_masked<bits<8> opc, string OpcodeStr,
 }
 
 // Group H multiclass: Byte unpack with immediate
-multiclass avx10_v2aux_unpackb<bits<8> opc, string OpcodeStr,
-                                X86VectorVTInfo _,
-                                SDPatternOperator OpNode> {
+multiclass avx10_v2aux_shuffle_base<bits<8> opc, string OpcodeStr,
+                                    X86VectorVTInfo _,
+                                    X86FoldableSchedWrite sched> {
   let ImmT = Imm8 in {
-    defm rri : AVX512_maskable<opc, MRMSrcReg, _,
-                    (outs _.RC:$dst),
+    defm rri : AVX512_maskable<opc, MRMSrcReg, _, (outs _.RC:$dst),
                     (ins _.RC:$src1, u8imm:$src2),
                     OpcodeStr, "$src2, $src1", "$src1, $src2",
-                    (_.VT (OpNode (_.VT _.RC:$src1), (i8 timm:$src2)))>,
-                    Sched<[WriteShuffle]>;
+                    (_.VT (!cast<Intrinsic>("int_x86_avx10_"#OpcodeStr#"_"#_.Size)
+                            (_.VT _.RC:$src1), (i8 timm:$src2)))>,
+                    Sched<[sched]>, EVEX, EVEX_CD8<8, CD8VF>;
     let mayLoad = 1 in
-    defm rmi : AVX512_maskable<opc, MRMSrcMem, _,
-                    (outs _.RC:$dst),
-                    (ins _.MemOp:$src1, u8imm:$src2),
-                    OpcodeStr, "$src2, $src1", "$src1, $src2",
-                    (_.VT (OpNode (_.VT (load addr:$src1)), (i8 timm:$src2)))>,
-                    Sched<[WriteShuffle.Folded]>;
+      defm rmi : AVX512_maskable<opc, MRMSrcMem, _, (outs _.RC:$dst),
+                      (ins _.MemOp:$src1, u8imm:$src2),
+                      OpcodeStr, "$src2, $src1", "$src1, $src2",
+                      (_.VT (!cast<Intrinsic>("int_x86_avx10_"#OpcodeStr#"_"#_.Size)
+                              (_.VT (load addr:$src1)), (i8 timm:$src2)))>,
+                      Sched<[sched.Folded]>, EVEX, EVEX_CD8<8, CD8VF>;
   }
 }
 
+multiclass avx10_v2aux_shuffle_b<bits<8> opc, string OpcodeStr> {
+  defm Z    : avx10_v2aux_shuffle_base<opc, OpcodeStr, v64i8_info,
+                                       WriteShuffleZ>, EVEX_V512;
+  defm Z256 : avx10_v2aux_shuffle_base<opc, OpcodeStr, v32i8x_info,
+                                       WriteShuffle256>, EVEX_V256;
+  defm Z128 : avx10_v2aux_shuffle_base<opc, OpcodeStr, v16i8x_info,
+                                       WriteShuffle>, EVEX_V128;
+}
+
 //===----------------------------------------------------------------------===//
 // AVX10 V2 AUX Instruction Definitions
 //===----------------------------------------------------------------------===//
@@ -552,13 +561,5 @@ let Predicates = [HasAVX10_V2_AUX] in {
 //-------------------------------------------------
 
 let Predicates = [HasAVX10_V2_AUX] in {
-  defm VUNPACKBZ    : avx10_v2aux_unpackb<0x3D, "vunpackb", v64i8_info,
-                                          int_x86_avx10_vunpackb_512>,
-                      EVEX_V512, TA, PS, EVEX, EVEX_CD8<8, CD8VF>;
-  defm VUNPACKBZ256 : avx10_v2aux_unpackb<0x3D, "vunpackb", v32i8x_info,
-                                          int_x86_avx10_vunpackb_256>,
-                      EVEX_V256, TA, PS, EVEX, EVEX_CD8<8, CD8VF>;
-  defm VUNPACKBZ128 : avx10_v2aux_unpackb<0x3D, "vunpackb", v16i8x_info,
-                                          int_x86_avx10_vunpackb_128>,
-                      EVEX_V128, TA, PS, EVEX, EVEX_CD8<8, CD8VF>;
+  defm VUNPACKB : avx10_v2aux_shuffle_b<0x3D, "vunpackb">, TA, PS;
 }

>From 6c8c779a8c65336e889e81433e5992413e666236 Mon Sep 17 00:00:00 2001
From: Ganesh Gopalasubramanian <Ganesh.Gopalasubramanian at amd.com>
Date: Thu, 30 Jul 2026 15:46:05 +0530
Subject: [PATCH 06/16] Address review comments for AVX10_V2_AUX

- Rename the convert builtins to <mnemonic>_<size>. Update the matching
  ClangBuiltin strings, headers and tests.
- Factor the narrowing and widening convert multiclasses into Z/Z256/Z128
  wrappers and align the AVX512_maskable continuation lines.
- Replace the Group A-H section labels with descriptions
- Detect avx10v2aux in getHostCPUFeatures via CPUID leaf 0x24 sub-leaf 1,
  ECX bit 3.
- Add the -mavx10v2aux/-mno-avx10v2aux driver flag, with driver and
  preprocessor tests and a release note.
- Make AVX10_V2_AUX imply AVX10.1 rather than AVX10.2, matching the spec's
  "Requires AVX10".
- Revert the multiversioning entry in X86TargetParser.def until libgcc
  assigns an ABI value.
- Move the vunpackb range checks to avx10_2_v2aux-builtins-errors.c
---
 clang/docs/ReleaseNotes.md                    |   1 +
 clang/include/clang/Basic/BuiltinsX86.td      |  66 ++--
 clang/include/clang/Options/Options.td        |   2 +
 clang/lib/Headers/avx10_2_512v2auxintrin.h    |  22 +-
 clang/lib/Headers/avx10_2_v2auxintrin.h       |  44 +--
 clang/lib/Headers/cpuid.h                     |   2 +-
 .../X86/avx10_2_v2aux-builtins-errors.c       |  39 +++
 .../test/CodeGen/X86/avx10_2_v2aux-builtins.c |  95 +++++-
 clang/test/Driver/x86-target-features.c       |   5 +
 clang/test/Preprocessor/x86_target_features.c |   8 +
 clang/test/Sema/builtins-x86.c                |  14 +-
 llvm/include/llvm/IR/IntrinsicsX86.td         |  66 ++--
 .../llvm/TargetParser/X86TargetParser.def     |   4 +-
 llvm/lib/Target/X86/X86.td                    |   4 +-
 llvm/lib/Target/X86/X86InstrAVX10_V2_AUX.td   | 319 ++++++++----------
 llvm/lib/TargetParser/Host.cpp                |   6 +
 llvm/lib/TargetParser/X86TargetParser.cpp     |   2 +-
 17 files changed, 391 insertions(+), 308 deletions(-)
 create mode 100644 clang/test/CodeGen/X86/avx10_2_v2aux-builtins-errors.c

diff --git a/clang/docs/ReleaseNotes.md b/clang/docs/ReleaseNotes.md
index b49a9db3d5fca4..7d2ea4b911e1d2 100644
--- a/clang/docs/ReleaseNotes.md
+++ b/clang/docs/ReleaseNotes.md
@@ -970,6 +970,7 @@ latest release, please see the [Clang Web Site](https://clang.llvm.org) or the
   - Support intrinsic of `_mm256_maskz_bitrev_epi8`.
   - Support intrinsic of `_mm_bitrev_epi8`.
   - Support intrinsic of `_mm256_bitrev_epi8`.
+- Support ISA of `AVX10_V2_AUX` (`-mavx10v2aux`).
 - Removed support for `AMX-TF32` (`-mamx-tf32`) and `TMMULTF32PS` instruction.
 
 #### Arm and AArch64 Support
diff --git a/clang/include/clang/Basic/BuiltinsX86.td b/clang/include/clang/Basic/BuiltinsX86.td
index 41fa98054ad6e9..6009d13c893518 100644
--- a/clang/include/clang/Basic/BuiltinsX86.td
+++ b/clang/include/clang/Basic/BuiltinsX86.td
@@ -5200,151 +5200,151 @@ let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in
 
 // VCVTBF82PS
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
-  def vcvtbf8_2ps128 : X86Builtin<"_Vector<4, float>(_Vector<16, char>)">;
+  def vcvtbf82ps_128 : X86Builtin<"_Vector<4, float>(_Vector<16, char>)">;
 }
 
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
-  def vcvtbf8_2ps256 : X86Builtin<"_Vector<8, float>(_Vector<16, char>)">;
+  def vcvtbf82ps_256 : X86Builtin<"_Vector<8, float>(_Vector<16, char>)">;
 }
 
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
-  def vcvtbf8_2ps512 : X86Builtin<"_Vector<16, float>(_Vector<16, char>)">;
+  def vcvtbf82ps_512 : X86Builtin<"_Vector<16, float>(_Vector<16, char>)">;
 }
 
 // VCVTHF82PS
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
-  def vcvthf8_2ps128 : X86Builtin<"_Vector<4, float>(_Vector<16, char>)">;
+  def vcvthf82ps_128 : X86Builtin<"_Vector<4, float>(_Vector<16, char>)">;
 }
 
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
-  def vcvthf8_2ps256 : X86Builtin<"_Vector<8, float>(_Vector<16, char>)">;
+  def vcvthf82ps_256 : X86Builtin<"_Vector<8, float>(_Vector<16, char>)">;
 }
 
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
-  def vcvthf8_2ps512 : X86Builtin<"_Vector<16, float>(_Vector<16, char>)">;
+  def vcvthf82ps_512 : X86Builtin<"_Vector<16, float>(_Vector<16, char>)">;
 }
 
 // Group E: Same-size reg-only conversions (no masking)
 
 // VCVTBF82BF6S
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
-  def vcvtbf82bf6s128 : X86Builtin<"_Vector<16, char>(_Vector<16, char>)">;
+  def vcvtbf82bf6s_128 : X86Builtin<"_Vector<16, char>(_Vector<16, char>)">;
 }
 
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
-  def vcvtbf82bf6s256 : X86Builtin<"_Vector<32, char>(_Vector<32, char>)">;
+  def vcvtbf82bf6s_256 : X86Builtin<"_Vector<32, char>(_Vector<32, char>)">;
 }
 
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
-  def vcvtbf82bf6s512 : X86Builtin<"_Vector<64, char>(_Vector<64, char>)">;
+  def vcvtbf82bf6s_512 : X86Builtin<"_Vector<64, char>(_Vector<64, char>)">;
 }
 
 // VCVTHF82HF6S
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
-  def vcvthf82hf6s128 : X86Builtin<"_Vector<16, char>(_Vector<16, char>)">;
+  def vcvthf82hf6s_128 : X86Builtin<"_Vector<16, char>(_Vector<16, char>)">;
 }
 
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
-  def vcvthf82hf6s256 : X86Builtin<"_Vector<32, char>(_Vector<32, char>)">;
+  def vcvthf82hf6s_256 : X86Builtin<"_Vector<32, char>(_Vector<32, char>)">;
 }
 
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
-  def vcvthf82hf6s512 : X86Builtin<"_Vector<64, char>(_Vector<64, char>)">;
+  def vcvthf82hf6s_512 : X86Builtin<"_Vector<64, char>(_Vector<64, char>)">;
 }
 
 // Group F: Expanding/same-size conversions (no masking in intrinsic; use selectb for masking)
 
 // VCVTBF42HF8
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
-  def vcvtbf42hf8128 : X86Builtin<"_Vector<16, char>(_Vector<16, char>)">;
+  def vcvtbf42hf8_128 : X86Builtin<"_Vector<16, char>(_Vector<16, char>)">;
 }
 
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
-  def vcvtbf42hf8256 : X86Builtin<"_Vector<32, char>(_Vector<16, char>)">;
+  def vcvtbf42hf8_256 : X86Builtin<"_Vector<32, char>(_Vector<16, char>)">;
 }
 
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
-  def vcvtbf42hf8512 : X86Builtin<"_Vector<64, char>(_Vector<32, char>)">;
+  def vcvtbf42hf8_512 : X86Builtin<"_Vector<64, char>(_Vector<32, char>)">;
 }
 
 // VCVTBF62HF8
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
-  def vcvtbf62hf8128 : X86Builtin<"_Vector<16, char>(_Vector<16, char>)">;
+  def vcvtbf62hf8_128 : X86Builtin<"_Vector<16, char>(_Vector<16, char>)">;
 }
 
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
-  def vcvtbf62hf8256 : X86Builtin<"_Vector<32, char>(_Vector<32, char>)">;
+  def vcvtbf62hf8_256 : X86Builtin<"_Vector<32, char>(_Vector<32, char>)">;
 }
 
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
-  def vcvtbf62hf8512 : X86Builtin<"_Vector<64, char>(_Vector<64, char>)">;
+  def vcvtbf62hf8_512 : X86Builtin<"_Vector<64, char>(_Vector<64, char>)">;
 }
 
 // VCVTHF62HF8
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
-  def vcvthf62hf8128 : X86Builtin<"_Vector<16, char>(_Vector<16, char>)">;
+  def vcvthf62hf8_128 : X86Builtin<"_Vector<16, char>(_Vector<16, char>)">;
 }
 
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
-  def vcvthf62hf8256 : X86Builtin<"_Vector<32, char>(_Vector<32, char>)">;
+  def vcvthf62hf8_256 : X86Builtin<"_Vector<32, char>(_Vector<32, char>)">;
 }
 
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
-  def vcvthf62hf8512 : X86Builtin<"_Vector<64, char>(_Vector<64, char>)">;
+  def vcvthf62hf8_512 : X86Builtin<"_Vector<64, char>(_Vector<64, char>)">;
 }
 
 // Group D: VCVTBF82BF4S / VCVTHF82BF4S - FP8 to FP4 truncating conversions
 
 // VCVTBF82BF4S - FP8 E5M2 to FP4 E2M1 with saturation (output is half size)
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
-  def vcvtbf82bf4s128 : X86Builtin<"_Vector<16, char>(_Vector<16, char>)">;
+  def vcvtbf82bf4s_128 : X86Builtin<"_Vector<16, char>(_Vector<16, char>)">;
 }
 
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
-  def vcvtbf82bf4s256 : X86Builtin<"_Vector<16, char>(_Vector<32, char>)">;
+  def vcvtbf82bf4s_256 : X86Builtin<"_Vector<16, char>(_Vector<32, char>)">;
 }
 
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
-  def vcvtbf82bf4s512 : X86Builtin<"_Vector<32, char>(_Vector<64, char>)">;
+  def vcvtbf82bf4s_512 : X86Builtin<"_Vector<32, char>(_Vector<64, char>)">;
 }
 
 // VCVTBF82BF4S memory store variants (no masking - spec does not support masks)
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
-  def vcvtbf82bf4s128mem : X86Builtin<"void(void *, _Vector<16, char>)">;
+  def vcvtbf82bf4s_128_mem : X86Builtin<"void(void *, _Vector<16, char>)">;
 }
 
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
-  def vcvtbf82bf4s256mem : X86Builtin<"void(void *, _Vector<32, char>)">;
+  def vcvtbf82bf4s_256_mem : X86Builtin<"void(void *, _Vector<32, char>)">;
 }
 
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
-  def vcvtbf82bf4s512mem : X86Builtin<"void(void *, _Vector<64, char>)">;
+  def vcvtbf82bf4s_512_mem : X86Builtin<"void(void *, _Vector<64, char>)">;
 }
 
 // VCVTHF82BF4S - FP8 E4M3 to FP4 E2M1 with saturation (output is half size)
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
-  def vcvthf82bf4s128 : X86Builtin<"_Vector<16, char>(_Vector<16, char>)">;
+  def vcvthf82bf4s_128 : X86Builtin<"_Vector<16, char>(_Vector<16, char>)">;
 }
 
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
-  def vcvthf82bf4s256 : X86Builtin<"_Vector<16, char>(_Vector<32, char>)">;
+  def vcvthf82bf4s_256 : X86Builtin<"_Vector<16, char>(_Vector<32, char>)">;
 }
 
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
-  def vcvthf82bf4s512 : X86Builtin<"_Vector<32, char>(_Vector<64, char>)">;
+  def vcvthf82bf4s_512 : X86Builtin<"_Vector<32, char>(_Vector<64, char>)">;
 }
 
 // VCVTHF82BF4S memory store variants (no masking - spec does not support masks)
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
-  def vcvthf82bf4s128mem : X86Builtin<"void(void *, _Vector<16, char>)">;
+  def vcvthf82bf4s_128_mem : X86Builtin<"void(void *, _Vector<16, char>)">;
 }
 
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
-  def vcvthf82bf4s256mem : X86Builtin<"void(void *, _Vector<32, char>)">;
+  def vcvthf82bf4s_256_mem : X86Builtin<"void(void *, _Vector<32, char>)">;
 }
 
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
-  def vcvthf82bf4s512mem : X86Builtin<"void(void *, _Vector<64, char>)">;
+  def vcvthf82bf4s_512_mem : X86Builtin<"void(void *, _Vector<64, char>)">;
 }
 
 // Group H: VUNPACKB - Byte unpack with immediate (no masking in intrinsic)
diff --git a/clang/include/clang/Options/Options.td b/clang/include/clang/Options/Options.td
index 59bfd621a59949..1d6ced2c1c817b 100644
--- a/clang/include/clang/Options/Options.td
+++ b/clang/include/clang/Options/Options.td
@@ -7209,6 +7209,8 @@ def mavx10_1 : Flag<["-"], "mavx10.1">, Group<m_x86_Features_Group>;
 def mno_avx10_1 : Flag<["-"], "mno-avx10.1">, Group<m_x86_Features_Group>;
 def mavx10_2 : Flag<["-"], "mavx10.2">, Group<m_x86_Features_Group>;
 def mno_avx10_2 : Flag<["-"], "mno-avx10.2">, Group<m_x86_Features_Group>;
+def mavx10v2aux : Flag<["-"], "mavx10v2aux">, Group<m_x86_Features_Group>;
+def mno_avx10v2aux : Flag<["-"], "mno-avx10v2aux">, Group<m_x86_Features_Group>;
 def mavx2 : Flag<["-"], "mavx2">, Group<m_x86_Features_Group>;
 def mno_avx2 : Flag<["-"], "mno-avx2">, Group<m_x86_Features_Group>;
 def mavx512f : Flag<["-"], "mavx512f">, Group<m_x86_Features_Group>;
diff --git a/clang/lib/Headers/avx10_2_512v2auxintrin.h b/clang/lib/Headers/avx10_2_512v2auxintrin.h
index adf697371ec0b8..34a5b09f46a260 100644
--- a/clang/lib/Headers/avx10_2_512v2auxintrin.h
+++ b/clang/lib/Headers/avx10_2_512v2auxintrin.h
@@ -657,7 +657,7 @@ _mm512_maskz_cvts_biasps_hf8(__mmask16 __U, __m512i __A, __m512 __B) {
 /// \returns
 ///    A 512-bit vector of [16 x float] containing the converted values.
 static __inline__ __m512 __DEFAULT_FN_ATTRS512 _mm512_cvtbf8_ps(__m128i __A) {
-  return (__m512)__builtin_ia32_vcvtbf8_2ps512((__v16qi)__A);
+  return (__m512)__builtin_ia32_vcvtbf82ps_512((__v16qi)__A);
 }
 
 /// Convert packed BF8 (8-bit) floating-point elements in \a __A to packed
@@ -716,7 +716,7 @@ _mm512_maskz_cvtbf8_ps(__mmask16 __U, __m128i __A) {
 /// \returns
 ///    A 512-bit vector of [16 x float] containing the converted values.
 static __inline__ __m512 __DEFAULT_FN_ATTRS512 _mm512_cvthf8_ps(__m128i __A) {
-  return (__m512)__builtin_ia32_vcvthf8_2ps512((__v16qi)__A);
+  return (__m512)__builtin_ia32_vcvthf82ps_512((__v16qi)__A);
 }
 
 /// Convert packed HF8 (8-bit) floating-point elements in \a __A to packed
@@ -776,7 +776,7 @@ _mm512_maskz_cvthf8_ps(__mmask16 __U, __m128i __A) {
 ///    A 512-bit vector of [64 x i8] containing the converted BF6 values.
 static __inline__ __m512i __DEFAULT_FN_ATTRS512
 _mm512_cvtbf8_bf6s(__m512i __A) {
-  return (__m512i)__builtin_ia32_vcvtbf82bf6s512((__v64qi)__A);
+  return (__m512i)__builtin_ia32_vcvtbf82bf6s_512((__v64qi)__A);
 }
 
 /// Convert packed HF8 (8-bit) floating-point elements in \a __A to packed
@@ -793,7 +793,7 @@ _mm512_cvtbf8_bf6s(__m512i __A) {
 ///    A 512-bit vector of [64 x i8] containing the converted HF6 values.
 static __inline__ __m512i __DEFAULT_FN_ATTRS512
 _mm512_cvthf8_hf6s(__m512i __A) {
-  return (__m512i)__builtin_ia32_vcvthf82hf6s512((__v64qi)__A);
+  return (__m512i)__builtin_ia32_vcvthf82hf6s_512((__v64qi)__A);
 }
 
 /// Convert packed BF8 (8-bit) floating-point elements in \a __A to packed
@@ -810,7 +810,7 @@ _mm512_cvthf8_hf6s(__m512i __A) {
 ///    A 256-bit vector of [32 x i8] containing the converted BF4 values.
 static __inline__ __m256i __DEFAULT_FN_ATTRS512
 _mm512_cvtbf8_bf4s(__m512i __A) {
-  return (__m256i)__builtin_ia32_vcvtbf82bf4s512((__v64qi)__A);
+  return (__m256i)__builtin_ia32_vcvtbf82bf4s_512((__v64qi)__A);
 }
 
 /// Convert packed HF8 (8-bit) floating-point elements in \a __A to packed
@@ -827,7 +827,7 @@ _mm512_cvtbf8_bf4s(__m512i __A) {
 ///    A 256-bit vector of [32 x i8] containing the converted BF4 values.
 static __inline__ __m256i __DEFAULT_FN_ATTRS512
 _mm512_cvthf8_bf4s(__m512i __A) {
-  return (__m256i)__builtin_ia32_vcvthf82bf4s512((__v64qi)__A);
+  return (__m256i)__builtin_ia32_vcvthf82bf4s_512((__v64qi)__A);
 }
 
 /// Convert packed BF8 (8-bit) floating-point elements in \a __A to packed
@@ -845,7 +845,7 @@ _mm512_cvthf8_bf4s(__m512i __A) {
 ///    A 512-bit vector of [64 x i8] containing BF8 values.
 static __inline__ void __DEFAULT_FN_ATTRS512
 _mm512_cvtbf8_bf4s_storeu(void *__P, __m512i __A) {
-  __builtin_ia32_vcvtbf82bf4s512mem(__P, (__v64qi)__A);
+  __builtin_ia32_vcvtbf82bf4s_512_mem(__P, (__v64qi)__A);
 }
 
 /// Convert packed HF8 (8-bit) floating-point elements in \a __A to packed
@@ -863,7 +863,7 @@ _mm512_cvtbf8_bf4s_storeu(void *__P, __m512i __A) {
 ///    A 512-bit vector of [64 x i8] containing HF8 values.
 static __inline__ void __DEFAULT_FN_ATTRS512
 _mm512_cvthf8_bf4s_storeu(void *__P, __m512i __A) {
-  __builtin_ia32_vcvthf82bf4s512mem(__P, (__v64qi)__A);
+  __builtin_ia32_vcvthf82bf4s_512_mem(__P, (__v64qi)__A);
 }
 
 /// Convert packed BF4 (4-bit) floating-point elements in \a __A to packed
@@ -879,7 +879,7 @@ _mm512_cvthf8_bf4s_storeu(void *__P, __m512i __A) {
 /// \returns
 ///    A 512-bit vector of [64 x i8] containing the converted HF8 values.
 static __inline__ __m512i __DEFAULT_FN_ATTRS512 _mm512_cvtbf4_hf8(__m256i __A) {
-  return (__m512i)__builtin_ia32_vcvtbf42hf8512((__v32qi)__A);
+  return (__m512i)__builtin_ia32_vcvtbf42hf8_512((__v32qi)__A);
 }
 
 /// Convert packed BF4 (4-bit) floating-point elements in \a __A to packed
@@ -937,7 +937,7 @@ _mm512_maskz_cvtbf4_hf8(__mmask64 __U, __m256i __A) {
 /// \returns
 ///    A 512-bit vector of [64 x i8] containing the converted HF8 values.
 static __inline__ __m512i __DEFAULT_FN_ATTRS512 _mm512_cvtbf6_hf8(__m512i __A) {
-  return (__m512i)__builtin_ia32_vcvtbf62hf8512((__v64qi)__A);
+  return (__m512i)__builtin_ia32_vcvtbf62hf8_512((__v64qi)__A);
 }
 
 /// Convert packed BF6 (6-bit) floating-point elements in \a __A to packed
@@ -995,7 +995,7 @@ _mm512_maskz_cvtbf6_hf8(__mmask64 __U, __m512i __A) {
 /// \returns
 ///    A 512-bit vector of [64 x i8] containing the converted HF8 values.
 static __inline__ __m512i __DEFAULT_FN_ATTRS512 _mm512_cvthf6_hf8(__m512i __A) {
-  return (__m512i)__builtin_ia32_vcvthf62hf8512((__v64qi)__A);
+  return (__m512i)__builtin_ia32_vcvthf62hf8_512((__v64qi)__A);
 }
 
 /// Convert packed HF6 (6-bit) floating-point elements in \a __A to packed
diff --git a/clang/lib/Headers/avx10_2_v2auxintrin.h b/clang/lib/Headers/avx10_2_v2auxintrin.h
index 9da0209b1e77cc..03940740b713e5 100644
--- a/clang/lib/Headers/avx10_2_v2auxintrin.h
+++ b/clang/lib/Headers/avx10_2_v2auxintrin.h
@@ -1278,7 +1278,7 @@ _mm256_maskz_cvts_biasps_hf8(__mmask8 __U, __m256i __A, __m256 __B) {
 /// \returns
 ///    A 128-bit vector of [4 x float] containing the converted values.
 static __inline__ __m128 __DEFAULT_FN_ATTRS128 _mm_cvtbf8_ps(__m128i __A) {
-  return (__m128)__builtin_ia32_vcvtbf8_2ps128((__v16qi)__A);
+  return (__m128)__builtin_ia32_vcvtbf82ps_128((__v16qi)__A);
 }
 
 /// Convert packed BF8 (8-bit) floating-point elements in \a __A to packed
@@ -1337,7 +1337,7 @@ _mm_maskz_cvtbf8_ps(__mmask8 __U, __m128i __A) {
 /// \returns
 ///    A 256-bit vector of [8 x float] containing the converted values.
 static __inline__ __m256 __DEFAULT_FN_ATTRS256 _mm256_cvtbf8_ps(__m128i __A) {
-  return (__m256)__builtin_ia32_vcvtbf8_2ps256((__v16qi)__A);
+  return (__m256)__builtin_ia32_vcvtbf82ps_256((__v16qi)__A);
 }
 
 /// Convert packed BF8 (8-bit) floating-point elements in \a __A to packed
@@ -1396,7 +1396,7 @@ _mm256_maskz_cvtbf8_ps(__mmask8 __U, __m128i __A) {
 /// \returns
 ///    A 128-bit vector of [4 x float] containing the converted values.
 static __inline__ __m128 __DEFAULT_FN_ATTRS128 _mm_cvthf8_ps(__m128i __A) {
-  return (__m128)__builtin_ia32_vcvthf8_2ps128((__v16qi)__A);
+  return (__m128)__builtin_ia32_vcvthf82ps_128((__v16qi)__A);
 }
 
 /// Convert packed HF8 (8-bit) floating-point elements in \a __A to packed
@@ -1455,7 +1455,7 @@ _mm_maskz_cvthf8_ps(__mmask8 __U, __m128i __A) {
 /// \returns
 ///    A 256-bit vector of [8 x float] containing the converted values.
 static __inline__ __m256 __DEFAULT_FN_ATTRS256 _mm256_cvthf8_ps(__m128i __A) {
-  return (__m256)__builtin_ia32_vcvthf8_2ps256((__v16qi)__A);
+  return (__m256)__builtin_ia32_vcvthf82ps_256((__v16qi)__A);
 }
 
 /// Convert packed HF8 (8-bit) floating-point elements in \a __A to packed
@@ -1514,7 +1514,7 @@ _mm256_maskz_cvthf8_ps(__mmask8 __U, __m128i __A) {
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted BF6 values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvtbf8_bf6s(__m128i __A) {
-  return (__m128i)__builtin_ia32_vcvtbf82bf6s128((__v16qi)__A);
+  return (__m128i)__builtin_ia32_vcvtbf82bf6s_128((__v16qi)__A);
 }
 
 /// Convert packed BF8 (8-bit) floating-point elements in \a __A to packed
@@ -1531,7 +1531,7 @@ static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvtbf8_bf6s(__m128i __A) {
 ///    A 256-bit vector of [32 x i8] containing the converted BF6 values.
 static __inline__ __m256i __DEFAULT_FN_ATTRS256
 _mm256_cvtbf8_bf6s(__m256i __A) {
-  return (__m256i)__builtin_ia32_vcvtbf82bf6s256((__v32qi)__A);
+  return (__m256i)__builtin_ia32_vcvtbf82bf6s_256((__v32qi)__A);
 }
 
 /// Convert packed HF8 (8-bit) floating-point elements in \a __A to packed
@@ -1547,7 +1547,7 @@ _mm256_cvtbf8_bf6s(__m256i __A) {
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted HF6 values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvthf8_hf6s(__m128i __A) {
-  return (__m128i)__builtin_ia32_vcvthf82hf6s128((__v16qi)__A);
+  return (__m128i)__builtin_ia32_vcvthf82hf6s_128((__v16qi)__A);
 }
 
 /// Convert packed HF8 (8-bit) floating-point elements in \a __A to packed
@@ -1564,7 +1564,7 @@ static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvthf8_hf6s(__m128i __A) {
 ///    A 256-bit vector of [32 x i8] containing the converted HF6 values.
 static __inline__ __m256i __DEFAULT_FN_ATTRS256
 _mm256_cvthf8_hf6s(__m256i __A) {
-  return (__m256i)__builtin_ia32_vcvthf82hf6s256((__v32qi)__A);
+  return (__m256i)__builtin_ia32_vcvthf82hf6s_256((__v32qi)__A);
 }
 
 /// Convert packed BF8 (8-bit) floating-point elements in \a __A to packed
@@ -1581,7 +1581,7 @@ _mm256_cvthf8_hf6s(__m256i __A) {
 ///    A 128-bit vector of [16 x i8] containing the converted BF4 values
 ///    (lower 8 bytes used).
 static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvtbf8_bf4s(__m128i __A) {
-  return (__m128i)__builtin_ia32_vcvtbf82bf4s128((__v16qi)__A);
+  return (__m128i)__builtin_ia32_vcvtbf82bf4s_128((__v16qi)__A);
 }
 
 /// Convert packed BF8 (8-bit) floating-point elements in \a __A to packed
@@ -1598,7 +1598,7 @@ static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvtbf8_bf4s(__m128i __A) {
 ///    A 128-bit vector of [16 x i8] containing the converted BF4 values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_cvtbf8_bf4s(__m256i __A) {
-  return (__m128i)__builtin_ia32_vcvtbf82bf4s256((__v32qi)__A);
+  return (__m128i)__builtin_ia32_vcvtbf82bf4s_256((__v32qi)__A);
 }
 
 /// Convert packed HF8 (8-bit) floating-point elements in \a __A to packed
@@ -1615,7 +1615,7 @@ _mm256_cvtbf8_bf4s(__m256i __A) {
 ///    A 128-bit vector of [16 x i8] containing the converted BF4 values
 ///    (lower 8 bytes used).
 static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvthf8_bf4s(__m128i __A) {
-  return (__m128i)__builtin_ia32_vcvthf82bf4s128((__v16qi)__A);
+  return (__m128i)__builtin_ia32_vcvthf82bf4s_128((__v16qi)__A);
 }
 
 /// Convert packed HF8 (8-bit) floating-point elements in \a __A to packed
@@ -1632,7 +1632,7 @@ static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvthf8_bf4s(__m128i __A) {
 ///    A 128-bit vector of [16 x i8] containing the converted BF4 values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_cvthf8_bf4s(__m256i __A) {
-  return (__m128i)__builtin_ia32_vcvthf82bf4s256((__v32qi)__A);
+  return (__m128i)__builtin_ia32_vcvthf82bf4s_256((__v32qi)__A);
 }
 
 /// Convert packed BF8 (8-bit) floating-point elements in \a __A to packed
@@ -1650,7 +1650,7 @@ _mm256_cvthf8_bf4s(__m256i __A) {
 ///    A 128-bit vector of [16 x i8] containing BF8 values.
 static __inline__ void __DEFAULT_FN_ATTRS128
 _mm_cvtbf8_bf4s_storeu(void *__P, __m128i __A) {
-  __builtin_ia32_vcvtbf82bf4s128mem(__P, (__v16qi)__A);
+  __builtin_ia32_vcvtbf82bf4s_128_mem(__P, (__v16qi)__A);
 }
 
 /// Convert packed BF8 (8-bit) floating-point elements in \a __A to packed
@@ -1668,7 +1668,7 @@ _mm_cvtbf8_bf4s_storeu(void *__P, __m128i __A) {
 ///    A 256-bit vector of [32 x i8] containing BF8 values.
 static __inline__ void __DEFAULT_FN_ATTRS256
 _mm256_cvtbf8_bf4s_storeu(void *__P, __m256i __A) {
-  __builtin_ia32_vcvtbf82bf4s256mem(__P, (__v32qi)__A);
+  __builtin_ia32_vcvtbf82bf4s_256_mem(__P, (__v32qi)__A);
 }
 
 /// Convert packed HF8 (8-bit) floating-point elements in \a __A to packed
@@ -1686,7 +1686,7 @@ _mm256_cvtbf8_bf4s_storeu(void *__P, __m256i __A) {
 ///    A 128-bit vector of [16 x i8] containing HF8 values.
 static __inline__ void __DEFAULT_FN_ATTRS128
 _mm_cvthf8_bf4s_storeu(void *__P, __m128i __A) {
-  __builtin_ia32_vcvthf82bf4s128mem(__P, (__v16qi)__A);
+  __builtin_ia32_vcvthf82bf4s_128_mem(__P, (__v16qi)__A);
 }
 
 /// Convert packed HF8 (8-bit) floating-point elements in \a __A to packed
@@ -1704,7 +1704,7 @@ _mm_cvthf8_bf4s_storeu(void *__P, __m128i __A) {
 ///    A 256-bit vector of [32 x i8] containing HF8 values.
 static __inline__ void __DEFAULT_FN_ATTRS256
 _mm256_cvthf8_bf4s_storeu(void *__P, __m256i __A) {
-  __builtin_ia32_vcvthf82bf4s256mem(__P, (__v32qi)__A);
+  __builtin_ia32_vcvthf82bf4s_256_mem(__P, (__v32qi)__A);
 }
 
 /// Convert packed BF4 (4-bit) floating-point elements in \a __A to packed
@@ -1720,7 +1720,7 @@ _mm256_cvthf8_bf4s_storeu(void *__P, __m256i __A) {
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted HF8 values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvtbf4_hf8(__m128i __A) {
-  return (__m128i)__builtin_ia32_vcvtbf42hf8128((__v16qi)__A);
+  return (__m128i)__builtin_ia32_vcvtbf42hf8_128((__v16qi)__A);
 }
 
 /// Convert packed BF4 (4-bit) floating-point elements in \a __A to packed
@@ -1778,7 +1778,7 @@ _mm_maskz_cvtbf4_hf8(__mmask16 __U, __m128i __A) {
 /// \returns
 ///    A 256-bit vector of [32 x i8] containing the converted HF8 values.
 static __inline__ __m256i __DEFAULT_FN_ATTRS256 _mm256_cvtbf4_hf8(__m128i __A) {
-  return (__m256i)__builtin_ia32_vcvtbf42hf8256((__v16qi)__A);
+  return (__m256i)__builtin_ia32_vcvtbf42hf8_256((__v16qi)__A);
 }
 
 /// Convert packed BF4 (4-bit) floating-point elements in \a __A to packed
@@ -1836,7 +1836,7 @@ _mm256_maskz_cvtbf4_hf8(__mmask32 __U, __m128i __A) {
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted HF8 values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvtbf6_hf8(__m128i __A) {
-  return (__m128i)__builtin_ia32_vcvtbf62hf8128((__v16qi)__A);
+  return (__m128i)__builtin_ia32_vcvtbf62hf8_128((__v16qi)__A);
 }
 
 /// Convert packed BF6 (6-bit) floating-point elements in \a __A to packed
@@ -1894,7 +1894,7 @@ _mm_maskz_cvtbf6_hf8(__mmask16 __U, __m128i __A) {
 /// \returns
 ///    A 256-bit vector of [32 x i8] containing the converted HF8 values.
 static __inline__ __m256i __DEFAULT_FN_ATTRS256 _mm256_cvtbf6_hf8(__m256i __A) {
-  return (__m256i)__builtin_ia32_vcvtbf62hf8256((__v32qi)__A);
+  return (__m256i)__builtin_ia32_vcvtbf62hf8_256((__v32qi)__A);
 }
 
 /// Convert packed BF6 (6-bit) floating-point elements in \a __A to packed
@@ -1952,7 +1952,7 @@ _mm256_maskz_cvtbf6_hf8(__mmask32 __U, __m256i __A) {
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted HF8 values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvthf6_hf8(__m128i __A) {
-  return (__m128i)__builtin_ia32_vcvthf62hf8128((__v16qi)__A);
+  return (__m128i)__builtin_ia32_vcvthf62hf8_128((__v16qi)__A);
 }
 
 /// Convert packed HF6 (6-bit) floating-point elements in \a __A to packed
@@ -2010,7 +2010,7 @@ _mm_maskz_cvthf6_hf8(__mmask16 __U, __m128i __A) {
 /// \returns
 ///    A 256-bit vector of [32 x i8] containing the converted HF8 values.
 static __inline__ __m256i __DEFAULT_FN_ATTRS256 _mm256_cvthf6_hf8(__m256i __A) {
-  return (__m256i)__builtin_ia32_vcvthf62hf8256((__v32qi)__A);
+  return (__m256i)__builtin_ia32_vcvthf62hf8_256((__v32qi)__A);
 }
 
 /// Convert packed HF6 (6-bit) floating-point elements in \a __A to packed
diff --git a/clang/lib/Headers/cpuid.h b/clang/lib/Headers/cpuid.h
index a8c6adbc3e2dc8..4b3bef7fca6858 100644
--- a/clang/lib/Headers/cpuid.h
+++ b/clang/lib/Headers/cpuid.h
@@ -222,7 +222,7 @@
 #define bit_AVX10         0x00080000
 #define bit_APXF          0x00200000
 
-/* Features in %ecx for leaf 24 sub-leaf 1 */
+/* Features in %ecx for leaf 0x24 sub-leaf 1 */
 #define bit_AVX10_V2_AUX 0x00000008
 
 /* Features in %eax for leaf 13 sub-leaf 1 */
diff --git a/clang/test/CodeGen/X86/avx10_2_v2aux-builtins-errors.c b/clang/test/CodeGen/X86/avx10_2_v2aux-builtins-errors.c
new file mode 100644
index 00000000000000..fa620ab58cb3bb
--- /dev/null
+++ b/clang/test/CodeGen/X86/avx10_2_v2aux-builtins-errors.c
@@ -0,0 +1,39 @@
+// RUN: %clang_cc1 -flax-vector-conversions=none -ffreestanding %s -triple=x86_64-unknown-unknown -target-feature +avx10v2aux -Wno-invalid-feature-combination -Wall -Werror -verify
+
+#include <immintrin.h>
+
+__m128i test_mm_unpackb_epi8(__m128i __A) {
+  return _mm_unpackb_epi8(__A, 64); // expected-error {{argument value 64 is outside the valid range [0, 63]}}
+}
+
+__m128i test_mm_mask_unpackb_epi8(__m128i __W, __mmask16 __U, __m128i __A) {
+  return _mm_mask_unpackb_epi8(__W, __U, __A, 64); // expected-error {{argument value 64 is outside the valid range [0, 63]}}
+}
+
+__m128i test_mm_maskz_unpackb_epi8(__mmask16 __U, __m128i __A) {
+  return _mm_maskz_unpackb_epi8(__U, __A, 64); // expected-error {{argument value 64 is outside the valid range [0, 63]}}
+}
+
+__m256i test_mm256_unpackb_epi8(__m256i __A) {
+  return _mm256_unpackb_epi8(__A, 64); // expected-error {{argument value 64 is outside the valid range [0, 63]}}
+}
+
+__m256i test_mm256_mask_unpackb_epi8(__m256i __W, __mmask32 __U, __m256i __A) {
+  return _mm256_mask_unpackb_epi8(__W, __U, __A, 64); // expected-error {{argument value 64 is outside the valid range [0, 63]}}
+}
+
+__m256i test_mm256_maskz_unpackb_epi8(__mmask32 __U, __m256i __A) {
+  return _mm256_maskz_unpackb_epi8(__U, __A, 64); // expected-error {{argument value 64 is outside the valid range [0, 63]}}
+}
+
+__m512i test_mm512_unpackb_epi8(__m512i __A) {
+  return _mm512_unpackb_epi8(__A, 64); // expected-error {{argument value 64 is outside the valid range [0, 63]}}
+}
+
+__m512i test_mm512_mask_unpackb_epi8(__m512i __W, __mmask64 __U, __m512i __A) {
+  return _mm512_mask_unpackb_epi8(__W, __U, __A, 64); // expected-error {{argument value 64 is outside the valid range [0, 63]}}
+}
+
+__m512i test_mm512_maskz_unpackb_epi8(__mmask64 __U, __m512i __A) {
+  return _mm512_maskz_unpackb_epi8(__U, __A, 64); // expected-error {{argument value 64 is outside the valid range [0, 63]}}
+}
diff --git a/clang/test/CodeGen/X86/avx10_2_v2aux-builtins.c b/clang/test/CodeGen/X86/avx10_2_v2aux-builtins.c
index c652af1e4d238d..36f96174c497ec 100644
--- a/clang/test/CodeGen/X86/avx10_2_v2aux-builtins.c
+++ b/clang/test/CodeGen/X86/avx10_2_v2aux-builtins.c
@@ -6,8 +6,9 @@
 #include <immintrin.h>
 
 //
-// Group A: VCVTPS2BF8 / VCVTPS2BF8S / VCVTPS2HF8 / VCVTPS2HF8S /
-//          VCVTROPS2HF8 / VCVTROPS2HF8S
+// Convert from FP32 to FP8
+// VCVTPS2BF8 / VCVTPS2BF8S / VCVTPS2HF8 / VCVTPS2HF8S /
+// VCVTROPS2HF8 / VCVTROPS2HF8S
 //
 
 // VCVTPS2BF8 - 128-bit
@@ -407,8 +408,9 @@ __m128i test_mm512_maskz_cvts_rops_hf8(__mmask16 __U, __m512 __A) {
 }
 
 //
-// Group B: VCVTBIASPS2BF8 / VCVTBIASPS2BF8S / VCVTBIASPS2HF8 /
-//          VCVTBIASPS2HF8S
+// Convert from FP32 to FP8 with bias
+// VCVTBIASPS2BF8 / VCVTBIASPS2BF8S / VCVTBIASPS2HF8 /
+// VCVTBIASPS2HF8S
 //
 
 // VCVTBIASPS2BF8 - 128-bit
@@ -676,7 +678,8 @@ __m128i test_mm512_maskz_cvts_biasps_hf8(__mmask16 __U, __m512i __A, __m512 __B)
 }
 
 //
-// Group C: VCVTBF82PS / VCVTHF82PS
+// Convert from FP8 to FP32
+// VCVTBF82PS / VCVTHF82PS
 //
 
 // VCVTBF82PS - 128-bit
@@ -812,7 +815,8 @@ __m512 test_mm512_maskz_cvthf8_ps(__mmask16 __U, __m128i __A) {
 }
 
 //
-// Group D: VCVTBF82BF4S / VCVTHF82BF4S (FP8 to FP4 truncating conversions)
+// Convert from FP8 to FP4
+// VCVTBF82BF4S / VCVTHF82BF4S
 //
 
 // VCVTBF82BF4S - register forms
@@ -896,7 +900,8 @@ void test_mm512_cvthf8_bf4s_storeu(void *__P, __m512i __A) {
 }
 
 //
-// Group E: VCVTBF82BF6S / VCVTHF82HF6S
+// Convert from FP8 to FP6
+// VCVTBF82BF6S / VCVTHF82HF6S
 //
 
 // VCVTBF82BF6S
@@ -940,7 +945,8 @@ __m512i test_mm512_cvthf8_hf6s(__m512i __A) {
 }
 
 //
-// Group F: VCVTBF42HF8 / VCVTBF62HF8 / VCVTHF62HF8
+// Convert from FP4 to FP8
+// VCVTBF42HF8
 //
 
 // VCVTBF42HF8 - 128-bit
@@ -1009,6 +1015,11 @@ __m512i test_mm512_maskz_cvtbf4_hf8(__mmask64 __U, __m256i __A) {
   return _mm512_maskz_cvtbf4_hf8(__U, __A);
 }
 
+//
+// Convert from FP6 to FP8
+// VCVTBF62HF8 / VCVTHF62HF8
+//
+
 // VCVTBF62HF8 - 128-bit
 
 __m128i test_mm_cvtbf6_hf8(__m128i __A) {
@@ -1142,7 +1153,8 @@ __m512i test_mm512_maskz_cvthf6_hf8(__mmask64 __U, __m512i __A) {
 }
 
 //
-// Group H: VUNPACKB
+// Unpack to Byte
+// VUNPACKB
 //
 
 // VUNPACKB - 128-bit
@@ -1212,9 +1224,72 @@ __m512i test_mm512_maskz_unpackb_epi8(__mmask64 __U, __m512i __A) {
 }
 
 //
-// VPMOVSSDB - Symmetric Signed Saturation DWord to Byte (memory store)
+// Down convert DWord to Byte with symmetric signed saturation
+// VPMOVSSDB
 //
 
+// VPMOVSSDB - 128-bit
+
+__m128i test_mm_cvtss_epi32_epi8(__m128i __A) {
+  // CHECK-LABEL: @test_mm_cvtss_epi32_epi8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.128(<4 x i32> %{{.*}}, <16 x i8> %{{.*}}, i8 -1)
+  return _mm_cvtss_epi32_epi8(__A);
+}
+
+__m128i test_mm_mask_cvtss_epi32_epi8(__m128i __W, __mmask8 __U, __m128i __A) {
+  // CHECK-LABEL: @test_mm_mask_cvtss_epi32_epi8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.128(<4 x i32> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
+  return _mm_mask_cvtss_epi32_epi8(__W, __U, __A);
+}
+
+__m128i test_mm_maskz_cvtss_epi32_epi8(__mmask8 __U, __m128i __A) {
+  // CHECK-LABEL: @test_mm_maskz_cvtss_epi32_epi8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.128(<4 x i32> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
+  return _mm_maskz_cvtss_epi32_epi8(__U, __A);
+}
+
+// VPMOVSSDB - 256-bit
+
+__m128i test_mm256_cvtss_epi32_epi8(__m256i __A) {
+  // CHECK-LABEL: @test_mm256_cvtss_epi32_epi8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.256(<8 x i32> %{{.*}}, <16 x i8> %{{.*}}, i8 -1)
+  return _mm256_cvtss_epi32_epi8(__A);
+}
+
+__m128i test_mm256_mask_cvtss_epi32_epi8(__m128i __W, __mmask8 __U, __m256i __A) {
+  // CHECK-LABEL: @test_mm256_mask_cvtss_epi32_epi8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.256(<8 x i32> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
+  return _mm256_mask_cvtss_epi32_epi8(__W, __U, __A);
+}
+
+__m128i test_mm256_maskz_cvtss_epi32_epi8(__mmask8 __U, __m256i __A) {
+  // CHECK-LABEL: @test_mm256_maskz_cvtss_epi32_epi8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.256(<8 x i32> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
+  return _mm256_maskz_cvtss_epi32_epi8(__U, __A);
+}
+
+// VPMOVSSDB - 512-bit
+
+__m128i test_mm512_cvtss_epi32_epi8(__m512i __A) {
+  // CHECK-LABEL: @test_mm512_cvtss_epi32_epi8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.512(<16 x i32> %{{.*}}, <16 x i8> %{{.*}}, i16 -1)
+  return _mm512_cvtss_epi32_epi8(__A);
+}
+
+__m128i test_mm512_mask_cvtss_epi32_epi8(__m128i __W, __mmask16 __U, __m512i __A) {
+  // CHECK-LABEL: @test_mm512_mask_cvtss_epi32_epi8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.512(<16 x i32> %{{.*}}, <16 x i8> %{{.*}}, i16 %{{.*}})
+  return _mm512_mask_cvtss_epi32_epi8(__W, __U, __A);
+}
+
+__m128i test_mm512_maskz_cvtss_epi32_epi8(__mmask16 __U, __m512i __A) {
+  // CHECK-LABEL: @test_mm512_maskz_cvtss_epi32_epi8(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.512(<16 x i32> %{{.*}}, <16 x i8> %{{.*}}, i16 %{{.*}})
+  return _mm512_maskz_cvtss_epi32_epi8(__U, __A);
+}
+
+// VPMOVSSDB - memory store
+
 void test_mm_mask_cvtss_epi32_storeu_epi8(void *__P, __mmask8 __M, __m128i __A) {
   // CHECK-LABEL: @test_mm_mask_cvtss_epi32_storeu_epi8(
   // CHECK: call void @llvm.x86.avx10.mask.pmovss.db.mem.128(ptr %{{.*}}, <4 x i32> %{{.*}}, i8 %{{.*}})
diff --git a/clang/test/Driver/x86-target-features.c b/clang/test/Driver/x86-target-features.c
index f18e820d154449..149bd50cc577f5 100644
--- a/clang/test/Driver/x86-target-features.c
+++ b/clang/test/Driver/x86-target-features.c
@@ -386,6 +386,11 @@
 // AVX10_2: "-target-feature" "+avx10.2"
 // NO-AVX10_2: "-target-feature" "-avx10.2"
 
+// RUN: %clang --target=i386 -mavx10v2aux %s -### -o %t.o 2>&1 -Werror | FileCheck -check-prefix=AVX10_V2_AUX %s
+// RUN: %clang --target=i386 -mno-avx10v2aux %s -### -o %t.o 2>&1 -Werror | FileCheck -check-prefix=NO-AVX10_V2_AUX %s
+// AVX10_V2_AUX: "-target-feature" "+avx10v2aux"
+// NO-AVX10_V2_AUX: "-target-feature" "-avx10v2aux"
+
 // RUN: %clang --target=i386 -musermsr %s -### -o %t.o 2>&1 | FileCheck -check-prefix=USERMSR %s
 // RUN: %clang --target=i386 -mno-usermsr %s -### -o %t.o 2>&1 | FileCheck -check-prefix=NO-USERMSR %s
 // USERMSR: "-target-feature" "+usermsr"
diff --git a/clang/test/Preprocessor/x86_target_features.c b/clang/test/Preprocessor/x86_target_features.c
index ac8ad64a469502..e3597dba7eee73 100644
--- a/clang/test/Preprocessor/x86_target_features.c
+++ b/clang/test/Preprocessor/x86_target_features.c
@@ -682,15 +682,23 @@
 // RUN: %clang -target i686-unknown-linux-gnu -march=atom -mavx10.1 -mno-avx512f -x c -E -dM -o - %s | FileCheck  -check-prefix=NO-AVX10_1 %s
 // RUN: %clang -target i686-unknown-linux-gnu -march=atom -mavx10.2 -x c -E -dM -o - %s | FileCheck  -check-prefixes=AVX10_1,AVX10_2 %s
 // RUN: %clang -target i686-unknown-linux-gnu -march=atom -mavx10.2 -mno-avx10.1 -x c -E -dM -o - %s | FileCheck  -check-prefixes=NO-AVX10_1,NO-AVX10_2 %s
+// RUN: %clang -target i686-unknown-linux-gnu -march=atom -mavx10v2aux -x c -E -dM -o - %s | FileCheck  -check-prefixes=AVX10_1,AVX10_V2_AUX %s
+// RUN: %clang -target i686-unknown-linux-gnu -march=atom -mavx10v2aux -x c -E -dM -o - %s | FileCheck  -check-prefix=NO-AVX10_2_ONLY %s
+// RUN: %clang -target i686-unknown-linux-gnu -march=atom -mno-avx10v2aux -x c -E -dM -o - %s | FileCheck  -check-prefix=NO-AVX10_V2_AUX %s
+// RUN: %clang -target i686-unknown-linux-gnu -march=atom -mavx10v2aux -mno-avx10.1 -x c -E -dM -o - %s | FileCheck  -check-prefix=NO-AVX10_V2_AUX %s
 // AVX10_1: #define __AVX10_1_512__ 1
 // AVX10_1: #define __AVX10_1__ 1
 // AVX10_2: #define __AVX10_2_512__ 1
 // AVX10_2: #define __AVX10_2__ 1
+// AVX10_V2_AUX: #define __AVX10_V2_AUX__ 1
 // AVX10_1: #define __AVX512F__ 1
 // NO-AVX10_1-NOT: __AVX10_1_512__
 // NO-AVX10_1-NOT: __AVX10_1__
 // NO-AVX10_1-NOT: __AVX10_2_512__
 // NO-AVX10_1-NOT: __AVX10_2__
+// NO-AVX10_V2_AUX-NOT: __AVX10_V2_AUX__
+// NO-AVX10_2_ONLY-NOT: __AVX10_2_512__
+// NO-AVX10_2_ONLY-NOT: __AVX10_2__
 // NO-AVX10_2: #define __AVX512F__ 1
 
 // RUN: %clang -target i686-unknown-linux-gnu -march=atom -musermsr -x c -E -dM -o - %s | FileCheck  -check-prefix=USERMSR %s
diff --git a/clang/test/Sema/builtins-x86.c b/clang/test/Sema/builtins-x86.c
index 51417ccdd733a7..0ae4e2cade1f3d 100644
--- a/clang/test/Sema/builtins-x86.c
+++ b/clang/test/Sema/builtins-x86.c
@@ -192,16 +192,4 @@ unsigned char test_lwpins64(unsigned long long data2, unsigned long long data1,
 
 void test_lwpval64(unsigned long long data2, unsigned long long data1, unsigned int flags) {
   __builtin_ia32_lwpval64(data2, data1, flags); // expected-error {{argument to '__builtin_ia32_lwpval64' must be a constant integer}}
-}
-
-__m128i test__builtin_ia32_vunpackb128(__m128i __a) {
-  return __builtin_ia32_vunpackb128(__a, 64); // expected-error {{argument value 64 is outside the valid range [0, 63]}}
-}
-
-__m256i test__builtin_ia32_vunpackb256(__m256i __a) {
-  return __builtin_ia32_vunpackb256(__a, 64); // expected-error {{argument value 64 is outside the valid range [0, 63]}}
-}
-
-__m512i test__builtin_ia32_vunpackb512(__m512i __a) {
-  return __builtin_ia32_vunpackb512(__a, 64); // expected-error {{argument value 64 is outside the valid range [0, 63]}}
-}
+}
\ No newline at end of file
diff --git a/llvm/include/llvm/IR/IntrinsicsX86.td b/llvm/include/llvm/IR/IntrinsicsX86.td
index 79180301dff1d0..8ccc2a50b4258e 100644
--- a/llvm/include/llvm/IR/IntrinsicsX86.td
+++ b/llvm/include/llvm/IR/IntrinsicsX86.td
@@ -7136,102 +7136,102 @@ def int_x86_avx10_vcvtbiasps2hf8s_512 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2
 // Group C: 8bit -> PS expanding conversions
 
 // VCVTBF82PS
-def int_x86_avx10_vcvtbf82ps_128 : ClangBuiltin<"__builtin_ia32_vcvtbf8_2ps128">,
+def int_x86_avx10_vcvtbf82ps_128 : ClangBuiltin<"__builtin_ia32_vcvtbf82ps_128">,
         DefaultAttrsIntrinsic<[llvm_v4f32_ty], [llvm_v16i8_ty], [IntrNoMem]>;
-def int_x86_avx10_vcvtbf82ps_256 : ClangBuiltin<"__builtin_ia32_vcvtbf8_2ps256">,
+def int_x86_avx10_vcvtbf82ps_256 : ClangBuiltin<"__builtin_ia32_vcvtbf82ps_256">,
         DefaultAttrsIntrinsic<[llvm_v8f32_ty], [llvm_v16i8_ty], [IntrNoMem]>;
-def int_x86_avx10_vcvtbf82ps_512 : ClangBuiltin<"__builtin_ia32_vcvtbf8_2ps512">,
+def int_x86_avx10_vcvtbf82ps_512 : ClangBuiltin<"__builtin_ia32_vcvtbf82ps_512">,
         DefaultAttrsIntrinsic<[llvm_v16f32_ty], [llvm_v16i8_ty], [IntrNoMem]>;
 
 // VCVTHF82PS
-def int_x86_avx10_vcvthf82ps_128 : ClangBuiltin<"__builtin_ia32_vcvthf8_2ps128">,
+def int_x86_avx10_vcvthf82ps_128 : ClangBuiltin<"__builtin_ia32_vcvthf82ps_128">,
         DefaultAttrsIntrinsic<[llvm_v4f32_ty], [llvm_v16i8_ty], [IntrNoMem]>;
-def int_x86_avx10_vcvthf82ps_256 : ClangBuiltin<"__builtin_ia32_vcvthf8_2ps256">,
+def int_x86_avx10_vcvthf82ps_256 : ClangBuiltin<"__builtin_ia32_vcvthf82ps_256">,
         DefaultAttrsIntrinsic<[llvm_v8f32_ty], [llvm_v16i8_ty], [IntrNoMem]>;
-def int_x86_avx10_vcvthf82ps_512 : ClangBuiltin<"__builtin_ia32_vcvthf8_2ps512">,
+def int_x86_avx10_vcvthf82ps_512 : ClangBuiltin<"__builtin_ia32_vcvthf82ps_512">,
         DefaultAttrsIntrinsic<[llvm_v16f32_ty], [llvm_v16i8_ty], [IntrNoMem]>;
 
 // Group E: Same-size reg-only conversions (no masking)
 
 // VCVTBF82BF6S
-def int_x86_avx10_vcvtbf82bf6s_128 : ClangBuiltin<"__builtin_ia32_vcvtbf82bf6s128">,
+def int_x86_avx10_vcvtbf82bf6s_128 : ClangBuiltin<"__builtin_ia32_vcvtbf82bf6s_128">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16i8_ty], [IntrNoMem]>;
-def int_x86_avx10_vcvtbf82bf6s_256 : ClangBuiltin<"__builtin_ia32_vcvtbf82bf6s256">,
+def int_x86_avx10_vcvtbf82bf6s_256 : ClangBuiltin<"__builtin_ia32_vcvtbf82bf6s_256">,
         DefaultAttrsIntrinsic<[llvm_v32i8_ty], [llvm_v32i8_ty], [IntrNoMem]>;
-def int_x86_avx10_vcvtbf82bf6s_512 : ClangBuiltin<"__builtin_ia32_vcvtbf82bf6s512">,
+def int_x86_avx10_vcvtbf82bf6s_512 : ClangBuiltin<"__builtin_ia32_vcvtbf82bf6s_512">,
         DefaultAttrsIntrinsic<[llvm_v64i8_ty], [llvm_v64i8_ty], [IntrNoMem]>;
 
 // VCVTHF82HF6S
-def int_x86_avx10_vcvthf82hf6s_128 : ClangBuiltin<"__builtin_ia32_vcvthf82hf6s128">,
+def int_x86_avx10_vcvthf82hf6s_128 : ClangBuiltin<"__builtin_ia32_vcvthf82hf6s_128">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16i8_ty], [IntrNoMem]>;
-def int_x86_avx10_vcvthf82hf6s_256 : ClangBuiltin<"__builtin_ia32_vcvthf82hf6s256">,
+def int_x86_avx10_vcvthf82hf6s_256 : ClangBuiltin<"__builtin_ia32_vcvthf82hf6s_256">,
         DefaultAttrsIntrinsic<[llvm_v32i8_ty], [llvm_v32i8_ty], [IntrNoMem]>;
-def int_x86_avx10_vcvthf82hf6s_512 : ClangBuiltin<"__builtin_ia32_vcvthf82hf6s512">,
+def int_x86_avx10_vcvthf82hf6s_512 : ClangBuiltin<"__builtin_ia32_vcvthf82hf6s_512">,
         DefaultAttrsIntrinsic<[llvm_v64i8_ty], [llvm_v64i8_ty], [IntrNoMem]>;
 
 // Group F: Expanding/same-size conversions with masking
 
 // VCVTBF42HF8: expanding (half-size input -> full output)
-def int_x86_avx10_vcvtbf42hf8_128 : ClangBuiltin<"__builtin_ia32_vcvtbf42hf8128">,
+def int_x86_avx10_vcvtbf42hf8_128 : ClangBuiltin<"__builtin_ia32_vcvtbf42hf8_128">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16i8_ty], [IntrNoMem]>;
-def int_x86_avx10_vcvtbf42hf8_256 : ClangBuiltin<"__builtin_ia32_vcvtbf42hf8256">,
+def int_x86_avx10_vcvtbf42hf8_256 : ClangBuiltin<"__builtin_ia32_vcvtbf42hf8_256">,
         DefaultAttrsIntrinsic<[llvm_v32i8_ty], [llvm_v16i8_ty], [IntrNoMem]>;
-def int_x86_avx10_vcvtbf42hf8_512 : ClangBuiltin<"__builtin_ia32_vcvtbf42hf8512">,
+def int_x86_avx10_vcvtbf42hf8_512 : ClangBuiltin<"__builtin_ia32_vcvtbf42hf8_512">,
         DefaultAttrsIntrinsic<[llvm_v64i8_ty], [llvm_v32i8_ty], [IntrNoMem]>;
 
 // VCVTBF62HF8: same-size, reg-only
-def int_x86_avx10_vcvtbf62hf8_128 : ClangBuiltin<"__builtin_ia32_vcvtbf62hf8128">,
+def int_x86_avx10_vcvtbf62hf8_128 : ClangBuiltin<"__builtin_ia32_vcvtbf62hf8_128">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16i8_ty], [IntrNoMem]>;
-def int_x86_avx10_vcvtbf62hf8_256 : ClangBuiltin<"__builtin_ia32_vcvtbf62hf8256">,
+def int_x86_avx10_vcvtbf62hf8_256 : ClangBuiltin<"__builtin_ia32_vcvtbf62hf8_256">,
         DefaultAttrsIntrinsic<[llvm_v32i8_ty], [llvm_v32i8_ty], [IntrNoMem]>;
-def int_x86_avx10_vcvtbf62hf8_512 : ClangBuiltin<"__builtin_ia32_vcvtbf62hf8512">,
+def int_x86_avx10_vcvtbf62hf8_512 : ClangBuiltin<"__builtin_ia32_vcvtbf62hf8_512">,
         DefaultAttrsIntrinsic<[llvm_v64i8_ty], [llvm_v64i8_ty], [IntrNoMem]>;
 
 // VCVTHF62HF8: same-size, reg-only
-def int_x86_avx10_vcvthf62hf8_128 : ClangBuiltin<"__builtin_ia32_vcvthf62hf8128">,
+def int_x86_avx10_vcvthf62hf8_128 : ClangBuiltin<"__builtin_ia32_vcvthf62hf8_128">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16i8_ty], [IntrNoMem]>;
-def int_x86_avx10_vcvthf62hf8_256 : ClangBuiltin<"__builtin_ia32_vcvthf62hf8256">,
+def int_x86_avx10_vcvthf62hf8_256 : ClangBuiltin<"__builtin_ia32_vcvthf62hf8_256">,
         DefaultAttrsIntrinsic<[llvm_v32i8_ty], [llvm_v32i8_ty], [IntrNoMem]>;
-def int_x86_avx10_vcvthf62hf8_512 : ClangBuiltin<"__builtin_ia32_vcvthf62hf8512">,
+def int_x86_avx10_vcvthf62hf8_512 : ClangBuiltin<"__builtin_ia32_vcvthf62hf8_512">,
         DefaultAttrsIntrinsic<[llvm_v64i8_ty], [llvm_v64i8_ty], [IntrNoMem]>;
 
 // Group D: VCVTBF82BF4S / VCVTHF82BF4S - FP8 to FP4 truncating conversions
 
 // VCVTBF82BF4S: FP8 E5M2 to FP4 E2M1 with saturation (truncating, output half size)
-def int_x86_avx10_vcvtbf82bf4s_128 : ClangBuiltin<"__builtin_ia32_vcvtbf82bf4s128">,
+def int_x86_avx10_vcvtbf82bf4s_128 : ClangBuiltin<"__builtin_ia32_vcvtbf82bf4s_128">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16i8_ty], [IntrNoMem]>;
-def int_x86_avx10_vcvtbf82bf4s_256 : ClangBuiltin<"__builtin_ia32_vcvtbf82bf4s256">,
+def int_x86_avx10_vcvtbf82bf4s_256 : ClangBuiltin<"__builtin_ia32_vcvtbf82bf4s_256">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v32i8_ty], [IntrNoMem]>;
-def int_x86_avx10_vcvtbf82bf4s_512 : ClangBuiltin<"__builtin_ia32_vcvtbf82bf4s512">,
+def int_x86_avx10_vcvtbf82bf4s_512 : ClangBuiltin<"__builtin_ia32_vcvtbf82bf4s_512">,
         DefaultAttrsIntrinsic<[llvm_v32i8_ty], [llvm_v64i8_ty], [IntrNoMem]>;
 
 // VCVTHF82BF4S: FP8 E4M3 to FP4 E2M1 with saturation (truncating, output half size)
-def int_x86_avx10_vcvthf82bf4s_128 : ClangBuiltin<"__builtin_ia32_vcvthf82bf4s128">,
+def int_x86_avx10_vcvthf82bf4s_128 : ClangBuiltin<"__builtin_ia32_vcvthf82bf4s_128">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16i8_ty], [IntrNoMem]>;
-def int_x86_avx10_vcvthf82bf4s_256 : ClangBuiltin<"__builtin_ia32_vcvthf82bf4s256">,
+def int_x86_avx10_vcvthf82bf4s_256 : ClangBuiltin<"__builtin_ia32_vcvthf82bf4s_256">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v32i8_ty], [IntrNoMem]>;
-def int_x86_avx10_vcvthf82bf4s_512 : ClangBuiltin<"__builtin_ia32_vcvthf82bf4s512">,
+def int_x86_avx10_vcvthf82bf4s_512 : ClangBuiltin<"__builtin_ia32_vcvthf82bf4s_512">,
         DefaultAttrsIntrinsic<[llvm_v32i8_ty], [llvm_v64i8_ty], [IntrNoMem]>;
 
 // VCVTBF82BF4S memory store: FP8 E5M2 to FP4 E2M1, store to memory
-def int_x86_avx10_vcvtbf82bf4s_128_mem : ClangBuiltin<"__builtin_ia32_vcvtbf82bf4s128mem">,
+def int_x86_avx10_vcvtbf82bf4s_128_mem : ClangBuiltin<"__builtin_ia32_vcvtbf82bf4s_128_mem">,
         DefaultAttrsIntrinsic<[], [llvm_ptr_ty, llvm_v16i8_ty],
                               [IntrWriteMem, IntrArgMemOnly, NoCapture<ArgIndex<0>>]>;
-def int_x86_avx10_vcvtbf82bf4s_256_mem : ClangBuiltin<"__builtin_ia32_vcvtbf82bf4s256mem">,
+def int_x86_avx10_vcvtbf82bf4s_256_mem : ClangBuiltin<"__builtin_ia32_vcvtbf82bf4s_256_mem">,
         DefaultAttrsIntrinsic<[], [llvm_ptr_ty, llvm_v32i8_ty],
                               [IntrWriteMem, IntrArgMemOnly, NoCapture<ArgIndex<0>>]>;
-def int_x86_avx10_vcvtbf82bf4s_512_mem : ClangBuiltin<"__builtin_ia32_vcvtbf82bf4s512mem">,
+def int_x86_avx10_vcvtbf82bf4s_512_mem : ClangBuiltin<"__builtin_ia32_vcvtbf82bf4s_512_mem">,
         DefaultAttrsIntrinsic<[], [llvm_ptr_ty, llvm_v64i8_ty],
                               [IntrWriteMem, IntrArgMemOnly, NoCapture<ArgIndex<0>>]>;
 
 // VCVTHF82BF4S memory store: FP8 E4M3 to FP4 E2M1, store to memory
-def int_x86_avx10_vcvthf82bf4s_128_mem : ClangBuiltin<"__builtin_ia32_vcvthf82bf4s128mem">,
+def int_x86_avx10_vcvthf82bf4s_128_mem : ClangBuiltin<"__builtin_ia32_vcvthf82bf4s_128_mem">,
         DefaultAttrsIntrinsic<[], [llvm_ptr_ty, llvm_v16i8_ty],
                               [IntrWriteMem, IntrArgMemOnly, NoCapture<ArgIndex<0>>]>;
-def int_x86_avx10_vcvthf82bf4s_256_mem : ClangBuiltin<"__builtin_ia32_vcvthf82bf4s256mem">,
+def int_x86_avx10_vcvthf82bf4s_256_mem : ClangBuiltin<"__builtin_ia32_vcvthf82bf4s_256_mem">,
         DefaultAttrsIntrinsic<[], [llvm_ptr_ty, llvm_v32i8_ty],
                               [IntrWriteMem, IntrArgMemOnly, NoCapture<ArgIndex<0>>]>;
-def int_x86_avx10_vcvthf82bf4s_512_mem : ClangBuiltin<"__builtin_ia32_vcvthf82bf4s512mem">,
+def int_x86_avx10_vcvthf82bf4s_512_mem : ClangBuiltin<"__builtin_ia32_vcvthf82bf4s_512_mem">,
         DefaultAttrsIntrinsic<[], [llvm_ptr_ty, llvm_v64i8_ty],
                               [IntrWriteMem, IntrArgMemOnly, NoCapture<ArgIndex<0>>]>;
 
diff --git a/llvm/include/llvm/TargetParser/X86TargetParser.def b/llvm/include/llvm/TargetParser/X86TargetParser.def
index 5f5749308f3414..03153045ae4cea 100644
--- a/llvm/include/llvm/TargetParser/X86TargetParser.def
+++ b/llvm/include/llvm/TargetParser/X86TargetParser.def
@@ -265,7 +265,7 @@ X86_FEATURE       (NDD,                "ndd")
 X86_FEATURE       (EGPR,               "egpr")
 X86_FEATURE       (ZU,                 "zu")
 X86_FEATURE       (JMPABS,             "jmpabs")
-X86_FEATURE_COMPAT(AVX10_V2_AUX,       "avx10v2aux",          36, 123)
+X86_FEATURE       (AVX10_V2_AUX,       "avx10v2aux")
 
 // These features aren't really CPU features, but the frontend can set them.
 X86_FEATURE       (RETPOLINE_EXTERNAL_THUNK,    "retpoline-external-thunk")
@@ -275,7 +275,7 @@ X86_FEATURE       (LVI_CFI,                     "lvi-cfi")
 X86_FEATURE       (LVI_LOAD_HARDENING,          "lvi-load-hardening")
 
 // Max number of priorities. Priorities form a consecutive range.
-#define MAX_PRIORITY 36
+#define MAX_PRIORITY 35
 
 #undef X86_FEATURE_COMPAT
 #undef X86_FEATURE
diff --git a/llvm/lib/Target/X86/X86.td b/llvm/lib/Target/X86/X86.td
index 9057dc38f08774..d2b418ebf05428 100644
--- a/llvm/lib/Target/X86/X86.td
+++ b/llvm/lib/Target/X86/X86.td
@@ -362,8 +362,8 @@ def FeatureAVX10_2_512 : SubtargetFeature<"avx10.2-512", "HasAVX10_2_512", "true
                                           "Support AVX10.2 instruction",
                                           [FeatureAVX10_2]>;
 def FeatureAVX10_V2_AUX : SubtargetFeature<"avx10v2aux", "HasAVX10_V2_AUX", "true",
-                                          "Support AVX10 V2 AUX instructions",
-                                          [FeatureAVX10_2]>;
+                                           "Support AVX10 V2 AUX instructions",
+                                           [FeatureAVX10_1]>;
 def FeatureEGPR : SubtargetFeature<"egpr", "HasEGPR", "true",
                                    "Support extended general purpose register">;
 def FeaturePush2Pop2 : SubtargetFeature<"push2pop2", "HasPush2Pop2", "true",
diff --git a/llvm/lib/Target/X86/X86InstrAVX10_V2_AUX.td b/llvm/lib/Target/X86/X86InstrAVX10_V2_AUX.td
index ba77c5fd240d2e..a0fa96157a7599 100644
--- a/llvm/lib/Target/X86/X86InstrAVX10_V2_AUX.td
+++ b/llvm/lib/Target/X86/X86InstrAVX10_V2_AUX.td
@@ -15,7 +15,7 @@
 // AVX10 V2 AUX Multiclass Definitions
 //===----------------------------------------------------------------------===//
 
-// Group A multiclass: PS(f32) -> i8 truncating conversion (quarter-size output)
+// Convert from FP32 to FP8: truncating conversion, quarter-size output
 // Output is always xmm for all VL variants.
 // Adapts avx512_vcvt_fp for each VL level with explicit dest/src type mappings.
 multiclass avx10_v2aux_cvt_trunc_ps2i8<bits<8> opc, string OpcodeStr,
@@ -126,7 +126,7 @@ multiclass avx10_v2aux_cvt_trunc_ps2i8<bits<8> opc, string OpcodeStr,
             (!cast<Instruction>(NAME # "Z128rmbkz") VK4WM:$mask, addr:$src)>;
 }
 
-// Group B multiclass: 3-operand bias PS(f32) -> i8 conversion (quarter-size output)
+// Convert from FP32 to FP8 with bias: 3-operand, quarter-size output
 // bias + f32 source -> i8 dest
 // Reuses avx10_convert_3op_packed from X86InstrAVX10.td for each VL variant.
 multiclass avx10_v2aux_cvt_3op_ps<bits<8> opc, string OpcodeStr,
@@ -224,7 +224,7 @@ multiclass avx10_v2aux_cvt_3op_ps<bits<8> opc, string OpcodeStr,
                                 VR128X:$src1, addr:$src2)>;
 }
 
-// Group C multiclass: i8 -> f32 expanding conversion (4x expansion, no broadcast)
+// Convert from FP8 to FP32: expanding conversion (4x, no broadcast)
 multiclass avx10_v2aux_cvt_2op_i8_to_f32<bits<8> opc, string OpcodeStr,
                                          SDNode OpNode> {
   defm Z : avx10_convert_2op_nomb_packed<opc, OpcodeStr, v16f32_info,
@@ -238,86 +238,122 @@ multiclass avx10_v2aux_cvt_2op_i8_to_f32<bits<8> opc, string OpcodeStr,
                                             WriteCvtPH2PSZ>, EVEX_V256;
 }
 
-// Group D multiclass: Truncating conversion (reg/mem dest, no masking)
+// Convert from FP8 to FP4: truncating conversion (reg/mem dest, no masking)
 // Uses MRMDestReg/MRMDestMem since destination can be memory operand.
 // Source is in reg field, destination is in r/m field.
-multiclass avx10_v2aux_cvt_trunc<bits<8> opc, string OpcodeStr,
-                                  X86VectorVTInfo SrcInfo,
-                                  X86VectorVTInfo DestInfo,
-                                  X86MemOperand x86memop,
-                                  SDPatternOperator OpNode> {
+multiclass avx10_v2aux_cvt_trunc_base<bits<8> opc, string OpcodeStr,
+                                      X86VectorVTInfo _src,
+                                      X86VectorVTInfo _dest,
+                                      X86MemOperand x86memop,
+                                      X86FoldableSchedWrite sched> {
   let hasSideEffects = 0 in {
-    def rr : I<opc, MRMDestReg, (outs DestInfo.RC:$dst),
-               (ins SrcInfo.RC:$src),
+    def rr : I<opc, MRMDestReg, (outs _dest.RC:$dst),
+               (ins _src.RC:$src),
                OpcodeStr # "\t{$src, $dst|$dst, $src}", []>,
-               EVEX, Sched<[WriteCvtPH2PSZ]>;
+             Sched<[sched]>, EVEX, EVEX_CD8<8, CD8VH>;
     let mayStore = 1 in
-    def mr : I<opc, MRMDestMem, (outs),
-               (ins x86memop:$dst, SrcInfo.RC:$src),
-               OpcodeStr # "\t{$src, $dst|$dst, $src}", []>,
-               EVEX, Sched<[WriteCvtPH2PSZ.Folded]>;
+      def mr : I<opc, MRMDestMem, (outs),
+                 (ins x86memop:$dst, _src.RC:$src),
+                 OpcodeStr # "\t{$src, $dst|$dst, $src}", []>,
+               Sched<[sched.Folded]>, EVEX, EVEX_CD8<8, CD8VH>;
   }
 
-  // Intrinsic pattern
-  def : Pat<(DestInfo.VT (OpNode (SrcInfo.VT SrcInfo.RC:$src))),
-            (!cast<Instruction>(NAME # "rr") SrcInfo.RC:$src)>;
+  def : Pat<(_dest.VT (!cast<Intrinsic>("int_x86_avx10_"#OpcodeStr#"_"#_src.Size)
+                        (_src.VT _src.RC:$src))),
+            (!cast<Instruction>(NAME # "rr") _src.RC:$src)>;
+  def : Pat<(!cast<Intrinsic>("int_x86_avx10_"#OpcodeStr#"_"#_src.Size#"_mem")
+                addr:$dst, (_src.VT _src.RC:$src)),
+            (!cast<Instruction>(NAME # "mr") addr:$dst, _src.RC:$src)>;
+}
+
+multiclass avx10_v2aux_cvt_trunc_b<bits<8> opc, string OpcodeStr> {
+  defm Z    : avx10_v2aux_cvt_trunc_base<opc, OpcodeStr, v64i8_info,
+                                         v32i8x_info, i256mem,
+                                         WriteCvtPH2PSZ>, EVEX_V512;
+  defm Z256 : avx10_v2aux_cvt_trunc_base<opc, OpcodeStr, v32i8x_info,
+                                         v16i8x_info, i128mem,
+                                         WriteCvtPH2PSY>, EVEX_V256;
+  defm Z128 : avx10_v2aux_cvt_trunc_base<opc, OpcodeStr, v16i8x_info,
+                                         v16i8x_info, i64mem,
+                                         WriteCvtPH2PS>, EVEX_V128;
 }
 
-// Group F helper: expanding conversion with masking, no broadcast (reg+mem)
-multiclass avx10_v2aux_cvt_expand_masked<bits<8> opc, string OpcodeStr,
-                                         X86VectorVTInfo _dest,
-                                         X86VectorVTInfo _src,
-                                         X86MemOperand x86memop,
-                                         SDPatternOperator OpNode> {
-  let ExeDomain = _dest.ExeDomain in {
-    defm rr : AVX512_maskable<opc, MRMSrcReg, _dest,
-                (outs _dest.RC:$dst),
-                (ins _src.RC:$src),
-                OpcodeStr, "$src", "$src",
-                (_dest.VT (OpNode (_src.VT _src.RC:$src)))>,
-               EVEX, Sched<[WriteCvtPH2PSZ]>;
+// Convert from FP4 to FP8: expanding with masking, no broadcast (reg+mem)
+multiclass avx10_v2aux_cvt_expand_masked_base<bits<8> opc, string OpcodeStr,
+                                              X86VectorVTInfo _,
+                                              X86VectorVTInfo _src,
+                                              X86MemOperand x86memop,
+                                              X86FoldableSchedWrite sched> {
+  let ExeDomain = _.ExeDomain in {
+    defm rr : AVX512_maskable<opc, MRMSrcReg, _, (outs _.RC:$dst),
+                              (ins _src.RC:$src),
+                              OpcodeStr, "$src", "$src",
+                              (_.VT (!cast<Intrinsic>("int_x86_avx10_"#OpcodeStr#"_"#_.Size)
+                                     (_src.VT _src.RC:$src)))>,
+                              Sched<[sched]>, EVEX, EVEX_CD8<8, CD8VH>;
     let mayLoad = 1 in
-    defm rm : AVX512_maskable<opc, MRMSrcMem, _dest,
-                (outs _dest.RC:$dst),
-                (ins x86memop:$src),
-                OpcodeStr, "$src", "$src",
-                (_dest.VT (OpNode (_src.VT (load addr:$src))))>,
-               EVEX, Sched<[WriteCvtPH2PSZ.Folded]>;
+      defm rm : AVX512_maskable<opc, MRMSrcMem, _, (outs _.RC:$dst),
+                                (ins x86memop:$src),
+                                OpcodeStr, "$src", "$src",
+                                (_.VT (!cast<Intrinsic>("int_x86_avx10_"#OpcodeStr#"_"#_.Size)
+                                       (_src.VT (load addr:$src))))>,
+                                Sched<[sched.Folded]>, EVEX, EVEX_CD8<8, CD8VH>;
   }
 }
 
-// Group F helper: widening conversion with masking (reg-only, 6-bit to 8-bit)
-multiclass avx10_v2aux_cvt_widen_masked<bits<8> opc, string OpcodeStr,
-                                        X86VectorVTInfo _,
-                                        SDPatternOperator OpNode> {
+multiclass avx10_v2aux_cvt_expand_masked_b<bits<8> opc, string OpcodeStr> {
+  defm Z    : avx10_v2aux_cvt_expand_masked_base<opc, OpcodeStr, v64i8_info,
+                                                 v32i8x_info, f256mem,
+                                                 WriteCvtPH2PSZ>, EVEX_V512;
+  defm Z256 : avx10_v2aux_cvt_expand_masked_base<opc, OpcodeStr, v32i8x_info,
+                                                 v16i8x_info, f128mem,
+                                                 WriteCvtPH2PSY>, EVEX_V256;
+  defm Z128 : avx10_v2aux_cvt_expand_masked_base<opc, OpcodeStr, v16i8x_info,
+                                                 v16i8x_info, f64mem,
+                                                 WriteCvtPH2PS>, EVEX_V128;
+}
+
+// Convert from FP6 to FP8: widening conversion with masking (reg-only)
+multiclass avx10_v2aux_cvt_widen_masked_base<bits<8> opc, string OpcodeStr,
+                                             X86VectorVTInfo _,
+                                             X86FoldableSchedWrite sched> {
   let ExeDomain = _.ExeDomain in {
-    defm rr : AVX512_maskable<opc, MRMSrcReg, _,
-                (outs _.RC:$dst),
-                (ins _.RC:$src),
-                OpcodeStr, "$src", "$src",
-                (_.VT (OpNode (_.VT _.RC:$src)))>,
-               EVEX, Sched<[WriteCvtPH2PSZ]>;
+    defm rr : AVX512_maskable<opc, MRMSrcReg, _, (outs _.RC:$dst),
+                              (ins _.RC:$src),
+                              OpcodeStr, "$src", "$src",
+                              (_.VT (!cast<Intrinsic>("int_x86_avx10_"#OpcodeStr#"_"#_.Size)
+                                     (_.VT _.RC:$src)))>,
+                              Sched<[sched]>, EVEX;
   }
 }
 
-// Group H multiclass: Byte unpack with immediate
+multiclass avx10_v2aux_cvt_widen_masked_b<bits<8> opc, string OpcodeStr> {
+  defm Z    : avx10_v2aux_cvt_widen_masked_base<opc, OpcodeStr, v64i8_info,
+                                                WriteCvtPH2PSZ>, EVEX_V512;
+  defm Z256 : avx10_v2aux_cvt_widen_masked_base<opc, OpcodeStr, v32i8x_info,
+                                                WriteCvtPH2PSY>, EVEX_V256;
+  defm Z128 : avx10_v2aux_cvt_widen_masked_base<opc, OpcodeStr, v16i8x_info,
+                                                WriteCvtPH2PS>, EVEX_V128;
+}
+
+// Unpack to Byte: byte unpack with immediate
 multiclass avx10_v2aux_shuffle_base<bits<8> opc, string OpcodeStr,
                                     X86VectorVTInfo _,
                                     X86FoldableSchedWrite sched> {
   let ImmT = Imm8 in {
     defm rri : AVX512_maskable<opc, MRMSrcReg, _, (outs _.RC:$dst),
-                    (ins _.RC:$src1, u8imm:$src2),
-                    OpcodeStr, "$src2, $src1", "$src1, $src2",
-                    (_.VT (!cast<Intrinsic>("int_x86_avx10_"#OpcodeStr#"_"#_.Size)
-                            (_.VT _.RC:$src1), (i8 timm:$src2)))>,
-                    Sched<[sched]>, EVEX, EVEX_CD8<8, CD8VF>;
+                               (ins _.RC:$src1, u8imm:$src2),
+                               OpcodeStr, "$src2, $src1", "$src1, $src2",
+                               (_.VT (!cast<Intrinsic>("int_x86_avx10_"#OpcodeStr#"_"#_.Size)
+                                      (_.VT _.RC:$src1), (i8 timm:$src2)))>,
+                               Sched<[sched]>, EVEX, EVEX_CD8<8, CD8VF>;
     let mayLoad = 1 in
       defm rmi : AVX512_maskable<opc, MRMSrcMem, _, (outs _.RC:$dst),
-                      (ins _.MemOp:$src1, u8imm:$src2),
-                      OpcodeStr, "$src2, $src1", "$src1, $src2",
-                      (_.VT (!cast<Intrinsic>("int_x86_avx10_"#OpcodeStr#"_"#_.Size)
-                              (_.VT (load addr:$src1)), (i8 timm:$src2)))>,
-                      Sched<[sched.Folded]>, EVEX, EVEX_CD8<8, CD8VF>;
+                                 (ins _.MemOp:$src1, u8imm:$src2),
+                                 OpcodeStr, "$src2, $src1", "$src1, $src2",
+                                 (_.VT (!cast<Intrinsic>("int_x86_avx10_"#OpcodeStr#"_"#_.Size)
+                                        (_.VT (load addr:$src1)), (i8 timm:$src2)))>,
+                                 Sched<[sched.Folded]>, EVEX, EVEX_CD8<8, CD8VF>;
   }
 }
 
@@ -335,7 +371,7 @@ multiclass avx10_v2aux_shuffle_b<bits<8> opc, string OpcodeStr> {
 //===----------------------------------------------------------------------===//
 
 //-------------------------------------------------
-// Group A: PS->8bit truncating conversions
+// Convert from FP32 to FP8
 //-------------------------------------------------
 
 let Predicates = [HasAVX10_V2_AUX] in {
@@ -360,7 +396,7 @@ let Predicates = [HasAVX10_V2_AUX] in {
 }
 
 //-------------------------------------------------
-// Group B: Bias PS->8bit conversions (3-operand)
+// Convert from FP32 to FP8 with bias
 //-------------------------------------------------
 
 let Predicates = [HasAVX10_V2_AUX] in {
@@ -383,7 +419,7 @@ let Predicates = [HasAVX10_V2_AUX] in {
 }
 
 //-------------------------------------------------
-// Group C: 8bit->PS expanding conversions
+// Convert from FP8 to FP32
 //-------------------------------------------------
 
 let Predicates = [HasAVX10_V2_AUX] in {
@@ -396,148 +432,71 @@ let Predicates = [HasAVX10_V2_AUX] in {
 }
 
 //-------------------------------------------------
-// Group D: BF8/HF8->BF4S truncating conversions
+// Convert from FP8 to FP4
 //-------------------------------------------------
 
 let Predicates = [HasAVX10_V2_AUX] in {
-  defm VCVTBF82BF4SZ    : avx10_v2aux_cvt_trunc<0x3D, "vcvtbf82bf4s",
-                                                v64i8_info, v32i8x_info, i256mem,
-                                                int_x86_avx10_vcvtbf82bf4s_512>,
-                          T_MAP5, XS, REX_W, EVEX_V512, EVEX_CD8<8, CD8VH>;
-  defm VCVTBF82BF4SZ256 : avx10_v2aux_cvt_trunc<0x3D, "vcvtbf82bf4s",
-                                                v32i8x_info, v16i8x_info, i128mem,
-                                                int_x86_avx10_vcvtbf82bf4s_256>,
-                          T_MAP5, XS, REX_W, EVEX_V256, EVEX_CD8<8, CD8VH>;
-  defm VCVTBF82BF4SZ128 : avx10_v2aux_cvt_trunc<0x3D, "vcvtbf82bf4s",
-                                                v16i8x_info, v16i8x_info, i64mem,
-                                                int_x86_avx10_vcvtbf82bf4s_128>,
-                          T_MAP5, XS, REX_W, EVEX_V128, EVEX_CD8<8, CD8VH>;
-
-  defm VCVTHF82BF4SZ    : avx10_v2aux_cvt_trunc<0x3D, "vcvthf82bf4s",
-                                                v64i8_info, v32i8x_info, i256mem,
-                                                int_x86_avx10_vcvthf82bf4s_512>,
-                          T_MAP5, XS, EVEX_V512, EVEX_CD8<8, CD8VH>;
-  defm VCVTHF82BF4SZ256 : avx10_v2aux_cvt_trunc<0x3D, "vcvthf82bf4s",
-                                                v32i8x_info, v16i8x_info, i128mem,
-                                                int_x86_avx10_vcvthf82bf4s_256>,
-                          T_MAP5, XS, EVEX_V256, EVEX_CD8<8, CD8VH>;
-  defm VCVTHF82BF4SZ128 : avx10_v2aux_cvt_trunc<0x3D, "vcvthf82bf4s",
-                                                v16i8x_info, v16i8x_info, i64mem,
-                                                int_x86_avx10_vcvthf82bf4s_128>,
-                          T_MAP5, XS, EVEX_V128, EVEX_CD8<8, CD8VH>;
-
-  // Memory store patterns for VCVTBF82BF4S
-  def : Pat<(int_x86_avx10_vcvtbf82bf4s_512_mem addr:$dst, VR512:$src),
-            (VCVTBF82BF4SZmr addr:$dst, VR512:$src)>;
-  def : Pat<(int_x86_avx10_vcvtbf82bf4s_256_mem addr:$dst, VR256X:$src),
-            (VCVTBF82BF4SZ256mr addr:$dst, VR256X:$src)>;
-  def : Pat<(int_x86_avx10_vcvtbf82bf4s_128_mem addr:$dst, VR128X:$src),
-            (VCVTBF82BF4SZ128mr addr:$dst, VR128X:$src)>;
-
-  // Memory store patterns for VCVTHF82BF4S
-  def : Pat<(int_x86_avx10_vcvthf82bf4s_512_mem addr:$dst, VR512:$src),
-            (VCVTHF82BF4SZmr addr:$dst, VR512:$src)>;
-  def : Pat<(int_x86_avx10_vcvthf82bf4s_256_mem addr:$dst, VR256X:$src),
-            (VCVTHF82BF4SZ256mr addr:$dst, VR256X:$src)>;
-  def : Pat<(int_x86_avx10_vcvthf82bf4s_128_mem addr:$dst, VR128X:$src),
-            (VCVTHF82BF4SZ128mr addr:$dst, VR128X:$src)>;
+  defm VCVTBF82BF4S : avx10_v2aux_cvt_trunc_b<0x3D, "vcvtbf82bf4s">,
+                      T_MAP5, XS, REX_W;
+  defm VCVTHF82BF4S : avx10_v2aux_cvt_trunc_b<0x3D, "vcvthf82bf4s">,
+                      T_MAP5, XS;
 }
 
 //-------------------------------------------------
-// Group E: Narrowing reg-only conversions (8-bit to 6-bit, no masking)
+// Convert from FP8 to FP6
 //-------------------------------------------------
 
-// Group E multiclass: Narrowing conversion reg-only (no masking, 8-bit to 6-bit)
-multiclass avx10_v2aux_cvt_narrow<bits<8> opc, string OpcodeStr,
-                                  X86VectorVTInfo Info,
-                                  SDPatternOperator OpNode> {
-  let hasSideEffects = 0 in {
-    def rr : I<opc, MRMSrcReg, (outs Info.RC:$dst),
-               (ins Info.RC:$src),
-               OpcodeStr # "\t{$src, $dst|$dst, $src}", []>,
-             EVEX, Sched<[WriteCvtPH2PSZ]>;
-  }
+// Convert from FP8 to FP6: narrowing conversion, reg-only, no masking
+multiclass avx10_v2aux_cvt_narrow_base<bits<8> opc, string OpcodeStr,
+                                       X86VectorVTInfo _,
+                                       X86FoldableSchedWrite sched> {
+  def rr : I<opc, MRMSrcReg, (outs _.RC:$dst),
+             (ins _.RC:$src),
+             OpcodeStr # "\t{$src, $dst|$dst, $src}",
+             [(set _.RC:$dst,
+               (_.VT (!cast<Intrinsic>("int_x86_avx10_"#OpcodeStr#"_"#_.Size)
+                       (_.VT _.RC:$src))))]>,
+           EVEX, Sched<[sched]>;
+}
 
-  // Intrinsic pattern
-  def : Pat<(Info.VT (OpNode (Info.VT Info.RC:$src))),
-            (!cast<Instruction>(NAME # "rr") Info.RC:$src)>;
+multiclass avx10_v2aux_cvt_narrow_b<bits<8> opc, string OpcodeStr> {
+  defm Z    : avx10_v2aux_cvt_narrow_base<opc, OpcodeStr, v64i8_info,
+                                          WriteCvtPH2PSZ>, EVEX_V512;
+  defm Z256 : avx10_v2aux_cvt_narrow_base<opc, OpcodeStr, v32i8x_info,
+                                          WriteCvtPH2PSY>, EVEX_V256;
+  defm Z128 : avx10_v2aux_cvt_narrow_base<opc, OpcodeStr, v16i8x_info,
+                                          WriteCvtPH2PS>, EVEX_V128;
 }
 
 let Predicates = [HasAVX10_V2_AUX] in {
-  defm VCVTBF82BF6SZ    : avx10_v2aux_cvt_narrow<0x3E, "vcvtbf82bf6s",
-                                                 v64i8_info, int_x86_avx10_vcvtbf82bf6s_512>,
-                          T_MAP5, XS, REX_W, EVEX_V512;
-  defm VCVTBF82BF6SZ256 : avx10_v2aux_cvt_narrow<0x3E, "vcvtbf82bf6s",
-                                                 v32i8x_info, int_x86_avx10_vcvtbf82bf6s_256>,
-                          T_MAP5, XS, REX_W, EVEX_V256;
-  defm VCVTBF82BF6SZ128 : avx10_v2aux_cvt_narrow<0x3E, "vcvtbf82bf6s",
-                                                 v16i8x_info, int_x86_avx10_vcvtbf82bf6s_128>,
-                          T_MAP5, XS, REX_W, EVEX_V128;
-
-  defm VCVTHF82HF6SZ    : avx10_v2aux_cvt_narrow<0x3C, "vcvthf82hf6s",
-                                                 v64i8_info, int_x86_avx10_vcvthf82hf6s_512>,
-                          T_MAP5, XS, EVEX_V512;
-  defm VCVTHF82HF6SZ256 : avx10_v2aux_cvt_narrow<0x3C, "vcvthf82hf6s",
-                                                 v32i8x_info, int_x86_avx10_vcvthf82hf6s_256>,
-                          T_MAP5, XS, EVEX_V256;
-  defm VCVTHF82HF6SZ128 : avx10_v2aux_cvt_narrow<0x3C, "vcvthf82hf6s",
-                                                 v16i8x_info, int_x86_avx10_vcvthf82hf6s_128>,
-                          T_MAP5, XS, EVEX_V128;
+  defm VCVTBF82BF6S : avx10_v2aux_cvt_narrow_b<0x3E, "vcvtbf82bf6s">,
+                      T_MAP5, XS, REX_W;
+  defm VCVTHF82HF6S : avx10_v2aux_cvt_narrow_b<0x3C, "vcvthf82hf6s">,
+                      T_MAP5, XS;
 }
 
 //-------------------------------------------------
-// Group F: Expanding/widening conversions with masking
+// Convert from FP4 to FP8 and from FP6 to FP8
 //-------------------------------------------------
 
 // VCVTBF42HF8: expanding (2x), with masking, no broadcast
 // Z128: xmm{k}{z}, xmm/m64  Z256: ymm{k}{z}, xmm/m128  Z: zmm{k}{z}, ymm/m256
 let Predicates = [HasAVX10_V2_AUX] in {
-  defm VCVTBF42HF8Z : avx10_v2aux_cvt_expand_masked<0x37, "vcvtbf42hf8",
-                                                    v64i8_info, v32i8x_info, f256mem,
-                                                    int_x86_avx10_vcvtbf42hf8_512>,
-                      T_MAP5, PS, EVEX_V512, EVEX_CD8<8, CD8VH>;
-  defm VCVTBF42HF8Z128 : avx10_v2aux_cvt_expand_masked<0x37, "vcvtbf42hf8",
-                                                       v16i8x_info, v16i8x_info, f64mem,
-                                                       int_x86_avx10_vcvtbf42hf8_128>,
-                         T_MAP5, PS, EVEX_V128, EVEX_CD8<8, CD8VH>;
-  defm VCVTBF42HF8Z256 : avx10_v2aux_cvt_expand_masked<0x37, "vcvtbf42hf8",
-                                                       v32i8x_info, v16i8x_info, f128mem,
-                                                       int_x86_avx10_vcvtbf42hf8_256>,
-                         T_MAP5, PS, EVEX_V256, EVEX_CD8<8, CD8VH>;
+  defm VCVTBF42HF8 : avx10_v2aux_cvt_expand_masked_b<0x37, "vcvtbf42hf8">,
+                     T_MAP5, PS;
 }
 let Predicates = [HasAVX10_V2_AUX] in {
-  // VCVTBF62HF8: widening (6-bit to 8-bit) with masking, reg-only (66.MAP5.W1)
-  defm VCVTBF62HF8Z    : avx10_v2aux_cvt_widen_masked<0x37, "vcvtbf62hf8",
-                                                      v64i8_info,
-                                                      int_x86_avx10_vcvtbf62hf8_512>,
-                         T_MAP5, PD, REX_W, EVEX_V512;
-  defm VCVTBF62HF8Z256 : avx10_v2aux_cvt_widen_masked<0x37, "vcvtbf62hf8",
-                                                      v32i8x_info,
-                                                      int_x86_avx10_vcvtbf62hf8_256>,
-                         T_MAP5, PD, REX_W, EVEX_V256;
-  defm VCVTBF62HF8Z128 : avx10_v2aux_cvt_widen_masked<0x37, "vcvtbf62hf8",
-                                                      v16i8x_info,
-                                                      int_x86_avx10_vcvtbf62hf8_128>,
-                         T_MAP5, PD, REX_W, EVEX_V128;
-
-  // VCVTHF62HF8: widening (6-bit to 8-bit) with masking, reg-only (66.MAP5.W0)
-  defm VCVTHF62HF8Z    : avx10_v2aux_cvt_widen_masked<0x37, "vcvthf62hf8",
-                                                      v64i8_info,
-                                                      int_x86_avx10_vcvthf62hf8_512>,
-                         T_MAP5, PD, EVEX_V512;
-  defm VCVTHF62HF8Z256 : avx10_v2aux_cvt_widen_masked<0x37, "vcvthf62hf8",
-                                                      v32i8x_info,
-                                                      int_x86_avx10_vcvthf62hf8_256>,
-                         T_MAP5, PD, EVEX_V256;
-  defm VCVTHF62HF8Z128 : avx10_v2aux_cvt_widen_masked<0x37, "vcvthf62hf8",
-                                                      v16i8x_info,
-                                                      int_x86_avx10_vcvthf62hf8_128>,
-                         T_MAP5, PD, EVEX_V128;
+  // VCVTBF62HF8: widening (6-bit to 8-bit) with masking, reg-only
+  defm VCVTBF62HF8 : avx10_v2aux_cvt_widen_masked_b<0x37, "vcvtbf62hf8">,
+                     T_MAP5, PD, REX_W;
+
+  // VCVTHF62HF8: widening (6-bit to 8-bit) with masking, reg-only
+  defm VCVTHF62HF8 : avx10_v2aux_cvt_widen_masked_b<0x37, "vcvthf62hf8">,
+                     T_MAP5, PD;
 }
 
 //-------------------------------------------------
-// Group G: VPMOVSSDB - Integer DWord->Byte symmetric signed saturation
-// F3.0F38.W0 0x41
+// Down convert DWord to Byte with symmetric signed saturation
 //-------------------------------------------------
 
 let Predicates = [HasAVX10_V2_AUX] in {
@@ -557,7 +516,7 @@ let Predicates = [HasAVX10_V2_AUX] in {
 }
 
 //-------------------------------------------------
-// Group H: VUNPACKB - Byte unpack with immediate
+// Unpack to Byte
 //-------------------------------------------------
 
 let Predicates = [HasAVX10_V2_AUX] in {
diff --git a/llvm/lib/TargetParser/Host.cpp b/llvm/lib/TargetParser/Host.cpp
index 4337a7d2dc2c9d..df5809250333f0 100644
--- a/llvm/lib/TargetParser/Host.cpp
+++ b/llvm/lib/TargetParser/Host.cpp
@@ -2271,6 +2271,12 @@ StringMap<bool> sys::getHostCPUFeatures() {
   Features["avx10.1"] = HasAVX10 && AVX10Ver >= 1;
   Features["avx10.2"] = HasAVX10 && AVX10Ver >= 2;
 
+  bool HasLeaf24Subleaf1 =
+      HasLeaf24 && EAX >= 1 &&
+      !getX86CpuIDAndInfoEx(0x24, 0x1, &EAX, &EBX, &ECX, &EDX);
+  Features["avx10v2aux"] =
+      HasAVX10 && HasLeaf24Subleaf1 && ((ECX >> 3) & 1) && HasAVX512Save;
+
   return Features;
 }
 #elif defined(__linux__) && (defined(__arm__) || defined(__aarch64__))
diff --git a/llvm/lib/TargetParser/X86TargetParser.cpp b/llvm/lib/TargetParser/X86TargetParser.cpp
index ce1b1ae7695af0..27a1a66ff6ff6a 100644
--- a/llvm/lib/TargetParser/X86TargetParser.cpp
+++ b/llvm/lib/TargetParser/X86TargetParser.cpp
@@ -669,7 +669,7 @@ constexpr FeatureBitset ImpliedFeaturesAVX10_1 =
     FeatureAVX512VBMI2 | FeatureAVX512BITALG | FeatureAVX512FP16 |
     FeatureAVX512DQ | FeatureAVX512VL;
 constexpr FeatureBitset ImpliedFeaturesAVX10_2 = FeatureAVX10_1;
-constexpr FeatureBitset ImpliedFeaturesAVX10_V2_AUX = FeatureAVX10_2;
+constexpr FeatureBitset ImpliedFeaturesAVX10_V2_AUX = FeatureAVX10_1;
 
 // APX Features
 constexpr FeatureBitset ImpliedFeaturesEGPR = {};

>From c1add95505bb66ac0cfee21d7b97d8bbe9749228 Mon Sep 17 00:00:00 2001
From: Ganesh Gopalasubramanian <Ganesh.Gopalasubramanian at amd.com>
Date: Thu, 30 Jul 2026 20:57:40 +0530
Subject: [PATCH 07/16] [X86] Fix clang-format alignment in vpmovssdb mask
 intrinsics

---
 clang/lib/Headers/avx10_2_512v2auxintrin.h | 2 +-
 clang/lib/Headers/avx10_2_v2auxintrin.h    | 4 ++--
 2 files changed, 3 insertions(+), 3 deletions(-)

diff --git a/clang/lib/Headers/avx10_2_512v2auxintrin.h b/clang/lib/Headers/avx10_2_512v2auxintrin.h
index 34a5b09f46a260..f15564f9742762 100644
--- a/clang/lib/Headers/avx10_2_512v2auxintrin.h
+++ b/clang/lib/Headers/avx10_2_512v2auxintrin.h
@@ -1136,7 +1136,7 @@ _mm512_cvtss_epi32_epi8(__m512i __A) {
 static __inline__ __m128i __DEFAULT_FN_ATTRS512
 _mm512_mask_cvtss_epi32_epi8(__m128i __W, __mmask16 __U, __m512i __A) {
   return (__m128i)__builtin_ia32_vpmovssdb512_mask((__v16si)__A, (__v16qi)__W,
-                                                  __U);
+                                                   __U);
 }
 
 /// Convert packed signed 32-bit integers in \a __A to packed 8-bit integers
diff --git a/clang/lib/Headers/avx10_2_v2auxintrin.h b/clang/lib/Headers/avx10_2_v2auxintrin.h
index 03940740b713e5..b377db5c29926a 100644
--- a/clang/lib/Headers/avx10_2_v2auxintrin.h
+++ b/clang/lib/Headers/avx10_2_v2auxintrin.h
@@ -2210,7 +2210,7 @@ _mm_cvtss_epi32_epi8(__m128i __A) {
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_mask_cvtss_epi32_epi8(__m128i __W, __mmask8 __U, __m128i __A) {
   return (__m128i)__builtin_ia32_vpmovssdb128_mask((__v4si)__A, (__v16qi)__W,
-                                                  __U);
+                                                   __U);
 }
 
 /// Convert packed signed 32-bit integers in \a __A to packed 8-bit integers
@@ -2269,7 +2269,7 @@ _mm256_cvtss_epi32_epi8(__m256i __A) {
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_mask_cvtss_epi32_epi8(__m128i __W, __mmask8 __U, __m256i __A) {
   return (__m128i)__builtin_ia32_vpmovssdb256_mask((__v8si)__A, (__v16qi)__W,
-                                                  __U);
+                                                   __U);
 }
 
 /// Convert packed signed 32-bit integers in \a __A to packed 8-bit integers

>From eb51f72dae8252653234a3c2ae1dc8d14742f775 Mon Sep 17 00:00:00 2001
From: Ganesh Gopalasubramanian <Ganesh.Gopalasubramanian at amd.com>
Date: Mon, 24 Aug 2026 14:08:01 +0530
Subject: [PATCH 08/16] [X86][AVX10_V2_AUX] Address review feedback on masking,
 naming and load folding

Follow-up to the initial AVX10_V2_AUX support, covering several independent
review points.

Masked sub-width converts
- The FP32->FP8 conversions narrow 4:1, Use direct masks instead of selects.
- Lower through TRUNCATE_TO_REG / TRUNCATE2_TO_REG.
- 512 bits destination is fully written, so no masked intrinsic is declared.

Bias operands
- Model the bias operand of the VCVTBIASPS2{BF8,BF8S,HF8,HF8S} families as a
vector of dwords rather than bytes.

Intrinsic naming
- Sync it with 'cvt', with GCC and as per the convention established in
19d2023a6668 ("[X86][AVX10.2] Use 's_' for saturate-convert intrinsics").

Load folding
- At 128 bits VCVTBF42HF8 reads only m64, but the rm pattern demanded a full
16-byte load because the multiclass used one parameter for both the register
class and the load type. _mm_loadu_si64 lowers to X86ISD::VZEXT_LOAD, which the
pattern cannot match, so the load did not fold. Give the multiclass a ld_dag
parameter and override it at Z128 with a vzload, mirroring avx512_cvtph2ps, and
add the companion scalar_to_vector pattern.
---
 clang/include/clang/Basic/BuiltinsX86.td      |   66 +-
 clang/lib/Headers/avx10_2_512v2auxintrin.h    |   64 +-
 clang/lib/Headers/avx10_2_v2auxintrin.h       |  476 +++--
 .../X86/avx10_2_v2aux-builtins-errors.c       |   36 +-
 .../test/CodeGen/X86/avx10_2_v2aux-builtins.c |  466 +++--
 clang/test/Preprocessor/x86_target_features.c |    5 +-
 clang/test/Sema/builtins-x86.c                |    2 +-
 llvm/include/llvm/IR/IntrinsicsX86.td         |  160 +-
 llvm/lib/Target/X86/X86InstrAVX10_V2_AUX.td   |  139 +-
 llvm/lib/Target/X86/X86InstrFragmentsSIMD.td  |   26 +-
 llvm/lib/Target/X86/X86IntrinsicsInfo.h       |   40 +
 .../CodeGen/X86/avx10_v2aux-intrinsics.ll     | 1806 ++++++++++++++++-
 12 files changed, 2622 insertions(+), 664 deletions(-)

diff --git a/clang/include/clang/Basic/BuiltinsX86.td b/clang/include/clang/Basic/BuiltinsX86.td
index 6009d13c893518..79282cfc11bcf7 100644
--- a/clang/include/clang/Basic/BuiltinsX86.td
+++ b/clang/include/clang/Basic/BuiltinsX86.td
@@ -5062,15 +5062,17 @@ let Features = "avx10.2", Attributes = [NoThrow, Const, RequiredVectorWidth<512>
 
 // AVX10 V2 AUX - Convert instructions
 
-// Group A: PS(f32) -> i8 truncating conversions (quarter-size: output always v16i8)
+// Convert from FP32 to FP8
 
 // VCVTPS2BF8
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
   def vcvtps2bf8_128 : X86Builtin<"_Vector<16, char>(_Vector<4, float>)">;
+  def vcvtps2bf8_128_mask : X86Builtin<"_Vector<16, char>(_Vector<4, float>, _Vector<16, char>, unsigned char)">;
 }
 
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
   def vcvtps2bf8_256 : X86Builtin<"_Vector<16, char>(_Vector<8, float>)">;
+  def vcvtps2bf8_256_mask : X86Builtin<"_Vector<16, char>(_Vector<8, float>, _Vector<16, char>, unsigned char)">;
 }
 
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
@@ -5080,10 +5082,12 @@ let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in
 // VCVTPS2BF8S
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
   def vcvtps2bf8s_128 : X86Builtin<"_Vector<16, char>(_Vector<4, float>)">;
+  def vcvtps2bf8s_128_mask : X86Builtin<"_Vector<16, char>(_Vector<4, float>, _Vector<16, char>, unsigned char)">;
 }
 
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
   def vcvtps2bf8s_256 : X86Builtin<"_Vector<16, char>(_Vector<8, float>)">;
+  def vcvtps2bf8s_256_mask : X86Builtin<"_Vector<16, char>(_Vector<8, float>, _Vector<16, char>, unsigned char)">;
 }
 
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
@@ -5093,10 +5097,12 @@ let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in
 // VCVTPS2HF8
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
   def vcvtps2hf8_128 : X86Builtin<"_Vector<16, char>(_Vector<4, float>)">;
+  def vcvtps2hf8_128_mask : X86Builtin<"_Vector<16, char>(_Vector<4, float>, _Vector<16, char>, unsigned char)">;
 }
 
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
   def vcvtps2hf8_256 : X86Builtin<"_Vector<16, char>(_Vector<8, float>)">;
+  def vcvtps2hf8_256_mask : X86Builtin<"_Vector<16, char>(_Vector<8, float>, _Vector<16, char>, unsigned char)">;
 }
 
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
@@ -5106,10 +5112,12 @@ let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in
 // VCVTPS2HF8S
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
   def vcvtps2hf8s_128 : X86Builtin<"_Vector<16, char>(_Vector<4, float>)">;
+  def vcvtps2hf8s_128_mask : X86Builtin<"_Vector<16, char>(_Vector<4, float>, _Vector<16, char>, unsigned char)">;
 }
 
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
   def vcvtps2hf8s_256 : X86Builtin<"_Vector<16, char>(_Vector<8, float>)">;
+  def vcvtps2hf8s_256_mask : X86Builtin<"_Vector<16, char>(_Vector<8, float>, _Vector<16, char>, unsigned char)">;
 }
 
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
@@ -5119,10 +5127,12 @@ let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in
 // VCVTROPS2HF8
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
   def vcvtrops2hf8_128 : X86Builtin<"_Vector<16, char>(_Vector<4, float>)">;
+  def vcvtrops2hf8_128_mask : X86Builtin<"_Vector<16, char>(_Vector<4, float>, _Vector<16, char>, unsigned char)">;
 }
 
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
   def vcvtrops2hf8_256 : X86Builtin<"_Vector<16, char>(_Vector<8, float>)">;
+  def vcvtrops2hf8_256_mask : X86Builtin<"_Vector<16, char>(_Vector<8, float>, _Vector<16, char>, unsigned char)">;
 }
 
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
@@ -5132,71 +5142,81 @@ let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in
 // VCVTROPS2HF8S
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
   def vcvtrops2hf8s_128 : X86Builtin<"_Vector<16, char>(_Vector<4, float>)">;
+  def vcvtrops2hf8s_128_mask : X86Builtin<"_Vector<16, char>(_Vector<4, float>, _Vector<16, char>, unsigned char)">;
 }
 
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
   def vcvtrops2hf8s_256 : X86Builtin<"_Vector<16, char>(_Vector<8, float>)">;
+  def vcvtrops2hf8s_256_mask : X86Builtin<"_Vector<16, char>(_Vector<8, float>, _Vector<16, char>, unsigned char)">;
 }
 
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
   def vcvtrops2hf8s_512 : X86Builtin<"_Vector<16, char>(_Vector<16, float>)">;
 }
 
-// Group B: Bias PS -> i8 conversions (2-operand: bias + f32 source -> i8 dest)
+// Convert from FP32 to FP8 with bias
 
 // VCVTBIASPS2BF8
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
-  def vcvtbiasps2bf8_128 : X86Builtin<"_Vector<16, char>(_Vector<16, char>, _Vector<4, float>)">;
+  def vcvtbiasps2bf8_128 : X86Builtin<"_Vector<16, char>(_Vector<4, int>, _Vector<4, float>)">;
+  def vcvtbiasps2bf8_128_mask : X86Builtin<"_Vector<16, char>(_Vector<4, int>, _Vector<4, float>, _Vector<16, char>, unsigned char)">;
 }
 
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
-  def vcvtbiasps2bf8_256 : X86Builtin<"_Vector<16, char>(_Vector<32, char>, _Vector<8, float>)">;
+  def vcvtbiasps2bf8_256 : X86Builtin<"_Vector<16, char>(_Vector<8, int>, _Vector<8, float>)">;
+  def vcvtbiasps2bf8_256_mask : X86Builtin<"_Vector<16, char>(_Vector<8, int>, _Vector<8, float>, _Vector<16, char>, unsigned char)">;
 }
 
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
-  def vcvtbiasps2bf8_512 : X86Builtin<"_Vector<16, char>(_Vector<64, char>, _Vector<16, float>)">;
+  def vcvtbiasps2bf8_512 : X86Builtin<"_Vector<16, char>(_Vector<16, int>, _Vector<16, float>)">;
 }
 
 // VCVTBIASPS2BF8S
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
-  def vcvtbiasps2bf8s_128 : X86Builtin<"_Vector<16, char>(_Vector<16, char>, _Vector<4, float>)">;
+  def vcvtbiasps2bf8s_128 : X86Builtin<"_Vector<16, char>(_Vector<4, int>, _Vector<4, float>)">;
+  def vcvtbiasps2bf8s_128_mask : X86Builtin<"_Vector<16, char>(_Vector<4, int>, _Vector<4, float>, _Vector<16, char>, unsigned char)">;
 }
 
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
-  def vcvtbiasps2bf8s_256 : X86Builtin<"_Vector<16, char>(_Vector<32, char>, _Vector<8, float>)">;
+  def vcvtbiasps2bf8s_256 : X86Builtin<"_Vector<16, char>(_Vector<8, int>, _Vector<8, float>)">;
+  def vcvtbiasps2bf8s_256_mask : X86Builtin<"_Vector<16, char>(_Vector<8, int>, _Vector<8, float>, _Vector<16, char>, unsigned char)">;
 }
 
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
-  def vcvtbiasps2bf8s_512 : X86Builtin<"_Vector<16, char>(_Vector<64, char>, _Vector<16, float>)">;
+  def vcvtbiasps2bf8s_512 : X86Builtin<"_Vector<16, char>(_Vector<16, int>, _Vector<16, float>)">;
 }
 
 // VCVTBIASPS2HF8
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
-  def vcvtbiasps2hf8_128 : X86Builtin<"_Vector<16, char>(_Vector<16, char>, _Vector<4, float>)">;
+  def vcvtbiasps2hf8_128 : X86Builtin<"_Vector<16, char>(_Vector<4, int>, _Vector<4, float>)">;
+  def vcvtbiasps2hf8_128_mask : X86Builtin<"_Vector<16, char>(_Vector<4, int>, _Vector<4, float>, _Vector<16, char>, unsigned char)">;
 }
 
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
-  def vcvtbiasps2hf8_256 : X86Builtin<"_Vector<16, char>(_Vector<32, char>, _Vector<8, float>)">;
+  def vcvtbiasps2hf8_256 : X86Builtin<"_Vector<16, char>(_Vector<8, int>, _Vector<8, float>)">;
+  def vcvtbiasps2hf8_256_mask : X86Builtin<"_Vector<16, char>(_Vector<8, int>, _Vector<8, float>, _Vector<16, char>, unsigned char)">;
 }
 
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
-  def vcvtbiasps2hf8_512 : X86Builtin<"_Vector<16, char>(_Vector<64, char>, _Vector<16, float>)">;
+  def vcvtbiasps2hf8_512 : X86Builtin<"_Vector<16, char>(_Vector<16, int>, _Vector<16, float>)">;
 }
 
 // VCVTBIASPS2HF8S
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
-  def vcvtbiasps2hf8s_128 : X86Builtin<"_Vector<16, char>(_Vector<16, char>, _Vector<4, float>)">;
+  def vcvtbiasps2hf8s_128 : X86Builtin<"_Vector<16, char>(_Vector<4, int>, _Vector<4, float>)">;
+  def vcvtbiasps2hf8s_128_mask : X86Builtin<"_Vector<16, char>(_Vector<4, int>, _Vector<4, float>, _Vector<16, char>, unsigned char)">;
 }
 
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
-  def vcvtbiasps2hf8s_256 : X86Builtin<"_Vector<16, char>(_Vector<32, char>, _Vector<8, float>)">;
+  def vcvtbiasps2hf8s_256 : X86Builtin<"_Vector<16, char>(_Vector<8, int>, _Vector<8, float>)">;
+  def vcvtbiasps2hf8s_256_mask : X86Builtin<"_Vector<16, char>(_Vector<8, int>, _Vector<8, float>, _Vector<16, char>, unsigned char)">;
 }
 
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
-  def vcvtbiasps2hf8s_512 : X86Builtin<"_Vector<16, char>(_Vector<64, char>, _Vector<16, float>)">;
+  def vcvtbiasps2hf8s_512 : X86Builtin<"_Vector<16, char>(_Vector<16, int>, _Vector<16, float>)">;
 }
 
-// Group C: 8bit -> PS expanding conversions
+// Convert from FP8 to FP32
 
 // VCVTBF82PS
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
@@ -5224,7 +5244,7 @@ let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in
   def vcvthf82ps_512 : X86Builtin<"_Vector<16, float>(_Vector<16, char>)">;
 }
 
-// Group E: Same-size reg-only conversions (no masking)
+// Convert from FP8 to FP6
 
 // VCVTBF82BF6S
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
@@ -5252,7 +5272,7 @@ let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in
   def vcvthf82hf6s_512 : X86Builtin<"_Vector<64, char>(_Vector<64, char>)">;
 }
 
-// Group F: Expanding/same-size conversions (no masking in intrinsic; use selectb for masking)
+// Convert from FP4 to FP8 and from FP6 to FP8
 
 // VCVTBF42HF8
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
@@ -5293,9 +5313,9 @@ let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in
   def vcvthf62hf8_512 : X86Builtin<"_Vector<64, char>(_Vector<64, char>)">;
 }
 
-// Group D: VCVTBF82BF4S / VCVTHF82BF4S - FP8 to FP4 truncating conversions
+// Convert from FP8 to FP4
 
-// VCVTBF82BF4S - FP8 E5M2 to FP4 E2M1 with saturation (output is half size)
+// VCVTBF82BF4S
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
   def vcvtbf82bf4s_128 : X86Builtin<"_Vector<16, char>(_Vector<16, char>)">;
 }
@@ -5308,7 +5328,7 @@ let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in
   def vcvtbf82bf4s_512 : X86Builtin<"_Vector<32, char>(_Vector<64, char>)">;
 }
 
-// VCVTBF82BF4S memory store variants (no masking - spec does not support masks)
+// VCVTBF82BF4S memory store
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
   def vcvtbf82bf4s_128_mem : X86Builtin<"void(void *, _Vector<16, char>)">;
 }
@@ -5321,7 +5341,7 @@ let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in
   def vcvtbf82bf4s_512_mem : X86Builtin<"void(void *, _Vector<64, char>)">;
 }
 
-// VCVTHF82BF4S - FP8 E4M3 to FP4 E2M1 with saturation (output is half size)
+// VCVTHF82BF4S
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
   def vcvthf82bf4s_128 : X86Builtin<"_Vector<16, char>(_Vector<16, char>)">;
 }
@@ -5334,7 +5354,7 @@ let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in
   def vcvthf82bf4s_512 : X86Builtin<"_Vector<32, char>(_Vector<64, char>)">;
 }
 
-// VCVTHF82BF4S memory store variants (no masking - spec does not support masks)
+// VCVTHF82BF4S memory store
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
   def vcvthf82bf4s_128_mem : X86Builtin<"void(void *, _Vector<16, char>)">;
 }
@@ -5347,7 +5367,7 @@ let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in
   def vcvthf82bf4s_512_mem : X86Builtin<"void(void *, _Vector<64, char>)">;
 }
 
-// Group H: VUNPACKB - Byte unpack with immediate (no masking in intrinsic)
+// Unpack to Byte
 
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
   def vunpackb128 : X86Builtin<"_Vector<16, char>(_Vector<16, char>, _Constant unsigned char)">;
diff --git a/clang/lib/Headers/avx10_2_512v2auxintrin.h b/clang/lib/Headers/avx10_2_512v2auxintrin.h
index f15564f9742762..21898966126fe2 100644
--- a/clang/lib/Headers/avx10_2_512v2auxintrin.h
+++ b/clang/lib/Headers/avx10_2_512v2auxintrin.h
@@ -387,14 +387,14 @@ _mm512_maskz_cvts_rops_hf8(__mmask16 __U, __m512 __A) {
 /// This intrinsic corresponds to the \c VCVTBIASPS2BF8 instruction.
 ///
 /// \param __A
-///    A 512-bit vector of [64 x i8] containing bias values.
+///    A 512-bit vector of [16 x i32] containing bias values.
 /// \param __B
 ///    A 512-bit vector of [16 x float].
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS512
 _mm512_cvtbiasps_bf8(__m512i __A, __m512 __B) {
-  return (__m128i)__builtin_ia32_vcvtbiasps2bf8_512((__v64qi)__A, (__v16sf)__B);
+  return (__m128i)__builtin_ia32_vcvtbiasps2bf8_512((__v16si)__A, (__v16sf)__B);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __B
@@ -410,7 +410,7 @@ _mm512_cvtbiasps_bf8(__m512i __A, __m512 __B) {
 /// \param __U
 ///    A 16-bit mask indicating which elements to write.
 /// \param __A
-///    A 512-bit vector of [64 x i8] containing bias values.
+///    A 512-bit vector of [16 x i32] containing bias values.
 /// \param __B
 ///    A 512-bit vector of [16 x float].
 /// \returns
@@ -432,7 +432,7 @@ _mm512_mask_cvtbiasps_bf8(__m128i __W, __mmask16 __U, __m512i __A, __m512 __B) {
 /// \param __U
 ///    A 16-bit mask indicating which elements to write (zero otherwise).
 /// \param __A
-///    A 512-bit vector of [64 x i8] containing bias values.
+///    A 512-bit vector of [16 x i32] containing bias values.
 /// \param __B
 ///    A 512-bit vector of [16 x float].
 /// \returns
@@ -453,14 +453,14 @@ _mm512_maskz_cvtbiasps_bf8(__mmask16 __U, __m512i __A, __m512 __B) {
 /// This intrinsic corresponds to the \c VCVTBIASPS2BF8S instruction.
 ///
 /// \param __A
-///    A 512-bit vector of [64 x i8] containing bias values.
+///    A 512-bit vector of [16 x i32] containing bias values.
 /// \param __B
 ///    A 512-bit vector of [16 x float].
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS512
 _mm512_cvts_biasps_bf8(__m512i __A, __m512 __B) {
-  return (__m128i)__builtin_ia32_vcvtbiasps2bf8s_512((__v64qi)__A,
+  return (__m128i)__builtin_ia32_vcvtbiasps2bf8s_512((__v16si)__A,
                                                      (__v16sf)__B);
 }
 
@@ -477,7 +477,7 @@ _mm512_cvts_biasps_bf8(__m512i __A, __m512 __B) {
 /// \param __U
 ///    A 16-bit mask indicating which elements to write.
 /// \param __A
-///    A 512-bit vector of [64 x i8] containing bias values.
+///    A 512-bit vector of [16 x i32] containing bias values.
 /// \param __B
 ///    A 512-bit vector of [16 x float].
 /// \returns
@@ -499,7 +499,7 @@ static __inline__ __m128i __DEFAULT_FN_ATTRS512 _mm512_mask_cvts_biasps_bf8(
 /// \param __U
 ///    A 16-bit mask indicating which elements to write (zero otherwise).
 /// \param __A
-///    A 512-bit vector of [64 x i8] containing bias values.
+///    A 512-bit vector of [16 x i32] containing bias values.
 /// \param __B
 ///    A 512-bit vector of [16 x float].
 /// \returns
@@ -520,14 +520,14 @@ _mm512_maskz_cvts_biasps_bf8(__mmask16 __U, __m512i __A, __m512 __B) {
 /// This intrinsic corresponds to the \c VCVTBIASPS2HF8 instruction.
 ///
 /// \param __A
-///    A 512-bit vector of [64 x i8] containing bias values.
+///    A 512-bit vector of [16 x i32] containing bias values.
 /// \param __B
 ///    A 512-bit vector of [16 x float].
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS512
 _mm512_cvtbiasps_hf8(__m512i __A, __m512 __B) {
-  return (__m128i)__builtin_ia32_vcvtbiasps2hf8_512((__v64qi)__A, (__v16sf)__B);
+  return (__m128i)__builtin_ia32_vcvtbiasps2hf8_512((__v16si)__A, (__v16sf)__B);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __B
@@ -543,7 +543,7 @@ _mm512_cvtbiasps_hf8(__m512i __A, __m512 __B) {
 /// \param __U
 ///    A 16-bit mask indicating which elements to write.
 /// \param __A
-///    A 512-bit vector of [64 x i8] containing bias values.
+///    A 512-bit vector of [16 x i32] containing bias values.
 /// \param __B
 ///    A 512-bit vector of [16 x float].
 /// \returns
@@ -565,7 +565,7 @@ _mm512_mask_cvtbiasps_hf8(__m128i __W, __mmask16 __U, __m512i __A, __m512 __B) {
 /// \param __U
 ///    A 16-bit mask indicating which elements to write (zero otherwise).
 /// \param __A
-///    A 512-bit vector of [64 x i8] containing bias values.
+///    A 512-bit vector of [16 x i32] containing bias values.
 /// \param __B
 ///    A 512-bit vector of [16 x float].
 /// \returns
@@ -586,14 +586,14 @@ _mm512_maskz_cvtbiasps_hf8(__mmask16 __U, __m512i __A, __m512 __B) {
 /// This intrinsic corresponds to the \c VCVTBIASPS2HF8S instruction.
 ///
 /// \param __A
-///    A 512-bit vector of [64 x i8] containing bias values.
+///    A 512-bit vector of [16 x i32] containing bias values.
 /// \param __B
 ///    A 512-bit vector of [16 x float].
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS512
 _mm512_cvts_biasps_hf8(__m512i __A, __m512 __B) {
-  return (__m128i)__builtin_ia32_vcvtbiasps2hf8s_512((__v64qi)__A,
+  return (__m128i)__builtin_ia32_vcvtbiasps2hf8s_512((__v16si)__A,
                                                      (__v16sf)__B);
 }
 
@@ -610,7 +610,7 @@ _mm512_cvts_biasps_hf8(__m512i __A, __m512 __B) {
 /// \param __U
 ///    A 16-bit mask indicating which elements to write.
 /// \param __A
-///    A 512-bit vector of [64 x i8] containing bias values.
+///    A 512-bit vector of [16 x i32] containing bias values.
 /// \param __B
 ///    A 512-bit vector of [16 x float].
 /// \returns
@@ -632,7 +632,7 @@ static __inline__ __m128i __DEFAULT_FN_ATTRS512 _mm512_mask_cvts_biasps_hf8(
 /// \param __U
 ///    A 16-bit mask indicating which elements to write (zero otherwise).
 /// \param __A
-///    A 512-bit vector of [64 x i8] containing bias values.
+///    A 512-bit vector of [16 x i32] containing bias values.
 /// \param __B
 ///    A 512-bit vector of [16 x float].
 /// \returns
@@ -775,7 +775,7 @@ _mm512_maskz_cvthf8_ps(__mmask16 __U, __m128i __A) {
 /// \returns
 ///    A 512-bit vector of [64 x i8] containing the converted BF6 values.
 static __inline__ __m512i __DEFAULT_FN_ATTRS512
-_mm512_cvtbf8_bf6s(__m512i __A) {
+_mm512_cvts_bf8_bf6(__m512i __A) {
   return (__m512i)__builtin_ia32_vcvtbf82bf6s_512((__v64qi)__A);
 }
 
@@ -792,7 +792,7 @@ _mm512_cvtbf8_bf6s(__m512i __A) {
 /// \returns
 ///    A 512-bit vector of [64 x i8] containing the converted HF6 values.
 static __inline__ __m512i __DEFAULT_FN_ATTRS512
-_mm512_cvthf8_hf6s(__m512i __A) {
+_mm512_cvts_hf8_hf6(__m512i __A) {
   return (__m512i)__builtin_ia32_vcvthf82hf6s_512((__v64qi)__A);
 }
 
@@ -809,7 +809,7 @@ _mm512_cvthf8_hf6s(__m512i __A) {
 /// \returns
 ///    A 256-bit vector of [32 x i8] containing the converted BF4 values.
 static __inline__ __m256i __DEFAULT_FN_ATTRS512
-_mm512_cvtbf8_bf4s(__m512i __A) {
+_mm512_cvts_bf8_bf4(__m512i __A) {
   return (__m256i)__builtin_ia32_vcvtbf82bf4s_512((__v64qi)__A);
 }
 
@@ -826,7 +826,7 @@ _mm512_cvtbf8_bf4s(__m512i __A) {
 /// \returns
 ///    A 256-bit vector of [32 x i8] containing the converted BF4 values.
 static __inline__ __m256i __DEFAULT_FN_ATTRS512
-_mm512_cvthf8_bf4s(__m512i __A) {
+_mm512_cvts_hf8_bf4(__m512i __A) {
   return (__m256i)__builtin_ia32_vcvthf82bf4s_512((__v64qi)__A);
 }
 
@@ -844,7 +844,7 @@ _mm512_cvthf8_bf4s(__m512i __A) {
 /// \param __A
 ///    A 512-bit vector of [64 x i8] containing BF8 values.
 static __inline__ void __DEFAULT_FN_ATTRS512
-_mm512_cvtbf8_bf4s_storeu(void *__P, __m512i __A) {
+_mm512_cvts_bf8_bf4_storeu(void *__P, __m512i __A) {
   __builtin_ia32_vcvtbf82bf4s_512_mem(__P, (__v64qi)__A);
 }
 
@@ -862,7 +862,7 @@ _mm512_cvtbf8_bf4s_storeu(void *__P, __m512i __A) {
 /// \param __A
 ///    A 512-bit vector of [64 x i8] containing HF8 values.
 static __inline__ void __DEFAULT_FN_ATTRS512
-_mm512_cvthf8_bf4s_storeu(void *__P, __m512i __A) {
+_mm512_cvts_hf8_bf4_storeu(void *__P, __m512i __A) {
   __builtin_ia32_vcvthf82bf4s_512_mem(__P, (__v64qi)__A);
 }
 
@@ -1053,7 +1053,7 @@ _mm512_maskz_cvthf6_hf8(__mmask64 __U, __m512i __A) {
 ///    An immediate value specifying the unpack operation.
 /// \returns
 ///    A 512-bit vector of [64 x i8] containing the unpacked values.
-#define _mm512_unpackb_epi8(A, imm)                                            \
+#define _mm512_unpack_epi8(A, imm)                                             \
   ((__m512i)__builtin_ia32_vunpackb512((__v64qi)(__m512i)(A), (int)(imm)))
 
 /// Unpack bytes from \a A according to the immediate value \a imm, and store
@@ -1073,9 +1073,9 @@ _mm512_maskz_cvthf6_hf8(__mmask64 __U, __m512i __A) {
 ///    An immediate value specifying the unpack operation.
 /// \returns
 ///    A 512-bit vector of [64 x i8] containing the unpacked values.
-#define _mm512_mask_unpackb_epi8(W, U, A, imm)                                 \
+#define _mm512_mask_unpack_epi8(W, U, A, imm)                                  \
   ((__m512i)__builtin_ia32_selectb_512(                                        \
-      (__mmask64)(U), (__v64qi)_mm512_unpackb_epi8((A), (imm)),                \
+      (__mmask64)(U), (__v64qi)_mm512_unpack_epi8((A), (imm)),                 \
       (__v64qi)(__m512i)(W)))
 
 /// Unpack bytes from \a A according to the immediate value \a imm, and store
@@ -1093,13 +1093,11 @@ _mm512_maskz_cvthf6_hf8(__mmask64 __U, __m512i __A) {
 ///    An immediate value specifying the unpack operation.
 /// \returns
 ///    A 512-bit vector of [64 x i8] containing the unpacked values.
-#define _mm512_maskz_unpackb_epi8(U, A, imm)                                   \
+#define _mm512_maskz_unpack_epi8(U, A, imm)                                    \
   ((__m512i)__builtin_ia32_selectb_512(                                        \
-      (__mmask64)(U), (__v64qi)_mm512_unpackb_epi8((A), (imm)),                \
+      (__mmask64)(U), (__v64qi)_mm512_unpack_epi8((A), (imm)),                 \
       (__v64qi)_mm512_setzero_si512()))
 
-/* VPMOVSSDB - Symmetric Signed Saturation DWord to Byte */
-
 /// Convert packed signed 32-bit integers in \a __A to packed 8-bit integers
 ///    with symmetric signed saturation (clamp to [-127, +127]), and store
 ///    the results in a 128-bit vector.
@@ -1113,7 +1111,7 @@ _mm512_maskz_cvthf6_hf8(__mmask64 __U, __m512i __A) {
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS512
-_mm512_cvtss_epi32_epi8(__m512i __A) {
+_mm512_cvtssepi32_epi8(__m512i __A) {
   return (__m128i)__builtin_ia32_vpmovssdb512_mask(
       (__v16si)__A, (__v16qi)_mm_setzero_si128(), (__mmask16)-1);
 }
@@ -1134,7 +1132,7 @@ _mm512_cvtss_epi32_epi8(__m512i __A) {
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS512
-_mm512_mask_cvtss_epi32_epi8(__m128i __W, __mmask16 __U, __m512i __A) {
+_mm512_mask_cvtssepi32_epi8(__m128i __W, __mmask16 __U, __m512i __A) {
   return (__m128i)__builtin_ia32_vpmovssdb512_mask((__v16si)__A, (__v16qi)__W,
                                                    __U);
 }
@@ -1153,7 +1151,7 @@ _mm512_mask_cvtss_epi32_epi8(__m128i __W, __mmask16 __U, __m512i __A) {
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS512
-_mm512_maskz_cvtss_epi32_epi8(__mmask16 __U, __m512i __A) {
+_mm512_maskz_cvtssepi32_epi8(__mmask16 __U, __m512i __A) {
   return (__m128i)__builtin_ia32_vpmovssdb512_mask(
       (__v16si)__A, (__v16qi)_mm_setzero_si128(), __U);
 }
@@ -1173,7 +1171,7 @@ _mm512_maskz_cvtss_epi32_epi8(__mmask16 __U, __m512i __A) {
 /// \param __A
 ///    A 512-bit vector of [16 x i32].
 static __inline__ void __DEFAULT_FN_ATTRS512
-_mm512_mask_cvtss_epi32_storeu_epi8(void *__P, __mmask16 __M, __m512i __A) {
+_mm512_mask_cvtssepi32_storeu_epi8(void *__P, __mmask16 __M, __m512i __A) {
   __builtin_ia32_vpmovssdb512mem_mask((__v16qi *)__P, (__v16si)__A, __M);
 }
 
diff --git a/clang/lib/Headers/avx10_2_v2auxintrin.h b/clang/lib/Headers/avx10_2_v2auxintrin.h
index b377db5c29926a..425edab16e5eb1 100644
--- a/clang/lib/Headers/avx10_2_v2auxintrin.h
+++ b/clang/lib/Headers/avx10_2_v2auxintrin.h
@@ -56,12 +56,14 @@ static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvtps_bf8(__m128 __A) {
 /// \param __A
 ///    A 128-bit vector of [4 x float].
 /// \returns
-///    A 128-bit vector of [16 x i8] containing the converted values.
+///    A 128-bit vector of [16 x i8]. The lower 4 bytes contain the converted
+///    values, or the corresponding bytes of \a __W where the mask bit is clear;
+///    the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_mask_cvtps_bf8(__m128i __W,
                                                                    __mmask8 __U,
                                                                    __m128 __A) {
-  return (__m128i)__builtin_ia32_selectb_128(
-      (__mmask16)__U, (__v16qi)_mm_cvtps_bf8(__A), (__v16qi)__W);
+  return (__m128i)__builtin_ia32_vcvtps2bf8_128_mask((__v4sf)__A, (__v16qi)__W,
+                                                     (__mmask8)__U);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __A
@@ -77,12 +79,12 @@ static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_mask_cvtps_bf8(__m128i __W,
 /// \param __A
 ///    A 128-bit vector of [4 x float].
 /// \returns
-///    A 128-bit vector of [16 x i8] containing the converted values.
+///    A 128-bit vector of [16 x i8]. The lower 4 bytes contain the converted
+///    values, or zero where the mask bit is clear; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_maskz_cvtps_bf8(__mmask8 __U, __m128 __A) {
-  return (__m128i)__builtin_ia32_selectb_128((__mmask16)__U,
-                                             (__v16qi)_mm_cvtps_bf8(__A),
-                                             (__v16qi)_mm_setzero_si128());
+  return (__m128i)__builtin_ia32_vcvtps2bf8_128_mask(
+      (__v4sf)__A, (__v16qi)_mm_setzero_si128(), (__mmask8)__U);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __A
@@ -117,11 +119,13 @@ static __inline__ __m128i __DEFAULT_FN_ATTRS256 _mm256_cvtps_bf8(__m256 __A) {
 /// \param __A
 ///    A 256-bit vector of [8 x float].
 /// \returns
-///    A 128-bit vector of [16 x i8] containing the converted values.
+///    A 128-bit vector of [16 x i8]. The lower 8 bytes contain the converted
+///    values, or the corresponding bytes of \a __W where the mask bit is clear;
+///    the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_mask_cvtps_bf8(__m128i __W, __mmask8 __U, __m256 __A) {
-  return (__m128i)__builtin_ia32_selectb_128(
-      (__mmask16)__U, (__v16qi)_mm256_cvtps_bf8(__A), (__v16qi)__W);
+  return (__m128i)__builtin_ia32_vcvtps2bf8_256_mask((__v8sf)__A, (__v16qi)__W,
+                                                     (__mmask8)__U);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __A
@@ -137,12 +141,12 @@ _mm256_mask_cvtps_bf8(__m128i __W, __mmask8 __U, __m256 __A) {
 /// \param __A
 ///    A 256-bit vector of [8 x float].
 /// \returns
-///    A 128-bit vector of [16 x i8] containing the converted values.
+///    A 128-bit vector of [16 x i8]. The lower 8 bytes contain the converted
+///    values, or zero where the mask bit is clear; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_maskz_cvtps_bf8(__mmask8 __U, __m256 __A) {
-  return (__m128i)__builtin_ia32_selectb_128((__mmask16)__U,
-                                             (__v16qi)_mm256_cvtps_bf8(__A),
-                                             (__v16qi)_mm_setzero_si128());
+  return (__m128i)__builtin_ia32_vcvtps2bf8_256_mask(
+      (__v8sf)__A, (__v16qi)_mm_setzero_si128(), (__mmask8)__U);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __A
@@ -176,11 +180,13 @@ static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvts_ps_bf8(__m128 __A) {
 /// \param __A
 ///    A 128-bit vector of [4 x float].
 /// \returns
-///    A 128-bit vector of [16 x i8] containing the converted values.
+///    A 128-bit vector of [16 x i8]. The lower 4 bytes contain the converted
+///    values, or the corresponding bytes of \a __W where the mask bit is clear;
+///    the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_mask_cvts_ps_bf8(__m128i __W, __mmask8 __U, __m128 __A) {
-  return (__m128i)__builtin_ia32_selectb_128(
-      (__mmask16)__U, (__v16qi)_mm_cvts_ps_bf8(__A), (__v16qi)__W);
+  return (__m128i)__builtin_ia32_vcvtps2bf8s_128_mask((__v4sf)__A, (__v16qi)__W,
+                                                      (__mmask8)__U);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __A
@@ -196,12 +202,12 @@ _mm_mask_cvts_ps_bf8(__m128i __W, __mmask8 __U, __m128 __A) {
 /// \param __A
 ///    A 128-bit vector of [4 x float].
 /// \returns
-///    A 128-bit vector of [16 x i8] containing the converted values.
+///    A 128-bit vector of [16 x i8]. The lower 4 bytes contain the converted
+///    values, or zero where the mask bit is clear; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_maskz_cvts_ps_bf8(__mmask8 __U, __m128 __A) {
-  return (__m128i)__builtin_ia32_selectb_128((__mmask16)__U,
-                                             (__v16qi)_mm_cvts_ps_bf8(__A),
-                                             (__v16qi)_mm_setzero_si128());
+  return (__m128i)__builtin_ia32_vcvtps2bf8s_128_mask(
+      (__v4sf)__A, (__v16qi)_mm_setzero_si128(), (__mmask8)__U);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __A
@@ -235,11 +241,13 @@ static __inline__ __m128i __DEFAULT_FN_ATTRS256 _mm256_cvts_ps_bf8(__m256 __A) {
 /// \param __A
 ///    A 256-bit vector of [8 x float].
 /// \returns
-///    A 128-bit vector of [16 x i8] containing the converted values.
+///    A 128-bit vector of [16 x i8]. The lower 8 bytes contain the converted
+///    values, or the corresponding bytes of \a __W where the mask bit is clear;
+///    the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_mask_cvts_ps_bf8(__m128i __W, __mmask8 __U, __m256 __A) {
-  return (__m128i)__builtin_ia32_selectb_128(
-      (__mmask16)__U, (__v16qi)_mm256_cvts_ps_bf8(__A), (__v16qi)__W);
+  return (__m128i)__builtin_ia32_vcvtps2bf8s_256_mask((__v8sf)__A, (__v16qi)__W,
+                                                      (__mmask8)__U);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __A
@@ -255,12 +263,12 @@ _mm256_mask_cvts_ps_bf8(__m128i __W, __mmask8 __U, __m256 __A) {
 /// \param __A
 ///    A 256-bit vector of [8 x float].
 /// \returns
-///    A 128-bit vector of [16 x i8] containing the converted values.
+///    A 128-bit vector of [16 x i8]. The lower 8 bytes contain the converted
+///    values, or zero where the mask bit is clear; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_maskz_cvts_ps_bf8(__mmask8 __U, __m256 __A) {
-  return (__m128i)__builtin_ia32_selectb_128((__mmask16)__U,
-                                             (__v16qi)_mm256_cvts_ps_bf8(__A),
-                                             (__v16qi)_mm_setzero_si128());
+  return (__m128i)__builtin_ia32_vcvtps2bf8s_256_mask(
+      (__v8sf)__A, (__v16qi)_mm_setzero_si128(), (__mmask8)__U);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __A
@@ -294,12 +302,14 @@ static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvtps_hf8(__m128 __A) {
 /// \param __A
 ///    A 128-bit vector of [4 x float].
 /// \returns
-///    A 128-bit vector of [16 x i8] containing the converted values.
+///    A 128-bit vector of [16 x i8]. The lower 4 bytes contain the converted
+///    values, or the corresponding bytes of \a __W where the mask bit is clear;
+///    the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_mask_cvtps_hf8(__m128i __W,
                                                                    __mmask8 __U,
                                                                    __m128 __A) {
-  return (__m128i)__builtin_ia32_selectb_128(
-      (__mmask16)__U, (__v16qi)_mm_cvtps_hf8(__A), (__v16qi)__W);
+  return (__m128i)__builtin_ia32_vcvtps2hf8_128_mask((__v4sf)__A, (__v16qi)__W,
+                                                     (__mmask8)__U);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __A
@@ -315,12 +325,12 @@ static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_mask_cvtps_hf8(__m128i __W,
 /// \param __A
 ///    A 128-bit vector of [4 x float].
 /// \returns
-///    A 128-bit vector of [16 x i8] containing the converted values.
+///    A 128-bit vector of [16 x i8]. The lower 4 bytes contain the converted
+///    values, or zero where the mask bit is clear; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_maskz_cvtps_hf8(__mmask8 __U, __m128 __A) {
-  return (__m128i)__builtin_ia32_selectb_128((__mmask16)__U,
-                                             (__v16qi)_mm_cvtps_hf8(__A),
-                                             (__v16qi)_mm_setzero_si128());
+  return (__m128i)__builtin_ia32_vcvtps2hf8_128_mask(
+      (__v4sf)__A, (__v16qi)_mm_setzero_si128(), (__mmask8)__U);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __A
@@ -354,11 +364,13 @@ static __inline__ __m128i __DEFAULT_FN_ATTRS256 _mm256_cvtps_hf8(__m256 __A) {
 /// \param __A
 ///    A 256-bit vector of [8 x float].
 /// \returns
-///    A 128-bit vector of [16 x i8] containing the converted values.
+///    A 128-bit vector of [16 x i8]. The lower 8 bytes contain the converted
+///    values, or the corresponding bytes of \a __W where the mask bit is clear;
+///    the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_mask_cvtps_hf8(__m128i __W, __mmask8 __U, __m256 __A) {
-  return (__m128i)__builtin_ia32_selectb_128(
-      (__mmask16)__U, (__v16qi)_mm256_cvtps_hf8(__A), (__v16qi)__W);
+  return (__m128i)__builtin_ia32_vcvtps2hf8_256_mask((__v8sf)__A, (__v16qi)__W,
+                                                     (__mmask8)__U);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __A
@@ -374,12 +386,12 @@ _mm256_mask_cvtps_hf8(__m128i __W, __mmask8 __U, __m256 __A) {
 /// \param __A
 ///    A 256-bit vector of [8 x float].
 /// \returns
-///    A 128-bit vector of [16 x i8] containing the converted values.
+///    A 128-bit vector of [16 x i8]. The lower 8 bytes contain the converted
+///    values, or zero where the mask bit is clear; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_maskz_cvtps_hf8(__mmask8 __U, __m256 __A) {
-  return (__m128i)__builtin_ia32_selectb_128((__mmask16)__U,
-                                             (__v16qi)_mm256_cvtps_hf8(__A),
-                                             (__v16qi)_mm_setzero_si128());
+  return (__m128i)__builtin_ia32_vcvtps2hf8_256_mask(
+      (__v8sf)__A, (__v16qi)_mm_setzero_si128(), (__mmask8)__U);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __A
@@ -413,11 +425,13 @@ static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvts_ps_hf8(__m128 __A) {
 /// \param __A
 ///    A 128-bit vector of [4 x float].
 /// \returns
-///    A 128-bit vector of [16 x i8] containing the converted values.
+///    A 128-bit vector of [16 x i8]. The lower 4 bytes contain the converted
+///    values, or the corresponding bytes of \a __W where the mask bit is clear;
+///    the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_mask_cvts_ps_hf8(__m128i __W, __mmask8 __U, __m128 __A) {
-  return (__m128i)__builtin_ia32_selectb_128(
-      (__mmask16)__U, (__v16qi)_mm_cvts_ps_hf8(__A), (__v16qi)__W);
+  return (__m128i)__builtin_ia32_vcvtps2hf8s_128_mask((__v4sf)__A, (__v16qi)__W,
+                                                      (__mmask8)__U);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __A
@@ -433,12 +447,12 @@ _mm_mask_cvts_ps_hf8(__m128i __W, __mmask8 __U, __m128 __A) {
 /// \param __A
 ///    A 128-bit vector of [4 x float].
 /// \returns
-///    A 128-bit vector of [16 x i8] containing the converted values.
+///    A 128-bit vector of [16 x i8]. The lower 4 bytes contain the converted
+///    values, or zero where the mask bit is clear; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_maskz_cvts_ps_hf8(__mmask8 __U, __m128 __A) {
-  return (__m128i)__builtin_ia32_selectb_128((__mmask16)__U,
-                                             (__v16qi)_mm_cvts_ps_hf8(__A),
-                                             (__v16qi)_mm_setzero_si128());
+  return (__m128i)__builtin_ia32_vcvtps2hf8s_128_mask(
+      (__v4sf)__A, (__v16qi)_mm_setzero_si128(), (__mmask8)__U);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __A
@@ -472,11 +486,13 @@ static __inline__ __m128i __DEFAULT_FN_ATTRS256 _mm256_cvts_ps_hf8(__m256 __A) {
 /// \param __A
 ///    A 256-bit vector of [8 x float].
 /// \returns
-///    A 128-bit vector of [16 x i8] containing the converted values.
+///    A 128-bit vector of [16 x i8]. The lower 8 bytes contain the converted
+///    values, or the corresponding bytes of \a __W where the mask bit is clear;
+///    the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_mask_cvts_ps_hf8(__m128i __W, __mmask8 __U, __m256 __A) {
-  return (__m128i)__builtin_ia32_selectb_128(
-      (__mmask16)__U, (__v16qi)_mm256_cvts_ps_hf8(__A), (__v16qi)__W);
+  return (__m128i)__builtin_ia32_vcvtps2hf8s_256_mask((__v8sf)__A, (__v16qi)__W,
+                                                      (__mmask8)__U);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __A
@@ -492,12 +508,12 @@ _mm256_mask_cvts_ps_hf8(__m128i __W, __mmask8 __U, __m256 __A) {
 /// \param __A
 ///    A 256-bit vector of [8 x float].
 /// \returns
-///    A 128-bit vector of [16 x i8] containing the converted values.
+///    A 128-bit vector of [16 x i8]. The lower 8 bytes contain the converted
+///    values, or zero where the mask bit is clear; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_maskz_cvts_ps_hf8(__mmask8 __U, __m256 __A) {
-  return (__m128i)__builtin_ia32_selectb_128((__mmask16)__U,
-                                             (__v16qi)_mm256_cvts_ps_hf8(__A),
-                                             (__v16qi)_mm_setzero_si128());
+  return (__m128i)__builtin_ia32_vcvtps2hf8s_256_mask(
+      (__v8sf)__A, (__v16qi)_mm_setzero_si128(), (__mmask8)__U);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __A
@@ -531,11 +547,13 @@ static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvtrops_hf8(__m128 __A) {
 /// \param __A
 ///    A 128-bit vector of [4 x float].
 /// \returns
-///    A 128-bit vector of [16 x i8] containing the converted values.
+///    A 128-bit vector of [16 x i8]. The lower 4 bytes contain the converted
+///    values, or the corresponding bytes of \a __W where the mask bit is clear;
+///    the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_mask_cvtrops_hf8(__m128i __W, __mmask8 __U, __m128 __A) {
-  return (__m128i)__builtin_ia32_selectb_128(
-      (__mmask16)__U, (__v16qi)_mm_cvtrops_hf8(__A), (__v16qi)__W);
+  return (__m128i)__builtin_ia32_vcvtrops2hf8_128_mask(
+      (__v4sf)__A, (__v16qi)__W, (__mmask8)__U);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __A
@@ -551,12 +569,12 @@ _mm_mask_cvtrops_hf8(__m128i __W, __mmask8 __U, __m128 __A) {
 /// \param __A
 ///    A 128-bit vector of [4 x float].
 /// \returns
-///    A 128-bit vector of [16 x i8] containing the converted values.
+///    A 128-bit vector of [16 x i8]. The lower 4 bytes contain the converted
+///    values, or zero where the mask bit is clear; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_maskz_cvtrops_hf8(__mmask8 __U, __m128 __A) {
-  return (__m128i)__builtin_ia32_selectb_128((__mmask16)__U,
-                                             (__v16qi)_mm_cvtrops_hf8(__A),
-                                             (__v16qi)_mm_setzero_si128());
+  return (__m128i)__builtin_ia32_vcvtrops2hf8_128_mask(
+      (__v4sf)__A, (__v16qi)_mm_setzero_si128(), (__mmask8)__U);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __A
@@ -590,11 +608,13 @@ static __inline__ __m128i __DEFAULT_FN_ATTRS256 _mm256_cvtrops_hf8(__m256 __A) {
 /// \param __A
 ///    A 256-bit vector of [8 x float].
 /// \returns
-///    A 128-bit vector of [16 x i8] containing the converted values.
+///    A 128-bit vector of [16 x i8]. The lower 8 bytes contain the converted
+///    values, or the corresponding bytes of \a __W where the mask bit is clear;
+///    the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_mask_cvtrops_hf8(__m128i __W, __mmask8 __U, __m256 __A) {
-  return (__m128i)__builtin_ia32_selectb_128(
-      (__mmask16)__U, (__v16qi)_mm256_cvtrops_hf8(__A), (__v16qi)__W);
+  return (__m128i)__builtin_ia32_vcvtrops2hf8_256_mask(
+      (__v8sf)__A, (__v16qi)__W, (__mmask8)__U);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __A
@@ -610,12 +630,12 @@ _mm256_mask_cvtrops_hf8(__m128i __W, __mmask8 __U, __m256 __A) {
 /// \param __A
 ///    A 256-bit vector of [8 x float].
 /// \returns
-///    A 128-bit vector of [16 x i8] containing the converted values.
+///    A 128-bit vector of [16 x i8]. The lower 8 bytes contain the converted
+///    values, or zero where the mask bit is clear; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_maskz_cvtrops_hf8(__mmask8 __U, __m256 __A) {
-  return (__m128i)__builtin_ia32_selectb_128((__mmask16)__U,
-                                             (__v16qi)_mm256_cvtrops_hf8(__A),
-                                             (__v16qi)_mm_setzero_si128());
+  return (__m128i)__builtin_ia32_vcvtrops2hf8_256_mask(
+      (__v8sf)__A, (__v16qi)_mm_setzero_si128(), (__mmask8)__U);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __A
@@ -649,11 +669,13 @@ static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvts_rops_hf8(__m128 __A) {
 /// \param __A
 ///    A 128-bit vector of [4 x float].
 /// \returns
-///    A 128-bit vector of [16 x i8] containing the converted values.
+///    A 128-bit vector of [16 x i8]. The lower 4 bytes contain the converted
+///    values, or the corresponding bytes of \a __W where the mask bit is clear;
+///    the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_mask_cvts_rops_hf8(__m128i __W, __mmask8 __U, __m128 __A) {
-  return (__m128i)__builtin_ia32_selectb_128(
-      (__mmask16)__U, (__v16qi)_mm_cvts_rops_hf8(__A), (__v16qi)__W);
+  return (__m128i)__builtin_ia32_vcvtrops2hf8s_128_mask(
+      (__v4sf)__A, (__v16qi)__W, (__mmask8)__U);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __A
@@ -669,12 +691,12 @@ _mm_mask_cvts_rops_hf8(__m128i __W, __mmask8 __U, __m128 __A) {
 /// \param __A
 ///    A 128-bit vector of [4 x float].
 /// \returns
-///    A 128-bit vector of [16 x i8] containing the converted values.
+///    A 128-bit vector of [16 x i8]. The lower 4 bytes contain the converted
+///    values, or zero where the mask bit is clear; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_maskz_cvts_rops_hf8(__mmask8 __U, __m128 __A) {
-  return (__m128i)__builtin_ia32_selectb_128((__mmask16)__U,
-                                             (__v16qi)_mm_cvts_rops_hf8(__A),
-                                             (__v16qi)_mm_setzero_si128());
+  return (__m128i)__builtin_ia32_vcvtrops2hf8s_128_mask(
+      (__v4sf)__A, (__v16qi)_mm_setzero_si128(), (__mmask8)__U);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __A
@@ -709,11 +731,13 @@ _mm256_cvts_rops_hf8(__m256 __A) {
 /// \param __A
 ///    A 256-bit vector of [8 x float].
 /// \returns
-///    A 128-bit vector of [16 x i8] containing the converted values.
+///    A 128-bit vector of [16 x i8]. The lower 8 bytes contain the converted
+///    values, or the corresponding bytes of \a __W where the mask bit is clear;
+///    the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_mask_cvts_rops_hf8(__m128i __W, __mmask8 __U, __m256 __A) {
-  return (__m128i)__builtin_ia32_selectb_128(
-      (__mmask16)__U, (__v16qi)_mm256_cvts_rops_hf8(__A), (__v16qi)__W);
+  return (__m128i)__builtin_ia32_vcvtrops2hf8s_256_mask(
+      (__v8sf)__A, (__v16qi)__W, (__mmask8)__U);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __A
@@ -729,12 +753,12 @@ _mm256_mask_cvts_rops_hf8(__m128i __W, __mmask8 __U, __m256 __A) {
 /// \param __A
 ///    A 256-bit vector of [8 x float].
 /// \returns
-///    A 128-bit vector of [16 x i8] containing the converted values.
+///    A 128-bit vector of [16 x i8]. The lower 8 bytes contain the converted
+///    values, or zero where the mask bit is clear; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_maskz_cvts_rops_hf8(__mmask8 __U, __m256 __A) {
-  return (__m128i)__builtin_ia32_selectb_128((__mmask16)__U,
-                                             (__v16qi)_mm256_cvts_rops_hf8(__A),
-                                             (__v16qi)_mm_setzero_si128());
+  return (__m128i)__builtin_ia32_vcvtrops2hf8s_256_mask(
+      (__v8sf)__A, (__v16qi)_mm_setzero_si128(), (__mmask8)__U);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __B
@@ -746,14 +770,14 @@ _mm256_maskz_cvts_rops_hf8(__mmask8 __U, __m256 __A) {
 /// This intrinsic corresponds to the \c VCVTBIASPS2BF8 instruction.
 ///
 /// \param __A
-///    A 128-bit vector of [16 x i8] containing bias values.
+///    A 128-bit vector of [4 x i32] containing bias values.
 /// \param __B
 ///    A 128-bit vector of [4 x float].
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvtbiasps_bf8(__m128i __A,
                                                                   __m128 __B) {
-  return (__m128i)__builtin_ia32_vcvtbiasps2bf8_128((__v16qi)__A, (__v4sf)__B);
+  return (__m128i)__builtin_ia32_vcvtbiasps2bf8_128((__v4si)__A, (__v4sf)__B);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __B
@@ -769,15 +793,17 @@ static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvtbiasps_bf8(__m128i __A,
 /// \param __U
 ///    A 8-bit mask indicating which elements to write.
 /// \param __A
-///    A 128-bit vector of [16 x i8] containing bias values.
+///    A 128-bit vector of [4 x i32] containing bias values.
 /// \param __B
 ///    A 128-bit vector of [4 x float].
 /// \returns
-///    A 128-bit vector of [16 x i8] containing the converted values.
+///    A 128-bit vector of [16 x i8]. The lower 4 bytes contain the converted
+///    values, or the corresponding bytes of \a __W where the mask bit is clear;
+///    the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_mask_cvtbiasps_bf8(__m128i __W, __mmask8 __U, __m128i __A, __m128 __B) {
-  return (__m128i)__builtin_ia32_selectb_128(
-      (__mmask16)__U, (__v16qi)_mm_cvtbiasps_bf8(__A, __B), (__v16qi)__W);
+  return (__m128i)__builtin_ia32_vcvtbiasps2bf8_128_mask(
+      (__v4si)__A, (__v4sf)__B, (__v16qi)__W, (__mmask8)__U);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __B
@@ -791,16 +817,16 @@ _mm_mask_cvtbiasps_bf8(__m128i __W, __mmask8 __U, __m128i __A, __m128 __B) {
 /// \param __U
 ///    A 8-bit mask indicating which elements to write (zero otherwise).
 /// \param __A
-///    A 128-bit vector of [16 x i8] containing bias values.
+///    A 128-bit vector of [4 x i32] containing bias values.
 /// \param __B
 ///    A 128-bit vector of [4 x float].
 /// \returns
-///    A 128-bit vector of [16 x i8] containing the converted values.
+///    A 128-bit vector of [16 x i8]. The lower 4 bytes contain the converted
+///    values, or zero where the mask bit is clear; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_maskz_cvtbiasps_bf8(__mmask8 __U, __m128i __A, __m128 __B) {
-  return (__m128i)__builtin_ia32_selectb_128(
-      (__mmask16)__U, (__v16qi)_mm_cvtbiasps_bf8(__A, __B),
-      (__v16qi)_mm_setzero_si128());
+  return (__m128i)__builtin_ia32_vcvtbiasps2bf8_128_mask(
+      (__v4si)__A, (__v4sf)__B, (__v16qi)_mm_setzero_si128(), (__mmask8)__U);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __B
@@ -812,14 +838,14 @@ _mm_maskz_cvtbiasps_bf8(__mmask8 __U, __m128i __A, __m128 __B) {
 /// This intrinsic corresponds to the \c VCVTBIASPS2BF8 instruction.
 ///
 /// \param __A
-///    A 256-bit vector of [32 x i8] containing bias values.
+///    A 256-bit vector of [8 x i32] containing bias values.
 /// \param __B
 ///    A 256-bit vector of [8 x float].
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_cvtbiasps_bf8(__m256i __A, __m256 __B) {
-  return (__m128i)__builtin_ia32_vcvtbiasps2bf8_256((__v32qi)__A, (__v8sf)__B);
+  return (__m128i)__builtin_ia32_vcvtbiasps2bf8_256((__v8si)__A, (__v8sf)__B);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __B
@@ -835,15 +861,17 @@ _mm256_cvtbiasps_bf8(__m256i __A, __m256 __B) {
 /// \param __U
 ///    A 8-bit mask indicating which elements to write.
 /// \param __A
-///    A 256-bit vector of [32 x i8] containing bias values.
+///    A 256-bit vector of [8 x i32] containing bias values.
 /// \param __B
 ///    A 256-bit vector of [8 x float].
 /// \returns
-///    A 128-bit vector of [16 x i8] containing the converted values.
+///    A 128-bit vector of [16 x i8]. The lower 8 bytes contain the converted
+///    values, or the corresponding bytes of \a __W where the mask bit is clear;
+///    the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_mask_cvtbiasps_bf8(__m128i __W, __mmask8 __U, __m256i __A, __m256 __B) {
-  return (__m128i)__builtin_ia32_selectb_128(
-      (__mmask16)__U, (__v16qi)_mm256_cvtbiasps_bf8(__A, __B), (__v16qi)__W);
+  return (__m128i)__builtin_ia32_vcvtbiasps2bf8_256_mask(
+      (__v8si)__A, (__v8sf)__B, (__v16qi)__W, (__mmask8)__U);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __B
@@ -857,16 +885,16 @@ _mm256_mask_cvtbiasps_bf8(__m128i __W, __mmask8 __U, __m256i __A, __m256 __B) {
 /// \param __U
 ///    A 8-bit mask indicating which elements to write (zero otherwise).
 /// \param __A
-///    A 256-bit vector of [32 x i8] containing bias values.
+///    A 256-bit vector of [8 x i32] containing bias values.
 /// \param __B
 ///    A 256-bit vector of [8 x float].
 /// \returns
-///    A 128-bit vector of [16 x i8] containing the converted values.
+///    A 128-bit vector of [16 x i8]. The lower 8 bytes contain the converted
+///    values, or zero where the mask bit is clear; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_maskz_cvtbiasps_bf8(__mmask8 __U, __m256i __A, __m256 __B) {
-  return (__m128i)__builtin_ia32_selectb_128(
-      (__mmask16)__U, (__v16qi)_mm256_cvtbiasps_bf8(__A, __B),
-      (__v16qi)_mm_setzero_si128());
+  return (__m128i)__builtin_ia32_vcvtbiasps2bf8_256_mask(
+      (__v8si)__A, (__v8sf)__B, (__v16qi)_mm_setzero_si128(), (__mmask8)__U);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __B
@@ -878,14 +906,14 @@ _mm256_maskz_cvtbiasps_bf8(__mmask8 __U, __m256i __A, __m256 __B) {
 /// This intrinsic corresponds to the \c VCVTBIASPS2BF8S instruction.
 ///
 /// \param __A
-///    A 128-bit vector of [16 x i8] containing bias values.
+///    A 128-bit vector of [4 x i32] containing bias values.
 /// \param __B
 ///    A 128-bit vector of [4 x float].
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_cvts_biasps_bf8(__m128i __A, __m128 __B) {
-  return (__m128i)__builtin_ia32_vcvtbiasps2bf8s_128((__v16qi)__A, (__v4sf)__B);
+  return (__m128i)__builtin_ia32_vcvtbiasps2bf8s_128((__v4si)__A, (__v4sf)__B);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __B
@@ -901,15 +929,17 @@ _mm_cvts_biasps_bf8(__m128i __A, __m128 __B) {
 /// \param __U
 ///    A 8-bit mask indicating which elements to write.
 /// \param __A
-///    A 128-bit vector of [16 x i8] containing bias values.
+///    A 128-bit vector of [4 x i32] containing bias values.
 /// \param __B
 ///    A 128-bit vector of [4 x float].
 /// \returns
-///    A 128-bit vector of [16 x i8] containing the converted values.
+///    A 128-bit vector of [16 x i8]. The lower 4 bytes contain the converted
+///    values, or the corresponding bytes of \a __W where the mask bit is clear;
+///    the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_mask_cvts_biasps_bf8(__m128i __W, __mmask8 __U, __m128i __A, __m128 __B) {
-  return (__m128i)__builtin_ia32_selectb_128(
-      (__mmask16)__U, (__v16qi)_mm_cvts_biasps_bf8(__A, __B), (__v16qi)__W);
+  return (__m128i)__builtin_ia32_vcvtbiasps2bf8s_128_mask(
+      (__v4si)__A, (__v4sf)__B, (__v16qi)__W, (__mmask8)__U);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __B
@@ -923,16 +953,16 @@ _mm_mask_cvts_biasps_bf8(__m128i __W, __mmask8 __U, __m128i __A, __m128 __B) {
 /// \param __U
 ///    A 8-bit mask indicating which elements to write (zero otherwise).
 /// \param __A
-///    A 128-bit vector of [16 x i8] containing bias values.
+///    A 128-bit vector of [4 x i32] containing bias values.
 /// \param __B
 ///    A 128-bit vector of [4 x float].
 /// \returns
-///    A 128-bit vector of [16 x i8] containing the converted values.
+///    A 128-bit vector of [16 x i8]. The lower 4 bytes contain the converted
+///    values, or zero where the mask bit is clear; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_maskz_cvts_biasps_bf8(__mmask8 __U, __m128i __A, __m128 __B) {
-  return (__m128i)__builtin_ia32_selectb_128(
-      (__mmask16)__U, (__v16qi)_mm_cvts_biasps_bf8(__A, __B),
-      (__v16qi)_mm_setzero_si128());
+  return (__m128i)__builtin_ia32_vcvtbiasps2bf8s_128_mask(
+      (__v4si)__A, (__v4sf)__B, (__v16qi)_mm_setzero_si128(), (__mmask8)__U);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __B
@@ -944,14 +974,14 @@ _mm_maskz_cvts_biasps_bf8(__mmask8 __U, __m128i __A, __m128 __B) {
 /// This intrinsic corresponds to the \c VCVTBIASPS2BF8S instruction.
 ///
 /// \param __A
-///    A 256-bit vector of [32 x i8] containing bias values.
+///    A 256-bit vector of [8 x i32] containing bias values.
 /// \param __B
 ///    A 256-bit vector of [8 x float].
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_cvts_biasps_bf8(__m256i __A, __m256 __B) {
-  return (__m128i)__builtin_ia32_vcvtbiasps2bf8s_256((__v32qi)__A, (__v8sf)__B);
+  return (__m128i)__builtin_ia32_vcvtbiasps2bf8s_256((__v8si)__A, (__v8sf)__B);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __B
@@ -967,15 +997,17 @@ _mm256_cvts_biasps_bf8(__m256i __A, __m256 __B) {
 /// \param __U
 ///    A 8-bit mask indicating which elements to write.
 /// \param __A
-///    A 256-bit vector of [32 x i8] containing bias values.
+///    A 256-bit vector of [8 x i32] containing bias values.
 /// \param __B
 ///    A 256-bit vector of [8 x float].
 /// \returns
-///    A 128-bit vector of [16 x i8] containing the converted values.
+///    A 128-bit vector of [16 x i8]. The lower 8 bytes contain the converted
+///    values, or the corresponding bytes of \a __W where the mask bit is clear;
+///    the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256 _mm256_mask_cvts_biasps_bf8(
     __m128i __W, __mmask8 __U, __m256i __A, __m256 __B) {
-  return (__m128i)__builtin_ia32_selectb_128(
-      (__mmask16)__U, (__v16qi)_mm256_cvts_biasps_bf8(__A, __B), (__v16qi)__W);
+  return (__m128i)__builtin_ia32_vcvtbiasps2bf8s_256_mask(
+      (__v8si)__A, (__v8sf)__B, (__v16qi)__W, (__mmask8)__U);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __B
@@ -989,16 +1021,16 @@ static __inline__ __m128i __DEFAULT_FN_ATTRS256 _mm256_mask_cvts_biasps_bf8(
 /// \param __U
 ///    A 8-bit mask indicating which elements to write (zero otherwise).
 /// \param __A
-///    A 256-bit vector of [32 x i8] containing bias values.
+///    A 256-bit vector of [8 x i32] containing bias values.
 /// \param __B
 ///    A 256-bit vector of [8 x float].
 /// \returns
-///    A 128-bit vector of [16 x i8] containing the converted values.
+///    A 128-bit vector of [16 x i8]. The lower 8 bytes contain the converted
+///    values, or zero where the mask bit is clear; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_maskz_cvts_biasps_bf8(__mmask8 __U, __m256i __A, __m256 __B) {
-  return (__m128i)__builtin_ia32_selectb_128(
-      (__mmask16)__U, (__v16qi)_mm256_cvts_biasps_bf8(__A, __B),
-      (__v16qi)_mm_setzero_si128());
+  return (__m128i)__builtin_ia32_vcvtbiasps2bf8s_256_mask(
+      (__v8si)__A, (__v8sf)__B, (__v16qi)_mm_setzero_si128(), (__mmask8)__U);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __B
@@ -1010,14 +1042,14 @@ _mm256_maskz_cvts_biasps_bf8(__mmask8 __U, __m256i __A, __m256 __B) {
 /// This intrinsic corresponds to the \c VCVTBIASPS2HF8 instruction.
 ///
 /// \param __A
-///    A 128-bit vector of [16 x i8] containing bias values.
+///    A 128-bit vector of [4 x i32] containing bias values.
 /// \param __B
 ///    A 128-bit vector of [4 x float].
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvtbiasps_hf8(__m128i __A,
                                                                   __m128 __B) {
-  return (__m128i)__builtin_ia32_vcvtbiasps2hf8_128((__v16qi)__A, (__v4sf)__B);
+  return (__m128i)__builtin_ia32_vcvtbiasps2hf8_128((__v4si)__A, (__v4sf)__B);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __B
@@ -1033,15 +1065,17 @@ static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvtbiasps_hf8(__m128i __A,
 /// \param __U
 ///    A 8-bit mask indicating which elements to write.
 /// \param __A
-///    A 128-bit vector of [16 x i8] containing bias values.
+///    A 128-bit vector of [4 x i32] containing bias values.
 /// \param __B
 ///    A 128-bit vector of [4 x float].
 /// \returns
-///    A 128-bit vector of [16 x i8] containing the converted values.
+///    A 128-bit vector of [16 x i8]. The lower 4 bytes contain the converted
+///    values, or the corresponding bytes of \a __W where the mask bit is clear;
+///    the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_mask_cvtbiasps_hf8(__m128i __W, __mmask8 __U, __m128i __A, __m128 __B) {
-  return (__m128i)__builtin_ia32_selectb_128(
-      (__mmask16)__U, (__v16qi)_mm_cvtbiasps_hf8(__A, __B), (__v16qi)__W);
+  return (__m128i)__builtin_ia32_vcvtbiasps2hf8_128_mask(
+      (__v4si)__A, (__v4sf)__B, (__v16qi)__W, (__mmask8)__U);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __B
@@ -1055,16 +1089,16 @@ _mm_mask_cvtbiasps_hf8(__m128i __W, __mmask8 __U, __m128i __A, __m128 __B) {
 /// \param __U
 ///    A 8-bit mask indicating which elements to write (zero otherwise).
 /// \param __A
-///    A 128-bit vector of [16 x i8] containing bias values.
+///    A 128-bit vector of [4 x i32] containing bias values.
 /// \param __B
 ///    A 128-bit vector of [4 x float].
 /// \returns
-///    A 128-bit vector of [16 x i8] containing the converted values.
+///    A 128-bit vector of [16 x i8]. The lower 4 bytes contain the converted
+///    values, or zero where the mask bit is clear; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_maskz_cvtbiasps_hf8(__mmask8 __U, __m128i __A, __m128 __B) {
-  return (__m128i)__builtin_ia32_selectb_128(
-      (__mmask16)__U, (__v16qi)_mm_cvtbiasps_hf8(__A, __B),
-      (__v16qi)_mm_setzero_si128());
+  return (__m128i)__builtin_ia32_vcvtbiasps2hf8_128_mask(
+      (__v4si)__A, (__v4sf)__B, (__v16qi)_mm_setzero_si128(), (__mmask8)__U);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __B
@@ -1076,14 +1110,14 @@ _mm_maskz_cvtbiasps_hf8(__mmask8 __U, __m128i __A, __m128 __B) {
 /// This intrinsic corresponds to the \c VCVTBIASPS2HF8 instruction.
 ///
 /// \param __A
-///    A 256-bit vector of [32 x i8] containing bias values.
+///    A 256-bit vector of [8 x i32] containing bias values.
 /// \param __B
 ///    A 256-bit vector of [8 x float].
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_cvtbiasps_hf8(__m256i __A, __m256 __B) {
-  return (__m128i)__builtin_ia32_vcvtbiasps2hf8_256((__v32qi)__A, (__v8sf)__B);
+  return (__m128i)__builtin_ia32_vcvtbiasps2hf8_256((__v8si)__A, (__v8sf)__B);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __B
@@ -1099,15 +1133,17 @@ _mm256_cvtbiasps_hf8(__m256i __A, __m256 __B) {
 /// \param __U
 ///    A 8-bit mask indicating which elements to write.
 /// \param __A
-///    A 256-bit vector of [32 x i8] containing bias values.
+///    A 256-bit vector of [8 x i32] containing bias values.
 /// \param __B
 ///    A 256-bit vector of [8 x float].
 /// \returns
-///    A 128-bit vector of [16 x i8] containing the converted values.
+///    A 128-bit vector of [16 x i8]. The lower 8 bytes contain the converted
+///    values, or the corresponding bytes of \a __W where the mask bit is clear;
+///    the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_mask_cvtbiasps_hf8(__m128i __W, __mmask8 __U, __m256i __A, __m256 __B) {
-  return (__m128i)__builtin_ia32_selectb_128(
-      (__mmask16)__U, (__v16qi)_mm256_cvtbiasps_hf8(__A, __B), (__v16qi)__W);
+  return (__m128i)__builtin_ia32_vcvtbiasps2hf8_256_mask(
+      (__v8si)__A, (__v8sf)__B, (__v16qi)__W, (__mmask8)__U);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __B
@@ -1121,16 +1157,16 @@ _mm256_mask_cvtbiasps_hf8(__m128i __W, __mmask8 __U, __m256i __A, __m256 __B) {
 /// \param __U
 ///    A 8-bit mask indicating which elements to write (zero otherwise).
 /// \param __A
-///    A 256-bit vector of [32 x i8] containing bias values.
+///    A 256-bit vector of [8 x i32] containing bias values.
 /// \param __B
 ///    A 256-bit vector of [8 x float].
 /// \returns
-///    A 128-bit vector of [16 x i8] containing the converted values.
+///    A 128-bit vector of [16 x i8]. The lower 8 bytes contain the converted
+///    values, or zero where the mask bit is clear; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_maskz_cvtbiasps_hf8(__mmask8 __U, __m256i __A, __m256 __B) {
-  return (__m128i)__builtin_ia32_selectb_128(
-      (__mmask16)__U, (__v16qi)_mm256_cvtbiasps_hf8(__A, __B),
-      (__v16qi)_mm_setzero_si128());
+  return (__m128i)__builtin_ia32_vcvtbiasps2hf8_256_mask(
+      (__v8si)__A, (__v8sf)__B, (__v16qi)_mm_setzero_si128(), (__mmask8)__U);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __B
@@ -1142,14 +1178,14 @@ _mm256_maskz_cvtbiasps_hf8(__mmask8 __U, __m256i __A, __m256 __B) {
 /// This intrinsic corresponds to the \c VCVTBIASPS2HF8S instruction.
 ///
 /// \param __A
-///    A 128-bit vector of [16 x i8] containing bias values.
+///    A 128-bit vector of [4 x i32] containing bias values.
 /// \param __B
 ///    A 128-bit vector of [4 x float].
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_cvts_biasps_hf8(__m128i __A, __m128 __B) {
-  return (__m128i)__builtin_ia32_vcvtbiasps2hf8s_128((__v16qi)__A, (__v4sf)__B);
+  return (__m128i)__builtin_ia32_vcvtbiasps2hf8s_128((__v4si)__A, (__v4sf)__B);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __B
@@ -1165,15 +1201,17 @@ _mm_cvts_biasps_hf8(__m128i __A, __m128 __B) {
 /// \param __U
 ///    A 8-bit mask indicating which elements to write.
 /// \param __A
-///    A 128-bit vector of [16 x i8] containing bias values.
+///    A 128-bit vector of [4 x i32] containing bias values.
 /// \param __B
 ///    A 128-bit vector of [4 x float].
 /// \returns
-///    A 128-bit vector of [16 x i8] containing the converted values.
+///    A 128-bit vector of [16 x i8]. The lower 4 bytes contain the converted
+///    values, or the corresponding bytes of \a __W where the mask bit is clear;
+///    the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_mask_cvts_biasps_hf8(__m128i __W, __mmask8 __U, __m128i __A, __m128 __B) {
-  return (__m128i)__builtin_ia32_selectb_128(
-      (__mmask16)__U, (__v16qi)_mm_cvts_biasps_hf8(__A, __B), (__v16qi)__W);
+  return (__m128i)__builtin_ia32_vcvtbiasps2hf8s_128_mask(
+      (__v4si)__A, (__v4sf)__B, (__v16qi)__W, (__mmask8)__U);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __B
@@ -1187,16 +1225,16 @@ _mm_mask_cvts_biasps_hf8(__m128i __W, __mmask8 __U, __m128i __A, __m128 __B) {
 /// \param __U
 ///    A 8-bit mask indicating which elements to write (zero otherwise).
 /// \param __A
-///    A 128-bit vector of [16 x i8] containing bias values.
+///    A 128-bit vector of [4 x i32] containing bias values.
 /// \param __B
 ///    A 128-bit vector of [4 x float].
 /// \returns
-///    A 128-bit vector of [16 x i8] containing the converted values.
+///    A 128-bit vector of [16 x i8]. The lower 4 bytes contain the converted
+///    values, or zero where the mask bit is clear; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_maskz_cvts_biasps_hf8(__mmask8 __U, __m128i __A, __m128 __B) {
-  return (__m128i)__builtin_ia32_selectb_128(
-      (__mmask16)__U, (__v16qi)_mm_cvts_biasps_hf8(__A, __B),
-      (__v16qi)_mm_setzero_si128());
+  return (__m128i)__builtin_ia32_vcvtbiasps2hf8s_128_mask(
+      (__v4si)__A, (__v4sf)__B, (__v16qi)_mm_setzero_si128(), (__mmask8)__U);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __B
@@ -1208,14 +1246,14 @@ _mm_maskz_cvts_biasps_hf8(__mmask8 __U, __m128i __A, __m128 __B) {
 /// This intrinsic corresponds to the \c VCVTBIASPS2HF8S instruction.
 ///
 /// \param __A
-///    A 256-bit vector of [32 x i8] containing bias values.
+///    A 256-bit vector of [8 x i32] containing bias values.
 /// \param __B
 ///    A 256-bit vector of [8 x float].
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_cvts_biasps_hf8(__m256i __A, __m256 __B) {
-  return (__m128i)__builtin_ia32_vcvtbiasps2hf8s_256((__v32qi)__A, (__v8sf)__B);
+  return (__m128i)__builtin_ia32_vcvtbiasps2hf8s_256((__v8si)__A, (__v8sf)__B);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __B
@@ -1231,15 +1269,17 @@ _mm256_cvts_biasps_hf8(__m256i __A, __m256 __B) {
 /// \param __U
 ///    A 8-bit mask indicating which elements to write.
 /// \param __A
-///    A 256-bit vector of [32 x i8] containing bias values.
+///    A 256-bit vector of [8 x i32] containing bias values.
 /// \param __B
 ///    A 256-bit vector of [8 x float].
 /// \returns
-///    A 128-bit vector of [16 x i8] containing the converted values.
+///    A 128-bit vector of [16 x i8]. The lower 8 bytes contain the converted
+///    values, or the corresponding bytes of \a __W where the mask bit is clear;
+///    the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256 _mm256_mask_cvts_biasps_hf8(
     __m128i __W, __mmask8 __U, __m256i __A, __m256 __B) {
-  return (__m128i)__builtin_ia32_selectb_128(
-      (__mmask16)__U, (__v16qi)_mm256_cvts_biasps_hf8(__A, __B), (__v16qi)__W);
+  return (__m128i)__builtin_ia32_vcvtbiasps2hf8s_256_mask(
+      (__v8si)__A, (__v8sf)__B, (__v16qi)__W, (__mmask8)__U);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __B
@@ -1253,16 +1293,16 @@ static __inline__ __m128i __DEFAULT_FN_ATTRS256 _mm256_mask_cvts_biasps_hf8(
 /// \param __U
 ///    A 8-bit mask indicating which elements to write (zero otherwise).
 /// \param __A
-///    A 256-bit vector of [32 x i8] containing bias values.
+///    A 256-bit vector of [8 x i32] containing bias values.
 /// \param __B
 ///    A 256-bit vector of [8 x float].
 /// \returns
-///    A 128-bit vector of [16 x i8] containing the converted values.
+///    A 128-bit vector of [16 x i8]. The lower 8 bytes contain the converted
+///    values, or zero where the mask bit is clear; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_maskz_cvts_biasps_hf8(__mmask8 __U, __m256i __A, __m256 __B) {
-  return (__m128i)__builtin_ia32_selectb_128(
-      (__mmask16)__U, (__v16qi)_mm256_cvts_biasps_hf8(__A, __B),
-      (__v16qi)_mm_setzero_si128());
+  return (__m128i)__builtin_ia32_vcvtbiasps2hf8s_256_mask(
+      (__v8si)__A, (__v8sf)__B, (__v16qi)_mm_setzero_si128(), (__mmask8)__U);
 }
 
 /// Convert packed BF8 (8-bit) floating-point elements in \a __A to packed
@@ -1513,7 +1553,7 @@ _mm256_maskz_cvthf8_ps(__mmask8 __U, __m128i __A) {
 ///    A 128-bit vector of [16 x i8] containing BF8 values.
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted BF6 values.
-static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvtbf8_bf6s(__m128i __A) {
+static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvts_bf8_bf6(__m128i __A) {
   return (__m128i)__builtin_ia32_vcvtbf82bf6s_128((__v16qi)__A);
 }
 
@@ -1530,7 +1570,7 @@ static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvtbf8_bf6s(__m128i __A) {
 /// \returns
 ///    A 256-bit vector of [32 x i8] containing the converted BF6 values.
 static __inline__ __m256i __DEFAULT_FN_ATTRS256
-_mm256_cvtbf8_bf6s(__m256i __A) {
+_mm256_cvts_bf8_bf6(__m256i __A) {
   return (__m256i)__builtin_ia32_vcvtbf82bf6s_256((__v32qi)__A);
 }
 
@@ -1546,7 +1586,7 @@ _mm256_cvtbf8_bf6s(__m256i __A) {
 ///    A 128-bit vector of [16 x i8] containing HF8 values.
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted HF6 values.
-static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvthf8_hf6s(__m128i __A) {
+static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvts_hf8_hf6(__m128i __A) {
   return (__m128i)__builtin_ia32_vcvthf82hf6s_128((__v16qi)__A);
 }
 
@@ -1563,7 +1603,7 @@ static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvthf8_hf6s(__m128i __A) {
 /// \returns
 ///    A 256-bit vector of [32 x i8] containing the converted HF6 values.
 static __inline__ __m256i __DEFAULT_FN_ATTRS256
-_mm256_cvthf8_hf6s(__m256i __A) {
+_mm256_cvts_hf8_hf6(__m256i __A) {
   return (__m256i)__builtin_ia32_vcvthf82hf6s_256((__v32qi)__A);
 }
 
@@ -1580,7 +1620,7 @@ _mm256_cvthf8_hf6s(__m256i __A) {
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted BF4 values
 ///    (lower 8 bytes used).
-static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvtbf8_bf4s(__m128i __A) {
+static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvts_bf8_bf4(__m128i __A) {
   return (__m128i)__builtin_ia32_vcvtbf82bf4s_128((__v16qi)__A);
 }
 
@@ -1597,7 +1637,7 @@ static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvtbf8_bf4s(__m128i __A) {
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted BF4 values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
-_mm256_cvtbf8_bf4s(__m256i __A) {
+_mm256_cvts_bf8_bf4(__m256i __A) {
   return (__m128i)__builtin_ia32_vcvtbf82bf4s_256((__v32qi)__A);
 }
 
@@ -1614,7 +1654,7 @@ _mm256_cvtbf8_bf4s(__m256i __A) {
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted BF4 values
 ///    (lower 8 bytes used).
-static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvthf8_bf4s(__m128i __A) {
+static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvts_hf8_bf4(__m128i __A) {
   return (__m128i)__builtin_ia32_vcvthf82bf4s_128((__v16qi)__A);
 }
 
@@ -1631,7 +1671,7 @@ static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvthf8_bf4s(__m128i __A) {
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted BF4 values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
-_mm256_cvthf8_bf4s(__m256i __A) {
+_mm256_cvts_hf8_bf4(__m256i __A) {
   return (__m128i)__builtin_ia32_vcvthf82bf4s_256((__v32qi)__A);
 }
 
@@ -1649,7 +1689,7 @@ _mm256_cvthf8_bf4s(__m256i __A) {
 /// \param __A
 ///    A 128-bit vector of [16 x i8] containing BF8 values.
 static __inline__ void __DEFAULT_FN_ATTRS128
-_mm_cvtbf8_bf4s_storeu(void *__P, __m128i __A) {
+_mm_cvts_bf8_bf4_storeu(void *__P, __m128i __A) {
   __builtin_ia32_vcvtbf82bf4s_128_mem(__P, (__v16qi)__A);
 }
 
@@ -1667,7 +1707,7 @@ _mm_cvtbf8_bf4s_storeu(void *__P, __m128i __A) {
 /// \param __A
 ///    A 256-bit vector of [32 x i8] containing BF8 values.
 static __inline__ void __DEFAULT_FN_ATTRS256
-_mm256_cvtbf8_bf4s_storeu(void *__P, __m256i __A) {
+_mm256_cvts_bf8_bf4_storeu(void *__P, __m256i __A) {
   __builtin_ia32_vcvtbf82bf4s_256_mem(__P, (__v32qi)__A);
 }
 
@@ -1685,7 +1725,7 @@ _mm256_cvtbf8_bf4s_storeu(void *__P, __m256i __A) {
 /// \param __A
 ///    A 128-bit vector of [16 x i8] containing HF8 values.
 static __inline__ void __DEFAULT_FN_ATTRS128
-_mm_cvthf8_bf4s_storeu(void *__P, __m128i __A) {
+_mm_cvts_hf8_bf4_storeu(void *__P, __m128i __A) {
   __builtin_ia32_vcvthf82bf4s_128_mem(__P, (__v16qi)__A);
 }
 
@@ -1703,7 +1743,7 @@ _mm_cvthf8_bf4s_storeu(void *__P, __m128i __A) {
 /// \param __A
 ///    A 256-bit vector of [32 x i8] containing HF8 values.
 static __inline__ void __DEFAULT_FN_ATTRS256
-_mm256_cvthf8_bf4s_storeu(void *__P, __m256i __A) {
+_mm256_cvts_hf8_bf4_storeu(void *__P, __m256i __A) {
   __builtin_ia32_vcvthf82bf4s_256_mem(__P, (__v32qi)__A);
 }
 
@@ -2055,6 +2095,36 @@ _mm256_maskz_cvthf6_hf8(__mmask32 __U, __m256i __A) {
       __U, (__v32qi)_mm256_cvthf6_hf8(__A), (__v32qi)_mm256_setzero_si256());
 }
 
+/// Compose the \c size field of the immediate operand of \c VUNPACKB, which
+///    selects the size in bits of the packed source elements.
+///
+/// \headerfile <immintrin.h>
+///
+/// \param n
+///    The packed element size in bits, in the range [2, 7].
+/// \returns
+///    Bits [4:2] of the immediate operand.
+#define _MM_UNPACKB_SIZE(n) (((n)&0x7) << 2)
+
+/// Compose the \c start field of the immediate operand of \c VUNPACKB, which
+///    selects which block of packed elements is extracted from the source.
+///
+/// \headerfile <immintrin.h>
+///
+/// \param s
+///    The starting offset, in the range [0, 3]. The permitted values depend on
+///    the element size; only offsets that allow a full extraction are valid.
+/// \returns
+///    Bits [1:0] of the immediate operand.
+#define _MM_UNPACKB_START(s) (((s)&0x3) << 0)
+
+/// The \c sign \c ext field of the immediate operand of \c VUNPACKB,
+///    requesting that unpacked elements be sign-extended to 8 bits instead of
+///    zero-extended.
+///
+/// \headerfile <immintrin.h>
+#define _MM_UNPACKB_SEXT (1 << 5)
+
 /// Unpack bytes from \a A according to the immediate value \a imm, and store
 ///    the results in a 128-bit vector.
 ///
@@ -2068,7 +2138,7 @@ _mm256_maskz_cvthf6_hf8(__mmask32 __U, __m256i __A) {
 ///    An immediate value specifying the unpack operation.
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the unpacked values.
-#define _mm_unpackb_epi8(A, imm)                                               \
+#define _mm_unpack_epi8(A, imm)                                                \
   ((__m128i)__builtin_ia32_vunpackb128((__v16qi)(__m128i)(A), (int)(imm)))
 
 /// Unpack bytes from \a A according to the immediate value \a imm, and store
@@ -2088,9 +2158,9 @@ _mm256_maskz_cvthf6_hf8(__mmask32 __U, __m256i __A) {
 ///    An immediate value specifying the unpack operation.
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the unpacked values.
-#define _mm_mask_unpackb_epi8(W, U, A, imm)                                    \
+#define _mm_mask_unpack_epi8(W, U, A, imm)                                     \
   ((__m128i)__builtin_ia32_selectb_128((__mmask16)(U),                         \
-                                       (__v16qi)_mm_unpackb_epi8((A), (imm)),  \
+                                       (__v16qi)_mm_unpack_epi8((A), (imm)),   \
                                        (__v16qi)(__m128i)(W)))
 
 /// Unpack bytes from \a A according to the immediate value \a imm, and store
@@ -2108,9 +2178,9 @@ _mm256_maskz_cvthf6_hf8(__mmask32 __U, __m256i __A) {
 ///    An immediate value specifying the unpack operation.
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the unpacked values.
-#define _mm_maskz_unpackb_epi8(U, A, imm)                                      \
+#define _mm_maskz_unpack_epi8(U, A, imm)                                       \
   ((__m128i)__builtin_ia32_selectb_128((__mmask16)(U),                         \
-                                       (__v16qi)_mm_unpackb_epi8((A), (imm)),  \
+                                       (__v16qi)_mm_unpack_epi8((A), (imm)),   \
                                        (__v16qi)_mm_setzero_si128()))
 
 /// Unpack bytes from \a A according to the immediate value \a imm, and store
@@ -2126,7 +2196,7 @@ _mm256_maskz_cvthf6_hf8(__mmask32 __U, __m256i __A) {
 ///    An immediate value specifying the unpack operation.
 /// \returns
 ///    A 256-bit vector of [32 x i8] containing the unpacked values.
-#define _mm256_unpackb_epi8(A, imm)                                            \
+#define _mm256_unpack_epi8(A, imm)                                             \
   ((__m256i)__builtin_ia32_vunpackb256((__v32qi)(__m256i)(A), (int)(imm)))
 
 /// Unpack bytes from \a A according to the immediate value \a imm, and store
@@ -2146,9 +2216,9 @@ _mm256_maskz_cvthf6_hf8(__mmask32 __U, __m256i __A) {
 ///    An immediate value specifying the unpack operation.
 /// \returns
 ///    A 256-bit vector of [32 x i8] containing the unpacked values.
-#define _mm256_mask_unpackb_epi8(W, U, A, imm)                                 \
+#define _mm256_mask_unpack_epi8(W, U, A, imm)                                  \
   ((__m256i)__builtin_ia32_selectb_256(                                        \
-      (__mmask32)(U), (__v32qi)_mm256_unpackb_epi8((A), (imm)),                \
+      (__mmask32)(U), (__v32qi)_mm256_unpack_epi8((A), (imm)),                 \
       (__v32qi)(__m256i)(W)))
 
 /// Unpack bytes from \a A according to the immediate value \a imm, and store
@@ -2166,13 +2236,11 @@ _mm256_maskz_cvthf6_hf8(__mmask32 __U, __m256i __A) {
 ///    An immediate value specifying the unpack operation.
 /// \returns
 ///    A 256-bit vector of [32 x i8] containing the unpacked values.
-#define _mm256_maskz_unpackb_epi8(U, A, imm)                                   \
+#define _mm256_maskz_unpack_epi8(U, A, imm)                                    \
   ((__m256i)__builtin_ia32_selectb_256(                                        \
-      (__mmask32)(U), (__v32qi)_mm256_unpackb_epi8((A), (imm)),                \
+      (__mmask32)(U), (__v32qi)_mm256_unpack_epi8((A), (imm)),                 \
       (__v32qi)_mm256_setzero_si256()))
 
-/* VPMOVSSDB - Symmetric Signed Saturation DWord to Byte */
-
 /// Convert packed signed 32-bit integers in \a __A to packed 8-bit integers
 ///    with symmetric signed saturation (clamp to [-127, +127]), and store
 ///    the results in a 128-bit vector.
@@ -2187,7 +2255,7 @@ _mm256_maskz_cvthf6_hf8(__mmask32 __U, __m256i __A) {
 ///    A 128-bit vector of [16 x i8]. The lower 4 bytes contain the converted
 ///    values; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
-_mm_cvtss_epi32_epi8(__m128i __A) {
+_mm_cvtssepi32_epi8(__m128i __A) {
   return (__m128i)__builtin_ia32_vpmovssdb128_mask(
       (__v4si)__A, (__v16qi)_mm_setzero_si128(), (__mmask8)-1);
 }
@@ -2208,7 +2276,7 @@ _mm_cvtss_epi32_epi8(__m128i __A) {
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
-_mm_mask_cvtss_epi32_epi8(__m128i __W, __mmask8 __U, __m128i __A) {
+_mm_mask_cvtssepi32_epi8(__m128i __W, __mmask8 __U, __m128i __A) {
   return (__m128i)__builtin_ia32_vpmovssdb128_mask((__v4si)__A, (__v16qi)__W,
                                                    __U);
 }
@@ -2227,7 +2295,7 @@ _mm_mask_cvtss_epi32_epi8(__m128i __W, __mmask8 __U, __m128i __A) {
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
-_mm_maskz_cvtss_epi32_epi8(__mmask8 __U, __m128i __A) {
+_mm_maskz_cvtssepi32_epi8(__mmask8 __U, __m128i __A) {
   return (__m128i)__builtin_ia32_vpmovssdb128_mask(
       (__v4si)__A, (__v16qi)_mm_setzero_si128(), __U);
 }
@@ -2246,7 +2314,7 @@ _mm_maskz_cvtss_epi32_epi8(__mmask8 __U, __m128i __A) {
 ///    A 128-bit vector of [16 x i8]. The lower 8 bytes contain the converted
 ///    values; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
-_mm256_cvtss_epi32_epi8(__m256i __A) {
+_mm256_cvtssepi32_epi8(__m256i __A) {
   return (__m128i)__builtin_ia32_vpmovssdb256_mask(
       (__v8si)__A, (__v16qi)_mm_setzero_si128(), (__mmask8)-1);
 }
@@ -2267,7 +2335,7 @@ _mm256_cvtss_epi32_epi8(__m256i __A) {
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
-_mm256_mask_cvtss_epi32_epi8(__m128i __W, __mmask8 __U, __m256i __A) {
+_mm256_mask_cvtssepi32_epi8(__m128i __W, __mmask8 __U, __m256i __A) {
   return (__m128i)__builtin_ia32_vpmovssdb256_mask((__v8si)__A, (__v16qi)__W,
                                                    __U);
 }
@@ -2286,7 +2354,7 @@ _mm256_mask_cvtss_epi32_epi8(__m128i __W, __mmask8 __U, __m256i __A) {
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
-_mm256_maskz_cvtss_epi32_epi8(__mmask8 __U, __m256i __A) {
+_mm256_maskz_cvtssepi32_epi8(__mmask8 __U, __m256i __A) {
   return (__m128i)__builtin_ia32_vpmovssdb256_mask(
       (__v8si)__A, (__v16qi)_mm_setzero_si128(), __U);
 }
@@ -2306,7 +2374,7 @@ _mm256_maskz_cvtss_epi32_epi8(__mmask8 __U, __m256i __A) {
 /// \param __A
 ///    A 128-bit vector of [4 x i32].
 static __inline__ void __DEFAULT_FN_ATTRS128
-_mm_mask_cvtss_epi32_storeu_epi8(void *__P, __mmask8 __M, __m128i __A) {
+_mm_mask_cvtssepi32_storeu_epi8(void *__P, __mmask8 __M, __m128i __A) {
   __builtin_ia32_vpmovssdb128mem_mask((__v16qi *)__P, (__v4si)__A, __M);
 }
 
@@ -2325,7 +2393,7 @@ _mm_mask_cvtss_epi32_storeu_epi8(void *__P, __mmask8 __M, __m128i __A) {
 /// \param __A
 ///    A 256-bit vector of [8 x i32].
 static __inline__ void __DEFAULT_FN_ATTRS256
-_mm256_mask_cvtss_epi32_storeu_epi8(void *__P, __mmask8 __M, __m256i __A) {
+_mm256_mask_cvtssepi32_storeu_epi8(void *__P, __mmask8 __M, __m256i __A) {
   __builtin_ia32_vpmovssdb256mem_mask((__v16qi *)__P, (__v8si)__A, __M);
 }
 
diff --git a/clang/test/CodeGen/X86/avx10_2_v2aux-builtins-errors.c b/clang/test/CodeGen/X86/avx10_2_v2aux-builtins-errors.c
index fa620ab58cb3bb..ab9e227f07a80c 100644
--- a/clang/test/CodeGen/X86/avx10_2_v2aux-builtins-errors.c
+++ b/clang/test/CodeGen/X86/avx10_2_v2aux-builtins-errors.c
@@ -2,38 +2,38 @@
 
 #include <immintrin.h>
 
-__m128i test_mm_unpackb_epi8(__m128i __A) {
-  return _mm_unpackb_epi8(__A, 64); // expected-error {{argument value 64 is outside the valid range [0, 63]}}
+__m128i test_mm_unpack_epi8(__m128i __A) {
+  return _mm_unpack_epi8(__A, 64); // expected-error {{argument value 64 is outside the valid range [0, 63]}}
 }
 
-__m128i test_mm_mask_unpackb_epi8(__m128i __W, __mmask16 __U, __m128i __A) {
-  return _mm_mask_unpackb_epi8(__W, __U, __A, 64); // expected-error {{argument value 64 is outside the valid range [0, 63]}}
+__m128i test_mm_mask_unpack_epi8(__m128i __W, __mmask16 __U, __m128i __A) {
+  return _mm_mask_unpack_epi8(__W, __U, __A, 64); // expected-error {{argument value 64 is outside the valid range [0, 63]}}
 }
 
-__m128i test_mm_maskz_unpackb_epi8(__mmask16 __U, __m128i __A) {
-  return _mm_maskz_unpackb_epi8(__U, __A, 64); // expected-error {{argument value 64 is outside the valid range [0, 63]}}
+__m128i test_mm_maskz_unpack_epi8(__mmask16 __U, __m128i __A) {
+  return _mm_maskz_unpack_epi8(__U, __A, 64); // expected-error {{argument value 64 is outside the valid range [0, 63]}}
 }
 
-__m256i test_mm256_unpackb_epi8(__m256i __A) {
-  return _mm256_unpackb_epi8(__A, 64); // expected-error {{argument value 64 is outside the valid range [0, 63]}}
+__m256i test_mm256_unpack_epi8(__m256i __A) {
+  return _mm256_unpack_epi8(__A, 64); // expected-error {{argument value 64 is outside the valid range [0, 63]}}
 }
 
-__m256i test_mm256_mask_unpackb_epi8(__m256i __W, __mmask32 __U, __m256i __A) {
-  return _mm256_mask_unpackb_epi8(__W, __U, __A, 64); // expected-error {{argument value 64 is outside the valid range [0, 63]}}
+__m256i test_mm256_mask_unpack_epi8(__m256i __W, __mmask32 __U, __m256i __A) {
+  return _mm256_mask_unpack_epi8(__W, __U, __A, 64); // expected-error {{argument value 64 is outside the valid range [0, 63]}}
 }
 
-__m256i test_mm256_maskz_unpackb_epi8(__mmask32 __U, __m256i __A) {
-  return _mm256_maskz_unpackb_epi8(__U, __A, 64); // expected-error {{argument value 64 is outside the valid range [0, 63]}}
+__m256i test_mm256_maskz_unpack_epi8(__mmask32 __U, __m256i __A) {
+  return _mm256_maskz_unpack_epi8(__U, __A, 64); // expected-error {{argument value 64 is outside the valid range [0, 63]}}
 }
 
-__m512i test_mm512_unpackb_epi8(__m512i __A) {
-  return _mm512_unpackb_epi8(__A, 64); // expected-error {{argument value 64 is outside the valid range [0, 63]}}
+__m512i test_mm512_unpack_epi8(__m512i __A) {
+  return _mm512_unpack_epi8(__A, 64); // expected-error {{argument value 64 is outside the valid range [0, 63]}}
 }
 
-__m512i test_mm512_mask_unpackb_epi8(__m512i __W, __mmask64 __U, __m512i __A) {
-  return _mm512_mask_unpackb_epi8(__W, __U, __A, 64); // expected-error {{argument value 64 is outside the valid range [0, 63]}}
+__m512i test_mm512_mask_unpack_epi8(__m512i __W, __mmask64 __U, __m512i __A) {
+  return _mm512_mask_unpack_epi8(__W, __U, __A, 64); // expected-error {{argument value 64 is outside the valid range [0, 63]}}
 }
 
-__m512i test_mm512_maskz_unpackb_epi8(__mmask64 __U, __m512i __A) {
-  return _mm512_maskz_unpackb_epi8(__U, __A, 64); // expected-error {{argument value 64 is outside the valid range [0, 63]}}
+__m512i test_mm512_maskz_unpack_epi8(__mmask64 __U, __m512i __A) {
+  return _mm512_maskz_unpack_epi8(__U, __A, 64); // expected-error {{argument value 64 is outside the valid range [0, 63]}}
 }
diff --git a/clang/test/CodeGen/X86/avx10_2_v2aux-builtins.c b/clang/test/CodeGen/X86/avx10_2_v2aux-builtins.c
index 36f96174c497ec..ec1be83f941cf3 100644
--- a/clang/test/CodeGen/X86/avx10_2_v2aux-builtins.c
+++ b/clang/test/CodeGen/X86/avx10_2_v2aux-builtins.c
@@ -21,15 +21,14 @@ __m128i test_mm_cvtps_bf8(__m128 __A) {
 
 __m128i test_mm_mask_cvtps_bf8(__m128i __W, __mmask8 __U, __m128 __A) {
   // CHECK-LABEL: @test_mm_mask_cvtps_bf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8.128(<4 x float> %{{.*}})
-  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.128(<4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm_mask_cvtps_bf8(__W, __U, __A);
 }
 
 __m128i test_mm_maskz_cvtps_bf8(__mmask8 __U, __m128 __A) {
   // CHECK-LABEL: @test_mm_maskz_cvtps_bf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8.128(<4 x float> %{{.*}})
-  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
+  // CHECK: zeroinitializer
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.128(<4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm_maskz_cvtps_bf8(__U, __A);
 }
 
@@ -43,15 +42,14 @@ __m128i test_mm256_cvtps_bf8(__m256 __A) {
 
 __m128i test_mm256_mask_cvtps_bf8(__m128i __W, __mmask8 __U, __m256 __A) {
   // CHECK-LABEL: @test_mm256_mask_cvtps_bf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8.256(<8 x float> %{{.*}})
-  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.256(<8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm256_mask_cvtps_bf8(__W, __U, __A);
 }
 
 __m128i test_mm256_maskz_cvtps_bf8(__mmask8 __U, __m256 __A) {
   // CHECK-LABEL: @test_mm256_maskz_cvtps_bf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8.256(<8 x float> %{{.*}})
-  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
+  // CHECK: zeroinitializer
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.256(<8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm256_maskz_cvtps_bf8(__U, __A);
 }
 
@@ -73,6 +71,7 @@ __m128i test_mm512_mask_cvtps_bf8(__m128i __W, __mmask16 __U, __m512 __A) {
 __m128i test_mm512_maskz_cvtps_bf8(__mmask16 __U, __m512 __A) {
   // CHECK-LABEL: @test_mm512_maskz_cvtps_bf8(
   // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8.512(<16 x float> %{{.*}})
+  // CHECK: zeroinitializer
   // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm512_maskz_cvtps_bf8(__U, __A);
 }
@@ -87,15 +86,14 @@ __m128i test_mm_cvts_ps_bf8(__m128 __A) {
 
 __m128i test_mm_mask_cvts_ps_bf8(__m128i __W, __mmask8 __U, __m128 __A) {
   // CHECK-LABEL: @test_mm_mask_cvts_ps_bf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8s.128(<4 x float> %{{.*}})
-  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.128(<4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm_mask_cvts_ps_bf8(__W, __U, __A);
 }
 
 __m128i test_mm_maskz_cvts_ps_bf8(__mmask8 __U, __m128 __A) {
   // CHECK-LABEL: @test_mm_maskz_cvts_ps_bf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8s.128(<4 x float> %{{.*}})
-  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
+  // CHECK: zeroinitializer
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.128(<4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm_maskz_cvts_ps_bf8(__U, __A);
 }
 
@@ -109,15 +107,14 @@ __m128i test_mm256_cvts_ps_bf8(__m256 __A) {
 
 __m128i test_mm256_mask_cvts_ps_bf8(__m128i __W, __mmask8 __U, __m256 __A) {
   // CHECK-LABEL: @test_mm256_mask_cvts_ps_bf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8s.256(<8 x float> %{{.*}})
-  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.256(<8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm256_mask_cvts_ps_bf8(__W, __U, __A);
 }
 
 __m128i test_mm256_maskz_cvts_ps_bf8(__mmask8 __U, __m256 __A) {
   // CHECK-LABEL: @test_mm256_maskz_cvts_ps_bf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8s.256(<8 x float> %{{.*}})
-  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
+  // CHECK: zeroinitializer
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.256(<8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm256_maskz_cvts_ps_bf8(__U, __A);
 }
 
@@ -139,6 +136,7 @@ __m128i test_mm512_mask_cvts_ps_bf8(__m128i __W, __mmask16 __U, __m512 __A) {
 __m128i test_mm512_maskz_cvts_ps_bf8(__mmask16 __U, __m512 __A) {
   // CHECK-LABEL: @test_mm512_maskz_cvts_ps_bf8(
   // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8s.512(<16 x float> %{{.*}})
+  // CHECK: zeroinitializer
   // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm512_maskz_cvts_ps_bf8(__U, __A);
 }
@@ -153,15 +151,14 @@ __m128i test_mm_cvtps_hf8(__m128 __A) {
 
 __m128i test_mm_mask_cvtps_hf8(__m128i __W, __mmask8 __U, __m128 __A) {
   // CHECK-LABEL: @test_mm_mask_cvtps_hf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8.128(<4 x float> %{{.*}})
-  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.128(<4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm_mask_cvtps_hf8(__W, __U, __A);
 }
 
 __m128i test_mm_maskz_cvtps_hf8(__mmask8 __U, __m128 __A) {
   // CHECK-LABEL: @test_mm_maskz_cvtps_hf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8.128(<4 x float> %{{.*}})
-  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
+  // CHECK: zeroinitializer
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.128(<4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm_maskz_cvtps_hf8(__U, __A);
 }
 
@@ -175,15 +172,14 @@ __m128i test_mm256_cvtps_hf8(__m256 __A) {
 
 __m128i test_mm256_mask_cvtps_hf8(__m128i __W, __mmask8 __U, __m256 __A) {
   // CHECK-LABEL: @test_mm256_mask_cvtps_hf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8.256(<8 x float> %{{.*}})
-  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.256(<8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm256_mask_cvtps_hf8(__W, __U, __A);
 }
 
 __m128i test_mm256_maskz_cvtps_hf8(__mmask8 __U, __m256 __A) {
   // CHECK-LABEL: @test_mm256_maskz_cvtps_hf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8.256(<8 x float> %{{.*}})
-  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
+  // CHECK: zeroinitializer
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.256(<8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm256_maskz_cvtps_hf8(__U, __A);
 }
 
@@ -205,6 +201,7 @@ __m128i test_mm512_mask_cvtps_hf8(__m128i __W, __mmask16 __U, __m512 __A) {
 __m128i test_mm512_maskz_cvtps_hf8(__mmask16 __U, __m512 __A) {
   // CHECK-LABEL: @test_mm512_maskz_cvtps_hf8(
   // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8.512(<16 x float> %{{.*}})
+  // CHECK: zeroinitializer
   // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm512_maskz_cvtps_hf8(__U, __A);
 }
@@ -219,15 +216,14 @@ __m128i test_mm_cvts_ps_hf8(__m128 __A) {
 
 __m128i test_mm_mask_cvts_ps_hf8(__m128i __W, __mmask8 __U, __m128 __A) {
   // CHECK-LABEL: @test_mm_mask_cvts_ps_hf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8s.128(<4 x float> %{{.*}})
-  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.128(<4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm_mask_cvts_ps_hf8(__W, __U, __A);
 }
 
 __m128i test_mm_maskz_cvts_ps_hf8(__mmask8 __U, __m128 __A) {
   // CHECK-LABEL: @test_mm_maskz_cvts_ps_hf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8s.128(<4 x float> %{{.*}})
-  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
+  // CHECK: zeroinitializer
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.128(<4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm_maskz_cvts_ps_hf8(__U, __A);
 }
 
@@ -241,15 +237,14 @@ __m128i test_mm256_cvts_ps_hf8(__m256 __A) {
 
 __m128i test_mm256_mask_cvts_ps_hf8(__m128i __W, __mmask8 __U, __m256 __A) {
   // CHECK-LABEL: @test_mm256_mask_cvts_ps_hf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8s.256(<8 x float> %{{.*}})
-  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.256(<8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm256_mask_cvts_ps_hf8(__W, __U, __A);
 }
 
 __m128i test_mm256_maskz_cvts_ps_hf8(__mmask8 __U, __m256 __A) {
   // CHECK-LABEL: @test_mm256_maskz_cvts_ps_hf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8s.256(<8 x float> %{{.*}})
-  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
+  // CHECK: zeroinitializer
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.256(<8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm256_maskz_cvts_ps_hf8(__U, __A);
 }
 
@@ -271,6 +266,7 @@ __m128i test_mm512_mask_cvts_ps_hf8(__m128i __W, __mmask16 __U, __m512 __A) {
 __m128i test_mm512_maskz_cvts_ps_hf8(__mmask16 __U, __m512 __A) {
   // CHECK-LABEL: @test_mm512_maskz_cvts_ps_hf8(
   // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8s.512(<16 x float> %{{.*}})
+  // CHECK: zeroinitializer
   // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm512_maskz_cvts_ps_hf8(__U, __A);
 }
@@ -285,15 +281,14 @@ __m128i test_mm_cvtrops_hf8(__m128 __A) {
 
 __m128i test_mm_mask_cvtrops_hf8(__m128i __W, __mmask8 __U, __m128 __A) {
   // CHECK-LABEL: @test_mm_mask_cvtrops_hf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8.128(<4 x float> %{{.*}})
-  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.128(<4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm_mask_cvtrops_hf8(__W, __U, __A);
 }
 
 __m128i test_mm_maskz_cvtrops_hf8(__mmask8 __U, __m128 __A) {
   // CHECK-LABEL: @test_mm_maskz_cvtrops_hf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8.128(<4 x float> %{{.*}})
-  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
+  // CHECK: zeroinitializer
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.128(<4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm_maskz_cvtrops_hf8(__U, __A);
 }
 
@@ -307,15 +302,14 @@ __m128i test_mm256_cvtrops_hf8(__m256 __A) {
 
 __m128i test_mm256_mask_cvtrops_hf8(__m128i __W, __mmask8 __U, __m256 __A) {
   // CHECK-LABEL: @test_mm256_mask_cvtrops_hf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8.256(<8 x float> %{{.*}})
-  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.256(<8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm256_mask_cvtrops_hf8(__W, __U, __A);
 }
 
 __m128i test_mm256_maskz_cvtrops_hf8(__mmask8 __U, __m256 __A) {
   // CHECK-LABEL: @test_mm256_maskz_cvtrops_hf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8.256(<8 x float> %{{.*}})
-  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
+  // CHECK: zeroinitializer
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.256(<8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm256_maskz_cvtrops_hf8(__U, __A);
 }
 
@@ -337,6 +331,7 @@ __m128i test_mm512_mask_cvtrops_hf8(__m128i __W, __mmask16 __U, __m512 __A) {
 __m128i test_mm512_maskz_cvtrops_hf8(__mmask16 __U, __m512 __A) {
   // CHECK-LABEL: @test_mm512_maskz_cvtrops_hf8(
   // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8.512(<16 x float> %{{.*}})
+  // CHECK: zeroinitializer
   // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm512_maskz_cvtrops_hf8(__U, __A);
 }
@@ -351,15 +346,14 @@ __m128i test_mm_cvts_rops_hf8(__m128 __A) {
 
 __m128i test_mm_mask_cvts_rops_hf8(__m128i __W, __mmask8 __U, __m128 __A) {
   // CHECK-LABEL: @test_mm_mask_cvts_rops_hf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s.128(<4 x float> %{{.*}})
-  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.128(<4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm_mask_cvts_rops_hf8(__W, __U, __A);
 }
 
 __m128i test_mm_maskz_cvts_rops_hf8(__mmask8 __U, __m128 __A) {
   // CHECK-LABEL: @test_mm_maskz_cvts_rops_hf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s.128(<4 x float> %{{.*}})
-  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
+  // CHECK: zeroinitializer
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.128(<4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm_maskz_cvts_rops_hf8(__U, __A);
 }
 
@@ -373,15 +367,14 @@ __m128i test_mm256_cvts_rops_hf8(__m256 __A) {
 
 __m128i test_mm256_mask_cvts_rops_hf8(__m128i __W, __mmask8 __U, __m256 __A) {
   // CHECK-LABEL: @test_mm256_mask_cvts_rops_hf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s.256(<8 x float> %{{.*}})
-  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.256(<8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm256_mask_cvts_rops_hf8(__W, __U, __A);
 }
 
 __m128i test_mm256_maskz_cvts_rops_hf8(__mmask8 __U, __m256 __A) {
   // CHECK-LABEL: @test_mm256_maskz_cvts_rops_hf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s.256(<8 x float> %{{.*}})
-  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
+  // CHECK: zeroinitializer
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.256(<8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm256_maskz_cvts_rops_hf8(__U, __A);
 }
 
@@ -403,6 +396,7 @@ __m128i test_mm512_mask_cvts_rops_hf8(__m128i __W, __mmask16 __U, __m512 __A) {
 __m128i test_mm512_maskz_cvts_rops_hf8(__mmask16 __U, __m512 __A) {
   // CHECK-LABEL: @test_mm512_maskz_cvts_rops_hf8(
   // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s.512(<16 x float> %{{.*}})
+  // CHECK: zeroinitializer
   // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm512_maskz_cvts_rops_hf8(__U, __A);
 }
@@ -417,21 +411,20 @@ __m128i test_mm512_maskz_cvts_rops_hf8(__mmask16 __U, __m512 __A) {
 
 __m128i test_mm_cvtbiasps_bf8(__m128i __A, __m128 __B) {
   // CHECK-LABEL: @test_mm_cvtbiasps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8.128(<16 x i8> %{{.*}}, <4 x float> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8.128(<4 x i32> %{{.*}}, <4 x float> %{{.*}})
   return _mm_cvtbiasps_bf8(__A, __B);
 }
 
 __m128i test_mm_mask_cvtbiasps_bf8(__m128i __W, __mmask8 __U, __m128i __A, __m128 __B) {
   // CHECK-LABEL: @test_mm_mask_cvtbiasps_bf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8.128(<16 x i8> %{{.*}}, <4 x float> %{{.*}})
-  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.128(<4 x i32> %{{.*}}, <4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm_mask_cvtbiasps_bf8(__W, __U, __A, __B);
 }
 
 __m128i test_mm_maskz_cvtbiasps_bf8(__mmask8 __U, __m128i __A, __m128 __B) {
   // CHECK-LABEL: @test_mm_maskz_cvtbiasps_bf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8.128(<16 x i8> %{{.*}}, <4 x float> %{{.*}})
-  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
+  // CHECK: zeroinitializer
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.128(<4 x i32> %{{.*}}, <4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm_maskz_cvtbiasps_bf8(__U, __A, __B);
 }
 
@@ -439,21 +432,20 @@ __m128i test_mm_maskz_cvtbiasps_bf8(__mmask8 __U, __m128i __A, __m128 __B) {
 
 __m128i test_mm256_cvtbiasps_bf8(__m256i __A, __m256 __B) {
   // CHECK-LABEL: @test_mm256_cvtbiasps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8.256(<32 x i8> %{{.*}}, <8 x float> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8.256(<8 x i32> %{{.*}}, <8 x float> %{{.*}})
   return _mm256_cvtbiasps_bf8(__A, __B);
 }
 
 __m128i test_mm256_mask_cvtbiasps_bf8(__m128i __W, __mmask8 __U, __m256i __A, __m256 __B) {
   // CHECK-LABEL: @test_mm256_mask_cvtbiasps_bf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8.256(<32 x i8> %{{.*}}, <8 x float> %{{.*}})
-  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.256(<8 x i32> %{{.*}}, <8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm256_mask_cvtbiasps_bf8(__W, __U, __A, __B);
 }
 
 __m128i test_mm256_maskz_cvtbiasps_bf8(__mmask8 __U, __m256i __A, __m256 __B) {
   // CHECK-LABEL: @test_mm256_maskz_cvtbiasps_bf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8.256(<32 x i8> %{{.*}}, <8 x float> %{{.*}})
-  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
+  // CHECK: zeroinitializer
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.256(<8 x i32> %{{.*}}, <8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm256_maskz_cvtbiasps_bf8(__U, __A, __B);
 }
 
@@ -461,20 +453,21 @@ __m128i test_mm256_maskz_cvtbiasps_bf8(__mmask8 __U, __m256i __A, __m256 __B) {
 
 __m128i test_mm512_cvtbiasps_bf8(__m512i __A, __m512 __B) {
   // CHECK-LABEL: @test_mm512_cvtbiasps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8.512(<64 x i8> %{{.*}}, <16 x float> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8.512(<16 x i32> %{{.*}}, <16 x float> %{{.*}})
   return _mm512_cvtbiasps_bf8(__A, __B);
 }
 
 __m128i test_mm512_mask_cvtbiasps_bf8(__m128i __W, __mmask16 __U, __m512i __A, __m512 __B) {
   // CHECK-LABEL: @test_mm512_mask_cvtbiasps_bf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8.512(<64 x i8> %{{.*}}, <16 x float> %{{.*}})
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8.512(<16 x i32> %{{.*}}, <16 x float> %{{.*}})
   // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm512_mask_cvtbiasps_bf8(__W, __U, __A, __B);
 }
 
 __m128i test_mm512_maskz_cvtbiasps_bf8(__mmask16 __U, __m512i __A, __m512 __B) {
   // CHECK-LABEL: @test_mm512_maskz_cvtbiasps_bf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8.512(<64 x i8> %{{.*}}, <16 x float> %{{.*}})
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8.512(<16 x i32> %{{.*}}, <16 x float> %{{.*}})
+  // CHECK: zeroinitializer
   // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm512_maskz_cvtbiasps_bf8(__U, __A, __B);
 }
@@ -483,21 +476,20 @@ __m128i test_mm512_maskz_cvtbiasps_bf8(__mmask16 __U, __m512i __A, __m512 __B) {
 
 __m128i test_mm_cvts_biasps_bf8(__m128i __A, __m128 __B) {
   // CHECK-LABEL: @test_mm_cvts_biasps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s.128(<16 x i8> %{{.*}}, <4 x float> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s.128(<4 x i32> %{{.*}}, <4 x float> %{{.*}})
   return _mm_cvts_biasps_bf8(__A, __B);
 }
 
 __m128i test_mm_mask_cvts_biasps_bf8(__m128i __W, __mmask8 __U, __m128i __A, __m128 __B) {
   // CHECK-LABEL: @test_mm_mask_cvts_biasps_bf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s.128(<16 x i8> %{{.*}}, <4 x float> %{{.*}})
-  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.128(<4 x i32> %{{.*}}, <4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm_mask_cvts_biasps_bf8(__W, __U, __A, __B);
 }
 
 __m128i test_mm_maskz_cvts_biasps_bf8(__mmask8 __U, __m128i __A, __m128 __B) {
   // CHECK-LABEL: @test_mm_maskz_cvts_biasps_bf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s.128(<16 x i8> %{{.*}}, <4 x float> %{{.*}})
-  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
+  // CHECK: zeroinitializer
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.128(<4 x i32> %{{.*}}, <4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm_maskz_cvts_biasps_bf8(__U, __A, __B);
 }
 
@@ -505,21 +497,20 @@ __m128i test_mm_maskz_cvts_biasps_bf8(__mmask8 __U, __m128i __A, __m128 __B) {
 
 __m128i test_mm256_cvts_biasps_bf8(__m256i __A, __m256 __B) {
   // CHECK-LABEL: @test_mm256_cvts_biasps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s.256(<32 x i8> %{{.*}}, <8 x float> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s.256(<8 x i32> %{{.*}}, <8 x float> %{{.*}})
   return _mm256_cvts_biasps_bf8(__A, __B);
 }
 
 __m128i test_mm256_mask_cvts_biasps_bf8(__m128i __W, __mmask8 __U, __m256i __A, __m256 __B) {
   // CHECK-LABEL: @test_mm256_mask_cvts_biasps_bf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s.256(<32 x i8> %{{.*}}, <8 x float> %{{.*}})
-  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.256(<8 x i32> %{{.*}}, <8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm256_mask_cvts_biasps_bf8(__W, __U, __A, __B);
 }
 
 __m128i test_mm256_maskz_cvts_biasps_bf8(__mmask8 __U, __m256i __A, __m256 __B) {
   // CHECK-LABEL: @test_mm256_maskz_cvts_biasps_bf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s.256(<32 x i8> %{{.*}}, <8 x float> %{{.*}})
-  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
+  // CHECK: zeroinitializer
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.256(<8 x i32> %{{.*}}, <8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm256_maskz_cvts_biasps_bf8(__U, __A, __B);
 }
 
@@ -527,20 +518,21 @@ __m128i test_mm256_maskz_cvts_biasps_bf8(__mmask8 __U, __m256i __A, __m256 __B)
 
 __m128i test_mm512_cvts_biasps_bf8(__m512i __A, __m512 __B) {
   // CHECK-LABEL: @test_mm512_cvts_biasps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s.512(<64 x i8> %{{.*}}, <16 x float> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s.512(<16 x i32> %{{.*}}, <16 x float> %{{.*}})
   return _mm512_cvts_biasps_bf8(__A, __B);
 }
 
 __m128i test_mm512_mask_cvts_biasps_bf8(__m128i __W, __mmask16 __U, __m512i __A, __m512 __B) {
   // CHECK-LABEL: @test_mm512_mask_cvts_biasps_bf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s.512(<64 x i8> %{{.*}}, <16 x float> %{{.*}})
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s.512(<16 x i32> %{{.*}}, <16 x float> %{{.*}})
   // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm512_mask_cvts_biasps_bf8(__W, __U, __A, __B);
 }
 
 __m128i test_mm512_maskz_cvts_biasps_bf8(__mmask16 __U, __m512i __A, __m512 __B) {
   // CHECK-LABEL: @test_mm512_maskz_cvts_biasps_bf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s.512(<64 x i8> %{{.*}}, <16 x float> %{{.*}})
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s.512(<16 x i32> %{{.*}}, <16 x float> %{{.*}})
+  // CHECK: zeroinitializer
   // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm512_maskz_cvts_biasps_bf8(__U, __A, __B);
 }
@@ -549,21 +541,20 @@ __m128i test_mm512_maskz_cvts_biasps_bf8(__mmask16 __U, __m512i __A, __m512 __B)
 
 __m128i test_mm_cvtbiasps_hf8(__m128i __A, __m128 __B) {
   // CHECK-LABEL: @test_mm_cvtbiasps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8.128(<16 x i8> %{{.*}}, <4 x float> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8.128(<4 x i32> %{{.*}}, <4 x float> %{{.*}})
   return _mm_cvtbiasps_hf8(__A, __B);
 }
 
 __m128i test_mm_mask_cvtbiasps_hf8(__m128i __W, __mmask8 __U, __m128i __A, __m128 __B) {
   // CHECK-LABEL: @test_mm_mask_cvtbiasps_hf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8.128(<16 x i8> %{{.*}}, <4 x float> %{{.*}})
-  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.128(<4 x i32> %{{.*}}, <4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm_mask_cvtbiasps_hf8(__W, __U, __A, __B);
 }
 
 __m128i test_mm_maskz_cvtbiasps_hf8(__mmask8 __U, __m128i __A, __m128 __B) {
   // CHECK-LABEL: @test_mm_maskz_cvtbiasps_hf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8.128(<16 x i8> %{{.*}}, <4 x float> %{{.*}})
-  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
+  // CHECK: zeroinitializer
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.128(<4 x i32> %{{.*}}, <4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm_maskz_cvtbiasps_hf8(__U, __A, __B);
 }
 
@@ -571,21 +562,20 @@ __m128i test_mm_maskz_cvtbiasps_hf8(__mmask8 __U, __m128i __A, __m128 __B) {
 
 __m128i test_mm256_cvtbiasps_hf8(__m256i __A, __m256 __B) {
   // CHECK-LABEL: @test_mm256_cvtbiasps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8.256(<32 x i8> %{{.*}}, <8 x float> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8.256(<8 x i32> %{{.*}}, <8 x float> %{{.*}})
   return _mm256_cvtbiasps_hf8(__A, __B);
 }
 
 __m128i test_mm256_mask_cvtbiasps_hf8(__m128i __W, __mmask8 __U, __m256i __A, __m256 __B) {
   // CHECK-LABEL: @test_mm256_mask_cvtbiasps_hf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8.256(<32 x i8> %{{.*}}, <8 x float> %{{.*}})
-  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.256(<8 x i32> %{{.*}}, <8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm256_mask_cvtbiasps_hf8(__W, __U, __A, __B);
 }
 
 __m128i test_mm256_maskz_cvtbiasps_hf8(__mmask8 __U, __m256i __A, __m256 __B) {
   // CHECK-LABEL: @test_mm256_maskz_cvtbiasps_hf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8.256(<32 x i8> %{{.*}}, <8 x float> %{{.*}})
-  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
+  // CHECK: zeroinitializer
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.256(<8 x i32> %{{.*}}, <8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm256_maskz_cvtbiasps_hf8(__U, __A, __B);
 }
 
@@ -593,20 +583,21 @@ __m128i test_mm256_maskz_cvtbiasps_hf8(__mmask8 __U, __m256i __A, __m256 __B) {
 
 __m128i test_mm512_cvtbiasps_hf8(__m512i __A, __m512 __B) {
   // CHECK-LABEL: @test_mm512_cvtbiasps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8.512(<64 x i8> %{{.*}}, <16 x float> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8.512(<16 x i32> %{{.*}}, <16 x float> %{{.*}})
   return _mm512_cvtbiasps_hf8(__A, __B);
 }
 
 __m128i test_mm512_mask_cvtbiasps_hf8(__m128i __W, __mmask16 __U, __m512i __A, __m512 __B) {
   // CHECK-LABEL: @test_mm512_mask_cvtbiasps_hf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8.512(<64 x i8> %{{.*}}, <16 x float> %{{.*}})
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8.512(<16 x i32> %{{.*}}, <16 x float> %{{.*}})
   // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm512_mask_cvtbiasps_hf8(__W, __U, __A, __B);
 }
 
 __m128i test_mm512_maskz_cvtbiasps_hf8(__mmask16 __U, __m512i __A, __m512 __B) {
   // CHECK-LABEL: @test_mm512_maskz_cvtbiasps_hf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8.512(<64 x i8> %{{.*}}, <16 x float> %{{.*}})
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8.512(<16 x i32> %{{.*}}, <16 x float> %{{.*}})
+  // CHECK: zeroinitializer
   // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm512_maskz_cvtbiasps_hf8(__U, __A, __B);
 }
@@ -615,21 +606,20 @@ __m128i test_mm512_maskz_cvtbiasps_hf8(__mmask16 __U, __m512i __A, __m512 __B) {
 
 __m128i test_mm_cvts_biasps_hf8(__m128i __A, __m128 __B) {
   // CHECK-LABEL: @test_mm_cvts_biasps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s.128(<16 x i8> %{{.*}}, <4 x float> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s.128(<4 x i32> %{{.*}}, <4 x float> %{{.*}})
   return _mm_cvts_biasps_hf8(__A, __B);
 }
 
 __m128i test_mm_mask_cvts_biasps_hf8(__m128i __W, __mmask8 __U, __m128i __A, __m128 __B) {
   // CHECK-LABEL: @test_mm_mask_cvts_biasps_hf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s.128(<16 x i8> %{{.*}}, <4 x float> %{{.*}})
-  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.128(<4 x i32> %{{.*}}, <4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm_mask_cvts_biasps_hf8(__W, __U, __A, __B);
 }
 
 __m128i test_mm_maskz_cvts_biasps_hf8(__mmask8 __U, __m128i __A, __m128 __B) {
   // CHECK-LABEL: @test_mm_maskz_cvts_biasps_hf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s.128(<16 x i8> %{{.*}}, <4 x float> %{{.*}})
-  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
+  // CHECK: zeroinitializer
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.128(<4 x i32> %{{.*}}, <4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm_maskz_cvts_biasps_hf8(__U, __A, __B);
 }
 
@@ -637,21 +627,20 @@ __m128i test_mm_maskz_cvts_biasps_hf8(__mmask8 __U, __m128i __A, __m128 __B) {
 
 __m128i test_mm256_cvts_biasps_hf8(__m256i __A, __m256 __B) {
   // CHECK-LABEL: @test_mm256_cvts_biasps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s.256(<32 x i8> %{{.*}}, <8 x float> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s.256(<8 x i32> %{{.*}}, <8 x float> %{{.*}})
   return _mm256_cvts_biasps_hf8(__A, __B);
 }
 
 __m128i test_mm256_mask_cvts_biasps_hf8(__m128i __W, __mmask8 __U, __m256i __A, __m256 __B) {
   // CHECK-LABEL: @test_mm256_mask_cvts_biasps_hf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s.256(<32 x i8> %{{.*}}, <8 x float> %{{.*}})
-  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.256(<8 x i32> %{{.*}}, <8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm256_mask_cvts_biasps_hf8(__W, __U, __A, __B);
 }
 
 __m128i test_mm256_maskz_cvts_biasps_hf8(__mmask8 __U, __m256i __A, __m256 __B) {
   // CHECK-LABEL: @test_mm256_maskz_cvts_biasps_hf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s.256(<32 x i8> %{{.*}}, <8 x float> %{{.*}})
-  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
+  // CHECK: zeroinitializer
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.256(<8 x i32> %{{.*}}, <8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm256_maskz_cvts_biasps_hf8(__U, __A, __B);
 }
 
@@ -659,20 +648,21 @@ __m128i test_mm256_maskz_cvts_biasps_hf8(__mmask8 __U, __m256i __A, __m256 __B)
 
 __m128i test_mm512_cvts_biasps_hf8(__m512i __A, __m512 __B) {
   // CHECK-LABEL: @test_mm512_cvts_biasps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s.512(<64 x i8> %{{.*}}, <16 x float> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s.512(<16 x i32> %{{.*}}, <16 x float> %{{.*}})
   return _mm512_cvts_biasps_hf8(__A, __B);
 }
 
 __m128i test_mm512_mask_cvts_biasps_hf8(__m128i __W, __mmask16 __U, __m512i __A, __m512 __B) {
   // CHECK-LABEL: @test_mm512_mask_cvts_biasps_hf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s.512(<64 x i8> %{{.*}}, <16 x float> %{{.*}})
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s.512(<16 x i32> %{{.*}}, <16 x float> %{{.*}})
   // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm512_mask_cvts_biasps_hf8(__W, __U, __A, __B);
 }
 
 __m128i test_mm512_maskz_cvts_biasps_hf8(__mmask16 __U, __m512i __A, __m512 __B) {
   // CHECK-LABEL: @test_mm512_maskz_cvts_biasps_hf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s.512(<64 x i8> %{{.*}}, <16 x float> %{{.*}})
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s.512(<16 x i32> %{{.*}}, <16 x float> %{{.*}})
+  // CHECK: zeroinitializer
   // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm512_maskz_cvts_biasps_hf8(__U, __A, __B);
 }
@@ -700,6 +690,7 @@ __m128 test_mm_mask_cvtbf8_ps(__m128 __W, __mmask8 __U, __m128i __A) {
 __m128 test_mm_maskz_cvtbf8_ps(__mmask8 __U, __m128i __A) {
   // CHECK-LABEL: @test_mm_maskz_cvtbf8_ps(
   // CHECK: [[RES:%.*]] = call <4 x float> @llvm.x86.avx10.vcvtbf82ps.128(<16 x i8> %{{.*}})
+  // CHECK: zeroinitializer
   // CHECK: select <4 x i1> %{{.*}}, <4 x float> [[RES]], <4 x float> %{{.*}}
   return _mm_maskz_cvtbf8_ps(__U, __A);
 }
@@ -722,6 +713,7 @@ __m256 test_mm256_mask_cvtbf8_ps(__m256 __W, __mmask8 __U, __m128i __A) {
 __m256 test_mm256_maskz_cvtbf8_ps(__mmask8 __U, __m128i __A) {
   // CHECK-LABEL: @test_mm256_maskz_cvtbf8_ps(
   // CHECK: [[RES:%.*]] = call <8 x float> @llvm.x86.avx10.vcvtbf82ps.256(<16 x i8> %{{.*}})
+  // CHECK: zeroinitializer
   // CHECK: select <8 x i1> %{{.*}}, <8 x float> [[RES]], <8 x float> %{{.*}}
   return _mm256_maskz_cvtbf8_ps(__U, __A);
 }
@@ -744,6 +736,7 @@ __m512 test_mm512_mask_cvtbf8_ps(__m512 __W, __mmask16 __U, __m128i __A) {
 __m512 test_mm512_maskz_cvtbf8_ps(__mmask16 __U, __m128i __A) {
   // CHECK-LABEL: @test_mm512_maskz_cvtbf8_ps(
   // CHECK: [[RES:%.*]] = call <16 x float> @llvm.x86.avx10.vcvtbf82ps.512(<16 x i8> %{{.*}})
+  // CHECK: zeroinitializer
   // CHECK: select <16 x i1> %{{.*}}, <16 x float> [[RES]], <16 x float> %{{.*}}
   return _mm512_maskz_cvtbf8_ps(__U, __A);
 }
@@ -766,6 +759,7 @@ __m128 test_mm_mask_cvthf8_ps(__m128 __W, __mmask8 __U, __m128i __A) {
 __m128 test_mm_maskz_cvthf8_ps(__mmask8 __U, __m128i __A) {
   // CHECK-LABEL: @test_mm_maskz_cvthf8_ps(
   // CHECK: [[RES:%.*]] = call <4 x float> @llvm.x86.avx10.vcvthf82ps.128(<16 x i8> %{{.*}})
+  // CHECK: zeroinitializer
   // CHECK: select <4 x i1> %{{.*}}, <4 x float> [[RES]], <4 x float> %{{.*}}
   return _mm_maskz_cvthf8_ps(__U, __A);
 }
@@ -788,6 +782,7 @@ __m256 test_mm256_mask_cvthf8_ps(__m256 __W, __mmask8 __U, __m128i __A) {
 __m256 test_mm256_maskz_cvthf8_ps(__mmask8 __U, __m128i __A) {
   // CHECK-LABEL: @test_mm256_maskz_cvthf8_ps(
   // CHECK: [[RES:%.*]] = call <8 x float> @llvm.x86.avx10.vcvthf82ps.256(<16 x i8> %{{.*}})
+  // CHECK: zeroinitializer
   // CHECK: select <8 x i1> %{{.*}}, <8 x float> [[RES]], <8 x float> %{{.*}}
   return _mm256_maskz_cvthf8_ps(__U, __A);
 }
@@ -810,6 +805,7 @@ __m512 test_mm512_mask_cvthf8_ps(__m512 __W, __mmask16 __U, __m128i __A) {
 __m512 test_mm512_maskz_cvthf8_ps(__mmask16 __U, __m128i __A) {
   // CHECK-LABEL: @test_mm512_maskz_cvthf8_ps(
   // CHECK: [[RES:%.*]] = call <16 x float> @llvm.x86.avx10.vcvthf82ps.512(<16 x i8> %{{.*}})
+  // CHECK: zeroinitializer
   // CHECK: select <16 x i1> %{{.*}}, <16 x float> [[RES]], <16 x float> %{{.*}}
   return _mm512_maskz_cvthf8_ps(__U, __A);
 }
@@ -821,82 +817,82 @@ __m512 test_mm512_maskz_cvthf8_ps(__mmask16 __U, __m128i __A) {
 
 // VCVTBF82BF4S - register forms
 
-__m128i test_mm_cvtbf8_bf4s(__m128i __A) {
-  // CHECK-LABEL: @test_mm_cvtbf8_bf4s(
+__m128i test_mm_cvts_bf8_bf4(__m128i __A) {
+  // CHECK-LABEL: @test_mm_cvts_bf8_bf4(
   // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbf82bf4s.128(<16 x i8> %{{.*}})
-  return _mm_cvtbf8_bf4s(__A);
+  return _mm_cvts_bf8_bf4(__A);
 }
 
-__m128i test_mm256_cvtbf8_bf4s(__m256i __A) {
-  // CHECK-LABEL: @test_mm256_cvtbf8_bf4s(
+__m128i test_mm256_cvts_bf8_bf4(__m256i __A) {
+  // CHECK-LABEL: @test_mm256_cvts_bf8_bf4(
   // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbf82bf4s.256(<32 x i8> %{{.*}})
-  return _mm256_cvtbf8_bf4s(__A);
+  return _mm256_cvts_bf8_bf4(__A);
 }
 
-__m256i test_mm512_cvtbf8_bf4s(__m512i __A) {
-  // CHECK-LABEL: @test_mm512_cvtbf8_bf4s(
+__m256i test_mm512_cvts_bf8_bf4(__m512i __A) {
+  // CHECK-LABEL: @test_mm512_cvts_bf8_bf4(
   // CHECK: call <32 x i8> @llvm.x86.avx10.vcvtbf82bf4s.512(<64 x i8> %{{.*}})
-  return _mm512_cvtbf8_bf4s(__A);
+  return _mm512_cvts_bf8_bf4(__A);
 }
 
 // VCVTHF82BF4S - register forms
 
-__m128i test_mm_cvthf8_bf4s(__m128i __A) {
-  // CHECK-LABEL: @test_mm_cvthf8_bf4s(
+__m128i test_mm_cvts_hf8_bf4(__m128i __A) {
+  // CHECK-LABEL: @test_mm_cvts_hf8_bf4(
   // CHECK: call <16 x i8> @llvm.x86.avx10.vcvthf82bf4s.128(<16 x i8> %{{.*}})
-  return _mm_cvthf8_bf4s(__A);
+  return _mm_cvts_hf8_bf4(__A);
 }
 
-__m128i test_mm256_cvthf8_bf4s(__m256i __A) {
-  // CHECK-LABEL: @test_mm256_cvthf8_bf4s(
+__m128i test_mm256_cvts_hf8_bf4(__m256i __A) {
+  // CHECK-LABEL: @test_mm256_cvts_hf8_bf4(
   // CHECK: call <16 x i8> @llvm.x86.avx10.vcvthf82bf4s.256(<32 x i8> %{{.*}})
-  return _mm256_cvthf8_bf4s(__A);
+  return _mm256_cvts_hf8_bf4(__A);
 }
 
-__m256i test_mm512_cvthf8_bf4s(__m512i __A) {
-  // CHECK-LABEL: @test_mm512_cvthf8_bf4s(
+__m256i test_mm512_cvts_hf8_bf4(__m512i __A) {
+  // CHECK-LABEL: @test_mm512_cvts_hf8_bf4(
   // CHECK: call <32 x i8> @llvm.x86.avx10.vcvthf82bf4s.512(<64 x i8> %{{.*}})
-  return _mm512_cvthf8_bf4s(__A);
+  return _mm512_cvts_hf8_bf4(__A);
 }
 
 // VCVTBF82BF4S - memory store forms
 
-void test_mm_cvtbf8_bf4s_storeu(void *__P, __m128i __A) {
-  // CHECK-LABEL: @test_mm_cvtbf8_bf4s_storeu(
+void test_mm_cvts_bf8_bf4_storeu(void *__P, __m128i __A) {
+  // CHECK-LABEL: @test_mm_cvts_bf8_bf4_storeu(
   // CHECK: call void @llvm.x86.avx10.vcvtbf82bf4s.128.mem(ptr %{{.*}}, <16 x i8> %{{.*}})
-  _mm_cvtbf8_bf4s_storeu(__P, __A);
+  _mm_cvts_bf8_bf4_storeu(__P, __A);
 }
 
-void test_mm256_cvtbf8_bf4s_storeu(void *__P, __m256i __A) {
-  // CHECK-LABEL: @test_mm256_cvtbf8_bf4s_storeu(
+void test_mm256_cvts_bf8_bf4_storeu(void *__P, __m256i __A) {
+  // CHECK-LABEL: @test_mm256_cvts_bf8_bf4_storeu(
   // CHECK: call void @llvm.x86.avx10.vcvtbf82bf4s.256.mem(ptr %{{.*}}, <32 x i8> %{{.*}})
-  _mm256_cvtbf8_bf4s_storeu(__P, __A);
+  _mm256_cvts_bf8_bf4_storeu(__P, __A);
 }
 
-void test_mm512_cvtbf8_bf4s_storeu(void *__P, __m512i __A) {
-  // CHECK-LABEL: @test_mm512_cvtbf8_bf4s_storeu(
+void test_mm512_cvts_bf8_bf4_storeu(void *__P, __m512i __A) {
+  // CHECK-LABEL: @test_mm512_cvts_bf8_bf4_storeu(
   // CHECK: call void @llvm.x86.avx10.vcvtbf82bf4s.512.mem(ptr %{{.*}}, <64 x i8> %{{.*}})
-  _mm512_cvtbf8_bf4s_storeu(__P, __A);
+  _mm512_cvts_bf8_bf4_storeu(__P, __A);
 }
 
 // VCVTHF82BF4S - memory store forms
 
-void test_mm_cvthf8_bf4s_storeu(void *__P, __m128i __A) {
-  // CHECK-LABEL: @test_mm_cvthf8_bf4s_storeu(
+void test_mm_cvts_hf8_bf4_storeu(void *__P, __m128i __A) {
+  // CHECK-LABEL: @test_mm_cvts_hf8_bf4_storeu(
   // CHECK: call void @llvm.x86.avx10.vcvthf82bf4s.128.mem(ptr %{{.*}}, <16 x i8> %{{.*}})
-  _mm_cvthf8_bf4s_storeu(__P, __A);
+  _mm_cvts_hf8_bf4_storeu(__P, __A);
 }
 
-void test_mm256_cvthf8_bf4s_storeu(void *__P, __m256i __A) {
-  // CHECK-LABEL: @test_mm256_cvthf8_bf4s_storeu(
+void test_mm256_cvts_hf8_bf4_storeu(void *__P, __m256i __A) {
+  // CHECK-LABEL: @test_mm256_cvts_hf8_bf4_storeu(
   // CHECK: call void @llvm.x86.avx10.vcvthf82bf4s.256.mem(ptr %{{.*}}, <32 x i8> %{{.*}})
-  _mm256_cvthf8_bf4s_storeu(__P, __A);
+  _mm256_cvts_hf8_bf4_storeu(__P, __A);
 }
 
-void test_mm512_cvthf8_bf4s_storeu(void *__P, __m512i __A) {
-  // CHECK-LABEL: @test_mm512_cvthf8_bf4s_storeu(
+void test_mm512_cvts_hf8_bf4_storeu(void *__P, __m512i __A) {
+  // CHECK-LABEL: @test_mm512_cvts_hf8_bf4_storeu(
   // CHECK: call void @llvm.x86.avx10.vcvthf82bf4s.512.mem(ptr %{{.*}}, <64 x i8> %{{.*}})
-  _mm512_cvthf8_bf4s_storeu(__P, __A);
+  _mm512_cvts_hf8_bf4_storeu(__P, __A);
 }
 
 //
@@ -906,42 +902,42 @@ void test_mm512_cvthf8_bf4s_storeu(void *__P, __m512i __A) {
 
 // VCVTBF82BF6S
 
-__m128i test_mm_cvtbf8_bf6s(__m128i __A) {
-  // CHECK-LABEL: @test_mm_cvtbf8_bf6s(
+__m128i test_mm_cvts_bf8_bf6(__m128i __A) {
+  // CHECK-LABEL: @test_mm_cvts_bf8_bf6(
   // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbf82bf6s.128(<16 x i8> %{{.*}})
-  return _mm_cvtbf8_bf6s(__A);
+  return _mm_cvts_bf8_bf6(__A);
 }
 
-__m256i test_mm256_cvtbf8_bf6s(__m256i __A) {
-  // CHECK-LABEL: @test_mm256_cvtbf8_bf6s(
+__m256i test_mm256_cvts_bf8_bf6(__m256i __A) {
+  // CHECK-LABEL: @test_mm256_cvts_bf8_bf6(
   // CHECK: call <32 x i8> @llvm.x86.avx10.vcvtbf82bf6s.256(<32 x i8> %{{.*}})
-  return _mm256_cvtbf8_bf6s(__A);
+  return _mm256_cvts_bf8_bf6(__A);
 }
 
-__m512i test_mm512_cvtbf8_bf6s(__m512i __A) {
-  // CHECK-LABEL: @test_mm512_cvtbf8_bf6s(
+__m512i test_mm512_cvts_bf8_bf6(__m512i __A) {
+  // CHECK-LABEL: @test_mm512_cvts_bf8_bf6(
   // CHECK: call <64 x i8> @llvm.x86.avx10.vcvtbf82bf6s.512(<64 x i8> %{{.*}})
-  return _mm512_cvtbf8_bf6s(__A);
+  return _mm512_cvts_bf8_bf6(__A);
 }
 
 // VCVTHF82HF6S
 
-__m128i test_mm_cvthf8_hf6s(__m128i __A) {
-  // CHECK-LABEL: @test_mm_cvthf8_hf6s(
+__m128i test_mm_cvts_hf8_hf6(__m128i __A) {
+  // CHECK-LABEL: @test_mm_cvts_hf8_hf6(
   // CHECK: call <16 x i8> @llvm.x86.avx10.vcvthf82hf6s.128(<16 x i8> %{{.*}})
-  return _mm_cvthf8_hf6s(__A);
+  return _mm_cvts_hf8_hf6(__A);
 }
 
-__m256i test_mm256_cvthf8_hf6s(__m256i __A) {
-  // CHECK-LABEL: @test_mm256_cvthf8_hf6s(
+__m256i test_mm256_cvts_hf8_hf6(__m256i __A) {
+  // CHECK-LABEL: @test_mm256_cvts_hf8_hf6(
   // CHECK: call <32 x i8> @llvm.x86.avx10.vcvthf82hf6s.256(<32 x i8> %{{.*}})
-  return _mm256_cvthf8_hf6s(__A);
+  return _mm256_cvts_hf8_hf6(__A);
 }
 
-__m512i test_mm512_cvthf8_hf6s(__m512i __A) {
-  // CHECK-LABEL: @test_mm512_cvthf8_hf6s(
+__m512i test_mm512_cvts_hf8_hf6(__m512i __A) {
+  // CHECK-LABEL: @test_mm512_cvts_hf8_hf6(
   // CHECK: call <64 x i8> @llvm.x86.avx10.vcvthf82hf6s.512(<64 x i8> %{{.*}})
-  return _mm512_cvthf8_hf6s(__A);
+  return _mm512_cvts_hf8_hf6(__A);
 }
 
 //
@@ -967,6 +963,7 @@ __m128i test_mm_mask_cvtbf4_hf8(__m128i __W, __mmask16 __U, __m128i __A) {
 __m128i test_mm_maskz_cvtbf4_hf8(__mmask16 __U, __m128i __A) {
   // CHECK-LABEL: @test_mm_maskz_cvtbf4_hf8(
   // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbf42hf8.128(<16 x i8> %{{.*}})
+  // CHECK: zeroinitializer
   // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm_maskz_cvtbf4_hf8(__U, __A);
 }
@@ -989,6 +986,7 @@ __m256i test_mm256_mask_cvtbf4_hf8(__m256i __W, __mmask32 __U, __m128i __A) {
 __m256i test_mm256_maskz_cvtbf4_hf8(__mmask32 __U, __m128i __A) {
   // CHECK-LABEL: @test_mm256_maskz_cvtbf4_hf8(
   // CHECK: [[RES:%.*]] = call <32 x i8> @llvm.x86.avx10.vcvtbf42hf8.256(<16 x i8> %{{.*}})
+  // CHECK: zeroinitializer
   // CHECK: select <32 x i1> %{{.*}}, <32 x i8> [[RES]], <32 x i8> %{{.*}}
   return _mm256_maskz_cvtbf4_hf8(__U, __A);
 }
@@ -1011,6 +1009,7 @@ __m512i test_mm512_mask_cvtbf4_hf8(__m512i __W, __mmask64 __U, __m256i __A) {
 __m512i test_mm512_maskz_cvtbf4_hf8(__mmask64 __U, __m256i __A) {
   // CHECK-LABEL: @test_mm512_maskz_cvtbf4_hf8(
   // CHECK: [[RES:%.*]] = call <64 x i8> @llvm.x86.avx10.vcvtbf42hf8.512(<32 x i8> %{{.*}})
+  // CHECK: zeroinitializer
   // CHECK: select <64 x i1> %{{.*}}, <64 x i8> [[RES]], <64 x i8> %{{.*}}
   return _mm512_maskz_cvtbf4_hf8(__U, __A);
 }
@@ -1038,6 +1037,7 @@ __m128i test_mm_mask_cvtbf6_hf8(__m128i __W, __mmask16 __U, __m128i __A) {
 __m128i test_mm_maskz_cvtbf6_hf8(__mmask16 __U, __m128i __A) {
   // CHECK-LABEL: @test_mm_maskz_cvtbf6_hf8(
   // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbf62hf8.128(<16 x i8> %{{.*}})
+  // CHECK: zeroinitializer
   // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm_maskz_cvtbf6_hf8(__U, __A);
 }
@@ -1060,6 +1060,7 @@ __m256i test_mm256_mask_cvtbf6_hf8(__m256i __W, __mmask32 __U, __m256i __A) {
 __m256i test_mm256_maskz_cvtbf6_hf8(__mmask32 __U, __m256i __A) {
   // CHECK-LABEL: @test_mm256_maskz_cvtbf6_hf8(
   // CHECK: [[RES:%.*]] = call <32 x i8> @llvm.x86.avx10.vcvtbf62hf8.256(<32 x i8> %{{.*}})
+  // CHECK: zeroinitializer
   // CHECK: select <32 x i1> %{{.*}}, <32 x i8> [[RES]], <32 x i8> %{{.*}}
   return _mm256_maskz_cvtbf6_hf8(__U, __A);
 }
@@ -1082,6 +1083,7 @@ __m512i test_mm512_mask_cvtbf6_hf8(__m512i __W, __mmask64 __U, __m512i __A) {
 __m512i test_mm512_maskz_cvtbf6_hf8(__mmask64 __U, __m512i __A) {
   // CHECK-LABEL: @test_mm512_maskz_cvtbf6_hf8(
   // CHECK: [[RES:%.*]] = call <64 x i8> @llvm.x86.avx10.vcvtbf62hf8.512(<64 x i8> %{{.*}})
+  // CHECK: zeroinitializer
   // CHECK: select <64 x i1> %{{.*}}, <64 x i8> [[RES]], <64 x i8> %{{.*}}
   return _mm512_maskz_cvtbf6_hf8(__U, __A);
 }
@@ -1104,6 +1106,7 @@ __m128i test_mm_mask_cvthf6_hf8(__m128i __W, __mmask16 __U, __m128i __A) {
 __m128i test_mm_maskz_cvthf6_hf8(__mmask16 __U, __m128i __A) {
   // CHECK-LABEL: @test_mm_maskz_cvthf6_hf8(
   // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvthf62hf8.128(<16 x i8> %{{.*}})
+  // CHECK: zeroinitializer
   // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm_maskz_cvthf6_hf8(__U, __A);
 }
@@ -1126,6 +1129,7 @@ __m256i test_mm256_mask_cvthf6_hf8(__m256i __W, __mmask32 __U, __m256i __A) {
 __m256i test_mm256_maskz_cvthf6_hf8(__mmask32 __U, __m256i __A) {
   // CHECK-LABEL: @test_mm256_maskz_cvthf6_hf8(
   // CHECK: [[RES:%.*]] = call <32 x i8> @llvm.x86.avx10.vcvthf62hf8.256(<32 x i8> %{{.*}})
+  // CHECK: zeroinitializer
   // CHECK: select <32 x i1> %{{.*}}, <32 x i8> [[RES]], <32 x i8> %{{.*}}
   return _mm256_maskz_cvthf6_hf8(__U, __A);
 }
@@ -1148,6 +1152,7 @@ __m512i test_mm512_mask_cvthf6_hf8(__m512i __W, __mmask64 __U, __m512i __A) {
 __m512i test_mm512_maskz_cvthf6_hf8(__mmask64 __U, __m512i __A) {
   // CHECK-LABEL: @test_mm512_maskz_cvthf6_hf8(
   // CHECK: [[RES:%.*]] = call <64 x i8> @llvm.x86.avx10.vcvthf62hf8.512(<64 x i8> %{{.*}})
+  // CHECK: zeroinitializer
   // CHECK: select <64 x i1> %{{.*}}, <64 x i8> [[RES]], <64 x i8> %{{.*}}
   return _mm512_maskz_cvthf6_hf8(__U, __A);
 }
@@ -1159,68 +1164,92 @@ __m512i test_mm512_maskz_cvthf6_hf8(__mmask64 __U, __m512i __A) {
 
 // VUNPACKB - 128-bit
 
-__m128i test_mm_unpackb_epi8(__m128i __A) {
-  // CHECK-LABEL: @test_mm_unpackb_epi8(
+__m128i test_mm_unpack_epi8(__m128i __A) {
+  // CHECK-LABEL: @test_mm_unpack_epi8(
   // CHECK: call <16 x i8> @llvm.x86.avx10.vunpackb.128(<16 x i8> %{{.*}}, i8 1)
-  return _mm_unpackb_epi8(__A, 1);
+  return _mm_unpack_epi8(__A, 1);
 }
 
-__m128i test_mm_mask_unpackb_epi8(__m128i __W, __mmask16 __U, __m128i __A) {
-  // CHECK-LABEL: @test_mm_mask_unpackb_epi8(
+__m128i test_mm_mask_unpack_epi8(__m128i __W, __mmask16 __U, __m128i __A) {
+  // CHECK-LABEL: @test_mm_mask_unpack_epi8(
   // CHECK: call <16 x i8> @llvm.x86.avx10.vunpackb.128(<16 x i8> %{{.*}}, i8 1)
   // CHECK: select <16 x i1> %{{.*}}, <16 x i8> %{{.*}}, <16 x i8> %{{.*}}
-  return _mm_mask_unpackb_epi8(__W, __U, __A, 1);
+  return _mm_mask_unpack_epi8(__W, __U, __A, 1);
 }
 
-__m128i test_mm_maskz_unpackb_epi8(__mmask16 __U, __m128i __A) {
-  // CHECK-LABEL: @test_mm_maskz_unpackb_epi8(
+__m128i test_mm_maskz_unpack_epi8(__mmask16 __U, __m128i __A) {
+  // CHECK-LABEL: @test_mm_maskz_unpack_epi8(
   // CHECK: call <16 x i8> @llvm.x86.avx10.vunpackb.128(<16 x i8> %{{.*}}, i8 1)
+  // CHECK: zeroinitializer
   // CHECK: select <16 x i1> %{{.*}}, <16 x i8> %{{.*}}, <16 x i8> %{{.*}}
-  return _mm_maskz_unpackb_epi8(__U, __A, 1);
+  return _mm_maskz_unpack_epi8(__U, __A, 1);
 }
 
 // VUNPACKB - 256-bit
 
-__m256i test_mm256_unpackb_epi8(__m256i __A) {
-  // CHECK-LABEL: @test_mm256_unpackb_epi8(
+__m256i test_mm256_unpack_epi8(__m256i __A) {
+  // CHECK-LABEL: @test_mm256_unpack_epi8(
   // CHECK: call <32 x i8> @llvm.x86.avx10.vunpackb.256(<32 x i8> %{{.*}}, i8 2)
-  return _mm256_unpackb_epi8(__A, 2);
+  return _mm256_unpack_epi8(__A, 2);
 }
 
-__m256i test_mm256_mask_unpackb_epi8(__m256i __W, __mmask32 __U, __m256i __A) {
-  // CHECK-LABEL: @test_mm256_mask_unpackb_epi8(
+__m256i test_mm256_mask_unpack_epi8(__m256i __W, __mmask32 __U, __m256i __A) {
+  // CHECK-LABEL: @test_mm256_mask_unpack_epi8(
   // CHECK: call <32 x i8> @llvm.x86.avx10.vunpackb.256(<32 x i8> %{{.*}}, i8 2)
   // CHECK: select <32 x i1> %{{.*}}, <32 x i8> %{{.*}}, <32 x i8> %{{.*}}
-  return _mm256_mask_unpackb_epi8(__W, __U, __A, 2);
+  return _mm256_mask_unpack_epi8(__W, __U, __A, 2);
 }
 
-__m256i test_mm256_maskz_unpackb_epi8(__mmask32 __U, __m256i __A) {
-  // CHECK-LABEL: @test_mm256_maskz_unpackb_epi8(
+__m256i test_mm256_maskz_unpack_epi8(__mmask32 __U, __m256i __A) {
+  // CHECK-LABEL: @test_mm256_maskz_unpack_epi8(
   // CHECK: call <32 x i8> @llvm.x86.avx10.vunpackb.256(<32 x i8> %{{.*}}, i8 2)
+  // CHECK: zeroinitializer
   // CHECK: select <32 x i1> %{{.*}}, <32 x i8> %{{.*}}, <32 x i8> %{{.*}}
-  return _mm256_maskz_unpackb_epi8(__U, __A, 2);
+  return _mm256_maskz_unpack_epi8(__U, __A, 2);
 }
 
 // VUNPACKB - 512-bit
 
-__m512i test_mm512_unpackb_epi8(__m512i __A) {
-  // CHECK-LABEL: @test_mm512_unpackb_epi8(
+__m512i test_mm512_unpack_epi8(__m512i __A) {
+  // CHECK-LABEL: @test_mm512_unpack_epi8(
   // CHECK: call <64 x i8> @llvm.x86.avx10.vunpackb.512(<64 x i8> %{{.*}}, i8 3)
-  return _mm512_unpackb_epi8(__A, 3);
+  return _mm512_unpack_epi8(__A, 3);
 }
 
-__m512i test_mm512_mask_unpackb_epi8(__m512i __W, __mmask64 __U, __m512i __A) {
-  // CHECK-LABEL: @test_mm512_mask_unpackb_epi8(
+__m512i test_mm512_mask_unpack_epi8(__m512i __W, __mmask64 __U, __m512i __A) {
+  // CHECK-LABEL: @test_mm512_mask_unpack_epi8(
   // CHECK: call <64 x i8> @llvm.x86.avx10.vunpackb.512(<64 x i8> %{{.*}}, i8 3)
   // CHECK: select <64 x i1> %{{.*}}, <64 x i8> %{{.*}}, <64 x i8> %{{.*}}
-  return _mm512_mask_unpackb_epi8(__W, __U, __A, 3);
+  return _mm512_mask_unpack_epi8(__W, __U, __A, 3);
 }
 
-__m512i test_mm512_maskz_unpackb_epi8(__mmask64 __U, __m512i __A) {
-  // CHECK-LABEL: @test_mm512_maskz_unpackb_epi8(
+__m512i test_mm512_maskz_unpack_epi8(__mmask64 __U, __m512i __A) {
+  // CHECK-LABEL: @test_mm512_maskz_unpack_epi8(
   // CHECK: call <64 x i8> @llvm.x86.avx10.vunpackb.512(<64 x i8> %{{.*}}, i8 3)
+  // CHECK: zeroinitializer
   // CHECK: select <64 x i1> %{{.*}}, <64 x i8> %{{.*}}, <64 x i8> %{{.*}}
-  return _mm512_maskz_unpackb_epi8(__U, __A, 3);
+  return _mm512_maskz_unpack_epi8(__U, __A, 3);
+}
+
+// VUNPACKB - immediate composition macros
+
+__m512i test_mm512_unpack_epi8_compose(__m512i __A) {
+  // CHECK-LABEL: @test_mm512_unpack_epi8_compose(
+  // CHECK: call <64 x i8> @llvm.x86.avx10.vunpackb.512(<64 x i8> %{{.*}}, i8 45)
+  return _mm512_unpack_epi8(
+      __A, _MM_UNPACKB_SIZE(3) | _MM_UNPACKB_START(1) | _MM_UNPACKB_SEXT);
+}
+
+__m256i test_mm256_unpack_epi8_compose(__m256i __A) {
+  // CHECK-LABEL: @test_mm256_unpack_epi8_compose(
+  // CHECK: call <32 x i8> @llvm.x86.avx10.vunpackb.256(<32 x i8> %{{.*}}, i8 9)
+  return _mm256_unpack_epi8(__A, _MM_UNPACKB_SIZE(2) | _MM_UNPACKB_START(1));
+}
+
+__m128i test_mm_unpack_epi8_compose(__m128i __A) {
+  // CHECK-LABEL: @test_mm_unpack_epi8_compose(
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vunpackb.128(<16 x i8> %{{.*}}, i8 60)
+  return _mm_unpack_epi8(__A, _MM_UNPACKB_SIZE(7) | _MM_UNPACKB_SEXT);
 }
 
 //
@@ -1230,80 +1259,83 @@ __m512i test_mm512_maskz_unpackb_epi8(__mmask64 __U, __m512i __A) {
 
 // VPMOVSSDB - 128-bit
 
-__m128i test_mm_cvtss_epi32_epi8(__m128i __A) {
-  // CHECK-LABEL: @test_mm_cvtss_epi32_epi8(
+__m128i test_mm_cvtssepi32_epi8(__m128i __A) {
+  // CHECK-LABEL: @test_mm_cvtssepi32_epi8(
   // CHECK: call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.128(<4 x i32> %{{.*}}, <16 x i8> %{{.*}}, i8 -1)
-  return _mm_cvtss_epi32_epi8(__A);
+  return _mm_cvtssepi32_epi8(__A);
 }
 
-__m128i test_mm_mask_cvtss_epi32_epi8(__m128i __W, __mmask8 __U, __m128i __A) {
-  // CHECK-LABEL: @test_mm_mask_cvtss_epi32_epi8(
+__m128i test_mm_mask_cvtssepi32_epi8(__m128i __W, __mmask8 __U, __m128i __A) {
+  // CHECK-LABEL: @test_mm_mask_cvtssepi32_epi8(
   // CHECK: call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.128(<4 x i32> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
-  return _mm_mask_cvtss_epi32_epi8(__W, __U, __A);
+  return _mm_mask_cvtssepi32_epi8(__W, __U, __A);
 }
 
-__m128i test_mm_maskz_cvtss_epi32_epi8(__mmask8 __U, __m128i __A) {
-  // CHECK-LABEL: @test_mm_maskz_cvtss_epi32_epi8(
+__m128i test_mm_maskz_cvtssepi32_epi8(__mmask8 __U, __m128i __A) {
+  // CHECK-LABEL: @test_mm_maskz_cvtssepi32_epi8(
+  // CHECK: zeroinitializer
   // CHECK: call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.128(<4 x i32> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
-  return _mm_maskz_cvtss_epi32_epi8(__U, __A);
+  return _mm_maskz_cvtssepi32_epi8(__U, __A);
 }
 
 // VPMOVSSDB - 256-bit
 
-__m128i test_mm256_cvtss_epi32_epi8(__m256i __A) {
-  // CHECK-LABEL: @test_mm256_cvtss_epi32_epi8(
+__m128i test_mm256_cvtssepi32_epi8(__m256i __A) {
+  // CHECK-LABEL: @test_mm256_cvtssepi32_epi8(
   // CHECK: call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.256(<8 x i32> %{{.*}}, <16 x i8> %{{.*}}, i8 -1)
-  return _mm256_cvtss_epi32_epi8(__A);
+  return _mm256_cvtssepi32_epi8(__A);
 }
 
-__m128i test_mm256_mask_cvtss_epi32_epi8(__m128i __W, __mmask8 __U, __m256i __A) {
-  // CHECK-LABEL: @test_mm256_mask_cvtss_epi32_epi8(
+__m128i test_mm256_mask_cvtssepi32_epi8(__m128i __W, __mmask8 __U, __m256i __A) {
+  // CHECK-LABEL: @test_mm256_mask_cvtssepi32_epi8(
   // CHECK: call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.256(<8 x i32> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
-  return _mm256_mask_cvtss_epi32_epi8(__W, __U, __A);
+  return _mm256_mask_cvtssepi32_epi8(__W, __U, __A);
 }
 
-__m128i test_mm256_maskz_cvtss_epi32_epi8(__mmask8 __U, __m256i __A) {
-  // CHECK-LABEL: @test_mm256_maskz_cvtss_epi32_epi8(
+__m128i test_mm256_maskz_cvtssepi32_epi8(__mmask8 __U, __m256i __A) {
+  // CHECK-LABEL: @test_mm256_maskz_cvtssepi32_epi8(
+  // CHECK: zeroinitializer
   // CHECK: call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.256(<8 x i32> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
-  return _mm256_maskz_cvtss_epi32_epi8(__U, __A);
+  return _mm256_maskz_cvtssepi32_epi8(__U, __A);
 }
 
 // VPMOVSSDB - 512-bit
 
-__m128i test_mm512_cvtss_epi32_epi8(__m512i __A) {
-  // CHECK-LABEL: @test_mm512_cvtss_epi32_epi8(
+__m128i test_mm512_cvtssepi32_epi8(__m512i __A) {
+  // CHECK-LABEL: @test_mm512_cvtssepi32_epi8(
   // CHECK: call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.512(<16 x i32> %{{.*}}, <16 x i8> %{{.*}}, i16 -1)
-  return _mm512_cvtss_epi32_epi8(__A);
+  return _mm512_cvtssepi32_epi8(__A);
 }
 
-__m128i test_mm512_mask_cvtss_epi32_epi8(__m128i __W, __mmask16 __U, __m512i __A) {
-  // CHECK-LABEL: @test_mm512_mask_cvtss_epi32_epi8(
+__m128i test_mm512_mask_cvtssepi32_epi8(__m128i __W, __mmask16 __U, __m512i __A) {
+  // CHECK-LABEL: @test_mm512_mask_cvtssepi32_epi8(
   // CHECK: call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.512(<16 x i32> %{{.*}}, <16 x i8> %{{.*}}, i16 %{{.*}})
-  return _mm512_mask_cvtss_epi32_epi8(__W, __U, __A);
+  return _mm512_mask_cvtssepi32_epi8(__W, __U, __A);
 }
 
-__m128i test_mm512_maskz_cvtss_epi32_epi8(__mmask16 __U, __m512i __A) {
-  // CHECK-LABEL: @test_mm512_maskz_cvtss_epi32_epi8(
+__m128i test_mm512_maskz_cvtssepi32_epi8(__mmask16 __U, __m512i __A) {
+  // CHECK-LABEL: @test_mm512_maskz_cvtssepi32_epi8(
+  // CHECK: zeroinitializer
   // CHECK: call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.512(<16 x i32> %{{.*}}, <16 x i8> %{{.*}}, i16 %{{.*}})
-  return _mm512_maskz_cvtss_epi32_epi8(__U, __A);
+  return _mm512_maskz_cvtssepi32_epi8(__U, __A);
 }
 
 // VPMOVSSDB - memory store
 
-void test_mm_mask_cvtss_epi32_storeu_epi8(void *__P, __mmask8 __M, __m128i __A) {
-  // CHECK-LABEL: @test_mm_mask_cvtss_epi32_storeu_epi8(
+void test_mm_mask_cvtssepi32_storeu_epi8(void *__P, __mmask8 __M, __m128i __A) {
+  // CHECK-LABEL: @test_mm_mask_cvtssepi32_storeu_epi8(
   // CHECK: call void @llvm.x86.avx10.mask.pmovss.db.mem.128(ptr %{{.*}}, <4 x i32> %{{.*}}, i8 %{{.*}})
-  _mm_mask_cvtss_epi32_storeu_epi8(__P, __M, __A);
+  _mm_mask_cvtssepi32_storeu_epi8(__P, __M, __A);
 }
 
-void test_mm256_mask_cvtss_epi32_storeu_epi8(void *__P, __mmask8 __M, __m256i __A) {
-  // CHECK-LABEL: @test_mm256_mask_cvtss_epi32_storeu_epi8(
+void test_mm256_mask_cvtssepi32_storeu_epi8(void *__P, __mmask8 __M, __m256i __A) {
+  // CHECK-LABEL: @test_mm256_mask_cvtssepi32_storeu_epi8(
   // CHECK: call void @llvm.x86.avx10.mask.pmovss.db.mem.256(ptr %{{.*}}, <8 x i32> %{{.*}}, i8 %{{.*}})
-  _mm256_mask_cvtss_epi32_storeu_epi8(__P, __M, __A);
+  _mm256_mask_cvtssepi32_storeu_epi8(__P, __M, __A);
 }
 
-void test_mm512_mask_cvtss_epi32_storeu_epi8(void *__P, __mmask16 __M, __m512i __A) {
-  // CHECK-LABEL: @test_mm512_mask_cvtss_epi32_storeu_epi8(
+void test_mm512_mask_cvtssepi32_storeu_epi8(void *__P, __mmask16 __M, __m512i __A) {
+  // CHECK-LABEL: @test_mm512_mask_cvtssepi32_storeu_epi8(
   // CHECK: call void @llvm.x86.avx10.mask.pmovss.db.mem.512(ptr %{{.*}}, <16 x i32> %{{.*}}, i16 %{{.*}})
-  _mm512_mask_cvtss_epi32_storeu_epi8(__P, __M, __A);
+  _mm512_mask_cvtssepi32_storeu_epi8(__P, __M, __A);
 }
diff --git a/clang/test/Preprocessor/x86_target_features.c b/clang/test/Preprocessor/x86_target_features.c
index a8b114abda7cf2..90b785d3c0694e 100644
--- a/clang/test/Preprocessor/x86_target_features.c
+++ b/clang/test/Preprocessor/x86_target_features.c
@@ -692,11 +692,12 @@
 // RUN: %clang -target i686-unknown-linux-gnu -march=atom -mavx10.2 -x c -E -dM -o - %s | FileCheck  -check-prefixes=AVX10_1,AVX10_2 %s
 // RUN: %clang -target i686-unknown-linux-gnu -march=atom -mavx10.2 -mno-avx10.1 -x c -E -dM -o - %s | FileCheck  -check-prefixes=NO-AVX10_1,NO-AVX10_2 %s
 // RUN: %clang -target i686-unknown-linux-gnu -march=atom -mavx10v2aux -x c -E -dM -o - %s | FileCheck  -check-prefixes=AVX10_1,AVX10_V2_AUX %s
-// RUN: %clang -target i686-unknown-linux-gnu -march=atom -mavx10v2aux -x c -E -dM -o - %s | FileCheck  -check-prefix=NO-AVX10_2_ONLY %s
 // RUN: %clang -target i686-unknown-linux-gnu -march=atom -mno-avx10v2aux -x c -E -dM -o - %s | FileCheck  -check-prefix=NO-AVX10_V2_AUX %s
 // RUN: %clang -target i686-unknown-linux-gnu -march=atom -mavx10v2aux -mno-avx10.1 -x c -E -dM -o - %s | FileCheck  -check-prefix=NO-AVX10_V2_AUX %s
 // AVX10_1: #define __AVX10_1_512__ 1
 // AVX10_1: #define __AVX10_1__ 1
+// AVX10_V2_AUX-NOT: #define __AVX10_2_512__ 1
+// AVX10_V2_AUX-NOT: #define __AVX10_2__ 1
 // AVX10_2: #define __AVX10_2_512__ 1
 // AVX10_2: #define __AVX10_2__ 1
 // AVX10_V2_AUX: #define __AVX10_V2_AUX__ 1
@@ -706,8 +707,6 @@
 // NO-AVX10_1-NOT: __AVX10_2_512__
 // NO-AVX10_1-NOT: __AVX10_2__
 // NO-AVX10_V2_AUX-NOT: __AVX10_V2_AUX__
-// NO-AVX10_2_ONLY-NOT: __AVX10_2_512__
-// NO-AVX10_2_ONLY-NOT: __AVX10_2__
 // NO-AVX10_2: #define __AVX512F__ 1
 
 // RUN: %clang -target i686-unknown-linux-gnu -march=atom -musermsr -x c -E -dM -o - %s | FileCheck  -check-prefix=USERMSR %s
diff --git a/clang/test/Sema/builtins-x86.c b/clang/test/Sema/builtins-x86.c
index 0ae4e2cade1f3d..7d9cdce3d78948 100644
--- a/clang/test/Sema/builtins-x86.c
+++ b/clang/test/Sema/builtins-x86.c
@@ -192,4 +192,4 @@ unsigned char test_lwpins64(unsigned long long data2, unsigned long long data1,
 
 void test_lwpval64(unsigned long long data2, unsigned long long data1, unsigned int flags) {
   __builtin_ia32_lwpval64(data2, data1, flags); // expected-error {{argument to '__builtin_ia32_lwpval64' must be a constant integer}}
-}
\ No newline at end of file
+}
diff --git a/llvm/include/llvm/IR/IntrinsicsX86.td b/llvm/include/llvm/IR/IntrinsicsX86.td
index 8ccc2a50b4258e..fef29301704a51 100644
--- a/llvm/include/llvm/IR/IntrinsicsX86.td
+++ b/llvm/include/llvm/IR/IntrinsicsX86.td
@@ -7049,7 +7049,7 @@ def int_x86_avx10_mask_vcvtph2hf8s512 : ClangBuiltin<"__builtin_ia32_vcvtph2hf8s
 
 // AVX10 V2 AUX - Convert instructions
 let TargetPrefix = "x86" in {
-// Group A: PS(f32) -> i8 truncating conversions (quarter-size: output always v16i8)
+// Convert from FP32 to FP8
 
 // VCVTPS2BF8
 def int_x86_avx10_vcvtps2bf8_128 : ClangBuiltin<"__builtin_ia32_vcvtps2bf8_128">,
@@ -7058,6 +7058,16 @@ def int_x86_avx10_vcvtps2bf8_256 : ClangBuiltin<"__builtin_ia32_vcvtps2bf8_256">
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v8f32_ty], [IntrNoMem]>;
 def int_x86_avx10_vcvtps2bf8_512 : ClangBuiltin<"__builtin_ia32_vcvtps2bf8_512">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16f32_ty], [IntrNoMem]>;
+def int_x86_avx10_mask_vcvtps2bf8_128 :
+        ClangBuiltin<"__builtin_ia32_vcvtps2bf8_128_mask">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty],
+                              [llvm_v4f32_ty, llvm_v16i8_ty, llvm_i8_ty],
+                              [IntrNoMem]>;
+def int_x86_avx10_mask_vcvtps2bf8_256 :
+        ClangBuiltin<"__builtin_ia32_vcvtps2bf8_256_mask">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty],
+                              [llvm_v8f32_ty, llvm_v16i8_ty, llvm_i8_ty],
+                              [IntrNoMem]>;
 
 // VCVTPS2BF8S
 def int_x86_avx10_vcvtps2bf8s_128 : ClangBuiltin<"__builtin_ia32_vcvtps2bf8s_128">,
@@ -7066,6 +7076,16 @@ def int_x86_avx10_vcvtps2bf8s_256 : ClangBuiltin<"__builtin_ia32_vcvtps2bf8s_256
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v8f32_ty], [IntrNoMem]>;
 def int_x86_avx10_vcvtps2bf8s_512 : ClangBuiltin<"__builtin_ia32_vcvtps2bf8s_512">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16f32_ty], [IntrNoMem]>;
+def int_x86_avx10_mask_vcvtps2bf8s_128 :
+        ClangBuiltin<"__builtin_ia32_vcvtps2bf8s_128_mask">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty],
+                              [llvm_v4f32_ty, llvm_v16i8_ty, llvm_i8_ty],
+                              [IntrNoMem]>;
+def int_x86_avx10_mask_vcvtps2bf8s_256 :
+        ClangBuiltin<"__builtin_ia32_vcvtps2bf8s_256_mask">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty],
+                              [llvm_v8f32_ty, llvm_v16i8_ty, llvm_i8_ty],
+                              [IntrNoMem]>;
 
 // VCVTPS2HF8
 def int_x86_avx10_vcvtps2hf8_128 : ClangBuiltin<"__builtin_ia32_vcvtps2hf8_128">,
@@ -7074,6 +7094,16 @@ def int_x86_avx10_vcvtps2hf8_256 : ClangBuiltin<"__builtin_ia32_vcvtps2hf8_256">
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v8f32_ty], [IntrNoMem]>;
 def int_x86_avx10_vcvtps2hf8_512 : ClangBuiltin<"__builtin_ia32_vcvtps2hf8_512">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16f32_ty], [IntrNoMem]>;
+def int_x86_avx10_mask_vcvtps2hf8_128 :
+        ClangBuiltin<"__builtin_ia32_vcvtps2hf8_128_mask">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty],
+                              [llvm_v4f32_ty, llvm_v16i8_ty, llvm_i8_ty],
+                              [IntrNoMem]>;
+def int_x86_avx10_mask_vcvtps2hf8_256 :
+        ClangBuiltin<"__builtin_ia32_vcvtps2hf8_256_mask">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty],
+                              [llvm_v8f32_ty, llvm_v16i8_ty, llvm_i8_ty],
+                              [IntrNoMem]>;
 
 // VCVTPS2HF8S
 def int_x86_avx10_vcvtps2hf8s_128 : ClangBuiltin<"__builtin_ia32_vcvtps2hf8s_128">,
@@ -7082,6 +7112,16 @@ def int_x86_avx10_vcvtps2hf8s_256 : ClangBuiltin<"__builtin_ia32_vcvtps2hf8s_256
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v8f32_ty], [IntrNoMem]>;
 def int_x86_avx10_vcvtps2hf8s_512 : ClangBuiltin<"__builtin_ia32_vcvtps2hf8s_512">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16f32_ty], [IntrNoMem]>;
+def int_x86_avx10_mask_vcvtps2hf8s_128 :
+        ClangBuiltin<"__builtin_ia32_vcvtps2hf8s_128_mask">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty],
+                              [llvm_v4f32_ty, llvm_v16i8_ty, llvm_i8_ty],
+                              [IntrNoMem]>;
+def int_x86_avx10_mask_vcvtps2hf8s_256 :
+        ClangBuiltin<"__builtin_ia32_vcvtps2hf8s_256_mask">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty],
+                              [llvm_v8f32_ty, llvm_v16i8_ty, llvm_i8_ty],
+                              [IntrNoMem]>;
 
 // VCVTROPS2HF8
 def int_x86_avx10_vcvtrops2hf8_128 : ClangBuiltin<"__builtin_ia32_vcvtrops2hf8_128">,
@@ -7090,6 +7130,16 @@ def int_x86_avx10_vcvtrops2hf8_256 : ClangBuiltin<"__builtin_ia32_vcvtrops2hf8_2
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v8f32_ty], [IntrNoMem]>;
 def int_x86_avx10_vcvtrops2hf8_512 : ClangBuiltin<"__builtin_ia32_vcvtrops2hf8_512">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16f32_ty], [IntrNoMem]>;
+def int_x86_avx10_mask_vcvtrops2hf8_128 :
+        ClangBuiltin<"__builtin_ia32_vcvtrops2hf8_128_mask">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty],
+                              [llvm_v4f32_ty, llvm_v16i8_ty, llvm_i8_ty],
+                              [IntrNoMem]>;
+def int_x86_avx10_mask_vcvtrops2hf8_256 :
+        ClangBuiltin<"__builtin_ia32_vcvtrops2hf8_256_mask">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty],
+                              [llvm_v8f32_ty, llvm_v16i8_ty, llvm_i8_ty],
+                              [IntrNoMem]>;
 
 // VCVTROPS2HF8S
 def int_x86_avx10_vcvtrops2hf8s_128 : ClangBuiltin<"__builtin_ia32_vcvtrops2hf8s_128">,
@@ -7098,42 +7148,100 @@ def int_x86_avx10_vcvtrops2hf8s_256 : ClangBuiltin<"__builtin_ia32_vcvtrops2hf8s
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v8f32_ty], [IntrNoMem]>;
 def int_x86_avx10_vcvtrops2hf8s_512 : ClangBuiltin<"__builtin_ia32_vcvtrops2hf8s_512">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16f32_ty], [IntrNoMem]>;
+def int_x86_avx10_mask_vcvtrops2hf8s_128 :
+        ClangBuiltin<"__builtin_ia32_vcvtrops2hf8s_128_mask">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty],
+                              [llvm_v4f32_ty, llvm_v16i8_ty, llvm_i8_ty],
+                              [IntrNoMem]>;
+def int_x86_avx10_mask_vcvtrops2hf8s_256 :
+        ClangBuiltin<"__builtin_ia32_vcvtrops2hf8s_256_mask">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty],
+                              [llvm_v8f32_ty, llvm_v16i8_ty, llvm_i8_ty],
+                              [IntrNoMem]>;
 
-// Group B: Bias PS -> i8 conversions (2-operand: bias + f32 source -> i8 dest)
+// Convert from FP32 to FP8 with bias
 
 // VCVTBIASPS2BF8
 def int_x86_avx10_vcvtbiasps2bf8_128 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2bf8_128">,
-        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16i8_ty, llvm_v4f32_ty], [IntrNoMem]>;
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v4i32_ty, llvm_v4f32_ty], [IntrNoMem]>;
 def int_x86_avx10_vcvtbiasps2bf8_256 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2bf8_256">,
-        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v32i8_ty, llvm_v8f32_ty], [IntrNoMem]>;
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v8i32_ty, llvm_v8f32_ty], [IntrNoMem]>;
 def int_x86_avx10_vcvtbiasps2bf8_512 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2bf8_512">,
-        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v64i8_ty, llvm_v16f32_ty], [IntrNoMem]>;
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16i32_ty, llvm_v16f32_ty], [IntrNoMem]>;
+def int_x86_avx10_mask_vcvtbiasps2bf8_128 :
+        ClangBuiltin<"__builtin_ia32_vcvtbiasps2bf8_128_mask">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty],
+                              [llvm_v4i32_ty, llvm_v4f32_ty,
+                               llvm_v16i8_ty, llvm_i8_ty],
+                              [IntrNoMem]>;
+def int_x86_avx10_mask_vcvtbiasps2bf8_256 :
+        ClangBuiltin<"__builtin_ia32_vcvtbiasps2bf8_256_mask">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty],
+                              [llvm_v8i32_ty, llvm_v8f32_ty,
+                               llvm_v16i8_ty, llvm_i8_ty],
+                              [IntrNoMem]>;
 
 // VCVTBIASPS2BF8S
 def int_x86_avx10_vcvtbiasps2bf8s_128 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2bf8s_128">,
-        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16i8_ty, llvm_v4f32_ty], [IntrNoMem]>;
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v4i32_ty, llvm_v4f32_ty], [IntrNoMem]>;
 def int_x86_avx10_vcvtbiasps2bf8s_256 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2bf8s_256">,
-        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v32i8_ty, llvm_v8f32_ty], [IntrNoMem]>;
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v8i32_ty, llvm_v8f32_ty], [IntrNoMem]>;
 def int_x86_avx10_vcvtbiasps2bf8s_512 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2bf8s_512">,
-        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v64i8_ty, llvm_v16f32_ty], [IntrNoMem]>;
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16i32_ty, llvm_v16f32_ty], [IntrNoMem]>;
+def int_x86_avx10_mask_vcvtbiasps2bf8s_128 :
+        ClangBuiltin<"__builtin_ia32_vcvtbiasps2bf8s_128_mask">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty],
+                              [llvm_v4i32_ty, llvm_v4f32_ty,
+                               llvm_v16i8_ty, llvm_i8_ty],
+                              [IntrNoMem]>;
+def int_x86_avx10_mask_vcvtbiasps2bf8s_256 :
+        ClangBuiltin<"__builtin_ia32_vcvtbiasps2bf8s_256_mask">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty],
+                              [llvm_v8i32_ty, llvm_v8f32_ty,
+                               llvm_v16i8_ty, llvm_i8_ty],
+                              [IntrNoMem]>;
 
 // VCVTBIASPS2HF8
 def int_x86_avx10_vcvtbiasps2hf8_128 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2hf8_128">,
-        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16i8_ty, llvm_v4f32_ty], [IntrNoMem]>;
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v4i32_ty, llvm_v4f32_ty], [IntrNoMem]>;
 def int_x86_avx10_vcvtbiasps2hf8_256 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2hf8_256">,
-        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v32i8_ty, llvm_v8f32_ty], [IntrNoMem]>;
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v8i32_ty, llvm_v8f32_ty], [IntrNoMem]>;
 def int_x86_avx10_vcvtbiasps2hf8_512 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2hf8_512">,
-        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v64i8_ty, llvm_v16f32_ty], [IntrNoMem]>;
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16i32_ty, llvm_v16f32_ty], [IntrNoMem]>;
+def int_x86_avx10_mask_vcvtbiasps2hf8_128 :
+        ClangBuiltin<"__builtin_ia32_vcvtbiasps2hf8_128_mask">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty],
+                              [llvm_v4i32_ty, llvm_v4f32_ty,
+                               llvm_v16i8_ty, llvm_i8_ty],
+                              [IntrNoMem]>;
+def int_x86_avx10_mask_vcvtbiasps2hf8_256 :
+        ClangBuiltin<"__builtin_ia32_vcvtbiasps2hf8_256_mask">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty],
+                              [llvm_v8i32_ty, llvm_v8f32_ty,
+                               llvm_v16i8_ty, llvm_i8_ty],
+                              [IntrNoMem]>;
 
 // VCVTBIASPS2HF8S
 def int_x86_avx10_vcvtbiasps2hf8s_128 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2hf8s_128">,
-        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16i8_ty, llvm_v4f32_ty], [IntrNoMem]>;
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v4i32_ty, llvm_v4f32_ty], [IntrNoMem]>;
 def int_x86_avx10_vcvtbiasps2hf8s_256 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2hf8s_256">,
-        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v32i8_ty, llvm_v8f32_ty], [IntrNoMem]>;
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v8i32_ty, llvm_v8f32_ty], [IntrNoMem]>;
 def int_x86_avx10_vcvtbiasps2hf8s_512 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2hf8s_512">,
-        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v64i8_ty, llvm_v16f32_ty], [IntrNoMem]>;
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16i32_ty, llvm_v16f32_ty], [IntrNoMem]>;
+def int_x86_avx10_mask_vcvtbiasps2hf8s_128 :
+        ClangBuiltin<"__builtin_ia32_vcvtbiasps2hf8s_128_mask">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty],
+                              [llvm_v4i32_ty, llvm_v4f32_ty,
+                               llvm_v16i8_ty, llvm_i8_ty],
+                              [IntrNoMem]>;
+def int_x86_avx10_mask_vcvtbiasps2hf8s_256 :
+        ClangBuiltin<"__builtin_ia32_vcvtbiasps2hf8s_256_mask">,
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty],
+                              [llvm_v8i32_ty, llvm_v8f32_ty,
+                               llvm_v16i8_ty, llvm_i8_ty],
+                              [IntrNoMem]>;
 
-// Group C: 8bit -> PS expanding conversions
+// Convert from FP8 to FP32
 
 // VCVTBF82PS
 def int_x86_avx10_vcvtbf82ps_128 : ClangBuiltin<"__builtin_ia32_vcvtbf82ps_128">,
@@ -7151,7 +7259,7 @@ def int_x86_avx10_vcvthf82ps_256 : ClangBuiltin<"__builtin_ia32_vcvthf82ps_256">
 def int_x86_avx10_vcvthf82ps_512 : ClangBuiltin<"__builtin_ia32_vcvthf82ps_512">,
         DefaultAttrsIntrinsic<[llvm_v16f32_ty], [llvm_v16i8_ty], [IntrNoMem]>;
 
-// Group E: Same-size reg-only conversions (no masking)
+// Convert from FP8 to FP6
 
 // VCVTBF82BF6S
 def int_x86_avx10_vcvtbf82bf6s_128 : ClangBuiltin<"__builtin_ia32_vcvtbf82bf6s_128">,
@@ -7169,9 +7277,9 @@ def int_x86_avx10_vcvthf82hf6s_256 : ClangBuiltin<"__builtin_ia32_vcvthf82hf6s_2
 def int_x86_avx10_vcvthf82hf6s_512 : ClangBuiltin<"__builtin_ia32_vcvthf82hf6s_512">,
         DefaultAttrsIntrinsic<[llvm_v64i8_ty], [llvm_v64i8_ty], [IntrNoMem]>;
 
-// Group F: Expanding/same-size conversions with masking
+// Convert from FP4 to FP8 and from FP6 to FP8
 
-// VCVTBF42HF8: expanding (half-size input -> full output)
+// VCVTBF42HF8
 def int_x86_avx10_vcvtbf42hf8_128 : ClangBuiltin<"__builtin_ia32_vcvtbf42hf8_128">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16i8_ty], [IntrNoMem]>;
 def int_x86_avx10_vcvtbf42hf8_256 : ClangBuiltin<"__builtin_ia32_vcvtbf42hf8_256">,
@@ -7179,7 +7287,7 @@ def int_x86_avx10_vcvtbf42hf8_256 : ClangBuiltin<"__builtin_ia32_vcvtbf42hf8_256
 def int_x86_avx10_vcvtbf42hf8_512 : ClangBuiltin<"__builtin_ia32_vcvtbf42hf8_512">,
         DefaultAttrsIntrinsic<[llvm_v64i8_ty], [llvm_v32i8_ty], [IntrNoMem]>;
 
-// VCVTBF62HF8: same-size, reg-only
+// VCVTBF62HF8
 def int_x86_avx10_vcvtbf62hf8_128 : ClangBuiltin<"__builtin_ia32_vcvtbf62hf8_128">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16i8_ty], [IntrNoMem]>;
 def int_x86_avx10_vcvtbf62hf8_256 : ClangBuiltin<"__builtin_ia32_vcvtbf62hf8_256">,
@@ -7187,7 +7295,7 @@ def int_x86_avx10_vcvtbf62hf8_256 : ClangBuiltin<"__builtin_ia32_vcvtbf62hf8_256
 def int_x86_avx10_vcvtbf62hf8_512 : ClangBuiltin<"__builtin_ia32_vcvtbf62hf8_512">,
         DefaultAttrsIntrinsic<[llvm_v64i8_ty], [llvm_v64i8_ty], [IntrNoMem]>;
 
-// VCVTHF62HF8: same-size, reg-only
+// VCVTHF62HF8
 def int_x86_avx10_vcvthf62hf8_128 : ClangBuiltin<"__builtin_ia32_vcvthf62hf8_128">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16i8_ty], [IntrNoMem]>;
 def int_x86_avx10_vcvthf62hf8_256 : ClangBuiltin<"__builtin_ia32_vcvthf62hf8_256">,
@@ -7195,9 +7303,9 @@ def int_x86_avx10_vcvthf62hf8_256 : ClangBuiltin<"__builtin_ia32_vcvthf62hf8_256
 def int_x86_avx10_vcvthf62hf8_512 : ClangBuiltin<"__builtin_ia32_vcvthf62hf8_512">,
         DefaultAttrsIntrinsic<[llvm_v64i8_ty], [llvm_v64i8_ty], [IntrNoMem]>;
 
-// Group D: VCVTBF82BF4S / VCVTHF82BF4S - FP8 to FP4 truncating conversions
+// Convert from FP8 to FP4
 
-// VCVTBF82BF4S: FP8 E5M2 to FP4 E2M1 with saturation (truncating, output half size)
+// VCVTBF82BF4S
 def int_x86_avx10_vcvtbf82bf4s_128 : ClangBuiltin<"__builtin_ia32_vcvtbf82bf4s_128">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16i8_ty], [IntrNoMem]>;
 def int_x86_avx10_vcvtbf82bf4s_256 : ClangBuiltin<"__builtin_ia32_vcvtbf82bf4s_256">,
@@ -7205,7 +7313,7 @@ def int_x86_avx10_vcvtbf82bf4s_256 : ClangBuiltin<"__builtin_ia32_vcvtbf82bf4s_2
 def int_x86_avx10_vcvtbf82bf4s_512 : ClangBuiltin<"__builtin_ia32_vcvtbf82bf4s_512">,
         DefaultAttrsIntrinsic<[llvm_v32i8_ty], [llvm_v64i8_ty], [IntrNoMem]>;
 
-// VCVTHF82BF4S: FP8 E4M3 to FP4 E2M1 with saturation (truncating, output half size)
+// VCVTHF82BF4S
 def int_x86_avx10_vcvthf82bf4s_128 : ClangBuiltin<"__builtin_ia32_vcvthf82bf4s_128">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16i8_ty], [IntrNoMem]>;
 def int_x86_avx10_vcvthf82bf4s_256 : ClangBuiltin<"__builtin_ia32_vcvthf82bf4s_256">,
@@ -7213,7 +7321,7 @@ def int_x86_avx10_vcvthf82bf4s_256 : ClangBuiltin<"__builtin_ia32_vcvthf82bf4s_2
 def int_x86_avx10_vcvthf82bf4s_512 : ClangBuiltin<"__builtin_ia32_vcvthf82bf4s_512">,
         DefaultAttrsIntrinsic<[llvm_v32i8_ty], [llvm_v64i8_ty], [IntrNoMem]>;
 
-// VCVTBF82BF4S memory store: FP8 E5M2 to FP4 E2M1, store to memory
+// VCVTBF82BF4S memory store
 def int_x86_avx10_vcvtbf82bf4s_128_mem : ClangBuiltin<"__builtin_ia32_vcvtbf82bf4s_128_mem">,
         DefaultAttrsIntrinsic<[], [llvm_ptr_ty, llvm_v16i8_ty],
                               [IntrWriteMem, IntrArgMemOnly, NoCapture<ArgIndex<0>>]>;
@@ -7224,7 +7332,7 @@ def int_x86_avx10_vcvtbf82bf4s_512_mem : ClangBuiltin<"__builtin_ia32_vcvtbf82bf
         DefaultAttrsIntrinsic<[], [llvm_ptr_ty, llvm_v64i8_ty],
                               [IntrWriteMem, IntrArgMemOnly, NoCapture<ArgIndex<0>>]>;
 
-// VCVTHF82BF4S memory store: FP8 E4M3 to FP4 E2M1, store to memory
+// VCVTHF82BF4S memory store
 def int_x86_avx10_vcvthf82bf4s_128_mem : ClangBuiltin<"__builtin_ia32_vcvthf82bf4s_128_mem">,
         DefaultAttrsIntrinsic<[], [llvm_ptr_ty, llvm_v16i8_ty],
                               [IntrWriteMem, IntrArgMemOnly, NoCapture<ArgIndex<0>>]>;
@@ -7235,7 +7343,7 @@ def int_x86_avx10_vcvthf82bf4s_512_mem : ClangBuiltin<"__builtin_ia32_vcvthf82bf
         DefaultAttrsIntrinsic<[], [llvm_ptr_ty, llvm_v64i8_ty],
                               [IntrWriteMem, IntrArgMemOnly, NoCapture<ArgIndex<0>>]>;
 
-// Group H: VUNPACKB - Byte unpack with immediate
+// Unpack to Byte
 
 def int_x86_avx10_vunpackb_128 : ClangBuiltin<"__builtin_ia32_vunpackb128">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16i8_ty, llvm_i8_ty],
diff --git a/llvm/lib/Target/X86/X86InstrAVX10_V2_AUX.td b/llvm/lib/Target/X86/X86InstrAVX10_V2_AUX.td
index a0fa96157a7599..2b8851f49a1ffe 100644
--- a/llvm/lib/Target/X86/X86InstrAVX10_V2_AUX.td
+++ b/llvm/lib/Target/X86/X86InstrAVX10_V2_AUX.td
@@ -17,7 +17,6 @@
 
 // Convert from FP32 to FP8: truncating conversion, quarter-size output
 // Output is always xmm for all VL variants.
-// Adapts avx512_vcvt_fp for each VL level with explicit dest/src type mappings.
 multiclass avx10_v2aux_cvt_trunc_ps2i8<bits<8> opc, string OpcodeStr,
                                         SDPatternOperator OpNode,
                                         SDPatternOperator MaskOpNode> {
@@ -44,16 +43,16 @@ multiclass avx10_v2aux_cvt_trunc_ps2i8<bits<8> opc, string OpcodeStr,
   // InstAliases for x/y suffixes
   def : InstAlias<OpcodeStr#"x\t{$src, $dst|$dst, $src}",
                   (!cast<Instruction>(NAME # "Z128rr") VR128X:$dst,
-                  VR128X:$src), 0>;
+                   VR128X:$src), 0>;
   def : InstAlias<OpcodeStr#"x\t{$src, $dst|$dst, $src}",
                   (!cast<Instruction>(NAME # "Z128rm") VR128X:$dst,
-                  f128mem:$src), 0, "intel">;
+                   f128mem:$src), 0, "intel">;
   def : InstAlias<OpcodeStr#"y\t{$src, $dst|$dst, $src}",
                   (!cast<Instruction>(NAME # "Z256rr") VR128X:$dst,
-                  VR256X:$src), 0>;
+                   VR256X:$src), 0>;
   def : InstAlias<OpcodeStr#"y\t{$src, $dst|$dst, $src}",
                   (!cast<Instruction>(NAME # "Z256rm") VR128X:$dst,
-                  f256mem:$src), 0, "intel">;
+                   f256mem:$src), 0, "intel">;
 
   // Explicit patterns for Z256 (8 source elements, VK8WM mask)
   // Unmasked
@@ -61,33 +60,33 @@ multiclass avx10_v2aux_cvt_trunc_ps2i8<bits<8> opc, string OpcodeStr,
             (!cast<Instruction>(NAME # "Z256rr") VR256X:$src)>;
   // Masked (merge)
   def : Pat<(MaskOpNode (v8f32 VR256X:$src), (v16i8 VR128X:$src0),
-                         VK8WM:$mask),
+                        VK8WM:$mask),
             (!cast<Instruction>(NAME # "Z256rrk") VR128X:$src0, VK8WM:$mask,
                                 VR256X:$src)>;
   // Masked (zero)
   def : Pat<(MaskOpNode (v8f32 VR256X:$src), v16i8x_info.ImmAllZerosV,
-                         VK8WM:$mask),
+                        VK8WM:$mask),
             (!cast<Instruction>(NAME # "Z256rrkz") VK8WM:$mask,
                                 VR256X:$src)>;
   // Memory
   def : Pat<(v16i8 (OpNode (loadv8f32 addr:$src))),
             (!cast<Instruction>(NAME # "Z256rm") addr:$src)>;
   def : Pat<(MaskOpNode (loadv8f32 addr:$src), (v16i8 VR128X:$src0),
-                         VK8WM:$mask),
+                        VK8WM:$mask),
             (!cast<Instruction>(NAME # "Z256rmk") VR128X:$src0, VK8WM:$mask,
                                 addr:$src)>;
   def : Pat<(MaskOpNode (loadv8f32 addr:$src), v16i8x_info.ImmAllZerosV,
-                         VK8WM:$mask),
+                        VK8WM:$mask),
             (!cast<Instruction>(NAME # "Z256rmkz") VK8WM:$mask, addr:$src)>;
   // Broadcast
   def : Pat<(v16i8 (OpNode (v8f32 (X86VBroadcastld32 addr:$src)))),
             (!cast<Instruction>(NAME # "Z256rmb") addr:$src)>;
   def : Pat<(MaskOpNode (v8f32 (X86VBroadcastld32 addr:$src)),
-                          (v16i8 VR128X:$src0), VK8WM:$mask),
+                        (v16i8 VR128X:$src0), VK8WM:$mask),
             (!cast<Instruction>(NAME # "Z256rmbk") VR128X:$src0, VK8WM:$mask,
                                 addr:$src)>;
   def : Pat<(MaskOpNode (v8f32 (X86VBroadcastld32 addr:$src)),
-                          v16i8x_info.ImmAllZerosV, VK8WM:$mask),
+                        v16i8x_info.ImmAllZerosV, VK8WM:$mask),
             (!cast<Instruction>(NAME # "Z256rmbkz") VK8WM:$mask, addr:$src)>;
 
   // Explicit patterns for Z128 (4 source elements, VK4WM mask)
@@ -96,130 +95,129 @@ multiclass avx10_v2aux_cvt_trunc_ps2i8<bits<8> opc, string OpcodeStr,
             (!cast<Instruction>(NAME # "Z128rr") VR128X:$src)>;
   // Masked (merge)
   def : Pat<(MaskOpNode (v4f32 VR128X:$src), (v16i8 VR128X:$src0),
-                         VK4WM:$mask),
+                        VK4WM:$mask),
             (!cast<Instruction>(NAME # "Z128rrk") VR128X:$src0, VK4WM:$mask,
                                 VR128X:$src)>;
   // Masked (zero)
   def : Pat<(MaskOpNode (v4f32 VR128X:$src), v16i8x_info.ImmAllZerosV,
-                         VK4WM:$mask),
+                        VK4WM:$mask),
             (!cast<Instruction>(NAME # "Z128rrkz") VK4WM:$mask,
                                 VR128X:$src)>;
   // Memory
   def : Pat<(v16i8 (OpNode (loadv4f32 addr:$src))),
             (!cast<Instruction>(NAME # "Z128rm") addr:$src)>;
   def : Pat<(MaskOpNode (loadv4f32 addr:$src), (v16i8 VR128X:$src0),
-                         VK4WM:$mask),
+                        VK4WM:$mask),
             (!cast<Instruction>(NAME # "Z128rmk") VR128X:$src0, VK4WM:$mask,
                                 addr:$src)>;
   def : Pat<(MaskOpNode (loadv4f32 addr:$src), v16i8x_info.ImmAllZerosV,
-                         VK4WM:$mask),
+                        VK4WM:$mask),
             (!cast<Instruction>(NAME # "Z128rmkz") VK4WM:$mask, addr:$src)>;
   // Broadcast
   def : Pat<(v16i8 (OpNode (v4f32 (X86VBroadcastld32 addr:$src)))),
             (!cast<Instruction>(NAME # "Z128rmb") addr:$src)>;
   def : Pat<(MaskOpNode (v4f32 (X86VBroadcastld32 addr:$src)),
-                          (v16i8 VR128X:$src0), VK4WM:$mask),
+                        (v16i8 VR128X:$src0), VK4WM:$mask),
             (!cast<Instruction>(NAME # "Z128rmbk") VR128X:$src0, VK4WM:$mask,
                                 addr:$src)>;
   def : Pat<(MaskOpNode (v4f32 (X86VBroadcastld32 addr:$src)),
-                          v16i8x_info.ImmAllZerosV, VK4WM:$mask),
+                        v16i8x_info.ImmAllZerosV, VK4WM:$mask),
             (!cast<Instruction>(NAME # "Z128rmbkz") VK4WM:$mask, addr:$src)>;
 }
 
 // Convert from FP32 to FP8 with bias: 3-operand, quarter-size output
-// bias + f32 source -> i8 dest
-// Reuses avx10_convert_3op_packed from X86InstrAVX10.td for each VL variant.
+// bias dword per FP32 lane + f32 source -> i8 dest
 multiclass avx10_v2aux_cvt_3op_ps<bits<8> opc, string OpcodeStr,
                                   SDPatternOperator OpNode,
                                   SDPatternOperator MaskOpNode> {
-  // Z (512-bit): bias=v64i8(zmm), src=v16f32(zmm), dst=v16i8(xmm)
+  // Z (512-bit): bias=v16i32(zmm), src=v16f32(zmm), dst=v16i8(xmm)
   // Element counts match (16), so vselect_mask works directly.
   defm Z : avx10_convert_3op_packed<opc, OpcodeStr, v16i8x_info,
-             v64i8_info, v16f32_info, OpNode, OpNode, WriteCvtPH2PSZ>,
+             v16i32_info, v16f32_info, OpNode, OpNode, WriteCvtPH2PSZ>,
              EVEX_V512, EVEX_CD8<32, CD8VF>;
   // Z256/Z128: use null_frag because element count mismatch between
   // dest (v16i8) and source (v8f32/v4f32) prevents vselect_mask from
   // generating correct masked patterns. Explicit Pat patterns below.
   defm Z256 : avx10_convert_3op_packed<opc, OpcodeStr, v16i8x_info,
-                v32i8x_info, v8f32x_info,
+                v8i32x_info, v8f32x_info,
                 null_frag, null_frag, WriteCvtPH2PSZ>,
                 EVEX_V256, EVEX_CD8<32, CD8VF>;
   defm Z128 : avx10_convert_3op_packed<opc, OpcodeStr, v16i8x_info,
-                v16i8x_info, v4f32x_info,
+                v4i32x_info, v4f32x_info,
                 null_frag, null_frag, WriteCvtPH2PSZ>,
                 EVEX_V128, EVEX_CD8<32, CD8VF>;
 
   // Explicit patterns for Z256 (8 source elements, VK8WM mask)
-  def : Pat<(v16i8 (OpNode (v32i8 VR256X:$src1), (v8f32 VR256X:$src2))),
+  def : Pat<(v16i8 (OpNode (v8i32 VR256X:$src1), (v8f32 VR256X:$src2))),
             (!cast<Instruction>(NAME # "Z256rr") VR256X:$src1, VR256X:$src2)>;
-  def : Pat<(MaskOpNode (v32i8 VR256X:$src1), (v8f32 VR256X:$src2),
-                         (v16i8 VR128X:$src0), VK8WM:$mask),
+  def : Pat<(MaskOpNode (v8i32 VR256X:$src1), (v8f32 VR256X:$src2),
+                        (v16i8 VR128X:$src0), VK8WM:$mask),
             (!cast<Instruction>(NAME # "Z256rrk") VR128X:$src0, VK8WM:$mask,
                                 VR256X:$src1, VR256X:$src2)>;
-  def : Pat<(MaskOpNode (v32i8 VR256X:$src1), (v8f32 VR256X:$src2),
-                         v16i8x_info.ImmAllZerosV, VK8WM:$mask),
+  def : Pat<(MaskOpNode (v8i32 VR256X:$src1), (v8f32 VR256X:$src2),
+                        v16i8x_info.ImmAllZerosV, VK8WM:$mask),
             (!cast<Instruction>(NAME # "Z256rrkz") VK8WM:$mask,
                                 VR256X:$src1, VR256X:$src2)>;
   // Memory
-  def : Pat<(v16i8 (OpNode (v32i8 VR256X:$src1), (loadv8f32 addr:$src2))),
+  def : Pat<(v16i8 (OpNode (v8i32 VR256X:$src1), (loadv8f32 addr:$src2))),
             (!cast<Instruction>(NAME # "Z256rm") VR256X:$src1, addr:$src2)>;
-  def : Pat<(MaskOpNode (v32i8 VR256X:$src1), (loadv8f32 addr:$src2),
-                         (v16i8 VR128X:$src0), VK8WM:$mask),
+  def : Pat<(MaskOpNode (v8i32 VR256X:$src1), (loadv8f32 addr:$src2),
+                        (v16i8 VR128X:$src0), VK8WM:$mask),
             (!cast<Instruction>(NAME # "Z256rmk") VR128X:$src0, VK8WM:$mask,
                                 VR256X:$src1, addr:$src2)>;
-  def : Pat<(MaskOpNode (v32i8 VR256X:$src1), (loadv8f32 addr:$src2),
-                         v16i8x_info.ImmAllZerosV, VK8WM:$mask),
+  def : Pat<(MaskOpNode (v8i32 VR256X:$src1), (loadv8f32 addr:$src2),
+                        v16i8x_info.ImmAllZerosV, VK8WM:$mask),
             (!cast<Instruction>(NAME # "Z256rmkz") VK8WM:$mask,
                                 VR256X:$src1, addr:$src2)>;
   // Broadcast
-  def : Pat<(v16i8 (OpNode (v32i8 VR256X:$src1),
-                            (v8f32 (X86VBroadcastld32 addr:$src2)))),
+  def : Pat<(v16i8 (OpNode (v8i32 VR256X:$src1),
+                           (v8f32 (X86VBroadcastld32 addr:$src2)))),
             (!cast<Instruction>(NAME # "Z256rmb") VR256X:$src1, addr:$src2)>;
-  def : Pat<(MaskOpNode (v32i8 VR256X:$src1),
-                          (v8f32 (X86VBroadcastld32 addr:$src2)),
-                          (v16i8 VR128X:$src0), VK8WM:$mask),
+  def : Pat<(MaskOpNode (v8i32 VR256X:$src1),
+                        (v8f32 (X86VBroadcastld32 addr:$src2)),
+                        (v16i8 VR128X:$src0), VK8WM:$mask),
             (!cast<Instruction>(NAME # "Z256rmbk") VR128X:$src0, VK8WM:$mask,
                                 VR256X:$src1, addr:$src2)>;
-  def : Pat<(MaskOpNode (v32i8 VR256X:$src1),
-                          (v8f32 (X86VBroadcastld32 addr:$src2)),
-                          v16i8x_info.ImmAllZerosV, VK8WM:$mask),
+  def : Pat<(MaskOpNode (v8i32 VR256X:$src1),
+                        (v8f32 (X86VBroadcastld32 addr:$src2)),
+                        v16i8x_info.ImmAllZerosV, VK8WM:$mask),
             (!cast<Instruction>(NAME # "Z256rmbkz") VK8WM:$mask,
                                 VR256X:$src1, addr:$src2)>;
 
   // Explicit patterns for Z128 (4 source elements, VK4WM mask)
-  def : Pat<(v16i8 (OpNode (v16i8 VR128X:$src1), (v4f32 VR128X:$src2))),
+  def : Pat<(v16i8 (OpNode (v4i32 VR128X:$src1), (v4f32 VR128X:$src2))),
             (!cast<Instruction>(NAME # "Z128rr") VR128X:$src1, VR128X:$src2)>;
-  def : Pat<(MaskOpNode (v16i8 VR128X:$src1), (v4f32 VR128X:$src2),
-                         (v16i8 VR128X:$src0), VK4WM:$mask),
+  def : Pat<(MaskOpNode (v4i32 VR128X:$src1), (v4f32 VR128X:$src2),
+                        (v16i8 VR128X:$src0), VK4WM:$mask),
             (!cast<Instruction>(NAME # "Z128rrk") VR128X:$src0, VK4WM:$mask,
                                 VR128X:$src1, VR128X:$src2)>;
-  def : Pat<(MaskOpNode (v16i8 VR128X:$src1), (v4f32 VR128X:$src2),
-                         v16i8x_info.ImmAllZerosV, VK4WM:$mask),
+  def : Pat<(MaskOpNode (v4i32 VR128X:$src1), (v4f32 VR128X:$src2),
+                        v16i8x_info.ImmAllZerosV, VK4WM:$mask),
             (!cast<Instruction>(NAME # "Z128rrkz") VK4WM:$mask,
                                 VR128X:$src1, VR128X:$src2)>;
   // Memory
-  def : Pat<(v16i8 (OpNode (v16i8 VR128X:$src1), (loadv4f32 addr:$src2))),
+  def : Pat<(v16i8 (OpNode (v4i32 VR128X:$src1), (loadv4f32 addr:$src2))),
             (!cast<Instruction>(NAME # "Z128rm") VR128X:$src1, addr:$src2)>;
-  def : Pat<(MaskOpNode (v16i8 VR128X:$src1), (loadv4f32 addr:$src2),
-                         (v16i8 VR128X:$src0), VK4WM:$mask),
+  def : Pat<(MaskOpNode (v4i32 VR128X:$src1), (loadv4f32 addr:$src2),
+                        (v16i8 VR128X:$src0), VK4WM:$mask),
             (!cast<Instruction>(NAME # "Z128rmk") VR128X:$src0, VK4WM:$mask,
                                 VR128X:$src1, addr:$src2)>;
-  def : Pat<(MaskOpNode (v16i8 VR128X:$src1), (loadv4f32 addr:$src2),
-                         v16i8x_info.ImmAllZerosV, VK4WM:$mask),
+  def : Pat<(MaskOpNode (v4i32 VR128X:$src1), (loadv4f32 addr:$src2),
+                        v16i8x_info.ImmAllZerosV, VK4WM:$mask),
             (!cast<Instruction>(NAME # "Z128rmkz") VK4WM:$mask,
                                 VR128X:$src1, addr:$src2)>;
   // Broadcast
-  def : Pat<(v16i8 (OpNode (v16i8 VR128X:$src1),
-                            (v4f32 (X86VBroadcastld32 addr:$src2)))),
+  def : Pat<(v16i8 (OpNode (v4i32 VR128X:$src1),
+                           (v4f32 (X86VBroadcastld32 addr:$src2)))),
             (!cast<Instruction>(NAME # "Z128rmb") VR128X:$src1, addr:$src2)>;
-  def : Pat<(MaskOpNode (v16i8 VR128X:$src1),
-                          (v4f32 (X86VBroadcastld32 addr:$src2)),
-                          (v16i8 VR128X:$src0), VK4WM:$mask),
+  def : Pat<(MaskOpNode (v4i32 VR128X:$src1),
+                        (v4f32 (X86VBroadcastld32 addr:$src2)),
+                        (v16i8 VR128X:$src0), VK4WM:$mask),
             (!cast<Instruction>(NAME # "Z128rmbk") VR128X:$src0, VK4WM:$mask,
                                 VR128X:$src1, addr:$src2)>;
-  def : Pat<(MaskOpNode (v16i8 VR128X:$src1),
-                          (v4f32 (X86VBroadcastld32 addr:$src2)),
-                          v16i8x_info.ImmAllZerosV, VK4WM:$mask),
+  def : Pat<(MaskOpNode (v4i32 VR128X:$src1),
+                        (v4f32 (X86VBroadcastld32 addr:$src2)),
+                        v16i8x_info.ImmAllZerosV, VK4WM:$mask),
             (!cast<Instruction>(NAME # "Z128rmbkz") VK4WM:$mask,
                                 VR128X:$src1, addr:$src2)>;
 }
@@ -259,10 +257,10 @@ multiclass avx10_v2aux_cvt_trunc_base<bits<8> opc, string OpcodeStr,
   }
 
   def : Pat<(_dest.VT (!cast<Intrinsic>("int_x86_avx10_"#OpcodeStr#"_"#_src.Size)
-                        (_src.VT _src.RC:$src))),
+                       (_src.VT _src.RC:$src))),
             (!cast<Instruction>(NAME # "rr") _src.RC:$src)>;
   def : Pat<(!cast<Intrinsic>("int_x86_avx10_"#OpcodeStr#"_"#_src.Size#"_mem")
-                addr:$dst, (_src.VT _src.RC:$src)),
+             addr:$dst, (_src.VT _src.RC:$src)),
             (!cast<Instruction>(NAME # "mr") addr:$dst, _src.RC:$src)>;
 }
 
@@ -283,7 +281,8 @@ multiclass avx10_v2aux_cvt_expand_masked_base<bits<8> opc, string OpcodeStr,
                                               X86VectorVTInfo _,
                                               X86VectorVTInfo _src,
                                               X86MemOperand x86memop,
-                                              X86FoldableSchedWrite sched> {
+                                              X86FoldableSchedWrite sched,
+                                              dag ld_dag = (load addr:$src)> {
   let ExeDomain = _.ExeDomain in {
     defm rr : AVX512_maskable<opc, MRMSrcReg, _, (outs _.RC:$dst),
                               (ins _src.RC:$src),
@@ -296,7 +295,7 @@ multiclass avx10_v2aux_cvt_expand_masked_base<bits<8> opc, string OpcodeStr,
                                 (ins x86memop:$src),
                                 OpcodeStr, "$src", "$src",
                                 (_.VT (!cast<Intrinsic>("int_x86_avx10_"#OpcodeStr#"_"#_.Size)
-                                       (_src.VT (load addr:$src))))>,
+                                       (_src.VT ld_dag)))>,
                                 Sched<[sched.Folded]>, EVEX, EVEX_CD8<8, CD8VH>;
   }
 }
@@ -308,9 +307,12 @@ multiclass avx10_v2aux_cvt_expand_masked_b<bits<8> opc, string OpcodeStr> {
   defm Z256 : avx10_v2aux_cvt_expand_masked_base<opc, OpcodeStr, v32i8x_info,
                                                  v16i8x_info, f128mem,
                                                  WriteCvtPH2PSY>, EVEX_V256;
+  // Z128 reads only 8 bytes, so match a vzload rather than a full 16-byte load.
   defm Z128 : avx10_v2aux_cvt_expand_masked_base<opc, OpcodeStr, v16i8x_info,
                                                  v16i8x_info, f64mem,
-                                                 WriteCvtPH2PS>, EVEX_V128;
+                                                 WriteCvtPH2PS,
+                                                 (bitconvert (v2i64 (X86vzload64 addr:$src)))>,
+                                                 EVEX_V128;
 }
 
 // Convert from FP6 to FP8: widening conversion with masking (reg-only)
@@ -484,6 +486,11 @@ let Predicates = [HasAVX10_V2_AUX] in {
 let Predicates = [HasAVX10_V2_AUX] in {
   defm VCVTBF42HF8 : avx10_v2aux_cvt_expand_masked_b<0x37, "vcvtbf42hf8">,
                      T_MAP5, PS;
+
+  // Pattern match vcvtbf42hf8 of a scalar i64 load.
+  def : Pat<(v16i8 (int_x86_avx10_vcvtbf42hf8_128 (v16i8 (bitconvert
+              (v2i64 (scalar_to_vector (loadi64 addr:$src))))))),
+            (VCVTBF42HF8Z128rm addr:$src)>;
 }
 let Predicates = [HasAVX10_V2_AUX] in {
   // VCVTBF62HF8: widening (6-bit to 8-bit) with masking, reg-only
@@ -508,10 +515,10 @@ let Predicates = [HasAVX10_V2_AUX] in {
   // Explicit patterns for 512-bit VMTRUNCSS (intrinsic lowering produces this
   // SDNode directly, but avx512_trunc_db generates vselect_mask patterns for Z)
   def : Pat<(v16i8 (X86vmtruncss (v16i32 VR512:$src), (v16i8 VR128X:$src0),
-                                  VK16WM:$mask)),
+                                 VK16WM:$mask)),
             (VPMOVSSDBZrrk VR128X:$src0, VK16WM:$mask, VR512:$src)>;
   def : Pat<(v16i8 (X86vmtruncss (v16i32 VR512:$src), v16i8x_info.ImmAllZerosV,
-                                  VK16WM:$mask)),
+                                 VK16WM:$mask)),
             (VPMOVSSDBZrrkz VK16WM:$mask, VR512:$src)>;
 }
 
diff --git a/llvm/lib/Target/X86/X86InstrFragmentsSIMD.td b/llvm/lib/Target/X86/X86InstrFragmentsSIMD.td
index 7b7307b55c2589..bc6a159d1158e3 100644
--- a/llvm/lib/Target/X86/X86InstrFragmentsSIMD.td
+++ b/llvm/lib/Target/X86/X86InstrFragmentsSIMD.td
@@ -1214,12 +1214,12 @@ def SDTAVX10CONVERT_I8F32_MASK : SDTypeProfile<1, 3, [
   SDTCisSameNumEltsAs<1, 3>
 ]>;
 
-def SDTAVX10CONVERT_2I8F32 : SDTypeProfile<1, 2, [
-  SDTCVecEltisVT<0, i8>, SDTCVecEltisVT<1, i8>, SDTCVecEltisVT<2, f32>
+def SDTAVX10CONVERT_I8I32F32 : SDTypeProfile<1, 2, [
+  SDTCVecEltisVT<0, i8>, SDTCVecEltisVT<1, i32>, SDTCVecEltisVT<2, f32>
 ]>;
 
-def SDTAVX10CONVERT_2I8F32_MASK : SDTypeProfile<1, 4, [
-  SDTCVecEltisVT<0, i8>, SDTCVecEltisVT<1, i8>,
+def SDTAVX10CONVERT_I8I32F32_MASK : SDTypeProfile<1, 4, [
+  SDTCVecEltisVT<0, i8>, SDTCVecEltisVT<1, i32>,
   SDTCVecEltisVT<2, f32>, SDTCisSameAs<0, 3>, SDTCVecEltisVT<4, i1>,
   SDTCisSameNumEltsAs<2, 4>
 ]>;
@@ -1246,15 +1246,15 @@ def X86vmcvtrops2hf8  : SDNode<"X86ISD::VMCVTROPS2HF8",  SDTAVX10CONVERT_I8F32_M
 def X86vmcvtrops2hf8s : SDNode<"X86ISD::VMCVTROPS2HF8S", SDTAVX10CONVERT_I8F32_MASK>;
 
 // Group B: Bias PS->8bit (3 operand)
-def X86vcvtbiasps2bf8  : SDNode<"X86ISD::VCVTBIASPS2BF8",  SDTAVX10CONVERT_2I8F32>;
-def X86vcvtbiasps2bf8s : SDNode<"X86ISD::VCVTBIASPS2BF8S", SDTAVX10CONVERT_2I8F32>;
-def X86vcvtbiasps2hf8  : SDNode<"X86ISD::VCVTBIASPS2HF8",  SDTAVX10CONVERT_2I8F32>;
-def X86vcvtbiasps2hf8s : SDNode<"X86ISD::VCVTBIASPS2HF8S", SDTAVX10CONVERT_2I8F32>;
-
-def X86vmcvtbiasps2bf8  : SDNode<"X86ISD::VMCVTBIASPS2BF8",  SDTAVX10CONVERT_2I8F32_MASK>;
-def X86vmcvtbiasps2bf8s : SDNode<"X86ISD::VMCVTBIASPS2BF8S", SDTAVX10CONVERT_2I8F32_MASK>;
-def X86vmcvtbiasps2hf8  : SDNode<"X86ISD::VMCVTBIASPS2HF8",  SDTAVX10CONVERT_2I8F32_MASK>;
-def X86vmcvtbiasps2hf8s : SDNode<"X86ISD::VMCVTBIASPS2HF8S", SDTAVX10CONVERT_2I8F32_MASK>;
+def X86vcvtbiasps2bf8  : SDNode<"X86ISD::VCVTBIASPS2BF8",  SDTAVX10CONVERT_I8I32F32>;
+def X86vcvtbiasps2bf8s : SDNode<"X86ISD::VCVTBIASPS2BF8S", SDTAVX10CONVERT_I8I32F32>;
+def X86vcvtbiasps2hf8  : SDNode<"X86ISD::VCVTBIASPS2HF8",  SDTAVX10CONVERT_I8I32F32>;
+def X86vcvtbiasps2hf8s : SDNode<"X86ISD::VCVTBIASPS2HF8S", SDTAVX10CONVERT_I8I32F32>;
+
+def X86vmcvtbiasps2bf8  : SDNode<"X86ISD::VMCVTBIASPS2BF8",  SDTAVX10CONVERT_I8I32F32_MASK>;
+def X86vmcvtbiasps2bf8s : SDNode<"X86ISD::VMCVTBIASPS2BF8S", SDTAVX10CONVERT_I8I32F32_MASK>;
+def X86vmcvtbiasps2hf8  : SDNode<"X86ISD::VMCVTBIASPS2HF8",  SDTAVX10CONVERT_I8I32F32_MASK>;
+def X86vmcvtbiasps2hf8s : SDNode<"X86ISD::VMCVTBIASPS2HF8S", SDTAVX10CONVERT_I8I32F32_MASK>;
 
 // Group C: 8bit->PS expanding
 def X86vcvtbf82ps  : SDNode<"X86ISD::VCVTBF82PS",  SDTAVX10CONVERT_F32I8>;
diff --git a/llvm/lib/Target/X86/X86IntrinsicsInfo.h b/llvm/lib/Target/X86/X86IntrinsicsInfo.h
index e172cdd105a6a9..79c99e7c841c7e 100644
--- a/llvm/lib/Target/X86/X86IntrinsicsInfo.h
+++ b/llvm/lib/Target/X86/X86IntrinsicsInfo.h
@@ -480,6 +480,22 @@ static const IntrinsicData IntrinsicsWithoutChain[] = {
                        X86ISD::VCVTBIASPH2HF8S, 0),
     X86_INTRINSIC_DATA(avx10_mask_vcvtbiasph2hf8s512, INTR_TYPE_2OP_MASK,
                        X86ISD::VCVTBIASPH2HF8S, 0),
+    X86_INTRINSIC_DATA(avx10_mask_vcvtbiasps2bf8_128, TRUNCATE2_TO_REG,
+                       X86ISD::VCVTBIASPS2BF8, X86ISD::VMCVTBIASPS2BF8),
+    X86_INTRINSIC_DATA(avx10_mask_vcvtbiasps2bf8_256, TRUNCATE2_TO_REG,
+                       X86ISD::VCVTBIASPS2BF8, X86ISD::VMCVTBIASPS2BF8),
+    X86_INTRINSIC_DATA(avx10_mask_vcvtbiasps2bf8s_128, TRUNCATE2_TO_REG,
+                       X86ISD::VCVTBIASPS2BF8S, X86ISD::VMCVTBIASPS2BF8S),
+    X86_INTRINSIC_DATA(avx10_mask_vcvtbiasps2bf8s_256, TRUNCATE2_TO_REG,
+                       X86ISD::VCVTBIASPS2BF8S, X86ISD::VMCVTBIASPS2BF8S),
+    X86_INTRINSIC_DATA(avx10_mask_vcvtbiasps2hf8_128, TRUNCATE2_TO_REG,
+                       X86ISD::VCVTBIASPS2HF8, X86ISD::VMCVTBIASPS2HF8),
+    X86_INTRINSIC_DATA(avx10_mask_vcvtbiasps2hf8_256, TRUNCATE2_TO_REG,
+                       X86ISD::VCVTBIASPS2HF8, X86ISD::VMCVTBIASPS2HF8),
+    X86_INTRINSIC_DATA(avx10_mask_vcvtbiasps2hf8s_128, TRUNCATE2_TO_REG,
+                       X86ISD::VCVTBIASPS2HF8S, X86ISD::VMCVTBIASPS2HF8S),
+    X86_INTRINSIC_DATA(avx10_mask_vcvtbiasps2hf8s_256, TRUNCATE2_TO_REG,
+                       X86ISD::VCVTBIASPS2HF8S, X86ISD::VMCVTBIASPS2HF8S),
     X86_INTRINSIC_DATA(avx10_mask_vcvthf82ph128, INTR_TYPE_1OP_MASK,
                        X86ISD::VCVTHF82PH, 0),
     X86_INTRINSIC_DATA(avx10_mask_vcvthf82ph256, INTR_TYPE_1OP_MASK,
@@ -522,6 +538,22 @@ static const IntrinsicData IntrinsicsWithoutChain[] = {
                        X86ISD::CVTP2IUBS, 0),
     X86_INTRINSIC_DATA(avx10_mask_vcvtph2iubs512, INTR_TYPE_1OP_MASK,
                        X86ISD::CVTP2IUBS, X86ISD::CVTP2IUBS_RND),
+    X86_INTRINSIC_DATA(avx10_mask_vcvtps2bf8_128, TRUNCATE_TO_REG,
+                       X86ISD::VCVTPS2BF8, X86ISD::VMCVTPS2BF8),
+    X86_INTRINSIC_DATA(avx10_mask_vcvtps2bf8_256, TRUNCATE_TO_REG,
+                       X86ISD::VCVTPS2BF8, X86ISD::VMCVTPS2BF8),
+    X86_INTRINSIC_DATA(avx10_mask_vcvtps2bf8s_128, TRUNCATE_TO_REG,
+                       X86ISD::VCVTPS2BF8S, X86ISD::VMCVTPS2BF8S),
+    X86_INTRINSIC_DATA(avx10_mask_vcvtps2bf8s_256, TRUNCATE_TO_REG,
+                       X86ISD::VCVTPS2BF8S, X86ISD::VMCVTPS2BF8S),
+    X86_INTRINSIC_DATA(avx10_mask_vcvtps2hf8_128, TRUNCATE_TO_REG,
+                       X86ISD::VCVTPS2HF8, X86ISD::VMCVTPS2HF8),
+    X86_INTRINSIC_DATA(avx10_mask_vcvtps2hf8_256, TRUNCATE_TO_REG,
+                       X86ISD::VCVTPS2HF8, X86ISD::VMCVTPS2HF8),
+    X86_INTRINSIC_DATA(avx10_mask_vcvtps2hf8s_128, TRUNCATE_TO_REG,
+                       X86ISD::VCVTPS2HF8S, X86ISD::VMCVTPS2HF8S),
+    X86_INTRINSIC_DATA(avx10_mask_vcvtps2hf8s_256, TRUNCATE_TO_REG,
+                       X86ISD::VCVTPS2HF8S, X86ISD::VMCVTPS2HF8S),
     X86_INTRINSIC_DATA(avx10_mask_vcvtps2ibs128, INTR_TYPE_1OP_MASK,
                        X86ISD::CVTP2IBS, 0),
     X86_INTRINSIC_DATA(avx10_mask_vcvtps2ibs256, INTR_TYPE_1OP_MASK,
@@ -534,6 +566,14 @@ static const IntrinsicData IntrinsicsWithoutChain[] = {
                        X86ISD::CVTP2IUBS, 0),
     X86_INTRINSIC_DATA(avx10_mask_vcvtps2iubs512, INTR_TYPE_1OP_MASK,
                        X86ISD::CVTP2IUBS, X86ISD::CVTP2IUBS_RND),
+    X86_INTRINSIC_DATA(avx10_mask_vcvtrops2hf8_128, TRUNCATE_TO_REG,
+                       X86ISD::VCVTROPS2HF8, X86ISD::VMCVTROPS2HF8),
+    X86_INTRINSIC_DATA(avx10_mask_vcvtrops2hf8_256, TRUNCATE_TO_REG,
+                       X86ISD::VCVTROPS2HF8, X86ISD::VMCVTROPS2HF8),
+    X86_INTRINSIC_DATA(avx10_mask_vcvtrops2hf8s_128, TRUNCATE_TO_REG,
+                       X86ISD::VCVTROPS2HF8S, X86ISD::VMCVTROPS2HF8S),
+    X86_INTRINSIC_DATA(avx10_mask_vcvtrops2hf8s_256, TRUNCATE_TO_REG,
+                       X86ISD::VCVTROPS2HF8S, X86ISD::VMCVTROPS2HF8S),
     X86_INTRINSIC_DATA(avx10_mask_vcvttpd2dqs_128, CVTPD2DQ_MASK,
                        X86ISD::CVTTP2SIS, X86ISD::MCVTTP2SIS),
     X86_INTRINSIC_DATA(avx10_mask_vcvttpd2dqs_256, INTR_TYPE_1OP_MASK,
diff --git a/llvm/test/CodeGen/X86/avx10_v2aux-intrinsics.ll b/llvm/test/CodeGen/X86/avx10_v2aux-intrinsics.ll
index 046ed29628a57d..b697cb7cec2727 100644
--- a/llvm/test/CodeGen/X86/avx10_v2aux-intrinsics.ll
+++ b/llvm/test/CodeGen/X86/avx10_v2aux-intrinsics.ll
@@ -494,41 +494,41 @@ define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8s_mem_512(ptr %ptr_a) {
   ret <16 x i8> %ret
 }
 
-declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8.128(<16 x i8>, <4 x float>)
-declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8.256(<32 x i8>, <8 x float>)
-declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8.512(<64 x i8>, <16 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8.128(<4 x i32>, <4 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8.256(<8 x i32>, <8 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8.512(<16 x i32>, <16 x float>)
 
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8_128(<16 x i8> %A, <4 x float> %b) {
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8_128(<4 x i32> %A, <4 x float> %b) {
 ; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2bf8_128:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvtbiasps2bf8 %xmm1, %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x39,0xc1]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8.128(<16 x i8> %A, <4 x float> %b)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8.128(<4 x i32> %A, <4 x float> %b)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8_256(<32 x i8> %A, <8 x float> %b) {
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8_256(<8 x i32> %A, <8 x float> %b) {
 ; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2bf8_256:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvtbiasps2bf8 %ymm1, %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x28,0x39,0xc1]
 ; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8.256(<32 x i8> %A, <8 x float> %b)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8.256(<8 x i32> %A, <8 x float> %b)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8_512(<64 x i8> %A, <16 x float> %b) {
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8_512(<16 x i32> %A, <16 x float> %b) {
 ; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2bf8_512:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvtbiasps2bf8 %zmm1, %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x39,0xc1]
 ; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8.512(<64 x i8> %A, <16 x float> %b)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8.512(<16 x i32> %A, <16 x float> %b)
   ret <16 x i8> %ret
 }
 
 ; Memory folding tests for vcvtbiasps2bf8 (second operand from memory)
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8_mem_128(<16 x i8> %A, ptr %ptr_b) {
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8_mem_128(<4 x i32> %A, ptr %ptr_b) {
 ; X64-LABEL: test_int_x86_avx10_vcvtbiasps2bf8_mem_128:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vcvtbiasps2bf8 (%rdi), %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x39,0x07]
@@ -540,11 +540,11 @@ define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8_mem_128(<16 x i8> %A, ptr %p
 ; X86-NEXT:    vcvtbiasps2bf8 (%eax), %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x39,0x00]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %b = load <4 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8.128(<16 x i8> %A, <4 x float> %b)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8.128(<4 x i32> %A, <4 x float> %b)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8_mem_256(<32 x i8> %A, ptr %ptr_b) {
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8_mem_256(<8 x i32> %A, ptr %ptr_b) {
 ; X64-LABEL: test_int_x86_avx10_vcvtbiasps2bf8_mem_256:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vcvtbiasps2bf8 (%rdi), %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x28,0x39,0x07]
@@ -558,11 +558,11 @@ define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8_mem_256(<32 x i8> %A, ptr %p
 ; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %b = load <8 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8.256(<32 x i8> %A, <8 x float> %b)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8.256(<8 x i32> %A, <8 x float> %b)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8_mem_512(<64 x i8> %A, ptr %ptr_b) {
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8_mem_512(<16 x i32> %A, ptr %ptr_b) {
 ; X64-LABEL: test_int_x86_avx10_vcvtbiasps2bf8_mem_512:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vcvtbiasps2bf8 (%rdi), %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x39,0x07]
@@ -576,45 +576,45 @@ define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8_mem_512(<64 x i8> %A, ptr %p
 ; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %b = load <16 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8.512(<64 x i8> %A, <16 x float> %b)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8.512(<16 x i32> %A, <16 x float> %b)
   ret <16 x i8> %ret
 }
 
-declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s.128(<16 x i8>, <4 x float>)
-declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s.256(<32 x i8>, <8 x float>)
-declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s.512(<64 x i8>, <16 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s.128(<4 x i32>, <4 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s.256(<8 x i32>, <8 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s.512(<16 x i32>, <16 x float>)
 
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8s_128(<16 x i8> %A, <4 x float> %b) {
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8s_128(<4 x i32> %A, <4 x float> %b) {
 ; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2bf8s_128:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvtbiasps2bf8s %xmm1, %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x3b,0xc1]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s.128(<16 x i8> %A, <4 x float> %b)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s.128(<4 x i32> %A, <4 x float> %b)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8s_256(<32 x i8> %A, <8 x float> %b) {
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8s_256(<8 x i32> %A, <8 x float> %b) {
 ; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2bf8s_256:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvtbiasps2bf8s %ymm1, %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x28,0x3b,0xc1]
 ; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s.256(<32 x i8> %A, <8 x float> %b)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s.256(<8 x i32> %A, <8 x float> %b)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8s_512(<64 x i8> %A, <16 x float> %b) {
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8s_512(<16 x i32> %A, <16 x float> %b) {
 ; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2bf8s_512:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvtbiasps2bf8s %zmm1, %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x3b,0xc1]
 ; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s.512(<64 x i8> %A, <16 x float> %b)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s.512(<16 x i32> %A, <16 x float> %b)
   ret <16 x i8> %ret
 }
 
 ; Memory folding tests for vcvtbiasps2bf8s
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8s_mem_128(<16 x i8> %A, ptr %ptr_b) {
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8s_mem_128(<4 x i32> %A, ptr %ptr_b) {
 ; X64-LABEL: test_int_x86_avx10_vcvtbiasps2bf8s_mem_128:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vcvtbiasps2bf8s (%rdi), %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x3b,0x07]
@@ -626,11 +626,11 @@ define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8s_mem_128(<16 x i8> %A, ptr %
 ; X86-NEXT:    vcvtbiasps2bf8s (%eax), %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x3b,0x00]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %b = load <4 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s.128(<16 x i8> %A, <4 x float> %b)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s.128(<4 x i32> %A, <4 x float> %b)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8s_mem_256(<32 x i8> %A, ptr %ptr_b) {
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8s_mem_256(<8 x i32> %A, ptr %ptr_b) {
 ; X64-LABEL: test_int_x86_avx10_vcvtbiasps2bf8s_mem_256:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vcvtbiasps2bf8s (%rdi), %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x28,0x3b,0x07]
@@ -644,11 +644,11 @@ define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8s_mem_256(<32 x i8> %A, ptr %
 ; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %b = load <8 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s.256(<32 x i8> %A, <8 x float> %b)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s.256(<8 x i32> %A, <8 x float> %b)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8s_mem_512(<64 x i8> %A, ptr %ptr_b) {
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8s_mem_512(<16 x i32> %A, ptr %ptr_b) {
 ; X64-LABEL: test_int_x86_avx10_vcvtbiasps2bf8s_mem_512:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vcvtbiasps2bf8s (%rdi), %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x3b,0x07]
@@ -662,45 +662,45 @@ define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8s_mem_512(<64 x i8> %A, ptr %
 ; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %b = load <16 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s.512(<64 x i8> %A, <16 x float> %b)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s.512(<16 x i32> %A, <16 x float> %b)
   ret <16 x i8> %ret
 }
 
-declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8.128(<16 x i8>, <4 x float>)
-declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8.256(<32 x i8>, <8 x float>)
-declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8.512(<64 x i8>, <16 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8.128(<4 x i32>, <4 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8.256(<8 x i32>, <8 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8.512(<16 x i32>, <16 x float>)
 
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8_128(<16 x i8> %A, <4 x float> %b) {
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8_128(<4 x i32> %A, <4 x float> %b) {
 ; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2hf8_128:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvtbiasps2hf8 %xmm1, %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x38,0xc1]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8.128(<16 x i8> %A, <4 x float> %b)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8.128(<4 x i32> %A, <4 x float> %b)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8_256(<32 x i8> %A, <8 x float> %b) {
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8_256(<8 x i32> %A, <8 x float> %b) {
 ; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2hf8_256:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvtbiasps2hf8 %ymm1, %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x28,0x38,0xc1]
 ; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8.256(<32 x i8> %A, <8 x float> %b)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8.256(<8 x i32> %A, <8 x float> %b)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8_512(<64 x i8> %A, <16 x float> %b) {
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8_512(<16 x i32> %A, <16 x float> %b) {
 ; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2hf8_512:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvtbiasps2hf8 %zmm1, %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x38,0xc1]
 ; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8.512(<64 x i8> %A, <16 x float> %b)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8.512(<16 x i32> %A, <16 x float> %b)
   ret <16 x i8> %ret
 }
 
 ; Memory folding tests for vcvtbiasps2hf8
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8_mem_128(<16 x i8> %A, ptr %ptr_b) {
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8_mem_128(<4 x i32> %A, ptr %ptr_b) {
 ; X64-LABEL: test_int_x86_avx10_vcvtbiasps2hf8_mem_128:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vcvtbiasps2hf8 (%rdi), %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x38,0x07]
@@ -712,11 +712,11 @@ define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8_mem_128(<16 x i8> %A, ptr %p
 ; X86-NEXT:    vcvtbiasps2hf8 (%eax), %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x38,0x00]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %b = load <4 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8.128(<16 x i8> %A, <4 x float> %b)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8.128(<4 x i32> %A, <4 x float> %b)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8_mem_256(<32 x i8> %A, ptr %ptr_b) {
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8_mem_256(<8 x i32> %A, ptr %ptr_b) {
 ; X64-LABEL: test_int_x86_avx10_vcvtbiasps2hf8_mem_256:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vcvtbiasps2hf8 (%rdi), %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x28,0x38,0x07]
@@ -730,11 +730,11 @@ define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8_mem_256(<32 x i8> %A, ptr %p
 ; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %b = load <8 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8.256(<32 x i8> %A, <8 x float> %b)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8.256(<8 x i32> %A, <8 x float> %b)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8_mem_512(<64 x i8> %A, ptr %ptr_b) {
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8_mem_512(<16 x i32> %A, ptr %ptr_b) {
 ; X64-LABEL: test_int_x86_avx10_vcvtbiasps2hf8_mem_512:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vcvtbiasps2hf8 (%rdi), %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x38,0x07]
@@ -748,45 +748,45 @@ define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8_mem_512(<64 x i8> %A, ptr %p
 ; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %b = load <16 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8.512(<64 x i8> %A, <16 x float> %b)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8.512(<16 x i32> %A, <16 x float> %b)
   ret <16 x i8> %ret
 }
 
-declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s.128(<16 x i8>, <4 x float>)
-declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s.256(<32 x i8>, <8 x float>)
-declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s.512(<64 x i8>, <16 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s.128(<4 x i32>, <4 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s.256(<8 x i32>, <8 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s.512(<16 x i32>, <16 x float>)
 
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8s_128(<16 x i8> %A, <4 x float> %b) {
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8s_128(<4 x i32> %A, <4 x float> %b) {
 ; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2hf8s_128:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvtbiasps2hf8s %xmm1, %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x3a,0xc1]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s.128(<16 x i8> %A, <4 x float> %b)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s.128(<4 x i32> %A, <4 x float> %b)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8s_256(<32 x i8> %A, <8 x float> %b) {
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8s_256(<8 x i32> %A, <8 x float> %b) {
 ; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2hf8s_256:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvtbiasps2hf8s %ymm1, %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x28,0x3a,0xc1]
 ; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s.256(<32 x i8> %A, <8 x float> %b)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s.256(<8 x i32> %A, <8 x float> %b)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8s_512(<64 x i8> %A, <16 x float> %b) {
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8s_512(<16 x i32> %A, <16 x float> %b) {
 ; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2hf8s_512:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvtbiasps2hf8s %zmm1, %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x3a,0xc1]
 ; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s.512(<64 x i8> %A, <16 x float> %b)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s.512(<16 x i32> %A, <16 x float> %b)
   ret <16 x i8> %ret
 }
 
 ; Memory folding tests for vcvtbiasps2hf8s
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8s_mem_128(<16 x i8> %A, ptr %ptr_b) {
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8s_mem_128(<4 x i32> %A, ptr %ptr_b) {
 ; X64-LABEL: test_int_x86_avx10_vcvtbiasps2hf8s_mem_128:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vcvtbiasps2hf8s (%rdi), %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x3a,0x07]
@@ -798,11 +798,11 @@ define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8s_mem_128(<16 x i8> %A, ptr %
 ; X86-NEXT:    vcvtbiasps2hf8s (%eax), %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x3a,0x00]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %b = load <4 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s.128(<16 x i8> %A, <4 x float> %b)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s.128(<4 x i32> %A, <4 x float> %b)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8s_mem_256(<32 x i8> %A, ptr %ptr_b) {
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8s_mem_256(<8 x i32> %A, ptr %ptr_b) {
 ; X64-LABEL: test_int_x86_avx10_vcvtbiasps2hf8s_mem_256:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vcvtbiasps2hf8s (%rdi), %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x28,0x3a,0x07]
@@ -816,11 +816,11 @@ define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8s_mem_256(<32 x i8> %A, ptr %
 ; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %b = load <8 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s.256(<32 x i8> %A, <8 x float> %b)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s.256(<8 x i32> %A, <8 x float> %b)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8s_mem_512(<64 x i8> %A, ptr %ptr_b) {
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8s_mem_512(<16 x i32> %A, ptr %ptr_b) {
 ; X64-LABEL: test_int_x86_avx10_vcvtbiasps2hf8s_mem_512:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vcvtbiasps2hf8s (%rdi), %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x3a,0x07]
@@ -834,7 +834,7 @@ define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8s_mem_512(<64 x i8> %A, ptr %
 ; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %b = load <16 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s.512(<64 x i8> %A, <16 x float> %b)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s.512(<16 x i32> %A, <16 x float> %b)
   ret <16 x i8> %ret
 }
 
@@ -1428,6 +1428,130 @@ define <64 x i8> @test_int_x86_avx10_vcvtbf42hf8_mem_512(ptr %ptr_a) {
   ret <64 x i8> %ret
 }
 
+; The 128-bit form only reads 8 bytes, so a 64-bit zero-extending load
+; (_mm_loadu_si64) must fold into the memory operand too.
+define <16 x i8> @test_int_x86_avx10_vcvtbf42hf8_vzload_128(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf42hf8_vzload_128:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtbf42hf8 (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x37,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbf42hf8_vzload_128:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtbf42hf8 (%eax), %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x37,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %l = load i64, ptr %ptr_a, align 1
+  %v = insertelement <2 x i64> <i64 poison, i64 0>, i64 %l, i64 0
+  %a = bitcast <2 x i64> %v to <16 x i8>
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf42hf8.128(<16 x i8> %a)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtbf42hf8_vzload_mask_128(ptr %ptr_a, <16 x i8> %src, i16 %mask) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf42hf8_vzload_mask_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtbf42hf8 (%rdi), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x37,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbf42hf8_vzload_mask_128:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtbf42hf8 (%eax), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x37,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %l = load i64, ptr %ptr_a, align 1
+  %v = insertelement <2 x i64> <i64 poison, i64 0>, i64 %l, i64 0
+  %a = bitcast <2 x i64> %v to <16 x i8>
+  %cvt = call <16 x i8> @llvm.x86.avx10.vcvtbf42hf8.128(<16 x i8> %a)
+  %msk = bitcast i16 %mask to <16 x i1>
+  %ret = select <16 x i1> %msk, <16 x i8> %cvt, <16 x i8> %src
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtbf42hf8_vzload_maskz_128(ptr %ptr_a, i16 %mask) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf42hf8_vzload_maskz_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtbf42hf8 (%rdi), %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x37,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbf42hf8_vzload_maskz_128:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtbf42hf8 (%eax), %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x37,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %l = load i64, ptr %ptr_a, align 1
+  %v = insertelement <2 x i64> <i64 poison, i64 0>, i64 %l, i64 0
+  %a = bitcast <2 x i64> %v to <16 x i8>
+  %cvt = call <16 x i8> @llvm.x86.avx10.vcvtbf42hf8.128(<16 x i8> %a)
+  %msk = bitcast i16 %mask to <16 x i1>
+  %ret = select <16 x i1> %msk, <16 x i8> %cvt, <16 x i8> zeroinitializer
+  ret <16 x i8> %ret
+}
+
+; Same, spelled as a scalar_to_vector of an i64 load (_mm_loadl_epi64).
+define <16 x i8> @test_int_x86_avx10_vcvtbf42hf8_s2v_128(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf42hf8_s2v_128:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtbf42hf8 (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x37,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbf42hf8_s2v_128:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtbf42hf8 (%eax), %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x37,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %l = load i64, ptr %ptr_a, align 8
+  %v = insertelement <2 x i64> undef, i64 %l, i64 0
+  %a = bitcast <2 x i64> %v to <16 x i8>
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf42hf8.128(<16 x i8> %a)
+  ret <16 x i8> %ret
+}
+
+; Masked variants of the plain 16-byte load, which must keep folding.
+define <16 x i8> @test_int_x86_avx10_vcvtbf42hf8_mem_mask_128(ptr %ptr_a, <16 x i8> %src, i16 %mask) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf42hf8_mem_mask_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtbf42hf8 (%rdi), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x37,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbf42hf8_mem_mask_128:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtbf42hf8 (%eax), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x37,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <16 x i8>, ptr %ptr_a
+  %cvt = call <16 x i8> @llvm.x86.avx10.vcvtbf42hf8.128(<16 x i8> %a)
+  %msk = bitcast i16 %mask to <16 x i1>
+  %ret = select <16 x i1> %msk, <16 x i8> %cvt, <16 x i8> %src
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtbf42hf8_mem_maskz_128(ptr %ptr_a, i16 %mask) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf42hf8_mem_maskz_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtbf42hf8 (%rdi), %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x37,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbf42hf8_mem_maskz_128:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtbf42hf8 (%eax), %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x37,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <16 x i8>, ptr %ptr_a
+  %cvt = call <16 x i8> @llvm.x86.avx10.vcvtbf42hf8.128(<16 x i8> %a)
+  %msk = bitcast i16 %mask to <16 x i1>
+  %ret = select <16 x i1> %msk, <16 x i8> %cvt, <16 x i8> zeroinitializer
+  ret <16 x i8> %ret
+}
+
 define <16 x i8> @test_int_x86_avx10_vcvtbf62hf8_128(<16 x i8> %a) {
 ; CHECK-LABEL: test_int_x86_avx10_vcvtbf62hf8_128:
 ; CHECK:       # %bb.0:
@@ -1860,3 +1984,1565 @@ define <16 x i8> @test_int_x86_avx10_pmovssdb_mem_512(ptr %ptr_a) {
   %ret = call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.512(<16 x i32> %a, <16 x i8> zeroinitializer, i16 -1)
   ret <16 x i8> %ret
 }
+
+; Masked 128/256-bit forms of vcvtps2bf8
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtps2bf8_128(<4 x float> %b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtps2bf8_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtps2bf8 %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x39,0xc8]
+; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtps2bf8_128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtps2bf8 %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x39,0xc8]
+; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.128(<4 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2bf8_128(<4 x float> %b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtps2bf8 %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x89,0x39,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8_128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtps2bf8 %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x89,0x39,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.128(<4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtps2bf8_mem_128(ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtps2bf8_mem_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtps2bf8x (%rdi), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x39,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtps2bf8_mem_128:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtps2bf8x (%eax), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x39,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %b = load <4 x float>, ptr %ptr_b
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.128(<4 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2bf8_bcst_128(ptr %ptr_b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8_bcst_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtps2bf8 (%rdi){1to4}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x99,0x39,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8_bcst_128:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtps2bf8 (%eax){1to4}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x99,0x39,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ld = load float, ptr %ptr_b
+  %ins = insertelement <4 x float> poison, float %ld, i32 0
+  %b = shufflevector <4 x float> %ins, <4 x float> poison, <4 x i32> zeroinitializer
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.128(<4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtps2bf8_256(<8 x float> %b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtps2bf8_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtps2bf8 %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x39,0xc8]
+; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtps2bf8_256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtps2bf8 %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x39,0xc8]
+; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.256(<8 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2bf8_256(<8 x float> %b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtps2bf8 %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xa9,0x39,0xc0]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8_256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtps2bf8 %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xa9,0x39,0xc0]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.256(<8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtps2bf8_mem_256(ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtps2bf8_mem_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtps2bf8y (%rdi), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x39,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtps2bf8_mem_256:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtps2bf8y (%eax), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x39,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %b = load <8 x float>, ptr %ptr_b
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.256(<8 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2bf8_bcst_256(ptr %ptr_b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8_bcst_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtps2bf8 (%rdi){1to8}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xb9,0x39,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8_bcst_256:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtps2bf8 (%eax){1to8}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xb9,0x39,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ld = load float, ptr %ptr_b
+  %ins = insertelement <8 x float> poison, float %ld, i32 0
+  %b = shufflevector <8 x float> %ins, <8 x float> poison, <8 x i32> zeroinitializer
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.256(<8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.128(<4 x float>, <16 x i8>, i8)
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.256(<8 x float>, <16 x i8>, i8)
+
+; Masked 128/256-bit forms of vcvtps2bf8s
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtps2bf8s_128(<4 x float> %b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtps2bf8s_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtps2bf8s %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x3b,0xc8]
+; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtps2bf8s_128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtps2bf8s %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x3b,0xc8]
+; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.128(<4 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2bf8s_128(<4 x float> %b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8s_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtps2bf8s %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x89,0x3b,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8s_128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtps2bf8s %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x89,0x3b,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.128(<4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtps2bf8s_mem_128(ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtps2bf8s_mem_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtps2bf8sx (%rdi), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x3b,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtps2bf8s_mem_128:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtps2bf8sx (%eax), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x3b,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %b = load <4 x float>, ptr %ptr_b
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.128(<4 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2bf8s_bcst_128(ptr %ptr_b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8s_bcst_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtps2bf8s (%rdi){1to4}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x99,0x3b,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8s_bcst_128:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtps2bf8s (%eax){1to4}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x99,0x3b,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ld = load float, ptr %ptr_b
+  %ins = insertelement <4 x float> poison, float %ld, i32 0
+  %b = shufflevector <4 x float> %ins, <4 x float> poison, <4 x i32> zeroinitializer
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.128(<4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtps2bf8s_256(<8 x float> %b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtps2bf8s_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtps2bf8s %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x3b,0xc8]
+; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtps2bf8s_256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtps2bf8s %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x3b,0xc8]
+; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.256(<8 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2bf8s_256(<8 x float> %b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8s_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtps2bf8s %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xa9,0x3b,0xc0]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8s_256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtps2bf8s %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xa9,0x3b,0xc0]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.256(<8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtps2bf8s_mem_256(ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtps2bf8s_mem_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtps2bf8sy (%rdi), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x3b,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtps2bf8s_mem_256:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtps2bf8sy (%eax), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x3b,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %b = load <8 x float>, ptr %ptr_b
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.256(<8 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2bf8s_bcst_256(ptr %ptr_b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8s_bcst_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtps2bf8s (%rdi){1to8}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xb9,0x3b,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8s_bcst_256:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtps2bf8s (%eax){1to8}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xb9,0x3b,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ld = load float, ptr %ptr_b
+  %ins = insertelement <8 x float> poison, float %ld, i32 0
+  %b = shufflevector <8 x float> %ins, <8 x float> poison, <8 x i32> zeroinitializer
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.256(<8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.128(<4 x float>, <16 x i8>, i8)
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.256(<8 x float>, <16 x i8>, i8)
+
+; Masked 128/256-bit forms of vcvtps2hf8
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtps2hf8_128(<4 x float> %b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtps2hf8_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtps2hf8 %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x38,0xc8]
+; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtps2hf8_128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtps2hf8 %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x38,0xc8]
+; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.128(<4 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2hf8_128(<4 x float> %b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtps2hf8 %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x89,0x38,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8_128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtps2hf8 %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x89,0x38,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.128(<4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtps2hf8_mem_128(ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtps2hf8_mem_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtps2hf8x (%rdi), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x38,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtps2hf8_mem_128:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtps2hf8x (%eax), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x38,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %b = load <4 x float>, ptr %ptr_b
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.128(<4 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2hf8_bcst_128(ptr %ptr_b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8_bcst_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtps2hf8 (%rdi){1to4}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x99,0x38,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8_bcst_128:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtps2hf8 (%eax){1to4}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x99,0x38,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ld = load float, ptr %ptr_b
+  %ins = insertelement <4 x float> poison, float %ld, i32 0
+  %b = shufflevector <4 x float> %ins, <4 x float> poison, <4 x i32> zeroinitializer
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.128(<4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtps2hf8_256(<8 x float> %b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtps2hf8_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtps2hf8 %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x38,0xc8]
+; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtps2hf8_256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtps2hf8 %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x38,0xc8]
+; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.256(<8 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2hf8_256(<8 x float> %b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtps2hf8 %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xa9,0x38,0xc0]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8_256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtps2hf8 %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xa9,0x38,0xc0]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.256(<8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtps2hf8_mem_256(ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtps2hf8_mem_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtps2hf8y (%rdi), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x38,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtps2hf8_mem_256:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtps2hf8y (%eax), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x38,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %b = load <8 x float>, ptr %ptr_b
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.256(<8 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2hf8_bcst_256(ptr %ptr_b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8_bcst_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtps2hf8 (%rdi){1to8}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xb9,0x38,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8_bcst_256:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtps2hf8 (%eax){1to8}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xb9,0x38,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ld = load float, ptr %ptr_b
+  %ins = insertelement <8 x float> poison, float %ld, i32 0
+  %b = shufflevector <8 x float> %ins, <8 x float> poison, <8 x i32> zeroinitializer
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.256(<8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.128(<4 x float>, <16 x i8>, i8)
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.256(<8 x float>, <16 x i8>, i8)
+
+; Masked 128/256-bit forms of vcvtps2hf8s
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtps2hf8s_128(<4 x float> %b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtps2hf8s_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtps2hf8s %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x3a,0xc8]
+; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtps2hf8s_128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtps2hf8s %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x3a,0xc8]
+; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.128(<4 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2hf8s_128(<4 x float> %b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8s_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtps2hf8s %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x89,0x3a,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8s_128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtps2hf8s %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x89,0x3a,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.128(<4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtps2hf8s_mem_128(ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtps2hf8s_mem_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtps2hf8sx (%rdi), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x3a,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtps2hf8s_mem_128:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtps2hf8sx (%eax), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x3a,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %b = load <4 x float>, ptr %ptr_b
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.128(<4 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2hf8s_bcst_128(ptr %ptr_b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8s_bcst_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtps2hf8s (%rdi){1to4}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x99,0x3a,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8s_bcst_128:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtps2hf8s (%eax){1to4}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x99,0x3a,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ld = load float, ptr %ptr_b
+  %ins = insertelement <4 x float> poison, float %ld, i32 0
+  %b = shufflevector <4 x float> %ins, <4 x float> poison, <4 x i32> zeroinitializer
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.128(<4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtps2hf8s_256(<8 x float> %b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtps2hf8s_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtps2hf8s %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x3a,0xc8]
+; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtps2hf8s_256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtps2hf8s %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x3a,0xc8]
+; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.256(<8 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2hf8s_256(<8 x float> %b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8s_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtps2hf8s %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xa9,0x3a,0xc0]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8s_256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtps2hf8s %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xa9,0x3a,0xc0]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.256(<8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtps2hf8s_mem_256(ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtps2hf8s_mem_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtps2hf8sy (%rdi), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x3a,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtps2hf8s_mem_256:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtps2hf8sy (%eax), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x3a,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %b = load <8 x float>, ptr %ptr_b
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.256(<8 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2hf8s_bcst_256(ptr %ptr_b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8s_bcst_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtps2hf8s (%rdi){1to8}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xb9,0x3a,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8s_bcst_256:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtps2hf8s (%eax){1to8}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xb9,0x3a,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ld = load float, ptr %ptr_b
+  %ins = insertelement <8 x float> poison, float %ld, i32 0
+  %b = shufflevector <8 x float> %ins, <8 x float> poison, <8 x i32> zeroinitializer
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.256(<8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.128(<4 x float>, <16 x i8>, i8)
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.256(<8 x float>, <16 x i8>, i8)
+
+; Masked 128/256-bit forms of vcvtrops2hf8
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtrops2hf8_128(<4 x float> %b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtrops2hf8 %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7d,0x09,0x38,0xc8]
+; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8_128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtrops2hf8 %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7d,0x09,0x38,0xc8]
+; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.128(<4 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtrops2hf8_128(<4 x float> %b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtrops2hf8 %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0x89,0x38,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8_128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtrops2hf8 %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0x89,0x38,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.128(<4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtrops2hf8_mem_128(ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8_mem_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtrops2hf8x (%rdi), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7d,0x09,0x38,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8_mem_128:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtrops2hf8x (%eax), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7d,0x09,0x38,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %b = load <4 x float>, ptr %ptr_b
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.128(<4 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtrops2hf8_bcst_128(ptr %ptr_b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8_bcst_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtrops2hf8 (%rdi){1to4}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0x99,0x38,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8_bcst_128:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtrops2hf8 (%eax){1to4}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0x99,0x38,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ld = load float, ptr %ptr_b
+  %ins = insertelement <4 x float> poison, float %ld, i32 0
+  %b = shufflevector <4 x float> %ins, <4 x float> poison, <4 x i32> zeroinitializer
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.128(<4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtrops2hf8_256(<8 x float> %b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtrops2hf8 %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7d,0x29,0x38,0xc8]
+; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8_256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtrops2hf8 %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7d,0x29,0x38,0xc8]
+; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.256(<8 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtrops2hf8_256(<8 x float> %b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtrops2hf8 %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0xa9,0x38,0xc0]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8_256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtrops2hf8 %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0xa9,0x38,0xc0]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.256(<8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtrops2hf8_mem_256(ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8_mem_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtrops2hf8y (%rdi), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7d,0x29,0x38,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8_mem_256:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtrops2hf8y (%eax), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7d,0x29,0x38,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %b = load <8 x float>, ptr %ptr_b
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.256(<8 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtrops2hf8_bcst_256(ptr %ptr_b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8_bcst_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtrops2hf8 (%rdi){1to8}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0xb9,0x38,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8_bcst_256:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtrops2hf8 (%eax){1to8}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0xb9,0x38,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ld = load float, ptr %ptr_b
+  %ins = insertelement <8 x float> poison, float %ld, i32 0
+  %b = shufflevector <8 x float> %ins, <8 x float> poison, <8 x i32> zeroinitializer
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.256(<8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.128(<4 x float>, <16 x i8>, i8)
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.256(<8 x float>, <16 x i8>, i8)
+
+; Masked 128/256-bit forms of vcvtrops2hf8s
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtrops2hf8s_128(<4 x float> %b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8s_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtrops2hf8s %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7d,0x09,0x3a,0xc8]
+; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8s_128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtrops2hf8s %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7d,0x09,0x3a,0xc8]
+; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.128(<4 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtrops2hf8s_128(<4 x float> %b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8s_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtrops2hf8s %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0x89,0x3a,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8s_128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtrops2hf8s %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0x89,0x3a,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.128(<4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtrops2hf8s_mem_128(ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8s_mem_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtrops2hf8sx (%rdi), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7d,0x09,0x3a,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8s_mem_128:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtrops2hf8sx (%eax), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7d,0x09,0x3a,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %b = load <4 x float>, ptr %ptr_b
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.128(<4 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtrops2hf8s_bcst_128(ptr %ptr_b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8s_bcst_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtrops2hf8s (%rdi){1to4}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0x99,0x3a,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8s_bcst_128:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtrops2hf8s (%eax){1to4}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0x99,0x3a,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ld = load float, ptr %ptr_b
+  %ins = insertelement <4 x float> poison, float %ld, i32 0
+  %b = shufflevector <4 x float> %ins, <4 x float> poison, <4 x i32> zeroinitializer
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.128(<4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtrops2hf8s_256(<8 x float> %b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8s_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtrops2hf8s %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7d,0x29,0x3a,0xc8]
+; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8s_256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtrops2hf8s %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7d,0x29,0x3a,0xc8]
+; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.256(<8 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtrops2hf8s_256(<8 x float> %b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8s_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtrops2hf8s %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0xa9,0x3a,0xc0]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8s_256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtrops2hf8s %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0xa9,0x3a,0xc0]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.256(<8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtrops2hf8s_mem_256(ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8s_mem_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtrops2hf8sy (%rdi), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7d,0x29,0x3a,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8s_mem_256:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtrops2hf8sy (%eax), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7d,0x29,0x3a,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %b = load <8 x float>, ptr %ptr_b
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.256(<8 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtrops2hf8s_bcst_256(ptr %ptr_b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8s_bcst_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtrops2hf8s (%rdi){1to8}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0xb9,0x3a,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8s_bcst_256:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtrops2hf8s (%eax){1to8}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0xb9,0x3a,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ld = load float, ptr %ptr_b
+  %ins = insertelement <8 x float> poison, float %ld, i32 0
+  %b = shufflevector <8 x float> %ins, <8 x float> poison, <8 x i32> zeroinitializer
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.256(<8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.128(<4 x float>, <16 x i8>, i8)
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.256(<8 x float>, <16 x i8>, i8)
+
+; Masked 128/256-bit forms of vcvtbiasps2bf8
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2bf8_128(<4 x i32> %A, <4 x float> %b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbiasps2bf8 %xmm1, %xmm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x39,0xd1]
+; X64-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8_128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2bf8 %xmm1, %xmm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x39,0xd1]
+; X86-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.128(<4 x i32> %A, <4 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2bf8_128(<4 x i32> %A, <4 x float> %b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbiasps2bf8 %xmm1, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x39,0xc1]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8_128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2bf8 %xmm1, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x39,0xc1]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.128(<4 x i32> %A, <4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2bf8_mem_128(<4 x i32> %A, ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8_mem_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtbiasps2bf8 (%rdi), %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x39,0x0f]
+; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8_mem_128:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtbiasps2bf8 (%eax), %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x39,0x08]
+; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %b = load <4 x float>, ptr %ptr_b
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.128(<4 x i32> %A, <4 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2bf8_bcst_128(<4 x i32> %A, ptr %ptr_b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8_bcst_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtbiasps2bf8 (%rdi){1to4}, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x99,0x39,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8_bcst_128:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtbiasps2bf8 (%eax){1to4}, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x99,0x39,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ld = load float, ptr %ptr_b
+  %ins = insertelement <4 x float> poison, float %ld, i32 0
+  %b = shufflevector <4 x float> %ins, <4 x float> poison, <4 x i32> zeroinitializer
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.128(<4 x i32> %A, <4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2bf8_256(<8 x i32> %A, <8 x float> %b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbiasps2bf8 %ymm1, %ymm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x39,0xd1]
+; X64-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8_256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2bf8 %ymm1, %ymm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x39,0xd1]
+; X86-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.256(<8 x i32> %A, <8 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2bf8_256(<8 x i32> %A, <8 x float> %b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbiasps2bf8 %ymm1, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xa9,0x39,0xc1]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8_256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2bf8 %ymm1, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xa9,0x39,0xc1]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.256(<8 x i32> %A, <8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2bf8_mem_256(<8 x i32> %A, ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8_mem_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtbiasps2bf8 (%rdi), %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x39,0x0f]
+; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8_mem_256:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtbiasps2bf8 (%eax), %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x39,0x08]
+; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %b = load <8 x float>, ptr %ptr_b
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.256(<8 x i32> %A, <8 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2bf8_bcst_256(<8 x i32> %A, ptr %ptr_b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8_bcst_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtbiasps2bf8 (%rdi){1to8}, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xb9,0x39,0x07]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8_bcst_256:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtbiasps2bf8 (%eax){1to8}, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xb9,0x39,0x00]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ld = load float, ptr %ptr_b
+  %ins = insertelement <8 x float> poison, float %ld, i32 0
+  %b = shufflevector <8 x float> %ins, <8 x float> poison, <8 x i32> zeroinitializer
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.256(<8 x i32> %A, <8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.128(<4 x i32>, <4 x float>, <16 x i8>, i8)
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.256(<8 x i32>, <8 x float>, <16 x i8>, i8)
+
+; Masked 128/256-bit forms of vcvtbiasps2bf8s
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2bf8s_128(<4 x i32> %A, <4 x float> %b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8s_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbiasps2bf8s %xmm1, %xmm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x3b,0xd1]
+; X64-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8s_128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2bf8s %xmm1, %xmm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x3b,0xd1]
+; X86-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.128(<4 x i32> %A, <4 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2bf8s_128(<4 x i32> %A, <4 x float> %b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8s_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbiasps2bf8s %xmm1, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x3b,0xc1]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8s_128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2bf8s %xmm1, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x3b,0xc1]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.128(<4 x i32> %A, <4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2bf8s_mem_128(<4 x i32> %A, ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8s_mem_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtbiasps2bf8s (%rdi), %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x3b,0x0f]
+; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8s_mem_128:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtbiasps2bf8s (%eax), %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x3b,0x08]
+; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %b = load <4 x float>, ptr %ptr_b
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.128(<4 x i32> %A, <4 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2bf8s_bcst_128(<4 x i32> %A, ptr %ptr_b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8s_bcst_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtbiasps2bf8s (%rdi){1to4}, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x99,0x3b,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8s_bcst_128:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtbiasps2bf8s (%eax){1to4}, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x99,0x3b,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ld = load float, ptr %ptr_b
+  %ins = insertelement <4 x float> poison, float %ld, i32 0
+  %b = shufflevector <4 x float> %ins, <4 x float> poison, <4 x i32> zeroinitializer
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.128(<4 x i32> %A, <4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2bf8s_256(<8 x i32> %A, <8 x float> %b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8s_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbiasps2bf8s %ymm1, %ymm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x3b,0xd1]
+; X64-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8s_256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2bf8s %ymm1, %ymm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x3b,0xd1]
+; X86-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.256(<8 x i32> %A, <8 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2bf8s_256(<8 x i32> %A, <8 x float> %b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8s_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbiasps2bf8s %ymm1, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xa9,0x3b,0xc1]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8s_256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2bf8s %ymm1, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xa9,0x3b,0xc1]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.256(<8 x i32> %A, <8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2bf8s_mem_256(<8 x i32> %A, ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8s_mem_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtbiasps2bf8s (%rdi), %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x3b,0x0f]
+; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8s_mem_256:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtbiasps2bf8s (%eax), %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x3b,0x08]
+; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %b = load <8 x float>, ptr %ptr_b
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.256(<8 x i32> %A, <8 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2bf8s_bcst_256(<8 x i32> %A, ptr %ptr_b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8s_bcst_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtbiasps2bf8s (%rdi){1to8}, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xb9,0x3b,0x07]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8s_bcst_256:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtbiasps2bf8s (%eax){1to8}, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xb9,0x3b,0x00]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ld = load float, ptr %ptr_b
+  %ins = insertelement <8 x float> poison, float %ld, i32 0
+  %b = shufflevector <8 x float> %ins, <8 x float> poison, <8 x i32> zeroinitializer
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.256(<8 x i32> %A, <8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.128(<4 x i32>, <4 x float>, <16 x i8>, i8)
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.256(<8 x i32>, <8 x float>, <16 x i8>, i8)
+
+; Masked 128/256-bit forms of vcvtbiasps2hf8
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2hf8_128(<4 x i32> %A, <4 x float> %b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbiasps2hf8 %xmm1, %xmm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x38,0xd1]
+; X64-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8_128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2hf8 %xmm1, %xmm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x38,0xd1]
+; X86-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.128(<4 x i32> %A, <4 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2hf8_128(<4 x i32> %A, <4 x float> %b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbiasps2hf8 %xmm1, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x38,0xc1]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8_128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2hf8 %xmm1, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x38,0xc1]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.128(<4 x i32> %A, <4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2hf8_mem_128(<4 x i32> %A, ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8_mem_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtbiasps2hf8 (%rdi), %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x38,0x0f]
+; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8_mem_128:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtbiasps2hf8 (%eax), %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x38,0x08]
+; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %b = load <4 x float>, ptr %ptr_b
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.128(<4 x i32> %A, <4 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2hf8_bcst_128(<4 x i32> %A, ptr %ptr_b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8_bcst_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtbiasps2hf8 (%rdi){1to4}, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x99,0x38,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8_bcst_128:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtbiasps2hf8 (%eax){1to4}, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x99,0x38,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ld = load float, ptr %ptr_b
+  %ins = insertelement <4 x float> poison, float %ld, i32 0
+  %b = shufflevector <4 x float> %ins, <4 x float> poison, <4 x i32> zeroinitializer
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.128(<4 x i32> %A, <4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2hf8_256(<8 x i32> %A, <8 x float> %b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbiasps2hf8 %ymm1, %ymm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x38,0xd1]
+; X64-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8_256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2hf8 %ymm1, %ymm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x38,0xd1]
+; X86-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.256(<8 x i32> %A, <8 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2hf8_256(<8 x i32> %A, <8 x float> %b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbiasps2hf8 %ymm1, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xa9,0x38,0xc1]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8_256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2hf8 %ymm1, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xa9,0x38,0xc1]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.256(<8 x i32> %A, <8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2hf8_mem_256(<8 x i32> %A, ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8_mem_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtbiasps2hf8 (%rdi), %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x38,0x0f]
+; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8_mem_256:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtbiasps2hf8 (%eax), %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x38,0x08]
+; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %b = load <8 x float>, ptr %ptr_b
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.256(<8 x i32> %A, <8 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2hf8_bcst_256(<8 x i32> %A, ptr %ptr_b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8_bcst_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtbiasps2hf8 (%rdi){1to8}, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xb9,0x38,0x07]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8_bcst_256:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtbiasps2hf8 (%eax){1to8}, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xb9,0x38,0x00]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ld = load float, ptr %ptr_b
+  %ins = insertelement <8 x float> poison, float %ld, i32 0
+  %b = shufflevector <8 x float> %ins, <8 x float> poison, <8 x i32> zeroinitializer
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.256(<8 x i32> %A, <8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.128(<4 x i32>, <4 x float>, <16 x i8>, i8)
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.256(<8 x i32>, <8 x float>, <16 x i8>, i8)
+
+; Masked 128/256-bit forms of vcvtbiasps2hf8s
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2hf8s_128(<4 x i32> %A, <4 x float> %b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8s_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbiasps2hf8s %xmm1, %xmm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x3a,0xd1]
+; X64-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8s_128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2hf8s %xmm1, %xmm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x3a,0xd1]
+; X86-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.128(<4 x i32> %A, <4 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2hf8s_128(<4 x i32> %A, <4 x float> %b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8s_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbiasps2hf8s %xmm1, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x3a,0xc1]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8s_128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2hf8s %xmm1, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x3a,0xc1]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.128(<4 x i32> %A, <4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2hf8s_mem_128(<4 x i32> %A, ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8s_mem_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtbiasps2hf8s (%rdi), %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x3a,0x0f]
+; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8s_mem_128:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtbiasps2hf8s (%eax), %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x3a,0x08]
+; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %b = load <4 x float>, ptr %ptr_b
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.128(<4 x i32> %A, <4 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2hf8s_bcst_128(<4 x i32> %A, ptr %ptr_b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8s_bcst_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtbiasps2hf8s (%rdi){1to4}, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x99,0x3a,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8s_bcst_128:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtbiasps2hf8s (%eax){1to4}, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x99,0x3a,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ld = load float, ptr %ptr_b
+  %ins = insertelement <4 x float> poison, float %ld, i32 0
+  %b = shufflevector <4 x float> %ins, <4 x float> poison, <4 x i32> zeroinitializer
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.128(<4 x i32> %A, <4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2hf8s_256(<8 x i32> %A, <8 x float> %b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8s_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbiasps2hf8s %ymm1, %ymm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x3a,0xd1]
+; X64-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8s_256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2hf8s %ymm1, %ymm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x3a,0xd1]
+; X86-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.256(<8 x i32> %A, <8 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2hf8s_256(<8 x i32> %A, <8 x float> %b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8s_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbiasps2hf8s %ymm1, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xa9,0x3a,0xc1]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8s_256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2hf8s %ymm1, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xa9,0x3a,0xc1]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.256(<8 x i32> %A, <8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2hf8s_mem_256(<8 x i32> %A, ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8s_mem_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtbiasps2hf8s (%rdi), %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x3a,0x0f]
+; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8s_mem_256:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtbiasps2hf8s (%eax), %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x3a,0x08]
+; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %b = load <8 x float>, ptr %ptr_b
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.256(<8 x i32> %A, <8 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2hf8s_bcst_256(<8 x i32> %A, ptr %ptr_b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8s_bcst_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtbiasps2hf8s (%rdi){1to8}, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xb9,0x3a,0x07]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8s_bcst_256:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtbiasps2hf8s (%eax){1to8}, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xb9,0x3a,0x00]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ld = load float, ptr %ptr_b
+  %ins = insertelement <8 x float> poison, float %ld, i32 0
+  %b = shufflevector <8 x float> %ins, <8 x float> poison, <8 x i32> zeroinitializer
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.256(<8 x i32> %A, <8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.128(<4 x i32>, <4 x float>, <16 x i8>, i8)
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.256(<8 x i32>, <8 x float>, <16 x i8>, i8)

>From b3ee24821fb72e1aa87f95c88b90e7fae96e40a1 Mon Sep 17 00:00:00 2001
From: Ganesh Gopalasubramanian <Ganesh.Gopalasubramanian at amd.com>
Date: Tue, 25 Aug 2026 06:05:24 +0530
Subject: [PATCH 09/16] [X86][AVX10_V2_AUX] Use poison instead of undef in the
 scalar_to_vector test

---
 llvm/test/CodeGen/X86/avx10_v2aux-intrinsics.ll | 2 +-
 1 file changed, 1 insertion(+), 1 deletion(-)

diff --git a/llvm/test/CodeGen/X86/avx10_v2aux-intrinsics.ll b/llvm/test/CodeGen/X86/avx10_v2aux-intrinsics.ll
index b697cb7cec2727..7a78236b8e8c69 100644
--- a/llvm/test/CodeGen/X86/avx10_v2aux-intrinsics.ll
+++ b/llvm/test/CodeGen/X86/avx10_v2aux-intrinsics.ll
@@ -1505,7 +1505,7 @@ define <16 x i8> @test_int_x86_avx10_vcvtbf42hf8_s2v_128(ptr %ptr_a) {
 ; X86-NEXT:    vcvtbf42hf8 (%eax), %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x37,0x00]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %l = load i64, ptr %ptr_a, align 8
-  %v = insertelement <2 x i64> undef, i64 %l, i64 0
+  %v = insertelement <2 x i64> poison, i64 %l, i64 0
   %a = bitcast <2 x i64> %v to <16 x i8>
   %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf42hf8.128(<16 x i8> %a)
   ret <16 x i8> %ret

>From 7aa3acc35447fce925b398a89f634855bbcbd74d Mon Sep 17 00:00:00 2001
From: Ganesh Gopalasubramanian <Ganesh.Gopalasubramanian at amd.com>
Date: Tue, 25 Aug 2026 06:29:59 +0530
Subject: [PATCH 10/16] [X86][AVX10_V2_AUX] Fix clang-format spacing in the
 _MM_UNPACKB_* macros

---
 clang/lib/Headers/avx10_2_v2auxintrin.h | 4 ++--
 1 file changed, 2 insertions(+), 2 deletions(-)

diff --git a/clang/lib/Headers/avx10_2_v2auxintrin.h b/clang/lib/Headers/avx10_2_v2auxintrin.h
index 425edab16e5eb1..47fd5aedd74c9e 100644
--- a/clang/lib/Headers/avx10_2_v2auxintrin.h
+++ b/clang/lib/Headers/avx10_2_v2auxintrin.h
@@ -2104,7 +2104,7 @@ _mm256_maskz_cvthf6_hf8(__mmask32 __U, __m256i __A) {
 ///    The packed element size in bits, in the range [2, 7].
 /// \returns
 ///    Bits [4:2] of the immediate operand.
-#define _MM_UNPACKB_SIZE(n) (((n)&0x7) << 2)
+#define _MM_UNPACKB_SIZE(n) (((n) & 0x7) << 2)
 
 /// Compose the \c start field of the immediate operand of \c VUNPACKB, which
 ///    selects which block of packed elements is extracted from the source.
@@ -2116,7 +2116,7 @@ _mm256_maskz_cvthf6_hf8(__mmask32 __U, __m256i __A) {
 ///    the element size; only offsets that allow a full extraction are valid.
 /// \returns
 ///    Bits [1:0] of the immediate operand.
-#define _MM_UNPACKB_START(s) (((s)&0x3) << 0)
+#define _MM_UNPACKB_START(s) (((s) & 0x3) << 0)
 
 /// The \c sign \c ext field of the immediate operand of \c VUNPACKB,
 ///    requesting that unpacked elements be sign-extended to 8 bits instead of

>From 01656964c69161f3247773de81b88d305b5560b5 Mon Sep 17 00:00:00 2001
From: Ganesh Gopalasubramanian <Ganesh.Gopalasubramanian at amd.com>
Date: Tue, 1 Sep 2026 00:21:02 +0530
Subject: [PATCH 11/16] [X86][AVX10_V2_AUX] Address review comments

Rename _mm{,256,512}_cvtssepi32_epi8 and their mask, maskz and storeu
forms to _mm{,256,512}_cvtss_epi32_epi8, matching GCC.

Drop the underscore before the width in the LLVM intrinsic records for
the convert families, so int_x86_avx10_vcvtps2bf8_128 becomes
int_x86_avx10_vcvtps2bf8128.

Remove the vcvt{bf8,hf8}2bf4s_*_mem builtins, which were never exposed
through the headers. The store folding they were meant to cover is
handled by patterns instead.

Extend the MC tests with the encodings that were missing, and add the
vpmovssdb store folding tests.
---
 clang/include/clang/Basic/BuiltinsX86.td      |   26 -
 clang/lib/Headers/avx10_2_512v2auxintrin.h    |   72 +-
 clang/lib/Headers/avx10_2_v2auxintrin.h       |  148 +-
 .../test/CodeGen/X86/avx10_2_v2aux-builtins.c |  567 ++---
 llvm/include/llvm/IR/IntrinsicsX86.td         |  176 +-
 llvm/lib/Target/X86/X86InstrAVX10_V2_AUX.td   |   43 +-
 llvm/lib/Target/X86/X86IntrinsicsInfo.h       |  124 +-
 .../CodeGen/X86/avx10_v2aux-intrinsics.ll     | 2091 ++++++++++-------
 .../MC/Disassembler/X86/avx10_v2_aux-32.txt   |  120 +
 .../MC/Disassembler/X86/avx10_v2_aux-64.txt   |  120 +
 llvm/test/MC/X86/avx10_v2_aux-att-32.s        |  142 +-
 llvm/test/MC/X86/avx10_v2_aux-att-64.s        |  142 +-
 llvm/test/MC/X86/avx10_v2_aux-intel-32.s      |  142 +-
 llvm/test/MC/X86/avx10_v2_aux-intel-64.s      |  142 +-
 14 files changed, 2217 insertions(+), 1838 deletions(-)

diff --git a/clang/include/clang/Basic/BuiltinsX86.td b/clang/include/clang/Basic/BuiltinsX86.td
index 79282cfc11bcf7..e4b7702c8b7847 100644
--- a/clang/include/clang/Basic/BuiltinsX86.td
+++ b/clang/include/clang/Basic/BuiltinsX86.td
@@ -5328,19 +5328,6 @@ let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in
   def vcvtbf82bf4s_512 : X86Builtin<"_Vector<32, char>(_Vector<64, char>)">;
 }
 
-// VCVTBF82BF4S memory store
-let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
-  def vcvtbf82bf4s_128_mem : X86Builtin<"void(void *, _Vector<16, char>)">;
-}
-
-let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
-  def vcvtbf82bf4s_256_mem : X86Builtin<"void(void *, _Vector<32, char>)">;
-}
-
-let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
-  def vcvtbf82bf4s_512_mem : X86Builtin<"void(void *, _Vector<64, char>)">;
-}
-
 // VCVTHF82BF4S
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
   def vcvthf82bf4s_128 : X86Builtin<"_Vector<16, char>(_Vector<16, char>)">;
@@ -5354,19 +5341,6 @@ let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in
   def vcvthf82bf4s_512 : X86Builtin<"_Vector<32, char>(_Vector<64, char>)">;
 }
 
-// VCVTHF82BF4S memory store
-let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
-  def vcvthf82bf4s_128_mem : X86Builtin<"void(void *, _Vector<16, char>)">;
-}
-
-let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
-  def vcvthf82bf4s_256_mem : X86Builtin<"void(void *, _Vector<32, char>)">;
-}
-
-let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
-  def vcvthf82bf4s_512_mem : X86Builtin<"void(void *, _Vector<64, char>)">;
-}
-
 // Unpack to Byte
 
 let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
diff --git a/clang/lib/Headers/avx10_2_512v2auxintrin.h b/clang/lib/Headers/avx10_2_512v2auxintrin.h
index 21898966126fe2..d83be98bffe82f 100644
--- a/clang/lib/Headers/avx10_2_512v2auxintrin.h
+++ b/clang/lib/Headers/avx10_2_512v2auxintrin.h
@@ -773,7 +773,8 @@ _mm512_maskz_cvthf8_ps(__mmask16 __U, __m128i __A) {
 /// \param __A
 ///    A 512-bit vector of [64 x i8] containing BF8 values.
 /// \returns
-///    A 512-bit vector of [64 x i8] containing the converted BF6 values.
+///    A 512-bit vector containing 64 converted packed BF6 values in the
+///    lower bits.
 static __inline__ __m512i __DEFAULT_FN_ATTRS512
 _mm512_cvts_bf8_bf6(__m512i __A) {
   return (__m512i)__builtin_ia32_vcvtbf82bf6s_512((__v64qi)__A);
@@ -790,7 +791,8 @@ _mm512_cvts_bf8_bf6(__m512i __A) {
 /// \param __A
 ///    A 512-bit vector of [64 x i8] containing HF8 values.
 /// \returns
-///    A 512-bit vector of [64 x i8] containing the converted HF6 values.
+///    A 512-bit vector containing 64 converted packed HF6 values in the
+///    lower bits.
 static __inline__ __m512i __DEFAULT_FN_ATTRS512
 _mm512_cvts_hf8_hf6(__m512i __A) {
   return (__m512i)__builtin_ia32_vcvthf82hf6s_512((__v64qi)__A);
@@ -807,7 +809,7 @@ _mm512_cvts_hf8_hf6(__m512i __A) {
 /// \param __A
 ///    A 512-bit vector of [64 x i8] containing BF8 values.
 /// \returns
-///    A 256-bit vector of [32 x i8] containing the converted BF4 values.
+///    A 256-bit vector containing 64 converted packed BF4 values.
 static __inline__ __m256i __DEFAULT_FN_ATTRS512
 _mm512_cvts_bf8_bf4(__m512i __A) {
   return (__m256i)__builtin_ia32_vcvtbf82bf4s_512((__v64qi)__A);
@@ -824,48 +826,12 @@ _mm512_cvts_bf8_bf4(__m512i __A) {
 /// \param __A
 ///    A 512-bit vector of [64 x i8] containing HF8 values.
 /// \returns
-///    A 256-bit vector of [32 x i8] containing the converted BF4 values.
+///    A 256-bit vector containing 64 converted packed BF4 values.
 static __inline__ __m256i __DEFAULT_FN_ATTRS512
 _mm512_cvts_hf8_bf4(__m512i __A) {
   return (__m256i)__builtin_ia32_vcvthf82bf4s_512((__v64qi)__A);
 }
 
-/// Convert packed BF8 (8-bit) floating-point elements in \a __A to packed
-///    BF4 (4-bit) floating-point elements with saturation, and store the
-///    results to memory at \a __P.
-///
-/// \headerfile <immintrin.h>
-///
-/// This intrinsic corresponds to the \c VCVTBF82BF4S instruction.
-///
-/// \param __P
-///    A pointer to a 256-bit memory location. The address does not need to be
-///    aligned.
-/// \param __A
-///    A 512-bit vector of [64 x i8] containing BF8 values.
-static __inline__ void __DEFAULT_FN_ATTRS512
-_mm512_cvts_bf8_bf4_storeu(void *__P, __m512i __A) {
-  __builtin_ia32_vcvtbf82bf4s_512_mem(__P, (__v64qi)__A);
-}
-
-/// Convert packed HF8 (8-bit) floating-point elements in \a __A to packed
-///    BF4 (4-bit) floating-point elements with saturation, and store the
-///    results to memory at \a __P.
-///
-/// \headerfile <immintrin.h>
-///
-/// This intrinsic corresponds to the \c VCVTHF82BF4S instruction.
-///
-/// \param __P
-///    A pointer to a 256-bit memory location. The address does not need to be
-///    aligned.
-/// \param __A
-///    A 512-bit vector of [64 x i8] containing HF8 values.
-static __inline__ void __DEFAULT_FN_ATTRS512
-_mm512_cvts_hf8_bf4_storeu(void *__P, __m512i __A) {
-  __builtin_ia32_vcvthf82bf4s_512_mem(__P, (__v64qi)__A);
-}
-
 /// Convert packed BF4 (4-bit) floating-point elements in \a __A to packed
 ///    HF8 (8-bit) floating-point elements, and store the results in a 512-bit
 ///    vector.
@@ -875,7 +841,7 @@ _mm512_cvts_hf8_bf4_storeu(void *__P, __m512i __A) {
 /// This intrinsic corresponds to the \c VCVTBF42HF8 instruction.
 ///
 /// \param __A
-///    A 256-bit vector of [32 x i8] containing BF4 values.
+///    A 256-bit vector containing 64 packed BF4 values.
 /// \returns
 ///    A 512-bit vector of [64 x i8] containing the converted HF8 values.
 static __inline__ __m512i __DEFAULT_FN_ATTRS512 _mm512_cvtbf4_hf8(__m256i __A) {
@@ -895,7 +861,7 @@ static __inline__ __m512i __DEFAULT_FN_ATTRS512 _mm512_cvtbf4_hf8(__m256i __A) {
 /// \param __U
 ///    A 64-bit mask indicating which elements to write.
 /// \param __A
-///    A 256-bit vector of [32 x i8] containing BF4 values.
+///    A 256-bit vector containing 64 packed BF4 values.
 /// \returns
 ///    A 512-bit vector of [64 x i8] containing the converted HF8 values.
 static __inline__ __m512i __DEFAULT_FN_ATTRS512
@@ -915,7 +881,7 @@ _mm512_mask_cvtbf4_hf8(__m512i __W, __mmask64 __U, __m256i __A) {
 /// \param __U
 ///    A 64-bit mask indicating which elements to write (zero otherwise).
 /// \param __A
-///    A 256-bit vector of [32 x i8] containing BF4 values.
+///    A 256-bit vector containing 64 packed BF4 values.
 /// \returns
 ///    A 512-bit vector of [64 x i8] containing the converted HF8 values.
 static __inline__ __m512i __DEFAULT_FN_ATTRS512
@@ -933,7 +899,7 @@ _mm512_maskz_cvtbf4_hf8(__mmask64 __U, __m256i __A) {
 /// This intrinsic corresponds to the \c VCVTBF62HF8 instruction.
 ///
 /// \param __A
-///    A 512-bit vector of [64 x i8] containing BF6 values.
+///    A 512-bit vector containing 64 packed BF6 values in the lower bits.
 /// \returns
 ///    A 512-bit vector of [64 x i8] containing the converted HF8 values.
 static __inline__ __m512i __DEFAULT_FN_ATTRS512 _mm512_cvtbf6_hf8(__m512i __A) {
@@ -953,7 +919,7 @@ static __inline__ __m512i __DEFAULT_FN_ATTRS512 _mm512_cvtbf6_hf8(__m512i __A) {
 /// \param __U
 ///    A 64-bit mask indicating which elements to write.
 /// \param __A
-///    A 512-bit vector of [64 x i8] containing BF6 values.
+///    A 512-bit vector containing 64 packed BF6 values in the lower bits.
 /// \returns
 ///    A 512-bit vector of [64 x i8] containing the converted HF8 values.
 static __inline__ __m512i __DEFAULT_FN_ATTRS512
@@ -973,7 +939,7 @@ _mm512_mask_cvtbf6_hf8(__m512i __W, __mmask64 __U, __m512i __A) {
 /// \param __U
 ///    A 64-bit mask indicating which elements to write (zero otherwise).
 /// \param __A
-///    A 512-bit vector of [64 x i8] containing BF6 values.
+///    A 512-bit vector containing 64 packed BF6 values in the lower bits.
 /// \returns
 ///    A 512-bit vector of [64 x i8] containing the converted HF8 values.
 static __inline__ __m512i __DEFAULT_FN_ATTRS512
@@ -991,7 +957,7 @@ _mm512_maskz_cvtbf6_hf8(__mmask64 __U, __m512i __A) {
 /// This intrinsic corresponds to the \c VCVTHF62HF8 instruction.
 ///
 /// \param __A
-///    A 512-bit vector of [64 x i8] containing HF6 values.
+///    A 512-bit vector containing 64 packed HF6 values in the lower bits.
 /// \returns
 ///    A 512-bit vector of [64 x i8] containing the converted HF8 values.
 static __inline__ __m512i __DEFAULT_FN_ATTRS512 _mm512_cvthf6_hf8(__m512i __A) {
@@ -1011,7 +977,7 @@ static __inline__ __m512i __DEFAULT_FN_ATTRS512 _mm512_cvthf6_hf8(__m512i __A) {
 /// \param __U
 ///    A 64-bit mask indicating which elements to write.
 /// \param __A
-///    A 512-bit vector of [64 x i8] containing HF6 values.
+///    A 512-bit vector containing 64 packed HF6 values in the lower bits.
 /// \returns
 ///    A 512-bit vector of [64 x i8] containing the converted HF8 values.
 static __inline__ __m512i __DEFAULT_FN_ATTRS512
@@ -1031,7 +997,7 @@ _mm512_mask_cvthf6_hf8(__m512i __W, __mmask64 __U, __m512i __A) {
 /// \param __U
 ///    A 64-bit mask indicating which elements to write (zero otherwise).
 /// \param __A
-///    A 512-bit vector of [64 x i8] containing HF6 values.
+///    A 512-bit vector containing 64 packed HF6 values in the lower bits.
 /// \returns
 ///    A 512-bit vector of [64 x i8] containing the converted HF8 values.
 static __inline__ __m512i __DEFAULT_FN_ATTRS512
@@ -1111,7 +1077,7 @@ _mm512_maskz_cvthf6_hf8(__mmask64 __U, __m512i __A) {
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS512
-_mm512_cvtssepi32_epi8(__m512i __A) {
+_mm512_cvtss_epi32_epi8(__m512i __A) {
   return (__m128i)__builtin_ia32_vpmovssdb512_mask(
       (__v16si)__A, (__v16qi)_mm_setzero_si128(), (__mmask16)-1);
 }
@@ -1132,7 +1098,7 @@ _mm512_cvtssepi32_epi8(__m512i __A) {
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS512
-_mm512_mask_cvtssepi32_epi8(__m128i __W, __mmask16 __U, __m512i __A) {
+_mm512_mask_cvtss_epi32_epi8(__m128i __W, __mmask16 __U, __m512i __A) {
   return (__m128i)__builtin_ia32_vpmovssdb512_mask((__v16si)__A, (__v16qi)__W,
                                                    __U);
 }
@@ -1151,7 +1117,7 @@ _mm512_mask_cvtssepi32_epi8(__m128i __W, __mmask16 __U, __m512i __A) {
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS512
-_mm512_maskz_cvtssepi32_epi8(__mmask16 __U, __m512i __A) {
+_mm512_maskz_cvtss_epi32_epi8(__mmask16 __U, __m512i __A) {
   return (__m128i)__builtin_ia32_vpmovssdb512_mask(
       (__v16si)__A, (__v16qi)_mm_setzero_si128(), __U);
 }
@@ -1171,7 +1137,7 @@ _mm512_maskz_cvtssepi32_epi8(__mmask16 __U, __m512i __A) {
 /// \param __A
 ///    A 512-bit vector of [16 x i32].
 static __inline__ void __DEFAULT_FN_ATTRS512
-_mm512_mask_cvtssepi32_storeu_epi8(void *__P, __mmask16 __M, __m512i __A) {
+_mm512_mask_cvtss_epi32_storeu_epi8(void *__P, __mmask16 __M, __m512i __A) {
   __builtin_ia32_vpmovssdb512mem_mask((__v16qi *)__P, (__v16si)__A, __M);
 }
 
diff --git a/clang/lib/Headers/avx10_2_v2auxintrin.h b/clang/lib/Headers/avx10_2_v2auxintrin.h
index 47fd5aedd74c9e..fe9fe8068e94d4 100644
--- a/clang/lib/Headers/avx10_2_v2auxintrin.h
+++ b/clang/lib/Headers/avx10_2_v2auxintrin.h
@@ -1552,7 +1552,8 @@ _mm256_maskz_cvthf8_ps(__mmask8 __U, __m128i __A) {
 /// \param __A
 ///    A 128-bit vector of [16 x i8] containing BF8 values.
 /// \returns
-///    A 128-bit vector of [16 x i8] containing the converted BF6 values.
+///    A 128-bit vector containing 16 converted packed BF6 values in the
+///    lower bits.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvts_bf8_bf6(__m128i __A) {
   return (__m128i)__builtin_ia32_vcvtbf82bf6s_128((__v16qi)__A);
 }
@@ -1568,7 +1569,8 @@ static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvts_bf8_bf6(__m128i __A) {
 /// \param __A
 ///    A 256-bit vector of [32 x i8] containing BF8 values.
 /// \returns
-///    A 256-bit vector of [32 x i8] containing the converted BF6 values.
+///    A 256-bit vector containing 32 converted packed BF6 values in the
+///    lower bits.
 static __inline__ __m256i __DEFAULT_FN_ATTRS256
 _mm256_cvts_bf8_bf6(__m256i __A) {
   return (__m256i)__builtin_ia32_vcvtbf82bf6s_256((__v32qi)__A);
@@ -1585,7 +1587,8 @@ _mm256_cvts_bf8_bf6(__m256i __A) {
 /// \param __A
 ///    A 128-bit vector of [16 x i8] containing HF8 values.
 /// \returns
-///    A 128-bit vector of [16 x i8] containing the converted HF6 values.
+///    A 128-bit vector containing 16 converted packed HF6 values in the
+///    lower bits.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvts_hf8_hf6(__m128i __A) {
   return (__m128i)__builtin_ia32_vcvthf82hf6s_128((__v16qi)__A);
 }
@@ -1601,7 +1604,8 @@ static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvts_hf8_hf6(__m128i __A) {
 /// \param __A
 ///    A 256-bit vector of [32 x i8] containing HF8 values.
 /// \returns
-///    A 256-bit vector of [32 x i8] containing the converted HF6 values.
+///    A 256-bit vector containing 32 converted packed HF6 values in the
+///    lower bits.
 static __inline__ __m256i __DEFAULT_FN_ATTRS256
 _mm256_cvts_hf8_hf6(__m256i __A) {
   return (__m256i)__builtin_ia32_vcvthf82hf6s_256((__v32qi)__A);
@@ -1618,8 +1622,8 @@ _mm256_cvts_hf8_hf6(__m256i __A) {
 /// \param __A
 ///    A 128-bit vector of [16 x i8] containing BF8 values.
 /// \returns
-///    A 128-bit vector of [16 x i8] containing the converted BF4 values
-///    (lower 8 bytes used).
+///    A 128-bit vector containing 16 converted packed BF4 values in the
+///    lower bits.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvts_bf8_bf4(__m128i __A) {
   return (__m128i)__builtin_ia32_vcvtbf82bf4s_128((__v16qi)__A);
 }
@@ -1635,7 +1639,7 @@ static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvts_bf8_bf4(__m128i __A) {
 /// \param __A
 ///    A 256-bit vector of [32 x i8] containing BF8 values.
 /// \returns
-///    A 128-bit vector of [16 x i8] containing the converted BF4 values.
+///    A 128-bit vector containing 32 converted packed BF4 values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_cvts_bf8_bf4(__m256i __A) {
   return (__m128i)__builtin_ia32_vcvtbf82bf4s_256((__v32qi)__A);
@@ -1652,8 +1656,8 @@ _mm256_cvts_bf8_bf4(__m256i __A) {
 /// \param __A
 ///    A 128-bit vector of [16 x i8] containing HF8 values.
 /// \returns
-///    A 128-bit vector of [16 x i8] containing the converted BF4 values
-///    (lower 8 bytes used).
+///    A 128-bit vector containing 16 converted packed BF4 values in the
+///    lower bits.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvts_hf8_bf4(__m128i __A) {
   return (__m128i)__builtin_ia32_vcvthf82bf4s_128((__v16qi)__A);
 }
@@ -1669,84 +1673,12 @@ static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvts_hf8_bf4(__m128i __A) {
 /// \param __A
 ///    A 256-bit vector of [32 x i8] containing HF8 values.
 /// \returns
-///    A 128-bit vector of [16 x i8] containing the converted BF4 values.
+///    A 128-bit vector containing 32 converted packed BF4 values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_cvts_hf8_bf4(__m256i __A) {
   return (__m128i)__builtin_ia32_vcvthf82bf4s_256((__v32qi)__A);
 }
 
-/// Convert packed BF8 (8-bit) floating-point elements in \a __A to packed
-///    BF4 (4-bit) floating-point elements with saturation, and store the
-///    results to memory at \a __P.
-///
-/// \headerfile <immintrin.h>
-///
-/// This intrinsic corresponds to the \c VCVTBF82BF4S instruction.
-///
-/// \param __P
-///    A pointer to a 64-bit memory location. The address does not need to be
-///    aligned.
-/// \param __A
-///    A 128-bit vector of [16 x i8] containing BF8 values.
-static __inline__ void __DEFAULT_FN_ATTRS128
-_mm_cvts_bf8_bf4_storeu(void *__P, __m128i __A) {
-  __builtin_ia32_vcvtbf82bf4s_128_mem(__P, (__v16qi)__A);
-}
-
-/// Convert packed BF8 (8-bit) floating-point elements in \a __A to packed
-///    BF4 (4-bit) floating-point elements with saturation, and store the
-///    results to memory at \a __P.
-///
-/// \headerfile <immintrin.h>
-///
-/// This intrinsic corresponds to the \c VCVTBF82BF4S instruction.
-///
-/// \param __P
-///    A pointer to a 128-bit memory location. The address does not need to be
-///    aligned.
-/// \param __A
-///    A 256-bit vector of [32 x i8] containing BF8 values.
-static __inline__ void __DEFAULT_FN_ATTRS256
-_mm256_cvts_bf8_bf4_storeu(void *__P, __m256i __A) {
-  __builtin_ia32_vcvtbf82bf4s_256_mem(__P, (__v32qi)__A);
-}
-
-/// Convert packed HF8 (8-bit) floating-point elements in \a __A to packed
-///    BF4 (4-bit) floating-point elements with saturation, and store the
-///    results to memory at \a __P.
-///
-/// \headerfile <immintrin.h>
-///
-/// This intrinsic corresponds to the \c VCVTHF82BF4S instruction.
-///
-/// \param __P
-///    A pointer to a 64-bit memory location. The address does not need to be
-///    aligned.
-/// \param __A
-///    A 128-bit vector of [16 x i8] containing HF8 values.
-static __inline__ void __DEFAULT_FN_ATTRS128
-_mm_cvts_hf8_bf4_storeu(void *__P, __m128i __A) {
-  __builtin_ia32_vcvthf82bf4s_128_mem(__P, (__v16qi)__A);
-}
-
-/// Convert packed HF8 (8-bit) floating-point elements in \a __A to packed
-///    BF4 (4-bit) floating-point elements with saturation, and store the
-///    results to memory at \a __P.
-///
-/// \headerfile <immintrin.h>
-///
-/// This intrinsic corresponds to the \c VCVTHF82BF4S instruction.
-///
-/// \param __P
-///    A pointer to a 128-bit memory location. The address does not need to be
-///    aligned.
-/// \param __A
-///    A 256-bit vector of [32 x i8] containing HF8 values.
-static __inline__ void __DEFAULT_FN_ATTRS256
-_mm256_cvts_hf8_bf4_storeu(void *__P, __m256i __A) {
-  __builtin_ia32_vcvthf82bf4s_256_mem(__P, (__v32qi)__A);
-}
-
 /// Convert packed BF4 (4-bit) floating-point elements in \a __A to packed
 ///    HF8 (8-bit) floating-point elements, and store the results in a 128-bit
 ///    vector.
@@ -1756,7 +1688,7 @@ _mm256_cvts_hf8_bf4_storeu(void *__P, __m256i __A) {
 /// This intrinsic corresponds to the \c VCVTBF42HF8 instruction.
 ///
 /// \param __A
-///    A 128-bit vector of [16 x i8] containing BF4 values.
+///    A 128-bit vector containing 16 packed BF4 values in the lower bits.
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted HF8 values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvtbf4_hf8(__m128i __A) {
@@ -1776,7 +1708,7 @@ static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvtbf4_hf8(__m128i __A) {
 /// \param __U
 ///    A 16-bit mask indicating which elements to write.
 /// \param __A
-///    A 128-bit vector of [16 x i8] containing BF4 values.
+///    A 128-bit vector containing 16 packed BF4 values in the lower bits.
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted HF8 values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
@@ -1796,7 +1728,7 @@ _mm_mask_cvtbf4_hf8(__m128i __W, __mmask16 __U, __m128i __A) {
 /// \param __U
 ///    A 16-bit mask indicating which elements to write (zero otherwise).
 /// \param __A
-///    A 128-bit vector of [16 x i8] containing BF4 values.
+///    A 128-bit vector containing 16 packed BF4 values in the lower bits.
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted HF8 values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
@@ -1814,7 +1746,7 @@ _mm_maskz_cvtbf4_hf8(__mmask16 __U, __m128i __A) {
 /// This intrinsic corresponds to the \c VCVTBF42HF8 instruction.
 ///
 /// \param __A
-///    A 128-bit vector of [16 x i8] containing BF4 values.
+///    A 128-bit vector containing 32 packed BF4 values.
 /// \returns
 ///    A 256-bit vector of [32 x i8] containing the converted HF8 values.
 static __inline__ __m256i __DEFAULT_FN_ATTRS256 _mm256_cvtbf4_hf8(__m128i __A) {
@@ -1834,7 +1766,7 @@ static __inline__ __m256i __DEFAULT_FN_ATTRS256 _mm256_cvtbf4_hf8(__m128i __A) {
 /// \param __U
 ///    A 32-bit mask indicating which elements to write.
 /// \param __A
-///    A 128-bit vector of [16 x i8] containing BF4 values.
+///    A 128-bit vector containing 32 packed BF4 values.
 /// \returns
 ///    A 256-bit vector of [32 x i8] containing the converted HF8 values.
 static __inline__ __m256i __DEFAULT_FN_ATTRS256
@@ -1854,7 +1786,7 @@ _mm256_mask_cvtbf4_hf8(__m256i __W, __mmask32 __U, __m128i __A) {
 /// \param __U
 ///    A 32-bit mask indicating which elements to write (zero otherwise).
 /// \param __A
-///    A 128-bit vector of [16 x i8] containing BF4 values.
+///    A 128-bit vector containing 32 packed BF4 values.
 /// \returns
 ///    A 256-bit vector of [32 x i8] containing the converted HF8 values.
 static __inline__ __m256i __DEFAULT_FN_ATTRS256
@@ -1872,7 +1804,7 @@ _mm256_maskz_cvtbf4_hf8(__mmask32 __U, __m128i __A) {
 /// This intrinsic corresponds to the \c VCVTBF62HF8 instruction.
 ///
 /// \param __A
-///    A 128-bit vector of [16 x i8] containing BF6 values.
+///    A 128-bit vector containing 16 packed BF6 values in the lower bits.
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted HF8 values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvtbf6_hf8(__m128i __A) {
@@ -1892,7 +1824,7 @@ static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvtbf6_hf8(__m128i __A) {
 /// \param __U
 ///    A 16-bit mask indicating which elements to write.
 /// \param __A
-///    A 128-bit vector of [16 x i8] containing BF6 values.
+///    A 128-bit vector containing 16 packed BF6 values in the lower bits.
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted HF8 values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
@@ -1912,7 +1844,7 @@ _mm_mask_cvtbf6_hf8(__m128i __W, __mmask16 __U, __m128i __A) {
 /// \param __U
 ///    A 16-bit mask indicating which elements to write (zero otherwise).
 /// \param __A
-///    A 128-bit vector of [16 x i8] containing BF6 values.
+///    A 128-bit vector containing 16 packed BF6 values in the lower bits.
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted HF8 values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
@@ -1930,7 +1862,7 @@ _mm_maskz_cvtbf6_hf8(__mmask16 __U, __m128i __A) {
 /// This intrinsic corresponds to the \c VCVTBF62HF8 instruction.
 ///
 /// \param __A
-///    A 256-bit vector of [32 x i8] containing BF6 values.
+///    A 256-bit vector containing 32 packed BF6 values in the lower bits.
 /// \returns
 ///    A 256-bit vector of [32 x i8] containing the converted HF8 values.
 static __inline__ __m256i __DEFAULT_FN_ATTRS256 _mm256_cvtbf6_hf8(__m256i __A) {
@@ -1950,7 +1882,7 @@ static __inline__ __m256i __DEFAULT_FN_ATTRS256 _mm256_cvtbf6_hf8(__m256i __A) {
 /// \param __U
 ///    A 32-bit mask indicating which elements to write.
 /// \param __A
-///    A 256-bit vector of [32 x i8] containing BF6 values.
+///    A 256-bit vector containing 32 packed BF6 values in the lower bits.
 /// \returns
 ///    A 256-bit vector of [32 x i8] containing the converted HF8 values.
 static __inline__ __m256i __DEFAULT_FN_ATTRS256
@@ -1970,7 +1902,7 @@ _mm256_mask_cvtbf6_hf8(__m256i __W, __mmask32 __U, __m256i __A) {
 /// \param __U
 ///    A 32-bit mask indicating which elements to write (zero otherwise).
 /// \param __A
-///    A 256-bit vector of [32 x i8] containing BF6 values.
+///    A 256-bit vector containing 32 packed BF6 values in the lower bits.
 /// \returns
 ///    A 256-bit vector of [32 x i8] containing the converted HF8 values.
 static __inline__ __m256i __DEFAULT_FN_ATTRS256
@@ -1988,7 +1920,7 @@ _mm256_maskz_cvtbf6_hf8(__mmask32 __U, __m256i __A) {
 /// This intrinsic corresponds to the \c VCVTHF62HF8 instruction.
 ///
 /// \param __A
-///    A 128-bit vector of [16 x i8] containing HF6 values.
+///    A 128-bit vector containing 16 packed HF6 values in the lower bits.
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted HF8 values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvthf6_hf8(__m128i __A) {
@@ -2008,7 +1940,7 @@ static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvthf6_hf8(__m128i __A) {
 /// \param __U
 ///    A 16-bit mask indicating which elements to write.
 /// \param __A
-///    A 128-bit vector of [16 x i8] containing HF6 values.
+///    A 128-bit vector containing 16 packed HF6 values in the lower bits.
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted HF8 values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
@@ -2028,7 +1960,7 @@ _mm_mask_cvthf6_hf8(__m128i __W, __mmask16 __U, __m128i __A) {
 /// \param __U
 ///    A 16-bit mask indicating which elements to write (zero otherwise).
 /// \param __A
-///    A 128-bit vector of [16 x i8] containing HF6 values.
+///    A 128-bit vector containing 16 packed HF6 values in the lower bits.
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted HF8 values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
@@ -2046,7 +1978,7 @@ _mm_maskz_cvthf6_hf8(__mmask16 __U, __m128i __A) {
 /// This intrinsic corresponds to the \c VCVTHF62HF8 instruction.
 ///
 /// \param __A
-///    A 256-bit vector of [32 x i8] containing HF6 values.
+///    A 256-bit vector containing 32 packed HF6 values in the lower bits.
 /// \returns
 ///    A 256-bit vector of [32 x i8] containing the converted HF8 values.
 static __inline__ __m256i __DEFAULT_FN_ATTRS256 _mm256_cvthf6_hf8(__m256i __A) {
@@ -2066,7 +1998,7 @@ static __inline__ __m256i __DEFAULT_FN_ATTRS256 _mm256_cvthf6_hf8(__m256i __A) {
 /// \param __U
 ///    A 32-bit mask indicating which elements to write.
 /// \param __A
-///    A 256-bit vector of [32 x i8] containing HF6 values.
+///    A 256-bit vector containing 32 packed HF6 values in the lower bits.
 /// \returns
 ///    A 256-bit vector of [32 x i8] containing the converted HF8 values.
 static __inline__ __m256i __DEFAULT_FN_ATTRS256
@@ -2086,7 +2018,7 @@ _mm256_mask_cvthf6_hf8(__m256i __W, __mmask32 __U, __m256i __A) {
 /// \param __U
 ///    A 32-bit mask indicating which elements to write (zero otherwise).
 /// \param __A
-///    A 256-bit vector of [32 x i8] containing HF6 values.
+///    A 256-bit vector containing 32 packed HF6 values in the lower bits.
 /// \returns
 ///    A 256-bit vector of [32 x i8] containing the converted HF8 values.
 static __inline__ __m256i __DEFAULT_FN_ATTRS256
@@ -2255,7 +2187,7 @@ _mm256_maskz_cvthf6_hf8(__mmask32 __U, __m256i __A) {
 ///    A 128-bit vector of [16 x i8]. The lower 4 bytes contain the converted
 ///    values; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
-_mm_cvtssepi32_epi8(__m128i __A) {
+_mm_cvtss_epi32_epi8(__m128i __A) {
   return (__m128i)__builtin_ia32_vpmovssdb128_mask(
       (__v4si)__A, (__v16qi)_mm_setzero_si128(), (__mmask8)-1);
 }
@@ -2276,7 +2208,7 @@ _mm_cvtssepi32_epi8(__m128i __A) {
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
-_mm_mask_cvtssepi32_epi8(__m128i __W, __mmask8 __U, __m128i __A) {
+_mm_mask_cvtss_epi32_epi8(__m128i __W, __mmask8 __U, __m128i __A) {
   return (__m128i)__builtin_ia32_vpmovssdb128_mask((__v4si)__A, (__v16qi)__W,
                                                    __U);
 }
@@ -2295,7 +2227,7 @@ _mm_mask_cvtssepi32_epi8(__m128i __W, __mmask8 __U, __m128i __A) {
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
-_mm_maskz_cvtssepi32_epi8(__mmask8 __U, __m128i __A) {
+_mm_maskz_cvtss_epi32_epi8(__mmask8 __U, __m128i __A) {
   return (__m128i)__builtin_ia32_vpmovssdb128_mask(
       (__v4si)__A, (__v16qi)_mm_setzero_si128(), __U);
 }
@@ -2314,7 +2246,7 @@ _mm_maskz_cvtssepi32_epi8(__mmask8 __U, __m128i __A) {
 ///    A 128-bit vector of [16 x i8]. The lower 8 bytes contain the converted
 ///    values; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
-_mm256_cvtssepi32_epi8(__m256i __A) {
+_mm256_cvtss_epi32_epi8(__m256i __A) {
   return (__m128i)__builtin_ia32_vpmovssdb256_mask(
       (__v8si)__A, (__v16qi)_mm_setzero_si128(), (__mmask8)-1);
 }
@@ -2335,7 +2267,7 @@ _mm256_cvtssepi32_epi8(__m256i __A) {
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
-_mm256_mask_cvtssepi32_epi8(__m128i __W, __mmask8 __U, __m256i __A) {
+_mm256_mask_cvtss_epi32_epi8(__m128i __W, __mmask8 __U, __m256i __A) {
   return (__m128i)__builtin_ia32_vpmovssdb256_mask((__v8si)__A, (__v16qi)__W,
                                                    __U);
 }
@@ -2354,7 +2286,7 @@ _mm256_mask_cvtssepi32_epi8(__m128i __W, __mmask8 __U, __m256i __A) {
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
-_mm256_maskz_cvtssepi32_epi8(__mmask8 __U, __m256i __A) {
+_mm256_maskz_cvtss_epi32_epi8(__mmask8 __U, __m256i __A) {
   return (__m128i)__builtin_ia32_vpmovssdb256_mask(
       (__v8si)__A, (__v16qi)_mm_setzero_si128(), __U);
 }
@@ -2374,7 +2306,7 @@ _mm256_maskz_cvtssepi32_epi8(__mmask8 __U, __m256i __A) {
 /// \param __A
 ///    A 128-bit vector of [4 x i32].
 static __inline__ void __DEFAULT_FN_ATTRS128
-_mm_mask_cvtssepi32_storeu_epi8(void *__P, __mmask8 __M, __m128i __A) {
+_mm_mask_cvtss_epi32_storeu_epi8(void *__P, __mmask8 __M, __m128i __A) {
   __builtin_ia32_vpmovssdb128mem_mask((__v16qi *)__P, (__v4si)__A, __M);
 }
 
@@ -2393,7 +2325,7 @@ _mm_mask_cvtssepi32_storeu_epi8(void *__P, __mmask8 __M, __m128i __A) {
 /// \param __A
 ///    A 256-bit vector of [8 x i32].
 static __inline__ void __DEFAULT_FN_ATTRS256
-_mm256_mask_cvtssepi32_storeu_epi8(void *__P, __mmask8 __M, __m256i __A) {
+_mm256_mask_cvtss_epi32_storeu_epi8(void *__P, __mmask8 __M, __m256i __A) {
   __builtin_ia32_vpmovssdb256mem_mask((__v16qi *)__P, (__v8si)__A, __M);
 }
 
diff --git a/clang/test/CodeGen/X86/avx10_2_v2aux-builtins.c b/clang/test/CodeGen/X86/avx10_2_v2aux-builtins.c
index ec1be83f941cf3..c19892ad2131fa 100644
--- a/clang/test/CodeGen/X86/avx10_2_v2aux-builtins.c
+++ b/clang/test/CodeGen/X86/avx10_2_v2aux-builtins.c
@@ -5,1165 +5,983 @@
 
 #include <immintrin.h>
 
-//
-// Convert from FP32 to FP8
-// VCVTPS2BF8 / VCVTPS2BF8S / VCVTPS2HF8 / VCVTPS2HF8S /
-// VCVTROPS2HF8 / VCVTROPS2HF8S
-//
-
-// VCVTPS2BF8 - 128-bit
-
 __m128i test_mm_cvtps_bf8(__m128 __A) {
   // CHECK-LABEL: @test_mm_cvtps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtps2bf8.128(<4 x float> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtps2bf8128(<4 x float> %{{.*}})
   return _mm_cvtps_bf8(__A);
 }
 
 __m128i test_mm_mask_cvtps_bf8(__m128i __W, __mmask8 __U, __m128 __A) {
   // CHECK-LABEL: @test_mm_mask_cvtps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.128(<4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8128(<4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm_mask_cvtps_bf8(__W, __U, __A);
 }
 
 __m128i test_mm_maskz_cvtps_bf8(__mmask8 __U, __m128 __A) {
   // CHECK-LABEL: @test_mm_maskz_cvtps_bf8(
   // CHECK: zeroinitializer
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.128(<4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8128(<4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm_maskz_cvtps_bf8(__U, __A);
 }
 
-// VCVTPS2BF8 - 256-bit
-
 __m128i test_mm256_cvtps_bf8(__m256 __A) {
   // CHECK-LABEL: @test_mm256_cvtps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtps2bf8.256(<8 x float> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtps2bf8256(<8 x float> %{{.*}})
   return _mm256_cvtps_bf8(__A);
 }
 
 __m128i test_mm256_mask_cvtps_bf8(__m128i __W, __mmask8 __U, __m256 __A) {
   // CHECK-LABEL: @test_mm256_mask_cvtps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.256(<8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8256(<8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm256_mask_cvtps_bf8(__W, __U, __A);
 }
 
 __m128i test_mm256_maskz_cvtps_bf8(__mmask8 __U, __m256 __A) {
   // CHECK-LABEL: @test_mm256_maskz_cvtps_bf8(
   // CHECK: zeroinitializer
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.256(<8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8256(<8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm256_maskz_cvtps_bf8(__U, __A);
 }
 
-// VCVTPS2BF8 - 512-bit
-
 __m128i test_mm512_cvtps_bf8(__m512 __A) {
   // CHECK-LABEL: @test_mm512_cvtps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtps2bf8.512(<16 x float> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtps2bf8512(<16 x float> %{{.*}})
   return _mm512_cvtps_bf8(__A);
 }
 
 __m128i test_mm512_mask_cvtps_bf8(__m128i __W, __mmask16 __U, __m512 __A) {
   // CHECK-LABEL: @test_mm512_mask_cvtps_bf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8.512(<16 x float> %{{.*}})
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8512(<16 x float> %{{.*}})
   // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm512_mask_cvtps_bf8(__W, __U, __A);
 }
 
 __m128i test_mm512_maskz_cvtps_bf8(__mmask16 __U, __m512 __A) {
   // CHECK-LABEL: @test_mm512_maskz_cvtps_bf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8.512(<16 x float> %{{.*}})
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8512(<16 x float> %{{.*}})
   // CHECK: zeroinitializer
   // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm512_maskz_cvtps_bf8(__U, __A);
 }
 
-// VCVTPS2BF8S - 128-bit
-
 __m128i test_mm_cvts_ps_bf8(__m128 __A) {
   // CHECK-LABEL: @test_mm_cvts_ps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtps2bf8s.128(<4 x float> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtps2bf8s128(<4 x float> %{{.*}})
   return _mm_cvts_ps_bf8(__A);
 }
 
 __m128i test_mm_mask_cvts_ps_bf8(__m128i __W, __mmask8 __U, __m128 __A) {
   // CHECK-LABEL: @test_mm_mask_cvts_ps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.128(<4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s128(<4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm_mask_cvts_ps_bf8(__W, __U, __A);
 }
 
 __m128i test_mm_maskz_cvts_ps_bf8(__mmask8 __U, __m128 __A) {
   // CHECK-LABEL: @test_mm_maskz_cvts_ps_bf8(
   // CHECK: zeroinitializer
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.128(<4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s128(<4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm_maskz_cvts_ps_bf8(__U, __A);
 }
 
-// VCVTPS2BF8S - 256-bit
-
 __m128i test_mm256_cvts_ps_bf8(__m256 __A) {
   // CHECK-LABEL: @test_mm256_cvts_ps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtps2bf8s.256(<8 x float> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtps2bf8s256(<8 x float> %{{.*}})
   return _mm256_cvts_ps_bf8(__A);
 }
 
 __m128i test_mm256_mask_cvts_ps_bf8(__m128i __W, __mmask8 __U, __m256 __A) {
   // CHECK-LABEL: @test_mm256_mask_cvts_ps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.256(<8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s256(<8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm256_mask_cvts_ps_bf8(__W, __U, __A);
 }
 
 __m128i test_mm256_maskz_cvts_ps_bf8(__mmask8 __U, __m256 __A) {
   // CHECK-LABEL: @test_mm256_maskz_cvts_ps_bf8(
   // CHECK: zeroinitializer
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.256(<8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s256(<8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm256_maskz_cvts_ps_bf8(__U, __A);
 }
 
-// VCVTPS2BF8S - 512-bit
-
 __m128i test_mm512_cvts_ps_bf8(__m512 __A) {
   // CHECK-LABEL: @test_mm512_cvts_ps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtps2bf8s.512(<16 x float> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtps2bf8s512(<16 x float> %{{.*}})
   return _mm512_cvts_ps_bf8(__A);
 }
 
 __m128i test_mm512_mask_cvts_ps_bf8(__m128i __W, __mmask16 __U, __m512 __A) {
   // CHECK-LABEL: @test_mm512_mask_cvts_ps_bf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8s.512(<16 x float> %{{.*}})
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8s512(<16 x float> %{{.*}})
   // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm512_mask_cvts_ps_bf8(__W, __U, __A);
 }
 
 __m128i test_mm512_maskz_cvts_ps_bf8(__mmask16 __U, __m512 __A) {
   // CHECK-LABEL: @test_mm512_maskz_cvts_ps_bf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8s.512(<16 x float> %{{.*}})
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8s512(<16 x float> %{{.*}})
   // CHECK: zeroinitializer
   // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm512_maskz_cvts_ps_bf8(__U, __A);
 }
 
-// VCVTPS2HF8 - 128-bit
-
 __m128i test_mm_cvtps_hf8(__m128 __A) {
   // CHECK-LABEL: @test_mm_cvtps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtps2hf8.128(<4 x float> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtps2hf8128(<4 x float> %{{.*}})
   return _mm_cvtps_hf8(__A);
 }
 
 __m128i test_mm_mask_cvtps_hf8(__m128i __W, __mmask8 __U, __m128 __A) {
   // CHECK-LABEL: @test_mm_mask_cvtps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.128(<4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8128(<4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm_mask_cvtps_hf8(__W, __U, __A);
 }
 
 __m128i test_mm_maskz_cvtps_hf8(__mmask8 __U, __m128 __A) {
   // CHECK-LABEL: @test_mm_maskz_cvtps_hf8(
   // CHECK: zeroinitializer
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.128(<4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8128(<4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm_maskz_cvtps_hf8(__U, __A);
 }
 
-// VCVTPS2HF8 - 256-bit
-
 __m128i test_mm256_cvtps_hf8(__m256 __A) {
   // CHECK-LABEL: @test_mm256_cvtps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtps2hf8.256(<8 x float> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtps2hf8256(<8 x float> %{{.*}})
   return _mm256_cvtps_hf8(__A);
 }
 
 __m128i test_mm256_mask_cvtps_hf8(__m128i __W, __mmask8 __U, __m256 __A) {
   // CHECK-LABEL: @test_mm256_mask_cvtps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.256(<8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8256(<8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm256_mask_cvtps_hf8(__W, __U, __A);
 }
 
 __m128i test_mm256_maskz_cvtps_hf8(__mmask8 __U, __m256 __A) {
   // CHECK-LABEL: @test_mm256_maskz_cvtps_hf8(
   // CHECK: zeroinitializer
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.256(<8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8256(<8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm256_maskz_cvtps_hf8(__U, __A);
 }
 
-// VCVTPS2HF8 - 512-bit
-
 __m128i test_mm512_cvtps_hf8(__m512 __A) {
   // CHECK-LABEL: @test_mm512_cvtps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtps2hf8.512(<16 x float> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtps2hf8512(<16 x float> %{{.*}})
   return _mm512_cvtps_hf8(__A);
 }
 
 __m128i test_mm512_mask_cvtps_hf8(__m128i __W, __mmask16 __U, __m512 __A) {
   // CHECK-LABEL: @test_mm512_mask_cvtps_hf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8.512(<16 x float> %{{.*}})
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8512(<16 x float> %{{.*}})
   // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm512_mask_cvtps_hf8(__W, __U, __A);
 }
 
 __m128i test_mm512_maskz_cvtps_hf8(__mmask16 __U, __m512 __A) {
   // CHECK-LABEL: @test_mm512_maskz_cvtps_hf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8.512(<16 x float> %{{.*}})
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8512(<16 x float> %{{.*}})
   // CHECK: zeroinitializer
   // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm512_maskz_cvtps_hf8(__U, __A);
 }
 
-// VCVTPS2HF8S - 128-bit
-
 __m128i test_mm_cvts_ps_hf8(__m128 __A) {
   // CHECK-LABEL: @test_mm_cvts_ps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtps2hf8s.128(<4 x float> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtps2hf8s128(<4 x float> %{{.*}})
   return _mm_cvts_ps_hf8(__A);
 }
 
 __m128i test_mm_mask_cvts_ps_hf8(__m128i __W, __mmask8 __U, __m128 __A) {
   // CHECK-LABEL: @test_mm_mask_cvts_ps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.128(<4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s128(<4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm_mask_cvts_ps_hf8(__W, __U, __A);
 }
 
 __m128i test_mm_maskz_cvts_ps_hf8(__mmask8 __U, __m128 __A) {
   // CHECK-LABEL: @test_mm_maskz_cvts_ps_hf8(
   // CHECK: zeroinitializer
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.128(<4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s128(<4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm_maskz_cvts_ps_hf8(__U, __A);
 }
 
-// VCVTPS2HF8S - 256-bit
-
 __m128i test_mm256_cvts_ps_hf8(__m256 __A) {
   // CHECK-LABEL: @test_mm256_cvts_ps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtps2hf8s.256(<8 x float> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtps2hf8s256(<8 x float> %{{.*}})
   return _mm256_cvts_ps_hf8(__A);
 }
 
 __m128i test_mm256_mask_cvts_ps_hf8(__m128i __W, __mmask8 __U, __m256 __A) {
   // CHECK-LABEL: @test_mm256_mask_cvts_ps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.256(<8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s256(<8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm256_mask_cvts_ps_hf8(__W, __U, __A);
 }
 
 __m128i test_mm256_maskz_cvts_ps_hf8(__mmask8 __U, __m256 __A) {
   // CHECK-LABEL: @test_mm256_maskz_cvts_ps_hf8(
   // CHECK: zeroinitializer
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.256(<8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s256(<8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm256_maskz_cvts_ps_hf8(__U, __A);
 }
 
-// VCVTPS2HF8S - 512-bit
-
 __m128i test_mm512_cvts_ps_hf8(__m512 __A) {
   // CHECK-LABEL: @test_mm512_cvts_ps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtps2hf8s.512(<16 x float> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtps2hf8s512(<16 x float> %{{.*}})
   return _mm512_cvts_ps_hf8(__A);
 }
 
 __m128i test_mm512_mask_cvts_ps_hf8(__m128i __W, __mmask16 __U, __m512 __A) {
   // CHECK-LABEL: @test_mm512_mask_cvts_ps_hf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8s.512(<16 x float> %{{.*}})
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8s512(<16 x float> %{{.*}})
   // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm512_mask_cvts_ps_hf8(__W, __U, __A);
 }
 
 __m128i test_mm512_maskz_cvts_ps_hf8(__mmask16 __U, __m512 __A) {
   // CHECK-LABEL: @test_mm512_maskz_cvts_ps_hf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8s.512(<16 x float> %{{.*}})
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8s512(<16 x float> %{{.*}})
   // CHECK: zeroinitializer
   // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm512_maskz_cvts_ps_hf8(__U, __A);
 }
 
-// VCVTROPS2HF8 - 128-bit
-
 __m128i test_mm_cvtrops_hf8(__m128 __A) {
   // CHECK-LABEL: @test_mm_cvtrops_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8.128(<4 x float> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8128(<4 x float> %{{.*}})
   return _mm_cvtrops_hf8(__A);
 }
 
 __m128i test_mm_mask_cvtrops_hf8(__m128i __W, __mmask8 __U, __m128 __A) {
   // CHECK-LABEL: @test_mm_mask_cvtrops_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.128(<4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8128(<4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm_mask_cvtrops_hf8(__W, __U, __A);
 }
 
 __m128i test_mm_maskz_cvtrops_hf8(__mmask8 __U, __m128 __A) {
   // CHECK-LABEL: @test_mm_maskz_cvtrops_hf8(
   // CHECK: zeroinitializer
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.128(<4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8128(<4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm_maskz_cvtrops_hf8(__U, __A);
 }
 
-// VCVTROPS2HF8 - 256-bit
-
 __m128i test_mm256_cvtrops_hf8(__m256 __A) {
   // CHECK-LABEL: @test_mm256_cvtrops_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8.256(<8 x float> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8256(<8 x float> %{{.*}})
   return _mm256_cvtrops_hf8(__A);
 }
 
 __m128i test_mm256_mask_cvtrops_hf8(__m128i __W, __mmask8 __U, __m256 __A) {
   // CHECK-LABEL: @test_mm256_mask_cvtrops_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.256(<8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8256(<8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm256_mask_cvtrops_hf8(__W, __U, __A);
 }
 
 __m128i test_mm256_maskz_cvtrops_hf8(__mmask8 __U, __m256 __A) {
   // CHECK-LABEL: @test_mm256_maskz_cvtrops_hf8(
   // CHECK: zeroinitializer
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.256(<8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8256(<8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm256_maskz_cvtrops_hf8(__U, __A);
 }
 
-// VCVTROPS2HF8 - 512-bit
-
 __m128i test_mm512_cvtrops_hf8(__m512 __A) {
   // CHECK-LABEL: @test_mm512_cvtrops_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8.512(<16 x float> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8512(<16 x float> %{{.*}})
   return _mm512_cvtrops_hf8(__A);
 }
 
 __m128i test_mm512_mask_cvtrops_hf8(__m128i __W, __mmask16 __U, __m512 __A) {
   // CHECK-LABEL: @test_mm512_mask_cvtrops_hf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8.512(<16 x float> %{{.*}})
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8512(<16 x float> %{{.*}})
   // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm512_mask_cvtrops_hf8(__W, __U, __A);
 }
 
 __m128i test_mm512_maskz_cvtrops_hf8(__mmask16 __U, __m512 __A) {
   // CHECK-LABEL: @test_mm512_maskz_cvtrops_hf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8.512(<16 x float> %{{.*}})
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8512(<16 x float> %{{.*}})
   // CHECK: zeroinitializer
   // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm512_maskz_cvtrops_hf8(__U, __A);
 }
 
-// VCVTROPS2HF8S - 128-bit
-
 __m128i test_mm_cvts_rops_hf8(__m128 __A) {
   // CHECK-LABEL: @test_mm_cvts_rops_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s.128(<4 x float> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s128(<4 x float> %{{.*}})
   return _mm_cvts_rops_hf8(__A);
 }
 
 __m128i test_mm_mask_cvts_rops_hf8(__m128i __W, __mmask8 __U, __m128 __A) {
   // CHECK-LABEL: @test_mm_mask_cvts_rops_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.128(<4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s128(<4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm_mask_cvts_rops_hf8(__W, __U, __A);
 }
 
 __m128i test_mm_maskz_cvts_rops_hf8(__mmask8 __U, __m128 __A) {
   // CHECK-LABEL: @test_mm_maskz_cvts_rops_hf8(
   // CHECK: zeroinitializer
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.128(<4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s128(<4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm_maskz_cvts_rops_hf8(__U, __A);
 }
 
-// VCVTROPS2HF8S - 256-bit
-
 __m128i test_mm256_cvts_rops_hf8(__m256 __A) {
   // CHECK-LABEL: @test_mm256_cvts_rops_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s.256(<8 x float> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s256(<8 x float> %{{.*}})
   return _mm256_cvts_rops_hf8(__A);
 }
 
 __m128i test_mm256_mask_cvts_rops_hf8(__m128i __W, __mmask8 __U, __m256 __A) {
   // CHECK-LABEL: @test_mm256_mask_cvts_rops_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.256(<8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s256(<8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm256_mask_cvts_rops_hf8(__W, __U, __A);
 }
 
 __m128i test_mm256_maskz_cvts_rops_hf8(__mmask8 __U, __m256 __A) {
   // CHECK-LABEL: @test_mm256_maskz_cvts_rops_hf8(
   // CHECK: zeroinitializer
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.256(<8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s256(<8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm256_maskz_cvts_rops_hf8(__U, __A);
 }
 
-// VCVTROPS2HF8S - 512-bit
-
 __m128i test_mm512_cvts_rops_hf8(__m512 __A) {
   // CHECK-LABEL: @test_mm512_cvts_rops_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s.512(<16 x float> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s512(<16 x float> %{{.*}})
   return _mm512_cvts_rops_hf8(__A);
 }
 
 __m128i test_mm512_mask_cvts_rops_hf8(__m128i __W, __mmask16 __U, __m512 __A) {
   // CHECK-LABEL: @test_mm512_mask_cvts_rops_hf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s.512(<16 x float> %{{.*}})
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s512(<16 x float> %{{.*}})
   // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm512_mask_cvts_rops_hf8(__W, __U, __A);
 }
 
 __m128i test_mm512_maskz_cvts_rops_hf8(__mmask16 __U, __m512 __A) {
   // CHECK-LABEL: @test_mm512_maskz_cvts_rops_hf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s.512(<16 x float> %{{.*}})
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s512(<16 x float> %{{.*}})
   // CHECK: zeroinitializer
   // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm512_maskz_cvts_rops_hf8(__U, __A);
 }
 
-//
-// Convert from FP32 to FP8 with bias
-// VCVTBIASPS2BF8 / VCVTBIASPS2BF8S / VCVTBIASPS2HF8 /
-// VCVTBIASPS2HF8S
-//
-
-// VCVTBIASPS2BF8 - 128-bit
-
 __m128i test_mm_cvtbiasps_bf8(__m128i __A, __m128 __B) {
   // CHECK-LABEL: @test_mm_cvtbiasps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8.128(<4 x i32> %{{.*}}, <4 x float> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8128(<4 x i32> %{{.*}}, <4 x float> %{{.*}})
   return _mm_cvtbiasps_bf8(__A, __B);
 }
 
 __m128i test_mm_mask_cvtbiasps_bf8(__m128i __W, __mmask8 __U, __m128i __A, __m128 __B) {
   // CHECK-LABEL: @test_mm_mask_cvtbiasps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.128(<4 x i32> %{{.*}}, <4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8128(<4 x i32> %{{.*}}, <4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm_mask_cvtbiasps_bf8(__W, __U, __A, __B);
 }
 
 __m128i test_mm_maskz_cvtbiasps_bf8(__mmask8 __U, __m128i __A, __m128 __B) {
   // CHECK-LABEL: @test_mm_maskz_cvtbiasps_bf8(
   // CHECK: zeroinitializer
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.128(<4 x i32> %{{.*}}, <4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8128(<4 x i32> %{{.*}}, <4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm_maskz_cvtbiasps_bf8(__U, __A, __B);
 }
 
-// VCVTBIASPS2BF8 - 256-bit
-
 __m128i test_mm256_cvtbiasps_bf8(__m256i __A, __m256 __B) {
   // CHECK-LABEL: @test_mm256_cvtbiasps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8.256(<8 x i32> %{{.*}}, <8 x float> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8256(<8 x i32> %{{.*}}, <8 x float> %{{.*}})
   return _mm256_cvtbiasps_bf8(__A, __B);
 }
 
 __m128i test_mm256_mask_cvtbiasps_bf8(__m128i __W, __mmask8 __U, __m256i __A, __m256 __B) {
   // CHECK-LABEL: @test_mm256_mask_cvtbiasps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.256(<8 x i32> %{{.*}}, <8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8256(<8 x i32> %{{.*}}, <8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm256_mask_cvtbiasps_bf8(__W, __U, __A, __B);
 }
 
 __m128i test_mm256_maskz_cvtbiasps_bf8(__mmask8 __U, __m256i __A, __m256 __B) {
   // CHECK-LABEL: @test_mm256_maskz_cvtbiasps_bf8(
   // CHECK: zeroinitializer
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.256(<8 x i32> %{{.*}}, <8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8256(<8 x i32> %{{.*}}, <8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm256_maskz_cvtbiasps_bf8(__U, __A, __B);
 }
 
-// VCVTBIASPS2BF8 - 512-bit
-
 __m128i test_mm512_cvtbiasps_bf8(__m512i __A, __m512 __B) {
   // CHECK-LABEL: @test_mm512_cvtbiasps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8.512(<16 x i32> %{{.*}}, <16 x float> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8512(<16 x i32> %{{.*}}, <16 x float> %{{.*}})
   return _mm512_cvtbiasps_bf8(__A, __B);
 }
 
 __m128i test_mm512_mask_cvtbiasps_bf8(__m128i __W, __mmask16 __U, __m512i __A, __m512 __B) {
   // CHECK-LABEL: @test_mm512_mask_cvtbiasps_bf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8.512(<16 x i32> %{{.*}}, <16 x float> %{{.*}})
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8512(<16 x i32> %{{.*}}, <16 x float> %{{.*}})
   // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm512_mask_cvtbiasps_bf8(__W, __U, __A, __B);
 }
 
 __m128i test_mm512_maskz_cvtbiasps_bf8(__mmask16 __U, __m512i __A, __m512 __B) {
   // CHECK-LABEL: @test_mm512_maskz_cvtbiasps_bf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8.512(<16 x i32> %{{.*}}, <16 x float> %{{.*}})
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8512(<16 x i32> %{{.*}}, <16 x float> %{{.*}})
   // CHECK: zeroinitializer
   // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm512_maskz_cvtbiasps_bf8(__U, __A, __B);
 }
 
-// VCVTBIASPS2BF8S - 128-bit
-
 __m128i test_mm_cvts_biasps_bf8(__m128i __A, __m128 __B) {
   // CHECK-LABEL: @test_mm_cvts_biasps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s.128(<4 x i32> %{{.*}}, <4 x float> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s128(<4 x i32> %{{.*}}, <4 x float> %{{.*}})
   return _mm_cvts_biasps_bf8(__A, __B);
 }
 
 __m128i test_mm_mask_cvts_biasps_bf8(__m128i __W, __mmask8 __U, __m128i __A, __m128 __B) {
   // CHECK-LABEL: @test_mm_mask_cvts_biasps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.128(<4 x i32> %{{.*}}, <4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s128(<4 x i32> %{{.*}}, <4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm_mask_cvts_biasps_bf8(__W, __U, __A, __B);
 }
 
 __m128i test_mm_maskz_cvts_biasps_bf8(__mmask8 __U, __m128i __A, __m128 __B) {
   // CHECK-LABEL: @test_mm_maskz_cvts_biasps_bf8(
   // CHECK: zeroinitializer
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.128(<4 x i32> %{{.*}}, <4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s128(<4 x i32> %{{.*}}, <4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm_maskz_cvts_biasps_bf8(__U, __A, __B);
 }
 
-// VCVTBIASPS2BF8S - 256-bit
-
 __m128i test_mm256_cvts_biasps_bf8(__m256i __A, __m256 __B) {
   // CHECK-LABEL: @test_mm256_cvts_biasps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s.256(<8 x i32> %{{.*}}, <8 x float> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s256(<8 x i32> %{{.*}}, <8 x float> %{{.*}})
   return _mm256_cvts_biasps_bf8(__A, __B);
 }
 
 __m128i test_mm256_mask_cvts_biasps_bf8(__m128i __W, __mmask8 __U, __m256i __A, __m256 __B) {
   // CHECK-LABEL: @test_mm256_mask_cvts_biasps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.256(<8 x i32> %{{.*}}, <8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s256(<8 x i32> %{{.*}}, <8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm256_mask_cvts_biasps_bf8(__W, __U, __A, __B);
 }
 
 __m128i test_mm256_maskz_cvts_biasps_bf8(__mmask8 __U, __m256i __A, __m256 __B) {
   // CHECK-LABEL: @test_mm256_maskz_cvts_biasps_bf8(
   // CHECK: zeroinitializer
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.256(<8 x i32> %{{.*}}, <8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s256(<8 x i32> %{{.*}}, <8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm256_maskz_cvts_biasps_bf8(__U, __A, __B);
 }
 
-// VCVTBIASPS2BF8S - 512-bit
-
 __m128i test_mm512_cvts_biasps_bf8(__m512i __A, __m512 __B) {
   // CHECK-LABEL: @test_mm512_cvts_biasps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s.512(<16 x i32> %{{.*}}, <16 x float> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s512(<16 x i32> %{{.*}}, <16 x float> %{{.*}})
   return _mm512_cvts_biasps_bf8(__A, __B);
 }
 
 __m128i test_mm512_mask_cvts_biasps_bf8(__m128i __W, __mmask16 __U, __m512i __A, __m512 __B) {
   // CHECK-LABEL: @test_mm512_mask_cvts_biasps_bf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s.512(<16 x i32> %{{.*}}, <16 x float> %{{.*}})
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s512(<16 x i32> %{{.*}}, <16 x float> %{{.*}})
   // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm512_mask_cvts_biasps_bf8(__W, __U, __A, __B);
 }
 
 __m128i test_mm512_maskz_cvts_biasps_bf8(__mmask16 __U, __m512i __A, __m512 __B) {
   // CHECK-LABEL: @test_mm512_maskz_cvts_biasps_bf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s.512(<16 x i32> %{{.*}}, <16 x float> %{{.*}})
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s512(<16 x i32> %{{.*}}, <16 x float> %{{.*}})
   // CHECK: zeroinitializer
   // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm512_maskz_cvts_biasps_bf8(__U, __A, __B);
 }
 
-// VCVTBIASPS2HF8 - 128-bit
-
 __m128i test_mm_cvtbiasps_hf8(__m128i __A, __m128 __B) {
   // CHECK-LABEL: @test_mm_cvtbiasps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8.128(<4 x i32> %{{.*}}, <4 x float> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8128(<4 x i32> %{{.*}}, <4 x float> %{{.*}})
   return _mm_cvtbiasps_hf8(__A, __B);
 }
 
 __m128i test_mm_mask_cvtbiasps_hf8(__m128i __W, __mmask8 __U, __m128i __A, __m128 __B) {
   // CHECK-LABEL: @test_mm_mask_cvtbiasps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.128(<4 x i32> %{{.*}}, <4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8128(<4 x i32> %{{.*}}, <4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm_mask_cvtbiasps_hf8(__W, __U, __A, __B);
 }
 
 __m128i test_mm_maskz_cvtbiasps_hf8(__mmask8 __U, __m128i __A, __m128 __B) {
   // CHECK-LABEL: @test_mm_maskz_cvtbiasps_hf8(
   // CHECK: zeroinitializer
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.128(<4 x i32> %{{.*}}, <4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8128(<4 x i32> %{{.*}}, <4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm_maskz_cvtbiasps_hf8(__U, __A, __B);
 }
 
-// VCVTBIASPS2HF8 - 256-bit
-
 __m128i test_mm256_cvtbiasps_hf8(__m256i __A, __m256 __B) {
   // CHECK-LABEL: @test_mm256_cvtbiasps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8.256(<8 x i32> %{{.*}}, <8 x float> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8256(<8 x i32> %{{.*}}, <8 x float> %{{.*}})
   return _mm256_cvtbiasps_hf8(__A, __B);
 }
 
 __m128i test_mm256_mask_cvtbiasps_hf8(__m128i __W, __mmask8 __U, __m256i __A, __m256 __B) {
   // CHECK-LABEL: @test_mm256_mask_cvtbiasps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.256(<8 x i32> %{{.*}}, <8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8256(<8 x i32> %{{.*}}, <8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm256_mask_cvtbiasps_hf8(__W, __U, __A, __B);
 }
 
 __m128i test_mm256_maskz_cvtbiasps_hf8(__mmask8 __U, __m256i __A, __m256 __B) {
   // CHECK-LABEL: @test_mm256_maskz_cvtbiasps_hf8(
   // CHECK: zeroinitializer
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.256(<8 x i32> %{{.*}}, <8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8256(<8 x i32> %{{.*}}, <8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm256_maskz_cvtbiasps_hf8(__U, __A, __B);
 }
 
-// VCVTBIASPS2HF8 - 512-bit
-
 __m128i test_mm512_cvtbiasps_hf8(__m512i __A, __m512 __B) {
   // CHECK-LABEL: @test_mm512_cvtbiasps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8.512(<16 x i32> %{{.*}}, <16 x float> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8512(<16 x i32> %{{.*}}, <16 x float> %{{.*}})
   return _mm512_cvtbiasps_hf8(__A, __B);
 }
 
 __m128i test_mm512_mask_cvtbiasps_hf8(__m128i __W, __mmask16 __U, __m512i __A, __m512 __B) {
   // CHECK-LABEL: @test_mm512_mask_cvtbiasps_hf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8.512(<16 x i32> %{{.*}}, <16 x float> %{{.*}})
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8512(<16 x i32> %{{.*}}, <16 x float> %{{.*}})
   // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm512_mask_cvtbiasps_hf8(__W, __U, __A, __B);
 }
 
 __m128i test_mm512_maskz_cvtbiasps_hf8(__mmask16 __U, __m512i __A, __m512 __B) {
   // CHECK-LABEL: @test_mm512_maskz_cvtbiasps_hf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8.512(<16 x i32> %{{.*}}, <16 x float> %{{.*}})
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8512(<16 x i32> %{{.*}}, <16 x float> %{{.*}})
   // CHECK: zeroinitializer
   // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm512_maskz_cvtbiasps_hf8(__U, __A, __B);
 }
 
-// VCVTBIASPS2HF8S - 128-bit
-
 __m128i test_mm_cvts_biasps_hf8(__m128i __A, __m128 __B) {
   // CHECK-LABEL: @test_mm_cvts_biasps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s.128(<4 x i32> %{{.*}}, <4 x float> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s128(<4 x i32> %{{.*}}, <4 x float> %{{.*}})
   return _mm_cvts_biasps_hf8(__A, __B);
 }
 
 __m128i test_mm_mask_cvts_biasps_hf8(__m128i __W, __mmask8 __U, __m128i __A, __m128 __B) {
   // CHECK-LABEL: @test_mm_mask_cvts_biasps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.128(<4 x i32> %{{.*}}, <4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s128(<4 x i32> %{{.*}}, <4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm_mask_cvts_biasps_hf8(__W, __U, __A, __B);
 }
 
 __m128i test_mm_maskz_cvts_biasps_hf8(__mmask8 __U, __m128i __A, __m128 __B) {
   // CHECK-LABEL: @test_mm_maskz_cvts_biasps_hf8(
   // CHECK: zeroinitializer
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.128(<4 x i32> %{{.*}}, <4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s128(<4 x i32> %{{.*}}, <4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm_maskz_cvts_biasps_hf8(__U, __A, __B);
 }
 
-// VCVTBIASPS2HF8S - 256-bit
-
 __m128i test_mm256_cvts_biasps_hf8(__m256i __A, __m256 __B) {
   // CHECK-LABEL: @test_mm256_cvts_biasps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s.256(<8 x i32> %{{.*}}, <8 x float> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s256(<8 x i32> %{{.*}}, <8 x float> %{{.*}})
   return _mm256_cvts_biasps_hf8(__A, __B);
 }
 
 __m128i test_mm256_mask_cvts_biasps_hf8(__m128i __W, __mmask8 __U, __m256i __A, __m256 __B) {
   // CHECK-LABEL: @test_mm256_mask_cvts_biasps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.256(<8 x i32> %{{.*}}, <8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s256(<8 x i32> %{{.*}}, <8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm256_mask_cvts_biasps_hf8(__W, __U, __A, __B);
 }
 
 __m128i test_mm256_maskz_cvts_biasps_hf8(__mmask8 __U, __m256i __A, __m256 __B) {
   // CHECK-LABEL: @test_mm256_maskz_cvts_biasps_hf8(
   // CHECK: zeroinitializer
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.256(<8 x i32> %{{.*}}, <8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s256(<8 x i32> %{{.*}}, <8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm256_maskz_cvts_biasps_hf8(__U, __A, __B);
 }
 
-// VCVTBIASPS2HF8S - 512-bit
-
 __m128i test_mm512_cvts_biasps_hf8(__m512i __A, __m512 __B) {
   // CHECK-LABEL: @test_mm512_cvts_biasps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s.512(<16 x i32> %{{.*}}, <16 x float> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s512(<16 x i32> %{{.*}}, <16 x float> %{{.*}})
   return _mm512_cvts_biasps_hf8(__A, __B);
 }
 
 __m128i test_mm512_mask_cvts_biasps_hf8(__m128i __W, __mmask16 __U, __m512i __A, __m512 __B) {
   // CHECK-LABEL: @test_mm512_mask_cvts_biasps_hf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s.512(<16 x i32> %{{.*}}, <16 x float> %{{.*}})
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s512(<16 x i32> %{{.*}}, <16 x float> %{{.*}})
   // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm512_mask_cvts_biasps_hf8(__W, __U, __A, __B);
 }
 
 __m128i test_mm512_maskz_cvts_biasps_hf8(__mmask16 __U, __m512i __A, __m512 __B) {
   // CHECK-LABEL: @test_mm512_maskz_cvts_biasps_hf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s.512(<16 x i32> %{{.*}}, <16 x float> %{{.*}})
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s512(<16 x i32> %{{.*}}, <16 x float> %{{.*}})
   // CHECK: zeroinitializer
   // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm512_maskz_cvts_biasps_hf8(__U, __A, __B);
 }
 
-//
-// Convert from FP8 to FP32
-// VCVTBF82PS / VCVTHF82PS
-//
-
-// VCVTBF82PS - 128-bit
-
 __m128 test_mm_cvtbf8_ps(__m128i __A) {
   // CHECK-LABEL: @test_mm_cvtbf8_ps(
-  // CHECK: call <4 x float> @llvm.x86.avx10.vcvtbf82ps.128(<16 x i8> %{{.*}})
+  // CHECK: call <4 x float> @llvm.x86.avx10.vcvtbf82ps128(<16 x i8> %{{.*}})
   return _mm_cvtbf8_ps(__A);
 }
 
 __m128 test_mm_mask_cvtbf8_ps(__m128 __W, __mmask8 __U, __m128i __A) {
   // CHECK-LABEL: @test_mm_mask_cvtbf8_ps(
-  // CHECK: [[RES:%.*]] = call <4 x float> @llvm.x86.avx10.vcvtbf82ps.128(<16 x i8> %{{.*}})
+  // CHECK: [[RES:%.*]] = call <4 x float> @llvm.x86.avx10.vcvtbf82ps128(<16 x i8> %{{.*}})
   // CHECK: select <4 x i1> %{{.*}}, <4 x float> [[RES]], <4 x float> %{{.*}}
   return _mm_mask_cvtbf8_ps(__W, __U, __A);
 }
 
 __m128 test_mm_maskz_cvtbf8_ps(__mmask8 __U, __m128i __A) {
   // CHECK-LABEL: @test_mm_maskz_cvtbf8_ps(
-  // CHECK: [[RES:%.*]] = call <4 x float> @llvm.x86.avx10.vcvtbf82ps.128(<16 x i8> %{{.*}})
+  // CHECK: [[RES:%.*]] = call <4 x float> @llvm.x86.avx10.vcvtbf82ps128(<16 x i8> %{{.*}})
   // CHECK: zeroinitializer
   // CHECK: select <4 x i1> %{{.*}}, <4 x float> [[RES]], <4 x float> %{{.*}}
   return _mm_maskz_cvtbf8_ps(__U, __A);
 }
 
-// VCVTBF82PS - 256-bit
-
 __m256 test_mm256_cvtbf8_ps(__m128i __A) {
   // CHECK-LABEL: @test_mm256_cvtbf8_ps(
-  // CHECK: call <8 x float> @llvm.x86.avx10.vcvtbf82ps.256(<16 x i8> %{{.*}})
+  // CHECK: call <8 x float> @llvm.x86.avx10.vcvtbf82ps256(<16 x i8> %{{.*}})
   return _mm256_cvtbf8_ps(__A);
 }
 
 __m256 test_mm256_mask_cvtbf8_ps(__m256 __W, __mmask8 __U, __m128i __A) {
   // CHECK-LABEL: @test_mm256_mask_cvtbf8_ps(
-  // CHECK: [[RES:%.*]] = call <8 x float> @llvm.x86.avx10.vcvtbf82ps.256(<16 x i8> %{{.*}})
+  // CHECK: [[RES:%.*]] = call <8 x float> @llvm.x86.avx10.vcvtbf82ps256(<16 x i8> %{{.*}})
   // CHECK: select <8 x i1> %{{.*}}, <8 x float> [[RES]], <8 x float> %{{.*}}
   return _mm256_mask_cvtbf8_ps(__W, __U, __A);
 }
 
 __m256 test_mm256_maskz_cvtbf8_ps(__mmask8 __U, __m128i __A) {
   // CHECK-LABEL: @test_mm256_maskz_cvtbf8_ps(
-  // CHECK: [[RES:%.*]] = call <8 x float> @llvm.x86.avx10.vcvtbf82ps.256(<16 x i8> %{{.*}})
+  // CHECK: [[RES:%.*]] = call <8 x float> @llvm.x86.avx10.vcvtbf82ps256(<16 x i8> %{{.*}})
   // CHECK: zeroinitializer
   // CHECK: select <8 x i1> %{{.*}}, <8 x float> [[RES]], <8 x float> %{{.*}}
   return _mm256_maskz_cvtbf8_ps(__U, __A);
 }
 
-// VCVTBF82PS - 512-bit
-
 __m512 test_mm512_cvtbf8_ps(__m128i __A) {
   // CHECK-LABEL: @test_mm512_cvtbf8_ps(
-  // CHECK: call <16 x float> @llvm.x86.avx10.vcvtbf82ps.512(<16 x i8> %{{.*}})
+  // CHECK: call <16 x float> @llvm.x86.avx10.vcvtbf82ps512(<16 x i8> %{{.*}})
   return _mm512_cvtbf8_ps(__A);
 }
 
 __m512 test_mm512_mask_cvtbf8_ps(__m512 __W, __mmask16 __U, __m128i __A) {
   // CHECK-LABEL: @test_mm512_mask_cvtbf8_ps(
-  // CHECK: [[RES:%.*]] = call <16 x float> @llvm.x86.avx10.vcvtbf82ps.512(<16 x i8> %{{.*}})
+  // CHECK: [[RES:%.*]] = call <16 x float> @llvm.x86.avx10.vcvtbf82ps512(<16 x i8> %{{.*}})
   // CHECK: select <16 x i1> %{{.*}}, <16 x float> [[RES]], <16 x float> %{{.*}}
   return _mm512_mask_cvtbf8_ps(__W, __U, __A);
 }
 
 __m512 test_mm512_maskz_cvtbf8_ps(__mmask16 __U, __m128i __A) {
   // CHECK-LABEL: @test_mm512_maskz_cvtbf8_ps(
-  // CHECK: [[RES:%.*]] = call <16 x float> @llvm.x86.avx10.vcvtbf82ps.512(<16 x i8> %{{.*}})
+  // CHECK: [[RES:%.*]] = call <16 x float> @llvm.x86.avx10.vcvtbf82ps512(<16 x i8> %{{.*}})
   // CHECK: zeroinitializer
   // CHECK: select <16 x i1> %{{.*}}, <16 x float> [[RES]], <16 x float> %{{.*}}
   return _mm512_maskz_cvtbf8_ps(__U, __A);
 }
 
-// VCVTHF82PS - 128-bit
-
 __m128 test_mm_cvthf8_ps(__m128i __A) {
   // CHECK-LABEL: @test_mm_cvthf8_ps(
-  // CHECK: call <4 x float> @llvm.x86.avx10.vcvthf82ps.128(<16 x i8> %{{.*}})
+  // CHECK: call <4 x float> @llvm.x86.avx10.vcvthf82ps128(<16 x i8> %{{.*}})
   return _mm_cvthf8_ps(__A);
 }
 
 __m128 test_mm_mask_cvthf8_ps(__m128 __W, __mmask8 __U, __m128i __A) {
   // CHECK-LABEL: @test_mm_mask_cvthf8_ps(
-  // CHECK: [[RES:%.*]] = call <4 x float> @llvm.x86.avx10.vcvthf82ps.128(<16 x i8> %{{.*}})
+  // CHECK: [[RES:%.*]] = call <4 x float> @llvm.x86.avx10.vcvthf82ps128(<16 x i8> %{{.*}})
   // CHECK: select <4 x i1> %{{.*}}, <4 x float> [[RES]], <4 x float> %{{.*}}
   return _mm_mask_cvthf8_ps(__W, __U, __A);
 }
 
 __m128 test_mm_maskz_cvthf8_ps(__mmask8 __U, __m128i __A) {
   // CHECK-LABEL: @test_mm_maskz_cvthf8_ps(
-  // CHECK: [[RES:%.*]] = call <4 x float> @llvm.x86.avx10.vcvthf82ps.128(<16 x i8> %{{.*}})
+  // CHECK: [[RES:%.*]] = call <4 x float> @llvm.x86.avx10.vcvthf82ps128(<16 x i8> %{{.*}})
   // CHECK: zeroinitializer
   // CHECK: select <4 x i1> %{{.*}}, <4 x float> [[RES]], <4 x float> %{{.*}}
   return _mm_maskz_cvthf8_ps(__U, __A);
 }
 
-// VCVTHF82PS - 256-bit
-
 __m256 test_mm256_cvthf8_ps(__m128i __A) {
   // CHECK-LABEL: @test_mm256_cvthf8_ps(
-  // CHECK: call <8 x float> @llvm.x86.avx10.vcvthf82ps.256(<16 x i8> %{{.*}})
+  // CHECK: call <8 x float> @llvm.x86.avx10.vcvthf82ps256(<16 x i8> %{{.*}})
   return _mm256_cvthf8_ps(__A);
 }
 
 __m256 test_mm256_mask_cvthf8_ps(__m256 __W, __mmask8 __U, __m128i __A) {
   // CHECK-LABEL: @test_mm256_mask_cvthf8_ps(
-  // CHECK: [[RES:%.*]] = call <8 x float> @llvm.x86.avx10.vcvthf82ps.256(<16 x i8> %{{.*}})
+  // CHECK: [[RES:%.*]] = call <8 x float> @llvm.x86.avx10.vcvthf82ps256(<16 x i8> %{{.*}})
   // CHECK: select <8 x i1> %{{.*}}, <8 x float> [[RES]], <8 x float> %{{.*}}
   return _mm256_mask_cvthf8_ps(__W, __U, __A);
 }
 
 __m256 test_mm256_maskz_cvthf8_ps(__mmask8 __U, __m128i __A) {
   // CHECK-LABEL: @test_mm256_maskz_cvthf8_ps(
-  // CHECK: [[RES:%.*]] = call <8 x float> @llvm.x86.avx10.vcvthf82ps.256(<16 x i8> %{{.*}})
+  // CHECK: [[RES:%.*]] = call <8 x float> @llvm.x86.avx10.vcvthf82ps256(<16 x i8> %{{.*}})
   // CHECK: zeroinitializer
   // CHECK: select <8 x i1> %{{.*}}, <8 x float> [[RES]], <8 x float> %{{.*}}
   return _mm256_maskz_cvthf8_ps(__U, __A);
 }
 
-// VCVTHF82PS - 512-bit
-
 __m512 test_mm512_cvthf8_ps(__m128i __A) {
   // CHECK-LABEL: @test_mm512_cvthf8_ps(
-  // CHECK: call <16 x float> @llvm.x86.avx10.vcvthf82ps.512(<16 x i8> %{{.*}})
+  // CHECK: call <16 x float> @llvm.x86.avx10.vcvthf82ps512(<16 x i8> %{{.*}})
   return _mm512_cvthf8_ps(__A);
 }
 
 __m512 test_mm512_mask_cvthf8_ps(__m512 __W, __mmask16 __U, __m128i __A) {
   // CHECK-LABEL: @test_mm512_mask_cvthf8_ps(
-  // CHECK: [[RES:%.*]] = call <16 x float> @llvm.x86.avx10.vcvthf82ps.512(<16 x i8> %{{.*}})
+  // CHECK: [[RES:%.*]] = call <16 x float> @llvm.x86.avx10.vcvthf82ps512(<16 x i8> %{{.*}})
   // CHECK: select <16 x i1> %{{.*}}, <16 x float> [[RES]], <16 x float> %{{.*}}
   return _mm512_mask_cvthf8_ps(__W, __U, __A);
 }
 
 __m512 test_mm512_maskz_cvthf8_ps(__mmask16 __U, __m128i __A) {
   // CHECK-LABEL: @test_mm512_maskz_cvthf8_ps(
-  // CHECK: [[RES:%.*]] = call <16 x float> @llvm.x86.avx10.vcvthf82ps.512(<16 x i8> %{{.*}})
+  // CHECK: [[RES:%.*]] = call <16 x float> @llvm.x86.avx10.vcvthf82ps512(<16 x i8> %{{.*}})
   // CHECK: zeroinitializer
   // CHECK: select <16 x i1> %{{.*}}, <16 x float> [[RES]], <16 x float> %{{.*}}
   return _mm512_maskz_cvthf8_ps(__U, __A);
 }
 
-//
-// Convert from FP8 to FP4
-// VCVTBF82BF4S / VCVTHF82BF4S
-//
-
-// VCVTBF82BF4S - register forms
-
 __m128i test_mm_cvts_bf8_bf4(__m128i __A) {
   // CHECK-LABEL: @test_mm_cvts_bf8_bf4(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbf82bf4s.128(<16 x i8> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbf82bf4s128(<16 x i8> %{{.*}})
   return _mm_cvts_bf8_bf4(__A);
 }
 
 __m128i test_mm256_cvts_bf8_bf4(__m256i __A) {
   // CHECK-LABEL: @test_mm256_cvts_bf8_bf4(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbf82bf4s.256(<32 x i8> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbf82bf4s256(<32 x i8> %{{.*}})
   return _mm256_cvts_bf8_bf4(__A);
 }
 
 __m256i test_mm512_cvts_bf8_bf4(__m512i __A) {
   // CHECK-LABEL: @test_mm512_cvts_bf8_bf4(
-  // CHECK: call <32 x i8> @llvm.x86.avx10.vcvtbf82bf4s.512(<64 x i8> %{{.*}})
+  // CHECK: call <32 x i8> @llvm.x86.avx10.vcvtbf82bf4s512(<64 x i8> %{{.*}})
   return _mm512_cvts_bf8_bf4(__A);
 }
 
-// VCVTHF82BF4S - register forms
-
 __m128i test_mm_cvts_hf8_bf4(__m128i __A) {
   // CHECK-LABEL: @test_mm_cvts_hf8_bf4(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvthf82bf4s.128(<16 x i8> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvthf82bf4s128(<16 x i8> %{{.*}})
   return _mm_cvts_hf8_bf4(__A);
 }
 
 __m128i test_mm256_cvts_hf8_bf4(__m256i __A) {
   // CHECK-LABEL: @test_mm256_cvts_hf8_bf4(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvthf82bf4s.256(<32 x i8> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvthf82bf4s256(<32 x i8> %{{.*}})
   return _mm256_cvts_hf8_bf4(__A);
 }
 
 __m256i test_mm512_cvts_hf8_bf4(__m512i __A) {
   // CHECK-LABEL: @test_mm512_cvts_hf8_bf4(
-  // CHECK: call <32 x i8> @llvm.x86.avx10.vcvthf82bf4s.512(<64 x i8> %{{.*}})
+  // CHECK: call <32 x i8> @llvm.x86.avx10.vcvthf82bf4s512(<64 x i8> %{{.*}})
   return _mm512_cvts_hf8_bf4(__A);
 }
 
-// VCVTBF82BF4S - memory store forms
-
-void test_mm_cvts_bf8_bf4_storeu(void *__P, __m128i __A) {
-  // CHECK-LABEL: @test_mm_cvts_bf8_bf4_storeu(
-  // CHECK: call void @llvm.x86.avx10.vcvtbf82bf4s.128.mem(ptr %{{.*}}, <16 x i8> %{{.*}})
-  _mm_cvts_bf8_bf4_storeu(__P, __A);
-}
-
-void test_mm256_cvts_bf8_bf4_storeu(void *__P, __m256i __A) {
-  // CHECK-LABEL: @test_mm256_cvts_bf8_bf4_storeu(
-  // CHECK: call void @llvm.x86.avx10.vcvtbf82bf4s.256.mem(ptr %{{.*}}, <32 x i8> %{{.*}})
-  _mm256_cvts_bf8_bf4_storeu(__P, __A);
-}
-
-void test_mm512_cvts_bf8_bf4_storeu(void *__P, __m512i __A) {
-  // CHECK-LABEL: @test_mm512_cvts_bf8_bf4_storeu(
-  // CHECK: call void @llvm.x86.avx10.vcvtbf82bf4s.512.mem(ptr %{{.*}}, <64 x i8> %{{.*}})
-  _mm512_cvts_bf8_bf4_storeu(__P, __A);
-}
-
-// VCVTHF82BF4S - memory store forms
-
-void test_mm_cvts_hf8_bf4_storeu(void *__P, __m128i __A) {
-  // CHECK-LABEL: @test_mm_cvts_hf8_bf4_storeu(
-  // CHECK: call void @llvm.x86.avx10.vcvthf82bf4s.128.mem(ptr %{{.*}}, <16 x i8> %{{.*}})
-  _mm_cvts_hf8_bf4_storeu(__P, __A);
-}
-
-void test_mm256_cvts_hf8_bf4_storeu(void *__P, __m256i __A) {
-  // CHECK-LABEL: @test_mm256_cvts_hf8_bf4_storeu(
-  // CHECK: call void @llvm.x86.avx10.vcvthf82bf4s.256.mem(ptr %{{.*}}, <32 x i8> %{{.*}})
-  _mm256_cvts_hf8_bf4_storeu(__P, __A);
-}
-
-void test_mm512_cvts_hf8_bf4_storeu(void *__P, __m512i __A) {
-  // CHECK-LABEL: @test_mm512_cvts_hf8_bf4_storeu(
-  // CHECK: call void @llvm.x86.avx10.vcvthf82bf4s.512.mem(ptr %{{.*}}, <64 x i8> %{{.*}})
-  _mm512_cvts_hf8_bf4_storeu(__P, __A);
-}
-
-//
-// Convert from FP8 to FP6
-// VCVTBF82BF6S / VCVTHF82HF6S
-//
-
-// VCVTBF82BF6S
-
 __m128i test_mm_cvts_bf8_bf6(__m128i __A) {
   // CHECK-LABEL: @test_mm_cvts_bf8_bf6(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbf82bf6s.128(<16 x i8> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbf82bf6s128(<16 x i8> %{{.*}})
   return _mm_cvts_bf8_bf6(__A);
 }
 
 __m256i test_mm256_cvts_bf8_bf6(__m256i __A) {
   // CHECK-LABEL: @test_mm256_cvts_bf8_bf6(
-  // CHECK: call <32 x i8> @llvm.x86.avx10.vcvtbf82bf6s.256(<32 x i8> %{{.*}})
+  // CHECK: call <32 x i8> @llvm.x86.avx10.vcvtbf82bf6s256(<32 x i8> %{{.*}})
   return _mm256_cvts_bf8_bf6(__A);
 }
 
 __m512i test_mm512_cvts_bf8_bf6(__m512i __A) {
   // CHECK-LABEL: @test_mm512_cvts_bf8_bf6(
-  // CHECK: call <64 x i8> @llvm.x86.avx10.vcvtbf82bf6s.512(<64 x i8> %{{.*}})
+  // CHECK: call <64 x i8> @llvm.x86.avx10.vcvtbf82bf6s512(<64 x i8> %{{.*}})
   return _mm512_cvts_bf8_bf6(__A);
 }
 
-// VCVTHF82HF6S
-
 __m128i test_mm_cvts_hf8_hf6(__m128i __A) {
   // CHECK-LABEL: @test_mm_cvts_hf8_hf6(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvthf82hf6s.128(<16 x i8> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvthf82hf6s128(<16 x i8> %{{.*}})
   return _mm_cvts_hf8_hf6(__A);
 }
 
 __m256i test_mm256_cvts_hf8_hf6(__m256i __A) {
   // CHECK-LABEL: @test_mm256_cvts_hf8_hf6(
-  // CHECK: call <32 x i8> @llvm.x86.avx10.vcvthf82hf6s.256(<32 x i8> %{{.*}})
+  // CHECK: call <32 x i8> @llvm.x86.avx10.vcvthf82hf6s256(<32 x i8> %{{.*}})
   return _mm256_cvts_hf8_hf6(__A);
 }
 
 __m512i test_mm512_cvts_hf8_hf6(__m512i __A) {
   // CHECK-LABEL: @test_mm512_cvts_hf8_hf6(
-  // CHECK: call <64 x i8> @llvm.x86.avx10.vcvthf82hf6s.512(<64 x i8> %{{.*}})
+  // CHECK: call <64 x i8> @llvm.x86.avx10.vcvthf82hf6s512(<64 x i8> %{{.*}})
   return _mm512_cvts_hf8_hf6(__A);
 }
 
-//
-// Convert from FP4 to FP8
-// VCVTBF42HF8
-//
-
-// VCVTBF42HF8 - 128-bit
-
 __m128i test_mm_cvtbf4_hf8(__m128i __A) {
   // CHECK-LABEL: @test_mm_cvtbf4_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbf42hf8.128(<16 x i8> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbf42hf8128(<16 x i8> %{{.*}})
   return _mm_cvtbf4_hf8(__A);
 }
 
 __m128i test_mm_mask_cvtbf4_hf8(__m128i __W, __mmask16 __U, __m128i __A) {
   // CHECK-LABEL: @test_mm_mask_cvtbf4_hf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbf42hf8.128(<16 x i8> %{{.*}})
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbf42hf8128(<16 x i8> %{{.*}})
   // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm_mask_cvtbf4_hf8(__W, __U, __A);
 }
 
 __m128i test_mm_maskz_cvtbf4_hf8(__mmask16 __U, __m128i __A) {
   // CHECK-LABEL: @test_mm_maskz_cvtbf4_hf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbf42hf8.128(<16 x i8> %{{.*}})
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbf42hf8128(<16 x i8> %{{.*}})
   // CHECK: zeroinitializer
   // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm_maskz_cvtbf4_hf8(__U, __A);
 }
 
-// VCVTBF42HF8 - 256-bit
-
 __m256i test_mm256_cvtbf4_hf8(__m128i __A) {
   // CHECK-LABEL: @test_mm256_cvtbf4_hf8(
-  // CHECK: call <32 x i8> @llvm.x86.avx10.vcvtbf42hf8.256(<16 x i8> %{{.*}})
+  // CHECK: call <32 x i8> @llvm.x86.avx10.vcvtbf42hf8256(<16 x i8> %{{.*}})
   return _mm256_cvtbf4_hf8(__A);
 }
 
 __m256i test_mm256_mask_cvtbf4_hf8(__m256i __W, __mmask32 __U, __m128i __A) {
   // CHECK-LABEL: @test_mm256_mask_cvtbf4_hf8(
-  // CHECK: [[RES:%.*]] = call <32 x i8> @llvm.x86.avx10.vcvtbf42hf8.256(<16 x i8> %{{.*}})
+  // CHECK: [[RES:%.*]] = call <32 x i8> @llvm.x86.avx10.vcvtbf42hf8256(<16 x i8> %{{.*}})
   // CHECK: select <32 x i1> %{{.*}}, <32 x i8> [[RES]], <32 x i8> %{{.*}}
   return _mm256_mask_cvtbf4_hf8(__W, __U, __A);
 }
 
 __m256i test_mm256_maskz_cvtbf4_hf8(__mmask32 __U, __m128i __A) {
   // CHECK-LABEL: @test_mm256_maskz_cvtbf4_hf8(
-  // CHECK: [[RES:%.*]] = call <32 x i8> @llvm.x86.avx10.vcvtbf42hf8.256(<16 x i8> %{{.*}})
+  // CHECK: [[RES:%.*]] = call <32 x i8> @llvm.x86.avx10.vcvtbf42hf8256(<16 x i8> %{{.*}})
   // CHECK: zeroinitializer
   // CHECK: select <32 x i1> %{{.*}}, <32 x i8> [[RES]], <32 x i8> %{{.*}}
   return _mm256_maskz_cvtbf4_hf8(__U, __A);
 }
 
-// VCVTBF42HF8 - 512-bit
-
 __m512i test_mm512_cvtbf4_hf8(__m256i __A) {
   // CHECK-LABEL: @test_mm512_cvtbf4_hf8(
-  // CHECK: call <64 x i8> @llvm.x86.avx10.vcvtbf42hf8.512(<32 x i8> %{{.*}})
+  // CHECK: call <64 x i8> @llvm.x86.avx10.vcvtbf42hf8512(<32 x i8> %{{.*}})
   return _mm512_cvtbf4_hf8(__A);
 }
 
 __m512i test_mm512_mask_cvtbf4_hf8(__m512i __W, __mmask64 __U, __m256i __A) {
   // CHECK-LABEL: @test_mm512_mask_cvtbf4_hf8(
-  // CHECK: [[RES:%.*]] = call <64 x i8> @llvm.x86.avx10.vcvtbf42hf8.512(<32 x i8> %{{.*}})
+  // CHECK: [[RES:%.*]] = call <64 x i8> @llvm.x86.avx10.vcvtbf42hf8512(<32 x i8> %{{.*}})
   // CHECK: select <64 x i1> %{{.*}}, <64 x i8> [[RES]], <64 x i8> %{{.*}}
   return _mm512_mask_cvtbf4_hf8(__W, __U, __A);
 }
 
 __m512i test_mm512_maskz_cvtbf4_hf8(__mmask64 __U, __m256i __A) {
   // CHECK-LABEL: @test_mm512_maskz_cvtbf4_hf8(
-  // CHECK: [[RES:%.*]] = call <64 x i8> @llvm.x86.avx10.vcvtbf42hf8.512(<32 x i8> %{{.*}})
+  // CHECK: [[RES:%.*]] = call <64 x i8> @llvm.x86.avx10.vcvtbf42hf8512(<32 x i8> %{{.*}})
   // CHECK: zeroinitializer
   // CHECK: select <64 x i1> %{{.*}}, <64 x i8> [[RES]], <64 x i8> %{{.*}}
   return _mm512_maskz_cvtbf4_hf8(__U, __A);
 }
 
-//
-// Convert from FP6 to FP8
-// VCVTBF62HF8 / VCVTHF62HF8
-//
-
-// VCVTBF62HF8 - 128-bit
-
 __m128i test_mm_cvtbf6_hf8(__m128i __A) {
   // CHECK-LABEL: @test_mm_cvtbf6_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbf62hf8.128(<16 x i8> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbf62hf8128(<16 x i8> %{{.*}})
   return _mm_cvtbf6_hf8(__A);
 }
 
 __m128i test_mm_mask_cvtbf6_hf8(__m128i __W, __mmask16 __U, __m128i __A) {
   // CHECK-LABEL: @test_mm_mask_cvtbf6_hf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbf62hf8.128(<16 x i8> %{{.*}})
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbf62hf8128(<16 x i8> %{{.*}})
   // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm_mask_cvtbf6_hf8(__W, __U, __A);
 }
 
 __m128i test_mm_maskz_cvtbf6_hf8(__mmask16 __U, __m128i __A) {
   // CHECK-LABEL: @test_mm_maskz_cvtbf6_hf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbf62hf8.128(<16 x i8> %{{.*}})
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbf62hf8128(<16 x i8> %{{.*}})
   // CHECK: zeroinitializer
   // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm_maskz_cvtbf6_hf8(__U, __A);
 }
 
-// VCVTBF62HF8 - 256-bit
-
 __m256i test_mm256_cvtbf6_hf8(__m256i __A) {
   // CHECK-LABEL: @test_mm256_cvtbf6_hf8(
-  // CHECK: call <32 x i8> @llvm.x86.avx10.vcvtbf62hf8.256(<32 x i8> %{{.*}})
+  // CHECK: call <32 x i8> @llvm.x86.avx10.vcvtbf62hf8256(<32 x i8> %{{.*}})
   return _mm256_cvtbf6_hf8(__A);
 }
 
 __m256i test_mm256_mask_cvtbf6_hf8(__m256i __W, __mmask32 __U, __m256i __A) {
   // CHECK-LABEL: @test_mm256_mask_cvtbf6_hf8(
-  // CHECK: [[RES:%.*]] = call <32 x i8> @llvm.x86.avx10.vcvtbf62hf8.256(<32 x i8> %{{.*}})
+  // CHECK: [[RES:%.*]] = call <32 x i8> @llvm.x86.avx10.vcvtbf62hf8256(<32 x i8> %{{.*}})
   // CHECK: select <32 x i1> %{{.*}}, <32 x i8> [[RES]], <32 x i8> %{{.*}}
   return _mm256_mask_cvtbf6_hf8(__W, __U, __A);
 }
 
 __m256i test_mm256_maskz_cvtbf6_hf8(__mmask32 __U, __m256i __A) {
   // CHECK-LABEL: @test_mm256_maskz_cvtbf6_hf8(
-  // CHECK: [[RES:%.*]] = call <32 x i8> @llvm.x86.avx10.vcvtbf62hf8.256(<32 x i8> %{{.*}})
+  // CHECK: [[RES:%.*]] = call <32 x i8> @llvm.x86.avx10.vcvtbf62hf8256(<32 x i8> %{{.*}})
   // CHECK: zeroinitializer
   // CHECK: select <32 x i1> %{{.*}}, <32 x i8> [[RES]], <32 x i8> %{{.*}}
   return _mm256_maskz_cvtbf6_hf8(__U, __A);
 }
 
-// VCVTBF62HF8 - 512-bit
-
 __m512i test_mm512_cvtbf6_hf8(__m512i __A) {
   // CHECK-LABEL: @test_mm512_cvtbf6_hf8(
-  // CHECK: call <64 x i8> @llvm.x86.avx10.vcvtbf62hf8.512(<64 x i8> %{{.*}})
+  // CHECK: call <64 x i8> @llvm.x86.avx10.vcvtbf62hf8512(<64 x i8> %{{.*}})
   return _mm512_cvtbf6_hf8(__A);
 }
 
 __m512i test_mm512_mask_cvtbf6_hf8(__m512i __W, __mmask64 __U, __m512i __A) {
   // CHECK-LABEL: @test_mm512_mask_cvtbf6_hf8(
-  // CHECK: [[RES:%.*]] = call <64 x i8> @llvm.x86.avx10.vcvtbf62hf8.512(<64 x i8> %{{.*}})
+  // CHECK: [[RES:%.*]] = call <64 x i8> @llvm.x86.avx10.vcvtbf62hf8512(<64 x i8> %{{.*}})
   // CHECK: select <64 x i1> %{{.*}}, <64 x i8> [[RES]], <64 x i8> %{{.*}}
   return _mm512_mask_cvtbf6_hf8(__W, __U, __A);
 }
 
 __m512i test_mm512_maskz_cvtbf6_hf8(__mmask64 __U, __m512i __A) {
   // CHECK-LABEL: @test_mm512_maskz_cvtbf6_hf8(
-  // CHECK: [[RES:%.*]] = call <64 x i8> @llvm.x86.avx10.vcvtbf62hf8.512(<64 x i8> %{{.*}})
+  // CHECK: [[RES:%.*]] = call <64 x i8> @llvm.x86.avx10.vcvtbf62hf8512(<64 x i8> %{{.*}})
   // CHECK: zeroinitializer
   // CHECK: select <64 x i1> %{{.*}}, <64 x i8> [[RES]], <64 x i8> %{{.*}}
   return _mm512_maskz_cvtbf6_hf8(__U, __A);
 }
 
-// VCVTHF62HF8 - 128-bit
-
 __m128i test_mm_cvthf6_hf8(__m128i __A) {
   // CHECK-LABEL: @test_mm_cvthf6_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvthf62hf8.128(<16 x i8> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvthf62hf8128(<16 x i8> %{{.*}})
   return _mm_cvthf6_hf8(__A);
 }
 
 __m128i test_mm_mask_cvthf6_hf8(__m128i __W, __mmask16 __U, __m128i __A) {
   // CHECK-LABEL: @test_mm_mask_cvthf6_hf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvthf62hf8.128(<16 x i8> %{{.*}})
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvthf62hf8128(<16 x i8> %{{.*}})
   // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm_mask_cvthf6_hf8(__W, __U, __A);
 }
 
 __m128i test_mm_maskz_cvthf6_hf8(__mmask16 __U, __m128i __A) {
   // CHECK-LABEL: @test_mm_maskz_cvthf6_hf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvthf62hf8.128(<16 x i8> %{{.*}})
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvthf62hf8128(<16 x i8> %{{.*}})
   // CHECK: zeroinitializer
   // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm_maskz_cvthf6_hf8(__U, __A);
 }
 
-// VCVTHF62HF8 - 256-bit
-
 __m256i test_mm256_cvthf6_hf8(__m256i __A) {
   // CHECK-LABEL: @test_mm256_cvthf6_hf8(
-  // CHECK: call <32 x i8> @llvm.x86.avx10.vcvthf62hf8.256(<32 x i8> %{{.*}})
+  // CHECK: call <32 x i8> @llvm.x86.avx10.vcvthf62hf8256(<32 x i8> %{{.*}})
   return _mm256_cvthf6_hf8(__A);
 }
 
 __m256i test_mm256_mask_cvthf6_hf8(__m256i __W, __mmask32 __U, __m256i __A) {
   // CHECK-LABEL: @test_mm256_mask_cvthf6_hf8(
-  // CHECK: [[RES:%.*]] = call <32 x i8> @llvm.x86.avx10.vcvthf62hf8.256(<32 x i8> %{{.*}})
+  // CHECK: [[RES:%.*]] = call <32 x i8> @llvm.x86.avx10.vcvthf62hf8256(<32 x i8> %{{.*}})
   // CHECK: select <32 x i1> %{{.*}}, <32 x i8> [[RES]], <32 x i8> %{{.*}}
   return _mm256_mask_cvthf6_hf8(__W, __U, __A);
 }
 
 __m256i test_mm256_maskz_cvthf6_hf8(__mmask32 __U, __m256i __A) {
   // CHECK-LABEL: @test_mm256_maskz_cvthf6_hf8(
-  // CHECK: [[RES:%.*]] = call <32 x i8> @llvm.x86.avx10.vcvthf62hf8.256(<32 x i8> %{{.*}})
+  // CHECK: [[RES:%.*]] = call <32 x i8> @llvm.x86.avx10.vcvthf62hf8256(<32 x i8> %{{.*}})
   // CHECK: zeroinitializer
   // CHECK: select <32 x i1> %{{.*}}, <32 x i8> [[RES]], <32 x i8> %{{.*}}
   return _mm256_maskz_cvthf6_hf8(__U, __A);
 }
 
-// VCVTHF62HF8 - 512-bit
-
 __m512i test_mm512_cvthf6_hf8(__m512i __A) {
   // CHECK-LABEL: @test_mm512_cvthf6_hf8(
-  // CHECK: call <64 x i8> @llvm.x86.avx10.vcvthf62hf8.512(<64 x i8> %{{.*}})
+  // CHECK: call <64 x i8> @llvm.x86.avx10.vcvthf62hf8512(<64 x i8> %{{.*}})
   return _mm512_cvthf6_hf8(__A);
 }
 
 __m512i test_mm512_mask_cvthf6_hf8(__m512i __W, __mmask64 __U, __m512i __A) {
   // CHECK-LABEL: @test_mm512_mask_cvthf6_hf8(
-  // CHECK: [[RES:%.*]] = call <64 x i8> @llvm.x86.avx10.vcvthf62hf8.512(<64 x i8> %{{.*}})
+  // CHECK: [[RES:%.*]] = call <64 x i8> @llvm.x86.avx10.vcvthf62hf8512(<64 x i8> %{{.*}})
   // CHECK: select <64 x i1> %{{.*}}, <64 x i8> [[RES]], <64 x i8> %{{.*}}
   return _mm512_mask_cvthf6_hf8(__W, __U, __A);
 }
 
 __m512i test_mm512_maskz_cvthf6_hf8(__mmask64 __U, __m512i __A) {
   // CHECK-LABEL: @test_mm512_maskz_cvthf6_hf8(
-  // CHECK: [[RES:%.*]] = call <64 x i8> @llvm.x86.avx10.vcvthf62hf8.512(<64 x i8> %{{.*}})
+  // CHECK: [[RES:%.*]] = call <64 x i8> @llvm.x86.avx10.vcvthf62hf8512(<64 x i8> %{{.*}})
   // CHECK: zeroinitializer
   // CHECK: select <64 x i1> %{{.*}}, <64 x i8> [[RES]], <64 x i8> %{{.*}}
   return _mm512_maskz_cvthf6_hf8(__U, __A);
 }
 
-//
-// Unpack to Byte
-// VUNPACKB
-//
-
-// VUNPACKB - 128-bit
-
 __m128i test_mm_unpack_epi8(__m128i __A) {
   // CHECK-LABEL: @test_mm_unpack_epi8(
   // CHECK: call <16 x i8> @llvm.x86.avx10.vunpackb.128(<16 x i8> %{{.*}}, i8 1)
@@ -1185,8 +1003,6 @@ __m128i test_mm_maskz_unpack_epi8(__mmask16 __U, __m128i __A) {
   return _mm_maskz_unpack_epi8(__U, __A, 1);
 }
 
-// VUNPACKB - 256-bit
-
 __m256i test_mm256_unpack_epi8(__m256i __A) {
   // CHECK-LABEL: @test_mm256_unpack_epi8(
   // CHECK: call <32 x i8> @llvm.x86.avx10.vunpackb.256(<32 x i8> %{{.*}}, i8 2)
@@ -1208,8 +1024,6 @@ __m256i test_mm256_maskz_unpack_epi8(__mmask32 __U, __m256i __A) {
   return _mm256_maskz_unpack_epi8(__U, __A, 2);
 }
 
-// VUNPACKB - 512-bit
-
 __m512i test_mm512_unpack_epi8(__m512i __A) {
   // CHECK-LABEL: @test_mm512_unpack_epi8(
   // CHECK: call <64 x i8> @llvm.x86.avx10.vunpackb.512(<64 x i8> %{{.*}}, i8 3)
@@ -1231,8 +1045,6 @@ __m512i test_mm512_maskz_unpack_epi8(__mmask64 __U, __m512i __A) {
   return _mm512_maskz_unpack_epi8(__U, __A, 3);
 }
 
-// VUNPACKB - immediate composition macros
-
 __m512i test_mm512_unpack_epi8_compose(__m512i __A) {
   // CHECK-LABEL: @test_mm512_unpack_epi8_compose(
   // CHECK: call <64 x i8> @llvm.x86.avx10.vunpackb.512(<64 x i8> %{{.*}}, i8 45)
@@ -1252,90 +1064,77 @@ __m128i test_mm_unpack_epi8_compose(__m128i __A) {
   return _mm_unpack_epi8(__A, _MM_UNPACKB_SIZE(7) | _MM_UNPACKB_SEXT);
 }
 
-//
-// Down convert DWord to Byte with symmetric signed saturation
-// VPMOVSSDB
-//
-
-// VPMOVSSDB - 128-bit
-
-__m128i test_mm_cvtssepi32_epi8(__m128i __A) {
-  // CHECK-LABEL: @test_mm_cvtssepi32_epi8(
+__m128i test_mm_cvtss_epi32_epi8(__m128i __A) {
+  // CHECK-LABEL: @test_mm_cvtss_epi32_epi8(
   // CHECK: call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.128(<4 x i32> %{{.*}}, <16 x i8> %{{.*}}, i8 -1)
-  return _mm_cvtssepi32_epi8(__A);
+  return _mm_cvtss_epi32_epi8(__A);
 }
 
-__m128i test_mm_mask_cvtssepi32_epi8(__m128i __W, __mmask8 __U, __m128i __A) {
-  // CHECK-LABEL: @test_mm_mask_cvtssepi32_epi8(
+__m128i test_mm_mask_cvtss_epi32_epi8(__m128i __W, __mmask8 __U, __m128i __A) {
+  // CHECK-LABEL: @test_mm_mask_cvtss_epi32_epi8(
   // CHECK: call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.128(<4 x i32> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
-  return _mm_mask_cvtssepi32_epi8(__W, __U, __A);
+  return _mm_mask_cvtss_epi32_epi8(__W, __U, __A);
 }
 
-__m128i test_mm_maskz_cvtssepi32_epi8(__mmask8 __U, __m128i __A) {
-  // CHECK-LABEL: @test_mm_maskz_cvtssepi32_epi8(
+__m128i test_mm_maskz_cvtss_epi32_epi8(__mmask8 __U, __m128i __A) {
+  // CHECK-LABEL: @test_mm_maskz_cvtss_epi32_epi8(
   // CHECK: zeroinitializer
   // CHECK: call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.128(<4 x i32> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
-  return _mm_maskz_cvtssepi32_epi8(__U, __A);
+  return _mm_maskz_cvtss_epi32_epi8(__U, __A);
 }
 
-// VPMOVSSDB - 256-bit
-
-__m128i test_mm256_cvtssepi32_epi8(__m256i __A) {
-  // CHECK-LABEL: @test_mm256_cvtssepi32_epi8(
+__m128i test_mm256_cvtss_epi32_epi8(__m256i __A) {
+  // CHECK-LABEL: @test_mm256_cvtss_epi32_epi8(
   // CHECK: call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.256(<8 x i32> %{{.*}}, <16 x i8> %{{.*}}, i8 -1)
-  return _mm256_cvtssepi32_epi8(__A);
+  return _mm256_cvtss_epi32_epi8(__A);
 }
 
-__m128i test_mm256_mask_cvtssepi32_epi8(__m128i __W, __mmask8 __U, __m256i __A) {
-  // CHECK-LABEL: @test_mm256_mask_cvtssepi32_epi8(
+__m128i test_mm256_mask_cvtss_epi32_epi8(__m128i __W, __mmask8 __U, __m256i __A) {
+  // CHECK-LABEL: @test_mm256_mask_cvtss_epi32_epi8(
   // CHECK: call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.256(<8 x i32> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
-  return _mm256_mask_cvtssepi32_epi8(__W, __U, __A);
+  return _mm256_mask_cvtss_epi32_epi8(__W, __U, __A);
 }
 
-__m128i test_mm256_maskz_cvtssepi32_epi8(__mmask8 __U, __m256i __A) {
-  // CHECK-LABEL: @test_mm256_maskz_cvtssepi32_epi8(
+__m128i test_mm256_maskz_cvtss_epi32_epi8(__mmask8 __U, __m256i __A) {
+  // CHECK-LABEL: @test_mm256_maskz_cvtss_epi32_epi8(
   // CHECK: zeroinitializer
   // CHECK: call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.256(<8 x i32> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
-  return _mm256_maskz_cvtssepi32_epi8(__U, __A);
+  return _mm256_maskz_cvtss_epi32_epi8(__U, __A);
 }
 
-// VPMOVSSDB - 512-bit
-
-__m128i test_mm512_cvtssepi32_epi8(__m512i __A) {
-  // CHECK-LABEL: @test_mm512_cvtssepi32_epi8(
+__m128i test_mm512_cvtss_epi32_epi8(__m512i __A) {
+  // CHECK-LABEL: @test_mm512_cvtss_epi32_epi8(
   // CHECK: call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.512(<16 x i32> %{{.*}}, <16 x i8> %{{.*}}, i16 -1)
-  return _mm512_cvtssepi32_epi8(__A);
+  return _mm512_cvtss_epi32_epi8(__A);
 }
 
-__m128i test_mm512_mask_cvtssepi32_epi8(__m128i __W, __mmask16 __U, __m512i __A) {
-  // CHECK-LABEL: @test_mm512_mask_cvtssepi32_epi8(
+__m128i test_mm512_mask_cvtss_epi32_epi8(__m128i __W, __mmask16 __U, __m512i __A) {
+  // CHECK-LABEL: @test_mm512_mask_cvtss_epi32_epi8(
   // CHECK: call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.512(<16 x i32> %{{.*}}, <16 x i8> %{{.*}}, i16 %{{.*}})
-  return _mm512_mask_cvtssepi32_epi8(__W, __U, __A);
+  return _mm512_mask_cvtss_epi32_epi8(__W, __U, __A);
 }
 
-__m128i test_mm512_maskz_cvtssepi32_epi8(__mmask16 __U, __m512i __A) {
-  // CHECK-LABEL: @test_mm512_maskz_cvtssepi32_epi8(
+__m128i test_mm512_maskz_cvtss_epi32_epi8(__mmask16 __U, __m512i __A) {
+  // CHECK-LABEL: @test_mm512_maskz_cvtss_epi32_epi8(
   // CHECK: zeroinitializer
   // CHECK: call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.512(<16 x i32> %{{.*}}, <16 x i8> %{{.*}}, i16 %{{.*}})
-  return _mm512_maskz_cvtssepi32_epi8(__U, __A);
+  return _mm512_maskz_cvtss_epi32_epi8(__U, __A);
 }
 
-// VPMOVSSDB - memory store
-
-void test_mm_mask_cvtssepi32_storeu_epi8(void *__P, __mmask8 __M, __m128i __A) {
-  // CHECK-LABEL: @test_mm_mask_cvtssepi32_storeu_epi8(
+void test_mm_mask_cvtss_epi32_storeu_epi8(void *__P, __mmask8 __M, __m128i __A) {
+  // CHECK-LABEL: @test_mm_mask_cvtss_epi32_storeu_epi8(
   // CHECK: call void @llvm.x86.avx10.mask.pmovss.db.mem.128(ptr %{{.*}}, <4 x i32> %{{.*}}, i8 %{{.*}})
-  _mm_mask_cvtssepi32_storeu_epi8(__P, __M, __A);
+  _mm_mask_cvtss_epi32_storeu_epi8(__P, __M, __A);
 }
 
-void test_mm256_mask_cvtssepi32_storeu_epi8(void *__P, __mmask8 __M, __m256i __A) {
-  // CHECK-LABEL: @test_mm256_mask_cvtssepi32_storeu_epi8(
+void test_mm256_mask_cvtss_epi32_storeu_epi8(void *__P, __mmask8 __M, __m256i __A) {
+  // CHECK-LABEL: @test_mm256_mask_cvtss_epi32_storeu_epi8(
   // CHECK: call void @llvm.x86.avx10.mask.pmovss.db.mem.256(ptr %{{.*}}, <8 x i32> %{{.*}}, i8 %{{.*}})
-  _mm256_mask_cvtssepi32_storeu_epi8(__P, __M, __A);
+  _mm256_mask_cvtss_epi32_storeu_epi8(__P, __M, __A);
 }
 
-void test_mm512_mask_cvtssepi32_storeu_epi8(void *__P, __mmask16 __M, __m512i __A) {
-  // CHECK-LABEL: @test_mm512_mask_cvtssepi32_storeu_epi8(
+void test_mm512_mask_cvtss_epi32_storeu_epi8(void *__P, __mmask16 __M, __m512i __A) {
+  // CHECK-LABEL: @test_mm512_mask_cvtss_epi32_storeu_epi8(
   // CHECK: call void @llvm.x86.avx10.mask.pmovss.db.mem.512(ptr %{{.*}}, <16 x i32> %{{.*}}, i16 %{{.*}})
-  _mm512_mask_cvtssepi32_storeu_epi8(__P, __M, __A);
+  _mm512_mask_cvtss_epi32_storeu_epi8(__P, __M, __A);
 }
diff --git a/llvm/include/llvm/IR/IntrinsicsX86.td b/llvm/include/llvm/IR/IntrinsicsX86.td
index fef29301704a51..35d82cb6b26f14 100644
--- a/llvm/include/llvm/IR/IntrinsicsX86.td
+++ b/llvm/include/llvm/IR/IntrinsicsX86.td
@@ -7052,108 +7052,108 @@ let TargetPrefix = "x86" in {
 // Convert from FP32 to FP8
 
 // VCVTPS2BF8
-def int_x86_avx10_vcvtps2bf8_128 : ClangBuiltin<"__builtin_ia32_vcvtps2bf8_128">,
+def int_x86_avx10_vcvtps2bf8128 : ClangBuiltin<"__builtin_ia32_vcvtps2bf8_128">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v4f32_ty], [IntrNoMem]>;
-def int_x86_avx10_vcvtps2bf8_256 : ClangBuiltin<"__builtin_ia32_vcvtps2bf8_256">,
+def int_x86_avx10_vcvtps2bf8256 : ClangBuiltin<"__builtin_ia32_vcvtps2bf8_256">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v8f32_ty], [IntrNoMem]>;
-def int_x86_avx10_vcvtps2bf8_512 : ClangBuiltin<"__builtin_ia32_vcvtps2bf8_512">,
+def int_x86_avx10_vcvtps2bf8512 : ClangBuiltin<"__builtin_ia32_vcvtps2bf8_512">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16f32_ty], [IntrNoMem]>;
-def int_x86_avx10_mask_vcvtps2bf8_128 :
+def int_x86_avx10_mask_vcvtps2bf8128 :
         ClangBuiltin<"__builtin_ia32_vcvtps2bf8_128_mask">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty],
                               [llvm_v4f32_ty, llvm_v16i8_ty, llvm_i8_ty],
                               [IntrNoMem]>;
-def int_x86_avx10_mask_vcvtps2bf8_256 :
+def int_x86_avx10_mask_vcvtps2bf8256 :
         ClangBuiltin<"__builtin_ia32_vcvtps2bf8_256_mask">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty],
                               [llvm_v8f32_ty, llvm_v16i8_ty, llvm_i8_ty],
                               [IntrNoMem]>;
 
 // VCVTPS2BF8S
-def int_x86_avx10_vcvtps2bf8s_128 : ClangBuiltin<"__builtin_ia32_vcvtps2bf8s_128">,
+def int_x86_avx10_vcvtps2bf8s128 : ClangBuiltin<"__builtin_ia32_vcvtps2bf8s_128">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v4f32_ty], [IntrNoMem]>;
-def int_x86_avx10_vcvtps2bf8s_256 : ClangBuiltin<"__builtin_ia32_vcvtps2bf8s_256">,
+def int_x86_avx10_vcvtps2bf8s256 : ClangBuiltin<"__builtin_ia32_vcvtps2bf8s_256">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v8f32_ty], [IntrNoMem]>;
-def int_x86_avx10_vcvtps2bf8s_512 : ClangBuiltin<"__builtin_ia32_vcvtps2bf8s_512">,
+def int_x86_avx10_vcvtps2bf8s512 : ClangBuiltin<"__builtin_ia32_vcvtps2bf8s_512">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16f32_ty], [IntrNoMem]>;
-def int_x86_avx10_mask_vcvtps2bf8s_128 :
+def int_x86_avx10_mask_vcvtps2bf8s128 :
         ClangBuiltin<"__builtin_ia32_vcvtps2bf8s_128_mask">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty],
                               [llvm_v4f32_ty, llvm_v16i8_ty, llvm_i8_ty],
                               [IntrNoMem]>;
-def int_x86_avx10_mask_vcvtps2bf8s_256 :
+def int_x86_avx10_mask_vcvtps2bf8s256 :
         ClangBuiltin<"__builtin_ia32_vcvtps2bf8s_256_mask">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty],
                               [llvm_v8f32_ty, llvm_v16i8_ty, llvm_i8_ty],
                               [IntrNoMem]>;
 
 // VCVTPS2HF8
-def int_x86_avx10_vcvtps2hf8_128 : ClangBuiltin<"__builtin_ia32_vcvtps2hf8_128">,
+def int_x86_avx10_vcvtps2hf8128 : ClangBuiltin<"__builtin_ia32_vcvtps2hf8_128">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v4f32_ty], [IntrNoMem]>;
-def int_x86_avx10_vcvtps2hf8_256 : ClangBuiltin<"__builtin_ia32_vcvtps2hf8_256">,
+def int_x86_avx10_vcvtps2hf8256 : ClangBuiltin<"__builtin_ia32_vcvtps2hf8_256">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v8f32_ty], [IntrNoMem]>;
-def int_x86_avx10_vcvtps2hf8_512 : ClangBuiltin<"__builtin_ia32_vcvtps2hf8_512">,
+def int_x86_avx10_vcvtps2hf8512 : ClangBuiltin<"__builtin_ia32_vcvtps2hf8_512">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16f32_ty], [IntrNoMem]>;
-def int_x86_avx10_mask_vcvtps2hf8_128 :
+def int_x86_avx10_mask_vcvtps2hf8128 :
         ClangBuiltin<"__builtin_ia32_vcvtps2hf8_128_mask">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty],
                               [llvm_v4f32_ty, llvm_v16i8_ty, llvm_i8_ty],
                               [IntrNoMem]>;
-def int_x86_avx10_mask_vcvtps2hf8_256 :
+def int_x86_avx10_mask_vcvtps2hf8256 :
         ClangBuiltin<"__builtin_ia32_vcvtps2hf8_256_mask">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty],
                               [llvm_v8f32_ty, llvm_v16i8_ty, llvm_i8_ty],
                               [IntrNoMem]>;
 
 // VCVTPS2HF8S
-def int_x86_avx10_vcvtps2hf8s_128 : ClangBuiltin<"__builtin_ia32_vcvtps2hf8s_128">,
+def int_x86_avx10_vcvtps2hf8s128 : ClangBuiltin<"__builtin_ia32_vcvtps2hf8s_128">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v4f32_ty], [IntrNoMem]>;
-def int_x86_avx10_vcvtps2hf8s_256 : ClangBuiltin<"__builtin_ia32_vcvtps2hf8s_256">,
+def int_x86_avx10_vcvtps2hf8s256 : ClangBuiltin<"__builtin_ia32_vcvtps2hf8s_256">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v8f32_ty], [IntrNoMem]>;
-def int_x86_avx10_vcvtps2hf8s_512 : ClangBuiltin<"__builtin_ia32_vcvtps2hf8s_512">,
+def int_x86_avx10_vcvtps2hf8s512 : ClangBuiltin<"__builtin_ia32_vcvtps2hf8s_512">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16f32_ty], [IntrNoMem]>;
-def int_x86_avx10_mask_vcvtps2hf8s_128 :
+def int_x86_avx10_mask_vcvtps2hf8s128 :
         ClangBuiltin<"__builtin_ia32_vcvtps2hf8s_128_mask">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty],
                               [llvm_v4f32_ty, llvm_v16i8_ty, llvm_i8_ty],
                               [IntrNoMem]>;
-def int_x86_avx10_mask_vcvtps2hf8s_256 :
+def int_x86_avx10_mask_vcvtps2hf8s256 :
         ClangBuiltin<"__builtin_ia32_vcvtps2hf8s_256_mask">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty],
                               [llvm_v8f32_ty, llvm_v16i8_ty, llvm_i8_ty],
                               [IntrNoMem]>;
 
 // VCVTROPS2HF8
-def int_x86_avx10_vcvtrops2hf8_128 : ClangBuiltin<"__builtin_ia32_vcvtrops2hf8_128">,
+def int_x86_avx10_vcvtrops2hf8128 : ClangBuiltin<"__builtin_ia32_vcvtrops2hf8_128">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v4f32_ty], [IntrNoMem]>;
-def int_x86_avx10_vcvtrops2hf8_256 : ClangBuiltin<"__builtin_ia32_vcvtrops2hf8_256">,
+def int_x86_avx10_vcvtrops2hf8256 : ClangBuiltin<"__builtin_ia32_vcvtrops2hf8_256">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v8f32_ty], [IntrNoMem]>;
-def int_x86_avx10_vcvtrops2hf8_512 : ClangBuiltin<"__builtin_ia32_vcvtrops2hf8_512">,
+def int_x86_avx10_vcvtrops2hf8512 : ClangBuiltin<"__builtin_ia32_vcvtrops2hf8_512">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16f32_ty], [IntrNoMem]>;
-def int_x86_avx10_mask_vcvtrops2hf8_128 :
+def int_x86_avx10_mask_vcvtrops2hf8128 :
         ClangBuiltin<"__builtin_ia32_vcvtrops2hf8_128_mask">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty],
                               [llvm_v4f32_ty, llvm_v16i8_ty, llvm_i8_ty],
                               [IntrNoMem]>;
-def int_x86_avx10_mask_vcvtrops2hf8_256 :
+def int_x86_avx10_mask_vcvtrops2hf8256 :
         ClangBuiltin<"__builtin_ia32_vcvtrops2hf8_256_mask">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty],
                               [llvm_v8f32_ty, llvm_v16i8_ty, llvm_i8_ty],
                               [IntrNoMem]>;
 
 // VCVTROPS2HF8S
-def int_x86_avx10_vcvtrops2hf8s_128 : ClangBuiltin<"__builtin_ia32_vcvtrops2hf8s_128">,
+def int_x86_avx10_vcvtrops2hf8s128 : ClangBuiltin<"__builtin_ia32_vcvtrops2hf8s_128">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v4f32_ty], [IntrNoMem]>;
-def int_x86_avx10_vcvtrops2hf8s_256 : ClangBuiltin<"__builtin_ia32_vcvtrops2hf8s_256">,
+def int_x86_avx10_vcvtrops2hf8s256 : ClangBuiltin<"__builtin_ia32_vcvtrops2hf8s_256">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v8f32_ty], [IntrNoMem]>;
-def int_x86_avx10_vcvtrops2hf8s_512 : ClangBuiltin<"__builtin_ia32_vcvtrops2hf8s_512">,
+def int_x86_avx10_vcvtrops2hf8s512 : ClangBuiltin<"__builtin_ia32_vcvtrops2hf8s_512">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16f32_ty], [IntrNoMem]>;
-def int_x86_avx10_mask_vcvtrops2hf8s_128 :
+def int_x86_avx10_mask_vcvtrops2hf8s128 :
         ClangBuiltin<"__builtin_ia32_vcvtrops2hf8s_128_mask">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty],
                               [llvm_v4f32_ty, llvm_v16i8_ty, llvm_i8_ty],
                               [IntrNoMem]>;
-def int_x86_avx10_mask_vcvtrops2hf8s_256 :
+def int_x86_avx10_mask_vcvtrops2hf8s256 :
         ClangBuiltin<"__builtin_ia32_vcvtrops2hf8s_256_mask">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty],
                               [llvm_v8f32_ty, llvm_v16i8_ty, llvm_i8_ty],
@@ -7162,19 +7162,19 @@ def int_x86_avx10_mask_vcvtrops2hf8s_256 :
 // Convert from FP32 to FP8 with bias
 
 // VCVTBIASPS2BF8
-def int_x86_avx10_vcvtbiasps2bf8_128 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2bf8_128">,
+def int_x86_avx10_vcvtbiasps2bf8128 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2bf8_128">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v4i32_ty, llvm_v4f32_ty], [IntrNoMem]>;
-def int_x86_avx10_vcvtbiasps2bf8_256 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2bf8_256">,
+def int_x86_avx10_vcvtbiasps2bf8256 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2bf8_256">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v8i32_ty, llvm_v8f32_ty], [IntrNoMem]>;
-def int_x86_avx10_vcvtbiasps2bf8_512 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2bf8_512">,
+def int_x86_avx10_vcvtbiasps2bf8512 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2bf8_512">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16i32_ty, llvm_v16f32_ty], [IntrNoMem]>;
-def int_x86_avx10_mask_vcvtbiasps2bf8_128 :
+def int_x86_avx10_mask_vcvtbiasps2bf8128 :
         ClangBuiltin<"__builtin_ia32_vcvtbiasps2bf8_128_mask">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty],
                               [llvm_v4i32_ty, llvm_v4f32_ty,
                                llvm_v16i8_ty, llvm_i8_ty],
                               [IntrNoMem]>;
-def int_x86_avx10_mask_vcvtbiasps2bf8_256 :
+def int_x86_avx10_mask_vcvtbiasps2bf8256 :
         ClangBuiltin<"__builtin_ia32_vcvtbiasps2bf8_256_mask">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty],
                               [llvm_v8i32_ty, llvm_v8f32_ty,
@@ -7182,19 +7182,19 @@ def int_x86_avx10_mask_vcvtbiasps2bf8_256 :
                               [IntrNoMem]>;
 
 // VCVTBIASPS2BF8S
-def int_x86_avx10_vcvtbiasps2bf8s_128 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2bf8s_128">,
+def int_x86_avx10_vcvtbiasps2bf8s128 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2bf8s_128">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v4i32_ty, llvm_v4f32_ty], [IntrNoMem]>;
-def int_x86_avx10_vcvtbiasps2bf8s_256 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2bf8s_256">,
+def int_x86_avx10_vcvtbiasps2bf8s256 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2bf8s_256">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v8i32_ty, llvm_v8f32_ty], [IntrNoMem]>;
-def int_x86_avx10_vcvtbiasps2bf8s_512 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2bf8s_512">,
+def int_x86_avx10_vcvtbiasps2bf8s512 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2bf8s_512">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16i32_ty, llvm_v16f32_ty], [IntrNoMem]>;
-def int_x86_avx10_mask_vcvtbiasps2bf8s_128 :
+def int_x86_avx10_mask_vcvtbiasps2bf8s128 :
         ClangBuiltin<"__builtin_ia32_vcvtbiasps2bf8s_128_mask">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty],
                               [llvm_v4i32_ty, llvm_v4f32_ty,
                                llvm_v16i8_ty, llvm_i8_ty],
                               [IntrNoMem]>;
-def int_x86_avx10_mask_vcvtbiasps2bf8s_256 :
+def int_x86_avx10_mask_vcvtbiasps2bf8s256 :
         ClangBuiltin<"__builtin_ia32_vcvtbiasps2bf8s_256_mask">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty],
                               [llvm_v8i32_ty, llvm_v8f32_ty,
@@ -7202,19 +7202,19 @@ def int_x86_avx10_mask_vcvtbiasps2bf8s_256 :
                               [IntrNoMem]>;
 
 // VCVTBIASPS2HF8
-def int_x86_avx10_vcvtbiasps2hf8_128 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2hf8_128">,
+def int_x86_avx10_vcvtbiasps2hf8128 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2hf8_128">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v4i32_ty, llvm_v4f32_ty], [IntrNoMem]>;
-def int_x86_avx10_vcvtbiasps2hf8_256 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2hf8_256">,
+def int_x86_avx10_vcvtbiasps2hf8256 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2hf8_256">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v8i32_ty, llvm_v8f32_ty], [IntrNoMem]>;
-def int_x86_avx10_vcvtbiasps2hf8_512 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2hf8_512">,
+def int_x86_avx10_vcvtbiasps2hf8512 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2hf8_512">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16i32_ty, llvm_v16f32_ty], [IntrNoMem]>;
-def int_x86_avx10_mask_vcvtbiasps2hf8_128 :
+def int_x86_avx10_mask_vcvtbiasps2hf8128 :
         ClangBuiltin<"__builtin_ia32_vcvtbiasps2hf8_128_mask">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty],
                               [llvm_v4i32_ty, llvm_v4f32_ty,
                                llvm_v16i8_ty, llvm_i8_ty],
                               [IntrNoMem]>;
-def int_x86_avx10_mask_vcvtbiasps2hf8_256 :
+def int_x86_avx10_mask_vcvtbiasps2hf8256 :
         ClangBuiltin<"__builtin_ia32_vcvtbiasps2hf8_256_mask">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty],
                               [llvm_v8i32_ty, llvm_v8f32_ty,
@@ -7222,19 +7222,19 @@ def int_x86_avx10_mask_vcvtbiasps2hf8_256 :
                               [IntrNoMem]>;
 
 // VCVTBIASPS2HF8S
-def int_x86_avx10_vcvtbiasps2hf8s_128 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2hf8s_128">,
+def int_x86_avx10_vcvtbiasps2hf8s128 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2hf8s_128">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v4i32_ty, llvm_v4f32_ty], [IntrNoMem]>;
-def int_x86_avx10_vcvtbiasps2hf8s_256 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2hf8s_256">,
+def int_x86_avx10_vcvtbiasps2hf8s256 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2hf8s_256">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v8i32_ty, llvm_v8f32_ty], [IntrNoMem]>;
-def int_x86_avx10_vcvtbiasps2hf8s_512 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2hf8s_512">,
+def int_x86_avx10_vcvtbiasps2hf8s512 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2hf8s_512">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16i32_ty, llvm_v16f32_ty], [IntrNoMem]>;
-def int_x86_avx10_mask_vcvtbiasps2hf8s_128 :
+def int_x86_avx10_mask_vcvtbiasps2hf8s128 :
         ClangBuiltin<"__builtin_ia32_vcvtbiasps2hf8s_128_mask">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty],
                               [llvm_v4i32_ty, llvm_v4f32_ty,
                                llvm_v16i8_ty, llvm_i8_ty],
                               [IntrNoMem]>;
-def int_x86_avx10_mask_vcvtbiasps2hf8s_256 :
+def int_x86_avx10_mask_vcvtbiasps2hf8s256 :
         ClangBuiltin<"__builtin_ia32_vcvtbiasps2hf8s_256_mask">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty],
                               [llvm_v8i32_ty, llvm_v8f32_ty,
@@ -7244,105 +7244,83 @@ def int_x86_avx10_mask_vcvtbiasps2hf8s_256 :
 // Convert from FP8 to FP32
 
 // VCVTBF82PS
-def int_x86_avx10_vcvtbf82ps_128 : ClangBuiltin<"__builtin_ia32_vcvtbf82ps_128">,
+def int_x86_avx10_vcvtbf82ps128 : ClangBuiltin<"__builtin_ia32_vcvtbf82ps_128">,
         DefaultAttrsIntrinsic<[llvm_v4f32_ty], [llvm_v16i8_ty], [IntrNoMem]>;
-def int_x86_avx10_vcvtbf82ps_256 : ClangBuiltin<"__builtin_ia32_vcvtbf82ps_256">,
+def int_x86_avx10_vcvtbf82ps256 : ClangBuiltin<"__builtin_ia32_vcvtbf82ps_256">,
         DefaultAttrsIntrinsic<[llvm_v8f32_ty], [llvm_v16i8_ty], [IntrNoMem]>;
-def int_x86_avx10_vcvtbf82ps_512 : ClangBuiltin<"__builtin_ia32_vcvtbf82ps_512">,
+def int_x86_avx10_vcvtbf82ps512 : ClangBuiltin<"__builtin_ia32_vcvtbf82ps_512">,
         DefaultAttrsIntrinsic<[llvm_v16f32_ty], [llvm_v16i8_ty], [IntrNoMem]>;
 
 // VCVTHF82PS
-def int_x86_avx10_vcvthf82ps_128 : ClangBuiltin<"__builtin_ia32_vcvthf82ps_128">,
+def int_x86_avx10_vcvthf82ps128 : ClangBuiltin<"__builtin_ia32_vcvthf82ps_128">,
         DefaultAttrsIntrinsic<[llvm_v4f32_ty], [llvm_v16i8_ty], [IntrNoMem]>;
-def int_x86_avx10_vcvthf82ps_256 : ClangBuiltin<"__builtin_ia32_vcvthf82ps_256">,
+def int_x86_avx10_vcvthf82ps256 : ClangBuiltin<"__builtin_ia32_vcvthf82ps_256">,
         DefaultAttrsIntrinsic<[llvm_v8f32_ty], [llvm_v16i8_ty], [IntrNoMem]>;
-def int_x86_avx10_vcvthf82ps_512 : ClangBuiltin<"__builtin_ia32_vcvthf82ps_512">,
+def int_x86_avx10_vcvthf82ps512 : ClangBuiltin<"__builtin_ia32_vcvthf82ps_512">,
         DefaultAttrsIntrinsic<[llvm_v16f32_ty], [llvm_v16i8_ty], [IntrNoMem]>;
 
 // Convert from FP8 to FP6
 
 // VCVTBF82BF6S
-def int_x86_avx10_vcvtbf82bf6s_128 : ClangBuiltin<"__builtin_ia32_vcvtbf82bf6s_128">,
+def int_x86_avx10_vcvtbf82bf6s128 : ClangBuiltin<"__builtin_ia32_vcvtbf82bf6s_128">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16i8_ty], [IntrNoMem]>;
-def int_x86_avx10_vcvtbf82bf6s_256 : ClangBuiltin<"__builtin_ia32_vcvtbf82bf6s_256">,
+def int_x86_avx10_vcvtbf82bf6s256 : ClangBuiltin<"__builtin_ia32_vcvtbf82bf6s_256">,
         DefaultAttrsIntrinsic<[llvm_v32i8_ty], [llvm_v32i8_ty], [IntrNoMem]>;
-def int_x86_avx10_vcvtbf82bf6s_512 : ClangBuiltin<"__builtin_ia32_vcvtbf82bf6s_512">,
+def int_x86_avx10_vcvtbf82bf6s512 : ClangBuiltin<"__builtin_ia32_vcvtbf82bf6s_512">,
         DefaultAttrsIntrinsic<[llvm_v64i8_ty], [llvm_v64i8_ty], [IntrNoMem]>;
 
 // VCVTHF82HF6S
-def int_x86_avx10_vcvthf82hf6s_128 : ClangBuiltin<"__builtin_ia32_vcvthf82hf6s_128">,
+def int_x86_avx10_vcvthf82hf6s128 : ClangBuiltin<"__builtin_ia32_vcvthf82hf6s_128">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16i8_ty], [IntrNoMem]>;
-def int_x86_avx10_vcvthf82hf6s_256 : ClangBuiltin<"__builtin_ia32_vcvthf82hf6s_256">,
+def int_x86_avx10_vcvthf82hf6s256 : ClangBuiltin<"__builtin_ia32_vcvthf82hf6s_256">,
         DefaultAttrsIntrinsic<[llvm_v32i8_ty], [llvm_v32i8_ty], [IntrNoMem]>;
-def int_x86_avx10_vcvthf82hf6s_512 : ClangBuiltin<"__builtin_ia32_vcvthf82hf6s_512">,
+def int_x86_avx10_vcvthf82hf6s512 : ClangBuiltin<"__builtin_ia32_vcvthf82hf6s_512">,
         DefaultAttrsIntrinsic<[llvm_v64i8_ty], [llvm_v64i8_ty], [IntrNoMem]>;
 
 // Convert from FP4 to FP8 and from FP6 to FP8
 
 // VCVTBF42HF8
-def int_x86_avx10_vcvtbf42hf8_128 : ClangBuiltin<"__builtin_ia32_vcvtbf42hf8_128">,
+def int_x86_avx10_vcvtbf42hf8128 : ClangBuiltin<"__builtin_ia32_vcvtbf42hf8_128">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16i8_ty], [IntrNoMem]>;
-def int_x86_avx10_vcvtbf42hf8_256 : ClangBuiltin<"__builtin_ia32_vcvtbf42hf8_256">,
+def int_x86_avx10_vcvtbf42hf8256 : ClangBuiltin<"__builtin_ia32_vcvtbf42hf8_256">,
         DefaultAttrsIntrinsic<[llvm_v32i8_ty], [llvm_v16i8_ty], [IntrNoMem]>;
-def int_x86_avx10_vcvtbf42hf8_512 : ClangBuiltin<"__builtin_ia32_vcvtbf42hf8_512">,
+def int_x86_avx10_vcvtbf42hf8512 : ClangBuiltin<"__builtin_ia32_vcvtbf42hf8_512">,
         DefaultAttrsIntrinsic<[llvm_v64i8_ty], [llvm_v32i8_ty], [IntrNoMem]>;
 
 // VCVTBF62HF8
-def int_x86_avx10_vcvtbf62hf8_128 : ClangBuiltin<"__builtin_ia32_vcvtbf62hf8_128">,
+def int_x86_avx10_vcvtbf62hf8128 : ClangBuiltin<"__builtin_ia32_vcvtbf62hf8_128">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16i8_ty], [IntrNoMem]>;
-def int_x86_avx10_vcvtbf62hf8_256 : ClangBuiltin<"__builtin_ia32_vcvtbf62hf8_256">,
+def int_x86_avx10_vcvtbf62hf8256 : ClangBuiltin<"__builtin_ia32_vcvtbf62hf8_256">,
         DefaultAttrsIntrinsic<[llvm_v32i8_ty], [llvm_v32i8_ty], [IntrNoMem]>;
-def int_x86_avx10_vcvtbf62hf8_512 : ClangBuiltin<"__builtin_ia32_vcvtbf62hf8_512">,
+def int_x86_avx10_vcvtbf62hf8512 : ClangBuiltin<"__builtin_ia32_vcvtbf62hf8_512">,
         DefaultAttrsIntrinsic<[llvm_v64i8_ty], [llvm_v64i8_ty], [IntrNoMem]>;
 
 // VCVTHF62HF8
-def int_x86_avx10_vcvthf62hf8_128 : ClangBuiltin<"__builtin_ia32_vcvthf62hf8_128">,
+def int_x86_avx10_vcvthf62hf8128 : ClangBuiltin<"__builtin_ia32_vcvthf62hf8_128">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16i8_ty], [IntrNoMem]>;
-def int_x86_avx10_vcvthf62hf8_256 : ClangBuiltin<"__builtin_ia32_vcvthf62hf8_256">,
+def int_x86_avx10_vcvthf62hf8256 : ClangBuiltin<"__builtin_ia32_vcvthf62hf8_256">,
         DefaultAttrsIntrinsic<[llvm_v32i8_ty], [llvm_v32i8_ty], [IntrNoMem]>;
-def int_x86_avx10_vcvthf62hf8_512 : ClangBuiltin<"__builtin_ia32_vcvthf62hf8_512">,
+def int_x86_avx10_vcvthf62hf8512 : ClangBuiltin<"__builtin_ia32_vcvthf62hf8_512">,
         DefaultAttrsIntrinsic<[llvm_v64i8_ty], [llvm_v64i8_ty], [IntrNoMem]>;
 
 // Convert from FP8 to FP4
 
 // VCVTBF82BF4S
-def int_x86_avx10_vcvtbf82bf4s_128 : ClangBuiltin<"__builtin_ia32_vcvtbf82bf4s_128">,
+def int_x86_avx10_vcvtbf82bf4s128 : ClangBuiltin<"__builtin_ia32_vcvtbf82bf4s_128">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16i8_ty], [IntrNoMem]>;
-def int_x86_avx10_vcvtbf82bf4s_256 : ClangBuiltin<"__builtin_ia32_vcvtbf82bf4s_256">,
+def int_x86_avx10_vcvtbf82bf4s256 : ClangBuiltin<"__builtin_ia32_vcvtbf82bf4s_256">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v32i8_ty], [IntrNoMem]>;
-def int_x86_avx10_vcvtbf82bf4s_512 : ClangBuiltin<"__builtin_ia32_vcvtbf82bf4s_512">,
+def int_x86_avx10_vcvtbf82bf4s512 : ClangBuiltin<"__builtin_ia32_vcvtbf82bf4s_512">,
         DefaultAttrsIntrinsic<[llvm_v32i8_ty], [llvm_v64i8_ty], [IntrNoMem]>;
 
 // VCVTHF82BF4S
-def int_x86_avx10_vcvthf82bf4s_128 : ClangBuiltin<"__builtin_ia32_vcvthf82bf4s_128">,
+def int_x86_avx10_vcvthf82bf4s128 : ClangBuiltin<"__builtin_ia32_vcvthf82bf4s_128">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16i8_ty], [IntrNoMem]>;
-def int_x86_avx10_vcvthf82bf4s_256 : ClangBuiltin<"__builtin_ia32_vcvthf82bf4s_256">,
+def int_x86_avx10_vcvthf82bf4s256 : ClangBuiltin<"__builtin_ia32_vcvthf82bf4s_256">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v32i8_ty], [IntrNoMem]>;
-def int_x86_avx10_vcvthf82bf4s_512 : ClangBuiltin<"__builtin_ia32_vcvthf82bf4s_512">,
+def int_x86_avx10_vcvthf82bf4s512 : ClangBuiltin<"__builtin_ia32_vcvthf82bf4s_512">,
         DefaultAttrsIntrinsic<[llvm_v32i8_ty], [llvm_v64i8_ty], [IntrNoMem]>;
 
-// VCVTBF82BF4S memory store
-def int_x86_avx10_vcvtbf82bf4s_128_mem : ClangBuiltin<"__builtin_ia32_vcvtbf82bf4s_128_mem">,
-        DefaultAttrsIntrinsic<[], [llvm_ptr_ty, llvm_v16i8_ty],
-                              [IntrWriteMem, IntrArgMemOnly, NoCapture<ArgIndex<0>>]>;
-def int_x86_avx10_vcvtbf82bf4s_256_mem : ClangBuiltin<"__builtin_ia32_vcvtbf82bf4s_256_mem">,
-        DefaultAttrsIntrinsic<[], [llvm_ptr_ty, llvm_v32i8_ty],
-                              [IntrWriteMem, IntrArgMemOnly, NoCapture<ArgIndex<0>>]>;
-def int_x86_avx10_vcvtbf82bf4s_512_mem : ClangBuiltin<"__builtin_ia32_vcvtbf82bf4s_512_mem">,
-        DefaultAttrsIntrinsic<[], [llvm_ptr_ty, llvm_v64i8_ty],
-                              [IntrWriteMem, IntrArgMemOnly, NoCapture<ArgIndex<0>>]>;
-
-// VCVTHF82BF4S memory store
-def int_x86_avx10_vcvthf82bf4s_128_mem : ClangBuiltin<"__builtin_ia32_vcvthf82bf4s_128_mem">,
-        DefaultAttrsIntrinsic<[], [llvm_ptr_ty, llvm_v16i8_ty],
-                              [IntrWriteMem, IntrArgMemOnly, NoCapture<ArgIndex<0>>]>;
-def int_x86_avx10_vcvthf82bf4s_256_mem : ClangBuiltin<"__builtin_ia32_vcvthf82bf4s_256_mem">,
-        DefaultAttrsIntrinsic<[], [llvm_ptr_ty, llvm_v32i8_ty],
-                              [IntrWriteMem, IntrArgMemOnly, NoCapture<ArgIndex<0>>]>;
-def int_x86_avx10_vcvthf82bf4s_512_mem : ClangBuiltin<"__builtin_ia32_vcvthf82bf4s_512_mem">,
-        DefaultAttrsIntrinsic<[], [llvm_ptr_ty, llvm_v64i8_ty],
-                              [IntrWriteMem, IntrArgMemOnly, NoCapture<ArgIndex<0>>]>;
-
 // Unpack to Byte
 
 def int_x86_avx10_vunpackb_128 : ClangBuiltin<"__builtin_ia32_vunpackb128">,
diff --git a/llvm/lib/Target/X86/X86InstrAVX10_V2_AUX.td b/llvm/lib/Target/X86/X86InstrAVX10_V2_AUX.td
index 2b8851f49a1ffe..0e09d7db4b1f01 100644
--- a/llvm/lib/Target/X86/X86InstrAVX10_V2_AUX.td
+++ b/llvm/lib/Target/X86/X86InstrAVX10_V2_AUX.td
@@ -256,12 +256,9 @@ multiclass avx10_v2aux_cvt_trunc_base<bits<8> opc, string OpcodeStr,
                Sched<[sched.Folded]>, EVEX, EVEX_CD8<8, CD8VH>;
   }
 
-  def : Pat<(_dest.VT (!cast<Intrinsic>("int_x86_avx10_"#OpcodeStr#"_"#_src.Size)
+  def : Pat<(_dest.VT (!cast<Intrinsic>("int_x86_avx10_"#OpcodeStr#_src.Size)
                        (_src.VT _src.RC:$src))),
             (!cast<Instruction>(NAME # "rr") _src.RC:$src)>;
-  def : Pat<(!cast<Intrinsic>("int_x86_avx10_"#OpcodeStr#"_"#_src.Size#"_mem")
-             addr:$dst, (_src.VT _src.RC:$src)),
-            (!cast<Instruction>(NAME # "mr") addr:$dst, _src.RC:$src)>;
 }
 
 multiclass avx10_v2aux_cvt_trunc_b<bits<8> opc, string OpcodeStr> {
@@ -287,14 +284,14 @@ multiclass avx10_v2aux_cvt_expand_masked_base<bits<8> opc, string OpcodeStr,
     defm rr : AVX512_maskable<opc, MRMSrcReg, _, (outs _.RC:$dst),
                               (ins _src.RC:$src),
                               OpcodeStr, "$src", "$src",
-                              (_.VT (!cast<Intrinsic>("int_x86_avx10_"#OpcodeStr#"_"#_.Size)
+                              (_.VT (!cast<Intrinsic>("int_x86_avx10_"#OpcodeStr#_.Size)
                                      (_src.VT _src.RC:$src)))>,
                               Sched<[sched]>, EVEX, EVEX_CD8<8, CD8VH>;
     let mayLoad = 1 in
       defm rm : AVX512_maskable<opc, MRMSrcMem, _, (outs _.RC:$dst),
                                 (ins x86memop:$src),
                                 OpcodeStr, "$src", "$src",
-                                (_.VT (!cast<Intrinsic>("int_x86_avx10_"#OpcodeStr#"_"#_.Size)
+                                (_.VT (!cast<Intrinsic>("int_x86_avx10_"#OpcodeStr#_.Size)
                                        (_src.VT ld_dag)))>,
                                 Sched<[sched.Folded]>, EVEX, EVEX_CD8<8, CD8VH>;
   }
@@ -323,7 +320,7 @@ multiclass avx10_v2aux_cvt_widen_masked_base<bits<8> opc, string OpcodeStr,
     defm rr : AVX512_maskable<opc, MRMSrcReg, _, (outs _.RC:$dst),
                               (ins _.RC:$src),
                               OpcodeStr, "$src", "$src",
-                              (_.VT (!cast<Intrinsic>("int_x86_avx10_"#OpcodeStr#"_"#_.Size)
+                              (_.VT (!cast<Intrinsic>("int_x86_avx10_"#OpcodeStr#_.Size)
                                      (_.VT _.RC:$src)))>,
                               Sched<[sched]>, EVEX;
   }
@@ -442,6 +439,34 @@ let Predicates = [HasAVX10_V2_AUX] in {
                       T_MAP5, XS, REX_W;
   defm VCVTHF82BF4S : avx10_v2aux_cvt_trunc_b<0x3D, "vcvthf82bf4s">,
                       T_MAP5, XS;
+
+  def : Pat<(store (int_x86_avx10_vcvtbf82bf4s512 VR512:$src), addr:$dst),
+            (VCVTBF82BF4SZmr addr:$dst, VR512:$src)>;
+  def : Pat<(store (int_x86_avx10_vcvtbf82bf4s256 VR256X:$src), addr:$dst),
+            (VCVTBF82BF4SZ256mr addr:$dst, VR256X:$src)>;
+  // The 128-bit form writes only the low 8 bytes of its destination.
+  def : Pat<(store (i64 (extractelt
+                         (bc_v2i64 (int_x86_avx10_vcvtbf82bf4s128 VR128X:$src)),
+                         (iPTR 0))), addr:$dst),
+            (VCVTBF82BF4SZ128mr addr:$dst, VR128X:$src)>;
+  def : Pat<(store (f64 (extractelt
+                         (bc_v2f64 (int_x86_avx10_vcvtbf82bf4s128 VR128X:$src)),
+                         (iPTR 0))), addr:$dst),
+            (VCVTBF82BF4SZ128mr addr:$dst, VR128X:$src)>;
+
+  def : Pat<(store (int_x86_avx10_vcvthf82bf4s512 VR512:$src), addr:$dst),
+            (VCVTHF82BF4SZmr addr:$dst, VR512:$src)>;
+  def : Pat<(store (int_x86_avx10_vcvthf82bf4s256 VR256X:$src), addr:$dst),
+            (VCVTHF82BF4SZ256mr addr:$dst, VR256X:$src)>;
+  // The 128-bit form writes only the low 8 bytes of its destination.
+  def : Pat<(store (i64 (extractelt
+                         (bc_v2i64 (int_x86_avx10_vcvthf82bf4s128 VR128X:$src)),
+                         (iPTR 0))), addr:$dst),
+            (VCVTHF82BF4SZ128mr addr:$dst, VR128X:$src)>;
+  def : Pat<(store (f64 (extractelt
+                         (bc_v2f64 (int_x86_avx10_vcvthf82bf4s128 VR128X:$src)),
+                         (iPTR 0))), addr:$dst),
+            (VCVTHF82BF4SZ128mr addr:$dst, VR128X:$src)>;
 }
 
 //-------------------------------------------------
@@ -456,7 +481,7 @@ multiclass avx10_v2aux_cvt_narrow_base<bits<8> opc, string OpcodeStr,
              (ins _.RC:$src),
              OpcodeStr # "\t{$src, $dst|$dst, $src}",
              [(set _.RC:$dst,
-               (_.VT (!cast<Intrinsic>("int_x86_avx10_"#OpcodeStr#"_"#_.Size)
+               (_.VT (!cast<Intrinsic>("int_x86_avx10_"#OpcodeStr#_.Size)
                        (_.VT _.RC:$src))))]>,
            EVEX, Sched<[sched]>;
 }
@@ -488,7 +513,7 @@ let Predicates = [HasAVX10_V2_AUX] in {
                      T_MAP5, PS;
 
   // Pattern match vcvtbf42hf8 of a scalar i64 load.
-  def : Pat<(v16i8 (int_x86_avx10_vcvtbf42hf8_128 (v16i8 (bitconvert
+  def : Pat<(v16i8 (int_x86_avx10_vcvtbf42hf8128 (v16i8 (bitconvert
               (v2i64 (scalar_to_vector (loadi64 addr:$src))))))),
             (VCVTBF42HF8Z128rm addr:$src)>;
 }
diff --git a/llvm/lib/Target/X86/X86IntrinsicsInfo.h b/llvm/lib/Target/X86/X86IntrinsicsInfo.h
index 79c99e7c841c7e..213c508b4a8b07 100644
--- a/llvm/lib/Target/X86/X86IntrinsicsInfo.h
+++ b/llvm/lib/Target/X86/X86IntrinsicsInfo.h
@@ -480,21 +480,21 @@ static const IntrinsicData IntrinsicsWithoutChain[] = {
                        X86ISD::VCVTBIASPH2HF8S, 0),
     X86_INTRINSIC_DATA(avx10_mask_vcvtbiasph2hf8s512, INTR_TYPE_2OP_MASK,
                        X86ISD::VCVTBIASPH2HF8S, 0),
-    X86_INTRINSIC_DATA(avx10_mask_vcvtbiasps2bf8_128, TRUNCATE2_TO_REG,
+    X86_INTRINSIC_DATA(avx10_mask_vcvtbiasps2bf8128, TRUNCATE2_TO_REG,
                        X86ISD::VCVTBIASPS2BF8, X86ISD::VMCVTBIASPS2BF8),
-    X86_INTRINSIC_DATA(avx10_mask_vcvtbiasps2bf8_256, TRUNCATE2_TO_REG,
+    X86_INTRINSIC_DATA(avx10_mask_vcvtbiasps2bf8256, TRUNCATE2_TO_REG,
                        X86ISD::VCVTBIASPS2BF8, X86ISD::VMCVTBIASPS2BF8),
-    X86_INTRINSIC_DATA(avx10_mask_vcvtbiasps2bf8s_128, TRUNCATE2_TO_REG,
+    X86_INTRINSIC_DATA(avx10_mask_vcvtbiasps2bf8s128, TRUNCATE2_TO_REG,
                        X86ISD::VCVTBIASPS2BF8S, X86ISD::VMCVTBIASPS2BF8S),
-    X86_INTRINSIC_DATA(avx10_mask_vcvtbiasps2bf8s_256, TRUNCATE2_TO_REG,
+    X86_INTRINSIC_DATA(avx10_mask_vcvtbiasps2bf8s256, TRUNCATE2_TO_REG,
                        X86ISD::VCVTBIASPS2BF8S, X86ISD::VMCVTBIASPS2BF8S),
-    X86_INTRINSIC_DATA(avx10_mask_vcvtbiasps2hf8_128, TRUNCATE2_TO_REG,
+    X86_INTRINSIC_DATA(avx10_mask_vcvtbiasps2hf8128, TRUNCATE2_TO_REG,
                        X86ISD::VCVTBIASPS2HF8, X86ISD::VMCVTBIASPS2HF8),
-    X86_INTRINSIC_DATA(avx10_mask_vcvtbiasps2hf8_256, TRUNCATE2_TO_REG,
+    X86_INTRINSIC_DATA(avx10_mask_vcvtbiasps2hf8256, TRUNCATE2_TO_REG,
                        X86ISD::VCVTBIASPS2HF8, X86ISD::VMCVTBIASPS2HF8),
-    X86_INTRINSIC_DATA(avx10_mask_vcvtbiasps2hf8s_128, TRUNCATE2_TO_REG,
+    X86_INTRINSIC_DATA(avx10_mask_vcvtbiasps2hf8s128, TRUNCATE2_TO_REG,
                        X86ISD::VCVTBIASPS2HF8S, X86ISD::VMCVTBIASPS2HF8S),
-    X86_INTRINSIC_DATA(avx10_mask_vcvtbiasps2hf8s_256, TRUNCATE2_TO_REG,
+    X86_INTRINSIC_DATA(avx10_mask_vcvtbiasps2hf8s256, TRUNCATE2_TO_REG,
                        X86ISD::VCVTBIASPS2HF8S, X86ISD::VMCVTBIASPS2HF8S),
     X86_INTRINSIC_DATA(avx10_mask_vcvthf82ph128, INTR_TYPE_1OP_MASK,
                        X86ISD::VCVTHF82PH, 0),
@@ -538,21 +538,21 @@ static const IntrinsicData IntrinsicsWithoutChain[] = {
                        X86ISD::CVTP2IUBS, 0),
     X86_INTRINSIC_DATA(avx10_mask_vcvtph2iubs512, INTR_TYPE_1OP_MASK,
                        X86ISD::CVTP2IUBS, X86ISD::CVTP2IUBS_RND),
-    X86_INTRINSIC_DATA(avx10_mask_vcvtps2bf8_128, TRUNCATE_TO_REG,
+    X86_INTRINSIC_DATA(avx10_mask_vcvtps2bf8128, TRUNCATE_TO_REG,
                        X86ISD::VCVTPS2BF8, X86ISD::VMCVTPS2BF8),
-    X86_INTRINSIC_DATA(avx10_mask_vcvtps2bf8_256, TRUNCATE_TO_REG,
+    X86_INTRINSIC_DATA(avx10_mask_vcvtps2bf8256, TRUNCATE_TO_REG,
                        X86ISD::VCVTPS2BF8, X86ISD::VMCVTPS2BF8),
-    X86_INTRINSIC_DATA(avx10_mask_vcvtps2bf8s_128, TRUNCATE_TO_REG,
+    X86_INTRINSIC_DATA(avx10_mask_vcvtps2bf8s128, TRUNCATE_TO_REG,
                        X86ISD::VCVTPS2BF8S, X86ISD::VMCVTPS2BF8S),
-    X86_INTRINSIC_DATA(avx10_mask_vcvtps2bf8s_256, TRUNCATE_TO_REG,
+    X86_INTRINSIC_DATA(avx10_mask_vcvtps2bf8s256, TRUNCATE_TO_REG,
                        X86ISD::VCVTPS2BF8S, X86ISD::VMCVTPS2BF8S),
-    X86_INTRINSIC_DATA(avx10_mask_vcvtps2hf8_128, TRUNCATE_TO_REG,
+    X86_INTRINSIC_DATA(avx10_mask_vcvtps2hf8128, TRUNCATE_TO_REG,
                        X86ISD::VCVTPS2HF8, X86ISD::VMCVTPS2HF8),
-    X86_INTRINSIC_DATA(avx10_mask_vcvtps2hf8_256, TRUNCATE_TO_REG,
+    X86_INTRINSIC_DATA(avx10_mask_vcvtps2hf8256, TRUNCATE_TO_REG,
                        X86ISD::VCVTPS2HF8, X86ISD::VMCVTPS2HF8),
-    X86_INTRINSIC_DATA(avx10_mask_vcvtps2hf8s_128, TRUNCATE_TO_REG,
+    X86_INTRINSIC_DATA(avx10_mask_vcvtps2hf8s128, TRUNCATE_TO_REG,
                        X86ISD::VCVTPS2HF8S, X86ISD::VMCVTPS2HF8S),
-    X86_INTRINSIC_DATA(avx10_mask_vcvtps2hf8s_256, TRUNCATE_TO_REG,
+    X86_INTRINSIC_DATA(avx10_mask_vcvtps2hf8s256, TRUNCATE_TO_REG,
                        X86ISD::VCVTPS2HF8S, X86ISD::VMCVTPS2HF8S),
     X86_INTRINSIC_DATA(avx10_mask_vcvtps2ibs128, INTR_TYPE_1OP_MASK,
                        X86ISD::CVTP2IBS, 0),
@@ -566,13 +566,13 @@ static const IntrinsicData IntrinsicsWithoutChain[] = {
                        X86ISD::CVTP2IUBS, 0),
     X86_INTRINSIC_DATA(avx10_mask_vcvtps2iubs512, INTR_TYPE_1OP_MASK,
                        X86ISD::CVTP2IUBS, X86ISD::CVTP2IUBS_RND),
-    X86_INTRINSIC_DATA(avx10_mask_vcvtrops2hf8_128, TRUNCATE_TO_REG,
+    X86_INTRINSIC_DATA(avx10_mask_vcvtrops2hf8128, TRUNCATE_TO_REG,
                        X86ISD::VCVTROPS2HF8, X86ISD::VMCVTROPS2HF8),
-    X86_INTRINSIC_DATA(avx10_mask_vcvtrops2hf8_256, TRUNCATE_TO_REG,
+    X86_INTRINSIC_DATA(avx10_mask_vcvtrops2hf8256, TRUNCATE_TO_REG,
                        X86ISD::VCVTROPS2HF8, X86ISD::VMCVTROPS2HF8),
-    X86_INTRINSIC_DATA(avx10_mask_vcvtrops2hf8s_128, TRUNCATE_TO_REG,
+    X86_INTRINSIC_DATA(avx10_mask_vcvtrops2hf8s128, TRUNCATE_TO_REG,
                        X86ISD::VCVTROPS2HF8S, X86ISD::VMCVTROPS2HF8S),
-    X86_INTRINSIC_DATA(avx10_mask_vcvtrops2hf8s_256, TRUNCATE_TO_REG,
+    X86_INTRINSIC_DATA(avx10_mask_vcvtrops2hf8s256, TRUNCATE_TO_REG,
                        X86ISD::VCVTROPS2HF8S, X86ISD::VMCVTROPS2HF8S),
     X86_INTRINSIC_DATA(avx10_mask_vcvttpd2dqs_128, CVTPD2DQ_MASK,
                        X86ISD::CVTTP2SIS, X86ISD::MCVTTP2SIS),
@@ -715,77 +715,77 @@ static const IntrinsicData IntrinsicsWithoutChain[] = {
                        0),
     X86_INTRINSIC_DATA(avx10_vcvtbf162iubs512, INTR_TYPE_1OP, X86ISD::CVTP2IUBS,
                        0),
-    X86_INTRINSIC_DATA(avx10_vcvtbf82ps_128, INTR_TYPE_1OP, X86ISD::VCVTBF82PS,
+    X86_INTRINSIC_DATA(avx10_vcvtbf82ps128, INTR_TYPE_1OP, X86ISD::VCVTBF82PS,
                        0),
-    X86_INTRINSIC_DATA(avx10_vcvtbf82ps_256, INTR_TYPE_1OP, X86ISD::VCVTBF82PS,
+    X86_INTRINSIC_DATA(avx10_vcvtbf82ps256, INTR_TYPE_1OP, X86ISD::VCVTBF82PS,
                        0),
-    X86_INTRINSIC_DATA(avx10_vcvtbf82ps_512, INTR_TYPE_1OP, X86ISD::VCVTBF82PS,
+    X86_INTRINSIC_DATA(avx10_vcvtbf82ps512, INTR_TYPE_1OP, X86ISD::VCVTBF82PS,
                        0),
-    X86_INTRINSIC_DATA(avx10_vcvtbiasps2bf8_128, INTR_TYPE_2OP,
+    X86_INTRINSIC_DATA(avx10_vcvtbiasps2bf8128, INTR_TYPE_2OP,
                        X86ISD::VCVTBIASPS2BF8, 0),
-    X86_INTRINSIC_DATA(avx10_vcvtbiasps2bf8_256, INTR_TYPE_2OP,
+    X86_INTRINSIC_DATA(avx10_vcvtbiasps2bf8256, INTR_TYPE_2OP,
                        X86ISD::VCVTBIASPS2BF8, 0),
-    X86_INTRINSIC_DATA(avx10_vcvtbiasps2bf8_512, INTR_TYPE_2OP,
+    X86_INTRINSIC_DATA(avx10_vcvtbiasps2bf8512, INTR_TYPE_2OP,
                        X86ISD::VCVTBIASPS2BF8, 0),
-    X86_INTRINSIC_DATA(avx10_vcvtbiasps2bf8s_128, INTR_TYPE_2OP,
+    X86_INTRINSIC_DATA(avx10_vcvtbiasps2bf8s128, INTR_TYPE_2OP,
                        X86ISD::VCVTBIASPS2BF8S, 0),
-    X86_INTRINSIC_DATA(avx10_vcvtbiasps2bf8s_256, INTR_TYPE_2OP,
+    X86_INTRINSIC_DATA(avx10_vcvtbiasps2bf8s256, INTR_TYPE_2OP,
                        X86ISD::VCVTBIASPS2BF8S, 0),
-    X86_INTRINSIC_DATA(avx10_vcvtbiasps2bf8s_512, INTR_TYPE_2OP,
+    X86_INTRINSIC_DATA(avx10_vcvtbiasps2bf8s512, INTR_TYPE_2OP,
                        X86ISD::VCVTBIASPS2BF8S, 0),
-    X86_INTRINSIC_DATA(avx10_vcvtbiasps2hf8_128, INTR_TYPE_2OP,
+    X86_INTRINSIC_DATA(avx10_vcvtbiasps2hf8128, INTR_TYPE_2OP,
                        X86ISD::VCVTBIASPS2HF8, 0),
-    X86_INTRINSIC_DATA(avx10_vcvtbiasps2hf8_256, INTR_TYPE_2OP,
+    X86_INTRINSIC_DATA(avx10_vcvtbiasps2hf8256, INTR_TYPE_2OP,
                        X86ISD::VCVTBIASPS2HF8, 0),
-    X86_INTRINSIC_DATA(avx10_vcvtbiasps2hf8_512, INTR_TYPE_2OP,
+    X86_INTRINSIC_DATA(avx10_vcvtbiasps2hf8512, INTR_TYPE_2OP,
                        X86ISD::VCVTBIASPS2HF8, 0),
-    X86_INTRINSIC_DATA(avx10_vcvtbiasps2hf8s_128, INTR_TYPE_2OP,
+    X86_INTRINSIC_DATA(avx10_vcvtbiasps2hf8s128, INTR_TYPE_2OP,
                        X86ISD::VCVTBIASPS2HF8S, 0),
-    X86_INTRINSIC_DATA(avx10_vcvtbiasps2hf8s_256, INTR_TYPE_2OP,
+    X86_INTRINSIC_DATA(avx10_vcvtbiasps2hf8s256, INTR_TYPE_2OP,
                        X86ISD::VCVTBIASPS2HF8S, 0),
-    X86_INTRINSIC_DATA(avx10_vcvtbiasps2hf8s_512, INTR_TYPE_2OP,
+    X86_INTRINSIC_DATA(avx10_vcvtbiasps2hf8s512, INTR_TYPE_2OP,
                        X86ISD::VCVTBIASPS2HF8S, 0),
-    X86_INTRINSIC_DATA(avx10_vcvthf82ps_128, INTR_TYPE_1OP, X86ISD::VCVTHF82PS,
+    X86_INTRINSIC_DATA(avx10_vcvthf82ps128, INTR_TYPE_1OP, X86ISD::VCVTHF82PS,
                        0),
-    X86_INTRINSIC_DATA(avx10_vcvthf82ps_256, INTR_TYPE_1OP, X86ISD::VCVTHF82PS,
+    X86_INTRINSIC_DATA(avx10_vcvthf82ps256, INTR_TYPE_1OP, X86ISD::VCVTHF82PS,
                        0),
-    X86_INTRINSIC_DATA(avx10_vcvthf82ps_512, INTR_TYPE_1OP, X86ISD::VCVTHF82PS,
+    X86_INTRINSIC_DATA(avx10_vcvthf82ps512, INTR_TYPE_1OP, X86ISD::VCVTHF82PS,
                        0),
-    X86_INTRINSIC_DATA(avx10_vcvtps2bf8_128, INTR_TYPE_1OP, X86ISD::VCVTPS2BF8,
+    X86_INTRINSIC_DATA(avx10_vcvtps2bf8128, INTR_TYPE_1OP, X86ISD::VCVTPS2BF8,
                        0),
-    X86_INTRINSIC_DATA(avx10_vcvtps2bf8_256, INTR_TYPE_1OP, X86ISD::VCVTPS2BF8,
+    X86_INTRINSIC_DATA(avx10_vcvtps2bf8256, INTR_TYPE_1OP, X86ISD::VCVTPS2BF8,
                        0),
-    X86_INTRINSIC_DATA(avx10_vcvtps2bf8_512, INTR_TYPE_1OP, X86ISD::VCVTPS2BF8,
+    X86_INTRINSIC_DATA(avx10_vcvtps2bf8512, INTR_TYPE_1OP, X86ISD::VCVTPS2BF8,
                        0),
-    X86_INTRINSIC_DATA(avx10_vcvtps2bf8s_128, INTR_TYPE_1OP,
-                       X86ISD::VCVTPS2BF8S, 0),
-    X86_INTRINSIC_DATA(avx10_vcvtps2bf8s_256, INTR_TYPE_1OP,
-                       X86ISD::VCVTPS2BF8S, 0),
-    X86_INTRINSIC_DATA(avx10_vcvtps2bf8s_512, INTR_TYPE_1OP,
-                       X86ISD::VCVTPS2BF8S, 0),
-    X86_INTRINSIC_DATA(avx10_vcvtps2hf8_128, INTR_TYPE_1OP, X86ISD::VCVTPS2HF8,
+    X86_INTRINSIC_DATA(avx10_vcvtps2bf8s128, INTR_TYPE_1OP, X86ISD::VCVTPS2BF8S,
                        0),
-    X86_INTRINSIC_DATA(avx10_vcvtps2hf8_256, INTR_TYPE_1OP, X86ISD::VCVTPS2HF8,
+    X86_INTRINSIC_DATA(avx10_vcvtps2bf8s256, INTR_TYPE_1OP, X86ISD::VCVTPS2BF8S,
                        0),
-    X86_INTRINSIC_DATA(avx10_vcvtps2hf8_512, INTR_TYPE_1OP, X86ISD::VCVTPS2HF8,
+    X86_INTRINSIC_DATA(avx10_vcvtps2bf8s512, INTR_TYPE_1OP, X86ISD::VCVTPS2BF8S,
                        0),
-    X86_INTRINSIC_DATA(avx10_vcvtps2hf8s_128, INTR_TYPE_1OP,
-                       X86ISD::VCVTPS2HF8S, 0),
-    X86_INTRINSIC_DATA(avx10_vcvtps2hf8s_256, INTR_TYPE_1OP,
-                       X86ISD::VCVTPS2HF8S, 0),
-    X86_INTRINSIC_DATA(avx10_vcvtps2hf8s_512, INTR_TYPE_1OP,
-                       X86ISD::VCVTPS2HF8S, 0),
-    X86_INTRINSIC_DATA(avx10_vcvtrops2hf8_128, INTR_TYPE_1OP,
+    X86_INTRINSIC_DATA(avx10_vcvtps2hf8128, INTR_TYPE_1OP, X86ISD::VCVTPS2HF8,
+                       0),
+    X86_INTRINSIC_DATA(avx10_vcvtps2hf8256, INTR_TYPE_1OP, X86ISD::VCVTPS2HF8,
+                       0),
+    X86_INTRINSIC_DATA(avx10_vcvtps2hf8512, INTR_TYPE_1OP, X86ISD::VCVTPS2HF8,
+                       0),
+    X86_INTRINSIC_DATA(avx10_vcvtps2hf8s128, INTR_TYPE_1OP, X86ISD::VCVTPS2HF8S,
+                       0),
+    X86_INTRINSIC_DATA(avx10_vcvtps2hf8s256, INTR_TYPE_1OP, X86ISD::VCVTPS2HF8S,
+                       0),
+    X86_INTRINSIC_DATA(avx10_vcvtps2hf8s512, INTR_TYPE_1OP, X86ISD::VCVTPS2HF8S,
+                       0),
+    X86_INTRINSIC_DATA(avx10_vcvtrops2hf8128, INTR_TYPE_1OP,
                        X86ISD::VCVTROPS2HF8, 0),
-    X86_INTRINSIC_DATA(avx10_vcvtrops2hf8_256, INTR_TYPE_1OP,
+    X86_INTRINSIC_DATA(avx10_vcvtrops2hf8256, INTR_TYPE_1OP,
                        X86ISD::VCVTROPS2HF8, 0),
-    X86_INTRINSIC_DATA(avx10_vcvtrops2hf8_512, INTR_TYPE_1OP,
+    X86_INTRINSIC_DATA(avx10_vcvtrops2hf8512, INTR_TYPE_1OP,
                        X86ISD::VCVTROPS2HF8, 0),
-    X86_INTRINSIC_DATA(avx10_vcvtrops2hf8s_128, INTR_TYPE_1OP,
+    X86_INTRINSIC_DATA(avx10_vcvtrops2hf8s128, INTR_TYPE_1OP,
                        X86ISD::VCVTROPS2HF8S, 0),
-    X86_INTRINSIC_DATA(avx10_vcvtrops2hf8s_256, INTR_TYPE_1OP,
+    X86_INTRINSIC_DATA(avx10_vcvtrops2hf8s256, INTR_TYPE_1OP,
                        X86ISD::VCVTROPS2HF8S, 0),
-    X86_INTRINSIC_DATA(avx10_vcvtrops2hf8s_512, INTR_TYPE_1OP,
+    X86_INTRINSIC_DATA(avx10_vcvtrops2hf8s512, INTR_TYPE_1OP,
                        X86ISD::VCVTROPS2HF8S, 0),
     X86_INTRINSIC_DATA(avx10_vcvttbf162ibs128, INTR_TYPE_1OP, X86ISD::CVTTP2IBS,
                        0),
diff --git a/llvm/test/CodeGen/X86/avx10_v2aux-intrinsics.ll b/llvm/test/CodeGen/X86/avx10_v2aux-intrinsics.ll
index 7a78236b8e8c69..a52ce4355b7dd1 100644
--- a/llvm/test/CodeGen/X86/avx10_v2aux-intrinsics.ll
+++ b/llvm/test/CodeGen/X86/avx10_v2aux-intrinsics.ll
@@ -2,1062 +2,1281 @@
 ; RUN: llc < %s -verify-machineinstrs -mtriple=x86_64-unknown-unknown --show-mc-encoding -mattr=+avx10v2aux | FileCheck %s --check-prefixes=CHECK,X64
 ; RUN: llc < %s -verify-machineinstrs -mtriple=i686-unknown-unknown --show-mc-encoding -mattr=+avx10v2aux | FileCheck %s --check-prefixes=CHECK,X86
 
-define <16 x i8> @test_int_x86_avx10_vcvtps2bf8_128(<4 x float> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtps2bf8_128:
+define <16 x i8> @test_int_x86_avx10_vcvtps2bf8128(<4 x float> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtps2bf8128:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvtps2bf8 %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x39,0xc0]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8.128(<4 x float> %a)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8128(<4 x float> %a)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtps2bf8_256(<8 x float> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtps2bf8_256:
+define <16 x i8> @test_int_x86_avx10_vcvtps2bf8256(<8 x float> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtps2bf8256:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvtps2bf8 %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x39,0xc0]
 ; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8.256(<8 x float> %a)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8256(<8 x float> %a)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtps2bf8_512(<16 x float> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtps2bf8_512:
+define <16 x i8> @test_int_x86_avx10_vcvtps2bf8512(<16 x float> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtps2bf8512:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvtps2bf8 %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x39,0xc0]
 ; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8.512(<16 x float> %a)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8512(<16 x float> %a)
   ret <16 x i8> %ret
 }
 
-declare <16 x i8> @llvm.x86.avx10.vcvtps2bf8.128(<4 x float>)
-declare <16 x i8> @llvm.x86.avx10.vcvtps2bf8.256(<8 x float>)
-declare <16 x i8> @llvm.x86.avx10.vcvtps2bf8.512(<16 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtps2bf8128(<4 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtps2bf8256(<8 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtps2bf8512(<16 x float>)
 
-; Memory folding tests for vcvtps2bf8
-define <16 x i8> @test_int_x86_avx10_vcvtps2bf8_mem_128(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtps2bf8_mem_128:
+define <16 x i8> @test_int_x86_avx10_vcvtps2bf8128_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtps2bf8128_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vcvtps2bf8x (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x39,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_vcvtps2bf8_mem_128:
+; X86-LABEL: test_int_x86_avx10_vcvtps2bf8128_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    vcvtps2bf8x (%eax), %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x39,0x00]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %a = load <4 x float>, ptr %ptr_a
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8.128(<4 x float> %a)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8128(<4 x float> %a)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtps2bf8_mem_256(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtps2bf8_mem_256:
+define <16 x i8> @test_int_x86_avx10_vcvtps2bf8256_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtps2bf8256_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vcvtps2bf8y (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x39,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_vcvtps2bf8_mem_256:
+; X86-LABEL: test_int_x86_avx10_vcvtps2bf8256_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    vcvtps2bf8y (%eax), %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x39,0x00]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %a = load <8 x float>, ptr %ptr_a
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8.256(<8 x float> %a)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8256(<8 x float> %a)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtps2bf8_mem_512(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtps2bf8_mem_512:
+define <16 x i8> @test_int_x86_avx10_vcvtps2bf8512_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtps2bf8512_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vcvtps2bf8 (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x39,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_vcvtps2bf8_mem_512:
+; X86-LABEL: test_int_x86_avx10_vcvtps2bf8512_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    vcvtps2bf8 (%eax), %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x39,0x00]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %a = load <16 x float>, ptr %ptr_a
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8.512(<16 x float> %a)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8512(<16 x float> %a)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtps2bf8s_128(<4 x float> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtps2bf8s_128:
+define <16 x i8> @test_int_x86_avx10_vcvtps2bf8s128(<4 x float> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtps2bf8s128:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvtps2bf8s %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x3b,0xc0]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8s.128(<4 x float> %a)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8s128(<4 x float> %a)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtps2bf8s_256(<8 x float> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtps2bf8s_256:
+define <16 x i8> @test_int_x86_avx10_vcvtps2bf8s256(<8 x float> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtps2bf8s256:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvtps2bf8s %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x3b,0xc0]
 ; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8s.256(<8 x float> %a)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8s256(<8 x float> %a)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtps2bf8s_512(<16 x float> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtps2bf8s_512:
+define <16 x i8> @test_int_x86_avx10_vcvtps2bf8s512(<16 x float> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtps2bf8s512:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvtps2bf8s %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x3b,0xc0]
 ; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8s.512(<16 x float> %a)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8s512(<16 x float> %a)
   ret <16 x i8> %ret
 }
 
-declare <16 x i8> @llvm.x86.avx10.vcvtps2bf8s.128(<4 x float>)
-declare <16 x i8> @llvm.x86.avx10.vcvtps2bf8s.256(<8 x float>)
-declare <16 x i8> @llvm.x86.avx10.vcvtps2bf8s.512(<16 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtps2bf8s128(<4 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtps2bf8s256(<8 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtps2bf8s512(<16 x float>)
 
-; Memory folding tests for vcvtps2bf8s
-define <16 x i8> @test_int_x86_avx10_vcvtps2bf8s_mem_128(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtps2bf8s_mem_128:
+define <16 x i8> @test_int_x86_avx10_vcvtps2bf8s128_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtps2bf8s128_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vcvtps2bf8sx (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x3b,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_vcvtps2bf8s_mem_128:
+; X86-LABEL: test_int_x86_avx10_vcvtps2bf8s128_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    vcvtps2bf8sx (%eax), %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x3b,0x00]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %a = load <4 x float>, ptr %ptr_a
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8s.128(<4 x float> %a)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8s128(<4 x float> %a)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtps2bf8s_mem_256(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtps2bf8s_mem_256:
+define <16 x i8> @test_int_x86_avx10_vcvtps2bf8s256_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtps2bf8s256_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vcvtps2bf8sy (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x3b,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_vcvtps2bf8s_mem_256:
+; X86-LABEL: test_int_x86_avx10_vcvtps2bf8s256_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    vcvtps2bf8sy (%eax), %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x3b,0x00]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %a = load <8 x float>, ptr %ptr_a
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8s.256(<8 x float> %a)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8s256(<8 x float> %a)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtps2bf8s_mem_512(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtps2bf8s_mem_512:
+define <16 x i8> @test_int_x86_avx10_vcvtps2bf8s512_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtps2bf8s512_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vcvtps2bf8s (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x3b,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_vcvtps2bf8s_mem_512:
+; X86-LABEL: test_int_x86_avx10_vcvtps2bf8s512_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    vcvtps2bf8s (%eax), %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x3b,0x00]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %a = load <16 x float>, ptr %ptr_a
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8s.512(<16 x float> %a)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8s512(<16 x float> %a)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtps2hf8_128(<4 x float> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtps2hf8_128:
+define <16 x i8> @test_int_x86_avx10_vcvtps2hf8128(<4 x float> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtps2hf8128:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvtps2hf8 %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x38,0xc0]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8.128(<4 x float> %a)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8128(<4 x float> %a)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtps2hf8_256(<8 x float> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtps2hf8_256:
+define <16 x i8> @test_int_x86_avx10_vcvtps2hf8256(<8 x float> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtps2hf8256:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvtps2hf8 %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x38,0xc0]
 ; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8.256(<8 x float> %a)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8256(<8 x float> %a)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtps2hf8_512(<16 x float> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtps2hf8_512:
+define <16 x i8> @test_int_x86_avx10_vcvtps2hf8512(<16 x float> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtps2hf8512:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvtps2hf8 %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x38,0xc0]
 ; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8.512(<16 x float> %a)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8512(<16 x float> %a)
   ret <16 x i8> %ret
 }
 
-declare <16 x i8> @llvm.x86.avx10.vcvtps2hf8.128(<4 x float>)
-declare <16 x i8> @llvm.x86.avx10.vcvtps2hf8.256(<8 x float>)
-declare <16 x i8> @llvm.x86.avx10.vcvtps2hf8.512(<16 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtps2hf8128(<4 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtps2hf8256(<8 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtps2hf8512(<16 x float>)
 
-; Memory folding tests for vcvtps2hf8
-define <16 x i8> @test_int_x86_avx10_vcvtps2hf8_mem_128(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtps2hf8_mem_128:
+define <16 x i8> @test_int_x86_avx10_vcvtps2hf8128_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtps2hf8128_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vcvtps2hf8x (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x38,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_vcvtps2hf8_mem_128:
+; X86-LABEL: test_int_x86_avx10_vcvtps2hf8128_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    vcvtps2hf8x (%eax), %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x38,0x00]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %a = load <4 x float>, ptr %ptr_a
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8.128(<4 x float> %a)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8128(<4 x float> %a)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtps2hf8_mem_256(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtps2hf8_mem_256:
+define <16 x i8> @test_int_x86_avx10_vcvtps2hf8256_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtps2hf8256_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vcvtps2hf8y (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x38,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_vcvtps2hf8_mem_256:
+; X86-LABEL: test_int_x86_avx10_vcvtps2hf8256_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    vcvtps2hf8y (%eax), %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x38,0x00]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %a = load <8 x float>, ptr %ptr_a
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8.256(<8 x float> %a)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8256(<8 x float> %a)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtps2hf8_mem_512(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtps2hf8_mem_512:
+define <16 x i8> @test_int_x86_avx10_vcvtps2hf8512_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtps2hf8512_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vcvtps2hf8 (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x38,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_vcvtps2hf8_mem_512:
+; X86-LABEL: test_int_x86_avx10_vcvtps2hf8512_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    vcvtps2hf8 (%eax), %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x38,0x00]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %a = load <16 x float>, ptr %ptr_a
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8.512(<16 x float> %a)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8512(<16 x float> %a)
   ret <16 x i8> %ret
 }
 
-declare <16 x i8> @llvm.x86.avx10.vcvtps2hf8s.128(<4 x float>)
-declare <16 x i8> @llvm.x86.avx10.vcvtps2hf8s.256(<8 x float>)
-declare <16 x i8> @llvm.x86.avx10.vcvtps2hf8s.512(<16 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtps2hf8s128(<4 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtps2hf8s256(<8 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtps2hf8s512(<16 x float>)
 
-define <16 x i8> @test_int_x86_avx10_vcvtps2hf8s_128(<4 x float> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtps2hf8s_128:
+define <16 x i8> @test_int_x86_avx10_vcvtps2hf8s128(<4 x float> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtps2hf8s128:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvtps2hf8s %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x3a,0xc0]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8s.128(<4 x float> %a)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8s128(<4 x float> %a)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtps2hf8s_256(<8 x float> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtps2hf8s_256:
+define <16 x i8> @test_int_x86_avx10_vcvtps2hf8s256(<8 x float> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtps2hf8s256:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvtps2hf8s %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x3a,0xc0]
 ; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8s.256(<8 x float> %a)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8s256(<8 x float> %a)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtps2hf8s_512(<16 x float> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtps2hf8s_512:
+define <16 x i8> @test_int_x86_avx10_vcvtps2hf8s512(<16 x float> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtps2hf8s512:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvtps2hf8s %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x3a,0xc0]
 ; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8s.512(<16 x float> %a)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8s512(<16 x float> %a)
   ret <16 x i8> %ret
 }
 
-; Memory folding tests for vcvtps2hf8s
-define <16 x i8> @test_int_x86_avx10_vcvtps2hf8s_mem_128(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtps2hf8s_mem_128:
+define <16 x i8> @test_int_x86_avx10_vcvtps2hf8s128_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtps2hf8s128_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vcvtps2hf8sx (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x3a,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_vcvtps2hf8s_mem_128:
+; X86-LABEL: test_int_x86_avx10_vcvtps2hf8s128_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    vcvtps2hf8sx (%eax), %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x3a,0x00]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %a = load <4 x float>, ptr %ptr_a
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8s.128(<4 x float> %a)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8s128(<4 x float> %a)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtps2hf8s_mem_256(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtps2hf8s_mem_256:
+define <16 x i8> @test_int_x86_avx10_vcvtps2hf8s256_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtps2hf8s256_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vcvtps2hf8sy (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x3a,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_vcvtps2hf8s_mem_256:
+; X86-LABEL: test_int_x86_avx10_vcvtps2hf8s256_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    vcvtps2hf8sy (%eax), %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x3a,0x00]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %a = load <8 x float>, ptr %ptr_a
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8s.256(<8 x float> %a)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8s256(<8 x float> %a)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtps2hf8s_mem_512(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtps2hf8s_mem_512:
+define <16 x i8> @test_int_x86_avx10_vcvtps2hf8s512_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtps2hf8s512_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vcvtps2hf8s (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x3a,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_vcvtps2hf8s_mem_512:
+; X86-LABEL: test_int_x86_avx10_vcvtps2hf8s512_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    vcvtps2hf8s (%eax), %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x3a,0x00]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %a = load <16 x float>, ptr %ptr_a
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8s.512(<16 x float> %a)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8s512(<16 x float> %a)
   ret <16 x i8> %ret
 }
 
-declare <16 x i8> @llvm.x86.avx10.vcvtrops2hf8.128(<4 x float>)
-declare <16 x i8> @llvm.x86.avx10.vcvtrops2hf8.256(<8 x float>)
-declare <16 x i8> @llvm.x86.avx10.vcvtrops2hf8.512(<16 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtrops2hf8128(<4 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtrops2hf8256(<8 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtrops2hf8512(<16 x float>)
 
-define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8_128(<4 x float> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtrops2hf8_128:
+define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8128(<4 x float> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtrops2hf8128:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvtrops2hf8 %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7d,0x08,0x38,0xc0]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8.128(<4 x float> %a)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8128(<4 x float> %a)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8_256(<8 x float> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtrops2hf8_256:
+define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8256(<8 x float> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtrops2hf8256:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvtrops2hf8 %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7d,0x28,0x38,0xc0]
 ; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8.256(<8 x float> %a)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8256(<8 x float> %a)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8_512(<16 x float> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtrops2hf8_512:
+define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8512(<16 x float> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtrops2hf8512:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvtrops2hf8 %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7d,0x48,0x38,0xc0]
 ; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8.512(<16 x float> %a)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8512(<16 x float> %a)
   ret <16 x i8> %ret
 }
 
-; Memory folding tests for vcvtrops2hf8
-define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8_mem_128(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtrops2hf8_mem_128:
+define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8128_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtrops2hf8128_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vcvtrops2hf8x (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7d,0x08,0x38,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_vcvtrops2hf8_mem_128:
+; X86-LABEL: test_int_x86_avx10_vcvtrops2hf8128_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    vcvtrops2hf8x (%eax), %xmm0 # encoding: [0x62,0xf5,0x7d,0x08,0x38,0x00]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %a = load <4 x float>, ptr %ptr_a
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8.128(<4 x float> %a)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8128(<4 x float> %a)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8_mem_256(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtrops2hf8_mem_256:
+define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8256_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtrops2hf8256_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vcvtrops2hf8y (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7d,0x28,0x38,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_vcvtrops2hf8_mem_256:
+; X86-LABEL: test_int_x86_avx10_vcvtrops2hf8256_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    vcvtrops2hf8y (%eax), %xmm0 # encoding: [0x62,0xf5,0x7d,0x28,0x38,0x00]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %a = load <8 x float>, ptr %ptr_a
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8.256(<8 x float> %a)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8256(<8 x float> %a)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8_mem_512(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtrops2hf8_mem_512:
+define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8512_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtrops2hf8512_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vcvtrops2hf8 (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7d,0x48,0x38,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_vcvtrops2hf8_mem_512:
+; X86-LABEL: test_int_x86_avx10_vcvtrops2hf8512_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    vcvtrops2hf8 (%eax), %xmm0 # encoding: [0x62,0xf5,0x7d,0x48,0x38,0x00]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %a = load <16 x float>, ptr %ptr_a
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8.512(<16 x float> %a)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8512(<16 x float> %a)
   ret <16 x i8> %ret
 }
 
-declare <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s.128(<4 x float>)
-declare <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s.256(<8 x float>)
-declare <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s.512(<16 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s128(<4 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s256(<8 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s512(<16 x float>)
 
-define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8s_128(<4 x float> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtrops2hf8s_128:
+define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8s128(<4 x float> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtrops2hf8s128:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvtrops2hf8s %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7d,0x08,0x3a,0xc0]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s.128(<4 x float> %a)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s128(<4 x float> %a)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8s_256(<8 x float> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtrops2hf8s_256:
+define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8s256(<8 x float> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtrops2hf8s256:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvtrops2hf8s %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7d,0x28,0x3a,0xc0]
 ; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s.256(<8 x float> %a)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s256(<8 x float> %a)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8s_512(<16 x float> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtrops2hf8s_512:
+define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8s512(<16 x float> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtrops2hf8s512:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvtrops2hf8s %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7d,0x48,0x3a,0xc0]
 ; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s.512(<16 x float> %a)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s512(<16 x float> %a)
   ret <16 x i8> %ret
 }
 
-; Memory folding tests for vcvtrops2hf8s
-define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8s_mem_128(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtrops2hf8s_mem_128:
+define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8s128_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtrops2hf8s128_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vcvtrops2hf8sx (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7d,0x08,0x3a,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_vcvtrops2hf8s_mem_128:
+; X86-LABEL: test_int_x86_avx10_vcvtrops2hf8s128_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    vcvtrops2hf8sx (%eax), %xmm0 # encoding: [0x62,0xf5,0x7d,0x08,0x3a,0x00]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %a = load <4 x float>, ptr %ptr_a
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s.128(<4 x float> %a)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s128(<4 x float> %a)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8s_mem_256(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtrops2hf8s_mem_256:
+define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8s256_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtrops2hf8s256_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vcvtrops2hf8sy (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7d,0x28,0x3a,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_vcvtrops2hf8s_mem_256:
+; X86-LABEL: test_int_x86_avx10_vcvtrops2hf8s256_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    vcvtrops2hf8sy (%eax), %xmm0 # encoding: [0x62,0xf5,0x7d,0x28,0x3a,0x00]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %a = load <8 x float>, ptr %ptr_a
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s.256(<8 x float> %a)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s256(<8 x float> %a)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8s_mem_512(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtrops2hf8s_mem_512:
+define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8s512_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtrops2hf8s512_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vcvtrops2hf8s (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7d,0x48,0x3a,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_vcvtrops2hf8s_mem_512:
+; X86-LABEL: test_int_x86_avx10_vcvtrops2hf8s512_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    vcvtrops2hf8s (%eax), %xmm0 # encoding: [0x62,0xf5,0x7d,0x48,0x3a,0x00]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %a = load <16 x float>, ptr %ptr_a
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s.512(<16 x float> %a)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s512(<16 x float> %a)
   ret <16 x i8> %ret
 }
 
-declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8.128(<4 x i32>, <4 x float>)
-declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8.256(<8 x i32>, <8 x float>)
-declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8.512(<16 x i32>, <16 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8128(<4 x i32>, <4 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8256(<8 x i32>, <8 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8512(<16 x i32>, <16 x float>)
 
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8_128(<4 x i32> %A, <4 x float> %b) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2bf8_128:
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8128(<4 x i32> %A, <4 x float> %b) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2bf8128:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvtbiasps2bf8 %xmm1, %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x39,0xc1]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8.128(<4 x i32> %A, <4 x float> %b)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8128(<4 x i32> %A, <4 x float> %b)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8_256(<8 x i32> %A, <8 x float> %b) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2bf8_256:
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8256(<8 x i32> %A, <8 x float> %b) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2bf8256:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvtbiasps2bf8 %ymm1, %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x28,0x39,0xc1]
 ; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8.256(<8 x i32> %A, <8 x float> %b)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8256(<8 x i32> %A, <8 x float> %b)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8_512(<16 x i32> %A, <16 x float> %b) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2bf8_512:
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8512(<16 x i32> %A, <16 x float> %b) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2bf8512:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvtbiasps2bf8 %zmm1, %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x39,0xc1]
 ; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8.512(<16 x i32> %A, <16 x float> %b)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8512(<16 x i32> %A, <16 x float> %b)
   ret <16 x i8> %ret
 }
 
-; Memory folding tests for vcvtbiasps2bf8 (second operand from memory)
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8_mem_128(<4 x i32> %A, ptr %ptr_b) {
-; X64-LABEL: test_int_x86_avx10_vcvtbiasps2bf8_mem_128:
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8128_mem(<4 x i32> %A, ptr %ptr_b) {
+; X64-LABEL: test_int_x86_avx10_vcvtbiasps2bf8128_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vcvtbiasps2bf8 (%rdi), %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x39,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_vcvtbiasps2bf8_mem_128:
+; X86-LABEL: test_int_x86_avx10_vcvtbiasps2bf8128_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    vcvtbiasps2bf8 (%eax), %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x39,0x00]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %b = load <4 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8.128(<4 x i32> %A, <4 x float> %b)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8128(<4 x i32> %A, <4 x float> %b)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8_mem_256(<8 x i32> %A, ptr %ptr_b) {
-; X64-LABEL: test_int_x86_avx10_vcvtbiasps2bf8_mem_256:
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8256_mem(<8 x i32> %A, ptr %ptr_b) {
+; X64-LABEL: test_int_x86_avx10_vcvtbiasps2bf8256_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vcvtbiasps2bf8 (%rdi), %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x28,0x39,0x07]
 ; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_vcvtbiasps2bf8_mem_256:
+; X86-LABEL: test_int_x86_avx10_vcvtbiasps2bf8256_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    vcvtbiasps2bf8 (%eax), %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x28,0x39,0x00]
 ; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %b = load <8 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8.256(<8 x i32> %A, <8 x float> %b)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8256(<8 x i32> %A, <8 x float> %b)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8_mem_512(<16 x i32> %A, ptr %ptr_b) {
-; X64-LABEL: test_int_x86_avx10_vcvtbiasps2bf8_mem_512:
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8512_mem(<16 x i32> %A, ptr %ptr_b) {
+; X64-LABEL: test_int_x86_avx10_vcvtbiasps2bf8512_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vcvtbiasps2bf8 (%rdi), %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x39,0x07]
 ; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_vcvtbiasps2bf8_mem_512:
+; X86-LABEL: test_int_x86_avx10_vcvtbiasps2bf8512_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    vcvtbiasps2bf8 (%eax), %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x39,0x00]
 ; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %b = load <16 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8.512(<16 x i32> %A, <16 x float> %b)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8512(<16 x i32> %A, <16 x float> %b)
   ret <16 x i8> %ret
 }
 
-declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s.128(<4 x i32>, <4 x float>)
-declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s.256(<8 x i32>, <8 x float>)
-declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s.512(<16 x i32>, <16 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s128(<4 x i32>, <4 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s256(<8 x i32>, <8 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s512(<16 x i32>, <16 x float>)
 
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8s_128(<4 x i32> %A, <4 x float> %b) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2bf8s_128:
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8s128(<4 x i32> %A, <4 x float> %b) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2bf8s128:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvtbiasps2bf8s %xmm1, %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x3b,0xc1]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s.128(<4 x i32> %A, <4 x float> %b)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s128(<4 x i32> %A, <4 x float> %b)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8s_256(<8 x i32> %A, <8 x float> %b) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2bf8s_256:
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8s256(<8 x i32> %A, <8 x float> %b) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2bf8s256:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvtbiasps2bf8s %ymm1, %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x28,0x3b,0xc1]
 ; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s.256(<8 x i32> %A, <8 x float> %b)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s256(<8 x i32> %A, <8 x float> %b)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8s_512(<16 x i32> %A, <16 x float> %b) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2bf8s_512:
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8s512(<16 x i32> %A, <16 x float> %b) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2bf8s512:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvtbiasps2bf8s %zmm1, %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x3b,0xc1]
 ; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s.512(<16 x i32> %A, <16 x float> %b)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s512(<16 x i32> %A, <16 x float> %b)
   ret <16 x i8> %ret
 }
 
-; Memory folding tests for vcvtbiasps2bf8s
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8s_mem_128(<4 x i32> %A, ptr %ptr_b) {
-; X64-LABEL: test_int_x86_avx10_vcvtbiasps2bf8s_mem_128:
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8s128_mem(<4 x i32> %A, ptr %ptr_b) {
+; X64-LABEL: test_int_x86_avx10_vcvtbiasps2bf8s128_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vcvtbiasps2bf8s (%rdi), %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x3b,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_vcvtbiasps2bf8s_mem_128:
+; X86-LABEL: test_int_x86_avx10_vcvtbiasps2bf8s128_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    vcvtbiasps2bf8s (%eax), %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x3b,0x00]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %b = load <4 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s.128(<4 x i32> %A, <4 x float> %b)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s128(<4 x i32> %A, <4 x float> %b)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8s_mem_256(<8 x i32> %A, ptr %ptr_b) {
-; X64-LABEL: test_int_x86_avx10_vcvtbiasps2bf8s_mem_256:
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8s256_mem(<8 x i32> %A, ptr %ptr_b) {
+; X64-LABEL: test_int_x86_avx10_vcvtbiasps2bf8s256_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vcvtbiasps2bf8s (%rdi), %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x28,0x3b,0x07]
 ; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_vcvtbiasps2bf8s_mem_256:
+; X86-LABEL: test_int_x86_avx10_vcvtbiasps2bf8s256_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    vcvtbiasps2bf8s (%eax), %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x28,0x3b,0x00]
 ; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %b = load <8 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s.256(<8 x i32> %A, <8 x float> %b)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s256(<8 x i32> %A, <8 x float> %b)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8s_mem_512(<16 x i32> %A, ptr %ptr_b) {
-; X64-LABEL: test_int_x86_avx10_vcvtbiasps2bf8s_mem_512:
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8s512_mem(<16 x i32> %A, ptr %ptr_b) {
+; X64-LABEL: test_int_x86_avx10_vcvtbiasps2bf8s512_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vcvtbiasps2bf8s (%rdi), %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x3b,0x07]
 ; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_vcvtbiasps2bf8s_mem_512:
+; X86-LABEL: test_int_x86_avx10_vcvtbiasps2bf8s512_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    vcvtbiasps2bf8s (%eax), %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x3b,0x00]
 ; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %b = load <16 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s.512(<16 x i32> %A, <16 x float> %b)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s512(<16 x i32> %A, <16 x float> %b)
   ret <16 x i8> %ret
 }
 
-declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8.128(<4 x i32>, <4 x float>)
-declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8.256(<8 x i32>, <8 x float>)
-declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8.512(<16 x i32>, <16 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8128(<4 x i32>, <4 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8256(<8 x i32>, <8 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8512(<16 x i32>, <16 x float>)
 
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8_128(<4 x i32> %A, <4 x float> %b) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2hf8_128:
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8128(<4 x i32> %A, <4 x float> %b) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2hf8128:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvtbiasps2hf8 %xmm1, %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x38,0xc1]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8.128(<4 x i32> %A, <4 x float> %b)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8128(<4 x i32> %A, <4 x float> %b)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8_256(<8 x i32> %A, <8 x float> %b) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2hf8_256:
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8256(<8 x i32> %A, <8 x float> %b) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2hf8256:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvtbiasps2hf8 %ymm1, %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x28,0x38,0xc1]
 ; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8.256(<8 x i32> %A, <8 x float> %b)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8256(<8 x i32> %A, <8 x float> %b)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8_512(<16 x i32> %A, <16 x float> %b) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2hf8_512:
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8512(<16 x i32> %A, <16 x float> %b) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2hf8512:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvtbiasps2hf8 %zmm1, %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x38,0xc1]
 ; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8.512(<16 x i32> %A, <16 x float> %b)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8512(<16 x i32> %A, <16 x float> %b)
   ret <16 x i8> %ret
 }
 
-; Memory folding tests for vcvtbiasps2hf8
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8_mem_128(<4 x i32> %A, ptr %ptr_b) {
-; X64-LABEL: test_int_x86_avx10_vcvtbiasps2hf8_mem_128:
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8128_mem(<4 x i32> %A, ptr %ptr_b) {
+; X64-LABEL: test_int_x86_avx10_vcvtbiasps2hf8128_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vcvtbiasps2hf8 (%rdi), %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x38,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_vcvtbiasps2hf8_mem_128:
+; X86-LABEL: test_int_x86_avx10_vcvtbiasps2hf8128_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    vcvtbiasps2hf8 (%eax), %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x38,0x00]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %b = load <4 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8.128(<4 x i32> %A, <4 x float> %b)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8128(<4 x i32> %A, <4 x float> %b)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8_mem_256(<8 x i32> %A, ptr %ptr_b) {
-; X64-LABEL: test_int_x86_avx10_vcvtbiasps2hf8_mem_256:
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8256_mem(<8 x i32> %A, ptr %ptr_b) {
+; X64-LABEL: test_int_x86_avx10_vcvtbiasps2hf8256_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vcvtbiasps2hf8 (%rdi), %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x28,0x38,0x07]
 ; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_vcvtbiasps2hf8_mem_256:
+; X86-LABEL: test_int_x86_avx10_vcvtbiasps2hf8256_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    vcvtbiasps2hf8 (%eax), %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x28,0x38,0x00]
 ; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %b = load <8 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8.256(<8 x i32> %A, <8 x float> %b)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8256(<8 x i32> %A, <8 x float> %b)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8_mem_512(<16 x i32> %A, ptr %ptr_b) {
-; X64-LABEL: test_int_x86_avx10_vcvtbiasps2hf8_mem_512:
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8512_mem(<16 x i32> %A, ptr %ptr_b) {
+; X64-LABEL: test_int_x86_avx10_vcvtbiasps2hf8512_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vcvtbiasps2hf8 (%rdi), %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x38,0x07]
 ; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_vcvtbiasps2hf8_mem_512:
+; X86-LABEL: test_int_x86_avx10_vcvtbiasps2hf8512_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    vcvtbiasps2hf8 (%eax), %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x38,0x00]
 ; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %b = load <16 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8.512(<16 x i32> %A, <16 x float> %b)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8512(<16 x i32> %A, <16 x float> %b)
   ret <16 x i8> %ret
 }
 
-declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s.128(<4 x i32>, <4 x float>)
-declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s.256(<8 x i32>, <8 x float>)
-declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s.512(<16 x i32>, <16 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s128(<4 x i32>, <4 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s256(<8 x i32>, <8 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s512(<16 x i32>, <16 x float>)
 
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8s_128(<4 x i32> %A, <4 x float> %b) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2hf8s_128:
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8s128(<4 x i32> %A, <4 x float> %b) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2hf8s128:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvtbiasps2hf8s %xmm1, %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x3a,0xc1]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s.128(<4 x i32> %A, <4 x float> %b)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s128(<4 x i32> %A, <4 x float> %b)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8s_256(<8 x i32> %A, <8 x float> %b) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2hf8s_256:
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8s256(<8 x i32> %A, <8 x float> %b) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2hf8s256:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvtbiasps2hf8s %ymm1, %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x28,0x3a,0xc1]
 ; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s.256(<8 x i32> %A, <8 x float> %b)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s256(<8 x i32> %A, <8 x float> %b)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8s_512(<16 x i32> %A, <16 x float> %b) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2hf8s_512:
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8s512(<16 x i32> %A, <16 x float> %b) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2hf8s512:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvtbiasps2hf8s %zmm1, %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x3a,0xc1]
 ; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s.512(<16 x i32> %A, <16 x float> %b)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s512(<16 x i32> %A, <16 x float> %b)
   ret <16 x i8> %ret
 }
 
-; Memory folding tests for vcvtbiasps2hf8s
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8s_mem_128(<4 x i32> %A, ptr %ptr_b) {
-; X64-LABEL: test_int_x86_avx10_vcvtbiasps2hf8s_mem_128:
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8s128_mem(<4 x i32> %A, ptr %ptr_b) {
+; X64-LABEL: test_int_x86_avx10_vcvtbiasps2hf8s128_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vcvtbiasps2hf8s (%rdi), %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x3a,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_vcvtbiasps2hf8s_mem_128:
+; X86-LABEL: test_int_x86_avx10_vcvtbiasps2hf8s128_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    vcvtbiasps2hf8s (%eax), %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x3a,0x00]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %b = load <4 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s.128(<4 x i32> %A, <4 x float> %b)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s128(<4 x i32> %A, <4 x float> %b)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8s_mem_256(<8 x i32> %A, ptr %ptr_b) {
-; X64-LABEL: test_int_x86_avx10_vcvtbiasps2hf8s_mem_256:
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8s256_mem(<8 x i32> %A, ptr %ptr_b) {
+; X64-LABEL: test_int_x86_avx10_vcvtbiasps2hf8s256_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vcvtbiasps2hf8s (%rdi), %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x28,0x3a,0x07]
 ; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_vcvtbiasps2hf8s_mem_256:
+; X86-LABEL: test_int_x86_avx10_vcvtbiasps2hf8s256_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    vcvtbiasps2hf8s (%eax), %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x28,0x3a,0x00]
 ; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %b = load <8 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s.256(<8 x i32> %A, <8 x float> %b)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s256(<8 x i32> %A, <8 x float> %b)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8s_mem_512(<16 x i32> %A, ptr %ptr_b) {
-; X64-LABEL: test_int_x86_avx10_vcvtbiasps2hf8s_mem_512:
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8s512_mem(<16 x i32> %A, ptr %ptr_b) {
+; X64-LABEL: test_int_x86_avx10_vcvtbiasps2hf8s512_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vcvtbiasps2hf8s (%rdi), %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x3a,0x07]
 ; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_vcvtbiasps2hf8s_mem_512:
+; X86-LABEL: test_int_x86_avx10_vcvtbiasps2hf8s512_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    vcvtbiasps2hf8s (%eax), %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x3a,0x00]
 ; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %b = load <16 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s.512(<16 x i32> %A, <16 x float> %b)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s512(<16 x i32> %A, <16 x float> %b)
   ret <16 x i8> %ret
 }
 
-declare <4 x float> @llvm.x86.avx10.vcvtbf82ps.128(<16 x i8>)
-declare <8 x float> @llvm.x86.avx10.vcvtbf82ps.256(<16 x i8>)
-declare <16 x float> @llvm.x86.avx10.vcvtbf82ps.512(<16 x i8>)
+declare <4 x float> @llvm.x86.avx10.vcvtbf82ps128(<16 x i8>)
+declare <8 x float> @llvm.x86.avx10.vcvtbf82ps256(<16 x i8>)
+declare <16 x float> @llvm.x86.avx10.vcvtbf82ps512(<16 x i8>)
 
-define <4 x float> @test_int_x86_avx10_vcvtbf82ps_128(<16 x i8> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbf82ps_128:
+define <4 x float> @test_int_x86_avx10_vcvtbf82ps128(<16 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbf82ps128:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvtbf82ps %xmm0, %xmm0 # encoding: [0x62,0xf5,0xfc,0x08,0x36,0xc0]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <4 x float> @llvm.x86.avx10.vcvtbf82ps.128(<16 x i8> %a)
+  %ret = call <4 x float> @llvm.x86.avx10.vcvtbf82ps128(<16 x i8> %a)
   ret <4 x float> %ret
 }
 
-define <8 x float> @test_int_x86_avx10_vcvtbf82ps_256(<16 x i8> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbf82ps_256:
+define <8 x float> @test_int_x86_avx10_vcvtbf82ps256(<16 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbf82ps256:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvtbf82ps %xmm0, %ymm0 # encoding: [0x62,0xf5,0xfc,0x28,0x36,0xc0]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <8 x float> @llvm.x86.avx10.vcvtbf82ps.256(<16 x i8> %a)
+  %ret = call <8 x float> @llvm.x86.avx10.vcvtbf82ps256(<16 x i8> %a)
   ret <8 x float> %ret
 }
 
-define <16 x float> @test_int_x86_avx10_vcvtbf82ps_512(<16 x i8> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbf82ps_512:
+define <16 x float> @test_int_x86_avx10_vcvtbf82ps512(<16 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbf82ps512:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvtbf82ps %xmm0, %zmm0 # encoding: [0x62,0xf5,0xfc,0x48,0x36,0xc0]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x float> @llvm.x86.avx10.vcvtbf82ps.512(<16 x i8> %a)
+  %ret = call <16 x float> @llvm.x86.avx10.vcvtbf82ps512(<16 x i8> %a)
   ret <16 x float> %ret
 }
 
-; Memory folding tests for vcvtbf82ps
-define <4 x float> @test_int_x86_avx10_vcvtbf82ps_mem_128(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtbf82ps_mem_128:
+define <4 x float> @test_int_x86_avx10_vcvtbf82ps128_mask(<16 x i8> %a, <4 x float> %src, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf82ps128_mask:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbf82ps %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0xfc,0x09,0x36,0xc8]
+; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbf82ps128_mask:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbf82ps %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0xfc,0x09,0x36,0xc8]
+; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %cvt = call <4 x float> @llvm.x86.avx10.vcvtbf82ps128(<16 x i8> %a)
+  %bst = bitcast i8 %mask to <8 x i1>
+  %ext = shufflevector <8 x i1> %bst, <8 x i1> poison, <4 x i32> <i32 0, i32 1, i32 2, i32 3>
+  %ret = select <4 x i1> %ext, <4 x float> %cvt, <4 x float> %src
+  ret <4 x float> %ret
+}
+
+define <4 x float> @test_int_x86_avx10_vcvtbf82ps128_maskz(<16 x i8> %a, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf82ps128_maskz:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbf82ps %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0xfc,0x89,0x36,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbf82ps128_maskz:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbf82ps %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0xfc,0x89,0x36,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %cvt = call <4 x float> @llvm.x86.avx10.vcvtbf82ps128(<16 x i8> %a)
+  %bst = bitcast i8 %mask to <8 x i1>
+  %ext = shufflevector <8 x i1> %bst, <8 x i1> poison, <4 x i32> <i32 0, i32 1, i32 2, i32 3>
+  %ret = select <4 x i1> %ext, <4 x float> %cvt, <4 x float> zeroinitializer
+  ret <4 x float> %ret
+}
+
+define <8 x float> @test_int_x86_avx10_vcvtbf82ps256_mask(<16 x i8> %a, <8 x float> %src, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf82ps256_mask:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbf82ps %xmm0, %ymm1 {%k1} # encoding: [0x62,0xf5,0xfc,0x29,0x36,0xc8]
+; X64-NEXT:    vmovaps %ymm1, %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfc,0x28,0xc1]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbf82ps256_mask:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbf82ps %xmm0, %ymm1 {%k1} # encoding: [0x62,0xf5,0xfc,0x29,0x36,0xc8]
+; X86-NEXT:    vmovaps %ymm1, %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfc,0x28,0xc1]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %cvt = call <8 x float> @llvm.x86.avx10.vcvtbf82ps256(<16 x i8> %a)
+  %bst = bitcast i8 %mask to <8 x i1>
+  %ret = select <8 x i1> %bst, <8 x float> %cvt, <8 x float> %src
+  ret <8 x float> %ret
+}
+
+define <8 x float> @test_int_x86_avx10_vcvtbf82ps256_maskz(<16 x i8> %a, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf82ps256_maskz:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbf82ps %xmm0, %ymm0 {%k1} {z} # encoding: [0x62,0xf5,0xfc,0xa9,0x36,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbf82ps256_maskz:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbf82ps %xmm0, %ymm0 {%k1} {z} # encoding: [0x62,0xf5,0xfc,0xa9,0x36,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %cvt = call <8 x float> @llvm.x86.avx10.vcvtbf82ps256(<16 x i8> %a)
+  %bst = bitcast i8 %mask to <8 x i1>
+  %ret = select <8 x i1> %bst, <8 x float> %cvt, <8 x float> zeroinitializer
+  ret <8 x float> %ret
+}
+
+define <16 x float> @test_int_x86_avx10_vcvtbf82ps512_mask(<16 x i8> %a, <16 x float> %src, i16 %mask) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf82ps512_mask:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbf82ps %xmm0, %zmm1 {%k1} # encoding: [0x62,0xf5,0xfc,0x49,0x36,0xc8]
+; X64-NEXT:    vmovaps %zmm1, %zmm0 # encoding: [0x62,0xf1,0x7c,0x48,0x28,0xc1]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbf82ps512_mask:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbf82ps %xmm0, %zmm1 {%k1} # encoding: [0x62,0xf5,0xfc,0x49,0x36,0xc8]
+; X86-NEXT:    vmovaps %zmm1, %zmm0 # encoding: [0x62,0xf1,0x7c,0x48,0x28,0xc1]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %cvt = call <16 x float> @llvm.x86.avx10.vcvtbf82ps512(<16 x i8> %a)
+  %bst = bitcast i16 %mask to <16 x i1>
+  %ret = select <16 x i1> %bst, <16 x float> %cvt, <16 x float> %src
+  ret <16 x float> %ret
+}
+
+define <16 x float> @test_int_x86_avx10_vcvtbf82ps512_maskz(<16 x i8> %a, i16 %mask) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf82ps512_maskz:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbf82ps %xmm0, %zmm0 {%k1} {z} # encoding: [0x62,0xf5,0xfc,0xc9,0x36,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbf82ps512_maskz:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbf82ps %xmm0, %zmm0 {%k1} {z} # encoding: [0x62,0xf5,0xfc,0xc9,0x36,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %cvt = call <16 x float> @llvm.x86.avx10.vcvtbf82ps512(<16 x i8> %a)
+  %bst = bitcast i16 %mask to <16 x i1>
+  %ret = select <16 x i1> %bst, <16 x float> %cvt, <16 x float> zeroinitializer
+  ret <16 x float> %ret
+}
+
+define <4 x float> @test_int_x86_avx10_vcvtbf82ps128_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf82ps128_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vcvtbf82ps (%rdi), %xmm0 # encoding: [0x62,0xf5,0xfc,0x08,0x36,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_vcvtbf82ps_mem_128:
+; X86-LABEL: test_int_x86_avx10_vcvtbf82ps128_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    vcvtbf82ps (%eax), %xmm0 # encoding: [0x62,0xf5,0xfc,0x08,0x36,0x00]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %a = load <16 x i8>, ptr %ptr_a
-  %ret = call <4 x float> @llvm.x86.avx10.vcvtbf82ps.128(<16 x i8> %a)
+  %ret = call <4 x float> @llvm.x86.avx10.vcvtbf82ps128(<16 x i8> %a)
   ret <4 x float> %ret
 }
 
-define <8 x float> @test_int_x86_avx10_vcvtbf82ps_mem_256(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtbf82ps_mem_256:
+define <8 x float> @test_int_x86_avx10_vcvtbf82ps256_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf82ps256_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vcvtbf82ps (%rdi), %ymm0 # encoding: [0x62,0xf5,0xfc,0x28,0x36,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_vcvtbf82ps_mem_256:
+; X86-LABEL: test_int_x86_avx10_vcvtbf82ps256_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    vcvtbf82ps (%eax), %ymm0 # encoding: [0x62,0xf5,0xfc,0x28,0x36,0x00]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %a = load <16 x i8>, ptr %ptr_a
-  %ret = call <8 x float> @llvm.x86.avx10.vcvtbf82ps.256(<16 x i8> %a)
+  %ret = call <8 x float> @llvm.x86.avx10.vcvtbf82ps256(<16 x i8> %a)
   ret <8 x float> %ret
 }
 
-define <16 x float> @test_int_x86_avx10_vcvtbf82ps_mem_512(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtbf82ps_mem_512:
+define <16 x float> @test_int_x86_avx10_vcvtbf82ps512_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf82ps512_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vcvtbf82ps (%rdi), %zmm0 # encoding: [0x62,0xf5,0xfc,0x48,0x36,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_vcvtbf82ps_mem_512:
+; X86-LABEL: test_int_x86_avx10_vcvtbf82ps512_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    vcvtbf82ps (%eax), %zmm0 # encoding: [0x62,0xf5,0xfc,0x48,0x36,0x00]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %a = load <16 x i8>, ptr %ptr_a
-  %ret = call <16 x float> @llvm.x86.avx10.vcvtbf82ps.512(<16 x i8> %a)
+  %ret = call <16 x float> @llvm.x86.avx10.vcvtbf82ps512(<16 x i8> %a)
   ret <16 x float> %ret
 }
 
-declare <4 x float> @llvm.x86.avx10.vcvthf82ps.128(<16 x i8>)
-declare <8 x float> @llvm.x86.avx10.vcvthf82ps.256(<16 x i8>)
-declare <16 x float> @llvm.x86.avx10.vcvthf82ps.512(<16 x i8>)
+declare <4 x float> @llvm.x86.avx10.vcvthf82ps128(<16 x i8>)
+declare <8 x float> @llvm.x86.avx10.vcvthf82ps256(<16 x i8>)
+declare <16 x float> @llvm.x86.avx10.vcvthf82ps512(<16 x i8>)
 
-define <4 x float> @test_int_x86_avx10_vcvthf82ps_128(<16 x i8> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvthf82ps_128:
+define <4 x float> @test_int_x86_avx10_vcvthf82ps128(<16 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvthf82ps128:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvthf82ps %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x36,0xc0]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <4 x float> @llvm.x86.avx10.vcvthf82ps.128(<16 x i8> %a)
+  %ret = call <4 x float> @llvm.x86.avx10.vcvthf82ps128(<16 x i8> %a)
   ret <4 x float> %ret
 }
 
-define <8 x float> @test_int_x86_avx10_vcvthf82ps_256(<16 x i8> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvthf82ps_256:
+define <8 x float> @test_int_x86_avx10_vcvthf82ps256(<16 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvthf82ps256:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvthf82ps %xmm0, %ymm0 # encoding: [0x62,0xf5,0x7c,0x28,0x36,0xc0]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <8 x float> @llvm.x86.avx10.vcvthf82ps.256(<16 x i8> %a)
+  %ret = call <8 x float> @llvm.x86.avx10.vcvthf82ps256(<16 x i8> %a)
   ret <8 x float> %ret
 }
 
-define <16 x float> @test_int_x86_avx10_vcvthf82ps_512(<16 x i8> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvthf82ps_512:
+define <16 x float> @test_int_x86_avx10_vcvthf82ps512(<16 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvthf82ps512:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvthf82ps %xmm0, %zmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x36,0xc0]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x float> @llvm.x86.avx10.vcvthf82ps.512(<16 x i8> %a)
+  %ret = call <16 x float> @llvm.x86.avx10.vcvthf82ps512(<16 x i8> %a)
+  ret <16 x float> %ret
+}
+
+define <4 x float> @test_int_x86_avx10_vcvthf82ps128_mask(<16 x i8> %a, <4 x float> %src, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_vcvthf82ps128_mask:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvthf82ps %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x36,0xc8]
+; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvthf82ps128_mask:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvthf82ps %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x36,0xc8]
+; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %cvt = call <4 x float> @llvm.x86.avx10.vcvthf82ps128(<16 x i8> %a)
+  %bst = bitcast i8 %mask to <8 x i1>
+  %ext = shufflevector <8 x i1> %bst, <8 x i1> poison, <4 x i32> <i32 0, i32 1, i32 2, i32 3>
+  %ret = select <4 x i1> %ext, <4 x float> %cvt, <4 x float> %src
+  ret <4 x float> %ret
+}
+
+define <4 x float> @test_int_x86_avx10_vcvthf82ps128_maskz(<16 x i8> %a, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_vcvthf82ps128_maskz:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvthf82ps %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x36,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvthf82ps128_maskz:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvthf82ps %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x36,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %cvt = call <4 x float> @llvm.x86.avx10.vcvthf82ps128(<16 x i8> %a)
+  %bst = bitcast i8 %mask to <8 x i1>
+  %ext = shufflevector <8 x i1> %bst, <8 x i1> poison, <4 x i32> <i32 0, i32 1, i32 2, i32 3>
+  %ret = select <4 x i1> %ext, <4 x float> %cvt, <4 x float> zeroinitializer
+  ret <4 x float> %ret
+}
+
+define <8 x float> @test_int_x86_avx10_vcvthf82ps256_mask(<16 x i8> %a, <8 x float> %src, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_vcvthf82ps256_mask:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvthf82ps %xmm0, %ymm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x36,0xc8]
+; X64-NEXT:    vmovaps %ymm1, %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfc,0x28,0xc1]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvthf82ps256_mask:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvthf82ps %xmm0, %ymm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x36,0xc8]
+; X86-NEXT:    vmovaps %ymm1, %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfc,0x28,0xc1]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %cvt = call <8 x float> @llvm.x86.avx10.vcvthf82ps256(<16 x i8> %a)
+  %bst = bitcast i8 %mask to <8 x i1>
+  %ret = select <8 x i1> %bst, <8 x float> %cvt, <8 x float> %src
+  ret <8 x float> %ret
+}
+
+define <8 x float> @test_int_x86_avx10_vcvthf82ps256_maskz(<16 x i8> %a, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_vcvthf82ps256_maskz:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvthf82ps %xmm0, %ymm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xa9,0x36,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvthf82ps256_maskz:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvthf82ps %xmm0, %ymm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xa9,0x36,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %cvt = call <8 x float> @llvm.x86.avx10.vcvthf82ps256(<16 x i8> %a)
+  %bst = bitcast i8 %mask to <8 x i1>
+  %ret = select <8 x i1> %bst, <8 x float> %cvt, <8 x float> zeroinitializer
+  ret <8 x float> %ret
+}
+
+define <16 x float> @test_int_x86_avx10_vcvthf82ps512_mask(<16 x i8> %a, <16 x float> %src, i16 %mask) {
+; X64-LABEL: test_int_x86_avx10_vcvthf82ps512_mask:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvthf82ps %xmm0, %zmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x49,0x36,0xc8]
+; X64-NEXT:    vmovaps %zmm1, %zmm0 # encoding: [0x62,0xf1,0x7c,0x48,0x28,0xc1]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvthf82ps512_mask:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvthf82ps %xmm0, %zmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x49,0x36,0xc8]
+; X86-NEXT:    vmovaps %zmm1, %zmm0 # encoding: [0x62,0xf1,0x7c,0x48,0x28,0xc1]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %cvt = call <16 x float> @llvm.x86.avx10.vcvthf82ps512(<16 x i8> %a)
+  %bst = bitcast i16 %mask to <16 x i1>
+  %ret = select <16 x i1> %bst, <16 x float> %cvt, <16 x float> %src
+  ret <16 x float> %ret
+}
+
+define <16 x float> @test_int_x86_avx10_vcvthf82ps512_maskz(<16 x i8> %a, i16 %mask) {
+; X64-LABEL: test_int_x86_avx10_vcvthf82ps512_maskz:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvthf82ps %xmm0, %zmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xc9,0x36,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvthf82ps512_maskz:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvthf82ps %xmm0, %zmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xc9,0x36,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %cvt = call <16 x float> @llvm.x86.avx10.vcvthf82ps512(<16 x i8> %a)
+  %bst = bitcast i16 %mask to <16 x i1>
+  %ret = select <16 x i1> %bst, <16 x float> %cvt, <16 x float> zeroinitializer
   ret <16 x float> %ret
 }
 
-; Memory folding tests for vcvthf82ps
-define <4 x float> @test_int_x86_avx10_vcvthf82ps_mem_128(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvthf82ps_mem_128:
+define <4 x float> @test_int_x86_avx10_vcvthf82ps128_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvthf82ps128_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vcvthf82ps (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x36,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_vcvthf82ps_mem_128:
+; X86-LABEL: test_int_x86_avx10_vcvthf82ps128_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    vcvthf82ps (%eax), %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x36,0x00]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %a = load <16 x i8>, ptr %ptr_a
-  %ret = call <4 x float> @llvm.x86.avx10.vcvthf82ps.128(<16 x i8> %a)
+  %ret = call <4 x float> @llvm.x86.avx10.vcvthf82ps128(<16 x i8> %a)
   ret <4 x float> %ret
 }
 
-define <8 x float> @test_int_x86_avx10_vcvthf82ps_mem_256(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvthf82ps_mem_256:
+define <8 x float> @test_int_x86_avx10_vcvthf82ps256_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvthf82ps256_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vcvthf82ps (%rdi), %ymm0 # encoding: [0x62,0xf5,0x7c,0x28,0x36,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_vcvthf82ps_mem_256:
+; X86-LABEL: test_int_x86_avx10_vcvthf82ps256_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    vcvthf82ps (%eax), %ymm0 # encoding: [0x62,0xf5,0x7c,0x28,0x36,0x00]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %a = load <16 x i8>, ptr %ptr_a
-  %ret = call <8 x float> @llvm.x86.avx10.vcvthf82ps.256(<16 x i8> %a)
+  %ret = call <8 x float> @llvm.x86.avx10.vcvthf82ps256(<16 x i8> %a)
   ret <8 x float> %ret
 }
 
-define <16 x float> @test_int_x86_avx10_vcvthf82ps_mem_512(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvthf82ps_mem_512:
+define <16 x float> @test_int_x86_avx10_vcvthf82ps512_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvthf82ps512_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vcvthf82ps (%rdi), %zmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x36,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_vcvthf82ps_mem_512:
+; X86-LABEL: test_int_x86_avx10_vcvthf82ps512_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    vcvthf82ps (%eax), %zmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x36,0x00]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %a = load <16 x i8>, ptr %ptr_a
-  %ret = call <16 x float> @llvm.x86.avx10.vcvthf82ps.512(<16 x i8> %a)
+  %ret = call <16 x float> @llvm.x86.avx10.vcvthf82ps512(<16 x i8> %a)
   ret <16 x float> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtbf82bf4s_128(<16 x i8> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbf82bf4s_128:
+define <16 x i8> @test_int_x86_avx10_vcvtbf82bf4s128(<16 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbf82bf4s128:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvtbf82bf4s %xmm0, %xmm0 # encoding: [0x62,0xf5,0xfe,0x08,0x3d,0xc0]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf82bf4s.128(<16 x i8> %a)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf82bf4s128(<16 x i8> %a)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtbf82bf4s_256(<32 x i8> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbf82bf4s_256:
+define <16 x i8> @test_int_x86_avx10_vcvtbf82bf4s256(<32 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbf82bf4s256:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvtbf82bf4s %ymm0, %xmm0 # encoding: [0x62,0xf5,0xfe,0x28,0x3d,0xc0]
 ; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf82bf4s.256(<32 x i8> %a)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf82bf4s256(<32 x i8> %a)
   ret <16 x i8> %ret
 }
 
-define <32 x i8> @test_int_x86_avx10_vcvtbf82bf4s_512(<64 x i8> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbf82bf4s_512:
+define <32 x i8> @test_int_x86_avx10_vcvtbf82bf4s512(<64 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbf82bf4s512:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvtbf82bf4s %zmm0, %ymm0 # encoding: [0x62,0xf5,0xfe,0x48,0x3d,0xc0]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <32 x i8> @llvm.x86.avx10.vcvtbf82bf4s.512(<64 x i8> %a)
+  %ret = call <32 x i8> @llvm.x86.avx10.vcvtbf82bf4s512(<64 x i8> %a)
   ret <32 x i8> %ret
 }
 
-declare <16 x i8> @llvm.x86.avx10.vcvtbf82bf4s.128(<16 x i8>)
-declare <16 x i8> @llvm.x86.avx10.vcvtbf82bf4s.256(<32 x i8>)
-declare <32 x i8> @llvm.x86.avx10.vcvtbf82bf4s.512(<64 x i8>)
+declare <16 x i8> @llvm.x86.avx10.vcvtbf82bf4s128(<16 x i8>)
+declare <16 x i8> @llvm.x86.avx10.vcvtbf82bf4s256(<32 x i8>)
+declare <32 x i8> @llvm.x86.avx10.vcvtbf82bf4s512(<64 x i8>)
 
-; Memory folding tests for vcvtbf82bf4s
-define <16 x i8> @test_int_x86_avx10_vcvtbf82bf4s_mem_128(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtbf82bf4s_mem_128:
+define <16 x i8> @test_int_x86_avx10_vcvtbf82bf4s128_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf82bf4s128_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vmovaps (%rdi), %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0x07]
 ; X64-NEXT:    vcvtbf82bf4s %xmm0, %xmm0 # encoding: [0x62,0xf5,0xfe,0x08,0x3d,0xc0]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_vcvtbf82bf4s_mem_128:
+; X86-LABEL: test_int_x86_avx10_vcvtbf82bf4s128_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    vmovaps (%eax), %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0x00]
 ; X86-NEXT:    vcvtbf82bf4s %xmm0, %xmm0 # encoding: [0x62,0xf5,0xfe,0x08,0x3d,0xc0]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %a = load <16 x i8>, ptr %ptr_a
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf82bf4s.128(<16 x i8> %a)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf82bf4s128(<16 x i8> %a)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtbf82bf4s_mem_256(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtbf82bf4s_mem_256:
+define <16 x i8> @test_int_x86_avx10_vcvtbf82bf4s256_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf82bf4s256_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vmovaps (%rdi), %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfc,0x28,0x07]
 ; X64-NEXT:    vcvtbf82bf4s %ymm0, %xmm0 # encoding: [0x62,0xf5,0xfe,0x28,0x3d,0xc0]
 ; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_vcvtbf82bf4s_mem_256:
+; X86-LABEL: test_int_x86_avx10_vcvtbf82bf4s256_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    vmovaps (%eax), %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfc,0x28,0x00]
@@ -1065,88 +1284,87 @@ define <16 x i8> @test_int_x86_avx10_vcvtbf82bf4s_mem_256(ptr %ptr_a) {
 ; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %a = load <32 x i8>, ptr %ptr_a
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf82bf4s.256(<32 x i8> %a)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf82bf4s256(<32 x i8> %a)
   ret <16 x i8> %ret
 }
 
-define <32 x i8> @test_int_x86_avx10_vcvtbf82bf4s_mem_512(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtbf82bf4s_mem_512:
+define <32 x i8> @test_int_x86_avx10_vcvtbf82bf4s512_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf82bf4s512_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vmovaps (%rdi), %zmm0 # encoding: [0x62,0xf1,0x7c,0x48,0x28,0x07]
 ; X64-NEXT:    vcvtbf82bf4s %zmm0, %ymm0 # encoding: [0x62,0xf5,0xfe,0x48,0x3d,0xc0]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_vcvtbf82bf4s_mem_512:
+; X86-LABEL: test_int_x86_avx10_vcvtbf82bf4s512_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    vmovaps (%eax), %zmm0 # encoding: [0x62,0xf1,0x7c,0x48,0x28,0x00]
 ; X86-NEXT:    vcvtbf82bf4s %zmm0, %ymm0 # encoding: [0x62,0xf5,0xfe,0x48,0x3d,0xc0]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %a = load <64 x i8>, ptr %ptr_a
-  %ret = call <32 x i8> @llvm.x86.avx10.vcvtbf82bf4s.512(<64 x i8> %a)
+  %ret = call <32 x i8> @llvm.x86.avx10.vcvtbf82bf4s512(<64 x i8> %a)
   ret <32 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvthf82bf4s_128(<16 x i8> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvthf82bf4s_128:
+define <16 x i8> @test_int_x86_avx10_vcvthf82bf4s128(<16 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvthf82bf4s128:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvthf82bf4s %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x3d,0xc0]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvthf82bf4s.128(<16 x i8> %a)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvthf82bf4s128(<16 x i8> %a)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvthf82bf4s_256(<32 x i8> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvthf82bf4s_256:
+define <16 x i8> @test_int_x86_avx10_vcvthf82bf4s256(<32 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvthf82bf4s256:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvthf82bf4s %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x3d,0xc0]
 ; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvthf82bf4s.256(<32 x i8> %a)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvthf82bf4s256(<32 x i8> %a)
   ret <16 x i8> %ret
 }
 
-define <32 x i8> @test_int_x86_avx10_vcvthf82bf4s_512(<64 x i8> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvthf82bf4s_512:
+define <32 x i8> @test_int_x86_avx10_vcvthf82bf4s512(<64 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvthf82bf4s512:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvthf82bf4s %zmm0, %ymm0 # encoding: [0x62,0xf5,0x7e,0x48,0x3d,0xc0]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <32 x i8> @llvm.x86.avx10.vcvthf82bf4s.512(<64 x i8> %a)
+  %ret = call <32 x i8> @llvm.x86.avx10.vcvthf82bf4s512(<64 x i8> %a)
   ret <32 x i8> %ret
 }
 
-declare <16 x i8> @llvm.x86.avx10.vcvthf82bf4s.128(<16 x i8>)
-declare <16 x i8> @llvm.x86.avx10.vcvthf82bf4s.256(<32 x i8>)
-declare <32 x i8> @llvm.x86.avx10.vcvthf82bf4s.512(<64 x i8>)
+declare <16 x i8> @llvm.x86.avx10.vcvthf82bf4s128(<16 x i8>)
+declare <16 x i8> @llvm.x86.avx10.vcvthf82bf4s256(<32 x i8>)
+declare <32 x i8> @llvm.x86.avx10.vcvthf82bf4s512(<64 x i8>)
 
-; Memory folding tests for vcvthf82bf4s
-define <16 x i8> @test_int_x86_avx10_vcvthf82bf4s_mem_128(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvthf82bf4s_mem_128:
+define <16 x i8> @test_int_x86_avx10_vcvthf82bf4s128_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvthf82bf4s128_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vmovaps (%rdi), %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0x07]
 ; X64-NEXT:    vcvthf82bf4s %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x3d,0xc0]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_vcvthf82bf4s_mem_128:
+; X86-LABEL: test_int_x86_avx10_vcvthf82bf4s128_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    vmovaps (%eax), %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0x00]
 ; X86-NEXT:    vcvthf82bf4s %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x3d,0xc0]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %a = load <16 x i8>, ptr %ptr_a
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvthf82bf4s.128(<16 x i8> %a)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvthf82bf4s128(<16 x i8> %a)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvthf82bf4s_mem_256(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvthf82bf4s_mem_256:
+define <16 x i8> @test_int_x86_avx10_vcvthf82bf4s256_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvthf82bf4s256_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vmovaps (%rdi), %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfc,0x28,0x07]
 ; X64-NEXT:    vcvthf82bf4s %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x3d,0xc0]
 ; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_vcvthf82bf4s_mem_256:
+; X86-LABEL: test_int_x86_avx10_vcvthf82bf4s256_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    vmovaps (%eax), %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfc,0x28,0x00]
@@ -1154,289 +1372,392 @@ define <16 x i8> @test_int_x86_avx10_vcvthf82bf4s_mem_256(ptr %ptr_a) {
 ; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %a = load <32 x i8>, ptr %ptr_a
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvthf82bf4s.256(<32 x i8> %a)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvthf82bf4s256(<32 x i8> %a)
   ret <16 x i8> %ret
 }
 
-define <32 x i8> @test_int_x86_avx10_vcvthf82bf4s_mem_512(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvthf82bf4s_mem_512:
+define <32 x i8> @test_int_x86_avx10_vcvthf82bf4s512_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvthf82bf4s512_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vmovaps (%rdi), %zmm0 # encoding: [0x62,0xf1,0x7c,0x48,0x28,0x07]
 ; X64-NEXT:    vcvthf82bf4s %zmm0, %ymm0 # encoding: [0x62,0xf5,0x7e,0x48,0x3d,0xc0]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_vcvthf82bf4s_mem_512:
+; X86-LABEL: test_int_x86_avx10_vcvthf82bf4s512_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    vmovaps (%eax), %zmm0 # encoding: [0x62,0xf1,0x7c,0x48,0x28,0x00]
 ; X86-NEXT:    vcvthf82bf4s %zmm0, %ymm0 # encoding: [0x62,0xf5,0x7e,0x48,0x3d,0xc0]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %a = load <64 x i8>, ptr %ptr_a
-  %ret = call <32 x i8> @llvm.x86.avx10.vcvthf82bf4s.512(<64 x i8> %a)
+  %ret = call <32 x i8> @llvm.x86.avx10.vcvthf82bf4s512(<64 x i8> %a)
   ret <32 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtbf82bf6s_128(<16 x i8> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbf82bf6s_128:
+define void @test_int_x86_avx10_vcvtbf82bf4s128_store(ptr %ptr, <16 x i8> %a) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf82bf4s128_store:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtbf82bf4s %xmm0, (%rdi) # encoding: [0x62,0xf5,0xfe,0x08,0x3d,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbf82bf4s128_store:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtbf82bf4s %xmm0, (%eax) # encoding: [0x62,0xf5,0xfe,0x08,0x3d,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf82bf4s128(<16 x i8> %a)
+  %cast = bitcast <16 x i8> %ret to <2 x i64>
+  %low = extractelement <2 x i64> %cast, i64 0
+  store i64 %low, ptr %ptr, align 1
+  ret void
+}
+
+define void @test_int_x86_avx10_vcvtbf82bf4s256_store(ptr %ptr, <32 x i8> %a) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf82bf4s256_store:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtbf82bf4s %ymm0, (%rdi) # encoding: [0x62,0xf5,0xfe,0x28,0x3d,0x07]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbf82bf4s256_store:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtbf82bf4s %ymm0, (%eax) # encoding: [0x62,0xf5,0xfe,0x28,0x3d,0x00]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf82bf4s256(<32 x i8> %a)
+  store <16 x i8> %ret, ptr %ptr, align 1
+  ret void
+}
+
+define void @test_int_x86_avx10_vcvtbf82bf4s512_store(ptr %ptr, <64 x i8> %a) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf82bf4s512_store:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtbf82bf4s %zmm0, (%rdi) # encoding: [0x62,0xf5,0xfe,0x48,0x3d,0x07]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbf82bf4s512_store:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtbf82bf4s %zmm0, (%eax) # encoding: [0x62,0xf5,0xfe,0x48,0x3d,0x00]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <32 x i8> @llvm.x86.avx10.vcvtbf82bf4s512(<64 x i8> %a)
+  store <32 x i8> %ret, ptr %ptr, align 1
+  ret void
+}
+
+define void @test_int_x86_avx10_vcvthf82bf4s128_store(ptr %ptr, <16 x i8> %a) {
+; X64-LABEL: test_int_x86_avx10_vcvthf82bf4s128_store:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvthf82bf4s %xmm0, (%rdi) # encoding: [0x62,0xf5,0x7e,0x08,0x3d,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvthf82bf4s128_store:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvthf82bf4s %xmm0, (%eax) # encoding: [0x62,0xf5,0x7e,0x08,0x3d,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvthf82bf4s128(<16 x i8> %a)
+  %cast = bitcast <16 x i8> %ret to <2 x double>
+  %low = extractelement <2 x double> %cast, i64 0
+  store double %low, ptr %ptr, align 1
+  ret void
+}
+
+define void @test_int_x86_avx10_vcvthf82bf4s256_store(ptr %ptr, <32 x i8> %a) {
+; X64-LABEL: test_int_x86_avx10_vcvthf82bf4s256_store:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvthf82bf4s %ymm0, (%rdi) # encoding: [0x62,0xf5,0x7e,0x28,0x3d,0x07]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvthf82bf4s256_store:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvthf82bf4s %ymm0, (%eax) # encoding: [0x62,0xf5,0x7e,0x28,0x3d,0x00]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvthf82bf4s256(<32 x i8> %a)
+  store <16 x i8> %ret, ptr %ptr, align 1
+  ret void
+}
+
+define void @test_int_x86_avx10_vcvthf82bf4s512_store(ptr %ptr, <64 x i8> %a) {
+; X64-LABEL: test_int_x86_avx10_vcvthf82bf4s512_store:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvthf82bf4s %zmm0, (%rdi) # encoding: [0x62,0xf5,0x7e,0x48,0x3d,0x07]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvthf82bf4s512_store:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvthf82bf4s %zmm0, (%eax) # encoding: [0x62,0xf5,0x7e,0x48,0x3d,0x00]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <32 x i8> @llvm.x86.avx10.vcvthf82bf4s512(<64 x i8> %a)
+  store <32 x i8> %ret, ptr %ptr, align 1
+  ret void
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtbf82bf6s128(<16 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbf82bf6s128:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvtbf82bf6s %xmm0, %xmm0 # encoding: [0x62,0xf5,0xfe,0x08,0x3e,0xc0]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf82bf6s.128(<16 x i8> %a)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf82bf6s128(<16 x i8> %a)
   ret <16 x i8> %ret
 }
 
-define <32 x i8> @test_int_x86_avx10_vcvtbf82bf6s_256(<32 x i8> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbf82bf6s_256:
+define <32 x i8> @test_int_x86_avx10_vcvtbf82bf6s256(<32 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbf82bf6s256:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvtbf82bf6s %ymm0, %ymm0 # encoding: [0x62,0xf5,0xfe,0x28,0x3e,0xc0]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <32 x i8> @llvm.x86.avx10.vcvtbf82bf6s.256(<32 x i8> %a)
+  %ret = call <32 x i8> @llvm.x86.avx10.vcvtbf82bf6s256(<32 x i8> %a)
   ret <32 x i8> %ret
 }
 
-define <64 x i8> @test_int_x86_avx10_vcvtbf82bf6s_512(<64 x i8> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbf82bf6s_512:
+define <64 x i8> @test_int_x86_avx10_vcvtbf82bf6s512(<64 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbf82bf6s512:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvtbf82bf6s %zmm0, %zmm0 # encoding: [0x62,0xf5,0xfe,0x48,0x3e,0xc0]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <64 x i8> @llvm.x86.avx10.vcvtbf82bf6s.512(<64 x i8> %a)
+  %ret = call <64 x i8> @llvm.x86.avx10.vcvtbf82bf6s512(<64 x i8> %a)
   ret <64 x i8> %ret
 }
 
-declare <16 x i8> @llvm.x86.avx10.vcvtbf82bf6s.128(<16 x i8>)
-declare <32 x i8> @llvm.x86.avx10.vcvtbf82bf6s.256(<32 x i8>)
-declare <64 x i8> @llvm.x86.avx10.vcvtbf82bf6s.512(<64 x i8>)
+declare <16 x i8> @llvm.x86.avx10.vcvtbf82bf6s128(<16 x i8>)
+declare <32 x i8> @llvm.x86.avx10.vcvtbf82bf6s256(<32 x i8>)
+declare <64 x i8> @llvm.x86.avx10.vcvtbf82bf6s512(<64 x i8>)
 
-; Memory folding tests for vcvtbf82bf6s
-define <16 x i8> @test_int_x86_avx10_vcvtbf82bf6s_mem_128(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtbf82bf6s_mem_128:
+define <16 x i8> @test_int_x86_avx10_vcvtbf82bf6s128_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf82bf6s128_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vmovaps (%rdi), %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0x07]
 ; X64-NEXT:    vcvtbf82bf6s %xmm0, %xmm0 # encoding: [0x62,0xf5,0xfe,0x08,0x3e,0xc0]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_vcvtbf82bf6s_mem_128:
+; X86-LABEL: test_int_x86_avx10_vcvtbf82bf6s128_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    vmovaps (%eax), %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0x00]
 ; X86-NEXT:    vcvtbf82bf6s %xmm0, %xmm0 # encoding: [0x62,0xf5,0xfe,0x08,0x3e,0xc0]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %a = load <16 x i8>, ptr %ptr_a
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf82bf6s.128(<16 x i8> %a)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf82bf6s128(<16 x i8> %a)
   ret <16 x i8> %ret
 }
 
-define <32 x i8> @test_int_x86_avx10_vcvtbf82bf6s_mem_256(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtbf82bf6s_mem_256:
+define <32 x i8> @test_int_x86_avx10_vcvtbf82bf6s256_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf82bf6s256_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vmovaps (%rdi), %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfc,0x28,0x07]
 ; X64-NEXT:    vcvtbf82bf6s %ymm0, %ymm0 # encoding: [0x62,0xf5,0xfe,0x28,0x3e,0xc0]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_vcvtbf82bf6s_mem_256:
+; X86-LABEL: test_int_x86_avx10_vcvtbf82bf6s256_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    vmovaps (%eax), %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfc,0x28,0x00]
 ; X86-NEXT:    vcvtbf82bf6s %ymm0, %ymm0 # encoding: [0x62,0xf5,0xfe,0x28,0x3e,0xc0]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %a = load <32 x i8>, ptr %ptr_a
-  %ret = call <32 x i8> @llvm.x86.avx10.vcvtbf82bf6s.256(<32 x i8> %a)
+  %ret = call <32 x i8> @llvm.x86.avx10.vcvtbf82bf6s256(<32 x i8> %a)
   ret <32 x i8> %ret
 }
 
-define <64 x i8> @test_int_x86_avx10_vcvtbf82bf6s_mem_512(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtbf82bf6s_mem_512:
+define <64 x i8> @test_int_x86_avx10_vcvtbf82bf6s512_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf82bf6s512_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vmovaps (%rdi), %zmm0 # encoding: [0x62,0xf1,0x7c,0x48,0x28,0x07]
 ; X64-NEXT:    vcvtbf82bf6s %zmm0, %zmm0 # encoding: [0x62,0xf5,0xfe,0x48,0x3e,0xc0]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_vcvtbf82bf6s_mem_512:
+; X86-LABEL: test_int_x86_avx10_vcvtbf82bf6s512_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    vmovaps (%eax), %zmm0 # encoding: [0x62,0xf1,0x7c,0x48,0x28,0x00]
 ; X86-NEXT:    vcvtbf82bf6s %zmm0, %zmm0 # encoding: [0x62,0xf5,0xfe,0x48,0x3e,0xc0]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %a = load <64 x i8>, ptr %ptr_a
-  %ret = call <64 x i8> @llvm.x86.avx10.vcvtbf82bf6s.512(<64 x i8> %a)
+  %ret = call <64 x i8> @llvm.x86.avx10.vcvtbf82bf6s512(<64 x i8> %a)
   ret <64 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvthf82hf6s_128(<16 x i8> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvthf82hf6s_128:
+define <16 x i8> @test_int_x86_avx10_vcvthf82hf6s128(<16 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvthf82hf6s128:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvthf82hf6s %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x3c,0xc0]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvthf82hf6s.128(<16 x i8> %a)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvthf82hf6s128(<16 x i8> %a)
   ret <16 x i8> %ret
 }
 
-define <32 x i8> @test_int_x86_avx10_vcvthf82hf6s_256(<32 x i8> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvthf82hf6s_256:
+define <32 x i8> @test_int_x86_avx10_vcvthf82hf6s256(<32 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvthf82hf6s256:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvthf82hf6s %ymm0, %ymm0 # encoding: [0x62,0xf5,0x7e,0x28,0x3c,0xc0]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <32 x i8> @llvm.x86.avx10.vcvthf82hf6s.256(<32 x i8> %a)
+  %ret = call <32 x i8> @llvm.x86.avx10.vcvthf82hf6s256(<32 x i8> %a)
   ret <32 x i8> %ret
 }
 
-define <64 x i8> @test_int_x86_avx10_vcvthf82hf6s_512(<64 x i8> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvthf82hf6s_512:
+define <64 x i8> @test_int_x86_avx10_vcvthf82hf6s512(<64 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvthf82hf6s512:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvthf82hf6s %zmm0, %zmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x3c,0xc0]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <64 x i8> @llvm.x86.avx10.vcvthf82hf6s.512(<64 x i8> %a)
+  %ret = call <64 x i8> @llvm.x86.avx10.vcvthf82hf6s512(<64 x i8> %a)
   ret <64 x i8> %ret
 }
 
-declare <16 x i8> @llvm.x86.avx10.vcvthf82hf6s.128(<16 x i8>)
-declare <32 x i8> @llvm.x86.avx10.vcvthf82hf6s.256(<32 x i8>)
-declare <64 x i8> @llvm.x86.avx10.vcvthf82hf6s.512(<64 x i8>)
+declare <16 x i8> @llvm.x86.avx10.vcvthf82hf6s128(<16 x i8>)
+declare <32 x i8> @llvm.x86.avx10.vcvthf82hf6s256(<32 x i8>)
+declare <64 x i8> @llvm.x86.avx10.vcvthf82hf6s512(<64 x i8>)
 
-; Memory folding tests for vcvthf82hf6s
-define <16 x i8> @test_int_x86_avx10_vcvthf82hf6s_mem_128(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvthf82hf6s_mem_128:
+define <16 x i8> @test_int_x86_avx10_vcvthf82hf6s128_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvthf82hf6s128_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vmovaps (%rdi), %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0x07]
 ; X64-NEXT:    vcvthf82hf6s %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x3c,0xc0]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_vcvthf82hf6s_mem_128:
+; X86-LABEL: test_int_x86_avx10_vcvthf82hf6s128_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    vmovaps (%eax), %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0x00]
 ; X86-NEXT:    vcvthf82hf6s %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x3c,0xc0]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %a = load <16 x i8>, ptr %ptr_a
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvthf82hf6s.128(<16 x i8> %a)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvthf82hf6s128(<16 x i8> %a)
   ret <16 x i8> %ret
 }
 
-define <32 x i8> @test_int_x86_avx10_vcvthf82hf6s_mem_256(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvthf82hf6s_mem_256:
+define <32 x i8> @test_int_x86_avx10_vcvthf82hf6s256_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvthf82hf6s256_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vmovaps (%rdi), %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfc,0x28,0x07]
 ; X64-NEXT:    vcvthf82hf6s %ymm0, %ymm0 # encoding: [0x62,0xf5,0x7e,0x28,0x3c,0xc0]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_vcvthf82hf6s_mem_256:
+; X86-LABEL: test_int_x86_avx10_vcvthf82hf6s256_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    vmovaps (%eax), %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfc,0x28,0x00]
 ; X86-NEXT:    vcvthf82hf6s %ymm0, %ymm0 # encoding: [0x62,0xf5,0x7e,0x28,0x3c,0xc0]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %a = load <32 x i8>, ptr %ptr_a
-  %ret = call <32 x i8> @llvm.x86.avx10.vcvthf82hf6s.256(<32 x i8> %a)
+  %ret = call <32 x i8> @llvm.x86.avx10.vcvthf82hf6s256(<32 x i8> %a)
   ret <32 x i8> %ret
 }
 
-define <64 x i8> @test_int_x86_avx10_vcvthf82hf6s_mem_512(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvthf82hf6s_mem_512:
+define <64 x i8> @test_int_x86_avx10_vcvthf82hf6s512_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvthf82hf6s512_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vmovaps (%rdi), %zmm0 # encoding: [0x62,0xf1,0x7c,0x48,0x28,0x07]
 ; X64-NEXT:    vcvthf82hf6s %zmm0, %zmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x3c,0xc0]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_vcvthf82hf6s_mem_512:
+; X86-LABEL: test_int_x86_avx10_vcvthf82hf6s512_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    vmovaps (%eax), %zmm0 # encoding: [0x62,0xf1,0x7c,0x48,0x28,0x00]
 ; X86-NEXT:    vcvthf82hf6s %zmm0, %zmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x3c,0xc0]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %a = load <64 x i8>, ptr %ptr_a
-  %ret = call <64 x i8> @llvm.x86.avx10.vcvthf82hf6s.512(<64 x i8> %a)
+  %ret = call <64 x i8> @llvm.x86.avx10.vcvthf82hf6s512(<64 x i8> %a)
   ret <64 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtbf42hf8_128(<16 x i8> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbf42hf8_128:
+define <16 x i8> @test_int_x86_avx10_vcvtbf42hf8128(<16 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbf42hf8128:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvtbf42hf8 %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x37,0xc0]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf42hf8.128(<16 x i8> %a)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf42hf8128(<16 x i8> %a)
   ret <16 x i8> %ret
 }
 
-define <32 x i8> @test_int_x86_avx10_vcvtbf42hf8_256(<16 x i8> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbf42hf8_256:
+define <32 x i8> @test_int_x86_avx10_vcvtbf42hf8256(<16 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbf42hf8256:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvtbf42hf8 %xmm0, %ymm0 # encoding: [0x62,0xf5,0x7c,0x28,0x37,0xc0]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <32 x i8> @llvm.x86.avx10.vcvtbf42hf8.256(<16 x i8> %a)
+  %ret = call <32 x i8> @llvm.x86.avx10.vcvtbf42hf8256(<16 x i8> %a)
   ret <32 x i8> %ret
 }
 
-define <64 x i8> @test_int_x86_avx10_vcvtbf42hf8_512(<32 x i8> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbf42hf8_512:
+define <64 x i8> @test_int_x86_avx10_vcvtbf42hf8512(<32 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbf42hf8512:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvtbf42hf8 %ymm0, %zmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x37,0xc0]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <64 x i8> @llvm.x86.avx10.vcvtbf42hf8.512(<32 x i8> %a)
+  %ret = call <64 x i8> @llvm.x86.avx10.vcvtbf42hf8512(<32 x i8> %a)
   ret <64 x i8> %ret
 }
 
-declare <16 x i8> @llvm.x86.avx10.vcvtbf42hf8.128(<16 x i8>)
-declare <32 x i8> @llvm.x86.avx10.vcvtbf42hf8.256(<16 x i8>)
-declare <64 x i8> @llvm.x86.avx10.vcvtbf42hf8.512(<32 x i8>)
+declare <16 x i8> @llvm.x86.avx10.vcvtbf42hf8128(<16 x i8>)
+declare <32 x i8> @llvm.x86.avx10.vcvtbf42hf8256(<16 x i8>)
+declare <64 x i8> @llvm.x86.avx10.vcvtbf42hf8512(<32 x i8>)
 
-; Memory folding tests for vcvtbf42hf8
-define <16 x i8> @test_int_x86_avx10_vcvtbf42hf8_mem_128(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtbf42hf8_mem_128:
+define <16 x i8> @test_int_x86_avx10_vcvtbf42hf8128_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf42hf8128_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vcvtbf42hf8 (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x37,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_vcvtbf42hf8_mem_128:
+; X86-LABEL: test_int_x86_avx10_vcvtbf42hf8128_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    vcvtbf42hf8 (%eax), %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x37,0x00]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %a = load <16 x i8>, ptr %ptr_a
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf42hf8.128(<16 x i8> %a)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf42hf8128(<16 x i8> %a)
   ret <16 x i8> %ret
 }
 
-define <32 x i8> @test_int_x86_avx10_vcvtbf42hf8_mem_256(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtbf42hf8_mem_256:
+define <32 x i8> @test_int_x86_avx10_vcvtbf42hf8256_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf42hf8256_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vcvtbf42hf8 (%rdi), %ymm0 # encoding: [0x62,0xf5,0x7c,0x28,0x37,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_vcvtbf42hf8_mem_256:
+; X86-LABEL: test_int_x86_avx10_vcvtbf42hf8256_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    vcvtbf42hf8 (%eax), %ymm0 # encoding: [0x62,0xf5,0x7c,0x28,0x37,0x00]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %a = load <16 x i8>, ptr %ptr_a
-  %ret = call <32 x i8> @llvm.x86.avx10.vcvtbf42hf8.256(<16 x i8> %a)
+  %ret = call <32 x i8> @llvm.x86.avx10.vcvtbf42hf8256(<16 x i8> %a)
   ret <32 x i8> %ret
 }
 
-define <64 x i8> @test_int_x86_avx10_vcvtbf42hf8_mem_512(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtbf42hf8_mem_512:
+define <64 x i8> @test_int_x86_avx10_vcvtbf42hf8512_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf42hf8512_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vcvtbf42hf8 (%rdi), %zmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x37,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_vcvtbf42hf8_mem_512:
+; X86-LABEL: test_int_x86_avx10_vcvtbf42hf8512_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    vcvtbf42hf8 (%eax), %zmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x37,0x00]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %a = load <32 x i8>, ptr %ptr_a
-  %ret = call <64 x i8> @llvm.x86.avx10.vcvtbf42hf8.512(<32 x i8> %a)
+  %ret = call <64 x i8> @llvm.x86.avx10.vcvtbf42hf8512(<32 x i8> %a)
   ret <64 x i8> %ret
 }
 
-; The 128-bit form only reads 8 bytes, so a 64-bit zero-extending load
-; (_mm_loadu_si64) must fold into the memory operand too.
-define <16 x i8> @test_int_x86_avx10_vcvtbf42hf8_vzload_128(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtbf42hf8_vzload_128:
+define <16 x i8> @test_int_x86_avx10_vcvtbf42hf8128_vzload(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf42hf8128_vzload:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vcvtbf42hf8 (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x37,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_vcvtbf42hf8_vzload_128:
+; X86-LABEL: test_int_x86_avx10_vcvtbf42hf8128_vzload:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    vcvtbf42hf8 (%eax), %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x37,0x00]
@@ -1444,18 +1765,18 @@ define <16 x i8> @test_int_x86_avx10_vcvtbf42hf8_vzload_128(ptr %ptr_a) {
   %l = load i64, ptr %ptr_a, align 1
   %v = insertelement <2 x i64> <i64 poison, i64 0>, i64 %l, i64 0
   %a = bitcast <2 x i64> %v to <16 x i8>
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf42hf8.128(<16 x i8> %a)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf42hf8128(<16 x i8> %a)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtbf42hf8_vzload_mask_128(ptr %ptr_a, <16 x i8> %src, i16 %mask) {
-; X64-LABEL: test_int_x86_avx10_vcvtbf42hf8_vzload_mask_128:
+define <16 x i8> @test_int_x86_avx10_vcvtbf42hf8128_vzload_mask(ptr %ptr_a, <16 x i8> %src, i16 %mask) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf42hf8128_vzload_mask:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
 ; X64-NEXT:    vcvtbf42hf8 (%rdi), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x37,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_vcvtbf42hf8_vzload_mask_128:
+; X86-LABEL: test_int_x86_avx10_vcvtbf42hf8128_vzload_mask:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x08]
@@ -1464,20 +1785,20 @@ define <16 x i8> @test_int_x86_avx10_vcvtbf42hf8_vzload_mask_128(ptr %ptr_a, <16
   %l = load i64, ptr %ptr_a, align 1
   %v = insertelement <2 x i64> <i64 poison, i64 0>, i64 %l, i64 0
   %a = bitcast <2 x i64> %v to <16 x i8>
-  %cvt = call <16 x i8> @llvm.x86.avx10.vcvtbf42hf8.128(<16 x i8> %a)
+  %cvt = call <16 x i8> @llvm.x86.avx10.vcvtbf42hf8128(<16 x i8> %a)
   %msk = bitcast i16 %mask to <16 x i1>
   %ret = select <16 x i1> %msk, <16 x i8> %cvt, <16 x i8> %src
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtbf42hf8_vzload_maskz_128(ptr %ptr_a, i16 %mask) {
-; X64-LABEL: test_int_x86_avx10_vcvtbf42hf8_vzload_maskz_128:
+define <16 x i8> @test_int_x86_avx10_vcvtbf42hf8128_vzload_maskz(ptr %ptr_a, i16 %mask) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf42hf8128_vzload_maskz:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
 ; X64-NEXT:    vcvtbf42hf8 (%rdi), %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x37,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_vcvtbf42hf8_vzload_maskz_128:
+; X86-LABEL: test_int_x86_avx10_vcvtbf42hf8128_vzload_maskz:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x08]
@@ -1486,20 +1807,19 @@ define <16 x i8> @test_int_x86_avx10_vcvtbf42hf8_vzload_maskz_128(ptr %ptr_a, i1
   %l = load i64, ptr %ptr_a, align 1
   %v = insertelement <2 x i64> <i64 poison, i64 0>, i64 %l, i64 0
   %a = bitcast <2 x i64> %v to <16 x i8>
-  %cvt = call <16 x i8> @llvm.x86.avx10.vcvtbf42hf8.128(<16 x i8> %a)
+  %cvt = call <16 x i8> @llvm.x86.avx10.vcvtbf42hf8128(<16 x i8> %a)
   %msk = bitcast i16 %mask to <16 x i1>
   %ret = select <16 x i1> %msk, <16 x i8> %cvt, <16 x i8> zeroinitializer
   ret <16 x i8> %ret
 }
 
-; Same, spelled as a scalar_to_vector of an i64 load (_mm_loadl_epi64).
-define <16 x i8> @test_int_x86_avx10_vcvtbf42hf8_s2v_128(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtbf42hf8_s2v_128:
+define <16 x i8> @test_int_x86_avx10_vcvtbf42hf8_s2v128(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf42hf8_s2v128:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vcvtbf42hf8 (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x37,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_vcvtbf42hf8_s2v_128:
+; X86-LABEL: test_int_x86_avx10_vcvtbf42hf8_s2v128:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    vcvtbf42hf8 (%eax), %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x37,0x00]
@@ -1507,220 +1827,217 @@ define <16 x i8> @test_int_x86_avx10_vcvtbf42hf8_s2v_128(ptr %ptr_a) {
   %l = load i64, ptr %ptr_a, align 8
   %v = insertelement <2 x i64> poison, i64 %l, i64 0
   %a = bitcast <2 x i64> %v to <16 x i8>
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf42hf8.128(<16 x i8> %a)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf42hf8128(<16 x i8> %a)
   ret <16 x i8> %ret
 }
 
-; Masked variants of the plain 16-byte load, which must keep folding.
-define <16 x i8> @test_int_x86_avx10_vcvtbf42hf8_mem_mask_128(ptr %ptr_a, <16 x i8> %src, i16 %mask) {
-; X64-LABEL: test_int_x86_avx10_vcvtbf42hf8_mem_mask_128:
+define <16 x i8> @test_int_x86_avx10_vcvtbf42hf8128_mem_mask(ptr %ptr_a, <16 x i8> %src, i16 %mask) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf42hf8128_mem_mask:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
 ; X64-NEXT:    vcvtbf42hf8 (%rdi), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x37,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_vcvtbf42hf8_mem_mask_128:
+; X86-LABEL: test_int_x86_avx10_vcvtbf42hf8128_mem_mask:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x08]
 ; X86-NEXT:    vcvtbf42hf8 (%eax), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x37,0x00]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %a = load <16 x i8>, ptr %ptr_a
-  %cvt = call <16 x i8> @llvm.x86.avx10.vcvtbf42hf8.128(<16 x i8> %a)
+  %cvt = call <16 x i8> @llvm.x86.avx10.vcvtbf42hf8128(<16 x i8> %a)
   %msk = bitcast i16 %mask to <16 x i1>
   %ret = select <16 x i1> %msk, <16 x i8> %cvt, <16 x i8> %src
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtbf42hf8_mem_maskz_128(ptr %ptr_a, i16 %mask) {
-; X64-LABEL: test_int_x86_avx10_vcvtbf42hf8_mem_maskz_128:
+define <16 x i8> @test_int_x86_avx10_vcvtbf42hf8128_mem_maskz(ptr %ptr_a, i16 %mask) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf42hf8128_mem_maskz:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
 ; X64-NEXT:    vcvtbf42hf8 (%rdi), %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x37,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_vcvtbf42hf8_mem_maskz_128:
+; X86-LABEL: test_int_x86_avx10_vcvtbf42hf8128_mem_maskz:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x08]
 ; X86-NEXT:    vcvtbf42hf8 (%eax), %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x37,0x00]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %a = load <16 x i8>, ptr %ptr_a
-  %cvt = call <16 x i8> @llvm.x86.avx10.vcvtbf42hf8.128(<16 x i8> %a)
+  %cvt = call <16 x i8> @llvm.x86.avx10.vcvtbf42hf8128(<16 x i8> %a)
   %msk = bitcast i16 %mask to <16 x i1>
   %ret = select <16 x i1> %msk, <16 x i8> %cvt, <16 x i8> zeroinitializer
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvtbf62hf8_128(<16 x i8> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbf62hf8_128:
+define <16 x i8> @test_int_x86_avx10_vcvtbf62hf8128(<16 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbf62hf8128:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvtbf62hf8 %xmm0, %xmm0 # encoding: [0x62,0xf5,0xfd,0x08,0x37,0xc0]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf62hf8.128(<16 x i8> %a)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf62hf8128(<16 x i8> %a)
   ret <16 x i8> %ret
 }
 
-define <32 x i8> @test_int_x86_avx10_vcvtbf62hf8_256(<32 x i8> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbf62hf8_256:
+define <32 x i8> @test_int_x86_avx10_vcvtbf62hf8256(<32 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbf62hf8256:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvtbf62hf8 %ymm0, %ymm0 # encoding: [0x62,0xf5,0xfd,0x28,0x37,0xc0]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <32 x i8> @llvm.x86.avx10.vcvtbf62hf8.256(<32 x i8> %a)
+  %ret = call <32 x i8> @llvm.x86.avx10.vcvtbf62hf8256(<32 x i8> %a)
   ret <32 x i8> %ret
 }
 
-define <64 x i8> @test_int_x86_avx10_vcvtbf62hf8_512(<64 x i8> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbf62hf8_512:
+define <64 x i8> @test_int_x86_avx10_vcvtbf62hf8512(<64 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbf62hf8512:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvtbf62hf8 %zmm0, %zmm0 # encoding: [0x62,0xf5,0xfd,0x48,0x37,0xc0]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <64 x i8> @llvm.x86.avx10.vcvtbf62hf8.512(<64 x i8> %a)
+  %ret = call <64 x i8> @llvm.x86.avx10.vcvtbf62hf8512(<64 x i8> %a)
   ret <64 x i8> %ret
 }
 
-declare <16 x i8> @llvm.x86.avx10.vcvtbf62hf8.128(<16 x i8>)
-declare <32 x i8> @llvm.x86.avx10.vcvtbf62hf8.256(<32 x i8>)
-declare <64 x i8> @llvm.x86.avx10.vcvtbf62hf8.512(<64 x i8>)
+declare <16 x i8> @llvm.x86.avx10.vcvtbf62hf8128(<16 x i8>)
+declare <32 x i8> @llvm.x86.avx10.vcvtbf62hf8256(<32 x i8>)
+declare <64 x i8> @llvm.x86.avx10.vcvtbf62hf8512(<64 x i8>)
 
-; Memory folding tests for vcvtbf62hf8
-define <16 x i8> @test_int_x86_avx10_vcvtbf62hf8_mem_128(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtbf62hf8_mem_128:
+define <16 x i8> @test_int_x86_avx10_vcvtbf62hf8128_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf62hf8128_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vmovdqa (%rdi), %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf9,0x6f,0x07]
 ; X64-NEXT:    vcvtbf62hf8 %xmm0, %xmm0 # encoding: [0x62,0xf5,0xfd,0x08,0x37,0xc0]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_vcvtbf62hf8_mem_128:
+; X86-LABEL: test_int_x86_avx10_vcvtbf62hf8128_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    vmovdqa (%eax), %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf9,0x6f,0x00]
 ; X86-NEXT:    vcvtbf62hf8 %xmm0, %xmm0 # encoding: [0x62,0xf5,0xfd,0x08,0x37,0xc0]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %a = load <16 x i8>, ptr %ptr_a
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf62hf8.128(<16 x i8> %a)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf62hf8128(<16 x i8> %a)
   ret <16 x i8> %ret
 }
 
-define <32 x i8> @test_int_x86_avx10_vcvtbf62hf8_mem_256(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtbf62hf8_mem_256:
+define <32 x i8> @test_int_x86_avx10_vcvtbf62hf8256_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf62hf8256_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vmovdqa (%rdi), %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfd,0x6f,0x07]
 ; X64-NEXT:    vcvtbf62hf8 %ymm0, %ymm0 # encoding: [0x62,0xf5,0xfd,0x28,0x37,0xc0]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_vcvtbf62hf8_mem_256:
+; X86-LABEL: test_int_x86_avx10_vcvtbf62hf8256_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    vmovdqa (%eax), %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfd,0x6f,0x00]
 ; X86-NEXT:    vcvtbf62hf8 %ymm0, %ymm0 # encoding: [0x62,0xf5,0xfd,0x28,0x37,0xc0]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %a = load <32 x i8>, ptr %ptr_a
-  %ret = call <32 x i8> @llvm.x86.avx10.vcvtbf62hf8.256(<32 x i8> %a)
+  %ret = call <32 x i8> @llvm.x86.avx10.vcvtbf62hf8256(<32 x i8> %a)
   ret <32 x i8> %ret
 }
 
-define <64 x i8> @test_int_x86_avx10_vcvtbf62hf8_mem_512(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtbf62hf8_mem_512:
+define <64 x i8> @test_int_x86_avx10_vcvtbf62hf8512_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf62hf8512_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vmovdqa64 (%rdi), %zmm0 # encoding: [0x62,0xf1,0xfd,0x48,0x6f,0x07]
 ; X64-NEXT:    vcvtbf62hf8 %zmm0, %zmm0 # encoding: [0x62,0xf5,0xfd,0x48,0x37,0xc0]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_vcvtbf62hf8_mem_512:
+; X86-LABEL: test_int_x86_avx10_vcvtbf62hf8512_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    vmovdqa64 (%eax), %zmm0 # encoding: [0x62,0xf1,0xfd,0x48,0x6f,0x00]
 ; X86-NEXT:    vcvtbf62hf8 %zmm0, %zmm0 # encoding: [0x62,0xf5,0xfd,0x48,0x37,0xc0]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %a = load <64 x i8>, ptr %ptr_a
-  %ret = call <64 x i8> @llvm.x86.avx10.vcvtbf62hf8.512(<64 x i8> %a)
+  %ret = call <64 x i8> @llvm.x86.avx10.vcvtbf62hf8512(<64 x i8> %a)
   ret <64 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_vcvthf62hf8_128(<16 x i8> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvthf62hf8_128:
+define <16 x i8> @test_int_x86_avx10_vcvthf62hf8128(<16 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvthf62hf8128:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvthf62hf8 %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7d,0x08,0x37,0xc0]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvthf62hf8.128(<16 x i8> %a)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvthf62hf8128(<16 x i8> %a)
   ret <16 x i8> %ret
 }
 
-define <32 x i8> @test_int_x86_avx10_vcvthf62hf8_256(<32 x i8> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvthf62hf8_256:
+define <32 x i8> @test_int_x86_avx10_vcvthf62hf8256(<32 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvthf62hf8256:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvthf62hf8 %ymm0, %ymm0 # encoding: [0x62,0xf5,0x7d,0x28,0x37,0xc0]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <32 x i8> @llvm.x86.avx10.vcvthf62hf8.256(<32 x i8> %a)
+  %ret = call <32 x i8> @llvm.x86.avx10.vcvthf62hf8256(<32 x i8> %a)
   ret <32 x i8> %ret
 }
 
-define <64 x i8> @test_int_x86_avx10_vcvthf62hf8_512(<64 x i8> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvthf62hf8_512:
+define <64 x i8> @test_int_x86_avx10_vcvthf62hf8512(<64 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvthf62hf8512:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vcvthf62hf8 %zmm0, %zmm0 # encoding: [0x62,0xf5,0x7d,0x48,0x37,0xc0]
 ; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <64 x i8> @llvm.x86.avx10.vcvthf62hf8.512(<64 x i8> %a)
+  %ret = call <64 x i8> @llvm.x86.avx10.vcvthf62hf8512(<64 x i8> %a)
   ret <64 x i8> %ret
 }
 
-declare <16 x i8> @llvm.x86.avx10.vcvthf62hf8.128(<16 x i8>)
-declare <32 x i8> @llvm.x86.avx10.vcvthf62hf8.256(<32 x i8>)
-declare <64 x i8> @llvm.x86.avx10.vcvthf62hf8.512(<64 x i8>)
+declare <16 x i8> @llvm.x86.avx10.vcvthf62hf8128(<16 x i8>)
+declare <32 x i8> @llvm.x86.avx10.vcvthf62hf8256(<32 x i8>)
+declare <64 x i8> @llvm.x86.avx10.vcvthf62hf8512(<64 x i8>)
 
-; Memory folding tests for vcvthf62hf8
-define <16 x i8> @test_int_x86_avx10_vcvthf62hf8_mem_128(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvthf62hf8_mem_128:
+define <16 x i8> @test_int_x86_avx10_vcvthf62hf8128_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvthf62hf8128_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vmovdqa (%rdi), %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf9,0x6f,0x07]
 ; X64-NEXT:    vcvthf62hf8 %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7d,0x08,0x37,0xc0]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_vcvthf62hf8_mem_128:
+; X86-LABEL: test_int_x86_avx10_vcvthf62hf8128_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    vmovdqa (%eax), %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf9,0x6f,0x00]
 ; X86-NEXT:    vcvthf62hf8 %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7d,0x08,0x37,0xc0]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %a = load <16 x i8>, ptr %ptr_a
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvthf62hf8.128(<16 x i8> %a)
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvthf62hf8128(<16 x i8> %a)
   ret <16 x i8> %ret
 }
 
-define <32 x i8> @test_int_x86_avx10_vcvthf62hf8_mem_256(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvthf62hf8_mem_256:
+define <32 x i8> @test_int_x86_avx10_vcvthf62hf8256_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvthf62hf8256_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vmovdqa (%rdi), %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfd,0x6f,0x07]
 ; X64-NEXT:    vcvthf62hf8 %ymm0, %ymm0 # encoding: [0x62,0xf5,0x7d,0x28,0x37,0xc0]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_vcvthf62hf8_mem_256:
+; X86-LABEL: test_int_x86_avx10_vcvthf62hf8256_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    vmovdqa (%eax), %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfd,0x6f,0x00]
 ; X86-NEXT:    vcvthf62hf8 %ymm0, %ymm0 # encoding: [0x62,0xf5,0x7d,0x28,0x37,0xc0]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %a = load <32 x i8>, ptr %ptr_a
-  %ret = call <32 x i8> @llvm.x86.avx10.vcvthf62hf8.256(<32 x i8> %a)
+  %ret = call <32 x i8> @llvm.x86.avx10.vcvthf62hf8256(<32 x i8> %a)
   ret <32 x i8> %ret
 }
 
-define <64 x i8> @test_int_x86_avx10_vcvthf62hf8_mem_512(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvthf62hf8_mem_512:
+define <64 x i8> @test_int_x86_avx10_vcvthf62hf8512_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvthf62hf8512_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    vmovdqa64 (%rdi), %zmm0 # encoding: [0x62,0xf1,0xfd,0x48,0x6f,0x07]
 ; X64-NEXT:    vcvthf62hf8 %zmm0, %zmm0 # encoding: [0x62,0xf5,0x7d,0x48,0x37,0xc0]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_vcvthf62hf8_mem_512:
+; X86-LABEL: test_int_x86_avx10_vcvthf62hf8512_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    vmovdqa64 (%eax), %zmm0 # encoding: [0x62,0xf1,0xfd,0x48,0x6f,0x00]
 ; X86-NEXT:    vcvthf62hf8 %zmm0, %zmm0 # encoding: [0x62,0xf5,0x7d,0x48,0x37,0xc0]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %a = load <64 x i8>, ptr %ptr_a
-  %ret = call <64 x i8> @llvm.x86.avx10.vcvthf62hf8.512(<64 x i8> %a)
+  %ret = call <64 x i8> @llvm.x86.avx10.vcvthf62hf8512(<64 x i8> %a)
   ret <64 x i8> %ret
 }
 
@@ -1755,7 +2072,6 @@ declare <16 x i8> @llvm.x86.avx10.vunpackb.128(<16 x i8>, i8)
 declare <32 x i8> @llvm.x86.avx10.vunpackb.256(<32 x i8>, i8)
 declare <64 x i8> @llvm.x86.avx10.vunpackb.512(<64 x i8>, i8)
 
-; Memory folding tests for vunpackb
 define <16 x i8> @test_int_x86_avx10_vunpackb_mem_128(ptr %ptr_a) {
 ; X64-LABEL: test_int_x86_avx10_vunpackb_mem_128:
 ; X64:       # %bb.0:
@@ -1836,8 +2152,10 @@ define <16 x i8> @test_int_x86_avx10_pmovssdb_512(<16 x i32> %a) {
 declare <16 x i8> @llvm.x86.avx10.mask.pmovss.db.128(<4 x i32>, <16 x i8>, i8)
 declare <16 x i8> @llvm.x86.avx10.mask.pmovss.db.256(<8 x i32>, <16 x i8>, i8)
 declare <16 x i8> @llvm.x86.avx10.mask.pmovss.db.512(<16 x i32>, <16 x i8>, i16)
+declare void @llvm.x86.avx10.mask.pmovss.db.mem.128(ptr, <4 x i32>, i8)
+declare void @llvm.x86.avx10.mask.pmovss.db.mem.256(ptr, <8 x i32>, i8)
+declare void @llvm.x86.avx10.mask.pmovss.db.mem.512(ptr, <16 x i32>, i16)
 
-; Masked tests for vpmovssdb
 define <16 x i8> @test_int_x86_avx10_mask_pmovssdb_128(<4 x i32> %a, <16 x i8> %passthru, i8 %mask) {
 ; X64-LABEL: test_int_x86_avx10_mask_pmovssdb_128:
 ; X64:       # %bb.0:
@@ -1926,7 +2244,6 @@ define <16 x i8> @test_int_x86_avx10_mask_pmovssdb_512(<16 x i32> %a, <16 x i8>
   ret <16 x i8> %add2
 }
 
-; Memory folding tests for vpmovssdb
 define <16 x i8> @test_int_x86_avx10_pmovssdb_mem_128(ptr %ptr_a) {
 ; X64-LABEL: test_int_x86_avx10_pmovssdb_mem_128:
 ; X64:       # %bb.0:
@@ -1985,68 +2302,130 @@ define <16 x i8> @test_int_x86_avx10_pmovssdb_mem_512(ptr %ptr_a) {
   ret <16 x i8> %ret
 }
 
-; Masked 128/256-bit forms of vcvtps2bf8
+define void @test_int_x86_avx10_mask_pmovssdb_store_128(ptr %ptr, <4 x i32> %a, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_pmovssdb_store_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vpmovssdb %xmm0, (%rdi) # encoding: [0x62,0xf2,0x7e,0x08,0x41,0x07]
+; X64-NEXT:    vpmovssdb %xmm0, (%rdi) {%k1} # encoding: [0x62,0xf2,0x7e,0x09,0x41,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_pmovssdb_store_128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vpmovssdb %xmm0, (%eax) # encoding: [0x62,0xf2,0x7e,0x08,0x41,0x00]
+; X86-NEXT:    vpmovssdb %xmm0, (%eax) {%k1} # encoding: [0x62,0xf2,0x7e,0x09,0x41,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  call void @llvm.x86.avx10.mask.pmovss.db.mem.128(ptr %ptr, <4 x i32> %a, i8 -1)
+  call void @llvm.x86.avx10.mask.pmovss.db.mem.128(ptr %ptr, <4 x i32> %a, i8 %mask)
+  ret void
+}
 
-define <16 x i8> @test_int_x86_avx10_mask_vcvtps2bf8_128(<4 x float> %b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtps2bf8_128:
+define void @test_int_x86_avx10_mask_pmovssdb_store_256(ptr %ptr, <8 x i32> %a, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_pmovssdb_store_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vpmovssdb %ymm0, (%rdi) # encoding: [0x62,0xf2,0x7e,0x28,0x41,0x07]
+; X64-NEXT:    vpmovssdb %ymm0, (%rdi) {%k1} # encoding: [0x62,0xf2,0x7e,0x29,0x41,0x07]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_pmovssdb_store_256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vpmovssdb %ymm0, (%eax) # encoding: [0x62,0xf2,0x7e,0x28,0x41,0x00]
+; X86-NEXT:    vpmovssdb %ymm0, (%eax) {%k1} # encoding: [0x62,0xf2,0x7e,0x29,0x41,0x00]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  call void @llvm.x86.avx10.mask.pmovss.db.mem.256(ptr %ptr, <8 x i32> %a, i8 -1)
+  call void @llvm.x86.avx10.mask.pmovss.db.mem.256(ptr %ptr, <8 x i32> %a, i8 %mask)
+  ret void
+}
+
+define void @test_int_x86_avx10_mask_pmovssdb_store_512(ptr %ptr, <16 x i32> %a, i16 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_pmovssdb_store_512:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vpmovssdb %zmm0, (%rdi) # encoding: [0x62,0xf2,0x7e,0x48,0x41,0x07]
+; X64-NEXT:    vpmovssdb %zmm0, (%rdi) {%k1} # encoding: [0x62,0xf2,0x7e,0x49,0x41,0x07]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_pmovssdb_store_512:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vpmovssdb %zmm0, (%eax) # encoding: [0x62,0xf2,0x7e,0x48,0x41,0x00]
+; X86-NEXT:    vpmovssdb %zmm0, (%eax) {%k1} # encoding: [0x62,0xf2,0x7e,0x49,0x41,0x00]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  call void @llvm.x86.avx10.mask.pmovss.db.mem.512(ptr %ptr, <16 x i32> %a, i16 -1)
+  call void @llvm.x86.avx10.mask.pmovss.db.mem.512(ptr %ptr, <16 x i32> %a, i16 %mask)
+  ret void
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtps2bf8128(<4 x float> %b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtps2bf8128:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
 ; X64-NEXT:    vcvtps2bf8 %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x39,0xc8]
 ; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtps2bf8_128:
+; X86-LABEL: test_int_x86_avx10_mask_vcvtps2bf8128:
 ; X86:       # %bb.0:
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
 ; X86-NEXT:    vcvtps2bf8 %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x39,0xc8]
 ; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
 ; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.128(<4 x float> %b, <16 x i8> %src0, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8128(<4 x float> %b, <16 x i8> %src0, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2bf8_128(<4 x float> %b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8_128:
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2bf8128(<4 x float> %b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8128:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
 ; X64-NEXT:    vcvtps2bf8 %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x89,0x39,0xc0]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8_128:
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8128:
 ; X86:       # %bb.0:
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
 ; X86-NEXT:    vcvtps2bf8 %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x89,0x39,0xc0]
 ; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.128(<4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8128(<4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_mask_vcvtps2bf8_mem_128(ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtps2bf8_mem_128:
+define <16 x i8> @test_int_x86_avx10_mask_vcvtps2bf8128_mem(ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtps2bf8128_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
 ; X64-NEXT:    vcvtps2bf8x (%rdi), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x39,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtps2bf8_mem_128:
+; X86-LABEL: test_int_x86_avx10_mask_vcvtps2bf8128_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
 ; X86-NEXT:    vcvtps2bf8x (%eax), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x39,0x00]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %b = load <4 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.128(<4 x float> %b, <16 x i8> %src0, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8128(<4 x float> %b, <16 x i8> %src0, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2bf8_bcst_128(ptr %ptr_b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8_bcst_128:
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2bf8128_bcst(ptr %ptr_b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8128_bcst:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
 ; X64-NEXT:    vcvtps2bf8 (%rdi){1to4}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x99,0x39,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8_bcst_128:
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8128_bcst:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
@@ -2055,12 +2434,12 @@ define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2bf8_bcst_128(ptr %ptr_b, i8 %m
   %ld = load float, ptr %ptr_b
   %ins = insertelement <4 x float> poison, float %ld, i32 0
   %b = shufflevector <4 x float> %ins, <4 x float> poison, <4 x i32> zeroinitializer
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.128(<4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8128(<4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_mask_vcvtps2bf8_256(<8 x float> %b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtps2bf8_256:
+define <16 x i8> @test_int_x86_avx10_mask_vcvtps2bf8256(<8 x float> %b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtps2bf8256:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
 ; X64-NEXT:    vcvtps2bf8 %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x39,0xc8]
@@ -2068,61 +2447,61 @@ define <16 x i8> @test_int_x86_avx10_mask_vcvtps2bf8_256(<8 x float> %b, <16 x i
 ; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtps2bf8_256:
+; X86-LABEL: test_int_x86_avx10_mask_vcvtps2bf8256:
 ; X86:       # %bb.0:
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
 ; X86-NEXT:    vcvtps2bf8 %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x39,0xc8]
 ; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
 ; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.256(<8 x float> %b, <16 x i8> %src0, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8256(<8 x float> %b, <16 x i8> %src0, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2bf8_256(<8 x float> %b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8_256:
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2bf8256(<8 x float> %b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8256:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
 ; X64-NEXT:    vcvtps2bf8 %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xa9,0x39,0xc0]
 ; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8_256:
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8256:
 ; X86:       # %bb.0:
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
 ; X86-NEXT:    vcvtps2bf8 %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xa9,0x39,0xc0]
 ; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.256(<8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8256(<8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_mask_vcvtps2bf8_mem_256(ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtps2bf8_mem_256:
+define <16 x i8> @test_int_x86_avx10_mask_vcvtps2bf8256_mem(ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtps2bf8256_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
 ; X64-NEXT:    vcvtps2bf8y (%rdi), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x39,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtps2bf8_mem_256:
+; X86-LABEL: test_int_x86_avx10_mask_vcvtps2bf8256_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
 ; X86-NEXT:    vcvtps2bf8y (%eax), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x39,0x00]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %b = load <8 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.256(<8 x float> %b, <16 x i8> %src0, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8256(<8 x float> %b, <16 x i8> %src0, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2bf8_bcst_256(ptr %ptr_b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8_bcst_256:
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2bf8256_bcst(ptr %ptr_b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8256_bcst:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
 ; X64-NEXT:    vcvtps2bf8 (%rdi){1to8}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xb9,0x39,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8_bcst_256:
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8256_bcst:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
@@ -2131,75 +2510,73 @@ define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2bf8_bcst_256(ptr %ptr_b, i8 %m
   %ld = load float, ptr %ptr_b
   %ins = insertelement <8 x float> poison, float %ld, i32 0
   %b = shufflevector <8 x float> %ins, <8 x float> poison, <8 x i32> zeroinitializer
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.256(<8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8256(<8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
   ret <16 x i8> %ret
 }
 
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.128(<4 x float>, <16 x i8>, i8)
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8.256(<8 x float>, <16 x i8>, i8)
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8128(<4 x float>, <16 x i8>, i8)
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8256(<8 x float>, <16 x i8>, i8)
 
-; Masked 128/256-bit forms of vcvtps2bf8s
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtps2bf8s_128(<4 x float> %b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtps2bf8s_128:
+define <16 x i8> @test_int_x86_avx10_mask_vcvtps2bf8s128(<4 x float> %b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtps2bf8s128:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
 ; X64-NEXT:    vcvtps2bf8s %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x3b,0xc8]
 ; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtps2bf8s_128:
+; X86-LABEL: test_int_x86_avx10_mask_vcvtps2bf8s128:
 ; X86:       # %bb.0:
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
 ; X86-NEXT:    vcvtps2bf8s %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x3b,0xc8]
 ; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
 ; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.128(<4 x float> %b, <16 x i8> %src0, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s128(<4 x float> %b, <16 x i8> %src0, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2bf8s_128(<4 x float> %b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8s_128:
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2bf8s128(<4 x float> %b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8s128:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
 ; X64-NEXT:    vcvtps2bf8s %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x89,0x3b,0xc0]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8s_128:
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8s128:
 ; X86:       # %bb.0:
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
 ; X86-NEXT:    vcvtps2bf8s %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x89,0x3b,0xc0]
 ; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.128(<4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s128(<4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_mask_vcvtps2bf8s_mem_128(ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtps2bf8s_mem_128:
+define <16 x i8> @test_int_x86_avx10_mask_vcvtps2bf8s128_mem(ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtps2bf8s128_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
 ; X64-NEXT:    vcvtps2bf8sx (%rdi), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x3b,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtps2bf8s_mem_128:
+; X86-LABEL: test_int_x86_avx10_mask_vcvtps2bf8s128_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
 ; X86-NEXT:    vcvtps2bf8sx (%eax), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x3b,0x00]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %b = load <4 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.128(<4 x float> %b, <16 x i8> %src0, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s128(<4 x float> %b, <16 x i8> %src0, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2bf8s_bcst_128(ptr %ptr_b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8s_bcst_128:
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2bf8s128_bcst(ptr %ptr_b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8s128_bcst:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
 ; X64-NEXT:    vcvtps2bf8s (%rdi){1to4}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x99,0x3b,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8s_bcst_128:
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8s128_bcst:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
@@ -2208,12 +2585,12 @@ define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2bf8s_bcst_128(ptr %ptr_b, i8 %
   %ld = load float, ptr %ptr_b
   %ins = insertelement <4 x float> poison, float %ld, i32 0
   %b = shufflevector <4 x float> %ins, <4 x float> poison, <4 x i32> zeroinitializer
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.128(<4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s128(<4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_mask_vcvtps2bf8s_256(<8 x float> %b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtps2bf8s_256:
+define <16 x i8> @test_int_x86_avx10_mask_vcvtps2bf8s256(<8 x float> %b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtps2bf8s256:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
 ; X64-NEXT:    vcvtps2bf8s %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x3b,0xc8]
@@ -2221,61 +2598,61 @@ define <16 x i8> @test_int_x86_avx10_mask_vcvtps2bf8s_256(<8 x float> %b, <16 x
 ; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtps2bf8s_256:
+; X86-LABEL: test_int_x86_avx10_mask_vcvtps2bf8s256:
 ; X86:       # %bb.0:
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
 ; X86-NEXT:    vcvtps2bf8s %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x3b,0xc8]
 ; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
 ; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.256(<8 x float> %b, <16 x i8> %src0, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s256(<8 x float> %b, <16 x i8> %src0, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2bf8s_256(<8 x float> %b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8s_256:
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2bf8s256(<8 x float> %b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8s256:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
 ; X64-NEXT:    vcvtps2bf8s %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xa9,0x3b,0xc0]
 ; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8s_256:
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8s256:
 ; X86:       # %bb.0:
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
 ; X86-NEXT:    vcvtps2bf8s %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xa9,0x3b,0xc0]
 ; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.256(<8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s256(<8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_mask_vcvtps2bf8s_mem_256(ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtps2bf8s_mem_256:
+define <16 x i8> @test_int_x86_avx10_mask_vcvtps2bf8s256_mem(ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtps2bf8s256_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
 ; X64-NEXT:    vcvtps2bf8sy (%rdi), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x3b,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtps2bf8s_mem_256:
+; X86-LABEL: test_int_x86_avx10_mask_vcvtps2bf8s256_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
 ; X86-NEXT:    vcvtps2bf8sy (%eax), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x3b,0x00]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %b = load <8 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.256(<8 x float> %b, <16 x i8> %src0, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s256(<8 x float> %b, <16 x i8> %src0, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2bf8s_bcst_256(ptr %ptr_b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8s_bcst_256:
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2bf8s256_bcst(ptr %ptr_b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8s256_bcst:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
 ; X64-NEXT:    vcvtps2bf8s (%rdi){1to8}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xb9,0x3b,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8s_bcst_256:
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8s256_bcst:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
@@ -2284,75 +2661,73 @@ define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2bf8s_bcst_256(ptr %ptr_b, i8 %
   %ld = load float, ptr %ptr_b
   %ins = insertelement <8 x float> poison, float %ld, i32 0
   %b = shufflevector <8 x float> %ins, <8 x float> poison, <8 x i32> zeroinitializer
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.256(<8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s256(<8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
   ret <16 x i8> %ret
 }
 
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.128(<4 x float>, <16 x i8>, i8)
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s.256(<8 x float>, <16 x i8>, i8)
-
-; Masked 128/256-bit forms of vcvtps2hf8
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s128(<4 x float>, <16 x i8>, i8)
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s256(<8 x float>, <16 x i8>, i8)
 
-define <16 x i8> @test_int_x86_avx10_mask_vcvtps2hf8_128(<4 x float> %b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtps2hf8_128:
+define <16 x i8> @test_int_x86_avx10_mask_vcvtps2hf8128(<4 x float> %b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtps2hf8128:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
 ; X64-NEXT:    vcvtps2hf8 %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x38,0xc8]
 ; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtps2hf8_128:
+; X86-LABEL: test_int_x86_avx10_mask_vcvtps2hf8128:
 ; X86:       # %bb.0:
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
 ; X86-NEXT:    vcvtps2hf8 %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x38,0xc8]
 ; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
 ; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.128(<4 x float> %b, <16 x i8> %src0, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8128(<4 x float> %b, <16 x i8> %src0, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2hf8_128(<4 x float> %b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8_128:
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2hf8128(<4 x float> %b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8128:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
 ; X64-NEXT:    vcvtps2hf8 %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x89,0x38,0xc0]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8_128:
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8128:
 ; X86:       # %bb.0:
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
 ; X86-NEXT:    vcvtps2hf8 %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x89,0x38,0xc0]
 ; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.128(<4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8128(<4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_mask_vcvtps2hf8_mem_128(ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtps2hf8_mem_128:
+define <16 x i8> @test_int_x86_avx10_mask_vcvtps2hf8128_mem(ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtps2hf8128_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
 ; X64-NEXT:    vcvtps2hf8x (%rdi), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x38,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtps2hf8_mem_128:
+; X86-LABEL: test_int_x86_avx10_mask_vcvtps2hf8128_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
 ; X86-NEXT:    vcvtps2hf8x (%eax), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x38,0x00]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %b = load <4 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.128(<4 x float> %b, <16 x i8> %src0, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8128(<4 x float> %b, <16 x i8> %src0, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2hf8_bcst_128(ptr %ptr_b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8_bcst_128:
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2hf8128_bcst(ptr %ptr_b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8128_bcst:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
 ; X64-NEXT:    vcvtps2hf8 (%rdi){1to4}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x99,0x38,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8_bcst_128:
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8128_bcst:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
@@ -2361,12 +2736,12 @@ define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2hf8_bcst_128(ptr %ptr_b, i8 %m
   %ld = load float, ptr %ptr_b
   %ins = insertelement <4 x float> poison, float %ld, i32 0
   %b = shufflevector <4 x float> %ins, <4 x float> poison, <4 x i32> zeroinitializer
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.128(<4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8128(<4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_mask_vcvtps2hf8_256(<8 x float> %b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtps2hf8_256:
+define <16 x i8> @test_int_x86_avx10_mask_vcvtps2hf8256(<8 x float> %b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtps2hf8256:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
 ; X64-NEXT:    vcvtps2hf8 %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x38,0xc8]
@@ -2374,61 +2749,61 @@ define <16 x i8> @test_int_x86_avx10_mask_vcvtps2hf8_256(<8 x float> %b, <16 x i
 ; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtps2hf8_256:
+; X86-LABEL: test_int_x86_avx10_mask_vcvtps2hf8256:
 ; X86:       # %bb.0:
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
 ; X86-NEXT:    vcvtps2hf8 %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x38,0xc8]
 ; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
 ; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.256(<8 x float> %b, <16 x i8> %src0, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8256(<8 x float> %b, <16 x i8> %src0, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2hf8_256(<8 x float> %b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8_256:
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2hf8256(<8 x float> %b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8256:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
 ; X64-NEXT:    vcvtps2hf8 %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xa9,0x38,0xc0]
 ; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8_256:
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8256:
 ; X86:       # %bb.0:
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
 ; X86-NEXT:    vcvtps2hf8 %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xa9,0x38,0xc0]
 ; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.256(<8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8256(<8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_mask_vcvtps2hf8_mem_256(ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtps2hf8_mem_256:
+define <16 x i8> @test_int_x86_avx10_mask_vcvtps2hf8256_mem(ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtps2hf8256_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
 ; X64-NEXT:    vcvtps2hf8y (%rdi), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x38,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtps2hf8_mem_256:
+; X86-LABEL: test_int_x86_avx10_mask_vcvtps2hf8256_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
 ; X86-NEXT:    vcvtps2hf8y (%eax), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x38,0x00]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %b = load <8 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.256(<8 x float> %b, <16 x i8> %src0, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8256(<8 x float> %b, <16 x i8> %src0, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2hf8_bcst_256(ptr %ptr_b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8_bcst_256:
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2hf8256_bcst(ptr %ptr_b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8256_bcst:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
 ; X64-NEXT:    vcvtps2hf8 (%rdi){1to8}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xb9,0x38,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8_bcst_256:
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8256_bcst:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
@@ -2437,75 +2812,73 @@ define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2hf8_bcst_256(ptr %ptr_b, i8 %m
   %ld = load float, ptr %ptr_b
   %ins = insertelement <8 x float> poison, float %ld, i32 0
   %b = shufflevector <8 x float> %ins, <8 x float> poison, <8 x i32> zeroinitializer
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.256(<8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8256(<8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
   ret <16 x i8> %ret
 }
 
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.128(<4 x float>, <16 x i8>, i8)
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8.256(<8 x float>, <16 x i8>, i8)
-
-; Masked 128/256-bit forms of vcvtps2hf8s
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8128(<4 x float>, <16 x i8>, i8)
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8256(<8 x float>, <16 x i8>, i8)
 
-define <16 x i8> @test_int_x86_avx10_mask_vcvtps2hf8s_128(<4 x float> %b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtps2hf8s_128:
+define <16 x i8> @test_int_x86_avx10_mask_vcvtps2hf8s128(<4 x float> %b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtps2hf8s128:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
 ; X64-NEXT:    vcvtps2hf8s %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x3a,0xc8]
 ; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtps2hf8s_128:
+; X86-LABEL: test_int_x86_avx10_mask_vcvtps2hf8s128:
 ; X86:       # %bb.0:
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
 ; X86-NEXT:    vcvtps2hf8s %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x3a,0xc8]
 ; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
 ; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.128(<4 x float> %b, <16 x i8> %src0, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s128(<4 x float> %b, <16 x i8> %src0, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2hf8s_128(<4 x float> %b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8s_128:
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2hf8s128(<4 x float> %b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8s128:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
 ; X64-NEXT:    vcvtps2hf8s %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x89,0x3a,0xc0]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8s_128:
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8s128:
 ; X86:       # %bb.0:
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
 ; X86-NEXT:    vcvtps2hf8s %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x89,0x3a,0xc0]
 ; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.128(<4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s128(<4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_mask_vcvtps2hf8s_mem_128(ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtps2hf8s_mem_128:
+define <16 x i8> @test_int_x86_avx10_mask_vcvtps2hf8s128_mem(ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtps2hf8s128_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
 ; X64-NEXT:    vcvtps2hf8sx (%rdi), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x3a,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtps2hf8s_mem_128:
+; X86-LABEL: test_int_x86_avx10_mask_vcvtps2hf8s128_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
 ; X86-NEXT:    vcvtps2hf8sx (%eax), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x3a,0x00]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %b = load <4 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.128(<4 x float> %b, <16 x i8> %src0, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s128(<4 x float> %b, <16 x i8> %src0, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2hf8s_bcst_128(ptr %ptr_b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8s_bcst_128:
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2hf8s128_bcst(ptr %ptr_b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8s128_bcst:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
 ; X64-NEXT:    vcvtps2hf8s (%rdi){1to4}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x99,0x3a,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8s_bcst_128:
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8s128_bcst:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
@@ -2514,12 +2887,12 @@ define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2hf8s_bcst_128(ptr %ptr_b, i8 %
   %ld = load float, ptr %ptr_b
   %ins = insertelement <4 x float> poison, float %ld, i32 0
   %b = shufflevector <4 x float> %ins, <4 x float> poison, <4 x i32> zeroinitializer
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.128(<4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s128(<4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_mask_vcvtps2hf8s_256(<8 x float> %b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtps2hf8s_256:
+define <16 x i8> @test_int_x86_avx10_mask_vcvtps2hf8s256(<8 x float> %b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtps2hf8s256:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
 ; X64-NEXT:    vcvtps2hf8s %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x3a,0xc8]
@@ -2527,61 +2900,61 @@ define <16 x i8> @test_int_x86_avx10_mask_vcvtps2hf8s_256(<8 x float> %b, <16 x
 ; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtps2hf8s_256:
+; X86-LABEL: test_int_x86_avx10_mask_vcvtps2hf8s256:
 ; X86:       # %bb.0:
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
 ; X86-NEXT:    vcvtps2hf8s %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x3a,0xc8]
 ; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
 ; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.256(<8 x float> %b, <16 x i8> %src0, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s256(<8 x float> %b, <16 x i8> %src0, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2hf8s_256(<8 x float> %b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8s_256:
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2hf8s256(<8 x float> %b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8s256:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
 ; X64-NEXT:    vcvtps2hf8s %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xa9,0x3a,0xc0]
 ; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8s_256:
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8s256:
 ; X86:       # %bb.0:
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
 ; X86-NEXT:    vcvtps2hf8s %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xa9,0x3a,0xc0]
 ; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.256(<8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s256(<8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_mask_vcvtps2hf8s_mem_256(ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtps2hf8s_mem_256:
+define <16 x i8> @test_int_x86_avx10_mask_vcvtps2hf8s256_mem(ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtps2hf8s256_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
 ; X64-NEXT:    vcvtps2hf8sy (%rdi), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x3a,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtps2hf8s_mem_256:
+; X86-LABEL: test_int_x86_avx10_mask_vcvtps2hf8s256_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
 ; X86-NEXT:    vcvtps2hf8sy (%eax), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x3a,0x00]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %b = load <8 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.256(<8 x float> %b, <16 x i8> %src0, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s256(<8 x float> %b, <16 x i8> %src0, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2hf8s_bcst_256(ptr %ptr_b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8s_bcst_256:
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2hf8s256_bcst(ptr %ptr_b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8s256_bcst:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
 ; X64-NEXT:    vcvtps2hf8s (%rdi){1to8}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xb9,0x3a,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8s_bcst_256:
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8s256_bcst:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
@@ -2590,75 +2963,73 @@ define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2hf8s_bcst_256(ptr %ptr_b, i8 %
   %ld = load float, ptr %ptr_b
   %ins = insertelement <8 x float> poison, float %ld, i32 0
   %b = shufflevector <8 x float> %ins, <8 x float> poison, <8 x i32> zeroinitializer
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.256(<8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s256(<8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
   ret <16 x i8> %ret
 }
 
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.128(<4 x float>, <16 x i8>, i8)
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s.256(<8 x float>, <16 x i8>, i8)
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s128(<4 x float>, <16 x i8>, i8)
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s256(<8 x float>, <16 x i8>, i8)
 
-; Masked 128/256-bit forms of vcvtrops2hf8
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtrops2hf8_128(<4 x float> %b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8_128:
+define <16 x i8> @test_int_x86_avx10_mask_vcvtrops2hf8128(<4 x float> %b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8128:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
 ; X64-NEXT:    vcvtrops2hf8 %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7d,0x09,0x38,0xc8]
 ; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8_128:
+; X86-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8128:
 ; X86:       # %bb.0:
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
 ; X86-NEXT:    vcvtrops2hf8 %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7d,0x09,0x38,0xc8]
 ; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
 ; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.128(<4 x float> %b, <16 x i8> %src0, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8128(<4 x float> %b, <16 x i8> %src0, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtrops2hf8_128(<4 x float> %b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8_128:
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtrops2hf8128(<4 x float> %b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8128:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
 ; X64-NEXT:    vcvtrops2hf8 %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0x89,0x38,0xc0]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8_128:
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8128:
 ; X86:       # %bb.0:
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
 ; X86-NEXT:    vcvtrops2hf8 %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0x89,0x38,0xc0]
 ; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.128(<4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8128(<4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_mask_vcvtrops2hf8_mem_128(ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8_mem_128:
+define <16 x i8> @test_int_x86_avx10_mask_vcvtrops2hf8128_mem(ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8128_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
 ; X64-NEXT:    vcvtrops2hf8x (%rdi), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7d,0x09,0x38,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8_mem_128:
+; X86-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8128_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
 ; X86-NEXT:    vcvtrops2hf8x (%eax), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7d,0x09,0x38,0x00]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %b = load <4 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.128(<4 x float> %b, <16 x i8> %src0, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8128(<4 x float> %b, <16 x i8> %src0, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtrops2hf8_bcst_128(ptr %ptr_b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8_bcst_128:
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtrops2hf8128_bcst(ptr %ptr_b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8128_bcst:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
 ; X64-NEXT:    vcvtrops2hf8 (%rdi){1to4}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0x99,0x38,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8_bcst_128:
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8128_bcst:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
@@ -2667,12 +3038,12 @@ define <16 x i8> @test_int_x86_avx10_maskz_vcvtrops2hf8_bcst_128(ptr %ptr_b, i8
   %ld = load float, ptr %ptr_b
   %ins = insertelement <4 x float> poison, float %ld, i32 0
   %b = shufflevector <4 x float> %ins, <4 x float> poison, <4 x i32> zeroinitializer
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.128(<4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8128(<4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_mask_vcvtrops2hf8_256(<8 x float> %b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8_256:
+define <16 x i8> @test_int_x86_avx10_mask_vcvtrops2hf8256(<8 x float> %b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8256:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
 ; X64-NEXT:    vcvtrops2hf8 %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7d,0x29,0x38,0xc8]
@@ -2680,61 +3051,61 @@ define <16 x i8> @test_int_x86_avx10_mask_vcvtrops2hf8_256(<8 x float> %b, <16 x
 ; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8_256:
+; X86-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8256:
 ; X86:       # %bb.0:
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
 ; X86-NEXT:    vcvtrops2hf8 %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7d,0x29,0x38,0xc8]
 ; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
 ; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.256(<8 x float> %b, <16 x i8> %src0, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8256(<8 x float> %b, <16 x i8> %src0, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtrops2hf8_256(<8 x float> %b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8_256:
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtrops2hf8256(<8 x float> %b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8256:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
 ; X64-NEXT:    vcvtrops2hf8 %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0xa9,0x38,0xc0]
 ; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8_256:
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8256:
 ; X86:       # %bb.0:
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
 ; X86-NEXT:    vcvtrops2hf8 %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0xa9,0x38,0xc0]
 ; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.256(<8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8256(<8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_mask_vcvtrops2hf8_mem_256(ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8_mem_256:
+define <16 x i8> @test_int_x86_avx10_mask_vcvtrops2hf8256_mem(ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8256_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
 ; X64-NEXT:    vcvtrops2hf8y (%rdi), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7d,0x29,0x38,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8_mem_256:
+; X86-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8256_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
 ; X86-NEXT:    vcvtrops2hf8y (%eax), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7d,0x29,0x38,0x00]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %b = load <8 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.256(<8 x float> %b, <16 x i8> %src0, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8256(<8 x float> %b, <16 x i8> %src0, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtrops2hf8_bcst_256(ptr %ptr_b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8_bcst_256:
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtrops2hf8256_bcst(ptr %ptr_b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8256_bcst:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
 ; X64-NEXT:    vcvtrops2hf8 (%rdi){1to8}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0xb9,0x38,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8_bcst_256:
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8256_bcst:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
@@ -2743,75 +3114,73 @@ define <16 x i8> @test_int_x86_avx10_maskz_vcvtrops2hf8_bcst_256(ptr %ptr_b, i8
   %ld = load float, ptr %ptr_b
   %ins = insertelement <8 x float> poison, float %ld, i32 0
   %b = shufflevector <8 x float> %ins, <8 x float> poison, <8 x i32> zeroinitializer
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.256(<8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8256(<8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
   ret <16 x i8> %ret
 }
 
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.128(<4 x float>, <16 x i8>, i8)
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8.256(<8 x float>, <16 x i8>, i8)
-
-; Masked 128/256-bit forms of vcvtrops2hf8s
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8128(<4 x float>, <16 x i8>, i8)
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8256(<8 x float>, <16 x i8>, i8)
 
-define <16 x i8> @test_int_x86_avx10_mask_vcvtrops2hf8s_128(<4 x float> %b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8s_128:
+define <16 x i8> @test_int_x86_avx10_mask_vcvtrops2hf8s128(<4 x float> %b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8s128:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
 ; X64-NEXT:    vcvtrops2hf8s %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7d,0x09,0x3a,0xc8]
 ; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8s_128:
+; X86-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8s128:
 ; X86:       # %bb.0:
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
 ; X86-NEXT:    vcvtrops2hf8s %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7d,0x09,0x3a,0xc8]
 ; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
 ; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.128(<4 x float> %b, <16 x i8> %src0, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s128(<4 x float> %b, <16 x i8> %src0, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtrops2hf8s_128(<4 x float> %b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8s_128:
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtrops2hf8s128(<4 x float> %b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8s128:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
 ; X64-NEXT:    vcvtrops2hf8s %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0x89,0x3a,0xc0]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8s_128:
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8s128:
 ; X86:       # %bb.0:
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
 ; X86-NEXT:    vcvtrops2hf8s %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0x89,0x3a,0xc0]
 ; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.128(<4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s128(<4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_mask_vcvtrops2hf8s_mem_128(ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8s_mem_128:
+define <16 x i8> @test_int_x86_avx10_mask_vcvtrops2hf8s128_mem(ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8s128_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
 ; X64-NEXT:    vcvtrops2hf8sx (%rdi), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7d,0x09,0x3a,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8s_mem_128:
+; X86-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8s128_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
 ; X86-NEXT:    vcvtrops2hf8sx (%eax), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7d,0x09,0x3a,0x00]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %b = load <4 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.128(<4 x float> %b, <16 x i8> %src0, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s128(<4 x float> %b, <16 x i8> %src0, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtrops2hf8s_bcst_128(ptr %ptr_b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8s_bcst_128:
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtrops2hf8s128_bcst(ptr %ptr_b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8s128_bcst:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
 ; X64-NEXT:    vcvtrops2hf8s (%rdi){1to4}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0x99,0x3a,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8s_bcst_128:
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8s128_bcst:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
@@ -2820,12 +3189,12 @@ define <16 x i8> @test_int_x86_avx10_maskz_vcvtrops2hf8s_bcst_128(ptr %ptr_b, i8
   %ld = load float, ptr %ptr_b
   %ins = insertelement <4 x float> poison, float %ld, i32 0
   %b = shufflevector <4 x float> %ins, <4 x float> poison, <4 x i32> zeroinitializer
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.128(<4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s128(<4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_mask_vcvtrops2hf8s_256(<8 x float> %b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8s_256:
+define <16 x i8> @test_int_x86_avx10_mask_vcvtrops2hf8s256(<8 x float> %b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8s256:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
 ; X64-NEXT:    vcvtrops2hf8s %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7d,0x29,0x3a,0xc8]
@@ -2833,61 +3202,61 @@ define <16 x i8> @test_int_x86_avx10_mask_vcvtrops2hf8s_256(<8 x float> %b, <16
 ; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8s_256:
+; X86-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8s256:
 ; X86:       # %bb.0:
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
 ; X86-NEXT:    vcvtrops2hf8s %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7d,0x29,0x3a,0xc8]
 ; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
 ; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.256(<8 x float> %b, <16 x i8> %src0, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s256(<8 x float> %b, <16 x i8> %src0, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtrops2hf8s_256(<8 x float> %b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8s_256:
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtrops2hf8s256(<8 x float> %b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8s256:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
 ; X64-NEXT:    vcvtrops2hf8s %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0xa9,0x3a,0xc0]
 ; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8s_256:
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8s256:
 ; X86:       # %bb.0:
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
 ; X86-NEXT:    vcvtrops2hf8s %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0xa9,0x3a,0xc0]
 ; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.256(<8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s256(<8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_mask_vcvtrops2hf8s_mem_256(ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8s_mem_256:
+define <16 x i8> @test_int_x86_avx10_mask_vcvtrops2hf8s256_mem(ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8s256_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
 ; X64-NEXT:    vcvtrops2hf8sy (%rdi), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7d,0x29,0x3a,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8s_mem_256:
+; X86-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8s256_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
 ; X86-NEXT:    vcvtrops2hf8sy (%eax), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7d,0x29,0x3a,0x00]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %b = load <8 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.256(<8 x float> %b, <16 x i8> %src0, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s256(<8 x float> %b, <16 x i8> %src0, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtrops2hf8s_bcst_256(ptr %ptr_b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8s_bcst_256:
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtrops2hf8s256_bcst(ptr %ptr_b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8s256_bcst:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
 ; X64-NEXT:    vcvtrops2hf8s (%rdi){1to8}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0xb9,0x3a,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8s_bcst_256:
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8s256_bcst:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
@@ -2896,58 +3265,56 @@ define <16 x i8> @test_int_x86_avx10_maskz_vcvtrops2hf8s_bcst_256(ptr %ptr_b, i8
   %ld = load float, ptr %ptr_b
   %ins = insertelement <8 x float> poison, float %ld, i32 0
   %b = shufflevector <8 x float> %ins, <8 x float> poison, <8 x i32> zeroinitializer
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.256(<8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s256(<8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
   ret <16 x i8> %ret
 }
 
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.128(<4 x float>, <16 x i8>, i8)
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s.256(<8 x float>, <16 x i8>, i8)
-
-; Masked 128/256-bit forms of vcvtbiasps2bf8
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s128(<4 x float>, <16 x i8>, i8)
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s256(<8 x float>, <16 x i8>, i8)
 
-define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2bf8_128(<4 x i32> %A, <4 x float> %b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8_128:
+define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2bf8128(<4 x i32> %A, <4 x float> %b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8128:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
 ; X64-NEXT:    vcvtbiasps2bf8 %xmm1, %xmm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x39,0xd1]
 ; X64-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8_128:
+; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8128:
 ; X86:       # %bb.0:
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
 ; X86-NEXT:    vcvtbiasps2bf8 %xmm1, %xmm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x39,0xd1]
 ; X86-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
 ; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.128(<4 x i32> %A, <4 x float> %b, <16 x i8> %src0, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8128(<4 x i32> %A, <4 x float> %b, <16 x i8> %src0, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2bf8_128(<4 x i32> %A, <4 x float> %b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8_128:
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2bf8128(<4 x i32> %A, <4 x float> %b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8128:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
 ; X64-NEXT:    vcvtbiasps2bf8 %xmm1, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x39,0xc1]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8_128:
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8128:
 ; X86:       # %bb.0:
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
 ; X86-NEXT:    vcvtbiasps2bf8 %xmm1, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x39,0xc1]
 ; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.128(<4 x i32> %A, <4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8128(<4 x i32> %A, <4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2bf8_mem_128(<4 x i32> %A, ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8_mem_128:
+define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2bf8128_mem(<4 x i32> %A, ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8128_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
 ; X64-NEXT:    vcvtbiasps2bf8 (%rdi), %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x39,0x0f]
 ; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8_mem_128:
+; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8128_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
@@ -2955,18 +3322,18 @@ define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2bf8_mem_128(<4 x i32> %A, p
 ; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %b = load <4 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.128(<4 x i32> %A, <4 x float> %b, <16 x i8> %src0, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8128(<4 x i32> %A, <4 x float> %b, <16 x i8> %src0, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2bf8_bcst_128(<4 x i32> %A, ptr %ptr_b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8_bcst_128:
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2bf8128_bcst(<4 x i32> %A, ptr %ptr_b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8128_bcst:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
 ; X64-NEXT:    vcvtbiasps2bf8 (%rdi){1to4}, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x99,0x39,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8_bcst_128:
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8128_bcst:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
@@ -2975,12 +3342,12 @@ define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2bf8_bcst_128(<4 x i32> %A,
   %ld = load float, ptr %ptr_b
   %ins = insertelement <4 x float> poison, float %ld, i32 0
   %b = shufflevector <4 x float> %ins, <4 x float> poison, <4 x i32> zeroinitializer
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.128(<4 x i32> %A, <4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8128(<4 x i32> %A, <4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2bf8_256(<8 x i32> %A, <8 x float> %b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8_256:
+define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2bf8256(<8 x i32> %A, <8 x float> %b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8256:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
 ; X64-NEXT:    vcvtbiasps2bf8 %ymm1, %ymm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x39,0xd1]
@@ -2988,37 +3355,37 @@ define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2bf8_256(<8 x i32> %A, <8 x
 ; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8_256:
+; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8256:
 ; X86:       # %bb.0:
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
 ; X86-NEXT:    vcvtbiasps2bf8 %ymm1, %ymm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x39,0xd1]
 ; X86-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
 ; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.256(<8 x i32> %A, <8 x float> %b, <16 x i8> %src0, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8256(<8 x i32> %A, <8 x float> %b, <16 x i8> %src0, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2bf8_256(<8 x i32> %A, <8 x float> %b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8_256:
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2bf8256(<8 x i32> %A, <8 x float> %b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8256:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
 ; X64-NEXT:    vcvtbiasps2bf8 %ymm1, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xa9,0x39,0xc1]
 ; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8_256:
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8256:
 ; X86:       # %bb.0:
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
 ; X86-NEXT:    vcvtbiasps2bf8 %ymm1, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xa9,0x39,0xc1]
 ; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.256(<8 x i32> %A, <8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8256(<8 x i32> %A, <8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2bf8_mem_256(<8 x i32> %A, ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8_mem_256:
+define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2bf8256_mem(<8 x i32> %A, ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8256_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
 ; X64-NEXT:    vcvtbiasps2bf8 (%rdi), %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x39,0x0f]
@@ -3026,7 +3393,7 @@ define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2bf8_mem_256(<8 x i32> %A, p
 ; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8_mem_256:
+; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8256_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
@@ -3035,19 +3402,19 @@ define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2bf8_mem_256(<8 x i32> %A, p
 ; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %b = load <8 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.256(<8 x i32> %A, <8 x float> %b, <16 x i8> %src0, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8256(<8 x i32> %A, <8 x float> %b, <16 x i8> %src0, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2bf8_bcst_256(<8 x i32> %A, ptr %ptr_b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8_bcst_256:
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2bf8256_bcst(<8 x i32> %A, ptr %ptr_b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8256_bcst:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
 ; X64-NEXT:    vcvtbiasps2bf8 (%rdi){1to8}, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xb9,0x39,0x07]
 ; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8_bcst_256:
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8256_bcst:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
@@ -3057,58 +3424,56 @@ define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2bf8_bcst_256(<8 x i32> %A,
   %ld = load float, ptr %ptr_b
   %ins = insertelement <8 x float> poison, float %ld, i32 0
   %b = shufflevector <8 x float> %ins, <8 x float> poison, <8 x i32> zeroinitializer
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.256(<8 x i32> %A, <8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8256(<8 x i32> %A, <8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
   ret <16 x i8> %ret
 }
 
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.128(<4 x i32>, <4 x float>, <16 x i8>, i8)
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8.256(<8 x i32>, <8 x float>, <16 x i8>, i8)
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8128(<4 x i32>, <4 x float>, <16 x i8>, i8)
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8256(<8 x i32>, <8 x float>, <16 x i8>, i8)
 
-; Masked 128/256-bit forms of vcvtbiasps2bf8s
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2bf8s_128(<4 x i32> %A, <4 x float> %b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8s_128:
+define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2bf8s128(<4 x i32> %A, <4 x float> %b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8s128:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
 ; X64-NEXT:    vcvtbiasps2bf8s %xmm1, %xmm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x3b,0xd1]
 ; X64-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8s_128:
+; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8s128:
 ; X86:       # %bb.0:
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
 ; X86-NEXT:    vcvtbiasps2bf8s %xmm1, %xmm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x3b,0xd1]
 ; X86-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
 ; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.128(<4 x i32> %A, <4 x float> %b, <16 x i8> %src0, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s128(<4 x i32> %A, <4 x float> %b, <16 x i8> %src0, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2bf8s_128(<4 x i32> %A, <4 x float> %b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8s_128:
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2bf8s128(<4 x i32> %A, <4 x float> %b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8s128:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
 ; X64-NEXT:    vcvtbiasps2bf8s %xmm1, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x3b,0xc1]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8s_128:
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8s128:
 ; X86:       # %bb.0:
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
 ; X86-NEXT:    vcvtbiasps2bf8s %xmm1, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x3b,0xc1]
 ; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.128(<4 x i32> %A, <4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s128(<4 x i32> %A, <4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2bf8s_mem_128(<4 x i32> %A, ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8s_mem_128:
+define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2bf8s128_mem(<4 x i32> %A, ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8s128_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
 ; X64-NEXT:    vcvtbiasps2bf8s (%rdi), %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x3b,0x0f]
 ; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8s_mem_128:
+; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8s128_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
@@ -3116,18 +3481,18 @@ define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2bf8s_mem_128(<4 x i32> %A,
 ; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %b = load <4 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.128(<4 x i32> %A, <4 x float> %b, <16 x i8> %src0, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s128(<4 x i32> %A, <4 x float> %b, <16 x i8> %src0, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2bf8s_bcst_128(<4 x i32> %A, ptr %ptr_b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8s_bcst_128:
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2bf8s128_bcst(<4 x i32> %A, ptr %ptr_b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8s128_bcst:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
 ; X64-NEXT:    vcvtbiasps2bf8s (%rdi){1to4}, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x99,0x3b,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8s_bcst_128:
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8s128_bcst:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
@@ -3136,12 +3501,12 @@ define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2bf8s_bcst_128(<4 x i32> %A
   %ld = load float, ptr %ptr_b
   %ins = insertelement <4 x float> poison, float %ld, i32 0
   %b = shufflevector <4 x float> %ins, <4 x float> poison, <4 x i32> zeroinitializer
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.128(<4 x i32> %A, <4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s128(<4 x i32> %A, <4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2bf8s_256(<8 x i32> %A, <8 x float> %b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8s_256:
+define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2bf8s256(<8 x i32> %A, <8 x float> %b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8s256:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
 ; X64-NEXT:    vcvtbiasps2bf8s %ymm1, %ymm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x3b,0xd1]
@@ -3149,37 +3514,37 @@ define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2bf8s_256(<8 x i32> %A, <8 x
 ; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8s_256:
+; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8s256:
 ; X86:       # %bb.0:
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
 ; X86-NEXT:    vcvtbiasps2bf8s %ymm1, %ymm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x3b,0xd1]
 ; X86-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
 ; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.256(<8 x i32> %A, <8 x float> %b, <16 x i8> %src0, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s256(<8 x i32> %A, <8 x float> %b, <16 x i8> %src0, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2bf8s_256(<8 x i32> %A, <8 x float> %b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8s_256:
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2bf8s256(<8 x i32> %A, <8 x float> %b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8s256:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
 ; X64-NEXT:    vcvtbiasps2bf8s %ymm1, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xa9,0x3b,0xc1]
 ; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8s_256:
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8s256:
 ; X86:       # %bb.0:
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
 ; X86-NEXT:    vcvtbiasps2bf8s %ymm1, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xa9,0x3b,0xc1]
 ; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.256(<8 x i32> %A, <8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s256(<8 x i32> %A, <8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2bf8s_mem_256(<8 x i32> %A, ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8s_mem_256:
+define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2bf8s256_mem(<8 x i32> %A, ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8s256_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
 ; X64-NEXT:    vcvtbiasps2bf8s (%rdi), %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x3b,0x0f]
@@ -3187,7 +3552,7 @@ define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2bf8s_mem_256(<8 x i32> %A,
 ; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8s_mem_256:
+; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8s256_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
@@ -3196,19 +3561,19 @@ define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2bf8s_mem_256(<8 x i32> %A,
 ; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %b = load <8 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.256(<8 x i32> %A, <8 x float> %b, <16 x i8> %src0, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s256(<8 x i32> %A, <8 x float> %b, <16 x i8> %src0, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2bf8s_bcst_256(<8 x i32> %A, ptr %ptr_b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8s_bcst_256:
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2bf8s256_bcst(<8 x i32> %A, ptr %ptr_b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8s256_bcst:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
 ; X64-NEXT:    vcvtbiasps2bf8s (%rdi){1to8}, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xb9,0x3b,0x07]
 ; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8s_bcst_256:
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8s256_bcst:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
@@ -3218,58 +3583,56 @@ define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2bf8s_bcst_256(<8 x i32> %A
   %ld = load float, ptr %ptr_b
   %ins = insertelement <8 x float> poison, float %ld, i32 0
   %b = shufflevector <8 x float> %ins, <8 x float> poison, <8 x i32> zeroinitializer
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.256(<8 x i32> %A, <8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s256(<8 x i32> %A, <8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
   ret <16 x i8> %ret
 }
 
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.128(<4 x i32>, <4 x float>, <16 x i8>, i8)
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s.256(<8 x i32>, <8 x float>, <16 x i8>, i8)
-
-; Masked 128/256-bit forms of vcvtbiasps2hf8
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s128(<4 x i32>, <4 x float>, <16 x i8>, i8)
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s256(<8 x i32>, <8 x float>, <16 x i8>, i8)
 
-define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2hf8_128(<4 x i32> %A, <4 x float> %b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8_128:
+define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2hf8128(<4 x i32> %A, <4 x float> %b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8128:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
 ; X64-NEXT:    vcvtbiasps2hf8 %xmm1, %xmm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x38,0xd1]
 ; X64-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8_128:
+; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8128:
 ; X86:       # %bb.0:
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
 ; X86-NEXT:    vcvtbiasps2hf8 %xmm1, %xmm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x38,0xd1]
 ; X86-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
 ; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.128(<4 x i32> %A, <4 x float> %b, <16 x i8> %src0, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8128(<4 x i32> %A, <4 x float> %b, <16 x i8> %src0, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2hf8_128(<4 x i32> %A, <4 x float> %b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8_128:
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2hf8128(<4 x i32> %A, <4 x float> %b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8128:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
 ; X64-NEXT:    vcvtbiasps2hf8 %xmm1, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x38,0xc1]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8_128:
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8128:
 ; X86:       # %bb.0:
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
 ; X86-NEXT:    vcvtbiasps2hf8 %xmm1, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x38,0xc1]
 ; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.128(<4 x i32> %A, <4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8128(<4 x i32> %A, <4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2hf8_mem_128(<4 x i32> %A, ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8_mem_128:
+define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2hf8128_mem(<4 x i32> %A, ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8128_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
 ; X64-NEXT:    vcvtbiasps2hf8 (%rdi), %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x38,0x0f]
 ; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8_mem_128:
+; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8128_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
@@ -3277,18 +3640,18 @@ define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2hf8_mem_128(<4 x i32> %A, p
 ; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %b = load <4 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.128(<4 x i32> %A, <4 x float> %b, <16 x i8> %src0, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8128(<4 x i32> %A, <4 x float> %b, <16 x i8> %src0, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2hf8_bcst_128(<4 x i32> %A, ptr %ptr_b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8_bcst_128:
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2hf8128_bcst(<4 x i32> %A, ptr %ptr_b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8128_bcst:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
 ; X64-NEXT:    vcvtbiasps2hf8 (%rdi){1to4}, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x99,0x38,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8_bcst_128:
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8128_bcst:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
@@ -3297,12 +3660,12 @@ define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2hf8_bcst_128(<4 x i32> %A,
   %ld = load float, ptr %ptr_b
   %ins = insertelement <4 x float> poison, float %ld, i32 0
   %b = shufflevector <4 x float> %ins, <4 x float> poison, <4 x i32> zeroinitializer
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.128(<4 x i32> %A, <4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8128(<4 x i32> %A, <4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2hf8_256(<8 x i32> %A, <8 x float> %b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8_256:
+define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2hf8256(<8 x i32> %A, <8 x float> %b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8256:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
 ; X64-NEXT:    vcvtbiasps2hf8 %ymm1, %ymm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x38,0xd1]
@@ -3310,37 +3673,37 @@ define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2hf8_256(<8 x i32> %A, <8 x
 ; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8_256:
+; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8256:
 ; X86:       # %bb.0:
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
 ; X86-NEXT:    vcvtbiasps2hf8 %ymm1, %ymm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x38,0xd1]
 ; X86-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
 ; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.256(<8 x i32> %A, <8 x float> %b, <16 x i8> %src0, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8256(<8 x i32> %A, <8 x float> %b, <16 x i8> %src0, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2hf8_256(<8 x i32> %A, <8 x float> %b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8_256:
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2hf8256(<8 x i32> %A, <8 x float> %b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8256:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
 ; X64-NEXT:    vcvtbiasps2hf8 %ymm1, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xa9,0x38,0xc1]
 ; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8_256:
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8256:
 ; X86:       # %bb.0:
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
 ; X86-NEXT:    vcvtbiasps2hf8 %ymm1, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xa9,0x38,0xc1]
 ; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.256(<8 x i32> %A, <8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8256(<8 x i32> %A, <8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2hf8_mem_256(<8 x i32> %A, ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8_mem_256:
+define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2hf8256_mem(<8 x i32> %A, ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8256_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
 ; X64-NEXT:    vcvtbiasps2hf8 (%rdi), %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x38,0x0f]
@@ -3348,7 +3711,7 @@ define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2hf8_mem_256(<8 x i32> %A, p
 ; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8_mem_256:
+; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8256_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
@@ -3357,19 +3720,19 @@ define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2hf8_mem_256(<8 x i32> %A, p
 ; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %b = load <8 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.256(<8 x i32> %A, <8 x float> %b, <16 x i8> %src0, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8256(<8 x i32> %A, <8 x float> %b, <16 x i8> %src0, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2hf8_bcst_256(<8 x i32> %A, ptr %ptr_b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8_bcst_256:
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2hf8256_bcst(<8 x i32> %A, ptr %ptr_b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8256_bcst:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
 ; X64-NEXT:    vcvtbiasps2hf8 (%rdi){1to8}, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xb9,0x38,0x07]
 ; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8_bcst_256:
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8256_bcst:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
@@ -3379,58 +3742,56 @@ define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2hf8_bcst_256(<8 x i32> %A,
   %ld = load float, ptr %ptr_b
   %ins = insertelement <8 x float> poison, float %ld, i32 0
   %b = shufflevector <8 x float> %ins, <8 x float> poison, <8 x i32> zeroinitializer
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.256(<8 x i32> %A, <8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8256(<8 x i32> %A, <8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
   ret <16 x i8> %ret
 }
 
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.128(<4 x i32>, <4 x float>, <16 x i8>, i8)
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8.256(<8 x i32>, <8 x float>, <16 x i8>, i8)
-
-; Masked 128/256-bit forms of vcvtbiasps2hf8s
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8128(<4 x i32>, <4 x float>, <16 x i8>, i8)
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8256(<8 x i32>, <8 x float>, <16 x i8>, i8)
 
-define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2hf8s_128(<4 x i32> %A, <4 x float> %b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8s_128:
+define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2hf8s128(<4 x i32> %A, <4 x float> %b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8s128:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
 ; X64-NEXT:    vcvtbiasps2hf8s %xmm1, %xmm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x3a,0xd1]
 ; X64-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8s_128:
+; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8s128:
 ; X86:       # %bb.0:
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
 ; X86-NEXT:    vcvtbiasps2hf8s %xmm1, %xmm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x3a,0xd1]
 ; X86-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
 ; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.128(<4 x i32> %A, <4 x float> %b, <16 x i8> %src0, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s128(<4 x i32> %A, <4 x float> %b, <16 x i8> %src0, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2hf8s_128(<4 x i32> %A, <4 x float> %b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8s_128:
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2hf8s128(<4 x i32> %A, <4 x float> %b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8s128:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
 ; X64-NEXT:    vcvtbiasps2hf8s %xmm1, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x3a,0xc1]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8s_128:
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8s128:
 ; X86:       # %bb.0:
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
 ; X86-NEXT:    vcvtbiasps2hf8s %xmm1, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x3a,0xc1]
 ; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.128(<4 x i32> %A, <4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s128(<4 x i32> %A, <4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2hf8s_mem_128(<4 x i32> %A, ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8s_mem_128:
+define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2hf8s128_mem(<4 x i32> %A, ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8s128_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
 ; X64-NEXT:    vcvtbiasps2hf8s (%rdi), %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x3a,0x0f]
 ; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8s_mem_128:
+; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8s128_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
@@ -3438,18 +3799,18 @@ define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2hf8s_mem_128(<4 x i32> %A,
 ; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %b = load <4 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.128(<4 x i32> %A, <4 x float> %b, <16 x i8> %src0, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s128(<4 x i32> %A, <4 x float> %b, <16 x i8> %src0, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2hf8s_bcst_128(<4 x i32> %A, ptr %ptr_b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8s_bcst_128:
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2hf8s128_bcst(<4 x i32> %A, ptr %ptr_b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8s128_bcst:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
 ; X64-NEXT:    vcvtbiasps2hf8s (%rdi){1to4}, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x99,0x3a,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8s_bcst_128:
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8s128_bcst:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
@@ -3458,12 +3819,12 @@ define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2hf8s_bcst_128(<4 x i32> %A
   %ld = load float, ptr %ptr_b
   %ins = insertelement <4 x float> poison, float %ld, i32 0
   %b = shufflevector <4 x float> %ins, <4 x float> poison, <4 x i32> zeroinitializer
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.128(<4 x i32> %A, <4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s128(<4 x i32> %A, <4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2hf8s_256(<8 x i32> %A, <8 x float> %b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8s_256:
+define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2hf8s256(<8 x i32> %A, <8 x float> %b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8s256:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
 ; X64-NEXT:    vcvtbiasps2hf8s %ymm1, %ymm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x3a,0xd1]
@@ -3471,37 +3832,37 @@ define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2hf8s_256(<8 x i32> %A, <8 x
 ; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8s_256:
+; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8s256:
 ; X86:       # %bb.0:
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
 ; X86-NEXT:    vcvtbiasps2hf8s %ymm1, %ymm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x3a,0xd1]
 ; X86-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
 ; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.256(<8 x i32> %A, <8 x float> %b, <16 x i8> %src0, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s256(<8 x i32> %A, <8 x float> %b, <16 x i8> %src0, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2hf8s_256(<8 x i32> %A, <8 x float> %b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8s_256:
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2hf8s256(<8 x i32> %A, <8 x float> %b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8s256:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
 ; X64-NEXT:    vcvtbiasps2hf8s %ymm1, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xa9,0x3a,0xc1]
 ; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8s_256:
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8s256:
 ; X86:       # %bb.0:
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
 ; X86-NEXT:    vcvtbiasps2hf8s %ymm1, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xa9,0x3a,0xc1]
 ; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.256(<8 x i32> %A, <8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s256(<8 x i32> %A, <8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2hf8s_mem_256(<8 x i32> %A, ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8s_mem_256:
+define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2hf8s256_mem(<8 x i32> %A, ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8s256_mem:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
 ; X64-NEXT:    vcvtbiasps2hf8s (%rdi), %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x3a,0x0f]
@@ -3509,7 +3870,7 @@ define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2hf8s_mem_256(<8 x i32> %A,
 ; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8s_mem_256:
+; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8s256_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
@@ -3518,19 +3879,19 @@ define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2hf8s_mem_256(<8 x i32> %A,
 ; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %b = load <8 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.256(<8 x i32> %A, <8 x float> %b, <16 x i8> %src0, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s256(<8 x i32> %A, <8 x float> %b, <16 x i8> %src0, i8 %mask)
   ret <16 x i8> %ret
 }
 
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2hf8s_bcst_256(<8 x i32> %A, ptr %ptr_b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8s_bcst_256:
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2hf8s256_bcst(<8 x i32> %A, ptr %ptr_b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8s256_bcst:
 ; X64:       # %bb.0:
 ; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
 ; X64-NEXT:    vcvtbiasps2hf8s (%rdi){1to8}, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xb9,0x3a,0x07]
 ; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8s_bcst_256:
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8s256_bcst:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
 ; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
@@ -3540,9 +3901,9 @@ define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2hf8s_bcst_256(<8 x i32> %A
   %ld = load float, ptr %ptr_b
   %ins = insertelement <8 x float> poison, float %ld, i32 0
   %b = shufflevector <8 x float> %ins, <8 x float> poison, <8 x i32> zeroinitializer
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.256(<8 x i32> %A, <8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s256(<8 x i32> %A, <8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
   ret <16 x i8> %ret
 }
 
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.128(<4 x i32>, <4 x float>, <16 x i8>, i8)
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s.256(<8 x i32>, <8 x float>, <16 x i8>, i8)
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s128(<4 x i32>, <4 x float>, <16 x i8>, i8)
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s256(<8 x i32>, <8 x float>, <16 x i8>, i8)
diff --git a/llvm/test/MC/Disassembler/X86/avx10_v2_aux-32.txt b/llvm/test/MC/Disassembler/X86/avx10_v2_aux-32.txt
index 6dacc4e33bde00..684a46cf364701 100644
--- a/llvm/test/MC/Disassembler/X86/avx10_v2_aux-32.txt
+++ b/llvm/test/MC/Disassembler/X86/avx10_v2_aux-32.txt
@@ -20,6 +20,18 @@
 # ATT:   vcvtps2bf8 %zmm1, %xmm0 {%k1} {z}
 # INTEL: vcvtps2bf8 xmm0 {k1} {z}, zmm1
 0x62,0xf5,0x7e,0xc9,0x39,0xc1
+
+# ATT:   vcvtps2bf8 (%edi){1to16}, %xmm0
+# INTEL: vcvtps2bf8 xmm0, dword ptr [edi]{1to16}
+0x62,0xf5,0x7e,0x58,0x39,0x07
+
+# ATT:   vcvtps2bf8 (%edi){1to8}, %xmm0
+# INTEL: vcvtps2bf8 xmm0, dword ptr [edi]{1to8}
+0x62,0xf5,0x7e,0x38,0x39,0x07
+
+# ATT:   vcvtps2bf8 (%edi){1to4}, %xmm0
+# INTEL: vcvtps2bf8 xmm0, dword ptr [edi]{1to4}
+0x62,0xf5,0x7e,0x18,0x39,0x07
 # ATT:   vcvtps2bf8s %zmm1, %xmm0
 # INTEL: vcvtps2bf8s xmm0, zmm1
 0x62,0xf5,0x7e,0x48,0x3b,0xc1
@@ -39,6 +51,18 @@
 # ATT:   vcvtps2bf8s %zmm1, %xmm0 {%k1} {z}
 # INTEL: vcvtps2bf8s xmm0 {k1} {z}, zmm1
 0x62,0xf5,0x7e,0xc9,0x3b,0xc1
+
+# ATT:   vcvtps2bf8s (%edi){1to16}, %xmm0
+# INTEL: vcvtps2bf8s xmm0, dword ptr [edi]{1to16}
+0x62,0xf5,0x7e,0x58,0x3b,0x07
+
+# ATT:   vcvtps2bf8s (%edi){1to8}, %xmm0
+# INTEL: vcvtps2bf8s xmm0, dword ptr [edi]{1to8}
+0x62,0xf5,0x7e,0x38,0x3b,0x07
+
+# ATT:   vcvtps2bf8s (%edi){1to4}, %xmm0
+# INTEL: vcvtps2bf8s xmm0, dword ptr [edi]{1to4}
+0x62,0xf5,0x7e,0x18,0x3b,0x07
 # ATT:   vcvtps2hf8 %zmm1, %xmm0
 # INTEL: vcvtps2hf8 xmm0, zmm1
 0x62,0xf5,0x7e,0x48,0x38,0xc1
@@ -58,6 +82,18 @@
 # ATT:   vcvtps2hf8 %zmm1, %xmm0 {%k1} {z}
 # INTEL: vcvtps2hf8 xmm0 {k1} {z}, zmm1
 0x62,0xf5,0x7e,0xc9,0x38,0xc1
+
+# ATT:   vcvtps2hf8 (%edi){1to16}, %xmm0
+# INTEL: vcvtps2hf8 xmm0, dword ptr [edi]{1to16}
+0x62,0xf5,0x7e,0x58,0x38,0x07
+
+# ATT:   vcvtps2hf8 (%edi){1to8}, %xmm0
+# INTEL: vcvtps2hf8 xmm0, dword ptr [edi]{1to8}
+0x62,0xf5,0x7e,0x38,0x38,0x07
+
+# ATT:   vcvtps2hf8 (%edi){1to4}, %xmm0
+# INTEL: vcvtps2hf8 xmm0, dword ptr [edi]{1to4}
+0x62,0xf5,0x7e,0x18,0x38,0x07
 # ATT:   vcvtps2hf8s %zmm1, %xmm0
 # INTEL: vcvtps2hf8s xmm0, zmm1
 0x62,0xf5,0x7e,0x48,0x3a,0xc1
@@ -77,6 +113,18 @@
 # ATT:   vcvtps2hf8s %zmm1, %xmm0 {%k1} {z}
 # INTEL: vcvtps2hf8s xmm0 {k1} {z}, zmm1
 0x62,0xf5,0x7e,0xc9,0x3a,0xc1
+
+# ATT:   vcvtps2hf8s (%edi){1to16}, %xmm0
+# INTEL: vcvtps2hf8s xmm0, dword ptr [edi]{1to16}
+0x62,0xf5,0x7e,0x58,0x3a,0x07
+
+# ATT:   vcvtps2hf8s (%edi){1to8}, %xmm0
+# INTEL: vcvtps2hf8s xmm0, dword ptr [edi]{1to8}
+0x62,0xf5,0x7e,0x38,0x3a,0x07
+
+# ATT:   vcvtps2hf8s (%edi){1to4}, %xmm0
+# INTEL: vcvtps2hf8s xmm0, dword ptr [edi]{1to4}
+0x62,0xf5,0x7e,0x18,0x3a,0x07
 # ATT:   vcvtrops2hf8 %zmm1, %xmm0
 # INTEL: vcvtrops2hf8 xmm0, zmm1
 0x62,0xf5,0x7d,0x48,0x38,0xc1
@@ -96,6 +144,18 @@
 # ATT:   vcvtrops2hf8 %zmm1, %xmm0 {%k1} {z}
 # INTEL: vcvtrops2hf8 xmm0 {k1} {z}, zmm1
 0x62,0xf5,0x7d,0xc9,0x38,0xc1
+
+# ATT:   vcvtrops2hf8 (%edi){1to16}, %xmm0
+# INTEL: vcvtrops2hf8 xmm0, dword ptr [edi]{1to16}
+0x62,0xf5,0x7d,0x58,0x38,0x07
+
+# ATT:   vcvtrops2hf8 (%edi){1to8}, %xmm0
+# INTEL: vcvtrops2hf8 xmm0, dword ptr [edi]{1to8}
+0x62,0xf5,0x7d,0x38,0x38,0x07
+
+# ATT:   vcvtrops2hf8 (%edi){1to4}, %xmm0
+# INTEL: vcvtrops2hf8 xmm0, dword ptr [edi]{1to4}
+0x62,0xf5,0x7d,0x18,0x38,0x07
 # ATT:   vcvtrops2hf8s %zmm1, %xmm0
 # INTEL: vcvtrops2hf8s xmm0, zmm1
 0x62,0xf5,0x7d,0x48,0x3a,0xc1
@@ -116,6 +176,18 @@
 # INTEL: vcvtrops2hf8s xmm0 {k1} {z}, zmm1
 0x62,0xf5,0x7d,0xc9,0x3a,0xc1
 
+# ATT:   vcvtrops2hf8s (%edi){1to16}, %xmm0
+# INTEL: vcvtrops2hf8s xmm0, dword ptr [edi]{1to16}
+0x62,0xf5,0x7d,0x58,0x3a,0x07
+
+# ATT:   vcvtrops2hf8s (%edi){1to8}, %xmm0
+# INTEL: vcvtrops2hf8s xmm0, dword ptr [edi]{1to8}
+0x62,0xf5,0x7d,0x38,0x3a,0x07
+
+# ATT:   vcvtrops2hf8s (%edi){1to4}, %xmm0
+# INTEL: vcvtrops2hf8s xmm0, dword ptr [edi]{1to4}
+0x62,0xf5,0x7d,0x18,0x3a,0x07
+
 # ATT:   vcvtbiasps2bf8 %zmm2, %zmm1, %xmm0
 # INTEL: vcvtbiasps2bf8 xmm0, zmm1, zmm2
 0x62,0xf5,0x74,0x48,0x39,0xc2
@@ -135,6 +207,18 @@
 # ATT:   vcvtbiasps2bf8 %zmm2, %zmm1, %xmm0 {%k1} {z}
 # INTEL: vcvtbiasps2bf8 xmm0 {k1} {z}, zmm1, zmm2
 0x62,0xf5,0x74,0xc9,0x39,0xc2
+
+# ATT:   vcvtbiasps2bf8 (%edi){1to16}, %zmm1, %xmm0
+# INTEL: vcvtbiasps2bf8 xmm0, zmm1, dword ptr [edi]{1to16}
+0x62,0xf5,0x74,0x58,0x39,0x07
+
+# ATT:   vcvtbiasps2bf8 (%edi){1to8}, %ymm1, %xmm0
+# INTEL: vcvtbiasps2bf8 xmm0, ymm1, dword ptr [edi]{1to8}
+0x62,0xf5,0x74,0x38,0x39,0x07
+
+# ATT:   vcvtbiasps2bf8 (%edi){1to4}, %xmm1, %xmm0
+# INTEL: vcvtbiasps2bf8 xmm0, xmm1, dword ptr [edi]{1to4}
+0x62,0xf5,0x74,0x18,0x39,0x07
 # ATT:   vcvtbiasps2bf8s %zmm2, %zmm1, %xmm0
 # INTEL: vcvtbiasps2bf8s xmm0, zmm1, zmm2
 0x62,0xf5,0x74,0x48,0x3b,0xc2
@@ -154,6 +238,18 @@
 # ATT:   vcvtbiasps2bf8s %zmm2, %zmm1, %xmm0 {%k1} {z}
 # INTEL: vcvtbiasps2bf8s xmm0 {k1} {z}, zmm1, zmm2
 0x62,0xf5,0x74,0xc9,0x3b,0xc2
+
+# ATT:   vcvtbiasps2bf8s (%edi){1to16}, %zmm1, %xmm0
+# INTEL: vcvtbiasps2bf8s xmm0, zmm1, dword ptr [edi]{1to16}
+0x62,0xf5,0x74,0x58,0x3b,0x07
+
+# ATT:   vcvtbiasps2bf8s (%edi){1to8}, %ymm1, %xmm0
+# INTEL: vcvtbiasps2bf8s xmm0, ymm1, dword ptr [edi]{1to8}
+0x62,0xf5,0x74,0x38,0x3b,0x07
+
+# ATT:   vcvtbiasps2bf8s (%edi){1to4}, %xmm1, %xmm0
+# INTEL: vcvtbiasps2bf8s xmm0, xmm1, dword ptr [edi]{1to4}
+0x62,0xf5,0x74,0x18,0x3b,0x07
 # ATT:   vcvtbiasps2hf8 %zmm2, %zmm1, %xmm0
 # INTEL: vcvtbiasps2hf8 xmm0, zmm1, zmm2
 0x62,0xf5,0x74,0x48,0x38,0xc2
@@ -173,6 +269,18 @@
 # ATT:   vcvtbiasps2hf8 %zmm2, %zmm1, %xmm0 {%k1} {z}
 # INTEL: vcvtbiasps2hf8 xmm0 {k1} {z}, zmm1, zmm2
 0x62,0xf5,0x74,0xc9,0x38,0xc2
+
+# ATT:   vcvtbiasps2hf8 (%edi){1to16}, %zmm1, %xmm0
+# INTEL: vcvtbiasps2hf8 xmm0, zmm1, dword ptr [edi]{1to16}
+0x62,0xf5,0x74,0x58,0x38,0x07
+
+# ATT:   vcvtbiasps2hf8 (%edi){1to8}, %ymm1, %xmm0
+# INTEL: vcvtbiasps2hf8 xmm0, ymm1, dword ptr [edi]{1to8}
+0x62,0xf5,0x74,0x38,0x38,0x07
+
+# ATT:   vcvtbiasps2hf8 (%edi){1to4}, %xmm1, %xmm0
+# INTEL: vcvtbiasps2hf8 xmm0, xmm1, dword ptr [edi]{1to4}
+0x62,0xf5,0x74,0x18,0x38,0x07
 # ATT:   vcvtbiasps2hf8s %zmm2, %zmm1, %xmm0
 # INTEL: vcvtbiasps2hf8s xmm0, zmm1, zmm2
 0x62,0xf5,0x74,0x48,0x3a,0xc2
@@ -193,6 +301,18 @@
 # INTEL: vcvtbiasps2hf8s xmm0 {k1} {z}, zmm1, zmm2
 0x62,0xf5,0x74,0xc9,0x3a,0xc2
 
+# ATT:   vcvtbiasps2hf8s (%edi){1to16}, %zmm1, %xmm0
+# INTEL: vcvtbiasps2hf8s xmm0, zmm1, dword ptr [edi]{1to16}
+0x62,0xf5,0x74,0x58,0x3a,0x07
+
+# ATT:   vcvtbiasps2hf8s (%edi){1to8}, %ymm1, %xmm0
+# INTEL: vcvtbiasps2hf8s xmm0, ymm1, dword ptr [edi]{1to8}
+0x62,0xf5,0x74,0x38,0x3a,0x07
+
+# ATT:   vcvtbiasps2hf8s (%edi){1to4}, %xmm1, %xmm0
+# INTEL: vcvtbiasps2hf8s xmm0, xmm1, dword ptr [edi]{1to4}
+0x62,0xf5,0x74,0x18,0x3a,0x07
+
 # ATT:   vcvtbf82ps %xmm1, %zmm0
 # INTEL: vcvtbf82ps zmm0, xmm1
 0x62,0xf5,0xfc,0x48,0x36,0xc1
diff --git a/llvm/test/MC/Disassembler/X86/avx10_v2_aux-64.txt b/llvm/test/MC/Disassembler/X86/avx10_v2_aux-64.txt
index 498b10303f855e..58f40b96da4551 100644
--- a/llvm/test/MC/Disassembler/X86/avx10_v2_aux-64.txt
+++ b/llvm/test/MC/Disassembler/X86/avx10_v2_aux-64.txt
@@ -20,6 +20,18 @@
 # ATT:   vcvtps2bf8 %zmm1, %xmm0 {%k1} {z}
 # INTEL: vcvtps2bf8 xmm0 {k1} {z}, zmm1
 0x62,0xf5,0x7e,0xc9,0x39,0xc1
+
+# ATT:   vcvtps2bf8 (%rdi){1to16}, %xmm0
+# INTEL: vcvtps2bf8 xmm0, dword ptr [rdi]{1to16}
+0x62,0xf5,0x7e,0x58,0x39,0x07
+
+# ATT:   vcvtps2bf8 (%rdi){1to8}, %xmm0
+# INTEL: vcvtps2bf8 xmm0, dword ptr [rdi]{1to8}
+0x62,0xf5,0x7e,0x38,0x39,0x07
+
+# ATT:   vcvtps2bf8 (%rdi){1to4}, %xmm0
+# INTEL: vcvtps2bf8 xmm0, dword ptr [rdi]{1to4}
+0x62,0xf5,0x7e,0x18,0x39,0x07
 # ATT:   vcvtps2bf8s %zmm1, %xmm0
 # INTEL: vcvtps2bf8s xmm0, zmm1
 0x62,0xf5,0x7e,0x48,0x3b,0xc1
@@ -39,6 +51,18 @@
 # ATT:   vcvtps2bf8s %zmm1, %xmm0 {%k1} {z}
 # INTEL: vcvtps2bf8s xmm0 {k1} {z}, zmm1
 0x62,0xf5,0x7e,0xc9,0x3b,0xc1
+
+# ATT:   vcvtps2bf8s (%rdi){1to16}, %xmm0
+# INTEL: vcvtps2bf8s xmm0, dword ptr [rdi]{1to16}
+0x62,0xf5,0x7e,0x58,0x3b,0x07
+
+# ATT:   vcvtps2bf8s (%rdi){1to8}, %xmm0
+# INTEL: vcvtps2bf8s xmm0, dword ptr [rdi]{1to8}
+0x62,0xf5,0x7e,0x38,0x3b,0x07
+
+# ATT:   vcvtps2bf8s (%rdi){1to4}, %xmm0
+# INTEL: vcvtps2bf8s xmm0, dword ptr [rdi]{1to4}
+0x62,0xf5,0x7e,0x18,0x3b,0x07
 # ATT:   vcvtps2hf8 %zmm1, %xmm0
 # INTEL: vcvtps2hf8 xmm0, zmm1
 0x62,0xf5,0x7e,0x48,0x38,0xc1
@@ -58,6 +82,18 @@
 # ATT:   vcvtps2hf8 %zmm1, %xmm0 {%k1} {z}
 # INTEL: vcvtps2hf8 xmm0 {k1} {z}, zmm1
 0x62,0xf5,0x7e,0xc9,0x38,0xc1
+
+# ATT:   vcvtps2hf8 (%rdi){1to16}, %xmm0
+# INTEL: vcvtps2hf8 xmm0, dword ptr [rdi]{1to16}
+0x62,0xf5,0x7e,0x58,0x38,0x07
+
+# ATT:   vcvtps2hf8 (%rdi){1to8}, %xmm0
+# INTEL: vcvtps2hf8 xmm0, dword ptr [rdi]{1to8}
+0x62,0xf5,0x7e,0x38,0x38,0x07
+
+# ATT:   vcvtps2hf8 (%rdi){1to4}, %xmm0
+# INTEL: vcvtps2hf8 xmm0, dword ptr [rdi]{1to4}
+0x62,0xf5,0x7e,0x18,0x38,0x07
 # ATT:   vcvtps2hf8s %zmm1, %xmm0
 # INTEL: vcvtps2hf8s xmm0, zmm1
 0x62,0xf5,0x7e,0x48,0x3a,0xc1
@@ -77,6 +113,18 @@
 # ATT:   vcvtps2hf8s %zmm1, %xmm0 {%k1} {z}
 # INTEL: vcvtps2hf8s xmm0 {k1} {z}, zmm1
 0x62,0xf5,0x7e,0xc9,0x3a,0xc1
+
+# ATT:   vcvtps2hf8s (%rdi){1to16}, %xmm0
+# INTEL: vcvtps2hf8s xmm0, dword ptr [rdi]{1to16}
+0x62,0xf5,0x7e,0x58,0x3a,0x07
+
+# ATT:   vcvtps2hf8s (%rdi){1to8}, %xmm0
+# INTEL: vcvtps2hf8s xmm0, dword ptr [rdi]{1to8}
+0x62,0xf5,0x7e,0x38,0x3a,0x07
+
+# ATT:   vcvtps2hf8s (%rdi){1to4}, %xmm0
+# INTEL: vcvtps2hf8s xmm0, dword ptr [rdi]{1to4}
+0x62,0xf5,0x7e,0x18,0x3a,0x07
 # ATT:   vcvtrops2hf8 %zmm1, %xmm0
 # INTEL: vcvtrops2hf8 xmm0, zmm1
 0x62,0xf5,0x7d,0x48,0x38,0xc1
@@ -96,6 +144,18 @@
 # ATT:   vcvtrops2hf8 %zmm1, %xmm0 {%k1} {z}
 # INTEL: vcvtrops2hf8 xmm0 {k1} {z}, zmm1
 0x62,0xf5,0x7d,0xc9,0x38,0xc1
+
+# ATT:   vcvtrops2hf8 (%rdi){1to16}, %xmm0
+# INTEL: vcvtrops2hf8 xmm0, dword ptr [rdi]{1to16}
+0x62,0xf5,0x7d,0x58,0x38,0x07
+
+# ATT:   vcvtrops2hf8 (%rdi){1to8}, %xmm0
+# INTEL: vcvtrops2hf8 xmm0, dword ptr [rdi]{1to8}
+0x62,0xf5,0x7d,0x38,0x38,0x07
+
+# ATT:   vcvtrops2hf8 (%rdi){1to4}, %xmm0
+# INTEL: vcvtrops2hf8 xmm0, dword ptr [rdi]{1to4}
+0x62,0xf5,0x7d,0x18,0x38,0x07
 # ATT:   vcvtrops2hf8s %zmm1, %xmm0
 # INTEL: vcvtrops2hf8s xmm0, zmm1
 0x62,0xf5,0x7d,0x48,0x3a,0xc1
@@ -116,6 +176,18 @@
 # INTEL: vcvtrops2hf8s xmm0 {k1} {z}, zmm1
 0x62,0xf5,0x7d,0xc9,0x3a,0xc1
 
+# ATT:   vcvtrops2hf8s (%rdi){1to16}, %xmm0
+# INTEL: vcvtrops2hf8s xmm0, dword ptr [rdi]{1to16}
+0x62,0xf5,0x7d,0x58,0x3a,0x07
+
+# ATT:   vcvtrops2hf8s (%rdi){1to8}, %xmm0
+# INTEL: vcvtrops2hf8s xmm0, dword ptr [rdi]{1to8}
+0x62,0xf5,0x7d,0x38,0x3a,0x07
+
+# ATT:   vcvtrops2hf8s (%rdi){1to4}, %xmm0
+# INTEL: vcvtrops2hf8s xmm0, dword ptr [rdi]{1to4}
+0x62,0xf5,0x7d,0x18,0x3a,0x07
+
 # ATT:   vcvtbiasps2bf8 %zmm2, %zmm1, %xmm0
 # INTEL: vcvtbiasps2bf8 xmm0, zmm1, zmm2
 0x62,0xf5,0x74,0x48,0x39,0xc2
@@ -135,6 +207,18 @@
 # ATT:   vcvtbiasps2bf8 %zmm2, %zmm1, %xmm0 {%k1} {z}
 # INTEL: vcvtbiasps2bf8 xmm0 {k1} {z}, zmm1, zmm2
 0x62,0xf5,0x74,0xc9,0x39,0xc2
+
+# ATT:   vcvtbiasps2bf8 (%rdi){1to16}, %zmm1, %xmm0
+# INTEL: vcvtbiasps2bf8 xmm0, zmm1, dword ptr [rdi]{1to16}
+0x62,0xf5,0x74,0x58,0x39,0x07
+
+# ATT:   vcvtbiasps2bf8 (%rdi){1to8}, %ymm1, %xmm0
+# INTEL: vcvtbiasps2bf8 xmm0, ymm1, dword ptr [rdi]{1to8}
+0x62,0xf5,0x74,0x38,0x39,0x07
+
+# ATT:   vcvtbiasps2bf8 (%rdi){1to4}, %xmm1, %xmm0
+# INTEL: vcvtbiasps2bf8 xmm0, xmm1, dword ptr [rdi]{1to4}
+0x62,0xf5,0x74,0x18,0x39,0x07
 # ATT:   vcvtbiasps2bf8s %zmm2, %zmm1, %xmm0
 # INTEL: vcvtbiasps2bf8s xmm0, zmm1, zmm2
 0x62,0xf5,0x74,0x48,0x3b,0xc2
@@ -154,6 +238,18 @@
 # ATT:   vcvtbiasps2bf8s %zmm2, %zmm1, %xmm0 {%k1} {z}
 # INTEL: vcvtbiasps2bf8s xmm0 {k1} {z}, zmm1, zmm2
 0x62,0xf5,0x74,0xc9,0x3b,0xc2
+
+# ATT:   vcvtbiasps2bf8s (%rdi){1to16}, %zmm1, %xmm0
+# INTEL: vcvtbiasps2bf8s xmm0, zmm1, dword ptr [rdi]{1to16}
+0x62,0xf5,0x74,0x58,0x3b,0x07
+
+# ATT:   vcvtbiasps2bf8s (%rdi){1to8}, %ymm1, %xmm0
+# INTEL: vcvtbiasps2bf8s xmm0, ymm1, dword ptr [rdi]{1to8}
+0x62,0xf5,0x74,0x38,0x3b,0x07
+
+# ATT:   vcvtbiasps2bf8s (%rdi){1to4}, %xmm1, %xmm0
+# INTEL: vcvtbiasps2bf8s xmm0, xmm1, dword ptr [rdi]{1to4}
+0x62,0xf5,0x74,0x18,0x3b,0x07
 # ATT:   vcvtbiasps2hf8 %zmm2, %zmm1, %xmm0
 # INTEL: vcvtbiasps2hf8 xmm0, zmm1, zmm2
 0x62,0xf5,0x74,0x48,0x38,0xc2
@@ -173,6 +269,18 @@
 # ATT:   vcvtbiasps2hf8 %zmm2, %zmm1, %xmm0 {%k1} {z}
 # INTEL: vcvtbiasps2hf8 xmm0 {k1} {z}, zmm1, zmm2
 0x62,0xf5,0x74,0xc9,0x38,0xc2
+
+# ATT:   vcvtbiasps2hf8 (%rdi){1to16}, %zmm1, %xmm0
+# INTEL: vcvtbiasps2hf8 xmm0, zmm1, dword ptr [rdi]{1to16}
+0x62,0xf5,0x74,0x58,0x38,0x07
+
+# ATT:   vcvtbiasps2hf8 (%rdi){1to8}, %ymm1, %xmm0
+# INTEL: vcvtbiasps2hf8 xmm0, ymm1, dword ptr [rdi]{1to8}
+0x62,0xf5,0x74,0x38,0x38,0x07
+
+# ATT:   vcvtbiasps2hf8 (%rdi){1to4}, %xmm1, %xmm0
+# INTEL: vcvtbiasps2hf8 xmm0, xmm1, dword ptr [rdi]{1to4}
+0x62,0xf5,0x74,0x18,0x38,0x07
 # ATT:   vcvtbiasps2hf8s %zmm2, %zmm1, %xmm0
 # INTEL: vcvtbiasps2hf8s xmm0, zmm1, zmm2
 0x62,0xf5,0x74,0x48,0x3a,0xc2
@@ -193,6 +301,18 @@
 # INTEL: vcvtbiasps2hf8s xmm0 {k1} {z}, zmm1, zmm2
 0x62,0xf5,0x74,0xc9,0x3a,0xc2
 
+# ATT:   vcvtbiasps2hf8s (%rdi){1to16}, %zmm1, %xmm0
+# INTEL: vcvtbiasps2hf8s xmm0, zmm1, dword ptr [rdi]{1to16}
+0x62,0xf5,0x74,0x58,0x3a,0x07
+
+# ATT:   vcvtbiasps2hf8s (%rdi){1to8}, %ymm1, %xmm0
+# INTEL: vcvtbiasps2hf8s xmm0, ymm1, dword ptr [rdi]{1to8}
+0x62,0xf5,0x74,0x38,0x3a,0x07
+
+# ATT:   vcvtbiasps2hf8s (%rdi){1to4}, %xmm1, %xmm0
+# INTEL: vcvtbiasps2hf8s xmm0, xmm1, dword ptr [rdi]{1to4}
+0x62,0xf5,0x74,0x18,0x3a,0x07
+
 # ATT:   vcvtbf82ps %xmm1, %zmm0
 # INTEL: vcvtbf82ps zmm0, xmm1
 0x62,0xf5,0xfc,0x48,0x36,0xc1
diff --git a/llvm/test/MC/X86/avx10_v2_aux-att-32.s b/llvm/test/MC/X86/avx10_v2_aux-att-32.s
index 3ac02c13f2587b..c7f6671eea8f52 100644
--- a/llvm/test/MC/X86/avx10_v2_aux-att-32.s
+++ b/llvm/test/MC/X86/avx10_v2_aux-att-32.s
@@ -1,11 +1,5 @@
 // RUN: llvm-mc -triple i386 --show-encoding -mattr=+avx10v2aux,+avx512vl %s | FileCheck %s
 
-//
-// Group A: PS->8bit truncating conversions
-//
-
-// vcvtps2bf8
-
 // CHECK: vcvtps2bf8 %zmm1, %xmm0
 // CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x39,0xc1]
           vcvtps2bf8 %zmm1, %xmm0
@@ -42,7 +36,13 @@
 // CHECK: encoding: [0x62,0xf5,0x7e,0x58,0x39,0x07]
           vcvtps2bf8 (%edi){1to16}, %xmm0
 
-// vcvtps2bf8s
+// CHECK: vcvtps2bf8 (%edi){1to8}, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x38,0x39,0x07]
+          vcvtps2bf8 (%edi){1to8}, %xmm0
+
+// CHECK: vcvtps2bf8 (%edi){1to4}, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x18,0x39,0x07]
+          vcvtps2bf8 (%edi){1to4}, %xmm0
 
 // CHECK: vcvtps2bf8s %zmm1, %xmm0
 // CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x3b,0xc1]
@@ -80,7 +80,13 @@
 // CHECK: encoding: [0x62,0xf5,0x7e,0x58,0x3b,0x07]
           vcvtps2bf8s (%edi){1to16}, %xmm0
 
-// vcvtps2hf8
+// CHECK: vcvtps2bf8s (%edi){1to8}, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x38,0x3b,0x07]
+          vcvtps2bf8s (%edi){1to8}, %xmm0
+
+// CHECK: vcvtps2bf8s (%edi){1to4}, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x18,0x3b,0x07]
+          vcvtps2bf8s (%edi){1to4}, %xmm0
 
 // CHECK: vcvtps2hf8 %zmm1, %xmm0
 // CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x38,0xc1]
@@ -118,7 +124,13 @@
 // CHECK: encoding: [0x62,0xf5,0x7e,0x58,0x38,0x07]
           vcvtps2hf8 (%edi){1to16}, %xmm0
 
-// vcvtps2hf8s
+// CHECK: vcvtps2hf8 (%edi){1to8}, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x38,0x38,0x07]
+          vcvtps2hf8 (%edi){1to8}, %xmm0
+
+// CHECK: vcvtps2hf8 (%edi){1to4}, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x18,0x38,0x07]
+          vcvtps2hf8 (%edi){1to4}, %xmm0
 
 // CHECK: vcvtps2hf8s %zmm1, %xmm0
 // CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x3a,0xc1]
@@ -156,7 +168,13 @@
 // CHECK: encoding: [0x62,0xf5,0x7e,0x58,0x3a,0x07]
           vcvtps2hf8s (%edi){1to16}, %xmm0
 
-// vcvtrops2hf8
+// CHECK: vcvtps2hf8s (%edi){1to8}, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x38,0x3a,0x07]
+          vcvtps2hf8s (%edi){1to8}, %xmm0
+
+// CHECK: vcvtps2hf8s (%edi){1to4}, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x18,0x3a,0x07]
+          vcvtps2hf8s (%edi){1to4}, %xmm0
 
 // CHECK: vcvtrops2hf8 %zmm1, %xmm0
 // CHECK: encoding: [0x62,0xf5,0x7d,0x48,0x38,0xc1]
@@ -194,7 +212,13 @@
 // CHECK: encoding: [0x62,0xf5,0x7d,0x58,0x38,0x07]
           vcvtrops2hf8 (%edi){1to16}, %xmm0
 
-// vcvtrops2hf8s
+// CHECK: vcvtrops2hf8 (%edi){1to8}, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7d,0x38,0x38,0x07]
+          vcvtrops2hf8 (%edi){1to8}, %xmm0
+
+// CHECK: vcvtrops2hf8 (%edi){1to4}, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7d,0x18,0x38,0x07]
+          vcvtrops2hf8 (%edi){1to4}, %xmm0
 
 // CHECK: vcvtrops2hf8s %zmm1, %xmm0
 // CHECK: encoding: [0x62,0xf5,0x7d,0x48,0x3a,0xc1]
@@ -232,11 +256,13 @@
 // CHECK: encoding: [0x62,0xf5,0x7d,0x58,0x3a,0x07]
           vcvtrops2hf8s (%edi){1to16}, %xmm0
 
-//
-// Group B: Bias PS->8bit conversions (3-operand)
-//
+// CHECK: vcvtrops2hf8s (%edi){1to8}, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7d,0x38,0x3a,0x07]
+          vcvtrops2hf8s (%edi){1to8}, %xmm0
 
-// vcvtbiasps2bf8
+// CHECK: vcvtrops2hf8s (%edi){1to4}, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7d,0x18,0x3a,0x07]
+          vcvtrops2hf8s (%edi){1to4}, %xmm0
 
 // CHECK: vcvtbiasps2bf8 %zmm2, %zmm1, %xmm0
 // CHECK: encoding: [0x62,0xf5,0x74,0x48,0x39,0xc2]
@@ -270,7 +296,17 @@
 // CHECK: encoding: [0x62,0xf5,0x74,0x08,0x39,0x07]
           vcvtbiasps2bf8 (%edi), %xmm1, %xmm0
 
-// vcvtbiasps2bf8s
+// CHECK: vcvtbiasps2bf8 (%edi){1to16}, %zmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x58,0x39,0x07]
+          vcvtbiasps2bf8 (%edi){1to16}, %zmm1, %xmm0
+
+// CHECK: vcvtbiasps2bf8 (%edi){1to8}, %ymm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x38,0x39,0x07]
+          vcvtbiasps2bf8 (%edi){1to8}, %ymm1, %xmm0
+
+// CHECK: vcvtbiasps2bf8 (%edi){1to4}, %xmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x18,0x39,0x07]
+          vcvtbiasps2bf8 (%edi){1to4}, %xmm1, %xmm0
 
 // CHECK: vcvtbiasps2bf8s %zmm2, %zmm1, %xmm0
 // CHECK: encoding: [0x62,0xf5,0x74,0x48,0x3b,0xc2]
@@ -304,7 +340,17 @@
 // CHECK: encoding: [0x62,0xf5,0x74,0x08,0x3b,0x07]
           vcvtbiasps2bf8s (%edi), %xmm1, %xmm0
 
-// vcvtbiasps2hf8
+// CHECK: vcvtbiasps2bf8s (%edi){1to16}, %zmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x58,0x3b,0x07]
+          vcvtbiasps2bf8s (%edi){1to16}, %zmm1, %xmm0
+
+// CHECK: vcvtbiasps2bf8s (%edi){1to8}, %ymm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x38,0x3b,0x07]
+          vcvtbiasps2bf8s (%edi){1to8}, %ymm1, %xmm0
+
+// CHECK: vcvtbiasps2bf8s (%edi){1to4}, %xmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x18,0x3b,0x07]
+          vcvtbiasps2bf8s (%edi){1to4}, %xmm1, %xmm0
 
 // CHECK: vcvtbiasps2hf8 %zmm2, %zmm1, %xmm0
 // CHECK: encoding: [0x62,0xf5,0x74,0x48,0x38,0xc2]
@@ -338,7 +384,17 @@
 // CHECK: encoding: [0x62,0xf5,0x74,0x08,0x38,0x07]
           vcvtbiasps2hf8 (%edi), %xmm1, %xmm0
 
-// vcvtbiasps2hf8s
+// CHECK: vcvtbiasps2hf8 (%edi){1to16}, %zmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x58,0x38,0x07]
+          vcvtbiasps2hf8 (%edi){1to16}, %zmm1, %xmm0
+
+// CHECK: vcvtbiasps2hf8 (%edi){1to8}, %ymm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x38,0x38,0x07]
+          vcvtbiasps2hf8 (%edi){1to8}, %ymm1, %xmm0
+
+// CHECK: vcvtbiasps2hf8 (%edi){1to4}, %xmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x18,0x38,0x07]
+          vcvtbiasps2hf8 (%edi){1to4}, %xmm1, %xmm0
 
 // CHECK: vcvtbiasps2hf8s %zmm2, %zmm1, %xmm0
 // CHECK: encoding: [0x62,0xf5,0x74,0x48,0x3a,0xc2]
@@ -372,11 +428,17 @@
 // CHECK: encoding: [0x62,0xf5,0x74,0x08,0x3a,0x07]
           vcvtbiasps2hf8s (%edi), %xmm1, %xmm0
 
-//
-// Group C: 8bit->PS expanding conversions
-//
+// CHECK: vcvtbiasps2hf8s (%edi){1to16}, %zmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x58,0x3a,0x07]
+          vcvtbiasps2hf8s (%edi){1to16}, %zmm1, %xmm0
+
+// CHECK: vcvtbiasps2hf8s (%edi){1to8}, %ymm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x38,0x3a,0x07]
+          vcvtbiasps2hf8s (%edi){1to8}, %ymm1, %xmm0
 
-// vcvtbf82ps
+// CHECK: vcvtbiasps2hf8s (%edi){1to4}, %xmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x18,0x3a,0x07]
+          vcvtbiasps2hf8s (%edi){1to4}, %xmm1, %xmm0
 
 // CHECK: vcvtbf82ps %xmm1, %zmm0
 // CHECK: encoding: [0x62,0xf5,0xfc,0x48,0x36,0xc1]
@@ -410,8 +472,6 @@
 // CHECK: encoding: [0x62,0xf5,0xfc,0x08,0x36,0x07]
           vcvtbf82ps (%edi), %xmm0
 
-// vcvthf82ps
-
 // CHECK: vcvthf82ps %xmm1, %zmm0
 // CHECK: encoding: [0x62,0xf5,0x7c,0x48,0x36,0xc1]
           vcvthf82ps %xmm1, %zmm0
@@ -444,12 +504,6 @@
 // CHECK: encoding: [0x62,0xf5,0x7c,0x08,0x36,0x07]
           vcvthf82ps (%edi), %xmm0
 
-//
-// Group D: BF8/HF8->BF4S truncations
-//
-
-// vcvtbf82bf4s
-
 // CHECK: vcvtbf82bf4s %zmm1, %ymm0
 // CHECK: encoding: [0x62,0xf5,0xfe,0x48,0x3d,0xc8]
           vcvtbf82bf4s %zmm1, %ymm0
@@ -474,8 +528,6 @@
 // CHECK: encoding: [0x62,0xf5,0xfe,0x08,0x3d,0x0f]
           vcvtbf82bf4s %xmm1, (%edi)
 
-// vcvthf82bf4s
-
 // CHECK: vcvthf82bf4s %zmm1, %ymm0
 // CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x3d,0xc8]
           vcvthf82bf4s %zmm1, %ymm0
@@ -500,12 +552,6 @@
 // CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x3d,0x0f]
           vcvthf82bf4s %xmm1, (%edi)
 
-//
-// Group E: Same-size reg-only conversions (no masking)
-//
-
-// vcvtbf82bf6s
-
 // CHECK: vcvtbf82bf6s %zmm1, %zmm0
 // CHECK: encoding: [0x62,0xf5,0xfe,0x48,0x3e,0xc1]
           vcvtbf82bf6s %zmm1, %zmm0
@@ -518,8 +564,6 @@
 // CHECK: encoding: [0x62,0xf5,0xfe,0x08,0x3e,0xc1]
           vcvtbf82bf6s %xmm1, %xmm0
 
-// vcvthf82hf6s
-
 // CHECK: vcvthf82hf6s %zmm1, %zmm0
 // CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x3c,0xc1]
           vcvthf82hf6s %zmm1, %zmm0
@@ -532,12 +576,6 @@
 // CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x3c,0xc1]
           vcvthf82hf6s %xmm1, %xmm0
 
-//
-// Group F: Expanding/same-size conversions with masking
-//
-
-// vcvtbf42hf8
-
 // CHECK: vcvtbf42hf8 %ymm1, %zmm0
 // CHECK: encoding: [0x62,0xf5,0x7c,0x48,0x37,0xc1]
           vcvtbf42hf8 %ymm1, %zmm0
@@ -570,8 +608,6 @@
 // CHECK: encoding: [0x62,0xf5,0x7c,0x08,0x37,0x07]
           vcvtbf42hf8 (%edi), %xmm0
 
-// vcvtbf62hf8
-
 // CHECK: vcvtbf62hf8 %zmm1, %zmm0
 // CHECK: encoding: [0x62,0xf5,0xfd,0x48,0x37,0xc1]
           vcvtbf62hf8 %zmm1, %zmm0
@@ -592,8 +628,6 @@
 // CHECK: encoding: [0x62,0xf5,0xfd,0xc9,0x37,0xc1]
           vcvtbf62hf8 %zmm1, %zmm0 {%k1} {z}
 
-// vcvthf62hf8
-
 // CHECK: vcvthf62hf8 %zmm1, %zmm0
 // CHECK: encoding: [0x62,0xf5,0x7d,0x48,0x37,0xc1]
           vcvthf62hf8 %zmm1, %zmm0
@@ -614,10 +648,6 @@
 // CHECK: encoding: [0x62,0xf5,0x7d,0xc9,0x37,0xc1]
           vcvthf62hf8 %zmm1, %zmm0 {%k1} {z}
 
-//
-// Group G: VPMOVSSDB - Integer DWord->Byte signed saturation
-//
-
 // CHECK: vpmovssdb %zmm1, %xmm0
 // CHECK: encoding: [0x62,0xf2,0x7e,0x48,0x41,0xc8]
           vpmovssdb %zmm1, %xmm0
@@ -650,10 +680,6 @@
 // CHECK: encoding: [0x62,0xf2,0x7e,0x08,0x41,0x0f]
           vpmovssdb %xmm1, (%edi)
 
-//
-// Group H: VUNPACKB - Byte unpack with immediate
-//
-
 // CHECK: vunpackb $1, %zmm1, %zmm0
 // CHECK: encoding: [0x62,0xf3,0x7c,0x48,0x3d,0xc1,0x01]
           vunpackb $1, %zmm1, %zmm0
diff --git a/llvm/test/MC/X86/avx10_v2_aux-att-64.s b/llvm/test/MC/X86/avx10_v2_aux-att-64.s
index fb84b55a896cf0..4da5db7ddd49e1 100644
--- a/llvm/test/MC/X86/avx10_v2_aux-att-64.s
+++ b/llvm/test/MC/X86/avx10_v2_aux-att-64.s
@@ -1,11 +1,5 @@
 // RUN: llvm-mc -triple x86_64 --show-encoding -mattr=+avx10v2aux,+avx512vl %s | FileCheck %s
 
-//
-// Group A: PS->8bit truncating conversions
-//
-
-// vcvtps2bf8
-
 // CHECK: vcvtps2bf8 %zmm1, %xmm0
 // CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x39,0xc1]
           vcvtps2bf8 %zmm1, %xmm0
@@ -42,7 +36,13 @@
 // CHECK: encoding: [0x62,0xf5,0x7e,0x58,0x39,0x07]
           vcvtps2bf8 (%rdi){1to16}, %xmm0
 
-// vcvtps2bf8s
+// CHECK: vcvtps2bf8 (%rdi){1to8}, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x38,0x39,0x07]
+          vcvtps2bf8 (%rdi){1to8}, %xmm0
+
+// CHECK: vcvtps2bf8 (%rdi){1to4}, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x18,0x39,0x07]
+          vcvtps2bf8 (%rdi){1to4}, %xmm0
 
 // CHECK: vcvtps2bf8s %zmm1, %xmm0
 // CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x3b,0xc1]
@@ -80,7 +80,13 @@
 // CHECK: encoding: [0x62,0xf5,0x7e,0x58,0x3b,0x07]
           vcvtps2bf8s (%rdi){1to16}, %xmm0
 
-// vcvtps2hf8
+// CHECK: vcvtps2bf8s (%rdi){1to8}, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x38,0x3b,0x07]
+          vcvtps2bf8s (%rdi){1to8}, %xmm0
+
+// CHECK: vcvtps2bf8s (%rdi){1to4}, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x18,0x3b,0x07]
+          vcvtps2bf8s (%rdi){1to4}, %xmm0
 
 // CHECK: vcvtps2hf8 %zmm1, %xmm0
 // CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x38,0xc1]
@@ -118,7 +124,13 @@
 // CHECK: encoding: [0x62,0xf5,0x7e,0x58,0x38,0x07]
           vcvtps2hf8 (%rdi){1to16}, %xmm0
 
-// vcvtps2hf8s
+// CHECK: vcvtps2hf8 (%rdi){1to8}, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x38,0x38,0x07]
+          vcvtps2hf8 (%rdi){1to8}, %xmm0
+
+// CHECK: vcvtps2hf8 (%rdi){1to4}, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x18,0x38,0x07]
+          vcvtps2hf8 (%rdi){1to4}, %xmm0
 
 // CHECK: vcvtps2hf8s %zmm1, %xmm0
 // CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x3a,0xc1]
@@ -156,7 +168,13 @@
 // CHECK: encoding: [0x62,0xf5,0x7e,0x58,0x3a,0x07]
           vcvtps2hf8s (%rdi){1to16}, %xmm0
 
-// vcvtrops2hf8
+// CHECK: vcvtps2hf8s (%rdi){1to8}, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x38,0x3a,0x07]
+          vcvtps2hf8s (%rdi){1to8}, %xmm0
+
+// CHECK: vcvtps2hf8s (%rdi){1to4}, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x18,0x3a,0x07]
+          vcvtps2hf8s (%rdi){1to4}, %xmm0
 
 // CHECK: vcvtrops2hf8 %zmm1, %xmm0
 // CHECK: encoding: [0x62,0xf5,0x7d,0x48,0x38,0xc1]
@@ -194,7 +212,13 @@
 // CHECK: encoding: [0x62,0xf5,0x7d,0x58,0x38,0x07]
           vcvtrops2hf8 (%rdi){1to16}, %xmm0
 
-// vcvtrops2hf8s
+// CHECK: vcvtrops2hf8 (%rdi){1to8}, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7d,0x38,0x38,0x07]
+          vcvtrops2hf8 (%rdi){1to8}, %xmm0
+
+// CHECK: vcvtrops2hf8 (%rdi){1to4}, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7d,0x18,0x38,0x07]
+          vcvtrops2hf8 (%rdi){1to4}, %xmm0
 
 // CHECK: vcvtrops2hf8s %zmm1, %xmm0
 // CHECK: encoding: [0x62,0xf5,0x7d,0x48,0x3a,0xc1]
@@ -232,11 +256,13 @@
 // CHECK: encoding: [0x62,0xf5,0x7d,0x58,0x3a,0x07]
           vcvtrops2hf8s (%rdi){1to16}, %xmm0
 
-//
-// Group B: Bias PS->8bit conversions (3-operand)
-//
+// CHECK: vcvtrops2hf8s (%rdi){1to8}, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7d,0x38,0x3a,0x07]
+          vcvtrops2hf8s (%rdi){1to8}, %xmm0
 
-// vcvtbiasps2bf8
+// CHECK: vcvtrops2hf8s (%rdi){1to4}, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7d,0x18,0x3a,0x07]
+          vcvtrops2hf8s (%rdi){1to4}, %xmm0
 
 // CHECK: vcvtbiasps2bf8 %zmm2, %zmm1, %xmm0
 // CHECK: encoding: [0x62,0xf5,0x74,0x48,0x39,0xc2]
@@ -270,7 +296,17 @@
 // CHECK: encoding: [0x62,0xf5,0x74,0xc9,0x39,0xc2]
           vcvtbiasps2bf8 %zmm2, %zmm1, %xmm0 {%k1} {z}
 
-// vcvtbiasps2bf8s
+// CHECK: vcvtbiasps2bf8 (%rdi){1to16}, %zmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x58,0x39,0x07]
+          vcvtbiasps2bf8 (%rdi){1to16}, %zmm1, %xmm0
+
+// CHECK: vcvtbiasps2bf8 (%rdi){1to8}, %ymm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x38,0x39,0x07]
+          vcvtbiasps2bf8 (%rdi){1to8}, %ymm1, %xmm0
+
+// CHECK: vcvtbiasps2bf8 (%rdi){1to4}, %xmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x18,0x39,0x07]
+          vcvtbiasps2bf8 (%rdi){1to4}, %xmm1, %xmm0
 
 // CHECK: vcvtbiasps2bf8s %zmm2, %zmm1, %xmm0
 // CHECK: encoding: [0x62,0xf5,0x74,0x48,0x3b,0xc2]
@@ -304,7 +340,17 @@
 // CHECK: encoding: [0x62,0xf5,0x74,0xc9,0x3b,0xc2]
           vcvtbiasps2bf8s %zmm2, %zmm1, %xmm0 {%k1} {z}
 
-// vcvtbiasps2hf8
+// CHECK: vcvtbiasps2bf8s (%rdi){1to16}, %zmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x58,0x3b,0x07]
+          vcvtbiasps2bf8s (%rdi){1to16}, %zmm1, %xmm0
+
+// CHECK: vcvtbiasps2bf8s (%rdi){1to8}, %ymm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x38,0x3b,0x07]
+          vcvtbiasps2bf8s (%rdi){1to8}, %ymm1, %xmm0
+
+// CHECK: vcvtbiasps2bf8s (%rdi){1to4}, %xmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x18,0x3b,0x07]
+          vcvtbiasps2bf8s (%rdi){1to4}, %xmm1, %xmm0
 
 // CHECK: vcvtbiasps2hf8 %zmm2, %zmm1, %xmm0
 // CHECK: encoding: [0x62,0xf5,0x74,0x48,0x38,0xc2]
@@ -338,7 +384,17 @@
 // CHECK: encoding: [0x62,0xf5,0x74,0xc9,0x38,0xc2]
           vcvtbiasps2hf8 %zmm2, %zmm1, %xmm0 {%k1} {z}
 
-// vcvtbiasps2hf8s
+// CHECK: vcvtbiasps2hf8 (%rdi){1to16}, %zmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x58,0x38,0x07]
+          vcvtbiasps2hf8 (%rdi){1to16}, %zmm1, %xmm0
+
+// CHECK: vcvtbiasps2hf8 (%rdi){1to8}, %ymm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x38,0x38,0x07]
+          vcvtbiasps2hf8 (%rdi){1to8}, %ymm1, %xmm0
+
+// CHECK: vcvtbiasps2hf8 (%rdi){1to4}, %xmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x18,0x38,0x07]
+          vcvtbiasps2hf8 (%rdi){1to4}, %xmm1, %xmm0
 
 // CHECK: vcvtbiasps2hf8s %zmm2, %zmm1, %xmm0
 // CHECK: encoding: [0x62,0xf5,0x74,0x48,0x3a,0xc2]
@@ -372,11 +428,17 @@
 // CHECK: encoding: [0x62,0xf5,0x74,0xc9,0x3a,0xc2]
           vcvtbiasps2hf8s %zmm2, %zmm1, %xmm0 {%k1} {z}
 
-//
-// Group C: 8bit->PS expanding conversions
-//
+// CHECK: vcvtbiasps2hf8s (%rdi){1to16}, %zmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x58,0x3a,0x07]
+          vcvtbiasps2hf8s (%rdi){1to16}, %zmm1, %xmm0
+
+// CHECK: vcvtbiasps2hf8s (%rdi){1to8}, %ymm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x38,0x3a,0x07]
+          vcvtbiasps2hf8s (%rdi){1to8}, %ymm1, %xmm0
 
-// vcvtbf82ps
+// CHECK: vcvtbiasps2hf8s (%rdi){1to4}, %xmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x18,0x3a,0x07]
+          vcvtbiasps2hf8s (%rdi){1to4}, %xmm1, %xmm0
 
 // CHECK: vcvtbf82ps %xmm1, %zmm0
 // CHECK: encoding: [0x62,0xf5,0xfc,0x48,0x36,0xc1]
@@ -410,8 +472,6 @@
 // CHECK: encoding: [0x62,0xf5,0xfc,0xc9,0x36,0xc1]
           vcvtbf82ps %xmm1, %zmm0 {%k1} {z}
 
-// vcvthf82ps
-
 // CHECK: vcvthf82ps %xmm1, %zmm0
 // CHECK: encoding: [0x62,0xf5,0x7c,0x48,0x36,0xc1]
           vcvthf82ps %xmm1, %zmm0
@@ -444,12 +504,6 @@
 // CHECK: encoding: [0x62,0xf5,0x7c,0xc9,0x36,0xc1]
           vcvthf82ps %xmm1, %zmm0 {%k1} {z}
 
-//
-// Group D: BF8/HF8->BF4S store-like truncations
-//
-
-// vcvtbf82bf4s
-
 // CHECK: vcvtbf82bf4s %zmm1, %ymm0
 // CHECK: encoding: [0x62,0xf5,0xfe,0x48,0x3d,0xc8]
           vcvtbf82bf4s %zmm1, %ymm0
@@ -474,8 +528,6 @@
 // CHECK: encoding: [0x62,0xf5,0xfe,0x08,0x3d,0x0f]
           vcvtbf82bf4s %xmm1, (%rdi)
 
-// vcvthf82bf4s
-
 // CHECK: vcvthf82bf4s %zmm1, %ymm0
 // CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x3d,0xc8]
           vcvthf82bf4s %zmm1, %ymm0
@@ -500,12 +552,6 @@
 // CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x3d,0x0f]
           vcvthf82bf4s %xmm1, (%rdi)
 
-//
-// Group E: Same-size reg-only conversions (no masking)
-//
-
-// vcvtbf82bf6s
-
 // CHECK: vcvtbf82bf6s %zmm1, %zmm0
 // CHECK: encoding: [0x62,0xf5,0xfe,0x48,0x3e,0xc1]
           vcvtbf82bf6s %zmm1, %zmm0
@@ -518,8 +564,6 @@
 // CHECK: encoding: [0x62,0xf5,0xfe,0x08,0x3e,0xc1]
           vcvtbf82bf6s %xmm1, %xmm0
 
-// vcvthf82hf6s
-
 // CHECK: vcvthf82hf6s %zmm1, %zmm0
 // CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x3c,0xc1]
           vcvthf82hf6s %zmm1, %zmm0
@@ -532,12 +576,6 @@
 // CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x3c,0xc1]
           vcvthf82hf6s %xmm1, %xmm0
 
-//
-// Group F: Expanding/same-size conversions with masking
-//
-
-// vcvtbf42hf8
-
 // CHECK: vcvtbf42hf8 %ymm1, %zmm0
 // CHECK: encoding: [0x62,0xf5,0x7c,0x48,0x37,0xc1]
           vcvtbf42hf8 %ymm1, %zmm0
@@ -570,8 +608,6 @@
 // CHECK: encoding: [0x62,0xf5,0x7c,0xc9,0x37,0xc1]
           vcvtbf42hf8 %ymm1, %zmm0 {%k1} {z}
 
-// vcvtbf62hf8
-
 // CHECK: vcvtbf62hf8 %zmm1, %zmm0
 // CHECK: encoding: [0x62,0xf5,0xfd,0x48,0x37,0xc1]
           vcvtbf62hf8 %zmm1, %zmm0
@@ -592,8 +628,6 @@
 // CHECK: encoding: [0x62,0xf5,0xfd,0xc9,0x37,0xc1]
           vcvtbf62hf8 %zmm1, %zmm0 {%k1} {z}
 
-// vcvthf62hf8
-
 // CHECK: vcvthf62hf8 %zmm1, %zmm0
 // CHECK: encoding: [0x62,0xf5,0x7d,0x48,0x37,0xc1]
           vcvthf62hf8 %zmm1, %zmm0
@@ -614,10 +648,6 @@
 // CHECK: encoding: [0x62,0xf5,0x7d,0xc9,0x37,0xc1]
           vcvthf62hf8 %zmm1, %zmm0 {%k1} {z}
 
-//
-// Group G: VPMOVSSDB - Integer DWord->Byte signed saturation
-//
-
 // CHECK: vpmovssdb %zmm1, %xmm0
 // CHECK: encoding: [0x62,0xf2,0x7e,0x48,0x41,0xc8]
           vpmovssdb %zmm1, %xmm0
@@ -650,10 +680,6 @@
 // CHECK: encoding: [0x62,0xf2,0x7e,0xc9,0x41,0xc8]
           vpmovssdb %zmm1, %xmm0 {%k1} {z}
 
-//
-// Group H: VUNPACKB - Byte unpack with immediate
-//
-
 // CHECK: vunpackb $1, %zmm1, %zmm0
 // CHECK: encoding: [0x62,0xf3,0x7c,0x48,0x3d,0xc1,0x01]
           vunpackb $1, %zmm1, %zmm0
diff --git a/llvm/test/MC/X86/avx10_v2_aux-intel-32.s b/llvm/test/MC/X86/avx10_v2_aux-intel-32.s
index b415359ac87818..54893e196d1c91 100644
--- a/llvm/test/MC/X86/avx10_v2_aux-intel-32.s
+++ b/llvm/test/MC/X86/avx10_v2_aux-intel-32.s
@@ -1,11 +1,5 @@
 // RUN: llvm-mc -triple i386 -x86-asm-syntax=intel -output-asm-variant=1 --show-encoding -mattr=+avx10v2aux,+avx512vl %s | FileCheck %s
 
-//
-// Group A: PS->8bit truncating conversions
-//
-
-// vcvtps2bf8
-
 // CHECK: vcvtps2bf8 xmm0, zmm1
 // CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x39,0xc1]
           vcvtps2bf8 xmm0, zmm1
@@ -42,7 +36,13 @@
 // CHECK: encoding: [0x62,0xf5,0x7e,0x58,0x39,0x07]
           vcvtps2bf8 xmm0, dword ptr [edi]{1to16}
 
-// vcvtps2bf8s
+// CHECK: vcvtps2bf8 xmm0, dword ptr [edi]{1to8}
+// CHECK: encoding: [0x62,0xf5,0x7e,0x38,0x39,0x07]
+          vcvtps2bf8 xmm0, dword ptr [edi]{1to8}
+
+// CHECK: vcvtps2bf8 xmm0, dword ptr [edi]{1to4}
+// CHECK: encoding: [0x62,0xf5,0x7e,0x18,0x39,0x07]
+          vcvtps2bf8 xmm0, dword ptr [edi]{1to4}
 
 // CHECK: vcvtps2bf8s xmm0, zmm1
 // CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x3b,0xc1]
@@ -80,7 +80,13 @@
 // CHECK: encoding: [0x62,0xf5,0x7e,0x58,0x3b,0x07]
           vcvtps2bf8s xmm0, dword ptr [edi]{1to16}
 
-// vcvtps2hf8
+// CHECK: vcvtps2bf8s xmm0, dword ptr [edi]{1to8}
+// CHECK: encoding: [0x62,0xf5,0x7e,0x38,0x3b,0x07]
+          vcvtps2bf8s xmm0, dword ptr [edi]{1to8}
+
+// CHECK: vcvtps2bf8s xmm0, dword ptr [edi]{1to4}
+// CHECK: encoding: [0x62,0xf5,0x7e,0x18,0x3b,0x07]
+          vcvtps2bf8s xmm0, dword ptr [edi]{1to4}
 
 // CHECK: vcvtps2hf8 xmm0, zmm1
 // CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x38,0xc1]
@@ -118,7 +124,13 @@
 // CHECK: encoding: [0x62,0xf5,0x7e,0x58,0x38,0x07]
           vcvtps2hf8 xmm0, dword ptr [edi]{1to16}
 
-// vcvtps2hf8s
+// CHECK: vcvtps2hf8 xmm0, dword ptr [edi]{1to8}
+// CHECK: encoding: [0x62,0xf5,0x7e,0x38,0x38,0x07]
+          vcvtps2hf8 xmm0, dword ptr [edi]{1to8}
+
+// CHECK: vcvtps2hf8 xmm0, dword ptr [edi]{1to4}
+// CHECK: encoding: [0x62,0xf5,0x7e,0x18,0x38,0x07]
+          vcvtps2hf8 xmm0, dword ptr [edi]{1to4}
 
 // CHECK: vcvtps2hf8s xmm0, zmm1
 // CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x3a,0xc1]
@@ -156,7 +168,13 @@
 // CHECK: encoding: [0x62,0xf5,0x7e,0x58,0x3a,0x07]
           vcvtps2hf8s xmm0, dword ptr [edi]{1to16}
 
-// vcvtrops2hf8
+// CHECK: vcvtps2hf8s xmm0, dword ptr [edi]{1to8}
+// CHECK: encoding: [0x62,0xf5,0x7e,0x38,0x3a,0x07]
+          vcvtps2hf8s xmm0, dword ptr [edi]{1to8}
+
+// CHECK: vcvtps2hf8s xmm0, dword ptr [edi]{1to4}
+// CHECK: encoding: [0x62,0xf5,0x7e,0x18,0x3a,0x07]
+          vcvtps2hf8s xmm0, dword ptr [edi]{1to4}
 
 // CHECK: vcvtrops2hf8 xmm0, zmm1
 // CHECK: encoding: [0x62,0xf5,0x7d,0x48,0x38,0xc1]
@@ -194,7 +212,13 @@
 // CHECK: encoding: [0x62,0xf5,0x7d,0x58,0x38,0x07]
           vcvtrops2hf8 xmm0, dword ptr [edi]{1to16}
 
-// vcvtrops2hf8s
+// CHECK: vcvtrops2hf8 xmm0, dword ptr [edi]{1to8}
+// CHECK: encoding: [0x62,0xf5,0x7d,0x38,0x38,0x07]
+          vcvtrops2hf8 xmm0, dword ptr [edi]{1to8}
+
+// CHECK: vcvtrops2hf8 xmm0, dword ptr [edi]{1to4}
+// CHECK: encoding: [0x62,0xf5,0x7d,0x18,0x38,0x07]
+          vcvtrops2hf8 xmm0, dword ptr [edi]{1to4}
 
 // CHECK: vcvtrops2hf8s xmm0, zmm1
 // CHECK: encoding: [0x62,0xf5,0x7d,0x48,0x3a,0xc1]
@@ -232,11 +256,13 @@
 // CHECK: encoding: [0x62,0xf5,0x7d,0x58,0x3a,0x07]
           vcvtrops2hf8s xmm0, dword ptr [edi]{1to16}
 
-//
-// Group B: Bias PS->8bit conversions (3-operand)
-//
+// CHECK: vcvtrops2hf8s xmm0, dword ptr [edi]{1to8}
+// CHECK: encoding: [0x62,0xf5,0x7d,0x38,0x3a,0x07]
+          vcvtrops2hf8s xmm0, dword ptr [edi]{1to8}
 
-// vcvtbiasps2bf8
+// CHECK: vcvtrops2hf8s xmm0, dword ptr [edi]{1to4}
+// CHECK: encoding: [0x62,0xf5,0x7d,0x18,0x3a,0x07]
+          vcvtrops2hf8s xmm0, dword ptr [edi]{1to4}
 
 // CHECK: vcvtbiasps2bf8 xmm0, zmm1, zmm2
 // CHECK: encoding: [0x62,0xf5,0x74,0x48,0x39,0xc2]
@@ -270,7 +296,17 @@
 // CHECK: encoding: [0x62,0xf5,0x74,0x08,0x39,0x07]
           vcvtbiasps2bf8 xmm0, xmm1, xmmword ptr [edi]
 
-// vcvtbiasps2bf8s
+// CHECK: vcvtbiasps2bf8 xmm0, zmm1, dword ptr [edi]{1to16}
+// CHECK: encoding: [0x62,0xf5,0x74,0x58,0x39,0x07]
+          vcvtbiasps2bf8 xmm0, zmm1, dword ptr [edi]{1to16}
+
+// CHECK: vcvtbiasps2bf8 xmm0, ymm1, dword ptr [edi]{1to8}
+// CHECK: encoding: [0x62,0xf5,0x74,0x38,0x39,0x07]
+          vcvtbiasps2bf8 xmm0, ymm1, dword ptr [edi]{1to8}
+
+// CHECK: vcvtbiasps2bf8 xmm0, xmm1, dword ptr [edi]{1to4}
+// CHECK: encoding: [0x62,0xf5,0x74,0x18,0x39,0x07]
+          vcvtbiasps2bf8 xmm0, xmm1, dword ptr [edi]{1to4}
 
 // CHECK: vcvtbiasps2bf8s xmm0, zmm1, zmm2
 // CHECK: encoding: [0x62,0xf5,0x74,0x48,0x3b,0xc2]
@@ -304,7 +340,17 @@
 // CHECK: encoding: [0x62,0xf5,0x74,0x08,0x3b,0x07]
           vcvtbiasps2bf8s xmm0, xmm1, xmmword ptr [edi]
 
-// vcvtbiasps2hf8
+// CHECK: vcvtbiasps2bf8s xmm0, zmm1, dword ptr [edi]{1to16}
+// CHECK: encoding: [0x62,0xf5,0x74,0x58,0x3b,0x07]
+          vcvtbiasps2bf8s xmm0, zmm1, dword ptr [edi]{1to16}
+
+// CHECK: vcvtbiasps2bf8s xmm0, ymm1, dword ptr [edi]{1to8}
+// CHECK: encoding: [0x62,0xf5,0x74,0x38,0x3b,0x07]
+          vcvtbiasps2bf8s xmm0, ymm1, dword ptr [edi]{1to8}
+
+// CHECK: vcvtbiasps2bf8s xmm0, xmm1, dword ptr [edi]{1to4}
+// CHECK: encoding: [0x62,0xf5,0x74,0x18,0x3b,0x07]
+          vcvtbiasps2bf8s xmm0, xmm1, dword ptr [edi]{1to4}
 
 // CHECK: vcvtbiasps2hf8 xmm0, zmm1, zmm2
 // CHECK: encoding: [0x62,0xf5,0x74,0x48,0x38,0xc2]
@@ -338,7 +384,17 @@
 // CHECK: encoding: [0x62,0xf5,0x74,0x08,0x38,0x07]
           vcvtbiasps2hf8 xmm0, xmm1, xmmword ptr [edi]
 
-// vcvtbiasps2hf8s
+// CHECK: vcvtbiasps2hf8 xmm0, zmm1, dword ptr [edi]{1to16}
+// CHECK: encoding: [0x62,0xf5,0x74,0x58,0x38,0x07]
+          vcvtbiasps2hf8 xmm0, zmm1, dword ptr [edi]{1to16}
+
+// CHECK: vcvtbiasps2hf8 xmm0, ymm1, dword ptr [edi]{1to8}
+// CHECK: encoding: [0x62,0xf5,0x74,0x38,0x38,0x07]
+          vcvtbiasps2hf8 xmm0, ymm1, dword ptr [edi]{1to8}
+
+// CHECK: vcvtbiasps2hf8 xmm0, xmm1, dword ptr [edi]{1to4}
+// CHECK: encoding: [0x62,0xf5,0x74,0x18,0x38,0x07]
+          vcvtbiasps2hf8 xmm0, xmm1, dword ptr [edi]{1to4}
 
 // CHECK: vcvtbiasps2hf8s xmm0, zmm1, zmm2
 // CHECK: encoding: [0x62,0xf5,0x74,0x48,0x3a,0xc2]
@@ -372,11 +428,17 @@
 // CHECK: encoding: [0x62,0xf5,0x74,0x08,0x3a,0x07]
           vcvtbiasps2hf8s xmm0, xmm1, xmmword ptr [edi]
 
-//
-// Group C: 8bit->PS expanding conversions
-//
+// CHECK: vcvtbiasps2hf8s xmm0, zmm1, dword ptr [edi]{1to16}
+// CHECK: encoding: [0x62,0xf5,0x74,0x58,0x3a,0x07]
+          vcvtbiasps2hf8s xmm0, zmm1, dword ptr [edi]{1to16}
+
+// CHECK: vcvtbiasps2hf8s xmm0, ymm1, dword ptr [edi]{1to8}
+// CHECK: encoding: [0x62,0xf5,0x74,0x38,0x3a,0x07]
+          vcvtbiasps2hf8s xmm0, ymm1, dword ptr [edi]{1to8}
 
-// vcvtbf82ps
+// CHECK: vcvtbiasps2hf8s xmm0, xmm1, dword ptr [edi]{1to4}
+// CHECK: encoding: [0x62,0xf5,0x74,0x18,0x3a,0x07]
+          vcvtbiasps2hf8s xmm0, xmm1, dword ptr [edi]{1to4}
 
 // CHECK: vcvtbf82ps zmm0, xmm1
 // CHECK: encoding: [0x62,0xf5,0xfc,0x48,0x36,0xc1]
@@ -410,8 +472,6 @@
 // CHECK: encoding: [0x62,0xf5,0xfc,0x08,0x36,0x07]
           vcvtbf82ps xmm0, dword ptr [edi]
 
-// vcvthf82ps
-
 // CHECK: vcvthf82ps zmm0, xmm1
 // CHECK: encoding: [0x62,0xf5,0x7c,0x48,0x36,0xc1]
           vcvthf82ps zmm0, xmm1
@@ -444,12 +504,6 @@
 // CHECK: encoding: [0x62,0xf5,0x7c,0x08,0x36,0x07]
           vcvthf82ps xmm0, dword ptr [edi]
 
-//
-// Group D: BF8/HF8->BF4S truncations
-//
-
-// vcvtbf82bf4s
-
 // CHECK: vcvtbf82bf4s ymm0, zmm1
 // CHECK: encoding: [0x62,0xf5,0xfe,0x48,0x3d,0xc8]
           vcvtbf82bf4s ymm0, zmm1
@@ -474,8 +528,6 @@
 // CHECK: encoding: [0x62,0xf5,0xfe,0x08,0x3d,0x0f]
           vcvtbf82bf4s qword ptr [edi], xmm1
 
-// vcvthf82bf4s
-
 // CHECK: vcvthf82bf4s ymm0, zmm1
 // CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x3d,0xc8]
           vcvthf82bf4s ymm0, zmm1
@@ -500,12 +552,6 @@
 // CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x3d,0x0f]
           vcvthf82bf4s qword ptr [edi], xmm1
 
-//
-// Group E: Same-size reg-only conversions (no masking)
-//
-
-// vcvtbf82bf6s
-
 // CHECK: vcvtbf82bf6s zmm0, zmm1
 // CHECK: encoding: [0x62,0xf5,0xfe,0x48,0x3e,0xc1]
           vcvtbf82bf6s zmm0, zmm1
@@ -518,8 +564,6 @@
 // CHECK: encoding: [0x62,0xf5,0xfe,0x08,0x3e,0xc1]
           vcvtbf82bf6s xmm0, xmm1
 
-// vcvthf82hf6s
-
 // CHECK: vcvthf82hf6s zmm0, zmm1
 // CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x3c,0xc1]
           vcvthf82hf6s zmm0, zmm1
@@ -532,12 +576,6 @@
 // CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x3c,0xc1]
           vcvthf82hf6s xmm0, xmm1
 
-//
-// Group F: Expanding/same-size conversions with masking
-//
-
-// vcvtbf42hf8
-
 // CHECK: vcvtbf42hf8 zmm0, ymm1
 // CHECK: encoding: [0x62,0xf5,0x7c,0x48,0x37,0xc1]
           vcvtbf42hf8 zmm0, ymm1
@@ -570,8 +608,6 @@
 // CHECK: encoding: [0x62,0xf5,0x7c,0x08,0x37,0x07]
           vcvtbf42hf8 xmm0, qword ptr [edi]
 
-// vcvtbf62hf8
-
 // CHECK: vcvtbf62hf8 zmm0, zmm1
 // CHECK: encoding: [0x62,0xf5,0xfd,0x48,0x37,0xc1]
           vcvtbf62hf8 zmm0, zmm1
@@ -592,8 +628,6 @@
 // CHECK: encoding: [0x62,0xf5,0xfd,0xc9,0x37,0xc1]
           vcvtbf62hf8 zmm0 {k1} {z}, zmm1
 
-// vcvthf62hf8
-
 // CHECK: vcvthf62hf8 zmm0, zmm1
 // CHECK: encoding: [0x62,0xf5,0x7d,0x48,0x37,0xc1]
           vcvthf62hf8 zmm0, zmm1
@@ -614,10 +648,6 @@
 // CHECK: encoding: [0x62,0xf5,0x7d,0xc9,0x37,0xc1]
           vcvthf62hf8 zmm0 {k1} {z}, zmm1
 
-//
-// Group G: VPMOVSSDB - Integer DWord->Byte signed saturation
-//
-
 // CHECK: vpmovssdb xmm0, zmm1
 // CHECK: encoding: [0x62,0xf2,0x7e,0x48,0x41,0xc8]
           vpmovssdb xmm0, zmm1
@@ -650,10 +680,6 @@
 // CHECK: encoding: [0x62,0xf2,0x7e,0x08,0x41,0x0f]
           vpmovssdb dword ptr [edi], xmm1
 
-//
-// Group H: VUNPACKB - Byte unpack with immediate
-//
-
 // CHECK: vunpackb zmm0, zmm1, 1
 // CHECK: encoding: [0x62,0xf3,0x7c,0x48,0x3d,0xc1,0x01]
           vunpackb zmm0, zmm1, 1
diff --git a/llvm/test/MC/X86/avx10_v2_aux-intel-64.s b/llvm/test/MC/X86/avx10_v2_aux-intel-64.s
index b5bc7360ffffa5..204a374fd3b8bb 100644
--- a/llvm/test/MC/X86/avx10_v2_aux-intel-64.s
+++ b/llvm/test/MC/X86/avx10_v2_aux-intel-64.s
@@ -1,11 +1,5 @@
 // RUN: llvm-mc -triple x86_64 -x86-asm-syntax=intel -output-asm-variant=1 --show-encoding -mattr=+avx10v2aux,+avx512vl %s | FileCheck %s
 
-//
-// Group A: PS->8bit truncating conversions
-//
-
-// vcvtps2bf8
-
 // CHECK: vcvtps2bf8 xmm0, zmm1
 // CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x39,0xc1]
           vcvtps2bf8 xmm0, zmm1
@@ -42,7 +36,13 @@
 // CHECK: encoding: [0x62,0xf5,0x7e,0x58,0x39,0x07]
           vcvtps2bf8 xmm0, dword ptr [rdi]{1to16}
 
-// vcvtps2bf8s
+// CHECK: vcvtps2bf8 xmm0, dword ptr [rdi]{1to8}
+// CHECK: encoding: [0x62,0xf5,0x7e,0x38,0x39,0x07]
+          vcvtps2bf8 xmm0, dword ptr [rdi]{1to8}
+
+// CHECK: vcvtps2bf8 xmm0, dword ptr [rdi]{1to4}
+// CHECK: encoding: [0x62,0xf5,0x7e,0x18,0x39,0x07]
+          vcvtps2bf8 xmm0, dword ptr [rdi]{1to4}
 
 // CHECK: vcvtps2bf8s xmm0, zmm1
 // CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x3b,0xc1]
@@ -80,7 +80,13 @@
 // CHECK: encoding: [0x62,0xf5,0x7e,0x58,0x3b,0x07]
           vcvtps2bf8s xmm0, dword ptr [rdi]{1to16}
 
-// vcvtps2hf8
+// CHECK: vcvtps2bf8s xmm0, dword ptr [rdi]{1to8}
+// CHECK: encoding: [0x62,0xf5,0x7e,0x38,0x3b,0x07]
+          vcvtps2bf8s xmm0, dword ptr [rdi]{1to8}
+
+// CHECK: vcvtps2bf8s xmm0, dword ptr [rdi]{1to4}
+// CHECK: encoding: [0x62,0xf5,0x7e,0x18,0x3b,0x07]
+          vcvtps2bf8s xmm0, dword ptr [rdi]{1to4}
 
 // CHECK: vcvtps2hf8 xmm0, zmm1
 // CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x38,0xc1]
@@ -118,7 +124,13 @@
 // CHECK: encoding: [0x62,0xf5,0x7e,0x58,0x38,0x07]
           vcvtps2hf8 xmm0, dword ptr [rdi]{1to16}
 
-// vcvtps2hf8s
+// CHECK: vcvtps2hf8 xmm0, dword ptr [rdi]{1to8}
+// CHECK: encoding: [0x62,0xf5,0x7e,0x38,0x38,0x07]
+          vcvtps2hf8 xmm0, dword ptr [rdi]{1to8}
+
+// CHECK: vcvtps2hf8 xmm0, dword ptr [rdi]{1to4}
+// CHECK: encoding: [0x62,0xf5,0x7e,0x18,0x38,0x07]
+          vcvtps2hf8 xmm0, dword ptr [rdi]{1to4}
 
 // CHECK: vcvtps2hf8s xmm0, zmm1
 // CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x3a,0xc1]
@@ -156,7 +168,13 @@
 // CHECK: encoding: [0x62,0xf5,0x7e,0x58,0x3a,0x07]
           vcvtps2hf8s xmm0, dword ptr [rdi]{1to16}
 
-// vcvtrops2hf8
+// CHECK: vcvtps2hf8s xmm0, dword ptr [rdi]{1to8}
+// CHECK: encoding: [0x62,0xf5,0x7e,0x38,0x3a,0x07]
+          vcvtps2hf8s xmm0, dword ptr [rdi]{1to8}
+
+// CHECK: vcvtps2hf8s xmm0, dword ptr [rdi]{1to4}
+// CHECK: encoding: [0x62,0xf5,0x7e,0x18,0x3a,0x07]
+          vcvtps2hf8s xmm0, dword ptr [rdi]{1to4}
 
 // CHECK: vcvtrops2hf8 xmm0, zmm1
 // CHECK: encoding: [0x62,0xf5,0x7d,0x48,0x38,0xc1]
@@ -194,7 +212,13 @@
 // CHECK: encoding: [0x62,0xf5,0x7d,0x58,0x38,0x07]
           vcvtrops2hf8 xmm0, dword ptr [rdi]{1to16}
 
-// vcvtrops2hf8s
+// CHECK: vcvtrops2hf8 xmm0, dword ptr [rdi]{1to8}
+// CHECK: encoding: [0x62,0xf5,0x7d,0x38,0x38,0x07]
+          vcvtrops2hf8 xmm0, dword ptr [rdi]{1to8}
+
+// CHECK: vcvtrops2hf8 xmm0, dword ptr [rdi]{1to4}
+// CHECK: encoding: [0x62,0xf5,0x7d,0x18,0x38,0x07]
+          vcvtrops2hf8 xmm0, dword ptr [rdi]{1to4}
 
 // CHECK: vcvtrops2hf8s xmm0, zmm1
 // CHECK: encoding: [0x62,0xf5,0x7d,0x48,0x3a,0xc1]
@@ -232,11 +256,13 @@
 // CHECK: encoding: [0x62,0xf5,0x7d,0x58,0x3a,0x07]
           vcvtrops2hf8s xmm0, dword ptr [rdi]{1to16}
 
-//
-// Group B: Bias PS->8bit conversions (3-operand)
-//
+// CHECK: vcvtrops2hf8s xmm0, dword ptr [rdi]{1to8}
+// CHECK: encoding: [0x62,0xf5,0x7d,0x38,0x3a,0x07]
+          vcvtrops2hf8s xmm0, dword ptr [rdi]{1to8}
 
-// vcvtbiasps2bf8
+// CHECK: vcvtrops2hf8s xmm0, dword ptr [rdi]{1to4}
+// CHECK: encoding: [0x62,0xf5,0x7d,0x18,0x3a,0x07]
+          vcvtrops2hf8s xmm0, dword ptr [rdi]{1to4}
 
 // CHECK: vcvtbiasps2bf8 xmm0, zmm1, zmm2
 // CHECK: encoding: [0x62,0xf5,0x74,0x48,0x39,0xc2]
@@ -270,7 +296,17 @@
 // CHECK: encoding: [0x62,0xf5,0x74,0xc9,0x39,0xc2]
           vcvtbiasps2bf8 xmm0 {k1} {z}, zmm1, zmm2
 
-// vcvtbiasps2bf8s
+// CHECK: vcvtbiasps2bf8 xmm0, zmm1, dword ptr [rdi]{1to16}
+// CHECK: encoding: [0x62,0xf5,0x74,0x58,0x39,0x07]
+          vcvtbiasps2bf8 xmm0, zmm1, dword ptr [rdi]{1to16}
+
+// CHECK: vcvtbiasps2bf8 xmm0, ymm1, dword ptr [rdi]{1to8}
+// CHECK: encoding: [0x62,0xf5,0x74,0x38,0x39,0x07]
+          vcvtbiasps2bf8 xmm0, ymm1, dword ptr [rdi]{1to8}
+
+// CHECK: vcvtbiasps2bf8 xmm0, xmm1, dword ptr [rdi]{1to4}
+// CHECK: encoding: [0x62,0xf5,0x74,0x18,0x39,0x07]
+          vcvtbiasps2bf8 xmm0, xmm1, dword ptr [rdi]{1to4}
 
 // CHECK: vcvtbiasps2bf8s xmm0, zmm1, zmm2
 // CHECK: encoding: [0x62,0xf5,0x74,0x48,0x3b,0xc2]
@@ -304,7 +340,17 @@
 // CHECK: encoding: [0x62,0xf5,0x74,0xc9,0x3b,0xc2]
           vcvtbiasps2bf8s xmm0 {k1} {z}, zmm1, zmm2
 
-// vcvtbiasps2hf8
+// CHECK: vcvtbiasps2bf8s xmm0, zmm1, dword ptr [rdi]{1to16}
+// CHECK: encoding: [0x62,0xf5,0x74,0x58,0x3b,0x07]
+          vcvtbiasps2bf8s xmm0, zmm1, dword ptr [rdi]{1to16}
+
+// CHECK: vcvtbiasps2bf8s xmm0, ymm1, dword ptr [rdi]{1to8}
+// CHECK: encoding: [0x62,0xf5,0x74,0x38,0x3b,0x07]
+          vcvtbiasps2bf8s xmm0, ymm1, dword ptr [rdi]{1to8}
+
+// CHECK: vcvtbiasps2bf8s xmm0, xmm1, dword ptr [rdi]{1to4}
+// CHECK: encoding: [0x62,0xf5,0x74,0x18,0x3b,0x07]
+          vcvtbiasps2bf8s xmm0, xmm1, dword ptr [rdi]{1to4}
 
 // CHECK: vcvtbiasps2hf8 xmm0, zmm1, zmm2
 // CHECK: encoding: [0x62,0xf5,0x74,0x48,0x38,0xc2]
@@ -338,7 +384,17 @@
 // CHECK: encoding: [0x62,0xf5,0x74,0xc9,0x38,0xc2]
           vcvtbiasps2hf8 xmm0 {k1} {z}, zmm1, zmm2
 
-// vcvtbiasps2hf8s
+// CHECK: vcvtbiasps2hf8 xmm0, zmm1, dword ptr [rdi]{1to16}
+// CHECK: encoding: [0x62,0xf5,0x74,0x58,0x38,0x07]
+          vcvtbiasps2hf8 xmm0, zmm1, dword ptr [rdi]{1to16}
+
+// CHECK: vcvtbiasps2hf8 xmm0, ymm1, dword ptr [rdi]{1to8}
+// CHECK: encoding: [0x62,0xf5,0x74,0x38,0x38,0x07]
+          vcvtbiasps2hf8 xmm0, ymm1, dword ptr [rdi]{1to8}
+
+// CHECK: vcvtbiasps2hf8 xmm0, xmm1, dword ptr [rdi]{1to4}
+// CHECK: encoding: [0x62,0xf5,0x74,0x18,0x38,0x07]
+          vcvtbiasps2hf8 xmm0, xmm1, dword ptr [rdi]{1to4}
 
 // CHECK: vcvtbiasps2hf8s xmm0, zmm1, zmm2
 // CHECK: encoding: [0x62,0xf5,0x74,0x48,0x3a,0xc2]
@@ -372,11 +428,17 @@
 // CHECK: encoding: [0x62,0xf5,0x74,0xc9,0x3a,0xc2]
           vcvtbiasps2hf8s xmm0 {k1} {z}, zmm1, zmm2
 
-//
-// Group C: 8bit->PS expanding conversions
-//
+// CHECK: vcvtbiasps2hf8s xmm0, zmm1, dword ptr [rdi]{1to16}
+// CHECK: encoding: [0x62,0xf5,0x74,0x58,0x3a,0x07]
+          vcvtbiasps2hf8s xmm0, zmm1, dword ptr [rdi]{1to16}
+
+// CHECK: vcvtbiasps2hf8s xmm0, ymm1, dword ptr [rdi]{1to8}
+// CHECK: encoding: [0x62,0xf5,0x74,0x38,0x3a,0x07]
+          vcvtbiasps2hf8s xmm0, ymm1, dword ptr [rdi]{1to8}
 
-// vcvtbf82ps
+// CHECK: vcvtbiasps2hf8s xmm0, xmm1, dword ptr [rdi]{1to4}
+// CHECK: encoding: [0x62,0xf5,0x74,0x18,0x3a,0x07]
+          vcvtbiasps2hf8s xmm0, xmm1, dword ptr [rdi]{1to4}
 
 // CHECK: vcvtbf82ps zmm0, xmm1
 // CHECK: encoding: [0x62,0xf5,0xfc,0x48,0x36,0xc1]
@@ -410,8 +472,6 @@
 // CHECK: encoding: [0x62,0xf5,0xfc,0xc9,0x36,0xc1]
           vcvtbf82ps zmm0 {k1} {z}, xmm1
 
-// vcvthf82ps
-
 // CHECK: vcvthf82ps zmm0, xmm1
 // CHECK: encoding: [0x62,0xf5,0x7c,0x48,0x36,0xc1]
           vcvthf82ps zmm0, xmm1
@@ -444,12 +504,6 @@
 // CHECK: encoding: [0x62,0xf5,0x7c,0xc9,0x36,0xc1]
           vcvthf82ps zmm0 {k1} {z}, xmm1
 
-//
-// Group D: BF8/HF8->BF4S store-like truncations
-//
-
-// vcvtbf82bf4s
-
 // CHECK: vcvtbf82bf4s ymm0, zmm1
 // CHECK: encoding: [0x62,0xf5,0xfe,0x48,0x3d,0xc8]
           vcvtbf82bf4s ymm0, zmm1
@@ -474,8 +528,6 @@
 // CHECK: encoding: [0x62,0xf5,0xfe,0x08,0x3d,0x0f]
           vcvtbf82bf4s qword ptr [rdi], xmm1
 
-// vcvthf82bf4s
-
 // CHECK: vcvthf82bf4s ymm0, zmm1
 // CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x3d,0xc8]
           vcvthf82bf4s ymm0, zmm1
@@ -500,12 +552,6 @@
 // CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x3d,0x0f]
           vcvthf82bf4s qword ptr [rdi], xmm1
 
-//
-// Group E: Same-size reg-only conversions (no masking)
-//
-
-// vcvtbf82bf6s
-
 // CHECK: vcvtbf82bf6s zmm0, zmm1
 // CHECK: encoding: [0x62,0xf5,0xfe,0x48,0x3e,0xc1]
           vcvtbf82bf6s zmm0, zmm1
@@ -518,8 +564,6 @@
 // CHECK: encoding: [0x62,0xf5,0xfe,0x08,0x3e,0xc1]
           vcvtbf82bf6s xmm0, xmm1
 
-// vcvthf82hf6s
-
 // CHECK: vcvthf82hf6s zmm0, zmm1
 // CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x3c,0xc1]
           vcvthf82hf6s zmm0, zmm1
@@ -532,12 +576,6 @@
 // CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x3c,0xc1]
           vcvthf82hf6s xmm0, xmm1
 
-//
-// Group F: Expanding/same-size conversions with masking
-//
-
-// vcvtbf42hf8
-
 // CHECK: vcvtbf42hf8 zmm0, ymm1
 // CHECK: encoding: [0x62,0xf5,0x7c,0x48,0x37,0xc1]
           vcvtbf42hf8 zmm0, ymm1
@@ -570,8 +608,6 @@
 // CHECK: encoding: [0x62,0xf5,0x7c,0xc9,0x37,0xc1]
           vcvtbf42hf8 zmm0 {k1} {z}, ymm1
 
-// vcvtbf62hf8
-
 // CHECK: vcvtbf62hf8 zmm0, zmm1
 // CHECK: encoding: [0x62,0xf5,0xfd,0x48,0x37,0xc1]
           vcvtbf62hf8 zmm0, zmm1
@@ -592,8 +628,6 @@
 // CHECK: encoding: [0x62,0xf5,0xfd,0xc9,0x37,0xc1]
           vcvtbf62hf8 zmm0 {k1} {z}, zmm1
 
-// vcvthf62hf8
-
 // CHECK: vcvthf62hf8 zmm0, zmm1
 // CHECK: encoding: [0x62,0xf5,0x7d,0x48,0x37,0xc1]
           vcvthf62hf8 zmm0, zmm1
@@ -614,10 +648,6 @@
 // CHECK: encoding: [0x62,0xf5,0x7d,0xc9,0x37,0xc1]
           vcvthf62hf8 zmm0 {k1} {z}, zmm1
 
-//
-// Group G: VPMOVSSDB - Integer DWord->Byte signed saturation
-//
-
 // CHECK: vpmovssdb xmm0, zmm1
 // CHECK: encoding: [0x62,0xf2,0x7e,0x48,0x41,0xc8]
           vpmovssdb xmm0, zmm1
@@ -650,10 +680,6 @@
 // CHECK: encoding: [0x62,0xf2,0x7e,0xc9,0x41,0xc8]
           vpmovssdb xmm0 {k1} {z}, zmm1
 
-//
-// Group H: VUNPACKB - Byte unpack with immediate
-//
-
 // CHECK: vunpackb zmm0, zmm1, 1
 // CHECK: encoding: [0x62,0xf3,0x7c,0x48,0x3d,0xc1,0x01]
           vunpackb zmm0, zmm1, 1

>From 784b633f986aa8db2608b38025cc32c6f23cac79 Mon Sep 17 00:00:00 2001
From: Ganesh Gopalasubramanian <Ganesh.Gopalasubramanian at amd.com>
Date: Tue, 1 Sep 2026 09:10:08 +0530
Subject: [PATCH 12/16] Add -avx10v2aux to attr-target-x86 test

---
 clang/test/CodeGen/attr-target-x86.c | 2 +-
 1 file changed, 1 insertion(+), 1 deletion(-)

diff --git a/clang/test/CodeGen/attr-target-x86.c b/clang/test/CodeGen/attr-target-x86.c
index 69be8abd42e9ee..3656a479f55faa 100644
--- a/clang/test/CodeGen/attr-target-x86.c
+++ b/clang/test/CodeGen/attr-target-x86.c
@@ -506,7 +506,7 @@ __attribute__((target("fxsr")))
 void f_fxsr(void) {}
 
 // CHECK: [[f_general_regs_only]] = {{.*}}"target-cpu"="i686"
-// CHECK-SAME: "target-features"="+cmov,+cx8,-aes,-amx-avx512,-avx,-avx10.1,-avx10.2,-avx2,-avx512bf16,-avx512bitalg,-avx512bmm,-avx512bw,-avx512cd,-avx512dq,-avx512f,-avx512fp16,-avx512ifma,-avx512vbmi,-avx512vbmi2,-avx512vl,-avx512vnni,-avx512vp2intersect,-avx512vpopcntdq,-avxifma,-avxneconvert,-avxvnni,-avxvnniint16,-avxvnniint8,-f16c,-fma,-fma4,-gfni,-kl,-mmx,-pclmul,-sha,-sha512,-sm3,-sm4,-sse,-sse2,-sse3,-sse4.1,-sse4.2,-sse4a,-ssse3,-vaes,-vpclmulqdq,-widekl,-x87,-xop"
+// CHECK-SAME: "target-features"="+cmov,+cx8,-aes,-amx-avx512,-avx,-avx10.1,-avx10.2,-avx10v2aux,-avx2,-avx512bf16,-avx512bitalg,-avx512bmm,-avx512bw,-avx512cd,-avx512dq,-avx512f,-avx512fp16,-avx512ifma,-avx512vbmi,-avx512vbmi2,-avx512vl,-avx512vnni,-avx512vp2intersect,-avx512vpopcntdq,-avxifma,-avxneconvert,-avxvnni,-avxvnniint16,-avxvnniint8,-f16c,-fma,-fma4,-gfni,-kl,-mmx,-pclmul,-sha,-sha512,-sm3,-sm4,-sse,-sse2,-sse3,-sse4.1,-sse4.2,-sse4a,-ssse3,-vaes,-vpclmulqdq,-widekl,-x87,-xop"
 __attribute__((target("general-regs-only")))
 void f_general_regs_only(void) {}
 

>From 28dbec8a940128feb313a11947d267c2fc688931 Mon Sep 17 00:00:00 2001
From: Ganesh Gopalasubramanian <Ganesh.Gopalasubramanian at amd.com>
Date: Wed, 2 Sep 2026 16:54:57 +0530
Subject: [PATCH 13/16] Split the large intrinsic test file into multiple test
 files

---
 .../X86/avx10_v2aux-cvt-fp4-fp6-intrinsics.ll |  826 ++++
 .../X86/avx10_v2aux-cvt-fp8-ps-intrinsics.ll  |  389 ++
 .../X86/avx10_v2aux-cvt-ps-fp8-intrinsics.ll  |  833 ++++
 .../CodeGen/X86/avx10_v2aux-intrinsics.ll     | 3909 -----------------
 ...avx10_v2aux-mask-bias-ps-fp8-intrinsics.ll |  639 +++
 .../avx10_v2aux-mask-cvt-ps-fp8-intrinsics.ll |  909 ++++
 .../X86/avx10_v2aux-pmovssdb-intrinsics.ll    |  249 ++
 .../X86/avx10_v2aux-unpackb-intrinsics.ll     |   82 +
 8 files changed, 3927 insertions(+), 3909 deletions(-)
 create mode 100644 llvm/test/CodeGen/X86/avx10_v2aux-cvt-fp4-fp6-intrinsics.ll
 create mode 100644 llvm/test/CodeGen/X86/avx10_v2aux-cvt-fp8-ps-intrinsics.ll
 create mode 100644 llvm/test/CodeGen/X86/avx10_v2aux-cvt-ps-fp8-intrinsics.ll
 delete mode 100644 llvm/test/CodeGen/X86/avx10_v2aux-intrinsics.ll
 create mode 100644 llvm/test/CodeGen/X86/avx10_v2aux-mask-bias-ps-fp8-intrinsics.ll
 create mode 100644 llvm/test/CodeGen/X86/avx10_v2aux-mask-cvt-ps-fp8-intrinsics.ll
 create mode 100644 llvm/test/CodeGen/X86/avx10_v2aux-pmovssdb-intrinsics.ll
 create mode 100644 llvm/test/CodeGen/X86/avx10_v2aux-unpackb-intrinsics.ll

diff --git a/llvm/test/CodeGen/X86/avx10_v2aux-cvt-fp4-fp6-intrinsics.ll b/llvm/test/CodeGen/X86/avx10_v2aux-cvt-fp4-fp6-intrinsics.ll
new file mode 100644
index 00000000000000..09ca861bf73f98
--- /dev/null
+++ b/llvm/test/CodeGen/X86/avx10_v2aux-cvt-fp4-fp6-intrinsics.ll
@@ -0,0 +1,826 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
+; RUN: llc < %s -verify-machineinstrs -mtriple=x86_64-unknown-unknown --show-mc-encoding -mattr=+avx10v2aux | FileCheck %s --check-prefixes=CHECK,X64
+; RUN: llc < %s -verify-machineinstrs -mtriple=i686-unknown-unknown --show-mc-encoding -mattr=+avx10v2aux | FileCheck %s --check-prefixes=CHECK,X86
+
+define <16 x i8> @test_int_x86_avx10_vcvtbf82bf4s128(<16 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbf82bf4s128:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbf82bf4s %xmm0, %xmm0 # encoding: [0x62,0xf5,0xfe,0x08,0x3d,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf82bf4s128(<16 x i8> %a)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtbf82bf4s256(<32 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbf82bf4s256:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbf82bf4s %ymm0, %xmm0 # encoding: [0x62,0xf5,0xfe,0x28,0x3d,0xc0]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf82bf4s256(<32 x i8> %a)
+  ret <16 x i8> %ret
+}
+
+define <32 x i8> @test_int_x86_avx10_vcvtbf82bf4s512(<64 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbf82bf4s512:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbf82bf4s %zmm0, %ymm0 # encoding: [0x62,0xf5,0xfe,0x48,0x3d,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <32 x i8> @llvm.x86.avx10.vcvtbf82bf4s512(<64 x i8> %a)
+  ret <32 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.vcvtbf82bf4s128(<16 x i8>)
+declare <16 x i8> @llvm.x86.avx10.vcvtbf82bf4s256(<32 x i8>)
+declare <32 x i8> @llvm.x86.avx10.vcvtbf82bf4s512(<64 x i8>)
+
+define <16 x i8> @test_int_x86_avx10_vcvtbf82bf4s128_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf82bf4s128_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    vmovaps (%rdi), %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0x07]
+; X64-NEXT:    vcvtbf82bf4s %xmm0, %xmm0 # encoding: [0x62,0xf5,0xfe,0x08,0x3d,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbf82bf4s128_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vmovaps (%eax), %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0x00]
+; X86-NEXT:    vcvtbf82bf4s %xmm0, %xmm0 # encoding: [0x62,0xf5,0xfe,0x08,0x3d,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <16 x i8>, ptr %ptr_a
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf82bf4s128(<16 x i8> %a)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtbf82bf4s256_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf82bf4s256_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    vmovaps (%rdi), %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfc,0x28,0x07]
+; X64-NEXT:    vcvtbf82bf4s %ymm0, %xmm0 # encoding: [0x62,0xf5,0xfe,0x28,0x3d,0xc0]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbf82bf4s256_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vmovaps (%eax), %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfc,0x28,0x00]
+; X86-NEXT:    vcvtbf82bf4s %ymm0, %xmm0 # encoding: [0x62,0xf5,0xfe,0x28,0x3d,0xc0]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <32 x i8>, ptr %ptr_a
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf82bf4s256(<32 x i8> %a)
+  ret <16 x i8> %ret
+}
+
+define <32 x i8> @test_int_x86_avx10_vcvtbf82bf4s512_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf82bf4s512_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    vmovaps (%rdi), %zmm0 # encoding: [0x62,0xf1,0x7c,0x48,0x28,0x07]
+; X64-NEXT:    vcvtbf82bf4s %zmm0, %ymm0 # encoding: [0x62,0xf5,0xfe,0x48,0x3d,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbf82bf4s512_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vmovaps (%eax), %zmm0 # encoding: [0x62,0xf1,0x7c,0x48,0x28,0x00]
+; X86-NEXT:    vcvtbf82bf4s %zmm0, %ymm0 # encoding: [0x62,0xf5,0xfe,0x48,0x3d,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <64 x i8>, ptr %ptr_a
+  %ret = call <32 x i8> @llvm.x86.avx10.vcvtbf82bf4s512(<64 x i8> %a)
+  ret <32 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvthf82bf4s128(<16 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvthf82bf4s128:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvthf82bf4s %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x3d,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvthf82bf4s128(<16 x i8> %a)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvthf82bf4s256(<32 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvthf82bf4s256:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvthf82bf4s %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x3d,0xc0]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvthf82bf4s256(<32 x i8> %a)
+  ret <16 x i8> %ret
+}
+
+define <32 x i8> @test_int_x86_avx10_vcvthf82bf4s512(<64 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvthf82bf4s512:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvthf82bf4s %zmm0, %ymm0 # encoding: [0x62,0xf5,0x7e,0x48,0x3d,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <32 x i8> @llvm.x86.avx10.vcvthf82bf4s512(<64 x i8> %a)
+  ret <32 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.vcvthf82bf4s128(<16 x i8>)
+declare <16 x i8> @llvm.x86.avx10.vcvthf82bf4s256(<32 x i8>)
+declare <32 x i8> @llvm.x86.avx10.vcvthf82bf4s512(<64 x i8>)
+
+define <16 x i8> @test_int_x86_avx10_vcvthf82bf4s128_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvthf82bf4s128_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    vmovaps (%rdi), %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0x07]
+; X64-NEXT:    vcvthf82bf4s %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x3d,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvthf82bf4s128_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vmovaps (%eax), %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0x00]
+; X86-NEXT:    vcvthf82bf4s %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x3d,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <16 x i8>, ptr %ptr_a
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvthf82bf4s128(<16 x i8> %a)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvthf82bf4s256_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvthf82bf4s256_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    vmovaps (%rdi), %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfc,0x28,0x07]
+; X64-NEXT:    vcvthf82bf4s %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x3d,0xc0]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvthf82bf4s256_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vmovaps (%eax), %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfc,0x28,0x00]
+; X86-NEXT:    vcvthf82bf4s %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x3d,0xc0]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <32 x i8>, ptr %ptr_a
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvthf82bf4s256(<32 x i8> %a)
+  ret <16 x i8> %ret
+}
+
+define <32 x i8> @test_int_x86_avx10_vcvthf82bf4s512_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvthf82bf4s512_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    vmovaps (%rdi), %zmm0 # encoding: [0x62,0xf1,0x7c,0x48,0x28,0x07]
+; X64-NEXT:    vcvthf82bf4s %zmm0, %ymm0 # encoding: [0x62,0xf5,0x7e,0x48,0x3d,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvthf82bf4s512_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vmovaps (%eax), %zmm0 # encoding: [0x62,0xf1,0x7c,0x48,0x28,0x00]
+; X86-NEXT:    vcvthf82bf4s %zmm0, %ymm0 # encoding: [0x62,0xf5,0x7e,0x48,0x3d,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <64 x i8>, ptr %ptr_a
+  %ret = call <32 x i8> @llvm.x86.avx10.vcvthf82bf4s512(<64 x i8> %a)
+  ret <32 x i8> %ret
+}
+
+define void @test_int_x86_avx10_vcvtbf82bf4s128_store(ptr %ptr, <16 x i8> %a) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf82bf4s128_store:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtbf82bf4s %xmm0, (%rdi) # encoding: [0x62,0xf5,0xfe,0x08,0x3d,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbf82bf4s128_store:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtbf82bf4s %xmm0, (%eax) # encoding: [0x62,0xf5,0xfe,0x08,0x3d,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf82bf4s128(<16 x i8> %a)
+  %cast = bitcast <16 x i8> %ret to <2 x i64>
+  %low = extractelement <2 x i64> %cast, i64 0
+  store i64 %low, ptr %ptr, align 1
+  ret void
+}
+
+define void @test_int_x86_avx10_vcvtbf82bf4s256_store(ptr %ptr, <32 x i8> %a) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf82bf4s256_store:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtbf82bf4s %ymm0, (%rdi) # encoding: [0x62,0xf5,0xfe,0x28,0x3d,0x07]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbf82bf4s256_store:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtbf82bf4s %ymm0, (%eax) # encoding: [0x62,0xf5,0xfe,0x28,0x3d,0x00]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf82bf4s256(<32 x i8> %a)
+  store <16 x i8> %ret, ptr %ptr, align 1
+  ret void
+}
+
+define void @test_int_x86_avx10_vcvtbf82bf4s512_store(ptr %ptr, <64 x i8> %a) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf82bf4s512_store:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtbf82bf4s %zmm0, (%rdi) # encoding: [0x62,0xf5,0xfe,0x48,0x3d,0x07]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbf82bf4s512_store:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtbf82bf4s %zmm0, (%eax) # encoding: [0x62,0xf5,0xfe,0x48,0x3d,0x00]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <32 x i8> @llvm.x86.avx10.vcvtbf82bf4s512(<64 x i8> %a)
+  store <32 x i8> %ret, ptr %ptr, align 1
+  ret void
+}
+
+define void @test_int_x86_avx10_vcvthf82bf4s128_store(ptr %ptr, <16 x i8> %a) {
+; X64-LABEL: test_int_x86_avx10_vcvthf82bf4s128_store:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvthf82bf4s %xmm0, (%rdi) # encoding: [0x62,0xf5,0x7e,0x08,0x3d,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvthf82bf4s128_store:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvthf82bf4s %xmm0, (%eax) # encoding: [0x62,0xf5,0x7e,0x08,0x3d,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvthf82bf4s128(<16 x i8> %a)
+  %cast = bitcast <16 x i8> %ret to <2 x double>
+  %low = extractelement <2 x double> %cast, i64 0
+  store double %low, ptr %ptr, align 1
+  ret void
+}
+
+define void @test_int_x86_avx10_vcvthf82bf4s256_store(ptr %ptr, <32 x i8> %a) {
+; X64-LABEL: test_int_x86_avx10_vcvthf82bf4s256_store:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvthf82bf4s %ymm0, (%rdi) # encoding: [0x62,0xf5,0x7e,0x28,0x3d,0x07]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvthf82bf4s256_store:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvthf82bf4s %ymm0, (%eax) # encoding: [0x62,0xf5,0x7e,0x28,0x3d,0x00]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvthf82bf4s256(<32 x i8> %a)
+  store <16 x i8> %ret, ptr %ptr, align 1
+  ret void
+}
+
+define void @test_int_x86_avx10_vcvthf82bf4s512_store(ptr %ptr, <64 x i8> %a) {
+; X64-LABEL: test_int_x86_avx10_vcvthf82bf4s512_store:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvthf82bf4s %zmm0, (%rdi) # encoding: [0x62,0xf5,0x7e,0x48,0x3d,0x07]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvthf82bf4s512_store:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvthf82bf4s %zmm0, (%eax) # encoding: [0x62,0xf5,0x7e,0x48,0x3d,0x00]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <32 x i8> @llvm.x86.avx10.vcvthf82bf4s512(<64 x i8> %a)
+  store <32 x i8> %ret, ptr %ptr, align 1
+  ret void
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtbf82bf6s128(<16 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbf82bf6s128:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbf82bf6s %xmm0, %xmm0 # encoding: [0x62,0xf5,0xfe,0x08,0x3e,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf82bf6s128(<16 x i8> %a)
+  ret <16 x i8> %ret
+}
+
+define <32 x i8> @test_int_x86_avx10_vcvtbf82bf6s256(<32 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbf82bf6s256:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbf82bf6s %ymm0, %ymm0 # encoding: [0x62,0xf5,0xfe,0x28,0x3e,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <32 x i8> @llvm.x86.avx10.vcvtbf82bf6s256(<32 x i8> %a)
+  ret <32 x i8> %ret
+}
+
+define <64 x i8> @test_int_x86_avx10_vcvtbf82bf6s512(<64 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbf82bf6s512:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbf82bf6s %zmm0, %zmm0 # encoding: [0x62,0xf5,0xfe,0x48,0x3e,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <64 x i8> @llvm.x86.avx10.vcvtbf82bf6s512(<64 x i8> %a)
+  ret <64 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.vcvtbf82bf6s128(<16 x i8>)
+declare <32 x i8> @llvm.x86.avx10.vcvtbf82bf6s256(<32 x i8>)
+declare <64 x i8> @llvm.x86.avx10.vcvtbf82bf6s512(<64 x i8>)
+
+define <16 x i8> @test_int_x86_avx10_vcvtbf82bf6s128_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf82bf6s128_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    vmovaps (%rdi), %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0x07]
+; X64-NEXT:    vcvtbf82bf6s %xmm0, %xmm0 # encoding: [0x62,0xf5,0xfe,0x08,0x3e,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbf82bf6s128_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vmovaps (%eax), %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0x00]
+; X86-NEXT:    vcvtbf82bf6s %xmm0, %xmm0 # encoding: [0x62,0xf5,0xfe,0x08,0x3e,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <16 x i8>, ptr %ptr_a
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf82bf6s128(<16 x i8> %a)
+  ret <16 x i8> %ret
+}
+
+define <32 x i8> @test_int_x86_avx10_vcvtbf82bf6s256_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf82bf6s256_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    vmovaps (%rdi), %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfc,0x28,0x07]
+; X64-NEXT:    vcvtbf82bf6s %ymm0, %ymm0 # encoding: [0x62,0xf5,0xfe,0x28,0x3e,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbf82bf6s256_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vmovaps (%eax), %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfc,0x28,0x00]
+; X86-NEXT:    vcvtbf82bf6s %ymm0, %ymm0 # encoding: [0x62,0xf5,0xfe,0x28,0x3e,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <32 x i8>, ptr %ptr_a
+  %ret = call <32 x i8> @llvm.x86.avx10.vcvtbf82bf6s256(<32 x i8> %a)
+  ret <32 x i8> %ret
+}
+
+define <64 x i8> @test_int_x86_avx10_vcvtbf82bf6s512_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf82bf6s512_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    vmovaps (%rdi), %zmm0 # encoding: [0x62,0xf1,0x7c,0x48,0x28,0x07]
+; X64-NEXT:    vcvtbf82bf6s %zmm0, %zmm0 # encoding: [0x62,0xf5,0xfe,0x48,0x3e,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbf82bf6s512_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vmovaps (%eax), %zmm0 # encoding: [0x62,0xf1,0x7c,0x48,0x28,0x00]
+; X86-NEXT:    vcvtbf82bf6s %zmm0, %zmm0 # encoding: [0x62,0xf5,0xfe,0x48,0x3e,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <64 x i8>, ptr %ptr_a
+  %ret = call <64 x i8> @llvm.x86.avx10.vcvtbf82bf6s512(<64 x i8> %a)
+  ret <64 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvthf82hf6s128(<16 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvthf82hf6s128:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvthf82hf6s %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x3c,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvthf82hf6s128(<16 x i8> %a)
+  ret <16 x i8> %ret
+}
+
+define <32 x i8> @test_int_x86_avx10_vcvthf82hf6s256(<32 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvthf82hf6s256:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvthf82hf6s %ymm0, %ymm0 # encoding: [0x62,0xf5,0x7e,0x28,0x3c,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <32 x i8> @llvm.x86.avx10.vcvthf82hf6s256(<32 x i8> %a)
+  ret <32 x i8> %ret
+}
+
+define <64 x i8> @test_int_x86_avx10_vcvthf82hf6s512(<64 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvthf82hf6s512:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvthf82hf6s %zmm0, %zmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x3c,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <64 x i8> @llvm.x86.avx10.vcvthf82hf6s512(<64 x i8> %a)
+  ret <64 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.vcvthf82hf6s128(<16 x i8>)
+declare <32 x i8> @llvm.x86.avx10.vcvthf82hf6s256(<32 x i8>)
+declare <64 x i8> @llvm.x86.avx10.vcvthf82hf6s512(<64 x i8>)
+
+define <16 x i8> @test_int_x86_avx10_vcvthf82hf6s128_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvthf82hf6s128_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    vmovaps (%rdi), %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0x07]
+; X64-NEXT:    vcvthf82hf6s %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x3c,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvthf82hf6s128_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vmovaps (%eax), %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0x00]
+; X86-NEXT:    vcvthf82hf6s %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x3c,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <16 x i8>, ptr %ptr_a
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvthf82hf6s128(<16 x i8> %a)
+  ret <16 x i8> %ret
+}
+
+define <32 x i8> @test_int_x86_avx10_vcvthf82hf6s256_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvthf82hf6s256_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    vmovaps (%rdi), %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfc,0x28,0x07]
+; X64-NEXT:    vcvthf82hf6s %ymm0, %ymm0 # encoding: [0x62,0xf5,0x7e,0x28,0x3c,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvthf82hf6s256_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vmovaps (%eax), %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfc,0x28,0x00]
+; X86-NEXT:    vcvthf82hf6s %ymm0, %ymm0 # encoding: [0x62,0xf5,0x7e,0x28,0x3c,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <32 x i8>, ptr %ptr_a
+  %ret = call <32 x i8> @llvm.x86.avx10.vcvthf82hf6s256(<32 x i8> %a)
+  ret <32 x i8> %ret
+}
+
+define <64 x i8> @test_int_x86_avx10_vcvthf82hf6s512_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvthf82hf6s512_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    vmovaps (%rdi), %zmm0 # encoding: [0x62,0xf1,0x7c,0x48,0x28,0x07]
+; X64-NEXT:    vcvthf82hf6s %zmm0, %zmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x3c,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvthf82hf6s512_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vmovaps (%eax), %zmm0 # encoding: [0x62,0xf1,0x7c,0x48,0x28,0x00]
+; X86-NEXT:    vcvthf82hf6s %zmm0, %zmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x3c,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <64 x i8>, ptr %ptr_a
+  %ret = call <64 x i8> @llvm.x86.avx10.vcvthf82hf6s512(<64 x i8> %a)
+  ret <64 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtbf42hf8128(<16 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbf42hf8128:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbf42hf8 %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x37,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf42hf8128(<16 x i8> %a)
+  ret <16 x i8> %ret
+}
+
+define <32 x i8> @test_int_x86_avx10_vcvtbf42hf8256(<16 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbf42hf8256:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbf42hf8 %xmm0, %ymm0 # encoding: [0x62,0xf5,0x7c,0x28,0x37,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <32 x i8> @llvm.x86.avx10.vcvtbf42hf8256(<16 x i8> %a)
+  ret <32 x i8> %ret
+}
+
+define <64 x i8> @test_int_x86_avx10_vcvtbf42hf8512(<32 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbf42hf8512:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbf42hf8 %ymm0, %zmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x37,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <64 x i8> @llvm.x86.avx10.vcvtbf42hf8512(<32 x i8> %a)
+  ret <64 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.vcvtbf42hf8128(<16 x i8>)
+declare <32 x i8> @llvm.x86.avx10.vcvtbf42hf8256(<16 x i8>)
+declare <64 x i8> @llvm.x86.avx10.vcvtbf42hf8512(<32 x i8>)
+
+define <16 x i8> @test_int_x86_avx10_vcvtbf42hf8128_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf42hf8128_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtbf42hf8 (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x37,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbf42hf8128_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtbf42hf8 (%eax), %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x37,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <16 x i8>, ptr %ptr_a
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf42hf8128(<16 x i8> %a)
+  ret <16 x i8> %ret
+}
+
+define <32 x i8> @test_int_x86_avx10_vcvtbf42hf8256_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf42hf8256_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtbf42hf8 (%rdi), %ymm0 # encoding: [0x62,0xf5,0x7c,0x28,0x37,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbf42hf8256_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtbf42hf8 (%eax), %ymm0 # encoding: [0x62,0xf5,0x7c,0x28,0x37,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <16 x i8>, ptr %ptr_a
+  %ret = call <32 x i8> @llvm.x86.avx10.vcvtbf42hf8256(<16 x i8> %a)
+  ret <32 x i8> %ret
+}
+
+define <64 x i8> @test_int_x86_avx10_vcvtbf42hf8512_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf42hf8512_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtbf42hf8 (%rdi), %zmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x37,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbf42hf8512_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtbf42hf8 (%eax), %zmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x37,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <32 x i8>, ptr %ptr_a
+  %ret = call <64 x i8> @llvm.x86.avx10.vcvtbf42hf8512(<32 x i8> %a)
+  ret <64 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtbf42hf8128_vzload(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf42hf8128_vzload:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtbf42hf8 (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x37,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbf42hf8128_vzload:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtbf42hf8 (%eax), %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x37,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %l = load i64, ptr %ptr_a, align 1
+  %v = insertelement <2 x i64> <i64 poison, i64 0>, i64 %l, i64 0
+  %a = bitcast <2 x i64> %v to <16 x i8>
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf42hf8128(<16 x i8> %a)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtbf42hf8128_vzload_mask(ptr %ptr_a, <16 x i8> %src, i16 %mask) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf42hf8128_vzload_mask:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtbf42hf8 (%rdi), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x37,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbf42hf8128_vzload_mask:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtbf42hf8 (%eax), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x37,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %l = load i64, ptr %ptr_a, align 1
+  %v = insertelement <2 x i64> <i64 poison, i64 0>, i64 %l, i64 0
+  %a = bitcast <2 x i64> %v to <16 x i8>
+  %cvt = call <16 x i8> @llvm.x86.avx10.vcvtbf42hf8128(<16 x i8> %a)
+  %msk = bitcast i16 %mask to <16 x i1>
+  %ret = select <16 x i1> %msk, <16 x i8> %cvt, <16 x i8> %src
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtbf42hf8128_vzload_maskz(ptr %ptr_a, i16 %mask) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf42hf8128_vzload_maskz:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtbf42hf8 (%rdi), %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x37,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbf42hf8128_vzload_maskz:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtbf42hf8 (%eax), %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x37,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %l = load i64, ptr %ptr_a, align 1
+  %v = insertelement <2 x i64> <i64 poison, i64 0>, i64 %l, i64 0
+  %a = bitcast <2 x i64> %v to <16 x i8>
+  %cvt = call <16 x i8> @llvm.x86.avx10.vcvtbf42hf8128(<16 x i8> %a)
+  %msk = bitcast i16 %mask to <16 x i1>
+  %ret = select <16 x i1> %msk, <16 x i8> %cvt, <16 x i8> zeroinitializer
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtbf42hf8_s2v128(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf42hf8_s2v128:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtbf42hf8 (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x37,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbf42hf8_s2v128:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtbf42hf8 (%eax), %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x37,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %l = load i64, ptr %ptr_a, align 8
+  %v = insertelement <2 x i64> poison, i64 %l, i64 0
+  %a = bitcast <2 x i64> %v to <16 x i8>
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf42hf8128(<16 x i8> %a)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtbf42hf8128_mem_mask(ptr %ptr_a, <16 x i8> %src, i16 %mask) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf42hf8128_mem_mask:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtbf42hf8 (%rdi), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x37,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbf42hf8128_mem_mask:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtbf42hf8 (%eax), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x37,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <16 x i8>, ptr %ptr_a
+  %cvt = call <16 x i8> @llvm.x86.avx10.vcvtbf42hf8128(<16 x i8> %a)
+  %msk = bitcast i16 %mask to <16 x i1>
+  %ret = select <16 x i1> %msk, <16 x i8> %cvt, <16 x i8> %src
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtbf42hf8128_mem_maskz(ptr %ptr_a, i16 %mask) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf42hf8128_mem_maskz:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtbf42hf8 (%rdi), %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x37,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbf42hf8128_mem_maskz:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtbf42hf8 (%eax), %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x37,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <16 x i8>, ptr %ptr_a
+  %cvt = call <16 x i8> @llvm.x86.avx10.vcvtbf42hf8128(<16 x i8> %a)
+  %msk = bitcast i16 %mask to <16 x i1>
+  %ret = select <16 x i1> %msk, <16 x i8> %cvt, <16 x i8> zeroinitializer
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtbf62hf8128(<16 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbf62hf8128:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbf62hf8 %xmm0, %xmm0 # encoding: [0x62,0xf5,0xfd,0x08,0x37,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf62hf8128(<16 x i8> %a)
+  ret <16 x i8> %ret
+}
+
+define <32 x i8> @test_int_x86_avx10_vcvtbf62hf8256(<32 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbf62hf8256:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbf62hf8 %ymm0, %ymm0 # encoding: [0x62,0xf5,0xfd,0x28,0x37,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <32 x i8> @llvm.x86.avx10.vcvtbf62hf8256(<32 x i8> %a)
+  ret <32 x i8> %ret
+}
+
+define <64 x i8> @test_int_x86_avx10_vcvtbf62hf8512(<64 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbf62hf8512:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbf62hf8 %zmm0, %zmm0 # encoding: [0x62,0xf5,0xfd,0x48,0x37,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <64 x i8> @llvm.x86.avx10.vcvtbf62hf8512(<64 x i8> %a)
+  ret <64 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.vcvtbf62hf8128(<16 x i8>)
+declare <32 x i8> @llvm.x86.avx10.vcvtbf62hf8256(<32 x i8>)
+declare <64 x i8> @llvm.x86.avx10.vcvtbf62hf8512(<64 x i8>)
+
+define <16 x i8> @test_int_x86_avx10_vcvtbf62hf8128_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf62hf8128_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    vmovdqa (%rdi), %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf9,0x6f,0x07]
+; X64-NEXT:    vcvtbf62hf8 %xmm0, %xmm0 # encoding: [0x62,0xf5,0xfd,0x08,0x37,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbf62hf8128_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vmovdqa (%eax), %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf9,0x6f,0x00]
+; X86-NEXT:    vcvtbf62hf8 %xmm0, %xmm0 # encoding: [0x62,0xf5,0xfd,0x08,0x37,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <16 x i8>, ptr %ptr_a
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf62hf8128(<16 x i8> %a)
+  ret <16 x i8> %ret
+}
+
+define <32 x i8> @test_int_x86_avx10_vcvtbf62hf8256_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf62hf8256_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    vmovdqa (%rdi), %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfd,0x6f,0x07]
+; X64-NEXT:    vcvtbf62hf8 %ymm0, %ymm0 # encoding: [0x62,0xf5,0xfd,0x28,0x37,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbf62hf8256_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vmovdqa (%eax), %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfd,0x6f,0x00]
+; X86-NEXT:    vcvtbf62hf8 %ymm0, %ymm0 # encoding: [0x62,0xf5,0xfd,0x28,0x37,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <32 x i8>, ptr %ptr_a
+  %ret = call <32 x i8> @llvm.x86.avx10.vcvtbf62hf8256(<32 x i8> %a)
+  ret <32 x i8> %ret
+}
+
+define <64 x i8> @test_int_x86_avx10_vcvtbf62hf8512_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf62hf8512_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    vmovdqa64 (%rdi), %zmm0 # encoding: [0x62,0xf1,0xfd,0x48,0x6f,0x07]
+; X64-NEXT:    vcvtbf62hf8 %zmm0, %zmm0 # encoding: [0x62,0xf5,0xfd,0x48,0x37,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbf62hf8512_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vmovdqa64 (%eax), %zmm0 # encoding: [0x62,0xf1,0xfd,0x48,0x6f,0x00]
+; X86-NEXT:    vcvtbf62hf8 %zmm0, %zmm0 # encoding: [0x62,0xf5,0xfd,0x48,0x37,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <64 x i8>, ptr %ptr_a
+  %ret = call <64 x i8> @llvm.x86.avx10.vcvtbf62hf8512(<64 x i8> %a)
+  ret <64 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvthf62hf8128(<16 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvthf62hf8128:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvthf62hf8 %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7d,0x08,0x37,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvthf62hf8128(<16 x i8> %a)
+  ret <16 x i8> %ret
+}
+
+define <32 x i8> @test_int_x86_avx10_vcvthf62hf8256(<32 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvthf62hf8256:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvthf62hf8 %ymm0, %ymm0 # encoding: [0x62,0xf5,0x7d,0x28,0x37,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <32 x i8> @llvm.x86.avx10.vcvthf62hf8256(<32 x i8> %a)
+  ret <32 x i8> %ret
+}
+
+define <64 x i8> @test_int_x86_avx10_vcvthf62hf8512(<64 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvthf62hf8512:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvthf62hf8 %zmm0, %zmm0 # encoding: [0x62,0xf5,0x7d,0x48,0x37,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <64 x i8> @llvm.x86.avx10.vcvthf62hf8512(<64 x i8> %a)
+  ret <64 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.vcvthf62hf8128(<16 x i8>)
+declare <32 x i8> @llvm.x86.avx10.vcvthf62hf8256(<32 x i8>)
+declare <64 x i8> @llvm.x86.avx10.vcvthf62hf8512(<64 x i8>)
+
+define <16 x i8> @test_int_x86_avx10_vcvthf62hf8128_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvthf62hf8128_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    vmovdqa (%rdi), %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf9,0x6f,0x07]
+; X64-NEXT:    vcvthf62hf8 %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7d,0x08,0x37,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvthf62hf8128_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vmovdqa (%eax), %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf9,0x6f,0x00]
+; X86-NEXT:    vcvthf62hf8 %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7d,0x08,0x37,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <16 x i8>, ptr %ptr_a
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvthf62hf8128(<16 x i8> %a)
+  ret <16 x i8> %ret
+}
+
+define <32 x i8> @test_int_x86_avx10_vcvthf62hf8256_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvthf62hf8256_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    vmovdqa (%rdi), %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfd,0x6f,0x07]
+; X64-NEXT:    vcvthf62hf8 %ymm0, %ymm0 # encoding: [0x62,0xf5,0x7d,0x28,0x37,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvthf62hf8256_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vmovdqa (%eax), %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfd,0x6f,0x00]
+; X86-NEXT:    vcvthf62hf8 %ymm0, %ymm0 # encoding: [0x62,0xf5,0x7d,0x28,0x37,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <32 x i8>, ptr %ptr_a
+  %ret = call <32 x i8> @llvm.x86.avx10.vcvthf62hf8256(<32 x i8> %a)
+  ret <32 x i8> %ret
+}
+
+define <64 x i8> @test_int_x86_avx10_vcvthf62hf8512_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvthf62hf8512_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    vmovdqa64 (%rdi), %zmm0 # encoding: [0x62,0xf1,0xfd,0x48,0x6f,0x07]
+; X64-NEXT:    vcvthf62hf8 %zmm0, %zmm0 # encoding: [0x62,0xf5,0x7d,0x48,0x37,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvthf62hf8512_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vmovdqa64 (%eax), %zmm0 # encoding: [0x62,0xf1,0xfd,0x48,0x6f,0x00]
+; X86-NEXT:    vcvthf62hf8 %zmm0, %zmm0 # encoding: [0x62,0xf5,0x7d,0x48,0x37,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <64 x i8>, ptr %ptr_a
+  %ret = call <64 x i8> @llvm.x86.avx10.vcvthf62hf8512(<64 x i8> %a)
+  ret <64 x i8> %ret
+}
diff --git a/llvm/test/CodeGen/X86/avx10_v2aux-cvt-fp8-ps-intrinsics.ll b/llvm/test/CodeGen/X86/avx10_v2aux-cvt-fp8-ps-intrinsics.ll
new file mode 100644
index 00000000000000..9e4301489c9017
--- /dev/null
+++ b/llvm/test/CodeGen/X86/avx10_v2aux-cvt-fp8-ps-intrinsics.ll
@@ -0,0 +1,389 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
+; RUN: llc < %s -verify-machineinstrs -mtriple=x86_64-unknown-unknown --show-mc-encoding -mattr=+avx10v2aux | FileCheck %s --check-prefixes=CHECK,X64
+; RUN: llc < %s -verify-machineinstrs -mtriple=i686-unknown-unknown --show-mc-encoding -mattr=+avx10v2aux | FileCheck %s --check-prefixes=CHECK,X86
+
+define <4 x float> @test_int_x86_avx10_vcvtbf82ps128(<16 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbf82ps128:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbf82ps %xmm0, %xmm0 # encoding: [0x62,0xf5,0xfc,0x08,0x36,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <4 x float> @llvm.x86.avx10.vcvtbf82ps128(<16 x i8> %a)
+  ret <4 x float> %ret
+}
+
+define <8 x float> @test_int_x86_avx10_vcvtbf82ps256(<16 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbf82ps256:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbf82ps %xmm0, %ymm0 # encoding: [0x62,0xf5,0xfc,0x28,0x36,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <8 x float> @llvm.x86.avx10.vcvtbf82ps256(<16 x i8> %a)
+  ret <8 x float> %ret
+}
+
+define <16 x float> @test_int_x86_avx10_vcvtbf82ps512(<16 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbf82ps512:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbf82ps %xmm0, %zmm0 # encoding: [0x62,0xf5,0xfc,0x48,0x36,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x float> @llvm.x86.avx10.vcvtbf82ps512(<16 x i8> %a)
+  ret <16 x float> %ret
+}
+
+define <4 x float> @test_int_x86_avx10_vcvtbf82ps128_mask(<16 x i8> %a, <4 x float> %src, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf82ps128_mask:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbf82ps %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0xfc,0x09,0x36,0xc8]
+; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbf82ps128_mask:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbf82ps %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0xfc,0x09,0x36,0xc8]
+; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %cvt = call <4 x float> @llvm.x86.avx10.vcvtbf82ps128(<16 x i8> %a)
+  %bst = bitcast i8 %mask to <8 x i1>
+  %ext = shufflevector <8 x i1> %bst, <8 x i1> poison, <4 x i32> <i32 0, i32 1, i32 2, i32 3>
+  %ret = select <4 x i1> %ext, <4 x float> %cvt, <4 x float> %src
+  ret <4 x float> %ret
+}
+
+define <4 x float> @test_int_x86_avx10_vcvtbf82ps128_maskz(<16 x i8> %a, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf82ps128_maskz:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbf82ps %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0xfc,0x89,0x36,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbf82ps128_maskz:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbf82ps %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0xfc,0x89,0x36,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %cvt = call <4 x float> @llvm.x86.avx10.vcvtbf82ps128(<16 x i8> %a)
+  %bst = bitcast i8 %mask to <8 x i1>
+  %ext = shufflevector <8 x i1> %bst, <8 x i1> poison, <4 x i32> <i32 0, i32 1, i32 2, i32 3>
+  %ret = select <4 x i1> %ext, <4 x float> %cvt, <4 x float> zeroinitializer
+  ret <4 x float> %ret
+}
+
+define <8 x float> @test_int_x86_avx10_vcvtbf82ps256_mask(<16 x i8> %a, <8 x float> %src, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf82ps256_mask:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbf82ps %xmm0, %ymm1 {%k1} # encoding: [0x62,0xf5,0xfc,0x29,0x36,0xc8]
+; X64-NEXT:    vmovaps %ymm1, %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfc,0x28,0xc1]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbf82ps256_mask:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbf82ps %xmm0, %ymm1 {%k1} # encoding: [0x62,0xf5,0xfc,0x29,0x36,0xc8]
+; X86-NEXT:    vmovaps %ymm1, %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfc,0x28,0xc1]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %cvt = call <8 x float> @llvm.x86.avx10.vcvtbf82ps256(<16 x i8> %a)
+  %bst = bitcast i8 %mask to <8 x i1>
+  %ret = select <8 x i1> %bst, <8 x float> %cvt, <8 x float> %src
+  ret <8 x float> %ret
+}
+
+define <8 x float> @test_int_x86_avx10_vcvtbf82ps256_maskz(<16 x i8> %a, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf82ps256_maskz:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbf82ps %xmm0, %ymm0 {%k1} {z} # encoding: [0x62,0xf5,0xfc,0xa9,0x36,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbf82ps256_maskz:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbf82ps %xmm0, %ymm0 {%k1} {z} # encoding: [0x62,0xf5,0xfc,0xa9,0x36,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %cvt = call <8 x float> @llvm.x86.avx10.vcvtbf82ps256(<16 x i8> %a)
+  %bst = bitcast i8 %mask to <8 x i1>
+  %ret = select <8 x i1> %bst, <8 x float> %cvt, <8 x float> zeroinitializer
+  ret <8 x float> %ret
+}
+
+define <16 x float> @test_int_x86_avx10_vcvtbf82ps512_mask(<16 x i8> %a, <16 x float> %src, i16 %mask) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf82ps512_mask:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbf82ps %xmm0, %zmm1 {%k1} # encoding: [0x62,0xf5,0xfc,0x49,0x36,0xc8]
+; X64-NEXT:    vmovaps %zmm1, %zmm0 # encoding: [0x62,0xf1,0x7c,0x48,0x28,0xc1]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbf82ps512_mask:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbf82ps %xmm0, %zmm1 {%k1} # encoding: [0x62,0xf5,0xfc,0x49,0x36,0xc8]
+; X86-NEXT:    vmovaps %zmm1, %zmm0 # encoding: [0x62,0xf1,0x7c,0x48,0x28,0xc1]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %cvt = call <16 x float> @llvm.x86.avx10.vcvtbf82ps512(<16 x i8> %a)
+  %bst = bitcast i16 %mask to <16 x i1>
+  %ret = select <16 x i1> %bst, <16 x float> %cvt, <16 x float> %src
+  ret <16 x float> %ret
+}
+
+define <16 x float> @test_int_x86_avx10_vcvtbf82ps512_maskz(<16 x i8> %a, i16 %mask) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf82ps512_maskz:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbf82ps %xmm0, %zmm0 {%k1} {z} # encoding: [0x62,0xf5,0xfc,0xc9,0x36,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbf82ps512_maskz:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbf82ps %xmm0, %zmm0 {%k1} {z} # encoding: [0x62,0xf5,0xfc,0xc9,0x36,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %cvt = call <16 x float> @llvm.x86.avx10.vcvtbf82ps512(<16 x i8> %a)
+  %bst = bitcast i16 %mask to <16 x i1>
+  %ret = select <16 x i1> %bst, <16 x float> %cvt, <16 x float> zeroinitializer
+  ret <16 x float> %ret
+}
+
+define <4 x float> @test_int_x86_avx10_vcvtbf82ps128_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf82ps128_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtbf82ps (%rdi), %xmm0 # encoding: [0x62,0xf5,0xfc,0x08,0x36,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbf82ps128_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtbf82ps (%eax), %xmm0 # encoding: [0x62,0xf5,0xfc,0x08,0x36,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <16 x i8>, ptr %ptr_a
+  %ret = call <4 x float> @llvm.x86.avx10.vcvtbf82ps128(<16 x i8> %a)
+  ret <4 x float> %ret
+}
+
+define <8 x float> @test_int_x86_avx10_vcvtbf82ps256_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf82ps256_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtbf82ps (%rdi), %ymm0 # encoding: [0x62,0xf5,0xfc,0x28,0x36,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbf82ps256_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtbf82ps (%eax), %ymm0 # encoding: [0x62,0xf5,0xfc,0x28,0x36,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <16 x i8>, ptr %ptr_a
+  %ret = call <8 x float> @llvm.x86.avx10.vcvtbf82ps256(<16 x i8> %a)
+  ret <8 x float> %ret
+}
+
+define <16 x float> @test_int_x86_avx10_vcvtbf82ps512_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf82ps512_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtbf82ps (%rdi), %zmm0 # encoding: [0x62,0xf5,0xfc,0x48,0x36,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbf82ps512_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtbf82ps (%eax), %zmm0 # encoding: [0x62,0xf5,0xfc,0x48,0x36,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <16 x i8>, ptr %ptr_a
+  %ret = call <16 x float> @llvm.x86.avx10.vcvtbf82ps512(<16 x i8> %a)
+  ret <16 x float> %ret
+}
+
+declare <4 x float> @llvm.x86.avx10.vcvthf82ps128(<16 x i8>)
+declare <8 x float> @llvm.x86.avx10.vcvthf82ps256(<16 x i8>)
+declare <16 x float> @llvm.x86.avx10.vcvthf82ps512(<16 x i8>)
+
+define <4 x float> @test_int_x86_avx10_vcvthf82ps128(<16 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvthf82ps128:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvthf82ps %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x36,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <4 x float> @llvm.x86.avx10.vcvthf82ps128(<16 x i8> %a)
+  ret <4 x float> %ret
+}
+
+define <8 x float> @test_int_x86_avx10_vcvthf82ps256(<16 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvthf82ps256:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvthf82ps %xmm0, %ymm0 # encoding: [0x62,0xf5,0x7c,0x28,0x36,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <8 x float> @llvm.x86.avx10.vcvthf82ps256(<16 x i8> %a)
+  ret <8 x float> %ret
+}
+
+define <16 x float> @test_int_x86_avx10_vcvthf82ps512(<16 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvthf82ps512:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvthf82ps %xmm0, %zmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x36,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x float> @llvm.x86.avx10.vcvthf82ps512(<16 x i8> %a)
+  ret <16 x float> %ret
+}
+
+define <4 x float> @test_int_x86_avx10_vcvthf82ps128_mask(<16 x i8> %a, <4 x float> %src, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_vcvthf82ps128_mask:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvthf82ps %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x36,0xc8]
+; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvthf82ps128_mask:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvthf82ps %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x36,0xc8]
+; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %cvt = call <4 x float> @llvm.x86.avx10.vcvthf82ps128(<16 x i8> %a)
+  %bst = bitcast i8 %mask to <8 x i1>
+  %ext = shufflevector <8 x i1> %bst, <8 x i1> poison, <4 x i32> <i32 0, i32 1, i32 2, i32 3>
+  %ret = select <4 x i1> %ext, <4 x float> %cvt, <4 x float> %src
+  ret <4 x float> %ret
+}
+
+define <4 x float> @test_int_x86_avx10_vcvthf82ps128_maskz(<16 x i8> %a, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_vcvthf82ps128_maskz:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvthf82ps %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x36,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvthf82ps128_maskz:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvthf82ps %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x36,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %cvt = call <4 x float> @llvm.x86.avx10.vcvthf82ps128(<16 x i8> %a)
+  %bst = bitcast i8 %mask to <8 x i1>
+  %ext = shufflevector <8 x i1> %bst, <8 x i1> poison, <4 x i32> <i32 0, i32 1, i32 2, i32 3>
+  %ret = select <4 x i1> %ext, <4 x float> %cvt, <4 x float> zeroinitializer
+  ret <4 x float> %ret
+}
+
+define <8 x float> @test_int_x86_avx10_vcvthf82ps256_mask(<16 x i8> %a, <8 x float> %src, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_vcvthf82ps256_mask:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvthf82ps %xmm0, %ymm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x36,0xc8]
+; X64-NEXT:    vmovaps %ymm1, %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfc,0x28,0xc1]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvthf82ps256_mask:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvthf82ps %xmm0, %ymm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x36,0xc8]
+; X86-NEXT:    vmovaps %ymm1, %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfc,0x28,0xc1]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %cvt = call <8 x float> @llvm.x86.avx10.vcvthf82ps256(<16 x i8> %a)
+  %bst = bitcast i8 %mask to <8 x i1>
+  %ret = select <8 x i1> %bst, <8 x float> %cvt, <8 x float> %src
+  ret <8 x float> %ret
+}
+
+define <8 x float> @test_int_x86_avx10_vcvthf82ps256_maskz(<16 x i8> %a, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_vcvthf82ps256_maskz:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvthf82ps %xmm0, %ymm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xa9,0x36,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvthf82ps256_maskz:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvthf82ps %xmm0, %ymm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xa9,0x36,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %cvt = call <8 x float> @llvm.x86.avx10.vcvthf82ps256(<16 x i8> %a)
+  %bst = bitcast i8 %mask to <8 x i1>
+  %ret = select <8 x i1> %bst, <8 x float> %cvt, <8 x float> zeroinitializer
+  ret <8 x float> %ret
+}
+
+define <16 x float> @test_int_x86_avx10_vcvthf82ps512_mask(<16 x i8> %a, <16 x float> %src, i16 %mask) {
+; X64-LABEL: test_int_x86_avx10_vcvthf82ps512_mask:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvthf82ps %xmm0, %zmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x49,0x36,0xc8]
+; X64-NEXT:    vmovaps %zmm1, %zmm0 # encoding: [0x62,0xf1,0x7c,0x48,0x28,0xc1]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvthf82ps512_mask:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvthf82ps %xmm0, %zmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x49,0x36,0xc8]
+; X86-NEXT:    vmovaps %zmm1, %zmm0 # encoding: [0x62,0xf1,0x7c,0x48,0x28,0xc1]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %cvt = call <16 x float> @llvm.x86.avx10.vcvthf82ps512(<16 x i8> %a)
+  %bst = bitcast i16 %mask to <16 x i1>
+  %ret = select <16 x i1> %bst, <16 x float> %cvt, <16 x float> %src
+  ret <16 x float> %ret
+}
+
+define <16 x float> @test_int_x86_avx10_vcvthf82ps512_maskz(<16 x i8> %a, i16 %mask) {
+; X64-LABEL: test_int_x86_avx10_vcvthf82ps512_maskz:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvthf82ps %xmm0, %zmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xc9,0x36,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvthf82ps512_maskz:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvthf82ps %xmm0, %zmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xc9,0x36,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %cvt = call <16 x float> @llvm.x86.avx10.vcvthf82ps512(<16 x i8> %a)
+  %bst = bitcast i16 %mask to <16 x i1>
+  %ret = select <16 x i1> %bst, <16 x float> %cvt, <16 x float> zeroinitializer
+  ret <16 x float> %ret
+}
+
+define <4 x float> @test_int_x86_avx10_vcvthf82ps128_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvthf82ps128_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvthf82ps (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x36,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvthf82ps128_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvthf82ps (%eax), %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x36,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <16 x i8>, ptr %ptr_a
+  %ret = call <4 x float> @llvm.x86.avx10.vcvthf82ps128(<16 x i8> %a)
+  ret <4 x float> %ret
+}
+
+define <8 x float> @test_int_x86_avx10_vcvthf82ps256_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvthf82ps256_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvthf82ps (%rdi), %ymm0 # encoding: [0x62,0xf5,0x7c,0x28,0x36,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvthf82ps256_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvthf82ps (%eax), %ymm0 # encoding: [0x62,0xf5,0x7c,0x28,0x36,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <16 x i8>, ptr %ptr_a
+  %ret = call <8 x float> @llvm.x86.avx10.vcvthf82ps256(<16 x i8> %a)
+  ret <8 x float> %ret
+}
+
+define <16 x float> @test_int_x86_avx10_vcvthf82ps512_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvthf82ps512_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvthf82ps (%rdi), %zmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x36,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvthf82ps512_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvthf82ps (%eax), %zmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x36,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <16 x i8>, ptr %ptr_a
+  %ret = call <16 x float> @llvm.x86.avx10.vcvthf82ps512(<16 x i8> %a)
+  ret <16 x float> %ret
+}
diff --git a/llvm/test/CodeGen/X86/avx10_v2aux-cvt-ps-fp8-intrinsics.ll b/llvm/test/CodeGen/X86/avx10_v2aux-cvt-ps-fp8-intrinsics.ll
new file mode 100644
index 00000000000000..7ac6a0f36e9c75
--- /dev/null
+++ b/llvm/test/CodeGen/X86/avx10_v2aux-cvt-ps-fp8-intrinsics.ll
@@ -0,0 +1,833 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
+; RUN: llc < %s -verify-machineinstrs -mtriple=x86_64-unknown-unknown --show-mc-encoding -mattr=+avx10v2aux | FileCheck %s --check-prefixes=CHECK,X64
+; RUN: llc < %s -verify-machineinstrs -mtriple=i686-unknown-unknown --show-mc-encoding -mattr=+avx10v2aux | FileCheck %s --check-prefixes=CHECK,X86
+
+define <16 x i8> @test_int_x86_avx10_vcvtps2bf8128(<4 x float> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtps2bf8128:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtps2bf8 %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x39,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8128(<4 x float> %a)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtps2bf8256(<8 x float> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtps2bf8256:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtps2bf8 %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x39,0xc0]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8256(<8 x float> %a)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtps2bf8512(<16 x float> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtps2bf8512:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtps2bf8 %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x39,0xc0]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8512(<16 x float> %a)
+  ret <16 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.vcvtps2bf8128(<4 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtps2bf8256(<8 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtps2bf8512(<16 x float>)
+
+define <16 x i8> @test_int_x86_avx10_vcvtps2bf8128_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtps2bf8128_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtps2bf8x (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x39,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtps2bf8128_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtps2bf8x (%eax), %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x39,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <4 x float>, ptr %ptr_a
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8128(<4 x float> %a)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtps2bf8256_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtps2bf8256_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtps2bf8y (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x39,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtps2bf8256_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtps2bf8y (%eax), %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x39,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <8 x float>, ptr %ptr_a
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8256(<8 x float> %a)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtps2bf8512_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtps2bf8512_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtps2bf8 (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x39,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtps2bf8512_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtps2bf8 (%eax), %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x39,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <16 x float>, ptr %ptr_a
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8512(<16 x float> %a)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtps2bf8s128(<4 x float> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtps2bf8s128:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtps2bf8s %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x3b,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8s128(<4 x float> %a)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtps2bf8s256(<8 x float> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtps2bf8s256:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtps2bf8s %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x3b,0xc0]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8s256(<8 x float> %a)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtps2bf8s512(<16 x float> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtps2bf8s512:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtps2bf8s %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x3b,0xc0]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8s512(<16 x float> %a)
+  ret <16 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.vcvtps2bf8s128(<4 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtps2bf8s256(<8 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtps2bf8s512(<16 x float>)
+
+define <16 x i8> @test_int_x86_avx10_vcvtps2bf8s128_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtps2bf8s128_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtps2bf8sx (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x3b,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtps2bf8s128_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtps2bf8sx (%eax), %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x3b,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <4 x float>, ptr %ptr_a
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8s128(<4 x float> %a)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtps2bf8s256_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtps2bf8s256_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtps2bf8sy (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x3b,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtps2bf8s256_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtps2bf8sy (%eax), %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x3b,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <8 x float>, ptr %ptr_a
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8s256(<8 x float> %a)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtps2bf8s512_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtps2bf8s512_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtps2bf8s (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x3b,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtps2bf8s512_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtps2bf8s (%eax), %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x3b,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <16 x float>, ptr %ptr_a
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8s512(<16 x float> %a)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtps2hf8128(<4 x float> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtps2hf8128:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtps2hf8 %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x38,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8128(<4 x float> %a)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtps2hf8256(<8 x float> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtps2hf8256:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtps2hf8 %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x38,0xc0]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8256(<8 x float> %a)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtps2hf8512(<16 x float> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtps2hf8512:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtps2hf8 %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x38,0xc0]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8512(<16 x float> %a)
+  ret <16 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.vcvtps2hf8128(<4 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtps2hf8256(<8 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtps2hf8512(<16 x float>)
+
+define <16 x i8> @test_int_x86_avx10_vcvtps2hf8128_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtps2hf8128_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtps2hf8x (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x38,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtps2hf8128_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtps2hf8x (%eax), %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x38,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <4 x float>, ptr %ptr_a
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8128(<4 x float> %a)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtps2hf8256_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtps2hf8256_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtps2hf8y (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x38,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtps2hf8256_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtps2hf8y (%eax), %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x38,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <8 x float>, ptr %ptr_a
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8256(<8 x float> %a)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtps2hf8512_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtps2hf8512_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtps2hf8 (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x38,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtps2hf8512_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtps2hf8 (%eax), %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x38,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <16 x float>, ptr %ptr_a
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8512(<16 x float> %a)
+  ret <16 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.vcvtps2hf8s128(<4 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtps2hf8s256(<8 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtps2hf8s512(<16 x float>)
+
+define <16 x i8> @test_int_x86_avx10_vcvtps2hf8s128(<4 x float> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtps2hf8s128:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtps2hf8s %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x3a,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8s128(<4 x float> %a)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtps2hf8s256(<8 x float> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtps2hf8s256:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtps2hf8s %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x3a,0xc0]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8s256(<8 x float> %a)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtps2hf8s512(<16 x float> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtps2hf8s512:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtps2hf8s %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x3a,0xc0]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8s512(<16 x float> %a)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtps2hf8s128_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtps2hf8s128_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtps2hf8sx (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x3a,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtps2hf8s128_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtps2hf8sx (%eax), %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x3a,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <4 x float>, ptr %ptr_a
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8s128(<4 x float> %a)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtps2hf8s256_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtps2hf8s256_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtps2hf8sy (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x3a,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtps2hf8s256_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtps2hf8sy (%eax), %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x3a,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <8 x float>, ptr %ptr_a
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8s256(<8 x float> %a)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtps2hf8s512_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtps2hf8s512_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtps2hf8s (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x3a,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtps2hf8s512_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtps2hf8s (%eax), %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x3a,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <16 x float>, ptr %ptr_a
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8s512(<16 x float> %a)
+  ret <16 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.vcvtrops2hf8128(<4 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtrops2hf8256(<8 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtrops2hf8512(<16 x float>)
+
+define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8128(<4 x float> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtrops2hf8128:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtrops2hf8 %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7d,0x08,0x38,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8128(<4 x float> %a)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8256(<8 x float> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtrops2hf8256:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtrops2hf8 %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7d,0x28,0x38,0xc0]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8256(<8 x float> %a)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8512(<16 x float> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtrops2hf8512:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtrops2hf8 %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7d,0x48,0x38,0xc0]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8512(<16 x float> %a)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8128_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtrops2hf8128_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtrops2hf8x (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7d,0x08,0x38,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtrops2hf8128_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtrops2hf8x (%eax), %xmm0 # encoding: [0x62,0xf5,0x7d,0x08,0x38,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <4 x float>, ptr %ptr_a
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8128(<4 x float> %a)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8256_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtrops2hf8256_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtrops2hf8y (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7d,0x28,0x38,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtrops2hf8256_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtrops2hf8y (%eax), %xmm0 # encoding: [0x62,0xf5,0x7d,0x28,0x38,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <8 x float>, ptr %ptr_a
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8256(<8 x float> %a)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8512_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtrops2hf8512_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtrops2hf8 (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7d,0x48,0x38,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtrops2hf8512_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtrops2hf8 (%eax), %xmm0 # encoding: [0x62,0xf5,0x7d,0x48,0x38,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <16 x float>, ptr %ptr_a
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8512(<16 x float> %a)
+  ret <16 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s128(<4 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s256(<8 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s512(<16 x float>)
+
+define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8s128(<4 x float> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtrops2hf8s128:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtrops2hf8s %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7d,0x08,0x3a,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s128(<4 x float> %a)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8s256(<8 x float> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtrops2hf8s256:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtrops2hf8s %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7d,0x28,0x3a,0xc0]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s256(<8 x float> %a)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8s512(<16 x float> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtrops2hf8s512:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtrops2hf8s %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7d,0x48,0x3a,0xc0]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s512(<16 x float> %a)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8s128_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtrops2hf8s128_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtrops2hf8sx (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7d,0x08,0x3a,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtrops2hf8s128_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtrops2hf8sx (%eax), %xmm0 # encoding: [0x62,0xf5,0x7d,0x08,0x3a,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <4 x float>, ptr %ptr_a
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s128(<4 x float> %a)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8s256_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtrops2hf8s256_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtrops2hf8sy (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7d,0x28,0x3a,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtrops2hf8s256_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtrops2hf8sy (%eax), %xmm0 # encoding: [0x62,0xf5,0x7d,0x28,0x3a,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <8 x float>, ptr %ptr_a
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s256(<8 x float> %a)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8s512_mem(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vcvtrops2hf8s512_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtrops2hf8s (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7d,0x48,0x3a,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtrops2hf8s512_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtrops2hf8s (%eax), %xmm0 # encoding: [0x62,0xf5,0x7d,0x48,0x3a,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <16 x float>, ptr %ptr_a
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s512(<16 x float> %a)
+  ret <16 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8128(<4 x i32>, <4 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8256(<8 x i32>, <8 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8512(<16 x i32>, <16 x float>)
+
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8128(<4 x i32> %A, <4 x float> %b) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2bf8128:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbiasps2bf8 %xmm1, %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x39,0xc1]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8128(<4 x i32> %A, <4 x float> %b)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8256(<8 x i32> %A, <8 x float> %b) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2bf8256:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbiasps2bf8 %ymm1, %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x28,0x39,0xc1]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8256(<8 x i32> %A, <8 x float> %b)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8512(<16 x i32> %A, <16 x float> %b) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2bf8512:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbiasps2bf8 %zmm1, %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x39,0xc1]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8512(<16 x i32> %A, <16 x float> %b)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8128_mem(<4 x i32> %A, ptr %ptr_b) {
+; X64-LABEL: test_int_x86_avx10_vcvtbiasps2bf8128_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtbiasps2bf8 (%rdi), %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x39,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbiasps2bf8128_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2bf8 (%eax), %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x39,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %b = load <4 x float>, ptr %ptr_b
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8128(<4 x i32> %A, <4 x float> %b)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8256_mem(<8 x i32> %A, ptr %ptr_b) {
+; X64-LABEL: test_int_x86_avx10_vcvtbiasps2bf8256_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtbiasps2bf8 (%rdi), %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x28,0x39,0x07]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbiasps2bf8256_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2bf8 (%eax), %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x28,0x39,0x00]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %b = load <8 x float>, ptr %ptr_b
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8256(<8 x i32> %A, <8 x float> %b)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8512_mem(<16 x i32> %A, ptr %ptr_b) {
+; X64-LABEL: test_int_x86_avx10_vcvtbiasps2bf8512_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtbiasps2bf8 (%rdi), %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x39,0x07]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbiasps2bf8512_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2bf8 (%eax), %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x39,0x00]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %b = load <16 x float>, ptr %ptr_b
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8512(<16 x i32> %A, <16 x float> %b)
+  ret <16 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s128(<4 x i32>, <4 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s256(<8 x i32>, <8 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s512(<16 x i32>, <16 x float>)
+
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8s128(<4 x i32> %A, <4 x float> %b) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2bf8s128:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbiasps2bf8s %xmm1, %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x3b,0xc1]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s128(<4 x i32> %A, <4 x float> %b)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8s256(<8 x i32> %A, <8 x float> %b) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2bf8s256:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbiasps2bf8s %ymm1, %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x28,0x3b,0xc1]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s256(<8 x i32> %A, <8 x float> %b)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8s512(<16 x i32> %A, <16 x float> %b) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2bf8s512:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbiasps2bf8s %zmm1, %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x3b,0xc1]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s512(<16 x i32> %A, <16 x float> %b)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8s128_mem(<4 x i32> %A, ptr %ptr_b) {
+; X64-LABEL: test_int_x86_avx10_vcvtbiasps2bf8s128_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtbiasps2bf8s (%rdi), %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x3b,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbiasps2bf8s128_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2bf8s (%eax), %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x3b,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %b = load <4 x float>, ptr %ptr_b
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s128(<4 x i32> %A, <4 x float> %b)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8s256_mem(<8 x i32> %A, ptr %ptr_b) {
+; X64-LABEL: test_int_x86_avx10_vcvtbiasps2bf8s256_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtbiasps2bf8s (%rdi), %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x28,0x3b,0x07]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbiasps2bf8s256_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2bf8s (%eax), %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x28,0x3b,0x00]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %b = load <8 x float>, ptr %ptr_b
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s256(<8 x i32> %A, <8 x float> %b)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8s512_mem(<16 x i32> %A, ptr %ptr_b) {
+; X64-LABEL: test_int_x86_avx10_vcvtbiasps2bf8s512_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtbiasps2bf8s (%rdi), %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x3b,0x07]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbiasps2bf8s512_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2bf8s (%eax), %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x3b,0x00]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %b = load <16 x float>, ptr %ptr_b
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s512(<16 x i32> %A, <16 x float> %b)
+  ret <16 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8128(<4 x i32>, <4 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8256(<8 x i32>, <8 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8512(<16 x i32>, <16 x float>)
+
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8128(<4 x i32> %A, <4 x float> %b) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2hf8128:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbiasps2hf8 %xmm1, %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x38,0xc1]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8128(<4 x i32> %A, <4 x float> %b)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8256(<8 x i32> %A, <8 x float> %b) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2hf8256:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbiasps2hf8 %ymm1, %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x28,0x38,0xc1]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8256(<8 x i32> %A, <8 x float> %b)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8512(<16 x i32> %A, <16 x float> %b) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2hf8512:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbiasps2hf8 %zmm1, %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x38,0xc1]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8512(<16 x i32> %A, <16 x float> %b)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8128_mem(<4 x i32> %A, ptr %ptr_b) {
+; X64-LABEL: test_int_x86_avx10_vcvtbiasps2hf8128_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtbiasps2hf8 (%rdi), %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x38,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbiasps2hf8128_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2hf8 (%eax), %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x38,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %b = load <4 x float>, ptr %ptr_b
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8128(<4 x i32> %A, <4 x float> %b)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8256_mem(<8 x i32> %A, ptr %ptr_b) {
+; X64-LABEL: test_int_x86_avx10_vcvtbiasps2hf8256_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtbiasps2hf8 (%rdi), %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x28,0x38,0x07]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbiasps2hf8256_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2hf8 (%eax), %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x28,0x38,0x00]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %b = load <8 x float>, ptr %ptr_b
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8256(<8 x i32> %A, <8 x float> %b)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8512_mem(<16 x i32> %A, ptr %ptr_b) {
+; X64-LABEL: test_int_x86_avx10_vcvtbiasps2hf8512_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtbiasps2hf8 (%rdi), %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x38,0x07]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbiasps2hf8512_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2hf8 (%eax), %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x38,0x00]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %b = load <16 x float>, ptr %ptr_b
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8512(<16 x i32> %A, <16 x float> %b)
+  ret <16 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s128(<4 x i32>, <4 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s256(<8 x i32>, <8 x float>)
+declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s512(<16 x i32>, <16 x float>)
+
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8s128(<4 x i32> %A, <4 x float> %b) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2hf8s128:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbiasps2hf8s %xmm1, %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x3a,0xc1]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s128(<4 x i32> %A, <4 x float> %b)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8s256(<8 x i32> %A, <8 x float> %b) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2hf8s256:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbiasps2hf8s %ymm1, %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x28,0x3a,0xc1]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s256(<8 x i32> %A, <8 x float> %b)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8s512(<16 x i32> %A, <16 x float> %b) {
+; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2hf8s512:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vcvtbiasps2hf8s %zmm1, %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x3a,0xc1]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s512(<16 x i32> %A, <16 x float> %b)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8s128_mem(<4 x i32> %A, ptr %ptr_b) {
+; X64-LABEL: test_int_x86_avx10_vcvtbiasps2hf8s128_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtbiasps2hf8s (%rdi), %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x3a,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbiasps2hf8s128_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2hf8s (%eax), %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x3a,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %b = load <4 x float>, ptr %ptr_b
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s128(<4 x i32> %A, <4 x float> %b)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8s256_mem(<8 x i32> %A, ptr %ptr_b) {
+; X64-LABEL: test_int_x86_avx10_vcvtbiasps2hf8s256_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtbiasps2hf8s (%rdi), %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x28,0x3a,0x07]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbiasps2hf8s256_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2hf8s (%eax), %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x28,0x3a,0x00]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %b = load <8 x float>, ptr %ptr_b
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s256(<8 x i32> %A, <8 x float> %b)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8s512_mem(<16 x i32> %A, ptr %ptr_b) {
+; X64-LABEL: test_int_x86_avx10_vcvtbiasps2hf8s512_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    vcvtbiasps2hf8s (%rdi), %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x3a,0x07]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbiasps2hf8s512_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2hf8s (%eax), %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x3a,0x00]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %b = load <16 x float>, ptr %ptr_b
+  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s512(<16 x i32> %A, <16 x float> %b)
+  ret <16 x i8> %ret
+}
+
+declare <4 x float> @llvm.x86.avx10.vcvtbf82ps128(<16 x i8>)
+declare <8 x float> @llvm.x86.avx10.vcvtbf82ps256(<16 x i8>)
+declare <16 x float> @llvm.x86.avx10.vcvtbf82ps512(<16 x i8>)
diff --git a/llvm/test/CodeGen/X86/avx10_v2aux-intrinsics.ll b/llvm/test/CodeGen/X86/avx10_v2aux-intrinsics.ll
deleted file mode 100644
index a52ce4355b7dd1..00000000000000
--- a/llvm/test/CodeGen/X86/avx10_v2aux-intrinsics.ll
+++ /dev/null
@@ -1,3909 +0,0 @@
-; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
-; RUN: llc < %s -verify-machineinstrs -mtriple=x86_64-unknown-unknown --show-mc-encoding -mattr=+avx10v2aux | FileCheck %s --check-prefixes=CHECK,X64
-; RUN: llc < %s -verify-machineinstrs -mtriple=i686-unknown-unknown --show-mc-encoding -mattr=+avx10v2aux | FileCheck %s --check-prefixes=CHECK,X86
-
-define <16 x i8> @test_int_x86_avx10_vcvtps2bf8128(<4 x float> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtps2bf8128:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtps2bf8 %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x39,0xc0]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8128(<4 x float> %a)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtps2bf8256(<8 x float> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtps2bf8256:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtps2bf8 %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x39,0xc0]
-; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8256(<8 x float> %a)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtps2bf8512(<16 x float> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtps2bf8512:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtps2bf8 %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x39,0xc0]
-; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8512(<16 x float> %a)
-  ret <16 x i8> %ret
-}
-
-declare <16 x i8> @llvm.x86.avx10.vcvtps2bf8128(<4 x float>)
-declare <16 x i8> @llvm.x86.avx10.vcvtps2bf8256(<8 x float>)
-declare <16 x i8> @llvm.x86.avx10.vcvtps2bf8512(<16 x float>)
-
-define <16 x i8> @test_int_x86_avx10_vcvtps2bf8128_mem(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtps2bf8128_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    vcvtps2bf8x (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x39,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvtps2bf8128_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vcvtps2bf8x (%eax), %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x39,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %a = load <4 x float>, ptr %ptr_a
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8128(<4 x float> %a)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtps2bf8256_mem(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtps2bf8256_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    vcvtps2bf8y (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x39,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvtps2bf8256_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vcvtps2bf8y (%eax), %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x39,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %a = load <8 x float>, ptr %ptr_a
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8256(<8 x float> %a)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtps2bf8512_mem(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtps2bf8512_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    vcvtps2bf8 (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x39,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvtps2bf8512_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vcvtps2bf8 (%eax), %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x39,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %a = load <16 x float>, ptr %ptr_a
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8512(<16 x float> %a)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtps2bf8s128(<4 x float> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtps2bf8s128:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtps2bf8s %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x3b,0xc0]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8s128(<4 x float> %a)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtps2bf8s256(<8 x float> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtps2bf8s256:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtps2bf8s %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x3b,0xc0]
-; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8s256(<8 x float> %a)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtps2bf8s512(<16 x float> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtps2bf8s512:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtps2bf8s %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x3b,0xc0]
-; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8s512(<16 x float> %a)
-  ret <16 x i8> %ret
-}
-
-declare <16 x i8> @llvm.x86.avx10.vcvtps2bf8s128(<4 x float>)
-declare <16 x i8> @llvm.x86.avx10.vcvtps2bf8s256(<8 x float>)
-declare <16 x i8> @llvm.x86.avx10.vcvtps2bf8s512(<16 x float>)
-
-define <16 x i8> @test_int_x86_avx10_vcvtps2bf8s128_mem(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtps2bf8s128_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    vcvtps2bf8sx (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x3b,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvtps2bf8s128_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vcvtps2bf8sx (%eax), %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x3b,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %a = load <4 x float>, ptr %ptr_a
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8s128(<4 x float> %a)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtps2bf8s256_mem(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtps2bf8s256_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    vcvtps2bf8sy (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x3b,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvtps2bf8s256_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vcvtps2bf8sy (%eax), %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x3b,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %a = load <8 x float>, ptr %ptr_a
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8s256(<8 x float> %a)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtps2bf8s512_mem(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtps2bf8s512_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    vcvtps2bf8s (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x3b,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvtps2bf8s512_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vcvtps2bf8s (%eax), %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x3b,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %a = load <16 x float>, ptr %ptr_a
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8s512(<16 x float> %a)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtps2hf8128(<4 x float> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtps2hf8128:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtps2hf8 %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x38,0xc0]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8128(<4 x float> %a)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtps2hf8256(<8 x float> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtps2hf8256:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtps2hf8 %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x38,0xc0]
-; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8256(<8 x float> %a)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtps2hf8512(<16 x float> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtps2hf8512:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtps2hf8 %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x38,0xc0]
-; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8512(<16 x float> %a)
-  ret <16 x i8> %ret
-}
-
-declare <16 x i8> @llvm.x86.avx10.vcvtps2hf8128(<4 x float>)
-declare <16 x i8> @llvm.x86.avx10.vcvtps2hf8256(<8 x float>)
-declare <16 x i8> @llvm.x86.avx10.vcvtps2hf8512(<16 x float>)
-
-define <16 x i8> @test_int_x86_avx10_vcvtps2hf8128_mem(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtps2hf8128_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    vcvtps2hf8x (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x38,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvtps2hf8128_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vcvtps2hf8x (%eax), %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x38,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %a = load <4 x float>, ptr %ptr_a
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8128(<4 x float> %a)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtps2hf8256_mem(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtps2hf8256_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    vcvtps2hf8y (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x38,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvtps2hf8256_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vcvtps2hf8y (%eax), %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x38,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %a = load <8 x float>, ptr %ptr_a
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8256(<8 x float> %a)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtps2hf8512_mem(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtps2hf8512_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    vcvtps2hf8 (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x38,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvtps2hf8512_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vcvtps2hf8 (%eax), %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x38,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %a = load <16 x float>, ptr %ptr_a
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8512(<16 x float> %a)
-  ret <16 x i8> %ret
-}
-
-declare <16 x i8> @llvm.x86.avx10.vcvtps2hf8s128(<4 x float>)
-declare <16 x i8> @llvm.x86.avx10.vcvtps2hf8s256(<8 x float>)
-declare <16 x i8> @llvm.x86.avx10.vcvtps2hf8s512(<16 x float>)
-
-define <16 x i8> @test_int_x86_avx10_vcvtps2hf8s128(<4 x float> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtps2hf8s128:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtps2hf8s %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x3a,0xc0]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8s128(<4 x float> %a)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtps2hf8s256(<8 x float> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtps2hf8s256:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtps2hf8s %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x3a,0xc0]
-; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8s256(<8 x float> %a)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtps2hf8s512(<16 x float> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtps2hf8s512:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtps2hf8s %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x3a,0xc0]
-; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8s512(<16 x float> %a)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtps2hf8s128_mem(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtps2hf8s128_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    vcvtps2hf8sx (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x3a,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvtps2hf8s128_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vcvtps2hf8sx (%eax), %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x3a,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %a = load <4 x float>, ptr %ptr_a
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8s128(<4 x float> %a)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtps2hf8s256_mem(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtps2hf8s256_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    vcvtps2hf8sy (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x3a,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvtps2hf8s256_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vcvtps2hf8sy (%eax), %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x3a,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %a = load <8 x float>, ptr %ptr_a
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8s256(<8 x float> %a)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtps2hf8s512_mem(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtps2hf8s512_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    vcvtps2hf8s (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x3a,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvtps2hf8s512_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vcvtps2hf8s (%eax), %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x3a,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %a = load <16 x float>, ptr %ptr_a
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8s512(<16 x float> %a)
-  ret <16 x i8> %ret
-}
-
-declare <16 x i8> @llvm.x86.avx10.vcvtrops2hf8128(<4 x float>)
-declare <16 x i8> @llvm.x86.avx10.vcvtrops2hf8256(<8 x float>)
-declare <16 x i8> @llvm.x86.avx10.vcvtrops2hf8512(<16 x float>)
-
-define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8128(<4 x float> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtrops2hf8128:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtrops2hf8 %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7d,0x08,0x38,0xc0]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8128(<4 x float> %a)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8256(<8 x float> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtrops2hf8256:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtrops2hf8 %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7d,0x28,0x38,0xc0]
-; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8256(<8 x float> %a)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8512(<16 x float> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtrops2hf8512:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtrops2hf8 %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7d,0x48,0x38,0xc0]
-; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8512(<16 x float> %a)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8128_mem(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtrops2hf8128_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    vcvtrops2hf8x (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7d,0x08,0x38,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvtrops2hf8128_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vcvtrops2hf8x (%eax), %xmm0 # encoding: [0x62,0xf5,0x7d,0x08,0x38,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %a = load <4 x float>, ptr %ptr_a
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8128(<4 x float> %a)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8256_mem(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtrops2hf8256_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    vcvtrops2hf8y (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7d,0x28,0x38,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvtrops2hf8256_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vcvtrops2hf8y (%eax), %xmm0 # encoding: [0x62,0xf5,0x7d,0x28,0x38,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %a = load <8 x float>, ptr %ptr_a
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8256(<8 x float> %a)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8512_mem(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtrops2hf8512_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    vcvtrops2hf8 (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7d,0x48,0x38,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvtrops2hf8512_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vcvtrops2hf8 (%eax), %xmm0 # encoding: [0x62,0xf5,0x7d,0x48,0x38,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %a = load <16 x float>, ptr %ptr_a
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8512(<16 x float> %a)
-  ret <16 x i8> %ret
-}
-
-declare <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s128(<4 x float>)
-declare <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s256(<8 x float>)
-declare <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s512(<16 x float>)
-
-define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8s128(<4 x float> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtrops2hf8s128:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtrops2hf8s %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7d,0x08,0x3a,0xc0]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s128(<4 x float> %a)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8s256(<8 x float> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtrops2hf8s256:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtrops2hf8s %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7d,0x28,0x3a,0xc0]
-; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s256(<8 x float> %a)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8s512(<16 x float> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtrops2hf8s512:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtrops2hf8s %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7d,0x48,0x3a,0xc0]
-; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s512(<16 x float> %a)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8s128_mem(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtrops2hf8s128_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    vcvtrops2hf8sx (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7d,0x08,0x3a,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvtrops2hf8s128_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vcvtrops2hf8sx (%eax), %xmm0 # encoding: [0x62,0xf5,0x7d,0x08,0x3a,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %a = load <4 x float>, ptr %ptr_a
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s128(<4 x float> %a)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8s256_mem(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtrops2hf8s256_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    vcvtrops2hf8sy (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7d,0x28,0x3a,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvtrops2hf8s256_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vcvtrops2hf8sy (%eax), %xmm0 # encoding: [0x62,0xf5,0x7d,0x28,0x3a,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %a = load <8 x float>, ptr %ptr_a
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s256(<8 x float> %a)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8s512_mem(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtrops2hf8s512_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    vcvtrops2hf8s (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7d,0x48,0x3a,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvtrops2hf8s512_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vcvtrops2hf8s (%eax), %xmm0 # encoding: [0x62,0xf5,0x7d,0x48,0x3a,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %a = load <16 x float>, ptr %ptr_a
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s512(<16 x float> %a)
-  ret <16 x i8> %ret
-}
-
-declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8128(<4 x i32>, <4 x float>)
-declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8256(<8 x i32>, <8 x float>)
-declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8512(<16 x i32>, <16 x float>)
-
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8128(<4 x i32> %A, <4 x float> %b) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2bf8128:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtbiasps2bf8 %xmm1, %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x39,0xc1]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8128(<4 x i32> %A, <4 x float> %b)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8256(<8 x i32> %A, <8 x float> %b) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2bf8256:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtbiasps2bf8 %ymm1, %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x28,0x39,0xc1]
-; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8256(<8 x i32> %A, <8 x float> %b)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8512(<16 x i32> %A, <16 x float> %b) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2bf8512:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtbiasps2bf8 %zmm1, %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x39,0xc1]
-; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8512(<16 x i32> %A, <16 x float> %b)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8128_mem(<4 x i32> %A, ptr %ptr_b) {
-; X64-LABEL: test_int_x86_avx10_vcvtbiasps2bf8128_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    vcvtbiasps2bf8 (%rdi), %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x39,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvtbiasps2bf8128_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vcvtbiasps2bf8 (%eax), %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x39,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %b = load <4 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8128(<4 x i32> %A, <4 x float> %b)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8256_mem(<8 x i32> %A, ptr %ptr_b) {
-; X64-LABEL: test_int_x86_avx10_vcvtbiasps2bf8256_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    vcvtbiasps2bf8 (%rdi), %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x28,0x39,0x07]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvtbiasps2bf8256_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vcvtbiasps2bf8 (%eax), %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x28,0x39,0x00]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %b = load <8 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8256(<8 x i32> %A, <8 x float> %b)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8512_mem(<16 x i32> %A, ptr %ptr_b) {
-; X64-LABEL: test_int_x86_avx10_vcvtbiasps2bf8512_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    vcvtbiasps2bf8 (%rdi), %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x39,0x07]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvtbiasps2bf8512_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vcvtbiasps2bf8 (%eax), %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x39,0x00]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %b = load <16 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8512(<16 x i32> %A, <16 x float> %b)
-  ret <16 x i8> %ret
-}
-
-declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s128(<4 x i32>, <4 x float>)
-declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s256(<8 x i32>, <8 x float>)
-declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s512(<16 x i32>, <16 x float>)
-
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8s128(<4 x i32> %A, <4 x float> %b) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2bf8s128:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtbiasps2bf8s %xmm1, %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x3b,0xc1]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s128(<4 x i32> %A, <4 x float> %b)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8s256(<8 x i32> %A, <8 x float> %b) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2bf8s256:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtbiasps2bf8s %ymm1, %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x28,0x3b,0xc1]
-; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s256(<8 x i32> %A, <8 x float> %b)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8s512(<16 x i32> %A, <16 x float> %b) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2bf8s512:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtbiasps2bf8s %zmm1, %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x3b,0xc1]
-; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s512(<16 x i32> %A, <16 x float> %b)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8s128_mem(<4 x i32> %A, ptr %ptr_b) {
-; X64-LABEL: test_int_x86_avx10_vcvtbiasps2bf8s128_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    vcvtbiasps2bf8s (%rdi), %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x3b,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvtbiasps2bf8s128_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vcvtbiasps2bf8s (%eax), %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x3b,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %b = load <4 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s128(<4 x i32> %A, <4 x float> %b)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8s256_mem(<8 x i32> %A, ptr %ptr_b) {
-; X64-LABEL: test_int_x86_avx10_vcvtbiasps2bf8s256_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    vcvtbiasps2bf8s (%rdi), %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x28,0x3b,0x07]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvtbiasps2bf8s256_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vcvtbiasps2bf8s (%eax), %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x28,0x3b,0x00]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %b = load <8 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s256(<8 x i32> %A, <8 x float> %b)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2bf8s512_mem(<16 x i32> %A, ptr %ptr_b) {
-; X64-LABEL: test_int_x86_avx10_vcvtbiasps2bf8s512_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    vcvtbiasps2bf8s (%rdi), %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x3b,0x07]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvtbiasps2bf8s512_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vcvtbiasps2bf8s (%eax), %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x3b,0x00]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %b = load <16 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s512(<16 x i32> %A, <16 x float> %b)
-  ret <16 x i8> %ret
-}
-
-declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8128(<4 x i32>, <4 x float>)
-declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8256(<8 x i32>, <8 x float>)
-declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8512(<16 x i32>, <16 x float>)
-
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8128(<4 x i32> %A, <4 x float> %b) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2hf8128:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtbiasps2hf8 %xmm1, %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x38,0xc1]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8128(<4 x i32> %A, <4 x float> %b)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8256(<8 x i32> %A, <8 x float> %b) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2hf8256:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtbiasps2hf8 %ymm1, %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x28,0x38,0xc1]
-; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8256(<8 x i32> %A, <8 x float> %b)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8512(<16 x i32> %A, <16 x float> %b) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2hf8512:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtbiasps2hf8 %zmm1, %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x38,0xc1]
-; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8512(<16 x i32> %A, <16 x float> %b)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8128_mem(<4 x i32> %A, ptr %ptr_b) {
-; X64-LABEL: test_int_x86_avx10_vcvtbiasps2hf8128_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    vcvtbiasps2hf8 (%rdi), %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x38,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvtbiasps2hf8128_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vcvtbiasps2hf8 (%eax), %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x38,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %b = load <4 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8128(<4 x i32> %A, <4 x float> %b)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8256_mem(<8 x i32> %A, ptr %ptr_b) {
-; X64-LABEL: test_int_x86_avx10_vcvtbiasps2hf8256_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    vcvtbiasps2hf8 (%rdi), %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x28,0x38,0x07]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvtbiasps2hf8256_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vcvtbiasps2hf8 (%eax), %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x28,0x38,0x00]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %b = load <8 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8256(<8 x i32> %A, <8 x float> %b)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8512_mem(<16 x i32> %A, ptr %ptr_b) {
-; X64-LABEL: test_int_x86_avx10_vcvtbiasps2hf8512_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    vcvtbiasps2hf8 (%rdi), %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x38,0x07]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvtbiasps2hf8512_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vcvtbiasps2hf8 (%eax), %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x38,0x00]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %b = load <16 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8512(<16 x i32> %A, <16 x float> %b)
-  ret <16 x i8> %ret
-}
-
-declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s128(<4 x i32>, <4 x float>)
-declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s256(<8 x i32>, <8 x float>)
-declare <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s512(<16 x i32>, <16 x float>)
-
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8s128(<4 x i32> %A, <4 x float> %b) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2hf8s128:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtbiasps2hf8s %xmm1, %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x3a,0xc1]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s128(<4 x i32> %A, <4 x float> %b)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8s256(<8 x i32> %A, <8 x float> %b) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2hf8s256:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtbiasps2hf8s %ymm1, %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x28,0x3a,0xc1]
-; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s256(<8 x i32> %A, <8 x float> %b)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8s512(<16 x i32> %A, <16 x float> %b) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbiasps2hf8s512:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtbiasps2hf8s %zmm1, %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x3a,0xc1]
-; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s512(<16 x i32> %A, <16 x float> %b)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8s128_mem(<4 x i32> %A, ptr %ptr_b) {
-; X64-LABEL: test_int_x86_avx10_vcvtbiasps2hf8s128_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    vcvtbiasps2hf8s (%rdi), %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x3a,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvtbiasps2hf8s128_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vcvtbiasps2hf8s (%eax), %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x3a,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %b = load <4 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s128(<4 x i32> %A, <4 x float> %b)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8s256_mem(<8 x i32> %A, ptr %ptr_b) {
-; X64-LABEL: test_int_x86_avx10_vcvtbiasps2hf8s256_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    vcvtbiasps2hf8s (%rdi), %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x28,0x3a,0x07]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvtbiasps2hf8s256_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vcvtbiasps2hf8s (%eax), %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x28,0x3a,0x00]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %b = load <8 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s256(<8 x i32> %A, <8 x float> %b)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtbiasps2hf8s512_mem(<16 x i32> %A, ptr %ptr_b) {
-; X64-LABEL: test_int_x86_avx10_vcvtbiasps2hf8s512_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    vcvtbiasps2hf8s (%rdi), %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x3a,0x07]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvtbiasps2hf8s512_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vcvtbiasps2hf8s (%eax), %zmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x3a,0x00]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %b = load <16 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s512(<16 x i32> %A, <16 x float> %b)
-  ret <16 x i8> %ret
-}
-
-declare <4 x float> @llvm.x86.avx10.vcvtbf82ps128(<16 x i8>)
-declare <8 x float> @llvm.x86.avx10.vcvtbf82ps256(<16 x i8>)
-declare <16 x float> @llvm.x86.avx10.vcvtbf82ps512(<16 x i8>)
-
-define <4 x float> @test_int_x86_avx10_vcvtbf82ps128(<16 x i8> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbf82ps128:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtbf82ps %xmm0, %xmm0 # encoding: [0x62,0xf5,0xfc,0x08,0x36,0xc0]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <4 x float> @llvm.x86.avx10.vcvtbf82ps128(<16 x i8> %a)
-  ret <4 x float> %ret
-}
-
-define <8 x float> @test_int_x86_avx10_vcvtbf82ps256(<16 x i8> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbf82ps256:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtbf82ps %xmm0, %ymm0 # encoding: [0x62,0xf5,0xfc,0x28,0x36,0xc0]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <8 x float> @llvm.x86.avx10.vcvtbf82ps256(<16 x i8> %a)
-  ret <8 x float> %ret
-}
-
-define <16 x float> @test_int_x86_avx10_vcvtbf82ps512(<16 x i8> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbf82ps512:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtbf82ps %xmm0, %zmm0 # encoding: [0x62,0xf5,0xfc,0x48,0x36,0xc0]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x float> @llvm.x86.avx10.vcvtbf82ps512(<16 x i8> %a)
-  ret <16 x float> %ret
-}
-
-define <4 x float> @test_int_x86_avx10_vcvtbf82ps128_mask(<16 x i8> %a, <4 x float> %src, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_vcvtbf82ps128_mask:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtbf82ps %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0xfc,0x09,0x36,0xc8]
-; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvtbf82ps128_mask:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtbf82ps %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0xfc,0x09,0x36,0xc8]
-; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %cvt = call <4 x float> @llvm.x86.avx10.vcvtbf82ps128(<16 x i8> %a)
-  %bst = bitcast i8 %mask to <8 x i1>
-  %ext = shufflevector <8 x i1> %bst, <8 x i1> poison, <4 x i32> <i32 0, i32 1, i32 2, i32 3>
-  %ret = select <4 x i1> %ext, <4 x float> %cvt, <4 x float> %src
-  ret <4 x float> %ret
-}
-
-define <4 x float> @test_int_x86_avx10_vcvtbf82ps128_maskz(<16 x i8> %a, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_vcvtbf82ps128_maskz:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtbf82ps %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0xfc,0x89,0x36,0xc0]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvtbf82ps128_maskz:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtbf82ps %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0xfc,0x89,0x36,0xc0]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %cvt = call <4 x float> @llvm.x86.avx10.vcvtbf82ps128(<16 x i8> %a)
-  %bst = bitcast i8 %mask to <8 x i1>
-  %ext = shufflevector <8 x i1> %bst, <8 x i1> poison, <4 x i32> <i32 0, i32 1, i32 2, i32 3>
-  %ret = select <4 x i1> %ext, <4 x float> %cvt, <4 x float> zeroinitializer
-  ret <4 x float> %ret
-}
-
-define <8 x float> @test_int_x86_avx10_vcvtbf82ps256_mask(<16 x i8> %a, <8 x float> %src, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_vcvtbf82ps256_mask:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtbf82ps %xmm0, %ymm1 {%k1} # encoding: [0x62,0xf5,0xfc,0x29,0x36,0xc8]
-; X64-NEXT:    vmovaps %ymm1, %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfc,0x28,0xc1]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvtbf82ps256_mask:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtbf82ps %xmm0, %ymm1 {%k1} # encoding: [0x62,0xf5,0xfc,0x29,0x36,0xc8]
-; X86-NEXT:    vmovaps %ymm1, %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfc,0x28,0xc1]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %cvt = call <8 x float> @llvm.x86.avx10.vcvtbf82ps256(<16 x i8> %a)
-  %bst = bitcast i8 %mask to <8 x i1>
-  %ret = select <8 x i1> %bst, <8 x float> %cvt, <8 x float> %src
-  ret <8 x float> %ret
-}
-
-define <8 x float> @test_int_x86_avx10_vcvtbf82ps256_maskz(<16 x i8> %a, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_vcvtbf82ps256_maskz:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtbf82ps %xmm0, %ymm0 {%k1} {z} # encoding: [0x62,0xf5,0xfc,0xa9,0x36,0xc0]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvtbf82ps256_maskz:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtbf82ps %xmm0, %ymm0 {%k1} {z} # encoding: [0x62,0xf5,0xfc,0xa9,0x36,0xc0]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %cvt = call <8 x float> @llvm.x86.avx10.vcvtbf82ps256(<16 x i8> %a)
-  %bst = bitcast i8 %mask to <8 x i1>
-  %ret = select <8 x i1> %bst, <8 x float> %cvt, <8 x float> zeroinitializer
-  ret <8 x float> %ret
-}
-
-define <16 x float> @test_int_x86_avx10_vcvtbf82ps512_mask(<16 x i8> %a, <16 x float> %src, i16 %mask) {
-; X64-LABEL: test_int_x86_avx10_vcvtbf82ps512_mask:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtbf82ps %xmm0, %zmm1 {%k1} # encoding: [0x62,0xf5,0xfc,0x49,0x36,0xc8]
-; X64-NEXT:    vmovaps %zmm1, %zmm0 # encoding: [0x62,0xf1,0x7c,0x48,0x28,0xc1]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvtbf82ps512_mask:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtbf82ps %xmm0, %zmm1 {%k1} # encoding: [0x62,0xf5,0xfc,0x49,0x36,0xc8]
-; X86-NEXT:    vmovaps %zmm1, %zmm0 # encoding: [0x62,0xf1,0x7c,0x48,0x28,0xc1]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %cvt = call <16 x float> @llvm.x86.avx10.vcvtbf82ps512(<16 x i8> %a)
-  %bst = bitcast i16 %mask to <16 x i1>
-  %ret = select <16 x i1> %bst, <16 x float> %cvt, <16 x float> %src
-  ret <16 x float> %ret
-}
-
-define <16 x float> @test_int_x86_avx10_vcvtbf82ps512_maskz(<16 x i8> %a, i16 %mask) {
-; X64-LABEL: test_int_x86_avx10_vcvtbf82ps512_maskz:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtbf82ps %xmm0, %zmm0 {%k1} {z} # encoding: [0x62,0xf5,0xfc,0xc9,0x36,0xc0]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvtbf82ps512_maskz:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtbf82ps %xmm0, %zmm0 {%k1} {z} # encoding: [0x62,0xf5,0xfc,0xc9,0x36,0xc0]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %cvt = call <16 x float> @llvm.x86.avx10.vcvtbf82ps512(<16 x i8> %a)
-  %bst = bitcast i16 %mask to <16 x i1>
-  %ret = select <16 x i1> %bst, <16 x float> %cvt, <16 x float> zeroinitializer
-  ret <16 x float> %ret
-}
-
-define <4 x float> @test_int_x86_avx10_vcvtbf82ps128_mem(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtbf82ps128_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    vcvtbf82ps (%rdi), %xmm0 # encoding: [0x62,0xf5,0xfc,0x08,0x36,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvtbf82ps128_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vcvtbf82ps (%eax), %xmm0 # encoding: [0x62,0xf5,0xfc,0x08,0x36,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %a = load <16 x i8>, ptr %ptr_a
-  %ret = call <4 x float> @llvm.x86.avx10.vcvtbf82ps128(<16 x i8> %a)
-  ret <4 x float> %ret
-}
-
-define <8 x float> @test_int_x86_avx10_vcvtbf82ps256_mem(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtbf82ps256_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    vcvtbf82ps (%rdi), %ymm0 # encoding: [0x62,0xf5,0xfc,0x28,0x36,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvtbf82ps256_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vcvtbf82ps (%eax), %ymm0 # encoding: [0x62,0xf5,0xfc,0x28,0x36,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %a = load <16 x i8>, ptr %ptr_a
-  %ret = call <8 x float> @llvm.x86.avx10.vcvtbf82ps256(<16 x i8> %a)
-  ret <8 x float> %ret
-}
-
-define <16 x float> @test_int_x86_avx10_vcvtbf82ps512_mem(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtbf82ps512_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    vcvtbf82ps (%rdi), %zmm0 # encoding: [0x62,0xf5,0xfc,0x48,0x36,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvtbf82ps512_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vcvtbf82ps (%eax), %zmm0 # encoding: [0x62,0xf5,0xfc,0x48,0x36,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %a = load <16 x i8>, ptr %ptr_a
-  %ret = call <16 x float> @llvm.x86.avx10.vcvtbf82ps512(<16 x i8> %a)
-  ret <16 x float> %ret
-}
-
-declare <4 x float> @llvm.x86.avx10.vcvthf82ps128(<16 x i8>)
-declare <8 x float> @llvm.x86.avx10.vcvthf82ps256(<16 x i8>)
-declare <16 x float> @llvm.x86.avx10.vcvthf82ps512(<16 x i8>)
-
-define <4 x float> @test_int_x86_avx10_vcvthf82ps128(<16 x i8> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvthf82ps128:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvthf82ps %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x36,0xc0]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <4 x float> @llvm.x86.avx10.vcvthf82ps128(<16 x i8> %a)
-  ret <4 x float> %ret
-}
-
-define <8 x float> @test_int_x86_avx10_vcvthf82ps256(<16 x i8> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvthf82ps256:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvthf82ps %xmm0, %ymm0 # encoding: [0x62,0xf5,0x7c,0x28,0x36,0xc0]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <8 x float> @llvm.x86.avx10.vcvthf82ps256(<16 x i8> %a)
-  ret <8 x float> %ret
-}
-
-define <16 x float> @test_int_x86_avx10_vcvthf82ps512(<16 x i8> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvthf82ps512:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvthf82ps %xmm0, %zmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x36,0xc0]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x float> @llvm.x86.avx10.vcvthf82ps512(<16 x i8> %a)
-  ret <16 x float> %ret
-}
-
-define <4 x float> @test_int_x86_avx10_vcvthf82ps128_mask(<16 x i8> %a, <4 x float> %src, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_vcvthf82ps128_mask:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvthf82ps %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x36,0xc8]
-; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvthf82ps128_mask:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvthf82ps %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x36,0xc8]
-; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %cvt = call <4 x float> @llvm.x86.avx10.vcvthf82ps128(<16 x i8> %a)
-  %bst = bitcast i8 %mask to <8 x i1>
-  %ext = shufflevector <8 x i1> %bst, <8 x i1> poison, <4 x i32> <i32 0, i32 1, i32 2, i32 3>
-  %ret = select <4 x i1> %ext, <4 x float> %cvt, <4 x float> %src
-  ret <4 x float> %ret
-}
-
-define <4 x float> @test_int_x86_avx10_vcvthf82ps128_maskz(<16 x i8> %a, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_vcvthf82ps128_maskz:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvthf82ps %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x36,0xc0]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvthf82ps128_maskz:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvthf82ps %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x36,0xc0]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %cvt = call <4 x float> @llvm.x86.avx10.vcvthf82ps128(<16 x i8> %a)
-  %bst = bitcast i8 %mask to <8 x i1>
-  %ext = shufflevector <8 x i1> %bst, <8 x i1> poison, <4 x i32> <i32 0, i32 1, i32 2, i32 3>
-  %ret = select <4 x i1> %ext, <4 x float> %cvt, <4 x float> zeroinitializer
-  ret <4 x float> %ret
-}
-
-define <8 x float> @test_int_x86_avx10_vcvthf82ps256_mask(<16 x i8> %a, <8 x float> %src, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_vcvthf82ps256_mask:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvthf82ps %xmm0, %ymm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x36,0xc8]
-; X64-NEXT:    vmovaps %ymm1, %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfc,0x28,0xc1]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvthf82ps256_mask:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvthf82ps %xmm0, %ymm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x36,0xc8]
-; X86-NEXT:    vmovaps %ymm1, %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfc,0x28,0xc1]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %cvt = call <8 x float> @llvm.x86.avx10.vcvthf82ps256(<16 x i8> %a)
-  %bst = bitcast i8 %mask to <8 x i1>
-  %ret = select <8 x i1> %bst, <8 x float> %cvt, <8 x float> %src
-  ret <8 x float> %ret
-}
-
-define <8 x float> @test_int_x86_avx10_vcvthf82ps256_maskz(<16 x i8> %a, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_vcvthf82ps256_maskz:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvthf82ps %xmm0, %ymm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xa9,0x36,0xc0]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvthf82ps256_maskz:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvthf82ps %xmm0, %ymm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xa9,0x36,0xc0]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %cvt = call <8 x float> @llvm.x86.avx10.vcvthf82ps256(<16 x i8> %a)
-  %bst = bitcast i8 %mask to <8 x i1>
-  %ret = select <8 x i1> %bst, <8 x float> %cvt, <8 x float> zeroinitializer
-  ret <8 x float> %ret
-}
-
-define <16 x float> @test_int_x86_avx10_vcvthf82ps512_mask(<16 x i8> %a, <16 x float> %src, i16 %mask) {
-; X64-LABEL: test_int_x86_avx10_vcvthf82ps512_mask:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvthf82ps %xmm0, %zmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x49,0x36,0xc8]
-; X64-NEXT:    vmovaps %zmm1, %zmm0 # encoding: [0x62,0xf1,0x7c,0x48,0x28,0xc1]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvthf82ps512_mask:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvthf82ps %xmm0, %zmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x49,0x36,0xc8]
-; X86-NEXT:    vmovaps %zmm1, %zmm0 # encoding: [0x62,0xf1,0x7c,0x48,0x28,0xc1]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %cvt = call <16 x float> @llvm.x86.avx10.vcvthf82ps512(<16 x i8> %a)
-  %bst = bitcast i16 %mask to <16 x i1>
-  %ret = select <16 x i1> %bst, <16 x float> %cvt, <16 x float> %src
-  ret <16 x float> %ret
-}
-
-define <16 x float> @test_int_x86_avx10_vcvthf82ps512_maskz(<16 x i8> %a, i16 %mask) {
-; X64-LABEL: test_int_x86_avx10_vcvthf82ps512_maskz:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvthf82ps %xmm0, %zmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xc9,0x36,0xc0]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvthf82ps512_maskz:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvthf82ps %xmm0, %zmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xc9,0x36,0xc0]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %cvt = call <16 x float> @llvm.x86.avx10.vcvthf82ps512(<16 x i8> %a)
-  %bst = bitcast i16 %mask to <16 x i1>
-  %ret = select <16 x i1> %bst, <16 x float> %cvt, <16 x float> zeroinitializer
-  ret <16 x float> %ret
-}
-
-define <4 x float> @test_int_x86_avx10_vcvthf82ps128_mem(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvthf82ps128_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    vcvthf82ps (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x36,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvthf82ps128_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vcvthf82ps (%eax), %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x36,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %a = load <16 x i8>, ptr %ptr_a
-  %ret = call <4 x float> @llvm.x86.avx10.vcvthf82ps128(<16 x i8> %a)
-  ret <4 x float> %ret
-}
-
-define <8 x float> @test_int_x86_avx10_vcvthf82ps256_mem(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvthf82ps256_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    vcvthf82ps (%rdi), %ymm0 # encoding: [0x62,0xf5,0x7c,0x28,0x36,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvthf82ps256_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vcvthf82ps (%eax), %ymm0 # encoding: [0x62,0xf5,0x7c,0x28,0x36,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %a = load <16 x i8>, ptr %ptr_a
-  %ret = call <8 x float> @llvm.x86.avx10.vcvthf82ps256(<16 x i8> %a)
-  ret <8 x float> %ret
-}
-
-define <16 x float> @test_int_x86_avx10_vcvthf82ps512_mem(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvthf82ps512_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    vcvthf82ps (%rdi), %zmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x36,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvthf82ps512_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vcvthf82ps (%eax), %zmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x36,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %a = load <16 x i8>, ptr %ptr_a
-  %ret = call <16 x float> @llvm.x86.avx10.vcvthf82ps512(<16 x i8> %a)
-  ret <16 x float> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtbf82bf4s128(<16 x i8> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbf82bf4s128:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtbf82bf4s %xmm0, %xmm0 # encoding: [0x62,0xf5,0xfe,0x08,0x3d,0xc0]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf82bf4s128(<16 x i8> %a)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtbf82bf4s256(<32 x i8> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbf82bf4s256:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtbf82bf4s %ymm0, %xmm0 # encoding: [0x62,0xf5,0xfe,0x28,0x3d,0xc0]
-; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf82bf4s256(<32 x i8> %a)
-  ret <16 x i8> %ret
-}
-
-define <32 x i8> @test_int_x86_avx10_vcvtbf82bf4s512(<64 x i8> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbf82bf4s512:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtbf82bf4s %zmm0, %ymm0 # encoding: [0x62,0xf5,0xfe,0x48,0x3d,0xc0]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <32 x i8> @llvm.x86.avx10.vcvtbf82bf4s512(<64 x i8> %a)
-  ret <32 x i8> %ret
-}
-
-declare <16 x i8> @llvm.x86.avx10.vcvtbf82bf4s128(<16 x i8>)
-declare <16 x i8> @llvm.x86.avx10.vcvtbf82bf4s256(<32 x i8>)
-declare <32 x i8> @llvm.x86.avx10.vcvtbf82bf4s512(<64 x i8>)
-
-define <16 x i8> @test_int_x86_avx10_vcvtbf82bf4s128_mem(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtbf82bf4s128_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    vmovaps (%rdi), %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0x07]
-; X64-NEXT:    vcvtbf82bf4s %xmm0, %xmm0 # encoding: [0x62,0xf5,0xfe,0x08,0x3d,0xc0]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvtbf82bf4s128_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vmovaps (%eax), %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0x00]
-; X86-NEXT:    vcvtbf82bf4s %xmm0, %xmm0 # encoding: [0x62,0xf5,0xfe,0x08,0x3d,0xc0]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %a = load <16 x i8>, ptr %ptr_a
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf82bf4s128(<16 x i8> %a)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtbf82bf4s256_mem(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtbf82bf4s256_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    vmovaps (%rdi), %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfc,0x28,0x07]
-; X64-NEXT:    vcvtbf82bf4s %ymm0, %xmm0 # encoding: [0x62,0xf5,0xfe,0x28,0x3d,0xc0]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvtbf82bf4s256_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vmovaps (%eax), %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfc,0x28,0x00]
-; X86-NEXT:    vcvtbf82bf4s %ymm0, %xmm0 # encoding: [0x62,0xf5,0xfe,0x28,0x3d,0xc0]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %a = load <32 x i8>, ptr %ptr_a
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf82bf4s256(<32 x i8> %a)
-  ret <16 x i8> %ret
-}
-
-define <32 x i8> @test_int_x86_avx10_vcvtbf82bf4s512_mem(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtbf82bf4s512_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    vmovaps (%rdi), %zmm0 # encoding: [0x62,0xf1,0x7c,0x48,0x28,0x07]
-; X64-NEXT:    vcvtbf82bf4s %zmm0, %ymm0 # encoding: [0x62,0xf5,0xfe,0x48,0x3d,0xc0]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvtbf82bf4s512_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vmovaps (%eax), %zmm0 # encoding: [0x62,0xf1,0x7c,0x48,0x28,0x00]
-; X86-NEXT:    vcvtbf82bf4s %zmm0, %ymm0 # encoding: [0x62,0xf5,0xfe,0x48,0x3d,0xc0]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %a = load <64 x i8>, ptr %ptr_a
-  %ret = call <32 x i8> @llvm.x86.avx10.vcvtbf82bf4s512(<64 x i8> %a)
-  ret <32 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvthf82bf4s128(<16 x i8> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvthf82bf4s128:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvthf82bf4s %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x3d,0xc0]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvthf82bf4s128(<16 x i8> %a)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvthf82bf4s256(<32 x i8> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvthf82bf4s256:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvthf82bf4s %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x3d,0xc0]
-; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvthf82bf4s256(<32 x i8> %a)
-  ret <16 x i8> %ret
-}
-
-define <32 x i8> @test_int_x86_avx10_vcvthf82bf4s512(<64 x i8> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvthf82bf4s512:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvthf82bf4s %zmm0, %ymm0 # encoding: [0x62,0xf5,0x7e,0x48,0x3d,0xc0]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <32 x i8> @llvm.x86.avx10.vcvthf82bf4s512(<64 x i8> %a)
-  ret <32 x i8> %ret
-}
-
-declare <16 x i8> @llvm.x86.avx10.vcvthf82bf4s128(<16 x i8>)
-declare <16 x i8> @llvm.x86.avx10.vcvthf82bf4s256(<32 x i8>)
-declare <32 x i8> @llvm.x86.avx10.vcvthf82bf4s512(<64 x i8>)
-
-define <16 x i8> @test_int_x86_avx10_vcvthf82bf4s128_mem(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvthf82bf4s128_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    vmovaps (%rdi), %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0x07]
-; X64-NEXT:    vcvthf82bf4s %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x3d,0xc0]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvthf82bf4s128_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vmovaps (%eax), %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0x00]
-; X86-NEXT:    vcvthf82bf4s %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x3d,0xc0]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %a = load <16 x i8>, ptr %ptr_a
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvthf82bf4s128(<16 x i8> %a)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvthf82bf4s256_mem(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvthf82bf4s256_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    vmovaps (%rdi), %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfc,0x28,0x07]
-; X64-NEXT:    vcvthf82bf4s %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x3d,0xc0]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvthf82bf4s256_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vmovaps (%eax), %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfc,0x28,0x00]
-; X86-NEXT:    vcvthf82bf4s %ymm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x28,0x3d,0xc0]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %a = load <32 x i8>, ptr %ptr_a
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvthf82bf4s256(<32 x i8> %a)
-  ret <16 x i8> %ret
-}
-
-define <32 x i8> @test_int_x86_avx10_vcvthf82bf4s512_mem(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvthf82bf4s512_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    vmovaps (%rdi), %zmm0 # encoding: [0x62,0xf1,0x7c,0x48,0x28,0x07]
-; X64-NEXT:    vcvthf82bf4s %zmm0, %ymm0 # encoding: [0x62,0xf5,0x7e,0x48,0x3d,0xc0]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvthf82bf4s512_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vmovaps (%eax), %zmm0 # encoding: [0x62,0xf1,0x7c,0x48,0x28,0x00]
-; X86-NEXT:    vcvthf82bf4s %zmm0, %ymm0 # encoding: [0x62,0xf5,0x7e,0x48,0x3d,0xc0]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %a = load <64 x i8>, ptr %ptr_a
-  %ret = call <32 x i8> @llvm.x86.avx10.vcvthf82bf4s512(<64 x i8> %a)
-  ret <32 x i8> %ret
-}
-
-define void @test_int_x86_avx10_vcvtbf82bf4s128_store(ptr %ptr, <16 x i8> %a) {
-; X64-LABEL: test_int_x86_avx10_vcvtbf82bf4s128_store:
-; X64:       # %bb.0:
-; X64-NEXT:    vcvtbf82bf4s %xmm0, (%rdi) # encoding: [0x62,0xf5,0xfe,0x08,0x3d,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvtbf82bf4s128_store:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vcvtbf82bf4s %xmm0, (%eax) # encoding: [0x62,0xf5,0xfe,0x08,0x3d,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf82bf4s128(<16 x i8> %a)
-  %cast = bitcast <16 x i8> %ret to <2 x i64>
-  %low = extractelement <2 x i64> %cast, i64 0
-  store i64 %low, ptr %ptr, align 1
-  ret void
-}
-
-define void @test_int_x86_avx10_vcvtbf82bf4s256_store(ptr %ptr, <32 x i8> %a) {
-; X64-LABEL: test_int_x86_avx10_vcvtbf82bf4s256_store:
-; X64:       # %bb.0:
-; X64-NEXT:    vcvtbf82bf4s %ymm0, (%rdi) # encoding: [0x62,0xf5,0xfe,0x28,0x3d,0x07]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvtbf82bf4s256_store:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vcvtbf82bf4s %ymm0, (%eax) # encoding: [0x62,0xf5,0xfe,0x28,0x3d,0x00]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf82bf4s256(<32 x i8> %a)
-  store <16 x i8> %ret, ptr %ptr, align 1
-  ret void
-}
-
-define void @test_int_x86_avx10_vcvtbf82bf4s512_store(ptr %ptr, <64 x i8> %a) {
-; X64-LABEL: test_int_x86_avx10_vcvtbf82bf4s512_store:
-; X64:       # %bb.0:
-; X64-NEXT:    vcvtbf82bf4s %zmm0, (%rdi) # encoding: [0x62,0xf5,0xfe,0x48,0x3d,0x07]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvtbf82bf4s512_store:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vcvtbf82bf4s %zmm0, (%eax) # encoding: [0x62,0xf5,0xfe,0x48,0x3d,0x00]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <32 x i8> @llvm.x86.avx10.vcvtbf82bf4s512(<64 x i8> %a)
-  store <32 x i8> %ret, ptr %ptr, align 1
-  ret void
-}
-
-define void @test_int_x86_avx10_vcvthf82bf4s128_store(ptr %ptr, <16 x i8> %a) {
-; X64-LABEL: test_int_x86_avx10_vcvthf82bf4s128_store:
-; X64:       # %bb.0:
-; X64-NEXT:    vcvthf82bf4s %xmm0, (%rdi) # encoding: [0x62,0xf5,0x7e,0x08,0x3d,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvthf82bf4s128_store:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vcvthf82bf4s %xmm0, (%eax) # encoding: [0x62,0xf5,0x7e,0x08,0x3d,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvthf82bf4s128(<16 x i8> %a)
-  %cast = bitcast <16 x i8> %ret to <2 x double>
-  %low = extractelement <2 x double> %cast, i64 0
-  store double %low, ptr %ptr, align 1
-  ret void
-}
-
-define void @test_int_x86_avx10_vcvthf82bf4s256_store(ptr %ptr, <32 x i8> %a) {
-; X64-LABEL: test_int_x86_avx10_vcvthf82bf4s256_store:
-; X64:       # %bb.0:
-; X64-NEXT:    vcvthf82bf4s %ymm0, (%rdi) # encoding: [0x62,0xf5,0x7e,0x28,0x3d,0x07]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvthf82bf4s256_store:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vcvthf82bf4s %ymm0, (%eax) # encoding: [0x62,0xf5,0x7e,0x28,0x3d,0x00]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvthf82bf4s256(<32 x i8> %a)
-  store <16 x i8> %ret, ptr %ptr, align 1
-  ret void
-}
-
-define void @test_int_x86_avx10_vcvthf82bf4s512_store(ptr %ptr, <64 x i8> %a) {
-; X64-LABEL: test_int_x86_avx10_vcvthf82bf4s512_store:
-; X64:       # %bb.0:
-; X64-NEXT:    vcvthf82bf4s %zmm0, (%rdi) # encoding: [0x62,0xf5,0x7e,0x48,0x3d,0x07]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvthf82bf4s512_store:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vcvthf82bf4s %zmm0, (%eax) # encoding: [0x62,0xf5,0x7e,0x48,0x3d,0x00]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <32 x i8> @llvm.x86.avx10.vcvthf82bf4s512(<64 x i8> %a)
-  store <32 x i8> %ret, ptr %ptr, align 1
-  ret void
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtbf82bf6s128(<16 x i8> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbf82bf6s128:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtbf82bf6s %xmm0, %xmm0 # encoding: [0x62,0xf5,0xfe,0x08,0x3e,0xc0]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf82bf6s128(<16 x i8> %a)
-  ret <16 x i8> %ret
-}
-
-define <32 x i8> @test_int_x86_avx10_vcvtbf82bf6s256(<32 x i8> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbf82bf6s256:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtbf82bf6s %ymm0, %ymm0 # encoding: [0x62,0xf5,0xfe,0x28,0x3e,0xc0]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <32 x i8> @llvm.x86.avx10.vcvtbf82bf6s256(<32 x i8> %a)
-  ret <32 x i8> %ret
-}
-
-define <64 x i8> @test_int_x86_avx10_vcvtbf82bf6s512(<64 x i8> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbf82bf6s512:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtbf82bf6s %zmm0, %zmm0 # encoding: [0x62,0xf5,0xfe,0x48,0x3e,0xc0]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <64 x i8> @llvm.x86.avx10.vcvtbf82bf6s512(<64 x i8> %a)
-  ret <64 x i8> %ret
-}
-
-declare <16 x i8> @llvm.x86.avx10.vcvtbf82bf6s128(<16 x i8>)
-declare <32 x i8> @llvm.x86.avx10.vcvtbf82bf6s256(<32 x i8>)
-declare <64 x i8> @llvm.x86.avx10.vcvtbf82bf6s512(<64 x i8>)
-
-define <16 x i8> @test_int_x86_avx10_vcvtbf82bf6s128_mem(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtbf82bf6s128_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    vmovaps (%rdi), %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0x07]
-; X64-NEXT:    vcvtbf82bf6s %xmm0, %xmm0 # encoding: [0x62,0xf5,0xfe,0x08,0x3e,0xc0]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvtbf82bf6s128_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vmovaps (%eax), %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0x00]
-; X86-NEXT:    vcvtbf82bf6s %xmm0, %xmm0 # encoding: [0x62,0xf5,0xfe,0x08,0x3e,0xc0]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %a = load <16 x i8>, ptr %ptr_a
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf82bf6s128(<16 x i8> %a)
-  ret <16 x i8> %ret
-}
-
-define <32 x i8> @test_int_x86_avx10_vcvtbf82bf6s256_mem(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtbf82bf6s256_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    vmovaps (%rdi), %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfc,0x28,0x07]
-; X64-NEXT:    vcvtbf82bf6s %ymm0, %ymm0 # encoding: [0x62,0xf5,0xfe,0x28,0x3e,0xc0]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvtbf82bf6s256_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vmovaps (%eax), %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfc,0x28,0x00]
-; X86-NEXT:    vcvtbf82bf6s %ymm0, %ymm0 # encoding: [0x62,0xf5,0xfe,0x28,0x3e,0xc0]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %a = load <32 x i8>, ptr %ptr_a
-  %ret = call <32 x i8> @llvm.x86.avx10.vcvtbf82bf6s256(<32 x i8> %a)
-  ret <32 x i8> %ret
-}
-
-define <64 x i8> @test_int_x86_avx10_vcvtbf82bf6s512_mem(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtbf82bf6s512_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    vmovaps (%rdi), %zmm0 # encoding: [0x62,0xf1,0x7c,0x48,0x28,0x07]
-; X64-NEXT:    vcvtbf82bf6s %zmm0, %zmm0 # encoding: [0x62,0xf5,0xfe,0x48,0x3e,0xc0]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvtbf82bf6s512_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vmovaps (%eax), %zmm0 # encoding: [0x62,0xf1,0x7c,0x48,0x28,0x00]
-; X86-NEXT:    vcvtbf82bf6s %zmm0, %zmm0 # encoding: [0x62,0xf5,0xfe,0x48,0x3e,0xc0]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %a = load <64 x i8>, ptr %ptr_a
-  %ret = call <64 x i8> @llvm.x86.avx10.vcvtbf82bf6s512(<64 x i8> %a)
-  ret <64 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvthf82hf6s128(<16 x i8> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvthf82hf6s128:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvthf82hf6s %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x3c,0xc0]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvthf82hf6s128(<16 x i8> %a)
-  ret <16 x i8> %ret
-}
-
-define <32 x i8> @test_int_x86_avx10_vcvthf82hf6s256(<32 x i8> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvthf82hf6s256:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvthf82hf6s %ymm0, %ymm0 # encoding: [0x62,0xf5,0x7e,0x28,0x3c,0xc0]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <32 x i8> @llvm.x86.avx10.vcvthf82hf6s256(<32 x i8> %a)
-  ret <32 x i8> %ret
-}
-
-define <64 x i8> @test_int_x86_avx10_vcvthf82hf6s512(<64 x i8> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvthf82hf6s512:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvthf82hf6s %zmm0, %zmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x3c,0xc0]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <64 x i8> @llvm.x86.avx10.vcvthf82hf6s512(<64 x i8> %a)
-  ret <64 x i8> %ret
-}
-
-declare <16 x i8> @llvm.x86.avx10.vcvthf82hf6s128(<16 x i8>)
-declare <32 x i8> @llvm.x86.avx10.vcvthf82hf6s256(<32 x i8>)
-declare <64 x i8> @llvm.x86.avx10.vcvthf82hf6s512(<64 x i8>)
-
-define <16 x i8> @test_int_x86_avx10_vcvthf82hf6s128_mem(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvthf82hf6s128_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    vmovaps (%rdi), %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0x07]
-; X64-NEXT:    vcvthf82hf6s %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x3c,0xc0]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvthf82hf6s128_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vmovaps (%eax), %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0x00]
-; X86-NEXT:    vcvthf82hf6s %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7e,0x08,0x3c,0xc0]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %a = load <16 x i8>, ptr %ptr_a
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvthf82hf6s128(<16 x i8> %a)
-  ret <16 x i8> %ret
-}
-
-define <32 x i8> @test_int_x86_avx10_vcvthf82hf6s256_mem(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvthf82hf6s256_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    vmovaps (%rdi), %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfc,0x28,0x07]
-; X64-NEXT:    vcvthf82hf6s %ymm0, %ymm0 # encoding: [0x62,0xf5,0x7e,0x28,0x3c,0xc0]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvthf82hf6s256_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vmovaps (%eax), %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfc,0x28,0x00]
-; X86-NEXT:    vcvthf82hf6s %ymm0, %ymm0 # encoding: [0x62,0xf5,0x7e,0x28,0x3c,0xc0]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %a = load <32 x i8>, ptr %ptr_a
-  %ret = call <32 x i8> @llvm.x86.avx10.vcvthf82hf6s256(<32 x i8> %a)
-  ret <32 x i8> %ret
-}
-
-define <64 x i8> @test_int_x86_avx10_vcvthf82hf6s512_mem(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvthf82hf6s512_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    vmovaps (%rdi), %zmm0 # encoding: [0x62,0xf1,0x7c,0x48,0x28,0x07]
-; X64-NEXT:    vcvthf82hf6s %zmm0, %zmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x3c,0xc0]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvthf82hf6s512_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vmovaps (%eax), %zmm0 # encoding: [0x62,0xf1,0x7c,0x48,0x28,0x00]
-; X86-NEXT:    vcvthf82hf6s %zmm0, %zmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x3c,0xc0]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %a = load <64 x i8>, ptr %ptr_a
-  %ret = call <64 x i8> @llvm.x86.avx10.vcvthf82hf6s512(<64 x i8> %a)
-  ret <64 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtbf42hf8128(<16 x i8> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbf42hf8128:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtbf42hf8 %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x37,0xc0]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf42hf8128(<16 x i8> %a)
-  ret <16 x i8> %ret
-}
-
-define <32 x i8> @test_int_x86_avx10_vcvtbf42hf8256(<16 x i8> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbf42hf8256:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtbf42hf8 %xmm0, %ymm0 # encoding: [0x62,0xf5,0x7c,0x28,0x37,0xc0]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <32 x i8> @llvm.x86.avx10.vcvtbf42hf8256(<16 x i8> %a)
-  ret <32 x i8> %ret
-}
-
-define <64 x i8> @test_int_x86_avx10_vcvtbf42hf8512(<32 x i8> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbf42hf8512:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtbf42hf8 %ymm0, %zmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x37,0xc0]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <64 x i8> @llvm.x86.avx10.vcvtbf42hf8512(<32 x i8> %a)
-  ret <64 x i8> %ret
-}
-
-declare <16 x i8> @llvm.x86.avx10.vcvtbf42hf8128(<16 x i8>)
-declare <32 x i8> @llvm.x86.avx10.vcvtbf42hf8256(<16 x i8>)
-declare <64 x i8> @llvm.x86.avx10.vcvtbf42hf8512(<32 x i8>)
-
-define <16 x i8> @test_int_x86_avx10_vcvtbf42hf8128_mem(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtbf42hf8128_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    vcvtbf42hf8 (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x37,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvtbf42hf8128_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vcvtbf42hf8 (%eax), %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x37,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %a = load <16 x i8>, ptr %ptr_a
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf42hf8128(<16 x i8> %a)
-  ret <16 x i8> %ret
-}
-
-define <32 x i8> @test_int_x86_avx10_vcvtbf42hf8256_mem(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtbf42hf8256_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    vcvtbf42hf8 (%rdi), %ymm0 # encoding: [0x62,0xf5,0x7c,0x28,0x37,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvtbf42hf8256_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vcvtbf42hf8 (%eax), %ymm0 # encoding: [0x62,0xf5,0x7c,0x28,0x37,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %a = load <16 x i8>, ptr %ptr_a
-  %ret = call <32 x i8> @llvm.x86.avx10.vcvtbf42hf8256(<16 x i8> %a)
-  ret <32 x i8> %ret
-}
-
-define <64 x i8> @test_int_x86_avx10_vcvtbf42hf8512_mem(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtbf42hf8512_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    vcvtbf42hf8 (%rdi), %zmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x37,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvtbf42hf8512_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vcvtbf42hf8 (%eax), %zmm0 # encoding: [0x62,0xf5,0x7c,0x48,0x37,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %a = load <32 x i8>, ptr %ptr_a
-  %ret = call <64 x i8> @llvm.x86.avx10.vcvtbf42hf8512(<32 x i8> %a)
-  ret <64 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtbf42hf8128_vzload(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtbf42hf8128_vzload:
-; X64:       # %bb.0:
-; X64-NEXT:    vcvtbf42hf8 (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x37,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvtbf42hf8128_vzload:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vcvtbf42hf8 (%eax), %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x37,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %l = load i64, ptr %ptr_a, align 1
-  %v = insertelement <2 x i64> <i64 poison, i64 0>, i64 %l, i64 0
-  %a = bitcast <2 x i64> %v to <16 x i8>
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf42hf8128(<16 x i8> %a)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtbf42hf8128_vzload_mask(ptr %ptr_a, <16 x i8> %src, i16 %mask) {
-; X64-LABEL: test_int_x86_avx10_vcvtbf42hf8128_vzload_mask:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
-; X64-NEXT:    vcvtbf42hf8 (%rdi), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x37,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvtbf42hf8128_vzload_mask:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x08]
-; X86-NEXT:    vcvtbf42hf8 (%eax), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x37,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %l = load i64, ptr %ptr_a, align 1
-  %v = insertelement <2 x i64> <i64 poison, i64 0>, i64 %l, i64 0
-  %a = bitcast <2 x i64> %v to <16 x i8>
-  %cvt = call <16 x i8> @llvm.x86.avx10.vcvtbf42hf8128(<16 x i8> %a)
-  %msk = bitcast i16 %mask to <16 x i1>
-  %ret = select <16 x i1> %msk, <16 x i8> %cvt, <16 x i8> %src
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtbf42hf8128_vzload_maskz(ptr %ptr_a, i16 %mask) {
-; X64-LABEL: test_int_x86_avx10_vcvtbf42hf8128_vzload_maskz:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
-; X64-NEXT:    vcvtbf42hf8 (%rdi), %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x37,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvtbf42hf8128_vzload_maskz:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x08]
-; X86-NEXT:    vcvtbf42hf8 (%eax), %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x37,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %l = load i64, ptr %ptr_a, align 1
-  %v = insertelement <2 x i64> <i64 poison, i64 0>, i64 %l, i64 0
-  %a = bitcast <2 x i64> %v to <16 x i8>
-  %cvt = call <16 x i8> @llvm.x86.avx10.vcvtbf42hf8128(<16 x i8> %a)
-  %msk = bitcast i16 %mask to <16 x i1>
-  %ret = select <16 x i1> %msk, <16 x i8> %cvt, <16 x i8> zeroinitializer
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtbf42hf8_s2v128(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtbf42hf8_s2v128:
-; X64:       # %bb.0:
-; X64-NEXT:    vcvtbf42hf8 (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x37,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvtbf42hf8_s2v128:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vcvtbf42hf8 (%eax), %xmm0 # encoding: [0x62,0xf5,0x7c,0x08,0x37,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %l = load i64, ptr %ptr_a, align 8
-  %v = insertelement <2 x i64> poison, i64 %l, i64 0
-  %a = bitcast <2 x i64> %v to <16 x i8>
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf42hf8128(<16 x i8> %a)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtbf42hf8128_mem_mask(ptr %ptr_a, <16 x i8> %src, i16 %mask) {
-; X64-LABEL: test_int_x86_avx10_vcvtbf42hf8128_mem_mask:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
-; X64-NEXT:    vcvtbf42hf8 (%rdi), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x37,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvtbf42hf8128_mem_mask:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x08]
-; X86-NEXT:    vcvtbf42hf8 (%eax), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x37,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %a = load <16 x i8>, ptr %ptr_a
-  %cvt = call <16 x i8> @llvm.x86.avx10.vcvtbf42hf8128(<16 x i8> %a)
-  %msk = bitcast i16 %mask to <16 x i1>
-  %ret = select <16 x i1> %msk, <16 x i8> %cvt, <16 x i8> %src
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtbf42hf8128_mem_maskz(ptr %ptr_a, i16 %mask) {
-; X64-LABEL: test_int_x86_avx10_vcvtbf42hf8128_mem_maskz:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
-; X64-NEXT:    vcvtbf42hf8 (%rdi), %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x37,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvtbf42hf8128_mem_maskz:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x08]
-; X86-NEXT:    vcvtbf42hf8 (%eax), %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x37,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %a = load <16 x i8>, ptr %ptr_a
-  %cvt = call <16 x i8> @llvm.x86.avx10.vcvtbf42hf8128(<16 x i8> %a)
-  %msk = bitcast i16 %mask to <16 x i1>
-  %ret = select <16 x i1> %msk, <16 x i8> %cvt, <16 x i8> zeroinitializer
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvtbf62hf8128(<16 x i8> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbf62hf8128:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtbf62hf8 %xmm0, %xmm0 # encoding: [0x62,0xf5,0xfd,0x08,0x37,0xc0]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf62hf8128(<16 x i8> %a)
-  ret <16 x i8> %ret
-}
-
-define <32 x i8> @test_int_x86_avx10_vcvtbf62hf8256(<32 x i8> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbf62hf8256:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtbf62hf8 %ymm0, %ymm0 # encoding: [0x62,0xf5,0xfd,0x28,0x37,0xc0]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <32 x i8> @llvm.x86.avx10.vcvtbf62hf8256(<32 x i8> %a)
-  ret <32 x i8> %ret
-}
-
-define <64 x i8> @test_int_x86_avx10_vcvtbf62hf8512(<64 x i8> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvtbf62hf8512:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvtbf62hf8 %zmm0, %zmm0 # encoding: [0x62,0xf5,0xfd,0x48,0x37,0xc0]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <64 x i8> @llvm.x86.avx10.vcvtbf62hf8512(<64 x i8> %a)
-  ret <64 x i8> %ret
-}
-
-declare <16 x i8> @llvm.x86.avx10.vcvtbf62hf8128(<16 x i8>)
-declare <32 x i8> @llvm.x86.avx10.vcvtbf62hf8256(<32 x i8>)
-declare <64 x i8> @llvm.x86.avx10.vcvtbf62hf8512(<64 x i8>)
-
-define <16 x i8> @test_int_x86_avx10_vcvtbf62hf8128_mem(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtbf62hf8128_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    vmovdqa (%rdi), %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf9,0x6f,0x07]
-; X64-NEXT:    vcvtbf62hf8 %xmm0, %xmm0 # encoding: [0x62,0xf5,0xfd,0x08,0x37,0xc0]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvtbf62hf8128_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vmovdqa (%eax), %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf9,0x6f,0x00]
-; X86-NEXT:    vcvtbf62hf8 %xmm0, %xmm0 # encoding: [0x62,0xf5,0xfd,0x08,0x37,0xc0]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %a = load <16 x i8>, ptr %ptr_a
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvtbf62hf8128(<16 x i8> %a)
-  ret <16 x i8> %ret
-}
-
-define <32 x i8> @test_int_x86_avx10_vcvtbf62hf8256_mem(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtbf62hf8256_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    vmovdqa (%rdi), %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfd,0x6f,0x07]
-; X64-NEXT:    vcvtbf62hf8 %ymm0, %ymm0 # encoding: [0x62,0xf5,0xfd,0x28,0x37,0xc0]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvtbf62hf8256_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vmovdqa (%eax), %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfd,0x6f,0x00]
-; X86-NEXT:    vcvtbf62hf8 %ymm0, %ymm0 # encoding: [0x62,0xf5,0xfd,0x28,0x37,0xc0]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %a = load <32 x i8>, ptr %ptr_a
-  %ret = call <32 x i8> @llvm.x86.avx10.vcvtbf62hf8256(<32 x i8> %a)
-  ret <32 x i8> %ret
-}
-
-define <64 x i8> @test_int_x86_avx10_vcvtbf62hf8512_mem(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvtbf62hf8512_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    vmovdqa64 (%rdi), %zmm0 # encoding: [0x62,0xf1,0xfd,0x48,0x6f,0x07]
-; X64-NEXT:    vcvtbf62hf8 %zmm0, %zmm0 # encoding: [0x62,0xf5,0xfd,0x48,0x37,0xc0]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvtbf62hf8512_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vmovdqa64 (%eax), %zmm0 # encoding: [0x62,0xf1,0xfd,0x48,0x6f,0x00]
-; X86-NEXT:    vcvtbf62hf8 %zmm0, %zmm0 # encoding: [0x62,0xf5,0xfd,0x48,0x37,0xc0]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %a = load <64 x i8>, ptr %ptr_a
-  %ret = call <64 x i8> @llvm.x86.avx10.vcvtbf62hf8512(<64 x i8> %a)
-  ret <64 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vcvthf62hf8128(<16 x i8> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvthf62hf8128:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvthf62hf8 %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7d,0x08,0x37,0xc0]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvthf62hf8128(<16 x i8> %a)
-  ret <16 x i8> %ret
-}
-
-define <32 x i8> @test_int_x86_avx10_vcvthf62hf8256(<32 x i8> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvthf62hf8256:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvthf62hf8 %ymm0, %ymm0 # encoding: [0x62,0xf5,0x7d,0x28,0x37,0xc0]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <32 x i8> @llvm.x86.avx10.vcvthf62hf8256(<32 x i8> %a)
-  ret <32 x i8> %ret
-}
-
-define <64 x i8> @test_int_x86_avx10_vcvthf62hf8512(<64 x i8> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vcvthf62hf8512:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vcvthf62hf8 %zmm0, %zmm0 # encoding: [0x62,0xf5,0x7d,0x48,0x37,0xc0]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <64 x i8> @llvm.x86.avx10.vcvthf62hf8512(<64 x i8> %a)
-  ret <64 x i8> %ret
-}
-
-declare <16 x i8> @llvm.x86.avx10.vcvthf62hf8128(<16 x i8>)
-declare <32 x i8> @llvm.x86.avx10.vcvthf62hf8256(<32 x i8>)
-declare <64 x i8> @llvm.x86.avx10.vcvthf62hf8512(<64 x i8>)
-
-define <16 x i8> @test_int_x86_avx10_vcvthf62hf8128_mem(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvthf62hf8128_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    vmovdqa (%rdi), %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf9,0x6f,0x07]
-; X64-NEXT:    vcvthf62hf8 %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7d,0x08,0x37,0xc0]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvthf62hf8128_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vmovdqa (%eax), %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf9,0x6f,0x00]
-; X86-NEXT:    vcvthf62hf8 %xmm0, %xmm0 # encoding: [0x62,0xf5,0x7d,0x08,0x37,0xc0]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %a = load <16 x i8>, ptr %ptr_a
-  %ret = call <16 x i8> @llvm.x86.avx10.vcvthf62hf8128(<16 x i8> %a)
-  ret <16 x i8> %ret
-}
-
-define <32 x i8> @test_int_x86_avx10_vcvthf62hf8256_mem(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvthf62hf8256_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    vmovdqa (%rdi), %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfd,0x6f,0x07]
-; X64-NEXT:    vcvthf62hf8 %ymm0, %ymm0 # encoding: [0x62,0xf5,0x7d,0x28,0x37,0xc0]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvthf62hf8256_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vmovdqa (%eax), %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfd,0x6f,0x00]
-; X86-NEXT:    vcvthf62hf8 %ymm0, %ymm0 # encoding: [0x62,0xf5,0x7d,0x28,0x37,0xc0]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %a = load <32 x i8>, ptr %ptr_a
-  %ret = call <32 x i8> @llvm.x86.avx10.vcvthf62hf8256(<32 x i8> %a)
-  ret <32 x i8> %ret
-}
-
-define <64 x i8> @test_int_x86_avx10_vcvthf62hf8512_mem(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vcvthf62hf8512_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    vmovdqa64 (%rdi), %zmm0 # encoding: [0x62,0xf1,0xfd,0x48,0x6f,0x07]
-; X64-NEXT:    vcvthf62hf8 %zmm0, %zmm0 # encoding: [0x62,0xf5,0x7d,0x48,0x37,0xc0]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vcvthf62hf8512_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vmovdqa64 (%eax), %zmm0 # encoding: [0x62,0xf1,0xfd,0x48,0x6f,0x00]
-; X86-NEXT:    vcvthf62hf8 %zmm0, %zmm0 # encoding: [0x62,0xf5,0x7d,0x48,0x37,0xc0]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %a = load <64 x i8>, ptr %ptr_a
-  %ret = call <64 x i8> @llvm.x86.avx10.vcvthf62hf8512(<64 x i8> %a)
-  ret <64 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_vunpackb_128(<16 x i8> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vunpackb_128:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vunpackb $1, %xmm0, %xmm0 # encoding: [0x62,0xf3,0x7c,0x08,0x3d,0xc0,0x01]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.vunpackb.128(<16 x i8> %a, i8 1)
-  ret <16 x i8> %ret
-}
-
-define <32 x i8> @test_int_x86_avx10_vunpackb_256(<32 x i8> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vunpackb_256:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vunpackb $2, %ymm0, %ymm0 # encoding: [0x62,0xf3,0x7c,0x28,0x3d,0xc0,0x02]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <32 x i8> @llvm.x86.avx10.vunpackb.256(<32 x i8> %a, i8 2)
-  ret <32 x i8> %ret
-}
-
-define <64 x i8> @test_int_x86_avx10_vunpackb_512(<64 x i8> %a) {
-; CHECK-LABEL: test_int_x86_avx10_vunpackb_512:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vunpackb $3, %zmm0, %zmm0 # encoding: [0x62,0xf3,0x7c,0x48,0x3d,0xc0,0x03]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <64 x i8> @llvm.x86.avx10.vunpackb.512(<64 x i8> %a, i8 3)
-  ret <64 x i8> %ret
-}
-
-declare <16 x i8> @llvm.x86.avx10.vunpackb.128(<16 x i8>, i8)
-declare <32 x i8> @llvm.x86.avx10.vunpackb.256(<32 x i8>, i8)
-declare <64 x i8> @llvm.x86.avx10.vunpackb.512(<64 x i8>, i8)
-
-define <16 x i8> @test_int_x86_avx10_vunpackb_mem_128(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vunpackb_mem_128:
-; X64:       # %bb.0:
-; X64-NEXT:    vunpackb $1, (%rdi), %xmm0 # encoding: [0x62,0xf3,0x7c,0x08,0x3d,0x07,0x01]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vunpackb_mem_128:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vunpackb $1, (%eax), %xmm0 # encoding: [0x62,0xf3,0x7c,0x08,0x3d,0x00,0x01]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %a = load <16 x i8>, ptr %ptr_a
-  %ret = call <16 x i8> @llvm.x86.avx10.vunpackb.128(<16 x i8> %a, i8 1)
-  ret <16 x i8> %ret
-}
-
-define <32 x i8> @test_int_x86_avx10_vunpackb_mem_256(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vunpackb_mem_256:
-; X64:       # %bb.0:
-; X64-NEXT:    vunpackb $2, (%rdi), %ymm0 # encoding: [0x62,0xf3,0x7c,0x28,0x3d,0x07,0x02]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vunpackb_mem_256:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vunpackb $2, (%eax), %ymm0 # encoding: [0x62,0xf3,0x7c,0x28,0x3d,0x00,0x02]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %a = load <32 x i8>, ptr %ptr_a
-  %ret = call <32 x i8> @llvm.x86.avx10.vunpackb.256(<32 x i8> %a, i8 2)
-  ret <32 x i8> %ret
-}
-
-define <64 x i8> @test_int_x86_avx10_vunpackb_mem_512(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_vunpackb_mem_512:
-; X64:       # %bb.0:
-; X64-NEXT:    vunpackb $3, (%rdi), %zmm0 # encoding: [0x62,0xf3,0x7c,0x48,0x3d,0x07,0x03]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_vunpackb_mem_512:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vunpackb $3, (%eax), %zmm0 # encoding: [0x62,0xf3,0x7c,0x48,0x3d,0x00,0x03]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %a = load <64 x i8>, ptr %ptr_a
-  %ret = call <64 x i8> @llvm.x86.avx10.vunpackb.512(<64 x i8> %a, i8 3)
-  ret <64 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_pmovssdb_128(<4 x i32> %a) {
-; CHECK-LABEL: test_int_x86_avx10_pmovssdb_128:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vpmovssdb %xmm0, %xmm0 # encoding: [0x62,0xf2,0x7e,0x08,0x41,0xc0]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.128(<4 x i32> %a, <16 x i8> zeroinitializer, i8 -1)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_pmovssdb_256(<8 x i32> %a) {
-; CHECK-LABEL: test_int_x86_avx10_pmovssdb_256:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vpmovssdb %ymm0, %xmm0 # encoding: [0x62,0xf2,0x7e,0x28,0x41,0xc0]
-; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.256(<8 x i32> %a, <16 x i8> zeroinitializer, i8 -1)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_pmovssdb_512(<16 x i32> %a) {
-; CHECK-LABEL: test_int_x86_avx10_pmovssdb_512:
-; CHECK:       # %bb.0:
-; CHECK-NEXT:    vpmovssdb %zmm0, %xmm0 # encoding: [0x62,0xf2,0x7e,0x48,0x41,0xc0]
-; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.512(<16 x i32> %a, <16 x i8> zeroinitializer, i16 -1)
-  ret <16 x i8> %ret
-}
-
-declare <16 x i8> @llvm.x86.avx10.mask.pmovss.db.128(<4 x i32>, <16 x i8>, i8)
-declare <16 x i8> @llvm.x86.avx10.mask.pmovss.db.256(<8 x i32>, <16 x i8>, i8)
-declare <16 x i8> @llvm.x86.avx10.mask.pmovss.db.512(<16 x i32>, <16 x i8>, i16)
-declare void @llvm.x86.avx10.mask.pmovss.db.mem.128(ptr, <4 x i32>, i8)
-declare void @llvm.x86.avx10.mask.pmovss.db.mem.256(ptr, <8 x i32>, i8)
-declare void @llvm.x86.avx10.mask.pmovss.db.mem.512(ptr, <16 x i32>, i16)
-
-define <16 x i8> @test_int_x86_avx10_mask_pmovssdb_128(<4 x i32> %a, <16 x i8> %passthru, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_pmovssdb_128:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vpmovssdb %xmm0, %xmm2 # encoding: [0x62,0xf2,0x7e,0x08,0x41,0xc2]
-; X64-NEXT:    vpmovssdb %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf2,0x7e,0x09,0x41,0xc1]
-; X64-NEXT:    vpaddb %xmm1, %xmm2, %xmm1 # EVEX TO VEX Compression encoding: [0xc5,0xe9,0xfc,0xc9]
-; X64-NEXT:    vpmovssdb %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf2,0x7e,0x89,0x41,0xc0]
-; X64-NEXT:    vpaddb %xmm0, %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf1,0xfc,0xc0]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_pmovssdb_128:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vpmovssdb %xmm0, %xmm2 # encoding: [0x62,0xf2,0x7e,0x08,0x41,0xc2]
-; X86-NEXT:    vpmovssdb %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf2,0x7e,0x09,0x41,0xc1]
-; X86-NEXT:    vpaddb %xmm1, %xmm2, %xmm1 # EVEX TO VEX Compression encoding: [0xc5,0xe9,0xfc,0xc9]
-; X86-NEXT:    vpmovssdb %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf2,0x7e,0x89,0x41,0xc0]
-; X86-NEXT:    vpaddb %xmm0, %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf1,0xfc,0xc0]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %res0 = call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.128(<4 x i32> %a, <16 x i8> %passthru, i8 -1)
-  %res1 = call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.128(<4 x i32> %a, <16 x i8> %passthru, i8 %mask)
-  %res2 = call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.128(<4 x i32> %a, <16 x i8> zeroinitializer, i8 %mask)
-  %add1 = add <16 x i8> %res0, %res1
-  %add2 = add <16 x i8> %add1, %res2
-  ret <16 x i8> %add2
-}
-
-define <16 x i8> @test_int_x86_avx10_mask_pmovssdb_256(<8 x i32> %a, <16 x i8> %passthru, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_pmovssdb_256:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vpmovssdb %ymm0, %xmm2 # encoding: [0x62,0xf2,0x7e,0x28,0x41,0xc2]
-; X64-NEXT:    vpmovssdb %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf2,0x7e,0x29,0x41,0xc1]
-; X64-NEXT:    vpaddb %xmm1, %xmm2, %xmm1 # EVEX TO VEX Compression encoding: [0xc5,0xe9,0xfc,0xc9]
-; X64-NEXT:    vpmovssdb %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf2,0x7e,0xa9,0x41,0xc0]
-; X64-NEXT:    vpaddb %xmm0, %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf1,0xfc,0xc0]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_pmovssdb_256:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vpmovssdb %ymm0, %xmm2 # encoding: [0x62,0xf2,0x7e,0x28,0x41,0xc2]
-; X86-NEXT:    vpmovssdb %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf2,0x7e,0x29,0x41,0xc1]
-; X86-NEXT:    vpaddb %xmm1, %xmm2, %xmm1 # EVEX TO VEX Compression encoding: [0xc5,0xe9,0xfc,0xc9]
-; X86-NEXT:    vpmovssdb %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf2,0x7e,0xa9,0x41,0xc0]
-; X86-NEXT:    vpaddb %xmm0, %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf1,0xfc,0xc0]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %res0 = call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.256(<8 x i32> %a, <16 x i8> %passthru, i8 -1)
-  %res1 = call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.256(<8 x i32> %a, <16 x i8> %passthru, i8 %mask)
-  %res2 = call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.256(<8 x i32> %a, <16 x i8> zeroinitializer, i8 %mask)
-  %add1 = add <16 x i8> %res0, %res1
-  %add2 = add <16 x i8> %add1, %res2
-  ret <16 x i8> %add2
-}
-
-define <16 x i8> @test_int_x86_avx10_mask_pmovssdb_512(<16 x i32> %a, <16 x i8> %passthru, i16 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_pmovssdb_512:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vpmovssdb %zmm0, %xmm2 # encoding: [0x62,0xf2,0x7e,0x48,0x41,0xc2]
-; X64-NEXT:    vpmovssdb %zmm0, %xmm1 {%k1} # encoding: [0x62,0xf2,0x7e,0x49,0x41,0xc1]
-; X64-NEXT:    vpaddb %xmm1, %xmm2, %xmm1 # EVEX TO VEX Compression encoding: [0xc5,0xe9,0xfc,0xc9]
-; X64-NEXT:    vpmovssdb %zmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf2,0x7e,0xc9,0x41,0xc0]
-; X64-NEXT:    vpaddb %xmm0, %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf1,0xfc,0xc0]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_pmovssdb_512:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vpmovssdb %zmm0, %xmm2 # encoding: [0x62,0xf2,0x7e,0x48,0x41,0xc2]
-; X86-NEXT:    vpmovssdb %zmm0, %xmm1 {%k1} # encoding: [0x62,0xf2,0x7e,0x49,0x41,0xc1]
-; X86-NEXT:    vpaddb %xmm1, %xmm2, %xmm1 # EVEX TO VEX Compression encoding: [0xc5,0xe9,0xfc,0xc9]
-; X86-NEXT:    vpmovssdb %zmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf2,0x7e,0xc9,0x41,0xc0]
-; X86-NEXT:    vpaddb %xmm0, %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf1,0xfc,0xc0]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %res0 = call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.512(<16 x i32> %a, <16 x i8> %passthru, i16 -1)
-  %res1 = call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.512(<16 x i32> %a, <16 x i8> %passthru, i16 %mask)
-  %res2 = call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.512(<16 x i32> %a, <16 x i8> zeroinitializer, i16 %mask)
-  %add1 = add <16 x i8> %res0, %res1
-  %add2 = add <16 x i8> %add1, %res2
-  ret <16 x i8> %add2
-}
-
-define <16 x i8> @test_int_x86_avx10_pmovssdb_mem_128(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_pmovssdb_mem_128:
-; X64:       # %bb.0:
-; X64-NEXT:    vmovdqa (%rdi), %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf9,0x6f,0x07]
-; X64-NEXT:    vpmovssdb %xmm0, %xmm0 # encoding: [0x62,0xf2,0x7e,0x08,0x41,0xc0]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_pmovssdb_mem_128:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vmovdqa (%eax), %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf9,0x6f,0x00]
-; X86-NEXT:    vpmovssdb %xmm0, %xmm0 # encoding: [0x62,0xf2,0x7e,0x08,0x41,0xc0]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %a = load <4 x i32>, ptr %ptr_a
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.128(<4 x i32> %a, <16 x i8> zeroinitializer, i8 -1)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_pmovssdb_mem_256(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_pmovssdb_mem_256:
-; X64:       # %bb.0:
-; X64-NEXT:    vmovdqa (%rdi), %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfd,0x6f,0x07]
-; X64-NEXT:    vpmovssdb %ymm0, %xmm0 # encoding: [0x62,0xf2,0x7e,0x28,0x41,0xc0]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_pmovssdb_mem_256:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vmovdqa (%eax), %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfd,0x6f,0x00]
-; X86-NEXT:    vpmovssdb %ymm0, %xmm0 # encoding: [0x62,0xf2,0x7e,0x28,0x41,0xc0]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %a = load <8 x i32>, ptr %ptr_a
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.256(<8 x i32> %a, <16 x i8> zeroinitializer, i8 -1)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_pmovssdb_mem_512(ptr %ptr_a) {
-; X64-LABEL: test_int_x86_avx10_pmovssdb_mem_512:
-; X64:       # %bb.0:
-; X64-NEXT:    vmovdqa64 (%rdi), %zmm0 # encoding: [0x62,0xf1,0xfd,0x48,0x6f,0x07]
-; X64-NEXT:    vpmovssdb %zmm0, %xmm0 # encoding: [0x62,0xf2,0x7e,0x48,0x41,0xc0]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_pmovssdb_mem_512:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vmovdqa64 (%eax), %zmm0 # encoding: [0x62,0xf1,0xfd,0x48,0x6f,0x00]
-; X86-NEXT:    vpmovssdb %zmm0, %xmm0 # encoding: [0x62,0xf2,0x7e,0x48,0x41,0xc0]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %a = load <16 x i32>, ptr %ptr_a
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.512(<16 x i32> %a, <16 x i8> zeroinitializer, i16 -1)
-  ret <16 x i8> %ret
-}
-
-define void @test_int_x86_avx10_mask_pmovssdb_store_128(ptr %ptr, <4 x i32> %a, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_pmovssdb_store_128:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
-; X64-NEXT:    vpmovssdb %xmm0, (%rdi) # encoding: [0x62,0xf2,0x7e,0x08,0x41,0x07]
-; X64-NEXT:    vpmovssdb %xmm0, (%rdi) {%k1} # encoding: [0x62,0xf2,0x7e,0x09,0x41,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_pmovssdb_store_128:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vpmovssdb %xmm0, (%eax) # encoding: [0x62,0xf2,0x7e,0x08,0x41,0x00]
-; X86-NEXT:    vpmovssdb %xmm0, (%eax) {%k1} # encoding: [0x62,0xf2,0x7e,0x09,0x41,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  call void @llvm.x86.avx10.mask.pmovss.db.mem.128(ptr %ptr, <4 x i32> %a, i8 -1)
-  call void @llvm.x86.avx10.mask.pmovss.db.mem.128(ptr %ptr, <4 x i32> %a, i8 %mask)
-  ret void
-}
-
-define void @test_int_x86_avx10_mask_pmovssdb_store_256(ptr %ptr, <8 x i32> %a, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_pmovssdb_store_256:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
-; X64-NEXT:    vpmovssdb %ymm0, (%rdi) # encoding: [0x62,0xf2,0x7e,0x28,0x41,0x07]
-; X64-NEXT:    vpmovssdb %ymm0, (%rdi) {%k1} # encoding: [0x62,0xf2,0x7e,0x29,0x41,0x07]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_pmovssdb_store_256:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vpmovssdb %ymm0, (%eax) # encoding: [0x62,0xf2,0x7e,0x28,0x41,0x00]
-; X86-NEXT:    vpmovssdb %ymm0, (%eax) {%k1} # encoding: [0x62,0xf2,0x7e,0x29,0x41,0x00]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  call void @llvm.x86.avx10.mask.pmovss.db.mem.256(ptr %ptr, <8 x i32> %a, i8 -1)
-  call void @llvm.x86.avx10.mask.pmovss.db.mem.256(ptr %ptr, <8 x i32> %a, i8 %mask)
-  ret void
-}
-
-define void @test_int_x86_avx10_mask_pmovssdb_store_512(ptr %ptr, <16 x i32> %a, i16 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_pmovssdb_store_512:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
-; X64-NEXT:    vpmovssdb %zmm0, (%rdi) # encoding: [0x62,0xf2,0x7e,0x48,0x41,0x07]
-; X64-NEXT:    vpmovssdb %zmm0, (%rdi) {%k1} # encoding: [0x62,0xf2,0x7e,0x49,0x41,0x07]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_pmovssdb_store_512:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x08]
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vpmovssdb %zmm0, (%eax) # encoding: [0x62,0xf2,0x7e,0x48,0x41,0x00]
-; X86-NEXT:    vpmovssdb %zmm0, (%eax) {%k1} # encoding: [0x62,0xf2,0x7e,0x49,0x41,0x00]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  call void @llvm.x86.avx10.mask.pmovss.db.mem.512(ptr %ptr, <16 x i32> %a, i16 -1)
-  call void @llvm.x86.avx10.mask.pmovss.db.mem.512(ptr %ptr, <16 x i32> %a, i16 %mask)
-  ret void
-}
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtps2bf8128(<4 x float> %b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtps2bf8128:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtps2bf8 %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x39,0xc8]
-; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtps2bf8128:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtps2bf8 %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x39,0xc8]
-; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8128(<4 x float> %b, <16 x i8> %src0, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2bf8128(<4 x float> %b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8128:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtps2bf8 %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x89,0x39,0xc0]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8128:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtps2bf8 %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x89,0x39,0xc0]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8128(<4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtps2bf8128_mem(ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtps2bf8128_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
-; X64-NEXT:    vcvtps2bf8x (%rdi), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x39,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtps2bf8128_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
-; X86-NEXT:    vcvtps2bf8x (%eax), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x39,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %b = load <4 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8128(<4 x float> %b, <16 x i8> %src0, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2bf8128_bcst(ptr %ptr_b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8128_bcst:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
-; X64-NEXT:    vcvtps2bf8 (%rdi){1to4}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x99,0x39,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8128_bcst:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
-; X86-NEXT:    vcvtps2bf8 (%eax){1to4}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x99,0x39,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ld = load float, ptr %ptr_b
-  %ins = insertelement <4 x float> poison, float %ld, i32 0
-  %b = shufflevector <4 x float> %ins, <4 x float> poison, <4 x i32> zeroinitializer
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8128(<4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtps2bf8256(<8 x float> %b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtps2bf8256:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtps2bf8 %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x39,0xc8]
-; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtps2bf8256:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtps2bf8 %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x39,0xc8]
-; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8256(<8 x float> %b, <16 x i8> %src0, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2bf8256(<8 x float> %b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8256:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtps2bf8 %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xa9,0x39,0xc0]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8256:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtps2bf8 %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xa9,0x39,0xc0]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8256(<8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtps2bf8256_mem(ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtps2bf8256_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
-; X64-NEXT:    vcvtps2bf8y (%rdi), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x39,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtps2bf8256_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
-; X86-NEXT:    vcvtps2bf8y (%eax), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x39,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %b = load <8 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8256(<8 x float> %b, <16 x i8> %src0, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2bf8256_bcst(ptr %ptr_b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8256_bcst:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
-; X64-NEXT:    vcvtps2bf8 (%rdi){1to8}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xb9,0x39,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8256_bcst:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
-; X86-NEXT:    vcvtps2bf8 (%eax){1to8}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xb9,0x39,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ld = load float, ptr %ptr_b
-  %ins = insertelement <8 x float> poison, float %ld, i32 0
-  %b = shufflevector <8 x float> %ins, <8 x float> poison, <8 x i32> zeroinitializer
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8256(<8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8128(<4 x float>, <16 x i8>, i8)
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8256(<8 x float>, <16 x i8>, i8)
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtps2bf8s128(<4 x float> %b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtps2bf8s128:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtps2bf8s %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x3b,0xc8]
-; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtps2bf8s128:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtps2bf8s %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x3b,0xc8]
-; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s128(<4 x float> %b, <16 x i8> %src0, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2bf8s128(<4 x float> %b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8s128:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtps2bf8s %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x89,0x3b,0xc0]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8s128:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtps2bf8s %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x89,0x3b,0xc0]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s128(<4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtps2bf8s128_mem(ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtps2bf8s128_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
-; X64-NEXT:    vcvtps2bf8sx (%rdi), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x3b,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtps2bf8s128_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
-; X86-NEXT:    vcvtps2bf8sx (%eax), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x3b,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %b = load <4 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s128(<4 x float> %b, <16 x i8> %src0, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2bf8s128_bcst(ptr %ptr_b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8s128_bcst:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
-; X64-NEXT:    vcvtps2bf8s (%rdi){1to4}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x99,0x3b,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8s128_bcst:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
-; X86-NEXT:    vcvtps2bf8s (%eax){1to4}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x99,0x3b,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ld = load float, ptr %ptr_b
-  %ins = insertelement <4 x float> poison, float %ld, i32 0
-  %b = shufflevector <4 x float> %ins, <4 x float> poison, <4 x i32> zeroinitializer
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s128(<4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtps2bf8s256(<8 x float> %b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtps2bf8s256:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtps2bf8s %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x3b,0xc8]
-; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtps2bf8s256:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtps2bf8s %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x3b,0xc8]
-; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s256(<8 x float> %b, <16 x i8> %src0, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2bf8s256(<8 x float> %b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8s256:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtps2bf8s %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xa9,0x3b,0xc0]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8s256:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtps2bf8s %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xa9,0x3b,0xc0]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s256(<8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtps2bf8s256_mem(ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtps2bf8s256_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
-; X64-NEXT:    vcvtps2bf8sy (%rdi), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x3b,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtps2bf8s256_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
-; X86-NEXT:    vcvtps2bf8sy (%eax), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x3b,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %b = load <8 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s256(<8 x float> %b, <16 x i8> %src0, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2bf8s256_bcst(ptr %ptr_b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8s256_bcst:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
-; X64-NEXT:    vcvtps2bf8s (%rdi){1to8}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xb9,0x3b,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8s256_bcst:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
-; X86-NEXT:    vcvtps2bf8s (%eax){1to8}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xb9,0x3b,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ld = load float, ptr %ptr_b
-  %ins = insertelement <8 x float> poison, float %ld, i32 0
-  %b = shufflevector <8 x float> %ins, <8 x float> poison, <8 x i32> zeroinitializer
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s256(<8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s128(<4 x float>, <16 x i8>, i8)
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s256(<8 x float>, <16 x i8>, i8)
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtps2hf8128(<4 x float> %b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtps2hf8128:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtps2hf8 %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x38,0xc8]
-; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtps2hf8128:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtps2hf8 %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x38,0xc8]
-; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8128(<4 x float> %b, <16 x i8> %src0, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2hf8128(<4 x float> %b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8128:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtps2hf8 %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x89,0x38,0xc0]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8128:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtps2hf8 %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x89,0x38,0xc0]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8128(<4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtps2hf8128_mem(ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtps2hf8128_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
-; X64-NEXT:    vcvtps2hf8x (%rdi), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x38,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtps2hf8128_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
-; X86-NEXT:    vcvtps2hf8x (%eax), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x38,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %b = load <4 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8128(<4 x float> %b, <16 x i8> %src0, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2hf8128_bcst(ptr %ptr_b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8128_bcst:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
-; X64-NEXT:    vcvtps2hf8 (%rdi){1to4}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x99,0x38,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8128_bcst:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
-; X86-NEXT:    vcvtps2hf8 (%eax){1to4}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x99,0x38,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ld = load float, ptr %ptr_b
-  %ins = insertelement <4 x float> poison, float %ld, i32 0
-  %b = shufflevector <4 x float> %ins, <4 x float> poison, <4 x i32> zeroinitializer
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8128(<4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtps2hf8256(<8 x float> %b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtps2hf8256:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtps2hf8 %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x38,0xc8]
-; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtps2hf8256:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtps2hf8 %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x38,0xc8]
-; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8256(<8 x float> %b, <16 x i8> %src0, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2hf8256(<8 x float> %b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8256:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtps2hf8 %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xa9,0x38,0xc0]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8256:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtps2hf8 %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xa9,0x38,0xc0]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8256(<8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtps2hf8256_mem(ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtps2hf8256_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
-; X64-NEXT:    vcvtps2hf8y (%rdi), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x38,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtps2hf8256_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
-; X86-NEXT:    vcvtps2hf8y (%eax), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x38,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %b = load <8 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8256(<8 x float> %b, <16 x i8> %src0, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2hf8256_bcst(ptr %ptr_b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8256_bcst:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
-; X64-NEXT:    vcvtps2hf8 (%rdi){1to8}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xb9,0x38,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8256_bcst:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
-; X86-NEXT:    vcvtps2hf8 (%eax){1to8}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xb9,0x38,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ld = load float, ptr %ptr_b
-  %ins = insertelement <8 x float> poison, float %ld, i32 0
-  %b = shufflevector <8 x float> %ins, <8 x float> poison, <8 x i32> zeroinitializer
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8256(<8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8128(<4 x float>, <16 x i8>, i8)
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8256(<8 x float>, <16 x i8>, i8)
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtps2hf8s128(<4 x float> %b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtps2hf8s128:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtps2hf8s %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x3a,0xc8]
-; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtps2hf8s128:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtps2hf8s %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x3a,0xc8]
-; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s128(<4 x float> %b, <16 x i8> %src0, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2hf8s128(<4 x float> %b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8s128:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtps2hf8s %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x89,0x3a,0xc0]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8s128:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtps2hf8s %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x89,0x3a,0xc0]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s128(<4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtps2hf8s128_mem(ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtps2hf8s128_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
-; X64-NEXT:    vcvtps2hf8sx (%rdi), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x3a,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtps2hf8s128_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
-; X86-NEXT:    vcvtps2hf8sx (%eax), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x3a,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %b = load <4 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s128(<4 x float> %b, <16 x i8> %src0, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2hf8s128_bcst(ptr %ptr_b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8s128_bcst:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
-; X64-NEXT:    vcvtps2hf8s (%rdi){1to4}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x99,0x3a,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8s128_bcst:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
-; X86-NEXT:    vcvtps2hf8s (%eax){1to4}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x99,0x3a,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ld = load float, ptr %ptr_b
-  %ins = insertelement <4 x float> poison, float %ld, i32 0
-  %b = shufflevector <4 x float> %ins, <4 x float> poison, <4 x i32> zeroinitializer
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s128(<4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtps2hf8s256(<8 x float> %b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtps2hf8s256:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtps2hf8s %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x3a,0xc8]
-; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtps2hf8s256:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtps2hf8s %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x3a,0xc8]
-; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s256(<8 x float> %b, <16 x i8> %src0, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2hf8s256(<8 x float> %b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8s256:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtps2hf8s %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xa9,0x3a,0xc0]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8s256:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtps2hf8s %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xa9,0x3a,0xc0]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s256(<8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtps2hf8s256_mem(ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtps2hf8s256_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
-; X64-NEXT:    vcvtps2hf8sy (%rdi), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x3a,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtps2hf8s256_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
-; X86-NEXT:    vcvtps2hf8sy (%eax), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x3a,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %b = load <8 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s256(<8 x float> %b, <16 x i8> %src0, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2hf8s256_bcst(ptr %ptr_b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8s256_bcst:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
-; X64-NEXT:    vcvtps2hf8s (%rdi){1to8}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xb9,0x3a,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8s256_bcst:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
-; X86-NEXT:    vcvtps2hf8s (%eax){1to8}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xb9,0x3a,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ld = load float, ptr %ptr_b
-  %ins = insertelement <8 x float> poison, float %ld, i32 0
-  %b = shufflevector <8 x float> %ins, <8 x float> poison, <8 x i32> zeroinitializer
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s256(<8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s128(<4 x float>, <16 x i8>, i8)
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s256(<8 x float>, <16 x i8>, i8)
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtrops2hf8128(<4 x float> %b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8128:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtrops2hf8 %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7d,0x09,0x38,0xc8]
-; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8128:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtrops2hf8 %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7d,0x09,0x38,0xc8]
-; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8128(<4 x float> %b, <16 x i8> %src0, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtrops2hf8128(<4 x float> %b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8128:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtrops2hf8 %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0x89,0x38,0xc0]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8128:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtrops2hf8 %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0x89,0x38,0xc0]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8128(<4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtrops2hf8128_mem(ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8128_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
-; X64-NEXT:    vcvtrops2hf8x (%rdi), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7d,0x09,0x38,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8128_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
-; X86-NEXT:    vcvtrops2hf8x (%eax), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7d,0x09,0x38,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %b = load <4 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8128(<4 x float> %b, <16 x i8> %src0, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtrops2hf8128_bcst(ptr %ptr_b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8128_bcst:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
-; X64-NEXT:    vcvtrops2hf8 (%rdi){1to4}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0x99,0x38,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8128_bcst:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
-; X86-NEXT:    vcvtrops2hf8 (%eax){1to4}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0x99,0x38,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ld = load float, ptr %ptr_b
-  %ins = insertelement <4 x float> poison, float %ld, i32 0
-  %b = shufflevector <4 x float> %ins, <4 x float> poison, <4 x i32> zeroinitializer
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8128(<4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtrops2hf8256(<8 x float> %b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8256:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtrops2hf8 %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7d,0x29,0x38,0xc8]
-; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8256:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtrops2hf8 %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7d,0x29,0x38,0xc8]
-; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8256(<8 x float> %b, <16 x i8> %src0, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtrops2hf8256(<8 x float> %b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8256:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtrops2hf8 %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0xa9,0x38,0xc0]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8256:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtrops2hf8 %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0xa9,0x38,0xc0]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8256(<8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtrops2hf8256_mem(ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8256_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
-; X64-NEXT:    vcvtrops2hf8y (%rdi), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7d,0x29,0x38,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8256_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
-; X86-NEXT:    vcvtrops2hf8y (%eax), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7d,0x29,0x38,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %b = load <8 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8256(<8 x float> %b, <16 x i8> %src0, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtrops2hf8256_bcst(ptr %ptr_b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8256_bcst:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
-; X64-NEXT:    vcvtrops2hf8 (%rdi){1to8}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0xb9,0x38,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8256_bcst:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
-; X86-NEXT:    vcvtrops2hf8 (%eax){1to8}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0xb9,0x38,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ld = load float, ptr %ptr_b
-  %ins = insertelement <8 x float> poison, float %ld, i32 0
-  %b = shufflevector <8 x float> %ins, <8 x float> poison, <8 x i32> zeroinitializer
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8256(<8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8128(<4 x float>, <16 x i8>, i8)
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8256(<8 x float>, <16 x i8>, i8)
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtrops2hf8s128(<4 x float> %b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8s128:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtrops2hf8s %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7d,0x09,0x3a,0xc8]
-; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8s128:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtrops2hf8s %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7d,0x09,0x3a,0xc8]
-; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s128(<4 x float> %b, <16 x i8> %src0, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtrops2hf8s128(<4 x float> %b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8s128:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtrops2hf8s %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0x89,0x3a,0xc0]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8s128:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtrops2hf8s %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0x89,0x3a,0xc0]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s128(<4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtrops2hf8s128_mem(ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8s128_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
-; X64-NEXT:    vcvtrops2hf8sx (%rdi), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7d,0x09,0x3a,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8s128_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
-; X86-NEXT:    vcvtrops2hf8sx (%eax), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7d,0x09,0x3a,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %b = load <4 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s128(<4 x float> %b, <16 x i8> %src0, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtrops2hf8s128_bcst(ptr %ptr_b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8s128_bcst:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
-; X64-NEXT:    vcvtrops2hf8s (%rdi){1to4}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0x99,0x3a,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8s128_bcst:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
-; X86-NEXT:    vcvtrops2hf8s (%eax){1to4}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0x99,0x3a,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ld = load float, ptr %ptr_b
-  %ins = insertelement <4 x float> poison, float %ld, i32 0
-  %b = shufflevector <4 x float> %ins, <4 x float> poison, <4 x i32> zeroinitializer
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s128(<4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtrops2hf8s256(<8 x float> %b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8s256:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtrops2hf8s %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7d,0x29,0x3a,0xc8]
-; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8s256:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtrops2hf8s %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7d,0x29,0x3a,0xc8]
-; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s256(<8 x float> %b, <16 x i8> %src0, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtrops2hf8s256(<8 x float> %b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8s256:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtrops2hf8s %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0xa9,0x3a,0xc0]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8s256:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtrops2hf8s %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0xa9,0x3a,0xc0]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s256(<8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtrops2hf8s256_mem(ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8s256_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
-; X64-NEXT:    vcvtrops2hf8sy (%rdi), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7d,0x29,0x3a,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8s256_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
-; X86-NEXT:    vcvtrops2hf8sy (%eax), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7d,0x29,0x3a,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %b = load <8 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s256(<8 x float> %b, <16 x i8> %src0, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtrops2hf8s256_bcst(ptr %ptr_b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8s256_bcst:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
-; X64-NEXT:    vcvtrops2hf8s (%rdi){1to8}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0xb9,0x3a,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8s256_bcst:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
-; X86-NEXT:    vcvtrops2hf8s (%eax){1to8}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0xb9,0x3a,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ld = load float, ptr %ptr_b
-  %ins = insertelement <8 x float> poison, float %ld, i32 0
-  %b = shufflevector <8 x float> %ins, <8 x float> poison, <8 x i32> zeroinitializer
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s256(<8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s128(<4 x float>, <16 x i8>, i8)
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s256(<8 x float>, <16 x i8>, i8)
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2bf8128(<4 x i32> %A, <4 x float> %b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8128:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtbiasps2bf8 %xmm1, %xmm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x39,0xd1]
-; X64-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8128:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtbiasps2bf8 %xmm1, %xmm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x39,0xd1]
-; X86-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8128(<4 x i32> %A, <4 x float> %b, <16 x i8> %src0, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2bf8128(<4 x i32> %A, <4 x float> %b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8128:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtbiasps2bf8 %xmm1, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x39,0xc1]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8128:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtbiasps2bf8 %xmm1, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x39,0xc1]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8128(<4 x i32> %A, <4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2bf8128_mem(<4 x i32> %A, ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8128_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
-; X64-NEXT:    vcvtbiasps2bf8 (%rdi), %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x39,0x0f]
-; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8128_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
-; X86-NEXT:    vcvtbiasps2bf8 (%eax), %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x39,0x08]
-; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %b = load <4 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8128(<4 x i32> %A, <4 x float> %b, <16 x i8> %src0, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2bf8128_bcst(<4 x i32> %A, ptr %ptr_b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8128_bcst:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
-; X64-NEXT:    vcvtbiasps2bf8 (%rdi){1to4}, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x99,0x39,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8128_bcst:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
-; X86-NEXT:    vcvtbiasps2bf8 (%eax){1to4}, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x99,0x39,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ld = load float, ptr %ptr_b
-  %ins = insertelement <4 x float> poison, float %ld, i32 0
-  %b = shufflevector <4 x float> %ins, <4 x float> poison, <4 x i32> zeroinitializer
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8128(<4 x i32> %A, <4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2bf8256(<8 x i32> %A, <8 x float> %b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8256:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtbiasps2bf8 %ymm1, %ymm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x39,0xd1]
-; X64-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8256:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtbiasps2bf8 %ymm1, %ymm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x39,0xd1]
-; X86-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8256(<8 x i32> %A, <8 x float> %b, <16 x i8> %src0, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2bf8256(<8 x i32> %A, <8 x float> %b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8256:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtbiasps2bf8 %ymm1, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xa9,0x39,0xc1]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8256:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtbiasps2bf8 %ymm1, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xa9,0x39,0xc1]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8256(<8 x i32> %A, <8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2bf8256_mem(<8 x i32> %A, ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8256_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
-; X64-NEXT:    vcvtbiasps2bf8 (%rdi), %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x39,0x0f]
-; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8256_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
-; X86-NEXT:    vcvtbiasps2bf8 (%eax), %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x39,0x08]
-; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %b = load <8 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8256(<8 x i32> %A, <8 x float> %b, <16 x i8> %src0, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2bf8256_bcst(<8 x i32> %A, ptr %ptr_b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8256_bcst:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
-; X64-NEXT:    vcvtbiasps2bf8 (%rdi){1to8}, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xb9,0x39,0x07]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8256_bcst:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
-; X86-NEXT:    vcvtbiasps2bf8 (%eax){1to8}, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xb9,0x39,0x00]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ld = load float, ptr %ptr_b
-  %ins = insertelement <8 x float> poison, float %ld, i32 0
-  %b = shufflevector <8 x float> %ins, <8 x float> poison, <8 x i32> zeroinitializer
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8256(<8 x i32> %A, <8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8128(<4 x i32>, <4 x float>, <16 x i8>, i8)
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8256(<8 x i32>, <8 x float>, <16 x i8>, i8)
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2bf8s128(<4 x i32> %A, <4 x float> %b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8s128:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtbiasps2bf8s %xmm1, %xmm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x3b,0xd1]
-; X64-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8s128:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtbiasps2bf8s %xmm1, %xmm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x3b,0xd1]
-; X86-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s128(<4 x i32> %A, <4 x float> %b, <16 x i8> %src0, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2bf8s128(<4 x i32> %A, <4 x float> %b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8s128:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtbiasps2bf8s %xmm1, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x3b,0xc1]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8s128:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtbiasps2bf8s %xmm1, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x3b,0xc1]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s128(<4 x i32> %A, <4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2bf8s128_mem(<4 x i32> %A, ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8s128_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
-; X64-NEXT:    vcvtbiasps2bf8s (%rdi), %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x3b,0x0f]
-; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8s128_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
-; X86-NEXT:    vcvtbiasps2bf8s (%eax), %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x3b,0x08]
-; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %b = load <4 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s128(<4 x i32> %A, <4 x float> %b, <16 x i8> %src0, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2bf8s128_bcst(<4 x i32> %A, ptr %ptr_b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8s128_bcst:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
-; X64-NEXT:    vcvtbiasps2bf8s (%rdi){1to4}, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x99,0x3b,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8s128_bcst:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
-; X86-NEXT:    vcvtbiasps2bf8s (%eax){1to4}, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x99,0x3b,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ld = load float, ptr %ptr_b
-  %ins = insertelement <4 x float> poison, float %ld, i32 0
-  %b = shufflevector <4 x float> %ins, <4 x float> poison, <4 x i32> zeroinitializer
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s128(<4 x i32> %A, <4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2bf8s256(<8 x i32> %A, <8 x float> %b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8s256:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtbiasps2bf8s %ymm1, %ymm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x3b,0xd1]
-; X64-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8s256:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtbiasps2bf8s %ymm1, %ymm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x3b,0xd1]
-; X86-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s256(<8 x i32> %A, <8 x float> %b, <16 x i8> %src0, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2bf8s256(<8 x i32> %A, <8 x float> %b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8s256:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtbiasps2bf8s %ymm1, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xa9,0x3b,0xc1]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8s256:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtbiasps2bf8s %ymm1, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xa9,0x3b,0xc1]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s256(<8 x i32> %A, <8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2bf8s256_mem(<8 x i32> %A, ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8s256_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
-; X64-NEXT:    vcvtbiasps2bf8s (%rdi), %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x3b,0x0f]
-; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8s256_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
-; X86-NEXT:    vcvtbiasps2bf8s (%eax), %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x3b,0x08]
-; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %b = load <8 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s256(<8 x i32> %A, <8 x float> %b, <16 x i8> %src0, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2bf8s256_bcst(<8 x i32> %A, ptr %ptr_b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8s256_bcst:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
-; X64-NEXT:    vcvtbiasps2bf8s (%rdi){1to8}, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xb9,0x3b,0x07]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8s256_bcst:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
-; X86-NEXT:    vcvtbiasps2bf8s (%eax){1to8}, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xb9,0x3b,0x00]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ld = load float, ptr %ptr_b
-  %ins = insertelement <8 x float> poison, float %ld, i32 0
-  %b = shufflevector <8 x float> %ins, <8 x float> poison, <8 x i32> zeroinitializer
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s256(<8 x i32> %A, <8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s128(<4 x i32>, <4 x float>, <16 x i8>, i8)
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s256(<8 x i32>, <8 x float>, <16 x i8>, i8)
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2hf8128(<4 x i32> %A, <4 x float> %b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8128:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtbiasps2hf8 %xmm1, %xmm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x38,0xd1]
-; X64-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8128:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtbiasps2hf8 %xmm1, %xmm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x38,0xd1]
-; X86-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8128(<4 x i32> %A, <4 x float> %b, <16 x i8> %src0, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2hf8128(<4 x i32> %A, <4 x float> %b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8128:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtbiasps2hf8 %xmm1, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x38,0xc1]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8128:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtbiasps2hf8 %xmm1, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x38,0xc1]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8128(<4 x i32> %A, <4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2hf8128_mem(<4 x i32> %A, ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8128_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
-; X64-NEXT:    vcvtbiasps2hf8 (%rdi), %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x38,0x0f]
-; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8128_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
-; X86-NEXT:    vcvtbiasps2hf8 (%eax), %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x38,0x08]
-; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %b = load <4 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8128(<4 x i32> %A, <4 x float> %b, <16 x i8> %src0, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2hf8128_bcst(<4 x i32> %A, ptr %ptr_b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8128_bcst:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
-; X64-NEXT:    vcvtbiasps2hf8 (%rdi){1to4}, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x99,0x38,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8128_bcst:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
-; X86-NEXT:    vcvtbiasps2hf8 (%eax){1to4}, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x99,0x38,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ld = load float, ptr %ptr_b
-  %ins = insertelement <4 x float> poison, float %ld, i32 0
-  %b = shufflevector <4 x float> %ins, <4 x float> poison, <4 x i32> zeroinitializer
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8128(<4 x i32> %A, <4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2hf8256(<8 x i32> %A, <8 x float> %b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8256:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtbiasps2hf8 %ymm1, %ymm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x38,0xd1]
-; X64-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8256:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtbiasps2hf8 %ymm1, %ymm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x38,0xd1]
-; X86-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8256(<8 x i32> %A, <8 x float> %b, <16 x i8> %src0, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2hf8256(<8 x i32> %A, <8 x float> %b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8256:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtbiasps2hf8 %ymm1, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xa9,0x38,0xc1]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8256:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtbiasps2hf8 %ymm1, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xa9,0x38,0xc1]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8256(<8 x i32> %A, <8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2hf8256_mem(<8 x i32> %A, ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8256_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
-; X64-NEXT:    vcvtbiasps2hf8 (%rdi), %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x38,0x0f]
-; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8256_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
-; X86-NEXT:    vcvtbiasps2hf8 (%eax), %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x38,0x08]
-; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %b = load <8 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8256(<8 x i32> %A, <8 x float> %b, <16 x i8> %src0, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2hf8256_bcst(<8 x i32> %A, ptr %ptr_b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8256_bcst:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
-; X64-NEXT:    vcvtbiasps2hf8 (%rdi){1to8}, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xb9,0x38,0x07]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8256_bcst:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
-; X86-NEXT:    vcvtbiasps2hf8 (%eax){1to8}, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xb9,0x38,0x00]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ld = load float, ptr %ptr_b
-  %ins = insertelement <8 x float> poison, float %ld, i32 0
-  %b = shufflevector <8 x float> %ins, <8 x float> poison, <8 x i32> zeroinitializer
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8256(<8 x i32> %A, <8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8128(<4 x i32>, <4 x float>, <16 x i8>, i8)
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8256(<8 x i32>, <8 x float>, <16 x i8>, i8)
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2hf8s128(<4 x i32> %A, <4 x float> %b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8s128:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtbiasps2hf8s %xmm1, %xmm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x3a,0xd1]
-; X64-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8s128:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtbiasps2hf8s %xmm1, %xmm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x3a,0xd1]
-; X86-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s128(<4 x i32> %A, <4 x float> %b, <16 x i8> %src0, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2hf8s128(<4 x i32> %A, <4 x float> %b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8s128:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtbiasps2hf8s %xmm1, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x3a,0xc1]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8s128:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtbiasps2hf8s %xmm1, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x3a,0xc1]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s128(<4 x i32> %A, <4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2hf8s128_mem(<4 x i32> %A, ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8s128_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
-; X64-NEXT:    vcvtbiasps2hf8s (%rdi), %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x3a,0x0f]
-; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8s128_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
-; X86-NEXT:    vcvtbiasps2hf8s (%eax), %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x3a,0x08]
-; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %b = load <4 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s128(<4 x i32> %A, <4 x float> %b, <16 x i8> %src0, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2hf8s128_bcst(<4 x i32> %A, ptr %ptr_b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8s128_bcst:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
-; X64-NEXT:    vcvtbiasps2hf8s (%rdi){1to4}, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x99,0x3a,0x07]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8s128_bcst:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
-; X86-NEXT:    vcvtbiasps2hf8s (%eax){1to4}, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x99,0x3a,0x00]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ld = load float, ptr %ptr_b
-  %ins = insertelement <4 x float> poison, float %ld, i32 0
-  %b = shufflevector <4 x float> %ins, <4 x float> poison, <4 x i32> zeroinitializer
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s128(<4 x i32> %A, <4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2hf8s256(<8 x i32> %A, <8 x float> %b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8s256:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtbiasps2hf8s %ymm1, %ymm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x3a,0xd1]
-; X64-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8s256:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtbiasps2hf8s %ymm1, %ymm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x3a,0xd1]
-; X86-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s256(<8 x i32> %A, <8 x float> %b, <16 x i8> %src0, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2hf8s256(<8 x i32> %A, <8 x float> %b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8s256:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
-; X64-NEXT:    vcvtbiasps2hf8s %ymm1, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xa9,0x3a,0xc1]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8s256:
-; X86:       # %bb.0:
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
-; X86-NEXT:    vcvtbiasps2hf8s %ymm1, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xa9,0x3a,0xc1]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s256(<8 x i32> %A, <8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2hf8s256_mem(<8 x i32> %A, ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8s256_mem:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
-; X64-NEXT:    vcvtbiasps2hf8s (%rdi), %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x3a,0x0f]
-; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8s256_mem:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
-; X86-NEXT:    vcvtbiasps2hf8s (%eax), %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x3a,0x08]
-; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %b = load <8 x float>, ptr %ptr_b
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s256(<8 x i32> %A, <8 x float> %b, <16 x i8> %src0, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2hf8s256_bcst(<8 x i32> %A, ptr %ptr_b, i8 %mask) {
-; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8s256_bcst:
-; X64:       # %bb.0:
-; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
-; X64-NEXT:    vcvtbiasps2hf8s (%rdi){1to8}, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xb9,0x3a,0x07]
-; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X64-NEXT:    retq # encoding: [0xc3]
-;
-; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8s256_bcst:
-; X86:       # %bb.0:
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
-; X86-NEXT:    vcvtbiasps2hf8s (%eax){1to8}, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xb9,0x3a,0x00]
-; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
-; X86-NEXT:    retl # encoding: [0xc3]
-  %ld = load float, ptr %ptr_b
-  %ins = insertelement <8 x float> poison, float %ld, i32 0
-  %b = shufflevector <8 x float> %ins, <8 x float> poison, <8 x i32> zeroinitializer
-  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s256(<8 x i32> %A, <8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
-  ret <16 x i8> %ret
-}
-
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s128(<4 x i32>, <4 x float>, <16 x i8>, i8)
-declare <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s256(<8 x i32>, <8 x float>, <16 x i8>, i8)
diff --git a/llvm/test/CodeGen/X86/avx10_v2aux-mask-bias-ps-fp8-intrinsics.ll b/llvm/test/CodeGen/X86/avx10_v2aux-mask-bias-ps-fp8-intrinsics.ll
new file mode 100644
index 00000000000000..9b7b91deb92d3d
--- /dev/null
+++ b/llvm/test/CodeGen/X86/avx10_v2aux-mask-bias-ps-fp8-intrinsics.ll
@@ -0,0 +1,639 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
+; RUN: llc < %s -verify-machineinstrs -mtriple=x86_64-unknown-unknown --show-mc-encoding -mattr=+avx10v2aux | FileCheck %s --check-prefix=X64
+; RUN: llc < %s -verify-machineinstrs -mtriple=i686-unknown-unknown --show-mc-encoding -mattr=+avx10v2aux | FileCheck %s --check-prefix=X86
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2bf8128(<4 x i32> %A, <4 x float> %b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbiasps2bf8 %xmm1, %xmm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x39,0xd1]
+; X64-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2bf8 %xmm1, %xmm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x39,0xd1]
+; X86-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8128(<4 x i32> %A, <4 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2bf8128(<4 x i32> %A, <4 x float> %b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbiasps2bf8 %xmm1, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x39,0xc1]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2bf8 %xmm1, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x39,0xc1]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8128(<4 x i32> %A, <4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2bf8128_mem(<4 x i32> %A, ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8128_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtbiasps2bf8 (%rdi), %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x39,0x0f]
+; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8128_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtbiasps2bf8 (%eax), %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x39,0x08]
+; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %b = load <4 x float>, ptr %ptr_b
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8128(<4 x i32> %A, <4 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2bf8128_bcst(<4 x i32> %A, ptr %ptr_b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8128_bcst:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtbiasps2bf8 (%rdi){1to4}, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x99,0x39,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8128_bcst:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtbiasps2bf8 (%eax){1to4}, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x99,0x39,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ld = load float, ptr %ptr_b
+  %ins = insertelement <4 x float> poison, float %ld, i32 0
+  %b = shufflevector <4 x float> %ins, <4 x float> poison, <4 x i32> zeroinitializer
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8128(<4 x i32> %A, <4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2bf8256(<8 x i32> %A, <8 x float> %b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbiasps2bf8 %ymm1, %ymm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x39,0xd1]
+; X64-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2bf8 %ymm1, %ymm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x39,0xd1]
+; X86-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8256(<8 x i32> %A, <8 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2bf8256(<8 x i32> %A, <8 x float> %b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbiasps2bf8 %ymm1, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xa9,0x39,0xc1]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2bf8 %ymm1, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xa9,0x39,0xc1]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8256(<8 x i32> %A, <8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2bf8256_mem(<8 x i32> %A, ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8256_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtbiasps2bf8 (%rdi), %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x39,0x0f]
+; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8256_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtbiasps2bf8 (%eax), %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x39,0x08]
+; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %b = load <8 x float>, ptr %ptr_b
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8256(<8 x i32> %A, <8 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2bf8256_bcst(<8 x i32> %A, ptr %ptr_b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8256_bcst:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtbiasps2bf8 (%rdi){1to8}, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xb9,0x39,0x07]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8256_bcst:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtbiasps2bf8 (%eax){1to8}, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xb9,0x39,0x00]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ld = load float, ptr %ptr_b
+  %ins = insertelement <8 x float> poison, float %ld, i32 0
+  %b = shufflevector <8 x float> %ins, <8 x float> poison, <8 x i32> zeroinitializer
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8256(<8 x i32> %A, <8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8128(<4 x i32>, <4 x float>, <16 x i8>, i8)
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8256(<8 x i32>, <8 x float>, <16 x i8>, i8)
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2bf8s128(<4 x i32> %A, <4 x float> %b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8s128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbiasps2bf8s %xmm1, %xmm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x3b,0xd1]
+; X64-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8s128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2bf8s %xmm1, %xmm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x3b,0xd1]
+; X86-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s128(<4 x i32> %A, <4 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2bf8s128(<4 x i32> %A, <4 x float> %b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8s128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbiasps2bf8s %xmm1, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x3b,0xc1]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8s128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2bf8s %xmm1, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x3b,0xc1]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s128(<4 x i32> %A, <4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2bf8s128_mem(<4 x i32> %A, ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8s128_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtbiasps2bf8s (%rdi), %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x3b,0x0f]
+; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8s128_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtbiasps2bf8s (%eax), %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x3b,0x08]
+; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %b = load <4 x float>, ptr %ptr_b
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s128(<4 x i32> %A, <4 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2bf8s128_bcst(<4 x i32> %A, ptr %ptr_b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8s128_bcst:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtbiasps2bf8s (%rdi){1to4}, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x99,0x3b,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8s128_bcst:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtbiasps2bf8s (%eax){1to4}, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x99,0x3b,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ld = load float, ptr %ptr_b
+  %ins = insertelement <4 x float> poison, float %ld, i32 0
+  %b = shufflevector <4 x float> %ins, <4 x float> poison, <4 x i32> zeroinitializer
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s128(<4 x i32> %A, <4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2bf8s256(<8 x i32> %A, <8 x float> %b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8s256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbiasps2bf8s %ymm1, %ymm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x3b,0xd1]
+; X64-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8s256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2bf8s %ymm1, %ymm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x3b,0xd1]
+; X86-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s256(<8 x i32> %A, <8 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2bf8s256(<8 x i32> %A, <8 x float> %b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8s256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbiasps2bf8s %ymm1, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xa9,0x3b,0xc1]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8s256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2bf8s %ymm1, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xa9,0x3b,0xc1]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s256(<8 x i32> %A, <8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2bf8s256_mem(<8 x i32> %A, ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8s256_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtbiasps2bf8s (%rdi), %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x3b,0x0f]
+; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2bf8s256_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtbiasps2bf8s (%eax), %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x3b,0x08]
+; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %b = load <8 x float>, ptr %ptr_b
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s256(<8 x i32> %A, <8 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2bf8s256_bcst(<8 x i32> %A, ptr %ptr_b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8s256_bcst:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtbiasps2bf8s (%rdi){1to8}, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xb9,0x3b,0x07]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2bf8s256_bcst:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtbiasps2bf8s (%eax){1to8}, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xb9,0x3b,0x00]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ld = load float, ptr %ptr_b
+  %ins = insertelement <8 x float> poison, float %ld, i32 0
+  %b = shufflevector <8 x float> %ins, <8 x float> poison, <8 x i32> zeroinitializer
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s256(<8 x i32> %A, <8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s128(<4 x i32>, <4 x float>, <16 x i8>, i8)
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s256(<8 x i32>, <8 x float>, <16 x i8>, i8)
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2hf8128(<4 x i32> %A, <4 x float> %b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbiasps2hf8 %xmm1, %xmm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x38,0xd1]
+; X64-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2hf8 %xmm1, %xmm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x38,0xd1]
+; X86-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8128(<4 x i32> %A, <4 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2hf8128(<4 x i32> %A, <4 x float> %b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbiasps2hf8 %xmm1, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x38,0xc1]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2hf8 %xmm1, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x38,0xc1]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8128(<4 x i32> %A, <4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2hf8128_mem(<4 x i32> %A, ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8128_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtbiasps2hf8 (%rdi), %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x38,0x0f]
+; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8128_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtbiasps2hf8 (%eax), %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x38,0x08]
+; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %b = load <4 x float>, ptr %ptr_b
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8128(<4 x i32> %A, <4 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2hf8128_bcst(<4 x i32> %A, ptr %ptr_b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8128_bcst:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtbiasps2hf8 (%rdi){1to4}, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x99,0x38,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8128_bcst:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtbiasps2hf8 (%eax){1to4}, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x99,0x38,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ld = load float, ptr %ptr_b
+  %ins = insertelement <4 x float> poison, float %ld, i32 0
+  %b = shufflevector <4 x float> %ins, <4 x float> poison, <4 x i32> zeroinitializer
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8128(<4 x i32> %A, <4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2hf8256(<8 x i32> %A, <8 x float> %b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbiasps2hf8 %ymm1, %ymm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x38,0xd1]
+; X64-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2hf8 %ymm1, %ymm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x38,0xd1]
+; X86-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8256(<8 x i32> %A, <8 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2hf8256(<8 x i32> %A, <8 x float> %b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbiasps2hf8 %ymm1, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xa9,0x38,0xc1]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2hf8 %ymm1, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xa9,0x38,0xc1]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8256(<8 x i32> %A, <8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2hf8256_mem(<8 x i32> %A, ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8256_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtbiasps2hf8 (%rdi), %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x38,0x0f]
+; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8256_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtbiasps2hf8 (%eax), %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x38,0x08]
+; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %b = load <8 x float>, ptr %ptr_b
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8256(<8 x i32> %A, <8 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2hf8256_bcst(<8 x i32> %A, ptr %ptr_b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8256_bcst:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtbiasps2hf8 (%rdi){1to8}, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xb9,0x38,0x07]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8256_bcst:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtbiasps2hf8 (%eax){1to8}, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xb9,0x38,0x00]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ld = load float, ptr %ptr_b
+  %ins = insertelement <8 x float> poison, float %ld, i32 0
+  %b = shufflevector <8 x float> %ins, <8 x float> poison, <8 x i32> zeroinitializer
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8256(<8 x i32> %A, <8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8128(<4 x i32>, <4 x float>, <16 x i8>, i8)
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8256(<8 x i32>, <8 x float>, <16 x i8>, i8)
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2hf8s128(<4 x i32> %A, <4 x float> %b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8s128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbiasps2hf8s %xmm1, %xmm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x3a,0xd1]
+; X64-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8s128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2hf8s %xmm1, %xmm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x3a,0xd1]
+; X86-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s128(<4 x i32> %A, <4 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2hf8s128(<4 x i32> %A, <4 x float> %b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8s128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbiasps2hf8s %xmm1, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x3a,0xc1]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8s128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2hf8s %xmm1, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x89,0x3a,0xc1]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s128(<4 x i32> %A, <4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2hf8s128_mem(<4 x i32> %A, ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8s128_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtbiasps2hf8s (%rdi), %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x3a,0x0f]
+; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8s128_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtbiasps2hf8s (%eax), %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x09,0x3a,0x08]
+; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %b = load <4 x float>, ptr %ptr_b
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s128(<4 x i32> %A, <4 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2hf8s128_bcst(<4 x i32> %A, ptr %ptr_b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8s128_bcst:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtbiasps2hf8s (%rdi){1to4}, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x99,0x3a,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8s128_bcst:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtbiasps2hf8s (%eax){1to4}, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0x99,0x3a,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ld = load float, ptr %ptr_b
+  %ins = insertelement <4 x float> poison, float %ld, i32 0
+  %b = shufflevector <4 x float> %ins, <4 x float> poison, <4 x i32> zeroinitializer
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s128(<4 x i32> %A, <4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2hf8s256(<8 x i32> %A, <8 x float> %b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8s256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbiasps2hf8s %ymm1, %ymm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x3a,0xd1]
+; X64-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8s256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2hf8s %ymm1, %ymm0, %xmm2 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x3a,0xd1]
+; X86-NEXT:    vmovaps %xmm2, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc2]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s256(<8 x i32> %A, <8 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2hf8s256(<8 x i32> %A, <8 x float> %b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8s256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbiasps2hf8s %ymm1, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xa9,0x3a,0xc1]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8s256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbiasps2hf8s %ymm1, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xa9,0x3a,0xc1]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s256(<8 x i32> %A, <8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtbiasps2hf8s256_mem(<8 x i32> %A, ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8s256_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtbiasps2hf8s (%rdi), %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x3a,0x0f]
+; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtbiasps2hf8s256_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtbiasps2hf8s (%eax), %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7c,0x29,0x3a,0x08]
+; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %b = load <8 x float>, ptr %ptr_b
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s256(<8 x i32> %A, <8 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtbiasps2hf8s256_bcst(<8 x i32> %A, ptr %ptr_b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8s256_bcst:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtbiasps2hf8s (%rdi){1to8}, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xb9,0x3a,0x07]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtbiasps2hf8s256_bcst:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtbiasps2hf8s (%eax){1to8}, %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7c,0xb9,0x3a,0x00]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ld = load float, ptr %ptr_b
+  %ins = insertelement <8 x float> poison, float %ld, i32 0
+  %b = shufflevector <8 x float> %ins, <8 x float> poison, <8 x i32> zeroinitializer
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s256(<8 x i32> %A, <8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s128(<4 x i32>, <4 x float>, <16 x i8>, i8)
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s256(<8 x i32>, <8 x float>, <16 x i8>, i8)
diff --git a/llvm/test/CodeGen/X86/avx10_v2aux-mask-cvt-ps-fp8-intrinsics.ll b/llvm/test/CodeGen/X86/avx10_v2aux-mask-cvt-ps-fp8-intrinsics.ll
new file mode 100644
index 00000000000000..c33edf45fa1f32
--- /dev/null
+++ b/llvm/test/CodeGen/X86/avx10_v2aux-mask-cvt-ps-fp8-intrinsics.ll
@@ -0,0 +1,909 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
+; RUN: llc < %s -verify-machineinstrs -mtriple=x86_64-unknown-unknown --show-mc-encoding -mattr=+avx10v2aux | FileCheck %s --check-prefix=X64
+; RUN: llc < %s -verify-machineinstrs -mtriple=i686-unknown-unknown --show-mc-encoding -mattr=+avx10v2aux | FileCheck %s --check-prefix=X86
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtps2bf8128(<4 x float> %b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtps2bf8128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtps2bf8 %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x39,0xc8]
+; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtps2bf8128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtps2bf8 %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x39,0xc8]
+; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8128(<4 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2bf8128(<4 x float> %b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtps2bf8 %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x89,0x39,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtps2bf8 %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x89,0x39,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8128(<4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtps2bf8128_mem(ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtps2bf8128_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtps2bf8x (%rdi), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x39,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtps2bf8128_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtps2bf8x (%eax), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x39,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %b = load <4 x float>, ptr %ptr_b
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8128(<4 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2bf8128_bcst(ptr %ptr_b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8128_bcst:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtps2bf8 (%rdi){1to4}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x99,0x39,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8128_bcst:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtps2bf8 (%eax){1to4}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x99,0x39,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ld = load float, ptr %ptr_b
+  %ins = insertelement <4 x float> poison, float %ld, i32 0
+  %b = shufflevector <4 x float> %ins, <4 x float> poison, <4 x i32> zeroinitializer
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8128(<4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtps2bf8256(<8 x float> %b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtps2bf8256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtps2bf8 %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x39,0xc8]
+; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtps2bf8256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtps2bf8 %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x39,0xc8]
+; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8256(<8 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2bf8256(<8 x float> %b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtps2bf8 %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xa9,0x39,0xc0]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtps2bf8 %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xa9,0x39,0xc0]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8256(<8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtps2bf8256_mem(ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtps2bf8256_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtps2bf8y (%rdi), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x39,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtps2bf8256_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtps2bf8y (%eax), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x39,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %b = load <8 x float>, ptr %ptr_b
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8256(<8 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2bf8256_bcst(ptr %ptr_b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8256_bcst:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtps2bf8 (%rdi){1to8}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xb9,0x39,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8256_bcst:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtps2bf8 (%eax){1to8}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xb9,0x39,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ld = load float, ptr %ptr_b
+  %ins = insertelement <8 x float> poison, float %ld, i32 0
+  %b = shufflevector <8 x float> %ins, <8 x float> poison, <8 x i32> zeroinitializer
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8256(<8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8128(<4 x float>, <16 x i8>, i8)
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8256(<8 x float>, <16 x i8>, i8)
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtps2bf8s128(<4 x float> %b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtps2bf8s128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtps2bf8s %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x3b,0xc8]
+; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtps2bf8s128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtps2bf8s %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x3b,0xc8]
+; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s128(<4 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2bf8s128(<4 x float> %b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8s128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtps2bf8s %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x89,0x3b,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8s128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtps2bf8s %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x89,0x3b,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s128(<4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtps2bf8s128_mem(ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtps2bf8s128_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtps2bf8sx (%rdi), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x3b,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtps2bf8s128_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtps2bf8sx (%eax), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x3b,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %b = load <4 x float>, ptr %ptr_b
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s128(<4 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2bf8s128_bcst(ptr %ptr_b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8s128_bcst:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtps2bf8s (%rdi){1to4}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x99,0x3b,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8s128_bcst:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtps2bf8s (%eax){1to4}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x99,0x3b,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ld = load float, ptr %ptr_b
+  %ins = insertelement <4 x float> poison, float %ld, i32 0
+  %b = shufflevector <4 x float> %ins, <4 x float> poison, <4 x i32> zeroinitializer
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s128(<4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtps2bf8s256(<8 x float> %b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtps2bf8s256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtps2bf8s %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x3b,0xc8]
+; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtps2bf8s256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtps2bf8s %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x3b,0xc8]
+; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s256(<8 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2bf8s256(<8 x float> %b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8s256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtps2bf8s %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xa9,0x3b,0xc0]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8s256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtps2bf8s %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xa9,0x3b,0xc0]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s256(<8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtps2bf8s256_mem(ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtps2bf8s256_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtps2bf8sy (%rdi), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x3b,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtps2bf8s256_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtps2bf8sy (%eax), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x3b,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %b = load <8 x float>, ptr %ptr_b
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s256(<8 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2bf8s256_bcst(ptr %ptr_b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8s256_bcst:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtps2bf8s (%rdi){1to8}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xb9,0x3b,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2bf8s256_bcst:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtps2bf8s (%eax){1to8}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xb9,0x3b,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ld = load float, ptr %ptr_b
+  %ins = insertelement <8 x float> poison, float %ld, i32 0
+  %b = shufflevector <8 x float> %ins, <8 x float> poison, <8 x i32> zeroinitializer
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s256(<8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s128(<4 x float>, <16 x i8>, i8)
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s256(<8 x float>, <16 x i8>, i8)
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtps2hf8128(<4 x float> %b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtps2hf8128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtps2hf8 %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x38,0xc8]
+; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtps2hf8128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtps2hf8 %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x38,0xc8]
+; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8128(<4 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2hf8128(<4 x float> %b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtps2hf8 %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x89,0x38,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtps2hf8 %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x89,0x38,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8128(<4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtps2hf8128_mem(ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtps2hf8128_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtps2hf8x (%rdi), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x38,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtps2hf8128_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtps2hf8x (%eax), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x38,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %b = load <4 x float>, ptr %ptr_b
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8128(<4 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2hf8128_bcst(ptr %ptr_b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8128_bcst:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtps2hf8 (%rdi){1to4}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x99,0x38,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8128_bcst:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtps2hf8 (%eax){1to4}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x99,0x38,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ld = load float, ptr %ptr_b
+  %ins = insertelement <4 x float> poison, float %ld, i32 0
+  %b = shufflevector <4 x float> %ins, <4 x float> poison, <4 x i32> zeroinitializer
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8128(<4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtps2hf8256(<8 x float> %b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtps2hf8256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtps2hf8 %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x38,0xc8]
+; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtps2hf8256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtps2hf8 %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x38,0xc8]
+; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8256(<8 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2hf8256(<8 x float> %b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtps2hf8 %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xa9,0x38,0xc0]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtps2hf8 %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xa9,0x38,0xc0]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8256(<8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtps2hf8256_mem(ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtps2hf8256_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtps2hf8y (%rdi), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x38,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtps2hf8256_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtps2hf8y (%eax), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x38,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %b = load <8 x float>, ptr %ptr_b
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8256(<8 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2hf8256_bcst(ptr %ptr_b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8256_bcst:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtps2hf8 (%rdi){1to8}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xb9,0x38,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8256_bcst:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtps2hf8 (%eax){1to8}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xb9,0x38,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ld = load float, ptr %ptr_b
+  %ins = insertelement <8 x float> poison, float %ld, i32 0
+  %b = shufflevector <8 x float> %ins, <8 x float> poison, <8 x i32> zeroinitializer
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8256(<8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8128(<4 x float>, <16 x i8>, i8)
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8256(<8 x float>, <16 x i8>, i8)
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtps2hf8s128(<4 x float> %b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtps2hf8s128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtps2hf8s %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x3a,0xc8]
+; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtps2hf8s128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtps2hf8s %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x3a,0xc8]
+; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s128(<4 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2hf8s128(<4 x float> %b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8s128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtps2hf8s %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x89,0x3a,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8s128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtps2hf8s %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x89,0x3a,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s128(<4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtps2hf8s128_mem(ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtps2hf8s128_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtps2hf8sx (%rdi), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x3a,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtps2hf8s128_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtps2hf8sx (%eax), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7e,0x09,0x3a,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %b = load <4 x float>, ptr %ptr_b
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s128(<4 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2hf8s128_bcst(ptr %ptr_b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8s128_bcst:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtps2hf8s (%rdi){1to4}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x99,0x3a,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8s128_bcst:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtps2hf8s (%eax){1to4}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0x99,0x3a,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ld = load float, ptr %ptr_b
+  %ins = insertelement <4 x float> poison, float %ld, i32 0
+  %b = shufflevector <4 x float> %ins, <4 x float> poison, <4 x i32> zeroinitializer
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s128(<4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtps2hf8s256(<8 x float> %b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtps2hf8s256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtps2hf8s %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x3a,0xc8]
+; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtps2hf8s256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtps2hf8s %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x3a,0xc8]
+; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s256(<8 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2hf8s256(<8 x float> %b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8s256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtps2hf8s %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xa9,0x3a,0xc0]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8s256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtps2hf8s %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xa9,0x3a,0xc0]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s256(<8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtps2hf8s256_mem(ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtps2hf8s256_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtps2hf8sy (%rdi), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x3a,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtps2hf8s256_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtps2hf8sy (%eax), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7e,0x29,0x3a,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %b = load <8 x float>, ptr %ptr_b
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s256(<8 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtps2hf8s256_bcst(ptr %ptr_b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8s256_bcst:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtps2hf8s (%rdi){1to8}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xb9,0x3a,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtps2hf8s256_bcst:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtps2hf8s (%eax){1to8}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7e,0xb9,0x3a,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ld = load float, ptr %ptr_b
+  %ins = insertelement <8 x float> poison, float %ld, i32 0
+  %b = shufflevector <8 x float> %ins, <8 x float> poison, <8 x i32> zeroinitializer
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s256(<8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s128(<4 x float>, <16 x i8>, i8)
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s256(<8 x float>, <16 x i8>, i8)
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtrops2hf8128(<4 x float> %b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtrops2hf8 %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7d,0x09,0x38,0xc8]
+; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtrops2hf8 %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7d,0x09,0x38,0xc8]
+; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8128(<4 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtrops2hf8128(<4 x float> %b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtrops2hf8 %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0x89,0x38,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtrops2hf8 %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0x89,0x38,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8128(<4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtrops2hf8128_mem(ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8128_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtrops2hf8x (%rdi), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7d,0x09,0x38,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8128_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtrops2hf8x (%eax), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7d,0x09,0x38,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %b = load <4 x float>, ptr %ptr_b
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8128(<4 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtrops2hf8128_bcst(ptr %ptr_b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8128_bcst:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtrops2hf8 (%rdi){1to4}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0x99,0x38,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8128_bcst:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtrops2hf8 (%eax){1to4}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0x99,0x38,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ld = load float, ptr %ptr_b
+  %ins = insertelement <4 x float> poison, float %ld, i32 0
+  %b = shufflevector <4 x float> %ins, <4 x float> poison, <4 x i32> zeroinitializer
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8128(<4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtrops2hf8256(<8 x float> %b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtrops2hf8 %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7d,0x29,0x38,0xc8]
+; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtrops2hf8 %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7d,0x29,0x38,0xc8]
+; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8256(<8 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtrops2hf8256(<8 x float> %b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtrops2hf8 %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0xa9,0x38,0xc0]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtrops2hf8 %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0xa9,0x38,0xc0]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8256(<8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtrops2hf8256_mem(ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8256_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtrops2hf8y (%rdi), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7d,0x29,0x38,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8256_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtrops2hf8y (%eax), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7d,0x29,0x38,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %b = load <8 x float>, ptr %ptr_b
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8256(<8 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtrops2hf8256_bcst(ptr %ptr_b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8256_bcst:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtrops2hf8 (%rdi){1to8}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0xb9,0x38,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8256_bcst:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtrops2hf8 (%eax){1to8}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0xb9,0x38,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ld = load float, ptr %ptr_b
+  %ins = insertelement <8 x float> poison, float %ld, i32 0
+  %b = shufflevector <8 x float> %ins, <8 x float> poison, <8 x i32> zeroinitializer
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8256(<8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8128(<4 x float>, <16 x i8>, i8)
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8256(<8 x float>, <16 x i8>, i8)
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtrops2hf8s128(<4 x float> %b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8s128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtrops2hf8s %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7d,0x09,0x3a,0xc8]
+; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8s128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtrops2hf8s %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7d,0x09,0x3a,0xc8]
+; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s128(<4 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtrops2hf8s128(<4 x float> %b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8s128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtrops2hf8s %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0x89,0x3a,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8s128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtrops2hf8s %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0x89,0x3a,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s128(<4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtrops2hf8s128_mem(ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8s128_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtrops2hf8sx (%rdi), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7d,0x09,0x3a,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8s128_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtrops2hf8sx (%eax), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7d,0x09,0x3a,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %b = load <4 x float>, ptr %ptr_b
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s128(<4 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtrops2hf8s128_bcst(ptr %ptr_b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8s128_bcst:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtrops2hf8s (%rdi){1to4}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0x99,0x3a,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8s128_bcst:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtrops2hf8s (%eax){1to4}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0x99,0x3a,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ld = load float, ptr %ptr_b
+  %ins = insertelement <4 x float> poison, float %ld, i32 0
+  %b = shufflevector <4 x float> %ins, <4 x float> poison, <4 x i32> zeroinitializer
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s128(<4 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtrops2hf8s256(<8 x float> %b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8s256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtrops2hf8s %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7d,0x29,0x3a,0xc8]
+; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8s256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtrops2hf8s %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7d,0x29,0x3a,0xc8]
+; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s256(<8 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtrops2hf8s256(<8 x float> %b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8s256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtrops2hf8s %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0xa9,0x3a,0xc0]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8s256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtrops2hf8s %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0xa9,0x3a,0xc0]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s256(<8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_vcvtrops2hf8s256_mem(ptr %ptr_b, <16 x i8> %src0, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8s256_mem:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtrops2hf8sy (%rdi), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7d,0x29,0x3a,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_vcvtrops2hf8s256_mem:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtrops2hf8sy (%eax), %xmm0 {%k1} # encoding: [0x62,0xf5,0x7d,0x29,0x3a,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %b = load <8 x float>, ptr %ptr_b
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s256(<8 x float> %b, <16 x i8> %src0, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_maskz_vcvtrops2hf8s256_bcst(ptr %ptr_b, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8s256_bcst:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vcvtrops2hf8s (%rdi){1to8}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0xb9,0x3a,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_maskz_vcvtrops2hf8s256_bcst:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vcvtrops2hf8s (%eax){1to8}, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0xb9,0x3a,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %ld = load float, ptr %ptr_b
+  %ins = insertelement <8 x float> poison, float %ld, i32 0
+  %b = shufflevector <8 x float> %ins, <8 x float> poison, <8 x i32> zeroinitializer
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s256(<8 x float> %b, <16 x i8> zeroinitializer, i8 %mask)
+  ret <16 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s128(<4 x float>, <16 x i8>, i8)
+declare <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s256(<8 x float>, <16 x i8>, i8)
diff --git a/llvm/test/CodeGen/X86/avx10_v2aux-pmovssdb-intrinsics.ll b/llvm/test/CodeGen/X86/avx10_v2aux-pmovssdb-intrinsics.ll
new file mode 100644
index 00000000000000..77807b58a70753
--- /dev/null
+++ b/llvm/test/CodeGen/X86/avx10_v2aux-pmovssdb-intrinsics.ll
@@ -0,0 +1,249 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
+; RUN: llc < %s -verify-machineinstrs -mtriple=x86_64-unknown-unknown --show-mc-encoding -mattr=+avx10v2aux | FileCheck %s --check-prefixes=CHECK,X64
+; RUN: llc < %s -verify-machineinstrs -mtriple=i686-unknown-unknown --show-mc-encoding -mattr=+avx10v2aux | FileCheck %s --check-prefixes=CHECK,X86
+
+define <16 x i8> @test_int_x86_avx10_pmovssdb_128(<4 x i32> %a) {
+; CHECK-LABEL: test_int_x86_avx10_pmovssdb_128:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vpmovssdb %xmm0, %xmm0 # encoding: [0x62,0xf2,0x7e,0x08,0x41,0xc0]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.128(<4 x i32> %a, <16 x i8> zeroinitializer, i8 -1)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_pmovssdb_256(<8 x i32> %a) {
+; CHECK-LABEL: test_int_x86_avx10_pmovssdb_256:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vpmovssdb %ymm0, %xmm0 # encoding: [0x62,0xf2,0x7e,0x28,0x41,0xc0]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.256(<8 x i32> %a, <16 x i8> zeroinitializer, i8 -1)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_pmovssdb_512(<16 x i32> %a) {
+; CHECK-LABEL: test_int_x86_avx10_pmovssdb_512:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vpmovssdb %zmm0, %xmm0 # encoding: [0x62,0xf2,0x7e,0x48,0x41,0xc0]
+; CHECK-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.512(<16 x i32> %a, <16 x i8> zeroinitializer, i16 -1)
+  ret <16 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.mask.pmovss.db.128(<4 x i32>, <16 x i8>, i8)
+declare <16 x i8> @llvm.x86.avx10.mask.pmovss.db.256(<8 x i32>, <16 x i8>, i8)
+declare <16 x i8> @llvm.x86.avx10.mask.pmovss.db.512(<16 x i32>, <16 x i8>, i16)
+declare void @llvm.x86.avx10.mask.pmovss.db.mem.128(ptr, <4 x i32>, i8)
+declare void @llvm.x86.avx10.mask.pmovss.db.mem.256(ptr, <8 x i32>, i8)
+declare void @llvm.x86.avx10.mask.pmovss.db.mem.512(ptr, <16 x i32>, i16)
+
+define <16 x i8> @test_int_x86_avx10_mask_pmovssdb_128(<4 x i32> %a, <16 x i8> %passthru, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_pmovssdb_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vpmovssdb %xmm0, %xmm2 # encoding: [0x62,0xf2,0x7e,0x08,0x41,0xc2]
+; X64-NEXT:    vpmovssdb %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf2,0x7e,0x09,0x41,0xc1]
+; X64-NEXT:    vpaddb %xmm1, %xmm2, %xmm1 # EVEX TO VEX Compression encoding: [0xc5,0xe9,0xfc,0xc9]
+; X64-NEXT:    vpmovssdb %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf2,0x7e,0x89,0x41,0xc0]
+; X64-NEXT:    vpaddb %xmm0, %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf1,0xfc,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_pmovssdb_128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vpmovssdb %xmm0, %xmm2 # encoding: [0x62,0xf2,0x7e,0x08,0x41,0xc2]
+; X86-NEXT:    vpmovssdb %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf2,0x7e,0x09,0x41,0xc1]
+; X86-NEXT:    vpaddb %xmm1, %xmm2, %xmm1 # EVEX TO VEX Compression encoding: [0xc5,0xe9,0xfc,0xc9]
+; X86-NEXT:    vpmovssdb %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf2,0x7e,0x89,0x41,0xc0]
+; X86-NEXT:    vpaddb %xmm0, %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf1,0xfc,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %res0 = call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.128(<4 x i32> %a, <16 x i8> %passthru, i8 -1)
+  %res1 = call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.128(<4 x i32> %a, <16 x i8> %passthru, i8 %mask)
+  %res2 = call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.128(<4 x i32> %a, <16 x i8> zeroinitializer, i8 %mask)
+  %add1 = add <16 x i8> %res0, %res1
+  %add2 = add <16 x i8> %add1, %res2
+  ret <16 x i8> %add2
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_pmovssdb_256(<8 x i32> %a, <16 x i8> %passthru, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_pmovssdb_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vpmovssdb %ymm0, %xmm2 # encoding: [0x62,0xf2,0x7e,0x28,0x41,0xc2]
+; X64-NEXT:    vpmovssdb %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf2,0x7e,0x29,0x41,0xc1]
+; X64-NEXT:    vpaddb %xmm1, %xmm2, %xmm1 # EVEX TO VEX Compression encoding: [0xc5,0xe9,0xfc,0xc9]
+; X64-NEXT:    vpmovssdb %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf2,0x7e,0xa9,0x41,0xc0]
+; X64-NEXT:    vpaddb %xmm0, %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf1,0xfc,0xc0]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_pmovssdb_256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vpmovssdb %ymm0, %xmm2 # encoding: [0x62,0xf2,0x7e,0x28,0x41,0xc2]
+; X86-NEXT:    vpmovssdb %ymm0, %xmm1 {%k1} # encoding: [0x62,0xf2,0x7e,0x29,0x41,0xc1]
+; X86-NEXT:    vpaddb %xmm1, %xmm2, %xmm1 # EVEX TO VEX Compression encoding: [0xc5,0xe9,0xfc,0xc9]
+; X86-NEXT:    vpmovssdb %ymm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf2,0x7e,0xa9,0x41,0xc0]
+; X86-NEXT:    vpaddb %xmm0, %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf1,0xfc,0xc0]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %res0 = call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.256(<8 x i32> %a, <16 x i8> %passthru, i8 -1)
+  %res1 = call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.256(<8 x i32> %a, <16 x i8> %passthru, i8 %mask)
+  %res2 = call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.256(<8 x i32> %a, <16 x i8> zeroinitializer, i8 %mask)
+  %add1 = add <16 x i8> %res0, %res1
+  %add2 = add <16 x i8> %add1, %res2
+  ret <16 x i8> %add2
+}
+
+define <16 x i8> @test_int_x86_avx10_mask_pmovssdb_512(<16 x i32> %a, <16 x i8> %passthru, i16 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_pmovssdb_512:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vpmovssdb %zmm0, %xmm2 # encoding: [0x62,0xf2,0x7e,0x48,0x41,0xc2]
+; X64-NEXT:    vpmovssdb %zmm0, %xmm1 {%k1} # encoding: [0x62,0xf2,0x7e,0x49,0x41,0xc1]
+; X64-NEXT:    vpaddb %xmm1, %xmm2, %xmm1 # EVEX TO VEX Compression encoding: [0xc5,0xe9,0xfc,0xc9]
+; X64-NEXT:    vpmovssdb %zmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf2,0x7e,0xc9,0x41,0xc0]
+; X64-NEXT:    vpaddb %xmm0, %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf1,0xfc,0xc0]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_pmovssdb_512:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vpmovssdb %zmm0, %xmm2 # encoding: [0x62,0xf2,0x7e,0x48,0x41,0xc2]
+; X86-NEXT:    vpmovssdb %zmm0, %xmm1 {%k1} # encoding: [0x62,0xf2,0x7e,0x49,0x41,0xc1]
+; X86-NEXT:    vpaddb %xmm1, %xmm2, %xmm1 # EVEX TO VEX Compression encoding: [0xc5,0xe9,0xfc,0xc9]
+; X86-NEXT:    vpmovssdb %zmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf2,0x7e,0xc9,0x41,0xc0]
+; X86-NEXT:    vpaddb %xmm0, %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf1,0xfc,0xc0]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %res0 = call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.512(<16 x i32> %a, <16 x i8> %passthru, i16 -1)
+  %res1 = call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.512(<16 x i32> %a, <16 x i8> %passthru, i16 %mask)
+  %res2 = call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.512(<16 x i32> %a, <16 x i8> zeroinitializer, i16 %mask)
+  %add1 = add <16 x i8> %res0, %res1
+  %add2 = add <16 x i8> %add1, %res2
+  ret <16 x i8> %add2
+}
+
+define <16 x i8> @test_int_x86_avx10_pmovssdb_mem_128(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_pmovssdb_mem_128:
+; X64:       # %bb.0:
+; X64-NEXT:    vmovdqa (%rdi), %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf9,0x6f,0x07]
+; X64-NEXT:    vpmovssdb %xmm0, %xmm0 # encoding: [0x62,0xf2,0x7e,0x08,0x41,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_pmovssdb_mem_128:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vmovdqa (%eax), %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf9,0x6f,0x00]
+; X86-NEXT:    vpmovssdb %xmm0, %xmm0 # encoding: [0x62,0xf2,0x7e,0x08,0x41,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <4 x i32>, ptr %ptr_a
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.128(<4 x i32> %a, <16 x i8> zeroinitializer, i8 -1)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_pmovssdb_mem_256(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_pmovssdb_mem_256:
+; X64:       # %bb.0:
+; X64-NEXT:    vmovdqa (%rdi), %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfd,0x6f,0x07]
+; X64-NEXT:    vpmovssdb %ymm0, %xmm0 # encoding: [0x62,0xf2,0x7e,0x28,0x41,0xc0]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_pmovssdb_mem_256:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vmovdqa (%eax), %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfd,0x6f,0x00]
+; X86-NEXT:    vpmovssdb %ymm0, %xmm0 # encoding: [0x62,0xf2,0x7e,0x28,0x41,0xc0]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <8 x i32>, ptr %ptr_a
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.256(<8 x i32> %a, <16 x i8> zeroinitializer, i8 -1)
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_pmovssdb_mem_512(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_pmovssdb_mem_512:
+; X64:       # %bb.0:
+; X64-NEXT:    vmovdqa64 (%rdi), %zmm0 # encoding: [0x62,0xf1,0xfd,0x48,0x6f,0x07]
+; X64-NEXT:    vpmovssdb %zmm0, %xmm0 # encoding: [0x62,0xf2,0x7e,0x48,0x41,0xc0]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_pmovssdb_mem_512:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vmovdqa64 (%eax), %zmm0 # encoding: [0x62,0xf1,0xfd,0x48,0x6f,0x00]
+; X86-NEXT:    vpmovssdb %zmm0, %xmm0 # encoding: [0x62,0xf2,0x7e,0x48,0x41,0xc0]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <16 x i32>, ptr %ptr_a
+  %ret = call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.512(<16 x i32> %a, <16 x i8> zeroinitializer, i16 -1)
+  ret <16 x i8> %ret
+}
+
+define void @test_int_x86_avx10_mask_pmovssdb_store_128(ptr %ptr, <4 x i32> %a, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_pmovssdb_store_128:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vpmovssdb %xmm0, (%rdi) # encoding: [0x62,0xf2,0x7e,0x08,0x41,0x07]
+; X64-NEXT:    vpmovssdb %xmm0, (%rdi) {%k1} # encoding: [0x62,0xf2,0x7e,0x09,0x41,0x07]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_pmovssdb_store_128:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vpmovssdb %xmm0, (%eax) # encoding: [0x62,0xf2,0x7e,0x08,0x41,0x00]
+; X86-NEXT:    vpmovssdb %xmm0, (%eax) {%k1} # encoding: [0x62,0xf2,0x7e,0x09,0x41,0x00]
+; X86-NEXT:    retl # encoding: [0xc3]
+  call void @llvm.x86.avx10.mask.pmovss.db.mem.128(ptr %ptr, <4 x i32> %a, i8 -1)
+  call void @llvm.x86.avx10.mask.pmovss.db.mem.128(ptr %ptr, <4 x i32> %a, i8 %mask)
+  ret void
+}
+
+define void @test_int_x86_avx10_mask_pmovssdb_store_256(ptr %ptr, <8 x i32> %a, i8 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_pmovssdb_store_256:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vpmovssdb %ymm0, (%rdi) # encoding: [0x62,0xf2,0x7e,0x28,0x41,0x07]
+; X64-NEXT:    vpmovssdb %ymm0, (%rdi) {%k1} # encoding: [0x62,0xf2,0x7e,0x29,0x41,0x07]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_pmovssdb_store_256:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovb {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf9,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vpmovssdb %ymm0, (%eax) # encoding: [0x62,0xf2,0x7e,0x28,0x41,0x00]
+; X86-NEXT:    vpmovssdb %ymm0, (%eax) {%k1} # encoding: [0x62,0xf2,0x7e,0x29,0x41,0x00]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  call void @llvm.x86.avx10.mask.pmovss.db.mem.256(ptr %ptr, <8 x i32> %a, i8 -1)
+  call void @llvm.x86.avx10.mask.pmovss.db.mem.256(ptr %ptr, <8 x i32> %a, i8 %mask)
+  ret void
+}
+
+define void @test_int_x86_avx10_mask_pmovssdb_store_512(ptr %ptr, <16 x i32> %a, i16 %mask) {
+; X64-LABEL: test_int_x86_avx10_mask_pmovssdb_store_512:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vpmovssdb %zmm0, (%rdi) # encoding: [0x62,0xf2,0x7e,0x48,0x41,0x07]
+; X64-NEXT:    vpmovssdb %zmm0, (%rdi) {%k1} # encoding: [0x62,0xf2,0x7e,0x49,0x41,0x07]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_mask_pmovssdb_store_512:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vpmovssdb %zmm0, (%eax) # encoding: [0x62,0xf2,0x7e,0x48,0x41,0x00]
+; X86-NEXT:    vpmovssdb %zmm0, (%eax) {%k1} # encoding: [0x62,0xf2,0x7e,0x49,0x41,0x00]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  call void @llvm.x86.avx10.mask.pmovss.db.mem.512(ptr %ptr, <16 x i32> %a, i16 -1)
+  call void @llvm.x86.avx10.mask.pmovss.db.mem.512(ptr %ptr, <16 x i32> %a, i16 %mask)
+  ret void
+}
diff --git a/llvm/test/CodeGen/X86/avx10_v2aux-unpackb-intrinsics.ll b/llvm/test/CodeGen/X86/avx10_v2aux-unpackb-intrinsics.ll
new file mode 100644
index 00000000000000..7d0292c6afd417
--- /dev/null
+++ b/llvm/test/CodeGen/X86/avx10_v2aux-unpackb-intrinsics.ll
@@ -0,0 +1,82 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
+; RUN: llc < %s -verify-machineinstrs -mtriple=x86_64-unknown-unknown --show-mc-encoding -mattr=+avx10v2aux | FileCheck %s --check-prefixes=CHECK,X64
+; RUN: llc < %s -verify-machineinstrs -mtriple=i686-unknown-unknown --show-mc-encoding -mattr=+avx10v2aux | FileCheck %s --check-prefixes=CHECK,X86
+
+define <16 x i8> @test_int_x86_avx10_vunpackb_128(<16 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vunpackb_128:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vunpackb $1, %xmm0, %xmm0 # encoding: [0x62,0xf3,0x7c,0x08,0x3d,0xc0,0x01]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <16 x i8> @llvm.x86.avx10.vunpackb.128(<16 x i8> %a, i8 1)
+  ret <16 x i8> %ret
+}
+
+define <32 x i8> @test_int_x86_avx10_vunpackb_256(<32 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vunpackb_256:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vunpackb $2, %ymm0, %ymm0 # encoding: [0x62,0xf3,0x7c,0x28,0x3d,0xc0,0x02]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <32 x i8> @llvm.x86.avx10.vunpackb.256(<32 x i8> %a, i8 2)
+  ret <32 x i8> %ret
+}
+
+define <64 x i8> @test_int_x86_avx10_vunpackb_512(<64 x i8> %a) {
+; CHECK-LABEL: test_int_x86_avx10_vunpackb_512:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vunpackb $3, %zmm0, %zmm0 # encoding: [0x62,0xf3,0x7c,0x48,0x3d,0xc0,0x03]
+; CHECK-NEXT:    ret{{[l|q]}} # encoding: [0xc3]
+  %ret = call <64 x i8> @llvm.x86.avx10.vunpackb.512(<64 x i8> %a, i8 3)
+  ret <64 x i8> %ret
+}
+
+declare <16 x i8> @llvm.x86.avx10.vunpackb.128(<16 x i8>, i8)
+declare <32 x i8> @llvm.x86.avx10.vunpackb.256(<32 x i8>, i8)
+declare <64 x i8> @llvm.x86.avx10.vunpackb.512(<64 x i8>, i8)
+
+define <16 x i8> @test_int_x86_avx10_vunpackb_mem_128(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vunpackb_mem_128:
+; X64:       # %bb.0:
+; X64-NEXT:    vunpackb $1, (%rdi), %xmm0 # encoding: [0x62,0xf3,0x7c,0x08,0x3d,0x07,0x01]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vunpackb_mem_128:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vunpackb $1, (%eax), %xmm0 # encoding: [0x62,0xf3,0x7c,0x08,0x3d,0x00,0x01]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <16 x i8>, ptr %ptr_a
+  %ret = call <16 x i8> @llvm.x86.avx10.vunpackb.128(<16 x i8> %a, i8 1)
+  ret <16 x i8> %ret
+}
+
+define <32 x i8> @test_int_x86_avx10_vunpackb_mem_256(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vunpackb_mem_256:
+; X64:       # %bb.0:
+; X64-NEXT:    vunpackb $2, (%rdi), %ymm0 # encoding: [0x62,0xf3,0x7c,0x28,0x3d,0x07,0x02]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vunpackb_mem_256:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vunpackb $2, (%eax), %ymm0 # encoding: [0x62,0xf3,0x7c,0x28,0x3d,0x00,0x02]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <32 x i8>, ptr %ptr_a
+  %ret = call <32 x i8> @llvm.x86.avx10.vunpackb.256(<32 x i8> %a, i8 2)
+  ret <32 x i8> %ret
+}
+
+define <64 x i8> @test_int_x86_avx10_vunpackb_mem_512(ptr %ptr_a) {
+; X64-LABEL: test_int_x86_avx10_vunpackb_mem_512:
+; X64:       # %bb.0:
+; X64-NEXT:    vunpackb $3, (%rdi), %zmm0 # encoding: [0x62,0xf3,0x7c,0x48,0x3d,0x07,0x03]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vunpackb_mem_512:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vunpackb $3, (%eax), %zmm0 # encoding: [0x62,0xf3,0x7c,0x48,0x3d,0x00,0x03]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <64 x i8>, ptr %ptr_a
+  %ret = call <64 x i8> @llvm.x86.avx10.vunpackb.512(<64 x i8> %a, i8 3)
+  ret <64 x i8> %ret
+}

>From 9bad74ee71fcdf7c5e2c01b8fdbc8d21c7e716be Mon Sep 17 00:00:00 2001
From: Ganesh Gopalasubramanian <Ganesh.Gopalasubramanian at amd.com>
Date: Tue, 8 Sep 2026 01:26:43 +0530
Subject: [PATCH 14/16] [X86][AVX10_V2_AUX] Align cpu_supports with GCC and
 tighten VUNPACKB immediates.

Make avx10v2aux X86_FEATURE_COMPAT with ABI bit 124 and FMV priority 0.
Detect it from CPUID 24.1 ECX[3]. Have AVX10 gate matching the spec, gcc.
Form VUNPACKB imm via _MM_UNPACKB_* helpers; reject reserved encodings in Sema.
---
 clang/include/clang/Basic/BuiltinsX86.td      | 120 ++++-----
 .../clang/Basic/DiagnosticSemaKinds.td        |   2 +
 clang/include/clang/Sema/SemaX86.h            |   1 +
 clang/lib/Headers/avx10_2_512v2auxintrin.h    |  30 ++-
 clang/lib/Headers/avx10_2_v2auxintrin.h       | 160 ++++++++----
 clang/lib/Sema/SemaX86.cpp                    |  50 +++-
 .../X86/avx10_2_v2aux-builtins-errors.c       |  28 +++
 .../test/CodeGen/X86/avx10_2_v2aux-builtins.c | 103 +++++---
 clang/test/CodeGen/attr-target-x86.c          |   2 +-
 clang/test/CodeGen/builtin-cpu-supports-all.c |   5 +
 clang/test/CodeGen/target-builtin-noerror.c   |   1 +
 clang/test/Sema/attr-target-mv.c              |   4 +
 compiler-rt/lib/builtins/cpu_model/x86.c      |  11 +-
 .../llvm/TargetParser/X86TargetParser.def     |   2 +-
 llvm/lib/Target/X86/X86ISelLowering.cpp       |  17 +-
 llvm/lib/Target/X86/X86InstrAVX10_V2_AUX.td   |  44 +++-
 llvm/lib/Target/X86/X86InstrFragmentsSIMD.td  |   4 +-
 llvm/lib/TargetParser/Host.cpp                |   3 +-
 .../X86/avx10_v2aux-cvt-fp4-fp6-intrinsics.ll | 228 ++++++++++++++++++
 .../X86/avx10_v2aux-pmovssdb-intrinsics.ll    |  18 ++
 .../X86/avx10_v2aux-unpackb-intrinsics.ll     | 154 ++++++++++++
 llvm/test/MC/X86/avx10_v2_aux-att-32.s        |  40 +++
 llvm/test/MC/X86/avx10_v2_aux-att-64.s        |  40 +++
 llvm/test/TableGen/x86-fold-tables.inc        |   3 -
 llvm/utils/TableGen/X86ManualFoldTables.def   |   3 +
 25 files changed, 907 insertions(+), 166 deletions(-)

diff --git a/clang/include/clang/Basic/BuiltinsX86.td b/clang/include/clang/Basic/BuiltinsX86.td
index 446b55d8848cf9..b96912528ac333 100644
--- a/clang/include/clang/Basic/BuiltinsX86.td
+++ b/clang/include/clang/Basic/BuiltinsX86.td
@@ -5068,293 +5068,293 @@ let Features = "avx10.2", Attributes = [NoThrow, Const, RequiredVectorWidth<512>
 // Convert from FP32 to FP8
 
 // VCVTPS2BF8
-let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<128>] in {
   def vcvtps2bf8_128 : X86Builtin<"_Vector<16, char>(_Vector<4, float>)">;
   def vcvtps2bf8_128_mask : X86Builtin<"_Vector<16, char>(_Vector<4, float>, _Vector<16, char>, unsigned char)">;
 }
 
-let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<256>] in {
   def vcvtps2bf8_256 : X86Builtin<"_Vector<16, char>(_Vector<8, float>)">;
   def vcvtps2bf8_256_mask : X86Builtin<"_Vector<16, char>(_Vector<8, float>, _Vector<16, char>, unsigned char)">;
 }
 
-let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<512>] in {
   def vcvtps2bf8_512 : X86Builtin<"_Vector<16, char>(_Vector<16, float>)">;
 }
 
 // VCVTPS2BF8S
-let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<128>] in {
   def vcvtps2bf8s_128 : X86Builtin<"_Vector<16, char>(_Vector<4, float>)">;
   def vcvtps2bf8s_128_mask : X86Builtin<"_Vector<16, char>(_Vector<4, float>, _Vector<16, char>, unsigned char)">;
 }
 
-let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<256>] in {
   def vcvtps2bf8s_256 : X86Builtin<"_Vector<16, char>(_Vector<8, float>)">;
   def vcvtps2bf8s_256_mask : X86Builtin<"_Vector<16, char>(_Vector<8, float>, _Vector<16, char>, unsigned char)">;
 }
 
-let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<512>] in {
   def vcvtps2bf8s_512 : X86Builtin<"_Vector<16, char>(_Vector<16, float>)">;
 }
 
 // VCVTPS2HF8
-let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<128>] in {
   def vcvtps2hf8_128 : X86Builtin<"_Vector<16, char>(_Vector<4, float>)">;
   def vcvtps2hf8_128_mask : X86Builtin<"_Vector<16, char>(_Vector<4, float>, _Vector<16, char>, unsigned char)">;
 }
 
-let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<256>] in {
   def vcvtps2hf8_256 : X86Builtin<"_Vector<16, char>(_Vector<8, float>)">;
   def vcvtps2hf8_256_mask : X86Builtin<"_Vector<16, char>(_Vector<8, float>, _Vector<16, char>, unsigned char)">;
 }
 
-let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<512>] in {
   def vcvtps2hf8_512 : X86Builtin<"_Vector<16, char>(_Vector<16, float>)">;
 }
 
 // VCVTPS2HF8S
-let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<128>] in {
   def vcvtps2hf8s_128 : X86Builtin<"_Vector<16, char>(_Vector<4, float>)">;
   def vcvtps2hf8s_128_mask : X86Builtin<"_Vector<16, char>(_Vector<4, float>, _Vector<16, char>, unsigned char)">;
 }
 
-let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<256>] in {
   def vcvtps2hf8s_256 : X86Builtin<"_Vector<16, char>(_Vector<8, float>)">;
   def vcvtps2hf8s_256_mask : X86Builtin<"_Vector<16, char>(_Vector<8, float>, _Vector<16, char>, unsigned char)">;
 }
 
-let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<512>] in {
   def vcvtps2hf8s_512 : X86Builtin<"_Vector<16, char>(_Vector<16, float>)">;
 }
 
 // VCVTROPS2HF8
-let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<128>] in {
   def vcvtrops2hf8_128 : X86Builtin<"_Vector<16, char>(_Vector<4, float>)">;
   def vcvtrops2hf8_128_mask : X86Builtin<"_Vector<16, char>(_Vector<4, float>, _Vector<16, char>, unsigned char)">;
 }
 
-let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<256>] in {
   def vcvtrops2hf8_256 : X86Builtin<"_Vector<16, char>(_Vector<8, float>)">;
   def vcvtrops2hf8_256_mask : X86Builtin<"_Vector<16, char>(_Vector<8, float>, _Vector<16, char>, unsigned char)">;
 }
 
-let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<512>] in {
   def vcvtrops2hf8_512 : X86Builtin<"_Vector<16, char>(_Vector<16, float>)">;
 }
 
 // VCVTROPS2HF8S
-let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<128>] in {
   def vcvtrops2hf8s_128 : X86Builtin<"_Vector<16, char>(_Vector<4, float>)">;
   def vcvtrops2hf8s_128_mask : X86Builtin<"_Vector<16, char>(_Vector<4, float>, _Vector<16, char>, unsigned char)">;
 }
 
-let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<256>] in {
   def vcvtrops2hf8s_256 : X86Builtin<"_Vector<16, char>(_Vector<8, float>)">;
   def vcvtrops2hf8s_256_mask : X86Builtin<"_Vector<16, char>(_Vector<8, float>, _Vector<16, char>, unsigned char)">;
 }
 
-let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<512>] in {
   def vcvtrops2hf8s_512 : X86Builtin<"_Vector<16, char>(_Vector<16, float>)">;
 }
 
 // Convert from FP32 to FP8 with bias
 
 // VCVTBIASPS2BF8
-let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<128>] in {
   def vcvtbiasps2bf8_128 : X86Builtin<"_Vector<16, char>(_Vector<4, int>, _Vector<4, float>)">;
   def vcvtbiasps2bf8_128_mask : X86Builtin<"_Vector<16, char>(_Vector<4, int>, _Vector<4, float>, _Vector<16, char>, unsigned char)">;
 }
 
-let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<256>] in {
   def vcvtbiasps2bf8_256 : X86Builtin<"_Vector<16, char>(_Vector<8, int>, _Vector<8, float>)">;
   def vcvtbiasps2bf8_256_mask : X86Builtin<"_Vector<16, char>(_Vector<8, int>, _Vector<8, float>, _Vector<16, char>, unsigned char)">;
 }
 
-let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<512>] in {
   def vcvtbiasps2bf8_512 : X86Builtin<"_Vector<16, char>(_Vector<16, int>, _Vector<16, float>)">;
 }
 
 // VCVTBIASPS2BF8S
-let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<128>] in {
   def vcvtbiasps2bf8s_128 : X86Builtin<"_Vector<16, char>(_Vector<4, int>, _Vector<4, float>)">;
   def vcvtbiasps2bf8s_128_mask : X86Builtin<"_Vector<16, char>(_Vector<4, int>, _Vector<4, float>, _Vector<16, char>, unsigned char)">;
 }
 
-let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<256>] in {
   def vcvtbiasps2bf8s_256 : X86Builtin<"_Vector<16, char>(_Vector<8, int>, _Vector<8, float>)">;
   def vcvtbiasps2bf8s_256_mask : X86Builtin<"_Vector<16, char>(_Vector<8, int>, _Vector<8, float>, _Vector<16, char>, unsigned char)">;
 }
 
-let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<512>] in {
   def vcvtbiasps2bf8s_512 : X86Builtin<"_Vector<16, char>(_Vector<16, int>, _Vector<16, float>)">;
 }
 
 // VCVTBIASPS2HF8
-let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<128>] in {
   def vcvtbiasps2hf8_128 : X86Builtin<"_Vector<16, char>(_Vector<4, int>, _Vector<4, float>)">;
   def vcvtbiasps2hf8_128_mask : X86Builtin<"_Vector<16, char>(_Vector<4, int>, _Vector<4, float>, _Vector<16, char>, unsigned char)">;
 }
 
-let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<256>] in {
   def vcvtbiasps2hf8_256 : X86Builtin<"_Vector<16, char>(_Vector<8, int>, _Vector<8, float>)">;
   def vcvtbiasps2hf8_256_mask : X86Builtin<"_Vector<16, char>(_Vector<8, int>, _Vector<8, float>, _Vector<16, char>, unsigned char)">;
 }
 
-let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<512>] in {
   def vcvtbiasps2hf8_512 : X86Builtin<"_Vector<16, char>(_Vector<16, int>, _Vector<16, float>)">;
 }
 
 // VCVTBIASPS2HF8S
-let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<128>] in {
   def vcvtbiasps2hf8s_128 : X86Builtin<"_Vector<16, char>(_Vector<4, int>, _Vector<4, float>)">;
   def vcvtbiasps2hf8s_128_mask : X86Builtin<"_Vector<16, char>(_Vector<4, int>, _Vector<4, float>, _Vector<16, char>, unsigned char)">;
 }
 
-let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<256>] in {
   def vcvtbiasps2hf8s_256 : X86Builtin<"_Vector<16, char>(_Vector<8, int>, _Vector<8, float>)">;
   def vcvtbiasps2hf8s_256_mask : X86Builtin<"_Vector<16, char>(_Vector<8, int>, _Vector<8, float>, _Vector<16, char>, unsigned char)">;
 }
 
-let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<512>] in {
   def vcvtbiasps2hf8s_512 : X86Builtin<"_Vector<16, char>(_Vector<16, int>, _Vector<16, float>)">;
 }
 
 // Convert from FP8 to FP32
 
 // VCVTBF82PS
-let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<128>] in {
   def vcvtbf82ps_128 : X86Builtin<"_Vector<4, float>(_Vector<16, char>)">;
 }
 
-let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<256>] in {
   def vcvtbf82ps_256 : X86Builtin<"_Vector<8, float>(_Vector<16, char>)">;
 }
 
-let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<512>] in {
   def vcvtbf82ps_512 : X86Builtin<"_Vector<16, float>(_Vector<16, char>)">;
 }
 
 // VCVTHF82PS
-let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<128>] in {
   def vcvthf82ps_128 : X86Builtin<"_Vector<4, float>(_Vector<16, char>)">;
 }
 
-let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<256>] in {
   def vcvthf82ps_256 : X86Builtin<"_Vector<8, float>(_Vector<16, char>)">;
 }
 
-let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<512>] in {
   def vcvthf82ps_512 : X86Builtin<"_Vector<16, float>(_Vector<16, char>)">;
 }
 
 // Convert from FP8 to FP6
 
 // VCVTBF82BF6S
-let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<128>] in {
   def vcvtbf82bf6s_128 : X86Builtin<"_Vector<16, char>(_Vector<16, char>)">;
 }
 
-let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<256>] in {
   def vcvtbf82bf6s_256 : X86Builtin<"_Vector<32, char>(_Vector<32, char>)">;
 }
 
-let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<512>] in {
   def vcvtbf82bf6s_512 : X86Builtin<"_Vector<64, char>(_Vector<64, char>)">;
 }
 
 // VCVTHF82HF6S
-let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<128>] in {
   def vcvthf82hf6s_128 : X86Builtin<"_Vector<16, char>(_Vector<16, char>)">;
 }
 
-let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<256>] in {
   def vcvthf82hf6s_256 : X86Builtin<"_Vector<32, char>(_Vector<32, char>)">;
 }
 
-let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<512>] in {
   def vcvthf82hf6s_512 : X86Builtin<"_Vector<64, char>(_Vector<64, char>)">;
 }
 
 // Convert from FP4 to FP8 and from FP6 to FP8
 
 // VCVTBF42HF8
-let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<128>] in {
   def vcvtbf42hf8_128 : X86Builtin<"_Vector<16, char>(_Vector<16, char>)">;
 }
 
-let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<256>] in {
   def vcvtbf42hf8_256 : X86Builtin<"_Vector<32, char>(_Vector<16, char>)">;
 }
 
-let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<512>] in {
   def vcvtbf42hf8_512 : X86Builtin<"_Vector<64, char>(_Vector<32, char>)">;
 }
 
 // VCVTBF62HF8
-let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<128>] in {
   def vcvtbf62hf8_128 : X86Builtin<"_Vector<16, char>(_Vector<16, char>)">;
 }
 
-let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<256>] in {
   def vcvtbf62hf8_256 : X86Builtin<"_Vector<32, char>(_Vector<32, char>)">;
 }
 
-let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<512>] in {
   def vcvtbf62hf8_512 : X86Builtin<"_Vector<64, char>(_Vector<64, char>)">;
 }
 
 // VCVTHF62HF8
-let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<128>] in {
   def vcvthf62hf8_128 : X86Builtin<"_Vector<16, char>(_Vector<16, char>)">;
 }
 
-let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<256>] in {
   def vcvthf62hf8_256 : X86Builtin<"_Vector<32, char>(_Vector<32, char>)">;
 }
 
-let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<512>] in {
   def vcvthf62hf8_512 : X86Builtin<"_Vector<64, char>(_Vector<64, char>)">;
 }
 
 // Convert from FP8 to FP4
 
 // VCVTBF82BF4S
-let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<128>] in {
   def vcvtbf82bf4s_128 : X86Builtin<"_Vector<16, char>(_Vector<16, char>)">;
 }
 
-let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<256>] in {
   def vcvtbf82bf4s_256 : X86Builtin<"_Vector<16, char>(_Vector<32, char>)">;
 }
 
-let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<512>] in {
   def vcvtbf82bf4s_512 : X86Builtin<"_Vector<32, char>(_Vector<64, char>)">;
 }
 
 // VCVTHF82BF4S
-let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<128>] in {
   def vcvthf82bf4s_128 : X86Builtin<"_Vector<16, char>(_Vector<16, char>)">;
 }
 
-let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<256>] in {
   def vcvthf82bf4s_256 : X86Builtin<"_Vector<16, char>(_Vector<32, char>)">;
 }
 
-let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<512>] in {
   def vcvthf82bf4s_512 : X86Builtin<"_Vector<32, char>(_Vector<64, char>)">;
 }
 
 // Unpack to Byte
 
-let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<128>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<128>] in {
   def vunpackb128 : X86Builtin<"_Vector<16, char>(_Vector<16, char>, _Constant unsigned char)">;
 }
 
-let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<256>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<256>] in {
   def vunpackb256 : X86Builtin<"_Vector<32, char>(_Vector<32, char>, _Constant unsigned char)">;
 }
 
-let Features = "avx10v2aux", Attributes = [NoThrow, RequiredVectorWidth<512>] in {
+let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<512>] in {
   def vunpackb512 : X86Builtin<"_Vector<64, char>(_Vector<64, char>, _Constant unsigned char)">;
 }
 
diff --git a/clang/include/clang/Basic/DiagnosticSemaKinds.td b/clang/include/clang/Basic/DiagnosticSemaKinds.td
index 24ecc88d2fbc03..42372d2dcae99b 100644
--- a/clang/include/clang/Basic/DiagnosticSemaKinds.td
+++ b/clang/include/clang/Basic/DiagnosticSemaKinds.td
@@ -11598,6 +11598,8 @@ def err_ppc_invalid_arg_type : Error<
   "argument %0 must be of type %1">;
 def err_x86_builtin_invalid_rounding : Error<
   "invalid rounding argument">;
+def err_x86_builtin_reserved_vunpackb_imm : Error<
+  "argument value %0 is a reserved VUNPACKB immediate">;
 def err_x86_builtin_invalid_scale : Error<
   "scale argument must be 1, 2, 4, or 8">;
 def err_x86_builtin_tile_arg_duplicate : Error<
diff --git a/clang/include/clang/Sema/SemaX86.h b/clang/include/clang/Sema/SemaX86.h
index f382d3d03dfa15..09a7831032abd8 100644
--- a/clang/include/clang/Sema/SemaX86.h
+++ b/clang/include/clang/Sema/SemaX86.h
@@ -26,6 +26,7 @@ class SemaX86 : public SemaBase {
   SemaX86(Sema &S);
 
   bool CheckBuiltinRoundingOrSAE(unsigned BuiltinID, CallExpr *TheCall);
+  bool CheckBuiltinVUnpackBImm(unsigned BuiltinID, CallExpr *TheCall);
   bool CheckBuiltinGatherScatterScale(unsigned BuiltinID, CallExpr *TheCall);
   bool CheckBuiltinTileArguments(unsigned BuiltinID, CallExpr *TheCall);
   bool CheckBuiltinTileArgumentsRange(CallExpr *TheCall, ArrayRef<int> ArgNums);
diff --git a/clang/lib/Headers/avx10_2_512v2auxintrin.h b/clang/lib/Headers/avx10_2_512v2auxintrin.h
index d83be98bffe82f..6e448fbb4a9155 100644
--- a/clang/lib/Headers/avx10_2_512v2auxintrin.h
+++ b/clang/lib/Headers/avx10_2_512v2auxintrin.h
@@ -1016,7 +1016,15 @@ _mm512_maskz_cvthf6_hf8(__mmask64 __U, __m512i __A) {
 /// \param A
 ///    A 512-bit vector of [64 x i8].
 /// \param imm
-///    An immediate value specifying the unpack operation.
+///    An 8-bit immediate selecting the packed element size, start offset, and
+///    optional sign-extend for \c VUNPACKB. Compose it with
+///    \c _MM_UNPACKB_SIZE, \c _MM_UNPACKB_START, and optionally
+///    \c _MM_UNPACKB_SEXT. Omitting \c _MM_UNPACKB_SEXT zero-extends unpacked
+///    elements to 8 bits. \n
+///    Example:
+///    <c>_MM_UNPACKB_SIZE(4) | _MM_UNPACKB_START(1) | _MM_UNPACKB_SEXT</c>
+///
+/// \see { _MM_UNPACKB_SIZE _MM_UNPACKB_START _MM_UNPACKB_SEXT }
 /// \returns
 ///    A 512-bit vector of [64 x i8] containing the unpacked values.
 #define _mm512_unpack_epi8(A, imm)                                             \
@@ -1036,7 +1044,15 @@ _mm512_maskz_cvthf6_hf8(__mmask64 __U, __m512i __A) {
 /// \param A
 ///    A 512-bit vector of [64 x i8].
 /// \param imm
-///    An immediate value specifying the unpack operation.
+///    An 8-bit immediate selecting the packed element size, start offset, and
+///    optional sign-extend for \c VUNPACKB. Compose it with
+///    \c _MM_UNPACKB_SIZE, \c _MM_UNPACKB_START, and optionally
+///    \c _MM_UNPACKB_SEXT. Omitting \c _MM_UNPACKB_SEXT zero-extends unpacked
+///    elements to 8 bits. \n
+///    Example:
+///    <c>_MM_UNPACKB_SIZE(4) | _MM_UNPACKB_START(1) | _MM_UNPACKB_SEXT</c>
+///
+/// \see { _MM_UNPACKB_SIZE _MM_UNPACKB_START _MM_UNPACKB_SEXT }
 /// \returns
 ///    A 512-bit vector of [64 x i8] containing the unpacked values.
 #define _mm512_mask_unpack_epi8(W, U, A, imm)                                  \
@@ -1056,7 +1072,15 @@ _mm512_maskz_cvthf6_hf8(__mmask64 __U, __m512i __A) {
 /// \param A
 ///    A 512-bit vector of [64 x i8].
 /// \param imm
-///    An immediate value specifying the unpack operation.
+///    An 8-bit immediate selecting the packed element size, start offset, and
+///    optional sign-extend for \c VUNPACKB. Compose it with
+///    \c _MM_UNPACKB_SIZE, \c _MM_UNPACKB_START, and optionally
+///    \c _MM_UNPACKB_SEXT. Omitting \c _MM_UNPACKB_SEXT zero-extends unpacked
+///    elements to 8 bits. \n
+///    Example:
+///    <c>_MM_UNPACKB_SIZE(4) | _MM_UNPACKB_START(1) | _MM_UNPACKB_SEXT</c>
+///
+/// \see { _MM_UNPACKB_SIZE _MM_UNPACKB_START _MM_UNPACKB_SEXT }
 /// \returns
 ///    A 512-bit vector of [64 x i8] containing the unpacked values.
 #define _mm512_maskz_unpack_epi8(U, A, imm)                                    \
diff --git a/clang/lib/Headers/avx10_2_v2auxintrin.h b/clang/lib/Headers/avx10_2_v2auxintrin.h
index fe9fe8068e94d4..dc14f2ba0dd420 100644
--- a/clang/lib/Headers/avx10_2_v2auxintrin.h
+++ b/clang/lib/Headers/avx10_2_v2auxintrin.h
@@ -83,8 +83,8 @@ static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_mask_cvtps_bf8(__m128i __W,
 ///    values, or zero where the mask bit is clear; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_maskz_cvtps_bf8(__mmask8 __U, __m128 __A) {
-  return (__m128i)__builtin_ia32_vcvtps2bf8_128_mask(
-      (__v4sf)__A, (__v16qi)_mm_setzero_si128(), (__mmask8)__U);
+  return (__m128i)__builtin_ia32_selectb_128(
+      (__mmask8)__U, (__v16qi)_mm_cvtps_bf8(__A), (__v16qi)_mm_setzero_si128());
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __A
@@ -145,8 +145,9 @@ _mm256_mask_cvtps_bf8(__m128i __W, __mmask8 __U, __m256 __A) {
 ///    values, or zero where the mask bit is clear; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_maskz_cvtps_bf8(__mmask8 __U, __m256 __A) {
-  return (__m128i)__builtin_ia32_vcvtps2bf8_256_mask(
-      (__v8sf)__A, (__v16qi)_mm_setzero_si128(), (__mmask8)__U);
+  return (__m128i)__builtin_ia32_selectb_128((__mmask8)__U,
+                                             (__v16qi)_mm256_cvtps_bf8(__A),
+                                             (__v16qi)_mm_setzero_si128());
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __A
@@ -206,8 +207,9 @@ _mm_mask_cvts_ps_bf8(__m128i __W, __mmask8 __U, __m128 __A) {
 ///    values, or zero where the mask bit is clear; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_maskz_cvts_ps_bf8(__mmask8 __U, __m128 __A) {
-  return (__m128i)__builtin_ia32_vcvtps2bf8s_128_mask(
-      (__v4sf)__A, (__v16qi)_mm_setzero_si128(), (__mmask8)__U);
+  return (__m128i)__builtin_ia32_selectb_128((__mmask8)__U,
+                                             (__v16qi)_mm_cvts_ps_bf8(__A),
+                                             (__v16qi)_mm_setzero_si128());
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __A
@@ -267,8 +269,9 @@ _mm256_mask_cvts_ps_bf8(__m128i __W, __mmask8 __U, __m256 __A) {
 ///    values, or zero where the mask bit is clear; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_maskz_cvts_ps_bf8(__mmask8 __U, __m256 __A) {
-  return (__m128i)__builtin_ia32_vcvtps2bf8s_256_mask(
-      (__v8sf)__A, (__v16qi)_mm_setzero_si128(), (__mmask8)__U);
+  return (__m128i)__builtin_ia32_selectb_128((__mmask8)__U,
+                                             (__v16qi)_mm256_cvts_ps_bf8(__A),
+                                             (__v16qi)_mm_setzero_si128());
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __A
@@ -329,8 +332,8 @@ static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_mask_cvtps_hf8(__m128i __W,
 ///    values, or zero where the mask bit is clear; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_maskz_cvtps_hf8(__mmask8 __U, __m128 __A) {
-  return (__m128i)__builtin_ia32_vcvtps2hf8_128_mask(
-      (__v4sf)__A, (__v16qi)_mm_setzero_si128(), (__mmask8)__U);
+  return (__m128i)__builtin_ia32_selectb_128(
+      (__mmask8)__U, (__v16qi)_mm_cvtps_hf8(__A), (__v16qi)_mm_setzero_si128());
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __A
@@ -390,8 +393,9 @@ _mm256_mask_cvtps_hf8(__m128i __W, __mmask8 __U, __m256 __A) {
 ///    values, or zero where the mask bit is clear; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_maskz_cvtps_hf8(__mmask8 __U, __m256 __A) {
-  return (__m128i)__builtin_ia32_vcvtps2hf8_256_mask(
-      (__v8sf)__A, (__v16qi)_mm_setzero_si128(), (__mmask8)__U);
+  return (__m128i)__builtin_ia32_selectb_128((__mmask8)__U,
+                                             (__v16qi)_mm256_cvtps_hf8(__A),
+                                             (__v16qi)_mm_setzero_si128());
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __A
@@ -451,8 +455,9 @@ _mm_mask_cvts_ps_hf8(__m128i __W, __mmask8 __U, __m128 __A) {
 ///    values, or zero where the mask bit is clear; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_maskz_cvts_ps_hf8(__mmask8 __U, __m128 __A) {
-  return (__m128i)__builtin_ia32_vcvtps2hf8s_128_mask(
-      (__v4sf)__A, (__v16qi)_mm_setzero_si128(), (__mmask8)__U);
+  return (__m128i)__builtin_ia32_selectb_128((__mmask8)__U,
+                                             (__v16qi)_mm_cvts_ps_hf8(__A),
+                                             (__v16qi)_mm_setzero_si128());
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __A
@@ -512,8 +517,9 @@ _mm256_mask_cvts_ps_hf8(__m128i __W, __mmask8 __U, __m256 __A) {
 ///    values, or zero where the mask bit is clear; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_maskz_cvts_ps_hf8(__mmask8 __U, __m256 __A) {
-  return (__m128i)__builtin_ia32_vcvtps2hf8s_256_mask(
-      (__v8sf)__A, (__v16qi)_mm_setzero_si128(), (__mmask8)__U);
+  return (__m128i)__builtin_ia32_selectb_128((__mmask8)__U,
+                                             (__v16qi)_mm256_cvts_ps_hf8(__A),
+                                             (__v16qi)_mm_setzero_si128());
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __A
@@ -573,8 +579,9 @@ _mm_mask_cvtrops_hf8(__m128i __W, __mmask8 __U, __m128 __A) {
 ///    values, or zero where the mask bit is clear; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_maskz_cvtrops_hf8(__mmask8 __U, __m128 __A) {
-  return (__m128i)__builtin_ia32_vcvtrops2hf8_128_mask(
-      (__v4sf)__A, (__v16qi)_mm_setzero_si128(), (__mmask8)__U);
+  return (__m128i)__builtin_ia32_selectb_128((__mmask8)__U,
+                                             (__v16qi)_mm_cvtrops_hf8(__A),
+                                             (__v16qi)_mm_setzero_si128());
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __A
@@ -634,8 +641,9 @@ _mm256_mask_cvtrops_hf8(__m128i __W, __mmask8 __U, __m256 __A) {
 ///    values, or zero where the mask bit is clear; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_maskz_cvtrops_hf8(__mmask8 __U, __m256 __A) {
-  return (__m128i)__builtin_ia32_vcvtrops2hf8_256_mask(
-      (__v8sf)__A, (__v16qi)_mm_setzero_si128(), (__mmask8)__U);
+  return (__m128i)__builtin_ia32_selectb_128((__mmask8)__U,
+                                             (__v16qi)_mm256_cvtrops_hf8(__A),
+                                             (__v16qi)_mm_setzero_si128());
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __A
@@ -695,8 +703,9 @@ _mm_mask_cvts_rops_hf8(__m128i __W, __mmask8 __U, __m128 __A) {
 ///    values, or zero where the mask bit is clear; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_maskz_cvts_rops_hf8(__mmask8 __U, __m128 __A) {
-  return (__m128i)__builtin_ia32_vcvtrops2hf8s_128_mask(
-      (__v4sf)__A, (__v16qi)_mm_setzero_si128(), (__mmask8)__U);
+  return (__m128i)__builtin_ia32_selectb_128((__mmask8)__U,
+                                             (__v16qi)_mm_cvts_rops_hf8(__A),
+                                             (__v16qi)_mm_setzero_si128());
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __A
@@ -757,8 +766,9 @@ _mm256_mask_cvts_rops_hf8(__m128i __W, __mmask8 __U, __m256 __A) {
 ///    values, or zero where the mask bit is clear; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_maskz_cvts_rops_hf8(__mmask8 __U, __m256 __A) {
-  return (__m128i)__builtin_ia32_vcvtrops2hf8s_256_mask(
-      (__v8sf)__A, (__v16qi)_mm_setzero_si128(), (__mmask8)__U);
+  return (__m128i)__builtin_ia32_selectb_128((__mmask8)__U,
+                                             (__v16qi)_mm256_cvts_rops_hf8(__A),
+                                             (__v16qi)_mm_setzero_si128());
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __B
@@ -825,8 +835,9 @@ _mm_mask_cvtbiasps_bf8(__m128i __W, __mmask8 __U, __m128i __A, __m128 __B) {
 ///    values, or zero where the mask bit is clear; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_maskz_cvtbiasps_bf8(__mmask8 __U, __m128i __A, __m128 __B) {
-  return (__m128i)__builtin_ia32_vcvtbiasps2bf8_128_mask(
-      (__v4si)__A, (__v4sf)__B, (__v16qi)_mm_setzero_si128(), (__mmask8)__U);
+  return (__m128i)__builtin_ia32_selectb_128(
+      (__mmask8)__U, (__v16qi)_mm_cvtbiasps_bf8(__A, __B),
+      (__v16qi)_mm_setzero_si128());
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __B
@@ -893,8 +904,9 @@ _mm256_mask_cvtbiasps_bf8(__m128i __W, __mmask8 __U, __m256i __A, __m256 __B) {
 ///    values, or zero where the mask bit is clear; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_maskz_cvtbiasps_bf8(__mmask8 __U, __m256i __A, __m256 __B) {
-  return (__m128i)__builtin_ia32_vcvtbiasps2bf8_256_mask(
-      (__v8si)__A, (__v8sf)__B, (__v16qi)_mm_setzero_si128(), (__mmask8)__U);
+  return (__m128i)__builtin_ia32_selectb_128(
+      (__mmask8)__U, (__v16qi)_mm256_cvtbiasps_bf8(__A, __B),
+      (__v16qi)_mm_setzero_si128());
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __B
@@ -961,8 +973,9 @@ _mm_mask_cvts_biasps_bf8(__m128i __W, __mmask8 __U, __m128i __A, __m128 __B) {
 ///    values, or zero where the mask bit is clear; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_maskz_cvts_biasps_bf8(__mmask8 __U, __m128i __A, __m128 __B) {
-  return (__m128i)__builtin_ia32_vcvtbiasps2bf8s_128_mask(
-      (__v4si)__A, (__v4sf)__B, (__v16qi)_mm_setzero_si128(), (__mmask8)__U);
+  return (__m128i)__builtin_ia32_selectb_128(
+      (__mmask8)__U, (__v16qi)_mm_cvts_biasps_bf8(__A, __B),
+      (__v16qi)_mm_setzero_si128());
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __B
@@ -1029,8 +1042,9 @@ static __inline__ __m128i __DEFAULT_FN_ATTRS256 _mm256_mask_cvts_biasps_bf8(
 ///    values, or zero where the mask bit is clear; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_maskz_cvts_biasps_bf8(__mmask8 __U, __m256i __A, __m256 __B) {
-  return (__m128i)__builtin_ia32_vcvtbiasps2bf8s_256_mask(
-      (__v8si)__A, (__v8sf)__B, (__v16qi)_mm_setzero_si128(), (__mmask8)__U);
+  return (__m128i)__builtin_ia32_selectb_128(
+      (__mmask8)__U, (__v16qi)_mm256_cvts_biasps_bf8(__A, __B),
+      (__v16qi)_mm_setzero_si128());
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __B
@@ -1097,8 +1111,9 @@ _mm_mask_cvtbiasps_hf8(__m128i __W, __mmask8 __U, __m128i __A, __m128 __B) {
 ///    values, or zero where the mask bit is clear; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_maskz_cvtbiasps_hf8(__mmask8 __U, __m128i __A, __m128 __B) {
-  return (__m128i)__builtin_ia32_vcvtbiasps2hf8_128_mask(
-      (__v4si)__A, (__v4sf)__B, (__v16qi)_mm_setzero_si128(), (__mmask8)__U);
+  return (__m128i)__builtin_ia32_selectb_128(
+      (__mmask8)__U, (__v16qi)_mm_cvtbiasps_hf8(__A, __B),
+      (__v16qi)_mm_setzero_si128());
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __B
@@ -1165,8 +1180,9 @@ _mm256_mask_cvtbiasps_hf8(__m128i __W, __mmask8 __U, __m256i __A, __m256 __B) {
 ///    values, or zero where the mask bit is clear; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_maskz_cvtbiasps_hf8(__mmask8 __U, __m256i __A, __m256 __B) {
-  return (__m128i)__builtin_ia32_vcvtbiasps2hf8_256_mask(
-      (__v8si)__A, (__v8sf)__B, (__v16qi)_mm_setzero_si128(), (__mmask8)__U);
+  return (__m128i)__builtin_ia32_selectb_128(
+      (__mmask8)__U, (__v16qi)_mm256_cvtbiasps_hf8(__A, __B),
+      (__v16qi)_mm_setzero_si128());
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __B
@@ -1233,8 +1249,9 @@ _mm_mask_cvts_biasps_hf8(__m128i __W, __mmask8 __U, __m128i __A, __m128 __B) {
 ///    values, or zero where the mask bit is clear; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_maskz_cvts_biasps_hf8(__mmask8 __U, __m128i __A, __m128 __B) {
-  return (__m128i)__builtin_ia32_vcvtbiasps2hf8s_128_mask(
-      (__v4si)__A, (__v4sf)__B, (__v16qi)_mm_setzero_si128(), (__mmask8)__U);
+  return (__m128i)__builtin_ia32_selectb_128(
+      (__mmask8)__U, (__v16qi)_mm_cvts_biasps_hf8(__A, __B),
+      (__v16qi)_mm_setzero_si128());
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __B
@@ -1301,8 +1318,9 @@ static __inline__ __m128i __DEFAULT_FN_ATTRS256 _mm256_mask_cvts_biasps_hf8(
 ///    values, or zero where the mask bit is clear; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_maskz_cvts_biasps_hf8(__mmask8 __U, __m256i __A, __m256 __B) {
-  return (__m128i)__builtin_ia32_vcvtbiasps2hf8s_256_mask(
-      (__v8si)__A, (__v8sf)__B, (__v16qi)_mm_setzero_si128(), (__mmask8)__U);
+  return (__m128i)__builtin_ia32_selectb_128(
+      (__mmask8)__U, (__v16qi)_mm256_cvts_biasps_hf8(__A, __B),
+      (__v16qi)_mm_setzero_si128());
 }
 
 /// Convert packed BF8 (8-bit) floating-point elements in \a __A to packed
@@ -2050,7 +2068,7 @@ _mm256_maskz_cvthf6_hf8(__mmask32 __U, __m256i __A) {
 ///    Bits [1:0] of the immediate operand.
 #define _MM_UNPACKB_START(s) (((s) & 0x3) << 0)
 
-/// The \c sign \c ext field of the immediate operand of \c VUNPACKB,
+/// The \c sign-extend field of the immediate operand of \c VUNPACKB,
 ///    requesting that unpacked elements be sign-extended to 8 bits instead of
 ///    zero-extended.
 ///
@@ -2067,7 +2085,15 @@ _mm256_maskz_cvthf6_hf8(__mmask32 __U, __m256i __A) {
 /// \param A
 ///    A 128-bit vector of [16 x i8].
 /// \param imm
-///    An immediate value specifying the unpack operation.
+///    An 8-bit immediate selecting the packed element size, start offset, and
+///    optional sign-extend for \c VUNPACKB. Compose it with
+///    \c _MM_UNPACKB_SIZE, \c _MM_UNPACKB_START, and optionally
+///    \c _MM_UNPACKB_SEXT. Omitting \c _MM_UNPACKB_SEXT zero-extends unpacked
+///    elements to 8 bits. \n
+///    Example:
+///    <c>_MM_UNPACKB_SIZE(4) | _MM_UNPACKB_START(1) | _MM_UNPACKB_SEXT</c>
+///
+/// \see { _MM_UNPACKB_SIZE _MM_UNPACKB_START _MM_UNPACKB_SEXT }
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the unpacked values.
 #define _mm_unpack_epi8(A, imm)                                                \
@@ -2087,7 +2113,15 @@ _mm256_maskz_cvthf6_hf8(__mmask32 __U, __m256i __A) {
 /// \param A
 ///    A 128-bit vector of [16 x i8].
 /// \param imm
-///    An immediate value specifying the unpack operation.
+///    An 8-bit immediate selecting the packed element size, start offset, and
+///    optional sign-extend for \c VUNPACKB. Compose it with
+///    \c _MM_UNPACKB_SIZE, \c _MM_UNPACKB_START, and optionally
+///    \c _MM_UNPACKB_SEXT. Omitting \c _MM_UNPACKB_SEXT zero-extends unpacked
+///    elements to 8 bits. \n
+///    Example:
+///    <c>_MM_UNPACKB_SIZE(4) | _MM_UNPACKB_START(1) | _MM_UNPACKB_SEXT</c>
+///
+/// \see { _MM_UNPACKB_SIZE _MM_UNPACKB_START _MM_UNPACKB_SEXT }
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the unpacked values.
 #define _mm_mask_unpack_epi8(W, U, A, imm)                                     \
@@ -2107,7 +2141,15 @@ _mm256_maskz_cvthf6_hf8(__mmask32 __U, __m256i __A) {
 /// \param A
 ///    A 128-bit vector of [16 x i8].
 /// \param imm
-///    An immediate value specifying the unpack operation.
+///    An 8-bit immediate selecting the packed element size, start offset, and
+///    optional sign-extend for \c VUNPACKB. Compose it with
+///    \c _MM_UNPACKB_SIZE, \c _MM_UNPACKB_START, and optionally
+///    \c _MM_UNPACKB_SEXT. Omitting \c _MM_UNPACKB_SEXT zero-extends unpacked
+///    elements to 8 bits. \n
+///    Example:
+///    <c>_MM_UNPACKB_SIZE(4) | _MM_UNPACKB_START(1) | _MM_UNPACKB_SEXT</c>
+///
+/// \see { _MM_UNPACKB_SIZE _MM_UNPACKB_START _MM_UNPACKB_SEXT }
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the unpacked values.
 #define _mm_maskz_unpack_epi8(U, A, imm)                                       \
@@ -2125,7 +2167,15 @@ _mm256_maskz_cvthf6_hf8(__mmask32 __U, __m256i __A) {
 /// \param A
 ///    A 256-bit vector of [32 x i8].
 /// \param imm
-///    An immediate value specifying the unpack operation.
+///    An 8-bit immediate selecting the packed element size, start offset, and
+///    optional sign-extend for \c VUNPACKB. Compose it with
+///    \c _MM_UNPACKB_SIZE, \c _MM_UNPACKB_START, and optionally
+///    \c _MM_UNPACKB_SEXT. Omitting \c _MM_UNPACKB_SEXT zero-extends unpacked
+///    elements to 8 bits. \n
+///    Example:
+///    <c>_MM_UNPACKB_SIZE(4) | _MM_UNPACKB_START(1) | _MM_UNPACKB_SEXT</c>
+///
+/// \see { _MM_UNPACKB_SIZE _MM_UNPACKB_START _MM_UNPACKB_SEXT }
 /// \returns
 ///    A 256-bit vector of [32 x i8] containing the unpacked values.
 #define _mm256_unpack_epi8(A, imm)                                             \
@@ -2145,7 +2195,15 @@ _mm256_maskz_cvthf6_hf8(__mmask32 __U, __m256i __A) {
 /// \param A
 ///    A 256-bit vector of [32 x i8].
 /// \param imm
-///    An immediate value specifying the unpack operation.
+///    An 8-bit immediate selecting the packed element size, start offset, and
+///    optional sign-extend for \c VUNPACKB. Compose it with
+///    \c _MM_UNPACKB_SIZE, \c _MM_UNPACKB_START, and optionally
+///    \c _MM_UNPACKB_SEXT. Omitting \c _MM_UNPACKB_SEXT zero-extends unpacked
+///    elements to 8 bits. \n
+///    Example:
+///    <c>_MM_UNPACKB_SIZE(4) | _MM_UNPACKB_START(1) | _MM_UNPACKB_SEXT</c>
+///
+/// \see { _MM_UNPACKB_SIZE _MM_UNPACKB_START _MM_UNPACKB_SEXT }
 /// \returns
 ///    A 256-bit vector of [32 x i8] containing the unpacked values.
 #define _mm256_mask_unpack_epi8(W, U, A, imm)                                  \
@@ -2165,7 +2223,15 @@ _mm256_maskz_cvthf6_hf8(__mmask32 __U, __m256i __A) {
 /// \param A
 ///    A 256-bit vector of [32 x i8].
 /// \param imm
-///    An immediate value specifying the unpack operation.
+///    An 8-bit immediate selecting the packed element size, start offset, and
+///    optional sign-extend for \c VUNPACKB. Compose it with
+///    \c _MM_UNPACKB_SIZE, \c _MM_UNPACKB_START, and optionally
+///    \c _MM_UNPACKB_SEXT. Omitting \c _MM_UNPACKB_SEXT zero-extends unpacked
+///    elements to 8 bits. \n
+///    Example:
+///    <c>_MM_UNPACKB_SIZE(4) | _MM_UNPACKB_START(1) | _MM_UNPACKB_SEXT</c>
+///
+/// \see { _MM_UNPACKB_SIZE _MM_UNPACKB_START _MM_UNPACKB_SEXT }
 /// \returns
 ///    A 256-bit vector of [32 x i8] containing the unpacked values.
 #define _mm256_maskz_unpack_epi8(U, A, imm)                                    \
diff --git a/clang/lib/Sema/SemaX86.cpp b/clang/lib/Sema/SemaX86.cpp
index 3b114f7c34a24a..24a6a2acc64f47 100644
--- a/clang/lib/Sema/SemaX86.cpp
+++ b/clang/lib/Sema/SemaX86.cpp
@@ -343,6 +343,48 @@ bool SemaX86::CheckBuiltinRoundingOrSAE(unsigned BuiltinID, CallExpr *TheCall) {
          << Arg->getSourceRange();
 }
 
+// Check if the VUNPACKB immediate encoding is legal.
+bool SemaX86::CheckBuiltinVUnpackBImm(unsigned BuiltinID, CallExpr *TheCall) {
+  unsigned ArgNum = 0;
+  switch (BuiltinID) {
+  default:
+    return false;
+  case X86::BI__builtin_ia32_vunpackb128:
+  case X86::BI__builtin_ia32_vunpackb256:
+  case X86::BI__builtin_ia32_vunpackb512:
+    ArgNum = 1;
+    break;
+  }
+
+  llvm::APSInt Result;
+
+  // We can't check the value of a dependent argument.
+  Expr *Arg = TheCall->getArg(ArgNum);
+  if (Arg->isTypeDependent() || Arg->isValueDependent())
+    return false;
+
+  // Check constant-ness first.
+  if (SemaRef.BuiltinConstantArg(TheCall, ArgNum, Result))
+    return true;
+
+  uint64_t Imm = Result.getZExtValue();
+  // Skip reserved-encoding checks when the range check already diagnosed
+  // the immediate.
+  if (Imm > 63)
+    return false;
+
+  // Make sure size (imm[4:2]) and start (imm[1:0]) form a defined pairing.
+  unsigned Size = (Imm >> 2) & 0x7;
+  unsigned Start = Imm & 0x3;
+  if (Size == 2 || ((Size == 3 || Size == 4) && Start <= 1) ||
+      (Size >= 5 && Start == 0))
+    return false;
+
+  return Diag(TheCall->getBeginLoc(),
+              diag::err_x86_builtin_reserved_vunpackb_imm)
+         << toString(Result, 10) << Arg->getSourceRange();
+}
+
 // Check if the gather/scatter scale is legal.
 bool SemaX86::CheckBuiltinGatherScatterScale(unsigned BuiltinID,
                                              CallExpr *TheCall) {
@@ -963,8 +1005,12 @@ bool SemaX86::CheckBuiltinFunctionCall(const TargetInfo &TI, unsigned BuiltinID,
   // template-generated or macro-generated dead code to potentially have out-of-
   // range values. These need to code generate, but don't need to necessarily
   // make any sense. We use a warning that defaults to an error.
-  return SemaRef.BuiltinConstantArgRange(TheCall, i, l, u,
-                                         /*RangeIsError*/ false);
+  if (SemaRef.BuiltinConstantArgRange(TheCall, i, l, u,
+                                      /*RangeIsError*/ false))
+    return true;
+
+  // If the intrinsic has a VUNPACKB immediate, make sure the encoding is valid.
+  return CheckBuiltinVUnpackBImm(BuiltinID, TheCall);
 }
 
 void SemaX86::handleAnyInterruptAttr(Decl *D, const ParsedAttr &AL) {
diff --git a/clang/test/CodeGen/X86/avx10_2_v2aux-builtins-errors.c b/clang/test/CodeGen/X86/avx10_2_v2aux-builtins-errors.c
index ab9e227f07a80c..4e7e055bbc7936 100644
--- a/clang/test/CodeGen/X86/avx10_2_v2aux-builtins-errors.c
+++ b/clang/test/CodeGen/X86/avx10_2_v2aux-builtins-errors.c
@@ -37,3 +37,31 @@ __m512i test_mm512_mask_unpack_epi8(__m512i __W, __mmask64 __U, __m512i __A) {
 __m512i test_mm512_maskz_unpack_epi8(__mmask64 __U, __m512i __A) {
   return _mm512_maskz_unpack_epi8(__U, __A, 64); // expected-error {{argument value 64 is outside the valid range [0, 63]}}
 }
+
+__m128i test_mm_unpack_epi8_reserved_size0(__m128i __A) {
+  return _mm_unpack_epi8(__A, 1); // expected-error {{argument value 1 is a reserved VUNPACKB immediate}}
+}
+
+__m256i test_mm256_unpack_epi8_reserved_size0(__m256i __A) {
+  return _mm256_unpack_epi8(__A, 2); // expected-error {{argument value 2 is a reserved VUNPACKB immediate}}
+}
+
+__m512i test_mm512_unpack_epi8_reserved_size0(__m512i __A) {
+  return _mm512_unpack_epi8(__A, 3); // expected-error {{argument value 3 is a reserved VUNPACKB immediate}}
+}
+
+__m128i test_mm_unpack_epi8_reserved_size1(__m128i __A) {
+  return _mm_unpack_epi8(__A, _MM_UNPACKB_SIZE(1)); // expected-error {{argument value 4 is a reserved VUNPACKB immediate}}
+}
+
+__m128i test_mm_unpack_epi8_reserved_start(__m128i __A) {
+  return _mm_unpack_epi8(__A, _MM_UNPACKB_SIZE(4) | _MM_UNPACKB_START(3)); // expected-error {{argument value 19 is a reserved VUNPACKB immediate}}
+}
+
+__m128i test_mm_unpack_epi8_reserved_size5_start(__m128i __A) {
+  return _mm_unpack_epi8(__A, _MM_UNPACKB_SIZE(5) | _MM_UNPACKB_START(1)); // expected-error {{argument value 21 is a reserved VUNPACKB immediate}}
+}
+
+__m128i test_mm_mask_unpack_epi8_reserved(__m128i __W, __mmask16 __U, __m128i __A) {
+  return _mm_mask_unpack_epi8(__W, __U, __A, 0); // expected-error {{argument value 0 is a reserved VUNPACKB immediate}}
+}
diff --git a/clang/test/CodeGen/X86/avx10_2_v2aux-builtins.c b/clang/test/CodeGen/X86/avx10_2_v2aux-builtins.c
index c19892ad2131fa..d31715c114883d 100644
--- a/clang/test/CodeGen/X86/avx10_2_v2aux-builtins.c
+++ b/clang/test/CodeGen/X86/avx10_2_v2aux-builtins.c
@@ -19,8 +19,9 @@ __m128i test_mm_mask_cvtps_bf8(__m128i __W, __mmask8 __U, __m128 __A) {
 
 __m128i test_mm_maskz_cvtps_bf8(__mmask8 __U, __m128 __A) {
   // CHECK-LABEL: @test_mm_maskz_cvtps_bf8(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8128(<4 x float> %{{.*}})
   // CHECK: zeroinitializer
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8128(<4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm_maskz_cvtps_bf8(__U, __A);
 }
 
@@ -38,8 +39,9 @@ __m128i test_mm256_mask_cvtps_bf8(__m128i __W, __mmask8 __U, __m256 __A) {
 
 __m128i test_mm256_maskz_cvtps_bf8(__mmask8 __U, __m256 __A) {
   // CHECK-LABEL: @test_mm256_maskz_cvtps_bf8(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8256(<8 x float> %{{.*}})
   // CHECK: zeroinitializer
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8256(<8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm256_maskz_cvtps_bf8(__U, __A);
 }
 
@@ -78,8 +80,9 @@ __m128i test_mm_mask_cvts_ps_bf8(__m128i __W, __mmask8 __U, __m128 __A) {
 
 __m128i test_mm_maskz_cvts_ps_bf8(__mmask8 __U, __m128 __A) {
   // CHECK-LABEL: @test_mm_maskz_cvts_ps_bf8(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8s128(<4 x float> %{{.*}})
   // CHECK: zeroinitializer
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s128(<4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm_maskz_cvts_ps_bf8(__U, __A);
 }
 
@@ -97,8 +100,9 @@ __m128i test_mm256_mask_cvts_ps_bf8(__m128i __W, __mmask8 __U, __m256 __A) {
 
 __m128i test_mm256_maskz_cvts_ps_bf8(__mmask8 __U, __m256 __A) {
   // CHECK-LABEL: @test_mm256_maskz_cvts_ps_bf8(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8s256(<8 x float> %{{.*}})
   // CHECK: zeroinitializer
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s256(<8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm256_maskz_cvts_ps_bf8(__U, __A);
 }
 
@@ -137,8 +141,9 @@ __m128i test_mm_mask_cvtps_hf8(__m128i __W, __mmask8 __U, __m128 __A) {
 
 __m128i test_mm_maskz_cvtps_hf8(__mmask8 __U, __m128 __A) {
   // CHECK-LABEL: @test_mm_maskz_cvtps_hf8(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8128(<4 x float> %{{.*}})
   // CHECK: zeroinitializer
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8128(<4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm_maskz_cvtps_hf8(__U, __A);
 }
 
@@ -156,8 +161,9 @@ __m128i test_mm256_mask_cvtps_hf8(__m128i __W, __mmask8 __U, __m256 __A) {
 
 __m128i test_mm256_maskz_cvtps_hf8(__mmask8 __U, __m256 __A) {
   // CHECK-LABEL: @test_mm256_maskz_cvtps_hf8(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8256(<8 x float> %{{.*}})
   // CHECK: zeroinitializer
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8256(<8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm256_maskz_cvtps_hf8(__U, __A);
 }
 
@@ -196,8 +202,9 @@ __m128i test_mm_mask_cvts_ps_hf8(__m128i __W, __mmask8 __U, __m128 __A) {
 
 __m128i test_mm_maskz_cvts_ps_hf8(__mmask8 __U, __m128 __A) {
   // CHECK-LABEL: @test_mm_maskz_cvts_ps_hf8(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8s128(<4 x float> %{{.*}})
   // CHECK: zeroinitializer
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s128(<4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm_maskz_cvts_ps_hf8(__U, __A);
 }
 
@@ -215,8 +222,9 @@ __m128i test_mm256_mask_cvts_ps_hf8(__m128i __W, __mmask8 __U, __m256 __A) {
 
 __m128i test_mm256_maskz_cvts_ps_hf8(__mmask8 __U, __m256 __A) {
   // CHECK-LABEL: @test_mm256_maskz_cvts_ps_hf8(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8s256(<8 x float> %{{.*}})
   // CHECK: zeroinitializer
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s256(<8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm256_maskz_cvts_ps_hf8(__U, __A);
 }
 
@@ -255,8 +263,9 @@ __m128i test_mm_mask_cvtrops_hf8(__m128i __W, __mmask8 __U, __m128 __A) {
 
 __m128i test_mm_maskz_cvtrops_hf8(__mmask8 __U, __m128 __A) {
   // CHECK-LABEL: @test_mm_maskz_cvtrops_hf8(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8128(<4 x float> %{{.*}})
   // CHECK: zeroinitializer
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8128(<4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm_maskz_cvtrops_hf8(__U, __A);
 }
 
@@ -274,8 +283,9 @@ __m128i test_mm256_mask_cvtrops_hf8(__m128i __W, __mmask8 __U, __m256 __A) {
 
 __m128i test_mm256_maskz_cvtrops_hf8(__mmask8 __U, __m256 __A) {
   // CHECK-LABEL: @test_mm256_maskz_cvtrops_hf8(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8256(<8 x float> %{{.*}})
   // CHECK: zeroinitializer
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8256(<8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm256_maskz_cvtrops_hf8(__U, __A);
 }
 
@@ -314,8 +324,9 @@ __m128i test_mm_mask_cvts_rops_hf8(__m128i __W, __mmask8 __U, __m128 __A) {
 
 __m128i test_mm_maskz_cvts_rops_hf8(__mmask8 __U, __m128 __A) {
   // CHECK-LABEL: @test_mm_maskz_cvts_rops_hf8(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s128(<4 x float> %{{.*}})
   // CHECK: zeroinitializer
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s128(<4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm_maskz_cvts_rops_hf8(__U, __A);
 }
 
@@ -333,8 +344,9 @@ __m128i test_mm256_mask_cvts_rops_hf8(__m128i __W, __mmask8 __U, __m256 __A) {
 
 __m128i test_mm256_maskz_cvts_rops_hf8(__mmask8 __U, __m256 __A) {
   // CHECK-LABEL: @test_mm256_maskz_cvts_rops_hf8(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s256(<8 x float> %{{.*}})
   // CHECK: zeroinitializer
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s256(<8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm256_maskz_cvts_rops_hf8(__U, __A);
 }
 
@@ -373,8 +385,9 @@ __m128i test_mm_mask_cvtbiasps_bf8(__m128i __W, __mmask8 __U, __m128i __A, __m12
 
 __m128i test_mm_maskz_cvtbiasps_bf8(__mmask8 __U, __m128i __A, __m128 __B) {
   // CHECK-LABEL: @test_mm_maskz_cvtbiasps_bf8(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8128(<4 x i32> %{{.*}}, <4 x float> %{{.*}})
   // CHECK: zeroinitializer
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8128(<4 x i32> %{{.*}}, <4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm_maskz_cvtbiasps_bf8(__U, __A, __B);
 }
 
@@ -392,8 +405,9 @@ __m128i test_mm256_mask_cvtbiasps_bf8(__m128i __W, __mmask8 __U, __m256i __A, __
 
 __m128i test_mm256_maskz_cvtbiasps_bf8(__mmask8 __U, __m256i __A, __m256 __B) {
   // CHECK-LABEL: @test_mm256_maskz_cvtbiasps_bf8(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8256(<8 x i32> %{{.*}}, <8 x float> %{{.*}})
   // CHECK: zeroinitializer
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8256(<8 x i32> %{{.*}}, <8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm256_maskz_cvtbiasps_bf8(__U, __A, __B);
 }
 
@@ -432,8 +446,9 @@ __m128i test_mm_mask_cvts_biasps_bf8(__m128i __W, __mmask8 __U, __m128i __A, __m
 
 __m128i test_mm_maskz_cvts_biasps_bf8(__mmask8 __U, __m128i __A, __m128 __B) {
   // CHECK-LABEL: @test_mm_maskz_cvts_biasps_bf8(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s128(<4 x i32> %{{.*}}, <4 x float> %{{.*}})
   // CHECK: zeroinitializer
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s128(<4 x i32> %{{.*}}, <4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm_maskz_cvts_biasps_bf8(__U, __A, __B);
 }
 
@@ -451,8 +466,9 @@ __m128i test_mm256_mask_cvts_biasps_bf8(__m128i __W, __mmask8 __U, __m256i __A,
 
 __m128i test_mm256_maskz_cvts_biasps_bf8(__mmask8 __U, __m256i __A, __m256 __B) {
   // CHECK-LABEL: @test_mm256_maskz_cvts_biasps_bf8(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s256(<8 x i32> %{{.*}}, <8 x float> %{{.*}})
   // CHECK: zeroinitializer
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s256(<8 x i32> %{{.*}}, <8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm256_maskz_cvts_biasps_bf8(__U, __A, __B);
 }
 
@@ -491,8 +507,9 @@ __m128i test_mm_mask_cvtbiasps_hf8(__m128i __W, __mmask8 __U, __m128i __A, __m12
 
 __m128i test_mm_maskz_cvtbiasps_hf8(__mmask8 __U, __m128i __A, __m128 __B) {
   // CHECK-LABEL: @test_mm_maskz_cvtbiasps_hf8(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8128(<4 x i32> %{{.*}}, <4 x float> %{{.*}})
   // CHECK: zeroinitializer
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8128(<4 x i32> %{{.*}}, <4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm_maskz_cvtbiasps_hf8(__U, __A, __B);
 }
 
@@ -510,8 +527,9 @@ __m128i test_mm256_mask_cvtbiasps_hf8(__m128i __W, __mmask8 __U, __m256i __A, __
 
 __m128i test_mm256_maskz_cvtbiasps_hf8(__mmask8 __U, __m256i __A, __m256 __B) {
   // CHECK-LABEL: @test_mm256_maskz_cvtbiasps_hf8(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8256(<8 x i32> %{{.*}}, <8 x float> %{{.*}})
   // CHECK: zeroinitializer
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8256(<8 x i32> %{{.*}}, <8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm256_maskz_cvtbiasps_hf8(__U, __A, __B);
 }
 
@@ -550,8 +568,9 @@ __m128i test_mm_mask_cvts_biasps_hf8(__m128i __W, __mmask8 __U, __m128i __A, __m
 
 __m128i test_mm_maskz_cvts_biasps_hf8(__mmask8 __U, __m128i __A, __m128 __B) {
   // CHECK-LABEL: @test_mm_maskz_cvts_biasps_hf8(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s128(<4 x i32> %{{.*}}, <4 x float> %{{.*}})
   // CHECK: zeroinitializer
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s128(<4 x i32> %{{.*}}, <4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm_maskz_cvts_biasps_hf8(__U, __A, __B);
 }
 
@@ -569,8 +588,9 @@ __m128i test_mm256_mask_cvts_biasps_hf8(__m128i __W, __mmask8 __U, __m256i __A,
 
 __m128i test_mm256_maskz_cvts_biasps_hf8(__mmask8 __U, __m256i __A, __m256 __B) {
   // CHECK-LABEL: @test_mm256_maskz_cvts_biasps_hf8(
+  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s256(<8 x i32> %{{.*}}, <8 x float> %{{.*}})
   // CHECK: zeroinitializer
-  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s256(<8 x i32> %{{.*}}, <8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
+  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
   return _mm256_maskz_cvts_biasps_hf8(__U, __A, __B);
 }
 
@@ -984,65 +1004,65 @@ __m512i test_mm512_maskz_cvthf6_hf8(__mmask64 __U, __m512i __A) {
 
 __m128i test_mm_unpack_epi8(__m128i __A) {
   // CHECK-LABEL: @test_mm_unpack_epi8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vunpackb.128(<16 x i8> %{{.*}}, i8 1)
-  return _mm_unpack_epi8(__A, 1);
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vunpackb.128(<16 x i8> %{{.*}}, i8 8)
+  return _mm_unpack_epi8(__A, 8);
 }
 
 __m128i test_mm_mask_unpack_epi8(__m128i __W, __mmask16 __U, __m128i __A) {
   // CHECK-LABEL: @test_mm_mask_unpack_epi8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vunpackb.128(<16 x i8> %{{.*}}, i8 1)
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vunpackb.128(<16 x i8> %{{.*}}, i8 8)
   // CHECK: select <16 x i1> %{{.*}}, <16 x i8> %{{.*}}, <16 x i8> %{{.*}}
-  return _mm_mask_unpack_epi8(__W, __U, __A, 1);
+  return _mm_mask_unpack_epi8(__W, __U, __A, 8);
 }
 
 __m128i test_mm_maskz_unpack_epi8(__mmask16 __U, __m128i __A) {
   // CHECK-LABEL: @test_mm_maskz_unpack_epi8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vunpackb.128(<16 x i8> %{{.*}}, i8 1)
+  // CHECK: call <16 x i8> @llvm.x86.avx10.vunpackb.128(<16 x i8> %{{.*}}, i8 8)
   // CHECK: zeroinitializer
   // CHECK: select <16 x i1> %{{.*}}, <16 x i8> %{{.*}}, <16 x i8> %{{.*}}
-  return _mm_maskz_unpack_epi8(__U, __A, 1);
+  return _mm_maskz_unpack_epi8(__U, __A, 8);
 }
 
 __m256i test_mm256_unpack_epi8(__m256i __A) {
   // CHECK-LABEL: @test_mm256_unpack_epi8(
-  // CHECK: call <32 x i8> @llvm.x86.avx10.vunpackb.256(<32 x i8> %{{.*}}, i8 2)
-  return _mm256_unpack_epi8(__A, 2);
+  // CHECK: call <32 x i8> @llvm.x86.avx10.vunpackb.256(<32 x i8> %{{.*}}, i8 9)
+  return _mm256_unpack_epi8(__A, 9);
 }
 
 __m256i test_mm256_mask_unpack_epi8(__m256i __W, __mmask32 __U, __m256i __A) {
   // CHECK-LABEL: @test_mm256_mask_unpack_epi8(
-  // CHECK: call <32 x i8> @llvm.x86.avx10.vunpackb.256(<32 x i8> %{{.*}}, i8 2)
+  // CHECK: call <32 x i8> @llvm.x86.avx10.vunpackb.256(<32 x i8> %{{.*}}, i8 9)
   // CHECK: select <32 x i1> %{{.*}}, <32 x i8> %{{.*}}, <32 x i8> %{{.*}}
-  return _mm256_mask_unpack_epi8(__W, __U, __A, 2);
+  return _mm256_mask_unpack_epi8(__W, __U, __A, 9);
 }
 
 __m256i test_mm256_maskz_unpack_epi8(__mmask32 __U, __m256i __A) {
   // CHECK-LABEL: @test_mm256_maskz_unpack_epi8(
-  // CHECK: call <32 x i8> @llvm.x86.avx10.vunpackb.256(<32 x i8> %{{.*}}, i8 2)
+  // CHECK: call <32 x i8> @llvm.x86.avx10.vunpackb.256(<32 x i8> %{{.*}}, i8 9)
   // CHECK: zeroinitializer
   // CHECK: select <32 x i1> %{{.*}}, <32 x i8> %{{.*}}, <32 x i8> %{{.*}}
-  return _mm256_maskz_unpack_epi8(__U, __A, 2);
+  return _mm256_maskz_unpack_epi8(__U, __A, 9);
 }
 
 __m512i test_mm512_unpack_epi8(__m512i __A) {
   // CHECK-LABEL: @test_mm512_unpack_epi8(
-  // CHECK: call <64 x i8> @llvm.x86.avx10.vunpackb.512(<64 x i8> %{{.*}}, i8 3)
-  return _mm512_unpack_epi8(__A, 3);
+  // CHECK: call <64 x i8> @llvm.x86.avx10.vunpackb.512(<64 x i8> %{{.*}}, i8 10)
+  return _mm512_unpack_epi8(__A, 10);
 }
 
 __m512i test_mm512_mask_unpack_epi8(__m512i __W, __mmask64 __U, __m512i __A) {
   // CHECK-LABEL: @test_mm512_mask_unpack_epi8(
-  // CHECK: call <64 x i8> @llvm.x86.avx10.vunpackb.512(<64 x i8> %{{.*}}, i8 3)
+  // CHECK: call <64 x i8> @llvm.x86.avx10.vunpackb.512(<64 x i8> %{{.*}}, i8 10)
   // CHECK: select <64 x i1> %{{.*}}, <64 x i8> %{{.*}}, <64 x i8> %{{.*}}
-  return _mm512_mask_unpack_epi8(__W, __U, __A, 3);
+  return _mm512_mask_unpack_epi8(__W, __U, __A, 10);
 }
 
 __m512i test_mm512_maskz_unpack_epi8(__mmask64 __U, __m512i __A) {
   // CHECK-LABEL: @test_mm512_maskz_unpack_epi8(
-  // CHECK: call <64 x i8> @llvm.x86.avx10.vunpackb.512(<64 x i8> %{{.*}}, i8 3)
+  // CHECK: call <64 x i8> @llvm.x86.avx10.vunpackb.512(<64 x i8> %{{.*}}, i8 10)
   // CHECK: zeroinitializer
   // CHECK: select <64 x i1> %{{.*}}, <64 x i8> %{{.*}}, <64 x i8> %{{.*}}
-  return _mm512_maskz_unpack_epi8(__U, __A, 3);
+  return _mm512_maskz_unpack_epi8(__U, __A, 10);
 }
 
 __m512i test_mm512_unpack_epi8_compose(__m512i __A) {
@@ -1064,6 +1084,13 @@ __m128i test_mm_unpack_epi8_compose(__m128i __A) {
   return _mm_unpack_epi8(__A, _MM_UNPACKB_SIZE(7) | _MM_UNPACKB_SEXT);
 }
 
+__m512i test_mm512_unpack_epi8_compose_example(__m512i __A) {
+  // CHECK-LABEL: @test_mm512_unpack_epi8_compose_example(
+  // CHECK: call <64 x i8> @llvm.x86.avx10.vunpackb.512(<64 x i8> %{{.*}}, i8 49)
+  return _mm512_unpack_epi8(
+      __A, _MM_UNPACKB_SIZE(4) | _MM_UNPACKB_START(1) | _MM_UNPACKB_SEXT);
+}
+
 __m128i test_mm_cvtss_epi32_epi8(__m128i __A) {
   // CHECK-LABEL: @test_mm_cvtss_epi32_epi8(
   // CHECK: call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.128(<4 x i32> %{{.*}}, <16 x i8> %{{.*}}, i8 -1)
diff --git a/clang/test/CodeGen/attr-target-x86.c b/clang/test/CodeGen/attr-target-x86.c
index 3656a479f55faa..deda204402a711 100644
--- a/clang/test/CodeGen/attr-target-x86.c
+++ b/clang/test/CodeGen/attr-target-x86.c
@@ -214,7 +214,7 @@ void f_x86_64_v4(void) {}
 __attribute__((target("avx10.1")))
 void f_avx10_1(void) {}
 
-// CHECK: [[f_avx10_v2_aux]] = {{.*}}"target-cpu"="i686" "target-features"="{{.*}}+avx10v2aux{{.*}}"
+// CHECK: [[f_avx10_v2_aux]] = {{.*}}"target-cpu"="i686" "target-features"="+avx,+avx10.1,+avx10v2aux,+avx2,+avx512bf16,+avx512bitalg,+avx512bw,+avx512cd,+avx512dq,+avx512f,+avx512fp16,+avx512ifma,+avx512vbmi,+avx512vbmi2,+avx512vl,+avx512vnni,+avx512vpopcntdq,+cmov,+crc32,+cx8,+f16c,+fma,+mmx,+popcnt,+sse,+sse2,+sse3,+sse4.1,+sse4.2,+ssse3,+x87,+xsave"
 __attribute__((target("avx10v2aux")))
 void f_avx10_v2_aux(void) {}
 
diff --git a/clang/test/CodeGen/builtin-cpu-supports-all.c b/clang/test/CodeGen/builtin-cpu-supports-all.c
index 73ccf88dffd876..12715f6fe22042 100644
--- a/clang/test/CodeGen/builtin-cpu-supports-all.c
+++ b/clang/test/CodeGen/builtin-cpu-supports-all.c
@@ -541,3 +541,8 @@ TEST_CPU_SUPPORTS(amx_movrs, "amx-movrs")
 // CHECK: [[LOAD:%[^ ]+]] = load i32, ptr getelementptr inbounds nuw (i8, ptr @__cpu_features2, i64 8)
 // CHECK: = and i32 [[LOAD]], 134217728
 TEST_CPU_SUPPORTS(avx512bmm, "avx512bmm")
+
+// CHECK-LABEL: define{{.*}} void @test_avx10v2aux(
+// CHECK: [[LOAD:%[^ ]+]] = load i32, ptr getelementptr inbounds nuw (i8, ptr @__cpu_features2, i64 8)
+// CHECK: = and i32 [[LOAD]], 268435456
+TEST_CPU_SUPPORTS(avx10v2aux, "avx10v2aux")
diff --git a/clang/test/CodeGen/target-builtin-noerror.c b/clang/test/CodeGen/target-builtin-noerror.c
index 86e74154cef4ce..71cd3e51f6ca87 100644
--- a/clang/test/CodeGen/target-builtin-noerror.c
+++ b/clang/test/CodeGen/target-builtin-noerror.c
@@ -144,6 +144,7 @@ void verifyfeaturestrings(void) {
   (void)__builtin_cpu_supports("usermsr");
   (void)__builtin_cpu_supports("avx10.1");
   (void)__builtin_cpu_supports("avx10.2");
+  (void)__builtin_cpu_supports("avx10v2aux");
   (void)__builtin_cpu_supports("movrs");
 }
 
diff --git a/clang/test/Sema/attr-target-mv.c b/clang/test/Sema/attr-target-mv.c
index e77a1888595f23..1f3ec79bf04c9d 100644
--- a/clang/test/Sema/attr-target-mv.c
+++ b/clang/test/Sema/attr-target-mv.c
@@ -189,6 +189,10 @@ int __attribute__((target("sha"))) no_priority3(void);
 int __attribute__((target("default"))) apxf_mv(void) { return 0; }
 int __attribute__((target("apxf"))) apxf_mv(void) { return 1; }
 
+int __attribute__((target("default"))) avx10v2aux_mv(void);
+// expected-error at +1 {{function multiversioning doesn't support feature 'avx10v2aux'}}
+int __attribute__((target("avx10v2aux"))) avx10v2aux_mv(void);
+
 // expected-error at +2 {{function multiversioning doesn't support feature 'ndd'}}
 // expected-note at +2 {{function multiversioning caused by this declaration}}
 int __attribute__((target("ndd"))) apx_sub(void);
diff --git a/compiler-rt/lib/builtins/cpu_model/x86.c b/compiler-rt/lib/builtins/cpu_model/x86.c
index ae1abf70396dae..e88bbe531570b2 100644
--- a/compiler-rt/lib/builtins/cpu_model/x86.c
+++ b/compiler-rt/lib/builtins/cpu_model/x86.c
@@ -242,6 +242,7 @@ enum ProcessorFeatures {
   FEATURE_MOVRS,
   FEATURE_AMX_MOVRS,
   FEATURE_AVX512BMM,
+  FEATURE_AVX10_V2_AUX = 124,
   CPU_FEATURE_MAX
 };
 
@@ -1132,6 +1133,7 @@ static void getAvailableFeatures(unsigned ECX, unsigned EDX, unsigned MaxLeaf,
     setFeature(FEATURE_USERMSR);
   if (HasLeaf7Subleaf1 && ((EDX >> 21) & 1) && HasAPXSave)
     setFeature(FEATURE_APXF);
+  bool HasAVX10 = HasLeaf7Subleaf1 && ((EDX >> 19) & 1) && HasAVX512Save;
 
   unsigned MaxLevel = 0;
   getX86CpuIDAndInfo(0, &MaxLevel, &EBX, &ECX, &EDX);
@@ -1155,12 +1157,19 @@ static void getAvailableFeatures(unsigned ECX, unsigned EDX, unsigned MaxLeaf,
 
   bool HasLeaf24 = MaxLevel >= 0x24 &&
                    !getX86CpuIDAndInfoEx(0x24, 0x0, &EAX, &EBX, &ECX, &EDX);
-  if (HasLeaf7Subleaf1 && ((EDX >> 19) & 1) && HasLeaf24) {
+  unsigned Leaf24MaxSubleaf = EAX;
+  if (HasAVX10 && HasLeaf24) {
     int AVX10Ver = EBX & 0xff;
     if (AVX10Ver >= 1)
       setFeature(FEATURE_AVX10_1);
     if (AVX10Ver >= 2)
       setFeature(FEATURE_AVX10_2);
+    if (Leaf24MaxSubleaf >= 1) {
+      unsigned EAX1, EBX1, ECX1, EDX1;
+      if (!getX86CpuIDAndInfoEx(0x24, 0x1, &EAX1, &EBX1, &ECX1, &EDX1) &&
+          ((ECX1 >> 3) & 1))
+        setFeature(FEATURE_AVX10_V2_AUX);
+    }
   }
 
   unsigned MaxExtLevel = 0;
diff --git a/llvm/include/llvm/TargetParser/X86TargetParser.def b/llvm/include/llvm/TargetParser/X86TargetParser.def
index 3aec4ec0eb4c46..2cf0a1feab13f2 100644
--- a/llvm/include/llvm/TargetParser/X86TargetParser.def
+++ b/llvm/include/llvm/TargetParser/X86TargetParser.def
@@ -244,6 +244,7 @@ X86_FEATURE_COMPAT(AMX_FP8,            "amx-fp8",                0, 120)
 X86_FEATURE_COMPAT(MOVRS,              "movrs",                  0, 121)
 X86_FEATURE_COMPAT(AMX_MOVRS,          "amx-movrs",              0, 122)
 X86_FEATURE_COMPAT(AVX512BMM,          "avx512bmm",              0, 123)
+X86_FEATURE_COMPAT(AVX10_V2_AUX,       "avx10v2aux",             0, 124)
 
 // Features we don't multiversion on.
 X86_FEATURE       (NF,                 "nf")
@@ -263,7 +264,6 @@ X86_FEATURE       (NDD,                "ndd")
 X86_FEATURE       (EGPR,               "egpr")
 X86_FEATURE       (ZU,                 "zu")
 X86_FEATURE       (JMPABS,             "jmpabs")
-X86_FEATURE       (AVX10_V2_AUX,       "avx10v2aux")
 
 // These features aren't really CPU features, but the frontend can set them.
 X86_FEATURE       (RETPOLINE_EXTERNAL_THUNK,    "retpoline-external-thunk")
diff --git a/llvm/lib/Target/X86/X86ISelLowering.cpp b/llvm/lib/Target/X86/X86ISelLowering.cpp
index cb8a60e8bf3324..bff9de55aec272 100644
--- a/llvm/lib/Target/X86/X86ISelLowering.cpp
+++ b/llvm/lib/Target/X86/X86ISelLowering.cpp
@@ -28780,15 +28780,15 @@ static SDValue LowerINTRINSIC_W_CHAIN(SDValue Op, const X86Subtarget &Subtarget,
       SDVTList VTs = DAG.getVTList(MVT::Other);
       if (isAllOnesConstant(Mask)) {
         SDValue Ops[] = {Chain, DataToTruncate, Addr};
-        return DAG.getMemIntrinsicNode(X86ISD::VTRUNCSTORSS, dl, VTs, Ops,
+        return DAG.getMemIntrinsicNode(X86ISD::VTRUNCSTORESS, dl, VTs, Ops,
                                        MemVT, MemIntr->getMemOperand());
       }
 
       MVT MaskVT = MVT::getVectorVT(MVT::i1, MemVT.getVectorNumElements());
       SDValue VMask = getMaskNode(Mask, MaskVT, Subtarget, DAG, dl);
       SDValue Ops[] = {Chain, DataToTruncate, Addr, VMask};
-      return DAG.getMemIntrinsicNode(X86ISD::VMTRUNCSTORSS, dl, VTs, Ops, MemVT,
-                                     MemIntr->getMemOperand());
+      return DAG.getMemIntrinsicNode(X86ISD::VMTRUNCSTORESS, dl, VTs, Ops,
+                                     MemVT, MemIntr->getMemOperand());
     }
     default:
       llvm_unreachable("Unsupported truncstore intrinsic");
@@ -55292,6 +55292,17 @@ static SDValue combineStore(SDNode *N, SelectionDAG &DAG,
                            St->getMemOperand(), DAG);
   }
 
+  // Try to fold a VTRUNCSS into a truncating store.
+  if (!St->isTruncatingStore() && StoredVal.getOpcode() == X86ISD::VTRUNCSS &&
+      StoredVal.hasOneUse() &&
+      TLI.isTruncStoreLegal(StoredVal.getOperand(0).getValueType(), VT,
+                            St->getAlign(), St->getAddressSpace())) {
+    SDVTList VTs = DAG.getVTList(MVT::Other);
+    SDValue Ops[] = {St->getChain(), StoredVal.getOperand(0), St->getBasePtr()};
+    return DAG.getMemIntrinsicNode(X86ISD::VTRUNCSTORESS, dl, VTs, Ops, VT,
+                                   St->getMemOperand());
+  }
+
   // Try to fold a extract_element(VTRUNC) pattern into a truncating store.
   if (!St->isTruncatingStore()) {
     auto IsExtractedElement = [](SDValue V) {
diff --git a/llvm/lib/Target/X86/X86InstrAVX10_V2_AUX.td b/llvm/lib/Target/X86/X86InstrAVX10_V2_AUX.td
index 0e09d7db4b1f01..30d5083c7b3393 100644
--- a/llvm/lib/Target/X86/X86InstrAVX10_V2_AUX.td
+++ b/llvm/lib/Target/X86/X86InstrAVX10_V2_AUX.td
@@ -40,19 +40,57 @@ multiclass avx10_v2aux_cvt_trunc_ps2i8<bits<8> opc, string OpcodeStr,
     }
   }
 
-  // InstAliases for x/y suffixes
+  // InstAliases for x/y suffixes. Priority 0 so they are accepted but not
+  // printed by default.
   def : InstAlias<OpcodeStr#"x\t{$src, $dst|$dst, $src}",
                   (!cast<Instruction>(NAME # "Z128rr") VR128X:$dst,
-                   VR128X:$src), 0>;
+                   VR128X:$src), 0, "att">;
+  def : InstAlias<OpcodeStr#"x\t{$src, $dst {${mask}}|$dst {${mask}}, $src}",
+                  (!cast<Instruction>(NAME # "Z128rrk") VR128X:$dst,
+                   VK4WM:$mask, VR128X:$src), 0, "att">;
+  def : InstAlias<OpcodeStr#"x\t{$src, $dst {${mask}} {z}|"
+                  "$dst {${mask}} {z}, $src}",
+                  (!cast<Instruction>(NAME # "Z128rrkz") VR128X:$dst,
+                   VK4WM:$mask, VR128X:$src), 0, "att">;
   def : InstAlias<OpcodeStr#"x\t{$src, $dst|$dst, $src}",
                   (!cast<Instruction>(NAME # "Z128rm") VR128X:$dst,
                    f128mem:$src), 0, "intel">;
+  def : InstAlias<OpcodeStr#"x\t{${src}{1to4}, $dst|$dst, ${src}{1to4}}",
+                  (!cast<Instruction>(NAME # "Z128rmb") VR128X:$dst,
+                   f32mem:$src), 0, "att">;
+  def : InstAlias<OpcodeStr#"x\t{${src}{1to4}, $dst {${mask}}|"
+                  "$dst {${mask}}, ${src}{1to4}}",
+                  (!cast<Instruction>(NAME # "Z128rmbk") VR128X:$dst,
+                   VK4WM:$mask, f32mem:$src), 0, "att">;
+  def : InstAlias<OpcodeStr#"x\t{${src}{1to4}, $dst {${mask}} {z}|"
+                  "$dst {${mask}} {z}, ${src}{1to4}}",
+                  (!cast<Instruction>(NAME # "Z128rmbkz") VR128X:$dst,
+                   VK4WM:$mask, f32mem:$src), 0, "att">;
+
   def : InstAlias<OpcodeStr#"y\t{$src, $dst|$dst, $src}",
                   (!cast<Instruction>(NAME # "Z256rr") VR128X:$dst,
-                   VR256X:$src), 0>;
+                   VR256X:$src), 0, "att">;
+  def : InstAlias<OpcodeStr#"y\t{$src, $dst {${mask}}|$dst {${mask}}, $src}",
+                  (!cast<Instruction>(NAME # "Z256rrk") VR128X:$dst,
+                   VK8WM:$mask, VR256X:$src), 0, "att">;
+  def : InstAlias<OpcodeStr#"y\t{$src, $dst {${mask}} {z}|"
+                  "$dst {${mask}} {z}, $src}",
+                  (!cast<Instruction>(NAME # "Z256rrkz") VR128X:$dst,
+                   VK8WM:$mask, VR256X:$src), 0, "att">;
   def : InstAlias<OpcodeStr#"y\t{$src, $dst|$dst, $src}",
                   (!cast<Instruction>(NAME # "Z256rm") VR128X:$dst,
                    f256mem:$src), 0, "intel">;
+  def : InstAlias<OpcodeStr#"y\t{${src}{1to8}, $dst|$dst, ${src}{1to8}}",
+                  (!cast<Instruction>(NAME # "Z256rmb") VR128X:$dst,
+                   f32mem:$src), 0, "att">;
+  def : InstAlias<OpcodeStr#"y\t{${src}{1to8}, $dst {${mask}}|"
+                  "$dst {${mask}}, ${src}{1to8}}",
+                  (!cast<Instruction>(NAME # "Z256rmbk") VR128X:$dst,
+                   VK8WM:$mask, f32mem:$src), 0, "att">;
+  def : InstAlias<OpcodeStr#"y\t{${src}{1to8}, $dst {${mask}} {z}|"
+                  "$dst {${mask}} {z}, ${src}{1to8}}",
+                  (!cast<Instruction>(NAME # "Z256rmbkz") VR128X:$dst,
+                   VK8WM:$mask, f32mem:$src), 0, "att">;
 
   // Explicit patterns for Z256 (8 source elements, VK8WM mask)
   // Unmasked
diff --git a/llvm/lib/Target/X86/X86InstrFragmentsSIMD.td b/llvm/lib/Target/X86/X86InstrFragmentsSIMD.td
index bc6a159d1158e3..841634a0c43222 100644
--- a/llvm/lib/Target/X86/X86InstrFragmentsSIMD.td
+++ b/llvm/lib/Target/X86/X86InstrFragmentsSIMD.td
@@ -1672,11 +1672,11 @@ def X86MTruncUSStore : SDNode<"X86ISD::VMTRUNCSTOREUS",  SDTX86MaskedStore,
                        [SDNPHasChain, SDNPMayStore, SDNPMemOperand]>;
 
 // Vector truncating store with symmetric signed saturation
-def X86TruncSSStore : SDNode<"X86ISD::VTRUNCSTORSS",  SDTStore,
+def X86TruncSSStore : SDNode<"X86ISD::VTRUNCSTORESS",  SDTStore,
                        [SDNPHasChain, SDNPMayStore, SDNPMemOperand]>;
 
 // Vector truncating masked store with symmetric signed saturation
-def X86MTruncSSStore : SDNode<"X86ISD::VMTRUNCSTORSS",  SDTX86MaskedStore,
+def X86MTruncSSStore : SDNode<"X86ISD::VMTRUNCSTORESS",  SDTX86MaskedStore,
                        [SDNPHasChain, SDNPMayStore, SDNPMemOperand]>;
 
 def truncstore_s_vi8 : PatFrag<(ops node:$val, node:$ptr),
diff --git a/llvm/lib/TargetParser/Host.cpp b/llvm/lib/TargetParser/Host.cpp
index a1a3f54581fa00..d89133d534b56d 100644
--- a/llvm/lib/TargetParser/Host.cpp
+++ b/llvm/lib/TargetParser/Host.cpp
@@ -2279,8 +2279,7 @@ StringMap<bool> sys::getHostCPUFeatures() {
   bool HasLeaf24Subleaf1 =
       HasLeaf24 && EAX >= 1 &&
       !getX86CpuIDAndInfoEx(0x24, 0x1, &EAX, &EBX, &ECX, &EDX);
-  Features["avx10v2aux"] =
-      HasAVX10 && HasLeaf24Subleaf1 && ((ECX >> 3) & 1) && HasAVX512Save;
+  Features["avx10v2aux"] = HasAVX10 && HasLeaf24Subleaf1 && ((ECX >> 3) & 1);
 
   return Features;
 }
diff --git a/llvm/test/CodeGen/X86/avx10_v2aux-cvt-fp4-fp6-intrinsics.ll b/llvm/test/CodeGen/X86/avx10_v2aux-cvt-fp4-fp6-intrinsics.ll
index 09ca861bf73f98..8cdf142ea84762 100644
--- a/llvm/test/CodeGen/X86/avx10_v2aux-cvt-fp4-fp6-intrinsics.ll
+++ b/llvm/test/CodeGen/X86/avx10_v2aux-cvt-fp4-fp6-intrinsics.ll
@@ -740,6 +740,120 @@ define <64 x i8> @test_int_x86_avx10_vcvtbf62hf8512_mem(ptr %ptr_a) {
   ret <64 x i8> %ret
 }
 
+define <16 x i8> @test_int_x86_avx10_vcvtbf62hf8128_mask(<16 x i8> %a, <16 x i8> %src, i16 %mask) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf62hf8128_mask:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbf62hf8 %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0xfd,0x09,0x37,0xc8]
+; X64-NEXT:    vmovdqa %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf9,0x6f,0xc1]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbf62hf8128_mask:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbf62hf8 %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0xfd,0x09,0x37,0xc8]
+; X86-NEXT:    vmovdqa %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf9,0x6f,0xc1]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %cvt = call <16 x i8> @llvm.x86.avx10.vcvtbf62hf8128(<16 x i8> %a)
+  %msk = bitcast i16 %mask to <16 x i1>
+  %ret = select <16 x i1> %msk, <16 x i8> %cvt, <16 x i8> %src
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvtbf62hf8128_maskz(<16 x i8> %a, i16 %mask) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf62hf8128_maskz:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbf62hf8 %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0xfd,0x89,0x37,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbf62hf8128_maskz:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbf62hf8 %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0xfd,0x89,0x37,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %cvt = call <16 x i8> @llvm.x86.avx10.vcvtbf62hf8128(<16 x i8> %a)
+  %msk = bitcast i16 %mask to <16 x i1>
+  %ret = select <16 x i1> %msk, <16 x i8> %cvt, <16 x i8> zeroinitializer
+  ret <16 x i8> %ret
+}
+
+define <32 x i8> @test_int_x86_avx10_vcvtbf62hf8256_mask(<32 x i8> %a, <32 x i8> %src, i32 %mask) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf62hf8256_mask:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbf62hf8 %ymm0, %ymm1 {%k1} # encoding: [0x62,0xf5,0xfd,0x29,0x37,0xc8]
+; X64-NEXT:    vmovdqa %ymm1, %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfd,0x6f,0xc1]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbf62hf8256_mask:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovd {{[0-9]+}}(%esp), %k1 # encoding: [0xc4,0xe1,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbf62hf8 %ymm0, %ymm1 {%k1} # encoding: [0x62,0xf5,0xfd,0x29,0x37,0xc8]
+; X86-NEXT:    vmovdqa %ymm1, %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfd,0x6f,0xc1]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %cvt = call <32 x i8> @llvm.x86.avx10.vcvtbf62hf8256(<32 x i8> %a)
+  %msk = bitcast i32 %mask to <32 x i1>
+  %ret = select <32 x i1> %msk, <32 x i8> %cvt, <32 x i8> %src
+  ret <32 x i8> %ret
+}
+
+define <32 x i8> @test_int_x86_avx10_vcvtbf62hf8256_maskz(<32 x i8> %a, i32 %mask) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf62hf8256_maskz:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbf62hf8 %ymm0, %ymm0 {%k1} {z} # encoding: [0x62,0xf5,0xfd,0xa9,0x37,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbf62hf8256_maskz:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovd {{[0-9]+}}(%esp), %k1 # encoding: [0xc4,0xe1,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbf62hf8 %ymm0, %ymm0 {%k1} {z} # encoding: [0x62,0xf5,0xfd,0xa9,0x37,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %cvt = call <32 x i8> @llvm.x86.avx10.vcvtbf62hf8256(<32 x i8> %a)
+  %msk = bitcast i32 %mask to <32 x i1>
+  %ret = select <32 x i1> %msk, <32 x i8> %cvt, <32 x i8> zeroinitializer
+  ret <32 x i8> %ret
+}
+
+define <64 x i8> @test_int_x86_avx10_vcvtbf62hf8512_mask(<64 x i8> %a, <64 x i8> %src, i64 %mask) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf62hf8512_mask:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovq %rdi, %k1 # encoding: [0xc4,0xe1,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbf62hf8 %zmm0, %zmm1 {%k1} # encoding: [0x62,0xf5,0xfd,0x49,0x37,0xc8]
+; X64-NEXT:    vmovdqa64 %zmm1, %zmm0 # encoding: [0x62,0xf1,0xfd,0x48,0x6f,0xc1]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbf62hf8512_mask:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovq {{[0-9]+}}(%esp), %k1 # encoding: [0xc4,0xe1,0xf8,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbf62hf8 %zmm0, %zmm1 {%k1} # encoding: [0x62,0xf5,0xfd,0x49,0x37,0xc8]
+; X86-NEXT:    vmovdqa64 %zmm1, %zmm0 # encoding: [0x62,0xf1,0xfd,0x48,0x6f,0xc1]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %cvt = call <64 x i8> @llvm.x86.avx10.vcvtbf62hf8512(<64 x i8> %a)
+  %msk = bitcast i64 %mask to <64 x i1>
+  %ret = select <64 x i1> %msk, <64 x i8> %cvt, <64 x i8> %src
+  ret <64 x i8> %ret
+}
+
+define <64 x i8> @test_int_x86_avx10_vcvtbf62hf8512_maskz(<64 x i8> %a, i64 %mask) {
+; X64-LABEL: test_int_x86_avx10_vcvtbf62hf8512_maskz:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovq %rdi, %k1 # encoding: [0xc4,0xe1,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvtbf62hf8 %zmm0, %zmm0 {%k1} {z} # encoding: [0x62,0xf5,0xfd,0xc9,0x37,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvtbf62hf8512_maskz:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovq {{[0-9]+}}(%esp), %k1 # encoding: [0xc4,0xe1,0xf8,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvtbf62hf8 %zmm0, %zmm0 {%k1} {z} # encoding: [0x62,0xf5,0xfd,0xc9,0x37,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %cvt = call <64 x i8> @llvm.x86.avx10.vcvtbf62hf8512(<64 x i8> %a)
+  %msk = bitcast i64 %mask to <64 x i1>
+  %ret = select <64 x i1> %msk, <64 x i8> %cvt, <64 x i8> zeroinitializer
+  ret <64 x i8> %ret
+}
+
 define <16 x i8> @test_int_x86_avx10_vcvthf62hf8128(<16 x i8> %a) {
 ; CHECK-LABEL: test_int_x86_avx10_vcvthf62hf8128:
 ; CHECK:       # %bb.0:
@@ -824,3 +938,117 @@ define <64 x i8> @test_int_x86_avx10_vcvthf62hf8512_mem(ptr %ptr_a) {
   %ret = call <64 x i8> @llvm.x86.avx10.vcvthf62hf8512(<64 x i8> %a)
   ret <64 x i8> %ret
 }
+
+define <16 x i8> @test_int_x86_avx10_vcvthf62hf8128_mask(<16 x i8> %a, <16 x i8> %src, i16 %mask) {
+; X64-LABEL: test_int_x86_avx10_vcvthf62hf8128_mask:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvthf62hf8 %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7d,0x09,0x37,0xc8]
+; X64-NEXT:    vmovdqa %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf9,0x6f,0xc1]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvthf62hf8128_mask:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvthf62hf8 %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf5,0x7d,0x09,0x37,0xc8]
+; X86-NEXT:    vmovdqa %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf9,0x6f,0xc1]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %cvt = call <16 x i8> @llvm.x86.avx10.vcvthf62hf8128(<16 x i8> %a)
+  %msk = bitcast i16 %mask to <16 x i1>
+  %ret = select <16 x i1> %msk, <16 x i8> %cvt, <16 x i8> %src
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vcvthf62hf8128_maskz(<16 x i8> %a, i16 %mask) {
+; X64-LABEL: test_int_x86_avx10_vcvthf62hf8128_maskz:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvthf62hf8 %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0x89,0x37,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvthf62hf8128_maskz:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvthf62hf8 %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0x89,0x37,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %cvt = call <16 x i8> @llvm.x86.avx10.vcvthf62hf8128(<16 x i8> %a)
+  %msk = bitcast i16 %mask to <16 x i1>
+  %ret = select <16 x i1> %msk, <16 x i8> %cvt, <16 x i8> zeroinitializer
+  ret <16 x i8> %ret
+}
+
+define <32 x i8> @test_int_x86_avx10_vcvthf62hf8256_mask(<32 x i8> %a, <32 x i8> %src, i32 %mask) {
+; X64-LABEL: test_int_x86_avx10_vcvthf62hf8256_mask:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvthf62hf8 %ymm0, %ymm1 {%k1} # encoding: [0x62,0xf5,0x7d,0x29,0x37,0xc8]
+; X64-NEXT:    vmovdqa %ymm1, %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfd,0x6f,0xc1]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvthf62hf8256_mask:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovd {{[0-9]+}}(%esp), %k1 # encoding: [0xc4,0xe1,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvthf62hf8 %ymm0, %ymm1 {%k1} # encoding: [0x62,0xf5,0x7d,0x29,0x37,0xc8]
+; X86-NEXT:    vmovdqa %ymm1, %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfd,0x6f,0xc1]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %cvt = call <32 x i8> @llvm.x86.avx10.vcvthf62hf8256(<32 x i8> %a)
+  %msk = bitcast i32 %mask to <32 x i1>
+  %ret = select <32 x i1> %msk, <32 x i8> %cvt, <32 x i8> %src
+  ret <32 x i8> %ret
+}
+
+define <32 x i8> @test_int_x86_avx10_vcvthf62hf8256_maskz(<32 x i8> %a, i32 %mask) {
+; X64-LABEL: test_int_x86_avx10_vcvthf62hf8256_maskz:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvthf62hf8 %ymm0, %ymm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0xa9,0x37,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvthf62hf8256_maskz:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovd {{[0-9]+}}(%esp), %k1 # encoding: [0xc4,0xe1,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvthf62hf8 %ymm0, %ymm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0xa9,0x37,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %cvt = call <32 x i8> @llvm.x86.avx10.vcvthf62hf8256(<32 x i8> %a)
+  %msk = bitcast i32 %mask to <32 x i1>
+  %ret = select <32 x i1> %msk, <32 x i8> %cvt, <32 x i8> zeroinitializer
+  ret <32 x i8> %ret
+}
+
+define <64 x i8> @test_int_x86_avx10_vcvthf62hf8512_mask(<64 x i8> %a, <64 x i8> %src, i64 %mask) {
+; X64-LABEL: test_int_x86_avx10_vcvthf62hf8512_mask:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovq %rdi, %k1 # encoding: [0xc4,0xe1,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvthf62hf8 %zmm0, %zmm1 {%k1} # encoding: [0x62,0xf5,0x7d,0x49,0x37,0xc8]
+; X64-NEXT:    vmovdqa64 %zmm1, %zmm0 # encoding: [0x62,0xf1,0xfd,0x48,0x6f,0xc1]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvthf62hf8512_mask:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovq {{[0-9]+}}(%esp), %k1 # encoding: [0xc4,0xe1,0xf8,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvthf62hf8 %zmm0, %zmm1 {%k1} # encoding: [0x62,0xf5,0x7d,0x49,0x37,0xc8]
+; X86-NEXT:    vmovdqa64 %zmm1, %zmm0 # encoding: [0x62,0xf1,0xfd,0x48,0x6f,0xc1]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %cvt = call <64 x i8> @llvm.x86.avx10.vcvthf62hf8512(<64 x i8> %a)
+  %msk = bitcast i64 %mask to <64 x i1>
+  %ret = select <64 x i1> %msk, <64 x i8> %cvt, <64 x i8> %src
+  ret <64 x i8> %ret
+}
+
+define <64 x i8> @test_int_x86_avx10_vcvthf62hf8512_maskz(<64 x i8> %a, i64 %mask) {
+; X64-LABEL: test_int_x86_avx10_vcvthf62hf8512_maskz:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovq %rdi, %k1 # encoding: [0xc4,0xe1,0xfb,0x92,0xcf]
+; X64-NEXT:    vcvthf62hf8 %zmm0, %zmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0xc9,0x37,0xc0]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vcvthf62hf8512_maskz:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovq {{[0-9]+}}(%esp), %k1 # encoding: [0xc4,0xe1,0xf8,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vcvthf62hf8 %zmm0, %zmm0 {%k1} {z} # encoding: [0x62,0xf5,0x7d,0xc9,0x37,0xc0]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %cvt = call <64 x i8> @llvm.x86.avx10.vcvthf62hf8512(<64 x i8> %a)
+  %msk = bitcast i64 %mask to <64 x i1>
+  %ret = select <64 x i1> %msk, <64 x i8> %cvt, <64 x i8> zeroinitializer
+  ret <64 x i8> %ret
+}
diff --git a/llvm/test/CodeGen/X86/avx10_v2aux-pmovssdb-intrinsics.ll b/llvm/test/CodeGen/X86/avx10_v2aux-pmovssdb-intrinsics.ll
index 77807b58a70753..85e5d2e4439a4a 100644
--- a/llvm/test/CodeGen/X86/avx10_v2aux-pmovssdb-intrinsics.ll
+++ b/llvm/test/CodeGen/X86/avx10_v2aux-pmovssdb-intrinsics.ll
@@ -247,3 +247,21 @@ define void @test_int_x86_avx10_mask_pmovssdb_store_512(ptr %ptr, <16 x i32> %a,
   call void @llvm.x86.avx10.mask.pmovss.db.mem.512(ptr %ptr, <16 x i32> %a, i16 %mask)
   ret void
 }
+
+define void @test_int_x86_avx10_pmovssdb_store_vtruncss_512(<16 x i32> %a, ptr %p) {
+; X64-LABEL: test_int_x86_avx10_pmovssdb_store_vtruncss_512:
+; X64:       # %bb.0:
+; X64-NEXT:    vpmovssdb %zmm0, (%rdi) # encoding: [0x62,0xf2,0x7e,0x48,0x41,0x07]
+; X64-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_pmovssdb_store_vtruncss_512:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    vpmovssdb %zmm0, (%eax) # encoding: [0x62,0xf2,0x7e,0x48,0x41,0x00]
+; X86-NEXT:    vzeroupper # encoding: [0xc5,0xf8,0x77]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %t = call <16 x i8> @llvm.x86.avx10.mask.pmovss.db.512(<16 x i32> %a, <16 x i8> zeroinitializer, i16 -1)
+  store <16 x i8> %t, ptr %p
+  ret void
+}
diff --git a/llvm/test/CodeGen/X86/avx10_v2aux-unpackb-intrinsics.ll b/llvm/test/CodeGen/X86/avx10_v2aux-unpackb-intrinsics.ll
index 7d0292c6afd417..e5c5a7f42d1f6f 100644
--- a/llvm/test/CodeGen/X86/avx10_v2aux-unpackb-intrinsics.ll
+++ b/llvm/test/CodeGen/X86/avx10_v2aux-unpackb-intrinsics.ll
@@ -80,3 +80,157 @@ define <64 x i8> @test_int_x86_avx10_vunpackb_mem_512(ptr %ptr_a) {
   %ret = call <64 x i8> @llvm.x86.avx10.vunpackb.512(<64 x i8> %a, i8 3)
   ret <64 x i8> %ret
 }
+
+define <16 x i8> @test_int_x86_avx10_vunpackb_128_mask(<16 x i8> %a, <16 x i8> %src, i16 %mask) {
+; X64-LABEL: test_int_x86_avx10_vunpackb_128_mask:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vunpackb $1, %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf3,0x7c,0x09,0x3d,0xc8,0x01]
+; X64-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vunpackb_128_mask:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vunpackb $1, %xmm0, %xmm1 {%k1} # encoding: [0x62,0xf3,0x7c,0x09,0x3d,0xc8,0x01]
+; X86-NEXT:    vmovaps %xmm1, %xmm0 # EVEX TO VEX Compression encoding: [0xc5,0xf8,0x28,0xc1]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %unp = call <16 x i8> @llvm.x86.avx10.vunpackb.128(<16 x i8> %a, i8 1)
+  %msk = bitcast i16 %mask to <16 x i1>
+  %ret = select <16 x i1> %msk, <16 x i8> %unp, <16 x i8> %src
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vunpackb_128_maskz(<16 x i8> %a, i16 %mask) {
+; X64-LABEL: test_int_x86_avx10_vunpackb_128_maskz:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vunpackb $1, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf3,0x7c,0x89,0x3d,0xc0,0x01]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vunpackb_128_maskz:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vunpackb $1, %xmm0, %xmm0 {%k1} {z} # encoding: [0x62,0xf3,0x7c,0x89,0x3d,0xc0,0x01]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %unp = call <16 x i8> @llvm.x86.avx10.vunpackb.128(<16 x i8> %a, i8 1)
+  %msk = bitcast i16 %mask to <16 x i1>
+  %ret = select <16 x i1> %msk, <16 x i8> %unp, <16 x i8> zeroinitializer
+  ret <16 x i8> %ret
+}
+
+define <32 x i8> @test_int_x86_avx10_vunpackb_256_mask(<32 x i8> %a, <32 x i8> %src, i32 %mask) {
+; X64-LABEL: test_int_x86_avx10_vunpackb_256_mask:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vunpackb $2, %ymm0, %ymm1 {%k1} # encoding: [0x62,0xf3,0x7c,0x29,0x3d,0xc8,0x02]
+; X64-NEXT:    vmovaps %ymm1, %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfc,0x28,0xc1]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vunpackb_256_mask:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovd {{[0-9]+}}(%esp), %k1 # encoding: [0xc4,0xe1,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vunpackb $2, %ymm0, %ymm1 {%k1} # encoding: [0x62,0xf3,0x7c,0x29,0x3d,0xc8,0x02]
+; X86-NEXT:    vmovaps %ymm1, %ymm0 # EVEX TO VEX Compression encoding: [0xc5,0xfc,0x28,0xc1]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %unp = call <32 x i8> @llvm.x86.avx10.vunpackb.256(<32 x i8> %a, i8 2)
+  %msk = bitcast i32 %mask to <32 x i1>
+  %ret = select <32 x i1> %msk, <32 x i8> %unp, <32 x i8> %src
+  ret <32 x i8> %ret
+}
+
+define <32 x i8> @test_int_x86_avx10_vunpackb_256_maskz(<32 x i8> %a, i32 %mask) {
+; X64-LABEL: test_int_x86_avx10_vunpackb_256_maskz:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %edi, %k1 # encoding: [0xc5,0xfb,0x92,0xcf]
+; X64-NEXT:    vunpackb $2, %ymm0, %ymm0 {%k1} {z} # encoding: [0x62,0xf3,0x7c,0xa9,0x3d,0xc0,0x02]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vunpackb_256_maskz:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovd {{[0-9]+}}(%esp), %k1 # encoding: [0xc4,0xe1,0xf9,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vunpackb $2, %ymm0, %ymm0 {%k1} {z} # encoding: [0x62,0xf3,0x7c,0xa9,0x3d,0xc0,0x02]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %unp = call <32 x i8> @llvm.x86.avx10.vunpackb.256(<32 x i8> %a, i8 2)
+  %msk = bitcast i32 %mask to <32 x i1>
+  %ret = select <32 x i1> %msk, <32 x i8> %unp, <32 x i8> zeroinitializer
+  ret <32 x i8> %ret
+}
+
+define <64 x i8> @test_int_x86_avx10_vunpackb_512_mask(<64 x i8> %a, <64 x i8> %src, i64 %mask) {
+; X64-LABEL: test_int_x86_avx10_vunpackb_512_mask:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovq %rdi, %k1 # encoding: [0xc4,0xe1,0xfb,0x92,0xcf]
+; X64-NEXT:    vunpackb $3, %zmm0, %zmm1 {%k1} # encoding: [0x62,0xf3,0x7c,0x49,0x3d,0xc8,0x03]
+; X64-NEXT:    vmovaps %zmm1, %zmm0 # encoding: [0x62,0xf1,0x7c,0x48,0x28,0xc1]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vunpackb_512_mask:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovq {{[0-9]+}}(%esp), %k1 # encoding: [0xc4,0xe1,0xf8,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vunpackb $3, %zmm0, %zmm1 {%k1} # encoding: [0x62,0xf3,0x7c,0x49,0x3d,0xc8,0x03]
+; X86-NEXT:    vmovaps %zmm1, %zmm0 # encoding: [0x62,0xf1,0x7c,0x48,0x28,0xc1]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %unp = call <64 x i8> @llvm.x86.avx10.vunpackb.512(<64 x i8> %a, i8 3)
+  %msk = bitcast i64 %mask to <64 x i1>
+  %ret = select <64 x i1> %msk, <64 x i8> %unp, <64 x i8> %src
+  ret <64 x i8> %ret
+}
+
+define <64 x i8> @test_int_x86_avx10_vunpackb_512_maskz(<64 x i8> %a, i64 %mask) {
+; X64-LABEL: test_int_x86_avx10_vunpackb_512_maskz:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovq %rdi, %k1 # encoding: [0xc4,0xe1,0xfb,0x92,0xcf]
+; X64-NEXT:    vunpackb $3, %zmm0, %zmm0 {%k1} {z} # encoding: [0x62,0xf3,0x7c,0xc9,0x3d,0xc0,0x03]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vunpackb_512_maskz:
+; X86:       # %bb.0:
+; X86-NEXT:    kmovq {{[0-9]+}}(%esp), %k1 # encoding: [0xc4,0xe1,0xf8,0x90,0x4c,0x24,0x04]
+; X86-NEXT:    vunpackb $3, %zmm0, %zmm0 {%k1} {z} # encoding: [0x62,0xf3,0x7c,0xc9,0x3d,0xc0,0x03]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %unp = call <64 x i8> @llvm.x86.avx10.vunpackb.512(<64 x i8> %a, i8 3)
+  %msk = bitcast i64 %mask to <64 x i1>
+  %ret = select <64 x i1> %msk, <64 x i8> %unp, <64 x i8> zeroinitializer
+  ret <64 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vunpackb_mem_128_mask(ptr %ptr_a, <16 x i8> %src, i16 %mask) {
+; X64-LABEL: test_int_x86_avx10_vunpackb_mem_128_mask:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vunpackb $1, (%rdi), %xmm0 {%k1} # encoding: [0x62,0xf3,0x7c,0x09,0x3d,0x07,0x01]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vunpackb_mem_128_mask:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vunpackb $1, (%eax), %xmm0 {%k1} # encoding: [0x62,0xf3,0x7c,0x09,0x3d,0x00,0x01]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <16 x i8>, ptr %ptr_a
+  %unp = call <16 x i8> @llvm.x86.avx10.vunpackb.128(<16 x i8> %a, i8 1)
+  %msk = bitcast i16 %mask to <16 x i1>
+  %ret = select <16 x i1> %msk, <16 x i8> %unp, <16 x i8> %src
+  ret <16 x i8> %ret
+}
+
+define <16 x i8> @test_int_x86_avx10_vunpackb_mem_128_maskz(ptr %ptr_a, i16 %mask) {
+; X64-LABEL: test_int_x86_avx10_vunpackb_mem_128_maskz:
+; X64:       # %bb.0:
+; X64-NEXT:    kmovd %esi, %k1 # encoding: [0xc5,0xfb,0x92,0xce]
+; X64-NEXT:    vunpackb $1, (%rdi), %xmm0 {%k1} {z} # encoding: [0x62,0xf3,0x7c,0x89,0x3d,0x07,0x01]
+; X64-NEXT:    retq # encoding: [0xc3]
+;
+; X86-LABEL: test_int_x86_avx10_vunpackb_mem_128_maskz:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
+; X86-NEXT:    kmovw {{[0-9]+}}(%esp), %k1 # encoding: [0xc5,0xf8,0x90,0x4c,0x24,0x08]
+; X86-NEXT:    vunpackb $1, (%eax), %xmm0 {%k1} {z} # encoding: [0x62,0xf3,0x7c,0x89,0x3d,0x00,0x01]
+; X86-NEXT:    retl # encoding: [0xc3]
+  %a = load <16 x i8>, ptr %ptr_a
+  %unp = call <16 x i8> @llvm.x86.avx10.vunpackb.128(<16 x i8> %a, i8 1)
+  %msk = bitcast i16 %mask to <16 x i1>
+  %ret = select <16 x i1> %msk, <16 x i8> %unp, <16 x i8> zeroinitializer
+  ret <16 x i8> %ret
+}
diff --git a/llvm/test/MC/X86/avx10_v2_aux-att-32.s b/llvm/test/MC/X86/avx10_v2_aux-att-32.s
index c7f6671eea8f52..41bbd8751b682a 100644
--- a/llvm/test/MC/X86/avx10_v2_aux-att-32.s
+++ b/llvm/test/MC/X86/avx10_v2_aux-att-32.s
@@ -24,6 +24,46 @@
 // CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x39,0x07]
           vcvtps2bf8x (%edi), %xmm0
 
+// CHECK: vcvtps2bf8 %xmm1, %xmm0 {%k1}
+// CHECK: encoding: [0x62,0xf5,0x7e,0x09,0x39,0xc1]
+          vcvtps2bf8x %xmm1, %xmm0 {%k1}
+
+// CHECK: vcvtps2bf8 %xmm1, %xmm0 {%k1} {z}
+// CHECK: encoding: [0x62,0xf5,0x7e,0x89,0x39,0xc1]
+          vcvtps2bf8x %xmm1, %xmm0 {%k1} {z}
+
+// CHECK: vcvtps2bf8 (%edi){1to4}, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x18,0x39,0x07]
+          vcvtps2bf8x (%edi){1to4}, %xmm0
+
+// CHECK: vcvtps2bf8 (%edi){1to4}, %xmm0 {%k1}
+// CHECK: encoding: [0x62,0xf5,0x7e,0x19,0x39,0x07]
+          vcvtps2bf8x (%edi){1to4}, %xmm0 {%k1}
+
+// CHECK: vcvtps2bf8 (%edi){1to4}, %xmm0 {%k1} {z}
+// CHECK: encoding: [0x62,0xf5,0x7e,0x99,0x39,0x07]
+          vcvtps2bf8x (%edi){1to4}, %xmm0 {%k1} {z}
+
+// CHECK: vcvtps2bf8 %ymm1, %xmm0 {%k1}
+// CHECK: encoding: [0x62,0xf5,0x7e,0x29,0x39,0xc1]
+          vcvtps2bf8y %ymm1, %xmm0 {%k1}
+
+// CHECK: vcvtps2bf8 %ymm1, %xmm0 {%k1} {z}
+// CHECK: encoding: [0x62,0xf5,0x7e,0xa9,0x39,0xc1]
+          vcvtps2bf8y %ymm1, %xmm0 {%k1} {z}
+
+// CHECK: vcvtps2bf8 (%edi){1to8}, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x38,0x39,0x07]
+          vcvtps2bf8y (%edi){1to8}, %xmm0
+
+// CHECK: vcvtps2bf8 (%edi){1to8}, %xmm0 {%k1}
+// CHECK: encoding: [0x62,0xf5,0x7e,0x39,0x39,0x07]
+          vcvtps2bf8y (%edi){1to8}, %xmm0 {%k1}
+
+// CHECK: vcvtps2bf8 (%edi){1to8}, %xmm0 {%k1} {z}
+// CHECK: encoding: [0x62,0xf5,0x7e,0xb9,0x39,0x07]
+          vcvtps2bf8y (%edi){1to8}, %xmm0 {%k1} {z}
+
 // CHECK: vcvtps2bf8 %zmm1, %xmm0 {%k1}
 // CHECK: encoding: [0x62,0xf5,0x7e,0x49,0x39,0xc1]
           vcvtps2bf8 %zmm1, %xmm0 {%k1}
diff --git a/llvm/test/MC/X86/avx10_v2_aux-att-64.s b/llvm/test/MC/X86/avx10_v2_aux-att-64.s
index 4da5db7ddd49e1..a24f65338bf90f 100644
--- a/llvm/test/MC/X86/avx10_v2_aux-att-64.s
+++ b/llvm/test/MC/X86/avx10_v2_aux-att-64.s
@@ -24,6 +24,46 @@
 // CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x39,0x07]
           vcvtps2bf8x (%rdi), %xmm0
 
+// CHECK: vcvtps2bf8 %xmm1, %xmm0 {%k1}
+// CHECK: encoding: [0x62,0xf5,0x7e,0x09,0x39,0xc1]
+          vcvtps2bf8x %xmm1, %xmm0 {%k1}
+
+// CHECK: vcvtps2bf8 %xmm1, %xmm0 {%k1} {z}
+// CHECK: encoding: [0x62,0xf5,0x7e,0x89,0x39,0xc1]
+          vcvtps2bf8x %xmm1, %xmm0 {%k1} {z}
+
+// CHECK: vcvtps2bf8 (%rdi){1to4}, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x18,0x39,0x07]
+          vcvtps2bf8x (%rdi){1to4}, %xmm0
+
+// CHECK: vcvtps2bf8 (%rdi){1to4}, %xmm0 {%k1}
+// CHECK: encoding: [0x62,0xf5,0x7e,0x19,0x39,0x07]
+          vcvtps2bf8x (%rdi){1to4}, %xmm0 {%k1}
+
+// CHECK: vcvtps2bf8 (%rdi){1to4}, %xmm0 {%k1} {z}
+// CHECK: encoding: [0x62,0xf5,0x7e,0x99,0x39,0x07]
+          vcvtps2bf8x (%rdi){1to4}, %xmm0 {%k1} {z}
+
+// CHECK: vcvtps2bf8 %ymm1, %xmm0 {%k1}
+// CHECK: encoding: [0x62,0xf5,0x7e,0x29,0x39,0xc1]
+          vcvtps2bf8y %ymm1, %xmm0 {%k1}
+
+// CHECK: vcvtps2bf8 %ymm1, %xmm0 {%k1} {z}
+// CHECK: encoding: [0x62,0xf5,0x7e,0xa9,0x39,0xc1]
+          vcvtps2bf8y %ymm1, %xmm0 {%k1} {z}
+
+// CHECK: vcvtps2bf8 (%rdi){1to8}, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x38,0x39,0x07]
+          vcvtps2bf8y (%rdi){1to8}, %xmm0
+
+// CHECK: vcvtps2bf8 (%rdi){1to8}, %xmm0 {%k1}
+// CHECK: encoding: [0x62,0xf5,0x7e,0x39,0x39,0x07]
+          vcvtps2bf8y (%rdi){1to8}, %xmm0 {%k1}
+
+// CHECK: vcvtps2bf8 (%rdi){1to8}, %xmm0 {%k1} {z}
+// CHECK: encoding: [0x62,0xf5,0x7e,0xb9,0x39,0x07]
+          vcvtps2bf8y (%rdi){1to8}, %xmm0 {%k1} {z}
+
 // CHECK: vcvtps2bf8 %zmm1, %xmm0 {%k1}
 // CHECK: encoding: [0x62,0xf5,0x7e,0x49,0x39,0xc1]
           vcvtps2bf8 %zmm1, %xmm0 {%k1}
diff --git a/llvm/test/TableGen/x86-fold-tables.inc b/llvm/test/TableGen/x86-fold-tables.inc
index a7d5258b501887..1f94fc7e28290c 100644
--- a/llvm/test/TableGen/x86-fold-tables.inc
+++ b/llvm/test/TableGen/x86-fold-tables.inc
@@ -328,9 +328,6 @@ static const X86FoldTableEntry Table2Addr[] = {
   {X86::SUB8ri_NF, X86::SUB8mi_NF, TB_NO_REVERSE},
   {X86::SUB8rr, X86::SUB8mr, TB_NO_REVERSE},
   {X86::SUB8rr_NF, X86::SUB8mr_NF, TB_NO_REVERSE},
-  {X86::VPMOVSSDBZ128rrk, X86::VPMOVSSDBZ128mrk, TB_NO_REVERSE},
-  {X86::VPMOVSSDBZ256rrk, X86::VPMOVSSDBZ256mrk, TB_NO_REVERSE},
-  {X86::VPMOVSSDBZrrk, X86::VPMOVSSDBZmrk, TB_NO_REVERSE},
   {X86::XOR16ri, X86::XOR16mi, TB_NO_REVERSE},
   {X86::XOR16ri8, X86::XOR16mi8, TB_NO_REVERSE},
   {X86::XOR16ri8_NF, X86::XOR16mi8_NF, TB_NO_REVERSE},
diff --git a/llvm/utils/TableGen/X86ManualFoldTables.def b/llvm/utils/TableGen/X86ManualFoldTables.def
index 693b3de69ae468..f39c5c4b427c4e 100644
--- a/llvm/utils/TableGen/X86ManualFoldTables.def
+++ b/llvm/utils/TableGen/X86ManualFoldTables.def
@@ -103,6 +103,9 @@ NOFOLD(VPMOVQWZrrk)
 NOFOLD(VPMOVSDBZ128rrk)
 NOFOLD(VPMOVSDBZ256rrk)
 NOFOLD(VPMOVSDBZrrk)
+NOFOLD(VPMOVSSDBZ128rrk)
+NOFOLD(VPMOVSSDBZ256rrk)
+NOFOLD(VPMOVSSDBZrrk)
 NOFOLD(VPMOVSDWZ128rrk)
 NOFOLD(VPMOVSDWZ256rrk)
 NOFOLD(VPMOVSDWZrrk)

>From ebd69541672c95c135e3e4bc9d370109e4422e7f Mon Sep 17 00:00:00 2001
From: Ganesh Gopalasubramanian <Ganesh.Gopalasubramanian at amd.com>
Date: Wed, 16 Sep 2026 01:18:13 +0530
Subject: [PATCH 15/16] [X86][AVX10_V2_AUX] Address review comments

---
 clang/include/clang/Sema/SemaX86.h            |  2 +-
 clang/lib/Sema/SemaX86.cpp                    | 34 +++++-------
 compiler-rt/lib/builtins/cpu_model/x86.c      |  4 +-
 llvm/lib/Target/X86/X86InstrAVX10_V2_AUX.td   | 35 +++++++++----
 .../X86/avx10_v2aux-cvt-ps-fp8-intrinsics.ll  | 24 ++++-----
 llvm/test/MC/X86/avx10_v2_aux-att-32.s        | 52 ++++++++-----------
 llvm/test/MC/X86/avx10_v2_aux-att-64.s        | 52 ++++++++-----------
 7 files changed, 98 insertions(+), 105 deletions(-)

diff --git a/clang/include/clang/Sema/SemaX86.h b/clang/include/clang/Sema/SemaX86.h
index 09a7831032abd8..00003142a50f39 100644
--- a/clang/include/clang/Sema/SemaX86.h
+++ b/clang/include/clang/Sema/SemaX86.h
@@ -26,7 +26,7 @@ class SemaX86 : public SemaBase {
   SemaX86(Sema &S);
 
   bool CheckBuiltinRoundingOrSAE(unsigned BuiltinID, CallExpr *TheCall);
-  bool CheckBuiltinVUnpackBImm(unsigned BuiltinID, CallExpr *TheCall);
+  bool CheckBuiltinVUnpackBImm(CallExpr *TheCall);
   bool CheckBuiltinGatherScatterScale(unsigned BuiltinID, CallExpr *TheCall);
   bool CheckBuiltinTileArguments(unsigned BuiltinID, CallExpr *TheCall);
   bool CheckBuiltinTileArgumentsRange(CallExpr *TheCall, ArrayRef<int> ArgNums);
diff --git a/clang/lib/Sema/SemaX86.cpp b/clang/lib/Sema/SemaX86.cpp
index 24a6a2acc64f47..8681972cb82c0d 100644
--- a/clang/lib/Sema/SemaX86.cpp
+++ b/clang/lib/Sema/SemaX86.cpp
@@ -344,17 +344,16 @@ bool SemaX86::CheckBuiltinRoundingOrSAE(unsigned BuiltinID, CallExpr *TheCall) {
 }
 
 // Check if the VUNPACKB immediate encoding is legal.
-bool SemaX86::CheckBuiltinVUnpackBImm(unsigned BuiltinID, CallExpr *TheCall) {
-  unsigned ArgNum = 0;
-  switch (BuiltinID) {
-  default:
-    return false;
-  case X86::BI__builtin_ia32_vunpackb128:
-  case X86::BI__builtin_ia32_vunpackb256:
-  case X86::BI__builtin_ia32_vunpackb512:
-    ArgNum = 1;
-    break;
-  }
+bool SemaX86::CheckBuiltinVUnpackBImm(CallExpr *TheCall) {
+  const unsigned ArgNum = 1;
+
+  // Note that we don't force a hard error on the range check here, allowing
+  // template-generated or macro-generated dead code to potentially have out-of-
+  // range values. These need to code generate, but don't need to necessarily
+  // make any sense. We use a warning that defaults to an error.
+  if (SemaRef.BuiltinConstantArgRange(TheCall, ArgNum, 0, 63,
+                                      /*RangeIsError*/ false))
+    return true;
 
   llvm::APSInt Result;
 
@@ -766,10 +765,7 @@ bool SemaX86::CheckBuiltinFunctionCall(const TargetInfo &TI, unsigned BuiltinID,
   case X86::BI__builtin_ia32_vunpackb128:
   case X86::BI__builtin_ia32_vunpackb256:
   case X86::BI__builtin_ia32_vunpackb512:
-    i = 1;
-    l = 0;
-    u = 63;
-    break;
+    return CheckBuiltinVUnpackBImm(TheCall);
   case X86::BI__builtin_ia32_cmpps:
   case X86::BI__builtin_ia32_cmpss:
   case X86::BI__builtin_ia32_cmppd:
@@ -1005,12 +1001,8 @@ bool SemaX86::CheckBuiltinFunctionCall(const TargetInfo &TI, unsigned BuiltinID,
   // template-generated or macro-generated dead code to potentially have out-of-
   // range values. These need to code generate, but don't need to necessarily
   // make any sense. We use a warning that defaults to an error.
-  if (SemaRef.BuiltinConstantArgRange(TheCall, i, l, u,
-                                      /*RangeIsError*/ false))
-    return true;
-
-  // If the intrinsic has a VUNPACKB immediate, make sure the encoding is valid.
-  return CheckBuiltinVUnpackBImm(BuiltinID, TheCall);
+  return SemaRef.BuiltinConstantArgRange(TheCall, i, l, u,
+                                         /*RangeIsError*/ false);
 }
 
 void SemaX86::handleAnyInterruptAttr(Decl *D, const ParsedAttr &AL) {
diff --git a/compiler-rt/lib/builtins/cpu_model/x86.c b/compiler-rt/lib/builtins/cpu_model/x86.c
index e88bbe531570b2..57251d8d98cb71 100644
--- a/compiler-rt/lib/builtins/cpu_model/x86.c
+++ b/compiler-rt/lib/builtins/cpu_model/x86.c
@@ -242,7 +242,7 @@ enum ProcessorFeatures {
   FEATURE_MOVRS,
   FEATURE_AMX_MOVRS,
   FEATURE_AVX512BMM,
-  FEATURE_AVX10_V2_AUX = 124,
+  FEATURE_AVX10_V2_AUX,
   CPU_FEATURE_MAX
 };
 
@@ -1133,7 +1133,7 @@ static void getAvailableFeatures(unsigned ECX, unsigned EDX, unsigned MaxLeaf,
     setFeature(FEATURE_USERMSR);
   if (HasLeaf7Subleaf1 && ((EDX >> 21) & 1) && HasAPXSave)
     setFeature(FEATURE_APXF);
-  bool HasAVX10 = HasLeaf7Subleaf1 && ((EDX >> 19) & 1) && HasAVX512Save;
+  bool HasAVX10 = HasLeaf7Subleaf1 && ((EDX >> 19) & 1);
 
   unsigned MaxLevel = 0;
   getX86CpuIDAndInfo(0, &MaxLevel, &EBX, &ECX, &EDX);
diff --git a/llvm/lib/Target/X86/X86InstrAVX10_V2_AUX.td b/llvm/lib/Target/X86/X86InstrAVX10_V2_AUX.td
index 30d5083c7b3393..411a954e0935f6 100644
--- a/llvm/lib/Target/X86/X86InstrAVX10_V2_AUX.td
+++ b/llvm/lib/Target/X86/X86InstrAVX10_V2_AUX.td
@@ -23,7 +23,8 @@ multiclass avx10_v2aux_cvt_trunc_ps2i8<bits<8> opc, string OpcodeStr,
   let ExeDomain = SSEPackedSingle in {
     let Uses = []<Register>, mayRaiseFPException = 0 in {
       defm Z : avx512_vcvt_fp<opc, OpcodeStr, v16i8x_info, v16f32_info,
-                              OpNode, OpNode, WriteCvtPH2PSZ>, EVEX_V512;
+                              OpNode, OpNode, WriteCvtPH2PSZ,
+                              v16f32_info.BroadcastStr, "{z}">, EVEX_V512;
       // Z256/Z128: use null_frag because element count mismatch between
       // dest (v16i8) and source (v8f32/v4f32) prevents avx512_vcvt_fp from
       // generating correct masked patterns. Explicit Pat patterns below.
@@ -40,8 +41,8 @@ multiclass avx10_v2aux_cvt_trunc_ps2i8<bits<8> opc, string OpcodeStr,
     }
   }
 
-  // InstAliases for x/y suffixes. Priority 0 so they are accepted but not
-  // printed by default.
+  // InstAliases for x/y/z suffixes (dest is always xmm). Priority 0 so they
+  // are accepted but not printed by default. Same precedent as vcvtpd2ph.
   def : InstAlias<OpcodeStr#"x\t{$src, $dst|$dst, $src}",
                   (!cast<Instruction>(NAME # "Z128rr") VR128X:$dst,
                    VR128X:$src), 0, "att">;
@@ -52,9 +53,6 @@ multiclass avx10_v2aux_cvt_trunc_ps2i8<bits<8> opc, string OpcodeStr,
                   "$dst {${mask}} {z}, $src}",
                   (!cast<Instruction>(NAME # "Z128rrkz") VR128X:$dst,
                    VK4WM:$mask, VR128X:$src), 0, "att">;
-  def : InstAlias<OpcodeStr#"x\t{$src, $dst|$dst, $src}",
-                  (!cast<Instruction>(NAME # "Z128rm") VR128X:$dst,
-                   f128mem:$src), 0, "intel">;
   def : InstAlias<OpcodeStr#"x\t{${src}{1to4}, $dst|$dst, ${src}{1to4}}",
                   (!cast<Instruction>(NAME # "Z128rmb") VR128X:$dst,
                    f32mem:$src), 0, "att">;
@@ -77,9 +75,6 @@ multiclass avx10_v2aux_cvt_trunc_ps2i8<bits<8> opc, string OpcodeStr,
                   "$dst {${mask}} {z}, $src}",
                   (!cast<Instruction>(NAME # "Z256rrkz") VR128X:$dst,
                    VK8WM:$mask, VR256X:$src), 0, "att">;
-  def : InstAlias<OpcodeStr#"y\t{$src, $dst|$dst, $src}",
-                  (!cast<Instruction>(NAME # "Z256rm") VR128X:$dst,
-                   f256mem:$src), 0, "intel">;
   def : InstAlias<OpcodeStr#"y\t{${src}{1to8}, $dst|$dst, ${src}{1to8}}",
                   (!cast<Instruction>(NAME # "Z256rmb") VR128X:$dst,
                    f32mem:$src), 0, "att">;
@@ -92,6 +87,28 @@ multiclass avx10_v2aux_cvt_trunc_ps2i8<bits<8> opc, string OpcodeStr,
                   (!cast<Instruction>(NAME # "Z256rmbkz") VR128X:$dst,
                    VK8WM:$mask, f32mem:$src), 0, "att">;
 
+  def : InstAlias<OpcodeStr#"z\t{$src, $dst|$dst, $src}",
+                  (!cast<Instruction>(NAME # "Zrr") VR128X:$dst,
+                   VR512:$src), 0, "att">;
+  def : InstAlias<OpcodeStr#"z\t{$src, $dst {${mask}}|$dst {${mask}}, $src}",
+                  (!cast<Instruction>(NAME # "Zrrk") VR128X:$dst,
+                   VK16WM:$mask, VR512:$src), 0, "att">;
+  def : InstAlias<OpcodeStr#"z\t{$src, $dst {${mask}} {z}|"
+                  "$dst {${mask}} {z}, $src}",
+                  (!cast<Instruction>(NAME # "Zrrkz") VR128X:$dst,
+                   VK16WM:$mask, VR512:$src), 0, "att">;
+  def : InstAlias<OpcodeStr#"z\t{${src}{1to16}, $dst|$dst, ${src}{1to16}}",
+                  (!cast<Instruction>(NAME # "Zrmb") VR128X:$dst,
+                   f32mem:$src), 0, "att">;
+  def : InstAlias<OpcodeStr#"z\t{${src}{1to16}, $dst {${mask}}|"
+                  "$dst {${mask}}, ${src}{1to16}}",
+                  (!cast<Instruction>(NAME # "Zrmbk") VR128X:$dst,
+                   VK16WM:$mask, f32mem:$src), 0, "att">;
+  def : InstAlias<OpcodeStr#"z\t{${src}{1to16}, $dst {${mask}} {z}|"
+                  "$dst {${mask}} {z}, ${src}{1to16}}",
+                  (!cast<Instruction>(NAME # "Zrmbkz") VR128X:$dst,
+                   VK16WM:$mask, f32mem:$src), 0, "att">;
+
   // Explicit patterns for Z256 (8 source elements, VK8WM mask)
   // Unmasked
   def : Pat<(v16i8 (OpNode (v8f32 VR256X:$src))),
diff --git a/llvm/test/CodeGen/X86/avx10_v2aux-cvt-ps-fp8-intrinsics.ll b/llvm/test/CodeGen/X86/avx10_v2aux-cvt-ps-fp8-intrinsics.ll
index 7ac6a0f36e9c75..bcba1280b159d9 100644
--- a/llvm/test/CodeGen/X86/avx10_v2aux-cvt-ps-fp8-intrinsics.ll
+++ b/llvm/test/CodeGen/X86/avx10_v2aux-cvt-ps-fp8-intrinsics.ll
@@ -70,13 +70,13 @@ define <16 x i8> @test_int_x86_avx10_vcvtps2bf8256_mem(ptr %ptr_a) {
 define <16 x i8> @test_int_x86_avx10_vcvtps2bf8512_mem(ptr %ptr_a) {
 ; X64-LABEL: test_int_x86_avx10_vcvtps2bf8512_mem:
 ; X64:       # %bb.0:
-; X64-NEXT:    vcvtps2bf8 (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x39,0x07]
+; X64-NEXT:    vcvtps2bf8z (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x39,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
 ; X86-LABEL: test_int_x86_avx10_vcvtps2bf8512_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vcvtps2bf8 (%eax), %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x39,0x00]
+; X86-NEXT:    vcvtps2bf8z (%eax), %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x39,0x00]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %a = load <16 x float>, ptr %ptr_a
   %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8512(<16 x float> %a)
@@ -151,13 +151,13 @@ define <16 x i8> @test_int_x86_avx10_vcvtps2bf8s256_mem(ptr %ptr_a) {
 define <16 x i8> @test_int_x86_avx10_vcvtps2bf8s512_mem(ptr %ptr_a) {
 ; X64-LABEL: test_int_x86_avx10_vcvtps2bf8s512_mem:
 ; X64:       # %bb.0:
-; X64-NEXT:    vcvtps2bf8s (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x3b,0x07]
+; X64-NEXT:    vcvtps2bf8sz (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x3b,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
 ; X86-LABEL: test_int_x86_avx10_vcvtps2bf8s512_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vcvtps2bf8s (%eax), %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x3b,0x00]
+; X86-NEXT:    vcvtps2bf8sz (%eax), %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x3b,0x00]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %a = load <16 x float>, ptr %ptr_a
   %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8s512(<16 x float> %a)
@@ -232,13 +232,13 @@ define <16 x i8> @test_int_x86_avx10_vcvtps2hf8256_mem(ptr %ptr_a) {
 define <16 x i8> @test_int_x86_avx10_vcvtps2hf8512_mem(ptr %ptr_a) {
 ; X64-LABEL: test_int_x86_avx10_vcvtps2hf8512_mem:
 ; X64:       # %bb.0:
-; X64-NEXT:    vcvtps2hf8 (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x38,0x07]
+; X64-NEXT:    vcvtps2hf8z (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x38,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
 ; X86-LABEL: test_int_x86_avx10_vcvtps2hf8512_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vcvtps2hf8 (%eax), %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x38,0x00]
+; X86-NEXT:    vcvtps2hf8z (%eax), %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x38,0x00]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %a = load <16 x float>, ptr %ptr_a
   %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8512(<16 x float> %a)
@@ -313,13 +313,13 @@ define <16 x i8> @test_int_x86_avx10_vcvtps2hf8s256_mem(ptr %ptr_a) {
 define <16 x i8> @test_int_x86_avx10_vcvtps2hf8s512_mem(ptr %ptr_a) {
 ; X64-LABEL: test_int_x86_avx10_vcvtps2hf8s512_mem:
 ; X64:       # %bb.0:
-; X64-NEXT:    vcvtps2hf8s (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x3a,0x07]
+; X64-NEXT:    vcvtps2hf8sz (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x3a,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
 ; X86-LABEL: test_int_x86_avx10_vcvtps2hf8s512_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vcvtps2hf8s (%eax), %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x3a,0x00]
+; X86-NEXT:    vcvtps2hf8sz (%eax), %xmm0 # encoding: [0x62,0xf5,0x7e,0x48,0x3a,0x00]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %a = load <16 x float>, ptr %ptr_a
   %ret = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8s512(<16 x float> %a)
@@ -394,13 +394,13 @@ define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8256_mem(ptr %ptr_a) {
 define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8512_mem(ptr %ptr_a) {
 ; X64-LABEL: test_int_x86_avx10_vcvtrops2hf8512_mem:
 ; X64:       # %bb.0:
-; X64-NEXT:    vcvtrops2hf8 (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7d,0x48,0x38,0x07]
+; X64-NEXT:    vcvtrops2hf8z (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7d,0x48,0x38,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
 ; X86-LABEL: test_int_x86_avx10_vcvtrops2hf8512_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vcvtrops2hf8 (%eax), %xmm0 # encoding: [0x62,0xf5,0x7d,0x48,0x38,0x00]
+; X86-NEXT:    vcvtrops2hf8z (%eax), %xmm0 # encoding: [0x62,0xf5,0x7d,0x48,0x38,0x00]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %a = load <16 x float>, ptr %ptr_a
   %ret = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8512(<16 x float> %a)
@@ -475,13 +475,13 @@ define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8s256_mem(ptr %ptr_a) {
 define <16 x i8> @test_int_x86_avx10_vcvtrops2hf8s512_mem(ptr %ptr_a) {
 ; X64-LABEL: test_int_x86_avx10_vcvtrops2hf8s512_mem:
 ; X64:       # %bb.0:
-; X64-NEXT:    vcvtrops2hf8s (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7d,0x48,0x3a,0x07]
+; X64-NEXT:    vcvtrops2hf8sz (%rdi), %xmm0 # encoding: [0x62,0xf5,0x7d,0x48,0x3a,0x07]
 ; X64-NEXT:    retq # encoding: [0xc3]
 ;
 ; X86-LABEL: test_int_x86_avx10_vcvtrops2hf8s512_mem:
 ; X86:       # %bb.0:
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax # encoding: [0x8b,0x44,0x24,0x04]
-; X86-NEXT:    vcvtrops2hf8s (%eax), %xmm0 # encoding: [0x62,0xf5,0x7d,0x48,0x3a,0x00]
+; X86-NEXT:    vcvtrops2hf8sz (%eax), %xmm0 # encoding: [0x62,0xf5,0x7d,0x48,0x3a,0x00]
 ; X86-NEXT:    retl # encoding: [0xc3]
   %a = load <16 x float>, ptr %ptr_a
   %ret = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s512(<16 x float> %a)
diff --git a/llvm/test/MC/X86/avx10_v2_aux-att-32.s b/llvm/test/MC/X86/avx10_v2_aux-att-32.s
index 41bbd8751b682a..ba16889de21591 100644
--- a/llvm/test/MC/X86/avx10_v2_aux-att-32.s
+++ b/llvm/test/MC/X86/avx10_v2_aux-att-32.s
@@ -12,9 +12,9 @@
 // CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x39,0xc1]
           vcvtps2bf8 %xmm1, %xmm0
 
-// CHECK: vcvtps2bf8 (%edi), %xmm0
+// CHECK: vcvtps2bf8z (%edi), %xmm0
 // CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x39,0x07]
-          vcvtps2bf8 (%edi), %xmm0
+          vcvtps2bf8z (%edi), %xmm0
 
 // CHECK: vcvtps2bf8y (%edi), %xmm0
 // CHECK: encoding: [0x62,0xf5,0x7e,0x28,0x39,0x07]
@@ -26,43 +26,43 @@
 
 // CHECK: vcvtps2bf8 %xmm1, %xmm0 {%k1}
 // CHECK: encoding: [0x62,0xf5,0x7e,0x09,0x39,0xc1]
-          vcvtps2bf8x %xmm1, %xmm0 {%k1}
+          vcvtps2bf8 %xmm1, %xmm0 {%k1}
 
 // CHECK: vcvtps2bf8 %xmm1, %xmm0 {%k1} {z}
 // CHECK: encoding: [0x62,0xf5,0x7e,0x89,0x39,0xc1]
-          vcvtps2bf8x %xmm1, %xmm0 {%k1} {z}
+          vcvtps2bf8 %xmm1, %xmm0 {%k1} {z}
 
 // CHECK: vcvtps2bf8 (%edi){1to4}, %xmm0
 // CHECK: encoding: [0x62,0xf5,0x7e,0x18,0x39,0x07]
-          vcvtps2bf8x (%edi){1to4}, %xmm0
+          vcvtps2bf8 (%edi){1to4}, %xmm0
 
 // CHECK: vcvtps2bf8 (%edi){1to4}, %xmm0 {%k1}
 // CHECK: encoding: [0x62,0xf5,0x7e,0x19,0x39,0x07]
-          vcvtps2bf8x (%edi){1to4}, %xmm0 {%k1}
+          vcvtps2bf8 (%edi){1to4}, %xmm0 {%k1}
 
 // CHECK: vcvtps2bf8 (%edi){1to4}, %xmm0 {%k1} {z}
 // CHECK: encoding: [0x62,0xf5,0x7e,0x99,0x39,0x07]
-          vcvtps2bf8x (%edi){1to4}, %xmm0 {%k1} {z}
+          vcvtps2bf8 (%edi){1to4}, %xmm0 {%k1} {z}
 
 // CHECK: vcvtps2bf8 %ymm1, %xmm0 {%k1}
 // CHECK: encoding: [0x62,0xf5,0x7e,0x29,0x39,0xc1]
-          vcvtps2bf8y %ymm1, %xmm0 {%k1}
+          vcvtps2bf8 %ymm1, %xmm0 {%k1}
 
 // CHECK: vcvtps2bf8 %ymm1, %xmm0 {%k1} {z}
 // CHECK: encoding: [0x62,0xf5,0x7e,0xa9,0x39,0xc1]
-          vcvtps2bf8y %ymm1, %xmm0 {%k1} {z}
+          vcvtps2bf8 %ymm1, %xmm0 {%k1} {z}
 
 // CHECK: vcvtps2bf8 (%edi){1to8}, %xmm0
 // CHECK: encoding: [0x62,0xf5,0x7e,0x38,0x39,0x07]
-          vcvtps2bf8y (%edi){1to8}, %xmm0
+          vcvtps2bf8 (%edi){1to8}, %xmm0
 
 // CHECK: vcvtps2bf8 (%edi){1to8}, %xmm0 {%k1}
 // CHECK: encoding: [0x62,0xf5,0x7e,0x39,0x39,0x07]
-          vcvtps2bf8y (%edi){1to8}, %xmm0 {%k1}
+          vcvtps2bf8 (%edi){1to8}, %xmm0 {%k1}
 
 // CHECK: vcvtps2bf8 (%edi){1to8}, %xmm0 {%k1} {z}
 // CHECK: encoding: [0x62,0xf5,0x7e,0xb9,0x39,0x07]
-          vcvtps2bf8y (%edi){1to8}, %xmm0 {%k1} {z}
+          vcvtps2bf8 (%edi){1to8}, %xmm0 {%k1} {z}
 
 // CHECK: vcvtps2bf8 %zmm1, %xmm0 {%k1}
 // CHECK: encoding: [0x62,0xf5,0x7e,0x49,0x39,0xc1]
@@ -76,14 +76,6 @@
 // CHECK: encoding: [0x62,0xf5,0x7e,0x58,0x39,0x07]
           vcvtps2bf8 (%edi){1to16}, %xmm0
 
-// CHECK: vcvtps2bf8 (%edi){1to8}, %xmm0
-// CHECK: encoding: [0x62,0xf5,0x7e,0x38,0x39,0x07]
-          vcvtps2bf8 (%edi){1to8}, %xmm0
-
-// CHECK: vcvtps2bf8 (%edi){1to4}, %xmm0
-// CHECK: encoding: [0x62,0xf5,0x7e,0x18,0x39,0x07]
-          vcvtps2bf8 (%edi){1to4}, %xmm0
-
 // CHECK: vcvtps2bf8s %zmm1, %xmm0
 // CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x3b,0xc1]
           vcvtps2bf8s %zmm1, %xmm0
@@ -96,9 +88,9 @@
 // CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x3b,0xc1]
           vcvtps2bf8s %xmm1, %xmm0
 
-// CHECK: vcvtps2bf8s (%edi), %xmm0
+// CHECK: vcvtps2bf8sz (%edi), %xmm0
 // CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x3b,0x07]
-          vcvtps2bf8s (%edi), %xmm0
+          vcvtps2bf8sz (%edi), %xmm0
 
 // CHECK: vcvtps2bf8sy (%edi), %xmm0
 // CHECK: encoding: [0x62,0xf5,0x7e,0x28,0x3b,0x07]
@@ -140,9 +132,9 @@
 // CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x38,0xc1]
           vcvtps2hf8 %xmm1, %xmm0
 
-// CHECK: vcvtps2hf8 (%edi), %xmm0
+// CHECK: vcvtps2hf8z (%edi), %xmm0
 // CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x38,0x07]
-          vcvtps2hf8 (%edi), %xmm0
+          vcvtps2hf8z (%edi), %xmm0
 
 // CHECK: vcvtps2hf8y (%edi), %xmm0
 // CHECK: encoding: [0x62,0xf5,0x7e,0x28,0x38,0x07]
@@ -184,9 +176,9 @@
 // CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x3a,0xc1]
           vcvtps2hf8s %xmm1, %xmm0
 
-// CHECK: vcvtps2hf8s (%edi), %xmm0
+// CHECK: vcvtps2hf8sz (%edi), %xmm0
 // CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x3a,0x07]
-          vcvtps2hf8s (%edi), %xmm0
+          vcvtps2hf8sz (%edi), %xmm0
 
 // CHECK: vcvtps2hf8sy (%edi), %xmm0
 // CHECK: encoding: [0x62,0xf5,0x7e,0x28,0x3a,0x07]
@@ -228,9 +220,9 @@
 // CHECK: encoding: [0x62,0xf5,0x7d,0x08,0x38,0xc1]
           vcvtrops2hf8 %xmm1, %xmm0
 
-// CHECK: vcvtrops2hf8 (%edi), %xmm0
+// CHECK: vcvtrops2hf8z (%edi), %xmm0
 // CHECK: encoding: [0x62,0xf5,0x7d,0x48,0x38,0x07]
-          vcvtrops2hf8 (%edi), %xmm0
+          vcvtrops2hf8z (%edi), %xmm0
 
 // CHECK: vcvtrops2hf8y (%edi), %xmm0
 // CHECK: encoding: [0x62,0xf5,0x7d,0x28,0x38,0x07]
@@ -272,9 +264,9 @@
 // CHECK: encoding: [0x62,0xf5,0x7d,0x08,0x3a,0xc1]
           vcvtrops2hf8s %xmm1, %xmm0
 
-// CHECK: vcvtrops2hf8s (%edi), %xmm0
+// CHECK: vcvtrops2hf8sz (%edi), %xmm0
 // CHECK: encoding: [0x62,0xf5,0x7d,0x48,0x3a,0x07]
-          vcvtrops2hf8s (%edi), %xmm0
+          vcvtrops2hf8sz (%edi), %xmm0
 
 // CHECK: vcvtrops2hf8sy (%edi), %xmm0
 // CHECK: encoding: [0x62,0xf5,0x7d,0x28,0x3a,0x07]
diff --git a/llvm/test/MC/X86/avx10_v2_aux-att-64.s b/llvm/test/MC/X86/avx10_v2_aux-att-64.s
index a24f65338bf90f..956f74912938fa 100644
--- a/llvm/test/MC/X86/avx10_v2_aux-att-64.s
+++ b/llvm/test/MC/X86/avx10_v2_aux-att-64.s
@@ -12,9 +12,9 @@
 // CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x39,0xc1]
           vcvtps2bf8 %xmm1, %xmm0
 
-// CHECK: vcvtps2bf8 (%rdi), %xmm0
+// CHECK: vcvtps2bf8z (%rdi), %xmm0
 // CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x39,0x07]
-          vcvtps2bf8 (%rdi), %xmm0
+          vcvtps2bf8z (%rdi), %xmm0
 
 // CHECK: vcvtps2bf8y (%rdi), %xmm0
 // CHECK: encoding: [0x62,0xf5,0x7e,0x28,0x39,0x07]
@@ -26,43 +26,43 @@
 
 // CHECK: vcvtps2bf8 %xmm1, %xmm0 {%k1}
 // CHECK: encoding: [0x62,0xf5,0x7e,0x09,0x39,0xc1]
-          vcvtps2bf8x %xmm1, %xmm0 {%k1}
+          vcvtps2bf8 %xmm1, %xmm0 {%k1}
 
 // CHECK: vcvtps2bf8 %xmm1, %xmm0 {%k1} {z}
 // CHECK: encoding: [0x62,0xf5,0x7e,0x89,0x39,0xc1]
-          vcvtps2bf8x %xmm1, %xmm0 {%k1} {z}
+          vcvtps2bf8 %xmm1, %xmm0 {%k1} {z}
 
 // CHECK: vcvtps2bf8 (%rdi){1to4}, %xmm0
 // CHECK: encoding: [0x62,0xf5,0x7e,0x18,0x39,0x07]
-          vcvtps2bf8x (%rdi){1to4}, %xmm0
+          vcvtps2bf8 (%rdi){1to4}, %xmm0
 
 // CHECK: vcvtps2bf8 (%rdi){1to4}, %xmm0 {%k1}
 // CHECK: encoding: [0x62,0xf5,0x7e,0x19,0x39,0x07]
-          vcvtps2bf8x (%rdi){1to4}, %xmm0 {%k1}
+          vcvtps2bf8 (%rdi){1to4}, %xmm0 {%k1}
 
 // CHECK: vcvtps2bf8 (%rdi){1to4}, %xmm0 {%k1} {z}
 // CHECK: encoding: [0x62,0xf5,0x7e,0x99,0x39,0x07]
-          vcvtps2bf8x (%rdi){1to4}, %xmm0 {%k1} {z}
+          vcvtps2bf8 (%rdi){1to4}, %xmm0 {%k1} {z}
 
 // CHECK: vcvtps2bf8 %ymm1, %xmm0 {%k1}
 // CHECK: encoding: [0x62,0xf5,0x7e,0x29,0x39,0xc1]
-          vcvtps2bf8y %ymm1, %xmm0 {%k1}
+          vcvtps2bf8 %ymm1, %xmm0 {%k1}
 
 // CHECK: vcvtps2bf8 %ymm1, %xmm0 {%k1} {z}
 // CHECK: encoding: [0x62,0xf5,0x7e,0xa9,0x39,0xc1]
-          vcvtps2bf8y %ymm1, %xmm0 {%k1} {z}
+          vcvtps2bf8 %ymm1, %xmm0 {%k1} {z}
 
 // CHECK: vcvtps2bf8 (%rdi){1to8}, %xmm0
 // CHECK: encoding: [0x62,0xf5,0x7e,0x38,0x39,0x07]
-          vcvtps2bf8y (%rdi){1to8}, %xmm0
+          vcvtps2bf8 (%rdi){1to8}, %xmm0
 
 // CHECK: vcvtps2bf8 (%rdi){1to8}, %xmm0 {%k1}
 // CHECK: encoding: [0x62,0xf5,0x7e,0x39,0x39,0x07]
-          vcvtps2bf8y (%rdi){1to8}, %xmm0 {%k1}
+          vcvtps2bf8 (%rdi){1to8}, %xmm0 {%k1}
 
 // CHECK: vcvtps2bf8 (%rdi){1to8}, %xmm0 {%k1} {z}
 // CHECK: encoding: [0x62,0xf5,0x7e,0xb9,0x39,0x07]
-          vcvtps2bf8y (%rdi){1to8}, %xmm0 {%k1} {z}
+          vcvtps2bf8 (%rdi){1to8}, %xmm0 {%k1} {z}
 
 // CHECK: vcvtps2bf8 %zmm1, %xmm0 {%k1}
 // CHECK: encoding: [0x62,0xf5,0x7e,0x49,0x39,0xc1]
@@ -76,14 +76,6 @@
 // CHECK: encoding: [0x62,0xf5,0x7e,0x58,0x39,0x07]
           vcvtps2bf8 (%rdi){1to16}, %xmm0
 
-// CHECK: vcvtps2bf8 (%rdi){1to8}, %xmm0
-// CHECK: encoding: [0x62,0xf5,0x7e,0x38,0x39,0x07]
-          vcvtps2bf8 (%rdi){1to8}, %xmm0
-
-// CHECK: vcvtps2bf8 (%rdi){1to4}, %xmm0
-// CHECK: encoding: [0x62,0xf5,0x7e,0x18,0x39,0x07]
-          vcvtps2bf8 (%rdi){1to4}, %xmm0
-
 // CHECK: vcvtps2bf8s %zmm1, %xmm0
 // CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x3b,0xc1]
           vcvtps2bf8s %zmm1, %xmm0
@@ -96,9 +88,9 @@
 // CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x3b,0xc1]
           vcvtps2bf8s %xmm1, %xmm0
 
-// CHECK: vcvtps2bf8s (%rdi), %xmm0
+// CHECK: vcvtps2bf8sz (%rdi), %xmm0
 // CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x3b,0x07]
-          vcvtps2bf8s (%rdi), %xmm0
+          vcvtps2bf8sz (%rdi), %xmm0
 
 // CHECK: vcvtps2bf8sy (%rdi), %xmm0
 // CHECK: encoding: [0x62,0xf5,0x7e,0x28,0x3b,0x07]
@@ -140,9 +132,9 @@
 // CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x38,0xc1]
           vcvtps2hf8 %xmm1, %xmm0
 
-// CHECK: vcvtps2hf8 (%rdi), %xmm0
+// CHECK: vcvtps2hf8z (%rdi), %xmm0
 // CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x38,0x07]
-          vcvtps2hf8 (%rdi), %xmm0
+          vcvtps2hf8z (%rdi), %xmm0
 
 // CHECK: vcvtps2hf8y (%rdi), %xmm0
 // CHECK: encoding: [0x62,0xf5,0x7e,0x28,0x38,0x07]
@@ -184,9 +176,9 @@
 // CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x3a,0xc1]
           vcvtps2hf8s %xmm1, %xmm0
 
-// CHECK: vcvtps2hf8s (%rdi), %xmm0
+// CHECK: vcvtps2hf8sz (%rdi), %xmm0
 // CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x3a,0x07]
-          vcvtps2hf8s (%rdi), %xmm0
+          vcvtps2hf8sz (%rdi), %xmm0
 
 // CHECK: vcvtps2hf8sy (%rdi), %xmm0
 // CHECK: encoding: [0x62,0xf5,0x7e,0x28,0x3a,0x07]
@@ -228,9 +220,9 @@
 // CHECK: encoding: [0x62,0xf5,0x7d,0x08,0x38,0xc1]
           vcvtrops2hf8 %xmm1, %xmm0
 
-// CHECK: vcvtrops2hf8 (%rdi), %xmm0
+// CHECK: vcvtrops2hf8z (%rdi), %xmm0
 // CHECK: encoding: [0x62,0xf5,0x7d,0x48,0x38,0x07]
-          vcvtrops2hf8 (%rdi), %xmm0
+          vcvtrops2hf8z (%rdi), %xmm0
 
 // CHECK: vcvtrops2hf8y (%rdi), %xmm0
 // CHECK: encoding: [0x62,0xf5,0x7d,0x28,0x38,0x07]
@@ -272,9 +264,9 @@
 // CHECK: encoding: [0x62,0xf5,0x7d,0x08,0x3a,0xc1]
           vcvtrops2hf8s %xmm1, %xmm0
 
-// CHECK: vcvtrops2hf8s (%rdi), %xmm0
+// CHECK: vcvtrops2hf8sz (%rdi), %xmm0
 // CHECK: encoding: [0x62,0xf5,0x7d,0x48,0x3a,0x07]
-          vcvtrops2hf8s (%rdi), %xmm0
+          vcvtrops2hf8sz (%rdi), %xmm0
 
 // CHECK: vcvtrops2hf8sy (%rdi), %xmm0
 // CHECK: encoding: [0x62,0xf5,0x7d,0x28,0x3a,0x07]

>From d4ede8d245aacaf201de7af1780cf282a40e2e7e Mon Sep 17 00:00:00 2001
From: Ganesh Gopalasubramanian <Ganesh.Gopalasubramanian at amd.com>
Date: Mon, 21 Sep 2026 23:04:06 +0530
Subject: [PATCH 16/16] [X86][AVX10_V2_AUX] Address review comments

For narrowing converts which write to xmm, 128/256 use _mask and remove select
  - unmasked - *_mask with -1
  - merge masking - *_mask - W, k
  - zero masking - *_mask - zero, k

Rename header files and sync them with gcc
Add disp32 tests
---
 clang/docs/ReleaseNotes.md                    |   2 +-
 clang/include/clang/Basic/BuiltinsX86.td      |  20 ---
 .../clang/Basic/DiagnosticSemaKinds.td        |   3 +-
 clang/lib/Headers/CMakeLists.txt              |   4 +-
 ...12v2auxintrin.h => avx10v2aux_512intrin.h} |  10 +-
 ...x10_2_v2auxintrin.h => avx10v2auxintrin.h} | 168 +++++++++---------
 clang/lib/Headers/immintrin.h                 |   4 +-
 clang/lib/Sema/SemaX86.cpp                    |   1 -
 .../X86/avx10_2_v2aux-builtins-errors.c       |  14 +-
 .../test/CodeGen/X86/avx10_2_v2aux-builtins.c | 100 +++++------
 llvm/docs/ReleaseNotes.md                     |   3 +
 llvm/include/llvm/IR/IntrinsicsX86.td         |  64 ++++---
 .../X86/avx10_v2aux-pmovssdb-intrinsics.ll    |   1 +
 llvm/test/MC/X86/avx10_v2_aux-att-32.s        |  74 ++++++++
 llvm/test/MC/X86/avx10_v2_aux-att-64.s        |  74 ++++++++
 llvm/test/MC/X86/avx10_v2_aux-intel-32.s      |  74 ++++++++
 llvm/test/MC/X86/avx10_v2_aux-intel-64.s      |  74 ++++++++
 17 files changed, 480 insertions(+), 210 deletions(-)
 rename clang/lib/Headers/{avx10_2_512v2auxintrin.h => avx10v2aux_512intrin.h} (99%)
 rename clang/lib/Headers/{avx10_2_v2auxintrin.h => avx10v2auxintrin.h} (94%)

diff --git a/clang/docs/ReleaseNotes.md b/clang/docs/ReleaseNotes.md
index 2daf540fd6e60e..3a09508ef0aa66 100644
--- a/clang/docs/ReleaseNotes.md
+++ b/clang/docs/ReleaseNotes.md
@@ -632,7 +632,7 @@ features cannot lower the translation-unit ABI level;
 
 #### X86 Support
 
-- Support ISA of `AVX10_V2_AUX` (`-mavx10v2aux`).
+- Support `AVX10_V2_AUX` ISA (`-mavx10v2aux`).
 
 #### Arm and AArch64 Support
 
diff --git a/clang/include/clang/Basic/BuiltinsX86.td b/clang/include/clang/Basic/BuiltinsX86.td
index b96912528ac333..6d459683202e58 100644
--- a/clang/include/clang/Basic/BuiltinsX86.td
+++ b/clang/include/clang/Basic/BuiltinsX86.td
@@ -5069,12 +5069,10 @@ let Features = "avx10.2", Attributes = [NoThrow, Const, RequiredVectorWidth<512>
 
 // VCVTPS2BF8
 let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<128>] in {
-  def vcvtps2bf8_128 : X86Builtin<"_Vector<16, char>(_Vector<4, float>)">;
   def vcvtps2bf8_128_mask : X86Builtin<"_Vector<16, char>(_Vector<4, float>, _Vector<16, char>, unsigned char)">;
 }
 
 let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<256>] in {
-  def vcvtps2bf8_256 : X86Builtin<"_Vector<16, char>(_Vector<8, float>)">;
   def vcvtps2bf8_256_mask : X86Builtin<"_Vector<16, char>(_Vector<8, float>, _Vector<16, char>, unsigned char)">;
 }
 
@@ -5084,12 +5082,10 @@ let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<5
 
 // VCVTPS2BF8S
 let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<128>] in {
-  def vcvtps2bf8s_128 : X86Builtin<"_Vector<16, char>(_Vector<4, float>)">;
   def vcvtps2bf8s_128_mask : X86Builtin<"_Vector<16, char>(_Vector<4, float>, _Vector<16, char>, unsigned char)">;
 }
 
 let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<256>] in {
-  def vcvtps2bf8s_256 : X86Builtin<"_Vector<16, char>(_Vector<8, float>)">;
   def vcvtps2bf8s_256_mask : X86Builtin<"_Vector<16, char>(_Vector<8, float>, _Vector<16, char>, unsigned char)">;
 }
 
@@ -5099,12 +5095,10 @@ let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<5
 
 // VCVTPS2HF8
 let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<128>] in {
-  def vcvtps2hf8_128 : X86Builtin<"_Vector<16, char>(_Vector<4, float>)">;
   def vcvtps2hf8_128_mask : X86Builtin<"_Vector<16, char>(_Vector<4, float>, _Vector<16, char>, unsigned char)">;
 }
 
 let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<256>] in {
-  def vcvtps2hf8_256 : X86Builtin<"_Vector<16, char>(_Vector<8, float>)">;
   def vcvtps2hf8_256_mask : X86Builtin<"_Vector<16, char>(_Vector<8, float>, _Vector<16, char>, unsigned char)">;
 }
 
@@ -5114,12 +5108,10 @@ let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<5
 
 // VCVTPS2HF8S
 let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<128>] in {
-  def vcvtps2hf8s_128 : X86Builtin<"_Vector<16, char>(_Vector<4, float>)">;
   def vcvtps2hf8s_128_mask : X86Builtin<"_Vector<16, char>(_Vector<4, float>, _Vector<16, char>, unsigned char)">;
 }
 
 let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<256>] in {
-  def vcvtps2hf8s_256 : X86Builtin<"_Vector<16, char>(_Vector<8, float>)">;
   def vcvtps2hf8s_256_mask : X86Builtin<"_Vector<16, char>(_Vector<8, float>, _Vector<16, char>, unsigned char)">;
 }
 
@@ -5129,12 +5121,10 @@ let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<5
 
 // VCVTROPS2HF8
 let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<128>] in {
-  def vcvtrops2hf8_128 : X86Builtin<"_Vector<16, char>(_Vector<4, float>)">;
   def vcvtrops2hf8_128_mask : X86Builtin<"_Vector<16, char>(_Vector<4, float>, _Vector<16, char>, unsigned char)">;
 }
 
 let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<256>] in {
-  def vcvtrops2hf8_256 : X86Builtin<"_Vector<16, char>(_Vector<8, float>)">;
   def vcvtrops2hf8_256_mask : X86Builtin<"_Vector<16, char>(_Vector<8, float>, _Vector<16, char>, unsigned char)">;
 }
 
@@ -5144,12 +5134,10 @@ let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<5
 
 // VCVTROPS2HF8S
 let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<128>] in {
-  def vcvtrops2hf8s_128 : X86Builtin<"_Vector<16, char>(_Vector<4, float>)">;
   def vcvtrops2hf8s_128_mask : X86Builtin<"_Vector<16, char>(_Vector<4, float>, _Vector<16, char>, unsigned char)">;
 }
 
 let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<256>] in {
-  def vcvtrops2hf8s_256 : X86Builtin<"_Vector<16, char>(_Vector<8, float>)">;
   def vcvtrops2hf8s_256_mask : X86Builtin<"_Vector<16, char>(_Vector<8, float>, _Vector<16, char>, unsigned char)">;
 }
 
@@ -5161,12 +5149,10 @@ let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<5
 
 // VCVTBIASPS2BF8
 let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<128>] in {
-  def vcvtbiasps2bf8_128 : X86Builtin<"_Vector<16, char>(_Vector<4, int>, _Vector<4, float>)">;
   def vcvtbiasps2bf8_128_mask : X86Builtin<"_Vector<16, char>(_Vector<4, int>, _Vector<4, float>, _Vector<16, char>, unsigned char)">;
 }
 
 let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<256>] in {
-  def vcvtbiasps2bf8_256 : X86Builtin<"_Vector<16, char>(_Vector<8, int>, _Vector<8, float>)">;
   def vcvtbiasps2bf8_256_mask : X86Builtin<"_Vector<16, char>(_Vector<8, int>, _Vector<8, float>, _Vector<16, char>, unsigned char)">;
 }
 
@@ -5176,12 +5162,10 @@ let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<5
 
 // VCVTBIASPS2BF8S
 let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<128>] in {
-  def vcvtbiasps2bf8s_128 : X86Builtin<"_Vector<16, char>(_Vector<4, int>, _Vector<4, float>)">;
   def vcvtbiasps2bf8s_128_mask : X86Builtin<"_Vector<16, char>(_Vector<4, int>, _Vector<4, float>, _Vector<16, char>, unsigned char)">;
 }
 
 let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<256>] in {
-  def vcvtbiasps2bf8s_256 : X86Builtin<"_Vector<16, char>(_Vector<8, int>, _Vector<8, float>)">;
   def vcvtbiasps2bf8s_256_mask : X86Builtin<"_Vector<16, char>(_Vector<8, int>, _Vector<8, float>, _Vector<16, char>, unsigned char)">;
 }
 
@@ -5191,12 +5175,10 @@ let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<5
 
 // VCVTBIASPS2HF8
 let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<128>] in {
-  def vcvtbiasps2hf8_128 : X86Builtin<"_Vector<16, char>(_Vector<4, int>, _Vector<4, float>)">;
   def vcvtbiasps2hf8_128_mask : X86Builtin<"_Vector<16, char>(_Vector<4, int>, _Vector<4, float>, _Vector<16, char>, unsigned char)">;
 }
 
 let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<256>] in {
-  def vcvtbiasps2hf8_256 : X86Builtin<"_Vector<16, char>(_Vector<8, int>, _Vector<8, float>)">;
   def vcvtbiasps2hf8_256_mask : X86Builtin<"_Vector<16, char>(_Vector<8, int>, _Vector<8, float>, _Vector<16, char>, unsigned char)">;
 }
 
@@ -5206,12 +5188,10 @@ let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<5
 
 // VCVTBIASPS2HF8S
 let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<128>] in {
-  def vcvtbiasps2hf8s_128 : X86Builtin<"_Vector<16, char>(_Vector<4, int>, _Vector<4, float>)">;
   def vcvtbiasps2hf8s_128_mask : X86Builtin<"_Vector<16, char>(_Vector<4, int>, _Vector<4, float>, _Vector<16, char>, unsigned char)">;
 }
 
 let Features = "avx10v2aux", Attributes = [NoThrow, Const, RequiredVectorWidth<256>] in {
-  def vcvtbiasps2hf8s_256 : X86Builtin<"_Vector<16, char>(_Vector<8, int>, _Vector<8, float>)">;
   def vcvtbiasps2hf8s_256_mask : X86Builtin<"_Vector<16, char>(_Vector<8, int>, _Vector<8, float>, _Vector<16, char>, unsigned char)">;
 }
 
diff --git a/clang/include/clang/Basic/DiagnosticSemaKinds.td b/clang/include/clang/Basic/DiagnosticSemaKinds.td
index 42372d2dcae99b..92346360ab98fb 100644
--- a/clang/include/clang/Basic/DiagnosticSemaKinds.td
+++ b/clang/include/clang/Basic/DiagnosticSemaKinds.td
@@ -11599,7 +11599,8 @@ def err_ppc_invalid_arg_type : Error<
 def err_x86_builtin_invalid_rounding : Error<
   "invalid rounding argument">;
 def err_x86_builtin_reserved_vunpackb_imm : Error<
-  "argument value %0 is a reserved VUNPACKB immediate">;
+  "argument value %0 is a reserved VUNPACKB encoding of size (bits [4:2]) and "
+  "start (bits [1:0])">;
 def err_x86_builtin_invalid_scale : Error<
   "scale argument must be 1, 2, 4, or 8">;
 def err_x86_builtin_tile_arg_duplicate : Error<
diff --git a/clang/lib/Headers/CMakeLists.txt b/clang/lib/Headers/CMakeLists.txt
index 9d73e541e00046..96bc177550697d 100644
--- a/clang/lib/Headers/CMakeLists.txt
+++ b/clang/lib/Headers/CMakeLists.txt
@@ -182,8 +182,8 @@ set(x86_files
   avx10_2_512niintrin.h
   avx10_2_512satcvtdsintrin.h
   avx10_2_512satcvtintrin.h
-  avx10_2_512v2auxintrin.h
-  avx10_2_v2auxintrin.h
+  avx10v2aux_512intrin.h
+  avx10v2auxintrin.h
   avx10_2bf16intrin.h
   avx10_2convertintrin.h
   avx10_2copyintrin.h
diff --git a/clang/lib/Headers/avx10_2_512v2auxintrin.h b/clang/lib/Headers/avx10v2aux_512intrin.h
similarity index 99%
rename from clang/lib/Headers/avx10_2_512v2auxintrin.h
rename to clang/lib/Headers/avx10v2aux_512intrin.h
index 6e448fbb4a9155..51e53325442584 100644
--- a/clang/lib/Headers/avx10_2_512v2auxintrin.h
+++ b/clang/lib/Headers/avx10v2aux_512intrin.h
@@ -1,4 +1,4 @@
-/*===--------- avx10_2_512v2auxintrin.h - AVX10_2_512V2AUX ---------------===
+/*===------------- avx10v2aux_512intrin.h - AVX10V2AUX 512 ----------------===
  *
  * Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
  * See https://llvm.org/LICENSE.txt for license information.
@@ -8,13 +8,13 @@
  */
 #ifndef __IMMINTRIN_H
 #error                                                                         \
-    "Never use <avx10_2_512v2auxintrin.h> directly; include <immintrin.h> instead."
+    "Never use <avx10v2aux_512intrin.h> directly; include <immintrin.h> instead."
 #endif // __IMMINTRIN_H
 
 #ifdef __SSE2__
 
-#ifndef __AVX10_2_512V2AUXINTRIN_H
-#define __AVX10_2_512V2AUXINTRIN_H
+#ifndef __AVX10V2AUX_512INTRIN_H
+#define __AVX10V2AUX_512INTRIN_H
 
 /* Define the default attributes for the functions in this file. */
 #define __DEFAULT_FN_ATTRS512                                                  \
@@ -1167,5 +1167,5 @@ _mm512_mask_cvtss_epi32_storeu_epi8(void *__P, __mmask16 __M, __m512i __A) {
 
 #undef __DEFAULT_FN_ATTRS512
 
-#endif // __AVX10_2_512V2AUXINTRIN_H
+#endif // __AVX10V2AUX_512INTRIN_H
 #endif // __SSE2__
diff --git a/clang/lib/Headers/avx10_2_v2auxintrin.h b/clang/lib/Headers/avx10v2auxintrin.h
similarity index 94%
rename from clang/lib/Headers/avx10_2_v2auxintrin.h
rename to clang/lib/Headers/avx10v2auxintrin.h
index dc14f2ba0dd420..2fffe57ce1f24f 100644
--- a/clang/lib/Headers/avx10_2_v2auxintrin.h
+++ b/clang/lib/Headers/avx10v2auxintrin.h
@@ -1,4 +1,4 @@
-/*===------------ avx10_2_v2auxintrin.h - AVX10_2_V2AUX -------------------===
+/*===---------------- avx10v2auxintrin.h - AVX10V2AUX ---------------------===
  *
  * Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
  * See https://llvm.org/LICENSE.txt for license information.
@@ -8,13 +8,13 @@
  */
 #ifndef __IMMINTRIN_H
 #error                                                                         \
-    "Never use <avx10_2_v2auxintrin.h> directly; include <immintrin.h> instead."
+    "Never use <avx10v2auxintrin.h> directly; include <immintrin.h> instead."
 #endif // __IMMINTRIN_H
 
 #ifdef __SSE2__
 
-#ifndef __AVX10_2_V2AUXINTRIN_H
-#define __AVX10_2_V2AUXINTRIN_H
+#ifndef __AVX10V2AUXINTRIN_H
+#define __AVX10V2AUXINTRIN_H
 
 /* Define the default attributes for the functions in this file. */
 #define __DEFAULT_FN_ATTRS128                                                  \
@@ -38,7 +38,8 @@
 ///    A 128-bit vector of [16 x i8]. The lower 4 bytes contain the converted
 ///    values; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvtps_bf8(__m128 __A) {
-  return (__m128i)__builtin_ia32_vcvtps2bf8_128((__v4sf)__A);
+  return (__m128i)__builtin_ia32_vcvtps2bf8_128_mask(
+      (__v4sf)__A, (__v16qi)_mm_undefined_si128(), (__mmask8)-1);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __A
@@ -83,8 +84,8 @@ static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_mask_cvtps_bf8(__m128i __W,
 ///    values, or zero where the mask bit is clear; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_maskz_cvtps_bf8(__mmask8 __U, __m128 __A) {
-  return (__m128i)__builtin_ia32_selectb_128(
-      (__mmask8)__U, (__v16qi)_mm_cvtps_bf8(__A), (__v16qi)_mm_setzero_si128());
+  return (__m128i)__builtin_ia32_vcvtps2bf8_128_mask(
+      (__v4sf)__A, (__v16qi)_mm_setzero_si128(), (__mmask8)__U);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __A
@@ -101,7 +102,8 @@ _mm_maskz_cvtps_bf8(__mmask8 __U, __m128 __A) {
 ///    A 128-bit vector of [16 x i8]. The lower 8 bytes contain the converted
 ///    values; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256 _mm256_cvtps_bf8(__m256 __A) {
-  return (__m128i)__builtin_ia32_vcvtps2bf8_256((__v8sf)__A);
+  return (__m128i)__builtin_ia32_vcvtps2bf8_256_mask(
+      (__v8sf)__A, (__v16qi)_mm_undefined_si128(), (__mmask8)-1);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __A
@@ -145,9 +147,8 @@ _mm256_mask_cvtps_bf8(__m128i __W, __mmask8 __U, __m256 __A) {
 ///    values, or zero where the mask bit is clear; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_maskz_cvtps_bf8(__mmask8 __U, __m256 __A) {
-  return (__m128i)__builtin_ia32_selectb_128((__mmask8)__U,
-                                             (__v16qi)_mm256_cvtps_bf8(__A),
-                                             (__v16qi)_mm_setzero_si128());
+  return (__m128i)__builtin_ia32_vcvtps2bf8_256_mask(
+      (__v8sf)__A, (__v16qi)_mm_setzero_si128(), (__mmask8)__U);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __A
@@ -163,7 +164,8 @@ _mm256_maskz_cvtps_bf8(__mmask8 __U, __m256 __A) {
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvts_ps_bf8(__m128 __A) {
-  return (__m128i)__builtin_ia32_vcvtps2bf8s_128((__v4sf)__A);
+  return (__m128i)__builtin_ia32_vcvtps2bf8s_128_mask(
+      (__v4sf)__A, (__v16qi)_mm_undefined_si128(), (__mmask8)-1);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __A
@@ -207,9 +209,8 @@ _mm_mask_cvts_ps_bf8(__m128i __W, __mmask8 __U, __m128 __A) {
 ///    values, or zero where the mask bit is clear; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_maskz_cvts_ps_bf8(__mmask8 __U, __m128 __A) {
-  return (__m128i)__builtin_ia32_selectb_128((__mmask8)__U,
-                                             (__v16qi)_mm_cvts_ps_bf8(__A),
-                                             (__v16qi)_mm_setzero_si128());
+  return (__m128i)__builtin_ia32_vcvtps2bf8s_128_mask(
+      (__v4sf)__A, (__v16qi)_mm_setzero_si128(), (__mmask8)__U);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __A
@@ -225,7 +226,8 @@ _mm_maskz_cvts_ps_bf8(__mmask8 __U, __m128 __A) {
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256 _mm256_cvts_ps_bf8(__m256 __A) {
-  return (__m128i)__builtin_ia32_vcvtps2bf8s_256((__v8sf)__A);
+  return (__m128i)__builtin_ia32_vcvtps2bf8s_256_mask(
+      (__v8sf)__A, (__v16qi)_mm_undefined_si128(), (__mmask8)-1);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __A
@@ -269,9 +271,8 @@ _mm256_mask_cvts_ps_bf8(__m128i __W, __mmask8 __U, __m256 __A) {
 ///    values, or zero where the mask bit is clear; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_maskz_cvts_ps_bf8(__mmask8 __U, __m256 __A) {
-  return (__m128i)__builtin_ia32_selectb_128((__mmask8)__U,
-                                             (__v16qi)_mm256_cvts_ps_bf8(__A),
-                                             (__v16qi)_mm_setzero_si128());
+  return (__m128i)__builtin_ia32_vcvtps2bf8s_256_mask(
+      (__v8sf)__A, (__v16qi)_mm_setzero_si128(), (__mmask8)__U);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __A
@@ -287,7 +288,8 @@ _mm256_maskz_cvts_ps_bf8(__mmask8 __U, __m256 __A) {
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvtps_hf8(__m128 __A) {
-  return (__m128i)__builtin_ia32_vcvtps2hf8_128((__v4sf)__A);
+  return (__m128i)__builtin_ia32_vcvtps2hf8_128_mask(
+      (__v4sf)__A, (__v16qi)_mm_undefined_si128(), (__mmask8)-1);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __A
@@ -332,8 +334,8 @@ static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_mask_cvtps_hf8(__m128i __W,
 ///    values, or zero where the mask bit is clear; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_maskz_cvtps_hf8(__mmask8 __U, __m128 __A) {
-  return (__m128i)__builtin_ia32_selectb_128(
-      (__mmask8)__U, (__v16qi)_mm_cvtps_hf8(__A), (__v16qi)_mm_setzero_si128());
+  return (__m128i)__builtin_ia32_vcvtps2hf8_128_mask(
+      (__v4sf)__A, (__v16qi)_mm_setzero_si128(), (__mmask8)__U);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __A
@@ -349,7 +351,8 @@ _mm_maskz_cvtps_hf8(__mmask8 __U, __m128 __A) {
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256 _mm256_cvtps_hf8(__m256 __A) {
-  return (__m128i)__builtin_ia32_vcvtps2hf8_256((__v8sf)__A);
+  return (__m128i)__builtin_ia32_vcvtps2hf8_256_mask(
+      (__v8sf)__A, (__v16qi)_mm_undefined_si128(), (__mmask8)-1);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __A
@@ -393,9 +396,8 @@ _mm256_mask_cvtps_hf8(__m128i __W, __mmask8 __U, __m256 __A) {
 ///    values, or zero where the mask bit is clear; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_maskz_cvtps_hf8(__mmask8 __U, __m256 __A) {
-  return (__m128i)__builtin_ia32_selectb_128((__mmask8)__U,
-                                             (__v16qi)_mm256_cvtps_hf8(__A),
-                                             (__v16qi)_mm_setzero_si128());
+  return (__m128i)__builtin_ia32_vcvtps2hf8_256_mask(
+      (__v8sf)__A, (__v16qi)_mm_setzero_si128(), (__mmask8)__U);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __A
@@ -411,7 +413,8 @@ _mm256_maskz_cvtps_hf8(__mmask8 __U, __m256 __A) {
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvts_ps_hf8(__m128 __A) {
-  return (__m128i)__builtin_ia32_vcvtps2hf8s_128((__v4sf)__A);
+  return (__m128i)__builtin_ia32_vcvtps2hf8s_128_mask(
+      (__v4sf)__A, (__v16qi)_mm_undefined_si128(), (__mmask8)-1);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __A
@@ -455,9 +458,8 @@ _mm_mask_cvts_ps_hf8(__m128i __W, __mmask8 __U, __m128 __A) {
 ///    values, or zero where the mask bit is clear; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_maskz_cvts_ps_hf8(__mmask8 __U, __m128 __A) {
-  return (__m128i)__builtin_ia32_selectb_128((__mmask8)__U,
-                                             (__v16qi)_mm_cvts_ps_hf8(__A),
-                                             (__v16qi)_mm_setzero_si128());
+  return (__m128i)__builtin_ia32_vcvtps2hf8s_128_mask(
+      (__v4sf)__A, (__v16qi)_mm_setzero_si128(), (__mmask8)__U);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __A
@@ -473,7 +475,8 @@ _mm_maskz_cvts_ps_hf8(__mmask8 __U, __m128 __A) {
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256 _mm256_cvts_ps_hf8(__m256 __A) {
-  return (__m128i)__builtin_ia32_vcvtps2hf8s_256((__v8sf)__A);
+  return (__m128i)__builtin_ia32_vcvtps2hf8s_256_mask(
+      (__v8sf)__A, (__v16qi)_mm_undefined_si128(), (__mmask8)-1);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __A
@@ -517,9 +520,8 @@ _mm256_mask_cvts_ps_hf8(__m128i __W, __mmask8 __U, __m256 __A) {
 ///    values, or zero where the mask bit is clear; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_maskz_cvts_ps_hf8(__mmask8 __U, __m256 __A) {
-  return (__m128i)__builtin_ia32_selectb_128((__mmask8)__U,
-                                             (__v16qi)_mm256_cvts_ps_hf8(__A),
-                                             (__v16qi)_mm_setzero_si128());
+  return (__m128i)__builtin_ia32_vcvtps2hf8s_256_mask(
+      (__v8sf)__A, (__v16qi)_mm_setzero_si128(), (__mmask8)__U);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __A
@@ -535,7 +537,8 @@ _mm256_maskz_cvts_ps_hf8(__mmask8 __U, __m256 __A) {
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvtrops_hf8(__m128 __A) {
-  return (__m128i)__builtin_ia32_vcvtrops2hf8_128((__v4sf)__A);
+  return (__m128i)__builtin_ia32_vcvtrops2hf8_128_mask(
+      (__v4sf)__A, (__v16qi)_mm_undefined_si128(), (__mmask8)-1);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __A
@@ -579,9 +582,8 @@ _mm_mask_cvtrops_hf8(__m128i __W, __mmask8 __U, __m128 __A) {
 ///    values, or zero where the mask bit is clear; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_maskz_cvtrops_hf8(__mmask8 __U, __m128 __A) {
-  return (__m128i)__builtin_ia32_selectb_128((__mmask8)__U,
-                                             (__v16qi)_mm_cvtrops_hf8(__A),
-                                             (__v16qi)_mm_setzero_si128());
+  return (__m128i)__builtin_ia32_vcvtrops2hf8_128_mask(
+      (__v4sf)__A, (__v16qi)_mm_setzero_si128(), (__mmask8)__U);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __A
@@ -597,7 +599,8 @@ _mm_maskz_cvtrops_hf8(__mmask8 __U, __m128 __A) {
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256 _mm256_cvtrops_hf8(__m256 __A) {
-  return (__m128i)__builtin_ia32_vcvtrops2hf8_256((__v8sf)__A);
+  return (__m128i)__builtin_ia32_vcvtrops2hf8_256_mask(
+      (__v8sf)__A, (__v16qi)_mm_undefined_si128(), (__mmask8)-1);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __A
@@ -641,9 +644,8 @@ _mm256_mask_cvtrops_hf8(__m128i __W, __mmask8 __U, __m256 __A) {
 ///    values, or zero where the mask bit is clear; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_maskz_cvtrops_hf8(__mmask8 __U, __m256 __A) {
-  return (__m128i)__builtin_ia32_selectb_128((__mmask8)__U,
-                                             (__v16qi)_mm256_cvtrops_hf8(__A),
-                                             (__v16qi)_mm_setzero_si128());
+  return (__m128i)__builtin_ia32_vcvtrops2hf8_256_mask(
+      (__v8sf)__A, (__v16qi)_mm_setzero_si128(), (__mmask8)__U);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __A
@@ -659,7 +661,8 @@ _mm256_maskz_cvtrops_hf8(__mmask8 __U, __m256 __A) {
 /// \returns
 ///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvts_rops_hf8(__m128 __A) {
-  return (__m128i)__builtin_ia32_vcvtrops2hf8s_128((__v4sf)__A);
+  return (__m128i)__builtin_ia32_vcvtrops2hf8s_128_mask(
+      (__v4sf)__A, (__v16qi)_mm_undefined_si128(), (__mmask8)-1);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __A
@@ -703,9 +706,8 @@ _mm_mask_cvts_rops_hf8(__m128i __W, __mmask8 __U, __m128 __A) {
 ///    values, or zero where the mask bit is clear; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_maskz_cvts_rops_hf8(__mmask8 __U, __m128 __A) {
-  return (__m128i)__builtin_ia32_selectb_128((__mmask8)__U,
-                                             (__v16qi)_mm_cvts_rops_hf8(__A),
-                                             (__v16qi)_mm_setzero_si128());
+  return (__m128i)__builtin_ia32_vcvtrops2hf8s_128_mask(
+      (__v4sf)__A, (__v16qi)_mm_setzero_si128(), (__mmask8)__U);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __A
@@ -722,7 +724,8 @@ _mm_maskz_cvts_rops_hf8(__mmask8 __U, __m128 __A) {
 ///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_cvts_rops_hf8(__m256 __A) {
-  return (__m128i)__builtin_ia32_vcvtrops2hf8s_256((__v8sf)__A);
+  return (__m128i)__builtin_ia32_vcvtrops2hf8s_256_mask(
+      (__v8sf)__A, (__v16qi)_mm_undefined_si128(), (__mmask8)-1);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __A
@@ -766,9 +769,8 @@ _mm256_mask_cvts_rops_hf8(__m128i __W, __mmask8 __U, __m256 __A) {
 ///    values, or zero where the mask bit is clear; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_maskz_cvts_rops_hf8(__mmask8 __U, __m256 __A) {
-  return (__m128i)__builtin_ia32_selectb_128((__mmask8)__U,
-                                             (__v16qi)_mm256_cvts_rops_hf8(__A),
-                                             (__v16qi)_mm_setzero_si128());
+  return (__m128i)__builtin_ia32_vcvtrops2hf8s_256_mask(
+      (__v8sf)__A, (__v16qi)_mm_setzero_si128(), (__mmask8)__U);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __B
@@ -787,7 +789,8 @@ _mm256_maskz_cvts_rops_hf8(__mmask8 __U, __m256 __A) {
 ///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvtbiasps_bf8(__m128i __A,
                                                                   __m128 __B) {
-  return (__m128i)__builtin_ia32_vcvtbiasps2bf8_128((__v4si)__A, (__v4sf)__B);
+  return (__m128i)__builtin_ia32_vcvtbiasps2bf8_128_mask(
+      (__v4si)__A, (__v4sf)__B, (__v16qi)_mm_undefined_si128(), (__mmask8)-1);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __B
@@ -835,9 +838,8 @@ _mm_mask_cvtbiasps_bf8(__m128i __W, __mmask8 __U, __m128i __A, __m128 __B) {
 ///    values, or zero where the mask bit is clear; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_maskz_cvtbiasps_bf8(__mmask8 __U, __m128i __A, __m128 __B) {
-  return (__m128i)__builtin_ia32_selectb_128(
-      (__mmask8)__U, (__v16qi)_mm_cvtbiasps_bf8(__A, __B),
-      (__v16qi)_mm_setzero_si128());
+  return (__m128i)__builtin_ia32_vcvtbiasps2bf8_128_mask(
+      (__v4si)__A, (__v4sf)__B, (__v16qi)_mm_setzero_si128(), (__mmask8)__U);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __B
@@ -856,7 +858,8 @@ _mm_maskz_cvtbiasps_bf8(__mmask8 __U, __m128i __A, __m128 __B) {
 ///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_cvtbiasps_bf8(__m256i __A, __m256 __B) {
-  return (__m128i)__builtin_ia32_vcvtbiasps2bf8_256((__v8si)__A, (__v8sf)__B);
+  return (__m128i)__builtin_ia32_vcvtbiasps2bf8_256_mask(
+      (__v8si)__A, (__v8sf)__B, (__v16qi)_mm_undefined_si128(), (__mmask8)-1);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __B
@@ -904,9 +907,8 @@ _mm256_mask_cvtbiasps_bf8(__m128i __W, __mmask8 __U, __m256i __A, __m256 __B) {
 ///    values, or zero where the mask bit is clear; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_maskz_cvtbiasps_bf8(__mmask8 __U, __m256i __A, __m256 __B) {
-  return (__m128i)__builtin_ia32_selectb_128(
-      (__mmask8)__U, (__v16qi)_mm256_cvtbiasps_bf8(__A, __B),
-      (__v16qi)_mm_setzero_si128());
+  return (__m128i)__builtin_ia32_vcvtbiasps2bf8_256_mask(
+      (__v8si)__A, (__v8sf)__B, (__v16qi)_mm_setzero_si128(), (__mmask8)__U);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __B
@@ -925,7 +927,8 @@ _mm256_maskz_cvtbiasps_bf8(__mmask8 __U, __m256i __A, __m256 __B) {
 ///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_cvts_biasps_bf8(__m128i __A, __m128 __B) {
-  return (__m128i)__builtin_ia32_vcvtbiasps2bf8s_128((__v4si)__A, (__v4sf)__B);
+  return (__m128i)__builtin_ia32_vcvtbiasps2bf8s_128_mask(
+      (__v4si)__A, (__v4sf)__B, (__v16qi)_mm_undefined_si128(), (__mmask8)-1);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __B
@@ -973,9 +976,8 @@ _mm_mask_cvts_biasps_bf8(__m128i __W, __mmask8 __U, __m128i __A, __m128 __B) {
 ///    values, or zero where the mask bit is clear; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_maskz_cvts_biasps_bf8(__mmask8 __U, __m128i __A, __m128 __B) {
-  return (__m128i)__builtin_ia32_selectb_128(
-      (__mmask8)__U, (__v16qi)_mm_cvts_biasps_bf8(__A, __B),
-      (__v16qi)_mm_setzero_si128());
+  return (__m128i)__builtin_ia32_vcvtbiasps2bf8s_128_mask(
+      (__v4si)__A, (__v4sf)__B, (__v16qi)_mm_setzero_si128(), (__mmask8)__U);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __B
@@ -994,7 +996,8 @@ _mm_maskz_cvts_biasps_bf8(__mmask8 __U, __m128i __A, __m128 __B) {
 ///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_cvts_biasps_bf8(__m256i __A, __m256 __B) {
-  return (__m128i)__builtin_ia32_vcvtbiasps2bf8s_256((__v8si)__A, (__v8sf)__B);
+  return (__m128i)__builtin_ia32_vcvtbiasps2bf8s_256_mask(
+      (__v8si)__A, (__v8sf)__B, (__v16qi)_mm_undefined_si128(), (__mmask8)-1);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __B
@@ -1042,9 +1045,8 @@ static __inline__ __m128i __DEFAULT_FN_ATTRS256 _mm256_mask_cvts_biasps_bf8(
 ///    values, or zero where the mask bit is clear; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_maskz_cvts_biasps_bf8(__mmask8 __U, __m256i __A, __m256 __B) {
-  return (__m128i)__builtin_ia32_selectb_128(
-      (__mmask8)__U, (__v16qi)_mm256_cvts_biasps_bf8(__A, __B),
-      (__v16qi)_mm_setzero_si128());
+  return (__m128i)__builtin_ia32_vcvtbiasps2bf8s_256_mask(
+      (__v8si)__A, (__v8sf)__B, (__v16qi)_mm_setzero_si128(), (__mmask8)__U);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __B
@@ -1063,7 +1065,8 @@ _mm256_maskz_cvts_biasps_bf8(__mmask8 __U, __m256i __A, __m256 __B) {
 ///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_cvtbiasps_hf8(__m128i __A,
                                                                   __m128 __B) {
-  return (__m128i)__builtin_ia32_vcvtbiasps2hf8_128((__v4si)__A, (__v4sf)__B);
+  return (__m128i)__builtin_ia32_vcvtbiasps2hf8_128_mask(
+      (__v4si)__A, (__v4sf)__B, (__v16qi)_mm_undefined_si128(), (__mmask8)-1);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __B
@@ -1111,9 +1114,8 @@ _mm_mask_cvtbiasps_hf8(__m128i __W, __mmask8 __U, __m128i __A, __m128 __B) {
 ///    values, or zero where the mask bit is clear; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_maskz_cvtbiasps_hf8(__mmask8 __U, __m128i __A, __m128 __B) {
-  return (__m128i)__builtin_ia32_selectb_128(
-      (__mmask8)__U, (__v16qi)_mm_cvtbiasps_hf8(__A, __B),
-      (__v16qi)_mm_setzero_si128());
+  return (__m128i)__builtin_ia32_vcvtbiasps2hf8_128_mask(
+      (__v4si)__A, (__v4sf)__B, (__v16qi)_mm_setzero_si128(), (__mmask8)__U);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __B
@@ -1132,7 +1134,8 @@ _mm_maskz_cvtbiasps_hf8(__mmask8 __U, __m128i __A, __m128 __B) {
 ///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_cvtbiasps_hf8(__m256i __A, __m256 __B) {
-  return (__m128i)__builtin_ia32_vcvtbiasps2hf8_256((__v8si)__A, (__v8sf)__B);
+  return (__m128i)__builtin_ia32_vcvtbiasps2hf8_256_mask(
+      (__v8si)__A, (__v8sf)__B, (__v16qi)_mm_undefined_si128(), (__mmask8)-1);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __B
@@ -1180,9 +1183,8 @@ _mm256_mask_cvtbiasps_hf8(__m128i __W, __mmask8 __U, __m256i __A, __m256 __B) {
 ///    values, or zero where the mask bit is clear; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_maskz_cvtbiasps_hf8(__mmask8 __U, __m256i __A, __m256 __B) {
-  return (__m128i)__builtin_ia32_selectb_128(
-      (__mmask8)__U, (__v16qi)_mm256_cvtbiasps_hf8(__A, __B),
-      (__v16qi)_mm_setzero_si128());
+  return (__m128i)__builtin_ia32_vcvtbiasps2hf8_256_mask(
+      (__v8si)__A, (__v8sf)__B, (__v16qi)_mm_setzero_si128(), (__mmask8)__U);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __B
@@ -1201,7 +1203,8 @@ _mm256_maskz_cvtbiasps_hf8(__mmask8 __U, __m256i __A, __m256 __B) {
 ///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_cvts_biasps_hf8(__m128i __A, __m128 __B) {
-  return (__m128i)__builtin_ia32_vcvtbiasps2hf8s_128((__v4si)__A, (__v4sf)__B);
+  return (__m128i)__builtin_ia32_vcvtbiasps2hf8s_128_mask(
+      (__v4si)__A, (__v4sf)__B, (__v16qi)_mm_undefined_si128(), (__mmask8)-1);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __B
@@ -1249,9 +1252,8 @@ _mm_mask_cvts_biasps_hf8(__m128i __W, __mmask8 __U, __m128i __A, __m128 __B) {
 ///    values, or zero where the mask bit is clear; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS128
 _mm_maskz_cvts_biasps_hf8(__mmask8 __U, __m128i __A, __m128 __B) {
-  return (__m128i)__builtin_ia32_selectb_128(
-      (__mmask8)__U, (__v16qi)_mm_cvts_biasps_hf8(__A, __B),
-      (__v16qi)_mm_setzero_si128());
+  return (__m128i)__builtin_ia32_vcvtbiasps2hf8s_128_mask(
+      (__v4si)__A, (__v4sf)__B, (__v16qi)_mm_setzero_si128(), (__mmask8)__U);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __B
@@ -1270,7 +1272,8 @@ _mm_maskz_cvts_biasps_hf8(__mmask8 __U, __m128i __A, __m128 __B) {
 ///    A 128-bit vector of [16 x i8] containing the converted values.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_cvts_biasps_hf8(__m256i __A, __m256 __B) {
-  return (__m128i)__builtin_ia32_vcvtbiasps2hf8s_256((__v8si)__A, (__v8sf)__B);
+  return (__m128i)__builtin_ia32_vcvtbiasps2hf8s_256_mask(
+      (__v8si)__A, (__v8sf)__B, (__v16qi)_mm_undefined_si128(), (__mmask8)-1);
 }
 
 /// Convert packed single-precision (32-bit) floating-point elements in \a __B
@@ -1318,9 +1321,8 @@ static __inline__ __m128i __DEFAULT_FN_ATTRS256 _mm256_mask_cvts_biasps_hf8(
 ///    values, or zero where the mask bit is clear; the upper bytes are zeroed.
 static __inline__ __m128i __DEFAULT_FN_ATTRS256
 _mm256_maskz_cvts_biasps_hf8(__mmask8 __U, __m256i __A, __m256 __B) {
-  return (__m128i)__builtin_ia32_selectb_128(
-      (__mmask8)__U, (__v16qi)_mm256_cvts_biasps_hf8(__A, __B),
-      (__v16qi)_mm_setzero_si128());
+  return (__m128i)__builtin_ia32_vcvtbiasps2hf8s_256_mask(
+      (__v8si)__A, (__v8sf)__B, (__v16qi)_mm_setzero_si128(), (__mmask8)__U);
 }
 
 /// Convert packed BF8 (8-bit) floating-point elements in \a __A to packed
@@ -2398,5 +2400,5 @@ _mm256_mask_cvtss_epi32_storeu_epi8(void *__P, __mmask8 __M, __m256i __A) {
 #undef __DEFAULT_FN_ATTRS128
 #undef __DEFAULT_FN_ATTRS256
 
-#endif // __AVX10_2_V2AUXINTRIN_H
+#endif // __AVX10V2AUXINTRIN_H
 #endif // __SSE2__
diff --git a/clang/lib/Headers/immintrin.h b/clang/lib/Headers/immintrin.h
index b386cb5f112461..ef3e638d4d0439 100644
--- a/clang/lib/Headers/immintrin.h
+++ b/clang/lib/Headers/immintrin.h
@@ -502,8 +502,8 @@ _storebe_i64(void * __P, long long __D) {
 #include <avx10_2_512satcvtdsintrin.h>
 #include <avx10_2_512satcvtintrin.h>
 
-#include <avx10_2_512v2auxintrin.h>
-#include <avx10_2_v2auxintrin.h>
+#include <avx10v2aux_512intrin.h>
+#include <avx10v2auxintrin.h>
 
 #include <sm4evexintrin.h>
 
diff --git a/clang/lib/Sema/SemaX86.cpp b/clang/lib/Sema/SemaX86.cpp
index 8681972cb82c0d..0c2e6281729589 100644
--- a/clang/lib/Sema/SemaX86.cpp
+++ b/clang/lib/Sema/SemaX86.cpp
@@ -343,7 +343,6 @@ bool SemaX86::CheckBuiltinRoundingOrSAE(unsigned BuiltinID, CallExpr *TheCall) {
          << Arg->getSourceRange();
 }
 
-// Check if the VUNPACKB immediate encoding is legal.
 bool SemaX86::CheckBuiltinVUnpackBImm(CallExpr *TheCall) {
   const unsigned ArgNum = 1;
 
diff --git a/clang/test/CodeGen/X86/avx10_2_v2aux-builtins-errors.c b/clang/test/CodeGen/X86/avx10_2_v2aux-builtins-errors.c
index 4e7e055bbc7936..7969b81745eb67 100644
--- a/clang/test/CodeGen/X86/avx10_2_v2aux-builtins-errors.c
+++ b/clang/test/CodeGen/X86/avx10_2_v2aux-builtins-errors.c
@@ -39,29 +39,29 @@ __m512i test_mm512_maskz_unpack_epi8(__mmask64 __U, __m512i __A) {
 }
 
 __m128i test_mm_unpack_epi8_reserved_size0(__m128i __A) {
-  return _mm_unpack_epi8(__A, 1); // expected-error {{argument value 1 is a reserved VUNPACKB immediate}}
+  return _mm_unpack_epi8(__A, 1); // expected-error {{argument value 1 is a reserved VUNPACKB encoding of size (bits [4:2]) and start (bits [1:0])}}
 }
 
 __m256i test_mm256_unpack_epi8_reserved_size0(__m256i __A) {
-  return _mm256_unpack_epi8(__A, 2); // expected-error {{argument value 2 is a reserved VUNPACKB immediate}}
+  return _mm256_unpack_epi8(__A, 2); // expected-error {{argument value 2 is a reserved VUNPACKB encoding of size (bits [4:2]) and start (bits [1:0])}}
 }
 
 __m512i test_mm512_unpack_epi8_reserved_size0(__m512i __A) {
-  return _mm512_unpack_epi8(__A, 3); // expected-error {{argument value 3 is a reserved VUNPACKB immediate}}
+  return _mm512_unpack_epi8(__A, 3); // expected-error {{argument value 3 is a reserved VUNPACKB encoding of size (bits [4:2]) and start (bits [1:0])}}
 }
 
 __m128i test_mm_unpack_epi8_reserved_size1(__m128i __A) {
-  return _mm_unpack_epi8(__A, _MM_UNPACKB_SIZE(1)); // expected-error {{argument value 4 is a reserved VUNPACKB immediate}}
+  return _mm_unpack_epi8(__A, _MM_UNPACKB_SIZE(1)); // expected-error {{argument value 4 is a reserved VUNPACKB encoding of size (bits [4:2]) and start (bits [1:0])}}
 }
 
 __m128i test_mm_unpack_epi8_reserved_start(__m128i __A) {
-  return _mm_unpack_epi8(__A, _MM_UNPACKB_SIZE(4) | _MM_UNPACKB_START(3)); // expected-error {{argument value 19 is a reserved VUNPACKB immediate}}
+  return _mm_unpack_epi8(__A, _MM_UNPACKB_SIZE(4) | _MM_UNPACKB_START(3)); // expected-error {{argument value 19 is a reserved VUNPACKB encoding of size (bits [4:2]) and start (bits [1:0])}}
 }
 
 __m128i test_mm_unpack_epi8_reserved_size5_start(__m128i __A) {
-  return _mm_unpack_epi8(__A, _MM_UNPACKB_SIZE(5) | _MM_UNPACKB_START(1)); // expected-error {{argument value 21 is a reserved VUNPACKB immediate}}
+  return _mm_unpack_epi8(__A, _MM_UNPACKB_SIZE(5) | _MM_UNPACKB_START(1)); // expected-error {{argument value 21 is a reserved VUNPACKB encoding of size (bits [4:2]) and start (bits [1:0])}}
 }
 
 __m128i test_mm_mask_unpack_epi8_reserved(__m128i __W, __mmask16 __U, __m128i __A) {
-  return _mm_mask_unpack_epi8(__W, __U, __A, 0); // expected-error {{argument value 0 is a reserved VUNPACKB immediate}}
+  return _mm_mask_unpack_epi8(__W, __U, __A, 0); // expected-error {{argument value 0 is a reserved VUNPACKB encoding of size (bits [4:2]) and start (bits [1:0])}}
 }
diff --git a/clang/test/CodeGen/X86/avx10_2_v2aux-builtins.c b/clang/test/CodeGen/X86/avx10_2_v2aux-builtins.c
index d31715c114883d..1b58a411699eb6 100644
--- a/clang/test/CodeGen/X86/avx10_2_v2aux-builtins.c
+++ b/clang/test/CodeGen/X86/avx10_2_v2aux-builtins.c
@@ -7,7 +7,7 @@
 
 __m128i test_mm_cvtps_bf8(__m128 __A) {
   // CHECK-LABEL: @test_mm_cvtps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtps2bf8128(<4 x float> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8128(<4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 -1)
   return _mm_cvtps_bf8(__A);
 }
 
@@ -19,15 +19,14 @@ __m128i test_mm_mask_cvtps_bf8(__m128i __W, __mmask8 __U, __m128 __A) {
 
 __m128i test_mm_maskz_cvtps_bf8(__mmask8 __U, __m128 __A) {
   // CHECK-LABEL: @test_mm_maskz_cvtps_bf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8128(<4 x float> %{{.*}})
   // CHECK: zeroinitializer
-  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8128(<4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm_maskz_cvtps_bf8(__U, __A);
 }
 
 __m128i test_mm256_cvtps_bf8(__m256 __A) {
   // CHECK-LABEL: @test_mm256_cvtps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtps2bf8256(<8 x float> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8256(<8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 -1)
   return _mm256_cvtps_bf8(__A);
 }
 
@@ -39,9 +38,8 @@ __m128i test_mm256_mask_cvtps_bf8(__m128i __W, __mmask8 __U, __m256 __A) {
 
 __m128i test_mm256_maskz_cvtps_bf8(__mmask8 __U, __m256 __A) {
   // CHECK-LABEL: @test_mm256_maskz_cvtps_bf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8256(<8 x float> %{{.*}})
   // CHECK: zeroinitializer
-  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8256(<8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm256_maskz_cvtps_bf8(__U, __A);
 }
 
@@ -68,7 +66,7 @@ __m128i test_mm512_maskz_cvtps_bf8(__mmask16 __U, __m512 __A) {
 
 __m128i test_mm_cvts_ps_bf8(__m128 __A) {
   // CHECK-LABEL: @test_mm_cvts_ps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtps2bf8s128(<4 x float> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s128(<4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 -1)
   return _mm_cvts_ps_bf8(__A);
 }
 
@@ -80,15 +78,14 @@ __m128i test_mm_mask_cvts_ps_bf8(__m128i __W, __mmask8 __U, __m128 __A) {
 
 __m128i test_mm_maskz_cvts_ps_bf8(__mmask8 __U, __m128 __A) {
   // CHECK-LABEL: @test_mm_maskz_cvts_ps_bf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8s128(<4 x float> %{{.*}})
   // CHECK: zeroinitializer
-  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s128(<4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm_maskz_cvts_ps_bf8(__U, __A);
 }
 
 __m128i test_mm256_cvts_ps_bf8(__m256 __A) {
   // CHECK-LABEL: @test_mm256_cvts_ps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtps2bf8s256(<8 x float> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s256(<8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 -1)
   return _mm256_cvts_ps_bf8(__A);
 }
 
@@ -100,9 +97,8 @@ __m128i test_mm256_mask_cvts_ps_bf8(__m128i __W, __mmask8 __U, __m256 __A) {
 
 __m128i test_mm256_maskz_cvts_ps_bf8(__mmask8 __U, __m256 __A) {
   // CHECK-LABEL: @test_mm256_maskz_cvts_ps_bf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2bf8s256(<8 x float> %{{.*}})
   // CHECK: zeroinitializer
-  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2bf8s256(<8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm256_maskz_cvts_ps_bf8(__U, __A);
 }
 
@@ -129,7 +125,7 @@ __m128i test_mm512_maskz_cvts_ps_bf8(__mmask16 __U, __m512 __A) {
 
 __m128i test_mm_cvtps_hf8(__m128 __A) {
   // CHECK-LABEL: @test_mm_cvtps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtps2hf8128(<4 x float> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8128(<4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 -1)
   return _mm_cvtps_hf8(__A);
 }
 
@@ -141,15 +137,14 @@ __m128i test_mm_mask_cvtps_hf8(__m128i __W, __mmask8 __U, __m128 __A) {
 
 __m128i test_mm_maskz_cvtps_hf8(__mmask8 __U, __m128 __A) {
   // CHECK-LABEL: @test_mm_maskz_cvtps_hf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8128(<4 x float> %{{.*}})
   // CHECK: zeroinitializer
-  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8128(<4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm_maskz_cvtps_hf8(__U, __A);
 }
 
 __m128i test_mm256_cvtps_hf8(__m256 __A) {
   // CHECK-LABEL: @test_mm256_cvtps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtps2hf8256(<8 x float> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8256(<8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 -1)
   return _mm256_cvtps_hf8(__A);
 }
 
@@ -161,9 +156,8 @@ __m128i test_mm256_mask_cvtps_hf8(__m128i __W, __mmask8 __U, __m256 __A) {
 
 __m128i test_mm256_maskz_cvtps_hf8(__mmask8 __U, __m256 __A) {
   // CHECK-LABEL: @test_mm256_maskz_cvtps_hf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8256(<8 x float> %{{.*}})
   // CHECK: zeroinitializer
-  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8256(<8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm256_maskz_cvtps_hf8(__U, __A);
 }
 
@@ -190,7 +184,7 @@ __m128i test_mm512_maskz_cvtps_hf8(__mmask16 __U, __m512 __A) {
 
 __m128i test_mm_cvts_ps_hf8(__m128 __A) {
   // CHECK-LABEL: @test_mm_cvts_ps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtps2hf8s128(<4 x float> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s128(<4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 -1)
   return _mm_cvts_ps_hf8(__A);
 }
 
@@ -202,15 +196,14 @@ __m128i test_mm_mask_cvts_ps_hf8(__m128i __W, __mmask8 __U, __m128 __A) {
 
 __m128i test_mm_maskz_cvts_ps_hf8(__mmask8 __U, __m128 __A) {
   // CHECK-LABEL: @test_mm_maskz_cvts_ps_hf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8s128(<4 x float> %{{.*}})
   // CHECK: zeroinitializer
-  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s128(<4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm_maskz_cvts_ps_hf8(__U, __A);
 }
 
 __m128i test_mm256_cvts_ps_hf8(__m256 __A) {
   // CHECK-LABEL: @test_mm256_cvts_ps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtps2hf8s256(<8 x float> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s256(<8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 -1)
   return _mm256_cvts_ps_hf8(__A);
 }
 
@@ -222,9 +215,8 @@ __m128i test_mm256_mask_cvts_ps_hf8(__m128i __W, __mmask8 __U, __m256 __A) {
 
 __m128i test_mm256_maskz_cvts_ps_hf8(__mmask8 __U, __m256 __A) {
   // CHECK-LABEL: @test_mm256_maskz_cvts_ps_hf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtps2hf8s256(<8 x float> %{{.*}})
   // CHECK: zeroinitializer
-  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtps2hf8s256(<8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm256_maskz_cvts_ps_hf8(__U, __A);
 }
 
@@ -251,7 +243,7 @@ __m128i test_mm512_maskz_cvts_ps_hf8(__mmask16 __U, __m512 __A) {
 
 __m128i test_mm_cvtrops_hf8(__m128 __A) {
   // CHECK-LABEL: @test_mm_cvtrops_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8128(<4 x float> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8128(<4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 -1)
   return _mm_cvtrops_hf8(__A);
 }
 
@@ -263,15 +255,14 @@ __m128i test_mm_mask_cvtrops_hf8(__m128i __W, __mmask8 __U, __m128 __A) {
 
 __m128i test_mm_maskz_cvtrops_hf8(__mmask8 __U, __m128 __A) {
   // CHECK-LABEL: @test_mm_maskz_cvtrops_hf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8128(<4 x float> %{{.*}})
   // CHECK: zeroinitializer
-  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8128(<4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm_maskz_cvtrops_hf8(__U, __A);
 }
 
 __m128i test_mm256_cvtrops_hf8(__m256 __A) {
   // CHECK-LABEL: @test_mm256_cvtrops_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8256(<8 x float> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8256(<8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 -1)
   return _mm256_cvtrops_hf8(__A);
 }
 
@@ -283,9 +274,8 @@ __m128i test_mm256_mask_cvtrops_hf8(__m128i __W, __mmask8 __U, __m256 __A) {
 
 __m128i test_mm256_maskz_cvtrops_hf8(__mmask8 __U, __m256 __A) {
   // CHECK-LABEL: @test_mm256_maskz_cvtrops_hf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8256(<8 x float> %{{.*}})
   // CHECK: zeroinitializer
-  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8256(<8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm256_maskz_cvtrops_hf8(__U, __A);
 }
 
@@ -312,7 +302,7 @@ __m128i test_mm512_maskz_cvtrops_hf8(__mmask16 __U, __m512 __A) {
 
 __m128i test_mm_cvts_rops_hf8(__m128 __A) {
   // CHECK-LABEL: @test_mm_cvts_rops_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s128(<4 x float> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s128(<4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 -1)
   return _mm_cvts_rops_hf8(__A);
 }
 
@@ -324,15 +314,14 @@ __m128i test_mm_mask_cvts_rops_hf8(__m128i __W, __mmask8 __U, __m128 __A) {
 
 __m128i test_mm_maskz_cvts_rops_hf8(__mmask8 __U, __m128 __A) {
   // CHECK-LABEL: @test_mm_maskz_cvts_rops_hf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s128(<4 x float> %{{.*}})
   // CHECK: zeroinitializer
-  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s128(<4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm_maskz_cvts_rops_hf8(__U, __A);
 }
 
 __m128i test_mm256_cvts_rops_hf8(__m256 __A) {
   // CHECK-LABEL: @test_mm256_cvts_rops_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s256(<8 x float> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s256(<8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 -1)
   return _mm256_cvts_rops_hf8(__A);
 }
 
@@ -344,9 +333,8 @@ __m128i test_mm256_mask_cvts_rops_hf8(__m128i __W, __mmask8 __U, __m256 __A) {
 
 __m128i test_mm256_maskz_cvts_rops_hf8(__mmask8 __U, __m256 __A) {
   // CHECK-LABEL: @test_mm256_maskz_cvts_rops_hf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtrops2hf8s256(<8 x float> %{{.*}})
   // CHECK: zeroinitializer
-  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtrops2hf8s256(<8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm256_maskz_cvts_rops_hf8(__U, __A);
 }
 
@@ -373,7 +361,7 @@ __m128i test_mm512_maskz_cvts_rops_hf8(__mmask16 __U, __m512 __A) {
 
 __m128i test_mm_cvtbiasps_bf8(__m128i __A, __m128 __B) {
   // CHECK-LABEL: @test_mm_cvtbiasps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8128(<4 x i32> %{{.*}}, <4 x float> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8128(<4 x i32> %{{.*}}, <4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 -1)
   return _mm_cvtbiasps_bf8(__A, __B);
 }
 
@@ -385,15 +373,14 @@ __m128i test_mm_mask_cvtbiasps_bf8(__m128i __W, __mmask8 __U, __m128i __A, __m12
 
 __m128i test_mm_maskz_cvtbiasps_bf8(__mmask8 __U, __m128i __A, __m128 __B) {
   // CHECK-LABEL: @test_mm_maskz_cvtbiasps_bf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8128(<4 x i32> %{{.*}}, <4 x float> %{{.*}})
   // CHECK: zeroinitializer
-  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8128(<4 x i32> %{{.*}}, <4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm_maskz_cvtbiasps_bf8(__U, __A, __B);
 }
 
 __m128i test_mm256_cvtbiasps_bf8(__m256i __A, __m256 __B) {
   // CHECK-LABEL: @test_mm256_cvtbiasps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8256(<8 x i32> %{{.*}}, <8 x float> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8256(<8 x i32> %{{.*}}, <8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 -1)
   return _mm256_cvtbiasps_bf8(__A, __B);
 }
 
@@ -405,9 +392,8 @@ __m128i test_mm256_mask_cvtbiasps_bf8(__m128i __W, __mmask8 __U, __m256i __A, __
 
 __m128i test_mm256_maskz_cvtbiasps_bf8(__mmask8 __U, __m256i __A, __m256 __B) {
   // CHECK-LABEL: @test_mm256_maskz_cvtbiasps_bf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8256(<8 x i32> %{{.*}}, <8 x float> %{{.*}})
   // CHECK: zeroinitializer
-  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8256(<8 x i32> %{{.*}}, <8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm256_maskz_cvtbiasps_bf8(__U, __A, __B);
 }
 
@@ -434,7 +420,7 @@ __m128i test_mm512_maskz_cvtbiasps_bf8(__mmask16 __U, __m512i __A, __m512 __B) {
 
 __m128i test_mm_cvts_biasps_bf8(__m128i __A, __m128 __B) {
   // CHECK-LABEL: @test_mm_cvts_biasps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s128(<4 x i32> %{{.*}}, <4 x float> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s128(<4 x i32> %{{.*}}, <4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 -1)
   return _mm_cvts_biasps_bf8(__A, __B);
 }
 
@@ -446,15 +432,14 @@ __m128i test_mm_mask_cvts_biasps_bf8(__m128i __W, __mmask8 __U, __m128i __A, __m
 
 __m128i test_mm_maskz_cvts_biasps_bf8(__mmask8 __U, __m128i __A, __m128 __B) {
   // CHECK-LABEL: @test_mm_maskz_cvts_biasps_bf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s128(<4 x i32> %{{.*}}, <4 x float> %{{.*}})
   // CHECK: zeroinitializer
-  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s128(<4 x i32> %{{.*}}, <4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm_maskz_cvts_biasps_bf8(__U, __A, __B);
 }
 
 __m128i test_mm256_cvts_biasps_bf8(__m256i __A, __m256 __B) {
   // CHECK-LABEL: @test_mm256_cvts_biasps_bf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s256(<8 x i32> %{{.*}}, <8 x float> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s256(<8 x i32> %{{.*}}, <8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 -1)
   return _mm256_cvts_biasps_bf8(__A, __B);
 }
 
@@ -466,9 +451,8 @@ __m128i test_mm256_mask_cvts_biasps_bf8(__m128i __W, __mmask8 __U, __m256i __A,
 
 __m128i test_mm256_maskz_cvts_biasps_bf8(__mmask8 __U, __m256i __A, __m256 __B) {
   // CHECK-LABEL: @test_mm256_maskz_cvts_biasps_bf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2bf8s256(<8 x i32> %{{.*}}, <8 x float> %{{.*}})
   // CHECK: zeroinitializer
-  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2bf8s256(<8 x i32> %{{.*}}, <8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm256_maskz_cvts_biasps_bf8(__U, __A, __B);
 }
 
@@ -495,7 +479,7 @@ __m128i test_mm512_maskz_cvts_biasps_bf8(__mmask16 __U, __m512i __A, __m512 __B)
 
 __m128i test_mm_cvtbiasps_hf8(__m128i __A, __m128 __B) {
   // CHECK-LABEL: @test_mm_cvtbiasps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8128(<4 x i32> %{{.*}}, <4 x float> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8128(<4 x i32> %{{.*}}, <4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 -1)
   return _mm_cvtbiasps_hf8(__A, __B);
 }
 
@@ -507,15 +491,14 @@ __m128i test_mm_mask_cvtbiasps_hf8(__m128i __W, __mmask8 __U, __m128i __A, __m12
 
 __m128i test_mm_maskz_cvtbiasps_hf8(__mmask8 __U, __m128i __A, __m128 __B) {
   // CHECK-LABEL: @test_mm_maskz_cvtbiasps_hf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8128(<4 x i32> %{{.*}}, <4 x float> %{{.*}})
   // CHECK: zeroinitializer
-  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8128(<4 x i32> %{{.*}}, <4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm_maskz_cvtbiasps_hf8(__U, __A, __B);
 }
 
 __m128i test_mm256_cvtbiasps_hf8(__m256i __A, __m256 __B) {
   // CHECK-LABEL: @test_mm256_cvtbiasps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8256(<8 x i32> %{{.*}}, <8 x float> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8256(<8 x i32> %{{.*}}, <8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 -1)
   return _mm256_cvtbiasps_hf8(__A, __B);
 }
 
@@ -527,9 +510,8 @@ __m128i test_mm256_mask_cvtbiasps_hf8(__m128i __W, __mmask8 __U, __m256i __A, __
 
 __m128i test_mm256_maskz_cvtbiasps_hf8(__mmask8 __U, __m256i __A, __m256 __B) {
   // CHECK-LABEL: @test_mm256_maskz_cvtbiasps_hf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8256(<8 x i32> %{{.*}}, <8 x float> %{{.*}})
   // CHECK: zeroinitializer
-  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8256(<8 x i32> %{{.*}}, <8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm256_maskz_cvtbiasps_hf8(__U, __A, __B);
 }
 
@@ -556,7 +538,7 @@ __m128i test_mm512_maskz_cvtbiasps_hf8(__mmask16 __U, __m512i __A, __m512 __B) {
 
 __m128i test_mm_cvts_biasps_hf8(__m128i __A, __m128 __B) {
   // CHECK-LABEL: @test_mm_cvts_biasps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s128(<4 x i32> %{{.*}}, <4 x float> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s128(<4 x i32> %{{.*}}, <4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 -1)
   return _mm_cvts_biasps_hf8(__A, __B);
 }
 
@@ -568,15 +550,14 @@ __m128i test_mm_mask_cvts_biasps_hf8(__m128i __W, __mmask8 __U, __m128i __A, __m
 
 __m128i test_mm_maskz_cvts_biasps_hf8(__mmask8 __U, __m128i __A, __m128 __B) {
   // CHECK-LABEL: @test_mm_maskz_cvts_biasps_hf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s128(<4 x i32> %{{.*}}, <4 x float> %{{.*}})
   // CHECK: zeroinitializer
-  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s128(<4 x i32> %{{.*}}, <4 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm_maskz_cvts_biasps_hf8(__U, __A, __B);
 }
 
 __m128i test_mm256_cvts_biasps_hf8(__m256i __A, __m256 __B) {
   // CHECK-LABEL: @test_mm256_cvts_biasps_hf8(
-  // CHECK: call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s256(<8 x i32> %{{.*}}, <8 x float> %{{.*}})
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s256(<8 x i32> %{{.*}}, <8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 -1)
   return _mm256_cvts_biasps_hf8(__A, __B);
 }
 
@@ -588,9 +569,8 @@ __m128i test_mm256_mask_cvts_biasps_hf8(__m128i __W, __mmask8 __U, __m256i __A,
 
 __m128i test_mm256_maskz_cvts_biasps_hf8(__mmask8 __U, __m256i __A, __m256 __B) {
   // CHECK-LABEL: @test_mm256_maskz_cvts_biasps_hf8(
-  // CHECK: [[RES:%.*]] = call <16 x i8> @llvm.x86.avx10.vcvtbiasps2hf8s256(<8 x i32> %{{.*}}, <8 x float> %{{.*}})
   // CHECK: zeroinitializer
-  // CHECK: select <16 x i1> %{{.*}}, <16 x i8> [[RES]], <16 x i8> %{{.*}}
+  // CHECK: call <16 x i8> @llvm.x86.avx10.mask.vcvtbiasps2hf8s256(<8 x i32> %{{.*}}, <8 x float> %{{.*}}, <16 x i8> %{{.*}}, i8 %{{.*}})
   return _mm256_maskz_cvts_biasps_hf8(__U, __A, __B);
 }
 
diff --git a/llvm/docs/ReleaseNotes.md b/llvm/docs/ReleaseNotes.md
index ef90a1f1f1c41d..d024bb2245dbb8 100644
--- a/llvm/docs/ReleaseNotes.md
+++ b/llvm/docs/ReleaseNotes.md
@@ -227,6 +227,9 @@ Makes programs 10x faster by doing Special New Thing.
 
 ### Changes to the X86 Backend
 
+* Added assembler and code generation support for the `AVX10_V2_AUX`
+  instruction set.
+
 ### Changes to the OCaml bindings
 
 ### Changes to the Python bindings
diff --git a/llvm/include/llvm/IR/IntrinsicsX86.td b/llvm/include/llvm/IR/IntrinsicsX86.td
index 35d82cb6b26f14..a315b234a39cfb 100644
--- a/llvm/include/llvm/IR/IntrinsicsX86.td
+++ b/llvm/include/llvm/IR/IntrinsicsX86.td
@@ -7052,9 +7052,9 @@ let TargetPrefix = "x86" in {
 // Convert from FP32 to FP8
 
 // VCVTPS2BF8
-def int_x86_avx10_vcvtps2bf8128 : ClangBuiltin<"__builtin_ia32_vcvtps2bf8_128">,
+def int_x86_avx10_vcvtps2bf8128 :
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v4f32_ty], [IntrNoMem]>;
-def int_x86_avx10_vcvtps2bf8256 : ClangBuiltin<"__builtin_ia32_vcvtps2bf8_256">,
+def int_x86_avx10_vcvtps2bf8256 :
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v8f32_ty], [IntrNoMem]>;
 def int_x86_avx10_vcvtps2bf8512 : ClangBuiltin<"__builtin_ia32_vcvtps2bf8_512">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16f32_ty], [IntrNoMem]>;
@@ -7070,9 +7070,9 @@ def int_x86_avx10_mask_vcvtps2bf8256 :
                               [IntrNoMem]>;
 
 // VCVTPS2BF8S
-def int_x86_avx10_vcvtps2bf8s128 : ClangBuiltin<"__builtin_ia32_vcvtps2bf8s_128">,
+def int_x86_avx10_vcvtps2bf8s128 :
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v4f32_ty], [IntrNoMem]>;
-def int_x86_avx10_vcvtps2bf8s256 : ClangBuiltin<"__builtin_ia32_vcvtps2bf8s_256">,
+def int_x86_avx10_vcvtps2bf8s256 :
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v8f32_ty], [IntrNoMem]>;
 def int_x86_avx10_vcvtps2bf8s512 : ClangBuiltin<"__builtin_ia32_vcvtps2bf8s_512">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16f32_ty], [IntrNoMem]>;
@@ -7088,9 +7088,9 @@ def int_x86_avx10_mask_vcvtps2bf8s256 :
                               [IntrNoMem]>;
 
 // VCVTPS2HF8
-def int_x86_avx10_vcvtps2hf8128 : ClangBuiltin<"__builtin_ia32_vcvtps2hf8_128">,
+def int_x86_avx10_vcvtps2hf8128 :
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v4f32_ty], [IntrNoMem]>;
-def int_x86_avx10_vcvtps2hf8256 : ClangBuiltin<"__builtin_ia32_vcvtps2hf8_256">,
+def int_x86_avx10_vcvtps2hf8256 :
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v8f32_ty], [IntrNoMem]>;
 def int_x86_avx10_vcvtps2hf8512 : ClangBuiltin<"__builtin_ia32_vcvtps2hf8_512">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16f32_ty], [IntrNoMem]>;
@@ -7106,9 +7106,9 @@ def int_x86_avx10_mask_vcvtps2hf8256 :
                               [IntrNoMem]>;
 
 // VCVTPS2HF8S
-def int_x86_avx10_vcvtps2hf8s128 : ClangBuiltin<"__builtin_ia32_vcvtps2hf8s_128">,
+def int_x86_avx10_vcvtps2hf8s128 :
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v4f32_ty], [IntrNoMem]>;
-def int_x86_avx10_vcvtps2hf8s256 : ClangBuiltin<"__builtin_ia32_vcvtps2hf8s_256">,
+def int_x86_avx10_vcvtps2hf8s256 :
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v8f32_ty], [IntrNoMem]>;
 def int_x86_avx10_vcvtps2hf8s512 : ClangBuiltin<"__builtin_ia32_vcvtps2hf8s_512">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16f32_ty], [IntrNoMem]>;
@@ -7124,9 +7124,9 @@ def int_x86_avx10_mask_vcvtps2hf8s256 :
                               [IntrNoMem]>;
 
 // VCVTROPS2HF8
-def int_x86_avx10_vcvtrops2hf8128 : ClangBuiltin<"__builtin_ia32_vcvtrops2hf8_128">,
+def int_x86_avx10_vcvtrops2hf8128 :
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v4f32_ty], [IntrNoMem]>;
-def int_x86_avx10_vcvtrops2hf8256 : ClangBuiltin<"__builtin_ia32_vcvtrops2hf8_256">,
+def int_x86_avx10_vcvtrops2hf8256 :
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v8f32_ty], [IntrNoMem]>;
 def int_x86_avx10_vcvtrops2hf8512 : ClangBuiltin<"__builtin_ia32_vcvtrops2hf8_512">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16f32_ty], [IntrNoMem]>;
@@ -7142,9 +7142,9 @@ def int_x86_avx10_mask_vcvtrops2hf8256 :
                               [IntrNoMem]>;
 
 // VCVTROPS2HF8S
-def int_x86_avx10_vcvtrops2hf8s128 : ClangBuiltin<"__builtin_ia32_vcvtrops2hf8s_128">,
+def int_x86_avx10_vcvtrops2hf8s128 :
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v4f32_ty], [IntrNoMem]>;
-def int_x86_avx10_vcvtrops2hf8s256 : ClangBuiltin<"__builtin_ia32_vcvtrops2hf8s_256">,
+def int_x86_avx10_vcvtrops2hf8s256 :
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v8f32_ty], [IntrNoMem]>;
 def int_x86_avx10_vcvtrops2hf8s512 : ClangBuiltin<"__builtin_ia32_vcvtrops2hf8s_512">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16f32_ty], [IntrNoMem]>;
@@ -7162,10 +7162,12 @@ def int_x86_avx10_mask_vcvtrops2hf8s256 :
 // Convert from FP32 to FP8 with bias
 
 // VCVTBIASPS2BF8
-def int_x86_avx10_vcvtbiasps2bf8128 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2bf8_128">,
-        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v4i32_ty, llvm_v4f32_ty], [IntrNoMem]>;
-def int_x86_avx10_vcvtbiasps2bf8256 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2bf8_256">,
-        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v8i32_ty, llvm_v8f32_ty], [IntrNoMem]>;
+def int_x86_avx10_vcvtbiasps2bf8128 :
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v4i32_ty, llvm_v4f32_ty],
+                              [IntrNoMem]>;
+def int_x86_avx10_vcvtbiasps2bf8256 :
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v8i32_ty, llvm_v8f32_ty],
+                              [IntrNoMem]>;
 def int_x86_avx10_vcvtbiasps2bf8512 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2bf8_512">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16i32_ty, llvm_v16f32_ty], [IntrNoMem]>;
 def int_x86_avx10_mask_vcvtbiasps2bf8128 :
@@ -7182,10 +7184,12 @@ def int_x86_avx10_mask_vcvtbiasps2bf8256 :
                               [IntrNoMem]>;
 
 // VCVTBIASPS2BF8S
-def int_x86_avx10_vcvtbiasps2bf8s128 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2bf8s_128">,
-        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v4i32_ty, llvm_v4f32_ty], [IntrNoMem]>;
-def int_x86_avx10_vcvtbiasps2bf8s256 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2bf8s_256">,
-        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v8i32_ty, llvm_v8f32_ty], [IntrNoMem]>;
+def int_x86_avx10_vcvtbiasps2bf8s128 :
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v4i32_ty, llvm_v4f32_ty],
+                              [IntrNoMem]>;
+def int_x86_avx10_vcvtbiasps2bf8s256 :
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v8i32_ty, llvm_v8f32_ty],
+                              [IntrNoMem]>;
 def int_x86_avx10_vcvtbiasps2bf8s512 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2bf8s_512">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16i32_ty, llvm_v16f32_ty], [IntrNoMem]>;
 def int_x86_avx10_mask_vcvtbiasps2bf8s128 :
@@ -7202,10 +7206,12 @@ def int_x86_avx10_mask_vcvtbiasps2bf8s256 :
                               [IntrNoMem]>;
 
 // VCVTBIASPS2HF8
-def int_x86_avx10_vcvtbiasps2hf8128 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2hf8_128">,
-        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v4i32_ty, llvm_v4f32_ty], [IntrNoMem]>;
-def int_x86_avx10_vcvtbiasps2hf8256 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2hf8_256">,
-        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v8i32_ty, llvm_v8f32_ty], [IntrNoMem]>;
+def int_x86_avx10_vcvtbiasps2hf8128 :
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v4i32_ty, llvm_v4f32_ty],
+                              [IntrNoMem]>;
+def int_x86_avx10_vcvtbiasps2hf8256 :
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v8i32_ty, llvm_v8f32_ty],
+                              [IntrNoMem]>;
 def int_x86_avx10_vcvtbiasps2hf8512 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2hf8_512">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16i32_ty, llvm_v16f32_ty], [IntrNoMem]>;
 def int_x86_avx10_mask_vcvtbiasps2hf8128 :
@@ -7222,10 +7228,12 @@ def int_x86_avx10_mask_vcvtbiasps2hf8256 :
                               [IntrNoMem]>;
 
 // VCVTBIASPS2HF8S
-def int_x86_avx10_vcvtbiasps2hf8s128 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2hf8s_128">,
-        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v4i32_ty, llvm_v4f32_ty], [IntrNoMem]>;
-def int_x86_avx10_vcvtbiasps2hf8s256 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2hf8s_256">,
-        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v8i32_ty, llvm_v8f32_ty], [IntrNoMem]>;
+def int_x86_avx10_vcvtbiasps2hf8s128 :
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v4i32_ty, llvm_v4f32_ty],
+                              [IntrNoMem]>;
+def int_x86_avx10_vcvtbiasps2hf8s256 :
+        DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v8i32_ty, llvm_v8f32_ty],
+                              [IntrNoMem]>;
 def int_x86_avx10_vcvtbiasps2hf8s512 : ClangBuiltin<"__builtin_ia32_vcvtbiasps2hf8s_512">,
         DefaultAttrsIntrinsic<[llvm_v16i8_ty], [llvm_v16i32_ty, llvm_v16f32_ty], [IntrNoMem]>;
 def int_x86_avx10_mask_vcvtbiasps2hf8s128 :
diff --git a/llvm/test/CodeGen/X86/avx10_v2aux-pmovssdb-intrinsics.ll b/llvm/test/CodeGen/X86/avx10_v2aux-pmovssdb-intrinsics.ll
index 85e5d2e4439a4a..b49d25e85ce85b 100644
--- a/llvm/test/CodeGen/X86/avx10_v2aux-pmovssdb-intrinsics.ll
+++ b/llvm/test/CodeGen/X86/avx10_v2aux-pmovssdb-intrinsics.ll
@@ -126,6 +126,7 @@ define <16 x i8> @test_int_x86_avx10_mask_pmovssdb_512(<16 x i32> %a, <16 x i8>
   ret <16 x i8> %add2
 }
 
+; Negative tests: VPMOVSSDB source loads must not fold.
 define <16 x i8> @test_int_x86_avx10_pmovssdb_mem_128(ptr %ptr_a) {
 ; X64-LABEL: test_int_x86_avx10_pmovssdb_mem_128:
 ; X64:       # %bb.0:
diff --git a/llvm/test/MC/X86/avx10_v2_aux-att-32.s b/llvm/test/MC/X86/avx10_v2_aux-att-32.s
index ba16889de21591..9a0a3aca1904c1 100644
--- a/llvm/test/MC/X86/avx10_v2_aux-att-32.s
+++ b/llvm/test/MC/X86/avx10_v2_aux-att-32.s
@@ -24,6 +24,43 @@
 // CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x39,0x07]
           vcvtps2bf8x (%edi), %xmm0
 
+// CHECK: vcvtps2bf8z 268435456(%esp,%esi,8), %xmm0 {%k1}
+// CHECK: encoding: [0x62,0xf5,0x7e,0x49,0x39,0x84,0xf4,0x00,0x00,0x00,0x10]
+          vcvtps2bf8z 268435456(%esp,%esi,8), %xmm0 {%k1}
+
+// CHECK: vcvtps2bf8y 268435456(%esp,%esi,8), %xmm0 {%k1}
+// CHECK: encoding: [0x62,0xf5,0x7e,0x29,0x39,0x84,0xf4,0x00,0x00,0x00,0x10]
+          vcvtps2bf8y 268435456(%esp,%esi,8), %xmm0 {%k1}
+
+// CHECK: vcvtps2bf8x 268435456(%esp,%esi,8), %xmm0 {%k1}
+// CHECK: encoding: [0x62,0xf5,0x7e,0x09,0x39,0x84,0xf4,0x00,0x00,0x00,0x10]
+          vcvtps2bf8x 268435456(%esp,%esi,8), %xmm0 {%k1}
+
+// CHECK: vcvtps2bf8z 8128(%ecx), %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x39,0x41,0x7f]
+          vcvtps2bf8z 8128(%ecx), %xmm0
+
+// CHECK: vcvtps2bf8y 4064(%ecx), %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x28,0x39,0x41,0x7f]
+          vcvtps2bf8y 4064(%ecx), %xmm0
+
+// CHECK: vcvtps2bf8x 2032(%ecx), %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x39,0x41,0x7f]
+          vcvtps2bf8x 2032(%ecx), %xmm0
+
+// CHECK: vcvtps2bf8 -512(%edx){1to16}, %xmm0 {%k1} {z}
+// CHECK: encoding: [0x62,0xf5,0x7e,0xd9,0x39,0x42,0x80]
+          vcvtps2bf8 -512(%edx){1to16}, %xmm0 {%k1} {z}
+
+// CHECK: vcvtps2bf8 -512(%edx){1to8}, %xmm0 {%k1} {z}
+// CHECK: encoding: [0x62,0xf5,0x7e,0xb9,0x39,0x42,0x80]
+          vcvtps2bf8 -512(%edx){1to8}, %xmm0 {%k1} {z}
+
+// CHECK: vcvtps2bf8 -512(%edx){1to4}, %xmm0 {%k1} {z}
+// CHECK: encoding: [0x62,0xf5,0x7e,0x99,0x39,0x42,0x80]
+          vcvtps2bf8 -512(%edx){1to4}, %xmm0 {%k1} {z}
+
+
 // CHECK: vcvtps2bf8 %xmm1, %xmm0 {%k1}
 // CHECK: encoding: [0x62,0xf5,0x7e,0x09,0x39,0xc1]
           vcvtps2bf8 %xmm1, %xmm0 {%k1}
@@ -328,6 +365,43 @@
 // CHECK: encoding: [0x62,0xf5,0x74,0x08,0x39,0x07]
           vcvtbiasps2bf8 (%edi), %xmm1, %xmm0
 
+// CHECK: vcvtbiasps2bf8 268435456(%esp,%esi,8), %zmm1, %xmm0 {%k1}
+// CHECK: encoding: [0x62,0xf5,0x74,0x49,0x39,0x84,0xf4,0x00,0x00,0x00,0x10]
+          vcvtbiasps2bf8 268435456(%esp,%esi,8), %zmm1, %xmm0 {%k1}
+
+// CHECK: vcvtbiasps2bf8 268435456(%esp,%esi,8), %ymm1, %xmm0 {%k1}
+// CHECK: encoding: [0x62,0xf5,0x74,0x29,0x39,0x84,0xf4,0x00,0x00,0x00,0x10]
+          vcvtbiasps2bf8 268435456(%esp,%esi,8), %ymm1, %xmm0 {%k1}
+
+// CHECK: vcvtbiasps2bf8 268435456(%esp,%esi,8), %xmm1, %xmm0 {%k1}
+// CHECK: encoding: [0x62,0xf5,0x74,0x09,0x39,0x84,0xf4,0x00,0x00,0x00,0x10]
+          vcvtbiasps2bf8 268435456(%esp,%esi,8), %xmm1, %xmm0 {%k1}
+
+// CHECK: vcvtbiasps2bf8 8128(%ecx), %zmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x48,0x39,0x41,0x7f]
+          vcvtbiasps2bf8 8128(%ecx), %zmm1, %xmm0
+
+// CHECK: vcvtbiasps2bf8 4064(%ecx), %ymm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x28,0x39,0x41,0x7f]
+          vcvtbiasps2bf8 4064(%ecx), %ymm1, %xmm0
+
+// CHECK: vcvtbiasps2bf8 2032(%ecx), %xmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x08,0x39,0x41,0x7f]
+          vcvtbiasps2bf8 2032(%ecx), %xmm1, %xmm0
+
+// CHECK: vcvtbiasps2bf8 -512(%edx){1to16}, %zmm1, %xmm0 {%k1} {z}
+// CHECK: encoding: [0x62,0xf5,0x74,0xd9,0x39,0x42,0x80]
+          vcvtbiasps2bf8 -512(%edx){1to16}, %zmm1, %xmm0 {%k1} {z}
+
+// CHECK: vcvtbiasps2bf8 -512(%edx){1to8}, %ymm1, %xmm0 {%k1} {z}
+// CHECK: encoding: [0x62,0xf5,0x74,0xb9,0x39,0x42,0x80]
+          vcvtbiasps2bf8 -512(%edx){1to8}, %ymm1, %xmm0 {%k1} {z}
+
+// CHECK: vcvtbiasps2bf8 -512(%edx){1to4}, %xmm1, %xmm0 {%k1} {z}
+// CHECK: encoding: [0x62,0xf5,0x74,0x99,0x39,0x42,0x80]
+          vcvtbiasps2bf8 -512(%edx){1to4}, %xmm1, %xmm0 {%k1} {z}
+
+
 // CHECK: vcvtbiasps2bf8 (%edi){1to16}, %zmm1, %xmm0
 // CHECK: encoding: [0x62,0xf5,0x74,0x58,0x39,0x07]
           vcvtbiasps2bf8 (%edi){1to16}, %zmm1, %xmm0
diff --git a/llvm/test/MC/X86/avx10_v2_aux-att-64.s b/llvm/test/MC/X86/avx10_v2_aux-att-64.s
index 956f74912938fa..a3ced1d0e977fd 100644
--- a/llvm/test/MC/X86/avx10_v2_aux-att-64.s
+++ b/llvm/test/MC/X86/avx10_v2_aux-att-64.s
@@ -24,6 +24,43 @@
 // CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x39,0x07]
           vcvtps2bf8x (%rdi), %xmm0
 
+// CHECK: vcvtps2bf8z 268435456(%rbp,%r14,8), %xmm0 {%k1}
+// CHECK: encoding: [0x62,0xb5,0x7e,0x49,0x39,0x84,0xf5,0x00,0x00,0x00,0x10]
+          vcvtps2bf8z 268435456(%rbp,%r14,8), %xmm0 {%k1}
+
+// CHECK: vcvtps2bf8y 268435456(%rbp,%r14,8), %xmm0 {%k1}
+// CHECK: encoding: [0x62,0xb5,0x7e,0x29,0x39,0x84,0xf5,0x00,0x00,0x00,0x10]
+          vcvtps2bf8y 268435456(%rbp,%r14,8), %xmm0 {%k1}
+
+// CHECK: vcvtps2bf8x 268435456(%rbp,%r14,8), %xmm0 {%k1}
+// CHECK: encoding: [0x62,0xb5,0x7e,0x09,0x39,0x84,0xf5,0x00,0x00,0x00,0x10]
+          vcvtps2bf8x 268435456(%rbp,%r14,8), %xmm0 {%k1}
+
+// CHECK: vcvtps2bf8z 8128(%rcx), %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x39,0x41,0x7f]
+          vcvtps2bf8z 8128(%rcx), %xmm0
+
+// CHECK: vcvtps2bf8y 4064(%rcx), %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x28,0x39,0x41,0x7f]
+          vcvtps2bf8y 4064(%rcx), %xmm0
+
+// CHECK: vcvtps2bf8x 2032(%rcx), %xmm0
+// CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x39,0x41,0x7f]
+          vcvtps2bf8x 2032(%rcx), %xmm0
+
+// CHECK: vcvtps2bf8 -512(%rdx){1to16}, %xmm0 {%k1} {z}
+// CHECK: encoding: [0x62,0xf5,0x7e,0xd9,0x39,0x42,0x80]
+          vcvtps2bf8 -512(%rdx){1to16}, %xmm0 {%k1} {z}
+
+// CHECK: vcvtps2bf8 -512(%rdx){1to8}, %xmm0 {%k1} {z}
+// CHECK: encoding: [0x62,0xf5,0x7e,0xb9,0x39,0x42,0x80]
+          vcvtps2bf8 -512(%rdx){1to8}, %xmm0 {%k1} {z}
+
+// CHECK: vcvtps2bf8 -512(%rdx){1to4}, %xmm0 {%k1} {z}
+// CHECK: encoding: [0x62,0xf5,0x7e,0x99,0x39,0x42,0x80]
+          vcvtps2bf8 -512(%rdx){1to4}, %xmm0 {%k1} {z}
+
+
 // CHECK: vcvtps2bf8 %xmm1, %xmm0 {%k1}
 // CHECK: encoding: [0x62,0xf5,0x7e,0x09,0x39,0xc1]
           vcvtps2bf8 %xmm1, %xmm0 {%k1}
@@ -320,6 +357,43 @@
 // CHECK: encoding: [0x62,0xf5,0x74,0x08,0x39,0x07]
           vcvtbiasps2bf8 (%rdi), %xmm1, %xmm0
 
+// CHECK: vcvtbiasps2bf8 268435456(%rbp,%r14,8), %zmm1, %xmm0 {%k1}
+// CHECK: encoding: [0x62,0xb5,0x74,0x49,0x39,0x84,0xf5,0x00,0x00,0x00,0x10]
+          vcvtbiasps2bf8 268435456(%rbp,%r14,8), %zmm1, %xmm0 {%k1}
+
+// CHECK: vcvtbiasps2bf8 268435456(%rbp,%r14,8), %ymm1, %xmm0 {%k1}
+// CHECK: encoding: [0x62,0xb5,0x74,0x29,0x39,0x84,0xf5,0x00,0x00,0x00,0x10]
+          vcvtbiasps2bf8 268435456(%rbp,%r14,8), %ymm1, %xmm0 {%k1}
+
+// CHECK: vcvtbiasps2bf8 268435456(%rbp,%r14,8), %xmm1, %xmm0 {%k1}
+// CHECK: encoding: [0x62,0xb5,0x74,0x09,0x39,0x84,0xf5,0x00,0x00,0x00,0x10]
+          vcvtbiasps2bf8 268435456(%rbp,%r14,8), %xmm1, %xmm0 {%k1}
+
+// CHECK: vcvtbiasps2bf8 8128(%rcx), %zmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x48,0x39,0x41,0x7f]
+          vcvtbiasps2bf8 8128(%rcx), %zmm1, %xmm0
+
+// CHECK: vcvtbiasps2bf8 4064(%rcx), %ymm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x28,0x39,0x41,0x7f]
+          vcvtbiasps2bf8 4064(%rcx), %ymm1, %xmm0
+
+// CHECK: vcvtbiasps2bf8 2032(%rcx), %xmm1, %xmm0
+// CHECK: encoding: [0x62,0xf5,0x74,0x08,0x39,0x41,0x7f]
+          vcvtbiasps2bf8 2032(%rcx), %xmm1, %xmm0
+
+// CHECK: vcvtbiasps2bf8 -512(%rdx){1to16}, %zmm1, %xmm0 {%k1} {z}
+// CHECK: encoding: [0x62,0xf5,0x74,0xd9,0x39,0x42,0x80]
+          vcvtbiasps2bf8 -512(%rdx){1to16}, %zmm1, %xmm0 {%k1} {z}
+
+// CHECK: vcvtbiasps2bf8 -512(%rdx){1to8}, %ymm1, %xmm0 {%k1} {z}
+// CHECK: encoding: [0x62,0xf5,0x74,0xb9,0x39,0x42,0x80]
+          vcvtbiasps2bf8 -512(%rdx){1to8}, %ymm1, %xmm0 {%k1} {z}
+
+// CHECK: vcvtbiasps2bf8 -512(%rdx){1to4}, %xmm1, %xmm0 {%k1} {z}
+// CHECK: encoding: [0x62,0xf5,0x74,0x99,0x39,0x42,0x80]
+          vcvtbiasps2bf8 -512(%rdx){1to4}, %xmm1, %xmm0 {%k1} {z}
+
+
 // CHECK: vcvtbiasps2bf8 %zmm2, %zmm1, %xmm0 {%k1}
 // CHECK: encoding: [0x62,0xf5,0x74,0x49,0x39,0xc2]
           vcvtbiasps2bf8 %zmm2, %zmm1, %xmm0 {%k1}
diff --git a/llvm/test/MC/X86/avx10_v2_aux-intel-32.s b/llvm/test/MC/X86/avx10_v2_aux-intel-32.s
index 54893e196d1c91..034c1536a9f4a6 100644
--- a/llvm/test/MC/X86/avx10_v2_aux-intel-32.s
+++ b/llvm/test/MC/X86/avx10_v2_aux-intel-32.s
@@ -32,6 +32,43 @@
 // CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x39,0x07]
           vcvtps2bf8 xmm0, xmmword ptr [edi]
 
+// CHECK: vcvtps2bf8 xmm0 {k1}, zmmword ptr [esp + 8*esi + 268435456]
+// CHECK: encoding: [0x62,0xf5,0x7e,0x49,0x39,0x84,0xf4,0x00,0x00,0x00,0x10]
+          vcvtps2bf8 xmm0 {k1}, zmmword ptr [esp + 8*esi + 268435456]
+
+// CHECK: vcvtps2bf8 xmm0 {k1}, ymmword ptr [esp + 8*esi + 268435456]
+// CHECK: encoding: [0x62,0xf5,0x7e,0x29,0x39,0x84,0xf4,0x00,0x00,0x00,0x10]
+          vcvtps2bf8 xmm0 {k1}, ymmword ptr [esp + 8*esi + 268435456]
+
+// CHECK: vcvtps2bf8 xmm0 {k1}, xmmword ptr [esp + 8*esi + 268435456]
+// CHECK: encoding: [0x62,0xf5,0x7e,0x09,0x39,0x84,0xf4,0x00,0x00,0x00,0x10]
+          vcvtps2bf8 xmm0 {k1}, xmmword ptr [esp + 8*esi + 268435456]
+
+// CHECK: vcvtps2bf8 xmm0, zmmword ptr [ecx + 8128]
+// CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x39,0x41,0x7f]
+          vcvtps2bf8 xmm0, zmmword ptr [ecx + 8128]
+
+// CHECK: vcvtps2bf8 xmm0, ymmword ptr [ecx + 4064]
+// CHECK: encoding: [0x62,0xf5,0x7e,0x28,0x39,0x41,0x7f]
+          vcvtps2bf8 xmm0, ymmword ptr [ecx + 4064]
+
+// CHECK: vcvtps2bf8 xmm0, xmmword ptr [ecx + 2032]
+// CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x39,0x41,0x7f]
+          vcvtps2bf8 xmm0, xmmword ptr [ecx + 2032]
+
+// CHECK: vcvtps2bf8 xmm0 {k1} {z}, dword ptr [edx - 512]{1to16}
+// CHECK: encoding: [0x62,0xf5,0x7e,0xd9,0x39,0x42,0x80]
+          vcvtps2bf8 xmm0 {k1} {z}, dword ptr [edx - 512]{1to16}
+
+// CHECK: vcvtps2bf8 xmm0 {k1} {z}, dword ptr [edx - 512]{1to8}
+// CHECK: encoding: [0x62,0xf5,0x7e,0xb9,0x39,0x42,0x80]
+          vcvtps2bf8 xmm0 {k1} {z}, dword ptr [edx - 512]{1to8}
+
+// CHECK: vcvtps2bf8 xmm0 {k1} {z}, dword ptr [edx - 512]{1to4}
+// CHECK: encoding: [0x62,0xf5,0x7e,0x99,0x39,0x42,0x80]
+          vcvtps2bf8 xmm0 {k1} {z}, dword ptr [edx - 512]{1to4}
+
+
 // CHECK: vcvtps2bf8 xmm0, dword ptr [edi]{1to16}
 // CHECK: encoding: [0x62,0xf5,0x7e,0x58,0x39,0x07]
           vcvtps2bf8 xmm0, dword ptr [edi]{1to16}
@@ -296,6 +333,43 @@
 // CHECK: encoding: [0x62,0xf5,0x74,0x08,0x39,0x07]
           vcvtbiasps2bf8 xmm0, xmm1, xmmword ptr [edi]
 
+// CHECK: vcvtbiasps2bf8 xmm0 {k1}, zmm1, zmmword ptr [esp + 8*esi + 268435456]
+// CHECK: encoding: [0x62,0xf5,0x74,0x49,0x39,0x84,0xf4,0x00,0x00,0x00,0x10]
+          vcvtbiasps2bf8 xmm0 {k1}, zmm1, zmmword ptr [esp + 8*esi + 268435456]
+
+// CHECK: vcvtbiasps2bf8 xmm0 {k1}, ymm1, ymmword ptr [esp + 8*esi + 268435456]
+// CHECK: encoding: [0x62,0xf5,0x74,0x29,0x39,0x84,0xf4,0x00,0x00,0x00,0x10]
+          vcvtbiasps2bf8 xmm0 {k1}, ymm1, ymmword ptr [esp + 8*esi + 268435456]
+
+// CHECK: vcvtbiasps2bf8 xmm0 {k1}, xmm1, xmmword ptr [esp + 8*esi + 268435456]
+// CHECK: encoding: [0x62,0xf5,0x74,0x09,0x39,0x84,0xf4,0x00,0x00,0x00,0x10]
+          vcvtbiasps2bf8 xmm0 {k1}, xmm1, xmmword ptr [esp + 8*esi + 268435456]
+
+// CHECK: vcvtbiasps2bf8 xmm0, zmm1, zmmword ptr [ecx + 8128]
+// CHECK: encoding: [0x62,0xf5,0x74,0x48,0x39,0x41,0x7f]
+          vcvtbiasps2bf8 xmm0, zmm1, zmmword ptr [ecx + 8128]
+
+// CHECK: vcvtbiasps2bf8 xmm0, ymm1, ymmword ptr [ecx + 4064]
+// CHECK: encoding: [0x62,0xf5,0x74,0x28,0x39,0x41,0x7f]
+          vcvtbiasps2bf8 xmm0, ymm1, ymmword ptr [ecx + 4064]
+
+// CHECK: vcvtbiasps2bf8 xmm0, xmm1, xmmword ptr [ecx + 2032]
+// CHECK: encoding: [0x62,0xf5,0x74,0x08,0x39,0x41,0x7f]
+          vcvtbiasps2bf8 xmm0, xmm1, xmmword ptr [ecx + 2032]
+
+// CHECK: vcvtbiasps2bf8 xmm0 {k1} {z}, zmm1, dword ptr [edx - 512]{1to16}
+// CHECK: encoding: [0x62,0xf5,0x74,0xd9,0x39,0x42,0x80]
+          vcvtbiasps2bf8 xmm0 {k1} {z}, zmm1, dword ptr [edx - 512]{1to16}
+
+// CHECK: vcvtbiasps2bf8 xmm0 {k1} {z}, ymm1, dword ptr [edx - 512]{1to8}
+// CHECK: encoding: [0x62,0xf5,0x74,0xb9,0x39,0x42,0x80]
+          vcvtbiasps2bf8 xmm0 {k1} {z}, ymm1, dword ptr [edx - 512]{1to8}
+
+// CHECK: vcvtbiasps2bf8 xmm0 {k1} {z}, xmm1, dword ptr [edx - 512]{1to4}
+// CHECK: encoding: [0x62,0xf5,0x74,0x99,0x39,0x42,0x80]
+          vcvtbiasps2bf8 xmm0 {k1} {z}, xmm1, dword ptr [edx - 512]{1to4}
+
+
 // CHECK: vcvtbiasps2bf8 xmm0, zmm1, dword ptr [edi]{1to16}
 // CHECK: encoding: [0x62,0xf5,0x74,0x58,0x39,0x07]
           vcvtbiasps2bf8 xmm0, zmm1, dword ptr [edi]{1to16}
diff --git a/llvm/test/MC/X86/avx10_v2_aux-intel-64.s b/llvm/test/MC/X86/avx10_v2_aux-intel-64.s
index 204a374fd3b8bb..eb828b22c4496e 100644
--- a/llvm/test/MC/X86/avx10_v2_aux-intel-64.s
+++ b/llvm/test/MC/X86/avx10_v2_aux-intel-64.s
@@ -24,6 +24,43 @@
 // CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x39,0x07]
           vcvtps2bf8 xmm0, xmmword ptr [rdi]
 
+// CHECK: vcvtps2bf8 xmm0 {k1}, zmmword ptr [rbp + 8*r14 + 268435456]
+// CHECK: encoding: [0x62,0xb5,0x7e,0x49,0x39,0x84,0xf5,0x00,0x00,0x00,0x10]
+          vcvtps2bf8 xmm0 {k1}, zmmword ptr [rbp + 8*r14 + 268435456]
+
+// CHECK: vcvtps2bf8 xmm0 {k1}, ymmword ptr [rbp + 8*r14 + 268435456]
+// CHECK: encoding: [0x62,0xb5,0x7e,0x29,0x39,0x84,0xf5,0x00,0x00,0x00,0x10]
+          vcvtps2bf8 xmm0 {k1}, ymmword ptr [rbp + 8*r14 + 268435456]
+
+// CHECK: vcvtps2bf8 xmm0 {k1}, xmmword ptr [rbp + 8*r14 + 268435456]
+// CHECK: encoding: [0x62,0xb5,0x7e,0x09,0x39,0x84,0xf5,0x00,0x00,0x00,0x10]
+          vcvtps2bf8 xmm0 {k1}, xmmword ptr [rbp + 8*r14 + 268435456]
+
+// CHECK: vcvtps2bf8 xmm0, zmmword ptr [rcx + 8128]
+// CHECK: encoding: [0x62,0xf5,0x7e,0x48,0x39,0x41,0x7f]
+          vcvtps2bf8 xmm0, zmmword ptr [rcx + 8128]
+
+// CHECK: vcvtps2bf8 xmm0, ymmword ptr [rcx + 4064]
+// CHECK: encoding: [0x62,0xf5,0x7e,0x28,0x39,0x41,0x7f]
+          vcvtps2bf8 xmm0, ymmword ptr [rcx + 4064]
+
+// CHECK: vcvtps2bf8 xmm0, xmmword ptr [rcx + 2032]
+// CHECK: encoding: [0x62,0xf5,0x7e,0x08,0x39,0x41,0x7f]
+          vcvtps2bf8 xmm0, xmmword ptr [rcx + 2032]
+
+// CHECK: vcvtps2bf8 xmm0 {k1} {z}, dword ptr [rdx - 512]{1to16}
+// CHECK: encoding: [0x62,0xf5,0x7e,0xd9,0x39,0x42,0x80]
+          vcvtps2bf8 xmm0 {k1} {z}, dword ptr [rdx - 512]{1to16}
+
+// CHECK: vcvtps2bf8 xmm0 {k1} {z}, dword ptr [rdx - 512]{1to8}
+// CHECK: encoding: [0x62,0xf5,0x7e,0xb9,0x39,0x42,0x80]
+          vcvtps2bf8 xmm0 {k1} {z}, dword ptr [rdx - 512]{1to8}
+
+// CHECK: vcvtps2bf8 xmm0 {k1} {z}, dword ptr [rdx - 512]{1to4}
+// CHECK: encoding: [0x62,0xf5,0x7e,0x99,0x39,0x42,0x80]
+          vcvtps2bf8 xmm0 {k1} {z}, dword ptr [rdx - 512]{1to4}
+
+
 // CHECK: vcvtps2bf8 xmm0 {k1}, zmm1
 // CHECK: encoding: [0x62,0xf5,0x7e,0x49,0x39,0xc1]
           vcvtps2bf8 xmm0 {k1}, zmm1
@@ -288,6 +325,43 @@
 // CHECK: encoding: [0x62,0xf5,0x74,0x08,0x39,0x07]
           vcvtbiasps2bf8 xmm0, xmm1, xmmword ptr [rdi]
 
+// CHECK: vcvtbiasps2bf8 xmm0 {k1}, zmm1, zmmword ptr [rbp + 8*r14 + 268435456]
+// CHECK: encoding: [0x62,0xb5,0x74,0x49,0x39,0x84,0xf5,0x00,0x00,0x00,0x10]
+          vcvtbiasps2bf8 xmm0 {k1}, zmm1, zmmword ptr [rbp + 8*r14 + 268435456]
+
+// CHECK: vcvtbiasps2bf8 xmm0 {k1}, ymm1, ymmword ptr [rbp + 8*r14 + 268435456]
+// CHECK: encoding: [0x62,0xb5,0x74,0x29,0x39,0x84,0xf5,0x00,0x00,0x00,0x10]
+          vcvtbiasps2bf8 xmm0 {k1}, ymm1, ymmword ptr [rbp + 8*r14 + 268435456]
+
+// CHECK: vcvtbiasps2bf8 xmm0 {k1}, xmm1, xmmword ptr [rbp + 8*r14 + 268435456]
+// CHECK: encoding: [0x62,0xb5,0x74,0x09,0x39,0x84,0xf5,0x00,0x00,0x00,0x10]
+          vcvtbiasps2bf8 xmm0 {k1}, xmm1, xmmword ptr [rbp + 8*r14 + 268435456]
+
+// CHECK: vcvtbiasps2bf8 xmm0, zmm1, zmmword ptr [rcx + 8128]
+// CHECK: encoding: [0x62,0xf5,0x74,0x48,0x39,0x41,0x7f]
+          vcvtbiasps2bf8 xmm0, zmm1, zmmword ptr [rcx + 8128]
+
+// CHECK: vcvtbiasps2bf8 xmm0, ymm1, ymmword ptr [rcx + 4064]
+// CHECK: encoding: [0x62,0xf5,0x74,0x28,0x39,0x41,0x7f]
+          vcvtbiasps2bf8 xmm0, ymm1, ymmword ptr [rcx + 4064]
+
+// CHECK: vcvtbiasps2bf8 xmm0, xmm1, xmmword ptr [rcx + 2032]
+// CHECK: encoding: [0x62,0xf5,0x74,0x08,0x39,0x41,0x7f]
+          vcvtbiasps2bf8 xmm0, xmm1, xmmword ptr [rcx + 2032]
+
+// CHECK: vcvtbiasps2bf8 xmm0 {k1} {z}, zmm1, dword ptr [rdx - 512]{1to16}
+// CHECK: encoding: [0x62,0xf5,0x74,0xd9,0x39,0x42,0x80]
+          vcvtbiasps2bf8 xmm0 {k1} {z}, zmm1, dword ptr [rdx - 512]{1to16}
+
+// CHECK: vcvtbiasps2bf8 xmm0 {k1} {z}, ymm1, dword ptr [rdx - 512]{1to8}
+// CHECK: encoding: [0x62,0xf5,0x74,0xb9,0x39,0x42,0x80]
+          vcvtbiasps2bf8 xmm0 {k1} {z}, ymm1, dword ptr [rdx - 512]{1to8}
+
+// CHECK: vcvtbiasps2bf8 xmm0 {k1} {z}, xmm1, dword ptr [rdx - 512]{1to4}
+// CHECK: encoding: [0x62,0xf5,0x74,0x99,0x39,0x42,0x80]
+          vcvtbiasps2bf8 xmm0 {k1} {z}, xmm1, dword ptr [rdx - 512]{1to4}
+
+
 // CHECK: vcvtbiasps2bf8 xmm0 {k1}, zmm1, zmm2
 // CHECK: encoding: [0x62,0xf5,0x74,0x49,0x39,0xc2]
           vcvtbiasps2bf8 xmm0 {k1}, zmm1, zmm2



More information about the llvm-commits mailing list