[clang] [CIR][AArch64] Handle constrained Neon FMA and sqrt (PR #218307)

Yair Ben Avraham via cfe-commits cfe-commits at lists.llvm.org
Tue Sep 29 03:01:57 PDT 2026


https://github.com/yairbenavraham updated https://github.com/llvm/llvm-project/pull/218307

>From fb3a027ecef08e1740ed35a91d8d1978c1154a74 Mon Sep 17 00:00:00 2001
From: Yair Ben Avraham <yairba at protonmail.com>
Date: Thu, 20 Aug 2026 07:38:56 +0300
Subject: [PATCH 01/16] [CIR][AArch64] Handle constrained Neon FMA and sqrt

Propagate expression FP options through AArch64 builtin emission and
attach the active constrained FP environment to CIR FMA and sqrt
operations.

Unconstrained FMA/sqrt coverage already lives under CodeGen/AArch64/neon.
Add command-line strict coverage there, plus CIR and CIR-to-LLVM coverage
to the existing test where a pragma overrides a command-line maytrap
setting.

Check that lowering does not fall back to generic CIR calls or
unconstrained LLVM intrinsics.

Assisted-by: Codex
Follow-up to #213800
---
 .../lib/CIR/CodeGen/CIRGenBuiltinAArch64.cpp  | 13 +++++-
 clang/test/CodeGen/AArch64/neon/fullfp16.c    | 10 +++++
 .../AArch64/neon/fused-multiple-fullfp16.c    |  2 +-
 .../CodeGen/AArch64/neon/fused-multiply.c     | 18 ++++++++
 clang/test/CodeGen/AArch64/neon/intrinsics.c  |  6 +++
 .../v8.2a-neon-intrinsics-constrained.c       | 41 +++++++++++++++++++
 6 files changed, 88 insertions(+), 2 deletions(-)

diff --git a/clang/lib/CIR/CodeGen/CIRGenBuiltinAArch64.cpp b/clang/lib/CIR/CodeGen/CIRGenBuiltinAArch64.cpp
index 7f504c3f717f2..e71c3e24af2a3 100644
--- a/clang/lib/CIR/CodeGen/CIRGenBuiltinAArch64.cpp
+++ b/clang/lib/CIR/CodeGen/CIRGenBuiltinAArch64.cpp
@@ -204,6 +204,16 @@ emitNeonCallToOp(CIRGenModule &cgm, CIRGenBuilderTy &builder,
                              builder.getStringAttr(intrinsicName.value()),
                              funcResTy, args)
         .getResult();
+  } else if constexpr (std::is_same_v<Operation, cir::FMAOp>) {
+    assert(args.size() == 3 && "fma expects three operands");
+    return Operation::create(builder, loc, funcResTy, args[0], args[1], args[2],
+                             builder.getConstrainedFPAttr())
+        .getResult();
+  } else if constexpr (std::is_same_v<Operation, cir::SqrtOp>) {
+    assert(args.size() == 1 && "sqrt expects one operand");
+    return Operation::create(builder, loc, funcResTy, args[0],
+                             builder.getConstrainedFPAttr())
+        .getResult();
   } else {
     return Operation::create(builder, loc, funcResTy, args).getResult();
   }
@@ -2487,6 +2497,8 @@ CIRGenFunction::emitAArch64BuiltinExpr(unsigned builtinID, const CallExpr *expr,
   // evaluation.
   assert(!cir::MissingFeatures::msvcBuiltins());
 
+  CIRGenFPOptionsRAII fpOptsRAII(*this, expr);
+
   // Some intrinsics are equivalent - if they are use the base intrinsic ID.
   auto it = llvm::find_if(neonEquivalentIntrinsicMap, [builtinID](auto &p) {
     return p.first == builtinID;
@@ -3386,7 +3398,6 @@ CIRGenFunction::emitAArch64BuiltinExpr(unsigned builtinID, const CallExpr *expr,
   }
   case NEON::BI__builtin_neon_vsqrt_v:
   case NEON::BI__builtin_neon_vsqrtq_v:
-    assert(!cir::MissingFeatures::emitConstrainedFPCall());
     return emitNeonCallToOp<cir::SqrtOp>(cgm, builder, {ty}, ops, std::nullopt,
                                          ty, loc);
   case NEON::BI__builtin_neon_vrbit_v:
diff --git a/clang/test/CodeGen/AArch64/neon/fullfp16.c b/clang/test/CodeGen/AArch64/neon/fullfp16.c
index c17518eb813de..ffab85ee93fbb 100644
--- a/clang/test/CodeGen/AArch64/neon/fullfp16.c
+++ b/clang/test/CodeGen/AArch64/neon/fullfp16.c
@@ -3,6 +3,8 @@
 // RUN:                   %clang_cc1_cg_arm64_neon -target-feature +fullfp16           -emit-llvm  %s -disable-O0-optnone | opt -S -passes=mem2reg             | FileCheck %s --check-prefixes=ALL,LLVM
 // RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 -fclangir -emit-llvm  %s -disable-O0-optnone | opt -S -passes=mem2reg,simplifycfg | FileCheck %s --check-prefixes=ALL,LLVM %}
 // RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 -fclangir -emit-cir   %s -disable-O0-optnone |                                      FileCheck %s --check-prefixes=ALL,CIR %}
+// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 -fexperimental-strict-floating-point -ffp-exception-behavior=strict -fclangir -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,simplifycfg | FileCheck %s --check-prefix=LLVM-STRICT --implicit-check-not=' @llvm.fma.' --implicit-check-not=' @llvm.sqrt.' %}
+// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 -fexperimental-strict-floating-point -ffp-exception-behavior=strict -fclangir -emit-cir  %s -disable-O0-optnone |                                      FileCheck %s --check-prefix=CIR-STRICT --implicit-check-not='cir.call_llvm_intrinsic "fma"' --implicit-check-not='cir.call_llvm_intrinsic "sqrt"' %}
 
 //=============================================================================
 // NOTES
@@ -309,6 +311,10 @@ float16_t test_vrndxh_f16(float16_t a) {
 //===------------------------------------------------------===//
 // 2.5.1.4.  Square root
 //===------------------------------------------------------===//
+// LLVM-STRICT-LABEL: @test_vsqrth_f16(
+// LLVM-STRICT: call half @llvm.experimental.constrained.sqrt.f16({{.*}}, metadata !"round.tonearest", metadata !"fpexcept.strict")
+// CIR-STRICT-LABEL: cir.func {{.*}}@test_vsqrth_f16(
+// CIR-STRICT: cir.sqrt %{{.*}} : !cir.f16 {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
 // ALL-LABEL: test_vsqrth_f16
 float16_t test_vsqrth_f16(float16_t a) {
 // CIR:  cir.sqrt
@@ -335,6 +341,10 @@ float16_t test_vnegh_f16(float16_t a) {
 //===------------------------------------------------------===//
 // 2.5.1.9.3 Fused multiply-accumulate
 //===------------------------------------------------------===//
+// LLVM-STRICT-LABEL: @test_vfmah_f16(
+// LLVM-STRICT: call half @llvm.experimental.constrained.fma.f16({{.*}}, metadata !"round.tonearest", metadata !"fpexcept.strict")
+// CIR-STRICT-LABEL: cir.func {{.*}}@test_vfmah_f16(
+// CIR-STRICT: cir.fma %{{.*}}, %{{.*}}, %{{.*}} : !cir.f16 {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
 // ALL-LABEL: test_vfmah_f16
 float16_t test_vfmah_f16(float16_t a, float16_t b, float16_t c) {
 // CIR: cir.fma {{.*}} : !cir.f16
diff --git a/clang/test/CodeGen/AArch64/neon/fused-multiple-fullfp16.c b/clang/test/CodeGen/AArch64/neon/fused-multiple-fullfp16.c
index 7231a1108a94f..30188ad3517b9 100644
--- a/clang/test/CodeGen/AArch64/neon/fused-multiple-fullfp16.c
+++ b/clang/test/CodeGen/AArch64/neon/fused-multiple-fullfp16.c
@@ -2,7 +2,7 @@
 
 // RUN:                   %clang_cc1_cg_arm64_neon -target-feature +fullfp16           -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,sroa | FileCheck %s --check-prefixes=ALL,LLVM
 // RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 -fclangir -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,sroa | FileCheck %s --check-prefixes=ALL,LLVM %}
-// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 -fclangir -emit-cir  %s -disable-O0-optnone |                               FileCheck %s --check-prefixes=ALL,CIR %}
+// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 -fclangir -emit-cir  %s -disable-O0-optnone |                               FileCheck %s --check-prefixes=ALL,CIR --implicit-check-not='cir.call_llvm_intrinsic "fma"' %}
 
 // ALL: {{[Mm]}}odule
 
diff --git a/clang/test/CodeGen/AArch64/neon/fused-multiply.c b/clang/test/CodeGen/AArch64/neon/fused-multiply.c
index 5d501d3f61734..bd1d842d7868a 100644
--- a/clang/test/CodeGen/AArch64/neon/fused-multiply.c
+++ b/clang/test/CodeGen/AArch64/neon/fused-multiply.c
@@ -3,6 +3,8 @@
 // RUN:                   %clang_cc1_cg_arm64_neon           -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,sroa | FileCheck %s --check-prefixes=ALL,LLVM
 // RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -fclangir -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,sroa | FileCheck %s --check-prefixes=ALL,LLVM %}
 // RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -fclangir -emit-cir  %s -disable-O0-optnone |                               FileCheck %s --check-prefixes=ALL,CIR %}
+// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -fexperimental-strict-floating-point -ffp-exception-behavior=strict -fclangir -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,sroa | FileCheck %s --check-prefix=LLVM-STRICT --implicit-check-not=' @llvm.fma.' %}
+// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -fexperimental-strict-floating-point -ffp-exception-behavior=strict -fclangir -emit-cir  %s -disable-O0-optnone |                               FileCheck %s --check-prefix=CIR-STRICT --implicit-check-not='cir.call_llvm_intrinsic "fma"' %}
 
 // ALL: {{[Mm]}}odule
 
@@ -50,7 +52,11 @@ float32x2_t test_vfma_f32(float32x2_t a, float32x2_t b, float32x2_t c) {
 }
 
 // LLVM-LABEL: @test_vfma_f64(
+// LLVM-STRICT-LABEL: @test_vfma_f64(
+// LLVM-STRICT: call <1 x double> @llvm.experimental.constrained.fma.v1f64({{.*}}, metadata !"round.tonearest", metadata !"fpexcept.strict")
 // CIR-LABEL: @vfma_f64(
+// CIR-STRICT-LABEL: cir.func {{.*}}@vfma_f64(
+// CIR-STRICT: cir.fma %{{.*}}, %{{.*}}, %{{.*}} : !cir.vector<1 x !cir.double> {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
 float64x1_t test_vfma_f64(float64x1_t a, float64x1_t b, float64x1_t c) {
 // CIR: cir.fma %{{.*}}, %{{.*}}, %{{.*}} : !cir.vector<1 x !cir.double>
 
@@ -222,6 +228,10 @@ float32x2_t test_vfma_laneq_f32(float32x2_t a, float32x2_t b, float32x4_t v) {
 }
 
 // ALL-LABEL: @test_vfma_laneq_f64(
+// LLVM-STRICT-LABEL: @test_vfma_laneq_f64(
+// LLVM-STRICT: call double @llvm.experimental.constrained.fma.f64({{.*}}, metadata !"round.tonearest", metadata !"fpexcept.strict")
+// CIR-STRICT-LABEL: cir.func {{.*}}@test_vfma_laneq_f64(
+// CIR-STRICT: cir.fma %{{.*}}, %{{.*}}, %{{.*}} : !cir.double {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
 float64x1_t test_vfma_laneq_f64(float64x1_t a, float64x1_t b,
                                  float64x2_t v) {
 // CIR: [[LANE:%.*]] = cir.vec.extract %{{.*}}[%{{.*}} : !u64i] : !cir.vector<2 x !cir.double>
@@ -291,6 +301,10 @@ float32x4_t test_vfmaq_laneq_f32(float32x4_t a, float32x4_t b,
 }
 
 // ALL-LABEL: @test_vfmaq_laneq_f64(
+// LLVM-STRICT-LABEL: @test_vfmaq_laneq_f64(
+// LLVM-STRICT: call <2 x double> @llvm.experimental.constrained.fma.v2f64({{.*}}, metadata !"round.tonearest", metadata !"fpexcept.strict")
+// CIR-STRICT-LABEL: cir.func {{.*}}@test_vfmaq_laneq_f64(
+// CIR-STRICT: cir.fma %{{.*}}, %{{.*}}, %{{.*}} : !cir.vector<2 x !cir.double> {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
 float64x2_t test_vfmaq_laneq_f64(float64x2_t a, float64x2_t b,
                                   float64x2_t v) {
 // CIR: [[LANE:%.*]] = cir.vec.shuffle(%{{.*}}, %{{.*}} : !cir.vector<2 x !cir.double>) [#cir.int<1> : !s32i, #cir.int<1> : !s32i] : !cir.vector<2 x !cir.double>
@@ -357,6 +371,10 @@ float64x2_t test_vfmaq_laneq_f64_0(float64x2_t a, float64x2_t b,
 }
 
 // ALL-LABEL: @test_vfmas_lane_f32(
+// LLVM-STRICT-LABEL: @test_vfmas_lane_f32(
+// LLVM-STRICT: call float @llvm.experimental.constrained.fma.f32({{.*}}, metadata !"round.tonearest", metadata !"fpexcept.strict")
+// CIR-STRICT-LABEL: cir.func {{.*}}@test_vfmas_lane_f32(
+// CIR-STRICT: cir.fma %{{.*}}, %{{.*}}, %{{.*}} : !cir.float {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
 float32_t test_vfmas_lane_f32(float32_t a, float32_t b, float32x2_t c) {
 // CIR: [[LANE:%.*]] = cir.vec.extract %{{.*}}[%{{.*}} : !u64i] : !cir.vector<2 x !cir.float>
 // CIR: cir.fma %{{.*}}, [[LANE]], %{{.*}} : !cir.float
diff --git a/clang/test/CodeGen/AArch64/neon/intrinsics.c b/clang/test/CodeGen/AArch64/neon/intrinsics.c
index 1393296648283..862572501eba1 100644
--- a/clang/test/CodeGen/AArch64/neon/intrinsics.c
+++ b/clang/test/CodeGen/AArch64/neon/intrinsics.c
@@ -3,6 +3,8 @@
 // RUN:                   %clang_cc1_cg_arm64_neon           -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,sroa | FileCheck %s --check-prefixes=ALL,LLVM
 // RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -fclangir -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,sroa | FileCheck %s --check-prefixes=ALL,LLVM %}
 // RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -fclangir -emit-cir  %s -disable-O0-optnone |                               FileCheck %s --check-prefixes=ALL,CIR %}
+// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -fexperimental-strict-floating-point -ffp-exception-behavior=strict -fclangir -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,sroa | FileCheck %s --check-prefix=LLVM-STRICT --implicit-check-not=' @llvm.sqrt.' %}
+// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -fexperimental-strict-floating-point -ffp-exception-behavior=strict -fclangir -emit-cir  %s -disable-O0-optnone |                               FileCheck %s --check-prefix=CIR-STRICT --implicit-check-not='cir.call_llvm_intrinsic "sqrt"' %}
 
 //=============================================================================
 // NOTES
@@ -1790,7 +1792,11 @@ float64x1_t test_vsqrt_f64(float64x1_t a) {
 }
 
 // LLVM-LABEL: @test_vsqrtq_f64(
+// LLVM-STRICT-LABEL: @test_vsqrtq_f64(
+// LLVM-STRICT: call <2 x double> @llvm.experimental.constrained.sqrt.v2f64({{.*}}, metadata !"round.tonearest", metadata !"fpexcept.strict")
 // CIR-LABEL: @vsqrtq_f64(
+// CIR-STRICT-LABEL: cir.func {{.*}}@vsqrtq_f64(
+// CIR-STRICT: cir.sqrt %{{.*}} : !cir.vector<2 x !cir.double> {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
 float64x2_t test_vsqrtq_f64(float64x2_t a) {
 // CIR: cir.sqrt %{{.*}} : !cir.vector<2 x !cir.double>
 
diff --git a/clang/test/CodeGen/AArch64/v8.2a-neon-intrinsics-constrained.c b/clang/test/CodeGen/AArch64/v8.2a-neon-intrinsics-constrained.c
index dac8b931ff210..fb41b7012cf8f 100644
--- a/clang/test/CodeGen/AArch64/v8.2a-neon-intrinsics-constrained.c
+++ b/clang/test/CodeGen/AArch64/v8.2a-neon-intrinsics-constrained.c
@@ -8,6 +8,15 @@
 // RUN: -flax-vector-conversions=none -disable-O0-optnone -emit-llvm -o - %s \
 // RUN: | opt -S -passes=mem2reg,sroa \
 // RUN: | FileCheck --check-prefix=CONSTRAINED --implicit-check-not=fpexcept.maytrap %s
+// RUN: %if cir-enabled %{%clang_cc1 -triple arm64-none-linux-gnu -target-feature +neon -target-feature +fullfp16 -target-feature +v8.2a \
+// RUN: -fexperimental-strict-floating-point -ffp-exception-behavior=maytrap -DEXCEPT=1 \
+// RUN: -flax-vector-conversions=none -disable-O0-optnone -fclangir -emit-cir -o - %s \
+// RUN: | FileCheck --check-prefix=CIR --implicit-check-not='cir.call_llvm_intrinsic "fma"' --implicit-check-not='cir.call_llvm_intrinsic "sqrt"' %s %}
+// RUN: %if cir-enabled %{%clang_cc1 -triple arm64-none-linux-gnu -target-feature +neon -target-feature +fullfp16 -target-feature +v8.2a \
+// RUN: -fexperimental-strict-floating-point -ffp-exception-behavior=maytrap -DEXCEPT=1 \
+// RUN: -flax-vector-conversions=none -disable-O0-optnone -fclangir -emit-llvm -o - %s \
+// RUN: | opt -S -passes=mem2reg,sroa \
+// RUN: | FileCheck --check-prefix=LLVM --implicit-check-not=fpexcept.maytrap --implicit-check-not=' @llvm.fma.' --implicit-check-not=' @llvm.sqrt.' %s %}
 
 // REQUIRES: aarch64-registered-target
 
@@ -39,6 +48,10 @@
 // CONSTRAINED-NEXT:    [[VSQRT_I:%.*]] = call <4 x half> @llvm.experimental.constrained.sqrt.v4f16(<4 x half> [[TMP2]], metadata !"round.tonearest", metadata !"fpexcept.strict") #[[ATTR2:[0-9]+]]
 // CONSTRAINED-NEXT:    ret <4 x half> [[VSQRT_I]]
 //
+// CIR-LABEL: cir.func {{.*}}@vsqrt_f16(
+// CIR: cir.sqrt %{{.*}} : !cir.vector<4 x !cir.f16> {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
+// LLVM-LABEL: @test_vsqrt_f16(
+// LLVM: call <4 x half> @llvm.experimental.constrained.sqrt.v4f16({{.*}}, metadata !"round.tonearest", metadata !"fpexcept.strict")
 float16x4_t test_vsqrt_f16(float16x4_t a) {
   return vsqrt_f16(a);
 }
@@ -61,6 +74,10 @@ float16x4_t test_vsqrt_f16(float16x4_t a) {
 // CONSTRAINED-NEXT:    [[VSQRT_I:%.*]] = call <8 x half> @llvm.experimental.constrained.sqrt.v8f16(<8 x half> [[TMP2]], metadata !"round.tonearest", metadata !"fpexcept.strict") #[[ATTR2]]
 // CONSTRAINED-NEXT:    ret <8 x half> [[VSQRT_I]]
 //
+// CIR-LABEL: cir.func {{.*}}@vsqrtq_f16(
+// CIR: cir.sqrt %{{.*}} : !cir.vector<8 x !cir.f16> {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
+// LLVM-LABEL: @test_vsqrtq_f16(
+// LLVM: call <8 x half> @llvm.experimental.constrained.sqrt.v8f16({{.*}}, metadata !"round.tonearest", metadata !"fpexcept.strict")
 float16x8_t test_vsqrtq_f16(float16x8_t a) {
   return vsqrtq_f16(a);
 }
@@ -95,6 +112,10 @@ float16x8_t test_vsqrtq_f16(float16x8_t a) {
 // CONSTRAINED-NEXT:    [[TMP9:%.*]] = call <4 x half> @llvm.experimental.constrained.fma.v4f16(<4 x half> [[TMP7]], <4 x half> [[TMP8]], <4 x half> [[TMP6]], metadata !"round.tonearest", metadata !"fpexcept.strict") #[[ATTR2]]
 // CONSTRAINED-NEXT:    ret <4 x half> [[TMP9]]
 //
+// CIR-LABEL: cir.func {{.*}}@vfma_f16(
+// CIR: cir.fma %{{.*}}, %{{.*}}, %{{.*}} : !cir.vector<4 x !cir.f16> {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
+// LLVM-LABEL: @test_vfma_f16(
+// LLVM: call <4 x half> @llvm.experimental.constrained.fma.v4f16({{.*}}, metadata !"round.tonearest", metadata !"fpexcept.strict")
 float16x4_t test_vfma_f16(float16x4_t a, float16x4_t b, float16x4_t c) {
   return vfma_f16(a, b, c);
 }
@@ -129,6 +150,10 @@ float16x4_t test_vfma_f16(float16x4_t a, float16x4_t b, float16x4_t c) {
 // CONSTRAINED-NEXT:    [[TMP9:%.*]] = call <8 x half> @llvm.experimental.constrained.fma.v8f16(<8 x half> [[TMP7]], <8 x half> [[TMP8]], <8 x half> [[TMP6]], metadata !"round.tonearest", metadata !"fpexcept.strict") #[[ATTR2]]
 // CONSTRAINED-NEXT:    ret <8 x half> [[TMP9]]
 //
+// CIR-LABEL: cir.func {{.*}}@vfmaq_f16(
+// CIR: cir.fma %{{.*}}, %{{.*}}, %{{.*}} : !cir.vector<8 x !cir.f16> {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
+// LLVM-LABEL: @test_vfmaq_f16(
+// LLVM: call <8 x half> @llvm.experimental.constrained.fma.v8f16({{.*}}, metadata !"round.tonearest", metadata !"fpexcept.strict")
 float16x8_t test_vfmaq_f16(float16x8_t a, float16x8_t b, float16x8_t c) {
   return vfmaq_f16(a, b, c);
 }
@@ -237,6 +262,10 @@ float16x8_t test_vfmsq_f16(float16x8_t a, float16x8_t b, float16x8_t c) {
 // CONSTRAINED-NEXT:    [[FMLA2:%.*]] = call <4 x half> @llvm.experimental.constrained.fma.v4f16(<4 x half> [[FMLA]], <4 x half> [[LANE]], <4 x half> [[FMLA1]], metadata !"round.tonearest", metadata !"fpexcept.strict") #[[ATTR2]]
 // CONSTRAINED-NEXT:    ret <4 x half> [[FMLA2]]
 //
+// CIR-LABEL: cir.func {{.*}}@test_vfma_lane_f16(
+// CIR: cir.fma %{{.*}}, %{{.*}}, %{{.*}} : !cir.vector<4 x !cir.f16> {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
+// LLVM-LABEL: @test_vfma_lane_f16(
+// LLVM: call <4 x half> @llvm.experimental.constrained.fma.v4f16({{.*}}, metadata !"round.tonearest", metadata !"fpexcept.strict")
 float16x4_t test_vfma_lane_f16(float16x4_t a, float16x4_t b, float16x4_t c) {
   return vfma_lane_f16(a, b, c, 3);
 }
@@ -309,6 +338,10 @@ float16x8_t test_vfmaq_lane_f16(float16x8_t a, float16x8_t b, float16x4_t c) {
 // CONSTRAINED-NEXT:    [[TMP9:%.*]] = call <4 x half> @llvm.experimental.constrained.fma.v4f16(<4 x half> [[LANE]], <4 x half> [[TMP7]], <4 x half> [[TMP6]], metadata !"round.tonearest", metadata !"fpexcept.strict") #[[ATTR2]]
 // CONSTRAINED-NEXT:    ret <4 x half> [[TMP9]]
 //
+// CIR-LABEL: cir.func {{.*}}@test_vfma_laneq_f16(
+// CIR: cir.fma %{{.*}}, %{{.*}}, %{{.*}} : !cir.vector<4 x !cir.f16> {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
+// LLVM-LABEL: @test_vfma_laneq_f16(
+// LLVM: call <4 x half> @llvm.experimental.constrained.fma.v4f16({{.*}}, metadata !"round.tonearest", metadata !"fpexcept.strict")
 float16x4_t test_vfma_laneq_f16(float16x4_t a, float16x4_t b, float16x8_t c) {
   return vfma_laneq_f16(a, b, c, 7);
 }
@@ -345,6 +378,10 @@ float16x4_t test_vfma_laneq_f16(float16x4_t a, float16x4_t b, float16x8_t c) {
 // CONSTRAINED-NEXT:    [[TMP9:%.*]] = call <8 x half> @llvm.experimental.constrained.fma.v8f16(<8 x half> [[LANE]], <8 x half> [[TMP7]], <8 x half> [[TMP6]], metadata !"round.tonearest", metadata !"fpexcept.strict") #[[ATTR2]]
 // CONSTRAINED-NEXT:    ret <8 x half> [[TMP9]]
 //
+// CIR-LABEL: cir.func {{.*}}@test_vfmaq_laneq_f16(
+// CIR: cir.fma %{{.*}}, %{{.*}}, %{{.*}} : !cir.vector<8 x !cir.f16> {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
+// LLVM-LABEL: @test_vfmaq_laneq_f16(
+// LLVM: call <8 x half> @llvm.experimental.constrained.fma.v8f16({{.*}}, metadata !"round.tonearest", metadata !"fpexcept.strict")
 float16x8_t test_vfmaq_laneq_f16(float16x8_t a, float16x8_t b, float16x8_t c) {
   return vfmaq_laneq_f16(a, b, c, 7);
 }
@@ -455,6 +492,10 @@ float16x8_t test_vfmaq_n_f16(float16x8_t a, float16x8_t b, float16_t c) {
 // CONSTRAINED-NEXT:    [[TMP0:%.*]] = call half @llvm.experimental.constrained.fma.f16(half [[B]], half [[EXTRACT]], half [[A]], metadata !"round.tonearest", metadata !"fpexcept.strict") #[[ATTR2]]
 // CONSTRAINED-NEXT:    ret half [[TMP0]]
 //
+// CIR-LABEL: cir.func {{.*}}@test_vfmah_lane_f16(
+// CIR: cir.fma %{{.*}}, %{{.*}}, %{{.*}} : !cir.f16 {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
+// LLVM-LABEL: @test_vfmah_lane_f16(
+// LLVM: call half @llvm.experimental.constrained.fma.f16({{.*}}, metadata !"round.tonearest", metadata !"fpexcept.strict")
 float16_t test_vfmah_lane_f16(float16_t a, float16_t b, float16x4_t c) {
   return vfmah_lane_f16(a, b, c, 3);
 }

>From 2b220b7c2d1cce3cbdda8820f9ceb2acd50065ad Mon Sep 17 00:00:00 2001
From: Yair Ben Avraham <yairba at protonmail.com>
Date: Wed, 26 Aug 2026 10:40:03 +0300
Subject: [PATCH 02/16] [CIR][AArch64] Avoid else after return

---
 clang/lib/CIR/CodeGen/CIRGenBuiltinAArch64.cpp | 12 ++++++++----
 1 file changed, 8 insertions(+), 4 deletions(-)

diff --git a/clang/lib/CIR/CodeGen/CIRGenBuiltinAArch64.cpp b/clang/lib/CIR/CodeGen/CIRGenBuiltinAArch64.cpp
index e71c3e24af2a3..63013e0163f0f 100644
--- a/clang/lib/CIR/CodeGen/CIRGenBuiltinAArch64.cpp
+++ b/clang/lib/CIR/CodeGen/CIRGenBuiltinAArch64.cpp
@@ -204,19 +204,23 @@ emitNeonCallToOp(CIRGenModule &cgm, CIRGenBuilderTy &builder,
                              builder.getStringAttr(intrinsicName.value()),
                              funcResTy, args)
         .getResult();
-  } else if constexpr (std::is_same_v<Operation, cir::FMAOp>) {
+  }
+  if constexpr (std::is_same_v<Operation, cir::FMAOp>) {
     assert(args.size() == 3 && "fma expects three operands");
     return Operation::create(builder, loc, funcResTy, args[0], args[1], args[2],
                              builder.getConstrainedFPAttr())
         .getResult();
-  } else if constexpr (std::is_same_v<Operation, cir::SqrtOp>) {
+  }
+  if constexpr (std::is_same_v<Operation, cir::SqrtOp>) {
     assert(args.size() == 1 && "sqrt expects one operand");
     return Operation::create(builder, loc, funcResTy, args[0],
                              builder.getConstrainedFPAttr())
         .getResult();
-  } else {
-    return Operation::create(builder, loc, funcResTy, args).getResult();
   }
+  if constexpr (!std::is_same_v<Operation, cir::LLVMIntrinsicCallOp> &&
+                !std::is_same_v<Operation, cir::FMAOp> &&
+                !std::is_same_v<Operation, cir::SqrtOp>)
+    return Operation::create(builder, loc, funcResTy, args).getResult();
 }
 
 // TODO(cir): Remove `cgm` from the list of arguments once all NYI(s) are gone.

>From 2e3b3dc80d1e175cf0d292acaef62765569d5b54 Mon Sep 17 00:00:00 2001
From: Yair Ben Avraham <yairba at protonmail.com>
Date: Wed, 26 Aug 2026 10:47:40 +0300
Subject: [PATCH 03/16] [CIR][AArch64] Move vfmas_lane_f32 constrained coverage

Reuse the existing scalar indexed constrained test for CIR coverage.
---
 .../AArch64/neon-scalar-x-indexed-elem-constrained.c | 12 ++++++++++++
 clang/test/CodeGen/AArch64/neon/fused-multiply.c     |  4 ----
 2 files changed, 12 insertions(+), 4 deletions(-)

diff --git a/clang/test/CodeGen/AArch64/neon-scalar-x-indexed-elem-constrained.c b/clang/test/CodeGen/AArch64/neon-scalar-x-indexed-elem-constrained.c
index 944929ccb5f42..e4e16e4f9876c 100644
--- a/clang/test/CodeGen/AArch64/neon-scalar-x-indexed-elem-constrained.c
+++ b/clang/test/CodeGen/AArch64/neon-scalar-x-indexed-elem-constrained.c
@@ -6,6 +6,14 @@
 // RUN: -ffp-exception-behavior=strict \
 // RUN: -disable-O0-optnone -emit-llvm -o - %s | opt -S -passes=mem2reg,sroa \
 // RUN: | FileCheck --check-prefix=CONSTRAINED %s
+// RUN: %if cir-enabled %{%clang_cc1 -triple arm64-none-linux-gnu -target-feature +neon -target-cpu cyclone \
+// RUN: -fexperimental-strict-floating-point -ffp-exception-behavior=strict \
+// RUN: -disable-O0-optnone -fclangir -emit-llvm -o - %s | opt -S -passes=mem2reg,sroa \
+// RUN: | FileCheck --check-prefix=LLVM --implicit-check-not=' @llvm.fma.' %s %}
+// RUN: %if cir-enabled %{%clang_cc1 -triple arm64-none-linux-gnu -target-feature +neon -target-cpu cyclone \
+// RUN: -fexperimental-strict-floating-point -ffp-exception-behavior=strict \
+// RUN: -disable-O0-optnone -fclangir -emit-cir -o - %s \
+// RUN: | FileCheck --check-prefix=CIR --implicit-check-not='cir.call_llvm_intrinsic "fma"' %s %}
 
 // REQUIRES: aarch64-registered-target
 
@@ -27,6 +35,10 @@
 // CONSTRAINED-NEXT:    [[TMP0:%.*]] = call float @llvm.experimental.constrained.fma.f32(float [[B]], float [[EXTRACT]], float [[A]], metadata !"round.tonearest", metadata !"fpexcept.strict") #[[ATTR2:[0-9]+]]
 // CONSTRAINED-NEXT:    ret float [[TMP0]]
 //
+// CIR-LABEL: cir.func {{.*}}@test_vfmas_lane_f32(
+// CIR: cir.fma %{{.*}}, %{{.*}}, %{{.*}} : !cir.float {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
+// LLVM-LABEL: @test_vfmas_lane_f32(
+// LLVM: call float @llvm.experimental.constrained.fma.f32({{.*}}, metadata !"round.tonearest", metadata !"fpexcept.strict")
 float32_t test_vfmas_lane_f32(float32_t a, float32_t b, float32x2_t c) {
   return vfmas_lane_f32(a, b, c, 1);
 }
diff --git a/clang/test/CodeGen/AArch64/neon/fused-multiply.c b/clang/test/CodeGen/AArch64/neon/fused-multiply.c
index bd1d842d7868a..e5320d7172353 100644
--- a/clang/test/CodeGen/AArch64/neon/fused-multiply.c
+++ b/clang/test/CodeGen/AArch64/neon/fused-multiply.c
@@ -371,10 +371,6 @@ float64x2_t test_vfmaq_laneq_f64_0(float64x2_t a, float64x2_t b,
 }
 
 // ALL-LABEL: @test_vfmas_lane_f32(
-// LLVM-STRICT-LABEL: @test_vfmas_lane_f32(
-// LLVM-STRICT: call float @llvm.experimental.constrained.fma.f32({{.*}}, metadata !"round.tonearest", metadata !"fpexcept.strict")
-// CIR-STRICT-LABEL: cir.func {{.*}}@test_vfmas_lane_f32(
-// CIR-STRICT: cir.fma %{{.*}}, %{{.*}}, %{{.*}} : !cir.float {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
 float32_t test_vfmas_lane_f32(float32_t a, float32_t b, float32x2_t c) {
 // CIR: [[LANE:%.*]] = cir.vec.extract %{{.*}}[%{{.*}} : !u64i] : !cir.vector<2 x !cir.float>
 // CIR: cir.fma %{{.*}}, [[LANE]], %{{.*}} : !cir.float

>From fece847dd13c3396c92953fcfd619484f20e6e5e Mon Sep 17 00:00:00 2001
From: Yair Ben Avraham <yairba at protonmail.com>
Date: Fri, 28 Aug 2026 13:20:46 +0300
Subject: [PATCH 04/16] [CIR][AArch64] Move laneq FMA constrained coverage

Reuse the existing scalar indexed constrained test for the remaining
f64 laneq FMA coverage.
---
 .../neon-scalar-x-indexed-elem-constrained.c  | 46 ++++++++++++++++++-
 .../CodeGen/AArch64/neon/fused-multiply.c     |  8 ----
 2 files changed, 45 insertions(+), 9 deletions(-)

diff --git a/clang/test/CodeGen/AArch64/neon-scalar-x-indexed-elem-constrained.c b/clang/test/CodeGen/AArch64/neon-scalar-x-indexed-elem-constrained.c
index e4e16e4f9876c..c8aab578b110a 100644
--- a/clang/test/CodeGen/AArch64/neon-scalar-x-indexed-elem-constrained.c
+++ b/clang/test/CodeGen/AArch64/neon-scalar-x-indexed-elem-constrained.c
@@ -223,10 +223,55 @@ float64x1_t test_vfms_lane_f64(float64x1_t a, float64x1_t b, float64x1_t v) {
 // CONSTRAINED-NEXT:    [[TMP10:%.*]] = bitcast double [[TMP9]] to <1 x double>
 // CONSTRAINED-NEXT:    ret <1 x double> [[TMP10]]
 //
+// CIR-LABEL: cir.func {{.*}}@test_vfma_laneq_f64(
+// CIR: cir.fma %{{.*}}, %{{.*}}, %{{.*}} : !cir.double {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
+// LLVM-LABEL: @test_vfma_laneq_f64(
+// LLVM: call double @llvm.experimental.constrained.fma.f64({{.*}}, metadata !"round.tonearest", metadata !"fpexcept.strict")
 float64x1_t test_vfma_laneq_f64(float64x1_t a, float64x1_t b, float64x2_t v) {
   return vfma_laneq_f64(a, b, v, 0);
 }
 
+// UNCONSTRAINED-LABEL: define dso_local <2 x double> @test_vfmaq_laneq_f64(
+// UNCONSTRAINED-SAME: <2 x double> noundef [[A:%.*]], <2 x double> noundef [[B:%.*]], <2 x double> noundef [[V:%.*]]) #[[ATTR0]] {
+// UNCONSTRAINED-NEXT:  [[ENTRY:.*:]]
+// UNCONSTRAINED-NEXT:    [[TMP0:%.*]] = bitcast <2 x double> [[A]] to <2 x i64>
+// UNCONSTRAINED-NEXT:    [[TMP1:%.*]] = bitcast <2 x double> [[B]] to <2 x i64>
+// UNCONSTRAINED-NEXT:    [[TMP2:%.*]] = bitcast <2 x double> [[V]] to <2 x i64>
+// UNCONSTRAINED-NEXT:    [[TMP3:%.*]] = bitcast <2 x i64> [[TMP0]] to <16 x i8>
+// UNCONSTRAINED-NEXT:    [[TMP4:%.*]] = bitcast <2 x i64> [[TMP1]] to <16 x i8>
+// UNCONSTRAINED-NEXT:    [[TMP5:%.*]] = bitcast <2 x i64> [[TMP2]] to <16 x i8>
+// UNCONSTRAINED-NEXT:    [[TMP6:%.*]] = bitcast <16 x i8> [[TMP3]] to <2 x double>
+// UNCONSTRAINED-NEXT:    [[TMP7:%.*]] = bitcast <16 x i8> [[TMP4]] to <2 x double>
+// UNCONSTRAINED-NEXT:    [[TMP8:%.*]] = bitcast <16 x i8> [[TMP5]] to <2 x double>
+// UNCONSTRAINED-NEXT:    [[LANE:%.*]] = shufflevector <2 x double> [[TMP8]], <2 x double> [[TMP8]], <2 x i32> <i32 1, i32 1>
+// UNCONSTRAINED-NEXT:    [[TMP9:%.*]] = call <2 x double> @llvm.fma.v2f64(<2 x double> [[LANE]], <2 x double> [[TMP7]], <2 x double> [[TMP6]])
+// UNCONSTRAINED-NEXT:    ret <2 x double> [[TMP9]]
+//
+// CONSTRAINED-LABEL: define dso_local <2 x double> @test_vfmaq_laneq_f64(
+// CONSTRAINED-SAME: <2 x double> noundef [[A:%.*]], <2 x double> noundef [[B:%.*]], <2 x double> noundef [[V:%.*]]) #[[ATTR0]] {
+// CONSTRAINED-NEXT:  [[ENTRY:.*:]]
+// CONSTRAINED-NEXT:    [[TMP0:%.*]] = bitcast <2 x double> [[A]] to <2 x i64>
+// CONSTRAINED-NEXT:    [[TMP1:%.*]] = bitcast <2 x double> [[B]] to <2 x i64>
+// CONSTRAINED-NEXT:    [[TMP2:%.*]] = bitcast <2 x double> [[V]] to <2 x i64>
+// CONSTRAINED-NEXT:    [[TMP3:%.*]] = bitcast <2 x i64> [[TMP0]] to <16 x i8>
+// CONSTRAINED-NEXT:    [[TMP4:%.*]] = bitcast <2 x i64> [[TMP1]] to <16 x i8>
+// CONSTRAINED-NEXT:    [[TMP5:%.*]] = bitcast <2 x i64> [[TMP2]] to <16 x i8>
+// CONSTRAINED-NEXT:    [[TMP6:%.*]] = bitcast <16 x i8> [[TMP3]] to <2 x double>
+// CONSTRAINED-NEXT:    [[TMP7:%.*]] = bitcast <16 x i8> [[TMP4]] to <2 x double>
+// CONSTRAINED-NEXT:    [[TMP8:%.*]] = bitcast <16 x i8> [[TMP5]] to <2 x double>
+// CONSTRAINED-NEXT:    [[LANE:%.*]] = shufflevector <2 x double> [[TMP8]], <2 x double> [[TMP8]], <2 x i32> <i32 1, i32 1>
+// CONSTRAINED-NEXT:    [[TMP9:%.*]] = call <2 x double> @llvm.experimental.constrained.fma.v2f64(<2 x double> [[LANE]], <2 x double> [[TMP7]], <2 x double> [[TMP6]], metadata !"round.tonearest", metadata !"fpexcept.strict") #[[ATTR2]]
+// CONSTRAINED-NEXT:    ret <2 x double> [[TMP9]]
+//
+// CIR-LABEL: cir.func {{.*}}@test_vfmaq_laneq_f64(
+// CIR: cir.fma %{{.*}}, %{{.*}}, %{{.*}} : !cir.vector<2 x !cir.double> {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
+// LLVM-LABEL: @test_vfmaq_laneq_f64(
+// LLVM: call <2 x double> @llvm.experimental.constrained.fma.v2f64({{.*}}, metadata !"round.tonearest", metadata !"fpexcept.strict")
+float64x2_t test_vfmaq_laneq_f64(float64x2_t a, float64x2_t b,
+                                 float64x2_t v) {
+  return vfmaq_laneq_f64(a, b, v, 1);
+}
+
 // UNCONSTRAINED-LABEL: define dso_local <1 x double> @test_vfms_laneq_f64(
 // UNCONSTRAINED-SAME: <1 x double> noundef [[A:%.*]], <1 x double> noundef [[B:%.*]], <2 x double> noundef [[V:%.*]]) #[[ATTR0]] {
 // UNCONSTRAINED-NEXT:  [[ENTRY:.*:]]
@@ -270,4 +315,3 @@ float64x1_t test_vfma_laneq_f64(float64x1_t a, float64x1_t b, float64x2_t v) {
 float64x1_t test_vfms_laneq_f64(float64x1_t a, float64x1_t b, float64x2_t v) {
   return vfms_laneq_f64(a, b, v, 0);
 }
-
diff --git a/clang/test/CodeGen/AArch64/neon/fused-multiply.c b/clang/test/CodeGen/AArch64/neon/fused-multiply.c
index e5320d7172353..4873abfd96eab 100644
--- a/clang/test/CodeGen/AArch64/neon/fused-multiply.c
+++ b/clang/test/CodeGen/AArch64/neon/fused-multiply.c
@@ -228,10 +228,6 @@ float32x2_t test_vfma_laneq_f32(float32x2_t a, float32x2_t b, float32x4_t v) {
 }
 
 // ALL-LABEL: @test_vfma_laneq_f64(
-// LLVM-STRICT-LABEL: @test_vfma_laneq_f64(
-// LLVM-STRICT: call double @llvm.experimental.constrained.fma.f64({{.*}}, metadata !"round.tonearest", metadata !"fpexcept.strict")
-// CIR-STRICT-LABEL: cir.func {{.*}}@test_vfma_laneq_f64(
-// CIR-STRICT: cir.fma %{{.*}}, %{{.*}}, %{{.*}} : !cir.double {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
 float64x1_t test_vfma_laneq_f64(float64x1_t a, float64x1_t b,
                                  float64x2_t v) {
 // CIR: [[LANE:%.*]] = cir.vec.extract %{{.*}}[%{{.*}} : !u64i] : !cir.vector<2 x !cir.double>
@@ -301,10 +297,6 @@ float32x4_t test_vfmaq_laneq_f32(float32x4_t a, float32x4_t b,
 }
 
 // ALL-LABEL: @test_vfmaq_laneq_f64(
-// LLVM-STRICT-LABEL: @test_vfmaq_laneq_f64(
-// LLVM-STRICT: call <2 x double> @llvm.experimental.constrained.fma.v2f64({{.*}}, metadata !"round.tonearest", metadata !"fpexcept.strict")
-// CIR-STRICT-LABEL: cir.func {{.*}}@test_vfmaq_laneq_f64(
-// CIR-STRICT: cir.fma %{{.*}}, %{{.*}}, %{{.*}} : !cir.vector<2 x !cir.double> {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
 float64x2_t test_vfmaq_laneq_f64(float64x2_t a, float64x2_t b,
                                   float64x2_t v) {
 // CIR: [[LANE:%.*]] = cir.vec.shuffle(%{{.*}}, %{{.*}} : !cir.vector<2 x !cir.double>) [#cir.int<1> : !s32i, #cir.int<1> : !s32i] : !cir.vector<2 x !cir.double>

>From 771593b25077966836baea86ddc794f6fc351e1c Mon Sep 17 00:00:00 2001
From: Yair Ben Avraham <yairba at protonmail.com>
Date: Fri, 28 Aug 2026 13:33:51 +0300
Subject: [PATCH 05/16] [CIR][AArch64] Move scalar FP16 constrained coverage

Reuse the existing v8.2a Neon constrained test for strict scalar FP16
FMA and square-root coverage.
---
 clang/test/CodeGen/AArch64/neon/fullfp16.c    | 10 ----
 .../v8.2a-neon-intrinsics-constrained.c       | 50 +++++++++++++++++++
 2 files changed, 50 insertions(+), 10 deletions(-)

diff --git a/clang/test/CodeGen/AArch64/neon/fullfp16.c b/clang/test/CodeGen/AArch64/neon/fullfp16.c
index ffab85ee93fbb..c17518eb813de 100644
--- a/clang/test/CodeGen/AArch64/neon/fullfp16.c
+++ b/clang/test/CodeGen/AArch64/neon/fullfp16.c
@@ -3,8 +3,6 @@
 // RUN:                   %clang_cc1_cg_arm64_neon -target-feature +fullfp16           -emit-llvm  %s -disable-O0-optnone | opt -S -passes=mem2reg             | FileCheck %s --check-prefixes=ALL,LLVM
 // RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 -fclangir -emit-llvm  %s -disable-O0-optnone | opt -S -passes=mem2reg,simplifycfg | FileCheck %s --check-prefixes=ALL,LLVM %}
 // RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 -fclangir -emit-cir   %s -disable-O0-optnone |                                      FileCheck %s --check-prefixes=ALL,CIR %}
-// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 -fexperimental-strict-floating-point -ffp-exception-behavior=strict -fclangir -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,simplifycfg | FileCheck %s --check-prefix=LLVM-STRICT --implicit-check-not=' @llvm.fma.' --implicit-check-not=' @llvm.sqrt.' %}
-// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 -fexperimental-strict-floating-point -ffp-exception-behavior=strict -fclangir -emit-cir  %s -disable-O0-optnone |                                      FileCheck %s --check-prefix=CIR-STRICT --implicit-check-not='cir.call_llvm_intrinsic "fma"' --implicit-check-not='cir.call_llvm_intrinsic "sqrt"' %}
 
 //=============================================================================
 // NOTES
@@ -311,10 +309,6 @@ float16_t test_vrndxh_f16(float16_t a) {
 //===------------------------------------------------------===//
 // 2.5.1.4.  Square root
 //===------------------------------------------------------===//
-// LLVM-STRICT-LABEL: @test_vsqrth_f16(
-// LLVM-STRICT: call half @llvm.experimental.constrained.sqrt.f16({{.*}}, metadata !"round.tonearest", metadata !"fpexcept.strict")
-// CIR-STRICT-LABEL: cir.func {{.*}}@test_vsqrth_f16(
-// CIR-STRICT: cir.sqrt %{{.*}} : !cir.f16 {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
 // ALL-LABEL: test_vsqrth_f16
 float16_t test_vsqrth_f16(float16_t a) {
 // CIR:  cir.sqrt
@@ -341,10 +335,6 @@ float16_t test_vnegh_f16(float16_t a) {
 //===------------------------------------------------------===//
 // 2.5.1.9.3 Fused multiply-accumulate
 //===------------------------------------------------------===//
-// LLVM-STRICT-LABEL: @test_vfmah_f16(
-// LLVM-STRICT: call half @llvm.experimental.constrained.fma.f16({{.*}}, metadata !"round.tonearest", metadata !"fpexcept.strict")
-// CIR-STRICT-LABEL: cir.func {{.*}}@test_vfmah_f16(
-// CIR-STRICT: cir.fma %{{.*}}, %{{.*}}, %{{.*}} : !cir.f16 {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
 // ALL-LABEL: test_vfmah_f16
 float16_t test_vfmah_f16(float16_t a, float16_t b, float16_t c) {
 // CIR: cir.fma {{.*}} : !cir.f16
diff --git a/clang/test/CodeGen/AArch64/v8.2a-neon-intrinsics-constrained.c b/clang/test/CodeGen/AArch64/v8.2a-neon-intrinsics-constrained.c
index fb41b7012cf8f..e9aece9dfa75f 100644
--- a/clang/test/CodeGen/AArch64/v8.2a-neon-intrinsics-constrained.c
+++ b/clang/test/CodeGen/AArch64/v8.2a-neon-intrinsics-constrained.c
@@ -17,6 +17,15 @@
 // RUN: -flax-vector-conversions=none -disable-O0-optnone -fclangir -emit-llvm -o - %s \
 // RUN: | opt -S -passes=mem2reg,sroa \
 // RUN: | FileCheck --check-prefix=LLVM --implicit-check-not=fpexcept.maytrap --implicit-check-not=' @llvm.fma.' --implicit-check-not=' @llvm.sqrt.' %s %}
+// RUN: %if cir-enabled %{%clang_cc1 -triple arm64-none-linux-gnu -target-feature +neon -target-feature +fullfp16 -target-feature +v8.2a \
+// RUN: -fexperimental-strict-floating-point -ffp-exception-behavior=strict \
+// RUN: -flax-vector-conversions=none -disable-O0-optnone -fclangir -emit-cir -o - %s \
+// RUN: | FileCheck --check-prefix=CIR --implicit-check-not='cir.call_llvm_intrinsic "fma"' --implicit-check-not='cir.call_llvm_intrinsic "sqrt"' %s %}
+// RUN: %if cir-enabled %{%clang_cc1 -triple arm64-none-linux-gnu -target-feature +neon -target-feature +fullfp16 -target-feature +v8.2a \
+// RUN: -fexperimental-strict-floating-point -ffp-exception-behavior=strict \
+// RUN: -flax-vector-conversions=none -disable-O0-optnone -fclangir -emit-llvm -o - %s \
+// RUN: | opt -S -passes=mem2reg,sroa \
+// RUN: | FileCheck --check-prefix=LLVM --implicit-check-not=' @llvm.fma.' --implicit-check-not=' @llvm.sqrt.' %s %}
 
 // REQUIRES: aarch64-registered-target
 
@@ -28,8 +37,29 @@
 #pragma float_control(except, on)
 #endif
 
+#include <arm_fp16.h>
 #include <arm_neon.h>
 
+// UNCONSTRAINED-LABEL: define dso_local half @test_vsqrth_f16(
+// UNCONSTRAINED-SAME: half noundef [[A:%.*]]) #[[ATTR0:[0-9]+]] {
+// UNCONSTRAINED-NEXT:  [[ENTRY:.*:]]
+// UNCONSTRAINED-NEXT:    [[SQR:%.*]] = call half @llvm.sqrt.f16(half [[A]])
+// UNCONSTRAINED-NEXT:    ret half [[SQR]]
+//
+// CONSTRAINED-LABEL: define dso_local half @test_vsqrth_f16(
+// CONSTRAINED-SAME: half noundef [[A:%.*]]) #[[ATTR0:[0-9]+]] {
+// CONSTRAINED-NEXT:  [[ENTRY:.*:]]
+// CONSTRAINED-NEXT:    [[SQR:%.*]] = call half @llvm.experimental.constrained.sqrt.f16(half [[A]], metadata !"round.tonearest", metadata !"fpexcept.strict") #[[ATTR2:[0-9]+]]
+// CONSTRAINED-NEXT:    ret half [[SQR]]
+//
+// CIR-LABEL: cir.func {{.*}}@test_vsqrth_f16(
+// CIR: cir.sqrt %{{.*}} : !cir.f16 {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
+// LLVM-LABEL: @test_vsqrth_f16(
+// LLVM: call half @llvm.experimental.constrained.sqrt.f16({{.*}}, metadata !"round.tonearest", metadata !"fpexcept.strict")
+float16_t test_vsqrth_f16(float16_t a) {
+  return vsqrth_f16(a);
+}
+
 // UNCONSTRAINED-LABEL: define dso_local <4 x half> @test_vsqrt_f16(
 // UNCONSTRAINED-SAME: <4 x half> noundef [[A:%.*]]) #[[ATTR0:[0-9]+]] {
 // UNCONSTRAINED-NEXT:  [[ENTRY:.*:]]
@@ -158,6 +188,26 @@ float16x8_t test_vfmaq_f16(float16x8_t a, float16x8_t b, float16x8_t c) {
   return vfmaq_f16(a, b, c);
 }
 
+// UNCONSTRAINED-LABEL: define dso_local half @test_vfmah_f16(
+// UNCONSTRAINED-SAME: half noundef [[A:%.*]], half noundef [[B:%.*]], half noundef [[C:%.*]]) #[[ATTR0]] {
+// UNCONSTRAINED-NEXT:  [[ENTRY:.*:]]
+// UNCONSTRAINED-NEXT:    [[FMA:%.*]] = call half @llvm.fma.f16(half [[B]], half [[C]], half [[A]])
+// UNCONSTRAINED-NEXT:    ret half [[FMA]]
+//
+// CONSTRAINED-LABEL: define dso_local half @test_vfmah_f16(
+// CONSTRAINED-SAME: half noundef [[A:%.*]], half noundef [[B:%.*]], half noundef [[C:%.*]]) #[[ATTR0]] {
+// CONSTRAINED-NEXT:  [[ENTRY:.*:]]
+// CONSTRAINED-NEXT:    [[FMA:%.*]] = call half @llvm.experimental.constrained.fma.f16(half [[B]], half [[C]], half [[A]], metadata !"round.tonearest", metadata !"fpexcept.strict") #[[ATTR2]]
+// CONSTRAINED-NEXT:    ret half [[FMA]]
+//
+// CIR-LABEL: cir.func {{.*}}@test_vfmah_f16(
+// CIR: cir.fma %{{.*}}, %{{.*}}, %{{.*}} : !cir.f16 {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
+// LLVM-LABEL: @test_vfmah_f16(
+// LLVM: call half @llvm.experimental.constrained.fma.f16({{.*}}, metadata !"round.tonearest", metadata !"fpexcept.strict")
+float16_t test_vfmah_f16(float16_t a, float16_t b, float16_t c) {
+  return vfmah_f16(a, b, c);
+}
+
 // UNCONSTRAINED-LABEL: define dso_local <4 x half> @test_vfms_f16(
 // UNCONSTRAINED-SAME: <4 x half> noundef [[A:%.*]], <4 x half> noundef [[B:%.*]], <4 x half> noundef [[C:%.*]]) #[[ATTR0]] {
 // UNCONSTRAINED-NEXT:  [[ENTRY:.*:]]

>From 8cbe19fc19d2490e2c362f568e695cbfeacc8590 Mon Sep 17 00:00:00 2001
From: Yair Ben Avraham <yairba at protonmail.com>
Date: Fri, 28 Aug 2026 13:48:02 +0300
Subject: [PATCH 06/16] [CIR][AArch64] Move FMA and sqrt constrained coverage

Move strict vfma_f64 and vsqrtq_f64 CIR coverage out of the ordinary
Neon tests and into a focused constrained test.

The existing legacy constrained files cannot run through CIR as a whole
because they contain unrelated AArch64 builtins that CIR does not yet lower.
---
 .../AArch64/neon/fma-sqrt-constrained.c       | 22 +++++++++++++++++++
 .../CodeGen/AArch64/neon/fused-multiply.c     |  6 -----
 clang/test/CodeGen/AArch64/neon/intrinsics.c  |  6 -----
 3 files changed, 22 insertions(+), 12 deletions(-)
 create mode 100644 clang/test/CodeGen/AArch64/neon/fma-sqrt-constrained.c

diff --git a/clang/test/CodeGen/AArch64/neon/fma-sqrt-constrained.c b/clang/test/CodeGen/AArch64/neon/fma-sqrt-constrained.c
new file mode 100644
index 0000000000000..8665bc39226d7
--- /dev/null
+++ b/clang/test/CodeGen/AArch64/neon/fma-sqrt-constrained.c
@@ -0,0 +1,22 @@
+// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -fexperimental-strict-floating-point -ffp-exception-behavior=strict -fclangir -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,sroa | FileCheck %s --check-prefix=LLVM --implicit-check-not=' @llvm.fma.' --implicit-check-not=' @llvm.sqrt.' %}
+// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -fexperimental-strict-floating-point -ffp-exception-behavior=strict -fclangir -emit-cir  %s -disable-O0-optnone |                               FileCheck %s --check-prefix=CIR --implicit-check-not='cir.call_llvm_intrinsic "fma"' --implicit-check-not='cir.call_llvm_intrinsic "sqrt"' %}
+
+// REQUIRES: aarch64-registered-target
+
+#include <arm_neon.h>
+
+// LLVM-LABEL: @test_vfma_f64(
+// LLVM: call <1 x double> @llvm.experimental.constrained.fma.v1f64({{.*}}, metadata !"round.tonearest", metadata !"fpexcept.strict")
+// CIR-LABEL: cir.func {{.*}}@test_vfma_f64(
+// CIR: cir.fma %{{.*}}, %{{.*}}, %{{.*}} : !cir.vector<1 x !cir.double> {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
+float64x1_t test_vfma_f64(float64x1_t a, float64x1_t b, float64x1_t c) {
+  return vfma_f64(a, b, c);
+}
+
+// LLVM-LABEL: @test_vsqrtq_f64(
+// LLVM: call <2 x double> @llvm.experimental.constrained.sqrt.v2f64({{.*}}, metadata !"round.tonearest", metadata !"fpexcept.strict")
+// CIR-LABEL: cir.func {{.*}}@test_vsqrtq_f64(
+// CIR: cir.sqrt %{{.*}} : !cir.vector<2 x !cir.double> {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
+float64x2_t test_vsqrtq_f64(float64x2_t a) {
+  return vsqrtq_f64(a);
+}
diff --git a/clang/test/CodeGen/AArch64/neon/fused-multiply.c b/clang/test/CodeGen/AArch64/neon/fused-multiply.c
index 4873abfd96eab..5d501d3f61734 100644
--- a/clang/test/CodeGen/AArch64/neon/fused-multiply.c
+++ b/clang/test/CodeGen/AArch64/neon/fused-multiply.c
@@ -3,8 +3,6 @@
 // RUN:                   %clang_cc1_cg_arm64_neon           -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,sroa | FileCheck %s --check-prefixes=ALL,LLVM
 // RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -fclangir -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,sroa | FileCheck %s --check-prefixes=ALL,LLVM %}
 // RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -fclangir -emit-cir  %s -disable-O0-optnone |                               FileCheck %s --check-prefixes=ALL,CIR %}
-// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -fexperimental-strict-floating-point -ffp-exception-behavior=strict -fclangir -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,sroa | FileCheck %s --check-prefix=LLVM-STRICT --implicit-check-not=' @llvm.fma.' %}
-// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -fexperimental-strict-floating-point -ffp-exception-behavior=strict -fclangir -emit-cir  %s -disable-O0-optnone |                               FileCheck %s --check-prefix=CIR-STRICT --implicit-check-not='cir.call_llvm_intrinsic "fma"' %}
 
 // ALL: {{[Mm]}}odule
 
@@ -52,11 +50,7 @@ float32x2_t test_vfma_f32(float32x2_t a, float32x2_t b, float32x2_t c) {
 }
 
 // LLVM-LABEL: @test_vfma_f64(
-// LLVM-STRICT-LABEL: @test_vfma_f64(
-// LLVM-STRICT: call <1 x double> @llvm.experimental.constrained.fma.v1f64({{.*}}, metadata !"round.tonearest", metadata !"fpexcept.strict")
 // CIR-LABEL: @vfma_f64(
-// CIR-STRICT-LABEL: cir.func {{.*}}@vfma_f64(
-// CIR-STRICT: cir.fma %{{.*}}, %{{.*}}, %{{.*}} : !cir.vector<1 x !cir.double> {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
 float64x1_t test_vfma_f64(float64x1_t a, float64x1_t b, float64x1_t c) {
 // CIR: cir.fma %{{.*}}, %{{.*}}, %{{.*}} : !cir.vector<1 x !cir.double>
 
diff --git a/clang/test/CodeGen/AArch64/neon/intrinsics.c b/clang/test/CodeGen/AArch64/neon/intrinsics.c
index 862572501eba1..1393296648283 100644
--- a/clang/test/CodeGen/AArch64/neon/intrinsics.c
+++ b/clang/test/CodeGen/AArch64/neon/intrinsics.c
@@ -3,8 +3,6 @@
 // RUN:                   %clang_cc1_cg_arm64_neon           -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,sroa | FileCheck %s --check-prefixes=ALL,LLVM
 // RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -fclangir -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,sroa | FileCheck %s --check-prefixes=ALL,LLVM %}
 // RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -fclangir -emit-cir  %s -disable-O0-optnone |                               FileCheck %s --check-prefixes=ALL,CIR %}
-// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -fexperimental-strict-floating-point -ffp-exception-behavior=strict -fclangir -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,sroa | FileCheck %s --check-prefix=LLVM-STRICT --implicit-check-not=' @llvm.sqrt.' %}
-// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -fexperimental-strict-floating-point -ffp-exception-behavior=strict -fclangir -emit-cir  %s -disable-O0-optnone |                               FileCheck %s --check-prefix=CIR-STRICT --implicit-check-not='cir.call_llvm_intrinsic "sqrt"' %}
 
 //=============================================================================
 // NOTES
@@ -1792,11 +1790,7 @@ float64x1_t test_vsqrt_f64(float64x1_t a) {
 }
 
 // LLVM-LABEL: @test_vsqrtq_f64(
-// LLVM-STRICT-LABEL: @test_vsqrtq_f64(
-// LLVM-STRICT: call <2 x double> @llvm.experimental.constrained.sqrt.v2f64({{.*}}, metadata !"round.tonearest", metadata !"fpexcept.strict")
 // CIR-LABEL: @vsqrtq_f64(
-// CIR-STRICT-LABEL: cir.func {{.*}}@vsqrtq_f64(
-// CIR-STRICT: cir.sqrt %{{.*}} : !cir.vector<2 x !cir.double> {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
 float64x2_t test_vsqrtq_f64(float64x2_t a) {
 // CIR: cir.sqrt %{{.*}} : !cir.vector<2 x !cir.double>
 

>From b5f0796081b97d4e234124bc799f3732e3742a53 Mon Sep 17 00:00:00 2001
From: Yair Ben Avraham <yairba at protonmail.com>
Date: Sun, 30 Aug 2026 07:05:55 +0300
Subject: [PATCH 07/16] [CIR][AArch64] Co-locate constrained FMA and sqrt tests

Place constrained FMA and sqrt coverage in files matching their regular
Neon counterparts. Move the new indexed FMA CIR coverage from the legacy
scalar indexed test and preserve the direct vfmaq_laneq_f64 checks.

Assisted-by: Codex
---
 .../neon-scalar-x-indexed-elem-constrained.c  | 57 --------------
 .../AArch64/neon/fused-multiply-constrained.c | 75 +++++++++++++++++++
 ...constrained.c => intrinsics-constrained.c} | 12 +--
 3 files changed, 77 insertions(+), 67 deletions(-)
 create mode 100644 clang/test/CodeGen/AArch64/neon/fused-multiply-constrained.c
 rename clang/test/CodeGen/AArch64/neon/{fma-sqrt-constrained.c => intrinsics-constrained.c} (62%)

diff --git a/clang/test/CodeGen/AArch64/neon-scalar-x-indexed-elem-constrained.c b/clang/test/CodeGen/AArch64/neon-scalar-x-indexed-elem-constrained.c
index c8aab578b110a..54930d384e247 100644
--- a/clang/test/CodeGen/AArch64/neon-scalar-x-indexed-elem-constrained.c
+++ b/clang/test/CodeGen/AArch64/neon-scalar-x-indexed-elem-constrained.c
@@ -6,14 +6,6 @@
 // RUN: -ffp-exception-behavior=strict \
 // RUN: -disable-O0-optnone -emit-llvm -o - %s | opt -S -passes=mem2reg,sroa \
 // RUN: | FileCheck --check-prefix=CONSTRAINED %s
-// RUN: %if cir-enabled %{%clang_cc1 -triple arm64-none-linux-gnu -target-feature +neon -target-cpu cyclone \
-// RUN: -fexperimental-strict-floating-point -ffp-exception-behavior=strict \
-// RUN: -disable-O0-optnone -fclangir -emit-llvm -o - %s | opt -S -passes=mem2reg,sroa \
-// RUN: | FileCheck --check-prefix=LLVM --implicit-check-not=' @llvm.fma.' %s %}
-// RUN: %if cir-enabled %{%clang_cc1 -triple arm64-none-linux-gnu -target-feature +neon -target-cpu cyclone \
-// RUN: -fexperimental-strict-floating-point -ffp-exception-behavior=strict \
-// RUN: -disable-O0-optnone -fclangir -emit-cir -o - %s \
-// RUN: | FileCheck --check-prefix=CIR --implicit-check-not='cir.call_llvm_intrinsic "fma"' %s %}
 
 // REQUIRES: aarch64-registered-target
 
@@ -35,10 +27,6 @@
 // CONSTRAINED-NEXT:    [[TMP0:%.*]] = call float @llvm.experimental.constrained.fma.f32(float [[B]], float [[EXTRACT]], float [[A]], metadata !"round.tonearest", metadata !"fpexcept.strict") #[[ATTR2:[0-9]+]]
 // CONSTRAINED-NEXT:    ret float [[TMP0]]
 //
-// CIR-LABEL: cir.func {{.*}}@test_vfmas_lane_f32(
-// CIR: cir.fma %{{.*}}, %{{.*}}, %{{.*}} : !cir.float {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
-// LLVM-LABEL: @test_vfmas_lane_f32(
-// LLVM: call float @llvm.experimental.constrained.fma.f32({{.*}}, metadata !"round.tonearest", metadata !"fpexcept.strict")
 float32_t test_vfmas_lane_f32(float32_t a, float32_t b, float32x2_t c) {
   return vfmas_lane_f32(a, b, c, 1);
 }
@@ -223,55 +211,10 @@ float64x1_t test_vfms_lane_f64(float64x1_t a, float64x1_t b, float64x1_t v) {
 // CONSTRAINED-NEXT:    [[TMP10:%.*]] = bitcast double [[TMP9]] to <1 x double>
 // CONSTRAINED-NEXT:    ret <1 x double> [[TMP10]]
 //
-// CIR-LABEL: cir.func {{.*}}@test_vfma_laneq_f64(
-// CIR: cir.fma %{{.*}}, %{{.*}}, %{{.*}} : !cir.double {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
-// LLVM-LABEL: @test_vfma_laneq_f64(
-// LLVM: call double @llvm.experimental.constrained.fma.f64({{.*}}, metadata !"round.tonearest", metadata !"fpexcept.strict")
 float64x1_t test_vfma_laneq_f64(float64x1_t a, float64x1_t b, float64x2_t v) {
   return vfma_laneq_f64(a, b, v, 0);
 }
 
-// UNCONSTRAINED-LABEL: define dso_local <2 x double> @test_vfmaq_laneq_f64(
-// UNCONSTRAINED-SAME: <2 x double> noundef [[A:%.*]], <2 x double> noundef [[B:%.*]], <2 x double> noundef [[V:%.*]]) #[[ATTR0]] {
-// UNCONSTRAINED-NEXT:  [[ENTRY:.*:]]
-// UNCONSTRAINED-NEXT:    [[TMP0:%.*]] = bitcast <2 x double> [[A]] to <2 x i64>
-// UNCONSTRAINED-NEXT:    [[TMP1:%.*]] = bitcast <2 x double> [[B]] to <2 x i64>
-// UNCONSTRAINED-NEXT:    [[TMP2:%.*]] = bitcast <2 x double> [[V]] to <2 x i64>
-// UNCONSTRAINED-NEXT:    [[TMP3:%.*]] = bitcast <2 x i64> [[TMP0]] to <16 x i8>
-// UNCONSTRAINED-NEXT:    [[TMP4:%.*]] = bitcast <2 x i64> [[TMP1]] to <16 x i8>
-// UNCONSTRAINED-NEXT:    [[TMP5:%.*]] = bitcast <2 x i64> [[TMP2]] to <16 x i8>
-// UNCONSTRAINED-NEXT:    [[TMP6:%.*]] = bitcast <16 x i8> [[TMP3]] to <2 x double>
-// UNCONSTRAINED-NEXT:    [[TMP7:%.*]] = bitcast <16 x i8> [[TMP4]] to <2 x double>
-// UNCONSTRAINED-NEXT:    [[TMP8:%.*]] = bitcast <16 x i8> [[TMP5]] to <2 x double>
-// UNCONSTRAINED-NEXT:    [[LANE:%.*]] = shufflevector <2 x double> [[TMP8]], <2 x double> [[TMP8]], <2 x i32> <i32 1, i32 1>
-// UNCONSTRAINED-NEXT:    [[TMP9:%.*]] = call <2 x double> @llvm.fma.v2f64(<2 x double> [[LANE]], <2 x double> [[TMP7]], <2 x double> [[TMP6]])
-// UNCONSTRAINED-NEXT:    ret <2 x double> [[TMP9]]
-//
-// CONSTRAINED-LABEL: define dso_local <2 x double> @test_vfmaq_laneq_f64(
-// CONSTRAINED-SAME: <2 x double> noundef [[A:%.*]], <2 x double> noundef [[B:%.*]], <2 x double> noundef [[V:%.*]]) #[[ATTR0]] {
-// CONSTRAINED-NEXT:  [[ENTRY:.*:]]
-// CONSTRAINED-NEXT:    [[TMP0:%.*]] = bitcast <2 x double> [[A]] to <2 x i64>
-// CONSTRAINED-NEXT:    [[TMP1:%.*]] = bitcast <2 x double> [[B]] to <2 x i64>
-// CONSTRAINED-NEXT:    [[TMP2:%.*]] = bitcast <2 x double> [[V]] to <2 x i64>
-// CONSTRAINED-NEXT:    [[TMP3:%.*]] = bitcast <2 x i64> [[TMP0]] to <16 x i8>
-// CONSTRAINED-NEXT:    [[TMP4:%.*]] = bitcast <2 x i64> [[TMP1]] to <16 x i8>
-// CONSTRAINED-NEXT:    [[TMP5:%.*]] = bitcast <2 x i64> [[TMP2]] to <16 x i8>
-// CONSTRAINED-NEXT:    [[TMP6:%.*]] = bitcast <16 x i8> [[TMP3]] to <2 x double>
-// CONSTRAINED-NEXT:    [[TMP7:%.*]] = bitcast <16 x i8> [[TMP4]] to <2 x double>
-// CONSTRAINED-NEXT:    [[TMP8:%.*]] = bitcast <16 x i8> [[TMP5]] to <2 x double>
-// CONSTRAINED-NEXT:    [[LANE:%.*]] = shufflevector <2 x double> [[TMP8]], <2 x double> [[TMP8]], <2 x i32> <i32 1, i32 1>
-// CONSTRAINED-NEXT:    [[TMP9:%.*]] = call <2 x double> @llvm.experimental.constrained.fma.v2f64(<2 x double> [[LANE]], <2 x double> [[TMP7]], <2 x double> [[TMP6]], metadata !"round.tonearest", metadata !"fpexcept.strict") #[[ATTR2]]
-// CONSTRAINED-NEXT:    ret <2 x double> [[TMP9]]
-//
-// CIR-LABEL: cir.func {{.*}}@test_vfmaq_laneq_f64(
-// CIR: cir.fma %{{.*}}, %{{.*}}, %{{.*}} : !cir.vector<2 x !cir.double> {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
-// LLVM-LABEL: @test_vfmaq_laneq_f64(
-// LLVM: call <2 x double> @llvm.experimental.constrained.fma.v2f64({{.*}}, metadata !"round.tonearest", metadata !"fpexcept.strict")
-float64x2_t test_vfmaq_laneq_f64(float64x2_t a, float64x2_t b,
-                                 float64x2_t v) {
-  return vfmaq_laneq_f64(a, b, v, 1);
-}
-
 // UNCONSTRAINED-LABEL: define dso_local <1 x double> @test_vfms_laneq_f64(
 // UNCONSTRAINED-SAME: <1 x double> noundef [[A:%.*]], <1 x double> noundef [[B:%.*]], <2 x double> noundef [[V:%.*]]) #[[ATTR0]] {
 // UNCONSTRAINED-NEXT:  [[ENTRY:.*:]]
diff --git a/clang/test/CodeGen/AArch64/neon/fused-multiply-constrained.c b/clang/test/CodeGen/AArch64/neon/fused-multiply-constrained.c
new file mode 100644
index 0000000000000..25bc87b376e42
--- /dev/null
+++ b/clang/test/CodeGen/AArch64/neon/fused-multiply-constrained.c
@@ -0,0 +1,75 @@
+// NOTE: Assertions have been autogenerated by utils/update_cc_test_checks.py UTC_ARGS: --version 5
+// RUN: %clang_cc1_cg_arm64_neon -target-cpu cyclone -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,sroa | FileCheck %s --check-prefix=UNCONSTRAINED
+// RUN: %clang_cc1_cg_arm64_neon -target-cpu cyclone -ffp-exception-behavior=strict -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,sroa | FileCheck %s --check-prefix=CONSTRAINED
+// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -fexperimental-strict-floating-point -ffp-exception-behavior=strict -fclangir -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,sroa | FileCheck %s --check-prefix=LLVM --implicit-check-not=' @llvm.fma.' %}
+// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -fexperimental-strict-floating-point -ffp-exception-behavior=strict -fclangir -emit-cir  %s -disable-O0-optnone |                               FileCheck %s --check-prefix=CIR --implicit-check-not='cir.call_llvm_intrinsic "fma"' %}
+
+// REQUIRES: aarch64-registered-target
+
+#include <arm_neon.h>
+
+// LLVM-LABEL: @test_vfma_f64(
+// LLVM: call <1 x double> @llvm.experimental.constrained.fma.v1f64({{.*}}, metadata !"round.tonearest", metadata !"fpexcept.strict")
+// CIR-LABEL: cir.func {{.*}}@test_vfma_f64(
+// CIR: cir.fma %{{.*}}, %{{.*}}, %{{.*}} : !cir.vector<1 x !cir.double> {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
+float64x1_t test_vfma_f64(float64x1_t a, float64x1_t b, float64x1_t c) {
+  return vfma_f64(a, b, c);
+}
+
+// CIR-LABEL: cir.func {{.*}}@test_vfma_laneq_f64(
+// CIR: cir.fma %{{.*}}, %{{.*}}, %{{.*}} : !cir.double {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
+// LLVM-LABEL: @test_vfma_laneq_f64(
+// LLVM: call double @llvm.experimental.constrained.fma.f64({{.*}}, metadata !"round.tonearest", metadata !"fpexcept.strict")
+float64x1_t test_vfma_laneq_f64(float64x1_t a, float64x1_t b,
+                                float64x2_t v) {
+  return vfma_laneq_f64(a, b, v, 0);
+}
+
+// UNCONSTRAINED-LABEL: define dso_local <2 x double> @test_vfmaq_laneq_f64(
+// UNCONSTRAINED-SAME: <2 x double> noundef [[A:%.*]], <2 x double> noundef [[B:%.*]], <2 x double> noundef [[V:%.*]]) #[[ATTR0:[0-9]+]] {
+// UNCONSTRAINED-NEXT:  [[ENTRY:.*:]]
+// UNCONSTRAINED-NEXT:    [[TMP0:%.*]] = bitcast <2 x double> [[A]] to <2 x i64>
+// UNCONSTRAINED-NEXT:    [[TMP1:%.*]] = bitcast <2 x double> [[B]] to <2 x i64>
+// UNCONSTRAINED-NEXT:    [[TMP2:%.*]] = bitcast <2 x double> [[V]] to <2 x i64>
+// UNCONSTRAINED-NEXT:    [[TMP3:%.*]] = bitcast <2 x i64> [[TMP0]] to <16 x i8>
+// UNCONSTRAINED-NEXT:    [[TMP4:%.*]] = bitcast <2 x i64> [[TMP1]] to <16 x i8>
+// UNCONSTRAINED-NEXT:    [[TMP5:%.*]] = bitcast <2 x i64> [[TMP2]] to <16 x i8>
+// UNCONSTRAINED-NEXT:    [[TMP6:%.*]] = bitcast <16 x i8> [[TMP3]] to <2 x double>
+// UNCONSTRAINED-NEXT:    [[TMP7:%.*]] = bitcast <16 x i8> [[TMP4]] to <2 x double>
+// UNCONSTRAINED-NEXT:    [[TMP8:%.*]] = bitcast <16 x i8> [[TMP5]] to <2 x double>
+// UNCONSTRAINED-NEXT:    [[LANE:%.*]] = shufflevector <2 x double> [[TMP8]], <2 x double> [[TMP8]], <2 x i32> <i32 1, i32 1>
+// UNCONSTRAINED-NEXT:    [[TMP9:%.*]] = call <2 x double> @llvm.fma.v2f64(<2 x double> [[LANE]], <2 x double> [[TMP7]], <2 x double> [[TMP6]])
+// UNCONSTRAINED-NEXT:    ret <2 x double> [[TMP9]]
+//
+// CONSTRAINED-LABEL: define dso_local <2 x double> @test_vfmaq_laneq_f64(
+// CONSTRAINED-SAME: <2 x double> noundef [[A:%.*]], <2 x double> noundef [[B:%.*]], <2 x double> noundef [[V:%.*]]) #[[ATTR0:[0-9]+]] {
+// CONSTRAINED-NEXT:  [[ENTRY:.*:]]
+// CONSTRAINED-NEXT:    [[TMP0:%.*]] = bitcast <2 x double> [[A]] to <2 x i64>
+// CONSTRAINED-NEXT:    [[TMP1:%.*]] = bitcast <2 x double> [[B]] to <2 x i64>
+// CONSTRAINED-NEXT:    [[TMP2:%.*]] = bitcast <2 x double> [[V]] to <2 x i64>
+// CONSTRAINED-NEXT:    [[TMP3:%.*]] = bitcast <2 x i64> [[TMP0]] to <16 x i8>
+// CONSTRAINED-NEXT:    [[TMP4:%.*]] = bitcast <2 x i64> [[TMP1]] to <16 x i8>
+// CONSTRAINED-NEXT:    [[TMP5:%.*]] = bitcast <2 x i64> [[TMP2]] to <16 x i8>
+// CONSTRAINED-NEXT:    [[TMP6:%.*]] = bitcast <16 x i8> [[TMP3]] to <2 x double>
+// CONSTRAINED-NEXT:    [[TMP7:%.*]] = bitcast <16 x i8> [[TMP4]] to <2 x double>
+// CONSTRAINED-NEXT:    [[TMP8:%.*]] = bitcast <16 x i8> [[TMP5]] to <2 x double>
+// CONSTRAINED-NEXT:    [[LANE:%.*]] = shufflevector <2 x double> [[TMP8]], <2 x double> [[TMP8]], <2 x i32> <i32 1, i32 1>
+// CONSTRAINED-NEXT:    [[TMP9:%.*]] = call <2 x double> @llvm.experimental.constrained.fma.v2f64(<2 x double> [[LANE]], <2 x double> [[TMP7]], <2 x double> [[TMP6]], metadata !"round.tonearest", metadata !"fpexcept.strict") #[[ATTR2:[0-9]+]]
+// CONSTRAINED-NEXT:    ret <2 x double> [[TMP9]]
+//
+// CIR-LABEL: cir.func {{.*}}@test_vfmaq_laneq_f64(
+// CIR: cir.fma %{{.*}}, %{{.*}}, %{{.*}} : !cir.vector<2 x !cir.double> {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
+// LLVM-LABEL: @test_vfmaq_laneq_f64(
+// LLVM: call <2 x double> @llvm.experimental.constrained.fma.v2f64({{.*}}, metadata !"round.tonearest", metadata !"fpexcept.strict")
+float64x2_t test_vfmaq_laneq_f64(float64x2_t a, float64x2_t b,
+                                 float64x2_t v) {
+  return vfmaq_laneq_f64(a, b, v, 1);
+}
+
+// CIR-LABEL: cir.func {{.*}}@test_vfmas_lane_f32(
+// CIR: cir.fma %{{.*}}, %{{.*}}, %{{.*}} : !cir.float {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
+// LLVM-LABEL: @test_vfmas_lane_f32(
+// LLVM: call float @llvm.experimental.constrained.fma.f32({{.*}}, metadata !"round.tonearest", metadata !"fpexcept.strict")
+float32_t test_vfmas_lane_f32(float32_t a, float32_t b, float32x2_t c) {
+  return vfmas_lane_f32(a, b, c, 1);
+}
diff --git a/clang/test/CodeGen/AArch64/neon/fma-sqrt-constrained.c b/clang/test/CodeGen/AArch64/neon/intrinsics-constrained.c
similarity index 62%
rename from clang/test/CodeGen/AArch64/neon/fma-sqrt-constrained.c
rename to clang/test/CodeGen/AArch64/neon/intrinsics-constrained.c
index 8665bc39226d7..d63f6a3373ebb 100644
--- a/clang/test/CodeGen/AArch64/neon/fma-sqrt-constrained.c
+++ b/clang/test/CodeGen/AArch64/neon/intrinsics-constrained.c
@@ -1,18 +1,10 @@
-// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -fexperimental-strict-floating-point -ffp-exception-behavior=strict -fclangir -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,sroa | FileCheck %s --check-prefix=LLVM --implicit-check-not=' @llvm.fma.' --implicit-check-not=' @llvm.sqrt.' %}
-// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -fexperimental-strict-floating-point -ffp-exception-behavior=strict -fclangir -emit-cir  %s -disable-O0-optnone |                               FileCheck %s --check-prefix=CIR --implicit-check-not='cir.call_llvm_intrinsic "fma"' --implicit-check-not='cir.call_llvm_intrinsic "sqrt"' %}
+// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -fexperimental-strict-floating-point -ffp-exception-behavior=strict -fclangir -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,sroa | FileCheck %s --check-prefix=LLVM --implicit-check-not=' @llvm.sqrt.' %}
+// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -fexperimental-strict-floating-point -ffp-exception-behavior=strict -fclangir -emit-cir  %s -disable-O0-optnone |                               FileCheck %s --check-prefix=CIR --implicit-check-not='cir.call_llvm_intrinsic "sqrt"' %}
 
 // REQUIRES: aarch64-registered-target
 
 #include <arm_neon.h>
 
-// LLVM-LABEL: @test_vfma_f64(
-// LLVM: call <1 x double> @llvm.experimental.constrained.fma.v1f64({{.*}}, metadata !"round.tonearest", metadata !"fpexcept.strict")
-// CIR-LABEL: cir.func {{.*}}@test_vfma_f64(
-// CIR: cir.fma %{{.*}}, %{{.*}}, %{{.*}} : !cir.vector<1 x !cir.double> {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
-float64x1_t test_vfma_f64(float64x1_t a, float64x1_t b, float64x1_t c) {
-  return vfma_f64(a, b, c);
-}
-
 // LLVM-LABEL: @test_vsqrtq_f64(
 // LLVM: call <2 x double> @llvm.experimental.constrained.sqrt.v2f64({{.*}}, metadata !"round.tonearest", metadata !"fpexcept.strict")
 // CIR-LABEL: cir.func {{.*}}@test_vsqrtq_f64(

>From fe6caa19ea2a2756f3d2a85b48621284018e5d14 Mon Sep 17 00:00:00 2001
From: Yair Ben Avraham <yairba at protonmail.com>
Date: Fri, 11 Sep 2026 06:56:06 +0300
Subject: [PATCH 08/16] [CIR][AArch64] Refine constrained FMA tests

Remove duplicate unconstrained coverage and share LLVM checks between the
classic and CIR pipelines. Drop the unnecessary Cyclone target, scope CIR
checks to emitted functions, and track operands through ABI conversions and
lane selection.
---
 .../AArch64/neon/fused-multiply-constrained.c | 121 ++++++++++--------
 1 file changed, 70 insertions(+), 51 deletions(-)

diff --git a/clang/test/CodeGen/AArch64/neon/fused-multiply-constrained.c b/clang/test/CodeGen/AArch64/neon/fused-multiply-constrained.c
index 25bc87b376e42..8920c40581e9f 100644
--- a/clang/test/CodeGen/AArch64/neon/fused-multiply-constrained.c
+++ b/clang/test/CodeGen/AArch64/neon/fused-multiply-constrained.c
@@ -1,75 +1,94 @@
-// NOTE: Assertions have been autogenerated by utils/update_cc_test_checks.py UTC_ARGS: --version 5
-// RUN: %clang_cc1_cg_arm64_neon -target-cpu cyclone -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,sroa | FileCheck %s --check-prefix=UNCONSTRAINED
-// RUN: %clang_cc1_cg_arm64_neon -target-cpu cyclone -ffp-exception-behavior=strict -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,sroa | FileCheck %s --check-prefix=CONSTRAINED
+// REQUIRES: aarch64-registered-target
+
+// RUN:                   %clang_cc1_cg_arm64_neon -ffp-exception-behavior=strict -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,sroa | FileCheck %s --check-prefix=LLVM
 // RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -fexperimental-strict-floating-point -ffp-exception-behavior=strict -fclangir -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,sroa | FileCheck %s --check-prefix=LLVM --implicit-check-not=' @llvm.fma.' %}
 // RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -fexperimental-strict-floating-point -ffp-exception-behavior=strict -fclangir -emit-cir  %s -disable-O0-optnone |                               FileCheck %s --check-prefix=CIR --implicit-check-not='cir.call_llvm_intrinsic "fma"' %}
 
-// REQUIRES: aarch64-registered-target
-
 #include <arm_neon.h>
 
 // LLVM-LABEL: @test_vfma_f64(
-// LLVM: call <1 x double> @llvm.experimental.constrained.fma.v1f64({{.*}}, metadata !"round.tonearest", metadata !"fpexcept.strict")
-// CIR-LABEL: cir.func {{.*}}@test_vfma_f64(
-// CIR: cir.fma %{{.*}}, %{{.*}}, %{{.*}} : !cir.vector<1 x !cir.double> {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
+// CIR-LABEL: @vfma_f64(
 float64x1_t test_vfma_f64(float64x1_t a, float64x1_t b, float64x1_t c) {
+// CIR: cir.fma %{{.*}}, %{{.*}}, %{{.*}} : !cir.vector<1 x !cir.double> {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
+
+// LLVM-SAME: <1 x double> {{.*}} [[A:%.*]], <1 x double> {{.*}} [[B:%.*]], <1 x double> {{.*}} [[C:%.*]]) {{.*}} {
+// LLVM: [[A_I:%.*]] = bitcast <1 x double> [[A]] to i64
+// LLVM: [[A_INSERT:%.*]] = insertelement <1 x i64> undef, i64 [[A_I]], i64 0
+// LLVM: [[B_I:%.*]] = bitcast <1 x double> [[B]] to i64
+// LLVM: [[B_INSERT:%.*]] = insertelement <1 x i64> undef, i64 [[B_I]], i64 0
+// LLVM: [[C_I:%.*]] = bitcast <1 x double> [[C]] to i64
+// LLVM: [[C_INSERT:%.*]] = insertelement <1 x i64> undef, i64 [[C_I]], i64 0
+// LLVM: [[A_BYTES:%.*]] = bitcast <1 x i64> [[A_INSERT]] to <8 x i8>
+// LLVM: [[B_BYTES:%.*]] = bitcast <1 x i64> [[B_INSERT]] to <8 x i8>
+// LLVM: [[C_BYTES:%.*]] = bitcast <1 x i64> [[C_INSERT]] to <8 x i8>
+// LLVM: [[A_CAST:%.*]] = bitcast <8 x i8> [[A_BYTES]] to <1 x double>
+// LLVM: [[B_CAST:%.*]] = bitcast <8 x i8> [[B_BYTES]] to <1 x double>
+// LLVM: [[C_CAST:%.*]] = bitcast <8 x i8> [[C_BYTES]] to <1 x double>
+// LLVM: [[FMA:%.*]] = call <1 x double> @llvm.experimental.constrained.fma.v1f64(<1 x double> [[B_CAST]], <1 x double> [[C_CAST]], <1 x double> [[A_CAST]], metadata !"round.tonearest", metadata !"fpexcept.strict")
+// LLVM: ret <1 x double> [[FMA]]
   return vfma_f64(a, b, c);
 }
 
-// CIR-LABEL: cir.func {{.*}}@test_vfma_laneq_f64(
-// CIR: cir.fma %{{.*}}, %{{.*}}, %{{.*}} : !cir.double {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
 // LLVM-LABEL: @test_vfma_laneq_f64(
-// LLVM: call double @llvm.experimental.constrained.fma.f64({{.*}}, metadata !"round.tonearest", metadata !"fpexcept.strict")
+// CIR-LABEL: @test_vfma_laneq_f64(
 float64x1_t test_vfma_laneq_f64(float64x1_t a, float64x1_t b,
-                                float64x2_t v) {
+                                 float64x2_t v) {
+// CIR: [[INDEX:%.*]] = cir.const #cir.int<0> : !u64i
+// CIR: [[LANE:%.*]] = cir.vec.extract %{{.*}}{{\[}}[[INDEX]] : !u64i] : !cir.vector<2 x !cir.double>
+// CIR: cir.fma %{{.*}}, [[LANE]], %{{.*}} : !cir.double {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
+
+// LLVM-SAME: <1 x double> {{.*}} [[A:%.*]], <1 x double> {{.*}} [[B:%.*]], <2 x double> {{.*}} [[V:%.*]]) {{.*}} {
+// LLVM: [[A_I:%.*]] = bitcast <1 x double> [[A]] to i64
+// LLVM: [[A_INSERT:%.*]] = insertelement <1 x i64> undef, i64 [[A_I]], i64 0
+// LLVM: [[B_I:%.*]] = bitcast <1 x double> [[B]] to i64
+// LLVM: [[B_INSERT:%.*]] = insertelement <1 x i64> undef, i64 [[B_I]], i64 0
+// LLVM: [[V_I:%.*]] = bitcast <2 x double> [[V]] to <2 x i64>
+// LLVM: [[A_BYTES:%.*]] = bitcast <1 x i64> [[A_INSERT]] to <8 x i8>
+// LLVM: [[B_BYTES:%.*]] = bitcast <1 x i64> [[B_INSERT]] to <8 x i8>
+// LLVM: [[V_BYTES:%.*]] = bitcast <2 x i64> [[V_I]] to <16 x i8>
+// LLVM: [[A_CAST:%.*]] = bitcast <8 x i8> [[A_BYTES]] to double
+// LLVM: [[B_CAST:%.*]] = bitcast <8 x i8> [[B_BYTES]] to double
+// LLVM: [[V_CAST:%.*]] = bitcast <16 x i8> [[V_BYTES]] to <2 x double>
+// LLVM: [[LANE:%.*]] = extractelement <2 x double> [[V_CAST]], i{{32|64}} 0
+// LLVM: [[FMA:%.*]] = call double @llvm.experimental.constrained.fma.f64(double [[B_CAST]], double [[LANE]], double [[A_CAST]], metadata !"round.tonearest", metadata !"fpexcept.strict")
+// LLVM: [[RESULT:%.*]] = bitcast double [[FMA]] to <1 x double>
+// LLVM: ret <1 x double> [[RESULT]]
   return vfma_laneq_f64(a, b, v, 0);
 }
 
-// UNCONSTRAINED-LABEL: define dso_local <2 x double> @test_vfmaq_laneq_f64(
-// UNCONSTRAINED-SAME: <2 x double> noundef [[A:%.*]], <2 x double> noundef [[B:%.*]], <2 x double> noundef [[V:%.*]]) #[[ATTR0:[0-9]+]] {
-// UNCONSTRAINED-NEXT:  [[ENTRY:.*:]]
-// UNCONSTRAINED-NEXT:    [[TMP0:%.*]] = bitcast <2 x double> [[A]] to <2 x i64>
-// UNCONSTRAINED-NEXT:    [[TMP1:%.*]] = bitcast <2 x double> [[B]] to <2 x i64>
-// UNCONSTRAINED-NEXT:    [[TMP2:%.*]] = bitcast <2 x double> [[V]] to <2 x i64>
-// UNCONSTRAINED-NEXT:    [[TMP3:%.*]] = bitcast <2 x i64> [[TMP0]] to <16 x i8>
-// UNCONSTRAINED-NEXT:    [[TMP4:%.*]] = bitcast <2 x i64> [[TMP1]] to <16 x i8>
-// UNCONSTRAINED-NEXT:    [[TMP5:%.*]] = bitcast <2 x i64> [[TMP2]] to <16 x i8>
-// UNCONSTRAINED-NEXT:    [[TMP6:%.*]] = bitcast <16 x i8> [[TMP3]] to <2 x double>
-// UNCONSTRAINED-NEXT:    [[TMP7:%.*]] = bitcast <16 x i8> [[TMP4]] to <2 x double>
-// UNCONSTRAINED-NEXT:    [[TMP8:%.*]] = bitcast <16 x i8> [[TMP5]] to <2 x double>
-// UNCONSTRAINED-NEXT:    [[LANE:%.*]] = shufflevector <2 x double> [[TMP8]], <2 x double> [[TMP8]], <2 x i32> <i32 1, i32 1>
-// UNCONSTRAINED-NEXT:    [[TMP9:%.*]] = call <2 x double> @llvm.fma.v2f64(<2 x double> [[LANE]], <2 x double> [[TMP7]], <2 x double> [[TMP6]])
-// UNCONSTRAINED-NEXT:    ret <2 x double> [[TMP9]]
-//
-// CONSTRAINED-LABEL: define dso_local <2 x double> @test_vfmaq_laneq_f64(
-// CONSTRAINED-SAME: <2 x double> noundef [[A:%.*]], <2 x double> noundef [[B:%.*]], <2 x double> noundef [[V:%.*]]) #[[ATTR0:[0-9]+]] {
-// CONSTRAINED-NEXT:  [[ENTRY:.*:]]
-// CONSTRAINED-NEXT:    [[TMP0:%.*]] = bitcast <2 x double> [[A]] to <2 x i64>
-// CONSTRAINED-NEXT:    [[TMP1:%.*]] = bitcast <2 x double> [[B]] to <2 x i64>
-// CONSTRAINED-NEXT:    [[TMP2:%.*]] = bitcast <2 x double> [[V]] to <2 x i64>
-// CONSTRAINED-NEXT:    [[TMP3:%.*]] = bitcast <2 x i64> [[TMP0]] to <16 x i8>
-// CONSTRAINED-NEXT:    [[TMP4:%.*]] = bitcast <2 x i64> [[TMP1]] to <16 x i8>
-// CONSTRAINED-NEXT:    [[TMP5:%.*]] = bitcast <2 x i64> [[TMP2]] to <16 x i8>
-// CONSTRAINED-NEXT:    [[TMP6:%.*]] = bitcast <16 x i8> [[TMP3]] to <2 x double>
-// CONSTRAINED-NEXT:    [[TMP7:%.*]] = bitcast <16 x i8> [[TMP4]] to <2 x double>
-// CONSTRAINED-NEXT:    [[TMP8:%.*]] = bitcast <16 x i8> [[TMP5]] to <2 x double>
-// CONSTRAINED-NEXT:    [[LANE:%.*]] = shufflevector <2 x double> [[TMP8]], <2 x double> [[TMP8]], <2 x i32> <i32 1, i32 1>
-// CONSTRAINED-NEXT:    [[TMP9:%.*]] = call <2 x double> @llvm.experimental.constrained.fma.v2f64(<2 x double> [[LANE]], <2 x double> [[TMP7]], <2 x double> [[TMP6]], metadata !"round.tonearest", metadata !"fpexcept.strict") #[[ATTR2:[0-9]+]]
-// CONSTRAINED-NEXT:    ret <2 x double> [[TMP9]]
-//
-// CIR-LABEL: cir.func {{.*}}@test_vfmaq_laneq_f64(
-// CIR: cir.fma %{{.*}}, %{{.*}}, %{{.*}} : !cir.vector<2 x !cir.double> {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
 // LLVM-LABEL: @test_vfmaq_laneq_f64(
-// LLVM: call <2 x double> @llvm.experimental.constrained.fma.v2f64({{.*}}, metadata !"round.tonearest", metadata !"fpexcept.strict")
+// CIR-LABEL: @test_vfmaq_laneq_f64(
 float64x2_t test_vfmaq_laneq_f64(float64x2_t a, float64x2_t b,
-                                 float64x2_t v) {
+                                  float64x2_t v) {
+// CIR: [[LANE:%.*]] = cir.vec.shuffle(%{{.*}}, %{{.*}} : !cir.vector<2 x !cir.double>) [#cir.int<1> : !s32i, #cir.int<1> : !s32i] : !cir.vector<2 x !cir.double>
+// CIR: cir.fma [[LANE]], %{{.*}}, %{{.*}} : !cir.vector<2 x !cir.double> {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
+
+// LLVM-SAME: <2 x double> {{.*}} [[A:%.*]], <2 x double> {{.*}} [[B:%.*]], <2 x double> {{.*}} [[V:%.*]]) {{.*}} {
+// LLVM: [[A_I:%.*]] = bitcast <2 x double> [[A]] to <2 x i64>
+// LLVM: [[B_I:%.*]] = bitcast <2 x double> [[B]] to <2 x i64>
+// LLVM: [[V_I:%.*]] = bitcast <2 x double> [[V]] to <2 x i64>
+// LLVM: [[A_BYTES:%.*]] = bitcast <2 x i64> [[A_I]] to <16 x i8>
+// LLVM: [[B_BYTES:%.*]] = bitcast <2 x i64> [[B_I]] to <16 x i8>
+// LLVM: [[V_BYTES:%.*]] = bitcast <2 x i64> [[V_I]] to <16 x i8>
+// LLVM: [[A_CAST:%.*]] = bitcast <16 x i8> [[A_BYTES]] to <2 x double>
+// LLVM: [[B_CAST:%.*]] = bitcast <16 x i8> [[B_BYTES]] to <2 x double>
+// LLVM: [[V_CAST:%.*]] = bitcast <16 x i8> [[V_BYTES]] to <2 x double>
+// LLVM: [[LANE:%.*]] = shufflevector <2 x double> [[V_CAST]], <2 x double> {{.*}}, <2 x i32> <i32 1, i32 1>
+// LLVM: [[FMA:%.*]] = call <2 x double> @llvm.experimental.constrained.fma.v2f64(<2 x double> [[LANE]], <2 x double> [[B_CAST]], <2 x double> [[A_CAST]], metadata !"round.tonearest", metadata !"fpexcept.strict")
+// LLVM: ret <2 x double> [[FMA]]
   return vfmaq_laneq_f64(a, b, v, 1);
 }
 
-// CIR-LABEL: cir.func {{.*}}@test_vfmas_lane_f32(
-// CIR: cir.fma %{{.*}}, %{{.*}}, %{{.*}} : !cir.float {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
 // LLVM-LABEL: @test_vfmas_lane_f32(
-// LLVM: call float @llvm.experimental.constrained.fma.f32({{.*}}, metadata !"round.tonearest", metadata !"fpexcept.strict")
+// CIR-LABEL: @test_vfmas_lane_f32(
 float32_t test_vfmas_lane_f32(float32_t a, float32_t b, float32x2_t c) {
+// CIR: [[INDEX:%.*]] = cir.const #cir.int<1> : !u64i
+// CIR: [[LANE:%.*]] = cir.vec.extract %{{.*}}{{\[}}[[INDEX]] : !u64i] : !cir.vector<2 x !cir.float>
+// CIR: cir.fma %{{.*}}, [[LANE]], %{{.*}} : !cir.float {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
+
+// LLVM-SAME: float {{.*}} [[A:%.*]], float {{.*}} [[B:%.*]], <2 x float> {{.*}} [[C:%.*]]) {{.*}} {
+// LLVM: [[LANE:%.*]] = extractelement <2 x float> [[C]], i{{32|64}} 1
+// LLVM: [[FMA:%.*]] = call float @llvm.experimental.constrained.fma.f32(float [[B]], float [[LANE]], float [[A]], metadata !"round.tonearest", metadata !"fpexcept.strict")
+// LLVM: ret float [[FMA]]
   return vfmas_lane_f32(a, b, c, 1);
 }

>From a362a0a257a81948503c96a800cd2b550f0f4f75 Mon Sep 17 00:00:00 2001
From: Yair Ben Avraham <yairba at protonmail.com>
Date: Fri, 11 Sep 2026 06:56:14 +0300
Subject: [PATCH 09/16] [CIR][AArch64] Move f16 constrained tests into Neon

Place the new CIR coverage beside its Neon test families. Preserve
command-line strict and pragma-overridden maytrap modes, add vfmaq_lane_f16,
and track operands through ABI conversions and lane selection.
---
 .../AArch64/neon/fullfp16-constrained.c       |  34 ++++
 .../fused-multiple-fullfp16-constrained.c     | 158 ++++++++++++++++++
 .../AArch64/neon/intrinsics-constrained.c     |  52 +++++-
 .../v8.2a-neon-intrinsics-constrained.c       |  91 ----------
 4 files changed, 238 insertions(+), 97 deletions(-)
 create mode 100644 clang/test/CodeGen/AArch64/neon/fullfp16-constrained.c
 create mode 100644 clang/test/CodeGen/AArch64/neon/fused-multiple-fullfp16-constrained.c

diff --git a/clang/test/CodeGen/AArch64/neon/fullfp16-constrained.c b/clang/test/CodeGen/AArch64/neon/fullfp16-constrained.c
new file mode 100644
index 0000000000000..8c9c1b61899c6
--- /dev/null
+++ b/clang/test/CodeGen/AArch64/neon/fullfp16-constrained.c
@@ -0,0 +1,34 @@
+// REQUIRES: aarch64-registered-target
+
+// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 -fexperimental-strict-floating-point -ffp-exception-behavior=maytrap -DEXCEPT=1 -fclangir -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,simplifycfg | FileCheck %s --check-prefix=LLVM --implicit-check-not=fpexcept.maytrap --implicit-check-not=' @llvm.fma.' --implicit-check-not=' @llvm.sqrt.' %}
+// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 -fexperimental-strict-floating-point -ffp-exception-behavior=maytrap -DEXCEPT=1 -fclangir -emit-cir  %s -disable-O0-optnone |                                      FileCheck %s --check-prefix=CIR --implicit-check-not='except_mode = maytrap' --implicit-check-not='cir.call_llvm_intrinsic "fma"' --implicit-check-not='cir.call_llvm_intrinsic "sqrt"' %}
+// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 -fexperimental-strict-floating-point -ffp-exception-behavior=strict             -fclangir -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,simplifycfg | FileCheck %s --check-prefix=LLVM --implicit-check-not=' @llvm.fma.' --implicit-check-not=' @llvm.sqrt.' %}
+// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 -fexperimental-strict-floating-point -ffp-exception-behavior=strict             -fclangir -emit-cir  %s -disable-O0-optnone |                                      FileCheck %s --check-prefix=CIR --implicit-check-not='cir.call_llvm_intrinsic "fma"' --implicit-check-not='cir.call_llvm_intrinsic "sqrt"' %}
+
+#if EXCEPT
+#pragma float_control(except, on)
+#endif
+
+#include <arm_fp16.h>
+
+// LLVM-LABEL: @test_vsqrth_f16(
+// CIR-LABEL: @test_vsqrth_f16(
+float16_t test_vsqrth_f16(float16_t a) {
+// CIR: cir.sqrt %{{.*}} : !cir.f16 {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
+
+// LLVM-SAME: half {{.*}} [[A:%.*]]) {{.*}} {
+// LLVM: [[SQRT:%.*]] = call half @llvm.experimental.constrained.sqrt.f16(half [[A]], metadata !"round.tonearest", metadata !"fpexcept.strict")
+// LLVM: ret half [[SQRT]]
+  return vsqrth_f16(a);
+}
+
+// LLVM-LABEL: @test_vfmah_f16(
+// CIR-LABEL: @test_vfmah_f16(
+float16_t test_vfmah_f16(float16_t a, float16_t b, float16_t c) {
+// CIR: cir.fma %{{.*}}, %{{.*}}, %{{.*}} : !cir.f16 {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
+
+// LLVM-SAME: half {{.*}} [[A:%.*]], half {{.*}} [[B:%.*]], half {{.*}} [[C:%.*]]) {{.*}} {
+// LLVM: [[FMA:%.*]] = call half @llvm.experimental.constrained.fma.f16(half [[B]], half [[C]], half [[A]], metadata !"round.tonearest", metadata !"fpexcept.strict")
+// LLVM: ret half [[FMA]]
+  return vfmah_f16(a, b, c);
+}
diff --git a/clang/test/CodeGen/AArch64/neon/fused-multiple-fullfp16-constrained.c b/clang/test/CodeGen/AArch64/neon/fused-multiple-fullfp16-constrained.c
new file mode 100644
index 0000000000000..907cf54b91134
--- /dev/null
+++ b/clang/test/CodeGen/AArch64/neon/fused-multiple-fullfp16-constrained.c
@@ -0,0 +1,158 @@
+// REQUIRES: aarch64-registered-target
+
+// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 -fexperimental-strict-floating-point -ffp-exception-behavior=maytrap -DEXCEPT=1 -fclangir -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,sroa | FileCheck %s --check-prefix=LLVM --implicit-check-not=fpexcept.maytrap --implicit-check-not=' @llvm.fma.' %}
+// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 -fexperimental-strict-floating-point -ffp-exception-behavior=maytrap -DEXCEPT=1 -fclangir -emit-cir  %s -disable-O0-optnone |                               FileCheck %s --check-prefix=CIR --implicit-check-not='except_mode = maytrap' --implicit-check-not='cir.call_llvm_intrinsic "fma"' %}
+// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 -fexperimental-strict-floating-point -ffp-exception-behavior=strict             -fclangir -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,sroa | FileCheck %s --check-prefix=LLVM --implicit-check-not=' @llvm.fma.' %}
+// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 -fexperimental-strict-floating-point -ffp-exception-behavior=strict             -fclangir -emit-cir  %s -disable-O0-optnone |                               FileCheck %s --check-prefix=CIR --implicit-check-not='cir.call_llvm_intrinsic "fma"' %}
+
+#if EXCEPT
+#pragma float_control(except, on)
+#endif
+
+#include <arm_neon.h>
+
+// LLVM-LABEL: @test_vfma_f16(
+// CIR-LABEL: @vfma_f16(
+float16x4_t test_vfma_f16(float16x4_t a, float16x4_t b, float16x4_t c) {
+// CIR: cir.fma %{{.*}}, %{{.*}}, %{{.*}} : !cir.vector<4 x !cir.f16> {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
+
+// LLVM-SAME: <4 x half> {{.*}} [[A:%.*]], <4 x half> {{.*}} [[B:%.*]], <4 x half> {{.*}} [[C:%.*]]) {{.*}} {
+// LLVM: [[A_I:%.*]] = bitcast <4 x half> [[A]] to <4 x i16>
+// LLVM: [[B_I:%.*]] = bitcast <4 x half> [[B]] to <4 x i16>
+// LLVM: [[C_I:%.*]] = bitcast <4 x half> [[C]] to <4 x i16>
+// LLVM: [[A_BYTES:%.*]] = bitcast <4 x i16> [[A_I]] to <8 x i8>
+// LLVM: [[B_BYTES:%.*]] = bitcast <4 x i16> [[B_I]] to <8 x i8>
+// LLVM: [[C_BYTES:%.*]] = bitcast <4 x i16> [[C_I]] to <8 x i8>
+// LLVM: [[A_CAST:%.*]] = bitcast <8 x i8> [[A_BYTES]] to <4 x half>
+// LLVM: [[B_CAST:%.*]] = bitcast <8 x i8> [[B_BYTES]] to <4 x half>
+// LLVM: [[C_CAST:%.*]] = bitcast <8 x i8> [[C_BYTES]] to <4 x half>
+// LLVM: [[FMA:%.*]] = call <4 x half> @llvm.experimental.constrained.fma.v4f16(<4 x half> [[B_CAST]], <4 x half> [[C_CAST]], <4 x half> [[A_CAST]], metadata !"round.tonearest", metadata !"fpexcept.strict")
+// LLVM: ret <4 x half> [[FMA]]
+  return vfma_f16(a, b, c);
+}
+
+// LLVM-LABEL: @test_vfmaq_f16(
+// CIR-LABEL: @vfmaq_f16(
+float16x8_t test_vfmaq_f16(float16x8_t a, float16x8_t b, float16x8_t c) {
+// CIR: cir.fma %{{.*}}, %{{.*}}, %{{.*}} : !cir.vector<8 x !cir.f16> {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
+
+// LLVM-SAME: <8 x half> {{.*}} [[A:%.*]], <8 x half> {{.*}} [[B:%.*]], <8 x half> {{.*}} [[C:%.*]]) {{.*}} {
+// LLVM: [[A_I:%.*]] = bitcast <8 x half> [[A]] to <8 x i16>
+// LLVM: [[B_I:%.*]] = bitcast <8 x half> [[B]] to <8 x i16>
+// LLVM: [[C_I:%.*]] = bitcast <8 x half> [[C]] to <8 x i16>
+// LLVM: [[A_BYTES:%.*]] = bitcast <8 x i16> [[A_I]] to <16 x i8>
+// LLVM: [[B_BYTES:%.*]] = bitcast <8 x i16> [[B_I]] to <16 x i8>
+// LLVM: [[C_BYTES:%.*]] = bitcast <8 x i16> [[C_I]] to <16 x i8>
+// LLVM: [[A_CAST:%.*]] = bitcast <16 x i8> [[A_BYTES]] to <8 x half>
+// LLVM: [[B_CAST:%.*]] = bitcast <16 x i8> [[B_BYTES]] to <8 x half>
+// LLVM: [[C_CAST:%.*]] = bitcast <16 x i8> [[C_BYTES]] to <8 x half>
+// LLVM: [[FMA:%.*]] = call <8 x half> @llvm.experimental.constrained.fma.v8f16(<8 x half> [[B_CAST]], <8 x half> [[C_CAST]], <8 x half> [[A_CAST]], metadata !"round.tonearest", metadata !"fpexcept.strict")
+// LLVM: ret <8 x half> [[FMA]]
+  return vfmaq_f16(a, b, c);
+}
+
+// LLVM-LABEL: @test_vfma_lane_f16(
+// CIR-LABEL: @test_vfma_lane_f16(
+float16x4_t test_vfma_lane_f16(float16x4_t a, float16x4_t b,
+                                float16x4_t c) {
+// CIR: [[LANE:%.*]] = cir.vec.shuffle(%{{.*}}, %{{.*}} : !cir.vector<4 x !cir.f16>) [#cir.int<3> : !s32i, #cir.int<3> : !s32i, #cir.int<3> : !s32i, #cir.int<3> : !s32i] : !cir.vector<4 x !cir.f16>
+// CIR: cir.fma %{{.*}}, [[LANE]], %{{.*}} : !cir.vector<4 x !cir.f16> {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
+
+// LLVM-SAME: <4 x half> {{.*}} [[A:%.*]], <4 x half> {{.*}} [[B:%.*]], <4 x half> {{.*}} [[C:%.*]]) {{.*}} {
+// LLVM: [[A_I:%.*]] = bitcast <4 x half> [[A]] to <4 x i16>
+// LLVM: [[B_I:%.*]] = bitcast <4 x half> [[B]] to <4 x i16>
+// LLVM: [[C_I:%.*]] = bitcast <4 x half> [[C]] to <4 x i16>
+// LLVM: [[A_BYTES:%.*]] = bitcast <4 x i16> [[A_I]] to <8 x i8>
+// LLVM: [[B_BYTES:%.*]] = bitcast <4 x i16> [[B_I]] to <8 x i8>
+// LLVM: [[C_BYTES:%.*]] = bitcast <4 x i16> [[C_I]] to <8 x i8>
+// LLVM: [[A_CAST:%.*]] = bitcast <8 x i8> [[A_BYTES]] to <4 x half>
+// LLVM: [[B_CAST:%.*]] = bitcast <8 x i8> [[B_BYTES]] to <4 x half>
+// LLVM: [[C_CAST:%.*]] = bitcast <8 x i8> [[C_BYTES]] to <4 x half>
+// LLVM: [[LANE:%.*]] = shufflevector <4 x half> [[C_CAST]], <4 x half> {{.*}}, <4 x i32> <i32 3, i32 3, i32 3, i32 3>
+// LLVM: [[FMA:%.*]] = call <4 x half> @llvm.experimental.constrained.fma.v4f16(<4 x half> [[B_CAST]], <4 x half> [[LANE]], <4 x half> [[A_CAST]], metadata !"round.tonearest", metadata !"fpexcept.strict")
+// LLVM: ret <4 x half> [[FMA]]
+  return vfma_lane_f16(a, b, c, 3);
+}
+
+// LLVM-LABEL: @test_vfmaq_lane_f16(
+// CIR-LABEL: @test_vfmaq_lane_f16(
+float16x8_t test_vfmaq_lane_f16(float16x8_t a, float16x8_t b,
+                                 float16x4_t c) {
+// CIR: [[LANE:%.*]] = cir.vec.shuffle(%{{.*}}, %{{.*}} : !cir.vector<4 x !cir.f16>) [#cir.int<3> : !s32i, #cir.int<3> : !s32i, #cir.int<3> : !s32i, #cir.int<3> : !s32i, #cir.int<3> : !s32i, #cir.int<3> : !s32i, #cir.int<3> : !s32i, #cir.int<3> : !s32i] : !cir.vector<8 x !cir.f16>
+// CIR: cir.fma %{{.*}}, [[LANE]], %{{.*}} : !cir.vector<8 x !cir.f16> {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
+
+// LLVM-SAME: <8 x half> {{.*}} [[A:%.*]], <8 x half> {{.*}} [[B:%.*]], <4 x half> {{.*}} [[C:%.*]]) {{.*}} {
+// LLVM: [[A_I:%.*]] = bitcast <8 x half> [[A]] to <8 x i16>
+// LLVM: [[B_I:%.*]] = bitcast <8 x half> [[B]] to <8 x i16>
+// LLVM: [[C_I:%.*]] = bitcast <4 x half> [[C]] to <4 x i16>
+// LLVM: [[A_BYTES:%.*]] = bitcast <8 x i16> [[A_I]] to <16 x i8>
+// LLVM: [[B_BYTES:%.*]] = bitcast <8 x i16> [[B_I]] to <16 x i8>
+// LLVM: [[C_BYTES:%.*]] = bitcast <4 x i16> [[C_I]] to <8 x i8>
+// LLVM: [[A_CAST:%.*]] = bitcast <16 x i8> [[A_BYTES]] to <8 x half>
+// LLVM: [[B_CAST:%.*]] = bitcast <16 x i8> [[B_BYTES]] to <8 x half>
+// LLVM: [[C_CAST:%.*]] = bitcast <8 x i8> [[C_BYTES]] to <4 x half>
+// LLVM: [[LANE:%.*]] = shufflevector <4 x half> [[C_CAST]], <4 x half> {{.*}}, <8 x i32> <i32 3, i32 3, i32 3, i32 3, i32 3, i32 3, i32 3, i32 3>
+// LLVM: [[FMA:%.*]] = call <8 x half> @llvm.experimental.constrained.fma.v8f16(<8 x half> [[B_CAST]], <8 x half> [[LANE]], <8 x half> [[A_CAST]], metadata !"round.tonearest", metadata !"fpexcept.strict")
+// LLVM: ret <8 x half> [[FMA]]
+  return vfmaq_lane_f16(a, b, c, 3);
+}
+
+// LLVM-LABEL: @test_vfma_laneq_f16(
+// CIR-LABEL: @test_vfma_laneq_f16(
+float16x4_t test_vfma_laneq_f16(float16x4_t a, float16x4_t b,
+                                 float16x8_t c) {
+// CIR: [[LANE:%.*]] = cir.vec.shuffle(%{{.*}}, %{{.*}} : !cir.vector<8 x !cir.f16>) [#cir.int<7> : !s32i, #cir.int<7> : !s32i, #cir.int<7> : !s32i, #cir.int<7> : !s32i] : !cir.vector<4 x !cir.f16>
+// CIR: cir.fma [[LANE]], %{{.*}}, %{{.*}} : !cir.vector<4 x !cir.f16> {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
+
+// LLVM-SAME: <4 x half> {{.*}} [[A:%.*]], <4 x half> {{.*}} [[B:%.*]], <8 x half> {{.*}} [[C:%.*]]) {{.*}} {
+// LLVM: [[A_I:%.*]] = bitcast <4 x half> [[A]] to <4 x i16>
+// LLVM: [[B_I:%.*]] = bitcast <4 x half> [[B]] to <4 x i16>
+// LLVM: [[C_I:%.*]] = bitcast <8 x half> [[C]] to <8 x i16>
+// LLVM: [[A_BYTES:%.*]] = bitcast <4 x i16> [[A_I]] to <8 x i8>
+// LLVM: [[B_BYTES:%.*]] = bitcast <4 x i16> [[B_I]] to <8 x i8>
+// LLVM: [[C_BYTES:%.*]] = bitcast <8 x i16> [[C_I]] to <16 x i8>
+// LLVM: [[A_CAST:%.*]] = bitcast <8 x i8> [[A_BYTES]] to <4 x half>
+// LLVM: [[B_CAST:%.*]] = bitcast <8 x i8> [[B_BYTES]] to <4 x half>
+// LLVM: [[C_CAST:%.*]] = bitcast <16 x i8> [[C_BYTES]] to <8 x half>
+// LLVM: [[LANE:%.*]] = shufflevector <8 x half> [[C_CAST]], <8 x half> {{.*}}, <4 x i32> <i32 7, i32 7, i32 7, i32 7>
+// LLVM: [[FMA:%.*]] = call <4 x half> @llvm.experimental.constrained.fma.v4f16(<4 x half> [[LANE]], <4 x half> [[B_CAST]], <4 x half> [[A_CAST]], metadata !"round.tonearest", metadata !"fpexcept.strict")
+// LLVM: ret <4 x half> [[FMA]]
+  return vfma_laneq_f16(a, b, c, 7);
+}
+
+// LLVM-LABEL: @test_vfmaq_laneq_f16(
+// CIR-LABEL: @test_vfmaq_laneq_f16(
+float16x8_t test_vfmaq_laneq_f16(float16x8_t a, float16x8_t b,
+                                  float16x8_t c) {
+// CIR: [[LANE:%.*]] = cir.vec.shuffle(%{{.*}}, %{{.*}} : !cir.vector<8 x !cir.f16>) [#cir.int<7> : !s32i, #cir.int<7> : !s32i, #cir.int<7> : !s32i, #cir.int<7> : !s32i, #cir.int<7> : !s32i, #cir.int<7> : !s32i, #cir.int<7> : !s32i, #cir.int<7> : !s32i] : !cir.vector<8 x !cir.f16>
+// CIR: cir.fma [[LANE]], %{{.*}}, %{{.*}} : !cir.vector<8 x !cir.f16> {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
+
+// LLVM-SAME: <8 x half> {{.*}} [[A:%.*]], <8 x half> {{.*}} [[B:%.*]], <8 x half> {{.*}} [[C:%.*]]) {{.*}} {
+// LLVM: [[A_I:%.*]] = bitcast <8 x half> [[A]] to <8 x i16>
+// LLVM: [[B_I:%.*]] = bitcast <8 x half> [[B]] to <8 x i16>
+// LLVM: [[C_I:%.*]] = bitcast <8 x half> [[C]] to <8 x i16>
+// LLVM: [[A_BYTES:%.*]] = bitcast <8 x i16> [[A_I]] to <16 x i8>
+// LLVM: [[B_BYTES:%.*]] = bitcast <8 x i16> [[B_I]] to <16 x i8>
+// LLVM: [[C_BYTES:%.*]] = bitcast <8 x i16> [[C_I]] to <16 x i8>
+// LLVM: [[A_CAST:%.*]] = bitcast <16 x i8> [[A_BYTES]] to <8 x half>
+// LLVM: [[B_CAST:%.*]] = bitcast <16 x i8> [[B_BYTES]] to <8 x half>
+// LLVM: [[C_CAST:%.*]] = bitcast <16 x i8> [[C_BYTES]] to <8 x half>
+// LLVM: [[LANE:%.*]] = shufflevector <8 x half> [[C_CAST]], <8 x half> {{.*}}, <8 x i32> <i32 7, i32 7, i32 7, i32 7, i32 7, i32 7, i32 7, i32 7>
+// LLVM: [[FMA:%.*]] = call <8 x half> @llvm.experimental.constrained.fma.v8f16(<8 x half> [[LANE]], <8 x half> [[B_CAST]], <8 x half> [[A_CAST]], metadata !"round.tonearest", metadata !"fpexcept.strict")
+// LLVM: ret <8 x half> [[FMA]]
+  return vfmaq_laneq_f16(a, b, c, 7);
+}
+
+// LLVM-LABEL: @test_vfmah_lane_f16(
+// CIR-LABEL: @test_vfmah_lane_f16(
+float16_t test_vfmah_lane_f16(float16_t a, float16_t b, float16x4_t c) {
+// CIR: [[INDEX:%.*]] = cir.const #cir.int<3> : !u64i
+// CIR: [[LANE:%.*]] = cir.vec.extract %{{.*}}{{\[}}[[INDEX]] : !u64i] : !cir.vector<4 x !cir.f16>
+// CIR: cir.fma %{{.*}}, [[LANE]], %{{.*}} : !cir.f16 {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
+
+// LLVM-SAME: half {{.*}} [[A:%.*]], half {{.*}} [[B:%.*]], <4 x half> {{.*}} [[C:%.*]]) {{.*}} {
+// LLVM: [[LANE:%.*]] = extractelement <4 x half> [[C]], i{{32|64}} 3
+// LLVM: [[FMA:%.*]] = call half @llvm.experimental.constrained.fma.f16(half [[B]], half [[LANE]], half [[A]], metadata !"round.tonearest", metadata !"fpexcept.strict")
+// LLVM: ret half [[FMA]]
+  return vfmah_lane_f16(a, b, c, 3);
+}
diff --git a/clang/test/CodeGen/AArch64/neon/intrinsics-constrained.c b/clang/test/CodeGen/AArch64/neon/intrinsics-constrained.c
index d63f6a3373ebb..212cc89ad354d 100644
--- a/clang/test/CodeGen/AArch64/neon/intrinsics-constrained.c
+++ b/clang/test/CodeGen/AArch64/neon/intrinsics-constrained.c
@@ -1,14 +1,54 @@
-// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -fexperimental-strict-floating-point -ffp-exception-behavior=strict -fclangir -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,sroa | FileCheck %s --check-prefix=LLVM --implicit-check-not=' @llvm.sqrt.' %}
-// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -fexperimental-strict-floating-point -ffp-exception-behavior=strict -fclangir -emit-cir  %s -disable-O0-optnone |                               FileCheck %s --check-prefix=CIR --implicit-check-not='cir.call_llvm_intrinsic "sqrt"' %}
-
 // REQUIRES: aarch64-registered-target
 
+// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 -fexperimental-strict-floating-point -ffp-exception-behavior=maytrap -DEXCEPT=1 -fclangir -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,sroa | FileCheck %s --check-prefix=LLVM --implicit-check-not=fpexcept.maytrap --implicit-check-not=' @llvm.sqrt.' %}
+// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 -fexperimental-strict-floating-point -ffp-exception-behavior=maytrap -DEXCEPT=1 -fclangir -emit-cir  %s -disable-O0-optnone |                               FileCheck %s --check-prefix=CIR --implicit-check-not='except_mode = maytrap' --implicit-check-not='cir.call_llvm_intrinsic "sqrt"' %}
+// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 -fexperimental-strict-floating-point -ffp-exception-behavior=strict             -fclangir -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,sroa | FileCheck %s --check-prefix=LLVM --implicit-check-not=' @llvm.sqrt.' %}
+// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 -fexperimental-strict-floating-point -ffp-exception-behavior=strict             -fclangir -emit-cir  %s -disable-O0-optnone |                               FileCheck %s --check-prefix=CIR --implicit-check-not='cir.call_llvm_intrinsic "sqrt"' %}
+
+#if EXCEPT
+#pragma float_control(except, on)
+#endif
+
 #include <arm_neon.h>
 
+// LLVM-LABEL: @test_vsqrt_f16(
+// CIR-LABEL: @vsqrt_f16(
+float16x4_t test_vsqrt_f16(float16x4_t a) {
+// CIR: cir.sqrt %{{.*}} : !cir.vector<4 x !cir.f16> {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
+
+// LLVM-SAME: <4 x half> {{.*}} [[A:%.*]]) {{.*}} {
+// LLVM: [[A_I:%.*]] = bitcast <4 x half> [[A]] to <4 x i16>
+// LLVM: [[A_BYTES:%.*]] = bitcast <4 x i16> [[A_I]] to <8 x i8>
+// LLVM: [[A_CAST:%.*]] = bitcast <8 x i8> [[A_BYTES]] to <4 x half>
+// LLVM: [[SQRT:%.*]] = call <4 x half> @llvm.experimental.constrained.sqrt.v4f16(<4 x half> [[A_CAST]], metadata !"round.tonearest", metadata !"fpexcept.strict")
+// LLVM: ret <4 x half> [[SQRT]]
+  return vsqrt_f16(a);
+}
+
+// LLVM-LABEL: @test_vsqrtq_f16(
+// CIR-LABEL: @vsqrtq_f16(
+float16x8_t test_vsqrtq_f16(float16x8_t a) {
+// CIR: cir.sqrt %{{.*}} : !cir.vector<8 x !cir.f16> {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
+
+// LLVM-SAME: <8 x half> {{.*}} [[A:%.*]]) {{.*}} {
+// LLVM: [[A_I:%.*]] = bitcast <8 x half> [[A]] to <8 x i16>
+// LLVM: [[A_BYTES:%.*]] = bitcast <8 x i16> [[A_I]] to <16 x i8>
+// LLVM: [[A_CAST:%.*]] = bitcast <16 x i8> [[A_BYTES]] to <8 x half>
+// LLVM: [[SQRT:%.*]] = call <8 x half> @llvm.experimental.constrained.sqrt.v8f16(<8 x half> [[A_CAST]], metadata !"round.tonearest", metadata !"fpexcept.strict")
+// LLVM: ret <8 x half> [[SQRT]]
+  return vsqrtq_f16(a);
+}
+
 // LLVM-LABEL: @test_vsqrtq_f64(
-// LLVM: call <2 x double> @llvm.experimental.constrained.sqrt.v2f64({{.*}}, metadata !"round.tonearest", metadata !"fpexcept.strict")
-// CIR-LABEL: cir.func {{.*}}@test_vsqrtq_f64(
-// CIR: cir.sqrt %{{.*}} : !cir.vector<2 x !cir.double> {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
+// CIR-LABEL: @vsqrtq_f64(
 float64x2_t test_vsqrtq_f64(float64x2_t a) {
+// CIR: cir.sqrt %{{.*}} : !cir.vector<2 x !cir.double> {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
+
+// LLVM-SAME: <2 x double> {{.*}} [[A:%.*]]) {{.*}} {
+// LLVM: [[A_I:%.*]] = bitcast <2 x double> [[A]] to <2 x i64>
+// LLVM: [[A_BYTES:%.*]] = bitcast <2 x i64> [[A_I]] to <16 x i8>
+// LLVM: [[A_CAST:%.*]] = bitcast <16 x i8> [[A_BYTES]] to <2 x double>
+// LLVM: [[SQRT:%.*]] = call <2 x double> @llvm.experimental.constrained.sqrt.v2f64(<2 x double> [[A_CAST]], metadata !"round.tonearest", metadata !"fpexcept.strict")
+// LLVM: ret <2 x double> [[SQRT]]
   return vsqrtq_f64(a);
 }
diff --git a/clang/test/CodeGen/AArch64/v8.2a-neon-intrinsics-constrained.c b/clang/test/CodeGen/AArch64/v8.2a-neon-intrinsics-constrained.c
index e9aece9dfa75f..dac8b931ff210 100644
--- a/clang/test/CodeGen/AArch64/v8.2a-neon-intrinsics-constrained.c
+++ b/clang/test/CodeGen/AArch64/v8.2a-neon-intrinsics-constrained.c
@@ -8,24 +8,6 @@
 // RUN: -flax-vector-conversions=none -disable-O0-optnone -emit-llvm -o - %s \
 // RUN: | opt -S -passes=mem2reg,sroa \
 // RUN: | FileCheck --check-prefix=CONSTRAINED --implicit-check-not=fpexcept.maytrap %s
-// RUN: %if cir-enabled %{%clang_cc1 -triple arm64-none-linux-gnu -target-feature +neon -target-feature +fullfp16 -target-feature +v8.2a \
-// RUN: -fexperimental-strict-floating-point -ffp-exception-behavior=maytrap -DEXCEPT=1 \
-// RUN: -flax-vector-conversions=none -disable-O0-optnone -fclangir -emit-cir -o - %s \
-// RUN: | FileCheck --check-prefix=CIR --implicit-check-not='cir.call_llvm_intrinsic "fma"' --implicit-check-not='cir.call_llvm_intrinsic "sqrt"' %s %}
-// RUN: %if cir-enabled %{%clang_cc1 -triple arm64-none-linux-gnu -target-feature +neon -target-feature +fullfp16 -target-feature +v8.2a \
-// RUN: -fexperimental-strict-floating-point -ffp-exception-behavior=maytrap -DEXCEPT=1 \
-// RUN: -flax-vector-conversions=none -disable-O0-optnone -fclangir -emit-llvm -o - %s \
-// RUN: | opt -S -passes=mem2reg,sroa \
-// RUN: | FileCheck --check-prefix=LLVM --implicit-check-not=fpexcept.maytrap --implicit-check-not=' @llvm.fma.' --implicit-check-not=' @llvm.sqrt.' %s %}
-// RUN: %if cir-enabled %{%clang_cc1 -triple arm64-none-linux-gnu -target-feature +neon -target-feature +fullfp16 -target-feature +v8.2a \
-// RUN: -fexperimental-strict-floating-point -ffp-exception-behavior=strict \
-// RUN: -flax-vector-conversions=none -disable-O0-optnone -fclangir -emit-cir -o - %s \
-// RUN: | FileCheck --check-prefix=CIR --implicit-check-not='cir.call_llvm_intrinsic "fma"' --implicit-check-not='cir.call_llvm_intrinsic "sqrt"' %s %}
-// RUN: %if cir-enabled %{%clang_cc1 -triple arm64-none-linux-gnu -target-feature +neon -target-feature +fullfp16 -target-feature +v8.2a \
-// RUN: -fexperimental-strict-floating-point -ffp-exception-behavior=strict \
-// RUN: -flax-vector-conversions=none -disable-O0-optnone -fclangir -emit-llvm -o - %s \
-// RUN: | opt -S -passes=mem2reg,sroa \
-// RUN: | FileCheck --check-prefix=LLVM --implicit-check-not=' @llvm.fma.' --implicit-check-not=' @llvm.sqrt.' %s %}
 
 // REQUIRES: aarch64-registered-target
 
@@ -37,29 +19,8 @@
 #pragma float_control(except, on)
 #endif
 
-#include <arm_fp16.h>
 #include <arm_neon.h>
 
-// UNCONSTRAINED-LABEL: define dso_local half @test_vsqrth_f16(
-// UNCONSTRAINED-SAME: half noundef [[A:%.*]]) #[[ATTR0:[0-9]+]] {
-// UNCONSTRAINED-NEXT:  [[ENTRY:.*:]]
-// UNCONSTRAINED-NEXT:    [[SQR:%.*]] = call half @llvm.sqrt.f16(half [[A]])
-// UNCONSTRAINED-NEXT:    ret half [[SQR]]
-//
-// CONSTRAINED-LABEL: define dso_local half @test_vsqrth_f16(
-// CONSTRAINED-SAME: half noundef [[A:%.*]]) #[[ATTR0:[0-9]+]] {
-// CONSTRAINED-NEXT:  [[ENTRY:.*:]]
-// CONSTRAINED-NEXT:    [[SQR:%.*]] = call half @llvm.experimental.constrained.sqrt.f16(half [[A]], metadata !"round.tonearest", metadata !"fpexcept.strict") #[[ATTR2:[0-9]+]]
-// CONSTRAINED-NEXT:    ret half [[SQR]]
-//
-// CIR-LABEL: cir.func {{.*}}@test_vsqrth_f16(
-// CIR: cir.sqrt %{{.*}} : !cir.f16 {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
-// LLVM-LABEL: @test_vsqrth_f16(
-// LLVM: call half @llvm.experimental.constrained.sqrt.f16({{.*}}, metadata !"round.tonearest", metadata !"fpexcept.strict")
-float16_t test_vsqrth_f16(float16_t a) {
-  return vsqrth_f16(a);
-}
-
 // UNCONSTRAINED-LABEL: define dso_local <4 x half> @test_vsqrt_f16(
 // UNCONSTRAINED-SAME: <4 x half> noundef [[A:%.*]]) #[[ATTR0:[0-9]+]] {
 // UNCONSTRAINED-NEXT:  [[ENTRY:.*:]]
@@ -78,10 +39,6 @@ float16_t test_vsqrth_f16(float16_t a) {
 // CONSTRAINED-NEXT:    [[VSQRT_I:%.*]] = call <4 x half> @llvm.experimental.constrained.sqrt.v4f16(<4 x half> [[TMP2]], metadata !"round.tonearest", metadata !"fpexcept.strict") #[[ATTR2:[0-9]+]]
 // CONSTRAINED-NEXT:    ret <4 x half> [[VSQRT_I]]
 //
-// CIR-LABEL: cir.func {{.*}}@vsqrt_f16(
-// CIR: cir.sqrt %{{.*}} : !cir.vector<4 x !cir.f16> {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
-// LLVM-LABEL: @test_vsqrt_f16(
-// LLVM: call <4 x half> @llvm.experimental.constrained.sqrt.v4f16({{.*}}, metadata !"round.tonearest", metadata !"fpexcept.strict")
 float16x4_t test_vsqrt_f16(float16x4_t a) {
   return vsqrt_f16(a);
 }
@@ -104,10 +61,6 @@ float16x4_t test_vsqrt_f16(float16x4_t a) {
 // CONSTRAINED-NEXT:    [[VSQRT_I:%.*]] = call <8 x half> @llvm.experimental.constrained.sqrt.v8f16(<8 x half> [[TMP2]], metadata !"round.tonearest", metadata !"fpexcept.strict") #[[ATTR2]]
 // CONSTRAINED-NEXT:    ret <8 x half> [[VSQRT_I]]
 //
-// CIR-LABEL: cir.func {{.*}}@vsqrtq_f16(
-// CIR: cir.sqrt %{{.*}} : !cir.vector<8 x !cir.f16> {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
-// LLVM-LABEL: @test_vsqrtq_f16(
-// LLVM: call <8 x half> @llvm.experimental.constrained.sqrt.v8f16({{.*}}, metadata !"round.tonearest", metadata !"fpexcept.strict")
 float16x8_t test_vsqrtq_f16(float16x8_t a) {
   return vsqrtq_f16(a);
 }
@@ -142,10 +95,6 @@ float16x8_t test_vsqrtq_f16(float16x8_t a) {
 // CONSTRAINED-NEXT:    [[TMP9:%.*]] = call <4 x half> @llvm.experimental.constrained.fma.v4f16(<4 x half> [[TMP7]], <4 x half> [[TMP8]], <4 x half> [[TMP6]], metadata !"round.tonearest", metadata !"fpexcept.strict") #[[ATTR2]]
 // CONSTRAINED-NEXT:    ret <4 x half> [[TMP9]]
 //
-// CIR-LABEL: cir.func {{.*}}@vfma_f16(
-// CIR: cir.fma %{{.*}}, %{{.*}}, %{{.*}} : !cir.vector<4 x !cir.f16> {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
-// LLVM-LABEL: @test_vfma_f16(
-// LLVM: call <4 x half> @llvm.experimental.constrained.fma.v4f16({{.*}}, metadata !"round.tonearest", metadata !"fpexcept.strict")
 float16x4_t test_vfma_f16(float16x4_t a, float16x4_t b, float16x4_t c) {
   return vfma_f16(a, b, c);
 }
@@ -180,34 +129,10 @@ float16x4_t test_vfma_f16(float16x4_t a, float16x4_t b, float16x4_t c) {
 // CONSTRAINED-NEXT:    [[TMP9:%.*]] = call <8 x half> @llvm.experimental.constrained.fma.v8f16(<8 x half> [[TMP7]], <8 x half> [[TMP8]], <8 x half> [[TMP6]], metadata !"round.tonearest", metadata !"fpexcept.strict") #[[ATTR2]]
 // CONSTRAINED-NEXT:    ret <8 x half> [[TMP9]]
 //
-// CIR-LABEL: cir.func {{.*}}@vfmaq_f16(
-// CIR: cir.fma %{{.*}}, %{{.*}}, %{{.*}} : !cir.vector<8 x !cir.f16> {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
-// LLVM-LABEL: @test_vfmaq_f16(
-// LLVM: call <8 x half> @llvm.experimental.constrained.fma.v8f16({{.*}}, metadata !"round.tonearest", metadata !"fpexcept.strict")
 float16x8_t test_vfmaq_f16(float16x8_t a, float16x8_t b, float16x8_t c) {
   return vfmaq_f16(a, b, c);
 }
 
-// UNCONSTRAINED-LABEL: define dso_local half @test_vfmah_f16(
-// UNCONSTRAINED-SAME: half noundef [[A:%.*]], half noundef [[B:%.*]], half noundef [[C:%.*]]) #[[ATTR0]] {
-// UNCONSTRAINED-NEXT:  [[ENTRY:.*:]]
-// UNCONSTRAINED-NEXT:    [[FMA:%.*]] = call half @llvm.fma.f16(half [[B]], half [[C]], half [[A]])
-// UNCONSTRAINED-NEXT:    ret half [[FMA]]
-//
-// CONSTRAINED-LABEL: define dso_local half @test_vfmah_f16(
-// CONSTRAINED-SAME: half noundef [[A:%.*]], half noundef [[B:%.*]], half noundef [[C:%.*]]) #[[ATTR0]] {
-// CONSTRAINED-NEXT:  [[ENTRY:.*:]]
-// CONSTRAINED-NEXT:    [[FMA:%.*]] = call half @llvm.experimental.constrained.fma.f16(half [[B]], half [[C]], half [[A]], metadata !"round.tonearest", metadata !"fpexcept.strict") #[[ATTR2]]
-// CONSTRAINED-NEXT:    ret half [[FMA]]
-//
-// CIR-LABEL: cir.func {{.*}}@test_vfmah_f16(
-// CIR: cir.fma %{{.*}}, %{{.*}}, %{{.*}} : !cir.f16 {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
-// LLVM-LABEL: @test_vfmah_f16(
-// LLVM: call half @llvm.experimental.constrained.fma.f16({{.*}}, metadata !"round.tonearest", metadata !"fpexcept.strict")
-float16_t test_vfmah_f16(float16_t a, float16_t b, float16_t c) {
-  return vfmah_f16(a, b, c);
-}
-
 // UNCONSTRAINED-LABEL: define dso_local <4 x half> @test_vfms_f16(
 // UNCONSTRAINED-SAME: <4 x half> noundef [[A:%.*]], <4 x half> noundef [[B:%.*]], <4 x half> noundef [[C:%.*]]) #[[ATTR0]] {
 // UNCONSTRAINED-NEXT:  [[ENTRY:.*:]]
@@ -312,10 +237,6 @@ float16x8_t test_vfmsq_f16(float16x8_t a, float16x8_t b, float16x8_t c) {
 // CONSTRAINED-NEXT:    [[FMLA2:%.*]] = call <4 x half> @llvm.experimental.constrained.fma.v4f16(<4 x half> [[FMLA]], <4 x half> [[LANE]], <4 x half> [[FMLA1]], metadata !"round.tonearest", metadata !"fpexcept.strict") #[[ATTR2]]
 // CONSTRAINED-NEXT:    ret <4 x half> [[FMLA2]]
 //
-// CIR-LABEL: cir.func {{.*}}@test_vfma_lane_f16(
-// CIR: cir.fma %{{.*}}, %{{.*}}, %{{.*}} : !cir.vector<4 x !cir.f16> {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
-// LLVM-LABEL: @test_vfma_lane_f16(
-// LLVM: call <4 x half> @llvm.experimental.constrained.fma.v4f16({{.*}}, metadata !"round.tonearest", metadata !"fpexcept.strict")
 float16x4_t test_vfma_lane_f16(float16x4_t a, float16x4_t b, float16x4_t c) {
   return vfma_lane_f16(a, b, c, 3);
 }
@@ -388,10 +309,6 @@ float16x8_t test_vfmaq_lane_f16(float16x8_t a, float16x8_t b, float16x4_t c) {
 // CONSTRAINED-NEXT:    [[TMP9:%.*]] = call <4 x half> @llvm.experimental.constrained.fma.v4f16(<4 x half> [[LANE]], <4 x half> [[TMP7]], <4 x half> [[TMP6]], metadata !"round.tonearest", metadata !"fpexcept.strict") #[[ATTR2]]
 // CONSTRAINED-NEXT:    ret <4 x half> [[TMP9]]
 //
-// CIR-LABEL: cir.func {{.*}}@test_vfma_laneq_f16(
-// CIR: cir.fma %{{.*}}, %{{.*}}, %{{.*}} : !cir.vector<4 x !cir.f16> {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
-// LLVM-LABEL: @test_vfma_laneq_f16(
-// LLVM: call <4 x half> @llvm.experimental.constrained.fma.v4f16({{.*}}, metadata !"round.tonearest", metadata !"fpexcept.strict")
 float16x4_t test_vfma_laneq_f16(float16x4_t a, float16x4_t b, float16x8_t c) {
   return vfma_laneq_f16(a, b, c, 7);
 }
@@ -428,10 +345,6 @@ float16x4_t test_vfma_laneq_f16(float16x4_t a, float16x4_t b, float16x8_t c) {
 // CONSTRAINED-NEXT:    [[TMP9:%.*]] = call <8 x half> @llvm.experimental.constrained.fma.v8f16(<8 x half> [[LANE]], <8 x half> [[TMP7]], <8 x half> [[TMP6]], metadata !"round.tonearest", metadata !"fpexcept.strict") #[[ATTR2]]
 // CONSTRAINED-NEXT:    ret <8 x half> [[TMP9]]
 //
-// CIR-LABEL: cir.func {{.*}}@test_vfmaq_laneq_f16(
-// CIR: cir.fma %{{.*}}, %{{.*}}, %{{.*}} : !cir.vector<8 x !cir.f16> {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
-// LLVM-LABEL: @test_vfmaq_laneq_f16(
-// LLVM: call <8 x half> @llvm.experimental.constrained.fma.v8f16({{.*}}, metadata !"round.tonearest", metadata !"fpexcept.strict")
 float16x8_t test_vfmaq_laneq_f16(float16x8_t a, float16x8_t b, float16x8_t c) {
   return vfmaq_laneq_f16(a, b, c, 7);
 }
@@ -542,10 +455,6 @@ float16x8_t test_vfmaq_n_f16(float16x8_t a, float16x8_t b, float16_t c) {
 // CONSTRAINED-NEXT:    [[TMP0:%.*]] = call half @llvm.experimental.constrained.fma.f16(half [[B]], half [[EXTRACT]], half [[A]], metadata !"round.tonearest", metadata !"fpexcept.strict") #[[ATTR2]]
 // CONSTRAINED-NEXT:    ret half [[TMP0]]
 //
-// CIR-LABEL: cir.func {{.*}}@test_vfmah_lane_f16(
-// CIR: cir.fma %{{.*}}, %{{.*}}, %{{.*}} : !cir.f16 {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
-// LLVM-LABEL: @test_vfmah_lane_f16(
-// LLVM: call half @llvm.experimental.constrained.fma.f16({{.*}}, metadata !"round.tonearest", metadata !"fpexcept.strict")
 float16_t test_vfmah_lane_f16(float16_t a, float16_t b, float16x4_t c) {
   return vfmah_lane_f16(a, b, c, 3);
 }

>From 2169b22c4328ac7fcda2587463127d4f40737c67 Mon Sep 17 00:00:00 2001
From: Yair Ben Avraham <yairba at protonmail.com>
Date: Fri, 11 Sep 2026 06:56:19 +0300
Subject: [PATCH 10/16] [CIR][AArch64] Revert unrelated test churn

Restore the unconstrained fullfp16 FMA RUN line and the trailing newline
in the legacy scalar-indexed constrained test so those files drop out of
the PR.
---
 .../CodeGen/AArch64/neon-scalar-x-indexed-elem-constrained.c    | 1 +
 clang/test/CodeGen/AArch64/neon/fused-multiple-fullfp16.c       | 2 +-
 2 files changed, 2 insertions(+), 1 deletion(-)

diff --git a/clang/test/CodeGen/AArch64/neon-scalar-x-indexed-elem-constrained.c b/clang/test/CodeGen/AArch64/neon-scalar-x-indexed-elem-constrained.c
index 54930d384e247..944929ccb5f42 100644
--- a/clang/test/CodeGen/AArch64/neon-scalar-x-indexed-elem-constrained.c
+++ b/clang/test/CodeGen/AArch64/neon-scalar-x-indexed-elem-constrained.c
@@ -258,3 +258,4 @@ float64x1_t test_vfma_laneq_f64(float64x1_t a, float64x1_t b, float64x2_t v) {
 float64x1_t test_vfms_laneq_f64(float64x1_t a, float64x1_t b, float64x2_t v) {
   return vfms_laneq_f64(a, b, v, 0);
 }
+
diff --git a/clang/test/CodeGen/AArch64/neon/fused-multiple-fullfp16.c b/clang/test/CodeGen/AArch64/neon/fused-multiple-fullfp16.c
index 30188ad3517b9..7231a1108a94f 100644
--- a/clang/test/CodeGen/AArch64/neon/fused-multiple-fullfp16.c
+++ b/clang/test/CodeGen/AArch64/neon/fused-multiple-fullfp16.c
@@ -2,7 +2,7 @@
 
 // RUN:                   %clang_cc1_cg_arm64_neon -target-feature +fullfp16           -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,sroa | FileCheck %s --check-prefixes=ALL,LLVM
 // RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 -fclangir -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,sroa | FileCheck %s --check-prefixes=ALL,LLVM %}
-// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 -fclangir -emit-cir  %s -disable-O0-optnone |                               FileCheck %s --check-prefixes=ALL,CIR --implicit-check-not='cir.call_llvm_intrinsic "fma"' %}
+// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 -fclangir -emit-cir  %s -disable-O0-optnone |                               FileCheck %s --check-prefixes=ALL,CIR %}
 
 // ALL: {{[Mm]}}odule
 

>From 2517dfc7e5ef73e60a958d8cc269d3c2f553846e Mon Sep 17 00:00:00 2001
From: Yair Ben Avraham <yairba at protonmail.com>
Date: Fri, 25 Sep 2026 06:06:58 +0300
Subject: [PATCH 11/16] [CIR][AArch64] Drop superseded FMA tests

Remove three legacy constrained FMA tests already covered by the Neon
fused-multiply tests for classic Clang and CIR.

Keep the remaining attribute captures self-contained after removing
the first indexed-element test.
---
 .../AArch64/neon-intrinsics-constrained.c     | 40 -----------
 .../neon-scalar-x-indexed-elem-constrained.c  | 67 +------------------
 2 files changed, 3 insertions(+), 104 deletions(-)

diff --git a/clang/test/CodeGen/AArch64/neon-intrinsics-constrained.c b/clang/test/CodeGen/AArch64/neon-intrinsics-constrained.c
index 50a2a629ea55b..09f2aec4176b4 100644
--- a/clang/test/CodeGen/AArch64/neon-intrinsics-constrained.c
+++ b/clang/test/CodeGen/AArch64/neon-intrinsics-constrained.c
@@ -1418,46 +1418,6 @@ float64x1_t test_vmls_f64(float64x1_t a, float64x1_t b, float64x1_t c) {
   return vmls_f64(a, b, c);
 }
 
-// UNCONSTRAINED-LABEL: define dso_local <1 x double> @test_vfma_f64(
-// UNCONSTRAINED-SAME: <1 x double> noundef [[A:%.*]], <1 x double> noundef [[B:%.*]], <1 x double> noundef [[C:%.*]]) #[[ATTR0]] {
-// UNCONSTRAINED-NEXT:  [[ENTRY:.*:]]
-// UNCONSTRAINED-NEXT:    [[TMP0:%.*]] = bitcast <1 x double> [[A]] to i64
-// UNCONSTRAINED-NEXT:    [[__P0_ADDR_I_SROA_0_0_VEC_INSERT:%.*]] = insertelement <1 x i64> undef, i64 [[TMP0]], i64 0
-// UNCONSTRAINED-NEXT:    [[TMP1:%.*]] = bitcast <1 x double> [[B]] to i64
-// UNCONSTRAINED-NEXT:    [[__P1_ADDR_I_SROA_0_0_VEC_INSERT:%.*]] = insertelement <1 x i64> undef, i64 [[TMP1]], i64 0
-// UNCONSTRAINED-NEXT:    [[TMP2:%.*]] = bitcast <1 x double> [[C]] to i64
-// UNCONSTRAINED-NEXT:    [[__P2_ADDR_I_SROA_0_0_VEC_INSERT:%.*]] = insertelement <1 x i64> undef, i64 [[TMP2]], i64 0
-// UNCONSTRAINED-NEXT:    [[TMP3:%.*]] = bitcast <1 x i64> [[__P0_ADDR_I_SROA_0_0_VEC_INSERT]] to <8 x i8>
-// UNCONSTRAINED-NEXT:    [[TMP4:%.*]] = bitcast <1 x i64> [[__P1_ADDR_I_SROA_0_0_VEC_INSERT]] to <8 x i8>
-// UNCONSTRAINED-NEXT:    [[TMP5:%.*]] = bitcast <1 x i64> [[__P2_ADDR_I_SROA_0_0_VEC_INSERT]] to <8 x i8>
-// UNCONSTRAINED-NEXT:    [[TMP6:%.*]] = bitcast <8 x i8> [[TMP3]] to <1 x double>
-// UNCONSTRAINED-NEXT:    [[TMP7:%.*]] = bitcast <8 x i8> [[TMP4]] to <1 x double>
-// UNCONSTRAINED-NEXT:    [[TMP8:%.*]] = bitcast <8 x i8> [[TMP5]] to <1 x double>
-// UNCONSTRAINED-NEXT:    [[TMP9:%.*]] = call <1 x double> @llvm.fma.v1f64(<1 x double> [[TMP7]], <1 x double> [[TMP8]], <1 x double> [[TMP6]])
-// UNCONSTRAINED-NEXT:    ret <1 x double> [[TMP9]]
-//
-// CONSTRAINED-LABEL: define dso_local <1 x double> @test_vfma_f64(
-// CONSTRAINED-SAME: <1 x double> noundef [[A:%.*]], <1 x double> noundef [[B:%.*]], <1 x double> noundef [[C:%.*]]) #[[ATTR0]] {
-// CONSTRAINED-NEXT:  [[ENTRY:.*:]]
-// CONSTRAINED-NEXT:    [[TMP0:%.*]] = bitcast <1 x double> [[A]] to i64
-// CONSTRAINED-NEXT:    [[__P0_ADDR_I_SROA_0_0_VEC_INSERT:%.*]] = insertelement <1 x i64> undef, i64 [[TMP0]], i64 0
-// CONSTRAINED-NEXT:    [[TMP1:%.*]] = bitcast <1 x double> [[B]] to i64
-// CONSTRAINED-NEXT:    [[__P1_ADDR_I_SROA_0_0_VEC_INSERT:%.*]] = insertelement <1 x i64> undef, i64 [[TMP1]], i64 0
-// CONSTRAINED-NEXT:    [[TMP2:%.*]] = bitcast <1 x double> [[C]] to i64
-// CONSTRAINED-NEXT:    [[__P2_ADDR_I_SROA_0_0_VEC_INSERT:%.*]] = insertelement <1 x i64> undef, i64 [[TMP2]], i64 0
-// CONSTRAINED-NEXT:    [[TMP3:%.*]] = bitcast <1 x i64> [[__P0_ADDR_I_SROA_0_0_VEC_INSERT]] to <8 x i8>
-// CONSTRAINED-NEXT:    [[TMP4:%.*]] = bitcast <1 x i64> [[__P1_ADDR_I_SROA_0_0_VEC_INSERT]] to <8 x i8>
-// CONSTRAINED-NEXT:    [[TMP5:%.*]] = bitcast <1 x i64> [[__P2_ADDR_I_SROA_0_0_VEC_INSERT]] to <8 x i8>
-// CONSTRAINED-NEXT:    [[TMP6:%.*]] = bitcast <8 x i8> [[TMP3]] to <1 x double>
-// CONSTRAINED-NEXT:    [[TMP7:%.*]] = bitcast <8 x i8> [[TMP4]] to <1 x double>
-// CONSTRAINED-NEXT:    [[TMP8:%.*]] = bitcast <8 x i8> [[TMP5]] to <1 x double>
-// CONSTRAINED-NEXT:    [[TMP9:%.*]] = call <1 x double> @llvm.experimental.constrained.fma.v1f64(<1 x double> [[TMP7]], <1 x double> [[TMP8]], <1 x double> [[TMP6]], metadata !"round.tonearest", metadata !"fpexcept.strict") #[[ATTR3]]
-// CONSTRAINED-NEXT:    ret <1 x double> [[TMP9]]
-//
-float64x1_t test_vfma_f64(float64x1_t a, float64x1_t b, float64x1_t c) {
-  return vfma_f64(a, b, c);
-}
-
 // UNCONSTRAINED-LABEL: define dso_local <1 x double> @test_vfms_f64(
 // UNCONSTRAINED-SAME: <1 x double> noundef [[A:%.*]], <1 x double> noundef [[B:%.*]], <1 x double> noundef [[C:%.*]]) #[[ATTR0]] {
 // UNCONSTRAINED-NEXT:  [[ENTRY:.*:]]
diff --git a/clang/test/CodeGen/AArch64/neon-scalar-x-indexed-elem-constrained.c b/clang/test/CodeGen/AArch64/neon-scalar-x-indexed-elem-constrained.c
index 944929ccb5f42..046bca2420a37 100644
--- a/clang/test/CodeGen/AArch64/neon-scalar-x-indexed-elem-constrained.c
+++ b/clang/test/CodeGen/AArch64/neon-scalar-x-indexed-elem-constrained.c
@@ -13,36 +13,18 @@
 
 #include <arm_neon.h>
 
-// UNCONSTRAINED-LABEL: define dso_local float @test_vfmas_lane_f32(
-// UNCONSTRAINED-SAME: float noundef [[A:%.*]], float noundef [[B:%.*]], <2 x float> noundef [[C:%.*]]) #[[ATTR0:[0-9]+]] {
-// UNCONSTRAINED-NEXT:  [[ENTRY:.*:]]
-// UNCONSTRAINED-NEXT:    [[EXTRACT:%.*]] = extractelement <2 x float> [[C]], i32 1
-// UNCONSTRAINED-NEXT:    [[TMP0:%.*]] = call float @llvm.fma.f32(float [[B]], float [[EXTRACT]], float [[A]])
-// UNCONSTRAINED-NEXT:    ret float [[TMP0]]
-//
-// CONSTRAINED-LABEL: define dso_local float @test_vfmas_lane_f32(
-// CONSTRAINED-SAME: float noundef [[A:%.*]], float noundef [[B:%.*]], <2 x float> noundef [[C:%.*]]) #[[ATTR0:[0-9]+]] {
-// CONSTRAINED-NEXT:  [[ENTRY:.*:]]
-// CONSTRAINED-NEXT:    [[EXTRACT:%.*]] = extractelement <2 x float> [[C]], i32 1
-// CONSTRAINED-NEXT:    [[TMP0:%.*]] = call float @llvm.experimental.constrained.fma.f32(float [[B]], float [[EXTRACT]], float [[A]], metadata !"round.tonearest", metadata !"fpexcept.strict") #[[ATTR2:[0-9]+]]
-// CONSTRAINED-NEXT:    ret float [[TMP0]]
-//
-float32_t test_vfmas_lane_f32(float32_t a, float32_t b, float32x2_t c) {
-  return vfmas_lane_f32(a, b, c, 1);
-}
-
 // UNCONSTRAINED-LABEL: define dso_local double @test_vfmad_lane_f64(
-// UNCONSTRAINED-SAME: double noundef [[A:%.*]], double noundef [[B:%.*]], <1 x double> noundef [[C:%.*]]) #[[ATTR0]] {
+// UNCONSTRAINED-SAME: double noundef [[A:%.*]], double noundef [[B:%.*]], <1 x double> noundef [[C:%.*]]) #[[ATTR0:[0-9]+]] {
 // UNCONSTRAINED-NEXT:  [[ENTRY:.*:]]
 // UNCONSTRAINED-NEXT:    [[EXTRACT:%.*]] = extractelement <1 x double> [[C]], i32 0
 // UNCONSTRAINED-NEXT:    [[TMP0:%.*]] = call double @llvm.fma.f64(double [[B]], double [[EXTRACT]], double [[A]])
 // UNCONSTRAINED-NEXT:    ret double [[TMP0]]
 //
 // CONSTRAINED-LABEL: define dso_local double @test_vfmad_lane_f64(
-// CONSTRAINED-SAME: double noundef [[A:%.*]], double noundef [[B:%.*]], <1 x double> noundef [[C:%.*]]) #[[ATTR0]] {
+// CONSTRAINED-SAME: double noundef [[A:%.*]], double noundef [[B:%.*]], <1 x double> noundef [[C:%.*]]) #[[ATTR0:[0-9]+]] {
 // CONSTRAINED-NEXT:  [[ENTRY:.*:]]
 // CONSTRAINED-NEXT:    [[EXTRACT:%.*]] = extractelement <1 x double> [[C]], i32 0
-// CONSTRAINED-NEXT:    [[TMP0:%.*]] = call double @llvm.experimental.constrained.fma.f64(double [[B]], double [[EXTRACT]], double [[A]], metadata !"round.tonearest", metadata !"fpexcept.strict") #[[ATTR2]]
+// CONSTRAINED-NEXT:    [[TMP0:%.*]] = call double @llvm.experimental.constrained.fma.f64(double [[B]], double [[EXTRACT]], double [[A]], metadata !"round.tonearest", metadata !"fpexcept.strict") #[[ATTR2:[0-9]+]]
 // CONSTRAINED-NEXT:    ret double [[TMP0]]
 //
 float64_t test_vfmad_lane_f64(float64_t a, float64_t b, float64x1_t c) {
@@ -173,48 +155,6 @@ float64x1_t test_vfms_lane_f64(float64x1_t a, float64x1_t b, float64x1_t v) {
   return vfms_lane_f64(a, b, v, 0);
 }
 
-// UNCONSTRAINED-LABEL: define dso_local <1 x double> @test_vfma_laneq_f64(
-// UNCONSTRAINED-SAME: <1 x double> noundef [[A:%.*]], <1 x double> noundef [[B:%.*]], <2 x double> noundef [[V:%.*]]) #[[ATTR0]] {
-// UNCONSTRAINED-NEXT:  [[ENTRY:.*:]]
-// UNCONSTRAINED-NEXT:    [[TMP0:%.*]] = bitcast <1 x double> [[A]] to i64
-// UNCONSTRAINED-NEXT:    [[__S0_SROA_0_0_VEC_INSERT:%.*]] = insertelement <1 x i64> undef, i64 [[TMP0]], i64 0
-// UNCONSTRAINED-NEXT:    [[TMP1:%.*]] = bitcast <1 x double> [[B]] to i64
-// UNCONSTRAINED-NEXT:    [[__S1_SROA_0_0_VEC_INSERT:%.*]] = insertelement <1 x i64> undef, i64 [[TMP1]], i64 0
-// UNCONSTRAINED-NEXT:    [[TMP2:%.*]] = bitcast <2 x double> [[V]] to <2 x i64>
-// UNCONSTRAINED-NEXT:    [[TMP3:%.*]] = bitcast <1 x i64> [[__S0_SROA_0_0_VEC_INSERT]] to <8 x i8>
-// UNCONSTRAINED-NEXT:    [[TMP4:%.*]] = bitcast <1 x i64> [[__S1_SROA_0_0_VEC_INSERT]] to <8 x i8>
-// UNCONSTRAINED-NEXT:    [[TMP5:%.*]] = bitcast <2 x i64> [[TMP2]] to <16 x i8>
-// UNCONSTRAINED-NEXT:    [[TMP6:%.*]] = bitcast <8 x i8> [[TMP3]] to double
-// UNCONSTRAINED-NEXT:    [[TMP7:%.*]] = bitcast <8 x i8> [[TMP4]] to double
-// UNCONSTRAINED-NEXT:    [[TMP8:%.*]] = bitcast <16 x i8> [[TMP5]] to <2 x double>
-// UNCONSTRAINED-NEXT:    [[EXTRACT:%.*]] = extractelement <2 x double> [[TMP8]], i32 0
-// UNCONSTRAINED-NEXT:    [[TMP9:%.*]] = call double @llvm.fma.f64(double [[TMP7]], double [[EXTRACT]], double [[TMP6]])
-// UNCONSTRAINED-NEXT:    [[TMP10:%.*]] = bitcast double [[TMP9]] to <1 x double>
-// UNCONSTRAINED-NEXT:    ret <1 x double> [[TMP10]]
-//
-// CONSTRAINED-LABEL: define dso_local <1 x double> @test_vfma_laneq_f64(
-// CONSTRAINED-SAME: <1 x double> noundef [[A:%.*]], <1 x double> noundef [[B:%.*]], <2 x double> noundef [[V:%.*]]) #[[ATTR0]] {
-// CONSTRAINED-NEXT:  [[ENTRY:.*:]]
-// CONSTRAINED-NEXT:    [[TMP0:%.*]] = bitcast <1 x double> [[A]] to i64
-// CONSTRAINED-NEXT:    [[__S0_SROA_0_0_VEC_INSERT:%.*]] = insertelement <1 x i64> undef, i64 [[TMP0]], i64 0
-// CONSTRAINED-NEXT:    [[TMP1:%.*]] = bitcast <1 x double> [[B]] to i64
-// CONSTRAINED-NEXT:    [[__S1_SROA_0_0_VEC_INSERT:%.*]] = insertelement <1 x i64> undef, i64 [[TMP1]], i64 0
-// CONSTRAINED-NEXT:    [[TMP2:%.*]] = bitcast <2 x double> [[V]] to <2 x i64>
-// CONSTRAINED-NEXT:    [[TMP3:%.*]] = bitcast <1 x i64> [[__S0_SROA_0_0_VEC_INSERT]] to <8 x i8>
-// CONSTRAINED-NEXT:    [[TMP4:%.*]] = bitcast <1 x i64> [[__S1_SROA_0_0_VEC_INSERT]] to <8 x i8>
-// CONSTRAINED-NEXT:    [[TMP5:%.*]] = bitcast <2 x i64> [[TMP2]] to <16 x i8>
-// CONSTRAINED-NEXT:    [[TMP6:%.*]] = bitcast <8 x i8> [[TMP3]] to double
-// CONSTRAINED-NEXT:    [[TMP7:%.*]] = bitcast <8 x i8> [[TMP4]] to double
-// CONSTRAINED-NEXT:    [[TMP8:%.*]] = bitcast <16 x i8> [[TMP5]] to <2 x double>
-// CONSTRAINED-NEXT:    [[EXTRACT:%.*]] = extractelement <2 x double> [[TMP8]], i32 0
-// CONSTRAINED-NEXT:    [[TMP9:%.*]] = call double @llvm.experimental.constrained.fma.f64(double [[TMP7]], double [[EXTRACT]], double [[TMP6]], metadata !"round.tonearest", metadata !"fpexcept.strict") #[[ATTR2]]
-// CONSTRAINED-NEXT:    [[TMP10:%.*]] = bitcast double [[TMP9]] to <1 x double>
-// CONSTRAINED-NEXT:    ret <1 x double> [[TMP10]]
-//
-float64x1_t test_vfma_laneq_f64(float64x1_t a, float64x1_t b, float64x2_t v) {
-  return vfma_laneq_f64(a, b, v, 0);
-}
-
 // UNCONSTRAINED-LABEL: define dso_local <1 x double> @test_vfms_laneq_f64(
 // UNCONSTRAINED-SAME: <1 x double> noundef [[A:%.*]], <1 x double> noundef [[B:%.*]], <2 x double> noundef [[V:%.*]]) #[[ATTR0]] {
 // UNCONSTRAINED-NEXT:  [[ENTRY:.*:]]
@@ -258,4 +198,3 @@ float64x1_t test_vfma_laneq_f64(float64x1_t a, float64x1_t b, float64x2_t v) {
 float64x1_t test_vfms_laneq_f64(float64x1_t a, float64x1_t b, float64x2_t v) {
   return vfms_laneq_f64(a, b, v, 0);
 }
-

>From d89174f4f0ea85809411c36086c56b538579b628 Mon Sep 17 00:00:00 2001
From: Yair Ben Avraham <yairba at protonmail.com>
Date: Fri, 25 Sep 2026 06:08:39 +0300
Subject: [PATCH 12/16] [CIR][AArch64] Migrate scalar FP16 strict tests

Check vsqrth_f16 and vfmah_f16 with classic Clang and CIR in the
shared Neon constrained test. Their unconstrained checks remain in
neon/fullfp16.c.

Remove the duplicate scalar blocks from the legacy FP16 test.
---
 .../CodeGen/AArch64/neon/fullfp16-constrained.c  |  1 +
 .../AArch64/v8.2a-fp16-intrinsics-constrained.c  | 16 ----------------
 2 files changed, 1 insertion(+), 16 deletions(-)

diff --git a/clang/test/CodeGen/AArch64/neon/fullfp16-constrained.c b/clang/test/CodeGen/AArch64/neon/fullfp16-constrained.c
index 8c9c1b61899c6..9f5b99b0aa587 100644
--- a/clang/test/CodeGen/AArch64/neon/fullfp16-constrained.c
+++ b/clang/test/CodeGen/AArch64/neon/fullfp16-constrained.c
@@ -1,5 +1,6 @@
 // REQUIRES: aarch64-registered-target
 
+// RUN: %clang_cc1_cg_arm64_neon -target-feature +fullfp16 -ffp-exception-behavior=strict -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,simplifycfg | FileCheck %s --check-prefix=LLVM --implicit-check-not=' @llvm.fma.' --implicit-check-not=' @llvm.sqrt.'
 // RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 -fexperimental-strict-floating-point -ffp-exception-behavior=maytrap -DEXCEPT=1 -fclangir -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,simplifycfg | FileCheck %s --check-prefix=LLVM --implicit-check-not=fpexcept.maytrap --implicit-check-not=' @llvm.fma.' --implicit-check-not=' @llvm.sqrt.' %}
 // RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 -fexperimental-strict-floating-point -ffp-exception-behavior=maytrap -DEXCEPT=1 -fclangir -emit-cir  %s -disable-O0-optnone |                                      FileCheck %s --check-prefix=CIR --implicit-check-not='except_mode = maytrap' --implicit-check-not='cir.call_llvm_intrinsic "fma"' --implicit-check-not='cir.call_llvm_intrinsic "sqrt"' %}
 // RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 -fexperimental-strict-floating-point -ffp-exception-behavior=strict             -fclangir -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,simplifycfg | FileCheck %s --check-prefix=LLVM --implicit-check-not=' @llvm.fma.' --implicit-check-not=' @llvm.sqrt.' %}
diff --git a/clang/test/CodeGen/AArch64/v8.2a-fp16-intrinsics-constrained.c b/clang/test/CodeGen/AArch64/v8.2a-fp16-intrinsics-constrained.c
index c9d5071c4eb83..4006171e40296 100644
--- a/clang/test/CodeGen/AArch64/v8.2a-fp16-intrinsics-constrained.c
+++ b/clang/test/CodeGen/AArch64/v8.2a-fp16-intrinsics-constrained.c
@@ -200,14 +200,6 @@ float16_t test_vrndxh_f16(float16_t a) {
   return vrndxh_f16(a);
 }
 
-// COMMON-LABEL: test_vsqrth_f16
-// UNCONSTRAINED:  [[SQR:%.*]] = call half @llvm.sqrt.f16(half %a)
-// CONSTRAINED:    [[SQR:%.*]] = call half @llvm.experimental.constrained.sqrt.f16(half %a, metadata !"round.tonearest", metadata !"fpexcept.strict")
-// COMMONIR:       ret half [[SQR]]
-float16_t test_vsqrth_f16(float16_t a) {
-  return vsqrth_f16(a);
-}
-
 // COMMON-LABEL: test_vaddh_f16
 // UNCONSTRAINED:  [[ADD:%.*]] = fadd half %a, %b
 // CONSTRAINED:    [[ADD:%.*]] = call half @llvm.experimental.constrained.fadd.f16(half %a, half %b, metadata !"round.tonearest", metadata !"fpexcept.strict")
@@ -285,14 +277,6 @@ float16_t test_vsubh_f16(float16_t a, float16_t b) {
   return vsubh_f16(a, b);
 }
 
-// COMMON-LABEL: test_vfmah_f16
-// UNCONSTRAINED:  [[FMA:%.*]] = call half @llvm.fma.f16(half %b, half %c, half %a)
-// CONSTRAINED:    [[FMA:%.*]] = call half @llvm.experimental.constrained.fma.f16(half %b, half %c, half %a, metadata !"round.tonearest", metadata !"fpexcept.strict")
-// COMMONIR:       ret half [[FMA]]
-float16_t test_vfmah_f16(float16_t a, float16_t b, float16_t c) {
-  return vfmah_f16(a, b, c);
-}
-
 // COMMON-LABEL: test_vfmsh_f16
 // COMMONIR:  [[SUB:%.*]] = fneg half %b
 // UNCONSTRAINED:  [[ADD:%.*]] = call half @llvm.fma.f16(half [[SUB]], half %c, half %a)

>From f51be4d5fd8baf4728e913be2f1eab493849be7d Mon Sep 17 00:00:00 2001
From: Yair Ben Avraham <yairba at protonmail.com>
Date: Fri, 25 Sep 2026 06:11:20 +0300
Subject: [PATCH 13/16] [CIR][AArch64] Migrate FP16 FMA strict tests

Check seven FP16 FMA forms with classic Clang and CIR in the shared
Neon constrained test. Accept independent lane-setup instructions in
either order while retaining operand and lane checks.

Remove only their superseded legacy blocks. Keep vector sqrt and other
unmigrated FP16 cases for a later change.
---
 .../fused-multiple-fullfp16-constrained.c     |  18 +-
 .../v8.2a-neon-intrinsics-constrained.c       | 230 ------------------
 2 files changed, 10 insertions(+), 238 deletions(-)

diff --git a/clang/test/CodeGen/AArch64/neon/fused-multiple-fullfp16-constrained.c b/clang/test/CodeGen/AArch64/neon/fused-multiple-fullfp16-constrained.c
index 907cf54b91134..d22fc1551dc44 100644
--- a/clang/test/CodeGen/AArch64/neon/fused-multiple-fullfp16-constrained.c
+++ b/clang/test/CodeGen/AArch64/neon/fused-multiple-fullfp16-constrained.c
@@ -1,5 +1,7 @@
 // REQUIRES: aarch64-registered-target
 
+// RUN: %clang_cc1_cg_arm64_neon -target-feature +fullfp16 -ffp-exception-behavior=maytrap -DEXCEPT=1 -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,sroa | FileCheck %s --check-prefix=LLVM --implicit-check-not=fpexcept.maytrap --implicit-check-not=' @llvm.fma.'
+// RUN: %clang_cc1_cg_arm64_neon -target-feature +fullfp16 -ffp-exception-behavior=strict -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,sroa | FileCheck %s --check-prefix=LLVM --implicit-check-not=' @llvm.fma.'
 // RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 -fexperimental-strict-floating-point -ffp-exception-behavior=maytrap -DEXCEPT=1 -fclangir -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,sroa | FileCheck %s --check-prefix=LLVM --implicit-check-not=fpexcept.maytrap --implicit-check-not=' @llvm.fma.' %}
 // RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 -fexperimental-strict-floating-point -ffp-exception-behavior=maytrap -DEXCEPT=1 -fclangir -emit-cir  %s -disable-O0-optnone |                               FileCheck %s --check-prefix=CIR --implicit-check-not='except_mode = maytrap' --implicit-check-not='cir.call_llvm_intrinsic "fma"' %}
 // RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 -fexperimental-strict-floating-point -ffp-exception-behavior=strict             -fclangir -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,sroa | FileCheck %s --check-prefix=LLVM --implicit-check-not=' @llvm.fma.' %}
@@ -65,10 +67,10 @@ float16x4_t test_vfma_lane_f16(float16x4_t a, float16x4_t b,
 // LLVM: [[A_BYTES:%.*]] = bitcast <4 x i16> [[A_I]] to <8 x i8>
 // LLVM: [[B_BYTES:%.*]] = bitcast <4 x i16> [[B_I]] to <8 x i8>
 // LLVM: [[C_BYTES:%.*]] = bitcast <4 x i16> [[C_I]] to <8 x i8>
-// LLVM: [[A_CAST:%.*]] = bitcast <8 x i8> [[A_BYTES]] to <4 x half>
-// LLVM: [[B_CAST:%.*]] = bitcast <8 x i8> [[B_BYTES]] to <4 x half>
-// LLVM: [[C_CAST:%.*]] = bitcast <8 x i8> [[C_BYTES]] to <4 x half>
-// LLVM: [[LANE:%.*]] = shufflevector <4 x half> [[C_CAST]], <4 x half> {{.*}}, <4 x i32> <i32 3, i32 3, i32 3, i32 3>
+// LLVM-DAG: [[A_CAST:%.*]] = bitcast <8 x i8> [[A_BYTES]] to <4 x half>
+// LLVM-DAG: [[B_CAST:%.*]] = bitcast <8 x i8> [[B_BYTES]] to <4 x half>
+// LLVM-DAG: [[C_CAST:%.*]] = bitcast <8 x i8> [[C_BYTES]] to <4 x half>
+// LLVM-DAG: [[LANE:%.*]] = shufflevector <4 x half> [[C_CAST]], <4 x half> {{.*}}, <4 x i32> <i32 3, i32 3, i32 3, i32 3>
 // LLVM: [[FMA:%.*]] = call <4 x half> @llvm.experimental.constrained.fma.v4f16(<4 x half> [[B_CAST]], <4 x half> [[LANE]], <4 x half> [[A_CAST]], metadata !"round.tonearest", metadata !"fpexcept.strict")
 // LLVM: ret <4 x half> [[FMA]]
   return vfma_lane_f16(a, b, c, 3);
@@ -88,10 +90,10 @@ float16x8_t test_vfmaq_lane_f16(float16x8_t a, float16x8_t b,
 // LLVM: [[A_BYTES:%.*]] = bitcast <8 x i16> [[A_I]] to <16 x i8>
 // LLVM: [[B_BYTES:%.*]] = bitcast <8 x i16> [[B_I]] to <16 x i8>
 // LLVM: [[C_BYTES:%.*]] = bitcast <4 x i16> [[C_I]] to <8 x i8>
-// LLVM: [[A_CAST:%.*]] = bitcast <16 x i8> [[A_BYTES]] to <8 x half>
-// LLVM: [[B_CAST:%.*]] = bitcast <16 x i8> [[B_BYTES]] to <8 x half>
-// LLVM: [[C_CAST:%.*]] = bitcast <8 x i8> [[C_BYTES]] to <4 x half>
-// LLVM: [[LANE:%.*]] = shufflevector <4 x half> [[C_CAST]], <4 x half> {{.*}}, <8 x i32> <i32 3, i32 3, i32 3, i32 3, i32 3, i32 3, i32 3, i32 3>
+// LLVM-DAG: [[A_CAST:%.*]] = bitcast <16 x i8> [[A_BYTES]] to <8 x half>
+// LLVM-DAG: [[B_CAST:%.*]] = bitcast <16 x i8> [[B_BYTES]] to <8 x half>
+// LLVM-DAG: [[C_CAST:%.*]] = bitcast <8 x i8> [[C_BYTES]] to <4 x half>
+// LLVM-DAG: [[LANE:%.*]] = shufflevector <4 x half> [[C_CAST]], <4 x half> {{.*}}, <8 x i32> <i32 3, i32 3, i32 3, i32 3, i32 3, i32 3, i32 3, i32 3>
 // LLVM: [[FMA:%.*]] = call <8 x half> @llvm.experimental.constrained.fma.v8f16(<8 x half> [[B_CAST]], <8 x half> [[LANE]], <8 x half> [[A_CAST]], metadata !"round.tonearest", metadata !"fpexcept.strict")
 // LLVM: ret <8 x half> [[FMA]]
   return vfmaq_lane_f16(a, b, c, 3);
diff --git a/clang/test/CodeGen/AArch64/v8.2a-neon-intrinsics-constrained.c b/clang/test/CodeGen/AArch64/v8.2a-neon-intrinsics-constrained.c
index dac8b931ff210..f04657d7bd0c2 100644
--- a/clang/test/CodeGen/AArch64/v8.2a-neon-intrinsics-constrained.c
+++ b/clang/test/CodeGen/AArch64/v8.2a-neon-intrinsics-constrained.c
@@ -65,74 +65,6 @@ float16x8_t test_vsqrtq_f16(float16x8_t a) {
   return vsqrtq_f16(a);
 }
 
-// UNCONSTRAINED-LABEL: define dso_local <4 x half> @test_vfma_f16(
-// UNCONSTRAINED-SAME: <4 x half> noundef [[A:%.*]], <4 x half> noundef [[B:%.*]], <4 x half> noundef [[C:%.*]]) #[[ATTR0]] {
-// UNCONSTRAINED-NEXT:  [[ENTRY:.*:]]
-// UNCONSTRAINED-NEXT:    [[TMP0:%.*]] = bitcast <4 x half> [[A]] to <4 x i16>
-// UNCONSTRAINED-NEXT:    [[TMP1:%.*]] = bitcast <4 x half> [[B]] to <4 x i16>
-// UNCONSTRAINED-NEXT:    [[TMP2:%.*]] = bitcast <4 x half> [[C]] to <4 x i16>
-// UNCONSTRAINED-NEXT:    [[TMP3:%.*]] = bitcast <4 x i16> [[TMP0]] to <8 x i8>
-// UNCONSTRAINED-NEXT:    [[TMP4:%.*]] = bitcast <4 x i16> [[TMP1]] to <8 x i8>
-// UNCONSTRAINED-NEXT:    [[TMP5:%.*]] = bitcast <4 x i16> [[TMP2]] to <8 x i8>
-// UNCONSTRAINED-NEXT:    [[TMP6:%.*]] = bitcast <8 x i8> [[TMP3]] to <4 x half>
-// UNCONSTRAINED-NEXT:    [[TMP7:%.*]] = bitcast <8 x i8> [[TMP4]] to <4 x half>
-// UNCONSTRAINED-NEXT:    [[TMP8:%.*]] = bitcast <8 x i8> [[TMP5]] to <4 x half>
-// UNCONSTRAINED-NEXT:    [[TMP9:%.*]] = call <4 x half> @llvm.fma.v4f16(<4 x half> [[TMP7]], <4 x half> [[TMP8]], <4 x half> [[TMP6]])
-// UNCONSTRAINED-NEXT:    ret <4 x half> [[TMP9]]
-//
-// CONSTRAINED-LABEL: define dso_local <4 x half> @test_vfma_f16(
-// CONSTRAINED-SAME: <4 x half> noundef [[A:%.*]], <4 x half> noundef [[B:%.*]], <4 x half> noundef [[C:%.*]]) #[[ATTR0]] {
-// CONSTRAINED-NEXT:  [[ENTRY:.*:]]
-// CONSTRAINED-NEXT:    [[TMP0:%.*]] = bitcast <4 x half> [[A]] to <4 x i16>
-// CONSTRAINED-NEXT:    [[TMP1:%.*]] = bitcast <4 x half> [[B]] to <4 x i16>
-// CONSTRAINED-NEXT:    [[TMP2:%.*]] = bitcast <4 x half> [[C]] to <4 x i16>
-// CONSTRAINED-NEXT:    [[TMP3:%.*]] = bitcast <4 x i16> [[TMP0]] to <8 x i8>
-// CONSTRAINED-NEXT:    [[TMP4:%.*]] = bitcast <4 x i16> [[TMP1]] to <8 x i8>
-// CONSTRAINED-NEXT:    [[TMP5:%.*]] = bitcast <4 x i16> [[TMP2]] to <8 x i8>
-// CONSTRAINED-NEXT:    [[TMP6:%.*]] = bitcast <8 x i8> [[TMP3]] to <4 x half>
-// CONSTRAINED-NEXT:    [[TMP7:%.*]] = bitcast <8 x i8> [[TMP4]] to <4 x half>
-// CONSTRAINED-NEXT:    [[TMP8:%.*]] = bitcast <8 x i8> [[TMP5]] to <4 x half>
-// CONSTRAINED-NEXT:    [[TMP9:%.*]] = call <4 x half> @llvm.experimental.constrained.fma.v4f16(<4 x half> [[TMP7]], <4 x half> [[TMP8]], <4 x half> [[TMP6]], metadata !"round.tonearest", metadata !"fpexcept.strict") #[[ATTR2]]
-// CONSTRAINED-NEXT:    ret <4 x half> [[TMP9]]
-//
-float16x4_t test_vfma_f16(float16x4_t a, float16x4_t b, float16x4_t c) {
-  return vfma_f16(a, b, c);
-}
-
-// UNCONSTRAINED-LABEL: define dso_local <8 x half> @test_vfmaq_f16(
-// UNCONSTRAINED-SAME: <8 x half> noundef [[A:%.*]], <8 x half> noundef [[B:%.*]], <8 x half> noundef [[C:%.*]]) #[[ATTR0]] {
-// UNCONSTRAINED-NEXT:  [[ENTRY:.*:]]
-// UNCONSTRAINED-NEXT:    [[TMP0:%.*]] = bitcast <8 x half> [[A]] to <8 x i16>
-// UNCONSTRAINED-NEXT:    [[TMP1:%.*]] = bitcast <8 x half> [[B]] to <8 x i16>
-// UNCONSTRAINED-NEXT:    [[TMP2:%.*]] = bitcast <8 x half> [[C]] to <8 x i16>
-// UNCONSTRAINED-NEXT:    [[TMP3:%.*]] = bitcast <8 x i16> [[TMP0]] to <16 x i8>
-// UNCONSTRAINED-NEXT:    [[TMP4:%.*]] = bitcast <8 x i16> [[TMP1]] to <16 x i8>
-// UNCONSTRAINED-NEXT:    [[TMP5:%.*]] = bitcast <8 x i16> [[TMP2]] to <16 x i8>
-// UNCONSTRAINED-NEXT:    [[TMP6:%.*]] = bitcast <16 x i8> [[TMP3]] to <8 x half>
-// UNCONSTRAINED-NEXT:    [[TMP7:%.*]] = bitcast <16 x i8> [[TMP4]] to <8 x half>
-// UNCONSTRAINED-NEXT:    [[TMP8:%.*]] = bitcast <16 x i8> [[TMP5]] to <8 x half>
-// UNCONSTRAINED-NEXT:    [[TMP9:%.*]] = call <8 x half> @llvm.fma.v8f16(<8 x half> [[TMP7]], <8 x half> [[TMP8]], <8 x half> [[TMP6]])
-// UNCONSTRAINED-NEXT:    ret <8 x half> [[TMP9]]
-//
-// CONSTRAINED-LABEL: define dso_local <8 x half> @test_vfmaq_f16(
-// CONSTRAINED-SAME: <8 x half> noundef [[A:%.*]], <8 x half> noundef [[B:%.*]], <8 x half> noundef [[C:%.*]]) #[[ATTR0]] {
-// CONSTRAINED-NEXT:  [[ENTRY:.*:]]
-// CONSTRAINED-NEXT:    [[TMP0:%.*]] = bitcast <8 x half> [[A]] to <8 x i16>
-// CONSTRAINED-NEXT:    [[TMP1:%.*]] = bitcast <8 x half> [[B]] to <8 x i16>
-// CONSTRAINED-NEXT:    [[TMP2:%.*]] = bitcast <8 x half> [[C]] to <8 x i16>
-// CONSTRAINED-NEXT:    [[TMP3:%.*]] = bitcast <8 x i16> [[TMP0]] to <16 x i8>
-// CONSTRAINED-NEXT:    [[TMP4:%.*]] = bitcast <8 x i16> [[TMP1]] to <16 x i8>
-// CONSTRAINED-NEXT:    [[TMP5:%.*]] = bitcast <8 x i16> [[TMP2]] to <16 x i8>
-// CONSTRAINED-NEXT:    [[TMP6:%.*]] = bitcast <16 x i8> [[TMP3]] to <8 x half>
-// CONSTRAINED-NEXT:    [[TMP7:%.*]] = bitcast <16 x i8> [[TMP4]] to <8 x half>
-// CONSTRAINED-NEXT:    [[TMP8:%.*]] = bitcast <16 x i8> [[TMP5]] to <8 x half>
-// CONSTRAINED-NEXT:    [[TMP9:%.*]] = call <8 x half> @llvm.experimental.constrained.fma.v8f16(<8 x half> [[TMP7]], <8 x half> [[TMP8]], <8 x half> [[TMP6]], metadata !"round.tonearest", metadata !"fpexcept.strict") #[[ATTR2]]
-// CONSTRAINED-NEXT:    ret <8 x half> [[TMP9]]
-//
-float16x8_t test_vfmaq_f16(float16x8_t a, float16x8_t b, float16x8_t c) {
-  return vfmaq_f16(a, b, c);
-}
-
 // UNCONSTRAINED-LABEL: define dso_local <4 x half> @test_vfms_f16(
 // UNCONSTRAINED-SAME: <4 x half> noundef [[A:%.*]], <4 x half> noundef [[B:%.*]], <4 x half> noundef [[C:%.*]]) #[[ATTR0]] {
 // UNCONSTRAINED-NEXT:  [[ENTRY:.*:]]
@@ -205,150 +137,6 @@ float16x8_t test_vfmsq_f16(float16x8_t a, float16x8_t b, float16x8_t c) {
   return vfmsq_f16(a, b, c);
 }
 
-// UNCONSTRAINED-LABEL: define dso_local <4 x half> @test_vfma_lane_f16(
-// UNCONSTRAINED-SAME: <4 x half> noundef [[A:%.*]], <4 x half> noundef [[B:%.*]], <4 x half> noundef [[C:%.*]]) #[[ATTR0]] {
-// UNCONSTRAINED-NEXT:  [[ENTRY:.*:]]
-// UNCONSTRAINED-NEXT:    [[TMP0:%.*]] = bitcast <4 x half> [[A]] to <4 x i16>
-// UNCONSTRAINED-NEXT:    [[TMP1:%.*]] = bitcast <4 x half> [[B]] to <4 x i16>
-// UNCONSTRAINED-NEXT:    [[TMP2:%.*]] = bitcast <4 x half> [[C]] to <4 x i16>
-// UNCONSTRAINED-NEXT:    [[TMP3:%.*]] = bitcast <4 x i16> [[TMP0]] to <8 x i8>
-// UNCONSTRAINED-NEXT:    [[TMP4:%.*]] = bitcast <4 x i16> [[TMP1]] to <8 x i8>
-// UNCONSTRAINED-NEXT:    [[TMP5:%.*]] = bitcast <4 x i16> [[TMP2]] to <8 x i8>
-// UNCONSTRAINED-NEXT:    [[TMP6:%.*]] = bitcast <8 x i8> [[TMP5]] to <4 x half>
-// UNCONSTRAINED-NEXT:    [[LANE:%.*]] = shufflevector <4 x half> [[TMP6]], <4 x half> [[TMP6]], <4 x i32> <i32 3, i32 3, i32 3, i32 3>
-// UNCONSTRAINED-NEXT:    [[FMLA:%.*]] = bitcast <8 x i8> [[TMP4]] to <4 x half>
-// UNCONSTRAINED-NEXT:    [[FMLA1:%.*]] = bitcast <8 x i8> [[TMP3]] to <4 x half>
-// UNCONSTRAINED-NEXT:    [[FMLA2:%.*]] = call <4 x half> @llvm.fma.v4f16(<4 x half> [[FMLA]], <4 x half> [[LANE]], <4 x half> [[FMLA1]])
-// UNCONSTRAINED-NEXT:    ret <4 x half> [[FMLA2]]
-//
-// CONSTRAINED-LABEL: define dso_local <4 x half> @test_vfma_lane_f16(
-// CONSTRAINED-SAME: <4 x half> noundef [[A:%.*]], <4 x half> noundef [[B:%.*]], <4 x half> noundef [[C:%.*]]) #[[ATTR0]] {
-// CONSTRAINED-NEXT:  [[ENTRY:.*:]]
-// CONSTRAINED-NEXT:    [[TMP0:%.*]] = bitcast <4 x half> [[A]] to <4 x i16>
-// CONSTRAINED-NEXT:    [[TMP1:%.*]] = bitcast <4 x half> [[B]] to <4 x i16>
-// CONSTRAINED-NEXT:    [[TMP2:%.*]] = bitcast <4 x half> [[C]] to <4 x i16>
-// CONSTRAINED-NEXT:    [[TMP3:%.*]] = bitcast <4 x i16> [[TMP0]] to <8 x i8>
-// CONSTRAINED-NEXT:    [[TMP4:%.*]] = bitcast <4 x i16> [[TMP1]] to <8 x i8>
-// CONSTRAINED-NEXT:    [[TMP5:%.*]] = bitcast <4 x i16> [[TMP2]] to <8 x i8>
-// CONSTRAINED-NEXT:    [[TMP6:%.*]] = bitcast <8 x i8> [[TMP5]] to <4 x half>
-// CONSTRAINED-NEXT:    [[LANE:%.*]] = shufflevector <4 x half> [[TMP6]], <4 x half> [[TMP6]], <4 x i32> <i32 3, i32 3, i32 3, i32 3>
-// CONSTRAINED-NEXT:    [[FMLA:%.*]] = bitcast <8 x i8> [[TMP4]] to <4 x half>
-// CONSTRAINED-NEXT:    [[FMLA1:%.*]] = bitcast <8 x i8> [[TMP3]] to <4 x half>
-// CONSTRAINED-NEXT:    [[FMLA2:%.*]] = call <4 x half> @llvm.experimental.constrained.fma.v4f16(<4 x half> [[FMLA]], <4 x half> [[LANE]], <4 x half> [[FMLA1]], metadata !"round.tonearest", metadata !"fpexcept.strict") #[[ATTR2]]
-// CONSTRAINED-NEXT:    ret <4 x half> [[FMLA2]]
-//
-float16x4_t test_vfma_lane_f16(float16x4_t a, float16x4_t b, float16x4_t c) {
-  return vfma_lane_f16(a, b, c, 3);
-}
-
-// UNCONSTRAINED-LABEL: define dso_local <8 x half> @test_vfmaq_lane_f16(
-// UNCONSTRAINED-SAME: <8 x half> noundef [[A:%.*]], <8 x half> noundef [[B:%.*]], <4 x half> noundef [[C:%.*]]) #[[ATTR0]] {
-// UNCONSTRAINED-NEXT:  [[ENTRY:.*:]]
-// UNCONSTRAINED-NEXT:    [[TMP0:%.*]] = bitcast <8 x half> [[A]] to <8 x i16>
-// UNCONSTRAINED-NEXT:    [[TMP1:%.*]] = bitcast <8 x half> [[B]] to <8 x i16>
-// UNCONSTRAINED-NEXT:    [[TMP2:%.*]] = bitcast <4 x half> [[C]] to <4 x i16>
-// UNCONSTRAINED-NEXT:    [[TMP3:%.*]] = bitcast <8 x i16> [[TMP0]] to <16 x i8>
-// UNCONSTRAINED-NEXT:    [[TMP4:%.*]] = bitcast <8 x i16> [[TMP1]] to <16 x i8>
-// UNCONSTRAINED-NEXT:    [[TMP5:%.*]] = bitcast <4 x i16> [[TMP2]] to <8 x i8>
-// UNCONSTRAINED-NEXT:    [[TMP6:%.*]] = bitcast <8 x i8> [[TMP5]] to <4 x half>
-// UNCONSTRAINED-NEXT:    [[LANE:%.*]] = shufflevector <4 x half> [[TMP6]], <4 x half> [[TMP6]], <8 x i32> <i32 3, i32 3, i32 3, i32 3, i32 3, i32 3, i32 3, i32 3>
-// UNCONSTRAINED-NEXT:    [[FMLA:%.*]] = bitcast <16 x i8> [[TMP4]] to <8 x half>
-// UNCONSTRAINED-NEXT:    [[FMLA1:%.*]] = bitcast <16 x i8> [[TMP3]] to <8 x half>
-// UNCONSTRAINED-NEXT:    [[FMLA2:%.*]] = call <8 x half> @llvm.fma.v8f16(<8 x half> [[FMLA]], <8 x half> [[LANE]], <8 x half> [[FMLA1]])
-// UNCONSTRAINED-NEXT:    ret <8 x half> [[FMLA2]]
-//
-// CONSTRAINED-LABEL: define dso_local <8 x half> @test_vfmaq_lane_f16(
-// CONSTRAINED-SAME: <8 x half> noundef [[A:%.*]], <8 x half> noundef [[B:%.*]], <4 x half> noundef [[C:%.*]]) #[[ATTR0]] {
-// CONSTRAINED-NEXT:  [[ENTRY:.*:]]
-// CONSTRAINED-NEXT:    [[TMP0:%.*]] = bitcast <8 x half> [[A]] to <8 x i16>
-// CONSTRAINED-NEXT:    [[TMP1:%.*]] = bitcast <8 x half> [[B]] to <8 x i16>
-// CONSTRAINED-NEXT:    [[TMP2:%.*]] = bitcast <4 x half> [[C]] to <4 x i16>
-// CONSTRAINED-NEXT:    [[TMP3:%.*]] = bitcast <8 x i16> [[TMP0]] to <16 x i8>
-// CONSTRAINED-NEXT:    [[TMP4:%.*]] = bitcast <8 x i16> [[TMP1]] to <16 x i8>
-// CONSTRAINED-NEXT:    [[TMP5:%.*]] = bitcast <4 x i16> [[TMP2]] to <8 x i8>
-// CONSTRAINED-NEXT:    [[TMP6:%.*]] = bitcast <8 x i8> [[TMP5]] to <4 x half>
-// CONSTRAINED-NEXT:    [[LANE:%.*]] = shufflevector <4 x half> [[TMP6]], <4 x half> [[TMP6]], <8 x i32> <i32 3, i32 3, i32 3, i32 3, i32 3, i32 3, i32 3, i32 3>
-// CONSTRAINED-NEXT:    [[FMLA:%.*]] = bitcast <16 x i8> [[TMP4]] to <8 x half>
-// CONSTRAINED-NEXT:    [[FMLA1:%.*]] = bitcast <16 x i8> [[TMP3]] to <8 x half>
-// CONSTRAINED-NEXT:    [[FMLA2:%.*]] = call <8 x half> @llvm.experimental.constrained.fma.v8f16(<8 x half> [[FMLA]], <8 x half> [[LANE]], <8 x half> [[FMLA1]], metadata !"round.tonearest", metadata !"fpexcept.strict") #[[ATTR2]]
-// CONSTRAINED-NEXT:    ret <8 x half> [[FMLA2]]
-//
-float16x8_t test_vfmaq_lane_f16(float16x8_t a, float16x8_t b, float16x4_t c) {
-  return vfmaq_lane_f16(a, b, c, 3);
-}
-
-// UNCONSTRAINED-LABEL: define dso_local <4 x half> @test_vfma_laneq_f16(
-// UNCONSTRAINED-SAME: <4 x half> noundef [[A:%.*]], <4 x half> noundef [[B:%.*]], <8 x half> noundef [[C:%.*]]) #[[ATTR0]] {
-// UNCONSTRAINED-NEXT:  [[ENTRY:.*:]]
-// UNCONSTRAINED-NEXT:    [[TMP0:%.*]] = bitcast <4 x half> [[A]] to <4 x i16>
-// UNCONSTRAINED-NEXT:    [[TMP1:%.*]] = bitcast <4 x half> [[B]] to <4 x i16>
-// UNCONSTRAINED-NEXT:    [[TMP2:%.*]] = bitcast <8 x half> [[C]] to <8 x i16>
-// UNCONSTRAINED-NEXT:    [[TMP3:%.*]] = bitcast <4 x i16> [[TMP0]] to <8 x i8>
-// UNCONSTRAINED-NEXT:    [[TMP4:%.*]] = bitcast <4 x i16> [[TMP1]] to <8 x i8>
-// UNCONSTRAINED-NEXT:    [[TMP5:%.*]] = bitcast <8 x i16> [[TMP2]] to <16 x i8>
-// UNCONSTRAINED-NEXT:    [[TMP6:%.*]] = bitcast <8 x i8> [[TMP3]] to <4 x half>
-// UNCONSTRAINED-NEXT:    [[TMP7:%.*]] = bitcast <8 x i8> [[TMP4]] to <4 x half>
-// UNCONSTRAINED-NEXT:    [[TMP8:%.*]] = bitcast <16 x i8> [[TMP5]] to <8 x half>
-// UNCONSTRAINED-NEXT:    [[LANE:%.*]] = shufflevector <8 x half> [[TMP8]], <8 x half> [[TMP8]], <4 x i32> <i32 7, i32 7, i32 7, i32 7>
-// UNCONSTRAINED-NEXT:    [[TMP9:%.*]] = call <4 x half> @llvm.fma.v4f16(<4 x half> [[LANE]], <4 x half> [[TMP7]], <4 x half> [[TMP6]])
-// UNCONSTRAINED-NEXT:    ret <4 x half> [[TMP9]]
-//
-// CONSTRAINED-LABEL: define dso_local <4 x half> @test_vfma_laneq_f16(
-// CONSTRAINED-SAME: <4 x half> noundef [[A:%.*]], <4 x half> noundef [[B:%.*]], <8 x half> noundef [[C:%.*]]) #[[ATTR0]] {
-// CONSTRAINED-NEXT:  [[ENTRY:.*:]]
-// CONSTRAINED-NEXT:    [[TMP0:%.*]] = bitcast <4 x half> [[A]] to <4 x i16>
-// CONSTRAINED-NEXT:    [[TMP1:%.*]] = bitcast <4 x half> [[B]] to <4 x i16>
-// CONSTRAINED-NEXT:    [[TMP2:%.*]] = bitcast <8 x half> [[C]] to <8 x i16>
-// CONSTRAINED-NEXT:    [[TMP3:%.*]] = bitcast <4 x i16> [[TMP0]] to <8 x i8>
-// CONSTRAINED-NEXT:    [[TMP4:%.*]] = bitcast <4 x i16> [[TMP1]] to <8 x i8>
-// CONSTRAINED-NEXT:    [[TMP5:%.*]] = bitcast <8 x i16> [[TMP2]] to <16 x i8>
-// CONSTRAINED-NEXT:    [[TMP6:%.*]] = bitcast <8 x i8> [[TMP3]] to <4 x half>
-// CONSTRAINED-NEXT:    [[TMP7:%.*]] = bitcast <8 x i8> [[TMP4]] to <4 x half>
-// CONSTRAINED-NEXT:    [[TMP8:%.*]] = bitcast <16 x i8> [[TMP5]] to <8 x half>
-// CONSTRAINED-NEXT:    [[LANE:%.*]] = shufflevector <8 x half> [[TMP8]], <8 x half> [[TMP8]], <4 x i32> <i32 7, i32 7, i32 7, i32 7>
-// CONSTRAINED-NEXT:    [[TMP9:%.*]] = call <4 x half> @llvm.experimental.constrained.fma.v4f16(<4 x half> [[LANE]], <4 x half> [[TMP7]], <4 x half> [[TMP6]], metadata !"round.tonearest", metadata !"fpexcept.strict") #[[ATTR2]]
-// CONSTRAINED-NEXT:    ret <4 x half> [[TMP9]]
-//
-float16x4_t test_vfma_laneq_f16(float16x4_t a, float16x4_t b, float16x8_t c) {
-  return vfma_laneq_f16(a, b, c, 7);
-}
-
-// UNCONSTRAINED-LABEL: define dso_local <8 x half> @test_vfmaq_laneq_f16(
-// UNCONSTRAINED-SAME: <8 x half> noundef [[A:%.*]], <8 x half> noundef [[B:%.*]], <8 x half> noundef [[C:%.*]]) #[[ATTR0]] {
-// UNCONSTRAINED-NEXT:  [[ENTRY:.*:]]
-// UNCONSTRAINED-NEXT:    [[TMP0:%.*]] = bitcast <8 x half> [[A]] to <8 x i16>
-// UNCONSTRAINED-NEXT:    [[TMP1:%.*]] = bitcast <8 x half> [[B]] to <8 x i16>
-// UNCONSTRAINED-NEXT:    [[TMP2:%.*]] = bitcast <8 x half> [[C]] to <8 x i16>
-// UNCONSTRAINED-NEXT:    [[TMP3:%.*]] = bitcast <8 x i16> [[TMP0]] to <16 x i8>
-// UNCONSTRAINED-NEXT:    [[TMP4:%.*]] = bitcast <8 x i16> [[TMP1]] to <16 x i8>
-// UNCONSTRAINED-NEXT:    [[TMP5:%.*]] = bitcast <8 x i16> [[TMP2]] to <16 x i8>
-// UNCONSTRAINED-NEXT:    [[TMP6:%.*]] = bitcast <16 x i8> [[TMP3]] to <8 x half>
-// UNCONSTRAINED-NEXT:    [[TMP7:%.*]] = bitcast <16 x i8> [[TMP4]] to <8 x half>
-// UNCONSTRAINED-NEXT:    [[TMP8:%.*]] = bitcast <16 x i8> [[TMP5]] to <8 x half>
-// UNCONSTRAINED-NEXT:    [[LANE:%.*]] = shufflevector <8 x half> [[TMP8]], <8 x half> [[TMP8]], <8 x i32> <i32 7, i32 7, i32 7, i32 7, i32 7, i32 7, i32 7, i32 7>
-// UNCONSTRAINED-NEXT:    [[TMP9:%.*]] = call <8 x half> @llvm.fma.v8f16(<8 x half> [[LANE]], <8 x half> [[TMP7]], <8 x half> [[TMP6]])
-// UNCONSTRAINED-NEXT:    ret <8 x half> [[TMP9]]
-//
-// CONSTRAINED-LABEL: define dso_local <8 x half> @test_vfmaq_laneq_f16(
-// CONSTRAINED-SAME: <8 x half> noundef [[A:%.*]], <8 x half> noundef [[B:%.*]], <8 x half> noundef [[C:%.*]]) #[[ATTR0]] {
-// CONSTRAINED-NEXT:  [[ENTRY:.*:]]
-// CONSTRAINED-NEXT:    [[TMP0:%.*]] = bitcast <8 x half> [[A]] to <8 x i16>
-// CONSTRAINED-NEXT:    [[TMP1:%.*]] = bitcast <8 x half> [[B]] to <8 x i16>
-// CONSTRAINED-NEXT:    [[TMP2:%.*]] = bitcast <8 x half> [[C]] to <8 x i16>
-// CONSTRAINED-NEXT:    [[TMP3:%.*]] = bitcast <8 x i16> [[TMP0]] to <16 x i8>
-// CONSTRAINED-NEXT:    [[TMP4:%.*]] = bitcast <8 x i16> [[TMP1]] to <16 x i8>
-// CONSTRAINED-NEXT:    [[TMP5:%.*]] = bitcast <8 x i16> [[TMP2]] to <16 x i8>
-// CONSTRAINED-NEXT:    [[TMP6:%.*]] = bitcast <16 x i8> [[TMP3]] to <8 x half>
-// CONSTRAINED-NEXT:    [[TMP7:%.*]] = bitcast <16 x i8> [[TMP4]] to <8 x half>
-// CONSTRAINED-NEXT:    [[TMP8:%.*]] = bitcast <16 x i8> [[TMP5]] to <8 x half>
-// CONSTRAINED-NEXT:    [[LANE:%.*]] = shufflevector <8 x half> [[TMP8]], <8 x half> [[TMP8]], <8 x i32> <i32 7, i32 7, i32 7, i32 7, i32 7, i32 7, i32 7, i32 7>
-// CONSTRAINED-NEXT:    [[TMP9:%.*]] = call <8 x half> @llvm.experimental.constrained.fma.v8f16(<8 x half> [[LANE]], <8 x half> [[TMP7]], <8 x half> [[TMP6]], metadata !"round.tonearest", metadata !"fpexcept.strict") #[[ATTR2]]
-// CONSTRAINED-NEXT:    ret <8 x half> [[TMP9]]
-//
-float16x8_t test_vfmaq_laneq_f16(float16x8_t a, float16x8_t b, float16x8_t c) {
-  return vfmaq_laneq_f16(a, b, c, 7);
-}
-
 // UNCONSTRAINED-LABEL: define dso_local <4 x half> @test_vfma_n_f16(
 // UNCONSTRAINED-SAME: <4 x half> noundef [[A:%.*]], <4 x half> noundef [[B:%.*]], half noundef [[C:%.*]]) #[[ATTR0]] {
 // UNCONSTRAINED-NEXT:  [[ENTRY:.*:]]
@@ -441,24 +229,6 @@ float16x8_t test_vfmaq_n_f16(float16x8_t a, float16x8_t b, float16_t c) {
   return vfmaq_n_f16(a, b, c);
 }
 
-// UNCONSTRAINED-LABEL: define dso_local half @test_vfmah_lane_f16(
-// UNCONSTRAINED-SAME: half noundef [[A:%.*]], half noundef [[B:%.*]], <4 x half> noundef [[C:%.*]]) #[[ATTR0]] {
-// UNCONSTRAINED-NEXT:  [[ENTRY:.*:]]
-// UNCONSTRAINED-NEXT:    [[EXTRACT:%.*]] = extractelement <4 x half> [[C]], i32 3
-// UNCONSTRAINED-NEXT:    [[TMP0:%.*]] = call half @llvm.fma.f16(half [[B]], half [[EXTRACT]], half [[A]])
-// UNCONSTRAINED-NEXT:    ret half [[TMP0]]
-//
-// CONSTRAINED-LABEL: define dso_local half @test_vfmah_lane_f16(
-// CONSTRAINED-SAME: half noundef [[A:%.*]], half noundef [[B:%.*]], <4 x half> noundef [[C:%.*]]) #[[ATTR0]] {
-// CONSTRAINED-NEXT:  [[ENTRY:.*:]]
-// CONSTRAINED-NEXT:    [[EXTRACT:%.*]] = extractelement <4 x half> [[C]], i32 3
-// CONSTRAINED-NEXT:    [[TMP0:%.*]] = call half @llvm.experimental.constrained.fma.f16(half [[B]], half [[EXTRACT]], half [[A]], metadata !"round.tonearest", metadata !"fpexcept.strict") #[[ATTR2]]
-// CONSTRAINED-NEXT:    ret half [[TMP0]]
-//
-float16_t test_vfmah_lane_f16(float16_t a, float16_t b, float16x4_t c) {
-  return vfmah_lane_f16(a, b, c, 3);
-}
-
 // UNCONSTRAINED-LABEL: define dso_local half @test_vfmah_laneq_f16(
 // UNCONSTRAINED-SAME: half noundef [[A:%.*]], half noundef [[B:%.*]], <8 x half> noundef [[C:%.*]]) #[[ATTR0]] {
 // UNCONSTRAINED-NEXT:  [[ENTRY:.*:]]

>From b67b80312f571e5c0c9a859a8e462c1785bf73f4 Mon Sep 17 00:00:00 2001
From: Yair Ben Avraham <yairba at protonmail.com>
Date: Fri, 25 Sep 2026 06:44:26 +0300
Subject: [PATCH 14/16] [CIR][AArch64] Migrate sqrt strict tests

Check the three existing Neon sqrt wrappers with classic Clang and
CIR under strict FP and pragma-overridden maytrap modes.

Remove their duplicate legacy constrained blocks. Unconstrained
coverage remains in the regular AArch64 and Neon tests. Rebind the
attribute captures in the surviving FP16 legacy checks.
---
 .../CodeGen/AArch64/neon-misc-constrained.c   | 22 --------
 .../AArch64/neon/intrinsics-constrained.c     |  2 +
 .../v8.2a-neon-intrinsics-constrained.c       | 50 ++-----------------
 3 files changed, 5 insertions(+), 69 deletions(-)

diff --git a/clang/test/CodeGen/AArch64/neon-misc-constrained.c b/clang/test/CodeGen/AArch64/neon-misc-constrained.c
index 49208892e3035..f1f638cb272c6 100644
--- a/clang/test/CodeGen/AArch64/neon-misc-constrained.c
+++ b/clang/test/CodeGen/AArch64/neon-misc-constrained.c
@@ -82,28 +82,6 @@ float32x4_t test_vsqrtq_f32(float32x4_t a) {
 }
 
 
-// UNCONSTRAINED-LABEL: define dso_local <2 x double> @test_vsqrtq_f64(
-// UNCONSTRAINED-SAME: <2 x double> noundef [[A:%.*]]) #[[ATTR0]] {
-// UNCONSTRAINED-NEXT:  [[ENTRY:.*:]]
-// UNCONSTRAINED-NEXT:    [[TMP0:%.*]] = bitcast <2 x double> [[A]] to <2 x i64>
-// UNCONSTRAINED-NEXT:    [[TMP1:%.*]] = bitcast <2 x i64> [[TMP0]] to <16 x i8>
-// UNCONSTRAINED-NEXT:    [[TMP2:%.*]] = bitcast <16 x i8> [[TMP1]] to <2 x double>
-// UNCONSTRAINED-NEXT:    [[VSQRT_I:%.*]] = call <2 x double> @llvm.sqrt.v2f64(<2 x double> [[TMP2]])
-// UNCONSTRAINED-NEXT:    ret <2 x double> [[VSQRT_I]]
-//
-// CONSTRAINED-LABEL: define dso_local <2 x double> @test_vsqrtq_f64(
-// CONSTRAINED-SAME: <2 x double> noundef [[A:%.*]]) #[[ATTR0]] {
-// CONSTRAINED-NEXT:  [[ENTRY:.*:]]
-// CONSTRAINED-NEXT:    [[TMP0:%.*]] = bitcast <2 x double> [[A]] to <2 x i64>
-// CONSTRAINED-NEXT:    [[TMP1:%.*]] = bitcast <2 x i64> [[TMP0]] to <16 x i8>
-// CONSTRAINED-NEXT:    [[TMP2:%.*]] = bitcast <16 x i8> [[TMP1]] to <2 x double>
-// CONSTRAINED-NEXT:    [[VSQRT_I:%.*]] = call <2 x double> @llvm.experimental.constrained.sqrt.v2f64(<2 x double> [[TMP2]], metadata !"round.tonearest", metadata !"fpexcept.strict") #[[ATTR2]]
-// CONSTRAINED-NEXT:    ret <2 x double> [[VSQRT_I]]
-//
-float64x2_t test_vsqrtq_f64(float64x2_t a) {
-  return vsqrtq_f64(a);
-}
-
 // UNCONSTRAINED-LABEL: define dso_local <4 x half> @test_vcvt_f16_f32(
 // UNCONSTRAINED-SAME: <4 x float> noundef [[A:%.*]]) #[[ATTR0]] {
 // UNCONSTRAINED-NEXT:  [[ENTRY:.*:]]
diff --git a/clang/test/CodeGen/AArch64/neon/intrinsics-constrained.c b/clang/test/CodeGen/AArch64/neon/intrinsics-constrained.c
index 212cc89ad354d..79e637f2ea898 100644
--- a/clang/test/CodeGen/AArch64/neon/intrinsics-constrained.c
+++ b/clang/test/CodeGen/AArch64/neon/intrinsics-constrained.c
@@ -1,5 +1,7 @@
 // REQUIRES: aarch64-registered-target
 
+// RUN: %clang_cc1_cg_arm64_neon -target-feature +fullfp16 -ffp-exception-behavior=maytrap -DEXCEPT=1 -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,sroa | FileCheck %s --check-prefix=LLVM --implicit-check-not=fpexcept.maytrap --implicit-check-not=' @llvm.sqrt.'
+// RUN: %clang_cc1_cg_arm64_neon -target-feature +fullfp16 -ffp-exception-behavior=strict -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,sroa | FileCheck %s --check-prefix=LLVM --implicit-check-not=' @llvm.sqrt.'
 // RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 -fexperimental-strict-floating-point -ffp-exception-behavior=maytrap -DEXCEPT=1 -fclangir -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,sroa | FileCheck %s --check-prefix=LLVM --implicit-check-not=fpexcept.maytrap --implicit-check-not=' @llvm.sqrt.' %}
 // RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 -fexperimental-strict-floating-point -ffp-exception-behavior=maytrap -DEXCEPT=1 -fclangir -emit-cir  %s -disable-O0-optnone |                               FileCheck %s --check-prefix=CIR --implicit-check-not='except_mode = maytrap' --implicit-check-not='cir.call_llvm_intrinsic "sqrt"' %}
 // RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 -fexperimental-strict-floating-point -ffp-exception-behavior=strict             -fclangir -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,sroa | FileCheck %s --check-prefix=LLVM --implicit-check-not=' @llvm.sqrt.' %}
diff --git a/clang/test/CodeGen/AArch64/v8.2a-neon-intrinsics-constrained.c b/clang/test/CodeGen/AArch64/v8.2a-neon-intrinsics-constrained.c
index f04657d7bd0c2..a30bfa8352646 100644
--- a/clang/test/CodeGen/AArch64/v8.2a-neon-intrinsics-constrained.c
+++ b/clang/test/CodeGen/AArch64/v8.2a-neon-intrinsics-constrained.c
@@ -21,52 +21,8 @@
 
 #include <arm_neon.h>
 
-// UNCONSTRAINED-LABEL: define dso_local <4 x half> @test_vsqrt_f16(
-// UNCONSTRAINED-SAME: <4 x half> noundef [[A:%.*]]) #[[ATTR0:[0-9]+]] {
-// UNCONSTRAINED-NEXT:  [[ENTRY:.*:]]
-// UNCONSTRAINED-NEXT:    [[TMP0:%.*]] = bitcast <4 x half> [[A]] to <4 x i16>
-// UNCONSTRAINED-NEXT:    [[TMP1:%.*]] = bitcast <4 x i16> [[TMP0]] to <8 x i8>
-// UNCONSTRAINED-NEXT:    [[TMP2:%.*]] = bitcast <8 x i8> [[TMP1]] to <4 x half>
-// UNCONSTRAINED-NEXT:    [[VSQRT_I:%.*]] = call <4 x half> @llvm.sqrt.v4f16(<4 x half> [[TMP2]])
-// UNCONSTRAINED-NEXT:    ret <4 x half> [[VSQRT_I]]
-//
-// CONSTRAINED-LABEL: define dso_local <4 x half> @test_vsqrt_f16(
-// CONSTRAINED-SAME: <4 x half> noundef [[A:%.*]]) #[[ATTR0:[0-9]+]] {
-// CONSTRAINED-NEXT:  [[ENTRY:.*:]]
-// CONSTRAINED-NEXT:    [[TMP0:%.*]] = bitcast <4 x half> [[A]] to <4 x i16>
-// CONSTRAINED-NEXT:    [[TMP1:%.*]] = bitcast <4 x i16> [[TMP0]] to <8 x i8>
-// CONSTRAINED-NEXT:    [[TMP2:%.*]] = bitcast <8 x i8> [[TMP1]] to <4 x half>
-// CONSTRAINED-NEXT:    [[VSQRT_I:%.*]] = call <4 x half> @llvm.experimental.constrained.sqrt.v4f16(<4 x half> [[TMP2]], metadata !"round.tonearest", metadata !"fpexcept.strict") #[[ATTR2:[0-9]+]]
-// CONSTRAINED-NEXT:    ret <4 x half> [[VSQRT_I]]
-//
-float16x4_t test_vsqrt_f16(float16x4_t a) {
-  return vsqrt_f16(a);
-}
-
-// UNCONSTRAINED-LABEL: define dso_local <8 x half> @test_vsqrtq_f16(
-// UNCONSTRAINED-SAME: <8 x half> noundef [[A:%.*]]) #[[ATTR0]] {
-// UNCONSTRAINED-NEXT:  [[ENTRY:.*:]]
-// UNCONSTRAINED-NEXT:    [[TMP0:%.*]] = bitcast <8 x half> [[A]] to <8 x i16>
-// UNCONSTRAINED-NEXT:    [[TMP1:%.*]] = bitcast <8 x i16> [[TMP0]] to <16 x i8>
-// UNCONSTRAINED-NEXT:    [[TMP2:%.*]] = bitcast <16 x i8> [[TMP1]] to <8 x half>
-// UNCONSTRAINED-NEXT:    [[VSQRT_I:%.*]] = call <8 x half> @llvm.sqrt.v8f16(<8 x half> [[TMP2]])
-// UNCONSTRAINED-NEXT:    ret <8 x half> [[VSQRT_I]]
-//
-// CONSTRAINED-LABEL: define dso_local <8 x half> @test_vsqrtq_f16(
-// CONSTRAINED-SAME: <8 x half> noundef [[A:%.*]]) #[[ATTR0]] {
-// CONSTRAINED-NEXT:  [[ENTRY:.*:]]
-// CONSTRAINED-NEXT:    [[TMP0:%.*]] = bitcast <8 x half> [[A]] to <8 x i16>
-// CONSTRAINED-NEXT:    [[TMP1:%.*]] = bitcast <8 x i16> [[TMP0]] to <16 x i8>
-// CONSTRAINED-NEXT:    [[TMP2:%.*]] = bitcast <16 x i8> [[TMP1]] to <8 x half>
-// CONSTRAINED-NEXT:    [[VSQRT_I:%.*]] = call <8 x half> @llvm.experimental.constrained.sqrt.v8f16(<8 x half> [[TMP2]], metadata !"round.tonearest", metadata !"fpexcept.strict") #[[ATTR2]]
-// CONSTRAINED-NEXT:    ret <8 x half> [[VSQRT_I]]
-//
-float16x8_t test_vsqrtq_f16(float16x8_t a) {
-  return vsqrtq_f16(a);
-}
-
 // UNCONSTRAINED-LABEL: define dso_local <4 x half> @test_vfms_f16(
-// UNCONSTRAINED-SAME: <4 x half> noundef [[A:%.*]], <4 x half> noundef [[B:%.*]], <4 x half> noundef [[C:%.*]]) #[[ATTR0]] {
+// UNCONSTRAINED-SAME: <4 x half> noundef [[A:%.*]], <4 x half> noundef [[B:%.*]], <4 x half> noundef [[C:%.*]]) #[[ATTR0:[0-9]+]] {
 // UNCONSTRAINED-NEXT:  [[ENTRY:.*:]]
 // UNCONSTRAINED-NEXT:    [[FNEG_I:%.*]] = fneg <4 x half> [[B]]
 // UNCONSTRAINED-NEXT:    [[TMP0:%.*]] = bitcast <4 x half> [[A]] to <4 x i16>
@@ -82,7 +38,7 @@ float16x8_t test_vsqrtq_f16(float16x8_t a) {
 // UNCONSTRAINED-NEXT:    ret <4 x half> [[TMP9]]
 //
 // CONSTRAINED-LABEL: define dso_local <4 x half> @test_vfms_f16(
-// CONSTRAINED-SAME: <4 x half> noundef [[A:%.*]], <4 x half> noundef [[B:%.*]], <4 x half> noundef [[C:%.*]]) #[[ATTR0]] {
+// CONSTRAINED-SAME: <4 x half> noundef [[A:%.*]], <4 x half> noundef [[B:%.*]], <4 x half> noundef [[C:%.*]]) #[[ATTR0:[0-9]+]] {
 // CONSTRAINED-NEXT:  [[ENTRY:.*:]]
 // CONSTRAINED-NEXT:    [[FNEG_I:%.*]] = fneg <4 x half> [[B]]
 // CONSTRAINED-NEXT:    [[TMP0:%.*]] = bitcast <4 x half> [[A]] to <4 x i16>
@@ -94,7 +50,7 @@ float16x8_t test_vsqrtq_f16(float16x8_t a) {
 // CONSTRAINED-NEXT:    [[TMP6:%.*]] = bitcast <8 x i8> [[TMP3]] to <4 x half>
 // CONSTRAINED-NEXT:    [[TMP7:%.*]] = bitcast <8 x i8> [[TMP4]] to <4 x half>
 // CONSTRAINED-NEXT:    [[TMP8:%.*]] = bitcast <8 x i8> [[TMP5]] to <4 x half>
-// CONSTRAINED-NEXT:    [[TMP9:%.*]] = call <4 x half> @llvm.experimental.constrained.fma.v4f16(<4 x half> [[TMP7]], <4 x half> [[TMP8]], <4 x half> [[TMP6]], metadata !"round.tonearest", metadata !"fpexcept.strict") #[[ATTR2]]
+// CONSTRAINED-NEXT:    [[TMP9:%.*]] = call <4 x half> @llvm.experimental.constrained.fma.v4f16(<4 x half> [[TMP7]], <4 x half> [[TMP8]], <4 x half> [[TMP6]], metadata !"round.tonearest", metadata !"fpexcept.strict") #[[ATTR2:[0-9]+]]
 // CONSTRAINED-NEXT:    ret <4 x half> [[TMP9]]
 //
 float16x4_t test_vfms_f16(float16x4_t a, float16x4_t b, float16x4_t c) {

>From fbd59dc24949d0b18111fa6bc35e37e864844faa Mon Sep 17 00:00:00 2001
From: Yair Ben Avraham <yairba at protonmail.com>
Date: Tue, 29 Sep 2026 11:20:47 +0300
Subject: [PATCH 15/16] [CIR][AArch64] Share matching constrained test labels

Use a common FileCheck prefix where CIR and LLVM IR have the same
function label. Keep distinct labels for inline Neon wrappers whose
names differ between the two representations.
---
 .../AArch64/neon/fullfp16-constrained.c       | 16 +++++------
 .../fused-multiple-fullfp16-constrained.c     | 27 ++++++++-----------
 .../AArch64/neon/fused-multiply-constrained.c | 15 +++++------
 3 files changed, 24 insertions(+), 34 deletions(-)

diff --git a/clang/test/CodeGen/AArch64/neon/fullfp16-constrained.c b/clang/test/CodeGen/AArch64/neon/fullfp16-constrained.c
index 9f5b99b0aa587..78b3d9a1a2375 100644
--- a/clang/test/CodeGen/AArch64/neon/fullfp16-constrained.c
+++ b/clang/test/CodeGen/AArch64/neon/fullfp16-constrained.c
@@ -1,10 +1,10 @@
 // REQUIRES: aarch64-registered-target
 
-// RUN: %clang_cc1_cg_arm64_neon -target-feature +fullfp16 -ffp-exception-behavior=strict -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,simplifycfg | FileCheck %s --check-prefix=LLVM --implicit-check-not=' @llvm.fma.' --implicit-check-not=' @llvm.sqrt.'
-// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 -fexperimental-strict-floating-point -ffp-exception-behavior=maytrap -DEXCEPT=1 -fclangir -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,simplifycfg | FileCheck %s --check-prefix=LLVM --implicit-check-not=fpexcept.maytrap --implicit-check-not=' @llvm.fma.' --implicit-check-not=' @llvm.sqrt.' %}
-// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 -fexperimental-strict-floating-point -ffp-exception-behavior=maytrap -DEXCEPT=1 -fclangir -emit-cir  %s -disable-O0-optnone |                                      FileCheck %s --check-prefix=CIR --implicit-check-not='except_mode = maytrap' --implicit-check-not='cir.call_llvm_intrinsic "fma"' --implicit-check-not='cir.call_llvm_intrinsic "sqrt"' %}
-// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 -fexperimental-strict-floating-point -ffp-exception-behavior=strict             -fclangir -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,simplifycfg | FileCheck %s --check-prefix=LLVM --implicit-check-not=' @llvm.fma.' --implicit-check-not=' @llvm.sqrt.' %}
-// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 -fexperimental-strict-floating-point -ffp-exception-behavior=strict             -fclangir -emit-cir  %s -disable-O0-optnone |                                      FileCheck %s --check-prefix=CIR --implicit-check-not='cir.call_llvm_intrinsic "fma"' --implicit-check-not='cir.call_llvm_intrinsic "sqrt"' %}
+// RUN: %clang_cc1_cg_arm64_neon -target-feature +fullfp16 -ffp-exception-behavior=strict -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,simplifycfg | FileCheck %s --check-prefixes=ALL,LLVM --implicit-check-not=' @llvm.fma.' --implicit-check-not=' @llvm.sqrt.'
+// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 -fexperimental-strict-floating-point -ffp-exception-behavior=maytrap -DEXCEPT=1 -fclangir -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,simplifycfg | FileCheck %s --check-prefixes=ALL,LLVM --implicit-check-not=fpexcept.maytrap --implicit-check-not=' @llvm.fma.' --implicit-check-not=' @llvm.sqrt.' %}
+// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 -fexperimental-strict-floating-point -ffp-exception-behavior=maytrap -DEXCEPT=1 -fclangir -emit-cir  %s -disable-O0-optnone |                                      FileCheck %s --check-prefixes=ALL,CIR --implicit-check-not='except_mode = maytrap' --implicit-check-not='cir.call_llvm_intrinsic "fma"' --implicit-check-not='cir.call_llvm_intrinsic "sqrt"' %}
+// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 -fexperimental-strict-floating-point -ffp-exception-behavior=strict             -fclangir -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,simplifycfg | FileCheck %s --check-prefixes=ALL,LLVM --implicit-check-not=' @llvm.fma.' --implicit-check-not=' @llvm.sqrt.' %}
+// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 -fexperimental-strict-floating-point -ffp-exception-behavior=strict             -fclangir -emit-cir  %s -disable-O0-optnone |                                      FileCheck %s --check-prefixes=ALL,CIR --implicit-check-not='cir.call_llvm_intrinsic "fma"' --implicit-check-not='cir.call_llvm_intrinsic "sqrt"' %}
 
 #if EXCEPT
 #pragma float_control(except, on)
@@ -12,8 +12,7 @@
 
 #include <arm_fp16.h>
 
-// LLVM-LABEL: @test_vsqrth_f16(
-// CIR-LABEL: @test_vsqrth_f16(
+// ALL-LABEL: @test_vsqrth_f16(
 float16_t test_vsqrth_f16(float16_t a) {
 // CIR: cir.sqrt %{{.*}} : !cir.f16 {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
 
@@ -23,8 +22,7 @@ float16_t test_vsqrth_f16(float16_t a) {
   return vsqrth_f16(a);
 }
 
-// LLVM-LABEL: @test_vfmah_f16(
-// CIR-LABEL: @test_vfmah_f16(
+// ALL-LABEL: @test_vfmah_f16(
 float16_t test_vfmah_f16(float16_t a, float16_t b, float16_t c) {
 // CIR: cir.fma %{{.*}}, %{{.*}}, %{{.*}} : !cir.f16 {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
 
diff --git a/clang/test/CodeGen/AArch64/neon/fused-multiple-fullfp16-constrained.c b/clang/test/CodeGen/AArch64/neon/fused-multiple-fullfp16-constrained.c
index d22fc1551dc44..665f4a8a6026d 100644
--- a/clang/test/CodeGen/AArch64/neon/fused-multiple-fullfp16-constrained.c
+++ b/clang/test/CodeGen/AArch64/neon/fused-multiple-fullfp16-constrained.c
@@ -1,11 +1,11 @@
 // REQUIRES: aarch64-registered-target
 
-// RUN: %clang_cc1_cg_arm64_neon -target-feature +fullfp16 -ffp-exception-behavior=maytrap -DEXCEPT=1 -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,sroa | FileCheck %s --check-prefix=LLVM --implicit-check-not=fpexcept.maytrap --implicit-check-not=' @llvm.fma.'
-// RUN: %clang_cc1_cg_arm64_neon -target-feature +fullfp16 -ffp-exception-behavior=strict -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,sroa | FileCheck %s --check-prefix=LLVM --implicit-check-not=' @llvm.fma.'
-// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 -fexperimental-strict-floating-point -ffp-exception-behavior=maytrap -DEXCEPT=1 -fclangir -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,sroa | FileCheck %s --check-prefix=LLVM --implicit-check-not=fpexcept.maytrap --implicit-check-not=' @llvm.fma.' %}
-// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 -fexperimental-strict-floating-point -ffp-exception-behavior=maytrap -DEXCEPT=1 -fclangir -emit-cir  %s -disable-O0-optnone |                               FileCheck %s --check-prefix=CIR --implicit-check-not='except_mode = maytrap' --implicit-check-not='cir.call_llvm_intrinsic "fma"' %}
-// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 -fexperimental-strict-floating-point -ffp-exception-behavior=strict             -fclangir -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,sroa | FileCheck %s --check-prefix=LLVM --implicit-check-not=' @llvm.fma.' %}
-// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 -fexperimental-strict-floating-point -ffp-exception-behavior=strict             -fclangir -emit-cir  %s -disable-O0-optnone |                               FileCheck %s --check-prefix=CIR --implicit-check-not='cir.call_llvm_intrinsic "fma"' %}
+// RUN: %clang_cc1_cg_arm64_neon -target-feature +fullfp16 -ffp-exception-behavior=maytrap -DEXCEPT=1 -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,sroa | FileCheck %s --check-prefixes=ALL,LLVM --implicit-check-not=fpexcept.maytrap --implicit-check-not=' @llvm.fma.'
+// RUN: %clang_cc1_cg_arm64_neon -target-feature +fullfp16 -ffp-exception-behavior=strict -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,sroa | FileCheck %s --check-prefixes=ALL,LLVM --implicit-check-not=' @llvm.fma.'
+// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 -fexperimental-strict-floating-point -ffp-exception-behavior=maytrap -DEXCEPT=1 -fclangir -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,sroa | FileCheck %s --check-prefixes=ALL,LLVM --implicit-check-not=fpexcept.maytrap --implicit-check-not=' @llvm.fma.' %}
+// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 -fexperimental-strict-floating-point -ffp-exception-behavior=maytrap -DEXCEPT=1 -fclangir -emit-cir  %s -disable-O0-optnone |                               FileCheck %s --check-prefixes=ALL,CIR --implicit-check-not='except_mode = maytrap' --implicit-check-not='cir.call_llvm_intrinsic "fma"' %}
+// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 -fexperimental-strict-floating-point -ffp-exception-behavior=strict             -fclangir -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,sroa | FileCheck %s --check-prefixes=ALL,LLVM --implicit-check-not=' @llvm.fma.' %}
+// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 -fexperimental-strict-floating-point -ffp-exception-behavior=strict             -fclangir -emit-cir  %s -disable-O0-optnone |                               FileCheck %s --check-prefixes=ALL,CIR --implicit-check-not='cir.call_llvm_intrinsic "fma"' %}
 
 #if EXCEPT
 #pragma float_control(except, on)
@@ -53,8 +53,7 @@ float16x8_t test_vfmaq_f16(float16x8_t a, float16x8_t b, float16x8_t c) {
   return vfmaq_f16(a, b, c);
 }
 
-// LLVM-LABEL: @test_vfma_lane_f16(
-// CIR-LABEL: @test_vfma_lane_f16(
+// ALL-LABEL: @test_vfma_lane_f16(
 float16x4_t test_vfma_lane_f16(float16x4_t a, float16x4_t b,
                                 float16x4_t c) {
 // CIR: [[LANE:%.*]] = cir.vec.shuffle(%{{.*}}, %{{.*}} : !cir.vector<4 x !cir.f16>) [#cir.int<3> : !s32i, #cir.int<3> : !s32i, #cir.int<3> : !s32i, #cir.int<3> : !s32i] : !cir.vector<4 x !cir.f16>
@@ -76,8 +75,7 @@ float16x4_t test_vfma_lane_f16(float16x4_t a, float16x4_t b,
   return vfma_lane_f16(a, b, c, 3);
 }
 
-// LLVM-LABEL: @test_vfmaq_lane_f16(
-// CIR-LABEL: @test_vfmaq_lane_f16(
+// ALL-LABEL: @test_vfmaq_lane_f16(
 float16x8_t test_vfmaq_lane_f16(float16x8_t a, float16x8_t b,
                                  float16x4_t c) {
 // CIR: [[LANE:%.*]] = cir.vec.shuffle(%{{.*}}, %{{.*}} : !cir.vector<4 x !cir.f16>) [#cir.int<3> : !s32i, #cir.int<3> : !s32i, #cir.int<3> : !s32i, #cir.int<3> : !s32i, #cir.int<3> : !s32i, #cir.int<3> : !s32i, #cir.int<3> : !s32i, #cir.int<3> : !s32i] : !cir.vector<8 x !cir.f16>
@@ -99,8 +97,7 @@ float16x8_t test_vfmaq_lane_f16(float16x8_t a, float16x8_t b,
   return vfmaq_lane_f16(a, b, c, 3);
 }
 
-// LLVM-LABEL: @test_vfma_laneq_f16(
-// CIR-LABEL: @test_vfma_laneq_f16(
+// ALL-LABEL: @test_vfma_laneq_f16(
 float16x4_t test_vfma_laneq_f16(float16x4_t a, float16x4_t b,
                                  float16x8_t c) {
 // CIR: [[LANE:%.*]] = cir.vec.shuffle(%{{.*}}, %{{.*}} : !cir.vector<8 x !cir.f16>) [#cir.int<7> : !s32i, #cir.int<7> : !s32i, #cir.int<7> : !s32i, #cir.int<7> : !s32i] : !cir.vector<4 x !cir.f16>
@@ -122,8 +119,7 @@ float16x4_t test_vfma_laneq_f16(float16x4_t a, float16x4_t b,
   return vfma_laneq_f16(a, b, c, 7);
 }
 
-// LLVM-LABEL: @test_vfmaq_laneq_f16(
-// CIR-LABEL: @test_vfmaq_laneq_f16(
+// ALL-LABEL: @test_vfmaq_laneq_f16(
 float16x8_t test_vfmaq_laneq_f16(float16x8_t a, float16x8_t b,
                                   float16x8_t c) {
 // CIR: [[LANE:%.*]] = cir.vec.shuffle(%{{.*}}, %{{.*}} : !cir.vector<8 x !cir.f16>) [#cir.int<7> : !s32i, #cir.int<7> : !s32i, #cir.int<7> : !s32i, #cir.int<7> : !s32i, #cir.int<7> : !s32i, #cir.int<7> : !s32i, #cir.int<7> : !s32i, #cir.int<7> : !s32i] : !cir.vector<8 x !cir.f16>
@@ -145,8 +141,7 @@ float16x8_t test_vfmaq_laneq_f16(float16x8_t a, float16x8_t b,
   return vfmaq_laneq_f16(a, b, c, 7);
 }
 
-// LLVM-LABEL: @test_vfmah_lane_f16(
-// CIR-LABEL: @test_vfmah_lane_f16(
+// ALL-LABEL: @test_vfmah_lane_f16(
 float16_t test_vfmah_lane_f16(float16_t a, float16_t b, float16x4_t c) {
 // CIR: [[INDEX:%.*]] = cir.const #cir.int<3> : !u64i
 // CIR: [[LANE:%.*]] = cir.vec.extract %{{.*}}{{\[}}[[INDEX]] : !u64i] : !cir.vector<4 x !cir.f16>
diff --git a/clang/test/CodeGen/AArch64/neon/fused-multiply-constrained.c b/clang/test/CodeGen/AArch64/neon/fused-multiply-constrained.c
index 8920c40581e9f..b06f799d7f17b 100644
--- a/clang/test/CodeGen/AArch64/neon/fused-multiply-constrained.c
+++ b/clang/test/CodeGen/AArch64/neon/fused-multiply-constrained.c
@@ -1,8 +1,8 @@
 // REQUIRES: aarch64-registered-target
 
-// RUN:                   %clang_cc1_cg_arm64_neon -ffp-exception-behavior=strict -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,sroa | FileCheck %s --check-prefix=LLVM
-// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -fexperimental-strict-floating-point -ffp-exception-behavior=strict -fclangir -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,sroa | FileCheck %s --check-prefix=LLVM --implicit-check-not=' @llvm.fma.' %}
-// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -fexperimental-strict-floating-point -ffp-exception-behavior=strict -fclangir -emit-cir  %s -disable-O0-optnone |                               FileCheck %s --check-prefix=CIR --implicit-check-not='cir.call_llvm_intrinsic "fma"' %}
+// RUN:                   %clang_cc1_cg_arm64_neon -ffp-exception-behavior=strict -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,sroa | FileCheck %s --check-prefixes=ALL,LLVM
+// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -fexperimental-strict-floating-point -ffp-exception-behavior=strict -fclangir -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,sroa | FileCheck %s --check-prefixes=ALL,LLVM --implicit-check-not=' @llvm.fma.' %}
+// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -fexperimental-strict-floating-point -ffp-exception-behavior=strict -fclangir -emit-cir  %s -disable-O0-optnone |                               FileCheck %s --check-prefixes=ALL,CIR --implicit-check-not='cir.call_llvm_intrinsic "fma"' %}
 
 #include <arm_neon.h>
 
@@ -29,8 +29,7 @@ float64x1_t test_vfma_f64(float64x1_t a, float64x1_t b, float64x1_t c) {
   return vfma_f64(a, b, c);
 }
 
-// LLVM-LABEL: @test_vfma_laneq_f64(
-// CIR-LABEL: @test_vfma_laneq_f64(
+// ALL-LABEL: @test_vfma_laneq_f64(
 float64x1_t test_vfma_laneq_f64(float64x1_t a, float64x1_t b,
                                  float64x2_t v) {
 // CIR: [[INDEX:%.*]] = cir.const #cir.int<0> : !u64i
@@ -56,8 +55,7 @@ float64x1_t test_vfma_laneq_f64(float64x1_t a, float64x1_t b,
   return vfma_laneq_f64(a, b, v, 0);
 }
 
-// LLVM-LABEL: @test_vfmaq_laneq_f64(
-// CIR-LABEL: @test_vfmaq_laneq_f64(
+// ALL-LABEL: @test_vfmaq_laneq_f64(
 float64x2_t test_vfmaq_laneq_f64(float64x2_t a, float64x2_t b,
                                   float64x2_t v) {
 // CIR: [[LANE:%.*]] = cir.vec.shuffle(%{{.*}}, %{{.*}} : !cir.vector<2 x !cir.double>) [#cir.int<1> : !s32i, #cir.int<1> : !s32i] : !cir.vector<2 x !cir.double>
@@ -79,8 +77,7 @@ float64x2_t test_vfmaq_laneq_f64(float64x2_t a, float64x2_t b,
   return vfmaq_laneq_f64(a, b, v, 1);
 }
 
-// LLVM-LABEL: @test_vfmas_lane_f32(
-// CIR-LABEL: @test_vfmas_lane_f32(
+// ALL-LABEL: @test_vfmas_lane_f32(
 float32_t test_vfmas_lane_f32(float32_t a, float32_t b, float32x2_t c) {
 // CIR: [[INDEX:%.*]] = cir.const #cir.int<1> : !u64i
 // CIR: [[LANE:%.*]] = cir.vec.extract %{{.*}}{{\[}}[[INDEX]] : !u64i] : !cir.vector<2 x !cir.float>

>From 28747598a63b39ea90aeb0bd1a7a8e27f028eccd Mon Sep 17 00:00:00 2001
From: Yair Ben Avraham <yairba at protonmail.com>
Date: Tue, 29 Sep 2026 12:59:23 +0300
Subject: [PATCH 16/16] [CIR][AArch64] Update fenv test syntax

Match the explicit upstream fenv assembly format for CIR fma and sqrt.

Keep the FP environment values and test coverage unchanged.
---
 .../CodeGen/AArch64/neon/fullfp16-constrained.c    |  4 ++--
 .../neon/fused-multiple-fullfp16-constrained.c     | 14 +++++++-------
 .../AArch64/neon/fused-multiply-constrained.c      |  8 ++++----
 .../CodeGen/AArch64/neon/intrinsics-constrained.c  |  6 +++---
 4 files changed, 16 insertions(+), 16 deletions(-)

diff --git a/clang/test/CodeGen/AArch64/neon/fullfp16-constrained.c b/clang/test/CodeGen/AArch64/neon/fullfp16-constrained.c
index 78b3d9a1a2375..1a668bd657e0d 100644
--- a/clang/test/CodeGen/AArch64/neon/fullfp16-constrained.c
+++ b/clang/test/CodeGen/AArch64/neon/fullfp16-constrained.c
@@ -14,7 +14,7 @@
 
 // ALL-LABEL: @test_vsqrth_f16(
 float16_t test_vsqrth_f16(float16_t a) {
-// CIR: cir.sqrt %{{.*}} : !cir.f16 {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
+// CIR: cir.sqrt %{{.*}} : !cir.f16 fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>
 
 // LLVM-SAME: half {{.*}} [[A:%.*]]) {{.*}} {
 // LLVM: [[SQRT:%.*]] = call half @llvm.experimental.constrained.sqrt.f16(half [[A]], metadata !"round.tonearest", metadata !"fpexcept.strict")
@@ -24,7 +24,7 @@ float16_t test_vsqrth_f16(float16_t a) {
 
 // ALL-LABEL: @test_vfmah_f16(
 float16_t test_vfmah_f16(float16_t a, float16_t b, float16_t c) {
-// CIR: cir.fma %{{.*}}, %{{.*}}, %{{.*}} : !cir.f16 {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
+// CIR: cir.fma %{{.*}}, %{{.*}}, %{{.*}} : !cir.f16 fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>
 
 // LLVM-SAME: half {{.*}} [[A:%.*]], half {{.*}} [[B:%.*]], half {{.*}} [[C:%.*]]) {{.*}} {
 // LLVM: [[FMA:%.*]] = call half @llvm.experimental.constrained.fma.f16(half [[B]], half [[C]], half [[A]], metadata !"round.tonearest", metadata !"fpexcept.strict")
diff --git a/clang/test/CodeGen/AArch64/neon/fused-multiple-fullfp16-constrained.c b/clang/test/CodeGen/AArch64/neon/fused-multiple-fullfp16-constrained.c
index 665f4a8a6026d..a56cb34014f15 100644
--- a/clang/test/CodeGen/AArch64/neon/fused-multiple-fullfp16-constrained.c
+++ b/clang/test/CodeGen/AArch64/neon/fused-multiple-fullfp16-constrained.c
@@ -16,7 +16,7 @@
 // LLVM-LABEL: @test_vfma_f16(
 // CIR-LABEL: @vfma_f16(
 float16x4_t test_vfma_f16(float16x4_t a, float16x4_t b, float16x4_t c) {
-// CIR: cir.fma %{{.*}}, %{{.*}}, %{{.*}} : !cir.vector<4 x !cir.f16> {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
+// CIR: cir.fma %{{.*}}, %{{.*}}, %{{.*}} : !cir.vector<4 x !cir.f16> fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>
 
 // LLVM-SAME: <4 x half> {{.*}} [[A:%.*]], <4 x half> {{.*}} [[B:%.*]], <4 x half> {{.*}} [[C:%.*]]) {{.*}} {
 // LLVM: [[A_I:%.*]] = bitcast <4 x half> [[A]] to <4 x i16>
@@ -36,7 +36,7 @@ float16x4_t test_vfma_f16(float16x4_t a, float16x4_t b, float16x4_t c) {
 // LLVM-LABEL: @test_vfmaq_f16(
 // CIR-LABEL: @vfmaq_f16(
 float16x8_t test_vfmaq_f16(float16x8_t a, float16x8_t b, float16x8_t c) {
-// CIR: cir.fma %{{.*}}, %{{.*}}, %{{.*}} : !cir.vector<8 x !cir.f16> {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
+// CIR: cir.fma %{{.*}}, %{{.*}}, %{{.*}} : !cir.vector<8 x !cir.f16> fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>
 
 // LLVM-SAME: <8 x half> {{.*}} [[A:%.*]], <8 x half> {{.*}} [[B:%.*]], <8 x half> {{.*}} [[C:%.*]]) {{.*}} {
 // LLVM: [[A_I:%.*]] = bitcast <8 x half> [[A]] to <8 x i16>
@@ -57,7 +57,7 @@ float16x8_t test_vfmaq_f16(float16x8_t a, float16x8_t b, float16x8_t c) {
 float16x4_t test_vfma_lane_f16(float16x4_t a, float16x4_t b,
                                 float16x4_t c) {
 // CIR: [[LANE:%.*]] = cir.vec.shuffle(%{{.*}}, %{{.*}} : !cir.vector<4 x !cir.f16>) [#cir.int<3> : !s32i, #cir.int<3> : !s32i, #cir.int<3> : !s32i, #cir.int<3> : !s32i] : !cir.vector<4 x !cir.f16>
-// CIR: cir.fma %{{.*}}, [[LANE]], %{{.*}} : !cir.vector<4 x !cir.f16> {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
+// CIR: cir.fma %{{.*}}, [[LANE]], %{{.*}} : !cir.vector<4 x !cir.f16> fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>
 
 // LLVM-SAME: <4 x half> {{.*}} [[A:%.*]], <4 x half> {{.*}} [[B:%.*]], <4 x half> {{.*}} [[C:%.*]]) {{.*}} {
 // LLVM: [[A_I:%.*]] = bitcast <4 x half> [[A]] to <4 x i16>
@@ -79,7 +79,7 @@ float16x4_t test_vfma_lane_f16(float16x4_t a, float16x4_t b,
 float16x8_t test_vfmaq_lane_f16(float16x8_t a, float16x8_t b,
                                  float16x4_t c) {
 // CIR: [[LANE:%.*]] = cir.vec.shuffle(%{{.*}}, %{{.*}} : !cir.vector<4 x !cir.f16>) [#cir.int<3> : !s32i, #cir.int<3> : !s32i, #cir.int<3> : !s32i, #cir.int<3> : !s32i, #cir.int<3> : !s32i, #cir.int<3> : !s32i, #cir.int<3> : !s32i, #cir.int<3> : !s32i] : !cir.vector<8 x !cir.f16>
-// CIR: cir.fma %{{.*}}, [[LANE]], %{{.*}} : !cir.vector<8 x !cir.f16> {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
+// CIR: cir.fma %{{.*}}, [[LANE]], %{{.*}} : !cir.vector<8 x !cir.f16> fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>
 
 // LLVM-SAME: <8 x half> {{.*}} [[A:%.*]], <8 x half> {{.*}} [[B:%.*]], <4 x half> {{.*}} [[C:%.*]]) {{.*}} {
 // LLVM: [[A_I:%.*]] = bitcast <8 x half> [[A]] to <8 x i16>
@@ -101,7 +101,7 @@ float16x8_t test_vfmaq_lane_f16(float16x8_t a, float16x8_t b,
 float16x4_t test_vfma_laneq_f16(float16x4_t a, float16x4_t b,
                                  float16x8_t c) {
 // CIR: [[LANE:%.*]] = cir.vec.shuffle(%{{.*}}, %{{.*}} : !cir.vector<8 x !cir.f16>) [#cir.int<7> : !s32i, #cir.int<7> : !s32i, #cir.int<7> : !s32i, #cir.int<7> : !s32i] : !cir.vector<4 x !cir.f16>
-// CIR: cir.fma [[LANE]], %{{.*}}, %{{.*}} : !cir.vector<4 x !cir.f16> {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
+// CIR: cir.fma [[LANE]], %{{.*}}, %{{.*}} : !cir.vector<4 x !cir.f16> fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>
 
 // LLVM-SAME: <4 x half> {{.*}} [[A:%.*]], <4 x half> {{.*}} [[B:%.*]], <8 x half> {{.*}} [[C:%.*]]) {{.*}} {
 // LLVM: [[A_I:%.*]] = bitcast <4 x half> [[A]] to <4 x i16>
@@ -123,7 +123,7 @@ float16x4_t test_vfma_laneq_f16(float16x4_t a, float16x4_t b,
 float16x8_t test_vfmaq_laneq_f16(float16x8_t a, float16x8_t b,
                                   float16x8_t c) {
 // CIR: [[LANE:%.*]] = cir.vec.shuffle(%{{.*}}, %{{.*}} : !cir.vector<8 x !cir.f16>) [#cir.int<7> : !s32i, #cir.int<7> : !s32i, #cir.int<7> : !s32i, #cir.int<7> : !s32i, #cir.int<7> : !s32i, #cir.int<7> : !s32i, #cir.int<7> : !s32i, #cir.int<7> : !s32i] : !cir.vector<8 x !cir.f16>
-// CIR: cir.fma [[LANE]], %{{.*}}, %{{.*}} : !cir.vector<8 x !cir.f16> {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
+// CIR: cir.fma [[LANE]], %{{.*}}, %{{.*}} : !cir.vector<8 x !cir.f16> fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>
 
 // LLVM-SAME: <8 x half> {{.*}} [[A:%.*]], <8 x half> {{.*}} [[B:%.*]], <8 x half> {{.*}} [[C:%.*]]) {{.*}} {
 // LLVM: [[A_I:%.*]] = bitcast <8 x half> [[A]] to <8 x i16>
@@ -145,7 +145,7 @@ float16x8_t test_vfmaq_laneq_f16(float16x8_t a, float16x8_t b,
 float16_t test_vfmah_lane_f16(float16_t a, float16_t b, float16x4_t c) {
 // CIR: [[INDEX:%.*]] = cir.const #cir.int<3> : !u64i
 // CIR: [[LANE:%.*]] = cir.vec.extract %{{.*}}{{\[}}[[INDEX]] : !u64i] : !cir.vector<4 x !cir.f16>
-// CIR: cir.fma %{{.*}}, [[LANE]], %{{.*}} : !cir.f16 {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
+// CIR: cir.fma %{{.*}}, [[LANE]], %{{.*}} : !cir.f16 fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>
 
 // LLVM-SAME: half {{.*}} [[A:%.*]], half {{.*}} [[B:%.*]], <4 x half> {{.*}} [[C:%.*]]) {{.*}} {
 // LLVM: [[LANE:%.*]] = extractelement <4 x half> [[C]], i{{32|64}} 3
diff --git a/clang/test/CodeGen/AArch64/neon/fused-multiply-constrained.c b/clang/test/CodeGen/AArch64/neon/fused-multiply-constrained.c
index b06f799d7f17b..52ee3f2062851 100644
--- a/clang/test/CodeGen/AArch64/neon/fused-multiply-constrained.c
+++ b/clang/test/CodeGen/AArch64/neon/fused-multiply-constrained.c
@@ -9,7 +9,7 @@
 // LLVM-LABEL: @test_vfma_f64(
 // CIR-LABEL: @vfma_f64(
 float64x1_t test_vfma_f64(float64x1_t a, float64x1_t b, float64x1_t c) {
-// CIR: cir.fma %{{.*}}, %{{.*}}, %{{.*}} : !cir.vector<1 x !cir.double> {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
+// CIR: cir.fma %{{.*}}, %{{.*}}, %{{.*}} : !cir.vector<1 x !cir.double> fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>
 
 // LLVM-SAME: <1 x double> {{.*}} [[A:%.*]], <1 x double> {{.*}} [[B:%.*]], <1 x double> {{.*}} [[C:%.*]]) {{.*}} {
 // LLVM: [[A_I:%.*]] = bitcast <1 x double> [[A]] to i64
@@ -34,7 +34,7 @@ float64x1_t test_vfma_laneq_f64(float64x1_t a, float64x1_t b,
                                  float64x2_t v) {
 // CIR: [[INDEX:%.*]] = cir.const #cir.int<0> : !u64i
 // CIR: [[LANE:%.*]] = cir.vec.extract %{{.*}}{{\[}}[[INDEX]] : !u64i] : !cir.vector<2 x !cir.double>
-// CIR: cir.fma %{{.*}}, [[LANE]], %{{.*}} : !cir.double {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
+// CIR: cir.fma %{{.*}}, [[LANE]], %{{.*}} : !cir.double fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>
 
 // LLVM-SAME: <1 x double> {{.*}} [[A:%.*]], <1 x double> {{.*}} [[B:%.*]], <2 x double> {{.*}} [[V:%.*]]) {{.*}} {
 // LLVM: [[A_I:%.*]] = bitcast <1 x double> [[A]] to i64
@@ -59,7 +59,7 @@ float64x1_t test_vfma_laneq_f64(float64x1_t a, float64x1_t b,
 float64x2_t test_vfmaq_laneq_f64(float64x2_t a, float64x2_t b,
                                   float64x2_t v) {
 // CIR: [[LANE:%.*]] = cir.vec.shuffle(%{{.*}}, %{{.*}} : !cir.vector<2 x !cir.double>) [#cir.int<1> : !s32i, #cir.int<1> : !s32i] : !cir.vector<2 x !cir.double>
-// CIR: cir.fma [[LANE]], %{{.*}}, %{{.*}} : !cir.vector<2 x !cir.double> {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
+// CIR: cir.fma [[LANE]], %{{.*}}, %{{.*}} : !cir.vector<2 x !cir.double> fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>
 
 // LLVM-SAME: <2 x double> {{.*}} [[A:%.*]], <2 x double> {{.*}} [[B:%.*]], <2 x double> {{.*}} [[V:%.*]]) {{.*}} {
 // LLVM: [[A_I:%.*]] = bitcast <2 x double> [[A]] to <2 x i64>
@@ -81,7 +81,7 @@ float64x2_t test_vfmaq_laneq_f64(float64x2_t a, float64x2_t b,
 float32_t test_vfmas_lane_f32(float32_t a, float32_t b, float32x2_t c) {
 // CIR: [[INDEX:%.*]] = cir.const #cir.int<1> : !u64i
 // CIR: [[LANE:%.*]] = cir.vec.extract %{{.*}}{{\[}}[[INDEX]] : !u64i] : !cir.vector<2 x !cir.float>
-// CIR: cir.fma %{{.*}}, [[LANE]], %{{.*}} : !cir.float {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
+// CIR: cir.fma %{{.*}}, [[LANE]], %{{.*}} : !cir.float fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>
 
 // LLVM-SAME: float {{.*}} [[A:%.*]], float {{.*}} [[B:%.*]], <2 x float> {{.*}} [[C:%.*]]) {{.*}} {
 // LLVM: [[LANE:%.*]] = extractelement <2 x float> [[C]], i{{32|64}} 1
diff --git a/clang/test/CodeGen/AArch64/neon/intrinsics-constrained.c b/clang/test/CodeGen/AArch64/neon/intrinsics-constrained.c
index 79e637f2ea898..ef915a0ccea1e 100644
--- a/clang/test/CodeGen/AArch64/neon/intrinsics-constrained.c
+++ b/clang/test/CodeGen/AArch64/neon/intrinsics-constrained.c
@@ -16,7 +16,7 @@
 // LLVM-LABEL: @test_vsqrt_f16(
 // CIR-LABEL: @vsqrt_f16(
 float16x4_t test_vsqrt_f16(float16x4_t a) {
-// CIR: cir.sqrt %{{.*}} : !cir.vector<4 x !cir.f16> {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
+// CIR: cir.sqrt %{{.*}} : !cir.vector<4 x !cir.f16> fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>
 
 // LLVM-SAME: <4 x half> {{.*}} [[A:%.*]]) {{.*}} {
 // LLVM: [[A_I:%.*]] = bitcast <4 x half> [[A]] to <4 x i16>
@@ -30,7 +30,7 @@ float16x4_t test_vsqrt_f16(float16x4_t a) {
 // LLVM-LABEL: @test_vsqrtq_f16(
 // CIR-LABEL: @vsqrtq_f16(
 float16x8_t test_vsqrtq_f16(float16x8_t a) {
-// CIR: cir.sqrt %{{.*}} : !cir.vector<8 x !cir.f16> {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
+// CIR: cir.sqrt %{{.*}} : !cir.vector<8 x !cir.f16> fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>
 
 // LLVM-SAME: <8 x half> {{.*}} [[A:%.*]]) {{.*}} {
 // LLVM: [[A_I:%.*]] = bitcast <8 x half> [[A]] to <8 x i16>
@@ -44,7 +44,7 @@ float16x8_t test_vsqrtq_f16(float16x8_t a) {
 // LLVM-LABEL: @test_vsqrtq_f64(
 // CIR-LABEL: @vsqrtq_f64(
 float64x2_t test_vsqrtq_f64(float64x2_t a) {
-// CIR: cir.sqrt %{{.*}} : !cir.vector<2 x !cir.double> {fenv = #cir.fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>}
+// CIR: cir.sqrt %{{.*}} : !cir.vector<2 x !cir.double> fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true>
 
 // LLVM-SAME: <2 x double> {{.*}} [[A:%.*]]) {{.*}} {
 // LLVM: [[A_I:%.*]] = bitcast <2 x double> [[A]] to <2 x i64>



More information about the cfe-commits mailing list