https://github.com/yairbenavraham updated 
https://github.com/llvm/llvm-project/pull/218307

>From bf471dadf0140da8601ebbaadf387f0cfdcf19bf Mon Sep 17 00:00:00 2001
From: Yair Ben Avraham <[email protected]>
Date: Mon, 5 Oct 2026 13:22:22 +0300
Subject: [PATCH] [CIR][AArch64] Handle constrained Neon FMA and sqrt

Propagate expression FP options through AArch64 builtin emission and
attach the active FP environment to CIR fma and sqrt operations. This
allows strict FP operations to lower to constrained LLVM intrinsics
without changing unconstrained behavior.

Co-locate classic Clang, CIR, and CIR-to-LLVM coverage in focused
-constrained.c Neon tests. Move matching classic constrained checks from
legacy files and remove superseded blocks while retaining unrelated
legacy tests.

Track FMA operands through ABI conversions and lane selection, verify
constrained results, and test FP16 pragma overrides of command-line
maytrap settings.

Assisted-by: Codex
Follow-up to #213800
---
 .../lib/CIR/CodeGen/CIRGenBuiltinAArch64.cpp  |  21 +-
 .../AArch64/neon-intrinsics-constrained.c     |  40 ---
 .../CodeGen/AArch64/neon-misc-constrained.c   |  22 --
 .../neon-scalar-x-indexed-elem-constrained.c  |  67 +----
 .../AArch64/neon/fullfp16-constrained.c       |  33 +++
 .../fused-multiple-fullfp16-constrained.c     | 155 ++++++++++
 .../AArch64/neon/fused-multiply-constrained.c |  91 ++++++
 .../AArch64/neon/intrinsics-constrained.c     |  56 ++++
 .../v8.2a-fp16-intrinsics-constrained.c       |  16 -
 .../v8.2a-neon-intrinsics-constrained.c       | 280 +-----------------
 10 files changed, 359 insertions(+), 422 deletions(-)
 create mode 100644 clang/test/CodeGen/AArch64/neon/fullfp16-constrained.c
 create mode 100644 
clang/test/CodeGen/AArch64/neon/fused-multiple-fullfp16-constrained.c
 create mode 100644 clang/test/CodeGen/AArch64/neon/fused-multiply-constrained.c
 create mode 100644 clang/test/CodeGen/AArch64/neon/intrinsics-constrained.c

diff --git a/clang/lib/CIR/CodeGen/CIRGenBuiltinAArch64.cpp 
b/clang/lib/CIR/CodeGen/CIRGenBuiltinAArch64.cpp
index 70253221c7b7d5..af0c3c6b1c2d81 100644
--- a/clang/lib/CIR/CodeGen/CIRGenBuiltinAArch64.cpp
+++ b/clang/lib/CIR/CodeGen/CIRGenBuiltinAArch64.cpp
@@ -204,12 +204,26 @@ emitNeonCallToOp(CIRGenModule &cgm, CIRGenBuilderTy 
&builder,
                              builder.getStringAttr(intrinsicName.value()),
                              funcResTy, args)
         .getResult();
-  } else {
+  }
+  if constexpr (std::is_same_v<Operation, cir::FMAOp>) {
+    assert(args.size() == 3 && "fma expects three operands");
+    return Operation::create(builder, loc, funcResTy, args[0], args[1], 
args[2],
+                             builder.getConstrainedFPAttr())
+        .getResult();
+  }
+  if constexpr (std::is_same_v<Operation, cir::SqrtOp>) {
+    assert(args.size() == 1 && "sqrt expects one operand");
+    return Operation::create(builder, loc, funcResTy, args[0],
+                             builder.getConstrainedFPAttr())
+        .getResult();
+  }
+  if constexpr (!std::is_same_v<Operation, cir::LLVMIntrinsicCallOp> &&
+                !std::is_same_v<Operation, cir::FMAOp> &&
+                !std::is_same_v<Operation, cir::SqrtOp>)
     return Operation::create(
                builder, loc, mlir::TypeRange{funcResTy}, args,
                cir::getDefaultProperties<Operation>(builder.getContext()))
         .getResult();
-  }
 }
 
 // TODO(cir): Remove `cgm` from the list of arguments once all NYI(s) are gone.
@@ -2709,6 +2723,8 @@ CIRGenFunction::emitAArch64BuiltinExpr(unsigned 
builtinID, const CallExpr *expr,
   // evaluation.
   assert(!cir::MissingFeatures::msvcBuiltins());
 
+  CIRGenFPOptionsRAII fpOptsRAII(*this, expr);
+
   // Some intrinsics are equivalent - if they are use the base intrinsic ID.
   auto it = llvm::find_if(neonEquivalentIntrinsicMap, [builtinID](auto &p) {
     return p.first == builtinID;
@@ -3630,7 +3646,6 @@ CIRGenFunction::emitAArch64BuiltinExpr(unsigned 
builtinID, const CallExpr *expr,
   }
   case NEON::BI__builtin_neon_vsqrt_v:
   case NEON::BI__builtin_neon_vsqrtq_v:
-    assert(!cir::MissingFeatures::emitConstrainedFPCall());
     return emitNeonCallToOp<cir::SqrtOp>(cgm, builder, {ty}, ops, std::nullopt,
                                          ty, loc);
   case NEON::BI__builtin_neon_vrbit_v:
diff --git a/clang/test/CodeGen/AArch64/neon-intrinsics-constrained.c 
b/clang/test/CodeGen/AArch64/neon-intrinsics-constrained.c
index 50a2a629ea55b3..09f2aec4176b42 100644
--- a/clang/test/CodeGen/AArch64/neon-intrinsics-constrained.c
+++ b/clang/test/CodeGen/AArch64/neon-intrinsics-constrained.c
@@ -1418,46 +1418,6 @@ float64x1_t test_vmls_f64(float64x1_t a, float64x1_t b, 
float64x1_t c) {
   return vmls_f64(a, b, c);
 }
 
-// UNCONSTRAINED-LABEL: define dso_local <1 x double> @test_vfma_f64(
-// UNCONSTRAINED-SAME: <1 x double> noundef [[A:%.*]], <1 x double> noundef 
[[B:%.*]], <1 x double> noundef [[C:%.*]]) #[[ATTR0]] {
-// UNCONSTRAINED-NEXT:  [[ENTRY:.*:]]
-// UNCONSTRAINED-NEXT:    [[TMP0:%.*]] = bitcast <1 x double> [[A]] to i64
-// UNCONSTRAINED-NEXT:    [[__P0_ADDR_I_SROA_0_0_VEC_INSERT:%.*]] = 
insertelement <1 x i64> undef, i64 [[TMP0]], i64 0
-// UNCONSTRAINED-NEXT:    [[TMP1:%.*]] = bitcast <1 x double> [[B]] to i64
-// UNCONSTRAINED-NEXT:    [[__P1_ADDR_I_SROA_0_0_VEC_INSERT:%.*]] = 
insertelement <1 x i64> undef, i64 [[TMP1]], i64 0
-// UNCONSTRAINED-NEXT:    [[TMP2:%.*]] = bitcast <1 x double> [[C]] to i64
-// UNCONSTRAINED-NEXT:    [[__P2_ADDR_I_SROA_0_0_VEC_INSERT:%.*]] = 
insertelement <1 x i64> undef, i64 [[TMP2]], i64 0
-// UNCONSTRAINED-NEXT:    [[TMP3:%.*]] = bitcast <1 x i64> 
[[__P0_ADDR_I_SROA_0_0_VEC_INSERT]] to <8 x i8>
-// UNCONSTRAINED-NEXT:    [[TMP4:%.*]] = bitcast <1 x i64> 
[[__P1_ADDR_I_SROA_0_0_VEC_INSERT]] to <8 x i8>
-// UNCONSTRAINED-NEXT:    [[TMP5:%.*]] = bitcast <1 x i64> 
[[__P2_ADDR_I_SROA_0_0_VEC_INSERT]] to <8 x i8>
-// UNCONSTRAINED-NEXT:    [[TMP6:%.*]] = bitcast <8 x i8> [[TMP3]] to <1 x 
double>
-// UNCONSTRAINED-NEXT:    [[TMP7:%.*]] = bitcast <8 x i8> [[TMP4]] to <1 x 
double>
-// UNCONSTRAINED-NEXT:    [[TMP8:%.*]] = bitcast <8 x i8> [[TMP5]] to <1 x 
double>
-// UNCONSTRAINED-NEXT:    [[TMP9:%.*]] = call <1 x double> @llvm.fma.v1f64(<1 
x double> [[TMP7]], <1 x double> [[TMP8]], <1 x double> [[TMP6]])
-// UNCONSTRAINED-NEXT:    ret <1 x double> [[TMP9]]
-//
-// CONSTRAINED-LABEL: define dso_local <1 x double> @test_vfma_f64(
-// CONSTRAINED-SAME: <1 x double> noundef [[A:%.*]], <1 x double> noundef 
[[B:%.*]], <1 x double> noundef [[C:%.*]]) #[[ATTR0]] {
-// CONSTRAINED-NEXT:  [[ENTRY:.*:]]
-// CONSTRAINED-NEXT:    [[TMP0:%.*]] = bitcast <1 x double> [[A]] to i64
-// CONSTRAINED-NEXT:    [[__P0_ADDR_I_SROA_0_0_VEC_INSERT:%.*]] = 
insertelement <1 x i64> undef, i64 [[TMP0]], i64 0
-// CONSTRAINED-NEXT:    [[TMP1:%.*]] = bitcast <1 x double> [[B]] to i64
-// CONSTRAINED-NEXT:    [[__P1_ADDR_I_SROA_0_0_VEC_INSERT:%.*]] = 
insertelement <1 x i64> undef, i64 [[TMP1]], i64 0
-// CONSTRAINED-NEXT:    [[TMP2:%.*]] = bitcast <1 x double> [[C]] to i64
-// CONSTRAINED-NEXT:    [[__P2_ADDR_I_SROA_0_0_VEC_INSERT:%.*]] = 
insertelement <1 x i64> undef, i64 [[TMP2]], i64 0
-// CONSTRAINED-NEXT:    [[TMP3:%.*]] = bitcast <1 x i64> 
[[__P0_ADDR_I_SROA_0_0_VEC_INSERT]] to <8 x i8>
-// CONSTRAINED-NEXT:    [[TMP4:%.*]] = bitcast <1 x i64> 
[[__P1_ADDR_I_SROA_0_0_VEC_INSERT]] to <8 x i8>
-// CONSTRAINED-NEXT:    [[TMP5:%.*]] = bitcast <1 x i64> 
[[__P2_ADDR_I_SROA_0_0_VEC_INSERT]] to <8 x i8>
-// CONSTRAINED-NEXT:    [[TMP6:%.*]] = bitcast <8 x i8> [[TMP3]] to <1 x 
double>
-// CONSTRAINED-NEXT:    [[TMP7:%.*]] = bitcast <8 x i8> [[TMP4]] to <1 x 
double>
-// CONSTRAINED-NEXT:    [[TMP8:%.*]] = bitcast <8 x i8> [[TMP5]] to <1 x 
double>
-// CONSTRAINED-NEXT:    [[TMP9:%.*]] = call <1 x double> 
@llvm.experimental.constrained.fma.v1f64(<1 x double> [[TMP7]], <1 x double> 
[[TMP8]], <1 x double> [[TMP6]], metadata !"round.tonearest", metadata 
!"fpexcept.strict") #[[ATTR3]]
-// CONSTRAINED-NEXT:    ret <1 x double> [[TMP9]]
-//
-float64x1_t test_vfma_f64(float64x1_t a, float64x1_t b, float64x1_t c) {
-  return vfma_f64(a, b, c);
-}
-
 // UNCONSTRAINED-LABEL: define dso_local <1 x double> @test_vfms_f64(
 // UNCONSTRAINED-SAME: <1 x double> noundef [[A:%.*]], <1 x double> noundef 
[[B:%.*]], <1 x double> noundef [[C:%.*]]) #[[ATTR0]] {
 // UNCONSTRAINED-NEXT:  [[ENTRY:.*:]]
diff --git a/clang/test/CodeGen/AArch64/neon-misc-constrained.c 
b/clang/test/CodeGen/AArch64/neon-misc-constrained.c
index 49208892e3035b..f1f638cb272c62 100644
--- a/clang/test/CodeGen/AArch64/neon-misc-constrained.c
+++ b/clang/test/CodeGen/AArch64/neon-misc-constrained.c
@@ -82,28 +82,6 @@ float32x4_t test_vsqrtq_f32(float32x4_t a) {
 }
 
 
-// UNCONSTRAINED-LABEL: define dso_local <2 x double> @test_vsqrtq_f64(
-// UNCONSTRAINED-SAME: <2 x double> noundef [[A:%.*]]) #[[ATTR0]] {
-// UNCONSTRAINED-NEXT:  [[ENTRY:.*:]]
-// UNCONSTRAINED-NEXT:    [[TMP0:%.*]] = bitcast <2 x double> [[A]] to <2 x 
i64>
-// UNCONSTRAINED-NEXT:    [[TMP1:%.*]] = bitcast <2 x i64> [[TMP0]] to <16 x 
i8>
-// UNCONSTRAINED-NEXT:    [[TMP2:%.*]] = bitcast <16 x i8> [[TMP1]] to <2 x 
double>
-// UNCONSTRAINED-NEXT:    [[VSQRT_I:%.*]] = call <2 x double> 
@llvm.sqrt.v2f64(<2 x double> [[TMP2]])
-// UNCONSTRAINED-NEXT:    ret <2 x double> [[VSQRT_I]]
-//
-// CONSTRAINED-LABEL: define dso_local <2 x double> @test_vsqrtq_f64(
-// CONSTRAINED-SAME: <2 x double> noundef [[A:%.*]]) #[[ATTR0]] {
-// CONSTRAINED-NEXT:  [[ENTRY:.*:]]
-// CONSTRAINED-NEXT:    [[TMP0:%.*]] = bitcast <2 x double> [[A]] to <2 x i64>
-// CONSTRAINED-NEXT:    [[TMP1:%.*]] = bitcast <2 x i64> [[TMP0]] to <16 x i8>
-// CONSTRAINED-NEXT:    [[TMP2:%.*]] = bitcast <16 x i8> [[TMP1]] to <2 x 
double>
-// CONSTRAINED-NEXT:    [[VSQRT_I:%.*]] = call <2 x double> 
@llvm.experimental.constrained.sqrt.v2f64(<2 x double> [[TMP2]], metadata 
!"round.tonearest", metadata !"fpexcept.strict") #[[ATTR2]]
-// CONSTRAINED-NEXT:    ret <2 x double> [[VSQRT_I]]
-//
-float64x2_t test_vsqrtq_f64(float64x2_t a) {
-  return vsqrtq_f64(a);
-}
-
 // UNCONSTRAINED-LABEL: define dso_local <4 x half> @test_vcvt_f16_f32(
 // UNCONSTRAINED-SAME: <4 x float> noundef [[A:%.*]]) #[[ATTR0]] {
 // UNCONSTRAINED-NEXT:  [[ENTRY:.*:]]
diff --git 
a/clang/test/CodeGen/AArch64/neon-scalar-x-indexed-elem-constrained.c 
b/clang/test/CodeGen/AArch64/neon-scalar-x-indexed-elem-constrained.c
index 944929ccb5f428..046bca2420a37a 100644
--- a/clang/test/CodeGen/AArch64/neon-scalar-x-indexed-elem-constrained.c
+++ b/clang/test/CodeGen/AArch64/neon-scalar-x-indexed-elem-constrained.c
@@ -13,36 +13,18 @@
 
 #include <arm_neon.h>
 
-// UNCONSTRAINED-LABEL: define dso_local float @test_vfmas_lane_f32(
-// UNCONSTRAINED-SAME: float noundef [[A:%.*]], float noundef [[B:%.*]], <2 x 
float> noundef [[C:%.*]]) #[[ATTR0:[0-9]+]] {
-// UNCONSTRAINED-NEXT:  [[ENTRY:.*:]]
-// UNCONSTRAINED-NEXT:    [[EXTRACT:%.*]] = extractelement <2 x float> [[C]], 
i32 1
-// UNCONSTRAINED-NEXT:    [[TMP0:%.*]] = call float @llvm.fma.f32(float [[B]], 
float [[EXTRACT]], float [[A]])
-// UNCONSTRAINED-NEXT:    ret float [[TMP0]]
-//
-// CONSTRAINED-LABEL: define dso_local float @test_vfmas_lane_f32(
-// CONSTRAINED-SAME: float noundef [[A:%.*]], float noundef [[B:%.*]], <2 x 
float> noundef [[C:%.*]]) #[[ATTR0:[0-9]+]] {
-// CONSTRAINED-NEXT:  [[ENTRY:.*:]]
-// CONSTRAINED-NEXT:    [[EXTRACT:%.*]] = extractelement <2 x float> [[C]], 
i32 1
-// CONSTRAINED-NEXT:    [[TMP0:%.*]] = call float 
@llvm.experimental.constrained.fma.f32(float [[B]], float [[EXTRACT]], float 
[[A]], metadata !"round.tonearest", metadata !"fpexcept.strict") 
#[[ATTR2:[0-9]+]]
-// CONSTRAINED-NEXT:    ret float [[TMP0]]
-//
-float32_t test_vfmas_lane_f32(float32_t a, float32_t b, float32x2_t c) {
-  return vfmas_lane_f32(a, b, c, 1);
-}
-
 // UNCONSTRAINED-LABEL: define dso_local double @test_vfmad_lane_f64(
-// UNCONSTRAINED-SAME: double noundef [[A:%.*]], double noundef [[B:%.*]], <1 
x double> noundef [[C:%.*]]) #[[ATTR0]] {
+// UNCONSTRAINED-SAME: double noundef [[A:%.*]], double noundef [[B:%.*]], <1 
x double> noundef [[C:%.*]]) #[[ATTR0:[0-9]+]] {
 // UNCONSTRAINED-NEXT:  [[ENTRY:.*:]]
 // UNCONSTRAINED-NEXT:    [[EXTRACT:%.*]] = extractelement <1 x double> [[C]], 
i32 0
 // UNCONSTRAINED-NEXT:    [[TMP0:%.*]] = call double @llvm.fma.f64(double 
[[B]], double [[EXTRACT]], double [[A]])
 // UNCONSTRAINED-NEXT:    ret double [[TMP0]]
 //
 // CONSTRAINED-LABEL: define dso_local double @test_vfmad_lane_f64(
-// CONSTRAINED-SAME: double noundef [[A:%.*]], double noundef [[B:%.*]], <1 x 
double> noundef [[C:%.*]]) #[[ATTR0]] {
+// CONSTRAINED-SAME: double noundef [[A:%.*]], double noundef [[B:%.*]], <1 x 
double> noundef [[C:%.*]]) #[[ATTR0:[0-9]+]] {
 // CONSTRAINED-NEXT:  [[ENTRY:.*:]]
 // CONSTRAINED-NEXT:    [[EXTRACT:%.*]] = extractelement <1 x double> [[C]], 
i32 0
-// CONSTRAINED-NEXT:    [[TMP0:%.*]] = call double 
@llvm.experimental.constrained.fma.f64(double [[B]], double [[EXTRACT]], double 
[[A]], metadata !"round.tonearest", metadata !"fpexcept.strict") #[[ATTR2]]
+// CONSTRAINED-NEXT:    [[TMP0:%.*]] = call double 
@llvm.experimental.constrained.fma.f64(double [[B]], double [[EXTRACT]], double 
[[A]], metadata !"round.tonearest", metadata !"fpexcept.strict") 
#[[ATTR2:[0-9]+]]
 // CONSTRAINED-NEXT:    ret double [[TMP0]]
 //
 float64_t test_vfmad_lane_f64(float64_t a, float64_t b, float64x1_t c) {
@@ -173,48 +155,6 @@ float64x1_t test_vfms_lane_f64(float64x1_t a, float64x1_t 
b, float64x1_t v) {
   return vfms_lane_f64(a, b, v, 0);
 }
 
-// UNCONSTRAINED-LABEL: define dso_local <1 x double> @test_vfma_laneq_f64(
-// UNCONSTRAINED-SAME: <1 x double> noundef [[A:%.*]], <1 x double> noundef 
[[B:%.*]], <2 x double> noundef [[V:%.*]]) #[[ATTR0]] {
-// UNCONSTRAINED-NEXT:  [[ENTRY:.*:]]
-// UNCONSTRAINED-NEXT:    [[TMP0:%.*]] = bitcast <1 x double> [[A]] to i64
-// UNCONSTRAINED-NEXT:    [[__S0_SROA_0_0_VEC_INSERT:%.*]] = insertelement <1 
x i64> undef, i64 [[TMP0]], i64 0
-// UNCONSTRAINED-NEXT:    [[TMP1:%.*]] = bitcast <1 x double> [[B]] to i64
-// UNCONSTRAINED-NEXT:    [[__S1_SROA_0_0_VEC_INSERT:%.*]] = insertelement <1 
x i64> undef, i64 [[TMP1]], i64 0
-// UNCONSTRAINED-NEXT:    [[TMP2:%.*]] = bitcast <2 x double> [[V]] to <2 x 
i64>
-// UNCONSTRAINED-NEXT:    [[TMP3:%.*]] = bitcast <1 x i64> 
[[__S0_SROA_0_0_VEC_INSERT]] to <8 x i8>
-// UNCONSTRAINED-NEXT:    [[TMP4:%.*]] = bitcast <1 x i64> 
[[__S1_SROA_0_0_VEC_INSERT]] to <8 x i8>
-// UNCONSTRAINED-NEXT:    [[TMP5:%.*]] = bitcast <2 x i64> [[TMP2]] to <16 x 
i8>
-// UNCONSTRAINED-NEXT:    [[TMP6:%.*]] = bitcast <8 x i8> [[TMP3]] to double
-// UNCONSTRAINED-NEXT:    [[TMP7:%.*]] = bitcast <8 x i8> [[TMP4]] to double
-// UNCONSTRAINED-NEXT:    [[TMP8:%.*]] = bitcast <16 x i8> [[TMP5]] to <2 x 
double>
-// UNCONSTRAINED-NEXT:    [[EXTRACT:%.*]] = extractelement <2 x double> 
[[TMP8]], i32 0
-// UNCONSTRAINED-NEXT:    [[TMP9:%.*]] = call double @llvm.fma.f64(double 
[[TMP7]], double [[EXTRACT]], double [[TMP6]])
-// UNCONSTRAINED-NEXT:    [[TMP10:%.*]] = bitcast double [[TMP9]] to <1 x 
double>
-// UNCONSTRAINED-NEXT:    ret <1 x double> [[TMP10]]
-//
-// CONSTRAINED-LABEL: define dso_local <1 x double> @test_vfma_laneq_f64(
-// CONSTRAINED-SAME: <1 x double> noundef [[A:%.*]], <1 x double> noundef 
[[B:%.*]], <2 x double> noundef [[V:%.*]]) #[[ATTR0]] {
-// CONSTRAINED-NEXT:  [[ENTRY:.*:]]
-// CONSTRAINED-NEXT:    [[TMP0:%.*]] = bitcast <1 x double> [[A]] to i64
-// CONSTRAINED-NEXT:    [[__S0_SROA_0_0_VEC_INSERT:%.*]] = insertelement <1 x 
i64> undef, i64 [[TMP0]], i64 0
-// CONSTRAINED-NEXT:    [[TMP1:%.*]] = bitcast <1 x double> [[B]] to i64
-// CONSTRAINED-NEXT:    [[__S1_SROA_0_0_VEC_INSERT:%.*]] = insertelement <1 x 
i64> undef, i64 [[TMP1]], i64 0
-// CONSTRAINED-NEXT:    [[TMP2:%.*]] = bitcast <2 x double> [[V]] to <2 x i64>
-// CONSTRAINED-NEXT:    [[TMP3:%.*]] = bitcast <1 x i64> 
[[__S0_SROA_0_0_VEC_INSERT]] to <8 x i8>
-// CONSTRAINED-NEXT:    [[TMP4:%.*]] = bitcast <1 x i64> 
[[__S1_SROA_0_0_VEC_INSERT]] to <8 x i8>
-// CONSTRAINED-NEXT:    [[TMP5:%.*]] = bitcast <2 x i64> [[TMP2]] to <16 x i8>
-// CONSTRAINED-NEXT:    [[TMP6:%.*]] = bitcast <8 x i8> [[TMP3]] to double
-// CONSTRAINED-NEXT:    [[TMP7:%.*]] = bitcast <8 x i8> [[TMP4]] to double
-// CONSTRAINED-NEXT:    [[TMP8:%.*]] = bitcast <16 x i8> [[TMP5]] to <2 x 
double>
-// CONSTRAINED-NEXT:    [[EXTRACT:%.*]] = extractelement <2 x double> 
[[TMP8]], i32 0
-// CONSTRAINED-NEXT:    [[TMP9:%.*]] = call double 
@llvm.experimental.constrained.fma.f64(double [[TMP7]], double [[EXTRACT]], 
double [[TMP6]], metadata !"round.tonearest", metadata !"fpexcept.strict") 
#[[ATTR2]]
-// CONSTRAINED-NEXT:    [[TMP10:%.*]] = bitcast double [[TMP9]] to <1 x double>
-// CONSTRAINED-NEXT:    ret <1 x double> [[TMP10]]
-//
-float64x1_t test_vfma_laneq_f64(float64x1_t a, float64x1_t b, float64x2_t v) {
-  return vfma_laneq_f64(a, b, v, 0);
-}
-
 // UNCONSTRAINED-LABEL: define dso_local <1 x double> @test_vfms_laneq_f64(
 // UNCONSTRAINED-SAME: <1 x double> noundef [[A:%.*]], <1 x double> noundef 
[[B:%.*]], <2 x double> noundef [[V:%.*]]) #[[ATTR0]] {
 // UNCONSTRAINED-NEXT:  [[ENTRY:.*:]]
@@ -258,4 +198,3 @@ float64x1_t test_vfma_laneq_f64(float64x1_t a, float64x1_t 
b, float64x2_t v) {
 float64x1_t test_vfms_laneq_f64(float64x1_t a, float64x1_t b, float64x2_t v) {
   return vfms_laneq_f64(a, b, v, 0);
 }
-
diff --git a/clang/test/CodeGen/AArch64/neon/fullfp16-constrained.c 
b/clang/test/CodeGen/AArch64/neon/fullfp16-constrained.c
new file mode 100644
index 00000000000000..1a668bd657e0de
--- /dev/null
+++ b/clang/test/CodeGen/AArch64/neon/fullfp16-constrained.c
@@ -0,0 +1,33 @@
+// REQUIRES: aarch64-registered-target
+
+// RUN: %clang_cc1_cg_arm64_neon -target-feature +fullfp16 
-ffp-exception-behavior=strict -emit-llvm %s -disable-O0-optnone | opt -S 
-passes=mem2reg,simplifycfg | FileCheck %s --check-prefixes=ALL,LLVM 
--implicit-check-not=' @llvm.fma.' --implicit-check-not=' @llvm.sqrt.'
+// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 
-fexperimental-strict-floating-point -ffp-exception-behavior=maytrap -DEXCEPT=1 
-fclangir -emit-llvm %s -disable-O0-optnone | opt -S 
-passes=mem2reg,simplifycfg | FileCheck %s --check-prefixes=ALL,LLVM 
--implicit-check-not=fpexcept.maytrap --implicit-check-not=' @llvm.fma.' 
--implicit-check-not=' @llvm.sqrt.' %}
+// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 
-fexperimental-strict-floating-point -ffp-exception-behavior=maytrap -DEXCEPT=1 
-fclangir -emit-cir  %s -disable-O0-optnone |                                   
   FileCheck %s --check-prefixes=ALL,CIR --implicit-check-not='except_mode = 
maytrap' --implicit-check-not='cir.call_llvm_intrinsic "fma"' 
--implicit-check-not='cir.call_llvm_intrinsic "sqrt"' %}
+// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 
-fexperimental-strict-floating-point -ffp-exception-behavior=strict             
-fclangir -emit-llvm %s -disable-O0-optnone | opt -S 
-passes=mem2reg,simplifycfg | FileCheck %s --check-prefixes=ALL,LLVM 
--implicit-check-not=' @llvm.fma.' --implicit-check-not=' @llvm.sqrt.' %}
+// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 
-fexperimental-strict-floating-point -ffp-exception-behavior=strict             
-fclangir -emit-cir  %s -disable-O0-optnone |                                   
   FileCheck %s --check-prefixes=ALL,CIR 
--implicit-check-not='cir.call_llvm_intrinsic "fma"' 
--implicit-check-not='cir.call_llvm_intrinsic "sqrt"' %}
+
+#if EXCEPT
+#pragma float_control(except, on)
+#endif
+
+#include <arm_fp16.h>
+
+// ALL-LABEL: @test_vsqrth_f16(
+float16_t test_vsqrth_f16(float16_t a) {
+// CIR: cir.sqrt %{{.*}} : !cir.f16 fenv<dynamic_rounding_mode = tonearest, 
except_mode = unknown, strict_except = true>
+
+// LLVM-SAME: half {{.*}} [[A:%.*]]) {{.*}} {
+// LLVM: [[SQRT:%.*]] = call half @llvm.experimental.constrained.sqrt.f16(half 
[[A]], metadata !"round.tonearest", metadata !"fpexcept.strict")
+// LLVM: ret half [[SQRT]]
+  return vsqrth_f16(a);
+}
+
+// ALL-LABEL: @test_vfmah_f16(
+float16_t test_vfmah_f16(float16_t a, float16_t b, float16_t c) {
+// CIR: cir.fma %{{.*}}, %{{.*}}, %{{.*}} : !cir.f16 
fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = 
true>
+
+// LLVM-SAME: half {{.*}} [[A:%.*]], half {{.*}} [[B:%.*]], half {{.*}} 
[[C:%.*]]) {{.*}} {
+// LLVM: [[FMA:%.*]] = call half @llvm.experimental.constrained.fma.f16(half 
[[B]], half [[C]], half [[A]], metadata !"round.tonearest", metadata 
!"fpexcept.strict")
+// LLVM: ret half [[FMA]]
+  return vfmah_f16(a, b, c);
+}
diff --git 
a/clang/test/CodeGen/AArch64/neon/fused-multiple-fullfp16-constrained.c 
b/clang/test/CodeGen/AArch64/neon/fused-multiple-fullfp16-constrained.c
new file mode 100644
index 00000000000000..a56cb34014f15e
--- /dev/null
+++ b/clang/test/CodeGen/AArch64/neon/fused-multiple-fullfp16-constrained.c
@@ -0,0 +1,155 @@
+// REQUIRES: aarch64-registered-target
+
+// RUN: %clang_cc1_cg_arm64_neon -target-feature +fullfp16 
-ffp-exception-behavior=maytrap -DEXCEPT=1 -emit-llvm %s -disable-O0-optnone | 
opt -S -passes=mem2reg,sroa | FileCheck %s --check-prefixes=ALL,LLVM 
--implicit-check-not=fpexcept.maytrap --implicit-check-not=' @llvm.fma.'
+// RUN: %clang_cc1_cg_arm64_neon -target-feature +fullfp16 
-ffp-exception-behavior=strict -emit-llvm %s -disable-O0-optnone | opt -S 
-passes=mem2reg,sroa | FileCheck %s --check-prefixes=ALL,LLVM 
--implicit-check-not=' @llvm.fma.'
+// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 
-fexperimental-strict-floating-point -ffp-exception-behavior=maytrap -DEXCEPT=1 
-fclangir -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,sroa | 
FileCheck %s --check-prefixes=ALL,LLVM --implicit-check-not=fpexcept.maytrap 
--implicit-check-not=' @llvm.fma.' %}
+// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 
-fexperimental-strict-floating-point -ffp-exception-behavior=maytrap -DEXCEPT=1 
-fclangir -emit-cir  %s -disable-O0-optnone |                               
FileCheck %s --check-prefixes=ALL,CIR --implicit-check-not='except_mode = 
maytrap' --implicit-check-not='cir.call_llvm_intrinsic "fma"' %}
+// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 
-fexperimental-strict-floating-point -ffp-exception-behavior=strict             
-fclangir -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,sroa | 
FileCheck %s --check-prefixes=ALL,LLVM --implicit-check-not=' @llvm.fma.' %}
+// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 
-fexperimental-strict-floating-point -ffp-exception-behavior=strict             
-fclangir -emit-cir  %s -disable-O0-optnone |                               
FileCheck %s --check-prefixes=ALL,CIR 
--implicit-check-not='cir.call_llvm_intrinsic "fma"' %}
+
+#if EXCEPT
+#pragma float_control(except, on)
+#endif
+
+#include <arm_neon.h>
+
+// LLVM-LABEL: @test_vfma_f16(
+// CIR-LABEL: @vfma_f16(
+float16x4_t test_vfma_f16(float16x4_t a, float16x4_t b, float16x4_t c) {
+// CIR: cir.fma %{{.*}}, %{{.*}}, %{{.*}} : !cir.vector<4 x !cir.f16> 
fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = 
true>
+
+// LLVM-SAME: <4 x half> {{.*}} [[A:%.*]], <4 x half> {{.*}} [[B:%.*]], <4 x 
half> {{.*}} [[C:%.*]]) {{.*}} {
+// LLVM: [[A_I:%.*]] = bitcast <4 x half> [[A]] to <4 x i16>
+// LLVM: [[B_I:%.*]] = bitcast <4 x half> [[B]] to <4 x i16>
+// LLVM: [[C_I:%.*]] = bitcast <4 x half> [[C]] to <4 x i16>
+// LLVM: [[A_BYTES:%.*]] = bitcast <4 x i16> [[A_I]] to <8 x i8>
+// LLVM: [[B_BYTES:%.*]] = bitcast <4 x i16> [[B_I]] to <8 x i8>
+// LLVM: [[C_BYTES:%.*]] = bitcast <4 x i16> [[C_I]] to <8 x i8>
+// LLVM: [[A_CAST:%.*]] = bitcast <8 x i8> [[A_BYTES]] to <4 x half>
+// LLVM: [[B_CAST:%.*]] = bitcast <8 x i8> [[B_BYTES]] to <4 x half>
+// LLVM: [[C_CAST:%.*]] = bitcast <8 x i8> [[C_BYTES]] to <4 x half>
+// LLVM: [[FMA:%.*]] = call <4 x half> 
@llvm.experimental.constrained.fma.v4f16(<4 x half> [[B_CAST]], <4 x half> 
[[C_CAST]], <4 x half> [[A_CAST]], metadata !"round.tonearest", metadata 
!"fpexcept.strict")
+// LLVM: ret <4 x half> [[FMA]]
+  return vfma_f16(a, b, c);
+}
+
+// LLVM-LABEL: @test_vfmaq_f16(
+// CIR-LABEL: @vfmaq_f16(
+float16x8_t test_vfmaq_f16(float16x8_t a, float16x8_t b, float16x8_t c) {
+// CIR: cir.fma %{{.*}}, %{{.*}}, %{{.*}} : !cir.vector<8 x !cir.f16> 
fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = 
true>
+
+// LLVM-SAME: <8 x half> {{.*}} [[A:%.*]], <8 x half> {{.*}} [[B:%.*]], <8 x 
half> {{.*}} [[C:%.*]]) {{.*}} {
+// LLVM: [[A_I:%.*]] = bitcast <8 x half> [[A]] to <8 x i16>
+// LLVM: [[B_I:%.*]] = bitcast <8 x half> [[B]] to <8 x i16>
+// LLVM: [[C_I:%.*]] = bitcast <8 x half> [[C]] to <8 x i16>
+// LLVM: [[A_BYTES:%.*]] = bitcast <8 x i16> [[A_I]] to <16 x i8>
+// LLVM: [[B_BYTES:%.*]] = bitcast <8 x i16> [[B_I]] to <16 x i8>
+// LLVM: [[C_BYTES:%.*]] = bitcast <8 x i16> [[C_I]] to <16 x i8>
+// LLVM: [[A_CAST:%.*]] = bitcast <16 x i8> [[A_BYTES]] to <8 x half>
+// LLVM: [[B_CAST:%.*]] = bitcast <16 x i8> [[B_BYTES]] to <8 x half>
+// LLVM: [[C_CAST:%.*]] = bitcast <16 x i8> [[C_BYTES]] to <8 x half>
+// LLVM: [[FMA:%.*]] = call <8 x half> 
@llvm.experimental.constrained.fma.v8f16(<8 x half> [[B_CAST]], <8 x half> 
[[C_CAST]], <8 x half> [[A_CAST]], metadata !"round.tonearest", metadata 
!"fpexcept.strict")
+// LLVM: ret <8 x half> [[FMA]]
+  return vfmaq_f16(a, b, c);
+}
+
+// ALL-LABEL: @test_vfma_lane_f16(
+float16x4_t test_vfma_lane_f16(float16x4_t a, float16x4_t b,
+                                float16x4_t c) {
+// CIR: [[LANE:%.*]] = cir.vec.shuffle(%{{.*}}, %{{.*}} : !cir.vector<4 x 
!cir.f16>) [#cir.int<3> : !s32i, #cir.int<3> : !s32i, #cir.int<3> : !s32i, 
#cir.int<3> : !s32i] : !cir.vector<4 x !cir.f16>
+// CIR: cir.fma %{{.*}}, [[LANE]], %{{.*}} : !cir.vector<4 x !cir.f16> 
fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = 
true>
+
+// LLVM-SAME: <4 x half> {{.*}} [[A:%.*]], <4 x half> {{.*}} [[B:%.*]], <4 x 
half> {{.*}} [[C:%.*]]) {{.*}} {
+// LLVM: [[A_I:%.*]] = bitcast <4 x half> [[A]] to <4 x i16>
+// LLVM: [[B_I:%.*]] = bitcast <4 x half> [[B]] to <4 x i16>
+// LLVM: [[C_I:%.*]] = bitcast <4 x half> [[C]] to <4 x i16>
+// LLVM: [[A_BYTES:%.*]] = bitcast <4 x i16> [[A_I]] to <8 x i8>
+// LLVM: [[B_BYTES:%.*]] = bitcast <4 x i16> [[B_I]] to <8 x i8>
+// LLVM: [[C_BYTES:%.*]] = bitcast <4 x i16> [[C_I]] to <8 x i8>
+// LLVM-DAG: [[A_CAST:%.*]] = bitcast <8 x i8> [[A_BYTES]] to <4 x half>
+// LLVM-DAG: [[B_CAST:%.*]] = bitcast <8 x i8> [[B_BYTES]] to <4 x half>
+// LLVM-DAG: [[C_CAST:%.*]] = bitcast <8 x i8> [[C_BYTES]] to <4 x half>
+// LLVM-DAG: [[LANE:%.*]] = shufflevector <4 x half> [[C_CAST]], <4 x half> 
{{.*}}, <4 x i32> <i32 3, i32 3, i32 3, i32 3>
+// LLVM: [[FMA:%.*]] = call <4 x half> 
@llvm.experimental.constrained.fma.v4f16(<4 x half> [[B_CAST]], <4 x half> 
[[LANE]], <4 x half> [[A_CAST]], metadata !"round.tonearest", metadata 
!"fpexcept.strict")
+// LLVM: ret <4 x half> [[FMA]]
+  return vfma_lane_f16(a, b, c, 3);
+}
+
+// ALL-LABEL: @test_vfmaq_lane_f16(
+float16x8_t test_vfmaq_lane_f16(float16x8_t a, float16x8_t b,
+                                 float16x4_t c) {
+// CIR: [[LANE:%.*]] = cir.vec.shuffle(%{{.*}}, %{{.*}} : !cir.vector<4 x 
!cir.f16>) [#cir.int<3> : !s32i, #cir.int<3> : !s32i, #cir.int<3> : !s32i, 
#cir.int<3> : !s32i, #cir.int<3> : !s32i, #cir.int<3> : !s32i, #cir.int<3> : 
!s32i, #cir.int<3> : !s32i] : !cir.vector<8 x !cir.f16>
+// CIR: cir.fma %{{.*}}, [[LANE]], %{{.*}} : !cir.vector<8 x !cir.f16> 
fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = 
true>
+
+// LLVM-SAME: <8 x half> {{.*}} [[A:%.*]], <8 x half> {{.*}} [[B:%.*]], <4 x 
half> {{.*}} [[C:%.*]]) {{.*}} {
+// LLVM: [[A_I:%.*]] = bitcast <8 x half> [[A]] to <8 x i16>
+// LLVM: [[B_I:%.*]] = bitcast <8 x half> [[B]] to <8 x i16>
+// LLVM: [[C_I:%.*]] = bitcast <4 x half> [[C]] to <4 x i16>
+// LLVM: [[A_BYTES:%.*]] = bitcast <8 x i16> [[A_I]] to <16 x i8>
+// LLVM: [[B_BYTES:%.*]] = bitcast <8 x i16> [[B_I]] to <16 x i8>
+// LLVM: [[C_BYTES:%.*]] = bitcast <4 x i16> [[C_I]] to <8 x i8>
+// LLVM-DAG: [[A_CAST:%.*]] = bitcast <16 x i8> [[A_BYTES]] to <8 x half>
+// LLVM-DAG: [[B_CAST:%.*]] = bitcast <16 x i8> [[B_BYTES]] to <8 x half>
+// LLVM-DAG: [[C_CAST:%.*]] = bitcast <8 x i8> [[C_BYTES]] to <4 x half>
+// LLVM-DAG: [[LANE:%.*]] = shufflevector <4 x half> [[C_CAST]], <4 x half> 
{{.*}}, <8 x i32> <i32 3, i32 3, i32 3, i32 3, i32 3, i32 3, i32 3, i32 3>
+// LLVM: [[FMA:%.*]] = call <8 x half> 
@llvm.experimental.constrained.fma.v8f16(<8 x half> [[B_CAST]], <8 x half> 
[[LANE]], <8 x half> [[A_CAST]], metadata !"round.tonearest", metadata 
!"fpexcept.strict")
+// LLVM: ret <8 x half> [[FMA]]
+  return vfmaq_lane_f16(a, b, c, 3);
+}
+
+// ALL-LABEL: @test_vfma_laneq_f16(
+float16x4_t test_vfma_laneq_f16(float16x4_t a, float16x4_t b,
+                                 float16x8_t c) {
+// CIR: [[LANE:%.*]] = cir.vec.shuffle(%{{.*}}, %{{.*}} : !cir.vector<8 x 
!cir.f16>) [#cir.int<7> : !s32i, #cir.int<7> : !s32i, #cir.int<7> : !s32i, 
#cir.int<7> : !s32i] : !cir.vector<4 x !cir.f16>
+// CIR: cir.fma [[LANE]], %{{.*}}, %{{.*}} : !cir.vector<4 x !cir.f16> 
fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = 
true>
+
+// LLVM-SAME: <4 x half> {{.*}} [[A:%.*]], <4 x half> {{.*}} [[B:%.*]], <8 x 
half> {{.*}} [[C:%.*]]) {{.*}} {
+// LLVM: [[A_I:%.*]] = bitcast <4 x half> [[A]] to <4 x i16>
+// LLVM: [[B_I:%.*]] = bitcast <4 x half> [[B]] to <4 x i16>
+// LLVM: [[C_I:%.*]] = bitcast <8 x half> [[C]] to <8 x i16>
+// LLVM: [[A_BYTES:%.*]] = bitcast <4 x i16> [[A_I]] to <8 x i8>
+// LLVM: [[B_BYTES:%.*]] = bitcast <4 x i16> [[B_I]] to <8 x i8>
+// LLVM: [[C_BYTES:%.*]] = bitcast <8 x i16> [[C_I]] to <16 x i8>
+// LLVM: [[A_CAST:%.*]] = bitcast <8 x i8> [[A_BYTES]] to <4 x half>
+// LLVM: [[B_CAST:%.*]] = bitcast <8 x i8> [[B_BYTES]] to <4 x half>
+// LLVM: [[C_CAST:%.*]] = bitcast <16 x i8> [[C_BYTES]] to <8 x half>
+// LLVM: [[LANE:%.*]] = shufflevector <8 x half> [[C_CAST]], <8 x half> 
{{.*}}, <4 x i32> <i32 7, i32 7, i32 7, i32 7>
+// LLVM: [[FMA:%.*]] = call <4 x half> 
@llvm.experimental.constrained.fma.v4f16(<4 x half> [[LANE]], <4 x half> 
[[B_CAST]], <4 x half> [[A_CAST]], metadata !"round.tonearest", metadata 
!"fpexcept.strict")
+// LLVM: ret <4 x half> [[FMA]]
+  return vfma_laneq_f16(a, b, c, 7);
+}
+
+// ALL-LABEL: @test_vfmaq_laneq_f16(
+float16x8_t test_vfmaq_laneq_f16(float16x8_t a, float16x8_t b,
+                                  float16x8_t c) {
+// CIR: [[LANE:%.*]] = cir.vec.shuffle(%{{.*}}, %{{.*}} : !cir.vector<8 x 
!cir.f16>) [#cir.int<7> : !s32i, #cir.int<7> : !s32i, #cir.int<7> : !s32i, 
#cir.int<7> : !s32i, #cir.int<7> : !s32i, #cir.int<7> : !s32i, #cir.int<7> : 
!s32i, #cir.int<7> : !s32i] : !cir.vector<8 x !cir.f16>
+// CIR: cir.fma [[LANE]], %{{.*}}, %{{.*}} : !cir.vector<8 x !cir.f16> 
fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = 
true>
+
+// LLVM-SAME: <8 x half> {{.*}} [[A:%.*]], <8 x half> {{.*}} [[B:%.*]], <8 x 
half> {{.*}} [[C:%.*]]) {{.*}} {
+// LLVM: [[A_I:%.*]] = bitcast <8 x half> [[A]] to <8 x i16>
+// LLVM: [[B_I:%.*]] = bitcast <8 x half> [[B]] to <8 x i16>
+// LLVM: [[C_I:%.*]] = bitcast <8 x half> [[C]] to <8 x i16>
+// LLVM: [[A_BYTES:%.*]] = bitcast <8 x i16> [[A_I]] to <16 x i8>
+// LLVM: [[B_BYTES:%.*]] = bitcast <8 x i16> [[B_I]] to <16 x i8>
+// LLVM: [[C_BYTES:%.*]] = bitcast <8 x i16> [[C_I]] to <16 x i8>
+// LLVM: [[A_CAST:%.*]] = bitcast <16 x i8> [[A_BYTES]] to <8 x half>
+// LLVM: [[B_CAST:%.*]] = bitcast <16 x i8> [[B_BYTES]] to <8 x half>
+// LLVM: [[C_CAST:%.*]] = bitcast <16 x i8> [[C_BYTES]] to <8 x half>
+// LLVM: [[LANE:%.*]] = shufflevector <8 x half> [[C_CAST]], <8 x half> 
{{.*}}, <8 x i32> <i32 7, i32 7, i32 7, i32 7, i32 7, i32 7, i32 7, i32 7>
+// LLVM: [[FMA:%.*]] = call <8 x half> 
@llvm.experimental.constrained.fma.v8f16(<8 x half> [[LANE]], <8 x half> 
[[B_CAST]], <8 x half> [[A_CAST]], metadata !"round.tonearest", metadata 
!"fpexcept.strict")
+// LLVM: ret <8 x half> [[FMA]]
+  return vfmaq_laneq_f16(a, b, c, 7);
+}
+
+// ALL-LABEL: @test_vfmah_lane_f16(
+float16_t test_vfmah_lane_f16(float16_t a, float16_t b, float16x4_t c) {
+// CIR: [[INDEX:%.*]] = cir.const #cir.int<3> : !u64i
+// CIR: [[LANE:%.*]] = cir.vec.extract %{{.*}}{{\[}}[[INDEX]] : !u64i] : 
!cir.vector<4 x !cir.f16>
+// CIR: cir.fma %{{.*}}, [[LANE]], %{{.*}} : !cir.f16 
fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = 
true>
+
+// LLVM-SAME: half {{.*}} [[A:%.*]], half {{.*}} [[B:%.*]], <4 x half> {{.*}} 
[[C:%.*]]) {{.*}} {
+// LLVM: [[LANE:%.*]] = extractelement <4 x half> [[C]], i{{32|64}} 3
+// LLVM: [[FMA:%.*]] = call half @llvm.experimental.constrained.fma.f16(half 
[[B]], half [[LANE]], half [[A]], metadata !"round.tonearest", metadata 
!"fpexcept.strict")
+// LLVM: ret half [[FMA]]
+  return vfmah_lane_f16(a, b, c, 3);
+}
diff --git a/clang/test/CodeGen/AArch64/neon/fused-multiply-constrained.c 
b/clang/test/CodeGen/AArch64/neon/fused-multiply-constrained.c
new file mode 100644
index 00000000000000..52ee3f20628511
--- /dev/null
+++ b/clang/test/CodeGen/AArch64/neon/fused-multiply-constrained.c
@@ -0,0 +1,91 @@
+// REQUIRES: aarch64-registered-target
+
+// RUN:                   %clang_cc1_cg_arm64_neon 
-ffp-exception-behavior=strict -emit-llvm %s -disable-O0-optnone | opt -S 
-passes=mem2reg,sroa | FileCheck %s --check-prefixes=ALL,LLVM
+// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon 
-fexperimental-strict-floating-point -ffp-exception-behavior=strict -fclangir 
-emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,sroa | FileCheck %s 
--check-prefixes=ALL,LLVM --implicit-check-not=' @llvm.fma.' %}
+// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon 
-fexperimental-strict-floating-point -ffp-exception-behavior=strict -fclangir 
-emit-cir  %s -disable-O0-optnone |                               FileCheck %s 
--check-prefixes=ALL,CIR --implicit-check-not='cir.call_llvm_intrinsic "fma"' %}
+
+#include <arm_neon.h>
+
+// LLVM-LABEL: @test_vfma_f64(
+// CIR-LABEL: @vfma_f64(
+float64x1_t test_vfma_f64(float64x1_t a, float64x1_t b, float64x1_t c) {
+// CIR: cir.fma %{{.*}}, %{{.*}}, %{{.*}} : !cir.vector<1 x !cir.double> 
fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = 
true>
+
+// LLVM-SAME: <1 x double> {{.*}} [[A:%.*]], <1 x double> {{.*}} [[B:%.*]], <1 
x double> {{.*}} [[C:%.*]]) {{.*}} {
+// LLVM: [[A_I:%.*]] = bitcast <1 x double> [[A]] to i64
+// LLVM: [[A_INSERT:%.*]] = insertelement <1 x i64> undef, i64 [[A_I]], i64 0
+// LLVM: [[B_I:%.*]] = bitcast <1 x double> [[B]] to i64
+// LLVM: [[B_INSERT:%.*]] = insertelement <1 x i64> undef, i64 [[B_I]], i64 0
+// LLVM: [[C_I:%.*]] = bitcast <1 x double> [[C]] to i64
+// LLVM: [[C_INSERT:%.*]] = insertelement <1 x i64> undef, i64 [[C_I]], i64 0
+// LLVM: [[A_BYTES:%.*]] = bitcast <1 x i64> [[A_INSERT]] to <8 x i8>
+// LLVM: [[B_BYTES:%.*]] = bitcast <1 x i64> [[B_INSERT]] to <8 x i8>
+// LLVM: [[C_BYTES:%.*]] = bitcast <1 x i64> [[C_INSERT]] to <8 x i8>
+// LLVM: [[A_CAST:%.*]] = bitcast <8 x i8> [[A_BYTES]] to <1 x double>
+// LLVM: [[B_CAST:%.*]] = bitcast <8 x i8> [[B_BYTES]] to <1 x double>
+// LLVM: [[C_CAST:%.*]] = bitcast <8 x i8> [[C_BYTES]] to <1 x double>
+// LLVM: [[FMA:%.*]] = call <1 x double> 
@llvm.experimental.constrained.fma.v1f64(<1 x double> [[B_CAST]], <1 x double> 
[[C_CAST]], <1 x double> [[A_CAST]], metadata !"round.tonearest", metadata 
!"fpexcept.strict")
+// LLVM: ret <1 x double> [[FMA]]
+  return vfma_f64(a, b, c);
+}
+
+// ALL-LABEL: @test_vfma_laneq_f64(
+float64x1_t test_vfma_laneq_f64(float64x1_t a, float64x1_t b,
+                                 float64x2_t v) {
+// CIR: [[INDEX:%.*]] = cir.const #cir.int<0> : !u64i
+// CIR: [[LANE:%.*]] = cir.vec.extract %{{.*}}{{\[}}[[INDEX]] : !u64i] : 
!cir.vector<2 x !cir.double>
+// CIR: cir.fma %{{.*}}, [[LANE]], %{{.*}} : !cir.double 
fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = 
true>
+
+// LLVM-SAME: <1 x double> {{.*}} [[A:%.*]], <1 x double> {{.*}} [[B:%.*]], <2 
x double> {{.*}} [[V:%.*]]) {{.*}} {
+// LLVM: [[A_I:%.*]] = bitcast <1 x double> [[A]] to i64
+// LLVM: [[A_INSERT:%.*]] = insertelement <1 x i64> undef, i64 [[A_I]], i64 0
+// LLVM: [[B_I:%.*]] = bitcast <1 x double> [[B]] to i64
+// LLVM: [[B_INSERT:%.*]] = insertelement <1 x i64> undef, i64 [[B_I]], i64 0
+// LLVM: [[V_I:%.*]] = bitcast <2 x double> [[V]] to <2 x i64>
+// LLVM: [[A_BYTES:%.*]] = bitcast <1 x i64> [[A_INSERT]] to <8 x i8>
+// LLVM: [[B_BYTES:%.*]] = bitcast <1 x i64> [[B_INSERT]] to <8 x i8>
+// LLVM: [[V_BYTES:%.*]] = bitcast <2 x i64> [[V_I]] to <16 x i8>
+// LLVM: [[A_CAST:%.*]] = bitcast <8 x i8> [[A_BYTES]] to double
+// LLVM: [[B_CAST:%.*]] = bitcast <8 x i8> [[B_BYTES]] to double
+// LLVM: [[V_CAST:%.*]] = bitcast <16 x i8> [[V_BYTES]] to <2 x double>
+// LLVM: [[LANE:%.*]] = extractelement <2 x double> [[V_CAST]], i{{32|64}} 0
+// LLVM: [[FMA:%.*]] = call double 
@llvm.experimental.constrained.fma.f64(double [[B_CAST]], double [[LANE]], 
double [[A_CAST]], metadata !"round.tonearest", metadata !"fpexcept.strict")
+// LLVM: [[RESULT:%.*]] = bitcast double [[FMA]] to <1 x double>
+// LLVM: ret <1 x double> [[RESULT]]
+  return vfma_laneq_f64(a, b, v, 0);
+}
+
+// ALL-LABEL: @test_vfmaq_laneq_f64(
+float64x2_t test_vfmaq_laneq_f64(float64x2_t a, float64x2_t b,
+                                  float64x2_t v) {
+// CIR: [[LANE:%.*]] = cir.vec.shuffle(%{{.*}}, %{{.*}} : !cir.vector<2 x 
!cir.double>) [#cir.int<1> : !s32i, #cir.int<1> : !s32i] : !cir.vector<2 x 
!cir.double>
+// CIR: cir.fma [[LANE]], %{{.*}}, %{{.*}} : !cir.vector<2 x !cir.double> 
fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = 
true>
+
+// LLVM-SAME: <2 x double> {{.*}} [[A:%.*]], <2 x double> {{.*}} [[B:%.*]], <2 
x double> {{.*}} [[V:%.*]]) {{.*}} {
+// LLVM: [[A_I:%.*]] = bitcast <2 x double> [[A]] to <2 x i64>
+// LLVM: [[B_I:%.*]] = bitcast <2 x double> [[B]] to <2 x i64>
+// LLVM: [[V_I:%.*]] = bitcast <2 x double> [[V]] to <2 x i64>
+// LLVM: [[A_BYTES:%.*]] = bitcast <2 x i64> [[A_I]] to <16 x i8>
+// LLVM: [[B_BYTES:%.*]] = bitcast <2 x i64> [[B_I]] to <16 x i8>
+// LLVM: [[V_BYTES:%.*]] = bitcast <2 x i64> [[V_I]] to <16 x i8>
+// LLVM: [[A_CAST:%.*]] = bitcast <16 x i8> [[A_BYTES]] to <2 x double>
+// LLVM: [[B_CAST:%.*]] = bitcast <16 x i8> [[B_BYTES]] to <2 x double>
+// LLVM: [[V_CAST:%.*]] = bitcast <16 x i8> [[V_BYTES]] to <2 x double>
+// LLVM: [[LANE:%.*]] = shufflevector <2 x double> [[V_CAST]], <2 x double> 
{{.*}}, <2 x i32> <i32 1, i32 1>
+// LLVM: [[FMA:%.*]] = call <2 x double> 
@llvm.experimental.constrained.fma.v2f64(<2 x double> [[LANE]], <2 x double> 
[[B_CAST]], <2 x double> [[A_CAST]], metadata !"round.tonearest", metadata 
!"fpexcept.strict")
+// LLVM: ret <2 x double> [[FMA]]
+  return vfmaq_laneq_f64(a, b, v, 1);
+}
+
+// ALL-LABEL: @test_vfmas_lane_f32(
+float32_t test_vfmas_lane_f32(float32_t a, float32_t b, float32x2_t c) {
+// CIR: [[INDEX:%.*]] = cir.const #cir.int<1> : !u64i
+// CIR: [[LANE:%.*]] = cir.vec.extract %{{.*}}{{\[}}[[INDEX]] : !u64i] : 
!cir.vector<2 x !cir.float>
+// CIR: cir.fma %{{.*}}, [[LANE]], %{{.*}} : !cir.float 
fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = 
true>
+
+// LLVM-SAME: float {{.*}} [[A:%.*]], float {{.*}} [[B:%.*]], <2 x float> 
{{.*}} [[C:%.*]]) {{.*}} {
+// LLVM: [[LANE:%.*]] = extractelement <2 x float> [[C]], i{{32|64}} 1
+// LLVM: [[FMA:%.*]] = call float @llvm.experimental.constrained.fma.f32(float 
[[B]], float [[LANE]], float [[A]], metadata !"round.tonearest", metadata 
!"fpexcept.strict")
+// LLVM: ret float [[FMA]]
+  return vfmas_lane_f32(a, b, c, 1);
+}
diff --git a/clang/test/CodeGen/AArch64/neon/intrinsics-constrained.c 
b/clang/test/CodeGen/AArch64/neon/intrinsics-constrained.c
new file mode 100644
index 00000000000000..ef915a0ccea1e9
--- /dev/null
+++ b/clang/test/CodeGen/AArch64/neon/intrinsics-constrained.c
@@ -0,0 +1,56 @@
+// REQUIRES: aarch64-registered-target
+
+// RUN: %clang_cc1_cg_arm64_neon -target-feature +fullfp16 
-ffp-exception-behavior=maytrap -DEXCEPT=1 -emit-llvm %s -disable-O0-optnone | 
opt -S -passes=mem2reg,sroa | FileCheck %s --check-prefix=LLVM 
--implicit-check-not=fpexcept.maytrap --implicit-check-not=' @llvm.sqrt.'
+// RUN: %clang_cc1_cg_arm64_neon -target-feature +fullfp16 
-ffp-exception-behavior=strict -emit-llvm %s -disable-O0-optnone | opt -S 
-passes=mem2reg,sroa | FileCheck %s --check-prefix=LLVM --implicit-check-not=' 
@llvm.sqrt.'
+// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 
-fexperimental-strict-floating-point -ffp-exception-behavior=maytrap -DEXCEPT=1 
-fclangir -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,sroa | 
FileCheck %s --check-prefix=LLVM --implicit-check-not=fpexcept.maytrap 
--implicit-check-not=' @llvm.sqrt.' %}
+// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 
-fexperimental-strict-floating-point -ffp-exception-behavior=maytrap -DEXCEPT=1 
-fclangir -emit-cir  %s -disable-O0-optnone |                               
FileCheck %s --check-prefix=CIR --implicit-check-not='except_mode = maytrap' 
--implicit-check-not='cir.call_llvm_intrinsic "sqrt"' %}
+// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 
-fexperimental-strict-floating-point -ffp-exception-behavior=strict             
-fclangir -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,sroa | 
FileCheck %s --check-prefix=LLVM --implicit-check-not=' @llvm.sqrt.' %}
+// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 
-fexperimental-strict-floating-point -ffp-exception-behavior=strict             
-fclangir -emit-cir  %s -disable-O0-optnone |                               
FileCheck %s --check-prefix=CIR --implicit-check-not='cir.call_llvm_intrinsic 
"sqrt"' %}
+
+#if EXCEPT
+#pragma float_control(except, on)
+#endif
+
+#include <arm_neon.h>
+
+// LLVM-LABEL: @test_vsqrt_f16(
+// CIR-LABEL: @vsqrt_f16(
+float16x4_t test_vsqrt_f16(float16x4_t a) {
+// CIR: cir.sqrt %{{.*}} : !cir.vector<4 x !cir.f16> 
fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = 
true>
+
+// LLVM-SAME: <4 x half> {{.*}} [[A:%.*]]) {{.*}} {
+// LLVM: [[A_I:%.*]] = bitcast <4 x half> [[A]] to <4 x i16>
+// LLVM: [[A_BYTES:%.*]] = bitcast <4 x i16> [[A_I]] to <8 x i8>
+// LLVM: [[A_CAST:%.*]] = bitcast <8 x i8> [[A_BYTES]] to <4 x half>
+// LLVM: [[SQRT:%.*]] = call <4 x half> 
@llvm.experimental.constrained.sqrt.v4f16(<4 x half> [[A_CAST]], metadata 
!"round.tonearest", metadata !"fpexcept.strict")
+// LLVM: ret <4 x half> [[SQRT]]
+  return vsqrt_f16(a);
+}
+
+// LLVM-LABEL: @test_vsqrtq_f16(
+// CIR-LABEL: @vsqrtq_f16(
+float16x8_t test_vsqrtq_f16(float16x8_t a) {
+// CIR: cir.sqrt %{{.*}} : !cir.vector<8 x !cir.f16> 
fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = 
true>
+
+// LLVM-SAME: <8 x half> {{.*}} [[A:%.*]]) {{.*}} {
+// LLVM: [[A_I:%.*]] = bitcast <8 x half> [[A]] to <8 x i16>
+// LLVM: [[A_BYTES:%.*]] = bitcast <8 x i16> [[A_I]] to <16 x i8>
+// LLVM: [[A_CAST:%.*]] = bitcast <16 x i8> [[A_BYTES]] to <8 x half>
+// LLVM: [[SQRT:%.*]] = call <8 x half> 
@llvm.experimental.constrained.sqrt.v8f16(<8 x half> [[A_CAST]], metadata 
!"round.tonearest", metadata !"fpexcept.strict")
+// LLVM: ret <8 x half> [[SQRT]]
+  return vsqrtq_f16(a);
+}
+
+// LLVM-LABEL: @test_vsqrtq_f64(
+// CIR-LABEL: @vsqrtq_f64(
+float64x2_t test_vsqrtq_f64(float64x2_t a) {
+// CIR: cir.sqrt %{{.*}} : !cir.vector<2 x !cir.double> 
fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = 
true>
+
+// LLVM-SAME: <2 x double> {{.*}} [[A:%.*]]) {{.*}} {
+// LLVM: [[A_I:%.*]] = bitcast <2 x double> [[A]] to <2 x i64>
+// LLVM: [[A_BYTES:%.*]] = bitcast <2 x i64> [[A_I]] to <16 x i8>
+// LLVM: [[A_CAST:%.*]] = bitcast <16 x i8> [[A_BYTES]] to <2 x double>
+// LLVM: [[SQRT:%.*]] = call <2 x double> 
@llvm.experimental.constrained.sqrt.v2f64(<2 x double> [[A_CAST]], metadata 
!"round.tonearest", metadata !"fpexcept.strict")
+// LLVM: ret <2 x double> [[SQRT]]
+  return vsqrtq_f64(a);
+}
diff --git a/clang/test/CodeGen/AArch64/v8.2a-fp16-intrinsics-constrained.c 
b/clang/test/CodeGen/AArch64/v8.2a-fp16-intrinsics-constrained.c
index c9d5071c4eb838..4006171e402964 100644
--- a/clang/test/CodeGen/AArch64/v8.2a-fp16-intrinsics-constrained.c
+++ b/clang/test/CodeGen/AArch64/v8.2a-fp16-intrinsics-constrained.c
@@ -200,14 +200,6 @@ float16_t test_vrndxh_f16(float16_t a) {
   return vrndxh_f16(a);
 }
 
-// COMMON-LABEL: test_vsqrth_f16
-// UNCONSTRAINED:  [[SQR:%.*]] = call half @llvm.sqrt.f16(half %a)
-// CONSTRAINED:    [[SQR:%.*]] = call half 
@llvm.experimental.constrained.sqrt.f16(half %a, metadata !"round.tonearest", 
metadata !"fpexcept.strict")
-// COMMONIR:       ret half [[SQR]]
-float16_t test_vsqrth_f16(float16_t a) {
-  return vsqrth_f16(a);
-}
-
 // COMMON-LABEL: test_vaddh_f16
 // UNCONSTRAINED:  [[ADD:%.*]] = fadd half %a, %b
 // CONSTRAINED:    [[ADD:%.*]] = call half 
@llvm.experimental.constrained.fadd.f16(half %a, half %b, metadata 
!"round.tonearest", metadata !"fpexcept.strict")
@@ -285,14 +277,6 @@ float16_t test_vsubh_f16(float16_t a, float16_t b) {
   return vsubh_f16(a, b);
 }
 
-// COMMON-LABEL: test_vfmah_f16
-// UNCONSTRAINED:  [[FMA:%.*]] = call half @llvm.fma.f16(half %b, half %c, 
half %a)
-// CONSTRAINED:    [[FMA:%.*]] = call half 
@llvm.experimental.constrained.fma.f16(half %b, half %c, half %a, metadata 
!"round.tonearest", metadata !"fpexcept.strict")
-// COMMONIR:       ret half [[FMA]]
-float16_t test_vfmah_f16(float16_t a, float16_t b, float16_t c) {
-  return vfmah_f16(a, b, c);
-}
-
 // COMMON-LABEL: test_vfmsh_f16
 // COMMONIR:  [[SUB:%.*]] = fneg half %b
 // UNCONSTRAINED:  [[ADD:%.*]] = call half @llvm.fma.f16(half [[SUB]], half 
%c, half %a)
diff --git a/clang/test/CodeGen/AArch64/v8.2a-neon-intrinsics-constrained.c 
b/clang/test/CodeGen/AArch64/v8.2a-neon-intrinsics-constrained.c
index dac8b931ff210d..a30bfa83526469 100644
--- a/clang/test/CodeGen/AArch64/v8.2a-neon-intrinsics-constrained.c
+++ b/clang/test/CodeGen/AArch64/v8.2a-neon-intrinsics-constrained.c
@@ -21,120 +21,8 @@
 
 #include <arm_neon.h>
 
-// UNCONSTRAINED-LABEL: define dso_local <4 x half> @test_vsqrt_f16(
-// UNCONSTRAINED-SAME: <4 x half> noundef [[A:%.*]]) #[[ATTR0:[0-9]+]] {
-// UNCONSTRAINED-NEXT:  [[ENTRY:.*:]]
-// UNCONSTRAINED-NEXT:    [[TMP0:%.*]] = bitcast <4 x half> [[A]] to <4 x i16>
-// UNCONSTRAINED-NEXT:    [[TMP1:%.*]] = bitcast <4 x i16> [[TMP0]] to <8 x i8>
-// UNCONSTRAINED-NEXT:    [[TMP2:%.*]] = bitcast <8 x i8> [[TMP1]] to <4 x 
half>
-// UNCONSTRAINED-NEXT:    [[VSQRT_I:%.*]] = call <4 x half> 
@llvm.sqrt.v4f16(<4 x half> [[TMP2]])
-// UNCONSTRAINED-NEXT:    ret <4 x half> [[VSQRT_I]]
-//
-// CONSTRAINED-LABEL: define dso_local <4 x half> @test_vsqrt_f16(
-// CONSTRAINED-SAME: <4 x half> noundef [[A:%.*]]) #[[ATTR0:[0-9]+]] {
-// CONSTRAINED-NEXT:  [[ENTRY:.*:]]
-// CONSTRAINED-NEXT:    [[TMP0:%.*]] = bitcast <4 x half> [[A]] to <4 x i16>
-// CONSTRAINED-NEXT:    [[TMP1:%.*]] = bitcast <4 x i16> [[TMP0]] to <8 x i8>
-// CONSTRAINED-NEXT:    [[TMP2:%.*]] = bitcast <8 x i8> [[TMP1]] to <4 x half>
-// CONSTRAINED-NEXT:    [[VSQRT_I:%.*]] = call <4 x half> 
@llvm.experimental.constrained.sqrt.v4f16(<4 x half> [[TMP2]], metadata 
!"round.tonearest", metadata !"fpexcept.strict") #[[ATTR2:[0-9]+]]
-// CONSTRAINED-NEXT:    ret <4 x half> [[VSQRT_I]]
-//
-float16x4_t test_vsqrt_f16(float16x4_t a) {
-  return vsqrt_f16(a);
-}
-
-// UNCONSTRAINED-LABEL: define dso_local <8 x half> @test_vsqrtq_f16(
-// UNCONSTRAINED-SAME: <8 x half> noundef [[A:%.*]]) #[[ATTR0]] {
-// UNCONSTRAINED-NEXT:  [[ENTRY:.*:]]
-// UNCONSTRAINED-NEXT:    [[TMP0:%.*]] = bitcast <8 x half> [[A]] to <8 x i16>
-// UNCONSTRAINED-NEXT:    [[TMP1:%.*]] = bitcast <8 x i16> [[TMP0]] to <16 x 
i8>
-// UNCONSTRAINED-NEXT:    [[TMP2:%.*]] = bitcast <16 x i8> [[TMP1]] to <8 x 
half>
-// UNCONSTRAINED-NEXT:    [[VSQRT_I:%.*]] = call <8 x half> 
@llvm.sqrt.v8f16(<8 x half> [[TMP2]])
-// UNCONSTRAINED-NEXT:    ret <8 x half> [[VSQRT_I]]
-//
-// CONSTRAINED-LABEL: define dso_local <8 x half> @test_vsqrtq_f16(
-// CONSTRAINED-SAME: <8 x half> noundef [[A:%.*]]) #[[ATTR0]] {
-// CONSTRAINED-NEXT:  [[ENTRY:.*:]]
-// CONSTRAINED-NEXT:    [[TMP0:%.*]] = bitcast <8 x half> [[A]] to <8 x i16>
-// CONSTRAINED-NEXT:    [[TMP1:%.*]] = bitcast <8 x i16> [[TMP0]] to <16 x i8>
-// CONSTRAINED-NEXT:    [[TMP2:%.*]] = bitcast <16 x i8> [[TMP1]] to <8 x half>
-// CONSTRAINED-NEXT:    [[VSQRT_I:%.*]] = call <8 x half> 
@llvm.experimental.constrained.sqrt.v8f16(<8 x half> [[TMP2]], metadata 
!"round.tonearest", metadata !"fpexcept.strict") #[[ATTR2]]
-// CONSTRAINED-NEXT:    ret <8 x half> [[VSQRT_I]]
-//
-float16x8_t test_vsqrtq_f16(float16x8_t a) {
-  return vsqrtq_f16(a);
-}
-
-// UNCONSTRAINED-LABEL: define dso_local <4 x half> @test_vfma_f16(
-// UNCONSTRAINED-SAME: <4 x half> noundef [[A:%.*]], <4 x half> noundef 
[[B:%.*]], <4 x half> noundef [[C:%.*]]) #[[ATTR0]] {
-// UNCONSTRAINED-NEXT:  [[ENTRY:.*:]]
-// UNCONSTRAINED-NEXT:    [[TMP0:%.*]] = bitcast <4 x half> [[A]] to <4 x i16>
-// UNCONSTRAINED-NEXT:    [[TMP1:%.*]] = bitcast <4 x half> [[B]] to <4 x i16>
-// UNCONSTRAINED-NEXT:    [[TMP2:%.*]] = bitcast <4 x half> [[C]] to <4 x i16>
-// UNCONSTRAINED-NEXT:    [[TMP3:%.*]] = bitcast <4 x i16> [[TMP0]] to <8 x i8>
-// UNCONSTRAINED-NEXT:    [[TMP4:%.*]] = bitcast <4 x i16> [[TMP1]] to <8 x i8>
-// UNCONSTRAINED-NEXT:    [[TMP5:%.*]] = bitcast <4 x i16> [[TMP2]] to <8 x i8>
-// UNCONSTRAINED-NEXT:    [[TMP6:%.*]] = bitcast <8 x i8> [[TMP3]] to <4 x 
half>
-// UNCONSTRAINED-NEXT:    [[TMP7:%.*]] = bitcast <8 x i8> [[TMP4]] to <4 x 
half>
-// UNCONSTRAINED-NEXT:    [[TMP8:%.*]] = bitcast <8 x i8> [[TMP5]] to <4 x 
half>
-// UNCONSTRAINED-NEXT:    [[TMP9:%.*]] = call <4 x half> @llvm.fma.v4f16(<4 x 
half> [[TMP7]], <4 x half> [[TMP8]], <4 x half> [[TMP6]])
-// UNCONSTRAINED-NEXT:    ret <4 x half> [[TMP9]]
-//
-// CONSTRAINED-LABEL: define dso_local <4 x half> @test_vfma_f16(
-// CONSTRAINED-SAME: <4 x half> noundef [[A:%.*]], <4 x half> noundef 
[[B:%.*]], <4 x half> noundef [[C:%.*]]) #[[ATTR0]] {
-// CONSTRAINED-NEXT:  [[ENTRY:.*:]]
-// CONSTRAINED-NEXT:    [[TMP0:%.*]] = bitcast <4 x half> [[A]] to <4 x i16>
-// CONSTRAINED-NEXT:    [[TMP1:%.*]] = bitcast <4 x half> [[B]] to <4 x i16>
-// CONSTRAINED-NEXT:    [[TMP2:%.*]] = bitcast <4 x half> [[C]] to <4 x i16>
-// CONSTRAINED-NEXT:    [[TMP3:%.*]] = bitcast <4 x i16> [[TMP0]] to <8 x i8>
-// CONSTRAINED-NEXT:    [[TMP4:%.*]] = bitcast <4 x i16> [[TMP1]] to <8 x i8>
-// CONSTRAINED-NEXT:    [[TMP5:%.*]] = bitcast <4 x i16> [[TMP2]] to <8 x i8>
-// CONSTRAINED-NEXT:    [[TMP6:%.*]] = bitcast <8 x i8> [[TMP3]] to <4 x half>
-// CONSTRAINED-NEXT:    [[TMP7:%.*]] = bitcast <8 x i8> [[TMP4]] to <4 x half>
-// CONSTRAINED-NEXT:    [[TMP8:%.*]] = bitcast <8 x i8> [[TMP5]] to <4 x half>
-// CONSTRAINED-NEXT:    [[TMP9:%.*]] = call <4 x half> 
@llvm.experimental.constrained.fma.v4f16(<4 x half> [[TMP7]], <4 x half> 
[[TMP8]], <4 x half> [[TMP6]], metadata !"round.tonearest", metadata 
!"fpexcept.strict") #[[ATTR2]]
-// CONSTRAINED-NEXT:    ret <4 x half> [[TMP9]]
-//
-float16x4_t test_vfma_f16(float16x4_t a, float16x4_t b, float16x4_t c) {
-  return vfma_f16(a, b, c);
-}
-
-// UNCONSTRAINED-LABEL: define dso_local <8 x half> @test_vfmaq_f16(
-// UNCONSTRAINED-SAME: <8 x half> noundef [[A:%.*]], <8 x half> noundef 
[[B:%.*]], <8 x half> noundef [[C:%.*]]) #[[ATTR0]] {
-// UNCONSTRAINED-NEXT:  [[ENTRY:.*:]]
-// UNCONSTRAINED-NEXT:    [[TMP0:%.*]] = bitcast <8 x half> [[A]] to <8 x i16>
-// UNCONSTRAINED-NEXT:    [[TMP1:%.*]] = bitcast <8 x half> [[B]] to <8 x i16>
-// UNCONSTRAINED-NEXT:    [[TMP2:%.*]] = bitcast <8 x half> [[C]] to <8 x i16>
-// UNCONSTRAINED-NEXT:    [[TMP3:%.*]] = bitcast <8 x i16> [[TMP0]] to <16 x 
i8>
-// UNCONSTRAINED-NEXT:    [[TMP4:%.*]] = bitcast <8 x i16> [[TMP1]] to <16 x 
i8>
-// UNCONSTRAINED-NEXT:    [[TMP5:%.*]] = bitcast <8 x i16> [[TMP2]] to <16 x 
i8>
-// UNCONSTRAINED-NEXT:    [[TMP6:%.*]] = bitcast <16 x i8> [[TMP3]] to <8 x 
half>
-// UNCONSTRAINED-NEXT:    [[TMP7:%.*]] = bitcast <16 x i8> [[TMP4]] to <8 x 
half>
-// UNCONSTRAINED-NEXT:    [[TMP8:%.*]] = bitcast <16 x i8> [[TMP5]] to <8 x 
half>
-// UNCONSTRAINED-NEXT:    [[TMP9:%.*]] = call <8 x half> @llvm.fma.v8f16(<8 x 
half> [[TMP7]], <8 x half> [[TMP8]], <8 x half> [[TMP6]])
-// UNCONSTRAINED-NEXT:    ret <8 x half> [[TMP9]]
-//
-// CONSTRAINED-LABEL: define dso_local <8 x half> @test_vfmaq_f16(
-// CONSTRAINED-SAME: <8 x half> noundef [[A:%.*]], <8 x half> noundef 
[[B:%.*]], <8 x half> noundef [[C:%.*]]) #[[ATTR0]] {
-// CONSTRAINED-NEXT:  [[ENTRY:.*:]]
-// CONSTRAINED-NEXT:    [[TMP0:%.*]] = bitcast <8 x half> [[A]] to <8 x i16>
-// CONSTRAINED-NEXT:    [[TMP1:%.*]] = bitcast <8 x half> [[B]] to <8 x i16>
-// CONSTRAINED-NEXT:    [[TMP2:%.*]] = bitcast <8 x half> [[C]] to <8 x i16>
-// CONSTRAINED-NEXT:    [[TMP3:%.*]] = bitcast <8 x i16> [[TMP0]] to <16 x i8>
-// CONSTRAINED-NEXT:    [[TMP4:%.*]] = bitcast <8 x i16> [[TMP1]] to <16 x i8>
-// CONSTRAINED-NEXT:    [[TMP5:%.*]] = bitcast <8 x i16> [[TMP2]] to <16 x i8>
-// CONSTRAINED-NEXT:    [[TMP6:%.*]] = bitcast <16 x i8> [[TMP3]] to <8 x half>
-// CONSTRAINED-NEXT:    [[TMP7:%.*]] = bitcast <16 x i8> [[TMP4]] to <8 x half>
-// CONSTRAINED-NEXT:    [[TMP8:%.*]] = bitcast <16 x i8> [[TMP5]] to <8 x half>
-// CONSTRAINED-NEXT:    [[TMP9:%.*]] = call <8 x half> 
@llvm.experimental.constrained.fma.v8f16(<8 x half> [[TMP7]], <8 x half> 
[[TMP8]], <8 x half> [[TMP6]], metadata !"round.tonearest", metadata 
!"fpexcept.strict") #[[ATTR2]]
-// CONSTRAINED-NEXT:    ret <8 x half> [[TMP9]]
-//
-float16x8_t test_vfmaq_f16(float16x8_t a, float16x8_t b, float16x8_t c) {
-  return vfmaq_f16(a, b, c);
-}
-
 // UNCONSTRAINED-LABEL: define dso_local <4 x half> @test_vfms_f16(
-// UNCONSTRAINED-SAME: <4 x half> noundef [[A:%.*]], <4 x half> noundef 
[[B:%.*]], <4 x half> noundef [[C:%.*]]) #[[ATTR0]] {
+// UNCONSTRAINED-SAME: <4 x half> noundef [[A:%.*]], <4 x half> noundef 
[[B:%.*]], <4 x half> noundef [[C:%.*]]) #[[ATTR0:[0-9]+]] {
 // UNCONSTRAINED-NEXT:  [[ENTRY:.*:]]
 // UNCONSTRAINED-NEXT:    [[FNEG_I:%.*]] = fneg <4 x half> [[B]]
 // UNCONSTRAINED-NEXT:    [[TMP0:%.*]] = bitcast <4 x half> [[A]] to <4 x i16>
@@ -150,7 +38,7 @@ float16x8_t test_vfmaq_f16(float16x8_t a, float16x8_t b, 
float16x8_t c) {
 // UNCONSTRAINED-NEXT:    ret <4 x half> [[TMP9]]
 //
 // CONSTRAINED-LABEL: define dso_local <4 x half> @test_vfms_f16(
-// CONSTRAINED-SAME: <4 x half> noundef [[A:%.*]], <4 x half> noundef 
[[B:%.*]], <4 x half> noundef [[C:%.*]]) #[[ATTR0]] {
+// CONSTRAINED-SAME: <4 x half> noundef [[A:%.*]], <4 x half> noundef 
[[B:%.*]], <4 x half> noundef [[C:%.*]]) #[[ATTR0:[0-9]+]] {
 // CONSTRAINED-NEXT:  [[ENTRY:.*:]]
 // CONSTRAINED-NEXT:    [[FNEG_I:%.*]] = fneg <4 x half> [[B]]
 // CONSTRAINED-NEXT:    [[TMP0:%.*]] = bitcast <4 x half> [[A]] to <4 x i16>
@@ -162,7 +50,7 @@ float16x8_t test_vfmaq_f16(float16x8_t a, float16x8_t b, 
float16x8_t c) {
 // CONSTRAINED-NEXT:    [[TMP6:%.*]] = bitcast <8 x i8> [[TMP3]] to <4 x half>
 // CONSTRAINED-NEXT:    [[TMP7:%.*]] = bitcast <8 x i8> [[TMP4]] to <4 x half>
 // CONSTRAINED-NEXT:    [[TMP8:%.*]] = bitcast <8 x i8> [[TMP5]] to <4 x half>
-// CONSTRAINED-NEXT:    [[TMP9:%.*]] = call <4 x half> 
@llvm.experimental.constrained.fma.v4f16(<4 x half> [[TMP7]], <4 x half> 
[[TMP8]], <4 x half> [[TMP6]], metadata !"round.tonearest", metadata 
!"fpexcept.strict") #[[ATTR2]]
+// CONSTRAINED-NEXT:    [[TMP9:%.*]] = call <4 x half> 
@llvm.experimental.constrained.fma.v4f16(<4 x half> [[TMP7]], <4 x half> 
[[TMP8]], <4 x half> [[TMP6]], metadata !"round.tonearest", metadata 
!"fpexcept.strict") #[[ATTR2:[0-9]+]]
 // CONSTRAINED-NEXT:    ret <4 x half> [[TMP9]]
 //
 float16x4_t test_vfms_f16(float16x4_t a, float16x4_t b, float16x4_t c) {
@@ -205,150 +93,6 @@ float16x8_t test_vfmsq_f16(float16x8_t a, float16x8_t b, 
float16x8_t c) {
   return vfmsq_f16(a, b, c);
 }
 
-// UNCONSTRAINED-LABEL: define dso_local <4 x half> @test_vfma_lane_f16(
-// UNCONSTRAINED-SAME: <4 x half> noundef [[A:%.*]], <4 x half> noundef 
[[B:%.*]], <4 x half> noundef [[C:%.*]]) #[[ATTR0]] {
-// UNCONSTRAINED-NEXT:  [[ENTRY:.*:]]
-// UNCONSTRAINED-NEXT:    [[TMP0:%.*]] = bitcast <4 x half> [[A]] to <4 x i16>
-// UNCONSTRAINED-NEXT:    [[TMP1:%.*]] = bitcast <4 x half> [[B]] to <4 x i16>
-// UNCONSTRAINED-NEXT:    [[TMP2:%.*]] = bitcast <4 x half> [[C]] to <4 x i16>
-// UNCONSTRAINED-NEXT:    [[TMP3:%.*]] = bitcast <4 x i16> [[TMP0]] to <8 x i8>
-// UNCONSTRAINED-NEXT:    [[TMP4:%.*]] = bitcast <4 x i16> [[TMP1]] to <8 x i8>
-// UNCONSTRAINED-NEXT:    [[TMP5:%.*]] = bitcast <4 x i16> [[TMP2]] to <8 x i8>
-// UNCONSTRAINED-NEXT:    [[TMP6:%.*]] = bitcast <8 x i8> [[TMP5]] to <4 x 
half>
-// UNCONSTRAINED-NEXT:    [[LANE:%.*]] = shufflevector <4 x half> [[TMP6]], <4 
x half> [[TMP6]], <4 x i32> <i32 3, i32 3, i32 3, i32 3>
-// UNCONSTRAINED-NEXT:    [[FMLA:%.*]] = bitcast <8 x i8> [[TMP4]] to <4 x 
half>
-// UNCONSTRAINED-NEXT:    [[FMLA1:%.*]] = bitcast <8 x i8> [[TMP3]] to <4 x 
half>
-// UNCONSTRAINED-NEXT:    [[FMLA2:%.*]] = call <4 x half> @llvm.fma.v4f16(<4 x 
half> [[FMLA]], <4 x half> [[LANE]], <4 x half> [[FMLA1]])
-// UNCONSTRAINED-NEXT:    ret <4 x half> [[FMLA2]]
-//
-// CONSTRAINED-LABEL: define dso_local <4 x half> @test_vfma_lane_f16(
-// CONSTRAINED-SAME: <4 x half> noundef [[A:%.*]], <4 x half> noundef 
[[B:%.*]], <4 x half> noundef [[C:%.*]]) #[[ATTR0]] {
-// CONSTRAINED-NEXT:  [[ENTRY:.*:]]
-// CONSTRAINED-NEXT:    [[TMP0:%.*]] = bitcast <4 x half> [[A]] to <4 x i16>
-// CONSTRAINED-NEXT:    [[TMP1:%.*]] = bitcast <4 x half> [[B]] to <4 x i16>
-// CONSTRAINED-NEXT:    [[TMP2:%.*]] = bitcast <4 x half> [[C]] to <4 x i16>
-// CONSTRAINED-NEXT:    [[TMP3:%.*]] = bitcast <4 x i16> [[TMP0]] to <8 x i8>
-// CONSTRAINED-NEXT:    [[TMP4:%.*]] = bitcast <4 x i16> [[TMP1]] to <8 x i8>
-// CONSTRAINED-NEXT:    [[TMP5:%.*]] = bitcast <4 x i16> [[TMP2]] to <8 x i8>
-// CONSTRAINED-NEXT:    [[TMP6:%.*]] = bitcast <8 x i8> [[TMP5]] to <4 x half>
-// CONSTRAINED-NEXT:    [[LANE:%.*]] = shufflevector <4 x half> [[TMP6]], <4 x 
half> [[TMP6]], <4 x i32> <i32 3, i32 3, i32 3, i32 3>
-// CONSTRAINED-NEXT:    [[FMLA:%.*]] = bitcast <8 x i8> [[TMP4]] to <4 x half>
-// CONSTRAINED-NEXT:    [[FMLA1:%.*]] = bitcast <8 x i8> [[TMP3]] to <4 x half>
-// CONSTRAINED-NEXT:    [[FMLA2:%.*]] = call <4 x half> 
@llvm.experimental.constrained.fma.v4f16(<4 x half> [[FMLA]], <4 x half> 
[[LANE]], <4 x half> [[FMLA1]], metadata !"round.tonearest", metadata 
!"fpexcept.strict") #[[ATTR2]]
-// CONSTRAINED-NEXT:    ret <4 x half> [[FMLA2]]
-//
-float16x4_t test_vfma_lane_f16(float16x4_t a, float16x4_t b, float16x4_t c) {
-  return vfma_lane_f16(a, b, c, 3);
-}
-
-// UNCONSTRAINED-LABEL: define dso_local <8 x half> @test_vfmaq_lane_f16(
-// UNCONSTRAINED-SAME: <8 x half> noundef [[A:%.*]], <8 x half> noundef 
[[B:%.*]], <4 x half> noundef [[C:%.*]]) #[[ATTR0]] {
-// UNCONSTRAINED-NEXT:  [[ENTRY:.*:]]
-// UNCONSTRAINED-NEXT:    [[TMP0:%.*]] = bitcast <8 x half> [[A]] to <8 x i16>
-// UNCONSTRAINED-NEXT:    [[TMP1:%.*]] = bitcast <8 x half> [[B]] to <8 x i16>
-// UNCONSTRAINED-NEXT:    [[TMP2:%.*]] = bitcast <4 x half> [[C]] to <4 x i16>
-// UNCONSTRAINED-NEXT:    [[TMP3:%.*]] = bitcast <8 x i16> [[TMP0]] to <16 x 
i8>
-// UNCONSTRAINED-NEXT:    [[TMP4:%.*]] = bitcast <8 x i16> [[TMP1]] to <16 x 
i8>
-// UNCONSTRAINED-NEXT:    [[TMP5:%.*]] = bitcast <4 x i16> [[TMP2]] to <8 x i8>
-// UNCONSTRAINED-NEXT:    [[TMP6:%.*]] = bitcast <8 x i8> [[TMP5]] to <4 x 
half>
-// UNCONSTRAINED-NEXT:    [[LANE:%.*]] = shufflevector <4 x half> [[TMP6]], <4 
x half> [[TMP6]], <8 x i32> <i32 3, i32 3, i32 3, i32 3, i32 3, i32 3, i32 3, 
i32 3>
-// UNCONSTRAINED-NEXT:    [[FMLA:%.*]] = bitcast <16 x i8> [[TMP4]] to <8 x 
half>
-// UNCONSTRAINED-NEXT:    [[FMLA1:%.*]] = bitcast <16 x i8> [[TMP3]] to <8 x 
half>
-// UNCONSTRAINED-NEXT:    [[FMLA2:%.*]] = call <8 x half> @llvm.fma.v8f16(<8 x 
half> [[FMLA]], <8 x half> [[LANE]], <8 x half> [[FMLA1]])
-// UNCONSTRAINED-NEXT:    ret <8 x half> [[FMLA2]]
-//
-// CONSTRAINED-LABEL: define dso_local <8 x half> @test_vfmaq_lane_f16(
-// CONSTRAINED-SAME: <8 x half> noundef [[A:%.*]], <8 x half> noundef 
[[B:%.*]], <4 x half> noundef [[C:%.*]]) #[[ATTR0]] {
-// CONSTRAINED-NEXT:  [[ENTRY:.*:]]
-// CONSTRAINED-NEXT:    [[TMP0:%.*]] = bitcast <8 x half> [[A]] to <8 x i16>
-// CONSTRAINED-NEXT:    [[TMP1:%.*]] = bitcast <8 x half> [[B]] to <8 x i16>
-// CONSTRAINED-NEXT:    [[TMP2:%.*]] = bitcast <4 x half> [[C]] to <4 x i16>
-// CONSTRAINED-NEXT:    [[TMP3:%.*]] = bitcast <8 x i16> [[TMP0]] to <16 x i8>
-// CONSTRAINED-NEXT:    [[TMP4:%.*]] = bitcast <8 x i16> [[TMP1]] to <16 x i8>
-// CONSTRAINED-NEXT:    [[TMP5:%.*]] = bitcast <4 x i16> [[TMP2]] to <8 x i8>
-// CONSTRAINED-NEXT:    [[TMP6:%.*]] = bitcast <8 x i8> [[TMP5]] to <4 x half>
-// CONSTRAINED-NEXT:    [[LANE:%.*]] = shufflevector <4 x half> [[TMP6]], <4 x 
half> [[TMP6]], <8 x i32> <i32 3, i32 3, i32 3, i32 3, i32 3, i32 3, i32 3, i32 
3>
-// CONSTRAINED-NEXT:    [[FMLA:%.*]] = bitcast <16 x i8> [[TMP4]] to <8 x half>
-// CONSTRAINED-NEXT:    [[FMLA1:%.*]] = bitcast <16 x i8> [[TMP3]] to <8 x 
half>
-// CONSTRAINED-NEXT:    [[FMLA2:%.*]] = call <8 x half> 
@llvm.experimental.constrained.fma.v8f16(<8 x half> [[FMLA]], <8 x half> 
[[LANE]], <8 x half> [[FMLA1]], metadata !"round.tonearest", metadata 
!"fpexcept.strict") #[[ATTR2]]
-// CONSTRAINED-NEXT:    ret <8 x half> [[FMLA2]]
-//
-float16x8_t test_vfmaq_lane_f16(float16x8_t a, float16x8_t b, float16x4_t c) {
-  return vfmaq_lane_f16(a, b, c, 3);
-}
-
-// UNCONSTRAINED-LABEL: define dso_local <4 x half> @test_vfma_laneq_f16(
-// UNCONSTRAINED-SAME: <4 x half> noundef [[A:%.*]], <4 x half> noundef 
[[B:%.*]], <8 x half> noundef [[C:%.*]]) #[[ATTR0]] {
-// UNCONSTRAINED-NEXT:  [[ENTRY:.*:]]
-// UNCONSTRAINED-NEXT:    [[TMP0:%.*]] = bitcast <4 x half> [[A]] to <4 x i16>
-// UNCONSTRAINED-NEXT:    [[TMP1:%.*]] = bitcast <4 x half> [[B]] to <4 x i16>
-// UNCONSTRAINED-NEXT:    [[TMP2:%.*]] = bitcast <8 x half> [[C]] to <8 x i16>
-// UNCONSTRAINED-NEXT:    [[TMP3:%.*]] = bitcast <4 x i16> [[TMP0]] to <8 x i8>
-// UNCONSTRAINED-NEXT:    [[TMP4:%.*]] = bitcast <4 x i16> [[TMP1]] to <8 x i8>
-// UNCONSTRAINED-NEXT:    [[TMP5:%.*]] = bitcast <8 x i16> [[TMP2]] to <16 x 
i8>
-// UNCONSTRAINED-NEXT:    [[TMP6:%.*]] = bitcast <8 x i8> [[TMP3]] to <4 x 
half>
-// UNCONSTRAINED-NEXT:    [[TMP7:%.*]] = bitcast <8 x i8> [[TMP4]] to <4 x 
half>
-// UNCONSTRAINED-NEXT:    [[TMP8:%.*]] = bitcast <16 x i8> [[TMP5]] to <8 x 
half>
-// UNCONSTRAINED-NEXT:    [[LANE:%.*]] = shufflevector <8 x half> [[TMP8]], <8 
x half> [[TMP8]], <4 x i32> <i32 7, i32 7, i32 7, i32 7>
-// UNCONSTRAINED-NEXT:    [[TMP9:%.*]] = call <4 x half> @llvm.fma.v4f16(<4 x 
half> [[LANE]], <4 x half> [[TMP7]], <4 x half> [[TMP6]])
-// UNCONSTRAINED-NEXT:    ret <4 x half> [[TMP9]]
-//
-// CONSTRAINED-LABEL: define dso_local <4 x half> @test_vfma_laneq_f16(
-// CONSTRAINED-SAME: <4 x half> noundef [[A:%.*]], <4 x half> noundef 
[[B:%.*]], <8 x half> noundef [[C:%.*]]) #[[ATTR0]] {
-// CONSTRAINED-NEXT:  [[ENTRY:.*:]]
-// CONSTRAINED-NEXT:    [[TMP0:%.*]] = bitcast <4 x half> [[A]] to <4 x i16>
-// CONSTRAINED-NEXT:    [[TMP1:%.*]] = bitcast <4 x half> [[B]] to <4 x i16>
-// CONSTRAINED-NEXT:    [[TMP2:%.*]] = bitcast <8 x half> [[C]] to <8 x i16>
-// CONSTRAINED-NEXT:    [[TMP3:%.*]] = bitcast <4 x i16> [[TMP0]] to <8 x i8>
-// CONSTRAINED-NEXT:    [[TMP4:%.*]] = bitcast <4 x i16> [[TMP1]] to <8 x i8>
-// CONSTRAINED-NEXT:    [[TMP5:%.*]] = bitcast <8 x i16> [[TMP2]] to <16 x i8>
-// CONSTRAINED-NEXT:    [[TMP6:%.*]] = bitcast <8 x i8> [[TMP3]] to <4 x half>
-// CONSTRAINED-NEXT:    [[TMP7:%.*]] = bitcast <8 x i8> [[TMP4]] to <4 x half>
-// CONSTRAINED-NEXT:    [[TMP8:%.*]] = bitcast <16 x i8> [[TMP5]] to <8 x half>
-// CONSTRAINED-NEXT:    [[LANE:%.*]] = shufflevector <8 x half> [[TMP8]], <8 x 
half> [[TMP8]], <4 x i32> <i32 7, i32 7, i32 7, i32 7>
-// CONSTRAINED-NEXT:    [[TMP9:%.*]] = call <4 x half> 
@llvm.experimental.constrained.fma.v4f16(<4 x half> [[LANE]], <4 x half> 
[[TMP7]], <4 x half> [[TMP6]], metadata !"round.tonearest", metadata 
!"fpexcept.strict") #[[ATTR2]]
-// CONSTRAINED-NEXT:    ret <4 x half> [[TMP9]]
-//
-float16x4_t test_vfma_laneq_f16(float16x4_t a, float16x4_t b, float16x8_t c) {
-  return vfma_laneq_f16(a, b, c, 7);
-}
-
-// UNCONSTRAINED-LABEL: define dso_local <8 x half> @test_vfmaq_laneq_f16(
-// UNCONSTRAINED-SAME: <8 x half> noundef [[A:%.*]], <8 x half> noundef 
[[B:%.*]], <8 x half> noundef [[C:%.*]]) #[[ATTR0]] {
-// UNCONSTRAINED-NEXT:  [[ENTRY:.*:]]
-// UNCONSTRAINED-NEXT:    [[TMP0:%.*]] = bitcast <8 x half> [[A]] to <8 x i16>
-// UNCONSTRAINED-NEXT:    [[TMP1:%.*]] = bitcast <8 x half> [[B]] to <8 x i16>
-// UNCONSTRAINED-NEXT:    [[TMP2:%.*]] = bitcast <8 x half> [[C]] to <8 x i16>
-// UNCONSTRAINED-NEXT:    [[TMP3:%.*]] = bitcast <8 x i16> [[TMP0]] to <16 x 
i8>
-// UNCONSTRAINED-NEXT:    [[TMP4:%.*]] = bitcast <8 x i16> [[TMP1]] to <16 x 
i8>
-// UNCONSTRAINED-NEXT:    [[TMP5:%.*]] = bitcast <8 x i16> [[TMP2]] to <16 x 
i8>
-// UNCONSTRAINED-NEXT:    [[TMP6:%.*]] = bitcast <16 x i8> [[TMP3]] to <8 x 
half>
-// UNCONSTRAINED-NEXT:    [[TMP7:%.*]] = bitcast <16 x i8> [[TMP4]] to <8 x 
half>
-// UNCONSTRAINED-NEXT:    [[TMP8:%.*]] = bitcast <16 x i8> [[TMP5]] to <8 x 
half>
-// UNCONSTRAINED-NEXT:    [[LANE:%.*]] = shufflevector <8 x half> [[TMP8]], <8 
x half> [[TMP8]], <8 x i32> <i32 7, i32 7, i32 7, i32 7, i32 7, i32 7, i32 7, 
i32 7>
-// UNCONSTRAINED-NEXT:    [[TMP9:%.*]] = call <8 x half> @llvm.fma.v8f16(<8 x 
half> [[LANE]], <8 x half> [[TMP7]], <8 x half> [[TMP6]])
-// UNCONSTRAINED-NEXT:    ret <8 x half> [[TMP9]]
-//
-// CONSTRAINED-LABEL: define dso_local <8 x half> @test_vfmaq_laneq_f16(
-// CONSTRAINED-SAME: <8 x half> noundef [[A:%.*]], <8 x half> noundef 
[[B:%.*]], <8 x half> noundef [[C:%.*]]) #[[ATTR0]] {
-// CONSTRAINED-NEXT:  [[ENTRY:.*:]]
-// CONSTRAINED-NEXT:    [[TMP0:%.*]] = bitcast <8 x half> [[A]] to <8 x i16>
-// CONSTRAINED-NEXT:    [[TMP1:%.*]] = bitcast <8 x half> [[B]] to <8 x i16>
-// CONSTRAINED-NEXT:    [[TMP2:%.*]] = bitcast <8 x half> [[C]] to <8 x i16>
-// CONSTRAINED-NEXT:    [[TMP3:%.*]] = bitcast <8 x i16> [[TMP0]] to <16 x i8>
-// CONSTRAINED-NEXT:    [[TMP4:%.*]] = bitcast <8 x i16> [[TMP1]] to <16 x i8>
-// CONSTRAINED-NEXT:    [[TMP5:%.*]] = bitcast <8 x i16> [[TMP2]] to <16 x i8>
-// CONSTRAINED-NEXT:    [[TMP6:%.*]] = bitcast <16 x i8> [[TMP3]] to <8 x half>
-// CONSTRAINED-NEXT:    [[TMP7:%.*]] = bitcast <16 x i8> [[TMP4]] to <8 x half>
-// CONSTRAINED-NEXT:    [[TMP8:%.*]] = bitcast <16 x i8> [[TMP5]] to <8 x half>
-// CONSTRAINED-NEXT:    [[LANE:%.*]] = shufflevector <8 x half> [[TMP8]], <8 x 
half> [[TMP8]], <8 x i32> <i32 7, i32 7, i32 7, i32 7, i32 7, i32 7, i32 7, i32 
7>
-// CONSTRAINED-NEXT:    [[TMP9:%.*]] = call <8 x half> 
@llvm.experimental.constrained.fma.v8f16(<8 x half> [[LANE]], <8 x half> 
[[TMP7]], <8 x half> [[TMP6]], metadata !"round.tonearest", metadata 
!"fpexcept.strict") #[[ATTR2]]
-// CONSTRAINED-NEXT:    ret <8 x half> [[TMP9]]
-//
-float16x8_t test_vfmaq_laneq_f16(float16x8_t a, float16x8_t b, float16x8_t c) {
-  return vfmaq_laneq_f16(a, b, c, 7);
-}
-
 // UNCONSTRAINED-LABEL: define dso_local <4 x half> @test_vfma_n_f16(
 // UNCONSTRAINED-SAME: <4 x half> noundef [[A:%.*]], <4 x half> noundef 
[[B:%.*]], half noundef [[C:%.*]]) #[[ATTR0]] {
 // UNCONSTRAINED-NEXT:  [[ENTRY:.*:]]
@@ -441,24 +185,6 @@ float16x8_t test_vfmaq_n_f16(float16x8_t a, float16x8_t b, 
float16_t c) {
   return vfmaq_n_f16(a, b, c);
 }
 
-// UNCONSTRAINED-LABEL: define dso_local half @test_vfmah_lane_f16(
-// UNCONSTRAINED-SAME: half noundef [[A:%.*]], half noundef [[B:%.*]], <4 x 
half> noundef [[C:%.*]]) #[[ATTR0]] {
-// UNCONSTRAINED-NEXT:  [[ENTRY:.*:]]
-// UNCONSTRAINED-NEXT:    [[EXTRACT:%.*]] = extractelement <4 x half> [[C]], 
i32 3
-// UNCONSTRAINED-NEXT:    [[TMP0:%.*]] = call half @llvm.fma.f16(half [[B]], 
half [[EXTRACT]], half [[A]])
-// UNCONSTRAINED-NEXT:    ret half [[TMP0]]
-//
-// CONSTRAINED-LABEL: define dso_local half @test_vfmah_lane_f16(
-// CONSTRAINED-SAME: half noundef [[A:%.*]], half noundef [[B:%.*]], <4 x 
half> noundef [[C:%.*]]) #[[ATTR0]] {
-// CONSTRAINED-NEXT:  [[ENTRY:.*:]]
-// CONSTRAINED-NEXT:    [[EXTRACT:%.*]] = extractelement <4 x half> [[C]], i32 
3
-// CONSTRAINED-NEXT:    [[TMP0:%.*]] = call half 
@llvm.experimental.constrained.fma.f16(half [[B]], half [[EXTRACT]], half 
[[A]], metadata !"round.tonearest", metadata !"fpexcept.strict") #[[ATTR2]]
-// CONSTRAINED-NEXT:    ret half [[TMP0]]
-//
-float16_t test_vfmah_lane_f16(float16_t a, float16_t b, float16x4_t c) {
-  return vfmah_lane_f16(a, b, c, 3);
-}
-
 // UNCONSTRAINED-LABEL: define dso_local half @test_vfmah_laneq_f16(
 // UNCONSTRAINED-SAME: half noundef [[A:%.*]], half noundef [[B:%.*]], <8 x 
half> noundef [[C:%.*]]) #[[ATTR0]] {
 // UNCONSTRAINED-NEXT:  [[ENTRY:.*:]]

_______________________________________________
cfe-commits mailing list
[email protected]
https://lists.llvm.org/cgi-bin/mailman/listinfo/cfe-commits

Reply via email to