https://github.com/yairbenavraham updated https://github.com/llvm/llvm-project/pull/218307
>From bf471dadf0140da8601ebbaadf387f0cfdcf19bf Mon Sep 17 00:00:00 2001 From: Yair Ben Avraham <[email protected]> Date: Mon, 5 Oct 2026 13:22:22 +0300 Subject: [PATCH] [CIR][AArch64] Handle constrained Neon FMA and sqrt Propagate expression FP options through AArch64 builtin emission and attach the active FP environment to CIR fma and sqrt operations. This allows strict FP operations to lower to constrained LLVM intrinsics without changing unconstrained behavior. Co-locate classic Clang, CIR, and CIR-to-LLVM coverage in focused -constrained.c Neon tests. Move matching classic constrained checks from legacy files and remove superseded blocks while retaining unrelated legacy tests. Track FMA operands through ABI conversions and lane selection, verify constrained results, and test FP16 pragma overrides of command-line maytrap settings. Assisted-by: Codex Follow-up to #213800 --- .../lib/CIR/CodeGen/CIRGenBuiltinAArch64.cpp | 21 +- .../AArch64/neon-intrinsics-constrained.c | 40 --- .../CodeGen/AArch64/neon-misc-constrained.c | 22 -- .../neon-scalar-x-indexed-elem-constrained.c | 67 +---- .../AArch64/neon/fullfp16-constrained.c | 33 +++ .../fused-multiple-fullfp16-constrained.c | 155 ++++++++++ .../AArch64/neon/fused-multiply-constrained.c | 91 ++++++ .../AArch64/neon/intrinsics-constrained.c | 56 ++++ .../v8.2a-fp16-intrinsics-constrained.c | 16 - .../v8.2a-neon-intrinsics-constrained.c | 280 +----------------- 10 files changed, 359 insertions(+), 422 deletions(-) create mode 100644 clang/test/CodeGen/AArch64/neon/fullfp16-constrained.c create mode 100644 clang/test/CodeGen/AArch64/neon/fused-multiple-fullfp16-constrained.c create mode 100644 clang/test/CodeGen/AArch64/neon/fused-multiply-constrained.c create mode 100644 clang/test/CodeGen/AArch64/neon/intrinsics-constrained.c diff --git a/clang/lib/CIR/CodeGen/CIRGenBuiltinAArch64.cpp b/clang/lib/CIR/CodeGen/CIRGenBuiltinAArch64.cpp index 70253221c7b7d5..af0c3c6b1c2d81 100644 --- a/clang/lib/CIR/CodeGen/CIRGenBuiltinAArch64.cpp +++ b/clang/lib/CIR/CodeGen/CIRGenBuiltinAArch64.cpp @@ -204,12 +204,26 @@ emitNeonCallToOp(CIRGenModule &cgm, CIRGenBuilderTy &builder, builder.getStringAttr(intrinsicName.value()), funcResTy, args) .getResult(); - } else { + } + if constexpr (std::is_same_v<Operation, cir::FMAOp>) { + assert(args.size() == 3 && "fma expects three operands"); + return Operation::create(builder, loc, funcResTy, args[0], args[1], args[2], + builder.getConstrainedFPAttr()) + .getResult(); + } + if constexpr (std::is_same_v<Operation, cir::SqrtOp>) { + assert(args.size() == 1 && "sqrt expects one operand"); + return Operation::create(builder, loc, funcResTy, args[0], + builder.getConstrainedFPAttr()) + .getResult(); + } + if constexpr (!std::is_same_v<Operation, cir::LLVMIntrinsicCallOp> && + !std::is_same_v<Operation, cir::FMAOp> && + !std::is_same_v<Operation, cir::SqrtOp>) return Operation::create( builder, loc, mlir::TypeRange{funcResTy}, args, cir::getDefaultProperties<Operation>(builder.getContext())) .getResult(); - } } // TODO(cir): Remove `cgm` from the list of arguments once all NYI(s) are gone. @@ -2709,6 +2723,8 @@ CIRGenFunction::emitAArch64BuiltinExpr(unsigned builtinID, const CallExpr *expr, // evaluation. assert(!cir::MissingFeatures::msvcBuiltins()); + CIRGenFPOptionsRAII fpOptsRAII(*this, expr); + // Some intrinsics are equivalent - if they are use the base intrinsic ID. auto it = llvm::find_if(neonEquivalentIntrinsicMap, [builtinID](auto &p) { return p.first == builtinID; @@ -3630,7 +3646,6 @@ CIRGenFunction::emitAArch64BuiltinExpr(unsigned builtinID, const CallExpr *expr, } case NEON::BI__builtin_neon_vsqrt_v: case NEON::BI__builtin_neon_vsqrtq_v: - assert(!cir::MissingFeatures::emitConstrainedFPCall()); return emitNeonCallToOp<cir::SqrtOp>(cgm, builder, {ty}, ops, std::nullopt, ty, loc); case NEON::BI__builtin_neon_vrbit_v: diff --git a/clang/test/CodeGen/AArch64/neon-intrinsics-constrained.c b/clang/test/CodeGen/AArch64/neon-intrinsics-constrained.c index 50a2a629ea55b3..09f2aec4176b42 100644 --- a/clang/test/CodeGen/AArch64/neon-intrinsics-constrained.c +++ b/clang/test/CodeGen/AArch64/neon-intrinsics-constrained.c @@ -1418,46 +1418,6 @@ float64x1_t test_vmls_f64(float64x1_t a, float64x1_t b, float64x1_t c) { return vmls_f64(a, b, c); } -// UNCONSTRAINED-LABEL: define dso_local <1 x double> @test_vfma_f64( -// UNCONSTRAINED-SAME: <1 x double> noundef [[A:%.*]], <1 x double> noundef [[B:%.*]], <1 x double> noundef [[C:%.*]]) #[[ATTR0]] { -// UNCONSTRAINED-NEXT: [[ENTRY:.*:]] -// UNCONSTRAINED-NEXT: [[TMP0:%.*]] = bitcast <1 x double> [[A]] to i64 -// UNCONSTRAINED-NEXT: [[__P0_ADDR_I_SROA_0_0_VEC_INSERT:%.*]] = insertelement <1 x i64> undef, i64 [[TMP0]], i64 0 -// UNCONSTRAINED-NEXT: [[TMP1:%.*]] = bitcast <1 x double> [[B]] to i64 -// UNCONSTRAINED-NEXT: [[__P1_ADDR_I_SROA_0_0_VEC_INSERT:%.*]] = insertelement <1 x i64> undef, i64 [[TMP1]], i64 0 -// UNCONSTRAINED-NEXT: [[TMP2:%.*]] = bitcast <1 x double> [[C]] to i64 -// UNCONSTRAINED-NEXT: [[__P2_ADDR_I_SROA_0_0_VEC_INSERT:%.*]] = insertelement <1 x i64> undef, i64 [[TMP2]], i64 0 -// UNCONSTRAINED-NEXT: [[TMP3:%.*]] = bitcast <1 x i64> [[__P0_ADDR_I_SROA_0_0_VEC_INSERT]] to <8 x i8> -// UNCONSTRAINED-NEXT: [[TMP4:%.*]] = bitcast <1 x i64> [[__P1_ADDR_I_SROA_0_0_VEC_INSERT]] to <8 x i8> -// UNCONSTRAINED-NEXT: [[TMP5:%.*]] = bitcast <1 x i64> [[__P2_ADDR_I_SROA_0_0_VEC_INSERT]] to <8 x i8> -// UNCONSTRAINED-NEXT: [[TMP6:%.*]] = bitcast <8 x i8> [[TMP3]] to <1 x double> -// UNCONSTRAINED-NEXT: [[TMP7:%.*]] = bitcast <8 x i8> [[TMP4]] to <1 x double> -// UNCONSTRAINED-NEXT: [[TMP8:%.*]] = bitcast <8 x i8> [[TMP5]] to <1 x double> -// UNCONSTRAINED-NEXT: [[TMP9:%.*]] = call <1 x double> @llvm.fma.v1f64(<1 x double> [[TMP7]], <1 x double> [[TMP8]], <1 x double> [[TMP6]]) -// UNCONSTRAINED-NEXT: ret <1 x double> [[TMP9]] -// -// CONSTRAINED-LABEL: define dso_local <1 x double> @test_vfma_f64( -// CONSTRAINED-SAME: <1 x double> noundef [[A:%.*]], <1 x double> noundef [[B:%.*]], <1 x double> noundef [[C:%.*]]) #[[ATTR0]] { -// CONSTRAINED-NEXT: [[ENTRY:.*:]] -// CONSTRAINED-NEXT: [[TMP0:%.*]] = bitcast <1 x double> [[A]] to i64 -// CONSTRAINED-NEXT: [[__P0_ADDR_I_SROA_0_0_VEC_INSERT:%.*]] = insertelement <1 x i64> undef, i64 [[TMP0]], i64 0 -// CONSTRAINED-NEXT: [[TMP1:%.*]] = bitcast <1 x double> [[B]] to i64 -// CONSTRAINED-NEXT: [[__P1_ADDR_I_SROA_0_0_VEC_INSERT:%.*]] = insertelement <1 x i64> undef, i64 [[TMP1]], i64 0 -// CONSTRAINED-NEXT: [[TMP2:%.*]] = bitcast <1 x double> [[C]] to i64 -// CONSTRAINED-NEXT: [[__P2_ADDR_I_SROA_0_0_VEC_INSERT:%.*]] = insertelement <1 x i64> undef, i64 [[TMP2]], i64 0 -// CONSTRAINED-NEXT: [[TMP3:%.*]] = bitcast <1 x i64> [[__P0_ADDR_I_SROA_0_0_VEC_INSERT]] to <8 x i8> -// CONSTRAINED-NEXT: [[TMP4:%.*]] = bitcast <1 x i64> [[__P1_ADDR_I_SROA_0_0_VEC_INSERT]] to <8 x i8> -// CONSTRAINED-NEXT: [[TMP5:%.*]] = bitcast <1 x i64> [[__P2_ADDR_I_SROA_0_0_VEC_INSERT]] to <8 x i8> -// CONSTRAINED-NEXT: [[TMP6:%.*]] = bitcast <8 x i8> [[TMP3]] to <1 x double> -// CONSTRAINED-NEXT: [[TMP7:%.*]] = bitcast <8 x i8> [[TMP4]] to <1 x double> -// CONSTRAINED-NEXT: [[TMP8:%.*]] = bitcast <8 x i8> [[TMP5]] to <1 x double> -// CONSTRAINED-NEXT: [[TMP9:%.*]] = call <1 x double> @llvm.experimental.constrained.fma.v1f64(<1 x double> [[TMP7]], <1 x double> [[TMP8]], <1 x double> [[TMP6]], metadata !"round.tonearest", metadata !"fpexcept.strict") #[[ATTR3]] -// CONSTRAINED-NEXT: ret <1 x double> [[TMP9]] -// -float64x1_t test_vfma_f64(float64x1_t a, float64x1_t b, float64x1_t c) { - return vfma_f64(a, b, c); -} - // UNCONSTRAINED-LABEL: define dso_local <1 x double> @test_vfms_f64( // UNCONSTRAINED-SAME: <1 x double> noundef [[A:%.*]], <1 x double> noundef [[B:%.*]], <1 x double> noundef [[C:%.*]]) #[[ATTR0]] { // UNCONSTRAINED-NEXT: [[ENTRY:.*:]] diff --git a/clang/test/CodeGen/AArch64/neon-misc-constrained.c b/clang/test/CodeGen/AArch64/neon-misc-constrained.c index 49208892e3035b..f1f638cb272c62 100644 --- a/clang/test/CodeGen/AArch64/neon-misc-constrained.c +++ b/clang/test/CodeGen/AArch64/neon-misc-constrained.c @@ -82,28 +82,6 @@ float32x4_t test_vsqrtq_f32(float32x4_t a) { } -// UNCONSTRAINED-LABEL: define dso_local <2 x double> @test_vsqrtq_f64( -// UNCONSTRAINED-SAME: <2 x double> noundef [[A:%.*]]) #[[ATTR0]] { -// UNCONSTRAINED-NEXT: [[ENTRY:.*:]] -// UNCONSTRAINED-NEXT: [[TMP0:%.*]] = bitcast <2 x double> [[A]] to <2 x i64> -// UNCONSTRAINED-NEXT: [[TMP1:%.*]] = bitcast <2 x i64> [[TMP0]] to <16 x i8> -// UNCONSTRAINED-NEXT: [[TMP2:%.*]] = bitcast <16 x i8> [[TMP1]] to <2 x double> -// UNCONSTRAINED-NEXT: [[VSQRT_I:%.*]] = call <2 x double> @llvm.sqrt.v2f64(<2 x double> [[TMP2]]) -// UNCONSTRAINED-NEXT: ret <2 x double> [[VSQRT_I]] -// -// CONSTRAINED-LABEL: define dso_local <2 x double> @test_vsqrtq_f64( -// CONSTRAINED-SAME: <2 x double> noundef [[A:%.*]]) #[[ATTR0]] { -// CONSTRAINED-NEXT: [[ENTRY:.*:]] -// CONSTRAINED-NEXT: [[TMP0:%.*]] = bitcast <2 x double> [[A]] to <2 x i64> -// CONSTRAINED-NEXT: [[TMP1:%.*]] = bitcast <2 x i64> [[TMP0]] to <16 x i8> -// CONSTRAINED-NEXT: [[TMP2:%.*]] = bitcast <16 x i8> [[TMP1]] to <2 x double> -// CONSTRAINED-NEXT: [[VSQRT_I:%.*]] = call <2 x double> @llvm.experimental.constrained.sqrt.v2f64(<2 x double> [[TMP2]], metadata !"round.tonearest", metadata !"fpexcept.strict") #[[ATTR2]] -// CONSTRAINED-NEXT: ret <2 x double> [[VSQRT_I]] -// -float64x2_t test_vsqrtq_f64(float64x2_t a) { - return vsqrtq_f64(a); -} - // UNCONSTRAINED-LABEL: define dso_local <4 x half> @test_vcvt_f16_f32( // UNCONSTRAINED-SAME: <4 x float> noundef [[A:%.*]]) #[[ATTR0]] { // UNCONSTRAINED-NEXT: [[ENTRY:.*:]] diff --git a/clang/test/CodeGen/AArch64/neon-scalar-x-indexed-elem-constrained.c b/clang/test/CodeGen/AArch64/neon-scalar-x-indexed-elem-constrained.c index 944929ccb5f428..046bca2420a37a 100644 --- a/clang/test/CodeGen/AArch64/neon-scalar-x-indexed-elem-constrained.c +++ b/clang/test/CodeGen/AArch64/neon-scalar-x-indexed-elem-constrained.c @@ -13,36 +13,18 @@ #include <arm_neon.h> -// UNCONSTRAINED-LABEL: define dso_local float @test_vfmas_lane_f32( -// UNCONSTRAINED-SAME: float noundef [[A:%.*]], float noundef [[B:%.*]], <2 x float> noundef [[C:%.*]]) #[[ATTR0:[0-9]+]] { -// UNCONSTRAINED-NEXT: [[ENTRY:.*:]] -// UNCONSTRAINED-NEXT: [[EXTRACT:%.*]] = extractelement <2 x float> [[C]], i32 1 -// UNCONSTRAINED-NEXT: [[TMP0:%.*]] = call float @llvm.fma.f32(float [[B]], float [[EXTRACT]], float [[A]]) -// UNCONSTRAINED-NEXT: ret float [[TMP0]] -// -// CONSTRAINED-LABEL: define dso_local float @test_vfmas_lane_f32( -// CONSTRAINED-SAME: float noundef [[A:%.*]], float noundef [[B:%.*]], <2 x float> noundef [[C:%.*]]) #[[ATTR0:[0-9]+]] { -// CONSTRAINED-NEXT: [[ENTRY:.*:]] -// CONSTRAINED-NEXT: [[EXTRACT:%.*]] = extractelement <2 x float> [[C]], i32 1 -// CONSTRAINED-NEXT: [[TMP0:%.*]] = call float @llvm.experimental.constrained.fma.f32(float [[B]], float [[EXTRACT]], float [[A]], metadata !"round.tonearest", metadata !"fpexcept.strict") #[[ATTR2:[0-9]+]] -// CONSTRAINED-NEXT: ret float [[TMP0]] -// -float32_t test_vfmas_lane_f32(float32_t a, float32_t b, float32x2_t c) { - return vfmas_lane_f32(a, b, c, 1); -} - // UNCONSTRAINED-LABEL: define dso_local double @test_vfmad_lane_f64( -// UNCONSTRAINED-SAME: double noundef [[A:%.*]], double noundef [[B:%.*]], <1 x double> noundef [[C:%.*]]) #[[ATTR0]] { +// UNCONSTRAINED-SAME: double noundef [[A:%.*]], double noundef [[B:%.*]], <1 x double> noundef [[C:%.*]]) #[[ATTR0:[0-9]+]] { // UNCONSTRAINED-NEXT: [[ENTRY:.*:]] // UNCONSTRAINED-NEXT: [[EXTRACT:%.*]] = extractelement <1 x double> [[C]], i32 0 // UNCONSTRAINED-NEXT: [[TMP0:%.*]] = call double @llvm.fma.f64(double [[B]], double [[EXTRACT]], double [[A]]) // UNCONSTRAINED-NEXT: ret double [[TMP0]] // // CONSTRAINED-LABEL: define dso_local double @test_vfmad_lane_f64( -// CONSTRAINED-SAME: double noundef [[A:%.*]], double noundef [[B:%.*]], <1 x double> noundef [[C:%.*]]) #[[ATTR0]] { +// CONSTRAINED-SAME: double noundef [[A:%.*]], double noundef [[B:%.*]], <1 x double> noundef [[C:%.*]]) #[[ATTR0:[0-9]+]] { // CONSTRAINED-NEXT: [[ENTRY:.*:]] // CONSTRAINED-NEXT: [[EXTRACT:%.*]] = extractelement <1 x double> [[C]], i32 0 -// CONSTRAINED-NEXT: [[TMP0:%.*]] = call double @llvm.experimental.constrained.fma.f64(double [[B]], double [[EXTRACT]], double [[A]], metadata !"round.tonearest", metadata !"fpexcept.strict") #[[ATTR2]] +// CONSTRAINED-NEXT: [[TMP0:%.*]] = call double @llvm.experimental.constrained.fma.f64(double [[B]], double [[EXTRACT]], double [[A]], metadata !"round.tonearest", metadata !"fpexcept.strict") #[[ATTR2:[0-9]+]] // CONSTRAINED-NEXT: ret double [[TMP0]] // float64_t test_vfmad_lane_f64(float64_t a, float64_t b, float64x1_t c) { @@ -173,48 +155,6 @@ float64x1_t test_vfms_lane_f64(float64x1_t a, float64x1_t b, float64x1_t v) { return vfms_lane_f64(a, b, v, 0); } -// UNCONSTRAINED-LABEL: define dso_local <1 x double> @test_vfma_laneq_f64( -// UNCONSTRAINED-SAME: <1 x double> noundef [[A:%.*]], <1 x double> noundef [[B:%.*]], <2 x double> noundef [[V:%.*]]) #[[ATTR0]] { -// UNCONSTRAINED-NEXT: [[ENTRY:.*:]] -// UNCONSTRAINED-NEXT: [[TMP0:%.*]] = bitcast <1 x double> [[A]] to i64 -// UNCONSTRAINED-NEXT: [[__S0_SROA_0_0_VEC_INSERT:%.*]] = insertelement <1 x i64> undef, i64 [[TMP0]], i64 0 -// UNCONSTRAINED-NEXT: [[TMP1:%.*]] = bitcast <1 x double> [[B]] to i64 -// UNCONSTRAINED-NEXT: [[__S1_SROA_0_0_VEC_INSERT:%.*]] = insertelement <1 x i64> undef, i64 [[TMP1]], i64 0 -// UNCONSTRAINED-NEXT: [[TMP2:%.*]] = bitcast <2 x double> [[V]] to <2 x i64> -// UNCONSTRAINED-NEXT: [[TMP3:%.*]] = bitcast <1 x i64> [[__S0_SROA_0_0_VEC_INSERT]] to <8 x i8> -// UNCONSTRAINED-NEXT: [[TMP4:%.*]] = bitcast <1 x i64> [[__S1_SROA_0_0_VEC_INSERT]] to <8 x i8> -// UNCONSTRAINED-NEXT: [[TMP5:%.*]] = bitcast <2 x i64> [[TMP2]] to <16 x i8> -// UNCONSTRAINED-NEXT: [[TMP6:%.*]] = bitcast <8 x i8> [[TMP3]] to double -// UNCONSTRAINED-NEXT: [[TMP7:%.*]] = bitcast <8 x i8> [[TMP4]] to double -// UNCONSTRAINED-NEXT: [[TMP8:%.*]] = bitcast <16 x i8> [[TMP5]] to <2 x double> -// UNCONSTRAINED-NEXT: [[EXTRACT:%.*]] = extractelement <2 x double> [[TMP8]], i32 0 -// UNCONSTRAINED-NEXT: [[TMP9:%.*]] = call double @llvm.fma.f64(double [[TMP7]], double [[EXTRACT]], double [[TMP6]]) -// UNCONSTRAINED-NEXT: [[TMP10:%.*]] = bitcast double [[TMP9]] to <1 x double> -// UNCONSTRAINED-NEXT: ret <1 x double> [[TMP10]] -// -// CONSTRAINED-LABEL: define dso_local <1 x double> @test_vfma_laneq_f64( -// CONSTRAINED-SAME: <1 x double> noundef [[A:%.*]], <1 x double> noundef [[B:%.*]], <2 x double> noundef [[V:%.*]]) #[[ATTR0]] { -// CONSTRAINED-NEXT: [[ENTRY:.*:]] -// CONSTRAINED-NEXT: [[TMP0:%.*]] = bitcast <1 x double> [[A]] to i64 -// CONSTRAINED-NEXT: [[__S0_SROA_0_0_VEC_INSERT:%.*]] = insertelement <1 x i64> undef, i64 [[TMP0]], i64 0 -// CONSTRAINED-NEXT: [[TMP1:%.*]] = bitcast <1 x double> [[B]] to i64 -// CONSTRAINED-NEXT: [[__S1_SROA_0_0_VEC_INSERT:%.*]] = insertelement <1 x i64> undef, i64 [[TMP1]], i64 0 -// CONSTRAINED-NEXT: [[TMP2:%.*]] = bitcast <2 x double> [[V]] to <2 x i64> -// CONSTRAINED-NEXT: [[TMP3:%.*]] = bitcast <1 x i64> [[__S0_SROA_0_0_VEC_INSERT]] to <8 x i8> -// CONSTRAINED-NEXT: [[TMP4:%.*]] = bitcast <1 x i64> [[__S1_SROA_0_0_VEC_INSERT]] to <8 x i8> -// CONSTRAINED-NEXT: [[TMP5:%.*]] = bitcast <2 x i64> [[TMP2]] to <16 x i8> -// CONSTRAINED-NEXT: [[TMP6:%.*]] = bitcast <8 x i8> [[TMP3]] to double -// CONSTRAINED-NEXT: [[TMP7:%.*]] = bitcast <8 x i8> [[TMP4]] to double -// CONSTRAINED-NEXT: [[TMP8:%.*]] = bitcast <16 x i8> [[TMP5]] to <2 x double> -// CONSTRAINED-NEXT: [[EXTRACT:%.*]] = extractelement <2 x double> [[TMP8]], i32 0 -// CONSTRAINED-NEXT: [[TMP9:%.*]] = call double @llvm.experimental.constrained.fma.f64(double [[TMP7]], double [[EXTRACT]], double [[TMP6]], metadata !"round.tonearest", metadata !"fpexcept.strict") #[[ATTR2]] -// CONSTRAINED-NEXT: [[TMP10:%.*]] = bitcast double [[TMP9]] to <1 x double> -// CONSTRAINED-NEXT: ret <1 x double> [[TMP10]] -// -float64x1_t test_vfma_laneq_f64(float64x1_t a, float64x1_t b, float64x2_t v) { - return vfma_laneq_f64(a, b, v, 0); -} - // UNCONSTRAINED-LABEL: define dso_local <1 x double> @test_vfms_laneq_f64( // UNCONSTRAINED-SAME: <1 x double> noundef [[A:%.*]], <1 x double> noundef [[B:%.*]], <2 x double> noundef [[V:%.*]]) #[[ATTR0]] { // UNCONSTRAINED-NEXT: [[ENTRY:.*:]] @@ -258,4 +198,3 @@ float64x1_t test_vfma_laneq_f64(float64x1_t a, float64x1_t b, float64x2_t v) { float64x1_t test_vfms_laneq_f64(float64x1_t a, float64x1_t b, float64x2_t v) { return vfms_laneq_f64(a, b, v, 0); } - diff --git a/clang/test/CodeGen/AArch64/neon/fullfp16-constrained.c b/clang/test/CodeGen/AArch64/neon/fullfp16-constrained.c new file mode 100644 index 00000000000000..1a668bd657e0de --- /dev/null +++ b/clang/test/CodeGen/AArch64/neon/fullfp16-constrained.c @@ -0,0 +1,33 @@ +// REQUIRES: aarch64-registered-target + +// RUN: %clang_cc1_cg_arm64_neon -target-feature +fullfp16 -ffp-exception-behavior=strict -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,simplifycfg | FileCheck %s --check-prefixes=ALL,LLVM --implicit-check-not=' @llvm.fma.' --implicit-check-not=' @llvm.sqrt.' +// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 -fexperimental-strict-floating-point -ffp-exception-behavior=maytrap -DEXCEPT=1 -fclangir -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,simplifycfg | FileCheck %s --check-prefixes=ALL,LLVM --implicit-check-not=fpexcept.maytrap --implicit-check-not=' @llvm.fma.' --implicit-check-not=' @llvm.sqrt.' %} +// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 -fexperimental-strict-floating-point -ffp-exception-behavior=maytrap -DEXCEPT=1 -fclangir -emit-cir %s -disable-O0-optnone | FileCheck %s --check-prefixes=ALL,CIR --implicit-check-not='except_mode = maytrap' --implicit-check-not='cir.call_llvm_intrinsic "fma"' --implicit-check-not='cir.call_llvm_intrinsic "sqrt"' %} +// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 -fexperimental-strict-floating-point -ffp-exception-behavior=strict -fclangir -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,simplifycfg | FileCheck %s --check-prefixes=ALL,LLVM --implicit-check-not=' @llvm.fma.' --implicit-check-not=' @llvm.sqrt.' %} +// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 -fexperimental-strict-floating-point -ffp-exception-behavior=strict -fclangir -emit-cir %s -disable-O0-optnone | FileCheck %s --check-prefixes=ALL,CIR --implicit-check-not='cir.call_llvm_intrinsic "fma"' --implicit-check-not='cir.call_llvm_intrinsic "sqrt"' %} + +#if EXCEPT +#pragma float_control(except, on) +#endif + +#include <arm_fp16.h> + +// ALL-LABEL: @test_vsqrth_f16( +float16_t test_vsqrth_f16(float16_t a) { +// CIR: cir.sqrt %{{.*}} : !cir.f16 fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true> + +// LLVM-SAME: half {{.*}} [[A:%.*]]) {{.*}} { +// LLVM: [[SQRT:%.*]] = call half @llvm.experimental.constrained.sqrt.f16(half [[A]], metadata !"round.tonearest", metadata !"fpexcept.strict") +// LLVM: ret half [[SQRT]] + return vsqrth_f16(a); +} + +// ALL-LABEL: @test_vfmah_f16( +float16_t test_vfmah_f16(float16_t a, float16_t b, float16_t c) { +// CIR: cir.fma %{{.*}}, %{{.*}}, %{{.*}} : !cir.f16 fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true> + +// LLVM-SAME: half {{.*}} [[A:%.*]], half {{.*}} [[B:%.*]], half {{.*}} [[C:%.*]]) {{.*}} { +// LLVM: [[FMA:%.*]] = call half @llvm.experimental.constrained.fma.f16(half [[B]], half [[C]], half [[A]], metadata !"round.tonearest", metadata !"fpexcept.strict") +// LLVM: ret half [[FMA]] + return vfmah_f16(a, b, c); +} diff --git a/clang/test/CodeGen/AArch64/neon/fused-multiple-fullfp16-constrained.c b/clang/test/CodeGen/AArch64/neon/fused-multiple-fullfp16-constrained.c new file mode 100644 index 00000000000000..a56cb34014f15e --- /dev/null +++ b/clang/test/CodeGen/AArch64/neon/fused-multiple-fullfp16-constrained.c @@ -0,0 +1,155 @@ +// REQUIRES: aarch64-registered-target + +// RUN: %clang_cc1_cg_arm64_neon -target-feature +fullfp16 -ffp-exception-behavior=maytrap -DEXCEPT=1 -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,sroa | FileCheck %s --check-prefixes=ALL,LLVM --implicit-check-not=fpexcept.maytrap --implicit-check-not=' @llvm.fma.' +// RUN: %clang_cc1_cg_arm64_neon -target-feature +fullfp16 -ffp-exception-behavior=strict -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,sroa | FileCheck %s --check-prefixes=ALL,LLVM --implicit-check-not=' @llvm.fma.' +// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 -fexperimental-strict-floating-point -ffp-exception-behavior=maytrap -DEXCEPT=1 -fclangir -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,sroa | FileCheck %s --check-prefixes=ALL,LLVM --implicit-check-not=fpexcept.maytrap --implicit-check-not=' @llvm.fma.' %} +// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 -fexperimental-strict-floating-point -ffp-exception-behavior=maytrap -DEXCEPT=1 -fclangir -emit-cir %s -disable-O0-optnone | FileCheck %s --check-prefixes=ALL,CIR --implicit-check-not='except_mode = maytrap' --implicit-check-not='cir.call_llvm_intrinsic "fma"' %} +// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 -fexperimental-strict-floating-point -ffp-exception-behavior=strict -fclangir -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,sroa | FileCheck %s --check-prefixes=ALL,LLVM --implicit-check-not=' @llvm.fma.' %} +// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 -fexperimental-strict-floating-point -ffp-exception-behavior=strict -fclangir -emit-cir %s -disable-O0-optnone | FileCheck %s --check-prefixes=ALL,CIR --implicit-check-not='cir.call_llvm_intrinsic "fma"' %} + +#if EXCEPT +#pragma float_control(except, on) +#endif + +#include <arm_neon.h> + +// LLVM-LABEL: @test_vfma_f16( +// CIR-LABEL: @vfma_f16( +float16x4_t test_vfma_f16(float16x4_t a, float16x4_t b, float16x4_t c) { +// CIR: cir.fma %{{.*}}, %{{.*}}, %{{.*}} : !cir.vector<4 x !cir.f16> fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true> + +// LLVM-SAME: <4 x half> {{.*}} [[A:%.*]], <4 x half> {{.*}} [[B:%.*]], <4 x half> {{.*}} [[C:%.*]]) {{.*}} { +// LLVM: [[A_I:%.*]] = bitcast <4 x half> [[A]] to <4 x i16> +// LLVM: [[B_I:%.*]] = bitcast <4 x half> [[B]] to <4 x i16> +// LLVM: [[C_I:%.*]] = bitcast <4 x half> [[C]] to <4 x i16> +// LLVM: [[A_BYTES:%.*]] = bitcast <4 x i16> [[A_I]] to <8 x i8> +// LLVM: [[B_BYTES:%.*]] = bitcast <4 x i16> [[B_I]] to <8 x i8> +// LLVM: [[C_BYTES:%.*]] = bitcast <4 x i16> [[C_I]] to <8 x i8> +// LLVM: [[A_CAST:%.*]] = bitcast <8 x i8> [[A_BYTES]] to <4 x half> +// LLVM: [[B_CAST:%.*]] = bitcast <8 x i8> [[B_BYTES]] to <4 x half> +// LLVM: [[C_CAST:%.*]] = bitcast <8 x i8> [[C_BYTES]] to <4 x half> +// LLVM: [[FMA:%.*]] = call <4 x half> @llvm.experimental.constrained.fma.v4f16(<4 x half> [[B_CAST]], <4 x half> [[C_CAST]], <4 x half> [[A_CAST]], metadata !"round.tonearest", metadata !"fpexcept.strict") +// LLVM: ret <4 x half> [[FMA]] + return vfma_f16(a, b, c); +} + +// LLVM-LABEL: @test_vfmaq_f16( +// CIR-LABEL: @vfmaq_f16( +float16x8_t test_vfmaq_f16(float16x8_t a, float16x8_t b, float16x8_t c) { +// CIR: cir.fma %{{.*}}, %{{.*}}, %{{.*}} : !cir.vector<8 x !cir.f16> fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true> + +// LLVM-SAME: <8 x half> {{.*}} [[A:%.*]], <8 x half> {{.*}} [[B:%.*]], <8 x half> {{.*}} [[C:%.*]]) {{.*}} { +// LLVM: [[A_I:%.*]] = bitcast <8 x half> [[A]] to <8 x i16> +// LLVM: [[B_I:%.*]] = bitcast <8 x half> [[B]] to <8 x i16> +// LLVM: [[C_I:%.*]] = bitcast <8 x half> [[C]] to <8 x i16> +// LLVM: [[A_BYTES:%.*]] = bitcast <8 x i16> [[A_I]] to <16 x i8> +// LLVM: [[B_BYTES:%.*]] = bitcast <8 x i16> [[B_I]] to <16 x i8> +// LLVM: [[C_BYTES:%.*]] = bitcast <8 x i16> [[C_I]] to <16 x i8> +// LLVM: [[A_CAST:%.*]] = bitcast <16 x i8> [[A_BYTES]] to <8 x half> +// LLVM: [[B_CAST:%.*]] = bitcast <16 x i8> [[B_BYTES]] to <8 x half> +// LLVM: [[C_CAST:%.*]] = bitcast <16 x i8> [[C_BYTES]] to <8 x half> +// LLVM: [[FMA:%.*]] = call <8 x half> @llvm.experimental.constrained.fma.v8f16(<8 x half> [[B_CAST]], <8 x half> [[C_CAST]], <8 x half> [[A_CAST]], metadata !"round.tonearest", metadata !"fpexcept.strict") +// LLVM: ret <8 x half> [[FMA]] + return vfmaq_f16(a, b, c); +} + +// ALL-LABEL: @test_vfma_lane_f16( +float16x4_t test_vfma_lane_f16(float16x4_t a, float16x4_t b, + float16x4_t c) { +// CIR: [[LANE:%.*]] = cir.vec.shuffle(%{{.*}}, %{{.*}} : !cir.vector<4 x !cir.f16>) [#cir.int<3> : !s32i, #cir.int<3> : !s32i, #cir.int<3> : !s32i, #cir.int<3> : !s32i] : !cir.vector<4 x !cir.f16> +// CIR: cir.fma %{{.*}}, [[LANE]], %{{.*}} : !cir.vector<4 x !cir.f16> fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true> + +// LLVM-SAME: <4 x half> {{.*}} [[A:%.*]], <4 x half> {{.*}} [[B:%.*]], <4 x half> {{.*}} [[C:%.*]]) {{.*}} { +// LLVM: [[A_I:%.*]] = bitcast <4 x half> [[A]] to <4 x i16> +// LLVM: [[B_I:%.*]] = bitcast <4 x half> [[B]] to <4 x i16> +// LLVM: [[C_I:%.*]] = bitcast <4 x half> [[C]] to <4 x i16> +// LLVM: [[A_BYTES:%.*]] = bitcast <4 x i16> [[A_I]] to <8 x i8> +// LLVM: [[B_BYTES:%.*]] = bitcast <4 x i16> [[B_I]] to <8 x i8> +// LLVM: [[C_BYTES:%.*]] = bitcast <4 x i16> [[C_I]] to <8 x i8> +// LLVM-DAG: [[A_CAST:%.*]] = bitcast <8 x i8> [[A_BYTES]] to <4 x half> +// LLVM-DAG: [[B_CAST:%.*]] = bitcast <8 x i8> [[B_BYTES]] to <4 x half> +// LLVM-DAG: [[C_CAST:%.*]] = bitcast <8 x i8> [[C_BYTES]] to <4 x half> +// LLVM-DAG: [[LANE:%.*]] = shufflevector <4 x half> [[C_CAST]], <4 x half> {{.*}}, <4 x i32> <i32 3, i32 3, i32 3, i32 3> +// LLVM: [[FMA:%.*]] = call <4 x half> @llvm.experimental.constrained.fma.v4f16(<4 x half> [[B_CAST]], <4 x half> [[LANE]], <4 x half> [[A_CAST]], metadata !"round.tonearest", metadata !"fpexcept.strict") +// LLVM: ret <4 x half> [[FMA]] + return vfma_lane_f16(a, b, c, 3); +} + +// ALL-LABEL: @test_vfmaq_lane_f16( +float16x8_t test_vfmaq_lane_f16(float16x8_t a, float16x8_t b, + float16x4_t c) { +// CIR: [[LANE:%.*]] = cir.vec.shuffle(%{{.*}}, %{{.*}} : !cir.vector<4 x !cir.f16>) [#cir.int<3> : !s32i, #cir.int<3> : !s32i, #cir.int<3> : !s32i, #cir.int<3> : !s32i, #cir.int<3> : !s32i, #cir.int<3> : !s32i, #cir.int<3> : !s32i, #cir.int<3> : !s32i] : !cir.vector<8 x !cir.f16> +// CIR: cir.fma %{{.*}}, [[LANE]], %{{.*}} : !cir.vector<8 x !cir.f16> fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true> + +// LLVM-SAME: <8 x half> {{.*}} [[A:%.*]], <8 x half> {{.*}} [[B:%.*]], <4 x half> {{.*}} [[C:%.*]]) {{.*}} { +// LLVM: [[A_I:%.*]] = bitcast <8 x half> [[A]] to <8 x i16> +// LLVM: [[B_I:%.*]] = bitcast <8 x half> [[B]] to <8 x i16> +// LLVM: [[C_I:%.*]] = bitcast <4 x half> [[C]] to <4 x i16> +// LLVM: [[A_BYTES:%.*]] = bitcast <8 x i16> [[A_I]] to <16 x i8> +// LLVM: [[B_BYTES:%.*]] = bitcast <8 x i16> [[B_I]] to <16 x i8> +// LLVM: [[C_BYTES:%.*]] = bitcast <4 x i16> [[C_I]] to <8 x i8> +// LLVM-DAG: [[A_CAST:%.*]] = bitcast <16 x i8> [[A_BYTES]] to <8 x half> +// LLVM-DAG: [[B_CAST:%.*]] = bitcast <16 x i8> [[B_BYTES]] to <8 x half> +// LLVM-DAG: [[C_CAST:%.*]] = bitcast <8 x i8> [[C_BYTES]] to <4 x half> +// LLVM-DAG: [[LANE:%.*]] = shufflevector <4 x half> [[C_CAST]], <4 x half> {{.*}}, <8 x i32> <i32 3, i32 3, i32 3, i32 3, i32 3, i32 3, i32 3, i32 3> +// LLVM: [[FMA:%.*]] = call <8 x half> @llvm.experimental.constrained.fma.v8f16(<8 x half> [[B_CAST]], <8 x half> [[LANE]], <8 x half> [[A_CAST]], metadata !"round.tonearest", metadata !"fpexcept.strict") +// LLVM: ret <8 x half> [[FMA]] + return vfmaq_lane_f16(a, b, c, 3); +} + +// ALL-LABEL: @test_vfma_laneq_f16( +float16x4_t test_vfma_laneq_f16(float16x4_t a, float16x4_t b, + float16x8_t c) { +// CIR: [[LANE:%.*]] = cir.vec.shuffle(%{{.*}}, %{{.*}} : !cir.vector<8 x !cir.f16>) [#cir.int<7> : !s32i, #cir.int<7> : !s32i, #cir.int<7> : !s32i, #cir.int<7> : !s32i] : !cir.vector<4 x !cir.f16> +// CIR: cir.fma [[LANE]], %{{.*}}, %{{.*}} : !cir.vector<4 x !cir.f16> fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true> + +// LLVM-SAME: <4 x half> {{.*}} [[A:%.*]], <4 x half> {{.*}} [[B:%.*]], <8 x half> {{.*}} [[C:%.*]]) {{.*}} { +// LLVM: [[A_I:%.*]] = bitcast <4 x half> [[A]] to <4 x i16> +// LLVM: [[B_I:%.*]] = bitcast <4 x half> [[B]] to <4 x i16> +// LLVM: [[C_I:%.*]] = bitcast <8 x half> [[C]] to <8 x i16> +// LLVM: [[A_BYTES:%.*]] = bitcast <4 x i16> [[A_I]] to <8 x i8> +// LLVM: [[B_BYTES:%.*]] = bitcast <4 x i16> [[B_I]] to <8 x i8> +// LLVM: [[C_BYTES:%.*]] = bitcast <8 x i16> [[C_I]] to <16 x i8> +// LLVM: [[A_CAST:%.*]] = bitcast <8 x i8> [[A_BYTES]] to <4 x half> +// LLVM: [[B_CAST:%.*]] = bitcast <8 x i8> [[B_BYTES]] to <4 x half> +// LLVM: [[C_CAST:%.*]] = bitcast <16 x i8> [[C_BYTES]] to <8 x half> +// LLVM: [[LANE:%.*]] = shufflevector <8 x half> [[C_CAST]], <8 x half> {{.*}}, <4 x i32> <i32 7, i32 7, i32 7, i32 7> +// LLVM: [[FMA:%.*]] = call <4 x half> @llvm.experimental.constrained.fma.v4f16(<4 x half> [[LANE]], <4 x half> [[B_CAST]], <4 x half> [[A_CAST]], metadata !"round.tonearest", metadata !"fpexcept.strict") +// LLVM: ret <4 x half> [[FMA]] + return vfma_laneq_f16(a, b, c, 7); +} + +// ALL-LABEL: @test_vfmaq_laneq_f16( +float16x8_t test_vfmaq_laneq_f16(float16x8_t a, float16x8_t b, + float16x8_t c) { +// CIR: [[LANE:%.*]] = cir.vec.shuffle(%{{.*}}, %{{.*}} : !cir.vector<8 x !cir.f16>) [#cir.int<7> : !s32i, #cir.int<7> : !s32i, #cir.int<7> : !s32i, #cir.int<7> : !s32i, #cir.int<7> : !s32i, #cir.int<7> : !s32i, #cir.int<7> : !s32i, #cir.int<7> : !s32i] : !cir.vector<8 x !cir.f16> +// CIR: cir.fma [[LANE]], %{{.*}}, %{{.*}} : !cir.vector<8 x !cir.f16> fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true> + +// LLVM-SAME: <8 x half> {{.*}} [[A:%.*]], <8 x half> {{.*}} [[B:%.*]], <8 x half> {{.*}} [[C:%.*]]) {{.*}} { +// LLVM: [[A_I:%.*]] = bitcast <8 x half> [[A]] to <8 x i16> +// LLVM: [[B_I:%.*]] = bitcast <8 x half> [[B]] to <8 x i16> +// LLVM: [[C_I:%.*]] = bitcast <8 x half> [[C]] to <8 x i16> +// LLVM: [[A_BYTES:%.*]] = bitcast <8 x i16> [[A_I]] to <16 x i8> +// LLVM: [[B_BYTES:%.*]] = bitcast <8 x i16> [[B_I]] to <16 x i8> +// LLVM: [[C_BYTES:%.*]] = bitcast <8 x i16> [[C_I]] to <16 x i8> +// LLVM: [[A_CAST:%.*]] = bitcast <16 x i8> [[A_BYTES]] to <8 x half> +// LLVM: [[B_CAST:%.*]] = bitcast <16 x i8> [[B_BYTES]] to <8 x half> +// LLVM: [[C_CAST:%.*]] = bitcast <16 x i8> [[C_BYTES]] to <8 x half> +// LLVM: [[LANE:%.*]] = shufflevector <8 x half> [[C_CAST]], <8 x half> {{.*}}, <8 x i32> <i32 7, i32 7, i32 7, i32 7, i32 7, i32 7, i32 7, i32 7> +// LLVM: [[FMA:%.*]] = call <8 x half> @llvm.experimental.constrained.fma.v8f16(<8 x half> [[LANE]], <8 x half> [[B_CAST]], <8 x half> [[A_CAST]], metadata !"round.tonearest", metadata !"fpexcept.strict") +// LLVM: ret <8 x half> [[FMA]] + return vfmaq_laneq_f16(a, b, c, 7); +} + +// ALL-LABEL: @test_vfmah_lane_f16( +float16_t test_vfmah_lane_f16(float16_t a, float16_t b, float16x4_t c) { +// CIR: [[INDEX:%.*]] = cir.const #cir.int<3> : !u64i +// CIR: [[LANE:%.*]] = cir.vec.extract %{{.*}}{{\[}}[[INDEX]] : !u64i] : !cir.vector<4 x !cir.f16> +// CIR: cir.fma %{{.*}}, [[LANE]], %{{.*}} : !cir.f16 fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true> + +// LLVM-SAME: half {{.*}} [[A:%.*]], half {{.*}} [[B:%.*]], <4 x half> {{.*}} [[C:%.*]]) {{.*}} { +// LLVM: [[LANE:%.*]] = extractelement <4 x half> [[C]], i{{32|64}} 3 +// LLVM: [[FMA:%.*]] = call half @llvm.experimental.constrained.fma.f16(half [[B]], half [[LANE]], half [[A]], metadata !"round.tonearest", metadata !"fpexcept.strict") +// LLVM: ret half [[FMA]] + return vfmah_lane_f16(a, b, c, 3); +} diff --git a/clang/test/CodeGen/AArch64/neon/fused-multiply-constrained.c b/clang/test/CodeGen/AArch64/neon/fused-multiply-constrained.c new file mode 100644 index 00000000000000..52ee3f20628511 --- /dev/null +++ b/clang/test/CodeGen/AArch64/neon/fused-multiply-constrained.c @@ -0,0 +1,91 @@ +// REQUIRES: aarch64-registered-target + +// RUN: %clang_cc1_cg_arm64_neon -ffp-exception-behavior=strict -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,sroa | FileCheck %s --check-prefixes=ALL,LLVM +// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -fexperimental-strict-floating-point -ffp-exception-behavior=strict -fclangir -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,sroa | FileCheck %s --check-prefixes=ALL,LLVM --implicit-check-not=' @llvm.fma.' %} +// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -fexperimental-strict-floating-point -ffp-exception-behavior=strict -fclangir -emit-cir %s -disable-O0-optnone | FileCheck %s --check-prefixes=ALL,CIR --implicit-check-not='cir.call_llvm_intrinsic "fma"' %} + +#include <arm_neon.h> + +// LLVM-LABEL: @test_vfma_f64( +// CIR-LABEL: @vfma_f64( +float64x1_t test_vfma_f64(float64x1_t a, float64x1_t b, float64x1_t c) { +// CIR: cir.fma %{{.*}}, %{{.*}}, %{{.*}} : !cir.vector<1 x !cir.double> fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true> + +// LLVM-SAME: <1 x double> {{.*}} [[A:%.*]], <1 x double> {{.*}} [[B:%.*]], <1 x double> {{.*}} [[C:%.*]]) {{.*}} { +// LLVM: [[A_I:%.*]] = bitcast <1 x double> [[A]] to i64 +// LLVM: [[A_INSERT:%.*]] = insertelement <1 x i64> undef, i64 [[A_I]], i64 0 +// LLVM: [[B_I:%.*]] = bitcast <1 x double> [[B]] to i64 +// LLVM: [[B_INSERT:%.*]] = insertelement <1 x i64> undef, i64 [[B_I]], i64 0 +// LLVM: [[C_I:%.*]] = bitcast <1 x double> [[C]] to i64 +// LLVM: [[C_INSERT:%.*]] = insertelement <1 x i64> undef, i64 [[C_I]], i64 0 +// LLVM: [[A_BYTES:%.*]] = bitcast <1 x i64> [[A_INSERT]] to <8 x i8> +// LLVM: [[B_BYTES:%.*]] = bitcast <1 x i64> [[B_INSERT]] to <8 x i8> +// LLVM: [[C_BYTES:%.*]] = bitcast <1 x i64> [[C_INSERT]] to <8 x i8> +// LLVM: [[A_CAST:%.*]] = bitcast <8 x i8> [[A_BYTES]] to <1 x double> +// LLVM: [[B_CAST:%.*]] = bitcast <8 x i8> [[B_BYTES]] to <1 x double> +// LLVM: [[C_CAST:%.*]] = bitcast <8 x i8> [[C_BYTES]] to <1 x double> +// LLVM: [[FMA:%.*]] = call <1 x double> @llvm.experimental.constrained.fma.v1f64(<1 x double> [[B_CAST]], <1 x double> [[C_CAST]], <1 x double> [[A_CAST]], metadata !"round.tonearest", metadata !"fpexcept.strict") +// LLVM: ret <1 x double> [[FMA]] + return vfma_f64(a, b, c); +} + +// ALL-LABEL: @test_vfma_laneq_f64( +float64x1_t test_vfma_laneq_f64(float64x1_t a, float64x1_t b, + float64x2_t v) { +// CIR: [[INDEX:%.*]] = cir.const #cir.int<0> : !u64i +// CIR: [[LANE:%.*]] = cir.vec.extract %{{.*}}{{\[}}[[INDEX]] : !u64i] : !cir.vector<2 x !cir.double> +// CIR: cir.fma %{{.*}}, [[LANE]], %{{.*}} : !cir.double fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true> + +// LLVM-SAME: <1 x double> {{.*}} [[A:%.*]], <1 x double> {{.*}} [[B:%.*]], <2 x double> {{.*}} [[V:%.*]]) {{.*}} { +// LLVM: [[A_I:%.*]] = bitcast <1 x double> [[A]] to i64 +// LLVM: [[A_INSERT:%.*]] = insertelement <1 x i64> undef, i64 [[A_I]], i64 0 +// LLVM: [[B_I:%.*]] = bitcast <1 x double> [[B]] to i64 +// LLVM: [[B_INSERT:%.*]] = insertelement <1 x i64> undef, i64 [[B_I]], i64 0 +// LLVM: [[V_I:%.*]] = bitcast <2 x double> [[V]] to <2 x i64> +// LLVM: [[A_BYTES:%.*]] = bitcast <1 x i64> [[A_INSERT]] to <8 x i8> +// LLVM: [[B_BYTES:%.*]] = bitcast <1 x i64> [[B_INSERT]] to <8 x i8> +// LLVM: [[V_BYTES:%.*]] = bitcast <2 x i64> [[V_I]] to <16 x i8> +// LLVM: [[A_CAST:%.*]] = bitcast <8 x i8> [[A_BYTES]] to double +// LLVM: [[B_CAST:%.*]] = bitcast <8 x i8> [[B_BYTES]] to double +// LLVM: [[V_CAST:%.*]] = bitcast <16 x i8> [[V_BYTES]] to <2 x double> +// LLVM: [[LANE:%.*]] = extractelement <2 x double> [[V_CAST]], i{{32|64}} 0 +// LLVM: [[FMA:%.*]] = call double @llvm.experimental.constrained.fma.f64(double [[B_CAST]], double [[LANE]], double [[A_CAST]], metadata !"round.tonearest", metadata !"fpexcept.strict") +// LLVM: [[RESULT:%.*]] = bitcast double [[FMA]] to <1 x double> +// LLVM: ret <1 x double> [[RESULT]] + return vfma_laneq_f64(a, b, v, 0); +} + +// ALL-LABEL: @test_vfmaq_laneq_f64( +float64x2_t test_vfmaq_laneq_f64(float64x2_t a, float64x2_t b, + float64x2_t v) { +// CIR: [[LANE:%.*]] = cir.vec.shuffle(%{{.*}}, %{{.*}} : !cir.vector<2 x !cir.double>) [#cir.int<1> : !s32i, #cir.int<1> : !s32i] : !cir.vector<2 x !cir.double> +// CIR: cir.fma [[LANE]], %{{.*}}, %{{.*}} : !cir.vector<2 x !cir.double> fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true> + +// LLVM-SAME: <2 x double> {{.*}} [[A:%.*]], <2 x double> {{.*}} [[B:%.*]], <2 x double> {{.*}} [[V:%.*]]) {{.*}} { +// LLVM: [[A_I:%.*]] = bitcast <2 x double> [[A]] to <2 x i64> +// LLVM: [[B_I:%.*]] = bitcast <2 x double> [[B]] to <2 x i64> +// LLVM: [[V_I:%.*]] = bitcast <2 x double> [[V]] to <2 x i64> +// LLVM: [[A_BYTES:%.*]] = bitcast <2 x i64> [[A_I]] to <16 x i8> +// LLVM: [[B_BYTES:%.*]] = bitcast <2 x i64> [[B_I]] to <16 x i8> +// LLVM: [[V_BYTES:%.*]] = bitcast <2 x i64> [[V_I]] to <16 x i8> +// LLVM: [[A_CAST:%.*]] = bitcast <16 x i8> [[A_BYTES]] to <2 x double> +// LLVM: [[B_CAST:%.*]] = bitcast <16 x i8> [[B_BYTES]] to <2 x double> +// LLVM: [[V_CAST:%.*]] = bitcast <16 x i8> [[V_BYTES]] to <2 x double> +// LLVM: [[LANE:%.*]] = shufflevector <2 x double> [[V_CAST]], <2 x double> {{.*}}, <2 x i32> <i32 1, i32 1> +// LLVM: [[FMA:%.*]] = call <2 x double> @llvm.experimental.constrained.fma.v2f64(<2 x double> [[LANE]], <2 x double> [[B_CAST]], <2 x double> [[A_CAST]], metadata !"round.tonearest", metadata !"fpexcept.strict") +// LLVM: ret <2 x double> [[FMA]] + return vfmaq_laneq_f64(a, b, v, 1); +} + +// ALL-LABEL: @test_vfmas_lane_f32( +float32_t test_vfmas_lane_f32(float32_t a, float32_t b, float32x2_t c) { +// CIR: [[INDEX:%.*]] = cir.const #cir.int<1> : !u64i +// CIR: [[LANE:%.*]] = cir.vec.extract %{{.*}}{{\[}}[[INDEX]] : !u64i] : !cir.vector<2 x !cir.float> +// CIR: cir.fma %{{.*}}, [[LANE]], %{{.*}} : !cir.float fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true> + +// LLVM-SAME: float {{.*}} [[A:%.*]], float {{.*}} [[B:%.*]], <2 x float> {{.*}} [[C:%.*]]) {{.*}} { +// LLVM: [[LANE:%.*]] = extractelement <2 x float> [[C]], i{{32|64}} 1 +// LLVM: [[FMA:%.*]] = call float @llvm.experimental.constrained.fma.f32(float [[B]], float [[LANE]], float [[A]], metadata !"round.tonearest", metadata !"fpexcept.strict") +// LLVM: ret float [[FMA]] + return vfmas_lane_f32(a, b, c, 1); +} diff --git a/clang/test/CodeGen/AArch64/neon/intrinsics-constrained.c b/clang/test/CodeGen/AArch64/neon/intrinsics-constrained.c new file mode 100644 index 00000000000000..ef915a0ccea1e9 --- /dev/null +++ b/clang/test/CodeGen/AArch64/neon/intrinsics-constrained.c @@ -0,0 +1,56 @@ +// REQUIRES: aarch64-registered-target + +// RUN: %clang_cc1_cg_arm64_neon -target-feature +fullfp16 -ffp-exception-behavior=maytrap -DEXCEPT=1 -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,sroa | FileCheck %s --check-prefix=LLVM --implicit-check-not=fpexcept.maytrap --implicit-check-not=' @llvm.sqrt.' +// RUN: %clang_cc1_cg_arm64_neon -target-feature +fullfp16 -ffp-exception-behavior=strict -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,sroa | FileCheck %s --check-prefix=LLVM --implicit-check-not=' @llvm.sqrt.' +// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 -fexperimental-strict-floating-point -ffp-exception-behavior=maytrap -DEXCEPT=1 -fclangir -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,sroa | FileCheck %s --check-prefix=LLVM --implicit-check-not=fpexcept.maytrap --implicit-check-not=' @llvm.sqrt.' %} +// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 -fexperimental-strict-floating-point -ffp-exception-behavior=maytrap -DEXCEPT=1 -fclangir -emit-cir %s -disable-O0-optnone | FileCheck %s --check-prefix=CIR --implicit-check-not='except_mode = maytrap' --implicit-check-not='cir.call_llvm_intrinsic "sqrt"' %} +// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 -fexperimental-strict-floating-point -ffp-exception-behavior=strict -fclangir -emit-llvm %s -disable-O0-optnone | opt -S -passes=mem2reg,sroa | FileCheck %s --check-prefix=LLVM --implicit-check-not=' @llvm.sqrt.' %} +// RUN: %if cir-enabled %{%clang_cc1_cg_arm64_neon -target-feature +fullfp16 -fexperimental-strict-floating-point -ffp-exception-behavior=strict -fclangir -emit-cir %s -disable-O0-optnone | FileCheck %s --check-prefix=CIR --implicit-check-not='cir.call_llvm_intrinsic "sqrt"' %} + +#if EXCEPT +#pragma float_control(except, on) +#endif + +#include <arm_neon.h> + +// LLVM-LABEL: @test_vsqrt_f16( +// CIR-LABEL: @vsqrt_f16( +float16x4_t test_vsqrt_f16(float16x4_t a) { +// CIR: cir.sqrt %{{.*}} : !cir.vector<4 x !cir.f16> fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true> + +// LLVM-SAME: <4 x half> {{.*}} [[A:%.*]]) {{.*}} { +// LLVM: [[A_I:%.*]] = bitcast <4 x half> [[A]] to <4 x i16> +// LLVM: [[A_BYTES:%.*]] = bitcast <4 x i16> [[A_I]] to <8 x i8> +// LLVM: [[A_CAST:%.*]] = bitcast <8 x i8> [[A_BYTES]] to <4 x half> +// LLVM: [[SQRT:%.*]] = call <4 x half> @llvm.experimental.constrained.sqrt.v4f16(<4 x half> [[A_CAST]], metadata !"round.tonearest", metadata !"fpexcept.strict") +// LLVM: ret <4 x half> [[SQRT]] + return vsqrt_f16(a); +} + +// LLVM-LABEL: @test_vsqrtq_f16( +// CIR-LABEL: @vsqrtq_f16( +float16x8_t test_vsqrtq_f16(float16x8_t a) { +// CIR: cir.sqrt %{{.*}} : !cir.vector<8 x !cir.f16> fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true> + +// LLVM-SAME: <8 x half> {{.*}} [[A:%.*]]) {{.*}} { +// LLVM: [[A_I:%.*]] = bitcast <8 x half> [[A]] to <8 x i16> +// LLVM: [[A_BYTES:%.*]] = bitcast <8 x i16> [[A_I]] to <16 x i8> +// LLVM: [[A_CAST:%.*]] = bitcast <16 x i8> [[A_BYTES]] to <8 x half> +// LLVM: [[SQRT:%.*]] = call <8 x half> @llvm.experimental.constrained.sqrt.v8f16(<8 x half> [[A_CAST]], metadata !"round.tonearest", metadata !"fpexcept.strict") +// LLVM: ret <8 x half> [[SQRT]] + return vsqrtq_f16(a); +} + +// LLVM-LABEL: @test_vsqrtq_f64( +// CIR-LABEL: @vsqrtq_f64( +float64x2_t test_vsqrtq_f64(float64x2_t a) { +// CIR: cir.sqrt %{{.*}} : !cir.vector<2 x !cir.double> fenv<dynamic_rounding_mode = tonearest, except_mode = unknown, strict_except = true> + +// LLVM-SAME: <2 x double> {{.*}} [[A:%.*]]) {{.*}} { +// LLVM: [[A_I:%.*]] = bitcast <2 x double> [[A]] to <2 x i64> +// LLVM: [[A_BYTES:%.*]] = bitcast <2 x i64> [[A_I]] to <16 x i8> +// LLVM: [[A_CAST:%.*]] = bitcast <16 x i8> [[A_BYTES]] to <2 x double> +// LLVM: [[SQRT:%.*]] = call <2 x double> @llvm.experimental.constrained.sqrt.v2f64(<2 x double> [[A_CAST]], metadata !"round.tonearest", metadata !"fpexcept.strict") +// LLVM: ret <2 x double> [[SQRT]] + return vsqrtq_f64(a); +} diff --git a/clang/test/CodeGen/AArch64/v8.2a-fp16-intrinsics-constrained.c b/clang/test/CodeGen/AArch64/v8.2a-fp16-intrinsics-constrained.c index c9d5071c4eb838..4006171e402964 100644 --- a/clang/test/CodeGen/AArch64/v8.2a-fp16-intrinsics-constrained.c +++ b/clang/test/CodeGen/AArch64/v8.2a-fp16-intrinsics-constrained.c @@ -200,14 +200,6 @@ float16_t test_vrndxh_f16(float16_t a) { return vrndxh_f16(a); } -// COMMON-LABEL: test_vsqrth_f16 -// UNCONSTRAINED: [[SQR:%.*]] = call half @llvm.sqrt.f16(half %a) -// CONSTRAINED: [[SQR:%.*]] = call half @llvm.experimental.constrained.sqrt.f16(half %a, metadata !"round.tonearest", metadata !"fpexcept.strict") -// COMMONIR: ret half [[SQR]] -float16_t test_vsqrth_f16(float16_t a) { - return vsqrth_f16(a); -} - // COMMON-LABEL: test_vaddh_f16 // UNCONSTRAINED: [[ADD:%.*]] = fadd half %a, %b // CONSTRAINED: [[ADD:%.*]] = call half @llvm.experimental.constrained.fadd.f16(half %a, half %b, metadata !"round.tonearest", metadata !"fpexcept.strict") @@ -285,14 +277,6 @@ float16_t test_vsubh_f16(float16_t a, float16_t b) { return vsubh_f16(a, b); } -// COMMON-LABEL: test_vfmah_f16 -// UNCONSTRAINED: [[FMA:%.*]] = call half @llvm.fma.f16(half %b, half %c, half %a) -// CONSTRAINED: [[FMA:%.*]] = call half @llvm.experimental.constrained.fma.f16(half %b, half %c, half %a, metadata !"round.tonearest", metadata !"fpexcept.strict") -// COMMONIR: ret half [[FMA]] -float16_t test_vfmah_f16(float16_t a, float16_t b, float16_t c) { - return vfmah_f16(a, b, c); -} - // COMMON-LABEL: test_vfmsh_f16 // COMMONIR: [[SUB:%.*]] = fneg half %b // UNCONSTRAINED: [[ADD:%.*]] = call half @llvm.fma.f16(half [[SUB]], half %c, half %a) diff --git a/clang/test/CodeGen/AArch64/v8.2a-neon-intrinsics-constrained.c b/clang/test/CodeGen/AArch64/v8.2a-neon-intrinsics-constrained.c index dac8b931ff210d..a30bfa83526469 100644 --- a/clang/test/CodeGen/AArch64/v8.2a-neon-intrinsics-constrained.c +++ b/clang/test/CodeGen/AArch64/v8.2a-neon-intrinsics-constrained.c @@ -21,120 +21,8 @@ #include <arm_neon.h> -// UNCONSTRAINED-LABEL: define dso_local <4 x half> @test_vsqrt_f16( -// UNCONSTRAINED-SAME: <4 x half> noundef [[A:%.*]]) #[[ATTR0:[0-9]+]] { -// UNCONSTRAINED-NEXT: [[ENTRY:.*:]] -// UNCONSTRAINED-NEXT: [[TMP0:%.*]] = bitcast <4 x half> [[A]] to <4 x i16> -// UNCONSTRAINED-NEXT: [[TMP1:%.*]] = bitcast <4 x i16> [[TMP0]] to <8 x i8> -// UNCONSTRAINED-NEXT: [[TMP2:%.*]] = bitcast <8 x i8> [[TMP1]] to <4 x half> -// UNCONSTRAINED-NEXT: [[VSQRT_I:%.*]] = call <4 x half> @llvm.sqrt.v4f16(<4 x half> [[TMP2]]) -// UNCONSTRAINED-NEXT: ret <4 x half> [[VSQRT_I]] -// -// CONSTRAINED-LABEL: define dso_local <4 x half> @test_vsqrt_f16( -// CONSTRAINED-SAME: <4 x half> noundef [[A:%.*]]) #[[ATTR0:[0-9]+]] { -// CONSTRAINED-NEXT: [[ENTRY:.*:]] -// CONSTRAINED-NEXT: [[TMP0:%.*]] = bitcast <4 x half> [[A]] to <4 x i16> -// CONSTRAINED-NEXT: [[TMP1:%.*]] = bitcast <4 x i16> [[TMP0]] to <8 x i8> -// CONSTRAINED-NEXT: [[TMP2:%.*]] = bitcast <8 x i8> [[TMP1]] to <4 x half> -// CONSTRAINED-NEXT: [[VSQRT_I:%.*]] = call <4 x half> @llvm.experimental.constrained.sqrt.v4f16(<4 x half> [[TMP2]], metadata !"round.tonearest", metadata !"fpexcept.strict") #[[ATTR2:[0-9]+]] -// CONSTRAINED-NEXT: ret <4 x half> [[VSQRT_I]] -// -float16x4_t test_vsqrt_f16(float16x4_t a) { - return vsqrt_f16(a); -} - -// UNCONSTRAINED-LABEL: define dso_local <8 x half> @test_vsqrtq_f16( -// UNCONSTRAINED-SAME: <8 x half> noundef [[A:%.*]]) #[[ATTR0]] { -// UNCONSTRAINED-NEXT: [[ENTRY:.*:]] -// UNCONSTRAINED-NEXT: [[TMP0:%.*]] = bitcast <8 x half> [[A]] to <8 x i16> -// UNCONSTRAINED-NEXT: [[TMP1:%.*]] = bitcast <8 x i16> [[TMP0]] to <16 x i8> -// UNCONSTRAINED-NEXT: [[TMP2:%.*]] = bitcast <16 x i8> [[TMP1]] to <8 x half> -// UNCONSTRAINED-NEXT: [[VSQRT_I:%.*]] = call <8 x half> @llvm.sqrt.v8f16(<8 x half> [[TMP2]]) -// UNCONSTRAINED-NEXT: ret <8 x half> [[VSQRT_I]] -// -// CONSTRAINED-LABEL: define dso_local <8 x half> @test_vsqrtq_f16( -// CONSTRAINED-SAME: <8 x half> noundef [[A:%.*]]) #[[ATTR0]] { -// CONSTRAINED-NEXT: [[ENTRY:.*:]] -// CONSTRAINED-NEXT: [[TMP0:%.*]] = bitcast <8 x half> [[A]] to <8 x i16> -// CONSTRAINED-NEXT: [[TMP1:%.*]] = bitcast <8 x i16> [[TMP0]] to <16 x i8> -// CONSTRAINED-NEXT: [[TMP2:%.*]] = bitcast <16 x i8> [[TMP1]] to <8 x half> -// CONSTRAINED-NEXT: [[VSQRT_I:%.*]] = call <8 x half> @llvm.experimental.constrained.sqrt.v8f16(<8 x half> [[TMP2]], metadata !"round.tonearest", metadata !"fpexcept.strict") #[[ATTR2]] -// CONSTRAINED-NEXT: ret <8 x half> [[VSQRT_I]] -// -float16x8_t test_vsqrtq_f16(float16x8_t a) { - return vsqrtq_f16(a); -} - -// UNCONSTRAINED-LABEL: define dso_local <4 x half> @test_vfma_f16( -// UNCONSTRAINED-SAME: <4 x half> noundef [[A:%.*]], <4 x half> noundef [[B:%.*]], <4 x half> noundef [[C:%.*]]) #[[ATTR0]] { -// UNCONSTRAINED-NEXT: [[ENTRY:.*:]] -// UNCONSTRAINED-NEXT: [[TMP0:%.*]] = bitcast <4 x half> [[A]] to <4 x i16> -// UNCONSTRAINED-NEXT: [[TMP1:%.*]] = bitcast <4 x half> [[B]] to <4 x i16> -// UNCONSTRAINED-NEXT: [[TMP2:%.*]] = bitcast <4 x half> [[C]] to <4 x i16> -// UNCONSTRAINED-NEXT: [[TMP3:%.*]] = bitcast <4 x i16> [[TMP0]] to <8 x i8> -// UNCONSTRAINED-NEXT: [[TMP4:%.*]] = bitcast <4 x i16> [[TMP1]] to <8 x i8> -// UNCONSTRAINED-NEXT: [[TMP5:%.*]] = bitcast <4 x i16> [[TMP2]] to <8 x i8> -// UNCONSTRAINED-NEXT: [[TMP6:%.*]] = bitcast <8 x i8> [[TMP3]] to <4 x half> -// UNCONSTRAINED-NEXT: [[TMP7:%.*]] = bitcast <8 x i8> [[TMP4]] to <4 x half> -// UNCONSTRAINED-NEXT: [[TMP8:%.*]] = bitcast <8 x i8> [[TMP5]] to <4 x half> -// UNCONSTRAINED-NEXT: [[TMP9:%.*]] = call <4 x half> @llvm.fma.v4f16(<4 x half> [[TMP7]], <4 x half> [[TMP8]], <4 x half> [[TMP6]]) -// UNCONSTRAINED-NEXT: ret <4 x half> [[TMP9]] -// -// CONSTRAINED-LABEL: define dso_local <4 x half> @test_vfma_f16( -// CONSTRAINED-SAME: <4 x half> noundef [[A:%.*]], <4 x half> noundef [[B:%.*]], <4 x half> noundef [[C:%.*]]) #[[ATTR0]] { -// CONSTRAINED-NEXT: [[ENTRY:.*:]] -// CONSTRAINED-NEXT: [[TMP0:%.*]] = bitcast <4 x half> [[A]] to <4 x i16> -// CONSTRAINED-NEXT: [[TMP1:%.*]] = bitcast <4 x half> [[B]] to <4 x i16> -// CONSTRAINED-NEXT: [[TMP2:%.*]] = bitcast <4 x half> [[C]] to <4 x i16> -// CONSTRAINED-NEXT: [[TMP3:%.*]] = bitcast <4 x i16> [[TMP0]] to <8 x i8> -// CONSTRAINED-NEXT: [[TMP4:%.*]] = bitcast <4 x i16> [[TMP1]] to <8 x i8> -// CONSTRAINED-NEXT: [[TMP5:%.*]] = bitcast <4 x i16> [[TMP2]] to <8 x i8> -// CONSTRAINED-NEXT: [[TMP6:%.*]] = bitcast <8 x i8> [[TMP3]] to <4 x half> -// CONSTRAINED-NEXT: [[TMP7:%.*]] = bitcast <8 x i8> [[TMP4]] to <4 x half> -// CONSTRAINED-NEXT: [[TMP8:%.*]] = bitcast <8 x i8> [[TMP5]] to <4 x half> -// CONSTRAINED-NEXT: [[TMP9:%.*]] = call <4 x half> @llvm.experimental.constrained.fma.v4f16(<4 x half> [[TMP7]], <4 x half> [[TMP8]], <4 x half> [[TMP6]], metadata !"round.tonearest", metadata !"fpexcept.strict") #[[ATTR2]] -// CONSTRAINED-NEXT: ret <4 x half> [[TMP9]] -// -float16x4_t test_vfma_f16(float16x4_t a, float16x4_t b, float16x4_t c) { - return vfma_f16(a, b, c); -} - -// UNCONSTRAINED-LABEL: define dso_local <8 x half> @test_vfmaq_f16( -// UNCONSTRAINED-SAME: <8 x half> noundef [[A:%.*]], <8 x half> noundef [[B:%.*]], <8 x half> noundef [[C:%.*]]) #[[ATTR0]] { -// UNCONSTRAINED-NEXT: [[ENTRY:.*:]] -// UNCONSTRAINED-NEXT: [[TMP0:%.*]] = bitcast <8 x half> [[A]] to <8 x i16> -// UNCONSTRAINED-NEXT: [[TMP1:%.*]] = bitcast <8 x half> [[B]] to <8 x i16> -// UNCONSTRAINED-NEXT: [[TMP2:%.*]] = bitcast <8 x half> [[C]] to <8 x i16> -// UNCONSTRAINED-NEXT: [[TMP3:%.*]] = bitcast <8 x i16> [[TMP0]] to <16 x i8> -// UNCONSTRAINED-NEXT: [[TMP4:%.*]] = bitcast <8 x i16> [[TMP1]] to <16 x i8> -// UNCONSTRAINED-NEXT: [[TMP5:%.*]] = bitcast <8 x i16> [[TMP2]] to <16 x i8> -// UNCONSTRAINED-NEXT: [[TMP6:%.*]] = bitcast <16 x i8> [[TMP3]] to <8 x half> -// UNCONSTRAINED-NEXT: [[TMP7:%.*]] = bitcast <16 x i8> [[TMP4]] to <8 x half> -// UNCONSTRAINED-NEXT: [[TMP8:%.*]] = bitcast <16 x i8> [[TMP5]] to <8 x half> -// UNCONSTRAINED-NEXT: [[TMP9:%.*]] = call <8 x half> @llvm.fma.v8f16(<8 x half> [[TMP7]], <8 x half> [[TMP8]], <8 x half> [[TMP6]]) -// UNCONSTRAINED-NEXT: ret <8 x half> [[TMP9]] -// -// CONSTRAINED-LABEL: define dso_local <8 x half> @test_vfmaq_f16( -// CONSTRAINED-SAME: <8 x half> noundef [[A:%.*]], <8 x half> noundef [[B:%.*]], <8 x half> noundef [[C:%.*]]) #[[ATTR0]] { -// CONSTRAINED-NEXT: [[ENTRY:.*:]] -// CONSTRAINED-NEXT: [[TMP0:%.*]] = bitcast <8 x half> [[A]] to <8 x i16> -// CONSTRAINED-NEXT: [[TMP1:%.*]] = bitcast <8 x half> [[B]] to <8 x i16> -// CONSTRAINED-NEXT: [[TMP2:%.*]] = bitcast <8 x half> [[C]] to <8 x i16> -// CONSTRAINED-NEXT: [[TMP3:%.*]] = bitcast <8 x i16> [[TMP0]] to <16 x i8> -// CONSTRAINED-NEXT: [[TMP4:%.*]] = bitcast <8 x i16> [[TMP1]] to <16 x i8> -// CONSTRAINED-NEXT: [[TMP5:%.*]] = bitcast <8 x i16> [[TMP2]] to <16 x i8> -// CONSTRAINED-NEXT: [[TMP6:%.*]] = bitcast <16 x i8> [[TMP3]] to <8 x half> -// CONSTRAINED-NEXT: [[TMP7:%.*]] = bitcast <16 x i8> [[TMP4]] to <8 x half> -// CONSTRAINED-NEXT: [[TMP8:%.*]] = bitcast <16 x i8> [[TMP5]] to <8 x half> -// CONSTRAINED-NEXT: [[TMP9:%.*]] = call <8 x half> @llvm.experimental.constrained.fma.v8f16(<8 x half> [[TMP7]], <8 x half> [[TMP8]], <8 x half> [[TMP6]], metadata !"round.tonearest", metadata !"fpexcept.strict") #[[ATTR2]] -// CONSTRAINED-NEXT: ret <8 x half> [[TMP9]] -// -float16x8_t test_vfmaq_f16(float16x8_t a, float16x8_t b, float16x8_t c) { - return vfmaq_f16(a, b, c); -} - // UNCONSTRAINED-LABEL: define dso_local <4 x half> @test_vfms_f16( -// UNCONSTRAINED-SAME: <4 x half> noundef [[A:%.*]], <4 x half> noundef [[B:%.*]], <4 x half> noundef [[C:%.*]]) #[[ATTR0]] { +// UNCONSTRAINED-SAME: <4 x half> noundef [[A:%.*]], <4 x half> noundef [[B:%.*]], <4 x half> noundef [[C:%.*]]) #[[ATTR0:[0-9]+]] { // UNCONSTRAINED-NEXT: [[ENTRY:.*:]] // UNCONSTRAINED-NEXT: [[FNEG_I:%.*]] = fneg <4 x half> [[B]] // UNCONSTRAINED-NEXT: [[TMP0:%.*]] = bitcast <4 x half> [[A]] to <4 x i16> @@ -150,7 +38,7 @@ float16x8_t test_vfmaq_f16(float16x8_t a, float16x8_t b, float16x8_t c) { // UNCONSTRAINED-NEXT: ret <4 x half> [[TMP9]] // // CONSTRAINED-LABEL: define dso_local <4 x half> @test_vfms_f16( -// CONSTRAINED-SAME: <4 x half> noundef [[A:%.*]], <4 x half> noundef [[B:%.*]], <4 x half> noundef [[C:%.*]]) #[[ATTR0]] { +// CONSTRAINED-SAME: <4 x half> noundef [[A:%.*]], <4 x half> noundef [[B:%.*]], <4 x half> noundef [[C:%.*]]) #[[ATTR0:[0-9]+]] { // CONSTRAINED-NEXT: [[ENTRY:.*:]] // CONSTRAINED-NEXT: [[FNEG_I:%.*]] = fneg <4 x half> [[B]] // CONSTRAINED-NEXT: [[TMP0:%.*]] = bitcast <4 x half> [[A]] to <4 x i16> @@ -162,7 +50,7 @@ float16x8_t test_vfmaq_f16(float16x8_t a, float16x8_t b, float16x8_t c) { // CONSTRAINED-NEXT: [[TMP6:%.*]] = bitcast <8 x i8> [[TMP3]] to <4 x half> // CONSTRAINED-NEXT: [[TMP7:%.*]] = bitcast <8 x i8> [[TMP4]] to <4 x half> // CONSTRAINED-NEXT: [[TMP8:%.*]] = bitcast <8 x i8> [[TMP5]] to <4 x half> -// CONSTRAINED-NEXT: [[TMP9:%.*]] = call <4 x half> @llvm.experimental.constrained.fma.v4f16(<4 x half> [[TMP7]], <4 x half> [[TMP8]], <4 x half> [[TMP6]], metadata !"round.tonearest", metadata !"fpexcept.strict") #[[ATTR2]] +// CONSTRAINED-NEXT: [[TMP9:%.*]] = call <4 x half> @llvm.experimental.constrained.fma.v4f16(<4 x half> [[TMP7]], <4 x half> [[TMP8]], <4 x half> [[TMP6]], metadata !"round.tonearest", metadata !"fpexcept.strict") #[[ATTR2:[0-9]+]] // CONSTRAINED-NEXT: ret <4 x half> [[TMP9]] // float16x4_t test_vfms_f16(float16x4_t a, float16x4_t b, float16x4_t c) { @@ -205,150 +93,6 @@ float16x8_t test_vfmsq_f16(float16x8_t a, float16x8_t b, float16x8_t c) { return vfmsq_f16(a, b, c); } -// UNCONSTRAINED-LABEL: define dso_local <4 x half> @test_vfma_lane_f16( -// UNCONSTRAINED-SAME: <4 x half> noundef [[A:%.*]], <4 x half> noundef [[B:%.*]], <4 x half> noundef [[C:%.*]]) #[[ATTR0]] { -// UNCONSTRAINED-NEXT: [[ENTRY:.*:]] -// UNCONSTRAINED-NEXT: [[TMP0:%.*]] = bitcast <4 x half> [[A]] to <4 x i16> -// UNCONSTRAINED-NEXT: [[TMP1:%.*]] = bitcast <4 x half> [[B]] to <4 x i16> -// UNCONSTRAINED-NEXT: [[TMP2:%.*]] = bitcast <4 x half> [[C]] to <4 x i16> -// UNCONSTRAINED-NEXT: [[TMP3:%.*]] = bitcast <4 x i16> [[TMP0]] to <8 x i8> -// UNCONSTRAINED-NEXT: [[TMP4:%.*]] = bitcast <4 x i16> [[TMP1]] to <8 x i8> -// UNCONSTRAINED-NEXT: [[TMP5:%.*]] = bitcast <4 x i16> [[TMP2]] to <8 x i8> -// UNCONSTRAINED-NEXT: [[TMP6:%.*]] = bitcast <8 x i8> [[TMP5]] to <4 x half> -// UNCONSTRAINED-NEXT: [[LANE:%.*]] = shufflevector <4 x half> [[TMP6]], <4 x half> [[TMP6]], <4 x i32> <i32 3, i32 3, i32 3, i32 3> -// UNCONSTRAINED-NEXT: [[FMLA:%.*]] = bitcast <8 x i8> [[TMP4]] to <4 x half> -// UNCONSTRAINED-NEXT: [[FMLA1:%.*]] = bitcast <8 x i8> [[TMP3]] to <4 x half> -// UNCONSTRAINED-NEXT: [[FMLA2:%.*]] = call <4 x half> @llvm.fma.v4f16(<4 x half> [[FMLA]], <4 x half> [[LANE]], <4 x half> [[FMLA1]]) -// UNCONSTRAINED-NEXT: ret <4 x half> [[FMLA2]] -// -// CONSTRAINED-LABEL: define dso_local <4 x half> @test_vfma_lane_f16( -// CONSTRAINED-SAME: <4 x half> noundef [[A:%.*]], <4 x half> noundef [[B:%.*]], <4 x half> noundef [[C:%.*]]) #[[ATTR0]] { -// CONSTRAINED-NEXT: [[ENTRY:.*:]] -// CONSTRAINED-NEXT: [[TMP0:%.*]] = bitcast <4 x half> [[A]] to <4 x i16> -// CONSTRAINED-NEXT: [[TMP1:%.*]] = bitcast <4 x half> [[B]] to <4 x i16> -// CONSTRAINED-NEXT: [[TMP2:%.*]] = bitcast <4 x half> [[C]] to <4 x i16> -// CONSTRAINED-NEXT: [[TMP3:%.*]] = bitcast <4 x i16> [[TMP0]] to <8 x i8> -// CONSTRAINED-NEXT: [[TMP4:%.*]] = bitcast <4 x i16> [[TMP1]] to <8 x i8> -// CONSTRAINED-NEXT: [[TMP5:%.*]] = bitcast <4 x i16> [[TMP2]] to <8 x i8> -// CONSTRAINED-NEXT: [[TMP6:%.*]] = bitcast <8 x i8> [[TMP5]] to <4 x half> -// CONSTRAINED-NEXT: [[LANE:%.*]] = shufflevector <4 x half> [[TMP6]], <4 x half> [[TMP6]], <4 x i32> <i32 3, i32 3, i32 3, i32 3> -// CONSTRAINED-NEXT: [[FMLA:%.*]] = bitcast <8 x i8> [[TMP4]] to <4 x half> -// CONSTRAINED-NEXT: [[FMLA1:%.*]] = bitcast <8 x i8> [[TMP3]] to <4 x half> -// CONSTRAINED-NEXT: [[FMLA2:%.*]] = call <4 x half> @llvm.experimental.constrained.fma.v4f16(<4 x half> [[FMLA]], <4 x half> [[LANE]], <4 x half> [[FMLA1]], metadata !"round.tonearest", metadata !"fpexcept.strict") #[[ATTR2]] -// CONSTRAINED-NEXT: ret <4 x half> [[FMLA2]] -// -float16x4_t test_vfma_lane_f16(float16x4_t a, float16x4_t b, float16x4_t c) { - return vfma_lane_f16(a, b, c, 3); -} - -// UNCONSTRAINED-LABEL: define dso_local <8 x half> @test_vfmaq_lane_f16( -// UNCONSTRAINED-SAME: <8 x half> noundef [[A:%.*]], <8 x half> noundef [[B:%.*]], <4 x half> noundef [[C:%.*]]) #[[ATTR0]] { -// UNCONSTRAINED-NEXT: [[ENTRY:.*:]] -// UNCONSTRAINED-NEXT: [[TMP0:%.*]] = bitcast <8 x half> [[A]] to <8 x i16> -// UNCONSTRAINED-NEXT: [[TMP1:%.*]] = bitcast <8 x half> [[B]] to <8 x i16> -// UNCONSTRAINED-NEXT: [[TMP2:%.*]] = bitcast <4 x half> [[C]] to <4 x i16> -// UNCONSTRAINED-NEXT: [[TMP3:%.*]] = bitcast <8 x i16> [[TMP0]] to <16 x i8> -// UNCONSTRAINED-NEXT: [[TMP4:%.*]] = bitcast <8 x i16> [[TMP1]] to <16 x i8> -// UNCONSTRAINED-NEXT: [[TMP5:%.*]] = bitcast <4 x i16> [[TMP2]] to <8 x i8> -// UNCONSTRAINED-NEXT: [[TMP6:%.*]] = bitcast <8 x i8> [[TMP5]] to <4 x half> -// UNCONSTRAINED-NEXT: [[LANE:%.*]] = shufflevector <4 x half> [[TMP6]], <4 x half> [[TMP6]], <8 x i32> <i32 3, i32 3, i32 3, i32 3, i32 3, i32 3, i32 3, i32 3> -// UNCONSTRAINED-NEXT: [[FMLA:%.*]] = bitcast <16 x i8> [[TMP4]] to <8 x half> -// UNCONSTRAINED-NEXT: [[FMLA1:%.*]] = bitcast <16 x i8> [[TMP3]] to <8 x half> -// UNCONSTRAINED-NEXT: [[FMLA2:%.*]] = call <8 x half> @llvm.fma.v8f16(<8 x half> [[FMLA]], <8 x half> [[LANE]], <8 x half> [[FMLA1]]) -// UNCONSTRAINED-NEXT: ret <8 x half> [[FMLA2]] -// -// CONSTRAINED-LABEL: define dso_local <8 x half> @test_vfmaq_lane_f16( -// CONSTRAINED-SAME: <8 x half> noundef [[A:%.*]], <8 x half> noundef [[B:%.*]], <4 x half> noundef [[C:%.*]]) #[[ATTR0]] { -// CONSTRAINED-NEXT: [[ENTRY:.*:]] -// CONSTRAINED-NEXT: [[TMP0:%.*]] = bitcast <8 x half> [[A]] to <8 x i16> -// CONSTRAINED-NEXT: [[TMP1:%.*]] = bitcast <8 x half> [[B]] to <8 x i16> -// CONSTRAINED-NEXT: [[TMP2:%.*]] = bitcast <4 x half> [[C]] to <4 x i16> -// CONSTRAINED-NEXT: [[TMP3:%.*]] = bitcast <8 x i16> [[TMP0]] to <16 x i8> -// CONSTRAINED-NEXT: [[TMP4:%.*]] = bitcast <8 x i16> [[TMP1]] to <16 x i8> -// CONSTRAINED-NEXT: [[TMP5:%.*]] = bitcast <4 x i16> [[TMP2]] to <8 x i8> -// CONSTRAINED-NEXT: [[TMP6:%.*]] = bitcast <8 x i8> [[TMP5]] to <4 x half> -// CONSTRAINED-NEXT: [[LANE:%.*]] = shufflevector <4 x half> [[TMP6]], <4 x half> [[TMP6]], <8 x i32> <i32 3, i32 3, i32 3, i32 3, i32 3, i32 3, i32 3, i32 3> -// CONSTRAINED-NEXT: [[FMLA:%.*]] = bitcast <16 x i8> [[TMP4]] to <8 x half> -// CONSTRAINED-NEXT: [[FMLA1:%.*]] = bitcast <16 x i8> [[TMP3]] to <8 x half> -// CONSTRAINED-NEXT: [[FMLA2:%.*]] = call <8 x half> @llvm.experimental.constrained.fma.v8f16(<8 x half> [[FMLA]], <8 x half> [[LANE]], <8 x half> [[FMLA1]], metadata !"round.tonearest", metadata !"fpexcept.strict") #[[ATTR2]] -// CONSTRAINED-NEXT: ret <8 x half> [[FMLA2]] -// -float16x8_t test_vfmaq_lane_f16(float16x8_t a, float16x8_t b, float16x4_t c) { - return vfmaq_lane_f16(a, b, c, 3); -} - -// UNCONSTRAINED-LABEL: define dso_local <4 x half> @test_vfma_laneq_f16( -// UNCONSTRAINED-SAME: <4 x half> noundef [[A:%.*]], <4 x half> noundef [[B:%.*]], <8 x half> noundef [[C:%.*]]) #[[ATTR0]] { -// UNCONSTRAINED-NEXT: [[ENTRY:.*:]] -// UNCONSTRAINED-NEXT: [[TMP0:%.*]] = bitcast <4 x half> [[A]] to <4 x i16> -// UNCONSTRAINED-NEXT: [[TMP1:%.*]] = bitcast <4 x half> [[B]] to <4 x i16> -// UNCONSTRAINED-NEXT: [[TMP2:%.*]] = bitcast <8 x half> [[C]] to <8 x i16> -// UNCONSTRAINED-NEXT: [[TMP3:%.*]] = bitcast <4 x i16> [[TMP0]] to <8 x i8> -// UNCONSTRAINED-NEXT: [[TMP4:%.*]] = bitcast <4 x i16> [[TMP1]] to <8 x i8> -// UNCONSTRAINED-NEXT: [[TMP5:%.*]] = bitcast <8 x i16> [[TMP2]] to <16 x i8> -// UNCONSTRAINED-NEXT: [[TMP6:%.*]] = bitcast <8 x i8> [[TMP3]] to <4 x half> -// UNCONSTRAINED-NEXT: [[TMP7:%.*]] = bitcast <8 x i8> [[TMP4]] to <4 x half> -// UNCONSTRAINED-NEXT: [[TMP8:%.*]] = bitcast <16 x i8> [[TMP5]] to <8 x half> -// UNCONSTRAINED-NEXT: [[LANE:%.*]] = shufflevector <8 x half> [[TMP8]], <8 x half> [[TMP8]], <4 x i32> <i32 7, i32 7, i32 7, i32 7> -// UNCONSTRAINED-NEXT: [[TMP9:%.*]] = call <4 x half> @llvm.fma.v4f16(<4 x half> [[LANE]], <4 x half> [[TMP7]], <4 x half> [[TMP6]]) -// UNCONSTRAINED-NEXT: ret <4 x half> [[TMP9]] -// -// CONSTRAINED-LABEL: define dso_local <4 x half> @test_vfma_laneq_f16( -// CONSTRAINED-SAME: <4 x half> noundef [[A:%.*]], <4 x half> noundef [[B:%.*]], <8 x half> noundef [[C:%.*]]) #[[ATTR0]] { -// CONSTRAINED-NEXT: [[ENTRY:.*:]] -// CONSTRAINED-NEXT: [[TMP0:%.*]] = bitcast <4 x half> [[A]] to <4 x i16> -// CONSTRAINED-NEXT: [[TMP1:%.*]] = bitcast <4 x half> [[B]] to <4 x i16> -// CONSTRAINED-NEXT: [[TMP2:%.*]] = bitcast <8 x half> [[C]] to <8 x i16> -// CONSTRAINED-NEXT: [[TMP3:%.*]] = bitcast <4 x i16> [[TMP0]] to <8 x i8> -// CONSTRAINED-NEXT: [[TMP4:%.*]] = bitcast <4 x i16> [[TMP1]] to <8 x i8> -// CONSTRAINED-NEXT: [[TMP5:%.*]] = bitcast <8 x i16> [[TMP2]] to <16 x i8> -// CONSTRAINED-NEXT: [[TMP6:%.*]] = bitcast <8 x i8> [[TMP3]] to <4 x half> -// CONSTRAINED-NEXT: [[TMP7:%.*]] = bitcast <8 x i8> [[TMP4]] to <4 x half> -// CONSTRAINED-NEXT: [[TMP8:%.*]] = bitcast <16 x i8> [[TMP5]] to <8 x half> -// CONSTRAINED-NEXT: [[LANE:%.*]] = shufflevector <8 x half> [[TMP8]], <8 x half> [[TMP8]], <4 x i32> <i32 7, i32 7, i32 7, i32 7> -// CONSTRAINED-NEXT: [[TMP9:%.*]] = call <4 x half> @llvm.experimental.constrained.fma.v4f16(<4 x half> [[LANE]], <4 x half> [[TMP7]], <4 x half> [[TMP6]], metadata !"round.tonearest", metadata !"fpexcept.strict") #[[ATTR2]] -// CONSTRAINED-NEXT: ret <4 x half> [[TMP9]] -// -float16x4_t test_vfma_laneq_f16(float16x4_t a, float16x4_t b, float16x8_t c) { - return vfma_laneq_f16(a, b, c, 7); -} - -// UNCONSTRAINED-LABEL: define dso_local <8 x half> @test_vfmaq_laneq_f16( -// UNCONSTRAINED-SAME: <8 x half> noundef [[A:%.*]], <8 x half> noundef [[B:%.*]], <8 x half> noundef [[C:%.*]]) #[[ATTR0]] { -// UNCONSTRAINED-NEXT: [[ENTRY:.*:]] -// UNCONSTRAINED-NEXT: [[TMP0:%.*]] = bitcast <8 x half> [[A]] to <8 x i16> -// UNCONSTRAINED-NEXT: [[TMP1:%.*]] = bitcast <8 x half> [[B]] to <8 x i16> -// UNCONSTRAINED-NEXT: [[TMP2:%.*]] = bitcast <8 x half> [[C]] to <8 x i16> -// UNCONSTRAINED-NEXT: [[TMP3:%.*]] = bitcast <8 x i16> [[TMP0]] to <16 x i8> -// UNCONSTRAINED-NEXT: [[TMP4:%.*]] = bitcast <8 x i16> [[TMP1]] to <16 x i8> -// UNCONSTRAINED-NEXT: [[TMP5:%.*]] = bitcast <8 x i16> [[TMP2]] to <16 x i8> -// UNCONSTRAINED-NEXT: [[TMP6:%.*]] = bitcast <16 x i8> [[TMP3]] to <8 x half> -// UNCONSTRAINED-NEXT: [[TMP7:%.*]] = bitcast <16 x i8> [[TMP4]] to <8 x half> -// UNCONSTRAINED-NEXT: [[TMP8:%.*]] = bitcast <16 x i8> [[TMP5]] to <8 x half> -// UNCONSTRAINED-NEXT: [[LANE:%.*]] = shufflevector <8 x half> [[TMP8]], <8 x half> [[TMP8]], <8 x i32> <i32 7, i32 7, i32 7, i32 7, i32 7, i32 7, i32 7, i32 7> -// UNCONSTRAINED-NEXT: [[TMP9:%.*]] = call <8 x half> @llvm.fma.v8f16(<8 x half> [[LANE]], <8 x half> [[TMP7]], <8 x half> [[TMP6]]) -// UNCONSTRAINED-NEXT: ret <8 x half> [[TMP9]] -// -// CONSTRAINED-LABEL: define dso_local <8 x half> @test_vfmaq_laneq_f16( -// CONSTRAINED-SAME: <8 x half> noundef [[A:%.*]], <8 x half> noundef [[B:%.*]], <8 x half> noundef [[C:%.*]]) #[[ATTR0]] { -// CONSTRAINED-NEXT: [[ENTRY:.*:]] -// CONSTRAINED-NEXT: [[TMP0:%.*]] = bitcast <8 x half> [[A]] to <8 x i16> -// CONSTRAINED-NEXT: [[TMP1:%.*]] = bitcast <8 x half> [[B]] to <8 x i16> -// CONSTRAINED-NEXT: [[TMP2:%.*]] = bitcast <8 x half> [[C]] to <8 x i16> -// CONSTRAINED-NEXT: [[TMP3:%.*]] = bitcast <8 x i16> [[TMP0]] to <16 x i8> -// CONSTRAINED-NEXT: [[TMP4:%.*]] = bitcast <8 x i16> [[TMP1]] to <16 x i8> -// CONSTRAINED-NEXT: [[TMP5:%.*]] = bitcast <8 x i16> [[TMP2]] to <16 x i8> -// CONSTRAINED-NEXT: [[TMP6:%.*]] = bitcast <16 x i8> [[TMP3]] to <8 x half> -// CONSTRAINED-NEXT: [[TMP7:%.*]] = bitcast <16 x i8> [[TMP4]] to <8 x half> -// CONSTRAINED-NEXT: [[TMP8:%.*]] = bitcast <16 x i8> [[TMP5]] to <8 x half> -// CONSTRAINED-NEXT: [[LANE:%.*]] = shufflevector <8 x half> [[TMP8]], <8 x half> [[TMP8]], <8 x i32> <i32 7, i32 7, i32 7, i32 7, i32 7, i32 7, i32 7, i32 7> -// CONSTRAINED-NEXT: [[TMP9:%.*]] = call <8 x half> @llvm.experimental.constrained.fma.v8f16(<8 x half> [[LANE]], <8 x half> [[TMP7]], <8 x half> [[TMP6]], metadata !"round.tonearest", metadata !"fpexcept.strict") #[[ATTR2]] -// CONSTRAINED-NEXT: ret <8 x half> [[TMP9]] -// -float16x8_t test_vfmaq_laneq_f16(float16x8_t a, float16x8_t b, float16x8_t c) { - return vfmaq_laneq_f16(a, b, c, 7); -} - // UNCONSTRAINED-LABEL: define dso_local <4 x half> @test_vfma_n_f16( // UNCONSTRAINED-SAME: <4 x half> noundef [[A:%.*]], <4 x half> noundef [[B:%.*]], half noundef [[C:%.*]]) #[[ATTR0]] { // UNCONSTRAINED-NEXT: [[ENTRY:.*:]] @@ -441,24 +185,6 @@ float16x8_t test_vfmaq_n_f16(float16x8_t a, float16x8_t b, float16_t c) { return vfmaq_n_f16(a, b, c); } -// UNCONSTRAINED-LABEL: define dso_local half @test_vfmah_lane_f16( -// UNCONSTRAINED-SAME: half noundef [[A:%.*]], half noundef [[B:%.*]], <4 x half> noundef [[C:%.*]]) #[[ATTR0]] { -// UNCONSTRAINED-NEXT: [[ENTRY:.*:]] -// UNCONSTRAINED-NEXT: [[EXTRACT:%.*]] = extractelement <4 x half> [[C]], i32 3 -// UNCONSTRAINED-NEXT: [[TMP0:%.*]] = call half @llvm.fma.f16(half [[B]], half [[EXTRACT]], half [[A]]) -// UNCONSTRAINED-NEXT: ret half [[TMP0]] -// -// CONSTRAINED-LABEL: define dso_local half @test_vfmah_lane_f16( -// CONSTRAINED-SAME: half noundef [[A:%.*]], half noundef [[B:%.*]], <4 x half> noundef [[C:%.*]]) #[[ATTR0]] { -// CONSTRAINED-NEXT: [[ENTRY:.*:]] -// CONSTRAINED-NEXT: [[EXTRACT:%.*]] = extractelement <4 x half> [[C]], i32 3 -// CONSTRAINED-NEXT: [[TMP0:%.*]] = call half @llvm.experimental.constrained.fma.f16(half [[B]], half [[EXTRACT]], half [[A]], metadata !"round.tonearest", metadata !"fpexcept.strict") #[[ATTR2]] -// CONSTRAINED-NEXT: ret half [[TMP0]] -// -float16_t test_vfmah_lane_f16(float16_t a, float16_t b, float16x4_t c) { - return vfmah_lane_f16(a, b, c, 3); -} - // UNCONSTRAINED-LABEL: define dso_local half @test_vfmah_laneq_f16( // UNCONSTRAINED-SAME: half noundef [[A:%.*]], half noundef [[B:%.*]], <8 x half> noundef [[C:%.*]]) #[[ATTR0]] { // UNCONSTRAINED-NEXT: [[ENTRY:.*:]] _______________________________________________ cfe-commits mailing list [email protected] https://lists.llvm.org/cgi-bin/mailman/listinfo/cfe-commits
