llvmorg-github-actions[bot] wrote:
<!--LLVM PR SUMMARY COMMENT--> @llvm/pr-subscribers-llvm-transforms Author: Divyansh Yadav (schizophrenicmaniac) <details> <summary>Changes</summary> Fixes #<!-- -->226631 This PR resolves a segmentation fault caused by a translation unit (TU) attempting to write to `.rodata` during dynamic initialization. ### Description When a static local variable of an inline function is evaluated, Clang may optimize it into a constant and place it in the `.rodata` section. However, if the variable relies on a value not visible across all translation units (and therefore fails to meet standard C++ rules for formal constant-initialization), another TU may fail to constant-fold it. That TU will instead emit a guard variable and a dynamic initializer that attempts to write to the variable at runtime. When linked together, if the `.rodata` definition from the first TU is chosen, the dynamic initializer from the second TU will cause a segmentation fault upon trying to write to read-only memory. ### Changes * Updated `CodeGenModule::EmitGlobalVarDefinition` and `CodeGenFunction::AddInitializerToStaticVarDecl` to verify `!hasConstantInitialization()` before marking a weak global/local static as `constant`. * If a variable is `isWeakForLinker()` and hasn't been formally constant-initialized across all TUs, it correctly defaults to a mutable data section (e.g. `.bss` or `.data`), avoiding the segfault. --- Patch is 88.78 KiB, truncated to 20.00 KiB below, full version: https://github.com/llvm/llvm-project/pull/226645.diff 15 Files Affected: - (modified) clang/lib/CodeGen/CGDecl.cpp (+6-2) - (modified) clang/lib/CodeGen/CodeGenModule.cpp (+7-2) - (modified) llvm/lib/Target/X86/X86ISelLowering.cpp (+70-1) - (modified) llvm/test/Analysis/CostModel/X86/fround.ll (+9-9) - (modified) llvm/test/CodeGen/X86/fp-round.ll (+284-249) - (modified) llvm/test/CodeGen/X86/fp16-libcalls.ll (+28-4) - (modified) llvm/test/CodeGen/X86/freeze-unary.ll (+32-4) - (modified) llvm/test/CodeGen/X86/ftrunc.ll (+190-147) - (modified) llvm/test/CodeGen/X86/isel-ftrunc.ll (+33-20) - (modified) llvm/test/CodeGen/X86/isint.ll (+57-13) - (modified) llvm/test/CodeGen/X86/llround-conv.ll (+28-2) - (modified) llvm/test/CodeGen/X86/lround-conv-i32.ll (+34-6) - (modified) llvm/test/CodeGen/X86/lround-conv-i64.ll (+14-1) - (modified) llvm/test/CodeGen/X86/unpredictable-brcond.ll (+16-4) - (modified) llvm/test/Transforms/SLPVectorizer/X86/fround.ll (+42-126) ``````````diff diff --git a/clang/lib/CodeGen/CGDecl.cpp b/clang/lib/CodeGen/CGDecl.cpp index e1ed66ae71243..1e9958b167030 100644 --- a/clang/lib/CodeGen/CGDecl.cpp +++ b/clang/lib/CodeGen/CGDecl.cpp @@ -393,8 +393,12 @@ CodeGenFunction::AddInitializerToStaticVarDecl(const VarDecl &D, bool NeedsDtor = D.needsDestruction(getContext()) == QualType::DK_cxx_destructor; - GV->setConstant( - D.getType().isConstantStorage(getContext(), true, !NeedsDtor)); + bool IsConstant = + D.getType().isConstantStorage(getContext(), true, !NeedsDtor); + if (IsConstant && GV->isWeakForLinker() && !D.hasConstantInitialization()) + IsConstant = false; + + GV->setConstant(IsConstant); GV->replaceInitializer(Init); emitter.finalize(GV); diff --git a/clang/lib/CodeGen/CodeGenModule.cpp b/clang/lib/CodeGen/CodeGenModule.cpp index 7274a8588670f..eb0118e8b41bd 100644 --- a/clang/lib/CodeGen/CodeGenModule.cpp +++ b/clang/lib/CodeGen/CodeGenModule.cpp @@ -6774,9 +6774,14 @@ void CodeGenModule::EmitGlobalVarDefinition(const VarDecl *D, emitter->finalize(GV); // If it is safe to mark the global 'constant', do so now. + bool IsConstant = !NeedsGlobalCtor && !NeedsGlobalDtor && + D->getType().isConstantStorage(getContext(), true, true); + if (IsConstant && GV->isWeakForLinker() && !D->hasConstantInitialization() && + !D->hasAttr<CUDAConstantAttr>()) + IsConstant = false; + GV->setConstant((D->hasAttr<CUDAConstantAttr>() && LangOpts.CUDAIsDevice) || - (!NeedsGlobalCtor && !NeedsGlobalDtor && - D->getType().isConstantStorage(getContext(), true, true))); + IsConstant); // If it is in a read-only section, mark it 'constant'. if (const SectionAttr *SA = D->getAttr<SectionAttr>()) { diff --git a/llvm/lib/Target/X86/X86ISelLowering.cpp b/llvm/lib/Target/X86/X86ISelLowering.cpp index e2f3b5f3cd3d5..9db2ab672eadd 100644 --- a/llvm/lib/Target/X86/X86ISelLowering.cpp +++ b/llvm/lib/Target/X86/X86ISelLowering.cpp @@ -632,6 +632,11 @@ X86TargetLowering::X86TargetLowering(const X86TargetMachine &TM, setLoadExtAction(ISD::EXTLOAD, MVT::f64, MVT::f32, Expand); for (auto VT : { MVT::f32, MVT::f64 }) { + if (VT == MVT::f32 || Subtarget.is64Bit()) { + setOperationAction(ISD::FTRUNC, VT, Custom); + setOperationAction(ISD::FROUND, VT, Custom); + } + // Use ANDPD to simulate FABS. setOperationAction(ISD::FABS, VT, Custom); @@ -1096,6 +1101,13 @@ X86TargetLowering::X86TargetLowering(const X86TargetMachine &TM, for (auto VT : { MVT::f64, MVT::v4f32, MVT::v2f64 }) SetFPMinMaxAction(VT); + for (auto VT : {MVT::v4f32, MVT::v2f64}) { + if (VT == MVT::v4f32 || Subtarget.is64Bit()) { + setOperationAction(ISD::FTRUNC, VT, Custom); + setOperationAction(ISD::FROUND, VT, Custom); + } + } + setOperationAction(ISD::MUL, MVT::v2i8, Custom); setOperationAction(ISD::MUL, MVT::v4i8, Custom); setOperationAction(ISD::MUL, MVT::v8i8, Custom); @@ -23178,6 +23190,54 @@ SDValue X86TargetLowering::lowerFaddFsub(SDValue Op, SelectionDAG &DAG) const { return lowerAddSubToHorizontalOp(Op, SDLoc(Op), DAG, Subtarget); } +static SDValue lowerFTRUNC_FROUND_SSE2(SDValue Op, SelectionDAG &DAG) { + SDLoc DL(Op); + SDValue N0 = Op.getOperand(0); + MVT VT = Op.getSimpleValueType(); + bool IsRound = Op.getOpcode() == ISD::FROUND; + + SDValue Abs = DAG.getNode(ISD::FABS, DL, VT, N0); + SDValue AbsBiased = Abs; + if (IsRound) { + const fltSemantics &Sem = VT.getFltSemantics(); + APFloat Bias = APFloat(0.5f); + bool Ignored; + Bias.convert(Sem, APFloat::rmNearestTiesToEven, &Ignored); + Bias.next(/*nextDown*/ true); + AbsBiased = + DAG.getNode(ISD::FADD, DL, VT, Abs, DAG.getConstantFP(Bias, DL, VT)); + } + + MVT IntVT; + if (VT == MVT::f32) + IntVT = MVT::i32; + else if (VT == MVT::f64) + IntVT = MVT::i64; + else if (VT == MVT::v4f32) + IntVT = MVT::v4i32; + else if (VT == MVT::v2f64) + IntVT = MVT::v2i64; + else + llvm_unreachable("Unexpected type"); + + const fltSemantics &Sem = VT.getFltSemantics(); + APFloat Bound = VT.getScalarType() == MVT::f32 ? APFloat(Sem, "0x1.0p23") + : APFloat(Sem, "0x1.0p52"); + SDValue Threshold = DAG.getConstantFP(Bound, DL, VT); + + EVT CCVT = DAG.getTargetLoweringInfo().getSetCCResultType( + DAG.getDataLayout(), *DAG.getContext(), VT); + SDValue IsLarge = DAG.getSetCC(DL, CCVT, Abs, Threshold, ISD::SETUGE); + + SDValue TruncInt = DAG.getNode(ISD::FP_TO_SINT, DL, IntVT, AbsBiased); + SDValue AbsTrunc = DAG.getNode(ISD::SINT_TO_FP, DL, VT, TruncInt); + + SDValue Trunc = DAG.getNode(ISD::FCOPYSIGN, DL, VT, AbsTrunc, N0); + + unsigned SelOpc = VT.isVector() ? ISD::VSELECT : ISD::SELECT; + return DAG.getNode(SelOpc, DL, VT, IsLarge, N0, Trunc); +} + /// ISD::FROUND is defined to round to nearest with ties rounding away from 0. /// This mode isn't supported in hardware on X86. But as long as we aren't /// compiling with trapping math, we can emulate this with @@ -34849,7 +34909,16 @@ SDValue X86TargetLowering::LowerOperation(SDValue Op, SelectionDAG &DAG) const { case ISD::STORE: return LowerStore(Op, Subtarget, DAG); case ISD::FADD: case ISD::FSUB: return lowerFaddFsub(Op, DAG); - case ISD::FROUND: return LowerFROUND(Op, DAG); + case ISD::FTRUNC: + case ISD::FROUND: { + MVT VT = Op.getSimpleValueType(); + if (Subtarget.hasSSE2() && !Subtarget.hasSSE41() && + (VT == MVT::f32 || VT == MVT::v4f32 || Subtarget.is64Bit())) + return lowerFTRUNC_FROUND_SSE2(Op, DAG); + if (Op.getOpcode() == ISD::FROUND) + return LowerFROUND(Op, DAG); + return SDValue(); + } case ISD::FABS: case ISD::FNEG: return LowerFABSorFNEG(Op, DAG); case ISD::FCOPYSIGN: return LowerFCOPYSIGN(Op, DAG); diff --git a/llvm/test/Analysis/CostModel/X86/fround.ll b/llvm/test/Analysis/CostModel/X86/fround.ll index ed7965af202d9..e40319462b116 100644 --- a/llvm/test/Analysis/CostModel/X86/fround.ll +++ b/llvm/test/Analysis/CostModel/X86/fround.ll @@ -267,15 +267,15 @@ define i32 @rint(i32 %arg) { define i32 @trunc(i32 %arg) { ; SSE2-LABEL: 'trunc' -; SSE2-NEXT: Cost Model: Found an estimated cost of 10 for instruction: %F32 = call float @llvm.trunc.f32(float undef) -; SSE2-NEXT: Cost Model: Found an estimated cost of 21 for instruction: %V2F32 = call <2 x float> @llvm.trunc.v2f32(<2 x float> undef) -; SSE2-NEXT: Cost Model: Found an estimated cost of 43 for instruction: %V4F32 = call <4 x float> @llvm.trunc.v4f32(<4 x float> undef) -; SSE2-NEXT: Cost Model: Found an estimated cost of 86 for instruction: %V8F32 = call <8 x float> @llvm.trunc.v8f32(<8 x float> undef) -; SSE2-NEXT: Cost Model: Found an estimated cost of 172 for instruction: %V16F32 = call <16 x float> @llvm.trunc.v16f32(<16 x float> undef) -; SSE2-NEXT: Cost Model: Found an estimated cost of 10 for instruction: %F64 = call double @llvm.trunc.f64(double undef) -; SSE2-NEXT: Cost Model: Found an estimated cost of 21 for instruction: %V2F64 = call <2 x double> @llvm.trunc.v2f64(<2 x double> undef) -; SSE2-NEXT: Cost Model: Found an estimated cost of 42 for instruction: %V4F64 = call <4 x double> @llvm.trunc.v4f64(<4 x double> undef) -; SSE2-NEXT: Cost Model: Found an estimated cost of 84 for instruction: %V8F64 = call <8 x double> @llvm.trunc.v8f64(<8 x double> undef) +; SSE2-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %F32 = call float @llvm.trunc.f32(float undef) +; SSE2-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %V2F32 = call <2 x float> @llvm.trunc.v2f32(<2 x float> undef) +; SSE2-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %V4F32 = call <4 x float> @llvm.trunc.v4f32(<4 x float> undef) +; SSE2-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V8F32 = call <8 x float> @llvm.trunc.v8f32(<8 x float> undef) +; SSE2-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V16F32 = call <16 x float> @llvm.trunc.v16f32(<16 x float> undef) +; SSE2-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %F64 = call double @llvm.trunc.f64(double undef) +; SSE2-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %V2F64 = call <2 x double> @llvm.trunc.v2f64(<2 x double> undef) +; SSE2-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %V4F64 = call <4 x double> @llvm.trunc.v4f64(<4 x double> undef) +; SSE2-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %V8F64 = call <8 x double> @llvm.trunc.v8f64(<8 x double> undef) ; SSE2-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret i32 undef ; ; SSE42-LABEL: 'trunc' diff --git a/llvm/test/CodeGen/X86/fp-round.ll b/llvm/test/CodeGen/X86/fp-round.ll index 8595b63fc8107..797e8fbe7dafd 100644 --- a/llvm/test/CodeGen/X86/fp-round.ll +++ b/llvm/test/CodeGen/X86/fp-round.ll @@ -11,7 +11,20 @@ define half @round_f16(half %h) { ; SSE2-NEXT: pushq %rax ; SSE2-NEXT: .cfi_def_cfa_offset 16 ; SSE2-NEXT: callq __extendhfsf2@PLT -; SSE2-NEXT: callq roundf@PLT +; SSE2-NEXT: movaps {{.*#+}} xmm1 = [NaN,NaN,NaN,NaN] +; SSE2-NEXT: movaps %xmm0, %xmm2 +; SSE2-NEXT: andps %xmm1, %xmm2 +; SSE2-NEXT: movss {{.*#+}} xmm3 = [4.9999997E-1,0.0E+0,0.0E+0,0.0E+0] +; SSE2-NEXT: addss %xmm2, %xmm3 +; SSE2-NEXT: cvttps2dq %xmm3, %xmm3 +; SSE2-NEXT: cvtdq2ps %xmm3, %xmm3 +; SSE2-NEXT: andps %xmm1, %xmm3 +; SSE2-NEXT: andnps %xmm0, %xmm1 +; SSE2-NEXT: orps %xmm1, %xmm3 +; SSE2-NEXT: cmpnltss {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %xmm2 +; SSE2-NEXT: andps %xmm2, %xmm0 +; SSE2-NEXT: andnps %xmm3, %xmm2 +; SSE2-NEXT: orps %xmm2, %xmm0 ; SSE2-NEXT: callq __truncsfhf2@PLT ; SSE2-NEXT: popq %rax ; SSE2-NEXT: .cfi_def_cfa_offset 8 @@ -74,7 +87,21 @@ entry: define float @round_f32(float %x) { ; SSE2-LABEL: round_f32: ; SSE2: # %bb.0: -; SSE2-NEXT: jmp roundf@PLT # TAILCALL +; SSE2-NEXT: movaps {{.*#+}} xmm1 = [NaN,NaN,NaN,NaN] +; SSE2-NEXT: movaps %xmm0, %xmm2 +; SSE2-NEXT: andps %xmm1, %xmm2 +; SSE2-NEXT: movss {{.*#+}} xmm3 = [4.9999997E-1,0.0E+0,0.0E+0,0.0E+0] +; SSE2-NEXT: addss %xmm2, %xmm3 +; SSE2-NEXT: cvttps2dq %xmm3, %xmm3 +; SSE2-NEXT: cvtdq2ps %xmm3, %xmm3 +; SSE2-NEXT: andps %xmm1, %xmm3 +; SSE2-NEXT: andnps %xmm0, %xmm1 +; SSE2-NEXT: orps %xmm1, %xmm3 +; SSE2-NEXT: cmpnltss {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %xmm2 +; SSE2-NEXT: andps %xmm2, %xmm0 +; SSE2-NEXT: andnps %xmm3, %xmm2 +; SSE2-NEXT: orps %xmm2, %xmm0 +; SSE2-NEXT: retq ; ; SSE41-LABEL: round_f32: ; SSE41: # %bb.0: @@ -117,7 +144,22 @@ define float @round_f32(float %x) { define double @round_f64(double %x) { ; SSE2-LABEL: round_f64: ; SSE2: # %bb.0: -; SSE2-NEXT: jmp round@PLT # TAILCALL +; SSE2-NEXT: movapd {{.*#+}} xmm1 = [NaN,NaN] +; SSE2-NEXT: movapd %xmm0, %xmm2 +; SSE2-NEXT: andpd %xmm1, %xmm2 +; SSE2-NEXT: movsd {{.*#+}} xmm3 = [4.9999999999999994E-1,0.0E+0] +; SSE2-NEXT: addsd %xmm2, %xmm3 +; SSE2-NEXT: cvttsd2si %xmm3, %rax +; SSE2-NEXT: xorps %xmm3, %xmm3 +; SSE2-NEXT: cvtsi2sd %rax, %xmm3 +; SSE2-NEXT: andpd %xmm1, %xmm3 +; SSE2-NEXT: andnpd %xmm0, %xmm1 +; SSE2-NEXT: orpd %xmm1, %xmm3 +; SSE2-NEXT: cmpnltsd {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %xmm2 +; SSE2-NEXT: andpd %xmm2, %xmm0 +; SSE2-NEXT: andnpd %xmm3, %xmm2 +; SSE2-NEXT: orpd %xmm2, %xmm0 +; SSE2-NEXT: retq ; ; SSE41-LABEL: round_f64: ; SSE41: # %bb.0: @@ -161,31 +203,20 @@ define double @round_f64(double %x) { define <4 x float> @round_v4f32(<4 x float> %x) { ; SSE2-LABEL: round_v4f32: ; SSE2: # %bb.0: -; SSE2-NEXT: subq $56, %rsp -; SSE2-NEXT: .cfi_def_cfa_offset 64 -; SSE2-NEXT: movaps %xmm0, {{[-0-9]+}}(%r{{[sb]}}p) # 16-byte Spill -; SSE2-NEXT: shufps {{.*#+}} xmm0 = xmm0[3,3,3,3] -; SSE2-NEXT: callq roundf@PLT -; SSE2-NEXT: movaps %xmm0, (%rsp) # 16-byte Spill -; SSE2-NEXT: movaps {{[-0-9]+}}(%r{{[sb]}}p), %xmm0 # 16-byte Reload -; SSE2-NEXT: movhlps {{.*#+}} xmm0 = xmm0[1,1] -; SSE2-NEXT: callq roundf@PLT -; SSE2-NEXT: unpcklps (%rsp), %xmm0 # 16-byte Folded Reload -; SSE2-NEXT: # xmm0 = xmm0[0],mem[0],xmm0[1],mem[1] -; SSE2-NEXT: movaps %xmm0, (%rsp) # 16-byte Spill -; SSE2-NEXT: movaps {{[-0-9]+}}(%r{{[sb]}}p), %xmm0 # 16-byte Reload -; SSE2-NEXT: callq roundf@PLT -; SSE2-NEXT: movaps %xmm0, {{[-0-9]+}}(%r{{[sb]}}p) # 16-byte Spill -; SSE2-NEXT: movaps {{[-0-9]+}}(%r{{[sb]}}p), %xmm0 # 16-byte Reload -; SSE2-NEXT: shufps {{.*#+}} xmm0 = xmm0[1,1,1,1] -; SSE2-NEXT: callq roundf@PLT -; SSE2-NEXT: movaps {{[-0-9]+}}(%r{{[sb]}}p), %xmm1 # 16-byte Reload -; SSE2-NEXT: unpcklps {{.*#+}} xmm1 = xmm1[0],xmm0[0],xmm1[1],xmm0[1] -; SSE2-NEXT: unpcklpd (%rsp), %xmm1 # 16-byte Folded Reload -; SSE2-NEXT: # xmm1 = xmm1[0],mem[0] -; SSE2-NEXT: movaps %xmm1, %xmm0 -; SSE2-NEXT: addq $56, %rsp -; SSE2-NEXT: .cfi_def_cfa_offset 8 +; SSE2-NEXT: movaps {{.*#+}} xmm1 = [NaN,NaN,NaN,NaN] +; SSE2-NEXT: movaps %xmm0, %xmm2 +; SSE2-NEXT: andps %xmm1, %xmm2 +; SSE2-NEXT: movaps {{.*#+}} xmm3 = [4.9999997E-1,4.9999997E-1,4.9999997E-1,4.9999997E-1] +; SSE2-NEXT: addps %xmm2, %xmm3 +; SSE2-NEXT: cvttps2dq %xmm3, %xmm3 +; SSE2-NEXT: cvtdq2ps %xmm3, %xmm3 +; SSE2-NEXT: andps %xmm1, %xmm3 +; SSE2-NEXT: andnps %xmm0, %xmm1 +; SSE2-NEXT: orps %xmm1, %xmm3 +; SSE2-NEXT: cmpnltps {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %xmm2 +; SSE2-NEXT: andps %xmm2, %xmm0 +; SSE2-NEXT: andnps %xmm3, %xmm2 +; SSE2-NEXT: orps %xmm2, %xmm0 ; SSE2-NEXT: retq ; ; SSE41-LABEL: round_v4f32: @@ -227,19 +258,25 @@ define <4 x float> @round_v4f32(<4 x float> %x) { define <2 x double> @round_v2f64(<2 x double> %x) { ; SSE2-LABEL: round_v2f64: ; SSE2: # %bb.0: -; SSE2-NEXT: subq $40, %rsp -; SSE2-NEXT: .cfi_def_cfa_offset 48 -; SSE2-NEXT: movaps %xmm0, (%rsp) # 16-byte Spill -; SSE2-NEXT: callq round@PLT -; SSE2-NEXT: movaps %xmm0, {{[-0-9]+}}(%r{{[sb]}}p) # 16-byte Spill -; SSE2-NEXT: movaps (%rsp), %xmm0 # 16-byte Reload -; SSE2-NEXT: movhlps {{.*#+}} xmm0 = xmm0[1,1] -; SSE2-NEXT: callq round@PLT -; SSE2-NEXT: movaps {{[-0-9]+}}(%r{{[sb]}}p), %xmm1 # 16-byte Reload -; SSE2-NEXT: movlhps {{.*#+}} xmm1 = xmm1[0],xmm0[0] -; SSE2-NEXT: movaps %xmm1, %xmm0 -; SSE2-NEXT: addq $40, %rsp -; SSE2-NEXT: .cfi_def_cfa_offset 8 +; SSE2-NEXT: movapd {{.*#+}} xmm1 = [NaN,NaN] +; SSE2-NEXT: movapd %xmm0, %xmm2 +; SSE2-NEXT: andpd %xmm1, %xmm2 +; SSE2-NEXT: movapd {{.*#+}} xmm3 = [4.9999999999999994E-1,4.9999999999999994E-1] +; SSE2-NEXT: addpd %xmm2, %xmm3 +; SSE2-NEXT: cvttsd2si %xmm3, %rax +; SSE2-NEXT: cvtsi2sd %rax, %xmm4 +; SSE2-NEXT: unpckhpd {{.*#+}} xmm3 = xmm3[1,1] +; SSE2-NEXT: cvttsd2si %xmm3, %rax +; SSE2-NEXT: xorps %xmm3, %xmm3 +; SSE2-NEXT: cvtsi2sd %rax, %xmm3 +; SSE2-NEXT: unpcklpd {{.*#+}} xmm4 = xmm4[0],xmm3[0] +; SSE2-NEXT: andpd %xmm1, %xmm4 +; SSE2-NEXT: andnpd %xmm0, %xmm1 +; SSE2-NEXT: orpd %xmm4, %xmm1 +; SSE2-NEXT: cmpnltpd {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %xmm2 +; SSE2-NEXT: andpd %xmm2, %xmm0 +; SSE2-NEXT: andnpd %xmm1, %xmm2 +; SSE2-NEXT: orpd %xmm2, %xmm0 ; SSE2-NEXT: retq ; ; SSE41-LABEL: round_v2f64: @@ -281,53 +318,35 @@ define <2 x double> @round_v2f64(<2 x double> %x) { define <8 x float> @round_v8f32(<8 x float> %x) { ; SSE2-LABEL: round_v8f32: ; SSE2: # %bb.0: -; SSE2-NEXT: subq $72, %rsp -; SSE2-NEXT: .cfi_def_cfa_offset 80 -; SSE2-NEXT: movaps %xmm1, {{[-0-9]+}}(%r{{[sb]}}p) # 16-byte Spill -; SSE2-NEXT: movaps %xmm0, (%rsp) # 16-byte Spill -; SSE2-NEXT: shufps {{.*#+}} xmm0 = xmm0[3,3,3,3] -; SSE2-NEXT: callq roundf@PLT -; SSE2-NEXT: movaps %xmm0, {{[-0-9]+}}(%r{{[sb]}}p) # 16-byte Spill -; SSE2-NEXT: movaps (%rsp), %xmm0 # 16-byte Reload -; SSE2-NEXT: movhlps {{.*#+}} xmm0 = xmm0[1,1] -; SSE2-NEXT: callq roundf@PLT -; SSE2-NEXT: unpcklps {{[-0-9]+}}(%r{{[sb]}}p), %xmm0 # 16-byte Folded Reload -; SSE2-NEXT: # xmm0 = xmm0[0],mem[0],xmm0[1],mem[1] -; SSE2-NEXT: movaps %xmm0, {{[-0-9]+}}(%r{{[sb]}}p) # 16-byte Spill -; SSE2-NEXT: movaps (%rsp), %xmm0 # 16-byte Reload -; SSE2-NEXT: callq roundf@PLT -; SSE2-NEXT: movaps %xmm0, {{[-0-9]+}}(%r{{[sb]}}p) # 16-byte Spill -; SSE2-NEXT: movaps (%rsp), %xmm0 # 16-byte Reload -; SSE2-NEXT: shufps {{.*#+}} xmm0 = xmm0[1,1,1,1] -; SSE2-NEXT: callq roundf@PLT -; SSE2-NEXT: movaps {{[-0-9]+}}(%r{{[sb]}}p), %xmm1 # 16-byte Reload -; SSE2-NEXT: unpcklps {{.*#+}} xmm1 = xmm1[0],xmm0[0],xmm1[1],xmm0[1] -; SSE2-NEXT: unpcklpd {{[-0-9]+}}(%r{{[sb]}}p), %xmm1 # 16-byte Folded Reload -; SSE2-NEXT: # xmm1 = xmm1[0],mem[0] -; SSE2-NEXT: movaps %xmm1, {{[-0-9]+}}(%r{{[sb]}}p) # 16-byte Spill -; SSE2-NEXT: movaps {{[-0-9]+}}(%r{{[sb]}}p), %xmm0 # 16-byte Reload -; SSE2-NEXT: shufps {{.*#+}} xmm0 = xmm0[3,3,3,3] -; SSE2-NEXT: callq roundf@PLT -; SSE2-NEXT: movaps %xmm0, (%rsp) # 16-byte Spill -; SSE2-NEXT: movaps {{[-0-9]+}}(%r{{[sb]}}p), %xmm0 # 16-byte Reload -; SSE2-NEXT: movhlps {{.*#+}} xmm0 = xmm0[1,1] -; SSE2-NEXT: callq roundf@PLT -; SSE2-NEXT: unpcklps (%rsp), %xmm0 # 16-byte Folded Reload -; SSE2-NEXT: # xmm0 = xmm0[0],mem[0],xmm0[1],mem[1] -; SSE2-NEXT: movaps %xmm0, (%rsp) # 16-byte Spill -; SSE2-NEXT: movaps {{[-0-9]+}}(%r{{[sb]}}p), %xmm0 # 16-byte Reload -; SSE2-NEXT: callq roundf@PLT -; SSE2-NEXT: movaps %xmm0, {{[-0-9]+}}(%r{{[sb]}}p) # 16-byte Spill -; SSE2-NEXT: movaps {{[-0-9]+}}(%r{{[sb]}}p), %xmm0 # 16-byte Reload -; SSE2-NEXT: shufps {{.*#+}} xmm0 = xmm0[1,1,1,1] -; SSE2-NEXT: callq roundf@PLT -; SSE2-NEXT: movaps {{[-0-9]+}}(%r{{[sb]}}p), %xmm1 # 16-byte Reload -; SSE2-NEXT: unpcklps {{.*#+}} xmm1 = xmm1[0],xmm0[0],xmm1[1],xmm0[1] -; SSE2-NEXT: unpcklpd (%rsp), %xmm1 # 16-byte Folded Reload -; SSE2-NEXT: # xmm1 = xmm1[0],mem[0] -; SSE2-NEXT: movaps {{[-0-9]+}}(%r{{[sb]}}p), %xmm0 # 16-byte Reload -; SSE2-NEXT: addq $72, %rsp -; SSE2-NEXT: .cfi_def_cfa_offset 8 +; SSE2-NEXT: movaps {{.*#+}} xmm3 = [NaN,NaN,NaN,NaN] +; SSE2-NEXT: movaps %xmm0, %xmm2 +; SSE2-NEXT: andps %xmm3, %xmm2 +; SSE2-NEXT: movaps {{.*#+}} xmm4 = [4.9999997E-1,4.9999997E-1,4.9999997E-1,4.9999997E-1] +; SSE2-NEXT: movaps %xmm2, %xmm5 +; SSE2-NEXT: addps %xmm4, %xmm5 +; SSE2-NEXT: cvttps2dq %xmm5, %xmm5 +; SSE2-NEXT: cvtdq2ps %xmm5, %xmm5 +; SSE2-NEXT: andps %xmm3, %xmm5 +; SSE2-NEXT: movaps %xmm3, %xmm6 +; SSE2-NEXT: movaps %xmm1, %xmm7 +; SSE2-NEXT: andps %xmm3, %xmm7 +; SSE2-NEXT: addps %xmm7, %xmm4 +; SSE2-NEXT: cvttps2dq %xmm4, %xmm4 +; SSE2-NEXT: cvtdq2ps %xmm4, %xmm4 +; SSE2-NEXT: andps %xmm3, %xmm4 +; SSE2-NEXT: andnps %xmm0, %xmm3 +; SSE2-NEXT: orps %xmm3, %xmm5 +; SSE2-NEXT: movaps {{.*#+}} xmm3 = [8.388608E+6,8.388608E+6,8.388608E+6,8.388608E+6] +; SSE2-NEXT: cmpnltps %xmm3, %xmm2 +; SSE2-NEXT: andps %xmm2, %xmm0 +; SSE2-NEXT: andnps %xmm5, %xmm2 +; SSE2-NEXT: orps %xmm2, %xmm0 +; SSE2-NEXT: andnps %xmm1, %xmm6 +; SSE2-NEXT: orps %xmm6, %xmm4 +; SSE2-NEXT: cmpnltps %xmm3, %xmm7 +; SSE2-NEXT: andps %xmm7, %xmm1 +; SSE2-NEXT: andnps %xmm4, %xmm7 +; SSE2-NEXT: orps %xmm7, %xmm1 ; SSE2-NEXT: retq ; ; SSE41-LABEL: round_v8f32: @@ -375,29 +394,46 @@ define <8 x float> @round_v8f32(<8 x float> %x) { define <4 x double> @round_v4f64(<4 x double> %x) { ; SSE2-LABEL: round_v4f64: ; SSE2: # %bb.0: -; SSE2-NEXT: subq $56, %rsp -; SSE2-NEXT: .cfi_def_cfa_offset 64 -; SSE2-NEXT: movaps %xmm1, {{[-0-9]+}}(%r{{[sb]}}p) # 16-byte Spill -; SSE2-NEXT: movaps %xmm0, (%rsp) # 16-byte Spill -; SSE2-NEXT: callq round@PLT -; SSE2-NEXT: movaps %xmm0, {{[-0-9]+}}(%r{{[sb]}}p) # 16-byte Spill -; SSE2-NEXT: movaps (%rsp), %xmm0 # 16-byte Reload -; SSE2-NEXT: movhlps {{.*#+}} xmm0 = xmm0[1,1] -; SSE2-NEXT: callq round@PLT -; SSE2-NEXT: movaps {{[-0-9]+}}(%r{{[sb]}}p), %xmm1 # 16-byte Reload -; SSE2-NEXT: movlhps {{.*#+}} xmm1 = xmm1[0],xmm0[0] -; SSE2-NEXT: movaps %xmm1, {{[-0-9]+}}(%r{{[sb]}}p) # 16-byte Spill -; SSE2-NEXT: movaps {{[-0-9]+}}(%... [truncated] `````````` </details> https://github.com/llvm/llvm-project/pull/226645 _______________________________________________ cfe-commits mailing list [email protected] https://lists.llvm.org/cgi-bin/mailman/listinfo/cfe-commits
