https://github.com/bob80905 updated https://github.com/llvm/llvm-project/pull/222163
>From 945e156c2ed79bc8a96cb2f095447482f4dcffb8 Mon Sep 17 00:00:00 2001 From: Joshua Batista <[email protected]> Date: Fri, 4 Sep 2026 14:54:47 -0700 Subject: [PATCH 1/6] First attempt implementing float InterlockedExchange --- clang/include/clang/Basic/Builtins.td | 3 +- .../clang/Basic/DiagnosticSemaKinds.td | 2 +- clang/lib/CodeGen/CGHLSLBuiltins.cpp | 9 ++++- clang/lib/Sema/HLSLBuiltinTypeDeclBuilder.cpp | 6 +++ clang/lib/Sema/HLSLExternalSemaSource.cpp | 14 +++++-- clang/lib/Sema/SemaHLSL.cpp | 12 ++++-- .../builtins/InterlockedExchange.hlsl | 11 +++++ ...ByteAddressBuffer-InterlockedExchange.hlsl | 16 ++++++++ ...sBuffer-InterlockedExchangeFloat-sm60.hlsl | 40 +++++++++++++++++++ .../BuiltIns/InterlockedExchange-errors.hlsl | 35 +++++++++------- llvm/lib/Target/DirectX/DXILLegalizePass.cpp | 29 ++++++++++++++ .../lib/Target/DirectX/DXILResourceAccess.cpp | 20 ++++++++-- .../DirectX/LegalizeAtomicExchangeFloat.ll | 39 ++++++++++++++++++ .../DirectX/ResourceAtomicExchangeFloat.ll | 38 ++++++++++++++++++ 14 files changed, 246 insertions(+), 28 deletions(-) create mode 100644 clang/test/SemaHLSL/BuiltIns/ByteAddressBuffer-InterlockedExchangeFloat-sm60.hlsl create mode 100644 llvm/test/CodeGen/DirectX/LegalizeAtomicExchangeFloat.ll create mode 100644 llvm/test/CodeGen/DirectX/ResourceAtomicExchangeFloat.ll diff --git a/clang/include/clang/Basic/Builtins.td b/clang/include/clang/Basic/Builtins.td index b91c56444e17e3..19a678ae67a3b4 100644 --- a/clang/include/clang/Basic/Builtins.td +++ b/clang/include/clang/Basic/Builtins.td @@ -5561,7 +5561,8 @@ def HLSLInterlockedAnd : LangBuiltin<"HLSL_LANG"> { def HLSLInterlockedExchange : LangBuiltin<"HLSL_LANG"> { let Spellings = ["__builtin_hlsl_interlocked_exchange"]; - let Attributes = [NoThrow]; + // Prevent inadvertent float -> double arg promotion. + let Attributes = [NoThrow, CustomTypeChecking]; let Prototype = "void (...)"; } diff --git a/clang/include/clang/Basic/DiagnosticSemaKinds.td b/clang/include/clang/Basic/DiagnosticSemaKinds.td index 36a18473f4d4cc..43b3a48edf04ae 100644 --- a/clang/include/clang/Basic/DiagnosticSemaKinds.td +++ b/clang/include/clang/Basic/DiagnosticSemaKinds.td @@ -13409,7 +13409,7 @@ def err_builtin_invalid_arg_type: Error< // An 'or' if non-empty second and third components are combined "%plural{0:|:%plural{0:|:or }2}3" // Third component: floating-point types - "%select{|floating-point|16 or 32 bit floating-point}3" + "%select{|floating-point|16 or 32 bit floating-point|32 bit floating-point}3" // A space after a non-empty third component "%plural{0:|: }3" "%plural{[0,3]:type|:types}1 (was %4)">; diff --git a/clang/lib/CodeGen/CGHLSLBuiltins.cpp b/clang/lib/CodeGen/CGHLSLBuiltins.cpp index 84808e7f88c6ac..0ce891a27550b6 100644 --- a/clang/lib/CodeGen/CGHLSLBuiltins.cpp +++ b/clang/lib/CodeGen/CGHLSLBuiltins.cpp @@ -317,8 +317,13 @@ static Value *handleInterlockedOp(CodeGenFunction &CGF, const CallExpr *E, LValue DestLV = CGF.EmitLValue(E->getArg(0)); Address DestAddr = DestLV.getAddress(); Value *Val = CGF.EmitScalarExpr(E->getArg(1)); - assert(E->getArg(1)->getType()->isIntegerType() && - "Intrinsic InterlockedOp value operand must be an integer"); + [[maybe_unused]] QualType ValTy = E->getArg(1)->getType(); + if (Op == llvm::AtomicRMWInst::Xchg) + assert((ValTy->isIntegerType() || ValTy->isFloatingType()) && + "InterlockedExchange value operand must be an integer or a float"); + else + assert(ValTy->isIntegerType() && + "Intrinsic InterlockedOp value operand must be an integer"); // Scopeless atomics will default to CrossDevice, which is illegal in Vulkan. // Set the memory scope: Workgroup for groupshared, otherwise Device. diff --git a/clang/lib/Sema/HLSLBuiltinTypeDeclBuilder.cpp b/clang/lib/Sema/HLSLBuiltinTypeDeclBuilder.cpp index 4922f67d0aec1b..91874dbca65b80 100644 --- a/clang/lib/Sema/HLSLBuiltinTypeDeclBuilder.cpp +++ b/clang/lib/Sema/HLSLBuiltinTypeDeclBuilder.cpp @@ -1765,6 +1765,12 @@ BuiltinTypeDeclBuilder::addByteAddressBufferInterlockedMethods() { addByteAddressBufferInterlockedMethod( "InterlockedExchange", AST.UnsignedIntTy, "__builtin_hlsl_interlocked_exchange", /*RequiresOriginalValue=*/true); + // The float exchange reuses the 32-bit integer DXIL operation, so it needs + // no capability bits and works from SM 6.0. ByteAddressBuffer carries no + // element type, so the method name states the type. + addByteAddressBufferInterlockedMethod("InterlockedExchangeFloat", AST.FloatTy, + "__builtin_hlsl_interlocked_exchange", + /*RequiresOriginalValue=*/true); addByteAddressBufferInterlockedMethod("InterlockedMax", AST.IntTy, "__builtin_hlsl_interlocked_max"); addByteAddressBufferInterlockedMethod("InterlockedMax", AST.UnsignedIntTy, diff --git a/clang/lib/Sema/HLSLExternalSemaSource.cpp b/clang/lib/Sema/HLSLExternalSemaSource.cpp index bf3becac043b3a..a25d38e865335c 100644 --- a/clang/lib/Sema/HLSLExternalSemaSource.cpp +++ b/clang/lib/Sema/HLSLExternalSemaSource.cpp @@ -888,13 +888,18 @@ static void buildAtomicOverload(Sema &S, NamespaceDecl *NS, StringRef FuncName, // Synthesize the InterlockedFunc overload set: {int, uint, int64_t, uint64_t} // x {groupshared, device} x {2-arg, 3-arg}. Operations that always report the // previous value, such as InterlockedExchange, only get the 3-arg form. +// InterlockedExchange also accepts float, which lowers to a bitwise exchange +// of the 32-bit pattern. static void defineHLSLInterlockedFunc(Sema &S, NamespaceDecl *NS, StringRef FuncName, StringRef BuiltinName, - bool RequiresOriginalValue = false) { + bool RequiresOriginalValue = false, + bool SupportsFloat = false) { ASTContext &AST = S.getASTContext(); // HLSL: int64_t == long, uint64_t == unsigned long (see hlsl_basic_types.h). - QualType Elems[] = {AST.IntTy, AST.UnsignedIntTy, AST.LongTy, - AST.UnsignedLongTy}; + SmallVector<QualType, 5> Elems = {AST.IntTy, AST.UnsignedIntTy, AST.LongTy, + AST.UnsignedLongTy}; + if (SupportsFloat) + Elems.push_back(AST.FloatTy); LangAS AddrSpaces[] = {LangAS::hlsl_groupshared, LangAS::hlsl_device}; for (QualType ElemTy : Elems) @@ -914,7 +919,8 @@ void HLSLExternalSemaSource::defineHLSLAtomicIntrinsics() { "__builtin_hlsl_interlocked_and"); defineHLSLInterlockedFunc(*SemaPtr, HLSLNamespace, "InterlockedExchange", "__builtin_hlsl_interlocked_exchange", - /*RequiresOriginalValue=*/true); + /*RequiresOriginalValue=*/true, + /*SupportsFloat=*/true); defineHLSLInterlockedFunc(*SemaPtr, HLSLNamespace, "InterlockedMax", "__builtin_hlsl_interlocked_max"); defineHLSLInterlockedFunc(*SemaPtr, HLSLNamespace, "InterlockedMin", diff --git a/clang/lib/Sema/SemaHLSL.cpp b/clang/lib/Sema/SemaHLSL.cpp index 84bf12fc680734..dfaaacf41ebee0 100644 --- a/clang/lib/Sema/SemaHLSL.cpp +++ b/clang/lib/Sema/SemaHLSL.cpp @@ -4632,11 +4632,17 @@ bool SemaHLSL::CheckBuiltinFunctionCall(unsigned BuiltinID, CallExpr *TheCall) { } QualType DestTy = TheCall->getArg(0)->getType().getUnqualifiedType(); - if (!DestTy->isIntegerType()) { + // InterlockedExchange also operates on float. DXIL lowers that as a + // bitwise exchange of the value's bit pattern, and DXC accepts 32-bit + // float only, so half and double are rejected. + const bool AllowsFloat = + BuiltinID == Builtin::BI__builtin_hlsl_interlocked_exchange; + if (!DestTy->isIntegerType() && + !(AllowsFloat && DestTy->isSpecificBuiltinType(BuiltinType::Float))) { SemaRef.Diag(TheCall->getArg(0)->getBeginLoc(), diag::err_builtin_invalid_arg_type) - << /*ordinal=*/1 << /*scalar*/ 1 << /*integer*/ 1 << /*no float*/ 0 - << DestTy; + << /*ordinal=*/1 << /*scalar*/ 1 << /*integer*/ 1 + << /*32 bit floating-point*/ (AllowsFloat ? 3 : 0) << DestTy; return true; } diff --git a/clang/test/CodeGenHLSL/builtins/InterlockedExchange.hlsl b/clang/test/CodeGenHLSL/builtins/InterlockedExchange.hlsl index 36240bf9f19ba1..5099a83ce42752 100644 --- a/clang/test/CodeGenHLSL/builtins/InterlockedExchange.hlsl +++ b/clang/test/CodeGenHLSL/builtins/InterlockedExchange.hlsl @@ -14,6 +14,7 @@ groupshared int gs_i32; groupshared uint gs_u32; groupshared int64_t gs_i64; groupshared uint64_t gs_u64; +groupshared float gs_f32; // CHECK-LABEL: define {{.*}}void @{{.*}}test_int_3arg // CHECK: %[[R:.*]] = atomicrmw xchg ptr addrspace(3) {{.*}}@gs_i32{{.*}}, i32 %{{.*}} syncscope("workgroup") monotonic @@ -42,3 +43,13 @@ export void test_int64_3arg(int64_t v, out int64_t orig) { export void test_uint64_3arg(uint64_t v, out uint64_t orig) { InterlockedExchange(gs_u64, v, orig); } + +// The float overload keeps the float type in the IR. DXIL converts it to an +// i32 exchange later, and SPIR-V selects OpAtomicExchange directly. +// CHECK-LABEL: define {{.*}}void @{{.*}}test_float_3arg +// DXCHECK: %[[R:.*]] = atomicrmw xchg ptr addrspace(3) {{.*}}@gs_f32{{.*}}, float %{{.*}} syncscope("workgroup") monotonic +// SPVCHECK: %[[R:.*]] = atomicrmw xchg ptr addrspace(3) {{.*}}@gs_f32{{.*}}, float %{{.*}} syncscope("workgroup") monotonic +// CHECK: store float %[[R]], ptr {{.*}} +export void test_float_3arg(float v, out float orig) { + InterlockedExchange(gs_f32, v, orig); +} diff --git a/clang/test/CodeGenHLSL/builtins/RWByteAddressBuffer-InterlockedExchange.hlsl b/clang/test/CodeGenHLSL/builtins/RWByteAddressBuffer-InterlockedExchange.hlsl index ce03bd0504053e..5185983e6b6359 100644 --- a/clang/test/CodeGenHLSL/builtins/RWByteAddressBuffer-InterlockedExchange.hlsl +++ b/clang/test/CodeGenHLSL/builtins/RWByteAddressBuffer-InterlockedExchange.hlsl @@ -38,3 +38,19 @@ export void test_bab_uint_3arg(uint off, uint v, out uint orig) { export void test_bab_uint64_3arg(uint off, uint64_t v, out uint64_t orig) { BAB.InterlockedExchange64(off, v, orig); } + +// ByteAddressBuffer holds no element type, so the float method carries the +// type in its name. The value keeps the float type here; DXIL converts it to +// an i32 exchange later. +// CHECK-LABEL: define {{.*}}void @{{.*}}test_bab_float_3arg +// DXCHECK: %[[HANDLE:.*]] = load target("dx.RawBuffer", i8, 1, 0), ptr {{.*}} +// DXCHECK: %[[PTR:.*]] = call ptr @llvm.dx.resource.getpointer.p0.tdx.RawBuffer_i8_1_0t.i32(target("dx.RawBuffer", i8, 1, 0) %[[HANDLE]], i32 %{{.*}}) +// DXCHECK: %[[R:.*]] = atomicrmw xchg ptr %[[PTR]], float %{{.*}} syncscope("device") monotonic +// DXCHECK: store float %[[R]], ptr {{.*}} +// SPVCHECK: %[[HANDLE:.*]] = load target("spirv.VulkanBuffer", [0 x i8], 12, 1), ptr {{.*}} +// SPVCHECK: %[[PTR:.*]] = call ptr addrspace(11) @llvm.spv.resource.getpointer.p11.tspirv.VulkanBuffer_a0i8_12_1t.i32(target("spirv.VulkanBuffer", [0 x i8], 12, 1) %[[HANDLE]], i32 %{{.*}}) +// SPVCHECK: %[[R:.*]] = atomicrmw xchg ptr addrspace(11) %[[PTR]], float %{{.*}} syncscope("device") monotonic +// SPVCHECK: store float %[[R]], ptr {{.*}} +export void test_bab_float_3arg(uint off, float v, out float orig) { + BAB.InterlockedExchangeFloat(off, v, orig); +} diff --git a/clang/test/SemaHLSL/BuiltIns/ByteAddressBuffer-InterlockedExchangeFloat-sm60.hlsl b/clang/test/SemaHLSL/BuiltIns/ByteAddressBuffer-InterlockedExchangeFloat-sm60.hlsl new file mode 100644 index 00000000000000..c4b1f34455f14f --- /dev/null +++ b/clang/test/SemaHLSL/BuiltIns/ByteAddressBuffer-InterlockedExchangeFloat-sm60.hlsl @@ -0,0 +1,40 @@ +// RUN: %clang_cc1 -std=hlsl202x -finclude-default-header \ +// RUN: -triple dxil-pc-shadermodel6.0-library %s -fsyntax-only -verify \ +// RUN: -verify-ignore-unexpected=warning + +// The float exchange reuses the 32-bit integer DXIL operation, so it needs no +// capability bits and works from SM 6.0. The 64-bit exchange needs SM 6.6. +// This file checks both halves, so it proves the two are gated differently. + +RWByteAddressBuffer BAB : register(u0); +RasterizerOrderedByteAddressBuffer ROVB : register(u1); +groupshared float gs_f32; +groupshared int64_t gs_i64; + +void sm60_bab_float_ok(uint off, float v, out float orig) { + BAB.InterlockedExchangeFloat(off, v, orig); +} + +void sm60_rovb_float_ok(uint off, float v, out float orig) { + ROVB.InterlockedExchangeFloat(off, v, orig); +} + +void sm60_free_function_ok(float v) { + float orig; + InterlockedExchange(gs_f32, v, orig); +} + +void sm60_direct_builtin_ok(float v) { + float orig; + __builtin_hlsl_interlocked_exchange(gs_f32, v, orig); +} + +void sm60_no_bab_exchange64(uint off, uint64_t v, out uint64_t orig) { + BAB.InterlockedExchange64(off, v, orig); + // expected-error@-1 {{no member named 'InterlockedExchange64' in 'hlsl::RWByteAddressBuffer'}} +} + +void sm60_no_direct_builtin_i64(int64_t v, out int64_t orig) { + __builtin_hlsl_interlocked_exchange(gs_i64, v, orig); + // expected-error@-1 {{'__builtin_hlsl_interlocked_exchange' requires shader model 6.6 or newer}} +} diff --git a/clang/test/SemaHLSL/BuiltIns/InterlockedExchange-errors.hlsl b/clang/test/SemaHLSL/BuiltIns/InterlockedExchange-errors.hlsl index bf6a1757b42f2c..3d70256bcddef9 100644 --- a/clang/test/SemaHLSL/BuiltIns/InterlockedExchange-errors.hlsl +++ b/clang/test/SemaHLSL/BuiltIns/InterlockedExchange-errors.hlsl @@ -3,53 +3,54 @@ // RUN: -disable-llvm-passes -verify // InterlockedExchange is provided as a set of address-space-qualified -// overloads (groupshared/device, {int,uint,int64_t,uint64_t}). It always -// reports the previous value, so there is no 2-argument form. +// overloads (groupshared/device, {int,uint,int64_t,uint64_t,float}). It always +// reports the previous value, so there is no 2-argument form. Only 32-bit +// float is accepted, so double has no overload. groupshared int gs_i32; -groupshared float gs_f32; +groupshared double gs_f64; struct S { int x; }; groupshared S gs_s; void too_few() { InterlockedExchange(gs_i32); // expected-error{{no matching function for call to 'InterlockedExchange'}} - // expected-note@*:* 8 {{candidate function}} + // expected-note@*:* 10 {{candidate function}} } void missing_original_value(int v) { InterlockedExchange(gs_i32, v); // expected-error{{no matching function for call to 'InterlockedExchange'}} - // expected-note@*:* 8 {{candidate function}} + // expected-note@*:* 10 {{candidate function}} } void too_many(int v, int extra) { int orig; InterlockedExchange(gs_i32, v, orig, extra); // expected-error{{no matching function for call to 'InterlockedExchange'}} - // expected-note@*:* 8 {{candidate function}} + // expected-note@*:* 10 {{candidate function}} } void local_dest(int v) { int dest; int orig; InterlockedExchange(dest, v, orig); // expected-error{{no matching function for call to 'InterlockedExchange'}} - // expected-note@*:* 8 {{candidate function}} + // expected-note@*:* 10 {{candidate function}} } -void float_dest(float v) { - float orig; - InterlockedExchange(gs_f32, v, orig); // expected-error{{no matching function for call to 'InterlockedExchange'}} - // expected-note@*:* 8 {{candidate function}} +void double_dest(double v) { + double orig; + InterlockedExchange(gs_f64, v, orig); // expected-error{{no matching function for call to 'InterlockedExchange'}} + // expected-note@*:* 10 {{candidate function}} } void struct_dest(int v) { int orig; InterlockedExchange(gs_s, v, orig); // expected-error{{no matching function for call to 'InterlockedExchange'}} - // expected-note@*:* 8 {{candidate function}} + // expected-note@*:* 10 {{candidate function}} } void mismatched_orig_type(int v) { uint orig; InterlockedExchange(gs_i32, v, orig); // expected-error{{no matching function for call to 'InterlockedExchange'}} - // expected-note@*:* 8 {{candidate function}} + // expected-note@*:* 10 {{candidate function}} } void direct_too_few() { @@ -72,7 +73,13 @@ void direct_non_integer_dest() { S local_s; S orig; __builtin_hlsl_interlocked_exchange(local_s, 1, orig); - // expected-error@-1 {{1st argument must be a scalar integer type (was 'S')}} + // expected-error@-1 {{1st argument must be a scalar integer or 32 bit floating-point type (was 'S')}} +} + +void direct_double_dest(double v) { + double orig; + __builtin_hlsl_interlocked_exchange(gs_f64, v, orig); + // expected-error@-1 {{1st argument must be a scalar integer or 32 bit floating-point type (was 'double')}} } void direct_nonlvalue_dest(int v) { diff --git a/llvm/lib/Target/DirectX/DXILLegalizePass.cpp b/llvm/lib/Target/DirectX/DXILLegalizePass.cpp index 9d4d7661ac9c92..c8037fbd84d0cc 100644 --- a/llvm/lib/Target/DirectX/DXILLegalizePass.cpp +++ b/llvm/lib/Target/DirectX/DXILLegalizePass.cpp @@ -356,6 +356,34 @@ static bool updateFnegToFsub(Instruction &I, return true; } +// DXIL has no floating-point atomic operation. A float exchange only moves the +// bit pattern, so exchange an integer of the same width instead. Opaque +// pointers keep the pointer operand type-agnostic, so only the value and the +// result need a cast. This matches what DXC emits for groupshared memory. +static bool +legalizeFloatAtomicExchange(Instruction &I, + SmallVectorImpl<Instruction *> &ToRemove, + DenseMap<Value *, Value *> &) { + auto *AI = dyn_cast<AtomicRMWInst>(&I); + if (!AI || AI->getOperation() != AtomicRMWInst::Xchg) + return false; + + Type *ValTy = AI->getValOperand()->getType(); + if (!ValTy->isFloatingPointTy()) + return false; + + IRBuilder<> Builder(AI); + Type *IntTy = Builder.getIntNTy(ValTy->getPrimitiveSizeInBits()); + Value *Val = Builder.CreateBitCast(AI->getValOperand(), IntTy); + AtomicRMWInst *NewAI = Builder.CreateAtomicRMW( + AtomicRMWInst::Xchg, AI->getPointerOperand(), Val, AI->getAlign(), + AI->getOrdering(), AI->getSyncScopeID()); + NewAI->copyMetadata(*AI); + AI->replaceAllUsesWith(Builder.CreateBitCast(NewAI, ValTy)); + ToRemove.push_back(AI); + return true; +} + static bool resolveUnreachableSwitchDefault(Instruction &I, SmallVectorImpl<Instruction *> &ToRemove, @@ -496,6 +524,7 @@ class DXILLegalizationPipeline { LegalizationPipeline[Stage1].push_back(fixI8UseChain); LegalizationPipeline[Stage1].push_back(legalizeFreeze); LegalizationPipeline[Stage1].push_back(updateFnegToFsub); + LegalizationPipeline[Stage1].push_back(legalizeFloatAtomicExchange); LegalizationPipeline[Stage1].push_back( downcastI64toI32InsertExtractElements); LegalizationPipeline[Stage2].push_back(legalizeScalarLoadStoreOnArrays); diff --git a/llvm/lib/Target/DirectX/DXILResourceAccess.cpp b/llvm/lib/Target/DirectX/DXILResourceAccess.cpp index a5cb99e17b2431..0dd5f69eff7cb9 100644 --- a/llvm/lib/Target/DirectX/DXILResourceAccess.cpp +++ b/llvm/lib/Target/DirectX/DXILResourceAccess.cpp @@ -383,17 +383,31 @@ static void emitAtomicBinOp(IRBuilder<> &Builder, AtomicRMWInst *AI, return; } + // DXIL has no floating-point atomic op. A float exchange only moves the bit + // pattern, so cast the value to an integer of the same width, exchange, and + // cast the result back. This matches what DXC emits. + Value *Val = AI->getValOperand(); + Type *ValTy = Val->getType(); + Type *OpTy = ValTy; + if (ValTy->isFloatingPointTy()) { + OpTy = Builder.getIntNTy(ValTy->getPrimitiveSizeInBits()); + Val = Builder.CreateBitCast(Val, OpTy); + } + SmallVector<Value *, 6> Args{ Handle, Builder.getInt32(static_cast<uint32_t>(*BinOpCode))}; append_range(Args, Coords); Args.append(3 - Coords.size(), PoisonValue::get(Builder.getInt32Ty())); - Args.push_back(AI->getValOperand()); + Args.push_back(Val); // Emit the target-independent intrinsic; DXILOpLowering lowers it to the // DXIL `AtomicBinOp` op and handles the target-ext-typed handle cast via // its `createTmpHandleCast` bookkeeping. - Value *Result = Builder.CreateIntrinsic( - AI->getType(), Intrinsic::dx_resource_atomic_binop, Args); + Value *Result = + Builder.CreateIntrinsic(OpTy, Intrinsic::dx_resource_atomic_binop, Args); + + if (OpTy != ValTy) + Result = Builder.CreateBitCast(Result, ValTy); AI->replaceAllUsesWith(Result); } diff --git a/llvm/test/CodeGen/DirectX/LegalizeAtomicExchangeFloat.ll b/llvm/test/CodeGen/DirectX/LegalizeAtomicExchangeFloat.ll new file mode 100644 index 00000000000000..446c76a2b801db --- /dev/null +++ b/llvm/test/CodeGen/DirectX/LegalizeAtomicExchangeFloat.ll @@ -0,0 +1,39 @@ +; RUN: opt -S -dxil-legalize -mtriple=dxil-pc-shadermodel6.0-compute %s | FileCheck %s + +; DXIL has no floating-point atomic op. A float exchange on groupshared memory +; only moves the bit pattern, so it becomes an i32 exchange with a bitcast on +; the value and on the result. Opaque pointers leave the pointer operand +; unchanged. + +target triple = "dxil-pc-shadermodel6.0-compute" + +@gs = external addrspace(3) global float + +; CHECK-LABEL: define float @gs_xchg_float +define float @gs_xchg_float(float %val) { + ; CHECK: [[CAST:%.*]] = bitcast float %val to i32 + ; CHECK: [[OLD:%.*]] = atomicrmw xchg ptr addrspace(3) @gs, i32 [[CAST]] syncscope("workgroup") monotonic + ; CHECK: [[RES:%.*]] = bitcast i32 [[OLD]] to float + %old = atomicrmw xchg ptr addrspace(3) @gs, float %val syncscope("workgroup") monotonic + ; CHECK: ret float [[RES]] + ret float %old +} + +; An integer exchange must pass through with no bitcast. +; CHECK-LABEL: define i32 @gs_xchg_i32 +define i32 @gs_xchg_i32(ptr addrspace(3) %p, i32 %val) { + ; CHECK-NOT: bitcast + ; CHECK: atomicrmw xchg ptr addrspace(3) %p, i32 %val syncscope("workgroup") monotonic + %old = atomicrmw xchg ptr addrspace(3) %p, i32 %val syncscope("workgroup") monotonic + ret i32 %old +} + +; Only exchange is rewritten. DXIL does not support float atomic add anywhere, +; but fadd must not be silently turned into an integer add. +; CHECK-LABEL: define float @gs_fadd_float +define float @gs_fadd_float(ptr addrspace(3) %p, float %val) { + ; CHECK-NOT: bitcast + ; CHECK: atomicrmw fadd ptr addrspace(3) %p, float %val syncscope("workgroup") monotonic + %old = atomicrmw fadd ptr addrspace(3) %p, float %val syncscope("workgroup") monotonic + ret float %old +} diff --git a/llvm/test/CodeGen/DirectX/ResourceAtomicExchangeFloat.ll b/llvm/test/CodeGen/DirectX/ResourceAtomicExchangeFloat.ll new file mode 100644 index 00000000000000..a35fce262f9a60 --- /dev/null +++ b/llvm/test/CodeGen/DirectX/ResourceAtomicExchangeFloat.ll @@ -0,0 +1,38 @@ +; RUN: opt -S -dxil-resource-access -dxil-op-lower -mtriple=dxil-pc-shadermodel6.0-compute %s | FileCheck %s + +; DXIL has no floating-point atomic op. A float exchange only moves the bit +; pattern, so it lowers to an i32 AtomicBinOp with a bitcast on the value and +; on the returned original value. This needs no capability bits, so it works +; from SM 6.0. + +target triple = "dxil-pc-shadermodel6.0-compute" + +; CHECK-LABEL: define float @bab_xchg_float +define float @bab_xchg_float(i32 %offset, float %val) { + %buffer = call target("dx.RawBuffer", i8, 1, 0, 0) + @llvm.dx.resource.handlefrombinding(i32 0, i32 0, i32 1, i32 0, ptr null) + %ptr = call ptr @llvm.dx.resource.getpointer( + target("dx.RawBuffer", i8, 1, 0, 0) %buffer, i32 %offset) + ; CHECK: [[CAST:%.*]] = bitcast float %val to i32 + ; CHECK: [[OLD:%.*]] = call i32 @dx.op.atomicBinOp.i32(i32 78, %dx.types.Handle %{{.*}}, i32 8, i32 %offset, i32 poison, i32 poison, i32 [[CAST]]) + ; CHECK: [[RES:%.*]] = bitcast i32 [[OLD]] to float + %old = atomicrmw xchg ptr %ptr, float %val monotonic + ; CHECK: ret float [[RES]] + ret float %old +} + +; A StructuredBuffer of float keeps the struct index in coord0 and the byte +; offset in coord1. +; CHECK-LABEL: define float @sbuf_xchg_float +define float @sbuf_xchg_float(i32 %index, float %val) { + %buffer = call target("dx.RawBuffer", float, 1, 0, 0) + @llvm.dx.resource.handlefrombinding(i32 0, i32 0, i32 1, i32 0, ptr null) + %ptr = call ptr @llvm.dx.resource.getpointer( + target("dx.RawBuffer", float, 1, 0, 0) %buffer, i32 %index) + ; CHECK: [[CAST:%.*]] = bitcast float %val to i32 + ; CHECK: [[OLD:%.*]] = call i32 @dx.op.atomicBinOp.i32(i32 78, %dx.types.Handle %{{.*}}, i32 8, i32 %index, i32 0, i32 poison, i32 [[CAST]]) + ; CHECK: [[RES:%.*]] = bitcast i32 [[OLD]] to float + %old = atomicrmw xchg ptr %ptr, float %val monotonic + ; CHECK: ret float [[RES]] + ret float %old +} >From fe2c83e6adf0536ae6fe018d5538987db7f312a6 Mon Sep 17 00:00:00 2001 From: Joshua Batista <[email protected]> Date: Thu, 17 Sep 2026 11:19:31 -0700 Subject: [PATCH 2/6] Add float InterlockedExchange coverage to the texture test --- .../test/CodeGenHLSL/builtins/RWTexture-Interlocked.hlsl | 8 +++++++- 1 file changed, 7 insertions(+), 1 deletion(-) diff --git a/clang/test/CodeGenHLSL/builtins/RWTexture-Interlocked.hlsl b/clang/test/CodeGenHLSL/builtins/RWTexture-Interlocked.hlsl index 5013b11c62636e..4a5d221a8ad4d8 100644 --- a/clang/test/CodeGenHLSL/builtins/RWTexture-Interlocked.hlsl +++ b/clang/test/CodeGenHLSL/builtins/RWTexture-Interlocked.hlsl @@ -19,6 +19,7 @@ RWTexture2D<int> Out : register(u0); RWTexture2DArray<uint> UOut : register(u1); +RWTexture2D<float> FOut : register(u2); // CHECK-LABEL: define void @main // DXCHECK: %[[PTR1:.*]] = call {{.*}} @llvm.dx.resource.getpointer.{{.*}}(target("dx.Texture", i32, 1, 0, 1, 2) %{{.*}}, <2 x i32> %{{.*}}) @@ -37,6 +38,8 @@ RWTexture2DArray<uint> UOut : register(u1); // DXCHECK: atomicrmw umax ptr %[[PTR7]], i32 1 syncscope("device") monotonic // DXCHECK: %[[PTR8:.*]] = call {{.*}} @llvm.dx.resource.getpointer.{{.*}}(target("dx.Texture", i32, 1, 0, 1, 2) %{{.*}}, <2 x i32> %{{.*}}) // DXCHECK: atomicrmw xchg ptr %[[PTR8]], i32 1 syncscope("device") monotonic +// DXCHECK: %[[PTR9:.*]] = call {{.*}} @llvm.dx.resource.getpointer.{{.*}}(target("dx.Texture", float, 1, 0, 0, 2) %{{.*}}, <2 x i32> %{{.*}}) +// DXCHECK: atomicrmw xchg ptr %[[PTR9]], float 1.000000e+00 syncscope("device") monotonic // SPVCHECK: %[[PTR1:.*]] = call {{.*}} @llvm.spv.resource.getpointer.{{.*}}(target("spirv.SignedImage", i32, {{.*}}) %{{.*}}, <2 x i32> %{{.*}}) // SPVCHECK: atomicrmw add ptr addrspace(11) %[[PTR1]], i32 1 syncscope("device") monotonic // SPVCHECK: %[[PTR2:.*]] = call {{.*}} @llvm.spv.resource.getpointer.{{.*}}(target("spirv.SignedImage", i32, {{.*}}) %{{.*}}, <2 x i32> %{{.*}}) @@ -53,6 +56,8 @@ RWTexture2DArray<uint> UOut : register(u1); // SPVCHECK: atomicrmw umax ptr addrspace(11) %[[PTR7]], i32 1 syncscope("device") monotonic // SPVCHECK: %[[PTR8:.*]] = call {{.*}} @llvm.spv.resource.getpointer.{{.*}}(target("spirv.SignedImage", i32, {{.*}}) %{{.*}}, <2 x i32> %{{.*}}) // SPVCHECK: atomicrmw xchg ptr addrspace(11) %[[PTR8]], i32 1 syncscope("device") monotonic +// SPVCHECK: %[[PTR9:.*]] = call {{.*}} @llvm.spv.resource.getpointer.{{.*}}(target("spirv.Image", float, {{.*}}) %{{.*}}, <2 x i32> %{{.*}}) +// SPVCHECK: atomicrmw xchg ptr addrspace(11) %[[PTR9]], float 1.000000e+00 syncscope("device") monotonic [shader("compute")] [numthreads(1,1,1)] void main(uint3 id : SV_DispatchThreadID) { @@ -63,7 +68,8 @@ void main(uint3 id : SV_DispatchThreadID) { InterlockedMin(UOut[id], 1u); InterlockedMax(Out[id.xy], 1); InterlockedMax(UOut[id], 1u); - // DXIL has no float texture atomic, so only the integer form is covered. int Orig; InterlockedExchange(Out[id.xy], 1, Orig); + float FOrig; + InterlockedExchange(FOut[id.xy], 1.0f, FOrig); } >From 5bfa2d291404f21433a4500192cc86d79fd693a0 Mon Sep 17 00:00:00 2001 From: Joshua Batista <[email protected]> Date: Thu, 17 Sep 2026 12:43:06 -0700 Subject: [PATCH 3/6] Lower atomicrmw on scalar float texture resources --- llvm/lib/Target/DirectX/DXILResourceAccess.cpp | 7 +++++-- .../ResourceAtomicBinOp-texture-nonscalar.ll | 8 ++++---- .../DirectX/ResourceAtomicExchangeFloat.ll | 18 ++++++++++++++++++ 3 files changed, 27 insertions(+), 6 deletions(-) diff --git a/llvm/lib/Target/DirectX/DXILResourceAccess.cpp b/llvm/lib/Target/DirectX/DXILResourceAccess.cpp index 0dd5f69eff7cb9..70f5407aeadcfd 100644 --- a/llvm/lib/Target/DirectX/DXILResourceAccess.cpp +++ b/llvm/lib/Target/DirectX/DXILResourceAccess.cpp @@ -424,10 +424,13 @@ static void createBufferAtomicBinOp(IntrinsicInst *II, AtomicRMWInst *AI, static void createTextureAtomicBinOp(IntrinsicInst *II, AtomicRMWInst *AI, dxil::ResourceTypeInfo &RTI) { + // A texture atomic operates on a whole texel, so a multi-component texel has + // no single addressable component. A scalar float texel is allowed, because + // emitAtomicBinOp exchanges its bit pattern as an integer. Type *ContainedType = RTI.getHandleTy()->getTypeParameter(0); - if (!ContainedType->isIntegerTy()) { + if (!ContainedType->isIntegerTy() && !ContainedType->isFloatingPointTy()) { reportFatalUsageError("DXIL atomicrmw requires a texture resource with a " - "scalar integer element type"); + "scalar element type"); return; } diff --git a/llvm/test/CodeGen/DirectX/ResourceAtomicBinOp-texture-nonscalar.ll b/llvm/test/CodeGen/DirectX/ResourceAtomicBinOp-texture-nonscalar.ll index 2abc27da6ffbd5..6c1b2c2ee2fd38 100644 --- a/llvm/test/CodeGen/DirectX/ResourceAtomicBinOp-texture-nonscalar.ll +++ b/llvm/test/CodeGen/DirectX/ResourceAtomicBinOp-texture-nonscalar.ll @@ -11,7 +11,7 @@ target triple = "dxil-pc-shadermodel6.6-compute" -; CHECK: DXIL atomicrmw requires a texture resource with a scalar integer element type +; CHECK: DXIL atomicrmw requires a texture resource with a scalar element type define i32 @atomic_texture1d_int2(i32 %coord, i32 %value) { %texture = call target("dx.Texture", <2 x i32>, 1, 0, 0, 1) @llvm.dx.resource.handlefrombinding(i32 0, i32 0, i32 1, i32 0, ptr null) @@ -25,7 +25,7 @@ define i32 @atomic_texture1d_int2(i32 %coord, i32 %value) { target triple = "dxil-pc-shadermodel6.6-compute" -; CHECK: DXIL atomicrmw requires a texture resource with a scalar integer element type +; CHECK: DXIL atomicrmw requires a texture resource with a scalar element type define i32 @atomic_texture2d_int4(<2 x i32> %coords, i32 %value) { %texture = call target("dx.Texture", <4 x i32>, 1, 0, 0, 2) @llvm.dx.resource.handlefrombinding(i32 0, i32 0, i32 1, i32 0, ptr null) @@ -39,7 +39,7 @@ define i32 @atomic_texture2d_int4(<2 x i32> %coords, i32 %value) { target triple = "dxil-pc-shadermodel6.6-compute" -; CHECK: DXIL atomicrmw requires a texture resource with a scalar integer element type +; CHECK: DXIL atomicrmw requires a texture resource with a scalar element type define i64 @atomic_texture2darray_i64x2(<3 x i32> %coords, i64 %value) { %texture = call target("dx.Texture", <2 x i64>, 1, 0, 0, 7) @llvm.dx.resource.handlefrombinding(i32 0, i32 0, i32 1, i32 0, ptr null) @@ -53,7 +53,7 @@ define i64 @atomic_texture2darray_i64x2(<3 x i32> %coords, i64 %value) { target triple = "dxil-pc-shadermodel6.6-compute" -; CHECK: DXIL atomicrmw requires a texture resource with a scalar integer element type +; CHECK: DXIL atomicrmw requires a texture resource with a scalar element type define i32 @atomic_texture3d_float4(<3 x i32> %coords, i32 %value) { %texture = call target("dx.Texture", <4 x float>, 1, 0, 0, 4) @llvm.dx.resource.handlefrombinding(i32 0, i32 0, i32 1, i32 0, ptr null) diff --git a/llvm/test/CodeGen/DirectX/ResourceAtomicExchangeFloat.ll b/llvm/test/CodeGen/DirectX/ResourceAtomicExchangeFloat.ll index a35fce262f9a60..ace0c929afb69b 100644 --- a/llvm/test/CodeGen/DirectX/ResourceAtomicExchangeFloat.ll +++ b/llvm/test/CodeGen/DirectX/ResourceAtomicExchangeFloat.ll @@ -36,3 +36,21 @@ define float @sbuf_xchg_float(i32 %index, float %val) { ; CHECK: ret float [[RES]] ret float %old } + +; A texture of scalar float lowers the same way, with one coordinate operand +; per texture dimension. +; CHECK-LABEL: define float @texture2d_xchg_float +define float @texture2d_xchg_float(<2 x i32> %coords, float %val) { + %texture = call target("dx.Texture", float, 1, 0, 0, 2) + @llvm.dx.resource.handlefrombinding(i32 0, i32 1, i32 1, i32 0, ptr null) + %ptr = call ptr @llvm.dx.resource.getpointer( + target("dx.Texture", float, 1, 0, 0, 2) %texture, <2 x i32> %coords) + ; CHECK: %[[X:.*]] = extractelement <2 x i32> %coords, i64 0 + ; CHECK: %[[Y:.*]] = extractelement <2 x i32> %coords, i64 1 + ; CHECK: [[CAST:%.*]] = bitcast float %val to i32 + ; CHECK: [[OLD:%.*]] = call i32 @dx.op.atomicBinOp.i32(i32 78, %dx.types.Handle %{{.*}}, i32 8, i32 %[[X]], i32 %[[Y]], i32 poison, i32 [[CAST]]) + ; CHECK: [[RES:%.*]] = bitcast i32 [[OLD]] to float + %old = atomicrmw xchg ptr %ptr, float %val monotonic + ; CHECK: ret float [[RES]] + ret float %old +} >From 7e3da5cebbc7c6ce63536cee8614de281dc9e7a1 Mon Sep 17 00:00:00 2001 From: Joshua Batista <[email protected]> Date: Thu, 17 Sep 2026 14:48:40 -0700 Subject: [PATCH 4/6] Remove comments on the float InterlockedExchange declarations --- clang/include/clang/Basic/Builtins.td | 1 - clang/lib/Sema/HLSLBuiltinTypeDeclBuilder.cpp | 3 --- 2 files changed, 4 deletions(-) diff --git a/clang/include/clang/Basic/Builtins.td b/clang/include/clang/Basic/Builtins.td index 19a678ae67a3b4..fcdbf4b885eb4b 100644 --- a/clang/include/clang/Basic/Builtins.td +++ b/clang/include/clang/Basic/Builtins.td @@ -5561,7 +5561,6 @@ def HLSLInterlockedAnd : LangBuiltin<"HLSL_LANG"> { def HLSLInterlockedExchange : LangBuiltin<"HLSL_LANG"> { let Spellings = ["__builtin_hlsl_interlocked_exchange"]; - // Prevent inadvertent float -> double arg promotion. let Attributes = [NoThrow, CustomTypeChecking]; let Prototype = "void (...)"; } diff --git a/clang/lib/Sema/HLSLBuiltinTypeDeclBuilder.cpp b/clang/lib/Sema/HLSLBuiltinTypeDeclBuilder.cpp index 91874dbca65b80..4fc974763a6fe5 100644 --- a/clang/lib/Sema/HLSLBuiltinTypeDeclBuilder.cpp +++ b/clang/lib/Sema/HLSLBuiltinTypeDeclBuilder.cpp @@ -1765,9 +1765,6 @@ BuiltinTypeDeclBuilder::addByteAddressBufferInterlockedMethods() { addByteAddressBufferInterlockedMethod( "InterlockedExchange", AST.UnsignedIntTy, "__builtin_hlsl_interlocked_exchange", /*RequiresOriginalValue=*/true); - // The float exchange reuses the 32-bit integer DXIL operation, so it needs - // no capability bits and works from SM 6.0. ByteAddressBuffer carries no - // element type, so the method name states the type. addByteAddressBufferInterlockedMethod("InterlockedExchangeFloat", AST.FloatTy, "__builtin_hlsl_interlocked_exchange", /*RequiresOriginalValue=*/true); >From ac4fcb79efd4f9ce07f87315ee92f5ce2fa781b8 Mon Sep 17 00:00:00 2001 From: Joshua Batista <[email protected]> Date: Wed, 23 Sep 2026 12:34:33 -0700 Subject: [PATCH 5/6] address Kaitlin, reject float atomic exchange of unsupported width --- llvm/lib/Target/DirectX/DXILLegalizePass.cpp | 10 ++++++- ...zeAtomicExchangeFloat-unsupported-width.ll | 30 +++++++++++++++++++ .../DirectX/LegalizeAtomicExchangeFloat.ll | 12 ++++++++ 3 files changed, 51 insertions(+), 1 deletion(-) create mode 100644 llvm/test/CodeGen/DirectX/LegalizeAtomicExchangeFloat-unsupported-width.ll diff --git a/llvm/lib/Target/DirectX/DXILLegalizePass.cpp b/llvm/lib/Target/DirectX/DXILLegalizePass.cpp index c8037fbd84d0cc..9cc864c5e47762 100644 --- a/llvm/lib/Target/DirectX/DXILLegalizePass.cpp +++ b/llvm/lib/Target/DirectX/DXILLegalizePass.cpp @@ -17,6 +17,7 @@ #include "llvm/IR/Instructions.h" #include "llvm/IR/Module.h" #include "llvm/Pass.h" +#include "llvm/Support/ErrorHandling.h" #include "llvm/Transforms/Utils/BasicBlockUtils.h" #include "llvm/Transforms/Utils/Local.h" #include <functional> @@ -372,8 +373,15 @@ legalizeFloatAtomicExchange(Instruction &I, if (!ValTy->isFloatingPointTy()) return false; + // DXIL has 32-bit and 64-bit atomics only. A float of any other width has no + // integer exchange to lower to. + unsigned Width = ValTy->getPrimitiveSizeInBits(); + if (Width != 32 && Width != 64) + reportFatalUsageError("DXIL atomic exchange requires a 32-bit or 64-bit " + "floating-point value"); + IRBuilder<> Builder(AI); - Type *IntTy = Builder.getIntNTy(ValTy->getPrimitiveSizeInBits()); + Type *IntTy = Builder.getIntNTy(Width); Value *Val = Builder.CreateBitCast(AI->getValOperand(), IntTy); AtomicRMWInst *NewAI = Builder.CreateAtomicRMW( AtomicRMWInst::Xchg, AI->getPointerOperand(), Val, AI->getAlign(), diff --git a/llvm/test/CodeGen/DirectX/LegalizeAtomicExchangeFloat-unsupported-width.ll b/llvm/test/CodeGen/DirectX/LegalizeAtomicExchangeFloat-unsupported-width.ll new file mode 100644 index 00000000000000..e75d74965da8d6 --- /dev/null +++ b/llvm/test/CodeGen/DirectX/LegalizeAtomicExchangeFloat-unsupported-width.ll @@ -0,0 +1,30 @@ +; RUN: split-file %s %t +; RUN: not opt -S -dxil-legalize -mtriple=dxil-pc-shadermodel6.0-compute %t/half.ll 2>&1 | FileCheck %t/half.ll +; RUN: not opt -S -dxil-legalize -mtriple=dxil-pc-shadermodel6.0-compute %t/fp128.ll 2>&1 | FileCheck %t/fp128.ll + +; DXIL has 32-bit and 64-bit atomics only, so a float exchange of any other +; width has no integer exchange to lower to. + +;--- half.ll + +target triple = "dxil-pc-shadermodel6.0-compute" + +@gs = external addrspace(3) global half + +; CHECK: DXIL atomic exchange requires a 32-bit or 64-bit floating-point value +define half @gs_xchg_half(half %val) { + %old = atomicrmw xchg ptr addrspace(3) @gs, half %val syncscope("workgroup") monotonic + ret half %old +} + +;--- fp128.ll + +target triple = "dxil-pc-shadermodel6.0-compute" + +@gs = external addrspace(3) global fp128 + +; CHECK: DXIL atomic exchange requires a 32-bit or 64-bit floating-point value +define fp128 @gs_xchg_fp128(fp128 %val) { + %old = atomicrmw xchg ptr addrspace(3) @gs, fp128 %val syncscope("workgroup") monotonic + ret fp128 %old +} \ No newline at end of file diff --git a/llvm/test/CodeGen/DirectX/LegalizeAtomicExchangeFloat.ll b/llvm/test/CodeGen/DirectX/LegalizeAtomicExchangeFloat.ll index 446c76a2b801db..8af5c533519bf2 100644 --- a/llvm/test/CodeGen/DirectX/LegalizeAtomicExchangeFloat.ll +++ b/llvm/test/CodeGen/DirectX/LegalizeAtomicExchangeFloat.ll @@ -8,6 +8,7 @@ target triple = "dxil-pc-shadermodel6.0-compute" @gs = external addrspace(3) global float +@gsd = external addrspace(3) global double ; CHECK-LABEL: define float @gs_xchg_float define float @gs_xchg_float(float %val) { @@ -19,6 +20,17 @@ define float @gs_xchg_float(float %val) { ret float %old } +; DXIL has a 64-bit atomic, so a double exchange becomes an i64 exchange. +; CHECK-LABEL: define double @gs_xchg_double +define double @gs_xchg_double(double %val) { + ; CHECK: [[CAST:%.*]] = bitcast double %val to i64 + ; CHECK: [[OLD:%.*]] = atomicrmw xchg ptr addrspace(3) @gsd, i64 [[CAST]] syncscope("workgroup") monotonic + ; CHECK: [[RES:%.*]] = bitcast i64 [[OLD]] to double + %old = atomicrmw xchg ptr addrspace(3) @gsd, double %val syncscope("workgroup") monotonic + ; CHECK: ret double [[RES]] + ret double %old +} + ; An integer exchange must pass through with no bitcast. ; CHECK-LABEL: define i32 @gs_xchg_i32 define i32 @gs_xchg_i32(ptr addrspace(3) %p, i32 %val) { >From 9b76fdf6eee789a7ee9e121d2052392d6721152d Mon Sep 17 00:00:00 2001 From: Joshua Batista <[email protected]> Date: Wed, 23 Sep 2026 12:53:24 -0700 Subject: [PATCH 6/6] add new line --- .../DirectX/LegalizeAtomicExchangeFloat-unsupported-width.ll | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/llvm/test/CodeGen/DirectX/LegalizeAtomicExchangeFloat-unsupported-width.ll b/llvm/test/CodeGen/DirectX/LegalizeAtomicExchangeFloat-unsupported-width.ll index e75d74965da8d6..ed344f8b27c000 100644 --- a/llvm/test/CodeGen/DirectX/LegalizeAtomicExchangeFloat-unsupported-width.ll +++ b/llvm/test/CodeGen/DirectX/LegalizeAtomicExchangeFloat-unsupported-width.ll @@ -27,4 +27,4 @@ target triple = "dxil-pc-shadermodel6.0-compute" define fp128 @gs_xchg_fp128(fp128 %val) { %old = atomicrmw xchg ptr addrspace(3) @gs, fp128 %val syncscope("workgroup") monotonic ret fp128 %old -} \ No newline at end of file +} _______________________________________________ llvm-branch-commits mailing list [email protected] https://lists.llvm.org/cgi-bin/mailman/listinfo/llvm-branch-commits
