https://github.com/bob80905 updated https://github.com/llvm/llvm-project/pull/222161
>From df2d2cda22f2f94e7edc660922044abc7f1a7e64 Mon Sep 17 00:00:00 2001 From: Joshua Batista <[email protected]> Date: Fri, 4 Sep 2026 10:43:20 -0700 Subject: [PATCH 1/2] First attempt implementing interlockedexchange --- clang/include/clang/Basic/Builtins.td | 6 + clang/lib/CodeGen/CGHLSLBuiltins.cpp | 3 + clang/lib/Sema/HLSLBuiltinTypeDeclBuilder.cpp | 15 ++- clang/lib/Sema/HLSLBuiltinTypeDeclBuilder.h | 3 +- clang/lib/Sema/HLSLExternalSemaSource.cpp | 15 ++- clang/lib/Sema/SemaHLSL.cpp | 24 ++-- .../builtins/InterlockedExchange.hlsl | 48 ++++++++ .../builtins/RWBuffer-Interlocked.hlsl | 6 + ...ByteAddressBuffer-InterlockedExchange.hlsl | 40 +++++++ ...ByteAddressBuffer-InterlockedExchange.hlsl | 29 +++++ ...ressBuffer-InterlockedExchange-errors.hlsl | 46 ++++++++ ...uffer-InterlockedExchange-sm65-errors.hlsl | 26 +++++ .../BuiltIns/InterlockedExchange-errors.hlsl | 107 ++++++++++++++++++ 13 files changed, 352 insertions(+), 16 deletions(-) create mode 100644 clang/test/CodeGenHLSL/builtins/InterlockedExchange.hlsl create mode 100644 clang/test/CodeGenHLSL/builtins/RWByteAddressBuffer-InterlockedExchange.hlsl create mode 100644 clang/test/CodeGenHLSL/builtins/RasterizerOrderedByteAddressBuffer-InterlockedExchange.hlsl create mode 100644 clang/test/SemaHLSL/BuiltIns/ByteAddressBuffer-InterlockedExchange-errors.hlsl create mode 100644 clang/test/SemaHLSL/BuiltIns/ByteAddressBuffer-InterlockedExchange-sm65-errors.hlsl create mode 100644 clang/test/SemaHLSL/BuiltIns/InterlockedExchange-errors.hlsl diff --git a/clang/include/clang/Basic/Builtins.td b/clang/include/clang/Basic/Builtins.td index 80d09c089ae952..b91c56444e17e3 100644 --- a/clang/include/clang/Basic/Builtins.td +++ b/clang/include/clang/Basic/Builtins.td @@ -5559,6 +5559,12 @@ def HLSLInterlockedAnd : LangBuiltin<"HLSL_LANG"> { let Prototype = "void (...)"; } +def HLSLInterlockedExchange : LangBuiltin<"HLSL_LANG"> { + let Spellings = ["__builtin_hlsl_interlocked_exchange"]; + let Attributes = [NoThrow]; + let Prototype = "void (...)"; +} + def HLSLInterlockedMax : LangBuiltin<"HLSL_LANG"> { let Spellings = ["__builtin_hlsl_interlocked_max"]; let Attributes = [NoThrow]; diff --git a/clang/lib/CodeGen/CGHLSLBuiltins.cpp b/clang/lib/CodeGen/CGHLSLBuiltins.cpp index bb0fe135ff8b08..84808e7f88c6ac 100644 --- a/clang/lib/CodeGen/CGHLSLBuiltins.cpp +++ b/clang/lib/CodeGen/CGHLSLBuiltins.cpp @@ -1461,6 +1461,9 @@ Value *CodeGenFunction::EmitHLSLBuiltinExpr(unsigned BuiltinID, case Builtin::BI__builtin_hlsl_interlocked_and: { return handleInterlockedOp(*this, E, llvm::AtomicRMWInst::And); } + case Builtin::BI__builtin_hlsl_interlocked_exchange: { + return handleInterlockedOp(*this, E, llvm::AtomicRMWInst::Xchg); + } case Builtin::BI__builtin_hlsl_interlocked_max: { llvm::AtomicRMWInst::BinOp Op = E->getArg(0)->getType()->hasSignedIntegerRepresentation() diff --git a/clang/lib/Sema/HLSLBuiltinTypeDeclBuilder.cpp b/clang/lib/Sema/HLSLBuiltinTypeDeclBuilder.cpp index 812f3d82c47a42..4922f67d0aec1b 100644 --- a/clang/lib/Sema/HLSLBuiltinTypeDeclBuilder.cpp +++ b/clang/lib/Sema/HLSLBuiltinTypeDeclBuilder.cpp @@ -1756,11 +1756,15 @@ BuiltinTypeDeclBuilder::addByteAddressBufferInterlockedMethods() { ASTContext &AST = SemaRef.getASTContext(); // This is a helper that declares two overloads with and without an out - // original-value parameter for each entry. + // original-value parameter for each entry, except where the original value + // is required. addByteAddressBufferInterlockedMethod("InterlockedAdd", AST.UnsignedIntTy, "__builtin_hlsl_interlocked_add"); addByteAddressBufferInterlockedMethod("InterlockedAnd", AST.UnsignedIntTy, "__builtin_hlsl_interlocked_and"); + addByteAddressBufferInterlockedMethod( + "InterlockedExchange", AST.UnsignedIntTy, + "__builtin_hlsl_interlocked_exchange", /*RequiresOriginalValue=*/true); addByteAddressBufferInterlockedMethod("InterlockedMax", AST.IntTy, "__builtin_hlsl_interlocked_max"); addByteAddressBufferInterlockedMethod("InterlockedMax", AST.UnsignedIntTy, @@ -1787,6 +1791,9 @@ BuiltinTypeDeclBuilder::addByteAddressBufferInterlockedMethods() { addByteAddressBufferInterlockedMethod("InterlockedAnd64", AST.UnsignedLongTy, "__builtin_hlsl_interlocked_and"); + addByteAddressBufferInterlockedMethod( + "InterlockedExchange64", AST.UnsignedLongTy, + "__builtin_hlsl_interlocked_exchange", /*RequiresOriginalValue=*/true); addByteAddressBufferInterlockedMethod("InterlockedMax64", AST.LongTy, "__builtin_hlsl_interlocked_max"); addByteAddressBufferInterlockedMethod("InterlockedMax64", @@ -2675,7 +2682,8 @@ BuiltinTypeDeclBuilder::addStoreFunction(DeclarationName &Name, bool IsConst, BuiltinTypeDeclBuilder & BuiltinTypeDeclBuilder::addByteAddressBufferInterlockedMethod( - StringRef MethodName, QualType ValueTy, StringRef BuiltinName) { + StringRef MethodName, QualType ValueTy, StringRef BuiltinName, + bool RequiresOriginalValue) { assert(!Record->isCompleteDefinition() && "record is already complete"); ASTContext &AST = SemaRef.getASTContext(); using PH = BuiltinTypeMethodBuilder::PlaceHolder; @@ -2704,7 +2712,8 @@ BuiltinTypeDeclBuilder::addByteAddressBufferInterlockedMethod( MMB.finalize(); }; - BuildOverload(/*WithOriginalValue=*/false); + if (!RequiresOriginalValue) + BuildOverload(/*WithOriginalValue=*/false); BuildOverload(/*WithOriginalValue=*/true); return *this; } diff --git a/clang/lib/Sema/HLSLBuiltinTypeDeclBuilder.h b/clang/lib/Sema/HLSLBuiltinTypeDeclBuilder.h index 62dea7fab80647..6d96b1df4cfea3 100644 --- a/clang/lib/Sema/HLSLBuiltinTypeDeclBuilder.h +++ b/clang/lib/Sema/HLSLBuiltinTypeDeclBuilder.h @@ -149,7 +149,8 @@ class BuiltinTypeDeclBuilder { bool TransposeArg = false); BuiltinTypeDeclBuilder & addByteAddressBufferInterlockedMethod(StringRef MethodName, QualType ValueTy, - StringRef BuiltinName); + StringRef BuiltinName, + bool RequiresOriginalValue = false); BuiltinTypeDeclBuilder &addAppendMethod(); BuiltinTypeDeclBuilder &addConsumeMethod(); diff --git a/clang/lib/Sema/HLSLExternalSemaSource.cpp b/clang/lib/Sema/HLSLExternalSemaSource.cpp index ad6cedc90bc25a..b3df9b9bd2b8bd 100644 --- a/clang/lib/Sema/HLSLExternalSemaSource.cpp +++ b/clang/lib/Sema/HLSLExternalSemaSource.cpp @@ -886,10 +886,11 @@ static void buildAtomicOverload(Sema &S, NamespaceDecl *NS, StringRef FuncName, } // Synthesize the InterlockedFunc overload set: {int, uint, int64_t, uint64_t} -// x {groupshared, device} x {2-arg, 3-arg}. +// x {groupshared, device} x {2-arg, 3-arg}. Operations that always report the +// previous value, such as InterlockedExchange, only get the 3-arg form. static void defineHLSLInterlockedFunc(Sema &S, NamespaceDecl *NS, - StringRef FuncName, - StringRef BuiltinName) { + StringRef FuncName, StringRef BuiltinName, + bool RequiresOriginalValue = false) { ASTContext &AST = S.getASTContext(); // HLSL: int64_t == long, uint64_t == unsigned long (see hlsl_basic_types.h). QualType Elems[] = {AST.IntTy, AST.UnsignedIntTy, AST.LongTy, @@ -898,8 +899,11 @@ static void defineHLSLInterlockedFunc(Sema &S, NamespaceDecl *NS, for (QualType ElemTy : Elems) for (LangAS AS : AddrSpaces) - for (bool ThreeArg : {false, true}) + for (bool ThreeArg : {false, true}) { + if (RequiresOriginalValue && !ThreeArg) + continue; buildAtomicOverload(S, NS, FuncName, BuiltinName, ElemTy, AS, ThreeArg); + } } void HLSLExternalSemaSource::defineHLSLAtomicIntrinsics() { @@ -907,6 +911,9 @@ void HLSLExternalSemaSource::defineHLSLAtomicIntrinsics() { "__builtin_hlsl_interlocked_add"); defineHLSLInterlockedFunc(*SemaPtr, HLSLNamespace, "InterlockedAnd", "__builtin_hlsl_interlocked_and"); + defineHLSLInterlockedFunc(*SemaPtr, HLSLNamespace, "InterlockedExchange", + "__builtin_hlsl_interlocked_exchange", + /*RequiresOriginalValue=*/true); defineHLSLInterlockedFunc(*SemaPtr, HLSLNamespace, "InterlockedMax", "__builtin_hlsl_interlocked_max"); defineHLSLInterlockedFunc(*SemaPtr, HLSLNamespace, "InterlockedMin", diff --git a/clang/lib/Sema/SemaHLSL.cpp b/clang/lib/Sema/SemaHLSL.cpp index a5429a90f962ba..84bf12fc680734 100644 --- a/clang/lib/Sema/SemaHLSL.cpp +++ b/clang/lib/Sema/SemaHLSL.cpp @@ -4602,6 +4602,7 @@ bool SemaHLSL::CheckBuiltinFunctionCall(unsigned BuiltinID, CallExpr *TheCall) { } case Builtin::BI__builtin_hlsl_interlocked_add: case Builtin::BI__builtin_hlsl_interlocked_and: + case Builtin::BI__builtin_hlsl_interlocked_exchange: case Builtin::BI__builtin_hlsl_interlocked_max: case Builtin::BI__builtin_hlsl_interlocked_min: case Builtin::BI__builtin_hlsl_interlocked_or: @@ -4613,15 +4614,22 @@ bool SemaHLSL::CheckBuiltinFunctionCall(unsigned BuiltinID, CallExpr *TheCall) { // argument count, integer-type matching, and the address-space requirement // on `dest`. The checks below are a safety net for callers that invoke the // builtin by its mangled name and would otherwise reach CodeGen unchecked. - if (TheCall->getNumArgs() < 2) { - SemaRef.Diag(TheCall->getEndLoc(), - diag::err_typecheck_call_too_few_args_at_least) - << /*callee_type=*/0 << /*min_arg_count=*/2 << TheCall->getNumArgs() - << /*is_non_object=*/0 << TheCall->getSourceRange(); - return true; + // InterlockedExchange always reports the previous value, so it requires + // `original_value` instead of accepting it as an optional argument. + if (BuiltinID == Builtin::BI__builtin_hlsl_interlocked_exchange) { + if (SemaRef.checkArgCount(TheCall, 3)) + return true; + } else { + if (TheCall->getNumArgs() < 2) { + SemaRef.Diag(TheCall->getEndLoc(), + diag::err_typecheck_call_too_few_args_at_least) + << /*callee_type=*/0 << /*min_arg_count=*/2 << TheCall->getNumArgs() + << /*is_non_object=*/0 << TheCall->getSourceRange(); + return true; + } + if (SemaRef.checkArgCountAtMost(TheCall, 3)) + return true; } - if (SemaRef.checkArgCountAtMost(TheCall, 3)) - return true; QualType DestTy = TheCall->getArg(0)->getType().getUnqualifiedType(); if (!DestTy->isIntegerType()) { diff --git a/clang/test/CodeGenHLSL/builtins/InterlockedExchange.hlsl b/clang/test/CodeGenHLSL/builtins/InterlockedExchange.hlsl new file mode 100644 index 00000000000000..ce47a08fb1a2bb --- /dev/null +++ b/clang/test/CodeGenHLSL/builtins/InterlockedExchange.hlsl @@ -0,0 +1,48 @@ +// RUN: %clang_cc1 -std=hlsl2021 -finclude-default-header -triple \ +// RUN: dxil-pc-shadermodel6.6-library %s -emit-llvm -disable-llvm-passes -o - | \ +// RUN: FileCheck %s --check-prefixes=CHECK,DXCHECK + +// RUN: %clang_cc1 -std=hlsl2021 -finclude-default-header -triple \ +// RUN: spirv-pc-vulkan-library %s -emit-llvm -disable-llvm-passes -o - | \ +// RUN: FileCheck %s --check-prefixes=CHECK,SPVCHECK + +// Test basic lowering of HLSL InterlockedExchange to `atomicrmw xchg +// monotonic`. InterlockedExchange always reports the previous value, so it +// only has a 3-argument form. + +groupshared int gs_i32; +groupshared uint gs_u32; +groupshared int64_t gs_i64; +groupshared uint64_t gs_u64; + +// CHECK-LABEL: define {{.*}}void @{{.*}}test_int_3arg +// DXCHECK: %[[R:.*]] = atomicrmw xchg ptr addrspace(3) {{.*}}@gs_i32{{.*}}, i32 %{{.*}} syncscope("workgroup") monotonic +// SPVCHECK: %[[R:.*]] = atomicrmw xchg ptr addrspace(3) {{.*}}@gs_i32{{.*}}, i32 %{{.*}} syncscope("workgroup") monotonic +// CHECK: store i32 %[[R]], ptr {{.*}} +export void test_int_3arg(int v, out int orig) { + InterlockedExchange(gs_i32, v, orig); +} + +// CHECK-LABEL: define {{.*}}void @{{.*}}test_uint_3arg +// DXCHECK: %[[R:.*]] = atomicrmw xchg ptr addrspace(3) {{.*}}@gs_u32{{.*}}, i32 %{{.*}} syncscope("workgroup") monotonic +// SPVCHECK: %[[R:.*]] = atomicrmw xchg ptr addrspace(3) {{.*}}@gs_u32{{.*}}, i32 %{{.*}} syncscope("workgroup") monotonic +// CHECK: store i32 %[[R]], ptr {{.*}} +export void test_uint_3arg(uint v, out uint orig) { + InterlockedExchange(gs_u32, v, orig); +} + +// CHECK-LABEL: define {{.*}}void @{{.*}}test_int64_3arg +// DXCHECK: %[[R:.*]] = atomicrmw xchg ptr addrspace(3) {{.*}}@gs_i64{{.*}}, i64 %{{.*}} syncscope("workgroup") monotonic +// SPVCHECK: %[[R:.*]] = atomicrmw xchg ptr addrspace(3) {{.*}}@gs_i64{{.*}}, i64 %{{.*}} syncscope("workgroup") monotonic +// CHECK: store i64 %[[R]], ptr {{.*}} +export void test_int64_3arg(int64_t v, out int64_t orig) { + InterlockedExchange(gs_i64, v, orig); +} + +// CHECK-LABEL: define {{.*}}void @{{.*}}test_uint64_3arg +// DXCHECK: %[[R:.*]] = atomicrmw xchg ptr addrspace(3) {{.*}}@gs_u64{{.*}}, i64 %{{.*}} syncscope("workgroup") monotonic +// SPVCHECK: %[[R:.*]] = atomicrmw xchg ptr addrspace(3) {{.*}}@gs_u64{{.*}}, i64 %{{.*}} syncscope("workgroup") monotonic +// CHECK: store i64 %[[R]], ptr {{.*}} +export void test_uint64_3arg(uint64_t v, out uint64_t orig) { + InterlockedExchange(gs_u64, v, orig); +} diff --git a/clang/test/CodeGenHLSL/builtins/RWBuffer-Interlocked.hlsl b/clang/test/CodeGenHLSL/builtins/RWBuffer-Interlocked.hlsl index 81c53ee4a5baa3..42b1813dc2dc3b 100644 --- a/clang/test/CodeGenHLSL/builtins/RWBuffer-Interlocked.hlsl +++ b/clang/test/CodeGenHLSL/builtins/RWBuffer-Interlocked.hlsl @@ -37,6 +37,8 @@ RWBuffer<uint> UOut : register(u1); // DXCHECK: atomicrmw max ptr %[[PTR7]], i32 1 syncscope("device") monotonic // DXCHECK: %[[PTR8:.*]] = call {{.*}} @llvm.dx.resource.getpointer.p0.tdx.TypedBuffer_i32_1_0_0t.i32(target("dx.TypedBuffer", i32, 1, 0, 0) %{{.*}}, i32 %{{.*}}) // DXCHECK: atomicrmw umax ptr %[[PTR8]], i32 1 syncscope("device") monotonic +// DXCHECK: %[[PTR9:.*]] = call {{.*}} @llvm.dx.resource.getpointer.p0.tdx.TypedBuffer_i32_1_0_1t.i32(target("dx.TypedBuffer", i32, 1, 0, 1) %{{.*}}, i32 %{{.*}}) +// DXCHECK: atomicrmw xchg ptr %[[PTR9]], i32 1 syncscope("device") monotonic // SPVCHECK: %[[PTR1:.*]] = call {{.*}} @llvm.spv.resource.getpointer.{{.*}}(target("spirv.SignedImage", i32, {{.*}}) %{{.*}}, i32 %{{.*}}) // SPVCHECK: atomicrmw add ptr addrspace(11) %[[PTR1]], i32 1 syncscope("device") monotonic // SPVCHECK: %[[PTR2:.*]] = call {{.*}} @llvm.spv.resource.getpointer.{{.*}}(target("spirv.SignedImage", i32, {{.*}}) %{{.*}}, i32 %{{.*}}) @@ -53,6 +55,8 @@ RWBuffer<uint> UOut : register(u1); // SPVCHECK: atomicrmw max ptr addrspace(11) %[[PTR7]], i32 1 syncscope("device") monotonic // SPVCHECK: %[[PTR8:.*]] = call {{.*}} @llvm.spv.resource.getpointer.{{.*}}(target("spirv.Image", i32, {{.*}}) %{{.*}}, i32 %{{.*}}) // SPVCHECK: atomicrmw umax ptr addrspace(11) %[[PTR8]], i32 1 syncscope("device") monotonic +// SPVCHECK: %[[PTR9:.*]] = call {{.*}} @llvm.spv.resource.getpointer.{{.*}}(target("spirv.SignedImage", i32, {{.*}}) %{{.*}}, i32 %{{.*}}) +// SPVCHECK: atomicrmw xchg ptr addrspace(11) %[[PTR9]], i32 1 syncscope("device") monotonic [shader("compute")] [numthreads(1,1,1)] void main(uint3 id : SV_DispatchThreadID) { @@ -64,4 +68,6 @@ void main(uint3 id : SV_DispatchThreadID) { InterlockedAnd(Out[id.x], 1); InterlockedMax(Out[id.x], 1); InterlockedMax(UOut[id.x], 1u); + int orig; + InterlockedExchange(Out[id.x], 1, orig); } diff --git a/clang/test/CodeGenHLSL/builtins/RWByteAddressBuffer-InterlockedExchange.hlsl b/clang/test/CodeGenHLSL/builtins/RWByteAddressBuffer-InterlockedExchange.hlsl new file mode 100644 index 00000000000000..ce03bd0504053e --- /dev/null +++ b/clang/test/CodeGenHLSL/builtins/RWByteAddressBuffer-InterlockedExchange.hlsl @@ -0,0 +1,40 @@ +// RUN: %clang_cc1 -std=hlsl202x -finclude-default-header -triple \ +// RUN: dxil-pc-shadermodel6.6-library %s -emit-llvm -disable-llvm-passes -o - | \ +// RUN: FileCheck %s --check-prefixes=CHECK,DXCHECK + +// RUN: %clang_cc1 -std=hlsl202x -finclude-default-header -triple \ +// RUN: spirv-pc-vulkan1.3-library %s -emit-llvm -disable-llvm-passes -o - | \ +// RUN: FileCheck %s --check-prefixes=CHECK,SPVCHECK + +// Test that the RWByteAddressBuffer::InterlockedExchange and +// InterlockedExchange64 member methods lower to `resource_getpointer -> +// atomicrmw xchg`, and that the returned original value is stored through the +// out parameter, for both DXIL and SPIR-V targets. + +RWByteAddressBuffer BAB : register(u0); + +// CHECK-LABEL: define {{.*}}void @{{.*}}test_bab_uint_3arg +// DXCHECK: %[[HANDLE:.*]] = load target("dx.RawBuffer", i8, 1, 0), ptr {{.*}} +// DXCHECK: %[[PTR:.*]] = call ptr @llvm.dx.resource.getpointer.p0.tdx.RawBuffer_i8_1_0t.i32(target("dx.RawBuffer", i8, 1, 0) %[[HANDLE]], i32 %{{.*}}) +// DXCHECK: %[[R:.*]] = atomicrmw xchg ptr %[[PTR]], i32 %{{.*}} syncscope("device") monotonic +// DXCHECK: store i32 %[[R]], ptr {{.*}} +// SPVCHECK: %[[HANDLE:.*]] = load target("spirv.VulkanBuffer", [0 x i8], 12, 1), ptr {{.*}} +// SPVCHECK: %[[PTR:.*]] = call ptr addrspace(11) @llvm.spv.resource.getpointer.p11.tspirv.VulkanBuffer_a0i8_12_1t.i32(target("spirv.VulkanBuffer", [0 x i8], 12, 1) %[[HANDLE]], i32 %{{.*}}) +// SPVCHECK: %[[R:.*]] = atomicrmw xchg ptr addrspace(11) %[[PTR]], i32 %{{.*}} syncscope("device") monotonic +// SPVCHECK: store i32 %[[R]], ptr {{.*}} +export void test_bab_uint_3arg(uint off, uint v, out uint orig) { + BAB.InterlockedExchange(off, v, orig); +} + +// CHECK-LABEL: define {{.*}}void @{{.*}}test_bab_uint64_3arg +// DXCHECK: %[[HANDLE:.*]] = load target("dx.RawBuffer", i8, 1, 0), ptr {{.*}} +// DXCHECK: %[[PTR:.*]] = call ptr @llvm.dx.resource.getpointer.p0.tdx.RawBuffer_i8_1_0t.i32(target("dx.RawBuffer", i8, 1, 0) %[[HANDLE]], i32 %{{.*}}) +// DXCHECK: %[[R:.*]] = atomicrmw xchg ptr %[[PTR]], i64 %{{.*}} syncscope("device") monotonic +// DXCHECK: store i64 %[[R]], ptr {{.*}} +// SPVCHECK: %[[HANDLE:.*]] = load target("spirv.VulkanBuffer", [0 x i8], 12, 1), ptr {{.*}} +// SPVCHECK: %[[PTR:.*]] = call ptr addrspace(11) @llvm.spv.resource.getpointer.p11.tspirv.VulkanBuffer_a0i8_12_1t.i32(target("spirv.VulkanBuffer", [0 x i8], 12, 1) %[[HANDLE]], i32 %{{.*}}) +// SPVCHECK: %[[R:.*]] = atomicrmw xchg ptr addrspace(11) %[[PTR]], i64 %{{.*}} syncscope("device") monotonic +// SPVCHECK: store i64 %[[R]], ptr {{.*}} +export void test_bab_uint64_3arg(uint off, uint64_t v, out uint64_t orig) { + BAB.InterlockedExchange64(off, v, orig); +} diff --git a/clang/test/CodeGenHLSL/builtins/RasterizerOrderedByteAddressBuffer-InterlockedExchange.hlsl b/clang/test/CodeGenHLSL/builtins/RasterizerOrderedByteAddressBuffer-InterlockedExchange.hlsl new file mode 100644 index 00000000000000..7a675e47ea47cc --- /dev/null +++ b/clang/test/CodeGenHLSL/builtins/RasterizerOrderedByteAddressBuffer-InterlockedExchange.hlsl @@ -0,0 +1,29 @@ +// RUN: %clang_cc1 -std=hlsl202x -finclude-default-header -triple \ +// RUN: dxil-pc-shadermodel6.6-library %s -emit-llvm -disable-llvm-passes -o - | \ +// RUN: FileCheck %s --check-prefixes=CHECK,DXCHECK + +// SPIR-V codegen for RasterizerOrderedByteAddressBuffer is not implemented +// yet (asserts in clang/lib/CodeGen/Targets/SPIR.cpp on +// `!ResAttrs.IsROV && "Rasterizer order views not implemented for SPIR-V yet"`). +// Add a `spirv-pc-vulkan1.3-library` RUN line here when SPIR-V ROV support +// lands. + +RasterizerOrderedByteAddressBuffer ROVB : register(u1); + +// CHECK-LABEL: define void @{{.*}}test_rovb_uint_3arg +// DXCHECK: %[[HANDLE:.*]] = load target("dx.RawBuffer", i8, 1, 1), ptr {{.*}} +// DXCHECK: %[[PTR:.*]] = call ptr @llvm.dx.resource.getpointer.p0.tdx.RawBuffer_i8_1_1t.i32(target("dx.RawBuffer", i8, 1, 1) %[[HANDLE]], i32 %{{.*}}) +// DXCHECK: %[[R:.*]] = atomicrmw xchg ptr %[[PTR]], i32 %{{.*}} syncscope("device") monotonic +// DXCHECK: store i32 %[[R]], ptr {{.*}} +export void test_rovb_uint_3arg(uint off, uint v, out uint orig) { + ROVB.InterlockedExchange(off, v, orig); +} + +// CHECK-LABEL: define void @{{.*}}test_rovb_uint64_3arg +// DXCHECK: %[[HANDLE:.*]] = load target("dx.RawBuffer", i8, 1, 1), ptr {{.*}} +// DXCHECK: %[[PTR:.*]] = call ptr @llvm.dx.resource.getpointer.p0.tdx.RawBuffer_i8_1_1t.i32(target("dx.RawBuffer", i8, 1, 1) %[[HANDLE]], i32 %{{.*}}) +// DXCHECK: %[[R:.*]] = atomicrmw xchg ptr %[[PTR]], i64 %{{.*}} syncscope("device") monotonic +// DXCHECK: store i64 %[[R]], ptr {{.*}} +export void test_rovb_uint64_3arg(uint off, uint64_t v, out uint64_t orig) { + ROVB.InterlockedExchange64(off, v, orig); +} diff --git a/clang/test/SemaHLSL/BuiltIns/ByteAddressBuffer-InterlockedExchange-errors.hlsl b/clang/test/SemaHLSL/BuiltIns/ByteAddressBuffer-InterlockedExchange-errors.hlsl new file mode 100644 index 00000000000000..ed221d45771069 --- /dev/null +++ b/clang/test/SemaHLSL/BuiltIns/ByteAddressBuffer-InterlockedExchange-errors.hlsl @@ -0,0 +1,46 @@ +// RUN: %clang_cc1 -std=hlsl202x -finclude-default-header \ +// RUN: -triple dxil-pc-shadermodel6.6-library %s -fsyntax-only -verify \ +// RUN: -verify-ignore-unexpected=note,warning + +// Unlike the other interlocked methods, InterlockedExchange declares a single +// overload per element type because the original value is required. There is +// therefore no overload set to fail against, so Clang reports the argument +// mismatch directly instead of 'no matching member function'. + +RWByteAddressBuffer BAB : register(u0); +RasterizerOrderedByteAddressBuffer ROVB : register(u1); + +struct S { int x; }; + +void too_few(uint off) { + BAB.InterlockedExchange(off); + // expected-error@-1 {{too few arguments to function call, expected 3, have 1}} +} + +void missing_original_value(uint off, uint v) { + BAB.InterlockedExchange(off, v); + // expected-error@-1 {{too few arguments to function call, expected 3, have 2}} +} + +void too_many(uint off, uint v, uint extra) { + uint orig; + BAB.InterlockedExchange(off, v, orig, extra); + // expected-error@-1 {{too many arguments to function call, expected 3, have 4}} +} + +void struct_value(uint off, S v) { + uint orig; + BAB.InterlockedExchange(off, v, orig); + // expected-error@-1 {{cannot initialize a parameter of type 'unsigned int' with an lvalue of type 'S'}} +} + +void rovb_missing_original_value(uint off, uint v) { + ROVB.InterlockedExchange(off, v); + // expected-error@-1 {{too few arguments to function call, expected 3, have 2}} +} + +void rovb_struct_value(uint off, S v) { + uint orig; + ROVB.InterlockedExchange(off, v, orig); + // expected-error@-1 {{cannot initialize a parameter of type 'unsigned int' with an lvalue of type 'S'}} +} diff --git a/clang/test/SemaHLSL/BuiltIns/ByteAddressBuffer-InterlockedExchange-sm65-errors.hlsl b/clang/test/SemaHLSL/BuiltIns/ByteAddressBuffer-InterlockedExchange-sm65-errors.hlsl new file mode 100644 index 00000000000000..3205f156827fa4 --- /dev/null +++ b/clang/test/SemaHLSL/BuiltIns/ByteAddressBuffer-InterlockedExchange-sm65-errors.hlsl @@ -0,0 +1,26 @@ +// RUN: %clang_cc1 -std=hlsl202x -finclude-default-header \ +// RUN: -triple dxil-pc-shadermodel6.5-library %s -fsyntax-only -verify \ +// RUN: -verify-ignore-unexpected=warning + +RWByteAddressBuffer BAB : register(u0); +RasterizerOrderedByteAddressBuffer ROVB : register(u1); + +void sm65_no_bab_exchange64(uint off, uint64_t v, out uint64_t orig) { + BAB.InterlockedExchange64(off, v, orig); + // expected-error@-1 {{no member named 'InterlockedExchange64' in 'hlsl::RWByteAddressBuffer'}} +} + +void sm65_no_rovb_exchange64(uint off, uint64_t v, out uint64_t orig) { + ROVB.InterlockedExchange64(off, v, orig); + // expected-error@-1 {{no member named 'InterlockedExchange64' in 'hlsl::RasterizerOrderedByteAddressBuffer'}} +} + +void sm65_bab_exchange32_ok(uint off, uint v, out uint orig) { + BAB.InterlockedExchange(off, v, orig); +} + +groupshared int64_t gs_i64; +void sm65_direct_builtin(int64_t v, out int64_t orig) { + __builtin_hlsl_interlocked_exchange(gs_i64, v, orig); + // expected-error@-1 {{'__builtin_hlsl_interlocked_exchange' requires shader model 6.6 or newer}} +} diff --git a/clang/test/SemaHLSL/BuiltIns/InterlockedExchange-errors.hlsl b/clang/test/SemaHLSL/BuiltIns/InterlockedExchange-errors.hlsl new file mode 100644 index 00000000000000..bf6a1757b42f2c --- /dev/null +++ b/clang/test/SemaHLSL/BuiltIns/InterlockedExchange-errors.hlsl @@ -0,0 +1,107 @@ +// RUN: %clang_cc1 -std=hlsl202x -finclude-default-header \ +// RUN: -triple dxil-pc-shadermodel6.6-library %s -emit-llvm-only \ +// RUN: -disable-llvm-passes -verify + +// InterlockedExchange is provided as a set of address-space-qualified +// overloads (groupshared/device, {int,uint,int64_t,uint64_t}). It always +// reports the previous value, so there is no 2-argument form. + +groupshared int gs_i32; +groupshared float gs_f32; +struct S { int x; }; +groupshared S gs_s; + +void too_few() { + InterlockedExchange(gs_i32); // expected-error{{no matching function for call to 'InterlockedExchange'}} + // expected-note@*:* 8 {{candidate function}} +} + +void missing_original_value(int v) { + InterlockedExchange(gs_i32, v); // expected-error{{no matching function for call to 'InterlockedExchange'}} + // expected-note@*:* 8 {{candidate function}} +} + +void too_many(int v, int extra) { + int orig; + InterlockedExchange(gs_i32, v, orig, extra); // expected-error{{no matching function for call to 'InterlockedExchange'}} + // expected-note@*:* 8 {{candidate function}} +} + +void local_dest(int v) { + int dest; + int orig; + InterlockedExchange(dest, v, orig); // expected-error{{no matching function for call to 'InterlockedExchange'}} + // expected-note@*:* 8 {{candidate function}} +} + +void float_dest(float v) { + float orig; + InterlockedExchange(gs_f32, v, orig); // expected-error{{no matching function for call to 'InterlockedExchange'}} + // expected-note@*:* 8 {{candidate function}} +} + +void struct_dest(int v) { + int orig; + InterlockedExchange(gs_s, v, orig); // expected-error{{no matching function for call to 'InterlockedExchange'}} + // expected-note@*:* 8 {{candidate function}} +} + +void mismatched_orig_type(int v) { + uint orig; + InterlockedExchange(gs_i32, v, orig); // expected-error{{no matching function for call to 'InterlockedExchange'}} + // expected-note@*:* 8 {{candidate function}} +} + +void direct_too_few() { + __builtin_hlsl_interlocked_exchange(gs_i32); + // expected-error@-1 {{too few arguments to function call, expected 3, have 1}} +} + +void direct_missing_original_value(int v) { + __builtin_hlsl_interlocked_exchange(gs_i32, v); + // expected-error@-1 {{too few arguments to function call, expected 3, have 2}} +} + +void direct_too_many(int v, int extra) { + int orig; + __builtin_hlsl_interlocked_exchange(gs_i32, v, orig, extra); + // expected-error@-1 {{too many arguments to function call, expected 3, have 4}} +} + +void direct_non_integer_dest() { + S local_s; + S orig; + __builtin_hlsl_interlocked_exchange(local_s, 1, orig); + // expected-error@-1 {{1st argument must be a scalar integer type (was 'S')}} +} + +void direct_nonlvalue_dest(int v) { + int orig; + __builtin_hlsl_interlocked_exchange(1, v, orig); + // expected-error@-1 {{cannot bind non-lvalue argument '1' to out parameter}} +} + +void direct_mismatched_value() { + uint value = 1; + int orig; + __builtin_hlsl_interlocked_exchange(gs_i32, value, orig); + // expected-error@-1 {{passing 'uint' (aka 'unsigned int') to parameter of incompatible type 'int'}} +} + +void direct_mismatched_orig(int v) { + uint orig; + __builtin_hlsl_interlocked_exchange(gs_i32, v, orig); + // expected-error@-1 {{passing 'uint' (aka 'unsigned int') to parameter of incompatible type 'int'}} +} + +void direct_nonlvalue_orig(int v) { + __builtin_hlsl_interlocked_exchange(gs_i32, v, 1); + // expected-error@-1 {{cannot bind non-lvalue argument '1' to out parameter}} +} + +void direct_default_as_dest(int v) { + int local; + int orig; + __builtin_hlsl_interlocked_exchange(local, v, orig); + // expected-error@-1 {{1st argument to atomic builtin must reference groupshared or device memory (was 'int')}} +} >From 00449f10bd2e5d38bda0ae7535d5a4cb1ef18942 Mon Sep 17 00:00:00 2001 From: Joshua Batista <[email protected]> Date: Wed, 16 Sep 2026 13:59:06 -0700 Subject: [PATCH 2/2] Add InterlockedExchange coverage to the texture test --- clang/test/CodeGenHLSL/builtins/RWTexture-Interlocked.hlsl | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/clang/test/CodeGenHLSL/builtins/RWTexture-Interlocked.hlsl b/clang/test/CodeGenHLSL/builtins/RWTexture-Interlocked.hlsl index 90c8c7e33836a1..5013b11c62636e 100644 --- a/clang/test/CodeGenHLSL/builtins/RWTexture-Interlocked.hlsl +++ b/clang/test/CodeGenHLSL/builtins/RWTexture-Interlocked.hlsl @@ -35,6 +35,8 @@ RWTexture2DArray<uint> UOut : register(u1); // DXCHECK: atomicrmw max ptr %[[PTR6]], i32 1 syncscope("device") monotonic // DXCHECK: %[[PTR7:.*]] = call {{.*}} @llvm.dx.resource.getpointer.{{.*}}(target("dx.Texture", i32, 1, 0, 0, 7) %{{.*}}, <3 x i32> %{{.*}}) // DXCHECK: atomicrmw umax ptr %[[PTR7]], i32 1 syncscope("device") monotonic +// DXCHECK: %[[PTR8:.*]] = call {{.*}} @llvm.dx.resource.getpointer.{{.*}}(target("dx.Texture", i32, 1, 0, 1, 2) %{{.*}}, <2 x i32> %{{.*}}) +// DXCHECK: atomicrmw xchg ptr %[[PTR8]], i32 1 syncscope("device") monotonic // SPVCHECK: %[[PTR1:.*]] = call {{.*}} @llvm.spv.resource.getpointer.{{.*}}(target("spirv.SignedImage", i32, {{.*}}) %{{.*}}, <2 x i32> %{{.*}}) // SPVCHECK: atomicrmw add ptr addrspace(11) %[[PTR1]], i32 1 syncscope("device") monotonic // SPVCHECK: %[[PTR2:.*]] = call {{.*}} @llvm.spv.resource.getpointer.{{.*}}(target("spirv.SignedImage", i32, {{.*}}) %{{.*}}, <2 x i32> %{{.*}}) @@ -49,6 +51,8 @@ RWTexture2DArray<uint> UOut : register(u1); // SPVCHECK: atomicrmw max ptr addrspace(11) %[[PTR6]], i32 1 syncscope("device") monotonic // SPVCHECK: %[[PTR7:.*]] = call {{.*}} @llvm.spv.resource.getpointer.{{.*}}(target("spirv.Image", i32, {{.*}}) %{{.*}}, <3 x i32> %{{.*}}) // SPVCHECK: atomicrmw umax ptr addrspace(11) %[[PTR7]], i32 1 syncscope("device") monotonic +// SPVCHECK: %[[PTR8:.*]] = call {{.*}} @llvm.spv.resource.getpointer.{{.*}}(target("spirv.SignedImage", i32, {{.*}}) %{{.*}}, <2 x i32> %{{.*}}) +// SPVCHECK: atomicrmw xchg ptr addrspace(11) %[[PTR8]], i32 1 syncscope("device") monotonic [shader("compute")] [numthreads(1,1,1)] void main(uint3 id : SV_DispatchThreadID) { @@ -59,4 +63,7 @@ void main(uint3 id : SV_DispatchThreadID) { InterlockedMin(UOut[id], 1u); InterlockedMax(Out[id.xy], 1); InterlockedMax(UOut[id], 1u); + // DXIL has no float texture atomic, so only the integer form is covered. + int Orig; + InterlockedExchange(Out[id.xy], 1, Orig); } _______________________________________________ cfe-commits mailing list [email protected] https://lists.llvm.org/cgi-bin/mailman/listinfo/cfe-commits
