https://github.com/arsenm created https://github.com/llvm/llvm-project/pull/229352
Improve the phrasing when the atomicrmw legalization emits the hardware instruction. Replace the unhelpful "due to an unsafe request" wording in the optimization remark with the actual reason the native instruction was legal to use. Co-authored-by: Claude Opus 5.5 <[email protected]> >From 2b68c681f577844373183fbe7e196e5401d7b8fd Mon Sep 17 00:00:00 2001 From: Matt Arsenault <[email protected]> Date: Tue, 6 Oct 2026 10:29:26 +0200 Subject: [PATCH] AMDGPU: Improve optimization remarks for atomic lowering Improve the phrasing when the atomicrmw legalization emits the hardware instruction. Replace the unhelpful "due to an unsafe request" wording in the optimization remark with the actual reason the native instruction was legal to use. Co-authored-by: Claude Opus 5.5 <[email protected]> --- .../atomics-unsafe-hw-remarks-gfx90a.cl | 6 +- llvm/lib/Target/AMDGPU/SIISelLowering.cpp | 156 +++++++++++++----- .../AMDGPU/atomics-hw-remarks-gfx908.ll | 115 +++++++++++++ .../AMDGPU/atomics-hw-remarks-gfx90a.ll | 19 +-- .../AMDGPU/atomics-hw-remarks-scope.ll | 41 +++++ 5 files changed, 281 insertions(+), 56 deletions(-) create mode 100644 llvm/test/CodeGen/AMDGPU/atomics-hw-remarks-gfx908.ll create mode 100644 llvm/test/CodeGen/AMDGPU/atomics-hw-remarks-scope.ll diff --git a/clang/test/CodeGenOpenCL/atomics-unsafe-hw-remarks-gfx90a.cl b/clang/test/CodeGenOpenCL/atomics-unsafe-hw-remarks-gfx90a.cl index 7fd20035ab430..0de6554d353c5 100644 --- a/clang/test/CodeGenOpenCL/atomics-unsafe-hw-remarks-gfx90a.cl +++ b/clang/test/CodeGenOpenCL/atomics-unsafe-hw-remarks-gfx90a.cl @@ -27,9 +27,9 @@ typedef enum memory_scope { #endif } memory_scope; -// GFX90A-HW-REMARK: Hardware instruction generated for atomic fadd operation at memory scope wavefront-one-as due to an unsafe request. [-Rpass=si-lower] -// GFX90A-HW-REMARK: Hardware instruction generated for atomic fadd operation at memory scope agent-one-as due to an unsafe request. [-Rpass=si-lower] -// GFX90A-HW-REMARK: Hardware instruction generated for atomic fadd operation at memory scope workgroup-one-as due to an unsafe request. [-Rpass=si-lower] +// GFX90A-HW-REMARK: hardware instruction generated for atomic fadd at wavefront-one-as scope since memory is not fine-grained (!amdgpu.no.fine.grained.memory), and denormals may be flushed (!atomic.ignore.denormal.mode) [-Rpass=si-lower] +// GFX90A-HW-REMARK: hardware instruction generated for atomic fadd at agent-one-as scope since memory is not fine-grained (!amdgpu.no.fine.grained.memory), and denormals may be flushed (!atomic.ignore.denormal.mode) [-Rpass=si-lower] +// GFX90A-HW-REMARK: hardware instruction generated for atomic fadd at workgroup-one-as scope since memory is not fine-grained (!amdgpu.no.fine.grained.memory), and denormals may be flushed (!atomic.ignore.denormal.mode) [-Rpass=si-lower] // GFX90A-HW-REMARK: global_atomic_add_f32 v{{[0-9]+}}, v[{{[0-9]+}}:{{[0-9]+}}], v{{[0-9]+}}, off glc // GFX90A-HW-REMARK: global_atomic_add_f32 v{{[0-9]+}}, v[{{[0-9]+}}:{{[0-9]+}}], v{{[0-9]+}}, off glc diff --git a/llvm/lib/Target/AMDGPU/SIISelLowering.cpp b/llvm/lib/Target/AMDGPU/SIISelLowering.cpp index 1fc3f8c396e4d..89f552c67846f 100644 --- a/llvm/lib/Target/AMDGPU/SIISelLowering.cpp +++ b/llvm/lib/Target/AMDGPU/SIISelLowering.cpp @@ -20815,26 +20815,88 @@ bool SITargetLowering::isKnownNeverNaNForTargetNode(SDValue Op, DAG, SNaN, Depth); } +namespace { + +/// Why a floating-point atomic instruction which may flush denormals is +/// acceptable. +enum class AtomicFlushDenormalReason { + Native, + IEEE, + IgnoreDenormalMode, + FunctionFlushesDenormals +}; + +/// Why a native floating-point atomic instruction is acceptable for a global +/// memory address. +enum class GlobalFPAtomicLegality { + Illegal, + AgentScopeFineGrainedRemoteMemory, + EmulatedSystemScope, + NoRemoteMemory, + NoFineGrainedMemory +}; + +} // end anonymous namespace + // On older subtargets, global FP atomic instructions have a hardcoded FP mode // and do not support FP32 denormals, and only support v2f16/f64 denormals. -static bool atomicIgnoresDenormalModeOrFPModeIsFTZ(const AtomicRMWInst *RMW) { +static AtomicFlushDenormalReason +getAtomicFlushDenormalReason(const AtomicRMWInst *RMW) { if (RMW->hasMetadata(LLVMContext::MD_atomic_ignore_denormal_mode)) - return true; + return AtomicFlushDenormalReason::IgnoreDenormalMode; const fltSemantics &Flt = RMW->getType()->getScalarType()->getFltSemantics(); auto DenormMode = RMW->getFunction()->getDenormalMode(Flt); - return DenormMode == DenormalMode::getPreserveSign(); + return DenormMode == DenormalMode::getPreserveSign() + ? AtomicFlushDenormalReason::FunctionFlushesDenormals + : AtomicFlushDenormalReason::IEEE; } -static OptimizationRemark emitAtomicRMWLegalRemark(const AtomicRMWInst *RMW) { +static OptimizationRemark +emitAtomicRMWLegalRemark(const AtomicRMWInst *RMW, + GlobalFPAtomicLegality MemLegality, + AtomicFlushDenormalReason DenormReason) { LLVMContext &Ctx = RMW->getContext(); - StringRef MemScope = - Ctx.getSyncScopeName(RMW->getSyncScopeID()).value_or("system"); + StringRef MemScope = Ctx.getSyncScopeName(RMW->getSyncScopeID()).value_or(""); + if (MemScope.empty()) + MemScope = "system"; + + OptimizationRemark R(DEBUG_TYPE, "Passed", RMW); + R << "hardware instruction generated for atomic " + << ore::NV("Operation", RMW->getOperationName(RMW->getOperation())) + << " at " << ore::NV("SyncScope", MemScope) << " scope since "; + + switch (MemLegality) { + case GlobalFPAtomicLegality::AgentScopeFineGrainedRemoteMemory: + R << "fine-grained remote memory atomics work below system scope"; + break; + case GlobalFPAtomicLegality::EmulatedSystemScope: + R << "system scope atomics are emulated in hardware"; + break; + case GlobalFPAtomicLegality::NoRemoteMemory: + R << "memory is not remote (!amdgpu.no.remote.memory)"; + break; + case GlobalFPAtomicLegality::NoFineGrainedMemory: + R << "memory is not fine-grained (!amdgpu.no.fine.grained.memory)"; + break; + case GlobalFPAtomicLegality::Illegal: + llvm_unreachable("remark for illegal atomic"); + } + + switch (DenormReason) { + case AtomicFlushDenormalReason::Native: + break; + case AtomicFlushDenormalReason::IgnoreDenormalMode: + R << ", and denormals may be flushed (!atomic.ignore.denormal.mode)"; + break; + case AtomicFlushDenormalReason::FunctionFlushesDenormals: + R << ", and the floating-point environment flushes denormals"; + break; + case AtomicFlushDenormalReason::IEEE: + llvm_unreachable("remark for illegal atomic"); + } - return OptimizationRemark(DEBUG_TYPE, "Passed", RMW) - << "Hardware instruction generated for atomic " - << RMW->getOperationName(RMW->getOperation()) - << " operation at memory scope " << MemScope; + return R; } static bool isV2F16OrV2BF16(Type *Ty) { @@ -20890,11 +20952,11 @@ static bool isAtomicRMWLegalXChgTy(const AtomicRMWInst *RMW) { return false; } -/// \returns true if it's valid to emit a native instruction for \p RMW, based -/// on the properties of the target memory. -static bool globalMemoryFPAtomicIsLegal(const GCNSubtarget &Subtarget, - const AtomicRMWInst *RMW, - bool HasSystemScope) { +/// \returns whether it's valid to emit a native instruction for \p RMW, and +/// why, based on the properties of the target memory. +static GlobalFPAtomicLegality +getGlobalMemoryFPAtomicLegality(const GCNSubtarget &Subtarget, + const AtomicRMWInst *RMW, bool HasSystemScope) { // The remote/fine-grained access logic is different from the integer // atomics. Without AgentScopeFineGrainedRemoteMemoryAtomics support, // fine-grained access does not work, even for a device local allocation. @@ -20904,13 +20966,15 @@ static bool globalMemoryFPAtomicIsLegal(const GCNSubtarget &Subtarget, if (HasSystemScope) { if (Subtarget.hasAgentScopeFineGrainedRemoteMemoryAtomics() && RMW->hasMetadata("amdgpu.no.remote.memory")) - return true; + return GlobalFPAtomicLegality::NoRemoteMemory; if (Subtarget.hasEmulatedSystemScopeAtomics()) - return true; + return GlobalFPAtomicLegality::EmulatedSystemScope; } else if (Subtarget.hasAgentScopeFineGrainedRemoteMemoryAtomics()) - return true; + return GlobalFPAtomicLegality::AgentScopeFineGrainedRemoteMemory; - return RMW->hasMetadata("amdgpu.no.fine.grained.memory"); + return RMW->hasMetadata("amdgpu.no.fine.grained.memory") + ? GlobalFPAtomicLegality::NoFineGrainedMemory + : GlobalFPAtomicLegality::Illegal; } /// \return Action to perform on AtomicRMWInsts for integer operations. @@ -20953,10 +21017,12 @@ SITargetLowering::shouldExpandAtomicRMWInIR(const AtomicRMWInst *RMW) const { flatInstrMayAccessPrivate(RMW)) return AtomicExpansionKind::CustomExpand; - auto ReportUnsafeHWInst = [=](TargetLowering::AtomicExpansionKind Kind) { + GlobalFPAtomicLegality MemLegality = GlobalFPAtomicLegality::Illegal; + AtomicFlushDenormalReason DenormReason = AtomicFlushDenormalReason::Native; + auto ReportHWInst = [&](TargetLowering::AtomicExpansionKind Kind) { OptimizationRemarkEmitter ORE(RMW->getFunction()); - ORE.emit([=]() { - return emitAtomicRMWLegalRemark(RMW) << " due to an unsafe request."; + ORE.emit([&]() { + return emitAtomicRMWLegalRemark(RMW, MemLegality, DenormReason); }); return Kind; }; @@ -21106,62 +21172,64 @@ SITargetLowering::shouldExpandAtomicRMWInIR(const AtomicRMWInst *RMW) const { // whether the target address resides in LDS or global memory. We consider // this flat-maybe-flush as will-flush. if (Ty->isFloatTy() && - !Subtarget->hasMemoryAtomicFaddF32DenormalSupport() && - !atomicIgnoresDenormalModeOrFPModeIsFTZ(RMW)) - return AtomicExpansionKind::CmpXChg; + !Subtarget->hasMemoryAtomicFaddF32DenormalSupport()) { + DenormReason = getAtomicFlushDenormalReason(RMW); + if (DenormReason == AtomicFlushDenormalReason::IEEE) + return AtomicExpansionKind::CmpXChg; + } - // FIXME: These ReportUnsafeHWInsts are imprecise. Some of these cases are - // safe. The message phrasing also should be better. - if (globalMemoryFPAtomicIsLegal(*Subtarget, RMW, HasSystemScope)) { + MemLegality = + getGlobalMemoryFPAtomicLegality(*Subtarget, RMW, HasSystemScope); + if (MemLegality != GlobalFPAtomicLegality::Illegal) { if (AS == AMDGPUAS::FLAT_ADDRESS) { // gfx942, gfx12 if (Subtarget->hasAtomicFlatPkAdd16Insts() && isV2F16OrV2BF16(Ty)) - return ReportUnsafeHWInst(AtomicExpansionKind::None); + return ReportHWInst(AtomicExpansionKind::None); } else if (AMDGPU::isExtendedGlobalAddrSpace(AS)) { // gfx90a, gfx942, gfx12 if (Subtarget->hasAtomicBufferGlobalPkAddF16Insts() && isV2F16(Ty)) - return ReportUnsafeHWInst(AtomicExpansionKind::None); + return ReportHWInst(AtomicExpansionKind::None); // gfx942, gfx12 if (Subtarget->hasAtomicGlobalPkAddBF16Inst() && isV2BF16(Ty)) - return ReportUnsafeHWInst(AtomicExpansionKind::None); + return ReportHWInst(AtomicExpansionKind::None); } else if (AS == AMDGPUAS::BUFFER_FAT_POINTER) { // gfx90a, gfx942, gfx12 if (Subtarget->hasAtomicBufferGlobalPkAddF16Insts() && isV2F16(Ty)) - return ReportUnsafeHWInst(AtomicExpansionKind::None); + return ReportHWInst(AtomicExpansionKind::None); // While gfx90a/gfx942 supports v2bf16 for global/flat, it does not for // buffer. gfx12 does have the buffer version. if (Subtarget->hasAtomicBufferPkAddBF16Inst() && isV2BF16(Ty)) - return ReportUnsafeHWInst(AtomicExpansionKind::None); + return ReportHWInst(AtomicExpansionKind::None); } // global and flat atomic fadd f64: gfx90a, gfx942. if (Subtarget->hasFlatBufferGlobalAtomicFaddF64Inst() && Ty->isDoubleTy()) - return ReportUnsafeHWInst(AtomicExpansionKind::None); + return ReportHWInst(AtomicExpansionKind::None); if (AS != AMDGPUAS::FLAT_ADDRESS) { if (Ty->isFloatTy()) { // global/buffer atomic fadd f32 no-rtn: gfx908, gfx90a, gfx942, // gfx11+. if (RMW->use_empty() && Subtarget->hasAtomicFaddNoRtnInsts()) - return ReportUnsafeHWInst(AtomicExpansionKind::None); + return ReportHWInst(AtomicExpansionKind::None); // global/buffer atomic fadd f32 rtn: gfx90a, gfx942, gfx11+. if (!RMW->use_empty() && Subtarget->hasAtomicFaddRtnInsts()) - return ReportUnsafeHWInst(AtomicExpansionKind::None); + return ReportHWInst(AtomicExpansionKind::None); } else { // gfx908 if (RMW->use_empty() && Subtarget->hasAtomicBufferGlobalPkAddF16NoRtnInsts() && isV2F16(Ty)) - return ReportUnsafeHWInst(AtomicExpansionKind::None); + return ReportHWInst(AtomicExpansionKind::None); } } // flat atomic fadd f32: gfx942, gfx11+. if (AS == AMDGPUAS::FLAT_ADDRESS && Ty->isFloatTy()) { if (Subtarget->hasFlatAtomicFaddF32Inst()) - return ReportUnsafeHWInst(AtomicExpansionKind::None); + return ReportHWInst(AtomicExpansionKind::None); // If it is in flat address space, and the type is float, we will try to // expand it, if the target supports global and lds atomic fadd. The @@ -21190,7 +21258,9 @@ SITargetLowering::shouldExpandAtomicRMWInIR(const AtomicRMWInst *RMW) const { : AtomicExpansionKind::CmpXChg; } - if (globalMemoryFPAtomicIsLegal(*Subtarget, RMW, HasSystemScope)) { + MemLegality = + getGlobalMemoryFPAtomicLegality(*Subtarget, RMW, HasSystemScope); + if (MemLegality != GlobalFPAtomicLegality::Illegal) { // For flat and global cases: // float, double in gfx7. Manual claims denormal support. // Removed in gfx8. @@ -21201,15 +21271,15 @@ SITargetLowering::shouldExpandAtomicRMWInIR(const AtomicRMWInst *RMW) const { // no f32. if (AS == AMDGPUAS::FLAT_ADDRESS) { if (Subtarget->hasAtomicFMinFMaxF32FlatInsts() && Ty->isFloatTy()) - return ReportUnsafeHWInst(AtomicExpansionKind::None); + return ReportHWInst(AtomicExpansionKind::None); if (Subtarget->hasAtomicFMinFMaxF64FlatInsts() && Ty->isDoubleTy()) - return ReportUnsafeHWInst(AtomicExpansionKind::None); + return ReportHWInst(AtomicExpansionKind::None); } else if (AMDGPU::isExtendedGlobalAddrSpace(AS) || AS == AMDGPUAS::BUFFER_FAT_POINTER) { if (Subtarget->hasAtomicFMinFMaxF32GlobalInsts() && Ty->isFloatTy()) - return ReportUnsafeHWInst(AtomicExpansionKind::None); + return ReportHWInst(AtomicExpansionKind::None); if (Subtarget->hasAtomicFMinFMaxF64GlobalInsts() && Ty->isDoubleTy()) - return ReportUnsafeHWInst(AtomicExpansionKind::None); + return ReportHWInst(AtomicExpansionKind::None); } } diff --git a/llvm/test/CodeGen/AMDGPU/atomics-hw-remarks-gfx908.ll b/llvm/test/CodeGen/AMDGPU/atomics-hw-remarks-gfx908.ll new file mode 100644 index 0000000000000..43623244bf6ef --- /dev/null +++ b/llvm/test/CodeGen/AMDGPU/atomics-hw-remarks-gfx908.ll @@ -0,0 +1,115 @@ +; RUN: llc -mtriple=amdgpu9.08 -pass-remarks='si-lower|atomic-expand' -filetype=null %s 2>&1 | \ +; RUN: FileCheck --implicit-check-not=remark: %s + +; No LDS f64 fadd instruction. +; CHECK: remark: <unknown>:0:0: A compare and swap loop was generated for an atomic fadd operation at system memory scope +define void @local_atomicrmw_fadd_f64__nortn(ptr addrspace(3) %ptr, double %val) #0 { + %ret = atomicrmw fadd ptr addrspace(3) %ptr, double %val seq_cst + ret void +} + +; CHECK: remark: <unknown>:0:0: A compare and swap loop was generated for an atomic fadd operation at agent memory scope +define void @global_atomicrmw_fadd_f32_agent__nortn(ptr addrspace(1) %ptr, float %val) #0 { + %ret = atomicrmw fadd ptr addrspace(1) %ptr, float %val syncscope("agent") monotonic, align 4 + ret void +} + +; CHECK: remark: <unknown>:0:0: A compare and swap loop was generated for an atomic fadd operation at agent memory scope +define float @global_atomicrmw_fadd_f32_agent__rtn(ptr addrspace(1) %ptr, float %val) #0 { + %ret = atomicrmw fadd ptr addrspace(1) %ptr, float %val syncscope("agent") monotonic, align 4 + ret float %ret +} + +; Denormals are not handled. +; CHECK: remark: <unknown>:0:0: A compare and swap loop was generated for an atomic fadd operation at agent memory scope +define void @global_atomicrmw_fadd_f32_agent__nortn__amdgpu_no_fine_grained_memory(ptr addrspace(1) %ptr, float %val) #0 { + %ret = atomicrmw fadd ptr addrspace(1) %ptr, float %val syncscope("agent") monotonic, align 4, !amdgpu.no.fine.grained.memory !0 + ret void +} + +; CHECK: remark: <unknown>:0:0: A compare and swap loop was generated for an atomic fadd operation at agent memory scope +define void @global_atomicrmw_fadd_f32_agent__nortn__amdgpu_no_remote_memory(ptr addrspace(1) %ptr, float %val) #0 { + %ret = atomicrmw fadd ptr addrspace(1) %ptr, float %val syncscope("agent") monotonic, align 4, !amdgpu.no.remote.memory !0 + ret void +} + +; CHECK: remark: <unknown>:0:0: hardware instruction generated for atomic fadd at agent scope since memory is not fine-grained (!amdgpu.no.fine.grained.memory), and denormals may be flushed (!atomic.ignore.denormal.mode) +define void @global_atomicrmw_fadd_f32_agent__nortn__amdgpu_no_fine_grained_memory__atomic_ignore_denormal_mode(ptr addrspace(1) %ptr, float %val) #0 { + %ret = atomicrmw fadd ptr addrspace(1) %ptr, float %val syncscope("agent") monotonic, align 4, !amdgpu.no.fine.grained.memory !0, !atomic.ignore.denormal.mode !0 + ret void +} + +; CHECK: remark: <unknown>:0:0: hardware instruction generated for atomic fadd at agent scope since memory is not fine-grained (!amdgpu.no.fine.grained.memory), and denormals may be flushed (!atomic.ignore.denormal.mode) +define void @global_atomicrmw_fadd_f32_agent__nortn__amdgpu_no_remote_memory__amdgpu_no_fine_grained_memory__atomic_ignore_denormal_mode(ptr addrspace(1) %ptr, float %val) #0 { + %ret = atomicrmw fadd ptr addrspace(1) %ptr, float %val syncscope("agent") monotonic, align 4, !amdgpu.no.remote.memory !0, !amdgpu.no.fine.grained.memory !0, !atomic.ignore.denormal.mode !0 + ret void +} + +; Remote memory is insufficient without agent scope fine-grained remote memory +; atomics. +; CHECK: remark: <unknown>:0:0: A compare and swap loop was generated for an atomic fadd operation at agent memory scope +define void @global_atomicrmw_fadd_f32_agent__nortn__amdgpu_no_remote_memory__atomic_ignore_denormal_mode(ptr addrspace(1) %ptr, float %val) #0 { + %ret = atomicrmw fadd ptr addrspace(1) %ptr, float %val syncscope("agent") monotonic, align 4, !amdgpu.no.remote.memory !0, !atomic.ignore.denormal.mode !0 + ret void +} + +; No f32 rtn instruction. +; CHECK: remark: <unknown>:0:0: A compare and swap loop was generated for an atomic fadd operation at agent memory scope +define float @global_atomicrmw_fadd_f32_agent__rtn__amdgpu_no_fine_grained_memory__atomic_ignore_denormal_mode(ptr addrspace(1) %ptr, float %val) #0 { + %ret = atomicrmw fadd ptr addrspace(1) %ptr, float %val syncscope("agent") monotonic, align 4, !amdgpu.no.fine.grained.memory !0, !atomic.ignore.denormal.mode !0 + ret float %ret +} + +; CHECK: remark: <unknown>:0:0: hardware instruction generated for atomic fadd at system scope since memory is not fine-grained (!amdgpu.no.fine.grained.memory), and denormals may be flushed (!atomic.ignore.denormal.mode) +define void @global_atomicrmw_fadd_f32_system__nortn__amdgpu_no_fine_grained_memory__atomic_ignore_denormal_mode(ptr addrspace(1) %ptr, float %val) #0 { + %ret = atomicrmw fadd ptr addrspace(1) %ptr, float %val monotonic, align 4, !amdgpu.no.fine.grained.memory !0, !atomic.ignore.denormal.mode !0 + ret void +} + +; CHECK: remark: <unknown>:0:0: hardware instruction generated for atomic fadd at agent scope since memory is not fine-grained (!amdgpu.no.fine.grained.memory), and the floating-point environment flushes denormals +define void @global_atomicrmw_fadd_f32_agent__nortn__amdgpu_no_fine_grained_memory__ftz(ptr addrspace(1) %ptr, float %val) #1 { + %ret = atomicrmw fadd ptr addrspace(1) %ptr, float %val syncscope("agent") monotonic, align 4, !amdgpu.no.fine.grained.memory !0 + ret void +} + +; CHECK: remark: <unknown>:0:0: A compare and swap loop was generated for an atomic fadd operation at agent memory scope +define void @global_atomicrmw_fadd_f32_agent__nortn__amdgpu_no_remote_memory__ftz(ptr addrspace(1) %ptr, float %val) #1 { + %ret = atomicrmw fadd ptr addrspace(1) %ptr, float %val syncscope("agent") monotonic, align 4, !amdgpu.no.remote.memory !0 + ret void +} + +; CHECK: remark: <unknown>:0:0: hardware instruction generated for atomic fadd at agent scope since memory is not fine-grained (!amdgpu.no.fine.grained.memory), and the floating-point environment flushes denormals +define void @global_atomicrmw_fadd_f32_agent__nortn__amdgpu_no_remote_memory__amdgpu_no_fine_grained_memory__ftz(ptr addrspace(1) %ptr, float %val) #1 { + %ret = atomicrmw fadd ptr addrspace(1) %ptr, float %val syncscope("agent") monotonic, align 4, !amdgpu.no.remote.memory !0, !amdgpu.no.fine.grained.memory !0 + ret void +} + +; CHECK: remark: <unknown>:0:0: hardware instruction generated for atomic fadd at agent scope since memory is not fine-grained (!amdgpu.no.fine.grained.memory) +define void @global_atomicrmw_fadd_v2f16_agent__nortn__amdgpu_no_fine_grained_memory(ptr addrspace(1) %ptr, <2 x half> %val) #0 { + %ret = atomicrmw fadd ptr addrspace(1) %ptr, <2 x half> %val syncscope("agent") monotonic, align 4, !amdgpu.no.fine.grained.memory !0 + ret void +} + +; CHECK: remark: <unknown>:0:0: A compare and swap loop was generated for an atomic fadd operation at agent memory scope +define void @global_atomicrmw_fadd_v2f16_agent__nortn__amdgpu_no_remote_memory(ptr addrspace(1) %ptr, <2 x half> %val) #0 { + %ret = atomicrmw fadd ptr addrspace(1) %ptr, <2 x half> %val syncscope("agent") monotonic, align 4, !amdgpu.no.remote.memory !0 + ret void +} + +; CHECK: remark: <unknown>:0:0: hardware instruction generated for atomic fadd at agent scope since memory is not fine-grained (!amdgpu.no.fine.grained.memory) +define void @global_atomicrmw_fadd_v2f16_agent__nortn__amdgpu_no_fine_grained_memory__amdgpu_no_remote_memory(ptr addrspace(1) %ptr, <2 x half> %val) #0 { + %ret = atomicrmw fadd ptr addrspace(1) %ptr, <2 x half> %val syncscope("agent") monotonic, align 4, !amdgpu.no.remote.memory !0, !amdgpu.no.fine.grained.memory !0 + ret void +} + +; No v2f16 rtn instruction. +; CHECK: remark: <unknown>:0:0: A compare and swap loop was generated for an atomic fadd operation at agent memory scope +define <2 x half> @global_atomicrmw_fadd_v2f16_agent__rtn__amdgpu_no_fine_grained_memory__amdgpu_no_remote_memory(ptr addrspace(1) %ptr, <2 x half> %val) #0 { + %ret = atomicrmw fadd ptr addrspace(1) %ptr, <2 x half> %val syncscope("agent") monotonic, align 4, !amdgpu.no.remote.memory !0, !amdgpu.no.fine.grained.memory !0 + ret <2 x half> %ret +} + +attributes #0 = { denormal_fpenv(ieee) } +attributes #1 = { denormal_fpenv(float: preservesign) } + +!0 = !{} diff --git a/llvm/test/CodeGen/AMDGPU/atomics-hw-remarks-gfx90a.ll b/llvm/test/CodeGen/AMDGPU/atomics-hw-remarks-gfx90a.ll index fd91f43ac3868..87d78d7be7162 100644 --- a/llvm/test/CodeGen/AMDGPU/atomics-hw-remarks-gfx90a.ll +++ b/llvm/test/CodeGen/AMDGPU/atomics-hw-remarks-gfx90a.ll @@ -1,14 +1,13 @@ -; RUN: llc -mtriple=amdgpu9.0a --pass-remarks=si-lower \ -; RUN: %s -o - 2>&1 | FileCheck %s --check-prefix=GFX90A-HW +; RUN: llc -mtriple=amdgpu9.0a --pass-remarks='si-lower|atomic-expand' %s -o - 2>&1 | FileCheck --check-prefix=GFX90A-HW %s -; GFX90A-HW: Hardware instruction generated for atomic fadd operation at memory scope agent due to an unsafe request. -; GFX90A-HW: Hardware instruction generated for atomic fadd operation at memory scope workgroup due to an unsafe request. -; GFX90A-HW: Hardware instruction generated for atomic fadd operation at memory scope wavefront due to an unsafe request. -; GFX90A-HW: Hardware instruction generated for atomic fadd operation at memory scope singlethread due to an unsafe request. -; GFX90A-HW: Hardware instruction generated for atomic fadd operation at memory scope agent-one-as due to an unsafe request. -; GFX90A-HW: Hardware instruction generated for atomic fadd operation at memory scope workgroup-one-as due to an unsafe request. -; GFX90A-HW: Hardware instruction generated for atomic fadd operation at memory scope wavefront-one-as due to an unsafe request. -; GFX90A-HW: Hardware instruction generated for atomic fadd operation at memory scope singlethread-one-as due to an unsafe request. +; GFX90A-HW: hardware instruction generated for atomic fadd at agent scope since memory is not fine-grained (!amdgpu.no.fine.grained.memory), and the floating-point environment flushes denormals +; GFX90A-HW: hardware instruction generated for atomic fadd at workgroup scope since memory is not fine-grained (!amdgpu.no.fine.grained.memory), and the floating-point environment flushes denormals +; GFX90A-HW: hardware instruction generated for atomic fadd at wavefront scope since memory is not fine-grained (!amdgpu.no.fine.grained.memory), and the floating-point environment flushes denormals +; GFX90A-HW: hardware instruction generated for atomic fadd at singlethread scope since memory is not fine-grained (!amdgpu.no.fine.grained.memory), and the floating-point environment flushes denormals +; GFX90A-HW: hardware instruction generated for atomic fadd at agent-one-as scope since memory is not fine-grained (!amdgpu.no.fine.grained.memory), and the floating-point environment flushes denormals +; GFX90A-HW: hardware instruction generated for atomic fadd at workgroup-one-as scope since memory is not fine-grained (!amdgpu.no.fine.grained.memory), and the floating-point environment flushes denormals +; GFX90A-HW: hardware instruction generated for atomic fadd at wavefront-one-as scope since memory is not fine-grained (!amdgpu.no.fine.grained.memory), and the floating-point environment flushes denormals +; GFX90A-HW: hardware instruction generated for atomic fadd at singlethread-one-as scope since memory is not fine-grained (!amdgpu.no.fine.grained.memory), and the floating-point environment flushes denormals ; GFX90A-HW-LABEL: atomic_add_unsafe_hw: ; GFX90A-HW: ds_add_f64 v0, v[2:3] diff --git a/llvm/test/CodeGen/AMDGPU/atomics-hw-remarks-scope.ll b/llvm/test/CodeGen/AMDGPU/atomics-hw-remarks-scope.ll new file mode 100644 index 0000000000000..ac8e9e0a5d78f --- /dev/null +++ b/llvm/test/CodeGen/AMDGPU/atomics-hw-remarks-scope.ll @@ -0,0 +1,41 @@ +; RUN: llc -mtriple=amdgpu9.42 -pass-remarks='si-lower|atomic-expand' -filetype=null %s 2>&1 | \ +; RUN: FileCheck %s --check-prefix=GFX942 --implicit-check-not=remark: +; RUN: llc -mtriple=amdgpu12.50 -pass-remarks='si-lower|atomic-expand' -filetype=null %s 2>&1 | \ +; RUN: FileCheck %s --check-prefix=GFX1250 --implicit-check-not=remark: + +; GFX942: remark: <unknown>:0:0: hardware instruction generated for atomic fadd at agent scope since fine-grained remote memory atomics work below system scope +; GFX1250: remark: <unknown>:0:0: hardware instruction generated for atomic fadd at agent scope since fine-grained remote memory atomics work below system scope +define void @global_atomicrmw_fadd_f32_agent(ptr addrspace(1) %ptr, float %val) { + %ret = atomicrmw fadd ptr addrspace(1) %ptr, float %val syncscope("agent") monotonic, align 4 + ret void +} + +; GFX942: remark: <unknown>:0:0: A compare and swap loop was generated for an atomic fadd operation at system memory scope +; GFX1250: remark: <unknown>:0:0: hardware instruction generated for atomic fadd at system scope since system scope atomics are emulated in hardware +define void @global_atomicrmw_fadd_f32_system(ptr addrspace(1) %ptr, float %val) { + %ret = atomicrmw fadd ptr addrspace(1) %ptr, float %val monotonic, align 4 + ret void +} + +; GFX942: remark: <unknown>:0:0: hardware instruction generated for atomic fadd at system scope since memory is not remote (!amdgpu.no.remote.memory) +; GFX1250: remark: <unknown>:0:0: hardware instruction generated for atomic fadd at system scope since memory is not remote (!amdgpu.no.remote.memory) +define void @global_atomicrmw_fadd_f32_system__amdgpu_no_remote_memory(ptr addrspace(1) %ptr, float %val) { + %ret = atomicrmw fadd ptr addrspace(1) %ptr, float %val monotonic, align 4, !amdgpu.no.remote.memory !0 + ret void +} + +; GFX942: remark: <unknown>:0:0: hardware instruction generated for atomic fadd at system scope since memory is not fine-grained (!amdgpu.no.fine.grained.memory) +; GFX1250: remark: <unknown>:0:0: hardware instruction generated for atomic fadd at system scope since system scope atomics are emulated in hardware +define void @global_atomicrmw_fadd_f32_system__amdgpu_no_fine_grained_memory(ptr addrspace(1) %ptr, float %val) { + %ret = atomicrmw fadd ptr addrspace(1) %ptr, float %val monotonic, align 4, !amdgpu.no.fine.grained.memory !0 + ret void +} + +; GFX942: remark: <unknown>:0:0: hardware instruction generated for atomic fmax at agent scope since fine-grained remote memory atomics work below system scope +; GFX1250: remark: <unknown>:0:0: hardware instruction generated for atomic fmax at agent scope since fine-grained remote memory atomics work below system scope +define void @global_atomicrmw_fmax_f64_agent(ptr addrspace(1) %ptr, double %val) { + %ret = atomicrmw fmax ptr addrspace(1) %ptr, double %val syncscope("agent") monotonic, align 8 + ret void +} + +!0 = !{} _______________________________________________ cfe-commits mailing list [email protected] https://lists.llvm.org/cgi-bin/mailman/listinfo/cfe-commits
