https://github.com/chinmaydd created https://github.com/llvm/llvm-project/pull/224852
Model LDS encoding granularity with dedicated features and expose the byte-valued `getLDSEncodingGranule` query for GPUKind and subarch. Migrate program resource register and PAL metadata encoding to the new query and remove `getLdsDwGranularity` from `AMDGPUBaseInfo` >From 47a076d2d3a0a93b14229299f4756e8dda633532 Mon Sep 17 00:00:00 2001 From: Chinmay Deshpande <[email protected]> Date: Sat, 19 Sep 2026 14:15:35 -0400 Subject: [PATCH] [AMDGPU] Add LDS encoding granularity to TargetParser Model LDS encoding granularity with dedicated features and expose the byte-valued getLDSEncodingGranule query for GPUKind and subarch. Keep encoding independent of the hardware allocation granularity used for occupancy; GFX10.3, GFX11 and GFX12.0 encode in 512-byte units while allocating 1024-byte blocks. Migrate program resource register and PAL metadata encoding to the new query and remove getLdsDwGranularity from AMDGPUBaseInfo. gfx9-4-generic uses gfx950's 1280-byte encoding granule independently of LDS capacity. Test encoding queries, feature membership, generic-target validation and encoded LDS sizes, including the GFX10.3 allocation/encoding distinction. Change-Id: I9d3c2c041605e9a45fa8fbda09fc3460a74953ea --- .../llvm/TargetParser/AMDGPUTargetParser.h | 5 ++ llvm/lib/Target/AMDGPU/AMDGPU.td | 20 +++++ llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp | 7 +- llvm/lib/Target/AMDGPU/AMDGPUFeatures.td | 15 ++++ llvm/lib/Target/AMDGPU/AMDGPUSubtarget.h | 1 + llvm/lib/Target/AMDGPU/R600Processors.td | 2 + .../Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp | 14 --- llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h | 5 -- llvm/lib/TargetParser/AMDGPUTargetParser.cpp | 35 +++++++- llvm/test/CodeGen/AMDGPU/lds-size-gfx1030.ll | 26 ++++++ .../CodeGen/AMDGPU/lds-size-gfx9-4-generic.ll | 34 +++++++ .../AMDGPUTargetDefLDSAllocGranularity.td | 42 +++++++-- .../TargetParser/TargetParserTest.cpp | 88 ++++++++++++------- 13 files changed, 235 insertions(+), 59 deletions(-) create mode 100644 llvm/test/CodeGen/AMDGPU/lds-size-gfx1030.ll create mode 100644 llvm/test/CodeGen/AMDGPU/lds-size-gfx9-4-generic.ll diff --git a/llvm/include/llvm/TargetParser/AMDGPUTargetParser.h b/llvm/include/llvm/TargetParser/AMDGPUTargetParser.h index 49245a4322ab6..1239baa200a73 100644 --- a/llvm/include/llvm/TargetParser/AMDGPUTargetParser.h +++ b/llvm/include/llvm/TargetParser/AMDGPUTargetParser.h @@ -245,6 +245,11 @@ LLVM_ABI unsigned getLDSBankCount(Triple::SubArchType SubArch); LLVM_ABI unsigned getLDSAllocGranule(GPUKind AK); LLVM_ABI unsigned getLDSAllocGranule(Triple::SubArchType SubArch); +/// \returns LDS size encoding granularity in bytes, used for program resource +/// registers and metadata. This can differ from the allocation granularity. +LLVM_ABI unsigned getLDSEncodingGranule(GPUKind AK); +LLVM_ABI unsigned getLDSEncodingGranule(Triple::SubArchType SubArch); + /// \returns Number of SIMDs a work-group's waves run on. All four SIMDs of the /// functional block in full-SIMD mode, half of them otherwise. constexpr unsigned getNumWorkGroupSIMDs(bool FullSIMDMode) { diff --git a/llvm/lib/Target/AMDGPU/AMDGPU.td b/llvm/lib/Target/AMDGPU/AMDGPU.td index 186191d76cb38..8ddbd8a44707a 100644 --- a/llvm/lib/Target/AMDGPU/AMDGPU.td +++ b/llvm/lib/Target/AMDGPU/AMDGPU.td @@ -1654,6 +1654,7 @@ def FeatureSouthernIslands : GCNSubtargetFeatureGeneration<"SOUTHERN_ISLANDS", [FeatureFP64, FeatureAddressableLocalMemorySize32768, FeatureHalfAddressablePhysicalLocalMemory, FeatureLDSAllocGranularity256, + FeatureLDSEncodingGranularity256, FeatureMIMG_R128, FeatureWavefrontSize64, FeatureSupportsWave64, FeatureSMemTimeInst, FeatureMadMacF32Insts, @@ -1674,6 +1675,7 @@ def FeatureSeaIslands : GCNSubtargetFeatureGeneration<"SEA_ISLANDS", "sea-islands", [FeatureFP64, FeatureAddressableLocalMemorySize65536, FeatureLDSAllocGranularity512, + FeatureLDSEncodingGranularity512, FeatureMIMG_R128, FeatureWavefrontSize64, FeatureSupportsWave64, FeatureFlatAddressSpace, FeatureCIInsts, FeatureMovrel, FeatureTrigReducedRange, @@ -1696,6 +1698,7 @@ def FeatureVolcanicIslands : GCNSubtargetFeatureGeneration<"VOLCANIC_ISLANDS", "volcanic-islands", [FeatureFP64, FeatureAddressableLocalMemorySize65536, FeatureLDSAllocGranularity512, + FeatureLDSEncodingGranularity512, FeatureMIMG_R128, FeatureWavefrontSize64, FeatureSupportsWave64, FeatureFlatAddressSpace, FeatureGCN3Encoding, FeatureCIInsts, Feature16BitInsts, @@ -1746,6 +1749,7 @@ def FeatureGFX9 : GCNSubtargetFeatureGeneration<"GFX9", def FeatureGFX10 : GCNSubtargetFeatureGeneration<"GFX10", "gfx10", [FeatureFP64, FeatureAddressableLocalMemorySize65536, + FeatureLDSEncodingGranularity512, FeatureHalfAddressablePhysicalLocalMemory, FeatureMIMG_R128, FeatureSupportsWave32, FeatureSupportsWave64, FeatureSupportsWGP, FeatureFlatAddressSpace, @@ -1781,6 +1785,7 @@ def FeatureGFX11 : GCNSubtargetFeatureGeneration<"GFX11", "gfx11", [FeatureFP64, FeatureAddressableLocalMemorySize65536, FeatureLDSAllocGranularity1024, + FeatureLDSEncodingGranularity512, FeatureHalfAddressablePhysicalLocalMemory, FeatureMIMG_R128, FeatureSupportsWave32, FeatureSupportsWave64, FeatureSupportsWGP, FeatureFlatAddressSpace, Feature16BitInsts, @@ -1951,6 +1956,7 @@ def FeatureISAVersion9_0_Common : FeatureSet< [FeatureGFX9, FeatureAddressableLocalMemorySize65536, FeatureLDSAllocGranularity512, + FeatureLDSEncodingGranularity512, FeatureLDSBankCount32, FeatureImageInsts, FeatureMadMacF32Insts]>; @@ -2101,6 +2107,7 @@ def FeatureISAVersion9_5_Common : FeatureSet< !listconcat(FeatureISAVersion9_4_Common.Features, [FeatureAddressableLocalMemorySize163840, FeatureLDSAllocGranularity1280, + FeatureLDSEncodingGranularity1280, FeatureLDSBankCount64, FeatureFP8Insts, FeatureFP8ConversionInsts, @@ -2124,6 +2131,7 @@ def FeatureISAVersion9_4_2 : FeatureSet< [ FeatureAddressableLocalMemorySize65536, FeatureLDSAllocGranularity512, + FeatureLDSEncodingGranularity512, FeatureLDSBankCount32, FeatureFP8Insts, FeatureFP8ConversionInsts, @@ -2136,6 +2144,7 @@ def FeatureISAVersion9_4_Generic : FeatureSet< !listconcat(FeatureISAVersion9_4_Common.Features, [FeatureAddressableLocalMemorySize65536, FeatureLDSAllocGranularity1280, + FeatureLDSEncodingGranularity1280, FeatureLDSBankCount32, FeatureRequiresCOV6])>; @@ -2348,6 +2357,7 @@ def FeatureISAVersion12 : FeatureSet< FeatureBackOffBarrier, FeatureAddressableLocalMemorySize65536, FeatureLDSAllocGranularity1024, + FeatureLDSEncodingGranularity512, FeatureHalfAddressablePhysicalLocalMemory, FeatureLDSBankCount32, FeatureDLInsts, @@ -2507,6 +2517,7 @@ def FeatureISAVersion12_50_STRICT : FeatureSet< [FeatureGFX1250_STRICT, FeatureAddressableLocalMemorySize327680, FeatureLDSAllocGranularity2048, + FeatureLDSEncodingGranularity2048, FeatureLDSBankCount64, FeatureWMMAN16Insts, FeatureVOP3PX2IncrementsVaVdstTwice, @@ -2538,6 +2549,7 @@ def FeatureISAVersion12_50 : FeatureSet< !listconcat(FeatureISAVersion12_50_Common.Features, [FeatureAddressableLocalMemorySize327680, FeatureLDSAllocGranularity2048, + FeatureLDSEncodingGranularity2048, FeatureLDSBankCount64, FeatureWMMAN16Insts, FeatureWMMAF4Insts, @@ -2570,6 +2582,7 @@ def FeatureISAVersion12_51 : FeatureSet< !listconcat(FeatureISAVersion12_50_Common.Features, [FeatureAddressableLocalMemorySize327680, FeatureLDSAllocGranularity2048, + FeatureLDSEncodingGranularity2048, FeatureLDSBankCount64, FeatureWMMAN16Insts, FeatureWMMAF4Insts, @@ -2610,6 +2623,7 @@ def FeatureISAVersion12_5_Generic: FeatureSet< !listconcat(FeatureISAVersion12_50_Common.Features, [FeatureAddressableLocalMemorySize327680, FeatureLDSAllocGranularity2048, + FeatureLDSEncodingGranularity2048, FeatureLDSBankCount64, FeatureBlock16ConversionScaleInsts, FeatureSetregVGPRMSBFixup, @@ -2629,6 +2643,7 @@ def FeatureISAVersion13 : FeatureSet< FeatureWaveMatchInsts, FeatureAddressableLocalMemorySize196608, FeatureLDSAllocGranularity1024, + FeatureLDSEncodingGranularity1024, Feature64BitLiterals, FeatureLDSBankCount32, FeatureDLInsts, @@ -3323,6 +3338,11 @@ def AMDGPUFrontendVisibleFeatures { FeatureLDSAllocGranularity1024, FeatureLDSAllocGranularity1280, FeatureLDSAllocGranularity2048, + FeatureLDSEncodingGranularity256, + FeatureLDSEncodingGranularity512, + FeatureLDSEncodingGranularity1024, + FeatureLDSEncodingGranularity1280, + FeatureLDSEncodingGranularity2048, ]; } diff --git a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp index 11c482c83f1e1..f25448a81f457 100644 --- a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp +++ b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp @@ -1440,7 +1440,8 @@ void AMDGPUAsmPrinter::getSIProgramInfo(SIProgramInfo &ProgInfo, ProgInfo.LDSSize = MFI->getLDSSize(); - unsigned LDSGranularityBytes = getLdsDwGranularity(STM) * 4; + unsigned LDSGranularityBytes = + AMDGPU::getLDSEncodingGranule(STM.getTargetID().getGPUKind()); ProgInfo.LDSBlocks = alignTo(ProgInfo.LDSSize, LDSGranularityBytes) / LDSGranularityBytes; @@ -1679,8 +1680,8 @@ static void EmitPALMetadataCommon(AMDGPUPALMetadata *MD, MD->updateHwStageMaximum( CC, ".lds_size", - (unsigned)(CurrentProgramInfo.LdsSize * getLdsDwGranularity(ST) * - sizeof(uint32_t))); + (unsigned)(CurrentProgramInfo.LdsSize * + AMDGPU::getLDSEncodingGranule(ST.getTargetID().getGPUKind()))); } // This is the equivalent of EmitProgramInfoSI above, but for when the OS type diff --git a/llvm/lib/Target/AMDGPU/AMDGPUFeatures.td b/llvm/lib/Target/AMDGPU/AMDGPUFeatures.td index 49c90acabb74b..cd3ac06f60fd1 100644 --- a/llvm/lib/Target/AMDGPU/AMDGPUFeatures.td +++ b/llvm/lib/Target/AMDGPU/AMDGPUFeatures.td @@ -60,6 +60,21 @@ def FeatureLDSAllocGranularity1024 : SubtargetFeatureLDSAllocGranularity<1024>; def FeatureLDSAllocGranularity1280 : SubtargetFeatureLDSAllocGranularity<1280>; def FeatureLDSAllocGranularity2048 : SubtargetFeatureLDSAllocGranularity<2048>; +// Units used to encode LDS size in program resource registers and metadata. +// These can differ from the hardware allocation granularity. A generic target's +// encoding granularity must be present on at least one covered GPU. +class SubtargetFeatureLDSEncodingGranularity <int Granularity> : AMDGPUGenericAnyFeature < + "lds-encoding-granularity-"#Granularity, + "LDSEncodingGranularity", + !cast<string>(Granularity), + "LDS encoding granularity in bytes.">; + +def FeatureLDSEncodingGranularity256 : SubtargetFeatureLDSEncodingGranularity<256>; +def FeatureLDSEncodingGranularity512 : SubtargetFeatureLDSEncodingGranularity<512>; +def FeatureLDSEncodingGranularity1024 : SubtargetFeatureLDSEncodingGranularity<1024>; +def FeatureLDSEncodingGranularity1280 : SubtargetFeatureLDSEncodingGranularity<1280>; +def FeatureLDSEncodingGranularity2048 : SubtargetFeatureLDSEncodingGranularity<2048>; + // Whether each wavefront size mode is available on the hardware, // independent of the active mode. def FeatureSupportsWave32 : SubtargetFeature<"supports-wave32", diff --git a/llvm/lib/Target/AMDGPU/AMDGPUSubtarget.h b/llvm/lib/Target/AMDGPU/AMDGPUSubtarget.h index e1f331229c463..d2ab61671c932 100644 --- a/llvm/lib/Target/AMDGPU/AMDGPUSubtarget.h +++ b/llvm/lib/Target/AMDGPU/AMDGPUSubtarget.h @@ -60,6 +60,7 @@ class AMDGPUSubtarget { unsigned LocalMemorySize = 0; unsigned AddressableLocalMemorySize = 0; unsigned LDSAllocationGranularity = 0; + unsigned LDSEncodingGranularity = 0; char WavefrontSizeLog2 = 0; unsigned FlatOffsetBitWidth = 0; diff --git a/llvm/lib/Target/AMDGPU/R600Processors.td b/llvm/lib/Target/AMDGPU/R600Processors.td index 04947b4c94b7c..a72c367fd74a2 100644 --- a/llvm/lib/Target/AMDGPU/R600Processors.td +++ b/llvm/lib/Target/AMDGPU/R600Processors.td @@ -61,6 +61,7 @@ def FeatureR700 : R600SubtargetFeatureGeneration<"R700", "r700", def FeatureEvergreen : R600SubtargetFeatureGeneration<"EVERGREEN", "evergreen", [FeatureFetchLimit16, FeatureAddressableLocalMemorySize32768, FeatureLDSAllocGranularity256, + FeatureLDSEncodingGranularity256, FeatureMadMacF32Insts] >; @@ -69,6 +70,7 @@ def FeatureNorthernIslands : R600SubtargetFeatureGeneration<"NORTHERN_ISLANDS", [FeatureFetchLimit16, FeatureWavefrontSize64, FeatureAddressableLocalMemorySize32768, FeatureLDSAllocGranularity256, + FeatureLDSEncodingGranularity256, FeatureMadMacF32Insts] >; diff --git a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp index 11b4dbc24c085..6200167a98354 100644 --- a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp +++ b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp @@ -3665,20 +3665,6 @@ bool isDPALU_DPP(const MCInstrDesc &OpDesc, const MCInstrInfo &MII, return hasAny64BitVGPROperands(OpDesc, MII, ST); } -unsigned getLdsDwGranularity(const MCSubtargetInfo &ST) { - if (ST.getFeatureBits().test(FeatureAddressableLocalMemorySize32768)) - return 64; - if (ST.getFeatureBits().test(FeatureAddressableLocalMemorySize65536)) - return 128; - if (ST.getFeatureBits().test(FeatureAddressableLocalMemorySize196608)) - return 256; - if (ST.getFeatureBits().test(FeatureAddressableLocalMemorySize163840)) - return 320; - if (ST.getFeatureBits().test(FeatureAddressableLocalMemorySize327680)) - return 512; - return 64; // In sync with getAddressableLocalMemorySize -} - bool isPackedSingleSGPRFP32Inst(unsigned Opc) { switch (Opc) { case AMDGPU::V_PK_ADD_F32_gfx1250: diff --git a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h index e124073f0466e..a9358689514bc 100644 --- a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h +++ b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h @@ -1812,11 +1812,6 @@ getVGPRLoweringOperandTables(const MCInstrDesc &Desc); /// \returns true if a memory instruction supports scale_offset modifier. bool supportsScaleOffset(const MCInstrInfo &MII, unsigned Opcode); -/// \returns lds block size in terms of dwords. \p -/// This is used to calculate the lds size encoded for PAL metadata 3.0+ which -/// must be defined in terms of bytes. -unsigned getLdsDwGranularity(const MCSubtargetInfo &ST); - class ClusterDimsAttr { public: enum class Kind { Unknown, NoCluster, VariableDims, FixedDims }; diff --git a/llvm/lib/TargetParser/AMDGPUTargetParser.cpp b/llvm/lib/TargetParser/AMDGPUTargetParser.cpp index 6cd174d0723a7..e81d0d964aebe 100644 --- a/llvm/lib/TargetParser/AMDGPUTargetParser.cpp +++ b/llvm/lib/TargetParser/AMDGPUTargetParser.cpp @@ -556,6 +556,34 @@ unsigned AMDGPU::getLDSAllocGranule(Triple::SubArchType SubArch) { return getLDSAllocGranule(getGPUKindFromSubArch(SubArch)); } +unsigned AMDGPU::getLDSEncodingGranule(GPUKind AK) { + const AMDGPUFeatureBitset &Features = getFeatureBitset(AK); + if (Features.none()) + return 256; + assert((Features.test(FEAT_LDS_ENCODING_GRANULARITY_256) || + Features.test(FEAT_LDS_ENCODING_GRANULARITY_512) || + Features.test(FEAT_LDS_ENCODING_GRANULARITY_1024) || + Features.test(FEAT_LDS_ENCODING_GRANULARITY_1280) || + Features.test(FEAT_LDS_ENCODING_GRANULARITY_2048)) && + "missing LDS encoding granularity feature"); + if (Features.test(FEAT_LDS_ENCODING_GRANULARITY_256)) + return 256; + if (Features.test(FEAT_LDS_ENCODING_GRANULARITY_512)) + return 512; + if (Features.test(FEAT_LDS_ENCODING_GRANULARITY_1024)) + return 1024; + if (Features.test(FEAT_LDS_ENCODING_GRANULARITY_1280)) + return 1280; + if (Features.test(FEAT_LDS_ENCODING_GRANULARITY_2048)) + return 2048; + + return 256; +} + +unsigned AMDGPU::getLDSEncodingGranule(Triple::SubArchType SubArch) { + return getLDSEncodingGranule(getGPUKindFromSubArch(SubArch)); +} + unsigned AMDGPU::getMaxWavesPerEU(GPUKind AK) { const GPUInfo *Info = getAMDGPUInfo(AK); return Info ? Info->MaxWavesPerEU : 10; @@ -597,7 +625,12 @@ static const AMDGPUFeatureBitset FrontendOnlyFeatures = { FEAT_LDS_ALLOC_GRANULARITY_512, FEAT_LDS_ALLOC_GRANULARITY_1024, FEAT_LDS_ALLOC_GRANULARITY_1280, - FEAT_LDS_ALLOC_GRANULARITY_2048}; + FEAT_LDS_ALLOC_GRANULARITY_2048, + FEAT_LDS_ENCODING_GRANULARITY_256, + FEAT_LDS_ENCODING_GRANULARITY_512, + FEAT_LDS_ENCODING_GRANULARITY_1024, + FEAT_LDS_ENCODING_GRANULARITY_1280, + FEAT_LDS_ENCODING_GRANULARITY_2048}; // Add a GPU's features (minus the frontend-only ones) to \p Features. With \p // Overwrite false, existing entries are kept so user -mattr overrides win. diff --git a/llvm/test/CodeGen/AMDGPU/lds-size-gfx1030.ll b/llvm/test/CodeGen/AMDGPU/lds-size-gfx1030.ll new file mode 100644 index 0000000000000..637add975dc9f --- /dev/null +++ b/llvm/test/CodeGen/AMDGPU/lds-size-gfx1030.ll @@ -0,0 +1,26 @@ +; RUN: llc -mtriple=amdgpu10.30-mesa-mesa3d < %s | FileCheck %s --check-prefixes=CHECK,MESA +; RUN: llc -mtriple=amdgpu10.30-amd-amdpal < %s | FileCheck %s --check-prefixes=CHECK,PAL + +; gfx1030 allocates LDS in 1024-byte blocks but encodes it in 512-byte units. +; 8252 bytes occupies 9216 bytes of LDS for occupancy, while its encoded size is +; 17 units (8704 bytes). Using the allocation granule for either encoding step +; would produce an incorrect block count or PAL metadata size. + +@lds = addrspace(3) global [2063 x i32] poison, align 4 + +; CHECK-LABEL: lds_granularity: +; MESA: granulated_lds_size = 17 +; CHECK: ; Occupancy: 7{{$}} +; PAL: .hardware_stages: +; PAL: .cs: +; PAL: .lds_size: 0x2200 +define amdgpu_kernel void @lds_granularity(i32 %index, i32 %value) #0 { + %ptr = getelementptr [2063 x i32], ptr addrspace(3) @lds, i32 0, i32 %index + store volatile i32 %value, ptr addrspace(3) %ptr + ret void +} + +attributes #0 = { "amdgpu-flat-work-group-size"="1,64" } + +!amdgpu.pal.metadata.msgpack = !{!0} +!0 = !{!"\81\AEamdpal.version\92\03\00"} diff --git a/llvm/test/CodeGen/AMDGPU/lds-size-gfx9-4-generic.ll b/llvm/test/CodeGen/AMDGPU/lds-size-gfx9-4-generic.ll new file mode 100644 index 0000000000000..c02e454ee10b3 --- /dev/null +++ b/llvm/test/CodeGen/AMDGPU/lds-size-gfx9-4-generic.ll @@ -0,0 +1,34 @@ +; RUN: llc -mtriple=amdgpu9.42-mesa-mesa3d < %s | FileCheck %s --check-prefix=GFX942 +; RUN: llc -mtriple=amdgpu9.50-mesa-mesa3d < %s | FileCheck %s --check-prefix=GFX950 +; RUN: llc -mtriple=amdgpu9.4-mesa-mesa3d < %s | FileCheck %s --check-prefix=GENERIC + +; gfx9-4-generic uses gfx950's 1280-byte allocation and encoding granules +; independently of its 64 KiB addressable LDS capacity. gfx942 uses 512 bytes +; for both granularities. + +@lds1280 = addrspace(3) global [320 x i32] poison, align 4 +@lds1284 = addrspace(3) global [321 x i32] poison, align 4 + +; GFX942-LABEL: one_granule: +; GFX942: granulated_lds_size = 3 +; GFX950-LABEL: one_granule: +; GFX950: granulated_lds_size = 1 +; GENERIC-LABEL: one_granule: +; GENERIC: granulated_lds_size = 1 +define amdgpu_kernel void @one_granule(i32 %index, i32 %value) { + %ptr = getelementptr [320 x i32], ptr addrspace(3) @lds1280, i32 0, i32 %index + store volatile i32 %value, ptr addrspace(3) %ptr + ret void +} + +; GFX942-LABEL: two_granules: +; GFX942: granulated_lds_size = 3 +; GFX950-LABEL: two_granules: +; GFX950: granulated_lds_size = 2 +; GENERIC-LABEL: two_granules: +; GENERIC: granulated_lds_size = 2 +define amdgpu_kernel void @two_granules(i32 %index, i32 %value) { + %ptr = getelementptr [321 x i32], ptr addrspace(3) @lds1284, i32 0, i32 %index + store volatile i32 %value, ptr addrspace(3) %ptr + ret void +} diff --git a/llvm/test/TableGen/AMDGPUTargetDefLDSAllocGranularity.td b/llvm/test/TableGen/AMDGPUTargetDefLDSAllocGranularity.td index 13136a7d00b59..b5e4337081a7d 100644 --- a/llvm/test/TableGen/AMDGPUTargetDefLDSAllocGranularity.td +++ b/llvm/test/TableGen/AMDGPUTargetDefLDSAllocGranularity.td @@ -3,6 +3,8 @@ // RUN: | FileCheck %t/valid.td // RUN: not llvm-tblgen -gen-amdgpu-target-def -I %t -I %p/../../include -I %p/../../lib/Target/AMDGPU %t/missing.td 2>&1 \ // RUN: | FileCheck %t/missing.td -DFILE=%t/missing.td --implicit-check-not="error:" +// RUN: not llvm-tblgen -gen-amdgpu-target-def -I %t -I %p/../../include -I %p/../../lib/Target/AMDGPU %t/missing-encoding.td 2>&1 \ +// RUN: | FileCheck %t/missing-encoding.td -DFILE=%t/missing-encoding.td --implicit-check-not="error:" //--- common.td include "llvm/Target/Target.td" @@ -12,25 +14,42 @@ def MyTarget : Target; def AMDGPUFrontendVisibleFeatures { list<SubtargetFeature> Features = [FeatureLDSAllocGranularity512, FeatureLDSAllocGranularity1024, - FeatureLDSAllocGranularity1280]; + FeatureLDSAllocGranularity1280, + FeatureLDSEncodingGranularity512, + FeatureLDSEncodingGranularity1024, + FeatureLDSEncodingGranularity1280]; } def GFX942 : AMDGPUProcessorModel<"gfx942", NoSchedModel, - [FeatureAddressableLocalMemorySize65536, FeatureLDSAllocGranularity512], + [FeatureAddressableLocalMemorySize65536, FeatureLDSAllocGranularity512, + FeatureLDSEncodingGranularity512], [9, 4, 2]>; def GFX950 : AMDGPUProcessorModel<"gfx950", NoSchedModel, - [FeatureAddressableLocalMemorySize163840, FeatureLDSAllocGranularity1280], + [FeatureAddressableLocalMemorySize163840, FeatureLDSAllocGranularity1280, + FeatureLDSEncodingGranularity1280], [9, 5, 0]>; //--- valid.td include "common.td" // Capacity and granularity can each come from a different covered GPU. def : AMDGPUProcessorModel<"gfx9-4-generic", NoSchedModel, - [FeatureAddressableLocalMemorySize65536, FeatureLDSAllocGranularity1280], + [FeatureAddressableLocalMemorySize65536, FeatureLDSAllocGranularity1280, + FeatureLDSEncodingGranularity1280], [9, 4, 0]> { let CoveredGPUs = [GFX942, GFX950]; } -// CHECK: Triple::AMDGPUSubArch9_4, AMDGPUFeatureBitset({FEAT_LDS_ALLOC_GRANULARITY_1280}), {9, 4, 0}, [[#]], 10, 65536, 32, 0}, +// CHECK: Triple::AMDGPUSubArch9_4, AMDGPUFeatureBitset({FEAT_LDS_ALLOC_GRANULARITY_1280, FEAT_LDS_ENCODING_GRANULARITY_1280}), {9, 4, 0}, [[#]], 10, 65536, 32, 0}, + +// Allocation and encoding use different units on gfx10.3, including its generic. +def GFX1030 : AMDGPUProcessorModel<"gfx1030", NoSchedModel, + [FeatureLDSAllocGranularity1024, FeatureLDSEncodingGranularity512], + [10, 3, 0]>; +def : AMDGPUProcessorModel<"gfx10-3-generic", NoSchedModel, + [FeatureLDSAllocGranularity1024, FeatureLDSEncodingGranularity512], + [10, 3, 0]> { + let CoveredGPUs = [GFX1030]; +} +// CHECK: Triple::AMDGPUSubArch10_3, AMDGPUFeatureBitset({FEAT_LDS_ALLOC_GRANULARITY_1024, FEAT_LDS_ENCODING_GRANULARITY_512}) //--- missing.td include "common.td" @@ -40,3 +59,16 @@ def : AMDGPUProcessorModel<"gfx9-4-generic", NoSchedModel, [9, 4, 0]> { let CoveredGPUs = [GFX942, GFX950]; } + +//--- missing-encoding.td +include "common.td" +// An allocation feature does not provide support for an encoding feature. +def GFX1030 : AMDGPUProcessorModel<"gfx1030", NoSchedModel, + [FeatureLDSAllocGranularity1024, FeatureLDSEncodingGranularity512], + [10, 3, 0]>; +// CHECK: [[FILE]]:[[#@LINE+1]]:1: error: generic target 'gfx10-3-generic' exposes feature 'lds-encoding-granularity-1024' not supported by any covered GPU +def : AMDGPUProcessorModel<"gfx10-3-generic", NoSchedModel, + [FeatureLDSAllocGranularity1024, FeatureLDSEncodingGranularity1024], + [10, 3, 0]> { + let CoveredGPUs = [GFX1030]; +} diff --git a/llvm/unittests/TargetParser/TargetParserTest.cpp b/llvm/unittests/TargetParser/TargetParserTest.cpp index 493e0a6215f4b..d381e8e949f74 100644 --- a/llvm/unittests/TargetParser/TargetParserTest.cpp +++ b/llvm/unittests/TargetParser/TargetParserTest.cpp @@ -2833,6 +2833,13 @@ TEST(TargetParserTest, testAMDGPUfillAMDGPUFeatureMap) { EXPECT_FALSE(HasFeature("gfx950", "lds-alloc-granularity-1280")); EXPECT_FALSE(HasFeature("gfx1310", "lds-alloc-granularity-1024")); EXPECT_FALSE(HasFeature("gfx1250", "lds-alloc-granularity-2048")); + + // Encoding granularity is also queried through the bitset only. + EXPECT_FALSE(HasFeature("gfx600", "lds-encoding-granularity-256")); + EXPECT_FALSE(HasFeature("gfx1030", "lds-encoding-granularity-512")); + EXPECT_FALSE(HasFeature("gfx950", "lds-encoding-granularity-1280")); + EXPECT_FALSE(HasFeature("gfx1310", "lds-encoding-granularity-1024")); + EXPECT_FALSE(HasFeature("gfx1250", "lds-encoding-granularity-2048")); } TEST(TargetParserTest, testAMDGPUgetFeatureBitset) { @@ -2887,30 +2894,41 @@ TEST(TargetParserTest, testAMDGPUHalfAddressableLDSFeature) { EXPECT_FALSE(Has(AMDGPU::GK_GFX1310)); } -TEST(TargetParserTest, testAMDGPULDSAllocGranularityFeatures) { +TEST(TargetParserTest, testAMDGPULDSGranularityFeatures) { auto Has = [](AMDGPU::GPUKind AK, AMDGPU::AMDGPUFeature Feature) { return AMDGPU::getFeatureBitset(AK).test(Feature); }; - auto Count = [&Has](AMDGPU::GPUKind AK) { + auto CountAlloc = [&Has](AMDGPU::GPUKind AK) { return Has(AK, AMDGPU::FEAT_LDS_ALLOC_GRANULARITY_256) + Has(AK, AMDGPU::FEAT_LDS_ALLOC_GRANULARITY_512) + Has(AK, AMDGPU::FEAT_LDS_ALLOC_GRANULARITY_1024) + Has(AK, AMDGPU::FEAT_LDS_ALLOC_GRANULARITY_1280) + Has(AK, AMDGPU::FEAT_LDS_ALLOC_GRANULARITY_2048); }; + auto CountEncoding = [&Has](AMDGPU::GPUKind AK) { + return Has(AK, AMDGPU::FEAT_LDS_ENCODING_GRANULARITY_256) + + Has(AK, AMDGPU::FEAT_LDS_ENCODING_GRANULARITY_512) + + Has(AK, AMDGPU::FEAT_LDS_ENCODING_GRANULARITY_1024) + + Has(AK, AMDGPU::FEAT_LDS_ENCODING_GRANULARITY_1280) + + Has(AK, AMDGPU::FEAT_LDS_ENCODING_GRANULARITY_2048); + }; - // Exactly one allocation granularity is set per GPU. + // Exactly one granularity of each kind is set per GPU, including generics. SmallVector<StringRef> AllGPUs; AMDGPU::fillValidArchListAMDGCN(AllGPUs, Triple::NoSubArch); for (StringRef Name : AllGPUs) { AMDGPU::GPUKind Kind = AMDGPU::parseArchAMDGCN(Name); - if (!AMDGPU::isPseudoTarget(Kind)) - EXPECT_EQ(Count(Kind), 1) << Name; + if (!AMDGPU::isPseudoTarget(Kind)) { + EXPECT_EQ(CountAlloc(Kind), 1) << Name; + EXPECT_EQ(CountEncoding(Kind), 1) << Name; + } } // The legacy pseudo-targets do not represent hardware. - EXPECT_EQ(Count(AMDGPU::GK_GENERIC), 0); - EXPECT_EQ(Count(AMDGPU::GK_GENERIC_HSA), 0); + EXPECT_EQ(CountAlloc(AMDGPU::GK_GENERIC), 0); + EXPECT_EQ(CountAlloc(AMDGPU::GK_GENERIC_HSA), 0); + EXPECT_EQ(CountEncoding(AMDGPU::GK_GENERIC), 0); + EXPECT_EQ(CountEncoding(AMDGPU::GK_GENERIC_HSA), 0); EXPECT_TRUE(Has(AMDGPU::GK_GFX600, AMDGPU::FEAT_LDS_ALLOC_GRANULARITY_256)); EXPECT_TRUE(Has(AMDGPU::GK_GFX900, AMDGPU::FEAT_LDS_ALLOC_GRANULARITY_512)); @@ -2918,13 +2936,15 @@ TEST(TargetParserTest, testAMDGPULDSAllocGranularityFeatures) { EXPECT_TRUE(Has(AMDGPU::GK_GFX1310, AMDGPU::FEAT_LDS_ALLOC_GRANULARITY_1024)); EXPECT_TRUE(Has(AMDGPU::GK_GFX1250, AMDGPU::FEAT_LDS_ALLOC_GRANULARITY_2048)); - // RDNA2 and later 64 KiB targets allocate LDS in 1024-byte blocks. + // RDNA2 and later 64 KiB targets allocate 1024 bytes but encode 512-byte + // units. for (AMDGPU::GPUKind Kind : {AMDGPU::GK_GFX1030, AMDGPU::GK_GFX1100, AMDGPU::GK_GFX1170, AMDGPU::GK_GFX1200, AMDGPU::GK_GFX10_3_GENERIC, AMDGPU::GK_GFX11_GENERIC, AMDGPU::GK_GFX11_7_GENERIC, AMDGPU::GK_GFX12_GENERIC}) { EXPECT_TRUE(Has(Kind, AMDGPU::FEAT_LDS_ALLOC_GRANULARITY_1024)); + EXPECT_TRUE(Has(Kind, AMDGPU::FEAT_LDS_ENCODING_GRANULARITY_512)); } // A generic target uses the largest allocation granularity of the GPUs it @@ -2933,6 +2953,8 @@ TEST(TargetParserTest, testAMDGPULDSAllocGranularityFeatures) { Has(AMDGPU::GK_GFX9_4_GENERIC, AMDGPU::FEAT_LDS_ALLOC_GRANULARITY_1280)); EXPECT_FALSE( Has(AMDGPU::GK_GFX9_4_GENERIC, AMDGPU::FEAT_LDS_ALLOC_GRANULARITY_512)); + EXPECT_TRUE(Has(AMDGPU::GK_GFX9_4_GENERIC, + AMDGPU::FEAT_LDS_ENCODING_GRANULARITY_1280)); } TEST(TargetParserTest, testAMDGPUfillValidArchListAMDGCN) { @@ -3390,43 +3412,47 @@ TEST(TargetParserTest, testAMDGPUgetAddressableLocalMemorySize) { 65536u); } -TEST(TargetParserTest, testAMDGPUgetLDSAllocGranule) { +TEST(TargetParserTest, testAMDGPUgetLDSGranules) { struct { AMDGPU::GPUKind Kind; Triple::SubArchType SubArch; unsigned Alloc; + unsigned Encoding; } Cases[] = { - {AMDGPU::GK_GFX600, Triple::AMDGPUSubArch600, 256}, - {AMDGPU::GK_GFX700, Triple::AMDGPUSubArch700, 512}, - {AMDGPU::GK_GFX900, Triple::AMDGPUSubArch900, 512}, - {AMDGPU::GK_GFX942, Triple::AMDGPUSubArch942, 512}, - {AMDGPU::GK_GFX950, Triple::AMDGPUSubArch950, 1280}, - {AMDGPU::GK_GFX1010, Triple::AMDGPUSubArch1010, 512}, - {AMDGPU::GK_GFX1030, Triple::AMDGPUSubArch1030, 1024}, - {AMDGPU::GK_GFX1100, Triple::AMDGPUSubArch1100, 1024}, - {AMDGPU::GK_GFX1150, Triple::AMDGPUSubArch1150, 1024}, - {AMDGPU::GK_GFX1170, Triple::AMDGPUSubArch1170, 1024}, - {AMDGPU::GK_GFX1200, Triple::AMDGPUSubArch1200, 1024}, - {AMDGPU::GK_GFX1250, Triple::AMDGPUSubArch1250, 2048}, - {AMDGPU::GK_GFX1251, Triple::AMDGPUSubArch1251, 2048}, - {AMDGPU::GK_GFX1310, Triple::AMDGPUSubArch1310, 1024}, - {AMDGPU::GK_GFX9_4_GENERIC, Triple::AMDGPUSubArch9_4, 1280}, - {AMDGPU::GK_GFX10_1_GENERIC, Triple::AMDGPUSubArch10_1, 512}, - {AMDGPU::GK_GFX10_3_GENERIC, Triple::AMDGPUSubArch10_3, 1024}, - {AMDGPU::GK_GFX11_GENERIC, Triple::AMDGPUSubArch11, 1024}, - {AMDGPU::GK_GFX11_7_GENERIC, Triple::AMDGPUSubArch11_7, 1024}, - {AMDGPU::GK_GFX12_GENERIC, Triple::AMDGPUSubArch12, 1024}, - {AMDGPU::GK_GFX12_5_GENERIC, Triple::AMDGPUSubArch12_5, 2048}, - {AMDGPU::GK_NONE, Triple::NoSubArch, 256}, + {AMDGPU::GK_GFX600, Triple::AMDGPUSubArch600, 256, 256}, + {AMDGPU::GK_GFX700, Triple::AMDGPUSubArch700, 512, 512}, + {AMDGPU::GK_GFX900, Triple::AMDGPUSubArch900, 512, 512}, + {AMDGPU::GK_GFX942, Triple::AMDGPUSubArch942, 512, 512}, + {AMDGPU::GK_GFX950, Triple::AMDGPUSubArch950, 1280, 1280}, + {AMDGPU::GK_GFX1010, Triple::AMDGPUSubArch1010, 512, 512}, + {AMDGPU::GK_GFX1030, Triple::AMDGPUSubArch1030, 1024, 512}, + {AMDGPU::GK_GFX1100, Triple::AMDGPUSubArch1100, 1024, 512}, + {AMDGPU::GK_GFX1150, Triple::AMDGPUSubArch1150, 1024, 512}, + {AMDGPU::GK_GFX1170, Triple::AMDGPUSubArch1170, 1024, 512}, + {AMDGPU::GK_GFX1200, Triple::AMDGPUSubArch1200, 1024, 512}, + {AMDGPU::GK_GFX1250, Triple::AMDGPUSubArch1250, 2048, 2048}, + {AMDGPU::GK_GFX1251, Triple::AMDGPUSubArch1251, 2048, 2048}, + {AMDGPU::GK_GFX1310, Triple::AMDGPUSubArch1310, 1024, 1024}, + {AMDGPU::GK_GFX9_4_GENERIC, Triple::AMDGPUSubArch9_4, 1280, 1280}, + {AMDGPU::GK_GFX10_1_GENERIC, Triple::AMDGPUSubArch10_1, 512, 512}, + {AMDGPU::GK_GFX10_3_GENERIC, Triple::AMDGPUSubArch10_3, 1024, 512}, + {AMDGPU::GK_GFX11_GENERIC, Triple::AMDGPUSubArch11, 1024, 512}, + {AMDGPU::GK_GFX11_7_GENERIC, Triple::AMDGPUSubArch11_7, 1024, 512}, + {AMDGPU::GK_GFX12_GENERIC, Triple::AMDGPUSubArch12, 1024, 512}, + {AMDGPU::GK_GFX12_5_GENERIC, Triple::AMDGPUSubArch12_5, 2048, 2048}, + {AMDGPU::GK_NONE, Triple::NoSubArch, 256, 256}, }; for (const auto &Case : Cases) { SCOPED_TRACE(AMDGPU::getArchNameAMDGCN(Case.Kind)); EXPECT_EQ(AMDGPU::getLDSAllocGranule(Case.Kind), Case.Alloc); EXPECT_EQ(AMDGPU::getLDSAllocGranule(Case.SubArch), Case.Alloc); + EXPECT_EQ(AMDGPU::getLDSEncodingGranule(Case.Kind), Case.Encoding); + EXPECT_EQ(AMDGPU::getLDSEncodingGranule(Case.SubArch), Case.Encoding); } for (AMDGPU::GPUKind Kind : {AMDGPU::GK_GENERIC, AMDGPU::GK_GENERIC_HSA}) { EXPECT_EQ(AMDGPU::getLDSAllocGranule(Kind), 256u); + EXPECT_EQ(AMDGPU::getLDSEncodingGranule(Kind), 256u); } } _______________________________________________ llvm-branch-commits mailing list [email protected] https://lists.llvm.org/cgi-bin/mailman/listinfo/llvm-branch-commits
