https://github.com/chinmaydd created 
https://github.com/llvm/llvm-project/pull/224852

Model LDS encoding granularity with dedicated features and expose the 
byte-valued `getLDSEncodingGranule` query for GPUKind and subarch.

Migrate program resource register and PAL metadata encoding to the new query 
and remove `getLdsDwGranularity` from `AMDGPUBaseInfo`

>From 47a076d2d3a0a93b14229299f4756e8dda633532 Mon Sep 17 00:00:00 2001
From: Chinmay Deshpande <[email protected]>
Date: Sat, 19 Sep 2026 14:15:35 -0400
Subject: [PATCH] [AMDGPU] Add LDS encoding granularity to TargetParser

Model LDS encoding granularity with dedicated features and expose the
byte-valued getLDSEncodingGranule query for GPUKind and subarch. Keep
encoding independent of the hardware allocation granularity used for
occupancy; GFX10.3, GFX11 and GFX12.0 encode in 512-byte units while
allocating 1024-byte blocks.

Migrate program resource register and PAL metadata encoding to the new
query and remove getLdsDwGranularity from AMDGPUBaseInfo. gfx9-4-generic
uses gfx950's 1280-byte encoding granule independently of LDS capacity.

Test encoding queries, feature membership, generic-target validation and
encoded LDS sizes, including the GFX10.3 allocation/encoding distinction.

Change-Id: I9d3c2c041605e9a45fa8fbda09fc3460a74953ea
---
 .../llvm/TargetParser/AMDGPUTargetParser.h    |  5 ++
 llvm/lib/Target/AMDGPU/AMDGPU.td              | 20 +++++
 llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp   |  7 +-
 llvm/lib/Target/AMDGPU/AMDGPUFeatures.td      | 15 ++++
 llvm/lib/Target/AMDGPU/AMDGPUSubtarget.h      |  1 +
 llvm/lib/Target/AMDGPU/R600Processors.td      |  2 +
 .../Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp    | 14 ---
 llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h |  5 --
 llvm/lib/TargetParser/AMDGPUTargetParser.cpp  | 35 +++++++-
 llvm/test/CodeGen/AMDGPU/lds-size-gfx1030.ll  | 26 ++++++
 .../CodeGen/AMDGPU/lds-size-gfx9-4-generic.ll | 34 +++++++
 .../AMDGPUTargetDefLDSAllocGranularity.td     | 42 +++++++--
 .../TargetParser/TargetParserTest.cpp         | 88 ++++++++++++-------
 13 files changed, 235 insertions(+), 59 deletions(-)
 create mode 100644 llvm/test/CodeGen/AMDGPU/lds-size-gfx1030.ll
 create mode 100644 llvm/test/CodeGen/AMDGPU/lds-size-gfx9-4-generic.ll

diff --git a/llvm/include/llvm/TargetParser/AMDGPUTargetParser.h 
b/llvm/include/llvm/TargetParser/AMDGPUTargetParser.h
index 49245a4322ab6..1239baa200a73 100644
--- a/llvm/include/llvm/TargetParser/AMDGPUTargetParser.h
+++ b/llvm/include/llvm/TargetParser/AMDGPUTargetParser.h
@@ -245,6 +245,11 @@ LLVM_ABI unsigned getLDSBankCount(Triple::SubArchType 
SubArch);
 LLVM_ABI unsigned getLDSAllocGranule(GPUKind AK);
 LLVM_ABI unsigned getLDSAllocGranule(Triple::SubArchType SubArch);
 
+/// \returns LDS size encoding granularity in bytes, used for program resource
+/// registers and metadata. This can differ from the allocation granularity.
+LLVM_ABI unsigned getLDSEncodingGranule(GPUKind AK);
+LLVM_ABI unsigned getLDSEncodingGranule(Triple::SubArchType SubArch);
+
 /// \returns Number of SIMDs a work-group's waves run on. All four SIMDs of the
 /// functional block in full-SIMD mode, half of them otherwise.
 constexpr unsigned getNumWorkGroupSIMDs(bool FullSIMDMode) {
diff --git a/llvm/lib/Target/AMDGPU/AMDGPU.td b/llvm/lib/Target/AMDGPU/AMDGPU.td
index 186191d76cb38..8ddbd8a44707a 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPU.td
+++ b/llvm/lib/Target/AMDGPU/AMDGPU.td
@@ -1654,6 +1654,7 @@ def FeatureSouthernIslands : 
GCNSubtargetFeatureGeneration<"SOUTHERN_ISLANDS",
   [FeatureFP64, FeatureAddressableLocalMemorySize32768,
   FeatureHalfAddressablePhysicalLocalMemory,
   FeatureLDSAllocGranularity256,
+  FeatureLDSEncodingGranularity256,
   FeatureMIMG_R128,
   FeatureWavefrontSize64, FeatureSupportsWave64, FeatureSMemTimeInst,
   FeatureMadMacF32Insts,
@@ -1674,6 +1675,7 @@ def FeatureSeaIslands : 
GCNSubtargetFeatureGeneration<"SEA_ISLANDS",
     "sea-islands",
   [FeatureFP64, FeatureAddressableLocalMemorySize65536,
   FeatureLDSAllocGranularity512,
+  FeatureLDSEncodingGranularity512,
   FeatureMIMG_R128,
   FeatureWavefrontSize64, FeatureSupportsWave64, FeatureFlatAddressSpace,
   FeatureCIInsts, FeatureMovrel, FeatureTrigReducedRange,
@@ -1696,6 +1698,7 @@ def FeatureVolcanicIslands : 
GCNSubtargetFeatureGeneration<"VOLCANIC_ISLANDS",
   "volcanic-islands",
   [FeatureFP64, FeatureAddressableLocalMemorySize65536,
    FeatureLDSAllocGranularity512,
+   FeatureLDSEncodingGranularity512,
    FeatureMIMG_R128,
    FeatureWavefrontSize64, FeatureSupportsWave64, FeatureFlatAddressSpace,
    FeatureGCN3Encoding, FeatureCIInsts, Feature16BitInsts,
@@ -1746,6 +1749,7 @@ def FeatureGFX9 : GCNSubtargetFeatureGeneration<"GFX9",
 def FeatureGFX10 : GCNSubtargetFeatureGeneration<"GFX10",
   "gfx10",
   [FeatureFP64, FeatureAddressableLocalMemorySize65536,
+   FeatureLDSEncodingGranularity512,
    FeatureHalfAddressablePhysicalLocalMemory, FeatureMIMG_R128,
    FeatureSupportsWave32, FeatureSupportsWave64, FeatureSupportsWGP,
    FeatureFlatAddressSpace,
@@ -1781,6 +1785,7 @@ def FeatureGFX11 : GCNSubtargetFeatureGeneration<"GFX11",
   "gfx11",
   [FeatureFP64, FeatureAddressableLocalMemorySize65536,
    FeatureLDSAllocGranularity1024,
+   FeatureLDSEncodingGranularity512,
    FeatureHalfAddressablePhysicalLocalMemory, FeatureMIMG_R128,
    FeatureSupportsWave32, FeatureSupportsWave64, FeatureSupportsWGP,
    FeatureFlatAddressSpace, Feature16BitInsts,
@@ -1951,6 +1956,7 @@ def FeatureISAVersion9_0_Common : FeatureSet<
   [FeatureGFX9,
    FeatureAddressableLocalMemorySize65536,
    FeatureLDSAllocGranularity512,
+   FeatureLDSEncodingGranularity512,
    FeatureLDSBankCount32,
    FeatureImageInsts,
    FeatureMadMacF32Insts]>;
@@ -2101,6 +2107,7 @@ def FeatureISAVersion9_5_Common : FeatureSet<
   !listconcat(FeatureISAVersion9_4_Common.Features,
   [FeatureAddressableLocalMemorySize163840,
    FeatureLDSAllocGranularity1280,
+   FeatureLDSEncodingGranularity1280,
    FeatureLDSBankCount64,
    FeatureFP8Insts,
    FeatureFP8ConversionInsts,
@@ -2124,6 +2131,7 @@ def FeatureISAVersion9_4_2 : FeatureSet<
     [
       FeatureAddressableLocalMemorySize65536,
       FeatureLDSAllocGranularity512,
+      FeatureLDSEncodingGranularity512,
       FeatureLDSBankCount32,
       FeatureFP8Insts,
       FeatureFP8ConversionInsts,
@@ -2136,6 +2144,7 @@ def FeatureISAVersion9_4_Generic : FeatureSet<
   !listconcat(FeatureISAVersion9_4_Common.Features,
     [FeatureAddressableLocalMemorySize65536,
      FeatureLDSAllocGranularity1280,
+     FeatureLDSEncodingGranularity1280,
      FeatureLDSBankCount32,
      FeatureRequiresCOV6])>;
 
@@ -2348,6 +2357,7 @@ def FeatureISAVersion12 : FeatureSet<
    FeatureBackOffBarrier,
    FeatureAddressableLocalMemorySize65536,
    FeatureLDSAllocGranularity1024,
+   FeatureLDSEncodingGranularity512,
    FeatureHalfAddressablePhysicalLocalMemory,
    FeatureLDSBankCount32,
    FeatureDLInsts,
@@ -2507,6 +2517,7 @@ def FeatureISAVersion12_50_STRICT : FeatureSet<
   [FeatureGFX1250_STRICT,
    FeatureAddressableLocalMemorySize327680,
    FeatureLDSAllocGranularity2048,
+   FeatureLDSEncodingGranularity2048,
    FeatureLDSBankCount64,
    FeatureWMMAN16Insts,
    FeatureVOP3PX2IncrementsVaVdstTwice,
@@ -2538,6 +2549,7 @@ def FeatureISAVersion12_50 : FeatureSet<
   !listconcat(FeatureISAVersion12_50_Common.Features,
   [FeatureAddressableLocalMemorySize327680,
    FeatureLDSAllocGranularity2048,
+   FeatureLDSEncodingGranularity2048,
    FeatureLDSBankCount64,
    FeatureWMMAN16Insts,
    FeatureWMMAF4Insts,
@@ -2570,6 +2582,7 @@ def FeatureISAVersion12_51 : FeatureSet<
   !listconcat(FeatureISAVersion12_50_Common.Features,
   [FeatureAddressableLocalMemorySize327680,
    FeatureLDSAllocGranularity2048,
+   FeatureLDSEncodingGranularity2048,
    FeatureLDSBankCount64,
    FeatureWMMAN16Insts,
    FeatureWMMAF4Insts,
@@ -2610,6 +2623,7 @@ def FeatureISAVersion12_5_Generic: FeatureSet<
   !listconcat(FeatureISAVersion12_50_Common.Features,
   [FeatureAddressableLocalMemorySize327680,
    FeatureLDSAllocGranularity2048,
+   FeatureLDSEncodingGranularity2048,
    FeatureLDSBankCount64,
    FeatureBlock16ConversionScaleInsts,
    FeatureSetregVGPRMSBFixup,
@@ -2629,6 +2643,7 @@ def FeatureISAVersion13 : FeatureSet<
    FeatureWaveMatchInsts,
    FeatureAddressableLocalMemorySize196608,
    FeatureLDSAllocGranularity1024,
+   FeatureLDSEncodingGranularity1024,
    Feature64BitLiterals,
    FeatureLDSBankCount32,
    FeatureDLInsts,
@@ -3323,6 +3338,11 @@ def AMDGPUFrontendVisibleFeatures {
   FeatureLDSAllocGranularity1024,
   FeatureLDSAllocGranularity1280,
   FeatureLDSAllocGranularity2048,
+  FeatureLDSEncodingGranularity256,
+  FeatureLDSEncodingGranularity512,
+  FeatureLDSEncodingGranularity1024,
+  FeatureLDSEncodingGranularity1280,
+  FeatureLDSEncodingGranularity2048,
   ];
 }
 
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp 
b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
index 11c482c83f1e1..f25448a81f457 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
@@ -1440,7 +1440,8 @@ void AMDGPUAsmPrinter::getSIProgramInfo(SIProgramInfo 
&ProgInfo,
 
   ProgInfo.LDSSize = MFI->getLDSSize();
 
-  unsigned LDSGranularityBytes = getLdsDwGranularity(STM) * 4;
+  unsigned LDSGranularityBytes =
+      AMDGPU::getLDSEncodingGranule(STM.getTargetID().getGPUKind());
   ProgInfo.LDSBlocks =
       alignTo(ProgInfo.LDSSize, LDSGranularityBytes) / LDSGranularityBytes;
 
@@ -1679,8 +1680,8 @@ static void EmitPALMetadataCommon(AMDGPUPALMetadata *MD,
 
   MD->updateHwStageMaximum(
       CC, ".lds_size",
-      (unsigned)(CurrentProgramInfo.LdsSize * getLdsDwGranularity(ST) *
-                 sizeof(uint32_t)));
+      (unsigned)(CurrentProgramInfo.LdsSize *
+                 
AMDGPU::getLDSEncodingGranule(ST.getTargetID().getGPUKind())));
 }
 
 // This is the equivalent of EmitProgramInfoSI above, but for when the OS type
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUFeatures.td 
b/llvm/lib/Target/AMDGPU/AMDGPUFeatures.td
index 49c90acabb74b..cd3ac06f60fd1 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUFeatures.td
+++ b/llvm/lib/Target/AMDGPU/AMDGPUFeatures.td
@@ -60,6 +60,21 @@ def FeatureLDSAllocGranularity1024 : 
SubtargetFeatureLDSAllocGranularity<1024>;
 def FeatureLDSAllocGranularity1280 : SubtargetFeatureLDSAllocGranularity<1280>;
 def FeatureLDSAllocGranularity2048 : SubtargetFeatureLDSAllocGranularity<2048>;
 
+// Units used to encode LDS size in program resource registers and metadata.
+// These can differ from the hardware allocation granularity. A generic 
target's
+// encoding granularity must be present on at least one covered GPU.
+class SubtargetFeatureLDSEncodingGranularity <int Granularity> : 
AMDGPUGenericAnyFeature <
+  "lds-encoding-granularity-"#Granularity,
+  "LDSEncodingGranularity",
+  !cast<string>(Granularity),
+  "LDS encoding granularity in bytes.">;
+
+def FeatureLDSEncodingGranularity256 : 
SubtargetFeatureLDSEncodingGranularity<256>;
+def FeatureLDSEncodingGranularity512 : 
SubtargetFeatureLDSEncodingGranularity<512>;
+def FeatureLDSEncodingGranularity1024 : 
SubtargetFeatureLDSEncodingGranularity<1024>;
+def FeatureLDSEncodingGranularity1280 : 
SubtargetFeatureLDSEncodingGranularity<1280>;
+def FeatureLDSEncodingGranularity2048 : 
SubtargetFeatureLDSEncodingGranularity<2048>;
+
 // Whether each wavefront size mode is available on the hardware,
 // independent of the active mode.
 def FeatureSupportsWave32 : SubtargetFeature<"supports-wave32",
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUSubtarget.h 
b/llvm/lib/Target/AMDGPU/AMDGPUSubtarget.h
index e1f331229c463..d2ab61671c932 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUSubtarget.h
+++ b/llvm/lib/Target/AMDGPU/AMDGPUSubtarget.h
@@ -60,6 +60,7 @@ class AMDGPUSubtarget {
   unsigned LocalMemorySize = 0;
   unsigned AddressableLocalMemorySize = 0;
   unsigned LDSAllocationGranularity = 0;
+  unsigned LDSEncodingGranularity = 0;
   char WavefrontSizeLog2 = 0;
   unsigned FlatOffsetBitWidth = 0;
 
diff --git a/llvm/lib/Target/AMDGPU/R600Processors.td 
b/llvm/lib/Target/AMDGPU/R600Processors.td
index 04947b4c94b7c..a72c367fd74a2 100644
--- a/llvm/lib/Target/AMDGPU/R600Processors.td
+++ b/llvm/lib/Target/AMDGPU/R600Processors.td
@@ -61,6 +61,7 @@ def FeatureR700 : R600SubtargetFeatureGeneration<"R700", 
"r700",
 def FeatureEvergreen : R600SubtargetFeatureGeneration<"EVERGREEN", "evergreen",
   [FeatureFetchLimit16, FeatureAddressableLocalMemorySize32768,
    FeatureLDSAllocGranularity256,
+   FeatureLDSEncodingGranularity256,
    FeatureMadMacF32Insts]
 >;
 
@@ -69,6 +70,7 @@ def FeatureNorthernIslands : 
R600SubtargetFeatureGeneration<"NORTHERN_ISLANDS",
   [FeatureFetchLimit16, FeatureWavefrontSize64,
    FeatureAddressableLocalMemorySize32768,
    FeatureLDSAllocGranularity256,
+   FeatureLDSEncodingGranularity256,
    FeatureMadMacF32Insts]
 >;
 
diff --git a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp 
b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
index 11b4dbc24c085..6200167a98354 100644
--- a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
@@ -3665,20 +3665,6 @@ bool isDPALU_DPP(const MCInstrDesc &OpDesc, const 
MCInstrInfo &MII,
   return hasAny64BitVGPROperands(OpDesc, MII, ST);
 }
 
-unsigned getLdsDwGranularity(const MCSubtargetInfo &ST) {
-  if (ST.getFeatureBits().test(FeatureAddressableLocalMemorySize32768))
-    return 64;
-  if (ST.getFeatureBits().test(FeatureAddressableLocalMemorySize65536))
-    return 128;
-  if (ST.getFeatureBits().test(FeatureAddressableLocalMemorySize196608))
-    return 256;
-  if (ST.getFeatureBits().test(FeatureAddressableLocalMemorySize163840))
-    return 320;
-  if (ST.getFeatureBits().test(FeatureAddressableLocalMemorySize327680))
-    return 512;
-  return 64; // In sync with getAddressableLocalMemorySize
-}
-
 bool isPackedSingleSGPRFP32Inst(unsigned Opc) {
   switch (Opc) {
   case AMDGPU::V_PK_ADD_F32_gfx1250:
diff --git a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h 
b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
index e124073f0466e..a9358689514bc 100644
--- a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
+++ b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
@@ -1812,11 +1812,6 @@ getVGPRLoweringOperandTables(const MCInstrDesc &Desc);
 /// \returns true if a memory instruction supports scale_offset modifier.
 bool supportsScaleOffset(const MCInstrInfo &MII, unsigned Opcode);
 
-/// \returns lds block size in terms of dwords. \p
-/// This is used to calculate the lds size encoded for PAL metadata 3.0+ which
-/// must be defined in terms of bytes.
-unsigned getLdsDwGranularity(const MCSubtargetInfo &ST);
-
 class ClusterDimsAttr {
 public:
   enum class Kind { Unknown, NoCluster, VariableDims, FixedDims };
diff --git a/llvm/lib/TargetParser/AMDGPUTargetParser.cpp 
b/llvm/lib/TargetParser/AMDGPUTargetParser.cpp
index 6cd174d0723a7..e81d0d964aebe 100644
--- a/llvm/lib/TargetParser/AMDGPUTargetParser.cpp
+++ b/llvm/lib/TargetParser/AMDGPUTargetParser.cpp
@@ -556,6 +556,34 @@ unsigned AMDGPU::getLDSAllocGranule(Triple::SubArchType 
SubArch) {
   return getLDSAllocGranule(getGPUKindFromSubArch(SubArch));
 }
 
+unsigned AMDGPU::getLDSEncodingGranule(GPUKind AK) {
+  const AMDGPUFeatureBitset &Features = getFeatureBitset(AK);
+  if (Features.none())
+    return 256;
+  assert((Features.test(FEAT_LDS_ENCODING_GRANULARITY_256) ||
+          Features.test(FEAT_LDS_ENCODING_GRANULARITY_512) ||
+          Features.test(FEAT_LDS_ENCODING_GRANULARITY_1024) ||
+          Features.test(FEAT_LDS_ENCODING_GRANULARITY_1280) ||
+          Features.test(FEAT_LDS_ENCODING_GRANULARITY_2048)) &&
+         "missing LDS encoding granularity feature");
+  if (Features.test(FEAT_LDS_ENCODING_GRANULARITY_256))
+    return 256;
+  if (Features.test(FEAT_LDS_ENCODING_GRANULARITY_512))
+    return 512;
+  if (Features.test(FEAT_LDS_ENCODING_GRANULARITY_1024))
+    return 1024;
+  if (Features.test(FEAT_LDS_ENCODING_GRANULARITY_1280))
+    return 1280;
+  if (Features.test(FEAT_LDS_ENCODING_GRANULARITY_2048))
+    return 2048;
+
+  return 256;
+}
+
+unsigned AMDGPU::getLDSEncodingGranule(Triple::SubArchType SubArch) {
+  return getLDSEncodingGranule(getGPUKindFromSubArch(SubArch));
+}
+
 unsigned AMDGPU::getMaxWavesPerEU(GPUKind AK) {
   const GPUInfo *Info = getAMDGPUInfo(AK);
   return Info ? Info->MaxWavesPerEU : 10;
@@ -597,7 +625,12 @@ static const AMDGPUFeatureBitset FrontendOnlyFeatures = {
     FEAT_LDS_ALLOC_GRANULARITY_512,
     FEAT_LDS_ALLOC_GRANULARITY_1024,
     FEAT_LDS_ALLOC_GRANULARITY_1280,
-    FEAT_LDS_ALLOC_GRANULARITY_2048};
+    FEAT_LDS_ALLOC_GRANULARITY_2048,
+    FEAT_LDS_ENCODING_GRANULARITY_256,
+    FEAT_LDS_ENCODING_GRANULARITY_512,
+    FEAT_LDS_ENCODING_GRANULARITY_1024,
+    FEAT_LDS_ENCODING_GRANULARITY_1280,
+    FEAT_LDS_ENCODING_GRANULARITY_2048};
 
 // Add a GPU's features (minus the frontend-only ones) to \p Features. With \p
 // Overwrite false, existing entries are kept so user -mattr overrides win.
diff --git a/llvm/test/CodeGen/AMDGPU/lds-size-gfx1030.ll 
b/llvm/test/CodeGen/AMDGPU/lds-size-gfx1030.ll
new file mode 100644
index 0000000000000..637add975dc9f
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/lds-size-gfx1030.ll
@@ -0,0 +1,26 @@
+; RUN: llc -mtriple=amdgpu10.30-mesa-mesa3d < %s | FileCheck %s 
--check-prefixes=CHECK,MESA
+; RUN: llc -mtriple=amdgpu10.30-amd-amdpal < %s | FileCheck %s 
--check-prefixes=CHECK,PAL
+
+; gfx1030 allocates LDS in 1024-byte blocks but encodes it in 512-byte units.
+; 8252 bytes occupies 9216 bytes of LDS for occupancy, while its encoded size 
is
+; 17 units (8704 bytes). Using the allocation granule for either encoding step
+; would produce an incorrect block count or PAL metadata size.
+
+@lds = addrspace(3) global [2063 x i32] poison, align 4
+
+; CHECK-LABEL: lds_granularity:
+; MESA: granulated_lds_size = 17
+; CHECK: ; Occupancy: 7{{$}}
+; PAL: .hardware_stages:
+; PAL: .cs:
+; PAL: .lds_size:       0x2200
+define amdgpu_kernel void @lds_granularity(i32 %index, i32 %value) #0 {
+  %ptr = getelementptr [2063 x i32], ptr addrspace(3) @lds, i32 0, i32 %index
+  store volatile i32 %value, ptr addrspace(3) %ptr
+  ret void
+}
+
+attributes #0 = { "amdgpu-flat-work-group-size"="1,64" }
+
+!amdgpu.pal.metadata.msgpack = !{!0}
+!0 = !{!"\81\AEamdpal.version\92\03\00"}
diff --git a/llvm/test/CodeGen/AMDGPU/lds-size-gfx9-4-generic.ll 
b/llvm/test/CodeGen/AMDGPU/lds-size-gfx9-4-generic.ll
new file mode 100644
index 0000000000000..c02e454ee10b3
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/lds-size-gfx9-4-generic.ll
@@ -0,0 +1,34 @@
+; RUN: llc -mtriple=amdgpu9.42-mesa-mesa3d < %s | FileCheck %s 
--check-prefix=GFX942
+; RUN: llc -mtriple=amdgpu9.50-mesa-mesa3d < %s | FileCheck %s 
--check-prefix=GFX950
+; RUN: llc -mtriple=amdgpu9.4-mesa-mesa3d < %s | FileCheck %s 
--check-prefix=GENERIC
+
+; gfx9-4-generic uses gfx950's 1280-byte allocation and encoding granules
+; independently of its 64 KiB addressable LDS capacity. gfx942 uses 512 bytes
+; for both granularities.
+
+@lds1280 = addrspace(3) global [320 x i32] poison, align 4
+@lds1284 = addrspace(3) global [321 x i32] poison, align 4
+
+; GFX942-LABEL: one_granule:
+; GFX942: granulated_lds_size = 3
+; GFX950-LABEL: one_granule:
+; GFX950: granulated_lds_size = 1
+; GENERIC-LABEL: one_granule:
+; GENERIC: granulated_lds_size = 1
+define amdgpu_kernel void @one_granule(i32 %index, i32 %value) {
+  %ptr = getelementptr [320 x i32], ptr addrspace(3) @lds1280, i32 0, i32 
%index
+  store volatile i32 %value, ptr addrspace(3) %ptr
+  ret void
+}
+
+; GFX942-LABEL: two_granules:
+; GFX942: granulated_lds_size = 3
+; GFX950-LABEL: two_granules:
+; GFX950: granulated_lds_size = 2
+; GENERIC-LABEL: two_granules:
+; GENERIC: granulated_lds_size = 2
+define amdgpu_kernel void @two_granules(i32 %index, i32 %value) {
+  %ptr = getelementptr [321 x i32], ptr addrspace(3) @lds1284, i32 0, i32 
%index
+  store volatile i32 %value, ptr addrspace(3) %ptr
+  ret void
+}
diff --git a/llvm/test/TableGen/AMDGPUTargetDefLDSAllocGranularity.td 
b/llvm/test/TableGen/AMDGPUTargetDefLDSAllocGranularity.td
index 13136a7d00b59..b5e4337081a7d 100644
--- a/llvm/test/TableGen/AMDGPUTargetDefLDSAllocGranularity.td
+++ b/llvm/test/TableGen/AMDGPUTargetDefLDSAllocGranularity.td
@@ -3,6 +3,8 @@
 // RUN:   | FileCheck %t/valid.td
 // RUN: not llvm-tblgen -gen-amdgpu-target-def -I %t -I %p/../../include -I 
%p/../../lib/Target/AMDGPU %t/missing.td 2>&1 \
 // RUN:   | FileCheck %t/missing.td -DFILE=%t/missing.td 
--implicit-check-not="error:"
+// RUN: not llvm-tblgen -gen-amdgpu-target-def -I %t -I %p/../../include -I 
%p/../../lib/Target/AMDGPU %t/missing-encoding.td 2>&1 \
+// RUN:   | FileCheck %t/missing-encoding.td -DFILE=%t/missing-encoding.td 
--implicit-check-not="error:"
 
 //--- common.td
 include "llvm/Target/Target.td"
@@ -12,25 +14,42 @@ def MyTarget : Target;
 def AMDGPUFrontendVisibleFeatures {
   list<SubtargetFeature> Features = [FeatureLDSAllocGranularity512,
                                    FeatureLDSAllocGranularity1024,
-                                   FeatureLDSAllocGranularity1280];
+                                   FeatureLDSAllocGranularity1280,
+                                   FeatureLDSEncodingGranularity512,
+                                   FeatureLDSEncodingGranularity1024,
+                                   FeatureLDSEncodingGranularity1280];
 }
 
 def GFX942 : AMDGPUProcessorModel<"gfx942", NoSchedModel,
-    [FeatureAddressableLocalMemorySize65536, FeatureLDSAllocGranularity512],
+    [FeatureAddressableLocalMemorySize65536, FeatureLDSAllocGranularity512,
+     FeatureLDSEncodingGranularity512],
     [9, 4, 2]>;
 def GFX950 : AMDGPUProcessorModel<"gfx950", NoSchedModel,
-    [FeatureAddressableLocalMemorySize163840, FeatureLDSAllocGranularity1280],
+    [FeatureAddressableLocalMemorySize163840, FeatureLDSAllocGranularity1280,
+     FeatureLDSEncodingGranularity1280],
     [9, 5, 0]>;
 
 //--- valid.td
 include "common.td"
 // Capacity and granularity can each come from a different covered GPU.
 def : AMDGPUProcessorModel<"gfx9-4-generic", NoSchedModel,
-    [FeatureAddressableLocalMemorySize65536, FeatureLDSAllocGranularity1280],
+    [FeatureAddressableLocalMemorySize65536, FeatureLDSAllocGranularity1280,
+     FeatureLDSEncodingGranularity1280],
     [9, 4, 0]> {
   let CoveredGPUs = [GFX942, GFX950];
 }
-// CHECK: Triple::AMDGPUSubArch9_4, 
AMDGPUFeatureBitset({FEAT_LDS_ALLOC_GRANULARITY_1280}), {9, 4, 0}, [[#]], 10, 
65536, 32, 0},
+// CHECK: Triple::AMDGPUSubArch9_4, 
AMDGPUFeatureBitset({FEAT_LDS_ALLOC_GRANULARITY_1280, 
FEAT_LDS_ENCODING_GRANULARITY_1280}), {9, 4, 0}, [[#]], 10, 65536, 32, 0},
+
+// Allocation and encoding use different units on gfx10.3, including its 
generic.
+def GFX1030 : AMDGPUProcessorModel<"gfx1030", NoSchedModel,
+    [FeatureLDSAllocGranularity1024, FeatureLDSEncodingGranularity512],
+    [10, 3, 0]>;
+def : AMDGPUProcessorModel<"gfx10-3-generic", NoSchedModel,
+    [FeatureLDSAllocGranularity1024, FeatureLDSEncodingGranularity512],
+    [10, 3, 0]> {
+  let CoveredGPUs = [GFX1030];
+}
+// CHECK: Triple::AMDGPUSubArch10_3, 
AMDGPUFeatureBitset({FEAT_LDS_ALLOC_GRANULARITY_1024, 
FEAT_LDS_ENCODING_GRANULARITY_512})
 
 //--- missing.td
 include "common.td"
@@ -40,3 +59,16 @@ def : AMDGPUProcessorModel<"gfx9-4-generic", NoSchedModel,
     [9, 4, 0]> {
   let CoveredGPUs = [GFX942, GFX950];
 }
+
+//--- missing-encoding.td
+include "common.td"
+// An allocation feature does not provide support for an encoding feature.
+def GFX1030 : AMDGPUProcessorModel<"gfx1030", NoSchedModel,
+    [FeatureLDSAllocGranularity1024, FeatureLDSEncodingGranularity512],
+    [10, 3, 0]>;
+// CHECK: [[FILE]]:[[#@LINE+1]]:1: error: generic target 'gfx10-3-generic' 
exposes feature 'lds-encoding-granularity-1024' not supported by any covered GPU
+def : AMDGPUProcessorModel<"gfx10-3-generic", NoSchedModel,
+    [FeatureLDSAllocGranularity1024, FeatureLDSEncodingGranularity1024],
+    [10, 3, 0]> {
+  let CoveredGPUs = [GFX1030];
+}
diff --git a/llvm/unittests/TargetParser/TargetParserTest.cpp 
b/llvm/unittests/TargetParser/TargetParserTest.cpp
index 493e0a6215f4b..d381e8e949f74 100644
--- a/llvm/unittests/TargetParser/TargetParserTest.cpp
+++ b/llvm/unittests/TargetParser/TargetParserTest.cpp
@@ -2833,6 +2833,13 @@ TEST(TargetParserTest, testAMDGPUfillAMDGPUFeatureMap) {
   EXPECT_FALSE(HasFeature("gfx950", "lds-alloc-granularity-1280"));
   EXPECT_FALSE(HasFeature("gfx1310", "lds-alloc-granularity-1024"));
   EXPECT_FALSE(HasFeature("gfx1250", "lds-alloc-granularity-2048"));
+
+  // Encoding granularity is also queried through the bitset only.
+  EXPECT_FALSE(HasFeature("gfx600", "lds-encoding-granularity-256"));
+  EXPECT_FALSE(HasFeature("gfx1030", "lds-encoding-granularity-512"));
+  EXPECT_FALSE(HasFeature("gfx950", "lds-encoding-granularity-1280"));
+  EXPECT_FALSE(HasFeature("gfx1310", "lds-encoding-granularity-1024"));
+  EXPECT_FALSE(HasFeature("gfx1250", "lds-encoding-granularity-2048"));
 }
 
 TEST(TargetParserTest, testAMDGPUgetFeatureBitset) {
@@ -2887,30 +2894,41 @@ TEST(TargetParserTest, 
testAMDGPUHalfAddressableLDSFeature) {
   EXPECT_FALSE(Has(AMDGPU::GK_GFX1310));
 }
 
-TEST(TargetParserTest, testAMDGPULDSAllocGranularityFeatures) {
+TEST(TargetParserTest, testAMDGPULDSGranularityFeatures) {
   auto Has = [](AMDGPU::GPUKind AK, AMDGPU::AMDGPUFeature Feature) {
     return AMDGPU::getFeatureBitset(AK).test(Feature);
   };
-  auto Count = [&Has](AMDGPU::GPUKind AK) {
+  auto CountAlloc = [&Has](AMDGPU::GPUKind AK) {
     return Has(AK, AMDGPU::FEAT_LDS_ALLOC_GRANULARITY_256) +
            Has(AK, AMDGPU::FEAT_LDS_ALLOC_GRANULARITY_512) +
            Has(AK, AMDGPU::FEAT_LDS_ALLOC_GRANULARITY_1024) +
            Has(AK, AMDGPU::FEAT_LDS_ALLOC_GRANULARITY_1280) +
            Has(AK, AMDGPU::FEAT_LDS_ALLOC_GRANULARITY_2048);
   };
+  auto CountEncoding = [&Has](AMDGPU::GPUKind AK) {
+    return Has(AK, AMDGPU::FEAT_LDS_ENCODING_GRANULARITY_256) +
+           Has(AK, AMDGPU::FEAT_LDS_ENCODING_GRANULARITY_512) +
+           Has(AK, AMDGPU::FEAT_LDS_ENCODING_GRANULARITY_1024) +
+           Has(AK, AMDGPU::FEAT_LDS_ENCODING_GRANULARITY_1280) +
+           Has(AK, AMDGPU::FEAT_LDS_ENCODING_GRANULARITY_2048);
+  };
 
-  // Exactly one allocation granularity is set per GPU.
+  // Exactly one granularity of each kind is set per GPU, including generics.
   SmallVector<StringRef> AllGPUs;
   AMDGPU::fillValidArchListAMDGCN(AllGPUs, Triple::NoSubArch);
   for (StringRef Name : AllGPUs) {
     AMDGPU::GPUKind Kind = AMDGPU::parseArchAMDGCN(Name);
-    if (!AMDGPU::isPseudoTarget(Kind))
-      EXPECT_EQ(Count(Kind), 1) << Name;
+    if (!AMDGPU::isPseudoTarget(Kind)) {
+      EXPECT_EQ(CountAlloc(Kind), 1) << Name;
+      EXPECT_EQ(CountEncoding(Kind), 1) << Name;
+    }
   }
 
   // The legacy pseudo-targets do not represent hardware.
-  EXPECT_EQ(Count(AMDGPU::GK_GENERIC), 0);
-  EXPECT_EQ(Count(AMDGPU::GK_GENERIC_HSA), 0);
+  EXPECT_EQ(CountAlloc(AMDGPU::GK_GENERIC), 0);
+  EXPECT_EQ(CountAlloc(AMDGPU::GK_GENERIC_HSA), 0);
+  EXPECT_EQ(CountEncoding(AMDGPU::GK_GENERIC), 0);
+  EXPECT_EQ(CountEncoding(AMDGPU::GK_GENERIC_HSA), 0);
 
   EXPECT_TRUE(Has(AMDGPU::GK_GFX600, AMDGPU::FEAT_LDS_ALLOC_GRANULARITY_256));
   EXPECT_TRUE(Has(AMDGPU::GK_GFX900, AMDGPU::FEAT_LDS_ALLOC_GRANULARITY_512));
@@ -2918,13 +2936,15 @@ TEST(TargetParserTest, 
testAMDGPULDSAllocGranularityFeatures) {
   EXPECT_TRUE(Has(AMDGPU::GK_GFX1310, 
AMDGPU::FEAT_LDS_ALLOC_GRANULARITY_1024));
   EXPECT_TRUE(Has(AMDGPU::GK_GFX1250, 
AMDGPU::FEAT_LDS_ALLOC_GRANULARITY_2048));
 
-  // RDNA2 and later 64 KiB targets allocate LDS in 1024-byte blocks.
+  // RDNA2 and later 64 KiB targets allocate 1024 bytes but encode 512-byte
+  // units.
   for (AMDGPU::GPUKind Kind :
        {AMDGPU::GK_GFX1030, AMDGPU::GK_GFX1100, AMDGPU::GK_GFX1170,
         AMDGPU::GK_GFX1200, AMDGPU::GK_GFX10_3_GENERIC,
         AMDGPU::GK_GFX11_GENERIC, AMDGPU::GK_GFX11_7_GENERIC,
         AMDGPU::GK_GFX12_GENERIC}) {
     EXPECT_TRUE(Has(Kind, AMDGPU::FEAT_LDS_ALLOC_GRANULARITY_1024));
+    EXPECT_TRUE(Has(Kind, AMDGPU::FEAT_LDS_ENCODING_GRANULARITY_512));
   }
 
   // A generic target uses the largest allocation granularity of the GPUs it
@@ -2933,6 +2953,8 @@ TEST(TargetParserTest, 
testAMDGPULDSAllocGranularityFeatures) {
       Has(AMDGPU::GK_GFX9_4_GENERIC, AMDGPU::FEAT_LDS_ALLOC_GRANULARITY_1280));
   EXPECT_FALSE(
       Has(AMDGPU::GK_GFX9_4_GENERIC, AMDGPU::FEAT_LDS_ALLOC_GRANULARITY_512));
+  EXPECT_TRUE(Has(AMDGPU::GK_GFX9_4_GENERIC,
+                  AMDGPU::FEAT_LDS_ENCODING_GRANULARITY_1280));
 }
 
 TEST(TargetParserTest, testAMDGPUfillValidArchListAMDGCN) {
@@ -3390,43 +3412,47 @@ TEST(TargetParserTest, 
testAMDGPUgetAddressableLocalMemorySize) {
       65536u);
 }
 
-TEST(TargetParserTest, testAMDGPUgetLDSAllocGranule) {
+TEST(TargetParserTest, testAMDGPUgetLDSGranules) {
   struct {
     AMDGPU::GPUKind Kind;
     Triple::SubArchType SubArch;
     unsigned Alloc;
+    unsigned Encoding;
   } Cases[] = {
-      {AMDGPU::GK_GFX600, Triple::AMDGPUSubArch600, 256},
-      {AMDGPU::GK_GFX700, Triple::AMDGPUSubArch700, 512},
-      {AMDGPU::GK_GFX900, Triple::AMDGPUSubArch900, 512},
-      {AMDGPU::GK_GFX942, Triple::AMDGPUSubArch942, 512},
-      {AMDGPU::GK_GFX950, Triple::AMDGPUSubArch950, 1280},
-      {AMDGPU::GK_GFX1010, Triple::AMDGPUSubArch1010, 512},
-      {AMDGPU::GK_GFX1030, Triple::AMDGPUSubArch1030, 1024},
-      {AMDGPU::GK_GFX1100, Triple::AMDGPUSubArch1100, 1024},
-      {AMDGPU::GK_GFX1150, Triple::AMDGPUSubArch1150, 1024},
-      {AMDGPU::GK_GFX1170, Triple::AMDGPUSubArch1170, 1024},
-      {AMDGPU::GK_GFX1200, Triple::AMDGPUSubArch1200, 1024},
-      {AMDGPU::GK_GFX1250, Triple::AMDGPUSubArch1250, 2048},
-      {AMDGPU::GK_GFX1251, Triple::AMDGPUSubArch1251, 2048},
-      {AMDGPU::GK_GFX1310, Triple::AMDGPUSubArch1310, 1024},
-      {AMDGPU::GK_GFX9_4_GENERIC, Triple::AMDGPUSubArch9_4, 1280},
-      {AMDGPU::GK_GFX10_1_GENERIC, Triple::AMDGPUSubArch10_1, 512},
-      {AMDGPU::GK_GFX10_3_GENERIC, Triple::AMDGPUSubArch10_3, 1024},
-      {AMDGPU::GK_GFX11_GENERIC, Triple::AMDGPUSubArch11, 1024},
-      {AMDGPU::GK_GFX11_7_GENERIC, Triple::AMDGPUSubArch11_7, 1024},
-      {AMDGPU::GK_GFX12_GENERIC, Triple::AMDGPUSubArch12, 1024},
-      {AMDGPU::GK_GFX12_5_GENERIC, Triple::AMDGPUSubArch12_5, 2048},
-      {AMDGPU::GK_NONE, Triple::NoSubArch, 256},
+      {AMDGPU::GK_GFX600, Triple::AMDGPUSubArch600, 256, 256},
+      {AMDGPU::GK_GFX700, Triple::AMDGPUSubArch700, 512, 512},
+      {AMDGPU::GK_GFX900, Triple::AMDGPUSubArch900, 512, 512},
+      {AMDGPU::GK_GFX942, Triple::AMDGPUSubArch942, 512, 512},
+      {AMDGPU::GK_GFX950, Triple::AMDGPUSubArch950, 1280, 1280},
+      {AMDGPU::GK_GFX1010, Triple::AMDGPUSubArch1010, 512, 512},
+      {AMDGPU::GK_GFX1030, Triple::AMDGPUSubArch1030, 1024, 512},
+      {AMDGPU::GK_GFX1100, Triple::AMDGPUSubArch1100, 1024, 512},
+      {AMDGPU::GK_GFX1150, Triple::AMDGPUSubArch1150, 1024, 512},
+      {AMDGPU::GK_GFX1170, Triple::AMDGPUSubArch1170, 1024, 512},
+      {AMDGPU::GK_GFX1200, Triple::AMDGPUSubArch1200, 1024, 512},
+      {AMDGPU::GK_GFX1250, Triple::AMDGPUSubArch1250, 2048, 2048},
+      {AMDGPU::GK_GFX1251, Triple::AMDGPUSubArch1251, 2048, 2048},
+      {AMDGPU::GK_GFX1310, Triple::AMDGPUSubArch1310, 1024, 1024},
+      {AMDGPU::GK_GFX9_4_GENERIC, Triple::AMDGPUSubArch9_4, 1280, 1280},
+      {AMDGPU::GK_GFX10_1_GENERIC, Triple::AMDGPUSubArch10_1, 512, 512},
+      {AMDGPU::GK_GFX10_3_GENERIC, Triple::AMDGPUSubArch10_3, 1024, 512},
+      {AMDGPU::GK_GFX11_GENERIC, Triple::AMDGPUSubArch11, 1024, 512},
+      {AMDGPU::GK_GFX11_7_GENERIC, Triple::AMDGPUSubArch11_7, 1024, 512},
+      {AMDGPU::GK_GFX12_GENERIC, Triple::AMDGPUSubArch12, 1024, 512},
+      {AMDGPU::GK_GFX12_5_GENERIC, Triple::AMDGPUSubArch12_5, 2048, 2048},
+      {AMDGPU::GK_NONE, Triple::NoSubArch, 256, 256},
   };
   for (const auto &Case : Cases) {
     SCOPED_TRACE(AMDGPU::getArchNameAMDGCN(Case.Kind));
     EXPECT_EQ(AMDGPU::getLDSAllocGranule(Case.Kind), Case.Alloc);
     EXPECT_EQ(AMDGPU::getLDSAllocGranule(Case.SubArch), Case.Alloc);
+    EXPECT_EQ(AMDGPU::getLDSEncodingGranule(Case.Kind), Case.Encoding);
+    EXPECT_EQ(AMDGPU::getLDSEncodingGranule(Case.SubArch), Case.Encoding);
   }
 
   for (AMDGPU::GPUKind Kind : {AMDGPU::GK_GENERIC, AMDGPU::GK_GENERIC_HSA}) {
     EXPECT_EQ(AMDGPU::getLDSAllocGranule(Kind), 256u);
+    EXPECT_EQ(AMDGPU::getLDSEncodingGranule(Kind), 256u);
   }
 }
 

_______________________________________________
llvm-branch-commits mailing list
[email protected]
https://lists.llvm.org/cgi-bin/mailman/listinfo/llvm-branch-commits

Reply via email to