================
@@ -427,15 +428,50 @@ class SIInsertWaitcnts {
     return SIInstrInfo::mayWriteLDSThroughDMA(MI) && isAsync(MI);
   }
 
-  bool shouldUpdateAsyncMark(const MachineInstr &MI,
-                             AMDGPU::InstCounterType T) const {
-    if (SIInstrInfo::usesTENSOR_CNT(MI))
-      return T == AMDGPU::TENSOR_CNT;
+  std::optional<AMDGPU::AsyncStage::Stage>
+  getAsyncStageToUpdate(const MachineInstr &MI,
+                        AMDGPU::InstCounterType T) const {
+    if (SIInstrInfo::usesTENSOR_CNT(MI) && T == AMDGPU::TENSOR_CNT)
+      return AMDGPU::AsyncStage::TENSOR;
     if (!isAsyncLdsDmaWrite(MI))
-      return false;
-    if (SIInstrInfo::usesASYNC_CNT(MI))
-      return T == AMDGPU::ASYNC_CNT;
-    return T == AMDGPU::LOAD_CNT;
+      return std::nullopt;
+    if (!SIInstrInfo::usesASYNC_CNT(MI) && T == AMDGPU::LOAD_CNT)
+      return AMDGPU::AsyncStage::BUFFER_GLOBAL_LOAD;
+    if (SIInstrInfo::usesASYNC_CNT(MI) && T == AMDGPU::ASYNC_CNT) {
+      switch (MI.getOpcode()) {
+      // Keep in sync with FLATInstructions.td.
+      case AMDGPU::GLOBAL_LOAD_ASYNC_TO_LDS_B8:
----------------
ssahasra wrote:

Uggh, as a next step, this really needs to be moved to tablegen. The names of 
async stages should be known in tablegen, and it should be possible to just 
declare an intrinsic and map it to the correct stage in one go.

https://github.com/llvm/llvm-project/pull/220442
_______________________________________________
cfe-commits mailing list
[email protected]
https://lists.llvm.org/cgi-bin/mailman/listinfo/cfe-commits

Reply via email to