================
@@ -427,15 +428,50 @@ class SIInsertWaitcnts {
return SIInstrInfo::mayWriteLDSThroughDMA(MI) && isAsync(MI);
}
- bool shouldUpdateAsyncMark(const MachineInstr &MI,
- AMDGPU::InstCounterType T) const {
- if (SIInstrInfo::usesTENSOR_CNT(MI))
- return T == AMDGPU::TENSOR_CNT;
+ std::optional<AMDGPU::AsyncStage::Stage>
+ getAsyncStageToUpdate(const MachineInstr &MI,
+ AMDGPU::InstCounterType T) const {
+ if (SIInstrInfo::usesTENSOR_CNT(MI) && T == AMDGPU::TENSOR_CNT)
+ return AMDGPU::AsyncStage::TENSOR;
if (!isAsyncLdsDmaWrite(MI))
- return false;
- if (SIInstrInfo::usesASYNC_CNT(MI))
- return T == AMDGPU::ASYNC_CNT;
- return T == AMDGPU::LOAD_CNT;
+ return std::nullopt;
+ if (!SIInstrInfo::usesASYNC_CNT(MI) && T == AMDGPU::LOAD_CNT)
+ return AMDGPU::AsyncStage::BUFFER_GLOBAL_LOAD;
+ if (SIInstrInfo::usesASYNC_CNT(MI) && T == AMDGPU::ASYNC_CNT) {
+ switch (MI.getOpcode()) {
+ // Keep in sync with FLATInstructions.td.
+ case AMDGPU::GLOBAL_LOAD_ASYNC_TO_LDS_B8:
----------------
ssahasra wrote:
Uggh, as a next step, this really needs to be moved to tablegen. The names of
async stages should be known in tablegen, and it should be possible to just
declare an intrinsic and map it to the correct stage in one go.
https://github.com/llvm/llvm-project/pull/220442
_______________________________________________
cfe-commits mailing list
[email protected]
https://lists.llvm.org/cgi-bin/mailman/listinfo/cfe-commits