llvmorg-github-actions[bot] wrote:
<!--LLVM PR SUMMARY COMMENT--> @llvm/pr-subscribers-backend-amdgpu Author: hidekisaito <details> <summary>Changes</summary> Stacked on the post-RA V_PERM_PK16 hazard fixup. V_PERM_PK16 must be immediately followed by a "safe" instruction (see SIInstrInfo::isVPermPk16SafeInstr) or the post-RA fixup has to insert a forced-EXEC V_NOP. Teach GCNHazardRecognizer to bias a safe follower into the slot right after a V_PERM_PK16 so that V_NOP can be avoided. Assisted-by: Opus 4.8 Medium --- Full diff: https://github.com/llvm/llvm-project/pull/214597.diff 4 Files Affected: - (modified) llvm/lib/Target/AMDGPU/GCNHazardRecognizer.cpp (+46-3) - (modified) llvm/lib/Target/AMDGPU/GCNHazardRecognizer.h (+21) - (added) llvm/test/CodeGen/AMDGPU/vperm-pk16-postmisched-hazard.mir (+33) - (added) llvm/test/CodeGen/AMDGPU/vperm-pk16-sched-softcost.mir (+56) ``````````diff diff --git a/llvm/lib/Target/AMDGPU/GCNHazardRecognizer.cpp b/llvm/lib/Target/AMDGPU/GCNHazardRecognizer.cpp index aeaa7a9101daf..a81186050f0b0 100644 --- a/llvm/lib/Target/AMDGPU/GCNHazardRecognizer.cpp +++ b/llvm/lib/Target/AMDGPU/GCNHazardRecognizer.cpp @@ -118,14 +118,16 @@ void GCNHazardRecognizer::Reset() { void GCNHazardRecognizer::schedulerReset() { LLVM_DEBUG({ if (CurrentCoExecStage.has_value() || CyclesUntilTRANS > 0 || - CyclesUntilVALU > 0) + CyclesUntilVALU > 0 || CyclesUntilPermPk16Safety > 0) dbgs() << " Scheduler Reset: clearing co-exec window, TRANS=" - << CyclesUntilTRANS << ", VALU=" << CyclesUntilVALU << "\n"; + << CyclesUntilTRANS << ", VALU=" << CyclesUntilVALU + << ", PermPk16=" << CyclesUntilPermPk16Safety << "\n"; }); CurrentCoExecStage = std::nullopt; CoExecWindowStartCycle = 0; CyclesUntilTRANS = 0; CyclesUntilVALU = 0; + CyclesUntilPermPk16Safety = 0; ActiveCoExecInfo = AMDGPU::CoExecInfo(); CoExecWindowLog.fill('.'); } @@ -168,7 +170,7 @@ void GCNHazardRecognizer::schedulerAdvanceCycle() { LLVM_DEBUG({ bool HasState = CurrentCoExecStage.has_value() || CyclesUntilTRANS > 0 || - CyclesUntilVALU > 0; + CyclesUntilVALU > 0 || CyclesUntilPermPk16Safety > 0; if (HasState) { dbgs() << " Scheduler AdvanceCycle:"; if (CurrentCoExecStage.has_value()) { @@ -184,6 +186,9 @@ void GCNHazardRecognizer::schedulerAdvanceCycle() { << (CyclesUntilTRANS - 1); if (CyclesUntilVALU > 0) dbgs() << " VALU=" << CyclesUntilVALU << "->" << (CyclesUntilVALU - 1); + if (CyclesUntilPermPk16Safety > 0) + dbgs() << " PermPk16=" << CyclesUntilPermPk16Safety << "->" + << (CyclesUntilPermPk16Safety - 1); dbgs() << "\n"; } }); @@ -193,6 +198,8 @@ void GCNHazardRecognizer::schedulerAdvanceCycle() { --CyclesUntilTRANS; if (CyclesUntilVALU > 0) --CyclesUntilVALU; + if (CyclesUntilPermPk16Safety > 0) + --CyclesUntilPermPk16Safety; // Advance WMMA co-execution window. if (CurrentCoExecStage.has_value()) { @@ -292,6 +299,21 @@ void GCNHazardRecognizer::updateMultiCycleVALUState(const MachineInstr &MI) { } } +void GCNHazardRecognizer::updateVPermPk16State(const MachineInstr &MI) { + if (!ST.hasVPermPk16Hazard()) + return; + if (!SIInstrInfo::isVPermPk16(MI.getOpcode())) + return; + + // V_PERM_PK16 must be immediately followed by a safe instruction to clear + // the V_PERM_PK16 hazard. Seed to 2: schedulerAdvanceCycle decrements once + // before the next pick's hazard check (same convention as CyclesUntilTRANS), + // so the counter is observed at 1 for the immediately following candidate. + CyclesUntilPermPk16Safety = 2; + LLVM_DEBUG( + dbgs() << " V_PERM_PK16 hazard set: CyclesUntilPermPk16Safety=2\n"); +} + AMDGPU::CoExecMaskT GCNHazardRecognizer::getCoExecMaskForMI(const MachineInstr &MI, const SIInstrInfo &TII) { @@ -408,6 +430,18 @@ GCNHazardRecognizer::checkMultiShadowHazard(const MachineInstr &MI) const { return StallCycles; } +unsigned +GCNHazardRecognizer::checkVPermPk16Hazard(const MachineInstr &MI) const { + if (!CyclesUntilPermPk16Safety) + return 0; + + if (TII.isVPermPk16SafeInstr(MI)) + return 0; + + // Prefer safe instruction over unsafe instruction. + return CyclesUntilPermPk16Safety; +} + void GCNHazardRecognizer::schedulerEmitInstruction(MachineInstr *MI) { LLVM_DEBUG({ bool InWindow = CurrentCoExecStage.has_value(); @@ -439,6 +473,7 @@ void GCNHazardRecognizer::schedulerEmitInstruction(MachineInstr *MI) { updateWMMAWindowState(*MI); updateTRANSState(*MI); updateMultiCycleVALUState(*MI); + updateVPermPk16State(*MI); } void GCNHazardRecognizer::EmitInstruction(SUnit *SU) { @@ -565,6 +600,13 @@ GCNHazardRecognizer::getHazardType(SUnit *SU, int Stalls) { return Hazard; if (checkMultiCycleVALUHazard(*MI) > 0) return Hazard; + // Post-RA the generic scheduler only consults getHazardType(), so gate the + // V_PERM_PK16 hazard here to bias a safe follower into the slot after a + // V_PERM_PK16. Pre-RA the CoExec strategy handles this as a soft cost via + // getHazardWaitStates() (which, unlike a hard gate, does not burn a cycle + // when no safe follower exists), so we do not gate it there. + if (isPostRA() && checkVPermPk16Hazard(*MI) > 0) + return Hazard; // The remaining checks are all defined by register dependences. if (!hasPhysRegs()) return NoHazard; @@ -721,6 +763,7 @@ unsigned GCNHazardRecognizer::getHazardWaitStates(MachineInstr *MI) const { W = std::max(W, checkTRANSHazard(*MI)); W = std::max(W, checkMultiCycleVALUHazard(*MI)); W = std::max(W, checkMultiShadowHazard(*MI)); + W = std::max(W, checkVPermPk16Hazard(*MI)); // The remaining checks are all defined by register dependences. if (!hasPhysRegs()) return W; diff --git a/llvm/lib/Target/AMDGPU/GCNHazardRecognizer.h b/llvm/lib/Target/AMDGPU/GCNHazardRecognizer.h index fdc6700e7508b..3ce6c8cda30e1 100644 --- a/llvm/lib/Target/AMDGPU/GCNHazardRecognizer.h +++ b/llvm/lib/Target/AMDGPU/GCNHazardRecognizer.h @@ -107,6 +107,17 @@ class GCNHazardRecognizer final : public ScheduleHazardRecognizer { /// Tracks cycles until next VALU after multi-cycle VALU (CVT hazard). unsigned CyclesUntilVALU = 0; + /// Set when a V_PERM_PK16 has just been emitted, marking that its immediately + /// following instruction must be a "safe" instruction (see + /// SIInstrInfo::isVPermPk16SafeInstr) to clear the hazard; otherwise a + /// forced-EXEC V_NOP is inserted post-RA. Tracked as a cycle counter for + /// consistency with the TRANS and multi-cycle-VALU shadows, but only the + /// immediately following pick observes it (as 1). Consumed to bias a safe + /// instruction into the slot after a V_PERM_PK16: pre-RA the CoExec strategy + /// treats it as a soft cost (getHazardWaitStates), and post-RA the generic + /// scheduler treats it as a hard hazard (getHazardType). + unsigned CyclesUntilPermPk16Safety = 0; + /// Debug: log of what was scheduled at each stage of the co-exec window. /// '.' = not yet reached, '-' = stall, else CoExecMask short char. std::array<char, AMDGPU::MaxCoExecStages> CoExecWindowLog; @@ -133,6 +144,13 @@ class GCNHazardRecognizer final : public ScheduleHazardRecognizer { /// instructions, return the number of stall cycles until one shadow clears. unsigned checkMultiShadowHazard(const MachineInstr &MI) const; + /// Check the V_PERM_PK16 hazard. Returns a nonzero stall count if a + /// V_PERM_PK16 was just emitted and \p MI is not a safe follower, biasing + /// the scheduler to place a safe instruction immediately after it. Pre-RA + /// this is consumed as a soft cost (getHazardWaitStates); post-RA it is + /// consumed as a hard hazard (getHazardType). + unsigned checkVPermPk16Hazard(const MachineInstr &MI) const; + /// Update WMMA window state when a WMMA instruction is emitted. void updateWMMAWindowState(const MachineInstr &MI); @@ -142,6 +160,9 @@ class GCNHazardRecognizer final : public ScheduleHazardRecognizer { /// Update multi-cycle VALU state when an instruction is emitted. void updateMultiCycleVALUState(const MachineInstr &MI); + /// Update V_PERM_PK16 hazard state when an instruction is emitted. + void updateVPermPk16State(const MachineInstr &MI); + /// Scheduler-mode part of EmitInstruction(). void schedulerEmitInstruction(MachineInstr *MI); diff --git a/llvm/test/CodeGen/AMDGPU/vperm-pk16-postmisched-hazard.mir b/llvm/test/CodeGen/AMDGPU/vperm-pk16-postmisched-hazard.mir new file mode 100644 index 0000000000000..24ab04a61596e --- /dev/null +++ b/llvm/test/CodeGen/AMDGPU/vperm-pk16-postmisched-hazard.mir @@ -0,0 +1,33 @@ +# NOTE: Assertions have been autogenerated by utils/update_mir_test_checks.py UTC_ARGS: --version 6 +# RUN: llc -mtriple=amdgpu12.50 -run-pass=postmisched %s -o - | FileCheck %s +# RUN: llc -mtriple=amdgpu12.50 -run-pass=postmisched,post-RA-hazard-rec %s -o - | FileCheck %s + +# Post-RA counterpart to vperm-pk16-sched-softcost.mir. The generic post-RA +# MachineScheduler has no soft-cost hook, so the V_PERM_PK16 hazard is modeled +# there as a hard hazard in getHazardType(). Both followers depend on the +# V_PERM_PK16 result, so neither can be hoisted above it; the scheduler biases +# the safe single-pass V_ADD_F32 into the slot right after the V_PERM_PK16 and +# pushes the unsafe multi-pass V_MUL_LO_U32 out of it. As a result the post-RA +# hazard fixup has no forced-EXEC V_NOP to insert, so the second RUN line (which +# additionally runs the hazard recognizer) matches the same output. + +--- +name: perm_prefers_safe_follower +tracksRegLiveness: true +body: | + bb.0: + liveins: $vgpr0, $vgpr1, $vgpr2_vgpr3, $vgpr5, $vgpr14_vgpr15 + ; CHECK-LABEL: name: perm_prefers_safe_follower + ; CHECK: liveins: $vgpr0, $vgpr1, $vgpr2_vgpr3, $vgpr5, $vgpr14_vgpr15 + ; CHECK-NEXT: {{ $}} + ; CHECK-NEXT: $vgpr6_vgpr7 = V_PERM_PK16_B4_U4_e64 killed $vgpr0, killed $vgpr1, killed $vgpr2_vgpr3, implicit $exec + ; CHECK-NEXT: $vgpr9 = V_ADD_F32_e32 0, killed $vgpr7, implicit $mode, implicit $exec + ; CHECK-NEXT: $vgpr8 = V_MUL_LO_U32_e64 killed $vgpr6, killed $vgpr5, implicit $exec + ; CHECK-NEXT: GLOBAL_STORE_DWORD killed $vgpr14_vgpr15, killed $vgpr8, 0, 0, implicit $exec + ; CHECK-NEXT: S_ENDPGM 0, implicit killed $vgpr9 + $vgpr6_vgpr7 = V_PERM_PK16_B4_U4_e64 $vgpr0, $vgpr1, $vgpr2_vgpr3, implicit $exec + $vgpr8 = V_MUL_LO_U32_e64 $vgpr6, $vgpr5, implicit $exec + $vgpr9 = V_ADD_F32_e32 0, $vgpr7, implicit $mode, implicit $exec + GLOBAL_STORE_DWORD $vgpr14_vgpr15, $vgpr8, 0, 0, implicit $exec + S_ENDPGM 0, implicit $vgpr9 +... diff --git a/llvm/test/CodeGen/AMDGPU/vperm-pk16-sched-softcost.mir b/llvm/test/CodeGen/AMDGPU/vperm-pk16-sched-softcost.mir new file mode 100644 index 0000000000000..e7b3e52d85c65 --- /dev/null +++ b/llvm/test/CodeGen/AMDGPU/vperm-pk16-sched-softcost.mir @@ -0,0 +1,56 @@ +# NOTE: Assertions have been autogenerated by utils/update_mir_test_checks.py UTC_ARGS: --version 6 +# RUN: llc -mtriple=amdgpu12.50 -run-pass=machine-scheduler -amdgpu-sched-strategy=coexec -verify-misched %s -o - | FileCheck -check-prefix=COEXEC %s + +# V_PERM_PK16 must be immediately followed by a "safe" instruction to clear the +# V_PERM_PK16 hazard, otherwise a forced-EXEC V_NOP has to be inserted post-RA. +# The co-execution scheduler models this as a soft cost (checkVPermPk16Hazard), +# so it biases a safe follower into the slot right after the V_PERM_PK16. The +# input program order places the unsafe multi-pass V_MUL_LO_U32 in that slot; the +# soft cost reorders the safe single-pass V_MOV_B32 into it and pushes the +# V_MUL_LO_U32 out (without the soft cost the co-exec scheduler keeps the input +# order, i.e. leaves V_MUL_LO_U32 in the slot). +# +# The default pre-RA machine-scheduler is intentionally not checked: it does not +# install GCNHazardRecognizer (it uses the generic recognizer while tracking vreg +# liveness), so it has no V_PERM_PK16 awareness and any safe follower it leaves +# is incidental. The post-RA counterpart is vperm-pk16-postmisched-hazard.mir. +# +# All source operands are defined in bb.0 so that the V_PERM_PK16 and both +# candidate followers are ready on entry to bb.1, making the placement decision +# depend only on the scheduling heuristics under test. + +--- +name: perm_prefers_safe_follower +tracksRegLiveness: true +body: | + ; COEXEC-LABEL: name: perm_prefers_safe_follower + ; COEXEC: bb.0: + ; COEXEC-NEXT: successors: %bb.1(0x80000000) + ; COEXEC-NEXT: {{ $}} + ; COEXEC-NEXT: [[DEF:%[0-9]+]]:vgpr_32 = IMPLICIT_DEF + ; COEXEC-NEXT: [[DEF1:%[0-9]+]]:vgpr_32 = IMPLICIT_DEF + ; COEXEC-NEXT: [[DEF2:%[0-9]+]]:vreg_64_align2 = IMPLICIT_DEF + ; COEXEC-NEXT: [[DEF3:%[0-9]+]]:vgpr_32 = IMPLICIT_DEF + ; COEXEC-NEXT: [[DEF4:%[0-9]+]]:vgpr_32 = IMPLICIT_DEF + ; COEXEC-NEXT: S_BRANCH %bb.1 + ; COEXEC-NEXT: {{ $}} + ; COEXEC-NEXT: bb.1: + ; COEXEC-NEXT: [[V_PERM_PK16_B4_U4_e64_:%[0-9]+]]:vreg_64_align2 = V_PERM_PK16_B4_U4_e64 [[DEF]], [[DEF1]], [[DEF2]], implicit $exec + ; COEXEC-NEXT: [[V_MOV_B32_e32_:%[0-9]+]]:vgpr_32 = V_MOV_B32_e32 0, implicit $exec + ; COEXEC-NEXT: [[V_MUL_LO_U32_e64_:%[0-9]+]]:vgpr_32 = V_MUL_LO_U32_e64 [[DEF3]], [[DEF4]], implicit $exec + ; COEXEC-NEXT: S_ENDPGM 0, implicit [[V_PERM_PK16_B4_U4_e64_]], implicit [[V_MUL_LO_U32_e64_]], implicit [[V_MOV_B32_e32_]] + bb.0: + successors: %bb.1 + %0:vgpr_32 = IMPLICIT_DEF + %1:vgpr_32 = IMPLICIT_DEF + %2:vreg_64_align2 = IMPLICIT_DEF + %3:vgpr_32 = IMPLICIT_DEF + %4:vgpr_32 = IMPLICIT_DEF + S_BRANCH %bb.1 + + bb.1: + %5:vreg_64_align2 = V_PERM_PK16_B4_U4_e64 %0, %1, %2, implicit $exec + %6:vgpr_32 = V_MUL_LO_U32_e64 %3, %4, implicit $exec + %7:vgpr_32 = V_MOV_B32_e32 0, implicit $exec + S_ENDPGM 0, implicit %5, implicit %6, implicit %7 +... `````````` </details> https://github.com/llvm/llvm-project/pull/214597 _______________________________________________ llvm-branch-commits mailing list [email protected] https://lists.llvm.org/cgi-bin/mailman/listinfo/llvm-branch-commits
