Signal GPU_RESET EVENTFD subscriptions after successful GPU reset completion.
GPU reset notifications allow userspace to observe that GPU recovery has completed and the device is operational again. GPU_RESET is a device-scoped event, so no queue object is used. All processes that subscribed to GPU_RESET on the device are notified. EVENTFD remains notification-only. v2: - Move GPU_RESET EVENTFD signaling to the path that confirms a real GPU hardware reset occurred. - Notify only after successful hardware reset completion. - Do not signal reset-skip or recovery-only paths. Cc: Alex Deucher <[email protected]> Cc: Christian König <[email protected]> Signed-off-by: Srinivasan Shanmugam <[email protected]> Reviewed-by: Alex Deucher <[email protected]> --- drivers/gpu/drm/amd/amdgpu/amdgpu_device.c | 51 +++++++++++++++++++++- 1 file changed, 49 insertions(+), 2 deletions(-) diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu_device.c b/drivers/gpu/drm/amd/amdgpu/amdgpu_device.c index c924a55fd16f..15aaceae13d1 100644 --- a/drivers/gpu/drm/amd/amdgpu/amdgpu_device.c +++ b/drivers/gpu/drm/amd/amdgpu/amdgpu_device.c @@ -5390,15 +5390,43 @@ static void amdgpu_device_halt_activities(struct amdgpu_device *adev, } } +/* + * Signal GPU_RESET EVENTFD subscriptions for all open DRM files. + * + * GPU reset is a device-wide event rather than being associated with a + * specific user queue. Notify every process that subscribed to the + * GPU_RESET event on this device. + */ +static void amdgpu_device_eventfd_signal_gpu_reset(struct amdgpu_device *adev) +{ + struct drm_device *ddev = adev_to_drm(adev); + struct drm_file *file; + + mutex_lock(&ddev->filelist_mutex); + + list_for_each_entry(file, &ddev->filelist, lhead) { + struct amdgpu_fpriv *fpriv = file->driver_priv; + + if (fpriv) + amdgpu_eventfd_signal(&fpriv->eventfd_mgr, + DRM_AMDGPU_EVENT_TYPE_GPU_RESET, + NULL); + } + + mutex_unlock(&ddev->filelist_mutex); +} + static int amdgpu_device_asic_reset(struct amdgpu_device *adev, struct list_head *device_list, struct amdgpu_reset_context *reset_context) { struct amdgpu_device *tmp_adev = NULL; + bool need_full_reset, skip_hw_reset; int retry_limit = AMDGPU_MAX_RETRY_LIMIT; int r = 0; -retry: /* Rest of adevs pre asic reset from XGMI hive. */ +retry: + /* Rest of adevs pre asic reset from XGMI hive. */ list_for_each_entry(tmp_adev, device_list, reset_list) { r = amdgpu_device_pre_asic_reset(tmp_adev, reset_context); /*TODO Should we stop ?*/ @@ -5430,10 +5458,30 @@ static int amdgpu_device_asic_reset(struct amdgpu_device *adev, } if (r) adev->asic_reset_res = r; + else + amdgpu_device_eventfd_signal_gpu_reset(adev); } else { + /* + * Save the reset decision before amdgpu_do_asic_reset(), + * which may clear AMDGPU_NEED_FULL_RESET before returning. + */ + need_full_reset = + test_bit(AMDGPU_NEED_FULL_RESET, &reset_context->flags); + skip_hw_reset = + test_bit(AMDGPU_SKIP_HW_RESET, &reset_context->flags); + r = amdgpu_do_asic_reset(device_list, reset_context); if (r && r == -EAGAIN) goto retry; + + /* + * Notify only after a real hardware reset completed. + * Do not notify recovery-only or reset-skip paths. + */ + if (!r && need_full_reset && !skip_hw_reset) { + list_for_each_entry(tmp_adev, device_list, reset_list) + amdgpu_device_eventfd_signal_gpu_reset(tmp_adev); + } } list_for_each_entry(tmp_adev, device_list, reset_list) { @@ -5530,7 +5578,6 @@ static void amdgpu_device_gpu_resume(struct amdgpu_device *adev, } } - /** * amdgpu_device_gpu_recover - reset the asic and recover scheduler * -- 2.34.1
