SVM fault handling and VM rebind work may still run when PCI error recovery starts or the device becomes permanently wedged. This work can validate BOs, update page tables or submit migration work after device I/O has been blocked.
Stop SVM fault handling, pagemap population, device-memory copies and preempt rebind work when device I/O is blocked. SVM invalidation still performs its software cleanup. Do not warn when TLB invalidation returns -ECANCELED because hardware access is blocked. Cc: Matthew Brost <[email protected]> Cc: Thomas Hellström <[email protected]> Cc: Himal Prasad Ghimiray <[email protected]> Cc: Rodrigo Vivi <[email protected]> Assisted-by: Claude:claude-opus-4-8 Signed-off-by: Arvind Yadav <[email protected]> --- drivers/gpu/drm/xe/xe_svm.c | 21 ++++++++++++++++++++- drivers/gpu/drm/xe/xe_vm.c | 16 ++++++++++++++++ 2 files changed, 36 insertions(+), 1 deletion(-) diff --git a/drivers/gpu/drm/xe/xe_svm.c b/drivers/gpu/drm/xe/xe_svm.c index 627a741293d5..9e78131bfa39 100644 --- a/drivers/gpu/drm/xe/xe_svm.c +++ b/drivers/gpu/drm/xe/xe_svm.c @@ -11,6 +11,7 @@ #include <drm/drm_pagemap_util.h> #include "xe_bo.h" +#include "xe_device.h" #include "xe_exec_queue_types.h" #include "xe_gt_stats.h" #include "xe_migrate.h" @@ -288,8 +289,11 @@ static void xe_svm_invalidate(struct drm_gpusvm *gpusvm, err = xe_tlb_inval_range_tilemask_submit(xe, vm->usm.asid, adj_start, adj_end, tile_mask, &batch); - if (!WARN_ON_ONCE(err)) + if (!err) xe_tlb_inval_batch_wait(&batch); + else if (!(err == -ECANCELED && + xe_device_io_blocked(xe))) + WARN_ON_ONCE(err); range_notifier_event_end: r = first; @@ -631,6 +635,11 @@ static int xe_svm_copy(struct page **pages, } XE_WARN_ON(spage && xe_page_to_vr(spage) != vr); + if (vr && xe_device_io_blocked(xe)) { + err = -ECANCELED; + goto err_out; + } + /* * CPU page and device page valid, capture physical address on * first device page, check if physical contiguous on subsequent @@ -1125,6 +1134,11 @@ static int xe_drm_pagemap_populate_mm(struct drm_pagemap *dpagemap, if (!drm_dev_enter(&xe->drm, &idx)) return -ENODEV; + if (xe_device_io_blocked(xe)) { + err = -ECANCELED; + goto out_drm; + } + xe_pm_runtime_get(xe); xe_validation_guard(&vctx, &xe->val, &exec, (struct xe_val_flags) {}, err) { @@ -1165,6 +1179,8 @@ static int xe_drm_pagemap_populate_mm(struct drm_pagemap *dpagemap, xe_bo_put(bo); } xe_pm_runtime_put(xe); + +out_drm: drm_dev_exit(idx); return err; @@ -1301,6 +1317,9 @@ static int __xe_svm_handle_pagefault(struct xe_vm *vm, struct xe_vma *vma, drm_gpusvm_range_put(&range->base); } + if (xe_device_io_blocked(vm->xe)) + return -ECANCELED; + /* Always process UNMAPs first so view SVM ranges is current */ err = xe_svm_garbage_collector(vm); if (err) diff --git a/drivers/gpu/drm/xe/xe_vm.c b/drivers/gpu/drm/xe/xe_vm.c index 19b3d0be7928..0cee5306fb9c 100644 --- a/drivers/gpu/drm/xe/xe_vm.c +++ b/drivers/gpu/drm/xe/xe_vm.c @@ -497,6 +497,11 @@ static void preempt_rebind_work_func(struct work_struct *w) } retry: + if (xe_device_io_blocked(vm->xe)) { + err = 0; + goto out_unlock_outer; + } + if (!try_wait_for_completion(&vm->xe->pm_block) && vm_suspend_rebind_worker(vm)) { up_write(&vm->lock); /* We don't actually block but don't make progress. */ @@ -518,6 +523,12 @@ static void preempt_rebind_work_func(struct work_struct *w) drm_exec_until_all_locked(&exec) { bool done = false; + if (xe_device_io_blocked(vm->xe)) { + xe_validation_ctx_fini(&ctx); + err = 0; + goto out_unlock_outer; + } + err = xe_preempt_work_begin(&exec, vm, &done); drm_exec_retry_on_contention(&exec); xe_validation_retry_on_oom(&ctx, &err); @@ -531,6 +542,11 @@ static void preempt_rebind_work_func(struct work_struct *w) if (err) goto out_unlock; + if (xe_device_io_blocked(vm->xe)) { + err = 0; + goto out_unlock; + } + xe_vm_set_validation_exec(vm, &exec); err = xe_vm_rebind(vm, true); xe_vm_set_validation_exec(vm, NULL); -- 2.43.0
