On Tue, Sep 01, 2026 at 02:41:44PM +0530, Yadav, Arvind wrote:
> 
> On 01-09-2026 02:25, Rodrigo Vivi wrote:
> > On Thu, Aug 27, 2026 at 03:47:52PM +0530, Arvind Yadav wrote:
> > > SVM fault handling and VM rebind work may still run when PCI error
> > > recovery starts or the device becomes permanently wedged. This work can
> > > validate BOs, update page tables or submit migration work after device
> > > I/O has been blocked.
> > > 
> > > Stop SVM fault handling, pagemap population, device-memory copies and
> > > preempt rebind work when device I/O is blocked.
> > > 
> > > SVM invalidation still performs its software cleanup. Do not warn when
> > > TLB invalidation returns -ECANCELED because hardware access is blocked.
> > > 
> > > Cc: Matthew Brost <[email protected]>
> > > Cc: Thomas Hellström <[email protected]>
> > > Cc: Himal Prasad Ghimiray <[email protected]>
> > > Cc: Rodrigo Vivi <[email protected]>
> > > Assisted-by: Claude:claude-opus-4-8
> > > Signed-off-by: Arvind Yadav <[email protected]>
> > > ---
> > >   drivers/gpu/drm/xe/xe_svm.c | 21 ++++++++++++++++++++-
> > >   drivers/gpu/drm/xe/xe_vm.c  | 16 ++++++++++++++++
> > >   2 files changed, 36 insertions(+), 1 deletion(-)
> > > 
> > > diff --git a/drivers/gpu/drm/xe/xe_svm.c b/drivers/gpu/drm/xe/xe_svm.c
> > > index 627a741293d5..9e78131bfa39 100644
> > > --- a/drivers/gpu/drm/xe/xe_svm.c
> > > +++ b/drivers/gpu/drm/xe/xe_svm.c
> > > @@ -11,6 +11,7 @@
> > >   #include <drm/drm_pagemap_util.h>
> > >   #include "xe_bo.h"
> > > +#include "xe_device.h"
> > >   #include "xe_exec_queue_types.h"
> > >   #include "xe_gt_stats.h"
> > >   #include "xe_migrate.h"
> > > @@ -288,8 +289,11 @@ static void xe_svm_invalidate(struct drm_gpusvm 
> > > *gpusvm,
> > >           err = xe_tlb_inval_range_tilemask_submit(xe, vm->usm.asid, 
> > > adj_start, adj_end,
> > >                                                    tile_mask, &batch);
> > > - if (!WARN_ON_ONCE(err))
> > > + if (!err)
> > >                   xe_tlb_inval_batch_wait(&batch);
> > > + else if (!(err == -ECANCELED &&
> > > +            xe_device_io_blocked(xe)))
> > > +         WARN_ON_ONCE(err);
> > if (!err)
> >          xe_tlb_inval_batch_wait(&batch);
> > else
> >          WARN_ON_ONCE(err != -ECANCELED ||
> >                       !xe_device_io_blocked(xe));
> 
> 
> Noted.
> 
> Thanks,
> Arvind
> 
> > 
> > >   range_notifier_event_end:
> > >           r = first;
> > > @@ -631,6 +635,11 @@ static int xe_svm_copy(struct page **pages,
> > >                   }
> > >                   XE_WARN_ON(spage && xe_page_to_vr(spage) != vr);
> > > +         if (vr && xe_device_io_blocked(xe)) {
> > > +                 err = -ECANCELED;
> > > +                 goto err_out;
> > > +         }
> > > +

Again a toctou given this state can immediately change.

Unless I'm missing something - my answer is basically no, likely for the
entire series unless this is a non-toctou check as otherwise we just have a
bunch of weakly ordered checks throughout the driver which don't
actually protect anything.

Matt

> > >                   /*
> > >                    * CPU page and device page valid, capture physical 
> > > address on
> > >                    * first device page, check if physical contiguous on 
> > > subsequent
> > > @@ -1125,6 +1134,11 @@ static int xe_drm_pagemap_populate_mm(struct 
> > > drm_pagemap *dpagemap,
> > >           if (!drm_dev_enter(&xe->drm, &idx))
> > >                   return -ENODEV;
> > > + if (xe_device_io_blocked(xe)) {
> > > +         err = -ECANCELED;
> > > +         goto out_drm;
> > > + }
> > > +
> > >           xe_pm_runtime_get(xe);
> > >           xe_validation_guard(&vctx, &xe->val, &exec, (struct 
> > > xe_val_flags) {}, err) {
> > > @@ -1165,6 +1179,8 @@ static int xe_drm_pagemap_populate_mm(struct 
> > > drm_pagemap *dpagemap,
> > >                   xe_bo_put(bo);
> > >           }
> > >           xe_pm_runtime_put(xe);
> > > +
> > > +out_drm:
> > >           drm_dev_exit(idx);
> > >           return err;
> > > @@ -1301,6 +1317,9 @@ static int __xe_svm_handle_pagefault(struct xe_vm 
> > > *vm, struct xe_vma *vma,
> > >                   drm_gpusvm_range_put(&range->base);
> > >           }
> > > + if (xe_device_io_blocked(vm->xe))
> > > +         return -ECANCELED;
> > > +
> > >           /* Always process UNMAPs first so view SVM ranges is current */
> > >           err = xe_svm_garbage_collector(vm);
> > >           if (err)
> > > diff --git a/drivers/gpu/drm/xe/xe_vm.c b/drivers/gpu/drm/xe/xe_vm.c
> > > index 19b3d0be7928..0cee5306fb9c 100644
> > > --- a/drivers/gpu/drm/xe/xe_vm.c
> > > +++ b/drivers/gpu/drm/xe/xe_vm.c
> > > @@ -497,6 +497,11 @@ static void preempt_rebind_work_func(struct 
> > > work_struct *w)
> > >           }
> > >   retry:
> > > + if (xe_device_io_blocked(vm->xe)) {
> > > +         err = 0;
> > > +         goto out_unlock_outer;
> > > + }
> > > +
> > >           if (!try_wait_for_completion(&vm->xe->pm_block) && 
> > > vm_suspend_rebind_worker(vm)) {
> > >                   up_write(&vm->lock);
> > >                   /* We don't actually block but don't make progress. */
> > > @@ -518,6 +523,12 @@ static void preempt_rebind_work_func(struct 
> > > work_struct *w)
> > >           drm_exec_until_all_locked(&exec) {
> > >                   bool done = false;
> > > +         if (xe_device_io_blocked(vm->xe)) {
> > > +                 xe_validation_ctx_fini(&ctx);
> > > +                 err = 0;
> > > +                 goto out_unlock_outer;
> > > +         }
> > > +
> > >                   err = xe_preempt_work_begin(&exec, vm, &done);
> > >                   drm_exec_retry_on_contention(&exec);
> > >                   xe_validation_retry_on_oom(&ctx, &err);
> > > @@ -531,6 +542,11 @@ static void preempt_rebind_work_func(struct 
> > > work_struct *w)
> > >           if (err)
> > >                   goto out_unlock;
> > > + if (xe_device_io_blocked(vm->xe)) {
> > > +         err = 0;
> > > +         goto out_unlock;
> > > + }
> > > +
> > >           xe_vm_set_validation_exec(vm, &exec);
> > >           err = xe_vm_rebind(vm, true);
> > >           xe_vm_set_validation_exec(vm, NULL);
> > > -- 
> > > 2.43.0
> > > 

Reply via email to