AMD General

> -----Original Message-----
> From: Lazar, Lijo <[email protected]>
> Sent: Friday, June 26, 2026 4:16 PM
> To: Yang, Stanley <[email protected]>; [email protected]
> Subject: Re: [PATCH 1/1] drm/amdgpu/ras: Resum RAS IP hw init during nps
> dynamic switch
>
>
>
> On 26-Jun-26 12:47 PM, Stanley.Yang wrote:
> > On an XGMI reset-on-init (NPS memory patition mode swith), RAS IP hw
> > fini, sw fini is called but hw init is skipped due to RAS IP block is
> > not included in hwinit mask, so need call RAS IP hw init during XGMI
> > reset-on-init.
> >
>
> After reset, we set it to default level.
>
> https://github.com/torvalds/linux/blob/master/drivers/gpu/drm/amd/amdg
> pu/amdgpu_device.c#L5211
>
> The default level includes all IP blocks, hence it's supposed to resume all
> blocks.
>
> With RAS as a separate ip block, ideally it should have resumed. Could you
> check why/how RAS IP block is missed?

static const struct amd_ip_funcs __maybe_unused ras_v1_0_ip_funcs = {
    .name = "ras_v1_0",
    .sw_init = amdgpu_ras_mgr_sw_init,
    .sw_fini = amdgpu_ras_mgr_sw_fini,
    .hw_init = amdgpu_ras_mgr_hw_init,
    .hw_fini = amdgpu_ras_mgr_hw_fini,
};

The RAS IP block does not register suspend and resume callback function, so RAS 
IP block hw init function missed under this case.

Regards,
Stanley
>
> Thanks,
> Lijo
> > Signed-off-by: Stanley.Yang <[email protected]>
> > ---
> >   drivers/gpu/drm/amd/amdgpu/amdgpu_ras.c       | 14 +++++++++++-
> >   drivers/gpu/drm/amd/amdgpu/amdgpu_ras.h       |  1 +
> >   drivers/gpu/drm/amd/amdgpu/amdgpu_xgmi.c      | 10 +++++++++
> >   .../gpu/drm/amd/ras/ras_mgr/amdgpu_ras_mgr.c  | 22
> +++++++++++++++++++
> >   .../gpu/drm/amd/ras/ras_mgr/amdgpu_ras_mgr.h  |  1 +
> >   5 files changed, 47 insertions(+), 1 deletion(-)
> >
> > diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu_ras.c
> > b/drivers/gpu/drm/amd/amdgpu/amdgpu_ras.c
> > index 7ef7c54ab982..e11c542a01b6 100644
> > --- a/drivers/gpu/drm/amd/amdgpu/amdgpu_ras.c
> > +++ b/drivers/gpu/drm/amd/amdgpu/amdgpu_ras.c
> > @@ -3857,7 +3857,14 @@ int amdgpu_ras_init_badpage_info(struct
> amdgpu_device *adev)
> >     if (!con || amdgpu_sriov_vf(adev))
> >             return 0;
> >
> > -   if (amdgpu_uniras_enabled(adev))
> > +   /*
> > +    * For the reset-on-init path (e.g. an NPS memory partition,
> > +    * switch) the RAS IP block hw_init has not been enabled and
> > +    * the amdgpu_uniras_enabled return false, check amdgpu ras
> > +    * context uniras_enabled flag, eepron init will be called
> > +    * during RAS IP block hw_init.
> > +    */
> > +   if (amdgpu_uniras_enabled(adev) || con->uniras_enabled)
> >             return 0;
> >
> >     control = &con->eeprom_control;
> > @@ -5859,3 +5866,8 @@ void amdgpu_ras_post_reset(struct
> amdgpu_device *adev,
> >                     amdgpu_ras_mgr_post_reset(tmp_adev);
> >     }
> >   }
> > +
> > +void amdgpu_ras_resume_after_reset(struct amdgpu_device *adev) {
> > +   amdgpu_ras_mgr_resume_after_reset(adev);
> > +}
> > diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu_ras.h
> > b/drivers/gpu/drm/amd/amdgpu/amdgpu_ras.h
> > index a86ab65aa2f0..ad24c7cf8936 100644
> > --- a/drivers/gpu/drm/amd/amdgpu/amdgpu_ras.h
> > +++ b/drivers/gpu/drm/amd/amdgpu/amdgpu_ras.h
> > @@ -1045,4 +1045,5 @@ void amdgpu_ras_pre_reset(struct
> amdgpu_device *adev,
> >                                       struct list_head *device_list);
> >   void amdgpu_ras_post_reset(struct amdgpu_device *adev,
> >                                       struct list_head *device_list);
> > +void amdgpu_ras_resume_after_reset(struct amdgpu_device *adev);
> >   #endif
> > diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu_xgmi.c
> > b/drivers/gpu/drm/amd/amdgpu/amdgpu_xgmi.c
> > index 9a4e8715742a..f175c8987aeb 100644
> > --- a/drivers/gpu/drm/amd/amdgpu/amdgpu_xgmi.c
> > +++ b/drivers/gpu/drm/amd/amdgpu/amdgpu_xgmi.c
> > @@ -1669,6 +1669,16 @@ static void
> amdgpu_xgmi_reset_on_init_work(struct work_struct *work)
> >             if (r && r != -EHWPOISON)
> >                     dev_err(tmp_adev->dev,
> >                             "error during bad page data initialization");
> > +
> > +           /*
> > +            * For the reset-on-init path (e.g. an NPS memory partition
> > +            * switch) the RAS IP block hw_init was skipped under the
> > +            * minimal init level, so uniras was never enabled. Bring it
> > +            * up now that the reset domain has been unlocked. This is a
> > +            * no-op for any other reset path where RAS is already
> > +            * initialized, and for non-uniras devices.
> > +            */
> > +           amdgpu_ras_resume_after_reset(tmp_adev);
> >     }
> >   }
> >
> > diff --git a/drivers/gpu/drm/amd/ras/ras_mgr/amdgpu_ras_mgr.c
> > b/drivers/gpu/drm/amd/ras/ras_mgr/amdgpu_ras_mgr.c
> > index f627a97797ed..a70e532b3d00 100644
> > --- a/drivers/gpu/drm/amd/ras/ras_mgr/amdgpu_ras_mgr.c
> > +++ b/drivers/gpu/drm/amd/ras/ras_mgr/amdgpu_ras_mgr.c
> > @@ -465,6 +465,28 @@ static int amdgpu_ras_mgr_hw_fini(struct
> amdgpu_ip_block *ip_block)
> >     return 0;
> >   }
> >
> > +int amdgpu_ras_mgr_resume_after_reset(struct amdgpu_device *adev) {
> > +   struct amdgpu_ras *con = amdgpu_ras_get_context(adev);
> > +   struct amdgpu_ras_mgr *ras_mgr =
> amdgpu_ras_mgr_get_context(adev);
> > +   struct amdgpu_ip_block *ip_block;
> > +
> > +   if (!con || !con->uniras_enabled)
> > +           return 0;
> > +
> > +   if (!ras_mgr || !ras_mgr->ras_core)
> > +           return -EINVAL;
> > +
> > +   if (ras_mgr->ras_is_ready)
> > +           return 0;
> > +
> > +   ip_block = amdgpu_device_ip_get_ip_block(adev,
> AMD_IP_BLOCK_TYPE_RAS);
> > +   if (!ip_block)
> > +           return -EINVAL;
> > +
> > +   return amdgpu_ras_mgr_hw_init(ip_block); }
> > +
> >   struct amdgpu_ras_mgr *amdgpu_ras_mgr_get_context(struct
> amdgpu_device *adev)
> >   {
> >     if (!adev || !adev->psp.ras_context.ras) diff --git
> > a/drivers/gpu/drm/amd/ras/ras_mgr/amdgpu_ras_mgr.h
> > b/drivers/gpu/drm/amd/ras/ras_mgr/amdgpu_ras_mgr.h
> > index 4f44a917d48b..3f80b9f1f0ac 100644
> > --- a/drivers/gpu/drm/amd/ras/ras_mgr/amdgpu_ras_mgr.h
> > +++ b/drivers/gpu/drm/amd/ras/ras_mgr/amdgpu_ras_mgr.h
> > @@ -82,6 +82,7 @@ int amdgpu_ras_mgr_handle_ras_cmd(struct
> amdgpu_device *adev,
> >             void *output, uint32_t out_size);
> >   int amdgpu_ras_mgr_pre_reset(struct amdgpu_device *adev);
> >   int amdgpu_ras_mgr_post_reset(struct amdgpu_device *adev);
> > +int amdgpu_ras_mgr_resume_after_reset(struct amdgpu_device *adev);
> >   int amdgpu_ras_mgr_lookup_bad_pages_in_a_row(struct amdgpu_device
> *adev,
> >             uint64_t addr, uint64_t *nps_page_addr, uint32_t
> max_page_count);
> >   #endif

Reply via email to