On 9/29/2026 3:34 PM, David Carlier wrote:
> The exporter records an importer when it answers NPA-REQ, so it can send
> NPA-REVOKE as soon as the BO is freed, before the importer has finished
> building the dma-buf for that handle. The revoke handler assumes a fully
> imported node: it dereferences imp_xa_node->dmabuf, which is still NULL
> until the import completes, and drops the xarray reference the importing
> thread still relies on. The importer then links the node and marks it
> READY regardless, so the node can be freed while still on the per-remote
> list.
>
> Only tear down a node that is READY. A PENDING node is still being
> imported and nothing has been handed to user-space yet, so only mark it
> for teardown and send NPA-RELEASE. The importer checks for teardown under
> the xarray lock before linking the node and marking it READY, and unwinds
> otherwise. As NPA-REVOKE always follows NPA-RSP, a revoke that finds the
> node NOT_READY is stale, and one that finds it in teardown hits a node
> that is already being released, so both are ignored.
>
> Fixes: 7cc82cd90d35 ("drm/amdgpu: Implement mechanism to revoke exported 
> memory")
> Assisted-by: LLM
> Signed-off-by: David Carlier <[email protected]>

Reviewed-by: Mukul Joshi <[email protected]>

I will get this tested first internally before pushing this.

Thanks,

Mukul

> ---
> Changes in v3:
> - Handle the node state with a switch: tear down READY nodes, mark
>   PENDING ones for teardown, ignore the rest (Mukul).
> - Drop the npa_done completion on a NOT_READY node: the exporter only
>   records the importer after sending NPA-RSP, so such a revoke is stale
>   (Mukul).
> - No longer send NPA-RELEASE for a node already in teardown (Mukul).
> - Use dev_dbg() for the revoked-during-import message (Mukul).
>
> Changes in v2:
> - Tear down only READY nodes, so a duplicate NPA-REVOKE for a node already
>   in teardown neither dereferences a NULL dmabuf nor drops the node
>   reference twice (Sashiko).
> - Complete npa_done when a revoke arrives before NPA-RSP, so the importer
>   fails right away instead of timing out into a connection reset (Sashiko).
> - Use the current Assisted-by format.
>
> Found by code analysis and compile-tested with W=1. Not tested on hardware,
> as it needs two UALink-connected accelerators in a vPod.
>
> v2: https://lore.kernel.org/all/[email protected]/
> v1: https://lore.kernel.org/all/[email protected]/
>
>  drivers/gpu/drm/amd/amdgpu/amdgpu_ualink.c | 51 +++++++++++++++++-----
>  1 file changed, 39 insertions(+), 12 deletions(-)
>
> diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu_ualink.c 
> b/drivers/gpu/drm/amd/amdgpu/amdgpu_ualink.c
> index 8411ea17172f..ea18e7f2e0a3 100644
> --- a/drivers/gpu/drm/amd/amdgpu/amdgpu_ualink.c
> +++ b/drivers/gpu/drm/amd/amdgpu/amdgpu_ualink.c
> @@ -3288,16 +3288,35 @@ static void 
> amdgpu_ualink_process_npa_revoke_msg(struct amdgpu_device *adev,
>               return;
>       }
>  
> -     WRITE_ONCE(imp_xa_node->node_state, AMDGPU_UALINK_NODE_TEARDOWN);
> -     list_del_init(&imp_xa_node->list);
> -     xa_unlock(&adev->ualink.imp_xa);
> +     switch (READ_ONCE(imp_xa_node->node_state)) {
> +     case AMDGPU_UALINK_NODE_READY:
> +             WRITE_ONCE(imp_xa_node->node_state, 
> AMDGPU_UALINK_NODE_TEARDOWN);
> +             list_del_init(&imp_xa_node->list);
> +             xa_unlock(&adev->ualink.imp_xa);
>  
> -     /* Invalidate the GPUVM mappings */
> -     bo = gem_to_amdgpu_bo(imp_xa_node->dmabuf->priv);
> -     amdgpu_ualink_invalidate_import_mappings(bo);
> +             /* Invalidate the GPUVM mappings */
> +             bo = gem_to_amdgpu_bo(imp_xa_node->dmabuf->priv);
> +             amdgpu_ualink_invalidate_import_mappings(bo);
>  
> -     /* Drop the refcount for the node */
> -     amdgpu_ualink_imp_xa_entry_put(imp_xa_node);
> +             /* Drop the refcount for the node */
> +             amdgpu_ualink_imp_xa_entry_put(imp_xa_node);
> +             break;
> +     case AMDGPU_UALINK_NODE_PENDING:
> +             /* The import is still building the dma-buf and nothing has
> +              * been handed to user-space yet. The importing thread sees
> +              * the teardown state and unwinds.
> +              */
> +             WRITE_ONCE(imp_xa_node->node_state, 
> AMDGPU_UALINK_NODE_TEARDOWN);
> +             xa_unlock(&adev->ualink.imp_xa);
> +             break;
> +     default:
> +             /* NPA-REVOKE always follows NPA-RSP, so a NOT_READY node means
> +              * a stale revoke, and a node in teardown is already being
> +              * released by whoever moved it there.
> +              */
> +             xa_unlock(&adev->ualink.imp_xa);
> +             return;
> +     }
>  
>       r = amdgpu_ualink_send_npa_release_msg(adev, remote_acc_id, handle);
>       if (r)
> @@ -3760,9 +3779,20 @@ static int amdgpu_ualink_do_import_handle(struct 
> amdgpu_device *adev,
>               return r;
>       }
>  
> -     /* Add this node to the imported handles list for the remote GPU */
> +     /* Add this node to the imported handles list for the remote GPU,
> +      * unless the exporter revoked the handle while the import was in
> +      * flight. The dmabuf is released with the last node reference.
> +      */
>       xa_lock(&adev->ualink.imp_xa);
> +     if (READ_ONCE(imp_xa_node->node_state) == AMDGPU_UALINK_NODE_TEARDOWN) {
> +             xa_unlock(&adev->ualink.imp_xa);
> +             dev_dbg(adev->dev,
> +                     "IMPORT: handle:%llx:%llx revoked during import\n",
> +                     handle.handle_hi, handle.handle_lo);
> +             return -EINVAL;
> +     }
>       list_add(&imp_xa_node->list, 
> &adev->ualink.imp_handles_list[remote_acc_id]);
> +     WRITE_ONCE(imp_xa_node->node_state, AMDGPU_UALINK_NODE_READY);
>       xa_unlock(&adev->ualink.imp_xa);
>  
>       return 0;
> @@ -3938,9 +3968,6 @@ int amdgpu_ualink_import_handle(struct drm_device *dev,
>                                       "IMPORT: XA import failed for 
> handle:%llx:%llx\n",
>                                       handle.handle_hi, handle.handle_lo);
>                       goto cleanup;
> -             } else {
> -                     WRITE_ONCE(imp_xa_node->node_state,
> -                                AMDGPU_UALINK_NODE_READY);
>               }
>       }
>  

Reply via email to