On 23-Jun-26 11:22 AM, Lazar, Lijo wrote:


On 23-Jun-26 9:53 AM, Xiang Liu wrote:
amdgpu_xcp_select_scheds() reads the per-XCP scheduler list.
Partition switching rebuilds the same table under xcp_lock.

Take xcp_lock around XCP scheduler selection and release.
This prevents readers from observing partially rebuilt state.

Also revalidate the selected XCP id before indexing the table.
An open file can outlive a switch to another partition mode.

Signed-off-by: Xiang Liu <[email protected]>
---
  drivers/gpu/drm/amd/amdgpu/amdgpu_xcp.c | 37 +++++++++++++++++--------
  1 file changed, 26 insertions(+), 11 deletions(-)

diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu_xcp.c b/drivers/gpu/ drm/amd/amdgpu/amdgpu_xcp.c
index 88e6eab91bc6..1db7d2ad01fc 100644
--- a/drivers/gpu/drm/amd/amdgpu/amdgpu_xcp.c
+++ b/drivers/gpu/drm/amd/amdgpu/amdgpu_xcp.c
@@ -469,16 +469,21 @@ void amdgpu_xcp_release_sched(struct amdgpu_device *adev,
  {
      struct drm_gpu_scheduler *sched;
      struct amdgpu_ring *ring;
+    struct amdgpu_xcp_mgr *xcp_mgr = adev->xcp_mgr;
-    if (!adev->xcp_mgr)
+    if (!xcp_mgr)
          return;
      sched = entity->entity.rq->sched;
-    if (drm_sched_wqueue_ready(sched)) {
-        ring = to_amdgpu_ring(entity->entity.rq->sched);
-        if (ring->xcp_id < MAX_XCP)
-            atomic_dec(&adev->xcp_mgr->xcp[ring->xcp_id].ref_cnt);
-    }
+    if (!drm_sched_wqueue_ready(sched))
+        return;
+
+    ring = to_amdgpu_ring(sched);
+
+    mutex_lock(&xcp_mgr->xcp_lock);
+    if (ring->xcp_id < xcp_mgr->num_xcps && xcp_mgr->xcp[ring- >xcp_id].valid)
+        atomic_dec(&xcp_mgr->xcp[ring->xcp_id].ref_cnt);
+    mutex_unlock(&xcp_mgr->xcp_lock);
  }
  int amdgpu_xcp_select_scheds(struct amdgpu_device *adev,
@@ -490,7 +495,9 @@ int amdgpu_xcp_select_scheds(struct amdgpu_device *adev,
      u32 sel_xcp_id;
      int i;
      struct amdgpu_xcp_mgr *xcp_mgr = adev->xcp_mgr;
+    int r = 0;
+    mutex_lock(&xcp_mgr->xcp_lock);
      if (fpriv->xcp_id == AMDGPU_XCP_NO_PARTITION) {
          u32 least_ref_cnt = ~0;
@@ -507,19 +514,27 @@ int amdgpu_xcp_select_scheds(struct amdgpu_device *adev,
      }
      sel_xcp_id = fpriv->xcp_id;
+    if (sel_xcp_id >= xcp_mgr->num_xcps || !xcp_mgr- >xcp[sel_xcp_id].valid) { +        dev_err(adev->dev, "Selected partition #%d is not valid.", sel_xcp_id);
+        r = -ENODEV;
+        goto out;
+    }
+
      if (xcp_mgr->xcp[sel_xcp_id].gpu_sched[hw_ip] [hw_prio].num_scheds) {
          *num_scheds =
-            xcp_mgr->xcp[fpriv->xcp_id].gpu_sched[hw_ip] [hw_prio].num_scheds; +            xcp_mgr->xcp[sel_xcp_id].gpu_sched[hw_ip] [hw_prio].num_scheds;
          *scheds =
-            xcp_mgr->xcp[fpriv->xcp_id].gpu_sched[hw_ip][hw_prio].sched;
-        atomic_inc(&adev->xcp_mgr->xcp[sel_xcp_id].ref_cnt);
+            xcp_mgr->xcp[sel_xcp_id].gpu_sched[hw_ip][hw_prio].sched;
+        atomic_inc(&xcp_mgr->xcp[sel_xcp_id].ref_cnt);
          dev_dbg(adev->dev, "Selected partition #%d", sel_xcp_id);
      } else {
          dev_err(adev->dev, "Failed to schedule partition #%d.", sel_xcp_id);
-        return -ENOENT;
+        r = -ENOENT;

Doesn't this require goto out as well? Or, for simplicity you could use guard(mutex)

Thanks,
Lijo>       }
-    return 0;

Please ignore. Didn't notice this one earlier.

Thanks,
Lijo

+out:
+    mutex_unlock(&xcp_mgr->xcp_lock);
+    return r;
  }
  static void amdgpu_set_xcp_id(struct amdgpu_device *adev,


Reply via email to