drm_sched_fence_get_timeline_name() returns fence->sched->name, and the
drm_sched_fence ops keep a .release callback, so the fence is not
ops-detached on signalling (dma_fence_signal_timestamp_locked() only clears
->ops for fences without .release/.wait). The callback therefore stays
reachable on a long-signalled, userspace-held finished fence and
unconditionally dereferences fence->sched.

A driver that allocates a drm_gpu_scheduler at per-context/per-queue/per-VM
granularity and frees it on an unprivileged context/fd close, while
exporting the resulting finished fence to userspace (drm_syncobj /
sync_file / dma_resv), leaves fence->sched dangling after the free. A
subsequent SYNC_IOC_FILE_INFO ioctl (which calls get_timeline_name()) then
reads the freed scheduler:

  BUG: KASAN: slab-use-after-free in drm_sched_fence_get_timeline_name

This is the same class as CVE-2025-38703 (drm/xe) and CVE-2025-71302
(drm/panthor), which were fixed per-driver. amdxdna, nouveau and msm
(VM_BIND) are still affected in mainline, so fix it in the core to cover any
per-context-scheduler driver at once.

Cache the scheduler's name pointer in the fence at init time, while the
scheduler is guaranteed alive, and return the cached value from
get_timeline_name() without dereferencing fence->sched. The timeline name is
not guaranteed by the contract to outlive the scheduler, so document in
struct drm_sched_init_args that the @name passed to drm_sched_init() must
follow the dma-fence safe access rules and outlive any exported fence. Every
in-tree driver passes a string literal, which satisfies this; drm/xe's
299bc6d50b1b keeps its dynamically-allocated name alive across the RCU grace
and can be simplified on top of this.

Fixes: 506aa8b02a8d ("dma-fence: Add safe access helpers and document the 
rules")
Cc: [email protected]
Signed-off-by: Jonghyuk Kim(MalHyuk) <[email protected]>
---
 drivers/gpu/drm/scheduler/sched_fence.c | 16 +++++++++++++++-
 include/drm/gpu_scheduler.h             | 18 +++++++++++++++++-
 2 files changed, 32 insertions(+), 2 deletions(-)

diff --git a/drivers/gpu/drm/scheduler/sched_fence.c 
b/drivers/gpu/drm/scheduler/sched_fence.c
index 096fe28aa9c9..a944eeeb25bd 100644
--- a/drivers/gpu/drm/scheduler/sched_fence.c
+++ b/drivers/gpu/drm/scheduler/sched_fence.c
@@ -92,7 +92,13 @@ static const char *drm_sched_fence_get_driver_name(struct 
dma_fence *fence)
 static const char *drm_sched_fence_get_timeline_name(struct dma_fence *f)
 {
        struct drm_sched_fence *fence = to_drm_sched_fence(f);
-       return (const char *)fence->sched->name;
+
+       /*
+        * Do not dereference fence->sched here: a userspace-held finished
+        * fence can outlive a per-context scheduler. Return the name cached
+        * in drm_sched_fence_init() instead.
+        */
+       return fence->sched_name;
 }
 
 static void drm_sched_fence_free_rcu(struct rcu_head *rcu)
@@ -228,6 +234,14 @@ void drm_sched_fence_init(struct drm_sched_fence *fence,
        unsigned seq;
 
        fence->sched = entity->rq->sched;
+       /*
+        * Cache the scheduler's timeline name. The finished fence may be
+        * exported to userspace and outlive @sched (per-context schedulers are
+        * freed on context teardown), so get_timeline_name() must not
+        * dereference @sched. The name is required to outlive any exported
+        * fence (see @name in struct drm_sched_init_args).
+        */
+       fence->sched_name = fence->sched->name;
        seq = atomic_inc_return(&entity->fence_seq);
        dma_fence_init(&fence->scheduled, &drm_sched_fence_ops_scheduled,
                       &fence->lock, entity->fence_context, seq);
diff --git a/include/drm/gpu_scheduler.h b/include/drm/gpu_scheduler.h
index 7a64cc11de08..412b8c4643f1 100644
--- a/include/drm/gpu_scheduler.h
+++ b/include/drm/gpu_scheduler.h
@@ -322,6 +322,17 @@ struct drm_sched_fence {
          * belongs to.
          */
        struct drm_gpu_scheduler        *sched;
+       /**
+        * @sched_name: the timeline name of @sched, cached at init time.
+        *
+        * &drm_sched_fence.finished may be exported to userspace (via a
+        * sync_file or drm_syncobj) and can outlive @sched: a driver using a
+        * per-context scheduler frees it on context teardown while a
+        * userspace-held finished fence still references it. The
+        * get_timeline_name() callback must therefore not dereference @sched;
+        * it returns this cached name instead.
+        */
+       const char                      *sched_name;
         /**
          * @lock: the lock used by the scheduled and the finished fences.
          */
@@ -646,7 +657,12 @@ struct drm_gpu_scheduler {
  * @timeout: timeout value in jiffies for submitted jobs.
  * @timeout_wq: workqueue to use for timeout work. If NULL, the system_wq is 
used.
  * @score: score atomic shared with other schedulers. May be NULL.
- * @name: name (typically the driver's name). Used for debugging
+ * @name: name (typically the driver's name). Used for debugging, and as the
+ *     dma-fence timeline name of the scheduler's fences. It must follow the
+ *     dma-fence safe access rules: a &drm_sched_fence.finished exported to
+ *     userspace can outlive the scheduler, so @name has to outlive any such
+ *     fence - use a string literal, or free it only after an RCU grace period
+ *     past the last exported fence. See drm_sched_fence_get_timeline_name().
  * @dev: associated device. Used for debugging
  */
 struct drm_sched_init_args {
-- 
2.43.0

Reply via email to