+ Christian's e-mail which he reads :)

On 10/09/2026 08:59, Philipp Stanner wrote:
The entity->last_scheduled field has always been set and read with
special RCU functions in addition to memory barriers.

This was added in

commit 70102d77ff22 ("drm/scheduler: add drm_sched_entity_error and use rcu for 
last_scheduled")

however, no proper justification for that mechanism was provided. There
seems to be no obvious reason, since the entity lock is available and
taken at all places that evaluate the last_scheduled field. The only
exception is drm_sched_entity_error(), which is not performance critical
in any way.

FWIW the entry paths that I found from the amdgpu side are from the job submission side where it is called a few times for each job via amdgpu_ctx_get_entity(), amdgpu_job_prepare_job() and amdgpu_vm_generation(). Last one being both in amdgpu_job_alloc() and amdgpu_job_run(). If I did not miss any, the change effectively adds four lock-unlock cycles to the job submission path. Plus one more for schedulers which use load balancing in some cases.

I don't know if avoiding that was the reason RCU was chosen, perhaps Christian remembers.

Also, to repeat what I said before, I benchmarked this a bit under extreme circumstances (4x non vsynced vkgears) and did not find a regression in fps.

Improve robustness, readability and maintainability by replacing RCU and
barriers with the lock.

Signed-off-by: Philipp Stanner <[email protected]>
---
  drivers/gpu/drm/scheduler/sched_entity.c | 50 ++++++++++--------------
  include/drm/gpu_scheduler.h              | 10 ++---
  2 files changed, 24 insertions(+), 36 deletions(-)

diff --git a/drivers/gpu/drm/scheduler/sched_entity.c 
b/drivers/gpu/drm/scheduler/sched_entity.c
index e168f445f2ab..fa4a9388e5c9 100644
--- a/drivers/gpu/drm/scheduler/sched_entity.c
+++ b/drivers/gpu/drm/scheduler/sched_entity.c
@@ -136,7 +136,6 @@ int drm_sched_entity_init(struct drm_sched_entity *entity,
                              DRM_SCHED_PRIORITY_KERNEL : priority;
        entity->num_sched_list = num_sched_list;
        entity->sched_list = num_sched_list > 1 ? sched_list : NULL;
-       RCU_INIT_POINTER(entity->last_scheduled, NULL);
        RB_CLEAR_NODE(&entity->rb_tree_node);
if (!sched_list[0]->sched_rq) {
@@ -233,10 +232,10 @@ int drm_sched_entity_error(struct drm_sched_entity 
*entity)
        struct dma_fence *fence;
        int r;
- rcu_read_lock();
-       fence = rcu_dereference(entity->last_scheduled);
+       spin_lock(&entity->lock);
+       fence = entity->last_scheduled;
        r = fence ? fence->error : 0;
-       rcu_read_unlock();
+       spin_unlock(&entity->lock);
return r;
  }
@@ -319,9 +318,10 @@ void drm_sched_entity_kill(struct drm_sched_entity *entity)
        /* Make sure this entity is not used by the scheduler at the moment */
        wait_for_completion(&entity->entity_idle);
- /* The entity is guaranteed to not be used by the scheduler */
-       prev = rcu_dereference_check(entity->last_scheduled, true);
+       spin_lock(&entity->lock);
+       prev = entity->last_scheduled;
        dma_fence_get(prev);

Here you can compact with the idiomatic:

 prev = dma_fence_get(entity->last_scheduled);

Otherwise looks plausible to me. Lets see what sashiko will have to say this time round.

Regards,

Tvrtko

+       spin_unlock(&entity->lock);
        while ((job = drm_sched_entity_queue_pop(entity))) {
                struct drm_sched_fence *s_fence = job->s_fence;
@@ -413,8 +413,7 @@ void drm_sched_entity_fini(struct drm_sched_entity *entity)
                entity->dependency = NULL;
        }
- dma_fence_put(rcu_dereference_check(entity->last_scheduled, true));
-       RCU_INIT_POINTER(entity->last_scheduled, NULL);
+       dma_fence_put(entity->last_scheduled);
        drm_sched_entity_stats_put(entity->stats);
  }
  EXPORT_SYMBOL(drm_sched_entity_fini);
@@ -536,6 +535,10 @@ drm_sched_job_dependency(struct drm_sched_job *job,
struct drm_sched_job *drm_sched_entity_pop_job(struct drm_sched_entity *entity)
  {
+       /* Helper to avoid dropping the reference while the entity lock is held,
+        * just to have some more robustness.
+        */
+       struct dma_fence *prev_last_scheduled;
        struct drm_sched_job *sched_job;
sched_job = drm_sched_entity_queue_peek(entity);
@@ -552,22 +555,15 @@ struct drm_sched_job *drm_sched_entity_pop_job(struct 
drm_sched_entity *entity)
        if (entity->guilty && atomic_read(entity->guilty))
                dma_fence_set_error(&sched_job->s_fence->finished, -ECANCELED);
- dma_fence_put(rcu_dereference_check(entity->last_scheduled, true));
-       rcu_assign_pointer(entity->last_scheduled,
-                          dma_fence_get(&sched_job->s_fence->finished));
-
-       /*
-        * If the queue is empty we allow drm_sched_entity_select_rq() to
-        * locklessly access ->last_scheduled. This only works if we set the
-        * pointer before we dequeue and if we a write barrier here.
-        */
-       smp_wmb();
-
        spin_lock(&entity->lock);
+       prev_last_scheduled = entity->last_scheduled;
+       entity->last_scheduled = dma_fence_get(&sched_job->s_fence->finished);
        spsc_queue_pop(&entity->job_queue);
        drm_sched_rq_pop_entity(entity);
        spin_unlock(&entity->lock);
+ dma_fence_put(prev_last_scheduled);
+
        /* Jobs and entities might have different lifecycles. Since we're
         * removing the job from the entities queue, set the jobs entity pointer
         * to NULL to prevent any future access of the entity through this job.
@@ -591,21 +587,15 @@ void drm_sched_entity_select_rq(struct drm_sched_entity 
*entity)
        if (spsc_queue_count(&entity->job_queue))
                return;
- /*
-        * Only when the queue is empty are we guaranteed that
-        * drm_sched_run_job_work() cannot change entity->last_scheduled. To
-        * enforce ordering we need a read barrier here. See
-        * drm_sched_entity_pop_job() for the other side.
-        */
-       smp_rmb();
-
-       fence = rcu_dereference_check(entity->last_scheduled, true);
+       spin_lock(&entity->lock);
+       fence = entity->last_scheduled;
/* stay on the same engine if the previous job hasn't finished */
-       if (fence && !dma_fence_is_signaled(fence))
+       if (fence && !dma_fence_is_signaled(fence)) {
+               spin_unlock(&entity->lock);
                return;
+       }
- spin_lock(&entity->lock);
        sched = drm_sched_pick_best(entity->sched_list, entity->num_sched_list);
        rq = sched ? sched->sched_rq[entity->rq_priority] : NULL;
        if (rq != entity->rq) {
diff --git a/include/drm/gpu_scheduler.h b/include/drm/gpu_scheduler.h
index 7a64cc11de08..aeeea6efc623 100644
--- a/include/drm/gpu_scheduler.h
+++ b/include/drm/gpu_scheduler.h
@@ -100,8 +100,8 @@ struct drm_sched_entity {
         * @lock:
         *
         * Lock protecting the run-queue (@rq) to which this entity belongs,
-        * @priority, the list of schedulers (@sched_list, @num_sched_list) and
-        * the @rr_ts field.
+        * @priority, @last_scheduled and the list of schedulers (@sched_list,
+        * @num_sched_list).
         */
        spinlock_t                      lock;
@@ -215,11 +215,9 @@ struct drm_sched_entity {
        /**
         * @last_scheduled:
         *
-        * Points to the finished fence of the last scheduled job. Only written
-        * by drm_sched_entity_pop_job(). Can be accessed locklessly from
-        * drm_sched_job_arm() if the queue is empty.
+        * Points to the finished fence of the last scheduled job.
         */
-       struct dma_fence __rcu          *last_scheduled;
+       struct dma_fence                *last_scheduled;
/**
         * @last_user: last group leader pushing a job into the entity.

Reply via email to