A global perfmon counts concurrent activity from every job, so no
dependency is added while one is set. Skip the accumulator for the jobs
carrying no perfmon while a global perfmon is enabled, which is every job
in the common case.

Jobs carrying their own perfmon stay tracked. The global perfmon may be
cleared before such a job runs, and the next job carrying a perfmon has
to wait for it, otherwise both are in flight at once and only one gets
programmed in the HW.

Reviewed-by: Iago Toral Quiroga <[email protected]>
Signed-off-by: Maíra Canal <[email protected]>
---
 drivers/gpu/drm/v3d/v3d_drv.h    |  3 ++-
 drivers/gpu/drm/v3d/v3d_submit.c | 18 ++++++++----------
 2 files changed, 10 insertions(+), 11 deletions(-)

diff --git a/drivers/gpu/drm/v3d/v3d_drv.h b/drivers/gpu/drm/v3d/v3d_drv.h
index 02b6ab7dc72b..d40610a34db9 100644
--- a/drivers/gpu/drm/v3d/v3d_drv.h
+++ b/drivers/gpu/drm/v3d/v3d_drv.h
@@ -201,7 +201,8 @@ struct v3d_dev {
                 * queue, which is used so that a new perfmon-carrying job can
                 * depend on every job currently in-flight across all queues.
                 *
-                * Finished fences are only tracked if @nperfmons > 0.
+                * Finished fences are only tracked if @nperfmons > 0 and no
+                * global perfmon is set.
                 */
                struct dma_fence *last_hw_fence[V3D_MAX_QUEUES];
        } perfmon_state;
diff --git a/drivers/gpu/drm/v3d/v3d_submit.c b/drivers/gpu/drm/v3d/v3d_submit.c
index bc3c43fd4fd9..7548faab67ba 100644
--- a/drivers/gpu/drm/v3d/v3d_submit.c
+++ b/drivers/gpu/drm/v3d/v3d_submit.c
@@ -359,15 +359,15 @@ v3d_attach_perfmon_to_jobs(struct v3d_submit *submit, u32 
perfmon_id)
  * track concurrent activity from all jobs.
  *
  * Keeping track of the in-flight jobs costs a fence merge per job, so it is
- * only done while at least one perfmon is alive. Jobs submitted while no
- * perfmon exists go untracked and may overlap the first measured job.
+ * only done while at least one perfmon is alive, and is skipped for the jobs
+ * carrying no perfmon while a global perfmon is set. Jobs submitted while
+ * tracking is off go untracked and may overlap the first measured job.
  */
 static int
 v3d_serialize_for_perfmon(struct v3d_job *job)
 {
        struct v3d_dev *v3d = job->v3d;
        struct dma_fence *merged;
-       bool is_global_perfmon;
        int ret;
 
        lockdep_assert_held(&v3d->sched_lock);
@@ -375,11 +375,10 @@ v3d_serialize_for_perfmon(struct v3d_job *job)
        if (!atomic_read(&v3d->perfmon_state.nperfmons))
                return 0;
 
-       scoped_guard(spinlock_irqsave, &v3d->perfmon_state.lock)
-               is_global_perfmon = !!v3d->global_perfmon;
-
-       if (is_global_perfmon)
-               goto publish;
+       scoped_guard(spinlock_irqsave, &v3d->perfmon_state.lock) {
+               if (!job->perfmon && v3d->global_perfmon)
+                       return 0;
+       }
 
        if (job->perfmon) {
                for (enum v3d_queue q = 0; q < V3D_MAX_QUEUES; q++) {
@@ -400,7 +399,6 @@ v3d_serialize_for_perfmon(struct v3d_job *job)
                        return ret;
        }
 
-publish:
        /*
         * Accumulate every in-flight job on this queue into one merged fence.
         * A HW queue is fed by several scheduler entities (one per-fd), so jobs
@@ -414,7 +412,7 @@ v3d_serialize_for_perfmon(struct v3d_job *job)
        dma_fence_put(v3d->perfmon_state.last_hw_fence[job->queue]);
        v3d->perfmon_state.last_hw_fence[job->queue] = merged;
 
-       if (job->perfmon && !is_global_perfmon) {
+       if (job->perfmon) {
                dma_fence_put(v3d->perfmon_state.fence);
                v3d->perfmon_state.fence = dma_fence_get(job->done_fence);
        }

-- 
2.55.0

Reply via email to