vc4_irq_finish_render_job() signals the job's fence, wakes the waiters, and queues the job done work while holding job_lock. This increases the job_lock contention as dma_fence_signal() takes the fence lock and runs every callback attached to the fence. So, job_lock is held, with interrupts disabled, for as long as those callbacks take.
Considering that these functions are unrelated to the job lists, hand the fence out of the critical section and signal it, wake the waiters and queue the work once job_lock is dropped. In order to split the end of the render job and the fence signalling, take a reference before handing it over, as the job done worker is free to release the exec as soon as the lock is dropped. On a Raspberry Pi 3, LOCK_STAT shows this section holding job_lock in about three quarters of its contentions, showing the relevance of reducing this critical section. Signed-off-by: Maíra Canal <[email protected]> --- drivers/gpu/drm/vc4/vc4_irq.c | 28 ++++++++++++++++++++++------ 1 file changed, 22 insertions(+), 6 deletions(-) diff --git a/drivers/gpu/drm/vc4/vc4_irq.c b/drivers/gpu/drm/vc4/vc4_irq.c index f87550f3e55d..dcee4a1e17ad 100644 --- a/drivers/gpu/drm/vc4/vc4_irq.c +++ b/drivers/gpu/drm/vc4/vc4_irq.c @@ -152,7 +152,7 @@ vc4_cancel_bin_job(struct drm_device *dev) vc4_submit_next_bin_job(dev); } -static void +static struct dma_fence * vc4_irq_finish_render_job(struct drm_device *dev) { struct vc4_dev *vc4 = to_vc4_dev(dev); @@ -160,7 +160,7 @@ vc4_irq_finish_render_job(struct drm_device *dev) struct vc4_exec_info *nextbin, *nextrender; if (!exec) - return; + return NULL; trace_vc4_rcl_end_irq(dev, exec->seqno); @@ -193,8 +193,17 @@ vc4_irq_finish_render_job(struct drm_device *dev) else if (nextbin && nextbin->perfmon != exec->perfmon) vc4_submit_next_bin_job(dev); - if (exec->fence) - dma_fence_signal(exec->fence); + return dma_fence_get(exec->fence); +} + +static void +vc4_irq_render_job_done(struct vc4_dev *vc4, struct dma_fence *fence) +{ + if (!fence) + return; + + dma_fence_signal(fence); + dma_fence_put(fence); wake_up_all(&vc4->job_wait_queue); schedule_work(&vc4->job_done_work); @@ -233,9 +242,13 @@ vc4_irq(int irq, void *arg) } if (intctl & V3D_INT_FRDONE) { + struct dma_fence *fence; + spin_lock(&vc4->job_lock); - vc4_irq_finish_render_job(dev); + fence = vc4_irq_finish_render_job(dev); spin_unlock(&vc4->job_lock); + + vc4_irq_render_job_done(vc4, fence); status = IRQ_HANDLED; } @@ -328,6 +341,7 @@ void vc4_irq_uninstall(struct drm_device *dev) void vc4_irq_reset(struct drm_device *dev) { struct vc4_dev *vc4 = to_vc4_dev(dev); + struct dma_fence *fence; unsigned long irqflags; if (WARN_ON_ONCE(vc4->gen > VC4_GEN_4)) @@ -346,6 +360,8 @@ void vc4_irq_reset(struct drm_device *dev) spin_lock_irqsave(&vc4->job_lock, irqflags); vc4_cancel_bin_job(dev); - vc4_irq_finish_render_job(dev); + fence = vc4_irq_finish_render_job(dev); spin_unlock_irqrestore(&vc4->job_lock, irqflags); + + vc4_irq_render_job_done(vc4, fence); } -- 2.55.0
