Currently there are two pathways used to submit a request to the DBC
FIFO. In preparation for using the DRM scheduler, merge these two into
one so that job submission logic is simpler.

Co-developed-by: Pranjal Ramajor Asha Kanojiya <[email protected]>
Signed-off-by: Pranjal Ramajor Asha Kanojiya <[email protected]>
Signed-off-by: Carl Vanderlip <[email protected]>
---
 drivers/accel/qaic/qaic_data.c | 171 +++++++++++++++++----------------
 1 file changed, 89 insertions(+), 82 deletions(-)

diff --git a/drivers/accel/qaic/qaic_data.c b/drivers/accel/qaic/qaic_data.c
index 3326d9da0e1e..0208400d2636 100644
--- a/drivers/accel/qaic/qaic_data.c
+++ b/drivers/accel/qaic/qaic_data.c
@@ -1164,47 +1164,88 @@ static inline u32 fifo_space_avail(u32 head, u32 tail, 
u32 q_size)
        return avail;
 }
 
-static inline int copy_exec_reqs(struct qaic_device *qdev, struct bo_slice 
*slice, u32 dbc_id,
-                                u32 head, u32 *ptail)
+static void submit_reqs_to_fifo(struct dma_bridge_chan *dbc, struct dbc_req 
*reqs,
+                              u32 num_req, u32 head, u32 tail)
 {
-       struct dma_bridge_chan *dbc = &qdev->dbc[dbc_id];
+       u32 num_submit;
+
+       if (tail + num_req > dbc->nelem) {
+               num_submit = dbc->nelem - tail;
+               memcpy(fifo_at(dbc->req_q_base, tail), reqs, sizeof(*reqs) * 
num_submit);
+               reqs += num_submit;
+               num_submit = num_req - num_submit;
+               memcpy(dbc->req_q_base, reqs, sizeof(*reqs) * num_submit);
+       } else {
+               memcpy(fifo_at(dbc->req_q_base, tail), reqs, sizeof(*reqs) * 
num_req);
+       }
+}
+
+/* Caller should be holding dbc->req_lock */
+static inline int qaic_submit_reqs_to_hw(struct bo_slice *slice,
+                                        unsigned int num_req,
+                                        bool is_partial, u32 partial_size)
+{
+       struct dma_bridge_chan *dbc = slice->bo->dbc;
        struct dbc_req *reqs = slice->reqs;
-       u32 tail = *ptail;
-       u32 avail;
+       struct qaic_bo *bo = slice->bo;
+       struct dbc_req *last_req;
+       u32 head, tail, avail;
 
+       qaic_data_get_fifo_info(dbc, &head, &tail);
        avail = fifo_space_avail(head, tail, dbc->nelem);
-       if (avail < slice->nents)
+
+       /* PCI link error */
+       if (head == U32_MAX || tail == U32_MAX)
+               return -ENODEV;
+
+       if (avail < num_req)
                return -EAGAIN;
 
-       if (tail + slice->nents > dbc->nelem) {
-               avail = dbc->nelem - tail;
-               avail = min_t(u32, avail, slice->nents);
-               memcpy(fifo_at(dbc->req_q_base, tail), reqs, sizeof(*reqs) * 
avail);
-               reqs += avail;
-               avail = slice->nents - avail;
-               if (avail)
-                       memcpy(dbc->req_q_base, reqs, sizeof(*reqs) * avail);
-       } else {
-               memcpy(fifo_at(dbc->req_q_base, tail), reqs, sizeof(*reqs) * 
slice->nents);
+       submit_reqs_to_fifo(dbc, reqs, num_req, head, tail);
+
+       if (is_partial) {
+               /*
+                * Copy over the last entry. Here we need to adjust len to the 
left over
+                * size, and set src and dst to the entry it is copied to.
+                */
+               last_req = fifo_at(dbc->req_q_base, (tail + num_req - 1) % 
dbc->nelem);
+               memcpy(last_req, &slice->reqs[slice->nents - 1], 
sizeof(*last_req));
+               last_req->len = cpu_to_le32(partial_size);
+               last_req->src_addr = reqs[num_req - 1].src_addr;
+               last_req->dest_addr = reqs[num_req - 1].dest_addr;
+               last_req->req_id = reqs[num_req - 1].req_id;
+               /* Disable DMA transfer */
+               if (!last_req->len)
+                       last_req->cmd &= ~GENMASK(1, 0);
        }
 
-       *ptail = (tail + slice->nents) % dbc->nelem;
+       tail = (tail + num_req) % dbc->nelem;
 
+       /* Submit_ts will be taken for the last job of this BO */
+       bo->perf_stats.req_submit_ts = ktime_get_ns();
+
+       /* Finalize commit to hardware */
+       dma_sync_sgtable_for_device(&dbc->qdev->pdev->dev, bo->sgt, bo->dir);
+       writel(tail, dbc->dbc_base + REQTP_OFF);
        return 0;
 }
 
-static inline int copy_partial_exec_reqs(struct qaic_device *qdev, struct 
bo_slice *slice,
-                                        u64 resize, struct dma_bridge_chan 
*dbc, u32 head,
-                                        u32 *ptail)
+static inline int copy_exec_reqs(struct bo_slice *slice)
+{
+       int ret;
+
+       ret = qaic_submit_reqs_to_hw(slice, slice->nents, false, 0);
+
+       return ret;
+}
+
+static inline int copy_partial_exec_reqs(struct bo_slice *slice, u64 resize)
 {
        struct dbc_req *reqs = slice->reqs;
-       struct dbc_req *last_req;
-       u32 tail = *ptail;
+       unsigned int nents_xfer;
        u64 last_bytes;
        u32 first_n;
-       u32 avail;
-
-       avail = fifo_space_avail(head, tail, dbc->nelem);
+       int ret;
 
        /*
         * After this for loop is complete, first_n represents the index
@@ -1219,50 +1260,15 @@ static inline int copy_partial_exec_reqs(struct 
qaic_device *qdev, struct bo_sli
                else
                        break;
 
-       if (avail < (first_n + 1))
-               return -EAGAIN;
-
-       if (first_n) {
-               if (tail + first_n > dbc->nelem) {
-                       avail = dbc->nelem - tail;
-                       avail = min_t(u32, avail, first_n);
-                       memcpy(fifo_at(dbc->req_q_base, tail), reqs, 
sizeof(*reqs) * avail);
-                       last_req = reqs + avail;
-                       avail = first_n - avail;
-                       if (avail)
-                               memcpy(dbc->req_q_base, last_req, sizeof(*reqs) 
* avail);
-               } else {
-                       memcpy(fifo_at(dbc->req_q_base, tail), reqs, 
sizeof(*reqs) * first_n);
-               }
-       }
-
-       /*
-        * Copy over the last entry. Here we need to adjust len to the left over
-        * size, and set src and dst to the entry it is copied to.
-        */
-       last_req = fifo_at(dbc->req_q_base, (tail + first_n) % dbc->nelem);
-       memcpy(last_req, reqs + slice->nents - 1, sizeof(*reqs));
-
-       /*
-        * last_bytes holds size of a DMA segment, maximum DMA segment size is
-        * set to UINT_MAX by qaic and hence last_bytes can never exceed u32
-        * range. So, by down sizing we are not corrupting the value.
-        */
-       last_req->len = cpu_to_le32((u32)last_bytes);
-       last_req->src_addr = reqs[first_n].src_addr;
-       last_req->dest_addr = reqs[first_n].dest_addr;
-       if (!last_bytes)
-               /* Disable DMA transfer */
-               last_req->cmd = GENMASK(7, 2) & reqs[first_n].cmd;
+       nents_xfer = first_n + 1;
 
-       *ptail = (tail + first_n + 1) % dbc->nelem;
+       ret = qaic_submit_reqs_to_hw(slice, nents_xfer, true, last_bytes);
 
-       return 0;
+       return ret;
 }
 
 static int send_bo_list_to_device(struct qaic_device *qdev, struct drm_file 
*file_priv,
-                                 struct execute_info *exec, struct 
dma_bridge_chan *dbc,
-                                 u32 head, u32 *tail)
+                                 struct execute_info *exec, struct 
dma_bridge_chan *dbc)
 {
        struct ww_acquire_ctx acquire_ctx;
        struct bo_slice *slice;
@@ -1309,37 +1315,40 @@ static int send_bo_list_to_device(struct qaic_device 
*qdev, struct drm_file *fil
 
                bo->req_id = dbc->next_req_id++;
 
+               spin_lock_irqsave(&dbc->xfer_lock, flags);
+               list_add_tail(&bo->xfer_list, &dbc->xfer_list);
+               spin_unlock_irqrestore(&dbc->xfer_lock, flags);
+
+               ret = qaic_acquire_bo_fence(bo);
+               if (ret)
+                       goto fence_failed;
+
                list_for_each_entry(slice, &bo->slices, slice) {
                        for (j = 0; j < slice->nents; j++)
                                slice->reqs[j].req_id = cpu_to_le16(bo->req_id);
 
                        if (exec->is_partial && (!resize || resize <= 
slice->offset))
                                /* Configure the slice for no DMA transfer */
-                               ret = copy_partial_exec_reqs(qdev, slice, 0, 
dbc, head, tail);
+                               ret = copy_partial_exec_reqs(slice, 0);
                        else if (exec->is_partial && resize < slice->offset + 
slice->size)
                                /* Configure the slice to be partially DMA 
transferred */
-                               ret = copy_partial_exec_reqs(qdev, slice,
-                                                            resize - 
slice->offset, dbc,
-                                                            head, tail);
+                               ret = copy_partial_exec_reqs(slice, resize - 
slice->offset);
                        else
-                               ret = copy_exec_reqs(qdev, slice, dbc->id, 
head, tail);
+                               ret = copy_exec_reqs(slice);
                        if (ret)
-                               goto unlock_bo;
+                               goto submit_failed;
                }
-
-               ret = qaic_acquire_bo_fence(bo);
-               if (ret)
-                       goto unlock_bo;
-
-               spin_lock_irqsave(&dbc->xfer_lock, flags);
-               list_add_tail(&bo->xfer_list, &dbc->xfer_list);
-               spin_unlock_irqrestore(&dbc->xfer_lock, flags);
-               dma_sync_sgtable_for_device(&qdev->pdev->dev, bo->sgt, bo->dir);
                mutex_unlock(&bo->lock);
        }
 
        goto unlock_resv;
 
+submit_failed:
+       qaic_remove_bo_fence(bo);
+fence_failed:
+       spin_lock_irqsave(&dbc->xfer_lock, flags);
+       list_del_init(&bo->xfer_list);
+       spin_unlock_irqrestore(&dbc->xfer_lock, flags);
 unlock_bo:
        drm_gem_object_put(&bo->base);
        mutex_unlock(&bo->lock);
@@ -1536,13 +1545,11 @@ static int __qaic_execute_bo_ioctl(struct drm_device 
*dev, void *data, struct dr
 
        exec.perf_stats.queue_level = head <= tail ? tail - head : dbc->nelem - 
(head - tail);
 
-       ret = send_bo_list_to_device(qdev, file_priv, &exec, dbc, head, &tail);
+       ret = send_bo_list_to_device(qdev, file_priv, &exec, dbc);
        if (ret)
                goto unlock_req_lock;
 
-       /* Finalize commit to hardware */
        exec.perf_stats.submit_ts = ktime_get_ns();
-       writel(tail, dbc->dbc_base + REQTP_OFF);
        mutex_unlock(&dbc->req_lock);
 
        update_profiling_data(&exec);
-- 
2.43.0

Reply via email to