Move per-exec ioctl information into common data structure in preparation for adding dma-resv fences which require all BO's associated with a fence to be locked with drm_gem_lock_reservations().
Co-developed-by: Pranjal Ramajor Asha Kanojiya <[email protected]> Signed-off-by: Pranjal Ramajor Asha Kanojiya <[email protected]> Signed-off-by: Carl Vanderlip <[email protected]> --- drivers/accel/qaic/qaic_data.c | 261 ++++++++++++++++++++++++--------- 1 file changed, 190 insertions(+), 71 deletions(-) diff --git a/drivers/accel/qaic/qaic_data.c b/drivers/accel/qaic/qaic_data.c index 4d55531bf1c9..64bc3cc50d1f 100644 --- a/drivers/accel/qaic/qaic_data.c +++ b/drivers/accel/qaic/qaic_data.c @@ -143,6 +143,47 @@ struct dbc_rsp { __le16 status; } __packed; +/* Common data structure for all the variants of execute IOCTL */ +struct execute_info { + /* Number of entries present in arrays */ + u32 count; + struct { + u64 received_ts; + u64 submit_ts; + u32 queue_level; + } perf_stats; + /* Array of pointers to BOs for this execution */ + struct qaic_bo **bo_arr; + /* Array of handles to above BOs, helpful for identifying failures */ + u32 *handle_arr; + /* Array of new sizes for BO transfer if request is_partial */ + u64 *resize_arr; + /* True if execution is of resized BO */ + bool is_partial; +}; + +/* Helper structure for retrieving info about execution entries */ +struct qaic_exec_data_ent { + /* Pointer to requested gem object */ + struct drm_gem_object *obj; + /* New size for BO transfer if request is partial execute */ + u64 resize; + /* Per-file handle for BO, helpful for identifying failures */ + u32 handle; +}; + +/* Helper structure for requesting info about execution entries */ +struct qaic_exec_data_req { + /* Pointer to drm file */ + struct drm_file *drm_file; + /* Pointer to beginning of array of execution entries from user */ + void *exec_data; + /* Is exec_data an array of partial execution entries? */ + bool is_partial; + /* Index of exec_data array this request is acting upon */ + int idx; +}; + static inline bool bo_queued(struct qaic_bo *bo) { return !list_empty(&bo->xfer_list); @@ -1222,41 +1263,43 @@ static inline int copy_partial_exec_reqs(struct qaic_device *qdev, struct bo_sli } static int send_bo_list_to_device(struct qaic_device *qdev, struct drm_file *file_priv, - struct qaic_execute_entry *exec, unsigned int count, - bool is_partial, struct dma_bridge_chan *dbc, u32 head, - u32 *tail) + struct execute_info *exec, struct dma_bridge_chan *dbc, + u32 head, u32 *tail) { - struct qaic_partial_execute_entry *pexec = (struct qaic_partial_execute_entry *)exec; - struct drm_gem_object *obj; + struct ww_acquire_ctx acquire_ctx; struct bo_slice *slice; unsigned long flags; struct qaic_bo *bo; + u64 resize; + u32 handle; int i, j; int ret; - for (i = 0; i < count; i++) { - /* - * ref count will be decremented when the transfer of this - * buffer is complete. It is inside dbc_irq_threaded_fn(). - */ - obj = drm_gem_object_lookup(file_priv, - is_partial ? pexec[i].handle : exec[i].handle); - if (!obj) { - ret = -ENOENT; - goto failed_to_send_bo; - } + ret = drm_gem_lock_reservations((struct drm_gem_object **)exec->bo_arr, + exec->count, &acquire_ctx); + if (ret) + return ret; - bo = to_qaic_bo(obj); + for (i = 0; i < exec->count; i++) { + resize = exec->is_partial ? exec->resize_arr[i] : 0; + bo = exec->bo_arr[i]; + handle = exec->handle_arr[i]; ret = mutex_lock_interruptible(&bo->lock); if (ret) goto failed_to_send_bo; + /* + * Take reference to object before sending to device, + * released when transfer is complete (in dbc_irq_threaded_fn()). + */ + drm_gem_object_get(&bo->base); + if (!bo->sliced) { ret = -EINVAL; goto unlock_bo; } - if (is_partial && pexec[i].resize > bo->base.size) { + if (exec->is_partial && resize > bo->base.size) { ret = -EINVAL; goto unlock_bo; } @@ -1274,13 +1317,13 @@ static int send_bo_list_to_device(struct qaic_device *qdev, struct drm_file *fil for (j = 0; j < slice->nents; j++) slice->reqs[j].req_id = cpu_to_le16(bo->req_id); - if (is_partial && (!pexec[i].resize || pexec[i].resize <= slice->offset)) + if (exec->is_partial && (!resize || resize <= slice->offset)) /* Configure the slice for no DMA transfer */ ret = copy_partial_exec_reqs(qdev, slice, 0, dbc, head, tail); - else if (is_partial && pexec[i].resize < slice->offset + slice->size) + else if (exec->is_partial && resize < slice->offset + slice->size) /* Configure the slice to be partially DMA transferred */ ret = copy_partial_exec_reqs(qdev, slice, - pexec[i].resize - slice->offset, dbc, + resize - slice->offset, dbc, head, tail); else ret = copy_exec_reqs(qdev, slice, dbc->id, head, tail); @@ -1296,82 +1339,124 @@ static int send_bo_list_to_device(struct qaic_device *qdev, struct drm_file *fil mutex_unlock(&bo->lock); } - return 0; + goto unlock_resv; unlock_bo: + drm_gem_object_put(&bo->base); mutex_unlock(&bo->lock); failed_to_send_bo: - if (likely(obj)) - drm_gem_object_put(obj); for (j = 0; j < i; j++) { + drm_gem_object_put(&exec->bo_arr[j]->base); spin_lock_irqsave(&dbc->xfer_lock, flags); bo = list_last_entry(&dbc->xfer_list, struct qaic_bo, xfer_list); - obj = &bo->base; list_del_init(&bo->xfer_list); spin_unlock_irqrestore(&dbc->xfer_lock, flags); dma_sync_sgtable_for_cpu(&qdev->pdev->dev, bo->sgt, bo->dir); - drm_gem_object_put(obj); } +unlock_resv: + drm_gem_unlock_reservations((struct drm_gem_object **)exec->bo_arr, + exec->count, &acquire_ctx); return ret; } -static void update_profiling_data(struct drm_file *file_priv, - struct qaic_execute_entry *exec, unsigned int count, - bool is_partial, u64 received_ts, u64 submit_ts, u32 queue_level) +static void update_profiling_data(struct execute_info *exec) { - struct qaic_partial_execute_entry *pexec = (struct qaic_partial_execute_entry *)exec; - struct drm_gem_object *obj; + u32 queue_level = exec->perf_stats.queue_level; struct qaic_bo *bo; int i; - for (i = 0; i < count; i++) { - /* - * Since we already committed the BO to hardware, the only way - * this should fail is a pending signal. We can't cancel the - * submit to hardware, so we have to just skip the profiling - * data. In case the signal is not fatal to the process, we - * return success so that the user doesn't try to resubmit. - */ - obj = drm_gem_object_lookup(file_priv, - is_partial ? pexec[i].handle : exec[i].handle); - if (!obj) - break; - bo = to_qaic_bo(obj); - bo->perf_stats.req_received_ts = received_ts; - bo->perf_stats.req_submit_ts = submit_ts; + for (i = 0; i < exec->count; i++) { + bo = exec->bo_arr[i]; + bo->perf_stats.req_received_ts = exec->perf_stats.received_ts; + bo->perf_stats.req_submit_ts = exec->perf_stats.submit_ts; bo->perf_stats.queue_level_before = queue_level; queue_level += bo->total_slice_nents; - drm_gem_object_put(obj); } } +static int exec_index_to_gem_info(struct qaic_exec_data_req req, struct qaic_exec_data_ent *ent) +{ + int idx = req.idx; + int ret = 0; + u32 handle; + + if (req.is_partial) { + struct qaic_partial_execute_entry *exec_ent_part = req.exec_data; + + handle = exec_ent_part[idx].handle; + ent->resize = exec_ent_part[idx].resize; + } else { + struct qaic_execute_entry *exec_ent = req.exec_data; + + handle = exec_ent[idx].handle; + } + + ent->obj = drm_gem_object_lookup(req.drm_file, handle); + if (!ent->obj) + ret = -ENOENT; + + ent->handle = handle; + + return ret; +} + +static int lookup_exec_data(struct drm_file *file_priv, void *exec_data, + struct execute_info *exec) +{ + struct qaic_exec_data_req req; + struct qaic_exec_data_ent ent; + int i, ret = 0; + + req.drm_file = file_priv; + req.exec_data = exec_data; + req.is_partial = exec->is_partial; + + for (i = 0; i < exec->count; i++) { + req.idx = i; + ret = exec_index_to_gem_info(req, &ent); + if (ret) + goto put_obj; + + exec->bo_arr[i] = to_qaic_bo(ent.obj); + exec->handle_arr[i] = ent.handle; + if (exec->resize_arr) + exec->resize_arr[i] = ent.resize; + } + + return ret; +put_obj: + for (i--; i >= 0; i--) + drm_gem_object_put(&exec->bo_arr[i]->base); + return ret; +} + static int __qaic_execute_bo_ioctl(struct drm_device *dev, void *data, struct drm_file *file_priv, bool is_partial) { + size_t usr_ent_size = is_partial ? sizeof(struct qaic_partial_execute_entry) : + sizeof(struct qaic_execute_entry); + int usr_rcu_id, qdev_rcu_id, ch_rcu_id; struct qaic_execute *args = data; - struct qaic_execute_entry *exec; struct dma_bridge_chan *dbc; - int usr_rcu_id, qdev_rcu_id; struct qaic_device *qdev; + struct execute_info exec; struct qaic_user *usr; - u64 received_ts; - u32 queue_level; - u64 submit_ts; - int rcu_id; - u32 head; - u32 tail; - u64 size; - int ret; + size_t elem_size_sum; + void *usr_exec_ent; + u32 head, tail; + int i, ret; - received_ts = ktime_get_ns(); + exec.perf_stats.received_ts = ktime_get_ns(); + exec.is_partial = is_partial; - size = is_partial ? sizeof(struct qaic_partial_execute_entry) : sizeof(*exec); if (args->hdr.count == 0) return -EINVAL; - exec = memdup_array_user(u64_to_user_ptr(args->data), args->hdr.count, size); - if (IS_ERR(exec)) - return PTR_ERR(exec); + exec.count = args->hdr.count; + + usr_exec_ent = memdup_array_user(u64_to_user_ptr(args->data), exec.count, usr_ent_size); + if (IS_ERR(usr_exec_ent)) + return PTR_ERR(usr_exec_ent); usr = file_priv->driver_priv; usr_rcu_id = srcu_read_lock(&usr->qddev_lock); @@ -1394,7 +1479,39 @@ static int __qaic_execute_bo_ioctl(struct drm_device *dev, void *data, struct dr dbc = &qdev->dbc[args->hdr.dbc_id]; - rcu_id = srcu_read_lock(&dbc->ch_lock); + /* + * Going to allocate one large array, and then create pointers to + * within it for sub-arrays that all have exec.count elements in them. + * + * bo_arr is an array of pointers to struct qaic_bo, but is used as an + * argument for drm_gem_lock_reservations(). + * + * handle_arr is an array of handles to each drm_gem_object embedded in + * the respective bo_arr element. It is used to help debugging by + * identifying particular buffers. + * + * resize_arr is an array of u64 sizes for partial executions. It is + * only defined if a partial execution is underway. + */ + elem_size_sum = sizeof(*exec.bo_arr) + sizeof(*exec.handle_arr); + elem_size_sum += is_partial ? sizeof(*exec.resize_arr) : 0; + exec.bo_arr = kcalloc(exec.count, elem_size_sum, GFP_KERNEL); + if (!exec.bo_arr) { + ret = -ENOMEM; + goto unlock_dev_srcu; + } + exec.handle_arr = ((void *)exec.bo_arr) + (sizeof(*exec.bo_arr) * exec.count); + if (is_partial) + exec.resize_arr = ((void *)exec.bo_arr) + + ((sizeof(*exec.bo_arr) + sizeof(*exec.handle_arr)) * exec.count); + else + exec.resize_arr = NULL; + + ret = lookup_exec_data(file_priv, usr_exec_ent, &exec); + if (ret) + goto free_bo_arr; + + ch_rcu_id = srcu_read_lock(&dbc->ch_lock); if (!dbc->usr || dbc->usr->handle != usr->handle) { ret = -EPERM; goto release_ch_rcu; @@ -1418,20 +1535,18 @@ static int __qaic_execute_bo_ioctl(struct drm_device *dev, void *data, struct dr goto unlock_req_lock; } - queue_level = head <= tail ? tail - head : dbc->nelem - (head - tail); + exec.perf_stats.queue_level = head <= tail ? tail - head : dbc->nelem - (head - tail); - ret = send_bo_list_to_device(qdev, file_priv, exec, args->hdr.count, is_partial, dbc, - head, &tail); + ret = send_bo_list_to_device(qdev, file_priv, &exec, dbc, head, &tail); if (ret) goto unlock_req_lock; /* Finalize commit to hardware */ - submit_ts = ktime_get_ns(); + exec.perf_stats.submit_ts = ktime_get_ns(); writel(tail, dbc->dbc_base + REQTP_OFF); mutex_unlock(&dbc->req_lock); - update_profiling_data(file_priv, exec, args->hdr.count, is_partial, received_ts, - submit_ts, queue_level); + update_profiling_data(&exec); if (datapath_polling) schedule_work(&dbc->poll_work); @@ -1440,12 +1555,16 @@ static int __qaic_execute_bo_ioctl(struct drm_device *dev, void *data, struct dr if (ret) mutex_unlock(&dbc->req_lock); release_ch_rcu: - srcu_read_unlock(&dbc->ch_lock, rcu_id); + srcu_read_unlock(&dbc->ch_lock, ch_rcu_id); + for (i = 0; i < exec.count; i++) + drm_gem_object_put(&exec.bo_arr[i]->base); +free_bo_arr: + kfree(exec.bo_arr); unlock_dev_srcu: srcu_read_unlock(&qdev->dev_lock, qdev_rcu_id); unlock_usr_srcu: srcu_read_unlock(&usr->qddev_lock, usr_rcu_id); - kfree(exec); + kfree(usr_exec_ent); return ret; } base-commit: 0915fb19e08f7f14a03df78c67d081bbfa8ff1fa -- 2.43.0
