Replace current FIFO submission process with DRM scheduler. Leverage
scheduler entity credit system to prevent overflow.

Since the network's FIFO size is unknown at device/scheduler creation, a
large fixed credit value is used (65536). During network activation, a
scaling ratio is calculated to map the real FIFO size to the fixed
credit limit. If the ratio would be a fraction, it is rounded up;
limiting FIFO utilization, but preventing overuse.

Co-developed-by: Pranjal Ramajor Asha Kanojiya <[email protected]>
Signed-off-by: Pranjal Ramajor Asha Kanojiya <[email protected]>
Co-developed-by: Youssef Samir <[email protected]>
Signed-off-by: Youssef Samir <[email protected]>
Signed-off-by: Carl Vanderlip <[email protected]>
---
 drivers/accel/qaic/Kconfig        |   1 +
 drivers/accel/qaic/Makefile       |   1 +
 drivers/accel/qaic/qaic.h         | 159 +++++++++++++--
 drivers/accel/qaic/qaic_control.c |   5 +
 drivers/accel/qaic/qaic_data.c    | 315 ++++++++++--------------------
 drivers/accel/qaic/qaic_debugfs.c |   4 +-
 drivers/accel/qaic/qaic_drv.c     |  14 +-
 drivers/accel/qaic/qaic_fence.c   | 101 ++++++----
 drivers/accel/qaic/qaic_sched.c   | 236 ++++++++++++++++++++++
 include/uapi/drm/qaic_accel.h     |   2 +-
 10 files changed, 563 insertions(+), 275 deletions(-)
 create mode 100644 drivers/accel/qaic/qaic_sched.c

diff --git a/drivers/accel/qaic/Kconfig b/drivers/accel/qaic/Kconfig
index 116e42d152ca..d43c06394bb0 100644
--- a/drivers/accel/qaic/Kconfig
+++ b/drivers/accel/qaic/Kconfig
@@ -10,6 +10,7 @@ config DRM_ACCEL_QAIC
        depends on MHI_BUS
        select CRC32
        select WANT_DEV_COREDUMP
+       select DRM_SCHED
        help
          Enables driver for Qualcomm's Cloud AI accelerator PCIe cards that are
          designed to accelerate Deep Learning inference workloads.
diff --git a/drivers/accel/qaic/Makefile b/drivers/accel/qaic/Makefile
index 90582b0693da..092f962e71d3 100644
--- a/drivers/accel/qaic/Makefile
+++ b/drivers/accel/qaic/Makefile
@@ -12,6 +12,7 @@ qaic-y := \
        qaic_drv.o \
        qaic_fence.o \
        qaic_ras.o \
+       qaic_sched.o \
        qaic_ssr.o \
        qaic_sysfs.o \
        qaic_timesync.o \
diff --git a/drivers/accel/qaic/qaic.h b/drivers/accel/qaic/qaic.h
index 4cdd6579aa89..40c7492ea121 100644
--- a/drivers/accel/qaic/qaic.h
+++ b/drivers/accel/qaic/qaic.h
@@ -18,12 +18,14 @@
 #include <linux/workqueue.h>
 #include <drm/drm_device.h>
 #include <drm/drm_gem.h>
+#include <drm/gpu_scheduler.h>
 
 #define QAIC_NAME              "qaic"
 #define QAIC_DBC_BASE          SZ_128K
 #define QAIC_DBC_SIZE          SZ_4K
 #define QAIC_SSR_DBC_SENTINEL  U32_MAX /* No ongoing SSR sentinel */
-#define QAIC_TIMELINE_LEN      32 /* Should fit max sized DBC id/BO handle */
+#define QAIC_TIMELINE_LEN      32 /* Should fit max sized DBC id/Job id */
+#define QAIC_DBC_NAME_LEN      8
 
 #define QAIC_NO_PARTITION      -1
 
@@ -34,6 +36,7 @@
 #define to_drm(qddev) (&(qddev)->drm)
 #define to_accel_kdev(qddev) (to_drm(qddev)->accel->kdev) /* Return Linux 
device of accel node */
 #define to_qaic_device(dev) (to_qaic_drm_device((dev))->qdev)
+#define to_qaic_job(job) container_of((job), struct qaic_job, base)
 
 enum aic_families {
        FAMILY_AIC100,
@@ -68,6 +71,32 @@ enum dbc_states {
 
 extern bool datapath_polling;
 
+struct qaic_job {
+       /*
+        * Base job structure, internal to DRM.
+        * Do not update any fields in base directly in this driver
+        */
+       struct drm_sched_job    base;
+       /* QAIC GEM object this job is referencing */
+       struct qaic_bo          *bo;
+       /* Slice this job is executing */
+       struct bo_slice         *slice;
+       /* DBC associated with this job */
+       struct dma_bridge_chan  *dbc;
+       /* QAIC will signal this fence after execution of this job */
+       struct qaic_fence       *irq_fence;
+       /* Node of list of jobs associated with a DBC */
+       struct list_head        queue;
+       struct kref             ref_count;
+       /* Number of requests this job executes */
+       unsigned int            num_req;
+       /* True if this job is partially executed */
+       bool                    partial;
+       u32                     partial_size;
+       /* Identifier for this job */
+       u16                     id;
+};
+
 struct qaic_user {
        /* Uniquely identifies this user for the device */
        int                     handle;
@@ -86,8 +115,8 @@ struct dma_bridge_chan {
        struct qaic_device      *qdev;
        /* ID of this DMA bridge channel(DBC) */
        unsigned int            id;
-       /* Synchronizes access to xfer_list */
-       spinlock_t              xfer_lock;
+       /* Synchronizes access to job queue */
+       spinlock_t              job_q_lock;
        /* Base address of request queue */
        void                    *req_q_base;
        /* Base address of response queue */
@@ -108,7 +137,7 @@ struct dma_bridge_chan {
         * memory handle can enqueue more than one request elements, all
         * this requests that belong to same memory handle have same request ID
         */
-       u16                     next_req_id;
+       atomic_t                next_req_id;
        /* true: DBC is in use; false: DBC not in use */
        bool                    in_use;
        /*
@@ -118,8 +147,6 @@ struct dma_bridge_chan {
        void __iomem            *dbc_base;
        /* Synchronizes access to Request queue's head and tail pointer */
        struct mutex            req_lock;
-       /* Head of list where each node is a memory handle queued in request 
queue */
-       struct list_head        xfer_list;
        /* Synchronizes DBC readers during cleanup */
        struct srcu_struct      ch_lock;
        /*
@@ -135,6 +162,23 @@ struct dma_bridge_chan {
        struct work_struct      poll_work;
        /* Represents various states of this DBC from enum dbc_states */
        unsigned int            state;
+       /*
+        * Name of this DBC. Format is "dbcXXX" where XXX is DBC id
+        * with leading 0's
+        */
+       char                            name[QAIC_DBC_NAME_LEN];
+       /* Job scheduler for this DBC */
+       struct drm_gpu_scheduler        sched;
+       /* Custom DRM scheduler workqueue for multi-device performance */
+       struct workqueue_struct         *submit_wq;
+       /* Scheduler entity for this DBC */
+       struct drm_sched_entity         sched_entity;
+       /* List head for jobs scheduled for execution */
+       struct list_head                job_queue;
+       /* Ratio to scale FIFO length to scheduler credit constant */
+       u32                             credit_ratio;
+       /* Total number of requests that have been queued to this DBC */
+       u32                             pending_reqs;
 };
 
 struct qaic_device {
@@ -238,6 +282,8 @@ struct qaic_fence {
        char                    timeline_name[QAIC_TIMELINE_LEN];
        /* Cached value of dbc_id for use in populating timeline_name */
        u32                     dbc_id;
+       /* ID of associated Job */
+       u16                     job_id;
 };
 
 struct qaic_bo {
@@ -257,24 +303,15 @@ struct qaic_bo {
        struct dma_bridge_chan  *dbc;
        /* Number of slice that belongs to this buffer */
        u32                     nr_slice;
-       /* Number of slice that have been transferred by DMA engine */
-       u32                     nr_slice_xfer_done;
        /*
         * If true then user has attached slicing information to this BO by
         * calling DRM_IOCTL_QAIC_ATTACH_SLICE_BO ioctl.
         */
        bool                    sliced;
-       /* Request ID of this BO if it is queued for execution */
-       u16                     req_id;
        /* Wait on this fence for DMA transfer of this BO */
-       struct qaic_fence       *fence;
+       struct dma_fence        *fence;
        /* Current fence context for this BO */
        u64                     fence_context;
-       /*
-        * Node in linked list where head is dbc->xfer_list.
-        * This link list contain BO's that are queued for DMA transfer.
-        */
-       struct list_head        xfer_list;
        /*
         * Node in linked list where head is dbc->bo_lists.
         * This link list contain BO's that are associated with the DBC it is
@@ -307,6 +344,8 @@ struct qaic_bo {
        struct mutex            lock;
        /* Callback to sync BO with CPU */
        struct dma_fence_cb     cb;
+       /* True if BO needs to be sync'd with device; otherwise no need */
+       bool                    need_dev_sync;
 };
 
 struct bo_slice {
@@ -331,6 +370,78 @@ struct bo_slice {
        u64                     offset;
 };
 
+struct dbc_req {
+       /*
+        * A request ID is assigned to each memory handle going in DMA queue.
+        * As a single memory handle can enqueue multiple elements in DMA queue
+        * all of them will have the same request ID.
+        */
+       __le16  req_id;
+       /* Future use */
+       __u8    seq_id;
+       /*
+        * Special encoded variable
+        * 7    0 - Do not force to generate MSI after DMA is completed
+        *      1 - Force to generate MSI after DMA is completed
+        * 6:5  Reserved
+        * 4    1 - Generate completion element in the response queue
+        *      0 - No Completion Code
+        * 3    0 - DMA request is a Link list transfer
+        *      1 - DMA request is a Bulk transfer
+        * 2    Reserved
+        * 1:0  00 - No DMA transfer involved
+        *      01 - DMA transfer is part of inbound transfer
+        *      10 - DMA transfer has outbound transfer
+        *      11 - NA
+        */
+       __u8    cmd;
+       __le32  resv;
+       /* Source address for the transfer */
+       __le64  src_addr;
+       /* Destination address for the transfer */
+       __le64  dest_addr;
+       /* Length of transfer request */
+       __le32  len;
+       __le32  resv2;
+       /* Doorbell address */
+       __le64  db_addr;
+       /*
+        * Special encoded variable
+        * 7    1 - Doorbell(db) write
+        *      0 - No doorbell write
+        * 6:2  Reserved
+        * 1:0  00 - 32 bit access, db address must be aligned to 32bit-boundary
+        *      01 - 16 bit access, db address must be aligned to 16bit-boundary
+        *      10 - 8 bit access, db address must be aligned to 8bit-boundary
+        *      11 - Reserved
+        */
+       __u8    db_len;
+       __u8    resv3;
+       __le16  resv4;
+       /* 32 bit data written to doorbell address */
+       __le32  db_data;
+       /*
+        * Special encoded variable
+        * All the fields of sem_cmdX are passed from user and all are ORed
+        * together to form sem_cmd.
+        * 0:11         Semaphore value
+        * 15:12        Reserved
+        * 20:16        Semaphore index
+        * 21           Reserved
+        * 22           Semaphore Sync
+        * 23           Reserved
+        * 26:24        Semaphore command
+        * 28:27        Reserved
+        * 29           Semaphore DMA out bound sync fence
+        * 30           Semaphore DMA in bound sync fence
+        * 31           Enable semaphore command
+        */
+       __le32  sem_cmd0;
+       __le32  sem_cmd1;
+       __le32  sem_cmd2;
+       __le32  sem_cmd3;
+} __packed;
+
 int get_dbc_req_elem_size(void);
 int get_dbc_rsp_elem_size(void);
 int get_cntl_version(struct qaic_device *qdev, struct qaic_user *usr, u16 
*major, u16 *minor);
@@ -350,6 +461,8 @@ void enable_dbc(struct qaic_device *qdev, u32 dbc_id, 
struct qaic_user *usr);
 void wakeup_dbc(struct qaic_device *qdev, u32 dbc_id);
 void release_dbc(struct qaic_device *qdev, u32 dbc_id);
 void qaic_data_get_fifo_info(struct dma_bridge_chan *dbc, u32 *head, u32 
*tail);
+int qaic_submit_reqs_to_hw(struct bo_slice *slice, unsigned int num_req,
+                          bool is_partial, u32 partial_size, u16 job_id);
 
 void wake_all_cntl(struct qaic_device *qdev);
 void qaic_dev_reset_clean_local_state(struct qaic_device *qdev);
@@ -370,8 +483,20 @@ void qaic_dbc_exit_ssr(struct qaic_device *qdev);
 
 /* qaic_fence.c */
 int qaic_fence_wait(struct dma_fence *fence, signed long timeout);
+struct qaic_fence *qaic_create_irq_fence(struct qaic_bo *bo, u64 seqno, u16 
job_id);
+int qaic_acquire_bo_fence(struct qaic_bo *bo, int job_count, struct list_head 
*job_list);
 void qaic_remove_bo_fence(struct qaic_bo *bo);
-int qaic_acquire_bo_fence(struct qaic_bo *bo);
+
+/* qaic_sched.c */
+int qaicm_sched_init(struct drm_device *drm, struct dma_bridge_chan *dbc);
+void qaic_sched_entity_init(struct dma_bridge_chan *dbc);
+struct qaic_job *qaic_create_job(struct drm_file *file_priv, struct bo_slice 
*slice,
+                                unsigned int num_req, u64 seq_no, struct 
list_head *tmp_list);
+void qaic_submit_job(struct dma_bridge_chan *dbc, struct qaic_job *job);
+void qaic_free_job(struct kref *ref);
+void qaic_put_job(struct qaic_job *job);
+void qaic_cleanup_job(struct qaic_job *job);
+void set_dbc_scaling_ratio(struct dma_bridge_chan *dbc, u32 nelem);
 
 /* qaic_sysfs.c */
 int qaic_sysfs_init(struct qaic_drm_device *qddev);
diff --git a/drivers/accel/qaic/qaic_control.c 
b/drivers/accel/qaic/qaic_control.c
index 2ccc55486aac..9c4400c0fd17 100644
--- a/drivers/accel/qaic/qaic_control.c
+++ b/drivers/accel/qaic/qaic_control.c
@@ -919,7 +919,12 @@ static int decode_deactivate(struct qaic_device *qdev, 
void *trans, u32 *msg_len
                 * Releasing resources failed on the device side, which puts
                 * us in a bind since they may still be in use, so enable the
                 * dbc. User is expected to retry deactivation.
+                *
+                * Scheduler entity must be destroyed before it's reinitialized
+                * in enable_dbc, otherwise entity list element points to itself
+                * and causes a cycle in any list it was a node of.
                 */
+               drm_sched_entity_destroy(&qdev->dbc[dbc_id].sched_entity);
                enable_dbc(qdev, dbc_id, usr);
                return -ECANCELED;
        }
diff --git a/drivers/accel/qaic/qaic_data.c b/drivers/accel/qaic/qaic_data.c
index 0208400d2636..63ca4a489a9a 100644
--- a/drivers/accel/qaic/qaic_data.c
+++ b/drivers/accel/qaic/qaic_data.c
@@ -64,78 +64,6 @@ module_param(datapath_poll_interval_us, uint, 0600);
 MODULE_PARM_DESC(datapath_poll_interval_us,
                 "Amount of time to sleep between activity when datapath 
polling is enabled");
 
-struct dbc_req {
-       /*
-        * A request ID is assigned to each memory handle going in DMA queue.
-        * As a single memory handle can enqueue multiple elements in DMA queue
-        * all of them will have the same request ID.
-        */
-       __le16  req_id;
-       /* Future use */
-       __u8    seq_id;
-       /*
-        * Special encoded variable
-        * 7    0 - Do not force to generate MSI after DMA is completed
-        *      1 - Force to generate MSI after DMA is completed
-        * 6:5  Reserved
-        * 4    1 - Generate completion element in the response queue
-        *      0 - No Completion Code
-        * 3    0 - DMA request is a Link list transfer
-        *      1 - DMA request is a Bulk transfer
-        * 2    Reserved
-        * 1:0  00 - No DMA transfer involved
-        *      01 - DMA transfer is part of inbound transfer
-        *      10 - DMA transfer has outbound transfer
-        *      11 - NA
-        */
-       __u8    cmd;
-       __le32  resv;
-       /* Source address for the transfer */
-       __le64  src_addr;
-       /* Destination address for the transfer */
-       __le64  dest_addr;
-       /* Length of transfer request */
-       __le32  len;
-       __le32  resv2;
-       /* Doorbell address */
-       __le64  db_addr;
-       /*
-        * Special encoded variable
-        * 7    1 - Doorbell(db) write
-        *      0 - No doorbell write
-        * 6:2  Reserved
-        * 1:0  00 - 32 bit access, db address must be aligned to 32bit-boundary
-        *      01 - 16 bit access, db address must be aligned to 16bit-boundary
-        *      10 - 8 bit access, db address must be aligned to 8bit-boundary
-        *      11 - Reserved
-        */
-       __u8    db_len;
-       __u8    resv3;
-       __le16  resv4;
-       /* 32 bit data written to doorbell address */
-       __le32  db_data;
-       /*
-        * Special encoded variable
-        * All the fields of sem_cmdX are passed from user and all are ORed
-        * together to form sem_cmd.
-        * 0:11         Semaphore value
-        * 15:12        Reserved
-        * 20:16        Semaphore index
-        * 21           Reserved
-        * 22           Semaphore Sync
-        * 23           Reserved
-        * 26:24        Semaphore command
-        * 28:27        Reserved
-        * 29           Semaphore DMA out bound sync fence
-        * 30           Semaphore DMA in bound sync fence
-        * 31           Enable semaphore command
-        */
-       __le32  sem_cmd0;
-       __le32  sem_cmd1;
-       __le32  sem_cmd2;
-       __le32  sem_cmd3;
-} __packed;
-
 struct dbc_rsp {
        /* Request ID of the memory handle whose DMA transaction is completed */
        __le16  req_id;
@@ -188,7 +116,7 @@ struct qaic_exec_data_req {
 static inline bool bo_queued(struct qaic_bo *bo)
 {
        if (bo->fence)
-               return !dma_fence_is_signaled(&bo->fence->base);
+               return !dma_fence_is_signaled(bo->fence);
        return false;
 }
 
@@ -751,7 +679,6 @@ static const struct drm_gem_object_funcs qaic_gem_funcs = {
 static void qaic_init_bo_lists(struct qaic_bo *bo)
 {
        INIT_LIST_HEAD(&bo->slices);
-       INIT_LIST_HEAD(&bo->xfer_list);
 }
 
 static struct qaic_bo *qaic_alloc_init_bo(void)
@@ -1181,9 +1108,8 @@ static void submit_reqs_to_fifo(struct dma_bridge_chan 
*dbc, struct dbc_req *req
 }
 
 /* Caller should be holding dbc->req_lock */
-static inline int qaic_submit_reqs_to_hw(struct bo_slice *slice,
-                                        unsigned int num_req,
-                                        bool is_partial, u32 partial_size)
+int qaic_submit_reqs_to_hw(struct bo_slice *slice, unsigned int num_req,
+                          bool is_partial, u32 partial_size, u16 job_id)
 {
        struct dma_bridge_chan *dbc = slice->bo->dbc;
        struct dbc_req *reqs = slice->reqs;
@@ -1225,27 +1151,30 @@ static inline int qaic_submit_reqs_to_hw(struct 
bo_slice *slice,
        bo->perf_stats.req_submit_ts = ktime_get_ns();
 
        /* Finalize commit to hardware */
-       dma_sync_sgtable_for_device(&dbc->qdev->pdev->dev, bo->sgt, bo->dir);
+       if (bo->need_dev_sync) {
+               bo->need_dev_sync = false;
+               dma_sync_sgtable_for_device(&dbc->qdev->pdev->dev, bo->sgt, 
bo->dir);
+       }
        writel(tail, dbc->dbc_base + REQTP_OFF);
        return 0;
 }
 
-static inline int copy_exec_reqs(struct bo_slice *slice)
+static inline void qaic_dbc_put_job(struct dma_bridge_chan *dbc, struct 
qaic_job *job)
 {
-       int ret;
-
-       ret = qaic_submit_reqs_to_hw(slice, slice->nents, false, 0);
-
-       return ret;
+       list_del_init(&job->queue);
+       dbc->pending_reqs -= job->num_req;
+       qaic_put_job(job);
 }
 
-static inline int copy_partial_exec_reqs(struct bo_slice *slice, u64 resize)
+static int create_slice_jobs(struct drm_file *file_priv, struct bo_slice 
*slice, bool partial,
+                            u64 resize, struct list_head *tmp_list)
 {
        struct dbc_req *reqs = slice->reqs;
+       struct qaic_job *job = NULL;
+       unsigned int job_count = 0;
        unsigned int nents_xfer;
        u64 last_bytes;
        u32 first_n;
-       int ret;
 
        /*
         * After this for loop is complete, first_n represents the index
@@ -1260,20 +1189,33 @@ static inline int copy_partial_exec_reqs(struct 
bo_slice *slice, u64 resize)
                else
                        break;
 
-       nents_xfer = first_n + 1;
+       if (partial)
+               nents_xfer = first_n + 1;
+       else
+               nents_xfer = slice->nents;
 
-       ret = qaic_submit_reqs_to_hw(slice, nents_xfer, true, last_bytes);
+       job = qaic_create_job(file_priv, slice, nents_xfer, 0, tmp_list);
+       if (IS_ERR(job))
+               return PTR_ERR(job);
+       job_count++;
 
-       return ret;
+       /* job points to the last job for this slice, only valid for partial 
execute ioctl */
+       job->partial_size = last_bytes;
+       job->partial = partial;
+
+       return job_count;
 }
 
-static int send_bo_list_to_device(struct qaic_device *qdev, struct drm_file 
*file_priv,
-                                 struct execute_info *exec, struct 
dma_bridge_chan *dbc)
+static int send_bo_list_to_sched(struct qaic_device *qdev, struct drm_file 
*file_priv,
+                                struct execute_info *exec, struct 
dma_bridge_chan *dbc)
 {
+       struct qaic_job *job, *job_i, *tmp;
        struct ww_acquire_ctx acquire_ctx;
        struct bo_slice *slice;
        unsigned long flags;
+       LIST_HEAD(tmp_list);
        struct qaic_bo *bo;
+       int job_count;
        u64 resize;
        u32 handle;
        int i, j;
@@ -1290,13 +1232,7 @@ static int send_bo_list_to_device(struct qaic_device 
*qdev, struct drm_file *fil
                handle = exec->handle_arr[i];
                ret = mutex_lock_interruptible(&bo->lock);
                if (ret)
-                       goto failed_to_send_bo;
-
-               /*
-                * Take reference to object before sending to device,
-                * released when transfer is complete (in 
dbc_irq_threaded_fn()).
-                */
-               drm_gem_object_get(&bo->base);
+                       goto free_jobs;
 
                if (!bo->sliced) {
                        ret = -EINVAL;
@@ -1313,54 +1249,54 @@ static int send_bo_list_to_device(struct qaic_device 
*qdev, struct drm_file *fil
                        goto unlock_bo;
                }
 
-               bo->req_id = dbc->next_req_id++;
-
-               spin_lock_irqsave(&dbc->xfer_lock, flags);
-               list_add_tail(&bo->xfer_list, &dbc->xfer_list);
-               spin_unlock_irqrestore(&dbc->xfer_lock, flags);
-
-               ret = qaic_acquire_bo_fence(bo);
-               if (ret)
-                       goto fence_failed;
-
+               job_count = 0;
                list_for_each_entry(slice, &bo->slices, slice) {
-                       for (j = 0; j < slice->nents; j++)
-                               slice->reqs[j].req_id = cpu_to_le16(bo->req_id);
-
                        if (exec->is_partial && (!resize || resize <= 
slice->offset))
                                /* Configure the slice for no DMA transfer */
-                               ret = copy_partial_exec_reqs(slice, 0);
+                               ret = create_slice_jobs(file_priv, slice, true, 
0, &tmp_list);
                        else if (exec->is_partial && resize < slice->offset + 
slice->size)
                                /* Configure the slice to be partially DMA 
transferred */
-                               ret = copy_partial_exec_reqs(slice, resize - 
slice->offset);
+                               ret = create_slice_jobs(file_priv, slice, true,
+                                                       resize - slice->offset, 
&tmp_list);
                        else
-                               ret = copy_exec_reqs(slice);
-                       if (ret)
-                               goto submit_failed;
+                               ret = create_slice_jobs(file_priv, slice, 
false, 0, &tmp_list);
+                       if (ret <= 0)
+                               goto unlock_bo;
+                       job_count += ret;
                }
+
+               ret = qaic_acquire_bo_fence(bo, job_count, &tmp_list);
+               if (ret)
+                       goto unlock_bo;
+
+               bo->need_dev_sync = true;
                mutex_unlock(&bo->lock);
        }
 
+       spin_lock_irqsave(&dbc->job_q_lock, flags);
+       job = list_last_entry(&dbc->job_queue, typeof(*job), queue);
+       list_splice_tail_init(&tmp_list, &dbc->job_queue);
+       list_for_each_entry_continue(job, &dbc->job_queue, queue)
+               qaic_submit_job(dbc, job);
+       spin_unlock_irqrestore(&dbc->job_q_lock, flags);
+
        goto unlock_resv;
 
-submit_failed:
-       qaic_remove_bo_fence(bo);
-fence_failed:
-       spin_lock_irqsave(&dbc->xfer_lock, flags);
-       list_del_init(&bo->xfer_list);
-       spin_unlock_irqrestore(&dbc->xfer_lock, flags);
 unlock_bo:
-       drm_gem_object_put(&bo->base);
        mutex_unlock(&bo->lock);
-failed_to_send_bo:
-       for (j = 0; j < i; j++) {
-               drm_gem_object_put(&exec->bo_arr[j]->base);
-               spin_lock_irqsave(&dbc->xfer_lock, flags);
-               bo = list_last_entry(&dbc->xfer_list, struct qaic_bo, 
xfer_list);
-               list_del_init(&bo->xfer_list);
-               spin_unlock_irqrestore(&dbc->xfer_lock, flags);
-               dma_sync_sgtable_for_cpu(&qdev->pdev->dev, bo->sgt, bo->dir);
+free_jobs:
+       for (j = i - 1; i > 0 && j >= 0; j--) {
+               bo = exec->bo_arr[j];
+               ret = mutex_lock_interruptible(&bo->lock);
+               if (!ret) {
+                       bo->need_dev_sync = false;
+                       dma_fence_put(bo->fence);
+                       bo->fence = NULL;
+                       mutex_unlock(&bo->lock);
+               }
        }
+       list_for_each_entry_safe_reverse(job_i, tmp, &tmp_list, queue)
+               qaic_cleanup_job(job_i);
 unlock_resv:
        drm_gem_unlock_reservations((struct drm_gem_object **)exec->bo_arr,
                                    exec->count, &acquire_ctx);
@@ -1451,7 +1387,6 @@ static int __qaic_execute_bo_ioctl(struct drm_device 
*dev, void *data, struct dr
        struct qaic_user *usr;
        size_t elem_size_sum;
        void *usr_exec_ent;
-       u32 head, tail;
        int i, ret;
 
        exec.perf_stats.received_ts = ktime_get_ns();
@@ -1530,36 +1465,20 @@ static int __qaic_execute_bo_ioctl(struct drm_device 
*dev, void *data, struct dr
                goto release_ch_rcu;
        }
 
-       ret = mutex_lock_interruptible(&dbc->req_lock);
-       if (ret)
-               goto release_ch_rcu;
-
-       head = readl(dbc->dbc_base + REQHP_OFF);
-       tail = readl(dbc->dbc_base + REQTP_OFF);
-
-       if (head == U32_MAX || tail == U32_MAX) {
-               /* PCI link error */
-               ret = -ENODEV;
-               goto unlock_req_lock;
-       }
+       /* Not locking to read pending_reqs since exact value not necessary */
+       exec.perf_stats.queue_level = dbc->pending_reqs;
 
-       exec.perf_stats.queue_level = head <= tail ? tail - head : dbc->nelem - 
(head - tail);
-
-       ret = send_bo_list_to_device(qdev, file_priv, &exec, dbc);
+       ret = send_bo_list_to_sched(qdev, file_priv, &exec, dbc);
        if (ret)
-               goto unlock_req_lock;
+               goto release_ch_rcu;
 
        exec.perf_stats.submit_ts = ktime_get_ns();
-       mutex_unlock(&dbc->req_lock);
 
        update_profiling_data(&exec);
 
        if (datapath_polling)
                schedule_work(&dbc->poll_work);
 
-unlock_req_lock:
-       if (ret)
-               mutex_unlock(&dbc->req_lock);
 release_ch_rcu:
        srcu_read_unlock(&dbc->ch_lock, ch_rcu_id);
        for (i = 0; i < exec.count; i++)
@@ -1683,13 +1602,13 @@ void qaic_irq_polling_work(struct work_struct *work)
                        srcu_read_unlock(&dbc->ch_lock, rcu_id);
                        return;
                }
-               spin_lock_irqsave(&dbc->xfer_lock, flags);
-               if (list_empty(&dbc->xfer_list)) {
-                       spin_unlock_irqrestore(&dbc->xfer_lock, flags);
+               spin_lock_irqsave(&dbc->job_q_lock, flags);
+               if (list_empty(&dbc->job_queue)) {
+                       spin_unlock_irqrestore(&dbc->job_q_lock, flags);
                        srcu_read_unlock(&dbc->ch_lock, rcu_id);
                        return;
                }
-               spin_unlock_irqrestore(&dbc->xfer_lock, flags);
+               spin_unlock_irqrestore(&dbc->job_q_lock, flags);
 
                head = readl(dbc->dbc_base + RSPHP_OFF);
                if (head == U32_MAX) { /* PCI link error */
@@ -1719,8 +1638,8 @@ irqreturn_t dbc_irq_threaded_fn(int irq, void *data)
        struct dma_bridge_chan *dbc = data;
        int event_count = NUM_EVENTS;
        int delay_count = NUM_DELAYS;
+       struct qaic_job *job, *tmp;
        struct qaic_device *qdev;
-       struct qaic_bo *bo, *i;
        struct dbc_rsp *rsp;
        unsigned long flags;
        int rcu_id;
@@ -1773,37 +1692,17 @@ irqreturn_t dbc_irq_threaded_fn(int irq, void *data)
                status = le16_to_cpu(rsp->status);
                if (status)
                        pci_dbg(qdev->pdev, "req_id %d failed with status 
%d\n", req_id, status);
-               spin_lock_irqsave(&dbc->xfer_lock, flags);
-               /*
-                * A BO can receive multiple interrupts, since a BO can be
-                * divided into multiple slices and a buffer receives as many
-                * interrupts as slices. So until it receives interrupts for
-                * all the slices we cannot mark that buffer complete.
-                */
-               list_for_each_entry_safe(bo, i, &dbc->xfer_list, xfer_list) {
-                       if (bo->req_id == req_id)
-                               bo->nr_slice_xfer_done++;
-                       else
-                               continue;
 
-                       if (bo->nr_slice_xfer_done < bo->nr_slice)
+               spin_lock_irqsave(&dbc->job_q_lock, flags);
+               list_for_each_entry_safe(job, tmp, &dbc->job_queue, queue) {
+                       if (job->id == req_id) {
+                               dma_fence_signal(&job->irq_fence->base);
+                               qaic_dbc_put_job(dbc, job);
                                break;
-
-                       /*
-                        * At this point we have received all the interrupts for
-                        * BO, which means BO execution is complete.
-                        */
-                       bo->nr_slice_xfer_done = 0;
-                       list_del_init(&bo->xfer_list);
-                       /*
-                        * fence is put when it is executed next, during 
xfer_list
-                        * cleanup, or when slice is detached
-                        */
-                       dma_fence_signal(&bo->fence->base);
-                       drm_gem_object_put(&bo->base);
-                       break;
+                       }
                }
-               spin_unlock_irqrestore(&dbc->xfer_lock, flags);
+               spin_unlock_irqrestore(&dbc->job_q_lock, flags);
+
                head = (head + 1) % dbc->nelem;
        }
 
@@ -1910,7 +1809,7 @@ int qaic_wait_bo_ioctl(struct drm_device *dev, void 
*data, struct drm_file *file
                goto put_obj;
        }
 
-       fence = &bo->fence->base;
+       fence = bo->fence;
        dma_fence_get(fence);
        mutex_unlock(&bo->lock);
 
@@ -2101,44 +2000,31 @@ int qaic_detach_slice_bo_ioctl(struct drm_device *dev, 
void *data, struct drm_fi
        return ret;
 }
 
-static void empty_xfer_list(struct qaic_device *qdev, struct dma_bridge_chan 
*dbc)
+static void empty_xfer_list(struct dma_bridge_chan *dbc)
 {
+       struct qaic_job *job, *tmp;
        unsigned long flags;
-       struct qaic_bo *bo;
 
-       spin_lock_irqsave(&dbc->xfer_lock, flags);
-       while (!list_empty(&dbc->xfer_list)) {
-               bo = list_first_entry(&dbc->xfer_list, typeof(*bo), xfer_list);
-               list_del_init(&bo->xfer_list);
-               spin_unlock_irqrestore(&dbc->xfer_lock, flags);
-               mutex_lock(&bo->lock);
-               bo->nr_slice_xfer_done = 0;
-               bo->req_id = 0;
-               bo->perf_stats.req_received_ts = 0;
-               bo->perf_stats.req_submit_ts = 0;
-               bo->perf_stats.req_processed_ts = 0;
-               bo->perf_stats.queue_level_before = 0;
-               dma_fence_get(&bo->fence->base);
-               dma_fence_set_error(&bo->fence->base, -ECANCELED);
-               dma_fence_signal(&bo->fence->base);
-               dma_fence_put(&bo->fence->base);
-               qaic_remove_bo_fence(bo);
-               mutex_unlock(&bo->lock);
-               drm_gem_object_put(&bo->base);
-               spin_lock_irqsave(&dbc->xfer_lock, flags);
+       spin_lock_irqsave(&dbc->job_q_lock, flags);
+       list_for_each_entry_safe_reverse(job, tmp, &dbc->job_queue, queue) {
+               dma_fence_get(&job->irq_fence->base);
+               dma_fence_set_error(&job->irq_fence->base, -ECANCELED);
+               dma_fence_signal(&job->irq_fence->base);
+               dma_fence_put(&job->irq_fence->base);
+               qaic_dbc_put_job(dbc, job);
        }
-       spin_unlock_irqrestore(&dbc->xfer_lock, flags);
+       spin_unlock_irqrestore(&dbc->job_q_lock, flags);
 }
 
-static void sync_empty_xfer_list(struct qaic_device *qdev, struct 
dma_bridge_chan *dbc)
+static void sync_empty_xfer_list(struct dma_bridge_chan *dbc)
 {
-       empty_xfer_list(qdev, dbc);
+       empty_xfer_list(dbc);
        synchronize_srcu(&dbc->ch_lock);
        /*
         * Threads holding channel lock, may add more elements in the xfer_list.
         * Flush out these elements from xfer_list.
         */
-       empty_xfer_list(qdev, dbc);
+       empty_xfer_list(dbc);
 }
 
 int disable_dbc(struct qaic_device *qdev, u32 dbc_id, struct qaic_user *usr)
@@ -2161,6 +2047,8 @@ int disable_dbc(struct qaic_device *qdev, u32 dbc_id, 
struct qaic_user *usr)
  */
 void enable_dbc(struct qaic_device *qdev, u32 dbc_id, struct qaic_user *usr)
 {
+       qaic_sched_entity_init(&qdev->dbc[dbc_id]);
+       set_dbc_scaling_ratio(&qdev->dbc[dbc_id], qdev->dbc[dbc_id].nelem);
        qdev->dbc[dbc_id].usr = usr;
 }
 
@@ -2169,7 +2057,7 @@ void wakeup_dbc(struct qaic_device *qdev, u32 dbc_id)
        struct dma_bridge_chan *dbc = &qdev->dbc[dbc_id];
 
        dbc->usr = NULL;
-       sync_empty_xfer_list(qdev, dbc);
+       sync_empty_xfer_list(dbc);
 }
 
 void release_dbc(struct qaic_device *qdev, u32 dbc_id)
@@ -2182,6 +2070,7 @@ void release_dbc(struct qaic_device *qdev, u32 dbc_id)
                return;
 
        wakeup_dbc(qdev, dbc_id);
+       drm_sched_entity_destroy(&dbc->sched_entity);
 
        dma_free_coherent(&qdev->pdev->dev, dbc->total_size, dbc->req_q_base, 
dbc->dma_addr);
        dbc->total_size = 0;
diff --git a/drivers/accel/qaic/qaic_debugfs.c 
b/drivers/accel/qaic/qaic_debugfs.c
index 909a4d774d90..2fac928d5406 100644
--- a/drivers/accel/qaic/qaic_debugfs.c
+++ b/drivers/accel/qaic/qaic_debugfs.c
@@ -100,7 +100,6 @@ void qaic_debugfs_init(struct qaic_drm_device *qddev)
        struct qaic_device *qdev = qddev->qdev;
        struct dentry *debugfs_root;
        struct dentry *debugfs_dir;
-       char name[QAIC_DBC_DIR_NAME];
        u32 i;
 
        debugfs_root = to_drm(qddev)->debugfs_root;
@@ -111,8 +110,7 @@ void qaic_debugfs_init(struct qaic_drm_device *qddev)
         * reasonable range.
         */
        for (i = 0; i < qdev->num_dbc && i < 256; ++i) {
-               snprintf(name, QAIC_DBC_DIR_NAME, "dbc%03u", i);
-               debugfs_dir = debugfs_create_dir(name, debugfs_root);
+               debugfs_dir = debugfs_create_dir(qdev->dbc[i].name, 
debugfs_root);
                debugfs_create_file("fifo_size", 0400, debugfs_dir, 
&qdev->dbc[i], &fifo_size_fops);
                debugfs_create_file("queued", 0400, debugfs_dir, &qdev->dbc[i], 
&queued_fops);
        }
diff --git a/drivers/accel/qaic/qaic_drv.c b/drivers/accel/qaic/qaic_drv.c
index 219f58fdfde8..46b86b3655ca 100644
--- a/drivers/accel/qaic/qaic_drv.c
+++ b/drivers/accel/qaic/qaic_drv.c
@@ -461,10 +461,11 @@ static struct qaic_device *create_qdev(struct pci_dev 
*pdev,
        INIT_LIST_HEAD(&qddev->users);
 
        for (i = 0; i < qdev->num_dbc; ++i) {
-               spin_lock_init(&qdev->dbc[i].xfer_lock);
+               spin_lock_init(&qdev->dbc[i].job_q_lock);
                qdev->dbc[i].qdev = qdev;
                qdev->dbc[i].id = i;
-               INIT_LIST_HEAD(&qdev->dbc[i].xfer_list);
+               INIT_LIST_HEAD(&qdev->dbc[i].job_queue);
+               scnprintf(qdev->dbc[i].name, ARRAY_SIZE(qdev->dbc[i].name), 
"dbc%03u", i);
                ret = qaicm_srcu_init(drm, &qdev->dbc[i].ch_lock);
                if (ret)
                        return NULL;
@@ -473,6 +474,9 @@ static struct qaic_device *create_qdev(struct pci_dev *pdev,
                ret = drmm_mutex_init(drm, &qdev->dbc[i].req_lock);
                if (ret)
                        return NULL;
+               ret = qaicm_sched_init(drm, &qdev->dbc[i]);
+               if (ret)
+                       return NULL;
        }
 
        return qdev;
@@ -708,9 +712,9 @@ static bool qaic_data_path_busy(struct qaic_device *qdev)
                        srcu_read_unlock(&dbc->ch_lock, ch_rcu_id);
                        continue;
                }
-               spin_lock_irqsave(&dbc->xfer_lock, flags);
-               ret = !list_empty(&dbc->xfer_list);
-               spin_unlock_irqrestore(&dbc->xfer_lock, flags);
+               spin_lock_irqsave(&dbc->job_q_lock, flags);
+               ret = !list_empty(&dbc->job_queue);
+               spin_unlock_irqrestore(&dbc->job_q_lock, flags);
                srcu_read_unlock(&dbc->ch_lock, ch_rcu_id);
                if (ret)
                        break;
diff --git a/drivers/accel/qaic/qaic_fence.c b/drivers/accel/qaic/qaic_fence.c
index ec7e555c454b..ed4c7f6a7768 100644
--- a/drivers/accel/qaic/qaic_fence.c
+++ b/drivers/accel/qaic/qaic_fence.c
@@ -2,9 +2,12 @@
 
 /* Copyright (c) 2024-2025 Qualcomm Innovation Center, Inc. All rights 
reserved. */
 
+#include <linux/dma-fence-array.h>
 #include <linux/dma-fence.h>
 #include <linux/dma-resv.h>
+#include <linux/err.h>
 #include <linux/ktime.h>
+#include <linux/slab.h>
 #include <linux/spinlock.h>
 #include <linux/sprintf.h>
 
@@ -20,8 +23,8 @@ static const char *qaic_fence_get_timeline_name(struct 
dma_fence *f)
        struct qaic_fence *qfence = container_of(f, struct qaic_fence, base);
 
        if (!qfence->timeline_name[0])
-               scnprintf(qfence->timeline_name, 
ARRAY_SIZE(qfence->timeline_name), "D%02u",
-                        qfence->dbc_id);
+               scnprintf(qfence->timeline_name, 
ARRAY_SIZE(qfence->timeline_name),
+                         "D%02u-JOB%05x", qfence->dbc_id, qfence->job_id);
        return qfence->timeline_name;
 }
 
@@ -64,7 +67,7 @@ int qaic_fence_wait(struct dma_fence *fence, signed long 
timeout)
        return ret;
 }
 
-static struct qaic_fence *qaic_create_fence(struct qaic_bo *bo)
+struct qaic_fence *qaic_create_irq_fence(struct qaic_bo *bo, u64 seqno, u16 
job_id)
 {
        struct qaic_fence *fence;
 
@@ -74,71 +77,97 @@ static struct qaic_fence *qaic_create_fence(struct qaic_bo 
*bo)
 
        spin_lock_init(&fence->lock);
        fence->dbc_id = bo->dbc->id;
-       dma_fence_init(&fence->base, &qaic_fence_ops, &fence->lock, 
bo->fence_context, 1);
+       fence->job_id = job_id;
+       dma_fence_init(&fence->base, &qaic_fence_ops, &fence->lock, 
bo->fence_context, seqno);
 
        return fence;
 }
 
-/* Caller should be holding bo->lock */
-static int qaic_set_bo_resv_fence(struct qaic_bo *bo, struct qaic_fence *fence)
+static int qaic_create_bo_fence(struct qaic_bo *bo, struct dma_fence 
**output_fence,
+                               int job_count, struct list_head *job_list)
 {
-       int ret;
-
-       ret = dma_resv_reserve_fences(bo->base.resv, 1);
-       if (ret)
-               goto err_out_fence;
+       struct dma_fence_array *fence_arr;
+       struct dma_fence **job_fences;
+       struct qaic_job *job;
+       int ret, i = 0;
 
-       /*
-        * Add a fence in dma-buf reservation object for this BO,
-        * enables buffer sharing across devices.
-        */
-       dma_resv_add_fence(bo->base.resv, &fence->base,
-                          dma_resv_usage_rw(bo->dir == DMA_TO_DEVICE));
+       job_fences = kcalloc(job_count, sizeof(*job_fences), GFP_KERNEL);
+       if (!job_fences)
+               return -ENOMEM;
 
-       ret = dma_fence_add_callback(&fence->base, &bo->cb, 
qaic_fence_sync_cb_func);
+       ret = dma_resv_reserve_fences(bo->base.resv, job_count);
        if (ret)
-               goto err_out_fence;
+               goto job_err;
+
+       /* Add all the fences that we need to wait on to resv */
+       list_for_each_entry_reverse(job, job_list, queue) {
+               if (i < job_count) {
+                       job_fences[i] = dma_fence_get(&job->irq_fence->base);
+                       dma_resv_add_fence(bo->base.resv, job_fences[i],
+                                          dma_resv_usage_rw(bo->dir == 
DMA_TO_DEVICE));
+                       i++;
+               } else {
+                       break;
+               }
+       }
 
-       return ret;
+       /* When successful, job_fences and the fences it contains are now owned 
by fence_arr */
+       fence_arr = dma_fence_array_create(job_count, job_fences, 
bo->fence_context, 1);
+       if (!fence_arr) {
+               ret = -ENOMEM;
+               goto fence_err;
+       }
 
-err_out_fence:
-       dma_fence_put(&fence->base);
-       bo->fence = NULL;
+       /* Output fence array so bo will only have one fence to wait on */
+       *output_fence = &fence_arr->base;
+       return 0;
+fence_err:
+       for (i = 0; i < job_count; i++)
+               dma_fence_put(job_fences[i]);
+job_err:
+       kfree(job_fences);
        return ret;
+
 }
 
 /* Caller should be holding bo->lock */
 void qaic_remove_bo_fence(struct qaic_bo *bo)
 {
        if (bo->fence) {
-               if (!dma_fence_is_signaled(&bo->fence->base)) {
-                       dma_fence_set_error(&bo->fence->base, -ECANCELED);
-                       dma_fence_signal(&bo->fence->base);
+               if (!dma_fence_is_signaled(bo->fence)) {
+                       dma_fence_set_error(bo->fence, -ECANCELED);
+                       dma_fence_signal(bo->fence);
                }
-               dma_fence_put(&bo->fence->base);
+               dma_fence_put(bo->fence);
        }
        bo->fence = NULL;
 }
 
 /* Caller should be holding bo->lock */
-static void qaic_replace_bo_fence(struct qaic_bo *bo, struct qaic_fence *fence)
+static void qaic_replace_bo_fence(struct qaic_bo *bo, struct dma_fence *fence)
 {
        qaic_remove_bo_fence(bo);
        bo->fence = fence;
 }
 
 /* Caller should be holding bo->lock */
-int qaic_acquire_bo_fence(struct qaic_bo *bo)
+int qaic_acquire_bo_fence(struct qaic_bo *bo, int job_count, struct list_head 
*job_list)
 {
-       struct qaic_fence *fence;
-       int ret = 0;
+       struct dma_fence *fence;
+       int ret;
 
-       fence = qaic_create_fence(bo);
-       if (IS_ERR(fence))
-               return PTR_ERR(fence);
+       ret = qaic_create_bo_fence(bo, &fence, job_count, job_list);
+       if (ret)
+               return ret;
+
+       ret = dma_fence_add_callback(fence, &bo->cb, qaic_fence_sync_cb_func);
+       if (ret) {
+               dma_fence_put(fence);
+               bo->fence = NULL;
+               return ret;
+       }
 
        qaic_replace_bo_fence(bo, fence);
-       ret = qaic_set_bo_resv_fence(bo, fence);
 
-       return ret;
+       return 0;
 }
diff --git a/drivers/accel/qaic/qaic_sched.c b/drivers/accel/qaic/qaic_sched.c
new file mode 100644
index 000000000000..b0992d63c2b6
--- /dev/null
+++ b/drivers/accel/qaic/qaic_sched.c
@@ -0,0 +1,236 @@
+// SPDX-License-Identifier: GPL-2.0-only
+
+/* Copyright (c) Qualcomm Technologies, Inc. and/or its subsidiaries. */
+
+#include <drm/drm_device.h>
+#include <drm/drm_file.h>
+#include <drm/drm_gem.h>
+#include <drm/drm_managed.h>
+#include <drm/gpu_scheduler.h>
+#include <linux/dma-fence.h>
+#include <linux/workqueue.h>
+
+#include "qaic.h"
+
+#define QAIC_CREDITS SZ_64K
+
+static void qaic_job_fini(struct drm_sched_job *sched_job)
+{
+       struct qaic_job *job = to_qaic_job(sched_job);
+
+       qaic_cleanup_job(job);
+}
+
+static struct dma_fence *qaic_job_run(struct drm_sched_job *sched_job)
+{
+       struct qaic_job *job = to_qaic_job(sched_job);
+       struct dma_bridge_chan *dbc = job->dbc;
+       struct bo_slice *slice = job->slice;
+       struct dma_fence *fence;
+       int ret;
+
+       fence = dma_fence_get(&job->irq_fence->base);
+
+       /* Job has been cancelled */
+       if (fence->error) {
+               ret = fence->error;
+               goto exit;
+       }
+
+       ret = mutex_lock_interruptible(&dbc->req_lock);
+       if (ret)
+               goto exit;
+
+       ret = qaic_submit_reqs_to_hw(slice, job->num_req, job->partial,
+                                    job->partial_size, job->id);
+       if (ret)
+               goto unlock_dbc;
+       mutex_unlock(&dbc->req_lock);
+
+       return fence;
+
+unlock_dbc:
+       mutex_unlock(&dbc->req_lock);
+exit:
+       dma_fence_put(fence);
+       return ERR_PTR(ret);
+}
+
+static const struct drm_sched_backend_ops qaic_sched_ops = {
+       .run_job = qaic_job_run,
+       .free_job = qaic_job_fini,
+};
+
+static void qaicm_sched_fini(struct drm_device *dev, void *res)
+{
+       struct dma_bridge_chan *dbc = res;
+
+       drm_sched_fini(&dbc->sched);
+       destroy_workqueue(dbc->submit_wq);
+}
+
+/*
+ * Currently, drm_sched is configured to always have credits equal to 64k, but
+ * the network can configure different HW queue sizes. Create a scaling ratio 
to
+ * convert sizes of HW queue to size in credits.
+ *
+ * The ratio expresses the cost of each queue element. For example, if 
nelem=10,
+ * then the ratio is 6554 credits consumed per queue element. This cost,
+ * together with the queue credit balance, limits the number of items in the
+ * queue at any time.
+ *
+ * The credit limit should be greater than the typical HW queue so that the
+ * scaling ratio is never less than 1. 'nelem' is bounds checked in
+ * encode_activate() to guarantee that it is less than QAIC_CREDITS. Round up
+ * the scaling ratio if it isn't perfectly divisible.
+ */
+void set_dbc_scaling_ratio(struct dma_bridge_chan *dbc, u32 nelem)
+{
+       if (!nelem) {
+               /*
+                * There is a chance of enable_dbc() being called after a 
failed deactivation of the
+                * dbc. In that case, set the credit ratio to a value that 
shouldn't let any further
+                * work be scheduled.
+                */
+               dbc->credit_ratio = QAIC_CREDITS;
+               return;
+       }
+       if (QAIC_CREDITS % nelem != 0)
+               pr_debug("Credits not evenly divisible by queue size: size = 
%d\n", nelem);
+       dbc->credit_ratio = DIV_ROUND_UP(QAIC_CREDITS, nelem);
+}
+
+int qaicm_sched_init(struct drm_device *drm, struct dma_bridge_chan *dbc)
+{
+       int ret;
+
+       /*
+        * Using out own workqueue here due to performance regression with
+        * large numbers of devices and jobs (32+ and 16+ respectively).
+        *
+        * Workqueues with WQ_UNBOUND + max_active > 1 also prevented this
+        * regression, but introduced the possibility of unordered
+        * submission to the HW queue. Under testing that did not seem to
+        * cause any issues, though eventually decided to go with this to
+        * minimize changes to existing HW interaction patterns.
+        */
+       dbc->submit_wq = alloc_workqueue("%s", WQ_MEM_RECLAIM | WQ_PERCPU, 1, 
dbc->name);
+       if (!dbc->submit_wq)
+               return -EINVAL;
+
+       const struct drm_sched_init_args args = {
+               .ops = &qaic_sched_ops,
+               .submit_wq = dbc->submit_wq,
+               .num_rqs = 1,
+               .credit_limit = (QAIC_CREDITS - 1),
+               .hang_limit = 1,
+               .timeout = MAX_SCHEDULE_TIMEOUT,
+               .name = dbc->name,
+               .dev = drm->dev,
+       };
+
+       ret = drm_sched_init(&dbc->sched, &args);
+       if (ret) {
+               destroy_workqueue(dbc->submit_wq);
+               return ret;
+       }
+
+       return drmm_add_action_or_reset(drm, qaicm_sched_fini, dbc);
+}
+
+void qaic_sched_entity_init(struct dma_bridge_chan *dbc)
+{
+       struct drm_gpu_scheduler *sched = &dbc->sched;
+
+       drm_sched_entity_init(&dbc->sched_entity, DRM_SCHED_PRIORITY_KERNEL, 
&sched, 1, NULL);
+}
+
+struct qaic_job *qaic_create_job(struct drm_file *file_priv, struct bo_slice 
*slice,
+                                unsigned int num_req, u64 seq_no, struct 
list_head *tmp_list)
+{
+       struct dma_bridge_chan *dbc = slice->bo->dbc;
+       struct qaic_bo *bo = slice->bo;
+       struct qaic_job *job;
+       u32 credits;
+       u16 req_id;
+       int ret, i;
+
+       if (check_mul_overflow((u32)num_req, dbc->credit_ratio, &credits))
+               return ERR_PTR(-EINVAL);
+
+       job = kzalloc_obj(*job);
+       if (!job)
+               return ERR_PTR(-ENOMEM);
+
+       req_id = (u16) atomic_inc_return(&dbc->next_req_id);
+
+       job->irq_fence = qaic_create_irq_fence(bo, seq_no, req_id);
+       if (IS_ERR(job->irq_fence)) {
+               ret = PTR_ERR(job->irq_fence);
+               goto free_job;
+       }
+
+       ret = drm_sched_job_init(&job->base, &dbc->sched_entity, credits,
+                                dbc, file_priv->client_id);
+       if (ret)
+               goto put_fence;
+
+       /*
+        * Make new job depend on the pre-existing fences in BO reservation.
+        * DRM scheduler will wait on every fence in the dependency list
+        * before running this job.
+        */
+       ret = drm_sched_job_add_implicit_dependencies(&job->base, &bo->base, 
true);
+       if (ret)
+               goto job_cleanup;
+
+       job->slice = slice;
+       job->num_req = num_req;
+       job->dbc = dbc;
+       INIT_LIST_HEAD(&job->queue);
+       job->id = req_id;
+       kref_init(&job->ref_count);
+       for (i = 0; i < job->num_req; i++)
+               slice->reqs[i].req_id = cpu_to_le16(job->id);
+
+       list_add_tail(&job->queue, tmp_list);
+       drm_gem_object_get(&bo->base);
+       job->bo = bo;
+       return job;
+
+job_cleanup:
+       drm_sched_job_cleanup(&job->base);
+put_fence:
+       dma_fence_put(&job->irq_fence->base);
+free_job:
+       kfree(job);
+       return ERR_PTR(ret);
+}
+
+void qaic_submit_job(struct dma_bridge_chan *dbc, struct qaic_job *job)
+{
+       kref_get(&job->ref_count);
+       drm_sched_job_arm(&job->base);
+       drm_sched_entity_push_job(&job->base);
+       dbc->pending_reqs += job->num_req;
+}
+
+void qaic_free_job(struct kref *ref)
+{
+       struct qaic_job *job = container_of(ref, struct qaic_job, ref_count);
+
+       dma_fence_put(&job->irq_fence->base);
+       drm_gem_object_put(&job->bo->base);
+       kfree(job);
+}
+
+void qaic_put_job(struct qaic_job *job)
+{
+       kref_put(&job->ref_count, qaic_free_job);
+}
+
+void qaic_cleanup_job(struct qaic_job *job)
+{
+       drm_sched_job_cleanup(&job->base);
+       qaic_put_job(job);
+}
diff --git a/include/uapi/drm/qaic_accel.h b/include/uapi/drm/qaic_accel.h
index c92d0309d583..5271bf66f5bd 100644
--- a/include/uapi/drm/qaic_accel.h
+++ b/include/uapi/drm/qaic_accel.h
@@ -346,7 +346,7 @@ struct qaic_perf_stats {
  * struct qaic_perf_stats_entry - Defines a BO perf info.
  * @handle: In. GEM handle of the BO to get perf stats for.
  * @queue_level_before: Out. Number of elements in the queue before this BO
- *                     was submitted.
+ *                     was submitted, including those not yet scheduled to HW.
  * @num_queue_element: Out. Number of elements added to the queue to submit
  *                    this BO.
  * @submit_latency_us: Out. Time taken by the driver to submit this BO.
-- 
2.43.0

Reply via email to