the issue only happens with oversubscription when gpu has no
workload, the root cause is mes oversubscription timer, so
disable mes timer and make a similar timer in kfd to resolve
the issue.

Signed-off-by: Eric Huang <[email protected]>
---
 drivers/gpu/drm/amd/amdgpu/mes_v11_0.c        |  1 -
 .../drm/amd/amdkfd/kfd_device_queue_manager.c | 39 ++++++++++++++++++-
 .../drm/amd/amdkfd/kfd_device_queue_manager.h |  1 +
 3 files changed, 39 insertions(+), 2 deletions(-)

diff --git a/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c 
b/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c
index bc0ee6ccfca7..40683db66a13 100644
--- a/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c
+++ b/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c
@@ -1019,7 +1019,6 @@ static int mes_v11_0_set_hw_resources(struct amdgpu_mes 
*mes)
        mes_set_hw_res_pkt.use_different_vmid_compute = 1;
        mes_set_hw_res_pkt.enable_reg_active_poll = 1;
        mes_set_hw_res_pkt.enable_level_process_quantum_check = 1;
-       mes_set_hw_res_pkt.oversubscription_timer = 50;
        if (adev->mes.use_rs64mem)
                mes_set_hw_res_pkt.use_rs64mem_for_proc_gang_ctx = 1;
 
diff --git a/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.c 
b/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.c
index ea9d87450eae..350e0ce308d3 100644
--- a/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.c
+++ b/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.c
@@ -47,6 +47,9 @@
 /* See unmap_queues_cpsch() */
 #define USE_DEFAULT_GRACE_PERIOD 0xffffffff
 
+/* Interval for notifying MES of work on unmapped queues during 
oversubscription */
+#define DQM_MES_UNMAP_NOTIFY_DELAY_MS 50
+
 static int set_pasid_vmid_mapping(struct device_queue_manager *dqm,
                                  u32 pasid, unsigned int vmid);
 
@@ -276,8 +279,16 @@ static int add_queue_mes(struct device_queue_manager *dqm, 
struct queue *q,
                        q->properties.doorbell_off);
                dev_err(adev->dev, "MES might be in unrecoverable state, issue 
a GPU reset\n");
                kfd_hws_hang(dqm);
+               return r;
        }
 
+       /* GFX11: start notify timer only once when oversubscription begins */
+       if (KFD_GC_VERSION(dqm->dev) >= IP_VERSION(11, 0, 0) &&
+           KFD_GC_VERSION(dqm->dev) < IP_VERSION(12, 0, 0) &&
+           dqm->active_cp_queue_count > get_cp_queues_num(dqm))
+               queue_delayed_work(system_wq, &dqm->notify_unmap_work,
+                                  
msecs_to_jiffies(DQM_MES_UNMAP_NOTIFY_DELAY_MS));
+
        return r;
 }
 
@@ -354,7 +365,16 @@ static void set_perfcount(struct device_queue_manager 
*dqm, int enable)
 static int remove_queue_mes(struct device_queue_manager *dqm, struct queue *q,
                            struct qcm_process_device *qpd)
 {
-       return remove_queue_mes_on_reset_option(dqm, q, qpd, false, false);
+       int r = remove_queue_mes_on_reset_option(dqm, q, qpd, false, false);
+
+       /* GFX11: stop notify timer when oversubscription clears */
+       if (!r &&
+           KFD_GC_VERSION(dqm->dev) >= IP_VERSION(11, 0, 0) &&
+           KFD_GC_VERSION(dqm->dev) < IP_VERSION(12, 0, 0) &&
+           dqm->active_cp_queue_count <= get_cp_queues_num(dqm))
+               cancel_delayed_work(&dqm->notify_unmap_work);
+
+       return r;
 }
 
 static int remove_all_kfd_queues_mes(struct device_queue_manager *dqm)
@@ -3088,6 +3108,20 @@ static void deallocate_hiq_sdma_mqd(struct kfd_node *dev,
        amdgpu_amdkfd_free_kernel_mem(dev->adev, &mqd->mem);
 }
 
+static void mes_notify_unmap_work_handler(struct work_struct *work)
+{
+       struct device_queue_manager *dqm =
+               container_of(work, struct device_queue_manager,
+                            notify_unmap_work.work);
+
+       amdgpu_mes_notify_unmap_queue((struct amdgpu_device *)dqm->dev->adev);
+
+       /* Re-arm if still oversubscribed */
+       if (READ_ONCE(dqm->active_cp_queue_count) > get_cp_queues_num(dqm))
+               queue_delayed_work(system_wq, &dqm->notify_unmap_work,
+                                  
msecs_to_jiffies(DQM_MES_UNMAP_NOTIFY_DELAY_MS));
+}
+
 struct device_queue_manager *device_queue_manager_init(struct kfd_node *dev)
 {
        struct device_queue_manager *dqm;
@@ -3213,6 +3247,8 @@ struct device_queue_manager 
*device_queue_manager_init(struct kfd_node *dev)
 
        if (!dqm->ops.initialize(dqm)) {
                init_waitqueue_head(&dqm->destroy_wait);
+               INIT_DELAYED_WORK(&dqm->notify_unmap_work,
+                                 mes_notify_unmap_work_handler);
                return dqm;
        }
 
@@ -3229,6 +3265,7 @@ struct device_queue_manager 
*device_queue_manager_init(struct kfd_node *dev)
 
 void device_queue_manager_uninit(struct device_queue_manager *dqm)
 {
+       cancel_delayed_work_sync(&dqm->notify_unmap_work);
        dqm->ops.stop(dqm);
        dqm->ops.uninitialize(dqm);
        if (!dqm->dev->kfd->shared_resources.enable_mes)
diff --git a/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.h 
b/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.h
index c9f9f7a87111..21cf3c16f3f9 100644
--- a/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.h
+++ b/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.h
@@ -281,6 +281,7 @@ struct device_queue_manager {
        uint32_t                wait_times;
 
        wait_queue_head_t       destroy_wait;
+       struct delayed_work     notify_unmap_work;
 
        /* for per-queue reset support */
        struct dqm_detect_hang_info *detect_hang_info;
-- 
2.34.1

Reply via email to