On Wed, Aug 5, 2026 at 6:23 PM Eric Huang <[email protected]> wrote: > > the issue only happens with oversubscription when gpu has no > workload, the root cause is mes oversubscription timer, so > disable mes timer and make a similar timer in kfd to resolve > the issue.
This should only be enabled on FW versions which support it. Additionally, we need similar treatment for KGD userqs if we make this change. Alex > > Signed-off-by: Eric Huang <[email protected]> > --- > drivers/gpu/drm/amd/amdgpu/mes_v11_0.c | 1 - > .../drm/amd/amdkfd/kfd_device_queue_manager.c | 39 ++++++++++++++++++- > .../drm/amd/amdkfd/kfd_device_queue_manager.h | 1 + > 3 files changed, 39 insertions(+), 2 deletions(-) > > diff --git a/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c > b/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c > index bc0ee6ccfca7..40683db66a13 100644 > --- a/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c > +++ b/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c > @@ -1019,7 +1019,6 @@ static int mes_v11_0_set_hw_resources(struct amdgpu_mes > *mes) > mes_set_hw_res_pkt.use_different_vmid_compute = 1; > mes_set_hw_res_pkt.enable_reg_active_poll = 1; > mes_set_hw_res_pkt.enable_level_process_quantum_check = 1; > - mes_set_hw_res_pkt.oversubscription_timer = 50; > if (adev->mes.use_rs64mem) > mes_set_hw_res_pkt.use_rs64mem_for_proc_gang_ctx = 1; > > diff --git a/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.c > b/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.c > index ea9d87450eae..350e0ce308d3 100644 > --- a/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.c > +++ b/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.c > @@ -47,6 +47,9 @@ > /* See unmap_queues_cpsch() */ > #define USE_DEFAULT_GRACE_PERIOD 0xffffffff > > +/* Interval for notifying MES of work on unmapped queues during > oversubscription */ > +#define DQM_MES_UNMAP_NOTIFY_DELAY_MS 50 > + > static int set_pasid_vmid_mapping(struct device_queue_manager *dqm, > u32 pasid, unsigned int vmid); > > @@ -276,8 +279,16 @@ static int add_queue_mes(struct device_queue_manager > *dqm, struct queue *q, > q->properties.doorbell_off); > dev_err(adev->dev, "MES might be in unrecoverable state, > issue a GPU reset\n"); > kfd_hws_hang(dqm); > + return r; > } > > + /* GFX11: start notify timer only once when oversubscription begins */ > + if (KFD_GC_VERSION(dqm->dev) >= IP_VERSION(11, 0, 0) && > + KFD_GC_VERSION(dqm->dev) < IP_VERSION(12, 0, 0) && > + dqm->active_cp_queue_count > get_cp_queues_num(dqm)) > + queue_delayed_work(system_wq, &dqm->notify_unmap_work, > + > msecs_to_jiffies(DQM_MES_UNMAP_NOTIFY_DELAY_MS)); > + > return r; > } > > @@ -354,7 +365,16 @@ static void set_perfcount(struct device_queue_manager > *dqm, int enable) > static int remove_queue_mes(struct device_queue_manager *dqm, struct queue > *q, > struct qcm_process_device *qpd) > { > - return remove_queue_mes_on_reset_option(dqm, q, qpd, false, false); > + int r = remove_queue_mes_on_reset_option(dqm, q, qpd, false, false); > + > + /* GFX11: stop notify timer when oversubscription clears */ > + if (!r && > + KFD_GC_VERSION(dqm->dev) >= IP_VERSION(11, 0, 0) && > + KFD_GC_VERSION(dqm->dev) < IP_VERSION(12, 0, 0) && > + dqm->active_cp_queue_count <= get_cp_queues_num(dqm)) > + cancel_delayed_work(&dqm->notify_unmap_work); > + > + return r; > } > > static int remove_all_kfd_queues_mes(struct device_queue_manager *dqm) > @@ -3088,6 +3108,20 @@ static void deallocate_hiq_sdma_mqd(struct kfd_node > *dev, > amdgpu_amdkfd_free_kernel_mem(dev->adev, &mqd->mem); > } > > +static void mes_notify_unmap_work_handler(struct work_struct *work) > +{ > + struct device_queue_manager *dqm = > + container_of(work, struct device_queue_manager, > + notify_unmap_work.work); > + > + amdgpu_mes_notify_unmap_queue((struct amdgpu_device *)dqm->dev->adev); > + > + /* Re-arm if still oversubscribed */ > + if (READ_ONCE(dqm->active_cp_queue_count) > get_cp_queues_num(dqm)) > + queue_delayed_work(system_wq, &dqm->notify_unmap_work, > + > msecs_to_jiffies(DQM_MES_UNMAP_NOTIFY_DELAY_MS)); > +} > + > struct device_queue_manager *device_queue_manager_init(struct kfd_node *dev) > { > struct device_queue_manager *dqm; > @@ -3213,6 +3247,8 @@ struct device_queue_manager > *device_queue_manager_init(struct kfd_node *dev) > > if (!dqm->ops.initialize(dqm)) { > init_waitqueue_head(&dqm->destroy_wait); > + INIT_DELAYED_WORK(&dqm->notify_unmap_work, > + mes_notify_unmap_work_handler); > return dqm; > } > > @@ -3229,6 +3265,7 @@ struct device_queue_manager > *device_queue_manager_init(struct kfd_node *dev) > > void device_queue_manager_uninit(struct device_queue_manager *dqm) > { > + cancel_delayed_work_sync(&dqm->notify_unmap_work); > dqm->ops.stop(dqm); > dqm->ops.uninitialize(dqm); > if (!dqm->dev->kfd->shared_resources.enable_mes) > diff --git a/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.h > b/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.h > index c9f9f7a87111..21cf3c16f3f9 100644 > --- a/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.h > +++ b/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.h > @@ -281,6 +281,7 @@ struct device_queue_manager { > uint32_t wait_times; > > wait_queue_head_t destroy_wait; > + struct delayed_work notify_unmap_work; > > /* for per-queue reset support */ > struct dqm_detect_hang_info *detect_hang_info; > -- > 2.34.1 >
